import json import sys import unittest from pathlib import Path MODULE_DIR = Path(__file__).resolve().parents[1] sys.path.insert(0, str(MODULE_DIR)) import prose_extract FIXTURE = ( MODULE_DIR.parents[1] / "Tests/NucleicCoreTests/Fixtures/context-switch-prose.json" ) class ProseParityTests(unittest.TestCase): """The shared fixture is the contract between this port and the Swift runtime. ``ContextSwitchTests.replyProseMatchesTheSharedParityFixture`` asserts the same file against ``HeuristicSummary.contextSwitchReplyProse``. A change to either extraction that is not mirrored in the other fails one of the two suites. """ def setUp(self): self.cases = json.loads(FIXTURE.read_text(encoding="utf-8")) self.assertTrue(self.cases, "shared prose parity fixture is empty") def test_matches_the_shared_runtime_fixture(self): for case in self.cases: with self.subTest(case["name"]): prose = prose_extract.reply_prose(case["reply"], case["characterLimit"]) self.assertEqual(prose, case["expectedProse"]) self.assertEqual( prose_extract.ends_in_question(prose), case["expectedEndsInQuestion"] ) def test_extracted_prose_never_exceeds_the_limit_in_graphemes(self): for case in self.cases: prose = prose_extract.reply_prose(case["reply"], case["characterLimit"]) if prose is None: continue with self.subTest(case["name"]): self.assertLessEqual( len(prose_extract.graphemes(prose)), case["characterLimit"] ) class GraphemeSegmentationTests(unittest.TestCase): def test_models_the_clusters_swift_treats_as_one_character(self): for text, expected in [ ("á", 1), # combining acute ("\U0001f468‍\U0001f469‍\U0001f467", 1), # family ZWJ sequence ("\U0001f44d\U0001f3fd", 1), # thumbs up + skin tone ("\U0001f1fa\U0001f1f8", 1), # regional indicator pair ("\U0001f1fa\U0001f1f8\U0001f1e9\U0001f1ea", 2), # two flags, not one run ("\r\n", 1), ("\n\n", 2), ("2️⃣", 1), # keycap ("plain", 5), ]: with self.subTest(repr(text)): self.assertEqual(len(prose_extract.graphemes(text)), expected) def test_reassembling_clusters_is_lossless(self): for text in ["", "ábc", "\U0001f468‍\U0001f469 x", "a\r\nb"]: with self.subTest(repr(text)): self.assertEqual("".join(prose_extract.graphemes(text)), text) class TrimTests(unittest.TestCase): def test_uses_the_swift_whitespace_set_rather_than_str_strip(self): # str.strip() would remove the information separators; Swift's # .whitespacesAndNewlines does not, so neither does the port. self.assertEqual(prose_extract.trim("Prose."), "Prose.") self.assertEqual(prose_extract.trim("  Prose. "), "Prose.") self.assertEqual(prose_extract.trim(" \t\r\n "), "") def test_whitespace_set_is_exactly_unicode_white_space(self): expected = { *range(0x0009, 0x000E), 0x0020, 0x0085, 0x00A0, 0x1680, *range(0x2000, 0x200B), 0x2028, 0x2029, 0x202F, 0x205F, 0x3000, } self.assertEqual({ord(c) for c in prose_extract.WHITESPACE}, expected) class FenceStrippingTests(unittest.TestCase): def test_drops_fenced_blocks_and_keeps_outside_line_structure(self): self.assertEqual( prose_extract.strip_fenced_code("a\n```\ncode\n```\nb"), "a\nb" ) # An unclosed fence takes the remainder: half-streamed code is still code. self.assertEqual(prose_extract.strip_fenced_code("a\n```\ncode"), "a") # A backtick fence does not close a tilde fence. self.assertEqual( prose_extract.strip_fenced_code("a\n~~~\n```\nx\n```\n~~~\nb"), "a\nb" ) if __name__ == "__main__": unittest.main()