Merge nucleic/jolly-coral-egret-smoz into dev
This commit is contained in:
@@ -0,0 +1,243 @@
|
||||
import sys
|
||||
import unittest
|
||||
from collections import Counter
|
||||
from pathlib import Path
|
||||
|
||||
|
||||
MODULE_DIR = Path(__file__).resolve().parents[1]
|
||||
sys.path.insert(0, str(MODULE_DIR))
|
||||
|
||||
import export_swe_chat_prose as prose_export
|
||||
|
||||
|
||||
def row(session, turn, ordinal, content, **overrides):
|
||||
value = {
|
||||
"session_id": session,
|
||||
"turn_id": turn,
|
||||
"conversation_turn_number": ordinal,
|
||||
"turn_number": ordinal,
|
||||
"turn_type": "user_prompt" if ordinal % 2 == 0 else "assistant_response",
|
||||
"role": "user" if ordinal % 2 == 0 else "assistant",
|
||||
"is_conversational": True,
|
||||
"is_continuation": False,
|
||||
"content": content,
|
||||
}
|
||||
value.update(overrides)
|
||||
return value
|
||||
|
||||
|
||||
def select(rows, *, max_per_session=3):
|
||||
"""Run both passes the way `export` does, over one in-memory row list."""
|
||||
|
||||
markers, funnel = prose_export.conversational_markers(rows)
|
||||
pairs = prose_export.select_turn_ends(markers, funnel, max_per_session=max_per_session)
|
||||
candidates = prose_export.attach_text(
|
||||
rows, pairs, sessions={}, funnel=funnel,
|
||||
character_limit=prose_export.DEFAULT_CHARACTER_LIMIT,
|
||||
)
|
||||
return candidates, funnel
|
||||
|
||||
|
||||
class TurnEndSelectionTests(unittest.TestCase):
|
||||
def test_takes_only_the_last_reply_of_a_multi_message_assistant_run(self):
|
||||
rows = [
|
||||
row("s1", "t0", 0, "Add the settings view."),
|
||||
row("s1", "t1", 1, "Working on it."),
|
||||
row("s1", "t2", 2, "Still working.", turn_type="assistant_response", role="assistant"),
|
||||
row("s1", "t3", 3, "The settings view is done."),
|
||||
]
|
||||
candidates, funnel = select(rows)
|
||||
self.assertEqual(1, len(candidates))
|
||||
self.assertEqual("The settings view is done.", candidates[0].prose)
|
||||
self.assertEqual("t3", candidates[0].pair.response_turn_id)
|
||||
self.assertEqual(2, funnel["rejectedMidTurnAssistantResponse"])
|
||||
|
||||
def test_keeps_the_final_reply_when_the_session_ends(self):
|
||||
rows = [
|
||||
row("s1", "t0", 0, "Ship it."),
|
||||
row("s1", "t1", 1, "Shipped. Anything else?"),
|
||||
]
|
||||
candidates, _ = select(rows)
|
||||
self.assertEqual(["Shipped. Anything else?"], [c.prose for c in candidates])
|
||||
|
||||
def test_pairs_each_reply_with_the_prompt_that_opened_its_own_turn(self):
|
||||
rows = [
|
||||
row("s1", "t0", 0, "First request."),
|
||||
row("s1", "t1", 1, "First answer."),
|
||||
row("s1", "t2", 2, "Second request."),
|
||||
row("s1", "t3", 3, "Second answer."),
|
||||
]
|
||||
candidates, _ = select(rows)
|
||||
self.assertEqual(
|
||||
[("First request.", "First answer."), ("Second request.", "Second answer.")],
|
||||
[(c.teacher_prompt, c.prose) for c in candidates],
|
||||
)
|
||||
|
||||
def test_orders_by_conversation_ordinal_not_row_arrival(self):
|
||||
rows = [
|
||||
row("s1", "t3", 3, "Second answer."),
|
||||
row("s1", "t1", 1, "First answer."),
|
||||
row("s1", "t2", 2, "Second request."),
|
||||
row("s1", "t0", 0, "First request."),
|
||||
]
|
||||
candidates, _ = select(rows)
|
||||
self.assertEqual(["First answer.", "Second answer."], [c.prose for c in candidates])
|
||||
|
||||
def test_drops_a_reply_with_no_preceding_user_prompt(self):
|
||||
rows = [row("s1", "t1", 1, "Orphan reply.")]
|
||||
candidates, funnel = select(rows)
|
||||
self.assertEqual([], candidates)
|
||||
self.assertEqual(1, funnel["rejectedNoPrecedingUserPrompt"])
|
||||
|
||||
def test_caps_replies_taken_from_one_session(self):
|
||||
rows = []
|
||||
for index in range(4):
|
||||
rows.append(row("s1", f"u{index}", index * 2, f"Request {index}."))
|
||||
rows.append(row("s1", f"a{index}", index * 2 + 1, f"Answer {index}."))
|
||||
candidates, funnel = select(rows, max_per_session=2)
|
||||
self.assertEqual(["Answer 0.", "Answer 1."], [c.prose for c in candidates])
|
||||
self.assertEqual(2, funnel["rejectedSessionCap"])
|
||||
|
||||
def test_rejects_non_conversational_continuation_and_malformed_rows(self):
|
||||
rows = [
|
||||
row("s1", "t0", 0, "Request."),
|
||||
row("s1", "t1", 1, "Tool output.", is_conversational=False),
|
||||
row("s1", "t2", 2, "Request.", is_continuation=True),
|
||||
row("s1", "t3", 3, "Reply.", conversation_turn_number=None),
|
||||
row("s1", "t4", 4, "Reply.", turn_type="tool_call", role="assistant"),
|
||||
]
|
||||
_, funnel = select(rows)
|
||||
self.assertEqual(1, funnel["rejectedNonConversational"])
|
||||
self.assertEqual(1, funnel["rejectedContinuation"])
|
||||
self.assertEqual(1, funnel["rejectedMalformedMarker"])
|
||||
self.assertEqual(1, funnel["rejectedTurnTypeOrRole"])
|
||||
|
||||
def test_rejects_a_duplicate_turn_id_within_a_session(self):
|
||||
rows = [row("s1", "t0", 0, "a"), row("s1", "t0", 2, "b")]
|
||||
with self.assertRaises(prose_export.DataError):
|
||||
prose_export.conversational_markers(rows)
|
||||
|
||||
|
||||
class ProseExtractionTests(unittest.TestCase):
|
||||
def test_student_text_is_the_runtime_extraction_not_the_raw_reply(self):
|
||||
reply = "The migration is done.\n```swift\nstruct View {}\n```\nTests pass."
|
||||
rows = [row("s1", "t0", 0, "Migrate it."), row("s1", "t1", 1, reply)]
|
||||
candidates, _ = select(rows)
|
||||
self.assertEqual("The migration is done.\nTests pass.", candidates[0].prose)
|
||||
self.assertFalse(candidates[0].tail_biased)
|
||||
|
||||
def test_flags_prose_the_character_limit_actually_truncated(self):
|
||||
rows = [
|
||||
row("s1", "t0", 0, "Do it."),
|
||||
row("s1", "t1", 1, "old words " * 6 + "Now the settings view."),
|
||||
]
|
||||
markers, funnel = prose_export.conversational_markers(rows)
|
||||
pairs = prose_export.select_turn_ends(markers, funnel, max_per_session=3)
|
||||
candidates = prose_export.attach_text(
|
||||
rows, pairs, sessions={}, funnel=funnel, character_limit=25
|
||||
)
|
||||
self.assertEqual("Now the settings view.", candidates[0].prose)
|
||||
self.assertTrue(candidates[0].tail_biased)
|
||||
|
||||
def test_drops_a_reply_that_is_entirely_code(self):
|
||||
rows = [
|
||||
row("s1", "t0", 0, "Show me the struct."),
|
||||
row("s1", "t1", 1, "```swift\nstruct View {}\n```"),
|
||||
]
|
||||
candidates, funnel = select(rows)
|
||||
self.assertEqual([], candidates)
|
||||
self.assertEqual(1, funnel["rejectedNoProseAfterExtraction"])
|
||||
|
||||
def test_drops_empty_and_nul_bearing_content(self):
|
||||
rows = [
|
||||
row("s1", "t0", 0, "Request."),
|
||||
row("s1", "t1", 1, " "),
|
||||
row("s1", "t2", 2, "Request."),
|
||||
row("s1", "t3", 3, "reply\x00"),
|
||||
]
|
||||
candidates, funnel = select(rows)
|
||||
self.assertEqual([], candidates)
|
||||
self.assertEqual(2, funnel["rejectedMalformedOrEmptyContent"])
|
||||
self.assertEqual(2, funnel["rejectedMissingText"])
|
||||
|
||||
def test_records_the_question_signal_and_hashes_the_student_text(self):
|
||||
rows = [
|
||||
row("s1", "t0", 0, "Finish the backend."),
|
||||
row("s1", "t1", 1, "Backend is done. Should I start the UI?"),
|
||||
]
|
||||
candidates, _ = select(rows)
|
||||
record = candidates[0].json("abc123", prose_export.DEFAULT_CHARACTER_LIMIT)
|
||||
self.assertTrue(record["endsInQuestion"])
|
||||
self.assertEqual(["t0", "t1"], record["sourceTurnIDs"])
|
||||
self.assertEqual("abc123", record["sourceRevision"])
|
||||
self.assertEqual(
|
||||
prose_export.prompt_hash("Backend is done. Should I start the UI?"),
|
||||
record["proseHash"],
|
||||
)
|
||||
self.assertEqual("Backend is done. Should I start the UI?", record["prose"])
|
||||
|
||||
|
||||
class CapAndDedupeTests(unittest.TestCase):
|
||||
def make(self, prose, repo, user):
|
||||
pair = prose_export.Pair("s", "t0", "t1", 1)
|
||||
return prose_export.Candidate(
|
||||
session_id="s", repo_id=repo, user_id=user, pair=pair,
|
||||
teacher_prompt="Request.", prose=prose, tail_biased=False,
|
||||
)
|
||||
|
||||
def test_dedupes_repeated_sign_off_prose_across_sessions(self):
|
||||
candidates, funnel = prose_export.cap_and_dedupe(
|
||||
[
|
||||
self.make("All tests pass.", "r1", "u1"),
|
||||
self.make(" all TESTS pass. ", "r2", "u2"),
|
||||
self.make("Renamed the module.", "r3", "u3"),
|
||||
],
|
||||
Counter(), max_per_repo=10, max_per_user=10,
|
||||
)
|
||||
self.assertEqual(["All tests pass.", "Renamed the module."], [c.prose for c in candidates])
|
||||
self.assertEqual(1, funnel["rejectedDuplicateProse"])
|
||||
|
||||
def test_caps_repository_and_user_concentration(self):
|
||||
candidates, funnel = prose_export.cap_and_dedupe(
|
||||
[
|
||||
self.make("One.", "r1", "u1"),
|
||||
self.make("Two.", "r1", "u2"),
|
||||
self.make("Three.", "r2", "u1"),
|
||||
self.make("Four.", "r2", "u2"),
|
||||
],
|
||||
Counter(), max_per_repo=1, max_per_user=1,
|
||||
)
|
||||
self.assertEqual(["One.", "Four."], [c.prose for c in candidates])
|
||||
self.assertEqual(1, funnel["rejectedRepoCap"])
|
||||
self.assertEqual(1, funnel["rejectedUserCap"])
|
||||
self.assertEqual(2, funnel["exportedCandidates"])
|
||||
|
||||
|
||||
class ArgumentTests(unittest.TestCase):
|
||||
def test_rejects_a_mutable_revision(self):
|
||||
for revision in ["main", ""]:
|
||||
with self.subTest(revision), self.assertRaises(prose_export.DataError):
|
||||
prose_export.export(
|
||||
conversations=[], sessions_path=[], revision=revision,
|
||||
output=Path("/dev/null"), manifest_path=Path("/dev/null"),
|
||||
max_per_repo=1, max_per_user=1, max_per_session=1, character_limit=1,
|
||||
)
|
||||
|
||||
def test_rejects_non_positive_caps_and_limits(self):
|
||||
for kwargs in [
|
||||
{"max_per_repo": 0}, {"max_per_user": 0},
|
||||
{"max_per_session": 0}, {"character_limit": 0},
|
||||
]:
|
||||
settings = {
|
||||
"max_per_repo": 1, "max_per_user": 1,
|
||||
"max_per_session": 1, "character_limit": 1, **kwargs,
|
||||
}
|
||||
with self.subTest(kwargs), self.assertRaises(prose_export.DataError):
|
||||
prose_export.export(
|
||||
conversations=[], sessions_path=[], revision="abc123",
|
||||
output=Path("/dev/null"), manifest_path=Path("/dev/null"), **settings,
|
||||
)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
unittest.main()
|
||||
@@ -0,0 +1,168 @@
|
||||
import json
|
||||
import sys
|
||||
import unittest
|
||||
from pathlib import Path
|
||||
|
||||
|
||||
MODULE_DIR = Path(__file__).resolve().parents[1]
|
||||
sys.path.insert(0, str(MODULE_DIR))
|
||||
|
||||
import label_nucleic_prompts as base
|
||||
import label_swe_chat_prose as prose_label
|
||||
from purpose_data import prompt_hash
|
||||
|
||||
|
||||
PROSE = "Rewired the settings screen to the new theme tokens."
|
||||
TEACHER_PROMPT = "Make the settings screen use the new theme."
|
||||
|
||||
|
||||
class LabelSWEChatProseTests(unittest.TestCase):
|
||||
def value(self, **overrides):
|
||||
value = {
|
||||
"schemaVersion": 1,
|
||||
"repoID": "repo",
|
||||
"userID": "user",
|
||||
"sessionID": "session",
|
||||
"sourceTurnIDs": ["one", "two"],
|
||||
"prose": PROSE,
|
||||
"teacherPrompt": TEACHER_PROMPT,
|
||||
"endsInQuestion": False,
|
||||
"tailBiased": False,
|
||||
}
|
||||
value["proseHash"] = prompt_hash(value["prose"])
|
||||
value["teacherPromptHash"] = prompt_hash(value["teacherPrompt"])
|
||||
value.update(overrides)
|
||||
return value
|
||||
|
||||
def candidate(self, **overrides):
|
||||
value = self.value(**overrides)
|
||||
line = base.SourceLine(1, json.dumps(value), "line-hash")
|
||||
return prose_label.candidates([line])[0]
|
||||
|
||||
def decision(self, candidate, recoverable, **overrides):
|
||||
item = {
|
||||
"id": candidate.id, "keep": True, "junkReason": None,
|
||||
"purpose": "frontendImpl", "secondary": None, "mixed": False,
|
||||
"difficulty": 0.4, "slice": "boundary", "lang": "en",
|
||||
"recoverableFromProse": recoverable,
|
||||
}
|
||||
item.update(overrides)
|
||||
return {"items": [item]}
|
||||
|
||||
def test_candidate_rejects_a_prose_hash_mismatch(self):
|
||||
with self.assertRaisesRegex(ValueError, "proseHash"):
|
||||
prose_label.candidates(
|
||||
[base.SourceLine(1, json.dumps(self.value(proseHash="bad")), "line-hash")]
|
||||
)
|
||||
|
||||
def test_candidate_rejects_a_teacher_prompt_hash_mismatch(self):
|
||||
with self.assertRaisesRegex(ValueError, "teacherPromptHash"):
|
||||
prose_label.candidates(
|
||||
[base.SourceLine(1, json.dumps(self.value(teacherPromptHash="bad")), "line-hash")]
|
||||
)
|
||||
|
||||
def test_candidate_requires_the_runtime_signals(self):
|
||||
for field in ("endsInQuestion", "tailBiased"):
|
||||
with self.subTest(field), self.assertRaisesRegex(ValueError, field):
|
||||
prose_label.candidates(
|
||||
[base.SourceLine(1, json.dumps(self.value(**{field: None})), "line-hash")]
|
||||
)
|
||||
|
||||
def test_candidate_rejects_duplicate_normalized_prose(self):
|
||||
lines = [
|
||||
base.SourceLine(1, json.dumps(self.value()), "a"),
|
||||
base.SourceLine(2, json.dumps(self.value(prose=f" {PROSE.upper()} ",
|
||||
proseHash=prompt_hash(PROSE))), "b"),
|
||||
]
|
||||
with self.assertRaisesRegex(ValueError, "duplicate normalized prose"):
|
||||
prose_label.candidates(lines)
|
||||
|
||||
def test_student_text_is_the_prose_never_the_user_prompt(self):
|
||||
candidate = self.candidate()
|
||||
decision = self.decision(candidate, True)["items"][0]
|
||||
record = prose_label.record_from_decision(candidate, decision)
|
||||
self.assertEqual(PROSE, record["prompt"])
|
||||
self.assertNotIn(TEACHER_PROMPT, json.dumps(record))
|
||||
self.assertEqual({"prompt", "purpose", "secondary", "mixed", "difficulty", "slice", "lang"},
|
||||
set(record))
|
||||
|
||||
def test_prose_that_needs_the_user_prompt_becomes_vague_eval(self):
|
||||
candidate = self.candidate()
|
||||
decisions = prose_label.validate_decisions(
|
||||
[candidate], self.decision(candidate, False)
|
||||
)
|
||||
self.assertEqual("vague-eval", decisions[0][1]["slice"])
|
||||
|
||||
def test_unrecoverable_mixed_decision_becomes_single_purpose_vague_eval(self):
|
||||
candidate = self.candidate()
|
||||
decision = self.decision(
|
||||
candidate, False, secondary="backendImpl", mixed=True, slice="mixed"
|
||||
)["items"][0]
|
||||
record = prose_label.record_from_decision(candidate, decision)
|
||||
self.assertEqual("vague-eval", record["slice"])
|
||||
self.assertFalse(record["mixed"])
|
||||
self.assertIsNone(record["secondary"])
|
||||
|
||||
def test_decision_requires_the_recoverability_verdict(self):
|
||||
candidate = self.candidate()
|
||||
payload = self.decision(candidate, None)
|
||||
with self.assertRaisesRegex(ValueError, "recoverableFromProse"):
|
||||
prose_label.validate_decisions([candidate], payload)
|
||||
|
||||
def test_rejected_decision_must_not_claim_recoverability(self):
|
||||
candidate = self.candidate()
|
||||
payload = self.decision(
|
||||
candidate, True, keep=False, junkReason="only offers work", purpose=None,
|
||||
secondary=None, mixed=None, difficulty=None, slice=None, lang=None,
|
||||
)
|
||||
with self.assertRaisesRegex(ValueError, "recoverableFromProse=null"):
|
||||
prose_label.validate_decisions([candidate], payload)
|
||||
|
||||
def test_state_and_audit_carry_no_user_prompt_text(self):
|
||||
candidate = self.candidate()
|
||||
state = prose_label._state(
|
||||
candidate, status="labeled", record=None, reason=None, recoverable=True
|
||||
)
|
||||
self.assertNotIn(TEACHER_PROMPT, json.dumps(state))
|
||||
self.assertEqual(prompt_hash(TEACHER_PROMPT), state["teacherPromptHash"])
|
||||
self.assertTrue(set(prose_label.AUDIT_FIELDS) <= set(state))
|
||||
self.assertNotIn("record", prose_label.AUDIT_FIELDS)
|
||||
|
||||
def test_labeling_prompt_quotes_both_texts_and_names_the_labeled_one(self):
|
||||
candidate = self.candidate(endsInQuestion=True, tailBiased=True)
|
||||
text = prose_label.labeling_prompt([candidate], 8_000, 24_000)
|
||||
payload = json.loads(text.split("<input_json>\n", 1)[1].split("\n</input_json>", 1)[0])
|
||||
item = payload["items"][0]
|
||||
self.assertEqual(PROSE, item["agent_reply_prose"])
|
||||
self.assertEqual(TEACHER_PROMPT, item["preceding_user_message"])
|
||||
self.assertTrue(item["reply_was_truncated_to_its_tail"])
|
||||
self.assertIn("Label only agent_reply_prose", text)
|
||||
# The must-not-fire rule from CONTEXT_SWITCH.md §4 lives in the teacher prompt.
|
||||
self.assertIn("only *asks* about or *offers* work", text)
|
||||
|
||||
def test_response_schema_requires_the_extra_verdict_field(self):
|
||||
schema = prose_label.response_schema([self.candidate()])
|
||||
item = schema["properties"]["items"]["items"]
|
||||
self.assertIn("recoverableFromProse", item["properties"])
|
||||
self.assertIn("recoverableFromProse", item["required"])
|
||||
|
||||
def test_batches_respect_size_and_character_budgets(self):
|
||||
candidates = [
|
||||
self.candidate(prose=f"Reply number {index}.",
|
||||
proseHash=prompt_hash(f"Reply number {index}."))
|
||||
for index in range(5)
|
||||
]
|
||||
batched = prose_label.batches(
|
||||
candidates, batch_size=2, batch_chars=100_000,
|
||||
max_prose_chars=8_000, max_prompt_chars=24_000,
|
||||
)
|
||||
self.assertEqual([2, 2, 1], [len(batch) for batch in batched])
|
||||
with self.assertRaisesRegex(ValueError, "above --batch-chars"):
|
||||
prose_label.batches(
|
||||
candidates, batch_size=2, batch_chars=10,
|
||||
max_prose_chars=8_000, max_prompt_chars=24_000,
|
||||
)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
unittest.main()
|
||||
@@ -0,0 +1,110 @@
|
||||
import json
|
||||
import sys
|
||||
import unittest
|
||||
from pathlib import Path
|
||||
|
||||
|
||||
MODULE_DIR = Path(__file__).resolve().parents[1]
|
||||
sys.path.insert(0, str(MODULE_DIR))
|
||||
|
||||
import prose_extract
|
||||
|
||||
FIXTURE = (
|
||||
MODULE_DIR.parents[1] / "Tests/NucleicCoreTests/Fixtures/context-switch-prose.json"
|
||||
)
|
||||
|
||||
|
||||
class ProseParityTests(unittest.TestCase):
|
||||
"""The shared fixture is the contract between this port and the Swift runtime.
|
||||
|
||||
``ContextSwitchTests.replyProseMatchesTheSharedParityFixture`` asserts the same file
|
||||
against ``HeuristicSummary.contextSwitchReplyProse``. A change to either extraction
|
||||
that is not mirrored in the other fails one of the two suites.
|
||||
"""
|
||||
|
||||
def setUp(self):
|
||||
self.cases = json.loads(FIXTURE.read_text(encoding="utf-8"))
|
||||
self.assertTrue(self.cases, "shared prose parity fixture is empty")
|
||||
|
||||
def test_matches_the_shared_runtime_fixture(self):
|
||||
for case in self.cases:
|
||||
with self.subTest(case["name"]):
|
||||
prose = prose_extract.reply_prose(case["reply"], case["characterLimit"])
|
||||
self.assertEqual(prose, case["expectedProse"])
|
||||
self.assertEqual(
|
||||
prose_extract.ends_in_question(prose), case["expectedEndsInQuestion"]
|
||||
)
|
||||
|
||||
def test_extracted_prose_never_exceeds_the_limit_in_graphemes(self):
|
||||
for case in self.cases:
|
||||
prose = prose_extract.reply_prose(case["reply"], case["characterLimit"])
|
||||
if prose is None:
|
||||
continue
|
||||
with self.subTest(case["name"]):
|
||||
self.assertLessEqual(
|
||||
len(prose_extract.graphemes(prose)), case["characterLimit"]
|
||||
)
|
||||
|
||||
|
||||
class GraphemeSegmentationTests(unittest.TestCase):
|
||||
def test_models_the_clusters_swift_treats_as_one_character(self):
|
||||
for text, expected in [
|
||||
("á", 1), # combining acute
|
||||
("\U0001f468\U0001f469\U0001f467", 1), # family ZWJ sequence
|
||||
("\U0001f44d\U0001f3fd", 1), # thumbs up + skin tone
|
||||
("\U0001f1fa\U0001f1f8", 1), # regional indicator pair
|
||||
("\U0001f1fa\U0001f1f8\U0001f1e9\U0001f1ea", 2), # two flags, not one run
|
||||
("\r\n", 1),
|
||||
("\n\n", 2),
|
||||
("2️⃣", 1), # keycap
|
||||
("plain", 5),
|
||||
]:
|
||||
with self.subTest(repr(text)):
|
||||
self.assertEqual(len(prose_extract.graphemes(text)), expected)
|
||||
|
||||
def test_reassembling_clusters_is_lossless(self):
|
||||
for text in ["", "ábc", "\U0001f468\U0001f469 x", "a\r\nb"]:
|
||||
with self.subTest(repr(text)):
|
||||
self.assertEqual("".join(prose_extract.graphemes(text)), text)
|
||||
|
||||
|
||||
class TrimTests(unittest.TestCase):
|
||||
def test_uses_the_swift_whitespace_set_rather_than_str_strip(self):
|
||||
# str.strip() would remove the information separators; Swift's
|
||||
# .whitespacesAndNewlines does not, so neither does the port.
|
||||
self.assertEqual(prose_extract.trim("Prose."), "Prose.")
|
||||
self.assertEqual(prose_extract.trim(" Prose. "), "Prose.")
|
||||
self.assertEqual(prose_extract.trim(" \t\r\n "), "")
|
||||
|
||||
def test_whitespace_set_is_exactly_unicode_white_space(self):
|
||||
expected = {
|
||||
*range(0x0009, 0x000E),
|
||||
0x0020,
|
||||
0x0085,
|
||||
0x00A0,
|
||||
0x1680,
|
||||
*range(0x2000, 0x200B),
|
||||
0x2028,
|
||||
0x2029,
|
||||
0x202F,
|
||||
0x205F,
|
||||
0x3000,
|
||||
}
|
||||
self.assertEqual({ord(c) for c in prose_extract.WHITESPACE}, expected)
|
||||
|
||||
|
||||
class FenceStrippingTests(unittest.TestCase):
|
||||
def test_drops_fenced_blocks_and_keeps_outside_line_structure(self):
|
||||
self.assertEqual(
|
||||
prose_extract.strip_fenced_code("a\n```\ncode\n```\nb"), "a\nb"
|
||||
)
|
||||
# An unclosed fence takes the remainder: half-streamed code is still code.
|
||||
self.assertEqual(prose_extract.strip_fenced_code("a\n```\ncode"), "a")
|
||||
# A backtick fence does not close a tilde fence.
|
||||
self.assertEqual(
|
||||
prose_extract.strip_fenced_code("a\n~~~\n```\nx\n```\n~~~\nb"), "a\nb"
|
||||
)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
unittest.main()
|
||||
Reference in New Issue
Block a user