From 4ad5b6581168e08f855475b434d924340aeeca0b Mon Sep 17 00:00:00 2001 From: Nucleic Date: Sun, 2 Aug 2026 16:38:29 -0700 Subject: [PATCH] Merge nucleic/fuzzy-dewy-urchin-hpvd into dev --- README.md | 15 ++++++++ rebuild_sol_high.py | 65 ++++++++++++++++++++++++++++++++-- tests/test_rebuild_sol_high.py | 53 +++++++++++++++++++++++++-- 3 files changed, 127 insertions(+), 6 deletions(-) diff --git a/README.md b/README.md index be3cd80..2a92ee6 100644 --- a/README.md +++ b/README.md @@ -59,6 +59,21 @@ lines. Once `status` reports `"complete": true`, promote the staged result expli --confirm overwrite-all-labels-with-sol-high ``` +If promotion reports a near-duplicate label conflict in the combined real-session +population, inspect both prompts and exclude the reviewed bad/ambiguous copy explicitly. +The 1-based line is relative to the reported `combined-real.labeled.jsonl`; repeat the +option for multiple decisions: + +```bash +"$PY" ml/purpose-classifier/rebuild_sol_high.py promote \ + --confirm overwrite-all-labels-with-sol-high \ + --exclude-real-line +``` + +Each exclusion is bound to the prompt hash, prior purpose, and review reason in the +combined dataset manifest and promotion result. This option does not change a teacher +label or relax duplicate detection. + Promotion first validates and curates the complete staged population. It then replaces the two canonical source files, regenerates all round-two mirrors, relabels shipped fixtures (teacher-rejected fixtures become the runtime `general` fallback), rebuilds the diff --git a/rebuild_sol_high.py b/rebuild_sol_high.py index 497b6a0..d3f9798 100644 --- a/rebuild_sol_high.py +++ b/rebuild_sol_high.py @@ -34,6 +34,7 @@ from purpose_data import ( file_sha256, jsonl_bytes, load_jsonl, + prompt_hash, validate_source_record, write_json, write_jsonl, @@ -463,7 +464,41 @@ def _replace_transaction(candidates: Sequence[tuple[Path, Path]], backup_root: P raise -def promote(stage: Path, confirmation: str | None) -> dict[str, Any]: +def _exclude_reviewed_real_lines( + records: Sequence[dict[str, Any]], + source_lines: Sequence[int], +) -> tuple[list[dict[str, Any]], list[dict[str, Any]]]: + unique_lines = sorted(set(source_lines)) + if len(unique_lines) != len(source_lines): + raise DataError("--exclude-real-line contains a duplicate line") + for line in unique_lines: + if not 1 <= line <= len(records): + raise DataError( + f"--exclude-real-line {line} is outside the combined real population " + f"(1-{len(records)})" + ) + + excluded = set(unique_lines) + review = [ + { + "sourceLine": line, + "promptHash": prompt_hash(records[line - 1]["prompt"]), + "purpose": records[line - 1]["purpose"], + "reason": "reviewed-near-duplicate-label-conflict", + } + for line in unique_lines + ] + return ( + [record for line, record in enumerate(records, 1) if line not in excluded], + review, + ) + + +def promote( + stage: Path, + confirmation: str | None, + excluded_real_lines: Sequence[int] = (), +) -> dict[str, Any]: if confirmation != CONFIRMATION: raise DataError(f"promotion requires --confirm {CONFIRMATION}") load_config(stage) @@ -514,9 +549,12 @@ def promote(stage: Path, confirmation: str | None) -> dict[str, Any]: state_by_name["fixtures"], ) combined_real = promotion / "combined-real.labeled.jsonl" - real_records = _records_from_states( + all_real_records = _records_from_states( state_by_name["history"] ) + _records_from_states(state_by_name["swe"]) + real_records, real_review = _exclude_reviewed_real_lines( + all_real_records, excluded_real_lines + ) write_jsonl(combined_real, real_records) public_dataset = promotion / "dataset-public" @@ -558,6 +596,10 @@ def promote(stage: Path, confirmation: str | None) -> dict[str, Any]: combined_manifest["policy"]["historyUsage"] = ( "Nucleic history and SWE-chat are training-only" ) + combined_manifest["policy"]["reviewedRealLabelConflicts"] = ( + "explicitly excluded before near-duplicate curation" + ) + combined_manifest["reviewedRealExclusions"] = real_review combined_manifest["sources"]["baseDataset"]["path"] = _relative( PUBLIC_DATASET_DESTINATION ) @@ -608,6 +650,7 @@ def promote(stage: Path, confirmation: str | None) -> dict[str, Any]: "teacher": {"model": MODEL, "reasoningEffort": REASONING_EFFORT}, "publicLabeled": sum(len(value) for value in public_records.values()), "realLabeled": len(real_records), + "realReviewExclusions": real_review, "combinedTrain": combined_manifest["outputs"]["train"]["records"], "validation": combined_manifest["outputs"]["validation"]["records"], "test": combined_manifest["outputs"]["test"]["records"], @@ -663,6 +706,16 @@ def build_parser() -> argparse.ArgumentParser: "promote", help="curate and replace canonical labels" ) promote_parser.add_argument("--confirm") + promote_parser.add_argument( + "--exclude-real-line", + action="append", + default=[], + type=int, + help=( + "exclude a reviewed 1-based line from the combined history+SWE training " + "augmentation; repeat for each adjudicated conflict" + ), + ) commands_parser = subparsers.add_parser( "train-commands", help="print clean from-base training commands" ) @@ -701,7 +754,13 @@ def main(argv: Sequence[str] | None = None) -> int: elif args.command == "status": print(json.dumps(status(stage), indent=2, sort_keys=True)) elif args.command == "promote": - print(json.dumps(promote(stage, args.confirm), indent=2, sort_keys=True)) + print( + json.dumps( + promote(stage, args.confirm, args.exclude_real_line), + indent=2, + sort_keys=True, + ) + ) else: print(train_commands(args.python)) except ( diff --git a/tests/test_rebuild_sol_high.py b/tests/test_rebuild_sol_high.py index 2eb7451..814741c 100644 --- a/tests/test_rebuild_sol_high.py +++ b/tests/test_rebuild_sol_high.py @@ -185,7 +185,18 @@ class RebuildSolHighTests(unittest.TestCase): history = root / "history.jsonl" purpose_data.write_jsonl(history, [{"prompt": "Add a private cache layer"}]) swe = root / "swe.jsonl" - purpose_data.write_jsonl(swe, [{"prompt": "Diagnose a unique worker crash"}]) + duplicate_words = [f"duplicate{index}" for index in range(100)] + first_duplicate = " ".join(duplicate_words) + duplicate_words[50] = "replacement" + second_duplicate = " ".join(duplicate_words) + purpose_data.write_jsonl( + swe, + [ + {"prompt": "Diagnose a unique worker crash"}, + {"prompt": first_duplicate}, + {"prompt": second_duplicate}, + ], + ) stage = script_dir / ".artifacts" / "sol-high-reset" public_dataset = script_dir / ".artifacts" / "dataset-public" combined_dataset = script_dir / ".artifacts" / "dataset-v1" @@ -222,7 +233,10 @@ class RebuildSolHighTests(unittest.TestCase): (item for item in records if item["prompt"] == prompt), None, ) - record = original or source_record(prompt) + record = original or source_record( + prompt, + "writing" if prompt == second_duplicate else "backendImpl", + ) states.append( { "schemaVersion": 1, @@ -240,10 +254,12 @@ class RebuildSolHighTests(unittest.TestCase): result = rebuild_sol_high.promote( stage, rebuild_sol_high.CONFIRMATION, + excluded_real_lines=[3], ) self.assertEqual(80, result["publicLabeled"]) - self.assertEqual(2, result["realLabeled"]) + self.assertEqual(3, result["realLabeled"]) + self.assertEqual(3, result["realReviewExclusions"][0]["sourceLine"]) self.assertTrue((combined_dataset / "train.jsonl").is_file()) self.assertGreater( len(purpose_data.load_jsonl(combined_dataset / "train.jsonl")), @@ -256,11 +272,42 @@ class RebuildSolHighTests(unittest.TestCase): "purpose-dataset-sol-high-v2", promoted_manifest["datasetVersion"], ) + combined_manifest = json.loads( + (combined_dataset / "manifest.json").read_text(encoding="utf-8") + ) + self.assertEqual( + result["realReviewExclusions"], + combined_manifest["reviewedRealExclusions"], + ) def test_promotion_requires_exact_confirmation(self): with self.assertRaisesRegex(purpose_data.DataError, "promotion requires"): rebuild_sol_high.promote(Path("unused"), None) + def test_reviewed_real_line_exclusions_are_auditable(self): + records = [ + source_record("Plan the cache migration", "planning"), + source_record("Review the cache migration", "review"), + ] + + retained, review = rebuild_sol_high._exclude_reviewed_real_lines(records, [1]) + + self.assertEqual([records[1]], retained) + self.assertEqual(1, review[0]["sourceLine"]) + self.assertEqual("planning", review[0]["purpose"]) + self.assertEqual( + "reviewed-near-duplicate-label-conflict", review[0]["reason"] + ) + self.assertEqual(64, len(review[0]["promptHash"])) + + def test_reviewed_real_line_exclusions_reject_invalid_lines(self): + records = [source_record("Plan the cache migration", "planning")] + + with self.assertRaisesRegex(purpose_data.DataError, "duplicate line"): + rebuild_sol_high._exclude_reviewed_real_lines(records, [1, 1]) + with self.assertRaisesRegex(purpose_data.DataError, "outside"): + rebuild_sol_high._exclude_reviewed_real_lines(records, [2]) + if __name__ == "__main__": unittest.main()