Merge nucleic/fuzzy-dewy-urchin-hpvd into dev

This commit is contained in:
2026-08-02 16:38:29 -07:00
parent f89561b6a8
commit 4ad5b65811
3 changed files with 127 additions and 6 deletions
+15
View File
@@ -59,6 +59,21 @@ lines. Once `status` reports `"complete": true`, promote the staged result expli
--confirm overwrite-all-labels-with-sol-high
```
If promotion reports a near-duplicate label conflict in the combined real-session
population, inspect both prompts and exclude the reviewed bad/ambiguous copy explicitly.
The 1-based line is relative to the reported `combined-real.labeled.jsonl`; repeat the
option for multiple decisions:
```bash
"$PY" ml/purpose-classifier/rebuild_sol_high.py promote \
--confirm overwrite-all-labels-with-sol-high \
--exclude-real-line <line>
```
Each exclusion is bound to the prompt hash, prior purpose, and review reason in the
combined dataset manifest and promotion result. This option does not change a teacher
label or relax duplicate detection.
Promotion first validates and curates the complete staged population. It then replaces
the two canonical source files, regenerates all round-two mirrors, relabels shipped
fixtures (teacher-rejected fixtures become the runtime `general` fallback), rebuilds the
+62 -3
View File
@@ -34,6 +34,7 @@ from purpose_data import (
file_sha256,
jsonl_bytes,
load_jsonl,
prompt_hash,
validate_source_record,
write_json,
write_jsonl,
@@ -463,7 +464,41 @@ def _replace_transaction(candidates: Sequence[tuple[Path, Path]], backup_root: P
raise
def promote(stage: Path, confirmation: str | None) -> dict[str, Any]:
def _exclude_reviewed_real_lines(
records: Sequence[dict[str, Any]],
source_lines: Sequence[int],
) -> tuple[list[dict[str, Any]], list[dict[str, Any]]]:
unique_lines = sorted(set(source_lines))
if len(unique_lines) != len(source_lines):
raise DataError("--exclude-real-line contains a duplicate line")
for line in unique_lines:
if not 1 <= line <= len(records):
raise DataError(
f"--exclude-real-line {line} is outside the combined real population "
f"(1-{len(records)})"
)
excluded = set(unique_lines)
review = [
{
"sourceLine": line,
"promptHash": prompt_hash(records[line - 1]["prompt"]),
"purpose": records[line - 1]["purpose"],
"reason": "reviewed-near-duplicate-label-conflict",
}
for line in unique_lines
]
return (
[record for line, record in enumerate(records, 1) if line not in excluded],
review,
)
def promote(
stage: Path,
confirmation: str | None,
excluded_real_lines: Sequence[int] = (),
) -> dict[str, Any]:
if confirmation != CONFIRMATION:
raise DataError(f"promotion requires --confirm {CONFIRMATION}")
load_config(stage)
@@ -514,9 +549,12 @@ def promote(stage: Path, confirmation: str | None) -> dict[str, Any]:
state_by_name["fixtures"],
)
combined_real = promotion / "combined-real.labeled.jsonl"
real_records = _records_from_states(
all_real_records = _records_from_states(
state_by_name["history"]
) + _records_from_states(state_by_name["swe"])
real_records, real_review = _exclude_reviewed_real_lines(
all_real_records, excluded_real_lines
)
write_jsonl(combined_real, real_records)
public_dataset = promotion / "dataset-public"
@@ -558,6 +596,10 @@ def promote(stage: Path, confirmation: str | None) -> dict[str, Any]:
combined_manifest["policy"]["historyUsage"] = (
"Nucleic history and SWE-chat are training-only"
)
combined_manifest["policy"]["reviewedRealLabelConflicts"] = (
"explicitly excluded before near-duplicate curation"
)
combined_manifest["reviewedRealExclusions"] = real_review
combined_manifest["sources"]["baseDataset"]["path"] = _relative(
PUBLIC_DATASET_DESTINATION
)
@@ -608,6 +650,7 @@ def promote(stage: Path, confirmation: str | None) -> dict[str, Any]:
"teacher": {"model": MODEL, "reasoningEffort": REASONING_EFFORT},
"publicLabeled": sum(len(value) for value in public_records.values()),
"realLabeled": len(real_records),
"realReviewExclusions": real_review,
"combinedTrain": combined_manifest["outputs"]["train"]["records"],
"validation": combined_manifest["outputs"]["validation"]["records"],
"test": combined_manifest["outputs"]["test"]["records"],
@@ -663,6 +706,16 @@ def build_parser() -> argparse.ArgumentParser:
"promote", help="curate and replace canonical labels"
)
promote_parser.add_argument("--confirm")
promote_parser.add_argument(
"--exclude-real-line",
action="append",
default=[],
type=int,
help=(
"exclude a reviewed 1-based line from the combined history+SWE training "
"augmentation; repeat for each adjudicated conflict"
),
)
commands_parser = subparsers.add_parser(
"train-commands", help="print clean from-base training commands"
)
@@ -701,7 +754,13 @@ def main(argv: Sequence[str] | None = None) -> int:
elif args.command == "status":
print(json.dumps(status(stage), indent=2, sort_keys=True))
elif args.command == "promote":
print(json.dumps(promote(stage, args.confirm), indent=2, sort_keys=True))
print(
json.dumps(
promote(stage, args.confirm, args.exclude_real_line),
indent=2,
sort_keys=True,
)
)
else:
print(train_commands(args.python))
except (
+50 -3
View File
@@ -185,7 +185,18 @@ class RebuildSolHighTests(unittest.TestCase):
history = root / "history.jsonl"
purpose_data.write_jsonl(history, [{"prompt": "Add a private cache layer"}])
swe = root / "swe.jsonl"
purpose_data.write_jsonl(swe, [{"prompt": "Diagnose a unique worker crash"}])
duplicate_words = [f"duplicate{index}" for index in range(100)]
first_duplicate = " ".join(duplicate_words)
duplicate_words[50] = "replacement"
second_duplicate = " ".join(duplicate_words)
purpose_data.write_jsonl(
swe,
[
{"prompt": "Diagnose a unique worker crash"},
{"prompt": first_duplicate},
{"prompt": second_duplicate},
],
)
stage = script_dir / ".artifacts" / "sol-high-reset"
public_dataset = script_dir / ".artifacts" / "dataset-public"
combined_dataset = script_dir / ".artifacts" / "dataset-v1"
@@ -222,7 +233,10 @@ class RebuildSolHighTests(unittest.TestCase):
(item for item in records if item["prompt"] == prompt),
None,
)
record = original or source_record(prompt)
record = original or source_record(
prompt,
"writing" if prompt == second_duplicate else "backendImpl",
)
states.append(
{
"schemaVersion": 1,
@@ -240,10 +254,12 @@ class RebuildSolHighTests(unittest.TestCase):
result = rebuild_sol_high.promote(
stage,
rebuild_sol_high.CONFIRMATION,
excluded_real_lines=[3],
)
self.assertEqual(80, result["publicLabeled"])
self.assertEqual(2, result["realLabeled"])
self.assertEqual(3, result["realLabeled"])
self.assertEqual(3, result["realReviewExclusions"][0]["sourceLine"])
self.assertTrue((combined_dataset / "train.jsonl").is_file())
self.assertGreater(
len(purpose_data.load_jsonl(combined_dataset / "train.jsonl")),
@@ -256,11 +272,42 @@ class RebuildSolHighTests(unittest.TestCase):
"purpose-dataset-sol-high-v2",
promoted_manifest["datasetVersion"],
)
combined_manifest = json.loads(
(combined_dataset / "manifest.json").read_text(encoding="utf-8")
)
self.assertEqual(
result["realReviewExclusions"],
combined_manifest["reviewedRealExclusions"],
)
def test_promotion_requires_exact_confirmation(self):
with self.assertRaisesRegex(purpose_data.DataError, "promotion requires"):
rebuild_sol_high.promote(Path("unused"), None)
def test_reviewed_real_line_exclusions_are_auditable(self):
records = [
source_record("Plan the cache migration", "planning"),
source_record("Review the cache migration", "review"),
]
retained, review = rebuild_sol_high._exclude_reviewed_real_lines(records, [1])
self.assertEqual([records[1]], retained)
self.assertEqual(1, review[0]["sourceLine"])
self.assertEqual("planning", review[0]["purpose"])
self.assertEqual(
"reviewed-near-duplicate-label-conflict", review[0]["reason"]
)
self.assertEqual(64, len(review[0]["promptHash"]))
def test_reviewed_real_line_exclusions_reject_invalid_lines(self):
records = [source_record("Plan the cache migration", "planning")]
with self.assertRaisesRegex(purpose_data.DataError, "duplicate line"):
rebuild_sol_high._exclude_reviewed_real_lines(records, [1, 1])
with self.assertRaisesRegex(purpose_data.DataError, "outside"):
rebuild_sol_high._exclude_reviewed_real_lines(records, [2])
if __name__ == "__main__":
unittest.main()