Merge nucleic/fuzzy-dewy-urchin-hpvd into dev
This commit is contained in:
@@ -59,6 +59,21 @@ lines. Once `status` reports `"complete": true`, promote the staged result expli
|
||||
--confirm overwrite-all-labels-with-sol-high
|
||||
```
|
||||
|
||||
If promotion reports a near-duplicate label conflict in the combined real-session
|
||||
population, inspect both prompts and exclude the reviewed bad/ambiguous copy explicitly.
|
||||
The 1-based line is relative to the reported `combined-real.labeled.jsonl`; repeat the
|
||||
option for multiple decisions:
|
||||
|
||||
```bash
|
||||
"$PY" ml/purpose-classifier/rebuild_sol_high.py promote \
|
||||
--confirm overwrite-all-labels-with-sol-high \
|
||||
--exclude-real-line <line>
|
||||
```
|
||||
|
||||
Each exclusion is bound to the prompt hash, prior purpose, and review reason in the
|
||||
combined dataset manifest and promotion result. This option does not change a teacher
|
||||
label or relax duplicate detection.
|
||||
|
||||
Promotion first validates and curates the complete staged population. It then replaces
|
||||
the two canonical source files, regenerates all round-two mirrors, relabels shipped
|
||||
fixtures (teacher-rejected fixtures become the runtime `general` fallback), rebuilds the
|
||||
|
||||
+62
-3
@@ -34,6 +34,7 @@ from purpose_data import (
|
||||
file_sha256,
|
||||
jsonl_bytes,
|
||||
load_jsonl,
|
||||
prompt_hash,
|
||||
validate_source_record,
|
||||
write_json,
|
||||
write_jsonl,
|
||||
@@ -463,7 +464,41 @@ def _replace_transaction(candidates: Sequence[tuple[Path, Path]], backup_root: P
|
||||
raise
|
||||
|
||||
|
||||
def promote(stage: Path, confirmation: str | None) -> dict[str, Any]:
|
||||
def _exclude_reviewed_real_lines(
|
||||
records: Sequence[dict[str, Any]],
|
||||
source_lines: Sequence[int],
|
||||
) -> tuple[list[dict[str, Any]], list[dict[str, Any]]]:
|
||||
unique_lines = sorted(set(source_lines))
|
||||
if len(unique_lines) != len(source_lines):
|
||||
raise DataError("--exclude-real-line contains a duplicate line")
|
||||
for line in unique_lines:
|
||||
if not 1 <= line <= len(records):
|
||||
raise DataError(
|
||||
f"--exclude-real-line {line} is outside the combined real population "
|
||||
f"(1-{len(records)})"
|
||||
)
|
||||
|
||||
excluded = set(unique_lines)
|
||||
review = [
|
||||
{
|
||||
"sourceLine": line,
|
||||
"promptHash": prompt_hash(records[line - 1]["prompt"]),
|
||||
"purpose": records[line - 1]["purpose"],
|
||||
"reason": "reviewed-near-duplicate-label-conflict",
|
||||
}
|
||||
for line in unique_lines
|
||||
]
|
||||
return (
|
||||
[record for line, record in enumerate(records, 1) if line not in excluded],
|
||||
review,
|
||||
)
|
||||
|
||||
|
||||
def promote(
|
||||
stage: Path,
|
||||
confirmation: str | None,
|
||||
excluded_real_lines: Sequence[int] = (),
|
||||
) -> dict[str, Any]:
|
||||
if confirmation != CONFIRMATION:
|
||||
raise DataError(f"promotion requires --confirm {CONFIRMATION}")
|
||||
load_config(stage)
|
||||
@@ -514,9 +549,12 @@ def promote(stage: Path, confirmation: str | None) -> dict[str, Any]:
|
||||
state_by_name["fixtures"],
|
||||
)
|
||||
combined_real = promotion / "combined-real.labeled.jsonl"
|
||||
real_records = _records_from_states(
|
||||
all_real_records = _records_from_states(
|
||||
state_by_name["history"]
|
||||
) + _records_from_states(state_by_name["swe"])
|
||||
real_records, real_review = _exclude_reviewed_real_lines(
|
||||
all_real_records, excluded_real_lines
|
||||
)
|
||||
write_jsonl(combined_real, real_records)
|
||||
|
||||
public_dataset = promotion / "dataset-public"
|
||||
@@ -558,6 +596,10 @@ def promote(stage: Path, confirmation: str | None) -> dict[str, Any]:
|
||||
combined_manifest["policy"]["historyUsage"] = (
|
||||
"Nucleic history and SWE-chat are training-only"
|
||||
)
|
||||
combined_manifest["policy"]["reviewedRealLabelConflicts"] = (
|
||||
"explicitly excluded before near-duplicate curation"
|
||||
)
|
||||
combined_manifest["reviewedRealExclusions"] = real_review
|
||||
combined_manifest["sources"]["baseDataset"]["path"] = _relative(
|
||||
PUBLIC_DATASET_DESTINATION
|
||||
)
|
||||
@@ -608,6 +650,7 @@ def promote(stage: Path, confirmation: str | None) -> dict[str, Any]:
|
||||
"teacher": {"model": MODEL, "reasoningEffort": REASONING_EFFORT},
|
||||
"publicLabeled": sum(len(value) for value in public_records.values()),
|
||||
"realLabeled": len(real_records),
|
||||
"realReviewExclusions": real_review,
|
||||
"combinedTrain": combined_manifest["outputs"]["train"]["records"],
|
||||
"validation": combined_manifest["outputs"]["validation"]["records"],
|
||||
"test": combined_manifest["outputs"]["test"]["records"],
|
||||
@@ -663,6 +706,16 @@ def build_parser() -> argparse.ArgumentParser:
|
||||
"promote", help="curate and replace canonical labels"
|
||||
)
|
||||
promote_parser.add_argument("--confirm")
|
||||
promote_parser.add_argument(
|
||||
"--exclude-real-line",
|
||||
action="append",
|
||||
default=[],
|
||||
type=int,
|
||||
help=(
|
||||
"exclude a reviewed 1-based line from the combined history+SWE training "
|
||||
"augmentation; repeat for each adjudicated conflict"
|
||||
),
|
||||
)
|
||||
commands_parser = subparsers.add_parser(
|
||||
"train-commands", help="print clean from-base training commands"
|
||||
)
|
||||
@@ -701,7 +754,13 @@ def main(argv: Sequence[str] | None = None) -> int:
|
||||
elif args.command == "status":
|
||||
print(json.dumps(status(stage), indent=2, sort_keys=True))
|
||||
elif args.command == "promote":
|
||||
print(json.dumps(promote(stage, args.confirm), indent=2, sort_keys=True))
|
||||
print(
|
||||
json.dumps(
|
||||
promote(stage, args.confirm, args.exclude_real_line),
|
||||
indent=2,
|
||||
sort_keys=True,
|
||||
)
|
||||
)
|
||||
else:
|
||||
print(train_commands(args.python))
|
||||
except (
|
||||
|
||||
@@ -185,7 +185,18 @@ class RebuildSolHighTests(unittest.TestCase):
|
||||
history = root / "history.jsonl"
|
||||
purpose_data.write_jsonl(history, [{"prompt": "Add a private cache layer"}])
|
||||
swe = root / "swe.jsonl"
|
||||
purpose_data.write_jsonl(swe, [{"prompt": "Diagnose a unique worker crash"}])
|
||||
duplicate_words = [f"duplicate{index}" for index in range(100)]
|
||||
first_duplicate = " ".join(duplicate_words)
|
||||
duplicate_words[50] = "replacement"
|
||||
second_duplicate = " ".join(duplicate_words)
|
||||
purpose_data.write_jsonl(
|
||||
swe,
|
||||
[
|
||||
{"prompt": "Diagnose a unique worker crash"},
|
||||
{"prompt": first_duplicate},
|
||||
{"prompt": second_duplicate},
|
||||
],
|
||||
)
|
||||
stage = script_dir / ".artifacts" / "sol-high-reset"
|
||||
public_dataset = script_dir / ".artifacts" / "dataset-public"
|
||||
combined_dataset = script_dir / ".artifacts" / "dataset-v1"
|
||||
@@ -222,7 +233,10 @@ class RebuildSolHighTests(unittest.TestCase):
|
||||
(item for item in records if item["prompt"] == prompt),
|
||||
None,
|
||||
)
|
||||
record = original or source_record(prompt)
|
||||
record = original or source_record(
|
||||
prompt,
|
||||
"writing" if prompt == second_duplicate else "backendImpl",
|
||||
)
|
||||
states.append(
|
||||
{
|
||||
"schemaVersion": 1,
|
||||
@@ -240,10 +254,12 @@ class RebuildSolHighTests(unittest.TestCase):
|
||||
result = rebuild_sol_high.promote(
|
||||
stage,
|
||||
rebuild_sol_high.CONFIRMATION,
|
||||
excluded_real_lines=[3],
|
||||
)
|
||||
|
||||
self.assertEqual(80, result["publicLabeled"])
|
||||
self.assertEqual(2, result["realLabeled"])
|
||||
self.assertEqual(3, result["realLabeled"])
|
||||
self.assertEqual(3, result["realReviewExclusions"][0]["sourceLine"])
|
||||
self.assertTrue((combined_dataset / "train.jsonl").is_file())
|
||||
self.assertGreater(
|
||||
len(purpose_data.load_jsonl(combined_dataset / "train.jsonl")),
|
||||
@@ -256,11 +272,42 @@ class RebuildSolHighTests(unittest.TestCase):
|
||||
"purpose-dataset-sol-high-v2",
|
||||
promoted_manifest["datasetVersion"],
|
||||
)
|
||||
combined_manifest = json.loads(
|
||||
(combined_dataset / "manifest.json").read_text(encoding="utf-8")
|
||||
)
|
||||
self.assertEqual(
|
||||
result["realReviewExclusions"],
|
||||
combined_manifest["reviewedRealExclusions"],
|
||||
)
|
||||
|
||||
def test_promotion_requires_exact_confirmation(self):
|
||||
with self.assertRaisesRegex(purpose_data.DataError, "promotion requires"):
|
||||
rebuild_sol_high.promote(Path("unused"), None)
|
||||
|
||||
def test_reviewed_real_line_exclusions_are_auditable(self):
|
||||
records = [
|
||||
source_record("Plan the cache migration", "planning"),
|
||||
source_record("Review the cache migration", "review"),
|
||||
]
|
||||
|
||||
retained, review = rebuild_sol_high._exclude_reviewed_real_lines(records, [1])
|
||||
|
||||
self.assertEqual([records[1]], retained)
|
||||
self.assertEqual(1, review[0]["sourceLine"])
|
||||
self.assertEqual("planning", review[0]["purpose"])
|
||||
self.assertEqual(
|
||||
"reviewed-near-duplicate-label-conflict", review[0]["reason"]
|
||||
)
|
||||
self.assertEqual(64, len(review[0]["promptHash"]))
|
||||
|
||||
def test_reviewed_real_line_exclusions_reject_invalid_lines(self):
|
||||
records = [source_record("Plan the cache migration", "planning")]
|
||||
|
||||
with self.assertRaisesRegex(purpose_data.DataError, "duplicate line"):
|
||||
rebuild_sol_high._exclude_reviewed_real_lines(records, [1, 1])
|
||||
with self.assertRaisesRegex(purpose_data.DataError, "outside"):
|
||||
rebuild_sol_high._exclude_reviewed_real_lines(records, [2])
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
unittest.main()
|
||||
|
||||
Reference in New Issue
Block a user