Merge nucleic/fuzzy-dewy-urchin-hpvd into dev
This commit is contained in:
@@ -59,6 +59,21 @@ lines. Once `status` reports `"complete": true`, promote the staged result expli
|
|||||||
--confirm overwrite-all-labels-with-sol-high
|
--confirm overwrite-all-labels-with-sol-high
|
||||||
```
|
```
|
||||||
|
|
||||||
|
If promotion reports a near-duplicate label conflict in the combined real-session
|
||||||
|
population, inspect both prompts and exclude the reviewed bad/ambiguous copy explicitly.
|
||||||
|
The 1-based line is relative to the reported `combined-real.labeled.jsonl`; repeat the
|
||||||
|
option for multiple decisions:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
"$PY" ml/purpose-classifier/rebuild_sol_high.py promote \
|
||||||
|
--confirm overwrite-all-labels-with-sol-high \
|
||||||
|
--exclude-real-line <line>
|
||||||
|
```
|
||||||
|
|
||||||
|
Each exclusion is bound to the prompt hash, prior purpose, and review reason in the
|
||||||
|
combined dataset manifest and promotion result. This option does not change a teacher
|
||||||
|
label or relax duplicate detection.
|
||||||
|
|
||||||
Promotion first validates and curates the complete staged population. It then replaces
|
Promotion first validates and curates the complete staged population. It then replaces
|
||||||
the two canonical source files, regenerates all round-two mirrors, relabels shipped
|
the two canonical source files, regenerates all round-two mirrors, relabels shipped
|
||||||
fixtures (teacher-rejected fixtures become the runtime `general` fallback), rebuilds the
|
fixtures (teacher-rejected fixtures become the runtime `general` fallback), rebuilds the
|
||||||
|
|||||||
+62
-3
@@ -34,6 +34,7 @@ from purpose_data import (
|
|||||||
file_sha256,
|
file_sha256,
|
||||||
jsonl_bytes,
|
jsonl_bytes,
|
||||||
load_jsonl,
|
load_jsonl,
|
||||||
|
prompt_hash,
|
||||||
validate_source_record,
|
validate_source_record,
|
||||||
write_json,
|
write_json,
|
||||||
write_jsonl,
|
write_jsonl,
|
||||||
@@ -463,7 +464,41 @@ def _replace_transaction(candidates: Sequence[tuple[Path, Path]], backup_root: P
|
|||||||
raise
|
raise
|
||||||
|
|
||||||
|
|
||||||
def promote(stage: Path, confirmation: str | None) -> dict[str, Any]:
|
def _exclude_reviewed_real_lines(
|
||||||
|
records: Sequence[dict[str, Any]],
|
||||||
|
source_lines: Sequence[int],
|
||||||
|
) -> tuple[list[dict[str, Any]], list[dict[str, Any]]]:
|
||||||
|
unique_lines = sorted(set(source_lines))
|
||||||
|
if len(unique_lines) != len(source_lines):
|
||||||
|
raise DataError("--exclude-real-line contains a duplicate line")
|
||||||
|
for line in unique_lines:
|
||||||
|
if not 1 <= line <= len(records):
|
||||||
|
raise DataError(
|
||||||
|
f"--exclude-real-line {line} is outside the combined real population "
|
||||||
|
f"(1-{len(records)})"
|
||||||
|
)
|
||||||
|
|
||||||
|
excluded = set(unique_lines)
|
||||||
|
review = [
|
||||||
|
{
|
||||||
|
"sourceLine": line,
|
||||||
|
"promptHash": prompt_hash(records[line - 1]["prompt"]),
|
||||||
|
"purpose": records[line - 1]["purpose"],
|
||||||
|
"reason": "reviewed-near-duplicate-label-conflict",
|
||||||
|
}
|
||||||
|
for line in unique_lines
|
||||||
|
]
|
||||||
|
return (
|
||||||
|
[record for line, record in enumerate(records, 1) if line not in excluded],
|
||||||
|
review,
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def promote(
|
||||||
|
stage: Path,
|
||||||
|
confirmation: str | None,
|
||||||
|
excluded_real_lines: Sequence[int] = (),
|
||||||
|
) -> dict[str, Any]:
|
||||||
if confirmation != CONFIRMATION:
|
if confirmation != CONFIRMATION:
|
||||||
raise DataError(f"promotion requires --confirm {CONFIRMATION}")
|
raise DataError(f"promotion requires --confirm {CONFIRMATION}")
|
||||||
load_config(stage)
|
load_config(stage)
|
||||||
@@ -514,9 +549,12 @@ def promote(stage: Path, confirmation: str | None) -> dict[str, Any]:
|
|||||||
state_by_name["fixtures"],
|
state_by_name["fixtures"],
|
||||||
)
|
)
|
||||||
combined_real = promotion / "combined-real.labeled.jsonl"
|
combined_real = promotion / "combined-real.labeled.jsonl"
|
||||||
real_records = _records_from_states(
|
all_real_records = _records_from_states(
|
||||||
state_by_name["history"]
|
state_by_name["history"]
|
||||||
) + _records_from_states(state_by_name["swe"])
|
) + _records_from_states(state_by_name["swe"])
|
||||||
|
real_records, real_review = _exclude_reviewed_real_lines(
|
||||||
|
all_real_records, excluded_real_lines
|
||||||
|
)
|
||||||
write_jsonl(combined_real, real_records)
|
write_jsonl(combined_real, real_records)
|
||||||
|
|
||||||
public_dataset = promotion / "dataset-public"
|
public_dataset = promotion / "dataset-public"
|
||||||
@@ -558,6 +596,10 @@ def promote(stage: Path, confirmation: str | None) -> dict[str, Any]:
|
|||||||
combined_manifest["policy"]["historyUsage"] = (
|
combined_manifest["policy"]["historyUsage"] = (
|
||||||
"Nucleic history and SWE-chat are training-only"
|
"Nucleic history and SWE-chat are training-only"
|
||||||
)
|
)
|
||||||
|
combined_manifest["policy"]["reviewedRealLabelConflicts"] = (
|
||||||
|
"explicitly excluded before near-duplicate curation"
|
||||||
|
)
|
||||||
|
combined_manifest["reviewedRealExclusions"] = real_review
|
||||||
combined_manifest["sources"]["baseDataset"]["path"] = _relative(
|
combined_manifest["sources"]["baseDataset"]["path"] = _relative(
|
||||||
PUBLIC_DATASET_DESTINATION
|
PUBLIC_DATASET_DESTINATION
|
||||||
)
|
)
|
||||||
@@ -608,6 +650,7 @@ def promote(stage: Path, confirmation: str | None) -> dict[str, Any]:
|
|||||||
"teacher": {"model": MODEL, "reasoningEffort": REASONING_EFFORT},
|
"teacher": {"model": MODEL, "reasoningEffort": REASONING_EFFORT},
|
||||||
"publicLabeled": sum(len(value) for value in public_records.values()),
|
"publicLabeled": sum(len(value) for value in public_records.values()),
|
||||||
"realLabeled": len(real_records),
|
"realLabeled": len(real_records),
|
||||||
|
"realReviewExclusions": real_review,
|
||||||
"combinedTrain": combined_manifest["outputs"]["train"]["records"],
|
"combinedTrain": combined_manifest["outputs"]["train"]["records"],
|
||||||
"validation": combined_manifest["outputs"]["validation"]["records"],
|
"validation": combined_manifest["outputs"]["validation"]["records"],
|
||||||
"test": combined_manifest["outputs"]["test"]["records"],
|
"test": combined_manifest["outputs"]["test"]["records"],
|
||||||
@@ -663,6 +706,16 @@ def build_parser() -> argparse.ArgumentParser:
|
|||||||
"promote", help="curate and replace canonical labels"
|
"promote", help="curate and replace canonical labels"
|
||||||
)
|
)
|
||||||
promote_parser.add_argument("--confirm")
|
promote_parser.add_argument("--confirm")
|
||||||
|
promote_parser.add_argument(
|
||||||
|
"--exclude-real-line",
|
||||||
|
action="append",
|
||||||
|
default=[],
|
||||||
|
type=int,
|
||||||
|
help=(
|
||||||
|
"exclude a reviewed 1-based line from the combined history+SWE training "
|
||||||
|
"augmentation; repeat for each adjudicated conflict"
|
||||||
|
),
|
||||||
|
)
|
||||||
commands_parser = subparsers.add_parser(
|
commands_parser = subparsers.add_parser(
|
||||||
"train-commands", help="print clean from-base training commands"
|
"train-commands", help="print clean from-base training commands"
|
||||||
)
|
)
|
||||||
@@ -701,7 +754,13 @@ def main(argv: Sequence[str] | None = None) -> int:
|
|||||||
elif args.command == "status":
|
elif args.command == "status":
|
||||||
print(json.dumps(status(stage), indent=2, sort_keys=True))
|
print(json.dumps(status(stage), indent=2, sort_keys=True))
|
||||||
elif args.command == "promote":
|
elif args.command == "promote":
|
||||||
print(json.dumps(promote(stage, args.confirm), indent=2, sort_keys=True))
|
print(
|
||||||
|
json.dumps(
|
||||||
|
promote(stage, args.confirm, args.exclude_real_line),
|
||||||
|
indent=2,
|
||||||
|
sort_keys=True,
|
||||||
|
)
|
||||||
|
)
|
||||||
else:
|
else:
|
||||||
print(train_commands(args.python))
|
print(train_commands(args.python))
|
||||||
except (
|
except (
|
||||||
|
|||||||
@@ -185,7 +185,18 @@ class RebuildSolHighTests(unittest.TestCase):
|
|||||||
history = root / "history.jsonl"
|
history = root / "history.jsonl"
|
||||||
purpose_data.write_jsonl(history, [{"prompt": "Add a private cache layer"}])
|
purpose_data.write_jsonl(history, [{"prompt": "Add a private cache layer"}])
|
||||||
swe = root / "swe.jsonl"
|
swe = root / "swe.jsonl"
|
||||||
purpose_data.write_jsonl(swe, [{"prompt": "Diagnose a unique worker crash"}])
|
duplicate_words = [f"duplicate{index}" for index in range(100)]
|
||||||
|
first_duplicate = " ".join(duplicate_words)
|
||||||
|
duplicate_words[50] = "replacement"
|
||||||
|
second_duplicate = " ".join(duplicate_words)
|
||||||
|
purpose_data.write_jsonl(
|
||||||
|
swe,
|
||||||
|
[
|
||||||
|
{"prompt": "Diagnose a unique worker crash"},
|
||||||
|
{"prompt": first_duplicate},
|
||||||
|
{"prompt": second_duplicate},
|
||||||
|
],
|
||||||
|
)
|
||||||
stage = script_dir / ".artifacts" / "sol-high-reset"
|
stage = script_dir / ".artifacts" / "sol-high-reset"
|
||||||
public_dataset = script_dir / ".artifacts" / "dataset-public"
|
public_dataset = script_dir / ".artifacts" / "dataset-public"
|
||||||
combined_dataset = script_dir / ".artifacts" / "dataset-v1"
|
combined_dataset = script_dir / ".artifacts" / "dataset-v1"
|
||||||
@@ -222,7 +233,10 @@ class RebuildSolHighTests(unittest.TestCase):
|
|||||||
(item for item in records if item["prompt"] == prompt),
|
(item for item in records if item["prompt"] == prompt),
|
||||||
None,
|
None,
|
||||||
)
|
)
|
||||||
record = original or source_record(prompt)
|
record = original or source_record(
|
||||||
|
prompt,
|
||||||
|
"writing" if prompt == second_duplicate else "backendImpl",
|
||||||
|
)
|
||||||
states.append(
|
states.append(
|
||||||
{
|
{
|
||||||
"schemaVersion": 1,
|
"schemaVersion": 1,
|
||||||
@@ -240,10 +254,12 @@ class RebuildSolHighTests(unittest.TestCase):
|
|||||||
result = rebuild_sol_high.promote(
|
result = rebuild_sol_high.promote(
|
||||||
stage,
|
stage,
|
||||||
rebuild_sol_high.CONFIRMATION,
|
rebuild_sol_high.CONFIRMATION,
|
||||||
|
excluded_real_lines=[3],
|
||||||
)
|
)
|
||||||
|
|
||||||
self.assertEqual(80, result["publicLabeled"])
|
self.assertEqual(80, result["publicLabeled"])
|
||||||
self.assertEqual(2, result["realLabeled"])
|
self.assertEqual(3, result["realLabeled"])
|
||||||
|
self.assertEqual(3, result["realReviewExclusions"][0]["sourceLine"])
|
||||||
self.assertTrue((combined_dataset / "train.jsonl").is_file())
|
self.assertTrue((combined_dataset / "train.jsonl").is_file())
|
||||||
self.assertGreater(
|
self.assertGreater(
|
||||||
len(purpose_data.load_jsonl(combined_dataset / "train.jsonl")),
|
len(purpose_data.load_jsonl(combined_dataset / "train.jsonl")),
|
||||||
@@ -256,11 +272,42 @@ class RebuildSolHighTests(unittest.TestCase):
|
|||||||
"purpose-dataset-sol-high-v2",
|
"purpose-dataset-sol-high-v2",
|
||||||
promoted_manifest["datasetVersion"],
|
promoted_manifest["datasetVersion"],
|
||||||
)
|
)
|
||||||
|
combined_manifest = json.loads(
|
||||||
|
(combined_dataset / "manifest.json").read_text(encoding="utf-8")
|
||||||
|
)
|
||||||
|
self.assertEqual(
|
||||||
|
result["realReviewExclusions"],
|
||||||
|
combined_manifest["reviewedRealExclusions"],
|
||||||
|
)
|
||||||
|
|
||||||
def test_promotion_requires_exact_confirmation(self):
|
def test_promotion_requires_exact_confirmation(self):
|
||||||
with self.assertRaisesRegex(purpose_data.DataError, "promotion requires"):
|
with self.assertRaisesRegex(purpose_data.DataError, "promotion requires"):
|
||||||
rebuild_sol_high.promote(Path("unused"), None)
|
rebuild_sol_high.promote(Path("unused"), None)
|
||||||
|
|
||||||
|
def test_reviewed_real_line_exclusions_are_auditable(self):
|
||||||
|
records = [
|
||||||
|
source_record("Plan the cache migration", "planning"),
|
||||||
|
source_record("Review the cache migration", "review"),
|
||||||
|
]
|
||||||
|
|
||||||
|
retained, review = rebuild_sol_high._exclude_reviewed_real_lines(records, [1])
|
||||||
|
|
||||||
|
self.assertEqual([records[1]], retained)
|
||||||
|
self.assertEqual(1, review[0]["sourceLine"])
|
||||||
|
self.assertEqual("planning", review[0]["purpose"])
|
||||||
|
self.assertEqual(
|
||||||
|
"reviewed-near-duplicate-label-conflict", review[0]["reason"]
|
||||||
|
)
|
||||||
|
self.assertEqual(64, len(review[0]["promptHash"]))
|
||||||
|
|
||||||
|
def test_reviewed_real_line_exclusions_reject_invalid_lines(self):
|
||||||
|
records = [source_record("Plan the cache migration", "planning")]
|
||||||
|
|
||||||
|
with self.assertRaisesRegex(purpose_data.DataError, "duplicate line"):
|
||||||
|
rebuild_sol_high._exclude_reviewed_real_lines(records, [1, 1])
|
||||||
|
with self.assertRaisesRegex(purpose_data.DataError, "outside"):
|
||||||
|
rebuild_sol_high._exclude_reviewed_real_lines(records, [2])
|
||||||
|
|
||||||
|
|
||||||
if __name__ == "__main__":
|
if __name__ == "__main__":
|
||||||
unittest.main()
|
unittest.main()
|
||||||
|
|||||||
Reference in New Issue
Block a user