Merge nucleic/dusky-ancient-shrew-r8gj into dev
This commit is contained in:
@@ -0,0 +1,268 @@
|
||||
import json
|
||||
import os
|
||||
import subprocess
|
||||
import sys
|
||||
import tempfile
|
||||
import unittest
|
||||
from pathlib import Path
|
||||
|
||||
|
||||
MODULE_DIR = Path(__file__).resolve().parents[1]
|
||||
sys.path.insert(0, str(MODULE_DIR))
|
||||
|
||||
import label_nucleic_prompts
|
||||
import purpose_data
|
||||
|
||||
|
||||
FAKE_CODEX = r"""#!/usr/bin/env python3
|
||||
import json
|
||||
import os
|
||||
import sys
|
||||
from pathlib import Path
|
||||
|
||||
args = sys.argv[1:]
|
||||
model_index = args.index("--model")
|
||||
if args[model_index + 1] != "gpt-5.6-terra":
|
||||
raise SystemExit("wrong model")
|
||||
if 'model_reasoning_effort="low"' not in args:
|
||||
raise SystemExit("wrong reasoning effort")
|
||||
|
||||
prompt = sys.stdin.read()
|
||||
payload_text = prompt.split("<input_json>\n", 1)[1].split("\n</input_json>", 1)[0]
|
||||
payload = json.loads(payload_text)
|
||||
items = []
|
||||
for item in payload["items"]:
|
||||
if "MODEL_JUNK" in item["prompt"]:
|
||||
decision = {
|
||||
"id": item["id"],
|
||||
"keep": False,
|
||||
"junkReason": "generated agent output",
|
||||
"purpose": None,
|
||||
"secondary": None,
|
||||
"mixed": None,
|
||||
"difficulty": None,
|
||||
"slice": None,
|
||||
"lang": None,
|
||||
}
|
||||
elif "PLAN_AND_BUILD" in item["prompt"]:
|
||||
decision = {
|
||||
"id": item["id"],
|
||||
"keep": True,
|
||||
"junkReason": None,
|
||||
"purpose": "planning",
|
||||
"secondary": "backendImpl",
|
||||
"mixed": True,
|
||||
"difficulty": 0.7,
|
||||
"slice": "mixed",
|
||||
"lang": "en",
|
||||
}
|
||||
else:
|
||||
decision = {
|
||||
"id": item["id"],
|
||||
"keep": True,
|
||||
"junkReason": None,
|
||||
"purpose": "writing",
|
||||
"secondary": None,
|
||||
"mixed": False,
|
||||
"difficulty": 0.3,
|
||||
"slice": "core",
|
||||
"lang": "en",
|
||||
}
|
||||
items.append(decision)
|
||||
|
||||
response_path = Path(args[args.index("--output-last-message") + 1])
|
||||
response_path.write_text(json.dumps({"items": items}), encoding="utf-8")
|
||||
with Path(os.environ["FAKE_CODEX_LOG"]).open("a", encoding="utf-8") as handle:
|
||||
handle.write("called\n")
|
||||
print(json.dumps({"items": items}))
|
||||
"""
|
||||
|
||||
|
||||
class LabelNucleicPromptsTests(unittest.TestCase):
|
||||
def make_fake_codex(self, root: Path) -> Path:
|
||||
path = root / "codex"
|
||||
path.write_text(FAKE_CODEX, encoding="utf-8")
|
||||
path.chmod(0o755)
|
||||
return path
|
||||
|
||||
def run_script(
|
||||
self,
|
||||
root: Path,
|
||||
input_path: Path,
|
||||
output_path: Path,
|
||||
fake_codex: Path,
|
||||
*extra: str,
|
||||
) -> subprocess.CompletedProcess[str]:
|
||||
environment = os.environ.copy()
|
||||
environment["FAKE_CODEX_LOG"] = str(root / "calls.log")
|
||||
return subprocess.run(
|
||||
[
|
||||
sys.executable,
|
||||
str(MODULE_DIR / "label_nucleic_prompts.py"),
|
||||
"--input",
|
||||
str(input_path),
|
||||
"--output",
|
||||
str(output_path),
|
||||
"--codex",
|
||||
str(fake_codex),
|
||||
"--batch-size",
|
||||
"2",
|
||||
*extra,
|
||||
],
|
||||
text=True,
|
||||
stdout=subprocess.PIPE,
|
||||
stderr=subprocess.PIPE,
|
||||
env=environment,
|
||||
check=False,
|
||||
)
|
||||
|
||||
def test_filters_labels_and_writes_exact_training_contract(self):
|
||||
with tempfile.TemporaryDirectory() as directory:
|
||||
root = Path(directory)
|
||||
fake_codex = self.make_fake_codex(root)
|
||||
input_path = root / "unlabeled.jsonl"
|
||||
input_path.write_text(
|
||||
"\n".join(
|
||||
[
|
||||
"",
|
||||
"not json",
|
||||
json.dumps(
|
||||
{
|
||||
"sessionID": "s1",
|
||||
"prompt": "Write the API documentation",
|
||||
"purpose": None,
|
||||
}
|
||||
),
|
||||
json.dumps(
|
||||
{
|
||||
"sessionID": "s2",
|
||||
"prompt": " write THE api documentation ",
|
||||
"purpose": None,
|
||||
}
|
||||
),
|
||||
json.dumps(
|
||||
{"sessionID": "s3", "prompt": "MODEL_JUNK transcript"}
|
||||
),
|
||||
json.dumps(
|
||||
{
|
||||
"sessionID": "s4",
|
||||
"prompt": "PLAN_AND_BUILD the queue migration",
|
||||
}
|
||||
),
|
||||
]
|
||||
)
|
||||
+ "\n",
|
||||
encoding="utf-8",
|
||||
)
|
||||
output_path = root / "labeled.jsonl"
|
||||
|
||||
result = self.run_script(
|
||||
root,
|
||||
input_path,
|
||||
output_path,
|
||||
fake_codex,
|
||||
)
|
||||
|
||||
self.assertEqual(0, result.returncode, result.stderr)
|
||||
records = purpose_data.load_jsonl(output_path)
|
||||
self.assertEqual(2, len(records))
|
||||
self.assertEqual(
|
||||
"Write the API documentation",
|
||||
records[0]["prompt"],
|
||||
)
|
||||
self.assertEqual("writing", records[0]["purpose"])
|
||||
self.assertEqual("planning", records[1]["purpose"])
|
||||
self.assertEqual("backendImpl", records[1]["secondary"])
|
||||
self.assertTrue(records[1]["mixed"])
|
||||
for index, record in enumerate(records, start=1):
|
||||
purpose_data.validate_source_record(record, f"record {index}")
|
||||
self.assertEqual(purpose_data.SOURCE_FIELDS, set(record))
|
||||
|
||||
rejects = purpose_data.load_jsonl(
|
||||
label_nucleic_prompts.rejects_path_for(output_path)
|
||||
)
|
||||
self.assertEqual(4, len(rejects))
|
||||
reasons = {record["reason"] for record in rejects}
|
||||
self.assertIn("blank_line", reasons)
|
||||
self.assertIn("invalid_json", reasons)
|
||||
self.assertIn("duplicate_prompt_of_line_3", reasons)
|
||||
self.assertIn("semantic_junk:generated agent output", reasons)
|
||||
self.assertTrue(
|
||||
label_nucleic_prompts.state_path_for(output_path).is_file()
|
||||
)
|
||||
|
||||
def test_resume_reuses_completed_state_without_calling_codex(self):
|
||||
with tempfile.TemporaryDirectory() as directory:
|
||||
root = Path(directory)
|
||||
fake_codex = self.make_fake_codex(root)
|
||||
input_path = root / "unlabeled.jsonl"
|
||||
input_path.write_text(
|
||||
json.dumps({"prompt": "Write the migration note"}) + "\n",
|
||||
encoding="utf-8",
|
||||
)
|
||||
output_path = root / "labeled.jsonl"
|
||||
|
||||
first = self.run_script(
|
||||
root,
|
||||
input_path,
|
||||
output_path,
|
||||
fake_codex,
|
||||
)
|
||||
second = self.run_script(
|
||||
root,
|
||||
input_path,
|
||||
output_path,
|
||||
fake_codex,
|
||||
"--resume",
|
||||
)
|
||||
|
||||
self.assertEqual(0, first.returncode, first.stderr)
|
||||
self.assertEqual(0, second.returncode, second.stderr)
|
||||
self.assertEqual(
|
||||
["called"],
|
||||
(root / "calls.log").read_text(encoding="utf-8").splitlines(),
|
||||
)
|
||||
self.assertIn("pending=0", second.stdout)
|
||||
|
||||
def test_head_tail_excerpt_is_bounded(self):
|
||||
prompt = "HEAD" + ("x" * 4_000) + "TAIL"
|
||||
excerpt = label_nucleic_prompts.excerpt_for_labeling(prompt, 1_000)
|
||||
|
||||
self.assertEqual(1_000, len(excerpt))
|
||||
self.assertTrue(excerpt.startswith("HEAD"))
|
||||
self.assertTrue(excerpt.endswith("TAIL"))
|
||||
self.assertIn("characters omitted for labeling", excerpt)
|
||||
|
||||
def test_retained_decision_must_obey_dataset_contract(self):
|
||||
source = label_nucleic_prompts.SourceLine(1, "{}", "hash")
|
||||
candidate = label_nucleic_prompts.Candidate(
|
||||
source=source,
|
||||
prompt="Plan and build it",
|
||||
prompt_hash="a" * 64,
|
||||
session_id=None,
|
||||
)
|
||||
response = {
|
||||
"items": [
|
||||
{
|
||||
"id": candidate.id,
|
||||
"keep": True,
|
||||
"junkReason": None,
|
||||
"purpose": "planning",
|
||||
"secondary": "backendImpl",
|
||||
"mixed": False,
|
||||
"difficulty": 0.7,
|
||||
"slice": "core",
|
||||
"lang": "en",
|
||||
}
|
||||
]
|
||||
}
|
||||
|
||||
with self.assertRaisesRegex(
|
||||
purpose_data.DataError,
|
||||
"mixed and secondary disagree",
|
||||
):
|
||||
label_nucleic_prompts.validate_decisions([candidate], response)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
unittest.main()
|
||||
Reference in New Issue
Block a user