Files
nucleic-purpose-classifier/tests/test_label_nucleic_prompts.py

300 lines
9.9 KiB
Python

import json
import os
import subprocess
import sys
import tempfile
import unittest
from pathlib import Path
MODULE_DIR = Path(__file__).resolve().parents[1]
sys.path.insert(0, str(MODULE_DIR))
import label_nucleic_prompts
import purpose_data
FAKE_CODEX = r"""#!/usr/bin/env python3
import json
import os
import sys
from pathlib import Path
args = sys.argv[1:]
model_index = args.index("--model")
if args[model_index + 1] != os.environ.get("EXPECTED_MODEL", "gpt-5.6-terra"):
raise SystemExit("wrong model")
expected_effort = os.environ.get("EXPECTED_REASONING_EFFORT", "low")
if f'model_reasoning_effort="{expected_effort}"' not in args:
raise SystemExit("wrong reasoning effort")
prompt = sys.stdin.read()
payload_text = prompt.split("<input_json>\n", 1)[1].split("\n</input_json>", 1)[0]
payload = json.loads(payload_text)
items = []
for item in payload["items"]:
if "MODEL_JUNK" in item["prompt"]:
decision = {
"id": item["id"],
"keep": False,
"junkReason": "generated agent output",
"purpose": None,
"secondary": None,
"mixed": None,
"difficulty": None,
"slice": None,
"lang": None,
}
elif "PLAN_AND_BUILD" in item["prompt"]:
decision = {
"id": item["id"],
"keep": True,
"junkReason": None,
"purpose": "planning",
"secondary": "backendImpl",
"mixed": True,
"difficulty": 0.7,
"slice": "mixed",
"lang": "en",
}
else:
decision = {
"id": item["id"],
"keep": True,
"junkReason": None,
"purpose": "writing",
"secondary": None,
"mixed": False,
"difficulty": 0.3,
"slice": "core",
"lang": "en",
}
items.append(decision)
response_path = Path(args[args.index("--output-last-message") + 1])
response_path.write_text(json.dumps({"items": items}), encoding="utf-8")
with Path(os.environ["FAKE_CODEX_LOG"]).open("a", encoding="utf-8") as handle:
handle.write("called\n")
print(json.dumps({"items": items}))
"""
class LabelNucleicPromptsTests(unittest.TestCase):
def make_fake_codex(self, root: Path) -> Path:
path = root / "codex"
path.write_text(FAKE_CODEX, encoding="utf-8")
path.chmod(0o755)
return path
def run_script(
self,
root: Path,
input_path: Path,
output_path: Path,
fake_codex: Path,
*extra: str,
) -> subprocess.CompletedProcess[str]:
environment = os.environ.copy()
environment["FAKE_CODEX_LOG"] = str(root / "calls.log")
if "--model" in extra:
environment["EXPECTED_MODEL"] = extra[extra.index("--model") + 1]
if "--reasoning-effort" in extra:
environment["EXPECTED_REASONING_EFFORT"] = extra[
extra.index("--reasoning-effort") + 1
]
return subprocess.run(
[
sys.executable,
str(MODULE_DIR / "label_nucleic_prompts.py"),
"--input",
str(input_path),
"--output",
str(output_path),
"--codex",
str(fake_codex),
"--batch-size",
"2",
*extra,
],
text=True,
stdout=subprocess.PIPE,
stderr=subprocess.PIPE,
env=environment,
check=False,
)
def test_filters_labels_and_writes_exact_training_contract(self):
with tempfile.TemporaryDirectory() as directory:
root = Path(directory)
fake_codex = self.make_fake_codex(root)
input_path = root / "unlabeled.jsonl"
input_path.write_text(
"\n".join(
[
"",
"not json",
json.dumps(
{
"sessionID": "s1",
"prompt": "Write the API documentation",
"purpose": None,
}
),
json.dumps(
{
"sessionID": "s2",
"prompt": " write THE api documentation ",
"purpose": None,
}
),
json.dumps(
{"sessionID": "s3", "prompt": "MODEL_JUNK transcript"}
),
json.dumps(
{
"sessionID": "s4",
"prompt": "PLAN_AND_BUILD the queue migration",
}
),
]
)
+ "\n",
encoding="utf-8",
)
output_path = root / "labeled.jsonl"
result = self.run_script(
root,
input_path,
output_path,
fake_codex,
)
self.assertEqual(0, result.returncode, result.stderr)
records = purpose_data.load_jsonl(output_path)
self.assertEqual(2, len(records))
self.assertEqual(
"Write the API documentation",
records[0]["prompt"],
)
self.assertEqual("writing", records[0]["purpose"])
self.assertEqual("planning", records[1]["purpose"])
self.assertEqual("backendImpl", records[1]["secondary"])
self.assertTrue(records[1]["mixed"])
for index, record in enumerate(records, start=1):
purpose_data.validate_source_record(record, f"record {index}")
self.assertEqual(purpose_data.SOURCE_FIELDS, set(record))
rejects = purpose_data.load_jsonl(
label_nucleic_prompts.rejects_path_for(output_path)
)
self.assertEqual(4, len(rejects))
reasons = {record["reason"] for record in rejects}
self.assertIn("blank_line", reasons)
self.assertIn("invalid_json", reasons)
self.assertIn("duplicate_prompt_of_line_3", reasons)
self.assertIn("semantic_junk:generated agent output", reasons)
self.assertTrue(
label_nucleic_prompts.state_path_for(output_path).is_file()
)
def test_resume_reuses_completed_state_without_calling_codex(self):
with tempfile.TemporaryDirectory() as directory:
root = Path(directory)
fake_codex = self.make_fake_codex(root)
input_path = root / "unlabeled.jsonl"
input_path.write_text(
json.dumps({"prompt": "Write the migration note"}) + "\n",
encoding="utf-8",
)
output_path = root / "labeled.jsonl"
first = self.run_script(
root,
input_path,
output_path,
fake_codex,
)
second = self.run_script(
root,
input_path,
output_path,
fake_codex,
"--resume",
)
self.assertEqual(0, first.returncode, first.stderr)
self.assertEqual(0, second.returncode, second.stderr)
self.assertEqual(
["called"],
(root / "calls.log").read_text(encoding="utf-8").splitlines(),
)
self.assertIn("pending=0", second.stdout)
def test_custom_model_and_reasoning_effort_are_forwarded(self):
with tempfile.TemporaryDirectory() as directory:
root = Path(directory)
fake_codex = self.make_fake_codex(root)
input_path = root / "unlabeled.jsonl"
input_path.write_text(
json.dumps({"prompt": "Write the migration note"}) + "\n",
encoding="utf-8",
)
result = self.run_script(
root,
input_path,
root / "labeled.jsonl",
fake_codex,
"--model",
"gpt-5.6-sol",
"--reasoning-effort",
"high",
)
self.assertEqual(0, result.returncode, result.stderr)
self.assertIn("model=gpt-5.6-sol reasoning=high", result.stdout)
def test_head_tail_excerpt_is_bounded(self):
prompt = "HEAD" + ("x" * 4_000) + "TAIL"
excerpt = label_nucleic_prompts.excerpt_for_labeling(prompt, 1_000)
self.assertEqual(1_000, len(excerpt))
self.assertTrue(excerpt.startswith("HEAD"))
self.assertTrue(excerpt.endswith("TAIL"))
self.assertIn("characters omitted for labeling", excerpt)
def test_retained_decision_must_obey_dataset_contract(self):
source = label_nucleic_prompts.SourceLine(1, "{}", "hash")
candidate = label_nucleic_prompts.Candidate(
source=source,
prompt="Plan and build it",
prompt_hash="a" * 64,
session_id=None,
)
response = {
"items": [
{
"id": candidate.id,
"keep": True,
"junkReason": None,
"purpose": "planning",
"secondary": "backendImpl",
"mixed": False,
"difficulty": 0.7,
"slice": "core",
"lang": "en",
}
]
}
with self.assertRaisesRegex(
purpose_data.DataError,
"mixed and secondary disagree",
):
label_nucleic_prompts.validate_decisions([candidate], response)
if __name__ == "__main__":
unittest.main()