Files
nucleic-purpose-classifier/data/generation-manifest.json
T

58 lines
1.8 KiB
JSON

{
"schemaVersion": 1,
"dataset": "purpose-classifier-source-v1",
"canonicalFiles": [
"purpose-prompts.jsonl",
"purpose-prompts-round2.jsonl"
],
"derivedBatchFiles": [
"round2-01.jsonl",
"round2-02.jsonl",
"round2-03.jsonl",
"round2-04.jsonl",
"round2-05.jsonl",
"round2-06.jsonl",
"round2-07.jsonl",
"round2-08.jsonl",
"round2-09.jsonl",
"round2-10.jsonl",
"round2-11.jsonl",
"round2-12.jsonl",
"round2-13.jsonl",
"round2-14.jsonl",
"round2-15.jsonl"
],
"generations": [
{
"file": "purpose-prompts.jsonl",
"model": "mixed frontier-model runs (legacy sol/opus aliases; exact model IDs were not retained)",
"date": "2026-07-29",
"prompt": "../datagen-prompt.md",
"topics": [
"web and mobile",
"backend and data",
"infrastructure and systems",
"developer tooling"
],
"notes": "The canonical aggregate was curated from the original per-model batches. Three exact overlaps with the shipped eval fixture were removed on 2026-07-30."
},
{
"file": "purpose-prompts-round2.jsonl",
"model": "Nucleic frontier-model generator (exact underlying model ID was not retained)",
"date": "2026-07-30",
"prompt": "../datagen-prompt-2.md",
"topics": [
"boundary confusion pairs",
"pasted context",
"mixed intent",
"non-English developer prompts"
],
"notes": "Corrective generation that counterbalances round-one label, slice, length, opener, and language drift."
}
],
"limitations": [
"The exact generator model IDs and sampling parameters were not recorded when the source corpora were created.",
"The round2-NN files are retained generation batches and duplicate the round-two canonical aggregate; dataset tooling must read canonicalFiles only."
]
}