{ "schemaVersion": 1, "experiment": "purpose-sol-high-v2-training", "recordedAt": "2026-08-03", "sourceCheckoutCommit": "531e9ba301a20dd124062afed47aa412244c407a", "status": "no-qualified-sol-high-v2-candidate", "summary": "The Sol-high label reset and promotion completed. The selected lite float checkpoint reached 93.7167% frozen accuracy, while the best deep validation checkpoint reached 77.6445%. Neither meets the shipping contract; frozen-set-driven tuning and further deep continuations are stopped.", "privacy": { "policy": "Private Nucleic-history and SWE-chat prompts remain in ignored artifacts.", "recordedIdentifiers": "Only aggregate counts, content hashes, reviewed source lines, and reviewed prompt hashes are versioned here." }, "labelReset": { "teacher": { "model": "gpt-5.6-sol", "reasoningEffort": "high" }, "inputDecisions": { "total": 15048, "labeled": 14704, "rejected": 344, "populations": { "canonicalPublic": { "labeled": 12007, "rejected": 207 }, "shippedFixtures": { "labeled": 89, "rejectedToGeneral": 3 }, "nucleicHistory": { "labeled": 1068, "rejected": 68 }, "sweChat": { "labeled": 1540, "rejected": 66 } } }, "promotion": { "backup": "ml/purpose-classifier/.artifacts/sol-high-reset/backups/20260802T234020Z", "publicLabeled": 12007, "realLabeledAfterReviewedExclusions": 2606, "combinedTrain": 12164, "validation": 1208, "syntheticTest": 1119, "reviewedRealExclusions": [ { "sourceLine": 1201, "promptHash": "124e98f338b935a7efd0d4e12295eb59f07ee68e4dc8ab65bc069b630ee93110", "priorPurpose": "planning", "reason": "reviewed-near-duplicate-label-conflict" }, { "sourceLine": 1481, "promptHash": "a6dd960a0857f2d8af5ef971582e480456a33599df8af27d50cddb5e9f828035", "priorPurpose": "refactor", "reason": "reviewed-near-duplicate-label-conflict" } ], "validationCommandResult": { "records": 12007, "errors": 0, "expectedShapeWarnings": 22 } } }, "dataset": { "version": "purpose-dataset-sol-high-v2", "public": { "inputRecords": 12007, "nearDuplicateExclusions": 18, "retainedRecords": 11989, "train": { "records": 9662, "sha256": "0e16940308dc7557c73b1b804bf7b715c86d263b613201588ec274169ee01e89" }, "validation": { "records": 1208, "scorableRecords": 917, "vagueEvalRecords": 291, "sha256": "301cd3d1c69e1bb92b811e857093ad12fb8c5c11bb402ee2993403d5e4467969" }, "syntheticTest": { "records": 1119, "vagueEvalRecords": 269, "sha256": "9997dbaa6d305ea56d8c161e39b925368761ab613a706eb8c2ba55d0e58b1906" }, "shippedFixtures": { "records": 92, "classifiableRecords": 89, "generalRecords": 3, "sha256": "876068ea26d7109365bcfbd1dc44ba3f1a0400c80dc7fdacf008b3fa07cd92cc" }, "logicalFrozenRecords": 1208, "scorableFrozenRecords": 939 }, "combinedTrainingArtifact": { "experimentVersion": "all-sol-high-training-augmentation-v2", "train": { "records": 12164, "sha256": "44a4bbccf2c8207f7f497e8138ee01007c687034a7c719604f50cbeba35e5bcc" }, "validation": { "records": 1208, "sha256": "301cd3d1c69e1bb92b811e857093ad12fb8c5c11bb402ee2993403d5e4467969" }, "syntheticTest": { "records": 1119, "sha256": "9997dbaa6d305ea56d8c161e39b925368761ab613a706eb8c2ba55d0e58b1906" }, "realInputAfterReviewedExclusions": { "records": 2606, "sha256": "ed38f4ffdb640a2a9decf1407c23ff8c8c8a150bbe1f55b1b051033d5bcc1ade" }, "acceptedRealAugmentation": 2502, "excludedRealAugmentation": { "total": 104, "vagueEval": 87, "nearHistoryDuplicate": 17 } }, "manifestHashes": { "datasetV1Manifest": "c6be5d0df6fd87ab3f5a254d3b8e280cae786cdc90e3cd0cdee02c6662dd27b1", "generationManifest": "bb707c81047cb1843562d1641abe344494ef0c4aab5af7ae56283d6769160b2a", "supersededCurationReview": "f8b47a08a10ff872f2a29e6f3321d4818ab439a2429d7162692b58c9281fdf47", "ignoredCombinedArtifactManifest": "f8756e6f3cf15b00de623d379f41fbd98dba7798d8ebfb9668d5dce4e767cd56" } }, "runs": [ { "id": "purpose-lite-sol-high-v2", "tier": "lite", "kind": "from-pretrained-float", "outputDirectory": "ml/purpose-classifier/outputs/purpose-lite-sol-high-v2", "resolvedTraining": { "baseModel": "sentence-transformers/all-MiniLM-L6-v2", "baseModelRevision": "1110a243fdf4706b3f48f1d95db1a4f5529b4d41", "device": "mps", "epochs": 3, "batchSize": 32, "learningRate": 0.00002, "warmupRatio": 0.1, "boundaryWeight": 1.0, "trainingSeconds": 191.07675279202522 }, "selectedValidation": { "epoch": 3, "records": 917, "accuracy": 0.9127589967284624, "macroRecall": 0.9127308084599176 }, "frozenEvaluation": { "scorable": { "correct": 876, "records": 939, "accuracy": 0.9329073482428115, "macroRecall": 0.9322276987713067 }, "hard": { "correct": 191, "records": 215, "accuracy": 0.8883720930232558 }, "shippedFixtures": { "correct": 79, "records": 89, "accuracy": 0.8876404494382022 }, "sliceAccuracy": { "boundary": 0.6862745098039216, "core": 0.9543307086614173, "mixed": 0.9195402298850575, "pastedContext": 0.987012987012987, "shippedFixtures": 0.8876404494382022 }, "latencyMilliseconds": { "median": 17.3936, "p95": 21.5198 }, "failedGates": [ "scoredAccuracyAtLeast95", "p95LatencyAtMost20Milliseconds" ] }, "decision": "rejected" }, { "id": "purpose-lite-sol-high-v2-boundary-cont", "tier": "lite", "kind": "validation-selected-boundary-continuation", "resumedFrom": "ml/purpose-classifier/outputs/purpose-lite-sol-high-v2/model", "outputDirectory": "ml/purpose-classifier/outputs/purpose-lite-sol-high-v2-boundary-cont", "resolvedTraining": { "device": "mps", "epochZeroValidationAccuracy": 0.9127589967284624, "epochsRequested": 3, "epochsCompleted": 3, "batchSize": 32, "learningRate": 0.000003, "warmupRatio": 0.0, "boundaryWeight": 2.0, "earlyStoppingPatience": 2, "trainingSeconds": 257.24039216700476 }, "selectedValidation": { "epoch": 1, "records": 917, "accuracy": 0.9203925845147219, "macroRecall": 0.9209676130783011 }, "frozenEvaluation": { "scorable": { "correct": 880, "records": 939, "accuracy": 0.9371671991480298, "macroRecall": 0.9368046325084637, "correctDecisionsShortOf95Percent": 13 }, "hard": { "correct": 196, "records": 215, "accuracy": 0.9116279069767442 }, "shippedFixtures": { "correct": 76, "records": 89, "accuracy": 0.8539325842696629 }, "sliceAccuracy": { "boundary": 0.7647058823529411, "core": 0.9574803149606299, "mixed": 0.9310344827586207, "pastedContext": 0.987012987012987, "shippedFixtures": 0.8539325842696629 }, "latencyMilliseconds": { "median": 6.192041502799839, "p95": 8.401416009292006 }, "failedGates": [ "scoredAccuracyAtLeast95" ] }, "decision": "rejected-as-shipping-candidate; retained-as-validation-teacher-and-diagnostic-baseline" }, { "id": "purpose-lite-sol-high-v2-teacher", "tier": "lite", "kind": "prompt-and-label-bound-logit-cache", "sourceModel": "ml/purpose-classifier/outputs/purpose-lite-sol-high-v2-boundary-cont/model", "output": "ml/purpose-classifier/outputs/purpose-lite-sol-high-v2-teacher.pt", "records": { "train": 12164, "validation": 1208 }, "sha256": "a4175f0dd28c7d6e6936aabd83239aa2cfa8799aea1aaa70d0fe461377766a88", "decision": "retained-for-reproducibility; deep distillation did not produce a viable candidate" }, { "id": "purpose-deep-sol-high-v2-base", "tier": "deep", "kind": "from-pretrained-multitask", "outputDirectory": "ml/purpose-classifier/outputs/purpose-deep-sol-high-v2-base", "resolvedTraining": { "baseModelRevision": "8949b909ec900327062f0ebf497f51aef5e6f0c8", "device": "metal", "epochs": 3, "batchSize": 4, "learningRate": 0.00002, "hardWeight": 2.0, "secondaryLossWeight": 0.25, "mixedLossWeight": 0.25, "difficultyLossWeight": 0.1, "trainingSeconds": 6963.924981958 }, "validationTrajectory": [ { "epoch": 1, "primaryAccuracy": 0.6063249727371864, "hardAccuracy": 0.5791666666666667 }, { "epoch": 2, "primaryAccuracy": 0.7022900763358778, "hardAccuracy": 0.6291666666666667 }, { "epoch": 3, "primaryAccuracy": 0.7339149400218102, "hardAccuracy": 0.675 } ], "selectedValidation": { "epoch": 3, "primaryRecords": 917, "hardRecords": 240, "primaryAccuracy": 0.7339149400218102, "hardAccuracy": 0.675, "selectionScore": 0.7044574700109052, "primaryMacroRecall": 0.7354311324151239, "perPurposeRecall": { "planning": 0.7192982456140351, "backendImpl": 0.7844827586206896, "frontendImpl": 0.7596153846153846, "quickFix": 0.7685185185185185, "refactor": 0.8448275862068966, "debugging": 0.6538461538461539, "review": 0.7739130434782608, "writing": 0.5789473684210527 }, "secondarySupportedMacroRecall": 0.3516156462585034, "mixedF1Raw": 0.30158730158730157, "mixedF1Calibrated": 0.3971119133574007, "difficultyMae": 0.1273345649242401, "vagueLowRate": 0.9759450171821306 }, "frozenEvaluation": null, "decision": "rejected-on-validation; no-frozen-look; do-not-run-large" }, { "id": "purpose-deep-sol-high-v2-base-distilled", "tier": "deep", "kind": "lite-teacher-distilled-continuation", "resumedFrom": "ml/purpose-classifier/outputs/purpose-deep-sol-high-v2-base/model", "outputDirectory": "ml/purpose-classifier/outputs/purpose-deep-sol-high-v2-base-distilled", "resolvedTraining": { "device": "metal", "epochs": 2, "batchSize": 4, "learningRate": 0.00001, "distillationWeight": 0.5, "distillationTemperature": 2.0, "hardWeight": 2.0, "secondaryLossWeight": 0.25, "mixedLossWeight": 0.25, "difficultyLossWeight": 0.1, "trainingSeconds": 4237.380279583012 }, "validationTrajectory": [ { "epoch": 1, "primaryAccuracy": 0.7611777535441657, "hardAccuracy": 0.7, "teacherAgreement": 0.7699018538713195 }, { "epoch": 2, "primaryAccuracy": 0.7764449291166848, "hardAccuracy": 0.7291666666666666, "teacherAgreement": 0.7851690294438386 } ], "selectedValidation": { "epoch": 2, "primaryRecords": 917, "hardRecords": 240, "primaryAccuracy": 0.7764449291166848, "hardAccuracy": 0.7291666666666666, "selectionScore": 0.7528057978916758, "primaryMacroRecall": 0.7781732980788059, "perPurposeRecall": { "planning": 0.8070175438596491, "backendImpl": 0.8017241379310345, "frontendImpl": 0.7884615384615384, "quickFix": 0.7870370370370371, "refactor": 0.9051724137931034, "debugging": 0.6692307692307692, "review": 0.7913043478260869, "writing": 0.6754385964912281 }, "secondarySupportedMacroRecall": 0.415391156462585, "mixedF1Raw": 0.366412213740458, "mixedF1Calibrated": 0.4733727810650888, "difficultyMae": 0.1244114488363266, "vagueLowRate": 0.9759450171821306, "teacherAgreement": 0.7851690294438386 }, "comparison": { "primaryGainOverDeepBasePercentagePoints": 4.253, "hardGainOverDeepBasePercentagePoints": 5.417, "primaryGapBehindSelectedLiteValidationPercentagePoints": 14.395 }, "frozenEvaluation": null, "decision": "rejected-on-validation; no-frozen-look; stop-continuations-and-do-not-run-large-or-max" } ], "reproductionCommands": { "note": "Use the purpose-classifier virtual environment as ${PYTHON}. These commands make defaults material; output directories are gitignored and their retained files must match artifactHashes.", "promotion": [ "${PYTHON}", "ml/purpose-classifier/rebuild_sol_high.py", "promote", "--confirm", "overwrite-all-labels-with-sol-high", "--exclude-real-line", "1201", "--exclude-real-line", "1481" ], "liteFromPretrained": [ "${PYTHON}", "-u", "ml/purpose-classifier/train.py", "--device", "mps", "--model", "sentence-transformers/all-MiniLM-L6-v2", "--dataset-dir", "ml/purpose-classifier/.artifacts/dataset-v1", "--epochs", "3", "--learning-rate", "2e-5", "--warmup-ratio", "0.1", "--boundary-weight", "1", "--early-stopping-patience", "2", "--output-dir", "ml/purpose-classifier/outputs/purpose-lite-sol-high-v2", "--overwrite-output" ], "liteFromPretrainedFrozenEvaluation": [ "${PYTHON}", "-u", "ml/purpose-classifier/eval.py", "--device", "mps", "--model-dir", "ml/purpose-classifier/outputs/purpose-lite-sol-high-v2/model", "--calibration", "ml/purpose-classifier/outputs/purpose-lite-sol-high-v2/calibration.json", "--report", "ml/purpose-classifier/outputs/purpose-lite-sol-high-v2/frozen-eval.json" ], "liteBoundaryContinuation": [ "${PYTHON}", "-u", "ml/purpose-classifier/train.py", "--device", "mps", "--model", "ml/purpose-classifier/outputs/purpose-lite-sol-high-v2/model", "--dataset-dir", "ml/purpose-classifier/.artifacts/dataset-v1", "--epochs", "3", "--learning-rate", "3e-6", "--warmup-ratio", "0", "--boundary-weight", "2", "--early-stopping-patience", "2", "--output-dir", "ml/purpose-classifier/outputs/purpose-lite-sol-high-v2-boundary-cont", "--overwrite-output" ], "liteBoundaryFrozenEvaluation": [ "${PYTHON}", "-u", "ml/purpose-classifier/eval.py", "--device", "mps", "--model-dir", "ml/purpose-classifier/outputs/purpose-lite-sol-high-v2-boundary-cont/model", "--calibration", "ml/purpose-classifier/outputs/purpose-lite-sol-high-v2-boundary-cont/calibration.json", "--report", "ml/purpose-classifier/outputs/purpose-lite-sol-high-v2-boundary-cont/frozen-eval.json" ], "teacherCache": [ "${PYTHON}", "-u", "ml/purpose-classifier/cache_teacher.py", "--device", "mps", "--dataset-dir", "ml/purpose-classifier/.artifacts/dataset-v1", "--model", "ml/purpose-classifier/outputs/purpose-lite-sol-high-v2-boundary-cont/model", "--output", "ml/purpose-classifier/outputs/purpose-lite-sol-high-v2-teacher.pt", "--batch-size", "16", "--progress-steps", "25", "--overwrite-output" ], "deepBase": [ "${PYTHON}", "-u", "ml/purpose-classifier/train_deep_mlx.py", "--variant", "base", "--device", "metal", "--dataset-dir", "ml/purpose-classifier/.artifacts/dataset-v1", "--epochs", "3", "--early-stopping-patience", "1", "--progress-steps", "10", "--output-dir", "ml/purpose-classifier/outputs/purpose-deep-sol-high-v2-base", "--overwrite-output" ], "deepDistilledContinuation": [ "${PYTHON}", "-u", "ml/purpose-classifier/train_deep_mlx.py", "--variant", "base", "--device", "metal", "--resume-from", "ml/purpose-classifier/outputs/purpose-deep-sol-high-v2-base/model", "--dataset-dir", "ml/purpose-classifier/.artifacts/dataset-v1", "--distillation-cache", "ml/purpose-classifier/outputs/purpose-lite-sol-high-v2-teacher.pt", "--distillation-weight", "0.5", "--distillation-temperature", "2", "--epochs", "2", "--learning-rate", "1e-5", "--early-stopping-patience", "1", "--progress-steps", "10", "--output-dir", "ml/purpose-classifier/outputs/purpose-deep-sol-high-v2-base-distilled", "--overwrite-output" ] }, "artifactHashes": { "purpose-lite-sol-high-v2": { "metrics.json": "ed6216f37e56e1c1af554d7c3cb614e3f1e2d0e856a2a6fc87c1aeaf5da86e15", "calibration.json": "7ebe788f86b43d5f9927ef891fc354bd77499b68b7c4f250e8eda613b8fe2168", "training-config.json": "b6547867830a29365a5a68bca1d725da8ee342fd67bf9297f6b826f454ab331a", "frozen-eval.json": "f6070bb82759521a8011d7705bf07d3660b4ba10b4181dbd9ea9589118e9e9ce", "model/model.safetensors": "0ca2284f443bcaf3b05c1a673f015f6856b3dc3152f936fa9a7d35ab81551f2e" }, "purpose-lite-sol-high-v2-boundary-cont": { "metrics.json": "bf97b5ac1926e735b9d210f9c6a286d7e8afe889a4f7d38aa4b899cf68e4dea5", "calibration.json": "e98349819b99819ec12806345a7c642e7716bb6302f4d796d7253681c05e20a4", "training-config.json": "cd3ce124b17589cef2f3c63170323f875ee9782d924d3c8067062d59125ece3e", "frozen-eval.json": "780c504b1bdddbb7b6cd9781583d4ac2976278ae379f7e8a6825d2e7fdc97aee", "model/model.safetensors": "c7f004dbaab01b3331ef7eb73d090e6d110afd531598e9998ac40730b608a2c6" }, "purpose-deep-sol-high-v2-base": { "metrics.json": "3ac5b0345da17fef3407eed2cf60888898734eaa401ec1ca9e5f6df2afd6ffa5", "calibration.json": "fbedfa506082aed36e9c120235f0d7a7fb9d6c944f4ad1622db7c27c1a509e1b", "training-config.json": "29fd34926f0f688e1b3c78f14f422ec8c59817b67346b672047f7307f6e487b5", "training-state.json": "221ae224898cbbfddd649b3dae17260bacfcfbb30edfa13c854fa0bbea36704e", "model/model.safetensors": "09205c022d95088300f04d90d52055ce218474f9bc5be6a85b428cdbacbe1782" }, "purpose-deep-sol-high-v2-base-distilled": { "metrics.json": "87a4c55ac9867fbb85e15d2f5e725fefb2eb34f98e91609136fbe8d3a009ba3c", "calibration.json": "2da1735b4879bd6caf8a1ff7bb5dbff65ba78c6335478bd949c9ad5093ad03e6", "training-config.json": "6242c711c8f070113987e57981f5879c72729000141b8559621b4ab6ec7b834e", "training-state.json": "e8093c0490a91590600e7d436858b025878c702490dde0bb59e98ff7c6bcaa53", "model/model.safetensors": "568c801c70a3caa49e0ab9afd10f4c36c415cf9b9bc4523d6d2bd37b9734c4b3" } }, "decisions": [ "The missing MiniLM classifier.weight and classifier.bias load report is expected because the pretrained encoder has no task-specific eight-way classifier head; downstream training initializes those two tensors.", "Do not tune another lite continuation against the Sol-high v2 frozen set.", "Do not QAT or export a Sol-high v2 lite candidate until a float checkpoint first clears the validation and frozen qualification policy without frozen-driven selection.", "Do not run another plain or distilled deep continuation, ModernBERT large, or purpose-max from these results.", "The previously deployed v1 W8A8 artifact is historical runtime evidence only and is not label-compatible with Sol-high v2." ], "nextSteps": [ { "order": 1, "action": "Implement a validation-only representation diagnostic harness.", "details": [ "Report the exact same 917-record primary and 240-record hard metrics used by the deep trainer.", "Cache ModernBERT representations with keys bound to model revision, tokenizer contract, dataset hash, and normalized prompt hash.", "Compare the current pretrained prediction-head CLS representation with attention-masked mean pooling.", "Fit only a regularized eight-way primary linear probe; exclude secondary, mixed-intent, and difficulty losses." ] }, { "order": 2, "action": "Apply a fail-fast probe gate before another full deep run.", "gate": { "minimumMeanPoolingGainOverClsPercentagePointsOnEitherOverallOrHard": 3.0, "minimumOverallAccuracy": 0.85, "perPurposeCollapseAllowed": false } }, { "order": 3, "action": "If the probe passes, add explicit --pooling cls|mean and --primary-only trainer options and run one from-pretrained ModernBERT-base experiment.", "selection": "Validation overall and hard primary accuracy only; introduce auxiliary heads only after primary representation viability is established." }, { "order": 4, "action": "Require a revised deep checkpoint to approach lite on validation before any frozen evaluation.", "gate": { "maximumPrimaryGapBehindLitePercentagePoints": 2.0, "minimumHardGainOverLitePercentagePoints": 3.0, "note": "First add the matching validation-hard report to lite so this comparison uses identical records and definitions." } }, { "order": 5, "action": "If ModernBERT pooling probes fail, compare sentence-trained encoder backbones with the same frozen probe protocol or remove the deep tier.", "prohibited": "Do not use model size as the next variable and do not run ModernBERT large." }, { "order": 6, "action": "Improve lite through validation-only error analysis and data work, then reserve a new independent holdout before another candidate cycle.", "prohibited": "Do not use the current frozen set for optimizer, data, calibration, or checkpoint selection." }, { "order": 7, "action": "Run QAT, export, target-runtime parity, latency, energy, and residency only after a float candidate qualifies.", "shippingGate": { "liteFrozenAccuracy": 0.95, "deepFrozenAccuracy": 0.97, "deepMinimumHardGainOverLitePercentagePoints": 5.0 } } ] }