636 lines
23 KiB
JSON
636 lines
23 KiB
JSON
{
|
|
"schemaVersion": 1,
|
|
"experiment": "purpose-sol-high-v2-training",
|
|
"recordedAt": "2026-08-03",
|
|
"sourceCheckoutCommit": "531e9ba301a20dd124062afed47aa412244c407a",
|
|
"status": "no-qualified-sol-high-v2-candidate",
|
|
"summary": "The Sol-high label reset and promotion completed. The selected lite float checkpoint reached 93.7167% frozen accuracy, while the best deep validation checkpoint reached 77.6445%. Neither meets the shipping contract; frozen-set-driven tuning and further deep continuations are stopped.",
|
|
"privacy": {
|
|
"policy": "Private Nucleic-history and SWE-chat prompts remain in ignored artifacts.",
|
|
"recordedIdentifiers": "Only aggregate counts, content hashes, reviewed source lines, and reviewed prompt hashes are versioned here."
|
|
},
|
|
"labelReset": {
|
|
"teacher": {
|
|
"model": "gpt-5.6-sol",
|
|
"reasoningEffort": "high"
|
|
},
|
|
"inputDecisions": {
|
|
"total": 15048,
|
|
"labeled": 14704,
|
|
"rejected": 344,
|
|
"populations": {
|
|
"canonicalPublic": {
|
|
"labeled": 12007,
|
|
"rejected": 207
|
|
},
|
|
"shippedFixtures": {
|
|
"labeled": 89,
|
|
"rejectedToGeneral": 3
|
|
},
|
|
"nucleicHistory": {
|
|
"labeled": 1068,
|
|
"rejected": 68
|
|
},
|
|
"sweChat": {
|
|
"labeled": 1540,
|
|
"rejected": 66
|
|
}
|
|
}
|
|
},
|
|
"promotion": {
|
|
"backup": "ml/purpose-classifier/.artifacts/sol-high-reset/backups/20260802T234020Z",
|
|
"publicLabeled": 12007,
|
|
"realLabeledAfterReviewedExclusions": 2606,
|
|
"combinedTrain": 12164,
|
|
"validation": 1208,
|
|
"syntheticTest": 1119,
|
|
"reviewedRealExclusions": [
|
|
{
|
|
"sourceLine": 1201,
|
|
"promptHash": "124e98f338b935a7efd0d4e12295eb59f07ee68e4dc8ab65bc069b630ee93110",
|
|
"priorPurpose": "planning",
|
|
"reason": "reviewed-near-duplicate-label-conflict"
|
|
},
|
|
{
|
|
"sourceLine": 1481,
|
|
"promptHash": "a6dd960a0857f2d8af5ef971582e480456a33599df8af27d50cddb5e9f828035",
|
|
"priorPurpose": "refactor",
|
|
"reason": "reviewed-near-duplicate-label-conflict"
|
|
}
|
|
],
|
|
"validationCommandResult": {
|
|
"records": 12007,
|
|
"errors": 0,
|
|
"expectedShapeWarnings": 22
|
|
}
|
|
}
|
|
},
|
|
"dataset": {
|
|
"version": "purpose-dataset-sol-high-v2",
|
|
"public": {
|
|
"inputRecords": 12007,
|
|
"nearDuplicateExclusions": 18,
|
|
"retainedRecords": 11989,
|
|
"train": {
|
|
"records": 9662,
|
|
"sha256": "0e16940308dc7557c73b1b804bf7b715c86d263b613201588ec274169ee01e89"
|
|
},
|
|
"validation": {
|
|
"records": 1208,
|
|
"scorableRecords": 917,
|
|
"vagueEvalRecords": 291,
|
|
"sha256": "301cd3d1c69e1bb92b811e857093ad12fb8c5c11bb402ee2993403d5e4467969"
|
|
},
|
|
"syntheticTest": {
|
|
"records": 1119,
|
|
"vagueEvalRecords": 269,
|
|
"sha256": "9997dbaa6d305ea56d8c161e39b925368761ab613a706eb8c2ba55d0e58b1906"
|
|
},
|
|
"shippedFixtures": {
|
|
"records": 92,
|
|
"classifiableRecords": 89,
|
|
"generalRecords": 3,
|
|
"sha256": "876068ea26d7109365bcfbd1dc44ba3f1a0400c80dc7fdacf008b3fa07cd92cc"
|
|
},
|
|
"logicalFrozenRecords": 1208,
|
|
"scorableFrozenRecords": 939
|
|
},
|
|
"combinedTrainingArtifact": {
|
|
"experimentVersion": "all-sol-high-training-augmentation-v2",
|
|
"train": {
|
|
"records": 12164,
|
|
"sha256": "44a4bbccf2c8207f7f497e8138ee01007c687034a7c719604f50cbeba35e5bcc"
|
|
},
|
|
"validation": {
|
|
"records": 1208,
|
|
"sha256": "301cd3d1c69e1bb92b811e857093ad12fb8c5c11bb402ee2993403d5e4467969"
|
|
},
|
|
"syntheticTest": {
|
|
"records": 1119,
|
|
"sha256": "9997dbaa6d305ea56d8c161e39b925368761ab613a706eb8c2ba55d0e58b1906"
|
|
},
|
|
"realInputAfterReviewedExclusions": {
|
|
"records": 2606,
|
|
"sha256": "ed38f4ffdb640a2a9decf1407c23ff8c8c8a150bbe1f55b1b051033d5bcc1ade"
|
|
},
|
|
"acceptedRealAugmentation": 2502,
|
|
"excludedRealAugmentation": {
|
|
"total": 104,
|
|
"vagueEval": 87,
|
|
"nearHistoryDuplicate": 17
|
|
}
|
|
},
|
|
"manifestHashes": {
|
|
"datasetV1Manifest": "c6be5d0df6fd87ab3f5a254d3b8e280cae786cdc90e3cd0cdee02c6662dd27b1",
|
|
"generationManifest": "bb707c81047cb1843562d1641abe344494ef0c4aab5af7ae56283d6769160b2a",
|
|
"supersededCurationReview": "f8b47a08a10ff872f2a29e6f3321d4818ab439a2429d7162692b58c9281fdf47",
|
|
"ignoredCombinedArtifactManifest": "f8756e6f3cf15b00de623d379f41fbd98dba7798d8ebfb9668d5dce4e767cd56"
|
|
}
|
|
},
|
|
"runs": [
|
|
{
|
|
"id": "purpose-lite-sol-high-v2",
|
|
"tier": "lite",
|
|
"kind": "from-pretrained-float",
|
|
"outputDirectory": "ml/purpose-classifier/outputs/purpose-lite-sol-high-v2",
|
|
"resolvedTraining": {
|
|
"baseModel": "sentence-transformers/all-MiniLM-L6-v2",
|
|
"baseModelRevision": "1110a243fdf4706b3f48f1d95db1a4f5529b4d41",
|
|
"device": "mps",
|
|
"epochs": 3,
|
|
"batchSize": 32,
|
|
"learningRate": 0.00002,
|
|
"warmupRatio": 0.1,
|
|
"boundaryWeight": 1.0,
|
|
"trainingSeconds": 191.07675279202522
|
|
},
|
|
"selectedValidation": {
|
|
"epoch": 3,
|
|
"records": 917,
|
|
"accuracy": 0.9127589967284624,
|
|
"macroRecall": 0.9127308084599176
|
|
},
|
|
"frozenEvaluation": {
|
|
"scorable": {
|
|
"correct": 876,
|
|
"records": 939,
|
|
"accuracy": 0.9329073482428115,
|
|
"macroRecall": 0.9322276987713067
|
|
},
|
|
"hard": {
|
|
"correct": 191,
|
|
"records": 215,
|
|
"accuracy": 0.8883720930232558
|
|
},
|
|
"shippedFixtures": {
|
|
"correct": 79,
|
|
"records": 89,
|
|
"accuracy": 0.8876404494382022
|
|
},
|
|
"sliceAccuracy": {
|
|
"boundary": 0.6862745098039216,
|
|
"core": 0.9543307086614173,
|
|
"mixed": 0.9195402298850575,
|
|
"pastedContext": 0.987012987012987,
|
|
"shippedFixtures": 0.8876404494382022
|
|
},
|
|
"latencyMilliseconds": {
|
|
"median": 17.3936,
|
|
"p95": 21.5198
|
|
},
|
|
"failedGates": [
|
|
"scoredAccuracyAtLeast95",
|
|
"p95LatencyAtMost20Milliseconds"
|
|
]
|
|
},
|
|
"decision": "rejected"
|
|
},
|
|
{
|
|
"id": "purpose-lite-sol-high-v2-boundary-cont",
|
|
"tier": "lite",
|
|
"kind": "validation-selected-boundary-continuation",
|
|
"resumedFrom": "ml/purpose-classifier/outputs/purpose-lite-sol-high-v2/model",
|
|
"outputDirectory": "ml/purpose-classifier/outputs/purpose-lite-sol-high-v2-boundary-cont",
|
|
"resolvedTraining": {
|
|
"device": "mps",
|
|
"epochZeroValidationAccuracy": 0.9127589967284624,
|
|
"epochsRequested": 3,
|
|
"epochsCompleted": 3,
|
|
"batchSize": 32,
|
|
"learningRate": 0.000003,
|
|
"warmupRatio": 0.0,
|
|
"boundaryWeight": 2.0,
|
|
"earlyStoppingPatience": 2,
|
|
"trainingSeconds": 257.24039216700476
|
|
},
|
|
"selectedValidation": {
|
|
"epoch": 1,
|
|
"records": 917,
|
|
"accuracy": 0.9203925845147219,
|
|
"macroRecall": 0.9209676130783011
|
|
},
|
|
"frozenEvaluation": {
|
|
"scorable": {
|
|
"correct": 880,
|
|
"records": 939,
|
|
"accuracy": 0.9371671991480298,
|
|
"macroRecall": 0.9368046325084637,
|
|
"correctDecisionsShortOf95Percent": 13
|
|
},
|
|
"hard": {
|
|
"correct": 196,
|
|
"records": 215,
|
|
"accuracy": 0.9116279069767442
|
|
},
|
|
"shippedFixtures": {
|
|
"correct": 76,
|
|
"records": 89,
|
|
"accuracy": 0.8539325842696629
|
|
},
|
|
"sliceAccuracy": {
|
|
"boundary": 0.7647058823529411,
|
|
"core": 0.9574803149606299,
|
|
"mixed": 0.9310344827586207,
|
|
"pastedContext": 0.987012987012987,
|
|
"shippedFixtures": 0.8539325842696629
|
|
},
|
|
"latencyMilliseconds": {
|
|
"median": 6.192041502799839,
|
|
"p95": 8.401416009292006
|
|
},
|
|
"failedGates": [
|
|
"scoredAccuracyAtLeast95"
|
|
]
|
|
},
|
|
"decision": "rejected-as-shipping-candidate; retained-as-validation-teacher-and-diagnostic-baseline"
|
|
},
|
|
{
|
|
"id": "purpose-lite-sol-high-v2-teacher",
|
|
"tier": "lite",
|
|
"kind": "prompt-and-label-bound-logit-cache",
|
|
"sourceModel": "ml/purpose-classifier/outputs/purpose-lite-sol-high-v2-boundary-cont/model",
|
|
"output": "ml/purpose-classifier/outputs/purpose-lite-sol-high-v2-teacher.pt",
|
|
"records": {
|
|
"train": 12164,
|
|
"validation": 1208
|
|
},
|
|
"sha256": "a4175f0dd28c7d6e6936aabd83239aa2cfa8799aea1aaa70d0fe461377766a88",
|
|
"decision": "retained-for-reproducibility; deep distillation did not produce a viable candidate"
|
|
},
|
|
{
|
|
"id": "purpose-deep-sol-high-v2-base",
|
|
"tier": "deep",
|
|
"kind": "from-pretrained-multitask",
|
|
"outputDirectory": "ml/purpose-classifier/outputs/purpose-deep-sol-high-v2-base",
|
|
"resolvedTraining": {
|
|
"baseModelRevision": "8949b909ec900327062f0ebf497f51aef5e6f0c8",
|
|
"device": "metal",
|
|
"epochs": 3,
|
|
"batchSize": 4,
|
|
"learningRate": 0.00002,
|
|
"hardWeight": 2.0,
|
|
"secondaryLossWeight": 0.25,
|
|
"mixedLossWeight": 0.25,
|
|
"difficultyLossWeight": 0.1,
|
|
"trainingSeconds": 6963.924981958
|
|
},
|
|
"validationTrajectory": [
|
|
{
|
|
"epoch": 1,
|
|
"primaryAccuracy": 0.6063249727371864,
|
|
"hardAccuracy": 0.5791666666666667
|
|
},
|
|
{
|
|
"epoch": 2,
|
|
"primaryAccuracy": 0.7022900763358778,
|
|
"hardAccuracy": 0.6291666666666667
|
|
},
|
|
{
|
|
"epoch": 3,
|
|
"primaryAccuracy": 0.7339149400218102,
|
|
"hardAccuracy": 0.675
|
|
}
|
|
],
|
|
"selectedValidation": {
|
|
"epoch": 3,
|
|
"primaryRecords": 917,
|
|
"hardRecords": 240,
|
|
"primaryAccuracy": 0.7339149400218102,
|
|
"hardAccuracy": 0.675,
|
|
"selectionScore": 0.7044574700109052,
|
|
"primaryMacroRecall": 0.7354311324151239,
|
|
"perPurposeRecall": {
|
|
"planning": 0.7192982456140351,
|
|
"backendImpl": 0.7844827586206896,
|
|
"frontendImpl": 0.7596153846153846,
|
|
"quickFix": 0.7685185185185185,
|
|
"refactor": 0.8448275862068966,
|
|
"debugging": 0.6538461538461539,
|
|
"review": 0.7739130434782608,
|
|
"writing": 0.5789473684210527
|
|
},
|
|
"secondarySupportedMacroRecall": 0.3516156462585034,
|
|
"mixedF1Raw": 0.30158730158730157,
|
|
"mixedF1Calibrated": 0.3971119133574007,
|
|
"difficultyMae": 0.1273345649242401,
|
|
"vagueLowRate": 0.9759450171821306
|
|
},
|
|
"frozenEvaluation": null,
|
|
"decision": "rejected-on-validation; no-frozen-look; do-not-run-large"
|
|
},
|
|
{
|
|
"id": "purpose-deep-sol-high-v2-base-distilled",
|
|
"tier": "deep",
|
|
"kind": "lite-teacher-distilled-continuation",
|
|
"resumedFrom": "ml/purpose-classifier/outputs/purpose-deep-sol-high-v2-base/model",
|
|
"outputDirectory": "ml/purpose-classifier/outputs/purpose-deep-sol-high-v2-base-distilled",
|
|
"resolvedTraining": {
|
|
"device": "metal",
|
|
"epochs": 2,
|
|
"batchSize": 4,
|
|
"learningRate": 0.00001,
|
|
"distillationWeight": 0.5,
|
|
"distillationTemperature": 2.0,
|
|
"hardWeight": 2.0,
|
|
"secondaryLossWeight": 0.25,
|
|
"mixedLossWeight": 0.25,
|
|
"difficultyLossWeight": 0.1,
|
|
"trainingSeconds": 4237.380279583012
|
|
},
|
|
"validationTrajectory": [
|
|
{
|
|
"epoch": 1,
|
|
"primaryAccuracy": 0.7611777535441657,
|
|
"hardAccuracy": 0.7,
|
|
"teacherAgreement": 0.7699018538713195
|
|
},
|
|
{
|
|
"epoch": 2,
|
|
"primaryAccuracy": 0.7764449291166848,
|
|
"hardAccuracy": 0.7291666666666666,
|
|
"teacherAgreement": 0.7851690294438386
|
|
}
|
|
],
|
|
"selectedValidation": {
|
|
"epoch": 2,
|
|
"primaryRecords": 917,
|
|
"hardRecords": 240,
|
|
"primaryAccuracy": 0.7764449291166848,
|
|
"hardAccuracy": 0.7291666666666666,
|
|
"selectionScore": 0.7528057978916758,
|
|
"primaryMacroRecall": 0.7781732980788059,
|
|
"perPurposeRecall": {
|
|
"planning": 0.8070175438596491,
|
|
"backendImpl": 0.8017241379310345,
|
|
"frontendImpl": 0.7884615384615384,
|
|
"quickFix": 0.7870370370370371,
|
|
"refactor": 0.9051724137931034,
|
|
"debugging": 0.6692307692307692,
|
|
"review": 0.7913043478260869,
|
|
"writing": 0.6754385964912281
|
|
},
|
|
"secondarySupportedMacroRecall": 0.415391156462585,
|
|
"mixedF1Raw": 0.366412213740458,
|
|
"mixedF1Calibrated": 0.4733727810650888,
|
|
"difficultyMae": 0.1244114488363266,
|
|
"vagueLowRate": 0.9759450171821306,
|
|
"teacherAgreement": 0.7851690294438386
|
|
},
|
|
"comparison": {
|
|
"primaryGainOverDeepBasePercentagePoints": 4.253,
|
|
"hardGainOverDeepBasePercentagePoints": 5.417,
|
|
"primaryGapBehindSelectedLiteValidationPercentagePoints": 14.395
|
|
},
|
|
"frozenEvaluation": null,
|
|
"decision": "rejected-on-validation; no-frozen-look; stop-continuations-and-do-not-run-large-or-max"
|
|
}
|
|
],
|
|
"reproductionCommands": {
|
|
"note": "Use the purpose-classifier virtual environment as ${PYTHON}. These commands make defaults material; output directories are gitignored and their retained files must match artifactHashes.",
|
|
"promotion": [
|
|
"${PYTHON}",
|
|
"ml/purpose-classifier/rebuild_sol_high.py",
|
|
"promote",
|
|
"--confirm",
|
|
"overwrite-all-labels-with-sol-high",
|
|
"--exclude-real-line",
|
|
"1201",
|
|
"--exclude-real-line",
|
|
"1481"
|
|
],
|
|
"liteFromPretrained": [
|
|
"${PYTHON}",
|
|
"-u",
|
|
"ml/purpose-classifier/train.py",
|
|
"--device",
|
|
"mps",
|
|
"--model",
|
|
"sentence-transformers/all-MiniLM-L6-v2",
|
|
"--dataset-dir",
|
|
"ml/purpose-classifier/.artifacts/dataset-v1",
|
|
"--epochs",
|
|
"3",
|
|
"--learning-rate",
|
|
"2e-5",
|
|
"--warmup-ratio",
|
|
"0.1",
|
|
"--boundary-weight",
|
|
"1",
|
|
"--early-stopping-patience",
|
|
"2",
|
|
"--output-dir",
|
|
"ml/purpose-classifier/outputs/purpose-lite-sol-high-v2",
|
|
"--overwrite-output"
|
|
],
|
|
"liteFromPretrainedFrozenEvaluation": [
|
|
"${PYTHON}",
|
|
"-u",
|
|
"ml/purpose-classifier/eval.py",
|
|
"--device",
|
|
"mps",
|
|
"--model-dir",
|
|
"ml/purpose-classifier/outputs/purpose-lite-sol-high-v2/model",
|
|
"--calibration",
|
|
"ml/purpose-classifier/outputs/purpose-lite-sol-high-v2/calibration.json",
|
|
"--report",
|
|
"ml/purpose-classifier/outputs/purpose-lite-sol-high-v2/frozen-eval.json"
|
|
],
|
|
"liteBoundaryContinuation": [
|
|
"${PYTHON}",
|
|
"-u",
|
|
"ml/purpose-classifier/train.py",
|
|
"--device",
|
|
"mps",
|
|
"--model",
|
|
"ml/purpose-classifier/outputs/purpose-lite-sol-high-v2/model",
|
|
"--dataset-dir",
|
|
"ml/purpose-classifier/.artifacts/dataset-v1",
|
|
"--epochs",
|
|
"3",
|
|
"--learning-rate",
|
|
"3e-6",
|
|
"--warmup-ratio",
|
|
"0",
|
|
"--boundary-weight",
|
|
"2",
|
|
"--early-stopping-patience",
|
|
"2",
|
|
"--output-dir",
|
|
"ml/purpose-classifier/outputs/purpose-lite-sol-high-v2-boundary-cont",
|
|
"--overwrite-output"
|
|
],
|
|
"liteBoundaryFrozenEvaluation": [
|
|
"${PYTHON}",
|
|
"-u",
|
|
"ml/purpose-classifier/eval.py",
|
|
"--device",
|
|
"mps",
|
|
"--model-dir",
|
|
"ml/purpose-classifier/outputs/purpose-lite-sol-high-v2-boundary-cont/model",
|
|
"--calibration",
|
|
"ml/purpose-classifier/outputs/purpose-lite-sol-high-v2-boundary-cont/calibration.json",
|
|
"--report",
|
|
"ml/purpose-classifier/outputs/purpose-lite-sol-high-v2-boundary-cont/frozen-eval.json"
|
|
],
|
|
"teacherCache": [
|
|
"${PYTHON}",
|
|
"-u",
|
|
"ml/purpose-classifier/cache_teacher.py",
|
|
"--device",
|
|
"mps",
|
|
"--dataset-dir",
|
|
"ml/purpose-classifier/.artifacts/dataset-v1",
|
|
"--model",
|
|
"ml/purpose-classifier/outputs/purpose-lite-sol-high-v2-boundary-cont/model",
|
|
"--output",
|
|
"ml/purpose-classifier/outputs/purpose-lite-sol-high-v2-teacher.pt",
|
|
"--batch-size",
|
|
"16",
|
|
"--progress-steps",
|
|
"25",
|
|
"--overwrite-output"
|
|
],
|
|
"deepBase": [
|
|
"${PYTHON}",
|
|
"-u",
|
|
"ml/purpose-classifier/train_deep_mlx.py",
|
|
"--variant",
|
|
"base",
|
|
"--device",
|
|
"metal",
|
|
"--dataset-dir",
|
|
"ml/purpose-classifier/.artifacts/dataset-v1",
|
|
"--epochs",
|
|
"3",
|
|
"--early-stopping-patience",
|
|
"1",
|
|
"--progress-steps",
|
|
"10",
|
|
"--output-dir",
|
|
"ml/purpose-classifier/outputs/purpose-deep-sol-high-v2-base",
|
|
"--overwrite-output"
|
|
],
|
|
"deepDistilledContinuation": [
|
|
"${PYTHON}",
|
|
"-u",
|
|
"ml/purpose-classifier/train_deep_mlx.py",
|
|
"--variant",
|
|
"base",
|
|
"--device",
|
|
"metal",
|
|
"--resume-from",
|
|
"ml/purpose-classifier/outputs/purpose-deep-sol-high-v2-base/model",
|
|
"--dataset-dir",
|
|
"ml/purpose-classifier/.artifacts/dataset-v1",
|
|
"--distillation-cache",
|
|
"ml/purpose-classifier/outputs/purpose-lite-sol-high-v2-teacher.pt",
|
|
"--distillation-weight",
|
|
"0.5",
|
|
"--distillation-temperature",
|
|
"2",
|
|
"--epochs",
|
|
"2",
|
|
"--learning-rate",
|
|
"1e-5",
|
|
"--early-stopping-patience",
|
|
"1",
|
|
"--progress-steps",
|
|
"10",
|
|
"--output-dir",
|
|
"ml/purpose-classifier/outputs/purpose-deep-sol-high-v2-base-distilled",
|
|
"--overwrite-output"
|
|
]
|
|
},
|
|
"artifactHashes": {
|
|
"purpose-lite-sol-high-v2": {
|
|
"metrics.json": "ed6216f37e56e1c1af554d7c3cb614e3f1e2d0e856a2a6fc87c1aeaf5da86e15",
|
|
"calibration.json": "7ebe788f86b43d5f9927ef891fc354bd77499b68b7c4f250e8eda613b8fe2168",
|
|
"training-config.json": "b6547867830a29365a5a68bca1d725da8ee342fd67bf9297f6b826f454ab331a",
|
|
"frozen-eval.json": "f6070bb82759521a8011d7705bf07d3660b4ba10b4181dbd9ea9589118e9e9ce",
|
|
"model/model.safetensors": "0ca2284f443bcaf3b05c1a673f015f6856b3dc3152f936fa9a7d35ab81551f2e"
|
|
},
|
|
"purpose-lite-sol-high-v2-boundary-cont": {
|
|
"metrics.json": "bf97b5ac1926e735b9d210f9c6a286d7e8afe889a4f7d38aa4b899cf68e4dea5",
|
|
"calibration.json": "e98349819b99819ec12806345a7c642e7716bb6302f4d796d7253681c05e20a4",
|
|
"training-config.json": "cd3ce124b17589cef2f3c63170323f875ee9782d924d3c8067062d59125ece3e",
|
|
"frozen-eval.json": "780c504b1bdddbb7b6cd9781583d4ac2976278ae379f7e8a6825d2e7fdc97aee",
|
|
"model/model.safetensors": "c7f004dbaab01b3331ef7eb73d090e6d110afd531598e9998ac40730b608a2c6"
|
|
},
|
|
"purpose-deep-sol-high-v2-base": {
|
|
"metrics.json": "3ac5b0345da17fef3407eed2cf60888898734eaa401ec1ca9e5f6df2afd6ffa5",
|
|
"calibration.json": "fbedfa506082aed36e9c120235f0d7a7fb9d6c944f4ad1622db7c27c1a509e1b",
|
|
"training-config.json": "29fd34926f0f688e1b3c78f14f422ec8c59817b67346b672047f7307f6e487b5",
|
|
"training-state.json": "221ae224898cbbfddd649b3dae17260bacfcfbb30edfa13c854fa0bbea36704e",
|
|
"model/model.safetensors": "09205c022d95088300f04d90d52055ce218474f9bc5be6a85b428cdbacbe1782"
|
|
},
|
|
"purpose-deep-sol-high-v2-base-distilled": {
|
|
"metrics.json": "87a4c55ac9867fbb85e15d2f5e725fefb2eb34f98e91609136fbe8d3a009ba3c",
|
|
"calibration.json": "2da1735b4879bd6caf8a1ff7bb5dbff65ba78c6335478bd949c9ad5093ad03e6",
|
|
"training-config.json": "6242c711c8f070113987e57981f5879c72729000141b8559621b4ab6ec7b834e",
|
|
"training-state.json": "e8093c0490a91590600e7d436858b025878c702490dde0bb59e98ff7c6bcaa53",
|
|
"model/model.safetensors": "568c801c70a3caa49e0ab9afd10f4c36c415cf9b9bc4523d6d2bd37b9734c4b3"
|
|
}
|
|
},
|
|
"decisions": [
|
|
"The missing MiniLM classifier.weight and classifier.bias load report is expected because the pretrained encoder has no task-specific eight-way classifier head; downstream training initializes those two tensors.",
|
|
"Do not tune another lite continuation against the Sol-high v2 frozen set.",
|
|
"Do not QAT or export a Sol-high v2 lite candidate until a float checkpoint first clears the validation and frozen qualification policy without frozen-driven selection.",
|
|
"Do not run another plain or distilled deep continuation, ModernBERT large, or purpose-max from these results.",
|
|
"The previously deployed v1 W8A8 artifact is historical runtime evidence only and is not label-compatible with Sol-high v2."
|
|
],
|
|
"nextSteps": [
|
|
{
|
|
"order": 1,
|
|
"action": "Implement a validation-only representation diagnostic harness.",
|
|
"details": [
|
|
"Report the exact same 917-record primary and 240-record hard metrics used by the deep trainer.",
|
|
"Cache ModernBERT representations with keys bound to model revision, tokenizer contract, dataset hash, and normalized prompt hash.",
|
|
"Compare the current pretrained prediction-head CLS representation with attention-masked mean pooling.",
|
|
"Fit only a regularized eight-way primary linear probe; exclude secondary, mixed-intent, and difficulty losses."
|
|
]
|
|
},
|
|
{
|
|
"order": 2,
|
|
"action": "Apply a fail-fast probe gate before another full deep run.",
|
|
"gate": {
|
|
"minimumMeanPoolingGainOverClsPercentagePointsOnEitherOverallOrHard": 3.0,
|
|
"minimumOverallAccuracy": 0.85,
|
|
"perPurposeCollapseAllowed": false
|
|
}
|
|
},
|
|
{
|
|
"order": 3,
|
|
"action": "If the probe passes, add explicit --pooling cls|mean and --primary-only trainer options and run one from-pretrained ModernBERT-base experiment.",
|
|
"selection": "Validation overall and hard primary accuracy only; introduce auxiliary heads only after primary representation viability is established."
|
|
},
|
|
{
|
|
"order": 4,
|
|
"action": "Require a revised deep checkpoint to approach lite on validation before any frozen evaluation.",
|
|
"gate": {
|
|
"maximumPrimaryGapBehindLitePercentagePoints": 2.0,
|
|
"minimumHardGainOverLitePercentagePoints": 3.0,
|
|
"note": "First add the matching validation-hard report to lite so this comparison uses identical records and definitions."
|
|
}
|
|
},
|
|
{
|
|
"order": 5,
|
|
"action": "If ModernBERT pooling probes fail, compare sentence-trained encoder backbones with the same frozen probe protocol or remove the deep tier.",
|
|
"prohibited": "Do not use model size as the next variable and do not run ModernBERT large."
|
|
},
|
|
{
|
|
"order": 6,
|
|
"action": "Improve lite through validation-only error analysis and data work, then reserve a new independent holdout before another candidate cycle.",
|
|
"prohibited": "Do not use the current frozen set for optimizer, data, calibration, or checkpoint selection."
|
|
},
|
|
{
|
|
"order": 7,
|
|
"action": "Run QAT, export, target-runtime parity, latency, energy, and residency only after a float candidate qualifies.",
|
|
"shippingGate": {
|
|
"liteFrozenAccuracy": 0.95,
|
|
"deepFrozenAccuracy": 0.97,
|
|
"deepMinimumHardGainOverLitePercentagePoints": 5.0
|
|
}
|
|
}
|
|
]
|
|
}
|