Files

636 lines
23 KiB
JSON
Raw Permalink Normal View History

{
"schemaVersion": 1,
"experiment": "purpose-sol-high-v2-training",
"recordedAt": "2026-08-03",
"sourceCheckoutCommit": "531e9ba301a20dd124062afed47aa412244c407a",
"status": "no-qualified-sol-high-v2-candidate",
"summary": "The Sol-high label reset and promotion completed. The selected lite float checkpoint reached 93.7167% frozen accuracy, while the best deep validation checkpoint reached 77.6445%. Neither meets the shipping contract; frozen-set-driven tuning and further deep continuations are stopped.",
"privacy": {
"policy": "Private Nucleic-history and SWE-chat prompts remain in ignored artifacts.",
"recordedIdentifiers": "Only aggregate counts, content hashes, reviewed source lines, and reviewed prompt hashes are versioned here."
},
"labelReset": {
"teacher": {
"model": "gpt-5.6-sol",
"reasoningEffort": "high"
},
"inputDecisions": {
"total": 15048,
"labeled": 14704,
"rejected": 344,
"populations": {
"canonicalPublic": {
"labeled": 12007,
"rejected": 207
},
"shippedFixtures": {
"labeled": 89,
"rejectedToGeneral": 3
},
"nucleicHistory": {
"labeled": 1068,
"rejected": 68
},
"sweChat": {
"labeled": 1540,
"rejected": 66
}
}
},
"promotion": {
"backup": "ml/purpose-classifier/.artifacts/sol-high-reset/backups/20260802T234020Z",
"publicLabeled": 12007,
"realLabeledAfterReviewedExclusions": 2606,
"combinedTrain": 12164,
"validation": 1208,
"syntheticTest": 1119,
"reviewedRealExclusions": [
{
"sourceLine": 1201,
"promptHash": "124e98f338b935a7efd0d4e12295eb59f07ee68e4dc8ab65bc069b630ee93110",
"priorPurpose": "planning",
"reason": "reviewed-near-duplicate-label-conflict"
},
{
"sourceLine": 1481,
"promptHash": "a6dd960a0857f2d8af5ef971582e480456a33599df8af27d50cddb5e9f828035",
"priorPurpose": "refactor",
"reason": "reviewed-near-duplicate-label-conflict"
}
],
"validationCommandResult": {
"records": 12007,
"errors": 0,
"expectedShapeWarnings": 22
}
}
},
"dataset": {
"version": "purpose-dataset-sol-high-v2",
"public": {
"inputRecords": 12007,
"nearDuplicateExclusions": 18,
"retainedRecords": 11989,
"train": {
"records": 9662,
"sha256": "0e16940308dc7557c73b1b804bf7b715c86d263b613201588ec274169ee01e89"
},
"validation": {
"records": 1208,
"scorableRecords": 917,
"vagueEvalRecords": 291,
"sha256": "301cd3d1c69e1bb92b811e857093ad12fb8c5c11bb402ee2993403d5e4467969"
},
"syntheticTest": {
"records": 1119,
"vagueEvalRecords": 269,
"sha256": "9997dbaa6d305ea56d8c161e39b925368761ab613a706eb8c2ba55d0e58b1906"
},
"shippedFixtures": {
"records": 92,
"classifiableRecords": 89,
"generalRecords": 3,
"sha256": "876068ea26d7109365bcfbd1dc44ba3f1a0400c80dc7fdacf008b3fa07cd92cc"
},
"logicalFrozenRecords": 1208,
"scorableFrozenRecords": 939
},
"combinedTrainingArtifact": {
"experimentVersion": "all-sol-high-training-augmentation-v2",
"train": {
"records": 12164,
"sha256": "44a4bbccf2c8207f7f497e8138ee01007c687034a7c719604f50cbeba35e5bcc"
},
"validation": {
"records": 1208,
"sha256": "301cd3d1c69e1bb92b811e857093ad12fb8c5c11bb402ee2993403d5e4467969"
},
"syntheticTest": {
"records": 1119,
"sha256": "9997dbaa6d305ea56d8c161e39b925368761ab613a706eb8c2ba55d0e58b1906"
},
"realInputAfterReviewedExclusions": {
"records": 2606,
"sha256": "ed38f4ffdb640a2a9decf1407c23ff8c8c8a150bbe1f55b1b051033d5bcc1ade"
},
"acceptedRealAugmentation": 2502,
"excludedRealAugmentation": {
"total": 104,
"vagueEval": 87,
"nearHistoryDuplicate": 17
}
},
"manifestHashes": {
"datasetV1Manifest": "c6be5d0df6fd87ab3f5a254d3b8e280cae786cdc90e3cd0cdee02c6662dd27b1",
"generationManifest": "bb707c81047cb1843562d1641abe344494ef0c4aab5af7ae56283d6769160b2a",
"supersededCurationReview": "f8b47a08a10ff872f2a29e6f3321d4818ab439a2429d7162692b58c9281fdf47",
"ignoredCombinedArtifactManifest": "f8756e6f3cf15b00de623d379f41fbd98dba7798d8ebfb9668d5dce4e767cd56"
}
},
"runs": [
{
"id": "purpose-lite-sol-high-v2",
"tier": "lite",
"kind": "from-pretrained-float",
"outputDirectory": "ml/purpose-classifier/outputs/purpose-lite-sol-high-v2",
"resolvedTraining": {
"baseModel": "sentence-transformers/all-MiniLM-L6-v2",
"baseModelRevision": "1110a243fdf4706b3f48f1d95db1a4f5529b4d41",
"device": "mps",
"epochs": 3,
"batchSize": 32,
"learningRate": 0.00002,
"warmupRatio": 0.1,
"boundaryWeight": 1.0,
"trainingSeconds": 191.07675279202522
},
"selectedValidation": {
"epoch": 3,
"records": 917,
"accuracy": 0.9127589967284624,
"macroRecall": 0.9127308084599176
},
"frozenEvaluation": {
"scorable": {
"correct": 876,
"records": 939,
"accuracy": 0.9329073482428115,
"macroRecall": 0.9322276987713067
},
"hard": {
"correct": 191,
"records": 215,
"accuracy": 0.8883720930232558
},
"shippedFixtures": {
"correct": 79,
"records": 89,
"accuracy": 0.8876404494382022
},
"sliceAccuracy": {
"boundary": 0.6862745098039216,
"core": 0.9543307086614173,
"mixed": 0.9195402298850575,
"pastedContext": 0.987012987012987,
"shippedFixtures": 0.8876404494382022
},
"latencyMilliseconds": {
"median": 17.3936,
"p95": 21.5198
},
"failedGates": [
"scoredAccuracyAtLeast95",
"p95LatencyAtMost20Milliseconds"
]
},
"decision": "rejected"
},
{
"id": "purpose-lite-sol-high-v2-boundary-cont",
"tier": "lite",
"kind": "validation-selected-boundary-continuation",
"resumedFrom": "ml/purpose-classifier/outputs/purpose-lite-sol-high-v2/model",
"outputDirectory": "ml/purpose-classifier/outputs/purpose-lite-sol-high-v2-boundary-cont",
"resolvedTraining": {
"device": "mps",
"epochZeroValidationAccuracy": 0.9127589967284624,
"epochsRequested": 3,
"epochsCompleted": 3,
"batchSize": 32,
"learningRate": 0.000003,
"warmupRatio": 0.0,
"boundaryWeight": 2.0,
"earlyStoppingPatience": 2,
"trainingSeconds": 257.24039216700476
},
"selectedValidation": {
"epoch": 1,
"records": 917,
"accuracy": 0.9203925845147219,
"macroRecall": 0.9209676130783011
},
"frozenEvaluation": {
"scorable": {
"correct": 880,
"records": 939,
"accuracy": 0.9371671991480298,
"macroRecall": 0.9368046325084637,
"correctDecisionsShortOf95Percent": 13
},
"hard": {
"correct": 196,
"records": 215,
"accuracy": 0.9116279069767442
},
"shippedFixtures": {
"correct": 76,
"records": 89,
"accuracy": 0.8539325842696629
},
"sliceAccuracy": {
"boundary": 0.7647058823529411,
"core": 0.9574803149606299,
"mixed": 0.9310344827586207,
"pastedContext": 0.987012987012987,
"shippedFixtures": 0.8539325842696629
},
"latencyMilliseconds": {
"median": 6.192041502799839,
"p95": 8.401416009292006
},
"failedGates": [
"scoredAccuracyAtLeast95"
]
},
"decision": "rejected-as-shipping-candidate; retained-as-validation-teacher-and-diagnostic-baseline"
},
{
"id": "purpose-lite-sol-high-v2-teacher",
"tier": "lite",
"kind": "prompt-and-label-bound-logit-cache",
"sourceModel": "ml/purpose-classifier/outputs/purpose-lite-sol-high-v2-boundary-cont/model",
"output": "ml/purpose-classifier/outputs/purpose-lite-sol-high-v2-teacher.pt",
"records": {
"train": 12164,
"validation": 1208
},
"sha256": "a4175f0dd28c7d6e6936aabd83239aa2cfa8799aea1aaa70d0fe461377766a88",
"decision": "retained-for-reproducibility; deep distillation did not produce a viable candidate"
},
{
"id": "purpose-deep-sol-high-v2-base",
"tier": "deep",
"kind": "from-pretrained-multitask",
"outputDirectory": "ml/purpose-classifier/outputs/purpose-deep-sol-high-v2-base",
"resolvedTraining": {
"baseModelRevision": "8949b909ec900327062f0ebf497f51aef5e6f0c8",
"device": "metal",
"epochs": 3,
"batchSize": 4,
"learningRate": 0.00002,
"hardWeight": 2.0,
"secondaryLossWeight": 0.25,
"mixedLossWeight": 0.25,
"difficultyLossWeight": 0.1,
"trainingSeconds": 6963.924981958
},
"validationTrajectory": [
{
"epoch": 1,
"primaryAccuracy": 0.6063249727371864,
"hardAccuracy": 0.5791666666666667
},
{
"epoch": 2,
"primaryAccuracy": 0.7022900763358778,
"hardAccuracy": 0.6291666666666667
},
{
"epoch": 3,
"primaryAccuracy": 0.7339149400218102,
"hardAccuracy": 0.675
}
],
"selectedValidation": {
"epoch": 3,
"primaryRecords": 917,
"hardRecords": 240,
"primaryAccuracy": 0.7339149400218102,
"hardAccuracy": 0.675,
"selectionScore": 0.7044574700109052,
"primaryMacroRecall": 0.7354311324151239,
"perPurposeRecall": {
"planning": 0.7192982456140351,
"backendImpl": 0.7844827586206896,
"frontendImpl": 0.7596153846153846,
"quickFix": 0.7685185185185185,
"refactor": 0.8448275862068966,
"debugging": 0.6538461538461539,
"review": 0.7739130434782608,
"writing": 0.5789473684210527
},
"secondarySupportedMacroRecall": 0.3516156462585034,
"mixedF1Raw": 0.30158730158730157,
"mixedF1Calibrated": 0.3971119133574007,
"difficultyMae": 0.1273345649242401,
"vagueLowRate": 0.9759450171821306
},
"frozenEvaluation": null,
"decision": "rejected-on-validation; no-frozen-look; do-not-run-large"
},
{
"id": "purpose-deep-sol-high-v2-base-distilled",
"tier": "deep",
"kind": "lite-teacher-distilled-continuation",
"resumedFrom": "ml/purpose-classifier/outputs/purpose-deep-sol-high-v2-base/model",
"outputDirectory": "ml/purpose-classifier/outputs/purpose-deep-sol-high-v2-base-distilled",
"resolvedTraining": {
"device": "metal",
"epochs": 2,
"batchSize": 4,
"learningRate": 0.00001,
"distillationWeight": 0.5,
"distillationTemperature": 2.0,
"hardWeight": 2.0,
"secondaryLossWeight": 0.25,
"mixedLossWeight": 0.25,
"difficultyLossWeight": 0.1,
"trainingSeconds": 4237.380279583012
},
"validationTrajectory": [
{
"epoch": 1,
"primaryAccuracy": 0.7611777535441657,
"hardAccuracy": 0.7,
"teacherAgreement": 0.7699018538713195
},
{
"epoch": 2,
"primaryAccuracy": 0.7764449291166848,
"hardAccuracy": 0.7291666666666666,
"teacherAgreement": 0.7851690294438386
}
],
"selectedValidation": {
"epoch": 2,
"primaryRecords": 917,
"hardRecords": 240,
"primaryAccuracy": 0.7764449291166848,
"hardAccuracy": 0.7291666666666666,
"selectionScore": 0.7528057978916758,
"primaryMacroRecall": 0.7781732980788059,
"perPurposeRecall": {
"planning": 0.8070175438596491,
"backendImpl": 0.8017241379310345,
"frontendImpl": 0.7884615384615384,
"quickFix": 0.7870370370370371,
"refactor": 0.9051724137931034,
"debugging": 0.6692307692307692,
"review": 0.7913043478260869,
"writing": 0.6754385964912281
},
"secondarySupportedMacroRecall": 0.415391156462585,
"mixedF1Raw": 0.366412213740458,
"mixedF1Calibrated": 0.4733727810650888,
"difficultyMae": 0.1244114488363266,
"vagueLowRate": 0.9759450171821306,
"teacherAgreement": 0.7851690294438386
},
"comparison": {
"primaryGainOverDeepBasePercentagePoints": 4.253,
"hardGainOverDeepBasePercentagePoints": 5.417,
"primaryGapBehindSelectedLiteValidationPercentagePoints": 14.395
},
"frozenEvaluation": null,
"decision": "rejected-on-validation; no-frozen-look; stop-continuations-and-do-not-run-large-or-max"
}
],
"reproductionCommands": {
"note": "Use the purpose-classifier virtual environment as ${PYTHON}. These commands make defaults material; output directories are gitignored and their retained files must match artifactHashes.",
"promotion": [
"${PYTHON}",
"ml/purpose-classifier/rebuild_sol_high.py",
"promote",
"--confirm",
"overwrite-all-labels-with-sol-high",
"--exclude-real-line",
"1201",
"--exclude-real-line",
"1481"
],
"liteFromPretrained": [
"${PYTHON}",
"-u",
"ml/purpose-classifier/train.py",
"--device",
"mps",
"--model",
"sentence-transformers/all-MiniLM-L6-v2",
"--dataset-dir",
"ml/purpose-classifier/.artifacts/dataset-v1",
"--epochs",
"3",
"--learning-rate",
"2e-5",
"--warmup-ratio",
"0.1",
"--boundary-weight",
"1",
"--early-stopping-patience",
"2",
"--output-dir",
"ml/purpose-classifier/outputs/purpose-lite-sol-high-v2",
"--overwrite-output"
],
"liteFromPretrainedFrozenEvaluation": [
"${PYTHON}",
"-u",
"ml/purpose-classifier/eval.py",
"--device",
"mps",
"--model-dir",
"ml/purpose-classifier/outputs/purpose-lite-sol-high-v2/model",
"--calibration",
"ml/purpose-classifier/outputs/purpose-lite-sol-high-v2/calibration.json",
"--report",
"ml/purpose-classifier/outputs/purpose-lite-sol-high-v2/frozen-eval.json"
],
"liteBoundaryContinuation": [
"${PYTHON}",
"-u",
"ml/purpose-classifier/train.py",
"--device",
"mps",
"--model",
"ml/purpose-classifier/outputs/purpose-lite-sol-high-v2/model",
"--dataset-dir",
"ml/purpose-classifier/.artifacts/dataset-v1",
"--epochs",
"3",
"--learning-rate",
"3e-6",
"--warmup-ratio",
"0",
"--boundary-weight",
"2",
"--early-stopping-patience",
"2",
"--output-dir",
"ml/purpose-classifier/outputs/purpose-lite-sol-high-v2-boundary-cont",
"--overwrite-output"
],
"liteBoundaryFrozenEvaluation": [
"${PYTHON}",
"-u",
"ml/purpose-classifier/eval.py",
"--device",
"mps",
"--model-dir",
"ml/purpose-classifier/outputs/purpose-lite-sol-high-v2-boundary-cont/model",
"--calibration",
"ml/purpose-classifier/outputs/purpose-lite-sol-high-v2-boundary-cont/calibration.json",
"--report",
"ml/purpose-classifier/outputs/purpose-lite-sol-high-v2-boundary-cont/frozen-eval.json"
],
"teacherCache": [
"${PYTHON}",
"-u",
"ml/purpose-classifier/cache_teacher.py",
"--device",
"mps",
"--dataset-dir",
"ml/purpose-classifier/.artifacts/dataset-v1",
"--model",
"ml/purpose-classifier/outputs/purpose-lite-sol-high-v2-boundary-cont/model",
"--output",
"ml/purpose-classifier/outputs/purpose-lite-sol-high-v2-teacher.pt",
"--batch-size",
"16",
"--progress-steps",
"25",
"--overwrite-output"
],
"deepBase": [
"${PYTHON}",
"-u",
"ml/purpose-classifier/train_deep_mlx.py",
"--variant",
"base",
"--device",
"metal",
"--dataset-dir",
"ml/purpose-classifier/.artifacts/dataset-v1",
"--epochs",
"3",
"--early-stopping-patience",
"1",
"--progress-steps",
"10",
"--output-dir",
"ml/purpose-classifier/outputs/purpose-deep-sol-high-v2-base",
"--overwrite-output"
],
"deepDistilledContinuation": [
"${PYTHON}",
"-u",
"ml/purpose-classifier/train_deep_mlx.py",
"--variant",
"base",
"--device",
"metal",
"--resume-from",
"ml/purpose-classifier/outputs/purpose-deep-sol-high-v2-base/model",
"--dataset-dir",
"ml/purpose-classifier/.artifacts/dataset-v1",
"--distillation-cache",
"ml/purpose-classifier/outputs/purpose-lite-sol-high-v2-teacher.pt",
"--distillation-weight",
"0.5",
"--distillation-temperature",
"2",
"--epochs",
"2",
"--learning-rate",
"1e-5",
"--early-stopping-patience",
"1",
"--progress-steps",
"10",
"--output-dir",
"ml/purpose-classifier/outputs/purpose-deep-sol-high-v2-base-distilled",
"--overwrite-output"
]
},
"artifactHashes": {
"purpose-lite-sol-high-v2": {
"metrics.json": "ed6216f37e56e1c1af554d7c3cb614e3f1e2d0e856a2a6fc87c1aeaf5da86e15",
"calibration.json": "7ebe788f86b43d5f9927ef891fc354bd77499b68b7c4f250e8eda613b8fe2168",
"training-config.json": "b6547867830a29365a5a68bca1d725da8ee342fd67bf9297f6b826f454ab331a",
"frozen-eval.json": "f6070bb82759521a8011d7705bf07d3660b4ba10b4181dbd9ea9589118e9e9ce",
"model/model.safetensors": "0ca2284f443bcaf3b05c1a673f015f6856b3dc3152f936fa9a7d35ab81551f2e"
},
"purpose-lite-sol-high-v2-boundary-cont": {
"metrics.json": "bf97b5ac1926e735b9d210f9c6a286d7e8afe889a4f7d38aa4b899cf68e4dea5",
"calibration.json": "e98349819b99819ec12806345a7c642e7716bb6302f4d796d7253681c05e20a4",
"training-config.json": "cd3ce124b17589cef2f3c63170323f875ee9782d924d3c8067062d59125ece3e",
"frozen-eval.json": "780c504b1bdddbb7b6cd9781583d4ac2976278ae379f7e8a6825d2e7fdc97aee",
"model/model.safetensors": "c7f004dbaab01b3331ef7eb73d090e6d110afd531598e9998ac40730b608a2c6"
},
"purpose-deep-sol-high-v2-base": {
"metrics.json": "3ac5b0345da17fef3407eed2cf60888898734eaa401ec1ca9e5f6df2afd6ffa5",
"calibration.json": "fbedfa506082aed36e9c120235f0d7a7fb9d6c944f4ad1622db7c27c1a509e1b",
"training-config.json": "29fd34926f0f688e1b3c78f14f422ec8c59817b67346b672047f7307f6e487b5",
"training-state.json": "221ae224898cbbfddd649b3dae17260bacfcfbb30edfa13c854fa0bbea36704e",
"model/model.safetensors": "09205c022d95088300f04d90d52055ce218474f9bc5be6a85b428cdbacbe1782"
},
"purpose-deep-sol-high-v2-base-distilled": {
"metrics.json": "87a4c55ac9867fbb85e15d2f5e725fefb2eb34f98e91609136fbe8d3a009ba3c",
"calibration.json": "2da1735b4879bd6caf8a1ff7bb5dbff65ba78c6335478bd949c9ad5093ad03e6",
"training-config.json": "6242c711c8f070113987e57981f5879c72729000141b8559621b4ab6ec7b834e",
"training-state.json": "e8093c0490a91590600e7d436858b025878c702490dde0bb59e98ff7c6bcaa53",
"model/model.safetensors": "568c801c70a3caa49e0ab9afd10f4c36c415cf9b9bc4523d6d2bd37b9734c4b3"
}
},
"decisions": [
"The missing MiniLM classifier.weight and classifier.bias load report is expected because the pretrained encoder has no task-specific eight-way classifier head; downstream training initializes those two tensors.",
"Do not tune another lite continuation against the Sol-high v2 frozen set.",
"Do not QAT or export a Sol-high v2 lite candidate until a float checkpoint first clears the validation and frozen qualification policy without frozen-driven selection.",
"Do not run another plain or distilled deep continuation, ModernBERT large, or purpose-max from these results.",
"The previously deployed v1 W8A8 artifact is historical runtime evidence only and is not label-compatible with Sol-high v2."
],
"nextSteps": [
{
"order": 1,
"action": "Implement a validation-only representation diagnostic harness.",
"details": [
"Report the exact same 917-record primary and 240-record hard metrics used by the deep trainer.",
"Cache ModernBERT representations with keys bound to model revision, tokenizer contract, dataset hash, and normalized prompt hash.",
"Compare the current pretrained prediction-head CLS representation with attention-masked mean pooling.",
"Fit only a regularized eight-way primary linear probe; exclude secondary, mixed-intent, and difficulty losses."
]
},
{
"order": 2,
"action": "Apply a fail-fast probe gate before another full deep run.",
"gate": {
"minimumMeanPoolingGainOverClsPercentagePointsOnEitherOverallOrHard": 3.0,
"minimumOverallAccuracy": 0.85,
"perPurposeCollapseAllowed": false
}
},
{
"order": 3,
"action": "If the probe passes, add explicit --pooling cls|mean and --primary-only trainer options and run one from-pretrained ModernBERT-base experiment.",
"selection": "Validation overall and hard primary accuracy only; introduce auxiliary heads only after primary representation viability is established."
},
{
"order": 4,
"action": "Require a revised deep checkpoint to approach lite on validation before any frozen evaluation.",
"gate": {
"maximumPrimaryGapBehindLitePercentagePoints": 2.0,
"minimumHardGainOverLitePercentagePoints": 3.0,
"note": "First add the matching validation-hard report to lite so this comparison uses identical records and definitions."
}
},
{
"order": 5,
"action": "If ModernBERT pooling probes fail, compare sentence-trained encoder backbones with the same frozen probe protocol or remove the deep tier.",
"prohibited": "Do not use model size as the next variable and do not run ModernBERT large."
},
{
"order": 6,
"action": "Improve lite through validation-only error analysis and data work, then reserve a new independent holdout before another candidate cycle.",
"prohibited": "Do not use the current frozen set for optimizer, data, calibration, or checkpoint selection."
},
{
"order": 7,
"action": "Run QAT, export, target-runtime parity, latency, energy, and residency only after a float candidate qualifies.",
"shippingGate": {
"liteFrozenAccuracy": 0.95,
"deepFrozenAccuracy": 0.97,
"deepMinimumHardGainOverLitePercentagePoints": 5.0
}
}
]
}