updating purpose-classifier data:

This commit is contained in:
2026-08-02 20:15:13 -07:00
parent 98311aa932
commit 0f639bfa05
21 changed files with 14248 additions and 14463 deletions
+60 -60
View File
@@ -1,12 +1,66 @@
{
"datasetVersion": "purpose-dataset-sol-high-v2",
"humanLabelAndDifficultyReview": {
"decision": "Every supervised target was regenerated with gpt-5.6-sol at high reasoning effort.",
"decisionBasis": "The dataset owner invalidated the original labels and required a complete Sol-high relabel.",
"decisionDate": "2026-08-02",
"generatedArtifact": "ml/purpose-classifier/.artifacts/human-review-v1.csv",
"populationRecords": 12193,
"reviewFields": [
"purpose",
"secondary",
"difficulty",
"slice",
"review status",
"notes"
],
"sampleFraction": 0.1,
"sampleRecords": 1219,
"seed": 10558743,
"status": "superseded-by-sol-high-relabel",
"strata": [
"purpose",
"slice",
"primary language"
]
},
"schemaVersion": 1,
"datasetVersion": "purpose-dataset-v1",
"semanticDuplicateReview": {
"auditVersion": "purpose-semantic-audit-v1",
"decision": "The prior semantic decisions were tied to the invalidated label population; the reset uses deterministic lexical curation only.",
"excluded": [
{
"droppedPromptHash": "7348d764d602cfff72dd2b1d369ad0375a96ea52b38174aa0bd6168a548bcc2f",
"matchedPromptHash": "3ad9cfbb664504f0e7ae84b905520937e7812d4e879fa706415782341f2cd400",
"similarity": 0.984897
},
{
"droppedPromptHash": "a0c6e7984e92adc11ea6855c39688d733bf7665f5f965939325b15f3ba39e937",
"matchedPromptHash": "17f79a890a323c08770193a94b100dadaa00116405bdb7d14c1b89abc3c3066a",
"similarity": 0.980998
},
{
"droppedPromptHash": "bfd7791069fd04c13797f39ba99224ad68ca2deebbc912f08b10815954a7d223",
"matchedPromptHash": "dc48031efb18f25a56a3beddcdd814f26473b23cd496482ce1884ae6f697417a",
"similarity": 0.973329
}
],
"excludedCandidates": 3,
"retained": [
{
"leftPromptHash": "70944e70e0062e947476f73b704ec01ecee19420d06fd174c58a41c04e8112f6",
"rightPromptHash": "cb3e63992cbf5830a222933dc7b44be39933baedcb46e335be02e3c379685517",
"similarity": 0.970376
}
],
"retainedCandidates": 1,
"reviewedCandidates": 4,
"status": "superseded-by-sol-high-relabel"
},
"wordTrigramExclusionReview": {
"status": "complete",
"reviewedRecords": 18,
"confirmedTemplateDuplicates": 18,
"rejectedExclusions": 0,
"decision": "All 18 pairs preserve the same task, requested outcome, and purpose label; differences are generated identifiers, ticket numbers, cosmetic context prefixes, or difficulty/slice metadata. Keep the earliest member and exclude the later template copy.",
"rejectedExclusions": 0,
"reviewedDroppedPromptHashes": [
"f29de1d646055017f80346055bb3ac1468c49af7ab0c6249f45f19cf386f9fa7",
"dabbcd0a19d2918199b5d0d35fac49fd5413cb17a5164252b789114fc68ec43b",
@@ -26,62 +80,8 @@
"fdc04376376dbffcaf9928e9edcef29376203640b49c200c2c117f78891125a1",
"432e67ead52d6cd57e4a5034cd8bcde14bd2da13f88cb5532c42fd8fbd8daf05",
"24f3759ff4a3fb9fc77bed4c8525f30f9af2479a6cffbb0c4842043f99a7d8f1"
]
},
"semanticDuplicateReview": {
"status": "complete",
"auditVersion": "purpose-semantic-audit-v1",
"reviewedCandidates": 4,
"excludedCandidates": 3,
"retainedCandidates": 1,
"decision": "Exclude three same-purpose generated template copies. One pair ('make the screen nicer' / 'Make this screen nicer') remains because it is a deliberate vague-language variant, lives entirely in validation, and does not contaminate the frozen evaluation boundary.",
"excluded": [
{
"droppedPromptHash": "7348d764d602cfff72dd2b1d369ad0375a96ea52b38174aa0bd6168a548bcc2f",
"matchedPromptHash": "3ad9cfbb664504f0e7ae84b905520937e7812d4e879fa706415782341f2cd400",
"similarity": 0.984897
},
{
"droppedPromptHash": "a0c6e7984e92adc11ea6855c39688d733bf7665f5f965939325b15f3ba39e937",
"matchedPromptHash": "17f79a890a323c08770193a94b100dadaa00116405bdb7d14c1b89abc3c3066a",
"similarity": 0.980998
},
{
"droppedPromptHash": "bfd7791069fd04c13797f39ba99224ad68ca2deebbc912f08b10815954a7d223",
"matchedPromptHash": "dc48031efb18f25a56a3beddcdd814f26473b23cd496482ce1884ae6f697417a",
"similarity": 0.973329
}
],
"retained": [
{
"leftPromptHash": "70944e70e0062e947476f73b704ec01ecee19420d06fd174c58a41c04e8112f6",
"rightPromptHash": "cb3e63992cbf5830a222933dc7b44be39933baedcb46e335be02e3c379685517",
"similarity": 0.970376
}
]
},
"humanLabelAndDifficultyReview": {
"status": "accepted-as-generated",
"populationRecords": 12193,
"sampleFraction": 0.1,
"sampleRecords": 1219,
"seed": 10558743,
"strata": [
"purpose",
"slice",
"primary language"
],
"reviewFields": [
"purpose",
"secondary",
"difficulty",
"slice",
"review status",
"notes"
],
"generatedArtifact": "ml/purpose-classifier/.artifacts/human-review-v1.csv",
"decisionDate": "2026-07-31",
"decisionBasis": "The dataset owner explicitly directed the project to assume the generated labels, secondary purposes, difficulties, and slices are correct and validated without completing the row-by-row sample.",
"decision": "Accept the curated generated population as-is. No relabel or reject decisions are inferred, and the blank CSV remains an optional future audit artifact rather than a rollout blocker."
"reviewedRecords": 18,
"status": "complete"
}
}