Merge nucleic/fuzzy-dewy-urchin-hpvd into dev

This commit is contained in:
2026-08-02 16:38:29 -07:00
parent f89561b6a8
commit 4ad5b65811
3 changed files with 127 additions and 6 deletions
+15
View File
@@ -59,6 +59,21 @@ lines. Once `status` reports `"complete": true`, promote the staged result expli
--confirm overwrite-all-labels-with-sol-high --confirm overwrite-all-labels-with-sol-high
``` ```
If promotion reports a near-duplicate label conflict in the combined real-session
population, inspect both prompts and exclude the reviewed bad/ambiguous copy explicitly.
The 1-based line is relative to the reported `combined-real.labeled.jsonl`; repeat the
option for multiple decisions:
```bash
"$PY" ml/purpose-classifier/rebuild_sol_high.py promote \
--confirm overwrite-all-labels-with-sol-high \
--exclude-real-line <line>
```
Each exclusion is bound to the prompt hash, prior purpose, and review reason in the
combined dataset manifest and promotion result. This option does not change a teacher
label or relax duplicate detection.
Promotion first validates and curates the complete staged population. It then replaces Promotion first validates and curates the complete staged population. It then replaces
the two canonical source files, regenerates all round-two mirrors, relabels shipped the two canonical source files, regenerates all round-two mirrors, relabels shipped
fixtures (teacher-rejected fixtures become the runtime `general` fallback), rebuilds the fixtures (teacher-rejected fixtures become the runtime `general` fallback), rebuilds the
+62 -3
View File
@@ -34,6 +34,7 @@ from purpose_data import (
file_sha256, file_sha256,
jsonl_bytes, jsonl_bytes,
load_jsonl, load_jsonl,
prompt_hash,
validate_source_record, validate_source_record,
write_json, write_json,
write_jsonl, write_jsonl,
@@ -463,7 +464,41 @@ def _replace_transaction(candidates: Sequence[tuple[Path, Path]], backup_root: P
raise raise
def promote(stage: Path, confirmation: str | None) -> dict[str, Any]: def _exclude_reviewed_real_lines(
records: Sequence[dict[str, Any]],
source_lines: Sequence[int],
) -> tuple[list[dict[str, Any]], list[dict[str, Any]]]:
unique_lines = sorted(set(source_lines))
if len(unique_lines) != len(source_lines):
raise DataError("--exclude-real-line contains a duplicate line")
for line in unique_lines:
if not 1 <= line <= len(records):
raise DataError(
f"--exclude-real-line {line} is outside the combined real population "
f"(1-{len(records)})"
)
excluded = set(unique_lines)
review = [
{
"sourceLine": line,
"promptHash": prompt_hash(records[line - 1]["prompt"]),
"purpose": records[line - 1]["purpose"],
"reason": "reviewed-near-duplicate-label-conflict",
}
for line in unique_lines
]
return (
[record for line, record in enumerate(records, 1) if line not in excluded],
review,
)
def promote(
stage: Path,
confirmation: str | None,
excluded_real_lines: Sequence[int] = (),
) -> dict[str, Any]:
if confirmation != CONFIRMATION: if confirmation != CONFIRMATION:
raise DataError(f"promotion requires --confirm {CONFIRMATION}") raise DataError(f"promotion requires --confirm {CONFIRMATION}")
load_config(stage) load_config(stage)
@@ -514,9 +549,12 @@ def promote(stage: Path, confirmation: str | None) -> dict[str, Any]:
state_by_name["fixtures"], state_by_name["fixtures"],
) )
combined_real = promotion / "combined-real.labeled.jsonl" combined_real = promotion / "combined-real.labeled.jsonl"
real_records = _records_from_states( all_real_records = _records_from_states(
state_by_name["history"] state_by_name["history"]
) + _records_from_states(state_by_name["swe"]) ) + _records_from_states(state_by_name["swe"])
real_records, real_review = _exclude_reviewed_real_lines(
all_real_records, excluded_real_lines
)
write_jsonl(combined_real, real_records) write_jsonl(combined_real, real_records)
public_dataset = promotion / "dataset-public" public_dataset = promotion / "dataset-public"
@@ -558,6 +596,10 @@ def promote(stage: Path, confirmation: str | None) -> dict[str, Any]:
combined_manifest["policy"]["historyUsage"] = ( combined_manifest["policy"]["historyUsage"] = (
"Nucleic history and SWE-chat are training-only" "Nucleic history and SWE-chat are training-only"
) )
combined_manifest["policy"]["reviewedRealLabelConflicts"] = (
"explicitly excluded before near-duplicate curation"
)
combined_manifest["reviewedRealExclusions"] = real_review
combined_manifest["sources"]["baseDataset"]["path"] = _relative( combined_manifest["sources"]["baseDataset"]["path"] = _relative(
PUBLIC_DATASET_DESTINATION PUBLIC_DATASET_DESTINATION
) )
@@ -608,6 +650,7 @@ def promote(stage: Path, confirmation: str | None) -> dict[str, Any]:
"teacher": {"model": MODEL, "reasoningEffort": REASONING_EFFORT}, "teacher": {"model": MODEL, "reasoningEffort": REASONING_EFFORT},
"publicLabeled": sum(len(value) for value in public_records.values()), "publicLabeled": sum(len(value) for value in public_records.values()),
"realLabeled": len(real_records), "realLabeled": len(real_records),
"realReviewExclusions": real_review,
"combinedTrain": combined_manifest["outputs"]["train"]["records"], "combinedTrain": combined_manifest["outputs"]["train"]["records"],
"validation": combined_manifest["outputs"]["validation"]["records"], "validation": combined_manifest["outputs"]["validation"]["records"],
"test": combined_manifest["outputs"]["test"]["records"], "test": combined_manifest["outputs"]["test"]["records"],
@@ -663,6 +706,16 @@ def build_parser() -> argparse.ArgumentParser:
"promote", help="curate and replace canonical labels" "promote", help="curate and replace canonical labels"
) )
promote_parser.add_argument("--confirm") promote_parser.add_argument("--confirm")
promote_parser.add_argument(
"--exclude-real-line",
action="append",
default=[],
type=int,
help=(
"exclude a reviewed 1-based line from the combined history+SWE training "
"augmentation; repeat for each adjudicated conflict"
),
)
commands_parser = subparsers.add_parser( commands_parser = subparsers.add_parser(
"train-commands", help="print clean from-base training commands" "train-commands", help="print clean from-base training commands"
) )
@@ -701,7 +754,13 @@ def main(argv: Sequence[str] | None = None) -> int:
elif args.command == "status": elif args.command == "status":
print(json.dumps(status(stage), indent=2, sort_keys=True)) print(json.dumps(status(stage), indent=2, sort_keys=True))
elif args.command == "promote": elif args.command == "promote":
print(json.dumps(promote(stage, args.confirm), indent=2, sort_keys=True)) print(
json.dumps(
promote(stage, args.confirm, args.exclude_real_line),
indent=2,
sort_keys=True,
)
)
else: else:
print(train_commands(args.python)) print(train_commands(args.python))
except ( except (
+50 -3
View File
@@ -185,7 +185,18 @@ class RebuildSolHighTests(unittest.TestCase):
history = root / "history.jsonl" history = root / "history.jsonl"
purpose_data.write_jsonl(history, [{"prompt": "Add a private cache layer"}]) purpose_data.write_jsonl(history, [{"prompt": "Add a private cache layer"}])
swe = root / "swe.jsonl" swe = root / "swe.jsonl"
purpose_data.write_jsonl(swe, [{"prompt": "Diagnose a unique worker crash"}]) duplicate_words = [f"duplicate{index}" for index in range(100)]
first_duplicate = " ".join(duplicate_words)
duplicate_words[50] = "replacement"
second_duplicate = " ".join(duplicate_words)
purpose_data.write_jsonl(
swe,
[
{"prompt": "Diagnose a unique worker crash"},
{"prompt": first_duplicate},
{"prompt": second_duplicate},
],
)
stage = script_dir / ".artifacts" / "sol-high-reset" stage = script_dir / ".artifacts" / "sol-high-reset"
public_dataset = script_dir / ".artifacts" / "dataset-public" public_dataset = script_dir / ".artifacts" / "dataset-public"
combined_dataset = script_dir / ".artifacts" / "dataset-v1" combined_dataset = script_dir / ".artifacts" / "dataset-v1"
@@ -222,7 +233,10 @@ class RebuildSolHighTests(unittest.TestCase):
(item for item in records if item["prompt"] == prompt), (item for item in records if item["prompt"] == prompt),
None, None,
) )
record = original or source_record(prompt) record = original or source_record(
prompt,
"writing" if prompt == second_duplicate else "backendImpl",
)
states.append( states.append(
{ {
"schemaVersion": 1, "schemaVersion": 1,
@@ -240,10 +254,12 @@ class RebuildSolHighTests(unittest.TestCase):
result = rebuild_sol_high.promote( result = rebuild_sol_high.promote(
stage, stage,
rebuild_sol_high.CONFIRMATION, rebuild_sol_high.CONFIRMATION,
excluded_real_lines=[3],
) )
self.assertEqual(80, result["publicLabeled"]) self.assertEqual(80, result["publicLabeled"])
self.assertEqual(2, result["realLabeled"]) self.assertEqual(3, result["realLabeled"])
self.assertEqual(3, result["realReviewExclusions"][0]["sourceLine"])
self.assertTrue((combined_dataset / "train.jsonl").is_file()) self.assertTrue((combined_dataset / "train.jsonl").is_file())
self.assertGreater( self.assertGreater(
len(purpose_data.load_jsonl(combined_dataset / "train.jsonl")), len(purpose_data.load_jsonl(combined_dataset / "train.jsonl")),
@@ -256,11 +272,42 @@ class RebuildSolHighTests(unittest.TestCase):
"purpose-dataset-sol-high-v2", "purpose-dataset-sol-high-v2",
promoted_manifest["datasetVersion"], promoted_manifest["datasetVersion"],
) )
combined_manifest = json.loads(
(combined_dataset / "manifest.json").read_text(encoding="utf-8")
)
self.assertEqual(
result["realReviewExclusions"],
combined_manifest["reviewedRealExclusions"],
)
def test_promotion_requires_exact_confirmation(self): def test_promotion_requires_exact_confirmation(self):
with self.assertRaisesRegex(purpose_data.DataError, "promotion requires"): with self.assertRaisesRegex(purpose_data.DataError, "promotion requires"):
rebuild_sol_high.promote(Path("unused"), None) rebuild_sol_high.promote(Path("unused"), None)
def test_reviewed_real_line_exclusions_are_auditable(self):
records = [
source_record("Plan the cache migration", "planning"),
source_record("Review the cache migration", "review"),
]
retained, review = rebuild_sol_high._exclude_reviewed_real_lines(records, [1])
self.assertEqual([records[1]], retained)
self.assertEqual(1, review[0]["sourceLine"])
self.assertEqual("planning", review[0]["purpose"])
self.assertEqual(
"reviewed-near-duplicate-label-conflict", review[0]["reason"]
)
self.assertEqual(64, len(review[0]["promptHash"]))
def test_reviewed_real_line_exclusions_reject_invalid_lines(self):
records = [source_record("Plan the cache migration", "planning")]
with self.assertRaisesRegex(purpose_data.DataError, "duplicate line"):
rebuild_sol_high._exclude_reviewed_real_lines(records, [1, 1])
with self.assertRaisesRegex(purpose_data.DataError, "outside"):
rebuild_sol_high._exclude_reviewed_real_lines(records, [2])
if __name__ == "__main__": if __name__ == "__main__":
unittest.main() unittest.main()