import json import os import subprocess import sys import tempfile import unittest from pathlib import Path MODULE_DIR = Path(__file__).resolve().parents[1] sys.path.insert(0, str(MODULE_DIR)) import label_nucleic_prompts import purpose_data FAKE_CODEX = r"""#!/usr/bin/env python3 import json import os import sys from pathlib import Path args = sys.argv[1:] model_index = args.index("--model") if args[model_index + 1] != "gpt-5.6-terra": raise SystemExit("wrong model") if 'model_reasoning_effort="low"' not in args: raise SystemExit("wrong reasoning effort") prompt = sys.stdin.read() payload_text = prompt.split("\n", 1)[1].split("\n", 1)[0] payload = json.loads(payload_text) items = [] for item in payload["items"]: if "MODEL_JUNK" in item["prompt"]: decision = { "id": item["id"], "keep": False, "junkReason": "generated agent output", "purpose": None, "secondary": None, "mixed": None, "difficulty": None, "slice": None, "lang": None, } elif "PLAN_AND_BUILD" in item["prompt"]: decision = { "id": item["id"], "keep": True, "junkReason": None, "purpose": "planning", "secondary": "backendImpl", "mixed": True, "difficulty": 0.7, "slice": "mixed", "lang": "en", } else: decision = { "id": item["id"], "keep": True, "junkReason": None, "purpose": "writing", "secondary": None, "mixed": False, "difficulty": 0.3, "slice": "core", "lang": "en", } items.append(decision) response_path = Path(args[args.index("--output-last-message") + 1]) response_path.write_text(json.dumps({"items": items}), encoding="utf-8") with Path(os.environ["FAKE_CODEX_LOG"]).open("a", encoding="utf-8") as handle: handle.write("called\n") print(json.dumps({"items": items})) """ class LabelNucleicPromptsTests(unittest.TestCase): def make_fake_codex(self, root: Path) -> Path: path = root / "codex" path.write_text(FAKE_CODEX, encoding="utf-8") path.chmod(0o755) return path def run_script( self, root: Path, input_path: Path, output_path: Path, fake_codex: Path, *extra: str, ) -> subprocess.CompletedProcess[str]: environment = os.environ.copy() environment["FAKE_CODEX_LOG"] = str(root / "calls.log") return subprocess.run( [ sys.executable, str(MODULE_DIR / "label_nucleic_prompts.py"), "--input", str(input_path), "--output", str(output_path), "--codex", str(fake_codex), "--batch-size", "2", *extra, ], text=True, stdout=subprocess.PIPE, stderr=subprocess.PIPE, env=environment, check=False, ) def test_filters_labels_and_writes_exact_training_contract(self): with tempfile.TemporaryDirectory() as directory: root = Path(directory) fake_codex = self.make_fake_codex(root) input_path = root / "unlabeled.jsonl" input_path.write_text( "\n".join( [ "", "not json", json.dumps( { "sessionID": "s1", "prompt": "Write the API documentation", "purpose": None, } ), json.dumps( { "sessionID": "s2", "prompt": " write THE api documentation ", "purpose": None, } ), json.dumps( {"sessionID": "s3", "prompt": "MODEL_JUNK transcript"} ), json.dumps( { "sessionID": "s4", "prompt": "PLAN_AND_BUILD the queue migration", } ), ] ) + "\n", encoding="utf-8", ) output_path = root / "labeled.jsonl" result = self.run_script( root, input_path, output_path, fake_codex, ) self.assertEqual(0, result.returncode, result.stderr) records = purpose_data.load_jsonl(output_path) self.assertEqual(2, len(records)) self.assertEqual( "Write the API documentation", records[0]["prompt"], ) self.assertEqual("writing", records[0]["purpose"]) self.assertEqual("planning", records[1]["purpose"]) self.assertEqual("backendImpl", records[1]["secondary"]) self.assertTrue(records[1]["mixed"]) for index, record in enumerate(records, start=1): purpose_data.validate_source_record(record, f"record {index}") self.assertEqual(purpose_data.SOURCE_FIELDS, set(record)) rejects = purpose_data.load_jsonl( label_nucleic_prompts.rejects_path_for(output_path) ) self.assertEqual(4, len(rejects)) reasons = {record["reason"] for record in rejects} self.assertIn("blank_line", reasons) self.assertIn("invalid_json", reasons) self.assertIn("duplicate_prompt_of_line_3", reasons) self.assertIn("semantic_junk:generated agent output", reasons) self.assertTrue( label_nucleic_prompts.state_path_for(output_path).is_file() ) def test_resume_reuses_completed_state_without_calling_codex(self): with tempfile.TemporaryDirectory() as directory: root = Path(directory) fake_codex = self.make_fake_codex(root) input_path = root / "unlabeled.jsonl" input_path.write_text( json.dumps({"prompt": "Write the migration note"}) + "\n", encoding="utf-8", ) output_path = root / "labeled.jsonl" first = self.run_script( root, input_path, output_path, fake_codex, ) second = self.run_script( root, input_path, output_path, fake_codex, "--resume", ) self.assertEqual(0, first.returncode, first.stderr) self.assertEqual(0, second.returncode, second.stderr) self.assertEqual( ["called"], (root / "calls.log").read_text(encoding="utf-8").splitlines(), ) self.assertIn("pending=0", second.stdout) def test_head_tail_excerpt_is_bounded(self): prompt = "HEAD" + ("x" * 4_000) + "TAIL" excerpt = label_nucleic_prompts.excerpt_for_labeling(prompt, 1_000) self.assertEqual(1_000, len(excerpt)) self.assertTrue(excerpt.startswith("HEAD")) self.assertTrue(excerpt.endswith("TAIL")) self.assertIn("characters omitted for labeling", excerpt) def test_retained_decision_must_obey_dataset_contract(self): source = label_nucleic_prompts.SourceLine(1, "{}", "hash") candidate = label_nucleic_prompts.Candidate( source=source, prompt="Plan and build it", prompt_hash="a" * 64, session_id=None, ) response = { "items": [ { "id": candidate.id, "keep": True, "junkReason": None, "purpose": "planning", "secondary": "backendImpl", "mixed": False, "difficulty": 0.7, "slice": "core", "lang": "en", } ] } with self.assertRaisesRegex( purpose_data.DataError, "mixed and secondary disagree", ): label_nucleic_prompts.validate_decisions([candidate], response) if __name__ == "__main__": unittest.main()