2026-07-30 23:01:28 -07:00
|
|
|
import json
|
|
|
|
|
import os
|
|
|
|
|
import subprocess
|
|
|
|
|
import sys
|
|
|
|
|
import tempfile
|
|
|
|
|
import unittest
|
|
|
|
|
from pathlib import Path
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
MODULE_DIR = Path(__file__).resolve().parents[1]
|
|
|
|
|
sys.path.insert(0, str(MODULE_DIR))
|
|
|
|
|
|
|
|
|
|
import label_nucleic_prompts
|
|
|
|
|
import purpose_data
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
FAKE_CODEX = r"""#!/usr/bin/env python3
|
|
|
|
|
import json
|
|
|
|
|
import os
|
|
|
|
|
import sys
|
|
|
|
|
from pathlib import Path
|
|
|
|
|
|
|
|
|
|
args = sys.argv[1:]
|
|
|
|
|
model_index = args.index("--model")
|
2026-08-01 21:26:31 -07:00
|
|
|
if args[model_index + 1] != os.environ.get("EXPECTED_MODEL", "gpt-5.6-terra"):
|
2026-07-30 23:01:28 -07:00
|
|
|
raise SystemExit("wrong model")
|
2026-08-01 21:26:31 -07:00
|
|
|
expected_effort = os.environ.get("EXPECTED_REASONING_EFFORT", "low")
|
|
|
|
|
if f'model_reasoning_effort="{expected_effort}"' not in args:
|
2026-07-30 23:01:28 -07:00
|
|
|
raise SystemExit("wrong reasoning effort")
|
|
|
|
|
|
|
|
|
|
prompt = sys.stdin.read()
|
|
|
|
|
payload_text = prompt.split("<input_json>\n", 1)[1].split("\n</input_json>", 1)[0]
|
|
|
|
|
payload = json.loads(payload_text)
|
|
|
|
|
items = []
|
|
|
|
|
for item in payload["items"]:
|
|
|
|
|
if "MODEL_JUNK" in item["prompt"]:
|
|
|
|
|
decision = {
|
|
|
|
|
"id": item["id"],
|
|
|
|
|
"keep": False,
|
|
|
|
|
"junkReason": "generated agent output",
|
|
|
|
|
"purpose": None,
|
|
|
|
|
"secondary": None,
|
|
|
|
|
"mixed": None,
|
|
|
|
|
"difficulty": None,
|
|
|
|
|
"slice": None,
|
|
|
|
|
"lang": None,
|
|
|
|
|
}
|
|
|
|
|
elif "PLAN_AND_BUILD" in item["prompt"]:
|
|
|
|
|
decision = {
|
|
|
|
|
"id": item["id"],
|
|
|
|
|
"keep": True,
|
|
|
|
|
"junkReason": None,
|
|
|
|
|
"purpose": "planning",
|
|
|
|
|
"secondary": "backendImpl",
|
|
|
|
|
"mixed": True,
|
|
|
|
|
"difficulty": 0.7,
|
|
|
|
|
"slice": "mixed",
|
|
|
|
|
"lang": "en",
|
|
|
|
|
}
|
|
|
|
|
else:
|
|
|
|
|
decision = {
|
|
|
|
|
"id": item["id"],
|
|
|
|
|
"keep": True,
|
|
|
|
|
"junkReason": None,
|
|
|
|
|
"purpose": "writing",
|
|
|
|
|
"secondary": None,
|
|
|
|
|
"mixed": False,
|
|
|
|
|
"difficulty": 0.3,
|
|
|
|
|
"slice": "core",
|
|
|
|
|
"lang": "en",
|
|
|
|
|
}
|
|
|
|
|
items.append(decision)
|
|
|
|
|
|
|
|
|
|
response_path = Path(args[args.index("--output-last-message") + 1])
|
|
|
|
|
response_path.write_text(json.dumps({"items": items}), encoding="utf-8")
|
|
|
|
|
with Path(os.environ["FAKE_CODEX_LOG"]).open("a", encoding="utf-8") as handle:
|
|
|
|
|
handle.write("called\n")
|
|
|
|
|
print(json.dumps({"items": items}))
|
|
|
|
|
"""
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
class LabelNucleicPromptsTests(unittest.TestCase):
|
|
|
|
|
def make_fake_codex(self, root: Path) -> Path:
|
|
|
|
|
path = root / "codex"
|
|
|
|
|
path.write_text(FAKE_CODEX, encoding="utf-8")
|
|
|
|
|
path.chmod(0o755)
|
|
|
|
|
return path
|
|
|
|
|
|
|
|
|
|
def run_script(
|
|
|
|
|
self,
|
|
|
|
|
root: Path,
|
|
|
|
|
input_path: Path,
|
|
|
|
|
output_path: Path,
|
|
|
|
|
fake_codex: Path,
|
|
|
|
|
*extra: str,
|
|
|
|
|
) -> subprocess.CompletedProcess[str]:
|
|
|
|
|
environment = os.environ.copy()
|
|
|
|
|
environment["FAKE_CODEX_LOG"] = str(root / "calls.log")
|
2026-08-01 21:26:31 -07:00
|
|
|
if "--model" in extra:
|
|
|
|
|
environment["EXPECTED_MODEL"] = extra[extra.index("--model") + 1]
|
|
|
|
|
if "--reasoning-effort" in extra:
|
|
|
|
|
environment["EXPECTED_REASONING_EFFORT"] = extra[
|
|
|
|
|
extra.index("--reasoning-effort") + 1
|
|
|
|
|
]
|
2026-07-30 23:01:28 -07:00
|
|
|
return subprocess.run(
|
|
|
|
|
[
|
|
|
|
|
sys.executable,
|
|
|
|
|
str(MODULE_DIR / "label_nucleic_prompts.py"),
|
|
|
|
|
"--input",
|
|
|
|
|
str(input_path),
|
|
|
|
|
"--output",
|
|
|
|
|
str(output_path),
|
|
|
|
|
"--codex",
|
|
|
|
|
str(fake_codex),
|
|
|
|
|
"--batch-size",
|
|
|
|
|
"2",
|
|
|
|
|
*extra,
|
|
|
|
|
],
|
|
|
|
|
text=True,
|
|
|
|
|
stdout=subprocess.PIPE,
|
|
|
|
|
stderr=subprocess.PIPE,
|
|
|
|
|
env=environment,
|
|
|
|
|
check=False,
|
|
|
|
|
)
|
|
|
|
|
|
|
|
|
|
def test_filters_labels_and_writes_exact_training_contract(self):
|
|
|
|
|
with tempfile.TemporaryDirectory() as directory:
|
|
|
|
|
root = Path(directory)
|
|
|
|
|
fake_codex = self.make_fake_codex(root)
|
|
|
|
|
input_path = root / "unlabeled.jsonl"
|
|
|
|
|
input_path.write_text(
|
|
|
|
|
"\n".join(
|
|
|
|
|
[
|
|
|
|
|
"",
|
|
|
|
|
"not json",
|
|
|
|
|
json.dumps(
|
|
|
|
|
{
|
|
|
|
|
"sessionID": "s1",
|
|
|
|
|
"prompt": "Write the API documentation",
|
|
|
|
|
"purpose": None,
|
|
|
|
|
}
|
|
|
|
|
),
|
|
|
|
|
json.dumps(
|
|
|
|
|
{
|
|
|
|
|
"sessionID": "s2",
|
|
|
|
|
"prompt": " write THE api documentation ",
|
|
|
|
|
"purpose": None,
|
|
|
|
|
}
|
|
|
|
|
),
|
|
|
|
|
json.dumps(
|
|
|
|
|
{"sessionID": "s3", "prompt": "MODEL_JUNK transcript"}
|
|
|
|
|
),
|
|
|
|
|
json.dumps(
|
|
|
|
|
{
|
|
|
|
|
"sessionID": "s4",
|
|
|
|
|
"prompt": "PLAN_AND_BUILD the queue migration",
|
|
|
|
|
}
|
|
|
|
|
),
|
|
|
|
|
]
|
|
|
|
|
)
|
|
|
|
|
+ "\n",
|
|
|
|
|
encoding="utf-8",
|
|
|
|
|
)
|
|
|
|
|
output_path = root / "labeled.jsonl"
|
|
|
|
|
|
|
|
|
|
result = self.run_script(
|
|
|
|
|
root,
|
|
|
|
|
input_path,
|
|
|
|
|
output_path,
|
|
|
|
|
fake_codex,
|
|
|
|
|
)
|
|
|
|
|
|
|
|
|
|
self.assertEqual(0, result.returncode, result.stderr)
|
|
|
|
|
records = purpose_data.load_jsonl(output_path)
|
|
|
|
|
self.assertEqual(2, len(records))
|
|
|
|
|
self.assertEqual(
|
|
|
|
|
"Write the API documentation",
|
|
|
|
|
records[0]["prompt"],
|
|
|
|
|
)
|
|
|
|
|
self.assertEqual("writing", records[0]["purpose"])
|
|
|
|
|
self.assertEqual("planning", records[1]["purpose"])
|
|
|
|
|
self.assertEqual("backendImpl", records[1]["secondary"])
|
|
|
|
|
self.assertTrue(records[1]["mixed"])
|
|
|
|
|
for index, record in enumerate(records, start=1):
|
|
|
|
|
purpose_data.validate_source_record(record, f"record {index}")
|
|
|
|
|
self.assertEqual(purpose_data.SOURCE_FIELDS, set(record))
|
|
|
|
|
|
|
|
|
|
rejects = purpose_data.load_jsonl(
|
|
|
|
|
label_nucleic_prompts.rejects_path_for(output_path)
|
|
|
|
|
)
|
|
|
|
|
self.assertEqual(4, len(rejects))
|
|
|
|
|
reasons = {record["reason"] for record in rejects}
|
|
|
|
|
self.assertIn("blank_line", reasons)
|
|
|
|
|
self.assertIn("invalid_json", reasons)
|
|
|
|
|
self.assertIn("duplicate_prompt_of_line_3", reasons)
|
|
|
|
|
self.assertIn("semantic_junk:generated agent output", reasons)
|
|
|
|
|
self.assertTrue(
|
|
|
|
|
label_nucleic_prompts.state_path_for(output_path).is_file()
|
|
|
|
|
)
|
|
|
|
|
|
|
|
|
|
def test_resume_reuses_completed_state_without_calling_codex(self):
|
|
|
|
|
with tempfile.TemporaryDirectory() as directory:
|
|
|
|
|
root = Path(directory)
|
|
|
|
|
fake_codex = self.make_fake_codex(root)
|
|
|
|
|
input_path = root / "unlabeled.jsonl"
|
|
|
|
|
input_path.write_text(
|
|
|
|
|
json.dumps({"prompt": "Write the migration note"}) + "\n",
|
|
|
|
|
encoding="utf-8",
|
|
|
|
|
)
|
|
|
|
|
output_path = root / "labeled.jsonl"
|
|
|
|
|
|
|
|
|
|
first = self.run_script(
|
|
|
|
|
root,
|
|
|
|
|
input_path,
|
|
|
|
|
output_path,
|
|
|
|
|
fake_codex,
|
|
|
|
|
)
|
|
|
|
|
second = self.run_script(
|
|
|
|
|
root,
|
|
|
|
|
input_path,
|
|
|
|
|
output_path,
|
|
|
|
|
fake_codex,
|
|
|
|
|
"--resume",
|
|
|
|
|
)
|
|
|
|
|
|
|
|
|
|
self.assertEqual(0, first.returncode, first.stderr)
|
|
|
|
|
self.assertEqual(0, second.returncode, second.stderr)
|
|
|
|
|
self.assertEqual(
|
|
|
|
|
["called"],
|
|
|
|
|
(root / "calls.log").read_text(encoding="utf-8").splitlines(),
|
|
|
|
|
)
|
|
|
|
|
self.assertIn("pending=0", second.stdout)
|
|
|
|
|
|
2026-08-01 21:26:31 -07:00
|
|
|
def test_custom_model_and_reasoning_effort_are_forwarded(self):
|
|
|
|
|
with tempfile.TemporaryDirectory() as directory:
|
|
|
|
|
root = Path(directory)
|
|
|
|
|
fake_codex = self.make_fake_codex(root)
|
|
|
|
|
input_path = root / "unlabeled.jsonl"
|
|
|
|
|
input_path.write_text(
|
|
|
|
|
json.dumps({"prompt": "Write the migration note"}) + "\n",
|
|
|
|
|
encoding="utf-8",
|
|
|
|
|
)
|
|
|
|
|
|
|
|
|
|
result = self.run_script(
|
|
|
|
|
root,
|
|
|
|
|
input_path,
|
|
|
|
|
root / "labeled.jsonl",
|
|
|
|
|
fake_codex,
|
|
|
|
|
"--model",
|
|
|
|
|
"gpt-5.6-sol",
|
|
|
|
|
"--reasoning-effort",
|
|
|
|
|
"high",
|
|
|
|
|
)
|
|
|
|
|
|
|
|
|
|
self.assertEqual(0, result.returncode, result.stderr)
|
|
|
|
|
self.assertIn("model=gpt-5.6-sol reasoning=high", result.stdout)
|
|
|
|
|
|
2026-07-30 23:01:28 -07:00
|
|
|
def test_head_tail_excerpt_is_bounded(self):
|
|
|
|
|
prompt = "HEAD" + ("x" * 4_000) + "TAIL"
|
|
|
|
|
excerpt = label_nucleic_prompts.excerpt_for_labeling(prompt, 1_000)
|
|
|
|
|
|
|
|
|
|
self.assertEqual(1_000, len(excerpt))
|
|
|
|
|
self.assertTrue(excerpt.startswith("HEAD"))
|
|
|
|
|
self.assertTrue(excerpt.endswith("TAIL"))
|
|
|
|
|
self.assertIn("characters omitted for labeling", excerpt)
|
|
|
|
|
|
|
|
|
|
def test_retained_decision_must_obey_dataset_contract(self):
|
|
|
|
|
source = label_nucleic_prompts.SourceLine(1, "{}", "hash")
|
|
|
|
|
candidate = label_nucleic_prompts.Candidate(
|
|
|
|
|
source=source,
|
|
|
|
|
prompt="Plan and build it",
|
|
|
|
|
prompt_hash="a" * 64,
|
|
|
|
|
session_id=None,
|
|
|
|
|
)
|
|
|
|
|
response = {
|
|
|
|
|
"items": [
|
|
|
|
|
{
|
|
|
|
|
"id": candidate.id,
|
|
|
|
|
"keep": True,
|
|
|
|
|
"junkReason": None,
|
|
|
|
|
"purpose": "planning",
|
|
|
|
|
"secondary": "backendImpl",
|
|
|
|
|
"mixed": False,
|
|
|
|
|
"difficulty": 0.7,
|
|
|
|
|
"slice": "core",
|
|
|
|
|
"lang": "en",
|
|
|
|
|
}
|
|
|
|
|
]
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
with self.assertRaisesRegex(
|
|
|
|
|
purpose_data.DataError,
|
|
|
|
|
"mixed and secondary disagree",
|
|
|
|
|
):
|
|
|
|
|
label_nucleic_prompts.validate_decisions([candidate], response)
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
if __name__ == "__main__":
|
|
|
|
|
unittest.main()
|