From 8ac00de0879d557ef91d5ef004aee6cd09f1f393 Mon Sep 17 00:00:00 2001 From: Nucleic Date: Sat, 1 Aug 2026 16:26:15 -0700 Subject: [PATCH] Merge nucleic/plucky-north-vole-sdna into dev --- label_swe_chat_prompts.py | 8 +++++++- tests/test_label_swe_chat_prompts.py | 9 +++++++++ 2 files changed, 16 insertions(+), 1 deletion(-) diff --git a/label_swe_chat_prompts.py b/label_swe_chat_prompts.py index 4934d70..5f60d48 100644 --- a/label_swe_chat_prompts.py +++ b/label_swe_chat_prompts.py @@ -311,6 +311,10 @@ def record_from_decision(candidate: Candidate, decision: dict[str, Any]) -> dict record = {field: decision[field] for field in SOURCE_FIELDS - {"prompt"}} record["prompt"] = candidate.prompt if not decision["recoverableFromFirst"]: + # `vague-eval` records model first-message uncertainty, so they cannot also + # claim a context-derived second deliverable. Preserve the primary label only. + record["secondary"] = None + record["mixed"] = False record["slice"] = "vague-eval" validate_source_record(record, candidate.id) return record @@ -334,7 +338,9 @@ def run(args: argparse.Namespace) -> dict[str, int]: states = _load_state(args.state, by_line) if args.resume else {} pending = [candidate for candidate in source if candidate.line.number not in states] if args.limit_sessions is not None: - pending = pending[:args.limit_sessions] + # A resumed bounded canary/dry run retains its original total limit rather + # than processing another full limit beyond already persisted decisions. + pending = pending[:max(0, args.limit_sessions - len(states))] batch_list = batches( pending, batch_size=args.batch_size, batch_chars=args.batch_chars, max_prompt_chars=args.max_prompt_chars, max_response_chars=args.max_response_chars, diff --git a/tests/test_label_swe_chat_prompts.py b/tests/test_label_swe_chat_prompts.py index 8639dbe..1adf42d 100644 --- a/tests/test_label_swe_chat_prompts.py +++ b/tests/test_label_swe_chat_prompts.py @@ -90,6 +90,15 @@ class LabelSWEChatPromptsTests(unittest.TestCase): self.assertEqual(candidate.prompt, record["prompt"]) self.assertNotIn(candidate.response, record.values()) + def test_unrecoverable_mixed_decision_becomes_single_purpose_vague_eval(self): + candidate = self.candidate() + decision = self.decision(candidate, False)["items"][0] + decision.update({"secondary": "frontendImpl", "mixed": True, "slice": "mixed"}) + record = label_swe_chat_prompts.record_from_decision(candidate, decision) + self.assertEqual("vague-eval", record["slice"]) + self.assertFalse(record["mixed"]) + self.assertIsNone(record["secondary"]) + if __name__ == "__main__": unittest.main()