Files
nucleic-purpose-classifier/tests/test_label_swe_chat_prompts.py
T

105 lines
4.2 KiB
Python
Raw Normal View History

import json
import sys
import tempfile
import unittest
from pathlib import Path
MODULE_DIR = Path(__file__).resolve().parents[1]
sys.path.insert(0, str(MODULE_DIR))
import label_nucleic_prompts as base
import label_swe_chat_prompts
from purpose_data import prompt_hash
class LabelSWEChatPromptsTests(unittest.TestCase):
def candidate(self):
value = {
"schemaVersion": 1,
"repoID": "repo",
"userID": "user",
"sessionID": "session",
"sourceTurnIDs": ["one", "two"],
"prompt": "What is making this test fail?",
"teacherResponse": "It fails only on CI.",
}
value["promptHash"] = prompt_hash(value["prompt"])
value["teacherResponseHash"] = prompt_hash(value["teacherResponse"])
line = base.SourceLine(1, json.dumps(value), "line-hash")
return label_swe_chat_prompts.candidates([line])[0]
def decision(self, candidate, recoverable):
return {
"items": [{
"id": candidate.id, "keep": True, "junkReason": None,
"purpose": "debugging", "secondary": None, "mixed": False,
"difficulty": 0.6, "slice": "boundary", "lang": "en",
"recoverableFromFirst": recoverable,
}]
}
def test_response_hash_is_checked_and_context_dependent_labels_become_vague_eval(self):
candidate = self.candidate()
decisions = label_swe_chat_prompts.validate_decisions(
[candidate], self.decision(candidate, False)
)
self.assertEqual("vague-eval", decisions[0][1]["slice"])
self.assertNotIn("It fails only on CI.", candidate.prompt)
def test_candidate_rejects_context_hash_mismatch(self):
candidate = self.candidate()
value = json.loads(candidate.line.raw)
value["teacherResponseHash"] = "bad"
with self.assertRaisesRegex(ValueError, "teacherResponseHash"):
label_swe_chat_prompts.candidates(
[base.SourceLine(1, json.dumps(value), "line-hash")]
)
def test_state_never_carries_later_text(self):
candidate = self.candidate()
state = label_swe_chat_prompts._state(
candidate, status="labeled", record=None, reason=None, recoverable=True
)
encoded = json.dumps(state)
self.assertNotIn("It fails only on CI.", encoded)
def test_response_sanitizer_preserves_all_prose_and_removes_code_payloads(self):
response = (
"I found the likely cause.\n\n"
"```swift\nlet secret = \"not teacher context\"\n```\n\n"
"The fix is to await the task.\n"
"<tool_use>{\"cmd\": \"rm -rf /\"}</tool_use>\n"
"Then rerun the focused test."
)
sanitized = label_swe_chat_prompts.response_for_labeling(response, 2_000)
self.assertIn("I found the likely cause.", sanitized)
self.assertIn("The fix is to await the task.", sanitized)
self.assertIn("Then rerun the focused test.", sanitized)
self.assertNotIn("let secret", sanitized)
self.assertNotIn("rm -rf", sanitized)
def test_response_sanitizer_refuses_to_truncate_prose(self):
with self.assertRaisesRegex(ValueError, "rather than truncating"):
label_swe_chat_prompts.response_for_labeling("x" * 1_001, 1_000)
def test_record_uses_only_the_first_user_message_as_student_text(self):
candidate = self.candidate()
decision = self.decision(candidate, True)["items"][0]
record = label_swe_chat_prompts.record_from_decision(candidate, decision)
self.assertEqual(candidate.prompt, record["prompt"])
self.assertNotIn(candidate.response, record.values())
def test_unrecoverable_mixed_decision_becomes_single_purpose_vague_eval(self):
candidate = self.candidate()
decision = self.decision(candidate, False)["items"][0]
decision.update({"secondary": "frontendImpl", "mixed": True, "slice": "mixed"})
record = label_swe_chat_prompts.record_from_decision(candidate, decision)
self.assertEqual("vague-eval", record["slice"])
self.assertFalse(record["mixed"])
self.assertIsNone(record["secondary"])
if __name__ == "__main__":
unittest.main()