74 lines
3.2 KiB
Python
74 lines
3.2 KiB
Python
import sys
|
|||
|
|
import unittest
|
||
|
|
from pathlib import Path
|
||
|
|
|
||
|
|
|
||
|
|
MODULE_DIR = Path(__file__).resolve().parents[1]
|
||
|
|
sys.path.insert(0, str(MODULE_DIR))
|
||
|
|
|
||
|
|
import export_swe_chat
|
||
|
|
|
||
|
|
|
||
|
|
def row(session, turn, prompt, **overrides):
|
||
|
|
value = {
|
||
|
|
"session_id": session,
|
||
|
|
"turn_id": turn,
|
||
|
|
"conversation_turn_number": int(turn[1:]),
|
||
|
|
"turn_number": int(turn[1:]),
|
||
|
|
"turn_type": "user_prompt",
|
||
|
|
"role": "user",
|
||
|
|
"is_conversational": True,
|
||
|
|
"is_continuation": False,
|
||
|
|
"content": prompt,
|
||
|
|
}
|
||
|
|
value.update(overrides)
|
||
|
|
return value
|
||
|
|
|
||
|
|
|
||
|
|
class ExportSWEChatTests(unittest.TestCase):
|
||
|
|
def test_selects_first_three_orders_dedupes_and_caps_sources(self):
|
||
|
|
rows = [
|
||
|
|
row("s1", "t3", "third", conversation_turn_number=3),
|
||
|
|
row("s1", "t1", "first", conversation_turn_number=1),
|
||
|
|
row("s1", "t2", "second", conversation_turn_number=2),
|
||
|
|
row("s2", "t1", " first ", conversation_turn_number=1),
|
||
|
|
row("s2", "t2", "later", conversation_turn_number=2),
|
||
|
|
row("s2", "t3", "later again", conversation_turn_number=3),
|
||
|
|
row("s3", "t1", "one"), row("s3", "t2", "two"),
|
||
|
|
row("s4", "t1", "another one"), row("s4", "t2", "another two"), row("s4", "t3", "another three"),
|
||
|
|
]
|
||
|
|
candidates, funnel = export_swe_chat.select_candidates(
|
||
|
|
rows,
|
||
|
|
sessions={"s1": ("repo", "user"), "s2": ("repo", "user"), "s4": ("repo", "user")},
|
||
|
|
max_per_repo=1, max_per_user=1,
|
||
|
|
)
|
||
|
|
self.assertEqual(1, len(candidates))
|
||
|
|
self.assertEqual(["t1", "t2", "t3"], [turn.turn_id for turn in candidates[0].turns])
|
||
|
|
self.assertEqual("first", candidates[0].turns[0].prompt)
|
||
|
|
self.assertEqual(1, funnel["rejectedDuplicateFirstPrompt"])
|
||
|
|
self.assertEqual(1, funnel["sessionsFewerThanThreeEligiblePrompts"])
|
||
|
|
self.assertEqual(1, funnel["rejectedRepoCap"])
|
||
|
|
|
||
|
|
def test_filters_non_user_continuation_empty_and_ambiguous_ordinals(self):
|
||
|
|
rows = [
|
||
|
|
row("s1", "t1", "x", role="assistant"), row("s1", "t2", "x", is_continuation=True), row("s1", "t3", ""),
|
||
|
|
row("s2", "t1", "a", conversation_turn_number=1, turn_number=1), row("s2", "t2", "b", conversation_turn_number=1, turn_number=1), row("s2", "t3", "c", conversation_turn_number=3),
|
||
|
|
]
|
||
|
|
candidates, funnel = export_swe_chat.select_candidates(rows, sessions={}, max_per_repo=10, max_per_user=10)
|
||
|
|
self.assertEqual([], candidates)
|
||
|
|
self.assertEqual(1, funnel["rejectedRole"])
|
||
|
|
self.assertEqual(1, funnel["rejectedContinuation"])
|
||
|
|
self.assertEqual(1, funnel["rejectedMalformedOrEmpty"])
|
||
|
|
self.assertEqual(1, funnel["sessionsAmbiguousTurnOrder"])
|
||
|
|
|
||
|
|
def test_candidate_only_retains_first_prompt_as_student_text(self):
|
||
|
|
turns = tuple(export_swe_chat.Turn("s", f"t{index}", index, index, prompt) for index, prompt in enumerate(("first", "second", "third"), start=1))
|
||
|
|
value = export_swe_chat.Candidate("s", "r", "u", turns).json("a" * 40)
|
||
|
|
self.assertEqual("first", value["prompt"])
|
||
|
|
self.assertEqual(["second", "third"], value["teacherContext"])
|
||
|
|
self.assertNotIn("second", value["contextPromptHashes"])
|
||
|
|
|
||
|
|
|
||
|
|
if __name__ == "__main__":
|
||
|
|
unittest.main()
|