2026-07-30 05:04:34 -07:00
|
|
|
import sys
|
|
|
|
|
import unittest
|
|
|
|
|
from pathlib import Path
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
MODULE_DIR = Path(__file__).resolve().parents[1]
|
|
|
|
|
sys.path.insert(0, str(MODULE_DIR))
|
|
|
|
|
|
|
|
|
|
import eval as purpose_eval
|
|
|
|
|
|
|
|
|
|
|
2026-07-30 20:56:46 -07:00
|
|
|
class CoreMLComputeUnitTests(unittest.TestCase):
|
|
|
|
|
class CoreMLTools:
|
|
|
|
|
class ComputeUnit:
|
|
|
|
|
ALL = "all-value"
|
|
|
|
|
CPU_ONLY = "cpu-value"
|
|
|
|
|
CPU_AND_GPU = "gpu-value"
|
|
|
|
|
CPU_AND_NE = "ne-value"
|
|
|
|
|
|
|
|
|
|
def test_maps_cli_compute_policies(self):
|
|
|
|
|
expected = {
|
|
|
|
|
"all": "all-value",
|
|
|
|
|
"cpu-only": "cpu-value",
|
|
|
|
|
"cpu-and-gpu": "gpu-value",
|
|
|
|
|
"cpu-and-ne": "ne-value",
|
|
|
|
|
}
|
|
|
|
|
for requested, value in expected.items():
|
|
|
|
|
with self.subTest(requested=requested):
|
|
|
|
|
self.assertEqual(
|
|
|
|
|
value,
|
|
|
|
|
purpose_eval._coreml_compute_unit(self.CoreMLTools, requested),
|
|
|
|
|
)
|
|
|
|
|
|
|
|
|
|
|
2026-07-30 05:04:34 -07:00
|
|
|
class TierDriftTests(unittest.TestCase):
|
|
|
|
|
def test_current_routing_matrix_bounds_every_label_pair(self):
|
|
|
|
|
records = [{"prompt": f"prompt {index}"} for index in range(8 * 8)]
|
|
|
|
|
actual = []
|
|
|
|
|
predicted = []
|
|
|
|
|
for expected in range(8):
|
|
|
|
|
for got in range(8):
|
|
|
|
|
actual.append(expected)
|
|
|
|
|
predicted.append(got)
|
|
|
|
|
report = purpose_eval.routing_tier_drift(records, actual, predicted)
|
|
|
|
|
self.assertTrue(report["passed"])
|
|
|
|
|
self.assertLessEqual(report["maximumTierDrift"], 1)
|
|
|
|
|
|
|
|
|
|
|
2026-07-30 17:28:44 -07:00
|
|
|
class PredictionAgreementTests(unittest.TestCase):
|
|
|
|
|
def test_reports_accuracy_transitions_and_scorable_agreement(self):
|
|
|
|
|
records = [
|
|
|
|
|
{"prompt": "one", "slice": "core"},
|
|
|
|
|
{"prompt": "two", "slice": "boundary"},
|
|
|
|
|
{"prompt": "three", "slice": "vague-eval"},
|
|
|
|
|
{"prompt": "four", "slice": "core"},
|
|
|
|
|
]
|
|
|
|
|
report = purpose_eval.prediction_agreement(
|
|
|
|
|
records,
|
|
|
|
|
actual=[0, 1, 2, 3],
|
|
|
|
|
reference=[0, 0, 3, 4],
|
|
|
|
|
candidate=[1, 1, 4, 5],
|
|
|
|
|
)
|
|
|
|
|
self.assertEqual(0.0, report["labelAgreement"])
|
|
|
|
|
self.assertEqual(0.0, report["scoredLabelAgreement"])
|
|
|
|
|
self.assertEqual(
|
|
|
|
|
{
|
|
|
|
|
"correctToIncorrect": 1,
|
|
|
|
|
"differentIncorrectLabel": 2,
|
|
|
|
|
"incorrectToCorrect": 1,
|
|
|
|
|
},
|
|
|
|
|
report["transitionCounts"],
|
|
|
|
|
)
|
|
|
|
|
self.assertEqual(2, report["bySlice"]["core"]["records"])
|
|
|
|
|
self.assertEqual(4, len(report["disagreements"]))
|
|
|
|
|
|
|
|
|
|
|
2026-07-30 05:04:34 -07:00
|
|
|
if __name__ == "__main__":
|
|
|
|
|
unittest.main()
|