diff --git a/README.md b/README.md index 33b3717..a8a8e14 100644 --- a/README.md +++ b/README.md @@ -239,7 +239,10 @@ and removes each package immediately after prediction; this avoids Core ML Tools one full weight copy per calibration step until process exit. It prints progress while it runs. The candidate uses per-tensor asymmetric uint8 activations, per-channel symmetric int8 linear weights, and per-tensor asymmetric uint8 embedding -weights. It fails the command if the resulting package exceeds 25 MiB. +weights. Activation quantization is limited to floating-point linear operations; applying +Core ML Tools' global policy also selects integer embedding-index additions and produces +an invalid quantize operation. It fails the command if the resulting package exceeds +25 MiB. Run the frozen gate with CPU+Neural Engine placement and compare labels directly with the accepted int8 ONNX artifact. Gated Core ML evaluation fails closed without diff --git a/quantize_coreml.py b/quantize_coreml.py index 628939c..0505b2a 100644 --- a/quantize_coreml.py +++ b/quantize_coreml.py @@ -86,7 +86,12 @@ def _optimization_configs(optimize: Any) -> tuple[Any, Any]: granularity="per_tensor", ) activation_config = optimize.coreml.OptimizationConfig( - global_config=activation, + # A global activation policy also selects integer `add` operations in the + # embedding/index path. Core ML's quantize op requires its floating-point scale + # to match a floating-point input, so quantize only accelerator-supported linear + # activations. Attention matmul activation quantization is not supported by this + # Core ML graph pass. + op_type_configs={"linear": activation}, ) linear_weight = optimize.coreml.OpLinearQuantizerConfig( diff --git a/tests/test_quantize_coreml.py b/tests/test_quantize_coreml.py index d01ebaa..8d4ed1f 100644 --- a/tests/test_quantize_coreml.py +++ b/tests/test_quantize_coreml.py @@ -37,12 +37,15 @@ class FakeOptimize: class CoreMLQuantizationConfigTests(unittest.TestCase): def test_matches_accepted_qdq_policy(self): activation, weights = quantize_coreml._optimization_configs(FakeOptimize) - self.assertEqual("linear", activation.global_config.values["mode"]) - self.assertIs(np.uint8, activation.global_config.values["dtype"]) + self.assertIsNone(activation.global_config) + activation_linear = activation.op_type_configs["linear"].values + self.assertEqual("linear", activation_linear["mode"]) + self.assertIs(np.uint8, activation_linear["dtype"]) self.assertEqual( "per_tensor", - activation.global_config.values["granularity"], + activation_linear["granularity"], ) + self.assertEqual({"linear"}, set(activation.op_type_configs)) linear = weights.op_type_configs["linear"].values self.assertEqual("linear_symmetric", linear["mode"])