Merge nucleic/sleek-ember-seal-uady into dev

This commit is contained in:
2026-07-30 21:47:07 -07:00
parent a374fa40b6
commit 82545b833b
3 changed files with 16 additions and 5 deletions
+4 -1
View File
@@ -239,7 +239,10 @@ and removes each package immediately after prediction; this avoids Core ML Tools
one full weight copy per calibration step until process exit. It prints progress while it one full weight copy per calibration step until process exit. It prints progress while it
runs. The candidate uses per-tensor asymmetric uint8 activations, runs. The candidate uses per-tensor asymmetric uint8 activations,
per-channel symmetric int8 linear weights, and per-tensor asymmetric uint8 embedding per-channel symmetric int8 linear weights, and per-tensor asymmetric uint8 embedding
weights. It fails the command if the resulting package exceeds 25 MiB. weights. Activation quantization is limited to floating-point linear operations; applying
Core ML Tools' global policy also selects integer embedding-index additions and produces
an invalid quantize operation. It fails the command if the resulting package exceeds
25 MiB.
Run the frozen gate with CPU+Neural Engine placement and compare labels directly with the Run the frozen gate with CPU+Neural Engine placement and compare labels directly with the
accepted int8 ONNX artifact. Gated Core ML evaluation fails closed without accepted int8 ONNX artifact. Gated Core ML evaluation fails closed without
+6 -1
View File
@@ -86,7 +86,12 @@ def _optimization_configs(optimize: Any) -> tuple[Any, Any]:
granularity="per_tensor", granularity="per_tensor",
) )
activation_config = optimize.coreml.OptimizationConfig( activation_config = optimize.coreml.OptimizationConfig(
global_config=activation, # A global activation policy also selects integer `add` operations in the
# embedding/index path. Core ML's quantize op requires its floating-point scale
# to match a floating-point input, so quantize only accelerator-supported linear
# activations. Attention matmul activation quantization is not supported by this
# Core ML graph pass.
op_type_configs={"linear": activation},
) )
linear_weight = optimize.coreml.OpLinearQuantizerConfig( linear_weight = optimize.coreml.OpLinearQuantizerConfig(
+6 -3
View File
@@ -37,12 +37,15 @@ class FakeOptimize:
class CoreMLQuantizationConfigTests(unittest.TestCase): class CoreMLQuantizationConfigTests(unittest.TestCase):
def test_matches_accepted_qdq_policy(self): def test_matches_accepted_qdq_policy(self):
activation, weights = quantize_coreml._optimization_configs(FakeOptimize) activation, weights = quantize_coreml._optimization_configs(FakeOptimize)
self.assertEqual("linear", activation.global_config.values["mode"]) self.assertIsNone(activation.global_config)
self.assertIs(np.uint8, activation.global_config.values["dtype"]) activation_linear = activation.op_type_configs["linear"].values
self.assertEqual("linear", activation_linear["mode"])
self.assertIs(np.uint8, activation_linear["dtype"])
self.assertEqual( self.assertEqual(
"per_tensor", "per_tensor",
activation.global_config.values["granularity"], activation_linear["granularity"],
) )
self.assertEqual({"linear"}, set(activation.op_type_configs))
linear = weights.op_type_configs["linear"].values linear = weights.op_type_configs["linear"].values
self.assertEqual("linear_symmetric", linear["mode"]) self.assertEqual("linear_symmetric", linear["mode"])