Merge nucleic/sleek-ember-seal-uady into dev
This commit is contained in:
@@ -239,7 +239,10 @@ and removes each package immediately after prediction; this avoids Core ML Tools
|
|||||||
one full weight copy per calibration step until process exit. It prints progress while it
|
one full weight copy per calibration step until process exit. It prints progress while it
|
||||||
runs. The candidate uses per-tensor asymmetric uint8 activations,
|
runs. The candidate uses per-tensor asymmetric uint8 activations,
|
||||||
per-channel symmetric int8 linear weights, and per-tensor asymmetric uint8 embedding
|
per-channel symmetric int8 linear weights, and per-tensor asymmetric uint8 embedding
|
||||||
weights. It fails the command if the resulting package exceeds 25 MiB.
|
weights. Activation quantization is limited to floating-point linear operations; applying
|
||||||
|
Core ML Tools' global policy also selects integer embedding-index additions and produces
|
||||||
|
an invalid quantize operation. It fails the command if the resulting package exceeds
|
||||||
|
25 MiB.
|
||||||
|
|
||||||
Run the frozen gate with CPU+Neural Engine placement and compare labels directly with the
|
Run the frozen gate with CPU+Neural Engine placement and compare labels directly with the
|
||||||
accepted int8 ONNX artifact. Gated Core ML evaluation fails closed without
|
accepted int8 ONNX artifact. Gated Core ML evaluation fails closed without
|
||||||
|
|||||||
+6
-1
@@ -86,7 +86,12 @@ def _optimization_configs(optimize: Any) -> tuple[Any, Any]:
|
|||||||
granularity="per_tensor",
|
granularity="per_tensor",
|
||||||
)
|
)
|
||||||
activation_config = optimize.coreml.OptimizationConfig(
|
activation_config = optimize.coreml.OptimizationConfig(
|
||||||
global_config=activation,
|
# A global activation policy also selects integer `add` operations in the
|
||||||
|
# embedding/index path. Core ML's quantize op requires its floating-point scale
|
||||||
|
# to match a floating-point input, so quantize only accelerator-supported linear
|
||||||
|
# activations. Attention matmul activation quantization is not supported by this
|
||||||
|
# Core ML graph pass.
|
||||||
|
op_type_configs={"linear": activation},
|
||||||
)
|
)
|
||||||
|
|
||||||
linear_weight = optimize.coreml.OpLinearQuantizerConfig(
|
linear_weight = optimize.coreml.OpLinearQuantizerConfig(
|
||||||
|
|||||||
@@ -37,12 +37,15 @@ class FakeOptimize:
|
|||||||
class CoreMLQuantizationConfigTests(unittest.TestCase):
|
class CoreMLQuantizationConfigTests(unittest.TestCase):
|
||||||
def test_matches_accepted_qdq_policy(self):
|
def test_matches_accepted_qdq_policy(self):
|
||||||
activation, weights = quantize_coreml._optimization_configs(FakeOptimize)
|
activation, weights = quantize_coreml._optimization_configs(FakeOptimize)
|
||||||
self.assertEqual("linear", activation.global_config.values["mode"])
|
self.assertIsNone(activation.global_config)
|
||||||
self.assertIs(np.uint8, activation.global_config.values["dtype"])
|
activation_linear = activation.op_type_configs["linear"].values
|
||||||
|
self.assertEqual("linear", activation_linear["mode"])
|
||||||
|
self.assertIs(np.uint8, activation_linear["dtype"])
|
||||||
self.assertEqual(
|
self.assertEqual(
|
||||||
"per_tensor",
|
"per_tensor",
|
||||||
activation.global_config.values["granularity"],
|
activation_linear["granularity"],
|
||||||
)
|
)
|
||||||
|
self.assertEqual({"linear"}, set(activation.op_type_configs))
|
||||||
|
|
||||||
linear = weights.op_type_configs["linear"].values
|
linear = weights.op_type_configs["linear"].values
|
||||||
self.assertEqual("linear_symmetric", linear["mode"])
|
self.assertEqual("linear_symmetric", linear["mode"])
|
||||||
|
|||||||
Reference in New Issue
Block a user