diff --git a/ENERGY_AND_RESIDENCY.md b/ENERGY_AND_RESIDENCY.md index 4dbc1d0..6a60993 100644 --- a/ENERGY_AND_RESIDENCY.md +++ b/ENERGY_AND_RESIDENCY.md @@ -19,7 +19,8 @@ placement. ## macOS -- Generate the ML Program with `convert_coreml.py`, then run the gated `eval.py +- Generate the float ML Program with `convert_coreml.py`, calibrate the W8A8 candidate + with `quantize_coreml.py`, then run the gated `eval.py --coreml-model ... --coreml-compute-units cpu-and-ne --compare-onnx ...` command from the README. Preserve the conversion manifest and frozen report with the model metrics. - Run `inspect_coreml.py` with `--compute-units cpu-and-ne`; preserve its full operation @@ -36,6 +37,13 @@ placement. per inference than CPU-only. On AC, a `.all` GPU retry must separately meet its declared GPU operation-share floor before it can activate a deep model. +The first float16 candidate is a diagnostic baseline only: it achieved 94.98% scored +accuracy and 97.97% agreement with the accepted ONNX artifact, so it fails the rollout +accuracy and parity gates despite 1.51 ms p95 latency. Its compute plan preferred the ANE +for 150/165 operations with placement information (90.91%) and 59.71% of estimated cost. +Do not spend energy-measurement time on that rejected package; repeat placement, latency, +and energy measurements on the first accuracy-qualified W8A8 package. + ## Windows - Record the Windows ML execution provider and assigned device after AOT compilation. diff --git a/README.md b/README.md index 36a77af..fdd4420 100644 --- a/README.md +++ b/README.md @@ -217,6 +217,28 @@ ml/purpose-classifier/venv/bin/python ml/purpose-classifier/convert_coreml.py \ --overwrite-output ``` +The direct float16 package is the conversion baseline, not the accepted Apple artifact. +On the first physical Apple-Silicon run it scored 94.98% (890/937), two correct decisions +behind the accepted ONNX graph, with 97.97% scorable label agreement. Calibrate a +Core ML-native W8A8 candidate with the same deterministic 256-record sample and QDQ policy +as the ONNX exporter: + +```bash +ml/purpose-classifier/venv/bin/python ml/purpose-classifier/quantize_coreml.py \ + --model \ + ml/purpose-classifier/outputs/purpose-lite-v1-distilled-qat-mlx-4e/coreml/purpose-lite-v1-fp16.mlpackage \ + --model-dir \ + ml/purpose-classifier/outputs/purpose-lite-v1-distilled-qat-mlx-4e/model \ + --output \ + ml/purpose-classifier/outputs/purpose-lite-v1-distilled-qat-mlx-4e/coreml/purpose-lite-v1-w8a8.mlpackage \ + --overwrite-output +``` + +Activation calibration is grouped to keep temporary Core ML packages bounded and prints +progress while it runs. The candidate uses per-tensor asymmetric uint8 activations, +per-channel symmetric int8 linear weights, and per-tensor asymmetric uint8 embedding +weights. It fails the command if the resulting package exceeds 25 MiB. + Run the frozen gate with CPU+Neural Engine placement and compare labels directly with the accepted int8 ONNX artifact. Gated Core ML evaluation fails closed without `--compare-onnx`, and requires at least 99.5% scorable label agreement: @@ -228,7 +250,7 @@ ml/purpose-classifier/venv/bin/python ml/purpose-classifier/eval.py \ --calibration \ ml/purpose-classifier/outputs/purpose-lite-v1-distilled-qat-mlx-4e/calibration.json \ --coreml-model \ - ml/purpose-classifier/outputs/purpose-lite-v1-distilled-qat-mlx-4e/coreml/purpose-lite-v1-fp16.mlpackage \ + ml/purpose-classifier/outputs/purpose-lite-v1-distilled-qat-mlx-4e/coreml/purpose-lite-v1-w8a8.mlpackage \ --coreml-compute-units cpu-and-ne \ --compare-onnx \ ml/purpose-classifier/outputs/purpose-lite-v1-distilled-qat-mlx-4e/export/purpose-lite-v1-int8-qdq.onnx \ @@ -243,7 +265,7 @@ energy comparison in `ENERGY_AND_RESIDENCY.md`: ```bash ml/purpose-classifier/venv/bin/python ml/purpose-classifier/inspect_coreml.py \ --model \ - ml/purpose-classifier/outputs/purpose-lite-v1-distilled-qat-mlx-4e/coreml/purpose-lite-v1-fp16.mlpackage \ + ml/purpose-classifier/outputs/purpose-lite-v1-distilled-qat-mlx-4e/coreml/purpose-lite-v1-w8a8.mlpackage \ --compute-units cpu-and-ne \ --report \ ml/purpose-classifier/outputs/purpose-lite-v1-distilled-qat-mlx-4e/coreml/ane-compute-plan.json diff --git a/quantize_coreml.py b/quantize_coreml.py new file mode 100644 index 0000000..3a6008e --- /dev/null +++ b/quantize_coreml.py @@ -0,0 +1,261 @@ +#!/usr/bin/env python3 +"""Calibrate a Core ML W8A8 candidate from the selected float16 ML Program.""" + +from __future__ import annotations + +import argparse +import hashlib +import shutil +import sys +from collections import Counter +from pathlib import Path +from typing import Any, Sequence + +import numpy as np + +from convert_coreml import COREMLTOOLS_VERSION, _package_manifest +from export import stratified_calibration_sample +from purpose_data import DataError, load_jsonl, prompt_hash, write_json +from train import encode_fixed_shape + + +SCRIPT_DIR = Path(__file__).resolve().parent +CANDIDATE_DIR = ( + SCRIPT_DIR / "outputs" / "purpose-lite-v1-distilled-qat-mlx-4e" +) +DEFAULT_MODEL = CANDIDATE_DIR / "coreml" / "purpose-lite-v1-fp16.mlpackage" +DEFAULT_MODEL_DIR = CANDIDATE_DIR / "model" +DEFAULT_VALIDATION = SCRIPT_DIR / ".artifacts" / "dataset-v1" / "validation.jsonl" +DEFAULT_OUTPUT = CANDIDATE_DIR / "coreml" / "purpose-lite-v1-w8a8.mlpackage" +SHIPPING_BUDGET_BYTES = 25 * 1024 * 1024 + + +def _tree_sha256(package: Path) -> str: + digest = hashlib.sha256() + for path in sorted(item for item in package.rglob("*") if item.is_file()): + relative = str(path.relative_to(package)).encode("utf-8") + digest.update(relative) + digest.update(b"\0") + with path.open("rb") as handle: + for chunk in iter(lambda: handle.read(1024 * 1024), b""): + digest.update(chunk) + return digest.hexdigest() + + +def _optimization_configs(optimize: Any) -> tuple[Any, Any]: + """Return the Core ML analogue of the accepted ONNX QDQ policy.""" + + activation = optimize.coreml.OpLinearQuantizerConfig( + mode="linear", + dtype=np.uint8, + granularity="per_tensor", + ) + activation_config = optimize.coreml.OptimizationConfig( + global_config=activation, + ) + + linear_weight = optimize.coreml.OpLinearQuantizerConfig( + mode="linear_symmetric", + dtype=np.int8, + granularity="per_channel", + weight_threshold=2048, + ) + embedding_weight = optimize.coreml.OpLinearQuantizerConfig( + mode="linear", + dtype=np.uint8, + granularity="per_tensor", + weight_threshold=2048, + ) + weight_config = optimize.coreml.OptimizationConfig( + op_type_configs={ + "gather": embedding_weight, + "linear": linear_weight, + "matmul": linear_weight, + } + ) + return activation_config, weight_config + + +def _validate_args(args: argparse.Namespace) -> None: + if not args.model.is_dir(): + raise DataError(f"{args.model}: source Core ML package is missing") + if not args.model_dir.is_dir(): + raise DataError(f"{args.model_dir}: tokenizer directory is missing") + if args.output.suffix != ".mlpackage": + raise DataError("Core ML output must end in .mlpackage") + try: + same_output = args.model.resolve() == args.output.resolve() + except OSError as exc: + raise DataError(f"cannot resolve Core ML package paths: {exc}") from exc + if same_output: + raise DataError("W8A8 output must not overwrite its float16 source package") + if args.output.exists() and not args.overwrite_output: + raise DataError( + f"{args.output}: output exists; pass --overwrite-output intentionally" + ) + + +def quantize(args: argparse.Namespace) -> dict[str, Any]: + _validate_args(args) + try: + import coremltools as ct + import coremltools.optimize as cto + import torch + from transformers import AutoTokenizer + except ImportError as exc: + raise DataError( + "Core ML quantization requires requirements-coreml.txt on macOS" + ) from exc + if ct.__version__ != COREMLTOOLS_VERSION: + raise DataError( + f"expected coremltools {COREMLTOOLS_VERSION}, found {ct.__version__}" + ) + + validation = load_jsonl(args.validation) + calibration = stratified_calibration_sample( + validation, + args.calibration_records, + seed=args.calibration_seed, + ) + tokenizer = AutoTokenizer.from_pretrained( + args.model_dir, + local_files_only=True, + ) + sample_data = [] + for record in calibration: + encoded = encode_fixed_shape(tokenizer, [record["prompt"]], torch) + sample_data.append( + { + name: value.numpy().astype(np.int32, copy=False) + for name, value in encoded.items() + } + ) + + print( + f"Core ML activation calibration: {len(sample_data)} records", + flush=True, + ) + source_model = ct.models.MLModel( + str(args.model), + compute_units=ct.ComputeUnit.CPU_ONLY, + ) + input_names = {item.name for item in source_model.get_spec().description.input} + expected_inputs = {"input_ids", "attention_mask", "token_type_ids"} + if input_names != expected_inputs: + raise DataError( + f"Core ML inputs changed: expected {sorted(expected_inputs)}, " + f"got {sorted(input_names)}" + ) + activation_config, weight_config = _optimization_configs(cto) + activation_quantized = cto.coreml.linear_quantize_activations( + source_model, + activation_config, + sample_data, + calibration_op_group_size=args.calibration_op_group_size, + ) + print("Core ML weight quantization: W8", flush=True) + quantized = cto.coreml.linear_quantize_weights( + activation_quantized, + weight_config, + ) + quantized.user_defined_metadata["com.nucleic.model.quantization"] = "W8A8" + quantized.user_defined_metadata["com.nucleic.model.quantizationCalibration"] = ( + f"stratified:{args.calibration_records}:seed={args.calibration_seed}" + ) + + if args.output.exists(): + if args.output.is_dir(): + shutil.rmtree(args.output) + else: + args.output.unlink() + args.output.parent.mkdir(parents=True, exist_ok=True) + quantized.save(str(args.output)) + + manifest = _package_manifest(args.output) + manifest.update( + { + "sourcePackage": str(args.model), + "sourcePackageSha256": _tree_sha256(args.model), + "coremltoolsVersion": ct.__version__, + "quantization": { + "name": "W8A8", + "activations": "per-tensor asymmetric uint8", + "linearWeights": "per-channel symmetric int8", + "embeddingWeights": "per-tensor asymmetric uint8", + }, + "calibrationRecords": len(calibration), + "calibrationSeed": args.calibration_seed, + "calibrationOpGroupSize": args.calibration_op_group_size, + "calibrationSample": { + "strategy": "stratified by purpose, slice, and primary language", + "purposeCounts": dict( + sorted(Counter(item["purpose"] for item in calibration).items()) + ), + "sliceCounts": dict( + sorted(Counter(item["slice"] for item in calibration).items()) + ), + "promptHashes": sorted( + prompt_hash(item["prompt"]) for item in calibration + ), + }, + "shippingBudgetBytes": args.shipping_budget_bytes, + "shippingBudgetPassed": ( + manifest["bytes"] <= args.shipping_budget_bytes + ), + } + ) + manifest_path = args.output.with_name(f"{args.output.stem}-manifest.json") + write_json(manifest_path, manifest) + print( + f"Core ML W8A8 package: {args.output} ({manifest['bytes']} bytes)", + flush=True, + ) + if not manifest["shippingBudgetPassed"]: + raise DataError( + f"{args.output}: {manifest['bytes']} bytes exceeds the " + f"{args.shipping_budget_bytes}-byte shipping budget" + ) + return manifest + + +def build_parser() -> argparse.ArgumentParser: + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("--model", type=Path, default=DEFAULT_MODEL) + parser.add_argument("--model-dir", type=Path, default=DEFAULT_MODEL_DIR) + parser.add_argument("--validation", type=Path, default=DEFAULT_VALIDATION) + parser.add_argument("--output", type=Path, default=DEFAULT_OUTPUT) + parser.add_argument("--calibration-records", type=int, default=256) + parser.add_argument("--calibration-seed", type=int, default=20260730) + parser.add_argument("--calibration-op-group-size", type=int, default=32) + parser.add_argument( + "--shipping-budget-bytes", + type=int, + default=SHIPPING_BUDGET_BYTES, + ) + parser.add_argument("--overwrite-output", action="store_true") + return parser + + +def main(argv: Sequence[str] | None = None) -> int: + parser = build_parser() + args = parser.parse_args(argv) + if ( + args.calibration_records <= 0 + or args.calibration_op_group_size == 0 + or args.calibration_op_group_size < -1 + or args.shipping_budget_bytes <= 0 + ): + parser.error( + "calibration records/budget must be positive and op group size must be " + "-1 or positive" + ) + try: + quantize(args) + except (DataError, OSError, RuntimeError, ValueError) as exc: + print(f"error: {exc}", file=sys.stderr) + return 1 + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/tests/test_quantize_coreml.py b/tests/test_quantize_coreml.py new file mode 100644 index 0000000..aab057f --- /dev/null +++ b/tests/test_quantize_coreml.py @@ -0,0 +1,79 @@ +import argparse +import sys +import tempfile +import unittest +from pathlib import Path + +import numpy as np + + +MODULE_DIR = Path(__file__).resolve().parents[1] +sys.path.insert(0, str(MODULE_DIR)) + +import quantize_coreml +from purpose_data import DataError + + +class FakeOpLinearQuantizerConfig: + def __init__(self, **values): + self.values = values + + +class FakeOptimizationConfig: + def __init__(self, *, global_config=None, op_type_configs=None): + self.global_config = global_config + self.op_type_configs = op_type_configs or {} + + +class FakeCoreML: + OpLinearQuantizerConfig = FakeOpLinearQuantizerConfig + OptimizationConfig = FakeOptimizationConfig + + +class FakeOptimize: + coreml = FakeCoreML + + +class CoreMLQuantizationConfigTests(unittest.TestCase): + def test_matches_accepted_qdq_policy(self): + activation, weights = quantize_coreml._optimization_configs(FakeOptimize) + self.assertEqual("linear", activation.global_config.values["mode"]) + self.assertIs(np.uint8, activation.global_config.values["dtype"]) + self.assertEqual( + "per_tensor", + activation.global_config.values["granularity"], + ) + + linear = weights.op_type_configs["linear"].values + self.assertEqual("linear_symmetric", linear["mode"]) + self.assertIs(np.int8, linear["dtype"]) + self.assertEqual("per_channel", linear["granularity"]) + self.assertIs( + weights.op_type_configs["linear"], + weights.op_type_configs["matmul"], + ) + + embedding = weights.op_type_configs["gather"].values + self.assertEqual("linear", embedding["mode"]) + self.assertIs(np.uint8, embedding["dtype"]) + self.assertEqual("per_tensor", embedding["granularity"]) + + def test_rejects_overwriting_source_package(self): + with tempfile.TemporaryDirectory() as temp: + root = Path(temp) + package = root / "model.mlpackage" + package.mkdir() + model_dir = root / "model" + model_dir.mkdir() + args = argparse.Namespace( + model=package, + model_dir=model_dir, + output=package, + overwrite_output=True, + ) + with self.assertRaisesRegex(DataError, "must not overwrite"): + quantize_coreml._validate_args(args) + + +if __name__ == "__main__": + unittest.main()