Merge nucleic/sleek-ember-seal-uady into dev

This commit is contained in:
2026-07-30 20:56:46 -07:00
parent e05521482f
commit 9e462a79fc
9 changed files with 832 additions and 8 deletions
+7
View File
@@ -19,6 +19,13 @@ placement.
## macOS
- Generate the ML Program with `convert_coreml.py`, then run the gated `eval.py
--coreml-model ... --coreml-compute-units cpu-and-ne --compare-onnx ...` command from
the README. Preserve the conversion manifest and frozen report with the model metrics.
- Run `inspect_coreml.py` with `--compute-units cpu-and-ne`; preserve its full operation
report and record both `neuralEngineOperationShare` and
`neuralEngineEstimatedCostShare`. A gated evaluation without the ONNX comparison is
invalid.
- Run the release Core ML artifact once with `.cpuAndNeuralEngine`, recording the
`MLComputePlan` ANE operation share and the artifact's required floor.
- Repeat with a diagnostic CPU-only configuration on the same Mac and power source.
+53
View File
@@ -196,6 +196,59 @@ routing-tier-drift gates pass. This is the current accuracy-qualified shipping c
latency and energy/residency still require measurement on the target Apple and Windows
accelerator runtimes.
### Convert and validate Core ML
Core ML Tools no longer maintains the legacy ONNX converter, so the Apple artifact is
converted directly from the selected Hugging Face checkpoint. `convert_coreml.py` uses a
fixed-shape export-only BERT forward to avoid dynamic Transformers masking helpers, checks
that forward against Transformers before conversion, writes an ML Program package, and
records hashes for every package file.
Install the pinned converter in the macOS environment and create the package:
```bash
ml/purpose-classifier/venv/bin/python -m pip install \
-r ml/purpose-classifier/requirements-coreml.txt
ml/purpose-classifier/venv/bin/python ml/purpose-classifier/convert_coreml.py \
--model-dir \
ml/purpose-classifier/outputs/purpose-lite-v1-distilled-qat-mlx-4e/model \
--output \
ml/purpose-classifier/outputs/purpose-lite-v1-distilled-qat-mlx-4e/coreml/purpose-lite-v1-fp16.mlpackage \
--overwrite-output
```
Run the frozen gate with CPU+Neural Engine placement and compare labels directly with the
accepted int8 ONNX artifact. Gated Core ML evaluation fails closed without
`--compare-onnx`, and requires at least 99.5% scorable label agreement:
```bash
ml/purpose-classifier/venv/bin/python ml/purpose-classifier/eval.py \
--model-dir \
ml/purpose-classifier/outputs/purpose-lite-v1-distilled-qat-mlx-4e/model \
--calibration \
ml/purpose-classifier/outputs/purpose-lite-v1-distilled-qat-mlx-4e/calibration.json \
--coreml-model \
ml/purpose-classifier/outputs/purpose-lite-v1-distilled-qat-mlx-4e/coreml/purpose-lite-v1-fp16.mlpackage \
--coreml-compute-units cpu-and-ne \
--compare-onnx \
ml/purpose-classifier/outputs/purpose-lite-v1-distilled-qat-mlx-4e/export/purpose-lite-v1-int8-qdq.onnx \
--report \
ml/purpose-classifier/outputs/purpose-lite-v1-distilled-qat-mlx-4e/coreml/ane-frozen-eval.json
```
Record the compute plan separately; this reports both operation-count and estimated-cost
ANE shares. Repeat evaluation with `--coreml-compute-units cpu-only --no-gate` before the
energy comparison in `ENERGY_AND_RESIDENCY.md`:
```bash
ml/purpose-classifier/venv/bin/python ml/purpose-classifier/inspect_coreml.py \
--model \
ml/purpose-classifier/outputs/purpose-lite-v1-distilled-qat-mlx-4e/coreml/purpose-lite-v1-fp16.mlpackage \
--compute-units cpu-and-ne \
--report \
ml/purpose-classifier/outputs/purpose-lite-v1-distilled-qat-mlx-4e/coreml/ane-compute-plan.json
```
For a wiring smoke test, use a small deterministic prefix:
```bash
+354
View File
@@ -0,0 +1,354 @@
#!/usr/bin/env python3
"""Convert the selected purpose-lite checkpoint to a fixed-shape Core ML package."""
from __future__ import annotations
import argparse
import hashlib
import json
import math
import shutil
import sys
from pathlib import Path
from typing import Any, Sequence
import numpy as np
import torch
import torch.nn.functional as functional
from purpose_data import LABELS, DataError, load_jsonl, write_json
from train import MAX_LENGTH, encode_fixed_shape
SCRIPT_DIR = Path(__file__).resolve().parent
DEFAULT_MODEL_DIR = (
SCRIPT_DIR
/ "outputs"
/ "purpose-lite-v1-distilled-qat-mlx-4e"
/ "model"
)
DEFAULT_VALIDATION = SCRIPT_DIR / ".artifacts" / "dataset-v1" / "validation.jsonl"
DEFAULT_OUTPUT = (
SCRIPT_DIR
/ "outputs"
/ "purpose-lite-v1-distilled-qat-mlx-4e"
/ "coreml"
/ "purpose-lite-v1-fp16.mlpackage"
)
COREMLTOOLS_VERSION = "9.0"
class FixedShapeBertForCoreML(torch.nn.Module):
"""Conversion-only BERT forward without Transformers' dynamic mask helpers."""
def __init__(self, model: Any) -> None:
super().__init__()
self.model = model
config = model.config
self.num_heads = int(config.num_attention_heads)
self.head_size = int(config.hidden_size) // self.num_heads
if int(config.hidden_size) % self.num_heads:
raise DataError("BERT hidden size must be divisible by attention heads")
if int(config.max_position_embeddings) < MAX_LENGTH:
raise DataError("BERT checkpoint cannot represent the fixed 128-token input")
self.register_buffer(
"fixed_position_ids",
torch.arange(MAX_LENGTH, dtype=torch.int64).reshape(1, MAX_LENGTH),
persistent=False,
)
def forward(
self,
input_ids: Any,
attention_mask: Any,
token_type_ids: Any,
) -> Any:
bert = self.model.bert
input_ids = input_ids.to(torch.int64)
token_type_ids = token_type_ids.to(torch.int64)
value = (
bert.embeddings.word_embeddings(input_ids)
+ bert.embeddings.position_embeddings(self.fixed_position_ids)
+ bert.embeddings.token_type_embeddings(token_type_ids)
)
value = bert.embeddings.LayerNorm(value)
zero = torch.zeros((), dtype=value.dtype, device=value.device)
hidden = torch.full((), -10000.0, dtype=value.dtype, device=value.device)
additive_mask = torch.where(
attention_mask[:, None, None, :] != 0,
zero,
hidden,
)
for layer in bert.encoder.layer:
self_attention = layer.attention.self
def split_heads(projected: Any) -> Any:
return projected.reshape(
1,
MAX_LENGTH,
self.num_heads,
self.head_size,
).permute(0, 2, 1, 3)
queries = split_heads(self_attention.query(value))
keys = split_heads(self_attention.key(value))
values = split_heads(self_attention.value(value))
scores = torch.matmul(queries, keys.transpose(-1, -2)) / math.sqrt(
self.head_size
)
probabilities = torch.softmax(scores + additive_mask, dim=-1)
context = torch.matmul(probabilities, values)
context = context.permute(0, 2, 1, 3).reshape(
1,
MAX_LENGTH,
-1,
)
attention_output = layer.attention.output.dense(context)
value = layer.attention.output.LayerNorm(value + attention_output)
intermediate = functional.gelu(
layer.intermediate.dense(value),
approximate="none",
)
value = layer.output.LayerNorm(
value + layer.output.dense(intermediate)
)
pooled = torch.tanh(bert.pooler.dense(value[:, 0]))
return self.model.classifier(pooled)
def _checkpoint_config(model_dir: Path) -> dict[str, Any]:
config_path = model_dir / "config.json"
try:
config = json.loads(config_path.read_text(encoding="utf-8"))
except (OSError, UnicodeError, json.JSONDecodeError) as exc:
raise DataError(f"{config_path}: cannot load model config: {exc}") from exc
configured_labels = [
config.get("id2label", {}).get(
str(index),
config.get("id2label", {}).get(index),
)
for index in range(len(LABELS))
]
if (
config.get("model_type") != "bert"
or config.get("hidden_size") != 384
or config.get("num_hidden_layers") != 6
):
raise DataError("Core ML conversion requires the purpose-lite BERT architecture")
if configured_labels != list(LABELS):
raise DataError("Core ML checkpoint label order does not match purpose-lite")
return config
def _sha256(path: Path) -> str:
digest = hashlib.sha256()
with path.open("rb") as handle:
for chunk in iter(lambda: handle.read(1024 * 1024), b""):
digest.update(chunk)
return digest.hexdigest()
def _package_manifest(package: Path) -> dict[str, Any]:
files = []
for path in sorted(item for item in package.rglob("*") if item.is_file()):
files.append(
{
"path": str(path.relative_to(package)),
"bytes": path.stat().st_size,
"sha256": _sha256(path),
}
)
return {
"schemaVersion": 1,
"modelVersion": "purpose-lite-v1",
"package": package.name,
"bytes": sum(item["bytes"] for item in files),
"files": files,
}
def _verify_wrapper_parity(
torch_model: Any,
wrapper: FixedShapeBertForCoreML,
tokenizer: Any,
records: Sequence[dict[str, Any]],
) -> float:
encoded = encode_fixed_shape(
tokenizer,
[record["prompt"] for record in records],
torch,
)
maximum_error = 0.0
with torch.inference_mode():
for index in range(len(records)):
item = {key: value[index : index + 1] for key, value in encoded.items()}
reference = torch_model(**item).logits
candidate = wrapper(
item["input_ids"].to(torch.int32),
item["attention_mask"].to(torch.int32),
item.get("token_type_ids", torch.zeros_like(item["input_ids"])).to(
torch.int32
),
)
error = float(torch.max(torch.abs(reference - candidate)).item())
maximum_error = max(maximum_error, error)
torch.testing.assert_close(candidate, reference, rtol=1e-5, atol=2e-5)
if int(candidate.argmax(dim=-1).item()) != int(
reference.argmax(dim=-1).item()
):
raise DataError("conversion wrapper changed a parity prediction")
return maximum_error
def convert(args: argparse.Namespace) -> dict[str, Any]:
_checkpoint_config(args.model_dir)
checkpoint = args.model_dir / "model.safetensors"
if not checkpoint.is_file():
raise DataError(f"{checkpoint}: checkpoint is missing")
if args.output.suffix != ".mlpackage":
raise DataError("Core ML output must end in .mlpackage")
if args.output.exists():
if not args.overwrite_output:
raise DataError(
f"{args.output}: output exists; pass --overwrite-output intentionally"
)
if args.output.is_dir():
shutil.rmtree(args.output)
else:
args.output.unlink()
args.output.parent.mkdir(parents=True, exist_ok=True)
try:
import coremltools as ct
from transformers import AutoModelForSequenceClassification, AutoTokenizer
except ImportError as exc:
raise DataError(
"Core ML conversion requires requirements-coreml.txt on macOS"
) from exc
if ct.__version__ != COREMLTOOLS_VERSION:
raise DataError(
f"expected coremltools {COREMLTOOLS_VERSION}, found {ct.__version__}"
)
model = AutoModelForSequenceClassification.from_pretrained(
args.model_dir,
local_files_only=True,
).eval()
tokenizer = AutoTokenizer.from_pretrained(args.model_dir, local_files_only=True)
wrapper = FixedShapeBertForCoreML(model).eval()
validation = load_jsonl(args.validation)
if len(validation) < args.parity_records:
raise DataError("validation split is smaller than --parity-records")
maximum_error = _verify_wrapper_parity(
model,
wrapper,
tokenizer,
validation[: args.parity_records],
)
print(f"conversion-wrapper parity: max_abs_error={maximum_error:.3g}")
example = (
torch.zeros((1, MAX_LENGTH), dtype=torch.int32),
torch.ones((1, MAX_LENGTH), dtype=torch.int32),
torch.zeros((1, MAX_LENGTH), dtype=torch.int32),
)
with torch.inference_mode():
traced = torch.jit.trace(wrapper, example, strict=True)
traced = torch.jit.freeze(traced)
deployment_target = getattr(ct.target, args.minimum_deployment_target, None)
if deployment_target is None:
raise DataError(
f"coremltools does not support {args.minimum_deployment_target}"
)
coreml_model = ct.convert(
traced,
convert_to="mlprogram",
minimum_deployment_target=deployment_target,
compute_precision=ct.precision.FLOAT16,
inputs=[
ct.TensorType(
name="input_ids",
shape=(1, MAX_LENGTH),
dtype=np.int32,
),
ct.TensorType(
name="attention_mask",
shape=(1, MAX_LENGTH),
dtype=np.int32,
),
ct.TensorType(
name="token_type_ids",
shape=(1, MAX_LENGTH),
dtype=np.int32,
),
],
outputs=[ct.TensorType(name="logits", dtype=np.float32)],
)
coreml_model.author = "Nucleic"
coreml_model.short_description = "purpose-lite-v1 prompt classifier"
coreml_model.version = "purpose-lite-v1"
coreml_model.user_defined_metadata["com.nucleic.model.version"] = (
"purpose-lite-v1"
)
coreml_model.user_defined_metadata["com.nucleic.model.labels"] = json.dumps(
list(LABELS),
separators=(",", ":"),
)
coreml_model.user_defined_metadata["com.nucleic.model.sourceSha256"] = _sha256(
checkpoint
)
coreml_model.user_defined_metadata["com.nucleic.model.fixedShape"] = "1x128"
coreml_model.user_defined_metadata["com.nucleic.model.minimumDeploymentTarget"] = (
args.minimum_deployment_target
)
coreml_model.save(str(args.output))
manifest = _package_manifest(args.output)
manifest.update(
{
"sourceCheckpoint": str(checkpoint),
"sourceCheckpointSha256": _sha256(checkpoint),
"coremltoolsVersion": ct.__version__,
"minimumDeploymentTarget": args.minimum_deployment_target,
"computePrecision": "float16",
"wrapperMaximumAbsoluteError": maximum_error,
}
)
manifest_path = args.output.with_name(f"{args.output.stem}-manifest.json")
write_json(manifest_path, manifest)
print(f"Core ML package: {args.output} ({manifest['bytes']} bytes)")
return manifest
def build_parser() -> argparse.ArgumentParser:
parser = argparse.ArgumentParser(description=__doc__)
parser.add_argument("--model-dir", type=Path, default=DEFAULT_MODEL_DIR)
parser.add_argument("--validation", type=Path, default=DEFAULT_VALIDATION)
parser.add_argument("--output", type=Path, default=DEFAULT_OUTPUT)
parser.add_argument("--parity-records", type=int, default=16)
parser.add_argument(
"--minimum-deployment-target",
default="macOS15",
choices=("macOS15", "macOS26"),
)
parser.add_argument("--overwrite-output", action="store_true")
return parser
def main(argv: Sequence[str] | None = None) -> int:
parser = build_parser()
args = parser.parse_args(argv)
if args.parity_records <= 0:
parser.error("--parity-records must be positive")
try:
convert(args)
except (AssertionError, DataError, OSError, RuntimeError, ValueError) as exc:
print(f"error: {exc}", file=sys.stderr)
return 1
return 0
if __name__ == "__main__":
raise SystemExit(main())
+130 -8
View File
@@ -13,6 +13,8 @@ from collections import Counter
from pathlib import Path
from typing import Any, Sequence
import numpy as np
from purpose_data import (
HARD_SLICES,
LABELS,
@@ -100,6 +102,16 @@ def _synchronize(torch: Any, device: Any) -> None:
torch.mps.synchronize()
def _coreml_compute_unit(coremltools: Any, requested: str) -> Any:
values = {
"all": coremltools.ComputeUnit.ALL,
"cpu-only": coremltools.ComputeUnit.CPU_ONLY,
"cpu-and-gpu": coremltools.ComputeUnit.CPU_AND_GPU,
"cpu-and-ne": coremltools.ComputeUnit.CPU_AND_NE,
}
return values[requested]
def routing_tier_drift(
records: Sequence[dict[str, Any]],
actual: Sequence[int],
@@ -245,8 +257,62 @@ def evaluate(args: argparse.Namespace) -> dict[str, Any]:
label_to_id = {label: index for index, label in enumerate(LABELS)}
tokenizer = AutoTokenizer.from_pretrained(args.model_dir, local_files_only=True)
onnx_session = None
coreml_model = None
reference_model = None
if args.onnx_model is not None:
reference_onnx_session = None
reference_onnx_input_names: set[str] = set()
if args.compare_pytorch and args.compare_onnx is not None:
raise DataError("choose only one parity reference")
if args.coreml_model is not None:
try:
import coremltools as ct
except ImportError as exc:
raise DataError(
"Core ML evaluation requires requirements-coreml.txt on macOS"
) from exc
if args.device != "auto":
raise DataError(
"Core ML compute placement uses --coreml-compute-units, not --device"
)
coreml_model = ct.models.MLModel(
str(args.coreml_model),
compute_units=_coreml_compute_unit(ct, args.coreml_compute_units),
)
coreml_input_names = {
item.name for item in coreml_model.get_spec().description.input
}
device = torch.device("cpu")
model = None
onnx_input_names = set()
runtime_name = f"coreml-{args.coreml_compute_units}"
if args.compare_pytorch:
reference_model = AutoModelForSequenceClassification.from_pretrained(
args.model_dir,
local_files_only=True,
).to(device)
reference_model.eval()
if args.compare_onnx is not None:
try:
import onnxruntime as ort
except ImportError as exc:
raise DataError(
"Core ML↔ONNX parity requires onnxruntime"
) from exc
reference_options = ort.SessionOptions()
reference_options.graph_optimization_level = (
ort.GraphOptimizationLevel.ORT_ENABLE_ALL
)
reference_onnx_session = ort.InferenceSession(
str(args.compare_onnx),
sess_options=reference_options,
providers=["CPUExecutionProvider"],
)
reference_onnx_input_names = {
item.name for item in reference_onnx_session.get_inputs()
}
elif args.onnx_model is not None:
if args.compare_onnx is not None:
raise DataError("--compare-onnx requires --coreml-model")
try:
import onnxruntime as ort
except ImportError as exc:
@@ -273,8 +339,12 @@ def evaluate(args: argparse.Namespace) -> dict[str, Any]:
).to(device)
reference_model.eval()
else:
if args.compare_onnx is not None:
raise DataError("--compare-onnx requires --coreml-model")
if args.compare_pytorch:
raise DataError("--compare-pytorch requires --onnx-model")
raise DataError(
"--compare-pytorch requires --onnx-model or --coreml-model"
)
device = _device(torch, args.device)
model = AutoModelForSequenceClassification.from_pretrained(
args.model_dir, local_files_only=True
@@ -284,6 +354,14 @@ def evaluate(args: argparse.Namespace) -> dict[str, Any]:
runtime_name = "pytorch"
def predict_logits(encoded: dict[str, Any]) -> Any:
if coreml_model is not None:
inputs = {
key: value.numpy().astype(np.int32, copy=False)
for key, value in encoded.items()
if key in coreml_input_names
}
output = np.asarray(coreml_model.predict(inputs)["logits"])
return torch.from_numpy(output.reshape(1, len(LABELS)))
if onnx_session is not None:
inputs = {
key: value.numpy()
@@ -300,7 +378,9 @@ def evaluate(args: argparse.Namespace) -> dict[str, Any]:
confidences: list[str] = []
margins: list[float] = []
reference_predictions: list[int] = []
inference_batch_size = 1 if onnx_session is not None else args.batch_size
inference_batch_size = (
1 if onnx_session is not None or coreml_model is not None else args.batch_size
)
with torch.inference_mode():
for start in range(0, len(records), inference_batch_size):
batch = records[start : start + inference_batch_size]
@@ -315,6 +395,19 @@ def evaluate(args: argparse.Namespace) -> dict[str, Any]:
reference_predictions.extend(
reference_logits.argmax(dim=-1).tolist()
)
elif reference_onnx_session is not None:
reference_inputs = {
key: value.numpy()
for key, value in encoded.items()
if key in reference_onnx_input_names
}
reference_logits = reference_onnx_session.run(
["logits"],
reference_inputs,
)[0]
reference_predictions.extend(
np.asarray(reference_logits).argmax(axis=-1).tolist()
)
distribution = torch.softmax(logits, dim=-1)
top = torch.topk(distribution, k=2, dim=-1)
batch_probabilities = top.values[:, 0].tolist()
@@ -465,7 +558,7 @@ def evaluate(args: argparse.Namespace) -> dict[str, Any]:
"modelVersion": calibration.get("modelVersion", args.model_dir.name),
"device": str(device),
"runtime": runtime_name,
"artifact": str(args.onnx_model or args.model_dir),
"artifact": str(args.coreml_model or args.onnx_model or args.model_dir),
"fixedInputShape": [1, MAX_LENGTH],
"overall": metrics,
"scoredClassification": scored_metrics,
@@ -508,13 +601,21 @@ def evaluate(args: argparse.Namespace) -> dict[str, Any]:
},
"misclassifications": misclassifications,
}
if reference_model is not None:
report["pytorchParity"] = prediction_agreement(
if reference_predictions:
parity_name = (
"onnxParity" if reference_onnx_session is not None else "pytorchParity"
)
parity = prediction_agreement(
records,
actual,
reference_predictions,
predicted,
)
report[parity_name] = parity
if reference_onnx_session is not None:
report["gates"]["onnxLabelAgreementAtLeast99_5Percent"] = (
parity["scoredLabelAgreement"] >= 0.995
)
write_json(args.report, report)
return report
@@ -522,15 +623,31 @@ def evaluate(args: argparse.Namespace) -> dict[str, Any]:
def build_parser() -> argparse.ArgumentParser:
parser = argparse.ArgumentParser(description=__doc__)
parser.add_argument("--model-dir", type=Path, default=DEFAULT_MODEL_DIR)
parser.add_argument(
runtime = parser.add_mutually_exclusive_group()
runtime.add_argument(
"--onnx-model",
type=Path,
help="score a fixed-shape ONNX artifact instead of the PyTorch checkpoint",
)
runtime.add_argument(
"--coreml-model",
type=Path,
help="score a fixed-shape Core ML package instead of the PyTorch checkpoint",
)
parser.add_argument(
"--compare-pytorch",
action="store_true",
help="include label-level drift from --model-dir when scoring ONNX",
help="include label-level drift from --model-dir when scoring an artifact",
)
parser.add_argument(
"--compare-onnx",
type=Path,
help="include Core ML label-level drift from this ONNX reference",
)
parser.add_argument(
"--coreml-compute-units",
choices=("all", "cpu-only", "cpu-and-gpu", "cpu-and-ne"),
default="cpu-and-ne",
)
parser.add_argument("--calibration", type=Path, default=DEFAULT_CALIBRATION)
parser.add_argument("--test", type=Path, default=DEFAULT_TEST)
@@ -552,6 +669,11 @@ def main(argv: Sequence[str] | None = None) -> int:
args = parser.parse_args(argv)
if args.batch_size <= 0 or args.latency_samples < 0:
parser.error("batch size must be positive and latency samples non-negative")
if args.coreml_model is not None and args.compare_onnx is None and not args.no_gate:
parser.error(
"gated Core ML evaluation requires --compare-onnx; use --no-gate only "
"for diagnostics"
)
try:
report = evaluate(args)
except (DataError, OSError, ValueError) as exc:
+153
View File
@@ -0,0 +1,153 @@
#!/usr/bin/env python3
"""Inspect Core ML operation placement and estimated accelerator cost share."""
from __future__ import annotations
import argparse
import platform
import sys
from collections import Counter, defaultdict
from pathlib import Path
from typing import Any, Sequence
from purpose_data import DataError, write_json
def _device_category(device: Any) -> str:
name = type(device).__name__.lower()
description = str(device).lower()
combined = f"{name} {description}"
if "neural" in combined:
return "neuralEngine"
if "gpu" in combined:
return "gpu"
if "cpu" in combined:
return "cpu"
return "unknown"
def _compute_unit(coremltools: Any, requested: str) -> Any:
values = {
"all": coremltools.ComputeUnit.ALL,
"cpu-only": coremltools.ComputeUnit.CPU_ONLY,
"cpu-and-gpu": coremltools.ComputeUnit.CPU_AND_GPU,
"cpu-and-ne": coremltools.ComputeUnit.CPU_AND_NE,
}
return values[requested]
def inspect(args: argparse.Namespace) -> dict[str, Any]:
if not args.model.exists():
raise DataError(f"{args.model}: Core ML model is missing")
try:
import coremltools as ct
except ImportError as exc:
raise DataError(
"Core ML inspection requires requirements-coreml.txt on macOS"
) from exc
compiled = ct.models.utils.compile_model(str(args.model))
compute_plan = ct.models.compute_plan.MLComputePlan.load_from_path(
path=str(compiled),
compute_units=_compute_unit(ct, args.compute_units),
)
program = compute_plan.model_structure.program
if program is None or "main" not in program.functions:
raise DataError("Core ML package is not an ML Program with a main function")
operations = list(program.functions["main"].block.operations)
if not operations:
raise DataError("Core ML compute plan contains no operations")
preferred_counts: Counter[str] = Counter()
preferred_costs: dict[str, float] = defaultdict(float)
supported_counts: Counter[str] = Counter()
operation_reports = []
operations_with_usage = 0
operations_with_cost = 0
total_cost = 0.0
for operation in operations:
usage = compute_plan.get_compute_device_usage_for_mlprogram_operation(
operation
)
cost = compute_plan.get_estimated_cost_for_mlprogram_operation(operation)
preferred = "unknown"
supported: list[str] = []
if usage is not None:
operations_with_usage += 1
preferred = _device_category(usage.preferred_compute_device)
preferred_counts[preferred] += 1
supported = sorted(
{_device_category(device) for device in usage.supported_compute_devices}
)
supported_counts.update(supported)
weight = None
if cost is not None:
operations_with_cost += 1
weight = float(cost.weight)
total_cost += weight
preferred_costs[preferred] += weight
operation_reports.append(
{
"operatorName": str(operation.operator_name),
"preferredDevice": preferred,
"supportedDevices": supported,
"estimatedCostWeight": weight,
}
)
ane_operations = preferred_counts["neuralEngine"]
ane_cost = preferred_costs["neuralEngine"]
report = {
"schemaVersion": 1,
"model": str(args.model),
"coremltoolsVersion": ct.__version__,
"machine": platform.machine(),
"macOS": platform.mac_ver()[0],
"computeUnits": args.compute_units,
"operations": len(operations),
"operationsWithDeviceUsage": operations_with_usage,
"operationsWithEstimatedCost": operations_with_cost,
"preferredOperationCounts": dict(sorted(preferred_counts.items())),
"supportedOperationCounts": dict(sorted(supported_counts.items())),
"preferredEstimatedCosts": dict(sorted(preferred_costs.items())),
"neuralEngineOperationShare": (
ane_operations / operations_with_usage if operations_with_usage else 0.0
),
"neuralEngineEstimatedCostShare": (
ane_cost / total_cost if total_cost else 0.0
),
"operationDetails": operation_reports,
}
write_json(args.report, report)
print(
"Core ML placement: "
f"ANE operations={report['neuralEngineOperationShare']:.2%} "
f"ANE estimated cost={report['neuralEngineEstimatedCostShare']:.2%}"
)
return report
def build_parser() -> argparse.ArgumentParser:
parser = argparse.ArgumentParser(description=__doc__)
parser.add_argument("--model", type=Path, required=True)
parser.add_argument("--report", type=Path, required=True)
parser.add_argument(
"--compute-units",
choices=("all", "cpu-only", "cpu-and-gpu", "cpu-and-ne"),
default="cpu-and-ne",
)
return parser
def main(argv: Sequence[str] | None = None) -> int:
args = build_parser().parse_args(argv)
try:
inspect(args)
except (DataError, OSError, RuntimeError, ValueError) as exc:
print(f"error: {exc}", file=sys.stderr)
return 1
return 0
if __name__ == "__main__":
raise SystemExit(main())
+2
View File
@@ -0,0 +1,2 @@
-r requirements.txt
coremltools==9.0
+84
View File
@@ -0,0 +1,84 @@
import json
import sys
import tempfile
import unittest
from pathlib import Path
import torch
from transformers import BertConfig, BertForSequenceClassification
MODULE_DIR = Path(__file__).resolve().parents[1]
sys.path.insert(0, str(MODULE_DIR))
import convert_coreml
from purpose_data import LABELS, DataError
class FixedShapeBertForCoreMLTests(unittest.TestCase):
def test_conversion_forward_matches_transformers(self):
torch.manual_seed(7)
config = BertConfig(
vocab_size=64,
hidden_size=16,
num_hidden_layers=1,
num_attention_heads=4,
intermediate_size=32,
max_position_embeddings=128,
type_vocab_size=2,
hidden_dropout_prob=0.0,
attention_probs_dropout_prob=0.0,
num_labels=len(LABELS),
)
model = BertForSequenceClassification(config).eval()
wrapper = convert_coreml.FixedShapeBertForCoreML(model).eval()
input_ids = torch.randint(0, config.vocab_size, (1, 128), dtype=torch.int32)
attention_mask = torch.zeros((1, 128), dtype=torch.int32)
attention_mask[:, :83] = 1
token_type_ids = torch.zeros((1, 128), dtype=torch.int32)
token_type_ids[:, 43:83] = 1
with torch.inference_mode():
reference = model(
input_ids=input_ids.long(),
attention_mask=attention_mask.long(),
token_type_ids=token_type_ids.long(),
).logits
candidate = wrapper(input_ids, attention_mask, token_type_ids)
torch.testing.assert_close(candidate, reference, rtol=1e-5, atol=2e-5)
traced = torch.jit.trace(
wrapper,
(input_ids, attention_mask, token_type_ids),
strict=True,
)
torch.testing.assert_close(
traced(input_ids, attention_mask, token_type_ids),
reference,
rtol=1e-5,
atol=2e-5,
)
class CheckpointConfigTests(unittest.TestCase):
def test_rejects_changed_label_order(self):
config = {
"model_type": "bert",
"hidden_size": 384,
"num_hidden_layers": 6,
"id2label": {
str(index): label
for index, label in enumerate(reversed(LABELS))
},
}
with tempfile.TemporaryDirectory() as temp:
model_dir = Path(temp)
(model_dir / "config.json").write_text(
json.dumps(config),
encoding="utf-8",
)
with self.assertRaisesRegex(DataError, "label order"):
convert_coreml._checkpoint_config(model_dir)
if __name__ == "__main__":
unittest.main()
+23
View File
@@ -9,6 +9,29 @@ sys.path.insert(0, str(MODULE_DIR))
import eval as purpose_eval
class CoreMLComputeUnitTests(unittest.TestCase):
class CoreMLTools:
class ComputeUnit:
ALL = "all-value"
CPU_ONLY = "cpu-value"
CPU_AND_GPU = "gpu-value"
CPU_AND_NE = "ne-value"
def test_maps_cli_compute_policies(self):
expected = {
"all": "all-value",
"cpu-only": "cpu-value",
"cpu-and-gpu": "gpu-value",
"cpu-and-ne": "ne-value",
}
for requested, value in expected.items():
with self.subTest(requested=requested):
self.assertEqual(
value,
purpose_eval._coreml_compute_unit(self.CoreMLTools, requested),
)
class TierDriftTests(unittest.TestCase):
def test_current_routing_matrix_bounds_every_label_pair(self):
records = [{"prompt": f"prompt {index}"} for index in range(8 * 8)]
+26
View File
@@ -0,0 +1,26 @@
import sys
import unittest
from pathlib import Path
MODULE_DIR = Path(__file__).resolve().parents[1]
sys.path.insert(0, str(MODULE_DIR))
import inspect_coreml
class CoreMLDeviceCategoryTests(unittest.TestCase):
def test_classifies_compute_device_types(self):
NeuralEngineDevice = type("MLNeuralEngineComputeDevice", (), {})
GPUDevice = type("MLGPUComputeDevice", (), {})
CPUDevice = type("MLCPUComputeDevice", (), {})
self.assertEqual(
"neuralEngine",
inspect_coreml._device_category(NeuralEngineDevice()),
)
self.assertEqual("gpu", inspect_coreml._device_category(GPUDevice()))
self.assertEqual("cpu", inspect_coreml._device_category(CPUDevice()))
if __name__ == "__main__":
unittest.main()