Merge nucleic/sleek-ember-seal-uady into dev
This commit is contained in:
@@ -19,6 +19,13 @@ placement.
|
||||
|
||||
## macOS
|
||||
|
||||
- Generate the ML Program with `convert_coreml.py`, then run the gated `eval.py
|
||||
--coreml-model ... --coreml-compute-units cpu-and-ne --compare-onnx ...` command from
|
||||
the README. Preserve the conversion manifest and frozen report with the model metrics.
|
||||
- Run `inspect_coreml.py` with `--compute-units cpu-and-ne`; preserve its full operation
|
||||
report and record both `neuralEngineOperationShare` and
|
||||
`neuralEngineEstimatedCostShare`. A gated evaluation without the ONNX comparison is
|
||||
invalid.
|
||||
- Run the release Core ML artifact once with `.cpuAndNeuralEngine`, recording the
|
||||
`MLComputePlan` ANE operation share and the artifact's required floor.
|
||||
- Repeat with a diagnostic CPU-only configuration on the same Mac and power source.
|
||||
|
||||
@@ -196,6 +196,59 @@ routing-tier-drift gates pass. This is the current accuracy-qualified shipping c
|
||||
latency and energy/residency still require measurement on the target Apple and Windows
|
||||
accelerator runtimes.
|
||||
|
||||
### Convert and validate Core ML
|
||||
|
||||
Core ML Tools no longer maintains the legacy ONNX converter, so the Apple artifact is
|
||||
converted directly from the selected Hugging Face checkpoint. `convert_coreml.py` uses a
|
||||
fixed-shape export-only BERT forward to avoid dynamic Transformers masking helpers, checks
|
||||
that forward against Transformers before conversion, writes an ML Program package, and
|
||||
records hashes for every package file.
|
||||
|
||||
Install the pinned converter in the macOS environment and create the package:
|
||||
|
||||
```bash
|
||||
ml/purpose-classifier/venv/bin/python -m pip install \
|
||||
-r ml/purpose-classifier/requirements-coreml.txt
|
||||
ml/purpose-classifier/venv/bin/python ml/purpose-classifier/convert_coreml.py \
|
||||
--model-dir \
|
||||
ml/purpose-classifier/outputs/purpose-lite-v1-distilled-qat-mlx-4e/model \
|
||||
--output \
|
||||
ml/purpose-classifier/outputs/purpose-lite-v1-distilled-qat-mlx-4e/coreml/purpose-lite-v1-fp16.mlpackage \
|
||||
--overwrite-output
|
||||
```
|
||||
|
||||
Run the frozen gate with CPU+Neural Engine placement and compare labels directly with the
|
||||
accepted int8 ONNX artifact. Gated Core ML evaluation fails closed without
|
||||
`--compare-onnx`, and requires at least 99.5% scorable label agreement:
|
||||
|
||||
```bash
|
||||
ml/purpose-classifier/venv/bin/python ml/purpose-classifier/eval.py \
|
||||
--model-dir \
|
||||
ml/purpose-classifier/outputs/purpose-lite-v1-distilled-qat-mlx-4e/model \
|
||||
--calibration \
|
||||
ml/purpose-classifier/outputs/purpose-lite-v1-distilled-qat-mlx-4e/calibration.json \
|
||||
--coreml-model \
|
||||
ml/purpose-classifier/outputs/purpose-lite-v1-distilled-qat-mlx-4e/coreml/purpose-lite-v1-fp16.mlpackage \
|
||||
--coreml-compute-units cpu-and-ne \
|
||||
--compare-onnx \
|
||||
ml/purpose-classifier/outputs/purpose-lite-v1-distilled-qat-mlx-4e/export/purpose-lite-v1-int8-qdq.onnx \
|
||||
--report \
|
||||
ml/purpose-classifier/outputs/purpose-lite-v1-distilled-qat-mlx-4e/coreml/ane-frozen-eval.json
|
||||
```
|
||||
|
||||
Record the compute plan separately; this reports both operation-count and estimated-cost
|
||||
ANE shares. Repeat evaluation with `--coreml-compute-units cpu-only --no-gate` before the
|
||||
energy comparison in `ENERGY_AND_RESIDENCY.md`:
|
||||
|
||||
```bash
|
||||
ml/purpose-classifier/venv/bin/python ml/purpose-classifier/inspect_coreml.py \
|
||||
--model \
|
||||
ml/purpose-classifier/outputs/purpose-lite-v1-distilled-qat-mlx-4e/coreml/purpose-lite-v1-fp16.mlpackage \
|
||||
--compute-units cpu-and-ne \
|
||||
--report \
|
||||
ml/purpose-classifier/outputs/purpose-lite-v1-distilled-qat-mlx-4e/coreml/ane-compute-plan.json
|
||||
```
|
||||
|
||||
For a wiring smoke test, use a small deterministic prefix:
|
||||
|
||||
```bash
|
||||
|
||||
@@ -0,0 +1,354 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Convert the selected purpose-lite checkpoint to a fixed-shape Core ML package."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import hashlib
|
||||
import json
|
||||
import math
|
||||
import shutil
|
||||
import sys
|
||||
from pathlib import Path
|
||||
from typing import Any, Sequence
|
||||
|
||||
import numpy as np
|
||||
import torch
|
||||
import torch.nn.functional as functional
|
||||
|
||||
from purpose_data import LABELS, DataError, load_jsonl, write_json
|
||||
from train import MAX_LENGTH, encode_fixed_shape
|
||||
|
||||
|
||||
SCRIPT_DIR = Path(__file__).resolve().parent
|
||||
DEFAULT_MODEL_DIR = (
|
||||
SCRIPT_DIR
|
||||
/ "outputs"
|
||||
/ "purpose-lite-v1-distilled-qat-mlx-4e"
|
||||
/ "model"
|
||||
)
|
||||
DEFAULT_VALIDATION = SCRIPT_DIR / ".artifacts" / "dataset-v1" / "validation.jsonl"
|
||||
DEFAULT_OUTPUT = (
|
||||
SCRIPT_DIR
|
||||
/ "outputs"
|
||||
/ "purpose-lite-v1-distilled-qat-mlx-4e"
|
||||
/ "coreml"
|
||||
/ "purpose-lite-v1-fp16.mlpackage"
|
||||
)
|
||||
COREMLTOOLS_VERSION = "9.0"
|
||||
|
||||
|
||||
class FixedShapeBertForCoreML(torch.nn.Module):
|
||||
"""Conversion-only BERT forward without Transformers' dynamic mask helpers."""
|
||||
|
||||
def __init__(self, model: Any) -> None:
|
||||
super().__init__()
|
||||
self.model = model
|
||||
config = model.config
|
||||
self.num_heads = int(config.num_attention_heads)
|
||||
self.head_size = int(config.hidden_size) // self.num_heads
|
||||
if int(config.hidden_size) % self.num_heads:
|
||||
raise DataError("BERT hidden size must be divisible by attention heads")
|
||||
if int(config.max_position_embeddings) < MAX_LENGTH:
|
||||
raise DataError("BERT checkpoint cannot represent the fixed 128-token input")
|
||||
self.register_buffer(
|
||||
"fixed_position_ids",
|
||||
torch.arange(MAX_LENGTH, dtype=torch.int64).reshape(1, MAX_LENGTH),
|
||||
persistent=False,
|
||||
)
|
||||
|
||||
def forward(
|
||||
self,
|
||||
input_ids: Any,
|
||||
attention_mask: Any,
|
||||
token_type_ids: Any,
|
||||
) -> Any:
|
||||
bert = self.model.bert
|
||||
input_ids = input_ids.to(torch.int64)
|
||||
token_type_ids = token_type_ids.to(torch.int64)
|
||||
value = (
|
||||
bert.embeddings.word_embeddings(input_ids)
|
||||
+ bert.embeddings.position_embeddings(self.fixed_position_ids)
|
||||
+ bert.embeddings.token_type_embeddings(token_type_ids)
|
||||
)
|
||||
value = bert.embeddings.LayerNorm(value)
|
||||
zero = torch.zeros((), dtype=value.dtype, device=value.device)
|
||||
hidden = torch.full((), -10000.0, dtype=value.dtype, device=value.device)
|
||||
additive_mask = torch.where(
|
||||
attention_mask[:, None, None, :] != 0,
|
||||
zero,
|
||||
hidden,
|
||||
)
|
||||
|
||||
for layer in bert.encoder.layer:
|
||||
self_attention = layer.attention.self
|
||||
|
||||
def split_heads(projected: Any) -> Any:
|
||||
return projected.reshape(
|
||||
1,
|
||||
MAX_LENGTH,
|
||||
self.num_heads,
|
||||
self.head_size,
|
||||
).permute(0, 2, 1, 3)
|
||||
|
||||
queries = split_heads(self_attention.query(value))
|
||||
keys = split_heads(self_attention.key(value))
|
||||
values = split_heads(self_attention.value(value))
|
||||
scores = torch.matmul(queries, keys.transpose(-1, -2)) / math.sqrt(
|
||||
self.head_size
|
||||
)
|
||||
probabilities = torch.softmax(scores + additive_mask, dim=-1)
|
||||
context = torch.matmul(probabilities, values)
|
||||
context = context.permute(0, 2, 1, 3).reshape(
|
||||
1,
|
||||
MAX_LENGTH,
|
||||
-1,
|
||||
)
|
||||
attention_output = layer.attention.output.dense(context)
|
||||
value = layer.attention.output.LayerNorm(value + attention_output)
|
||||
intermediate = functional.gelu(
|
||||
layer.intermediate.dense(value),
|
||||
approximate="none",
|
||||
)
|
||||
value = layer.output.LayerNorm(
|
||||
value + layer.output.dense(intermediate)
|
||||
)
|
||||
|
||||
pooled = torch.tanh(bert.pooler.dense(value[:, 0]))
|
||||
return self.model.classifier(pooled)
|
||||
|
||||
|
||||
def _checkpoint_config(model_dir: Path) -> dict[str, Any]:
|
||||
config_path = model_dir / "config.json"
|
||||
try:
|
||||
config = json.loads(config_path.read_text(encoding="utf-8"))
|
||||
except (OSError, UnicodeError, json.JSONDecodeError) as exc:
|
||||
raise DataError(f"{config_path}: cannot load model config: {exc}") from exc
|
||||
configured_labels = [
|
||||
config.get("id2label", {}).get(
|
||||
str(index),
|
||||
config.get("id2label", {}).get(index),
|
||||
)
|
||||
for index in range(len(LABELS))
|
||||
]
|
||||
if (
|
||||
config.get("model_type") != "bert"
|
||||
or config.get("hidden_size") != 384
|
||||
or config.get("num_hidden_layers") != 6
|
||||
):
|
||||
raise DataError("Core ML conversion requires the purpose-lite BERT architecture")
|
||||
if configured_labels != list(LABELS):
|
||||
raise DataError("Core ML checkpoint label order does not match purpose-lite")
|
||||
return config
|
||||
|
||||
|
||||
def _sha256(path: Path) -> str:
|
||||
digest = hashlib.sha256()
|
||||
with path.open("rb") as handle:
|
||||
for chunk in iter(lambda: handle.read(1024 * 1024), b""):
|
||||
digest.update(chunk)
|
||||
return digest.hexdigest()
|
||||
|
||||
|
||||
def _package_manifest(package: Path) -> dict[str, Any]:
|
||||
files = []
|
||||
for path in sorted(item for item in package.rglob("*") if item.is_file()):
|
||||
files.append(
|
||||
{
|
||||
"path": str(path.relative_to(package)),
|
||||
"bytes": path.stat().st_size,
|
||||
"sha256": _sha256(path),
|
||||
}
|
||||
)
|
||||
return {
|
||||
"schemaVersion": 1,
|
||||
"modelVersion": "purpose-lite-v1",
|
||||
"package": package.name,
|
||||
"bytes": sum(item["bytes"] for item in files),
|
||||
"files": files,
|
||||
}
|
||||
|
||||
|
||||
def _verify_wrapper_parity(
|
||||
torch_model: Any,
|
||||
wrapper: FixedShapeBertForCoreML,
|
||||
tokenizer: Any,
|
||||
records: Sequence[dict[str, Any]],
|
||||
) -> float:
|
||||
encoded = encode_fixed_shape(
|
||||
tokenizer,
|
||||
[record["prompt"] for record in records],
|
||||
torch,
|
||||
)
|
||||
maximum_error = 0.0
|
||||
with torch.inference_mode():
|
||||
for index in range(len(records)):
|
||||
item = {key: value[index : index + 1] for key, value in encoded.items()}
|
||||
reference = torch_model(**item).logits
|
||||
candidate = wrapper(
|
||||
item["input_ids"].to(torch.int32),
|
||||
item["attention_mask"].to(torch.int32),
|
||||
item.get("token_type_ids", torch.zeros_like(item["input_ids"])).to(
|
||||
torch.int32
|
||||
),
|
||||
)
|
||||
error = float(torch.max(torch.abs(reference - candidate)).item())
|
||||
maximum_error = max(maximum_error, error)
|
||||
torch.testing.assert_close(candidate, reference, rtol=1e-5, atol=2e-5)
|
||||
if int(candidate.argmax(dim=-1).item()) != int(
|
||||
reference.argmax(dim=-1).item()
|
||||
):
|
||||
raise DataError("conversion wrapper changed a parity prediction")
|
||||
return maximum_error
|
||||
|
||||
|
||||
def convert(args: argparse.Namespace) -> dict[str, Any]:
|
||||
_checkpoint_config(args.model_dir)
|
||||
checkpoint = args.model_dir / "model.safetensors"
|
||||
if not checkpoint.is_file():
|
||||
raise DataError(f"{checkpoint}: checkpoint is missing")
|
||||
if args.output.suffix != ".mlpackage":
|
||||
raise DataError("Core ML output must end in .mlpackage")
|
||||
if args.output.exists():
|
||||
if not args.overwrite_output:
|
||||
raise DataError(
|
||||
f"{args.output}: output exists; pass --overwrite-output intentionally"
|
||||
)
|
||||
if args.output.is_dir():
|
||||
shutil.rmtree(args.output)
|
||||
else:
|
||||
args.output.unlink()
|
||||
args.output.parent.mkdir(parents=True, exist_ok=True)
|
||||
|
||||
try:
|
||||
import coremltools as ct
|
||||
from transformers import AutoModelForSequenceClassification, AutoTokenizer
|
||||
except ImportError as exc:
|
||||
raise DataError(
|
||||
"Core ML conversion requires requirements-coreml.txt on macOS"
|
||||
) from exc
|
||||
if ct.__version__ != COREMLTOOLS_VERSION:
|
||||
raise DataError(
|
||||
f"expected coremltools {COREMLTOOLS_VERSION}, found {ct.__version__}"
|
||||
)
|
||||
|
||||
model = AutoModelForSequenceClassification.from_pretrained(
|
||||
args.model_dir,
|
||||
local_files_only=True,
|
||||
).eval()
|
||||
tokenizer = AutoTokenizer.from_pretrained(args.model_dir, local_files_only=True)
|
||||
wrapper = FixedShapeBertForCoreML(model).eval()
|
||||
validation = load_jsonl(args.validation)
|
||||
if len(validation) < args.parity_records:
|
||||
raise DataError("validation split is smaller than --parity-records")
|
||||
maximum_error = _verify_wrapper_parity(
|
||||
model,
|
||||
wrapper,
|
||||
tokenizer,
|
||||
validation[: args.parity_records],
|
||||
)
|
||||
print(f"conversion-wrapper parity: max_abs_error={maximum_error:.3g}")
|
||||
|
||||
example = (
|
||||
torch.zeros((1, MAX_LENGTH), dtype=torch.int32),
|
||||
torch.ones((1, MAX_LENGTH), dtype=torch.int32),
|
||||
torch.zeros((1, MAX_LENGTH), dtype=torch.int32),
|
||||
)
|
||||
with torch.inference_mode():
|
||||
traced = torch.jit.trace(wrapper, example, strict=True)
|
||||
traced = torch.jit.freeze(traced)
|
||||
deployment_target = getattr(ct.target, args.minimum_deployment_target, None)
|
||||
if deployment_target is None:
|
||||
raise DataError(
|
||||
f"coremltools does not support {args.minimum_deployment_target}"
|
||||
)
|
||||
coreml_model = ct.convert(
|
||||
traced,
|
||||
convert_to="mlprogram",
|
||||
minimum_deployment_target=deployment_target,
|
||||
compute_precision=ct.precision.FLOAT16,
|
||||
inputs=[
|
||||
ct.TensorType(
|
||||
name="input_ids",
|
||||
shape=(1, MAX_LENGTH),
|
||||
dtype=np.int32,
|
||||
),
|
||||
ct.TensorType(
|
||||
name="attention_mask",
|
||||
shape=(1, MAX_LENGTH),
|
||||
dtype=np.int32,
|
||||
),
|
||||
ct.TensorType(
|
||||
name="token_type_ids",
|
||||
shape=(1, MAX_LENGTH),
|
||||
dtype=np.int32,
|
||||
),
|
||||
],
|
||||
outputs=[ct.TensorType(name="logits", dtype=np.float32)],
|
||||
)
|
||||
coreml_model.author = "Nucleic"
|
||||
coreml_model.short_description = "purpose-lite-v1 prompt classifier"
|
||||
coreml_model.version = "purpose-lite-v1"
|
||||
coreml_model.user_defined_metadata["com.nucleic.model.version"] = (
|
||||
"purpose-lite-v1"
|
||||
)
|
||||
coreml_model.user_defined_metadata["com.nucleic.model.labels"] = json.dumps(
|
||||
list(LABELS),
|
||||
separators=(",", ":"),
|
||||
)
|
||||
coreml_model.user_defined_metadata["com.nucleic.model.sourceSha256"] = _sha256(
|
||||
checkpoint
|
||||
)
|
||||
coreml_model.user_defined_metadata["com.nucleic.model.fixedShape"] = "1x128"
|
||||
coreml_model.user_defined_metadata["com.nucleic.model.minimumDeploymentTarget"] = (
|
||||
args.minimum_deployment_target
|
||||
)
|
||||
coreml_model.save(str(args.output))
|
||||
|
||||
manifest = _package_manifest(args.output)
|
||||
manifest.update(
|
||||
{
|
||||
"sourceCheckpoint": str(checkpoint),
|
||||
"sourceCheckpointSha256": _sha256(checkpoint),
|
||||
"coremltoolsVersion": ct.__version__,
|
||||
"minimumDeploymentTarget": args.minimum_deployment_target,
|
||||
"computePrecision": "float16",
|
||||
"wrapperMaximumAbsoluteError": maximum_error,
|
||||
}
|
||||
)
|
||||
manifest_path = args.output.with_name(f"{args.output.stem}-manifest.json")
|
||||
write_json(manifest_path, manifest)
|
||||
print(f"Core ML package: {args.output} ({manifest['bytes']} bytes)")
|
||||
return manifest
|
||||
|
||||
|
||||
def build_parser() -> argparse.ArgumentParser:
|
||||
parser = argparse.ArgumentParser(description=__doc__)
|
||||
parser.add_argument("--model-dir", type=Path, default=DEFAULT_MODEL_DIR)
|
||||
parser.add_argument("--validation", type=Path, default=DEFAULT_VALIDATION)
|
||||
parser.add_argument("--output", type=Path, default=DEFAULT_OUTPUT)
|
||||
parser.add_argument("--parity-records", type=int, default=16)
|
||||
parser.add_argument(
|
||||
"--minimum-deployment-target",
|
||||
default="macOS15",
|
||||
choices=("macOS15", "macOS26"),
|
||||
)
|
||||
parser.add_argument("--overwrite-output", action="store_true")
|
||||
return parser
|
||||
|
||||
|
||||
def main(argv: Sequence[str] | None = None) -> int:
|
||||
parser = build_parser()
|
||||
args = parser.parse_args(argv)
|
||||
if args.parity_records <= 0:
|
||||
parser.error("--parity-records must be positive")
|
||||
try:
|
||||
convert(args)
|
||||
except (AssertionError, DataError, OSError, RuntimeError, ValueError) as exc:
|
||||
print(f"error: {exc}", file=sys.stderr)
|
||||
return 1
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
raise SystemExit(main())
|
||||
@@ -13,6 +13,8 @@ from collections import Counter
|
||||
from pathlib import Path
|
||||
from typing import Any, Sequence
|
||||
|
||||
import numpy as np
|
||||
|
||||
from purpose_data import (
|
||||
HARD_SLICES,
|
||||
LABELS,
|
||||
@@ -100,6 +102,16 @@ def _synchronize(torch: Any, device: Any) -> None:
|
||||
torch.mps.synchronize()
|
||||
|
||||
|
||||
def _coreml_compute_unit(coremltools: Any, requested: str) -> Any:
|
||||
values = {
|
||||
"all": coremltools.ComputeUnit.ALL,
|
||||
"cpu-only": coremltools.ComputeUnit.CPU_ONLY,
|
||||
"cpu-and-gpu": coremltools.ComputeUnit.CPU_AND_GPU,
|
||||
"cpu-and-ne": coremltools.ComputeUnit.CPU_AND_NE,
|
||||
}
|
||||
return values[requested]
|
||||
|
||||
|
||||
def routing_tier_drift(
|
||||
records: Sequence[dict[str, Any]],
|
||||
actual: Sequence[int],
|
||||
@@ -245,8 +257,62 @@ def evaluate(args: argparse.Namespace) -> dict[str, Any]:
|
||||
label_to_id = {label: index for index, label in enumerate(LABELS)}
|
||||
tokenizer = AutoTokenizer.from_pretrained(args.model_dir, local_files_only=True)
|
||||
onnx_session = None
|
||||
coreml_model = None
|
||||
reference_model = None
|
||||
if args.onnx_model is not None:
|
||||
reference_onnx_session = None
|
||||
reference_onnx_input_names: set[str] = set()
|
||||
if args.compare_pytorch and args.compare_onnx is not None:
|
||||
raise DataError("choose only one parity reference")
|
||||
if args.coreml_model is not None:
|
||||
try:
|
||||
import coremltools as ct
|
||||
except ImportError as exc:
|
||||
raise DataError(
|
||||
"Core ML evaluation requires requirements-coreml.txt on macOS"
|
||||
) from exc
|
||||
if args.device != "auto":
|
||||
raise DataError(
|
||||
"Core ML compute placement uses --coreml-compute-units, not --device"
|
||||
)
|
||||
coreml_model = ct.models.MLModel(
|
||||
str(args.coreml_model),
|
||||
compute_units=_coreml_compute_unit(ct, args.coreml_compute_units),
|
||||
)
|
||||
coreml_input_names = {
|
||||
item.name for item in coreml_model.get_spec().description.input
|
||||
}
|
||||
device = torch.device("cpu")
|
||||
model = None
|
||||
onnx_input_names = set()
|
||||
runtime_name = f"coreml-{args.coreml_compute_units}"
|
||||
if args.compare_pytorch:
|
||||
reference_model = AutoModelForSequenceClassification.from_pretrained(
|
||||
args.model_dir,
|
||||
local_files_only=True,
|
||||
).to(device)
|
||||
reference_model.eval()
|
||||
if args.compare_onnx is not None:
|
||||
try:
|
||||
import onnxruntime as ort
|
||||
except ImportError as exc:
|
||||
raise DataError(
|
||||
"Core ML↔ONNX parity requires onnxruntime"
|
||||
) from exc
|
||||
reference_options = ort.SessionOptions()
|
||||
reference_options.graph_optimization_level = (
|
||||
ort.GraphOptimizationLevel.ORT_ENABLE_ALL
|
||||
)
|
||||
reference_onnx_session = ort.InferenceSession(
|
||||
str(args.compare_onnx),
|
||||
sess_options=reference_options,
|
||||
providers=["CPUExecutionProvider"],
|
||||
)
|
||||
reference_onnx_input_names = {
|
||||
item.name for item in reference_onnx_session.get_inputs()
|
||||
}
|
||||
elif args.onnx_model is not None:
|
||||
if args.compare_onnx is not None:
|
||||
raise DataError("--compare-onnx requires --coreml-model")
|
||||
try:
|
||||
import onnxruntime as ort
|
||||
except ImportError as exc:
|
||||
@@ -273,8 +339,12 @@ def evaluate(args: argparse.Namespace) -> dict[str, Any]:
|
||||
).to(device)
|
||||
reference_model.eval()
|
||||
else:
|
||||
if args.compare_onnx is not None:
|
||||
raise DataError("--compare-onnx requires --coreml-model")
|
||||
if args.compare_pytorch:
|
||||
raise DataError("--compare-pytorch requires --onnx-model")
|
||||
raise DataError(
|
||||
"--compare-pytorch requires --onnx-model or --coreml-model"
|
||||
)
|
||||
device = _device(torch, args.device)
|
||||
model = AutoModelForSequenceClassification.from_pretrained(
|
||||
args.model_dir, local_files_only=True
|
||||
@@ -284,6 +354,14 @@ def evaluate(args: argparse.Namespace) -> dict[str, Any]:
|
||||
runtime_name = "pytorch"
|
||||
|
||||
def predict_logits(encoded: dict[str, Any]) -> Any:
|
||||
if coreml_model is not None:
|
||||
inputs = {
|
||||
key: value.numpy().astype(np.int32, copy=False)
|
||||
for key, value in encoded.items()
|
||||
if key in coreml_input_names
|
||||
}
|
||||
output = np.asarray(coreml_model.predict(inputs)["logits"])
|
||||
return torch.from_numpy(output.reshape(1, len(LABELS)))
|
||||
if onnx_session is not None:
|
||||
inputs = {
|
||||
key: value.numpy()
|
||||
@@ -300,7 +378,9 @@ def evaluate(args: argparse.Namespace) -> dict[str, Any]:
|
||||
confidences: list[str] = []
|
||||
margins: list[float] = []
|
||||
reference_predictions: list[int] = []
|
||||
inference_batch_size = 1 if onnx_session is not None else args.batch_size
|
||||
inference_batch_size = (
|
||||
1 if onnx_session is not None or coreml_model is not None else args.batch_size
|
||||
)
|
||||
with torch.inference_mode():
|
||||
for start in range(0, len(records), inference_batch_size):
|
||||
batch = records[start : start + inference_batch_size]
|
||||
@@ -315,6 +395,19 @@ def evaluate(args: argparse.Namespace) -> dict[str, Any]:
|
||||
reference_predictions.extend(
|
||||
reference_logits.argmax(dim=-1).tolist()
|
||||
)
|
||||
elif reference_onnx_session is not None:
|
||||
reference_inputs = {
|
||||
key: value.numpy()
|
||||
for key, value in encoded.items()
|
||||
if key in reference_onnx_input_names
|
||||
}
|
||||
reference_logits = reference_onnx_session.run(
|
||||
["logits"],
|
||||
reference_inputs,
|
||||
)[0]
|
||||
reference_predictions.extend(
|
||||
np.asarray(reference_logits).argmax(axis=-1).tolist()
|
||||
)
|
||||
distribution = torch.softmax(logits, dim=-1)
|
||||
top = torch.topk(distribution, k=2, dim=-1)
|
||||
batch_probabilities = top.values[:, 0].tolist()
|
||||
@@ -465,7 +558,7 @@ def evaluate(args: argparse.Namespace) -> dict[str, Any]:
|
||||
"modelVersion": calibration.get("modelVersion", args.model_dir.name),
|
||||
"device": str(device),
|
||||
"runtime": runtime_name,
|
||||
"artifact": str(args.onnx_model or args.model_dir),
|
||||
"artifact": str(args.coreml_model or args.onnx_model or args.model_dir),
|
||||
"fixedInputShape": [1, MAX_LENGTH],
|
||||
"overall": metrics,
|
||||
"scoredClassification": scored_metrics,
|
||||
@@ -508,13 +601,21 @@ def evaluate(args: argparse.Namespace) -> dict[str, Any]:
|
||||
},
|
||||
"misclassifications": misclassifications,
|
||||
}
|
||||
if reference_model is not None:
|
||||
report["pytorchParity"] = prediction_agreement(
|
||||
if reference_predictions:
|
||||
parity_name = (
|
||||
"onnxParity" if reference_onnx_session is not None else "pytorchParity"
|
||||
)
|
||||
parity = prediction_agreement(
|
||||
records,
|
||||
actual,
|
||||
reference_predictions,
|
||||
predicted,
|
||||
)
|
||||
report[parity_name] = parity
|
||||
if reference_onnx_session is not None:
|
||||
report["gates"]["onnxLabelAgreementAtLeast99_5Percent"] = (
|
||||
parity["scoredLabelAgreement"] >= 0.995
|
||||
)
|
||||
write_json(args.report, report)
|
||||
return report
|
||||
|
||||
@@ -522,15 +623,31 @@ def evaluate(args: argparse.Namespace) -> dict[str, Any]:
|
||||
def build_parser() -> argparse.ArgumentParser:
|
||||
parser = argparse.ArgumentParser(description=__doc__)
|
||||
parser.add_argument("--model-dir", type=Path, default=DEFAULT_MODEL_DIR)
|
||||
parser.add_argument(
|
||||
runtime = parser.add_mutually_exclusive_group()
|
||||
runtime.add_argument(
|
||||
"--onnx-model",
|
||||
type=Path,
|
||||
help="score a fixed-shape ONNX artifact instead of the PyTorch checkpoint",
|
||||
)
|
||||
runtime.add_argument(
|
||||
"--coreml-model",
|
||||
type=Path,
|
||||
help="score a fixed-shape Core ML package instead of the PyTorch checkpoint",
|
||||
)
|
||||
parser.add_argument(
|
||||
"--compare-pytorch",
|
||||
action="store_true",
|
||||
help="include label-level drift from --model-dir when scoring ONNX",
|
||||
help="include label-level drift from --model-dir when scoring an artifact",
|
||||
)
|
||||
parser.add_argument(
|
||||
"--compare-onnx",
|
||||
type=Path,
|
||||
help="include Core ML label-level drift from this ONNX reference",
|
||||
)
|
||||
parser.add_argument(
|
||||
"--coreml-compute-units",
|
||||
choices=("all", "cpu-only", "cpu-and-gpu", "cpu-and-ne"),
|
||||
default="cpu-and-ne",
|
||||
)
|
||||
parser.add_argument("--calibration", type=Path, default=DEFAULT_CALIBRATION)
|
||||
parser.add_argument("--test", type=Path, default=DEFAULT_TEST)
|
||||
@@ -552,6 +669,11 @@ def main(argv: Sequence[str] | None = None) -> int:
|
||||
args = parser.parse_args(argv)
|
||||
if args.batch_size <= 0 or args.latency_samples < 0:
|
||||
parser.error("batch size must be positive and latency samples non-negative")
|
||||
if args.coreml_model is not None and args.compare_onnx is None and not args.no_gate:
|
||||
parser.error(
|
||||
"gated Core ML evaluation requires --compare-onnx; use --no-gate only "
|
||||
"for diagnostics"
|
||||
)
|
||||
try:
|
||||
report = evaluate(args)
|
||||
except (DataError, OSError, ValueError) as exc:
|
||||
|
||||
@@ -0,0 +1,153 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Inspect Core ML operation placement and estimated accelerator cost share."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import platform
|
||||
import sys
|
||||
from collections import Counter, defaultdict
|
||||
from pathlib import Path
|
||||
from typing import Any, Sequence
|
||||
|
||||
from purpose_data import DataError, write_json
|
||||
|
||||
|
||||
def _device_category(device: Any) -> str:
|
||||
name = type(device).__name__.lower()
|
||||
description = str(device).lower()
|
||||
combined = f"{name} {description}"
|
||||
if "neural" in combined:
|
||||
return "neuralEngine"
|
||||
if "gpu" in combined:
|
||||
return "gpu"
|
||||
if "cpu" in combined:
|
||||
return "cpu"
|
||||
return "unknown"
|
||||
|
||||
|
||||
def _compute_unit(coremltools: Any, requested: str) -> Any:
|
||||
values = {
|
||||
"all": coremltools.ComputeUnit.ALL,
|
||||
"cpu-only": coremltools.ComputeUnit.CPU_ONLY,
|
||||
"cpu-and-gpu": coremltools.ComputeUnit.CPU_AND_GPU,
|
||||
"cpu-and-ne": coremltools.ComputeUnit.CPU_AND_NE,
|
||||
}
|
||||
return values[requested]
|
||||
|
||||
|
||||
def inspect(args: argparse.Namespace) -> dict[str, Any]:
|
||||
if not args.model.exists():
|
||||
raise DataError(f"{args.model}: Core ML model is missing")
|
||||
try:
|
||||
import coremltools as ct
|
||||
except ImportError as exc:
|
||||
raise DataError(
|
||||
"Core ML inspection requires requirements-coreml.txt on macOS"
|
||||
) from exc
|
||||
|
||||
compiled = ct.models.utils.compile_model(str(args.model))
|
||||
compute_plan = ct.models.compute_plan.MLComputePlan.load_from_path(
|
||||
path=str(compiled),
|
||||
compute_units=_compute_unit(ct, args.compute_units),
|
||||
)
|
||||
program = compute_plan.model_structure.program
|
||||
if program is None or "main" not in program.functions:
|
||||
raise DataError("Core ML package is not an ML Program with a main function")
|
||||
operations = list(program.functions["main"].block.operations)
|
||||
if not operations:
|
||||
raise DataError("Core ML compute plan contains no operations")
|
||||
|
||||
preferred_counts: Counter[str] = Counter()
|
||||
preferred_costs: dict[str, float] = defaultdict(float)
|
||||
supported_counts: Counter[str] = Counter()
|
||||
operation_reports = []
|
||||
operations_with_usage = 0
|
||||
operations_with_cost = 0
|
||||
total_cost = 0.0
|
||||
for operation in operations:
|
||||
usage = compute_plan.get_compute_device_usage_for_mlprogram_operation(
|
||||
operation
|
||||
)
|
||||
cost = compute_plan.get_estimated_cost_for_mlprogram_operation(operation)
|
||||
preferred = "unknown"
|
||||
supported: list[str] = []
|
||||
if usage is not None:
|
||||
operations_with_usage += 1
|
||||
preferred = _device_category(usage.preferred_compute_device)
|
||||
preferred_counts[preferred] += 1
|
||||
supported = sorted(
|
||||
{_device_category(device) for device in usage.supported_compute_devices}
|
||||
)
|
||||
supported_counts.update(supported)
|
||||
weight = None
|
||||
if cost is not None:
|
||||
operations_with_cost += 1
|
||||
weight = float(cost.weight)
|
||||
total_cost += weight
|
||||
preferred_costs[preferred] += weight
|
||||
operation_reports.append(
|
||||
{
|
||||
"operatorName": str(operation.operator_name),
|
||||
"preferredDevice": preferred,
|
||||
"supportedDevices": supported,
|
||||
"estimatedCostWeight": weight,
|
||||
}
|
||||
)
|
||||
|
||||
ane_operations = preferred_counts["neuralEngine"]
|
||||
ane_cost = preferred_costs["neuralEngine"]
|
||||
report = {
|
||||
"schemaVersion": 1,
|
||||
"model": str(args.model),
|
||||
"coremltoolsVersion": ct.__version__,
|
||||
"machine": platform.machine(),
|
||||
"macOS": platform.mac_ver()[0],
|
||||
"computeUnits": args.compute_units,
|
||||
"operations": len(operations),
|
||||
"operationsWithDeviceUsage": operations_with_usage,
|
||||
"operationsWithEstimatedCost": operations_with_cost,
|
||||
"preferredOperationCounts": dict(sorted(preferred_counts.items())),
|
||||
"supportedOperationCounts": dict(sorted(supported_counts.items())),
|
||||
"preferredEstimatedCosts": dict(sorted(preferred_costs.items())),
|
||||
"neuralEngineOperationShare": (
|
||||
ane_operations / operations_with_usage if operations_with_usage else 0.0
|
||||
),
|
||||
"neuralEngineEstimatedCostShare": (
|
||||
ane_cost / total_cost if total_cost else 0.0
|
||||
),
|
||||
"operationDetails": operation_reports,
|
||||
}
|
||||
write_json(args.report, report)
|
||||
print(
|
||||
"Core ML placement: "
|
||||
f"ANE operations={report['neuralEngineOperationShare']:.2%} "
|
||||
f"ANE estimated cost={report['neuralEngineEstimatedCostShare']:.2%}"
|
||||
)
|
||||
return report
|
||||
|
||||
|
||||
def build_parser() -> argparse.ArgumentParser:
|
||||
parser = argparse.ArgumentParser(description=__doc__)
|
||||
parser.add_argument("--model", type=Path, required=True)
|
||||
parser.add_argument("--report", type=Path, required=True)
|
||||
parser.add_argument(
|
||||
"--compute-units",
|
||||
choices=("all", "cpu-only", "cpu-and-gpu", "cpu-and-ne"),
|
||||
default="cpu-and-ne",
|
||||
)
|
||||
return parser
|
||||
|
||||
|
||||
def main(argv: Sequence[str] | None = None) -> int:
|
||||
args = build_parser().parse_args(argv)
|
||||
try:
|
||||
inspect(args)
|
||||
except (DataError, OSError, RuntimeError, ValueError) as exc:
|
||||
print(f"error: {exc}", file=sys.stderr)
|
||||
return 1
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
raise SystemExit(main())
|
||||
@@ -0,0 +1,2 @@
|
||||
-r requirements.txt
|
||||
coremltools==9.0
|
||||
@@ -0,0 +1,84 @@
|
||||
import json
|
||||
import sys
|
||||
import tempfile
|
||||
import unittest
|
||||
from pathlib import Path
|
||||
|
||||
import torch
|
||||
from transformers import BertConfig, BertForSequenceClassification
|
||||
|
||||
|
||||
MODULE_DIR = Path(__file__).resolve().parents[1]
|
||||
sys.path.insert(0, str(MODULE_DIR))
|
||||
|
||||
import convert_coreml
|
||||
from purpose_data import LABELS, DataError
|
||||
|
||||
|
||||
class FixedShapeBertForCoreMLTests(unittest.TestCase):
|
||||
def test_conversion_forward_matches_transformers(self):
|
||||
torch.manual_seed(7)
|
||||
config = BertConfig(
|
||||
vocab_size=64,
|
||||
hidden_size=16,
|
||||
num_hidden_layers=1,
|
||||
num_attention_heads=4,
|
||||
intermediate_size=32,
|
||||
max_position_embeddings=128,
|
||||
type_vocab_size=2,
|
||||
hidden_dropout_prob=0.0,
|
||||
attention_probs_dropout_prob=0.0,
|
||||
num_labels=len(LABELS),
|
||||
)
|
||||
model = BertForSequenceClassification(config).eval()
|
||||
wrapper = convert_coreml.FixedShapeBertForCoreML(model).eval()
|
||||
input_ids = torch.randint(0, config.vocab_size, (1, 128), dtype=torch.int32)
|
||||
attention_mask = torch.zeros((1, 128), dtype=torch.int32)
|
||||
attention_mask[:, :83] = 1
|
||||
token_type_ids = torch.zeros((1, 128), dtype=torch.int32)
|
||||
token_type_ids[:, 43:83] = 1
|
||||
with torch.inference_mode():
|
||||
reference = model(
|
||||
input_ids=input_ids.long(),
|
||||
attention_mask=attention_mask.long(),
|
||||
token_type_ids=token_type_ids.long(),
|
||||
).logits
|
||||
candidate = wrapper(input_ids, attention_mask, token_type_ids)
|
||||
torch.testing.assert_close(candidate, reference, rtol=1e-5, atol=2e-5)
|
||||
|
||||
traced = torch.jit.trace(
|
||||
wrapper,
|
||||
(input_ids, attention_mask, token_type_ids),
|
||||
strict=True,
|
||||
)
|
||||
torch.testing.assert_close(
|
||||
traced(input_ids, attention_mask, token_type_ids),
|
||||
reference,
|
||||
rtol=1e-5,
|
||||
atol=2e-5,
|
||||
)
|
||||
|
||||
|
||||
class CheckpointConfigTests(unittest.TestCase):
|
||||
def test_rejects_changed_label_order(self):
|
||||
config = {
|
||||
"model_type": "bert",
|
||||
"hidden_size": 384,
|
||||
"num_hidden_layers": 6,
|
||||
"id2label": {
|
||||
str(index): label
|
||||
for index, label in enumerate(reversed(LABELS))
|
||||
},
|
||||
}
|
||||
with tempfile.TemporaryDirectory() as temp:
|
||||
model_dir = Path(temp)
|
||||
(model_dir / "config.json").write_text(
|
||||
json.dumps(config),
|
||||
encoding="utf-8",
|
||||
)
|
||||
with self.assertRaisesRegex(DataError, "label order"):
|
||||
convert_coreml._checkpoint_config(model_dir)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
unittest.main()
|
||||
@@ -9,6 +9,29 @@ sys.path.insert(0, str(MODULE_DIR))
|
||||
import eval as purpose_eval
|
||||
|
||||
|
||||
class CoreMLComputeUnitTests(unittest.TestCase):
|
||||
class CoreMLTools:
|
||||
class ComputeUnit:
|
||||
ALL = "all-value"
|
||||
CPU_ONLY = "cpu-value"
|
||||
CPU_AND_GPU = "gpu-value"
|
||||
CPU_AND_NE = "ne-value"
|
||||
|
||||
def test_maps_cli_compute_policies(self):
|
||||
expected = {
|
||||
"all": "all-value",
|
||||
"cpu-only": "cpu-value",
|
||||
"cpu-and-gpu": "gpu-value",
|
||||
"cpu-and-ne": "ne-value",
|
||||
}
|
||||
for requested, value in expected.items():
|
||||
with self.subTest(requested=requested):
|
||||
self.assertEqual(
|
||||
value,
|
||||
purpose_eval._coreml_compute_unit(self.CoreMLTools, requested),
|
||||
)
|
||||
|
||||
|
||||
class TierDriftTests(unittest.TestCase):
|
||||
def test_current_routing_matrix_bounds_every_label_pair(self):
|
||||
records = [{"prompt": f"prompt {index}"} for index in range(8 * 8)]
|
||||
|
||||
@@ -0,0 +1,26 @@
|
||||
import sys
|
||||
import unittest
|
||||
from pathlib import Path
|
||||
|
||||
|
||||
MODULE_DIR = Path(__file__).resolve().parents[1]
|
||||
sys.path.insert(0, str(MODULE_DIR))
|
||||
|
||||
import inspect_coreml
|
||||
|
||||
|
||||
class CoreMLDeviceCategoryTests(unittest.TestCase):
|
||||
def test_classifies_compute_device_types(self):
|
||||
NeuralEngineDevice = type("MLNeuralEngineComputeDevice", (), {})
|
||||
GPUDevice = type("MLGPUComputeDevice", (), {})
|
||||
CPUDevice = type("MLCPUComputeDevice", (), {})
|
||||
self.assertEqual(
|
||||
"neuralEngine",
|
||||
inspect_coreml._device_category(NeuralEngineDevice()),
|
||||
)
|
||||
self.assertEqual("gpu", inspect_coreml._device_category(GPUDevice()))
|
||||
self.assertEqual("cpu", inspect_coreml._device_category(CPUDevice()))
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
unittest.main()
|
||||
Reference in New Issue
Block a user