Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
13 changes: 13 additions & 0 deletions docs/commands/perf.md
Original file line number Diff line number Diff line change
Expand Up @@ -421,6 +421,19 @@ not supported.
- **Random inputs do not represent real data distributions.** Latency numbers are accurate, but memory access patterns may differ from production because the generated tensors are uniform random values. For memory-bandwidth-sensitive models this can understate real-world latency.
- **Cross-device comparison.** To compare performance across devices, run `winml perf` separately with different `--device` values and compare the resulting JSON reports.

## Concrete input shapes for CGC

For local ONNX models, Runtime CGC and WinMLCG receive concrete named input
dimensions before compilation. With --input-data, shapes come from NPZ headers
without allocating input tensors. Otherwise, the existing --shape-config and
batch-size resolution rules apply. Rank, static axes, positive dimensions and
shared symbolic names must agree. The source ONNX is not rewritten.

Runtime CGC requires a Runtime compiler that supports symbolic-dimension
options. Anonymous dynamic axes are rejected rather than guessed, and these
overrides do not resolve internal data-dependent shapes. Input payload loading
and dtype conversion still occur at the normal input-allocation boundary.

## See also

- [winml eval](eval.md) — measure accuracy after benchmarking
Expand Down
87 changes: 87 additions & 0 deletions src/winml/modelkit/commands/perf.py
Original file line number Diff line number Diff line change
Expand Up @@ -907,6 +907,70 @@ def load_input_data(
return _load_input_data(path, io_config)


def _runtime_input_shapes(
model_path: Path, input_data: Path | None, shape_config: dict | None, batch_size: int
) -> tuple[dict[str, tuple[int, ...]], dict[str, int]]:
"""Resolve source-ONNX shapes before compilation without allocating input tensors."""
import zipfile

from ..onnx import get_io_config
from ..session.runtime_session import _symbolic_dimensions_for_inputs

io_config = get_io_config(model_path)
if not any(dim is None for shape in io_config["input_shapes"] for dim in shape):
return {}, {}
shapes: dict[str, tuple[int, ...]] = {}
if input_data is not None:
if input_data.suffix.lower() != ".npz":
raise click.UsageError("--input-data must be a named .npz archive.")
try:
with zipfile.ZipFile(input_data) as archive:
members = archive.namelist()
expected = [name + ".npy" for name in io_config["input_names"]]
if len(members) != len(expected) or set(members) != set(expected):
raise ValueError("archive keys must exactly match ONNX input names")
for name in io_config["input_names"]:
with archive.open(name + ".npy") as stream:
version = np.lib.format.read_magic(stream)
# NumPy has no public v3 header reader. Use the same
# version-aware, size-limited reader as np.load, without
# reading or allocating the array payload.
shape, _, dtype = np.lib.format._read_array_header( # type: ignore[attr-defined]
stream, version
)
if dtype.hasobject:
raise ValueError("object input arrays are unsupported")
shapes[name] = shape
except (OSError, ValueError, EOFError, zipfile.BadZipFile) as exc:
raise click.UsageError(f"Cannot read concrete --input-data shapes: {exc}") from exc
else:
for name, shape, symbolic in zip(
io_config["input_names"],
io_config["input_shapes"],
io_config["input_symbolic_shapes"],
strict=True,
):
full_shape = (shape_config or {}).get(name)
if isinstance(full_shape, (list, tuple)):
# Preserve original values for strict integer/static-axis validation.
shapes[name] = tuple(full_shape)
else:
shapes[name] = _resolve_shape(
shape, name, batch_size, symbolic_shape=symbolic, shape_config=shape_config
)
return shapes, _symbolic_dimensions_for_inputs(io_config, shapes)


def _ort_options_for_dimensions(dimensions: dict[str, int]) -> Any:
"""Create fresh ORT options with concrete dimensions before eager session creation."""
import onnxruntime as ort

options = ort.SessionOptions()
for name, extent in dimensions.items():
options.add_free_dimension_override_by_name(name, extent)
return options


def effective_batch_size(
inputs: dict[str, np.ndarray],
input_names: list[str],
Expand Down Expand Up @@ -1361,13 +1425,36 @@ def _load_model(self) -> None:
}

if is_onnx:
runtime_shapes: dict[str, tuple[int, ...]] = {}
runtime_cgc = self.config.runtime == "winml-runtime" and self._runtime_backend == "cgc"
ort_cgc = (
self.config.runtime == "winml-ort"
and self._ep_device.device.ep_name == "WinMLCGExecutionProvider"
)
if runtime_cgc or ort_cgc:
runtime_shapes, dimensions = _runtime_input_shapes(
model_path,
self.config.input_data,
self.config.shape_config,
self.config.batch_size,
)
if ort_cgc and dimensions:
common_kwargs["session_options"] = lambda: _ort_options_for_dimensions(
dimensions
)
with suppress_native_warnings(enabled=True):
self._model = WinMLAutoModel.from_onnx(
onnx_path=model_path,
skip_build=self.config.skip_build,
compile_provider_options=self.config.compile_ep_options,
**common_kwargs,
)
if runtime_cgc and runtime_shapes:
from ..session.runtime_session import WinMLRuntimeSession

# Keep the lazy import visible to CodeQL as well as type checkers.
runtime_session = cast(WinMLRuntimeSession, self._single._session) # noqa: TC006
runtime_session.set_input_shapes(runtime_shapes)
elif is_mlir:
with suppress_native_warnings(enabled=True):
self._model = WinMLAutoModel.from_mlir(
Expand Down
76 changes: 75 additions & 1 deletion src/winml/modelkit/session/runtime_session.py
Original file line number Diff line number Diff line change
Expand Up @@ -30,6 +30,7 @@
import ctypes
import json
import logging
import operator
import threading
from contextlib import contextmanager
from dataclasses import dataclass
Expand Down Expand Up @@ -246,6 +247,51 @@ def _stage_schema(wr: Any, stage: Any) -> Any:
return schema_type(interface, stage)


def _symbolic_dimensions_for_inputs(
io_config: dict[str, Any], input_shapes: Mapping[str, Any]
) -> dict[str, int]:
"""Validate concrete input shapes and resolve shared ONNX dimension names."""
names = io_config["input_names"]
if set(input_shapes) != set(names):
raise click.ClickException("Concrete input shapes must match the ONNX input names exactly.")
overrides: dict[str, int] = {}
for name, declared, symbolic in zip(
names, io_config["input_shapes"], io_config["input_symbolic_shapes"], strict=True
):
actual = input_shapes[name]
if len(actual) != len(declared):
raise click.ClickException(f"Input {name!r} rank does not match the ONNX schema.")
for axis, (value, fixed, symbol) in enumerate(zip(actual, declared, symbolic, strict=True)):
try:
extent = operator.index(value)
except TypeError as exc:
raise click.ClickException(
f"Input {name!r} axis {axis} must be an integer."
) from exc
if isinstance(value, bool) or not 0 <= extent <= (1 << 63) - 1:
raise click.ClickException(
f"Input {name!r} axis {axis} must be a nonnegative int64."
)
if fixed is not None:
if extent != fixed:
raise click.ClickException(
f"Input {name!r} axis {axis} is {extent}; ONNX requires {fixed}."
)
elif not isinstance(symbol, str) or not symbol:
# Anonymous axes cannot be bound by name. Let the compiler decide
# whether they need specialization (unused inputs may not).
continue
elif extent == 0:
raise click.ClickException(f"Input {name!r} axis {axis} must be a positive int64.")
elif symbol in overrides and overrides[symbol] != extent:
raise click.ClickException(
f"Conflicting concrete sizes for symbolic dimension {symbol!r}."
)
else:
overrides[symbol] = extent
return overrides


def _to_numpy(value: Any) -> np.ndarray:
"""Coerce a run input value (numpy array or torch tensor) to a NumPy array."""
import numpy as np
Expand Down Expand Up @@ -647,12 +693,26 @@ def __init__(
self._is_pinned: bool = False
self._has_named_bindings = False
self._compiled_artifacts: TemporaryDirectory[str] | None = None
self._symbolic_dimensions: dict[str, int] = {}
self._built = False

# Perf tracking, enabled inside perf().
self._perf_stats: PerfStats | None = None

# -- lifecycle ----------------------------------------------------------
def set_input_shapes(self, input_shapes: Mapping[str, Any]) -> None:
"""Set validated ONNX dimensions before compiling; never rewrite the model."""
from ..onnx import get_io_config

with self._lock:
if self._built:
raise ValueError("Input shapes must be configured before Runtime compilation.")
if self._is_mlir or self._backend != "cgc":
raise ValueError("Concrete compilation shapes require source ONNX and backend=cgc.")
self._symbolic_dimensions = _symbolic_dimensions_for_inputs(
get_io_config(self._model_path), input_shapes
)

def _ensure_built(self) -> None:
"""Import the runtime, resolve the target, load and build the pipeline.

Expand Down Expand Up @@ -695,9 +755,23 @@ def _load_onnx_on_cgc(
artifact_path = Path(self._compiled_artifacts.name) / "model.mlir"
with _translate_native_errors("build", device_class="gpu"):
compiler = resolved_target.execution_target.model_compiler()
options = None
try:
compiler.compile_to_file(source_model, str(artifact_path))
if self._symbolic_dimensions:
options = compiler.create_options()
if not options.supports_symbolic_dimensions:
raise click.ClickException(
"The installed Runtime compiler does not support "
"symbolic dimensions."
)
for name, extent in self._symbolic_dimensions.items():
options.symbolic_dimensions[name] = extent
compiler.compile_to_file(source_model, str(artifact_path), options=options)
else:
compiler.compile_to_file(source_model, str(artifact_path))
finally:
if options is not None:
options.close()
compiler.close()

with _translate_native_errors("load"):
Expand Down
158 changes: 158 additions & 0 deletions tests/unit/commands/test_perf_runtime_shapes.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,158 @@
# -------------------------------------------------------------------------
# Copyright (c) Microsoft Corporation. All rights reserved.
# Licensed under the MIT License.
# --------------------------------------------------------------------------
"""Concrete compilation shape handoff, with no native execution."""

from io import BytesIO
from pathlib import Path
from types import SimpleNamespace
from unittest.mock import Mock
from zipfile import ZipFile

import click
import numpy as np
import onnx
import pytest

from winml.modelkit.commands.perf import (
BenchmarkConfig,
PerfBenchmark,
_ort_options_for_dimensions,
_runtime_input_shapes,
)


def _model(tmp_path: Path) -> Path:
inputs = [
onnx.helper.make_tensor_value_info(name, onnx.TensorProto.FLOAT, ["batch", 3])
for name in ("left", "right")
]
output = onnx.helper.make_tensor_value_info("out", onnx.TensorProto.FLOAT, ["batch", 3])
model = onnx.helper.make_model(
onnx.helper.make_graph(
[onnx.helper.make_node("Add", ["left", "right"], ["out"])], "shapes", inputs, [output]
)
)
path = tmp_path / "model.onnx"
onnx.save(model, path)
return path


def test_npz_headers_override_shape_config_without_loading_tensors(tmp_path, monkeypatch):
model = _model(tmp_path)
before = model.read_bytes()
data = tmp_path / "inputs.npz"
np.savez(data, left=np.zeros((2, 3)), right=np.zeros((2, 3)))
monkeypatch.setattr(np, "load", Mock(side_effect=AssertionError("No tensor allocation")))
shapes, dimensions = _runtime_input_shapes(model, data, {"batch": 99}, 42)
assert shapes == {"left": (2, 3), "right": (2, 3)}
assert dimensions == {"batch": 2}
assert model.read_bytes() == before


def test_shape_config_resolves_symbols_without_inputs(tmp_path):
shapes, dimensions = _runtime_input_shapes(_model(tmp_path), None, {"batch": 4}, 1)
assert shapes == {"left": (4, 3), "right": (4, 3)}
assert dimensions == {"batch": 4}


@pytest.mark.parametrize(
"unused_shape, actual_shape", [([None, 3], (2, 3)), (["extent", 0], (2, 0))]
)
def test_unused_input_preserves_anonymous_and_static_zero_axes(
tmp_path, unused_shape, actual_shape
):
path = _model(tmp_path)
model = onnx.load(path)
model.graph.input.append(
onnx.helper.make_tensor_value_info("unused", onnx.TensorProto.FLOAT, unused_shape)
)
onnx.save(model, path)
data = tmp_path / "inputs.npz"
np.savez(data, left=np.zeros((2, 3)), right=np.zeros((2, 3)), unused=np.zeros(actual_shape))
shapes, dimensions = _runtime_input_shapes(path, data, None, 1)
assert shapes["unused"] == actual_shape
assert dimensions == ({"batch": 2, "extent": 2} if actual_shape[1] == 0 else {"batch": 2})


@pytest.mark.parametrize("version", [(1, 0), (2, 0), (3, 0)])
def test_npz_header_versions_without_reading_payload(tmp_path, version):
data = tmp_path / "headers.npz"
with ZipFile(data, "w") as archive:
for name in ("left", "right"):
stream = BytesIO()
np.lib.format.write_array(stream, np.zeros((2, 3), dtype=np.float32), version=version)
# Omit the payload: shape discovery must only read the header.
archive.writestr(name + ".npy", stream.getvalue()[:-24])
shapes, dimensions = _runtime_input_shapes(_model(tmp_path), data, None, 1)
assert shapes == {"left": (2, 3), "right": (2, 3)}
assert dimensions == {"batch": 2}


@pytest.mark.parametrize("right_shape", [(3, 3), (2, 4), (2, 3, 1)])
def test_npz_rejects_symbol_conflict_static_mismatch_and_rank(tmp_path, right_shape):
data = tmp_path / "inputs.npz"
np.savez(data, left=np.zeros((2, 3)), right=np.zeros(right_shape))
with pytest.raises(click.ClickException):
_runtime_input_shapes(_model(tmp_path), data, None, 1)


def test_ort_options_are_fresh_and_receive_dimensions_before_use(monkeypatch):
import onnxruntime as ort

created = []

def create():
value = SimpleNamespace(add_free_dimension_override_by_name=Mock())
created.append(value)
return value

monkeypatch.setattr(ort, "SessionOptions", create)
first = _ort_options_for_dimensions({"batch": 2})
second = _ort_options_for_dimensions({"batch": 2})
assert first is not second
for options in created:
options.add_free_dimension_override_by_name.assert_called_once_with("batch", 2)


@pytest.mark.parametrize("runtime", ["winml-runtime", "winml-ort"])
def test_perf_hands_shapes_to_runtime_before_compilation(tmp_path, monkeypatch, runtime):
from winml.modelkit.models import WinMLAutoModel

model_path = _model(tmp_path)
data = tmp_path / "inputs.npz"
np.savez(data, left=np.zeros((2, 3)), right=np.zeros((2, 3)))
config = BenchmarkConfig(
model_id=str(model_path),
runtime=runtime,
backend="cgc" if runtime == "winml-runtime" else None,
input_data=data,
device="gpu",
)
benchmark = PerfBenchmark(config)
benchmark._ep_device = SimpleNamespace(
device=SimpleNamespace(ep_name="WinMLCGExecutionProvider")
)
monkeypatch.setattr(benchmark, "_resolve_device_ep", lambda: None)
session = SimpleNamespace(set_input_shapes=Mock(), compile=Mock())
wrapper = SimpleNamespace(_session=session)
options = object()
configure = Mock(return_value=options)
monkeypatch.setattr("winml.modelkit.commands.perf._ort_options_for_dimensions", configure)

def construct(**kwargs):
if runtime == "winml-ort":
assert kwargs["session_options"]() is options
configure.assert_called_once_with({"batch": 2})
else:
assert "session_options" not in kwargs
return wrapper

monkeypatch.setattr(WinMLAutoModel, "from_onnx", construct)
benchmark._load_model()
if runtime == "winml-runtime":
session.set_input_shapes.assert_called_once_with({"left": (2, 3), "right": (2, 3)})
else:
session.set_input_shapes.assert_not_called()
session.compile.assert_not_called()
Loading