diff --git a/docs/commands/perf.md b/docs/commands/perf.md index 7abe93dd0..bc8045e88 100644 --- a/docs/commands/perf.md +++ b/docs/commands/perf.md @@ -421,6 +421,19 @@ not supported. - **Random inputs do not represent real data distributions.** Latency numbers are accurate, but memory access patterns may differ from production because the generated tensors are uniform random values. For memory-bandwidth-sensitive models this can understate real-world latency. - **Cross-device comparison.** To compare performance across devices, run `winml perf` separately with different `--device` values and compare the resulting JSON reports. +## Concrete input shapes for CGC + +For local ONNX models, Runtime CGC and WinMLCG receive concrete named input +dimensions before compilation. With --input-data, shapes come from NPZ headers +without allocating input tensors. Otherwise, the existing --shape-config and +batch-size resolution rules apply. Rank, static axes, positive dimensions and +shared symbolic names must agree. The source ONNX is not rewritten. + +Runtime CGC requires a Runtime compiler that supports symbolic-dimension +options. Anonymous dynamic axes are rejected rather than guessed, and these +overrides do not resolve internal data-dependent shapes. Input payload loading +and dtype conversion still occur at the normal input-allocation boundary. + ## See also - [winml eval](eval.md) — measure accuracy after benchmarking diff --git a/src/winml/modelkit/commands/perf.py b/src/winml/modelkit/commands/perf.py index 98061bb83..ee4bf2cec 100644 --- a/src/winml/modelkit/commands/perf.py +++ b/src/winml/modelkit/commands/perf.py @@ -907,6 +907,70 @@ def load_input_data( return _load_input_data(path, io_config) +def _runtime_input_shapes( + model_path: Path, input_data: Path | None, shape_config: dict | None, batch_size: int +) -> tuple[dict[str, tuple[int, ...]], dict[str, int]]: + """Resolve source-ONNX shapes before compilation without allocating input tensors.""" + import zipfile + + from ..onnx import get_io_config + from ..session.runtime_session import _symbolic_dimensions_for_inputs + + io_config = get_io_config(model_path) + if not any(dim is None for shape in io_config["input_shapes"] for dim in shape): + return {}, {} + shapes: dict[str, tuple[int, ...]] = {} + if input_data is not None: + if input_data.suffix.lower() != ".npz": + raise click.UsageError("--input-data must be a named .npz archive.") + try: + with zipfile.ZipFile(input_data) as archive: + members = archive.namelist() + expected = [name + ".npy" for name in io_config["input_names"]] + if len(members) != len(expected) or set(members) != set(expected): + raise ValueError("archive keys must exactly match ONNX input names") + for name in io_config["input_names"]: + with archive.open(name + ".npy") as stream: + version = np.lib.format.read_magic(stream) + # NumPy has no public v3 header reader. Use the same + # version-aware, size-limited reader as np.load, without + # reading or allocating the array payload. + shape, _, dtype = np.lib.format._read_array_header( # type: ignore[attr-defined] + stream, version + ) + if dtype.hasobject: + raise ValueError("object input arrays are unsupported") + shapes[name] = shape + except (OSError, ValueError, EOFError, zipfile.BadZipFile) as exc: + raise click.UsageError(f"Cannot read concrete --input-data shapes: {exc}") from exc + else: + for name, shape, symbolic in zip( + io_config["input_names"], + io_config["input_shapes"], + io_config["input_symbolic_shapes"], + strict=True, + ): + full_shape = (shape_config or {}).get(name) + if isinstance(full_shape, (list, tuple)): + # Preserve original values for strict integer/static-axis validation. + shapes[name] = tuple(full_shape) + else: + shapes[name] = _resolve_shape( + shape, name, batch_size, symbolic_shape=symbolic, shape_config=shape_config + ) + return shapes, _symbolic_dimensions_for_inputs(io_config, shapes) + + +def _ort_options_for_dimensions(dimensions: dict[str, int]) -> Any: + """Create fresh ORT options with concrete dimensions before eager session creation.""" + import onnxruntime as ort + + options = ort.SessionOptions() + for name, extent in dimensions.items(): + options.add_free_dimension_override_by_name(name, extent) + return options + + def effective_batch_size( inputs: dict[str, np.ndarray], input_names: list[str], @@ -1361,6 +1425,23 @@ def _load_model(self) -> None: } if is_onnx: + runtime_shapes: dict[str, tuple[int, ...]] = {} + runtime_cgc = self.config.runtime == "winml-runtime" and self._runtime_backend == "cgc" + ort_cgc = ( + self.config.runtime == "winml-ort" + and self._ep_device.device.ep_name == "WinMLCGExecutionProvider" + ) + if runtime_cgc or ort_cgc: + runtime_shapes, dimensions = _runtime_input_shapes( + model_path, + self.config.input_data, + self.config.shape_config, + self.config.batch_size, + ) + if ort_cgc and dimensions: + common_kwargs["session_options"] = lambda: _ort_options_for_dimensions( + dimensions + ) with suppress_native_warnings(enabled=True): self._model = WinMLAutoModel.from_onnx( onnx_path=model_path, @@ -1368,6 +1449,12 @@ def _load_model(self) -> None: compile_provider_options=self.config.compile_ep_options, **common_kwargs, ) + if runtime_cgc and runtime_shapes: + from ..session.runtime_session import WinMLRuntimeSession + + # Keep the lazy import visible to CodeQL as well as type checkers. + runtime_session = cast(WinMLRuntimeSession, self._single._session) # noqa: TC006 + runtime_session.set_input_shapes(runtime_shapes) elif is_mlir: with suppress_native_warnings(enabled=True): self._model = WinMLAutoModel.from_mlir( diff --git a/src/winml/modelkit/session/runtime_session.py b/src/winml/modelkit/session/runtime_session.py index 79fb92a2f..3f19d8db8 100644 --- a/src/winml/modelkit/session/runtime_session.py +++ b/src/winml/modelkit/session/runtime_session.py @@ -30,6 +30,7 @@ import ctypes import json import logging +import operator import threading from contextlib import contextmanager from dataclasses import dataclass @@ -246,6 +247,51 @@ def _stage_schema(wr: Any, stage: Any) -> Any: return schema_type(interface, stage) +def _symbolic_dimensions_for_inputs( + io_config: dict[str, Any], input_shapes: Mapping[str, Any] +) -> dict[str, int]: + """Validate concrete input shapes and resolve shared ONNX dimension names.""" + names = io_config["input_names"] + if set(input_shapes) != set(names): + raise click.ClickException("Concrete input shapes must match the ONNX input names exactly.") + overrides: dict[str, int] = {} + for name, declared, symbolic in zip( + names, io_config["input_shapes"], io_config["input_symbolic_shapes"], strict=True + ): + actual = input_shapes[name] + if len(actual) != len(declared): + raise click.ClickException(f"Input {name!r} rank does not match the ONNX schema.") + for axis, (value, fixed, symbol) in enumerate(zip(actual, declared, symbolic, strict=True)): + try: + extent = operator.index(value) + except TypeError as exc: + raise click.ClickException( + f"Input {name!r} axis {axis} must be an integer." + ) from exc + if isinstance(value, bool) or not 0 <= extent <= (1 << 63) - 1: + raise click.ClickException( + f"Input {name!r} axis {axis} must be a nonnegative int64." + ) + if fixed is not None: + if extent != fixed: + raise click.ClickException( + f"Input {name!r} axis {axis} is {extent}; ONNX requires {fixed}." + ) + elif not isinstance(symbol, str) or not symbol: + # Anonymous axes cannot be bound by name. Let the compiler decide + # whether they need specialization (unused inputs may not). + continue + elif extent == 0: + raise click.ClickException(f"Input {name!r} axis {axis} must be a positive int64.") + elif symbol in overrides and overrides[symbol] != extent: + raise click.ClickException( + f"Conflicting concrete sizes for symbolic dimension {symbol!r}." + ) + else: + overrides[symbol] = extent + return overrides + + def _to_numpy(value: Any) -> np.ndarray: """Coerce a run input value (numpy array or torch tensor) to a NumPy array.""" import numpy as np @@ -647,12 +693,26 @@ def __init__( self._is_pinned: bool = False self._has_named_bindings = False self._compiled_artifacts: TemporaryDirectory[str] | None = None + self._symbolic_dimensions: dict[str, int] = {} self._built = False # Perf tracking, enabled inside perf(). self._perf_stats: PerfStats | None = None # -- lifecycle ---------------------------------------------------------- + def set_input_shapes(self, input_shapes: Mapping[str, Any]) -> None: + """Set validated ONNX dimensions before compiling; never rewrite the model.""" + from ..onnx import get_io_config + + with self._lock: + if self._built: + raise ValueError("Input shapes must be configured before Runtime compilation.") + if self._is_mlir or self._backend != "cgc": + raise ValueError("Concrete compilation shapes require source ONNX and backend=cgc.") + self._symbolic_dimensions = _symbolic_dimensions_for_inputs( + get_io_config(self._model_path), input_shapes + ) + def _ensure_built(self) -> None: """Import the runtime, resolve the target, load and build the pipeline. @@ -695,9 +755,23 @@ def _load_onnx_on_cgc( artifact_path = Path(self._compiled_artifacts.name) / "model.mlir" with _translate_native_errors("build", device_class="gpu"): compiler = resolved_target.execution_target.model_compiler() + options = None try: - compiler.compile_to_file(source_model, str(artifact_path)) + if self._symbolic_dimensions: + options = compiler.create_options() + if not options.supports_symbolic_dimensions: + raise click.ClickException( + "The installed Runtime compiler does not support " + "symbolic dimensions." + ) + for name, extent in self._symbolic_dimensions.items(): + options.symbolic_dimensions[name] = extent + compiler.compile_to_file(source_model, str(artifact_path), options=options) + else: + compiler.compile_to_file(source_model, str(artifact_path)) finally: + if options is not None: + options.close() compiler.close() with _translate_native_errors("load"): diff --git a/tests/unit/commands/test_perf_runtime_shapes.py b/tests/unit/commands/test_perf_runtime_shapes.py new file mode 100644 index 000000000..059479b59 --- /dev/null +++ b/tests/unit/commands/test_perf_runtime_shapes.py @@ -0,0 +1,158 @@ +# ------------------------------------------------------------------------- +# Copyright (c) Microsoft Corporation. All rights reserved. +# Licensed under the MIT License. +# -------------------------------------------------------------------------- +"""Concrete compilation shape handoff, with no native execution.""" + +from io import BytesIO +from pathlib import Path +from types import SimpleNamespace +from unittest.mock import Mock +from zipfile import ZipFile + +import click +import numpy as np +import onnx +import pytest + +from winml.modelkit.commands.perf import ( + BenchmarkConfig, + PerfBenchmark, + _ort_options_for_dimensions, + _runtime_input_shapes, +) + + +def _model(tmp_path: Path) -> Path: + inputs = [ + onnx.helper.make_tensor_value_info(name, onnx.TensorProto.FLOAT, ["batch", 3]) + for name in ("left", "right") + ] + output = onnx.helper.make_tensor_value_info("out", onnx.TensorProto.FLOAT, ["batch", 3]) + model = onnx.helper.make_model( + onnx.helper.make_graph( + [onnx.helper.make_node("Add", ["left", "right"], ["out"])], "shapes", inputs, [output] + ) + ) + path = tmp_path / "model.onnx" + onnx.save(model, path) + return path + + +def test_npz_headers_override_shape_config_without_loading_tensors(tmp_path, monkeypatch): + model = _model(tmp_path) + before = model.read_bytes() + data = tmp_path / "inputs.npz" + np.savez(data, left=np.zeros((2, 3)), right=np.zeros((2, 3))) + monkeypatch.setattr(np, "load", Mock(side_effect=AssertionError("No tensor allocation"))) + shapes, dimensions = _runtime_input_shapes(model, data, {"batch": 99}, 42) + assert shapes == {"left": (2, 3), "right": (2, 3)} + assert dimensions == {"batch": 2} + assert model.read_bytes() == before + + +def test_shape_config_resolves_symbols_without_inputs(tmp_path): + shapes, dimensions = _runtime_input_shapes(_model(tmp_path), None, {"batch": 4}, 1) + assert shapes == {"left": (4, 3), "right": (4, 3)} + assert dimensions == {"batch": 4} + + +@pytest.mark.parametrize( + "unused_shape, actual_shape", [([None, 3], (2, 3)), (["extent", 0], (2, 0))] +) +def test_unused_input_preserves_anonymous_and_static_zero_axes( + tmp_path, unused_shape, actual_shape +): + path = _model(tmp_path) + model = onnx.load(path) + model.graph.input.append( + onnx.helper.make_tensor_value_info("unused", onnx.TensorProto.FLOAT, unused_shape) + ) + onnx.save(model, path) + data = tmp_path / "inputs.npz" + np.savez(data, left=np.zeros((2, 3)), right=np.zeros((2, 3)), unused=np.zeros(actual_shape)) + shapes, dimensions = _runtime_input_shapes(path, data, None, 1) + assert shapes["unused"] == actual_shape + assert dimensions == ({"batch": 2, "extent": 2} if actual_shape[1] == 0 else {"batch": 2}) + + +@pytest.mark.parametrize("version", [(1, 0), (2, 0), (3, 0)]) +def test_npz_header_versions_without_reading_payload(tmp_path, version): + data = tmp_path / "headers.npz" + with ZipFile(data, "w") as archive: + for name in ("left", "right"): + stream = BytesIO() + np.lib.format.write_array(stream, np.zeros((2, 3), dtype=np.float32), version=version) + # Omit the payload: shape discovery must only read the header. + archive.writestr(name + ".npy", stream.getvalue()[:-24]) + shapes, dimensions = _runtime_input_shapes(_model(tmp_path), data, None, 1) + assert shapes == {"left": (2, 3), "right": (2, 3)} + assert dimensions == {"batch": 2} + + +@pytest.mark.parametrize("right_shape", [(3, 3), (2, 4), (2, 3, 1)]) +def test_npz_rejects_symbol_conflict_static_mismatch_and_rank(tmp_path, right_shape): + data = tmp_path / "inputs.npz" + np.savez(data, left=np.zeros((2, 3)), right=np.zeros(right_shape)) + with pytest.raises(click.ClickException): + _runtime_input_shapes(_model(tmp_path), data, None, 1) + + +def test_ort_options_are_fresh_and_receive_dimensions_before_use(monkeypatch): + import onnxruntime as ort + + created = [] + + def create(): + value = SimpleNamespace(add_free_dimension_override_by_name=Mock()) + created.append(value) + return value + + monkeypatch.setattr(ort, "SessionOptions", create) + first = _ort_options_for_dimensions({"batch": 2}) + second = _ort_options_for_dimensions({"batch": 2}) + assert first is not second + for options in created: + options.add_free_dimension_override_by_name.assert_called_once_with("batch", 2) + + +@pytest.mark.parametrize("runtime", ["winml-runtime", "winml-ort"]) +def test_perf_hands_shapes_to_runtime_before_compilation(tmp_path, monkeypatch, runtime): + from winml.modelkit.models import WinMLAutoModel + + model_path = _model(tmp_path) + data = tmp_path / "inputs.npz" + np.savez(data, left=np.zeros((2, 3)), right=np.zeros((2, 3))) + config = BenchmarkConfig( + model_id=str(model_path), + runtime=runtime, + backend="cgc" if runtime == "winml-runtime" else None, + input_data=data, + device="gpu", + ) + benchmark = PerfBenchmark(config) + benchmark._ep_device = SimpleNamespace( + device=SimpleNamespace(ep_name="WinMLCGExecutionProvider") + ) + monkeypatch.setattr(benchmark, "_resolve_device_ep", lambda: None) + session = SimpleNamespace(set_input_shapes=Mock(), compile=Mock()) + wrapper = SimpleNamespace(_session=session) + options = object() + configure = Mock(return_value=options) + monkeypatch.setattr("winml.modelkit.commands.perf._ort_options_for_dimensions", configure) + + def construct(**kwargs): + if runtime == "winml-ort": + assert kwargs["session_options"]() is options + configure.assert_called_once_with({"batch": 2}) + else: + assert "session_options" not in kwargs + return wrapper + + monkeypatch.setattr(WinMLAutoModel, "from_onnx", construct) + benchmark._load_model() + if runtime == "winml-runtime": + session.set_input_shapes.assert_called_once_with({"left": (2, 3), "right": (2, 3)}) + else: + session.set_input_shapes.assert_not_called() + session.compile.assert_not_called() diff --git a/tests/unit/session/test_runtime_session.py b/tests/unit/session/test_runtime_session.py index 897cac4ec..f5ad8b673 100644 --- a/tests/unit/session/test_runtime_session.py +++ b/tests/unit/session/test_runtime_session.py @@ -24,6 +24,7 @@ _numpy_dtype_for, _ResolvedRuntimeTarget, _shape_with_dynamic_dims, + _symbolic_dimensions_for_inputs, ) @@ -206,6 +207,52 @@ def load_model(path: str) -> Any: session.reset() +@pytest.mark.parametrize("fails", [False, True]) +def test_cgc_compile_uses_public_symbolic_options_and_closes_them(monkeypatch, fails): + io = { + "input_names": ["x"], + "input_shapes": [[None, 3]], + "input_symbolic_shapes": [["batch", 3]], + } + monkeypatch.setattr("winml.modelkit.onnx.get_io_config", lambda _path: io) + options = SimpleNamespace( + supports_symbolic_dimensions=True, symbolic_dimensions={}, close=Mock() + ) + source = Mock() + compiler = Mock(create_options=Mock(return_value=options)) + if fails: + compiler.compile_to_file.side_effect = RuntimeError("compile failed") + runtime = Mock(load_model=Mock(side_effect=[source, _Model()])) + target = _ResolvedRuntimeTarget( + execution_target=SimpleNamespace(model_compiler=lambda: compiler), device_class="gpu" + ) + session = WinMLRuntimeSession("source.onnx", ep_device=_mlir_ep_device(), backend="cgc") + session.set_input_shapes({"x": (2, 3)}) + try: + if fails: + with pytest.raises(RuntimeError, match="compile failed"): + session._load_onnx_on_cgc(runtime, target) + else: + session._load_onnx_on_cgc(runtime, target) + assert options.symbolic_dimensions == {"batch": 2} + assert compiler.compile_to_file.call_args.kwargs == {"options": options} + options.close.assert_called_once_with() + compiler.close.assert_called_once_with() + finally: + session.reset() + + +@pytest.mark.parametrize("shape", [(True, 3), (1.5, 3), (0, 3), (1 << 63, 3)]) +def test_symbolic_dimensions_reject_invalid_extents(shape): + io = { + "input_names": ["x"], + "input_shapes": [[None, 3]], + "input_symbolic_shapes": [["batch", 3]], + } + with pytest.raises(click.ClickException): + _symbolic_dimensions_for_inputs(io, {"x": shape}) + + def test_schema_helpers() -> None: assert _numpy_dtype_for(SimpleNamespace(name="FLOAT16")) == "float16" assert _shape_with_dynamic_dims([0, -1, (1 << 64) - 1, 4]) == [