From 4e9700f365bd34363aa52e278fb8a41039e862eb Mon Sep 17 00:00:00 2001 From: Qiong Wu Date: Tue, 22 Sep 2026 08:17:46 +0800 Subject: [PATCH 1/3] fix(perf): bind concrete CGC input shapes before compilation --- docs/commands/perf.md | 20 ++- src/winml/modelkit/commands/perf.py | 84 ++++++++++++ src/winml/modelkit/ep_path.py | 15 +-- src/winml/modelkit/session/runtime_session.py | 73 ++++++++++- .../unit/commands/test_perf_runtime_shapes.py | 123 ++++++++++++++++++ .../unit/ep_path/test_winml_catalog_source.py | 51 +++++++- tests/unit/session/test_runtime_session.py | 47 +++++++ 7 files changed, 394 insertions(+), 19 deletions(-) create mode 100644 tests/unit/commands/test_perf_runtime_shapes.py diff --git a/docs/commands/perf.md b/docs/commands/perf.md index 7abe93dd0..e1d59118e 100644 --- a/docs/commands/perf.md +++ b/docs/commands/perf.md @@ -413,6 +413,7 @@ not supported. ## Common pitfalls +- **Provider discovery and installation.** Discovery lists already-ready catalog providers without preparing unrelated providers. An explicit provider request such as `--ep qnn` retries that provider's catalog with preparation enabled when no installed source is available. `WINMLCLI_EP_PATH` adds source precedence; it does not disable the other discovery sources. - **Warm-up too low on NPU.** The first several inferences on an NPU EP can be significantly slower due to kernel compilation and caching. The default of 10 warm-up iterations is usually enough for vision models, but transformer models with many operators may need `--warmup 30` or higher to reach steady-state latency. - **Hidden third-party diagnostics.** Normal `winml perf` output suppresses noisy native warning-level diagnostics and Hugging Face download/progress chatter so benchmark results stay readable. Use `-v`/`-vv` or set `WINMLCLI_SHOW_ALL_WARNINGS=1` to show those warnings when debugging provider or Hub issues. - **`--input-data` keys must match; dtypes are cast.** The `.npz` keys must equal the model's input names — a missing or unexpected key is a hard error (typo protection). Array dtypes are cast to the model's expected dtype with a warning (matching normal inference), so you don't have to hand-match widths. `.npy` files are not supported — save named arrays as `.npz`. When `--input-data` is set, `--batch-size` and `--shape-config` are ignored (the tensors define their own shapes). It is also rejected for `--module` mode, `--runtime ort-genai`, and composite (dual-encoder) models such as CLIP/SigLIP, where each sub-model has its own inputs that a single `.npz` cannot address. @@ -421,9 +422,22 @@ not supported. - **Random inputs do not represent real data distributions.** Latency numbers are accurate, but memory access patterns may differ from production because the generated tensors are uniform random values. For memory-bandwidth-sensitive models this can understate real-world latency. - **Cross-device comparison.** To compare performance across devices, run `winml perf` separately with different `--device` values and compare the resulting JSON reports. +## Concrete input shapes for CGC + +For local ONNX models, Runtime CGC and WinMLCG receive concrete named input +dimensions before compilation. With --input-data, shapes come from NPZ headers +without allocating input tensors. Otherwise, the existing --shape-config and +batch-size resolution rules apply. Rank, static axes, positive dimensions and +shared symbolic names must agree. The source ONNX is not rewritten. + +Runtime CGC requires a Runtime compiler that supports symbolic-dimension +options. Anonymous dynamic axes are rejected rather than guessed, and these +overrides do not resolve internal data-dependent shapes. Input payload loading +and dtype conversion still occur at the normal input-allocation boundary. + ## See also - [winml eval](eval.md) — measure accuracy after benchmarking -- [winml build](build.md) — build the quantized artifact that `perf` benchmarks -- [Load and export concept](../concepts/load-and-export.md) — how `--module` per-instance benchmarking works -- [ONNX & Execution Providers](../concepts/eps-and-devices.md) — understand `--device` vs `--ep` +- [winml build](build.md) — build the quantized artifact that perf benchmarks +- [Load and export concept](../concepts/load-and-export.md) — module benchmarking +- [ONNX & Execution Providers](../concepts/eps-and-devices.md) — devices and EPs diff --git a/src/winml/modelkit/commands/perf.py b/src/winml/modelkit/commands/perf.py index 98061bb83..dd179fe82 100644 --- a/src/winml/modelkit/commands/perf.py +++ b/src/winml/modelkit/commands/perf.py @@ -65,6 +65,7 @@ from ..session.monitor import ProcessMemoryTracker from ..session.monitor.ep_monitor import WinMLEPMonitor from ..session.monitor.op_metrics import TraceFallbackReason + from ..session.runtime_session import WinMLRuntimeSession from ..session.stats import PerfStats logger = logging.getLogger(__name__) @@ -907,6 +908,70 @@ def load_input_data( return _load_input_data(path, io_config) +def _runtime_input_shapes( + model_path: Path, input_data: Path | None, shape_config: dict | None, batch_size: int +) -> tuple[dict[str, tuple[int, ...]], dict[str, int]]: + """Resolve source-ONNX shapes before compilation without allocating input tensors.""" + import zipfile + + from ..onnx import get_io_config + from ..session.runtime_session import _symbolic_dimensions_for_inputs + + io_config = get_io_config(model_path) + if not any(dim is None for shape in io_config["input_shapes"] for dim in shape): + return {}, {} + shapes: dict[str, tuple[int, ...]] = {} + if input_data is not None: + if input_data.suffix.lower() != ".npz": + raise click.UsageError("--input-data must be a named .npz archive.") + try: + with zipfile.ZipFile(input_data) as archive: + members = archive.namelist() + expected = [name + ".npy" for name in io_config["input_names"]] + if len(members) != len(expected) or set(members) != set(expected): + raise ValueError("archive keys must exactly match ONNX input names") + for name in io_config["input_names"]: + with archive.open(name + ".npy") as stream: + version = np.lib.format.read_magic(stream) + if version == (1, 0): + shape, _, dtype = np.lib.format.read_array_header_1_0(stream) + elif version == (2, 0): + shape, _, dtype = np.lib.format.read_array_header_2_0(stream) + else: + raise ValueError(f"Unsupported NPY header version {version}") + if dtype.hasobject: + raise ValueError("object input arrays are unsupported") + shapes[name] = shape + except (OSError, ValueError, EOFError, zipfile.BadZipFile) as exc: + raise click.UsageError(f"Cannot read concrete --input-data shapes: {exc}") from exc + else: + for name, shape, symbolic in zip( + io_config["input_names"], + io_config["input_shapes"], + io_config["input_symbolic_shapes"], + strict=True, + ): + full_shape = (shape_config or {}).get(name) + if isinstance(full_shape, (list, tuple)): + # Preserve original values for strict integer/static-axis validation. + shapes[name] = tuple(full_shape) + else: + shapes[name] = _resolve_shape( + shape, name, batch_size, symbolic_shape=symbolic, shape_config=shape_config + ) + return shapes, _symbolic_dimensions_for_inputs(io_config, shapes) + + +def _ort_options_for_dimensions(dimensions: dict[str, int]) -> Any: + """Create fresh ORT options with concrete dimensions before eager session creation.""" + import onnxruntime as ort + + options = ort.SessionOptions() + for name, extent in dimensions.items(): + options.add_free_dimension_override_by_name(name, extent) + return options + + def effective_batch_size( inputs: dict[str, np.ndarray], input_names: list[str], @@ -1361,6 +1426,23 @@ def _load_model(self) -> None: } if is_onnx: + runtime_shapes: dict[str, tuple[int, ...]] = {} + runtime_cgc = self.config.runtime == "winml-runtime" and self._runtime_backend == "cgc" + ort_cgc = ( + self.config.runtime == "winml-ort" + and self._ep_device.device.ep_name == "WinMLCGExecutionProvider" + ) + if runtime_cgc or ort_cgc: + runtime_shapes, dimensions = _runtime_input_shapes( + model_path, + self.config.input_data, + self.config.shape_config, + self.config.batch_size, + ) + if ort_cgc and dimensions: + common_kwargs["session_options"] = lambda: _ort_options_for_dimensions( + dimensions + ) with suppress_native_warnings(enabled=True): self._model = WinMLAutoModel.from_onnx( onnx_path=model_path, @@ -1368,6 +1450,8 @@ def _load_model(self) -> None: compile_provider_options=self.config.compile_ep_options, **common_kwargs, ) + if runtime_cgc and runtime_shapes: + cast("WinMLRuntimeSession", self._single._session).set_input_shapes(runtime_shapes) elif is_mlir: with suppress_native_warnings(enabled=True): self._model = WinMLAutoModel.from_mlir( diff --git a/src/winml/modelkit/ep_path.py b/src/winml/modelkit/ep_path.py index bec8bcd35..f12db474d 100644 --- a/src/winml/modelkit/ep_path.py +++ b/src/winml/modelkit/ep_path.py @@ -1124,8 +1124,8 @@ class WinMLCatalogSource(EPSource): eps: Canonical EP names this source provides. Typically a single name, but listed as a tuple for symmetry with the other sources. - auto_download: If ``True``, providers in the ``NotPresent`` ready - state will be downloaded by ``ensure_ready()``. + auto_download: If ``True``, non-ready providers may be prepared or + downloaded by ``ensure_ready()``. Defaults to ``False`` to avoid surprising the user with a multi-second to multi-minute network operation on first call; see ``docs/ep-path-design.md`` Interaction section. @@ -1183,14 +1183,13 @@ def _resolve_provider(self, provider: Any) -> Iterator[EPEntry]: if getattr(provider, "name", None) != self.catalog_name: return - # Skip providers that are not present on this machine. The design - # doc explicitly forbids auto-downloading hundreds of MB without - # opt-in; we honor that via auto_download=False (the default). + # Discovery must not prepare unrelated providers. NotReady can also + # require installation; reserve every readiness transition for the + # explicit-provider acquisition path (auto_download=True). ready_state = getattr(provider, "ready_state", None) - if ready_state is not None and not self.auto_download and self._is_not_present(ready_state): + if not self.auto_download and not self._is_ready(ready_state): logger.debug( - "WinMLCatalogSource(%s): provider in NotPresent state; " - "skipping (auto_download=False)", + "WinMLCatalogSource(%s): provider is not ready; skipping (auto_download=False)", self.catalog_name, ) return diff --git a/src/winml/modelkit/session/runtime_session.py b/src/winml/modelkit/session/runtime_session.py index 79fb92a2f..4f4fe9142 100644 --- a/src/winml/modelkit/session/runtime_session.py +++ b/src/winml/modelkit/session/runtime_session.py @@ -30,6 +30,7 @@ import ctypes import json import logging +import operator import threading from contextlib import contextmanager from dataclasses import dataclass @@ -246,6 +247,48 @@ def _stage_schema(wr: Any, stage: Any) -> Any: return schema_type(interface, stage) +def _symbolic_dimensions_for_inputs( + io_config: dict[str, Any], input_shapes: Mapping[str, Any] +) -> dict[str, int]: + """Validate concrete input shapes and resolve shared ONNX dimension names.""" + names = io_config["input_names"] + if set(input_shapes) != set(names): + raise click.ClickException("Concrete input shapes must match the ONNX input names exactly.") + overrides: dict[str, int] = {} + for name, declared, symbolic in zip( + names, io_config["input_shapes"], io_config["input_symbolic_shapes"], strict=True + ): + actual = input_shapes[name] + if len(actual) != len(declared): + raise click.ClickException(f"Input {name!r} rank does not match the ONNX schema.") + for axis, (value, fixed, symbol) in enumerate(zip(actual, declared, symbolic, strict=True)): + try: + extent = operator.index(value) + except TypeError as exc: + raise click.ClickException( + f"Input {name!r} axis {axis} must be an integer." + ) from exc + if isinstance(value, bool) or not 0 < extent <= (1 << 63) - 1: + raise click.ClickException(f"Input {name!r} axis {axis} must be a positive int64.") + if fixed is not None: + if extent != fixed: + raise click.ClickException( + f"Input {name!r} axis {axis} is {extent}; ONNX requires {fixed}." + ) + elif not isinstance(symbol, str) or not symbol: + raise click.ClickException( + f"Input {name!r} axis {axis} has no symbolic dimension name; " + "the compiler's named-dimension API cannot bind it." + ) + elif symbol in overrides and overrides[symbol] != extent: + raise click.ClickException( + f"Conflicting concrete sizes for symbolic dimension {symbol!r}." + ) + else: + overrides[symbol] = extent + return overrides + + def _to_numpy(value: Any) -> np.ndarray: """Coerce a run input value (numpy array or torch tensor) to a NumPy array.""" import numpy as np @@ -647,12 +690,26 @@ def __init__( self._is_pinned: bool = False self._has_named_bindings = False self._compiled_artifacts: TemporaryDirectory[str] | None = None + self._symbolic_dimensions: dict[str, int] = {} self._built = False # Perf tracking, enabled inside perf(). self._perf_stats: PerfStats | None = None # -- lifecycle ---------------------------------------------------------- + def set_input_shapes(self, input_shapes: Mapping[str, Any]) -> None: + """Set validated ONNX dimensions before compiling; never rewrite the model.""" + from ..onnx import get_io_config + + with self._lock: + if self._built: + raise ValueError("Input shapes must be configured before Runtime compilation.") + if self._is_mlir or self._backend != "cgc": + raise ValueError("Concrete compilation shapes require source ONNX and backend=cgc.") + self._symbolic_dimensions = _symbolic_dimensions_for_inputs( + get_io_config(self._model_path), input_shapes + ) + def _ensure_built(self) -> None: """Import the runtime, resolve the target, load and build the pipeline. @@ -695,9 +752,23 @@ def _load_onnx_on_cgc( artifact_path = Path(self._compiled_artifacts.name) / "model.mlir" with _translate_native_errors("build", device_class="gpu"): compiler = resolved_target.execution_target.model_compiler() + options = None try: - compiler.compile_to_file(source_model, str(artifact_path)) + if self._symbolic_dimensions: + options = compiler.create_options() + if not options.supports_symbolic_dimensions: + raise click.ClickException( + "The installed Runtime compiler does not support " + "symbolic dimensions." + ) + for name, extent in self._symbolic_dimensions.items(): + options.symbolic_dimensions[name] = extent + compiler.compile_to_file(source_model, str(artifact_path), options=options) + else: + compiler.compile_to_file(source_model, str(artifact_path)) finally: + if options is not None: + options.close() compiler.close() with _translate_native_errors("load"): diff --git a/tests/unit/commands/test_perf_runtime_shapes.py b/tests/unit/commands/test_perf_runtime_shapes.py new file mode 100644 index 000000000..c5fe28059 --- /dev/null +++ b/tests/unit/commands/test_perf_runtime_shapes.py @@ -0,0 +1,123 @@ +# ------------------------------------------------------------------------- +# Copyright (c) Microsoft Corporation. All rights reserved. +# Licensed under the MIT License. +# -------------------------------------------------------------------------- +"""Concrete compilation shape handoff, with no native execution.""" + +from pathlib import Path +from types import SimpleNamespace +from unittest.mock import Mock + +import click +import numpy as np +import onnx +import pytest + +from winml.modelkit.commands.perf import ( + BenchmarkConfig, + PerfBenchmark, + _ort_options_for_dimensions, + _runtime_input_shapes, +) + + +def _model(tmp_path: Path) -> Path: + inputs = [ + onnx.helper.make_tensor_value_info(name, onnx.TensorProto.FLOAT, ["batch", 3]) + for name in ("left", "right") + ] + output = onnx.helper.make_tensor_value_info("out", onnx.TensorProto.FLOAT, ["batch", 3]) + model = onnx.helper.make_model( + onnx.helper.make_graph( + [onnx.helper.make_node("Add", ["left", "right"], ["out"])], "shapes", inputs, [output] + ) + ) + path = tmp_path / "model.onnx" + onnx.save(model, path) + return path + + +def test_npz_headers_override_shape_config_without_loading_tensors(tmp_path, monkeypatch): + model = _model(tmp_path) + before = model.read_bytes() + data = tmp_path / "inputs.npz" + np.savez(data, left=np.zeros((2, 3)), right=np.zeros((2, 3))) + monkeypatch.setattr(np, "load", Mock(side_effect=AssertionError("No tensor allocation"))) + shapes, dimensions = _runtime_input_shapes(model, data, {"batch": 99}, 42) + assert shapes == {"left": (2, 3), "right": (2, 3)} + assert dimensions == {"batch": 2} + assert model.read_bytes() == before + + +def test_shape_config_resolves_symbols_without_inputs(tmp_path): + shapes, dimensions = _runtime_input_shapes(_model(tmp_path), None, {"batch": 4}, 1) + assert shapes == {"left": (4, 3), "right": (4, 3)} + assert dimensions == {"batch": 4} + + +@pytest.mark.parametrize("right_shape", [(3, 3), (2, 4), (2, 3, 1)]) +def test_npz_rejects_symbol_conflict_static_mismatch_and_rank(tmp_path, right_shape): + data = tmp_path / "inputs.npz" + np.savez(data, left=np.zeros((2, 3)), right=np.zeros(right_shape)) + with pytest.raises(click.ClickException): + _runtime_input_shapes(_model(tmp_path), data, None, 1) + + +def test_ort_options_are_fresh_and_receive_dimensions_before_use(monkeypatch): + import onnxruntime as ort + + created = [] + + def create(): + value = SimpleNamespace(add_free_dimension_override_by_name=Mock()) + created.append(value) + return value + + monkeypatch.setattr(ort, "SessionOptions", create) + first = _ort_options_for_dimensions({"batch": 2}) + second = _ort_options_for_dimensions({"batch": 2}) + assert first is not second + for options in created: + options.add_free_dimension_override_by_name.assert_called_once_with("batch", 2) + + +@pytest.mark.parametrize("runtime", ["winml-runtime", "winml-ort"]) +def test_perf_hands_shapes_to_runtime_before_compilation(tmp_path, monkeypatch, runtime): + from winml.modelkit.models import WinMLAutoModel + + model_path = _model(tmp_path) + data = tmp_path / "inputs.npz" + np.savez(data, left=np.zeros((2, 3)), right=np.zeros((2, 3))) + config = BenchmarkConfig( + model_id=str(model_path), + runtime=runtime, + backend="cgc" if runtime == "winml-runtime" else None, + input_data=data, + device="gpu", + ) + benchmark = PerfBenchmark(config) + benchmark._ep_device = SimpleNamespace( + device=SimpleNamespace(ep_name="WinMLCGExecutionProvider") + ) + monkeypatch.setattr(benchmark, "_resolve_device_ep", lambda: None) + session = SimpleNamespace(set_input_shapes=Mock(), compile=Mock()) + wrapper = SimpleNamespace(_session=session) + options = object() + configure = Mock(return_value=options) + monkeypatch.setattr("winml.modelkit.commands.perf._ort_options_for_dimensions", configure) + + def construct(**kwargs): + if runtime == "winml-ort": + assert kwargs["session_options"]() is options + configure.assert_called_once_with({"batch": 2}) + else: + assert "session_options" not in kwargs + return wrapper + + monkeypatch.setattr(WinMLAutoModel, "from_onnx", construct) + benchmark._load_model() + if runtime == "winml-runtime": + session.set_input_shapes.assert_called_once_with({"left": (2, 3), "right": (2, 3)}) + else: + session.set_input_shapes.assert_not_called() + session.compile.assert_not_called() diff --git a/tests/unit/ep_path/test_winml_catalog_source.py b/tests/unit/ep_path/test_winml_catalog_source.py index 4e1790b7e..337d2c6f7 100644 --- a/tests/unit/ep_path/test_winml_catalog_source.py +++ b/tests/unit/ep_path/test_winml_catalog_source.py @@ -409,7 +409,9 @@ def ensure_ready_async( catalog = _FakeCatalog([_StuckProvider()]) # type: ignore[list-item] _install_windowsml_module(monkeypatch, catalog) - source = WinMLCatalogSource(catalog_name="VitisAI", eps=("VitisAIExecutionProvider",)) + source = WinMLCatalogSource( + catalog_name="VitisAI", eps=("VitisAIExecutionProvider",), auto_download=True + ) with caplog.at_level(logging.WARNING, logger="winml.modelkit.ep_path"): assert list(source.resolve()) == [] warn_messages = [r.getMessage() for r in caplog.records if r.levelno == logging.WARNING] @@ -443,7 +445,9 @@ def test_ensure_ready_raises_warns_and_continues( ) _install_windowsml_module(monkeypatch, catalog) - source = WinMLCatalogSource(catalog_name="OpenVINO", eps=("OpenVINOExecutionProvider",)) + source = WinMLCatalogSource( + catalog_name="OpenVINO", eps=("OpenVINOExecutionProvider",), auto_download=True + ) with caplog.at_level(logging.WARNING, logger="winml.modelkit.ep_path"): results = list(source.resolve()) # The good provider should still yield. @@ -490,7 +494,7 @@ def test_epcatalog_constructor_raises_yields_nothing( # --------------------------------------------------------------------------- -def test_not_ready_provider_is_prepared_and_yielded( +def test_not_ready_provider_is_prepared_and_yielded_with_opt_in( reset_catalog_singleton: None, monkeypatch: pytest.MonkeyPatch, tmp_path: Path, @@ -504,6 +508,7 @@ def test_not_ready_provider_is_prepared_and_yielded( WinMLCatalogSource( catalog_name="QNNExecutionProvider", eps=("QNNExecutionProvider",), + auto_download=True, ).resolve() ) @@ -532,14 +537,20 @@ def test_ready_provider_is_not_prepared_again( assert len(entries) == 1 -def test_not_present_provider_is_not_downloaded_by_default( +@pytest.mark.parametrize( + "ready_state", ["NotPresent", "NOT_PRESENT", "NotReady", "NOT_READY", None] +) +def test_non_ready_provider_is_not_prepared_by_default( reset_catalog_singleton: None, monkeypatch: pytest.MonkeyPatch, tmp_path: Path, + ready_state: str | None, ) -> None: dll = tmp_path / "qnn.dll" dll.write_bytes(b"") - provider = _FakeProvider("QNNExecutionProvider", "NotPresent", str(dll)) + provider = _FakeProvider("QNNExecutionProvider", ready_state or "Unknown", str(dll)) + if ready_state is None: + provider.ready_state = None # type: ignore[assignment] catalog = _FakeCatalog([provider]) _install_windowsml_module(monkeypatch, catalog) @@ -570,19 +581,23 @@ def test_not_present_provider_downloads_with_opt_in( assert provider.ensure_ready_calls == 1 +@pytest.mark.parametrize("ready_state", ["NotPresent", "NotReady"]) def test_explicit_catalog_resolution_downloads_requested_provider( reset_catalog_singleton: None, monkeypatch: pytest.MonkeyPatch, tmp_path: Path, + ready_state: str, ) -> None: dll = tmp_path / "qnn.dll" dll.write_bytes(b"") - provider = _FakeProvider("QNNExecutionProvider", "NotPresent", str(dll)) - _install_windowsml_module(monkeypatch, _FakeCatalog([provider])) + provider = _FakeProvider("QNNExecutionProvider", ready_state, str(dll)) + unrelated = _FakeProvider("OpenVINOExecutionProvider", "NotReady", "unrelated.dll") + _install_windowsml_module(monkeypatch, _FakeCatalog([provider, unrelated])) entries = _ep._resolve_requested_winml_catalog_ep("QNNExecutionProvider") assert provider.ensure_ready_calls == 1 + assert unrelated.ensure_ready_calls == 0 assert [entry.dll_path for entry in entries] == [dll] assert all( isinstance(entry.source, WinMLCatalogSource) and entry.source.auto_download @@ -595,6 +610,28 @@ def test_explicit_catalog_resolution_downloads_requested_provider( # --------------------------------------------------------------------------- +def test_discovery_keeps_ready_provider_without_preparing_unrelated_catalog( + reset_catalog_singleton: None, + monkeypatch: pytest.MonkeyPatch, + tmp_path: Path, +) -> None: + ready_dll = tmp_path / "ready.dll" + ready_dll.write_bytes(b"") + unrelated = _FakeProvider("QNNExecutionProvider", "NotReady", "unrelated.dll") + ready = _FakeProvider("WinMLCGExecutionProvider", "Ready", str(ready_dll)) + _install_windowsml_module(monkeypatch, _FakeCatalog([unrelated, ready])) + sources = [WinMLCatalogSource(catalog_name=p.name, eps=(p.name,)) for p in (unrelated, ready)] + monkeypatch.setattr(_ep, "_default_ep_sources", lambda: sources) + monkeypatch.delenv("WINMLCLI_EP_PATH", raising=False) + + entries = _ep.discover_all_eps() + + assert [entry.ep_name for entry in entries] == [ready.name] + assert entries[0].dll_path == ready_dll + assert unrelated.ensure_ready_calls == 0 + assert ready.ensure_ready_calls == 0 + + @pytest.mark.parametrize( ("value", "expected"), [ diff --git a/tests/unit/session/test_runtime_session.py b/tests/unit/session/test_runtime_session.py index 897cac4ec..f5ad8b673 100644 --- a/tests/unit/session/test_runtime_session.py +++ b/tests/unit/session/test_runtime_session.py @@ -24,6 +24,7 @@ _numpy_dtype_for, _ResolvedRuntimeTarget, _shape_with_dynamic_dims, + _symbolic_dimensions_for_inputs, ) @@ -206,6 +207,52 @@ def load_model(path: str) -> Any: session.reset() +@pytest.mark.parametrize("fails", [False, True]) +def test_cgc_compile_uses_public_symbolic_options_and_closes_them(monkeypatch, fails): + io = { + "input_names": ["x"], + "input_shapes": [[None, 3]], + "input_symbolic_shapes": [["batch", 3]], + } + monkeypatch.setattr("winml.modelkit.onnx.get_io_config", lambda _path: io) + options = SimpleNamespace( + supports_symbolic_dimensions=True, symbolic_dimensions={}, close=Mock() + ) + source = Mock() + compiler = Mock(create_options=Mock(return_value=options)) + if fails: + compiler.compile_to_file.side_effect = RuntimeError("compile failed") + runtime = Mock(load_model=Mock(side_effect=[source, _Model()])) + target = _ResolvedRuntimeTarget( + execution_target=SimpleNamespace(model_compiler=lambda: compiler), device_class="gpu" + ) + session = WinMLRuntimeSession("source.onnx", ep_device=_mlir_ep_device(), backend="cgc") + session.set_input_shapes({"x": (2, 3)}) + try: + if fails: + with pytest.raises(RuntimeError, match="compile failed"): + session._load_onnx_on_cgc(runtime, target) + else: + session._load_onnx_on_cgc(runtime, target) + assert options.symbolic_dimensions == {"batch": 2} + assert compiler.compile_to_file.call_args.kwargs == {"options": options} + options.close.assert_called_once_with() + compiler.close.assert_called_once_with() + finally: + session.reset() + + +@pytest.mark.parametrize("shape", [(True, 3), (1.5, 3), (0, 3), (1 << 63, 3)]) +def test_symbolic_dimensions_reject_invalid_extents(shape): + io = { + "input_names": ["x"], + "input_shapes": [[None, 3]], + "input_symbolic_shapes": [["batch", 3]], + } + with pytest.raises(click.ClickException): + _symbolic_dimensions_for_inputs(io, {"x": shape}) + + def test_schema_helpers() -> None: assert _numpy_dtype_for(SimpleNamespace(name="FLOAT16")) == "float16" assert _shape_with_dynamic_dims([0, -1, (1 << 64) - 1, 4]) == [ From 43d25621b77db8f7b314f30cea1997f1a0435f59 Mon Sep 17 00:00:00 2001 From: Qiong Wu Date: Tue, 22 Sep 2026 10:42:08 +0800 Subject: [PATCH 2/3] refactor: limit CGC fix to concrete input shapes --- docs/commands/perf.md | 7 ++- src/winml/modelkit/ep_path.py | 15 +++--- .../unit/ep_path/test_winml_catalog_source.py | 51 +++---------------- 3 files changed, 18 insertions(+), 55 deletions(-) diff --git a/docs/commands/perf.md b/docs/commands/perf.md index e1d59118e..bc8045e88 100644 --- a/docs/commands/perf.md +++ b/docs/commands/perf.md @@ -413,7 +413,6 @@ not supported. ## Common pitfalls -- **Provider discovery and installation.** Discovery lists already-ready catalog providers without preparing unrelated providers. An explicit provider request such as `--ep qnn` retries that provider's catalog with preparation enabled when no installed source is available. `WINMLCLI_EP_PATH` adds source precedence; it does not disable the other discovery sources. - **Warm-up too low on NPU.** The first several inferences on an NPU EP can be significantly slower due to kernel compilation and caching. The default of 10 warm-up iterations is usually enough for vision models, but transformer models with many operators may need `--warmup 30` or higher to reach steady-state latency. - **Hidden third-party diagnostics.** Normal `winml perf` output suppresses noisy native warning-level diagnostics and Hugging Face download/progress chatter so benchmark results stay readable. Use `-v`/`-vv` or set `WINMLCLI_SHOW_ALL_WARNINGS=1` to show those warnings when debugging provider or Hub issues. - **`--input-data` keys must match; dtypes are cast.** The `.npz` keys must equal the model's input names — a missing or unexpected key is a hard error (typo protection). Array dtypes are cast to the model's expected dtype with a warning (matching normal inference), so you don't have to hand-match widths. `.npy` files are not supported — save named arrays as `.npz`. When `--input-data` is set, `--batch-size` and `--shape-config` are ignored (the tensors define their own shapes). It is also rejected for `--module` mode, `--runtime ort-genai`, and composite (dual-encoder) models such as CLIP/SigLIP, where each sub-model has its own inputs that a single `.npz` cannot address. @@ -438,6 +437,6 @@ and dtype conversion still occur at the normal input-allocation boundary. ## See also - [winml eval](eval.md) — measure accuracy after benchmarking -- [winml build](build.md) — build the quantized artifact that perf benchmarks -- [Load and export concept](../concepts/load-and-export.md) — module benchmarking -- [ONNX & Execution Providers](../concepts/eps-and-devices.md) — devices and EPs +- [winml build](build.md) — build the quantized artifact that `perf` benchmarks +- [Load and export concept](../concepts/load-and-export.md) — how `--module` per-instance benchmarking works +- [ONNX & Execution Providers](../concepts/eps-and-devices.md) — understand `--device` vs `--ep` diff --git a/src/winml/modelkit/ep_path.py b/src/winml/modelkit/ep_path.py index f12db474d..bec8bcd35 100644 --- a/src/winml/modelkit/ep_path.py +++ b/src/winml/modelkit/ep_path.py @@ -1124,8 +1124,8 @@ class WinMLCatalogSource(EPSource): eps: Canonical EP names this source provides. Typically a single name, but listed as a tuple for symmetry with the other sources. - auto_download: If ``True``, non-ready providers may be prepared or - downloaded by ``ensure_ready()``. + auto_download: If ``True``, providers in the ``NotPresent`` ready + state will be downloaded by ``ensure_ready()``. Defaults to ``False`` to avoid surprising the user with a multi-second to multi-minute network operation on first call; see ``docs/ep-path-design.md`` Interaction section. @@ -1183,13 +1183,14 @@ def _resolve_provider(self, provider: Any) -> Iterator[EPEntry]: if getattr(provider, "name", None) != self.catalog_name: return - # Discovery must not prepare unrelated providers. NotReady can also - # require installation; reserve every readiness transition for the - # explicit-provider acquisition path (auto_download=True). + # Skip providers that are not present on this machine. The design + # doc explicitly forbids auto-downloading hundreds of MB without + # opt-in; we honor that via auto_download=False (the default). ready_state = getattr(provider, "ready_state", None) - if not self.auto_download and not self._is_ready(ready_state): + if ready_state is not None and not self.auto_download and self._is_not_present(ready_state): logger.debug( - "WinMLCatalogSource(%s): provider is not ready; skipping (auto_download=False)", + "WinMLCatalogSource(%s): provider in NotPresent state; " + "skipping (auto_download=False)", self.catalog_name, ) return diff --git a/tests/unit/ep_path/test_winml_catalog_source.py b/tests/unit/ep_path/test_winml_catalog_source.py index 337d2c6f7..4e1790b7e 100644 --- a/tests/unit/ep_path/test_winml_catalog_source.py +++ b/tests/unit/ep_path/test_winml_catalog_source.py @@ -409,9 +409,7 @@ def ensure_ready_async( catalog = _FakeCatalog([_StuckProvider()]) # type: ignore[list-item] _install_windowsml_module(monkeypatch, catalog) - source = WinMLCatalogSource( - catalog_name="VitisAI", eps=("VitisAIExecutionProvider",), auto_download=True - ) + source = WinMLCatalogSource(catalog_name="VitisAI", eps=("VitisAIExecutionProvider",)) with caplog.at_level(logging.WARNING, logger="winml.modelkit.ep_path"): assert list(source.resolve()) == [] warn_messages = [r.getMessage() for r in caplog.records if r.levelno == logging.WARNING] @@ -445,9 +443,7 @@ def test_ensure_ready_raises_warns_and_continues( ) _install_windowsml_module(monkeypatch, catalog) - source = WinMLCatalogSource( - catalog_name="OpenVINO", eps=("OpenVINOExecutionProvider",), auto_download=True - ) + source = WinMLCatalogSource(catalog_name="OpenVINO", eps=("OpenVINOExecutionProvider",)) with caplog.at_level(logging.WARNING, logger="winml.modelkit.ep_path"): results = list(source.resolve()) # The good provider should still yield. @@ -494,7 +490,7 @@ def test_epcatalog_constructor_raises_yields_nothing( # --------------------------------------------------------------------------- -def test_not_ready_provider_is_prepared_and_yielded_with_opt_in( +def test_not_ready_provider_is_prepared_and_yielded( reset_catalog_singleton: None, monkeypatch: pytest.MonkeyPatch, tmp_path: Path, @@ -508,7 +504,6 @@ def test_not_ready_provider_is_prepared_and_yielded_with_opt_in( WinMLCatalogSource( catalog_name="QNNExecutionProvider", eps=("QNNExecutionProvider",), - auto_download=True, ).resolve() ) @@ -537,20 +532,14 @@ def test_ready_provider_is_not_prepared_again( assert len(entries) == 1 -@pytest.mark.parametrize( - "ready_state", ["NotPresent", "NOT_PRESENT", "NotReady", "NOT_READY", None] -) -def test_non_ready_provider_is_not_prepared_by_default( +def test_not_present_provider_is_not_downloaded_by_default( reset_catalog_singleton: None, monkeypatch: pytest.MonkeyPatch, tmp_path: Path, - ready_state: str | None, ) -> None: dll = tmp_path / "qnn.dll" dll.write_bytes(b"") - provider = _FakeProvider("QNNExecutionProvider", ready_state or "Unknown", str(dll)) - if ready_state is None: - provider.ready_state = None # type: ignore[assignment] + provider = _FakeProvider("QNNExecutionProvider", "NotPresent", str(dll)) catalog = _FakeCatalog([provider]) _install_windowsml_module(monkeypatch, catalog) @@ -581,23 +570,19 @@ def test_not_present_provider_downloads_with_opt_in( assert provider.ensure_ready_calls == 1 -@pytest.mark.parametrize("ready_state", ["NotPresent", "NotReady"]) def test_explicit_catalog_resolution_downloads_requested_provider( reset_catalog_singleton: None, monkeypatch: pytest.MonkeyPatch, tmp_path: Path, - ready_state: str, ) -> None: dll = tmp_path / "qnn.dll" dll.write_bytes(b"") - provider = _FakeProvider("QNNExecutionProvider", ready_state, str(dll)) - unrelated = _FakeProvider("OpenVINOExecutionProvider", "NotReady", "unrelated.dll") - _install_windowsml_module(monkeypatch, _FakeCatalog([provider, unrelated])) + provider = _FakeProvider("QNNExecutionProvider", "NotPresent", str(dll)) + _install_windowsml_module(monkeypatch, _FakeCatalog([provider])) entries = _ep._resolve_requested_winml_catalog_ep("QNNExecutionProvider") assert provider.ensure_ready_calls == 1 - assert unrelated.ensure_ready_calls == 0 assert [entry.dll_path for entry in entries] == [dll] assert all( isinstance(entry.source, WinMLCatalogSource) and entry.source.auto_download @@ -610,28 +595,6 @@ def test_explicit_catalog_resolution_downloads_requested_provider( # --------------------------------------------------------------------------- -def test_discovery_keeps_ready_provider_without_preparing_unrelated_catalog( - reset_catalog_singleton: None, - monkeypatch: pytest.MonkeyPatch, - tmp_path: Path, -) -> None: - ready_dll = tmp_path / "ready.dll" - ready_dll.write_bytes(b"") - unrelated = _FakeProvider("QNNExecutionProvider", "NotReady", "unrelated.dll") - ready = _FakeProvider("WinMLCGExecutionProvider", "Ready", str(ready_dll)) - _install_windowsml_module(monkeypatch, _FakeCatalog([unrelated, ready])) - sources = [WinMLCatalogSource(catalog_name=p.name, eps=(p.name,)) for p in (unrelated, ready)] - monkeypatch.setattr(_ep, "_default_ep_sources", lambda: sources) - monkeypatch.delenv("WINMLCLI_EP_PATH", raising=False) - - entries = _ep.discover_all_eps() - - assert [entry.ep_name for entry in entries] == [ready.name] - assert entries[0].dll_path == ready_dll - assert unrelated.ensure_ready_calls == 0 - assert ready.ensure_ready_calls == 0 - - @pytest.mark.parametrize( ("value", "expected"), [ From 189ae94ad69dc016fcdb382066d6a140bbf7e63d Mon Sep 17 00:00:00 2001 From: Qiong Wu Date: Tue, 22 Sep 2026 14:47:22 +0800 Subject: [PATCH 3/3] fix(perf): preserve compatible inputs during CGC shape binding --- src/winml/modelkit/commands/perf.py | 19 +++++----- src/winml/modelkit/session/runtime_session.py | 15 ++++---- .../unit/commands/test_perf_runtime_shapes.py | 35 +++++++++++++++++++ 3 files changed, 55 insertions(+), 14 deletions(-) diff --git a/src/winml/modelkit/commands/perf.py b/src/winml/modelkit/commands/perf.py index dd179fe82..ee4bf2cec 100644 --- a/src/winml/modelkit/commands/perf.py +++ b/src/winml/modelkit/commands/perf.py @@ -65,7 +65,6 @@ from ..session.monitor import ProcessMemoryTracker from ..session.monitor.ep_monitor import WinMLEPMonitor from ..session.monitor.op_metrics import TraceFallbackReason - from ..session.runtime_session import WinMLRuntimeSession from ..session.stats import PerfStats logger = logging.getLogger(__name__) @@ -933,12 +932,12 @@ def _runtime_input_shapes( for name in io_config["input_names"]: with archive.open(name + ".npy") as stream: version = np.lib.format.read_magic(stream) - if version == (1, 0): - shape, _, dtype = np.lib.format.read_array_header_1_0(stream) - elif version == (2, 0): - shape, _, dtype = np.lib.format.read_array_header_2_0(stream) - else: - raise ValueError(f"Unsupported NPY header version {version}") + # NumPy has no public v3 header reader. Use the same + # version-aware, size-limited reader as np.load, without + # reading or allocating the array payload. + shape, _, dtype = np.lib.format._read_array_header( # type: ignore[attr-defined] + stream, version + ) if dtype.hasobject: raise ValueError("object input arrays are unsupported") shapes[name] = shape @@ -1451,7 +1450,11 @@ def _load_model(self) -> None: **common_kwargs, ) if runtime_cgc and runtime_shapes: - cast("WinMLRuntimeSession", self._single._session).set_input_shapes(runtime_shapes) + from ..session.runtime_session import WinMLRuntimeSession + + # Keep the lazy import visible to CodeQL as well as type checkers. + runtime_session = cast(WinMLRuntimeSession, self._single._session) # noqa: TC006 + runtime_session.set_input_shapes(runtime_shapes) elif is_mlir: with suppress_native_warnings(enabled=True): self._model = WinMLAutoModel.from_mlir( diff --git a/src/winml/modelkit/session/runtime_session.py b/src/winml/modelkit/session/runtime_session.py index 4f4fe9142..3f19d8db8 100644 --- a/src/winml/modelkit/session/runtime_session.py +++ b/src/winml/modelkit/session/runtime_session.py @@ -268,18 +268,21 @@ def _symbolic_dimensions_for_inputs( raise click.ClickException( f"Input {name!r} axis {axis} must be an integer." ) from exc - if isinstance(value, bool) or not 0 < extent <= (1 << 63) - 1: - raise click.ClickException(f"Input {name!r} axis {axis} must be a positive int64.") + if isinstance(value, bool) or not 0 <= extent <= (1 << 63) - 1: + raise click.ClickException( + f"Input {name!r} axis {axis} must be a nonnegative int64." + ) if fixed is not None: if extent != fixed: raise click.ClickException( f"Input {name!r} axis {axis} is {extent}; ONNX requires {fixed}." ) elif not isinstance(symbol, str) or not symbol: - raise click.ClickException( - f"Input {name!r} axis {axis} has no symbolic dimension name; " - "the compiler's named-dimension API cannot bind it." - ) + # Anonymous axes cannot be bound by name. Let the compiler decide + # whether they need specialization (unused inputs may not). + continue + elif extent == 0: + raise click.ClickException(f"Input {name!r} axis {axis} must be a positive int64.") elif symbol in overrides and overrides[symbol] != extent: raise click.ClickException( f"Conflicting concrete sizes for symbolic dimension {symbol!r}." diff --git a/tests/unit/commands/test_perf_runtime_shapes.py b/tests/unit/commands/test_perf_runtime_shapes.py index c5fe28059..059479b59 100644 --- a/tests/unit/commands/test_perf_runtime_shapes.py +++ b/tests/unit/commands/test_perf_runtime_shapes.py @@ -4,9 +4,11 @@ # -------------------------------------------------------------------------- """Concrete compilation shape handoff, with no native execution.""" +from io import BytesIO from pathlib import Path from types import SimpleNamespace from unittest.mock import Mock +from zipfile import ZipFile import click import numpy as np @@ -55,6 +57,39 @@ def test_shape_config_resolves_symbols_without_inputs(tmp_path): assert dimensions == {"batch": 4} +@pytest.mark.parametrize( + "unused_shape, actual_shape", [([None, 3], (2, 3)), (["extent", 0], (2, 0))] +) +def test_unused_input_preserves_anonymous_and_static_zero_axes( + tmp_path, unused_shape, actual_shape +): + path = _model(tmp_path) + model = onnx.load(path) + model.graph.input.append( + onnx.helper.make_tensor_value_info("unused", onnx.TensorProto.FLOAT, unused_shape) + ) + onnx.save(model, path) + data = tmp_path / "inputs.npz" + np.savez(data, left=np.zeros((2, 3)), right=np.zeros((2, 3)), unused=np.zeros(actual_shape)) + shapes, dimensions = _runtime_input_shapes(path, data, None, 1) + assert shapes["unused"] == actual_shape + assert dimensions == ({"batch": 2, "extent": 2} if actual_shape[1] == 0 else {"batch": 2}) + + +@pytest.mark.parametrize("version", [(1, 0), (2, 0), (3, 0)]) +def test_npz_header_versions_without_reading_payload(tmp_path, version): + data = tmp_path / "headers.npz" + with ZipFile(data, "w") as archive: + for name in ("left", "right"): + stream = BytesIO() + np.lib.format.write_array(stream, np.zeros((2, 3), dtype=np.float32), version=version) + # Omit the payload: shape discovery must only read the header. + archive.writestr(name + ".npy", stream.getvalue()[:-24]) + shapes, dimensions = _runtime_input_shapes(_model(tmp_path), data, None, 1) + assert shapes == {"left": (2, 3), "right": (2, 3)} + assert dimensions == {"batch": 2} + + @pytest.mark.parametrize("right_shape", [(3, 3), (2, 4), (2, 3, 1)]) def test_npz_rejects_symbol_conflict_static_mismatch_and_rank(tmp_path, right_shape): data = tmp_path / "inputs.npz"