From 2c96e075ec157ab014a45f47337bd5538abdf02b Mon Sep 17 00:00:00 2001 From: Supreet Singh Palne Date: Thu, 1 Oct 2026 14:19:50 -0700 Subject: [PATCH 1/6] Add Openvino support --- npu_config_npuw.json | 12 + src/winml/modelkit/commands/perf.py | 19 ++ .../modelkit/models/hf/qwen3/__init__.py | 4 + src/winml/modelkit/models/hf/qwen3/genai.py | 183 ++++++++++++- .../modelkit/models/winml/genai_bundle.py | 4 + tests/unit/commands/test_perf_genai.py | 64 +++++ tests/unit/models/qwen3/test_genai_config.py | 253 ++++++++++++++++++ .../winml/test_genai_bundle_registry.py | 8 + 8 files changed, 538 insertions(+), 9 deletions(-) create mode 100644 npu_config_npuw.json diff --git a/npu_config_npuw.json b/npu_config_npuw.json new file mode 100644 index 000000000..69154cec2 --- /dev/null +++ b/npu_config_npuw.json @@ -0,0 +1,12 @@ +{ + "NPU": { + "NPU_TURBO": "YES", + "NPU_QDQ_OPTIMIZATION": "YES", + "NPU_COMPILER_TYPE": "DRIVER", + "NPU_USE_NPUW": "YES", + "NPUW_DEVICES": "NPU", + "NPUW_PARALLEL_COMPILE": "YES", + "NPUW_GQA": "YES", + "CACHE_MODE": "OPTIMIZE_SPEED" + } +} \ No newline at end of file diff --git a/src/winml/modelkit/commands/perf.py b/src/winml/modelkit/commands/perf.py index c1395a950..36652c89e 100644 --- a/src/winml/modelkit/commands/perf.py +++ b/src/winml/modelkit/commands/perf.py @@ -2504,6 +2504,14 @@ def _autobuild_genai_bundle( build_ep, build_device = short_ep_name(target.ep), target.device # Do not reuse a bundle exported for a different execution provider. bundle_dir = bundle_dir.with_name(f"genai-bundle-{build_ep}-{build_device}") + openvino_config: Path | None = p.get("openvino_config") + if openvino_config is not None: + if build_ep != "openvino": + raise click.UsageError("--openvino-config requires --ep openvino.") + import hashlib + + config_digest = hashlib.sha256(openvino_config.read_bytes()).hexdigest()[:12] + bundle_dir = bundle_dir.with_name(f"{bundle_dir.name}-config-{config_digest}") build_cache_dir = cache_dir # --rebuild overwrites the cached bundle; a plain run reuses it. Checked # before any model resolution so a cache hit never touches the network. @@ -2564,6 +2572,9 @@ def _autobuild_genai_bundle( force_rebuild=force_rebuild, cache_dir=build_cache_dir, emit=lambda msg: console.print(msg, markup=False), + assemble_options=( + {"openvino_config_path": openvino_config} if openvino_config is not None else None + ), ) return bundle_dir, True @@ -2784,6 +2795,13 @@ def _validate_duration( help="[ort-genai] Max seconds to compile each EPContext stage before falling back " "to the original ONNX (requires --compile).", ) +@click.option( + "--openvino-config", + type=click.Path(exists=True, dir_okay=False, path_type=Path), + default=None, + help="[ort-genai] OpenVINO load-config JSON embedded while auto-building a model ID. " + "Requires --ep openvino.", +) @click.option( "--task", type=str, @@ -2964,6 +2982,7 @@ def perf( apply_template: bool, max_new_tokens: int, compile_timeout: int, + openvino_config: Path | None, task: str | None, submodel: str | None, iterations: int, diff --git a/src/winml/modelkit/models/hf/qwen3/__init__.py b/src/winml/modelkit/models/hf/qwen3/__init__.py index 6e60e07ee..0e66b9700 100644 --- a/src/winml/modelkit/models/hf/qwen3/__init__.py +++ b/src/winml/modelkit/models/hf/qwen3/__init__.py @@ -17,7 +17,9 @@ PipelineStage, build_decoder_pipeline_stages, build_genai_config, + build_npu_load_config, build_qwen3_transformer_only_stages, + openvino_stage_session_options, strip_gqa_default_attrs, write_genai_bundle, ) @@ -28,7 +30,9 @@ "PipelineStage", "build_decoder_pipeline_stages", "build_genai_config", + "build_npu_load_config", "build_qwen3_transformer_only_stages", + "openvino_stage_session_options", "strip_gqa_default_attrs", "write_genai_bundle", ] diff --git a/src/winml/modelkit/models/hf/qwen3/genai.py b/src/winml/modelkit/models/hf/qwen3/genai.py index 9b661fd30..ceb5aab67 100644 --- a/src/winml/modelkit/models/hf/qwen3/genai.py +++ b/src/winml/modelkit/models/hf/qwen3/genai.py @@ -11,11 +11,12 @@ This module adds the **Qwen3-specific** layer on top: the Qwen3 transformer stages run on an NPU backend, so this is where the per-EP ``session_options`` -are constructed. Two NPU execution providers are supported for the +are constructed. Three NPU execution providers are supported for the transformer (context/iterator) stages: * **QNN HTP** — Qualcomm Snapdragon NPU (``ep="qnn"``). * **VitisAI** — AMD Ryzen AI NPU (``ep="vitisai"``). +* **OpenVINO** — Intel NPU via the plugin EP (``ep="openvino"``). Keeping the EP-specific logic here lets the generic utilities stay universal while the Qwen3 bundle emits the correct per-EP ``genai_config.json``. @@ -23,6 +24,8 @@ from __future__ import annotations +import json +from pathlib import Path from typing import TYPE_CHECKING from ....onnx import strip_node_attrs @@ -51,15 +54,16 @@ if TYPE_CHECKING: from collections.abc import Callable, Sequence - from pathlib import Path import onnx # --------------------------------------------------------------------------- -# Qwen3-specific NPU execution-provider routing (QNN / VitisAI) +# Qwen3-specific NPU execution-provider routing (QNN / VitisAI / OpenVINO) # --------------------------------------------------------------------------- +_OPENVINO_CONFIG_ROLE_KEYS = frozenset({"CTX", "ITER", "HEAD"}) + def qnn_stage_session_options(log_id: str, soc_model: str = "60") -> dict: """Return the ``session_options`` block that routes a stage to QNN HTP. @@ -124,19 +128,134 @@ def vitisai_stage_session_options(log_id: str) -> dict: } -def _stage_session_options(ep: str, soc_model: str) -> tuple[dict | None, dict | None]: +def build_npu_load_config( + custom_config_path: str | Path | None = None, + *, + model_role: str | None = None, + weights_path: str | Path | None = None, +) -> str: + """Build the JSON-string ``load_config`` for the OpenVINO NPU plugin. + + Accepts a flat configuration (e.g. ``{"NPU": {...}}``) or a configuration + with ``CTX``/``ITER`` sections. A role-based file must contain the requested + role; missing sections are errors rather than silently using defaults. + ``HEAD`` is recognized only to detect role-based files; the LM head stays + on CPU. No legacy provider setup or tuning flags are injected. + + Args: + custom_config_path: Optional UTF-8 JSON file. Without one, only the + driver compiler default and an explicitly supplied weights path + are included. Custom values take precedence over defaults. + model_role: ``"CTX"`` or ``"ITER"``; required for role-based files. + weights_path: Default external-weights directory. A ``WEIGHTS_PATH`` + in the file takes precedence; relative file values resolve against + that file's directory. Emitted paths are absolute so loading a + derived bundle does not change their meaning. + + Returns: + Serialized OpenVINO load configuration, not a filename. + """ + if model_role not in (None, "CTX", "ITER"): + raise ValueError("OpenVINO model_role must be 'CTX' or 'ITER'") + + config: dict = {} + config_path = ( + Path(custom_config_path).expanduser().resolve() if custom_config_path is not None else None + ) + if config_path is not None: + try: + config = json.loads(config_path.read_text(encoding="utf-8-sig")) + except json.JSONDecodeError as exc: + raise ValueError(f"Invalid OpenVINO load config JSON in {config_path}: {exc}") from exc + if not isinstance(config, dict): + raise TypeError("OpenVINO load config must be a JSON object") + if _OPENVINO_CONFIG_ROLE_KEYS.intersection(config): + if model_role is None: + raise ValueError("A role-based OpenVINO load config requires model_role") + if model_role not in config: + raise ValueError(f"OpenVINO load config is missing the {model_role} section") + config = config[model_role] + if not isinstance(config, dict): + raise TypeError(f"OpenVINO {model_role} configuration must be a JSON object") + + npu_config = config.setdefault("NPU", {}) + if not isinstance(npu_config, dict): + raise TypeError("OpenVINO NPU configuration must be a JSON object") + npu_config.setdefault("NPU_COMPILER_TYPE", "DRIVER") + + if "WEIGHTS_PATH" in npu_config: + configured_weights = npu_config["WEIGHTS_PATH"] + if not isinstance(configured_weights, str) or not configured_weights.strip(): + raise ValueError("OpenVINO WEIGHTS_PATH must be a non-empty string") + resolved_weights = Path(configured_weights).expanduser() + if not resolved_weights.is_absolute() and config_path is not None: + resolved_weights = config_path.parent / resolved_weights + npu_config["WEIGHTS_PATH"] = str(resolved_weights.resolve()) + elif weights_path is not None: + npu_config["WEIGHTS_PATH"] = str(Path(weights_path).expanduser().resolve()) + + # Stable serialization keeps equivalent CTX/ITER options shareable even + # when their keys occur in a different order in the user's JSON file. + return json.dumps(config, sort_keys=True) + + +def openvino_stage_session_options( + log_id: str, + *, + custom_config_path: str | Path | None = None, + model_role: str | None = None, + weights_path: str | Path | None = None, +) -> dict: + """Return session options routing a transformer stage to the Intel NPU. + + ``load_config`` is a JSON string containing OpenVINO properties, not ORT + session entries. Plugin registration, ABI device binding, and EPContext + compilation remain owned by the existing session/compiler infrastructure. + See :func:`build_npu_load_config` for the optional configuration arguments. + """ + return { + "log_id": log_id, + "provider_options": [ + { + "openvino": { + "device_type": "NPU", + "load_config": build_npu_load_config( + custom_config_path, + model_role=model_role, + weights_path=weights_path, + ), + } + } + ], + "intra_op_num_threads": 2, + "inter_op_num_threads": 1, + } + + +def _stage_session_options( + ep: str, + soc_model: str, + *, + openvino_config_path: str | Path | None = None, + openvino_weights_path: str | Path | None = None, +) -> tuple[dict | None, dict | None]: """Return ``(context, iterator)`` session_options for the given EP. Routes the Qwen3 transformer (context/iterator) stages to an NPU backend: * ``ep="qnn"`` -> Qualcomm QNN HTP (``soc_model`` selects the Snapdragon SoC). * ``ep="vitisai"`` -> AMD Ryzen AI NPU. + * ``ep="openvino"`` -> Intel NPU, with optional per-role load configuration. Any other value (e.g. ``"cpu"``) leaves the stages on the default CPU provider. Short aliases and full ``*ExecutionProvider`` names are both accepted (normalized via :func:`normalize_ep_name`). """ canonical = normalize_ep_name(ep) + if canonical != "OpenVINOExecutionProvider" and ( + openvino_config_path is not None or openvino_weights_path is not None + ): + raise ValueError("OpenVINO configuration requires ep='openvino'") if canonical == "QNNExecutionProvider": return ( qnn_stage_session_options("onnxruntime-genai.context", soc_model=soc_model), @@ -147,6 +266,21 @@ def _stage_session_options(ep: str, soc_model: str) -> tuple[dict | None, dict | vitisai_stage_session_options("onnxruntime-genai.context"), vitisai_stage_session_options("onnxruntime-genai.iterator"), ) + if canonical == "OpenVINOExecutionProvider": + return ( + openvino_stage_session_options( + "onnxruntime-genai.context", + custom_config_path=openvino_config_path, + model_role="CTX", + weights_path=openvino_weights_path, + ), + openvino_stage_session_options( + "onnxruntime-genai.iterator", + custom_config_path=openvino_config_path, + model_role="ITER", + weights_path=openvino_weights_path, + ), + ) return None, None @@ -190,6 +324,8 @@ def build_qwen3_transformer_only_stages( lm_head_filename: str = DEFAULT_LM_HEAD_FILENAME, ep: str = "cpu", soc_model: str = "60", + openvino_config_path: str | Path | None = None, + openvino_weights_path: str | Path | None = None, ) -> tuple[list[PipelineStage], DecoderIOMapping]: """Build the Qwen3 4-stage pipeline, routing ctx/iter to the NPU per ``ep``. @@ -207,19 +343,28 @@ def build_qwen3_transformer_only_stages( embeddings_filename: Bundle filename for the embeddings model. lm_head_filename: Bundle filename for the lm_head model. ep: NPU execution provider for the ``context``/``iterator`` stages — - ``"qnn"`` (Qualcomm) or ``"vitisai"`` (AMD) injects that EP's - ``session_options`` so those stages run on the NPU while + ``"qnn"`` (Qualcomm), ``"vitisai"`` (AMD), or ``"openvino"`` (Intel) + injects that EP's ``session_options`` so those stages run on the NPU while ``embeddings`` and ``lm_head`` stay on CPU. ``"cpu"`` (default) omits them. soc_model: Snapdragon SoC model number forwarded to the QNN backend when ``ep="qnn"``. Default ``"60"`` targets Snapdragon 8 Gen 3. Ignored for non-QNN EPs. + openvino_config_path: Optional OpenVINO JSON file shared by both stages + or containing separate ``CTX``/``ITER`` sections. Only for OpenVINO. + openvino_weights_path: Optional default external-weights directory for + OpenVINO. Explicit ``WEIGHTS_PATH`` values in the JSON take precedence. Returns: ``(stages, decoder_io)`` — see :func:`~winml.modelkit.utils.genai.build_decoder_pipeline_stages`. """ - ctx_opts, iter_opts = _stage_session_options(ep, soc_model) + ctx_opts, iter_opts = _stage_session_options( + ep, + soc_model, + openvino_config_path=openvino_config_path, + openvino_weights_path=openvino_weights_path, + ) return build_decoder_pipeline_stages( context_onnx, iterator_onnx, @@ -250,6 +395,8 @@ def write_genai_bundle( ep: str = "cpu", soc_model: str = "60", transformer_onnx_passes: Sequence[Callable[[onnx.ModelProto], onnx.ModelProto]] | None = None, + openvino_config_path: str | Path | None = None, + openvino_weights_path: str | Path | None = None, ) -> Path: """Assemble a Qwen3 genai bundle, routing ctx/iter to the NPU per ``ep``. @@ -260,7 +407,8 @@ def write_genai_bundle( Args: ep: NPU execution provider routing the transformer (context/iterator) - stages — ``"qnn"`` (Qualcomm HTP) or ``"vitisai"`` (AMD Ryzen AI); + stages — ``"qnn"`` (Qualcomm HTP), ``"vitisai"`` (AMD Ryzen AI), + or ``"openvino"`` (Intel NPU); ``"cpu"`` (default) keeps every stage on CPU. soc_model: Snapdragon SoC model passed to the QNN backend when ``ep="qnn"``. Default ``"60"`` = Snapdragon 8 Gen 3 / X Elite. @@ -268,11 +416,25 @@ def write_genai_bundle( transformer_onnx_passes: Optional ONNX graph transforms applied to the copied context/iterator models before ``genai_config.json`` is written. Forwarded verbatim to the generic assembler. + openvino_config_path: Optional flat or per-role OpenVINO JSON file. + Its contents are embedded in stage provider options; the original + configuration file is not required at runtime. Different role + options prevent grouped compilation in the current runtime. + openvino_weights_path: Default external-weights directory for OpenVINO; + defaults to the output bundle directory. Explicit ``WEIGHTS_PATH`` + values in the JSON take precedence. Other EPs reject these options. Returns: Path to the written ``genai_config.json``. """ - ctx_opts, iter_opts = _stage_session_options(ep, soc_model) + if normalize_ep_name(ep) == "OpenVINOExecutionProvider" and openvino_weights_path is None: + openvino_weights_path = output_dir + ctx_opts, iter_opts = _stage_session_options( + ep, + soc_model, + openvino_config_path=openvino_config_path, + openvino_weights_path=openvino_weights_path, + ) return _write_genai_bundle( output_dir, context_onnx=context_onnx, @@ -302,7 +464,9 @@ def write_genai_bundle( "PipelineStage", "build_decoder_pipeline_stages", "build_genai_config", + "build_npu_load_config", "build_qwen3_transformer_only_stages", + "openvino_stage_session_options", "qnn_stage_session_options", "strip_gqa_default_attrs", "vitisai_stage_session_options", @@ -347,6 +511,7 @@ def write_genai_bundle( supported_targets=( GenaiTarget(ep="qnn", device="npu"), # Qualcomm Snapdragon NPU GenaiTarget(ep="vitisai", device="npu"), # AMD Ryzen AI NPU + GenaiTarget(ep="openvino", device="npu"), # Intel NPU GenaiTarget(ep="cpu", device="cpu"), ), transformer_onnx_passes=(strip_gqa_default_attrs,), diff --git a/src/winml/modelkit/models/winml/genai_bundle.py b/src/winml/modelkit/models/winml/genai_bundle.py index 419f48908..3d5cce491 100644 --- a/src/winml/modelkit/models/winml/genai_bundle.py +++ b/src/winml/modelkit/models/winml/genai_bundle.py @@ -246,6 +246,7 @@ def build_genai_bundle( force_rebuild: bool = False, cache_dir: str | Path | None = None, emit: Callable[[str], None] | None = None, + assemble_options: Mapping[str, object] | None = None, ) -> Path: """Build (or reuse) every bundle component and assemble the genai bundle. @@ -270,6 +271,8 @@ def build_genai_bundle( force_rebuild: Rebuild components even if cached. cache_dir: Build cache directory override. emit: Optional progress sink invoked with human-readable status lines. + assemble_options: Model-specific keyword arguments forwarded to the + registered bundle assembler. Returns: Path to the written ``genai_config.json``. @@ -382,6 +385,7 @@ def build_genai_bundle( soc_model=soc_model, transformer_onnx_passes=list(recipe.transformer_onnx_passes), **{f"{role}_src": path for role, path in companion_srcs.items()}, + **dict(assemble_options or {}), ) _emit(f" genai_config.json -> {config_path}") return Path(config_path) diff --git a/tests/unit/commands/test_perf_genai.py b/tests/unit/commands/test_perf_genai.py index 626cb361c..797cae26c 100644 --- a/tests/unit/commands/test_perf_genai.py +++ b/tests/unit/commands/test_perf_genai.py @@ -1458,6 +1458,70 @@ def test_hf_model_id_autobuilds_and_dispatches( assert cfg.device == "config" assert cfg.ep is None + def test_openvino_config_is_forwarded_to_autobuild( + self, runner: CliRunner, tmp_path: Path, capture_run: dict, monkeypatch + ) -> None: + import winml.modelkit.loader as loader_mod + import winml.modelkit.models.winml as winml_models + + config_path = tmp_path / "npu.json" + config_path.write_text('{"NPU": {"NPU_USE_NPUW": "YES"}}', encoding="utf-8") + monkeypatch.setenv("WINML_CACHE_DIR", str(tmp_path / "cache")) + monkeypatch.setattr( + loader_mod, "resolve_loader_config", _fake_resolve_loader_config("qwen3") + ) + build_calls: dict = {} + monkeypatch.setattr( + winml_models, "build_genai_bundle", _fake_build_genai_bundle(build_calls) + ) + + result = runner.invoke( + perf, + [ + "-m", + "Qwen/Qwen3-0.6B", + "--runtime", + "ort-genai", + "--ep", + "openvino", + "--device", + "npu", + "--openvino-config", + str(config_path), + ], + ) + + assert result.exit_code == 0, result.output + assert build_calls["build"]["assemble_options"] == { + "openvino_config_path": config_path + } + assert "-config-" in build_calls["build"]["output_dir"].name + assert capture_run["config"].bundle_dir == build_calls["build"]["output_dir"] + + def test_openvino_config_rejects_other_ep( + self, runner: CliRunner, tmp_path: Path, capture_run: dict + ) -> None: + config_path = tmp_path / "npu.json" + config_path.write_text("{}", encoding="utf-8") + + result = runner.invoke( + perf, + [ + "-m", + "Qwen/Qwen3-0.6B", + "--runtime", + "ort-genai", + "--ep", + "cpu", + "--openvino-config", + str(config_path), + ], + ) + + assert result.exit_code == 2, result.output + assert "--openvino-config requires --ep openvino" in result.output + assert "config" not in capture_run + @pytest.mark.parametrize( ("args", "resolved_ep", "build_ep", "build_device"), [ diff --git a/tests/unit/models/qwen3/test_genai_config.py b/tests/unit/models/qwen3/test_genai_config.py index 63dc7f78e..7683f3f8b 100644 --- a/tests/unit/models/qwen3/test_genai_config.py +++ b/tests/unit/models/qwen3/test_genai_config.py @@ -6,15 +6,21 @@ from __future__ import annotations +import json +from pathlib import Path from types import SimpleNamespace from typing import ClassVar from unittest.mock import patch +import pytest + from winml.modelkit.models.hf.qwen3 import ( DecoderIOMapping, PipelineStage, build_genai_config, + build_npu_load_config, build_qwen3_transformer_only_stages, + openvino_stage_session_options, write_genai_bundle, ) from winml.modelkit.models.hf.qwen3.genai import ( @@ -31,6 +37,16 @@ # --------------------------------------------------------------------------- +@pytest.fixture +def openvino_config_file(tmp_path): + def write_config(payload): + path = tmp_path / "openvino.json" + path.write_text(json.dumps(payload), encoding="utf-8") + return path + + return write_config + + def _mock_config( *, num_hidden_layers: int = 28, @@ -387,6 +403,137 @@ def test_single_layer_model(self) -> None: assert result == {"keys_": "keys_%d", "vals_": "vals_%d"} +# --------------------------------------------------------------------------- +# Tests: OpenVINO NPU options and load configuration +# --------------------------------------------------------------------------- + + +class TestOpenVINOSessionOptions: + def test_minimal_npu_options_without_custom_config(self) -> None: + options = openvino_stage_session_options("test.context") + assert options["log_id"] == "test.context" + assert options["intra_op_num_threads"] == 2 + assert options["inter_op_num_threads"] == 1 + provider = options["provider_options"][0]["openvino"] + assert provider["device_type"] == "NPU" + assert json.loads(provider["load_config"]) == {"NPU": {"NPU_COMPILER_TYPE": "DRIVER"}} + assert set(provider) == {"device_type", "load_config"} + + def test_flat_config_preserves_custom_settings(self, openvino_config_file) -> None: + payload = { + "NPU": { + "NPU_COMPILER_TYPE": "custom-compiler", + "NPU_TURBO": "NO", + "NPU_QDQ_OPTIMIZATION": "YES", + } + } + path = openvino_config_file(payload) + for role in ("CTX", "ITER"): + assert json.loads(build_npu_load_config(path, model_role=role)) == payload + assert json.loads(path.read_text(encoding="utf-8")) == payload + + @pytest.mark.parametrize("role", ["CTX", "ITER"]) + def test_selects_role_without_forwarding_other_sections( + self, openvino_config_file, role + ) -> None: + payload = { + "CTX": {"NPU": {"NPU_TURBO": "YES"}}, + "ITER": {"NPU": {"NPU_TURBO": "NO"}}, + "HEAD": {"NPU": {"NPU_TURBO": "YES"}}, + } + path = openvino_config_file(payload) + config = json.loads(build_npu_load_config(path, model_role=role)) + assert set(config) == {"NPU"} + assert config["NPU"] == { + **payload[role]["NPU"], + "NPU_COMPILER_TYPE": "DRIVER", + } + + def test_equivalent_role_configs_serialize_identically(self, openvino_config_file) -> None: + properties = {"NPU_TURBO": "YES", "NPU_QDQ_OPTIMIZATION": "YES"} + path = openvino_config_file( + { + "CTX": {"NPU": properties}, + "ITER": {"NPU": dict(reversed(list(properties.items())))}, + } + ) + assert build_npu_load_config(path, model_role="CTX") == build_npu_load_config( + path, model_role="ITER" + ) + + def test_weights_default_is_absolute(self, tmp_path, monkeypatch) -> None: + monkeypatch.chdir(tmp_path) + config = json.loads(build_npu_load_config(weights_path="model weights")) + path = Path(config["NPU"]["WEIGHTS_PATH"]) + assert path.is_absolute() + assert path == tmp_path / "model weights" + + def test_configured_weights_override_default(self, tmp_path, openvino_config_file) -> None: + weights_path = tmp_path / "custom weights" + path = openvino_config_file({"NPU": {"WEIGHTS_PATH": str(weights_path)}}) + config = json.loads(build_npu_load_config(path, weights_path=tmp_path / "fallback")) + assert config["NPU"]["WEIGHTS_PATH"] == str(weights_path.resolve()) + + def test_relative_configured_weights_use_config_directory( + self, tmp_path, monkeypatch, openvino_config_file + ) -> None: + path = openvino_config_file({"NPU": {"WEIGHTS_PATH": "weights"}}) + monkeypatch.chdir(tmp_path.parent) + config = json.loads(build_npu_load_config(path)) + assert config["NPU"]["WEIGHTS_PATH"] == str((path.parent / "weights").resolve()) + + @pytest.mark.parametrize( + ("payload", "role", "error", "message"), + [ + ([], None, TypeError, "load config must be a JSON object"), + (None, None, TypeError, "load config must be a JSON object"), + ({"CTX": {}}, None, ValueError, "requires model_role"), + ({"CTX": {}}, "ITER", ValueError, "missing the ITER section"), + ({"CTX": []}, "CTX", TypeError, "CTX configuration must be a JSON object"), + ({"NPU": []}, None, TypeError, "NPU configuration must be a JSON object"), + ({"NPU": None}, None, TypeError, "NPU configuration must be a JSON object"), + ( + {"NPU": {"WEIGHTS_PATH": 42}}, + None, + ValueError, + "WEIGHTS_PATH must be a non-empty string", + ), + ( + {"NPU": {"WEIGHTS_PATH": " "}}, + None, + ValueError, + "WEIGHTS_PATH must be a non-empty string", + ), + ], + ) + def test_invalid_config_is_rejected(self, openvino_config_file, payload, role, error, message): + path = openvino_config_file(payload) + with pytest.raises(error, match=message): + build_npu_load_config(path, model_role=role) + + def test_unsupported_role_is_rejected(self) -> None: + with pytest.raises(ValueError, match="model_role must be 'CTX' or 'ITER'"): + build_npu_load_config(model_role="HEAD") + + def test_missing_file_is_not_silently_ignored(self, tmp_path) -> None: + with pytest.raises(FileNotFoundError): + build_npu_load_config(tmp_path / "missing.json") + + def test_invalid_json_reports_config_path(self, tmp_path) -> None: + path = tmp_path / "invalid.json" + path.write_text("{", encoding="utf-8") + with pytest.raises(ValueError, match="Invalid OpenVINO load config JSON") as exc_info: + build_npu_load_config(path) + assert str(path) in str(exc_info.value) + + def test_utf8_bom_is_accepted(self, tmp_path) -> None: + payload = {"NPU": {"NPU_TURBO": "YES"}} + path = tmp_path / "bom.json" + path.write_text(json.dumps(payload), encoding="utf-8-sig") + config = json.loads(build_npu_load_config(path)) + assert config["NPU"]["NPU_TURBO"] == payload["NPU"]["NPU_TURBO"] + + # --------------------------------------------------------------------------- # Tests: build_qwen3_transformer_only_stages # --------------------------------------------------------------------------- @@ -561,6 +708,64 @@ def test_vitisai_ep_injects_session_options(self) -> None: assert vitisai_opts["no_linear_slice"] == "1" assert itr_opts["log_id"] == "onnxruntime-genai.iterator" + @pytest.mark.parametrize("ep", ["openvino", "OpenVINOExecutionProvider", "OPENVINO"]) + def test_openvino_only_routes_transformer_stages(self, ep) -> None: + with self._patch_onnx(): + stages, _ = build_qwen3_transformer_only_stages( + "ctx.onnx", "iter.onnx", num_layers=4, ep=ep + ) + stage_map = {stage.name: stage for stage in stages} + assert stage_map["embeddings"].session_options is None + assert stage_map["lm_head"].session_options is None + for name in ("context", "iterator"): + options = stage_map[name].session_options + assert options["log_id"] == f"onnxruntime-genai.{name}" + assert options["provider_options"][0]["openvino"]["device_type"] == "NPU" + assert ( + stage_map["context"].session_options["provider_options"] + == stage_map["iterator"].session_options["provider_options"] + ) + + def test_openvino_role_config_reaches_serialized_pipeline(self, openvino_config_file) -> None: + payload = { + "CTX": {"NPU": {"NPU_TURBO": "YES"}}, + "ITER": {"NPU": {"NPU_TURBO": "NO"}}, + } + path = openvino_config_file(payload) + with self._patch_onnx(): + stages, decoder_io = build_qwen3_transformer_only_stages( + "ctx.onnx", "iter.onnx", num_layers=4, ep="openvino", openvino_config_path=path + ) + config = build_genai_config( + _mock_config(num_hidden_layers=4), + max_cache_len=256, + prefill_seq_len=64, + pipeline=stages, + decoder_io=decoder_io, + ) + pipeline = json.loads(json.dumps(config))["model"]["decoder"]["pipeline"] + stage_map = {name: stage for entry in pipeline for name, stage in entry.items()} + for name, role in (("context", "CTX"), ("iterator", "ITER")): + provider = stage_map[name]["session_options"]["provider_options"][0]["openvino"] + assert isinstance(provider["load_config"], str) + assert ( + json.loads(provider["load_config"])["NPU"]["NPU_TURBO"] + == payload[role]["NPU"]["NPU_TURBO"] + ) + assert "session_options" not in stage_map["embeddings"] + assert "session_options" not in stage_map["lm_head"] + + @pytest.mark.parametrize("ep", ["cpu", "qnn", "vitisai"]) + def test_openvino_options_are_rejected_for_other_eps(self, ep, tmp_path) -> None: + with pytest.raises(ValueError, match="OpenVINO configuration requires"): + build_qwen3_transformer_only_stages( + "ctx.onnx", + "iter.onnx", + num_layers=4, + ep=ep, + openvino_config_path=tmp_path / "unused.json", + ) + # --------------------------------------------------------------------------- # Tests: write_genai_bundle wrapper (ep routing + transformer_onnx_passes) @@ -618,3 +823,51 @@ def test_cpu_ep_forwards_no_session_options(self) -> None: kwargs = mock_write.call_args.kwargs assert kwargs["context_session_options"] is None assert kwargs["iterator_session_options"] is None + + def test_openvino_defaults_weights_to_bundle_directory(self, tmp_path) -> None: + output_dir = tmp_path / "bundle" + with self._patch_generic() as mock_write: + write_genai_bundle(output_dir, ep="openvino", **self._COMMON) + kwargs = mock_write.call_args.kwargs + for name in ("context", "iterator"): + options = kwargs[f"{name}_session_options"] + assert options["log_id"] == f"onnxruntime-genai.{name}" + provider = options["provider_options"][0]["openvino"] + assert provider["device_type"] == "NPU" + config = json.loads(provider["load_config"]) + assert config["NPU"]["WEIGHTS_PATH"] == str(output_dir.resolve()) + assert "openvino_config_path" not in kwargs + assert "openvino_weights_path" not in kwargs + + def test_openvino_forwards_custom_role_options(self, tmp_path, openvino_config_file) -> None: + payload = { + "CTX": {"NPU": {"NPU_TURBO": "YES"}}, + "ITER": {"NPU": {"NPU_TURBO": "NO"}}, + } + path = openvino_config_file(payload) + weights_dir = tmp_path / "shared weights" + with self._patch_generic() as mock_write: + write_genai_bundle( + tmp_path / "bundle", + ep="OpenVINOExecutionProvider", + openvino_config_path=path, + openvino_weights_path=weights_dir, + **self._COMMON, + ) + kwargs = mock_write.call_args.kwargs + for name, role in (("context", "CTX"), ("iterator", "ITER")): + provider = kwargs[f"{name}_session_options"]["provider_options"][0]["openvino"] + config = json.loads(provider["load_config"]) + assert config["NPU"]["NPU_TURBO"] == payload[role]["NPU"]["NPU_TURBO"] + assert config["NPU"]["WEIGHTS_PATH"] == str(weights_dir.resolve()) + + def test_openvino_config_errors_prevent_assembly(self, tmp_path, openvino_config_file) -> None: + path = openvino_config_file({"CTX": {"NPU": {}}}) + with ( + self._patch_generic() as mock_write, + pytest.raises(ValueError, match="missing the ITER section"), + ): + write_genai_bundle( + tmp_path / "bundle", ep="openvino", openvino_config_path=path, **self._COMMON + ) + mock_write.assert_not_called() diff --git a/tests/unit/models/winml/test_genai_bundle_registry.py b/tests/unit/models/winml/test_genai_bundle_registry.py index a26ba08a9..064cb0533 100644 --- a/tests/unit/models/winml/test_genai_bundle_registry.py +++ b/tests/unit/models/winml/test_genai_bundle_registry.py @@ -37,6 +37,14 @@ def test_resolve_qwen3_returns_recipe(): assert len(recipe.transformer_onnx_passes) >= 1 +def test_qwen3_openvino_target_is_npu_only(): + recipe = resolve_genai_bundle("qwen3") + assert recipe is not None + assert {target.device for target in recipe.supported_targets if target.ep == "openvino"} == { + "npu" + } + + def test_resolve_unregistered_returns_none(): assert resolve_genai_bundle("bert") is None From b2b61ab7d328769be1093b2f3404728a223cab34 Mon Sep 17 00:00:00 2001 From: Supreet Singh Palne Date: Wed, 7 Oct 2026 13:59:43 -0700 Subject: [PATCH 2/6] Fix OpenVINO precompiled GenAI output and show response text --- npu_config_npuw.json | 6 +- src/winml/modelkit/commands/_perf_genai.py | 9 +++ src/winml/modelkit/session/genai_session.py | 38 +++++++++++ tests/unit/commands/test_perf_genai.py | 21 ++++-- tests/unit/session/test_genai_session.py | 71 +++++++++++++++++++++ 5 files changed, 136 insertions(+), 9 deletions(-) diff --git a/npu_config_npuw.json b/npu_config_npuw.json index 69154cec2..ac5225ae3 100644 --- a/npu_config_npuw.json +++ b/npu_config_npuw.json @@ -2,11 +2,7 @@ "NPU": { "NPU_TURBO": "YES", "NPU_QDQ_OPTIMIZATION": "YES", - "NPU_COMPILER_TYPE": "DRIVER", - "NPU_USE_NPUW": "YES", - "NPUW_DEVICES": "NPU", - "NPUW_PARALLEL_COMPILE": "YES", - "NPUW_GQA": "YES", + "NPU_COMPILER_TYPE": "PLUGIN", "CACHE_MODE": "OPTIMIZE_SPEED" } } \ No newline at end of file diff --git a/src/winml/modelkit/commands/_perf_genai.py b/src/winml/modelkit/commands/_perf_genai.py index 570549772..5dc7d64a6 100644 --- a/src/winml/modelkit/commands/_perf_genai.py +++ b/src/winml/modelkit/commands/_perf_genai.py @@ -357,6 +357,7 @@ class _RequestSample: decode_token_durations_ms: list[float] sequence_fetch_duration_ms: float detokenization_duration_ms: float + response_text: str = "" @property def model_ttft_duration_ms(self) -> float: @@ -457,6 +458,7 @@ class GenaiBenchmarkResult: timestamp: str = field(default_factory=lambda: datetime.now(timezone.utc).isoformat()) prompt_tokens: int = 0 generated_tokens: int = 0 + response_text: str = "" context_length: int | None = None effective_ep: str | None = None effective_device: str | None = None @@ -497,6 +499,7 @@ def to_dict(self) -> dict[str, Any]: }, "requests": [sample.to_dict() for sample in self.requests], "aggregate": self._round_aggregate(), + "response_text": self.response_text, } if self.memory_profile: result["memory"] = self.memory_profile @@ -747,6 +750,7 @@ def _time_one_generation( decode_token_durations_ms=[value * 1000.0 for value in timing.decode_s], sequence_fetch_duration_ms=timing.sequence_fetch_s * 1000.0, detokenization_duration_ms=timing.detokenization_s * 1000.0, + response_text=timing.response_text, ) def _aggregate( @@ -788,6 +792,7 @@ def _aggregate( effective_device=getattr(self._session, "effective_device", None), prompt_tokens=timed[0].prompt_tokens if timed else 0, generated_tokens=timed[0].generated_tokens if timed else 0, + response_text=timed[0].response_text if timed else "", context_length=self._session.context_length if self._session else None, load=load, requests=samples, @@ -849,6 +854,10 @@ def display_genai_report(result: GenaiBenchmarkResult, console: Console) -> None f"[dim]Generated:[/dim] {result.generated_tokens} tokens " f"(max_new_tokens={cfg.max_new_tokens})" ) + if result.response_text: + console.print() + console.print("[bold]Response[/bold]") + console.print(result.response_text, markup=False) load = result.load aggregate = result.aggregate diff --git a/src/winml/modelkit/session/genai_session.py b/src/winml/modelkit/session/genai_session.py index 1cae5f5a3..a0bc8a522 100644 --- a/src/winml/modelkit/session/genai_session.py +++ b/src/winml/modelkit/session/genai_session.py @@ -90,6 +90,9 @@ {"OpenVINOExecutionProvider", "VitisAIExecutionProvider"} ) +# Their shared EPContext blobs omit weights, which load from the source graphs' external data. +_WEIGHTLESS_SHARED_CONTEXT_EPS: frozenset[str] = frozenset({"OpenVINOExecutionProvider"}) + # --------------------------------------------------------------------------- # Module-level compilation worker. @@ -1615,6 +1618,7 @@ def _process_shared_group( ): self._patch_stage_filename(modified_cfg, stage_key, ctx.name) compiled_stage_filenames.add(onnx_filename) + self._link_shared_weight_sources(group, compiled_dir, modified_cfg) return True logger.info( @@ -1626,6 +1630,7 @@ def _process_shared_group( self._write_compile_marker(ctx, ea, eo) self._patch_stage_filename(modified_cfg, stage_key, ctx.name) compiled_stage_filenames.add(onnx_filename) + self._link_shared_weight_sources(group, compiled_dir, modified_cfg) return True logger.warning( @@ -1636,6 +1641,39 @@ def _process_shared_group( self._patch_stage_filename(modified_cfg, stage_key, onnx_filename) return False + def _link_shared_weight_sources( + self, + group: list[tuple[str, str, str, dict]], + compiled_dir: Path, + modified_cfg: dict, + ) -> None: + """Expose source weights to weightless shared EPContexts and enable sharing at load.""" + ep_name = normalize_ep_name(cast("EPNameOrAlias", group[0][2])) + if ep_name not in _WEIGHTLESS_SHARED_CONTEXT_EPS: + return + from ..onnx import get_external_data_files + + for _stage_key, onnx_filename, _ea, _eo in group: + src_onnx = self._bundle_dir / onnx_filename + for location in get_external_data_files(src_onnx): + src, dst = src_onnx.parent / location, compiled_dir / location + if dst.exists(): + continue + dst.parent.mkdir(parents=True, exist_ok=True) + try: + dst.symlink_to(src.resolve()) + except (OSError, NotImplementedError): + shutil.copy2(src, dst) + + stage_keys = {stage[0] for stage in group} + pipeline = modified_cfg.get("model", {}).get("decoder", {}).get("pipeline", []) + for stage_entry in pipeline: + if not isinstance(stage_entry, dict): + continue + for stage_key, stage_cfg in stage_entry.items(): + if stage_key in stage_keys and isinstance(stage_cfg, dict): + stage_cfg.setdefault("session_options", {})["ep.share_ep_contexts"] = "1" + def _compile_stages_shared( self, srcs: list[Path], diff --git a/tests/unit/commands/test_perf_genai.py b/tests/unit/commands/test_perf_genai.py index 797cae26c..dfaf9b2aa 100644 --- a/tests/unit/commands/test_perf_genai.py +++ b/tests/unit/commands/test_perf_genai.py @@ -58,6 +58,7 @@ def _timing( generator_create_s: float = 0.0, sequence_fetch_s: float = 0.0, detokenization_s: float = 0.0, + response_text: str = "", ) -> GenerationTiming: """Build a GenerationTiming with ``1 + len(decode_s)`` generated tokens.""" return GenerationTiming( @@ -69,6 +70,7 @@ def _timing( decode_s=list(decode_s), sequence_fetch_s=sequence_fetch_s, detokenization_s=detokenization_s, + response_text=response_text, ) @@ -587,7 +589,7 @@ def _result(self) -> GenaiBenchmarkResult: compile_timeout=120, ) session = _FakeSession( - [_timing(0.4, 0.6, [0.4, 0.4, 0.4])], + [_timing(0.4, 0.6, [0.4, 0.4, 0.4], response_text="Measured answer")], prompt_ids=[1, 2, 3], effective_ep="qnn", effective_device="npu", @@ -604,6 +606,7 @@ def test_to_dict_shape(self) -> None: "load", "requests", "aggregate", + "response_text", } assert d["schema_version"] == 2 info = d["benchmark_info"] @@ -622,6 +625,7 @@ def test_to_dict_shape(self) -> None: assert info["monitor"] is False assert info["apply_template"] is True assert info["prompt"] == "Benchmark this exact prompt" + assert d["response_text"] == "Measured answer" assert set(d["load"]) == { "session_load_duration_ms", "ep_registration_duration_ms", @@ -876,7 +880,9 @@ def _result(self) -> GenaiBenchmarkResult: iterations=1, warmup=0, ) - session = _FakeSession([_timing(0.4, 0.6, [0.4, 0.4, 0.4])]) + session = _FakeSession( + [_timing(0.4, 0.6, [0.4, 0.4, 0.4], response_text="Representative [/b] answer")] + ) bench = GenaiPerfBenchmark(cfg, session=session) return bench.run() @@ -887,9 +893,16 @@ def test_write_genai_report_writes_json(self, tmp_path: Path) -> None: assert out.exists() data = json.loads(out.read_text(encoding="utf-8")) assert data["benchmark_info"]["runtime"] == "ort-genai" + assert data["response_text"] == "Representative [/b] answer" + + def test_display_genai_report_shows_response_verbatim(self) -> None: + console = Console(file=StringIO(), width=200, force_terminal=False, record=True) + + display_genai_report(self._result(), console) - def test_display_genai_report_does_not_crash(self) -> None: - display_genai_report(self._result(), Console()) + output = console.export_text() + assert "Response" in output + assert "Representative [/b] answer" in output def test_display_genai_report_ep_none_does_not_crash(self) -> None: # ep=None renders as " (config)" without error. diff --git a/tests/unit/session/test_genai_session.py b/tests/unit/session/test_genai_session.py index e34c1dacb..9449d5bc3 100644 --- a/tests/unit/session/test_genai_session.py +++ b/tests/unit/session/test_genai_session.py @@ -2785,6 +2785,77 @@ def test_does_not_overwrite_existing_files(self, bundle_dir_with_pipeline: Path) assert existing.read_bytes() == b"already here" +# --------------------------------------------------------------------------- +# Tests: shared EPContext groups whose blobs omit weights +# --------------------------------------------------------------------------- + + +class TestSharedGroupExternalWeights: + @staticmethod + def _shared_group(tmp_path: Path, ep: str) -> tuple[GenaiSession, list, dict]: + import numpy as np + import onnx + from onnx import TensorProto, helper, numpy_helper + + pipeline = [] + group = [] + for stage_key, filename in (("context", "ctx.onnx"), ("iterator", "iter.onnx")): + weight = numpy_helper.from_array(np.ones((4, 4), dtype=np.float32), "w") + graph = helper.make_graph( + [helper.make_node("MatMul", ["x", "w"], ["y"])], + stage_key, + [helper.make_tensor_value_info("x", TensorProto.FLOAT, [1, 4])], + [helper.make_tensor_value_info("y", TensorProto.FLOAT, [1, 4])], + [weight], + ) + onnx.save_model( + helper.make_model(graph), + str(tmp_path / filename), + save_as_external_data=True, + location=f"{filename}.data", + size_threshold=0, + ) + options = {"provider_options": [{ep: {}}]} + pipeline.append({stage_key: {"filename": filename, "session_options": options}}) + group.append((stage_key, filename, ep, {})) + cfg = {"model": {"type": "decoder-pipeline", "decoder": {"pipeline": pipeline}}} + (tmp_path / "genai_config.json").write_text(json.dumps(cfg), encoding="utf-8") + return GenaiSession(tmp_path, compile=True), group, cfg + + @pytest.mark.parametrize("cached", [False, True]) + @pytest.mark.parametrize(("ep", "needs_weights"), [("openvino", True), ("qnn", False)]) + def test_weightless_blobs_get_weights_and_sharing_at_load( + self, tmp_path: Path, monkeypatch: pytest.MonkeyPatch, ep, needs_weights, cached + ) -> None: + session, group, cfg = self._shared_group(tmp_path, ep) + compiled_dir = tmp_path / "_compiled" + compiled_dir.mkdir() + + def fake_compile(srcs, ctx_outs, group): + for ctx in ctx_outs: + ctx.write_bytes(b"ep_ctx") + return True + + compile_spy = MagicMock(side_effect=fake_compile) + monkeypatch.setattr(session, "_compile_stages_shared", compile_spy) + monkeypatch.setattr(session, "_epcontext_is_fresh", lambda *_args: cached) + + assert session._process_shared_group(group, compiled_dir, cfg, set()) is True + + assert compile_spy.called is not cached + stages = {k: v for entry in cfg["model"]["decoder"]["pipeline"] for k, v in entry.items()} + for stage_key, filename, _ep, _opts in group: + source_data = tmp_path / f"{filename}.data" + linked = compiled_dir / source_data.name + sharing = stages[stage_key]["session_options"].get("ep.share_ep_contexts") + if needs_weights: + assert linked.read_bytes() == source_data.read_bytes() + assert sharing == "1" + else: + assert not linked.exists() + assert sharing is None + + # --------------------------------------------------------------------------- # Tests: _patch_stage_filename # --------------------------------------------------------------------------- From 053970fcd4eb8a7b37f6facaa9abfefac2dd3737 Mon Sep 17 00:00:00 2001 From: Qiong Wu Date: Fri, 9 Oct 2026 09:32:38 +0800 Subject: [PATCH 3/6] Address OpenVINO review feedback --- npu_config_npuw.json | 8 - src/winml/modelkit/commands/_perf_genai.py | 2 + src/winml/modelkit/commands/perf.py | 27 +-- src/winml/modelkit/models/hf/qwen3/genai.py | 102 ++--------- src/winml/modelkit/session/genai_session.py | 17 +- tests/unit/commands/test_perf_genai.py | 50 +++--- tests/unit/models/qwen3/test_genai_config.py | 175 ++----------------- tests/unit/session/test_genai_session.py | 111 ++++++++---- 8 files changed, 160 insertions(+), 332 deletions(-) delete mode 100644 npu_config_npuw.json diff --git a/npu_config_npuw.json b/npu_config_npuw.json deleted file mode 100644 index ac5225ae3..000000000 --- a/npu_config_npuw.json +++ /dev/null @@ -1,8 +0,0 @@ -{ - "NPU": { - "NPU_TURBO": "YES", - "NPU_QDQ_OPTIMIZATION": "YES", - "NPU_COMPILER_TYPE": "PLUGIN", - "CACHE_MODE": "OPTIMIZE_SPEED" - } -} \ No newline at end of file diff --git a/src/winml/modelkit/commands/_perf_genai.py b/src/winml/modelkit/commands/_perf_genai.py index 5dc7d64a6..98ac1a74a 100644 --- a/src/winml/modelkit/commands/_perf_genai.py +++ b/src/winml/modelkit/commands/_perf_genai.py @@ -328,6 +328,7 @@ class GenaiPerfConfig: model_id: str | None = None ep: EPNameOrAlias | None = None device: str = "auto" + provider_options: dict[str, str] | None = None prompt: str = _DEFAULT_PROMPT apply_template: bool = True max_new_tokens: int = 128 @@ -560,6 +561,7 @@ def _build_session(self) -> GenaiSession: self._config.bundle_dir, self._config.ep, device=self._session_device(), + provider_options=self._config.provider_options, context_length=self._config.context_length, compile=self._config.compile, compile_timeout=self._config.compile_timeout, diff --git a/src/winml/modelkit/commands/perf.py b/src/winml/modelkit/commands/perf.py index 36652c89e..c4cb9b74a 100644 --- a/src/winml/modelkit/commands/perf.py +++ b/src/winml/modelkit/commands/perf.py @@ -2390,7 +2390,6 @@ def _run_simple_loop( _GENAI_IGNORED_FLAGS: dict[str, str] = { "task": "--task", "precision": "--precision", - "ep_options": "--ep-options", "shape_config_path": "--shape-config", "input_specs": "--input-specs", "export_config": "--export-config", @@ -2504,14 +2503,6 @@ def _autobuild_genai_bundle( build_ep, build_device = short_ep_name(target.ep), target.device # Do not reuse a bundle exported for a different execution provider. bundle_dir = bundle_dir.with_name(f"genai-bundle-{build_ep}-{build_device}") - openvino_config: Path | None = p.get("openvino_config") - if openvino_config is not None: - if build_ep != "openvino": - raise click.UsageError("--openvino-config requires --ep openvino.") - import hashlib - - config_digest = hashlib.sha256(openvino_config.read_bytes()).hexdigest()[:12] - bundle_dir = bundle_dir.with_name(f"{bundle_dir.name}-config-{config_digest}") build_cache_dir = cache_dir # --rebuild overwrites the cached bundle; a plain run reuses it. Checked # before any model resolution so a cache hit never touches the network. @@ -2572,9 +2563,6 @@ def _autobuild_genai_bundle( force_rebuild=force_rebuild, cache_dir=build_cache_dir, emit=lambda msg: console.print(msg, markup=False), - assemble_options=( - {"openvino_config_path": openvino_config} if openvino_config is not None else None - ), ) return bundle_dir, True @@ -2627,6 +2615,11 @@ def _run_genai_runtime( ep = cast("EPNameOrAlias", short_ep_name(target.ep)) if target is not None else None if target is not None: device = target.device + provider_options: dict[str, str] | None = p.get("ep_options") + if provider_options and ep is None: + raise click.UsageError( + "--ep-options requires --ep or a concrete --device with --runtime ort-genai." + ) # Keep any bundle-lifetime resources alive across the benchmark. with contextlib.ExitStack() as stack: @@ -2689,6 +2682,7 @@ def _run_genai_runtime( model_id=model, ep=ep, device=device, + provider_options=provider_options, prompt=prompt, apply_template=p["apply_template"], max_new_tokens=p["max_new_tokens"], @@ -2795,13 +2789,6 @@ def _validate_duration( help="[ort-genai] Max seconds to compile each EPContext stage before falling back " "to the original ONNX (requires --compile).", ) -@click.option( - "--openvino-config", - type=click.Path(exists=True, dir_okay=False, path_type=Path), - default=None, - help="[ort-genai] OpenVINO load-config JSON embedded while auto-building a model ID. " - "Requires --ep openvino.", -) @click.option( "--task", type=str, @@ -2982,7 +2969,6 @@ def perf( apply_template: bool, max_new_tokens: int, compile_timeout: int, - openvino_config: Path | None, task: str | None, submodel: str | None, iterations: int, @@ -3061,6 +3047,7 @@ def perf( raise click.UsageError("A model is required via -m/--model.") ep_provider_options = cli_utils.parse_ep_options(ep_options) + ctx.params["ep_options"] = ep_provider_options if device_luid is not None and "device_id" in (ep_provider_options or {}): raise click.UsageError( "--device-luid cannot be combined with --ep-options device_id=...; " diff --git a/src/winml/modelkit/models/hf/qwen3/genai.py b/src/winml/modelkit/models/hf/qwen3/genai.py index ceb5aab67..35a444ca9 100644 --- a/src/winml/modelkit/models/hf/qwen3/genai.py +++ b/src/winml/modelkit/models/hf/qwen3/genai.py @@ -62,7 +62,12 @@ # Qwen3-specific NPU execution-provider routing (QNN / VitisAI / OpenVINO) # --------------------------------------------------------------------------- -_OPENVINO_CONFIG_ROLE_KEYS = frozenset({"CTX", "ITER", "HEAD"}) +_OPENVINO_NPU_DEFAULTS = { + "NPU_TURBO": "YES", + "NPU_QDQ_OPTIMIZATION": "YES", + "NPU_COMPILER_TYPE": "PLUGIN", + "CACHE_MODE": "OPTIMIZE_SPEED", +} def qnn_stage_session_options(log_id: str, soc_model: str = "60") -> dict: @@ -129,81 +134,28 @@ def vitisai_stage_session_options(log_id: str) -> dict: def build_npu_load_config( - custom_config_path: str | Path | None = None, *, - model_role: str | None = None, weights_path: str | Path | None = None, ) -> str: """Build the JSON-string ``load_config`` for the OpenVINO NPU plugin. - Accepts a flat configuration (e.g. ``{"NPU": {...}}``) or a configuration - with ``CTX``/``ITER`` sections. A role-based file must contain the requested - role; missing sections are errors rather than silently using defaults. - ``HEAD`` is recognized only to detect role-based files; the LM head stays - on CPU. No legacy provider setup or tuning flags are injected. - Args: - custom_config_path: Optional UTF-8 JSON file. Without one, only the - driver compiler default and an explicitly supplied weights path - are included. Custom values take precedence over defaults. - model_role: ``"CTX"`` or ``"ITER"``; required for role-based files. - weights_path: Default external-weights directory. A ``WEIGHTS_PATH`` - in the file takes precedence; relative file values resolve against - that file's directory. Emitted paths are absolute so loading a - derived bundle does not change their meaning. + weights_path: Optional external-weights directory. Emitted paths are + absolute so loading a derived bundle does not change their meaning. Returns: Serialized OpenVINO load configuration, not a filename. """ - if model_role not in (None, "CTX", "ITER"): - raise ValueError("OpenVINO model_role must be 'CTX' or 'ITER'") - - config: dict = {} - config_path = ( - Path(custom_config_path).expanduser().resolve() if custom_config_path is not None else None - ) - if config_path is not None: - try: - config = json.loads(config_path.read_text(encoding="utf-8-sig")) - except json.JSONDecodeError as exc: - raise ValueError(f"Invalid OpenVINO load config JSON in {config_path}: {exc}") from exc - if not isinstance(config, dict): - raise TypeError("OpenVINO load config must be a JSON object") - if _OPENVINO_CONFIG_ROLE_KEYS.intersection(config): - if model_role is None: - raise ValueError("A role-based OpenVINO load config requires model_role") - if model_role not in config: - raise ValueError(f"OpenVINO load config is missing the {model_role} section") - config = config[model_role] - if not isinstance(config, dict): - raise TypeError(f"OpenVINO {model_role} configuration must be a JSON object") - - npu_config = config.setdefault("NPU", {}) - if not isinstance(npu_config, dict): - raise TypeError("OpenVINO NPU configuration must be a JSON object") - npu_config.setdefault("NPU_COMPILER_TYPE", "DRIVER") - - if "WEIGHTS_PATH" in npu_config: - configured_weights = npu_config["WEIGHTS_PATH"] - if not isinstance(configured_weights, str) or not configured_weights.strip(): - raise ValueError("OpenVINO WEIGHTS_PATH must be a non-empty string") - resolved_weights = Path(configured_weights).expanduser() - if not resolved_weights.is_absolute() and config_path is not None: - resolved_weights = config_path.parent / resolved_weights - npu_config["WEIGHTS_PATH"] = str(resolved_weights.resolve()) - elif weights_path is not None: + npu_config = dict(_OPENVINO_NPU_DEFAULTS) + if weights_path is not None: npu_config["WEIGHTS_PATH"] = str(Path(weights_path).expanduser().resolve()) - # Stable serialization keeps equivalent CTX/ITER options shareable even - # when their keys occur in a different order in the user's JSON file. - return json.dumps(config, sort_keys=True) + return json.dumps({"NPU": npu_config}, sort_keys=True) def openvino_stage_session_options( log_id: str, *, - custom_config_path: str | Path | None = None, - model_role: str | None = None, weights_path: str | Path | None = None, ) -> dict: """Return session options routing a transformer stage to the Intel NPU. @@ -219,11 +171,7 @@ def openvino_stage_session_options( { "openvino": { "device_type": "NPU", - "load_config": build_npu_load_config( - custom_config_path, - model_role=model_role, - weights_path=weights_path, - ), + "load_config": build_npu_load_config(weights_path=weights_path), } } ], @@ -236,7 +184,6 @@ def _stage_session_options( ep: str, soc_model: str, *, - openvino_config_path: str | Path | None = None, openvino_weights_path: str | Path | None = None, ) -> tuple[dict | None, dict | None]: """Return ``(context, iterator)`` session_options for the given EP. @@ -245,16 +192,14 @@ def _stage_session_options( * ``ep="qnn"`` -> Qualcomm QNN HTP (``soc_model`` selects the Snapdragon SoC). * ``ep="vitisai"`` -> AMD Ryzen AI NPU. - * ``ep="openvino"`` -> Intel NPU, with optional per-role load configuration. + * ``ep="openvino"`` -> Intel NPU. Any other value (e.g. ``"cpu"``) leaves the stages on the default CPU provider. Short aliases and full ``*ExecutionProvider`` names are both accepted (normalized via :func:`normalize_ep_name`). """ canonical = normalize_ep_name(ep) - if canonical != "OpenVINOExecutionProvider" and ( - openvino_config_path is not None or openvino_weights_path is not None - ): + if canonical != "OpenVINOExecutionProvider" and openvino_weights_path is not None: raise ValueError("OpenVINO configuration requires ep='openvino'") if canonical == "QNNExecutionProvider": return ( @@ -270,14 +215,10 @@ def _stage_session_options( return ( openvino_stage_session_options( "onnxruntime-genai.context", - custom_config_path=openvino_config_path, - model_role="CTX", weights_path=openvino_weights_path, ), openvino_stage_session_options( "onnxruntime-genai.iterator", - custom_config_path=openvino_config_path, - model_role="ITER", weights_path=openvino_weights_path, ), ) @@ -324,7 +265,6 @@ def build_qwen3_transformer_only_stages( lm_head_filename: str = DEFAULT_LM_HEAD_FILENAME, ep: str = "cpu", soc_model: str = "60", - openvino_config_path: str | Path | None = None, openvino_weights_path: str | Path | None = None, ) -> tuple[list[PipelineStage], DecoderIOMapping]: """Build the Qwen3 4-stage pipeline, routing ctx/iter to the NPU per ``ep``. @@ -350,10 +290,8 @@ def build_qwen3_transformer_only_stages( soc_model: Snapdragon SoC model number forwarded to the QNN backend when ``ep="qnn"``. Default ``"60"`` targets Snapdragon 8 Gen 3. Ignored for non-QNN EPs. - openvino_config_path: Optional OpenVINO JSON file shared by both stages - or containing separate ``CTX``/``ITER`` sections. Only for OpenVINO. openvino_weights_path: Optional default external-weights directory for - OpenVINO. Explicit ``WEIGHTS_PATH`` values in the JSON take precedence. + OpenVINO. Returns: ``(stages, decoder_io)`` — see @@ -362,7 +300,6 @@ def build_qwen3_transformer_only_stages( ctx_opts, iter_opts = _stage_session_options( ep, soc_model, - openvino_config_path=openvino_config_path, openvino_weights_path=openvino_weights_path, ) return build_decoder_pipeline_stages( @@ -395,7 +332,6 @@ def write_genai_bundle( ep: str = "cpu", soc_model: str = "60", transformer_onnx_passes: Sequence[Callable[[onnx.ModelProto], onnx.ModelProto]] | None = None, - openvino_config_path: str | Path | None = None, openvino_weights_path: str | Path | None = None, ) -> Path: """Assemble a Qwen3 genai bundle, routing ctx/iter to the NPU per ``ep``. @@ -416,13 +352,8 @@ def write_genai_bundle( transformer_onnx_passes: Optional ONNX graph transforms applied to the copied context/iterator models before ``genai_config.json`` is written. Forwarded verbatim to the generic assembler. - openvino_config_path: Optional flat or per-role OpenVINO JSON file. - Its contents are embedded in stage provider options; the original - configuration file is not required at runtime. Different role - options prevent grouped compilation in the current runtime. openvino_weights_path: Default external-weights directory for OpenVINO; - defaults to the output bundle directory. Explicit ``WEIGHTS_PATH`` - values in the JSON take precedence. Other EPs reject these options. + defaults to the output bundle directory. Other EPs reject this option. Returns: Path to the written ``genai_config.json``. @@ -432,7 +363,6 @@ def write_genai_bundle( ctx_opts, iter_opts = _stage_session_options( ep, soc_model, - openvino_config_path=openvino_config_path, openvino_weights_path=openvino_weights_path, ) return _write_genai_bundle( diff --git a/src/winml/modelkit/session/genai_session.py b/src/winml/modelkit/session/genai_session.py index a0bc8a522..d972a2cef 100644 --- a/src/winml/modelkit/session/genai_session.py +++ b/src/winml/modelkit/session/genai_session.py @@ -410,6 +410,9 @@ class GenaiSession: *ep* should run on. Used only to synthesize ``device_type`` for device-parameterized EPs (OpenVINO/VitisAI) when a re-routed stage has no reusable options; ignored when respecting the bundle config. + provider_options: Explicit options merged onto each hardware stage + selected by *ep*. Values override options already stored in the + bundle, matching the ``--ep-options`` CLI precedence. context_length: Override for the static KV cache length. When ``None`` (default), read from ``genai_config.json``. Must match the ``--max-cache-len`` used during the winml-cli build. @@ -454,6 +457,7 @@ def __init__( ep: EPNameOrAlias | None = None, *, device: str | None = None, + provider_options: dict[str, str] | None = None, context_length: int | None = None, verbose: bool = False, compile: bool = False, @@ -469,6 +473,7 @@ def __init__( # VitisAI ``device_type``) when re-routing a stage that has no reusable # options for the target EP; QNN reuses the bundle's own ``backend_path``. self._device: str | None = device.lower() if device else None + self._provider_options = dict(provider_options or {}) # Set at load(): did the override actually take effect (rewrite/strip at # least one stage)? Drives :attr:`effective_ep` so the report never # claims an EP that never applied (flat/empty pipeline, all-CPU bundle). @@ -1110,6 +1115,7 @@ def _override_stage_provider(self, stage_cfg: dict[str, Any], borrow_opts: dict) opts = dict(borrow_opts) else: opts = self._default_opts_for_device(ep) + opts = {**opts, **self._provider_options} if not isinstance(so, dict): so = {} stage_cfg["session_options"] = so @@ -1657,13 +1663,16 @@ def _link_shared_weight_sources( src_onnx = self._bundle_dir / onnx_filename for location in get_external_data_files(src_onnx): src, dst = src_onnx.parent / location, compiled_dir / location - if dst.exists(): + if dst.is_symlink(): continue dst.parent.mkdir(parents=True, exist_ok=True) - try: - dst.symlink_to(src.resolve()) - except (OSError, NotImplementedError): + if dst.exists(): shutil.copy2(src, dst) + else: + try: + dst.symlink_to(src.resolve()) + except (OSError, NotImplementedError): + shutil.copy2(src, dst) stage_keys = {stage[0] for stage in group} pipeline = modified_cfg.get("model", {}).get("decoder", {}).get("pipeline", []) diff --git a/tests/unit/commands/test_perf_genai.py b/tests/unit/commands/test_perf_genai.py index dfaf9b2aa..f78a58221 100644 --- a/tests/unit/commands/test_perf_genai.py +++ b/tests/unit/commands/test_perf_genai.py @@ -800,15 +800,26 @@ def test_auto_falls_back_to_ep_primary_device(self) -> None: def test_build_session_forwards_device(self, monkeypatch) -> None: captured: dict = {} - def fake_ctor(bundle_dir, ep, *, device=None, **_kwargs): + def fake_ctor(bundle_dir, ep, *, device=None, provider_options=None, **_kwargs): captured["ep"] = ep captured["device"] = device + captured["provider_options"] = provider_options return _FakeSession([]) monkeypatch.setattr(perf_genai, "GenaiSession", fake_ctor) - cfg = GenaiPerfConfig(bundle_dir=Path("bundle"), ep="openvino", device="npu") + provider_options = {"load_config": '{"NPU":{"NPU_TURBO":"NO"}}'} + cfg = GenaiPerfConfig( + bundle_dir=Path("bundle"), + ep="openvino", + device="npu", + provider_options=provider_options, + ) GenaiPerfBenchmark(cfg)._build_session() - assert captured == {"ep": "openvino", "device": "npu"} + assert captured == { + "ep": "openvino", + "device": "npu", + "provider_options": provider_options, + } def test_build_hw_monitor_uses_cpu_when_effective_adapter_is_unproven( self, monkeypatch @@ -1471,14 +1482,12 @@ def test_hf_model_id_autobuilds_and_dispatches( assert cfg.device == "config" assert cfg.ep is None - def test_openvino_config_is_forwarded_to_autobuild( + def test_ep_options_are_forwarded_to_genai_session( self, runner: CliRunner, tmp_path: Path, capture_run: dict, monkeypatch ) -> None: import winml.modelkit.loader as loader_mod import winml.modelkit.models.winml as winml_models - config_path = tmp_path / "npu.json" - config_path.write_text('{"NPU": {"NPU_USE_NPUW": "YES"}}', encoding="utf-8") monkeypatch.setenv("WINML_CACHE_DIR", str(tmp_path / "cache")) monkeypatch.setattr( loader_mod, "resolve_loader_config", _fake_resolve_loader_config("qwen3") @@ -1499,40 +1508,37 @@ def test_openvino_config_is_forwarded_to_autobuild( "openvino", "--device", "npu", - "--openvino-config", - str(config_path), + "--ep-options", + 'load_config={"NPU":{"NPU_TURBO":"NO"}}', ], ) assert result.exit_code == 0, result.output - assert build_calls["build"]["assemble_options"] == { - "openvino_config_path": config_path + assert capture_run["config"].provider_options == { + "load_config": '{"NPU":{"NPU_TURBO":"NO"}}' } - assert "-config-" in build_calls["build"]["output_dir"].name assert capture_run["config"].bundle_dir == build_calls["build"]["output_dir"] - def test_openvino_config_rejects_other_ep( - self, runner: CliRunner, tmp_path: Path, capture_run: dict + def test_ep_options_require_genai_target( + self, runner: CliRunner, capture_run: dict, tmp_path: Path ) -> None: - config_path = tmp_path / "npu.json" - config_path.write_text("{}", encoding="utf-8") - + bundle_dir = tmp_path / "bundle" + bundle_dir.mkdir() + (bundle_dir / "genai_config.json").write_text("{}", encoding="utf-8") result = runner.invoke( perf, [ "-m", - "Qwen/Qwen3-0.6B", + str(bundle_dir), "--runtime", "ort-genai", - "--ep", - "cpu", - "--openvino-config", - str(config_path), + "--ep-options", + "load_config={}", ], ) assert result.exit_code == 2, result.output - assert "--openvino-config requires --ep openvino" in result.output + assert "--ep-options requires --ep or a concrete --device" in result.output assert "config" not in capture_run @pytest.mark.parametrize( diff --git a/tests/unit/models/qwen3/test_genai_config.py b/tests/unit/models/qwen3/test_genai_config.py index 7683f3f8b..775e1ebf9 100644 --- a/tests/unit/models/qwen3/test_genai_config.py +++ b/tests/unit/models/qwen3/test_genai_config.py @@ -37,16 +37,6 @@ # --------------------------------------------------------------------------- -@pytest.fixture -def openvino_config_file(tmp_path): - def write_config(payload): - path = tmp_path / "openvino.json" - path.write_text(json.dumps(payload), encoding="utf-8") - return path - - return write_config - - def _mock_config( *, num_hidden_layers: int = 28, @@ -409,57 +399,22 @@ def test_single_layer_model(self) -> None: class TestOpenVINOSessionOptions: - def test_minimal_npu_options_without_custom_config(self) -> None: + def test_default_npu_options(self) -> None: options = openvino_stage_session_options("test.context") assert options["log_id"] == "test.context" assert options["intra_op_num_threads"] == 2 assert options["inter_op_num_threads"] == 1 provider = options["provider_options"][0]["openvino"] assert provider["device_type"] == "NPU" - assert json.loads(provider["load_config"]) == {"NPU": {"NPU_COMPILER_TYPE": "DRIVER"}} - assert set(provider) == {"device_type", "load_config"} - - def test_flat_config_preserves_custom_settings(self, openvino_config_file) -> None: - payload = { + assert json.loads(provider["load_config"]) == { "NPU": { - "NPU_COMPILER_TYPE": "custom-compiler", - "NPU_TURBO": "NO", + "CACHE_MODE": "OPTIMIZE_SPEED", + "NPU_COMPILER_TYPE": "PLUGIN", "NPU_QDQ_OPTIMIZATION": "YES", + "NPU_TURBO": "YES", } } - path = openvino_config_file(payload) - for role in ("CTX", "ITER"): - assert json.loads(build_npu_load_config(path, model_role=role)) == payload - assert json.loads(path.read_text(encoding="utf-8")) == payload - - @pytest.mark.parametrize("role", ["CTX", "ITER"]) - def test_selects_role_without_forwarding_other_sections( - self, openvino_config_file, role - ) -> None: - payload = { - "CTX": {"NPU": {"NPU_TURBO": "YES"}}, - "ITER": {"NPU": {"NPU_TURBO": "NO"}}, - "HEAD": {"NPU": {"NPU_TURBO": "YES"}}, - } - path = openvino_config_file(payload) - config = json.loads(build_npu_load_config(path, model_role=role)) - assert set(config) == {"NPU"} - assert config["NPU"] == { - **payload[role]["NPU"], - "NPU_COMPILER_TYPE": "DRIVER", - } - - def test_equivalent_role_configs_serialize_identically(self, openvino_config_file) -> None: - properties = {"NPU_TURBO": "YES", "NPU_QDQ_OPTIMIZATION": "YES"} - path = openvino_config_file( - { - "CTX": {"NPU": properties}, - "ITER": {"NPU": dict(reversed(list(properties.items())))}, - } - ) - assert build_npu_load_config(path, model_role="CTX") == build_npu_load_config( - path, model_role="ITER" - ) + assert set(provider) == {"device_type", "load_config"} def test_weights_default_is_absolute(self, tmp_path, monkeypatch) -> None: monkeypatch.chdir(tmp_path) @@ -468,72 +423,6 @@ def test_weights_default_is_absolute(self, tmp_path, monkeypatch) -> None: assert path.is_absolute() assert path == tmp_path / "model weights" - def test_configured_weights_override_default(self, tmp_path, openvino_config_file) -> None: - weights_path = tmp_path / "custom weights" - path = openvino_config_file({"NPU": {"WEIGHTS_PATH": str(weights_path)}}) - config = json.loads(build_npu_load_config(path, weights_path=tmp_path / "fallback")) - assert config["NPU"]["WEIGHTS_PATH"] == str(weights_path.resolve()) - - def test_relative_configured_weights_use_config_directory( - self, tmp_path, monkeypatch, openvino_config_file - ) -> None: - path = openvino_config_file({"NPU": {"WEIGHTS_PATH": "weights"}}) - monkeypatch.chdir(tmp_path.parent) - config = json.loads(build_npu_load_config(path)) - assert config["NPU"]["WEIGHTS_PATH"] == str((path.parent / "weights").resolve()) - - @pytest.mark.parametrize( - ("payload", "role", "error", "message"), - [ - ([], None, TypeError, "load config must be a JSON object"), - (None, None, TypeError, "load config must be a JSON object"), - ({"CTX": {}}, None, ValueError, "requires model_role"), - ({"CTX": {}}, "ITER", ValueError, "missing the ITER section"), - ({"CTX": []}, "CTX", TypeError, "CTX configuration must be a JSON object"), - ({"NPU": []}, None, TypeError, "NPU configuration must be a JSON object"), - ({"NPU": None}, None, TypeError, "NPU configuration must be a JSON object"), - ( - {"NPU": {"WEIGHTS_PATH": 42}}, - None, - ValueError, - "WEIGHTS_PATH must be a non-empty string", - ), - ( - {"NPU": {"WEIGHTS_PATH": " "}}, - None, - ValueError, - "WEIGHTS_PATH must be a non-empty string", - ), - ], - ) - def test_invalid_config_is_rejected(self, openvino_config_file, payload, role, error, message): - path = openvino_config_file(payload) - with pytest.raises(error, match=message): - build_npu_load_config(path, model_role=role) - - def test_unsupported_role_is_rejected(self) -> None: - with pytest.raises(ValueError, match="model_role must be 'CTX' or 'ITER'"): - build_npu_load_config(model_role="HEAD") - - def test_missing_file_is_not_silently_ignored(self, tmp_path) -> None: - with pytest.raises(FileNotFoundError): - build_npu_load_config(tmp_path / "missing.json") - - def test_invalid_json_reports_config_path(self, tmp_path) -> None: - path = tmp_path / "invalid.json" - path.write_text("{", encoding="utf-8") - with pytest.raises(ValueError, match="Invalid OpenVINO load config JSON") as exc_info: - build_npu_load_config(path) - assert str(path) in str(exc_info.value) - - def test_utf8_bom_is_accepted(self, tmp_path) -> None: - payload = {"NPU": {"NPU_TURBO": "YES"}} - path = tmp_path / "bom.json" - path.write_text(json.dumps(payload), encoding="utf-8-sig") - config = json.loads(build_npu_load_config(path)) - assert config["NPU"]["NPU_TURBO"] == payload["NPU"]["NPU_TURBO"] - - # --------------------------------------------------------------------------- # Tests: build_qwen3_transformer_only_stages # --------------------------------------------------------------------------- @@ -726,15 +615,10 @@ def test_openvino_only_routes_transformer_stages(self, ep) -> None: == stage_map["iterator"].session_options["provider_options"] ) - def test_openvino_role_config_reaches_serialized_pipeline(self, openvino_config_file) -> None: - payload = { - "CTX": {"NPU": {"NPU_TURBO": "YES"}}, - "ITER": {"NPU": {"NPU_TURBO": "NO"}}, - } - path = openvino_config_file(payload) + def test_openvino_defaults_reach_serialized_pipeline(self) -> None: with self._patch_onnx(): stages, decoder_io = build_qwen3_transformer_only_stages( - "ctx.onnx", "iter.onnx", num_layers=4, ep="openvino", openvino_config_path=path + "ctx.onnx", "iter.onnx", num_layers=4, ep="openvino" ) config = build_genai_config( _mock_config(num_hidden_layers=4), @@ -745,28 +629,13 @@ def test_openvino_role_config_reaches_serialized_pipeline(self, openvino_config_ ) pipeline = json.loads(json.dumps(config))["model"]["decoder"]["pipeline"] stage_map = {name: stage for entry in pipeline for name, stage in entry.items()} - for name, role in (("context", "CTX"), ("iterator", "ITER")): + for name in ("context", "iterator"): provider = stage_map[name]["session_options"]["provider_options"][0]["openvino"] assert isinstance(provider["load_config"], str) - assert ( - json.loads(provider["load_config"])["NPU"]["NPU_TURBO"] - == payload[role]["NPU"]["NPU_TURBO"] - ) + assert json.loads(provider["load_config"])["NPU"]["NPU_TURBO"] == "YES" assert "session_options" not in stage_map["embeddings"] assert "session_options" not in stage_map["lm_head"] - @pytest.mark.parametrize("ep", ["cpu", "qnn", "vitisai"]) - def test_openvino_options_are_rejected_for_other_eps(self, ep, tmp_path) -> None: - with pytest.raises(ValueError, match="OpenVINO configuration requires"): - build_qwen3_transformer_only_stages( - "ctx.onnx", - "iter.onnx", - num_layers=4, - ep=ep, - openvino_config_path=tmp_path / "unused.json", - ) - - # --------------------------------------------------------------------------- # Tests: write_genai_bundle wrapper (ep routing + transformer_onnx_passes) # --------------------------------------------------------------------------- @@ -836,38 +705,20 @@ def test_openvino_defaults_weights_to_bundle_directory(self, tmp_path) -> None: assert provider["device_type"] == "NPU" config = json.loads(provider["load_config"]) assert config["NPU"]["WEIGHTS_PATH"] == str(output_dir.resolve()) - assert "openvino_config_path" not in kwargs assert "openvino_weights_path" not in kwargs - def test_openvino_forwards_custom_role_options(self, tmp_path, openvino_config_file) -> None: - payload = { - "CTX": {"NPU": {"NPU_TURBO": "YES"}}, - "ITER": {"NPU": {"NPU_TURBO": "NO"}}, - } - path = openvino_config_file(payload) + def test_openvino_forwards_custom_weights_path(self, tmp_path) -> None: weights_dir = tmp_path / "shared weights" with self._patch_generic() as mock_write: write_genai_bundle( tmp_path / "bundle", ep="OpenVINOExecutionProvider", - openvino_config_path=path, openvino_weights_path=weights_dir, **self._COMMON, ) kwargs = mock_write.call_args.kwargs - for name, role in (("context", "CTX"), ("iterator", "ITER")): + for name in ("context", "iterator"): provider = kwargs[f"{name}_session_options"]["provider_options"][0]["openvino"] config = json.loads(provider["load_config"]) - assert config["NPU"]["NPU_TURBO"] == payload[role]["NPU"]["NPU_TURBO"] + assert config["NPU"]["NPU_TURBO"] == "YES" assert config["NPU"]["WEIGHTS_PATH"] == str(weights_dir.resolve()) - - def test_openvino_config_errors_prevent_assembly(self, tmp_path, openvino_config_file) -> None: - path = openvino_config_file({"CTX": {"NPU": {}}}) - with ( - self._patch_generic() as mock_write, - pytest.raises(ValueError, match="missing the ITER section"), - ): - write_genai_bundle( - tmp_path / "bundle", ep="openvino", openvino_config_path=path, **self._COMMON - ) - mock_write.assert_not_called() diff --git a/tests/unit/session/test_genai_session.py b/tests/unit/session/test_genai_session.py index 9449d5bc3..a25d61d2c 100644 --- a/tests/unit/session/test_genai_session.py +++ b/tests/unit/session/test_genai_session.py @@ -902,6 +902,38 @@ def test_force_same_ep_preserves_existing_options(self, bundle_dir: Path) -> Non stage = effective["model"]["decoder"]["pipeline"][0]["context"] assert stage["session_options"]["provider_options"] == [{"qnn": opts}] + def test_explicit_provider_options_override_bundle_options(self, bundle_dir: Path) -> None: + session = GenaiSession( + bundle_dir, + ep="openvino", + provider_options={"load_config": '{"NPU":{"NPU_TURBO":"NO"}}'}, + ) + cfg = self._pipeline_cfg( + { + "context": { + "session_options": { + "provider_options": [ + { + "openvino": { + "device_type": "NPU", + "load_config": '{"NPU":{"NPU_TURBO":"YES"}}', + } + } + ] + } + } + } + ) + effective, changed = session._apply_ep_override(cfg) + assert changed is True + provider = effective["model"]["decoder"]["pipeline"][0]["context"]["session_options"][ + "provider_options" + ][0]["openvino"] + assert provider == { + "device_type": "NPU", + "load_config": '{"NPU":{"NPU_TURBO":"NO"}}', + } + def test_force_different_hardware_ep_drops_foreign_options(self, bundle_dir: Path) -> None: # Switching QNN -> DML must not carry QNN's backend_path across. session = GenaiSession(bundle_dir, ep="dml") @@ -1592,9 +1624,8 @@ class TestCompileStageSalvage: def _make_epcontext(path: Path, bin_name: str) -> None: """Code-generate a minimal, structurally valid EPContext ONNX model.""" import onnx - from onnx import TensorProto, helper - node = helper.make_node( + node = onnx.helper.make_node( "EPContext", inputs=[], outputs=["output"], @@ -1604,13 +1635,15 @@ def _make_epcontext(path: Path, bin_name: str) -> None: ep_cache_context=bin_name, main_context=1, ) - output_info = helper.make_tensor_value_info("output", TensorProto.FLOAT, [1, 4]) - graph = helper.make_graph([node], "epcontext_model", [], [output_info]) - model = helper.make_model( + output_info = onnx.helper.make_tensor_value_info( + "output", onnx.TensorProto.FLOAT, [1, 4] + ) + graph = onnx.helper.make_graph([node], "epcontext_model", [], [output_info]) + model = onnx.helper.make_model( graph, opset_imports=[ - helper.make_opsetid("", 17), - helper.make_opsetid("com.microsoft", 1), + onnx.helper.make_opsetid("", 17), + onnx.helper.make_opsetid("com.microsoft", 1), ], ) model.ir_version = 9 @@ -1629,9 +1662,8 @@ def _make_multipartition_epcontext( accept this artifact, not reject it at the secondary node. """ import onnx - from onnx import TensorProto, helper - main = helper.make_node( + main = onnx.helper.make_node( "EPContext", inputs=[], outputs=["main_out"], @@ -1644,7 +1676,7 @@ def _make_multipartition_epcontext( secondary_attrs: dict[str, int | str] = {"embed_mode": 0, "main_context": 0} if secondary_bin_name is not None: secondary_attrs["ep_cache_context"] = secondary_bin_name - secondary = helper.make_node( + secondary = onnx.helper.make_node( "EPContext", inputs=[], outputs=["secondary_out"], @@ -1652,16 +1684,20 @@ def _make_multipartition_epcontext( domain="com.microsoft", **secondary_attrs, ) - main_info = helper.make_tensor_value_info("main_out", TensorProto.FLOAT, [1, 4]) - secondary_info = helper.make_tensor_value_info("secondary_out", TensorProto.FLOAT, [1, 4]) - graph = helper.make_graph( + main_info = onnx.helper.make_tensor_value_info( + "main_out", onnx.TensorProto.FLOAT, [1, 4] + ) + secondary_info = onnx.helper.make_tensor_value_info( + "secondary_out", onnx.TensorProto.FLOAT, [1, 4] + ) + graph = onnx.helper.make_graph( [main, secondary], "epcontext_multipartition", [], [main_info, secondary_info] ) - model = helper.make_model( + model = onnx.helper.make_model( graph, opset_imports=[ - helper.make_opsetid("", 17), - helper.make_opsetid("com.microsoft", 1), + onnx.helper.make_opsetid("", 17), + onnx.helper.make_opsetid("com.microsoft", 1), ], ) model.ir_version = 9 @@ -1814,7 +1850,6 @@ def _write() -> None: def test_non_epcontext_leftover_is_not_salvaged(self, bundle_dir_with_pipeline: Path) -> None: """A structurally valid ONNX that is *not* an EPContext model is rejected.""" import onnx - from onnx import TensorProto, helper session = GenaiSession(bundle_dir_with_pipeline, ep="qnn", compile=True) compiled_dir = bundle_dir_with_pipeline / "_compiled" @@ -1824,12 +1859,12 @@ def test_non_epcontext_leftover_is_not_salvaged(self, bundle_dir_with_pipeline: def _write() -> None: # A plain Identity graph (valid ONNX, no EPContext node) next to source. - node = helper.make_node("Identity", inputs=["x"], outputs=["y"], name="id0") - x = helper.make_tensor_value_info("x", TensorProto.FLOAT, [1, 4]) - y = helper.make_tensor_value_info("y", TensorProto.FLOAT, [1, 4]) - model = helper.make_model( - helper.make_graph([node], "plain", [x], [y]), - opset_imports=[helper.make_opsetid("", 17)], + node = onnx.helper.make_node("Identity", inputs=["x"], outputs=["y"], name="id0") + x = onnx.helper.make_tensor_value_info("x", onnx.TensorProto.FLOAT, [1, 4]) + y = onnx.helper.make_tensor_value_info("y", onnx.TensorProto.FLOAT, [1, 4]) + model = onnx.helper.make_model( + onnx.helper.make_graph([node], "plain", [x], [y]), + opset_imports=[onnx.helper.make_opsetid("", 17)], ) model.ir_version = 9 onnx.save(model, str(auto_onnx)) @@ -2795,21 +2830,20 @@ class TestSharedGroupExternalWeights: def _shared_group(tmp_path: Path, ep: str) -> tuple[GenaiSession, list, dict]: import numpy as np import onnx - from onnx import TensorProto, helper, numpy_helper pipeline = [] group = [] for stage_key, filename in (("context", "ctx.onnx"), ("iterator", "iter.onnx")): - weight = numpy_helper.from_array(np.ones((4, 4), dtype=np.float32), "w") - graph = helper.make_graph( - [helper.make_node("MatMul", ["x", "w"], ["y"])], + weight = onnx.numpy_helper.from_array(np.ones((4, 4), dtype=np.float32), "w") + graph = onnx.helper.make_graph( + [onnx.helper.make_node("MatMul", ["x", "w"], ["y"])], stage_key, - [helper.make_tensor_value_info("x", TensorProto.FLOAT, [1, 4])], - [helper.make_tensor_value_info("y", TensorProto.FLOAT, [1, 4])], + [onnx.helper.make_tensor_value_info("x", onnx.TensorProto.FLOAT, [1, 4])], + [onnx.helper.make_tensor_value_info("y", onnx.TensorProto.FLOAT, [1, 4])], [weight], ) onnx.save_model( - helper.make_model(graph), + onnx.helper.make_model(graph), str(tmp_path / filename), save_as_external_data=True, location=f"{filename}.data", @@ -2822,6 +2856,23 @@ def _shared_group(tmp_path: Path, ep: str) -> tuple[GenaiSession, list, dict]: (tmp_path / "genai_config.json").write_text(json.dumps(cfg), encoding="utf-8") return GenaiSession(tmp_path, compile=True), group, cfg + def test_recompile_refreshes_copied_weight_sidecars( + self, tmp_path: Path, monkeypatch: pytest.MonkeyPatch + ) -> None: + session, group, cfg = self._shared_group(tmp_path, "openvino") + compiled_dir = tmp_path / "_compiled" + compiled_dir.mkdir() + monkeypatch.setattr(Path, "symlink_to", MagicMock(side_effect=OSError("denied"))) + + session._link_shared_weight_sources(group, compiled_dir, cfg) + source = tmp_path / "ctx.onnx.data" + copied = compiled_dir / source.name + assert copied.read_bytes() == source.read_bytes() + + source.write_bytes(b"recompiled weights") + session._link_shared_weight_sources(group, compiled_dir, cfg) + assert copied.read_bytes() == b"recompiled weights" + @pytest.mark.parametrize("cached", [False, True]) @pytest.mark.parametrize(("ep", "needs_weights"), [("openvino", True), ("qnn", False)]) def test_weightless_blobs_get_weights_and_sharing_at_load( From 1eb10647abaf6761a975f5f3a02680185d6e3460 Mon Sep 17 00:00:00 2001 From: Qiong Wu Date: Fri, 9 Oct 2026 12:11:42 +0800 Subject: [PATCH 4/6] Generalize OpenVINO NPU defaults Co-authored-by: Copilot App <223556219+Copilot@users.noreply.github.com> --- .../modelkit/models/hf/qwen3/__init__.py | 4 - src/winml/modelkit/models/hf/qwen3/genai.py | 191 +++++++----------- src/winml/modelkit/session/ep_device.py | 14 +- src/winml/modelkit/session/genai_session.py | 55 ++++- tests/unit/models/qwen3/test_genai_config.py | 33 --- tests/unit/session/test_ep_device.py | 15 ++ tests/unit/session/test_genai_session.py | 34 +++- 7 files changed, 173 insertions(+), 173 deletions(-) diff --git a/src/winml/modelkit/models/hf/qwen3/__init__.py b/src/winml/modelkit/models/hf/qwen3/__init__.py index 0e66b9700..6e60e07ee 100644 --- a/src/winml/modelkit/models/hf/qwen3/__init__.py +++ b/src/winml/modelkit/models/hf/qwen3/__init__.py @@ -17,9 +17,7 @@ PipelineStage, build_decoder_pipeline_stages, build_genai_config, - build_npu_load_config, build_qwen3_transformer_only_stages, - openvino_stage_session_options, strip_gqa_default_attrs, write_genai_bundle, ) @@ -30,9 +28,7 @@ "PipelineStage", "build_decoder_pipeline_stages", "build_genai_config", - "build_npu_load_config", "build_qwen3_transformer_only_stages", - "openvino_stage_session_options", "strip_gqa_default_attrs", "write_genai_bundle", ] diff --git a/src/winml/modelkit/models/hf/qwen3/genai.py b/src/winml/modelkit/models/hf/qwen3/genai.py index 35a444ca9..0920b1ed2 100644 --- a/src/winml/modelkit/models/hf/qwen3/genai.py +++ b/src/winml/modelkit/models/hf/qwen3/genai.py @@ -29,6 +29,7 @@ from typing import TYPE_CHECKING from ....onnx import strip_node_attrs +from ....session import lookup_device_spec, short_ep_name from ....utils.constants import normalize_ep_name from ....utils.genai import ( DEFAULT_CONTEXT_FILENAME, @@ -62,125 +63,41 @@ # Qwen3-specific NPU execution-provider routing (QNN / VitisAI / OpenVINO) # --------------------------------------------------------------------------- -_OPENVINO_NPU_DEFAULTS = { - "NPU_TURBO": "YES", - "NPU_QDQ_OPTIMIZATION": "YES", - "NPU_COMPILER_TYPE": "PLUGIN", - "CACHE_MODE": "OPTIMIZE_SPEED", -} - - -def qnn_stage_session_options(log_id: str, soc_model: str = "60") -> dict: - """Return the ``session_options`` block that routes a stage to QNN HTP. - - Args: - log_id: ORT log identifier (shown in ORT logs), e.g. - ``"onnxruntime-genai.context"``. - soc_model: Snapdragon SoC model number passed to the QNN HTP backend. - ``"60"`` targets Snapdragon 8 Gen 3 (X Elite). Change for other - SoCs (e.g. ``"55"`` for 8 Gen 2, ``"73"`` for 8 Elite). - - Returns: - Dict suitable for the ``session_options`` key of a pipeline stage in - ``genai_config.json``. - """ - return { - "log_id": log_id, - "provider_options": [ - { - "qnn": { - "backend_path": "QnnHtp.dll", - "htp_performance_mode": "burst", - "htp_graph_finalization_optimization_mode": "3", - "soc_model": soc_model, - } - } - ], - "intra_op_num_threads": 2, - "inter_op_num_threads": 1, - } - - -def vitisai_stage_session_options(log_id: str) -> dict: - """Return the ``session_options`` block that routes a stage to the AMD NPU. - - Routes a Qwen3 transformer stage to the AMD Ryzen AI NPU via the VitisAI - execution provider. The provider options match the AMD reference inference - configuration (``waic_target_vaiml_cpp_me`` VAIML C++ backend with the - XMC runner and linear-slice disabled). +def _stage_session_options( + log_id: str, + ep: str, + device: str, + *, + provider_options: dict[str, str] | None = None, + intra_op_num_threads: int = 2, +) -> dict: + """Return session options for any cataloged EP/device target. Args: - log_id: ORT log identifier (shown in ORT logs), e.g. - ``"onnxruntime-genai.context"``. + log_id: ORT log identifier. + ep: Execution provider alias or full name. + device: Concrete device category. + provider_options: Options overlaid on the catalog defaults. + intra_op_num_threads: Stage intra-op thread count. Returns: Dict suitable for the ``session_options`` key of a pipeline stage in ``genai_config.json``. """ + canonical = normalize_ep_name(ep) + spec = lookup_device_spec(canonical, device) + if spec is None: + raise ValueError(f"Unsupported GenAI stage target: ep={ep!r}, device={device!r}") + options = {**spec.default_provider_options, **(provider_options or {})} return { "log_id": log_id, - "provider_options": [ - { - "vitisai": { - "target": "waic_target_vaiml_cpp_me", - "xmc_runner_config": "1", - "no_linear_slice": "1", - } - } - ], - "intra_op_num_threads": 8, - "inter_op_num_threads": 1, - } - - -def build_npu_load_config( - *, - weights_path: str | Path | None = None, -) -> str: - """Build the JSON-string ``load_config`` for the OpenVINO NPU plugin. - - Args: - weights_path: Optional external-weights directory. Emitted paths are - absolute so loading a derived bundle does not change their meaning. - - Returns: - Serialized OpenVINO load configuration, not a filename. - """ - npu_config = dict(_OPENVINO_NPU_DEFAULTS) - if weights_path is not None: - npu_config["WEIGHTS_PATH"] = str(Path(weights_path).expanduser().resolve()) - - return json.dumps({"NPU": npu_config}, sort_keys=True) - - -def openvino_stage_session_options( - log_id: str, - *, - weights_path: str | Path | None = None, -) -> dict: - """Return session options routing a transformer stage to the Intel NPU. - - ``load_config`` is a JSON string containing OpenVINO properties, not ORT - session entries. Plugin registration, ABI device binding, and EPContext - compilation remain owned by the existing session/compiler infrastructure. - See :func:`build_npu_load_config` for the optional configuration arguments. - """ - return { - "log_id": log_id, - "provider_options": [ - { - "openvino": { - "device_type": "NPU", - "load_config": build_npu_load_config(weights_path=weights_path), - } - } - ], - "intra_op_num_threads": 2, + "provider_options": [{short_ep_name(canonical): options}], + "intra_op_num_threads": intra_op_num_threads, "inter_op_num_threads": 1, } -def _stage_session_options( +def _transformer_stage_session_options( ep: str, soc_model: str, *, @@ -202,24 +119,62 @@ def _stage_session_options( if canonical != "OpenVINOExecutionProvider" and openvino_weights_path is not None: raise ValueError("OpenVINO configuration requires ep='openvino'") if canonical == "QNNExecutionProvider": + provider_options = { + "backend_path": "QnnHtp.dll", + "soc_model": soc_model, + } return ( - qnn_stage_session_options("onnxruntime-genai.context", soc_model=soc_model), - qnn_stage_session_options("onnxruntime-genai.iterator", soc_model=soc_model), + _stage_session_options( + "onnxruntime-genai.context", ep, "npu", provider_options=provider_options + ), + _stage_session_options( + "onnxruntime-genai.iterator", ep, "npu", provider_options=provider_options + ), ) if canonical == "VitisAIExecutionProvider": + provider_options = { + "target": "waic_target_vaiml_cpp_me", + "xmc_runner_config": "1", + "no_linear_slice": "1", + } return ( - vitisai_stage_session_options("onnxruntime-genai.context"), - vitisai_stage_session_options("onnxruntime-genai.iterator"), + _stage_session_options( + "onnxruntime-genai.context", + ep, + "npu", + provider_options=provider_options, + intra_op_num_threads=8, + ), + _stage_session_options( + "onnxruntime-genai.iterator", + ep, + "npu", + provider_options=provider_options, + intra_op_num_threads=8, + ), ) if canonical == "OpenVINOExecutionProvider": + spec = lookup_device_spec(canonical, "npu") + if spec is None: + raise ValueError("OpenVINO NPU is missing from the EP/device catalog") + load_config = json.loads(spec.default_provider_options["load_config"]) + if openvino_weights_path is not None: + load_config["NPU"]["WEIGHTS_PATH"] = str( + Path(openvino_weights_path).expanduser().resolve() + ) + provider_options = {"load_config": json.dumps(load_config, sort_keys=True)} return ( - openvino_stage_session_options( + _stage_session_options( "onnxruntime-genai.context", - weights_path=openvino_weights_path, + ep, + "npu", + provider_options=provider_options, ), - openvino_stage_session_options( + _stage_session_options( "onnxruntime-genai.iterator", - weights_path=openvino_weights_path, + ep, + "npu", + provider_options=provider_options, ), ) return None, None @@ -297,7 +252,7 @@ def build_qwen3_transformer_only_stages( ``(stages, decoder_io)`` — see :func:`~winml.modelkit.utils.genai.build_decoder_pipeline_stages`. """ - ctx_opts, iter_opts = _stage_session_options( + ctx_opts, iter_opts = _transformer_stage_session_options( ep, soc_model, openvino_weights_path=openvino_weights_path, @@ -360,7 +315,7 @@ def write_genai_bundle( """ if normalize_ep_name(ep) == "OpenVINOExecutionProvider" and openvino_weights_path is None: openvino_weights_path = output_dir - ctx_opts, iter_opts = _stage_session_options( + ctx_opts, iter_opts = _transformer_stage_session_options( ep, soc_model, openvino_weights_path=openvino_weights_path, @@ -394,12 +349,8 @@ def write_genai_bundle( "PipelineStage", "build_decoder_pipeline_stages", "build_genai_config", - "build_npu_load_config", "build_qwen3_transformer_only_stages", - "openvino_stage_session_options", - "qnn_stage_session_options", "strip_gqa_default_attrs", - "vitisai_stage_session_options", "write_genai_bundle", ] diff --git a/src/winml/modelkit/session/ep_device.py b/src/winml/modelkit/session/ep_device.py index 4bba1fdf4..6f726c254 100644 --- a/src/winml/modelkit/session/ep_device.py +++ b/src/winml/modelkit/session/ep_device.py @@ -357,6 +357,7 @@ class EPDeviceSpec: ep: EPName device: DeviceType default_provider_options: Mapping[str, str] = field(default_factory=dict) + use_defaults_for_genai: bool = False provider_option_hints: Mapping[str, str] = field(default_factory=dict) @@ -386,7 +387,18 @@ class EPDeviceSpec: "backend_type": "htp", }, ), - EPDeviceSpec(ep="OpenVINOExecutionProvider", device="npu"), + EPDeviceSpec( + ep="OpenVINOExecutionProvider", + device="npu", + default_provider_options={ + "device_type": "NPU", + "load_config": ( + '{"NPU":{"CACHE_MODE":"OPTIMIZE_SPEED","NPU_COMPILER_TYPE":"PLUGIN",' + '"NPU_QDQ_OPTIMIZATION":"YES","NPU_TURBO":"YES"}}' + ), + }, + use_defaults_for_genai=True, + ), EPDeviceSpec(ep="VitisAIExecutionProvider", device="npu"), EPDeviceSpec(ep="OpenVINOExecutionProvider", device="gpu"), EPDeviceSpec(ep="MIGraphXExecutionProvider", device="gpu"), diff --git a/src/winml/modelkit/session/genai_session.py b/src/winml/modelkit/session/genai_session.py index d972a2cef..109886f35 100644 --- a/src/winml/modelkit/session/genai_session.py +++ b/src/winml/modelkit/session/genai_session.py @@ -64,6 +64,7 @@ from .ep_device import ( VALID_EPS, device_from_provider_option_hints, + lookup_device_spec, short_ep_name, ) from .ep_registry import WinMLEPRegistry @@ -1107,20 +1108,62 @@ def _override_stage_provider(self, stage_cfg: dict[str, Any], borrow_opts: dict) return alias = short_ep_name(ep) + device = self._device or EP_SUPPORTED_DEVICES[ep][0] + spec = lookup_device_spec(ep, device) + defaults = ( + dict(spec.default_provider_options) + if spec is not None and spec.use_defaults_for_genai + else {} + ) if self._stage_targets_ep(current_po, ep): - # Re-selecting the stage's own EP: preserve its shipped options - # verbatim (even when empty) — this is a byte-for-byte no-op. - opts = self._existing_opts_for_ep(current_po, ep) + opts = self._merge_provider_options( + defaults, + self._existing_opts_for_ep(current_po, ep), + ) elif borrow_opts: - opts = dict(borrow_opts) + opts = self._merge_provider_options(defaults, borrow_opts) else: - opts = self._default_opts_for_device(ep) - opts = {**opts, **self._provider_options} + opts = self._merge_provider_options(defaults, self._default_opts_for_device(ep)) + opts = self._merge_provider_options(opts, self._provider_options) if not isinstance(so, dict): so = {} stage_cfg["session_options"] = so so["provider_options"] = [{alias: opts}] + @staticmethod + def _merge_provider_options( + defaults: dict[str, Any], + overrides: dict[str, Any], + ) -> dict[str, Any]: + """Merge provider options, preserving nested JSON configuration defaults.""" + merged = {**defaults, **overrides} + for key in defaults.keys() & overrides.keys(): + try: + default_value = json.loads(defaults[key]) + override_value = json.loads(overrides[key]) + except (TypeError, json.JSONDecodeError): + continue + if isinstance(default_value, dict) and isinstance(override_value, dict): + merged[key] = json.dumps( + GenaiSession._merge_nested_dicts(default_value, override_value), + sort_keys=True, + separators=(",", ":"), + ) + return merged + + @staticmethod + def _merge_nested_dicts( + defaults: dict[str, Any], + overrides: dict[str, Any], + ) -> dict[str, Any]: + merged = copy.deepcopy(defaults) + for key, value in overrides.items(): + if isinstance(value, dict) and isinstance(merged.get(key), dict): + merged[key] = GenaiSession._merge_nested_dicts(merged[key], value) + else: + merged[key] = value + return merged + def _default_opts_for_device(self, ep: EPName) -> dict[str, Any]: """Synthesize provider options for *ep* when the bundle defines none. diff --git a/tests/unit/models/qwen3/test_genai_config.py b/tests/unit/models/qwen3/test_genai_config.py index 775e1ebf9..e8434e6d3 100644 --- a/tests/unit/models/qwen3/test_genai_config.py +++ b/tests/unit/models/qwen3/test_genai_config.py @@ -7,7 +7,6 @@ from __future__ import annotations import json -from pathlib import Path from types import SimpleNamespace from typing import ClassVar from unittest.mock import patch @@ -18,9 +17,7 @@ DecoderIOMapping, PipelineStage, build_genai_config, - build_npu_load_config, build_qwen3_transformer_only_stages, - openvino_stage_session_options, write_genai_bundle, ) from winml.modelkit.models.hf.qwen3.genai import ( @@ -393,36 +390,6 @@ def test_single_layer_model(self) -> None: assert result == {"keys_": "keys_%d", "vals_": "vals_%d"} -# --------------------------------------------------------------------------- -# Tests: OpenVINO NPU options and load configuration -# --------------------------------------------------------------------------- - - -class TestOpenVINOSessionOptions: - def test_default_npu_options(self) -> None: - options = openvino_stage_session_options("test.context") - assert options["log_id"] == "test.context" - assert options["intra_op_num_threads"] == 2 - assert options["inter_op_num_threads"] == 1 - provider = options["provider_options"][0]["openvino"] - assert provider["device_type"] == "NPU" - assert json.loads(provider["load_config"]) == { - "NPU": { - "CACHE_MODE": "OPTIMIZE_SPEED", - "NPU_COMPILER_TYPE": "PLUGIN", - "NPU_QDQ_OPTIMIZATION": "YES", - "NPU_TURBO": "YES", - } - } - assert set(provider) == {"device_type", "load_config"} - - def test_weights_default_is_absolute(self, tmp_path, monkeypatch) -> None: - monkeypatch.chdir(tmp_path) - config = json.loads(build_npu_load_config(weights_path="model weights")) - path = Path(config["NPU"]["WEIGHTS_PATH"]) - assert path.is_absolute() - assert path == tmp_path / "model weights" - # --------------------------------------------------------------------------- # Tests: build_qwen3_transformer_only_stages # --------------------------------------------------------------------------- diff --git a/tests/unit/session/test_ep_device.py b/tests/unit/session/test_ep_device.py index 2d088f947..ca99e8267 100644 --- a/tests/unit/session/test_ep_device.py +++ b/tests/unit/session/test_ep_device.py @@ -6,6 +6,7 @@ # tests/unit/session/test_ep_device.py """Unit tests for EPDeviceTarget descriptor and resolution helpers.""" +import json from unittest.mock import MagicMock, patch import pytest @@ -15,6 +16,7 @@ DeviceNotFound, EPDeviceTarget, expand_ep_name, + lookup_device_spec, resolve_device, short_ep_name, ) @@ -41,6 +43,19 @@ def test_ep_device_lowercase_invariant() -> None: assert ep_device.device == "npu" +def test_openvino_npu_defaults_are_cataloged() -> None: + spec = lookup_device_spec("OpenVINOExecutionProvider", "npu") + assert spec is not None + assert spec.use_defaults_for_genai is True + assert spec.default_provider_options["device_type"] == "NPU" + assert json.loads(spec.default_provider_options["load_config"])["NPU"] == { + "CACHE_MODE": "OPTIMIZE_SPEED", + "NPU_COMPILER_TYPE": "PLUGIN", + "NPU_QDQ_OPTIMIZATION": "YES", + "NPU_TURBO": "YES", + } + + def test_from_dict_forward_compat_with_optional_source() -> None: """EPDeviceTarget.from_dict round-trips both legacy and new JSON shapes. diff --git a/tests/unit/session/test_genai_session.py b/tests/unit/session/test_genai_session.py index a25d61d2c..9ede6d03c 100644 --- a/tests/unit/session/test_genai_session.py +++ b/tests/unit/session/test_genai_session.py @@ -851,9 +851,9 @@ def test_reroute_synthesizes_device_type_for_openvino(self, bundle_dir: Path) -> ) effective, _ = session._apply_ep_override(cfg) stage = effective["model"]["decoder"]["pipeline"][0]["context"] - assert stage["session_options"]["provider_options"] == [ - {"openvino": {"device_type": "NPU"}} - ] + provider = stage["session_options"]["provider_options"][0]["openvino"] + assert provider["device_type"] == "NPU" + assert json.loads(provider["load_config"])["NPU"]["NPU_TURBO"] == "YES" def test_reroute_openvino_defaults_device_type_without_device(self, bundle_dir: Path) -> None: # No --device: fall back to the EP's primary supported device (npu). @@ -863,9 +863,20 @@ def test_reroute_openvino_defaults_device_type_without_device(self, bundle_dir: ) effective, _ = session._apply_ep_override(cfg) stage = effective["model"]["decoder"]["pipeline"][0]["context"] - assert stage["session_options"]["provider_options"] == [ - {"openvino": {"device_type": "NPU"}} - ] + provider = stage["session_options"]["provider_options"][0]["openvino"] + assert provider["device_type"] == "NPU" + assert json.loads(provider["load_config"])["NPU"]["NPU_TURBO"] == "YES" + + def test_openvino_non_npu_does_not_apply_npu_defaults(self, bundle_dir: Path) -> None: + session = GenaiSession(bundle_dir, ep="openvino", device="gpu") + cfg = self._pipeline_cfg( + {"context": {"session_options": {"provider_options": [{"dml": {}}]}}} + ) + effective, _ = session._apply_ep_override(cfg) + provider = effective["model"]["decoder"]["pipeline"][0]["context"][ + "session_options" + ]["provider_options"][0]["openvino"] + assert provider == {"device_type": "GPU"} def test_reroute_synthesizes_device_type_for_vitisai(self, bundle_dir: Path) -> None: session = GenaiSession(bundle_dir, ep="vitisai", device="npu") @@ -929,9 +940,14 @@ def test_explicit_provider_options_override_bundle_options(self, bundle_dir: Pat provider = effective["model"]["decoder"]["pipeline"][0]["context"]["session_options"][ "provider_options" ][0]["openvino"] - assert provider == { - "device_type": "NPU", - "load_config": '{"NPU":{"NPU_TURBO":"NO"}}', + assert provider["device_type"] == "NPU" + assert json.loads(provider["load_config"]) == { + "NPU": { + "CACHE_MODE": "OPTIMIZE_SPEED", + "NPU_COMPILER_TYPE": "PLUGIN", + "NPU_QDQ_OPTIMIZATION": "YES", + "NPU_TURBO": "NO", + } } def test_force_different_hardware_ep_drops_foreign_options(self, bundle_dir: Path) -> None: From 651b747c9498ad2b2e11c48187daa67bbd2b0fa7 Mon Sep 17 00:00:00 2001 From: Qiong Wu Date: Fri, 9 Oct 2026 12:37:28 +0800 Subject: [PATCH 5/6] Fix strict mypy class annotations Co-authored-by: Copilot App <223556219+Copilot@users.noreply.github.com> --- src/winml/modelkit/export/value_range.py | 11 +++++------ 1 file changed, 5 insertions(+), 6 deletions(-) diff --git a/src/winml/modelkit/export/value_range.py b/src/winml/modelkit/export/value_range.py index b2d990596..cba2463f1 100644 --- a/src/winml/modelkit/export/value_range.py +++ b/src/winml/modelkit/export/value_range.py @@ -125,7 +125,7 @@ def intercept_value_ranges() -> Iterator[dict[str, dict[str, Any]]]: 'token_type_ids': {'min': 0, 'max': 2, 'method': 'random_int_tensor'}} """ captured: dict[str, dict] = {} - originals: dict = {} + originals: dict[str | tuple[type[Any], str], Any] = {} # Patch static tensor gen methods on the base class for method_name in _TENSOR_GEN_METHODS: @@ -138,14 +138,13 @@ def intercept_value_ranges() -> Iterator[dict[str, dict[str, Any]]]: ) # Patch generate() on all subclasses that override it - patched_classes = [] + patched_classes: list[type[Any]] = [] - def _patch_subclasses(base: type) -> None: + def _patch_subclasses(base: type[Any]) -> None: for cls in base.__subclasses__(): if "generate" in cls.__dict__: originals[(cls, "generate")] = cls.__dict__["generate"] - # Monkey-patch optimum's untyped generator hierarchy. - cls.generate = _make_generate_wrapper(cls.__dict__["generate"]) # type: ignore[attr-defined] + cls.generate = _make_generate_wrapper(cls.__dict__["generate"]) patched_classes.append(cls) _patch_subclasses(cls) @@ -162,4 +161,4 @@ def _patch_subclasses(base: type) -> None: staticmethod(originals[method_name]), ) for cls in patched_classes: - cls.generate = originals[(cls, "generate")] # type: ignore[attr-defined] + cls.generate = originals[cls, "generate"] From a3c625899e6505a9e426447103b9fa9ed6c8e236 Mon Sep 17 00:00:00 2001 From: Qiong Wu Date: Fri, 9 Oct 2026 13:42:32 +0800 Subject: [PATCH 6/6] Support OpenVINO GPU GenAI bundles Co-authored-by: Copilot App <223556219+Copilot@users.noreply.github.com> --- src/winml/modelkit/models/hf/qwen3/genai.py | 65 ++++++++++++------- .../modelkit/models/winml/genai_bundle.py | 1 + src/winml/modelkit/session/ep_device.py | 7 +- .../unit/commands/test_build_genai_bundle.py | 15 ++++- tests/unit/models/qwen3/test_genai_config.py | 27 ++++++++ .../winml/test_genai_bundle_orchestrator.py | 1 + .../winml/test_genai_bundle_registry.py | 5 +- 7 files changed, 94 insertions(+), 27 deletions(-) diff --git a/src/winml/modelkit/models/hf/qwen3/genai.py b/src/winml/modelkit/models/hf/qwen3/genai.py index 0920b1ed2..7c685e5ec 100644 --- a/src/winml/modelkit/models/hf/qwen3/genai.py +++ b/src/winml/modelkit/models/hf/qwen3/genai.py @@ -101,6 +101,7 @@ def _transformer_stage_session_options( ep: str, soc_model: str, *, + device: str = "npu", openvino_weights_path: str | Path | None = None, ) -> tuple[dict | None, dict | None]: """Return ``(context, iterator)`` session_options for the given EP. @@ -109,13 +110,14 @@ def _transformer_stage_session_options( * ``ep="qnn"`` -> Qualcomm QNN HTP (``soc_model`` selects the Snapdragon SoC). * ``ep="vitisai"`` -> AMD Ryzen AI NPU. - * ``ep="openvino"`` -> Intel NPU. + * ``ep="openvino"`` -> Intel NPU or GPU selected by ``device``. Any other value (e.g. ``"cpu"``) leaves the stages on the default CPU provider. Short aliases and full ``*ExecutionProvider`` names are both accepted (normalized via :func:`normalize_ep_name`). """ canonical = normalize_ep_name(ep) + device = device.lower() if canonical != "OpenVINOExecutionProvider" and openvino_weights_path is not None: raise ValueError("OpenVINO configuration requires ep='openvino'") if canonical == "QNNExecutionProvider": @@ -154,27 +156,31 @@ def _transformer_stage_session_options( ), ) if canonical == "OpenVINOExecutionProvider": - spec = lookup_device_spec(canonical, "npu") + spec = lookup_device_spec(canonical, device) if spec is None: - raise ValueError("OpenVINO NPU is missing from the EP/device catalog") - load_config = json.loads(spec.default_provider_options["load_config"]) - if openvino_weights_path is not None: - load_config["NPU"]["WEIGHTS_PATH"] = str( - Path(openvino_weights_path).expanduser().resolve() - ) - provider_options = {"load_config": json.dumps(load_config, sort_keys=True)} + raise ValueError(f"OpenVINO {device.upper()} is missing from the EP/device catalog") + openvino_options: dict[str, str] = {} + if device == "npu": + load_config = json.loads(spec.default_provider_options["load_config"]) + if openvino_weights_path is not None: + load_config["NPU"]["WEIGHTS_PATH"] = str( + Path(openvino_weights_path).expanduser().resolve() + ) + openvino_options["load_config"] = json.dumps(load_config, sort_keys=True) + elif openvino_weights_path is not None: + raise ValueError("OpenVINO weights path is only supported for device='npu'") return ( _stage_session_options( "onnxruntime-genai.context", ep, - "npu", - provider_options=provider_options, + device, + provider_options=openvino_options, ), _stage_session_options( "onnxruntime-genai.iterator", ep, - "npu", - provider_options=provider_options, + device, + provider_options=openvino_options, ), ) return None, None @@ -219,14 +225,15 @@ def build_qwen3_transformer_only_stages( embeddings_filename: str = DEFAULT_EMBEDDINGS_FILENAME, lm_head_filename: str = DEFAULT_LM_HEAD_FILENAME, ep: str = "cpu", + device: str = "npu", soc_model: str = "60", openvino_weights_path: str | Path | None = None, ) -> tuple[list[PipelineStage], DecoderIOMapping]: - """Build the Qwen3 4-stage pipeline, routing ctx/iter to the NPU per ``ep``. + """Build the Qwen3 4-stage pipeline, routing ctx/iter per ``ep``/``device``. Qwen3-specific wrapper over :func:`winml.modelkit.utils.genai.build_decoder_pipeline_stages` that injects - the NPU ``session_options`` for the transformer stages. Tensor names are + the accelerator ``session_options`` for the transformer stages. Tensor names are still discovered by introspecting the ONNX graphs, so nothing is hardcoded. Args: @@ -237,11 +244,12 @@ def build_qwen3_transformer_only_stages( iterator_filename: Bundle filename for the iterator model. embeddings_filename: Bundle filename for the embeddings model. lm_head_filename: Bundle filename for the lm_head model. - ep: NPU execution provider for the ``context``/``iterator`` stages — + ep: Execution provider for the ``context``/``iterator`` stages. ``"qnn"`` (Qualcomm), ``"vitisai"`` (AMD), or ``"openvino"`` (Intel) - injects that EP's ``session_options`` so those stages run on the NPU while - ``embeddings`` and ``lm_head`` stay on CPU. ``"cpu"`` (default) - omits them. + injects that EP's ``session_options`` while ``embeddings`` and + ``lm_head`` stay on CPU. ``"cpu"`` (default) omits them. + device: Device targeted by ``ep``. OpenVINO supports ``"npu"`` and + ``"gpu"``; the other accelerated recipe targets use ``"npu"``. soc_model: Snapdragon SoC model number forwarded to the QNN backend when ``ep="qnn"``. Default ``"60"`` targets Snapdragon 8 Gen 3. Ignored for non-QNN EPs. @@ -255,6 +263,7 @@ def build_qwen3_transformer_only_stages( ctx_opts, iter_opts = _transformer_stage_session_options( ep, soc_model, + device=device, openvino_weights_path=openvino_weights_path, ) return build_decoder_pipeline_stages( @@ -285,11 +294,12 @@ def write_genai_bundle( embeddings_filename: str = DEFAULT_EMBEDDINGS_FILENAME, lm_head_filename: str = DEFAULT_LM_HEAD_FILENAME, ep: str = "cpu", + device: str = "npu", soc_model: str = "60", transformer_onnx_passes: Sequence[Callable[[onnx.ModelProto], onnx.ModelProto]] | None = None, openvino_weights_path: str | Path | None = None, ) -> Path: - """Assemble a Qwen3 genai bundle, routing ctx/iter to the NPU per ``ep``. + """Assemble a Qwen3 genai bundle, routing ctx/iter per ``ep``/``device``. Qwen3-specific wrapper over :func:`winml.modelkit.utils.genai.write_genai_bundle` that supplies the NPU @@ -297,10 +307,12 @@ def write_genai_bundle( the description of every other argument. Args: - ep: NPU execution provider routing the transformer (context/iterator) + ep: Execution provider routing the transformer (context/iterator) stages — ``"qnn"`` (Qualcomm HTP), ``"vitisai"`` (AMD Ryzen AI), - or ``"openvino"`` (Intel NPU); + or ``"openvino"`` (Intel NPU/GPU); ``"cpu"`` (default) keeps every stage on CPU. + device: Device targeted by ``ep``. OpenVINO supports ``"npu"`` and + ``"gpu"``; the other accelerated recipe targets use ``"npu"``. soc_model: Snapdragon SoC model passed to the QNN backend when ``ep="qnn"``. Default ``"60"`` = Snapdragon 8 Gen 3 / X Elite. Ignored for non-QNN EPs. @@ -313,11 +325,17 @@ def write_genai_bundle( Returns: Path to the written ``genai_config.json``. """ - if normalize_ep_name(ep) == "OpenVINOExecutionProvider" and openvino_weights_path is None: + device = device.lower() + if ( + normalize_ep_name(ep) == "OpenVINOExecutionProvider" + and device == "npu" + and openvino_weights_path is None + ): openvino_weights_path = output_dir ctx_opts, iter_opts = _transformer_stage_session_options( ep, soc_model, + device=device, openvino_weights_path=openvino_weights_path, ) return _write_genai_bundle( @@ -393,6 +411,7 @@ def write_genai_bundle( GenaiTarget(ep="qnn", device="npu"), # Qualcomm Snapdragon NPU GenaiTarget(ep="vitisai", device="npu"), # AMD Ryzen AI NPU GenaiTarget(ep="openvino", device="npu"), # Intel NPU + GenaiTarget(ep="openvino", device="gpu"), # Intel GPU GenaiTarget(ep="cpu", device="cpu"), ), transformer_onnx_passes=(strip_gqa_default_attrs,), diff --git a/src/winml/modelkit/models/winml/genai_bundle.py b/src/winml/modelkit/models/winml/genai_bundle.py index 3d5cce491..8254ca93b 100644 --- a/src/winml/modelkit/models/winml/genai_bundle.py +++ b/src/winml/modelkit/models/winml/genai_bundle.py @@ -382,6 +382,7 @@ def build_genai_bundle( max_cache_len=max_cache_len, prefill_seq_len=prefill_seq_len, ep=ep, + device=device, soc_model=soc_model, transformer_onnx_passes=list(recipe.transformer_onnx_passes), **{f"{role}_src": path for role, path in companion_srcs.items()}, diff --git a/src/winml/modelkit/session/ep_device.py b/src/winml/modelkit/session/ep_device.py index 6f726c254..3a11a92fc 100644 --- a/src/winml/modelkit/session/ep_device.py +++ b/src/winml/modelkit/session/ep_device.py @@ -400,7 +400,12 @@ class EPDeviceSpec: use_defaults_for_genai=True, ), EPDeviceSpec(ep="VitisAIExecutionProvider", device="npu"), - EPDeviceSpec(ep="OpenVINOExecutionProvider", device="gpu"), + EPDeviceSpec( + ep="OpenVINOExecutionProvider", + device="gpu", + default_provider_options={"device_type": "GPU"}, + use_defaults_for_genai=True, + ), EPDeviceSpec(ep="MIGraphXExecutionProvider", device="gpu"), EPDeviceSpec(ep="TensorrtExecutionProvider", device="gpu"), EPDeviceSpec(ep="NvTensorRTRTXExecutionProvider", device="gpu"), diff --git a/tests/unit/commands/test_build_genai_bundle.py b/tests/unit/commands/test_build_genai_bundle.py index c642b267f..67151ec20 100644 --- a/tests/unit/commands/test_build_genai_bundle.py +++ b/tests/unit/commands/test_build_genai_bundle.py @@ -534,6 +534,7 @@ def test_use_cache_rejected_for_bundle(tmp_path: Path): ("resolved_target", "expected_ep", "expected_device"), [ (EPDeviceTarget(ep="QNNExecutionProvider", device="npu"), "qnn", "npu"), + (EPDeviceTarget(ep="OpenVINOExecutionProvider", device="gpu"), "openvino", "gpu"), (EPDeviceTarget(ep="CPUExecutionProvider", device="cpu"), "cpu", "cpu"), ], ) @@ -546,7 +547,7 @@ def test_export_type_optimized_resolves_target_and_builds( """``--export-type optimized`` resolves the host target, then builds its recipe. No ``--device``/``--ep`` is pinned, so the target is hardware-probed. - Both CPU and QNN/NPU are supported by the recipe. + CPU, QNN/NPU, and OpenVINO/GPU are supported by the recipe. """ out = tmp_path / "bundle" recorded: dict = {} @@ -743,6 +744,18 @@ def test_resolve_optimized_target_rejects_unsupported_ep(): _resolve_optimized_target(recipe, device="gpu", ep="dml") +def test_resolve_optimized_target_accepts_openvino_gpu(): + from winml.modelkit.commands.build import _resolve_optimized_target + from winml.modelkit.models.winml import resolve_genai_bundle + + recipe = resolve_genai_bundle("qwen3") + assert recipe is not None + assert _resolve_optimized_target(recipe, device="gpu", ep="openvino") == ( + "openvino", + "gpu", + ) + + def test_resolve_optimized_target_rejects_unsupported_device(): from winml.modelkit.commands.build import _resolve_optimized_target from winml.modelkit.models.winml import resolve_genai_bundle diff --git a/tests/unit/models/qwen3/test_genai_config.py b/tests/unit/models/qwen3/test_genai_config.py index e8434e6d3..ef6fccc85 100644 --- a/tests/unit/models/qwen3/test_genai_config.py +++ b/tests/unit/models/qwen3/test_genai_config.py @@ -603,6 +603,20 @@ def test_openvino_defaults_reach_serialized_pipeline(self) -> None: assert "session_options" not in stage_map["embeddings"] assert "session_options" not in stage_map["lm_head"] + def test_openvino_gpu_uses_gpu_without_npu_config(self) -> None: + with self._patch_onnx(): + stages, _ = build_qwen3_transformer_only_stages( + "ctx.onnx", + "iter.onnx", + num_layers=4, + ep="openvino", + device="gpu", + ) + stage_map = {stage.name: stage for stage in stages} + for name in ("context", "iterator"): + provider = stage_map[name].session_options["provider_options"][0]["openvino"] + assert provider == {"device_type": "GPU"} + # --------------------------------------------------------------------------- # Tests: write_genai_bundle wrapper (ep routing + transformer_onnx_passes) # --------------------------------------------------------------------------- @@ -674,6 +688,19 @@ def test_openvino_defaults_weights_to_bundle_directory(self, tmp_path) -> None: assert config["NPU"]["WEIGHTS_PATH"] == str(output_dir.resolve()) assert "openvino_weights_path" not in kwargs + def test_openvino_gpu_does_not_add_npu_weights_config(self, tmp_path) -> None: + with self._patch_generic() as mock_write: + write_genai_bundle( + tmp_path / "bundle", + ep="openvino", + device="gpu", + **self._COMMON, + ) + kwargs = mock_write.call_args.kwargs + for name in ("context", "iterator"): + provider = kwargs[f"{name}_session_options"]["provider_options"][0]["openvino"] + assert provider == {"device_type": "GPU"} + def test_openvino_forwards_custom_weights_path(self, tmp_path) -> None: weights_dir = tmp_path / "shared weights" with self._patch_generic() as mock_write: diff --git a/tests/unit/models/winml/test_genai_bundle_orchestrator.py b/tests/unit/models/winml/test_genai_bundle_orchestrator.py index 20d3ddd2b..399244cc1 100644 --- a/tests/unit/models/winml/test_genai_bundle_orchestrator.py +++ b/tests/unit/models/winml/test_genai_bundle_orchestrator.py @@ -183,6 +183,7 @@ def test_assembler_receives_paths_ep_and_passes(harness): assert Path(ak["embeddings_src"]) == onnx_file assert Path(ak["lm_head_src"]) == onnx_file assert ak["ep"] == "qnn" # short token forwarded verbatim to the assembler + assert ak["device"] == "npu" assert ak["soc_model"] == "60" assert ak["model_id"] == "m" assert ak["max_cache_len"] == 2048 diff --git a/tests/unit/models/winml/test_genai_bundle_registry.py b/tests/unit/models/winml/test_genai_bundle_registry.py index 064cb0533..298e9058b 100644 --- a/tests/unit/models/winml/test_genai_bundle_registry.py +++ b/tests/unit/models/winml/test_genai_bundle_registry.py @@ -37,11 +37,12 @@ def test_resolve_qwen3_returns_recipe(): assert len(recipe.transformer_onnx_passes) >= 1 -def test_qwen3_openvino_target_is_npu_only(): +def test_qwen3_openvino_targets_include_npu_and_gpu(): recipe = resolve_genai_bundle("qwen3") assert recipe is not None assert {target.device for target in recipe.supported_targets if target.ep == "openvino"} == { - "npu" + "npu", + "gpu", }