diff --git a/npu_config_npuw.json b/npu_config_npuw.json new file mode 100644 index 000000000..69154cec2 --- /dev/null +++ b/npu_config_npuw.json @@ -0,0 +1,12 @@ +{ + "NPU": { + "NPU_TURBO": "YES", + "NPU_QDQ_OPTIMIZATION": "YES", + "NPU_COMPILER_TYPE": "DRIVER", + "NPU_USE_NPUW": "YES", + "NPUW_DEVICES": "NPU", + "NPUW_PARALLEL_COMPILE": "YES", + "NPUW_GQA": "YES", + "CACHE_MODE": "OPTIMIZE_SPEED" + } +} \ No newline at end of file diff --git a/src/winml/modelkit/commands/perf.py b/src/winml/modelkit/commands/perf.py index c1395a950..36652c89e 100644 --- a/src/winml/modelkit/commands/perf.py +++ b/src/winml/modelkit/commands/perf.py @@ -2504,6 +2504,14 @@ def _autobuild_genai_bundle( build_ep, build_device = short_ep_name(target.ep), target.device # Do not reuse a bundle exported for a different execution provider. bundle_dir = bundle_dir.with_name(f"genai-bundle-{build_ep}-{build_device}") + openvino_config: Path | None = p.get("openvino_config") + if openvino_config is not None: + if build_ep != "openvino": + raise click.UsageError("--openvino-config requires --ep openvino.") + import hashlib + + config_digest = hashlib.sha256(openvino_config.read_bytes()).hexdigest()[:12] + bundle_dir = bundle_dir.with_name(f"{bundle_dir.name}-config-{config_digest}") build_cache_dir = cache_dir # --rebuild overwrites the cached bundle; a plain run reuses it. Checked # before any model resolution so a cache hit never touches the network. @@ -2564,6 +2572,9 @@ def _autobuild_genai_bundle( force_rebuild=force_rebuild, cache_dir=build_cache_dir, emit=lambda msg: console.print(msg, markup=False), + assemble_options=( + {"openvino_config_path": openvino_config} if openvino_config is not None else None + ), ) return bundle_dir, True @@ -2784,6 +2795,13 @@ def _validate_duration( help="[ort-genai] Max seconds to compile each EPContext stage before falling back " "to the original ONNX (requires --compile).", ) +@click.option( + "--openvino-config", + type=click.Path(exists=True, dir_okay=False, path_type=Path), + default=None, + help="[ort-genai] OpenVINO load-config JSON embedded while auto-building a model ID. " + "Requires --ep openvino.", +) @click.option( "--task", type=str, @@ -2964,6 +2982,7 @@ def perf( apply_template: bool, max_new_tokens: int, compile_timeout: int, + openvino_config: Path | None, task: str | None, submodel: str | None, iterations: int, diff --git a/src/winml/modelkit/models/hf/qwen3/__init__.py b/src/winml/modelkit/models/hf/qwen3/__init__.py index 6e60e07ee..0e66b9700 100644 --- a/src/winml/modelkit/models/hf/qwen3/__init__.py +++ b/src/winml/modelkit/models/hf/qwen3/__init__.py @@ -17,7 +17,9 @@ PipelineStage, build_decoder_pipeline_stages, build_genai_config, + build_npu_load_config, build_qwen3_transformer_only_stages, + openvino_stage_session_options, strip_gqa_default_attrs, write_genai_bundle, ) @@ -28,7 +30,9 @@ "PipelineStage", "build_decoder_pipeline_stages", "build_genai_config", + "build_npu_load_config", "build_qwen3_transformer_only_stages", + "openvino_stage_session_options", "strip_gqa_default_attrs", "write_genai_bundle", ] diff --git a/src/winml/modelkit/models/hf/qwen3/genai.py b/src/winml/modelkit/models/hf/qwen3/genai.py index 9b661fd30..ceb5aab67 100644 --- a/src/winml/modelkit/models/hf/qwen3/genai.py +++ b/src/winml/modelkit/models/hf/qwen3/genai.py @@ -11,11 +11,12 @@ This module adds the **Qwen3-specific** layer on top: the Qwen3 transformer stages run on an NPU backend, so this is where the per-EP ``session_options`` -are constructed. Two NPU execution providers are supported for the +are constructed. Three NPU execution providers are supported for the transformer (context/iterator) stages: * **QNN HTP** — Qualcomm Snapdragon NPU (``ep="qnn"``). * **VitisAI** — AMD Ryzen AI NPU (``ep="vitisai"``). +* **OpenVINO** — Intel NPU via the plugin EP (``ep="openvino"``). Keeping the EP-specific logic here lets the generic utilities stay universal while the Qwen3 bundle emits the correct per-EP ``genai_config.json``. @@ -23,6 +24,8 @@ from __future__ import annotations +import json +from pathlib import Path from typing import TYPE_CHECKING from ....onnx import strip_node_attrs @@ -51,15 +54,16 @@ if TYPE_CHECKING: from collections.abc import Callable, Sequence - from pathlib import Path import onnx # --------------------------------------------------------------------------- -# Qwen3-specific NPU execution-provider routing (QNN / VitisAI) +# Qwen3-specific NPU execution-provider routing (QNN / VitisAI / OpenVINO) # --------------------------------------------------------------------------- +_OPENVINO_CONFIG_ROLE_KEYS = frozenset({"CTX", "ITER", "HEAD"}) + def qnn_stage_session_options(log_id: str, soc_model: str = "60") -> dict: """Return the ``session_options`` block that routes a stage to QNN HTP. @@ -124,19 +128,134 @@ def vitisai_stage_session_options(log_id: str) -> dict: } -def _stage_session_options(ep: str, soc_model: str) -> tuple[dict | None, dict | None]: +def build_npu_load_config( + custom_config_path: str | Path | None = None, + *, + model_role: str | None = None, + weights_path: str | Path | None = None, +) -> str: + """Build the JSON-string ``load_config`` for the OpenVINO NPU plugin. + + Accepts a flat configuration (e.g. ``{"NPU": {...}}``) or a configuration + with ``CTX``/``ITER`` sections. A role-based file must contain the requested + role; missing sections are errors rather than silently using defaults. + ``HEAD`` is recognized only to detect role-based files; the LM head stays + on CPU. No legacy provider setup or tuning flags are injected. + + Args: + custom_config_path: Optional UTF-8 JSON file. Without one, only the + driver compiler default and an explicitly supplied weights path + are included. Custom values take precedence over defaults. + model_role: ``"CTX"`` or ``"ITER"``; required for role-based files. + weights_path: Default external-weights directory. A ``WEIGHTS_PATH`` + in the file takes precedence; relative file values resolve against + that file's directory. Emitted paths are absolute so loading a + derived bundle does not change their meaning. + + Returns: + Serialized OpenVINO load configuration, not a filename. + """ + if model_role not in (None, "CTX", "ITER"): + raise ValueError("OpenVINO model_role must be 'CTX' or 'ITER'") + + config: dict = {} + config_path = ( + Path(custom_config_path).expanduser().resolve() if custom_config_path is not None else None + ) + if config_path is not None: + try: + config = json.loads(config_path.read_text(encoding="utf-8-sig")) + except json.JSONDecodeError as exc: + raise ValueError(f"Invalid OpenVINO load config JSON in {config_path}: {exc}") from exc + if not isinstance(config, dict): + raise TypeError("OpenVINO load config must be a JSON object") + if _OPENVINO_CONFIG_ROLE_KEYS.intersection(config): + if model_role is None: + raise ValueError("A role-based OpenVINO load config requires model_role") + if model_role not in config: + raise ValueError(f"OpenVINO load config is missing the {model_role} section") + config = config[model_role] + if not isinstance(config, dict): + raise TypeError(f"OpenVINO {model_role} configuration must be a JSON object") + + npu_config = config.setdefault("NPU", {}) + if not isinstance(npu_config, dict): + raise TypeError("OpenVINO NPU configuration must be a JSON object") + npu_config.setdefault("NPU_COMPILER_TYPE", "DRIVER") + + if "WEIGHTS_PATH" in npu_config: + configured_weights = npu_config["WEIGHTS_PATH"] + if not isinstance(configured_weights, str) or not configured_weights.strip(): + raise ValueError("OpenVINO WEIGHTS_PATH must be a non-empty string") + resolved_weights = Path(configured_weights).expanduser() + if not resolved_weights.is_absolute() and config_path is not None: + resolved_weights = config_path.parent / resolved_weights + npu_config["WEIGHTS_PATH"] = str(resolved_weights.resolve()) + elif weights_path is not None: + npu_config["WEIGHTS_PATH"] = str(Path(weights_path).expanduser().resolve()) + + # Stable serialization keeps equivalent CTX/ITER options shareable even + # when their keys occur in a different order in the user's JSON file. + return json.dumps(config, sort_keys=True) + + +def openvino_stage_session_options( + log_id: str, + *, + custom_config_path: str | Path | None = None, + model_role: str | None = None, + weights_path: str | Path | None = None, +) -> dict: + """Return session options routing a transformer stage to the Intel NPU. + + ``load_config`` is a JSON string containing OpenVINO properties, not ORT + session entries. Plugin registration, ABI device binding, and EPContext + compilation remain owned by the existing session/compiler infrastructure. + See :func:`build_npu_load_config` for the optional configuration arguments. + """ + return { + "log_id": log_id, + "provider_options": [ + { + "openvino": { + "device_type": "NPU", + "load_config": build_npu_load_config( + custom_config_path, + model_role=model_role, + weights_path=weights_path, + ), + } + } + ], + "intra_op_num_threads": 2, + "inter_op_num_threads": 1, + } + + +def _stage_session_options( + ep: str, + soc_model: str, + *, + openvino_config_path: str | Path | None = None, + openvino_weights_path: str | Path | None = None, +) -> tuple[dict | None, dict | None]: """Return ``(context, iterator)`` session_options for the given EP. Routes the Qwen3 transformer (context/iterator) stages to an NPU backend: * ``ep="qnn"`` -> Qualcomm QNN HTP (``soc_model`` selects the Snapdragon SoC). * ``ep="vitisai"`` -> AMD Ryzen AI NPU. + * ``ep="openvino"`` -> Intel NPU, with optional per-role load configuration. Any other value (e.g. ``"cpu"``) leaves the stages on the default CPU provider. Short aliases and full ``*ExecutionProvider`` names are both accepted (normalized via :func:`normalize_ep_name`). """ canonical = normalize_ep_name(ep) + if canonical != "OpenVINOExecutionProvider" and ( + openvino_config_path is not None or openvino_weights_path is not None + ): + raise ValueError("OpenVINO configuration requires ep='openvino'") if canonical == "QNNExecutionProvider": return ( qnn_stage_session_options("onnxruntime-genai.context", soc_model=soc_model), @@ -147,6 +266,21 @@ def _stage_session_options(ep: str, soc_model: str) -> tuple[dict | None, dict | vitisai_stage_session_options("onnxruntime-genai.context"), vitisai_stage_session_options("onnxruntime-genai.iterator"), ) + if canonical == "OpenVINOExecutionProvider": + return ( + openvino_stage_session_options( + "onnxruntime-genai.context", + custom_config_path=openvino_config_path, + model_role="CTX", + weights_path=openvino_weights_path, + ), + openvino_stage_session_options( + "onnxruntime-genai.iterator", + custom_config_path=openvino_config_path, + model_role="ITER", + weights_path=openvino_weights_path, + ), + ) return None, None @@ -190,6 +324,8 @@ def build_qwen3_transformer_only_stages( lm_head_filename: str = DEFAULT_LM_HEAD_FILENAME, ep: str = "cpu", soc_model: str = "60", + openvino_config_path: str | Path | None = None, + openvino_weights_path: str | Path | None = None, ) -> tuple[list[PipelineStage], DecoderIOMapping]: """Build the Qwen3 4-stage pipeline, routing ctx/iter to the NPU per ``ep``. @@ -207,19 +343,28 @@ def build_qwen3_transformer_only_stages( embeddings_filename: Bundle filename for the embeddings model. lm_head_filename: Bundle filename for the lm_head model. ep: NPU execution provider for the ``context``/``iterator`` stages — - ``"qnn"`` (Qualcomm) or ``"vitisai"`` (AMD) injects that EP's - ``session_options`` so those stages run on the NPU while + ``"qnn"`` (Qualcomm), ``"vitisai"`` (AMD), or ``"openvino"`` (Intel) + injects that EP's ``session_options`` so those stages run on the NPU while ``embeddings`` and ``lm_head`` stay on CPU. ``"cpu"`` (default) omits them. soc_model: Snapdragon SoC model number forwarded to the QNN backend when ``ep="qnn"``. Default ``"60"`` targets Snapdragon 8 Gen 3. Ignored for non-QNN EPs. + openvino_config_path: Optional OpenVINO JSON file shared by both stages + or containing separate ``CTX``/``ITER`` sections. Only for OpenVINO. + openvino_weights_path: Optional default external-weights directory for + OpenVINO. Explicit ``WEIGHTS_PATH`` values in the JSON take precedence. Returns: ``(stages, decoder_io)`` — see :func:`~winml.modelkit.utils.genai.build_decoder_pipeline_stages`. """ - ctx_opts, iter_opts = _stage_session_options(ep, soc_model) + ctx_opts, iter_opts = _stage_session_options( + ep, + soc_model, + openvino_config_path=openvino_config_path, + openvino_weights_path=openvino_weights_path, + ) return build_decoder_pipeline_stages( context_onnx, iterator_onnx, @@ -250,6 +395,8 @@ def write_genai_bundle( ep: str = "cpu", soc_model: str = "60", transformer_onnx_passes: Sequence[Callable[[onnx.ModelProto], onnx.ModelProto]] | None = None, + openvino_config_path: str | Path | None = None, + openvino_weights_path: str | Path | None = None, ) -> Path: """Assemble a Qwen3 genai bundle, routing ctx/iter to the NPU per ``ep``. @@ -260,7 +407,8 @@ def write_genai_bundle( Args: ep: NPU execution provider routing the transformer (context/iterator) - stages — ``"qnn"`` (Qualcomm HTP) or ``"vitisai"`` (AMD Ryzen AI); + stages — ``"qnn"`` (Qualcomm HTP), ``"vitisai"`` (AMD Ryzen AI), + or ``"openvino"`` (Intel NPU); ``"cpu"`` (default) keeps every stage on CPU. soc_model: Snapdragon SoC model passed to the QNN backend when ``ep="qnn"``. Default ``"60"`` = Snapdragon 8 Gen 3 / X Elite. @@ -268,11 +416,25 @@ def write_genai_bundle( transformer_onnx_passes: Optional ONNX graph transforms applied to the copied context/iterator models before ``genai_config.json`` is written. Forwarded verbatim to the generic assembler. + openvino_config_path: Optional flat or per-role OpenVINO JSON file. + Its contents are embedded in stage provider options; the original + configuration file is not required at runtime. Different role + options prevent grouped compilation in the current runtime. + openvino_weights_path: Default external-weights directory for OpenVINO; + defaults to the output bundle directory. Explicit ``WEIGHTS_PATH`` + values in the JSON take precedence. Other EPs reject these options. Returns: Path to the written ``genai_config.json``. """ - ctx_opts, iter_opts = _stage_session_options(ep, soc_model) + if normalize_ep_name(ep) == "OpenVINOExecutionProvider" and openvino_weights_path is None: + openvino_weights_path = output_dir + ctx_opts, iter_opts = _stage_session_options( + ep, + soc_model, + openvino_config_path=openvino_config_path, + openvino_weights_path=openvino_weights_path, + ) return _write_genai_bundle( output_dir, context_onnx=context_onnx, @@ -302,7 +464,9 @@ def write_genai_bundle( "PipelineStage", "build_decoder_pipeline_stages", "build_genai_config", + "build_npu_load_config", "build_qwen3_transformer_only_stages", + "openvino_stage_session_options", "qnn_stage_session_options", "strip_gqa_default_attrs", "vitisai_stage_session_options", @@ -347,6 +511,7 @@ def write_genai_bundle( supported_targets=( GenaiTarget(ep="qnn", device="npu"), # Qualcomm Snapdragon NPU GenaiTarget(ep="vitisai", device="npu"), # AMD Ryzen AI NPU + GenaiTarget(ep="openvino", device="npu"), # Intel NPU GenaiTarget(ep="cpu", device="cpu"), ), transformer_onnx_passes=(strip_gqa_default_attrs,), diff --git a/src/winml/modelkit/models/winml/genai_bundle.py b/src/winml/modelkit/models/winml/genai_bundle.py index 419f48908..3d5cce491 100644 --- a/src/winml/modelkit/models/winml/genai_bundle.py +++ b/src/winml/modelkit/models/winml/genai_bundle.py @@ -246,6 +246,7 @@ def build_genai_bundle( force_rebuild: bool = False, cache_dir: str | Path | None = None, emit: Callable[[str], None] | None = None, + assemble_options: Mapping[str, object] | None = None, ) -> Path: """Build (or reuse) every bundle component and assemble the genai bundle. @@ -270,6 +271,8 @@ def build_genai_bundle( force_rebuild: Rebuild components even if cached. cache_dir: Build cache directory override. emit: Optional progress sink invoked with human-readable status lines. + assemble_options: Model-specific keyword arguments forwarded to the + registered bundle assembler. Returns: Path to the written ``genai_config.json``. @@ -382,6 +385,7 @@ def build_genai_bundle( soc_model=soc_model, transformer_onnx_passes=list(recipe.transformer_onnx_passes), **{f"{role}_src": path for role, path in companion_srcs.items()}, + **dict(assemble_options or {}), ) _emit(f" genai_config.json -> {config_path}") return Path(config_path) diff --git a/tests/unit/commands/test_perf_genai.py b/tests/unit/commands/test_perf_genai.py index 626cb361c..797cae26c 100644 --- a/tests/unit/commands/test_perf_genai.py +++ b/tests/unit/commands/test_perf_genai.py @@ -1458,6 +1458,70 @@ def test_hf_model_id_autobuilds_and_dispatches( assert cfg.device == "config" assert cfg.ep is None + def test_openvino_config_is_forwarded_to_autobuild( + self, runner: CliRunner, tmp_path: Path, capture_run: dict, monkeypatch + ) -> None: + import winml.modelkit.loader as loader_mod + import winml.modelkit.models.winml as winml_models + + config_path = tmp_path / "npu.json" + config_path.write_text('{"NPU": {"NPU_USE_NPUW": "YES"}}', encoding="utf-8") + monkeypatch.setenv("WINML_CACHE_DIR", str(tmp_path / "cache")) + monkeypatch.setattr( + loader_mod, "resolve_loader_config", _fake_resolve_loader_config("qwen3") + ) + build_calls: dict = {} + monkeypatch.setattr( + winml_models, "build_genai_bundle", _fake_build_genai_bundle(build_calls) + ) + + result = runner.invoke( + perf, + [ + "-m", + "Qwen/Qwen3-0.6B", + "--runtime", + "ort-genai", + "--ep", + "openvino", + "--device", + "npu", + "--openvino-config", + str(config_path), + ], + ) + + assert result.exit_code == 0, result.output + assert build_calls["build"]["assemble_options"] == { + "openvino_config_path": config_path + } + assert "-config-" in build_calls["build"]["output_dir"].name + assert capture_run["config"].bundle_dir == build_calls["build"]["output_dir"] + + def test_openvino_config_rejects_other_ep( + self, runner: CliRunner, tmp_path: Path, capture_run: dict + ) -> None: + config_path = tmp_path / "npu.json" + config_path.write_text("{}", encoding="utf-8") + + result = runner.invoke( + perf, + [ + "-m", + "Qwen/Qwen3-0.6B", + "--runtime", + "ort-genai", + "--ep", + "cpu", + "--openvino-config", + str(config_path), + ], + ) + + assert result.exit_code == 2, result.output + assert "--openvino-config requires --ep openvino" in result.output + assert "config" not in capture_run + @pytest.mark.parametrize( ("args", "resolved_ep", "build_ep", "build_device"), [ diff --git a/tests/unit/models/qwen3/test_genai_config.py b/tests/unit/models/qwen3/test_genai_config.py index 63dc7f78e..7683f3f8b 100644 --- a/tests/unit/models/qwen3/test_genai_config.py +++ b/tests/unit/models/qwen3/test_genai_config.py @@ -6,15 +6,21 @@ from __future__ import annotations +import json +from pathlib import Path from types import SimpleNamespace from typing import ClassVar from unittest.mock import patch +import pytest + from winml.modelkit.models.hf.qwen3 import ( DecoderIOMapping, PipelineStage, build_genai_config, + build_npu_load_config, build_qwen3_transformer_only_stages, + openvino_stage_session_options, write_genai_bundle, ) from winml.modelkit.models.hf.qwen3.genai import ( @@ -31,6 +37,16 @@ # --------------------------------------------------------------------------- +@pytest.fixture +def openvino_config_file(tmp_path): + def write_config(payload): + path = tmp_path / "openvino.json" + path.write_text(json.dumps(payload), encoding="utf-8") + return path + + return write_config + + def _mock_config( *, num_hidden_layers: int = 28, @@ -387,6 +403,137 @@ def test_single_layer_model(self) -> None: assert result == {"keys_": "keys_%d", "vals_": "vals_%d"} +# --------------------------------------------------------------------------- +# Tests: OpenVINO NPU options and load configuration +# --------------------------------------------------------------------------- + + +class TestOpenVINOSessionOptions: + def test_minimal_npu_options_without_custom_config(self) -> None: + options = openvino_stage_session_options("test.context") + assert options["log_id"] == "test.context" + assert options["intra_op_num_threads"] == 2 + assert options["inter_op_num_threads"] == 1 + provider = options["provider_options"][0]["openvino"] + assert provider["device_type"] == "NPU" + assert json.loads(provider["load_config"]) == {"NPU": {"NPU_COMPILER_TYPE": "DRIVER"}} + assert set(provider) == {"device_type", "load_config"} + + def test_flat_config_preserves_custom_settings(self, openvino_config_file) -> None: + payload = { + "NPU": { + "NPU_COMPILER_TYPE": "custom-compiler", + "NPU_TURBO": "NO", + "NPU_QDQ_OPTIMIZATION": "YES", + } + } + path = openvino_config_file(payload) + for role in ("CTX", "ITER"): + assert json.loads(build_npu_load_config(path, model_role=role)) == payload + assert json.loads(path.read_text(encoding="utf-8")) == payload + + @pytest.mark.parametrize("role", ["CTX", "ITER"]) + def test_selects_role_without_forwarding_other_sections( + self, openvino_config_file, role + ) -> None: + payload = { + "CTX": {"NPU": {"NPU_TURBO": "YES"}}, + "ITER": {"NPU": {"NPU_TURBO": "NO"}}, + "HEAD": {"NPU": {"NPU_TURBO": "YES"}}, + } + path = openvino_config_file(payload) + config = json.loads(build_npu_load_config(path, model_role=role)) + assert set(config) == {"NPU"} + assert config["NPU"] == { + **payload[role]["NPU"], + "NPU_COMPILER_TYPE": "DRIVER", + } + + def test_equivalent_role_configs_serialize_identically(self, openvino_config_file) -> None: + properties = {"NPU_TURBO": "YES", "NPU_QDQ_OPTIMIZATION": "YES"} + path = openvino_config_file( + { + "CTX": {"NPU": properties}, + "ITER": {"NPU": dict(reversed(list(properties.items())))}, + } + ) + assert build_npu_load_config(path, model_role="CTX") == build_npu_load_config( + path, model_role="ITER" + ) + + def test_weights_default_is_absolute(self, tmp_path, monkeypatch) -> None: + monkeypatch.chdir(tmp_path) + config = json.loads(build_npu_load_config(weights_path="model weights")) + path = Path(config["NPU"]["WEIGHTS_PATH"]) + assert path.is_absolute() + assert path == tmp_path / "model weights" + + def test_configured_weights_override_default(self, tmp_path, openvino_config_file) -> None: + weights_path = tmp_path / "custom weights" + path = openvino_config_file({"NPU": {"WEIGHTS_PATH": str(weights_path)}}) + config = json.loads(build_npu_load_config(path, weights_path=tmp_path / "fallback")) + assert config["NPU"]["WEIGHTS_PATH"] == str(weights_path.resolve()) + + def test_relative_configured_weights_use_config_directory( + self, tmp_path, monkeypatch, openvino_config_file + ) -> None: + path = openvino_config_file({"NPU": {"WEIGHTS_PATH": "weights"}}) + monkeypatch.chdir(tmp_path.parent) + config = json.loads(build_npu_load_config(path)) + assert config["NPU"]["WEIGHTS_PATH"] == str((path.parent / "weights").resolve()) + + @pytest.mark.parametrize( + ("payload", "role", "error", "message"), + [ + ([], None, TypeError, "load config must be a JSON object"), + (None, None, TypeError, "load config must be a JSON object"), + ({"CTX": {}}, None, ValueError, "requires model_role"), + ({"CTX": {}}, "ITER", ValueError, "missing the ITER section"), + ({"CTX": []}, "CTX", TypeError, "CTX configuration must be a JSON object"), + ({"NPU": []}, None, TypeError, "NPU configuration must be a JSON object"), + ({"NPU": None}, None, TypeError, "NPU configuration must be a JSON object"), + ( + {"NPU": {"WEIGHTS_PATH": 42}}, + None, + ValueError, + "WEIGHTS_PATH must be a non-empty string", + ), + ( + {"NPU": {"WEIGHTS_PATH": " "}}, + None, + ValueError, + "WEIGHTS_PATH must be a non-empty string", + ), + ], + ) + def test_invalid_config_is_rejected(self, openvino_config_file, payload, role, error, message): + path = openvino_config_file(payload) + with pytest.raises(error, match=message): + build_npu_load_config(path, model_role=role) + + def test_unsupported_role_is_rejected(self) -> None: + with pytest.raises(ValueError, match="model_role must be 'CTX' or 'ITER'"): + build_npu_load_config(model_role="HEAD") + + def test_missing_file_is_not_silently_ignored(self, tmp_path) -> None: + with pytest.raises(FileNotFoundError): + build_npu_load_config(tmp_path / "missing.json") + + def test_invalid_json_reports_config_path(self, tmp_path) -> None: + path = tmp_path / "invalid.json" + path.write_text("{", encoding="utf-8") + with pytest.raises(ValueError, match="Invalid OpenVINO load config JSON") as exc_info: + build_npu_load_config(path) + assert str(path) in str(exc_info.value) + + def test_utf8_bom_is_accepted(self, tmp_path) -> None: + payload = {"NPU": {"NPU_TURBO": "YES"}} + path = tmp_path / "bom.json" + path.write_text(json.dumps(payload), encoding="utf-8-sig") + config = json.loads(build_npu_load_config(path)) + assert config["NPU"]["NPU_TURBO"] == payload["NPU"]["NPU_TURBO"] + + # --------------------------------------------------------------------------- # Tests: build_qwen3_transformer_only_stages # --------------------------------------------------------------------------- @@ -561,6 +708,64 @@ def test_vitisai_ep_injects_session_options(self) -> None: assert vitisai_opts["no_linear_slice"] == "1" assert itr_opts["log_id"] == "onnxruntime-genai.iterator" + @pytest.mark.parametrize("ep", ["openvino", "OpenVINOExecutionProvider", "OPENVINO"]) + def test_openvino_only_routes_transformer_stages(self, ep) -> None: + with self._patch_onnx(): + stages, _ = build_qwen3_transformer_only_stages( + "ctx.onnx", "iter.onnx", num_layers=4, ep=ep + ) + stage_map = {stage.name: stage for stage in stages} + assert stage_map["embeddings"].session_options is None + assert stage_map["lm_head"].session_options is None + for name in ("context", "iterator"): + options = stage_map[name].session_options + assert options["log_id"] == f"onnxruntime-genai.{name}" + assert options["provider_options"][0]["openvino"]["device_type"] == "NPU" + assert ( + stage_map["context"].session_options["provider_options"] + == stage_map["iterator"].session_options["provider_options"] + ) + + def test_openvino_role_config_reaches_serialized_pipeline(self, openvino_config_file) -> None: + payload = { + "CTX": {"NPU": {"NPU_TURBO": "YES"}}, + "ITER": {"NPU": {"NPU_TURBO": "NO"}}, + } + path = openvino_config_file(payload) + with self._patch_onnx(): + stages, decoder_io = build_qwen3_transformer_only_stages( + "ctx.onnx", "iter.onnx", num_layers=4, ep="openvino", openvino_config_path=path + ) + config = build_genai_config( + _mock_config(num_hidden_layers=4), + max_cache_len=256, + prefill_seq_len=64, + pipeline=stages, + decoder_io=decoder_io, + ) + pipeline = json.loads(json.dumps(config))["model"]["decoder"]["pipeline"] + stage_map = {name: stage for entry in pipeline for name, stage in entry.items()} + for name, role in (("context", "CTX"), ("iterator", "ITER")): + provider = stage_map[name]["session_options"]["provider_options"][0]["openvino"] + assert isinstance(provider["load_config"], str) + assert ( + json.loads(provider["load_config"])["NPU"]["NPU_TURBO"] + == payload[role]["NPU"]["NPU_TURBO"] + ) + assert "session_options" not in stage_map["embeddings"] + assert "session_options" not in stage_map["lm_head"] + + @pytest.mark.parametrize("ep", ["cpu", "qnn", "vitisai"]) + def test_openvino_options_are_rejected_for_other_eps(self, ep, tmp_path) -> None: + with pytest.raises(ValueError, match="OpenVINO configuration requires"): + build_qwen3_transformer_only_stages( + "ctx.onnx", + "iter.onnx", + num_layers=4, + ep=ep, + openvino_config_path=tmp_path / "unused.json", + ) + # --------------------------------------------------------------------------- # Tests: write_genai_bundle wrapper (ep routing + transformer_onnx_passes) @@ -618,3 +823,51 @@ def test_cpu_ep_forwards_no_session_options(self) -> None: kwargs = mock_write.call_args.kwargs assert kwargs["context_session_options"] is None assert kwargs["iterator_session_options"] is None + + def test_openvino_defaults_weights_to_bundle_directory(self, tmp_path) -> None: + output_dir = tmp_path / "bundle" + with self._patch_generic() as mock_write: + write_genai_bundle(output_dir, ep="openvino", **self._COMMON) + kwargs = mock_write.call_args.kwargs + for name in ("context", "iterator"): + options = kwargs[f"{name}_session_options"] + assert options["log_id"] == f"onnxruntime-genai.{name}" + provider = options["provider_options"][0]["openvino"] + assert provider["device_type"] == "NPU" + config = json.loads(provider["load_config"]) + assert config["NPU"]["WEIGHTS_PATH"] == str(output_dir.resolve()) + assert "openvino_config_path" not in kwargs + assert "openvino_weights_path" not in kwargs + + def test_openvino_forwards_custom_role_options(self, tmp_path, openvino_config_file) -> None: + payload = { + "CTX": {"NPU": {"NPU_TURBO": "YES"}}, + "ITER": {"NPU": {"NPU_TURBO": "NO"}}, + } + path = openvino_config_file(payload) + weights_dir = tmp_path / "shared weights" + with self._patch_generic() as mock_write: + write_genai_bundle( + tmp_path / "bundle", + ep="OpenVINOExecutionProvider", + openvino_config_path=path, + openvino_weights_path=weights_dir, + **self._COMMON, + ) + kwargs = mock_write.call_args.kwargs + for name, role in (("context", "CTX"), ("iterator", "ITER")): + provider = kwargs[f"{name}_session_options"]["provider_options"][0]["openvino"] + config = json.loads(provider["load_config"]) + assert config["NPU"]["NPU_TURBO"] == payload[role]["NPU"]["NPU_TURBO"] + assert config["NPU"]["WEIGHTS_PATH"] == str(weights_dir.resolve()) + + def test_openvino_config_errors_prevent_assembly(self, tmp_path, openvino_config_file) -> None: + path = openvino_config_file({"CTX": {"NPU": {}}}) + with ( + self._patch_generic() as mock_write, + pytest.raises(ValueError, match="missing the ITER section"), + ): + write_genai_bundle( + tmp_path / "bundle", ep="openvino", openvino_config_path=path, **self._COMMON + ) + mock_write.assert_not_called() diff --git a/tests/unit/models/winml/test_genai_bundle_registry.py b/tests/unit/models/winml/test_genai_bundle_registry.py index a26ba08a9..064cb0533 100644 --- a/tests/unit/models/winml/test_genai_bundle_registry.py +++ b/tests/unit/models/winml/test_genai_bundle_registry.py @@ -37,6 +37,14 @@ def test_resolve_qwen3_returns_recipe(): assert len(recipe.transformer_onnx_passes) >= 1 +def test_qwen3_openvino_target_is_npu_only(): + recipe = resolve_genai_bundle("qwen3") + assert recipe is not None + assert {target.device for target in recipe.supported_targets if target.ep == "openvino"} == { + "npu" + } + + def test_resolve_unregistered_returns_none(): assert resolve_genai_bundle("bert") is None