Skip to content
Draft
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
12 changes: 12 additions & 0 deletions npu_config_npuw.json
Original file line number Diff line number Diff line change
@@ -0,0 +1,12 @@
{
"NPU": {
"NPU_TURBO": "YES",
"NPU_QDQ_OPTIMIZATION": "YES",
"NPU_COMPILER_TYPE": "DRIVER",
"NPU_USE_NPUW": "YES",
"NPUW_DEVICES": "NPU",
"NPUW_PARALLEL_COMPILE": "YES",
"NPUW_GQA": "YES",
"CACHE_MODE": "OPTIMIZE_SPEED"
}
}
19 changes: 19 additions & 0 deletions src/winml/modelkit/commands/perf.py
Original file line number Diff line number Diff line change
Expand Up @@ -2504,6 +2504,14 @@ def _autobuild_genai_bundle(
build_ep, build_device = short_ep_name(target.ep), target.device
# Do not reuse a bundle exported for a different execution provider.
bundle_dir = bundle_dir.with_name(f"genai-bundle-{build_ep}-{build_device}")
openvino_config: Path | None = p.get("openvino_config")
if openvino_config is not None:
if build_ep != "openvino":
raise click.UsageError("--openvino-config requires --ep openvino.")
import hashlib

config_digest = hashlib.sha256(openvino_config.read_bytes()).hexdigest()[:12]
bundle_dir = bundle_dir.with_name(f"{bundle_dir.name}-config-{config_digest}")
build_cache_dir = cache_dir
# --rebuild overwrites the cached bundle; a plain run reuses it. Checked
# before any model resolution so a cache hit never touches the network.
Expand Down Expand Up @@ -2564,6 +2572,9 @@ def _autobuild_genai_bundle(
force_rebuild=force_rebuild,
cache_dir=build_cache_dir,
emit=lambda msg: console.print(msg, markup=False),
assemble_options=(
{"openvino_config_path": openvino_config} if openvino_config is not None else None
),
)
return bundle_dir, True

Expand Down Expand Up @@ -2784,6 +2795,13 @@ def _validate_duration(
help="[ort-genai] Max seconds to compile each EPContext stage before falling back "
"to the original ONNX (requires --compile).",
)
@click.option(
"--openvino-config",
type=click.Path(exists=True, dir_okay=False, path_type=Path),
default=None,
help="[ort-genai] OpenVINO load-config JSON embedded while auto-building a model ID. "
"Requires --ep openvino.",
)
@click.option(
"--task",
type=str,
Expand Down Expand Up @@ -2964,6 +2982,7 @@ def perf(
apply_template: bool,
max_new_tokens: int,
compile_timeout: int,
openvino_config: Path | None,
task: str | None,
submodel: str | None,
iterations: int,
Expand Down
4 changes: 4 additions & 0 deletions src/winml/modelkit/models/hf/qwen3/__init__.py
Original file line number Diff line number Diff line change
Expand Up @@ -17,7 +17,9 @@
PipelineStage,
build_decoder_pipeline_stages,
build_genai_config,
build_npu_load_config,
build_qwen3_transformer_only_stages,
openvino_stage_session_options,
strip_gqa_default_attrs,
write_genai_bundle,
)
Expand All @@ -28,7 +30,9 @@
"PipelineStage",
"build_decoder_pipeline_stages",
"build_genai_config",
"build_npu_load_config",
"build_qwen3_transformer_only_stages",
"openvino_stage_session_options",
"strip_gqa_default_attrs",
"write_genai_bundle",
]
183 changes: 174 additions & 9 deletions src/winml/modelkit/models/hf/qwen3/genai.py
Original file line number Diff line number Diff line change
Expand Up @@ -11,18 +11,21 @@

This module adds the **Qwen3-specific** layer on top: the Qwen3 transformer
stages run on an NPU backend, so this is where the per-EP ``session_options``
are constructed. Two NPU execution providers are supported for the
are constructed. Three NPU execution providers are supported for the
transformer (context/iterator) stages:

* **QNN HTP** — Qualcomm Snapdragon NPU (``ep="qnn"``).
* **VitisAI** — AMD Ryzen AI NPU (``ep="vitisai"``).
* **OpenVINO** — Intel NPU via the plugin EP (``ep="openvino"``).

Keeping the EP-specific logic here lets the generic utilities stay universal
while the Qwen3 bundle emits the correct per-EP ``genai_config.json``.
"""

from __future__ import annotations

import json
from pathlib import Path
from typing import TYPE_CHECKING

from ....onnx import strip_node_attrs
Expand Down Expand Up @@ -51,15 +54,16 @@

if TYPE_CHECKING:
from collections.abc import Callable, Sequence
from pathlib import Path

import onnx


# ---------------------------------------------------------------------------
# Qwen3-specific NPU execution-provider routing (QNN / VitisAI)
# Qwen3-specific NPU execution-provider routing (QNN / VitisAI / OpenVINO)
# ---------------------------------------------------------------------------

_OPENVINO_CONFIG_ROLE_KEYS = frozenset({"CTX", "ITER", "HEAD"})


def qnn_stage_session_options(log_id: str, soc_model: str = "60") -> dict:
"""Return the ``session_options`` block that routes a stage to QNN HTP.
Expand Down Expand Up @@ -124,19 +128,134 @@ def vitisai_stage_session_options(log_id: str) -> dict:
}


def _stage_session_options(ep: str, soc_model: str) -> tuple[dict | None, dict | None]:
def build_npu_load_config(
custom_config_path: str | Path | None = None,
*,
model_role: str | None = None,
weights_path: str | Path | None = None,
) -> str:
"""Build the JSON-string ``load_config`` for the OpenVINO NPU plugin.

Accepts a flat configuration (e.g. ``{"NPU": {...}}``) or a configuration
with ``CTX``/``ITER`` sections. A role-based file must contain the requested
role; missing sections are errors rather than silently using defaults.
``HEAD`` is recognized only to detect role-based files; the LM head stays
on CPU. No legacy provider setup or tuning flags are injected.

Args:
custom_config_path: Optional UTF-8 JSON file. Without one, only the
driver compiler default and an explicitly supplied weights path
are included. Custom values take precedence over defaults.
model_role: ``"CTX"`` or ``"ITER"``; required for role-based files.
weights_path: Default external-weights directory. A ``WEIGHTS_PATH``
in the file takes precedence; relative file values resolve against
that file's directory. Emitted paths are absolute so loading a
derived bundle does not change their meaning.

Returns:
Serialized OpenVINO load configuration, not a filename.
"""
if model_role not in (None, "CTX", "ITER"):
raise ValueError("OpenVINO model_role must be 'CTX' or 'ITER'")

config: dict = {}
config_path = (
Path(custom_config_path).expanduser().resolve() if custom_config_path is not None else None
)
if config_path is not None:
try:
config = json.loads(config_path.read_text(encoding="utf-8-sig"))
except json.JSONDecodeError as exc:
raise ValueError(f"Invalid OpenVINO load config JSON in {config_path}: {exc}") from exc
if not isinstance(config, dict):
raise TypeError("OpenVINO load config must be a JSON object")
if _OPENVINO_CONFIG_ROLE_KEYS.intersection(config):
if model_role is None:
raise ValueError("A role-based OpenVINO load config requires model_role")
if model_role not in config:
raise ValueError(f"OpenVINO load config is missing the {model_role} section")
config = config[model_role]
if not isinstance(config, dict):
raise TypeError(f"OpenVINO {model_role} configuration must be a JSON object")

npu_config = config.setdefault("NPU", {})
if not isinstance(npu_config, dict):
raise TypeError("OpenVINO NPU configuration must be a JSON object")
npu_config.setdefault("NPU_COMPILER_TYPE", "DRIVER")

if "WEIGHTS_PATH" in npu_config:
configured_weights = npu_config["WEIGHTS_PATH"]
if not isinstance(configured_weights, str) or not configured_weights.strip():
raise ValueError("OpenVINO WEIGHTS_PATH must be a non-empty string")
resolved_weights = Path(configured_weights).expanduser()
if not resolved_weights.is_absolute() and config_path is not None:
resolved_weights = config_path.parent / resolved_weights
npu_config["WEIGHTS_PATH"] = str(resolved_weights.resolve())
elif weights_path is not None:
npu_config["WEIGHTS_PATH"] = str(Path(weights_path).expanduser().resolve())

# Stable serialization keeps equivalent CTX/ITER options shareable even
# when their keys occur in a different order in the user's JSON file.
return json.dumps(config, sort_keys=True)


def openvino_stage_session_options(
log_id: str,
*,
custom_config_path: str | Path | None = None,
model_role: str | None = None,
weights_path: str | Path | None = None,
) -> dict:
"""Return session options routing a transformer stage to the Intel NPU.

``load_config`` is a JSON string containing OpenVINO properties, not ORT
session entries. Plugin registration, ABI device binding, and EPContext
compilation remain owned by the existing session/compiler infrastructure.
See :func:`build_npu_load_config` for the optional configuration arguments.
"""
return {
"log_id": log_id,
"provider_options": [
{
"openvino": {
"device_type": "NPU",
"load_config": build_npu_load_config(
custom_config_path,
model_role=model_role,
weights_path=weights_path,
),
}
}
],
"intra_op_num_threads": 2,
"inter_op_num_threads": 1,
}


def _stage_session_options(
ep: str,
soc_model: str,
*,
openvino_config_path: str | Path | None = None,
openvino_weights_path: str | Path | None = None,
) -> tuple[dict | None, dict | None]:
"""Return ``(context, iterator)`` session_options for the given EP.

Routes the Qwen3 transformer (context/iterator) stages to an NPU backend:

* ``ep="qnn"`` -> Qualcomm QNN HTP (``soc_model`` selects the Snapdragon SoC).
* ``ep="vitisai"`` -> AMD Ryzen AI NPU.
* ``ep="openvino"`` -> Intel NPU, with optional per-role load configuration.

Any other value (e.g. ``"cpu"``) leaves the stages on the default CPU
provider. Short aliases and full ``*ExecutionProvider`` names are both
accepted (normalized via :func:`normalize_ep_name`).
"""
canonical = normalize_ep_name(ep)
if canonical != "OpenVINOExecutionProvider" and (
openvino_config_path is not None or openvino_weights_path is not None
):
raise ValueError("OpenVINO configuration requires ep='openvino'")
if canonical == "QNNExecutionProvider":
return (
qnn_stage_session_options("onnxruntime-genai.context", soc_model=soc_model),
Expand All @@ -147,6 +266,21 @@ def _stage_session_options(ep: str, soc_model: str) -> tuple[dict | None, dict |
vitisai_stage_session_options("onnxruntime-genai.context"),
vitisai_stage_session_options("onnxruntime-genai.iterator"),
)
if canonical == "OpenVINOExecutionProvider":
return (
openvino_stage_session_options(
"onnxruntime-genai.context",
custom_config_path=openvino_config_path,
model_role="CTX",
weights_path=openvino_weights_path,
),
openvino_stage_session_options(
"onnxruntime-genai.iterator",
custom_config_path=openvino_config_path,
model_role="ITER",
weights_path=openvino_weights_path,
),
)
return None, None


Expand Down Expand Up @@ -190,6 +324,8 @@ def build_qwen3_transformer_only_stages(
lm_head_filename: str = DEFAULT_LM_HEAD_FILENAME,
ep: str = "cpu",
soc_model: str = "60",
openvino_config_path: str | Path | None = None,
openvino_weights_path: str | Path | None = None,
) -> tuple[list[PipelineStage], DecoderIOMapping]:
"""Build the Qwen3 4-stage pipeline, routing ctx/iter to the NPU per ``ep``.

Expand All @@ -207,19 +343,28 @@ def build_qwen3_transformer_only_stages(
embeddings_filename: Bundle filename for the embeddings model.
lm_head_filename: Bundle filename for the lm_head model.
ep: NPU execution provider for the ``context``/``iterator`` stages —
``"qnn"`` (Qualcomm) or ``"vitisai"`` (AMD) injects that EP's
``session_options`` so those stages run on the NPU while
``"qnn"`` (Qualcomm), ``"vitisai"`` (AMD), or ``"openvino"`` (Intel)
injects that EP's ``session_options`` so those stages run on the NPU while
``embeddings`` and ``lm_head`` stay on CPU. ``"cpu"`` (default)
omits them.
soc_model: Snapdragon SoC model number forwarded to the QNN backend when
``ep="qnn"``. Default ``"60"`` targets Snapdragon 8 Gen 3. Ignored
for non-QNN EPs.
openvino_config_path: Optional OpenVINO JSON file shared by both stages
or containing separate ``CTX``/``ITER`` sections. Only for OpenVINO.
openvino_weights_path: Optional default external-weights directory for
OpenVINO. Explicit ``WEIGHTS_PATH`` values in the JSON take precedence.

Returns:
``(stages, decoder_io)`` — see
:func:`~winml.modelkit.utils.genai.build_decoder_pipeline_stages`.
"""
ctx_opts, iter_opts = _stage_session_options(ep, soc_model)
ctx_opts, iter_opts = _stage_session_options(
ep,
soc_model,
openvino_config_path=openvino_config_path,
openvino_weights_path=openvino_weights_path,
)
return build_decoder_pipeline_stages(
context_onnx,
iterator_onnx,
Expand Down Expand Up @@ -250,6 +395,8 @@ def write_genai_bundle(
ep: str = "cpu",
soc_model: str = "60",
transformer_onnx_passes: Sequence[Callable[[onnx.ModelProto], onnx.ModelProto]] | None = None,
openvino_config_path: str | Path | None = None,
openvino_weights_path: str | Path | None = None,
) -> Path:
"""Assemble a Qwen3 genai bundle, routing ctx/iter to the NPU per ``ep``.

Expand All @@ -260,19 +407,34 @@ def write_genai_bundle(

Args:
ep: NPU execution provider routing the transformer (context/iterator)
stages — ``"qnn"`` (Qualcomm HTP) or ``"vitisai"`` (AMD Ryzen AI);
stages — ``"qnn"`` (Qualcomm HTP), ``"vitisai"`` (AMD Ryzen AI),
or ``"openvino"`` (Intel NPU);
``"cpu"`` (default) keeps every stage on CPU.
soc_model: Snapdragon SoC model passed to the QNN backend when
``ep="qnn"``. Default ``"60"`` = Snapdragon 8 Gen 3 / X Elite.
Ignored for non-QNN EPs.
transformer_onnx_passes: Optional ONNX graph transforms applied to the
copied context/iterator models before ``genai_config.json`` is
written. Forwarded verbatim to the generic assembler.
openvino_config_path: Optional flat or per-role OpenVINO JSON file.
Its contents are embedded in stage provider options; the original
configuration file is not required at runtime. Different role
options prevent grouped compilation in the current runtime.
openvino_weights_path: Default external-weights directory for OpenVINO;
defaults to the output bundle directory. Explicit ``WEIGHTS_PATH``
values in the JSON take precedence. Other EPs reject these options.

Returns:
Path to the written ``genai_config.json``.
"""
ctx_opts, iter_opts = _stage_session_options(ep, soc_model)
if normalize_ep_name(ep) == "OpenVINOExecutionProvider" and openvino_weights_path is None:
openvino_weights_path = output_dir
ctx_opts, iter_opts = _stage_session_options(
ep,
soc_model,
openvino_config_path=openvino_config_path,
openvino_weights_path=openvino_weights_path,
)
return _write_genai_bundle(
output_dir,
context_onnx=context_onnx,
Expand Down Expand Up @@ -302,7 +464,9 @@ def write_genai_bundle(
"PipelineStage",
"build_decoder_pipeline_stages",
"build_genai_config",
"build_npu_load_config",
"build_qwen3_transformer_only_stages",
"openvino_stage_session_options",
"qnn_stage_session_options",
"strip_gqa_default_attrs",
"vitisai_stage_session_options",
Expand Down Expand Up @@ -347,6 +511,7 @@ def write_genai_bundle(
supported_targets=(
GenaiTarget(ep="qnn", device="npu"), # Qualcomm Snapdragon NPU
GenaiTarget(ep="vitisai", device="npu"), # AMD Ryzen AI NPU
GenaiTarget(ep="openvino", device="npu"), # Intel NPU
GenaiTarget(ep="cpu", device="cpu"),
),
transformer_onnx_passes=(strip_gqa_default_attrs,),
Expand Down
Loading
Loading