diff --git a/docs/README.md b/docs/README.md index d9d5e5eb6..55a61a99b 100644 --- a/docs/README.md +++ b/docs/README.md @@ -17,7 +17,7 @@ This folder hosts the source for the [winml-cli](https://github.com/microsoft/wi ``` docs/ ├── index.md ← landing page -├── getting-started/ ← 3 onboarding pages +├── getting-started/ ← 4 onboarding pages ├── concepts/ ← 12 conceptual pages in two sub-groups │ ├── how-it-works.md, graphs-and-ir.md, weight-and-activation.md, │ │ eps-and-devices.md, quantization.md (Fundamentals) @@ -64,7 +64,7 @@ uv run mkdocs build --strict - A new page added without an entry in `nav:` (gives a "not included in nav" warning) - A nav entry pointing at a file that doesn't exist -- A relative link like `[text](other-page.md)` whose target file is missing +- A relative link whose target file is missing - A markdown anchor like `[link](#section-heading)` that doesn't match any heading slug ## Publishing @@ -116,6 +116,7 @@ on: The following are present in `docs/` but **excluded from the published site** via the `exclude_docs:` block in `mkdocs.yml`. They are kept in-repo for contributors: - `docs/design/` — internal architecture decision records and design notes +- `docs/README.md` — contributor guidance for the documentation source - `docs/superpowers/` — specs, plans, and review notes accumulated during doc development - `docs/naming-convention.md` — internal naming conventions for code review - `docs/pytest-best-practices.md` — internal testing style guide diff --git a/docs/commands/build.md b/docs/commands/build.md index 9453f1f87..e3bbde914 100644 --- a/docs/commands/build.md +++ b/docs/commands/build.md @@ -27,7 +27,8 @@ $ winml build [options] | `--output-dir` | `-o` | path | `None` | Directory for all build artifacts. Mutually exclusive with `--use-cache`. | | `--use-cache/--no-use-cache` | | flag | `false` | Store artifacts in the winml-cli global cache (`~/.cache/winml/`). Mutually exclusive with `--output-dir`. | | `--rebuild/--no-rebuild` | | flag | `false` | Overwrite existing artifacts and re-run the full pipeline. | -| `--quant/--no-quant` | | flag | `true` | Run the quantization stage (use `--no-quant` to skip), overriding the config. | +| `--quant` | | flag | `true` | Run the quantization stage, overriding the config. | +| `--no-quant` | | flag | `false` | Skip the quantization stage, overriding the config. | | `--no-compile` / `--compile` | | flag | `None` | Override compilation. `--compile` forces enable (config must have a compile section). `--no-compile` forces skip. Default: inherit from config. | | `--optimize/--no-optimize` | | flag | `true` | Run the optimization stage (use `--no-optimize` to skip). | | `--ep` | | string | `None` | Target execution provider for the analyzer (e.g., `qnn`). Falls back to the compile config EP if not set. | diff --git a/docs/commands/optimize.md b/docs/commands/optimize.md index 950fa8ec3..81592d82b 100644 --- a/docs/commands/optimize.md +++ b/docs/commands/optimize.md @@ -19,7 +19,7 @@ $ winml optimize [options] | `--model` | `-m` | `PATH` | *(required unless listing)* | Input ONNX model file. Not required when `--list-capabilities` or `--list-rewrites` is used. | | `--output` | `-o` | `PATH` | `{input}_opt.onnx` | Output path for the optimized model. Defaults to the input filename with `_opt` inserted before the extension. | | `--config` | `-c` | `PATH` | *(none)* | YAML or JSON configuration file. Fields in the file override capability defaults; CLI flags override the file. | -| `--enable-ort-graph-optimization` / `--disable-ort-graph-optimization` | | flag | enabled | Run or skip ORTGraphPipe. Disabling it does not implicitly enable compatibility rewrites or disable other pipes. Config key: `ort-graph-optimization` (boolean). | +| `ort-graph-optimization` | | capability | enabled | Run or skip ORTGraphPipe. Disabling it does not implicitly enable compatibility rewrites or disable other pipes. Config key: `ort-graph-optimization` (boolean); its CLI flags are generated using the `--enable-` / `--disable-` pattern below. | | `--verbose` | `-v` | flag | off | Enable verbose output. | | `--list-capabilities` | `-l` | flag | off | Print all registered optimization capabilities grouped by category and exit. Add `--verbose` for descriptions and ORT names. | | `--list-rewrites` | | flag | off | Print all available pattern-rewrite families with their source-to-target mappings and exit. | diff --git a/docs/samples/qwen3-genai-bundle.md b/docs/samples/qwen3-genai-bundle.md index 16cb7680b..cbceb6514 100644 --- a/docs/samples/qwen3-genai-bundle.md +++ b/docs/samples/qwen3-genai-bundle.md @@ -7,8 +7,8 @@ runtime metadata and tokenizer that onnxruntime-genai loads together: | File | Role | Device | Precision | |------|------|--------|-----------| -| `ctx.onnx` | Transformer **prefill** graph (processes the prompt) | NPU (QNN or VitisAI) | `w8a16` | -| `iter.onnx` | Transformer **decode** graph (one token per step) | NPU (QNN or VitisAI) | `w8a16` | +| `ctx.onnx` | Transformer **prefill** graph (processes the prompt) | NPU (QNN, VitisAI, or OpenVINO) | `w8a16` | +| `iter.onnx` | Transformer **decode** graph (one token per step) | NPU (QNN, VitisAI, or OpenVINO) | `w8a16` | | `embeddings.onnx` | Token embedding lookup | CPU | `fp32` | | `lm_head.onnx` | Final vocab projection | CPU | `w4a32` | | `genai_config.json` + tokenizer | onnxruntime-genai runtime metadata | — | — | @@ -17,21 +17,25 @@ runtime metadata and tokenizer that onnxruntime-genai loads together: composite: prefill bakes in a context sequence length, decode is fixed to a single token. The embedding table and vocab projection stay on CPU. Splitting the model this way lets the compute-heavy transformer run on the selected NPU while -the memory-bound companions stay on CPU. +the memory-bound companions stay on CPU. On Intel, OpenVINO runs the `context` +and `iterator` stages on the NPU; the embedding lookup and `lm_head` projection +remain on CPU. ## Prerequisites - winml-cli installed and `winml` on your PATH. - A network connection to download Qwen3 weights from HuggingFace on first run. -- For NPU inference, a QNN (Qualcomm) or VitisAI (AMD) execution provider. - CPU inference does not require either NPU provider. +- For NPU inference, a compatible QNN (Qualcomm), VitisAI (AMD), or OpenVINO + (Intel) execution provider. Intel NPU runs also require a compatible Intel NPU + driver and OpenVINO NPU plugin, available through a compatible ONNX Runtime + OpenVINO EP installation. CPU inference does not require an NPU provider. ## Overall workflow ```mermaid graph LR A["winml build -m Qwen/Qwen3-0.6B --export-type optimized"] --> B[Genai bundle recipe] - B --> C[ctx.onnx / iter.onnx — NPU] + B --> C[ctx.onnx / iter.onnx — selected NPU] B --> D[embeddings.onnx — CPU] B --> E[lm_head.onnx — CPU] C --> F[genai_config.json + tokenizer] @@ -54,8 +58,9 @@ winml build -m Qwen/Qwen3-0.6B -o out/qwen3-bundle --export-type optimized This builds (or reuses from cache) all four components and assembles them, writing `out/qwen3-bundle/genai_config.json` alongside the ONNX graphs and tokenizer. `--output-dir` is required — the bundle is a directory — and `--use-cache` is not -supported for bundles. The recipe supports CPU, QNN/NPU, and VitisAI/NPU; -other resolved targets fail fast. Pin the provider to select a target explicitly: +supported for bundles. The recipe supports CPU, QNN/NPU, VitisAI/NPU, and +OpenVINO/NPU; other resolved targets fail fast. Pin the provider to select a +target explicitly: ```bash # Qualcomm Snapdragon NPU @@ -65,6 +70,10 @@ winml build -m Qwen/Qwen3-0.6B -o out/qwen3-bundle \ # AMD Ryzen AI NPU winml build -m Qwen/Qwen3-0.6B -o out/qwen3-bundle \ --export-type optimized --ep vitisai --device npu + +# Intel NPU (OpenVINO) +winml build -m Qwen/Qwen3-0.6B -o out/qwen3-bundle \ + --export-type optimized --ep openvino --device npu ``` The Qwen3 transformer's quantization scheme is fixed by its recipe (`w8a16`, the @@ -92,26 +101,7 @@ Force a clean rebuild of every component with `--rebuild`. `--ep`/`--device` that contradicts the recipe fails fast rather than silently reverting. -## Step 2: Tune context and prefill lengths (optional) - -`winml build` uses the recipe defaults: context length (static KV cache) `2048` -and prefill sequence length `64`. To change those, use the equivalent developer -script, which exposes the extra knobs and delegates to the same builder: - -```bash -uv run python scripts/qwen3.py export \ - --device npu \ - --output out/qwen3-bundle \ - --max-cache-len 4096 \ - --prefill-seq-len 128 -``` - -The script also accepts `--embeddings ` and `--lm-head ` to reuse -pre-built companions (skipping their builds), and `--force-rebuild` to rebuild -everything from scratch. The developer script's `--device npu` shortcut targets -QNN; use `winml build --ep vitisai --device npu` for an AMD bundle. - -## Step 3: Run the bundle (generate text) +## Step 2: Run the bundle (generate text) The assembled bundle runs through onnxruntime-genai. Benchmark prompt processing and token generation on the NPU with `winml perf`: @@ -124,6 +114,10 @@ winml perf -m out/qwen3-bundle --runtime ort-genai --device npu --compile \ # AMD Ryzen AI NPU winml perf -m out/qwen3-bundle --runtime ort-genai --device npu --ep vitisai --compile \ --compile-timeout 600 --max-new-tokens 20 --prompt "What is the capital of France?" + +# Intel NPU (OpenVINO) +winml perf -m out/qwen3-bundle --runtime ort-genai --device npu --ep openvino --compile \ + --compile-timeout 600 --max-new-tokens 20 --prompt "What is the capital of France?" ``` `winml perf` registers the selected WinML EP and runs the bundle's `context` and @@ -134,6 +128,9 @@ request/model TTFT, prefill throughput, steady-state decode throughput, full request latency, optional RAM/VRAM deltas, and a results JSON under `~/.cache/winml/perf/`. Exact weight-upload telemetry is currently `null` because onnxruntime-genai does not expose it; the estimate is labeled in JSON. +Intel NPU availability and performance depend on the installed driver, OpenVINO +plugin, hardware, and model compatibility; this Qwen3 recipe does not imply that +all Intel NPU models are supported. !!! tip "One command from a model id (auto-build)" `winml perf --runtime ort-genai` also accepts a HuggingFace **model id** directly. @@ -148,10 +145,38 @@ onnxruntime-genai does not expose it; the estimate is labeled in JSON. Without a device or EP override, the model-ID shortcut targets QNN/NPU. An explicit `--device` or `--ep` also selects the transformer build target, - not just the inference target. For example, a CPU run does not require QNN: + not just the inference target. To auto-build for an Intel NPU, explicitly + select OpenVINO: ```bash - winml perf -m Qwen/Qwen3-0.6B --runtime ort-genai --device cpu --no-compile \ + winml perf -m Qwen/Qwen3-0.6B --runtime ort-genai --ep openvino --device npu \ + --compile --compile-timeout 600 --max-new-tokens 20 \ + --prompt "What is the capital of France?" + ``` + + The optional `--openvino-config` flag embeds a vendor-specific OpenVINO + load-config JSON while auto-building a model ID; it requires `--ep openvino`. + The repository-root `npu_config_npuw.json` is an optional example with + vendor-specific NPU properties such as `NPU_TURBO`, + `NPU_QDQ_OPTIMIZATION`, `NPU_COMPILER_TYPE`, and `CACHE_MODE`; supported + values can depend on the driver and plugin version. Run the command from the + repository root to use that relative path: + + ```bash + winml perf -m Qwen/Qwen3-0.6B --runtime ort-genai --ep openvino --device npu \ + --openvino-config npu_config_npuw.json --compile --compile-timeout 600 \ + --max-new-tokens 20 --prompt "What is the capital of France?" + ``` + + These settings are optional; omit `--openvino-config` to use the driver's + default compiler configuration. The file is not required at inference time: + its values are embedded in the auto-built bundle. `--openvino-config` is for + model-ID auto-build, not for a prebuilt bundle; the prebuilt-bundle command + above uses the options already stored in that bundle. For example, a CPU run + does not require QNN: + + ```bash + winml perf -m Qwen/Qwen3-0.6B --runtime ort-genai --device cpu \ --warmup 2 --iterations 10 --max-new-tokens 20 \ --prompt "What is the capital of France?" ``` @@ -163,22 +188,6 @@ onnxruntime-genai does not expose it; the estimate is labeled in JSON. retain their recipe precisions. `-o/--output` stays the results-JSON path, and `--rebuild` forces a fresh bundle for the selected target. -!!! warning "`--compile` is required on the NPU" - The genai NPU path needs `--compile` (EPContext pre-compilation). The context - and iterator stages are compiled together when they use the same provider - options, allowing both EPContext graphs to reference one shared weight - `.bin`. Without `--compile`, onnxruntime-genai compiles the NPU context - in-memory at model-creation time, which can fault before the first token. - Use `--compile-timeout ` to bound compilation before falling back - to the original ONNX. - -!!! note "Known caveat: non-zero exit on teardown" - On Windows ARM64, after generation completes and the results JSON is saved, the - process may exit with a native `0xC0000374` (heap corruption) during - onnxruntime-genai / QNN-EP **teardown**. This fires after all work is done — the - generated tokens and the saved perf metrics are unaffected — and originates in the - native runtime below winml-cli, not in the bundle or the build. - ## How it maps to the composite system The bundle reuses winml-cli's existing composite-model machinery — it does not add diff --git a/mkdocs.yml b/mkdocs.yml index 262d1c577..32c0b5051 100644 --- a/mkdocs.yml +++ b/mkdocs.yml @@ -8,6 +8,7 @@ edit_uri: edit/main/docs/ docs_dir: docs exclude_docs: | + /README.md /design/ /naming-convention.md /pytest-best-practices.md @@ -88,6 +89,7 @@ nav: - Installation: getting-started/installation.md - Quickstart: getting-started/quickstart.md - UI Quickstart: getting-started/ui-quickstart.md + - CGC Onboarding: getting-started/cgc-onboarding.md - Use with AI Agent: - Overview: getting-started/agent-skill/index.md - Use WinML CLI: getting-started/agent-skill/use-winml-cli.md @@ -130,6 +132,8 @@ nav: - BERT — Config + Build + Perf: samples/bert-config-build.md - CLIP — Composite Models: samples/clip-composite.md - Qwen3 — Genai Bundle: samples/qwen3-genai-bundle.md + - ResNet-50 — PyTorch to CGIR: samples/resnet50-cgir.md + - YOLO11 — ONNX to CGIR: samples/yolo11-cgir.md - Tutorials: - Overview: tutorials/index.md - Hugging Face Model to NPU: tutorials/npu-convnext.md diff --git a/src/winml/modelkit/export/value_range.py b/src/winml/modelkit/export/value_range.py index b2d990596..fb8e67b52 100644 --- a/src/winml/modelkit/export/value_range.py +++ b/src/winml/modelkit/export/value_range.py @@ -125,7 +125,7 @@ def intercept_value_ranges() -> Iterator[dict[str, dict[str, Any]]]: 'token_type_ids': {'min': 0, 'max': 2, 'method': 'random_int_tensor'}} """ captured: dict[str, dict] = {} - originals: dict = {} + originals: dict[str | tuple[type[Any], str], Any] = {} # Patch static tensor gen methods on the base class for method_name in _TENSOR_GEN_METHODS: @@ -138,14 +138,14 @@ def intercept_value_ranges() -> Iterator[dict[str, dict[str, Any]]]: ) # Patch generate() on all subclasses that override it - patched_classes = [] + patched_classes: list[type[Any]] = [] - def _patch_subclasses(base: type) -> None: - for cls in base.__subclasses__(): + def _patch_subclasses(base: type[Any]) -> None: + subclasses: list[type[Any]] = base.__subclasses__() + for cls in subclasses: if "generate" in cls.__dict__: originals[(cls, "generate")] = cls.__dict__["generate"] - # Monkey-patch optimum's untyped generator hierarchy. - cls.generate = _make_generate_wrapper(cls.__dict__["generate"]) # type: ignore[attr-defined] + cls.generate = _make_generate_wrapper(cls.__dict__["generate"]) patched_classes.append(cls) _patch_subclasses(cls) @@ -162,4 +162,4 @@ def _patch_subclasses(base: type) -> None: staticmethod(originals[method_name]), ) for cls in patched_classes: - cls.generate = originals[(cls, "generate")] # type: ignore[attr-defined] + cls.generate = originals[cls, "generate"]