Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
12 changes: 12 additions & 0 deletions docs/commands/build.md
Original file line number Diff line number Diff line change
Expand Up @@ -24,6 +24,8 @@ $ winml build [options]
| `--model` | `-m` | string | `None` | Hugging Face model ID or path to an existing `.onnx` file. |
| `--backend` | | choice | `None` | Backend for auto-generated config. `cgc` selects CGC preparation and CGIR conversion; it cannot be combined with `--ep`. |
| `--export-type` | | choice | `generic` | Output selector: `generic` builds the stock single/composite ONNX model; `optimized` builds the family's registered runtime-optimized recipe (today the onnxruntime-genai CPU/NPU bundle) for the **resolved** `--ep`/`--device`. `optimized` fails fast if the family has no recipe or the resolved target is not one the recipe supports. |
| `--max-cache-len` | | integer | recipe default | Static KV cache (context) length for optimized GenAI bundles. Must be at least 1. |
| `--prefill-seq-len` | | integer | recipe default | Prefill sequence length for optimized GenAI bundles. Must be at least 1. |
| `--output-dir` | `-o` | path | `None` | Directory for all build artifacts. Mutually exclusive with `--use-cache`. |
| `--use-cache/--no-use-cache` | | flag | `false` | Store artifacts in the winml-cli global cache (`~/.cache/winml/`). Mutually exclusive with `--output-dir`. |
| `--rebuild/--no-rebuild` | | flag | `false` | Overwrite existing artifacts and re-run the full pipeline. |
Expand Down Expand Up @@ -110,6 +112,16 @@ complete [onnxruntime-genai](https://github.com/microsoft/onnxruntime-genai)
winml build -m Qwen/Qwen3-0.6B -o out/qwen3-bundle --export-type optimized
```

The recipe supplies the context and prefill lengths when no overrides are
given. To customize them while keeping the same one-command workflow:

```bash
winml build -m Qwen/Qwen3-0.6B -o out/qwen3-bundle \
--export-type optimized \
--max-cache-len 4096 \
--prefill-seq-len 128
```

`--export-type optimized` resolves `--ep`/`--device` the same way a generic
build does — an explicit value is honored, otherwise the host is probed — and
then builds the recipe for that resolved target. The Qwen3 recipe supports
Expand Down
10 changes: 5 additions & 5 deletions docs/samples/qwen3-genai-bundle.md
Original file line number Diff line number Diff line change
Expand Up @@ -95,18 +95,18 @@ Force a clean rebuild of every component with `--rebuild`.
## Step 2: Tune context and prefill lengths (optional)

`winml build` uses the recipe defaults: context length (static KV cache) `2048`
and prefill sequence length `64`. To change those, use the equivalent developer
script, which exposes the extra knobs and delegates to the same builder:
and prefill sequence length `64`. Override either value directly when needed:

```bash
uv run python scripts/qwen3.py export \
winml build -m Qwen/Qwen3-0.6B -o out/qwen3-bundle \
--export-type optimized \
--ep qnn \
--device npu \
--output out/qwen3-bundle \
--max-cache-len 4096 \
--prefill-seq-len 128
```

The script also accepts `--embeddings <onnx>` and `--lm-head <onnx>` to reuse
The developer script also accepts `--embeddings <onnx>` and `--lm-head <onnx>` to reuse
pre-built companions (skipping their builds), and `--force-rebuild` to rebuild
everything from scratch. The developer script's `--device npu` shortcut targets
QNN; use `winml build --ep vitisai --device npu` for an AMD bundle.
Expand Down
36 changes: 34 additions & 2 deletions src/winml/modelkit/commands/build.py
Original file line number Diff line number Diff line change
Expand Up @@ -613,6 +613,8 @@ def _maybe_build_genai_bundle(
device: str,
ep: EPNameOrAlias | None,
precision: str | None,
max_cache_len: int | None,
prefill_seq_len: int | None,
rebuild: bool,
submodel: str | None,
) -> bool:
Expand Down Expand Up @@ -773,6 +775,8 @@ def _maybe_build_genai_bundle(
ep=bundle_ep,
device=bundle_device,
precision=override_precision,
max_cache_len=max_cache_len,
prefill_seq_len=prefill_seq_len,
force_rebuild=rebuild,
)
console.print(
Expand Down Expand Up @@ -856,6 +860,24 @@ def _maybe_build_genai_bundle(
"an explicit --ep qnn on an NPU target still routes a registered family to its "
"optimized bundle (backward-compatible shortcut).",
)
@click.option(
"--max-cache-len",
type=click.IntRange(min=1),
default=None,
help=(
"Static KV cache (context) length for optimized GenAI bundles. "
"Defaults to the registered recipe's value."
),
)
@click.option(
"--prefill-seq-len",
type=click.IntRange(min=1),
default=None,
help=(
"Prefill sequence length for optimized GenAI bundles. "
"Defaults to the registered recipe's value."
),
)
@cli_utils.shape_config_option(
help_text="JSON with shape overrides for auto-generated HuggingFace export configs.",
)
Expand Down Expand Up @@ -900,6 +922,8 @@ def build(
device: str,
precision: str,
export_type: str,
max_cache_len: int | None,
prefill_seq_len: int | None,
shape_config: Path | None,
input_specs: Path | None,
export_config: Path | None,
Expand Down Expand Up @@ -1229,7 +1253,7 @@ def _patch_device(cfg: WinMLBuildConfig) -> None:
# model-specific value lives in the recipe. ``--export-type generic`` (or
# any other device/ep combination without the flag) keeps the stock
# single/composite build.
if _maybe_build_genai_bundle(
built_genai_bundle = _maybe_build_genai_bundle(
ctx,
export_type=export_type,
model=model,
Expand All @@ -1241,10 +1265,18 @@ def _patch_device(cfg: WinMLBuildConfig) -> None:
device=runtime_device,
ep=runtime_ep_value,
precision=precision,
max_cache_len=max_cache_len,
prefill_seq_len=prefill_seq_len,
rebuild=rebuild,
submodel=submodel,
):
)
if built_genai_bundle:
return
if max_cache_len is not None or prefill_seq_len is not None:
raise click.UsageError(
"--max-cache-len and --prefill-seq-len are only supported for an "
"optimized GenAI bundle build."
)

if isinstance(config_or_configs, list):
# ---- MODULE MODE: array config, one build per submodule ----
Expand Down
2 changes: 2 additions & 0 deletions tests/unit/commands/test_build.py
Original file line number Diff line number Diff line change
Expand Up @@ -258,6 +258,8 @@ def test_help_shows_all_options(self, runner: CliRunner) -> None:
assert "--verbose" in result.output
assert "--no-analyze" in result.output
assert "--max-optim-iterations" in result.output
assert "--max-cache-len" in result.output
assert "--prefill-seq-len" in result.output
assert "--shape-config" in result.output
assert "--input-specs" in result.output
assert "--export-config" in result.output
Expand Down
76 changes: 76 additions & 0 deletions tests/unit/commands/test_build_genai_bundle.py
Original file line number Diff line number Diff line change
Expand Up @@ -167,6 +167,8 @@ def test_registered_family_npu_qnn_routes_to_bundle(tmp_path: Path):
assert kwargs["device"] == "npu"
assert kwargs["force_rebuild"] is False
assert kwargs["precision"] is None
assert kwargs["max_cache_len"] is None
assert kwargs["prefill_seq_len"] is None
assert "emit" not in kwargs


Expand Down Expand Up @@ -353,6 +355,74 @@ def test_precision_override_forwarded(tmp_path: Path):
assert recorded["kwargs"]["precision"] == "w8a16"


def test_context_and_prefill_lengths_forwarded(tmp_path: Path):
recorded: dict = {}

with (
patch(_GENERATE_TARGET, return_value=_fake_config("qwen3")),
patch(_BUNDLE_TARGET, side_effect=_record_bundle(recorded)),
patch(_RUN_SINGLE_TARGET),
patch(_COMPOSITE_TARGET, return_value=None),
):
result = _invoke(
[
"-m",
"Qwen/Qwen3-0.6B",
"-o",
str(tmp_path / "bundle"),
"--export-type",
"optimized",
"--ep",
"qnn",
"--device",
"npu",
"--max-cache-len",
"4096",
"--prefill-seq-len",
"128",
]
)

assert result.exit_code == 0, result.output
assert recorded["kwargs"]["max_cache_len"] == 4096
assert recorded["kwargs"]["prefill_seq_len"] == 128


@pytest.mark.parametrize("option", ["--max-cache-len", "--prefill-seq-len"])
@pytest.mark.parametrize("value", ["0", "-1"])
def test_context_and_prefill_lengths_must_be_positive(option: str, value: str):
result = _invoke(["-m", "Qwen/Qwen3-0.6B", "-o", "out", option, value])

assert result.exit_code != 0
assert "x>=1" in result.output


def test_context_and_prefill_lengths_rejected_for_generic_build(tmp_path: Path):
with (
patch(_GENERATE_TARGET, return_value=_fake_config("qwen3")),
patch(_BUNDLE_TARGET) as bundle,
patch(_RUN_SINGLE_TARGET) as run_single,
patch(_COMPOSITE_TARGET, return_value=None),
):
result = _invoke(
[
"-m",
"Qwen/Qwen3-0.6B",
"-o",
str(tmp_path / "generic"),
"--export-type",
"generic",
"--max-cache-len",
"4096",
]
)

assert result.exit_code != 0
assert "only supported for an optimized GenAI bundle build" in result.output
bundle.assert_not_called()
run_single.assert_not_called()


def test_registered_family_auto_qnn_routes_to_bundle(tmp_path: Path):
"""``--device auto`` resolving to the NPU, with explicit ``--ep qnn``, routes."""
out = tmp_path / "bundle"
Expand Down Expand Up @@ -768,6 +838,8 @@ def test_optimized_rejects_onnx_input():
device="auto",
ep=None,
precision=None,
max_cache_len=None,
prefill_seq_len=None,
rebuild=False,
submodel=None,
)
Expand All @@ -789,6 +861,8 @@ def test_optimized_rejects_module_mode():
device="auto",
ep=None,
precision=None,
max_cache_len=None,
prefill_seq_len=None,
rebuild=False,
submodel=None,
)
Expand All @@ -810,6 +884,8 @@ def test_optimized_requires_model():
device="auto",
ep=None,
precision=None,
max_cache_len=None,
prefill_seq_len=None,
rebuild=False,
submodel=None,
)
Expand Down
Loading