diff --git a/docs/commands/build.md b/docs/commands/build.md index 9453f1f87..a2ff4601d 100644 --- a/docs/commands/build.md +++ b/docs/commands/build.md @@ -24,6 +24,8 @@ $ winml build [options] | `--model` | `-m` | string | `None` | Hugging Face model ID or path to an existing `.onnx` file. | | `--backend` | | choice | `None` | Backend for auto-generated config. `cgc` selects CGC preparation and CGIR conversion; it cannot be combined with `--ep`. | | `--export-type` | | choice | `generic` | Output selector: `generic` builds the stock single/composite ONNX model; `optimized` builds the family's registered runtime-optimized recipe (today the onnxruntime-genai CPU/NPU bundle) for the **resolved** `--ep`/`--device`. `optimized` fails fast if the family has no recipe or the resolved target is not one the recipe supports. | +| `--max-cache-len` | | integer | recipe default | Static KV cache (context) length for optimized GenAI bundles. Must be at least 1. | +| `--prefill-seq-len` | | integer | recipe default | Prefill sequence length for optimized GenAI bundles. Must be at least 1. | | `--output-dir` | `-o` | path | `None` | Directory for all build artifacts. Mutually exclusive with `--use-cache`. | | `--use-cache/--no-use-cache` | | flag | `false` | Store artifacts in the winml-cli global cache (`~/.cache/winml/`). Mutually exclusive with `--output-dir`. | | `--rebuild/--no-rebuild` | | flag | `false` | Overwrite existing artifacts and re-run the full pipeline. | @@ -110,6 +112,16 @@ complete [onnxruntime-genai](https://github.com/microsoft/onnxruntime-genai) winml build -m Qwen/Qwen3-0.6B -o out/qwen3-bundle --export-type optimized ``` +The recipe supplies the context and prefill lengths when no overrides are +given. To customize them while keeping the same one-command workflow: + +```bash +winml build -m Qwen/Qwen3-0.6B -o out/qwen3-bundle \ + --export-type optimized \ + --max-cache-len 4096 \ + --prefill-seq-len 128 +``` + `--export-type optimized` resolves `--ep`/`--device` the same way a generic build does — an explicit value is honored, otherwise the host is probed — and then builds the recipe for that resolved target. The Qwen3 recipe supports diff --git a/docs/samples/qwen3-genai-bundle.md b/docs/samples/qwen3-genai-bundle.md index 16cb7680b..32f75aab0 100644 --- a/docs/samples/qwen3-genai-bundle.md +++ b/docs/samples/qwen3-genai-bundle.md @@ -95,18 +95,18 @@ Force a clean rebuild of every component with `--rebuild`. ## Step 2: Tune context and prefill lengths (optional) `winml build` uses the recipe defaults: context length (static KV cache) `2048` -and prefill sequence length `64`. To change those, use the equivalent developer -script, which exposes the extra knobs and delegates to the same builder: +and prefill sequence length `64`. Override either value directly when needed: ```bash -uv run python scripts/qwen3.py export \ +winml build -m Qwen/Qwen3-0.6B -o out/qwen3-bundle \ + --export-type optimized \ + --ep qnn \ --device npu \ - --output out/qwen3-bundle \ --max-cache-len 4096 \ --prefill-seq-len 128 ``` -The script also accepts `--embeddings ` and `--lm-head ` to reuse +The developer script also accepts `--embeddings ` and `--lm-head ` to reuse pre-built companions (skipping their builds), and `--force-rebuild` to rebuild everything from scratch. The developer script's `--device npu` shortcut targets QNN; use `winml build --ep vitisai --device npu` for an AMD bundle. diff --git a/src/winml/modelkit/commands/build.py b/src/winml/modelkit/commands/build.py index 03eda5ae2..caaca1f6a 100644 --- a/src/winml/modelkit/commands/build.py +++ b/src/winml/modelkit/commands/build.py @@ -613,6 +613,8 @@ def _maybe_build_genai_bundle( device: str, ep: EPNameOrAlias | None, precision: str | None, + max_cache_len: int | None, + prefill_seq_len: int | None, rebuild: bool, submodel: str | None, ) -> bool: @@ -773,6 +775,8 @@ def _maybe_build_genai_bundle( ep=bundle_ep, device=bundle_device, precision=override_precision, + max_cache_len=max_cache_len, + prefill_seq_len=prefill_seq_len, force_rebuild=rebuild, ) console.print( @@ -856,6 +860,24 @@ def _maybe_build_genai_bundle( "an explicit --ep qnn on an NPU target still routes a registered family to its " "optimized bundle (backward-compatible shortcut).", ) +@click.option( + "--max-cache-len", + type=click.IntRange(min=1), + default=None, + help=( + "Static KV cache (context) length for optimized GenAI bundles. " + "Defaults to the registered recipe's value." + ), +) +@click.option( + "--prefill-seq-len", + type=click.IntRange(min=1), + default=None, + help=( + "Prefill sequence length for optimized GenAI bundles. " + "Defaults to the registered recipe's value." + ), +) @cli_utils.shape_config_option( help_text="JSON with shape overrides for auto-generated HuggingFace export configs.", ) @@ -900,6 +922,8 @@ def build( device: str, precision: str, export_type: str, + max_cache_len: int | None, + prefill_seq_len: int | None, shape_config: Path | None, input_specs: Path | None, export_config: Path | None, @@ -1229,7 +1253,7 @@ def _patch_device(cfg: WinMLBuildConfig) -> None: # model-specific value lives in the recipe. ``--export-type generic`` (or # any other device/ep combination without the flag) keeps the stock # single/composite build. - if _maybe_build_genai_bundle( + built_genai_bundle = _maybe_build_genai_bundle( ctx, export_type=export_type, model=model, @@ -1241,10 +1265,18 @@ def _patch_device(cfg: WinMLBuildConfig) -> None: device=runtime_device, ep=runtime_ep_value, precision=precision, + max_cache_len=max_cache_len, + prefill_seq_len=prefill_seq_len, rebuild=rebuild, submodel=submodel, - ): + ) + if built_genai_bundle: return + if max_cache_len is not None or prefill_seq_len is not None: + raise click.UsageError( + "--max-cache-len and --prefill-seq-len are only supported for an " + "optimized GenAI bundle build." + ) if isinstance(config_or_configs, list): # ---- MODULE MODE: array config, one build per submodule ---- diff --git a/tests/unit/commands/test_build.py b/tests/unit/commands/test_build.py index fb6fdd362..2b82cf8d6 100644 --- a/tests/unit/commands/test_build.py +++ b/tests/unit/commands/test_build.py @@ -258,6 +258,8 @@ def test_help_shows_all_options(self, runner: CliRunner) -> None: assert "--verbose" in result.output assert "--no-analyze" in result.output assert "--max-optim-iterations" in result.output + assert "--max-cache-len" in result.output + assert "--prefill-seq-len" in result.output assert "--shape-config" in result.output assert "--input-specs" in result.output assert "--export-config" in result.output diff --git a/tests/unit/commands/test_build_genai_bundle.py b/tests/unit/commands/test_build_genai_bundle.py index c642b267f..5a081a39c 100644 --- a/tests/unit/commands/test_build_genai_bundle.py +++ b/tests/unit/commands/test_build_genai_bundle.py @@ -167,6 +167,8 @@ def test_registered_family_npu_qnn_routes_to_bundle(tmp_path: Path): assert kwargs["device"] == "npu" assert kwargs["force_rebuild"] is False assert kwargs["precision"] is None + assert kwargs["max_cache_len"] is None + assert kwargs["prefill_seq_len"] is None assert "emit" not in kwargs @@ -353,6 +355,74 @@ def test_precision_override_forwarded(tmp_path: Path): assert recorded["kwargs"]["precision"] == "w8a16" +def test_context_and_prefill_lengths_forwarded(tmp_path: Path): + recorded: dict = {} + + with ( + patch(_GENERATE_TARGET, return_value=_fake_config("qwen3")), + patch(_BUNDLE_TARGET, side_effect=_record_bundle(recorded)), + patch(_RUN_SINGLE_TARGET), + patch(_COMPOSITE_TARGET, return_value=None), + ): + result = _invoke( + [ + "-m", + "Qwen/Qwen3-0.6B", + "-o", + str(tmp_path / "bundle"), + "--export-type", + "optimized", + "--ep", + "qnn", + "--device", + "npu", + "--max-cache-len", + "4096", + "--prefill-seq-len", + "128", + ] + ) + + assert result.exit_code == 0, result.output + assert recorded["kwargs"]["max_cache_len"] == 4096 + assert recorded["kwargs"]["prefill_seq_len"] == 128 + + +@pytest.mark.parametrize("option", ["--max-cache-len", "--prefill-seq-len"]) +@pytest.mark.parametrize("value", ["0", "-1"]) +def test_context_and_prefill_lengths_must_be_positive(option: str, value: str): + result = _invoke(["-m", "Qwen/Qwen3-0.6B", "-o", "out", option, value]) + + assert result.exit_code != 0 + assert "x>=1" in result.output + + +def test_context_and_prefill_lengths_rejected_for_generic_build(tmp_path: Path): + with ( + patch(_GENERATE_TARGET, return_value=_fake_config("qwen3")), + patch(_BUNDLE_TARGET) as bundle, + patch(_RUN_SINGLE_TARGET) as run_single, + patch(_COMPOSITE_TARGET, return_value=None), + ): + result = _invoke( + [ + "-m", + "Qwen/Qwen3-0.6B", + "-o", + str(tmp_path / "generic"), + "--export-type", + "generic", + "--max-cache-len", + "4096", + ] + ) + + assert result.exit_code != 0 + assert "only supported for an optimized GenAI bundle build" in result.output + bundle.assert_not_called() + run_single.assert_not_called() + + def test_registered_family_auto_qnn_routes_to_bundle(tmp_path: Path): """``--device auto`` resolving to the NPU, with explicit ``--ep qnn``, routes.""" out = tmp_path / "bundle" @@ -768,6 +838,8 @@ def test_optimized_rejects_onnx_input(): device="auto", ep=None, precision=None, + max_cache_len=None, + prefill_seq_len=None, rebuild=False, submodel=None, ) @@ -789,6 +861,8 @@ def test_optimized_rejects_module_mode(): device="auto", ep=None, precision=None, + max_cache_len=None, + prefill_seq_len=None, rebuild=False, submodel=None, ) @@ -810,6 +884,8 @@ def test_optimized_requires_model(): device="auto", ep=None, precision=None, + max_cache_len=None, + prefill_seq_len=None, rebuild=False, submodel=None, )