From be958a39d249e45b71ca7f7384246dd24a17613c Mon Sep 17 00:00:00 2001 From: Joshua Calafato Date: Thu, 17 Sep 2026 02:05:41 +0000 Subject: [PATCH 1/8] feat(qwen3_5): add native Edge execution Keep original-source dense FP16 and explicit 4B/9B DFlash execution family-owned on the recorded native SM80 profile. Preserve the native builder and unchanged quality gates; retain current E2E evidence while saving failed native-command diagnostics. Document exact historical model receipts separately from current source and native-build validation. Exclude Qwen3 and older, MoE, 27B, quantized sources, and unqualified platforms. Signed-off-by: Joshua Calafato --- families/qwen3_5/EDGE_LLM.md | 93 +++++++ families/qwen3_5/dispatch.py | 142 ++++++++++ families/qwen3_5/edge_llm.py | 223 +++++++++++++++ families/qwen3_5/model.py | 198 +++++++++----- families/qwen3_5/runtime/CMakeLists.txt | 25 +- families/qwen3_5/runtime/edge_llm/adapter.cpp | 256 ++++++++++++++++++ families/qwen3_5/runtime/edge_llm/adapter.h | 15 + families/qwen3_5/runtime/edge_llm/contract.h | 56 ++++ .../qwen3_5/runtime/edge_llm/device_link.cu | 5 + families/qwen3_5/runtime/edge_llm/request.h | 29 ++ families/qwen3_5/runtime/plugin.cpp | 11 + families/qwen3_5/tests/test_e2e.py | 5 +- website/docs/features/model-families.md | 11 + 13 files changed, 998 insertions(+), 71 deletions(-) create mode 100644 families/qwen3_5/EDGE_LLM.md create mode 100644 families/qwen3_5/dispatch.py create mode 100644 families/qwen3_5/edge_llm.py create mode 100644 families/qwen3_5/runtime/edge_llm/adapter.cpp create mode 100644 families/qwen3_5/runtime/edge_llm/adapter.h create mode 100644 families/qwen3_5/runtime/edge_llm/contract.h create mode 100644 families/qwen3_5/runtime/edge_llm/device_link.cu create mode 100644 families/qwen3_5/runtime/edge_llm/request.h diff --git a/families/qwen3_5/EDGE_LLM.md b/families/qwen3_5/EDGE_LLM.md new file mode 100644 index 0000000000..c99aa4a95b --- /dev/null +++ b/families/qwen3_5/EDGE_LLM.md @@ -0,0 +1,93 @@ +# Qwen3.5: optional native Edge-LLM execution + +This family owns checkpoint admission, configuration mapping, bundle composition, +paired execution, runtime orchestration, and validation. Edge owns the complete +network. The optional CMake package is pinned to official GitHub Edge-LLM +**0.10.1 / e8b29522938901f6df19ebeedd4b69bc8edbcd97**. Build on the executing GPU; +there is no cross-compilation or use of an alternate Edge checkout. + +## Scope and fallback + +The route admits the recorded dense 0.8B/2B/4B/9B text configurations, original +source weights, FP16 compute, batch one, TP/CP one, on native Linux x86 SM80. +It does not admit Qwen3 or older, Qwen3.5 MoE, 27B, quantized sources, additional +platforms, or image/video requests. Unmapped ordinary requests keep the existing +native builder. Edge preparation failures warn and retry that unchanged native +request once; publication failures never switch backends. + +An explicit DFlash companion is admitted only for the recorded 4B/9B base +topologies and linear block16 profile. The family checks draft architecture, +hidden/vocabulary widths, target layers, mask token, context capacity, and +unquantized weights. If Edge fails, native reports that the requested DFlash +variant is unsupported; it must never silently return an ordinary base bundle. + +## Build and runtime + +Enable `TRTMC_ENABLE_EDGELLM=ON` and expose its native installation using +`CMAKE_PREFIX_PATH`. The family checks the Edge pin, SM, CUDA, and TensorRT +identity. Existing ordinary public build requests remain the entrypoint; +paired requests use the existing `execution_variant="dflash"` and explicit +draft-checkpoint input contract. + +`edge_llm.py` maps those requests into the pinned Python direct builder +(`experimental.builder.cli`). These recorded FP16 profiles already passed that +flow; they are not additional ONNX conversions. The family packages all required +engine, tokenizer, chat-template, configuration, and external-weight files. +DFlash maps both speculative engines and keeps the draft checkpoint's tensor +payloads while normalizing its safetensors header for the upstream reader. + +Ordinary builds preserve the requested context. DFlash caps the input profile +at min(context, 1024), keeps the requested KV capacity, and uses block16 with +one draft step and top-k one. The C++ family adapter invokes the pinned Edge +runtime and preserves the explicit execution variant. Native model math and +its recurrent-state initialization remain unchanged. + +## Historical full-model qualification + +These are saved local build/inference receipts, **not fresh publication-head +inference**. Every row passed its own local Model Connect build, public/direct +comparison, independent HF comparison under unchanged family criteria, and the +original Edge basic/context-reuse workloads. All eight ordinary runs used +native SM80 FP16, batch one, context4096. The original 9B owning E2E also passed +at its context256 profile with strict CLI JSON parsing. + +| Qwen checkpoint | Exact revision | Edge basic ROUGE-1 / ROUGE-L | +| --- | --- | --- | +| Qwen3.5-0.8B | `2fc06364715b967f1860aea9cf38778875588b17` | 0.3955 / 0.2599 | +| Qwen3.5-0.8B-Base | `dc7cdfe2ee4154fa7e30f5b51ca41bfa40174e68` | 0.3567 / 0.2166 | +| Qwen3.5-2B | `15852e8c16360a2fea060d615a32b45270f8a8fc` | 0.3929 / 0.2500 | +| Qwen3.5-2B-Base | `b1485b2fa6dfa1287294f269f5fb618e03d52d7c` | 0.5455 / 0.3102 | +| Qwen3.5-4B | `851bf6e806efd8d0a36b00ddf55e13ccb7b8cd0a` | 0.4485 / 0.2667 | +| Qwen3.5-4B-Base | `1001bb4d826a52d1f399e183466143f4da7b741b` | 0.4246 / 0.2011 | +| Qwen3.5-9B | `c202236235762e1c871ad0ccb60c8ee5ba337b9a` | 0.5116 / 0.2558 | +| Qwen3.5-9B-Base | `68c46c4b3498877f3ef123c856ecfde50c39f404` | 0.4379 / 0.2130 | + +The paired receipts use the 4B/9B Instruct revisions above and these z-lab drafts: + +| Draft | Exact revision | Speculative iterations | Edge basic ROUGE-1 / ROUGE-L | +| --- | --- | --- | --- | +| Qwen3.5-4B-DFlash | `9a1996ccf887b79ab3af4fcbf8c1d1f4b5658bcf` | 14 | 0.4485 / 0.2667 | +| Qwen3.5-9B-DFlash | `5fc3b3d474760f18c516db87d84c37edbfd3ede6` | 10 | 0.5116 / 0.2558 | + +Both used native SM80 FP16, context2048/input1024, block16, top-k one/one +draft step. Public and direct outputs agreed; HF comparisons passed the +unchanged maximum NED0.15 gate. The 9B pair reused the source/reference-verified +CPU-FP32 HF baseline rather than claiming a newly generated reference. + +## Existing tests and replay gaps + +Only the existing `tests/test_e2e.py` diagnostic handling changes: save native +stdout/stderr, keep current evidence instrumentation, then enforce the return +code and strict JSON parsing. No quality threshold, fixture, timeout, or +comparison is weakened. No new test driver or test framework is published. + +The current owning manifests cover 0.8B/2B/4B/9B, not their Base variants or +DFlash pairs. Manifest presence is not a fresh-head model pass. The existing +recurrent-output-initializer C++ test remains registered. + +Successful full-model payloads were retired after preserving compact evidence. +Fresh full inference requires restoring these exact source revisions and +rebuilding locally. CPU/source checks, native compilation, and generic CLI +checks do not replace that work or qualify unlisted models. No independent +logit-bias coverage is claimed from an upstream fixture that repeats the +basic workload. diff --git a/families/qwen3_5/dispatch.py b/families/qwen3_5/dispatch.py new file mode 100644 index 0000000000..c2e57f7c1e --- /dev/null +++ b/families/qwen3_5/dispatch.py @@ -0,0 +1,142 @@ +# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-License-Identifier: Apache-2.0 + +"""Family-owned complete-network route map; native is the default.""" + +from __future__ import annotations + +import json +import logging +import os +from pathlib import Path +import tempfile +import traceback + +from . import edge_llm + +_LOG = logging.getLogger(__name__) + +# Only the native platform exercised by the recorded original-source profiles. +EDGE_DISPATCH = {("linux", "x86_64", 80, "fp16"): edge_llm.prepare} + +# Layers, hidden, intermediate, attention heads, KV heads, vocabulary. +# These configurations are verified from the retained per-model engine metadata. +DENSE_CONFIGS = { + (24, 1024, 3584, 8, 2, 248320), + (24, 2048, 6144, 8, 2, 248320), + (32, 2560, 9216, 16, 4, 248320), + (32, 4096, 12288, 16, 4, 248320), +} + + +def candidate(request, raw: dict) -> bool: + """Return whether this family's model/request contract can delegate to Edge.""" + config = raw.get("text_config", raw) + return ( + isinstance(config, dict) + and raw.get("model_type") == "qwen3_5" + and tuple( + config.get(key) + for key in ( + "num_hidden_layers", + "hidden_size", + "intermediate_size", + "num_attention_heads", + "num_key_value_heads", + "vocab_size", + ) + ) + in DENSE_CONFIGS + and ("output_gate_type" not in config or "mlp_only_layers" in config) + and config.get("linear_key_head_dim") == config.get("linear_value_head_dim") == 128 + and not config.get("num_experts") + and not raw.get("quantization_config") + and not config.get("quantization_config") + and not any( + (Path(request.model_dir) / name).exists() + for name in ("hf_quant_config.json", "quantize_config.json", "quant_config.json") + ) + and request.backend == "trt" + and request.task == "text_generation" + and request.precision.lower() == "fp16" + and request.quantization in {None, "none"} + and request.max_batch_size + == request.tensor_parallel_size + == request.context_parallel_size + == 1 + and not request.dynamic_kv_cache + and not request.fp32_layers + and request.graph_transform is None + and all( + value is None + for value in (request.image_height, request.image_width, request.video_num_frames) + ) + ) + + +def build(request, writer, native, *, draft_dir: Path | None = None) -> None: + """Dispatch locally or warn and retry native once with the original request. + + Args: + request: Unmodified Model Connect build request. + writer: Unpublished bundle writer. + native: This family's original native builder callback. + + Raises: + Exception: Common input/publication error, or native build error with + Edge cause after a failed preparation. Cancellation never retries. + """ + if not (Path(request.model_dir) / "config.json").is_file(): + native(request, writer) + return + raw = json.loads((Path(request.model_dir) / "config.json").read_text(encoding="utf-8")) + if not isinstance(raw, dict): + raise ValueError("checkpoint config.json must contain an object") + config = raw.get("text_config", raw) + if not isinstance(config, dict): + raise ValueError("checkpoint text_config must contain an object") + if not candidate(request, raw): + native(request, writer) + return + capacity = config.get("max_position_embeddings") + if type(capacity) is not int or capacity <= 0: + raise ValueError("checkpoint max_position_embeddings must be a positive integer") + if request.max_sequence_length and request.max_sequence_length > capacity: + raise ValueError("max_sequence_length exceeds checkpoint context capacity") + failure = None + descriptor, name = tempfile.mkstemp( + prefix=f".{request.output_path.name}.edge-", suffix=".log", dir=request.output_path.parent + ) + os.close(descriptor) + log_path = Path(name) + with tempfile.TemporaryDirectory(prefix="trtmc-qwen3_5-edge-") as directory: + try: + target = edge_llm.local_target() + key = (target["os"], target["arch"], target["sm"], request.precision.lower()) + adapter = EDGE_DISPATCH.get(key) + if adapter is not None: + options = {} if draft_dir is None else {"draft_dir": draft_dir} + files, marker = adapter(request, raw, target, Path(directory), log_path, **options) + except Exception as error: + failure = error + with log_path.open("a", encoding="utf-8") as log: + traceback.print_exception(error, file=log) + _LOG.warning( + "qwen3_5 Edge build failed: %s. Diagnostics: %s. " + "Retrying native once with the unchanged request.", + error, + log_path, + exc_info=True, + ) + else: + # Edge preparation did not touch writer; publication cannot fallback. + if adapter is not None: + edge_llm.publish(request, writer, files, marker) + return + log_path.unlink() # A platform non-match is not an Edge failure. + try: + native(request, writer) + except Exception as error: + if failure is not None: + raise error from failure + raise diff --git a/families/qwen3_5/edge_llm.py b/families/qwen3_5/edge_llm.py new file mode 100644 index 0000000000..100364d2a1 --- /dev/null +++ b/families/qwen3_5/edge_llm.py @@ -0,0 +1,223 @@ +# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-License-Identifier: Apache-2.0 + +"""Thin family-owned adapter to the pinned Edge direct-builder API.""" + +from __future__ import annotations + +import json +from pathlib import Path +import shutil +import subprocess +import struct + +from tensorrt_model_connect.build import cmake_prefixes, detect_local_platform + +EDGE_REVISION = "e8b29522938901f6df19ebeedd4b69bc8edbcd97" + + +def local_target() -> dict: + """Return the executing worker identity supplied by generic build mechanics.""" + return detect_local_platform() + + +def installed_package(target: dict) -> dict: + """Resolve CMake installation via standard prefixes; never install anything. + + Args: + target: Executing device and SDK identity. + + Returns: + Validated package metadata with absolute Python and plugin paths. + + Raises: + FileNotFoundError: No CMake installation or required artifact exists. + ValueError: Pin, architecture, SDK or contained-path contract differs. + """ + for prefix in cmake_prefixes(): + manifest = prefix / "share/trtmc/edge-llm.json" + if not manifest.is_file(): + continue + package = json.loads(manifest.read_text(encoding="utf-8")) + if package.get("schema_version") != 1 or package.get("revision") != EDGE_REVISION: + raise ValueError(f"Edge package has an unsupported revision/schema: {manifest}") + if package.get("version") != "0.10.1" or package.get("arch") != target["arch"]: + raise ValueError("Edge package version/architecture differs from executing worker") + if target["sm"] not in package.get("architectures", []): + raise ValueError("Edge package was not built for this local GPU") + cuda_version = ".".join(str(package.get("cuda_version", "")).split(".")[:2]) + if ( + cuda_version != target["cuda_version"] + or package.get("tensorrt_version") != target["tensorrt_version"] + ): + raise ValueError("Edge package CUDA/TensorRT differs from executing worker") + for name in ("python", "plugin"): + relative = Path(package[name]) + path = (prefix / relative).resolve() + if relative.is_absolute() or not path.is_relative_to(prefix.resolve()): + raise ValueError(f"Edge package {name} must be contained in its installation") + if not path.is_file(): + raise FileNotFoundError(f"Edge package {name} is missing: {path}") + package[name] = str(path) + return package + raise FileNotFoundError( + "Edge-LLM is not installed; enable the optional Edge-LLM CMake dependency " + "and set CMAKE_PREFIX_PATH to its install prefix" + ) + + +def copy_draft_safetensors(source: Path, destination: Path) -> None: + """Align the draft payload for pinned Edge mapped GPU reads, without changing tensors. + + Some official DFlash files omit safetensors JSON-header padding. Only trailing + JSON whitespace and its length prefix change; descriptors, relative offsets, + and the complete tensor payload are copied byte-for-byte with bounded memory. + The original pinned checkpoint is never modified. + """ + with source.open("rb") as incoming: + prefix = incoming.read(8) + if len(prefix) != 8: + raise ValueError("Truncated DFlash safetensors header") + length = struct.unpack(" tuple[dict, dict]: + """Map the request to Edge main(argv), returning complete unpublished assets. + + Edge owns model selection, configuration, conversion, graphs and engine + composition. The family adapter only maps text-generation arguments and + preserves the checkpoint needed by Edge external-weight APIs. + + Returns: + (section-name to file mapping, runtime marker). + + Raises: + Exception: Dependency, upstream build or artifact validation failed. + """ + package = installed_package(target) + checkpoint = staging / "edge_llm/checkpoint" + checkpoint.mkdir(parents=True) + for source in Path(request.model_dir).iterdir(): + if source.is_file() and ( + source.suffix in {".json", ".safetensors", ".model", ".jinja"} + or source.name in {"merges.txt", "vocab.txt"} + ): + shutil.copy2(source, checkpoint / source.name) + if not list(checkpoint.glob("*.safetensors")): + raise ValueError("Edge direct builder requires a safetensors checkpoint") + if draft_dir is not None: + draft_checkpoint = checkpoint / "draft" + draft_checkpoint.mkdir() + for source in draft_dir.iterdir(): + if source.is_file() and source.suffix in {".json", ".safetensors"}: + if source.suffix == ".safetensors": + copy_draft_safetensors(source, draft_checkpoint / source.name) + else: + shutil.copy2(source, draft_checkpoint / source.name) + if not list(draft_checkpoint.glob("*.safetensors")): + raise ValueError("DFlash direct builder requires draft safetensors") + config = raw.get("text_config", raw) + limit = request.max_sequence_length or min(int(config["max_position_embeddings"]), 256) + input_limit = min(limit, 1024) if draft_dir is not None else limit + engine = staging / "edge_llm/engine" + # Calling upstream main preserves its complete build/artifact orchestration. + command = [ + package["python"], + "-I", + "-c", + "from experimental.builder.cli import main; main()", + "--model-dir", + str(checkpoint), + "--engine-dir", + str(engine), + "--components", + "llm", + "--plugin-path", + package["plugin"], + "--dense", + "fp16", + "--max-input-len", + str(input_limit), + "--max-kv-cache-capacity", + str(limit), + "--max-batch-size", + "1", + "--externalize-weights", + "all", + ] + if draft_dir is not None: + command.extend( + [ + "--spec-type", + "dflash", + "--draft-model-dir", + str(draft_checkpoint), + "--max-verify-tree-size", + "16", + "--max-draft-tree-size", + "16", + ] + ) + if request.verbose: + command.append("--verbose") + with log_path.open("a", encoding="utf-8") as log: + subprocess.run(command, check=True, stdout=log, stderr=subprocess.STDOUT, cwd=staging) + required = ["tokenizer.json", "tokenizer_config.json", "processed_chat_template.json"] + required += ( + ["spec_base.engine", "spec_draft.engine", "base_config.json", "draft_config.json"] + if draft_dir is not None + else ["llm.engine", "config.json"] + ) + for name in required: + if not (engine / name).is_file() or (engine / name).stat().st_size == 0: + raise ValueError(f"Edge builder did not produce required artifact: {name}") + files = {} + for directory in (engine, checkpoint): + for path in sorted(directory.rglob("*")): + if path.is_symlink(): + raise ValueError(f"Edge output must not contain symlinks: {path}") + if path.is_file(): + files[path.relative_to(staging).as_posix()] = path + return files, { + "version": 1, + "edge_revision": EDGE_REVISION, + "target": target, + "precision": "fp16", + "max_sequence_length": limit, + "max_input_length": input_limit, + "max_batch_size": 1, + "artifacts": list(files), + **( + {"execution_variant": "dflash", "dflash_block_size": 16} + if draft_dir is not None + else {} + ), + } + + +def publish(request, writer, files: dict, marker: dict) -> None: + """Stream complete Edge sections; publication errors must not retry native.""" + writer.set_header(family=request.family, task=request.task, backend=request.backend) + for name, path in files.items(): + with path.open("rb") as source, writer.open_section(name) as destination: + shutil.copyfileobj(source, destination, length=1024 * 1024) + writer.add_json("edge_llm.json", marker) diff --git a/families/qwen3_5/model.py b/families/qwen3_5/model.py index 3af3d7c9f4..32ea38e79f 100644 --- a/families/qwen3_5/model.py +++ b/families/qwen3_5/model.py @@ -1361,74 +1361,134 @@ def _runtime_config(model_dir: Path, config: ModelConfig, model: _Qwen35Model, * def build(request: "BuildRequest", writer: "BundleWriter") -> None: - """Build one Qwen3.5 hybrid bundle through family-owned code only.""" - if request.dynamic_kv_cache: - raise NotImplementedError("qwen3_5 does not support dynamic_kv_cache") - - if request.image_height is not None: - raise NotImplementedError("qwen3_5 does not support image_height") - - if request.image_width is not None: - raise NotImplementedError("qwen3_5 does not support image_width") - - if request.video_num_frames is not None: - raise NotImplementedError("qwen3_5 does not support video_num_frames") - - if request.max_batch_size != 1: - raise NotImplementedError("qwen3_5 does not support max_batch_size") - - if request.context_parallel_size != 1: - raise ValueError("this family does not support context parallelism") - - if request.task != "text_generation": - raise ValueError("qwen3_5 supports only task=text_generation") - - model_dir = Path(request.model_dir) - config = ModelConfig.from_dir(model_dir) - if str(config.model_type).lower() not in {"qwen3_5", "qwen3.5"}: - raise ValueError(f"Qwen3.5 does not support model_type={config.model_type!r}") - precision = str(request.precision).lower() - if precision not in {"fp32", "fp16"}: - raise ValueError("Qwen3.5 precision must be fp32 or fp16") - max_sequence_length = _positive_int( - request.max_sequence_length or min(config.max_position_embeddings, 256), - "max_sequence_length", - ) - if max_sequence_length > config.max_position_embeddings: - raise ValueError("Qwen3.5 max_sequence_length exceeds checkpoint context capacity") - if request.tensor_parallel_size != 1: - raise NotImplementedError("Qwen3.5 does not expose a tensor-parallel builder") - if request.quantization not in {None, "none"}: - raise NotImplementedError("Qwen3.5 does not support quantized builds") - - model = _Qwen35Model() - config.raw["_model_dir"] = str(model_dir) - config.raw["_fp32_layers"] = list(request.fp32_layers) - config.raw["_resolved_build_precision"] = precision - weights = model.load_weights(str(model_dir), config) - plan = model.build_engine( - config, - weights, - max_sequence_length, - precision=precision, - verbose=bool(request.verbose), - debug_layer_outputs=False, - ) - - writer.set_header(family="qwen3_5", task=request.task, backend=request.backend) - writer.add_bytes("engine.plan", plan) - writer.add_json( - "runtime.json", - _runtime_config( - model_dir, + """Select complete-network offload or preserve the native Qwen3.5 builder.""" + from .dispatch import build as dispatch_build + + def _build_native(request: "BuildRequest", writer: "BundleWriter") -> None: + """Build one Qwen3.5 hybrid bundle through family-owned code only.""" + if request.dynamic_kv_cache: + raise NotImplementedError("qwen3_5 does not support dynamic_kv_cache") + + if request.image_height is not None: + raise NotImplementedError("qwen3_5 does not support image_height") + + if request.image_width is not None: + raise NotImplementedError("qwen3_5 does not support image_width") + + if request.video_num_frames is not None: + raise NotImplementedError("qwen3_5 does not support video_num_frames") + + if request.max_batch_size != 1: + raise NotImplementedError("qwen3_5 does not support max_batch_size") + + if request.context_parallel_size != 1: + raise ValueError("this family does not support context parallelism") + + if request.task != "text_generation": + raise ValueError("qwen3_5 supports only task=text_generation") + + model_dir = Path(request.model_dir) + config = ModelConfig.from_dir(model_dir) + if str(config.model_type).lower() not in {"qwen3_5", "qwen3.5"}: + raise ValueError(f"Qwen3.5 does not support model_type={config.model_type!r}") + precision = str(request.precision).lower() + if precision not in {"fp32", "fp16"}: + raise ValueError("Qwen3.5 precision must be fp32 or fp16") + max_sequence_length = _positive_int( + request.max_sequence_length or min(config.max_position_embeddings, 256), + "max_sequence_length", + ) + if max_sequence_length > config.max_position_embeddings: + raise ValueError("Qwen3.5 max_sequence_length exceeds checkpoint context capacity") + if request.tensor_parallel_size != 1: + raise NotImplementedError("Qwen3.5 does not expose a tensor-parallel builder") + if request.quantization not in {None, "none"}: + raise NotImplementedError("Qwen3.5 does not support quantized builds") + + model = _Qwen35Model() + config.raw["_model_dir"] = str(model_dir) + config.raw["_fp32_layers"] = list(request.fp32_layers) + config.raw["_resolved_build_precision"] = precision + weights = model.load_weights(str(model_dir), config) + plan = model.build_engine( config, - model, + weights, + max_sequence_length, precision=precision, - max_cache_length=max_sequence_length, - decoder_engine_layout="single", - ), - ) - for filename in _BUNDLE_FILES: - path = model_dir / filename - if path.is_file(): - writer.add_bytes(filename, path.read_bytes()) + verbose=bool(request.verbose), + debug_layer_outputs=False, + ) + + writer.set_header(family="qwen3_5", task=request.task, backend=request.backend) + writer.add_bytes("engine.plan", plan) + writer.add_json( + "runtime.json", + _runtime_config( + model_dir, + config, + model, + precision=precision, + max_cache_length=max_sequence_length, + decoder_engine_layout="single", + ), + ) + for filename in _BUNDLE_FILES: + path = model_dir / filename + if path.is_file(): + writer.add_bytes(filename, path.read_bytes()) + + dispatch_build(request, writer, _build_native) + + +def build_with_inputs(request, writer, execution) -> None: + """Build an explicit Qwen3.5 DFlash pair, never a base-only replacement.""" + from . import dispatch + + if execution.variant != "dflash" or tuple(x.role for x in execution.checkpoints) != ("draft",): + raise ValueError( + "Qwen3.5 paired execution requires variant=dflash and one draft checkpoint" + ) + draft_dir = execution.checkpoints[0].model_dir + raw = json.loads((request.model_dir / "config.json").read_text()) + draft = json.loads((draft_dir / "config.json").read_text()) + if not dispatch.candidate(request, raw): + raise ValueError("Qwen3.5 DFlash requires an admitted dense base request") + base = raw.get("text_config", raw) + if (base.get("hidden_size"), base.get("num_hidden_layers")) not in {(2560, 32), (4096, 32)}: + raise ValueError("Qwen3.5 DFlash is qualified only for the recorded 4B/9B base profiles") + if not isinstance(draft, dict) or draft.get("architectures") != ["DFlashDraftModel"]: + raise ValueError("Expected a DFlashDraftModel companion") + for name in ("hidden_size", "vocab_size"): + if type(draft.get(name)) is not int or draft[name] != base.get(name): + raise ValueError(f"Qwen3.5 DFlash base and draft disagree on {name}") + if draft.get("num_target_layers") != base.get("num_hidden_layers"): + raise ValueError("Qwen3.5 DFlash target layer count differs from base") + config = draft.get("dflash_config") + if not isinstance(config, dict) or config.get("block_size") != 16: + raise ValueError("Qwen3.5 DFlash currently maps the upstream linear block16 profile") + layers = config.get("target_layer_ids") + if ( + not isinstance(layers, list) + or not layers + or any(type(i) is not int or not 0 <= i < base["num_hidden_layers"] for i in layers) + or len(set(layers)) != len(layers) + ): + raise ValueError("Invalid Qwen3.5 DFlash target layer IDs") + mask = config.get("mask_token_id") + if type(mask) is not int or not 0 <= mask < base["vocab_size"]: + raise ValueError("Invalid Qwen3.5 DFlash mask token") + capacity = draft.get("max_position_embeddings") + limit = request.max_sequence_length or min(base["max_position_embeddings"], 256) + if type(capacity) is not int or not 16 < limit <= capacity: + raise ValueError("Requested context exceeds DFlash draft capacity or block minimum") + if draft.get("quantization_config") or any( + (draft_dir / name).exists() + for name in ("hf_quant_config.json", "quantize_config.json", "quant_config.json") + ): + raise ValueError("This Qwen3.5 DFlash profile requires unquantized draft weights") + + def native_pair(original_request, original_writer): + # Preserve the requested variant on fallback; native has no DFlash decoder. + raise NotImplementedError("Native Qwen3.5 does not implement the requested DFlash variant") + + dispatch.build(request, writer, native_pair, draft_dir=draft_dir) diff --git a/families/qwen3_5/runtime/CMakeLists.txt b/families/qwen3_5/runtime/CMakeLists.txt index 769359c270..8012ee0f97 100644 --- a/families/qwen3_5/runtime/CMakeLists.txt +++ b/families/qwen3_5/runtime/CMakeLists.txt @@ -28,7 +28,7 @@ target_link_libraries(trtmc_model_qwen3_5 PRIVATE ) target_compile_options(trtmc_model_qwen3_5 PRIVATE "$<$:-Wall;-Wextra;-Wpedantic>" - "$<$:-Wall;-Wextra>" + "$<$:-Xcompiler=-Wall,-Wextra>" ) set_target_properties(trtmc_model_qwen3_5 PROPERTIES LIBRARY_OUTPUT_DIRECTORY "${CMAKE_BINARY_DIR}" @@ -55,3 +55,26 @@ if(TRTMC_BUILD_TESTS) ) add_test(NAME ${test_name} COMMAND ${test_name}) endif() + +# Complete-network offload is family-owned and absent from native-only builds. +if(TARGET EdgeLLM::Core) + target_sources(trtmc_model_qwen3_5 PRIVATE + edge_llm/adapter.cpp + edge_llm/device_link.cu + ) + target_compile_definitions(trtmc_model_qwen3_5 PRIVATE TRTMC_HAS_EDGE_LLM=1) + target_link_libraries(trtmc_model_qwen3_5 PRIVATE EdgeLLM::Core) + set_target_properties(trtmc_model_qwen3_5 PROPERTIES + CUDA_ARCHITECTURES "${EdgeLLM_CUDA_ARCHITECTURE}" + CUDA_SEPARABLE_COMPILATION ON + CUDA_RESOLVE_DEVICE_SYMBOLS ON + ) +endif() + +if(TARGET EdgeLLM::Core) + add_custom_command(TARGET trtmc_model_qwen3_5 POST_BUILD + COMMAND ${CMAKE_COMMAND} -E copy_if_different + $ $ + VERBATIM + ) +endif() diff --git a/families/qwen3_5/runtime/edge_llm/adapter.cpp b/families/qwen3_5/runtime/edge_llm/adapter.cpp new file mode 100644 index 0000000000..cad34f553c --- /dev/null +++ b/families/qwen3_5/runtime/edge_llm/adapter.cpp @@ -0,0 +1,256 @@ +/* + * SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. + * SPDX-License-Identifier: Apache-2.0 + */ +#include "families/qwen3_5/runtime/edge_llm/adapter.h" + +#include "families/qwen3_5/runtime/edge_llm/request.h" + +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include + +namespace trtmc::qwen3_5::edge_llm { +namespace { +namespace fs = std::filesystem; + +/// Turn CUDA failures into caller-visible load or inference errors. +void check_cuda(cudaError_t result) { + if (result != cudaSuccess) + throw std::runtime_error(std::string("Qwen3.5 Edge CUDA error: ") + + cudaGetErrorString(result)); +} + +/// Reject an engine built for a different local GPU or CUDA/TensorRT runtime. +void validate_target(const nlohmann::json& target) { + utsname host{}; + if (uname(&host) != 0) + throw std::runtime_error("Cannot identify Qwen3.5 Edge runtime host"); + std::ifstream release("/etc/os-release"); + std::string line, os_version; + while (std::getline(release, line)) { + if (line.rfind("VERSION_ID=", 0) == 0) { + os_version = line.substr(11); + if (os_version.size() >= 2 && os_version.front() == char(34) && + os_version.back() == char(34)) + os_version = os_version.substr(1, os_version.size() - 2); + } + } + int device = 0, cuda_version = 0; + check_cuda(cudaGetDevice(&device)); + check_cuda(cudaRuntimeGetVersion(&cuda_version)); + cudaDeviceProp gpu{}; + check_cuda(cudaGetDeviceProperties(&gpu, device)); + const int trt_version = getInferLibVersion(); + const std::string trt = + std::to_string(trt_version / 10000) + "." + std::to_string((trt_version % 10000) / 100) + + "." + std::to_string(trt_version % 100) + "." + std::to_string(getInferLibBuildVersion()); + const std::string cuda = + std::to_string(cuda_version / 1000) + "." + std::to_string((cuda_version % 1000) / 10); + if (target.at("os") != "linux" || target.at("os_version") != os_version || + target.at("arch") != host.machine || target.at("sm") != gpu.major * 10 + gpu.minor || + target.at("cuda_version") != cuda || target.at("tensorrt_version") != trt) + throw std::runtime_error( + "Qwen3.5 Edge bundle requires its build GPU and CUDA/TensorRT stack"); +} + +/// Own extracted engine/checkpoint files until after the Edge runtime is destroyed. +class Artifacts { + public: + explicit Artifacts(const BundleReader& bundle, const nlohmann::json& marker) { + std::set names; + for (const auto& entry : marker.at("artifacts")) { + const auto name = entry.get(); + if (!safe_artifact_path(name) || !names.insert(name).second || + !bundle.find_section(name)) + throw std::runtime_error("Invalid Qwen3.5 Edge artifact: " + name); + } + std::vector required_files{ + "edge_llm/engine/tokenizer.json", "edge_llm/engine/tokenizer_config.json", + "edge_llm/engine/processed_chat_template.json", "edge_llm/checkpoint/config.json"}; + if (marker.value("execution_variant", "autoregressive") == "dflash") { + for (const auto* name : + {"spec_base.engine", "spec_draft.engine", "base_config.json", "draft_config.json"}) + required_files.push_back(std::string("edge_llm/engine/") + name); + required_files.push_back("edge_llm/checkpoint/draft/config.json"); + } else { + required_files.push_back("edge_llm/engine/llm.engine"); + required_files.push_back("edge_llm/engine/config.json"); + } + for (const auto& required : required_files) + if (!names.count(required) || bundle.find_section(required)->length == 0) + throw std::runtime_error("Required Qwen3.5 Edge artifact missing: " + required); + std::string pattern = (fs::temp_directory_path() / "trtmc-qwen3_5-edge-XXXXXX").string(); + if (!mkdtemp(pattern.data())) + throw std::runtime_error("Cannot create Qwen3.5 Edge artifact directory"); + root_ = pattern; + try { + for (const auto& name : names) { + const auto destination = root_ / name; + fs::create_directories(destination.parent_path()); + std::ofstream output(destination, std::ios::binary); + bundle.copy_section(name, output); + output.close(); + if (!output) + throw std::runtime_error("Cannot extract Qwen3.5 Edge artifact: " + name); + } + } catch (...) { + cleanup(); + throw; + } + } + ~Artifacts() { cleanup(); } + Artifacts(const Artifacts&) = delete; + Artifacts& operator=(const Artifacts&) = delete; + std::string engine() const { return (root_ / "edge_llm/engine").string(); } + std::string checkpoint() const { return (root_ / "edge_llm/checkpoint").string(); } + + private: + void cleanup() noexcept { + std::error_code ignored; + fs::remove_all(root_, ignored); + } + fs::path root_; +}; + +/// Close the plugin handle after runtime destruction; registrations remain mapped. +struct CloseLibrary { + void operator()(void* handle) const noexcept { + if (handle) + dlclose(handle); + } +}; + +/// Initialize the CMake-installed adjacent plugin without process-global environment mutation. +std::unique_ptr load_plugin() { + Dl_info location{}; + if (!dladdr(reinterpret_cast(&create), &location) || !location.dli_fname) + throw std::runtime_error("Cannot locate Qwen3.5 family library"); + const auto path = + fs::absolute(location.dli_fname).parent_path() / "libNvInfer_edgellm_plugin.so"; + std::unique_ptr plugin( + dlopen(path.c_str(), RTLD_NOW | RTLD_GLOBAL | RTLD_NODELETE)); + if (!plugin) + throw std::runtime_error("Cannot load CMake-installed Edge plugin: " + + std::string(dlerror())); + using Initialize = bool (*)(void*, const char*); + auto initialize = reinterpret_cast(dlsym(plugin.get(), "initEdgellmPlugins")); + if (!initialize || !initialize(static_cast(&trt_edgellm::gLogger), "")) + throw std::runtime_error("Cannot initialize Qwen3.5 Edge plugin"); + return plugin; +} + +/// Stream ownership is independent of construction success and outlives the Edge instance. +class Stream { + public: + Stream() { check_cuda(cudaStreamCreateWithFlags(&value_, cudaStreamNonBlocking)); } + ~Stream() { cudaStreamDestroy(value_); } + Stream(const Stream&) = delete; + Stream& operator=(const Stream&) = delete; + cudaStream_t get() const { return value_; } + + private: + cudaStream_t value_{nullptr}; +}; + +/// Delegate the complete linear DFlash block16 algorithm to the pinned runtime. +std::unique_ptr +make_runtime(const Artifacts& artifacts, const nlohmann::json& marker, cudaStream_t stream) { + using Runtime = trt_edgellm::rt::LLMInferenceRuntime; + if (marker.value("execution_variant", "autoregressive") == "dflash") { + trt_edgellm::rt::SpecDecodeDraftingConfig drafting{}; + drafting.draftingTopK = 1; + drafting.draftingStep = 1; + drafting.verifySize = 16; + drafting.dflashBlockSize = 16; + return std::make_unique( + artifacts.engine(), "", std::unordered_map{}, drafting, + stream, trt_edgellm::rt::ContextCacheConfig{}, artifacts.checkpoint(), + (fs::path(artifacts.checkpoint()) / "draft").string()); + } + return std::make_unique(artifacts.engine(), "", + std::unordered_map{}, stream, + trt_edgellm::rt::ContextCacheConfig{}, artifacts.checkpoint()); +} + +/// Thin persistent Edge API adapter; serialization prevents concurrent use of Edge request state. +class EdgeTask final : public ITextGeneration { + public: + EdgeTask(const BundleReader& bundle, const nlohmann::json& marker) + : artifacts_(bundle, marker), plugin_(load_plugin()), + runtime_(make_runtime(artifacts_, marker, stream_.get())), + capacity_(marker.at("max_sequence_length").get()), + input_limit_(marker.at("max_input_length").get()) {} + + std::int32_t default_max_new_tokens() const override { return std::min(128, capacity_ - 1); } + + /// Drain work from failed requests before destroying the runtime and its weight buffers. + ~EdgeTask() override { cudaStreamSynchronize(stream_.get()); } + + /// Invoke Edge once; failures propagate without attempting native inference. + TextResult generate(const std::string& prompt, const TextGenerationConfig& config) override { + auto request = make_request(prompt, config, default_max_new_tokens()); + std::lock_guard lock(mutex_); + const auto counts = runtime_->countPromptTokens(request); + if (counts.size() != 1) + throw std::runtime_error("Qwen3.5 Edge returned invalid prompt counts"); + validate_capacity(counts.front(), input_limit_, capacity_, request.maxGenerateLength); + trt_edgellm::rt::LLMGenerationResponse response{}; + if (!runtime_->handleRequest(request, response, stream_.get()) || + response.outputIds.size() != 1 || response.outputTexts.size() != 1 || + response.outputIds.front().empty() || + response.outputIds.front().size() > static_cast(request.maxGenerateLength)) + throw std::runtime_error("Qwen3.5 Edge generation failed"); + if (response.finishReasons.size() != 1 || + (response.finishReasons.front() != trt_edgellm::rt::FinishReason::kEndId && + response.finishReasons.front() != trt_edgellm::rt::FinishReason::kLength)) + throw std::runtime_error("Qwen3.5 Edge generation did not complete successfully"); + // This API does not expose per-request stage times; zero means unavailable. + return {std::move(response.outputTexts.front()), std::move(response.outputIds.front())}; + } + + private: + // Reverse destruction order keeps weights, plugin and stream alive throughout Edge teardown. + Artifacts artifacts_; + std::unique_ptr plugin_; + Stream stream_; + std::unique_ptr runtime_; + int capacity_; + int input_limit_; + std::mutex mutex_; +}; +} // namespace + +ITask* create(const BundleReader& bundle) { + const auto bytes = bundle.read_section("edge_llm.json"); + const auto marker = nlohmann::json::parse(bytes.begin(), bytes.end()); + if (marker.at("version") != 1 || marker.at("edge_revision") != kRevision || + marker.at("max_sequence_length").get() <= 1 || + marker.at("max_input_length").get() <= 0 || + marker.at("max_input_length").get() > marker.at("max_sequence_length").get() || + marker.at("max_batch_size") != 1 || marker.at("precision") != "fp16" || + !marker.at("artifacts").is_array()) + throw std::runtime_error("Invalid Qwen3.5 Edge bundle contract"); + const auto variant = marker.value("execution_variant", "autoregressive"); + if ((variant != "autoregressive" && variant != "dflash") || + (variant == "dflash" && marker.at("dflash_block_size") != 16)) + throw std::runtime_error("Unsupported Qwen3.5 Edge execution variant"); + validate_target(marker.at("target")); + return new EdgeTask(bundle, marker); +} + +} // namespace trtmc::qwen3_5::edge_llm diff --git a/families/qwen3_5/runtime/edge_llm/adapter.h b/families/qwen3_5/runtime/edge_llm/adapter.h new file mode 100644 index 0000000000..ac3277daae --- /dev/null +++ b/families/qwen3_5/runtime/edge_llm/adapter.h @@ -0,0 +1,15 @@ +/* + * SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. + * SPDX-License-Identifier: Apache-2.0 + */ +#pragma once + +#include "trtmc/bundle.h" +#include "trtmc/task.h" + +namespace trtmc::qwen3_5::edge_llm { + +/// Create a persistent Edge task from a self-contained bundle; throws on load failure. +ITask* create(const BundleReader& bundle); + +} // namespace trtmc::qwen3_5::edge_llm diff --git a/families/qwen3_5/runtime/edge_llm/contract.h b/families/qwen3_5/runtime/edge_llm/contract.h new file mode 100644 index 0000000000..fe9d58be94 --- /dev/null +++ b/families/qwen3_5/runtime/edge_llm/contract.h @@ -0,0 +1,56 @@ +/* + * SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. + * SPDX-License-Identifier: Apache-2.0 + */ +#pragma once + +#include "trtmc/task.h" + +#include +#include +#include +#include + +namespace trtmc::qwen3_5::edge_llm { + +inline constexpr const char* kRevision = "e8b29522938901f6df19ebeedd4b69bc8edbcd97"; + +/// Return whether an artifact is a normalized file below one of the two Edge roots. +inline bool safe_artifact_path(const std::string& name) { + if (name.find('\\') != std::string::npos || name.find('\0') != std::string::npos) + return false; + const std::filesystem::path path(name); + if (path.is_absolute() || path.filename().empty()) + return false; + for (const auto& part : path) + if (part == "." || part == "..") + return false; + return path.generic_string() == name && + (name.rfind("edge_llm/engine/", 0) == 0 || name.rfind("edge_llm/checkpoint/", 0) == 0); +} + +/// Reject invalid sampling settings and controls with no equivalent Edge request API. +inline void validate_generation(const TextGenerationConfig& c) { + if (!std::isfinite(c.temperature) || c.temperature < 0 || !std::isfinite(c.top_p) || + c.top_p <= 0 || c.top_p > 1 || c.top_k < 0) + throw std::invalid_argument("Invalid Qwen3.5 Edge sampling parameters"); + if (c.min_p != 0 || c.seed != -1 || c.eos_token_id != -1 || c.repetition_penalty != 1 || + !c.lora_adapter_id.empty() || c.stop_on_boxed_answer || + (c.text_generation_mode != "auto" && c.text_generation_mode != "autoregressive") || + c.source_language_token_id != -1 || c.forced_bos_token_id != -1 || c.guidance_scale != -1 || + c.cfg_scale != -1 || c.num_steps != -1 || c.sde_gamma != -1 || !c.initial_latents.empty() || + !c.condition_latents.empty() || !c.condition_mask.empty() || !c.sampling_steps.empty() || + !c.sde_noises.empty() || c.block_length != 0 || c.confidence_threshold != -1) + throw std::invalid_argument( + "Requested generation controls are unsupported by Qwen3.5 Edge"); +} + +/// Enforce prompt and total capacity without allowing Edge to silently clip generation. +inline void validate_capacity(int prompt_tokens, int input_limit, int capacity, + std::int64_t generated_tokens) { + if (prompt_tokens <= 0 || prompt_tokens > input_limit || generated_tokens <= 0 || + generated_tokens > static_cast(capacity) - prompt_tokens) + throw std::invalid_argument("Qwen3.5 Edge prompt and generation exceed bundle capacity"); +} + +} // namespace trtmc::qwen3_5::edge_llm diff --git a/families/qwen3_5/runtime/edge_llm/device_link.cu b/families/qwen3_5/runtime/edge_llm/device_link.cu new file mode 100644 index 0000000000..1ea4768d60 --- /dev/null +++ b/families/qwen3_5/runtime/edge_llm/device_link.cu @@ -0,0 +1,5 @@ +/* + * SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. + * SPDX-License-Identifier: Apache-2.0 + */ +// Enable the final CUDA device-link step for Edge's static runtime dependencies. diff --git a/families/qwen3_5/runtime/edge_llm/request.h b/families/qwen3_5/runtime/edge_llm/request.h new file mode 100644 index 0000000000..ba15809542 --- /dev/null +++ b/families/qwen3_5/runtime/edge_llm/request.h @@ -0,0 +1,29 @@ +/* + * SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. + * SPDX-License-Identifier: Apache-2.0 + */ +#pragma once + +#include "families/qwen3_5/runtime/edge_llm/contract.h" + +#include + +namespace trtmc::qwen3_5::edge_llm { + +/// Map Model Connect text arguments to the pinned Edge API; rejects unmapped controls. +inline trt_edgellm::rt::LLMGenerationRequest +make_request(const std::string& prompt, const TextGenerationConfig& config, int default_length) { + validate_generation(config); + trt_edgellm::rt::LLMGenerationRequest request{}; + request.requests.resize(1); + request.requests.front().messages.push_back({"user", {{"text", prompt}}}); + request.applyChatTemplate = config.use_chat_template; + request.enableThinking = config.enable_thinking; + request.temperature = config.temperature; + request.topK = config.top_k; + request.topP = config.top_p; + request.maxGenerateLength = config.max_new_tokens > 0 ? config.max_new_tokens : default_length; + return request; +} + +} // namespace trtmc::qwen3_5::edge_llm diff --git a/families/qwen3_5/runtime/plugin.cpp b/families/qwen3_5/runtime/plugin.cpp index 7786e34b43..d855bc7a2e 100644 --- a/families/qwen3_5/runtime/plugin.cpp +++ b/families/qwen3_5/runtime/plugin.cpp @@ -9,6 +9,9 @@ #include "families/qwen3_5/runtime/plugin_helpers.h" #include "families/qwen3_5/runtime/recurrent_state.h" #include "trtmc/runtime/family_factory.h" +#ifdef TRTMC_HAS_EDGE_LLM +#include "families/qwen3_5/runtime/edge_llm/adapter.h" +#endif #include #include @@ -119,6 +122,14 @@ std::string chat_template(const BundleReader& bundle) { } // namespace ITask* create(const FamilyContext& context) { + if (context.reader.find_section("edge_llm.json")) { +#ifdef TRTMC_HAS_EDGE_LLM + return edge_llm::create(context.reader); +#else + throw std::runtime_error("Qwen3.5 Edge bundle requires a runtime configured with " + "-DTRTMC_ENABLE_EDGELLM=ON; rebuild and install Model Connect"); +#endif + } const RuntimeConfig config = parse_runtime_config(context.reader); auto decoder = load_engine(context.backend, require_section(context.reader, "engine.plan"), "qwen3_5 decoder"); diff --git a/families/qwen3_5/tests/test_e2e.py b/families/qwen3_5/tests/test_e2e.py index cd5ad94989..7ee2a75f07 100644 --- a/families/qwen3_5/tests/test_e2e.py +++ b/families/qwen3_5/tests/test_e2e.py @@ -227,7 +227,7 @@ def _run_native( completed = subprocess.run( command, - check=True, + check=False, capture_output=True, text=True, timeout=600, @@ -235,6 +235,9 @@ def _run_native( ) record_evidence("commands", {"argv": getattr(completed, "args", None)}) record_evidence("native", {"stdout": getattr(completed, "stdout", None), "stderr": getattr(completed, "stderr", None)}) + (tmp_path / "native.stdout.log").write_text(completed.stdout, encoding="utf-8") + (tmp_path / "native.stderr.log").write_text(completed.stderr, encoding="utf-8") + completed.check_returncode() if tp_size == 1: return json.loads(completed.stdout) diff --git a/website/docs/features/model-families.md b/website/docs/features/model-families.md index 4b8cd3f5f8..6f4c9e1cbc 100644 --- a/website/docs/features/model-families.md +++ b/website/docs/features/model-families.md @@ -139,3 +139,14 @@ features through a small family-owned test helper without changing its gates. Quantized sources, InternVL3.5, and multi-image public requests are excluded. See the [family recipe](https://github.com/NVIDIA/TensorRT-Model-Connect/blob/main/families/internvl/edge_llm/README.md) for exact source revisions, evidence boundaries, and replay requirements. + +### Optional Qwen3.5 Edge execution + +The Qwen3.5 family maps the recorded original-source dense 0.8B/2B/4B/9B +configurations (Instruct and Base) and explicit 4B/9B DFlash pairs to the pinned +Edge-LLM 0.10.1 SDK on native x86 SM80, FP16, batch one. Historical local +build/public/direct/HF receipts are documented separately from current-head +checks. Qwen3 or older, MoE, 27B, quantized sources, and other platforms are not +qualified by this route. Existing quality gates are unchanged. +See the [family recipe](https://github.com/NVIDIA/TensorRT-Model-Connect/blob/main/families/qwen3_5/EDGE_LLM.md) +for exact revisions, paired execution, and replay gaps. From d51890e91b8d18e81335ff2b5eaa6044f42f4bdb Mon Sep 17 00:00:00 2001 From: Joshua Calafato Date: Mon, 21 Sep 2026 17:19:33 +0000 Subject: [PATCH 2/8] fix(qwen3_5): clean successful Edge diagnostics Signed-off-by: Joshua Calafato --- families/qwen3_5/dispatch.py | 1 + 1 file changed, 1 insertion(+) diff --git a/families/qwen3_5/dispatch.py b/families/qwen3_5/dispatch.py index c2e57f7c1e..9f0d01face 100644 --- a/families/qwen3_5/dispatch.py +++ b/families/qwen3_5/dispatch.py @@ -132,6 +132,7 @@ def build(request, writer, native, *, draft_dir: Path | None = None) -> None: # Edge preparation did not touch writer; publication cannot fallback. if adapter is not None: edge_llm.publish(request, writer, files, marker) + log_path.unlink() return log_path.unlink() # A platform non-match is not an Edge failure. try: From 4cf86cacc8797cb3175aa7d311032634d0fdec2b Mon Sep 17 00:00:00 2001 From: Joshua Calafato Date: Wed, 23 Sep 2026 04:34:50 +0000 Subject: [PATCH 3/8] refactor(qwen3_5): own Edge build selection Keep optional Edge dispatch, request options, adapters, and CMake wiring within family-owned edge_llm folders. Route explicit companions through the generic lazy CLI hook and the ordinary family build entrypoint; preserve native fallback semantics and quality gates. Signed-off-by: Joshua Calafato --- .../{EDGE_LLM.md => edge_llm/README.md} | 42 +++++- families/qwen3_5/edge_llm/__init__.py | 4 + .../{edge_llm.py => edge_llm/builder.py} | 0 families/qwen3_5/edge_llm/cli.py | 42 ++++++ families/qwen3_5/edge_llm/config.py | 88 ++++++++++++ families/qwen3_5/{ => edge_llm}/dispatch.py | 56 +++++++- families/qwen3_5/model.py | 63 ++------- families/qwen3_5/runtime/CMakeLists.txt | 23 +-- .../qwen3_5/runtime/edge_llm/Adapter.cmake | 28 ++++ families/qwen3_5/support.py | 1 + .../qwen3_5/tests/test_precision_contract.py | 133 ++++++++++++++++++ website/docs/features/model-families.md | 2 +- 12 files changed, 398 insertions(+), 84 deletions(-) rename families/qwen3_5/{EDGE_LLM.md => edge_llm/README.md} (75%) create mode 100644 families/qwen3_5/edge_llm/__init__.py rename families/qwen3_5/{edge_llm.py => edge_llm/builder.py} (100%) create mode 100644 families/qwen3_5/edge_llm/cli.py create mode 100644 families/qwen3_5/edge_llm/config.py rename families/qwen3_5/{ => edge_llm}/dispatch.py (64%) create mode 100644 families/qwen3_5/runtime/edge_llm/Adapter.cmake diff --git a/families/qwen3_5/EDGE_LLM.md b/families/qwen3_5/edge_llm/README.md similarity index 75% rename from families/qwen3_5/EDGE_LLM.md rename to families/qwen3_5/edge_llm/README.md index c99aa4a95b..be185c9261 100644 --- a/families/qwen3_5/EDGE_LLM.md +++ b/families/qwen3_5/edge_llm/README.md @@ -26,10 +26,9 @@ variant is unsupported; it must never silently return an ordinary base bundle. Enable `TRTMC_ENABLE_EDGELLM=ON` and expose its native installation using `CMAKE_PREFIX_PATH`. The family checks the Edge pin, SM, CUDA, and TensorRT identity. Existing ordinary public build requests remain the entrypoint; -paired requests use the existing `execution_variant="dflash"` and explicit -draft-checkpoint input contract. +paired requests use the family-owned CLI options shown below. -`edge_llm.py` maps those requests into the pinned Python direct builder +`edge_llm/builder.py` maps those requests into the pinned Python direct builder (`experimental.builder.cli`). These recorded FP16 profiles already passed that flow; they are not additional ONNX conversions. The family packages all required engine, tokenizer, chat-template, configuration, and external-weight files. @@ -76,10 +75,12 @@ CPU-FP32 HF baseline rather than claiming a newly generated reference. ## Existing tests and replay gaps -Only the existing `tests/test_e2e.py` diagnostic handling changes: save native +The existing `tests/test_e2e.py` diagnostic handling changes: save native stdout/stderr, keep current evidence instrumentation, then enforce the return code and strict JSON parsing. No quality threshold, fixture, timeout, or -comparison is weakened. No new test driver or test framework is published. +comparison is weakened. The existing precision-contract tests also cover the family-owned build options, +request identity, and paired-publication failure behavior. No new test driver or +test framework is published. The current owning manifests cover 0.8B/2B/4B/9B, not their Base variants or DFlash pairs. Manifest presence is not a fresh-head model pass. The existing @@ -91,3 +92,34 @@ rebuilding locally. CPU/source checks, native compilation, and generic CLI checks do not replace that work or qualify unlisted models. No independent logit-bias coverage is claimed from an upstream fixture that repeats the basic workload. + +## Family-owned build options + +The generic CLI loads this family's `edge_llm.cli` hook only after resolving +the model. The shared build API has no execution-mode or companion arguments. +All variant validation and builder selection remain in this family. + +```sh +trtmc build /path/to/target --family qwen3_5 --precision fp16 \ + --execution-variant dflash --companion draft=/path/to/draft \ + -o model.bundle +``` + +Put MODEL immediately after `build`. `trtmc build /path/to/target --help` +shows these family options using local metadata; remote-ID help does not download +a checkpoint. For Python callers, use this family's request extension: + +```python +from tensorrt_model_connect import build +from families.qwen3_5.edge_llm.config import ( + BuildExecutionInputs, NamedCheckpoint, with_execution, +) + +# request is an ordinary BuildRequest owned by this family; draft_path is a Path. +build(with_execution(request, BuildExecutionInputs( + "dflash", (NamedCheckpoint("draft", draft_path),), +))) +``` + +A failed explicit pair is never replaced by a base-only bundle. Previously +recorded full-model results above are historical, not fresh refactor-head E2Es. diff --git a/families/qwen3_5/edge_llm/__init__.py b/families/qwen3_5/edge_llm/__init__.py new file mode 100644 index 0000000000..df93d8cabe --- /dev/null +++ b/families/qwen3_5/edge_llm/__init__.py @@ -0,0 +1,4 @@ +# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-License-Identifier: Apache-2.0 + +"""Family-owned optional complete-network Edge offload.""" diff --git a/families/qwen3_5/edge_llm.py b/families/qwen3_5/edge_llm/builder.py similarity index 100% rename from families/qwen3_5/edge_llm.py rename to families/qwen3_5/edge_llm/builder.py diff --git a/families/qwen3_5/edge_llm/cli.py b/families/qwen3_5/edge_llm/cli.py new file mode 100644 index 0000000000..544932b751 --- /dev/null +++ b/families/qwen3_5/edge_llm/cli.py @@ -0,0 +1,42 @@ +# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-License-Identifier: Apache-2.0 + +"""Qwen35-owned CLI options for explicit paired Edge execution.""" + +import argparse +from pathlib import Path + +from .config import BuildExecutionInputs, NamedCheckpoint, with_execution + + +def add_build_arguments(parser: argparse.ArgumentParser) -> None: + """Register options only when the Qwen35 family has been resolved.""" + parser.add_argument("--execution-variant", choices=("dflash",), + help="Explicit Qwen35 paired execution mode") + parser.add_argument("--companion", action="append", default=[], metavar="ROLE=LOCAL_DIR", + help="Explicit local draft checkpoint; exactly one draft role is required") + + +def _execution_inputs(args: argparse.Namespace) -> BuildExecutionInputs | None: + """Parse only explicit local inputs; no variant list or model acquisition.""" + if args.command != "build": + return None + if args.execution_variant is None: + if args.companion: + raise ValueError("--companion requires --execution-variant") + return None + checkpoints = [] + for value in args.companion: + role, separator, directory = value.partition("=") + if not separator or not role or not directory: + raise ValueError("--companion must be ROLE=LOCAL_DIR") + if "://" in directory: + raise ValueError("--companion requires a local directory, not a URI") + checkpoints.append(NamedCheckpoint(role, Path(directory))) + return BuildExecutionInputs(args.execution_variant, tuple(checkpoints)) + + +def prepare_build_request(request, args): + """Attach a validated family-owned recipe before importing the GPU builder.""" + execution = _execution_inputs(args) + return request if execution is None else with_execution(request, execution) diff --git a/families/qwen3_5/edge_llm/config.py b/families/qwen3_5/edge_llm/config.py new file mode 100644 index 0000000000..e697330ba3 --- /dev/null +++ b/families/qwen3_5/edge_llm/config.py @@ -0,0 +1,88 @@ +# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-License-Identifier: Apache-2.0 + +"""Explicit Qwen35 paired-build inputs; no shared execution-mode contract.""" + +from dataclasses import dataclass, fields +from pathlib import Path +import re + +from tensorrt_model_connect.build import BuildRequest + + +_ID = re.compile(r"[a-z][a-z0-9_]*\Z") + + +def _validate_id(field: str, value: object) -> str: + if not isinstance(value, str) or _ID.fullmatch(value) is None: + raise ValueError(f"{field} must be a lowercase identifier containing only letters, digits, and underscores") + return value + + +@dataclass(frozen=True) +class NamedCheckpoint: + """One explicitly named local checkpoint; the family owns role semantics.""" + + role: str + model_dir: Path + + def __post_init__(self) -> None: + _validate_id("checkpoint role", self.role) + if not isinstance(self.model_dir, Path): + raise TypeError("checkpoint model_dir must be a Path") + if not self.model_dir.is_dir(): + raise ValueError(f"checkpoint must be an existing local directory: {self.model_dir}") + + +@dataclass(frozen=True) +class BuildExecutionInputs: + """Optional family-owned execution variant and immutable local companions. + + Qwen35 validates these inputs without fetching or inferring companions. + """ + + variant: str + checkpoints: tuple[NamedCheckpoint, ...] = () + + def __post_init__(self) -> None: + _validate_id("execution variant", self.variant) + if not isinstance(self.checkpoints, tuple) or any( + not isinstance(checkpoint, NamedCheckpoint) for checkpoint in self.checkpoints + ): + raise TypeError("checkpoints must be a tuple of NamedCheckpoint values") + roles = [checkpoint.role for checkpoint in self.checkpoints] + if len(roles) != len(set(roles)): + raise ValueError("checkpoint roles must be unique") + self.validate_local() + + def validate_local(self) -> None: + """Recheck local availability before dispatch, without acquiring inputs.""" + for checkpoint in self.checkpoints: + if not checkpoint.model_dir.is_dir(): + raise ValueError( + f"checkpoint must be an existing local directory: {checkpoint.model_dir}" + ) + + +@dataclass(frozen=True) +class Qwen35BuildRequest(BuildRequest): + """Ordinary build inputs plus an explicitly requested Qwen35 execution recipe.""" + + execution: BuildExecutionInputs | None = None + + def __post_init__(self) -> None: + super().__post_init__() + if self.family != "qwen3_5": + raise ValueError("Qwen35BuildRequest requires the qwen3_5 family") + if self.execution is not None: + if not isinstance(self.execution, BuildExecutionInputs): + raise TypeError("execution must be BuildExecutionInputs") + self.execution.validate_local() + + +def with_execution(request: BuildRequest, execution: BuildExecutionInputs) -> Qwen35BuildRequest: + """Preserve ordinary request fields and callback identity.""" + return Qwen35BuildRequest( + **{field.name: getattr(request, field.name) for field in fields(BuildRequest)}, + execution=execution, + ) diff --git a/families/qwen3_5/dispatch.py b/families/qwen3_5/edge_llm/dispatch.py similarity index 64% rename from families/qwen3_5/dispatch.py rename to families/qwen3_5/edge_llm/dispatch.py index 9f0d01face..06eee8abc7 100644 --- a/families/qwen3_5/dispatch.py +++ b/families/qwen3_5/edge_llm/dispatch.py @@ -12,7 +12,7 @@ import tempfile import traceback -from . import edge_llm +from . import builder as edge_llm _LOG = logging.getLogger(__name__) @@ -141,3 +141,57 @@ def build(request, writer, native, *, draft_dir: Path | None = None) -> None: if failure is not None: raise error from failure raise + + +def build_paired(request, writer, execution) -> None: + """Build an explicit Qwen3.5 DFlash pair, never a base-only replacement.""" + execution.validate_local() + + if execution.variant != "dflash" or tuple(x.role for x in execution.checkpoints) != ("draft",): + raise ValueError( + "Qwen3.5 paired execution requires variant=dflash and one draft checkpoint" + ) + draft_dir = execution.checkpoints[0].model_dir + raw = json.loads((request.model_dir / "config.json").read_text()) + draft = json.loads((draft_dir / "config.json").read_text()) + if not candidate(request, raw): + raise ValueError("Qwen3.5 DFlash requires an admitted dense base request") + base = raw.get("text_config", raw) + if (base.get("hidden_size"), base.get("num_hidden_layers")) not in {(2560, 32), (4096, 32)}: + raise ValueError("Qwen3.5 DFlash is qualified only for the recorded 4B/9B base profiles") + if not isinstance(draft, dict) or draft.get("architectures") != ["DFlashDraftModel"]: + raise ValueError("Expected a DFlashDraftModel companion") + for name in ("hidden_size", "vocab_size"): + if type(draft.get(name)) is not int or draft[name] != base.get(name): + raise ValueError(f"Qwen3.5 DFlash base and draft disagree on {name}") + if draft.get("num_target_layers") != base.get("num_hidden_layers"): + raise ValueError("Qwen3.5 DFlash target layer count differs from base") + config = draft.get("dflash_config") + if not isinstance(config, dict) or config.get("block_size") != 16: + raise ValueError("Qwen3.5 DFlash currently maps the upstream linear block16 profile") + layers = config.get("target_layer_ids") + if ( + not isinstance(layers, list) + or not layers + or any(type(i) is not int or not 0 <= i < base["num_hidden_layers"] for i in layers) + or len(set(layers)) != len(layers) + ): + raise ValueError("Invalid Qwen3.5 DFlash target layer IDs") + mask = config.get("mask_token_id") + if type(mask) is not int or not 0 <= mask < base["vocab_size"]: + raise ValueError("Invalid Qwen3.5 DFlash mask token") + capacity = draft.get("max_position_embeddings") + limit = request.max_sequence_length or min(base["max_position_embeddings"], 256) + if type(capacity) is not int or not 16 < limit <= capacity: + raise ValueError("Requested context exceeds DFlash draft capacity or block minimum") + if draft.get("quantization_config") or any( + (draft_dir / name).exists() + for name in ("hf_quant_config.json", "quantize_config.json", "quant_config.json") + ): + raise ValueError("This Qwen3.5 DFlash profile requires unquantized draft weights") + + def native_pair(original_request, original_writer): + # Preserve the requested variant on fallback; native has no DFlash decoder. + raise NotImplementedError("Native Qwen3.5 does not implement the requested DFlash variant") + + build(request, writer, native_pair, draft_dir=draft_dir) diff --git a/families/qwen3_5/model.py b/families/qwen3_5/model.py index 32ea38e79f..a2b94e4053 100644 --- a/families/qwen3_5/model.py +++ b/families/qwen3_5/model.py @@ -1362,7 +1362,14 @@ def _runtime_config(model_dir: Path, config: ModelConfig, model: _Qwen35Model, * def build(request: "BuildRequest", writer: "BundleWriter") -> None: """Select complete-network offload or preserve the native Qwen3.5 builder.""" - from .dispatch import build as dispatch_build + + from .edge_llm.config import Qwen35BuildRequest + from .edge_llm.dispatch import build_paired + + if isinstance(request, Qwen35BuildRequest) and request.execution is not None: + build_paired(request, writer, request.execution) + return + from .edge_llm.dispatch import build as dispatch_build def _build_native(request: "BuildRequest", writer: "BundleWriter") -> None: """Build one Qwen3.5 hybrid bundle through family-owned code only.""" @@ -1438,57 +1445,3 @@ def _build_native(request: "BuildRequest", writer: "BundleWriter") -> None: writer.add_bytes(filename, path.read_bytes()) dispatch_build(request, writer, _build_native) - - -def build_with_inputs(request, writer, execution) -> None: - """Build an explicit Qwen3.5 DFlash pair, never a base-only replacement.""" - from . import dispatch - - if execution.variant != "dflash" or tuple(x.role for x in execution.checkpoints) != ("draft",): - raise ValueError( - "Qwen3.5 paired execution requires variant=dflash and one draft checkpoint" - ) - draft_dir = execution.checkpoints[0].model_dir - raw = json.loads((request.model_dir / "config.json").read_text()) - draft = json.loads((draft_dir / "config.json").read_text()) - if not dispatch.candidate(request, raw): - raise ValueError("Qwen3.5 DFlash requires an admitted dense base request") - base = raw.get("text_config", raw) - if (base.get("hidden_size"), base.get("num_hidden_layers")) not in {(2560, 32), (4096, 32)}: - raise ValueError("Qwen3.5 DFlash is qualified only for the recorded 4B/9B base profiles") - if not isinstance(draft, dict) or draft.get("architectures") != ["DFlashDraftModel"]: - raise ValueError("Expected a DFlashDraftModel companion") - for name in ("hidden_size", "vocab_size"): - if type(draft.get(name)) is not int or draft[name] != base.get(name): - raise ValueError(f"Qwen3.5 DFlash base and draft disagree on {name}") - if draft.get("num_target_layers") != base.get("num_hidden_layers"): - raise ValueError("Qwen3.5 DFlash target layer count differs from base") - config = draft.get("dflash_config") - if not isinstance(config, dict) or config.get("block_size") != 16: - raise ValueError("Qwen3.5 DFlash currently maps the upstream linear block16 profile") - layers = config.get("target_layer_ids") - if ( - not isinstance(layers, list) - or not layers - or any(type(i) is not int or not 0 <= i < base["num_hidden_layers"] for i in layers) - or len(set(layers)) != len(layers) - ): - raise ValueError("Invalid Qwen3.5 DFlash target layer IDs") - mask = config.get("mask_token_id") - if type(mask) is not int or not 0 <= mask < base["vocab_size"]: - raise ValueError("Invalid Qwen3.5 DFlash mask token") - capacity = draft.get("max_position_embeddings") - limit = request.max_sequence_length or min(base["max_position_embeddings"], 256) - if type(capacity) is not int or not 16 < limit <= capacity: - raise ValueError("Requested context exceeds DFlash draft capacity or block minimum") - if draft.get("quantization_config") or any( - (draft_dir / name).exists() - for name in ("hf_quant_config.json", "quantize_config.json", "quant_config.json") - ): - raise ValueError("This Qwen3.5 DFlash profile requires unquantized draft weights") - - def native_pair(original_request, original_writer): - # Preserve the requested variant on fallback; native has no DFlash decoder. - raise NotImplementedError("Native Qwen3.5 does not implement the requested DFlash variant") - - dispatch.build(request, writer, native_pair, draft_dir=draft_dir) diff --git a/families/qwen3_5/runtime/CMakeLists.txt b/families/qwen3_5/runtime/CMakeLists.txt index 8012ee0f97..37be57e544 100644 --- a/families/qwen3_5/runtime/CMakeLists.txt +++ b/families/qwen3_5/runtime/CMakeLists.txt @@ -56,25 +56,4 @@ if(TRTMC_BUILD_TESTS) add_test(NAME ${test_name} COMMAND ${test_name}) endif() -# Complete-network offload is family-owned and absent from native-only builds. -if(TARGET EdgeLLM::Core) - target_sources(trtmc_model_qwen3_5 PRIVATE - edge_llm/adapter.cpp - edge_llm/device_link.cu - ) - target_compile_definitions(trtmc_model_qwen3_5 PRIVATE TRTMC_HAS_EDGE_LLM=1) - target_link_libraries(trtmc_model_qwen3_5 PRIVATE EdgeLLM::Core) - set_target_properties(trtmc_model_qwen3_5 PROPERTIES - CUDA_ARCHITECTURES "${EdgeLLM_CUDA_ARCHITECTURE}" - CUDA_SEPARABLE_COMPILATION ON - CUDA_RESOLVE_DEVICE_SYMBOLS ON - ) -endif() - -if(TARGET EdgeLLM::Core) - add_custom_command(TARGET trtmc_model_qwen3_5 POST_BUILD - COMMAND ${CMAKE_COMMAND} -E copy_if_different - $ $ - VERBATIM - ) -endif() +include(edge_llm/Adapter.cmake) diff --git a/families/qwen3_5/runtime/edge_llm/Adapter.cmake b/families/qwen3_5/runtime/edge_llm/Adapter.cmake new file mode 100644 index 0000000000..b1479727ba --- /dev/null +++ b/families/qwen3_5/runtime/edge_llm/Adapter.cmake @@ -0,0 +1,28 @@ +# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-License-Identifier: Apache-2.0 + +# Complete-network offload is family-owned and absent from native-only builds. +if(TARGET EdgeLLM::Core) + if(NOT TARGET EdgeLLM::Plugin) + message(FATAL_ERROR "Qwen35 Edge adapter requires the complete EdgeLLM package (Core and Plugin)") + endif() + target_sources(trtmc_model_qwen3_5 PRIVATE + "${CMAKE_CURRENT_LIST_DIR}/adapter.cpp" + "${CMAKE_CURRENT_LIST_DIR}/device_link.cu" + ) + target_compile_definitions(trtmc_model_qwen3_5 PRIVATE TRTMC_HAS_EDGE_LLM=1) + target_link_libraries(trtmc_model_qwen3_5 PRIVATE EdgeLLM::Core) + set_target_properties(trtmc_model_qwen3_5 PROPERTIES + CUDA_ARCHITECTURES "${EdgeLLM_CUDA_ARCHITECTURE}" + CUDA_SEPARABLE_COMPILATION ON + CUDA_RESOLVE_DEVICE_SYMBOLS ON + ) +endif() + +if(TARGET EdgeLLM::Core) + add_custom_command(TARGET trtmc_model_qwen3_5 POST_BUILD + COMMAND ${CMAKE_COMMAND} -E copy_if_different + $ $ + VERBATIM + ) +endif() diff --git a/families/qwen3_5/support.py b/families/qwen3_5/support.py index ff3e7c8d39..05cd5f797c 100644 --- a/families/qwen3_5/support.py +++ b/families/qwen3_5/support.py @@ -10,6 +10,7 @@ model_types=("qwen35", "qwen3.5", "qwen3_5"), tasks=("text_generation",), default_task="text_generation", + build_cli_module="edge_llm.cli", ) diff --git a/families/qwen3_5/tests/test_precision_contract.py b/families/qwen3_5/tests/test_precision_contract.py index 12273e7f42..ab7c13a935 100644 --- a/families/qwen3_5/tests/test_precision_contract.py +++ b/families/qwen3_5/tests/test_precision_contract.py @@ -67,3 +67,136 @@ def add_cast(self, tensor: _Tensor, dtype): assert prepared_ssm == [ssm_state] assert prepared_ssm[0].dtype == trt.float32 assert ssm_state not in network.cast_inputs + + +def _edge_cli_source(tmp_path): + import json + + source = tmp_path / "target" + source.mkdir() + (source / "config.json").write_text(json.dumps({"model_type": "qwen3_5"})) + draft = tmp_path / "draft" + draft.mkdir() + return source, draft + + +def test_edge_cli_uses_ordinary_family_build(tmp_path, monkeypatch): + from tensorrt_model_connect import build_cli + from families.qwen3_5.edge_llm import dispatch + from families.qwen3_5.edge_llm.config import Qwen35BuildRequest + + source, draft = _edge_cli_source(tmp_path) + output = tmp_path / "pair.bundle" + seen = [] + + def paired(request, writer, execution): + assert isinstance(request, Qwen35BuildRequest) + assert request.execution is execution + assert execution.variant == "dflash" + assert [(item.role, item.model_dir) for item in execution.checkpoints] == [("draft", draft)] + seen.append(request) + writer.set_header(family="qwen3_5", task=request.task, backend=request.backend) + writer.add_json("edge-test.json", {"variant": execution.variant}) + + monkeypatch.setattr(dispatch, "build_paired", paired) + assert build_cli.main([ + "build", str(source), "--precision", "fp16", "-o", str(output), + "--execution-variant", "dflash", "--companion", f"draft={draft}", + ]) == 0 + assert len(seen) == 1 + assert output.is_file() + + +@pytest.mark.parametrize("options", [ + ["--companion", "draft=/missing"], + ["--execution-variant", "dflash", "--companion", "missing_separator"], + ["--execution-variant", "dflash", "--companion", "=path"], + ["--execution-variant", "dflash", "--companion", "draft="], + ["--execution-variant", "dflash", "--companion", "draft=https://example.com/model"], +]) +def test_bad_edge_cli_inputs_fail_before_backend(tmp_path, monkeypatch, options): + import importlib + from tensorrt_model_connect import build_cli + + core = importlib.import_module("tensorrt_model_connect.build") + source, _ = _edge_cli_source(tmp_path) + monkeypatch.setattr(core, "_select_backend", lambda *_: pytest.fail("backend touched")) + monkeypatch.setattr(core, "BundleWriter", lambda *_: pytest.fail("writer created")) + with pytest.raises(ValueError): + build_cli.main(["build", str(source), "-o", str(tmp_path / "out"), *options]) + + +def test_edge_cli_help_is_family_owned(tmp_path, capsys): + from tensorrt_model_connect import build_cli + + source, _ = _edge_cli_source(tmp_path) + with pytest.raises(SystemExit) as caught: + build_cli.main(["build", str(source), "--help"]) + assert caught.value.code == 0 + help_text = capsys.readouterr().out + assert "--execution-variant {dflash}" in help_text + assert "--companion" in help_text + + +def test_edge_request_preserves_fields_and_family_owner(tmp_path): + import argparse + from families.qwen3_5.edge_llm import cli + from dataclasses import fields, replace + from tensorrt_model_connect.build import BuildRequest + from families.qwen3_5.edge_llm.config import ( + BuildExecutionInputs, NamedCheckpoint, with_execution, + ) + + source, draft = _edge_cli_source(tmp_path) + request = BuildRequest(source, tmp_path / "out", "qwen3_5", "text_generation", "fp16", + graph_transform=lambda layer: layer) + execution = BuildExecutionInputs("dflash", (NamedCheckpoint("draft", draft),)) + args = argparse.Namespace(command="build", execution_variant=None, companion=[]) + assert cli.prepare_build_request(request, args) is request + extended = with_execution(request, execution) + for field in fields(BuildRequest): + assert getattr(extended, field.name) is getattr(request, field.name) + with pytest.raises(ValueError, match="requires the qwen3_5 family"): + with_execution(replace(request, family="other"), execution) + with pytest.raises(ValueError, match="unique"): + BuildExecutionInputs("dflash", execution.checkpoints * 2) + + +@pytest.mark.parametrize("failure", [RuntimeError("paired build failed"), KeyboardInterrupt()]) +def test_edge_cli_failure_preserves_existing_bundle(tmp_path, monkeypatch, failure): + from tensorrt_model_connect import build_cli + from families.qwen3_5.edge_llm import dispatch + + source, draft = _edge_cli_source(tmp_path) + output = tmp_path / "pair.bundle" + output.write_bytes(b"previous publication") + + def fail(request, writer, execution): + writer.set_header(family="qwen3_5", task=request.task, backend=request.backend) + writer.add_json("edge-test.json", {"variant": execution.variant}) + raise failure + + monkeypatch.setattr(dispatch, "build_paired", fail) + with pytest.raises(type(failure)) as caught: + build_cli.main([ + "build", str(source), "-o", str(output), "--execution-variant", "dflash", + "--companion", f"draft={draft}", + ]) + assert caught.value is failure + assert output.read_bytes() == b"previous publication" + assert sorted(item.name for item in tmp_path.iterdir()) == ["draft", "pair.bundle", "target"] + + +def test_edge_pair_requires_draft_and_rechecks_local_inputs(tmp_path): + from tensorrt_model_connect.build import BuildRequest + from families.qwen3_5.edge_llm.config import BuildExecutionInputs, NamedCheckpoint + from families.qwen3_5.edge_llm.dispatch import build_paired + + source, draft = _edge_cli_source(tmp_path) + request = BuildRequest(source, tmp_path / "out", "qwen3_5", "text_generation", "fp16") + with pytest.raises(ValueError, match="paired execution requires"): + build_paired(request, None, BuildExecutionInputs("dflash")) + execution = BuildExecutionInputs("dflash", (NamedCheckpoint("draft", draft),)) + draft.rmdir() + with pytest.raises(ValueError, match="existing local directory"): + build_paired(request, None, execution) diff --git a/website/docs/features/model-families.md b/website/docs/features/model-families.md index 6f4c9e1cbc..fe8f3e66c7 100644 --- a/website/docs/features/model-families.md +++ b/website/docs/features/model-families.md @@ -148,5 +148,5 @@ Edge-LLM 0.10.1 SDK on native x86 SM80, FP16, batch one. Historical local build/public/direct/HF receipts are documented separately from current-head checks. Qwen3 or older, MoE, 27B, quantized sources, and other platforms are not qualified by this route. Existing quality gates are unchanged. -See the [family recipe](https://github.com/NVIDIA/TensorRT-Model-Connect/blob/main/families/qwen3_5/EDGE_LLM.md) +See the [family recipe](https://github.com/NVIDIA/TensorRT-Model-Connect/blob/main/families/qwen3_5/edge_llm/README.md) for exact revisions, paired execution, and replay gaps. From 88db85a204fe191714564881c7f2eb2a8410e47d Mon Sep 17 00:00:00 2001 From: Joshua Calafato Date: Wed, 23 Sep 2026 15:53:51 +0000 Subject: [PATCH 4/8] fix(qwen3_5): preserve optional SDK fallback Keep Edge routing family-owned, avoid failure diagnostics for an absent optional SDK, and retain errors for broken installations. Use output-local staging and existing regression coverage. Signed-off-by: Joshua Calafato --- families/qwen3_5/edge_llm/README.md | 7 +- families/qwen3_5/edge_llm/builder.py | 9 +++ families/qwen3_5/edge_llm/dispatch.py | 7 +- .../qwen3_5/tests/test_precision_contract.py | 76 +++++++++++++++++++ 4 files changed, 97 insertions(+), 2 deletions(-) diff --git a/families/qwen3_5/edge_llm/README.md b/families/qwen3_5/edge_llm/README.md index be185c9261..876fbcda30 100644 --- a/families/qwen3_5/edge_llm/README.md +++ b/families/qwen3_5/edge_llm/README.md @@ -105,7 +105,7 @@ trtmc build /path/to/target --family qwen3_5 --precision fp16 \ -o model.bundle ``` -Put MODEL immediately after `build`. `trtmc build /path/to/target --help` +Put MODEL before family-specific options; known core options may precede MODEL. `trtmc build /path/to/target --help` shows these family options using local metadata; remote-ID help does not download a checkpoint. For Python callers, use this family's request extension: @@ -123,3 +123,8 @@ build(with_execution(request, BuildExecutionInputs( A failed explicit pair is never replaced by a base-only bundle. Previously recorded full-model results above are historical, not fresh refactor-head E2Es. + +Ordinary builds without an installed optional Edge SDK select native without a +warning. Malformed or incomplete installed packages still retain diagnostics and +warn before native fallback. Temporary checkpoint/engine staging uses the bundle +output directory filesystem (choose a scratch-backed output), not system /tmp. diff --git a/families/qwen3_5/edge_llm/builder.py b/families/qwen3_5/edge_llm/builder.py index 100364d2a1..0e906049ad 100644 --- a/families/qwen3_5/edge_llm/builder.py +++ b/families/qwen3_5/edge_llm/builder.py @@ -21,6 +21,15 @@ def local_target() -> dict: return detect_local_platform() +def package_present() -> bool: + """Absence of the optional SDK is a non-match, not a failed Edge build.""" + for prefix in cmake_prefixes(): + manifest = prefix / "share/trtmc/edge-llm.json" + if manifest.exists() or manifest.is_symlink(): + return True + return False + + def installed_package(target: dict) -> dict: """Resolve CMake installation via standard prefixes; never install anything. diff --git a/families/qwen3_5/edge_llm/dispatch.py b/families/qwen3_5/edge_llm/dispatch.py index 06eee8abc7..96d490e9c4 100644 --- a/families/qwen3_5/edge_llm/dispatch.py +++ b/families/qwen3_5/edge_llm/dispatch.py @@ -98,6 +98,9 @@ def build(request, writer, native, *, draft_dir: Path | None = None) -> None: if not candidate(request, raw): native(request, writer) return + if draft_dir is None and not edge_llm.package_present(): + native(request, writer) + return capacity = config.get("max_position_embeddings") if type(capacity) is not int or capacity <= 0: raise ValueError("checkpoint max_position_embeddings must be a positive integer") @@ -109,7 +112,9 @@ def build(request, writer, native, *, draft_dir: Path | None = None) -> None: ) os.close(descriptor) log_path = Path(name) - with tempfile.TemporaryDirectory(prefix="trtmc-qwen3_5-edge-") as directory: + with tempfile.TemporaryDirectory( + prefix=f".{request.output_path.name}.edge-", dir=request.output_path.parent + ) as directory: try: target = edge_llm.local_target() key = (target["os"], target["arch"], target["sm"], request.precision.lower()) diff --git a/families/qwen3_5/tests/test_precision_contract.py b/families/qwen3_5/tests/test_precision_contract.py index ab7c13a935..1db969c816 100644 --- a/families/qwen3_5/tests/test_precision_contract.py +++ b/families/qwen3_5/tests/test_precision_contract.py @@ -200,3 +200,79 @@ def test_edge_pair_requires_draft_and_rechecks_local_inputs(tmp_path): draft.rmdir() with pytest.raises(ValueError, match="existing local directory"): build_paired(request, None, execution) + + +@pytest.mark.parametrize("mode", ["absent", "success", "corrupt", "failure", "cancel", "device_failure"]) +def test_edge_optional_package_and_output_local_staging(tmp_path, monkeypatch, caplog, mode): + import json + from tensorrt_model_connect.build import BuildRequest + from families.qwen3_5.edge_llm import builder, dispatch + + source = tmp_path / "target" + source.mkdir() + (source / "config.json").write_text(json.dumps({ + "max_position_embeddings": 4096, "hidden_size": 896, + })) + prefix = tmp_path / "package" + manifest = prefix / "share/trtmc/edge-llm.json" + if mode != "absent": + manifest.parent.mkdir(parents=True) + manifest.write_text("{" if mode == "corrupt" else "{}") + monkeypatch.setattr(builder, "cmake_prefixes", lambda: [prefix]) + monkeypatch.setattr(dispatch, "candidate", lambda *_: True) + for name in ("mapped_request", "request_matches", "platform_matches"): + if hasattr(dispatch, name): + monkeypatch.setattr(dispatch, name, lambda *_: True) + if hasattr(dispatch, "source_quantization"): + monkeypatch.setattr(dispatch, "source_quantization", lambda *_: "fp16") + if hasattr(builder, "request_weight_format"): + monkeypatch.setattr(builder, "request_weight_format", lambda *_: "fp16") + request = BuildRequest(source, tmp_path / "out", "qwen3_5", "text_generation", "fp16") + writer = object() + target = {"os": "linux", "arch": "x86_64", "sm": 80} + target_calls = [] + + def local_target(): + target_calls.append(True) + if mode == "device_failure": + raise RuntimeError("CUDA discovery failed") + return target + + monkeypatch.setattr(builder, "local_target", local_target) + stages, native_calls, publications = [], [], [] + + def prepare(original, raw, platform, staging, log): + assert original is request and platform is target + assert staging.parent == request.output_path.parent + assert staging.name.startswith(f".{request.output_path.name}.edge-") + stages.append(staging) + (staging / "large-checkpoint").write_bytes(b"fixture") + if mode == "corrupt": + builder.installed_package(target) + if mode == "failure": + raise FileNotFoundError("installed SDK artifact missing") + if mode == "cancel": + raise KeyboardInterrupt() + return {}, {} + + monkeypatch.setattr(dispatch, "EDGE_DISPATCH", {("linux", "x86_64", 80, "fp16"): prepare}) + monkeypatch.setattr(builder, "publish", lambda *args: publications.append(args)) + def native(*args): + native_calls.append(args) + if mode == "cancel": + with pytest.raises(KeyboardInterrupt): + dispatch.build(request, writer, native) + else: + dispatch.build(request, writer, native) + assert all(not path.exists() for path in stages) + assert native_calls == ([(request, writer)] if mode in { + "absent", "corrupt", "failure", "device_failure", + } else []) + assert len(publications) == (1 if mode == "success" else 0) + assert len(target_calls) == (0 if mode == "absent" else 1) + logs = list(tmp_path.glob(".out.edge-*.log")) + if mode in {"corrupt", "failure", "device_failure"}: + assert len(logs) == 1 and "Traceback" in logs[0].read_text() + assert "Retrying native once" in caplog.text + elif mode != "cancel": + assert not logs and "Edge build failed" not in caplog.text From 3b08465fdac7547fa4d8633d41825c2c6254da1a Mon Sep 17 00:00:00 2001 From: Joshua Calafato Date: Wed, 23 Sep 2026 19:46:58 +0000 Subject: [PATCH 5/8] refactor(qwen3_5): use declared family CLI Reuse the existing family CLI protocol instead of extending the shared parser. Own the command description, request contract and build lifecycle; keep Edge companion semantics inside this family. Preserve legacy callers through strict conversion and keep numerical acceptance gates unchanged. Signed-off-by: Joshua Calafato --- families/qwen3_5/build_request.py | 95 +++++++++++++ families/qwen3_5/cli.json | 129 ++++++++++++++++++ families/qwen3_5/cli.py | 59 ++++++++ families/qwen3_5/edge_llm/README.md | 15 +- families/qwen3_5/edge_llm/cli.py | 33 ++--- families/qwen3_5/edge_llm/config.py | 5 +- families/qwen3_5/model.py | 5 +- families/qwen3_5/support.py | 1 - .../qwen3_5/tests/test_precision_contract.py | 78 +++++++++-- website/docs/features/model-families.md | 5 + 10 files changed, 383 insertions(+), 42 deletions(-) create mode 100644 families/qwen3_5/build_request.py create mode 100644 families/qwen3_5/cli.json create mode 100644 families/qwen3_5/cli.py diff --git a/families/qwen3_5/build_request.py b/families/qwen3_5/build_request.py new file mode 100644 index 0000000000..c02a360dc1 --- /dev/null +++ b/families/qwen3_5/build_request.py @@ -0,0 +1,95 @@ +# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-License-Identifier: Apache-2.0 + +"""qwen3_5 build inputs and strict compatibility for existing Python callers.""" + +from __future__ import annotations + +from dataclasses import dataclass, fields +from pathlib import Path +import re +from typing import ClassVar + +from tensorrt_model_connect.graph_transform import GraphTransform + + +def _validate_id(field: str, value: str) -> None: + if not isinstance(value, str) or re.fullmatch(r"[a-z][a-z0-9_]*", value) is None: + raise ValueError(f"{field} must be a lowercase identifier") + + +@dataclass(frozen=True) +class BuildRequest: + """qwen3_5-owned inputs; unsupported legacy controls are read-only defaults.""" + + model_dir: Path + output_path: Path + family: str + task: str + precision: str + backend: str = "trt" + max_sequence_length: int | None = None + image_height: ClassVar[int | None] = None + image_width: ClassVar[int | None] = None + video_num_frames: ClassVar[int | None] = None + max_batch_size: ClassVar[int] = 1 + tensor_parallel_size: int = 1 + context_parallel_size: ClassVar[int] = 1 + quantization: ClassVar[str | None] = None + fp32_layers: tuple[int, ...] = () + dynamic_kv_cache: ClassVar[bool] = False + verbose: bool = False + graph_transform: GraphTransform | None = None + + def __post_init__(self) -> None: + if not self.precision: + raise ValueError("precision must be non-empty") + _validate_id("family", self.family) + _validate_id("task", self.task) + if self.backend not in {"trt", "trt_rtx"}: + raise ValueError("backend must be 'trt' or 'trt_rtx'") + if self.max_sequence_length is not None and self.max_sequence_length < 1: + raise ValueError("max_sequence_length must be positive") + for field in ("image_height", "image_width", "video_num_frames"): + value = getattr(self, field) + if value is not None and value < 1: + raise ValueError(f"{field} must be positive") + if self.max_batch_size < 1: + raise ValueError("max_batch_size must be positive") + if self.tensor_parallel_size < 1: + raise ValueError("tensor_parallel_size must be positive") + if self.context_parallel_size < 1: + raise ValueError("context_parallel_size must be positive") + if self.quantization is not None and not self.quantization: + raise ValueError("quantization must be non-empty when provided") + if any(layer < 0 for layer in self.fp32_layers): + raise ValueError("fp32_layers must contain non-negative indices") + if not isinstance(self.dynamic_kv_cache, bool): + raise ValueError("dynamic_kv_cache must be a bool") + if self.graph_transform is not None and not callable(self.graph_transform): + raise ValueError("graph_transform must be callable when provided") + + +def coerce_request(request: object) -> BuildRequest: + """Reject unsupported/unknown legacy inputs before converting to owner fields.""" + if isinstance(request, BuildRequest): + return request + unsupported = { + "image_height": None, + "image_width": None, + "video_num_frames": None, + "max_batch_size": 1, + "context_parallel_size": 1, + "quantization": None, + "dynamic_kv_cache": False, + } + for name, default in unsupported.items(): + value = getattr(request, name, default) + if name == "quantization" and value == "none": + continue + if value != default: + raise NotImplementedError(f"qwen3_5 does not support {name}") + names = {field.name for field in fields(BuildRequest)} + if unknown := set(vars(request)) - names - set(unsupported): + raise ValueError(f"unknown qwen3_5 build inputs: {sorted(unknown)}") + return BuildRequest(**{name: getattr(request, name) for name in names}) diff --git a/families/qwen3_5/cli.json b/families/qwen3_5/cli.json new file mode 100644 index 0000000000..6d616ec0ac --- /dev/null +++ b/families/qwen3_5/cli.json @@ -0,0 +1,129 @@ +{ + "version": 1, + "commands": [ + { + "name": "build", + "help": "Build one qwen3_5 TensorRT bundle", + "executor": "python", + "handler": "cli:build", + "arguments": [ + { + "name": "model", + "type": "string", + "help": "Hugging Face model ID or local snapshot" + }, + { + "name": "output", + "flags": [ + "-o", + "--output" + ], + "type": "path", + "required": true + }, + { + "name": "revision", + "flags": [ + "--revision" + ], + "type": "string" + }, + { + "name": "task", + "flags": [ + "--task" + ], + "type": "string", + "choices": [ + "text_generation" + ], + "default": "text_generation" + }, + { + "name": "precision", + "flags": [ + "--precision" + ], + "type": "string", + "choices": [ + "fp32", + "fp16" + ], + "default": "fp32" + }, + { + "name": "backend", + "flags": [ + "--backend" + ], + "type": "string", + "choices": [ + "trt", + "trt_rtx" + ], + "default": "trt" + }, + { + "name": "max_sequence_length", + "flags": [ + "--max-sequence-length" + ], + "type": "int" + }, + { + "name": "tensor_parallel_size", + "flags": [ + "--tensor-parallel-size" + ], + "type": "int", + "choices": [ + 1, + 2, + 4, + 8 + ], + "default": 1 + }, + { + "name": "verbose", + "flags": [ + "--verbose" + ], + "type": "bool", + "action": "store_true", + "default": false + }, + { + "name": "fp32_layers", + "flags": [ + "--fp32-layer" + ], + "type": "int", + "action": "append", + "default": [] + }, + { + "name": "execution_variant", + "flags": [ + "--execution-variant" + ], + "type": "string", + "choices": [ + "dflash" + ], + "help": "Explicit paired execution mode; no automatic draft discovery" + }, + { + "name": "companion", + "flags": [ + "--companion" + ], + "type": "string", + "action": "append", + "default": [], + "help": "Local companion checkpoint as ROLE=LOCAL_DIR" + } + ] + } + ] +} diff --git a/families/qwen3_5/cli.py b/families/qwen3_5/cli.py new file mode 100644 index 0000000000..b46ba93832 --- /dev/null +++ b/families/qwen3_5/cli.py @@ -0,0 +1,59 @@ +# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-License-Identifier: Apache-2.0 + +"""qwen3_5-owned build command and typed inputs; importing this module is CPU-only.""" + +from __future__ import annotations + +from dataclasses import replace +from pathlib import Path + +from tensorrt_model_connect.build import select_backend +from tensorrt_model_connect.bundle_writer import BundleWriter +from tensorrt_model_connect.graph_transform import graph_transform +from tensorrt_model_connect.model_support import load_model_metadata, resolve_family, resolve_model + +from .build_request import BuildRequest + +from .edge_llm.cli import execution_inputs +from .edge_llm.config import with_execution + + +def build_bundle(request: BuildRequest, output: Path) -> None: + """Build and publish through the owning model, preserving atomic failure.""" + request = replace(request, output_path=output) + select_backend(request.backend) + from .model import build as build_model + + writer = BundleWriter(output) + try: + with graph_transform(request.graph_transform): + build_model(request, writer) + writer.finish() + except BaseException: + writer.abort() + raise + +def build( + *, model: str, output: Path, revision: str | None = None, + task: str = "text_generation", precision: str = "fp32", backend: str = "trt", + max_sequence_length: int | None = None, tensor_parallel_size: int = 1, + verbose: bool = False, + fp32_layers: list[int] | tuple[int, ...] = (), + execution_variant: str | None = None, companion: list[str] | tuple[str, ...] = (), +) -> int: + """Run the declared owner command; help never imports this handler.""" + execution = execution_inputs(execution_variant, companion) + model_dir = resolve_model(model, revision) + resolve_family(load_model_metadata(model_dir), "qwen3_5") + request = BuildRequest( + model_dir=model_dir, output_path=output, family="qwen3_5", + task=task, precision=precision, backend=backend, + max_sequence_length=max_sequence_length, tensor_parallel_size=tensor_parallel_size, + verbose=verbose, + fp32_layers=tuple(fp32_layers), + ) + if execution is not None: + request = with_execution(request, execution) + build_bundle(request, output) + return 0 diff --git a/families/qwen3_5/edge_llm/README.md b/families/qwen3_5/edge_llm/README.md index 876fbcda30..7ecccc09ae 100644 --- a/families/qwen3_5/edge_llm/README.md +++ b/families/qwen3_5/edge_llm/README.md @@ -95,17 +95,18 @@ basic workload. ## Family-owned build options -The generic CLI loads this family's `edge_llm.cli` hook only after resolving -the model. The shared build API has no execution-mode or companion arguments. +The existing family CLI reads this owner's cli.json and invokes cli.py. +Edge-specific inputs and selection remain in edge_llm/; the shared parser, +CLI protocol and build API gain no new options or hooks. All variant validation and builder selection remain in this family. ```sh -trtmc build /path/to/target --family qwen3_5 --precision fp16 \ +trtmc qwen3_5 build /path/to/target --precision fp16 \ --execution-variant dflash --companion draft=/path/to/draft \ -o model.bundle ``` -Put MODEL before family-specific options; known core options may precede MODEL. `trtmc build /path/to/target --help` +Options may precede or follow MODEL. `trtmc qwen3_5 build /path/to/target --help` shows these family options using local metadata; remote-ID help does not download a checkpoint. For Python callers, use this family's request extension: @@ -128,3 +129,9 @@ Ordinary builds without an installed optional Edge SDK select native without a warning. Malformed or incomplete installed packages still retain diagnostics and warn before native fallback. Temporary checkpoint/engine staging uses the bundle output directory filesystem (choose a scratch-backed output), not system /tmp. + +## Declared build command + +This family uses the existing cli.json protocol introduced in #1310. The family\nowns its declaration, typed inputs and Python handler. The handler adapts those\ninputs to the unchanged builder API, preserving native/Edge dispatch and bundle\npublication. The legacy flat build command remains available for its existing\nordinary options; new family options use `trtmc qwen3_5 build`. +Help is offline and does not need a local checkpoint. No shared parser hook or +family registry entry is added. diff --git a/families/qwen3_5/edge_llm/cli.py b/families/qwen3_5/edge_llm/cli.py index 544932b751..373a51ce87 100644 --- a/families/qwen3_5/edge_llm/cli.py +++ b/families/qwen3_5/edge_llm/cli.py @@ -3,40 +3,27 @@ """Qwen35-owned CLI options for explicit paired Edge execution.""" -import argparse from pathlib import Path -from .config import BuildExecutionInputs, NamedCheckpoint, with_execution +from .config import BuildExecutionInputs, NamedCheckpoint -def add_build_arguments(parser: argparse.ArgumentParser) -> None: - """Register options only when the Qwen35 family has been resolved.""" - parser.add_argument("--execution-variant", choices=("dflash",), - help="Explicit Qwen35 paired execution mode") - parser.add_argument("--companion", action="append", default=[], metavar="ROLE=LOCAL_DIR", - help="Explicit local draft checkpoint; exactly one draft role is required") - - -def _execution_inputs(args: argparse.Namespace) -> BuildExecutionInputs | None: +def execution_inputs( + execution_variant: str | None, companion: list[str] | tuple[str, ...] = (), +) -> BuildExecutionInputs | None: """Parse only explicit local inputs; no variant list or model acquisition.""" - if args.command != "build": - return None - if args.execution_variant is None: - if args.companion: + if execution_variant is None: + if companion: raise ValueError("--companion requires --execution-variant") return None + if execution_variant not in ['dflash']: + raise ValueError("unsupported qwen3_5 execution variant") checkpoints = [] - for value in args.companion: + for value in companion: role, separator, directory = value.partition("=") if not separator or not role or not directory: raise ValueError("--companion must be ROLE=LOCAL_DIR") if "://" in directory: raise ValueError("--companion requires a local directory, not a URI") checkpoints.append(NamedCheckpoint(role, Path(directory))) - return BuildExecutionInputs(args.execution_variant, tuple(checkpoints)) - - -def prepare_build_request(request, args): - """Attach a validated family-owned recipe before importing the GPU builder.""" - execution = _execution_inputs(args) - return request if execution is None else with_execution(request, execution) + return BuildExecutionInputs(execution_variant, tuple(checkpoints)) diff --git a/families/qwen3_5/edge_llm/config.py b/families/qwen3_5/edge_llm/config.py index e697330ba3..07dd2c1361 100644 --- a/families/qwen3_5/edge_llm/config.py +++ b/families/qwen3_5/edge_llm/config.py @@ -7,7 +7,7 @@ from pathlib import Path import re -from tensorrt_model_connect.build import BuildRequest +from ..build_request import BuildRequest, coerce_request _ID = re.compile(r"[a-z][a-z0-9_]*\Z") @@ -81,7 +81,8 @@ def __post_init__(self) -> None: def with_execution(request: BuildRequest, execution: BuildExecutionInputs) -> Qwen35BuildRequest: - """Preserve ordinary request fields and callback identity.""" + """Preserve supported ordinary request fields and callback identity.""" + request = coerce_request(request) return Qwen35BuildRequest( **{field.name: getattr(request, field.name) for field in fields(BuildRequest)}, execution=execution, diff --git a/families/qwen3_5/model.py b/families/qwen3_5/model.py index a2b94e4053..59dfaf0e29 100644 --- a/families/qwen3_5/model.py +++ b/families/qwen3_5/model.py @@ -126,7 +126,7 @@ def cast_all(tensors): if TYPE_CHECKING: - from tensorrt_model_connect.build import BuildRequest + from .build_request import BuildRequest from tensorrt_model_connect.bundle_writer import BundleWriter @@ -1362,6 +1362,9 @@ def _runtime_config(model_dir: Path, config: ModelConfig, model: _Qwen35Model, * def build(request: "BuildRequest", writer: "BundleWriter") -> None: """Select complete-network offload or preserve the native Qwen3.5 builder.""" + from .build_request import coerce_request + + request = coerce_request(request) from .edge_llm.config import Qwen35BuildRequest from .edge_llm.dispatch import build_paired diff --git a/families/qwen3_5/support.py b/families/qwen3_5/support.py index 05cd5f797c..ff3e7c8d39 100644 --- a/families/qwen3_5/support.py +++ b/families/qwen3_5/support.py @@ -10,7 +10,6 @@ model_types=("qwen35", "qwen3.5", "qwen3_5"), tasks=("text_generation",), default_task="text_generation", - build_cli_module="edge_llm.cli", ) diff --git a/families/qwen3_5/tests/test_precision_contract.py b/families/qwen3_5/tests/test_precision_contract.py index 1db969c816..ad80ec81eb 100644 --- a/families/qwen3_5/tests/test_precision_contract.py +++ b/families/qwen3_5/tests/test_precision_contract.py @@ -81,7 +81,7 @@ def _edge_cli_source(tmp_path): def test_edge_cli_uses_ordinary_family_build(tmp_path, monkeypatch): - from tensorrt_model_connect import build_cli + from tensorrt_model_connect import family_cli as build_cli from families.qwen3_5.edge_llm import dispatch from families.qwen3_5.edge_llm.config import Qwen35BuildRequest @@ -99,7 +99,7 @@ def paired(request, writer, execution): writer.add_json("edge-test.json", {"variant": execution.variant}) monkeypatch.setattr(dispatch, "build_paired", paired) - assert build_cli.main([ + assert build_cli.main(["qwen3_5", "build", str(source), "--precision", "fp16", "-o", str(output), "--execution-variant", "dflash", "--companion", f"draft={draft}", ]) == 0 @@ -116,22 +116,22 @@ def paired(request, writer, execution): ]) def test_bad_edge_cli_inputs_fail_before_backend(tmp_path, monkeypatch, options): import importlib - from tensorrt_model_connect import build_cli + from tensorrt_model_connect import family_cli as build_cli core = importlib.import_module("tensorrt_model_connect.build") source, _ = _edge_cli_source(tmp_path) monkeypatch.setattr(core, "_select_backend", lambda *_: pytest.fail("backend touched")) monkeypatch.setattr(core, "BundleWriter", lambda *_: pytest.fail("writer created")) with pytest.raises(ValueError): - build_cli.main(["build", str(source), "-o", str(tmp_path / "out"), *options]) + build_cli.main(["qwen3_5", "build", str(source), "-o", str(tmp_path / "out"), *options]) def test_edge_cli_help_is_family_owned(tmp_path, capsys): - from tensorrt_model_connect import build_cli + from tensorrt_model_connect import family_cli as build_cli source, _ = _edge_cli_source(tmp_path) with pytest.raises(SystemExit) as caught: - build_cli.main(["build", str(source), "--help"]) + build_cli.main(["qwen3_5", "build", str(source), "--help"]) assert caught.value.code == 0 help_text = capsys.readouterr().out assert "--execution-variant {dflash}" in help_text @@ -139,7 +139,6 @@ def test_edge_cli_help_is_family_owned(tmp_path, capsys): def test_edge_request_preserves_fields_and_family_owner(tmp_path): - import argparse from families.qwen3_5.edge_llm import cli from dataclasses import fields, replace from tensorrt_model_connect.build import BuildRequest @@ -151,8 +150,7 @@ def test_edge_request_preserves_fields_and_family_owner(tmp_path): request = BuildRequest(source, tmp_path / "out", "qwen3_5", "text_generation", "fp16", graph_transform=lambda layer: layer) execution = BuildExecutionInputs("dflash", (NamedCheckpoint("draft", draft),)) - args = argparse.Namespace(command="build", execution_variant=None, companion=[]) - assert cli.prepare_build_request(request, args) is request + assert cli.execution_inputs(None) is None extended = with_execution(request, execution) for field in fields(BuildRequest): assert getattr(extended, field.name) is getattr(request, field.name) @@ -164,7 +162,7 @@ def test_edge_request_preserves_fields_and_family_owner(tmp_path): @pytest.mark.parametrize("failure", [RuntimeError("paired build failed"), KeyboardInterrupt()]) def test_edge_cli_failure_preserves_existing_bundle(tmp_path, monkeypatch, failure): - from tensorrt_model_connect import build_cli + from tensorrt_model_connect import family_cli as build_cli from families.qwen3_5.edge_llm import dispatch source, draft = _edge_cli_source(tmp_path) @@ -178,7 +176,7 @@ def fail(request, writer, execution): monkeypatch.setattr(dispatch, "build_paired", fail) with pytest.raises(type(failure)) as caught: - build_cli.main([ + build_cli.main(["qwen3_5", "build", str(source), "-o", str(output), "--execution-variant", "dflash", "--companion", f"draft={draft}", ]) @@ -276,3 +274,61 @@ def native(*args): assert "Retrying native once" in caplog.text elif mode != "cancel": assert not logs and "Edge build failed" not in caplog.text + + +@pytest.mark.parametrize("options", [[], ["--precision", "fp16", "--max-sequence-length", "64"]]) +def test_declared_build_matches_legacy_request(tmp_path, monkeypatch, options): + """Owner command preserves ordinary request defaults and explicit controls.""" + import json + from families.qwen3_5 import cli as owner + from tensorrt_model_connect import build_cli, family_cli + + source = tmp_path / "checkpoint" + source.mkdir() + (source / "config.json").write_text(json.dumps({"model_type": "qwen3_5"})) + output = tmp_path / "model.bundle" + captured = [] + monkeypatch.setattr(owner, "build_bundle", lambda request, output: captured.append(request)) + monkeypatch.setattr(build_cli, "build", captured.append) + args = [str(source), "-o", str(output), *options] + assert family_cli.main(["qwen3_5", "build", *args]) == 0 + assert build_cli.main(["build", *args, "--family", "qwen3_5"]) == 0 + assert len(captured) == 2 + from dataclasses import fields + assert isinstance(captured[0], owner.BuildRequest) + for field in fields(captured[1]): + assert getattr(captured[0], field.name) == getattr(captured[1], field.name) + from dataclasses import replace + from families.qwen3_5.build_request import coerce_request + assert coerce_request(captured[1]) == captured[0] + with pytest.raises(NotImplementedError, match="image_height"): + coerce_request(replace(captured[1], image_height=32)) + from types import SimpleNamespace + with pytest.raises(ValueError, match="unknown"): + coerce_request(SimpleNamespace(**vars(captured[1]), unexpected_option=True)) + assert captured[0].family == "qwen3_5" + assert captured[0].task == "text_generation" + assert captured[0].precision == ("fp16" if options else "fp32") + assert not output.exists() + + +def test_declared_help_is_offline_and_dependency_free(): + """Actual child-process help needs neither a checkpoint nor GPU imports.""" + import subprocess + import sys + + code = """ +import sys +from tensorrt_model_connect.family_cli import main +try: + main(["qwen3_5", "build", "--help"]) +except SystemExit as error: + assert error.code == 0 +else: + raise AssertionError("help did not exit") +assert "families.qwen3_5.cli" not in sys.modules +assert "tensorrt" not in sys.modules +assert "huggingface_hub" not in sys.modules +""" + result = subprocess.run([sys.executable, "-c", code], capture_output=True, text=True, check=True) + assert "trtmc qwen3_5 build" in result.stdout diff --git a/website/docs/features/model-families.md b/website/docs/features/model-families.md index fe8f3e66c7..1068e093fb 100644 --- a/website/docs/features/model-families.md +++ b/website/docs/features/model-families.md @@ -142,6 +142,11 @@ for exact source revisions, evidence boundaries, and replay requirements. ### Optional Qwen3.5 Edge execution +Use `trtmc qwen3_5 build MODEL -o model.bundle` with the owning +family\u0027s options. `trtmc qwen3_5 build --help` works offline without +a checkpoint or GPU imports. This uses the existing +[family CLI protocol](../extend/family-cli.md), not an extension to the shared parser. + The Qwen3.5 family maps the recorded original-source dense 0.8B/2B/4B/9B configurations (Instruct and Base) and explicit 4B/9B DFlash pairs to the pinned Edge-LLM 0.10.1 SDK on native x86 SM80, FP16, batch one. Historical local From 33280048805a4bd210567c2c1504aeeaf00c953e Mon Sep 17 00:00:00 2001 From: Joshua Calafato Date: Wed, 23 Sep 2026 21:43:31 +0000 Subject: [PATCH 6/8] fix(qwen3_5): address Edge review findings Signed-off-by: Joshua Calafato --- families/qwen3_5/edge_llm/README.md | 6 +++++- families/qwen3_5/edge_llm/dispatch.py | 3 +++ families/qwen3_5/tests/test_precision_contract.py | 9 ++++----- website/docs/features/model-families.md | 2 +- 4 files changed, 13 insertions(+), 7 deletions(-) diff --git a/families/qwen3_5/edge_llm/README.md b/families/qwen3_5/edge_llm/README.md index 7ecccc09ae..d5c65f329f 100644 --- a/families/qwen3_5/edge_llm/README.md +++ b/families/qwen3_5/edge_llm/README.md @@ -132,6 +132,10 @@ output directory filesystem (choose a scratch-backed output), not system /tmp. ## Declared build command -This family uses the existing cli.json protocol introduced in #1310. The family\nowns its declaration, typed inputs and Python handler. The handler adapts those\ninputs to the unchanged builder API, preserving native/Edge dispatch and bundle\npublication. The legacy flat build command remains available for its existing\nordinary options; new family options use `trtmc qwen3_5 build`. +This family uses the existing cli.json protocol introduced in #1310. The family +owns its declaration, typed inputs and Python handler. The handler adapts those +inputs to the unchanged builder API, preserving native/Edge dispatch and bundle +publication. The legacy flat build command remains available for its existing +ordinary options; new family options use `trtmc qwen3_5 build`. Help is offline and does not need a local checkpoint. No shared parser hook or family registry entry is added. diff --git a/families/qwen3_5/edge_llm/dispatch.py b/families/qwen3_5/edge_llm/dispatch.py index 96d490e9c4..d6f4fab008 100644 --- a/families/qwen3_5/edge_llm/dispatch.py +++ b/families/qwen3_5/edge_llm/dispatch.py @@ -133,6 +133,9 @@ def build(request, writer, native, *, draft_dir: Path | None = None) -> None: log_path, exc_info=True, ) + except BaseException: + log_path.unlink(missing_ok=True) + raise else: # Edge preparation did not touch writer; publication cannot fallback. if adapter is not None: diff --git a/families/qwen3_5/tests/test_precision_contract.py b/families/qwen3_5/tests/test_precision_contract.py index ad80ec81eb..36fc600c75 100644 --- a/families/qwen3_5/tests/test_precision_contract.py +++ b/families/qwen3_5/tests/test_precision_contract.py @@ -115,13 +115,12 @@ def paired(request, writer, execution): ["--execution-variant", "dflash", "--companion", "draft=https://example.com/model"], ]) def test_bad_edge_cli_inputs_fail_before_backend(tmp_path, monkeypatch, options): - import importlib from tensorrt_model_connect import family_cli as build_cli - core = importlib.import_module("tensorrt_model_connect.build") + from families.qwen3_5 import cli as owner source, _ = _edge_cli_source(tmp_path) - monkeypatch.setattr(core, "_select_backend", lambda *_: pytest.fail("backend touched")) - monkeypatch.setattr(core, "BundleWriter", lambda *_: pytest.fail("writer created")) + monkeypatch.setattr(owner, "select_backend", lambda *_: pytest.fail("backend touched")) + monkeypatch.setattr(owner, "BundleWriter", lambda *_: pytest.fail("writer created")) with pytest.raises(ValueError): build_cli.main(["qwen3_5", "build", str(source), "-o", str(tmp_path / "out"), *options]) @@ -272,7 +271,7 @@ def native(*args): if mode in {"corrupt", "failure", "device_failure"}: assert len(logs) == 1 and "Traceback" in logs[0].read_text() assert "Retrying native once" in caplog.text - elif mode != "cancel": + else: assert not logs and "Edge build failed" not in caplog.text diff --git a/website/docs/features/model-families.md b/website/docs/features/model-families.md index 1068e093fb..f3292546e0 100644 --- a/website/docs/features/model-families.md +++ b/website/docs/features/model-families.md @@ -143,7 +143,7 @@ for exact source revisions, evidence boundaries, and replay requirements. ### Optional Qwen3.5 Edge execution Use `trtmc qwen3_5 build MODEL -o model.bundle` with the owning -family\u0027s options. `trtmc qwen3_5 build --help` works offline without +family's options. `trtmc qwen3_5 build --help` works offline without a checkpoint or GPU imports. This uses the existing [family CLI protocol](../extend/family-cli.md), not an extension to the shared parser. From 3aef7d7e9b9da462dacf56cb41b10260abeaebcd Mon Sep 17 00:00:00 2001 From: Joshua Calafato Date: Thu, 1 Oct 2026 22:52:34 +0000 Subject: [PATCH 7/8] fix(qwen3_5): default paired CLI builds to fp16 Resolve omitted precision in the family handler for explicit DFlash execution while retaining ordinary fp32 defaults and explicit precision values. Extend the existing routing test, including unchanged bf16 rejection. Keep model admission and quality gates unchanged. Signed-off-by: Joshua Calafato --- families/qwen3_5/cli.json | 2 +- families/qwen3_5/cli.py | 4 +++- .../qwen3_5/tests/test_precision_contract.py | 18 ++++++++++++++---- 3 files changed, 18 insertions(+), 6 deletions(-) diff --git a/families/qwen3_5/cli.json b/families/qwen3_5/cli.json index 6d616ec0ac..15e163c89e 100644 --- a/families/qwen3_5/cli.json +++ b/families/qwen3_5/cli.json @@ -49,7 +49,7 @@ "fp32", "fp16" ], - "default": "fp32" + "help": "Build precision (default: fp16 for paired execution, fp32 otherwise)" }, { "name": "backend", diff --git a/families/qwen3_5/cli.py b/families/qwen3_5/cli.py index b46ba93832..919d15cd3a 100644 --- a/families/qwen3_5/cli.py +++ b/families/qwen3_5/cli.py @@ -36,7 +36,7 @@ def build_bundle(request: BuildRequest, output: Path) -> None: def build( *, model: str, output: Path, revision: str | None = None, - task: str = "text_generation", precision: str = "fp32", backend: str = "trt", + task: str = "text_generation", precision: str | None = None, backend: str = "trt", max_sequence_length: int | None = None, tensor_parallel_size: int = 1, verbose: bool = False, fp32_layers: list[int] | tuple[int, ...] = (), @@ -44,6 +44,8 @@ def build( ) -> int: """Run the declared owner command; help never imports this handler.""" execution = execution_inputs(execution_variant, companion) + if precision is None: + precision = "fp16" if execution is not None else "fp32" model_dir = resolve_model(model, revision) resolve_family(load_model_metadata(model_dir), "qwen3_5") request = BuildRequest( diff --git a/families/qwen3_5/tests/test_precision_contract.py b/families/qwen3_5/tests/test_precision_contract.py index 36fc600c75..8b149a60d8 100644 --- a/families/qwen3_5/tests/test_precision_contract.py +++ b/families/qwen3_5/tests/test_precision_contract.py @@ -80,7 +80,8 @@ def _edge_cli_source(tmp_path): return source, draft -def test_edge_cli_uses_ordinary_family_build(tmp_path, monkeypatch): +@pytest.mark.parametrize("precision", [None, "fp16", "fp32", "bf16"]) +def test_edge_cli_uses_ordinary_family_build(tmp_path, monkeypatch, precision): from tensorrt_model_connect import family_cli as build_cli from families.qwen3_5.edge_llm import dispatch from families.qwen3_5.edge_llm.config import Qwen35BuildRequest @@ -92,6 +93,7 @@ def test_edge_cli_uses_ordinary_family_build(tmp_path, monkeypatch): def paired(request, writer, execution): assert isinstance(request, Qwen35BuildRequest) assert request.execution is execution + assert request.precision == ("fp16" if precision is None else precision) assert execution.variant == "dflash" assert [(item.role, item.model_dir) for item in execution.checkpoints] == [("draft", draft)] seen.append(request) @@ -99,10 +101,18 @@ def paired(request, writer, execution): writer.add_json("edge-test.json", {"variant": execution.variant}) monkeypatch.setattr(dispatch, "build_paired", paired) - assert build_cli.main(["qwen3_5", - "build", str(source), "--precision", "fp16", "-o", str(output), + precision_args = [] if precision is None else ["--precision", precision] + arguments = ["qwen3_5", + "build", str(source), *precision_args, "-o", str(output), "--execution-variant", "dflash", "--companion", f"draft={draft}", - ]) == 0 + ] + if precision == "bf16": + with pytest.raises(SystemExit) as error: + build_cli.main(arguments) + assert error.value.code == 2 + assert not seen and not output.exists() + return + assert build_cli.main(arguments) == 0 assert len(seen) == 1 assert output.is_file() From 407fac671e7841a07e75a0e6e5b8ad8bf214c435 Mon Sep 17 00:00:00 2001 From: Joshua Calafato Date: Thu, 1 Oct 2026 23:30:08 +0000 Subject: [PATCH 8/8] fix(qwen3_5): drain Edge requests before returning Keep request and response storage alive until queued CUDA work completes, including exceptional exits. Correct the existing family request test to assert direct coercion identity. Signed-off-by: Joshua Calafato --- families/qwen3_5/runtime/edge_llm/adapter.cpp | 6 ++++++ families/qwen3_5/tests/test_precision_contract.py | 3 ++- 2 files changed, 8 insertions(+), 1 deletion(-) diff --git a/families/qwen3_5/runtime/edge_llm/adapter.cpp b/families/qwen3_5/runtime/edge_llm/adapter.cpp index cad34f553c..5d2cdbb5e3 100644 --- a/families/qwen3_5/runtime/edge_llm/adapter.cpp +++ b/families/qwen3_5/runtime/edge_llm/adapter.cpp @@ -210,6 +210,12 @@ class EdgeTask final : public ITextGeneration { throw std::runtime_error("Qwen3.5 Edge returned invalid prompt counts"); validate_capacity(counts.front(), input_limit_, capacity_, request.maxGenerateLength); trt_edgellm::rt::LLMGenerationResponse response{}; + // Complete queued work before response/request storage is destroyed, + // including exception paths in a persistent task. + struct Drain { + cudaStream_t stream; + ~Drain() { cudaStreamSynchronize(stream); } + } drain{stream_.get()}; if (!runtime_->handleRequest(request, response, stream_.get()) || response.outputIds.size() != 1 || response.outputTexts.size() != 1 || response.outputIds.front().empty() || diff --git a/families/qwen3_5/tests/test_precision_contract.py b/families/qwen3_5/tests/test_precision_contract.py index 8b149a60d8..ec0e782958 100644 --- a/families/qwen3_5/tests/test_precision_contract.py +++ b/families/qwen3_5/tests/test_precision_contract.py @@ -150,7 +150,7 @@ def test_edge_cli_help_is_family_owned(tmp_path, capsys): def test_edge_request_preserves_fields_and_family_owner(tmp_path): from families.qwen3_5.edge_llm import cli from dataclasses import fields, replace - from tensorrt_model_connect.build import BuildRequest + from families.qwen3_5.build_request import BuildRequest, coerce_request from families.qwen3_5.edge_llm.config import ( BuildExecutionInputs, NamedCheckpoint, with_execution, ) @@ -158,6 +158,7 @@ def test_edge_request_preserves_fields_and_family_owner(tmp_path): source, draft = _edge_cli_source(tmp_path) request = BuildRequest(source, tmp_path / "out", "qwen3_5", "text_generation", "fp16", graph_transform=lambda layer: layer) + assert coerce_request(request) is request execution = BuildExecutionInputs("dflash", (NamedCheckpoint("draft", draft),)) assert cli.execution_inputs(None) is None extended = with_execution(request, execution)