From 65ae21066a4bffcefa3a645faa8dccedf8b59658 Mon Sep 17 00:00:00 2001 From: Joshua Calafato Date: Thu, 17 Sep 2026 00:26:27 +0000 Subject: [PATCH 1/9] feat(nemotron-h): add native Edge execution Keep ordinary complete-network offload and explicit Lightning DFlash ONNX execution owned by the Nemotron-H family. Preserve native prompt, full EOS and source quantization contracts; never replace a requested pair with base-only decoding. Restore the scalar-prefill platform admission used by the recorded passing profiles. Keep the native fallback implementation unchanged, extend an existing regression check and distinguish historical model qualification from current publication checks. Signed-off-by: Joshua Calafato --- families/nemotron_h/EDGE_LLM.md | 113 +++++++ families/nemotron_h/dispatch.py | 136 ++++++++ families/nemotron_h/edge_config.py | 79 +++++ families/nemotron_h/edge_llm.py | 252 ++++++++++++++ families/nemotron_h/edge_quantization.py | 106 ++++++ families/nemotron_h/edge_tokenizer.py | 95 ++++++ families/nemotron_h/model.py | 187 +++++----- families/nemotron_h/runtime/CMakeLists.txt | 13 + .../nemotron_h/runtime/chat_templates.cpp | 18 +- .../nemotron_h/runtime/edge_llm/adapter.cpp | 320 ++++++++++++++++++ .../nemotron_h/runtime/edge_llm/adapter.h | 15 + .../nemotron_h/runtime/edge_llm/contract.h | 60 ++++ .../runtime/edge_llm/device_link.cu | 5 + .../nemotron_h/runtime/edge_llm/request.h | 68 ++++ .../nemotron_h/runtime/edge_llm/tokenizer.h | 29 ++ families/nemotron_h/runtime/plugin.cpp | 10 + families/nemotron_h/tests/test_e2e.py | 177 +++++++++- .../nemotron_h/tests/test_runtime_contract.py | 8 + website/docs/features/model-families.md | 17 + 19 files changed, 1615 insertions(+), 93 deletions(-) create mode 100644 families/nemotron_h/EDGE_LLM.md create mode 100644 families/nemotron_h/dispatch.py create mode 100644 families/nemotron_h/edge_config.py create mode 100644 families/nemotron_h/edge_llm.py create mode 100644 families/nemotron_h/edge_quantization.py create mode 100644 families/nemotron_h/edge_tokenizer.py create mode 100644 families/nemotron_h/runtime/edge_llm/adapter.cpp create mode 100644 families/nemotron_h/runtime/edge_llm/adapter.h create mode 100644 families/nemotron_h/runtime/edge_llm/contract.h create mode 100644 families/nemotron_h/runtime/edge_llm/device_link.cu create mode 100644 families/nemotron_h/runtime/edge_llm/request.h create mode 100644 families/nemotron_h/runtime/edge_llm/tokenizer.h diff --git a/families/nemotron_h/EDGE_LLM.md b/families/nemotron_h/EDGE_LLM.md new file mode 100644 index 0000000000..e16a4c1595 --- /dev/null +++ b/families/nemotron_h/EDGE_LLM.md @@ -0,0 +1,113 @@ +# Nemotron-H Edge-LLM execution + +The family owns source-policy admission, command mapping, tokenizer/EOS semantics, +engine composition and runtime orchestration. TensorRT Edge-LLM owns model graphs, +lowering and execution. Provision the [optional native SDK](../../cmake/edgellm/README.md) +from official GitHub Edge-LLM 0.10.1, revision +`e8b29522938901f6df19ebeedd4b69bc8edbcd97`. No internal source changes or +cross-compilation are used. + +## Ordinary and paired builds + +Ordinary compatible text builds use the installed experimental Python builder +with components=llm and original checkpoints. Plain source weights use FP16 +compute; packed FP8/NVFP4 or mixed policies remain in their declared formats. +The adapter preserves source scales, exclusions and KV policy. For packed +sources, only supported external-weight kinds are requested; FP16 bias tensors +are baked into the engine rather than externalized through a missing recipe. + +The explicit Lightning DFlash pair instead uses the original ONNX exporter and +native ONNX builder. Enable `TRTMC_EDGELLM_ALL_KERNELS=ON` and +`TRTMC_EDGELLM_ONNX=ON`, set `CMAKE_PREFIX_PATH` to the SDK installation, and +configure the runtime with `TRTMC_ENABLE_EDGELLM=ON`. Add +`--execution-variant dflash --companion draft=/path/to/draft` to the build CLI, +or pass `BuildExecutionInputs("dflash", (NamedCheckpoint("draft", draft_path),))` +to the Python build API. + +Both paired engines are mandatory. The ONNX graph supplies the intermediate +Mamba/replay states required to commit accepted draft tokens; the experimental +builder's final-only state outputs do not implement this pair. No base-only +engine is substituted. ONNX projection weights are baked into plans and are not +duplicated in the bundle. Successfully consumed ONNX intermediates are removed. + +Ordinary preparation failure warns, retains a diagnostic log and attempts the +unchanged native request once. Native fallback rejects packed checkpoints it +cannot interpret. Paired native fallback is unavailable and fails explicitly. +Cancellation, publication and runtime errors propagate without retry. + +## Request and runtime contracts + +Admission requires compatible Nemotron-H text topology and source quantization, +FP16 compute, TP1/batch1/context-parallel1, and no unmapped build controls. +The published platform routes are native x86_64 SM80 for plain sources and +SM120 for plain/FP8/NVFP4 sources. These routes are admission rules, not proof for +every compatible checkpoint or capacity. + +Do not gate head80 on optimized CuTe SSD availability. The pinned Mamba plugin +has a scalar-prefill fallback, which the recorded SM80/SM120 profiles exercise. +A previous optimized-kernel-only restriction was incorrect and is removed. + +The family renders and tokenizes using its native prompt semantics, then supplies +actual pretokenized IDs to the public Edge API. Full native EOS lists are +preserved with validated derivative tokenizer metadata; original source assets +are unchanged. Separate Jinja templates follow source-file precedence, and the +modern Nemotron ChatML empty-system/thinking suffix is retained. Empty prompts, +submitted counts, total capacity and generation completion are checked. + +The runtime is persistent and serialized, with scoped plugin, stream and artifact +ownership. Supported ordinary sampling controls are forwarded; unmapped controls +are rejected. DFlash uses block16/verify16 and is greedy-only because the pinned +runtime forces greedy verification. Runtime errors are not converted to native +inference. + +## Recorded model qualification + +These are **historical local Model Connect build/inference qualifications**, +not assertions that every publication head or CI executes these profiles. +CUDA 13.3 and TensorRT 11.1.0.106 were used. Ordinary MC profiles use capacity256; +the DFlash profile uses input/KV1024. All are text-only TP1/batch1. +Independent NED must be at most0.15; the original Edge128-token fixture gates +remain ROUGE-1 >=0.25 and ROUGE-L >=0.20. + +| Exact source model | GPU | Independent gate | MC ROUGE-1 / ROUGE-L | +| --- | --- | --- | --- | +| NVIDIA-Nemotron-3-Nano-4B-BF16 | SM120 | NED0.0 | 0.5371 / 0.2514 | +| NVIDIA-Nemotron-3-Nano-4B-FP8 | SM120 | NED0.0 | 0.5402 / 0.2644 | +| NVIDIA-Nemotron-Nano-9B-v2 | SM80 | Existing HF pytest gate passed; numeric NED not serialized | 0.4318 / 0.2727 | +| NVIDIA-Nemotron-Nano-9B-v2-FP8 | SM120 | NED0.0 | 0.3750 / 0.2614 | +| NVIDIA-Nemotron-Nano-9B-v2-NVFP4 | SM120 | NED0.0 | 0.4130 / 0.2391 | +| NVIDIA-Nemotron-3.5-Lightning-30B-A3B-NVFP4 | SM120 | NED0.0 | 0.4643 / 0.2500 | +| Same Lightning target + DFlash companion | SM120 | NED0.0 | 0.4848 / 0.2545 | + +All models are from the public `nvidia` namespace. Immutable revisions, in table +order, are: + +- 4B-BF16: `dfaf35de3e30f1867dd8dbc38a7fc9fb52d3914f`. +- 4B-FP8: `3fe6dab75665a93884214ad4b1b95cf02717d081`. +- 9B-v2: `6533e8de2c68e4536bf7c411d7a3ce5734111476`. +- 9B-v2-FP8: `8bc5eece2eb5514c4bca7f2ec655b91eb554f4c0`. +- 9B-v2-NVFP4: `8556c9164ddb43fe1f4f4ad730593b3c5e3f7328`. +- Lightning target: `bee7596271d1495f6992ae224aefde4410e816b8`. +- Lightning DFlash companion: `8abcc4db8f34a5080c31eef05d4467afc06c6b9e`. + +The independent references use builtin HF BF16 mathematical execution. Packed +sources are decoded with official ModelOpt routines and strict full-state loading; +these oracles do not emulate activation/KV quantization rounding. DFlash reused +a source-byte/reference-function-verified independent reference rather than +regenerating it. Its Edge fixture uses explicit greedy controls, not a claim of +sampling parity. + +Passing payloads were retired with approval; compact results and provenance were +retained. Replaying all seven model checks requires rebuilding those payloads. +Fresh publication compilation/unit/source checks must be reported separately. +Only the existing plain9B case is a registered owning E2E among these profiles; +the other exact models/pair used local recipes with existing family helpers. + +## Still outside these qualifications + +The separate direct-Edge9B-NVFP4 capacity1024 run failed ROUGE-L0.1957 against0.20 +on a different SM120 GPU. The MC256 pass does not resolve that failure. +The earlier 4B-BF16 capacity4096 compiler failure, long-context/context-reuse, +TP4, other models/platforms and stochastic equivalence remain outside this proof. +Nano30B, Super120B and Omni are not qualified by the table above. +No quality gate, failed result or hardware limitation is hidden by these passes. diff --git a/families/nemotron_h/dispatch.py b/families/nemotron_h/dispatch.py new file mode 100644 index 0000000000..50e1401a71 --- /dev/null +++ b/families/nemotron_h/dispatch.py @@ -0,0 +1,136 @@ +# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-License-Identifier: Apache-2.0 + +"""Family-owned Nemotron-H complete-network map and native preparation retry.""" + +from __future__ import annotations + +import json +import logging +import os +from pathlib import Path +import tempfile +import traceback + +from . import edge_llm +from .edge_quantization import source_quantization +from .edge_config import topology_matches + +_LOG = logging.getLogger(__name__) +# Native platforms with recorded passing family profiles; admission is not qualification. +PLATFORMS = {"x86_64": (80, 120)} +EDGE_DISPATCH = { + ("linux", arch, sm, precision): edge_llm.prepare + for arch, sms in PLATFORMS.items() + for sm in sms + for precision in ("fp16", "fp8", "nvfp4") + if precision == "fp16" or sm >= 100 +} + + +def request_matches(request) -> bool: + """Preserve native early rejection order for unmapped request controls.""" + return ( + request.family == "nemotron_h" and request.backend == "trt" and request.task == "text_generation" + and request.precision.lower() == "fp16" and request.quantization in {None, "none", "fp8", "nvfp4"} + and request.max_batch_size == request.tensor_parallel_size == request.context_parallel_size == 1 + and not request.dynamic_kv_cache and not request.fp32_layers and request.graph_transform is None + and all(value is None for value in (request.image_height, request.image_width, request.video_num_frames)) + ) + + +def candidate(request, raw: dict) -> bool: + """Preserve native tokenizer contract and require exact source hybrid policy.""" + if not request_matches(request): + return False + precision = source_quantization(request, raw) + if precision is None or not topology_matches(raw, precision): + return False + try: + from .edge_tokenizer import source_chat_template + source_chat_template(Path(request.model_dir)) + return True + except (OSError, ValueError): + return False + + +def platform_matches(raw: dict, target: dict) -> bool: + """Admit the native runtime, not just its optional optimized SSD kernel.""" + # MambaPlugin falls back to scalar prefill when SSD cannot implement a shape. + # Topology, source precision and the installed package are checked separately. + return target["sm"] >= 80 + + +def build(request, writer, native) -> None: + """Prepare Edge privately or warn and retry unchanged native once. + + Publication and cancellation errors propagate. Runtime fallback is not part + of this builder adapter; an Edge bundle is owned entirely by its runtime. + """ + if not request_matches(request): + native(request, writer) + return + raw = json.loads((Path(request.model_dir) / "config.json").read_text(encoding="utf-8")) + if not isinstance(raw, dict): + raise ValueError("checkpoint config.json must contain an object") + if not candidate(request, raw): + native(request, writer) + return + if edge_llm.sequence_length(request, raw) > raw["max_position_embeddings"]: + raise ValueError("Nemotron-H max_sequence_length exceeds checkpoint context capacity") + failure = None + descriptor, name = tempfile.mkstemp(prefix=f".{request.output_path.name}.edge-", suffix=".log", + dir=request.output_path.parent) + os.close(descriptor) + log_path = Path(name) + with tempfile.TemporaryDirectory(prefix="trtmc-nemotron-h-edge-") as directory: + try: + target = edge_llm.local_target() + key = (target["os"], target["arch"], target["sm"], source_quantization(request, raw)) + adapter = EDGE_DISPATCH.get(key) if platform_matches(raw, target) else None + if adapter is not None: + files, marker = adapter(request, raw, target, Path(directory), log_path) + except Exception as error: + failure = error + with log_path.open("a", encoding="utf-8") as log: + traceback.print_exception(error, file=log) + _LOG.warning("nemotron_h Edge build failed: %s. Diagnostics: %s. " + "Retrying native once with the unchanged request.", error, log_path, exc_info=True) + else: + if adapter is not None: + edge_llm.publish(request, writer, files, marker) + return + log_path.unlink() + try: + native(request, writer) + except Exception as error: + if failure is not None: + raise error from failure + raise + + +def build_dflash(request, writer, draft: Path) -> None: + """Preserve a paired request on failure; never replace it with a base-only build.""" + raw = json.loads((Path(request.model_dir) / "config.json").read_text(encoding="utf-8")) + if not candidate(request, raw) or source_quantization(request, raw) != "nvfp4": + raise ValueError("Nemotron-H DFlash ONNX requires a compatible NVFP4 target and TP1 text request") + if edge_llm.sequence_length(request, raw) > raw["max_position_embeddings"]: + raise ValueError("Nemotron-H max_sequence_length exceeds checkpoint context capacity") + descriptor, name = tempfile.mkstemp(prefix=f".{request.output_path.name}.edge-onnx-", suffix=".log", + dir=request.output_path.parent) + os.close(descriptor) + log_path = Path(name) + with tempfile.TemporaryDirectory(prefix="trtmc-nemotron-h-dflash-", dir=request.output_path.parent) as directory: + try: + target = edge_llm.local_target() + key = (target["os"], target["arch"], target["sm"], "nvfp4") + if key not in EDGE_DISPATCH or not platform_matches(raw, target): + raise ValueError("Nemotron-H DFlash ONNX is unavailable on this native platform") + files, marker = edge_llm.prepare_dflash(request, raw, target, Path(directory), log_path, draft) + except Exception as error: + with log_path.open("a", encoding="utf-8") as log: + traceback.print_exception(error, file=log) + _LOG.warning("Nemotron-H DFlash Edge ONNX build failed: %s. Diagnostics: %s. " + "Native paired execution is unavailable; refusing base-only fallback.", error, log_path) + raise NotImplementedError("Native Nemotron-H DFlash fallback is unavailable") from error + edge_llm.publish(request, writer, files, marker) diff --git a/families/nemotron_h/edge_config.py b/families/nemotron_h/edge_config.py new file mode 100644 index 0000000000..2fbebca8db --- /dev/null +++ b/families/nemotron_h/edge_config.py @@ -0,0 +1,79 @@ +# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-License-Identifier: Apache-2.0 +"""Native hybrid source topology gates; Edge still constructs every graph.""" + +from __future__ import annotations + +import math + +_PATTERN = {"M": "mamba", "-": "mlp", "*": "attention", "E": "moe"} + + +def layers(raw: dict) -> list[str] | None: + """Require complete consistent explicit hybrid topology, never filter unknowns.""" + forms = [] + for name in ("layers_block_type", "layer_types"): + if name in raw: + value = raw[name] + if not isinstance(value, list) or any(x not in _PATTERN.values() for x in value): + return None + forms.append(value) + if raw.get("hybrid_override_pattern"): + pattern = raw["hybrid_override_pattern"] + if not isinstance(pattern, str) or any(x not in _PATTERN for x in pattern): + return None + forms.append([_PATTERN[x] for x in pattern]) + if not forms or not forms[0] or any(x != forms[0] for x in forms): + return None + return forms[0] if len(forms[0]) == raw.get("num_hidden_layers") else None + + +def topology_matches(raw: dict, precision: str) -> bool: + """Admit only pinned dense/SSD or non-gated NVFP4 MoE semantics.""" + kinds = layers(raw) + if (raw.get("model_type") != "nemotron_h" + or raw.get("architectures") != ["NemotronHForCausalLM"] or kinds is None + or "mamba" not in kinds + or any(key in raw for key in ("text_config", "vision_config", "audio_config", "thinker_config", + "eagle_config", "dflash_config", "jetspec_config", "dspark_config")) + or raw.get("mamba_hidden_act", "silu") != "silu" + or raw.get("mlp_hidden_act", "relu2") != "relu2" + or raw.get("hidden_act", "relu2") != "relu2" + or raw.get("mamba_ssm_cache_dtype", "float32") != "float32" + or any(raw.get(key, False) for key in ("attention_bias", "mamba_proj_bias", "mlp_bias")) + or raw.get("rope_scaling") is not None or raw.get("hybrid_uses_rope", False) + or raw.get("norm_before_gate", False) or raw.get("conv_kernel", 4) != 4): + return False + if type(raw.get("bos_token_id", -1)) is not int or raw.get("bos_token_id", -1) < -1: + return False + keys = ("hidden_size", "num_hidden_layers", "num_attention_heads", "num_key_value_heads", + "head_dim", "mamba_num_heads", "mamba_head_dim", "n_groups", "ssm_state_size", + "max_position_embeddings", "vocab_size") + values = [raw.get(key) for key in keys] + if any(type(value) is not int or value <= 0 for value in values): + return False + hidden, _, heads, kv, head, mheads, mhead, groups, state, _, _ = values + if (heads % kv or head % 8 or mheads % groups or state not in {64, 128} + or mhead not in {64, 80, 128} or mhead == 80 and state != 128 + or raw.get("conv_dim", mheads * mhead + 2 * groups * state) + != mheads * mhead + 2 * groups * state): + return False + if "mlp" in kinds and (type(raw.get("intermediate_size")) is not int or raw["intermediate_size"] <= 0): + return False + if "moe" not in kinds: + return not raw.get("n_routed_experts", 0) + names = ("n_routed_experts", "num_experts_per_tok", "n_group", "topk_group", + "moe_intermediate_size", "moe_shared_expert_intermediate_size") + if any(type(raw.get(key)) is not int or raw[key] <= 0 for key in names): + return False + experts, topk, router_groups, selected_groups, intermediate, shared = [raw[key] for key in names] + scale = raw.get("routed_scaling_factor", 1.0) + latent = raw.get("moe_latent_size", hidden) + latent = hidden if latent is None else latent + return (precision == "nvfp4" and experts <= 512 and topk <= experts + and experts % router_groups == 0 and selected_groups <= router_groups + and topk <= selected_groups * (experts // router_groups) + and raw.get("n_shared_experts", 1) == 1 and intermediate % 16 == shared % 16 == 0 + and type(latent) is int and latent > 0 and latent % 16 == 0 + and isinstance(scale, (int, float)) and not isinstance(scale, bool) + and math.isfinite(scale) and scale > 0) diff --git a/families/nemotron_h/edge_llm.py b/families/nemotron_h/edge_llm.py new file mode 100644 index 0000000000..d4e1f5215b --- /dev/null +++ b/families/nemotron_h/edge_llm.py @@ -0,0 +1,252 @@ +# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-License-Identifier: Apache-2.0 + +"""Thin family-owned adapter to the pinned Edge direct-builder API.""" + +from __future__ import annotations + +import json +from pathlib import Path +import shutil +import subprocess + +from tensorrt_model_connect.build import cmake_prefixes, detect_local_platform, subprocess_environment + +EDGE_REVISION = "e8b29522938901f6df19ebeedd4b69bc8edbcd97" +_QUANTIZED_EXTERNAL_WEIGHTS = ( + "int4_ffn", "int4_moe", "nvfp4_moe", "nvfp4_tp", "lm_head", "embedding", +) + + +def local_target() -> dict: + """Return the executing worker identity supplied by generic build mechanics.""" + return detect_local_platform() + + +def installed_package(target: dict) -> dict: + """Resolve CMake installation via standard prefixes; never install anything. + + Args: + target: Executing device and SDK identity. + + Returns: + Validated package metadata with absolute Python and plugin paths. + + Raises: + FileNotFoundError: No CMake installation or required artifact exists. + ValueError: Pin, architecture, SDK or contained-path contract differs. + """ + for prefix in cmake_prefixes(): + manifest = prefix / "share/trtmc/edge-llm.json" + if not manifest.is_file(): + continue + package = json.loads(manifest.read_text(encoding="utf-8")) + if package.get("schema_version") != 1 or package.get("revision") != EDGE_REVISION: + raise ValueError(f"Edge package has an unsupported revision/schema: {manifest}") + if package.get("all_native_kernels") is not True: + raise ValueError("Nemotron-H requires an Edge package with all native operator groups") + if package.get("version") != "0.10.1" or package.get("arch") != target["arch"]: + raise ValueError("Edge package version/architecture differs from executing worker") + if target["sm"] not in package.get("architectures", []): + raise ValueError("Edge package was not built for this local GPU") + cuda_version = ".".join(str(package.get("cuda_version", "")).split(".")[:2]) + if cuda_version != target["cuda_version"] or package.get("tensorrt_version") != target["tensorrt_version"]: + raise ValueError("Edge package CUDA/TensorRT differs from executing worker") + for name in ("python", "plugin") + (("onnx_builder",) if package.get("onnx") is True else ()): + relative = Path(package[name]) + path = (prefix / relative).resolve() + if relative.is_absolute() or not path.is_relative_to(prefix.resolve()): + raise ValueError(f"Edge package {name} must be contained in its installation") + if not path.is_file(): + raise FileNotFoundError(f"Edge package {name} is missing: {path}") + package[name] = str(path) + return package + raise FileNotFoundError("Edge-LLM is not installed; enable the optional Edge-LLM CMake dependency " + "and set CMAKE_PREFIX_PATH to its install prefix") + + +def sequence_length(request, raw: dict) -> int: + """Resolve the same default capacity as this family's original native builder.""" + value = request.max_sequence_length or min(raw["max_position_embeddings"], 256) + if isinstance(value, bool): + raise ValueError("max_sequence_length must be a positive integer") + try: + result = int(value) + except (TypeError, ValueError, OverflowError) as error: + raise ValueError("max_sequence_length must be a positive integer") from error + if result < 1: + raise ValueError("max_sequence_length must be a positive integer") + return result + + +def prepare(request, raw: dict, target: dict, staging: Path, log_path: Path) -> tuple[dict, dict]: + """Map the request to Edge main(argv), returning complete unpublished assets. + + Edge owns model selection, configuration, conversion, graphs and engine + composition. The family adapter only maps text-generation arguments and + preserves the checkpoint needed by Edge external-weight APIs. + + Returns: + (section-name to file mapping, runtime marker). + + Raises: + Exception: Dependency, upstream build or artifact validation failed. + """ + from .edge_quantization import source_quantization + + source_precision = source_quantization(request, raw) + if source_precision is None: + raise ValueError("Unsupported Nemotron-H checkpoint precision") + package = installed_package(target) + checkpoint = staging / "edge_llm/checkpoint" + checkpoint.mkdir(parents=True) + for source in Path(request.model_dir).iterdir(): + if source.is_file() and (source.suffix in {".json", ".safetensors", ".model", ".jinja"} + or source.name in {"merges.txt", "vocab.txt"}): + shutil.copy2(source, checkpoint / source.name) + if not list(checkpoint.glob("*.safetensors")): + raise ValueError("Edge direct builder requires a safetensors checkpoint") + limit = sequence_length(request, raw) + engine = staging / "edge_llm/engine" + # Calling upstream main preserves its complete build/artifact orchestration. + command = [package["python"], "-I", "-c", + "from experimental.builder.cli import main; main()", + "--model-dir", str(checkpoint), "--engine-dir", str(engine), + "--components", "llm", "--plugin-path", package["plugin"], + "--dense", "fp16" if source_precision == "fp16" else "auto", "--max-input-len", str(limit), + "--max-kv-cache-capacity", str(limit), "--max-batch-size", "1"] + # The pinned builder cannot externalize FP16 biases on quantized projections: + # their checkpoint recipes are missing. Bake FP16 parameters (which can enlarge the plan), + # retaining the original packed quantized weights and every other supported kind. + external_weights = ("all",) if source_precision == "fp16" else _QUANTIZED_EXTERNAL_WEIGHTS + for kind in external_weights: + command.extend(("--externalize-weights", kind)) + if request.verbose: + command.append("--verbose") + with log_path.open("a", encoding="utf-8") as log: + subprocess.run(command, check=True, stdout=log, stderr=subprocess.STDOUT, cwd=staging) + for name in ("llm.engine", "config.json", "tokenizer.json", "tokenizer_config.json", "processed_chat_template.json"): + if not (engine / name).is_file() or (engine / name).stat().st_size == 0: + raise ValueError(f"Edge builder did not produce required artifact: {name}") + from .edge_tokenizer import prepare_tokenizer, source_chat_template + + runtime_tokenizer = staging / "edge_llm/runtime_tokenizer" + eos = prepare_tokenizer(checkpoint, engine, runtime_tokenizer, raw, + chat_template=source_chat_template(checkpoint)) + files = {} + for directory in (engine, checkpoint, runtime_tokenizer): + for path in sorted(directory.rglob("*")): + if path.is_symlink(): + raise ValueError(f"Edge output must not contain symlinks: {path}") + if path.is_file(): + files[path.relative_to(staging).as_posix()] = path + return files, { + "version": 1, "edge_revision": EDGE_REVISION, "target": target, "precision": "fp16", + "max_sequence_length": limit, "max_input_length": limit, "max_batch_size": 1, + "artifacts": list(files), "checkpoint_precision": source_precision, + "tokenizer_policy": "native_full_eos", "native_eos_token_ids": eos, + "native_bos_token_id": raw.get("bos_token_id", -1), + } + + +def publish(request, writer, files: dict, marker: dict) -> None: + """Stream complete Edge sections; publication errors must not retry native.""" + writer.set_header(family=request.family, task=request.task, backend=request.backend) + for name, path in files.items(): + with path.open("rb") as source, writer.open_section(name) as destination: + shutil.copyfileobj(source, destination, length=1024 * 1024) + writer.add_json("edge_llm.json", marker) + + +def prepare_dflash(request, raw: dict, target: dict, staging: Path, log_path: Path, + draft: Path) -> tuple[dict, dict]: + """Build both recurrent-state-aware graphs with the original ONNX toolchain. + + Export consumes the caller's immutable local checkpoints. ONNX plans bake + projection weights, so the bundle needs only engine assets and tokenizer + metadata, not a duplicate of either checkpoint's packed weights. + """ + from .edge_quantization import source_quantization + from .edge_tokenizer import prepare_tokenizer, source_chat_template + + package = installed_package(target) + if package.get("onnx") is not True: + raise ValueError("Nemotron-H DFlash requires an ONNX-enabled Edge SDK") + source = Path(request.model_dir).resolve() + draft = draft.resolve() + draft_config = json.loads((draft / "config.json").read_text(encoding="utf-8")) + dflash = draft_config.get("dflash_config", {}) + layer_ids = dflash.get("target_layer_ids", []) + if (draft_config.get("architectures") != ["DFlashDraftModel"] + or draft_config.get("hidden_size") != raw["hidden_size"] + or draft_config.get("vocab_size") != raw["vocab_size"] + or not layer_ids or len(set(layer_ids)) != len(layer_ids) + or any(type(i) is not int or not 0 <= i < raw["num_hidden_layers"] for i in layer_ids) + or dflash.get("block_size", 16) != 16): + raise ValueError("Nemotron-H DFlash companion geometry does not match its target") + if not list(source.glob("*.safetensors")) or not list(draft.glob("*.safetensors")): + raise ValueError("Nemotron-H DFlash requires both local safetensors checkpoints") + limit = sequence_length(request, raw) + if limit < 16: + raise ValueError("Nemotron-H DFlash requires capacity for a complete block16") + checkpoint = staging / "edge_llm/checkpoint" + checkpoint.mkdir(parents=True) + for name in ("config.json", "tokenizer.json", "tokenizer_config.json", "generation_config.json", + "chat_template.jinja"): + if (source / name).is_file(): + shutil.copy2(source / name, checkpoint / name) + engine = staging / "edge_llm/engine" + onnx = staging / "onnx" + # Upstream packed MoE layout is an exporter option, not a model rewrite. + env = subprocess_environment( + {"EDGELLM_PLUGIN_PATH": package["plugin"], + "EDGELLM_NVFP4_MOE_TARGET": "sm12x" if target["sm"] in {120, 121} else f"sm{target['sm']}"}, + prepend_paths={"LD_LIBRARY_PATH": str(Path(package["plugin"]).parent)}, + ) + for role, subdirectory, flag in (("draft", "dflash_draft", "--specDraft"), + ("base", "llm", "--specBase")): + commands = [ + [package["python"], "-I", "-m", "tensorrt_edgellm.scripts.export", + str(source), str(onnx), f"--dflash-{role}", "--dflash-draft-dir", str(draft), + "--skip-visual", "--skip-audio"], + [package["onnx_builder"], "--onnxDir", str(onnx / subdirectory), + "--engineDir", str(engine), flag, "--maxInputLen", str(min(limit, 1024)), + "--maxKVCacheCapacity", str(limit), "--maxBatchSize", "1", + "--maxVerifyTreeSize", "16", "--maxDraftTreeSize", "16"], + ] + with log_path.open("a", encoding="utf-8") as log: + for command in commands: + log.write(json.dumps(command) + "\n") + log.flush() + subprocess.run(command, check=True, stdout=log, stderr=subprocess.STDOUT, + cwd=staging, env=env) + # Only intermediates created by this preparation are removed. Source + # checkpoints and final engine assets remain untouched. + shutil.rmtree(onnx) + required = ("spec_base.engine", "spec_draft.engine", "base_config.json", "draft_config.json", + "embedding.safetensors", "tokenizer.json", "tokenizer_config.json", "processed_chat_template.json") + for name in required: + if not (engine / name).is_file() or (engine / name).stat().st_size == 0: + raise ValueError(f"Edge ONNX builder did not produce required artifact: {name}") + for role in ("base", "draft"): + config = json.loads((engine / f"{role}_config.json").read_text(encoding="utf-8")) + if config.get("spec_decode_type") != "dflash" or config.get("dflash_config", {}).get("block_size") != 16: + raise ValueError("Edge ONNX builder returned a different speculative execution contract") + tokenizer = staging / "edge_llm/runtime_tokenizer" + eos = prepare_tokenizer(checkpoint, engine, tokenizer, raw, + chat_template=source_chat_template(checkpoint)) + files = {} + for directory in (engine, checkpoint, tokenizer): + for path in sorted(directory.rglob("*")): + if path.is_symlink(): + raise ValueError(f"Edge output must not contain symlinks: {path}") + if path.is_file(): + files[path.relative_to(staging).as_posix()] = path + return files, { + "version": 1, "edge_revision": EDGE_REVISION, "target": target, "precision": "fp16", + "max_sequence_length": limit, "max_input_length": min(limit, 1024), "max_batch_size": 1, + "artifacts": list(files), "checkpoint_precision": source_quantization(request, raw), + "tokenizer_policy": "native_full_eos", "native_eos_token_ids": eos, + "native_bos_token_id": raw.get("bos_token_id", -1), "execution_variant": "dflash", + "builder_flow": "onnx", "dflash_block_size": 16, + } diff --git a/families/nemotron_h/edge_quantization.py b/families/nemotron_h/edge_quantization.py new file mode 100644 index 0000000000..d3ebf5fcbb --- /dev/null +++ b/families/nemotron_h/edge_quantization.py @@ -0,0 +1,106 @@ +# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-License-Identifier: Apache-2.0 +"""Exact family-owned source policy admission; never convert or calibrate weights.""" + +from __future__ import annotations + +import json +from pathlib import Path + + +def descriptor(value: dict) -> tuple | None: + """Normalize known ModelOpt policy fields, rejecting unknown packed layouts.""" + if not isinstance(value, dict): + return None + q = value.get("quantization", value) + if not isinstance(q, dict): + return None + if q.get("quant_method", "modelopt").lower() not in {"modelopt", ""}: + return None + algorithm = str(q.get("quant_algo", "")).upper() + if algorithm == "MIXED_PRECISION": + layers = q.get("quantized_layers") + if not isinstance(layers, dict) or not layers: + return None + algorithms = set() + for name, policy in layers.items(): + if not isinstance(name, str) or not name or not isinstance(policy, dict): + return None + algo = policy.get("quant_algo") + if algo not in {"FP8", "NVFP4", "W4A16_NVFP4"}: + return None + group = policy.get("group_size", 1 if algo == "FP8" else 16) + if type(group) is not int or group not in ({1, 16} if algo == "FP8" else {16}): + return None + if set(policy) - {"quant_algo", "group_size"}: + return None + algorithms.add(algo) + if algorithms not in ({"FP8", "NVFP4"}, {"FP8", "W4A16_NVFP4"}): + return None + kv = q.get("kv_cache_quant_algo") + scheme = q.get("kv_cache_scheme") + if scheme is not None: + if scheme != {"dynamic": False, "num_bits": 8, "type": "float"}: + return None + if kv not in {None, "FP8"}: + return None + kv = "FP8" + if kv != "FP8": + return None + # The pinned direct builder consumes the sidecar per-layer map. Keep + # its exact algorithms/scales; do not reinterpret ModelOpt config_groups. + return "nvfp4", kv, json.dumps(layers, sort_keys=True), "mixed" + precision = {"FP8": "fp8", "NVFP4": "nvfp4", "W4A16_NVFP4": "nvfp4"}.get(algorithm) + if precision is None: + return None + group = q.get("group_size", 1 if precision == "fp8" else 16) + # FP8 projections use scalar scaling, not grouped packing. ModelOpt emits + # either1 (parser default) or16 (export metadata); neither changes FP8 layout. + if type(group) is not int or group not in ({1, 16} if precision == "fp8" else {16}): + return None + if any(key in q for key in ("config_groups", "quantized_layers", "kv_cache_scheme")): + return None + if any(key not in {"quant_algo", "group_size", "kv_cache_quant_algo", "exclude_modules", + "ignore", "quant_method"} for key in q): + return None + excluded = q.get("exclude_modules", q.get("ignore", [])) + if (not isinstance(excluded, list) or any(not isinstance(x, str) for x in excluded) + or ("ignore" in q and "exclude_modules" in q and set(q["ignore"]) != set(excluded))): + return None + kv = q.get("kv_cache_quant_algo") + if kv not in {None, "FP8"}: + return None + return precision, kv, tuple(sorted(set(excluded))) + + +def source_quantization(request, raw: dict) -> str | None: + """Preserve consistent original/FP8/NVFP4 source, or return native admission.""" + try: + policies = [] + embedded = raw.get("quantization_config") + if embedded is not None: + policies.append(descriptor(embedded)) + for name in ("hf_quant_config.json", "quantize_config.json", "quant_config.json"): + path = Path(request.model_dir) / name + if path.exists(): + value = json.loads(path.read_text(encoding="utf-8")) + policies.append(descriptor(value)) + if policies: + if any(value is None or value != policies[0] for value in policies): + return None + # Upstream only consumes hf_quant_config or embedded metadata. + if embedded is None and not (Path(request.model_dir) / "hf_quant_config.json").is_file(): + return None + if policies[0][-1] == "mixed" and ( + embedded is None or not (Path(request.model_dir) / "hf_quant_config.json").is_file() + ): + return None + precision = policies[0][0] + else: + precision = "fp16" + requested = request.quantization + if requested is not None and requested != ("none" if precision == "fp16" else precision): + return None + return precision + except (OSError, ValueError, TypeError, AttributeError): + return None diff --git a/families/nemotron_h/edge_tokenizer.py b/families/nemotron_h/edge_tokenizer.py new file mode 100644 index 0000000000..5e773f0ce8 --- /dev/null +++ b/families/nemotron_h/edge_tokenizer.py @@ -0,0 +1,95 @@ +# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-License-Identifier: Apache-2.0 + +"""Serialize native full-EOS metadata without mutating original tokenizer assets.""" + +from __future__ import annotations + +import json +from pathlib import Path +import shutil + + +def _object(path: Path) -> dict: + """Read a JSON object or raise a preparation error with the source path.""" + value = json.loads(path.read_text(encoding="utf-8")) + if not isinstance(value, dict): + raise ValueError(f"Tokenizer metadata must contain an object: {path.name}") + return value + + +def source_chat_template(checkpoint: Path) -> str: + """Resolve the original standalone template with HF file precedence.""" + metadata = _object(checkpoint / "tokenizer_config.json") + standalone = checkpoint / "chat_template.jinja" + template = standalone.read_text(encoding="utf-8") if standalone.exists() else metadata.get("chat_template") + if not isinstance(template, str) or not template.strip(): + raise ValueError("Nemotron-H requires an embedded or standalone chat template") + return template + + +def native_eos_ids(checkpoint: Path, raw: dict) -> list[int]: + """Resolve native generation-config precedence and the complete stop vector.""" + generation = checkpoint / "generation_config.json" + config = _object(generation) if generation.exists() else {} + eos = config.get("eos_token_id", raw.get("eos_token_id")) + ids = eos if isinstance(eos, list) else [eos] + if (not ids or any(type(value) is not int or not 0 <= value < raw["vocab_size"] for value in ids) + or len(set(ids)) != len(ids)): + raise ValueError("Nemotron-H native EOS must contain unique vocabulary IDs") + return ids + + +def eos_token(tokenizer: dict, eos: int) -> str: + """Resolve one unambiguous original token string for an ID, without constants.""" + by_id: dict[int, set[str]] = {} + by_token: dict[str, set[int]] = {} + vocab = tokenizer.get("model", {}).get("vocab", {}) + if not isinstance(vocab, dict): + raise ValueError("Nemotron-H Edge requires a BPE vocabulary object") + entries = [(token, index) for token, index in vocab.items()] + for entry in tokenizer.get("added_tokens", []): + if not isinstance(entry, dict): + raise ValueError("Invalid Nemotron-H added token entry") + entries.append((entry.get("content"), entry.get("id"))) + for token, index in entries: + if not isinstance(token, str) or type(index) is not int or index < 0: + raise ValueError("Invalid Nemotron-H tokenizer token/ID mapping") + by_id.setdefault(index, set()).add(token) + by_token.setdefault(token, set()).add(index) + tokens = by_id.get(eos, set()) + if len(tokens) != 1: + raise ValueError("Native first EOS has missing or ambiguous tokenizer ID mapping") + token = next(iter(tokens)) + if not token or by_token[token] != {eos}: + raise ValueError("Native first EOS token maps to multiple IDs") + return token + + +def prepare_tokenizer(checkpoint: Path, engine: Path, destination: Path, raw: dict, + *, chat_template: str | None = None) -> list[int]: + """Create explicit derivative metadata; original source/engine files stay intact. + + Returns the full native EOS vector. Invalid source metadata is a + preparation failure, before bundle publication or runtime execution. + """ + tokenizer = _object(checkpoint / "tokenizer.json") + metadata = _object(checkpoint / "tokenizer_config.json") + if chat_template is not None: + if not isinstance(chat_template, str) or not chat_template.strip(): + raise ValueError("Explicit Edge chat template must be non-empty") + metadata["chat_template"] = chat_template + if not isinstance(metadata.get("chat_template"), str) or not metadata["chat_template"]: + raise ValueError("Nemotron-H native requires embedded tokenizer chat_template") + eos = native_eos_ids(checkpoint, raw) + for token_id in eos: + eos_token(tokenizer, token_id) + metadata["eos_token"] = eos_token(tokenizer, eos[0]) + destination.mkdir(parents=True) + # Runtime token IDs come from the byte-exact retained source vocabulary. + shutil.copy2(checkpoint / "tokenizer.json", destination / "tokenizer.json") + shutil.copy2(engine / "processed_chat_template.json", destination / "processed_chat_template.json") + (destination / "tokenizer_config.json").write_text( + json.dumps(metadata, ensure_ascii=False, indent=2) + "\n", encoding="utf-8" + ) + return eos diff --git a/families/nemotron_h/model.py b/families/nemotron_h/model.py index 7a3a72b02a..06ce9545b6 100644 --- a/families/nemotron_h/model.py +++ b/families/nemotron_h/model.py @@ -993,55 +993,82 @@ def _runtime_config( def build(request: "BuildRequest", writer: "BundleWriter") -> None: - """Build one Nemotron-H bundle.""" - if request.dynamic_kv_cache: - raise NotImplementedError("nemotron_h does not support dynamic_kv_cache") - - if request.image_height is not None: - raise NotImplementedError("nemotron_h does not support image_height") - - if request.image_width is not None: - raise NotImplementedError("nemotron_h does not support image_width") - - if request.video_num_frames is not None: - raise NotImplementedError("nemotron_h does not support video_num_frames") - - if request.max_batch_size != 1: - raise NotImplementedError("nemotron_h does not support max_batch_size") - - if request.context_parallel_size != 1: - raise ValueError("this family does not support context parallelism") - - if request.task != "text_generation": - raise ValueError("nemotron_h supports only task=text_generation") - - model_dir = Path(request.model_dir) - config = ModelConfig.from_dir(model_dir) - if str(config.model_type).lower() not in {"nemotron_h", "nemotron_hybrid"}: - raise ValueError(f"Nemotron-H does not support model_type={config.model_type!r}") - precision = str(request.precision).lower() - if precision not in {"fp32", "fp16", "bf16"}: - raise ValueError("Nemotron-H precision must be fp32, fp16, or bf16") - max_sequence_length = _positive_int( - request.max_sequence_length or min(config.max_position_embeddings, 256), - "max_sequence_length", - ) - if max_sequence_length > config.max_position_embeddings: - raise ValueError("Nemotron-H max_sequence_length exceeds checkpoint context capacity") - if request.quantization not in {None, "none"}: - raise NotImplementedError("Nemotron-H has no qualified family-owned quantized build") - if request.fp32_layers: - raise NotImplementedError("Nemotron-H does not expose mixed-precision layers") - parallel = ParallelConfig( - tp_size=_positive_int(request.tensor_parallel_size, "tensor_parallel_size") - ) - parallel.validate() - model = _NemotronHModel() - config.raw["_model_dir"] = str(model_dir) - weights = model.load_weights(str(model_dir), config) - writer.set_header(family="nemotron_h", task=request.task, backend=request.backend) - if parallel.enabled: - for rank in range(parallel.tp_size): + """Dispatch complete offload or execute the unchanged native implementation.""" + from .dispatch import build as dispatch_build + + def _build_native(request: "BuildRequest", writer: "BundleWriter") -> None: + """Build one Nemotron-H bundle.""" + if request.dynamic_kv_cache: + raise NotImplementedError("nemotron_h does not support dynamic_kv_cache") + + if request.image_height is not None: + raise NotImplementedError("nemotron_h does not support image_height") + + if request.image_width is not None: + raise NotImplementedError("nemotron_h does not support image_width") + + if request.video_num_frames is not None: + raise NotImplementedError("nemotron_h does not support video_num_frames") + + if request.max_batch_size != 1: + raise NotImplementedError("nemotron_h does not support max_batch_size") + + if request.context_parallel_size != 1: + raise ValueError("this family does not support context parallelism") + + if request.task != "text_generation": + raise ValueError("nemotron_h supports only task=text_generation") + + model_dir = Path(request.model_dir) + config = ModelConfig.from_dir(model_dir) + if str(config.model_type).lower() not in {"nemotron_h", "nemotron_hybrid"}: + raise ValueError(f"Nemotron-H does not support model_type={config.model_type!r}") + precision = str(request.precision).lower() + if precision not in {"fp32", "fp16", "bf16"}: + raise ValueError("Nemotron-H precision must be fp32, fp16, or bf16") + max_sequence_length = _positive_int( + request.max_sequence_length or min(config.max_position_embeddings, 256), + "max_sequence_length", + ) + if max_sequence_length > config.max_position_embeddings: + raise ValueError("Nemotron-H max_sequence_length exceeds checkpoint context capacity") + if request.quantization not in {None, "none"}: + raise NotImplementedError("Nemotron-H has no qualified family-owned quantized build") + # Native weight loading does not apply packed-source quantization scales. + # Auto/none controls cannot reinterpret an already quantized checkpoint. + if config.raw.get("quantization_config") is not None or any( + (model_dir / name).exists() + for name in ("hf_quant_config.json", "quantize_config.json", "quant_config.json") + ): + raise NotImplementedError( + "Nemotron-H native fallback cannot load a quantized checkpoint; " + "use the Edge builder on a supported platform" + ) + if request.fp32_layers: + raise NotImplementedError("Nemotron-H does not expose mixed-precision layers") + parallel = ParallelConfig( + tp_size=_positive_int(request.tensor_parallel_size, "tensor_parallel_size") + ) + parallel.validate() + model = _NemotronHModel() + config.raw["_model_dir"] = str(model_dir) + weights = model.load_weights(str(model_dir), config) + writer.set_header(family="nemotron_h", task=request.task, backend=request.backend) + if parallel.enabled: + for rank in range(parallel.tp_size): + plan = model.build_engine( + config, + weights, + max_sequence_length, + precision=precision, + quant_ctx=None, + verbose=bool(request.verbose), + debug_layer_outputs=False, + parallel_config=parallel.for_rank(rank), + ) + writer.add_bytes(f"engine.rank{rank}.plan", plan) + layout = "dual_profile" + else: plan = model.build_engine( config, weights, @@ -1050,37 +1077,35 @@ def build(request: "BuildRequest", writer: "BundleWriter") -> None: quant_ctx=None, verbose=bool(request.verbose), debug_layer_outputs=False, - parallel_config=parallel.for_rank(rank), + parallel_config=parallel, ) - writer.add_bytes(f"engine.rank{rank}.plan", plan) - layout = "dual_profile" - else: - plan = model.build_engine( - config, - weights, - max_sequence_length, - precision=precision, - quant_ctx=None, - verbose=bool(request.verbose), - debug_layer_outputs=False, - parallel_config=parallel, + writer.add_bytes("engine.plan", plan) + layout = "single" + writer.add_json( + "runtime.json", + _runtime_config( + model_dir, + config, + model, + precision=precision, + max_cache_length=max_sequence_length, + decoder_engine_layout=layout, + tensor_parallel_size=parallel.tp_size, + tensor_parallel_mode="tensor_parallel" if parallel.enabled else "single", + ), ) - writer.add_bytes("engine.plan", plan) - layout = "single" - writer.add_json( - "runtime.json", - _runtime_config( - model_dir, - config, - model, - precision=precision, - max_cache_length=max_sequence_length, - decoder_engine_layout=layout, - tensor_parallel_size=parallel.tp_size, - tensor_parallel_mode="tensor_parallel" if parallel.enabled else "single", - ), - ) - for filename in _BUNDLE_FILES: - path = model_dir / filename - if path.is_file(): - writer.add_bytes(filename, path.read_bytes()) + for filename in _BUNDLE_FILES: + path = model_dir / filename + if path.is_file(): + writer.add_bytes(filename, path.read_bytes()) + + dispatch_build(request, writer, _build_native) + + +def build_with_inputs(request, writer, execution) -> None: + """Keep the complete DFlash pair owned by the family ONNX adapter.""" + if execution.variant != "dflash" or tuple(x.role for x in execution.checkpoints) != ("draft",): + raise ValueError("Nemotron-H paired execution requires variant=dflash and one draft") + from .dispatch import build_dflash + + build_dflash(request, writer, execution.checkpoints[0].model_dir) diff --git a/families/nemotron_h/runtime/CMakeLists.txt b/families/nemotron_h/runtime/CMakeLists.txt index d6d06e7f2b..640b4c03ba 100644 --- a/families/nemotron_h/runtime/CMakeLists.txt +++ b/families/nemotron_h/runtime/CMakeLists.txt @@ -57,3 +57,16 @@ if(TRTMC_BUILD_TESTS) ) add_test(NAME ${test_name} COMMAND ${test_name}) endif() + +if(TARGET EdgeLLM::Core) + target_sources(trtmc_model_nemotron_h PRIVATE edge_llm/adapter.cpp edge_llm/device_link.cu) + target_compile_definitions(trtmc_model_nemotron_h PRIVATE TRTMC_HAS_EDGE_LLM=1) + target_link_libraries(trtmc_model_nemotron_h PRIVATE EdgeLLM::Core nlohmann_json::nlohmann_json) + set_target_properties(trtmc_model_nemotron_h PROPERTIES + CUDA_ARCHITECTURES "${EdgeLLM_CUDA_ARCHITECTURE}" + CUDA_SEPARABLE_COMPILATION ON CUDA_RESOLVE_DEVICE_SYMBOLS ON) + add_custom_command(TARGET trtmc_model_nemotron_h POST_BUILD + COMMAND ${CMAKE_COMMAND} -E copy_if_different + $ $ + VERBATIM) +endif() diff --git a/families/nemotron_h/runtime/chat_templates.cpp b/families/nemotron_h/runtime/chat_templates.cpp index bddc592966..3208745f4b 100644 --- a/families/nemotron_h/runtime/chat_templates.cpp +++ b/families/nemotron_h/runtime/chat_templates.cpp @@ -10,10 +10,16 @@ namespace trtmc { namespace { -std::string apply_chatml(const std::string& prompt, bool enable_thinking) { +std::string apply_chatml(const std::string& prompt, bool enable_thinking, + bool nemotron_format = false) { std::string r = "<|im_start|>user\n" + prompt + "<|im_end|>\n<|im_start|>assistant\n"; - if (!enable_thinking) + if (nemotron_format) { + // Modern Nemotron ChatML includes an empty system and a compact think suffix. + r = "<|im_start|>system\n<|im_end|>\n" + r; + r += enable_thinking ? "\n" : ""; + } else if (!enable_thinking) { r += "\n\n\n\n"; + } return r; } @@ -29,8 +35,12 @@ std::string apply_nemotron_h(const std::string& prompt, bool enable_thinking) { std::string nemotron_h_detect_chat_template_format(const std::string& jinja_template) { if (jinja_template.empty()) return {}; - if (jinja_template.find("<|im_start|>") != std::string::npos) + if (jinja_template.find("<|im_start|>") != std::string::npos) { + if (jinja_template.find("<|im_start|>system") != std::string::npos && + jinja_template.find("") != std::string::npos) + return "nemotron_chatml"; return "chatml"; + } if (jinja_template.find("") != std::string::npos) return "nemotron_h"; return {}; @@ -40,6 +50,8 @@ std::string nemotron_h_apply_chat_template(const std::string& format, const std: bool enable_thinking) { if (format.empty()) return prompt; + if (format == "nemotron_chatml") + return apply_chatml(prompt, enable_thinking, true); if (format == "chatml") return apply_chatml(prompt, enable_thinking); if (format == "nemotron_h") diff --git a/families/nemotron_h/runtime/edge_llm/adapter.cpp b/families/nemotron_h/runtime/edge_llm/adapter.cpp new file mode 100644 index 0000000000..3c439ae00d --- /dev/null +++ b/families/nemotron_h/runtime/edge_llm/adapter.cpp @@ -0,0 +1,320 @@ +/* + * SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. + * SPDX-License-Identifier: Apache-2.0 + */ +#include "families/nemotron_h/runtime/edge_llm/adapter.h" + +#include "families/nemotron_h/runtime/edge_llm/request.h" +#include "families/nemotron_h/runtime/edge_llm/tokenizer.h" + +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include + +namespace trtmc::nemotron_h::edge_llm { +namespace { +namespace fs = std::filesystem; + +/// Turn CUDA failures into caller-visible load or inference errors. +void check_cuda(cudaError_t result) { + if (result != cudaSuccess) + throw std::runtime_error(std::string("Nemotron-H Edge CUDA error: ") + + cudaGetErrorString(result)); +} + +/// Reject an engine built for a different local GPU or CUDA/TensorRT runtime. +void validate_target(const nlohmann::json& target) { + utsname host{}; + if (uname(&host) != 0) + throw std::runtime_error("Cannot identify Nemotron-H Edge runtime host"); + std::ifstream release("/etc/os-release"); + std::string line, os_version; + while (std::getline(release, line)) { + if (line.rfind("VERSION_ID=", 0) == 0) { + os_version = line.substr(11); + if (os_version.size() >= 2 && os_version.front() == char(34) && + os_version.back() == char(34)) + os_version = os_version.substr(1, os_version.size() - 2); + } + } + int device = 0, cuda_version = 0; + check_cuda(cudaGetDevice(&device)); + check_cuda(cudaRuntimeGetVersion(&cuda_version)); + cudaDeviceProp gpu{}; + check_cuda(cudaGetDeviceProperties(&gpu, device)); + const int trt_version = getInferLibVersion(); + const std::string trt = + std::to_string(trt_version / 10000) + "." + std::to_string((trt_version % 10000) / 100) + + "." + std::to_string(trt_version % 100) + "." + std::to_string(getInferLibBuildVersion()); + const std::string cuda = + std::to_string(cuda_version / 1000) + "." + std::to_string((cuda_version % 1000) / 10); + if (target.at("os") != "linux" || target.at("os_version") != os_version || + target.at("arch") != host.machine || target.at("sm") != gpu.major * 10 + gpu.minor || + target.at("cuda_version") != cuda || target.at("tensorrt_version") != trt) + throw std::runtime_error( + "Nemotron-H Edge bundle requires its build GPU and CUDA/TensorRT stack"); +} + +/// Own extracted engine/checkpoint files until after the Edge runtime is destroyed. +class Artifacts { + public: + explicit Artifacts(const BundleReader& bundle, const nlohmann::json& marker) { + std::set names; + for (const auto& entry : marker.at("artifacts")) { + const auto name = entry.get(); + if (!safe_artifact_path(name) || !names.insert(name).second || + !bundle.find_section(name)) + throw std::runtime_error("Invalid Nemotron-H Edge artifact: " + name); + } + const bool paired = marker.value("execution_variant", "autoregressive") == "dflash"; + const std::vector plans = + paired ? std::vector{"spec_base.engine", "spec_draft.engine", + "base_config.json", "draft_config.json", + "embedding.safetensors"} + : std::vector{"llm.engine", "config.json"}; + for (const auto& plan : plans) { + const auto required = "edge_llm/engine/" + plan; + if (!names.count(required) || bundle.find_section(required)->length == 0) + throw std::runtime_error("Required Nemotron-H Edge artifact missing: " + required); + } + for (const auto* required : + {"edge_llm/engine/tokenizer.json", "edge_llm/engine/tokenizer_config.json", + "edge_llm/engine/processed_chat_template.json", "edge_llm/checkpoint/config.json", + "edge_llm/runtime_tokenizer/tokenizer.json", + "edge_llm/runtime_tokenizer/tokenizer_config.json", + "edge_llm/runtime_tokenizer/processed_chat_template.json"}) + if (!names.count(required) || bundle.find_section(required)->length == 0) + throw std::runtime_error( + std::string("Required Nemotron-H Edge artifact missing: ") + required); + std::string pattern = (fs::temp_directory_path() / "trtmc-nemotron-h-edge-XXXXXX").string(); + if (!mkdtemp(pattern.data())) + throw std::runtime_error("Cannot create Nemotron-H Edge artifact directory"); + root_ = pattern; + try { + for (const auto& name : names) { + const auto destination = root_ / name; + fs::create_directories(destination.parent_path()); + std::ofstream output(destination, std::ios::binary); + bundle.copy_section(name, output); + output.close(); + if (!output) + throw std::runtime_error("Cannot extract Nemotron-H Edge artifact: " + name); + } + } catch (...) { + cleanup(); + throw; + } + } + ~Artifacts() { cleanup(); } + Artifacts(const Artifacts&) = delete; + Artifacts& operator=(const Artifacts&) = delete; + std::string engine() const { return (root_ / "edge_llm/engine").string(); } + std::string tokenizer() const { return (root_ / "edge_llm/runtime_tokenizer").string(); } + std::string checkpoint() const { return (root_ / "edge_llm/checkpoint").string(); } + + private: + void cleanup() noexcept { + std::error_code ignored; + fs::remove_all(root_, ignored); + } + fs::path root_; +}; + +/// Close the plugin handle after runtime destruction; registrations remain mapped. +struct CloseLibrary { + void operator()(void* handle) const noexcept { + if (handle) + dlclose(handle); + } +}; + +/// Initialize the CMake-installed adjacent plugin without process-global environment mutation. +std::unique_ptr load_plugin() { + Dl_info location{}; + if (!dladdr(reinterpret_cast(&create), &location) || !location.dli_fname) + throw std::runtime_error("Cannot locate Nemotron-H family library"); + const auto path = + fs::absolute(location.dli_fname).parent_path() / "libNvInfer_edgellm_plugin.so"; + std::unique_ptr plugin( + dlopen(path.c_str(), RTLD_NOW | RTLD_GLOBAL | RTLD_NODELETE)); + if (!plugin) + throw std::runtime_error("Cannot load CMake-installed Edge plugin: " + + std::string(dlerror())); + using Initialize = bool (*)(void*, const char*); + auto initialize = reinterpret_cast(dlsym(plugin.get(), "initEdgellmPlugins")); + if (!initialize || !initialize(static_cast(&trt_edgellm::gLogger), "")) + throw std::runtime_error("Cannot initialize Nemotron-H Edge plugin"); + return plugin; +} + +/// Stream ownership is independent of construction success and outlives the Edge instance. +class Stream { + public: + Stream() { check_cuda(cudaStreamCreateWithFlags(&value_, cudaStreamNonBlocking)); } + ~Stream() { cudaStreamDestroy(value_); } + Stream(const Stream&) = delete; + Stream& operator=(const Stream&) = delete; + cudaStream_t get() const { return value_; } + + private: + cudaStream_t value_{nullptr}; +}; + +/// Drain all queued work before request storage destruction or mutex release. +class RequestDrain { + public: + explicit RequestDrain(cudaStream_t stream) : stream_(stream) {} + ~RequestDrain() noexcept { cudaStreamSynchronize(stream_); } + void checked() { check_cuda(cudaStreamSynchronize(stream_)); } + + private: + cudaStream_t stream_; +}; + +/// The paired engine ABI fixes a linear block16 speculative schedule. +std::optional +drafting_config(const nlohmann::json& marker) { + if (marker.value("execution_variant", "autoregressive") != "dflash") + return std::nullopt; + trt_edgellm::rt::SpecDecodeDraftingConfig config{}; + config.draftingTopK = 1; + config.draftingStep = 1; + config.verifySize = 16; + config.dflashBlockSize = 16; + return config; +} + +/// Use public artifact injection; its coordinator retains this supplied tokenizer. +trt_edgellm::rt::ModelArtifacts +runtime_artifacts(const Artifacts& files, const nlohmann::json& marker, cudaStream_t stream) { + auto artifacts = trt_edgellm::rt::ModelArtifacts::loadFromEngineDir( + files.engine(), drafting_config(marker), files.checkpoint(), "", stream); + artifacts.tokenizer = load_tokenizer(files.tokenizer(), + marker.at("native_eos_token_ids").get>()); + artifacts.deployment.base.eosTokenIds = + marker.at("native_eos_token_ids").get>(); + return artifacts; +} + +/// Read the original source template resolved into family-owned runtime metadata. +std::string native_chat_format(const BundleReader& bundle) { + const auto bytes = bundle.read_section("edge_llm/runtime_tokenizer/tokenizer_config.json"); + const auto metadata = nlohmann::json::parse(bytes.begin(), bytes.end()); + const auto text = metadata.at("chat_template").get(); + if (text.empty()) + throw std::runtime_error("Nemotron-H Edge requires resolved source chat_template"); + return nemotron_h_detect_chat_template_format(text); +} + +/// Reuse the original family BPE implementation, including its special postprocessor. +std::unique_ptr native_tokenizer(const BundleReader& bundle) { + const auto bytes = bundle.read_section("edge_llm/checkpoint/tokenizer.json"); + return CreateBpeTokenizer(reinterpret_cast(bytes.data()), bytes.size(), true); +} + +/// Thin persistent Edge API adapter; serialization prevents concurrent use of Edge request state. +class EdgeTask final : public ITextGeneration { + public: + EdgeTask(const BundleReader& bundle, const nlohmann::json& marker) + : artifacts_(bundle, marker), native_tokenizer_(native_tokenizer(bundle)), + plugin_(load_plugin()), + runtime_(runtime_artifacts(artifacts_, marker, stream_.get()), artifacts_.engine(), "", + std::unordered_map{}, drafting_config(marker), + stream_.get()), + chat_format_(native_chat_format(bundle)), + capacity_(marker.at("max_sequence_length").get()), + input_limit_(marker.at("max_input_length").get()), + bos_id_(marker.at("native_bos_token_id").get()), + paired_(marker.value("execution_variant", "autoregressive") == "dflash") {} + + std::int32_t default_max_new_tokens() const override { return kDefaultMaxNewTokens; } + + /// Drain work from failed requests before destroying the runtime and its weight buffers. + ~EdgeTask() override { cudaStreamSynchronize(stream_.get()); } + + /// Invoke Edge once; failures propagate without attempting native inference. + TextResult generate(const std::string& prompt, const TextGenerationConfig& config) override { + auto request = make_request(prompt, config, chat_format_, *native_tokenizer_, bos_id_); + // The pinned DFlash decoder forces greedy verification. Reject sampling + // rather than silently replacing caller-requested stochastic semantics. + if (paired_ && request.temperature != 0.0F) + throw std::invalid_argument("Nemotron-H DFlash supports greedy generation only"); + const auto count = request.preTokenizedInputIds.front().size(); + if (count == 0) + return {}; + if (count > static_cast(input_limit_)) + throw std::invalid_argument("Nemotron-H Edge input exceeds bundle capacity"); + std::lock_guard lock(mutex_); + RequestDrain drain(stream_.get()); + validate_capacity(static_cast(count), input_limit_, capacity_, + request.maxGenerateLength); + trt_edgellm::rt::LLMGenerationResponse response{}; + if (!runtime_.handleRequest(request, response, stream_.get())) + throw std::runtime_error("Nemotron-H Edge generation failed"); + drain.checked(); + validate_response(response, input_limit_, capacity_, request.maxGenerateLength, + static_cast(count)); + // This API does not expose per-request stage times; zero means unavailable. + return {native_tokenizer_->decode(response.outputIds.front()), + std::move(response.outputIds.front())}; + } + + private: + // Reverse destruction order keeps weights, plugin and stream alive throughout Edge teardown. + Artifacts artifacts_; + std::unique_ptr native_tokenizer_; + std::unique_ptr plugin_; + Stream stream_; + trt_edgellm::rt::LLMInferenceRuntime runtime_; + std::string chat_format_; + int capacity_; + int input_limit_; + int bos_id_; + bool paired_; + std::mutex mutex_; +}; +} // namespace + +ITask* create(const BundleReader& bundle) { + const auto bytes = bundle.read_section("edge_llm.json"); + const auto marker = nlohmann::json::parse(bytes.begin(), bytes.end()); + if (marker.at("version") != 1 || marker.at("edge_revision") != kRevision || + marker.at("max_sequence_length").get() <= 1 || + marker.at("max_input_length").get() <= 0 || + marker.at("max_input_length").get() > marker.at("max_sequence_length").get() || + marker.at("max_batch_size") != 1 || marker.at("precision") != "fp16" || + !marker.at("artifacts").is_array() || marker.at("tokenizer_policy") != "native_full_eos" || + !marker.at("native_bos_token_id").is_number_integer() || + !marker.at("native_eos_token_ids").is_array() || marker.at("native_eos_token_ids").empty()) + throw std::runtime_error("Invalid Nemotron-H Edge bundle contract"); + const auto variant = marker.value("execution_variant", "autoregressive"); + if ((variant != "autoregressive" && variant != "dflash") || + (variant == "dflash" && (marker.value("builder_flow", "") != "onnx" || + marker.value("dflash_block_size", 0) != 16))) + throw std::runtime_error("Invalid Nemotron-H Edge execution variant"); + std::set stops; + for (const auto& id : marker.at("native_eos_token_ids")) { + if (!id.is_number_integer() || id.get() < 0 || !stops.insert(id.get()).second) + throw std::runtime_error("Invalid Nemotron-H Edge full EOS set"); + } + validate_target(marker.at("target")); + return new EdgeTask(bundle, marker); +} + +} // namespace trtmc::nemotron_h::edge_llm diff --git a/families/nemotron_h/runtime/edge_llm/adapter.h b/families/nemotron_h/runtime/edge_llm/adapter.h new file mode 100644 index 0000000000..b62c2fda39 --- /dev/null +++ b/families/nemotron_h/runtime/edge_llm/adapter.h @@ -0,0 +1,15 @@ +/* + * SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. + * SPDX-License-Identifier: Apache-2.0 + */ +#pragma once + +#include "trtmc/bundle.h" +#include "trtmc/task.h" + +namespace trtmc::nemotron_h::edge_llm { + +/// Create a persistent Edge task from a self-contained bundle; throws on load failure. +ITask* create(const BundleReader& bundle); + +} // namespace trtmc::nemotron_h::edge_llm diff --git a/families/nemotron_h/runtime/edge_llm/contract.h b/families/nemotron_h/runtime/edge_llm/contract.h new file mode 100644 index 0000000000..9a1c63c029 --- /dev/null +++ b/families/nemotron_h/runtime/edge_llm/contract.h @@ -0,0 +1,60 @@ +/* + * SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. + * SPDX-License-Identifier: Apache-2.0 + */ +#pragma once + +#include "trtmc/task.h" + +#include +#include +#include +#include + +namespace trtmc::nemotron_h::edge_llm { + +/// Match the native Nemotron-H task default; insufficient capacity is an explicit error. +inline constexpr int kDefaultMaxNewTokens = 128; + +inline constexpr const char* kRevision = "e8b29522938901f6df19ebeedd4b69bc8edbcd97"; + +/// Return whether an artifact is a normalized file below one of the three Edge roots. +inline bool safe_artifact_path(const std::string& name) { + if (name.find('\\') != std::string::npos || name.find('\0') != std::string::npos) + return false; + const std::filesystem::path path(name); + if (path.is_absolute() || path.filename().empty()) + return false; + for (const auto& part : path) + if (part == "." || part == "..") + return false; + return path.generic_string() == name && + (name.rfind("edge_llm/engine/", 0) == 0 || name.rfind("edge_llm/checkpoint/", 0) == 0 || + name.rfind("edge_llm/runtime_tokenizer/", 0) == 0); +} + +/// Reject invalid sampling settings and controls with no equivalent Edge request API. +inline void validate_generation(const TextGenerationConfig& c) { + if (!std::isfinite(c.temperature) || c.temperature < 0 || !std::isfinite(c.top_p) || + c.top_p < 0 || c.top_p > 1 || c.top_k < 0) + throw std::invalid_argument("Invalid Nemotron-H Edge sampling parameters"); + if (c.min_p != 0 || c.seed != -1 || c.eos_token_id != -1 || c.repetition_penalty != 1 || + !c.lora_adapter_id.empty() || c.stop_on_boxed_answer || + (c.text_generation_mode != "auto" && c.text_generation_mode != "autoregressive") || + c.source_language_token_id != -1 || c.forced_bos_token_id != -1 || c.guidance_scale != -1 || + c.cfg_scale != -1 || c.num_steps != -1 || c.sde_gamma != -1 || !c.initial_latents.empty() || + !c.condition_latents.empty() || !c.condition_mask.empty() || !c.sampling_steps.empty() || + !c.sde_noises.empty() || c.block_length != 0 || c.confidence_threshold != -1) + throw std::invalid_argument( + "Requested generation controls are unsupported by Nemotron-H Edge"); +} + +/// Enforce prompt and total capacity without allowing Edge to silently clip generation. +inline void validate_capacity(int prompt_tokens, int input_limit, int capacity, + std::int64_t generated_tokens) { + if (prompt_tokens <= 0 || prompt_tokens > input_limit || generated_tokens <= 0 || + generated_tokens > static_cast(capacity) - prompt_tokens) + throw std::invalid_argument("Nemotron-H Edge prompt and generation exceed bundle capacity"); +} + +} // namespace trtmc::nemotron_h::edge_llm diff --git a/families/nemotron_h/runtime/edge_llm/device_link.cu b/families/nemotron_h/runtime/edge_llm/device_link.cu new file mode 100644 index 0000000000..1ea4768d60 --- /dev/null +++ b/families/nemotron_h/runtime/edge_llm/device_link.cu @@ -0,0 +1,5 @@ +/* + * SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. + * SPDX-License-Identifier: Apache-2.0 + */ +// Enable the final CUDA device-link step for Edge's static runtime dependencies. diff --git a/families/nemotron_h/runtime/edge_llm/request.h b/families/nemotron_h/runtime/edge_llm/request.h new file mode 100644 index 0000000000..ff0901d2ea --- /dev/null +++ b/families/nemotron_h/runtime/edge_llm/request.h @@ -0,0 +1,68 @@ +/* + * SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. + * SPDX-License-Identifier: Apache-2.0 + */ +#pragma once +#include "families/nemotron_h/runtime/chat_templates.h" +#include "families/nemotron_h/runtime/edge_llm/contract.h" +#include "families/nemotron_h/runtime/tokenizer.h" + +#include +#include + +namespace trtmc::nemotron_h::edge_llm { + +/// Apply the existing native family renderer, then pass already framed raw text. +inline trt_edgellm::rt::LLMGenerationRequest make_request(const std::string& prompt, + const TextGenerationConfig& config, + const std::string& chat_format, + const ITokenizer& tokenizer, int bos_id) { + validate_generation(config); + const auto formatted = + config.use_chat_template && !chat_format.empty() + ? nemotron_h_apply_chat_template(chat_format, prompt, config.enable_thinking) + : prompt; + trt_edgellm::rt::LLMGenerationRequest request{}; + request.requests.resize(1); + request.requests.front().messages.push_back({"user", {{"text", formatted}}}); + auto ids = tokenizer.encode(formatted); + if (config.use_chat_template && !chat_format.empty() && ids.size() >= 2 && bos_id >= 0 && + ids[0] == bos_id && ids[1] == bos_id) + ids.erase(ids.begin()); + request.preTokenizedInputIds.push_back(std::move(ids)); + request.applyChatTemplate = false; + request.addGenerationPrompt = false; + request.enableThinking = false; + const bool greedy = config.temperature < 1e-6F || config.top_p <= 0 || + (config.top_k <= 1 && config.top_p >= 1.0F - 1e-6F); + request.temperature = greedy ? 0.0F : config.temperature; + request.topK = config.top_k; + request.topP = config.top_p <= 0 ? 1.0F : config.top_p; + request.maxGenerateLength = + config.max_new_tokens > 0 ? config.max_new_tokens : kDefaultMaxNewTokens; + return request; +} + +/// Validate the ORIGINAL requested budget even after early EOS or upstream clipping. +inline void validate_response(const trt_edgellm::rt::LLMGenerationResponse& response, + std::int32_t input_limit, std::int32_t capacity, + std::int64_t requested_budget, int submitted_count) { + namespace rt = trt_edgellm::rt; + if (response.outputIds.size() != 1 || response.outputTexts.size() != 1 || + response.outputIds.front().empty() || requested_budget <= 0 || + response.outputIds.front().size() > static_cast(requested_budget) || + response.finishReasons.size() != 1 || + (response.finishReasons.front() != rt::FinishReason::kEndId && + response.finishReasons.front() != rt::FinishReason::kLength)) + throw std::runtime_error("Nemotron-H Edge generation did not complete successfully"); + if (response.inputTokenCounts.size() != 1 || + response.inputTokenCounts.front() != submitted_count) + throw std::runtime_error("Nemotron-H Edge returned invalid input counts"); + validate_capacity(response.inputTokenCounts.front(), input_limit, capacity, requested_budget); + if (response.finishReasons.front() == rt::FinishReason::kLength && + response.outputIds.front().size() != static_cast(requested_budget)) + throw std::runtime_error( + "Nemotron-H Edge silently shortened the requested generation budget"); +} + +} // namespace trtmc::nemotron_h::edge_llm diff --git a/families/nemotron_h/runtime/edge_llm/tokenizer.h b/families/nemotron_h/runtime/edge_llm/tokenizer.h new file mode 100644 index 0000000000..d835ebb940 --- /dev/null +++ b/families/nemotron_h/runtime/edge_llm/tokenizer.h @@ -0,0 +1,29 @@ +/* + * SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. + * SPDX-License-Identifier: Apache-2.0 + */ +#pragma once +#include +#include +#include +#include + +namespace trtmc::nemotron_h::edge_llm { + +/// Load separately derived metadata through the public API and enforce native full EOS. +inline std::unique_ptr +load_tokenizer(const std::filesystem::path& directory, const std::vector& native_eos) { + auto tokenizer = std::make_unique(); + if (native_eos.empty() || native_eos.front() < 0 || !tokenizer->loadFromHF(directory) || + tokenizer->getEosId() != native_eos.front()) + throw std::runtime_error("Nemotron-H Edge derived tokenizer differs from native full EOS"); + for (const auto id : native_eos) + if (id < 0 || tokenizer->idToPiece(id, false).empty()) + throw std::runtime_error("Nemotron-H Edge EOS has no tokenizer vocabulary mapping"); + tokenizer->setAdditionalEosIds(std::vector(native_eos.begin() + 1, native_eos.end())); + if (tokenizer->getEosIds() != native_eos) + throw std::runtime_error("Nemotron-H Edge tokenizer retained an incorrect full EOS set"); + return tokenizer; +} + +} // namespace trtmc::nemotron_h::edge_llm diff --git a/families/nemotron_h/runtime/plugin.cpp b/families/nemotron_h/runtime/plugin.cpp index 7ab2faaff9..fb89bfb360 100644 --- a/families/nemotron_h/runtime/plugin.cpp +++ b/families/nemotron_h/runtime/plugin.cpp @@ -4,6 +4,9 @@ */ #include "families/nemotron_h/runtime/chat_templates.h" +#ifdef TRTMC_HAS_EDGE_LLM +#include "families/nemotron_h/runtime/edge_llm/adapter.h" +#endif #include "families/nemotron_h/runtime/distributed_runtime.h" #include "families/nemotron_h/runtime/hybrid_state.h" #include "families/nemotron_h/runtime/pipeline.h" @@ -215,5 +218,12 @@ ITask* create(const FamilyContext& context) { extern "C" trtmc::ITask* trtmc_create_family(const trtmc::FamilyContext& context) { if (context.kv_cache_size_bytes != 0) throw std::invalid_argument("nemotron_h does not support --kv-cache-size"); + if (context.reader.find_section("edge_llm.json")) { +#ifdef TRTMC_HAS_EDGE_LLM + return trtmc::nemotron_h::edge_llm::create(context.reader); +#else + throw std::runtime_error("Nemotron-H Edge bundle requires TRTMC_ENABLE_EDGELLM=ON"); +#endif + } return trtmc::nemotron_h::create(context); } diff --git a/families/nemotron_h/tests/test_e2e.py b/families/nemotron_h/tests/test_e2e.py index d56f55ec28..0d33d565d2 100644 --- a/families/nemotron_h/tests/test_e2e.py +++ b/families/nemotron_h/tests/test_e2e.py @@ -14,6 +14,7 @@ import shutil import subprocess from pathlib import Path +from unittest.mock import Mock import pytest @@ -133,7 +134,7 @@ def _thresholds(case_name: str) -> dict[str, float]: return thresholds -def _build_bundle(manifest: dict, model_dir: Path, bundle: Path) -> None: +def _build_bundle(manifest: dict, model_dir: Path, bundle: Path, *, execution=None) -> None: quantization = manifest.get("quantization") assert quantization is None or isinstance(quantization, str) fp32_layers = tuple(manifest.get("fp32_layers", ())) @@ -148,7 +149,8 @@ def _build_bundle(manifest: dict, model_dir: Path, bundle: Path) -> None: tensor_parallel_size=manifest["tensor_parallel_size"], quantization=quantization, fp32_layers=fp32_layers, - ) + ), + execution=execution, ) assert bundle.is_file() and bundle.stat().st_size > 0, bundle @@ -254,7 +256,14 @@ def _render_prompt(tokenizer, prompt: str, case: dict): if "enable_thinking" in case: options["enable_thinking"] = case["enable_thinking"] messages = [] - if case.get("enable_thinking") is False: + template = getattr(tokenizer, "chat_template", "") or "" + modern_chatml = ( + isinstance(template, str) + and "<|im_start|>system" in template + and "" in template + ) + # The modern source template honors enable_thinking itself, without /no_think. + if case.get("enable_thinking") is False and not modern_chatml: messages.append({"role": "system", "content": "/no_think"}) messages.append({"role": "user", "content": prompt}) rendered = tokenizer.apply_chat_template( @@ -339,7 +348,7 @@ def _hf_reference( actual_ids: list[int], torch, ) -> tuple[list[int], str, float | None, str]: - from transformers import AutoModelForCausalLM, AutoTokenizer + from transformers import AutoConfig, AutoModelForCausalLM, AutoTokenizer, GenerationConfig trust_remote_code = bool(manifest.get("trust_remote_code", False)) tokenizer = AutoTokenizer.from_pretrained( @@ -357,17 +366,148 @@ def _hf_reference( "bf16": torch.bfloat16, } assert reference_precision in dtypes, reference_precision - model = ( - AutoModelForCausalLM.from_pretrained( + reference_options = {} + if "reference_use_mamba_kernels" in case: + value = case["reference_use_mamba_kernels"] + assert isinstance(value, bool), "reference_use_mamba_kernels must be boolean" + reference_options["use_mamba_kernels"] = value + if case.get("reference_decode_modelopt_nvfp4", False): + # The declared BF16 oracle needs decoded weights, not packed NVFP4 bytes. + # This does not emulate compiled activation or KV rounding. + from modelopt.torch.quantization.qtensor import NVFP4QTensor + from modelopt.torch.export.quant_utils import QUANTIZATION_FP8, from_quantized_weight + from safetensors.torch import load_file + + assert not trust_remote_code and reference_precision == "bf16" + quant = json.loads((model_dir / "hf_quant_config.json").read_text())["quantization"] + assert quant["quant_algo"] in {"NVFP4", "MIXED_PRECISION"} + mixed = quant["quant_algo"] == "MIXED_PRECISION" + layers = quant.get("quantized_layers", {}) + if mixed: + assert layers and all( + policy["quant_algo"] == "FP8" + or (policy["quant_algo"] == "W4A16_NVFP4" and policy["group_size"] == 16) + for policy in layers.values() + ) + else: + assert quant["group_size"] == 16 + state = {} + for shard in sorted(model_dir.glob("*.safetensors")): + tensors = load_file(str(shard), device="cpu") + assert not state.keys() & tensors.keys(), "duplicate checkpoint tensors" + state.update(tensors) + packed = {key for key, value in state.items() if value.dtype == torch.uint8} + assert packed, "NVFP4 reference requires actual packed weights" + fp8 = { + key for key, value in state.items() + if key.endswith(".weight") and value.dtype == torch.float8_e4m3fn + } + quantized = packed | fp8 + if mixed: + assert quantized == {key + ".weight" for key in layers} + assert all(layers[key.removesuffix(".weight")]["quant_algo"] == "FP8" for key in fp8) + assert all( + layers[key.removesuffix(".weight")]["quant_algo"] == "W4A16_NVFP4" + for key in packed + ) + else: + assert not fp8 + for key in packed: + assert key.endswith(".weight"), key + weight = state[key] + assert weight.ndim == 2 and weight.shape[-1] % 8 == 0, key + prefix = key.removesuffix("weight") + scale = state[prefix + "weight_scale"] + double_scale = state[prefix + "weight_scale_2"] + shape = (weight.shape[0], weight.shape[1] * 2) + assert scale.dtype == torch.float8_e4m3fn + assert scale.shape == (shape[0], shape[1] // 16), key + assert torch.isfinite(scale.float()).all() and (scale.float() >= 0).all() + assert double_scale.numel() == 1 and torch.isfinite(double_scale).all() + assert (double_scale > 0).all() + state[key] = NVFP4QTensor(shape, torch.bfloat16, weight).dequantize( + dtype=torch.bfloat16, scale=scale, double_scale=double_scale, + block_sizes={-1: 16}, fast=False, + ) + assert state[key].shape == shape and torch.isfinite(state[key]).all(), key + for key in fp8: + weight = state[key] + scale = state[key.removesuffix("weight") + "weight_scale"] + assert weight.ndim == 2 and scale.numel() == 1, key + assert torch.isfinite(scale).all() and (scale > 0).all(), key + state[key] = from_quantized_weight( + weight, scale, QUANTIZATION_FP8, torch.bfloat16, + ) + assert state[key].shape == weight.shape and torch.isfinite(state[key]).all(), key + for key in list(state): + if key.endswith((".weight_scale", ".weight_scale_2", ".input_scale")): + assert key.rsplit(".", 1)[0] + ".weight" in quantized, key + scale = state.pop(key) + assert torch.isfinite(scale.float()).all() and (scale.float() >= 0).all(), key + # ModelOpt exports FP8 KV calibration buffers as k_scale/v_scale. + # The floating BF16 oracle does not use quantized KV; these are not weights. + kv_scales = { + key for key in state if key.endswith((".k_proj.k_scale", ".v_proj.v_scale")) + } + if kv_scales: + assert quant["kv_cache_quant_algo"] == "FP8" + k_prefixes = { + key.removesuffix(".k_proj.k_scale") + for key in kv_scales if key.endswith(".k_proj.k_scale") + } + v_prefixes = { + key.removesuffix(".v_proj.v_scale") + for key in kv_scales if key.endswith(".v_proj.v_scale") + } + assert k_prefixes == v_prefixes, "unpaired FP8 KV calibration" + for key in kv_scales: + scale = state[key] + weight = state[key.rsplit(".", 1)[0] + ".weight"] + assert weight.is_floating_point() and weight.ndim == 2, key + assert scale.dtype == torch.float32 and scale.numel() == 1, key + assert torch.isfinite(scale).all() and (scale > 0).all(), key + del state[key] + config = AutoConfig.from_pretrained( + model_dir, local_files_only=True, trust_remote_code=False, **reference_options, + ) + embedded = getattr(config, "quantization_config", None) + if mixed: + assert embedded["quant_method"] == "modelopt" + assert embedded["quant_algo"] == quant["quant_algo"] + assert embedded["quantized_layers"] == layers + # All declared packed/FP8 matrices are now independently decoded. + # Do not ask HF to quantize the already decoded BF16 state again. + del config.quantization_config + else: + assert not embedded + # Keep Transformers' official checkpoint-name conversion (backbone -> model). + from transformers import NemotronHForCausalLM + + assert isinstance(config, NemotronHForCausalLM.config_class) + model, loading = NemotronHForCausalLM.from_pretrained( + None, config=config, state_dict=state, torch_dtype=torch.bfloat16, + attn_implementation="eager", output_loading_info=True, + ) + assert all(not loading.get(key) for key in ( + "missing_keys", "unexpected_keys", "mismatched_keys", "error_msgs", + )), loading + if (model_dir / "generation_config.json").is_file(): + model.generation_config = GenerationConfig.from_pretrained( + model_dir, local_files_only=True, + ) + del state, tensors + else: + model = AutoModelForCausalLM.from_pretrained( model_dir, local_files_only=True, trust_remote_code=trust_remote_code, torch_dtype=dtypes[reference_precision], attn_implementation="eager", + **reference_options, ) - .eval() - .to("cuda") - ) + reference_device = case.get("reference_device", "cuda") + assert reference_device in {"cpu", "cuda"}, reference_device + model = model.eval().to(reference_device) inputs = _render_prompt(tokenizer, prompt, case).to(model.device) prompt_ids = inputs["input_ids"][0].tolist() if "expected_prompt_token_ids" in case: @@ -549,6 +689,25 @@ def test_reference_routes_keep_tp4_on_golden_and_single_gpu_on_hf() -> None: '"use_cache": False', ): assert required in source + for template, modern in ( + ("", False), + ("<|im_start|>", False), + ("<|im_start|>system ", True), + ): + tokenizer = Mock(chat_template=template) + tokenizer.apply_chat_template.return_value = "rendered" + _render_prompt( + tokenizer, "prompt", {"use_chat_template": True, "enable_thinking": False} + ) + messages = [{"role": "user", "content": "prompt"}] + if not modern: + messages.insert(0, {"role": "system", "content": "/no_think"}) + tokenizer.apply_chat_template.assert_called_once_with( + messages, tokenize=False, add_generation_prompt=True, enable_thinking=False + ) + tokenizer.assert_called_once_with( + "rendered", return_tensors="pt", add_special_tokens=False + ) _, case = _CASES["nemotron-h-nano-9b-tp4"] assert _reference_backend(case) == "golden_snapshot" assert _golden_reference(case) == "Paris" diff --git a/families/nemotron_h/tests/test_runtime_contract.py b/families/nemotron_h/tests/test_runtime_contract.py index 25a6d58e56..aa18538ad5 100644 --- a/families/nemotron_h/tests/test_runtime_contract.py +++ b/families/nemotron_h/tests/test_runtime_contract.py @@ -63,3 +63,11 @@ def test_runtime_keeps_checkpoint_prompt_and_builder_policies() -> None: source = path.read_text(encoding="utf-8") assert "builder_optimization_level = 3" in source assert "builder_optimization_level = 1" not in source + + # Edge has native scalar prefill even when the optimized head80 SSD is absent. + from families.nemotron_h.dispatch import EDGE_DISPATCH, platform_matches + + for sm in (80, 120): + assert platform_matches({"mamba_head_dim": 80}, {"sm": sm}) + assert ("linux", "x86_64", sm, "fp16") in EDGE_DISPATCH + assert ("linux", "x86_64", 80, "nvfp4") not in EDGE_DISPATCH diff --git a/website/docs/features/model-families.md b/website/docs/features/model-families.md index ae49254cc2..bd2ca81ddf 100644 --- a/website/docs/features/model-families.md +++ b/website/docs/features/model-families.md @@ -109,6 +109,23 @@ for exact revisions, capacities, sampling controls and quality evidence. The local paired qualification is not a registered manifest case and does not imply CI coverage of that pair or statistical sampling equivalence. +### Nemotron-H Edge-LLM execution + +The Nemotron-H family owns optional pinned Edge-LLM whole-network offload. +Ordinary compatible text builds use the experimental builder; the explicit +Lightning NVFP4/DFlash pair uses ONNX with +`--execution-variant dflash --companion draft=/path/to/draft`. +Native fallback is attempted only during ordinary preparation; it cannot +interpret unsupported packed checkpoints or replace a requested pair. + +See the [owning Nemotron-H recipe](https://github.com/NVIDIA/TensorRT-Model-Connect/blob/main/families/nemotron_h/EDGE_LLM.md) +for the six recorded ordinary profiles, the qualified greedy DFlash pair, +immutable revisions and unchanged quality gates. These are bounded historical +local qualifications, not catalog-wide or current-head CI proof. The separate +direct-Edge9B-NVFP4 quality failure and longer-context failures remain open. +The original plain9B case is registered in the owning E2E inventory; the other +exact profiles are not implied to run in CI. + ## Runtime and validation The directory name is also the runtime DSO identity: From fc5ba7196011f3ca10f8df0aac2b07428fc54187 Mon Sep 17 00:00:00 2001 From: Joshua Calafato Date: Mon, 21 Sep 2026 17:14:32 +0000 Subject: [PATCH 2/9] fix(nemotron-h): enforce Edge runtime contracts Reject unknown source chat framing instead of silently sending raw prompts. Remove diagnostics only after successful ordinary or DFlash publication. Signed-off-by: Joshua Calafato --- families/nemotron_h/dispatch.py | 2 ++ families/nemotron_h/runtime/edge_llm/adapter.cpp | 5 ++++- 2 files changed, 6 insertions(+), 1 deletion(-) diff --git a/families/nemotron_h/dispatch.py b/families/nemotron_h/dispatch.py index 50e1401a71..99a8363e7d 100644 --- a/families/nemotron_h/dispatch.py +++ b/families/nemotron_h/dispatch.py @@ -99,6 +99,7 @@ def build(request, writer, native) -> None: else: if adapter is not None: edge_llm.publish(request, writer, files, marker) + log_path.unlink() return log_path.unlink() try: @@ -134,3 +135,4 @@ def build_dflash(request, writer, draft: Path) -> None: "Native paired execution is unavailable; refusing base-only fallback.", error, log_path) raise NotImplementedError("Native Nemotron-H DFlash fallback is unavailable") from error edge_llm.publish(request, writer, files, marker) + log_path.unlink() diff --git a/families/nemotron_h/runtime/edge_llm/adapter.cpp b/families/nemotron_h/runtime/edge_llm/adapter.cpp index 3c439ae00d..1d4a45604f 100644 --- a/families/nemotron_h/runtime/edge_llm/adapter.cpp +++ b/families/nemotron_h/runtime/edge_llm/adapter.cpp @@ -219,7 +219,10 @@ std::string native_chat_format(const BundleReader& bundle) { const auto text = metadata.at("chat_template").get(); if (text.empty()) throw std::runtime_error("Nemotron-H Edge requires resolved source chat_template"); - return nemotron_h_detect_chat_template_format(text); + const auto format = nemotron_h_detect_chat_template_format(text); + if (format.empty()) + throw std::runtime_error("Nemotron-H Edge does not recognize source chat_template"); + return format; } /// Reuse the original family BPE implementation, including its special postprocessor. From 1beaff46d2f36dc17ef2d7c330432f3169517f32 Mon Sep 17 00:00:00 2001 From: Joshua Calafato Date: Wed, 23 Sep 2026 04:43:47 +0000 Subject: [PATCH 3/9] refactor(nemotron_h): own Edge build selection Keep optional Edge dispatch, request options, adapters, and CMake wiring within family-owned edge_llm folders. Route explicit companions through the generic lazy CLI hook and the ordinary family build entrypoint; preserve native fallback semantics and quality gates. Signed-off-by: Joshua Calafato --- .../{EDGE_LLM.md => edge_llm/README.md} | 33 ++++- families/nemotron_h/edge_llm/__init__.py | 4 + .../{edge_llm.py => edge_llm/builder.py} | 0 families/nemotron_h/edge_llm/cli.py | 42 ++++++ families/nemotron_h/edge_llm/config.py | 88 ++++++++++++ .../nemotron_h/{ => edge_llm}/dispatch.py | 11 +- .../nemotron_h/{ => edge_llm}/edge_config.py | 0 .../{ => edge_llm}/edge_quantization.py | 0 .../{ => edge_llm}/edge_tokenizer.py | 0 families/nemotron_h/model.py | 18 ++- families/nemotron_h/runtime/CMakeLists.txt | 13 +- .../nemotron_h/runtime/edge_llm/Adapter.cmake | 18 +++ families/nemotron_h/support.py | 1 + .../nemotron_h/tests/test_runtime_contract.py | 135 +++++++++++++++++- website/docs/features/model-families.md | 2 +- 15 files changed, 339 insertions(+), 26 deletions(-) rename families/nemotron_h/{EDGE_LLM.md => edge_llm/README.md} (84%) create mode 100644 families/nemotron_h/edge_llm/__init__.py rename families/nemotron_h/{edge_llm.py => edge_llm/builder.py} (100%) create mode 100644 families/nemotron_h/edge_llm/cli.py create mode 100644 families/nemotron_h/edge_llm/config.py rename families/nemotron_h/{ => edge_llm}/dispatch.py (93%) rename families/nemotron_h/{ => edge_llm}/edge_config.py (100%) rename families/nemotron_h/{ => edge_llm}/edge_quantization.py (100%) rename families/nemotron_h/{ => edge_llm}/edge_tokenizer.py (100%) create mode 100644 families/nemotron_h/runtime/edge_llm/Adapter.cmake diff --git a/families/nemotron_h/EDGE_LLM.md b/families/nemotron_h/edge_llm/README.md similarity index 84% rename from families/nemotron_h/EDGE_LLM.md rename to families/nemotron_h/edge_llm/README.md index e16a4c1595..fc4265a55d 100644 --- a/families/nemotron_h/EDGE_LLM.md +++ b/families/nemotron_h/edge_llm/README.md @@ -2,7 +2,7 @@ The family owns source-policy admission, command mapping, tokenizer/EOS semantics, engine composition and runtime orchestration. TensorRT Edge-LLM owns model graphs, -lowering and execution. Provision the [optional native SDK](../../cmake/edgellm/README.md) +lowering and execution. Provision the [optional native SDK](../../../cmake/edge_llm/README.md) from official GitHub Edge-LLM 0.10.1, revision `e8b29522938901f6df19ebeedd4b69bc8edbcd97`. No internal source changes or cross-compilation are used. @@ -111,3 +111,34 @@ The earlier 4B-BF16 capacity4096 compiler failure, long-context/context-reuse, TP4, other models/platforms and stochastic equivalence remain outside this proof. Nano30B, Super120B and Omni are not qualified by the table above. No quality gate, failed result or hardware limitation is hidden by these passes. + +## Family-owned build options + +The generic CLI loads this family's `edge_llm.cli` hook only after resolving +the model. The shared build API has no execution-mode or companion arguments. +All variant validation and builder selection remain in this family. + +```sh +trtmc build /path/to/target --family nemotron_h --precision fp16 \ + --execution-variant dflash --companion draft=/path/to/draft \ + -o model.bundle +``` + +Put MODEL immediately after `build`. `trtmc build /path/to/target --help` +shows these family options using local metadata; remote-ID help does not download +a checkpoint. For Python callers, use this family's request extension: + +```python +from tensorrt_model_connect import build +from families.nemotron_h.edge_llm.config import ( + BuildExecutionInputs, NamedCheckpoint, with_execution, +) + +# request is an ordinary BuildRequest owned by this family; draft_path is a Path. +build(with_execution(request, BuildExecutionInputs( + "dflash", (NamedCheckpoint("draft", draft_path),), +))) +``` + +A failed explicit pair is never replaced by a base-only bundle. Previously +recorded full-model results above are historical, not fresh refactor-head E2Es. diff --git a/families/nemotron_h/edge_llm/__init__.py b/families/nemotron_h/edge_llm/__init__.py new file mode 100644 index 0000000000..df93d8cabe --- /dev/null +++ b/families/nemotron_h/edge_llm/__init__.py @@ -0,0 +1,4 @@ +# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-License-Identifier: Apache-2.0 + +"""Family-owned optional complete-network Edge offload.""" diff --git a/families/nemotron_h/edge_llm.py b/families/nemotron_h/edge_llm/builder.py similarity index 100% rename from families/nemotron_h/edge_llm.py rename to families/nemotron_h/edge_llm/builder.py diff --git a/families/nemotron_h/edge_llm/cli.py b/families/nemotron_h/edge_llm/cli.py new file mode 100644 index 0000000000..b397c20273 --- /dev/null +++ b/families/nemotron_h/edge_llm/cli.py @@ -0,0 +1,42 @@ +# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-License-Identifier: Apache-2.0 + +"""NemotronH-owned CLI options for explicit paired Edge execution.""" + +import argparse +from pathlib import Path + +from .config import BuildExecutionInputs, NamedCheckpoint, with_execution + + +def add_build_arguments(parser: argparse.ArgumentParser) -> None: + """Register options only when the NemotronH family has been resolved.""" + parser.add_argument("--execution-variant", choices=("dflash",), + help="Explicit NemotronH paired execution mode") + parser.add_argument("--companion", action="append", default=[], metavar="ROLE=LOCAL_DIR", + help="Explicit local draft checkpoint; exactly one draft role is required") + + +def _execution_inputs(args: argparse.Namespace) -> BuildExecutionInputs | None: + """Parse only explicit local inputs; no variant list or model acquisition.""" + if args.command != "build": + return None + if args.execution_variant is None: + if args.companion: + raise ValueError("--companion requires --execution-variant") + return None + checkpoints = [] + for value in args.companion: + role, separator, directory = value.partition("=") + if not separator or not role or not directory: + raise ValueError("--companion must be ROLE=LOCAL_DIR") + if "://" in directory: + raise ValueError("--companion requires a local directory, not a URI") + checkpoints.append(NamedCheckpoint(role, Path(directory))) + return BuildExecutionInputs(args.execution_variant, tuple(checkpoints)) + + +def prepare_build_request(request, args): + """Attach a validated family-owned recipe before importing the GPU builder.""" + execution = _execution_inputs(args) + return request if execution is None else with_execution(request, execution) diff --git a/families/nemotron_h/edge_llm/config.py b/families/nemotron_h/edge_llm/config.py new file mode 100644 index 0000000000..27ae74f9c9 --- /dev/null +++ b/families/nemotron_h/edge_llm/config.py @@ -0,0 +1,88 @@ +# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-License-Identifier: Apache-2.0 + +"""Explicit NemotronH paired-build inputs; no shared execution-mode contract.""" + +from dataclasses import dataclass, fields +from pathlib import Path +import re + +from tensorrt_model_connect.build import BuildRequest + + +_ID = re.compile(r"[a-z][a-z0-9_]*\Z") + + +def _validate_id(field: str, value: object) -> str: + if not isinstance(value, str) or _ID.fullmatch(value) is None: + raise ValueError(f"{field} must be a lowercase identifier containing only letters, digits, and underscores") + return value + + +@dataclass(frozen=True) +class NamedCheckpoint: + """One explicitly named local checkpoint; the family owns role semantics.""" + + role: str + model_dir: Path + + def __post_init__(self) -> None: + _validate_id("checkpoint role", self.role) + if not isinstance(self.model_dir, Path): + raise TypeError("checkpoint model_dir must be a Path") + if not self.model_dir.is_dir(): + raise ValueError(f"checkpoint must be an existing local directory: {self.model_dir}") + + +@dataclass(frozen=True) +class BuildExecutionInputs: + """Optional family-owned execution variant and immutable local companions. + + NemotronH validates these inputs without fetching or inferring companions. + """ + + variant: str + checkpoints: tuple[NamedCheckpoint, ...] = () + + def __post_init__(self) -> None: + _validate_id("execution variant", self.variant) + if not isinstance(self.checkpoints, tuple) or any( + not isinstance(checkpoint, NamedCheckpoint) for checkpoint in self.checkpoints + ): + raise TypeError("checkpoints must be a tuple of NamedCheckpoint values") + roles = [checkpoint.role for checkpoint in self.checkpoints] + if len(roles) != len(set(roles)): + raise ValueError("checkpoint roles must be unique") + self.validate_local() + + def validate_local(self) -> None: + """Recheck local availability before dispatch, without acquiring inputs.""" + for checkpoint in self.checkpoints: + if not checkpoint.model_dir.is_dir(): + raise ValueError( + f"checkpoint must be an existing local directory: {checkpoint.model_dir}" + ) + + +@dataclass(frozen=True) +class NemotronHBuildRequest(BuildRequest): + """Ordinary build inputs plus an explicitly requested NemotronH execution recipe.""" + + execution: BuildExecutionInputs | None = None + + def __post_init__(self) -> None: + super().__post_init__() + if self.family != "nemotron_h": + raise ValueError("NemotronHBuildRequest requires the nemotron_h family") + if self.execution is not None: + if not isinstance(self.execution, BuildExecutionInputs): + raise TypeError("execution must be BuildExecutionInputs") + self.execution.validate_local() + + +def with_execution(request: BuildRequest, execution: BuildExecutionInputs) -> NemotronHBuildRequest: + """Preserve ordinary request fields and callback identity.""" + return NemotronHBuildRequest( + **{field.name: getattr(request, field.name) for field in fields(BuildRequest)}, + execution=execution, + ) diff --git a/families/nemotron_h/dispatch.py b/families/nemotron_h/edge_llm/dispatch.py similarity index 93% rename from families/nemotron_h/dispatch.py rename to families/nemotron_h/edge_llm/dispatch.py index 99a8363e7d..280f0d294a 100644 --- a/families/nemotron_h/dispatch.py +++ b/families/nemotron_h/edge_llm/dispatch.py @@ -12,7 +12,7 @@ import tempfile import traceback -from . import edge_llm +from . import builder as edge_llm from .edge_quantization import source_quantization from .edge_config import topology_matches @@ -136,3 +136,12 @@ def build_dflash(request, writer, draft: Path) -> None: raise NotImplementedError("Native Nemotron-H DFlash fallback is unavailable") from error edge_llm.publish(request, writer, files, marker) log_path.unlink() + + +def build_paired(request, writer, execution) -> None: + """Keep the complete DFlash pair owned by the family ONNX adapter.""" + execution.validate_local() + if execution.variant != "dflash" or tuple(x.role for x in execution.checkpoints) != ("draft",): + raise ValueError("Nemotron-H paired execution requires variant=dflash and one draft") + + build_dflash(request, writer, execution.checkpoints[0].model_dir) diff --git a/families/nemotron_h/edge_config.py b/families/nemotron_h/edge_llm/edge_config.py similarity index 100% rename from families/nemotron_h/edge_config.py rename to families/nemotron_h/edge_llm/edge_config.py diff --git a/families/nemotron_h/edge_quantization.py b/families/nemotron_h/edge_llm/edge_quantization.py similarity index 100% rename from families/nemotron_h/edge_quantization.py rename to families/nemotron_h/edge_llm/edge_quantization.py diff --git a/families/nemotron_h/edge_tokenizer.py b/families/nemotron_h/edge_llm/edge_tokenizer.py similarity index 100% rename from families/nemotron_h/edge_tokenizer.py rename to families/nemotron_h/edge_llm/edge_tokenizer.py diff --git a/families/nemotron_h/model.py b/families/nemotron_h/model.py index 06ce9545b6..89c3acdcb7 100644 --- a/families/nemotron_h/model.py +++ b/families/nemotron_h/model.py @@ -994,7 +994,14 @@ def _runtime_config( def build(request: "BuildRequest", writer: "BundleWriter") -> None: """Dispatch complete offload or execute the unchanged native implementation.""" - from .dispatch import build as dispatch_build + + from .edge_llm.config import NemotronHBuildRequest + from .edge_llm.dispatch import build_paired + + if isinstance(request, NemotronHBuildRequest) and request.execution is not None: + build_paired(request, writer, request.execution) + return + from .edge_llm.dispatch import build as dispatch_build def _build_native(request: "BuildRequest", writer: "BundleWriter") -> None: """Build one Nemotron-H bundle.""" @@ -1100,12 +1107,3 @@ def _build_native(request: "BuildRequest", writer: "BundleWriter") -> None: writer.add_bytes(filename, path.read_bytes()) dispatch_build(request, writer, _build_native) - - -def build_with_inputs(request, writer, execution) -> None: - """Keep the complete DFlash pair owned by the family ONNX adapter.""" - if execution.variant != "dflash" or tuple(x.role for x in execution.checkpoints) != ("draft",): - raise ValueError("Nemotron-H paired execution requires variant=dflash and one draft") - from .dispatch import build_dflash - - build_dflash(request, writer, execution.checkpoints[0].model_dir) diff --git a/families/nemotron_h/runtime/CMakeLists.txt b/families/nemotron_h/runtime/CMakeLists.txt index 640b4c03ba..e665f39099 100644 --- a/families/nemotron_h/runtime/CMakeLists.txt +++ b/families/nemotron_h/runtime/CMakeLists.txt @@ -58,15 +58,4 @@ if(TRTMC_BUILD_TESTS) add_test(NAME ${test_name} COMMAND ${test_name}) endif() -if(TARGET EdgeLLM::Core) - target_sources(trtmc_model_nemotron_h PRIVATE edge_llm/adapter.cpp edge_llm/device_link.cu) - target_compile_definitions(trtmc_model_nemotron_h PRIVATE TRTMC_HAS_EDGE_LLM=1) - target_link_libraries(trtmc_model_nemotron_h PRIVATE EdgeLLM::Core nlohmann_json::nlohmann_json) - set_target_properties(trtmc_model_nemotron_h PROPERTIES - CUDA_ARCHITECTURES "${EdgeLLM_CUDA_ARCHITECTURE}" - CUDA_SEPARABLE_COMPILATION ON CUDA_RESOLVE_DEVICE_SYMBOLS ON) - add_custom_command(TARGET trtmc_model_nemotron_h POST_BUILD - COMMAND ${CMAKE_COMMAND} -E copy_if_different - $ $ - VERBATIM) -endif() +include(edge_llm/Adapter.cmake) diff --git a/families/nemotron_h/runtime/edge_llm/Adapter.cmake b/families/nemotron_h/runtime/edge_llm/Adapter.cmake new file mode 100644 index 0000000000..fc5ab575c6 --- /dev/null +++ b/families/nemotron_h/runtime/edge_llm/Adapter.cmake @@ -0,0 +1,18 @@ +# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-License-Identifier: Apache-2.0 + +if(TARGET EdgeLLM::Core) + if(NOT TARGET EdgeLLM::Plugin) + message(FATAL_ERROR "NemotronH Edge adapter requires the complete EdgeLLM package (Core and Plugin)") + endif() + target_sources(trtmc_model_nemotron_h PRIVATE "${CMAKE_CURRENT_LIST_DIR}/adapter.cpp" "${CMAKE_CURRENT_LIST_DIR}/device_link.cu") + target_compile_definitions(trtmc_model_nemotron_h PRIVATE TRTMC_HAS_EDGE_LLM=1) + target_link_libraries(trtmc_model_nemotron_h PRIVATE EdgeLLM::Core nlohmann_json::nlohmann_json) + set_target_properties(trtmc_model_nemotron_h PROPERTIES + CUDA_ARCHITECTURES "${EdgeLLM_CUDA_ARCHITECTURE}" + CUDA_SEPARABLE_COMPILATION ON CUDA_RESOLVE_DEVICE_SYMBOLS ON) + add_custom_command(TARGET trtmc_model_nemotron_h POST_BUILD + COMMAND ${CMAKE_COMMAND} -E copy_if_different + $ $ + VERBATIM) +endif() diff --git a/families/nemotron_h/support.py b/families/nemotron_h/support.py index 0accd09428..11c331a813 100644 --- a/families/nemotron_h/support.py +++ b/families/nemotron_h/support.py @@ -10,4 +10,5 @@ model_types=("nemotron_h", "nemotron_hybrid", "nemotronh"), tasks=("text_generation",), default_task="text_generation", + build_cli_module="edge_llm.cli", ) diff --git a/families/nemotron_h/tests/test_runtime_contract.py b/families/nemotron_h/tests/test_runtime_contract.py index aa18538ad5..1cae6ae8e0 100644 --- a/families/nemotron_h/tests/test_runtime_contract.py +++ b/families/nemotron_h/tests/test_runtime_contract.py @@ -65,9 +65,142 @@ def test_runtime_keeps_checkpoint_prompt_and_builder_policies() -> None: assert "builder_optimization_level = 1" not in source # Edge has native scalar prefill even when the optimized head80 SSD is absent. - from families.nemotron_h.dispatch import EDGE_DISPATCH, platform_matches + from families.nemotron_h.edge_llm.dispatch import EDGE_DISPATCH, platform_matches for sm in (80, 120): assert platform_matches({"mamba_head_dim": 80}, {"sm": sm}) assert ("linux", "x86_64", sm, "fp16") in EDGE_DISPATCH assert ("linux", "x86_64", 80, "nvfp4") not in EDGE_DISPATCH + + +def _edge_cli_source(tmp_path): + import json + + source = tmp_path / "target" + source.mkdir() + (source / "config.json").write_text(json.dumps({"model_type": "nemotron_h"})) + draft = tmp_path / "draft" + draft.mkdir() + return source, draft + + +def test_edge_cli_uses_ordinary_family_build(tmp_path, monkeypatch): + from tensorrt_model_connect import build_cli + from families.nemotron_h.edge_llm import dispatch + from families.nemotron_h.edge_llm.config import NemotronHBuildRequest + + source, draft = _edge_cli_source(tmp_path) + output = tmp_path / "pair.bundle" + seen = [] + + def paired(request, writer, execution): + assert isinstance(request, NemotronHBuildRequest) + assert request.execution is execution + assert execution.variant == "dflash" + assert [(item.role, item.model_dir) for item in execution.checkpoints] == [("draft", draft)] + seen.append(request) + writer.set_header(family="nemotron_h", task=request.task, backend=request.backend) + writer.add_json("edge-test.json", {"variant": execution.variant}) + + monkeypatch.setattr(dispatch, "build_paired", paired) + assert build_cli.main([ + "build", str(source), "--precision", "fp16", "-o", str(output), + "--execution-variant", "dflash", "--companion", f"draft={draft}", + ]) == 0 + assert len(seen) == 1 + assert output.is_file() + + +@pytest.mark.parametrize("options", [ + ["--companion", "draft=/missing"], + ["--execution-variant", "dflash", "--companion", "missing_separator"], + ["--execution-variant", "dflash", "--companion", "=path"], + ["--execution-variant", "dflash", "--companion", "draft="], + ["--execution-variant", "dflash", "--companion", "draft=https://example.com/model"], +]) +def test_bad_edge_cli_inputs_fail_before_backend(tmp_path, monkeypatch, options): + import importlib + from tensorrt_model_connect import build_cli + + core = importlib.import_module("tensorrt_model_connect.build") + source, _ = _edge_cli_source(tmp_path) + monkeypatch.setattr(core, "_select_backend", lambda *_: pytest.fail("backend touched")) + monkeypatch.setattr(core, "BundleWriter", lambda *_: pytest.fail("writer created")) + with pytest.raises(ValueError): + build_cli.main(["build", str(source), "-o", str(tmp_path / "out"), *options]) + + +def test_edge_cli_help_is_family_owned(tmp_path, capsys): + from tensorrt_model_connect import build_cli + + source, _ = _edge_cli_source(tmp_path) + with pytest.raises(SystemExit) as caught: + build_cli.main(["build", str(source), "--help"]) + assert caught.value.code == 0 + help_text = capsys.readouterr().out + assert "--execution-variant {dflash}" in help_text + assert "--companion" in help_text + + +def test_edge_request_preserves_fields_and_family_owner(tmp_path): + import argparse + from families.nemotron_h.edge_llm import cli + from dataclasses import fields, replace + from tensorrt_model_connect.build import BuildRequest + from families.nemotron_h.edge_llm.config import ( + BuildExecutionInputs, NamedCheckpoint, with_execution, + ) + + source, draft = _edge_cli_source(tmp_path) + request = BuildRequest(source, tmp_path / "out", "nemotron_h", "text_generation", "fp16", + graph_transform=lambda layer: layer) + execution = BuildExecutionInputs("dflash", (NamedCheckpoint("draft", draft),)) + args = argparse.Namespace(command="build", execution_variant=None, companion=[]) + assert cli.prepare_build_request(request, args) is request + extended = with_execution(request, execution) + for field in fields(BuildRequest): + assert getattr(extended, field.name) is getattr(request, field.name) + with pytest.raises(ValueError, match="requires the nemotron_h family"): + with_execution(replace(request, family="other"), execution) + with pytest.raises(ValueError, match="unique"): + BuildExecutionInputs("dflash", execution.checkpoints * 2) + + +@pytest.mark.parametrize("failure", [RuntimeError("paired build failed"), KeyboardInterrupt()]) +def test_edge_cli_failure_preserves_existing_bundle(tmp_path, monkeypatch, failure): + from tensorrt_model_connect import build_cli + from families.nemotron_h.edge_llm import dispatch + + source, draft = _edge_cli_source(tmp_path) + output = tmp_path / "pair.bundle" + output.write_bytes(b"previous publication") + + def fail(request, writer, execution): + writer.set_header(family="nemotron_h", task=request.task, backend=request.backend) + writer.add_json("edge-test.json", {"variant": execution.variant}) + raise failure + + monkeypatch.setattr(dispatch, "build_paired", fail) + with pytest.raises(type(failure)) as caught: + build_cli.main([ + "build", str(source), "-o", str(output), "--execution-variant", "dflash", + "--companion", f"draft={draft}", + ]) + assert caught.value is failure + assert output.read_bytes() == b"previous publication" + assert sorted(item.name for item in tmp_path.iterdir()) == ["draft", "pair.bundle", "target"] + + +def test_edge_pair_requires_draft_and_rechecks_local_inputs(tmp_path): + from tensorrt_model_connect.build import BuildRequest + from families.nemotron_h.edge_llm.config import BuildExecutionInputs, NamedCheckpoint + from families.nemotron_h.edge_llm.dispatch import build_paired + + source, draft = _edge_cli_source(tmp_path) + request = BuildRequest(source, tmp_path / "out", "nemotron_h", "text_generation", "fp16") + with pytest.raises(ValueError, match="paired execution requires"): + build_paired(request, None, BuildExecutionInputs("dflash")) + execution = BuildExecutionInputs("dflash", (NamedCheckpoint("draft", draft),)) + draft.rmdir() + with pytest.raises(ValueError, match="existing local directory"): + build_paired(request, None, execution) diff --git a/website/docs/features/model-families.md b/website/docs/features/model-families.md index bd2ca81ddf..3a60097c8a 100644 --- a/website/docs/features/model-families.md +++ b/website/docs/features/model-families.md @@ -118,7 +118,7 @@ Lightning NVFP4/DFlash pair uses ONNX with Native fallback is attempted only during ordinary preparation; it cannot interpret unsupported packed checkpoints or replace a requested pair. -See the [owning Nemotron-H recipe](https://github.com/NVIDIA/TensorRT-Model-Connect/blob/main/families/nemotron_h/EDGE_LLM.md) +See the [owning Nemotron-H recipe](https://github.com/NVIDIA/TensorRT-Model-Connect/blob/main/families/nemotron_h/edge_llm/README.md) for the six recorded ordinary profiles, the qualified greedy DFlash pair, immutable revisions and unchanged quality gates. These are bounded historical local qualifications, not catalog-wide or current-head CI proof. The separate From 60e030dcb17b19728f59f8f76a34486156a54fe4 Mon Sep 17 00:00:00 2001 From: Joshua Calafato Date: Wed, 23 Sep 2026 05:01:27 +0000 Subject: [PATCH 4/9] docs(nemotron-h): clarify paired build inputs Signed-off-by: Joshua Calafato --- families/nemotron_h/edge_llm/README.md | 28 +++++++++---------- .../nemotron_h/edge_llm/edge_quantization.py | 2 +- website/docs/features/model-families.md | 4 +-- 3 files changed, 17 insertions(+), 17 deletions(-) diff --git a/families/nemotron_h/edge_llm/README.md b/families/nemotron_h/edge_llm/README.md index fc4265a55d..3759759092 100644 --- a/families/nemotron_h/edge_llm/README.md +++ b/families/nemotron_h/edge_llm/README.md @@ -21,8 +21,8 @@ native ONNX builder. Enable `TRTMC_EDGELLM_ALL_KERNELS=ON` and `TRTMC_EDGELLM_ONNX=ON`, set `CMAKE_PREFIX_PATH` to the SDK installation, and configure the runtime with `TRTMC_ENABLE_EDGELLM=ON`. Add `--execution-variant dflash --companion draft=/path/to/draft` to the build CLI, -or pass `BuildExecutionInputs("dflash", (NamedCheckpoint("draft", draft_path),))` -to the Python build API. +or wrap the request with this family's `with_execution(request, inputs)` +before calling `build`, as the Python example below shows. Both paired engines are mandatory. The ONNX graph supplies the intermediate Mamba/replay states required to commit accepted draft tokens; the experimental @@ -38,7 +38,7 @@ Cancellation, publication and runtime errors propagate without retry. ## Request and runtime contracts Admission requires compatible Nemotron-H text topology and source quantization, -FP16 compute, TP1/batch1/context-parallel1, and no unmapped build controls. +FP16 compute, TP 1/batch 1/context-parallel 1, and no unmapped build controls. The published platform routes are native x86_64 SM80 for plain sources and SM120 for plain/FP8/NVFP4 sources. These routes are admission rules, not proof for every compatible checkpoint or capacity. @@ -64,20 +64,20 @@ inference. These are **historical local Model Connect build/inference qualifications**, not assertions that every publication head or CI executes these profiles. -CUDA 13.3 and TensorRT 11.1.0.106 were used. Ordinary MC profiles use capacity256; -the DFlash profile uses input/KV1024. All are text-only TP1/batch1. -Independent NED must be at most0.15; the original Edge128-token fixture gates +CUDA 13.3 and TensorRT 11.1.0.106 were used. Ordinary MC profiles use capacity 256; +the DFlash profile uses input/KV1024. All are text-only TP 1/batch 1. +Independent NED must be at most 0.15; the original Edge 128-token fixture gates remain ROUGE-1 >=0.25 and ROUGE-L >=0.20. | Exact source model | GPU | Independent gate | MC ROUGE-1 / ROUGE-L | | --- | --- | --- | --- | -| NVIDIA-Nemotron-3-Nano-4B-BF16 | SM120 | NED0.0 | 0.5371 / 0.2514 | -| NVIDIA-Nemotron-3-Nano-4B-FP8 | SM120 | NED0.0 | 0.5402 / 0.2644 | +| NVIDIA-Nemotron-3-Nano-4B-BF16 | SM120 | NED 0.0 | 0.5371 / 0.2514 | +| NVIDIA-Nemotron-3-Nano-4B-FP8 | SM120 | NED 0.0 | 0.5402 / 0.2644 | | NVIDIA-Nemotron-Nano-9B-v2 | SM80 | Existing HF pytest gate passed; numeric NED not serialized | 0.4318 / 0.2727 | -| NVIDIA-Nemotron-Nano-9B-v2-FP8 | SM120 | NED0.0 | 0.3750 / 0.2614 | -| NVIDIA-Nemotron-Nano-9B-v2-NVFP4 | SM120 | NED0.0 | 0.4130 / 0.2391 | -| NVIDIA-Nemotron-3.5-Lightning-30B-A3B-NVFP4 | SM120 | NED0.0 | 0.4643 / 0.2500 | -| Same Lightning target + DFlash companion | SM120 | NED0.0 | 0.4848 / 0.2545 | +| NVIDIA-Nemotron-Nano-9B-v2-FP8 | SM120 | NED 0.0 | 0.3750 / 0.2614 | +| NVIDIA-Nemotron-Nano-9B-v2-NVFP4 | SM120 | NED 0.0 | 0.4130 / 0.2391 | +| NVIDIA-Nemotron-3.5-Lightning-30B-A3B-NVFP4 | SM120 | NED 0.0 | 0.4643 / 0.2500 | +| Same Lightning target + DFlash companion | SM120 | NED 0.0 | 0.4848 / 0.2545 | All models are from the public `nvidia` namespace. Immutable revisions, in table order, are: @@ -100,12 +100,12 @@ sampling parity. Passing payloads were retired with approval; compact results and provenance were retained. Replaying all seven model checks requires rebuilding those payloads. Fresh publication compilation/unit/source checks must be reported separately. -Only the existing plain9B case is a registered owning E2E among these profiles; +Only the existing plain 9B case is a registered owning E2E among these profiles; the other exact models/pair used local recipes with existing family helpers. ## Still outside these qualifications -The separate direct-Edge9B-NVFP4 capacity1024 run failed ROUGE-L0.1957 against0.20 +The separate direct-Edge 9B-NVFP4 capacity 1024 run failed ROUGE-L 0.1957 against 0.20 on a different SM120 GPU. The MC256 pass does not resolve that failure. The earlier 4B-BF16 capacity4096 compiler failure, long-context/context-reuse, TP4, other models/platforms and stochastic equivalence remain outside this proof. diff --git a/families/nemotron_h/edge_llm/edge_quantization.py b/families/nemotron_h/edge_llm/edge_quantization.py index d3ebf5fcbb..dde984468e 100644 --- a/families/nemotron_h/edge_llm/edge_quantization.py +++ b/families/nemotron_h/edge_llm/edge_quantization.py @@ -55,7 +55,7 @@ def descriptor(value: dict) -> tuple | None: return None group = q.get("group_size", 1 if precision == "fp8" else 16) # FP8 projections use scalar scaling, not grouped packing. ModelOpt emits - # either1 (parser default) or16 (export metadata); neither changes FP8 layout. + # either 1 (parser default) or 16 (export metadata); neither changes FP8 layout. if type(group) is not int or group not in ({1, 16} if precision == "fp8" else {16}): return None if any(key in q for key in ("config_groups", "quantized_layers", "kv_cache_scheme")): diff --git a/website/docs/features/model-families.md b/website/docs/features/model-families.md index 3a60097c8a..213a3bde0c 100644 --- a/website/docs/features/model-families.md +++ b/website/docs/features/model-families.md @@ -122,8 +122,8 @@ See the [owning Nemotron-H recipe](https://github.com/NVIDIA/TensorRT-Model-Conn for the six recorded ordinary profiles, the qualified greedy DFlash pair, immutable revisions and unchanged quality gates. These are bounded historical local qualifications, not catalog-wide or current-head CI proof. The separate -direct-Edge9B-NVFP4 quality failure and longer-context failures remain open. -The original plain9B case is registered in the owning E2E inventory; the other +direct-Edge 9B-NVFP4 quality failure and longer-context failures remain open. +The original plain 9B case is registered in the owning E2E inventory; the other exact profiles are not implied to run in CI. ## Runtime and validation From 73c7bc6bfcc65021ba1b79323db7471c05d4a7f5 Mon Sep 17 00:00:00 2001 From: Joshua Calafato Date: Wed, 23 Sep 2026 15:53:53 +0000 Subject: [PATCH 5/9] fix(nemotron_h): preserve optional SDK fallback Keep Edge routing family-owned, avoid failure diagnostics for an absent optional SDK, and retain errors for broken installations. Use output-local staging and existing regression coverage. Signed-off-by: Joshua Calafato --- families/nemotron_h/edge_llm/README.md | 7 +- families/nemotron_h/edge_llm/builder.py | 9 +++ families/nemotron_h/edge_llm/dispatch.py | 7 +- .../nemotron_h/tests/test_runtime_contract.py | 76 +++++++++++++++++++ 4 files changed, 97 insertions(+), 2 deletions(-) diff --git a/families/nemotron_h/edge_llm/README.md b/families/nemotron_h/edge_llm/README.md index 3759759092..d68fc90cf0 100644 --- a/families/nemotron_h/edge_llm/README.md +++ b/families/nemotron_h/edge_llm/README.md @@ -124,7 +124,7 @@ trtmc build /path/to/target --family nemotron_h --precision fp16 \ -o model.bundle ``` -Put MODEL immediately after `build`. `trtmc build /path/to/target --help` +Put MODEL before family-specific options; known core options may precede MODEL. `trtmc build /path/to/target --help` shows these family options using local metadata; remote-ID help does not download a checkpoint. For Python callers, use this family's request extension: @@ -142,3 +142,8 @@ build(with_execution(request, BuildExecutionInputs( A failed explicit pair is never replaced by a base-only bundle. Previously recorded full-model results above are historical, not fresh refactor-head E2Es. + +Ordinary builds without an installed optional Edge SDK select native without a +warning. Malformed or incomplete installed packages still retain diagnostics and +warn before native fallback. Temporary checkpoint/engine staging uses the bundle +output directory filesystem (choose a scratch-backed output), not system /tmp. diff --git a/families/nemotron_h/edge_llm/builder.py b/families/nemotron_h/edge_llm/builder.py index d4e1f5215b..5bad995487 100644 --- a/families/nemotron_h/edge_llm/builder.py +++ b/families/nemotron_h/edge_llm/builder.py @@ -23,6 +23,15 @@ def local_target() -> dict: return detect_local_platform() +def package_present() -> bool: + """Absence of the optional SDK is a non-match, not a failed Edge build.""" + for prefix in cmake_prefixes(): + manifest = prefix / "share/trtmc/edge-llm.json" + if manifest.exists() or manifest.is_symlink(): + return True + return False + + def installed_package(target: dict) -> dict: """Resolve CMake installation via standard prefixes; never install anything. diff --git a/families/nemotron_h/edge_llm/dispatch.py b/families/nemotron_h/edge_llm/dispatch.py index 280f0d294a..e9eaa1df57 100644 --- a/families/nemotron_h/edge_llm/dispatch.py +++ b/families/nemotron_h/edge_llm/dispatch.py @@ -76,6 +76,9 @@ def build(request, writer, native) -> None: if not candidate(request, raw): native(request, writer) return + if not edge_llm.package_present(): + native(request, writer) + return if edge_llm.sequence_length(request, raw) > raw["max_position_embeddings"]: raise ValueError("Nemotron-H max_sequence_length exceeds checkpoint context capacity") failure = None @@ -83,7 +86,9 @@ def build(request, writer, native) -> None: dir=request.output_path.parent) os.close(descriptor) log_path = Path(name) - with tempfile.TemporaryDirectory(prefix="trtmc-nemotron-h-edge-") as directory: + with tempfile.TemporaryDirectory( + prefix=f".{request.output_path.name}.edge-", dir=request.output_path.parent + ) as directory: try: target = edge_llm.local_target() key = (target["os"], target["arch"], target["sm"], source_quantization(request, raw)) diff --git a/families/nemotron_h/tests/test_runtime_contract.py b/families/nemotron_h/tests/test_runtime_contract.py index 1cae6ae8e0..5776c8fec1 100644 --- a/families/nemotron_h/tests/test_runtime_contract.py +++ b/families/nemotron_h/tests/test_runtime_contract.py @@ -204,3 +204,79 @@ def test_edge_pair_requires_draft_and_rechecks_local_inputs(tmp_path): draft.rmdir() with pytest.raises(ValueError, match="existing local directory"): build_paired(request, None, execution) + + +@pytest.mark.parametrize("mode", ["absent", "success", "corrupt", "failure", "cancel", "device_failure"]) +def test_edge_optional_package_and_output_local_staging(tmp_path, monkeypatch, caplog, mode): + import json + from tensorrt_model_connect.build import BuildRequest + from families.nemotron_h.edge_llm import builder, dispatch + + source = tmp_path / "target" + source.mkdir() + (source / "config.json").write_text(json.dumps({ + "max_position_embeddings": 4096, "hidden_size": 896, + })) + prefix = tmp_path / "package" + manifest = prefix / "share/trtmc/edge-llm.json" + if mode != "absent": + manifest.parent.mkdir(parents=True) + manifest.write_text("{" if mode == "corrupt" else "{}") + monkeypatch.setattr(builder, "cmake_prefixes", lambda: [prefix]) + monkeypatch.setattr(dispatch, "candidate", lambda *_: True) + for name in ("mapped_request", "request_matches", "platform_matches"): + if hasattr(dispatch, name): + monkeypatch.setattr(dispatch, name, lambda *_: True) + if hasattr(dispatch, "source_quantization"): + monkeypatch.setattr(dispatch, "source_quantization", lambda *_: "fp16") + if hasattr(builder, "request_weight_format"): + monkeypatch.setattr(builder, "request_weight_format", lambda *_: "fp16") + request = BuildRequest(source, tmp_path / "out", "nemotron_h", "text_generation", "fp16") + writer = object() + target = {"os": "linux", "arch": "x86_64", "sm": 80} + target_calls = [] + + def local_target(): + target_calls.append(True) + if mode == "device_failure": + raise RuntimeError("CUDA discovery failed") + return target + + monkeypatch.setattr(builder, "local_target", local_target) + stages, native_calls, publications = [], [], [] + + def prepare(original, raw, platform, staging, log): + assert original is request and platform is target + assert staging.parent == request.output_path.parent + assert staging.name.startswith(f".{request.output_path.name}.edge-") + stages.append(staging) + (staging / "large-checkpoint").write_bytes(b"fixture") + if mode == "corrupt": + builder.installed_package(target) + if mode == "failure": + raise FileNotFoundError("installed SDK artifact missing") + if mode == "cancel": + raise KeyboardInterrupt() + return {}, {} + + monkeypatch.setattr(dispatch, "EDGE_DISPATCH", {("linux", "x86_64", 80, "fp16"): prepare}) + monkeypatch.setattr(builder, "publish", lambda *args: publications.append(args)) + def native(*args): + native_calls.append(args) + if mode == "cancel": + with pytest.raises(KeyboardInterrupt): + dispatch.build(request, writer, native) + else: + dispatch.build(request, writer, native) + assert all(not path.exists() for path in stages) + assert native_calls == ([(request, writer)] if mode in { + "absent", "corrupt", "failure", "device_failure", + } else []) + assert len(publications) == (1 if mode == "success" else 0) + assert len(target_calls) == (0 if mode == "absent" else 1) + logs = list(tmp_path.glob(".out.edge-*.log")) + if mode in {"corrupt", "failure", "device_failure"}: + assert len(logs) == 1 and "Traceback" in logs[0].read_text() + assert "Retrying native once" in caplog.text + elif mode != "cancel": + assert not logs and "Edge build failed" not in caplog.text From 9986bdb6d25eac6325214da1aa20d6c4d7734d30 Mon Sep 17 00:00:00 2001 From: Joshua Calafato Date: Wed, 23 Sep 2026 19:46:59 +0000 Subject: [PATCH 6/9] refactor(nemotron_h): use declared family CLI Reuse the existing family CLI protocol instead of extending the shared parser. Own the command description, request contract and build lifecycle; keep Edge companion semantics inside this family. Preserve legacy callers through strict conversion and keep numerical acceptance gates unchanged. Signed-off-by: Joshua Calafato --- families/nemotron_h/build_request.py | 96 ++++++++++++++ families/nemotron_h/cli.json | 121 ++++++++++++++++++ families/nemotron_h/cli.py | 57 +++++++++ families/nemotron_h/edge_llm/README.md | 15 ++- families/nemotron_h/edge_llm/cli.py | 33 ++--- families/nemotron_h/edge_llm/config.py | 5 +- families/nemotron_h/model.py | 5 +- families/nemotron_h/support.py | 1 - .../nemotron_h/tests/test_runtime_contract.py | 78 +++++++++-- website/docs/features/model-families.md | 5 + 10 files changed, 374 insertions(+), 42 deletions(-) create mode 100644 families/nemotron_h/build_request.py create mode 100644 families/nemotron_h/cli.json create mode 100644 families/nemotron_h/cli.py diff --git a/families/nemotron_h/build_request.py b/families/nemotron_h/build_request.py new file mode 100644 index 0000000000..0bf48a4567 --- /dev/null +++ b/families/nemotron_h/build_request.py @@ -0,0 +1,96 @@ +# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-License-Identifier: Apache-2.0 + +"""nemotron_h build inputs and strict compatibility for existing Python callers.""" + +from __future__ import annotations + +from dataclasses import dataclass, fields +from pathlib import Path +import re +from typing import ClassVar + +from tensorrt_model_connect.graph_transform import GraphTransform + + +def _validate_id(field: str, value: str) -> None: + if not isinstance(value, str) or re.fullmatch(r"[a-z][a-z0-9_]*", value) is None: + raise ValueError(f"{field} must be a lowercase identifier") + + +@dataclass(frozen=True) +class BuildRequest: + """nemotron_h-owned inputs; unsupported legacy controls are read-only defaults.""" + + model_dir: Path + output_path: Path + family: str + task: str + precision: str + backend: str = "trt" + max_sequence_length: int | None = None + image_height: ClassVar[int | None] = None + image_width: ClassVar[int | None] = None + video_num_frames: ClassVar[int | None] = None + max_batch_size: ClassVar[int] = 1 + tensor_parallel_size: int = 1 + context_parallel_size: ClassVar[int] = 1 + quantization: ClassVar[str | None] = None + fp32_layers: ClassVar[tuple[int, ...]] = () + dynamic_kv_cache: ClassVar[bool] = False + verbose: bool = False + graph_transform: GraphTransform | None = None + + def __post_init__(self) -> None: + if not self.precision: + raise ValueError("precision must be non-empty") + _validate_id("family", self.family) + _validate_id("task", self.task) + if self.backend not in {"trt", "trt_rtx"}: + raise ValueError("backend must be 'trt' or 'trt_rtx'") + if self.max_sequence_length is not None and self.max_sequence_length < 1: + raise ValueError("max_sequence_length must be positive") + for field in ("image_height", "image_width", "video_num_frames"): + value = getattr(self, field) + if value is not None and value < 1: + raise ValueError(f"{field} must be positive") + if self.max_batch_size < 1: + raise ValueError("max_batch_size must be positive") + if self.tensor_parallel_size < 1: + raise ValueError("tensor_parallel_size must be positive") + if self.context_parallel_size < 1: + raise ValueError("context_parallel_size must be positive") + if self.quantization is not None and not self.quantization: + raise ValueError("quantization must be non-empty when provided") + if any(layer < 0 for layer in self.fp32_layers): + raise ValueError("fp32_layers must contain non-negative indices") + if not isinstance(self.dynamic_kv_cache, bool): + raise ValueError("dynamic_kv_cache must be a bool") + if self.graph_transform is not None and not callable(self.graph_transform): + raise ValueError("graph_transform must be callable when provided") + + +def coerce_request(request: object) -> BuildRequest: + """Reject unsupported/unknown legacy inputs before converting to owner fields.""" + if isinstance(request, BuildRequest): + return request + unsupported = { + "image_height": None, + "image_width": None, + "video_num_frames": None, + "max_batch_size": 1, + "context_parallel_size": 1, + "quantization": None, + "fp32_layers": (), + "dynamic_kv_cache": False, + } + for name, default in unsupported.items(): + value = getattr(request, name, default) + if name == "quantization" and value == "none": + continue + if value != default: + raise NotImplementedError(f"nemotron_h does not support {name}") + names = {field.name for field in fields(BuildRequest)} + if unknown := set(vars(request)) - names - set(unsupported): + raise ValueError(f"unknown nemotron_h build inputs: {sorted(unknown)}") + return BuildRequest(**{name: getattr(request, name) for name in names}) diff --git a/families/nemotron_h/cli.json b/families/nemotron_h/cli.json new file mode 100644 index 0000000000..9f4685f2e2 --- /dev/null +++ b/families/nemotron_h/cli.json @@ -0,0 +1,121 @@ +{ + "version": 1, + "commands": [ + { + "name": "build", + "help": "Build one nemotron_h TensorRT bundle", + "executor": "python", + "handler": "cli:build", + "arguments": [ + { + "name": "model", + "type": "string", + "help": "Hugging Face model ID or local snapshot" + }, + { + "name": "output", + "flags": [ + "-o", + "--output" + ], + "type": "path", + "required": true + }, + { + "name": "revision", + "flags": [ + "--revision" + ], + "type": "string" + }, + { + "name": "task", + "flags": [ + "--task" + ], + "type": "string", + "choices": [ + "text_generation" + ], + "default": "text_generation" + }, + { + "name": "precision", + "flags": [ + "--precision" + ], + "type": "string", + "choices": [ + "fp32", + "fp16", + "bf16" + ], + "default": "fp32" + }, + { + "name": "backend", + "flags": [ + "--backend" + ], + "type": "string", + "choices": [ + "trt", + "trt_rtx" + ], + "default": "trt" + }, + { + "name": "max_sequence_length", + "flags": [ + "--max-sequence-length" + ], + "type": "int" + }, + { + "name": "tensor_parallel_size", + "flags": [ + "--tensor-parallel-size" + ], + "type": "int", + "choices": [ + 1, + 2, + 4, + 8 + ], + "default": 1 + }, + { + "name": "verbose", + "flags": [ + "--verbose" + ], + "type": "bool", + "action": "store_true", + "default": false + }, + { + "name": "execution_variant", + "flags": [ + "--execution-variant" + ], + "type": "string", + "choices": [ + "dflash" + ], + "help": "Explicit paired execution mode; no automatic draft discovery" + }, + { + "name": "companion", + "flags": [ + "--companion" + ], + "type": "string", + "action": "append", + "default": [], + "help": "Local companion checkpoint as ROLE=LOCAL_DIR" + } + ] + } + ] +} diff --git a/families/nemotron_h/cli.py b/families/nemotron_h/cli.py new file mode 100644 index 0000000000..4f5f414169 --- /dev/null +++ b/families/nemotron_h/cli.py @@ -0,0 +1,57 @@ +# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-License-Identifier: Apache-2.0 + +"""nemotron_h-owned build command and typed inputs; importing this module is CPU-only.""" + +from __future__ import annotations + +from dataclasses import replace +from pathlib import Path + +from tensorrt_model_connect.build import select_backend +from tensorrt_model_connect.bundle_writer import BundleWriter +from tensorrt_model_connect.graph_transform import graph_transform +from tensorrt_model_connect.model_support import load_model_metadata, resolve_family, resolve_model + +from .build_request import BuildRequest + +from .edge_llm.cli import execution_inputs +from .edge_llm.config import with_execution + + +def build_bundle(request: BuildRequest, output: Path) -> None: + """Build and publish through the owning model, preserving atomic failure.""" + request = replace(request, output_path=output) + select_backend(request.backend) + from .model import build as build_model + + writer = BundleWriter(output) + try: + with graph_transform(request.graph_transform): + build_model(request, writer) + writer.finish() + except BaseException: + writer.abort() + raise + +def build( + *, model: str, output: Path, revision: str | None = None, + task: str = "text_generation", precision: str = "fp32", backend: str = "trt", + max_sequence_length: int | None = None, tensor_parallel_size: int = 1, + verbose: bool = False, + execution_variant: str | None = None, companion: list[str] | tuple[str, ...] = (), +) -> int: + """Run the declared owner command; help never imports this handler.""" + execution = execution_inputs(execution_variant, companion) + model_dir = resolve_model(model, revision) + resolve_family(load_model_metadata(model_dir), "nemotron_h") + request = BuildRequest( + model_dir=model_dir, output_path=output, family="nemotron_h", + task=task, precision=precision, backend=backend, + max_sequence_length=max_sequence_length, tensor_parallel_size=tensor_parallel_size, + verbose=verbose, + ) + if execution is not None: + request = with_execution(request, execution) + build_bundle(request, output) + return 0 diff --git a/families/nemotron_h/edge_llm/README.md b/families/nemotron_h/edge_llm/README.md index d68fc90cf0..0a1f033189 100644 --- a/families/nemotron_h/edge_llm/README.md +++ b/families/nemotron_h/edge_llm/README.md @@ -114,17 +114,18 @@ No quality gate, failed result or hardware limitation is hidden by these passes. ## Family-owned build options -The generic CLI loads this family's `edge_llm.cli` hook only after resolving -the model. The shared build API has no execution-mode or companion arguments. +The existing family CLI reads this owner's cli.json and invokes cli.py. +Edge-specific inputs and selection remain in edge_llm/; the shared parser, +CLI protocol and build API gain no new options or hooks. All variant validation and builder selection remain in this family. ```sh -trtmc build /path/to/target --family nemotron_h --precision fp16 \ +trtmc nemotron_h build /path/to/target --precision fp16 \ --execution-variant dflash --companion draft=/path/to/draft \ -o model.bundle ``` -Put MODEL before family-specific options; known core options may precede MODEL. `trtmc build /path/to/target --help` +Options may precede or follow MODEL. `trtmc nemotron_h build /path/to/target --help` shows these family options using local metadata; remote-ID help does not download a checkpoint. For Python callers, use this family's request extension: @@ -147,3 +148,9 @@ Ordinary builds without an installed optional Edge SDK select native without a warning. Malformed or incomplete installed packages still retain diagnostics and warn before native fallback. Temporary checkpoint/engine staging uses the bundle output directory filesystem (choose a scratch-backed output), not system /tmp. + +## Declared build command + +This family uses the existing cli.json protocol introduced in #1310. The family\nowns its declaration, typed inputs and Python handler. The handler adapts those\ninputs to the unchanged builder API, preserving native/Edge dispatch and bundle\npublication. The legacy flat build command remains available for its existing\nordinary options; new family options use `trtmc nemotron_h build`. +Help is offline and does not need a local checkpoint. No shared parser hook or +family registry entry is added. diff --git a/families/nemotron_h/edge_llm/cli.py b/families/nemotron_h/edge_llm/cli.py index b397c20273..ce6c3109f1 100644 --- a/families/nemotron_h/edge_llm/cli.py +++ b/families/nemotron_h/edge_llm/cli.py @@ -3,40 +3,27 @@ """NemotronH-owned CLI options for explicit paired Edge execution.""" -import argparse from pathlib import Path -from .config import BuildExecutionInputs, NamedCheckpoint, with_execution +from .config import BuildExecutionInputs, NamedCheckpoint -def add_build_arguments(parser: argparse.ArgumentParser) -> None: - """Register options only when the NemotronH family has been resolved.""" - parser.add_argument("--execution-variant", choices=("dflash",), - help="Explicit NemotronH paired execution mode") - parser.add_argument("--companion", action="append", default=[], metavar="ROLE=LOCAL_DIR", - help="Explicit local draft checkpoint; exactly one draft role is required") - - -def _execution_inputs(args: argparse.Namespace) -> BuildExecutionInputs | None: +def execution_inputs( + execution_variant: str | None, companion: list[str] | tuple[str, ...] = (), +) -> BuildExecutionInputs | None: """Parse only explicit local inputs; no variant list or model acquisition.""" - if args.command != "build": - return None - if args.execution_variant is None: - if args.companion: + if execution_variant is None: + if companion: raise ValueError("--companion requires --execution-variant") return None + if execution_variant not in ['dflash']: + raise ValueError("unsupported nemotron_h execution variant") checkpoints = [] - for value in args.companion: + for value in companion: role, separator, directory = value.partition("=") if not separator or not role or not directory: raise ValueError("--companion must be ROLE=LOCAL_DIR") if "://" in directory: raise ValueError("--companion requires a local directory, not a URI") checkpoints.append(NamedCheckpoint(role, Path(directory))) - return BuildExecutionInputs(args.execution_variant, tuple(checkpoints)) - - -def prepare_build_request(request, args): - """Attach a validated family-owned recipe before importing the GPU builder.""" - execution = _execution_inputs(args) - return request if execution is None else with_execution(request, execution) + return BuildExecutionInputs(execution_variant, tuple(checkpoints)) diff --git a/families/nemotron_h/edge_llm/config.py b/families/nemotron_h/edge_llm/config.py index 27ae74f9c9..f0b40b5fb8 100644 --- a/families/nemotron_h/edge_llm/config.py +++ b/families/nemotron_h/edge_llm/config.py @@ -7,7 +7,7 @@ from pathlib import Path import re -from tensorrt_model_connect.build import BuildRequest +from ..build_request import BuildRequest, coerce_request _ID = re.compile(r"[a-z][a-z0-9_]*\Z") @@ -81,7 +81,8 @@ def __post_init__(self) -> None: def with_execution(request: BuildRequest, execution: BuildExecutionInputs) -> NemotronHBuildRequest: - """Preserve ordinary request fields and callback identity.""" + """Preserve supported ordinary request fields and callback identity.""" + request = coerce_request(request) return NemotronHBuildRequest( **{field.name: getattr(request, field.name) for field in fields(BuildRequest)}, execution=execution, diff --git a/families/nemotron_h/model.py b/families/nemotron_h/model.py index 89c3acdcb7..6340a5a7a7 100644 --- a/families/nemotron_h/model.py +++ b/families/nemotron_h/model.py @@ -73,7 +73,7 @@ def _parse_layer_types(pattern: str) -> list[str]: if TYPE_CHECKING: - from tensorrt_model_connect.build import BuildRequest + from .build_request import BuildRequest from tensorrt_model_connect.bundle_writer import BundleWriter @@ -994,6 +994,9 @@ def _runtime_config( def build(request: "BuildRequest", writer: "BundleWriter") -> None: """Dispatch complete offload or execute the unchanged native implementation.""" + from .build_request import coerce_request + + request = coerce_request(request) from .edge_llm.config import NemotronHBuildRequest from .edge_llm.dispatch import build_paired diff --git a/families/nemotron_h/support.py b/families/nemotron_h/support.py index 11c331a813..0accd09428 100644 --- a/families/nemotron_h/support.py +++ b/families/nemotron_h/support.py @@ -10,5 +10,4 @@ model_types=("nemotron_h", "nemotron_hybrid", "nemotronh"), tasks=("text_generation",), default_task="text_generation", - build_cli_module="edge_llm.cli", ) diff --git a/families/nemotron_h/tests/test_runtime_contract.py b/families/nemotron_h/tests/test_runtime_contract.py index 5776c8fec1..72bc6aba60 100644 --- a/families/nemotron_h/tests/test_runtime_contract.py +++ b/families/nemotron_h/tests/test_runtime_contract.py @@ -85,7 +85,7 @@ def _edge_cli_source(tmp_path): def test_edge_cli_uses_ordinary_family_build(tmp_path, monkeypatch): - from tensorrt_model_connect import build_cli + from tensorrt_model_connect import family_cli as build_cli from families.nemotron_h.edge_llm import dispatch from families.nemotron_h.edge_llm.config import NemotronHBuildRequest @@ -103,7 +103,7 @@ def paired(request, writer, execution): writer.add_json("edge-test.json", {"variant": execution.variant}) monkeypatch.setattr(dispatch, "build_paired", paired) - assert build_cli.main([ + assert build_cli.main(["nemotron_h", "build", str(source), "--precision", "fp16", "-o", str(output), "--execution-variant", "dflash", "--companion", f"draft={draft}", ]) == 0 @@ -120,22 +120,22 @@ def paired(request, writer, execution): ]) def test_bad_edge_cli_inputs_fail_before_backend(tmp_path, monkeypatch, options): import importlib - from tensorrt_model_connect import build_cli + from tensorrt_model_connect import family_cli as build_cli core = importlib.import_module("tensorrt_model_connect.build") source, _ = _edge_cli_source(tmp_path) monkeypatch.setattr(core, "_select_backend", lambda *_: pytest.fail("backend touched")) monkeypatch.setattr(core, "BundleWriter", lambda *_: pytest.fail("writer created")) with pytest.raises(ValueError): - build_cli.main(["build", str(source), "-o", str(tmp_path / "out"), *options]) + build_cli.main(["nemotron_h", "build", str(source), "-o", str(tmp_path / "out"), *options]) def test_edge_cli_help_is_family_owned(tmp_path, capsys): - from tensorrt_model_connect import build_cli + from tensorrt_model_connect import family_cli as build_cli source, _ = _edge_cli_source(tmp_path) with pytest.raises(SystemExit) as caught: - build_cli.main(["build", str(source), "--help"]) + build_cli.main(["nemotron_h", "build", str(source), "--help"]) assert caught.value.code == 0 help_text = capsys.readouterr().out assert "--execution-variant {dflash}" in help_text @@ -143,7 +143,6 @@ def test_edge_cli_help_is_family_owned(tmp_path, capsys): def test_edge_request_preserves_fields_and_family_owner(tmp_path): - import argparse from families.nemotron_h.edge_llm import cli from dataclasses import fields, replace from tensorrt_model_connect.build import BuildRequest @@ -155,8 +154,7 @@ def test_edge_request_preserves_fields_and_family_owner(tmp_path): request = BuildRequest(source, tmp_path / "out", "nemotron_h", "text_generation", "fp16", graph_transform=lambda layer: layer) execution = BuildExecutionInputs("dflash", (NamedCheckpoint("draft", draft),)) - args = argparse.Namespace(command="build", execution_variant=None, companion=[]) - assert cli.prepare_build_request(request, args) is request + assert cli.execution_inputs(None) is None extended = with_execution(request, execution) for field in fields(BuildRequest): assert getattr(extended, field.name) is getattr(request, field.name) @@ -168,7 +166,7 @@ def test_edge_request_preserves_fields_and_family_owner(tmp_path): @pytest.mark.parametrize("failure", [RuntimeError("paired build failed"), KeyboardInterrupt()]) def test_edge_cli_failure_preserves_existing_bundle(tmp_path, monkeypatch, failure): - from tensorrt_model_connect import build_cli + from tensorrt_model_connect import family_cli as build_cli from families.nemotron_h.edge_llm import dispatch source, draft = _edge_cli_source(tmp_path) @@ -182,7 +180,7 @@ def fail(request, writer, execution): monkeypatch.setattr(dispatch, "build_paired", fail) with pytest.raises(type(failure)) as caught: - build_cli.main([ + build_cli.main(["nemotron_h", "build", str(source), "-o", str(output), "--execution-variant", "dflash", "--companion", f"draft={draft}", ]) @@ -280,3 +278,61 @@ def native(*args): assert "Retrying native once" in caplog.text elif mode != "cancel": assert not logs and "Edge build failed" not in caplog.text + + +@pytest.mark.parametrize("options", [[], ["--precision", "fp16", "--max-sequence-length", "64"]]) +def test_declared_build_matches_legacy_request(tmp_path, monkeypatch, options): + """Owner command preserves ordinary request defaults and explicit controls.""" + import json + from families.nemotron_h import cli as owner + from tensorrt_model_connect import build_cli, family_cli + + source = tmp_path / "checkpoint" + source.mkdir() + (source / "config.json").write_text(json.dumps({"model_type": "nemotron_h"})) + output = tmp_path / "model.bundle" + captured = [] + monkeypatch.setattr(owner, "build_bundle", lambda request, output: captured.append(request)) + monkeypatch.setattr(build_cli, "build", captured.append) + args = [str(source), "-o", str(output), *options] + assert family_cli.main(["nemotron_h", "build", *args]) == 0 + assert build_cli.main(["build", *args, "--family", "nemotron_h"]) == 0 + assert len(captured) == 2 + from dataclasses import fields + assert isinstance(captured[0], owner.BuildRequest) + for field in fields(captured[1]): + assert getattr(captured[0], field.name) == getattr(captured[1], field.name) + from dataclasses import replace + from families.nemotron_h.build_request import coerce_request + assert coerce_request(captured[1]) == captured[0] + with pytest.raises(NotImplementedError, match="image_height"): + coerce_request(replace(captured[1], image_height=32)) + from types import SimpleNamespace + with pytest.raises(ValueError, match="unknown"): + coerce_request(SimpleNamespace(**vars(captured[1]), unexpected_option=True)) + assert captured[0].family == "nemotron_h" + assert captured[0].task == "text_generation" + assert captured[0].precision == ("fp16" if options else "fp32") + assert not output.exists() + + +def test_declared_help_is_offline_and_dependency_free(): + """Actual child-process help needs neither a checkpoint nor GPU imports.""" + import subprocess + import sys + + code = """ +import sys +from tensorrt_model_connect.family_cli import main +try: + main(["nemotron_h", "build", "--help"]) +except SystemExit as error: + assert error.code == 0 +else: + raise AssertionError("help did not exit") +assert "families.nemotron_h.cli" not in sys.modules +assert "tensorrt" not in sys.modules +assert "huggingface_hub" not in sys.modules +""" + result = subprocess.run([sys.executable, "-c", code], capture_output=True, text=True, check=True) + assert "trtmc nemotron_h build" in result.stdout diff --git a/website/docs/features/model-families.md b/website/docs/features/model-families.md index 213a3bde0c..9a2d5ed328 100644 --- a/website/docs/features/model-families.md +++ b/website/docs/features/model-families.md @@ -111,6 +111,11 @@ imply CI coverage of that pair or statistical sampling equivalence. ### Nemotron-H Edge-LLM execution +Use `trtmc nemotron_h build MODEL -o model.bundle` with the owning +family\u0027s options. `trtmc nemotron_h build --help` works offline without +a checkpoint or GPU imports. This uses the existing +[family CLI protocol](../extend/family-cli.md), not an extension to the shared parser. + The Nemotron-H family owns optional pinned Edge-LLM whole-network offload. Ordinary compatible text builds use the experimental builder; the explicit Lightning NVFP4/DFlash pair uses ONNX with From 1e6e45cc7667bb2f6c0e9742a279eeef1f7f7346 Mon Sep 17 00:00:00 2001 From: Joshua Calafato Date: Wed, 23 Sep 2026 21:51:55 +0000 Subject: [PATCH 7/9] fix(nemotron_h): address Edge review findings Signed-off-by: Joshua Calafato --- families/nemotron_h/build_request.py | 2 ++ families/nemotron_h/cli.json | 3 +-- families/nemotron_h/cli.py | 2 ++ families/nemotron_h/edge_llm/README.md | 8 ++++++-- families/nemotron_h/edge_llm/dispatch.py | 3 +++ .../nemotron_h/tests/test_runtime_contract.py | 16 +++++++++++----- website/docs/features/model-families.md | 2 +- 7 files changed, 26 insertions(+), 10 deletions(-) diff --git a/families/nemotron_h/build_request.py b/families/nemotron_h/build_request.py index 0bf48a4567..9f52f82be7 100644 --- a/families/nemotron_h/build_request.py +++ b/families/nemotron_h/build_request.py @@ -88,6 +88,8 @@ def coerce_request(request: object) -> BuildRequest: value = getattr(request, name, default) if name == "quantization" and value == "none": continue + if name == "fp32_layers" and isinstance(value, (list, tuple)) and not value: + continue if value != default: raise NotImplementedError(f"nemotron_h does not support {name}") names = {field.name for field in fields(BuildRequest)} diff --git a/families/nemotron_h/cli.json b/families/nemotron_h/cli.json index 9f4685f2e2..9863f95b97 100644 --- a/families/nemotron_h/cli.json +++ b/families/nemotron_h/cli.json @@ -47,8 +47,7 @@ "type": "string", "choices": [ "fp32", - "fp16", - "bf16" + "fp16" ], "default": "fp32" }, diff --git a/families/nemotron_h/cli.py b/families/nemotron_h/cli.py index 4f5f414169..3d82801280 100644 --- a/families/nemotron_h/cli.py +++ b/families/nemotron_h/cli.py @@ -42,6 +42,8 @@ def build( execution_variant: str | None = None, companion: list[str] | tuple[str, ...] = (), ) -> int: """Run the declared owner command; help never imports this handler.""" + if precision not in {"fp32", "fp16"}: + raise ValueError("Nemotron-H precision must be fp32 or fp16") execution = execution_inputs(execution_variant, companion) model_dir = resolve_model(model, revision) resolve_family(load_model_metadata(model_dir), "nemotron_h") diff --git a/families/nemotron_h/edge_llm/README.md b/families/nemotron_h/edge_llm/README.md index 0a1f033189..5d8c71ff1d 100644 --- a/families/nemotron_h/edge_llm/README.md +++ b/families/nemotron_h/edge_llm/README.md @@ -107,7 +107,7 @@ the other exact models/pair used local recipes with existing family helpers. The separate direct-Edge 9B-NVFP4 capacity 1024 run failed ROUGE-L 0.1957 against 0.20 on a different SM120 GPU. The MC256 pass does not resolve that failure. -The earlier 4B-BF16 capacity4096 compiler failure, long-context/context-reuse, +The earlier 4B-BF16 capacity 4096 compiler failure, long-context/context-reuse, TP4, other models/platforms and stochastic equivalence remain outside this proof. Nano30B, Super120B and Omni are not qualified by the table above. No quality gate, failed result or hardware limitation is hidden by these passes. @@ -151,6 +151,10 @@ output directory filesystem (choose a scratch-backed output), not system /tmp. ## Declared build command -This family uses the existing cli.json protocol introduced in #1310. The family\nowns its declaration, typed inputs and Python handler. The handler adapts those\ninputs to the unchanged builder API, preserving native/Edge dispatch and bundle\npublication. The legacy flat build command remains available for its existing\nordinary options; new family options use `trtmc nemotron_h build`. +This family uses the existing cli.json protocol introduced in #1310. The family +owns its declaration, typed inputs and Python handler. The handler adapts those +inputs to the unchanged builder API, preserving native/Edge dispatch and bundle +publication. The legacy flat build command remains available for its existing +ordinary options; new family options use `trtmc nemotron_h build`. Help is offline and does not need a local checkpoint. No shared parser hook or family registry entry is added. diff --git a/families/nemotron_h/edge_llm/dispatch.py b/families/nemotron_h/edge_llm/dispatch.py index e9eaa1df57..1a8ed32377 100644 --- a/families/nemotron_h/edge_llm/dispatch.py +++ b/families/nemotron_h/edge_llm/dispatch.py @@ -101,6 +101,9 @@ def build(request, writer, native) -> None: traceback.print_exception(error, file=log) _LOG.warning("nemotron_h Edge build failed: %s. Diagnostics: %s. " "Retrying native once with the unchanged request.", error, log_path, exc_info=True) + except BaseException: + log_path.unlink(missing_ok=True) + raise else: if adapter is not None: edge_llm.publish(request, writer, files, marker) diff --git a/families/nemotron_h/tests/test_runtime_contract.py b/families/nemotron_h/tests/test_runtime_contract.py index 72bc6aba60..2a48fb12a9 100644 --- a/families/nemotron_h/tests/test_runtime_contract.py +++ b/families/nemotron_h/tests/test_runtime_contract.py @@ -119,13 +119,12 @@ def paired(request, writer, execution): ["--execution-variant", "dflash", "--companion", "draft=https://example.com/model"], ]) def test_bad_edge_cli_inputs_fail_before_backend(tmp_path, monkeypatch, options): - import importlib from tensorrt_model_connect import family_cli as build_cli - core = importlib.import_module("tensorrt_model_connect.build") + from families.nemotron_h import cli as owner source, _ = _edge_cli_source(tmp_path) - monkeypatch.setattr(core, "_select_backend", lambda *_: pytest.fail("backend touched")) - monkeypatch.setattr(core, "BundleWriter", lambda *_: pytest.fail("writer created")) + monkeypatch.setattr(owner, "select_backend", lambda *_: pytest.fail("backend touched")) + monkeypatch.setattr(owner, "BundleWriter", lambda *_: pytest.fail("writer created")) with pytest.raises(ValueError): build_cli.main(["nemotron_h", "build", str(source), "-o", str(tmp_path / "out"), *options]) @@ -276,7 +275,7 @@ def native(*args): if mode in {"corrupt", "failure", "device_failure"}: assert len(logs) == 1 and "Traceback" in logs[0].read_text() assert "Retrying native once" in caplog.text - elif mode != "cancel": + else: assert not logs and "Edge build failed" not in caplog.text @@ -305,6 +304,9 @@ def test_declared_build_matches_legacy_request(tmp_path, monkeypatch, options): from dataclasses import replace from families.nemotron_h.build_request import coerce_request assert coerce_request(captured[1]) == captured[0] + assert coerce_request(replace(captured[1], fp32_layers=[])) == captured[0] + with pytest.raises(NotImplementedError, match="fp32_layers"): + coerce_request(replace(captured[1], fp32_layers=[0])) with pytest.raises(NotImplementedError, match="image_height"): coerce_request(replace(captured[1], image_height=32)) from types import SimpleNamespace @@ -336,3 +338,7 @@ def test_declared_help_is_offline_and_dependency_free(): """ result = subprocess.run([sys.executable, "-c", code], capture_output=True, text=True, check=True) assert "trtmc nemotron_h build" in result.stdout + assert "bf16" not in result.stdout + from families.nemotron_h import cli as owner + with pytest.raises(ValueError, match="precision"): + owner.build(model="/missing", output=Path("/unused"), precision="bf16") diff --git a/website/docs/features/model-families.md b/website/docs/features/model-families.md index 9a2d5ed328..e4889a9a20 100644 --- a/website/docs/features/model-families.md +++ b/website/docs/features/model-families.md @@ -112,7 +112,7 @@ imply CI coverage of that pair or statistical sampling equivalence. ### Nemotron-H Edge-LLM execution Use `trtmc nemotron_h build MODEL -o model.bundle` with the owning -family\u0027s options. `trtmc nemotron_h build --help` works offline without +family's options. `trtmc nemotron_h build --help` works offline without a checkpoint or GPU imports. This uses the existing [family CLI protocol](../extend/family-cli.md), not an extension to the shared parser. From b757fde9c21f2b90eb23c7d66fbb835fe49163ac Mon Sep 17 00:00:00 2001 From: Joshua Calafato Date: Thu, 1 Oct 2026 02:37:45 +0000 Subject: [PATCH 8/9] fix(nemotron_h): keep E2E execution family-owned The E2E helper still passed execution to the removed shared build API. Put paired controls into the existing family-owned request before calling the unchanged generic builder. Extend the existing native and paired regression tests to exercise the E2E helper. Preserve all model inputs and quality gates. Signed-off-by: Joshua Calafato --- families/nemotron_h/tests/test_e2e.py | 28 ++++++++++--------- .../nemotron_h/tests/test_runtime_contract.py | 27 ++++++++++++++++++ 2 files changed, 42 insertions(+), 13 deletions(-) diff --git a/families/nemotron_h/tests/test_e2e.py b/families/nemotron_h/tests/test_e2e.py index 0d33d565d2..20baed7af5 100644 --- a/families/nemotron_h/tests/test_e2e.py +++ b/families/nemotron_h/tests/test_e2e.py @@ -138,20 +138,22 @@ def _build_bundle(manifest: dict, model_dir: Path, bundle: Path, *, execution=No quantization = manifest.get("quantization") assert quantization is None or isinstance(quantization, str) fp32_layers = tuple(manifest.get("fp32_layers", ())) - build( - BuildRequest( - model_dir=model_dir, - output_path=bundle, - family=_FAMILY, - task="text_generation", - precision=manifest["precision"], - max_sequence_length=manifest["max_sequence_length"], - tensor_parallel_size=manifest["tensor_parallel_size"], - quantization=quantization, - fp32_layers=fp32_layers, - ), - execution=execution, + request = BuildRequest( + model_dir=model_dir, + output_path=bundle, + family=_FAMILY, + task="text_generation", + precision=manifest["precision"], + max_sequence_length=manifest["max_sequence_length"], + tensor_parallel_size=manifest["tensor_parallel_size"], + quantization=quantization, + fp32_layers=fp32_layers, ) + if execution is not None: + from families.nemotron_h.edge_llm.config import with_execution + + request = with_execution(request, execution) + build(request) assert bundle.is_file() and bundle.stat().st_size > 0, bundle diff --git a/families/nemotron_h/tests/test_runtime_contract.py b/families/nemotron_h/tests/test_runtime_contract.py index 2a48fb12a9..520b7eb821 100644 --- a/families/nemotron_h/tests/test_runtime_contract.py +++ b/families/nemotron_h/tests/test_runtime_contract.py @@ -110,6 +110,18 @@ def paired(request, writer, execution): assert len(seen) == 1 assert output.is_file() + from families.nemotron_h.tests.test_e2e import _build_bundle + from families.nemotron_h.edge_llm.config import BuildExecutionInputs, NamedCheckpoint + + _build_bundle( + {"precision": "fp16", "max_sequence_length": 64, "tensor_parallel_size": 1}, + source, output, + execution=BuildExecutionInputs("dflash", (NamedCheckpoint("draft", draft),)), + ) + assert len(seen) == 2 + assert seen[-1].max_sequence_length == 64 + assert output.is_file() + @pytest.mark.parametrize("options", [ ["--companion", "draft=/missing"], @@ -317,6 +329,21 @@ def test_declared_build_matches_legacy_request(tmp_path, monkeypatch, options): assert captured[0].precision == ("fp16" if options else "fp32") assert not output.exists() + from families.nemotron_h.tests import test_e2e as e2e + + def capture_bundle(request): + captured.append(request) + request.output_path.write_bytes(b"test bundle") + + monkeypatch.setattr(e2e, "build", capture_bundle) + e2e._build_bundle( + {"precision": captured[1].precision, + "max_sequence_length": captured[1].max_sequence_length, + "tensor_parallel_size": captured[1].tensor_parallel_size}, + source, output, + ) + assert captured[-1] == captured[1] + def test_declared_help_is_offline_and_dependency_free(): """Actual child-process help needs neither a checkpoint nor GPU imports.""" From ded9926e95e95fa30308e8b7004497348091fb61 Mon Sep 17 00:00:00 2001 From: Joshua Calafato Date: Fri, 2 Oct 2026 18:23:07 +0000 Subject: [PATCH 9/9] fix(nemotron_h): align paired defaults and request lifetime Default omitted paired CLI precision to FP16 while preserving ordinary FP32 and explicit choices. Keep the response alive until the existing drain guard runs during exceptions, and check family-owned coercion identity in the existing test. Signed-off-by: Joshua Calafato --- families/nemotron_h/cli.json | 2 +- families/nemotron_h/cli.py | 6 ++++-- families/nemotron_h/runtime/edge_llm/adapter.cpp | 2 +- families/nemotron_h/tests/test_runtime_contract.py | 11 ++++++++--- 4 files changed, 14 insertions(+), 7 deletions(-) diff --git a/families/nemotron_h/cli.json b/families/nemotron_h/cli.json index 9863f95b97..18cb258b76 100644 --- a/families/nemotron_h/cli.json +++ b/families/nemotron_h/cli.json @@ -49,7 +49,7 @@ "fp32", "fp16" ], - "default": "fp32" + "help": "Compute precision (default: fp16 for paired execution, fp32 otherwise)" }, { "name": "backend", diff --git a/families/nemotron_h/cli.py b/families/nemotron_h/cli.py index 3d82801280..8392fea882 100644 --- a/families/nemotron_h/cli.py +++ b/families/nemotron_h/cli.py @@ -36,15 +36,17 @@ def build_bundle(request: BuildRequest, output: Path) -> None: def build( *, model: str, output: Path, revision: str | None = None, - task: str = "text_generation", precision: str = "fp32", backend: str = "trt", + task: str = "text_generation", precision: str | None = None, backend: str = "trt", max_sequence_length: int | None = None, tensor_parallel_size: int = 1, verbose: bool = False, execution_variant: str | None = None, companion: list[str] | tuple[str, ...] = (), ) -> int: """Run the declared owner command; help never imports this handler.""" + execution = execution_inputs(execution_variant, companion) + if precision is None: + precision = "fp16" if execution is not None else "fp32" if precision not in {"fp32", "fp16"}: raise ValueError("Nemotron-H precision must be fp32 or fp16") - execution = execution_inputs(execution_variant, companion) model_dir = resolve_model(model, revision) resolve_family(load_model_metadata(model_dir), "nemotron_h") request = BuildRequest( diff --git a/families/nemotron_h/runtime/edge_llm/adapter.cpp b/families/nemotron_h/runtime/edge_llm/adapter.cpp index 1d4a45604f..dec7abbc8f 100644 --- a/families/nemotron_h/runtime/edge_llm/adapter.cpp +++ b/families/nemotron_h/runtime/edge_llm/adapter.cpp @@ -264,10 +264,10 @@ class EdgeTask final : public ITextGeneration { if (count > static_cast(input_limit_)) throw std::invalid_argument("Nemotron-H Edge input exceeds bundle capacity"); std::lock_guard lock(mutex_); - RequestDrain drain(stream_.get()); validate_capacity(static_cast(count), input_limit_, capacity_, request.maxGenerateLength); trt_edgellm::rt::LLMGenerationResponse response{}; + RequestDrain drain(stream_.get()); if (!runtime_.handleRequest(request, response, stream_.get())) throw std::runtime_error("Nemotron-H Edge generation failed"); drain.checked(); diff --git a/families/nemotron_h/tests/test_runtime_contract.py b/families/nemotron_h/tests/test_runtime_contract.py index 520b7eb821..7ce303003f 100644 --- a/families/nemotron_h/tests/test_runtime_contract.py +++ b/families/nemotron_h/tests/test_runtime_contract.py @@ -84,7 +84,8 @@ def _edge_cli_source(tmp_path): return source, draft -def test_edge_cli_uses_ordinary_family_build(tmp_path, monkeypatch): +@pytest.mark.parametrize("precision", [None, "fp16", "fp32"]) +def test_edge_cli_uses_ordinary_family_build(tmp_path, monkeypatch, precision): from tensorrt_model_connect import family_cli as build_cli from families.nemotron_h.edge_llm import dispatch from families.nemotron_h.edge_llm.config import NemotronHBuildRequest @@ -103,11 +104,13 @@ def paired(request, writer, execution): writer.add_json("edge-test.json", {"variant": execution.variant}) monkeypatch.setattr(dispatch, "build_paired", paired) + options = ["--precision", precision] if precision else [] assert build_cli.main(["nemotron_h", - "build", str(source), "--precision", "fp16", "-o", str(output), + "build", str(source), *options, "-o", str(output), "--execution-variant", "dflash", "--companion", f"draft={draft}", ]) == 0 assert len(seen) == 1 + assert seen[0].precision == (precision or "fp16") assert output.is_file() from families.nemotron_h.tests.test_e2e import _build_bundle @@ -119,6 +122,7 @@ def paired(request, writer, execution): execution=BuildExecutionInputs("dflash", (NamedCheckpoint("draft", draft),)), ) assert len(seen) == 2 + assert seen[-1].precision == "fp16" assert seen[-1].max_sequence_length == 64 assert output.is_file() @@ -156,7 +160,7 @@ def test_edge_cli_help_is_family_owned(tmp_path, capsys): def test_edge_request_preserves_fields_and_family_owner(tmp_path): from families.nemotron_h.edge_llm import cli from dataclasses import fields, replace - from tensorrt_model_connect.build import BuildRequest + from families.nemotron_h.build_request import BuildRequest, coerce_request from families.nemotron_h.edge_llm.config import ( BuildExecutionInputs, NamedCheckpoint, with_execution, ) @@ -164,6 +168,7 @@ def test_edge_request_preserves_fields_and_family_owner(tmp_path): source, draft = _edge_cli_source(tmp_path) request = BuildRequest(source, tmp_path / "out", "nemotron_h", "text_generation", "fp16", graph_transform=lambda layer: layer) + assert coerce_request(request) is request execution = BuildExecutionInputs("dflash", (NamedCheckpoint("draft", draft),)) assert cli.execution_inputs(None) is None extended = with_execution(request, execution)