diff --git a/families/nemotron_h/build_request.py b/families/nemotron_h/build_request.py new file mode 100644 index 0000000000..9f52f82be7 --- /dev/null +++ b/families/nemotron_h/build_request.py @@ -0,0 +1,98 @@ +# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-License-Identifier: Apache-2.0 + +"""nemotron_h build inputs and strict compatibility for existing Python callers.""" + +from __future__ import annotations + +from dataclasses import dataclass, fields +from pathlib import Path +import re +from typing import ClassVar + +from tensorrt_model_connect.graph_transform import GraphTransform + + +def _validate_id(field: str, value: str) -> None: + if not isinstance(value, str) or re.fullmatch(r"[a-z][a-z0-9_]*", value) is None: + raise ValueError(f"{field} must be a lowercase identifier") + + +@dataclass(frozen=True) +class BuildRequest: + """nemotron_h-owned inputs; unsupported legacy controls are read-only defaults.""" + + model_dir: Path + output_path: Path + family: str + task: str + precision: str + backend: str = "trt" + max_sequence_length: int | None = None + image_height: ClassVar[int | None] = None + image_width: ClassVar[int | None] = None + video_num_frames: ClassVar[int | None] = None + max_batch_size: ClassVar[int] = 1 + tensor_parallel_size: int = 1 + context_parallel_size: ClassVar[int] = 1 + quantization: ClassVar[str | None] = None + fp32_layers: ClassVar[tuple[int, ...]] = () + dynamic_kv_cache: ClassVar[bool] = False + verbose: bool = False + graph_transform: GraphTransform | None = None + + def __post_init__(self) -> None: + if not self.precision: + raise ValueError("precision must be non-empty") + _validate_id("family", self.family) + _validate_id("task", self.task) + if self.backend not in {"trt", "trt_rtx"}: + raise ValueError("backend must be 'trt' or 'trt_rtx'") + if self.max_sequence_length is not None and self.max_sequence_length < 1: + raise ValueError("max_sequence_length must be positive") + for field in ("image_height", "image_width", "video_num_frames"): + value = getattr(self, field) + if value is not None and value < 1: + raise ValueError(f"{field} must be positive") + if self.max_batch_size < 1: + raise ValueError("max_batch_size must be positive") + if self.tensor_parallel_size < 1: + raise ValueError("tensor_parallel_size must be positive") + if self.context_parallel_size < 1: + raise ValueError("context_parallel_size must be positive") + if self.quantization is not None and not self.quantization: + raise ValueError("quantization must be non-empty when provided") + if any(layer < 0 for layer in self.fp32_layers): + raise ValueError("fp32_layers must contain non-negative indices") + if not isinstance(self.dynamic_kv_cache, bool): + raise ValueError("dynamic_kv_cache must be a bool") + if self.graph_transform is not None and not callable(self.graph_transform): + raise ValueError("graph_transform must be callable when provided") + + +def coerce_request(request: object) -> BuildRequest: + """Reject unsupported/unknown legacy inputs before converting to owner fields.""" + if isinstance(request, BuildRequest): + return request + unsupported = { + "image_height": None, + "image_width": None, + "video_num_frames": None, + "max_batch_size": 1, + "context_parallel_size": 1, + "quantization": None, + "fp32_layers": (), + "dynamic_kv_cache": False, + } + for name, default in unsupported.items(): + value = getattr(request, name, default) + if name == "quantization" and value == "none": + continue + if name == "fp32_layers" and isinstance(value, (list, tuple)) and not value: + continue + if value != default: + raise NotImplementedError(f"nemotron_h does not support {name}") + names = {field.name for field in fields(BuildRequest)} + if unknown := set(vars(request)) - names - set(unsupported): + raise ValueError(f"unknown nemotron_h build inputs: {sorted(unknown)}") + return BuildRequest(**{name: getattr(request, name) for name in names}) diff --git a/families/nemotron_h/cli.json b/families/nemotron_h/cli.json new file mode 100644 index 0000000000..18cb258b76 --- /dev/null +++ b/families/nemotron_h/cli.json @@ -0,0 +1,120 @@ +{ + "version": 1, + "commands": [ + { + "name": "build", + "help": "Build one nemotron_h TensorRT bundle", + "executor": "python", + "handler": "cli:build", + "arguments": [ + { + "name": "model", + "type": "string", + "help": "Hugging Face model ID or local snapshot" + }, + { + "name": "output", + "flags": [ + "-o", + "--output" + ], + "type": "path", + "required": true + }, + { + "name": "revision", + "flags": [ + "--revision" + ], + "type": "string" + }, + { + "name": "task", + "flags": [ + "--task" + ], + "type": "string", + "choices": [ + "text_generation" + ], + "default": "text_generation" + }, + { + "name": "precision", + "flags": [ + "--precision" + ], + "type": "string", + "choices": [ + "fp32", + "fp16" + ], + "help": "Compute precision (default: fp16 for paired execution, fp32 otherwise)" + }, + { + "name": "backend", + "flags": [ + "--backend" + ], + "type": "string", + "choices": [ + "trt", + "trt_rtx" + ], + "default": "trt" + }, + { + "name": "max_sequence_length", + "flags": [ + "--max-sequence-length" + ], + "type": "int" + }, + { + "name": "tensor_parallel_size", + "flags": [ + "--tensor-parallel-size" + ], + "type": "int", + "choices": [ + 1, + 2, + 4, + 8 + ], + "default": 1 + }, + { + "name": "verbose", + "flags": [ + "--verbose" + ], + "type": "bool", + "action": "store_true", + "default": false + }, + { + "name": "execution_variant", + "flags": [ + "--execution-variant" + ], + "type": "string", + "choices": [ + "dflash" + ], + "help": "Explicit paired execution mode; no automatic draft discovery" + }, + { + "name": "companion", + "flags": [ + "--companion" + ], + "type": "string", + "action": "append", + "default": [], + "help": "Local companion checkpoint as ROLE=LOCAL_DIR" + } + ] + } + ] +} diff --git a/families/nemotron_h/cli.py b/families/nemotron_h/cli.py new file mode 100644 index 0000000000..8392fea882 --- /dev/null +++ b/families/nemotron_h/cli.py @@ -0,0 +1,61 @@ +# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-License-Identifier: Apache-2.0 + +"""nemotron_h-owned build command and typed inputs; importing this module is CPU-only.""" + +from __future__ import annotations + +from dataclasses import replace +from pathlib import Path + +from tensorrt_model_connect.build import select_backend +from tensorrt_model_connect.bundle_writer import BundleWriter +from tensorrt_model_connect.graph_transform import graph_transform +from tensorrt_model_connect.model_support import load_model_metadata, resolve_family, resolve_model + +from .build_request import BuildRequest + +from .edge_llm.cli import execution_inputs +from .edge_llm.config import with_execution + + +def build_bundle(request: BuildRequest, output: Path) -> None: + """Build and publish through the owning model, preserving atomic failure.""" + request = replace(request, output_path=output) + select_backend(request.backend) + from .model import build as build_model + + writer = BundleWriter(output) + try: + with graph_transform(request.graph_transform): + build_model(request, writer) + writer.finish() + except BaseException: + writer.abort() + raise + +def build( + *, model: str, output: Path, revision: str | None = None, + task: str = "text_generation", precision: str | None = None, backend: str = "trt", + max_sequence_length: int | None = None, tensor_parallel_size: int = 1, + verbose: bool = False, + execution_variant: str | None = None, companion: list[str] | tuple[str, ...] = (), +) -> int: + """Run the declared owner command; help never imports this handler.""" + execution = execution_inputs(execution_variant, companion) + if precision is None: + precision = "fp16" if execution is not None else "fp32" + if precision not in {"fp32", "fp16"}: + raise ValueError("Nemotron-H precision must be fp32 or fp16") + model_dir = resolve_model(model, revision) + resolve_family(load_model_metadata(model_dir), "nemotron_h") + request = BuildRequest( + model_dir=model_dir, output_path=output, family="nemotron_h", + task=task, precision=precision, backend=backend, + max_sequence_length=max_sequence_length, tensor_parallel_size=tensor_parallel_size, + verbose=verbose, + ) + if execution is not None: + request = with_execution(request, execution) + build_bundle(request, output) + return 0 diff --git a/families/nemotron_h/edge_llm/README.md b/families/nemotron_h/edge_llm/README.md new file mode 100644 index 0000000000..5d8c71ff1d --- /dev/null +++ b/families/nemotron_h/edge_llm/README.md @@ -0,0 +1,160 @@ +# Nemotron-H Edge-LLM execution + +The family owns source-policy admission, command mapping, tokenizer/EOS semantics, +engine composition and runtime orchestration. TensorRT Edge-LLM owns model graphs, +lowering and execution. Provision the [optional native SDK](../../../cmake/edge_llm/README.md) +from official GitHub Edge-LLM 0.10.1, revision +`e8b29522938901f6df19ebeedd4b69bc8edbcd97`. No internal source changes or +cross-compilation are used. + +## Ordinary and paired builds + +Ordinary compatible text builds use the installed experimental Python builder +with components=llm and original checkpoints. Plain source weights use FP16 +compute; packed FP8/NVFP4 or mixed policies remain in their declared formats. +The adapter preserves source scales, exclusions and KV policy. For packed +sources, only supported external-weight kinds are requested; FP16 bias tensors +are baked into the engine rather than externalized through a missing recipe. + +The explicit Lightning DFlash pair instead uses the original ONNX exporter and +native ONNX builder. Enable `TRTMC_EDGELLM_ALL_KERNELS=ON` and +`TRTMC_EDGELLM_ONNX=ON`, set `CMAKE_PREFIX_PATH` to the SDK installation, and +configure the runtime with `TRTMC_ENABLE_EDGELLM=ON`. Add +`--execution-variant dflash --companion draft=/path/to/draft` to the build CLI, +or wrap the request with this family's `with_execution(request, inputs)` +before calling `build`, as the Python example below shows. + +Both paired engines are mandatory. The ONNX graph supplies the intermediate +Mamba/replay states required to commit accepted draft tokens; the experimental +builder's final-only state outputs do not implement this pair. No base-only +engine is substituted. ONNX projection weights are baked into plans and are not +duplicated in the bundle. Successfully consumed ONNX intermediates are removed. + +Ordinary preparation failure warns, retains a diagnostic log and attempts the +unchanged native request once. Native fallback rejects packed checkpoints it +cannot interpret. Paired native fallback is unavailable and fails explicitly. +Cancellation, publication and runtime errors propagate without retry. + +## Request and runtime contracts + +Admission requires compatible Nemotron-H text topology and source quantization, +FP16 compute, TP 1/batch 1/context-parallel 1, and no unmapped build controls. +The published platform routes are native x86_64 SM80 for plain sources and +SM120 for plain/FP8/NVFP4 sources. These routes are admission rules, not proof for +every compatible checkpoint or capacity. + +Do not gate head80 on optimized CuTe SSD availability. The pinned Mamba plugin +has a scalar-prefill fallback, which the recorded SM80/SM120 profiles exercise. +A previous optimized-kernel-only restriction was incorrect and is removed. + +The family renders and tokenizes using its native prompt semantics, then supplies +actual pretokenized IDs to the public Edge API. Full native EOS lists are +preserved with validated derivative tokenizer metadata; original source assets +are unchanged. Separate Jinja templates follow source-file precedence, and the +modern Nemotron ChatML empty-system/thinking suffix is retained. Empty prompts, +submitted counts, total capacity and generation completion are checked. + +The runtime is persistent and serialized, with scoped plugin, stream and artifact +ownership. Supported ordinary sampling controls are forwarded; unmapped controls +are rejected. DFlash uses block16/verify16 and is greedy-only because the pinned +runtime forces greedy verification. Runtime errors are not converted to native +inference. + +## Recorded model qualification + +These are **historical local Model Connect build/inference qualifications**, +not assertions that every publication head or CI executes these profiles. +CUDA 13.3 and TensorRT 11.1.0.106 were used. Ordinary MC profiles use capacity 256; +the DFlash profile uses input/KV1024. All are text-only TP 1/batch 1. +Independent NED must be at most 0.15; the original Edge 128-token fixture gates +remain ROUGE-1 >=0.25 and ROUGE-L >=0.20. + +| Exact source model | GPU | Independent gate | MC ROUGE-1 / ROUGE-L | +| --- | --- | --- | --- | +| NVIDIA-Nemotron-3-Nano-4B-BF16 | SM120 | NED 0.0 | 0.5371 / 0.2514 | +| NVIDIA-Nemotron-3-Nano-4B-FP8 | SM120 | NED 0.0 | 0.5402 / 0.2644 | +| NVIDIA-Nemotron-Nano-9B-v2 | SM80 | Existing HF pytest gate passed; numeric NED not serialized | 0.4318 / 0.2727 | +| NVIDIA-Nemotron-Nano-9B-v2-FP8 | SM120 | NED 0.0 | 0.3750 / 0.2614 | +| NVIDIA-Nemotron-Nano-9B-v2-NVFP4 | SM120 | NED 0.0 | 0.4130 / 0.2391 | +| NVIDIA-Nemotron-3.5-Lightning-30B-A3B-NVFP4 | SM120 | NED 0.0 | 0.4643 / 0.2500 | +| Same Lightning target + DFlash companion | SM120 | NED 0.0 | 0.4848 / 0.2545 | + +All models are from the public `nvidia` namespace. Immutable revisions, in table +order, are: + +- 4B-BF16: `dfaf35de3e30f1867dd8dbc38a7fc9fb52d3914f`. +- 4B-FP8: `3fe6dab75665a93884214ad4b1b95cf02717d081`. +- 9B-v2: `6533e8de2c68e4536bf7c411d7a3ce5734111476`. +- 9B-v2-FP8: `8bc5eece2eb5514c4bca7f2ec655b91eb554f4c0`. +- 9B-v2-NVFP4: `8556c9164ddb43fe1f4f4ad730593b3c5e3f7328`. +- Lightning target: `bee7596271d1495f6992ae224aefde4410e816b8`. +- Lightning DFlash companion: `8abcc4db8f34a5080c31eef05d4467afc06c6b9e`. + +The independent references use builtin HF BF16 mathematical execution. Packed +sources are decoded with official ModelOpt routines and strict full-state loading; +these oracles do not emulate activation/KV quantization rounding. DFlash reused +a source-byte/reference-function-verified independent reference rather than +regenerating it. Its Edge fixture uses explicit greedy controls, not a claim of +sampling parity. + +Passing payloads were retired with approval; compact results and provenance were +retained. Replaying all seven model checks requires rebuilding those payloads. +Fresh publication compilation/unit/source checks must be reported separately. +Only the existing plain 9B case is a registered owning E2E among these profiles; +the other exact models/pair used local recipes with existing family helpers. + +## Still outside these qualifications + +The separate direct-Edge 9B-NVFP4 capacity 1024 run failed ROUGE-L 0.1957 against 0.20 +on a different SM120 GPU. The MC256 pass does not resolve that failure. +The earlier 4B-BF16 capacity 4096 compiler failure, long-context/context-reuse, +TP4, other models/platforms and stochastic equivalence remain outside this proof. +Nano30B, Super120B and Omni are not qualified by the table above. +No quality gate, failed result or hardware limitation is hidden by these passes. + +## Family-owned build options + +The existing family CLI reads this owner's cli.json and invokes cli.py. +Edge-specific inputs and selection remain in edge_llm/; the shared parser, +CLI protocol and build API gain no new options or hooks. +All variant validation and builder selection remain in this family. + +```sh +trtmc nemotron_h build /path/to/target --precision fp16 \ + --execution-variant dflash --companion draft=/path/to/draft \ + -o model.bundle +``` + +Options may precede or follow MODEL. `trtmc nemotron_h build /path/to/target --help` +shows these family options using local metadata; remote-ID help does not download +a checkpoint. For Python callers, use this family's request extension: + +```python +from tensorrt_model_connect import build +from families.nemotron_h.edge_llm.config import ( + BuildExecutionInputs, NamedCheckpoint, with_execution, +) + +# request is an ordinary BuildRequest owned by this family; draft_path is a Path. +build(with_execution(request, BuildExecutionInputs( + "dflash", (NamedCheckpoint("draft", draft_path),), +))) +``` + +A failed explicit pair is never replaced by a base-only bundle. Previously +recorded full-model results above are historical, not fresh refactor-head E2Es. + +Ordinary builds without an installed optional Edge SDK select native without a +warning. Malformed or incomplete installed packages still retain diagnostics and +warn before native fallback. Temporary checkpoint/engine staging uses the bundle +output directory filesystem (choose a scratch-backed output), not system /tmp. + +## Declared build command + +This family uses the existing cli.json protocol introduced in #1310. The family +owns its declaration, typed inputs and Python handler. The handler adapts those +inputs to the unchanged builder API, preserving native/Edge dispatch and bundle +publication. The legacy flat build command remains available for its existing +ordinary options; new family options use `trtmc nemotron_h build`. +Help is offline and does not need a local checkpoint. No shared parser hook or +family registry entry is added. diff --git a/families/nemotron_h/edge_llm/__init__.py b/families/nemotron_h/edge_llm/__init__.py new file mode 100644 index 0000000000..df93d8cabe --- /dev/null +++ b/families/nemotron_h/edge_llm/__init__.py @@ -0,0 +1,4 @@ +# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-License-Identifier: Apache-2.0 + +"""Family-owned optional complete-network Edge offload.""" diff --git a/families/nemotron_h/edge_llm/builder.py b/families/nemotron_h/edge_llm/builder.py new file mode 100644 index 0000000000..5bad995487 --- /dev/null +++ b/families/nemotron_h/edge_llm/builder.py @@ -0,0 +1,261 @@ +# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-License-Identifier: Apache-2.0 + +"""Thin family-owned adapter to the pinned Edge direct-builder API.""" + +from __future__ import annotations + +import json +from pathlib import Path +import shutil +import subprocess + +from tensorrt_model_connect.build import cmake_prefixes, detect_local_platform, subprocess_environment + +EDGE_REVISION = "e8b29522938901f6df19ebeedd4b69bc8edbcd97" +_QUANTIZED_EXTERNAL_WEIGHTS = ( + "int4_ffn", "int4_moe", "nvfp4_moe", "nvfp4_tp", "lm_head", "embedding", +) + + +def local_target() -> dict: + """Return the executing worker identity supplied by generic build mechanics.""" + return detect_local_platform() + + +def package_present() -> bool: + """Absence of the optional SDK is a non-match, not a failed Edge build.""" + for prefix in cmake_prefixes(): + manifest = prefix / "share/trtmc/edge-llm.json" + if manifest.exists() or manifest.is_symlink(): + return True + return False + + +def installed_package(target: dict) -> dict: + """Resolve CMake installation via standard prefixes; never install anything. + + Args: + target: Executing device and SDK identity. + + Returns: + Validated package metadata with absolute Python and plugin paths. + + Raises: + FileNotFoundError: No CMake installation or required artifact exists. + ValueError: Pin, architecture, SDK or contained-path contract differs. + """ + for prefix in cmake_prefixes(): + manifest = prefix / "share/trtmc/edge-llm.json" + if not manifest.is_file(): + continue + package = json.loads(manifest.read_text(encoding="utf-8")) + if package.get("schema_version") != 1 or package.get("revision") != EDGE_REVISION: + raise ValueError(f"Edge package has an unsupported revision/schema: {manifest}") + if package.get("all_native_kernels") is not True: + raise ValueError("Nemotron-H requires an Edge package with all native operator groups") + if package.get("version") != "0.10.1" or package.get("arch") != target["arch"]: + raise ValueError("Edge package version/architecture differs from executing worker") + if target["sm"] not in package.get("architectures", []): + raise ValueError("Edge package was not built for this local GPU") + cuda_version = ".".join(str(package.get("cuda_version", "")).split(".")[:2]) + if cuda_version != target["cuda_version"] or package.get("tensorrt_version") != target["tensorrt_version"]: + raise ValueError("Edge package CUDA/TensorRT differs from executing worker") + for name in ("python", "plugin") + (("onnx_builder",) if package.get("onnx") is True else ()): + relative = Path(package[name]) + path = (prefix / relative).resolve() + if relative.is_absolute() or not path.is_relative_to(prefix.resolve()): + raise ValueError(f"Edge package {name} must be contained in its installation") + if not path.is_file(): + raise FileNotFoundError(f"Edge package {name} is missing: {path}") + package[name] = str(path) + return package + raise FileNotFoundError("Edge-LLM is not installed; enable the optional Edge-LLM CMake dependency " + "and set CMAKE_PREFIX_PATH to its install prefix") + + +def sequence_length(request, raw: dict) -> int: + """Resolve the same default capacity as this family's original native builder.""" + value = request.max_sequence_length or min(raw["max_position_embeddings"], 256) + if isinstance(value, bool): + raise ValueError("max_sequence_length must be a positive integer") + try: + result = int(value) + except (TypeError, ValueError, OverflowError) as error: + raise ValueError("max_sequence_length must be a positive integer") from error + if result < 1: + raise ValueError("max_sequence_length must be a positive integer") + return result + + +def prepare(request, raw: dict, target: dict, staging: Path, log_path: Path) -> tuple[dict, dict]: + """Map the request to Edge main(argv), returning complete unpublished assets. + + Edge owns model selection, configuration, conversion, graphs and engine + composition. The family adapter only maps text-generation arguments and + preserves the checkpoint needed by Edge external-weight APIs. + + Returns: + (section-name to file mapping, runtime marker). + + Raises: + Exception: Dependency, upstream build or artifact validation failed. + """ + from .edge_quantization import source_quantization + + source_precision = source_quantization(request, raw) + if source_precision is None: + raise ValueError("Unsupported Nemotron-H checkpoint precision") + package = installed_package(target) + checkpoint = staging / "edge_llm/checkpoint" + checkpoint.mkdir(parents=True) + for source in Path(request.model_dir).iterdir(): + if source.is_file() and (source.suffix in {".json", ".safetensors", ".model", ".jinja"} + or source.name in {"merges.txt", "vocab.txt"}): + shutil.copy2(source, checkpoint / source.name) + if not list(checkpoint.glob("*.safetensors")): + raise ValueError("Edge direct builder requires a safetensors checkpoint") + limit = sequence_length(request, raw) + engine = staging / "edge_llm/engine" + # Calling upstream main preserves its complete build/artifact orchestration. + command = [package["python"], "-I", "-c", + "from experimental.builder.cli import main; main()", + "--model-dir", str(checkpoint), "--engine-dir", str(engine), + "--components", "llm", "--plugin-path", package["plugin"], + "--dense", "fp16" if source_precision == "fp16" else "auto", "--max-input-len", str(limit), + "--max-kv-cache-capacity", str(limit), "--max-batch-size", "1"] + # The pinned builder cannot externalize FP16 biases on quantized projections: + # their checkpoint recipes are missing. Bake FP16 parameters (which can enlarge the plan), + # retaining the original packed quantized weights and every other supported kind. + external_weights = ("all",) if source_precision == "fp16" else _QUANTIZED_EXTERNAL_WEIGHTS + for kind in external_weights: + command.extend(("--externalize-weights", kind)) + if request.verbose: + command.append("--verbose") + with log_path.open("a", encoding="utf-8") as log: + subprocess.run(command, check=True, stdout=log, stderr=subprocess.STDOUT, cwd=staging) + for name in ("llm.engine", "config.json", "tokenizer.json", "tokenizer_config.json", "processed_chat_template.json"): + if not (engine / name).is_file() or (engine / name).stat().st_size == 0: + raise ValueError(f"Edge builder did not produce required artifact: {name}") + from .edge_tokenizer import prepare_tokenizer, source_chat_template + + runtime_tokenizer = staging / "edge_llm/runtime_tokenizer" + eos = prepare_tokenizer(checkpoint, engine, runtime_tokenizer, raw, + chat_template=source_chat_template(checkpoint)) + files = {} + for directory in (engine, checkpoint, runtime_tokenizer): + for path in sorted(directory.rglob("*")): + if path.is_symlink(): + raise ValueError(f"Edge output must not contain symlinks: {path}") + if path.is_file(): + files[path.relative_to(staging).as_posix()] = path + return files, { + "version": 1, "edge_revision": EDGE_REVISION, "target": target, "precision": "fp16", + "max_sequence_length": limit, "max_input_length": limit, "max_batch_size": 1, + "artifacts": list(files), "checkpoint_precision": source_precision, + "tokenizer_policy": "native_full_eos", "native_eos_token_ids": eos, + "native_bos_token_id": raw.get("bos_token_id", -1), + } + + +def publish(request, writer, files: dict, marker: dict) -> None: + """Stream complete Edge sections; publication errors must not retry native.""" + writer.set_header(family=request.family, task=request.task, backend=request.backend) + for name, path in files.items(): + with path.open("rb") as source, writer.open_section(name) as destination: + shutil.copyfileobj(source, destination, length=1024 * 1024) + writer.add_json("edge_llm.json", marker) + + +def prepare_dflash(request, raw: dict, target: dict, staging: Path, log_path: Path, + draft: Path) -> tuple[dict, dict]: + """Build both recurrent-state-aware graphs with the original ONNX toolchain. + + Export consumes the caller's immutable local checkpoints. ONNX plans bake + projection weights, so the bundle needs only engine assets and tokenizer + metadata, not a duplicate of either checkpoint's packed weights. + """ + from .edge_quantization import source_quantization + from .edge_tokenizer import prepare_tokenizer, source_chat_template + + package = installed_package(target) + if package.get("onnx") is not True: + raise ValueError("Nemotron-H DFlash requires an ONNX-enabled Edge SDK") + source = Path(request.model_dir).resolve() + draft = draft.resolve() + draft_config = json.loads((draft / "config.json").read_text(encoding="utf-8")) + dflash = draft_config.get("dflash_config", {}) + layer_ids = dflash.get("target_layer_ids", []) + if (draft_config.get("architectures") != ["DFlashDraftModel"] + or draft_config.get("hidden_size") != raw["hidden_size"] + or draft_config.get("vocab_size") != raw["vocab_size"] + or not layer_ids or len(set(layer_ids)) != len(layer_ids) + or any(type(i) is not int or not 0 <= i < raw["num_hidden_layers"] for i in layer_ids) + or dflash.get("block_size", 16) != 16): + raise ValueError("Nemotron-H DFlash companion geometry does not match its target") + if not list(source.glob("*.safetensors")) or not list(draft.glob("*.safetensors")): + raise ValueError("Nemotron-H DFlash requires both local safetensors checkpoints") + limit = sequence_length(request, raw) + if limit < 16: + raise ValueError("Nemotron-H DFlash requires capacity for a complete block16") + checkpoint = staging / "edge_llm/checkpoint" + checkpoint.mkdir(parents=True) + for name in ("config.json", "tokenizer.json", "tokenizer_config.json", "generation_config.json", + "chat_template.jinja"): + if (source / name).is_file(): + shutil.copy2(source / name, checkpoint / name) + engine = staging / "edge_llm/engine" + onnx = staging / "onnx" + # Upstream packed MoE layout is an exporter option, not a model rewrite. + env = subprocess_environment( + {"EDGELLM_PLUGIN_PATH": package["plugin"], + "EDGELLM_NVFP4_MOE_TARGET": "sm12x" if target["sm"] in {120, 121} else f"sm{target['sm']}"}, + prepend_paths={"LD_LIBRARY_PATH": str(Path(package["plugin"]).parent)}, + ) + for role, subdirectory, flag in (("draft", "dflash_draft", "--specDraft"), + ("base", "llm", "--specBase")): + commands = [ + [package["python"], "-I", "-m", "tensorrt_edgellm.scripts.export", + str(source), str(onnx), f"--dflash-{role}", "--dflash-draft-dir", str(draft), + "--skip-visual", "--skip-audio"], + [package["onnx_builder"], "--onnxDir", str(onnx / subdirectory), + "--engineDir", str(engine), flag, "--maxInputLen", str(min(limit, 1024)), + "--maxKVCacheCapacity", str(limit), "--maxBatchSize", "1", + "--maxVerifyTreeSize", "16", "--maxDraftTreeSize", "16"], + ] + with log_path.open("a", encoding="utf-8") as log: + for command in commands: + log.write(json.dumps(command) + "\n") + log.flush() + subprocess.run(command, check=True, stdout=log, stderr=subprocess.STDOUT, + cwd=staging, env=env) + # Only intermediates created by this preparation are removed. Source + # checkpoints and final engine assets remain untouched. + shutil.rmtree(onnx) + required = ("spec_base.engine", "spec_draft.engine", "base_config.json", "draft_config.json", + "embedding.safetensors", "tokenizer.json", "tokenizer_config.json", "processed_chat_template.json") + for name in required: + if not (engine / name).is_file() or (engine / name).stat().st_size == 0: + raise ValueError(f"Edge ONNX builder did not produce required artifact: {name}") + for role in ("base", "draft"): + config = json.loads((engine / f"{role}_config.json").read_text(encoding="utf-8")) + if config.get("spec_decode_type") != "dflash" or config.get("dflash_config", {}).get("block_size") != 16: + raise ValueError("Edge ONNX builder returned a different speculative execution contract") + tokenizer = staging / "edge_llm/runtime_tokenizer" + eos = prepare_tokenizer(checkpoint, engine, tokenizer, raw, + chat_template=source_chat_template(checkpoint)) + files = {} + for directory in (engine, checkpoint, tokenizer): + for path in sorted(directory.rglob("*")): + if path.is_symlink(): + raise ValueError(f"Edge output must not contain symlinks: {path}") + if path.is_file(): + files[path.relative_to(staging).as_posix()] = path + return files, { + "version": 1, "edge_revision": EDGE_REVISION, "target": target, "precision": "fp16", + "max_sequence_length": limit, "max_input_length": min(limit, 1024), "max_batch_size": 1, + "artifacts": list(files), "checkpoint_precision": source_quantization(request, raw), + "tokenizer_policy": "native_full_eos", "native_eos_token_ids": eos, + "native_bos_token_id": raw.get("bos_token_id", -1), "execution_variant": "dflash", + "builder_flow": "onnx", "dflash_block_size": 16, + } diff --git a/families/nemotron_h/edge_llm/cli.py b/families/nemotron_h/edge_llm/cli.py new file mode 100644 index 0000000000..ce6c3109f1 --- /dev/null +++ b/families/nemotron_h/edge_llm/cli.py @@ -0,0 +1,29 @@ +# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-License-Identifier: Apache-2.0 + +"""NemotronH-owned CLI options for explicit paired Edge execution.""" + +from pathlib import Path + +from .config import BuildExecutionInputs, NamedCheckpoint + + +def execution_inputs( + execution_variant: str | None, companion: list[str] | tuple[str, ...] = (), +) -> BuildExecutionInputs | None: + """Parse only explicit local inputs; no variant list or model acquisition.""" + if execution_variant is None: + if companion: + raise ValueError("--companion requires --execution-variant") + return None + if execution_variant not in ['dflash']: + raise ValueError("unsupported nemotron_h execution variant") + checkpoints = [] + for value in companion: + role, separator, directory = value.partition("=") + if not separator or not role or not directory: + raise ValueError("--companion must be ROLE=LOCAL_DIR") + if "://" in directory: + raise ValueError("--companion requires a local directory, not a URI") + checkpoints.append(NamedCheckpoint(role, Path(directory))) + return BuildExecutionInputs(execution_variant, tuple(checkpoints)) diff --git a/families/nemotron_h/edge_llm/config.py b/families/nemotron_h/edge_llm/config.py new file mode 100644 index 0000000000..f0b40b5fb8 --- /dev/null +++ b/families/nemotron_h/edge_llm/config.py @@ -0,0 +1,89 @@ +# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-License-Identifier: Apache-2.0 + +"""Explicit NemotronH paired-build inputs; no shared execution-mode contract.""" + +from dataclasses import dataclass, fields +from pathlib import Path +import re + +from ..build_request import BuildRequest, coerce_request + + +_ID = re.compile(r"[a-z][a-z0-9_]*\Z") + + +def _validate_id(field: str, value: object) -> str: + if not isinstance(value, str) or _ID.fullmatch(value) is None: + raise ValueError(f"{field} must be a lowercase identifier containing only letters, digits, and underscores") + return value + + +@dataclass(frozen=True) +class NamedCheckpoint: + """One explicitly named local checkpoint; the family owns role semantics.""" + + role: str + model_dir: Path + + def __post_init__(self) -> None: + _validate_id("checkpoint role", self.role) + if not isinstance(self.model_dir, Path): + raise TypeError("checkpoint model_dir must be a Path") + if not self.model_dir.is_dir(): + raise ValueError(f"checkpoint must be an existing local directory: {self.model_dir}") + + +@dataclass(frozen=True) +class BuildExecutionInputs: + """Optional family-owned execution variant and immutable local companions. + + NemotronH validates these inputs without fetching or inferring companions. + """ + + variant: str + checkpoints: tuple[NamedCheckpoint, ...] = () + + def __post_init__(self) -> None: + _validate_id("execution variant", self.variant) + if not isinstance(self.checkpoints, tuple) or any( + not isinstance(checkpoint, NamedCheckpoint) for checkpoint in self.checkpoints + ): + raise TypeError("checkpoints must be a tuple of NamedCheckpoint values") + roles = [checkpoint.role for checkpoint in self.checkpoints] + if len(roles) != len(set(roles)): + raise ValueError("checkpoint roles must be unique") + self.validate_local() + + def validate_local(self) -> None: + """Recheck local availability before dispatch, without acquiring inputs.""" + for checkpoint in self.checkpoints: + if not checkpoint.model_dir.is_dir(): + raise ValueError( + f"checkpoint must be an existing local directory: {checkpoint.model_dir}" + ) + + +@dataclass(frozen=True) +class NemotronHBuildRequest(BuildRequest): + """Ordinary build inputs plus an explicitly requested NemotronH execution recipe.""" + + execution: BuildExecutionInputs | None = None + + def __post_init__(self) -> None: + super().__post_init__() + if self.family != "nemotron_h": + raise ValueError("NemotronHBuildRequest requires the nemotron_h family") + if self.execution is not None: + if not isinstance(self.execution, BuildExecutionInputs): + raise TypeError("execution must be BuildExecutionInputs") + self.execution.validate_local() + + +def with_execution(request: BuildRequest, execution: BuildExecutionInputs) -> NemotronHBuildRequest: + """Preserve supported ordinary request fields and callback identity.""" + request = coerce_request(request) + return NemotronHBuildRequest( + **{field.name: getattr(request, field.name) for field in fields(BuildRequest)}, + execution=execution, + ) diff --git a/families/nemotron_h/edge_llm/dispatch.py b/families/nemotron_h/edge_llm/dispatch.py new file mode 100644 index 0000000000..1a8ed32377 --- /dev/null +++ b/families/nemotron_h/edge_llm/dispatch.py @@ -0,0 +1,155 @@ +# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-License-Identifier: Apache-2.0 + +"""Family-owned Nemotron-H complete-network map and native preparation retry.""" + +from __future__ import annotations + +import json +import logging +import os +from pathlib import Path +import tempfile +import traceback + +from . import builder as edge_llm +from .edge_quantization import source_quantization +from .edge_config import topology_matches + +_LOG = logging.getLogger(__name__) +# Native platforms with recorded passing family profiles; admission is not qualification. +PLATFORMS = {"x86_64": (80, 120)} +EDGE_DISPATCH = { + ("linux", arch, sm, precision): edge_llm.prepare + for arch, sms in PLATFORMS.items() + for sm in sms + for precision in ("fp16", "fp8", "nvfp4") + if precision == "fp16" or sm >= 100 +} + + +def request_matches(request) -> bool: + """Preserve native early rejection order for unmapped request controls.""" + return ( + request.family == "nemotron_h" and request.backend == "trt" and request.task == "text_generation" + and request.precision.lower() == "fp16" and request.quantization in {None, "none", "fp8", "nvfp4"} + and request.max_batch_size == request.tensor_parallel_size == request.context_parallel_size == 1 + and not request.dynamic_kv_cache and not request.fp32_layers and request.graph_transform is None + and all(value is None for value in (request.image_height, request.image_width, request.video_num_frames)) + ) + + +def candidate(request, raw: dict) -> bool: + """Preserve native tokenizer contract and require exact source hybrid policy.""" + if not request_matches(request): + return False + precision = source_quantization(request, raw) + if precision is None or not topology_matches(raw, precision): + return False + try: + from .edge_tokenizer import source_chat_template + source_chat_template(Path(request.model_dir)) + return True + except (OSError, ValueError): + return False + + +def platform_matches(raw: dict, target: dict) -> bool: + """Admit the native runtime, not just its optional optimized SSD kernel.""" + # MambaPlugin falls back to scalar prefill when SSD cannot implement a shape. + # Topology, source precision and the installed package are checked separately. + return target["sm"] >= 80 + + +def build(request, writer, native) -> None: + """Prepare Edge privately or warn and retry unchanged native once. + + Publication and cancellation errors propagate. Runtime fallback is not part + of this builder adapter; an Edge bundle is owned entirely by its runtime. + """ + if not request_matches(request): + native(request, writer) + return + raw = json.loads((Path(request.model_dir) / "config.json").read_text(encoding="utf-8")) + if not isinstance(raw, dict): + raise ValueError("checkpoint config.json must contain an object") + if not candidate(request, raw): + native(request, writer) + return + if not edge_llm.package_present(): + native(request, writer) + return + if edge_llm.sequence_length(request, raw) > raw["max_position_embeddings"]: + raise ValueError("Nemotron-H max_sequence_length exceeds checkpoint context capacity") + failure = None + descriptor, name = tempfile.mkstemp(prefix=f".{request.output_path.name}.edge-", suffix=".log", + dir=request.output_path.parent) + os.close(descriptor) + log_path = Path(name) + with tempfile.TemporaryDirectory( + prefix=f".{request.output_path.name}.edge-", dir=request.output_path.parent + ) as directory: + try: + target = edge_llm.local_target() + key = (target["os"], target["arch"], target["sm"], source_quantization(request, raw)) + adapter = EDGE_DISPATCH.get(key) if platform_matches(raw, target) else None + if adapter is not None: + files, marker = adapter(request, raw, target, Path(directory), log_path) + except Exception as error: + failure = error + with log_path.open("a", encoding="utf-8") as log: + traceback.print_exception(error, file=log) + _LOG.warning("nemotron_h Edge build failed: %s. Diagnostics: %s. " + "Retrying native once with the unchanged request.", error, log_path, exc_info=True) + except BaseException: + log_path.unlink(missing_ok=True) + raise + else: + if adapter is not None: + edge_llm.publish(request, writer, files, marker) + log_path.unlink() + return + log_path.unlink() + try: + native(request, writer) + except Exception as error: + if failure is not None: + raise error from failure + raise + + +def build_dflash(request, writer, draft: Path) -> None: + """Preserve a paired request on failure; never replace it with a base-only build.""" + raw = json.loads((Path(request.model_dir) / "config.json").read_text(encoding="utf-8")) + if not candidate(request, raw) or source_quantization(request, raw) != "nvfp4": + raise ValueError("Nemotron-H DFlash ONNX requires a compatible NVFP4 target and TP1 text request") + if edge_llm.sequence_length(request, raw) > raw["max_position_embeddings"]: + raise ValueError("Nemotron-H max_sequence_length exceeds checkpoint context capacity") + descriptor, name = tempfile.mkstemp(prefix=f".{request.output_path.name}.edge-onnx-", suffix=".log", + dir=request.output_path.parent) + os.close(descriptor) + log_path = Path(name) + with tempfile.TemporaryDirectory(prefix="trtmc-nemotron-h-dflash-", dir=request.output_path.parent) as directory: + try: + target = edge_llm.local_target() + key = (target["os"], target["arch"], target["sm"], "nvfp4") + if key not in EDGE_DISPATCH or not platform_matches(raw, target): + raise ValueError("Nemotron-H DFlash ONNX is unavailable on this native platform") + files, marker = edge_llm.prepare_dflash(request, raw, target, Path(directory), log_path, draft) + except Exception as error: + with log_path.open("a", encoding="utf-8") as log: + traceback.print_exception(error, file=log) + _LOG.warning("Nemotron-H DFlash Edge ONNX build failed: %s. Diagnostics: %s. " + "Native paired execution is unavailable; refusing base-only fallback.", error, log_path) + raise NotImplementedError("Native Nemotron-H DFlash fallback is unavailable") from error + edge_llm.publish(request, writer, files, marker) + log_path.unlink() + + +def build_paired(request, writer, execution) -> None: + """Keep the complete DFlash pair owned by the family ONNX adapter.""" + execution.validate_local() + if execution.variant != "dflash" or tuple(x.role for x in execution.checkpoints) != ("draft",): + raise ValueError("Nemotron-H paired execution requires variant=dflash and one draft") + + build_dflash(request, writer, execution.checkpoints[0].model_dir) diff --git a/families/nemotron_h/edge_llm/edge_config.py b/families/nemotron_h/edge_llm/edge_config.py new file mode 100644 index 0000000000..2fbebca8db --- /dev/null +++ b/families/nemotron_h/edge_llm/edge_config.py @@ -0,0 +1,79 @@ +# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-License-Identifier: Apache-2.0 +"""Native hybrid source topology gates; Edge still constructs every graph.""" + +from __future__ import annotations + +import math + +_PATTERN = {"M": "mamba", "-": "mlp", "*": "attention", "E": "moe"} + + +def layers(raw: dict) -> list[str] | None: + """Require complete consistent explicit hybrid topology, never filter unknowns.""" + forms = [] + for name in ("layers_block_type", "layer_types"): + if name in raw: + value = raw[name] + if not isinstance(value, list) or any(x not in _PATTERN.values() for x in value): + return None + forms.append(value) + if raw.get("hybrid_override_pattern"): + pattern = raw["hybrid_override_pattern"] + if not isinstance(pattern, str) or any(x not in _PATTERN for x in pattern): + return None + forms.append([_PATTERN[x] for x in pattern]) + if not forms or not forms[0] or any(x != forms[0] for x in forms): + return None + return forms[0] if len(forms[0]) == raw.get("num_hidden_layers") else None + + +def topology_matches(raw: dict, precision: str) -> bool: + """Admit only pinned dense/SSD or non-gated NVFP4 MoE semantics.""" + kinds = layers(raw) + if (raw.get("model_type") != "nemotron_h" + or raw.get("architectures") != ["NemotronHForCausalLM"] or kinds is None + or "mamba" not in kinds + or any(key in raw for key in ("text_config", "vision_config", "audio_config", "thinker_config", + "eagle_config", "dflash_config", "jetspec_config", "dspark_config")) + or raw.get("mamba_hidden_act", "silu") != "silu" + or raw.get("mlp_hidden_act", "relu2") != "relu2" + or raw.get("hidden_act", "relu2") != "relu2" + or raw.get("mamba_ssm_cache_dtype", "float32") != "float32" + or any(raw.get(key, False) for key in ("attention_bias", "mamba_proj_bias", "mlp_bias")) + or raw.get("rope_scaling") is not None or raw.get("hybrid_uses_rope", False) + or raw.get("norm_before_gate", False) or raw.get("conv_kernel", 4) != 4): + return False + if type(raw.get("bos_token_id", -1)) is not int or raw.get("bos_token_id", -1) < -1: + return False + keys = ("hidden_size", "num_hidden_layers", "num_attention_heads", "num_key_value_heads", + "head_dim", "mamba_num_heads", "mamba_head_dim", "n_groups", "ssm_state_size", + "max_position_embeddings", "vocab_size") + values = [raw.get(key) for key in keys] + if any(type(value) is not int or value <= 0 for value in values): + return False + hidden, _, heads, kv, head, mheads, mhead, groups, state, _, _ = values + if (heads % kv or head % 8 or mheads % groups or state not in {64, 128} + or mhead not in {64, 80, 128} or mhead == 80 and state != 128 + or raw.get("conv_dim", mheads * mhead + 2 * groups * state) + != mheads * mhead + 2 * groups * state): + return False + if "mlp" in kinds and (type(raw.get("intermediate_size")) is not int or raw["intermediate_size"] <= 0): + return False + if "moe" not in kinds: + return not raw.get("n_routed_experts", 0) + names = ("n_routed_experts", "num_experts_per_tok", "n_group", "topk_group", + "moe_intermediate_size", "moe_shared_expert_intermediate_size") + if any(type(raw.get(key)) is not int or raw[key] <= 0 for key in names): + return False + experts, topk, router_groups, selected_groups, intermediate, shared = [raw[key] for key in names] + scale = raw.get("routed_scaling_factor", 1.0) + latent = raw.get("moe_latent_size", hidden) + latent = hidden if latent is None else latent + return (precision == "nvfp4" and experts <= 512 and topk <= experts + and experts % router_groups == 0 and selected_groups <= router_groups + and topk <= selected_groups * (experts // router_groups) + and raw.get("n_shared_experts", 1) == 1 and intermediate % 16 == shared % 16 == 0 + and type(latent) is int and latent > 0 and latent % 16 == 0 + and isinstance(scale, (int, float)) and not isinstance(scale, bool) + and math.isfinite(scale) and scale > 0) diff --git a/families/nemotron_h/edge_llm/edge_quantization.py b/families/nemotron_h/edge_llm/edge_quantization.py new file mode 100644 index 0000000000..dde984468e --- /dev/null +++ b/families/nemotron_h/edge_llm/edge_quantization.py @@ -0,0 +1,106 @@ +# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-License-Identifier: Apache-2.0 +"""Exact family-owned source policy admission; never convert or calibrate weights.""" + +from __future__ import annotations + +import json +from pathlib import Path + + +def descriptor(value: dict) -> tuple | None: + """Normalize known ModelOpt policy fields, rejecting unknown packed layouts.""" + if not isinstance(value, dict): + return None + q = value.get("quantization", value) + if not isinstance(q, dict): + return None + if q.get("quant_method", "modelopt").lower() not in {"modelopt", ""}: + return None + algorithm = str(q.get("quant_algo", "")).upper() + if algorithm == "MIXED_PRECISION": + layers = q.get("quantized_layers") + if not isinstance(layers, dict) or not layers: + return None + algorithms = set() + for name, policy in layers.items(): + if not isinstance(name, str) or not name or not isinstance(policy, dict): + return None + algo = policy.get("quant_algo") + if algo not in {"FP8", "NVFP4", "W4A16_NVFP4"}: + return None + group = policy.get("group_size", 1 if algo == "FP8" else 16) + if type(group) is not int or group not in ({1, 16} if algo == "FP8" else {16}): + return None + if set(policy) - {"quant_algo", "group_size"}: + return None + algorithms.add(algo) + if algorithms not in ({"FP8", "NVFP4"}, {"FP8", "W4A16_NVFP4"}): + return None + kv = q.get("kv_cache_quant_algo") + scheme = q.get("kv_cache_scheme") + if scheme is not None: + if scheme != {"dynamic": False, "num_bits": 8, "type": "float"}: + return None + if kv not in {None, "FP8"}: + return None + kv = "FP8" + if kv != "FP8": + return None + # The pinned direct builder consumes the sidecar per-layer map. Keep + # its exact algorithms/scales; do not reinterpret ModelOpt config_groups. + return "nvfp4", kv, json.dumps(layers, sort_keys=True), "mixed" + precision = {"FP8": "fp8", "NVFP4": "nvfp4", "W4A16_NVFP4": "nvfp4"}.get(algorithm) + if precision is None: + return None + group = q.get("group_size", 1 if precision == "fp8" else 16) + # FP8 projections use scalar scaling, not grouped packing. ModelOpt emits + # either 1 (parser default) or 16 (export metadata); neither changes FP8 layout. + if type(group) is not int or group not in ({1, 16} if precision == "fp8" else {16}): + return None + if any(key in q for key in ("config_groups", "quantized_layers", "kv_cache_scheme")): + return None + if any(key not in {"quant_algo", "group_size", "kv_cache_quant_algo", "exclude_modules", + "ignore", "quant_method"} for key in q): + return None + excluded = q.get("exclude_modules", q.get("ignore", [])) + if (not isinstance(excluded, list) or any(not isinstance(x, str) for x in excluded) + or ("ignore" in q and "exclude_modules" in q and set(q["ignore"]) != set(excluded))): + return None + kv = q.get("kv_cache_quant_algo") + if kv not in {None, "FP8"}: + return None + return precision, kv, tuple(sorted(set(excluded))) + + +def source_quantization(request, raw: dict) -> str | None: + """Preserve consistent original/FP8/NVFP4 source, or return native admission.""" + try: + policies = [] + embedded = raw.get("quantization_config") + if embedded is not None: + policies.append(descriptor(embedded)) + for name in ("hf_quant_config.json", "quantize_config.json", "quant_config.json"): + path = Path(request.model_dir) / name + if path.exists(): + value = json.loads(path.read_text(encoding="utf-8")) + policies.append(descriptor(value)) + if policies: + if any(value is None or value != policies[0] for value in policies): + return None + # Upstream only consumes hf_quant_config or embedded metadata. + if embedded is None and not (Path(request.model_dir) / "hf_quant_config.json").is_file(): + return None + if policies[0][-1] == "mixed" and ( + embedded is None or not (Path(request.model_dir) / "hf_quant_config.json").is_file() + ): + return None + precision = policies[0][0] + else: + precision = "fp16" + requested = request.quantization + if requested is not None and requested != ("none" if precision == "fp16" else precision): + return None + return precision + except (OSError, ValueError, TypeError, AttributeError): + return None diff --git a/families/nemotron_h/edge_llm/edge_tokenizer.py b/families/nemotron_h/edge_llm/edge_tokenizer.py new file mode 100644 index 0000000000..5e773f0ce8 --- /dev/null +++ b/families/nemotron_h/edge_llm/edge_tokenizer.py @@ -0,0 +1,95 @@ +# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-License-Identifier: Apache-2.0 + +"""Serialize native full-EOS metadata without mutating original tokenizer assets.""" + +from __future__ import annotations + +import json +from pathlib import Path +import shutil + + +def _object(path: Path) -> dict: + """Read a JSON object or raise a preparation error with the source path.""" + value = json.loads(path.read_text(encoding="utf-8")) + if not isinstance(value, dict): + raise ValueError(f"Tokenizer metadata must contain an object: {path.name}") + return value + + +def source_chat_template(checkpoint: Path) -> str: + """Resolve the original standalone template with HF file precedence.""" + metadata = _object(checkpoint / "tokenizer_config.json") + standalone = checkpoint / "chat_template.jinja" + template = standalone.read_text(encoding="utf-8") if standalone.exists() else metadata.get("chat_template") + if not isinstance(template, str) or not template.strip(): + raise ValueError("Nemotron-H requires an embedded or standalone chat template") + return template + + +def native_eos_ids(checkpoint: Path, raw: dict) -> list[int]: + """Resolve native generation-config precedence and the complete stop vector.""" + generation = checkpoint / "generation_config.json" + config = _object(generation) if generation.exists() else {} + eos = config.get("eos_token_id", raw.get("eos_token_id")) + ids = eos if isinstance(eos, list) else [eos] + if (not ids or any(type(value) is not int or not 0 <= value < raw["vocab_size"] for value in ids) + or len(set(ids)) != len(ids)): + raise ValueError("Nemotron-H native EOS must contain unique vocabulary IDs") + return ids + + +def eos_token(tokenizer: dict, eos: int) -> str: + """Resolve one unambiguous original token string for an ID, without constants.""" + by_id: dict[int, set[str]] = {} + by_token: dict[str, set[int]] = {} + vocab = tokenizer.get("model", {}).get("vocab", {}) + if not isinstance(vocab, dict): + raise ValueError("Nemotron-H Edge requires a BPE vocabulary object") + entries = [(token, index) for token, index in vocab.items()] + for entry in tokenizer.get("added_tokens", []): + if not isinstance(entry, dict): + raise ValueError("Invalid Nemotron-H added token entry") + entries.append((entry.get("content"), entry.get("id"))) + for token, index in entries: + if not isinstance(token, str) or type(index) is not int or index < 0: + raise ValueError("Invalid Nemotron-H tokenizer token/ID mapping") + by_id.setdefault(index, set()).add(token) + by_token.setdefault(token, set()).add(index) + tokens = by_id.get(eos, set()) + if len(tokens) != 1: + raise ValueError("Native first EOS has missing or ambiguous tokenizer ID mapping") + token = next(iter(tokens)) + if not token or by_token[token] != {eos}: + raise ValueError("Native first EOS token maps to multiple IDs") + return token + + +def prepare_tokenizer(checkpoint: Path, engine: Path, destination: Path, raw: dict, + *, chat_template: str | None = None) -> list[int]: + """Create explicit derivative metadata; original source/engine files stay intact. + + Returns the full native EOS vector. Invalid source metadata is a + preparation failure, before bundle publication or runtime execution. + """ + tokenizer = _object(checkpoint / "tokenizer.json") + metadata = _object(checkpoint / "tokenizer_config.json") + if chat_template is not None: + if not isinstance(chat_template, str) or not chat_template.strip(): + raise ValueError("Explicit Edge chat template must be non-empty") + metadata["chat_template"] = chat_template + if not isinstance(metadata.get("chat_template"), str) or not metadata["chat_template"]: + raise ValueError("Nemotron-H native requires embedded tokenizer chat_template") + eos = native_eos_ids(checkpoint, raw) + for token_id in eos: + eos_token(tokenizer, token_id) + metadata["eos_token"] = eos_token(tokenizer, eos[0]) + destination.mkdir(parents=True) + # Runtime token IDs come from the byte-exact retained source vocabulary. + shutil.copy2(checkpoint / "tokenizer.json", destination / "tokenizer.json") + shutil.copy2(engine / "processed_chat_template.json", destination / "processed_chat_template.json") + (destination / "tokenizer_config.json").write_text( + json.dumps(metadata, ensure_ascii=False, indent=2) + "\n", encoding="utf-8" + ) + return eos diff --git a/families/nemotron_h/model.py b/families/nemotron_h/model.py index 7a3a72b02a..6340a5a7a7 100644 --- a/families/nemotron_h/model.py +++ b/families/nemotron_h/model.py @@ -73,7 +73,7 @@ def _parse_layer_types(pattern: str) -> list[str]: if TYPE_CHECKING: - from tensorrt_model_connect.build import BuildRequest + from .build_request import BuildRequest from tensorrt_model_connect.bundle_writer import BundleWriter @@ -993,55 +993,92 @@ def _runtime_config( def build(request: "BuildRequest", writer: "BundleWriter") -> None: - """Build one Nemotron-H bundle.""" - if request.dynamic_kv_cache: - raise NotImplementedError("nemotron_h does not support dynamic_kv_cache") - - if request.image_height is not None: - raise NotImplementedError("nemotron_h does not support image_height") - - if request.image_width is not None: - raise NotImplementedError("nemotron_h does not support image_width") - - if request.video_num_frames is not None: - raise NotImplementedError("nemotron_h does not support video_num_frames") - - if request.max_batch_size != 1: - raise NotImplementedError("nemotron_h does not support max_batch_size") - - if request.context_parallel_size != 1: - raise ValueError("this family does not support context parallelism") - - if request.task != "text_generation": - raise ValueError("nemotron_h supports only task=text_generation") - - model_dir = Path(request.model_dir) - config = ModelConfig.from_dir(model_dir) - if str(config.model_type).lower() not in {"nemotron_h", "nemotron_hybrid"}: - raise ValueError(f"Nemotron-H does not support model_type={config.model_type!r}") - precision = str(request.precision).lower() - if precision not in {"fp32", "fp16", "bf16"}: - raise ValueError("Nemotron-H precision must be fp32, fp16, or bf16") - max_sequence_length = _positive_int( - request.max_sequence_length or min(config.max_position_embeddings, 256), - "max_sequence_length", - ) - if max_sequence_length > config.max_position_embeddings: - raise ValueError("Nemotron-H max_sequence_length exceeds checkpoint context capacity") - if request.quantization not in {None, "none"}: - raise NotImplementedError("Nemotron-H has no qualified family-owned quantized build") - if request.fp32_layers: - raise NotImplementedError("Nemotron-H does not expose mixed-precision layers") - parallel = ParallelConfig( - tp_size=_positive_int(request.tensor_parallel_size, "tensor_parallel_size") - ) - parallel.validate() - model = _NemotronHModel() - config.raw["_model_dir"] = str(model_dir) - weights = model.load_weights(str(model_dir), config) - writer.set_header(family="nemotron_h", task=request.task, backend=request.backend) - if parallel.enabled: - for rank in range(parallel.tp_size): + """Dispatch complete offload or execute the unchanged native implementation.""" + from .build_request import coerce_request + + request = coerce_request(request) + + from .edge_llm.config import NemotronHBuildRequest + from .edge_llm.dispatch import build_paired + + if isinstance(request, NemotronHBuildRequest) and request.execution is not None: + build_paired(request, writer, request.execution) + return + from .edge_llm.dispatch import build as dispatch_build + + def _build_native(request: "BuildRequest", writer: "BundleWriter") -> None: + """Build one Nemotron-H bundle.""" + if request.dynamic_kv_cache: + raise NotImplementedError("nemotron_h does not support dynamic_kv_cache") + + if request.image_height is not None: + raise NotImplementedError("nemotron_h does not support image_height") + + if request.image_width is not None: + raise NotImplementedError("nemotron_h does not support image_width") + + if request.video_num_frames is not None: + raise NotImplementedError("nemotron_h does not support video_num_frames") + + if request.max_batch_size != 1: + raise NotImplementedError("nemotron_h does not support max_batch_size") + + if request.context_parallel_size != 1: + raise ValueError("this family does not support context parallelism") + + if request.task != "text_generation": + raise ValueError("nemotron_h supports only task=text_generation") + + model_dir = Path(request.model_dir) + config = ModelConfig.from_dir(model_dir) + if str(config.model_type).lower() not in {"nemotron_h", "nemotron_hybrid"}: + raise ValueError(f"Nemotron-H does not support model_type={config.model_type!r}") + precision = str(request.precision).lower() + if precision not in {"fp32", "fp16", "bf16"}: + raise ValueError("Nemotron-H precision must be fp32, fp16, or bf16") + max_sequence_length = _positive_int( + request.max_sequence_length or min(config.max_position_embeddings, 256), + "max_sequence_length", + ) + if max_sequence_length > config.max_position_embeddings: + raise ValueError("Nemotron-H max_sequence_length exceeds checkpoint context capacity") + if request.quantization not in {None, "none"}: + raise NotImplementedError("Nemotron-H has no qualified family-owned quantized build") + # Native weight loading does not apply packed-source quantization scales. + # Auto/none controls cannot reinterpret an already quantized checkpoint. + if config.raw.get("quantization_config") is not None or any( + (model_dir / name).exists() + for name in ("hf_quant_config.json", "quantize_config.json", "quant_config.json") + ): + raise NotImplementedError( + "Nemotron-H native fallback cannot load a quantized checkpoint; " + "use the Edge builder on a supported platform" + ) + if request.fp32_layers: + raise NotImplementedError("Nemotron-H does not expose mixed-precision layers") + parallel = ParallelConfig( + tp_size=_positive_int(request.tensor_parallel_size, "tensor_parallel_size") + ) + parallel.validate() + model = _NemotronHModel() + config.raw["_model_dir"] = str(model_dir) + weights = model.load_weights(str(model_dir), config) + writer.set_header(family="nemotron_h", task=request.task, backend=request.backend) + if parallel.enabled: + for rank in range(parallel.tp_size): + plan = model.build_engine( + config, + weights, + max_sequence_length, + precision=precision, + quant_ctx=None, + verbose=bool(request.verbose), + debug_layer_outputs=False, + parallel_config=parallel.for_rank(rank), + ) + writer.add_bytes(f"engine.rank{rank}.plan", plan) + layout = "dual_profile" + else: plan = model.build_engine( config, weights, @@ -1050,37 +1087,26 @@ def build(request: "BuildRequest", writer: "BundleWriter") -> None: quant_ctx=None, verbose=bool(request.verbose), debug_layer_outputs=False, - parallel_config=parallel.for_rank(rank), + parallel_config=parallel, ) - writer.add_bytes(f"engine.rank{rank}.plan", plan) - layout = "dual_profile" - else: - plan = model.build_engine( - config, - weights, - max_sequence_length, - precision=precision, - quant_ctx=None, - verbose=bool(request.verbose), - debug_layer_outputs=False, - parallel_config=parallel, + writer.add_bytes("engine.plan", plan) + layout = "single" + writer.add_json( + "runtime.json", + _runtime_config( + model_dir, + config, + model, + precision=precision, + max_cache_length=max_sequence_length, + decoder_engine_layout=layout, + tensor_parallel_size=parallel.tp_size, + tensor_parallel_mode="tensor_parallel" if parallel.enabled else "single", + ), ) - writer.add_bytes("engine.plan", plan) - layout = "single" - writer.add_json( - "runtime.json", - _runtime_config( - model_dir, - config, - model, - precision=precision, - max_cache_length=max_sequence_length, - decoder_engine_layout=layout, - tensor_parallel_size=parallel.tp_size, - tensor_parallel_mode="tensor_parallel" if parallel.enabled else "single", - ), - ) - for filename in _BUNDLE_FILES: - path = model_dir / filename - if path.is_file(): - writer.add_bytes(filename, path.read_bytes()) + for filename in _BUNDLE_FILES: + path = model_dir / filename + if path.is_file(): + writer.add_bytes(filename, path.read_bytes()) + + dispatch_build(request, writer, _build_native) diff --git a/families/nemotron_h/runtime/CMakeLists.txt b/families/nemotron_h/runtime/CMakeLists.txt index d6d06e7f2b..e665f39099 100644 --- a/families/nemotron_h/runtime/CMakeLists.txt +++ b/families/nemotron_h/runtime/CMakeLists.txt @@ -57,3 +57,5 @@ if(TRTMC_BUILD_TESTS) ) add_test(NAME ${test_name} COMMAND ${test_name}) endif() + +include(edge_llm/Adapter.cmake) diff --git a/families/nemotron_h/runtime/chat_templates.cpp b/families/nemotron_h/runtime/chat_templates.cpp index bddc592966..3208745f4b 100644 --- a/families/nemotron_h/runtime/chat_templates.cpp +++ b/families/nemotron_h/runtime/chat_templates.cpp @@ -10,10 +10,16 @@ namespace trtmc { namespace { -std::string apply_chatml(const std::string& prompt, bool enable_thinking) { +std::string apply_chatml(const std::string& prompt, bool enable_thinking, + bool nemotron_format = false) { std::string r = "<|im_start|>user\n" + prompt + "<|im_end|>\n<|im_start|>assistant\n"; - if (!enable_thinking) + if (nemotron_format) { + // Modern Nemotron ChatML includes an empty system and a compact think suffix. + r = "<|im_start|>system\n<|im_end|>\n" + r; + r += enable_thinking ? "\n" : ""; + } else if (!enable_thinking) { r += "\n\n\n\n"; + } return r; } @@ -29,8 +35,12 @@ std::string apply_nemotron_h(const std::string& prompt, bool enable_thinking) { std::string nemotron_h_detect_chat_template_format(const std::string& jinja_template) { if (jinja_template.empty()) return {}; - if (jinja_template.find("<|im_start|>") != std::string::npos) + if (jinja_template.find("<|im_start|>") != std::string::npos) { + if (jinja_template.find("<|im_start|>system") != std::string::npos && + jinja_template.find("") != std::string::npos) + return "nemotron_chatml"; return "chatml"; + } if (jinja_template.find("") != std::string::npos) return "nemotron_h"; return {}; @@ -40,6 +50,8 @@ std::string nemotron_h_apply_chat_template(const std::string& format, const std: bool enable_thinking) { if (format.empty()) return prompt; + if (format == "nemotron_chatml") + return apply_chatml(prompt, enable_thinking, true); if (format == "chatml") return apply_chatml(prompt, enable_thinking); if (format == "nemotron_h") diff --git a/families/nemotron_h/runtime/edge_llm/Adapter.cmake b/families/nemotron_h/runtime/edge_llm/Adapter.cmake new file mode 100644 index 0000000000..fc5ab575c6 --- /dev/null +++ b/families/nemotron_h/runtime/edge_llm/Adapter.cmake @@ -0,0 +1,18 @@ +# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-License-Identifier: Apache-2.0 + +if(TARGET EdgeLLM::Core) + if(NOT TARGET EdgeLLM::Plugin) + message(FATAL_ERROR "NemotronH Edge adapter requires the complete EdgeLLM package (Core and Plugin)") + endif() + target_sources(trtmc_model_nemotron_h PRIVATE "${CMAKE_CURRENT_LIST_DIR}/adapter.cpp" "${CMAKE_CURRENT_LIST_DIR}/device_link.cu") + target_compile_definitions(trtmc_model_nemotron_h PRIVATE TRTMC_HAS_EDGE_LLM=1) + target_link_libraries(trtmc_model_nemotron_h PRIVATE EdgeLLM::Core nlohmann_json::nlohmann_json) + set_target_properties(trtmc_model_nemotron_h PROPERTIES + CUDA_ARCHITECTURES "${EdgeLLM_CUDA_ARCHITECTURE}" + CUDA_SEPARABLE_COMPILATION ON CUDA_RESOLVE_DEVICE_SYMBOLS ON) + add_custom_command(TARGET trtmc_model_nemotron_h POST_BUILD + COMMAND ${CMAKE_COMMAND} -E copy_if_different + $ $ + VERBATIM) +endif() diff --git a/families/nemotron_h/runtime/edge_llm/adapter.cpp b/families/nemotron_h/runtime/edge_llm/adapter.cpp new file mode 100644 index 0000000000..dec7abbc8f --- /dev/null +++ b/families/nemotron_h/runtime/edge_llm/adapter.cpp @@ -0,0 +1,323 @@ +/* + * SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. + * SPDX-License-Identifier: Apache-2.0 + */ +#include "families/nemotron_h/runtime/edge_llm/adapter.h" + +#include "families/nemotron_h/runtime/edge_llm/request.h" +#include "families/nemotron_h/runtime/edge_llm/tokenizer.h" + +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include + +namespace trtmc::nemotron_h::edge_llm { +namespace { +namespace fs = std::filesystem; + +/// Turn CUDA failures into caller-visible load or inference errors. +void check_cuda(cudaError_t result) { + if (result != cudaSuccess) + throw std::runtime_error(std::string("Nemotron-H Edge CUDA error: ") + + cudaGetErrorString(result)); +} + +/// Reject an engine built for a different local GPU or CUDA/TensorRT runtime. +void validate_target(const nlohmann::json& target) { + utsname host{}; + if (uname(&host) != 0) + throw std::runtime_error("Cannot identify Nemotron-H Edge runtime host"); + std::ifstream release("/etc/os-release"); + std::string line, os_version; + while (std::getline(release, line)) { + if (line.rfind("VERSION_ID=", 0) == 0) { + os_version = line.substr(11); + if (os_version.size() >= 2 && os_version.front() == char(34) && + os_version.back() == char(34)) + os_version = os_version.substr(1, os_version.size() - 2); + } + } + int device = 0, cuda_version = 0; + check_cuda(cudaGetDevice(&device)); + check_cuda(cudaRuntimeGetVersion(&cuda_version)); + cudaDeviceProp gpu{}; + check_cuda(cudaGetDeviceProperties(&gpu, device)); + const int trt_version = getInferLibVersion(); + const std::string trt = + std::to_string(trt_version / 10000) + "." + std::to_string((trt_version % 10000) / 100) + + "." + std::to_string(trt_version % 100) + "." + std::to_string(getInferLibBuildVersion()); + const std::string cuda = + std::to_string(cuda_version / 1000) + "." + std::to_string((cuda_version % 1000) / 10); + if (target.at("os") != "linux" || target.at("os_version") != os_version || + target.at("arch") != host.machine || target.at("sm") != gpu.major * 10 + gpu.minor || + target.at("cuda_version") != cuda || target.at("tensorrt_version") != trt) + throw std::runtime_error( + "Nemotron-H Edge bundle requires its build GPU and CUDA/TensorRT stack"); +} + +/// Own extracted engine/checkpoint files until after the Edge runtime is destroyed. +class Artifacts { + public: + explicit Artifacts(const BundleReader& bundle, const nlohmann::json& marker) { + std::set names; + for (const auto& entry : marker.at("artifacts")) { + const auto name = entry.get(); + if (!safe_artifact_path(name) || !names.insert(name).second || + !bundle.find_section(name)) + throw std::runtime_error("Invalid Nemotron-H Edge artifact: " + name); + } + const bool paired = marker.value("execution_variant", "autoregressive") == "dflash"; + const std::vector plans = + paired ? std::vector{"spec_base.engine", "spec_draft.engine", + "base_config.json", "draft_config.json", + "embedding.safetensors"} + : std::vector{"llm.engine", "config.json"}; + for (const auto& plan : plans) { + const auto required = "edge_llm/engine/" + plan; + if (!names.count(required) || bundle.find_section(required)->length == 0) + throw std::runtime_error("Required Nemotron-H Edge artifact missing: " + required); + } + for (const auto* required : + {"edge_llm/engine/tokenizer.json", "edge_llm/engine/tokenizer_config.json", + "edge_llm/engine/processed_chat_template.json", "edge_llm/checkpoint/config.json", + "edge_llm/runtime_tokenizer/tokenizer.json", + "edge_llm/runtime_tokenizer/tokenizer_config.json", + "edge_llm/runtime_tokenizer/processed_chat_template.json"}) + if (!names.count(required) || bundle.find_section(required)->length == 0) + throw std::runtime_error( + std::string("Required Nemotron-H Edge artifact missing: ") + required); + std::string pattern = (fs::temp_directory_path() / "trtmc-nemotron-h-edge-XXXXXX").string(); + if (!mkdtemp(pattern.data())) + throw std::runtime_error("Cannot create Nemotron-H Edge artifact directory"); + root_ = pattern; + try { + for (const auto& name : names) { + const auto destination = root_ / name; + fs::create_directories(destination.parent_path()); + std::ofstream output(destination, std::ios::binary); + bundle.copy_section(name, output); + output.close(); + if (!output) + throw std::runtime_error("Cannot extract Nemotron-H Edge artifact: " + name); + } + } catch (...) { + cleanup(); + throw; + } + } + ~Artifacts() { cleanup(); } + Artifacts(const Artifacts&) = delete; + Artifacts& operator=(const Artifacts&) = delete; + std::string engine() const { return (root_ / "edge_llm/engine").string(); } + std::string tokenizer() const { return (root_ / "edge_llm/runtime_tokenizer").string(); } + std::string checkpoint() const { return (root_ / "edge_llm/checkpoint").string(); } + + private: + void cleanup() noexcept { + std::error_code ignored; + fs::remove_all(root_, ignored); + } + fs::path root_; +}; + +/// Close the plugin handle after runtime destruction; registrations remain mapped. +struct CloseLibrary { + void operator()(void* handle) const noexcept { + if (handle) + dlclose(handle); + } +}; + +/// Initialize the CMake-installed adjacent plugin without process-global environment mutation. +std::unique_ptr load_plugin() { + Dl_info location{}; + if (!dladdr(reinterpret_cast(&create), &location) || !location.dli_fname) + throw std::runtime_error("Cannot locate Nemotron-H family library"); + const auto path = + fs::absolute(location.dli_fname).parent_path() / "libNvInfer_edgellm_plugin.so"; + std::unique_ptr plugin( + dlopen(path.c_str(), RTLD_NOW | RTLD_GLOBAL | RTLD_NODELETE)); + if (!plugin) + throw std::runtime_error("Cannot load CMake-installed Edge plugin: " + + std::string(dlerror())); + using Initialize = bool (*)(void*, const char*); + auto initialize = reinterpret_cast(dlsym(plugin.get(), "initEdgellmPlugins")); + if (!initialize || !initialize(static_cast(&trt_edgellm::gLogger), "")) + throw std::runtime_error("Cannot initialize Nemotron-H Edge plugin"); + return plugin; +} + +/// Stream ownership is independent of construction success and outlives the Edge instance. +class Stream { + public: + Stream() { check_cuda(cudaStreamCreateWithFlags(&value_, cudaStreamNonBlocking)); } + ~Stream() { cudaStreamDestroy(value_); } + Stream(const Stream&) = delete; + Stream& operator=(const Stream&) = delete; + cudaStream_t get() const { return value_; } + + private: + cudaStream_t value_{nullptr}; +}; + +/// Drain all queued work before request storage destruction or mutex release. +class RequestDrain { + public: + explicit RequestDrain(cudaStream_t stream) : stream_(stream) {} + ~RequestDrain() noexcept { cudaStreamSynchronize(stream_); } + void checked() { check_cuda(cudaStreamSynchronize(stream_)); } + + private: + cudaStream_t stream_; +}; + +/// The paired engine ABI fixes a linear block16 speculative schedule. +std::optional +drafting_config(const nlohmann::json& marker) { + if (marker.value("execution_variant", "autoregressive") != "dflash") + return std::nullopt; + trt_edgellm::rt::SpecDecodeDraftingConfig config{}; + config.draftingTopK = 1; + config.draftingStep = 1; + config.verifySize = 16; + config.dflashBlockSize = 16; + return config; +} + +/// Use public artifact injection; its coordinator retains this supplied tokenizer. +trt_edgellm::rt::ModelArtifacts +runtime_artifacts(const Artifacts& files, const nlohmann::json& marker, cudaStream_t stream) { + auto artifacts = trt_edgellm::rt::ModelArtifacts::loadFromEngineDir( + files.engine(), drafting_config(marker), files.checkpoint(), "", stream); + artifacts.tokenizer = load_tokenizer(files.tokenizer(), + marker.at("native_eos_token_ids").get>()); + artifacts.deployment.base.eosTokenIds = + marker.at("native_eos_token_ids").get>(); + return artifacts; +} + +/// Read the original source template resolved into family-owned runtime metadata. +std::string native_chat_format(const BundleReader& bundle) { + const auto bytes = bundle.read_section("edge_llm/runtime_tokenizer/tokenizer_config.json"); + const auto metadata = nlohmann::json::parse(bytes.begin(), bytes.end()); + const auto text = metadata.at("chat_template").get(); + if (text.empty()) + throw std::runtime_error("Nemotron-H Edge requires resolved source chat_template"); + const auto format = nemotron_h_detect_chat_template_format(text); + if (format.empty()) + throw std::runtime_error("Nemotron-H Edge does not recognize source chat_template"); + return format; +} + +/// Reuse the original family BPE implementation, including its special postprocessor. +std::unique_ptr native_tokenizer(const BundleReader& bundle) { + const auto bytes = bundle.read_section("edge_llm/checkpoint/tokenizer.json"); + return CreateBpeTokenizer(reinterpret_cast(bytes.data()), bytes.size(), true); +} + +/// Thin persistent Edge API adapter; serialization prevents concurrent use of Edge request state. +class EdgeTask final : public ITextGeneration { + public: + EdgeTask(const BundleReader& bundle, const nlohmann::json& marker) + : artifacts_(bundle, marker), native_tokenizer_(native_tokenizer(bundle)), + plugin_(load_plugin()), + runtime_(runtime_artifacts(artifacts_, marker, stream_.get()), artifacts_.engine(), "", + std::unordered_map{}, drafting_config(marker), + stream_.get()), + chat_format_(native_chat_format(bundle)), + capacity_(marker.at("max_sequence_length").get()), + input_limit_(marker.at("max_input_length").get()), + bos_id_(marker.at("native_bos_token_id").get()), + paired_(marker.value("execution_variant", "autoregressive") == "dflash") {} + + std::int32_t default_max_new_tokens() const override { return kDefaultMaxNewTokens; } + + /// Drain work from failed requests before destroying the runtime and its weight buffers. + ~EdgeTask() override { cudaStreamSynchronize(stream_.get()); } + + /// Invoke Edge once; failures propagate without attempting native inference. + TextResult generate(const std::string& prompt, const TextGenerationConfig& config) override { + auto request = make_request(prompt, config, chat_format_, *native_tokenizer_, bos_id_); + // The pinned DFlash decoder forces greedy verification. Reject sampling + // rather than silently replacing caller-requested stochastic semantics. + if (paired_ && request.temperature != 0.0F) + throw std::invalid_argument("Nemotron-H DFlash supports greedy generation only"); + const auto count = request.preTokenizedInputIds.front().size(); + if (count == 0) + return {}; + if (count > static_cast(input_limit_)) + throw std::invalid_argument("Nemotron-H Edge input exceeds bundle capacity"); + std::lock_guard lock(mutex_); + validate_capacity(static_cast(count), input_limit_, capacity_, + request.maxGenerateLength); + trt_edgellm::rt::LLMGenerationResponse response{}; + RequestDrain drain(stream_.get()); + if (!runtime_.handleRequest(request, response, stream_.get())) + throw std::runtime_error("Nemotron-H Edge generation failed"); + drain.checked(); + validate_response(response, input_limit_, capacity_, request.maxGenerateLength, + static_cast(count)); + // This API does not expose per-request stage times; zero means unavailable. + return {native_tokenizer_->decode(response.outputIds.front()), + std::move(response.outputIds.front())}; + } + + private: + // Reverse destruction order keeps weights, plugin and stream alive throughout Edge teardown. + Artifacts artifacts_; + std::unique_ptr native_tokenizer_; + std::unique_ptr plugin_; + Stream stream_; + trt_edgellm::rt::LLMInferenceRuntime runtime_; + std::string chat_format_; + int capacity_; + int input_limit_; + int bos_id_; + bool paired_; + std::mutex mutex_; +}; +} // namespace + +ITask* create(const BundleReader& bundle) { + const auto bytes = bundle.read_section("edge_llm.json"); + const auto marker = nlohmann::json::parse(bytes.begin(), bytes.end()); + if (marker.at("version") != 1 || marker.at("edge_revision") != kRevision || + marker.at("max_sequence_length").get() <= 1 || + marker.at("max_input_length").get() <= 0 || + marker.at("max_input_length").get() > marker.at("max_sequence_length").get() || + marker.at("max_batch_size") != 1 || marker.at("precision") != "fp16" || + !marker.at("artifacts").is_array() || marker.at("tokenizer_policy") != "native_full_eos" || + !marker.at("native_bos_token_id").is_number_integer() || + !marker.at("native_eos_token_ids").is_array() || marker.at("native_eos_token_ids").empty()) + throw std::runtime_error("Invalid Nemotron-H Edge bundle contract"); + const auto variant = marker.value("execution_variant", "autoregressive"); + if ((variant != "autoregressive" && variant != "dflash") || + (variant == "dflash" && (marker.value("builder_flow", "") != "onnx" || + marker.value("dflash_block_size", 0) != 16))) + throw std::runtime_error("Invalid Nemotron-H Edge execution variant"); + std::set stops; + for (const auto& id : marker.at("native_eos_token_ids")) { + if (!id.is_number_integer() || id.get() < 0 || !stops.insert(id.get()).second) + throw std::runtime_error("Invalid Nemotron-H Edge full EOS set"); + } + validate_target(marker.at("target")); + return new EdgeTask(bundle, marker); +} + +} // namespace trtmc::nemotron_h::edge_llm diff --git a/families/nemotron_h/runtime/edge_llm/adapter.h b/families/nemotron_h/runtime/edge_llm/adapter.h new file mode 100644 index 0000000000..b62c2fda39 --- /dev/null +++ b/families/nemotron_h/runtime/edge_llm/adapter.h @@ -0,0 +1,15 @@ +/* + * SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. + * SPDX-License-Identifier: Apache-2.0 + */ +#pragma once + +#include "trtmc/bundle.h" +#include "trtmc/task.h" + +namespace trtmc::nemotron_h::edge_llm { + +/// Create a persistent Edge task from a self-contained bundle; throws on load failure. +ITask* create(const BundleReader& bundle); + +} // namespace trtmc::nemotron_h::edge_llm diff --git a/families/nemotron_h/runtime/edge_llm/contract.h b/families/nemotron_h/runtime/edge_llm/contract.h new file mode 100644 index 0000000000..9a1c63c029 --- /dev/null +++ b/families/nemotron_h/runtime/edge_llm/contract.h @@ -0,0 +1,60 @@ +/* + * SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. + * SPDX-License-Identifier: Apache-2.0 + */ +#pragma once + +#include "trtmc/task.h" + +#include +#include +#include +#include + +namespace trtmc::nemotron_h::edge_llm { + +/// Match the native Nemotron-H task default; insufficient capacity is an explicit error. +inline constexpr int kDefaultMaxNewTokens = 128; + +inline constexpr const char* kRevision = "e8b29522938901f6df19ebeedd4b69bc8edbcd97"; + +/// Return whether an artifact is a normalized file below one of the three Edge roots. +inline bool safe_artifact_path(const std::string& name) { + if (name.find('\\') != std::string::npos || name.find('\0') != std::string::npos) + return false; + const std::filesystem::path path(name); + if (path.is_absolute() || path.filename().empty()) + return false; + for (const auto& part : path) + if (part == "." || part == "..") + return false; + return path.generic_string() == name && + (name.rfind("edge_llm/engine/", 0) == 0 || name.rfind("edge_llm/checkpoint/", 0) == 0 || + name.rfind("edge_llm/runtime_tokenizer/", 0) == 0); +} + +/// Reject invalid sampling settings and controls with no equivalent Edge request API. +inline void validate_generation(const TextGenerationConfig& c) { + if (!std::isfinite(c.temperature) || c.temperature < 0 || !std::isfinite(c.top_p) || + c.top_p < 0 || c.top_p > 1 || c.top_k < 0) + throw std::invalid_argument("Invalid Nemotron-H Edge sampling parameters"); + if (c.min_p != 0 || c.seed != -1 || c.eos_token_id != -1 || c.repetition_penalty != 1 || + !c.lora_adapter_id.empty() || c.stop_on_boxed_answer || + (c.text_generation_mode != "auto" && c.text_generation_mode != "autoregressive") || + c.source_language_token_id != -1 || c.forced_bos_token_id != -1 || c.guidance_scale != -1 || + c.cfg_scale != -1 || c.num_steps != -1 || c.sde_gamma != -1 || !c.initial_latents.empty() || + !c.condition_latents.empty() || !c.condition_mask.empty() || !c.sampling_steps.empty() || + !c.sde_noises.empty() || c.block_length != 0 || c.confidence_threshold != -1) + throw std::invalid_argument( + "Requested generation controls are unsupported by Nemotron-H Edge"); +} + +/// Enforce prompt and total capacity without allowing Edge to silently clip generation. +inline void validate_capacity(int prompt_tokens, int input_limit, int capacity, + std::int64_t generated_tokens) { + if (prompt_tokens <= 0 || prompt_tokens > input_limit || generated_tokens <= 0 || + generated_tokens > static_cast(capacity) - prompt_tokens) + throw std::invalid_argument("Nemotron-H Edge prompt and generation exceed bundle capacity"); +} + +} // namespace trtmc::nemotron_h::edge_llm diff --git a/families/nemotron_h/runtime/edge_llm/device_link.cu b/families/nemotron_h/runtime/edge_llm/device_link.cu new file mode 100644 index 0000000000..1ea4768d60 --- /dev/null +++ b/families/nemotron_h/runtime/edge_llm/device_link.cu @@ -0,0 +1,5 @@ +/* + * SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. + * SPDX-License-Identifier: Apache-2.0 + */ +// Enable the final CUDA device-link step for Edge's static runtime dependencies. diff --git a/families/nemotron_h/runtime/edge_llm/request.h b/families/nemotron_h/runtime/edge_llm/request.h new file mode 100644 index 0000000000..ff0901d2ea --- /dev/null +++ b/families/nemotron_h/runtime/edge_llm/request.h @@ -0,0 +1,68 @@ +/* + * SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. + * SPDX-License-Identifier: Apache-2.0 + */ +#pragma once +#include "families/nemotron_h/runtime/chat_templates.h" +#include "families/nemotron_h/runtime/edge_llm/contract.h" +#include "families/nemotron_h/runtime/tokenizer.h" + +#include +#include + +namespace trtmc::nemotron_h::edge_llm { + +/// Apply the existing native family renderer, then pass already framed raw text. +inline trt_edgellm::rt::LLMGenerationRequest make_request(const std::string& prompt, + const TextGenerationConfig& config, + const std::string& chat_format, + const ITokenizer& tokenizer, int bos_id) { + validate_generation(config); + const auto formatted = + config.use_chat_template && !chat_format.empty() + ? nemotron_h_apply_chat_template(chat_format, prompt, config.enable_thinking) + : prompt; + trt_edgellm::rt::LLMGenerationRequest request{}; + request.requests.resize(1); + request.requests.front().messages.push_back({"user", {{"text", formatted}}}); + auto ids = tokenizer.encode(formatted); + if (config.use_chat_template && !chat_format.empty() && ids.size() >= 2 && bos_id >= 0 && + ids[0] == bos_id && ids[1] == bos_id) + ids.erase(ids.begin()); + request.preTokenizedInputIds.push_back(std::move(ids)); + request.applyChatTemplate = false; + request.addGenerationPrompt = false; + request.enableThinking = false; + const bool greedy = config.temperature < 1e-6F || config.top_p <= 0 || + (config.top_k <= 1 && config.top_p >= 1.0F - 1e-6F); + request.temperature = greedy ? 0.0F : config.temperature; + request.topK = config.top_k; + request.topP = config.top_p <= 0 ? 1.0F : config.top_p; + request.maxGenerateLength = + config.max_new_tokens > 0 ? config.max_new_tokens : kDefaultMaxNewTokens; + return request; +} + +/// Validate the ORIGINAL requested budget even after early EOS or upstream clipping. +inline void validate_response(const trt_edgellm::rt::LLMGenerationResponse& response, + std::int32_t input_limit, std::int32_t capacity, + std::int64_t requested_budget, int submitted_count) { + namespace rt = trt_edgellm::rt; + if (response.outputIds.size() != 1 || response.outputTexts.size() != 1 || + response.outputIds.front().empty() || requested_budget <= 0 || + response.outputIds.front().size() > static_cast(requested_budget) || + response.finishReasons.size() != 1 || + (response.finishReasons.front() != rt::FinishReason::kEndId && + response.finishReasons.front() != rt::FinishReason::kLength)) + throw std::runtime_error("Nemotron-H Edge generation did not complete successfully"); + if (response.inputTokenCounts.size() != 1 || + response.inputTokenCounts.front() != submitted_count) + throw std::runtime_error("Nemotron-H Edge returned invalid input counts"); + validate_capacity(response.inputTokenCounts.front(), input_limit, capacity, requested_budget); + if (response.finishReasons.front() == rt::FinishReason::kLength && + response.outputIds.front().size() != static_cast(requested_budget)) + throw std::runtime_error( + "Nemotron-H Edge silently shortened the requested generation budget"); +} + +} // namespace trtmc::nemotron_h::edge_llm diff --git a/families/nemotron_h/runtime/edge_llm/tokenizer.h b/families/nemotron_h/runtime/edge_llm/tokenizer.h new file mode 100644 index 0000000000..d835ebb940 --- /dev/null +++ b/families/nemotron_h/runtime/edge_llm/tokenizer.h @@ -0,0 +1,29 @@ +/* + * SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. + * SPDX-License-Identifier: Apache-2.0 + */ +#pragma once +#include +#include +#include +#include + +namespace trtmc::nemotron_h::edge_llm { + +/// Load separately derived metadata through the public API and enforce native full EOS. +inline std::unique_ptr +load_tokenizer(const std::filesystem::path& directory, const std::vector& native_eos) { + auto tokenizer = std::make_unique(); + if (native_eos.empty() || native_eos.front() < 0 || !tokenizer->loadFromHF(directory) || + tokenizer->getEosId() != native_eos.front()) + throw std::runtime_error("Nemotron-H Edge derived tokenizer differs from native full EOS"); + for (const auto id : native_eos) + if (id < 0 || tokenizer->idToPiece(id, false).empty()) + throw std::runtime_error("Nemotron-H Edge EOS has no tokenizer vocabulary mapping"); + tokenizer->setAdditionalEosIds(std::vector(native_eos.begin() + 1, native_eos.end())); + if (tokenizer->getEosIds() != native_eos) + throw std::runtime_error("Nemotron-H Edge tokenizer retained an incorrect full EOS set"); + return tokenizer; +} + +} // namespace trtmc::nemotron_h::edge_llm diff --git a/families/nemotron_h/runtime/plugin.cpp b/families/nemotron_h/runtime/plugin.cpp index 7ab2faaff9..fb89bfb360 100644 --- a/families/nemotron_h/runtime/plugin.cpp +++ b/families/nemotron_h/runtime/plugin.cpp @@ -4,6 +4,9 @@ */ #include "families/nemotron_h/runtime/chat_templates.h" +#ifdef TRTMC_HAS_EDGE_LLM +#include "families/nemotron_h/runtime/edge_llm/adapter.h" +#endif #include "families/nemotron_h/runtime/distributed_runtime.h" #include "families/nemotron_h/runtime/hybrid_state.h" #include "families/nemotron_h/runtime/pipeline.h" @@ -215,5 +218,12 @@ ITask* create(const FamilyContext& context) { extern "C" trtmc::ITask* trtmc_create_family(const trtmc::FamilyContext& context) { if (context.kv_cache_size_bytes != 0) throw std::invalid_argument("nemotron_h does not support --kv-cache-size"); + if (context.reader.find_section("edge_llm.json")) { +#ifdef TRTMC_HAS_EDGE_LLM + return trtmc::nemotron_h::edge_llm::create(context.reader); +#else + throw std::runtime_error("Nemotron-H Edge bundle requires TRTMC_ENABLE_EDGELLM=ON"); +#endif + } return trtmc::nemotron_h::create(context); } diff --git a/families/nemotron_h/tests/test_e2e.py b/families/nemotron_h/tests/test_e2e.py index d56f55ec28..20baed7af5 100644 --- a/families/nemotron_h/tests/test_e2e.py +++ b/families/nemotron_h/tests/test_e2e.py @@ -14,6 +14,7 @@ import shutil import subprocess from pathlib import Path +from unittest.mock import Mock import pytest @@ -133,23 +134,26 @@ def _thresholds(case_name: str) -> dict[str, float]: return thresholds -def _build_bundle(manifest: dict, model_dir: Path, bundle: Path) -> None: +def _build_bundle(manifest: dict, model_dir: Path, bundle: Path, *, execution=None) -> None: quantization = manifest.get("quantization") assert quantization is None or isinstance(quantization, str) fp32_layers = tuple(manifest.get("fp32_layers", ())) - build( - BuildRequest( - model_dir=model_dir, - output_path=bundle, - family=_FAMILY, - task="text_generation", - precision=manifest["precision"], - max_sequence_length=manifest["max_sequence_length"], - tensor_parallel_size=manifest["tensor_parallel_size"], - quantization=quantization, - fp32_layers=fp32_layers, - ) + request = BuildRequest( + model_dir=model_dir, + output_path=bundle, + family=_FAMILY, + task="text_generation", + precision=manifest["precision"], + max_sequence_length=manifest["max_sequence_length"], + tensor_parallel_size=manifest["tensor_parallel_size"], + quantization=quantization, + fp32_layers=fp32_layers, ) + if execution is not None: + from families.nemotron_h.edge_llm.config import with_execution + + request = with_execution(request, execution) + build(request) assert bundle.is_file() and bundle.stat().st_size > 0, bundle @@ -254,7 +258,14 @@ def _render_prompt(tokenizer, prompt: str, case: dict): if "enable_thinking" in case: options["enable_thinking"] = case["enable_thinking"] messages = [] - if case.get("enable_thinking") is False: + template = getattr(tokenizer, "chat_template", "") or "" + modern_chatml = ( + isinstance(template, str) + and "<|im_start|>system" in template + and "" in template + ) + # The modern source template honors enable_thinking itself, without /no_think. + if case.get("enable_thinking") is False and not modern_chatml: messages.append({"role": "system", "content": "/no_think"}) messages.append({"role": "user", "content": prompt}) rendered = tokenizer.apply_chat_template( @@ -339,7 +350,7 @@ def _hf_reference( actual_ids: list[int], torch, ) -> tuple[list[int], str, float | None, str]: - from transformers import AutoModelForCausalLM, AutoTokenizer + from transformers import AutoConfig, AutoModelForCausalLM, AutoTokenizer, GenerationConfig trust_remote_code = bool(manifest.get("trust_remote_code", False)) tokenizer = AutoTokenizer.from_pretrained( @@ -357,17 +368,148 @@ def _hf_reference( "bf16": torch.bfloat16, } assert reference_precision in dtypes, reference_precision - model = ( - AutoModelForCausalLM.from_pretrained( + reference_options = {} + if "reference_use_mamba_kernels" in case: + value = case["reference_use_mamba_kernels"] + assert isinstance(value, bool), "reference_use_mamba_kernels must be boolean" + reference_options["use_mamba_kernels"] = value + if case.get("reference_decode_modelopt_nvfp4", False): + # The declared BF16 oracle needs decoded weights, not packed NVFP4 bytes. + # This does not emulate compiled activation or KV rounding. + from modelopt.torch.quantization.qtensor import NVFP4QTensor + from modelopt.torch.export.quant_utils import QUANTIZATION_FP8, from_quantized_weight + from safetensors.torch import load_file + + assert not trust_remote_code and reference_precision == "bf16" + quant = json.loads((model_dir / "hf_quant_config.json").read_text())["quantization"] + assert quant["quant_algo"] in {"NVFP4", "MIXED_PRECISION"} + mixed = quant["quant_algo"] == "MIXED_PRECISION" + layers = quant.get("quantized_layers", {}) + if mixed: + assert layers and all( + policy["quant_algo"] == "FP8" + or (policy["quant_algo"] == "W4A16_NVFP4" and policy["group_size"] == 16) + for policy in layers.values() + ) + else: + assert quant["group_size"] == 16 + state = {} + for shard in sorted(model_dir.glob("*.safetensors")): + tensors = load_file(str(shard), device="cpu") + assert not state.keys() & tensors.keys(), "duplicate checkpoint tensors" + state.update(tensors) + packed = {key for key, value in state.items() if value.dtype == torch.uint8} + assert packed, "NVFP4 reference requires actual packed weights" + fp8 = { + key for key, value in state.items() + if key.endswith(".weight") and value.dtype == torch.float8_e4m3fn + } + quantized = packed | fp8 + if mixed: + assert quantized == {key + ".weight" for key in layers} + assert all(layers[key.removesuffix(".weight")]["quant_algo"] == "FP8" for key in fp8) + assert all( + layers[key.removesuffix(".weight")]["quant_algo"] == "W4A16_NVFP4" + for key in packed + ) + else: + assert not fp8 + for key in packed: + assert key.endswith(".weight"), key + weight = state[key] + assert weight.ndim == 2 and weight.shape[-1] % 8 == 0, key + prefix = key.removesuffix("weight") + scale = state[prefix + "weight_scale"] + double_scale = state[prefix + "weight_scale_2"] + shape = (weight.shape[0], weight.shape[1] * 2) + assert scale.dtype == torch.float8_e4m3fn + assert scale.shape == (shape[0], shape[1] // 16), key + assert torch.isfinite(scale.float()).all() and (scale.float() >= 0).all() + assert double_scale.numel() == 1 and torch.isfinite(double_scale).all() + assert (double_scale > 0).all() + state[key] = NVFP4QTensor(shape, torch.bfloat16, weight).dequantize( + dtype=torch.bfloat16, scale=scale, double_scale=double_scale, + block_sizes={-1: 16}, fast=False, + ) + assert state[key].shape == shape and torch.isfinite(state[key]).all(), key + for key in fp8: + weight = state[key] + scale = state[key.removesuffix("weight") + "weight_scale"] + assert weight.ndim == 2 and scale.numel() == 1, key + assert torch.isfinite(scale).all() and (scale > 0).all(), key + state[key] = from_quantized_weight( + weight, scale, QUANTIZATION_FP8, torch.bfloat16, + ) + assert state[key].shape == weight.shape and torch.isfinite(state[key]).all(), key + for key in list(state): + if key.endswith((".weight_scale", ".weight_scale_2", ".input_scale")): + assert key.rsplit(".", 1)[0] + ".weight" in quantized, key + scale = state.pop(key) + assert torch.isfinite(scale.float()).all() and (scale.float() >= 0).all(), key + # ModelOpt exports FP8 KV calibration buffers as k_scale/v_scale. + # The floating BF16 oracle does not use quantized KV; these are not weights. + kv_scales = { + key for key in state if key.endswith((".k_proj.k_scale", ".v_proj.v_scale")) + } + if kv_scales: + assert quant["kv_cache_quant_algo"] == "FP8" + k_prefixes = { + key.removesuffix(".k_proj.k_scale") + for key in kv_scales if key.endswith(".k_proj.k_scale") + } + v_prefixes = { + key.removesuffix(".v_proj.v_scale") + for key in kv_scales if key.endswith(".v_proj.v_scale") + } + assert k_prefixes == v_prefixes, "unpaired FP8 KV calibration" + for key in kv_scales: + scale = state[key] + weight = state[key.rsplit(".", 1)[0] + ".weight"] + assert weight.is_floating_point() and weight.ndim == 2, key + assert scale.dtype == torch.float32 and scale.numel() == 1, key + assert torch.isfinite(scale).all() and (scale > 0).all(), key + del state[key] + config = AutoConfig.from_pretrained( + model_dir, local_files_only=True, trust_remote_code=False, **reference_options, + ) + embedded = getattr(config, "quantization_config", None) + if mixed: + assert embedded["quant_method"] == "modelopt" + assert embedded["quant_algo"] == quant["quant_algo"] + assert embedded["quantized_layers"] == layers + # All declared packed/FP8 matrices are now independently decoded. + # Do not ask HF to quantize the already decoded BF16 state again. + del config.quantization_config + else: + assert not embedded + # Keep Transformers' official checkpoint-name conversion (backbone -> model). + from transformers import NemotronHForCausalLM + + assert isinstance(config, NemotronHForCausalLM.config_class) + model, loading = NemotronHForCausalLM.from_pretrained( + None, config=config, state_dict=state, torch_dtype=torch.bfloat16, + attn_implementation="eager", output_loading_info=True, + ) + assert all(not loading.get(key) for key in ( + "missing_keys", "unexpected_keys", "mismatched_keys", "error_msgs", + )), loading + if (model_dir / "generation_config.json").is_file(): + model.generation_config = GenerationConfig.from_pretrained( + model_dir, local_files_only=True, + ) + del state, tensors + else: + model = AutoModelForCausalLM.from_pretrained( model_dir, local_files_only=True, trust_remote_code=trust_remote_code, torch_dtype=dtypes[reference_precision], attn_implementation="eager", + **reference_options, ) - .eval() - .to("cuda") - ) + reference_device = case.get("reference_device", "cuda") + assert reference_device in {"cpu", "cuda"}, reference_device + model = model.eval().to(reference_device) inputs = _render_prompt(tokenizer, prompt, case).to(model.device) prompt_ids = inputs["input_ids"][0].tolist() if "expected_prompt_token_ids" in case: @@ -549,6 +691,25 @@ def test_reference_routes_keep_tp4_on_golden_and_single_gpu_on_hf() -> None: '"use_cache": False', ): assert required in source + for template, modern in ( + ("", False), + ("<|im_start|>", False), + ("<|im_start|>system ", True), + ): + tokenizer = Mock(chat_template=template) + tokenizer.apply_chat_template.return_value = "rendered" + _render_prompt( + tokenizer, "prompt", {"use_chat_template": True, "enable_thinking": False} + ) + messages = [{"role": "user", "content": "prompt"}] + if not modern: + messages.insert(0, {"role": "system", "content": "/no_think"}) + tokenizer.apply_chat_template.assert_called_once_with( + messages, tokenize=False, add_generation_prompt=True, enable_thinking=False + ) + tokenizer.assert_called_once_with( + "rendered", return_tensors="pt", add_special_tokens=False + ) _, case = _CASES["nemotron-h-nano-9b-tp4"] assert _reference_backend(case) == "golden_snapshot" assert _golden_reference(case) == "Paris" diff --git a/families/nemotron_h/tests/test_runtime_contract.py b/families/nemotron_h/tests/test_runtime_contract.py index 25a6d58e56..7ce303003f 100644 --- a/families/nemotron_h/tests/test_runtime_contract.py +++ b/families/nemotron_h/tests/test_runtime_contract.py @@ -63,3 +63,314 @@ def test_runtime_keeps_checkpoint_prompt_and_builder_policies() -> None: source = path.read_text(encoding="utf-8") assert "builder_optimization_level = 3" in source assert "builder_optimization_level = 1" not in source + + # Edge has native scalar prefill even when the optimized head80 SSD is absent. + from families.nemotron_h.edge_llm.dispatch import EDGE_DISPATCH, platform_matches + + for sm in (80, 120): + assert platform_matches({"mamba_head_dim": 80}, {"sm": sm}) + assert ("linux", "x86_64", sm, "fp16") in EDGE_DISPATCH + assert ("linux", "x86_64", 80, "nvfp4") not in EDGE_DISPATCH + + +def _edge_cli_source(tmp_path): + import json + + source = tmp_path / "target" + source.mkdir() + (source / "config.json").write_text(json.dumps({"model_type": "nemotron_h"})) + draft = tmp_path / "draft" + draft.mkdir() + return source, draft + + +@pytest.mark.parametrize("precision", [None, "fp16", "fp32"]) +def test_edge_cli_uses_ordinary_family_build(tmp_path, monkeypatch, precision): + from tensorrt_model_connect import family_cli as build_cli + from families.nemotron_h.edge_llm import dispatch + from families.nemotron_h.edge_llm.config import NemotronHBuildRequest + + source, draft = _edge_cli_source(tmp_path) + output = tmp_path / "pair.bundle" + seen = [] + + def paired(request, writer, execution): + assert isinstance(request, NemotronHBuildRequest) + assert request.execution is execution + assert execution.variant == "dflash" + assert [(item.role, item.model_dir) for item in execution.checkpoints] == [("draft", draft)] + seen.append(request) + writer.set_header(family="nemotron_h", task=request.task, backend=request.backend) + writer.add_json("edge-test.json", {"variant": execution.variant}) + + monkeypatch.setattr(dispatch, "build_paired", paired) + options = ["--precision", precision] if precision else [] + assert build_cli.main(["nemotron_h", + "build", str(source), *options, "-o", str(output), + "--execution-variant", "dflash", "--companion", f"draft={draft}", + ]) == 0 + assert len(seen) == 1 + assert seen[0].precision == (precision or "fp16") + assert output.is_file() + + from families.nemotron_h.tests.test_e2e import _build_bundle + from families.nemotron_h.edge_llm.config import BuildExecutionInputs, NamedCheckpoint + + _build_bundle( + {"precision": "fp16", "max_sequence_length": 64, "tensor_parallel_size": 1}, + source, output, + execution=BuildExecutionInputs("dflash", (NamedCheckpoint("draft", draft),)), + ) + assert len(seen) == 2 + assert seen[-1].precision == "fp16" + assert seen[-1].max_sequence_length == 64 + assert output.is_file() + + +@pytest.mark.parametrize("options", [ + ["--companion", "draft=/missing"], + ["--execution-variant", "dflash", "--companion", "missing_separator"], + ["--execution-variant", "dflash", "--companion", "=path"], + ["--execution-variant", "dflash", "--companion", "draft="], + ["--execution-variant", "dflash", "--companion", "draft=https://example.com/model"], +]) +def test_bad_edge_cli_inputs_fail_before_backend(tmp_path, monkeypatch, options): + from tensorrt_model_connect import family_cli as build_cli + + from families.nemotron_h import cli as owner + source, _ = _edge_cli_source(tmp_path) + monkeypatch.setattr(owner, "select_backend", lambda *_: pytest.fail("backend touched")) + monkeypatch.setattr(owner, "BundleWriter", lambda *_: pytest.fail("writer created")) + with pytest.raises(ValueError): + build_cli.main(["nemotron_h", "build", str(source), "-o", str(tmp_path / "out"), *options]) + + +def test_edge_cli_help_is_family_owned(tmp_path, capsys): + from tensorrt_model_connect import family_cli as build_cli + + source, _ = _edge_cli_source(tmp_path) + with pytest.raises(SystemExit) as caught: + build_cli.main(["nemotron_h", "build", str(source), "--help"]) + assert caught.value.code == 0 + help_text = capsys.readouterr().out + assert "--execution-variant {dflash}" in help_text + assert "--companion" in help_text + + +def test_edge_request_preserves_fields_and_family_owner(tmp_path): + from families.nemotron_h.edge_llm import cli + from dataclasses import fields, replace + from families.nemotron_h.build_request import BuildRequest, coerce_request + from families.nemotron_h.edge_llm.config import ( + BuildExecutionInputs, NamedCheckpoint, with_execution, + ) + + source, draft = _edge_cli_source(tmp_path) + request = BuildRequest(source, tmp_path / "out", "nemotron_h", "text_generation", "fp16", + graph_transform=lambda layer: layer) + assert coerce_request(request) is request + execution = BuildExecutionInputs("dflash", (NamedCheckpoint("draft", draft),)) + assert cli.execution_inputs(None) is None + extended = with_execution(request, execution) + for field in fields(BuildRequest): + assert getattr(extended, field.name) is getattr(request, field.name) + with pytest.raises(ValueError, match="requires the nemotron_h family"): + with_execution(replace(request, family="other"), execution) + with pytest.raises(ValueError, match="unique"): + BuildExecutionInputs("dflash", execution.checkpoints * 2) + + +@pytest.mark.parametrize("failure", [RuntimeError("paired build failed"), KeyboardInterrupt()]) +def test_edge_cli_failure_preserves_existing_bundle(tmp_path, monkeypatch, failure): + from tensorrt_model_connect import family_cli as build_cli + from families.nemotron_h.edge_llm import dispatch + + source, draft = _edge_cli_source(tmp_path) + output = tmp_path / "pair.bundle" + output.write_bytes(b"previous publication") + + def fail(request, writer, execution): + writer.set_header(family="nemotron_h", task=request.task, backend=request.backend) + writer.add_json("edge-test.json", {"variant": execution.variant}) + raise failure + + monkeypatch.setattr(dispatch, "build_paired", fail) + with pytest.raises(type(failure)) as caught: + build_cli.main(["nemotron_h", + "build", str(source), "-o", str(output), "--execution-variant", "dflash", + "--companion", f"draft={draft}", + ]) + assert caught.value is failure + assert output.read_bytes() == b"previous publication" + assert sorted(item.name for item in tmp_path.iterdir()) == ["draft", "pair.bundle", "target"] + + +def test_edge_pair_requires_draft_and_rechecks_local_inputs(tmp_path): + from tensorrt_model_connect.build import BuildRequest + from families.nemotron_h.edge_llm.config import BuildExecutionInputs, NamedCheckpoint + from families.nemotron_h.edge_llm.dispatch import build_paired + + source, draft = _edge_cli_source(tmp_path) + request = BuildRequest(source, tmp_path / "out", "nemotron_h", "text_generation", "fp16") + with pytest.raises(ValueError, match="paired execution requires"): + build_paired(request, None, BuildExecutionInputs("dflash")) + execution = BuildExecutionInputs("dflash", (NamedCheckpoint("draft", draft),)) + draft.rmdir() + with pytest.raises(ValueError, match="existing local directory"): + build_paired(request, None, execution) + + +@pytest.mark.parametrize("mode", ["absent", "success", "corrupt", "failure", "cancel", "device_failure"]) +def test_edge_optional_package_and_output_local_staging(tmp_path, monkeypatch, caplog, mode): + import json + from tensorrt_model_connect.build import BuildRequest + from families.nemotron_h.edge_llm import builder, dispatch + + source = tmp_path / "target" + source.mkdir() + (source / "config.json").write_text(json.dumps({ + "max_position_embeddings": 4096, "hidden_size": 896, + })) + prefix = tmp_path / "package" + manifest = prefix / "share/trtmc/edge-llm.json" + if mode != "absent": + manifest.parent.mkdir(parents=True) + manifest.write_text("{" if mode == "corrupt" else "{}") + monkeypatch.setattr(builder, "cmake_prefixes", lambda: [prefix]) + monkeypatch.setattr(dispatch, "candidate", lambda *_: True) + for name in ("mapped_request", "request_matches", "platform_matches"): + if hasattr(dispatch, name): + monkeypatch.setattr(dispatch, name, lambda *_: True) + if hasattr(dispatch, "source_quantization"): + monkeypatch.setattr(dispatch, "source_quantization", lambda *_: "fp16") + if hasattr(builder, "request_weight_format"): + monkeypatch.setattr(builder, "request_weight_format", lambda *_: "fp16") + request = BuildRequest(source, tmp_path / "out", "nemotron_h", "text_generation", "fp16") + writer = object() + target = {"os": "linux", "arch": "x86_64", "sm": 80} + target_calls = [] + + def local_target(): + target_calls.append(True) + if mode == "device_failure": + raise RuntimeError("CUDA discovery failed") + return target + + monkeypatch.setattr(builder, "local_target", local_target) + stages, native_calls, publications = [], [], [] + + def prepare(original, raw, platform, staging, log): + assert original is request and platform is target + assert staging.parent == request.output_path.parent + assert staging.name.startswith(f".{request.output_path.name}.edge-") + stages.append(staging) + (staging / "large-checkpoint").write_bytes(b"fixture") + if mode == "corrupt": + builder.installed_package(target) + if mode == "failure": + raise FileNotFoundError("installed SDK artifact missing") + if mode == "cancel": + raise KeyboardInterrupt() + return {}, {} + + monkeypatch.setattr(dispatch, "EDGE_DISPATCH", {("linux", "x86_64", 80, "fp16"): prepare}) + monkeypatch.setattr(builder, "publish", lambda *args: publications.append(args)) + def native(*args): + native_calls.append(args) + if mode == "cancel": + with pytest.raises(KeyboardInterrupt): + dispatch.build(request, writer, native) + else: + dispatch.build(request, writer, native) + assert all(not path.exists() for path in stages) + assert native_calls == ([(request, writer)] if mode in { + "absent", "corrupt", "failure", "device_failure", + } else []) + assert len(publications) == (1 if mode == "success" else 0) + assert len(target_calls) == (0 if mode == "absent" else 1) + logs = list(tmp_path.glob(".out.edge-*.log")) + if mode in {"corrupt", "failure", "device_failure"}: + assert len(logs) == 1 and "Traceback" in logs[0].read_text() + assert "Retrying native once" in caplog.text + else: + assert not logs and "Edge build failed" not in caplog.text + + +@pytest.mark.parametrize("options", [[], ["--precision", "fp16", "--max-sequence-length", "64"]]) +def test_declared_build_matches_legacy_request(tmp_path, monkeypatch, options): + """Owner command preserves ordinary request defaults and explicit controls.""" + import json + from families.nemotron_h import cli as owner + from tensorrt_model_connect import build_cli, family_cli + + source = tmp_path / "checkpoint" + source.mkdir() + (source / "config.json").write_text(json.dumps({"model_type": "nemotron_h"})) + output = tmp_path / "model.bundle" + captured = [] + monkeypatch.setattr(owner, "build_bundle", lambda request, output: captured.append(request)) + monkeypatch.setattr(build_cli, "build", captured.append) + args = [str(source), "-o", str(output), *options] + assert family_cli.main(["nemotron_h", "build", *args]) == 0 + assert build_cli.main(["build", *args, "--family", "nemotron_h"]) == 0 + assert len(captured) == 2 + from dataclasses import fields + assert isinstance(captured[0], owner.BuildRequest) + for field in fields(captured[1]): + assert getattr(captured[0], field.name) == getattr(captured[1], field.name) + from dataclasses import replace + from families.nemotron_h.build_request import coerce_request + assert coerce_request(captured[1]) == captured[0] + assert coerce_request(replace(captured[1], fp32_layers=[])) == captured[0] + with pytest.raises(NotImplementedError, match="fp32_layers"): + coerce_request(replace(captured[1], fp32_layers=[0])) + with pytest.raises(NotImplementedError, match="image_height"): + coerce_request(replace(captured[1], image_height=32)) + from types import SimpleNamespace + with pytest.raises(ValueError, match="unknown"): + coerce_request(SimpleNamespace(**vars(captured[1]), unexpected_option=True)) + assert captured[0].family == "nemotron_h" + assert captured[0].task == "text_generation" + assert captured[0].precision == ("fp16" if options else "fp32") + assert not output.exists() + + from families.nemotron_h.tests import test_e2e as e2e + + def capture_bundle(request): + captured.append(request) + request.output_path.write_bytes(b"test bundle") + + monkeypatch.setattr(e2e, "build", capture_bundle) + e2e._build_bundle( + {"precision": captured[1].precision, + "max_sequence_length": captured[1].max_sequence_length, + "tensor_parallel_size": captured[1].tensor_parallel_size}, + source, output, + ) + assert captured[-1] == captured[1] + + +def test_declared_help_is_offline_and_dependency_free(): + """Actual child-process help needs neither a checkpoint nor GPU imports.""" + import subprocess + import sys + + code = """ +import sys +from tensorrt_model_connect.family_cli import main +try: + main(["nemotron_h", "build", "--help"]) +except SystemExit as error: + assert error.code == 0 +else: + raise AssertionError("help did not exit") +assert "families.nemotron_h.cli" not in sys.modules +assert "tensorrt" not in sys.modules +assert "huggingface_hub" not in sys.modules +""" + result = subprocess.run([sys.executable, "-c", code], capture_output=True, text=True, check=True) + assert "trtmc nemotron_h build" in result.stdout + assert "bf16" not in result.stdout + from families.nemotron_h import cli as owner + with pytest.raises(ValueError, match="precision"): + owner.build(model="/missing", output=Path("/unused"), precision="bf16") diff --git a/website/docs/features/model-families.md b/website/docs/features/model-families.md index ae49254cc2..e4889a9a20 100644 --- a/website/docs/features/model-families.md +++ b/website/docs/features/model-families.md @@ -109,6 +109,28 @@ for exact revisions, capacities, sampling controls and quality evidence. The local paired qualification is not a registered manifest case and does not imply CI coverage of that pair or statistical sampling equivalence. +### Nemotron-H Edge-LLM execution + +Use `trtmc nemotron_h build MODEL -o model.bundle` with the owning +family's options. `trtmc nemotron_h build --help` works offline without +a checkpoint or GPU imports. This uses the existing +[family CLI protocol](../extend/family-cli.md), not an extension to the shared parser. + +The Nemotron-H family owns optional pinned Edge-LLM whole-network offload. +Ordinary compatible text builds use the experimental builder; the explicit +Lightning NVFP4/DFlash pair uses ONNX with +`--execution-variant dflash --companion draft=/path/to/draft`. +Native fallback is attempted only during ordinary preparation; it cannot +interpret unsupported packed checkpoints or replace a requested pair. + +See the [owning Nemotron-H recipe](https://github.com/NVIDIA/TensorRT-Model-Connect/blob/main/families/nemotron_h/edge_llm/README.md) +for the six recorded ordinary profiles, the qualified greedy DFlash pair, +immutable revisions and unchanged quality gates. These are bounded historical +local qualifications, not catalog-wide or current-head CI proof. The separate +direct-Edge 9B-NVFP4 quality failure and longer-context failures remain open. +The original plain 9B case is registered in the owning E2E inventory; the other +exact profiles are not implied to run in CI. + ## Runtime and validation The directory name is also the runtime DSO identity: