diff --git a/families/internvl/build_request.py b/families/internvl/build_request.py new file mode 100644 index 0000000000..d0de20899e --- /dev/null +++ b/families/internvl/build_request.py @@ -0,0 +1,98 @@ +# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-License-Identifier: Apache-2.0 + +"""internvl build inputs and strict compatibility for existing Python callers.""" + +from __future__ import annotations + +from dataclasses import dataclass, fields +from pathlib import Path +import re +from typing import ClassVar + +from tensorrt_model_connect.graph_transform import GraphTransform + + +def _validate_id(field: str, value: str) -> None: + if not isinstance(value, str) or re.fullmatch(r"[a-z][a-z0-9_]*", value) is None: + raise ValueError(f"{field} must be a lowercase identifier") + + +@dataclass(frozen=True) +class BuildRequest: + """internvl-owned inputs; unsupported legacy controls are read-only defaults.""" + + model_dir: Path + output_path: Path + family: str + task: str + precision: str + backend: str = "trt" + max_sequence_length: int | None = None + image_height: ClassVar[int | None] = None + image_width: ClassVar[int | None] = None + video_num_frames: ClassVar[int | None] = None + max_batch_size: ClassVar[int] = 1 + tensor_parallel_size: int = 1 + context_parallel_size: ClassVar[int] = 1 + quantization: ClassVar[str | None] = None + fp32_layers: ClassVar[tuple[int, ...]] = () + dynamic_kv_cache: ClassVar[bool] = False + verbose: bool = False + graph_transform: GraphTransform | None = None + + def __post_init__(self) -> None: + if not self.precision: + raise ValueError("precision must be non-empty") + _validate_id("family", self.family) + _validate_id("task", self.task) + if self.backend not in {"trt", "trt_rtx"}: + raise ValueError("backend must be 'trt' or 'trt_rtx'") + if self.max_sequence_length is not None and self.max_sequence_length < 1: + raise ValueError("max_sequence_length must be positive") + for field in ("image_height", "image_width", "video_num_frames"): + value = getattr(self, field) + if value is not None and value < 1: + raise ValueError(f"{field} must be positive") + if self.max_batch_size < 1: + raise ValueError("max_batch_size must be positive") + if self.tensor_parallel_size < 1: + raise ValueError("tensor_parallel_size must be positive") + if self.context_parallel_size < 1: + raise ValueError("context_parallel_size must be positive") + if self.quantization is not None and not self.quantization: + raise ValueError("quantization must be non-empty when provided") + if any(layer < 0 for layer in self.fp32_layers): + raise ValueError("fp32_layers must contain non-negative indices") + if not isinstance(self.dynamic_kv_cache, bool): + raise ValueError("dynamic_kv_cache must be a bool") + if self.graph_transform is not None and not callable(self.graph_transform): + raise ValueError("graph_transform must be callable when provided") + + +def coerce_request(request: object) -> BuildRequest: + """Reject unsupported/unknown legacy inputs before converting to owner fields.""" + if isinstance(request, BuildRequest): + return request + unsupported = { + "image_height": None, + "image_width": None, + "video_num_frames": None, + "max_batch_size": 1, + "context_parallel_size": 1, + "quantization": None, + "fp32_layers": (), + "dynamic_kv_cache": False, + } + for name, default in unsupported.items(): + value = getattr(request, name, default) + if name == "quantization" and value == "none": + continue + if name == "fp32_layers" and isinstance(value, (list, tuple)) and not value: + continue + if value != default: + raise NotImplementedError(f"internvl does not support {name}") + names = {field.name for field in fields(BuildRequest)} + if unknown := set(vars(request)) - names - set(unsupported): + raise ValueError(f"unknown internvl build inputs: {sorted(unknown)}") + return BuildRequest(**{name: getattr(request, name) for name in names}) diff --git a/families/internvl/cli.json b/families/internvl/cli.json new file mode 100644 index 0000000000..1f35566f07 --- /dev/null +++ b/families/internvl/cli.json @@ -0,0 +1,100 @@ +{ + "version": 1, + "commands": [ + { + "name": "build", + "help": "Build one internvl TensorRT bundle", + "executor": "python", + "handler": "cli:build", + "arguments": [ + { + "name": "model", + "type": "string", + "help": "Hugging Face model ID or local snapshot" + }, + { + "name": "output", + "flags": [ + "-o", + "--output" + ], + "type": "path", + "required": true + }, + { + "name": "revision", + "flags": [ + "--revision" + ], + "type": "string" + }, + { + "name": "task", + "flags": [ + "--task" + ], + "type": "string", + "choices": [ + "vision_language_generation" + ], + "default": "vision_language_generation" + }, + { + "name": "precision", + "flags": [ + "--precision" + ], + "type": "string", + "choices": [ + "fp32", + "fp16", + "bf16" + ], + "default": "fp32" + }, + { + "name": "backend", + "flags": [ + "--backend" + ], + "type": "string", + "choices": [ + "trt", + "trt_rtx" + ], + "default": "trt" + }, + { + "name": "max_sequence_length", + "flags": [ + "--max-sequence-length" + ], + "type": "int" + }, + { + "name": "tensor_parallel_size", + "flags": [ + "--tensor-parallel-size" + ], + "type": "int", + "choices": [ + 1, + 2, + 4, + 8 + ], + "default": 1 + }, + { + "name": "verbose", + "flags": [ + "--verbose" + ], + "type": "bool", + "action": "store_true", + "default": false + } + ] + } + ] +} diff --git a/families/internvl/cli.py b/families/internvl/cli.py new file mode 100644 index 0000000000..fcaa19a253 --- /dev/null +++ b/families/internvl/cli.py @@ -0,0 +1,50 @@ +# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-License-Identifier: Apache-2.0 + +"""internvl-owned build command and typed inputs; importing this module is CPU-only.""" + +from __future__ import annotations + +from dataclasses import replace +from pathlib import Path + +from tensorrt_model_connect.build import select_backend +from tensorrt_model_connect.bundle_writer import BundleWriter +from tensorrt_model_connect.graph_transform import graph_transform +from tensorrt_model_connect.model_support import load_model_metadata, resolve_family, resolve_model + +from .build_request import BuildRequest + + +def build_bundle(request: BuildRequest, output: Path) -> None: + """Build and publish through the owning model, preserving atomic failure.""" + request = replace(request, output_path=output) + select_backend(request.backend) + from .model import build as build_model + + writer = BundleWriter(output) + try: + with graph_transform(request.graph_transform): + build_model(request, writer) + writer.finish() + except BaseException: + writer.abort() + raise + +def build( + *, model: str, output: Path, revision: str | None = None, + task: str = "vision_language_generation", precision: str = "fp32", backend: str = "trt", + max_sequence_length: int | None = None, tensor_parallel_size: int = 1, + verbose: bool = False, +) -> int: + """Run the declared owner command; help never imports this handler.""" + model_dir = resolve_model(model, revision) + resolve_family(load_model_metadata(model_dir), "internvl") + request = BuildRequest( + model_dir=model_dir, output_path=output, family="internvl", + task=task, precision=precision, backend=backend, + max_sequence_length=max_sequence_length, tensor_parallel_size=tensor_parallel_size, + verbose=verbose, + ) + build_bundle(request, output) + return 0 diff --git a/families/internvl/edge_llm/README.md b/families/internvl/edge_llm/README.md new file mode 100644 index 0000000000..ef6375d6ac --- /dev/null +++ b/families/internvl/edge_llm/README.md @@ -0,0 +1,104 @@ +# InternVL: optional native Edge-LLM execution + +The family owns configuration admission, builder argument mapping, bundle +composition, runtime orchestration, and validation. Edge owns complete visual +and language networks. This route uses the official GitHub Edge-LLM **0.10.1** +snapshot **e8b29522938901f6df19ebeedd4b69bc8edbcd97**, provisioned through the +optional native CMake package. It does not use another checkout or cross-compile. + +## Scope + +Only original-source InternVL3 HF checkpoints with the recorded dense Qwen2 +text and 448-pixel vision configurations enter this route. Compute is FP16, +batch size and tensor/context parallelism are one. The 1B/2B/8B configurations +route on native x86 SM80; 14B routes on native x86 SM120. Other platforms, +quantized sources, InternVL3.5, multiple-image public requests, and speculative +decoding are not qualified by this change. + +The existing native builder remains the default for requests outside this +map. If Edge preparation fails, the family logs a warning and retries the +unchanged native request once. Publication errors never retry into a different +backend. Runtime loading dispatches on the family-owned Edge bundle marker. + +## Build and runtime + +Enable the generic optional SDK with `TRTMC_ENABLE_EDGELLM=ON`, build it on the +executing GPU, and expose its installation through `CMAKE_PREFIX_PATH`. +The installed package must match the pin, SM, CUDA, and TensorRT identity. +The family-owned `trtmc internvl build` command uses the existing CLI protocol. + +`edge_llm/builder.py` maps the request into the pinned Python direct builder with +`--components llm,visual`. It preserves both engines, processor/tokenizer +metadata, chat template, and the checkpoint required for external weights. +The family C++ adapter uses the installed Edge runtime API; it does not +implement either model graph. + +One public RGB image uses a 448-pixel tile and 256 visual tokens. The visual +profile retains 256 tokens per image and an aggregate token budget derived +from the requested context. That also permits the original Edge two-image +reference workload; it does **not** expand the Model Connect single-image API. +Edge owns preprocessing, normalization, visual inference, and token expansion. + +Requests retain the existing public greedy-generation controls and context +limits. Unsupported controls are rejected rather than silently ignored. +Text-only generation and image-bearing generation share a persistent Edge +runtime. Quantization is never inferred from a filename or applied implicitly. + +## Recorded full-model qualification + +The following are historical local build/inference receipts, not fresh-head +CI claims. All four passed public Model Connect versus direct Edge comparisons +and independent HF image/text comparisons with NED **0** and exact tokens. +Original Edge `vlm_basic` used its two-image fixture and unchanged gates. + +| Checkpoint (OpenGVLab) | Source revision | Native SM | Context | Edge ROUGE-1 / ROUGE-L | +| --- | --- | --- | --- | --- | +| InternVL3-1B-hf | `014c0583a0d4bedf29fbe2dbff4f865eb998e171` | 80 | 1024 | 0.5662 / 0.3193 | +| InternVL3-2B-hf | `cb57a075cb75a2e6d1b668b128d48bb00ae321d2` | 80 | 1024 | 0.6275 / 0.3958 | +| InternVL3-8B-hf | `259a3b64a14623c0ec91a045cb43f7c5af5fa6af` | 80 | 1024 | 0.5704 / 0.3654 | +| InternVL3-14B-hf | `e22931943e5336f85e06f4e2b38f3e5e6ee4de3b` | 120 | 1024 | 0.5330 / 0.3296 | + +The existing 2B and 8B owning E2Es also passed at their manifest context **384**, +including actual visual-feature health. The 8B golden-reference and 2B HF +comparison criteria were not changed. The 1B and 14B profiles have no owning +registered E2E manifest. TP2/TP4 manifests remain native and are not additional +Edge qualification. + +Successful full bundles and source checkpoints were retired after preserving +compact receipts. Replaying full inference requires restoring the exact source +and rebuilding locally. Current source/build checks are not substitutes for +that replay, nor proof of every model in the upstream catalog. + +## Existing image-health test + +Edge 0.10.1 has no visual-feature dump CLI. The test-only +`internvl_edge_vision_features` executable therefore calls the pinned +`MultimodalRunner` API and copies its actual FP16 output to a temporary file. +It contains no generation driver and is built only when both +`TRTMC_BUILD_TESTS` and the optional Edge package are enabled. + +The existing `tests/vision_oracle.py` selects this helper for Edge bundles. +It streams visual/checkpoint sections into temporary storage instead of loading +checkpoint weights into Python memory. Native bundles continue to use their +existing `vision.plan` path. `TRTMC_RUNTIME_ROOT` must point at the build tree +containing the helper and matching plugin. + +The owning E2E still requires nonempty, finite, nonzero features and then runs +its existing generation-quality comparison. No thresholds were weakened and no +new test framework was introduced. The standalone cosine contract remains +unchanged; a health assertion alone is not a claim of HF feature parity. + +Ordinary builds without an installed optional Edge SDK select native without a +warning. Malformed or incomplete installed packages still retain diagnostics and +warn before native fallback. Temporary checkpoint/engine staging uses the bundle +output directory filesystem (choose a scratch-backed output), not system /tmp. + +## Declared build command + +This family uses the existing cli.json protocol introduced in #1310. The family +owns its declaration, typed inputs and Python handler. The handler adapts those +inputs to the unchanged builder API, preserving native/Edge dispatch and bundle +publication. The legacy flat build command remains available for its existing +ordinary options; new family options use `trtmc internvl build`. +Help is offline and does not need a local checkpoint. No shared parser hook or +family registry entry is added. diff --git a/families/internvl/edge_llm/__init__.py b/families/internvl/edge_llm/__init__.py new file mode 100644 index 0000000000..df93d8cabe --- /dev/null +++ b/families/internvl/edge_llm/__init__.py @@ -0,0 +1,4 @@ +# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-License-Identifier: Apache-2.0 + +"""Family-owned optional complete-network Edge offload.""" diff --git a/families/internvl/edge_llm/builder.py b/families/internvl/edge_llm/builder.py new file mode 100644 index 0000000000..a2a5736228 --- /dev/null +++ b/families/internvl/edge_llm/builder.py @@ -0,0 +1,236 @@ +# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-License-Identifier: Apache-2.0 + +"""Thin family-owned adapter to the pinned Edge direct-builder API.""" + +from __future__ import annotations + +import json +from pathlib import Path +import shutil +import subprocess + +from tensorrt_model_connect.build import cmake_prefixes, detect_local_platform + + +EDGE_REVISION = "e8b29522938901f6df19ebeedd4b69bc8edbcd97" + + +def local_target() -> dict: + """Return the executing worker identity supplied by generic build mechanics.""" + return detect_local_platform() + + +def package_present() -> bool: + """Absence of the optional SDK is a non-match, not a failed Edge build.""" + for prefix in cmake_prefixes(): + manifest = prefix / "share/trtmc/edge-llm.json" + if manifest.exists() or manifest.is_symlink(): + return True + return False + + +def installed_package(target: dict) -> dict: + """Resolve CMake installation via standard prefixes; never install anything. + + Args: + target: Executing device and SDK identity. + + Returns: + Validated package metadata with absolute Python and plugin paths. + + Raises: + FileNotFoundError: No CMake installation or required artifact exists. + ValueError: Pin, architecture, SDK or contained-path contract differs. + """ + for prefix in cmake_prefixes(): + manifest = prefix / "share/trtmc/edge-llm.json" + if not manifest.is_file(): + continue + package = json.loads(manifest.read_text(encoding="utf-8")) + if package.get("schema_version") != 1 or package.get("revision") != EDGE_REVISION: + raise ValueError(f"Edge package has an unsupported revision/schema: {manifest}") + if package.get("version") != "0.10.1" or package.get("arch") != target["arch"]: + raise ValueError("Edge package version/architecture differs from executing worker") + if target["sm"] not in package.get("architectures", []): + raise ValueError("Edge package was not built for this local GPU") + cuda_version = ".".join(str(package.get("cuda_version", "")).split(".")[:2]) + if ( + cuda_version != target["cuda_version"] + or package.get("tensorrt_version") != target["tensorrt_version"] + ): + raise ValueError("Edge package CUDA/TensorRT differs from executing worker") + for name in ("python", "plugin"): + relative = Path(package[name]) + path = (prefix / relative).resolve() + if relative.is_absolute() or not path.is_relative_to(prefix.resolve()): + raise ValueError(f"Edge package {name} must be contained in its installation") + if not path.is_file(): + raise FileNotFoundError(f"Edge package {name} is missing: {path}") + package[name] = str(path) + return package + raise FileNotFoundError( + "Edge-LLM is not installed; enable the optional Edge-LLM CMake dependency " + "and set CMAKE_PREFIX_PATH to its install prefix" + ) + + +def component_weight_formats(model_dir: Path, raw: dict) -> dict: + """Resolve actual component layouts without overriding unknown quantized weights. + + Only original vision and decoder source weights are admitted. + Global sidecars affect both components in pinned Edge and are not admitted. + """ + if any( + (model_dir / name).exists() + for name in ("hf_quant_config.json", "quantize_config.json", "quant_config.json") + ): + raise ValueError("InternVL global quantization sidecars are not qualified") + text = raw.get("text_config", raw.get("llm_config", {})) + for component in (raw, raw.get("vision_config", {})): + if component.get("quantization_config") not in (None, {}): + raise ValueError("InternVL Edge requires unquantized root and vision metadata") + quant = text.get("quantization_config") + if quant is None or quant == {}: + return {"llm": "fp16", "visual": "fp16"} + raise ValueError("InternVL Edge publication requires unquantized decoder weights") + + +def request_weight_format(request, raw: dict) -> str: + """Preserve source for None; explicit none or named format must match it. + + FP16 request.precision is compute precision, not a dequantization request. + Neither calibration nor quantization is synthesized by this family adapter. + """ + source = component_weight_formats(Path(request.model_dir), raw)["llm"] + requested = source if request.quantization is None else request.quantization + requested = {"none": "fp16"}.get(requested, requested) + if requested != source: + raise ValueError(f"InternVL Edge source {source} differs from requested {requested}") + return source + + +def prepare(request, raw: dict, target: dict, staging: Path, log_path: Path) -> tuple[dict, dict]: + """Map the request to Edge main(argv), returning complete unpublished assets. + + Edge owns model selection, configuration, conversion, graphs and engine + composition. The family adapter only maps vision-language arguments and + preserves the checkpoint needed by Edge external-weight APIs. + + Returns: + (section-name to file mapping, runtime marker). + + Raises: + Exception: Dependency, upstream build or artifact validation failed. + """ + package = installed_package(target) + weight_format = request_weight_format(request, raw) + checkpoint = staging / "edge_llm/checkpoint" + checkpoint.mkdir(parents=True) + for source in Path(request.model_dir).iterdir(): + if source.is_file() and ( + source.suffix in {".json", ".safetensors", ".model", ".jinja"} + or source.name in {"merges.txt", "vocab.txt"} + ): + shutil.copy2(source, checkpoint / source.name) + if not list(checkpoint.glob("*.safetensors")): + raise ValueError("Edge direct builder requires a safetensors checkpoint") + config = raw.get("text_config", raw.get("llm_config")) + limit = request.max_sequence_length or min(config["max_position_embeddings"], 256) + if type(limit) is not int or limit <= 1 or limit > config["max_position_embeddings"]: + raise ValueError("Invalid InternVL Edge sequence capacity") + tokenizer = json.loads((checkpoint / "tokenizer.json").read_text(encoding="utf-8")) + tokenizer_config = json.loads( + (checkpoint / "tokenizer_config.json").read_text(encoding="utf-8") + ) + if tokenizer_config.get("add_bos_token") or tokenizer_config.get("add_eos_token"): + raise ValueError("InternVL Edge raw tokenizer cannot add BOS/EOS tokens") + post = tokenizer.get("post_processor") + if not isinstance(post, dict) or post.get("type") != "ByteLevel": + raise ValueError("InternVL Edge requires the documented raw ByteLevel tokenizer") + # Preserve one tile per image, but provision aggregate tiles for the requested context. + max_image_tokens = max(256, (limit // 256) * 256) + engine = staging / "edge_llm/engine" + # Calling upstream main preserves its complete build/artifact orchestration. + command = [ + package["python"], + "-I", + "-c", + "from experimental.builder.cli import main; main()", + "--model-dir", + str(checkpoint), + "--engine-dir", + str(engine), + "--components", + "llm,visual", + "--plugin-path", + package["plugin"], + "--dense", + "fp16" if weight_format == "fp16" else "auto", + "--max-input-len", + str(limit), + "--max-kv-cache-capacity", + str(limit), + "--max-batch-size", + "1", + "--min-image-tokens", + "256", + "--max-image-tokens", + str(max_image_tokens), + "--max-image-tokens-per-image", + "256", + ] + # Quantized attention FP16 biases lack an external-weight recipe in this pin. + # Keep them engine constants while preserving external packed INT4 weights. + kinds = ( + ("all",) + if weight_format == "fp16" + else ("int4_ffn", "int4_moe", "nvfp4_moe", "nvfp4_tp", "lm_head", "embedding") + ) + for kind in kinds: + command.extend(("--externalize-weights", kind)) + if request.verbose: + command.append("--verbose") + with log_path.open("a", encoding="utf-8") as log: + subprocess.run(command, check=True, stdout=log, stderr=subprocess.STDOUT, cwd=staging) + for name in ( + "visual/visual.engine", + "visual/config.json", + "llm.engine", + "config.json", + "tokenizer.json", + "tokenizer_config.json", + "processed_chat_template.json", + ): + if not (engine / name).is_file() or (engine / name).stat().st_size == 0: + raise ValueError(f"Edge builder did not produce required artifact: {name}") + files = {} + for directory in (engine, checkpoint): + for path in sorted(directory.rglob("*")): + if path.is_symlink(): + raise ValueError(f"Edge output must not contain symlinks: {path}") + if path.is_file(): + files[path.relative_to(staging).as_posix()] = path + return files, { + "version": 1, + "edge_revision": EDGE_REVISION, + "target": target, + "precision": "fp16", + "weight_format": weight_format, + "component_weight_formats": {"llm": weight_format, "visual": "fp16"}, + "visual_image_tokens": 256, + "visual_max_image_tokens": max_image_tokens, + "max_sequence_length": limit, + "max_input_length": limit, + "max_batch_size": 1, + "artifacts": list(files), + } + + +def publish(request, writer, files: dict, marker: dict) -> None: + """Stream complete Edge sections; publication errors must not retry native.""" + writer.set_header(family=request.family, task=request.task, backend=request.backend) + for name, path in files.items(): + with path.open("rb") as source, writer.open_section(name) as destination: + shutil.copyfileobj(source, destination, length=1024 * 1024) + writer.add_json("edge_llm.json", marker) diff --git a/families/internvl/edge_llm/dispatch.py b/families/internvl/edge_llm/dispatch.py new file mode 100644 index 0000000000..d579ad6707 --- /dev/null +++ b/families/internvl/edge_llm/dispatch.py @@ -0,0 +1,183 @@ +# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-License-Identifier: Apache-2.0 + +"""Family-owned complete-network route map; native is the default.""" + +from __future__ import annotations + +import json +import logging +import os +from pathlib import Path +import tempfile +import traceback + +from . import builder as edge_llm + +_LOG = logging.getLogger(__name__) + +# Only native platforms exercised by the recorded original-source profiles. +EDGE_DISPATCH = {("linux", "x86_64", sm, "fp16"): edge_llm.prepare for sm in (80, 120)} + +# Pinned documented text topologies, not repository-name heuristics. +# type, layers, hidden, intermediate, attention heads, KV heads, vocabulary. +INTERNVL_CONFIGS = { + ("qwen2", 24, 896, 4864, 14, 2, 151674), + ("qwen2", 28, 1536, 8960, 12, 2, 151674), + ("qwen2", 28, 3584, 18944, 28, 4, 151674), + ("qwen2", 48, 5120, 13824, 40, 8, 151674), +} + + +def mapped_request(request) -> bool: + """Preserve native-only controls before reading any checkpoint metadata.""" + return ( + request.backend == "trt" + and request.task == "vision_language_generation" + and request.precision.lower() == "fp16" + and request.quantization in {None, "none"} + and request.max_batch_size + == request.tensor_parallel_size + == request.context_parallel_size + == 1 + and not request.dynamic_kv_cache + and not request.fp32_layers + and getattr(request, "graph_transform", None) is None + and all( + value is None + for value in (request.image_height, request.image_width, request.video_num_frames) + ) + ) + + +def candidate(request, raw: dict) -> bool: + """Admit documented dense InternVL text/vision configurations and mapped controls.""" + config = raw.get("text_config", raw.get("llm_config", {})) + vision = raw.get("vision_config", {}) + if not isinstance(config, dict) or not isinstance(vision, dict): + return False + keys = ( + "model_type", + "num_hidden_layers", + "hidden_size", + "intermediate_size", + "num_attention_heads", + "num_key_value_heads", + "vocab_size", + ) + shape = tuple(config.get(key) for key in keys) + return ( + raw.get("model_type") in {"internvl", "internvl_chat"} + and not ("text_config" in raw and "llm_config" in raw) + and all(type(value) is int for value in shape[1:]) + and shape in INTERNVL_CONFIGS + and not any( + config.get(key) for key in ("num_experts", "num_local_experts", "moe_intermediate_size") + ) + and not any( + raw.get(key) + for key in ("audio_config", "draft_config", "eagle_config", "speculative_config") + ) + and config.get("hidden_act", "silu") == "silu" + and vision.get("image_size") in (448, [448, 448]) + and vision.get("patch_size") in (14, [14, 14]) + and vision.get("num_channels", 3) == 3 + and vision.get("model_type") in {"internvl_vision", "intern_vit_6b"} + and tuple( + vision.get(key) + for key in ( + "hidden_size", + "intermediate_size", + "num_hidden_layers", + "num_attention_heads", + ) + ) + == (1024, 4096, 24, 16) + and not vision.get("use_moe", False) + and raw.get("downsample_ratio", 0.5) == 0.5 + and raw.get("image_seq_length", 256) == 256 + and mapped_request(request) + ) + + +def build(request, writer, native) -> None: + """Dispatch locally or warn and retry native once with the original request. + + Args: + request: Unmodified Model Connect build request. + writer: Unpublished bundle writer. + native: This family's original native builder callback. + + Raises: + Exception: Common input/publication error, or native build error with + Edge cause after a failed preparation. Cancellation never retries. + """ + if not mapped_request(request) or not (Path(request.model_dir) / "config.json").is_file(): + native(request, writer) + return + raw = json.loads((Path(request.model_dir) / "config.json").read_text(encoding="utf-8")) + if not isinstance(raw, dict): + raise ValueError("checkpoint config.json must contain an object") + config = raw.get("text_config", raw.get("llm_config", raw)) + if not isinstance(config, dict): + raise ValueError("checkpoint text_config must contain an object") + if not candidate(request, raw): + native(request, writer) + return + if not edge_llm.package_present(): + native(request, writer) + return + failure = None + descriptor, name = tempfile.mkstemp( + prefix=f".{request.output_path.name}.edge-", suffix=".log", dir=request.output_path.parent + ) + os.close(descriptor) + log_path = Path(name) + with tempfile.TemporaryDirectory( + prefix=f".{request.output_path.name}.edge-", dir=request.output_path.parent + ) as directory: + try: + target = edge_llm.local_target() + weight_format = edge_llm.request_weight_format(request, raw) + key = (target["os"], target["arch"], target["sm"], weight_format) + adapter = EDGE_DISPATCH.get(key) + # The recorded 14B profile uses SM120; 1B/2B/8B use SM80. + if target["sm"] != (120 if config["hidden_size"] == 5120 else 80): + adapter = None + if adapter is not None: + # This is an Edge profile restriction, not a native build limit. + capacity = config.get("max_position_embeddings") + if type(capacity) is not int or capacity <= 0: + raise ValueError( + "checkpoint max_position_embeddings must be a positive integer" + ) + if request.max_sequence_length and request.max_sequence_length > capacity: + raise ValueError("max_sequence_length exceeds checkpoint context capacity") + files, marker = adapter(request, raw, target, Path(directory), log_path) + except Exception as error: + failure = error + with log_path.open("a", encoding="utf-8") as log: + traceback.print_exception(error, file=log) + _LOG.warning( + "internvl Edge build failed: %s. Diagnostics: %s. " + "Retrying native once with the unchanged request.", + error, + log_path, + exc_info=True, + ) + except BaseException: + log_path.unlink(missing_ok=True) + raise + else: + # Edge preparation did not touch writer; publication cannot fallback. + if adapter is not None: + edge_llm.publish(request, writer, files, marker) + log_path.unlink() + return + log_path.unlink() # A platform non-match is not an Edge failure. + try: + native(request, writer) + except Exception as error: + if failure is not None: + raise error from failure + raise diff --git a/families/internvl/model.py b/families/internvl/model.py index 0a2126c02d..21cefa2248 100644 --- a/families/internvl/model.py +++ b/families/internvl/model.py @@ -40,7 +40,7 @@ if TYPE_CHECKING: - from tensorrt_model_connect.build import BuildRequest + from .build_request import BuildRequest from tensorrt_model_connect.bundle_writer import BundleWriter @@ -392,115 +392,126 @@ def _tokenizer_runtime_contract(model_dir: Path) -> dict[str, object]: def build(request: "BuildRequest", writer: "BundleWriter") -> None: - """Build one InternVL vision-language bundle.""" - if request.dynamic_kv_cache: - raise NotImplementedError("internvl does not support dynamic_kv_cache") - - if request.image_height is not None: - raise NotImplementedError("internvl does not support image_height") - - if request.image_width is not None: - raise NotImplementedError("internvl does not support image_width") - - if request.video_num_frames is not None: - raise NotImplementedError("internvl does not support video_num_frames") - - if request.max_batch_size != 1: - raise NotImplementedError("internvl does not support max_batch_size") - - if request.context_parallel_size != 1: - raise ValueError("this family does not support context parallelism") - - if request.task != "vision_language_generation": - raise ValueError("internvl supports only task=vision_language_generation") - if request.quantization not in {None, "none"} or request.fp32_layers: - raise NotImplementedError("InternVL supports only non-quantized uniform-precision builds") - model_dir = Path(request.model_dir) - config = ModelConfig.from_dir(model_dir) - if str(config.model_type).lower() not in {"internvl_chat", "internvl3", "internvl"}: - raise ValueError(f"InternVL does not support model_type={config.model_type!r}") - precision = str(request.precision).lower() - max_length = int(request.max_sequence_length or min(config.max_position_embeddings, 256)) - parallel = ParallelConfig(tp_size=int(request.tensor_parallel_size)) - parallel.validate() - config.raw["_model_dir"] = str(model_dir) - model = _InternVLModel() - weights = model.load_weights(str(model_dir), config) - writer.set_header(family="internvl", task=request.task, backend=request.backend) - if parallel.enabled: - for rank in range(parallel.tp_size): - writer.add_bytes( - f"engine.rank{rank}.plan", - model.build_engine( - config, - weights, - max_length, - precision=precision, - quant_ctx=None, - verbose=request.verbose, - parallel_config=parallel.for_rank(rank), - ), + """Select complete-network offload or preserve the native InternVL builder.""" + from .build_request import coerce_request + + request = coerce_request(request) + from .edge_llm.dispatch import build as dispatch_build + + def _build_native(request: "BuildRequest", writer: "BundleWriter") -> None: + """Build one InternVL vision-language bundle.""" + if request.dynamic_kv_cache: + raise NotImplementedError("internvl does not support dynamic_kv_cache") + + if request.image_height is not None: + raise NotImplementedError("internvl does not support image_height") + + if request.image_width is not None: + raise NotImplementedError("internvl does not support image_width") + + if request.video_num_frames is not None: + raise NotImplementedError("internvl does not support video_num_frames") + + if request.max_batch_size != 1: + raise NotImplementedError("internvl does not support max_batch_size") + + if request.context_parallel_size != 1: + raise ValueError("this family does not support context parallelism") + + if request.task != "vision_language_generation": + raise ValueError("internvl supports only task=vision_language_generation") + if request.quantization not in {None, "none"} or request.fp32_layers: + raise NotImplementedError( + "InternVL supports only non-quantized uniform-precision builds" ) - else: - config.raw["_decoder_engine_role"] = "prefill" - prefill = model.build_engine( - config, - weights, - max_length, - precision=precision, - quant_ctx=None, - verbose=request.verbose, - parallel_config=parallel, - ) - config.raw["_decoder_engine_role"] = "decode" - decode = model.build_engine( - config, - weights, - max_length, - precision=precision, - quant_ctx=None, - verbose=request.verbose, - parallel_config=parallel, + model_dir = Path(request.model_dir) + config = ModelConfig.from_dir(model_dir) + if str(config.model_type).lower() not in {"internvl_chat", "internvl3", "internvl"}: + raise ValueError(f"InternVL does not support model_type={config.model_type!r}") + precision = str(request.precision).lower() + max_length = int(request.max_sequence_length or min(config.max_position_embeddings, 256)) + parallel = ParallelConfig(tp_size=int(request.tensor_parallel_size)) + parallel.validate() + config.raw["_model_dir"] = str(model_dir) + model = _InternVLModel() + weights = model.load_weights(str(model_dir), config) + writer.set_header(family="internvl", task=request.task, backend=request.backend) + if parallel.enabled: + for rank in range(parallel.tp_size): + writer.add_bytes( + f"engine.rank{rank}.plan", + model.build_engine( + config, + weights, + max_length, + precision=precision, + quant_ctx=None, + verbose=request.verbose, + parallel_config=parallel.for_rank(rank), + ), + ) + else: + config.raw["_decoder_engine_role"] = "prefill" + prefill = model.build_engine( + config, + weights, + max_length, + precision=precision, + quant_ctx=None, + verbose=request.verbose, + parallel_config=parallel, + ) + config.raw["_decoder_engine_role"] = "decode" + decode = model.build_engine( + config, + weights, + max_length, + precision=precision, + quant_ctx=None, + verbose=request.verbose, + parallel_config=parallel, + ) + config.raw.pop("_decoder_engine_role", None) + writer.add_bytes("engine.plan", decode) + writer.add_bytes("prefill.plan", prefill) + vision = model.build_vision_engine( + str(model_dir), config, weights, precision=precision, verbose=request.verbose ) - config.raw.pop("_decoder_engine_role", None) - writer.add_bytes("engine.plan", decode) - writer.add_bytes("prefill.plan", prefill) - vision = model.build_vision_engine( - str(model_dir), config, weights, precision=precision, verbose=request.verbose - ) - if vision is None: - raise RuntimeError("InternVL vision build returned no engine") - vl = model.get_vl_config(config) or {} - runtime = { - "tensor_parallel_size": parallel.tp_size, - "num_layers": config.num_hidden_layers, - "max_cache_length": max_length, - "vocab_size": config.vocab_size, - "id_bos": config.bos_token_id, - "id_eos": config.eos_token_id, - "image_token_id": int(vl.get("image_token_id", -1)), - "vision_output_dim": int(vl.get("vision_output_dim", config.hidden_size)), - "prefill_max_length": int(vl.get("prefill_max_length", max_length)), - "io_map": { - "cache_k_pattern": "cache_k_{i}", - "cache_v_pattern": "cache_v_{i}", - "present_k_pattern": "present_k_{i}", - "present_v_pattern": "present_v_{i}", - }, - } - runtime.update(vl) - writer.add_bytes("vision.plan", vision) - runtime.update(_tokenizer_runtime_contract(model_dir)) - writer.add_json("runtime.json", runtime) - for filename in ( - "tokenizer.json", - "tokenizer_config.json", - "chat_template.jinja", - "vocab.json", - "merges.txt", - "special_tokens_map.json", - "tokenizer.model", - ): - path = model_dir / filename - if path.is_file(): - writer.add_bytes(filename, path.read_bytes()) + if vision is None: + raise RuntimeError("InternVL vision build returned no engine") + vl = model.get_vl_config(config) or {} + runtime = { + "tensor_parallel_size": parallel.tp_size, + "num_layers": config.num_hidden_layers, + "max_cache_length": max_length, + "vocab_size": config.vocab_size, + "id_bos": config.bos_token_id, + "id_eos": config.eos_token_id, + "image_token_id": int(vl.get("image_token_id", -1)), + "vision_output_dim": int(vl.get("vision_output_dim", config.hidden_size)), + "prefill_max_length": int(vl.get("prefill_max_length", max_length)), + "io_map": { + "cache_k_pattern": "cache_k_{i}", + "cache_v_pattern": "cache_v_{i}", + "present_k_pattern": "present_k_{i}", + "present_v_pattern": "present_v_{i}", + }, + } + runtime.update(vl) + writer.add_bytes("vision.plan", vision) + runtime.update(_tokenizer_runtime_contract(model_dir)) + writer.add_json("runtime.json", runtime) + for filename in ( + "tokenizer.json", + "tokenizer_config.json", + "chat_template.jinja", + "vocab.json", + "merges.txt", + "special_tokens_map.json", + "tokenizer.model", + ): + path = model_dir / filename + if path.is_file(): + writer.add_bytes(filename, path.read_bytes()) + + dispatch_build(request, writer, _build_native) diff --git a/families/internvl/runtime/CMakeLists.txt b/families/internvl/runtime/CMakeLists.txt index e20248fa6c..cde6f40ca8 100644 --- a/families/internvl/runtime/CMakeLists.txt +++ b/families/internvl/runtime/CMakeLists.txt @@ -46,3 +46,5 @@ set_target_properties(trtmc_model_internvl PROPERTIES install(TARGETS trtmc_model_internvl LIBRARY DESTINATION ${CMAKE_INSTALL_LIBDIR} ) + +include(edge_llm/Adapter.cmake) diff --git a/families/internvl/runtime/edge_llm/Adapter.cmake b/families/internvl/runtime/edge_llm/Adapter.cmake new file mode 100644 index 0000000000..0deed0002a --- /dev/null +++ b/families/internvl/runtime/edge_llm/Adapter.cmake @@ -0,0 +1,44 @@ +# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-License-Identifier: Apache-2.0 + +# Complete-network offload is family-owned and absent from native-only builds. +if(TARGET EdgeLLM::Core) + if(NOT TARGET EdgeLLM::Plugin) + message(FATAL_ERROR "InternVL Edge adapter requires the complete EdgeLLM package (Core and Plugin)") + endif() + target_sources(trtmc_model_internvl PRIVATE + "${CMAKE_CURRENT_LIST_DIR}/adapter.cpp" + "${CMAKE_CURRENT_LIST_DIR}/device_link.cu" + ) + target_compile_definitions(trtmc_model_internvl PRIVATE TRTMC_HAS_EDGE_LLM=1) + target_link_libraries(trtmc_model_internvl PRIVATE EdgeLLM::Core) + set_target_properties(trtmc_model_internvl PROPERTIES + CUDA_ARCHITECTURES "${EdgeLLM_CUDA_ARCHITECTURE}" + CUDA_SEPARABLE_COMPILATION ON + CUDA_RESOLVE_DEVICE_SYMBOLS ON + ) +endif() + + +if(TARGET EdgeLLM::Core) + add_custom_command(TARGET trtmc_model_internvl POST_BUILD + COMMAND ${CMAKE_COMMAND} -E copy_if_different + $ $ + VERBATIM + ) +endif() + +# Edge 0.10.1 has no visual-feature dump CLI. This helper serves the existing +# InternVL image-health E2E and is not installed as a product executable. +if(TRTMC_BUILD_TESTS AND TARGET EdgeLLM::Core) + add_executable(internvl_edge_vision_features + "${CMAKE_CURRENT_LIST_DIR}/vision_features.cpp" + "${CMAKE_CURRENT_LIST_DIR}/device_link.cu" + ) + target_link_libraries(internvl_edge_vision_features PRIVATE EdgeLLM::Core ${CMAKE_DL_LIBS}) + set_target_properties(internvl_edge_vision_features PROPERTIES + CUDA_ARCHITECTURES "${EdgeLLM_CUDA_ARCHITECTURE}" + CUDA_SEPARABLE_COMPILATION ON + CUDA_RESOLVE_DEVICE_SYMBOLS ON + ) +endif() diff --git a/families/internvl/runtime/edge_llm/adapter.cpp b/families/internvl/runtime/edge_llm/adapter.cpp new file mode 100644 index 0000000000..d606a13727 --- /dev/null +++ b/families/internvl/runtime/edge_llm/adapter.cpp @@ -0,0 +1,256 @@ +/* + * SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. + * SPDX-License-Identifier: Apache-2.0 + */ +#include "families/internvl/runtime/edge_llm/adapter.h" + +#include "families/internvl/runtime/edge_llm/request.h" + +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include + +namespace trtmc::internvl::edge_llm { +namespace { +namespace fs = std::filesystem; + +/// Turn CUDA failures into caller-visible load or inference errors. +void check_cuda(cudaError_t result) { + if (result != cudaSuccess) + throw std::runtime_error(std::string("InternVL Edge CUDA error: ") + + cudaGetErrorString(result)); +} + +/// Reject an engine built for a different local GPU or CUDA/TensorRT runtime. +void validate_target(const nlohmann::json& target) { + utsname host{}; + if (uname(&host) != 0) + throw std::runtime_error("Cannot identify InternVL Edge runtime host"); + std::ifstream release("/etc/os-release"); + std::string line, os_version; + while (std::getline(release, line)) { + if (line.rfind("VERSION_ID=", 0) == 0) { + os_version = line.substr(11); + if (os_version.size() >= 2 && os_version.front() == char(34) && + os_version.back() == char(34)) + os_version = os_version.substr(1, os_version.size() - 2); + } + } + int device = 0, cuda_version = 0; + check_cuda(cudaGetDevice(&device)); + check_cuda(cudaRuntimeGetVersion(&cuda_version)); + cudaDeviceProp gpu{}; + check_cuda(cudaGetDeviceProperties(&gpu, device)); + const int trt_version = getInferLibVersion(); + const std::string trt = + std::to_string(trt_version / 10000) + "." + std::to_string((trt_version % 10000) / 100) + + "." + std::to_string(trt_version % 100) + "." + std::to_string(getInferLibBuildVersion()); + const std::string cuda = + std::to_string(cuda_version / 1000) + "." + std::to_string((cuda_version % 1000) / 10); + if (target.at("os") != "linux" || target.at("os_version") != os_version || + target.at("arch") != host.machine || target.at("sm") != gpu.major * 10 + gpu.minor || + target.at("cuda_version") != cuda || target.at("tensorrt_version") != trt) + throw std::runtime_error( + "InternVL Edge bundle requires its build GPU and CUDA/TensorRT stack"); +} + +/// Own extracted engine/checkpoint files until after the Edge runtime is destroyed. +class Artifacts { + public: + explicit Artifacts(const BundleReader& bundle, const nlohmann::json& marker) { + std::set names; + for (const auto& entry : marker.at("artifacts")) { + const auto name = entry.get(); + if (!safe_artifact_path(name) || !names.insert(name).second || + !bundle.find_section(name)) + throw std::runtime_error("Invalid InternVL Edge artifact: " + name); + } + for (const auto* required : + {"edge_llm/engine/visual/visual.engine", "edge_llm/engine/visual/config.json", + "edge_llm/engine/llm.engine", "edge_llm/engine/config.json", + "edge_llm/engine/tokenizer.json", "edge_llm/engine/tokenizer_config.json", + "edge_llm/engine/processed_chat_template.json", "edge_llm/checkpoint/config.json"}) + if (!names.count(required) || bundle.find_section(required)->length == 0) + throw std::runtime_error(std::string("Required InternVL Edge artifact missing: ") + + required); + std::string pattern = (fs::temp_directory_path() / "trtmc-internvl-edge-XXXXXX").string(); + if (!mkdtemp(pattern.data())) + throw std::runtime_error("Cannot create InternVL Edge artifact directory"); + root_ = pattern; + try { + for (const auto& name : names) { + const auto destination = root_ / name; + fs::create_directories(destination.parent_path()); + std::ofstream output(destination, std::ios::binary); + bundle.copy_section(name, output); + output.close(); + if (!output) + throw std::runtime_error("Cannot extract InternVL Edge artifact: " + name); + } + } catch (...) { + cleanup(); + throw; + } + } + ~Artifacts() { cleanup(); } + Artifacts(const Artifacts&) = delete; + Artifacts& operator=(const Artifacts&) = delete; + std::string engine() const { return (root_ / "edge_llm/engine").string(); } + std::string checkpoint() const { return (root_ / "edge_llm/checkpoint").string(); } + + private: + void cleanup() noexcept { + std::error_code ignored; + fs::remove_all(root_, ignored); + } + fs::path root_; +}; + +/// Close the plugin handle after runtime destruction; registrations remain mapped. +struct CloseLibrary { + void operator()(void* handle) const noexcept { + if (handle) + dlclose(handle); + } +}; + +/// Initialize the CMake-installed adjacent plugin without process-global environment mutation. +std::unique_ptr load_plugin() { + Dl_info location{}; + if (!dladdr(reinterpret_cast(&create), &location) || !location.dli_fname) + throw std::runtime_error("Cannot locate InternVL family library"); + const auto path = + fs::absolute(location.dli_fname).parent_path() / "libNvInfer_edgellm_plugin.so"; + std::unique_ptr plugin( + dlopen(path.c_str(), RTLD_NOW | RTLD_GLOBAL | RTLD_NODELETE)); + if (!plugin) + throw std::runtime_error("Cannot load CMake-installed Edge plugin: " + + std::string(dlerror())); + using Initialize = bool (*)(void*, const char*); + auto initialize = reinterpret_cast(dlsym(plugin.get(), "initEdgellmPlugins")); + if (!initialize || !initialize(static_cast(&trt_edgellm::gLogger), "")) + throw std::runtime_error("Cannot initialize InternVL Edge plugin"); + return plugin; +} + +/// Stream ownership is independent of construction success and outlives the Edge instance. +class Stream { + public: + Stream() { check_cuda(cudaStreamCreateWithFlags(&value_, cudaStreamNonBlocking)); } + ~Stream() { cudaStreamDestroy(value_); } + Stream(const Stream&) = delete; + Stream& operator=(const Stream&) = delete; + cudaStream_t get() const { return value_; } + + private: + cudaStream_t value_{nullptr}; +}; + +/// Validate the admitted raw-tokenizer policy from immutable bundle bytes. +void validate_bundle_tokenizer(const BundleReader& bundle) { + const auto bytes = bundle.read_section("edge_llm/engine/tokenizer.json"); + const auto config = bundle.read_section("edge_llm/engine/tokenizer_config.json"); + validate_raw_tokenizer(nlohmann::json::parse(bytes.begin(), bytes.end()), + nlohmann::json::parse(config.begin(), config.end())); +} + +/// Thin persistent Edge API adapter; serialization prevents concurrent use of Edge request state. +class EdgeTask final : public ITextGeneration, public IVisionLanguageGeneration { + public: + EdgeTask(const BundleReader& bundle, const nlohmann::json& marker) + : artifacts_(bundle, marker), plugin_(load_plugin()), + runtime_(artifacts_.engine(), artifacts_.engine(), + std::unordered_map{}, stream_.get(), + trt_edgellm::rt::ContextCacheConfig{}, artifacts_.checkpoint()), + capacity_(marker.at("max_sequence_length").get()), + input_limit_(marker.at("max_input_length").get()) { + validate_bundle_tokenizer(bundle); + } + + const char* task() const noexcept override { return IVisionLanguageGeneration::kTask; } + std::int32_t default_max_new_tokens() const override { + // Some profiles allow inputs up to the entire KV capacity. Reserve one + // output token there; a prompt occupying every slot still fails validation. + return std::max(1, capacity_ - input_limit_); + } + + /// Drain work from failed requests before destroying the runtime and its weight buffers. + ~EdgeTask() override { cudaStreamSynchronize(stream_.get()); } + + /// No-image calls retain native raw semantics, including ignored chat/thinking flags. + TextResult generate(const std::string& prompt, const TextGenerationConfig& config) override { + return generate(prompt, nullptr, 0, 0, config); + } + + /// Invoke the full Edge visual and LLM runtime once; never retry inference natively. + TextResult generate(const std::string& prompt, const float* pixels, std::int32_t height, + std::int32_t width, const TextGenerationConfig& config) override { + auto bytes = image_bytes(pixels, height, width); + auto request = make_request(prompt, config, !bytes.empty()); + if (!bytes.empty()) { + trt_edgellm::rt::Tensor tensor({1, height, width, 3}, trt_edgellm::rt::DeviceType::kCPU, + nvinfer1::DataType::kUINT8); + std::memcpy(tensor.rawPointer(), bytes.data(), bytes.size()); + request.requests.front().imageBuffers.emplace_back(std::move(tensor)); + } + const auto requested_budget = request.maxGenerateLength; + std::lock_guard lock(mutex_); + try { + trt_edgellm::rt::LLMGenerationResponse response{}; + if (!runtime_.handleRequest(request, response, stream_.get())) + throw std::runtime_error("InternVL Edge generation failed"); + // Actual media expansion plus ORIGINAL budget, even for early EOS. + validate_response(response, input_limit_, capacity_, requested_budget); + check_cuda(cudaStreamSynchronize(stream_.get())); + return {std::move(response.outputTexts.front()), std::move(response.outputIds.front())}; + } catch (...) { + // Image buffers must outlive outstanding DMA, including a failed request. + cudaStreamSynchronize(stream_.get()); + throw; + } + } + + private: + // Reverse destruction order keeps weights, plugin and stream alive throughout Edge teardown. + Artifacts artifacts_; + std::unique_ptr plugin_; + Stream stream_; + trt_edgellm::rt::LLMInferenceRuntime runtime_; + int capacity_; + int input_limit_; + std::mutex mutex_; +}; +} // namespace + +ITask* create(const BundleReader& bundle) { + const auto bytes = bundle.read_section("edge_llm.json"); + const auto marker = nlohmann::json::parse(bytes.begin(), bytes.end()); + if (marker.at("version") != 1 || marker.at("edge_revision") != kRevision || + marker.at("max_sequence_length").get() <= 1 || + marker.at("max_input_length").get() <= 0 || + marker.at("max_input_length").get() > marker.at("max_sequence_length").get() || + marker.at("max_batch_size") != 1 || marker.at("precision") != "fp16" || + marker.at("weight_format") != "fp16" || + marker.at("component_weight_formats").at("llm") != marker.at("weight_format") || + marker.at("component_weight_formats").at("visual") != "fp16" || + marker.at("visual_image_tokens") != 256 || !marker.at("artifacts").is_array()) + throw std::runtime_error("Invalid InternVL Edge bundle contract"); + validate_target(marker.at("target")); + return new EdgeTask(bundle, marker); +} + +} // namespace trtmc::internvl::edge_llm diff --git a/families/internvl/runtime/edge_llm/adapter.h b/families/internvl/runtime/edge_llm/adapter.h new file mode 100644 index 0000000000..7f60b62209 --- /dev/null +++ b/families/internvl/runtime/edge_llm/adapter.h @@ -0,0 +1,15 @@ +/* + * SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. + * SPDX-License-Identifier: Apache-2.0 + */ +#pragma once + +#include "trtmc/bundle.h" +#include "trtmc/task.h" + +namespace trtmc::internvl::edge_llm { + +/// Create a persistent Edge task from a self-contained bundle; throws on load failure. +ITask* create(const BundleReader& bundle); + +} // namespace trtmc::internvl::edge_llm diff --git a/families/internvl/runtime/edge_llm/contract.h b/families/internvl/runtime/edge_llm/contract.h new file mode 100644 index 0000000000..1782d67d25 --- /dev/null +++ b/families/internvl/runtime/edge_llm/contract.h @@ -0,0 +1,89 @@ +/* + * SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. + * SPDX-License-Identifier: Apache-2.0 + */ +#pragma once + +#include "trtmc/task.h" + +#include +#include +#include +#include +#include +#include +#include + +namespace trtmc::internvl::edge_llm { + +inline constexpr const char* kRevision = "e8b29522938901f6df19ebeedd4b69bc8edbcd97"; + +/// Return whether an artifact is a normalized file below one of the two Edge roots. +inline bool safe_artifact_path(const std::string& name) { + if (name.find('\\') != std::string::npos || name.find('\0') != std::string::npos) + return false; + const std::filesystem::path path(name); + if (path.is_absolute() || path.filename().empty()) + return false; + for (const auto& part : path) + if (part == "." || part == "..") + return false; + return path.generic_string() == name && + (name.rfind("edge_llm/engine/", 0) == 0 || name.rfind("edge_llm/checkpoint/", 0) == 0); +} + +/// Reject invalid sampling settings and controls with no equivalent Edge request API. +inline void validate_generation(const TextGenerationConfig& c) { + if (!std::isfinite(c.temperature) || c.temperature < 0 || !std::isfinite(c.top_p) || + c.top_p <= 0 || c.top_p > 1 || c.top_k < 0) + throw std::invalid_argument("Invalid InternVL Edge sampling parameters"); + if (c.min_p != 0 || c.seed != -1 || c.eos_token_id != -1 || c.repetition_penalty != 1 || + !c.lora_adapter_id.empty() || c.stop_on_boxed_answer || + (c.text_generation_mode != "auto" && c.text_generation_mode != "autoregressive") || + c.source_language_token_id != -1 || c.forced_bos_token_id != -1 || c.guidance_scale != -1 || + c.cfg_scale != -1 || c.num_steps != -1 || c.sde_gamma != -1 || !c.initial_latents.empty() || + !c.condition_latents.empty() || !c.condition_mask.empty() || !c.sampling_steps.empty() || + !c.sde_noises.empty() || c.block_length != 0 || c.confidence_threshold != -1) + throw std::invalid_argument( + "Requested generation controls are unsupported by InternVL Edge"); +} + +/// Enforce prompt and total capacity without allowing Edge to silently clip generation. +inline void validate_capacity(int prompt_tokens, int input_limit, int capacity, + std::int64_t generated_tokens) { + if (prompt_tokens <= 0 || prompt_tokens > input_limit || generated_tokens <= 0 || + generated_tokens > static_cast(capacity) - prompt_tokens) + throw std::invalid_argument("InternVL Edge prompt and generation exceed bundle capacity"); +} + +/// Validate external normalized-float HWC images before multiplying dimensions or allocating. +inline std::size_t image_elements(const float* pixels, std::int32_t height, std::int32_t width) { + if (!pixels && height == 0 && width == 0) + return 0; + if (!pixels || height <= 0 || width <= 0) + throw std::invalid_argument("InternVL Edge image requires pixels and positive dimensions"); + const auto h = static_cast(height); + const auto w = static_cast(width); + const auto maximum = + std::min(std::numeric_limits::max() / sizeof(float), + static_cast(std::numeric_limits::max())); + if (w > maximum / 3 || h > maximum / (w * 3)) + throw std::invalid_argument("InternVL Edge image dimensions overflow"); + return h * w * 3; +} + +/// Preserve native float-to-byte rounding and clamping; preprocessing belongs to Edge. +inline std::vector image_bytes(const float* pixels, std::int32_t height, + std::int32_t width) { + const auto size = image_elements(pixels, height, width); + std::vector result(size); + for (std::size_t i = 0; i < size; ++i) { + if (!std::isfinite(pixels[i])) + throw std::invalid_argument("InternVL Edge image contains non-finite pixels"); + result[i] = + static_cast(std::round(std::clamp(pixels[i], 0.0F, 1.0F) * 255.0F)); + } + return result; +} + +} // namespace trtmc::internvl::edge_llm diff --git a/families/internvl/runtime/edge_llm/device_link.cu b/families/internvl/runtime/edge_llm/device_link.cu new file mode 100644 index 0000000000..1ea4768d60 --- /dev/null +++ b/families/internvl/runtime/edge_llm/device_link.cu @@ -0,0 +1,5 @@ +/* + * SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. + * SPDX-License-Identifier: Apache-2.0 + */ +// Enable the final CUDA device-link step for Edge's static runtime dependencies. diff --git a/families/internvl/runtime/edge_llm/request.h b/families/internvl/runtime/edge_llm/request.h new file mode 100644 index 0000000000..927f47ffb4 --- /dev/null +++ b/families/internvl/runtime/edge_llm/request.h @@ -0,0 +1,66 @@ +/* + * SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. + * SPDX-License-Identifier: Apache-2.0 + */ +#pragma once + +#include "families/internvl/runtime/edge_llm/contract.h" + +#include +#include +#include + +namespace trtmc::internvl::edge_llm { + +/// ByteLevel postprocessing only changes offsets; reject unqualified token-ID framing. +inline void validate_raw_tokenizer(const nlohmann::json& tokenizer, + const nlohmann::json& config = nlohmann::json::object()) { + if (config.value("add_bos_token", false) || config.value("add_eos_token", false)) + throw std::invalid_argument("Unsupported InternVL Edge raw BOS/EOS tokenizer flags"); + if (!tokenizer.contains("post_processor") || !tokenizer.at("post_processor").is_object() || + tokenizer.at("post_processor").value("type", "") != "ByteLevel") + throw std::invalid_argument("Unsupported InternVL Edge raw tokenizer postprocessor"); +} + +/// Validate complete responses before exposing any partial output, even early EOS. +inline void validate_response(const trt_edgellm::rt::LLMGenerationResponse& response, + int input_limit, int capacity, std::int64_t requested_budget) { + if (response.outputIds.size() != 1 || response.outputTexts.size() != 1 || + response.inputTokenCounts.size() != 1 || response.finishReasons.size() != 1 || + response.outputIds.front().empty() || requested_budget <= 0 || + response.outputIds.front().size() > static_cast(requested_budget)) + throw std::runtime_error("InternVL Edge returned an invalid generation response"); + validate_capacity(response.inputTokenCounts.front(), input_limit, capacity, requested_budget); + const auto reason = response.finishReasons.front(); + if (reason != trt_edgellm::rt::FinishReason::kEndId && + reason != trt_edgellm::rt::FinishReason::kLength) + throw std::runtime_error("InternVL Edge generation did not complete successfully"); + if (reason == trt_edgellm::rt::FinishReason::kLength && + response.outputIds.front().size() != static_cast(requested_budget)) + throw std::runtime_error("InternVL Edge clipped the original generation budget"); +} + +/// Preserve native raw-no-image and fixed-system image mode, independent of ignored chat flags. +inline trt_edgellm::rt::LLMGenerationRequest +make_request(const std::string& prompt, const TextGenerationConfig& config, bool has_image) { + validate_generation(config); + trt_edgellm::rt::LLMGenerationRequest request{}; + request.requests.resize(1); + auto& messages = request.requests.front().messages; + if (has_image) { + messages.push_back({"system", {{"text", "You are a helpful assistant."}}}); + messages.push_back({"user", {{"image", ""}, {"text", prompt}}}); + } else { + messages.push_back({"user", {{"text", prompt}}}); + } + request.applyChatTemplate = has_image; + request.enableThinking = false; + request.temperature = config.temperature; + request.topK = config.top_k; + request.topP = config.top_p; + // Native accessor returns context, but explicit nonpositive request is the sentinel128. + request.maxGenerateLength = config.max_new_tokens > 0 ? config.max_new_tokens : 128; + return request; +} + +} // namespace trtmc::internvl::edge_llm diff --git a/families/internvl/runtime/edge_llm/vision_features.cpp b/families/internvl/runtime/edge_llm/vision_features.cpp new file mode 100644 index 0000000000..cba1bccb63 --- /dev/null +++ b/families/internvl/runtime/edge_llm/vision_features.cpp @@ -0,0 +1,117 @@ +/* + * SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. + * SPDX-License-Identifier: Apache-2.0 + */ +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include + +namespace { +struct Stream { + cudaStream_t value{nullptr}; + Stream() { + const auto status = cudaStreamCreateWithFlags(&value, cudaStreamNonBlocking); + if (status != cudaSuccess) + throw std::runtime_error(cudaGetErrorString(status)); + } + ~Stream() { cudaStreamDestroy(value); } +}; +struct CloseLibrary { + void operator()(void* handle) const noexcept { + if (handle) + dlclose(handle); + } +}; +int64_t positive_integer(const char* text) { + std::size_t end = 0; + const std::string input(text); + const auto value = std::stoll(input, &end); + if (end != input.size() || value <= 0 || value > std::numeric_limits::max()) + throw std::invalid_argument("Expected a positive int32 dimension"); + return value; +} +} // namespace + +// Test-only: execute Edge's actual visual engine, without an LLM or Model Connect adapter. +int main(int argc, char** argv) { + try { + if (argc != 9) + throw std::invalid_argument("Usage: internvl_edge_vision_features ENGINE CHECKPOINT " + "PLUGIN RGB H W CAPACITY OUTPUT"); + const auto height = positive_integer(argv[5]); + const auto width = positive_integer(argv[6]); + const auto capacity = positive_integer(argv[7]); + const auto image_bytes = static_cast(height) * width * 3; + if (std::filesystem::file_size(argv[4]) != image_bytes) + throw std::invalid_argument("RGB fixture size does not match H W"); + std::unique_ptr plugin( + dlopen(argv[3], RTLD_NOW | RTLD_GLOBAL | RTLD_NODELETE)); + if (!plugin) + throw std::runtime_error(std::string("Cannot load Edge plugin: ") + dlerror()); + using InitializeFn = bool (*)(void*, const char*); + auto initialize = reinterpret_cast(dlsym(plugin.get(), "initEdgellmPlugins")); + if (!initialize || !initialize(static_cast(&trt_edgellm::gLogger), "")) + throw std::runtime_error("Cannot initialize Edge plugin"); + Stream stream; + trt_edgellm::rt::LLMGenerationRequest request{}; + request.requests.resize(1); + trt_edgellm::rt::Tensor image({1, height, width, 3}, trt_edgellm::rt::DeviceType::kCPU, + nvinfer1::DataType::kUINT8); + std::ifstream input(argv[4], std::ios::binary); + if (!input.read(static_cast(image.rawPointer()), image_bytes)) + throw std::runtime_error("Cannot read RGB fixture"); + request.requests.front().imageBuffers.emplace_back(std::move(image)); + // Context memory must outlive the runner and all enqueued work. + trt_edgellm::rt::Tensor context_memory; + auto runner = trt_edgellm::rt::MultimodalRunner::create( + (std::filesystem::path(argv[1]) / "visual").string(), 1, capacity, stream.value, + argv[2]); + if (!runner) + throw std::runtime_error("Cannot create Edge visual runner"); + context_memory = + trt_edgellm::rt::Tensor({runner->getRequiredContextMemorySize()}, + trt_edgellm::rt::DeviceType::kGPU, nvinfer1::DataType::kUINT8); + try { + if (!runner->setContextMemory(context_memory)) + throw std::runtime_error("Cannot set visual context memory"); + std::vector> unused_ids; + if (!runner->preprocess(request, unused_ids, nullptr, std::nullopt, stream.value, + true) || + !runner->infer(stream.value)) + throw std::runtime_error("Edge visual inference failed"); + const auto& output = runner->getOutputEmbedding(); + if (output.getDataType() != nvinfer1::DataType::kHALF || + output.getShape().volume() <= 0) + throw std::runtime_error("Invalid Edge feature shape or dtype"); + std::vector features(output.getShape().volume()); + const auto bytes = features.size() * sizeof(uint16_t); + const auto copied = cudaMemcpyAsync(features.data(), output.rawPointer(), bytes, + cudaMemcpyDeviceToHost, stream.value); + const auto synced = cudaStreamSynchronize(stream.value); + if (copied != cudaSuccess || synced != cudaSuccess) + throw std::runtime_error("Cannot read Edge vision features"); + std::ofstream file(argv[8], std::ios::binary); + file.write(reinterpret_cast(features.data()), bytes); + file.close(); + if (!file) + throw std::runtime_error("Cannot write Edge vision features"); + } catch (...) { + cudaStreamSynchronize(stream.value); + throw; + } + return 0; + } catch (const std::exception& error) { + std::cerr << error.what() << '\n'; + return 1; + } +} diff --git a/families/internvl/runtime/plugin.cpp b/families/internvl/runtime/plugin.cpp index 15a328d5b2..65d753c241 100644 --- a/families/internvl/runtime/plugin.cpp +++ b/families/internvl/runtime/plugin.cpp @@ -3,6 +3,10 @@ * SPDX-License-Identifier: Apache-2.0 */ +#ifdef TRTMC_HAS_EDGE_LLM +#include "families/internvl/runtime/edge_llm/adapter.h" +#endif + #include "families/internvl/runtime/cuda_stream.h" #include "families/internvl/runtime/distributed_runtime.h" #include "families/internvl/runtime/pipeline.h" @@ -67,6 +71,13 @@ extern "C" trtmc::ITask* trtmc_create_family(const trtmc::FamilyContext& context if (context.kv_cache_size_bytes != 0) throw std::invalid_argument("internvl does not support --kv-cache-size"); using namespace trtmc; + if (context.reader.find_section("edge_llm.json")) { +#ifdef TRTMC_HAS_EDGE_LLM + return internvl::edge_llm::create(context.reader); +#else + throw std::runtime_error("InternVL Edge bundle requires the optional Edge CMake package"); +#endif + } const std::string runtime_text = internvl_factory::section_text(context.reader, "runtime.json"); const auto config = nlohmann::json::parse(runtime_text); if (!config.is_object()) diff --git a/families/internvl/tests/test_tp_contract.py b/families/internvl/tests/test_tp_contract.py index 8f4756d057..8ad2ae7a3b 100644 --- a/families/internvl/tests/test_tp_contract.py +++ b/families/internvl/tests/test_tp_contract.py @@ -3,6 +3,8 @@ from __future__ import annotations +import pytest + from types import SimpleNamespace import numpy as np @@ -93,6 +95,7 @@ def add_json(self, name, value): monkeypatch.setattr(model, "_tokenizer_runtime_contract", lambda _path: {}) writer = Writer() request = SimpleNamespace( + family="internvl", output_path=tmp_path / "model.bundle", graph_transform=None, model_dir=tmp_path, backend="trt", dynamic_kv_cache=False, @@ -149,3 +152,179 @@ def apply_chat_template(messages, **kwargs): prompt = _official_prompt(Processor(), user_prompt) assert prompt.count(user_prompt) == 1 assert "" in prompt + + +def test_ordinary_cli_keeps_edge_selection_in_the_family(tmp_path, monkeypatch): + import json + from tensorrt_model_connect import family_cli as build_cli + from families.internvl.edge_llm import dispatch + + source = tmp_path / "target" + source.mkdir() + (source / "config.json").write_text(json.dumps({"model_type": "internvl"})) + output = tmp_path / "model.bundle" + seen = [] + + def select(request, writer, native): + assert callable(native) + assert request.family == "internvl" + assert request.task == "vision_language_generation" + seen.append(request) + writer.set_header(family=request.family, task=request.task, backend=request.backend) + writer.add_json("edge-test.json", {"family": request.family}) + + monkeypatch.setattr(dispatch, "build", select) + assert build_cli.main(["internvl", "build", str(source), "-o", str(output)]) == 0 + assert len(seen) == 1 + assert output.is_file() + + +def test_internvl_does_not_register_unowned_companion_options(): + from tensorrt_model_connect.model_support import ModelMetadata + from families.internvl.support import describe + + support = describe(ModelMetadata({"model_type": "internvl"}, {})) + assert support is not None + from tensorrt_model_connect.family_cli import load_family_cli + declaration = load_family_cli("internvl") + flags = {flag for argument in declaration["commands"][0]["arguments"] + for flag in argument.get("flags", [])} + assert "--execution-variant" not in flags + assert "--companion" not in flags + + +@pytest.mark.parametrize("mode", ["absent", "success", "corrupt", "failure", "cancel", "device_failure"]) +def test_edge_optional_package_and_output_local_staging(tmp_path, monkeypatch, caplog, mode): + import json + from tensorrt_model_connect.build import BuildRequest + from families.internvl.edge_llm import builder, dispatch + + source = tmp_path / "target" + source.mkdir() + (source / "config.json").write_text(json.dumps({ + "max_position_embeddings": 4096, "hidden_size": 896, + })) + prefix = tmp_path / "package" + manifest = prefix / "share/trtmc/edge-llm.json" + if mode != "absent": + manifest.parent.mkdir(parents=True) + manifest.write_text("{" if mode == "corrupt" else "{}") + monkeypatch.setattr(builder, "cmake_prefixes", lambda: [prefix]) + monkeypatch.setattr(dispatch, "candidate", lambda *_: True) + for name in ("mapped_request", "request_matches", "platform_matches"): + if hasattr(dispatch, name): + monkeypatch.setattr(dispatch, name, lambda *_: True) + if hasattr(dispatch, "source_quantization"): + monkeypatch.setattr(dispatch, "source_quantization", lambda *_: "fp16") + if hasattr(builder, "request_weight_format"): + monkeypatch.setattr(builder, "request_weight_format", lambda *_: "fp16") + request = BuildRequest(source, tmp_path / "out", "internvl", "vision_language_generation", "fp16") + writer = object() + target = {"os": "linux", "arch": "x86_64", "sm": 80} + target_calls = [] + + def local_target(): + target_calls.append(True) + if mode == "device_failure": + raise RuntimeError("CUDA discovery failed") + return target + + monkeypatch.setattr(builder, "local_target", local_target) + stages, native_calls, publications = [], [], [] + + def prepare(original, raw, platform, staging, log): + assert original is request and platform is target + assert staging.parent == request.output_path.parent + assert staging.name.startswith(f".{request.output_path.name}.edge-") + stages.append(staging) + (staging / "large-checkpoint").write_bytes(b"fixture") + if mode == "corrupt": + builder.installed_package(target) + if mode == "failure": + raise FileNotFoundError("installed SDK artifact missing") + if mode == "cancel": + raise KeyboardInterrupt() + return {}, {} + + monkeypatch.setattr(dispatch, "EDGE_DISPATCH", {("linux", "x86_64", 80, "fp16"): prepare}) + monkeypatch.setattr(builder, "publish", lambda *args: publications.append(args)) + def native(*args): + native_calls.append(args) + if mode == "cancel": + with pytest.raises(KeyboardInterrupt): + dispatch.build(request, writer, native) + else: + dispatch.build(request, writer, native) + assert all(not path.exists() for path in stages) + assert native_calls == ([(request, writer)] if mode in { + "absent", "corrupt", "failure", "device_failure", + } else []) + assert len(publications) == (1 if mode == "success" else 0) + assert len(target_calls) == (0 if mode == "absent" else 1) + logs = list(tmp_path.glob(".out.edge-*.log")) + if mode in {"corrupt", "failure", "device_failure"}: + assert len(logs) == 1 and "Traceback" in logs[0].read_text() + assert "Retrying native once" in caplog.text + else: + assert not logs and "Edge build failed" not in caplog.text + + +@pytest.mark.parametrize("options", [[], ["--precision", "fp16", "--max-sequence-length", "64"]]) +def test_declared_build_matches_legacy_request(tmp_path, monkeypatch, options): + """Owner command preserves ordinary request defaults and explicit controls.""" + import json + from families.internvl import cli as owner + from tensorrt_model_connect import build_cli, family_cli + + source = tmp_path / "checkpoint" + source.mkdir() + (source / "config.json").write_text(json.dumps({"model_type": "internvl"})) + output = tmp_path / "model.bundle" + captured = [] + monkeypatch.setattr(owner, "build_bundle", lambda request, output: captured.append(request)) + monkeypatch.setattr(build_cli, "build", captured.append) + args = [str(source), "-o", str(output), *options] + assert family_cli.main(["internvl", "build", *args]) == 0 + assert build_cli.main(["build", *args, "--family", "internvl"]) == 0 + assert len(captured) == 2 + from dataclasses import fields + assert isinstance(captured[0], owner.BuildRequest) + for field in fields(captured[1]): + assert getattr(captured[0], field.name) == getattr(captured[1], field.name) + from dataclasses import replace + from families.internvl.build_request import coerce_request + assert coerce_request(captured[1]) == captured[0] + assert coerce_request(replace(captured[1], fp32_layers=[])) == captured[0] + with pytest.raises(NotImplementedError, match="fp32_layers"): + coerce_request(replace(captured[1], fp32_layers=[0])) + with pytest.raises(NotImplementedError, match="image_height"): + coerce_request(replace(captured[1], image_height=32)) + from types import SimpleNamespace + with pytest.raises(ValueError, match="unknown"): + coerce_request(SimpleNamespace(**vars(captured[1]), unexpected_option=True)) + assert captured[0].family == "internvl" + assert captured[0].task == "vision_language_generation" + assert captured[0].precision == ("fp16" if options else "fp32") + assert not output.exists() + + +def test_declared_help_is_offline_and_dependency_free(): + """Actual child-process help needs neither a checkpoint nor GPU imports.""" + import subprocess + import sys + + code = """ +import sys +from tensorrt_model_connect.family_cli import main +try: + main(["internvl", "build", "--help"]) +except SystemExit as error: + assert error.code == 0 +else: + raise AssertionError("help did not exit") +assert "families.internvl.cli" not in sys.modules +assert "tensorrt" not in sys.modules +assert "huggingface_hub" not in sys.modules +""" + result = subprocess.run([sys.executable, "-c", code], capture_output=True, text=True, check=True) + assert "trtmc internvl build" in result.stdout diff --git a/families/internvl/tests/test_vision_oracle.py b/families/internvl/tests/test_vision_oracle.py index fe372a8949..1f6b7469df 100644 --- a/families/internvl/tests/test_vision_oracle.py +++ b/families/internvl/tests/test_vision_oracle.py @@ -16,6 +16,7 @@ _bundle_section, _native_pixels, assert_vision_parity, + native_vision_features, ) @@ -49,10 +50,51 @@ def _bundle(path, sections: dict[str, bytes]) -> None: ) -def test_bundle_reader_selects_the_family_vision_plan(tmp_path) -> None: +def test_bundle_reader_selects_the_family_vision_plan(tmp_path, monkeypatch) -> None: bundle = tmp_path / "model.bundle" _bundle(bundle, {"vision.plan": b"VISION", "runtime.json": b"{}"}) assert _bundle_section(bundle, "vision.plan") == b"VISION" + output = tmp_path / "extracted/vision.plan" + _bundle_section(bundle, "vision.plan", output=output) + assert output.read_bytes() == b"VISION" + + image = tmp_path / "image.png" + Image.fromarray(np.full((2, 4, 3), 255, dtype=np.uint8)).save(image) + marker = { + "max_sequence_length": 1024, + "artifacts": [ + "edge_llm/engine/visual/visual.engine", + "edge_llm/checkpoint/model.safetensors", + ], + } + _bundle( + bundle, + { + "edge_llm.json": json.dumps(marker).encode(), + marker["artifacts"][0]: b"VISUAL", + marker["artifacts"][1]: b"WEIGHTS", + }, + ) + monkeypatch.setenv("TRTMC_RUNTIME_ROOT", str(tmp_path / "runtime")) + + def read_features(command, *, check, timeout): + from pathlib import Path + + assert check and timeout == 1800 + assert command[0].endswith("/families/internvl/internvl_edge_vision_features") + assert (Path(command[1]) / "visual/visual.engine").read_bytes() == b"VISUAL" + assert (Path(command[2]) / "model.safetensors").read_bytes() == b"WEIGHTS" + assert command[3].endswith("/libNvInfer_edgellm_plugin.so") + assert Path(command[4]).read_bytes() == bytes([255]) * 24 + assert command[5:8] == ["2", "4", "1024"] + np.asarray([1.0, 2.0], dtype=np.float16).tofile(command[8]) + + monkeypatch.setattr("families.internvl.tests.vision_oracle.subprocess.run", read_features) + assert np.array_equal(native_vision_features(bundle, image), [1.0, 2.0]) + marker["artifacts"] = ["edge_llm/checkpoint/../../escaped"] + _bundle(bundle, {"edge_llm.json": json.dumps(marker).encode()}) + with pytest.raises(ValueError, match="Unsafe Edge artifact"): + native_vision_features(bundle, image) def test_native_pixels_follow_the_family_runtime_contract(tmp_path) -> None: diff --git a/families/internvl/tests/vision_oracle.py b/families/internvl/tests/vision_oracle.py index c5cce57ed3..bbbd652b6d 100644 --- a/families/internvl/tests/vision_oracle.py +++ b/families/internvl/tests/vision_oracle.py @@ -4,8 +4,11 @@ from __future__ import annotations import json +import os import struct -from pathlib import Path +import subprocess +import tempfile +from pathlib import Path, PurePosixPath import numpy as np @@ -13,18 +16,76 @@ VISION_FEATURE_COSINE = 0.5 -def _bundle_section(bundle: Path, name: str) -> bytes: +def _bundle_section(bundle: Path, name: str, *, output: Path | None = None) -> bytes: + """Read metadata or stream a weight section without allocating the whole checkpoint.""" with bundle.open("rb") as stream: assert stream.read(8) == _BUNDLE_MAGIC encoded_length = stream.read(8) assert len(encoded_length) == 8 header_length = struct.unpack("= 0 and length > 0 + assert 16 + header_length + offset + length <= bundle.stat().st_size + stream.seek(16 + header_length + offset) + if output is None: + data = stream.read(length) + assert len(data) == length + return data + output.parent.mkdir(parents=True, exist_ok=True) + with output.open("wb") as destination: + while length: + chunk = stream.read(min(length, 64 * 1024)) + assert chunk + destination.write(chunk) + length -= len(chunk) + return b"" + + +def _edge_vision_features(bundle: Path, image_path: Path, marker: dict) -> np.ndarray: + """Keep the existing health test on actual features, not generated text.""" + from PIL import Image + + runtime = Path(os.environ["TRTMC_RUNTIME_ROOT"]) + with tempfile.TemporaryDirectory(prefix="internvl-vision-") as temporary: + root = Path(temporary) + for name in marker["artifacts"]: + path = PurePosixPath(name) + if ( + path.is_absolute() + or ".." in path.parts + or str(path) != name + or "\\" in name + or "\0" in name + or not name.startswith(("edge_llm/engine/", "edge_llm/checkpoint/")) + ): + raise ValueError(f"Unsafe Edge artifact: {name}") + if name.startswith(("edge_llm/engine/visual/", "edge_llm/checkpoint/")): + _bundle_section(bundle, name, output=root / name) + with Image.open(image_path) as source: + image = np.asarray(source.convert("RGB"), dtype=np.uint8) + rgb = root / "image.rgb" + rgb.write_bytes(image.tobytes()) + features = root / "features.fp16" + subprocess.run( + [ + str(runtime / "families/internvl/internvl_edge_vision_features"), + str(root / "edge_llm/engine"), + str(root / "edge_llm/checkpoint"), + str(runtime / "libNvInfer_edgellm_plugin.so"), + str(rgb), + str(image.shape[0]), + str(image.shape[1]), + str(marker["max_sequence_length"]), + str(features), + ], + check=True, + timeout=1800, + ) + return np.fromfile(features, dtype=np.float16).astype(np.float32) def _native_pixels(image_path: Path, config: dict) -> np.ndarray: @@ -89,6 +150,12 @@ def _execute_vision_plan(plan: bytes, inputs: dict[str, np.ndarray]) -> np.ndarr def native_vision_features(bundle: Path, image_path: Path) -> np.ndarray: + try: + marker = json.loads(_bundle_section(bundle, "edge_llm.json")) + except KeyError: + marker = None + if marker is not None: + return _edge_vision_features(bundle, image_path, marker) config = json.loads(_bundle_section(bundle, "runtime.json")) pixels = _native_pixels(image_path, config) return _execute_vision_plan(_bundle_section(bundle, "vision.plan"), {"pixel_values": pixels}) diff --git a/website/docs/features/model-families.md b/website/docs/features/model-families.md index f94173e67e..4b8cd3f5f8 100644 --- a/website/docs/features/model-families.md +++ b/website/docs/features/model-families.md @@ -122,3 +122,20 @@ by design so a team can implement, validate, change, and revert one family without modifying another. Shared code is limited to model-agnostic contracts and mechanics described in the [Architecture](../architecture/ai-native-horizontal-scaling.md). + +### Optional InternVL Edge execution + +Use `trtmc internvl build MODEL -o model.bundle` with the owning +family's options. `trtmc internvl build --help` works offline without +a checkpoint or GPU imports. This uses the existing +[family CLI protocol](../extend/family-cli.md), not an extension to the shared parser. + +The InternVL family can offload the recorded original-source InternVL3 +1B/2B/8B HF configurations on native x86 SM80 and 14B on native x86 SM120 to the +pinned Edge-LLM 0.10.1 SDK. These FP16, batch-one, TP-one profiles have historical +local public/direct/HF qualification; this is not catalog-wide or fresh-head CI +qualification. The existing 2B/8B image-health test reads actual Edge visual +features through a small family-owned test helper without changing its gates. +Quantized sources, InternVL3.5, and multi-image public requests are excluded. +See the [family recipe](https://github.com/NVIDIA/TensorRT-Model-Connect/blob/main/families/internvl/edge_llm/README.md) +for exact source revisions, evidence boundaries, and replay requirements.