Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
98 changes: 98 additions & 0 deletions families/internvl/build_request.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,98 @@
# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved.
# SPDX-License-Identifier: Apache-2.0

"""internvl build inputs and strict compatibility for existing Python callers."""

from __future__ import annotations

from dataclasses import dataclass, fields
from pathlib import Path
import re
from typing import ClassVar

from tensorrt_model_connect.graph_transform import GraphTransform


def _validate_id(field: str, value: str) -> None:
if not isinstance(value, str) or re.fullmatch(r"[a-z][a-z0-9_]*", value) is None:
raise ValueError(f"{field} must be a lowercase identifier")


@dataclass(frozen=True)
class BuildRequest:
"""internvl-owned inputs; unsupported legacy controls are read-only defaults."""

model_dir: Path
output_path: Path
family: str
task: str
precision: str
backend: str = "trt"
max_sequence_length: int | None = None
image_height: ClassVar[int | None] = None
image_width: ClassVar[int | None] = None
video_num_frames: ClassVar[int | None] = None
max_batch_size: ClassVar[int] = 1
tensor_parallel_size: int = 1
context_parallel_size: ClassVar[int] = 1
quantization: ClassVar[str | None] = None
fp32_layers: ClassVar[tuple[int, ...]] = ()
dynamic_kv_cache: ClassVar[bool] = False
verbose: bool = False
graph_transform: GraphTransform | None = None

def __post_init__(self) -> None:
if not self.precision:
raise ValueError("precision must be non-empty")
_validate_id("family", self.family)
_validate_id("task", self.task)
if self.backend not in {"trt", "trt_rtx"}:
raise ValueError("backend must be 'trt' or 'trt_rtx'")
if self.max_sequence_length is not None and self.max_sequence_length < 1:
raise ValueError("max_sequence_length must be positive")
for field in ("image_height", "image_width", "video_num_frames"):
value = getattr(self, field)
if value is not None and value < 1:
raise ValueError(f"{field} must be positive")
if self.max_batch_size < 1:
raise ValueError("max_batch_size must be positive")
if self.tensor_parallel_size < 1:
raise ValueError("tensor_parallel_size must be positive")
if self.context_parallel_size < 1:
raise ValueError("context_parallel_size must be positive")
if self.quantization is not None and not self.quantization:
raise ValueError("quantization must be non-empty when provided")
if any(layer < 0 for layer in self.fp32_layers):
raise ValueError("fp32_layers must contain non-negative indices")
if not isinstance(self.dynamic_kv_cache, bool):
raise ValueError("dynamic_kv_cache must be a bool")
if self.graph_transform is not None and not callable(self.graph_transform):
raise ValueError("graph_transform must be callable when provided")


def coerce_request(request: object) -> BuildRequest:
"""Reject unsupported/unknown legacy inputs before converting to owner fields."""
if isinstance(request, BuildRequest):
return request
unsupported = {
"image_height": None,
"image_width": None,
"video_num_frames": None,
"max_batch_size": 1,
"context_parallel_size": 1,
"quantization": None,
"fp32_layers": (),
"dynamic_kv_cache": False,
}
for name, default in unsupported.items():
value = getattr(request, name, default)
if name == "quantization" and value == "none":
continue
if name == "fp32_layers" and isinstance(value, (list, tuple)) and not value:
continue
if value != default:
raise NotImplementedError(f"internvl does not support {name}")
Comment thread
coderabbitai[bot] marked this conversation as resolved.
names = {field.name for field in fields(BuildRequest)}
if unknown := set(vars(request)) - names - set(unsupported):
raise ValueError(f"unknown internvl build inputs: {sorted(unknown)}")
return BuildRequest(**{name: getattr(request, name) for name in names})
100 changes: 100 additions & 0 deletions families/internvl/cli.json
Original file line number Diff line number Diff line change
@@ -0,0 +1,100 @@
{
"version": 1,
"commands": [
{
"name": "build",
"help": "Build one internvl TensorRT bundle",
"executor": "python",
"handler": "cli:build",
"arguments": [
{
"name": "model",
"type": "string",
"help": "Hugging Face model ID or local snapshot"
},
{
"name": "output",
"flags": [
"-o",
"--output"
],
"type": "path",
"required": true
},
{
"name": "revision",
"flags": [
"--revision"
],
"type": "string"
},
{
"name": "task",
"flags": [
"--task"
],
"type": "string",
"choices": [
"vision_language_generation"
],
"default": "vision_language_generation"
},
{
"name": "precision",
"flags": [
"--precision"
],
"type": "string",
"choices": [
"fp32",
"fp16",
"bf16"
],
"default": "fp32"
},
{
"name": "backend",
"flags": [
"--backend"
],
"type": "string",
"choices": [
"trt",
"trt_rtx"
],
"default": "trt"
},
{
"name": "max_sequence_length",
"flags": [
"--max-sequence-length"
],
"type": "int"
},
{
"name": "tensor_parallel_size",
"flags": [
"--tensor-parallel-size"
],
"type": "int",
"choices": [
1,
2,
4,
8
],
"default": 1
},
{
"name": "verbose",
"flags": [
"--verbose"
],
"type": "bool",
"action": "store_true",
"default": false
}
]
}
]
}
50 changes: 50 additions & 0 deletions families/internvl/cli.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,50 @@
# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved.
# SPDX-License-Identifier: Apache-2.0

"""internvl-owned build command and typed inputs; importing this module is CPU-only."""

from __future__ import annotations

from dataclasses import replace
from pathlib import Path

from tensorrt_model_connect.build import select_backend
from tensorrt_model_connect.bundle_writer import BundleWriter
from tensorrt_model_connect.graph_transform import graph_transform
from tensorrt_model_connect.model_support import load_model_metadata, resolve_family, resolve_model

from .build_request import BuildRequest


def build_bundle(request: BuildRequest, output: Path) -> None:
"""Build and publish through the owning model, preserving atomic failure."""
request = replace(request, output_path=output)
select_backend(request.backend)
from .model import build as build_model

writer = BundleWriter(output)
try:
with graph_transform(request.graph_transform):
build_model(request, writer)
writer.finish()
except BaseException:
writer.abort()
raise

def build(
*, model: str, output: Path, revision: str | None = None,
task: str = "vision_language_generation", precision: str = "fp32", backend: str = "trt",
max_sequence_length: int | None = None, tensor_parallel_size: int = 1,
verbose: bool = False,
) -> int:
"""Run the declared owner command; help never imports this handler."""
model_dir = resolve_model(model, revision)
resolve_family(load_model_metadata(model_dir), "internvl")
request = BuildRequest(
model_dir=model_dir, output_path=output, family="internvl",
task=task, precision=precision, backend=backend,
max_sequence_length=max_sequence_length, tensor_parallel_size=tensor_parallel_size,
verbose=verbose,
)
build_bundle(request, output)
return 0
104 changes: 104 additions & 0 deletions families/internvl/edge_llm/README.md
Original file line number Diff line number Diff line change
@@ -0,0 +1,104 @@
# InternVL: optional native Edge-LLM execution

The family owns configuration admission, builder argument mapping, bundle
composition, runtime orchestration, and validation. Edge owns complete visual
and language networks. This route uses the official GitHub Edge-LLM **0.10.1**
snapshot **e8b29522938901f6df19ebeedd4b69bc8edbcd97**, provisioned through the
optional native CMake package. It does not use another checkout or cross-compile.

## Scope

Only original-source InternVL3 HF checkpoints with the recorded dense Qwen2
text and 448-pixel vision configurations enter this route. Compute is FP16,
batch size and tensor/context parallelism are one. The 1B/2B/8B configurations
route on native x86 SM80; 14B routes on native x86 SM120. Other platforms,
quantized sources, InternVL3.5, multiple-image public requests, and speculative
decoding are not qualified by this change.

The existing native builder remains the default for requests outside this
map. If Edge preparation fails, the family logs a warning and retries the
unchanged native request once. Publication errors never retry into a different
backend. Runtime loading dispatches on the family-owned Edge bundle marker.

## Build and runtime

Enable the generic optional SDK with `TRTMC_ENABLE_EDGELLM=ON`, build it on the
executing GPU, and expose its installation through `CMAKE_PREFIX_PATH`.
The installed package must match the pin, SM, CUDA, and TensorRT identity.
The family-owned `trtmc internvl build` command uses the existing CLI protocol.

`edge_llm/builder.py` maps the request into the pinned Python direct builder with
`--components llm,visual`. It preserves both engines, processor/tokenizer
metadata, chat template, and the checkpoint required for external weights.
The family C++ adapter uses the installed Edge runtime API; it does not
implement either model graph.

One public RGB image uses a 448-pixel tile and 256 visual tokens. The visual
profile retains 256 tokens per image and an aggregate token budget derived
from the requested context. That also permits the original Edge two-image
reference workload; it does **not** expand the Model Connect single-image API.
Edge owns preprocessing, normalization, visual inference, and token expansion.

Requests retain the existing public greedy-generation controls and context
limits. Unsupported controls are rejected rather than silently ignored.
Text-only generation and image-bearing generation share a persistent Edge
runtime. Quantization is never inferred from a filename or applied implicitly.

## Recorded full-model qualification

The following are historical local build/inference receipts, not fresh-head
CI claims. All four passed public Model Connect versus direct Edge comparisons
and independent HF image/text comparisons with NED **0** and exact tokens.
Original Edge `vlm_basic` used its two-image fixture and unchanged gates.

| Checkpoint (OpenGVLab) | Source revision | Native SM | Context | Edge ROUGE-1 / ROUGE-L |
| --- | --- | --- | --- | --- |
| InternVL3-1B-hf | `014c0583a0d4bedf29fbe2dbff4f865eb998e171` | 80 | 1024 | 0.5662 / 0.3193 |
| InternVL3-2B-hf | `cb57a075cb75a2e6d1b668b128d48bb00ae321d2` | 80 | 1024 | 0.6275 / 0.3958 |
| InternVL3-8B-hf | `259a3b64a14623c0ec91a045cb43f7c5af5fa6af` | 80 | 1024 | 0.5704 / 0.3654 |
| InternVL3-14B-hf | `e22931943e5336f85e06f4e2b38f3e5e6ee4de3b` | 120 | 1024 | 0.5330 / 0.3296 |

The existing 2B and 8B owning E2Es also passed at their manifest context **384**,
including actual visual-feature health. The 8B golden-reference and 2B HF
comparison criteria were not changed. The 1B and 14B profiles have no owning
registered E2E manifest. TP2/TP4 manifests remain native and are not additional
Edge qualification.

Successful full bundles and source checkpoints were retired after preserving
compact receipts. Replaying full inference requires restoring the exact source
and rebuilding locally. Current source/build checks are not substitutes for
that replay, nor proof of every model in the upstream catalog.

## Existing image-health test

Edge 0.10.1 has no visual-feature dump CLI. The test-only
`internvl_edge_vision_features` executable therefore calls the pinned
`MultimodalRunner` API and copies its actual FP16 output to a temporary file.
It contains no generation driver and is built only when both
`TRTMC_BUILD_TESTS` and the optional Edge package are enabled.

The existing `tests/vision_oracle.py` selects this helper for Edge bundles.
It streams visual/checkpoint sections into temporary storage instead of loading
checkpoint weights into Python memory. Native bundles continue to use their
existing `vision.plan` path. `TRTMC_RUNTIME_ROOT` must point at the build tree
containing the helper and matching plugin.

The owning E2E still requires nonempty, finite, nonzero features and then runs
its existing generation-quality comparison. No thresholds were weakened and no
new test framework was introduced. The standalone cosine contract remains
unchanged; a health assertion alone is not a claim of HF feature parity.

Ordinary builds without an installed optional Edge SDK select native without a
warning. Malformed or incomplete installed packages still retain diagnostics and
warn before native fallback. Temporary checkpoint/engine staging uses the bundle
output directory filesystem (choose a scratch-backed output), not system /tmp.

## Declared build command

This family uses the existing cli.json protocol introduced in #1310. The family
owns its declaration, typed inputs and Python handler. The handler adapts those
inputs to the unchanged builder API, preserving native/Edge dispatch and bundle
publication. The legacy flat build command remains available for its existing
ordinary options; new family options use `trtmc internvl build`.
Help is offline and does not need a local checkpoint. No shared parser hook or
family registry entry is added.
4 changes: 4 additions & 0 deletions families/internvl/edge_llm/__init__.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,4 @@
# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved.
# SPDX-License-Identifier: Apache-2.0

"""Family-owned optional complete-network Edge offload."""
Loading
Loading