-
Notifications
You must be signed in to change notification settings - Fork 67
feat(internvl): add native Edge execution #1315
New issue
Have a question about this project? Sign up for a free GitHub account to open an issue and contact its maintainers and the community.
By clicking “Sign up for GitHub”, you agree to our terms of service and privacy statement. We’ll occasionally send you account related emails.
Already on GitHub? Sign in to your account
Merged
Merged
Changes from all commits
Commits
Show all changes
7 commits
Select commit
Hold shift + click to select a range
6c98357
feat(internvl): add native Edge execution
JCalafato 905a2c3
fix(internvl): reserve default output capacity
JCalafato b663c72
fix(internvl): preserve native capacity admission
JCalafato 8c7a529
refactor(internvl): own Edge build selection
JCalafato f2c2511
fix(internvl): preserve optional SDK fallback
JCalafato 0fee3aa
refactor(internvl): use declared family CLI
JCalafato 66e07a5
fix(internvl): address Edge review findings
JCalafato File filter
Filter by extension
Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
There are no files selected for viewing
This file contains hidden or bidirectional Unicode text that may be interpreted or compiled differently than what appears below. To review, open the file in an editor that reveals hidden Unicode characters.
Learn more about bidirectional Unicode characters
| Original file line number | Diff line number | Diff line change |
|---|---|---|
| @@ -0,0 +1,98 @@ | ||
| # SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. | ||
| # SPDX-License-Identifier: Apache-2.0 | ||
|
|
||
| """internvl build inputs and strict compatibility for existing Python callers.""" | ||
|
|
||
| from __future__ import annotations | ||
|
|
||
| from dataclasses import dataclass, fields | ||
| from pathlib import Path | ||
| import re | ||
| from typing import ClassVar | ||
|
|
||
| from tensorrt_model_connect.graph_transform import GraphTransform | ||
|
|
||
|
|
||
| def _validate_id(field: str, value: str) -> None: | ||
| if not isinstance(value, str) or re.fullmatch(r"[a-z][a-z0-9_]*", value) is None: | ||
| raise ValueError(f"{field} must be a lowercase identifier") | ||
|
|
||
|
|
||
| @dataclass(frozen=True) | ||
| class BuildRequest: | ||
| """internvl-owned inputs; unsupported legacy controls are read-only defaults.""" | ||
|
|
||
| model_dir: Path | ||
| output_path: Path | ||
| family: str | ||
| task: str | ||
| precision: str | ||
| backend: str = "trt" | ||
| max_sequence_length: int | None = None | ||
| image_height: ClassVar[int | None] = None | ||
| image_width: ClassVar[int | None] = None | ||
| video_num_frames: ClassVar[int | None] = None | ||
| max_batch_size: ClassVar[int] = 1 | ||
| tensor_parallel_size: int = 1 | ||
| context_parallel_size: ClassVar[int] = 1 | ||
| quantization: ClassVar[str | None] = None | ||
| fp32_layers: ClassVar[tuple[int, ...]] = () | ||
| dynamic_kv_cache: ClassVar[bool] = False | ||
| verbose: bool = False | ||
| graph_transform: GraphTransform | None = None | ||
|
|
||
| def __post_init__(self) -> None: | ||
| if not self.precision: | ||
| raise ValueError("precision must be non-empty") | ||
| _validate_id("family", self.family) | ||
| _validate_id("task", self.task) | ||
| if self.backend not in {"trt", "trt_rtx"}: | ||
| raise ValueError("backend must be 'trt' or 'trt_rtx'") | ||
| if self.max_sequence_length is not None and self.max_sequence_length < 1: | ||
| raise ValueError("max_sequence_length must be positive") | ||
| for field in ("image_height", "image_width", "video_num_frames"): | ||
| value = getattr(self, field) | ||
| if value is not None and value < 1: | ||
| raise ValueError(f"{field} must be positive") | ||
| if self.max_batch_size < 1: | ||
| raise ValueError("max_batch_size must be positive") | ||
| if self.tensor_parallel_size < 1: | ||
| raise ValueError("tensor_parallel_size must be positive") | ||
| if self.context_parallel_size < 1: | ||
| raise ValueError("context_parallel_size must be positive") | ||
| if self.quantization is not None and not self.quantization: | ||
| raise ValueError("quantization must be non-empty when provided") | ||
| if any(layer < 0 for layer in self.fp32_layers): | ||
| raise ValueError("fp32_layers must contain non-negative indices") | ||
| if not isinstance(self.dynamic_kv_cache, bool): | ||
| raise ValueError("dynamic_kv_cache must be a bool") | ||
| if self.graph_transform is not None and not callable(self.graph_transform): | ||
| raise ValueError("graph_transform must be callable when provided") | ||
|
|
||
|
|
||
| def coerce_request(request: object) -> BuildRequest: | ||
| """Reject unsupported/unknown legacy inputs before converting to owner fields.""" | ||
| if isinstance(request, BuildRequest): | ||
| return request | ||
| unsupported = { | ||
| "image_height": None, | ||
| "image_width": None, | ||
| "video_num_frames": None, | ||
| "max_batch_size": 1, | ||
| "context_parallel_size": 1, | ||
| "quantization": None, | ||
| "fp32_layers": (), | ||
| "dynamic_kv_cache": False, | ||
| } | ||
| for name, default in unsupported.items(): | ||
| value = getattr(request, name, default) | ||
| if name == "quantization" and value == "none": | ||
| continue | ||
| if name == "fp32_layers" and isinstance(value, (list, tuple)) and not value: | ||
| continue | ||
| if value != default: | ||
| raise NotImplementedError(f"internvl does not support {name}") | ||
| names = {field.name for field in fields(BuildRequest)} | ||
| if unknown := set(vars(request)) - names - set(unsupported): | ||
| raise ValueError(f"unknown internvl build inputs: {sorted(unknown)}") | ||
| return BuildRequest(**{name: getattr(request, name) for name in names}) | ||
This file contains hidden or bidirectional Unicode text that may be interpreted or compiled differently than what appears below. To review, open the file in an editor that reveals hidden Unicode characters.
Learn more about bidirectional Unicode characters
| Original file line number | Diff line number | Diff line change |
|---|---|---|
| @@ -0,0 +1,100 @@ | ||
| { | ||
| "version": 1, | ||
| "commands": [ | ||
| { | ||
| "name": "build", | ||
| "help": "Build one internvl TensorRT bundle", | ||
| "executor": "python", | ||
| "handler": "cli:build", | ||
| "arguments": [ | ||
| { | ||
| "name": "model", | ||
| "type": "string", | ||
| "help": "Hugging Face model ID or local snapshot" | ||
| }, | ||
| { | ||
| "name": "output", | ||
| "flags": [ | ||
| "-o", | ||
| "--output" | ||
| ], | ||
| "type": "path", | ||
| "required": true | ||
| }, | ||
| { | ||
| "name": "revision", | ||
| "flags": [ | ||
| "--revision" | ||
| ], | ||
| "type": "string" | ||
| }, | ||
| { | ||
| "name": "task", | ||
| "flags": [ | ||
| "--task" | ||
| ], | ||
| "type": "string", | ||
| "choices": [ | ||
| "vision_language_generation" | ||
| ], | ||
| "default": "vision_language_generation" | ||
| }, | ||
| { | ||
| "name": "precision", | ||
| "flags": [ | ||
| "--precision" | ||
| ], | ||
| "type": "string", | ||
| "choices": [ | ||
| "fp32", | ||
| "fp16", | ||
| "bf16" | ||
| ], | ||
| "default": "fp32" | ||
| }, | ||
| { | ||
| "name": "backend", | ||
| "flags": [ | ||
| "--backend" | ||
| ], | ||
| "type": "string", | ||
| "choices": [ | ||
| "trt", | ||
| "trt_rtx" | ||
| ], | ||
| "default": "trt" | ||
| }, | ||
| { | ||
| "name": "max_sequence_length", | ||
| "flags": [ | ||
| "--max-sequence-length" | ||
| ], | ||
| "type": "int" | ||
| }, | ||
| { | ||
| "name": "tensor_parallel_size", | ||
| "flags": [ | ||
| "--tensor-parallel-size" | ||
| ], | ||
| "type": "int", | ||
| "choices": [ | ||
| 1, | ||
| 2, | ||
| 4, | ||
| 8 | ||
| ], | ||
| "default": 1 | ||
| }, | ||
| { | ||
| "name": "verbose", | ||
| "flags": [ | ||
| "--verbose" | ||
| ], | ||
| "type": "bool", | ||
| "action": "store_true", | ||
| "default": false | ||
| } | ||
| ] | ||
| } | ||
| ] | ||
| } |
This file contains hidden or bidirectional Unicode text that may be interpreted or compiled differently than what appears below. To review, open the file in an editor that reveals hidden Unicode characters.
Learn more about bidirectional Unicode characters
| Original file line number | Diff line number | Diff line change |
|---|---|---|
| @@ -0,0 +1,50 @@ | ||
| # SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. | ||
| # SPDX-License-Identifier: Apache-2.0 | ||
|
|
||
| """internvl-owned build command and typed inputs; importing this module is CPU-only.""" | ||
|
|
||
| from __future__ import annotations | ||
|
|
||
| from dataclasses import replace | ||
| from pathlib import Path | ||
|
|
||
| from tensorrt_model_connect.build import select_backend | ||
| from tensorrt_model_connect.bundle_writer import BundleWriter | ||
| from tensorrt_model_connect.graph_transform import graph_transform | ||
| from tensorrt_model_connect.model_support import load_model_metadata, resolve_family, resolve_model | ||
|
|
||
| from .build_request import BuildRequest | ||
|
|
||
|
|
||
| def build_bundle(request: BuildRequest, output: Path) -> None: | ||
| """Build and publish through the owning model, preserving atomic failure.""" | ||
| request = replace(request, output_path=output) | ||
| select_backend(request.backend) | ||
| from .model import build as build_model | ||
|
|
||
| writer = BundleWriter(output) | ||
| try: | ||
| with graph_transform(request.graph_transform): | ||
| build_model(request, writer) | ||
| writer.finish() | ||
| except BaseException: | ||
| writer.abort() | ||
| raise | ||
|
|
||
| def build( | ||
| *, model: str, output: Path, revision: str | None = None, | ||
| task: str = "vision_language_generation", precision: str = "fp32", backend: str = "trt", | ||
| max_sequence_length: int | None = None, tensor_parallel_size: int = 1, | ||
| verbose: bool = False, | ||
| ) -> int: | ||
| """Run the declared owner command; help never imports this handler.""" | ||
| model_dir = resolve_model(model, revision) | ||
| resolve_family(load_model_metadata(model_dir), "internvl") | ||
| request = BuildRequest( | ||
| model_dir=model_dir, output_path=output, family="internvl", | ||
| task=task, precision=precision, backend=backend, | ||
| max_sequence_length=max_sequence_length, tensor_parallel_size=tensor_parallel_size, | ||
| verbose=verbose, | ||
| ) | ||
| build_bundle(request, output) | ||
| return 0 |
This file contains hidden or bidirectional Unicode text that may be interpreted or compiled differently than what appears below. To review, open the file in an editor that reveals hidden Unicode characters.
Learn more about bidirectional Unicode characters
| Original file line number | Diff line number | Diff line change |
|---|---|---|
| @@ -0,0 +1,104 @@ | ||
| # InternVL: optional native Edge-LLM execution | ||
|
|
||
| The family owns configuration admission, builder argument mapping, bundle | ||
| composition, runtime orchestration, and validation. Edge owns complete visual | ||
| and language networks. This route uses the official GitHub Edge-LLM **0.10.1** | ||
| snapshot **e8b29522938901f6df19ebeedd4b69bc8edbcd97**, provisioned through the | ||
| optional native CMake package. It does not use another checkout or cross-compile. | ||
|
|
||
| ## Scope | ||
|
|
||
| Only original-source InternVL3 HF checkpoints with the recorded dense Qwen2 | ||
| text and 448-pixel vision configurations enter this route. Compute is FP16, | ||
| batch size and tensor/context parallelism are one. The 1B/2B/8B configurations | ||
| route on native x86 SM80; 14B routes on native x86 SM120. Other platforms, | ||
| quantized sources, InternVL3.5, multiple-image public requests, and speculative | ||
| decoding are not qualified by this change. | ||
|
|
||
| The existing native builder remains the default for requests outside this | ||
| map. If Edge preparation fails, the family logs a warning and retries the | ||
| unchanged native request once. Publication errors never retry into a different | ||
| backend. Runtime loading dispatches on the family-owned Edge bundle marker. | ||
|
|
||
| ## Build and runtime | ||
|
|
||
| Enable the generic optional SDK with `TRTMC_ENABLE_EDGELLM=ON`, build it on the | ||
| executing GPU, and expose its installation through `CMAKE_PREFIX_PATH`. | ||
| The installed package must match the pin, SM, CUDA, and TensorRT identity. | ||
| The family-owned `trtmc internvl build` command uses the existing CLI protocol. | ||
|
|
||
| `edge_llm/builder.py` maps the request into the pinned Python direct builder with | ||
| `--components llm,visual`. It preserves both engines, processor/tokenizer | ||
| metadata, chat template, and the checkpoint required for external weights. | ||
| The family C++ adapter uses the installed Edge runtime API; it does not | ||
| implement either model graph. | ||
|
|
||
| One public RGB image uses a 448-pixel tile and 256 visual tokens. The visual | ||
| profile retains 256 tokens per image and an aggregate token budget derived | ||
| from the requested context. That also permits the original Edge two-image | ||
| reference workload; it does **not** expand the Model Connect single-image API. | ||
| Edge owns preprocessing, normalization, visual inference, and token expansion. | ||
|
|
||
| Requests retain the existing public greedy-generation controls and context | ||
| limits. Unsupported controls are rejected rather than silently ignored. | ||
| Text-only generation and image-bearing generation share a persistent Edge | ||
| runtime. Quantization is never inferred from a filename or applied implicitly. | ||
|
|
||
| ## Recorded full-model qualification | ||
|
|
||
| The following are historical local build/inference receipts, not fresh-head | ||
| CI claims. All four passed public Model Connect versus direct Edge comparisons | ||
| and independent HF image/text comparisons with NED **0** and exact tokens. | ||
| Original Edge `vlm_basic` used its two-image fixture and unchanged gates. | ||
|
|
||
| | Checkpoint (OpenGVLab) | Source revision | Native SM | Context | Edge ROUGE-1 / ROUGE-L | | ||
| | --- | --- | --- | --- | --- | | ||
| | InternVL3-1B-hf | `014c0583a0d4bedf29fbe2dbff4f865eb998e171` | 80 | 1024 | 0.5662 / 0.3193 | | ||
| | InternVL3-2B-hf | `cb57a075cb75a2e6d1b668b128d48bb00ae321d2` | 80 | 1024 | 0.6275 / 0.3958 | | ||
| | InternVL3-8B-hf | `259a3b64a14623c0ec91a045cb43f7c5af5fa6af` | 80 | 1024 | 0.5704 / 0.3654 | | ||
| | InternVL3-14B-hf | `e22931943e5336f85e06f4e2b38f3e5e6ee4de3b` | 120 | 1024 | 0.5330 / 0.3296 | | ||
|
|
||
| The existing 2B and 8B owning E2Es also passed at their manifest context **384**, | ||
| including actual visual-feature health. The 8B golden-reference and 2B HF | ||
| comparison criteria were not changed. The 1B and 14B profiles have no owning | ||
| registered E2E manifest. TP2/TP4 manifests remain native and are not additional | ||
| Edge qualification. | ||
|
|
||
| Successful full bundles and source checkpoints were retired after preserving | ||
| compact receipts. Replaying full inference requires restoring the exact source | ||
| and rebuilding locally. Current source/build checks are not substitutes for | ||
| that replay, nor proof of every model in the upstream catalog. | ||
|
|
||
| ## Existing image-health test | ||
|
|
||
| Edge 0.10.1 has no visual-feature dump CLI. The test-only | ||
| `internvl_edge_vision_features` executable therefore calls the pinned | ||
| `MultimodalRunner` API and copies its actual FP16 output to a temporary file. | ||
| It contains no generation driver and is built only when both | ||
| `TRTMC_BUILD_TESTS` and the optional Edge package are enabled. | ||
|
|
||
| The existing `tests/vision_oracle.py` selects this helper for Edge bundles. | ||
| It streams visual/checkpoint sections into temporary storage instead of loading | ||
| checkpoint weights into Python memory. Native bundles continue to use their | ||
| existing `vision.plan` path. `TRTMC_RUNTIME_ROOT` must point at the build tree | ||
| containing the helper and matching plugin. | ||
|
|
||
| The owning E2E still requires nonempty, finite, nonzero features and then runs | ||
| its existing generation-quality comparison. No thresholds were weakened and no | ||
| new test framework was introduced. The standalone cosine contract remains | ||
| unchanged; a health assertion alone is not a claim of HF feature parity. | ||
|
|
||
| Ordinary builds without an installed optional Edge SDK select native without a | ||
| warning. Malformed or incomplete installed packages still retain diagnostics and | ||
| warn before native fallback. Temporary checkpoint/engine staging uses the bundle | ||
| output directory filesystem (choose a scratch-backed output), not system /tmp. | ||
|
|
||
| ## Declared build command | ||
|
|
||
| This family uses the existing cli.json protocol introduced in #1310. The family | ||
| owns its declaration, typed inputs and Python handler. The handler adapts those | ||
| inputs to the unchanged builder API, preserving native/Edge dispatch and bundle | ||
| publication. The legacy flat build command remains available for its existing | ||
| ordinary options; new family options use `trtmc internvl build`. | ||
| Help is offline and does not need a local checkpoint. No shared parser hook or | ||
| family registry entry is added. |
This file contains hidden or bidirectional Unicode text that may be interpreted or compiled differently than what appears below. To review, open the file in an editor that reveals hidden Unicode characters.
Learn more about bidirectional Unicode characters
| Original file line number | Diff line number | Diff line change |
|---|---|---|
| @@ -0,0 +1,4 @@ | ||
| # SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. | ||
| # SPDX-License-Identifier: Apache-2.0 | ||
|
|
||
| """Family-owned optional complete-network Edge offload.""" |
Oops, something went wrong.
Oops, something went wrong.
Add this suggestion to a batch that can be applied as a single commit.
This suggestion is invalid because no changes were made to the code.
Suggestions cannot be applied while the pull request is closed.
Suggestions cannot be applied while viewing a subset of changes.
Only one suggestion per line can be applied in a batch.
Add this suggestion to a batch that can be applied as a single commit.
Applying suggestions on deleted lines is not supported.
You must change the existing code in this line in order to create a valid suggestion.
Outdated suggestions cannot be applied.
This suggestion has been applied or marked resolved.
Suggestions cannot be applied from pending reviews.
Suggestions cannot be applied on multi-line comments.
Suggestions cannot be applied while the pull request is queued to merge.
Suggestion cannot be applied right now. Please check back later.
Uh oh!
There was an error while loading. Please reload this page.