From c482a18c4e0054802bfb78025dd7ccfdf21c6b8b Mon Sep 17 00:00:00 2001 From: Joshua Calafato Date: Wed, 16 Sep 2026 23:13:19 +0000 Subject: [PATCH 01/20] feat(build): add optional native Edge-LLM SDK Provision the official pinned SDK through CMake with optional ONNX tools and native platform, capability, and exact JSON-header checks. Keep package discovery and dependency setup separate from model builds. Transport explicit family-owned companion inputs without shared model dispatch. Add bounded bundle extraction and separate executable diagnostics from machine-readable results. Document the optional build/runtime workflow and extend existing tests. Signed-off-by: Joshua Calafato --- CMakeLists.txt | 2 + apps/cli/main.cpp | 9 +- cmake/EdgeLLM.cmake | 119 +++++++ cmake/edgellm/CheckNative.cmake | 73 +++++ cmake/edgellm/EdgeLLMConfig.cmake.in | 50 +++ cmake/edgellm/Install.cmake.in | 35 +++ cmake/edgellm/Prepare.cmake.in | 49 +++ cmake/edgellm/README.md | 72 +++++ .../tensorrt_model_connect/__init__.py | 4 +- core/builder/tensorrt_model_connect/build.py | 129 +++++++- .../tensorrt_model_connect/build_cli.py | 72 +++-- core/builder/tests/test_build.py | 295 +++++++++++++++++- core/runtime/bundle/bundle_format.cpp | 27 ++ core/runtime/include/trtmc/bundle.h | 7 + core/runtime/tests/test_bundle_format_v1.cpp | 32 ++ tools/tests/test_architecture.py | 10 +- website/docs/api/python-builder.md | 30 ++ website/docs/architecture/build-pipeline.md | 11 +- website/docs/user-guides/configure-runtime.md | 23 ++ 19 files changed, 1020 insertions(+), 29 deletions(-) create mode 100644 cmake/EdgeLLM.cmake create mode 100644 cmake/edgellm/CheckNative.cmake create mode 100644 cmake/edgellm/EdgeLLMConfig.cmake.in create mode 100644 cmake/edgellm/Install.cmake.in create mode 100644 cmake/edgellm/Prepare.cmake.in create mode 100644 cmake/edgellm/README.md diff --git a/CMakeLists.txt b/CMakeLists.txt index 8a083fa6f4..bb21195fb3 100644 --- a/CMakeLists.txt +++ b/CMakeLists.txt @@ -45,6 +45,8 @@ find_library(TRTMC_TRT_LIBRARY REQUIRED ) +include("${CMAKE_CURRENT_SOURCE_DIR}/cmake/EdgeLLM.cmake") + option(TRTMC_ENABLE_BYOK "Enable the optional TVM-FFI BYOK bridge" ON) set(TRTMC_HAS_TVM_FFI OFF) if(TRTMC_ENABLE_BYOK) diff --git a/apps/cli/main.cpp b/apps/cli/main.cpp index dbb5c93bb8..8a27e4292e 100644 --- a/apps/cli/main.cpp +++ b/apps/cli/main.cpp @@ -8,5 +8,12 @@ #include int main(int argc, char** argv) { - return trtmc::cli::run(argc, argv, std::cout, std::cerr); + // The executable owns the console: keep result output machine-readable even + // when loaded libraries write C++ diagnostics to std::cout. Do not change + // library logger levels or the output behavior of embedded runtime APIs. + std::ostream result(std::cout.rdbuf()); + std::cout.rdbuf(std::cerr.rdbuf()); + const int status = trtmc::cli::run(argc, argv, result, std::cerr); + result.flush(); + return status; } diff --git a/cmake/EdgeLLM.cmake b/cmake/EdgeLLM.cmake new file mode 100644 index 0000000000..678e33dced --- /dev/null +++ b/cmake/EdgeLLM.cmake @@ -0,0 +1,119 @@ +# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-License-Identifier: Apache-2.0 + +# Optional native dependency provisioning. Model builds never acquire dependencies. +option(TRTMC_ENABLE_EDGELLM "Install the pinned native Edge-LLM SDK and builder" OFF) +if(NOT TRTMC_ENABLE_EDGELLM) + return() +endif() +if(CMAKE_CROSSCOMPILING) + message(FATAL_ERROR "Edge-LLM cross compilation is not supported") +endif() +include("${CMAKE_CURRENT_LIST_DIR}/edgellm/CheckNative.cmake") +option(TRTMC_EDGELLM_ALL_KERNELS "Build all pinned Edge operator groups supported by the native GPU" OFF) +option(TRTMC_EDGELLM_ONNX "Install the pinned ONNX exporter and native engine builder" OFF) +set(_edge_cute_groups "fmha|gdn") +set(_edge_cute_cli_groups "fmha,gdn") +if(TRTMC_EDGELLM_ALL_KERNELS) + set(_edge_cute_groups ALL) + set(_edge_cute_cli_groups ALL) +endif() +set(_edge_build_targets edgellmCore NvInfer_edgellm_plugin) +set(_edge_onnx_byproducts "") +if(TRTMC_EDGELLM_ONNX) + list(APPEND _edge_build_targets llm_build) + list(APPEND _edge_onnx_byproducts "${CMAKE_BINARY_DIR}/_deps/edgellm/install/bin/edgellm-onnx-build") +endif() +set(_edge_version "0.10.1") +set(_edge_revision "e8b29522938901f6df19ebeedd4b69bc8edbcd97") +set(_edge_root "${CMAKE_BINARY_DIR}/_deps/edgellm") +set(_edge_prefix "${_edge_root}/install") +find_package(EdgeLLM ${_edge_version} EXACT CONFIG QUIET) +if(EdgeLLM_FOUND AND NOT EdgeLLM_PREFIX STREQUAL _edge_prefix) + _edgellm_json_include(_edge_json_include) + _edgellm_check_json_headers("${_edge_json_include}" "${EdgeLLM_PREFIX}/include/edgellm/3rdParty/nlohmannJson") + if(NOT EdgeLLM_REVISION STREQUAL _edge_revision) + message(FATAL_ERROR "EdgeLLM package does not match the pinned GitHub revision") + endif() + if(TRTMC_EDGELLM_ALL_KERNELS AND NOT EdgeLLM_ALL_KERNELS) + message(FATAL_ERROR "EdgeLLM package lacks requested full native operator coverage; rebuild with TRTMC_EDGELLM_ALL_KERNELS=ON") + endif() + if(TRTMC_EDGELLM_ONNX AND (NOT EdgeLLM_ONNX OR NOT EXISTS "${EdgeLLM_ONNX_BUILDER}")) + message(FATAL_ERROR "EdgeLLM package lacks requested ONNX tools; rebuild with TRTMC_EDGELLM_ONNX=ON") + endif() + install(FILES "$" DESTINATION "${CMAKE_INSTALL_LIBDIR}" COMPONENT EdgeLLM) + return() +endif() + +include(ExternalProject) +include(CMakePackageConfigHelpers) +find_package(Python3 3.10 REQUIRED COMPONENTS Interpreter) +find_package(Threads REQUIRED) +set(TRTMC_EDGELLM_TRT_ROOT "$ENV{TRT_ROOT}" CACHE PATH "Native TensorRT SDK, including its Python wheel") +set(TRTMC_EDGELLM_CUDA_ARCHITECTURE "${CMAKE_CUDA_ARCHITECTURES}" CACHE STRING "One local GPU architecture for Edge-LLM") +set(TRTMC_EDGELLM_JOBS 2 CACHE STRING "Parallel Edge-LLM native and AOT compilation jobs") +set(TRTMC_EDGELLM_WHEELHOUSE "" CACHE PATH "Optional complete offline Python wheelhouse") +set(TRTMC_EDGELLM_GIT_MIRROR "" CACHE PATH "Optional local mirror of the pinned upstream Git repository") +if(NOT TRTMC_EDGELLM_CUDA_ARCHITECTURE MATCHES "^[0-9]+$") + message(FATAL_ERROR "Set TRTMC_EDGELLM_CUDA_ARCHITECTURE to one local GPU architecture, e.g. 80") +endif() +if(NOT EXISTS "${TRTMC_EDGELLM_TRT_ROOT}/include/NvInfer.h") + message(FATAL_ERROR "TRTMC_EDGELLM_TRT_ROOT must contain the native TensorRT SDK") +endif() +_edgellm_check_gpu("${TRTMC_EDGELLM_CUDA_ARCHITECTURE}") +_edgellm_trt_version("${TRTMC_EDGELLM_TRT_ROOT}/include" _edge_trt_version) +set(_edge_source "${_edge_root}/source") +set(_edge_build "${_edge_root}/build") +set(_edge_python "${_edge_prefix}/libexec/trtmc-edge-llm/bin/python") +set(_edge_repository "https://github.com/NVIDIA/TensorRT-Edge-LLM.git") +if(TRTMC_EDGELLM_GIT_MIRROR) + set(_edge_repository "${TRTMC_EDGELLM_GIT_MIRROR}") +endif() +set(_edge_template_dir "${CMAKE_CURRENT_LIST_DIR}/edgellm") +_edgellm_json_include(_edge_json_include) +file(MAKE_DIRECTORY "${_edge_prefix}/lib/cmake/EdgeLLM" "${_edge_prefix}/include/edgellm/cpp" + "${_edge_prefix}/include/edgellm/3rdParty/nlohmannJson/include" + "${_edge_prefix}/include/edgellm/3rdParty/stb" "${_edge_prefix}/include/edgellm/3rdParty/miniaudio") +configure_file("${_edge_template_dir}/CheckNative.cmake" "${_edge_prefix}/lib/cmake/EdgeLLM/CheckNative.cmake" COPYONLY) +foreach(_script IN ITEMS Prepare Install) + configure_file("${_edge_template_dir}/${_script}.cmake.in" "${_edge_root}/${_script}.cmake" @ONLY) +endforeach() +configure_file("${_edge_template_dir}/EdgeLLMConfig.cmake.in" + "${_edge_prefix}/lib/cmake/EdgeLLM/EdgeLLMConfig.cmake" @ONLY) +write_basic_package_version_file("${_edge_prefix}/lib/cmake/EdgeLLM/EdgeLLMConfigVersion.cmake" + VERSION "${_edge_version}" COMPATIBILITY ExactVersion) +ExternalProject_Add(trtmc_edgellm_dependency + PREFIX "${_edge_root}/ep" SOURCE_DIR "${_edge_source}" BINARY_DIR "${_edge_build}" + GIT_REPOSITORY "${_edge_repository}" GIT_TAG "${_edge_revision}" + GIT_SUBMODULES_RECURSE TRUE UPDATE_DISCONNECTED TRUE + LIST_SEPARATOR | + # Preparation installs tools; it does not patch upstream sources. Keep it in + # the configure step so template changes invalidate disconnected builds too. + CONFIGURE_COMMAND "${CMAKE_COMMAND}" -P "${_edge_root}/Prepare.cmake" + COMMAND "${_edge_prefix}/libexec/trtmc-edge-llm/bin/cmake" + -S -B -DCMAKE_BUILD_TYPE=Release -DCMAKE_POSITION_INDEPENDENT_CODE=ON + "-DCMAKE_CUDA_COMPILER=${CMAKE_CUDA_COMPILER}" + "-DCMAKE_CUDA_ARCHITECTURES=${TRTMC_EDGELLM_CUDA_ARCHITECTURE}" + "-DCUDA_DIR=${CUDAToolkit_LIBRARY_ROOT}" "-DCUDAToolkit_ROOT=${CUDAToolkit_LIBRARY_ROOT}" + "-DCUDA_CTK_VERSION=${CUDAToolkit_VERSION_MAJOR}.${CUDAToolkit_VERSION_MINOR}" + "-DTRT_PACKAGE_DIR=${TRTMC_EDGELLM_TRT_ROOT}" "-DPython3_EXECUTABLE=${_edge_python}" + -DEDGELLM_WHEEL_PAYLOAD_DIR=unused "-DENABLE_CUTE_DSL=${_edge_cute_groups}" + "-DCUTE_DSL_ARTIFACT_TAG=sm_${TRTMC_EDGELLM_CUDA_ARCHITECTURE}" + BUILD_COMMAND "${CMAKE_COMMAND}" --build --target ${_edge_build_targets} + --parallel "${TRTMC_EDGELLM_JOBS}" + INSTALL_COMMAND "${CMAKE_COMMAND}" -P "${_edge_root}/Install.cmake" + BUILD_BYPRODUCTS "${_edge_prefix}/lib/libedgellmCore.a" + "${_edge_prefix}/lib/libNvInfer_edgellm_plugin.so" + "${_edge_prefix}/lib/libcutedsl.a" ${_edge_onnx_byproducts} + LOG_DOWNLOAD ON LOG_CONFIGURE ON LOG_BUILD ON LOG_INSTALL ON LOG_OUTPUT_ON_FAILURE ON) +ExternalProject_Add_StepDependencies(trtmc_edgellm_dependency configure "${_edge_root}/Prepare.cmake") +ExternalProject_Add_StepDependencies(trtmc_edgellm_dependency install "${_edge_root}/Install.cmake") +# Generated package targets refer to declared future byproducts; their build dependency +# prevents consumers from compiling or linking until installation completes. +find_package(EdgeLLM ${_edge_version} EXACT CONFIG REQUIRED + PATHS "${_edge_prefix}/lib/cmake/EdgeLLM" NO_DEFAULT_PATH) +add_dependencies(EdgeLLM::Core trtmc_edgellm_dependency) +add_dependencies(EdgeLLM::Plugin trtmc_edgellm_dependency) +install(DIRECTORY "${_edge_prefix}/" DESTINATION . USE_SOURCE_PERMISSIONS COMPONENT EdgeLLM) +# Family DSOs may use lib64; their dynamically loaded plugin must remain adjacent. +install(FILES "$" DESTINATION "${CMAKE_INSTALL_LIBDIR}" COMPONENT EdgeLLM) diff --git a/cmake/edgellm/CheckNative.cmake b/cmake/edgellm/CheckNative.cmake new file mode 100644 index 0000000000..03dd575b39 --- /dev/null +++ b/cmake/edgellm/CheckNative.cmake @@ -0,0 +1,73 @@ +# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-License-Identifier: Apache-2.0 + +# Read the complete TensorRT SDK version using the compiler, including aliased macros. +# include_dir: native SDK include directory; output: caller variable receiving x.y.z.build. +function(_edgellm_trt_version include_dir output) + set(_version) + set(_probe "${CMAKE_CURRENT_BINARY_DIR}/CMakeFiles/edgellm-version.cpp") + file(WRITE "${_probe}" "#include \n") + foreach(_part IN ITEMS MAJOR MINOR PATCH BUILD) + file(APPEND "${_probe}" "TRTMC_EDGE_${_part}=NV_TENSORRT_${_part}\n") + endforeach() + execute_process(COMMAND "${CMAKE_CXX_COMPILER}" -E -P -I "${include_dir}" "${_probe}" + OUTPUT_VARIABLE _expanded COMMAND_ERROR_IS_FATAL ANY) + foreach(_part IN ITEMS MAJOR MINOR PATCH BUILD) + if(NOT _expanded MATCHES "TRTMC_EDGE_${_part}=[ \t]*([0-9]+)") + message(FATAL_ERROR "Cannot determine TensorRT ${_part} from ${include_dir}") + endif() + list(APPEND _version "${CMAKE_MATCH_1}") + endforeach() + list(JOIN _version "." _version) + set(${output} "${_version}" PARENT_SCOPE) +endfunction() + +# Require the installed package GPU architecture to be present on this build host. +function(_edgellm_check_gpu architecture) + execute_process(COMMAND nvidia-smi --query-gpu=compute_cap --format=csv,noheader + OUTPUT_VARIABLE _sms RESULT_VARIABLE _result OUTPUT_STRIP_TRAILING_WHITESPACE) + string(REPLACE "." "" _sms "${_sms}") + string(REPLACE "\n" ";" _sms "${_sms}") + if(NOT _result EQUAL 0 OR NOT architecture IN_LIST _sms) + message(FATAL_ERROR "EdgeLLM requires a local GPU with architecture ${architecture}") + endif() +endfunction() + +# A version label alone is not an ABI guarantee: development headers can retain +# 3.12.0 while changing parser layouts inside the same C++ ABI namespace. +function(_edgellm_json_include output) + get_target_property(_includes nlohmann_json::nlohmann_json INTERFACE_INCLUDE_DIRECTORIES) + foreach(_include IN LISTS _includes) + string(REGEX REPLACE "^\\$$" "\\1" _include "${_include}") + if(EXISTS "${_include}/nlohmann/json.hpp") + set(${output} "${_include}" PARENT_SCOPE) + return() + endif() + endforeach() + message(FATAL_ERROR "Cannot locate nlohmann_json headers for EdgeLLM ABI validation") +endfunction() + +function(_edgellm_check_json_headers include_dir vendor_dir) + set(_header "${include_dir}/nlohmann/json.hpp") + set(_single "${vendor_dir}/single_include/nlohmann/json.hpp") + set(_multiple "${vendor_dir}/include/nlohmann/json.hpp") + if(NOT EXISTS "${_header}" OR NOT EXISTS "${_single}" OR NOT EXISTS "${_multiple}") + message(FATAL_ERROR "Missing nlohmann_json headers for EdgeLLM ABI validation") + endif() + file(SHA256 "${_header}" _actual) + file(SHA256 "${_single}" _expected_single) + if(_actual STREQUAL _expected_single) + return() + endif() + file(GLOB_RECURSE _headers RELATIVE "${vendor_dir}/include" "${vendor_dir}/include/nlohmann/*.hpp") + foreach(_relative IN LISTS _headers) + if(EXISTS "${include_dir}/${_relative}") + file(SHA256 "${include_dir}/${_relative}" _actual) + file(SHA256 "${vendor_dir}/include/${_relative}" _expected) + if(_actual STREQUAL _expected) + continue() + endif() + endif() + message(FATAL_ERROR "EdgeLLM requires the pinned nlohmann_json headers, not only the same version label. Set nlohmann_json_DIR to an installation of the pinned Edge 3rdParty/nlohmannJson dependency. Mismatch: ${_relative}") + endforeach() +endfunction() diff --git a/cmake/edgellm/EdgeLLMConfig.cmake.in b/cmake/edgellm/EdgeLLMConfig.cmake.in new file mode 100644 index 0000000000..bade6c9dba --- /dev/null +++ b/cmake/edgellm/EdgeLLMConfig.cmake.in @@ -0,0 +1,50 @@ +# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-License-Identifier: Apache-2.0 +include(CMakeFindDependencyMacro) +find_dependency(CUDAToolkit) +find_dependency(Threads) +find_dependency(nlohmann_json 3.12.0 EXACT) +include("${CMAKE_CURRENT_LIST_DIR}/CheckNative.cmake") +get_filename_component(EdgeLLM_PREFIX "${CMAKE_CURRENT_LIST_DIR}/../../.." ABSOLUTE) +_edgellm_json_include(_edge_json_include) +# During first provisioning these future headers do not exist yet; Prepare +# performs the same check after checkout and before installing or building tools. +if(EXISTS "${EdgeLLM_PREFIX}/include/edgellm/3rdParty/nlohmannJson/include/nlohmann/json.hpp") + _edgellm_check_json_headers("${_edge_json_include}" "${EdgeLLM_PREFIX}/include/edgellm/3rdParty/nlohmannJson") +endif() +set(EdgeLLM_VERSION "@_edge_version@") +set(EdgeLLM_REVISION "@_edge_revision@") +# Tool availability only; model admission and orchestration remain family-owned. +set(EdgeLLM_ALL_KERNELS "@TRTMC_EDGELLM_ALL_KERNELS@") +set(EdgeLLM_ONNX "@TRTMC_EDGELLM_ONNX@") +set(EdgeLLM_ONNX_BUILDER "${EdgeLLM_PREFIX}/bin/edgellm-onnx-build") +set(EdgeLLM_CUDA_VERSION "@CUDAToolkit_VERSION@") +set(EdgeLLM_TENSORRT_VERSION "@_edge_trt_version@") +set(EdgeLLM_ARCH "@CMAKE_SYSTEM_PROCESSOR@") +set(EdgeLLM_CUDA_ARCHITECTURE "@TRTMC_EDGELLM_CUDA_ARCHITECTURE@") +if(CMAKE_CROSSCOMPILING OR NOT CMAKE_SYSTEM_PROCESSOR STREQUAL EdgeLLM_ARCH) + message(FATAL_ERROR "EdgeLLM is a native-only package for ${EdgeLLM_ARCH}") +endif() +if(NOT CUDAToolkit_VERSION_MAJOR EQUAL @CUDAToolkit_VERSION_MAJOR@ OR + NOT CUDAToolkit_VERSION_MINOR EQUAL @CUDAToolkit_VERSION_MINOR@) + message(FATAL_ERROR "EdgeLLM requires the CUDA SDK it was built with: ${EdgeLLM_CUDA_VERSION}") +endif() +set(EdgeLLM_PYTHON_EXECUTABLE "${EdgeLLM_PREFIX}/libexec/trtmc-edge-llm/bin/python") +set(EdgeLLM_BUILDER_LAUNCHER "${EdgeLLM_PREFIX}/bin/edgellm-builder") +find_path(EdgeLLM_TRT_INCLUDE_DIR NvInfer.h HINTS "$ENV{TRT_ROOT}" "@TRTMC_EDGELLM_TRT_ROOT@" PATH_SUFFIXES include REQUIRED) +find_library(EdgeLLM_TRT_LIBRARY nvinfer HINTS "$ENV{TRT_ROOT}" "@TRTMC_EDGELLM_TRT_ROOT@" PATH_SUFFIXES lib lib64 REQUIRED) +find_library(EdgeLLM_PARSER_LIBRARY nvonnxparser HINTS "$ENV{TRT_ROOT}" "@TRTMC_EDGELLM_TRT_ROOT@" PATH_SUFFIXES lib lib64 REQUIRED) +_edgellm_trt_version("${EdgeLLM_TRT_INCLUDE_DIR}" _edge_current_trt) +if(NOT _edge_current_trt STREQUAL EdgeLLM_TENSORRT_VERSION) + message(FATAL_ERROR "EdgeLLM requires TensorRT ${EdgeLLM_TENSORRT_VERSION}; found ${_edge_current_trt}") +endif() +_edgellm_check_gpu("${EdgeLLM_CUDA_ARCHITECTURE}") +if(NOT TARGET EdgeLLM::Core) + add_library(EdgeLLM::Core STATIC IMPORTED) + set_target_properties(EdgeLLM::Core PROPERTIES + IMPORTED_LOCATION "${EdgeLLM_PREFIX}/lib/libedgellmCore.a" + INTERFACE_INCLUDE_DIRECTORIES "${EdgeLLM_PREFIX}/include;${EdgeLLM_PREFIX}/include/edgellm/cpp;${EdgeLLM_PREFIX}/include/edgellm/3rdParty/nlohmannJson/include;${EdgeLLM_PREFIX}/include/edgellm/3rdParty/stb;${EdgeLLM_PREFIX}/include/edgellm/3rdParty/miniaudio;${EdgeLLM_TRT_INCLUDE_DIR}" + INTERFACE_LINK_LIBRARIES "${EdgeLLM_PREFIX}/lib/libcutedsl.a;${EdgeLLM_TRT_LIBRARY};${EdgeLLM_PARSER_LIBRARY};CUDA::cudart;CUDA::cuda_driver;Threads::Threads;${CMAKE_DL_LIBS}") + add_library(EdgeLLM::Plugin SHARED IMPORTED) + set_target_properties(EdgeLLM::Plugin PROPERTIES IMPORTED_LOCATION "${EdgeLLM_PREFIX}/lib/libNvInfer_edgellm_plugin.so") +endif() diff --git a/cmake/edgellm/Install.cmake.in b/cmake/edgellm/Install.cmake.in new file mode 100644 index 0000000000..562c720b40 --- /dev/null +++ b/cmake/edgellm/Install.cmake.in @@ -0,0 +1,35 @@ +# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-License-Identifier: Apache-2.0 +cmake_minimum_required(VERSION 3.20) +file(INSTALL "@_edge_build@/cpp/libedgellmCore.a" DESTINATION "@_edge_prefix@/lib") +file(INSTALL "@_edge_build@/libNvInfer_edgellm_plugin.so" DESTINATION "@_edge_prefix@/lib" FOLLOW_SYMLINK_CHAIN) +file(INSTALL "@_edge_source@/cpp/kernels/cuteDSLArtifact/@CMAKE_SYSTEM_PROCESSOR@/sm_@TRTMC_EDGELLM_CUDA_ARCHITECTURE@/libcutedsl_@CMAKE_SYSTEM_PROCESSOR@.a" + DESTINATION "@_edge_prefix@/lib" RENAME libcutedsl.a) +file(INSTALL "@_edge_source@/cpp" DESTINATION "@_edge_prefix@/include/edgellm" FILES_MATCHING PATTERN "*.h" PATTERN "*.cuh") +foreach(_third_party IN ITEMS nlohmannJson stb miniaudio) + file(INSTALL "@_edge_source@/3rdParty/${_third_party}" DESTINATION "@_edge_prefix@/include/edgellm/3rdParty" + FILES_MATCHING PATTERN "*.h" PATTERN "*.hpp") +endforeach() +file(MAKE_DIRECTORY "@_edge_prefix@/bin" "@_edge_prefix@/share/trtmc") +file(WRITE "@_edge_prefix@/bin/edgellm-builder" [=[#!/bin/sh +set -eu +prefix=$(CDPATH= cd -- "$(dirname -- "$0")/.." && pwd) +exec "$prefix/libexec/trtmc-edge-llm/bin/python" -I -c 'from experimental.builder.cli import main; main()' "$@" +]=]) +file(CHMOD "@_edge_prefix@/bin/edgellm-builder" PERMISSIONS OWNER_READ OWNER_WRITE OWNER_EXECUTE GROUP_READ GROUP_EXECUTE WORLD_READ WORLD_EXECUTE) +if("@TRTMC_EDGELLM_ONNX@") + file(INSTALL "@_edge_build@/examples/llm/llm_build" DESTINATION "@_edge_prefix@/bin" + TYPE PROGRAM RENAME edgellm-onnx-build) +endif() +set(_all_kernels false) +if("@TRTMC_EDGELLM_ALL_KERNELS@") + set(_all_kernels true) +endif() +set(_onnx false) +if("@TRTMC_EDGELLM_ONNX@") + set(_onnx true) +endif() +execute_process(COMMAND "@_edge_python@" -I -c "import tensorrt; print(tensorrt.__version__)" + OUTPUT_VARIABLE _trt_version OUTPUT_STRIP_TRAILING_WHITESPACE COMMAND_ERROR_IS_FATAL ANY) +# Paths are relative to the installation prefix, preserving relocatability. +file(WRITE "@_edge_prefix@/share/trtmc/edge-llm.json" "{\n \"schema_version\": 1,\n \"version\": \"@_edge_version@\",\n \"revision\": \"@_edge_revision@\",\n \"arch\": \"@CMAKE_SYSTEM_PROCESSOR@\",\n \"architectures\": [@TRTMC_EDGELLM_CUDA_ARCHITECTURE@],\n \"cuda_version\": \"@CUDAToolkit_VERSION_MAJOR@.@CUDAToolkit_VERSION_MINOR@\",\n \"tensorrt_version\": \"${_trt_version}\",\n \"python\": \"libexec/trtmc-edge-llm/bin/python\",\n \"builder\": \"bin/edgellm-builder\",\n \"all_native_kernels\": ${_all_kernels},\n \"onnx\": ${_onnx},\n \"onnx_builder\": \"bin/edgellm-onnx-build\",\n \"plugin\": \"lib/libNvInfer_edgellm_plugin.so\"\n}\n") diff --git a/cmake/edgellm/Prepare.cmake.in b/cmake/edgellm/Prepare.cmake.in new file mode 100644 index 0000000000..bf19447034 --- /dev/null +++ b/cmake/edgellm/Prepare.cmake.in @@ -0,0 +1,49 @@ +# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-License-Identifier: Apache-2.0 +# Executed only by the explicit CMake dependency build, never by model dispatch. +cmake_minimum_required(VERSION 3.20) +include("@_edge_template_dir@/CheckNative.cmake") +_edgellm_check_json_headers("@_edge_json_include@" "@_edge_source@/3rdParty/nlohmannJson") +function(run) + execute_process(COMMAND ${ARGV} COMMAND_ERROR_IS_FATAL ANY) +endfunction() +execute_process(COMMAND "@Python3_EXECUTABLE@" -I -c "import ensurepip" + RESULT_VARIABLE _has_ensurepip OUTPUT_QUIET ERROR_QUIET) +if(_has_ensurepip EQUAL 0) + run("@Python3_EXECUTABLE@" -I -m venv --copies "@_edge_prefix@/libexec/trtmc-edge-llm") +else() + # Debian minimal Python may omit ensurepip; use an already installed bootstrapper. + run("@Python3_EXECUTABLE@" -I -m virtualenv --copies --no-download --no-periodic-update + "@_edge_prefix@/libexec/trtmc-edge-llm") +endif() +set(_pip_options --isolated install --no-user) +if(NOT "@TRTMC_EDGELLM_WHEELHOUSE@" STREQUAL "") + list(APPEND _pip_options --no-index --find-links "@TRTMC_EDGELLM_WHEELHOUSE@") +endif() +file(GLOB _trt_wheels "@TRTMC_EDGELLM_TRT_ROOT@/python/tensorrt-*-cp@Python3_VERSION_MAJOR@@Python3_VERSION_MINOR@-none-linux_@CMAKE_SYSTEM_PROCESSOR@.whl") +list(LENGTH _trt_wheels _wheel_count) +if(NOT _wheel_count EQUAL 1) + message(FATAL_ERROR "Expected exactly one TensorRT SDK wheel matching the native Python ABI") +endif() +run("@_edge_python@" -I -m pip ${_pip_options} --report "@_edge_prefix@/pip-report.json" + ${_trt_wheels} numpy==2.2.6 transformers==5.14.1 jinja2==3.1.6 + scikit-build-core==0.11.6 wheel==0.45.1 cmake==3.31.10 ninja==1.13.0 + "cuda-python>=@CUDAToolkit_VERSION_MAJOR@.@CUDAToolkit_VERSION_MINOR@,<@CUDAToolkit_VERSION_MAJOR@.@CUDAToolkit_VERSION_MINOR@.999" + "nvidia-cutlass-dsl[cu@CUDAToolkit_VERSION_MAJOR@]==4.7.0" "cupy-cuda@CUDAToolkit_VERSION_MAJOR@x==13.6.0") +run("@_edge_python@" -I -m pip ${_pip_options} --no-deps --no-build-isolation "@_edge_source@") +if("@TRTMC_EDGELLM_ONNX@") + # Original exporter dependencies, isolated from the caller environment. Export + # is CPU-side; native TensorRT compilation still runs on the inference GPU. + set(_torch_options ${_pip_options}) + if("@TRTMC_EDGELLM_WHEELHOUSE@" STREQUAL "") + list(APPEND _torch_options --index-url https://download.pytorch.org/whl/cpu) + endif() + run("@_edge_python@" -I -m pip ${_torch_options} "torch==2.13.0") + run("@_edge_python@" -I -m pip ${_pip_options} "tensorrt-edgellm[export]==@_edge_version@") +endif() +run("@_edge_python@" -I -m pip --isolated check) +run("@_edge_python@" -I -c "print(__import__('tensorrt').__version__)") +set(ENV{PATH} "@CUDAToolkit_BIN_DIR@:$ENV{PATH}") +run("@_edge_python@" -I "@_edge_source@/kernelSrcs/build_cutedsl.py" + --gpu_arch "sm_@TRTMC_EDGELLM_CUDA_ARCHITECTURE@" --arch "@CMAKE_SYSTEM_PROCESSOR@" + --kernels "@_edge_cute_cli_groups@" --cuda-version "@CUDAToolkit_VERSION_MAJOR@.@CUDAToolkit_VERSION_MINOR@" --jobs "@TRTMC_EDGELLM_JOBS@") diff --git a/cmake/edgellm/README.md b/cmake/edgellm/README.md new file mode 100644 index 0000000000..89ae872efc --- /dev/null +++ b/cmake/edgellm/README.md @@ -0,0 +1,72 @@ +# Pinned native Edge-LLM package + +Edge-LLM is optional. The default `TRTMC_ENABLE_EDGELLM=OFF` neither downloads +nor builds it. Enable it once while installing Model Connect; ordinary model +builds only use the installed package and never fetch or install dependencies. +Cross compilation is rejected. Configure and build on the inference GPU host. + +```bash +cmake -S . -B build \ + -DTRTMC_ENABLE_EDGELLM=ON \ + -DTRTMC_EDGELLM_CUDA_ARCHITECTURE=80 \ + -DTRTMC_EDGELLM_TRT_ROOT="$TRT_ROOT" \ + -DCUDAToolkit_ROOT="$CUDA_ROOT" \ + -DCMAKE_CUDA_COMPILER="$CUDA_ROOT/bin/nvcc" \ + -DCMAKE_INSTALL_PREFIX="$PWD/install" +cmake --build build --parallel 8 +cmake --install build +export CMAKE_PREFIX_PATH="$PWD/install${CMAKE_PREFIX_PATH:+:$CMAKE_PREFIX_PATH}" +``` + +The regular project dependencies remain required, including nlohmann_json +**3.12.0 with the exact pinned upstream headers** when Edge is enabled. A +development snapshot can retain that version label but change parser layouts; +mixing it with the static SDK causes undefined behavior. The package checks +header content against its vendored dependency (single or multiple headers). +If rejected, install `3rdParty/nlohmannJson` from the pinned Edge checkout into +a separate prefix and configure with that installation’s `nlohmann_json_DIR`. The Python +interpreter needs `ensurepip` or an already installed `virtualenv` bootstrapper. +The native CUDA SDK must include NVCC, NVRTC, cuRAND headers and driver link +libraries; the TensorRT SDK must contain its matching CPython wheel. + +The provider first uses `find_package(EdgeLLM 0.10.1 EXACT CONFIG)`. If absent, +CMake `ExternalProject` clones the public NVIDIA TensorRT-Edge-LLM repository at +`e8b29522938901f6df19ebeedd4b69bc8edbcd97` (v0.10.1), initializes the pinned +submodules, builds the native core/plugin and FMHA/GDN CuTe archives, and installs +an isolated direct-builder Python environment. It does not modify the caller +Python environment. Downloads happen only during this +explicit dependency build. `TRTMC_EDGELLM_WHEELHOUSE` selects a complete offline +Python wheelhouse; `TRTMC_EDGELLM_GIT_MIRROR` optionally supplies a local Git +mirror, still checked out at the immutable upstream commit. + +Upstream 0.10.1 does not export a CMake SDK package, so these compact templates +supply that installation boundary. `EdgeLLM::Core` exposes the installed static +core, headers, CuTe archive and native dependencies. Consumers requiring CUDA +device linking enable separable compilation and device-symbol resolution. +`EdgeLLM::Plugin` identifies the plugin DSO; adapters load it, rather than linking +it twice. `EdgeLLM_PYTHON_EXECUTABLE` and `EdgeLLM_BUILDER_LAUNCHER` expose the +isolated upstream `experimental.builder.cli.main` API. + +`share/trtmc/edge-llm.json` records the pin, native architecture, CUDA/TensorRT +versions and prefix-relative Python/plugin paths. The manifest is written only +after successful installation. Build-tree package files live under +`build/_deps/edgellm/install`; `cmake --install` copies the package into the final +prefix. Package discovery rejects mismatched native CPU/GPU and SDK versions. +Model support and routing policies belong exclusively to the model families. + +Set `TRTMC_EDGELLM_ALL_KERNELS=ON` to provision all upstream operator groups +supported by the native GPU. Set `TRTMC_EDGELLM_ONNX=ON` to additionally install +the original Python exporter (including its pinned CPU PyTorch dependencies) +and original C++ `llm_build` executable as `bin/edgellm-onnx-build`. Families invoke +the exporter using the installed Python and obtain the native builder path from +`onnx_builder` in the manifest. These options describe SDK capabilities, not +qualified model support. Reusing an installed package that lacks a requested +capability is an error; no dependency installation occurs during model builds. +The CUDA and TensorRT shared libraries must remain available to the executable. + +Run the existing runtime and family checks against this installation +(some tests require a local GPU): + +```bash +ctest --test-dir build --output-on-failure +``` diff --git a/core/builder/tensorrt_model_connect/__init__.py b/core/builder/tensorrt_model_connect/__init__.py index 101f6dcb4b..407e0c8a14 100644 --- a/core/builder/tensorrt_model_connect/__init__.py +++ b/core/builder/tensorrt_model_connect/__init__.py @@ -3,12 +3,14 @@ """TensorRT Model Connect build API.""" -from .build import BuildRequest, build +from .build import BuildExecutionInputs, BuildRequest, NamedCheckpoint, build from .bundle_writer import BundleWriter from .graph_transform import GraphTransform __all__ = [ + "BuildExecutionInputs", "BuildRequest", + "NamedCheckpoint", "BundleWriter", "GraphTransform", "build", diff --git a/core/builder/tensorrt_model_connect/build.py b/core/builder/tensorrt_model_connect/build.py index f6fad3a61e..eb0e00bf46 100644 --- a/core/builder/tensorrt_model_connect/build.py +++ b/core/builder/tensorrt_model_connect/build.py @@ -7,6 +7,8 @@ import hashlib import importlib +import os +import platform import re import sys from dataclasses import dataclass @@ -20,6 +22,52 @@ _ID = re.compile(r"[a-z][a-z0-9_]*\Z") +@dataclass(frozen=True) +class NamedCheckpoint: + """One explicitly named local checkpoint; the family owns role semantics.""" + + role: str + model_dir: Path + + def __post_init__(self) -> None: + _validate_id("checkpoint role", self.role) + if not isinstance(self.model_dir, Path): + raise TypeError("checkpoint model_dir must be a Path") + if not self.model_dir.is_dir(): + raise ValueError(f"checkpoint must be an existing local directory: {self.model_dir}") + + +@dataclass(frozen=True) +class BuildExecutionInputs: + """Optional family-owned execution variant and immutable local companions. + + Core transports these inputs without interpreting variants, fetching models, + or inferring compatibility. A family must explicitly implement the capability. + """ + + variant: str + checkpoints: tuple[NamedCheckpoint, ...] = () + + def __post_init__(self) -> None: + _validate_id("execution variant", self.variant) + if not isinstance(self.checkpoints, tuple) or any( + not isinstance(checkpoint, NamedCheckpoint) for checkpoint in self.checkpoints + ): + raise TypeError("checkpoints must be a tuple of NamedCheckpoint values") + roles = [checkpoint.role for checkpoint in self.checkpoints] + if len(roles) != len(set(roles)): + raise ValueError("checkpoint roles must be unique") + self.validate_local() + + def validate_local(self) -> None: + """Recheck local availability before dispatch, without acquiring inputs.""" + for checkpoint in self.checkpoints: + if not checkpoint.model_dir.is_dir(): + raise ValueError( + f"checkpoint must be an existing local directory: {checkpoint.model_dir}" + ) + + @dataclass(frozen=True) class BuildRequest: """Inputs shared by the build core and one family-owned builder.""" @@ -72,6 +120,61 @@ def __post_init__(self) -> None: raise ValueError("graph_transform must be callable when provided") +def subprocess_environment(overrides: dict[str, str], *, + prepend_paths: dict[str, str] | None = None) -> dict[str, str]: + """Copy the parent environment for one child without mutating process state. + + Callers own explicit tool settings; this helper only merges values and + prepends search paths using the executing platform's path separator. + """ + environment = os.environ.copy() + environment.update(overrides) + for name, value in (prepend_paths or {}).items(): + previous = environment.get(name) + environment[name] = value + (os.pathsep + previous if previous else "") + return environment + + +def cmake_prefixes() -> list[Path]: + """Return explicit standard CMake prefixes followed by the Python prefix.""" + prefixes = [ + Path(value) for value in os.environ.get("CMAKE_PREFIX_PATH", "").split(os.pathsep) if value + ] + return [*prefixes, Path(sys.prefix)] + + +def detect_local_platform() -> dict: + """Return executing GPU and native SDK identity without selecting a model. + + Returns: + OS/release, CPU architecture, GPU SM, CUDA and TensorRT versions. + + Raises: + ImportError: Native SDK Python bindings are unavailable. + RuntimeError: CUDA cannot identify the executing device. + """ + import tensorrt as trt + from cuda.bindings import runtime + + def checked(result): + if int(result[0]) != 0: + raise RuntimeError(f"CUDA device discovery failed: {result[0]}") + return result[1] + + device = checked(runtime.cudaGetDevice()) + gpu = checked(runtime.cudaGetDeviceProperties(device)) + cuda = checked(runtime.cudaRuntimeGetVersion()) + release = platform.freedesktop_os_release() if sys.platform == "linux" else {} + return { + "os": sys.platform, + "os_version": release.get("VERSION_ID", platform.release()), + "arch": platform.machine(), + "sm": gpu.major * 10 + gpu.minor, + "cuda_version": f"{cuda // 1000}.{cuda % 1000 // 10}", + "tensorrt_version": trt.__version__, + } + + def _validate_id(field: str, value: object) -> str: if not isinstance(value, str) or _ID.fullmatch(value) is None: raise ValueError( @@ -135,16 +238,36 @@ def select_backend(backend: str) -> None: _select_backend = select_backend # Compatibility for existing Python callers. -def build(request: BuildRequest) -> None: - """Run one family builder and publish its bundle on success.""" +def build(request: BuildRequest, *, execution: BuildExecutionInputs | None = None) -> None: + """Run one family builder and atomically publish its bundle on success. + + Explicit execution inputs require the optional family build_with_inputs hook. + Missing support fails before constructing the writer; failures never retry a + different variant or silently invoke the ordinary builder. + """ + + if execution is not None: + if not isinstance(execution, BuildExecutionInputs): + raise TypeError("execution must be BuildExecutionInputs") + execution.validate_local() family = _resolve_family(request) _select_backend(request.backend) family_module = _load_family(family) + extended_build = None + if execution is not None: + extended_build = getattr(family_module, "build_with_inputs", None) + if not callable(extended_build): + raise NotImplementedError( + f"family {family!r} does not support explicit build execution inputs" + ) writer = BundleWriter(request.output_path) try: with graph_transform(request.graph_transform): - family_module.build(request, writer) + if execution is None: + family_module.build(request, writer) + else: + extended_build(request, writer, execution) writer.finish() except BaseException: writer.abort() diff --git a/core/builder/tensorrt_model_connect/build_cli.py b/core/builder/tensorrt_model_connect/build_cli.py index 837f5e4c47..cc091cf26a 100644 --- a/core/builder/tensorrt_model_connect/build_cli.py +++ b/core/builder/tensorrt_model_connect/build_cli.py @@ -12,7 +12,7 @@ from pathlib import Path from typing import Sequence -from .build import BuildRequest, _load_family, build +from .build import BuildExecutionInputs, BuildRequest, NamedCheckpoint, _load_family, build from .model_support import ( AmbiguousFamilyError, FamilyResolutionError, @@ -46,6 +46,14 @@ def _parser(prepare_family: object | None = None) -> argparse.ArgumentParser: build_parser.add_argument("--fp32-layer", type=int, action="append", default=[]) build_parser.add_argument("--dynamic-kv-cache", action="store_true") build_parser.add_argument("--verbose", action="store_true") + build_parser.add_argument("--execution-variant", help="Explicit family-owned execution variant") + build_parser.add_argument( + "--companion", + action="append", + default=[], + metavar="ROLE=LOCAL_DIR", + help="Named existing local checkpoint; repeat for multiple distinct roles", + ) prepare_parser = commands.add_parser( "prepare-structure", help="Prepare one structure request without rebuilding its model bundle", @@ -64,6 +72,25 @@ def _parser(prepare_family: object | None = None) -> argparse.ArgumentParser: return parser +def _execution_inputs(args: argparse.Namespace) -> BuildExecutionInputs | None: + """Parse only explicit local inputs; no variant list or model acquisition.""" + if args.command != "build": + return None + if args.execution_variant is None: + if args.companion: + raise ValueError("--companion requires --execution-variant") + return None + checkpoints = [] + for value in args.companion: + role, separator, directory = value.partition("=") + if not separator or not role or not directory: + raise ValueError("--companion must be ROLE=LOCAL_DIR") + if "://" in directory: + raise ValueError("--companion requires a local directory, not a URI") + checkpoints.append(NamedCheckpoint(role, Path(directory))) + return BuildExecutionInputs(args.execution_variant, tuple(checkpoints)) + + def main(argv: Sequence[str] | None = None) -> int: arguments = list(sys.argv[1:] if argv is None else argv) base_parser = _parser() @@ -88,6 +115,7 @@ def main(argv: Sequence[str] | None = None) -> int: return 2 family_module = _load_family(family) if preliminary.command == "prepare-structure" else None args = _parser(family_module).parse_args(arguments) + execution = _execution_inputs(args) if args.command == "prepare-structure": prepare = getattr(family_module, "prepare_structure_request", None) if not callable(prepare): @@ -111,27 +139,29 @@ def main(argv: Sequence[str] | None = None) -> int: f"family {family!r} does not support task {task!r}; " f"choose one of: {', '.join(support.tasks)}" ) - build( - BuildRequest( - model_dir=model_dir, - output_path=args.output, - precision=args.precision or support.default_precision, - backend=args.backend, - family=family, - task=task, - max_sequence_length=args.max_sequence_length, - image_height=args.image_height, - image_width=args.image_width, - video_num_frames=args.video_num_frames, - max_batch_size=args.max_batch_size, - tensor_parallel_size=args.tensor_parallel_size, - context_parallel_size=args.context_parallel_size, - quantization=args.quantization, - fp32_layers=tuple(args.fp32_layer), - dynamic_kv_cache=args.dynamic_kv_cache, - verbose=args.verbose, - ) + request = BuildRequest( + model_dir=model_dir, + output_path=args.output, + precision=args.precision or support.default_precision, + backend=args.backend, + family=family, + task=task, + max_sequence_length=args.max_sequence_length, + image_height=args.image_height, + image_width=args.image_width, + video_num_frames=args.video_num_frames, + max_batch_size=args.max_batch_size, + tensor_parallel_size=args.tensor_parallel_size, + context_parallel_size=args.context_parallel_size, + quantization=args.quantization, + fp32_layers=tuple(args.fp32_layer), + dynamic_kv_cache=args.dynamic_kv_cache, + verbose=args.verbose, ) + if execution is None: + build(request) + else: + build(request, execution=execution) return 0 diff --git a/core/builder/tests/test_build.py b/core/builder/tests/test_build.py index 5eff69de1b..fc94a5a99c 100644 --- a/core/builder/tests/test_build.py +++ b/core/builder/tests/test_build.py @@ -4,6 +4,7 @@ from __future__ import annotations import importlib +from contextlib import contextmanager import sys from dataclasses import FrozenInstanceError, replace from pathlib import Path @@ -11,7 +12,7 @@ import pytest -from tensorrt_model_connect import BuildRequest +from tensorrt_model_connect import BuildExecutionInputs, BuildRequest, NamedCheckpoint, build_cli build_core = importlib.import_module("tensorrt_model_connect.build") @@ -287,3 +288,295 @@ def abort(self) -> None: with pytest.raises(OSError, match="publish failed"): build_core.build(_request(tmp_path)) assert events == ["finish", "abort"] + + +@pytest.mark.parametrize("value", ["", "/one", "/one:/two", ":/one::/two:"]) +def test_cmake_prefixes_preserve_standard_search_order(monkeypatch, value): + monkeypatch.setenv("CMAKE_PREFIX_PATH", value) + monkeypatch.setattr(build_core.sys, "prefix", "/python") + expected = [Path(item) for item in value.split(build_core.os.pathsep) if item] + assert build_core.cmake_prefixes() == [*expected, Path("/python")] + + +def test_cmake_prefixes_without_environment_use_python_prefix(monkeypatch): + monkeypatch.delenv("CMAKE_PREFIX_PATH", raising=False) + monkeypatch.setattr(build_core.sys, "prefix", "/python") + assert build_core.cmake_prefixes() == [Path("/python")] + # Constructing explicit child-tool settings must not change the caller's + # package search order or mutate an inherited search path. + monkeypatch.setenv("TEST_TOOL_SEARCH_PATH", "/original") + monkeypatch.delenv("TEST_TOOL_NEW_PATH", raising=False) + child = build_core.subprocess_environment( + {"CMAKE_PREFIX_PATH": "/child"}, + prepend_paths={"TEST_TOOL_SEARCH_PATH": "/first", "TEST_TOOL_NEW_PATH": "/new"}, + ) + assert child["CMAKE_PREFIX_PATH"] == "/child" + assert child["TEST_TOOL_SEARCH_PATH"] == "/first" + build_core.os.pathsep + "/original" + assert child["TEST_TOOL_NEW_PATH"] == "/new" + assert build_core.cmake_prefixes() == [Path("/python")] + assert build_core.os.environ["TEST_TOOL_SEARCH_PATH"] == "/original" + assert "TEST_TOOL_NEW_PATH" not in build_core.os.environ + + +@pytest.fixture +def native_platform_bindings(monkeypatch): + from unittest.mock import Mock + + runtime = SimpleNamespace( + cudaGetDevice=Mock(return_value=(0, 3)), + cudaGetDeviceProperties=Mock(return_value=(0, SimpleNamespace(major=8, minor=6))), + cudaRuntimeGetVersion=Mock(return_value=(0, 13030)), + ) + monkeypatch.setitem(sys.modules, "tensorrt", SimpleNamespace(__version__="11.1.0.106")) + monkeypatch.setitem(sys.modules, "cuda.bindings", SimpleNamespace(runtime=runtime)) + monkeypatch.setattr(build_core.sys, "platform", "linux") + monkeypatch.setattr(build_core.platform, "machine", lambda: "x86_64") + monkeypatch.setattr( + build_core.platform, "freedesktop_os_release", lambda: {"VERSION_ID": "24.04"} + ) + monkeypatch.setattr(build_core.platform, "release", lambda: "fallback-release") + return runtime + + +def test_native_platform_uses_executing_cuda_device_and_full_sdk(native_platform_bindings): + assert build_core.detect_local_platform() == { + "os": "linux", + "os_version": "24.04", + "arch": "x86_64", + "sm": 86, + "cuda_version": "13.3", + "tensorrt_version": "11.1.0.106", + } + native_platform_bindings.cudaGetDevice.assert_called_once_with() + native_platform_bindings.cudaGetDeviceProperties.assert_called_once_with(3) + native_platform_bindings.cudaRuntimeGetVersion.assert_called_once_with() + + +@pytest.mark.parametrize( + "failing", ["cudaGetDevice", "cudaGetDeviceProperties", "cudaRuntimeGetVersion"] +) +def test_native_platform_propagates_cuda_discovery_failure(native_platform_bindings, failing): + getattr(native_platform_bindings, failing).return_value = (35,) + with pytest.raises(RuntimeError, match="CUDA device discovery failed: 35"): + build_core.detect_local_platform() + + +def test_native_platform_retains_nonlinux_identity(native_platform_bindings, monkeypatch): + monkeypatch.setattr(build_core.sys, "platform", "win32") + result = build_core.detect_local_platform() + assert result["os"] == "win32" + assert result["os_version"] == "fallback-release" + + +def execution_request(root: Path) -> BuildRequest: + return BuildRequest(root, root / "model.bundle", "example", "text_generation", "fp16") + + +def inputs(root: Path) -> BuildExecutionInputs: + return BuildExecutionInputs("paired", (NamedCheckpoint("draft", root),)) + + +def test_execution_inputs_are_immutable(tmp_path): + execution = inputs(tmp_path) + with pytest.raises(FrozenInstanceError): + execution.variant = "other" + with pytest.raises(FrozenInstanceError): + execution.checkpoints[0].role = "other" + assert execution.checkpoints[0].model_dir is tmp_path + + +@pytest.mark.parametrize("value", ["", "../bad", "UPPER", "a-b", "a.b"]) +def test_invalid_role_and_variant(tmp_path, value): + with pytest.raises(ValueError, match="lowercase identifier"): + NamedCheckpoint(value, tmp_path) + with pytest.raises(ValueError, match="lowercase identifier"): + BuildExecutionInputs(value) + + +def test_execution_requires_immutable_typed_companions(tmp_path): + checkpoint = NamedCheckpoint("draft", tmp_path) + with pytest.raises(TypeError, match="tuple"): + BuildExecutionInputs("paired", [checkpoint]) + with pytest.raises(TypeError, match="NamedCheckpoint"): + BuildExecutionInputs("paired", (object(),)) + with pytest.raises(ValueError, match="unique"): + BuildExecutionInputs("paired", (checkpoint, checkpoint)) + with pytest.raises(TypeError, match="Path"): + NamedCheckpoint("draft", str(tmp_path)) + + +def test_local_checkpoint_required_and_rechecked(tmp_path, monkeypatch): + with pytest.raises(ValueError, match="existing local directory"): + NamedCheckpoint("draft", tmp_path / "missing") + file = tmp_path / "file" + file.write_text("not a directory") + with pytest.raises(ValueError, match="existing local directory"): + NamedCheckpoint("draft", file) + directory = tmp_path / "companion" + directory.mkdir() + execution = inputs(directory) + directory.rmdir() + monkeypatch.setattr(build_core, "_select_backend", lambda _: pytest.fail("backend touched")) + with pytest.raises(ValueError, match="existing local directory"): + build_core.build(execution_request(tmp_path), execution=execution) + + +def test_untyped_execution_fails_before_side_effects(tmp_path, monkeypatch): + monkeypatch.setattr(build_core, "_select_backend", lambda _: pytest.fail("backend touched")) + with pytest.raises(TypeError, match="BuildExecutionInputs"): + build_core.build(execution_request(tmp_path), execution={"variant": "paired"}) + + +@pytest.mark.parametrize("hook", [None, 17]) +def test_missing_capability_fails_before_writer(tmp_path, monkeypatch, hook): + monkeypatch.setattr( + build_core, + "_load_family", + lambda _: SimpleNamespace( + build=lambda *_: pytest.fail("ordinary fallback invoked"), build_with_inputs=hook + ), + ) + monkeypatch.setattr(build_core, "BundleWriter", lambda _: pytest.fail("writer created")) + with pytest.raises(NotImplementedError, match="does not support explicit"): + build_core.build(execution_request(tmp_path), execution=inputs(tmp_path)) + + +def test_exact_envelope_and_existing_transaction_are_preserved(tmp_path, monkeypatch): + events = [] + original_request, execution = execution_request(tmp_path), inputs(tmp_path) + + @contextmanager + def transform(value): + assert value is original_request.graph_transform + events.append("enter") + yield + events.append("exit") + + class Writer: + def __init__(self, path): + assert path == original_request.output_path + events.append("writer") + + def finish(self): + events.append("finish") + + def abort(self): + pytest.fail("unexpected abort") + + def extended(actual_request, writer, actual_execution): + assert actual_request is original_request and actual_execution is execution + assert isinstance(writer, Writer) + events.append("extended") + + monkeypatch.setattr(build_core, "graph_transform", transform) + monkeypatch.setattr(build_core, "BundleWriter", Writer) + monkeypatch.setattr( + build_core, + "_load_family", + lambda _: SimpleNamespace( + build=lambda *_: pytest.fail("ordinary fallback invoked"), build_with_inputs=extended + ), + ) + build_core.build(original_request, execution=execution) + assert events == ["writer", "enter", "extended", "exit", "finish"] + + +@pytest.mark.parametrize("failure", [RuntimeError("failed"), KeyboardInterrupt()]) +def test_explicit_failure_aborts_real_writer_without_replacing_bundle( + tmp_path, monkeypatch, failure +): + build_request = execution_request(tmp_path) + build_request.output_path.write_bytes(b"previous valid publication") + + def extended(actual, writer, execution): + writer.set_header(family=actual.family, task=actual.task, backend=actual.backend) + writer.add_json("test.json", {"variant": execution.variant}) + raise failure + + monkeypatch.setattr( + build_core, + "_load_family", + lambda _: SimpleNamespace( + build=lambda *_: pytest.fail("ordinary fallback invoked"), build_with_inputs=extended + ), + ) + with pytest.raises(type(failure)) as caught: + build_core.build(build_request, execution=inputs(tmp_path)) + assert caught.value is failure + assert build_request.output_path.read_bytes() == b"previous valid publication" + assert sorted(path.name for path in tmp_path.iterdir()) == ["model.bundle"] + + +def test_variant_without_companions_is_explicit_and_supported(tmp_path, monkeypatch): + seen = [] + + def extended(actual, writer, execution): + seen.append(execution) + writer.set_header(family=actual.family, task=actual.task, backend=actual.backend) + writer.add_json("test.json", {"variant": execution.variant}) + + monkeypatch.setattr( + build_core, "_load_family", lambda _: SimpleNamespace(build_with_inputs=extended) + ) + execution = BuildExecutionInputs("embedded") + build_core.build(execution_request(tmp_path), execution=execution) + assert seen == [execution] and (tmp_path / "model.bundle").is_file() + + +def test_ordinary_build_ignores_available_optional_hook(tmp_path, monkeypatch): + def ordinary(actual, writer): + writer.set_header(family=actual.family, task=actual.task, backend=actual.backend) + writer.add_json("test.json", {"ordinary": True}) + + monkeypatch.setattr( + build_core, + "_load_family", + lambda _: SimpleNamespace( + build=ordinary, build_with_inputs=lambda *_: pytest.fail("optional hook invoked") + ), + ) + build_core.build(execution_request(tmp_path)) + + +@pytest.mark.parametrize( + "options", + [ + ["--companion", "draft=/missing"], + ["--execution-variant", "paired", "--companion", "missing_separator"], + ["--execution-variant", "paired", "--companion", "=path"], + ["--execution-variant", "paired", "--companion", "draft="], + ["--execution-variant", "paired", "--companion", "draft=https://example.com/model"], + ["--execution-variant", ""], + ], +) +def test_bad_cli_execution_rejected_before_primary_model_acquisition(monkeypatch, options): + monkeypatch.setattr(build_cli, "_resolve_model", lambda *_: pytest.fail("model acquisition")) + with pytest.raises(ValueError): + build_cli.main(["build", "model-id", "-o", "/tmp/example.bundle", *options]) + + +def test_cli_forwards_exact_variant_and_named_local_paths(tmp_path, monkeypatch): + (tmp_path / "config.json").write_text('{"model_type":"gpt2"}') + companion = tmp_path / "checkpoint=local" + companion.mkdir() + seen = [] + monkeypatch.setattr( + build_cli, "build", lambda request, **kwargs: seen.append((request, kwargs)) + ) + build_cli.main( + [ + "build", + str(tmp_path), + "-o", + str(tmp_path / "out.bundle"), + "--execution-variant", + "paired", + "--companion", + f"draft={companion}", + ] + ) + actual_request, kwargs = seen[0] + assert actual_request.model_dir == tmp_path + assert kwargs == { + "execution": BuildExecutionInputs("paired", (NamedCheckpoint("draft", companion),)) + } diff --git a/core/runtime/bundle/bundle_format.cpp b/core/runtime/bundle/bundle_format.cpp index eb11f20674..2c1661f088 100644 --- a/core/runtime/bundle/bundle_format.cpp +++ b/core/runtime/bundle/bundle_format.cpp @@ -6,6 +6,7 @@ #include "runtime/bundle/bundle_format.h" #include +#include #include #include #include @@ -215,6 +216,32 @@ std::vector BundleReader::read_section(std::string_view name) const { return data; } +void BundleReader::copy_section(std::string_view name, std::ostream& output) const { + const auto* section = find_section(name); + if (section == nullptr) + throw std::runtime_error("Bundle section not found: " + std::string(name)); + const auto offset = checked_section_file_offset(*section, data_offset_, file_size_, path_); + if (offset > static_cast(std::numeric_limits::max())) + throw std::runtime_error("Bundle section has an unsupported file offset: " + path_); + std::ifstream input(path_, std::ios::binary); + input.seekg(static_cast(offset)); + if (!input || !output) + throw std::runtime_error("Cannot copy bundle section: " + std::string(name)); + std::array buffer; + auto remaining = section->length; + while (remaining != 0) { + const auto count = + static_cast(std::min(remaining, buffer.size())); + input.read(buffer.data(), count); + if (!input) + throw std::runtime_error("Failed reading bundle section: " + std::string(name)); + output.write(buffer.data(), count); + if (!output) + throw std::runtime_error("Failed writing bundle section: " + std::string(name)); + remaining -= static_cast(count); + } +} + BundleInfo InspectBundle(const std::string& bundle_path) { return BundleReader(bundle_path).info(); } diff --git a/core/runtime/include/trtmc/bundle.h b/core/runtime/include/trtmc/bundle.h index 154d697ef1..2b32776d2e 100644 --- a/core/runtime/include/trtmc/bundle.h +++ b/core/runtime/include/trtmc/bundle.h @@ -6,6 +6,7 @@ #pragma once #include +#include #include #include #include @@ -44,6 +45,12 @@ class BundleReader { const BundleSectionInfo* find_section(std::string_view name) const noexcept; std::vector read_section(std::string_view name) const; + /// Copy a named section to an output stream using bounded working memory. + /// @param name Validated bundle section name. + /// @param output Caller-owned stream; may contain partial data on failure. + /// @throws std::runtime_error If the section is absent or input/output fails. + void copy_section(std::string_view name, std::ostream& output) const; + private: std::string path_; BundleInfo info_; diff --git a/core/runtime/tests/test_bundle_format_v1.cpp b/core/runtime/tests/test_bundle_format_v1.cpp index 0da636b49b..6c5faa420f 100644 --- a/core/runtime/tests/test_bundle_format_v1.cpp +++ b/core/runtime/tests/test_bundle_format_v1.cpp @@ -9,6 +9,7 @@ #include #include #include +#include #include #include #include @@ -52,6 +53,36 @@ bool read_throws(const std::filesystem::path& path) { } } +/// Verify bounded-chunk section boundaries, empty sections and late I/O failures. +void test_copy_section(const std::filesystem::path& directory) { + const auto path = directory / "stream.bundle"; + const std::string payload(2 * 64 * 1024 + 17, 'x'); + const std::string header = + R"({"format":1,"family":"fake","task":"text","backend":"fake","sections":{"data":{"offset":3,"length":)" + + std::to_string(payload.size()) + R"(},"empty":{"offset":0,"length":0}}})"; + write_bundle(path, header, "PRE" + payload + "POST"); + const trtmc::BundleReader reader(path.string()); + std::ostringstream output; + reader.copy_section("data", output); + check(output.str() == payload, "stream copy preserves boundaries across chunks"); + reader.copy_section("empty", output); + check(output.str() == payload, "empty section appends nothing"); + auto fails = [&](const char* section, std::ostream& destination) { + try { + reader.copy_section(section, destination); + return false; + } catch (const std::runtime_error&) { + return true; + } + }; + check(fails("missing", output), "stream copy rejects missing sections"); + std::ostringstream broken; + broken.setstate(std::ios::badbit); + check(fails("data", broken), "stream copy reports output failure"); + std::filesystem::resize_file(path, 16 + header.size() + 3 + payload.size() - 1); + check(fails("data", output), "stream copy detects truncation after validation"); +} + } // namespace int main() { @@ -103,6 +134,7 @@ int main() { "PLAN"); check(read_throws(out_of_bounds), "out of bounds section rejected"); + test_copy_section(directory); std::filesystem::remove_all(directory); std::cerr << (failures == 0 ? "ALL PASSED\n" : "SOME FAILED\n"); return failures; diff --git a/tools/tests/test_architecture.py b/tools/tests/test_architecture.py index 5986c7293e..c83201ca3b 100644 --- a/tools/tests/test_architecture.py +++ b/tools/tests/test_architecture.py @@ -631,7 +631,15 @@ def test_shared_python_and_native_trees_are_closed_minimal_sets() -> None: "qualification_tests/benchmark_qualification/performance/tests/test_structured_output_contracts.py", "qualification_tests/benchmark_qualification/performance/tests/test_timing_contracts.py", } - expected_cmake = {"cmake/trtmcConfig.cmake.in"} + expected_cmake = { + "cmake/trtmcConfig.cmake.in", + "cmake/EdgeLLM.cmake", + "cmake/edgellm/CheckNative.cmake", + "cmake/edgellm/EdgeLLMConfig.cmake.in", + "cmake/edgellm/Install.cmake.in", + "cmake/edgellm/Prepare.cmake.in", + "cmake/edgellm/README.md", + } expected_third_party = { "third_party/stb/stb_image.h", "third_party/stb/stb_image_resize2.h", diff --git a/website/docs/api/python-builder.md b/website/docs/api/python-builder.md index 980b803521..643776e18f 100644 --- a/website/docs/api/python-builder.md +++ b/website/docs/api/python-builder.md @@ -29,6 +29,36 @@ resolved API directly. decides whether that directory is a Hugging Face snapshot or a prepared checkpoint; `BuildRequest` does not perform another discovery pass. +## Optional execution inputs + +`build(request, execution=...)` accepts an optional, frozen +`BuildExecutionInputs` descriptor. It contains a family-owned `variant` string +and a tuple of `NamedCheckpoint(role, model_dir)` descriptors. Import these +public types from `tensorrt_model_connect`. Companion directories must already +exist locally; the core does not download them or infer compatible model pairs. +Roles must be unique. Variant and role names are lowercase identifiers. + +Providing execution inputs requires the selected family to implement +`build_with_inputs(request, writer, execution)`. The family validates the +variant, checkpoint roles, compatibility and execution semantics. A missing +hook fails before bundle creation; the core never substitutes ordinary +base-only generation or another variant. With no execution inputs, the +existing `build(request, writer)` family call is unchanged. + +The build CLI exposes the same optional contract: + +```text +trtmc build LOCAL_TARGET -o model.bundle \ + --execution-variant FAMILY_VARIANT \ + --companion ROLE=LOCAL_COMPANION_DIR +``` + +Replace the uppercase placeholders with values from the selected family's +recipe; they are not literal supported identifiers. Repeat `--companion` only +for distinct roles. A companion requires `--execution-variant`; URLs and +implicit companion downloads are unsupported. The generic API does not itself +qualify any speculative algorithm or checkpoint pair. + ## Optional graph transform `BuildRequest.graph_transform` is an in-place callback invoked on the completed diff --git a/website/docs/architecture/build-pipeline.md b/website/docs/architecture/build-pipeline.md index 3389c9cdfd..243172f4af 100644 --- a/website/docs/architecture/build-pipeline.md +++ b/website/docs/architecture/build-pipeline.md @@ -11,6 +11,7 @@ model ID/local snapshot -> choose family default task or validate --task -> import the selected families..model -> call build(BuildRequest, BundleWriter) + or the explicitly requested family build_with_inputs hook -> atomically publish format-1 bundle ``` @@ -36,12 +37,20 @@ sizes, family-owned quantization selection, FP32 layer overrides, direct dynamic-KV opt-in, and optional graph transform. Each family must implement or explicitly reject every non-default request it receives. +Optional `BuildExecutionInputs` travel separately from `BuildRequest`. Core +checks descriptor types, unique roles and existing local directories; only +the selected family interprets variant names and companion compatibility. +See the [Python Build API](../api/python-builder.md#optional-execution-inputs). + ## Family build `families//model.py` exposes a plain `build(request, writer)` function. It reads model config and weights, constructs the TensorRT network and engines, and writes family-owned named sections. Builder inheritance and shared model -topology helpers are forbidden. +topology helpers are forbidden. A family may instead delegate a complete +network to an installed optimized runtime through a family-owned adapter. +Model-specific admission, builder mapping, runtime orchestration and validation +remain in that family; shared dependency provisioning contains no model policy. The graph-transform callback, when present, receives the live TensorRT network immediately before serialization. This is the build-time half of the explicit diff --git a/website/docs/user-guides/configure-runtime.md b/website/docs/user-guides/configure-runtime.md index 82f3f62dca..fbbc07a82f 100644 --- a/website/docs/user-guides/configure-runtime.md +++ b/website/docs/user-guides/configure-runtime.md @@ -31,3 +31,26 @@ Unsupported values fail; they are not silently ignored. See [Configuration and Backends](../features/config-and-backends.md), [Quantization](../features/quantization.md), and [Multi-Device Execution](../features/multi-device.md). + +## Optional native Edge-LLM SDK + +Provision Edge-LLM explicitly when building Model Connect, not during model +builds or inference. `TRTMC_ENABLE_EDGELLM=ON` selects the public Edge-LLM +0.10.1 snapshot at `e8b29522938901f6df19ebeedd4b69bc8edbcd97`. The default is +`OFF`. Configure and build on the inference GPU host; cross compilation is +rejected. The package must match the native CPU/GPU and CUDA/TensorRT stack. +Runtime compilation must also use the exact pinned JSON dependency headers; +the SDK rejects same-version development headers with an incompatible C++ ABI. + +- `TRTMC_EDGELLM_ALL_KERNELS=ON` requests all upstream operator groups supported + by the local GPU. +- `TRTMC_EDGELLM_ONNX=ON` also installs the original Python exporter and native + C++ ONNX engine builder. It does not select a model's build flow. +- `CMAKE_PREFIX_PATH` points builders and runtime compilation to the installed + SDK. Reuse fails explicitly if a requested capability is absent. + +Follow the repository's [pinned SDK installation instructions](https://github.com/NVIDIA/TensorRT-Model-Connect/blob/main/cmake/edgellm/README.md) +for dependencies, native architecture selection and offline provisioning. +Each family decides whether and how to use the package. Installing the SDK +is not evidence that a model, precision, input modality or execution variant +has passed validation. Ordinary model builds never install missing SDK tools. From 9dbcac7d0c6364ab56f42adeb2c3ab9bec2c16b4 Mon Sep 17 00:00:00 2001 From: Joshua Calafato Date: Mon, 21 Sep 2026 17:03:37 +0000 Subject: [PATCH 02/20] fix(build): enforce native SDK compatibility Select TensorRT headers and libraries from one root, record the native SDK version, and tolerate missing Linux release metadata. Preserve fail-fast execution-input validation with main CLI discovery. Signed-off-by: Joshua Calafato --- cmake/edgellm/EdgeLLMConfig.cmake.in | 22 ++++++++++++++++--- cmake/edgellm/Install.cmake.in | 5 ++--- core/builder/tensorrt_model_connect/build.py | 10 ++++++--- .../tensorrt_model_connect/build_cli.py | 2 +- core/builder/tests/test_build.py | 13 +++++++++-- 5 files changed, 40 insertions(+), 12 deletions(-) diff --git a/cmake/edgellm/EdgeLLMConfig.cmake.in b/cmake/edgellm/EdgeLLMConfig.cmake.in index bade6c9dba..63f05b8dd8 100644 --- a/cmake/edgellm/EdgeLLMConfig.cmake.in +++ b/cmake/edgellm/EdgeLLMConfig.cmake.in @@ -31,9 +31,25 @@ if(NOT CUDAToolkit_VERSION_MAJOR EQUAL @CUDAToolkit_VERSION_MAJOR@ OR endif() set(EdgeLLM_PYTHON_EXECUTABLE "${EdgeLLM_PREFIX}/libexec/trtmc-edge-llm/bin/python") set(EdgeLLM_BUILDER_LAUNCHER "${EdgeLLM_PREFIX}/bin/edgellm-builder") -find_path(EdgeLLM_TRT_INCLUDE_DIR NvInfer.h HINTS "$ENV{TRT_ROOT}" "@TRTMC_EDGELLM_TRT_ROOT@" PATH_SUFFIXES include REQUIRED) -find_library(EdgeLLM_TRT_LIBRARY nvinfer HINTS "$ENV{TRT_ROOT}" "@TRTMC_EDGELLM_TRT_ROOT@" PATH_SUFFIXES lib lib64 REQUIRED) -find_library(EdgeLLM_PARSER_LIBRARY nvonnxparser HINTS "$ENV{TRT_ROOT}" "@TRTMC_EDGELLM_TRT_ROOT@" PATH_SUFFIXES lib lib64 REQUIRED) +# Select one SDK root before searching: never mix headers and libraries from +# different installations, including values left in an earlier CMake cache. +if(TRTMC_EDGELLM_TRT_ROOT) + set(_edge_trt_root "${TRTMC_EDGELLM_TRT_ROOT}") +elseif(DEFINED ENV{TRT_ROOT} AND NOT "$ENV{TRT_ROOT}" STREQUAL "") + set(_edge_trt_root "$ENV{TRT_ROOT}") +else() + set(_edge_trt_root "@TRTMC_EDGELLM_TRT_ROOT@") +endif() +if(NOT _edge_trt_root) + message(FATAL_ERROR "Set TRTMC_EDGELLM_TRT_ROOT to one complete native TensorRT SDK") +endif() +foreach(_artifact IN ITEMS EdgeLLM_TRT_INCLUDE_DIR EdgeLLM_TRT_LIBRARY EdgeLLM_PARSER_LIBRARY) + unset(${_artifact}) + unset(${_artifact} CACHE) +endforeach() +find_path(EdgeLLM_TRT_INCLUDE_DIR NvInfer.h PATHS "${_edge_trt_root}/include" NO_DEFAULT_PATH REQUIRED) +find_library(EdgeLLM_TRT_LIBRARY nvinfer PATHS "${_edge_trt_root}/lib" "${_edge_trt_root}/lib64" NO_DEFAULT_PATH REQUIRED) +find_library(EdgeLLM_PARSER_LIBRARY nvonnxparser PATHS "${_edge_trt_root}/lib" "${_edge_trt_root}/lib64" NO_DEFAULT_PATH REQUIRED) _edgellm_trt_version("${EdgeLLM_TRT_INCLUDE_DIR}" _edge_current_trt) if(NOT _edge_current_trt STREQUAL EdgeLLM_TENSORRT_VERSION) message(FATAL_ERROR "EdgeLLM requires TensorRT ${EdgeLLM_TENSORRT_VERSION}; found ${_edge_current_trt}") diff --git a/cmake/edgellm/Install.cmake.in b/cmake/edgellm/Install.cmake.in index 562c720b40..e71bd217df 100644 --- a/cmake/edgellm/Install.cmake.in +++ b/cmake/edgellm/Install.cmake.in @@ -29,7 +29,6 @@ set(_onnx false) if("@TRTMC_EDGELLM_ONNX@") set(_onnx true) endif() -execute_process(COMMAND "@_edge_python@" -I -c "import tensorrt; print(tensorrt.__version__)" - OUTPUT_VARIABLE _trt_version OUTPUT_STRIP_TRAILING_WHITESPACE COMMAND_ERROR_IS_FATAL ANY) +# Bundle compatibility uses the complete native SDK version, not the Python wheel label. # Paths are relative to the installation prefix, preserving relocatability. -file(WRITE "@_edge_prefix@/share/trtmc/edge-llm.json" "{\n \"schema_version\": 1,\n \"version\": \"@_edge_version@\",\n \"revision\": \"@_edge_revision@\",\n \"arch\": \"@CMAKE_SYSTEM_PROCESSOR@\",\n \"architectures\": [@TRTMC_EDGELLM_CUDA_ARCHITECTURE@],\n \"cuda_version\": \"@CUDAToolkit_VERSION_MAJOR@.@CUDAToolkit_VERSION_MINOR@\",\n \"tensorrt_version\": \"${_trt_version}\",\n \"python\": \"libexec/trtmc-edge-llm/bin/python\",\n \"builder\": \"bin/edgellm-builder\",\n \"all_native_kernels\": ${_all_kernels},\n \"onnx\": ${_onnx},\n \"onnx_builder\": \"bin/edgellm-onnx-build\",\n \"plugin\": \"lib/libNvInfer_edgellm_plugin.so\"\n}\n") +file(WRITE "@_edge_prefix@/share/trtmc/edge-llm.json" "{\n \"schema_version\": 1,\n \"version\": \"@_edge_version@\",\n \"revision\": \"@_edge_revision@\",\n \"arch\": \"@CMAKE_SYSTEM_PROCESSOR@\",\n \"architectures\": [@TRTMC_EDGELLM_CUDA_ARCHITECTURE@],\n \"cuda_version\": \"@CUDAToolkit_VERSION_MAJOR@.@CUDAToolkit_VERSION_MINOR@\",\n \"tensorrt_version\": \"@_edge_trt_version@\",\n \"python\": \"libexec/trtmc-edge-llm/bin/python\",\n \"builder\": \"bin/edgellm-builder\",\n \"all_native_kernels\": ${_all_kernels},\n \"onnx\": ${_onnx},\n \"onnx_builder\": \"bin/edgellm-onnx-build\",\n \"plugin\": \"lib/libNvInfer_edgellm_plugin.so\"\n}\n") diff --git a/core/builder/tensorrt_model_connect/build.py b/core/builder/tensorrt_model_connect/build.py index eb0e00bf46..aa19f29c4e 100644 --- a/core/builder/tensorrt_model_connect/build.py +++ b/core/builder/tensorrt_model_connect/build.py @@ -120,8 +120,9 @@ def __post_init__(self) -> None: raise ValueError("graph_transform must be callable when provided") -def subprocess_environment(overrides: dict[str, str], *, - prepend_paths: dict[str, str] | None = None) -> dict[str, str]: +def subprocess_environment( + overrides: dict[str, str], *, prepend_paths: dict[str, str] | None = None +) -> dict[str, str]: """Copy the parent environment for one child without mutating process state. Callers own explicit tool settings; this helper only merges values and @@ -164,7 +165,10 @@ def checked(result): device = checked(runtime.cudaGetDevice()) gpu = checked(runtime.cudaGetDeviceProperties(device)) cuda = checked(runtime.cudaRuntimeGetVersion()) - release = platform.freedesktop_os_release() if sys.platform == "linux" else {} + try: + release = platform.freedesktop_os_release() if sys.platform == "linux" else {} + except OSError: + release = {} return { "os": sys.platform, "os_version": release.get("VERSION_ID", platform.release()), diff --git a/core/builder/tensorrt_model_connect/build_cli.py b/core/builder/tensorrt_model_connect/build_cli.py index cc091cf26a..64f8d3c600 100644 --- a/core/builder/tensorrt_model_connect/build_cli.py +++ b/core/builder/tensorrt_model_connect/build_cli.py @@ -102,6 +102,7 @@ def main(argv: Sequence[str] | None = None) -> int: ): base_parser.error("MODEL must immediately follow prepare-structure") preliminary, _ = base_parser.parse_known_args(arguments) + execution = _execution_inputs(preliminary) model_dir = _resolve_model(preliminary.model, preliminary.revision) metadata = load_model_metadata(model_dir) try: @@ -115,7 +116,6 @@ def main(argv: Sequence[str] | None = None) -> int: return 2 family_module = _load_family(family) if preliminary.command == "prepare-structure" else None args = _parser(family_module).parse_args(arguments) - execution = _execution_inputs(args) if args.command == "prepare-structure": prepare = getattr(family_module, "prepare_structure_request", None) if not callable(prepare): diff --git a/core/builder/tests/test_build.py b/core/builder/tests/test_build.py index fc94a5a99c..b71172efc6 100644 --- a/core/builder/tests/test_build.py +++ b/core/builder/tests/test_build.py @@ -338,10 +338,19 @@ def native_platform_bindings(monkeypatch): return runtime -def test_native_platform_uses_executing_cuda_device_and_full_sdk(native_platform_bindings): +@pytest.mark.parametrize("release_available", [True, False]) +def test_native_platform_uses_executing_cuda_device_and_full_sdk( + native_platform_bindings, monkeypatch, release_available +): + if not release_available: + from unittest.mock import Mock + + monkeypatch.setattr( + build_core.platform, "freedesktop_os_release", Mock(side_effect=OSError("missing")) + ) assert build_core.detect_local_platform() == { "os": "linux", - "os_version": "24.04", + "os_version": "24.04" if release_available else "fallback-release", "arch": "x86_64", "sm": 86, "cuda_version": "13.3", From faca1394860575eb236ed062ce19caab39923cc9 Mon Sep 17 00:00:00 2001 From: Joshua Calafato Date: Mon, 21 Sep 2026 17:21:13 +0000 Subject: [PATCH 03/20] test(build): isolate execution input forwarding Signed-off-by: Joshua Calafato --- core/builder/tests/test_build.py | 9 ++++++++- 1 file changed, 8 insertions(+), 1 deletion(-) diff --git a/core/builder/tests/test_build.py b/core/builder/tests/test_build.py index b71172efc6..f6c5176823 100644 --- a/core/builder/tests/test_build.py +++ b/core/builder/tests/test_build.py @@ -565,7 +565,14 @@ def test_bad_cli_execution_rejected_before_primary_model_acquisition(monkeypatch def test_cli_forwards_exact_variant_and_named_local_paths(tmp_path, monkeypatch): - (tmp_path / "config.json").write_text('{"model_type":"gpt2"}') + from tensorrt_model_connect.model_support import FamilySupport + + (tmp_path / "config.json").write_text('{"model_type":"example_model"}') + monkeypatch.setattr( + build_cli, + "resolve_family", + lambda _: ("example_owner", FamilySupport(("example_task",), "example_task")), + ) companion = tmp_path / "checkpoint=local" companion.mkdir() seen = [] From 79616afb2af05e8c7288044d09d302ed465c3657 Mon Sep 17 00:00:00 2001 From: Joshua Calafato Date: Tue, 22 Sep 2026 16:27:38 +0000 Subject: [PATCH 04/20] test(builder): preserve explicit family execution inputs Signed-off-by: Joshua Calafato --- core/builder/tests/test_build.py | 10 ++++++++-- 1 file changed, 8 insertions(+), 2 deletions(-) diff --git a/core/builder/tests/test_build.py b/core/builder/tests/test_build.py index f6c5176823..69bb12f6ba 100644 --- a/core/builder/tests/test_build.py +++ b/core/builder/tests/test_build.py @@ -564,14 +564,18 @@ def test_bad_cli_execution_rejected_before_primary_model_acquisition(monkeypatch build_cli.main(["build", "model-id", "-o", "/tmp/example.bundle", *options]) -def test_cli_forwards_exact_variant_and_named_local_paths(tmp_path, monkeypatch): +@pytest.mark.parametrize("family", [None, "explicit_owner"]) +def test_cli_forwards_exact_variant_and_named_local_paths(tmp_path, monkeypatch, family): from tensorrt_model_connect.model_support import FamilySupport (tmp_path / "config.json").write_text('{"model_type":"example_model"}') monkeypatch.setattr( build_cli, "resolve_family", - lambda _: ("example_owner", FamilySupport(("example_task",), "example_task")), + lambda _, requested=None: ( + requested or "example_owner", + FamilySupport(("example_task",), "example_task"), + ), ) companion = tmp_path / "checkpoint=local" companion.mkdir() @@ -589,10 +593,12 @@ def test_cli_forwards_exact_variant_and_named_local_paths(tmp_path, monkeypatch) "paired", "--companion", f"draft={companion}", + *(["--family", family] if family is not None else []), ] ) actual_request, kwargs = seen[0] assert actual_request.model_dir == tmp_path + assert actual_request.family == (family or "example_owner") assert kwargs == { "execution": BuildExecutionInputs("paired", (NamedCheckpoint("draft", companion),)) } From 775ed5c8c3c76c3900818eb41ca65d366cad922c Mon Sep 17 00:00:00 2001 From: Joshua Calafato Date: Tue, 22 Sep 2026 20:51:02 +0000 Subject: [PATCH 05/20] refactor(build): delegate CLI options to families Keep core build dispatch unchanged. Let lightweight family support register CLI arguments and prepare a family-owned typed request; move Edge input and platform semantics out of core. Consolidate optional SDK provisioning under its Edge directory. Signed-off-by: Joshua Calafato --- CMakeLists.txt | 2 +- cmake/{edgellm => edge_llm}/CheckNative.cmake | 0 cmake/{ => edge_llm}/EdgeLLM.cmake | 4 +- .../EdgeLLMConfig.cmake.in | 0 cmake/{edgellm => edge_llm}/Install.cmake.in | 0 cmake/{edgellm => edge_llm}/Prepare.cmake.in | 0 cmake/{edgellm => edge_llm}/README.md | 0 .../tensorrt_model_connect/__init__.py | 4 +- core/builder/tensorrt_model_connect/build.py | 133 +------- .../tensorrt_model_connect/build_cli.py | 65 ++-- .../tensorrt_model_connect/model_support.py | 13 +- core/builder/tests/test_build.py | 317 +----------------- core/builder/tests/test_build_cli.py | 75 +++++ tools/tests/test_architecture.py | 12 +- website/docs/api/python-builder.md | 46 +-- website/docs/architecture/build-pipeline.md | 10 +- website/docs/user-guides/configure-runtime.md | 2 +- 17 files changed, 151 insertions(+), 532 deletions(-) rename cmake/{edgellm => edge_llm}/CheckNative.cmake (100%) rename cmake/{ => edge_llm}/EdgeLLM.cmake (98%) rename cmake/{edgellm => edge_llm}/EdgeLLMConfig.cmake.in (100%) rename cmake/{edgellm => edge_llm}/Install.cmake.in (100%) rename cmake/{edgellm => edge_llm}/Prepare.cmake.in (100%) rename cmake/{edgellm => edge_llm}/README.md (100%) diff --git a/CMakeLists.txt b/CMakeLists.txt index bb21195fb3..92e11c3ee0 100644 --- a/CMakeLists.txt +++ b/CMakeLists.txt @@ -45,7 +45,7 @@ find_library(TRTMC_TRT_LIBRARY REQUIRED ) -include("${CMAKE_CURRENT_SOURCE_DIR}/cmake/EdgeLLM.cmake") +include("${CMAKE_CURRENT_SOURCE_DIR}/cmake/edge_llm/EdgeLLM.cmake") option(TRTMC_ENABLE_BYOK "Enable the optional TVM-FFI BYOK bridge" ON) set(TRTMC_HAS_TVM_FFI OFF) diff --git a/cmake/edgellm/CheckNative.cmake b/cmake/edge_llm/CheckNative.cmake similarity index 100% rename from cmake/edgellm/CheckNative.cmake rename to cmake/edge_llm/CheckNative.cmake diff --git a/cmake/EdgeLLM.cmake b/cmake/edge_llm/EdgeLLM.cmake similarity index 98% rename from cmake/EdgeLLM.cmake rename to cmake/edge_llm/EdgeLLM.cmake index 678e33dced..d4c52f2b7a 100644 --- a/cmake/EdgeLLM.cmake +++ b/cmake/edge_llm/EdgeLLM.cmake @@ -9,7 +9,7 @@ endif() if(CMAKE_CROSSCOMPILING) message(FATAL_ERROR "Edge-LLM cross compilation is not supported") endif() -include("${CMAKE_CURRENT_LIST_DIR}/edgellm/CheckNative.cmake") +include("${CMAKE_CURRENT_LIST_DIR}/CheckNative.cmake") option(TRTMC_EDGELLM_ALL_KERNELS "Build all pinned Edge operator groups supported by the native GPU" OFF) option(TRTMC_EDGELLM_ONNX "Install the pinned ONNX exporter and native engine builder" OFF) set(_edge_cute_groups "fmha|gdn") @@ -26,7 +26,7 @@ if(TRTMC_EDGELLM_ONNX) endif() set(_edge_version "0.10.1") set(_edge_revision "e8b29522938901f6df19ebeedd4b69bc8edbcd97") -set(_edge_root "${CMAKE_BINARY_DIR}/_deps/edgellm") +set(_edge_root "${CMAKE_BINARY_DIR}/_deps") set(_edge_prefix "${_edge_root}/install") find_package(EdgeLLM ${_edge_version} EXACT CONFIG QUIET) if(EdgeLLM_FOUND AND NOT EdgeLLM_PREFIX STREQUAL _edge_prefix) diff --git a/cmake/edgellm/EdgeLLMConfig.cmake.in b/cmake/edge_llm/EdgeLLMConfig.cmake.in similarity index 100% rename from cmake/edgellm/EdgeLLMConfig.cmake.in rename to cmake/edge_llm/EdgeLLMConfig.cmake.in diff --git a/cmake/edgellm/Install.cmake.in b/cmake/edge_llm/Install.cmake.in similarity index 100% rename from cmake/edgellm/Install.cmake.in rename to cmake/edge_llm/Install.cmake.in diff --git a/cmake/edgellm/Prepare.cmake.in b/cmake/edge_llm/Prepare.cmake.in similarity index 100% rename from cmake/edgellm/Prepare.cmake.in rename to cmake/edge_llm/Prepare.cmake.in diff --git a/cmake/edgellm/README.md b/cmake/edge_llm/README.md similarity index 100% rename from cmake/edgellm/README.md rename to cmake/edge_llm/README.md diff --git a/core/builder/tensorrt_model_connect/__init__.py b/core/builder/tensorrt_model_connect/__init__.py index 407e0c8a14..101f6dcb4b 100644 --- a/core/builder/tensorrt_model_connect/__init__.py +++ b/core/builder/tensorrt_model_connect/__init__.py @@ -3,14 +3,12 @@ """TensorRT Model Connect build API.""" -from .build import BuildExecutionInputs, BuildRequest, NamedCheckpoint, build +from .build import BuildRequest, build from .bundle_writer import BundleWriter from .graph_transform import GraphTransform __all__ = [ - "BuildExecutionInputs", "BuildRequest", - "NamedCheckpoint", "BundleWriter", "GraphTransform", "build", diff --git a/core/builder/tensorrt_model_connect/build.py b/core/builder/tensorrt_model_connect/build.py index aa19f29c4e..f6fad3a61e 100644 --- a/core/builder/tensorrt_model_connect/build.py +++ b/core/builder/tensorrt_model_connect/build.py @@ -7,8 +7,6 @@ import hashlib import importlib -import os -import platform import re import sys from dataclasses import dataclass @@ -22,52 +20,6 @@ _ID = re.compile(r"[a-z][a-z0-9_]*\Z") -@dataclass(frozen=True) -class NamedCheckpoint: - """One explicitly named local checkpoint; the family owns role semantics.""" - - role: str - model_dir: Path - - def __post_init__(self) -> None: - _validate_id("checkpoint role", self.role) - if not isinstance(self.model_dir, Path): - raise TypeError("checkpoint model_dir must be a Path") - if not self.model_dir.is_dir(): - raise ValueError(f"checkpoint must be an existing local directory: {self.model_dir}") - - -@dataclass(frozen=True) -class BuildExecutionInputs: - """Optional family-owned execution variant and immutable local companions. - - Core transports these inputs without interpreting variants, fetching models, - or inferring compatibility. A family must explicitly implement the capability. - """ - - variant: str - checkpoints: tuple[NamedCheckpoint, ...] = () - - def __post_init__(self) -> None: - _validate_id("execution variant", self.variant) - if not isinstance(self.checkpoints, tuple) or any( - not isinstance(checkpoint, NamedCheckpoint) for checkpoint in self.checkpoints - ): - raise TypeError("checkpoints must be a tuple of NamedCheckpoint values") - roles = [checkpoint.role for checkpoint in self.checkpoints] - if len(roles) != len(set(roles)): - raise ValueError("checkpoint roles must be unique") - self.validate_local() - - def validate_local(self) -> None: - """Recheck local availability before dispatch, without acquiring inputs.""" - for checkpoint in self.checkpoints: - if not checkpoint.model_dir.is_dir(): - raise ValueError( - f"checkpoint must be an existing local directory: {checkpoint.model_dir}" - ) - - @dataclass(frozen=True) class BuildRequest: """Inputs shared by the build core and one family-owned builder.""" @@ -120,65 +72,6 @@ def __post_init__(self) -> None: raise ValueError("graph_transform must be callable when provided") -def subprocess_environment( - overrides: dict[str, str], *, prepend_paths: dict[str, str] | None = None -) -> dict[str, str]: - """Copy the parent environment for one child without mutating process state. - - Callers own explicit tool settings; this helper only merges values and - prepends search paths using the executing platform's path separator. - """ - environment = os.environ.copy() - environment.update(overrides) - for name, value in (prepend_paths or {}).items(): - previous = environment.get(name) - environment[name] = value + (os.pathsep + previous if previous else "") - return environment - - -def cmake_prefixes() -> list[Path]: - """Return explicit standard CMake prefixes followed by the Python prefix.""" - prefixes = [ - Path(value) for value in os.environ.get("CMAKE_PREFIX_PATH", "").split(os.pathsep) if value - ] - return [*prefixes, Path(sys.prefix)] - - -def detect_local_platform() -> dict: - """Return executing GPU and native SDK identity without selecting a model. - - Returns: - OS/release, CPU architecture, GPU SM, CUDA and TensorRT versions. - - Raises: - ImportError: Native SDK Python bindings are unavailable. - RuntimeError: CUDA cannot identify the executing device. - """ - import tensorrt as trt - from cuda.bindings import runtime - - def checked(result): - if int(result[0]) != 0: - raise RuntimeError(f"CUDA device discovery failed: {result[0]}") - return result[1] - - device = checked(runtime.cudaGetDevice()) - gpu = checked(runtime.cudaGetDeviceProperties(device)) - cuda = checked(runtime.cudaRuntimeGetVersion()) - try: - release = platform.freedesktop_os_release() if sys.platform == "linux" else {} - except OSError: - release = {} - return { - "os": sys.platform, - "os_version": release.get("VERSION_ID", platform.release()), - "arch": platform.machine(), - "sm": gpu.major * 10 + gpu.minor, - "cuda_version": f"{cuda // 1000}.{cuda % 1000 // 10}", - "tensorrt_version": trt.__version__, - } - - def _validate_id(field: str, value: object) -> str: if not isinstance(value, str) or _ID.fullmatch(value) is None: raise ValueError( @@ -242,36 +135,16 @@ def select_backend(backend: str) -> None: _select_backend = select_backend # Compatibility for existing Python callers. -def build(request: BuildRequest, *, execution: BuildExecutionInputs | None = None) -> None: - """Run one family builder and atomically publish its bundle on success. - - Explicit execution inputs require the optional family build_with_inputs hook. - Missing support fails before constructing the writer; failures never retry a - different variant or silently invoke the ordinary builder. - """ - - if execution is not None: - if not isinstance(execution, BuildExecutionInputs): - raise TypeError("execution must be BuildExecutionInputs") - execution.validate_local() +def build(request: BuildRequest) -> None: + """Run one family builder and publish its bundle on success.""" family = _resolve_family(request) _select_backend(request.backend) family_module = _load_family(family) - extended_build = None - if execution is not None: - extended_build = getattr(family_module, "build_with_inputs", None) - if not callable(extended_build): - raise NotImplementedError( - f"family {family!r} does not support explicit build execution inputs" - ) writer = BundleWriter(request.output_path) try: with graph_transform(request.graph_transform): - if execution is None: - family_module.build(request, writer) - else: - extended_build(request, writer, execution) + family_module.build(request, writer) writer.finish() except BaseException: writer.abort() diff --git a/core/builder/tensorrt_model_connect/build_cli.py b/core/builder/tensorrt_model_connect/build_cli.py index 64f8d3c600..e4b926fd53 100644 --- a/core/builder/tensorrt_model_connect/build_cli.py +++ b/core/builder/tensorrt_model_connect/build_cli.py @@ -12,22 +12,26 @@ from pathlib import Path from typing import Sequence -from .build import BuildExecutionInputs, BuildRequest, NamedCheckpoint, _load_family, build +from .build import BuildRequest, _load_family, build from .model_support import ( AmbiguousFamilyError, FamilyResolutionError, + FamilySupport, load_model_metadata, resolve_family, resolve_model, ) -def _parser(prepare_family: object | None = None) -> argparse.ArgumentParser: +def _parser( + prepare_family: object | None = None, *, build_support: FamilySupport | None = None, + require_output: bool = True, +) -> argparse.ArgumentParser: parser = argparse.ArgumentParser(prog="trtmc") commands = parser.add_subparsers(dest="command", required=True) build_parser = commands.add_parser("build", help="Build one TensorRT bundle") build_parser.add_argument("model", help="Hugging Face model ID or local snapshot") - build_parser.add_argument("-o", "--output", type=Path, required=True) + build_parser.add_argument("-o", "--output", type=Path, required=require_output) build_parser.add_argument( "--family", help="Select one compatible family instead of automatic resolution" ) @@ -46,14 +50,8 @@ def _parser(prepare_family: object | None = None) -> argparse.ArgumentParser: build_parser.add_argument("--fp32-layer", type=int, action="append", default=[]) build_parser.add_argument("--dynamic-kv-cache", action="store_true") build_parser.add_argument("--verbose", action="store_true") - build_parser.add_argument("--execution-variant", help="Explicit family-owned execution variant") - build_parser.add_argument( - "--companion", - action="append", - default=[], - metavar="ROLE=LOCAL_DIR", - help="Named existing local checkpoint; repeat for multiple distinct roles", - ) + if build_support is not None and callable(build_support.add_build_arguments): + build_support.add_build_arguments(build_parser) prepare_parser = commands.add_parser( "prepare-structure", help="Prepare one structure request without rebuilding its model bundle", @@ -72,28 +70,14 @@ def _parser(prepare_family: object | None = None) -> argparse.ArgumentParser: return parser -def _execution_inputs(args: argparse.Namespace) -> BuildExecutionInputs | None: - """Parse only explicit local inputs; no variant list or model acquisition.""" - if args.command != "build": - return None - if args.execution_variant is None: - if args.companion: - raise ValueError("--companion requires --execution-variant") - return None - checkpoints = [] - for value in args.companion: - role, separator, directory = value.partition("=") - if not separator or not role or not directory: - raise ValueError("--companion must be ROLE=LOCAL_DIR") - if "://" in directory: - raise ValueError("--companion requires a local directory, not a URI") - checkpoints.append(NamedCheckpoint(role, Path(directory))) - return BuildExecutionInputs(args.execution_variant, tuple(checkpoints)) - - def main(argv: Sequence[str] | None = None) -> int: arguments = list(sys.argv[1:] if argv is None else argv) - base_parser = _parser() + family_help = ( + len(arguments) > 1 and arguments[0] == "build" + and not arguments[1].startswith("-") + and any(arg in {"-h", "--help"} for arg in arguments) + ) + base_parser = _parser(require_output=not family_help) if ( len(arguments) > 1 and arguments[0] == "prepare-structure" @@ -101,8 +85,12 @@ def main(argv: Sequence[str] | None = None) -> int: and arguments[1].startswith("-") ): base_parser.error("MODEL must immediately follow prepare-structure") - preliminary, _ = base_parser.parse_known_args(arguments) - execution = _execution_inputs(preliminary) + preliminary_arguments = ( + [arg for arg in arguments if arg not in {"-h", "--help"}] if family_help else arguments + ) + preliminary, unknown = base_parser.parse_known_args(preliminary_arguments) + if unknown and arguments[0] == "build" and arguments[1].startswith("-"): + base_parser.error("MODEL must immediately follow build when family options are used") model_dir = _resolve_model(preliminary.model, preliminary.revision) metadata = load_model_metadata(model_dir) try: @@ -115,7 +103,7 @@ def main(argv: Sequence[str] | None = None) -> int: _print_family_error(error, arguments) return 2 family_module = _load_family(family) if preliminary.command == "prepare-structure" else None - args = _parser(family_module).parse_args(arguments) + args = _parser(family_module, build_support=support).parse_args(arguments) if args.command == "prepare-structure": prepare = getattr(family_module, "prepare_structure_request", None) if not callable(prepare): @@ -158,10 +146,11 @@ def main(argv: Sequence[str] | None = None) -> int: dynamic_kv_cache=args.dynamic_kv_cache, verbose=args.verbose, ) - if execution is None: - build(request) - else: - build(request, execution=execution) + if callable(support.prepare_build_request): + request = support.prepare_build_request(request, args) + if not isinstance(request, BuildRequest) or request.family != family: + raise TypeError("family prepare_build_request must preserve the owning BuildRequest") + build(request) return 0 diff --git a/core/builder/tensorrt_model_connect/model_support.py b/core/builder/tensorrt_model_connect/model_support.py index 04fd18c307..08103ed782 100644 --- a/core/builder/tensorrt_model_connect/model_support.py +++ b/core/builder/tensorrt_model_connect/model_support.py @@ -5,12 +5,17 @@ from __future__ import annotations +from argparse import ArgumentParser, Namespace import importlib import json import re from dataclasses import dataclass from pathlib import Path -from typing import Any, Callable +from typing import TYPE_CHECKING, Any, Callable + + +if TYPE_CHECKING: + from .build import BuildRequest _ID = re.compile(r"[a-z][a-z0-9_]*\Z") @@ -56,6 +61,8 @@ class FamilySupport: tasks: tuple[str, ...] default_task: str default_precision: str = "fp32" + add_build_arguments: Callable[[ArgumentParser], None] | None = None + prepare_build_request: Callable[["BuildRequest", Namespace], "BuildRequest"] | None = None def __post_init__(self) -> None: if not self.tasks or len(set(self.tasks)) != len(self.tasks): @@ -97,6 +104,8 @@ def family_support( tasks: tuple[str, ...], default_task: str, default_precision: str = "fp32", + add_build_arguments: Callable[[ArgumentParser], None] | None = None, + prepare_build_request: Callable[["BuildRequest", Namespace], "BuildRequest"] | None = None, ) -> DescribeSupport: """Create one exact, family-owned support function.""" @@ -110,6 +119,8 @@ def family_support( tasks=tasks, default_task=default_task, default_precision=default_precision, + add_build_arguments=add_build_arguments, + prepare_build_request=prepare_build_request, ) def describe(metadata: ModelMetadata) -> FamilySupport | None: diff --git a/core/builder/tests/test_build.py b/core/builder/tests/test_build.py index 69bb12f6ba..5eff69de1b 100644 --- a/core/builder/tests/test_build.py +++ b/core/builder/tests/test_build.py @@ -4,7 +4,6 @@ from __future__ import annotations import importlib -from contextlib import contextmanager import sys from dataclasses import FrozenInstanceError, replace from pathlib import Path @@ -12,7 +11,7 @@ import pytest -from tensorrt_model_connect import BuildExecutionInputs, BuildRequest, NamedCheckpoint, build_cli +from tensorrt_model_connect import BuildRequest build_core = importlib.import_module("tensorrt_model_connect.build") @@ -288,317 +287,3 @@ def abort(self) -> None: with pytest.raises(OSError, match="publish failed"): build_core.build(_request(tmp_path)) assert events == ["finish", "abort"] - - -@pytest.mark.parametrize("value", ["", "/one", "/one:/two", ":/one::/two:"]) -def test_cmake_prefixes_preserve_standard_search_order(monkeypatch, value): - monkeypatch.setenv("CMAKE_PREFIX_PATH", value) - monkeypatch.setattr(build_core.sys, "prefix", "/python") - expected = [Path(item) for item in value.split(build_core.os.pathsep) if item] - assert build_core.cmake_prefixes() == [*expected, Path("/python")] - - -def test_cmake_prefixes_without_environment_use_python_prefix(monkeypatch): - monkeypatch.delenv("CMAKE_PREFIX_PATH", raising=False) - monkeypatch.setattr(build_core.sys, "prefix", "/python") - assert build_core.cmake_prefixes() == [Path("/python")] - # Constructing explicit child-tool settings must not change the caller's - # package search order or mutate an inherited search path. - monkeypatch.setenv("TEST_TOOL_SEARCH_PATH", "/original") - monkeypatch.delenv("TEST_TOOL_NEW_PATH", raising=False) - child = build_core.subprocess_environment( - {"CMAKE_PREFIX_PATH": "/child"}, - prepend_paths={"TEST_TOOL_SEARCH_PATH": "/first", "TEST_TOOL_NEW_PATH": "/new"}, - ) - assert child["CMAKE_PREFIX_PATH"] == "/child" - assert child["TEST_TOOL_SEARCH_PATH"] == "/first" + build_core.os.pathsep + "/original" - assert child["TEST_TOOL_NEW_PATH"] == "/new" - assert build_core.cmake_prefixes() == [Path("/python")] - assert build_core.os.environ["TEST_TOOL_SEARCH_PATH"] == "/original" - assert "TEST_TOOL_NEW_PATH" not in build_core.os.environ - - -@pytest.fixture -def native_platform_bindings(monkeypatch): - from unittest.mock import Mock - - runtime = SimpleNamespace( - cudaGetDevice=Mock(return_value=(0, 3)), - cudaGetDeviceProperties=Mock(return_value=(0, SimpleNamespace(major=8, minor=6))), - cudaRuntimeGetVersion=Mock(return_value=(0, 13030)), - ) - monkeypatch.setitem(sys.modules, "tensorrt", SimpleNamespace(__version__="11.1.0.106")) - monkeypatch.setitem(sys.modules, "cuda.bindings", SimpleNamespace(runtime=runtime)) - monkeypatch.setattr(build_core.sys, "platform", "linux") - monkeypatch.setattr(build_core.platform, "machine", lambda: "x86_64") - monkeypatch.setattr( - build_core.platform, "freedesktop_os_release", lambda: {"VERSION_ID": "24.04"} - ) - monkeypatch.setattr(build_core.platform, "release", lambda: "fallback-release") - return runtime - - -@pytest.mark.parametrize("release_available", [True, False]) -def test_native_platform_uses_executing_cuda_device_and_full_sdk( - native_platform_bindings, monkeypatch, release_available -): - if not release_available: - from unittest.mock import Mock - - monkeypatch.setattr( - build_core.platform, "freedesktop_os_release", Mock(side_effect=OSError("missing")) - ) - assert build_core.detect_local_platform() == { - "os": "linux", - "os_version": "24.04" if release_available else "fallback-release", - "arch": "x86_64", - "sm": 86, - "cuda_version": "13.3", - "tensorrt_version": "11.1.0.106", - } - native_platform_bindings.cudaGetDevice.assert_called_once_with() - native_platform_bindings.cudaGetDeviceProperties.assert_called_once_with(3) - native_platform_bindings.cudaRuntimeGetVersion.assert_called_once_with() - - -@pytest.mark.parametrize( - "failing", ["cudaGetDevice", "cudaGetDeviceProperties", "cudaRuntimeGetVersion"] -) -def test_native_platform_propagates_cuda_discovery_failure(native_platform_bindings, failing): - getattr(native_platform_bindings, failing).return_value = (35,) - with pytest.raises(RuntimeError, match="CUDA device discovery failed: 35"): - build_core.detect_local_platform() - - -def test_native_platform_retains_nonlinux_identity(native_platform_bindings, monkeypatch): - monkeypatch.setattr(build_core.sys, "platform", "win32") - result = build_core.detect_local_platform() - assert result["os"] == "win32" - assert result["os_version"] == "fallback-release" - - -def execution_request(root: Path) -> BuildRequest: - return BuildRequest(root, root / "model.bundle", "example", "text_generation", "fp16") - - -def inputs(root: Path) -> BuildExecutionInputs: - return BuildExecutionInputs("paired", (NamedCheckpoint("draft", root),)) - - -def test_execution_inputs_are_immutable(tmp_path): - execution = inputs(tmp_path) - with pytest.raises(FrozenInstanceError): - execution.variant = "other" - with pytest.raises(FrozenInstanceError): - execution.checkpoints[0].role = "other" - assert execution.checkpoints[0].model_dir is tmp_path - - -@pytest.mark.parametrize("value", ["", "../bad", "UPPER", "a-b", "a.b"]) -def test_invalid_role_and_variant(tmp_path, value): - with pytest.raises(ValueError, match="lowercase identifier"): - NamedCheckpoint(value, tmp_path) - with pytest.raises(ValueError, match="lowercase identifier"): - BuildExecutionInputs(value) - - -def test_execution_requires_immutable_typed_companions(tmp_path): - checkpoint = NamedCheckpoint("draft", tmp_path) - with pytest.raises(TypeError, match="tuple"): - BuildExecutionInputs("paired", [checkpoint]) - with pytest.raises(TypeError, match="NamedCheckpoint"): - BuildExecutionInputs("paired", (object(),)) - with pytest.raises(ValueError, match="unique"): - BuildExecutionInputs("paired", (checkpoint, checkpoint)) - with pytest.raises(TypeError, match="Path"): - NamedCheckpoint("draft", str(tmp_path)) - - -def test_local_checkpoint_required_and_rechecked(tmp_path, monkeypatch): - with pytest.raises(ValueError, match="existing local directory"): - NamedCheckpoint("draft", tmp_path / "missing") - file = tmp_path / "file" - file.write_text("not a directory") - with pytest.raises(ValueError, match="existing local directory"): - NamedCheckpoint("draft", file) - directory = tmp_path / "companion" - directory.mkdir() - execution = inputs(directory) - directory.rmdir() - monkeypatch.setattr(build_core, "_select_backend", lambda _: pytest.fail("backend touched")) - with pytest.raises(ValueError, match="existing local directory"): - build_core.build(execution_request(tmp_path), execution=execution) - - -def test_untyped_execution_fails_before_side_effects(tmp_path, monkeypatch): - monkeypatch.setattr(build_core, "_select_backend", lambda _: pytest.fail("backend touched")) - with pytest.raises(TypeError, match="BuildExecutionInputs"): - build_core.build(execution_request(tmp_path), execution={"variant": "paired"}) - - -@pytest.mark.parametrize("hook", [None, 17]) -def test_missing_capability_fails_before_writer(tmp_path, monkeypatch, hook): - monkeypatch.setattr( - build_core, - "_load_family", - lambda _: SimpleNamespace( - build=lambda *_: pytest.fail("ordinary fallback invoked"), build_with_inputs=hook - ), - ) - monkeypatch.setattr(build_core, "BundleWriter", lambda _: pytest.fail("writer created")) - with pytest.raises(NotImplementedError, match="does not support explicit"): - build_core.build(execution_request(tmp_path), execution=inputs(tmp_path)) - - -def test_exact_envelope_and_existing_transaction_are_preserved(tmp_path, monkeypatch): - events = [] - original_request, execution = execution_request(tmp_path), inputs(tmp_path) - - @contextmanager - def transform(value): - assert value is original_request.graph_transform - events.append("enter") - yield - events.append("exit") - - class Writer: - def __init__(self, path): - assert path == original_request.output_path - events.append("writer") - - def finish(self): - events.append("finish") - - def abort(self): - pytest.fail("unexpected abort") - - def extended(actual_request, writer, actual_execution): - assert actual_request is original_request and actual_execution is execution - assert isinstance(writer, Writer) - events.append("extended") - - monkeypatch.setattr(build_core, "graph_transform", transform) - monkeypatch.setattr(build_core, "BundleWriter", Writer) - monkeypatch.setattr( - build_core, - "_load_family", - lambda _: SimpleNamespace( - build=lambda *_: pytest.fail("ordinary fallback invoked"), build_with_inputs=extended - ), - ) - build_core.build(original_request, execution=execution) - assert events == ["writer", "enter", "extended", "exit", "finish"] - - -@pytest.mark.parametrize("failure", [RuntimeError("failed"), KeyboardInterrupt()]) -def test_explicit_failure_aborts_real_writer_without_replacing_bundle( - tmp_path, monkeypatch, failure -): - build_request = execution_request(tmp_path) - build_request.output_path.write_bytes(b"previous valid publication") - - def extended(actual, writer, execution): - writer.set_header(family=actual.family, task=actual.task, backend=actual.backend) - writer.add_json("test.json", {"variant": execution.variant}) - raise failure - - monkeypatch.setattr( - build_core, - "_load_family", - lambda _: SimpleNamespace( - build=lambda *_: pytest.fail("ordinary fallback invoked"), build_with_inputs=extended - ), - ) - with pytest.raises(type(failure)) as caught: - build_core.build(build_request, execution=inputs(tmp_path)) - assert caught.value is failure - assert build_request.output_path.read_bytes() == b"previous valid publication" - assert sorted(path.name for path in tmp_path.iterdir()) == ["model.bundle"] - - -def test_variant_without_companions_is_explicit_and_supported(tmp_path, monkeypatch): - seen = [] - - def extended(actual, writer, execution): - seen.append(execution) - writer.set_header(family=actual.family, task=actual.task, backend=actual.backend) - writer.add_json("test.json", {"variant": execution.variant}) - - monkeypatch.setattr( - build_core, "_load_family", lambda _: SimpleNamespace(build_with_inputs=extended) - ) - execution = BuildExecutionInputs("embedded") - build_core.build(execution_request(tmp_path), execution=execution) - assert seen == [execution] and (tmp_path / "model.bundle").is_file() - - -def test_ordinary_build_ignores_available_optional_hook(tmp_path, monkeypatch): - def ordinary(actual, writer): - writer.set_header(family=actual.family, task=actual.task, backend=actual.backend) - writer.add_json("test.json", {"ordinary": True}) - - monkeypatch.setattr( - build_core, - "_load_family", - lambda _: SimpleNamespace( - build=ordinary, build_with_inputs=lambda *_: pytest.fail("optional hook invoked") - ), - ) - build_core.build(execution_request(tmp_path)) - - -@pytest.mark.parametrize( - "options", - [ - ["--companion", "draft=/missing"], - ["--execution-variant", "paired", "--companion", "missing_separator"], - ["--execution-variant", "paired", "--companion", "=path"], - ["--execution-variant", "paired", "--companion", "draft="], - ["--execution-variant", "paired", "--companion", "draft=https://example.com/model"], - ["--execution-variant", ""], - ], -) -def test_bad_cli_execution_rejected_before_primary_model_acquisition(monkeypatch, options): - monkeypatch.setattr(build_cli, "_resolve_model", lambda *_: pytest.fail("model acquisition")) - with pytest.raises(ValueError): - build_cli.main(["build", "model-id", "-o", "/tmp/example.bundle", *options]) - - -@pytest.mark.parametrize("family", [None, "explicit_owner"]) -def test_cli_forwards_exact_variant_and_named_local_paths(tmp_path, monkeypatch, family): - from tensorrt_model_connect.model_support import FamilySupport - - (tmp_path / "config.json").write_text('{"model_type":"example_model"}') - monkeypatch.setattr( - build_cli, - "resolve_family", - lambda _, requested=None: ( - requested or "example_owner", - FamilySupport(("example_task",), "example_task"), - ), - ) - companion = tmp_path / "checkpoint=local" - companion.mkdir() - seen = [] - monkeypatch.setattr( - build_cli, "build", lambda request, **kwargs: seen.append((request, kwargs)) - ) - build_cli.main( - [ - "build", - str(tmp_path), - "-o", - str(tmp_path / "out.bundle"), - "--execution-variant", - "paired", - "--companion", - f"draft={companion}", - *(["--family", family] if family is not None else []), - ] - ) - actual_request, kwargs = seen[0] - assert actual_request.model_dir == tmp_path - assert actual_request.family == (family or "example_owner") - assert kwargs == { - "execution": BuildExecutionInputs("paired", (NamedCheckpoint("draft", companion),)) - } diff --git a/core/builder/tests/test_build_cli.py b/core/builder/tests/test_build_cli.py index 8086a1524b..a4cdbe3d7e 100644 --- a/core/builder/tests/test_build_cli.py +++ b/core/builder/tests/test_build_cli.py @@ -541,3 +541,78 @@ def test_prepare_structure_requires_a_callable_family_hook( "-o", str(output), ]) assert not output.exists() + + +@pytest.mark.parametrize("explicit_family", [False, True]) +def test_build_family_hooks_forward_a_typed_request(monkeypatch, tmp_path, explicit_family): + from dataclasses import dataclass + + @dataclass(frozen=True) + class ExampleRequest(build_cli.BuildRequest): + example_setting: str = "" + + def add_arguments(parser): + parser.add_argument("--example-setting", required=True) + + def prepare(request, args): + return ExampleRequest(**vars(request), example_setting=args.example_setting) + + support = FamilySupport( + ("example_task",), "example_task", + add_build_arguments=add_arguments, prepare_build_request=prepare, + ) + (tmp_path / "config.json").write_text('{"model_type":"example_model"}') + monkeypatch.setattr(build_cli, "resolve_family", lambda metadata, *args: ("example_owner", support)) + monkeypatch.setattr(build_cli, "_load_family", lambda *_: pytest.fail("GPU builder imported by CLI")) + seen = [] + monkeypatch.setattr(build_cli, "build", seen.append) + arguments = ["build", str(tmp_path), "-o", str(tmp_path / "out"), "--example-setting", "verbatim"] + if explicit_family: + arguments += ["--family", "example_owner"] + assert build_cli.main(arguments) == 0 + assert len(seen) == 1 and isinstance(seen[0], ExampleRequest) + assert seen[0].example_setting == "verbatim" + assert seen[0].family == "example_owner" and seen[0].model_dir == tmp_path + + +def test_build_family_options_are_not_global(monkeypatch, tmp_path): + (tmp_path / "config.json").write_text('{"model_type":"example_model"}') + support = FamilySupport(("example_task",), "example_task") + monkeypatch.setattr(build_cli, "resolve_family", lambda _: ("example_owner", support)) + monkeypatch.setattr(build_cli, "build", lambda *_: pytest.fail("unknown option reached build")) + with pytest.raises(SystemExit) as error: + build_cli.main(["build", str(tmp_path), "-o", str(tmp_path / "out"), + "--example-setting", "not-owned"]) + assert error.value.code == 2 + + +def test_build_family_help_does_not_require_output_or_load_builder(monkeypatch, tmp_path, capsys): + (tmp_path / "config.json").write_text('{"model_type":"example_model"}') + support = FamilySupport( + ("example_task",), "example_task", + add_build_arguments=lambda parser: parser.add_argument("--example-setting"), + prepare_build_request=lambda *_: pytest.fail("help prepared a build"), + ) + monkeypatch.setattr(build_cli, "resolve_family", lambda _: ("example_owner", support)) + monkeypatch.setattr(build_cli, "_load_family", lambda *_: pytest.fail("help imported GPU builder")) + with pytest.raises(SystemExit) as error: + build_cli.main(["build", str(tmp_path), "--help"]) + assert error.value.code == 0 + assert "--example-setting" in capsys.readouterr().out + + +@pytest.mark.parametrize("wrong_owner", [False, True]) +def test_build_family_hook_cannot_replace_owner_or_contract(monkeypatch, tmp_path, wrong_owner): + from dataclasses import replace + + (tmp_path / "config.json").write_text('{"model_type":"example_model"}') + support = FamilySupport( + ("example_task",), "example_task", + prepare_build_request=lambda request, _: ( + replace(request, family="another_owner") if wrong_owner else object() + ), + ) + monkeypatch.setattr(build_cli, "resolve_family", lambda _: ("example_owner", support)) + monkeypatch.setattr(build_cli, "build", lambda *_: pytest.fail("invalid request reached build")) + with pytest.raises(TypeError, match="preserve the owning BuildRequest"): + build_cli.main(["build", str(tmp_path), "-o", str(tmp_path / "out")]) diff --git a/tools/tests/test_architecture.py b/tools/tests/test_architecture.py index c83201ca3b..c95a86f7af 100644 --- a/tools/tests/test_architecture.py +++ b/tools/tests/test_architecture.py @@ -633,12 +633,12 @@ def test_shared_python_and_native_trees_are_closed_minimal_sets() -> None: } expected_cmake = { "cmake/trtmcConfig.cmake.in", - "cmake/EdgeLLM.cmake", - "cmake/edgellm/CheckNative.cmake", - "cmake/edgellm/EdgeLLMConfig.cmake.in", - "cmake/edgellm/Install.cmake.in", - "cmake/edgellm/Prepare.cmake.in", - "cmake/edgellm/README.md", + "cmake/edge_llm/EdgeLLM.cmake", + "cmake/edge_llm/CheckNative.cmake", + "cmake/edge_llm/EdgeLLMConfig.cmake.in", + "cmake/edge_llm/Install.cmake.in", + "cmake/edge_llm/Prepare.cmake.in", + "cmake/edge_llm/README.md", } expected_third_party = { "third_party/stb/stb_image.h", diff --git a/website/docs/api/python-builder.md b/website/docs/api/python-builder.md index 643776e18f..0b6d20e048 100644 --- a/website/docs/api/python-builder.md +++ b/website/docs/api/python-builder.md @@ -29,35 +29,23 @@ resolved API directly. decides whether that directory is a Hugging Face snapshot or a prepared checkpoint; `BuildRequest` does not perform another discovery pass. -## Optional execution inputs - -`build(request, execution=...)` accepts an optional, frozen -`BuildExecutionInputs` descriptor. It contains a family-owned `variant` string -and a tuple of `NamedCheckpoint(role, model_dir)` descriptors. Import these -public types from `tensorrt_model_connect`. Companion directories must already -exist locally; the core does not download them or infer compatible model pairs. -Roles must be unique. Variant and role names are lowercase identifiers. - -Providing execution inputs requires the selected family to implement -`build_with_inputs(request, writer, execution)`. The family validates the -variant, checkpoint roles, compatibility and execution semantics. A missing -hook fails before bundle creation; the core never substitutes ordinary -base-only generation or another variant. With no execution inputs, the -existing `build(request, writer)` family call is unchanged. - -The build CLI exposes the same optional contract: - -```text -trtmc build LOCAL_TARGET -o model.bundle \ - --execution-variant FAMILY_VARIANT \ - --companion ROLE=LOCAL_COMPANION_DIR -``` - -Replace the uppercase placeholders with values from the selected family's -recipe; they are not literal supported identifiers. Repeat `--companion` only -for distinct roles. A companion requires `--execution-variant`; URLs and -implicit companion downloads are unsupported. The generic API does not itself -qualify any speculative algorithm or checkpoint pair. +## Family-owned build arguments + +A family may provide `add_build_arguments(parser)` and +`prepare_build_request(request, args)` through its lightweight `FamilySupport` +declaration. The CLI resolves the owner, registers only that family's options, +and lets it return a family-owned `BuildRequest` subclass. The hook must retain +the resolved family. Ordinary families need no changes. + +Core always calls the same `build(request)` and family `build(request, writer)` +entrypoints. There is no shared execution-variant list, companion interpretation, +GPU offload selection, or alternative-builder dispatch. The family owns all +extra fields, validation and execution choices. + +Put the model immediately after `build` when using family-specific options. +`trtmc build /path/to/model --help` shows the resolved family's options without +importing its GPU builder. Model resolution precedes family-specific argument +validation; request preparation precedes backend import and bundle creation. ## Optional graph transform diff --git a/website/docs/architecture/build-pipeline.md b/website/docs/architecture/build-pipeline.md index 243172f4af..5cedd2e161 100644 --- a/website/docs/architecture/build-pipeline.md +++ b/website/docs/architecture/build-pipeline.md @@ -11,7 +11,6 @@ model ID/local snapshot -> choose family default task or validate --task -> import the selected families..model -> call build(BuildRequest, BundleWriter) - or the explicitly requested family build_with_inputs hook -> atomically publish format-1 bundle ``` @@ -37,10 +36,11 @@ sizes, family-owned quantization selection, FP32 layer overrides, direct dynamic-KV opt-in, and optional graph transform. Each family must implement or explicitly reject every non-default request it receives. -Optional `BuildExecutionInputs` travel separately from `BuildRequest`. Core -checks descriptor types, unique roles and existing local directories; only -the selected family interprets variant names and companion compatibility. -See the [Python Build API](../api/python-builder.md#optional-execution-inputs). +Optional CLI extensions are declared by the owning family's lightweight +support module. The family registers its arguments and prepares a typed request +before any GPU builder is imported. Core retains one ordinary family build +entrypoint; execution selection, companion validation and runtime composition +remain inside the family. No other family needs to change. ## Family build diff --git a/website/docs/user-guides/configure-runtime.md b/website/docs/user-guides/configure-runtime.md index fbbc07a82f..c7335bb6c5 100644 --- a/website/docs/user-guides/configure-runtime.md +++ b/website/docs/user-guides/configure-runtime.md @@ -49,7 +49,7 @@ the SDK rejects same-version development headers with an incompatible C++ ABI. - `CMAKE_PREFIX_PATH` points builders and runtime compilation to the installed SDK. Reuse fails explicitly if a requested capability is absent. -Follow the repository's [pinned SDK installation instructions](https://github.com/NVIDIA/TensorRT-Model-Connect/blob/main/cmake/edgellm/README.md) +Follow the repository's [pinned SDK installation instructions](https://github.com/NVIDIA/TensorRT-Model-Connect/blob/main/cmake/edge_llm/README.md) for dependencies, native architecture selection and offline provisioning. Each family decides whether and how to use the package. Installing the SDK is not evidence that a model, precision, input modality or execution variant From be6dd18bdfdf0c405003be2bc0c0ba716c1a26ff Mon Sep 17 00:00:00 2001 From: Joshua Calafato Date: Tue, 22 Sep 2026 21:14:16 +0000 Subject: [PATCH 06/20] fix(build): keep family CLI declarations import-light Resolve family-local hook modules only after choosing the owner. Preserve ordinary build dispatch and neutral host mechanics without exposing execution selection in core. Correct CMake template relocation while preserving SDK build paths. Signed-off-by: Joshua Calafato --- cmake/edge_llm/EdgeLLM.cmake | 4 +- core/builder/tensorrt_model_connect/build.py | 61 +++++++++++++ .../tensorrt_model_connect/build_cli.py | 20 +++-- .../tensorrt_model_connect/model_support.py | 21 ++--- core/builder/tests/test_build.py | 87 +++++++++++++++++++ core/builder/tests/test_build_cli.py | 15 +++- core/builder/tests/test_model_support.py | 14 +++ website/docs/api/python-builder.md | 4 +- 8 files changed, 200 insertions(+), 26 deletions(-) diff --git a/cmake/edge_llm/EdgeLLM.cmake b/cmake/edge_llm/EdgeLLM.cmake index d4c52f2b7a..7c1ef92c9d 100644 --- a/cmake/edge_llm/EdgeLLM.cmake +++ b/cmake/edge_llm/EdgeLLM.cmake @@ -26,7 +26,7 @@ if(TRTMC_EDGELLM_ONNX) endif() set(_edge_version "0.10.1") set(_edge_revision "e8b29522938901f6df19ebeedd4b69bc8edbcd97") -set(_edge_root "${CMAKE_BINARY_DIR}/_deps") +set(_edge_root "${CMAKE_BINARY_DIR}/_deps/edgellm") set(_edge_prefix "${_edge_root}/install") find_package(EdgeLLM ${_edge_version} EXACT CONFIG QUIET) if(EdgeLLM_FOUND AND NOT EdgeLLM_PREFIX STREQUAL _edge_prefix) @@ -69,7 +69,7 @@ set(_edge_repository "https://github.com/NVIDIA/TensorRT-Edge-LLM.git") if(TRTMC_EDGELLM_GIT_MIRROR) set(_edge_repository "${TRTMC_EDGELLM_GIT_MIRROR}") endif() -set(_edge_template_dir "${CMAKE_CURRENT_LIST_DIR}/edgellm") +set(_edge_template_dir "${CMAKE_CURRENT_LIST_DIR}") _edgellm_json_include(_edge_json_include) file(MAKE_DIRECTORY "${_edge_prefix}/lib/cmake/EdgeLLM" "${_edge_prefix}/include/edgellm/cpp" "${_edge_prefix}/include/edgellm/3rdParty/nlohmannJson/include" diff --git a/core/builder/tensorrt_model_connect/build.py b/core/builder/tensorrt_model_connect/build.py index f6fad3a61e..6ec9bd5ef1 100644 --- a/core/builder/tensorrt_model_connect/build.py +++ b/core/builder/tensorrt_model_connect/build.py @@ -7,6 +7,8 @@ import hashlib import importlib +import os +import platform import re import sys from dataclasses import dataclass @@ -72,6 +74,65 @@ def __post_init__(self) -> None: raise ValueError("graph_transform must be callable when provided") +def subprocess_environment( + overrides: dict[str, str], *, prepend_paths: dict[str, str] | None = None +) -> dict[str, str]: + """Copy the parent environment for one child without mutating process state. + + Callers own explicit tool settings; this helper only merges values and + prepends search paths using the executing platform's path separator. + """ + environment = os.environ.copy() + environment.update(overrides) + for name, value in (prepend_paths or {}).items(): + previous = environment.get(name) + environment[name] = value + (os.pathsep + previous if previous else "") + return environment + + +def cmake_prefixes() -> list[Path]: + """Return explicit standard CMake prefixes followed by the Python prefix.""" + prefixes = [ + Path(value) for value in os.environ.get("CMAKE_PREFIX_PATH", "").split(os.pathsep) if value + ] + return [*prefixes, Path(sys.prefix)] + + +def detect_local_platform() -> dict: + """Return executing GPU and native SDK identity without selecting a model. + + Returns: + OS/release, CPU architecture, GPU SM, CUDA and TensorRT versions. + + Raises: + ImportError: Native SDK Python bindings are unavailable. + RuntimeError: CUDA cannot identify the executing device. + """ + import tensorrt as trt + from cuda.bindings import runtime + + def checked(result): + if int(result[0]) != 0: + raise RuntimeError(f"CUDA device discovery failed: {result[0]}") + return result[1] + + device = checked(runtime.cudaGetDevice()) + gpu = checked(runtime.cudaGetDeviceProperties(device)) + cuda = checked(runtime.cudaRuntimeGetVersion()) + try: + release = platform.freedesktop_os_release() if sys.platform == "linux" else {} + except OSError: + release = {} + return { + "os": sys.platform, + "os_version": release.get("VERSION_ID", platform.release()), + "arch": platform.machine(), + "sm": gpu.major * 10 + gpu.minor, + "cuda_version": f"{cuda // 1000}.{cuda % 1000 // 10}", + "tensorrt_version": trt.__version__, + } + + def _validate_id(field: str, value: object) -> str: if not isinstance(value, str) or _ID.fullmatch(value) is None: raise ValueError( diff --git a/core/builder/tensorrt_model_connect/build_cli.py b/core/builder/tensorrt_model_connect/build_cli.py index e4b926fd53..2754667a69 100644 --- a/core/builder/tensorrt_model_connect/build_cli.py +++ b/core/builder/tensorrt_model_connect/build_cli.py @@ -6,6 +6,7 @@ from __future__ import annotations import argparse +import importlib import json import shlex import sys @@ -16,7 +17,6 @@ from .model_support import ( AmbiguousFamilyError, FamilyResolutionError, - FamilySupport, load_model_metadata, resolve_family, resolve_model, @@ -24,7 +24,7 @@ def _parser( - prepare_family: object | None = None, *, build_support: FamilySupport | None = None, + prepare_family: object | None = None, *, build_hooks: object | None = None, require_output: bool = True, ) -> argparse.ArgumentParser: parser = argparse.ArgumentParser(prog="trtmc") @@ -50,8 +50,9 @@ def _parser( build_parser.add_argument("--fp32-layer", type=int, action="append", default=[]) build_parser.add_argument("--dynamic-kv-cache", action="store_true") build_parser.add_argument("--verbose", action="store_true") - if build_support is not None and callable(build_support.add_build_arguments): - build_support.add_build_arguments(build_parser) + add_build_arguments = getattr(build_hooks, "add_build_arguments", None) + if callable(add_build_arguments): + add_build_arguments(build_parser) prepare_parser = commands.add_parser( "prepare-structure", help="Prepare one structure request without rebuilding its model bundle", @@ -103,7 +104,11 @@ def main(argv: Sequence[str] | None = None) -> int: _print_family_error(error, arguments) return 2 family_module = _load_family(family) if preliminary.command == "prepare-structure" else None - args = _parser(family_module, build_support=support).parse_args(arguments) + build_hooks = ( + importlib.import_module(f"families.{family}.{support.build_cli_module}") + if preliminary.command == "build" and support.build_cli_module is not None else None + ) + args = _parser(family_module, build_hooks=build_hooks).parse_args(arguments) if args.command == "prepare-structure": prepare = getattr(family_module, "prepare_structure_request", None) if not callable(prepare): @@ -146,8 +151,9 @@ def main(argv: Sequence[str] | None = None) -> int: dynamic_kv_cache=args.dynamic_kv_cache, verbose=args.verbose, ) - if callable(support.prepare_build_request): - request = support.prepare_build_request(request, args) + prepare_build_request = getattr(build_hooks, "prepare_build_request", None) + if callable(prepare_build_request): + request = prepare_build_request(request, args) if not isinstance(request, BuildRequest) or request.family != family: raise TypeError("family prepare_build_request must preserve the owning BuildRequest") build(request) diff --git a/core/builder/tensorrt_model_connect/model_support.py b/core/builder/tensorrt_model_connect/model_support.py index 08103ed782..39262801da 100644 --- a/core/builder/tensorrt_model_connect/model_support.py +++ b/core/builder/tensorrt_model_connect/model_support.py @@ -5,17 +5,12 @@ from __future__ import annotations -from argparse import ArgumentParser, Namespace import importlib import json import re from dataclasses import dataclass from pathlib import Path -from typing import TYPE_CHECKING, Any, Callable - - -if TYPE_CHECKING: - from .build import BuildRequest +from typing import Any, Callable _ID = re.compile(r"[a-z][a-z0-9_]*\Z") @@ -61,10 +56,14 @@ class FamilySupport: tasks: tuple[str, ...] default_task: str default_precision: str = "fp32" - add_build_arguments: Callable[[ArgumentParser], None] | None = None - prepare_build_request: Callable[["BuildRequest", Namespace], "BuildRequest"] | None = None + build_cli_module: str | None = None def __post_init__(self) -> None: + if self.build_cli_module is not None and ( + not isinstance(self.build_cli_module, str) + or any(_ID.fullmatch(part) is None for part in self.build_cli_module.split(".")) + ): + raise ValueError("build_cli_module must name a module within the owning family") if not self.tasks or len(set(self.tasks)) != len(self.tasks): raise ValueError("family support tasks must be non-empty and unique") if any(_ID.fullmatch(task) is None for task in self.tasks): @@ -104,8 +103,7 @@ def family_support( tasks: tuple[str, ...], default_task: str, default_precision: str = "fp32", - add_build_arguments: Callable[[ArgumentParser], None] | None = None, - prepare_build_request: Callable[["BuildRequest", Namespace], "BuildRequest"] | None = None, + build_cli_module: str | None = None, ) -> DescribeSupport: """Create one exact, family-owned support function.""" @@ -119,8 +117,7 @@ def family_support( tasks=tasks, default_task=default_task, default_precision=default_precision, - add_build_arguments=add_build_arguments, - prepare_build_request=prepare_build_request, + build_cli_module=build_cli_module, ) def describe(metadata: ModelMetadata) -> FamilySupport | None: diff --git a/core/builder/tests/test_build.py b/core/builder/tests/test_build.py index 5eff69de1b..ac892f7b6e 100644 --- a/core/builder/tests/test_build.py +++ b/core/builder/tests/test_build.py @@ -287,3 +287,90 @@ def abort(self) -> None: with pytest.raises(OSError, match="publish failed"): build_core.build(_request(tmp_path)) assert events == ["finish", "abort"] + + +@pytest.mark.parametrize("value", ["", "/one", "/one:/two", ":/one::/two:"]) +def test_cmake_prefixes_preserve_standard_search_order(monkeypatch, value): + monkeypatch.setenv("CMAKE_PREFIX_PATH", value) + monkeypatch.setattr(build_core.sys, "prefix", "/python") + expected = [Path(item) for item in value.split(build_core.os.pathsep) if item] + assert build_core.cmake_prefixes() == [*expected, Path("/python")] + + +def test_cmake_prefixes_without_environment_use_python_prefix(monkeypatch): + monkeypatch.delenv("CMAKE_PREFIX_PATH", raising=False) + monkeypatch.setattr(build_core.sys, "prefix", "/python") + assert build_core.cmake_prefixes() == [Path("/python")] + # Constructing explicit child-tool settings must not change the caller's + # package search order or mutate an inherited search path. + monkeypatch.setenv("TEST_TOOL_SEARCH_PATH", "/original") + monkeypatch.delenv("TEST_TOOL_NEW_PATH", raising=False) + child = build_core.subprocess_environment( + {"CMAKE_PREFIX_PATH": "/child"}, + prepend_paths={"TEST_TOOL_SEARCH_PATH": "/first", "TEST_TOOL_NEW_PATH": "/new"}, + ) + assert child["CMAKE_PREFIX_PATH"] == "/child" + assert child["TEST_TOOL_SEARCH_PATH"] == "/first" + build_core.os.pathsep + "/original" + assert child["TEST_TOOL_NEW_PATH"] == "/new" + assert build_core.cmake_prefixes() == [Path("/python")] + assert build_core.os.environ["TEST_TOOL_SEARCH_PATH"] == "/original" + assert "TEST_TOOL_NEW_PATH" not in build_core.os.environ + + +@pytest.fixture +def native_platform_bindings(monkeypatch): + from unittest.mock import Mock + + runtime = SimpleNamespace( + cudaGetDevice=Mock(return_value=(0, 3)), + cudaGetDeviceProperties=Mock(return_value=(0, SimpleNamespace(major=8, minor=6))), + cudaRuntimeGetVersion=Mock(return_value=(0, 13030)), + ) + monkeypatch.setitem(sys.modules, "tensorrt", SimpleNamespace(__version__="11.1.0.106")) + monkeypatch.setitem(sys.modules, "cuda.bindings", SimpleNamespace(runtime=runtime)) + monkeypatch.setattr(build_core.sys, "platform", "linux") + monkeypatch.setattr(build_core.platform, "machine", lambda: "x86_64") + monkeypatch.setattr( + build_core.platform, "freedesktop_os_release", lambda: {"VERSION_ID": "24.04"} + ) + monkeypatch.setattr(build_core.platform, "release", lambda: "fallback-release") + return runtime + + +@pytest.mark.parametrize("release_available", [True, False]) +def test_native_platform_uses_executing_cuda_device_and_full_sdk( + native_platform_bindings, monkeypatch, release_available +): + if not release_available: + from unittest.mock import Mock + + monkeypatch.setattr( + build_core.platform, "freedesktop_os_release", Mock(side_effect=OSError("missing")) + ) + assert build_core.detect_local_platform() == { + "os": "linux", + "os_version": "24.04" if release_available else "fallback-release", + "arch": "x86_64", + "sm": 86, + "cuda_version": "13.3", + "tensorrt_version": "11.1.0.106", + } + native_platform_bindings.cudaGetDevice.assert_called_once_with() + native_platform_bindings.cudaGetDeviceProperties.assert_called_once_with(3) + native_platform_bindings.cudaRuntimeGetVersion.assert_called_once_with() + + +@pytest.mark.parametrize( + "failing", ["cudaGetDevice", "cudaGetDeviceProperties", "cudaRuntimeGetVersion"] +) +def test_native_platform_propagates_cuda_discovery_failure(native_platform_bindings, failing): + getattr(native_platform_bindings, failing).return_value = (35,) + with pytest.raises(RuntimeError, match="CUDA device discovery failed: 35"): + build_core.detect_local_platform() + + +def test_native_platform_retains_nonlinux_identity(native_platform_bindings, monkeypatch): + monkeypatch.setattr(build_core.sys, "platform", "win32") + result = build_core.detect_local_platform() + assert result["os"] == "win32" + assert result["os_version"] == "fallback-release" diff --git a/core/builder/tests/test_build_cli.py b/core/builder/tests/test_build_cli.py index a4cdbe3d7e..580d4e5bd7 100644 --- a/core/builder/tests/test_build_cli.py +++ b/core/builder/tests/test_build_cli.py @@ -559,8 +559,11 @@ def prepare(request, args): support = FamilySupport( ("example_task",), "example_task", - add_build_arguments=add_arguments, prepare_build_request=prepare, + build_cli_module="example_cli", ) + monkeypatch.setitem(sys.modules, "families.example_owner.example_cli", SimpleNamespace( + add_build_arguments=add_arguments, prepare_build_request=prepare, + )) (tmp_path / "config.json").write_text('{"model_type":"example_model"}') monkeypatch.setattr(build_cli, "resolve_family", lambda metadata, *args: ("example_owner", support)) monkeypatch.setattr(build_cli, "_load_family", lambda *_: pytest.fail("GPU builder imported by CLI")) @@ -590,9 +593,12 @@ def test_build_family_help_does_not_require_output_or_load_builder(monkeypatch, (tmp_path / "config.json").write_text('{"model_type":"example_model"}') support = FamilySupport( ("example_task",), "example_task", + build_cli_module="example_cli", + ) + monkeypatch.setitem(sys.modules, "families.example_owner.example_cli", SimpleNamespace( add_build_arguments=lambda parser: parser.add_argument("--example-setting"), prepare_build_request=lambda *_: pytest.fail("help prepared a build"), - ) + )) monkeypatch.setattr(build_cli, "resolve_family", lambda _: ("example_owner", support)) monkeypatch.setattr(build_cli, "_load_family", lambda *_: pytest.fail("help imported GPU builder")) with pytest.raises(SystemExit) as error: @@ -608,10 +614,13 @@ def test_build_family_hook_cannot_replace_owner_or_contract(monkeypatch, tmp_pat (tmp_path / "config.json").write_text('{"model_type":"example_model"}') support = FamilySupport( ("example_task",), "example_task", + build_cli_module="example_cli", + ) + monkeypatch.setitem(sys.modules, "families.example_owner.example_cli", SimpleNamespace( prepare_build_request=lambda request, _: ( replace(request, family="another_owner") if wrong_owner else object() ), - ) + )) monkeypatch.setattr(build_cli, "resolve_family", lambda _: ("example_owner", support)) monkeypatch.setattr(build_cli, "build", lambda *_: pytest.fail("invalid request reached build")) with pytest.raises(TypeError, match="preserve the owning BuildRequest"): diff --git a/core/builder/tests/test_model_support.py b/core/builder/tests/test_model_support.py index 0144dbdc7d..a3826c0c9e 100644 --- a/core/builder/tests/test_model_support.py +++ b/core/builder/tests/test_model_support.py @@ -310,3 +310,17 @@ def test_family_owned_exact_metadata_shapes_resolve_without_priority( ) -> None: family, _ = resolve_family(metadata) assert family == expected_family + +@pytest.mark.parametrize("module", ["", ".cli", "../other", "cli.", "a..b", "module-name", 12]) +def test_build_cli_module_must_be_owned_module_path(module): + with pytest.raises(ValueError, match="within the owning family"): + FamilySupport(("text_generation",), "text_generation", build_cli_module=module) + + +def test_family_build_cli_declaration_remains_import_free(): + declaration = family_support( + model_types=("example_model",), tasks=("text_generation",), + default_task="text_generation", build_cli_module="optional.cli", + ) + metadata = ModelMetadata({"model_type": "example_model"}, {}, frozenset()) + assert declaration(metadata).build_cli_module == "optional.cli" diff --git a/website/docs/api/python-builder.md b/website/docs/api/python-builder.md index 0b6d20e048..9b9afe7de6 100644 --- a/website/docs/api/python-builder.md +++ b/website/docs/api/python-builder.md @@ -32,8 +32,8 @@ checkpoint; `BuildRequest` does not perform another discovery pass. ## Family-owned build arguments A family may provide `add_build_arguments(parser)` and -`prepare_build_request(request, args)` through its lightweight `FamilySupport` -declaration. The CLI resolves the owner, registers only that family's options, +`prepare_build_request(request, args)` in a family-local module named by the lightweight +`FamilySupport.build_cli_module` declaration. The CLI resolves the owner, registers only that family's options, and lets it return a family-owned `BuildRequest` subclass. The hook must retain the resolved family. Ordinary families need no changes. From 97e4e150287454880af839e215aeba8d547cd872 Mon Sep 17 00:00:00 2001 From: Joshua Calafato Date: Tue, 22 Sep 2026 22:02:20 +0000 Subject: [PATCH 07/20] fix(cli): keep model help free of downloads Signed-off-by: Joshua Calafato --- .../tensorrt_model_connect/build_cli.py | 10 +++++++++- core/builder/tests/test_build_cli.py | 18 ++++++++++++++++++ website/docs/api/python-builder.md | 4 +++- 3 files changed, 30 insertions(+), 2 deletions(-) diff --git a/core/builder/tensorrt_model_connect/build_cli.py b/core/builder/tensorrt_model_connect/build_cli.py index 2754667a69..a77e4640d6 100644 --- a/core/builder/tensorrt_model_connect/build_cli.py +++ b/core/builder/tensorrt_model_connect/build_cli.py @@ -92,7 +92,15 @@ def main(argv: Sequence[str] | None = None) -> int: preliminary, unknown = base_parser.parse_known_args(preliminary_arguments) if unknown and arguments[0] == "build" and arguments[1].startswith("-"): base_parser.error("MODEL must immediately follow build when family options are used") - model_dir = _resolve_model(preliminary.model, preliminary.revision) + if family_help and not Path(preliminary.model).is_dir(): + # Remote or missing inputs cannot provide local metadata. Help must + # remain side-effect free instead of acquiring a checkpoint. + base_parser.parse_args(arguments) + return 0 + model_dir = ( + Path(preliminary.model) if family_help + else _resolve_model(preliminary.model, preliminary.revision) + ) metadata = load_model_metadata(model_dir) try: family, support = ( diff --git a/core/builder/tests/test_build_cli.py b/core/builder/tests/test_build_cli.py index 580d4e5bd7..3b31ffdafc 100644 --- a/core/builder/tests/test_build_cli.py +++ b/core/builder/tests/test_build_cli.py @@ -590,6 +590,7 @@ def test_build_family_options_are_not_global(monkeypatch, tmp_path): def test_build_family_help_does_not_require_output_or_load_builder(monkeypatch, tmp_path, capsys): + monkeypatch.setattr(build_cli, "_resolve_model", lambda *_: pytest.fail("help acquired a model")) (tmp_path / "config.json").write_text('{"model_type":"example_model"}') support = FamilySupport( ("example_task",), "example_task", @@ -625,3 +626,20 @@ def test_build_family_hook_cannot_replace_owner_or_contract(monkeypatch, tmp_pat monkeypatch.setattr(build_cli, "build", lambda *_: pytest.fail("invalid request reached build")) with pytest.raises(TypeError, match="preserve the owning BuildRequest"): build_cli.main(["build", str(tmp_path), "-o", str(tmp_path / "out")]) + + +@pytest.mark.parametrize("help_option", ["-h", "--help"]) +@pytest.mark.parametrize("explicit_family", [False, True]) +def test_remote_model_help_never_acquires_checkpoint(monkeypatch, capsys, help_option, explicit_family): + monkeypatch.setattr(build_cli, "_resolve_model", lambda *_: pytest.fail("help downloaded model")) + monkeypatch.setattr(build_cli, "resolve_family", lambda *_: pytest.fail("remote help resolved owner")) + monkeypatch.setattr(build_cli, "build", lambda *_: pytest.fail("help started build")) + arguments = ["build", "example-organization/uncached-model", help_option] + if explicit_family: + arguments += ["--family", "example_owner"] + with pytest.raises(SystemExit) as error: + build_cli.main(arguments) + assert error.value.code == 0 + output = capsys.readouterr() + assert "--precision" in output.out + assert not output.err diff --git a/website/docs/api/python-builder.md b/website/docs/api/python-builder.md index 9b9afe7de6..fb38c4f46e 100644 --- a/website/docs/api/python-builder.md +++ b/website/docs/api/python-builder.md @@ -44,7 +44,9 @@ extra fields, validation and execution choices. Put the model immediately after `build` when using family-specific options. `trtmc build /path/to/model --help` shows the resolved family's options without -importing its GPU builder. Model resolution precedes family-specific argument +importing its GPU builder. Help never downloads a checkpoint: remote model IDs +or missing local directories show generic build help instead. Model resolution +precedes family-specific argument validation; request preparation precedes backend import and bundle creation. ## Optional graph transform From 1f818dcf38923f0cc040e1163151ec61f7c38cb4 Mon Sep 17 00:00:00 2001 From: Joshua Calafato Date: Tue, 22 Sep 2026 22:24:22 +0000 Subject: [PATCH 08/20] fix(build): verify native SDK library and pip bootstrap Query the selected TensorRT library by absolute path and reject complete-version mismatches, including changed libraries in an existing CMake cache. Upgrade isolated pip only when the bootstrap predates report support, respecting an offline wheelhouse. Signed-off-by: Joshua Calafato --- cmake/edge_llm/CheckNative.cmake | 39 +++++++++++++++++++++++++++ cmake/edge_llm/EdgeLLM.cmake | 1 + cmake/edge_llm/EdgeLLMConfig.cmake.in | 1 + cmake/edge_llm/Prepare.cmake.in | 10 +++++++ 4 files changed, 51 insertions(+) diff --git a/cmake/edge_llm/CheckNative.cmake b/cmake/edge_llm/CheckNative.cmake index 03dd575b39..2d7118de5f 100644 --- a/cmake/edge_llm/CheckNative.cmake +++ b/cmake/edge_llm/CheckNative.cmake @@ -71,3 +71,42 @@ function(_edgellm_check_json_headers include_dir vendor_dir) message(FATAL_ERROR "EdgeLLM requires the pinned nlohmann_json headers, not only the same version label. Set nlohmann_json_DIR to an installation of the pinned Edge 3rdParty/nlohmannJson dependency. Mismatch: ${_relative}") endforeach() endfunction() + +# Verify the selected native library itself, not only its accompanying headers. +function(_edgellm_check_trt_library library expected) + if(CMAKE_CROSSCOMPILING) + message(FATAL_ERROR "TensorRT library validation requires native execution") + endif() + set(_probe "${CMAKE_CURRENT_BINARY_DIR}/CMakeFiles/edgellm-library-version.cpp") + file(WRITE "${_probe}" [=[ +#include +#include +int main(int argc, char** argv) { + if (argc != 2) return 1; + void* library = dlopen(argv[1], RTLD_NOW | RTLD_LOCAL); + if (!library) { std::cerr << dlerror(); return 2; } + const char* names[] = {"getInferLibMajorVersion", "getInferLibMinorVersion", + "getInferLibPatchVersion", "getInferLibBuildVersion"}; + for (int i = 0; i < 4; ++i) { + auto version = reinterpret_cast(dlsym(library, names[i])); + if (!version) { std::cerr << "Missing " << names[i]; dlclose(library); return 3; } + if (i) std::cout << "."; + std::cout << version(); + } + dlclose(library); + return 0; +} +]=]) + unset(_edge_version_run CACHE) + unset(_edge_version_compiled CACHE) + try_run(_edge_version_run _edge_version_compiled + "${CMAKE_CURRENT_BINARY_DIR}/CMakeFiles/edgellm-library-version" "${_probe}" + LINK_LIBRARIES "${CMAKE_DL_LIBS}" ARGS "${library}" + RUN_OUTPUT_VARIABLE _actual COMPILE_OUTPUT_VARIABLE _compile_output) + if(NOT _edge_version_compiled OR NOT _edge_version_run STREQUAL "0") + message(FATAL_ERROR "Cannot verify selected TensorRT library ${library}: ${_actual} ${_compile_output}") + endif() + if(NOT _actual STREQUAL expected) + message(FATAL_ERROR "EdgeLLM requires TensorRT library ${expected}; selected ${library} reports ${_actual}") + endif() +endfunction() diff --git a/cmake/edge_llm/EdgeLLM.cmake b/cmake/edge_llm/EdgeLLM.cmake index 7c1ef92c9d..a4b8565f49 100644 --- a/cmake/edge_llm/EdgeLLM.cmake +++ b/cmake/edge_llm/EdgeLLM.cmake @@ -30,6 +30,7 @@ set(_edge_root "${CMAKE_BINARY_DIR}/_deps/edgellm") set(_edge_prefix "${_edge_root}/install") find_package(EdgeLLM ${_edge_version} EXACT CONFIG QUIET) if(EdgeLLM_FOUND AND NOT EdgeLLM_PREFIX STREQUAL _edge_prefix) + _edgellm_check_trt_library("${EdgeLLM_TRT_LIBRARY}" "${EdgeLLM_TENSORRT_VERSION}") _edgellm_json_include(_edge_json_include) _edgellm_check_json_headers("${_edge_json_include}" "${EdgeLLM_PREFIX}/include/edgellm/3rdParty/nlohmannJson") if(NOT EdgeLLM_REVISION STREQUAL _edge_revision) diff --git a/cmake/edge_llm/EdgeLLMConfig.cmake.in b/cmake/edge_llm/EdgeLLMConfig.cmake.in index 63f05b8dd8..7f8da5df01 100644 --- a/cmake/edge_llm/EdgeLLMConfig.cmake.in +++ b/cmake/edge_llm/EdgeLLMConfig.cmake.in @@ -54,6 +54,7 @@ _edgellm_trt_version("${EdgeLLM_TRT_INCLUDE_DIR}" _edge_current_trt) if(NOT _edge_current_trt STREQUAL EdgeLLM_TENSORRT_VERSION) message(FATAL_ERROR "EdgeLLM requires TensorRT ${EdgeLLM_TENSORRT_VERSION}; found ${_edge_current_trt}") endif() +_edgellm_check_trt_library("${EdgeLLM_TRT_LIBRARY}" "${EdgeLLM_TENSORRT_VERSION}") _edgellm_check_gpu("${EdgeLLM_CUDA_ARCHITECTURE}") if(NOT TARGET EdgeLLM::Core) add_library(EdgeLLM::Core STATIC IMPORTED) diff --git a/cmake/edge_llm/Prepare.cmake.in b/cmake/edge_llm/Prepare.cmake.in index bf19447034..f2ba6f09f5 100644 --- a/cmake/edge_llm/Prepare.cmake.in +++ b/cmake/edge_llm/Prepare.cmake.in @@ -20,6 +20,16 @@ set(_pip_options --isolated install --no-user) if(NOT "@TRTMC_EDGELLM_WHEELHOUSE@" STREQUAL "") list(APPEND _pip_options --no-index --find-links "@TRTMC_EDGELLM_WHEELHOUSE@") endif() +# Older supported Python bootstraps may seed pip before --report was introduced. +execute_process(COMMAND "@_edge_python@" -I -m pip --version + OUTPUT_VARIABLE _pip_version OUTPUT_STRIP_TRAILING_WHITESPACE COMMAND_ERROR_IS_FATAL ANY) +if(NOT _pip_version MATCHES "^pip ([0-9]+\\.[0-9]+(\\.[0-9]+)?)") + message(FATAL_ERROR "Cannot determine the isolated environment pip version") +endif() +if(CMAKE_MATCH_1 VERSION_LESS "22.2") + # _pip_options keeps offline upgrades restricted to the explicit wheelhouse. + run("@_edge_python@" -I -m pip ${_pip_options} --upgrade "pip>=22.2") +endif() file(GLOB _trt_wheels "@TRTMC_EDGELLM_TRT_ROOT@/python/tensorrt-*-cp@Python3_VERSION_MAJOR@@Python3_VERSION_MINOR@-none-linux_@CMAKE_SYSTEM_PROCESSOR@.whl") list(LENGTH _trt_wheels _wheel_count) if(NOT _wheel_count EQUAL 1) From 068ad57e5b37de49f9ffe7124ad5bec49a516f0f Mon Sep 17 00:00:00 2001 From: Joshua Calafato Date: Tue, 22 Sep 2026 22:50:53 +0000 Subject: [PATCH 09/20] fix(build): honor requested Edge package architecture Signed-off-by: Joshua Calafato --- cmake/edge_llm/EdgeLLM.cmake | 11 +++++++---- 1 file changed, 7 insertions(+), 4 deletions(-) diff --git a/cmake/edge_llm/EdgeLLM.cmake b/cmake/edge_llm/EdgeLLM.cmake index a4b8565f49..279037ca54 100644 --- a/cmake/edge_llm/EdgeLLM.cmake +++ b/cmake/edge_llm/EdgeLLM.cmake @@ -28,8 +28,15 @@ set(_edge_version "0.10.1") set(_edge_revision "e8b29522938901f6df19ebeedd4b69bc8edbcd97") set(_edge_root "${CMAKE_BINARY_DIR}/_deps/edgellm") set(_edge_prefix "${_edge_root}/install") +set(TRTMC_EDGELLM_CUDA_ARCHITECTURE "${CMAKE_CUDA_ARCHITECTURES}" CACHE STRING "One local GPU architecture for Edge-LLM") +if(NOT TRTMC_EDGELLM_CUDA_ARCHITECTURE MATCHES "^[0-9]+$") + message(FATAL_ERROR "Set TRTMC_EDGELLM_CUDA_ARCHITECTURE to one local GPU architecture, e.g. 80") +endif() find_package(EdgeLLM ${_edge_version} EXACT CONFIG QUIET) if(EdgeLLM_FOUND AND NOT EdgeLLM_PREFIX STREQUAL _edge_prefix) + if(NOT EdgeLLM_CUDA_ARCHITECTURE STREQUAL TRTMC_EDGELLM_CUDA_ARCHITECTURE) + message(FATAL_ERROR "EdgeLLM package architecture ${EdgeLLM_CUDA_ARCHITECTURE} differs from requested ${TRTMC_EDGELLM_CUDA_ARCHITECTURE}") + endif() _edgellm_check_trt_library("${EdgeLLM_TRT_LIBRARY}" "${EdgeLLM_TENSORRT_VERSION}") _edgellm_json_include(_edge_json_include) _edgellm_check_json_headers("${_edge_json_include}" "${EdgeLLM_PREFIX}/include/edgellm/3rdParty/nlohmannJson") @@ -51,13 +58,9 @@ include(CMakePackageConfigHelpers) find_package(Python3 3.10 REQUIRED COMPONENTS Interpreter) find_package(Threads REQUIRED) set(TRTMC_EDGELLM_TRT_ROOT "$ENV{TRT_ROOT}" CACHE PATH "Native TensorRT SDK, including its Python wheel") -set(TRTMC_EDGELLM_CUDA_ARCHITECTURE "${CMAKE_CUDA_ARCHITECTURES}" CACHE STRING "One local GPU architecture for Edge-LLM") set(TRTMC_EDGELLM_JOBS 2 CACHE STRING "Parallel Edge-LLM native and AOT compilation jobs") set(TRTMC_EDGELLM_WHEELHOUSE "" CACHE PATH "Optional complete offline Python wheelhouse") set(TRTMC_EDGELLM_GIT_MIRROR "" CACHE PATH "Optional local mirror of the pinned upstream Git repository") -if(NOT TRTMC_EDGELLM_CUDA_ARCHITECTURE MATCHES "^[0-9]+$") - message(FATAL_ERROR "Set TRTMC_EDGELLM_CUDA_ARCHITECTURE to one local GPU architecture, e.g. 80") -endif() if(NOT EXISTS "${TRTMC_EDGELLM_TRT_ROOT}/include/NvInfer.h") message(FATAL_ERROR "TRTMC_EDGELLM_TRT_ROOT must contain the native TensorRT SDK") endif() From 59778d21e37dd6e0f42b2aad9b0d091d582b3ccf Mon Sep 17 00:00:00 2001 From: Joshua Calafato Date: Wed, 23 Sep 2026 15:15:15 +0000 Subject: [PATCH 10/20] fix(build): harden family CLI and SDK discovery Keep staged CLI parsing side-effect free and preserve core options before the model. Identify the configured CUDA toolkit independently of cuda-python, exclude stale generated packages during reconfiguration, and install complete plugin symlink chains. Signed-off-by: Joshua Calafato --- cmake/edge_llm/EdgeLLM.cmake | 28 +++++++++++++- cmake/edge_llm/README.md | 4 ++ core/builder/tensorrt_model_connect/build.py | 26 ++++++++++++- .../tensorrt_model_connect/build_cli.py | 33 +++++++++++++---- core/builder/tests/test_build.py | 37 +++++++++++++++++-- core/builder/tests/test_build_cli.py | 37 +++++++++++++++++-- 6 files changed, 147 insertions(+), 18 deletions(-) diff --git a/cmake/edge_llm/EdgeLLM.cmake b/cmake/edge_llm/EdgeLLM.cmake index 279037ca54..3dba21264c 100644 --- a/cmake/edge_llm/EdgeLLM.cmake +++ b/cmake/edge_llm/EdgeLLM.cmake @@ -32,7 +32,30 @@ set(TRTMC_EDGELLM_CUDA_ARCHITECTURE "${CMAKE_CUDA_ARCHITECTURES}" CACHE STRING " if(NOT TRTMC_EDGELLM_CUDA_ARCHITECTURE MATCHES "^[0-9]+$") message(FATAL_ERROR "Set TRTMC_EDGELLM_CUDA_ARCHITECTURE to one local GPU architecture, e.g. 80") endif() +# Do not import this build tree's previous generated package before regenerating +# it: its baked SDK checks and imported targets may describe the old configure. +set(_edge_package_dir "${_edge_prefix}/lib/cmake/EdgeLLM") +get_filename_component(_edge_package_real "${_edge_package_dir}" REALPATH) +if(EdgeLLM_DIR) + get_filename_component(_edge_cached_real "${EdgeLLM_DIR}" REALPATH) + if(_edge_cached_real STREQUAL _edge_package_real) + unset(EdgeLLM_DIR CACHE) + unset(EdgeLLM_DIR) + endif() +endif() +set(_edge_saved_ignore_path "${CMAKE_IGNORE_PATH}") +list(APPEND CMAKE_IGNORE_PATH "${_edge_package_dir}" "${_edge_package_real}") find_package(EdgeLLM ${_edge_version} EXACT CONFIG QUIET) +set(CMAKE_IGNORE_PATH "${_edge_saved_ignore_path}") + +function(_edgellm_install_plugin) + # Preserve the complete SONAME chain when lib and lib64 differ. Install-time + # expansion also honors cmake --install --prefix and DESTDIR. + install(CODE "file(INSTALL + DESTINATION \"\${CMAKE_INSTALL_PREFIX}/${CMAKE_INSTALL_LIBDIR}\" + TYPE SHARED_LIBRARY FOLLOW_SYMLINK_CHAIN + FILES \"$\")" COMPONENT EdgeLLM) +endfunction() if(EdgeLLM_FOUND AND NOT EdgeLLM_PREFIX STREQUAL _edge_prefix) if(NOT EdgeLLM_CUDA_ARCHITECTURE STREQUAL TRTMC_EDGELLM_CUDA_ARCHITECTURE) message(FATAL_ERROR "EdgeLLM package architecture ${EdgeLLM_CUDA_ARCHITECTURE} differs from requested ${TRTMC_EDGELLM_CUDA_ARCHITECTURE}") @@ -49,7 +72,7 @@ if(EdgeLLM_FOUND AND NOT EdgeLLM_PREFIX STREQUAL _edge_prefix) if(TRTMC_EDGELLM_ONNX AND (NOT EdgeLLM_ONNX OR NOT EXISTS "${EdgeLLM_ONNX_BUILDER}")) message(FATAL_ERROR "EdgeLLM package lacks requested ONNX tools; rebuild with TRTMC_EDGELLM_ONNX=ON") endif() - install(FILES "$" DESTINATION "${CMAKE_INSTALL_LIBDIR}" COMPONENT EdgeLLM) + _edgellm_install_plugin() return() endif() @@ -114,10 +137,11 @@ ExternalProject_Add_StepDependencies(trtmc_edgellm_dependency configure "${_edge ExternalProject_Add_StepDependencies(trtmc_edgellm_dependency install "${_edge_root}/Install.cmake") # Generated package targets refer to declared future byproducts; their build dependency # prevents consumers from compiling or linking until installation completes. +set(EdgeLLM_DIR "${_edge_package_dir}" CACHE PATH "Edge-LLM package directory" FORCE) find_package(EdgeLLM ${_edge_version} EXACT CONFIG REQUIRED PATHS "${_edge_prefix}/lib/cmake/EdgeLLM" NO_DEFAULT_PATH) add_dependencies(EdgeLLM::Core trtmc_edgellm_dependency) add_dependencies(EdgeLLM::Plugin trtmc_edgellm_dependency) install(DIRECTORY "${_edge_prefix}/" DESTINATION . USE_SOURCE_PERMISSIONS COMPONENT EdgeLLM) # Family DSOs may use lib64; their dynamically loaded plugin must remain adjacent. -install(FILES "$" DESTINATION "${CMAKE_INSTALL_LIBDIR}" COMPONENT EdgeLLM) +_edgellm_install_plugin() diff --git a/cmake/edge_llm/README.md b/cmake/edge_llm/README.md index 89ae872efc..c4268d32a5 100644 --- a/cmake/edge_llm/README.md +++ b/cmake/edge_llm/README.md @@ -63,6 +63,10 @@ the exporter using the installed Python and obtain the native builder path from qualified model support. Reusing an installed package that lacks a requested capability is an error; no dependency installation occurs during model builds. The CUDA and TensorRT shared libraries must remain available to the executable. +When building models, select the same native CUDA toolkit with CUDACXX (the +NVCC executable) or CUDAToolkit_ROOT (the SDK root); CUDA_HOME, CUDA_PATH, +and then NVCC on PATH are fallbacks. Platform admission reads this compiler's +release, not the independently versioned cuda-python binding's build toolkit. Run the existing runtime and family checks against this installation (some tests require a local GPU): diff --git a/core/builder/tensorrt_model_connect/build.py b/core/builder/tensorrt_model_connect/build.py index 6ec9bd5ef1..896b191867 100644 --- a/core/builder/tensorrt_model_connect/build.py +++ b/core/builder/tensorrt_model_connect/build.py @@ -11,6 +11,8 @@ import platform import re import sys +import shutil +import subprocess from dataclasses import dataclass from pathlib import Path from types import ModuleType @@ -98,6 +100,26 @@ def cmake_prefixes() -> list[Path]: return [*prefixes, Path(sys.prefix)] +def _cuda_toolkit_version() -> str: + """Identify the selected native compiler, not cuda-python's build toolkit.""" + compiler = os.environ.get("CUDACXX") + if not compiler: + root = next( + (os.environ[key] for key in ("CUDAToolkit_ROOT", "CUDA_HOME", "CUDA_PATH") + if os.environ.get(key)), None + ) + compiler = str(Path(root) / "bin" / "nvcc") if root else shutil.which("nvcc") + if not compiler: + raise RuntimeError("CUDA toolkit not found; set CUDAToolkit_ROOT or CUDACXX") + result = subprocess.run( + [compiler, "--version"], check=True, capture_output=True, text=True, + ) + version = re.search(r"release\s+(\d+\.\d+)", result.stdout) + if version is None: + raise RuntimeError(f"Cannot identify CUDA toolkit from {compiler} --version") + return version.group(1) + + def detect_local_platform() -> dict: """Return executing GPU and native SDK identity without selecting a model. @@ -118,7 +140,7 @@ def checked(result): device = checked(runtime.cudaGetDevice()) gpu = checked(runtime.cudaGetDeviceProperties(device)) - cuda = checked(runtime.cudaRuntimeGetVersion()) + cuda_version = _cuda_toolkit_version() try: release = platform.freedesktop_os_release() if sys.platform == "linux" else {} except OSError: @@ -128,7 +150,7 @@ def checked(result): "os_version": release.get("VERSION_ID", platform.release()), "arch": platform.machine(), "sm": gpu.major * 10 + gpu.minor, - "cuda_version": f"{cuda // 1000}.{cuda % 1000 // 10}", + "cuda_version": cuda_version, "tensorrt_version": trt.__version__, } diff --git a/core/builder/tensorrt_model_connect/build_cli.py b/core/builder/tensorrt_model_connect/build_cli.py index a77e4640d6..2d2465aec9 100644 --- a/core/builder/tensorrt_model_connect/build_cli.py +++ b/core/builder/tensorrt_model_connect/build_cli.py @@ -25,12 +25,13 @@ def _parser( prepare_family: object | None = None, *, build_hooks: object | None = None, - require_output: bool = True, + require_output: bool = True, require_model: bool = True, ) -> argparse.ArgumentParser: parser = argparse.ArgumentParser(prog="trtmc") commands = parser.add_subparsers(dest="command", required=True) - build_parser = commands.add_parser("build", help="Build one TensorRT bundle") - build_parser.add_argument("model", help="Hugging Face model ID or local snapshot") + build_parser = commands.add_parser("build", help="Build one TensorRT bundle", allow_abbrev=False) + if require_model: + build_parser.add_argument("model", help="Hugging Face model ID or local snapshot") build_parser.add_argument("-o", "--output", type=Path, required=require_output) build_parser.add_argument( "--family", help="Select one compatible family instead of automatic resolution" @@ -75,7 +76,6 @@ def main(argv: Sequence[str] | None = None) -> int: arguments = list(sys.argv[1:] if argv is None else argv) family_help = ( len(arguments) > 1 and arguments[0] == "build" - and not arguments[1].startswith("-") and any(arg in {"-h", "--help"} for arg in arguments) ) base_parser = _parser(require_output=not family_help) @@ -89,9 +89,19 @@ def main(argv: Sequence[str] | None = None) -> int: preliminary_arguments = ( [arg for arg in arguments if arg not in {"-h", "--help"}] if family_help else arguments ) - preliminary, unknown = base_parser.parse_known_args(preliminary_arguments) - if unknown and arguments[0] == "build" and arguments[1].startswith("-"): - base_parser.error("MODEL must immediately follow build when family options are used") + if arguments and arguments[0] == "build": + # Without a positional, parse_known_args preserves MODEL and family + # options in order. Never acquire a checkpoint from an unknown option's + # value; known core options may safely precede MODEL. + _, remaining = _parser(require_output=False, require_model=False).parse_known_args( + preliminary_arguments + ) + if remaining and remaining[0].startswith("-") and remaining[0] != "--": + base_parser.error("MODEL must precede family options") + if family_help and not remaining: + base_parser.parse_args(arguments) + return 0 + preliminary, _ = base_parser.parse_known_args(preliminary_arguments) if family_help and not Path(preliminary.model).is_dir(): # Remote or missing inputs cannot provide local metadata. Help must # remain side-effect free instead of acquiring a checkpoint. @@ -101,7 +111,14 @@ def main(argv: Sequence[str] | None = None) -> int: Path(preliminary.model) if family_help else _resolve_model(preliminary.model, preliminary.revision) ) - metadata = load_model_metadata(model_dir) + try: + metadata = load_model_metadata(model_dir) + except (ValueError, OSError): + if not family_help: + raise + # Empty/invalid local directories still have useful generic help. + base_parser.parse_args(arguments) + return 0 try: family, support = ( resolve_family(metadata, preliminary.family) diff --git a/core/builder/tests/test_build.py b/core/builder/tests/test_build.py index ac892f7b6e..35e87751fd 100644 --- a/core/builder/tests/test_build.py +++ b/core/builder/tests/test_build.py @@ -324,7 +324,7 @@ def native_platform_bindings(monkeypatch): runtime = SimpleNamespace( cudaGetDevice=Mock(return_value=(0, 3)), cudaGetDeviceProperties=Mock(return_value=(0, SimpleNamespace(major=8, minor=6))), - cudaRuntimeGetVersion=Mock(return_value=(0, 13030)), + cudaRuntimeGetVersion=Mock(return_value=(0, 13000)), ) monkeypatch.setitem(sys.modules, "tensorrt", SimpleNamespace(__version__="11.1.0.106")) monkeypatch.setitem(sys.modules, "cuda.bindings", SimpleNamespace(runtime=runtime)) @@ -334,6 +334,7 @@ def native_platform_bindings(monkeypatch): build_core.platform, "freedesktop_os_release", lambda: {"VERSION_ID": "24.04"} ) monkeypatch.setattr(build_core.platform, "release", lambda: "fallback-release") + monkeypatch.setattr(build_core, "_cuda_toolkit_version", lambda: "13.3") return runtime @@ -357,11 +358,11 @@ def test_native_platform_uses_executing_cuda_device_and_full_sdk( } native_platform_bindings.cudaGetDevice.assert_called_once_with() native_platform_bindings.cudaGetDeviceProperties.assert_called_once_with(3) - native_platform_bindings.cudaRuntimeGetVersion.assert_called_once_with() + native_platform_bindings.cudaRuntimeGetVersion.assert_not_called() @pytest.mark.parametrize( - "failing", ["cudaGetDevice", "cudaGetDeviceProperties", "cudaRuntimeGetVersion"] + "failing", ["cudaGetDevice", "cudaGetDeviceProperties"] ) def test_native_platform_propagates_cuda_discovery_failure(native_platform_bindings, failing): getattr(native_platform_bindings, failing).return_value = (35,) @@ -374,3 +375,33 @@ def test_native_platform_retains_nonlinux_identity(native_platform_bindings, mon result = build_core.detect_local_platform() assert result["os"] == "win32" assert result["os_version"] == "fallback-release" + +@pytest.mark.parametrize("source", ["CUDACXX", "CUDAToolkit_ROOT", "CUDA_HOME", "CUDA_PATH", "PATH"]) +def test_cuda_toolkit_version_uses_selected_compiler(monkeypatch, source): + from unittest.mock import Mock + + for name in ("CUDACXX", "CUDAToolkit_ROOT", "CUDA_HOME", "CUDA_PATH"): + monkeypatch.delenv(name, raising=False) + compiler = "/selected/bin/nvcc" + monkeypatch.setattr(build_core.shutil, "which", lambda _: compiler) + if source != "PATH": + monkeypatch.setenv(source, compiler if source == "CUDACXX" else "/selected") + run = Mock(return_value=SimpleNamespace(stdout="Cuda compilation tools, release 13.3, V13.3.1")) + monkeypatch.setattr(build_core.subprocess, "run", run) + assert build_core._cuda_toolkit_version() == "13.3" + run.assert_called_once_with([compiler, "--version"], check=True, capture_output=True, text=True) + + +def test_cuda_toolkit_version_does_not_guess_when_missing(monkeypatch): + for name in ("CUDACXX", "CUDAToolkit_ROOT", "CUDA_HOME", "CUDA_PATH"): + monkeypatch.delenv(name, raising=False) + monkeypatch.setattr(build_core.shutil, "which", lambda _: None) + with pytest.raises(RuntimeError, match="CUDA toolkit not found"): + build_core._cuda_toolkit_version() + + +def test_cuda_toolkit_version_rejects_unrecognized_output(monkeypatch): + monkeypatch.setenv("CUDACXX", "/selected/nvcc") + monkeypatch.setattr(build_core.subprocess, "run", lambda *_, **__: SimpleNamespace(stdout="")) + with pytest.raises(RuntimeError, match="Cannot identify CUDA toolkit"): + build_core._cuda_toolkit_version() diff --git a/core/builder/tests/test_build_cli.py b/core/builder/tests/test_build_cli.py index 3b31ffdafc..a5db0b1959 100644 --- a/core/builder/tests/test_build_cli.py +++ b/core/builder/tests/test_build_cli.py @@ -544,7 +544,11 @@ def test_prepare_structure_requires_a_callable_family_hook( @pytest.mark.parametrize("explicit_family", [False, True]) -def test_build_family_hooks_forward_a_typed_request(monkeypatch, tmp_path, explicit_family): +@pytest.mark.parametrize("core_prefix", [[], ["--verbose"], ["--family", "example_owner"]]) +@pytest.mark.parametrize("option", ["--example-setting", "--image"]) +def test_build_family_hooks_forward_a_typed_request( + monkeypatch, tmp_path, explicit_family, core_prefix, option +): from dataclasses import dataclass @dataclass(frozen=True) @@ -552,7 +556,7 @@ class ExampleRequest(build_cli.BuildRequest): example_setting: str = "" def add_arguments(parser): - parser.add_argument("--example-setting", required=True) + parser.add_argument(option, dest="example_setting", required=True) def prepare(request, args): return ExampleRequest(**vars(request), example_setting=args.example_setting) @@ -569,7 +573,7 @@ def prepare(request, args): monkeypatch.setattr(build_cli, "_load_family", lambda *_: pytest.fail("GPU builder imported by CLI")) seen = [] monkeypatch.setattr(build_cli, "build", seen.append) - arguments = ["build", str(tmp_path), "-o", str(tmp_path / "out"), "--example-setting", "verbatim"] + arguments = ["build", *core_prefix, str(tmp_path), "-o", str(tmp_path / "out"), option, "verbatim"] if explicit_family: arguments += ["--family", "example_owner"] assert build_cli.main(arguments) == 0 @@ -643,3 +647,30 @@ def test_remote_model_help_never_acquires_checkpoint(monkeypatch, capsys, help_o output = capsys.readouterr() assert "--precision" in output.out assert not output.err + +@pytest.mark.parametrize("metadata", [None, "", "{broken", "[]"]) +@pytest.mark.parametrize("prefix", [[], ["--family", "example_owner"]]) +def test_local_invalid_metadata_help_is_generic(monkeypatch, tmp_path, capsys, metadata, prefix): + if metadata is not None: + (tmp_path / "config.json").write_text(metadata) + monkeypatch.setattr(build_cli, "_resolve_model", lambda *_: pytest.fail("help acquired model")) + monkeypatch.setattr(build_cli, "build", lambda *_: pytest.fail("help built model")) + with pytest.raises(SystemExit) as error: + build_cli.main(["build", *prefix, str(tmp_path), "--help"]) + assert error.value.code == 0 + assert "--precision" in capsys.readouterr().out + + +@pytest.mark.parametrize("prefix", [[], ["--family", "example_owner"], ["--verbose"]]) +def test_unknown_build_option_before_model_never_acquires(monkeypatch, tmp_path, prefix): + monkeypatch.setattr(build_cli, "_resolve_model", lambda *_: pytest.fail("download before parsing")) + with pytest.raises(SystemExit) as error: + build_cli.main(["build", *prefix, "--example-setting", "not-a-model", + str(tmp_path), "-o", str(tmp_path / "out")]) + assert error.value.code == 2 + + +def test_non_help_invalid_metadata_is_not_hidden(tmp_path): + (tmp_path / "config.json").write_text("{broken") + with pytest.raises(ValueError): + build_cli.main(["build", str(tmp_path), "-o", str(tmp_path / "out")]) From a766c2cba0e1bed28765698217ae1a2fb263b222 Mon Sep 17 00:00:00 2001 From: Joshua Calafato Date: Wed, 23 Sep 2026 15:52:54 +0000 Subject: [PATCH 11/20] docs(build): clarify family option ordering Signed-off-by: Joshua Calafato --- website/docs/api/python-builder.md | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/website/docs/api/python-builder.md b/website/docs/api/python-builder.md index fb38c4f46e..f073864f50 100644 --- a/website/docs/api/python-builder.md +++ b/website/docs/api/python-builder.md @@ -42,7 +42,7 @@ entrypoints. There is no shared execution-variant list, companion interpretation GPU offload selection, or alternative-builder dispatch. The family owns all extra fields, validation and execution choices. -Put the model immediately after `build` when using family-specific options. +Put MODEL before family-specific options. Known core options may precede MODEL. `trtmc build /path/to/model --help` shows the resolved family's options without importing its GPU builder. Help never downloads a checkpoint: remote model IDs or missing local directories show generic build help instead. Model resolution From 3d39af4d644fa0957057b51bb826605bfe86517a Mon Sep 17 00:00:00 2001 From: Joshua Calafato Date: Wed, 23 Sep 2026 16:51:56 +0000 Subject: [PATCH 12/20] fix(build): align native SDK and compiler selection Reject different TensorRT headers or libraries between Model Connect and Edge provisioning/reuse. Parse flag-bearing compiler commands without a shell and bound version probes while preserving diagnostic causes. Signed-off-by: Joshua Calafato --- cmake/edge_llm/EdgeLLM.cmake | 27 ++++++++++++++ core/builder/tensorrt_model_connect/build.py | 14 +++++--- core/builder/tests/test_build.py | 37 +++++++++++++++++++- 3 files changed, 73 insertions(+), 5 deletions(-) diff --git a/cmake/edge_llm/EdgeLLM.cmake b/cmake/edge_llm/EdgeLLM.cmake index 3dba21264c..72d8ee2e70 100644 --- a/cmake/edge_llm/EdgeLLM.cmake +++ b/cmake/edge_llm/EdgeLLM.cmake @@ -10,6 +10,26 @@ if(CMAKE_CROSSCOMPILING) message(FATAL_ERROR "Edge-LLM cross compilation is not supported") endif() include("${CMAKE_CURRENT_LIST_DIR}/CheckNative.cmake") + +# Model Connect and the offload must link the same native TensorRT installation. +function(_edgellm_check_trt_selection include_dir library) + foreach(_kind IN ITEMS INCLUDE_DIR LIBRARY) + get_filename_component(_parent "${TRTMC_TRT_${_kind}}" REALPATH) + if(_kind STREQUAL "INCLUDE_DIR") + get_filename_component(_selected "${include_dir}" REALPATH) + else() + get_filename_component(_selected "${library}" REALPATH) + endif() + if(NOT _parent STREQUAL _selected) + message(FATAL_ERROR "Model Connect and EdgeLLM must use the same TensorRT ${_kind}: ${_parent} != ${_selected}. Set TRTMC_TRT_INCLUDE_DIR, TRTMC_TRT_LIBRARY and TRTMC_EDGELLM_TRT_ROOT to one SDK.") + endif() + endforeach() + _edgellm_trt_version("${TRTMC_TRT_INCLUDE_DIR}" _parent_version) + _edgellm_trt_version("${include_dir}" _selected_version) + if(NOT _parent_version STREQUAL _selected_version) + message(FATAL_ERROR "Model Connect and EdgeLLM TensorRT header versions differ") + endif() +endfunction() option(TRTMC_EDGELLM_ALL_KERNELS "Build all pinned Edge operator groups supported by the native GPU" OFF) option(TRTMC_EDGELLM_ONNX "Install the pinned ONNX exporter and native engine builder" OFF) set(_edge_cute_groups "fmha|gdn") @@ -57,6 +77,7 @@ function(_edgellm_install_plugin) FILES \"$\")" COMPONENT EdgeLLM) endfunction() if(EdgeLLM_FOUND AND NOT EdgeLLM_PREFIX STREQUAL _edge_prefix) + _edgellm_check_trt_selection("${EdgeLLM_TRT_INCLUDE_DIR}" "${EdgeLLM_TRT_LIBRARY}") if(NOT EdgeLLM_CUDA_ARCHITECTURE STREQUAL TRTMC_EDGELLM_CUDA_ARCHITECTURE) message(FATAL_ERROR "EdgeLLM package architecture ${EdgeLLM_CUDA_ARCHITECTURE} differs from requested ${TRTMC_EDGELLM_CUDA_ARCHITECTURE}") endif() @@ -87,6 +108,12 @@ set(TRTMC_EDGELLM_GIT_MIRROR "" CACHE PATH "Optional local mirror of the pinned if(NOT EXISTS "${TRTMC_EDGELLM_TRT_ROOT}/include/NvInfer.h") message(FATAL_ERROR "TRTMC_EDGELLM_TRT_ROOT must contain the native TensorRT SDK") endif() +unset(_edge_selected_trt_library CACHE) +unset(_edge_selected_trt_library) +find_library(_edge_selected_trt_library nvinfer + PATHS "${TRTMC_EDGELLM_TRT_ROOT}/lib" "${TRTMC_EDGELLM_TRT_ROOT}/lib64" + NO_DEFAULT_PATH REQUIRED) +_edgellm_check_trt_selection("${TRTMC_EDGELLM_TRT_ROOT}/include" "${_edge_selected_trt_library}") _edgellm_check_gpu("${TRTMC_EDGELLM_CUDA_ARCHITECTURE}") _edgellm_trt_version("${TRTMC_EDGELLM_TRT_ROOT}/include" _edge_trt_version) set(_edge_source "${_edge_root}/source") diff --git a/core/builder/tensorrt_model_connect/build.py b/core/builder/tensorrt_model_connect/build.py index 896b191867..0a684bd8c8 100644 --- a/core/builder/tensorrt_model_connect/build.py +++ b/core/builder/tensorrt_model_connect/build.py @@ -10,6 +10,7 @@ import os import platform import re +import shlex import sys import shutil import subprocess @@ -103,17 +104,22 @@ def cmake_prefixes() -> list[Path]: def _cuda_toolkit_version() -> str: """Identify the selected native compiler, not cuda-python's build toolkit.""" compiler = os.environ.get("CUDACXX") + command = shlex.split(compiler) if compiler else [] if not compiler: root = next( (os.environ[key] for key in ("CUDAToolkit_ROOT", "CUDA_HOME", "CUDA_PATH") if os.environ.get(key)), None ) compiler = str(Path(root) / "bin" / "nvcc") if root else shutil.which("nvcc") - if not compiler: + command = [compiler] if compiler else [] + if not command: raise RuntimeError("CUDA toolkit not found; set CUDAToolkit_ROOT or CUDACXX") - result = subprocess.run( - [compiler, "--version"], check=True, capture_output=True, text=True, - ) + try: + result = subprocess.run( + [*command, "--version"], check=True, capture_output=True, text=True, timeout=10, + ) + except (OSError, subprocess.SubprocessError) as error: + raise RuntimeError(f"Cannot query CUDA toolkit from {compiler}: {error}") from error version = re.search(r"release\s+(\d+\.\d+)", result.stdout) if version is None: raise RuntimeError(f"Cannot identify CUDA toolkit from {compiler} --version") diff --git a/core/builder/tests/test_build.py b/core/builder/tests/test_build.py index 35e87751fd..2655b87184 100644 --- a/core/builder/tests/test_build.py +++ b/core/builder/tests/test_build.py @@ -389,7 +389,9 @@ def test_cuda_toolkit_version_uses_selected_compiler(monkeypatch, source): run = Mock(return_value=SimpleNamespace(stdout="Cuda compilation tools, release 13.3, V13.3.1")) monkeypatch.setattr(build_core.subprocess, "run", run) assert build_core._cuda_toolkit_version() == "13.3" - run.assert_called_once_with([compiler, "--version"], check=True, capture_output=True, text=True) + run.assert_called_once_with( + [compiler, "--version"], check=True, capture_output=True, text=True, timeout=10, + ) def test_cuda_toolkit_version_does_not_guess_when_missing(monkeypatch): @@ -405,3 +407,36 @@ def test_cuda_toolkit_version_rejects_unrecognized_output(monkeypatch): monkeypatch.setattr(build_core.subprocess, "run", lambda *_, **__: SimpleNamespace(stdout="")) with pytest.raises(RuntimeError, match="Cannot identify CUDA toolkit"): build_core._cuda_toolkit_version() + +@pytest.mark.parametrize("source, value, expected", [ + ("CUDACXX", '"/tool kit/nvcc" --allow-unsupported-compiler', + ["/tool kit/nvcc", "--allow-unsupported-compiler"]), + ("CUDAToolkit_ROOT", "/tool kit", ["/tool kit/bin/nvcc"]), +]) +def test_cuda_toolkit_compiler_arguments_and_spaces(monkeypatch, source, value, expected): + from unittest.mock import Mock + + for name in ("CUDACXX", "CUDAToolkit_ROOT", "CUDA_HOME", "CUDA_PATH"): + monkeypatch.delenv(name, raising=False) + monkeypatch.setenv(source, value) + run = Mock(return_value=SimpleNamespace(stdout="release 13.3, V13.3.1")) + monkeypatch.setattr(build_core.subprocess, "run", run) + assert build_core._cuda_toolkit_version() == "13.3" + run.assert_called_once_with( + [*expected, "--version"], check=True, capture_output=True, text=True, timeout=10, + ) + + +@pytest.mark.parametrize("failure", [ + FileNotFoundError("compiler missing"), + build_core.subprocess.CalledProcessError(1, ["nvcc", "--version"]), + build_core.subprocess.TimeoutExpired(["nvcc", "--version"], 10), +]) +def test_cuda_toolkit_compiler_failures_preserve_cause(monkeypatch, failure): + from unittest.mock import Mock + + monkeypatch.setenv("CUDACXX", "/selected/nvcc") + monkeypatch.setattr(build_core.subprocess, "run", Mock(side_effect=failure)) + with pytest.raises(RuntimeError, match="Cannot query CUDA toolkit") as caught: + build_core._cuda_toolkit_version() + assert caught.value.__cause__ is failure From 502e684b7cf18292f8d96b48554432315bb872ec Mon Sep 17 00:00:00 2001 From: Joshua Calafato Date: Wed, 23 Sep 2026 17:12:36 +0000 Subject: [PATCH 13/20] fix(build): reject incomplete external Edge SDKs Signed-off-by: Joshua Calafato --- cmake/edge_llm/EdgeLLM.cmake | 13 +++++++++++++ 1 file changed, 13 insertions(+) diff --git a/cmake/edge_llm/EdgeLLM.cmake b/cmake/edge_llm/EdgeLLM.cmake index 72d8ee2e70..26e3447c41 100644 --- a/cmake/edge_llm/EdgeLLM.cmake +++ b/cmake/edge_llm/EdgeLLM.cmake @@ -76,7 +76,20 @@ function(_edgellm_install_plugin) TYPE SHARED_LIBRARY FOLLOW_SYMLINK_CHAIN FILES \"$\")" COMPONENT EdgeLLM) endfunction() +function(_edgellm_check_external_artifacts) + foreach(_target IN ITEMS EdgeLLM::Core EdgeLLM::Plugin) + get_target_property(_artifact ${_target} IMPORTED_LOCATION) + if(NOT EXISTS "${_artifact}" OR IS_DIRECTORY "${_artifact}") + message(FATAL_ERROR "External EdgeLLM package is incomplete: ${_target} missing at ${_artifact}") + endif() + endforeach() + if(NOT EXISTS "${EdgeLLM_PREFIX}/lib/libcutedsl.a" OR IS_DIRECTORY "${EdgeLLM_PREFIX}/lib/libcutedsl.a") + message(FATAL_ERROR "External EdgeLLM package is incomplete: missing libcutedsl.a") + endif() +endfunction() + if(EdgeLLM_FOUND AND NOT EdgeLLM_PREFIX STREQUAL _edge_prefix) + _edgellm_check_external_artifacts() _edgellm_check_trt_selection("${EdgeLLM_TRT_INCLUDE_DIR}" "${EdgeLLM_TRT_LIBRARY}") if(NOT EdgeLLM_CUDA_ARCHITECTURE STREQUAL TRTMC_EDGELLM_CUDA_ARCHITECTURE) message(FATAL_ERROR "EdgeLLM package architecture ${EdgeLLM_CUDA_ARCHITECTURE} differs from requested ${TRTMC_EDGELLM_CUDA_ARCHITECTURE}") From c1b73721db6be40de97db6683004f7f6a6df00ad Mon Sep 17 00:00:00 2001 From: Joshua Calafato Date: Wed, 23 Sep 2026 17:29:01 +0000 Subject: [PATCH 14/20] fix(build): harden pinned runtime SDK installation Verify the actual checkout before preparing dependencies. Install runtime Python interpreters and modules without build-tree console entrypoints; retain tools for reprovisioning. Normalize malformed compiler commands to the documented error contract. Signed-off-by: Joshua Calafato --- cmake/edge_llm/EdgeLLM.cmake | 9 ++++++++- cmake/edge_llm/Prepare.cmake.in | 9 +++++++++ cmake/edge_llm/README.md | 10 ++++++++++ core/builder/tensorrt_model_connect/build.py | 5 ++++- core/builder/tests/test_build.py | 8 ++++++++ 5 files changed, 39 insertions(+), 2 deletions(-) diff --git a/cmake/edge_llm/EdgeLLM.cmake b/cmake/edge_llm/EdgeLLM.cmake index 26e3447c41..6437bac427 100644 --- a/cmake/edge_llm/EdgeLLM.cmake +++ b/cmake/edge_llm/EdgeLLM.cmake @@ -182,6 +182,13 @@ find_package(EdgeLLM ${_edge_version} EXACT CONFIG REQUIRED PATHS "${_edge_prefix}/lib/cmake/EdgeLLM" NO_DEFAULT_PATH) add_dependencies(EdgeLLM::Core trtmc_edgellm_dependency) add_dependencies(EdgeLLM::Plugin trtmc_edgellm_dependency) -install(DIRECTORY "${_edge_prefix}/" DESTINATION . USE_SOURCE_PERMISSIONS COMPONENT EdgeLLM) +# Runtime consumers use the interpreter/modules and the prefix-relative launcher. +# Build-only console scripts/activation files embed build-tree paths; keep them +# available for rebuilding the dependency, but do not publish them in the SDK. +install(DIRECTORY "${_edge_prefix}/" DESTINATION . USE_SOURCE_PERMISSIONS COMPONENT EdgeLLM + PATTERN "libexec/trtmc-edge-llm/bin" EXCLUDE) +install(DIRECTORY "${_edge_prefix}/libexec/trtmc-edge-llm/bin/" + DESTINATION libexec/trtmc-edge-llm/bin USE_SOURCE_PERMISSIONS COMPONENT EdgeLLM + FILES_MATCHING REGEX "/python([0-9]+(\\.[0-9]+)?)?$") # Family DSOs may use lib64; their dynamically loaded plugin must remain adjacent. _edgellm_install_plugin() diff --git a/cmake/edge_llm/Prepare.cmake.in b/cmake/edge_llm/Prepare.cmake.in index f2ba6f09f5..b161f65953 100644 --- a/cmake/edge_llm/Prepare.cmake.in +++ b/cmake/edge_llm/Prepare.cmake.in @@ -2,6 +2,15 @@ # SPDX-License-Identifier: Apache-2.0 # Executed only by the explicit CMake dependency build, never by model dispatch. cmake_minimum_required(VERSION 3.20) +# Older CMake can skip a disconnected Git update when the requested pin changes. +# Never label an old checkout as the newly requested official snapshot. +find_package(Git REQUIRED) +execute_process(COMMAND "${GIT_EXECUTABLE}" -C "@_edge_source@" rev-parse HEAD + OUTPUT_VARIABLE _edge_checkout_revision OUTPUT_STRIP_TRAILING_WHITESPACE + COMMAND_ERROR_IS_FATAL ANY) +if(NOT _edge_checkout_revision STREQUAL "@_edge_revision@") + message(FATAL_ERROR "EdgeLLM checkout does not match the pinned revision; use a fresh dependency build directory") +endif() include("@_edge_template_dir@/CheckNative.cmake") _edgellm_check_json_headers("@_edge_json_include@" "@_edge_source@/3rdParty/nlohmannJson") function(run) diff --git a/cmake/edge_llm/README.md b/cmake/edge_llm/README.md index c4268d32a5..1490069b0b 100644 --- a/cmake/edge_llm/README.md +++ b/cmake/edge_llm/README.md @@ -74,3 +74,13 @@ Run the existing runtime and family checks against this installation ```bash ctest --test-dir build --output-on-failure ``` + +The installed private Python environment exposes its interpreter and modules, +not build-only console/activation scripts containing build-tree paths. Invoke +the exported interpreter with isolated module execution, or use the installed +prefix-relative builder launcher. CMake/Ninja entrypoints remain in the +dependency build environment for reprovisioning. Install into a clean prefix +when replacing an older SDK that included these private console scripts. +Preparation verifies the actual Git checkout against the official pin before +installing dependencies, including on CMake versions with older disconnected +update behavior. diff --git a/core/builder/tensorrt_model_connect/build.py b/core/builder/tensorrt_model_connect/build.py index 0a684bd8c8..5724df53e5 100644 --- a/core/builder/tensorrt_model_connect/build.py +++ b/core/builder/tensorrt_model_connect/build.py @@ -104,7 +104,10 @@ def cmake_prefixes() -> list[Path]: def _cuda_toolkit_version() -> str: """Identify the selected native compiler, not cuda-python's build toolkit.""" compiler = os.environ.get("CUDACXX") - command = shlex.split(compiler) if compiler else [] + try: + command = shlex.split(compiler) if compiler else [] + except ValueError as error: + raise RuntimeError(f"Invalid CUDACXX command: {error}") from error if not compiler: root = next( (os.environ[key] for key in ("CUDAToolkit_ROOT", "CUDA_HOME", "CUDA_PATH") diff --git a/core/builder/tests/test_build.py b/core/builder/tests/test_build.py index 2655b87184..ff0b0c1a0d 100644 --- a/core/builder/tests/test_build.py +++ b/core/builder/tests/test_build.py @@ -440,3 +440,11 @@ def test_cuda_toolkit_compiler_failures_preserve_cause(monkeypatch, failure): with pytest.raises(RuntimeError, match="Cannot query CUDA toolkit") as caught: build_core._cuda_toolkit_version() assert caught.value.__cause__ is failure + + +def test_cuda_toolkit_malformed_compiler_command_preserves_cause(monkeypatch): + monkeypatch.setenv("CUDACXX", '"unclosed compiler path') + monkeypatch.setattr(build_core.subprocess, "run", lambda *_a, **_k: pytest.fail("compiler ran")) + with pytest.raises(RuntimeError, match="Invalid CUDACXX command") as caught: + build_core._cuda_toolkit_version() + assert isinstance(caught.value.__cause__, ValueError) From 1bdfe17cd923bf5c8e3d13cb6e55e50bf0ea051c Mon Sep 17 00:00:00 2001 From: Joshua Calafato Date: Wed, 23 Sep 2026 18:02:47 +0000 Subject: [PATCH 15/20] fix(build): align CUDA dependency requirements Match the pinned upstream CUDA 12 and CUDA 13 CuPy versions, including the compatible NumPy constraint. Reject external packages missing their Python interpreter or builder launcher before model dispatch. Signed-off-by: Joshua Calafato --- cmake/edge_llm/EdgeLLM.cmake | 5 +++++ cmake/edge_llm/Prepare.cmake.in | 15 +++++++++++++-- 2 files changed, 18 insertions(+), 2 deletions(-) diff --git a/cmake/edge_llm/EdgeLLM.cmake b/cmake/edge_llm/EdgeLLM.cmake index 6437bac427..a446cd000d 100644 --- a/cmake/edge_llm/EdgeLLM.cmake +++ b/cmake/edge_llm/EdgeLLM.cmake @@ -77,6 +77,11 @@ function(_edgellm_install_plugin) FILES \"$\")" COMPONENT EdgeLLM) endfunction() function(_edgellm_check_external_artifacts) + foreach(_tool IN ITEMS EdgeLLM_PYTHON_EXECUTABLE EdgeLLM_BUILDER_LAUNCHER) + if(NOT EXISTS "${${_tool}}" OR IS_DIRECTORY "${${_tool}}") + message(FATAL_ERROR "External EdgeLLM package is incomplete: ${_tool} missing at ${${_tool}}") + endif() + endforeach() foreach(_target IN ITEMS EdgeLLM::Core EdgeLLM::Plugin) get_target_property(_artifact ${_target} IMPORTED_LOCATION) if(NOT EXISTS "${_artifact}" OR IS_DIRECTORY "${_artifact}") diff --git a/cmake/edge_llm/Prepare.cmake.in b/cmake/edge_llm/Prepare.cmake.in index b161f65953..e0b1b38c31 100644 --- a/cmake/edge_llm/Prepare.cmake.in +++ b/cmake/edge_llm/Prepare.cmake.in @@ -44,11 +44,22 @@ list(LENGTH _trt_wheels _wheel_count) if(NOT _wheel_count EQUAL 1) message(FATAL_ERROR "Expected exactly one TensorRT SDK wheel matching the native Python ABI") endif() +# Match the exact CuPy versions required by the pinned upstream CuTe builder. +# CuPy 12.3 also requires NumPy below 1.29. +if("@CUDAToolkit_VERSION_MAJOR@" STREQUAL "12") + set(_edge_cupy_version 12.3.0) + set(_edge_numpy_version 1.26.4) +elseif("@CUDAToolkit_VERSION_MAJOR@" STREQUAL "13") + set(_edge_cupy_version 13.6.0) + set(_edge_numpy_version 2.2.6) +else() + message(FATAL_ERROR "Pinned EdgeLLM supports CUDA major 12 or 13") +endif() run("@_edge_python@" -I -m pip ${_pip_options} --report "@_edge_prefix@/pip-report.json" - ${_trt_wheels} numpy==2.2.6 transformers==5.14.1 jinja2==3.1.6 + ${_trt_wheels} "numpy==${_edge_numpy_version}" transformers==5.14.1 jinja2==3.1.6 scikit-build-core==0.11.6 wheel==0.45.1 cmake==3.31.10 ninja==1.13.0 "cuda-python>=@CUDAToolkit_VERSION_MAJOR@.@CUDAToolkit_VERSION_MINOR@,<@CUDAToolkit_VERSION_MAJOR@.@CUDAToolkit_VERSION_MINOR@.999" - "nvidia-cutlass-dsl[cu@CUDAToolkit_VERSION_MAJOR@]==4.7.0" "cupy-cuda@CUDAToolkit_VERSION_MAJOR@x==13.6.0") + "nvidia-cutlass-dsl[cu@CUDAToolkit_VERSION_MAJOR@]==4.7.0" "cupy-cuda@CUDAToolkit_VERSION_MAJOR@x==${_edge_cupy_version}") run("@_edge_python@" -I -m pip ${_pip_options} --no-deps --no-build-isolation "@_edge_source@") if("@TRTMC_EDGELLM_ONNX@") # Original exporter dependencies, isolated from the caller environment. Export From 55c0ac70361467484fa7c48d2a3de7e4010bb989 Mon Sep 17 00:00:00 2001 From: Joshua Calafato Date: Wed, 23 Sep 2026 18:17:30 +0000 Subject: [PATCH 16/20] refactor(cli): remove redundant build argument hook Use the already merged family CLI declaration protocol for owner options. Restore shared build parsing and support contracts unchanged from main; retain unrelated SDK provisioning and native discovery mechanics. Signed-off-by: Joshua Calafato --- .../tensorrt_model_connect/build_cli.py | 108 ++++---------- .../tensorrt_model_connect/model_support.py | 8 -- core/builder/tests/test_build_cli.py | 133 ------------------ core/builder/tests/test_model_support.py | 14 -- website/docs/api/python-builder.md | 20 --- website/docs/architecture/build-pipeline.md | 6 - 6 files changed, 29 insertions(+), 260 deletions(-) diff --git a/core/builder/tensorrt_model_connect/build_cli.py b/core/builder/tensorrt_model_connect/build_cli.py index 2d2465aec9..837f5e4c47 100644 --- a/core/builder/tensorrt_model_connect/build_cli.py +++ b/core/builder/tensorrt_model_connect/build_cli.py @@ -6,7 +6,6 @@ from __future__ import annotations import argparse -import importlib import json import shlex import sys @@ -23,16 +22,12 @@ ) -def _parser( - prepare_family: object | None = None, *, build_hooks: object | None = None, - require_output: bool = True, require_model: bool = True, -) -> argparse.ArgumentParser: +def _parser(prepare_family: object | None = None) -> argparse.ArgumentParser: parser = argparse.ArgumentParser(prog="trtmc") commands = parser.add_subparsers(dest="command", required=True) - build_parser = commands.add_parser("build", help="Build one TensorRT bundle", allow_abbrev=False) - if require_model: - build_parser.add_argument("model", help="Hugging Face model ID or local snapshot") - build_parser.add_argument("-o", "--output", type=Path, required=require_output) + build_parser = commands.add_parser("build", help="Build one TensorRT bundle") + build_parser.add_argument("model", help="Hugging Face model ID or local snapshot") + build_parser.add_argument("-o", "--output", type=Path, required=True) build_parser.add_argument( "--family", help="Select one compatible family instead of automatic resolution" ) @@ -51,9 +46,6 @@ def _parser( build_parser.add_argument("--fp32-layer", type=int, action="append", default=[]) build_parser.add_argument("--dynamic-kv-cache", action="store_true") build_parser.add_argument("--verbose", action="store_true") - add_build_arguments = getattr(build_hooks, "add_build_arguments", None) - if callable(add_build_arguments): - add_build_arguments(build_parser) prepare_parser = commands.add_parser( "prepare-structure", help="Prepare one structure request without rebuilding its model bundle", @@ -74,11 +66,7 @@ def _parser( def main(argv: Sequence[str] | None = None) -> int: arguments = list(sys.argv[1:] if argv is None else argv) - family_help = ( - len(arguments) > 1 and arguments[0] == "build" - and any(arg in {"-h", "--help"} for arg in arguments) - ) - base_parser = _parser(require_output=not family_help) + base_parser = _parser() if ( len(arguments) > 1 and arguments[0] == "prepare-structure" @@ -86,39 +74,9 @@ def main(argv: Sequence[str] | None = None) -> int: and arguments[1].startswith("-") ): base_parser.error("MODEL must immediately follow prepare-structure") - preliminary_arguments = ( - [arg for arg in arguments if arg not in {"-h", "--help"}] if family_help else arguments - ) - if arguments and arguments[0] == "build": - # Without a positional, parse_known_args preserves MODEL and family - # options in order. Never acquire a checkpoint from an unknown option's - # value; known core options may safely precede MODEL. - _, remaining = _parser(require_output=False, require_model=False).parse_known_args( - preliminary_arguments - ) - if remaining and remaining[0].startswith("-") and remaining[0] != "--": - base_parser.error("MODEL must precede family options") - if family_help and not remaining: - base_parser.parse_args(arguments) - return 0 - preliminary, _ = base_parser.parse_known_args(preliminary_arguments) - if family_help and not Path(preliminary.model).is_dir(): - # Remote or missing inputs cannot provide local metadata. Help must - # remain side-effect free instead of acquiring a checkpoint. - base_parser.parse_args(arguments) - return 0 - model_dir = ( - Path(preliminary.model) if family_help - else _resolve_model(preliminary.model, preliminary.revision) - ) - try: - metadata = load_model_metadata(model_dir) - except (ValueError, OSError): - if not family_help: - raise - # Empty/invalid local directories still have useful generic help. - base_parser.parse_args(arguments) - return 0 + preliminary, _ = base_parser.parse_known_args(arguments) + model_dir = _resolve_model(preliminary.model, preliminary.revision) + metadata = load_model_metadata(model_dir) try: family, support = ( resolve_family(metadata, preliminary.family) @@ -129,11 +87,7 @@ def main(argv: Sequence[str] | None = None) -> int: _print_family_error(error, arguments) return 2 family_module = _load_family(family) if preliminary.command == "prepare-structure" else None - build_hooks = ( - importlib.import_module(f"families.{family}.{support.build_cli_module}") - if preliminary.command == "build" and support.build_cli_module is not None else None - ) - args = _parser(family_module, build_hooks=build_hooks).parse_args(arguments) + args = _parser(family_module).parse_args(arguments) if args.command == "prepare-structure": prepare = getattr(family_module, "prepare_structure_request", None) if not callable(prepare): @@ -157,31 +111,27 @@ def main(argv: Sequence[str] | None = None) -> int: f"family {family!r} does not support task {task!r}; " f"choose one of: {', '.join(support.tasks)}" ) - request = BuildRequest( - model_dir=model_dir, - output_path=args.output, - precision=args.precision or support.default_precision, - backend=args.backend, - family=family, - task=task, - max_sequence_length=args.max_sequence_length, - image_height=args.image_height, - image_width=args.image_width, - video_num_frames=args.video_num_frames, - max_batch_size=args.max_batch_size, - tensor_parallel_size=args.tensor_parallel_size, - context_parallel_size=args.context_parallel_size, - quantization=args.quantization, - fp32_layers=tuple(args.fp32_layer), - dynamic_kv_cache=args.dynamic_kv_cache, - verbose=args.verbose, + build( + BuildRequest( + model_dir=model_dir, + output_path=args.output, + precision=args.precision or support.default_precision, + backend=args.backend, + family=family, + task=task, + max_sequence_length=args.max_sequence_length, + image_height=args.image_height, + image_width=args.image_width, + video_num_frames=args.video_num_frames, + max_batch_size=args.max_batch_size, + tensor_parallel_size=args.tensor_parallel_size, + context_parallel_size=args.context_parallel_size, + quantization=args.quantization, + fp32_layers=tuple(args.fp32_layer), + dynamic_kv_cache=args.dynamic_kv_cache, + verbose=args.verbose, + ) ) - prepare_build_request = getattr(build_hooks, "prepare_build_request", None) - if callable(prepare_build_request): - request = prepare_build_request(request, args) - if not isinstance(request, BuildRequest) or request.family != family: - raise TypeError("family prepare_build_request must preserve the owning BuildRequest") - build(request) return 0 diff --git a/core/builder/tensorrt_model_connect/model_support.py b/core/builder/tensorrt_model_connect/model_support.py index 39262801da..04fd18c307 100644 --- a/core/builder/tensorrt_model_connect/model_support.py +++ b/core/builder/tensorrt_model_connect/model_support.py @@ -56,14 +56,8 @@ class FamilySupport: tasks: tuple[str, ...] default_task: str default_precision: str = "fp32" - build_cli_module: str | None = None def __post_init__(self) -> None: - if self.build_cli_module is not None and ( - not isinstance(self.build_cli_module, str) - or any(_ID.fullmatch(part) is None for part in self.build_cli_module.split(".")) - ): - raise ValueError("build_cli_module must name a module within the owning family") if not self.tasks or len(set(self.tasks)) != len(self.tasks): raise ValueError("family support tasks must be non-empty and unique") if any(_ID.fullmatch(task) is None for task in self.tasks): @@ -103,7 +97,6 @@ def family_support( tasks: tuple[str, ...], default_task: str, default_precision: str = "fp32", - build_cli_module: str | None = None, ) -> DescribeSupport: """Create one exact, family-owned support function.""" @@ -117,7 +110,6 @@ def family_support( tasks=tasks, default_task=default_task, default_precision=default_precision, - build_cli_module=build_cli_module, ) def describe(metadata: ModelMetadata) -> FamilySupport | None: diff --git a/core/builder/tests/test_build_cli.py b/core/builder/tests/test_build_cli.py index a5db0b1959..8086a1524b 100644 --- a/core/builder/tests/test_build_cli.py +++ b/core/builder/tests/test_build_cli.py @@ -541,136 +541,3 @@ def test_prepare_structure_requires_a_callable_family_hook( "-o", str(output), ]) assert not output.exists() - - -@pytest.mark.parametrize("explicit_family", [False, True]) -@pytest.mark.parametrize("core_prefix", [[], ["--verbose"], ["--family", "example_owner"]]) -@pytest.mark.parametrize("option", ["--example-setting", "--image"]) -def test_build_family_hooks_forward_a_typed_request( - monkeypatch, tmp_path, explicit_family, core_prefix, option -): - from dataclasses import dataclass - - @dataclass(frozen=True) - class ExampleRequest(build_cli.BuildRequest): - example_setting: str = "" - - def add_arguments(parser): - parser.add_argument(option, dest="example_setting", required=True) - - def prepare(request, args): - return ExampleRequest(**vars(request), example_setting=args.example_setting) - - support = FamilySupport( - ("example_task",), "example_task", - build_cli_module="example_cli", - ) - monkeypatch.setitem(sys.modules, "families.example_owner.example_cli", SimpleNamespace( - add_build_arguments=add_arguments, prepare_build_request=prepare, - )) - (tmp_path / "config.json").write_text('{"model_type":"example_model"}') - monkeypatch.setattr(build_cli, "resolve_family", lambda metadata, *args: ("example_owner", support)) - monkeypatch.setattr(build_cli, "_load_family", lambda *_: pytest.fail("GPU builder imported by CLI")) - seen = [] - monkeypatch.setattr(build_cli, "build", seen.append) - arguments = ["build", *core_prefix, str(tmp_path), "-o", str(tmp_path / "out"), option, "verbatim"] - if explicit_family: - arguments += ["--family", "example_owner"] - assert build_cli.main(arguments) == 0 - assert len(seen) == 1 and isinstance(seen[0], ExampleRequest) - assert seen[0].example_setting == "verbatim" - assert seen[0].family == "example_owner" and seen[0].model_dir == tmp_path - - -def test_build_family_options_are_not_global(monkeypatch, tmp_path): - (tmp_path / "config.json").write_text('{"model_type":"example_model"}') - support = FamilySupport(("example_task",), "example_task") - monkeypatch.setattr(build_cli, "resolve_family", lambda _: ("example_owner", support)) - monkeypatch.setattr(build_cli, "build", lambda *_: pytest.fail("unknown option reached build")) - with pytest.raises(SystemExit) as error: - build_cli.main(["build", str(tmp_path), "-o", str(tmp_path / "out"), - "--example-setting", "not-owned"]) - assert error.value.code == 2 - - -def test_build_family_help_does_not_require_output_or_load_builder(monkeypatch, tmp_path, capsys): - monkeypatch.setattr(build_cli, "_resolve_model", lambda *_: pytest.fail("help acquired a model")) - (tmp_path / "config.json").write_text('{"model_type":"example_model"}') - support = FamilySupport( - ("example_task",), "example_task", - build_cli_module="example_cli", - ) - monkeypatch.setitem(sys.modules, "families.example_owner.example_cli", SimpleNamespace( - add_build_arguments=lambda parser: parser.add_argument("--example-setting"), - prepare_build_request=lambda *_: pytest.fail("help prepared a build"), - )) - monkeypatch.setattr(build_cli, "resolve_family", lambda _: ("example_owner", support)) - monkeypatch.setattr(build_cli, "_load_family", lambda *_: pytest.fail("help imported GPU builder")) - with pytest.raises(SystemExit) as error: - build_cli.main(["build", str(tmp_path), "--help"]) - assert error.value.code == 0 - assert "--example-setting" in capsys.readouterr().out - - -@pytest.mark.parametrize("wrong_owner", [False, True]) -def test_build_family_hook_cannot_replace_owner_or_contract(monkeypatch, tmp_path, wrong_owner): - from dataclasses import replace - - (tmp_path / "config.json").write_text('{"model_type":"example_model"}') - support = FamilySupport( - ("example_task",), "example_task", - build_cli_module="example_cli", - ) - monkeypatch.setitem(sys.modules, "families.example_owner.example_cli", SimpleNamespace( - prepare_build_request=lambda request, _: ( - replace(request, family="another_owner") if wrong_owner else object() - ), - )) - monkeypatch.setattr(build_cli, "resolve_family", lambda _: ("example_owner", support)) - monkeypatch.setattr(build_cli, "build", lambda *_: pytest.fail("invalid request reached build")) - with pytest.raises(TypeError, match="preserve the owning BuildRequest"): - build_cli.main(["build", str(tmp_path), "-o", str(tmp_path / "out")]) - - -@pytest.mark.parametrize("help_option", ["-h", "--help"]) -@pytest.mark.parametrize("explicit_family", [False, True]) -def test_remote_model_help_never_acquires_checkpoint(monkeypatch, capsys, help_option, explicit_family): - monkeypatch.setattr(build_cli, "_resolve_model", lambda *_: pytest.fail("help downloaded model")) - monkeypatch.setattr(build_cli, "resolve_family", lambda *_: pytest.fail("remote help resolved owner")) - monkeypatch.setattr(build_cli, "build", lambda *_: pytest.fail("help started build")) - arguments = ["build", "example-organization/uncached-model", help_option] - if explicit_family: - arguments += ["--family", "example_owner"] - with pytest.raises(SystemExit) as error: - build_cli.main(arguments) - assert error.value.code == 0 - output = capsys.readouterr() - assert "--precision" in output.out - assert not output.err - -@pytest.mark.parametrize("metadata", [None, "", "{broken", "[]"]) -@pytest.mark.parametrize("prefix", [[], ["--family", "example_owner"]]) -def test_local_invalid_metadata_help_is_generic(monkeypatch, tmp_path, capsys, metadata, prefix): - if metadata is not None: - (tmp_path / "config.json").write_text(metadata) - monkeypatch.setattr(build_cli, "_resolve_model", lambda *_: pytest.fail("help acquired model")) - monkeypatch.setattr(build_cli, "build", lambda *_: pytest.fail("help built model")) - with pytest.raises(SystemExit) as error: - build_cli.main(["build", *prefix, str(tmp_path), "--help"]) - assert error.value.code == 0 - assert "--precision" in capsys.readouterr().out - - -@pytest.mark.parametrize("prefix", [[], ["--family", "example_owner"], ["--verbose"]]) -def test_unknown_build_option_before_model_never_acquires(monkeypatch, tmp_path, prefix): - monkeypatch.setattr(build_cli, "_resolve_model", lambda *_: pytest.fail("download before parsing")) - with pytest.raises(SystemExit) as error: - build_cli.main(["build", *prefix, "--example-setting", "not-a-model", - str(tmp_path), "-o", str(tmp_path / "out")]) - assert error.value.code == 2 - - -def test_non_help_invalid_metadata_is_not_hidden(tmp_path): - (tmp_path / "config.json").write_text("{broken") - with pytest.raises(ValueError): - build_cli.main(["build", str(tmp_path), "-o", str(tmp_path / "out")]) diff --git a/core/builder/tests/test_model_support.py b/core/builder/tests/test_model_support.py index a3826c0c9e..0144dbdc7d 100644 --- a/core/builder/tests/test_model_support.py +++ b/core/builder/tests/test_model_support.py @@ -310,17 +310,3 @@ def test_family_owned_exact_metadata_shapes_resolve_without_priority( ) -> None: family, _ = resolve_family(metadata) assert family == expected_family - -@pytest.mark.parametrize("module", ["", ".cli", "../other", "cli.", "a..b", "module-name", 12]) -def test_build_cli_module_must_be_owned_module_path(module): - with pytest.raises(ValueError, match="within the owning family"): - FamilySupport(("text_generation",), "text_generation", build_cli_module=module) - - -def test_family_build_cli_declaration_remains_import_free(): - declaration = family_support( - model_types=("example_model",), tasks=("text_generation",), - default_task="text_generation", build_cli_module="optional.cli", - ) - metadata = ModelMetadata({"model_type": "example_model"}, {}, frozenset()) - assert declaration(metadata).build_cli_module == "optional.cli" diff --git a/website/docs/api/python-builder.md b/website/docs/api/python-builder.md index f073864f50..980b803521 100644 --- a/website/docs/api/python-builder.md +++ b/website/docs/api/python-builder.md @@ -29,26 +29,6 @@ resolved API directly. decides whether that directory is a Hugging Face snapshot or a prepared checkpoint; `BuildRequest` does not perform another discovery pass. -## Family-owned build arguments - -A family may provide `add_build_arguments(parser)` and -`prepare_build_request(request, args)` in a family-local module named by the lightweight -`FamilySupport.build_cli_module` declaration. The CLI resolves the owner, registers only that family's options, -and lets it return a family-owned `BuildRequest` subclass. The hook must retain -the resolved family. Ordinary families need no changes. - -Core always calls the same `build(request)` and family `build(request, writer)` -entrypoints. There is no shared execution-variant list, companion interpretation, -GPU offload selection, or alternative-builder dispatch. The family owns all -extra fields, validation and execution choices. - -Put MODEL before family-specific options. Known core options may precede MODEL. -`trtmc build /path/to/model --help` shows the resolved family's options without -importing its GPU builder. Help never downloads a checkpoint: remote model IDs -or missing local directories show generic build help instead. Model resolution -precedes family-specific argument -validation; request preparation precedes backend import and bundle creation. - ## Optional graph transform `BuildRequest.graph_transform` is an in-place callback invoked on the completed diff --git a/website/docs/architecture/build-pipeline.md b/website/docs/architecture/build-pipeline.md index 5cedd2e161..84ade7448b 100644 --- a/website/docs/architecture/build-pipeline.md +++ b/website/docs/architecture/build-pipeline.md @@ -36,12 +36,6 @@ sizes, family-owned quantization selection, FP32 layer overrides, direct dynamic-KV opt-in, and optional graph transform. Each family must implement or explicitly reject every non-default request it receives. -Optional CLI extensions are declared by the owning family's lightweight -support module. The family registers its arguments and prepares a typed request -before any GPU builder is imported. Core retains one ordinary family build -entrypoint; execution selection, companion validation and runtime composition -remain inside the family. No other family needs to change. - ## Family build `families//model.py` exposes a plain `build(request, writer)` function. From 69b2d520b10c1178cf6e7f9c6fc31bb4438ab762 Mon Sep 17 00:00:00 2001 From: Joshua Calafato Date: Wed, 23 Sep 2026 21:14:09 +0000 Subject: [PATCH 17/20] fix(edge-llm): isolate CUDA 12 kernel dependencies Signed-off-by: Joshua Calafato --- cmake/edge_llm/EdgeLLM.cmake | 3 ++ cmake/edge_llm/Prepare.cmake.in | 59 +++++++++++++++++---------------- cmake/edge_llm/README.md | 9 +++++ 3 files changed, 42 insertions(+), 29 deletions(-) diff --git a/cmake/edge_llm/EdgeLLM.cmake b/cmake/edge_llm/EdgeLLM.cmake index a446cd000d..a2e907a0de 100644 --- a/cmake/edge_llm/EdgeLLM.cmake +++ b/cmake/edge_llm/EdgeLLM.cmake @@ -118,6 +118,9 @@ endif() include(ExternalProject) include(CMakePackageConfigHelpers) find_package(Python3 3.10 REQUIRED COMPONENTS Interpreter) +if(CUDAToolkit_VERSION_MAJOR EQUAL 12 AND Python3_VERSION VERSION_GREATER_EQUAL "3.13") + message(FATAL_ERROR "Pinned CUDA 12 CuPy kernels require Python 3.10-3.12; select Python3_EXECUTABLE accordingly") +endif() find_package(Threads REQUIRED) set(TRTMC_EDGELLM_TRT_ROOT "$ENV{TRT_ROOT}" CACHE PATH "Native TensorRT SDK, including its Python wheel") set(TRTMC_EDGELLM_JOBS 2 CACHE STRING "Parallel Edge-LLM native and AOT compilation jobs") diff --git a/cmake/edge_llm/Prepare.cmake.in b/cmake/edge_llm/Prepare.cmake.in index e0b1b38c31..9c4dce1614 100644 --- a/cmake/edge_llm/Prepare.cmake.in +++ b/cmake/edge_llm/Prepare.cmake.in @@ -16,50 +16,51 @@ _edgellm_check_json_headers("@_edge_json_include@" "@_edge_source@/3rdParty/nloh function(run) execute_process(COMMAND ${ARGV} COMMAND_ERROR_IS_FATAL ANY) endfunction() -execute_process(COMMAND "@Python3_EXECUTABLE@" -I -c "import ensurepip" - RESULT_VARIABLE _has_ensurepip OUTPUT_QUIET ERROR_QUIET) -if(_has_ensurepip EQUAL 0) - run("@Python3_EXECUTABLE@" -I -m venv --copies "@_edge_prefix@/libexec/trtmc-edge-llm") -else() - # Debian minimal Python may omit ensurepip; use an already installed bootstrapper. - run("@Python3_EXECUTABLE@" -I -m virtualenv --copies --no-download --no-periodic-update - "@_edge_prefix@/libexec/trtmc-edge-llm") -endif() set(_pip_options --isolated install --no-user) if(NOT "@TRTMC_EDGELLM_WHEELHOUSE@" STREQUAL "") list(APPEND _pip_options --no-index --find-links "@TRTMC_EDGELLM_WHEELHOUSE@") endif() -# Older supported Python bootstraps may seed pip before --report was introduced. -execute_process(COMMAND "@_edge_python@" -I -m pip --version - OUTPUT_VARIABLE _pip_version OUTPUT_STRIP_TRAILING_WHITESPACE COMMAND_ERROR_IS_FATAL ANY) -if(NOT _pip_version MATCHES "^pip ([0-9]+\\.[0-9]+(\\.[0-9]+)?)") - message(FATAL_ERROR "Cannot determine the isolated environment pip version") -endif() -if(CMAKE_MATCH_1 VERSION_LESS "22.2") - # _pip_options keeps offline upgrades restricted to the explicit wheelhouse. - run("@_edge_python@" -I -m pip ${_pip_options} --upgrade "pip>=22.2") -endif() +function(create_environment destination) + execute_process(COMMAND "@Python3_EXECUTABLE@" -I -c "import ensurepip" + RESULT_VARIABLE _has_ensurepip OUTPUT_QUIET ERROR_QUIET) + if(_has_ensurepip EQUAL 0) + run("@Python3_EXECUTABLE@" -I -m venv --copies "${destination}") + else() + run("@Python3_EXECUTABLE@" -I -m virtualenv --copies --no-download --no-periodic-update + "${destination}") + endif() + # Pin even when ensurepip seeds a newer version; offline installs stay offline. + run("${destination}/bin/python" -I -m pip ${_pip_options} "pip==24.0") +endfunction() +create_environment("@_edge_prefix@/libexec/trtmc-edge-llm") file(GLOB _trt_wheels "@TRTMC_EDGELLM_TRT_ROOT@/python/tensorrt-*-cp@Python3_VERSION_MAJOR@@Python3_VERSION_MINOR@-none-linux_@CMAKE_SYSTEM_PROCESSOR@.whl") list(LENGTH _trt_wheels _wheel_count) if(NOT _wheel_count EQUAL 1) message(FATAL_ERROR "Expected exactly one TensorRT SDK wheel matching the native Python ABI") endif() -# Match the exact CuPy versions required by the pinned upstream CuTe builder. -# CuPy 12.3 also requires NumPy below 1.29. +# CUDA 12 CuPy needs NumPy 1.x, while the installed Edge SDK needs NumPy 2.x. +# Keep its AOT-only environment outside the installed SDK and check both. +set(_edge_kernel_python "@_edge_python@") if("@CUDAToolkit_VERSION_MAJOR@" STREQUAL "12") - set(_edge_cupy_version 12.3.0) - set(_edge_numpy_version 1.26.4) + set(_edge_cuda_python_version 12.9.7) + set(_edge_kernel_python "@_edge_root@/kernel-python/bin/python") + create_environment("@_edge_root@/kernel-python") + run("${_edge_kernel_python}" -I -m pip ${_pip_options} + --report "@_edge_root@/kernel-pip-report.json" + numpy==1.26.4 cupy-cuda12x==12.3.0 cuda-python==12.9.7 + "nvidia-cutlass-dsl[cu12]==4.7.0") + run("${_edge_kernel_python}" -I -m pip --isolated check) + set(_edge_kernel_dependencies "") elseif("@CUDAToolkit_VERSION_MAJOR@" STREQUAL "13") - set(_edge_cupy_version 13.6.0) - set(_edge_numpy_version 2.2.6) + set(_edge_cuda_python_version 13.3.1) + set(_edge_kernel_dependencies "nvidia-cutlass-dsl[cu13]==4.7.0" cupy-cuda13x==13.6.0) else() message(FATAL_ERROR "Pinned EdgeLLM supports CUDA major 12 or 13") endif() run("@_edge_python@" -I -m pip ${_pip_options} --report "@_edge_prefix@/pip-report.json" - ${_trt_wheels} "numpy==${_edge_numpy_version}" transformers==5.14.1 jinja2==3.1.6 + ${_trt_wheels} numpy==2.2.6 transformers==5.14.1 jinja2==3.1.6 scikit-build-core==0.11.6 wheel==0.45.1 cmake==3.31.10 ninja==1.13.0 - "cuda-python>=@CUDAToolkit_VERSION_MAJOR@.@CUDAToolkit_VERSION_MINOR@,<@CUDAToolkit_VERSION_MAJOR@.@CUDAToolkit_VERSION_MINOR@.999" - "nvidia-cutlass-dsl[cu@CUDAToolkit_VERSION_MAJOR@]==4.7.0" "cupy-cuda@CUDAToolkit_VERSION_MAJOR@x==${_edge_cupy_version}") + "cuda-python==${_edge_cuda_python_version}" ${_edge_kernel_dependencies}) run("@_edge_python@" -I -m pip ${_pip_options} --no-deps --no-build-isolation "@_edge_source@") if("@TRTMC_EDGELLM_ONNX@") # Original exporter dependencies, isolated from the caller environment. Export @@ -74,6 +75,6 @@ endif() run("@_edge_python@" -I -m pip --isolated check) run("@_edge_python@" -I -c "print(__import__('tensorrt').__version__)") set(ENV{PATH} "@CUDAToolkit_BIN_DIR@:$ENV{PATH}") -run("@_edge_python@" -I "@_edge_source@/kernelSrcs/build_cutedsl.py" +run("${_edge_kernel_python}" -I "@_edge_source@/kernelSrcs/build_cutedsl.py" --gpu_arch "sm_@TRTMC_EDGELLM_CUDA_ARCHITECTURE@" --arch "@CMAKE_SYSTEM_PROCESSOR@" --kernels "@_edge_cute_cli_groups@" --cuda-version "@CUDAToolkit_VERSION_MAJOR@.@CUDAToolkit_VERSION_MINOR@" --jobs "@TRTMC_EDGELLM_JOBS@") diff --git a/cmake/edge_llm/README.md b/cmake/edge_llm/README.md index 1490069b0b..1f10ba7cb3 100644 --- a/cmake/edge_llm/README.md +++ b/cmake/edge_llm/README.md @@ -84,3 +84,12 @@ when replacing an older SDK that included these private console scripts. Preparation verifies the actual Git checkout against the official pin before installing dependencies, including on CMake versions with older disconnected update behavior. + +CUDA 12 provisioning requires Python 3.10-3.12. Its pinned CuPy 12.3 kernel +compiler uses a separate build-only environment with NumPy 1.26.4; the installed +SDK and ONNX exporter use NumPy 2.2.6. Both environments must pass pip check. +CUDA 13 uses the SDK environment for kernel compilation. Bootstrap pip is pinned +to 24.0 and cuda-python to 12.9.7 (CUDA 12) or 13.3.1 (CUDA 13); these bindings +do not determine the native toolkit identity. Offline wheelhouses must include +these exact pins and the dependencies for both environments. The kernel-only +environment and its dependency report remain under the dependency build root. From a2c1189af2930de7bb86f84023097bb6bb6bf0ae Mon Sep 17 00:00:00 2001 From: Joshua Calafato Date: Wed, 23 Sep 2026 21:37:15 +0000 Subject: [PATCH 18/20] fix(edge-llm): isolate CUDA 12 kernel dependencies Signed-off-by: Joshua Calafato --- cmake/edge_llm/EdgeLLM.cmake | 3 +++ cmake/edge_llm/Prepare.cmake.in | 2 +- cmake/edge_llm/README.md | 3 ++- 3 files changed, 6 insertions(+), 2 deletions(-) diff --git a/cmake/edge_llm/EdgeLLM.cmake b/cmake/edge_llm/EdgeLLM.cmake index a2e907a0de..116add9b72 100644 --- a/cmake/edge_llm/EdgeLLM.cmake +++ b/cmake/edge_llm/EdgeLLM.cmake @@ -121,6 +121,9 @@ find_package(Python3 3.10 REQUIRED COMPONENTS Interpreter) if(CUDAToolkit_VERSION_MAJOR EQUAL 12 AND Python3_VERSION VERSION_GREATER_EQUAL "3.13") message(FATAL_ERROR "Pinned CUDA 12 CuPy kernels require Python 3.10-3.12; select Python3_EXECUTABLE accordingly") endif() +if(Python3_VERSION VERSION_GREATER_EQUAL "3.14") + message(FATAL_ERROR "Pinned EdgeLLM NumPy requires Python 3.10-3.13; select Python3_EXECUTABLE accordingly") +endif() find_package(Threads REQUIRED) set(TRTMC_EDGELLM_TRT_ROOT "$ENV{TRT_ROOT}" CACHE PATH "Native TensorRT SDK, including its Python wheel") set(TRTMC_EDGELLM_JOBS 2 CACHE STRING "Parallel Edge-LLM native and AOT compilation jobs") diff --git a/cmake/edge_llm/Prepare.cmake.in b/cmake/edge_llm/Prepare.cmake.in index 9c4dce1614..75c69fc4fd 100644 --- a/cmake/edge_llm/Prepare.cmake.in +++ b/cmake/edge_llm/Prepare.cmake.in @@ -30,7 +30,7 @@ function(create_environment destination) "${destination}") endif() # Pin even when ensurepip seeds a newer version; offline installs stay offline. - run("${destination}/bin/python" -I -m pip ${_pip_options} "pip==24.0") + run("${destination}/bin/python" -I -m pip ${_pip_options} "pip==26.2.1") endfunction() create_environment("@_edge_prefix@/libexec/trtmc-edge-llm") file(GLOB _trt_wheels "@TRTMC_EDGELLM_TRT_ROOT@/python/tensorrt-*-cp@Python3_VERSION_MAJOR@@Python3_VERSION_MINOR@-none-linux_@CMAKE_SYSTEM_PROCESSOR@.whl") diff --git a/cmake/edge_llm/README.md b/cmake/edge_llm/README.md index 1f10ba7cb3..280bf1697c 100644 --- a/cmake/edge_llm/README.md +++ b/cmake/edge_llm/README.md @@ -89,7 +89,8 @@ CUDA 12 provisioning requires Python 3.10-3.12. Its pinned CuPy 12.3 kernel compiler uses a separate build-only environment with NumPy 1.26.4; the installed SDK and ONNX exporter use NumPy 2.2.6. Both environments must pass pip check. CUDA 13 uses the SDK environment for kernel compilation. Bootstrap pip is pinned -to 24.0 and cuda-python to 12.9.7 (CUDA 12) or 13.3.1 (CUDA 13); these bindings +to 26.2.1 and cuda-python to 12.9.7 (CUDA 12) or 13.3.1 (CUDA 13); these bindings do not determine the native toolkit identity. Offline wheelhouses must include these exact pins and the dependencies for both environments. The kernel-only environment and its dependency report remain under the dependency build root. +CUDA 13 provisioning requires Python 3.10-3.13 because of the pinned NumPy wheel support. From ca48ee345d5e93628f26354285d0fda66e6fe2e9 Mon Sep 17 00:00:00 2001 From: Joshua Calafato Date: Wed, 23 Sep 2026 22:08:02 +0000 Subject: [PATCH 19/20] fix(edge-llm): match Python to the native TRT SDK Signed-off-by: Joshua Calafato --- cmake/edge_llm/Prepare.cmake.in | 6 +++--- cmake/edge_llm/README.md | 1 + 2 files changed, 4 insertions(+), 3 deletions(-) diff --git a/cmake/edge_llm/Prepare.cmake.in b/cmake/edge_llm/Prepare.cmake.in index 75c69fc4fd..a561d44d74 100644 --- a/cmake/edge_llm/Prepare.cmake.in +++ b/cmake/edge_llm/Prepare.cmake.in @@ -33,10 +33,10 @@ function(create_environment destination) run("${destination}/bin/python" -I -m pip ${_pip_options} "pip==26.2.1") endfunction() create_environment("@_edge_prefix@/libexec/trtmc-edge-llm") -file(GLOB _trt_wheels "@TRTMC_EDGELLM_TRT_ROOT@/python/tensorrt-*-cp@Python3_VERSION_MAJOR@@Python3_VERSION_MINOR@-none-linux_@CMAKE_SYSTEM_PROCESSOR@.whl") +file(GLOB _trt_wheels "@TRTMC_EDGELLM_TRT_ROOT@/python/tensorrt-@_edge_trt_version@-cp@Python3_VERSION_MAJOR@@Python3_VERSION_MINOR@-none-linux_@CMAKE_SYSTEM_PROCESSOR@.whl") list(LENGTH _trt_wheels _wheel_count) if(NOT _wheel_count EQUAL 1) - message(FATAL_ERROR "Expected exactly one TensorRT SDK wheel matching the native Python ABI") + message(FATAL_ERROR "Expected the TensorRT @_edge_trt_version@ SDK wheel matching the native Python ABI") endif() # CUDA 12 CuPy needs NumPy 1.x, while the installed Edge SDK needs NumPy 2.x. # Keep its AOT-only environment outside the installed SDK and check both. @@ -73,7 +73,7 @@ if("@TRTMC_EDGELLM_ONNX@") run("@_edge_python@" -I -m pip ${_pip_options} "tensorrt-edgellm[export]==@_edge_version@") endif() run("@_edge_python@" -I -m pip --isolated check) -run("@_edge_python@" -I -c "print(__import__('tensorrt').__version__)") +run("@_edge_python@" -I -c "__import__('sys').exit(0 if __import__('tensorrt').__version__ == '@_edge_trt_version@' else 'TensorRT Python ' + __import__('tensorrt').__version__ + ' differs from native SDK @_edge_trt_version@')") set(ENV{PATH} "@CUDAToolkit_BIN_DIR@:$ENV{PATH}") run("${_edge_kernel_python}" -I "@_edge_source@/kernelSrcs/build_cutedsl.py" --gpu_arch "sm_@TRTMC_EDGELLM_CUDA_ARCHITECTURE@" --arch "@CMAKE_SYSTEM_PROCESSOR@" diff --git a/cmake/edge_llm/README.md b/cmake/edge_llm/README.md index 280bf1697c..067ac370f5 100644 --- a/cmake/edge_llm/README.md +++ b/cmake/edge_llm/README.md @@ -94,3 +94,4 @@ do not determine the native toolkit identity. Offline wheelhouses must include these exact pins and the dependencies for both environments. The kernel-only environment and its dependency report remain under the dependency build root. CUDA 13 provisioning requires Python 3.10-3.13 because of the pinned NumPy wheel support. +The selected Python wheel and imported TensorRT version must match the exact native SDK header/library version; an ABI-compatible wheel from another release is rejected. From b45cd19ec339854b5362c60a385c367d8498433e Mon Sep 17 00:00:00 2001 From: Joshua Calafato Date: Thu, 24 Sep 2026 06:05:11 +0000 Subject: [PATCH 20/20] fix(tests): honor offline checkpoint loading Pass the Hub offline setting explicitly to snapshot_download. Pinned revisions can otherwise request uncached tree metadata in Hub 1.32 even when the checkpoint was staged before entering the offline runner. Keep network isolation, pinned revisions, and numerical quality gates unchanged. This repairs existing smoke tests, not model support scope. Signed-off-by: Joshua Calafato --- families/qwen/tests/test_e2e.py | 3 ++- 1 file changed, 2 insertions(+), 1 deletion(-) diff --git a/families/qwen/tests/test_e2e.py b/families/qwen/tests/test_e2e.py index 2c62d307ff..2f15085986 100644 --- a/families/qwen/tests/test_e2e.py +++ b/families/qwen/tests/test_e2e.py @@ -104,12 +104,13 @@ def _required_environment(tp_size: int): def _checkpoint(manifest: dict) -> Path: - from huggingface_hub import snapshot_download + from huggingface_hub import constants, snapshot_download path = Path( snapshot_download( repo_id=manifest["hf_id"], revision=manifest.get("hf_revision"), + local_files_only=constants.HF_HUB_OFFLINE, ) ) assert (path / "config.json").is_file(), path