diff --git a/CMakeLists.txt b/CMakeLists.txt index 8a083fa6f4..92e11c3ee0 100644 --- a/CMakeLists.txt +++ b/CMakeLists.txt @@ -45,6 +45,8 @@ find_library(TRTMC_TRT_LIBRARY REQUIRED ) +include("${CMAKE_CURRENT_SOURCE_DIR}/cmake/edge_llm/EdgeLLM.cmake") + option(TRTMC_ENABLE_BYOK "Enable the optional TVM-FFI BYOK bridge" ON) set(TRTMC_HAS_TVM_FFI OFF) if(TRTMC_ENABLE_BYOK) diff --git a/apps/cli/main.cpp b/apps/cli/main.cpp index dbb5c93bb8..8a27e4292e 100644 --- a/apps/cli/main.cpp +++ b/apps/cli/main.cpp @@ -8,5 +8,12 @@ #include int main(int argc, char** argv) { - return trtmc::cli::run(argc, argv, std::cout, std::cerr); + // The executable owns the console: keep result output machine-readable even + // when loaded libraries write C++ diagnostics to std::cout. Do not change + // library logger levels or the output behavior of embedded runtime APIs. + std::ostream result(std::cout.rdbuf()); + std::cout.rdbuf(std::cerr.rdbuf()); + const int status = trtmc::cli::run(argc, argv, result, std::cerr); + result.flush(); + return status; } diff --git a/cmake/edge_llm/CheckNative.cmake b/cmake/edge_llm/CheckNative.cmake new file mode 100644 index 0000000000..2d7118de5f --- /dev/null +++ b/cmake/edge_llm/CheckNative.cmake @@ -0,0 +1,112 @@ +# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-License-Identifier: Apache-2.0 + +# Read the complete TensorRT SDK version using the compiler, including aliased macros. +# include_dir: native SDK include directory; output: caller variable receiving x.y.z.build. +function(_edgellm_trt_version include_dir output) + set(_version) + set(_probe "${CMAKE_CURRENT_BINARY_DIR}/CMakeFiles/edgellm-version.cpp") + file(WRITE "${_probe}" "#include \n") + foreach(_part IN ITEMS MAJOR MINOR PATCH BUILD) + file(APPEND "${_probe}" "TRTMC_EDGE_${_part}=NV_TENSORRT_${_part}\n") + endforeach() + execute_process(COMMAND "${CMAKE_CXX_COMPILER}" -E -P -I "${include_dir}" "${_probe}" + OUTPUT_VARIABLE _expanded COMMAND_ERROR_IS_FATAL ANY) + foreach(_part IN ITEMS MAJOR MINOR PATCH BUILD) + if(NOT _expanded MATCHES "TRTMC_EDGE_${_part}=[ \t]*([0-9]+)") + message(FATAL_ERROR "Cannot determine TensorRT ${_part} from ${include_dir}") + endif() + list(APPEND _version "${CMAKE_MATCH_1}") + endforeach() + list(JOIN _version "." _version) + set(${output} "${_version}" PARENT_SCOPE) +endfunction() + +# Require the installed package GPU architecture to be present on this build host. +function(_edgellm_check_gpu architecture) + execute_process(COMMAND nvidia-smi --query-gpu=compute_cap --format=csv,noheader + OUTPUT_VARIABLE _sms RESULT_VARIABLE _result OUTPUT_STRIP_TRAILING_WHITESPACE) + string(REPLACE "." "" _sms "${_sms}") + string(REPLACE "\n" ";" _sms "${_sms}") + if(NOT _result EQUAL 0 OR NOT architecture IN_LIST _sms) + message(FATAL_ERROR "EdgeLLM requires a local GPU with architecture ${architecture}") + endif() +endfunction() + +# A version label alone is not an ABI guarantee: development headers can retain +# 3.12.0 while changing parser layouts inside the same C++ ABI namespace. +function(_edgellm_json_include output) + get_target_property(_includes nlohmann_json::nlohmann_json INTERFACE_INCLUDE_DIRECTORIES) + foreach(_include IN LISTS _includes) + string(REGEX REPLACE "^\\$$" "\\1" _include "${_include}") + if(EXISTS "${_include}/nlohmann/json.hpp") + set(${output} "${_include}" PARENT_SCOPE) + return() + endif() + endforeach() + message(FATAL_ERROR "Cannot locate nlohmann_json headers for EdgeLLM ABI validation") +endfunction() + +function(_edgellm_check_json_headers include_dir vendor_dir) + set(_header "${include_dir}/nlohmann/json.hpp") + set(_single "${vendor_dir}/single_include/nlohmann/json.hpp") + set(_multiple "${vendor_dir}/include/nlohmann/json.hpp") + if(NOT EXISTS "${_header}" OR NOT EXISTS "${_single}" OR NOT EXISTS "${_multiple}") + message(FATAL_ERROR "Missing nlohmann_json headers for EdgeLLM ABI validation") + endif() + file(SHA256 "${_header}" _actual) + file(SHA256 "${_single}" _expected_single) + if(_actual STREQUAL _expected_single) + return() + endif() + file(GLOB_RECURSE _headers RELATIVE "${vendor_dir}/include" "${vendor_dir}/include/nlohmann/*.hpp") + foreach(_relative IN LISTS _headers) + if(EXISTS "${include_dir}/${_relative}") + file(SHA256 "${include_dir}/${_relative}" _actual) + file(SHA256 "${vendor_dir}/include/${_relative}" _expected) + if(_actual STREQUAL _expected) + continue() + endif() + endif() + message(FATAL_ERROR "EdgeLLM requires the pinned nlohmann_json headers, not only the same version label. Set nlohmann_json_DIR to an installation of the pinned Edge 3rdParty/nlohmannJson dependency. Mismatch: ${_relative}") + endforeach() +endfunction() + +# Verify the selected native library itself, not only its accompanying headers. +function(_edgellm_check_trt_library library expected) + if(CMAKE_CROSSCOMPILING) + message(FATAL_ERROR "TensorRT library validation requires native execution") + endif() + set(_probe "${CMAKE_CURRENT_BINARY_DIR}/CMakeFiles/edgellm-library-version.cpp") + file(WRITE "${_probe}" [=[ +#include +#include +int main(int argc, char** argv) { + if (argc != 2) return 1; + void* library = dlopen(argv[1], RTLD_NOW | RTLD_LOCAL); + if (!library) { std::cerr << dlerror(); return 2; } + const char* names[] = {"getInferLibMajorVersion", "getInferLibMinorVersion", + "getInferLibPatchVersion", "getInferLibBuildVersion"}; + for (int i = 0; i < 4; ++i) { + auto version = reinterpret_cast(dlsym(library, names[i])); + if (!version) { std::cerr << "Missing " << names[i]; dlclose(library); return 3; } + if (i) std::cout << "."; + std::cout << version(); + } + dlclose(library); + return 0; +} +]=]) + unset(_edge_version_run CACHE) + unset(_edge_version_compiled CACHE) + try_run(_edge_version_run _edge_version_compiled + "${CMAKE_CURRENT_BINARY_DIR}/CMakeFiles/edgellm-library-version" "${_probe}" + LINK_LIBRARIES "${CMAKE_DL_LIBS}" ARGS "${library}" + RUN_OUTPUT_VARIABLE _actual COMPILE_OUTPUT_VARIABLE _compile_output) + if(NOT _edge_version_compiled OR NOT _edge_version_run STREQUAL "0") + message(FATAL_ERROR "Cannot verify selected TensorRT library ${library}: ${_actual} ${_compile_output}") + endif() + if(NOT _actual STREQUAL expected) + message(FATAL_ERROR "EdgeLLM requires TensorRT library ${expected}; selected ${library} reports ${_actual}") + endif() +endfunction() diff --git a/cmake/edge_llm/EdgeLLM.cmake b/cmake/edge_llm/EdgeLLM.cmake new file mode 100644 index 0000000000..116add9b72 --- /dev/null +++ b/cmake/edge_llm/EdgeLLM.cmake @@ -0,0 +1,205 @@ +# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-License-Identifier: Apache-2.0 + +# Optional native dependency provisioning. Model builds never acquire dependencies. +option(TRTMC_ENABLE_EDGELLM "Install the pinned native Edge-LLM SDK and builder" OFF) +if(NOT TRTMC_ENABLE_EDGELLM) + return() +endif() +if(CMAKE_CROSSCOMPILING) + message(FATAL_ERROR "Edge-LLM cross compilation is not supported") +endif() +include("${CMAKE_CURRENT_LIST_DIR}/CheckNative.cmake") + +# Model Connect and the offload must link the same native TensorRT installation. +function(_edgellm_check_trt_selection include_dir library) + foreach(_kind IN ITEMS INCLUDE_DIR LIBRARY) + get_filename_component(_parent "${TRTMC_TRT_${_kind}}" REALPATH) + if(_kind STREQUAL "INCLUDE_DIR") + get_filename_component(_selected "${include_dir}" REALPATH) + else() + get_filename_component(_selected "${library}" REALPATH) + endif() + if(NOT _parent STREQUAL _selected) + message(FATAL_ERROR "Model Connect and EdgeLLM must use the same TensorRT ${_kind}: ${_parent} != ${_selected}. Set TRTMC_TRT_INCLUDE_DIR, TRTMC_TRT_LIBRARY and TRTMC_EDGELLM_TRT_ROOT to one SDK.") + endif() + endforeach() + _edgellm_trt_version("${TRTMC_TRT_INCLUDE_DIR}" _parent_version) + _edgellm_trt_version("${include_dir}" _selected_version) + if(NOT _parent_version STREQUAL _selected_version) + message(FATAL_ERROR "Model Connect and EdgeLLM TensorRT header versions differ") + endif() +endfunction() +option(TRTMC_EDGELLM_ALL_KERNELS "Build all pinned Edge operator groups supported by the native GPU" OFF) +option(TRTMC_EDGELLM_ONNX "Install the pinned ONNX exporter and native engine builder" OFF) +set(_edge_cute_groups "fmha|gdn") +set(_edge_cute_cli_groups "fmha,gdn") +if(TRTMC_EDGELLM_ALL_KERNELS) + set(_edge_cute_groups ALL) + set(_edge_cute_cli_groups ALL) +endif() +set(_edge_build_targets edgellmCore NvInfer_edgellm_plugin) +set(_edge_onnx_byproducts "") +if(TRTMC_EDGELLM_ONNX) + list(APPEND _edge_build_targets llm_build) + list(APPEND _edge_onnx_byproducts "${CMAKE_BINARY_DIR}/_deps/edgellm/install/bin/edgellm-onnx-build") +endif() +set(_edge_version "0.10.1") +set(_edge_revision "e8b29522938901f6df19ebeedd4b69bc8edbcd97") +set(_edge_root "${CMAKE_BINARY_DIR}/_deps/edgellm") +set(_edge_prefix "${_edge_root}/install") +set(TRTMC_EDGELLM_CUDA_ARCHITECTURE "${CMAKE_CUDA_ARCHITECTURES}" CACHE STRING "One local GPU architecture for Edge-LLM") +if(NOT TRTMC_EDGELLM_CUDA_ARCHITECTURE MATCHES "^[0-9]+$") + message(FATAL_ERROR "Set TRTMC_EDGELLM_CUDA_ARCHITECTURE to one local GPU architecture, e.g. 80") +endif() +# Do not import this build tree's previous generated package before regenerating +# it: its baked SDK checks and imported targets may describe the old configure. +set(_edge_package_dir "${_edge_prefix}/lib/cmake/EdgeLLM") +get_filename_component(_edge_package_real "${_edge_package_dir}" REALPATH) +if(EdgeLLM_DIR) + get_filename_component(_edge_cached_real "${EdgeLLM_DIR}" REALPATH) + if(_edge_cached_real STREQUAL _edge_package_real) + unset(EdgeLLM_DIR CACHE) + unset(EdgeLLM_DIR) + endif() +endif() +set(_edge_saved_ignore_path "${CMAKE_IGNORE_PATH}") +list(APPEND CMAKE_IGNORE_PATH "${_edge_package_dir}" "${_edge_package_real}") +find_package(EdgeLLM ${_edge_version} EXACT CONFIG QUIET) +set(CMAKE_IGNORE_PATH "${_edge_saved_ignore_path}") + +function(_edgellm_install_plugin) + # Preserve the complete SONAME chain when lib and lib64 differ. Install-time + # expansion also honors cmake --install --prefix and DESTDIR. + install(CODE "file(INSTALL + DESTINATION \"\${CMAKE_INSTALL_PREFIX}/${CMAKE_INSTALL_LIBDIR}\" + TYPE SHARED_LIBRARY FOLLOW_SYMLINK_CHAIN + FILES \"$\")" COMPONENT EdgeLLM) +endfunction() +function(_edgellm_check_external_artifacts) + foreach(_tool IN ITEMS EdgeLLM_PYTHON_EXECUTABLE EdgeLLM_BUILDER_LAUNCHER) + if(NOT EXISTS "${${_tool}}" OR IS_DIRECTORY "${${_tool}}") + message(FATAL_ERROR "External EdgeLLM package is incomplete: ${_tool} missing at ${${_tool}}") + endif() + endforeach() + foreach(_target IN ITEMS EdgeLLM::Core EdgeLLM::Plugin) + get_target_property(_artifact ${_target} IMPORTED_LOCATION) + if(NOT EXISTS "${_artifact}" OR IS_DIRECTORY "${_artifact}") + message(FATAL_ERROR "External EdgeLLM package is incomplete: ${_target} missing at ${_artifact}") + endif() + endforeach() + if(NOT EXISTS "${EdgeLLM_PREFIX}/lib/libcutedsl.a" OR IS_DIRECTORY "${EdgeLLM_PREFIX}/lib/libcutedsl.a") + message(FATAL_ERROR "External EdgeLLM package is incomplete: missing libcutedsl.a") + endif() +endfunction() + +if(EdgeLLM_FOUND AND NOT EdgeLLM_PREFIX STREQUAL _edge_prefix) + _edgellm_check_external_artifacts() + _edgellm_check_trt_selection("${EdgeLLM_TRT_INCLUDE_DIR}" "${EdgeLLM_TRT_LIBRARY}") + if(NOT EdgeLLM_CUDA_ARCHITECTURE STREQUAL TRTMC_EDGELLM_CUDA_ARCHITECTURE) + message(FATAL_ERROR "EdgeLLM package architecture ${EdgeLLM_CUDA_ARCHITECTURE} differs from requested ${TRTMC_EDGELLM_CUDA_ARCHITECTURE}") + endif() + _edgellm_check_trt_library("${EdgeLLM_TRT_LIBRARY}" "${EdgeLLM_TENSORRT_VERSION}") + _edgellm_json_include(_edge_json_include) + _edgellm_check_json_headers("${_edge_json_include}" "${EdgeLLM_PREFIX}/include/edgellm/3rdParty/nlohmannJson") + if(NOT EdgeLLM_REVISION STREQUAL _edge_revision) + message(FATAL_ERROR "EdgeLLM package does not match the pinned GitHub revision") + endif() + if(TRTMC_EDGELLM_ALL_KERNELS AND NOT EdgeLLM_ALL_KERNELS) + message(FATAL_ERROR "EdgeLLM package lacks requested full native operator coverage; rebuild with TRTMC_EDGELLM_ALL_KERNELS=ON") + endif() + if(TRTMC_EDGELLM_ONNX AND (NOT EdgeLLM_ONNX OR NOT EXISTS "${EdgeLLM_ONNX_BUILDER}")) + message(FATAL_ERROR "EdgeLLM package lacks requested ONNX tools; rebuild with TRTMC_EDGELLM_ONNX=ON") + endif() + _edgellm_install_plugin() + return() +endif() + +include(ExternalProject) +include(CMakePackageConfigHelpers) +find_package(Python3 3.10 REQUIRED COMPONENTS Interpreter) +if(CUDAToolkit_VERSION_MAJOR EQUAL 12 AND Python3_VERSION VERSION_GREATER_EQUAL "3.13") + message(FATAL_ERROR "Pinned CUDA 12 CuPy kernels require Python 3.10-3.12; select Python3_EXECUTABLE accordingly") +endif() +if(Python3_VERSION VERSION_GREATER_EQUAL "3.14") + message(FATAL_ERROR "Pinned EdgeLLM NumPy requires Python 3.10-3.13; select Python3_EXECUTABLE accordingly") +endif() +find_package(Threads REQUIRED) +set(TRTMC_EDGELLM_TRT_ROOT "$ENV{TRT_ROOT}" CACHE PATH "Native TensorRT SDK, including its Python wheel") +set(TRTMC_EDGELLM_JOBS 2 CACHE STRING "Parallel Edge-LLM native and AOT compilation jobs") +set(TRTMC_EDGELLM_WHEELHOUSE "" CACHE PATH "Optional complete offline Python wheelhouse") +set(TRTMC_EDGELLM_GIT_MIRROR "" CACHE PATH "Optional local mirror of the pinned upstream Git repository") +if(NOT EXISTS "${TRTMC_EDGELLM_TRT_ROOT}/include/NvInfer.h") + message(FATAL_ERROR "TRTMC_EDGELLM_TRT_ROOT must contain the native TensorRT SDK") +endif() +unset(_edge_selected_trt_library CACHE) +unset(_edge_selected_trt_library) +find_library(_edge_selected_trt_library nvinfer + PATHS "${TRTMC_EDGELLM_TRT_ROOT}/lib" "${TRTMC_EDGELLM_TRT_ROOT}/lib64" + NO_DEFAULT_PATH REQUIRED) +_edgellm_check_trt_selection("${TRTMC_EDGELLM_TRT_ROOT}/include" "${_edge_selected_trt_library}") +_edgellm_check_gpu("${TRTMC_EDGELLM_CUDA_ARCHITECTURE}") +_edgellm_trt_version("${TRTMC_EDGELLM_TRT_ROOT}/include" _edge_trt_version) +set(_edge_source "${_edge_root}/source") +set(_edge_build "${_edge_root}/build") +set(_edge_python "${_edge_prefix}/libexec/trtmc-edge-llm/bin/python") +set(_edge_repository "https://github.com/NVIDIA/TensorRT-Edge-LLM.git") +if(TRTMC_EDGELLM_GIT_MIRROR) + set(_edge_repository "${TRTMC_EDGELLM_GIT_MIRROR}") +endif() +set(_edge_template_dir "${CMAKE_CURRENT_LIST_DIR}") +_edgellm_json_include(_edge_json_include) +file(MAKE_DIRECTORY "${_edge_prefix}/lib/cmake/EdgeLLM" "${_edge_prefix}/include/edgellm/cpp" + "${_edge_prefix}/include/edgellm/3rdParty/nlohmannJson/include" + "${_edge_prefix}/include/edgellm/3rdParty/stb" "${_edge_prefix}/include/edgellm/3rdParty/miniaudio") +configure_file("${_edge_template_dir}/CheckNative.cmake" "${_edge_prefix}/lib/cmake/EdgeLLM/CheckNative.cmake" COPYONLY) +foreach(_script IN ITEMS Prepare Install) + configure_file("${_edge_template_dir}/${_script}.cmake.in" "${_edge_root}/${_script}.cmake" @ONLY) +endforeach() +configure_file("${_edge_template_dir}/EdgeLLMConfig.cmake.in" + "${_edge_prefix}/lib/cmake/EdgeLLM/EdgeLLMConfig.cmake" @ONLY) +write_basic_package_version_file("${_edge_prefix}/lib/cmake/EdgeLLM/EdgeLLMConfigVersion.cmake" + VERSION "${_edge_version}" COMPATIBILITY ExactVersion) +ExternalProject_Add(trtmc_edgellm_dependency + PREFIX "${_edge_root}/ep" SOURCE_DIR "${_edge_source}" BINARY_DIR "${_edge_build}" + GIT_REPOSITORY "${_edge_repository}" GIT_TAG "${_edge_revision}" + GIT_SUBMODULES_RECURSE TRUE UPDATE_DISCONNECTED TRUE + LIST_SEPARATOR | + # Preparation installs tools; it does not patch upstream sources. Keep it in + # the configure step so template changes invalidate disconnected builds too. + CONFIGURE_COMMAND "${CMAKE_COMMAND}" -P "${_edge_root}/Prepare.cmake" + COMMAND "${_edge_prefix}/libexec/trtmc-edge-llm/bin/cmake" + -S -B -DCMAKE_BUILD_TYPE=Release -DCMAKE_POSITION_INDEPENDENT_CODE=ON + "-DCMAKE_CUDA_COMPILER=${CMAKE_CUDA_COMPILER}" + "-DCMAKE_CUDA_ARCHITECTURES=${TRTMC_EDGELLM_CUDA_ARCHITECTURE}" + "-DCUDA_DIR=${CUDAToolkit_LIBRARY_ROOT}" "-DCUDAToolkit_ROOT=${CUDAToolkit_LIBRARY_ROOT}" + "-DCUDA_CTK_VERSION=${CUDAToolkit_VERSION_MAJOR}.${CUDAToolkit_VERSION_MINOR}" + "-DTRT_PACKAGE_DIR=${TRTMC_EDGELLM_TRT_ROOT}" "-DPython3_EXECUTABLE=${_edge_python}" + -DEDGELLM_WHEEL_PAYLOAD_DIR=unused "-DENABLE_CUTE_DSL=${_edge_cute_groups}" + "-DCUTE_DSL_ARTIFACT_TAG=sm_${TRTMC_EDGELLM_CUDA_ARCHITECTURE}" + BUILD_COMMAND "${CMAKE_COMMAND}" --build --target ${_edge_build_targets} + --parallel "${TRTMC_EDGELLM_JOBS}" + INSTALL_COMMAND "${CMAKE_COMMAND}" -P "${_edge_root}/Install.cmake" + BUILD_BYPRODUCTS "${_edge_prefix}/lib/libedgellmCore.a" + "${_edge_prefix}/lib/libNvInfer_edgellm_plugin.so" + "${_edge_prefix}/lib/libcutedsl.a" ${_edge_onnx_byproducts} + LOG_DOWNLOAD ON LOG_CONFIGURE ON LOG_BUILD ON LOG_INSTALL ON LOG_OUTPUT_ON_FAILURE ON) +ExternalProject_Add_StepDependencies(trtmc_edgellm_dependency configure "${_edge_root}/Prepare.cmake") +ExternalProject_Add_StepDependencies(trtmc_edgellm_dependency install "${_edge_root}/Install.cmake") +# Generated package targets refer to declared future byproducts; their build dependency +# prevents consumers from compiling or linking until installation completes. +set(EdgeLLM_DIR "${_edge_package_dir}" CACHE PATH "Edge-LLM package directory" FORCE) +find_package(EdgeLLM ${_edge_version} EXACT CONFIG REQUIRED + PATHS "${_edge_prefix}/lib/cmake/EdgeLLM" NO_DEFAULT_PATH) +add_dependencies(EdgeLLM::Core trtmc_edgellm_dependency) +add_dependencies(EdgeLLM::Plugin trtmc_edgellm_dependency) +# Runtime consumers use the interpreter/modules and the prefix-relative launcher. +# Build-only console scripts/activation files embed build-tree paths; keep them +# available for rebuilding the dependency, but do not publish them in the SDK. +install(DIRECTORY "${_edge_prefix}/" DESTINATION . USE_SOURCE_PERMISSIONS COMPONENT EdgeLLM + PATTERN "libexec/trtmc-edge-llm/bin" EXCLUDE) +install(DIRECTORY "${_edge_prefix}/libexec/trtmc-edge-llm/bin/" + DESTINATION libexec/trtmc-edge-llm/bin USE_SOURCE_PERMISSIONS COMPONENT EdgeLLM + FILES_MATCHING REGEX "/python([0-9]+(\\.[0-9]+)?)?$") +# Family DSOs may use lib64; their dynamically loaded plugin must remain adjacent. +_edgellm_install_plugin() diff --git a/cmake/edge_llm/EdgeLLMConfig.cmake.in b/cmake/edge_llm/EdgeLLMConfig.cmake.in new file mode 100644 index 0000000000..7f8da5df01 --- /dev/null +++ b/cmake/edge_llm/EdgeLLMConfig.cmake.in @@ -0,0 +1,67 @@ +# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-License-Identifier: Apache-2.0 +include(CMakeFindDependencyMacro) +find_dependency(CUDAToolkit) +find_dependency(Threads) +find_dependency(nlohmann_json 3.12.0 EXACT) +include("${CMAKE_CURRENT_LIST_DIR}/CheckNative.cmake") +get_filename_component(EdgeLLM_PREFIX "${CMAKE_CURRENT_LIST_DIR}/../../.." ABSOLUTE) +_edgellm_json_include(_edge_json_include) +# During first provisioning these future headers do not exist yet; Prepare +# performs the same check after checkout and before installing or building tools. +if(EXISTS "${EdgeLLM_PREFIX}/include/edgellm/3rdParty/nlohmannJson/include/nlohmann/json.hpp") + _edgellm_check_json_headers("${_edge_json_include}" "${EdgeLLM_PREFIX}/include/edgellm/3rdParty/nlohmannJson") +endif() +set(EdgeLLM_VERSION "@_edge_version@") +set(EdgeLLM_REVISION "@_edge_revision@") +# Tool availability only; model admission and orchestration remain family-owned. +set(EdgeLLM_ALL_KERNELS "@TRTMC_EDGELLM_ALL_KERNELS@") +set(EdgeLLM_ONNX "@TRTMC_EDGELLM_ONNX@") +set(EdgeLLM_ONNX_BUILDER "${EdgeLLM_PREFIX}/bin/edgellm-onnx-build") +set(EdgeLLM_CUDA_VERSION "@CUDAToolkit_VERSION@") +set(EdgeLLM_TENSORRT_VERSION "@_edge_trt_version@") +set(EdgeLLM_ARCH "@CMAKE_SYSTEM_PROCESSOR@") +set(EdgeLLM_CUDA_ARCHITECTURE "@TRTMC_EDGELLM_CUDA_ARCHITECTURE@") +if(CMAKE_CROSSCOMPILING OR NOT CMAKE_SYSTEM_PROCESSOR STREQUAL EdgeLLM_ARCH) + message(FATAL_ERROR "EdgeLLM is a native-only package for ${EdgeLLM_ARCH}") +endif() +if(NOT CUDAToolkit_VERSION_MAJOR EQUAL @CUDAToolkit_VERSION_MAJOR@ OR + NOT CUDAToolkit_VERSION_MINOR EQUAL @CUDAToolkit_VERSION_MINOR@) + message(FATAL_ERROR "EdgeLLM requires the CUDA SDK it was built with: ${EdgeLLM_CUDA_VERSION}") +endif() +set(EdgeLLM_PYTHON_EXECUTABLE "${EdgeLLM_PREFIX}/libexec/trtmc-edge-llm/bin/python") +set(EdgeLLM_BUILDER_LAUNCHER "${EdgeLLM_PREFIX}/bin/edgellm-builder") +# Select one SDK root before searching: never mix headers and libraries from +# different installations, including values left in an earlier CMake cache. +if(TRTMC_EDGELLM_TRT_ROOT) + set(_edge_trt_root "${TRTMC_EDGELLM_TRT_ROOT}") +elseif(DEFINED ENV{TRT_ROOT} AND NOT "$ENV{TRT_ROOT}" STREQUAL "") + set(_edge_trt_root "$ENV{TRT_ROOT}") +else() + set(_edge_trt_root "@TRTMC_EDGELLM_TRT_ROOT@") +endif() +if(NOT _edge_trt_root) + message(FATAL_ERROR "Set TRTMC_EDGELLM_TRT_ROOT to one complete native TensorRT SDK") +endif() +foreach(_artifact IN ITEMS EdgeLLM_TRT_INCLUDE_DIR EdgeLLM_TRT_LIBRARY EdgeLLM_PARSER_LIBRARY) + unset(${_artifact}) + unset(${_artifact} CACHE) +endforeach() +find_path(EdgeLLM_TRT_INCLUDE_DIR NvInfer.h PATHS "${_edge_trt_root}/include" NO_DEFAULT_PATH REQUIRED) +find_library(EdgeLLM_TRT_LIBRARY nvinfer PATHS "${_edge_trt_root}/lib" "${_edge_trt_root}/lib64" NO_DEFAULT_PATH REQUIRED) +find_library(EdgeLLM_PARSER_LIBRARY nvonnxparser PATHS "${_edge_trt_root}/lib" "${_edge_trt_root}/lib64" NO_DEFAULT_PATH REQUIRED) +_edgellm_trt_version("${EdgeLLM_TRT_INCLUDE_DIR}" _edge_current_trt) +if(NOT _edge_current_trt STREQUAL EdgeLLM_TENSORRT_VERSION) + message(FATAL_ERROR "EdgeLLM requires TensorRT ${EdgeLLM_TENSORRT_VERSION}; found ${_edge_current_trt}") +endif() +_edgellm_check_trt_library("${EdgeLLM_TRT_LIBRARY}" "${EdgeLLM_TENSORRT_VERSION}") +_edgellm_check_gpu("${EdgeLLM_CUDA_ARCHITECTURE}") +if(NOT TARGET EdgeLLM::Core) + add_library(EdgeLLM::Core STATIC IMPORTED) + set_target_properties(EdgeLLM::Core PROPERTIES + IMPORTED_LOCATION "${EdgeLLM_PREFIX}/lib/libedgellmCore.a" + INTERFACE_INCLUDE_DIRECTORIES "${EdgeLLM_PREFIX}/include;${EdgeLLM_PREFIX}/include/edgellm/cpp;${EdgeLLM_PREFIX}/include/edgellm/3rdParty/nlohmannJson/include;${EdgeLLM_PREFIX}/include/edgellm/3rdParty/stb;${EdgeLLM_PREFIX}/include/edgellm/3rdParty/miniaudio;${EdgeLLM_TRT_INCLUDE_DIR}" + INTERFACE_LINK_LIBRARIES "${EdgeLLM_PREFIX}/lib/libcutedsl.a;${EdgeLLM_TRT_LIBRARY};${EdgeLLM_PARSER_LIBRARY};CUDA::cudart;CUDA::cuda_driver;Threads::Threads;${CMAKE_DL_LIBS}") + add_library(EdgeLLM::Plugin SHARED IMPORTED) + set_target_properties(EdgeLLM::Plugin PROPERTIES IMPORTED_LOCATION "${EdgeLLM_PREFIX}/lib/libNvInfer_edgellm_plugin.so") +endif() diff --git a/cmake/edge_llm/Install.cmake.in b/cmake/edge_llm/Install.cmake.in new file mode 100644 index 0000000000..e71bd217df --- /dev/null +++ b/cmake/edge_llm/Install.cmake.in @@ -0,0 +1,34 @@ +# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-License-Identifier: Apache-2.0 +cmake_minimum_required(VERSION 3.20) +file(INSTALL "@_edge_build@/cpp/libedgellmCore.a" DESTINATION "@_edge_prefix@/lib") +file(INSTALL "@_edge_build@/libNvInfer_edgellm_plugin.so" DESTINATION "@_edge_prefix@/lib" FOLLOW_SYMLINK_CHAIN) +file(INSTALL "@_edge_source@/cpp/kernels/cuteDSLArtifact/@CMAKE_SYSTEM_PROCESSOR@/sm_@TRTMC_EDGELLM_CUDA_ARCHITECTURE@/libcutedsl_@CMAKE_SYSTEM_PROCESSOR@.a" + DESTINATION "@_edge_prefix@/lib" RENAME libcutedsl.a) +file(INSTALL "@_edge_source@/cpp" DESTINATION "@_edge_prefix@/include/edgellm" FILES_MATCHING PATTERN "*.h" PATTERN "*.cuh") +foreach(_third_party IN ITEMS nlohmannJson stb miniaudio) + file(INSTALL "@_edge_source@/3rdParty/${_third_party}" DESTINATION "@_edge_prefix@/include/edgellm/3rdParty" + FILES_MATCHING PATTERN "*.h" PATTERN "*.hpp") +endforeach() +file(MAKE_DIRECTORY "@_edge_prefix@/bin" "@_edge_prefix@/share/trtmc") +file(WRITE "@_edge_prefix@/bin/edgellm-builder" [=[#!/bin/sh +set -eu +prefix=$(CDPATH= cd -- "$(dirname -- "$0")/.." && pwd) +exec "$prefix/libexec/trtmc-edge-llm/bin/python" -I -c 'from experimental.builder.cli import main; main()' "$@" +]=]) +file(CHMOD "@_edge_prefix@/bin/edgellm-builder" PERMISSIONS OWNER_READ OWNER_WRITE OWNER_EXECUTE GROUP_READ GROUP_EXECUTE WORLD_READ WORLD_EXECUTE) +if("@TRTMC_EDGELLM_ONNX@") + file(INSTALL "@_edge_build@/examples/llm/llm_build" DESTINATION "@_edge_prefix@/bin" + TYPE PROGRAM RENAME edgellm-onnx-build) +endif() +set(_all_kernels false) +if("@TRTMC_EDGELLM_ALL_KERNELS@") + set(_all_kernels true) +endif() +set(_onnx false) +if("@TRTMC_EDGELLM_ONNX@") + set(_onnx true) +endif() +# Bundle compatibility uses the complete native SDK version, not the Python wheel label. +# Paths are relative to the installation prefix, preserving relocatability. +file(WRITE "@_edge_prefix@/share/trtmc/edge-llm.json" "{\n \"schema_version\": 1,\n \"version\": \"@_edge_version@\",\n \"revision\": \"@_edge_revision@\",\n \"arch\": \"@CMAKE_SYSTEM_PROCESSOR@\",\n \"architectures\": [@TRTMC_EDGELLM_CUDA_ARCHITECTURE@],\n \"cuda_version\": \"@CUDAToolkit_VERSION_MAJOR@.@CUDAToolkit_VERSION_MINOR@\",\n \"tensorrt_version\": \"@_edge_trt_version@\",\n \"python\": \"libexec/trtmc-edge-llm/bin/python\",\n \"builder\": \"bin/edgellm-builder\",\n \"all_native_kernels\": ${_all_kernels},\n \"onnx\": ${_onnx},\n \"onnx_builder\": \"bin/edgellm-onnx-build\",\n \"plugin\": \"lib/libNvInfer_edgellm_plugin.so\"\n}\n") diff --git a/cmake/edge_llm/Prepare.cmake.in b/cmake/edge_llm/Prepare.cmake.in new file mode 100644 index 0000000000..a561d44d74 --- /dev/null +++ b/cmake/edge_llm/Prepare.cmake.in @@ -0,0 +1,80 @@ +# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-License-Identifier: Apache-2.0 +# Executed only by the explicit CMake dependency build, never by model dispatch. +cmake_minimum_required(VERSION 3.20) +# Older CMake can skip a disconnected Git update when the requested pin changes. +# Never label an old checkout as the newly requested official snapshot. +find_package(Git REQUIRED) +execute_process(COMMAND "${GIT_EXECUTABLE}" -C "@_edge_source@" rev-parse HEAD + OUTPUT_VARIABLE _edge_checkout_revision OUTPUT_STRIP_TRAILING_WHITESPACE + COMMAND_ERROR_IS_FATAL ANY) +if(NOT _edge_checkout_revision STREQUAL "@_edge_revision@") + message(FATAL_ERROR "EdgeLLM checkout does not match the pinned revision; use a fresh dependency build directory") +endif() +include("@_edge_template_dir@/CheckNative.cmake") +_edgellm_check_json_headers("@_edge_json_include@" "@_edge_source@/3rdParty/nlohmannJson") +function(run) + execute_process(COMMAND ${ARGV} COMMAND_ERROR_IS_FATAL ANY) +endfunction() +set(_pip_options --isolated install --no-user) +if(NOT "@TRTMC_EDGELLM_WHEELHOUSE@" STREQUAL "") + list(APPEND _pip_options --no-index --find-links "@TRTMC_EDGELLM_WHEELHOUSE@") +endif() +function(create_environment destination) + execute_process(COMMAND "@Python3_EXECUTABLE@" -I -c "import ensurepip" + RESULT_VARIABLE _has_ensurepip OUTPUT_QUIET ERROR_QUIET) + if(_has_ensurepip EQUAL 0) + run("@Python3_EXECUTABLE@" -I -m venv --copies "${destination}") + else() + run("@Python3_EXECUTABLE@" -I -m virtualenv --copies --no-download --no-periodic-update + "${destination}") + endif() + # Pin even when ensurepip seeds a newer version; offline installs stay offline. + run("${destination}/bin/python" -I -m pip ${_pip_options} "pip==26.2.1") +endfunction() +create_environment("@_edge_prefix@/libexec/trtmc-edge-llm") +file(GLOB _trt_wheels "@TRTMC_EDGELLM_TRT_ROOT@/python/tensorrt-@_edge_trt_version@-cp@Python3_VERSION_MAJOR@@Python3_VERSION_MINOR@-none-linux_@CMAKE_SYSTEM_PROCESSOR@.whl") +list(LENGTH _trt_wheels _wheel_count) +if(NOT _wheel_count EQUAL 1) + message(FATAL_ERROR "Expected the TensorRT @_edge_trt_version@ SDK wheel matching the native Python ABI") +endif() +# CUDA 12 CuPy needs NumPy 1.x, while the installed Edge SDK needs NumPy 2.x. +# Keep its AOT-only environment outside the installed SDK and check both. +set(_edge_kernel_python "@_edge_python@") +if("@CUDAToolkit_VERSION_MAJOR@" STREQUAL "12") + set(_edge_cuda_python_version 12.9.7) + set(_edge_kernel_python "@_edge_root@/kernel-python/bin/python") + create_environment("@_edge_root@/kernel-python") + run("${_edge_kernel_python}" -I -m pip ${_pip_options} + --report "@_edge_root@/kernel-pip-report.json" + numpy==1.26.4 cupy-cuda12x==12.3.0 cuda-python==12.9.7 + "nvidia-cutlass-dsl[cu12]==4.7.0") + run("${_edge_kernel_python}" -I -m pip --isolated check) + set(_edge_kernel_dependencies "") +elseif("@CUDAToolkit_VERSION_MAJOR@" STREQUAL "13") + set(_edge_cuda_python_version 13.3.1) + set(_edge_kernel_dependencies "nvidia-cutlass-dsl[cu13]==4.7.0" cupy-cuda13x==13.6.0) +else() + message(FATAL_ERROR "Pinned EdgeLLM supports CUDA major 12 or 13") +endif() +run("@_edge_python@" -I -m pip ${_pip_options} --report "@_edge_prefix@/pip-report.json" + ${_trt_wheels} numpy==2.2.6 transformers==5.14.1 jinja2==3.1.6 + scikit-build-core==0.11.6 wheel==0.45.1 cmake==3.31.10 ninja==1.13.0 + "cuda-python==${_edge_cuda_python_version}" ${_edge_kernel_dependencies}) +run("@_edge_python@" -I -m pip ${_pip_options} --no-deps --no-build-isolation "@_edge_source@") +if("@TRTMC_EDGELLM_ONNX@") + # Original exporter dependencies, isolated from the caller environment. Export + # is CPU-side; native TensorRT compilation still runs on the inference GPU. + set(_torch_options ${_pip_options}) + if("@TRTMC_EDGELLM_WHEELHOUSE@" STREQUAL "") + list(APPEND _torch_options --index-url https://download.pytorch.org/whl/cpu) + endif() + run("@_edge_python@" -I -m pip ${_torch_options} "torch==2.13.0") + run("@_edge_python@" -I -m pip ${_pip_options} "tensorrt-edgellm[export]==@_edge_version@") +endif() +run("@_edge_python@" -I -m pip --isolated check) +run("@_edge_python@" -I -c "__import__('sys').exit(0 if __import__('tensorrt').__version__ == '@_edge_trt_version@' else 'TensorRT Python ' + __import__('tensorrt').__version__ + ' differs from native SDK @_edge_trt_version@')") +set(ENV{PATH} "@CUDAToolkit_BIN_DIR@:$ENV{PATH}") +run("${_edge_kernel_python}" -I "@_edge_source@/kernelSrcs/build_cutedsl.py" + --gpu_arch "sm_@TRTMC_EDGELLM_CUDA_ARCHITECTURE@" --arch "@CMAKE_SYSTEM_PROCESSOR@" + --kernels "@_edge_cute_cli_groups@" --cuda-version "@CUDAToolkit_VERSION_MAJOR@.@CUDAToolkit_VERSION_MINOR@" --jobs "@TRTMC_EDGELLM_JOBS@") diff --git a/cmake/edge_llm/README.md b/cmake/edge_llm/README.md new file mode 100644 index 0000000000..067ac370f5 --- /dev/null +++ b/cmake/edge_llm/README.md @@ -0,0 +1,97 @@ +# Pinned native Edge-LLM package + +Edge-LLM is optional. The default `TRTMC_ENABLE_EDGELLM=OFF` neither downloads +nor builds it. Enable it once while installing Model Connect; ordinary model +builds only use the installed package and never fetch or install dependencies. +Cross compilation is rejected. Configure and build on the inference GPU host. + +```bash +cmake -S . -B build \ + -DTRTMC_ENABLE_EDGELLM=ON \ + -DTRTMC_EDGELLM_CUDA_ARCHITECTURE=80 \ + -DTRTMC_EDGELLM_TRT_ROOT="$TRT_ROOT" \ + -DCUDAToolkit_ROOT="$CUDA_ROOT" \ + -DCMAKE_CUDA_COMPILER="$CUDA_ROOT/bin/nvcc" \ + -DCMAKE_INSTALL_PREFIX="$PWD/install" +cmake --build build --parallel 8 +cmake --install build +export CMAKE_PREFIX_PATH="$PWD/install${CMAKE_PREFIX_PATH:+:$CMAKE_PREFIX_PATH}" +``` + +The regular project dependencies remain required, including nlohmann_json +**3.12.0 with the exact pinned upstream headers** when Edge is enabled. A +development snapshot can retain that version label but change parser layouts; +mixing it with the static SDK causes undefined behavior. The package checks +header content against its vendored dependency (single or multiple headers). +If rejected, install `3rdParty/nlohmannJson` from the pinned Edge checkout into +a separate prefix and configure with that installation’s `nlohmann_json_DIR`. The Python +interpreter needs `ensurepip` or an already installed `virtualenv` bootstrapper. +The native CUDA SDK must include NVCC, NVRTC, cuRAND headers and driver link +libraries; the TensorRT SDK must contain its matching CPython wheel. + +The provider first uses `find_package(EdgeLLM 0.10.1 EXACT CONFIG)`. If absent, +CMake `ExternalProject` clones the public NVIDIA TensorRT-Edge-LLM repository at +`e8b29522938901f6df19ebeedd4b69bc8edbcd97` (v0.10.1), initializes the pinned +submodules, builds the native core/plugin and FMHA/GDN CuTe archives, and installs +an isolated direct-builder Python environment. It does not modify the caller +Python environment. Downloads happen only during this +explicit dependency build. `TRTMC_EDGELLM_WHEELHOUSE` selects a complete offline +Python wheelhouse; `TRTMC_EDGELLM_GIT_MIRROR` optionally supplies a local Git +mirror, still checked out at the immutable upstream commit. + +Upstream 0.10.1 does not export a CMake SDK package, so these compact templates +supply that installation boundary. `EdgeLLM::Core` exposes the installed static +core, headers, CuTe archive and native dependencies. Consumers requiring CUDA +device linking enable separable compilation and device-symbol resolution. +`EdgeLLM::Plugin` identifies the plugin DSO; adapters load it, rather than linking +it twice. `EdgeLLM_PYTHON_EXECUTABLE` and `EdgeLLM_BUILDER_LAUNCHER` expose the +isolated upstream `experimental.builder.cli.main` API. + +`share/trtmc/edge-llm.json` records the pin, native architecture, CUDA/TensorRT +versions and prefix-relative Python/plugin paths. The manifest is written only +after successful installation. Build-tree package files live under +`build/_deps/edgellm/install`; `cmake --install` copies the package into the final +prefix. Package discovery rejects mismatched native CPU/GPU and SDK versions. +Model support and routing policies belong exclusively to the model families. + +Set `TRTMC_EDGELLM_ALL_KERNELS=ON` to provision all upstream operator groups +supported by the native GPU. Set `TRTMC_EDGELLM_ONNX=ON` to additionally install +the original Python exporter (including its pinned CPU PyTorch dependencies) +and original C++ `llm_build` executable as `bin/edgellm-onnx-build`. Families invoke +the exporter using the installed Python and obtain the native builder path from +`onnx_builder` in the manifest. These options describe SDK capabilities, not +qualified model support. Reusing an installed package that lacks a requested +capability is an error; no dependency installation occurs during model builds. +The CUDA and TensorRT shared libraries must remain available to the executable. +When building models, select the same native CUDA toolkit with CUDACXX (the +NVCC executable) or CUDAToolkit_ROOT (the SDK root); CUDA_HOME, CUDA_PATH, +and then NVCC on PATH are fallbacks. Platform admission reads this compiler's +release, not the independently versioned cuda-python binding's build toolkit. + +Run the existing runtime and family checks against this installation +(some tests require a local GPU): + +```bash +ctest --test-dir build --output-on-failure +``` + +The installed private Python environment exposes its interpreter and modules, +not build-only console/activation scripts containing build-tree paths. Invoke +the exported interpreter with isolated module execution, or use the installed +prefix-relative builder launcher. CMake/Ninja entrypoints remain in the +dependency build environment for reprovisioning. Install into a clean prefix +when replacing an older SDK that included these private console scripts. +Preparation verifies the actual Git checkout against the official pin before +installing dependencies, including on CMake versions with older disconnected +update behavior. + +CUDA 12 provisioning requires Python 3.10-3.12. Its pinned CuPy 12.3 kernel +compiler uses a separate build-only environment with NumPy 1.26.4; the installed +SDK and ONNX exporter use NumPy 2.2.6. Both environments must pass pip check. +CUDA 13 uses the SDK environment for kernel compilation. Bootstrap pip is pinned +to 26.2.1 and cuda-python to 12.9.7 (CUDA 12) or 13.3.1 (CUDA 13); these bindings +do not determine the native toolkit identity. Offline wheelhouses must include +these exact pins and the dependencies for both environments. The kernel-only +environment and its dependency report remain under the dependency build root. +CUDA 13 provisioning requires Python 3.10-3.13 because of the pinned NumPy wheel support. +The selected Python wheel and imported TensorRT version must match the exact native SDK header/library version; an ABI-compatible wheel from another release is rejected. diff --git a/core/builder/tensorrt_model_connect/build.py b/core/builder/tensorrt_model_connect/build.py index f6fad3a61e..5724df53e5 100644 --- a/core/builder/tensorrt_model_connect/build.py +++ b/core/builder/tensorrt_model_connect/build.py @@ -7,8 +7,13 @@ import hashlib import importlib +import os +import platform import re +import shlex import sys +import shutil +import subprocess from dataclasses import dataclass from pathlib import Path from types import ModuleType @@ -72,6 +77,93 @@ def __post_init__(self) -> None: raise ValueError("graph_transform must be callable when provided") +def subprocess_environment( + overrides: dict[str, str], *, prepend_paths: dict[str, str] | None = None +) -> dict[str, str]: + """Copy the parent environment for one child without mutating process state. + + Callers own explicit tool settings; this helper only merges values and + prepends search paths using the executing platform's path separator. + """ + environment = os.environ.copy() + environment.update(overrides) + for name, value in (prepend_paths or {}).items(): + previous = environment.get(name) + environment[name] = value + (os.pathsep + previous if previous else "") + return environment + + +def cmake_prefixes() -> list[Path]: + """Return explicit standard CMake prefixes followed by the Python prefix.""" + prefixes = [ + Path(value) for value in os.environ.get("CMAKE_PREFIX_PATH", "").split(os.pathsep) if value + ] + return [*prefixes, Path(sys.prefix)] + + +def _cuda_toolkit_version() -> str: + """Identify the selected native compiler, not cuda-python's build toolkit.""" + compiler = os.environ.get("CUDACXX") + try: + command = shlex.split(compiler) if compiler else [] + except ValueError as error: + raise RuntimeError(f"Invalid CUDACXX command: {error}") from error + if not compiler: + root = next( + (os.environ[key] for key in ("CUDAToolkit_ROOT", "CUDA_HOME", "CUDA_PATH") + if os.environ.get(key)), None + ) + compiler = str(Path(root) / "bin" / "nvcc") if root else shutil.which("nvcc") + command = [compiler] if compiler else [] + if not command: + raise RuntimeError("CUDA toolkit not found; set CUDAToolkit_ROOT or CUDACXX") + try: + result = subprocess.run( + [*command, "--version"], check=True, capture_output=True, text=True, timeout=10, + ) + except (OSError, subprocess.SubprocessError) as error: + raise RuntimeError(f"Cannot query CUDA toolkit from {compiler}: {error}") from error + version = re.search(r"release\s+(\d+\.\d+)", result.stdout) + if version is None: + raise RuntimeError(f"Cannot identify CUDA toolkit from {compiler} --version") + return version.group(1) + + +def detect_local_platform() -> dict: + """Return executing GPU and native SDK identity without selecting a model. + + Returns: + OS/release, CPU architecture, GPU SM, CUDA and TensorRT versions. + + Raises: + ImportError: Native SDK Python bindings are unavailable. + RuntimeError: CUDA cannot identify the executing device. + """ + import tensorrt as trt + from cuda.bindings import runtime + + def checked(result): + if int(result[0]) != 0: + raise RuntimeError(f"CUDA device discovery failed: {result[0]}") + return result[1] + + device = checked(runtime.cudaGetDevice()) + gpu = checked(runtime.cudaGetDeviceProperties(device)) + cuda_version = _cuda_toolkit_version() + try: + release = platform.freedesktop_os_release() if sys.platform == "linux" else {} + except OSError: + release = {} + return { + "os": sys.platform, + "os_version": release.get("VERSION_ID", platform.release()), + "arch": platform.machine(), + "sm": gpu.major * 10 + gpu.minor, + "cuda_version": cuda_version, + "tensorrt_version": trt.__version__, + } + + def _validate_id(field: str, value: object) -> str: if not isinstance(value, str) or _ID.fullmatch(value) is None: raise ValueError( diff --git a/core/builder/tests/test_build.py b/core/builder/tests/test_build.py index 5eff69de1b..ff0b0c1a0d 100644 --- a/core/builder/tests/test_build.py +++ b/core/builder/tests/test_build.py @@ -287,3 +287,164 @@ def abort(self) -> None: with pytest.raises(OSError, match="publish failed"): build_core.build(_request(tmp_path)) assert events == ["finish", "abort"] + + +@pytest.mark.parametrize("value", ["", "/one", "/one:/two", ":/one::/two:"]) +def test_cmake_prefixes_preserve_standard_search_order(monkeypatch, value): + monkeypatch.setenv("CMAKE_PREFIX_PATH", value) + monkeypatch.setattr(build_core.sys, "prefix", "/python") + expected = [Path(item) for item in value.split(build_core.os.pathsep) if item] + assert build_core.cmake_prefixes() == [*expected, Path("/python")] + + +def test_cmake_prefixes_without_environment_use_python_prefix(monkeypatch): + monkeypatch.delenv("CMAKE_PREFIX_PATH", raising=False) + monkeypatch.setattr(build_core.sys, "prefix", "/python") + assert build_core.cmake_prefixes() == [Path("/python")] + # Constructing explicit child-tool settings must not change the caller's + # package search order or mutate an inherited search path. + monkeypatch.setenv("TEST_TOOL_SEARCH_PATH", "/original") + monkeypatch.delenv("TEST_TOOL_NEW_PATH", raising=False) + child = build_core.subprocess_environment( + {"CMAKE_PREFIX_PATH": "/child"}, + prepend_paths={"TEST_TOOL_SEARCH_PATH": "/first", "TEST_TOOL_NEW_PATH": "/new"}, + ) + assert child["CMAKE_PREFIX_PATH"] == "/child" + assert child["TEST_TOOL_SEARCH_PATH"] == "/first" + build_core.os.pathsep + "/original" + assert child["TEST_TOOL_NEW_PATH"] == "/new" + assert build_core.cmake_prefixes() == [Path("/python")] + assert build_core.os.environ["TEST_TOOL_SEARCH_PATH"] == "/original" + assert "TEST_TOOL_NEW_PATH" not in build_core.os.environ + + +@pytest.fixture +def native_platform_bindings(monkeypatch): + from unittest.mock import Mock + + runtime = SimpleNamespace( + cudaGetDevice=Mock(return_value=(0, 3)), + cudaGetDeviceProperties=Mock(return_value=(0, SimpleNamespace(major=8, minor=6))), + cudaRuntimeGetVersion=Mock(return_value=(0, 13000)), + ) + monkeypatch.setitem(sys.modules, "tensorrt", SimpleNamespace(__version__="11.1.0.106")) + monkeypatch.setitem(sys.modules, "cuda.bindings", SimpleNamespace(runtime=runtime)) + monkeypatch.setattr(build_core.sys, "platform", "linux") + monkeypatch.setattr(build_core.platform, "machine", lambda: "x86_64") + monkeypatch.setattr( + build_core.platform, "freedesktop_os_release", lambda: {"VERSION_ID": "24.04"} + ) + monkeypatch.setattr(build_core.platform, "release", lambda: "fallback-release") + monkeypatch.setattr(build_core, "_cuda_toolkit_version", lambda: "13.3") + return runtime + + +@pytest.mark.parametrize("release_available", [True, False]) +def test_native_platform_uses_executing_cuda_device_and_full_sdk( + native_platform_bindings, monkeypatch, release_available +): + if not release_available: + from unittest.mock import Mock + + monkeypatch.setattr( + build_core.platform, "freedesktop_os_release", Mock(side_effect=OSError("missing")) + ) + assert build_core.detect_local_platform() == { + "os": "linux", + "os_version": "24.04" if release_available else "fallback-release", + "arch": "x86_64", + "sm": 86, + "cuda_version": "13.3", + "tensorrt_version": "11.1.0.106", + } + native_platform_bindings.cudaGetDevice.assert_called_once_with() + native_platform_bindings.cudaGetDeviceProperties.assert_called_once_with(3) + native_platform_bindings.cudaRuntimeGetVersion.assert_not_called() + + +@pytest.mark.parametrize( + "failing", ["cudaGetDevice", "cudaGetDeviceProperties"] +) +def test_native_platform_propagates_cuda_discovery_failure(native_platform_bindings, failing): + getattr(native_platform_bindings, failing).return_value = (35,) + with pytest.raises(RuntimeError, match="CUDA device discovery failed: 35"): + build_core.detect_local_platform() + + +def test_native_platform_retains_nonlinux_identity(native_platform_bindings, monkeypatch): + monkeypatch.setattr(build_core.sys, "platform", "win32") + result = build_core.detect_local_platform() + assert result["os"] == "win32" + assert result["os_version"] == "fallback-release" + +@pytest.mark.parametrize("source", ["CUDACXX", "CUDAToolkit_ROOT", "CUDA_HOME", "CUDA_PATH", "PATH"]) +def test_cuda_toolkit_version_uses_selected_compiler(monkeypatch, source): + from unittest.mock import Mock + + for name in ("CUDACXX", "CUDAToolkit_ROOT", "CUDA_HOME", "CUDA_PATH"): + monkeypatch.delenv(name, raising=False) + compiler = "/selected/bin/nvcc" + monkeypatch.setattr(build_core.shutil, "which", lambda _: compiler) + if source != "PATH": + monkeypatch.setenv(source, compiler if source == "CUDACXX" else "/selected") + run = Mock(return_value=SimpleNamespace(stdout="Cuda compilation tools, release 13.3, V13.3.1")) + monkeypatch.setattr(build_core.subprocess, "run", run) + assert build_core._cuda_toolkit_version() == "13.3" + run.assert_called_once_with( + [compiler, "--version"], check=True, capture_output=True, text=True, timeout=10, + ) + + +def test_cuda_toolkit_version_does_not_guess_when_missing(monkeypatch): + for name in ("CUDACXX", "CUDAToolkit_ROOT", "CUDA_HOME", "CUDA_PATH"): + monkeypatch.delenv(name, raising=False) + monkeypatch.setattr(build_core.shutil, "which", lambda _: None) + with pytest.raises(RuntimeError, match="CUDA toolkit not found"): + build_core._cuda_toolkit_version() + + +def test_cuda_toolkit_version_rejects_unrecognized_output(monkeypatch): + monkeypatch.setenv("CUDACXX", "/selected/nvcc") + monkeypatch.setattr(build_core.subprocess, "run", lambda *_, **__: SimpleNamespace(stdout="")) + with pytest.raises(RuntimeError, match="Cannot identify CUDA toolkit"): + build_core._cuda_toolkit_version() + +@pytest.mark.parametrize("source, value, expected", [ + ("CUDACXX", '"/tool kit/nvcc" --allow-unsupported-compiler', + ["/tool kit/nvcc", "--allow-unsupported-compiler"]), + ("CUDAToolkit_ROOT", "/tool kit", ["/tool kit/bin/nvcc"]), +]) +def test_cuda_toolkit_compiler_arguments_and_spaces(monkeypatch, source, value, expected): + from unittest.mock import Mock + + for name in ("CUDACXX", "CUDAToolkit_ROOT", "CUDA_HOME", "CUDA_PATH"): + monkeypatch.delenv(name, raising=False) + monkeypatch.setenv(source, value) + run = Mock(return_value=SimpleNamespace(stdout="release 13.3, V13.3.1")) + monkeypatch.setattr(build_core.subprocess, "run", run) + assert build_core._cuda_toolkit_version() == "13.3" + run.assert_called_once_with( + [*expected, "--version"], check=True, capture_output=True, text=True, timeout=10, + ) + + +@pytest.mark.parametrize("failure", [ + FileNotFoundError("compiler missing"), + build_core.subprocess.CalledProcessError(1, ["nvcc", "--version"]), + build_core.subprocess.TimeoutExpired(["nvcc", "--version"], 10), +]) +def test_cuda_toolkit_compiler_failures_preserve_cause(monkeypatch, failure): + from unittest.mock import Mock + + monkeypatch.setenv("CUDACXX", "/selected/nvcc") + monkeypatch.setattr(build_core.subprocess, "run", Mock(side_effect=failure)) + with pytest.raises(RuntimeError, match="Cannot query CUDA toolkit") as caught: + build_core._cuda_toolkit_version() + assert caught.value.__cause__ is failure + + +def test_cuda_toolkit_malformed_compiler_command_preserves_cause(monkeypatch): + monkeypatch.setenv("CUDACXX", '"unclosed compiler path') + monkeypatch.setattr(build_core.subprocess, "run", lambda *_a, **_k: pytest.fail("compiler ran")) + with pytest.raises(RuntimeError, match="Invalid CUDACXX command") as caught: + build_core._cuda_toolkit_version() + assert isinstance(caught.value.__cause__, ValueError) diff --git a/core/runtime/bundle/bundle_format.cpp b/core/runtime/bundle/bundle_format.cpp index eb11f20674..2c1661f088 100644 --- a/core/runtime/bundle/bundle_format.cpp +++ b/core/runtime/bundle/bundle_format.cpp @@ -6,6 +6,7 @@ #include "runtime/bundle/bundle_format.h" #include +#include #include #include #include @@ -215,6 +216,32 @@ std::vector BundleReader::read_section(std::string_view name) const { return data; } +void BundleReader::copy_section(std::string_view name, std::ostream& output) const { + const auto* section = find_section(name); + if (section == nullptr) + throw std::runtime_error("Bundle section not found: " + std::string(name)); + const auto offset = checked_section_file_offset(*section, data_offset_, file_size_, path_); + if (offset > static_cast(std::numeric_limits::max())) + throw std::runtime_error("Bundle section has an unsupported file offset: " + path_); + std::ifstream input(path_, std::ios::binary); + input.seekg(static_cast(offset)); + if (!input || !output) + throw std::runtime_error("Cannot copy bundle section: " + std::string(name)); + std::array buffer; + auto remaining = section->length; + while (remaining != 0) { + const auto count = + static_cast(std::min(remaining, buffer.size())); + input.read(buffer.data(), count); + if (!input) + throw std::runtime_error("Failed reading bundle section: " + std::string(name)); + output.write(buffer.data(), count); + if (!output) + throw std::runtime_error("Failed writing bundle section: " + std::string(name)); + remaining -= static_cast(count); + } +} + BundleInfo InspectBundle(const std::string& bundle_path) { return BundleReader(bundle_path).info(); } diff --git a/core/runtime/include/trtmc/bundle.h b/core/runtime/include/trtmc/bundle.h index 154d697ef1..2b32776d2e 100644 --- a/core/runtime/include/trtmc/bundle.h +++ b/core/runtime/include/trtmc/bundle.h @@ -6,6 +6,7 @@ #pragma once #include +#include #include #include #include @@ -44,6 +45,12 @@ class BundleReader { const BundleSectionInfo* find_section(std::string_view name) const noexcept; std::vector read_section(std::string_view name) const; + /// Copy a named section to an output stream using bounded working memory. + /// @param name Validated bundle section name. + /// @param output Caller-owned stream; may contain partial data on failure. + /// @throws std::runtime_error If the section is absent or input/output fails. + void copy_section(std::string_view name, std::ostream& output) const; + private: std::string path_; BundleInfo info_; diff --git a/core/runtime/tests/test_bundle_format_v1.cpp b/core/runtime/tests/test_bundle_format_v1.cpp index 0da636b49b..6c5faa420f 100644 --- a/core/runtime/tests/test_bundle_format_v1.cpp +++ b/core/runtime/tests/test_bundle_format_v1.cpp @@ -9,6 +9,7 @@ #include #include #include +#include #include #include #include @@ -52,6 +53,36 @@ bool read_throws(const std::filesystem::path& path) { } } +/// Verify bounded-chunk section boundaries, empty sections and late I/O failures. +void test_copy_section(const std::filesystem::path& directory) { + const auto path = directory / "stream.bundle"; + const std::string payload(2 * 64 * 1024 + 17, 'x'); + const std::string header = + R"({"format":1,"family":"fake","task":"text","backend":"fake","sections":{"data":{"offset":3,"length":)" + + std::to_string(payload.size()) + R"(},"empty":{"offset":0,"length":0}}})"; + write_bundle(path, header, "PRE" + payload + "POST"); + const trtmc::BundleReader reader(path.string()); + std::ostringstream output; + reader.copy_section("data", output); + check(output.str() == payload, "stream copy preserves boundaries across chunks"); + reader.copy_section("empty", output); + check(output.str() == payload, "empty section appends nothing"); + auto fails = [&](const char* section, std::ostream& destination) { + try { + reader.copy_section(section, destination); + return false; + } catch (const std::runtime_error&) { + return true; + } + }; + check(fails("missing", output), "stream copy rejects missing sections"); + std::ostringstream broken; + broken.setstate(std::ios::badbit); + check(fails("data", broken), "stream copy reports output failure"); + std::filesystem::resize_file(path, 16 + header.size() + 3 + payload.size() - 1); + check(fails("data", output), "stream copy detects truncation after validation"); +} + } // namespace int main() { @@ -103,6 +134,7 @@ int main() { "PLAN"); check(read_throws(out_of_bounds), "out of bounds section rejected"); + test_copy_section(directory); std::filesystem::remove_all(directory); std::cerr << (failures == 0 ? "ALL PASSED\n" : "SOME FAILED\n"); return failures; diff --git a/families/qwen/tests/test_e2e.py b/families/qwen/tests/test_e2e.py index 2c62d307ff..2f15085986 100644 --- a/families/qwen/tests/test_e2e.py +++ b/families/qwen/tests/test_e2e.py @@ -104,12 +104,13 @@ def _required_environment(tp_size: int): def _checkpoint(manifest: dict) -> Path: - from huggingface_hub import snapshot_download + from huggingface_hub import constants, snapshot_download path = Path( snapshot_download( repo_id=manifest["hf_id"], revision=manifest.get("hf_revision"), + local_files_only=constants.HF_HUB_OFFLINE, ) ) assert (path / "config.json").is_file(), path diff --git a/tools/tests/test_architecture.py b/tools/tests/test_architecture.py index 5986c7293e..c95a86f7af 100644 --- a/tools/tests/test_architecture.py +++ b/tools/tests/test_architecture.py @@ -631,7 +631,15 @@ def test_shared_python_and_native_trees_are_closed_minimal_sets() -> None: "qualification_tests/benchmark_qualification/performance/tests/test_structured_output_contracts.py", "qualification_tests/benchmark_qualification/performance/tests/test_timing_contracts.py", } - expected_cmake = {"cmake/trtmcConfig.cmake.in"} + expected_cmake = { + "cmake/trtmcConfig.cmake.in", + "cmake/edge_llm/EdgeLLM.cmake", + "cmake/edge_llm/CheckNative.cmake", + "cmake/edge_llm/EdgeLLMConfig.cmake.in", + "cmake/edge_llm/Install.cmake.in", + "cmake/edge_llm/Prepare.cmake.in", + "cmake/edge_llm/README.md", + } expected_third_party = { "third_party/stb/stb_image.h", "third_party/stb/stb_image_resize2.h", diff --git a/website/docs/architecture/build-pipeline.md b/website/docs/architecture/build-pipeline.md index 3389c9cdfd..84ade7448b 100644 --- a/website/docs/architecture/build-pipeline.md +++ b/website/docs/architecture/build-pipeline.md @@ -41,7 +41,10 @@ explicitly reject every non-default request it receives. `families//model.py` exposes a plain `build(request, writer)` function. It reads model config and weights, constructs the TensorRT network and engines, and writes family-owned named sections. Builder inheritance and shared model -topology helpers are forbidden. +topology helpers are forbidden. A family may instead delegate a complete +network to an installed optimized runtime through a family-owned adapter. +Model-specific admission, builder mapping, runtime orchestration and validation +remain in that family; shared dependency provisioning contains no model policy. The graph-transform callback, when present, receives the live TensorRT network immediately before serialization. This is the build-time half of the explicit diff --git a/website/docs/user-guides/configure-runtime.md b/website/docs/user-guides/configure-runtime.md index 82f3f62dca..c7335bb6c5 100644 --- a/website/docs/user-guides/configure-runtime.md +++ b/website/docs/user-guides/configure-runtime.md @@ -31,3 +31,26 @@ Unsupported values fail; they are not silently ignored. See [Configuration and Backends](../features/config-and-backends.md), [Quantization](../features/quantization.md), and [Multi-Device Execution](../features/multi-device.md). + +## Optional native Edge-LLM SDK + +Provision Edge-LLM explicitly when building Model Connect, not during model +builds or inference. `TRTMC_ENABLE_EDGELLM=ON` selects the public Edge-LLM +0.10.1 snapshot at `e8b29522938901f6df19ebeedd4b69bc8edbcd97`. The default is +`OFF`. Configure and build on the inference GPU host; cross compilation is +rejected. The package must match the native CPU/GPU and CUDA/TensorRT stack. +Runtime compilation must also use the exact pinned JSON dependency headers; +the SDK rejects same-version development headers with an incompatible C++ ABI. + +- `TRTMC_EDGELLM_ALL_KERNELS=ON` requests all upstream operator groups supported + by the local GPU. +- `TRTMC_EDGELLM_ONNX=ON` also installs the original Python exporter and native + C++ ONNX engine builder. It does not select a model's build flow. +- `CMAKE_PREFIX_PATH` points builders and runtime compilation to the installed + SDK. Reuse fails explicitly if a requested capability is absent. + +Follow the repository's [pinned SDK installation instructions](https://github.com/NVIDIA/TensorRT-Model-Connect/blob/main/cmake/edge_llm/README.md) +for dependencies, native architecture selection and offline provisioning. +Each family decides whether and how to use the package. Installing the SDK +is not evidence that a model, precision, input modality or execution variant +has passed validation. Ordinary model builds never install missing SDK tools.