From a101406c436f4f18e200590ea768c4c77b3509eb Mon Sep 17 00:00:00 2001 From: Prabhsimran Singh Date: Thu, 1 Oct 2026 18:12:58 +0000 Subject: [PATCH 01/15] ci: build release archives in github CI - build/ publish release archives on version tags, a daily nightly that skips an unchanged main, and PRs that change release packaging - target x86-64-v3 with GGML_NATIVE=OFF, reject AVX-512/AMX/AVX-VNNI code, and run CPU archive on emulated Haswell CPU - package Linux on a glibc 2.31 baseline and load bundled libraries through DT_RPATH - ship ITN/TN on Linux and macOS: build_itn_deps.sh STATIC=1 links OpenFST, Sparrowhawk, protobuf, and RE2 privately - add macOS and Windows packagers, smoke tests for TN/ITN, and the release guide --- .github/workflows/release.yml | 668 ++++++++++++++++++++++++++ THIRD_PARTY_NOTICES.md | 21 + docker/Dockerfile.release-linux | 241 ++++++++++ docs/build.md | 9 + docs/development/README.md | 2 + docs/development/releasing.md | 88 ++++ docs/install.md | 5 + scripts/build_itn_deps.sh | 109 +++-- scripts/build_sentencepiece_static.sh | 9 +- scripts/release/check_release.py | 192 ++++++++ scripts/release/package-linux.sh | 368 ++++++++++++++ scripts/release/package-macos.sh | 170 +++++++ scripts/windows/build.ps1 | 7 +- scripts/windows/package-release.ps1 | 109 +++++ src/common/CMakeLists.txt | 88 ++-- tests/ci/model_smoke.py | 55 +++ 16 files changed, 2081 insertions(+), 60 deletions(-) create mode 100644 .github/workflows/release.yml create mode 100644 docker/Dockerfile.release-linux create mode 100644 docs/development/releasing.md create mode 100755 scripts/release/check_release.py create mode 100755 scripts/release/package-linux.sh create mode 100755 scripts/release/package-macos.sh create mode 100644 scripts/windows/package-release.ps1 diff --git a/.github/workflows/release.yml b/.github/workflows/release.yml new file mode 100644 index 0000000..5a4a5bf --- /dev/null +++ b/.github/workflows/release.yml @@ -0,0 +1,668 @@ +name: Release + +# Builds the portable release archives that scripts/install.sh and +# scripts/install.ps1 download, checks them, and publishes them. +# +# push of a vX.Y.Z tag -> draft release vX.Y.Z; a maintainer reviews and publishes it +# nightly schedule -> prerelease "nightly", rebuilt in place when main has moved +# workflow_dispatch -> dry run (build and check only) or a nightly +# pull request -> dry run when it changes release packaging (pull-request/N +# branches pushed by copy-pr-bot, as in gpu.yml) +# +# x86_64 binaries target x86-64-v3 (AVX2): the archives are built with +# GGML_NATIVE=OFF, scanned for AVX-512/AMX instructions, and the Linux CPU +# archive runs under an emulated Haswell CPU before anything is published. +# +# Linux and macOS archives include text normalization (ITN/TN). The grammars are +# not built here: they are taken from the latest release, used by the smoke +# tests, and published again with every release. + +on: + push: + tags: ["v*"] + branches: ["pull-request/[0-9]+"] + schedule: + - cron: "0 6 * * *" + workflow_dispatch: + inputs: + channel: + description: "dry-run builds and checks without publishing" + type: choice + options: [dry-run, nightly] + default: dry-run + +concurrency: + group: release-${{ github.ref }} + cancel-in-progress: ${{ startsWith(github.ref, 'refs/heads/pull-request/') }} + +permissions: + contents: read + +env: + NEMO_SPEECH_MODEL_DIR: ${{ github.workspace }}/.ci-models + JFK_AUDIO: test_files/asr/wav/test/jfk.wav + +jobs: + prepare: + name: Resolve version + if: github.repository == 'NVIDIA/NeMo-Speech.cpp' + runs-on: ubuntu-latest + outputs: + version: ${{ steps.version.outputs.version }} + channel: ${{ steps.version.outputs.channel }} + build: ${{ steps.version.outputs.build }} + steps: + - name: Checkout repository + uses: actions/checkout@df4cb1c069e1874edd31b4311f1884172cec0e10 # v6.0.3 + with: + persist-credentials: false + fetch-depth: 0 + + - name: Resolve version and channel + id: version + env: + EVENT: ${{ github.event_name }} + INPUT_CHANNEL: ${{ inputs.channel }} + RELEASE_PATHS: >- + ^(\.github/workflows/release\.yml|docker/Dockerfile\.release-linux|scripts/build_itn_deps\.sh|scripts/build_sentencepiece_static\.sh|scripts/release/|scripts/windows/(build|package-release)\.ps1|tests/ci/model_smoke\.py) + run: | + set -euo pipefail + declared="$(sed -n 's/^NEMO_SPEECH_VERSION:[[:space:]]*//p' VERSION)" + build=true + case "$EVENT:$GITHUB_REF_TYPE" in + push:branch) + # Dry run only when the pull request changes release packaging. A + # paths filter cannot tell: copy-pr-bot mirrors commits that + # already exist, so the push lists no changed files. + channel=dry-run + version="$declared" + git fetch --no-tags --quiet origin main + if ! git diff --name-only FETCH_HEAD...HEAD | grep -qE "$RELEASE_PATHS"; then + build=false + echo "::notice::no release packaging changes; skipping" + fi + ;; + push:tag) + if [ "$GITHUB_REF_NAME" != "v$declared" ]; then + echo "::error::tag $GITHUB_REF_NAME does not match VERSION ($declared)" + exit 1 + fi + channel=release + version="$declared" + ;; + schedule:*) channel=nightly; version=nightly ;; + *) + channel="$INPUT_CHANNEL" + if [ "$channel" = nightly ]; then version=nightly; else version="$declared"; fi + ;; + esac + if [ "$EVENT" = schedule ]; then + # The nightly tag points at the commit it was built from; the + # peeled entry, when present, sorts last. + published="$(git ls-remote origin refs/tags/nightly 'refs/tags/nightly^{}' | + tail -n 1 | cut -f 1)" + if [ "$published" = "$GITHUB_SHA" ]; then + build=false + echo "::notice::nightly is already built from $GITHUB_SHA; skipping" + fi + fi + echo "channel=$channel" >> "$GITHUB_OUTPUT" + echo "version=$version" >> "$GITHUB_OUTPUT" + echo "build=$build" >> "$GITHUB_OUTPUT" + echo "Building $version ($channel): $build" + + grammars: + name: Text-normalization grammars + needs: prepare + if: needs.prepare.outputs.build == 'true' + runs-on: ubuntu-24.04 + steps: + - name: Download from the latest release + run: | + set -euo pipefail + mkdir -p grammars + for name in itn_configs tn_configs; do + curl -fsSL --retry 3 -o "grammars/$name.tar.bz2" \ + "https://github.com/$GITHUB_REPOSITORY/releases/latest/download/$name.tar.bz2" + done + + - name: Upload grammars + uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1 + with: + name: grammars + path: grammars/* + if-no-files-found: error + + linux: + name: Linux ${{ matrix.arch }} ${{ matrix.artifact_backend }} + needs: [prepare, grammars] + if: needs.prepare.outputs.build == 'true' + runs-on: ${{ matrix.runner }} + timeout-minutes: 360 + strategy: + fail-fast: false + matrix: + include: + - { arch: x86_64, runner: ubuntu-24.04, platform: linux/amd64, backend: cpu, artifact_backend: cpu, target: artifact } + - { arch: x86_64, runner: ubuntu-24.04, platform: linux/amd64, backend: vulkan, artifact_backend: vulkan, target: artifact } + - arch: x86_64 + runner: ${{ vars.RELEASE_LINUX_X64_CUDA_RUNNER || 'ubuntu-24.04' }} + platform: linux/amd64 + backend: cuda + artifact_backend: cuda + target: cuda-artifact + cuda_image: nvcr.io/nvidia/cuda:12.8.1-devel-ubuntu24.04 + cuda_arch: "75-real;80-virtual;86-real;89-real;120a-real" + - { arch: aarch64, runner: ubuntu-24.04-arm, platform: linux/arm64, backend: cpu, artifact_backend: cpu, target: artifact } + - { arch: aarch64, runner: ubuntu-24.04-arm, platform: linux/arm64, backend: vulkan, artifact_backend: vulkan, target: artifact } + - arch: aarch64 + runner: ${{ vars.RELEASE_LINUX_ARM64_CUDA_RUNNER || 'ubuntu-24.04-arm' }} + platform: linux/arm64 + backend: cuda + artifact_backend: cuda12 + target: cuda-artifact + cuda_image: nvcr.io/nvidia/cuda:12.8.1-devel-ubuntu24.04 + cuda_arch: "87-real;87-virtual" + - arch: aarch64 + runner: ${{ vars.RELEASE_LINUX_ARM64_CUDA_RUNNER || 'ubuntu-24.04-arm' }} + platform: linux/arm64 + backend: cuda + artifact_backend: cuda13 + target: cuda-artifact + cuda_image: nvcr.io/nvidia/cuda:13.0.0-devel-ubuntu24.04 + cuda_arch: "110a-real;121a-real;110-virtual" + steps: + - name: Checkout repository + uses: actions/checkout@df4cb1c069e1874edd31b4311f1884172cec0e10 # v6.0.3 + with: + persist-credentials: false + + - name: Initialize submodules + run: | + git submodule update --init --depth 1 llama.cpp third_party/cpp-httplib + + - name: Free disk space + # Make room for the CUDA image and build tree. + if: matrix.backend == 'cuda' + run: | + sudo rm -rf /usr/share/dotnet /usr/local/lib/android /opt/ghc /opt/hostedtoolcache/CodeQL + docker system prune -af + df -h / + + - name: Build and package + env: + PLATFORM: ${{ matrix.platform }} + TARGET: ${{ matrix.target }} + BACKEND: ${{ matrix.backend }} + ARTIFACT_BACKEND: ${{ matrix.artifact_backend }} + CUDA_IMAGE: ${{ matrix.cuda_image }} + CUDA_ARCH: ${{ matrix.cuda_arch }} + RELEASE_VERSION: ${{ needs.prepare.outputs.version }} + run: | + set -euo pipefail + args=(--build-arg "BACKEND=$BACKEND" --build-arg "ARTIFACT_BACKEND=$ARTIFACT_BACKEND" + --build-arg "RELEASE_VERSION=$RELEASE_VERSION" --build-arg "JOBS=$(nproc)") + if [ "$BACKEND" = cuda ]; then + args+=(--build-arg "CUDA_IMAGE=$CUDA_IMAGE" --build-arg "CUDA_ARCH=$CUDA_ARCH") + fi + docker buildx build --platform "$PLATFORM" -f docker/Dockerfile.release-linux \ + --target "$TARGET" "${args[@]}" \ + --output type=local,dest=release-artifacts . + test "$(find release-artifacts -maxdepth 1 -name '*.tar.gz' | wc -l)" -eq 1 + (cd release-artifacts && sha256sum --check ./*.sha256) + + - name: Cache models + if: matrix.backend == 'cpu' + uses: actions/cache@55cc8345863c7cc4c66a329aec7e433d2d1c52a9 # v6.1.0 + with: + path: ${{ env.NEMO_SPEECH_MODEL_DIR }} + key: models-${{ hashFiles('models/index.json') }} + + - name: Download grammars + if: matrix.backend == 'cpu' + uses: actions/download-artifact@3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c # v8.0.1 + with: + name: grammars + path: ${{ runner.temp }}/grammars + + - name: Smoke test + # CPU archives run the model smoke test; Vulkan archives must start (the + # hosted runner has no GPU); CUDA archives are tested in gpu-smoke. + if: matrix.backend != 'cuda' + env: + BACKEND: ${{ matrix.backend }} + run: | + set -euo pipefail + mkdir -p "$RUNNER_TEMP/extract" + tar -xzf release-artifacts/*.tar.gz -C "$RUNNER_TEMP/extract" + cli="$(echo "$RUNNER_TEMP"/extract/*/bin/nemo-speech)" + if [ "$BACKEND" = cpu ]; then + grammars="$RUNNER_TEMP/grammars" + tar -xjf "$grammars/itn_configs.tar.bz2" -C "$grammars" + tar -xjf "$grammars/tn_configs.tar.bz2" -C "$grammars" + python3 tests/ci/model_smoke.py --binary "$cli" --backend cpu --audio "$JFK_AUDIO" \ + --itn-model-dir "$grammars/itn_configs/en" --tn-model-dir "$grammars/tn_configs" + else + sudo apt-get update && sudo apt-get install -y --no-install-recommends libvulkan1 + "$cli" --version + fi + + - name: Upload archive + uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1 + with: + name: release-linux-${{ matrix.arch }}-${{ matrix.artifact_backend }} + path: release-artifacts/* + if-no-files-found: error + + macos: + name: macOS ${{ matrix.arch }} ${{ matrix.backend }} + needs: [prepare, grammars] + if: needs.prepare.outputs.build == 'true' + runs-on: ${{ matrix.runner }} + timeout-minutes: 180 + strategy: + fail-fast: false + matrix: + include: + - { arch: aarch64, runner: macos-15, backend: cpu, preset: cpu-server, cmake_args: "-DGGML_METAL=OFF" } + - { arch: aarch64, runner: macos-15, backend: metal, preset: metal-server, cmake_args: "" } + - { arch: x86_64, runner: macos-15-intel, backend: cpu, preset: cpu-server, cmake_args: "-DGGML_METAL=OFF" } + env: + MACOSX_DEPLOYMENT_TARGET: "13.3" + steps: + - name: Checkout repository + uses: actions/checkout@df4cb1c069e1874edd31b4311f1884172cec0e10 # v6.0.3 + with: + persist-credentials: false + + - name: Initialize submodules + run: | + git submodule update --init --depth 1 llama.cpp third_party/cpp-httplib + + - name: Install dependencies + run: brew install bash ninja + + - name: Build SentencePiece + run: JOBS="$(sysctl -n hw.ncpu)" scripts/build_sentencepiece_static.sh + + - name: Build the text-normalization dependencies + run: STATIC=1 JOBS="$(sysctl -n hw.ncpu)" scripts/build_itn_deps.sh + + - name: Configure and build + env: + PRESET: ${{ matrix.preset }} + CMAKE_ARGS: ${{ matrix.cmake_args }} + run: | + set -euo pipefail + # configure.sh needs bash 4+; the runner's default bash is 3.2. + # shellcheck disable=SC2086 + "$(brew --prefix)/bin/bash" scripts/configure.sh "$PRESET" \ + -DGGML_NATIVE=OFF \ + -DNEMO_SPEECH_BUILD_GRPC=OFF -DNEMO_SPEECH_WITH_GRPC=OFF \ + -DNEMO_SPEECH_WITH_NORM=ON \ + -DCMAKE_OSX_DEPLOYMENT_TARGET="$MACOSX_DEPLOYMENT_TARGET" \ + $CMAKE_ARGS + cmake --build --preset "$PRESET" --parallel + cmake --install "build/$PRESET" --prefix "$RUNNER_TEMP/install" + + - name: Package + env: + BACKEND: ${{ matrix.backend }} + ARCH: ${{ matrix.arch }} + RELEASE_VERSION: ${{ needs.prepare.outputs.version }} + run: | + scripts/release/package-macos.sh --install-prefix "$RUNNER_TEMP/install" \ + --backend "$BACKEND" --arch "$ARCH" --version "$RELEASE_VERSION" \ + --min-macos "$MACOSX_DEPLOYMENT_TARGET" --output-dir release-artifacts + + - name: Cache models + uses: actions/cache@55cc8345863c7cc4c66a329aec7e433d2d1c52a9 # v6.1.0 + with: + path: ${{ env.NEMO_SPEECH_MODEL_DIR }} + key: models-${{ hashFiles('models/index.json') }} + + - name: Download grammars + uses: actions/download-artifact@3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c # v8.0.1 + with: + name: grammars + path: ${{ runner.temp }}/grammars + + - name: Smoke test + # The hosted VM's paravirtual GPU cannot run ggml Metal inference + # (see build.yml), so every archive runs on the CPU backend. + run: | + set -euo pipefail + mkdir -p "$RUNNER_TEMP/extract" + tar -xzf release-artifacts/*.tar.gz -C "$RUNNER_TEMP/extract" + cli="$(echo "$RUNNER_TEMP"/extract/*/bin/nemo-speech)" + grammars="$RUNNER_TEMP/grammars" + tar -xjf "$grammars/itn_configs.tar.bz2" -C "$grammars" + tar -xjf "$grammars/tn_configs.tar.bz2" -C "$grammars" + python3 tests/ci/model_smoke.py --binary "$cli" --backend cpu --audio "$JFK_AUDIO" \ + --itn-model-dir "$grammars/itn_configs/en" --tn-model-dir "$grammars/tn_configs" + + - name: Upload archive + uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1 + with: + name: release-macos-${{ matrix.arch }}-${{ matrix.backend }} + path: release-artifacts/* + if-no-files-found: error + + windows: + name: Windows x86_64 ${{ matrix.backend }} + needs: prepare + if: needs.prepare.outputs.build == 'true' + runs-on: ${{ matrix.runner }} + timeout-minutes: 360 + strategy: + fail-fast: false + matrix: + include: + - { backend: cpu, runner: windows-2022 } + - { backend: vulkan, runner: windows-2022 } + - backend: cuda + runner: ${{ vars.RELEASE_WINDOWS_CUDA_RUNNER || 'windows-2022' }} + steps: + - name: Checkout repository + uses: actions/checkout@df4cb1c069e1874edd31b4311f1884172cec0e10 # v6.0.3 + with: + persist-credentials: false + + - name: Initialize submodules + shell: bash + run: | + git submodule update --init --depth 1 llama.cpp third_party/cpp-httplib + + - name: Install the Vulkan SDK + if: matrix.backend == 'vulkan' + shell: pwsh + run: | + $ErrorActionPreference = 'Stop' + choco install -y --no-progress vulkan-sdk + $vulkan = [Environment]::GetEnvironmentVariable('VULKAN_SDK', 'Machine') + if (-not $vulkan) { throw 'VULKAN_SDK was not set by the installer' } + Add-Content -Path $env:GITHUB_ENV -Value "VULKAN_SDK=$vulkan" + Add-Content -Path $env:GITHUB_PATH -Value "$vulkan\Bin" + + - name: Install the CUDA toolkit + if: matrix.backend == 'cuda' + uses: Jimver/cuda-toolkit@b8bf9c6c28f8a92fbb04dcfcaee872e60c57462d # v0.2.36 + with: + cuda: "13.0.0" + method: network + + - name: Restore vcpkg binaries + id: vcpkg-cache + uses: actions/cache/restore@55cc8345863c7cc4c66a329aec7e433d2d1c52a9 # v6.1.0 + with: + path: ~\AppData\Local\vcpkg\archives + key: vcpkg-windows-x64-${{ hashFiles('vcpkg.json', 'scripts/windows/build.ps1') }} + restore-keys: vcpkg-windows-x64- + + - name: Build + id: build + shell: pwsh + env: + BACKEND: ${{ matrix.backend }} + run: | + $arguments = @{ + Backend = $env:BACKEND + Profile = 'server' + BuildDir = "${{ github.workspace }}\build\release-$env:BACKEND" + CMakeArgs = @('-DGGML_NATIVE=OFF') + } + if ($env:BACKEND -eq 'cuda') { + $arguments.CudaArch = '75-real;80-virtual;86-real;89-real;120a-real' + $arguments.CublasShim = $true + } + scripts\windows\build.ps1 @arguments + + - name: Save vcpkg binaries + if: steps.build.outcome == 'success' && steps.vcpkg-cache.outputs.cache-hit != 'true' + uses: actions/cache/save@55cc8345863c7cc4c66a329aec7e433d2d1c52a9 # v6.1.0 + with: + path: ~\AppData\Local\vcpkg\archives + key: ${{ steps.vcpkg-cache.outputs.cache-primary-key }} + + - name: Package + shell: pwsh + env: + BACKEND: ${{ matrix.backend }} + RELEASE_VERSION: ${{ needs.prepare.outputs.version }} + run: | + scripts\windows\package-release.ps1 -BuildDir "build\release-$env:BACKEND" ` + -Backend $env:BACKEND -Version $env:RELEASE_VERSION -OutputDir release-artifacts + + - name: Cache models + if: matrix.backend == 'cpu' + uses: actions/cache@55cc8345863c7cc4c66a329aec7e433d2d1c52a9 # v6.1.0 + with: + path: ${{ env.NEMO_SPEECH_MODEL_DIR }} + key: models-${{ hashFiles('models/index.json') }} + + - name: Smoke test + # GPU archives need a GPU driver at load time; the CUDA archive is tested in gpu-smoke. + if: matrix.backend == 'cpu' + shell: pwsh + run: | + $ErrorActionPreference = 'Stop' + $extract = Join-Path $env:RUNNER_TEMP 'extract' + Expand-Archive -Path (Get-Item release-artifacts\*.zip).FullName -DestinationPath $extract + $cli = (Get-Item "$extract\*\bin\nemo-speech.exe").FullName + python tests\ci\model_smoke.py --binary $cli --backend cpu --audio $env:JFK_AUDIO + if ($LASTEXITCODE -ne 0) { throw "model smoke test failed ($LASTEXITCODE)" } + + - name: Upload archive + uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1 + with: + name: release-windows-x86_64-${{ matrix.backend }} + path: release-artifacts/* + if-no-files-found: error + + verify: + name: Verify archives + needs: [prepare, linux, macos, windows] + runs-on: ubuntu-24.04 + steps: + - name: Checkout repository + uses: actions/checkout@df4cb1c069e1874edd31b4311f1884172cec0e10 # v6.0.3 + with: + persist-credentials: false + + - name: Download archives + uses: actions/download-artifact@3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c # v8.0.1 + with: + pattern: release-* + path: dist + merge-multiple: true + + - name: Check the archive set + env: + VERSION: ${{ needs.prepare.outputs.version }} + run: | + set -euo pipefail + expected="" + for name in linux-x86_64-cpu linux-x86_64-vulkan linux-x86_64-cuda \ + linux-aarch64-cpu linux-aarch64-vulkan linux-aarch64-cuda12 linux-aarch64-cuda13 \ + macos-aarch64-cpu macos-aarch64-metal macos-x86_64-cpu; do + expected="$expected nemo-speech-$VERSION-$name.tar.gz" + done + for name in windows-x86_64-cpu windows-x86_64-vulkan windows-x86_64-cuda; do + expected="$expected nemo-speech-$VERSION-$name.zip" + done + actual="$(cd dist && ls -- *.tar.gz *.zip | sort | tr '\n' ' ')" + wanted="$(printf '%s\n' $expected | sort | tr '\n' ' ')" + if [ "$actual" != "$wanted" ]; then + echo "::error::archive set differs from the expected 13" + diff <(printf '%s\n' $wanted) <(printf '%s\n' $actual) || true + exit 1 + fi + + - name: Check checksums, layout, and the x86_64 instruction baseline + env: + VERSION: ${{ needs.prepare.outputs.version }} + run: | + sudo apt-get update && sudo apt-get install -y --no-install-recommends llvm + python3 scripts/release/check_release.py archives dist --version "$VERSION" + + baseline: + name: x86_64 baseline (emulated Haswell CPU) + needs: [prepare, linux] + runs-on: ubuntu-24.04 + timeout-minutes: 120 + steps: + - name: Checkout repository + uses: actions/checkout@df4cb1c069e1874edd31b4311f1884172cec0e10 # v6.0.3 + with: + persist-credentials: false + + - name: Download the Linux x86_64 CPU archive + uses: actions/download-artifact@3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c # v8.0.1 + with: + name: release-linux-x86_64-cpu + path: dist + + - name: Cache models + uses: actions/cache@55cc8345863c7cc4c66a329aec7e433d2d1c52a9 # v6.1.0 + with: + path: ${{ env.NEMO_SPEECH_MODEL_DIR }} + key: models-${{ hashFiles('models/index.json') }} + + - name: Transcribe on an AVX2-only CPU + # Hosted runners usually have AVX-512, so a native run would not catch + # AVX-512 code. QEMU's Haswell model has AVX2/FMA/F16C and no AVX-512, + # so such an instruction raises SIGILL here. + run: | + set -euo pipefail + sudo apt-get update && sudo apt-get install -y --no-install-recommends qemu-user + mkdir -p "$RUNNER_TEMP/extract" + tar -xzf dist/*.tar.gz -C "$RUNNER_TEMP/extract" + cli="$(echo "$RUNNER_TEMP"/extract/*/bin/nemo-speech)" + "$cli" pull nvidia/nemotron-3.5-asr-streaming-0.6b + text="$(qemu-x86_64 -cpu Haswell "$cli" transcribe "$JFK_AUDIO" --backend cpu)" + echo "$text" + echo "$text" | grep -qi "fellow americans" + + gpu-smoke-linux: + name: Linux x86_64 CUDA archive (L4) + needs: [prepare, grammars, linux] + runs-on: linux-amd64-gpu-l4-latest-1 + timeout-minutes: 60 + container: + image: nvcr.io/nvidia/cuda@sha256:1e8ac7a54c184a1af8ef2167f28fa98281892a835c981ebcddb1fad04bdd452d # 13.0.0-devel-ubuntu24.04 + options: -u root --security-opt seccomp=unconfined --shm-size 16g + env: + NVIDIA_VISIBLE_DEVICES: ${{ env.NVIDIA_VISIBLE_DEVICES }} + steps: + - name: Install prerequisites + run: apt-get update && apt-get install -y --no-install-recommends git python3 ca-certificates bzip2 + + - name: Checkout repository + uses: actions/checkout@df4cb1c069e1874edd31b4311f1884172cec0e10 # v6.0.3 + with: + persist-credentials: false + + - name: Download the CUDA archive + uses: actions/download-artifact@3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c # v8.0.1 + with: + name: release-linux-x86_64-cuda + path: dist + + - name: Download grammars + uses: actions/download-artifact@3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c # v8.0.1 + with: + name: grammars + path: /tmp/grammars + + - name: Model smoke test + run: | + set -eu + mkdir -p /tmp/extract + tar -xzf dist/*.tar.gz -C /tmp/extract + tar -xjf /tmp/grammars/itn_configs.tar.bz2 -C /tmp/grammars + tar -xjf /tmp/grammars/tn_configs.tar.bz2 -C /tmp/grammars + python3 tests/ci/model_smoke.py --binary "$(echo /tmp/extract/*/bin/nemo-speech)" \ + --backend cuda --audio "$JFK_AUDIO" \ + --itn-model-dir /tmp/grammars/itn_configs/en --tn-model-dir /tmp/grammars/tn_configs + + gpu-smoke-windows: + name: Windows x86_64 CUDA archive (L4) + needs: [prepare, windows] + runs-on: windows-amd64-gpu-l4-latest-1 + timeout-minutes: 60 + # Non-blocking, like the other jobs on this runner (see gpu.yml). + continue-on-error: true + steps: + - name: Checkout repository + uses: actions/checkout@df4cb1c069e1874edd31b4311f1884172cec0e10 # v6.0.3 + with: + persist-credentials: false + + - name: Download the CUDA archive + uses: actions/download-artifact@3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c # v8.0.1 + with: + name: release-windows-x86_64-cuda + path: dist + + - name: Model smoke test + shell: pwsh + run: | + $ErrorActionPreference = 'Stop' + $extract = Join-Path $env:RUNNER_TEMP 'extract' + Expand-Archive -Path (Get-Item dist\*.zip).FullName -DestinationPath $extract + $cli = (Get-Item "$extract\*\bin\nemo-speech.exe").FullName + python tests\ci\model_smoke.py --binary $cli --backend cuda --audio $env:JFK_AUDIO + if ($LASTEXITCODE -ne 0) { throw "model smoke test failed ($LASTEXITCODE)" } + + publish: + name: Publish (${{ needs.prepare.outputs.channel }}) + needs: [prepare, grammars, verify, baseline, gpu-smoke-linux, gpu-smoke-windows] + if: needs.prepare.outputs.channel != 'dry-run' + runs-on: ubuntu-latest + # Requires approval when the "release" environment has required reviewers. + environment: release + permissions: + contents: write + id-token: write + attestations: write + steps: + - name: Download archives + uses: actions/download-artifact@3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c # v8.0.1 + with: + pattern: release-* + path: dist + merge-multiple: true + + - name: Write SHA256SUMS + run: cd dist && sha256sum -- *.tar.gz *.zip > SHA256SUMS + + - name: Attest build provenance + uses: actions/attest-build-provenance@4d101475d8b20a2381f78447822ac1eab6504dd8 # v4.2.2 + with: + subject-path: | + dist/*.tar.gz + dist/*.zip + + - name: Add the text-normalization grammars + uses: actions/download-artifact@3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c # v8.0.1 + with: + name: grammars + path: dist + + - name: Create the draft release + if: needs.prepare.outputs.channel == 'release' + env: + GH_TOKEN: ${{ github.token }} + VERSION: ${{ needs.prepare.outputs.version }} + run: | + gh release create "v$VERSION" --repo "$GITHUB_REPOSITORY" --draft --verify-tag \ + --title "NeMo-Speech.cpp $VERSION" --generate-notes dist/* + + - name: Replace the nightly prerelease + if: needs.prepare.outputs.channel == 'nightly' + env: + GH_TOKEN: ${{ github.token }} + run: | + gh release delete nightly --repo "$GITHUB_REPOSITORY" --yes --cleanup-tag || true + gh release create nightly --repo "$GITHUB_REPOSITORY" --prerelease --target "$GITHUB_SHA" \ + --title "Nightly" --notes "Built from $GITHUB_SHA." dist/* diff --git a/THIRD_PARTY_NOTICES.md b/THIRD_PARTY_NOTICES.md index 282bd20..d41194a 100644 --- a/THIRD_PARTY_NOTICES.md +++ b/THIRD_PARTY_NOTICES.md @@ -169,6 +169,27 @@ SentencePiece runtime and its bundled Abseil, protobuf-lite, and Darts-clone components. Their Apache 2.0 and BSD license texts are installed under `share/licenses/nemo-speech/third_party/sentencepiece/`. +### Text normalization runtime + +`scripts/build_itn_deps.sh` builds these pinned sources for ITN and TN: + +- OpenFST: [`sarane22/openfst`](https://github.com/sarane22/openfst), revision + `fc23b4cf529429284b874a26f28b15c6cc94f404`; Copyright 2005-2024 Google LLC; + Apache License 2.0 +- Sparrowhawk: [`sarane22/sparrowhawk`](https://github.com/sarane22/sparrowhawk), + revision `8b082acc507312077a096be8398584a13832c490`; Copyright 2015 and + onwards Google, Inc.; Apache License 2.0 +- Protocol Buffers: + [`protocolbuffers/protobuf`](https://github.com/protocolbuffers/protobuf) + v21.12; Copyright 2008 Google Inc.; BSD 3-Clause License +- RE2: [`google/re2`](https://github.com/google/re2) 2023-03-01; Copyright (c) + 2009 The RE2 Authors; BSD 3-Clause License + +Linux and macOS release archives statically link all four into +`libnemo_speech_text_normalization`. Builds that use system Protocol Buffers and +RE2 link those instead. The license texts are installed under +`share/licenses/nemo-speech/third_party/`. + ### whisper.cpp sample audio The ASR quick-start fixture at `test_files/asr/wav/test/jfk.wav` is copied from diff --git a/docker/Dockerfile.release-linux b/docker/Dockerfile.release-linux new file mode 100644 index 0000000..656a7a9 --- /dev/null +++ b/docker/Dockerfile.release-linux @@ -0,0 +1,241 @@ +# syntax=docker/dockerfile:1.7 +# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-License-Identifier: Apache-2.0 +# Portable x86_64 and aarch64 CPU/Vulkan/CUDA release archives. +# +# Build an installer-compatible archive into release-artifacts/ from the +# repository root (.github/workflows/release.yml runs the same commands): +# docker build --platform=linux/amd64 -f docker/Dockerfile.release-linux \ +# --build-arg BACKEND=vulkan --target artifact \ +# --output type=local,dest=release-artifacts . +# docker build --platform=linux/amd64 -f docker/Dockerfile.release-linux \ +# --target cuda-artifact \ +# --output type=local,dest=release-artifacts . + +ARG CUDA_IMAGE=nvcr.io/nvidia/cuda:12.8.1-devel-ubuntu24.04 + +FROM ubuntu:24.04 AS vulkan-tools + +ENV DEBIAN_FRONTEND=noninteractive + +RUN apt-get update && apt-get install -y --no-install-recommends \ + ca-certificates \ + glslc \ + libvulkan-dev \ + spirv-headers \ + && rm -rf /var/lib/apt/lists/* \ + && set -eux; \ + mkdir -p /opt/vulkan-tools/bin /opt/vulkan-tools/lib; \ + cp -L "$(command -v glslc)" /opt/vulkan-tools/bin/glslc; \ + ldd "$(command -v glslc)" \ + | awk '$1 ~ /^\// { print $1 } $2 == "=>" && $3 ~ /^\// { print $3 }' \ + | sort -u \ + | while read -r dependency; do \ + cp -L "$dependency" "/opt/vulkan-tools/lib/$(basename "$dependency")"; \ + done; \ + loader="$(ldd "$(command -v glslc)" \ + | awk '$1 ~ /ld-linux/ { print $1 } $2 == "=>" && $3 ~ /ld-linux/ { print $3 }' \ + | head -n 1)"; \ + test -n "$loader"; \ + cp -L "$loader" /opt/vulkan-tools/lib/ld-linux.so + +FROM ${CUDA_IMAGE} AS cuda-tools + +RUN set -eux; \ + cuda_home="$(readlink -f /usr/local/cuda)"; \ + mkdir -p /opt/cuda; \ + cp -a "$cuda_home"/. /opt/cuda/; \ + cuda_license="$(find /usr/share/doc -maxdepth 2 -type f \ + -path '*/cuda-cudart-*/copyright' -print | sort | head -n 1)"; \ + test -n "$cuda_license"; \ + cp "$cuda_license" /opt/cuda-cudart-copyright + +FROM ubuntu:20.04 AS portable-base + +ARG TARGETARCH +ARG CMAKE_VERSION=3.31.6 +ARG CMAKE_SHA256_AMD64=5a1133ff103c71eb5120e2cc3de922733e7d8a26a98ae716397e8676adb367bf +ARG CMAKE_SHA256_ARM64=b4cc788d63112b2749b40627e719eb5d3b8ed8f00c36d77189f4019cfe64bc9e +ARG JOBS=4 + +ENV DEBIAN_FRONTEND=noninteractive \ + CMAKE_BUILD_PARALLEL_LEVEL=${JOBS} \ + PATH=/opt/cmake/bin:/usr/local/sbin:/usr/local/bin:/usr/sbin:/usr/bin:/sbin:/bin + +# Use the target architecture's native toolchain on the portable baseline. +RUN test "$(dpkg --print-architecture)" = "${TARGETARCH}" \ + && apt-get update && apt-get install -y --no-install-recommends \ + binutils \ + ca-certificates \ + curl \ + g++-9 \ + gcc-9 \ + git \ + libvulkan-dev \ + ninja-build \ + pkg-config \ + python3 \ + && update-alternatives --install /usr/bin/gcc gcc /usr/bin/gcc-9 100 \ + && update-alternatives --install /usr/bin/g++ g++ /usr/bin/g++-9 100 \ + && update-alternatives --install /usr/bin/cc cc /usr/bin/gcc-9 100 \ + && update-alternatives --install /usr/bin/c++ c++ /usr/bin/g++-9 100 \ + && rm -rf /var/lib/apt/lists/* + +RUN set -eux; \ + case "${TARGETARCH}" in \ + amd64) \ + cmake_arch=x86_64; \ + cmake_sha256="${CMAKE_SHA256_AMD64}" \ + ;; \ + arm64) \ + cmake_arch=aarch64; \ + cmake_sha256="${CMAKE_SHA256_ARM64}" \ + ;; \ + *) echo "unsupported TARGETARCH: ${TARGETARCH}" >&2; exit 2 ;; \ + esac; \ + curl -fsSL \ + "https://github.com/Kitware/CMake/releases/download/v${CMAKE_VERSION}/cmake-${CMAKE_VERSION}-linux-${cmake_arch}.tar.gz" \ + -o /tmp/cmake.tar.gz; \ + echo "${cmake_sha256} /tmp/cmake.tar.gz" | sha256sum -c -; \ + mkdir -p /opt/cmake; \ + tar -xzf /tmp/cmake.tar.gz --strip-components=1 -C /opt/cmake; \ + rm /tmp/cmake.tar.gz; \ + cmake --version; \ + ninja --version + +FROM portable-base AS sentencepiece-dependency + +WORKDIR /work +COPY scripts/build_sentencepiece_static.sh /work/scripts/build_sentencepiece_static.sh + +# SentencePiece is a core ASR dependency. Build the repository's pinned static +# revision once per target architecture and share it across every release +# backend instead of relying on a package from the portable base image. +RUN scripts/build_sentencepiece_static.sh + +FROM portable-base AS itn-dependency + +RUN apt-get update && apt-get install -y --no-install-recommends make perl \ + && rm -rf /var/lib/apt/lists/* + +WORKDIR /work +COPY scripts/build_itn_deps.sh /work/scripts/build_itn_deps.sh +COPY src/common/text_normalization/compat /work/src/common/text_normalization/compat + +# Text normalization (ITN/TN): OpenFST, Sparrowhawk, protobuf, and RE2 as +# static archives, linked privately into libnemo_speech_text_normalization. +RUN STATIC=1 scripts/build_itn_deps.sh + +FROM portable-base AS builder + +ARG BACKEND +ARG ARTIFACT_BACKEND +# Archive version; defaults to VERSION ("nightly" for the nightly channel). +ARG RELEASE_VERSION= + +# Ubuntu 20.04 supplies the release ABI baseline. Current Vulkan headers and +# glslc are build-only inputs copied from the newer tool stage. +COPY --from=vulkan-tools /opt/vulkan-tools /opt/vulkan-tools +COPY --from=vulkan-tools /usr/include/vulkan /usr/include/vulkan +COPY --from=vulkan-tools /usr/include/vk_video /usr/include/vk_video +COPY --from=vulkan-tools /usr/include/spirv /usr/include/spirv +COPY --from=vulkan-tools /usr/share/cmake/SPIRV-Headers /usr/share/cmake/SPIRV-Headers + +RUN printf '%s\n' \ + '#!/bin/sh' \ + 'exec /opt/vulkan-tools/lib/ld-linux.so --library-path /opt/vulkan-tools/lib /opt/vulkan-tools/bin/glslc "$@"' \ + > /usr/local/bin/glslc \ + && chmod 0755 /usr/local/bin/glslc \ + && glslc --version + +WORKDIR /work +COPY . /work +COPY --from=sentencepiece-dependency /work/.deps/sentencepiece /work/.deps/sentencepiece +COPY --from=itn-dependency /work/.deps/itn /work/.deps/itn + +RUN case "${BACKEND}" in cpu|vulkan) ;; \ + *) echo "BACKEND must be cpu or vulkan" >&2; exit 2 ;; \ + esac \ + && scripts/configure.sh "${BACKEND}-server" \ + -DGGML_NATIVE=OFF \ + -DNEMO_SPEECH_BUILD_GRPC=OFF \ + -DNEMO_SPEECH_WITH_GRPC=OFF \ + -DNEMO_SPEECH_WITH_NORM=ON \ + -DCMAKE_INSTALL_PREFIX=/opt/nemo-speech \ + "-DCMAKE_CXX_STANDARD_LIBRARIES:STRING=-lstdc++fs -lanl" \ + "-DCMAKE_EXE_LINKER_FLAGS=-Wl,--disable-new-dtags" \ + "-DCMAKE_SHARED_LINKER_FLAGS=-Wl,--disable-new-dtags" \ + && cmake --build --preset "${BACKEND}-server" \ + && cmake --install "build/${BACKEND}-server" --prefix /opt/nemo-speech \ + && artifact_backend="${ARTIFACT_BACKEND:-${BACKEND}}" \ + && scripts/release/package-linux.sh \ + --install-prefix /opt/nemo-speech \ + --backend "${BACKEND}" \ + --artifact-backend "${artifact_backend}" \ + --output-dir /release \ + --max-glibc 2.31 \ + ${RELEASE_VERSION:+--version "$RELEASE_VERSION"} + +FROM scratch AS artifact +COPY --from=builder /release/ / + +FROM portable-base AS cuda-builder + +ARG TARGETARCH +ARG CUDA_ARCH= +ARG ARTIFACT_BACKEND=cuda +ARG RELEASE_VERSION= + +ENV CUDA_HOME=/usr/local/cuda \ + PATH=/opt/cmake/bin:/usr/local/cuda/bin:/usr/local/sbin:/usr/local/bin:/usr/sbin:/usr/bin:/sbin:/bin + +COPY --from=cuda-tools /opt/cuda /usr/local/cuda +COPY --from=cuda-tools /opt/cuda-cudart-copyright /opt/cuda-cudart-copyright + +RUN set -eux; \ + nvcc --version; \ + cuda_version="$(nvcc --version \ + | sed -n 's/.*release \([0-9]*\)\.\([0-9]*\).*/\1-\2/p' \ + | head -n 1)"; \ + test -n "$cuda_version"; \ + install -Dm0644 /opt/cuda-cudart-copyright \ + "/usr/share/doc/cuda-cudart-${cuda_version}/copyright" + +WORKDIR /work +COPY . /work +COPY --from=sentencepiece-dependency /work/.deps/sentencepiece /work/.deps/sentencepiece +COPY --from=itn-dependency /work/.deps/itn /work/.deps/itn + +RUN set -eux; \ + cuda_arch="${CUDA_ARCH}"; \ + if [ -z "$cuda_arch" ]; then \ + case "${TARGETARCH}" in \ + amd64) cuda_arch='75-real;80-virtual;86-real;89-real;120a-real' ;; \ + arm64) cuda_arch='87-real;110a-real;121a-real;87-virtual' ;; \ + *) echo "unsupported TARGETARCH: ${TARGETARCH}" >&2; exit 2 ;; \ + esac; \ + fi; \ + scripts/configure.sh cuda-server \ + -DGGML_NATIVE=OFF \ + -DGGML_CUDA_NCCL=OFF \ + -DNEMO_SPEECH_CUBLAS_SHIM=ON \ + -DNEMO_SPEECH_BUILD_GRPC=OFF \ + -DNEMO_SPEECH_WITH_GRPC=OFF \ + -DNEMO_SPEECH_WITH_NORM=ON \ + -DCMAKE_INSTALL_PREFIX=/opt/nemo-speech \ + "-DCMAKE_CXX_STANDARD_LIBRARIES:STRING=-lstdc++fs -lanl" \ + "-DCMAKE_EXE_LINKER_FLAGS=-Wl,--disable-new-dtags" \ + "-DCMAKE_SHARED_LINKER_FLAGS=-Wl,--disable-new-dtags" \ + "-DCMAKE_CUDA_ARCHITECTURES=${cuda_arch}"; \ + cmake --build --preset cuda-server; \ + cmake --install build/cuda-server --prefix /opt/nemo-speech; \ + scripts/release/package-linux.sh \ + --install-prefix /opt/nemo-speech \ + --backend cuda \ + --artifact-backend "${ARTIFACT_BACKEND}" \ + --output-dir /release \ + --max-glibc 2.31 \ + ${RELEASE_VERSION:+--version "$RELEASE_VERSION"} + +FROM scratch AS cuda-artifact +COPY --from=cuda-builder /release/ / diff --git a/docs/build.md b/docs/build.md index 1c8d91e..42d6e67 100644 --- a/docs/build.md +++ b/docs/build.md @@ -196,6 +196,15 @@ prefix without `sudo`: CC=gcc-12 CXX=g++-12 scripts/build_itn_deps.sh ``` +With `STATIC=1`, the default on macOS, the script also builds pinned Protobuf +and RE2 and installs static archives only. `libnemo_speech_text_normalization` +then carries the whole stack privately, so neither system package is needed at +build or run time. The release archives use this mode: + +```bash +STATIC=1 scripts/build_itn_deps.sh +``` + On Linux, normalization builds also require the static SentencePiece dependency: diff --git a/docs/development/README.md b/docs/development/README.md index bc3896f..d9e3399 100644 --- a/docs/development/README.md +++ b/docs/development/README.md @@ -16,3 +16,5 @@ the server want [ASR configuration](../asr/configuration.md), - [`cublas-shim.md`](cublas-shim.md) - the in-tree drop-in cuBLAS replacement under `kernels/` and where the custom GPU kernels live. - [Windows build notes](windows-build.md) +- [`releasing.md`](releasing.md) - how the release workflow builds, checks, and + publishes the binary archives. diff --git a/docs/development/releasing.md b/docs/development/releasing.md new file mode 100644 index 0000000..6950245 --- /dev/null +++ b/docs/development/releasing.md @@ -0,0 +1,88 @@ +# Releasing + +[`.github/workflows/release.yml`](../../.github/workflows/release.yml) builds +the binary archives that `scripts/install.sh` and `scripts/install.ps1` +download, checks them, and publishes them to GitHub Releases. + +| Platform | Archives | Built on | +|---|---|---| +| Linux x86_64 | `cpu`, `vulkan`, `cuda` | `docker/Dockerfile.release-linux` | +| Linux aarch64 | `cpu`, `vulkan`, `cuda12` (Orin), `cuda13` (Thor, DGX Spark) | `docker/Dockerfile.release-linux` | +| macOS | `aarch64-cpu`, `aarch64-metal`, `x86_64-cpu` | `macos-15`, `macos-15-intel` | +| Windows x86_64 | `cpu`, `vulkan`, `cuda` | `scripts/windows/build.ps1` | + +Archives are named `nemo-speech----.tar.gz` +(`.zip` on Windows), contain a single directory of the same name, and ship +with a `.sha256` file. + +Linux and macOS archives include ITN and TN: `scripts/build_itn_deps.sh` with +`STATIC=1` builds OpenFST, Sparrowhawk, Protobuf, and RE2 as static archives, +and `libnemo_speech_text_normalization` links them privately. The grammars, +`itn_configs.tar.bz2` and `tn_configs.tar.bz2`, are not built by the workflow; +each run takes them from the latest release and publishes them again. + +## Cutting a release + +1. Set `NEMO_SPEECH_VERSION` in `VERSION` and merge it to `main`. +2. Push the matching tag: `git tag v0.2.0 && git push origin v0.2.0`. The + workflow fails if the tag and `VERSION` differ. +3. Approve the `release` environment when the build and checks finish. The + workflow creates a draft release with the archives, `SHA256SUMS`, build + provenance attestations, and the text-normalization grammars. +4. Review the draft and publish it. + +A daily scheduled run publishes the `nightly` prerelease, which +`install.sh --channel nightly` installs. It is skipped when `main` has not +moved since the current nightly; running the workflow manually with `nightly` +always rebuilds. + +Pull requests that change release packaging (the workflow, the release +Dockerfile and packagers, the dependency build scripts, or the model smoke test) +run the workflow as a dry run once copy-pr-bot mirrors them to a +`pull-request/` branch. Run the workflow manually with +`dry-run` to build and check everything without publishing. + +## What every run checks + +- **CPU baseline:** x86_64 archives are built with `GGML_NATIVE=OFF` and must + not contain instructions beyond x86-64-v3 (AVX2, FMA, F16C, BMI2); + `scripts/release/check_release.py` disassembles every x86_64 binary. The + Linux x86_64 CPU archive also transcribes audio under QEMU's Haswell model, + which has no AVX-512. +- **Self-contained packages:** Linux archives need glibc 2.31 or newer and find + their bundled libraries through `DT_RPATH`, which `LD_LIBRARY_PATH` cannot + override. macOS archives need macOS 13.3 or newer, link only system + libraries, and are ad-hoc signed. Windows archives bundle every DLL they + import except Windows and GPU driver libraries. +- **Smoke tests:** CPU archives run `tests/ci/model_smoke.py` on their runner, + and the CUDA archives run it on L4 GPUs. On Linux and macOS the test also + synthesizes digits through TN and transcribes them back through ITN. + +## Runners + +CUDA builds are the slowest jobs. Set these repository variables to run them +on larger runners: + +| Variable | Default | +|---|---| +| `RELEASE_LINUX_X64_CUDA_RUNNER` | `ubuntu-24.04` | +| `RELEASE_LINUX_ARM64_CUDA_RUNNER` | `ubuntu-24.04-arm` | +| `RELEASE_WINDOWS_CUDA_RUNNER` | `windows-2022` | + +## Building an archive locally + +```sh +# Linux, from the repository root (CUDA: --target cuda-artifact) +docker build --platform=linux/amd64 -f docker/Dockerfile.release-linux \ + --build-arg BACKEND=cpu --target artifact \ + --output type=local,dest=release-artifacts . + +# macOS, after scripts/build_itn_deps.sh and installing a preset configured +# with -DNEMO_SPEECH_WITH_NORM=ON +scripts/release/package-macos.sh --install-prefix --backend metal --arch aarch64 +``` + +```powershell +# Windows, after scripts\windows\build.ps1 -Profile server -CMakeArgs '-DGGML_NATIVE=OFF' +scripts\windows\package-release.ps1 -BuildDir -Backend cpu +``` diff --git a/docs/install.md b/docs/install.md index c29d4ee..24a94c3 100644 --- a/docs/install.md +++ b/docs/install.md @@ -37,6 +37,11 @@ x86_64 CUDA archive supports Turing-class GPUs (compute capability 7.5, including RTX 20-series) and newer. On an older GPU, select `--backend cpu` or `--backend vulkan`, or build from source with a compatible CUDA toolkit. +Linux and macOS archives include inverse text normalization for transcripts +(`--itn-model-dir`) and text normalization for synthesis (`--tn-model-dir`). +The grammars are published with each release as `itn_configs.tar.bz2` and +`tn_configs.tar.bz2`. + The installer selects CUDA when `nvidia-smi` is available, Metal on Apple Silicon, and CPU otherwise. Override the backend or force a source build: diff --git a/scripts/build_itn_deps.sh b/scripts/build_itn_deps.sh index 3a04c7f..eb6353e 100755 --- a/scripts/build_itn_deps.sh +++ b/scripts/build_itn_deps.sh @@ -3,14 +3,20 @@ # SPDX-License-Identifier: Apache-2.0 # Build the Sparrowhawk ITN stack (OpenFST 1.8 + Sparrowhawk) from pinned # sarane22 forks and install to a user-writable project prefix, enabling -# -DNEMO_SPEECH_WITH_ITN=ON. Shared by the x86_64 and aarch64 images. +# -DNEMO_SPEECH_WITH_NORM=ON. Shared by the x86_64 and aarch64 images. # -# Expects: protobuf headers + protoc, re2, autotools, and gcc-12 as CC/CXX. The -# Dockerfiles set CC/CXX=gcc-12 for this step (gcc-13/14 ICE on OpenFST's heavy -# templates at -O2) while the runtime itself builds with gcc-13. Sparrowhawk uses -# an in-tree, OpenFST-only compatibility implementation for the tiny subset of +# Expects autotools and, on Linux, gcc-12 as CC/CXX. docker/Dockerfile sets +# CC/CXX=gcc-12 for this step (gcc-13/14 ICE on OpenFST's heavy templates at +# -O2) while the runtime itself builds with gcc-13. Sparrowhawk uses an in-tree, +# OpenFST-only compatibility implementation for the tiny subset of # thrax::GrmManager that it calls; no Thrax or fstscript library is built/linked. # +# STATIC=0 (Linux default) builds shared libraries against the system protobuf +# headers, protoc, and RE2. STATIC=1 (macOS default) also builds pinned protobuf +# and RE2, and installs position-independent static archives only, so +# nemo_speech_text_normalization carries the whole stack privately; the release +# archives use this mode. It additionally requires CMake. +# # Usage: scripts/build_itn_deps.sh [WORKDIR] (default: ./.deps/itn-build) set -euo pipefail @@ -20,14 +26,21 @@ PREFIX="${PREFIX:-$REPO/.deps/itn}" JOBS="${JOBS:-8}" # Cap parallelism: OpenFST's template-heavy translation units can OOM cc1plus. JOBS="$(( JOBS < 4 ? JOBS : 4 ))" +if [ "$(uname -s)" = Darwin ]; then + STATIC="${STATIC:-1}" +else + STATIC="${STATIC:-0}" +fi CXXO="-std=c++17 -O2" SHIM="$REPO/src/common/text_normalization/compat/sparrowhawk_compat.h" ITN_COMPAT="$REPO/src/common/text_normalization/compat" +LICENSE_DIR="$PREFIX/share/licenses/nemo-speech/third_party" clone() { # clone if [ ! -d "$2" ]; then - git clone "$1" "$2" - git -C "$2" checkout --quiet "$3" + git init --quiet "$2" + git -C "$2" fetch --quiet --depth 1 "$1" "$3" + git -C "$2" checkout --quiet FETCH_HEAD elif [ "$(git -C "$2" rev-parse HEAD)" != "$3" ]; then echo "$2 exists at the wrong revision; remove it or choose a clean WORKDIR" >&2 echo " expected: $3" >&2 @@ -36,19 +49,70 @@ clone() { # clone fi } +# Stamp the shipped autotools output newer than its inputs so make does not try +# to regenerate it with whichever autoconf/automake the host has. +stamp_autotools() { + touch -t 202001010000 configure.ac acinclude.m4 2>/dev/null || true + [ -d m4 ] && touch -t 202001010000 m4/*.m4 2>/dev/null || true + find . -name 'Makefile.am' -exec touch -t 202001010000 {} + + touch -t 202001020000 aclocal.m4 + touch -t 202001030000 configure + find . -name '*.in' -exec touch -t 202001030000 {} + +} + +install_license() { # install_license + install -d "$LICENSE_DIR/$2" + install -m 0644 "$1" "$LICENSE_DIR/$2/$(basename "$1")" +} + +LIBRARY_KIND=() +if [ "$STATIC" = 1 ]; then + LIBRARY_KIND=(--disable-shared --enable-static --with-pic) +fi + mkdir -p "$WORK" cd "$WORK" # Pinned OpenFST and Sparrowhawk compatibility revisions. clone https://github.com/sarane22/openfst.git openfst fc23b4cf529429284b874a26f28b15c6cc94f404 clone https://github.com/sarane22/sparrowhawk.git sparrowhawk 8b082acc507312077a096be8398584a13832c490 +# --------------------------------------------------------- protobuf and RE2 +# The last releases that do not require Abseil. +if [ "$STATIC" = 1 ]; then + clone https://github.com/protocolbuffers/protobuf.git protobuf f0dc78d7e6e331b8c6bb2d5283e06aa26883ca7c # v21.12 + clone https://github.com/google/re2.git re2 3a8436ac436124a57a4e22d5c8713a2d42b381d7 # 2023-03-01 + cmake -S "$WORK/protobuf" -B "$WORK/protobuf/build" \ + -DCMAKE_BUILD_TYPE=Release \ + -DCMAKE_INSTALL_PREFIX="$PREFIX" \ + -DCMAKE_INSTALL_LIBDIR=lib \ + -DCMAKE_POSITION_INDEPENDENT_CODE=ON \ + -Dprotobuf_BUILD_SHARED_LIBS=OFF \ + -Dprotobuf_BUILD_TESTS=OFF \ + -Dprotobuf_WITH_ZLIB=OFF + cmake --build "$WORK/protobuf/build" --target install -j "$JOBS" + cmake -S "$WORK/re2" -B "$WORK/re2/build" \ + -DCMAKE_BUILD_TYPE=Release \ + -DCMAKE_INSTALL_PREFIX="$PREFIX" \ + -DCMAKE_INSTALL_LIBDIR=lib \ + -DCMAKE_POSITION_INDEPENDENT_CODE=ON \ + -DBUILD_SHARED_LIBS=OFF \ + -DRE2_BUILD_TESTING=OFF + cmake --build "$WORK/re2/build" --target install -j "$JOBS" + # Sparrowhawk's configure and Makefiles run protoc from PATH. + export PATH="$PREFIX/bin:$PATH" + install_license "$WORK/protobuf/LICENSE" protobuf + install_license "$WORK/re2/LICENSE" re2 +fi + # ---------------------------------------------------------------- OpenFST 1.8 cd "$WORK/openfst" # FST_FLAGS_v rename missed by the fork. -sed -i 's/\bFLAGS_v\b/FST_FLAGS_v/g' src/include/fst/label-reachable.h +perl -pi -e 's/\bFLAGS_v\b/FST_FLAGS_v/g' src/include/fst/label-reachable.h # FAR + PDT cover Sparrowhawk's runtime grammar formats. Disable command-line # tools/script wrappers: the runtime calls the typed C++ OpenFST API directly. -./configure --prefix="$PREFIX" --enable-far --enable-pdt --disable-bin CXXFLAGS="$CXXO" +stamp_autotools +./configure --prefix="$PREFIX" --enable-far --enable-pdt --disable-bin \ + ${LIBRARY_KIND[@]+"${LIBRARY_KIND[@]}"} CXXFLAGS="$CXXO" make -j"$JOBS" make install # Stage all core headers plus the FAR/PDT extension templates used by the @@ -67,25 +131,17 @@ fi # -------------------------------------------------------------- Sparrowhawk cd "$WORK/sparrowhawk" # (1) Autoconf tarball pins -std=c++11; OpenFST 1.8 headers need C++17. -sed -i 's/-std=c++11/-std=c++17/g' configure -[ -f configure.ac ] && sed -i 's/-std=c++11/-std=c++17/g' configure.ac || true -./configure --prefix="$PREFIX" --disable-bin \ +perl -pi -e 's/-std=c\+\+11/-std=c++17/g' configure +[ -f configure.ac ] && perl -pi -e 's/-std=c\+\+11/-std=c++17/g' configure.ac || true +# (2) Stamp after the edits so make does not try to regenerate configure. +stamp_autotools +./configure --prefix="$PREFIX" --disable-bin ${LIBRARY_KIND[@]+"${LIBRARY_KIND[@]}"} \ CPPFLAGS="-I$ITN_COMPAT -I$PREFIX/include" \ LDFLAGS="-L$PREFIX/lib" CXXFLAGS="$CXXO" -# (2) Stamp generated autotools files so make does not try to regenerate them. -touch -d '2020-01-01 00:00:00' configure.ac acinclude.m4 2>/dev/null || true -[ -d m4 ] && touch -d '2020-01-01 00:00:00' m4/*.m4 2>/dev/null || true -find . -name 'Makefile.am' -exec touch -d '2020-01-01 00:00:00' {} + -touch -d '2020-01-02 00:00:00' aclocal.m4 -touch -d '2020-01-03 00:00:00' configure -find . -name '*.in' -exec touch -d '2020-01-03 00:00:00' {} + -touch -d '2020-01-04 00:00:00' config.status -find . -name 'Makefile' -exec touch -d '2020-01-05 00:00:00' {} + - # (3) Build + install the proto stubs, library, and headers with the OpenFST 1.8 # compat shim force-included. src/bin (normalizer_main CLI) is skipped: the -# the runtime links libsparrowhawk directly. Put the compatibility include first so +# runtime links libsparrowhawk directly. Put the compatibility include first so # Sparrowhawk resolves without the Thrax project. CPPF="-I$ITN_COMPAT -I$PREFIX/include -include $SHIM -funsigned-char" make -C src/proto CPPFLAGS="$CPPF" @@ -101,13 +157,12 @@ for f in "$PREFIX"/lib/lib{fst,fstfar,sparrowhawk}.so.*; do [ -f "$f" ] && [ ! -L "$f" ] && strip --strip-unneeded "$f" done -LICENSE_DIR="$PREFIX/share/licenses/nemo-speech/third_party" -install -Dm0644 "$WORK/openfst/COPYING" "$LICENSE_DIR/openfst/COPYING" -install -Dm0644 "$WORK/sparrowhawk/LICENSE" "$LICENSE_DIR/sparrowhawk/LICENSE" +install_license "$WORK/openfst/COPYING" openfst +install_license "$WORK/sparrowhawk/LICENSE" sparrowhawk if command -v ldconfig >/dev/null 2>&1 && [ "$(id -u)" -eq 0 ]; then ldconfig fi echo echo "ITN stack installed to $PREFIX:" -ls -1 "$PREFIX"/lib/libsparrowhawk.so "$PREFIX"/lib/libfstfar.so "$PREFIX"/lib/libfst.so +ls -1 "$PREFIX"/lib/lib{sparrowhawk,fstfar,fst}.* diff --git a/scripts/build_sentencepiece_static.sh b/scripts/build_sentencepiece_static.sh index 05f5b20..70fe338 100755 --- a/scripts/build_sentencepiece_static.sh +++ b/scripts/build_sentencepiece_static.sh @@ -32,7 +32,8 @@ install -m 0644 "$BUILD/src/libsentencepiece.a" "$PREFIX/lib/libsentencepiece.a" install -m 0644 "$SOURCE/src/sentencepiece_processor.h" "$PREFIX/include/sentencepiece_processor.h" LICENSE_DIR="$PREFIX/share/licenses/nemo-speech/third_party/sentencepiece" -install -Dm0644 "$SOURCE/LICENSE" "$LICENSE_DIR/LICENSE" -install -Dm0644 "$SOURCE/third_party/absl/LICENSE" "$LICENSE_DIR/absl-LICENSE" -install -Dm0644 "$SOURCE/third_party/darts_clone/LICENSE" "$LICENSE_DIR/darts-clone-LICENSE" -install -Dm0644 "$SOURCE/third_party/protobuf-lite/LICENSE" "$LICENSE_DIR/protobuf-lite-LICENSE" +install -d "$LICENSE_DIR" +install -m 0644 "$SOURCE/LICENSE" "$LICENSE_DIR/LICENSE" +install -m 0644 "$SOURCE/third_party/absl/LICENSE" "$LICENSE_DIR/absl-LICENSE" +install -m 0644 "$SOURCE/third_party/darts_clone/LICENSE" "$LICENSE_DIR/darts-clone-LICENSE" +install -m 0644 "$SOURCE/third_party/protobuf-lite/LICENSE" "$LICENSE_DIR/protobuf-lite-LICENSE" diff --git a/scripts/release/check_release.py b/scripts/release/check_release.py new file mode 100755 index 0000000..722b0ef --- /dev/null +++ b/scripts/release/check_release.py @@ -0,0 +1,192 @@ +#!/usr/bin/env python3 +# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-License-Identifier: Apache-2.0 +"""Check release archives before they are published. + + check_release.py isa PATH... Reject AVX-512/AMX code in x86_64 binaries under PATH. + check_release.py archives DIR --version VERSION + Check every archive in DIR: checksum, name, layout, + required files, and the x86_64 instruction baseline. + +x86_64 archives target x86-64-v3 (AVX2, FMA, F16C, BMI2). ELF, PE, and Mach-O files are +disassembled with llvm-objdump when available, otherwise with objdump. +""" +from __future__ import annotations + +import argparse +import hashlib +import pathlib +import re +import shutil +import struct +import subprocess +import sys +import tarfile +import tempfile +import zipfile + +ARCHIVE = re.compile( + r"^nemo-speech-(?Pnightly|[0-9][0-9A-Za-z.+-]*)-(?Plinux|macos|windows)-" + r"(?Px86_64|aarch64)-(?Pcpu|vulkan|metal|cuda|cuda12|cuda13)" + r"\.(?Ptar\.gz|zip)$" +) + +# AT&T syntax, as printed by objdump and llvm-objdump. +BEYOND_X86_64_V3 = re.compile( + r"%zmm\d+|%[xy]mm(?:1[6-9]|2\d|3[01])\b|%k[0-7]\b|\{%k[0-7]\}" + r"|\b(?:vpternlog[dq]|vmovdqu(?:8|16|32|64)|vmovdqa(?:32|64)|vperm[it]2\w*|vpcompress\w*" + r"|vpexpand\w*|vfixupimm\w*|vgetexp\w*|vgetmant\w*|vrndscale\w*|vreduce\w*|vscalef\w*" + r"|vpmovm2\w*|vpbroadcastm\w*|vpdpbusds?|vpdpwssds?|vdpbf16ps|vcvtne2ps2bf16" + r"|kmov[bwdq]|kand\w*|kor\w*|kxor\w*|knot\w*|ktest\w*|kshift\w*|kunpck\w*" + r"|tdp\w+|tileload\w*|tilestored|tilezero|ldtilecfg|sttilecfg|tilerelease)\b" +) + + +def binary_arch(path: pathlib.Path) -> str | None: + """Return 'x86_64', 'aarch64', 'other', or None for files that are not executables.""" + try: + with path.open("rb") as f: + head = f.read(64) + if head[:4] == b"\x7fELF": + machine = struct.unpack_from("= 64: + f.seek(struct.unpack_from(" str: + for tool in ("llvm-objdump", "objdump"): + if shutil.which(tool): + return tool + raise SystemExit("error: llvm-objdump or objdump is required") + + +def scan_isa(paths: list[pathlib.Path]) -> list[str]: + tool = disassembler() + failures = [] + for root in paths: + files = ( + [root] + if root.is_file() + else sorted(p for p in root.rglob("*") if p.is_file() and not p.is_symlink()) + ) + for path in files: + arch = binary_arch(path) + if arch == "other": + failures.append(f"{path}: unsupported binary format or architecture") + if arch != "x86_64": + continue + result = subprocess.run( + [tool, "-d", "--no-show-raw-insn", str(path)], + capture_output=True, + text=True, + errors="replace", + ) + if result.returncode != 0 or "file format not recognized" in result.stderr: + failures.append(f"{path}: {tool} could not disassemble it") + continue + hits = [ + line.strip() for line in result.stdout.splitlines() if BEYOND_X86_64_V3.search(line) + ] + if hits: + failures.append( + f"{path}: {len(hits)} instructions beyond x86-64-v3, first: {hits[0]}" + ) + return failures + + +def extract(archive: pathlib.Path, dest: pathlib.Path) -> None: + if archive.name.endswith(".zip"): + with zipfile.ZipFile(archive) as z: + z.extractall(dest) + else: + with tarfile.open(archive) as t: + if hasattr(tarfile, "data_filter"): + t.extractall(dest, filter="data") + else: + t.extractall(dest) + + +def check_archives(directory: pathlib.Path, version: str) -> list[str]: + failures = [] + archives = sorted(p for p in directory.iterdir() if p.name.endswith((".tar.gz", ".zip"))) + if not archives: + return [f"{directory}: no archives"] + for archive in archives: + match = ARCHIVE.match(archive.name) + if not match: + failures.append(f"{archive.name}: unexpected archive name") + continue + if match["version"] != version: + failures.append(f"{archive.name}: version is not {version}") + expected_ext = "zip" if match["os"] == "windows" else "tar.gz" + if match["ext"] != expected_ext: + failures.append(f"{archive.name}: {match['os']} archives must be .{expected_ext}") + checksum = archive.with_name(archive.name + ".sha256") + digest = hashlib.sha256(archive.read_bytes()).hexdigest() + if not checksum.is_file(): + failures.append(f"{archive.name}: missing .sha256") + elif checksum.read_text().split() != [digest, archive.name]: + failures.append(f"{archive.name}: .sha256 does not match") + + package = archive.name[: -len("." + match["ext"])] + with tempfile.TemporaryDirectory() as tmp: + tmp_path = pathlib.Path(tmp) + extract(archive, tmp_path) + entries = list(tmp_path.iterdir()) + if [e.name for e in entries] != [package] or not entries[0].is_dir(): + failures.append(f"{archive.name}: must contain exactly one directory, {package}/") + continue + root = entries[0] + exe = "nemo-speech.exe" if match["os"] == "windows" else "nemo-speech" + for required in ( + f"bin/{exe}", + "share/licenses/nemo-speech/LICENSE", + "share/licenses/nemo-speech/THIRD_PARTY_NOTICES.md", + ): + if not (root / required).is_file(): + failures.append(f"{archive.name}: missing {required}") + if match["arch"] == "x86_64": + failures += [f"{archive.name}: {f}" for f in scan_isa([root])] + print(f"checked {archive.name}", flush=True) + return failures + + +def main() -> None: + parser = argparse.ArgumentParser( + description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter + ) + sub = parser.add_subparsers(dest="command", required=True) + isa = sub.add_parser("isa") + isa.add_argument("paths", nargs="+", type=pathlib.Path) + archives = sub.add_parser("archives") + archives.add_argument("directory", type=pathlib.Path) + archives.add_argument("--version", required=True) + args = parser.parse_args() + + if args.command == "isa": + failures = scan_isa(args.paths) + else: + failures = check_archives(args.directory, args.version) + for failure in failures: + print(f"error: {failure}", file=sys.stderr) + if failures: + raise SystemExit(1) + print("OK") + + +if __name__ == "__main__": + main() diff --git a/scripts/release/package-linux.sh b/scripts/release/package-linux.sh new file mode 100755 index 0000000..deb3aca --- /dev/null +++ b/scripts/release/package-linux.sh @@ -0,0 +1,368 @@ +#!/usr/bin/env bash +# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-License-Identifier: Apache-2.0 +# Package a portable Linux installation for the binary installer. +set -euo pipefail + +ROOT="$(cd "$(dirname "${BASH_SOURCE[0]}")/../.." && pwd)" +install_prefix= +output_dir="$ROOT/release-artifacts" +backend= +artifact_backend= +max_glibc=2.31 +version= + +usage() { + cat <<'EOF' +Usage: scripts/release/package-linux.sh --install-prefix DIR --backend cpu|vulkan|cuda [OPTION ...] + +Options: + --artifact-backend NAME + Backend label used in the archive name + --output-dir DIR Destination for the archive and checksum + --version VERSION Override the version read from VERSION ("nightly" for + the nightly channel) + --max-glibc VERSION Reject binaries requiring a newer glibc (default: 2.31) + -h, --help + +The installed project and GCC runtimes are packaged together; the project must +be built with text normalization (-DNEMO_SPEECH_WITH_NORM=ON and the static +dependencies from scripts/build_itn_deps.sh). CUDA archives also include +libcudart. glibc, GPU drivers, and the Vulkan loader remain host +dependencies. x86_64 binaries must not use instructions beyond x86-64-v3 (AVX2), +and project binaries must find bundled libraries through DT_RPATH, which +LD_LIBRARY_PATH cannot override. +EOF +} + +while [[ $# -gt 0 ]]; do + case "$1" in + --install-prefix) + [[ $# -ge 2 ]] || { echo "error: --install-prefix requires a value" >&2; exit 2; } + install_prefix=$2 + shift 2 + ;; + --backend) + [[ $# -ge 2 ]] || { echo "error: --backend requires a value" >&2; exit 2; } + backend=$2 + shift 2 + ;; + --artifact-backend) + [[ $# -ge 2 ]] || { echo "error: --artifact-backend requires a value" >&2; exit 2; } + artifact_backend=$2 + shift 2 + ;; + --output-dir) + [[ $# -ge 2 ]] || { echo "error: --output-dir requires a value" >&2; exit 2; } + output_dir=$2 + shift 2 + ;; + --version) + [[ $# -ge 2 ]] || { echo "error: --version requires a value" >&2; exit 2; } + version=$2 + shift 2 + ;; + --max-glibc) + [[ $# -ge 2 ]] || { echo "error: --max-glibc requires a value" >&2; exit 2; } + max_glibc=$2 + shift 2 + ;; + -h|--help) + usage + exit 0 + ;; + *) + echo "error: unknown option '$1'" >&2 + usage >&2 + exit 2 + ;; + esac +done + +[[ -n "$install_prefix" ]] || { echo "error: --install-prefix is required" >&2; exit 2; } +[[ -d "$install_prefix" ]] || { echo "error: install prefix does not exist: $install_prefix" >&2; exit 1; } +[[ -x "$install_prefix/bin/nemo-speech" ]] || { + echo "error: install prefix does not contain bin/nemo-speech" >&2 + exit 1 +} +case "$backend" in + cpu|vulkan|cuda) ;; + *) + echo "error: --backend must be cpu, vulkan, or cuda" >&2 + exit 2 + ;; +esac +if [[ -z "$artifact_backend" ]]; then + artifact_backend=$backend +fi +case "$backend:$artifact_backend" in + cpu:cpu|vulkan:vulkan|cuda:cuda|cuda:cuda12|cuda:cuda13) ;; + *) + echo "error: invalid artifact backend '$artifact_backend' for '$backend'" >&2 + exit 2 + ;; +esac +[[ "$max_glibc" =~ ^[0-9]+(\.[0-9]+)+$ ]] || { + echo "error: --max-glibc must be a dotted version" >&2 + exit 2 +} + +if [[ -z "$version" ]]; then + version="$(sed -n 's/^NEMO_SPEECH_VERSION:[[:space:]]*//p' "$ROOT/VERSION")" +fi +[[ "$version" == nightly || "$version" =~ ^[0-9]+\.[0-9]+\.[0-9]+([-.][0-9A-Za-z.]+)?$ ]] || { + echo "error: invalid release version '$version'" >&2 + exit 1 +} + +case "$(uname -m)" in + x86_64|amd64) arch=x86_64 ;; + aarch64|arm64) arch=aarch64 ;; + *) + echo "error: unsupported architecture: $(uname -m)" >&2 + exit 1 + ;; +esac + +for command_name in cc c++ ldd readelf strip tar gzip sha256sum sort; do + command -v "$command_name" >/dev/null 2>&1 || { + echo "error: required command not found: $command_name" >&2 + exit 1 + } +done + +package_name="nemo-speech-${version}-linux-${arch}-${artifact_backend}" +archive="${output_dir}/${package_name}.tar.gz" +work_dir="$(mktemp -d "${TMPDIR:-/tmp}/nemo-speech-package.XXXXXX")" +trap 'rm -rf "$work_dir"' EXIT HUP INT TERM +package_root="$work_dir/$package_name" + +mkdir -p "$package_root" "$output_dir" +cp -a "$install_prefix"/. "$package_root"/ + +forbidden_payload="$(find "$package_root" \ + \( -name 'riva_server' -o -name 'riva_server.exe' \ + -o -name 'libgrpc*.so*' -o -path '*/riva-common' \) \ + -print -quit)" +[[ -z "$forbidden_payload" ]] || { + echo "error: release contains Riva gRPC payload: $forbidden_payload" >&2 + exit 1 +} + +sentencepiece_license_dir="$package_root/share/licenses/nemo-speech/third_party/sentencepiece" +for license_file in LICENSE absl-LICENSE darts-clone-LICENSE protobuf-lite-LICENSE; do + [[ -f "$sentencepiece_license_dir/$license_file" ]] || { + echo "error: release is missing SentencePiece notice: $license_file" >&2 + exit 1 + } +done + +# Release archives ship ITN/TN with its statically linked dependencies. +[[ -n "$(find "$package_root/lib" -maxdepth 1 -name 'libnemo_speech_text_normalization.so*' -print -quit)" ]] || { + echo "error: release is missing text normalization; configure with -DNEMO_SPEECH_WITH_NORM=ON" >&2 + exit 1 +} +for license_file in openfst/COPYING sparrowhawk/LICENSE protobuf/LICENSE re2/LICENSE; do + [[ -f "$package_root/share/licenses/nemo-speech/third_party/$license_file" ]] || { + echo "error: release is missing text normalization notice: $license_file" >&2 + exit 1 + } +done + +runtime_license_dir="$package_root/share/licenses/nemo-speech/third_party/gcc-runtime" +mkdir -p "$package_root/lib" "$runtime_license_dir" +for runtime in libstdc++.so.6 libgcc_s.so.1 libgomp.so.1 libatomic.so.1; do + if [[ "$runtime" == libstdc++* ]]; then + compiler=c++ + else + compiler=cc + fi + runtime_path="$("$compiler" -print-file-name="$runtime")" + [[ "$runtime_path" != "$runtime" && -f "$runtime_path" ]] || { + echo "error: $compiler could not locate $runtime" >&2 + exit 1 + } + cp -L "$runtime_path" "$package_root/lib/$runtime" +done + +compiler_major="$(cc -dumpfullversion -dumpversion | cut -d. -f1)" +runtime_copyright="/usr/share/doc/gcc-${compiler_major}-base/copyright" +if [[ ! -f "$runtime_copyright" ]]; then + runtime_copyright="$(find /usr/share/doc -maxdepth 2 -type f \ + \( -path '*/gcc-*-base/copyright' -o -path '*/libstdc++*/copyright' \) \ + -print | sort | head -n 1)" +fi +[[ -n "$runtime_copyright" && -f "$runtime_copyright" ]] || { + echo "error: could not locate the GCC runtime copyright file" >&2 + exit 1 +} +cp "$runtime_copyright" "$runtime_license_dir/copyright" + +if [[ "$backend" == cuda ]]; then + cuda_home="${CUDA_HOME:-${CUDA_PATH:-}}" + if [[ -z "$cuda_home" ]] && command -v nvcc >/dev/null 2>&1; then + cuda_home="$(cd "$(dirname "$(command -v nvcc)")/.." && pwd)" + fi + [[ -n "$cuda_home" && -d "$cuda_home" ]] || { + echo "error: CUDA_HOME is required when packaging a CUDA build" >&2 + exit 1 + } + cuda_lib_dir= + case "$arch" in + x86_64) + cuda_targets=(x86_64-linux) + ;; + aarch64) + cuda_targets=(aarch64-linux sbsa-linux) + ;; + esac + for cuda_target in "${cuda_targets[@]}"; do + candidate="$cuda_home/targets/$cuda_target/lib" + if [[ -d "$candidate" ]]; then + cuda_lib_dir="$candidate" + break + fi + done + [[ -n "$cuda_lib_dir" ]] || { + echo "error: CUDA target libraries were not found for $arch under $cuda_home/targets" >&2 + exit 1 + } + cudart_path="$(find -L "$cuda_lib_dir" -maxdepth 1 -type f \ + -name 'libcudart.so.*' -print | sort -V | tail -n 1)" + [[ -n "$cudart_path" ]] || { + echo "error: libcudart was not found under $cuda_lib_dir" >&2 + exit 1 + } + cudart_path="$(readlink -f "$cudart_path")" + cudart_soname="$(readelf -d "$cudart_path" | + sed -n 's/.*Library soname: \[\([^]]*\)\].*/\1/p')" + [[ -n "$cudart_soname" ]] || { + echo "error: libcudart does not declare a SONAME: $cudart_path" >&2 + exit 1 + } + cp -L "$cudart_path" "$package_root/lib/$cudart_soname" + + cuda_version="$("$cuda_home/bin/nvcc" --version | + sed -n 's/.*release \([0-9]*\)\.\([0-9]*\).*/\1-\2/p' | head -n 1)" + actual_cuda_major="${cuda_version%%-*}" + if [[ "$artifact_backend" == cuda12 || "$artifact_backend" == cuda13 ]]; then + expected_cuda_major="${artifact_backend#cuda}" + [[ "$actual_cuda_major" == "$expected_cuda_major" ]] || { + echo "error: artifact label '$artifact_backend' does not match CUDA $cuda_version" >&2 + exit 1 + } + fi + cublas_soname="libcublas.so.${actual_cuda_major}" + cublas_shim="$package_root/lib/$cublas_soname" + [[ -f "$cublas_shim" ]] || { + echo "error: CUDA release does not contain the $cublas_soname shim" >&2 + exit 1 + } + actual_cublas_soname="$(readelf -d "$cublas_shim" | + sed -n 's/.*Library soname: \[\([^]]*\)\].*/\1/p')" + [[ "$actual_cublas_soname" == "$cublas_soname" ]] || { + echo "error: cuBLAS shim SONAME is '$actual_cublas_soname'; expected '$cublas_soname'" >&2 + exit 1 + } + readelf --version-info "$cublas_shim" | grep -Fq "$cublas_soname" || { + echo "error: cuBLAS shim does not export the $cublas_soname symbol version" >&2 + exit 1 + } + ggml_cuda="$(find -L "$package_root/lib" -maxdepth 1 -type f \ + -name 'libggml-cuda.so.*' -print | sort -V | tail -n 1)" + [[ -n "$ggml_cuda" ]] || { + echo "error: CUDA release does not contain libggml-cuda" >&2 + exit 1 + } + required_cublas="$(readelf -d "$ggml_cuda" | + sed -n 's/.*Shared library: \[\(libcublas\.so\.[^]]*\)\].*/\1/p')" + [[ "$required_cublas" == "$cublas_soname" ]] || { + echo "error: libggml-cuda requires '$required_cublas'; expected '$cublas_soname'" >&2 + exit 1 + } + cuda_license="/usr/share/doc/cuda-cudart-${cuda_version}/copyright" + [[ -f "$cuda_license" ]] || { + echo "error: CUDA runtime license was not found: $cuda_license" >&2 + exit 1 + } + install -Dm0644 "$cuda_license" \ + "$package_root/share/licenses/nemo-speech/nvidia/cuda-runtime/copyright" +fi + +elf_candidates="$work_dir/elf-candidates" +find "$package_root/bin" "$package_root/lib" -type f -print0 > "$elf_candidates" +while IFS= read -r -d '' file; do + if readelf -h "$file" >/dev/null 2>&1; then + strip --strip-unneeded "$file" + fi +done < "$elf_candidates" + +abi_versions="$work_dir/glibc-versions" +missing_dependencies="$work_dir/missing-dependencies" +: > "$abi_versions" +: > "$missing_dependencies" +while IFS= read -r -d '' file; do + readelf -h "$file" >/dev/null 2>&1 || continue + readelf --version-info "$file" 2>/dev/null | + grep -Eo 'GLIBC_[0-9]+(\.[0-9]+)*' | + sed 's/^GLIBC_//' >> "$abi_versions" || true + LD_LIBRARY_PATH="$package_root/lib${LD_LIBRARY_PATH:+:$LD_LIBRARY_PATH}" \ + ldd "$file" 2>/dev/null | + awk -v file="$file" -v backend="$backend" \ + '$2 == "not" && $3 == "found" { + if (!(backend == "cuda" && $1 == "libcuda.so.1")) { + print file ": " $1 + } + }' \ + >> "$missing_dependencies" || true +done < "$elf_candidates" + +if [[ -s "$missing_dependencies" ]]; then + echo "error: packaged ELF dependencies are unresolved:" >&2 + sed 's/^/ /' "$missing_dependencies" >&2 + exit 1 +fi + +highest_glibc="$(sort -Vu "$abi_versions" | tail -n 1)" +[[ -n "$highest_glibc" ]] || { + echo "error: no glibc requirements were found in the package" >&2 + exit 1 +} +if [[ "$highest_glibc" != "$max_glibc" ]] && + [[ "$(printf '%s\n%s\n' "$highest_glibc" "$max_glibc" | sort -V | tail -n 1)" == "$highest_glibc" ]]; then + echo "error: package requires GLIBC_$highest_glibc; maximum is GLIBC_$max_glibc" >&2 + exit 1 +fi + +runpath_files="$work_dir/runpath-files" +: > "$runpath_files" +while IFS= read -r -d '' file; do + readelf -h "$file" >/dev/null 2>&1 || continue + if readelf -d "$file" 2>/dev/null | grep -q '(RUNPATH)'; then + echo "$file" >> "$runpath_files" + fi +done < "$elf_candidates" +if [[ -s "$runpath_files" ]]; then + echo "error: these binaries use DT_RUNPATH; link with -Wl,--disable-new-dtags:" >&2 + sed 's/^/ /' "$runpath_files" >&2 + exit 1 +fi + +if [[ "$arch" == x86_64 ]]; then + python3 "$ROOT/scripts/release/check_release.py" isa "$package_root" +fi + +source_date_epoch="${SOURCE_DATE_EPOCH:-0}" +tar --sort=name \ + --mtime="@$source_date_epoch" \ + --owner=0 --group=0 --numeric-owner \ + -C "$work_dir" -cf - "$package_name" | + gzip -n -9 > "$archive" +( + cd "$output_dir" + sha256sum "$(basename "$archive")" > "$(basename "$archive").sha256" +) + +echo "Created: $archive" +echo "SHA-256: $archive.sha256" +echo "Required: GLIBC_$highest_glibc or newer" diff --git a/scripts/release/package-macos.sh b/scripts/release/package-macos.sh new file mode 100755 index 0000000..20b02c6 --- /dev/null +++ b/scripts/release/package-macos.sh @@ -0,0 +1,170 @@ +#!/usr/bin/env bash +# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-License-Identifier: Apache-2.0 +# Package a macOS installation for the binary installer. +set -euo pipefail + +ROOT="$(cd "$(dirname "${BASH_SOURCE[0]}")/../.." && pwd)" +install_prefix= +output_dir="$ROOT/release-artifacts" +backend= +arch= +version= +min_macos=13.3 + +usage() { + cat <<'EOF' +Usage: scripts/release/package-macos.sh --install-prefix DIR --backend cpu|metal --arch aarch64|x86_64 [OPTION ...] + +Options: + --output-dir DIR Destination for the archive and checksum + --version VERSION Override the version read from VERSION ("nightly" for + the nightly channel) + --min-macos VERSION Reject binaries built for a newer macOS (default: 13.3) + -h, --help + +The project must be built with text normalization (-DNEMO_SPEECH_WITH_NORM=ON). +Every Mach-O file must be built for --arch and link only system libraries or +libraries inside the package. Binaries are stripped of local symbols and +ad-hoc signed. +EOF +} + +while [[ $# -gt 0 ]]; do + case "$1" in + --install-prefix|--backend|--arch|--output-dir|--version|--min-macos) + [[ $# -ge 2 ]] || { echo "error: $1 requires a value" >&2; exit 2; } + case "$1" in + --install-prefix) install_prefix=$2 ;; + --backend) backend=$2 ;; + --arch) arch=$2 ;; + --output-dir) output_dir=$2 ;; + --version) version=$2 ;; + --min-macos) min_macos=$2 ;; + esac + shift 2 + ;; + -h|--help) usage; exit 0 ;; + *) echo "error: unknown option: $1" >&2; usage >&2; exit 2 ;; + esac +done + +[[ -n "$install_prefix" && -x "$install_prefix/bin/nemo-speech" ]] || { + echo "error: --install-prefix must contain bin/nemo-speech" >&2 + exit 2 +} +case "$backend" in cpu|metal) ;; *) echo "error: --backend must be cpu or metal" >&2; exit 2 ;; esac +case "$arch" in + aarch64) macho_arch=arm64 ;; + x86_64) macho_arch=x86_64 ;; + *) echo "error: --arch must be aarch64 or x86_64" >&2; exit 2 ;; +esac +if [[ "$backend" == metal && "$arch" != aarch64 ]]; then + echo "error: Metal archives are built for aarch64 only" >&2 + exit 2 +fi +if [[ -z "$version" ]]; then + version="$(sed -n 's/^NEMO_SPEECH_VERSION:[[:space:]]*//p' "$ROOT/VERSION")" +fi +[[ "$version" == nightly || "$version" =~ ^[0-9]+\.[0-9]+\.[0-9]+([-.][0-9A-Za-z.]+)?$ ]] || { + echo "error: invalid release version '$version'" >&2 + exit 1 +} +for command_name in otool lipo codesign strip tar gzip shasum; do + command -v "$command_name" >/dev/null 2>&1 || { + echo "error: required command not found: $command_name" >&2 + exit 1 + } +done + +package_name="nemo-speech-${version}-macos-${arch}-${backend}" +archive="${output_dir}/${package_name}.tar.gz" +work_dir="$(mktemp -d "${TMPDIR:-/tmp}/nemo-speech-package.XXXXXX")" +trap 'rm -rf "$work_dir"' EXIT HUP INT TERM +package_root="$work_dir/$package_name" + +mkdir -p "$package_root" "$output_dir" +cp -R "$install_prefix"/. "$package_root"/ + +if find "$package_root" \( -name 'riva_server' -o -name 'libgrpc*' \) -print | grep -q .; then + echo "error: release contains Riva gRPC payload" >&2 + exit 1 +fi +sentencepiece_license_dir="$package_root/share/licenses/nemo-speech/third_party/sentencepiece" +[[ -f "$sentencepiece_license_dir/LICENSE" ]] || { + echo "error: release is missing the SentencePiece notice" >&2 + exit 1 +} + +# Release archives ship ITN/TN with its statically linked dependencies. +[[ -f "$package_root/lib/libnemo_speech_text_normalization.dylib" ]] || { + echo "error: release is missing text normalization; configure with -DNEMO_SPEECH_WITH_NORM=ON" >&2 + exit 1 +} +for license_file in openfst/COPYING sparrowhawk/LICENSE protobuf/LICENSE re2/LICENSE; do + [[ -f "$package_root/share/licenses/nemo-speech/third_party/$license_file" ]] || { + echo "error: release is missing text normalization notice: $license_file" >&2 + exit 1 + } +done + +version_le() { # version_le A B: A <= B + [[ "$(printf '%s\n%s\n' "$1" "$2" | sort -t. -k1,1n -k2,2n -k3,3n | head -n 1)" == "$1" ]] +} + +failures="$work_dir/failures" +: > "$failures" +macho_files="$work_dir/macho-files" +find "$package_root/bin" "$package_root/lib" -type f -print | LC_ALL=C sort > "$macho_files" +while IFS= read -r file; do + otool -h "$file" >/dev/null 2>&1 || continue + archs="$(lipo -archs "$file" 2>/dev/null || true)" + [[ "$archs" == "$macho_arch" ]] || echo "$file: built for '$archs', expected $macho_arch" >> "$failures" + + minos="$(otool -l "$file" | awk '$1 == "minos" { print $2; exit }')" + if [[ -n "$minos" ]] && ! version_le "$minos" "$min_macos"; then + echo "$file: requires macOS $minos, newer than $min_macos" >> "$failures" + fi + + # Dependencies: system libraries or @rpath/@loader_path entries present in the package. + otool -L "$file" | tail -n +2 | awk '{ print $1 }' | while IFS= read -r dependency; do + case "$dependency" in + /usr/lib/*|/System/Library/*) ;; + @rpath/*|@loader_path/*|@executable_path/*) + name="${dependency##*/}" + [[ -e "$package_root/lib/$name" || -e "$package_root/bin/$name" ]] || + echo "$file: $dependency is not in the package" >> "$failures" + ;; + *) echo "$file: links $dependency, which is outside the package" >> "$failures" ;; + esac + done +done < "$macho_files" + +if [[ -s "$failures" ]]; then + echo "error: the package is not self-contained:" >&2 + sed 's/^/ /' "$failures" >&2 + exit 1 +fi + +# Strip local symbols, then ad-hoc sign: install-time rpath edits invalidate the +# linker's signature, and arm64 macOS refuses to run unsigned code. +while IFS= read -r file; do + otool -h "$file" >/dev/null 2>&1 || continue + strip -x "$file" + codesign --force --sign - "$file" +done < "$macho_files" + +source_date_epoch="${SOURCE_DATE_EPOCH:-0}" +stamp="$(date -u -r "$source_date_epoch" +%Y%m%d%H%M.%S)" +find "$package_root" -exec touch -h -t "$stamp" {} + +(cd "$work_dir" && find "$package_name" -print | LC_ALL=C sort > "$work_dir/files") +COPYFILE_DISABLE=1 tar --no-mac-metadata --no-xattrs --uid 0 --gid 0 --uname root --gname wheel \ + -C "$work_dir" -n -T "$work_dir/files" -cf - | gzip -n -9 > "$archive" +( + cd "$output_dir" + shasum -a 256 "$(basename "$archive")" > "$(basename "$archive").sha256" +) + +echo "Created: $archive" +echo "SHA-256: $archive.sha256" +echo "Requires: macOS $min_macos or newer ($macho_arch)" diff --git a/scripts/windows/build.ps1 b/scripts/windows/build.ps1 index 07a7bfa..053b75b 100644 --- a/scripts/windows/build.ps1 +++ b/scripts/windows/build.ps1 @@ -77,6 +77,9 @@ .PARAMETER Architecture Target architecture: auto (the host), x64, or arm64. +.PARAMETER CMakeArgs + Extra arguments appended to the CMake configure command, for example + -CMakeArgs '-DGGML_NATIVE=OFF'. .EXAMPLE pwsh scripts\windows\build.ps1 -Backend cuda -Profile server @@ -112,7 +115,8 @@ param( [ValidateSet('auto', 'msvc', 'clang-cl')] [string]$Compiler = 'auto', [int]$Jobs = 0, - [switch]$DryRun + [switch]$DryRun, + [string[]]$CMakeArgs = @() ) $ErrorActionPreference = 'Stop' @@ -443,6 +447,7 @@ switch ($Backend) { } } +$cmakeArgs += $CMakeArgs Write-Host "==> cmake $($cmakeArgs -join ' ')" & cmake @cmakeArgs if ($LASTEXITCODE -ne 0) { throw "CMake configure failed ($LASTEXITCODE)" } diff --git a/scripts/windows/package-release.ps1 b/scripts/windows/package-release.ps1 new file mode 100644 index 0000000..5529ddd --- /dev/null +++ b/scripts/windows/package-release.ps1 @@ -0,0 +1,109 @@ +# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-License-Identifier: Apache-2.0 +<# +.SYNOPSIS + Package a Windows build for the binary installer. +.DESCRIPTION + Installs -BuildDir into a directory named after the archive, checks that the + package is self-contained, and writes + nemo-speech--windows--.zip and its .sha256 to + -OutputDir. Every DLL a packaged binary imports must ship in bin\ or be a + Windows or GPU driver library. +.PARAMETER BuildDir + Build tree produced by scripts\windows\build.ps1. +.PARAMETER Backend + cpu, vulkan, or cuda. +.PARAMETER Version + Release version, or nightly. Defaults to the one in VERSION. +.PARAMETER OutputDir + Destination for the archive and checksum (default: release-artifacts). +.EXAMPLE + pwsh scripts\windows\package-release.ps1 -BuildDir build\release-cuda -Backend cuda +#> +[CmdletBinding()] +param( + [Parameter(Mandatory)] + [string]$BuildDir, + [Parameter(Mandatory)] + [ValidateSet('cpu', 'vulkan', 'cuda')] + [string]$Backend, + [string]$Version, + [string]$OutputDir +) + +$ErrorActionPreference = 'Stop' +$RepoRoot = Split-Path -Parent (Split-Path -Parent $PSScriptRoot) +if (-not $OutputDir) { $OutputDir = Join-Path $RepoRoot 'release-artifacts' } +if (-not $Version) { + $Version = ((Get-Content (Join-Path $RepoRoot 'VERSION')) -match '^NEMO_SPEECH_VERSION:' | + Select-Object -First 1) -replace '^NEMO_SPEECH_VERSION:\s*', '' +} +if ($Version -ne 'nightly' -and $Version -notmatch '^[0-9]+\.[0-9]+\.[0-9]+([-.][0-9A-Za-z.]+)?$') { + throw "invalid release version '$Version'" +} +$arch = switch ([System.Runtime.InteropServices.RuntimeInformation]::OSArchitecture) { + 'X64' { 'x86_64' } + 'Arm64' { 'aarch64' } + default { throw "unsupported architecture: $_" } +} + +$name = "nemo-speech-$Version-windows-$arch-$Backend" +$staging = Join-Path ([System.IO.Path]::GetTempPath()) "nemo-speech-package-$PID" +$root = Join-Path $staging $name +New-Item -ItemType Directory -Force -Path $root, $OutputDir | Out-Null +try { + & cmake --install $BuildDir --config Release --prefix $root + if ($LASTEXITCODE -ne 0) { throw "cmake --install failed ($LASTEXITCODE)" } + + $bin = Join-Path $root 'bin' + foreach ($required in @('bin\nemo-speech.exe', 'share\licenses\nemo-speech\LICENSE', + 'share\licenses\nemo-speech\THIRD_PARTY_NOTICES.md')) { + if (-not (Test-Path (Join-Path $root $required))) { throw "package is missing $required" } + } + if (Get-ChildItem -Recurse $root -Include 'riva_server.exe', 'grpc*.dll') { + throw 'release contains Riva gRPC payload' + } + if ($Backend -eq 'cuda') { + if (-not (Get-ChildItem $bin -Filter 'ggml-cuda.dll')) { throw 'CUDA package is missing ggml-cuda.dll' } + if (-not (Get-ChildItem $bin -Filter 'cublas64_*.dll')) { + throw 'CUDA package is missing the cuBLAS shim; build with -CublasShim' + } + } + + # Every imported DLL must be bundled or provided by Windows or the GPU driver. + $vswhere = "${env:ProgramFiles(x86)}\Microsoft Visual Studio\Installer\vswhere.exe" + $dumpbin = & $vswhere -latest -products * -find '**\Hostx64\x64\dumpbin.exe' | Select-Object -First 1 + if (-not $dumpbin) { throw 'dumpbin.exe was not found (Visual Studio C++ tools are required)' } + $system = '^(api-ms-win-.*|ext-ms-.*|kernel32|kernelbase|user32|gdi32|advapi32|shell32|ole32|oleaut32|' + + 'ws2_32|wsock32|mswsock|bcrypt|crypt32|ncrypt|secur32|dbghelp|shlwapi|winmm|ntdll|userenv|' + + 'psapi|version|iphlpapi|setupapi|cfgmgr32|comctl32|comdlg32|rpcrt4|dnsapi|powrprof|winhttp|' + + 'wininet|normaliz|avrt|ucrtbase|nvcuda|nvml|vulkan-1)\.dll$' + $bundled = @{} + Get-ChildItem $bin -Filter '*.dll' | ForEach-Object { $bundled[$_.Name.ToLowerInvariant()] = $true } + $missing = @() + foreach ($file in Get-ChildItem $bin -Include '*.dll', '*.exe' -Recurse) { + $dependencies = & $dumpbin /nologo /dependents $file.FullName | ForEach-Object { + if ($_ -match '^\s+(\S+\.dll)\s*$') { $Matches[1].ToLowerInvariant() } + } + foreach ($dependency in $dependencies) { + if (-not $bundled.ContainsKey($dependency) -and $dependency -notmatch $system) { + $missing += "$($file.Name): $dependency" + } + } + } + if ($missing) { + throw "the package is not self-contained:`n $($missing -join "`n ")" + } + + $zip = Join-Path (Resolve-Path $OutputDir) "$name.zip" + if (Test-Path $zip) { Remove-Item $zip } + Add-Type -AssemblyName System.IO.Compression.FileSystem + [System.IO.Compression.ZipFile]::CreateFromDirectory( + $root, $zip, [System.IO.Compression.CompressionLevel]::Optimal, $true) + $hash = (Get-FileHash -Algorithm SHA256 $zip).Hash.ToLowerInvariant() + [System.IO.File]::WriteAllText("$zip.sha256", "$hash $name.zip`n") + Write-Host "Created: $zip" + Write-Host "SHA-256: $zip.sha256" +} finally { + Remove-Item -Recurse -Force -ErrorAction SilentlyContinue $staging +} diff --git a/src/common/CMakeLists.txt b/src/common/CMakeLists.txt index 9b9a18d..2ab30c0 100644 --- a/src/common/CMakeLists.txt +++ b/src/common/CMakeLists.txt @@ -23,9 +23,11 @@ set_target_properties(nemo_speech_common PROPERTIES POSITION_INDEPENDENT_CODE ON # Shared WFST grammar runtime for ASR ITN and TTS TN. Keep this separate from # nemo_speech_common so builds that do not request text normalization retain -# no OpenFST/Sparrowhawk dependency. +# no OpenFST/Sparrowhawk dependency. It is a shared library of its own so that +# its protobuf runtime stays out of the ASR library, which links SentencePiece's +# bundled protobuf, and so ASR and TTS share one copy. if(NEMO_SPEECH_WITH_NORM) - add_library(nemo_speech_text_normalization STATIC + add_library(nemo_speech_text_normalization SHARED text_normalization/fst_normalizer.cpp) target_compile_definitions( nemo_speech_text_normalization PUBLIC NEMO_SPEECH_WITH_NORM=1) @@ -34,46 +36,70 @@ if(NEMO_SPEECH_WITH_NORM) find_library(SPARROWHAWK_LIB sparrowhawk HINTS "${_NEMO_SPEECH_ITN_PREFIX}/lib" REQUIRED) find_library(FSTFAR_LIB fstfar HINTS "${_NEMO_SPEECH_ITN_PREFIX}/lib" REQUIRED) find_library(FST_LIB fst HINTS "${_NEMO_SPEECH_ITN_PREFIX}/lib" REQUIRED) - find_library(RE2_LIB re2 REQUIRED) + find_library(RE2_LIB re2 HINTS "${_NEMO_SPEECH_ITN_PREFIX}/lib" REQUIRED) find_path(SPARROWHAWK_INCLUDE_DIR sparrowhawk/normalizer.h HINTS "${_NEMO_SPEECH_ITN_PREFIX}/include" REQUIRED) find_path(OPENFST_INCLUDE_DIR fst/fst.h HINTS "${_NEMO_SPEECH_ITN_PREFIX}/include" REQUIRED) + find_package(Threads REQUIRED) - find_package(Protobuf REQUIRED) - find_library(ITN_PROTOBUF_LIB - NAMES libprotobuf.so.25.3.0 libprotobuf.so.25 - PATHS /lib /lib/x86_64-linux-gnu /usr/lib /usr/lib/x86_64-linux-gnu - NO_DEFAULT_PATH) - if(NOT ITN_PROTOBUF_LIB) - if(TARGET protobuf::libprotobuf) - set(ITN_PROTOBUF_LIB protobuf::libprotobuf) - else() - set(ITN_PROTOBUF_LIB protobuf) + # scripts/build_itn_deps.sh with STATIC=1 installs protobuf and RE2 into + # the same prefix; otherwise use the system protobuf. + find_library(ITN_PROTOBUF_LIB protobuf + PATHS "${_NEMO_SPEECH_ITN_PREFIX}/lib" NO_DEFAULT_PATH) + set(ABSL_LIBS) + if(ITN_PROTOBUF_LIB) + set(ITN_PROTOBUF_LIBS ${ITN_PROTOBUF_LIB}) + else() + find_package(Protobuf REQUIRED) + find_library(ITN_PROTOBUF_LIB + NAMES libprotobuf.so.25.3.0 libprotobuf.so.25 + PATHS /lib /lib/x86_64-linux-gnu /usr/lib /usr/lib/x86_64-linux-gnu + NO_DEFAULT_PATH) + if(NOT ITN_PROTOBUF_LIB) + if(TARGET protobuf::libprotobuf) + set(ITN_PROTOBUF_LIB protobuf::libprotobuf) + else() + set(ITN_PROTOBUF_LIB protobuf) + endif() + endif() + find_library(UTF8_RANGE_LIB NAMES utf8_range_lib PATHS /usr/lib /lib) + file(GLOB ABSL_LIBS /usr/lib/libabsl_*.so /lib/libabsl_*.so) + set(ITN_PROTOBUF_LIBS ${ITN_PROTOBUF_LIB}) + if(UTF8_RANGE_LIB) + list(APPEND ITN_PROTOBUF_LIBS ${UTF8_RANGE_LIB}) endif() - endif() - find_library(UTF8_RANGE_LIB NAMES utf8_range_lib PATHS /usr/lib /lib) - file(GLOB ABSL_LIBS /usr/lib/libabsl_*.so /lib/libabsl_*.so) - set(ITN_PROTOBUF_LIBS ${ITN_PROTOBUF_LIB}) - if(UTF8_RANGE_LIB) - list(APPEND ITN_PROTOBUF_LIBS ${UTF8_RANGE_LIB}) endif() target_include_directories(nemo_speech_text_normalization PUBLIC ${CMAKE_CURRENT_SOURCE_DIR}/text_normalization + PRIVATE ${CMAKE_CURRENT_SOURCE_DIR}/text_normalization/compat) target_include_directories(nemo_speech_text_normalization SYSTEM - PUBLIC - ${OPENFST_INCLUDE_DIR} PRIVATE + ${OPENFST_INCLUDE_DIR} ${SPARROWHAWK_INCLUDE_DIR}) - target_link_libraries(nemo_speech_text_normalization PUBLIC - ${SPARROWHAWK_LIB} ${FSTFAR_LIB} ${FST_LIB} - ${ITN_PROTOBUF_LIBS} - -Wl,--as-needed ${ABSL_LIBS} -Wl,--no-as-needed - ${RE2_LIB}) - set_target_properties( - nemo_speech_text_normalization PROPERTIES POSITION_INDEPENDENT_CODE ON) + # OpenFST registers its FST types from static initializers; keep them when + # linking the static archive. + set(_NEMO_SPEECH_FST_LINK ${FST_LIB}) + if(FST_LIB MATCHES "\\${CMAKE_STATIC_LIBRARY_SUFFIX}$") + set(_NEMO_SPEECH_FST_LINK "$") + endif() + target_link_libraries(nemo_speech_text_normalization PRIVATE + ${SPARROWHAWK_LIB} ${FSTFAR_LIB} ${_NEMO_SPEECH_FST_LINK} + ${ITN_PROTOBUF_LIBS}) + if(ABSL_LIBS) + target_link_libraries(nemo_speech_text_normalization PRIVATE + -Wl,--as-needed ${ABSL_LIBS} -Wl,--no-as-needed) + endif() + target_link_libraries(nemo_speech_text_normalization PRIVATE + ${RE2_LIB} Threads::Threads ${CMAKE_DL_LIBS}) + if(UNIX AND NOT APPLE) + # Keep the symbols of statically linked dependencies private. + target_link_options(nemo_speech_text_normalization PRIVATE "LINKER:--exclude-libs,ALL") + endif() + install(TARGETS nemo_speech_text_normalization + LIBRARY DESTINATION ${CMAKE_INSTALL_LIBDIR}) file(GLOB _NEMO_SPEECH_NORM_RUNTIME_LIBS "${_NEMO_SPEECH_ITN_PREFIX}/lib/libsparrowhawk${CMAKE_SHARED_LIBRARY_SUFFIX}*" @@ -84,4 +110,10 @@ if(NEMO_SPEECH_WITH_NORM) "${_NEMO_SPEECH_ITN_PREFIX}/lib/libfstfar.*${CMAKE_SHARED_LIBRARY_SUFFIX}") install(FILES ${_NEMO_SPEECH_NORM_RUNTIME_LIBS} DESTINATION ${CMAKE_INSTALL_LIBDIR}) + set(_NEMO_SPEECH_ITN_LICENSE_DIR + "${_NEMO_SPEECH_ITN_PREFIX}/share/licenses/nemo-speech/third_party") + if(EXISTS "${_NEMO_SPEECH_ITN_LICENSE_DIR}") + install(DIRECTORY "${_NEMO_SPEECH_ITN_LICENSE_DIR}/" + DESTINATION "${NEMO_SPEECH_LICENSE_INSTALL_DIR}/third_party") + endif() endif() diff --git a/tests/ci/model_smoke.py b/tests/ci/model_smoke.py index 6dde2fb..07b2130 100644 --- a/tests/ci/model_smoke.py +++ b/tests/ci/model_smoke.py @@ -3,6 +3,9 @@ # SPDX-License-Identifier: Apache-2.0 """CLI smoke test with real models: ASR, diarization, and a TTS round trip. +With grammar directories, it also checks text normalization: TN must turn digits +into words before synthesis, and ITN must turn them back in the transcript. + Models are pulled by the CLI from the indexed Hugging Face repos into NEMO_SPEECH_MODEL_DIR, so a warm cache makes this cheap. Text comparisons use a similarity ratio rather than exact match so backend numerics cannot flake it. @@ -24,6 +27,8 @@ "ask what you can do for your country" ) TTS_TEXT = "The quick brown fox jumps over the lazy dog." +NORM_TEXT = "I paid 25 dollars for 3 tickets." +NORM_SPOKEN = "i paid twenty five dollars for three tickets" FIXTURES = pathlib.Path(__file__).resolve().parents[2] / "test_files" / "diar" @@ -126,16 +131,36 @@ def main() -> None: parser.add_argument("--backend", default="cpu") parser.add_argument("--audio", required=True) parser.add_argument("--skip-tts", action="store_true") + parser.add_argument("--itn-model-dir", help="ITN grammar directory") + parser.add_argument("--tn-model-dir", help="TN grammar directory") args = parser.parse_args() with tempfile.TemporaryDirectory(prefix="nemo-speech-smoke-") as temporary: work = pathlib.Path(temporary) + if args.itn_model_dir or args.tn_model_dir: + features = json.loads(run(args.binary, "doctor", "--json"))["features"] + check(features.get("text_normalization") is True, "build includes text normalization") + text = run(args.binary, "transcribe", args.audio, "--backend", args.backend) ratio = similarity(text, JFK_TEXT) print(f"asr transcript: {text.strip()!r} (similarity {ratio:.2f})") check(ratio >= 0.9, "ASR offline transcript matches the reference") + if args.itn_model_dir: + text = run( + args.binary, + "transcribe", + args.audio, + "--backend", + args.backend, + "--itn-model-dir", + args.itn_model_dir, + ) + ratio = similarity(text, JFK_TEXT) + print(f"asr transcript with itn: {text.strip()!r} (similarity {ratio:.2f})") + check(ratio >= 0.9, "ASR transcript with ITN matches the reference") + streamed = run( args.binary, "transcribe", @@ -219,6 +244,36 @@ def main() -> None: print(f"tts round trip: {round_trip.strip()!r} (similarity {ratio:.2f})") check(ratio >= 0.8, "TTS output transcribes back to the input text") + if args.tn_model_dir: + wav = work / "tn.wav" + run( + args.binary, + "synthesize", + NORM_TEXT, + "--backend", + args.backend, + "--tn-model-dir", + args.tn_model_dir, + "--output", + str(wav), + ) + spoken = run(args.binary, "transcribe", str(wav), "--backend", args.backend) + ratio = similarity(spoken, NORM_SPOKEN) + print(f"tn round trip: {spoken.strip()!r} (similarity {ratio:.2f})") + check(ratio >= 0.8, "TN spells out numbers before synthesis") + if args.itn_model_dir: + written = run( + args.binary, + "transcribe", + str(wav), + "--backend", + args.backend, + "--itn-model-dir", + args.itn_model_dir, + ) + print(f"itn round trip: {written.strip()!r}") + check("$25" in written, "ITN writes spoken numbers in written form") + if __name__ == "__main__": main() From 1f97c7532f63aeb56810925fdd93646bdaa4f234 Mon Sep 17 00:00:00 2001 From: Prabhsimran Singh Date: Fri, 2 Oct 2026 00:05:23 +0530 Subject: [PATCH 02/15] build: bump SentencePiece pin to v0.2.1 for CMake 4 compatibility --- THIRD_PARTY_NOTICES.md | 2 +- scripts/build_sentencepiece_static.sh | 2 +- 2 files changed, 2 insertions(+), 2 deletions(-) diff --git a/THIRD_PARTY_NOTICES.md b/THIRD_PARTY_NOTICES.md index d41194a..6d71c60 100644 --- a/THIRD_PARTY_NOTICES.md +++ b/THIRD_PARTY_NOTICES.md @@ -160,7 +160,7 @@ The command-line microphone capture layer compiles miniaudio directly into ### SentencePiece - Source: [`google/sentencepiece`](https://github.com/google/sentencepiece), - revision `17d7580d6407802f85855d2cc9190634e2c95624` + revision `31646a467d2051eb904e0b45de3a73e91fe1c1e3` - Copyright 2018 Google Inc. - License: Apache License 2.0 diff --git a/scripts/build_sentencepiece_static.sh b/scripts/build_sentencepiece_static.sh index 70fe338..00bd999 100755 --- a/scripts/build_sentencepiece_static.sh +++ b/scripts/build_sentencepiece_static.sh @@ -12,7 +12,7 @@ JOBS="${JOBS:-8}" JOBS="$(( JOBS < 4 ? JOBS : 4 ))" SOURCE="$WORK/source" BUILD="$WORK/build" -COMMIT=17d7580d6407802f85855d2cc9190634e2c95624 +COMMIT=31646a467d2051eb904e0b45de3a73e91fe1c1e3 if [ ! -d "$SOURCE/.git" ]; then git clone --filter=blob:none --no-checkout https://github.com/google/sentencepiece.git "$SOURCE" From dc7a0d2a0f4be6eec139af41de722356f9cab642 Mon Sep 17 00:00:00 2001 From: Prabhsimran Singh Date: Fri, 2 Oct 2026 00:55:08 +0530 Subject: [PATCH 03/15] ci: fix macOS and Windows release builds - build_itn_deps.sh: build RE2/protobuf out of tree; patch OpenFST for Clang 20+ - macOS: include for mkdtemp; link static SentencePiece hidden - package-macos.sh: detect Mach-O with lipo instead of otool - Windows: bundle vcomp140.dll when ggml uses OpenMP --- CMakeLists.txt | 7 ++++++- scripts/build_itn_deps.sh | 13 +++++++++---- scripts/release/package-macos.sh | 6 +++--- src/asr/CMakeLists.txt | 20 +++++++++++++++----- src/tts/preproc/text_normalizer.cpp | 2 ++ 5 files changed, 35 insertions(+), 13 deletions(-) diff --git a/CMakeLists.txt b/CMakeLists.txt index 685b661..83b095c 100644 --- a/CMakeLists.txt +++ b/CMakeLists.txt @@ -24,7 +24,12 @@ add_compile_definitions(NEMO_SPEECH_VERSION_STR="${NEMO_SPEECH_VERSION}") include(GNUInstallDirs) include(CMakePackageConfigHelpers) if(WIN32) - # Install the Visual C++ runtime required by Windows packages. + # Install the Visual C++ runtime required by Windows packages, plus the + # OpenMP runtime (vcomp140.dll) that ggml's CPU threading links against + # unless GGML_OPENMP is OFF. + if(NOT DEFINED GGML_OPENMP OR GGML_OPENMP) + set(CMAKE_INSTALL_OPENMP_LIBRARIES TRUE) + endif() include(InstallRequiredSystemLibraries) endif() set(NEMO_SPEECH_LICENSE_INSTALL_DIR diff --git a/scripts/build_itn_deps.sh b/scripts/build_itn_deps.sh index eb6353e..7fe288d 100755 --- a/scripts/build_itn_deps.sh +++ b/scripts/build_itn_deps.sh @@ -81,7 +81,9 @@ clone https://github.com/sarane22/sparrowhawk.git sparrowhawk 8b082acc507312077a if [ "$STATIC" = 1 ]; then clone https://github.com/protocolbuffers/protobuf.git protobuf f0dc78d7e6e331b8c6bb2d5283e06aa26883ca7c # v21.12 clone https://github.com/google/re2.git re2 3a8436ac436124a57a4e22d5c8713a2d42b381d7 # 2023-03-01 - cmake -S "$WORK/protobuf" -B "$WORK/protobuf/build" \ + # Build outside the source trees: RE2 ships a Bazel BUILD file, which + # "re2/build" resolves to on case-insensitive filesystems (macOS). + cmake -S "$WORK/protobuf" -B "$WORK/protobuf-build" \ -DCMAKE_BUILD_TYPE=Release \ -DCMAKE_INSTALL_PREFIX="$PREFIX" \ -DCMAKE_INSTALL_LIBDIR=lib \ @@ -89,15 +91,15 @@ if [ "$STATIC" = 1 ]; then -Dprotobuf_BUILD_SHARED_LIBS=OFF \ -Dprotobuf_BUILD_TESTS=OFF \ -Dprotobuf_WITH_ZLIB=OFF - cmake --build "$WORK/protobuf/build" --target install -j "$JOBS" - cmake -S "$WORK/re2" -B "$WORK/re2/build" \ + cmake --build "$WORK/protobuf-build" --target install -j "$JOBS" + cmake -S "$WORK/re2" -B "$WORK/re2-build" \ -DCMAKE_BUILD_TYPE=Release \ -DCMAKE_INSTALL_PREFIX="$PREFIX" \ -DCMAKE_INSTALL_LIBDIR=lib \ -DCMAKE_POSITION_INDEPENDENT_CODE=ON \ -DBUILD_SHARED_LIBS=OFF \ -DRE2_BUILD_TESTING=OFF - cmake --build "$WORK/re2/build" --target install -j "$JOBS" + cmake --build "$WORK/re2-build" --target install -j "$JOBS" # Sparrowhawk's configure and Makefiles run protoc from PATH. export PATH="$PREFIX/bin:$PATH" install_license "$WORK/protobuf/LICENSE" protobuf @@ -108,6 +110,9 @@ fi cd "$WORK/openfst" # FST_FLAGS_v rename missed by the fork. perl -pi -e 's/\bFLAGS_v\b/FST_FLAGS_v/g' src/include/fst/label-reachable.h +# VectorHashBiTable's copy constructor reads a nonexistent member (fixed +# upstream); Clang 20+ rejects it without instantiation. +perl -pi -e 's/selector_\(table\.s_\)/selector_(table.selector_)/' src/include/fst/bi-table.h # FAR + PDT cover Sparrowhawk's runtime grammar formats. Disable command-line # tools/script wrappers: the runtime calls the typed C++ OpenFST API directly. stamp_autotools diff --git a/scripts/release/package-macos.sh b/scripts/release/package-macos.sh index 20b02c6..4a4c08c 100755 --- a/scripts/release/package-macos.sh +++ b/scripts/release/package-macos.sh @@ -117,8 +117,8 @@ failures="$work_dir/failures" macho_files="$work_dir/macho-files" find "$package_root/bin" "$package_root/lib" -type f -print | LC_ALL=C sort > "$macho_files" while IFS= read -r file; do - otool -h "$file" >/dev/null 2>&1 || continue - archs="$(lipo -archs "$file" 2>/dev/null || true)" + # lipo, not otool: llvm-otool exits 0 for files that are not Mach-O. + archs="$(lipo -archs "$file" 2>/dev/null)" || continue [[ "$archs" == "$macho_arch" ]] || echo "$file: built for '$archs', expected $macho_arch" >> "$failures" minos="$(otool -l "$file" | awk '$1 == "minos" { print $2; exit }')" @@ -149,7 +149,7 @@ fi # Strip local symbols, then ad-hoc sign: install-time rpath edits invalidate the # linker's signature, and arm64 macOS refuses to run unsigned code. while IFS= read -r file; do - otool -h "$file" >/dev/null 2>&1 || continue + lipo -archs "$file" >/dev/null 2>&1 || continue strip -x "$file" codesign --force --sign - "$file" done < "$macho_files" diff --git a/src/asr/CMakeLists.txt b/src/asr/CMakeLists.txt index e140856..7c09ea0 100644 --- a/src/asr/CMakeLists.txt +++ b/src/asr/CMakeLists.txt @@ -63,8 +63,8 @@ target_link_libraries(nemo_speech_asr PUBLIC nemo_speech_runtime_ggml) # SentencePiece is a core ASR dependency: the RNNT head tokenizes word-boosting # phrases with the model's embedded tokenizer (context biasing in # rnnt_greedy_decoder.cpp / RnntModel::encode_phrase); the flashlight decoder -# also encodes OOV boost phrases with it. Prefer a static archive on ELF -# platforms so its bundled protobuf symbols stay private. +# also encodes OOV boost phrases with it. Prefer a static archive on ELF and +# Apple platforms so its bundled protobuf symbols stay private. if(UNIX AND NOT APPLE) set(_NEMO_SPEECH_LIBRARY_SUFFIXES "${CMAKE_FIND_LIBRARY_SUFFIXES}") set(CMAKE_FIND_LIBRARY_SUFFIXES ".a") @@ -72,6 +72,11 @@ if(UNIX AND NOT APPLE) HINTS "${NEMO_SPEECH_DEPENDENCY_PREFIX}/sentencepiece/lib") set(CMAKE_FIND_LIBRARY_SUFFIXES "${_NEMO_SPEECH_LIBRARY_SUFFIXES}") unset(_NEMO_SPEECH_LIBRARY_SUFFIXES) +elseif(APPLE) + # Only the project-built archive: Homebrew's libsentencepiece.a needs its + # shared abseil, which release archives cannot carry. + find_library(SENTENCEPIECE_STATIC_LIB libsentencepiece.a + PATHS "${NEMO_SPEECH_DEPENDENCY_PREFIX}/sentencepiece/lib" NO_DEFAULT_PATH) endif() if(SENTENCEPIECE_STATIC_LIB) find_path(SENTENCEPIECE_INCLUDE_DIR sentencepiece_processor.h @@ -79,10 +84,15 @@ if(SENTENCEPIECE_STATIC_LIB) if(NOT SENTENCEPIECE_INCLUDE_DIR) message(FATAL_ERROR "SentencePiece archive found without sentencepiece_processor.h") endif() - target_link_libraries(nemo_speech_asr PRIVATE ${SENTENCEPIECE_STATIC_LIB}) target_include_directories(nemo_speech_asr PRIVATE ${SENTENCEPIECE_INCLUDE_DIR}) - target_link_options( - nemo_speech_asr PRIVATE "LINKER:--exclude-libs,libsentencepiece.a") + if(APPLE) + target_link_options( + nemo_speech_asr PRIVATE "LINKER:-load_hidden,${SENTENCEPIECE_STATIC_LIB}") + else() + target_link_libraries(nemo_speech_asr PRIVATE ${SENTENCEPIECE_STATIC_LIB}) + target_link_options( + nemo_speech_asr PRIVATE "LINKER:--exclude-libs,libsentencepiece.a") + endif() elseif(NEMO_SPEECH_WITH_NORM AND UNIX AND NOT APPLE) message(FATAL_ERROR "normalization requires private static SentencePiece; " diff --git a/src/tts/preproc/text_normalizer.cpp b/src/tts/preproc/text_normalizer.cpp index d2a49a0..1a95367 100644 --- a/src/tts/preproc/text_normalizer.cpp +++ b/src/tts/preproc/text_normalizer.cpp @@ -9,6 +9,8 @@ #include #ifdef NEMO_SPEECH_WITH_NORM +#include // mkdtemp (macOS declares it only here) + #include #include #include From 2cfd184dc4fa07f02524483d3a5129c105ea32ab Mon Sep 17 00:00:00 2001 From: Prabhsimran Singh Date: Fri, 2 Oct 2026 02:20:27 +0530 Subject: [PATCH 04/15] ci: address CodeRabbit review on release workflow | fix: gpu smoke jobs --- .github/workflows/release.yml | 43 ++++++++++++++++++++++++++------ docker/Dockerfile.release-linux | 4 ++- docs/development/releasing.md | 27 +++++++++++++------- scripts/release/package-linux.sh | 6 +++++ 4 files changed, 62 insertions(+), 18 deletions(-) diff --git a/.github/workflows/release.yml b/.github/workflows/release.yml index 5a4a5bf..3f3aa40 100644 --- a/.github/workflows/release.yml +++ b/.github/workflows/release.yml @@ -14,8 +14,9 @@ name: Release # archive runs under an emulated Haswell CPU before anything is published. # # Linux and macOS archives include text normalization (ITN/TN). The grammars are -# not built here: they are taken from the latest release, used by the smoke -# tests, and published again with every release. +# not built here: they are taken from the release pinned in the grammars job, +# checked against recorded SHA-256 digests, used by the smoke tests, and +# published again with every release. on: push: @@ -64,7 +65,7 @@ jobs: EVENT: ${{ github.event_name }} INPUT_CHANNEL: ${{ inputs.channel }} RELEASE_PATHS: >- - ^(\.github/workflows/release\.yml|docker/Dockerfile\.release-linux|scripts/build_itn_deps\.sh|scripts/build_sentencepiece_static\.sh|scripts/release/|scripts/windows/(build|package-release)\.ps1|tests/ci/model_smoke\.py) + ^(\.github/workflows/release\.yml|docker/Dockerfile\.release-linux|scripts/build_itn_deps\.sh|scripts/build_sentencepiece_static\.sh|scripts/release/|scripts/windows/(build|package-release)\.ps1|CMakeLists\.txt|src/(asr|common)/CMakeLists\.txt|tests/ci/model_smoke\.py) run: | set -euo pipefail declared="$(sed -n 's/^NEMO_SPEECH_VERSION:[[:space:]]*//p' VERSION)" @@ -93,6 +94,10 @@ jobs: schedule:*) channel=nightly; version=nightly ;; *) channel="$INPUT_CHANNEL" + if [ "$channel" = nightly ] && [ "$GITHUB_REF" != refs/heads/main ]; then + echo "::error::nightly can only be published from main" + exit 1 + fi if [ "$channel" = nightly ]; then version=nightly; else version="$declared"; fi ;; esac @@ -117,14 +122,22 @@ jobs: if: needs.prepare.outputs.build == 'true' runs-on: ubuntu-24.04 steps: - - name: Download from the latest release + - name: Download the pinned grammars + # To ship new grammars, attach them to a release and update the tag and + # digests here. + env: + GRAMMARS_RELEASE: v0.1.0 run: | set -euo pipefail mkdir -p grammars for name in itn_configs tn_configs; do curl -fsSL --retry 3 -o "grammars/$name.tar.bz2" \ - "https://github.com/$GITHUB_REPOSITORY/releases/latest/download/$name.tar.bz2" + "https://github.com/$GITHUB_REPOSITORY/releases/download/$GRAMMARS_RELEASE/$name.tar.bz2" done + (cd grammars && sha256sum --check --strict) <<'EOF' + 880c9365d1d52c17450bd930950b0e58eca294421b7be62eb71666fb21b8997f itn_configs.tar.bz2 + 2ca242c6d29f551eba3663d7e508c0d9dad10440e287628234154c6d1a72c7bc tn_configs.tar.bz2 + EOF - name: Upload grammars uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1 @@ -555,7 +568,8 @@ jobs: NVIDIA_VISIBLE_DEVICES: ${{ env.NVIDIA_VISIBLE_DEVICES }} steps: - name: Install prerequisites - run: apt-get update && apt-get install -y --no-install-recommends git python3 ca-certificates bzip2 + # The CLI downloads models with curl. + run: apt-get update && apt-get install -y --no-install-recommends git curl python3 ca-certificates bzip2 - name: Checkout repository uses: actions/checkout@df4cb1c069e1874edd31b4311f1884172cec0e10 # v6.0.3 @@ -568,6 +582,12 @@ jobs: name: release-linux-x86_64-cuda path: dist + - name: Cache models + uses: actions/cache@55cc8345863c7cc4c66a329aec7e433d2d1c52a9 # v6.1.0 + with: + path: ${{ env.NEMO_SPEECH_MODEL_DIR }} + key: models-${{ hashFiles('models/index.json') }} + - name: Download grammars uses: actions/download-artifact@3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c # v8.0.1 with: @@ -590,8 +610,9 @@ jobs: needs: [prepare, windows] runs-on: windows-amd64-gpu-l4-latest-1 timeout-minutes: 60 - # Non-blocking, like the other jobs on this runner (see gpu.yml). - continue-on-error: true + # Non-blocking, like the other jobs on this runner (see gpu.yml), except + # for tagged releases. + continue-on-error: ${{ needs.prepare.outputs.channel != 'release' }} steps: - name: Checkout repository uses: actions/checkout@df4cb1c069e1874edd31b4311f1884172cec0e10 # v6.0.3 @@ -604,6 +625,12 @@ jobs: name: release-windows-x86_64-cuda path: dist + - name: Cache models + uses: actions/cache@55cc8345863c7cc4c66a329aec7e433d2d1c52a9 # v6.1.0 + with: + path: ${{ env.NEMO_SPEECH_MODEL_DIR }} + key: models-${{ hashFiles('models/index.json') }} + - name: Model smoke test shell: pwsh run: | diff --git a/docker/Dockerfile.release-linux b/docker/Dockerfile.release-linux index 656a7a9..d89ddb9 100644 --- a/docker/Dockerfile.release-linux +++ b/docker/Dockerfile.release-linux @@ -206,12 +206,14 @@ COPY . /work COPY --from=sentencepiece-dependency /work/.deps/sentencepiece /work/.deps/sentencepiece COPY --from=itn-dependency /work/.deps/itn /work/.deps/itn +# The default architectures match the default CUDA 12.8 image; CUDA 13 builds +# (sm_110, sm_121) pass CUDA_ARCH. RUN set -eux; \ cuda_arch="${CUDA_ARCH}"; \ if [ -z "$cuda_arch" ]; then \ case "${TARGETARCH}" in \ amd64) cuda_arch='75-real;80-virtual;86-real;89-real;120a-real' ;; \ - arm64) cuda_arch='87-real;110a-real;121a-real;87-virtual' ;; \ + arm64) cuda_arch='87-real;87-virtual' ;; \ *) echo "unsupported TARGETARCH: ${TARGETARCH}" >&2; exit 2 ;; \ esac; \ fi; \ diff --git a/docs/development/releasing.md b/docs/development/releasing.md index 6950245..3cad802 100644 --- a/docs/development/releasing.md +++ b/docs/development/releasing.md @@ -19,7 +19,9 @@ Linux and macOS archives include ITN and TN: `scripts/build_itn_deps.sh` with `STATIC=1` builds OpenFST, Sparrowhawk, Protobuf, and RE2 as static archives, and `libnemo_speech_text_normalization` links them privately. The grammars, `itn_configs.tar.bz2` and `tn_configs.tar.bz2`, are not built by the workflow; -each run takes them from the latest release and publishes them again. +each run downloads them from the release pinned in the `grammars` job, checks +their SHA-256 digests, and publishes them again. To ship new grammars, attach +them to a release and update the tag and digests there. ## Cutting a release @@ -37,7 +39,8 @@ moved since the current nightly; running the workflow manually with `nightly` always rebuilds. Pull requests that change release packaging (the workflow, the release -Dockerfile and packagers, the dependency build scripts, or the model smoke test) +Dockerfile and packagers, the dependency build scripts, the CMake files that +link the bundled dependencies, or the model smoke test) run the workflow as a dry run once copy-pr-bot mirrors them to a `pull-request/` branch. Run the workflow manually with `dry-run` to build and check everything without publishing. @@ -51,12 +54,16 @@ run the workflow as a dry run once copy-pr-bot mirrors them to a which has no AVX-512. - **Self-contained packages:** Linux archives need glibc 2.31 or newer and find their bundled libraries through `DT_RPATH`, which `LD_LIBRARY_PATH` cannot - override. macOS archives need macOS 13.3 or newer, link only system + override. Vulkan archives use the host's `libstdc++` and `libgcc_s`, which + the host's Vulkan drivers also need. macOS archives need macOS 13.3 or newer, link only system libraries, and are ad-hoc signed. Windows archives bundle every DLL they import except Windows and GPU driver libraries. -- **Smoke tests:** CPU archives run `tests/ci/model_smoke.py` on their runner, - and the CUDA archives run it on L4 GPUs. On Linux and macOS the test also - synthesizes digits through TN and transcribes them back through ITN. +- **Smoke tests:** CPU archives run `tests/ci/model_smoke.py` on their runner. + The x86_64 Linux and Windows CUDA archives run it on L4 GPUs; the Windows run + blocks only tagged releases. Linux Vulkan archives only start (`--version`); + the aarch64 CUDA and Windows Vulkan archives are not run. On Linux and macOS + the test also synthesizes digits through TN and transcribes them back through + ITN. ## Runners @@ -77,12 +84,14 @@ docker build --platform=linux/amd64 -f docker/Dockerfile.release-linux \ --build-arg BACKEND=cpu --target artifact \ --output type=local,dest=release-artifacts . -# macOS, after scripts/build_itn_deps.sh and installing a preset configured -# with -DNEMO_SPEECH_WITH_NORM=ON +# macOS, after scripts/build_sentencepiece_static.sh, scripts/build_itn_deps.sh, +# and installing a preset configured with -DNEMO_SPEECH_WITH_NORM=ON +# -DGGML_NATIVE=OFF -DCMAKE_OSX_DEPLOYMENT_TARGET=13.3, all with +# MACOSX_DEPLOYMENT_TARGET=13.3 in the environment scripts/release/package-macos.sh --install-prefix --backend metal --arch aarch64 ``` ```powershell -# Windows, after scripts\windows\build.ps1 -Profile server -CMakeArgs '-DGGML_NATIVE=OFF' +# Windows, after scripts\windows\build.ps1 -Backend cpu -Profile server -CMakeArgs '-DGGML_NATIVE=OFF' scripts\windows\package-release.ps1 -BuildDir -Backend cpu ``` diff --git a/scripts/release/package-linux.sh b/scripts/release/package-linux.sh index deb3aca..45912cf 100755 --- a/scripts/release/package-linux.sh +++ b/scripts/release/package-linux.sh @@ -172,6 +172,12 @@ done runtime_license_dir="$package_root/share/licenses/nemo-speech/third_party/gcc-runtime" mkdir -p "$package_root/lib" "$runtime_license_dir" for runtime in libstdc++.so.6 libgcc_s.so.1 libgomp.so.1 libatomic.so.1; do + # Vulkan drivers (Mesa ICDs) need the host's newer C++ runtime; a bundled + # copy loaded first through DT_RPATH would make them fail to load. + if [[ "$backend" == vulkan ]] && + [[ "$runtime" == libstdc++.so.6 || "$runtime" == libgcc_s.so.1 ]]; then + continue + fi if [[ "$runtime" == libstdc++* ]]; then compiler=c++ else From a05e2f9806a2c61b2bdee677d27e5980228541aa Mon Sep 17 00:00:00 2001 From: Prabhsimran Singh Date: Fri, 2 Oct 2026 03:09:43 +0530 Subject: [PATCH 05/15] ci: nightly version stamp, release cache hardening, greptile config --- .github/workflows/pre-commit.yml | 38 ------------ .github/workflows/release.yml | 41 ++++++++----- .greptile/config.json | 58 ++++++++++++++++++ CMakeLists.txt | 15 +++++ README.md | 8 ++- docker/Dockerfile.release-linux | 6 ++ docs/development/releasing.md | 5 +- docs/install.md | 29 ++++----- greptile.json | 100 +++++++++++++++++++++++++++++++ 9 files changed, 230 insertions(+), 70 deletions(-) delete mode 100644 .github/workflows/pre-commit.yml create mode 100644 .greptile/config.json create mode 100644 greptile.json diff --git a/.github/workflows/pre-commit.yml b/.github/workflows/pre-commit.yml deleted file mode 100644 index e381975..0000000 --- a/.github/workflows/pre-commit.yml +++ /dev/null @@ -1,38 +0,0 @@ -name: Pre-commit Formatting - -on: - pull_request: - types: [opened, synchronize, reopened, ready_for_review] - push: - branches: [main] - workflow_dispatch: - -concurrency: - group: pre-commit-${{ github.workflow }}-${{ github.ref }} - cancel-in-progress: true - -permissions: - contents: read - -jobs: - pre-commit: - name: Formatting and lint checks - if: github.event_name != 'pull_request' || !github.event.pull_request.draft - runs-on: ubuntu-latest - timeout-minutes: 15 - steps: - - name: Checkout repository - uses: actions/checkout@df4cb1c069e1874edd31b4311f1884172cec0e10 # v6.0.3 - with: - persist-credentials: false - - - name: Set up Python - uses: actions/setup-python@ece7cb06caefa5fff74198d8649806c4678c61a1 # v6 - with: - python-version: "3.12" - - - name: Install pre-commit - run: python -m pip install --disable-pip-version-check pre-commit==4.6.2 - - - name: Run pre-commit - run: pre-commit run --all-files --show-diff-on-failure diff --git a/.github/workflows/release.yml b/.github/workflows/release.yml index 3f3aa40..dfa1ea7 100644 --- a/.github/workflows/release.yml +++ b/.github/workflows/release.yml @@ -40,6 +40,8 @@ permissions: contents: read env: + # Release jobs only restore the model cache, so they never write entries that + # a later release could read; build.yml and gpu.yml save it from main. NEMO_SPEECH_MODEL_DIR: ${{ github.workspace }}/.ci-models JFK_AUDIO: test_files/asr/wav/test/jfk.wav @@ -52,6 +54,7 @@ jobs: version: ${{ steps.version.outputs.version }} channel: ${{ steps.version.outputs.channel }} build: ${{ steps.version.outputs.build }} + version_metadata: ${{ steps.version.outputs.version_metadata }} steps: - name: Checkout repository uses: actions/checkout@df4cb1c069e1874edd31b4311f1884172cec0e10 # v6.0.3 @@ -111,9 +114,14 @@ jobs: echo "::notice::nightly is already built from $GITHUB_SHA; skipping" fi fi + # Nightly binaries report VERSION plus build metadata, e.g. + # 0.2.0+nightly.a1b2c3d, so they are not mistaken for the release. + metadata="" + if [ "$channel" = nightly ]; then metadata="nightly.${GITHUB_SHA::7}"; fi echo "channel=$channel" >> "$GITHUB_OUTPUT" echo "version=$version" >> "$GITHUB_OUTPUT" echo "build=$build" >> "$GITHUB_OUTPUT" + echo "version_metadata=$metadata" >> "$GITHUB_OUTPUT" echo "Building $version ($channel): $build" grammars: @@ -211,10 +219,12 @@ jobs: CUDA_IMAGE: ${{ matrix.cuda_image }} CUDA_ARCH: ${{ matrix.cuda_arch }} RELEASE_VERSION: ${{ needs.prepare.outputs.version }} + VERSION_METADATA: ${{ needs.prepare.outputs.version_metadata }} run: | set -euo pipefail args=(--build-arg "BACKEND=$BACKEND" --build-arg "ARTIFACT_BACKEND=$ARTIFACT_BACKEND" - --build-arg "RELEASE_VERSION=$RELEASE_VERSION" --build-arg "JOBS=$(nproc)") + --build-arg "RELEASE_VERSION=$RELEASE_VERSION" --build-arg "JOBS=$(nproc)" + --build-arg "VERSION_METADATA=$VERSION_METADATA") if [ "$BACKEND" = cuda ]; then args+=(--build-arg "CUDA_IMAGE=$CUDA_IMAGE" --build-arg "CUDA_ARCH=$CUDA_ARCH") fi @@ -224,9 +234,9 @@ jobs: test "$(find release-artifacts -maxdepth 1 -name '*.tar.gz' | wc -l)" -eq 1 (cd release-artifacts && sha256sum --check ./*.sha256) - - name: Cache models + - name: Restore cached models if: matrix.backend == 'cpu' - uses: actions/cache@55cc8345863c7cc4c66a329aec7e433d2d1c52a9 # v6.1.0 + uses: actions/cache/restore@55cc8345863c7cc4c66a329aec7e433d2d1c52a9 # v6.1.0 with: path: ${{ env.NEMO_SPEECH_MODEL_DIR }} key: models-${{ hashFiles('models/index.json') }} @@ -305,6 +315,7 @@ jobs: env: PRESET: ${{ matrix.preset }} CMAKE_ARGS: ${{ matrix.cmake_args }} + VERSION_METADATA: ${{ needs.prepare.outputs.version_metadata }} run: | set -euo pipefail # configure.sh needs bash 4+; the runner's default bash is 3.2. @@ -314,6 +325,7 @@ jobs: -DNEMO_SPEECH_BUILD_GRPC=OFF -DNEMO_SPEECH_WITH_GRPC=OFF \ -DNEMO_SPEECH_WITH_NORM=ON \ -DCMAKE_OSX_DEPLOYMENT_TARGET="$MACOSX_DEPLOYMENT_TARGET" \ + -DNEMO_SPEECH_VERSION_METADATA="$VERSION_METADATA" \ $CMAKE_ARGS cmake --build --preset "$PRESET" --parallel cmake --install "build/$PRESET" --prefix "$RUNNER_TEMP/install" @@ -328,8 +340,8 @@ jobs: --backend "$BACKEND" --arch "$ARCH" --version "$RELEASE_VERSION" \ --min-macos "$MACOSX_DEPLOYMENT_TARGET" --output-dir release-artifacts - - name: Cache models - uses: actions/cache@55cc8345863c7cc4c66a329aec7e433d2d1c52a9 # v6.1.0 + - name: Restore cached models + uses: actions/cache/restore@55cc8345863c7cc4c66a329aec7e433d2d1c52a9 # v6.1.0 with: path: ${{ env.NEMO_SPEECH_MODEL_DIR }} key: models-${{ hashFiles('models/index.json') }} @@ -417,12 +429,13 @@ jobs: shell: pwsh env: BACKEND: ${{ matrix.backend }} + VERSION_METADATA: ${{ needs.prepare.outputs.version_metadata }} run: | $arguments = @{ Backend = $env:BACKEND Profile = 'server' BuildDir = "${{ github.workspace }}\build\release-$env:BACKEND" - CMakeArgs = @('-DGGML_NATIVE=OFF') + CMakeArgs = @('-DGGML_NATIVE=OFF', "-DNEMO_SPEECH_VERSION_METADATA=$env:VERSION_METADATA") } if ($env:BACKEND -eq 'cuda') { $arguments.CudaArch = '75-real;80-virtual;86-real;89-real;120a-real' @@ -446,9 +459,9 @@ jobs: scripts\windows\package-release.ps1 -BuildDir "build\release-$env:BACKEND" ` -Backend $env:BACKEND -Version $env:RELEASE_VERSION -OutputDir release-artifacts - - name: Cache models + - name: Restore cached models if: matrix.backend == 'cpu' - uses: actions/cache@55cc8345863c7cc4c66a329aec7e433d2d1c52a9 # v6.1.0 + uses: actions/cache/restore@55cc8345863c7cc4c66a329aec7e433d2d1c52a9 # v6.1.0 with: path: ${{ env.NEMO_SPEECH_MODEL_DIR }} key: models-${{ hashFiles('models/index.json') }} @@ -535,8 +548,8 @@ jobs: name: release-linux-x86_64-cpu path: dist - - name: Cache models - uses: actions/cache@55cc8345863c7cc4c66a329aec7e433d2d1c52a9 # v6.1.0 + - name: Restore cached models + uses: actions/cache/restore@55cc8345863c7cc4c66a329aec7e433d2d1c52a9 # v6.1.0 with: path: ${{ env.NEMO_SPEECH_MODEL_DIR }} key: models-${{ hashFiles('models/index.json') }} @@ -582,8 +595,8 @@ jobs: name: release-linux-x86_64-cuda path: dist - - name: Cache models - uses: actions/cache@55cc8345863c7cc4c66a329aec7e433d2d1c52a9 # v6.1.0 + - name: Restore cached models + uses: actions/cache/restore@55cc8345863c7cc4c66a329aec7e433d2d1c52a9 # v6.1.0 with: path: ${{ env.NEMO_SPEECH_MODEL_DIR }} key: models-${{ hashFiles('models/index.json') }} @@ -625,8 +638,8 @@ jobs: name: release-windows-x86_64-cuda path: dist - - name: Cache models - uses: actions/cache@55cc8345863c7cc4c66a329aec7e433d2d1c52a9 # v6.1.0 + - name: Restore cached models + uses: actions/cache/restore@55cc8345863c7cc4c66a329aec7e433d2d1c52a9 # v6.1.0 with: path: ${{ env.NEMO_SPEECH_MODEL_DIR }} key: models-${{ hashFiles('models/index.json') }} diff --git a/.greptile/config.json b/.greptile/config.json new file mode 100644 index 0000000..419f14f --- /dev/null +++ b/.greptile/config.json @@ -0,0 +1,58 @@ +{ + "strictness": 2, + "commentTypes": ["logic", "syntax", "style"], + "triggerOnUpdates": true, + "includeBranches": ["main"], + "excludeAuthors": ["dependabot[bot]", "renovate[bot]", "pre-commit-ci[bot]"], + "ignorePatterns": "third_party/**\ntest_files/**\n", + "fixWithAI": false, + "shouldUpdateDescription": false, + "statusCommentsEnabled": false, + "summarySection": { + "included": true, + "collapsible": true, + "defaultOpen": false + }, + "issuesTableSection": { + "included": true, + "collapsible": true, + "defaultOpen": false + }, + "confidenceScoreSection": { + "included": false, + "collapsible": false, + "defaultOpen": false + }, + "sequenceDiagramSection": { + "included": false, + "collapsible": false, + "defaultOpen": false + }, + "instructions": "NeMo-Speech.cpp is a C++17 speech runtime (ASR, diarization, TTS, translation, VoiceChat) built on ggml from a pinned llama.cpp submodule, with CPU, CUDA, Metal and Vulkan backends and portable release archives for Linux, macOS and Windows. patches/ holds the project's own changes to llama.cpp and ggml, including custom GPU kernels; review them like any other code.\n\nPrioritize findings in this order: correctness, performance, portability, maintainability. Report only problems the change introduces or makes reachable, each with the concrete input, backend or platform where it fails and what to do about it. Do not comment on formatting, naming or anything pre-commit already checks.", + "rules": [ + { + "id": "correctness", + "rule": "Look for wrong results, crashes, memory errors, races, uninitialized or stale state, unhandled edge cases (empty input, boundary sizes, partial tiles or chunks), numerical instability, and fallback paths that behave differently from the fast path. Streaming results must not depend on chunk size.", + "scope": ["src/**", "app/**", "server/**", "include/**", "patches/**", "kernels/**", "conversion/**", "convert_model.py", "scripts/**", ".github/workflows/**"], + "severity": "high" + }, + { + "id": "performance", + "rule": "Look for regressions on hot paths: added host-device synchronization, per-chunk allocations, redundant work, extra copies between backends, ops that fall back to the CPU, and caches or reuse that never take effect.", + "scope": ["src/**", "server/**", "patches/**", "kernels/**"], + "severity": "medium" + }, + { + "id": "portability", + "rule": "Look for assumptions tied to one GPU architecture, backend, operating system or compiler. Hardware limits and features must be checked with a working fallback, and code must build with GCC, Clang, Apple Clang and MSVC. Release archives must stay self-contained and run on every supported target.", + "scope": ["src/**", "app/**", "server/**", "include/**", "patches/**", "kernels/**", "scripts/**", "docker/**", "CMakeLists.txt", "**/CMakeLists.txt"], + "severity": "medium" + }, + { + "id": "maintainability", + "rule": "Look for duplicated logic that should reuse an existing helper, platform-specific branches scattered across call sites, dead or unreachable code, unexplained magic constants, and documentation that no longer matches the code.", + "scope": ["src/**", "app/**", "server/**", "include/**", "patches/**", "kernels/**", "conversion/**", "scripts/**", "docs/**", "README.md"], + "severity": "low" + } + ] +} diff --git a/CMakeLists.txt b/CMakeLists.txt index 83b095c..3a1397c 100644 --- a/CMakeLists.txt +++ b/CMakeLists.txt @@ -17,6 +17,21 @@ endforeach() if(NOT NEMO_SPEECH_VERSION OR NOT NEMO_SPEECH_PROJECT_VERSION) message(FATAL_ERROR "could not parse NEMO_SPEECH_VERSION from ${CMAKE_CURRENT_SOURCE_DIR}/VERSION") endif() +# Semver build metadata for the reported version, e.g. nightly. for +# nightly release builds, which then report 0.2.0+nightly.. +set(NEMO_SPEECH_VERSION_METADATA "" CACHE STRING + "Build metadata appended to the reported version (dot-separated [0-9A-Za-z-] identifiers)") +if(NEMO_SPEECH_VERSION_METADATA) + if(NOT NEMO_SPEECH_VERSION_METADATA MATCHES "^[0-9A-Za-z-]+(\\.[0-9A-Za-z-]+)*$") + message(FATAL_ERROR + "NEMO_SPEECH_VERSION_METADATA must be dot-separated [0-9A-Za-z-] identifiers") + endif() + if(NEMO_SPEECH_VERSION MATCHES "\\+") + string(APPEND NEMO_SPEECH_VERSION ".${NEMO_SPEECH_VERSION_METADATA}") + else() + string(APPEND NEMO_SPEECH_VERSION "+${NEMO_SPEECH_VERSION_METADATA}") + endif() +endif() project(nemo_speech VERSION ${NEMO_SPEECH_PROJECT_VERSION} LANGUAGES C CXX) add_compile_definitions(NEMO_SPEECH_VERSION_STR="${NEMO_SPEECH_VERSION}") diff --git a/README.md b/README.md index a119077..d343214 100644 --- a/README.md +++ b/README.md @@ -50,9 +50,11 @@ See [BENCHMARK.md](BENCHMARK.md) for the methodology and more results. ## Installation > [!IMPORTANT] -> **For the best performance and the latest features, build natively from source.** A native -> build is compiled for your machine, and release tags can be out of sync with the -> `main` branch. See [Build from source](#build-from-source). +> **For the best performance, build natively from source.** A native build is compiled for +> your machine. Tagged releases are cut periodically and can trail the `main` branch; for +> prebuilt binaries of the latest `main`, pass `--channel nightly` (`-Channel nightly` on +> Windows) to install the nightly prerelease, which is rebuilt daily. See +> [Build from source](#build-from-source). Install the `nemo-speech` CLI for the detected platform and backend: diff --git a/docker/Dockerfile.release-linux b/docker/Dockerfile.release-linux index d89ddb9..58e16d0 100644 --- a/docker/Dockerfile.release-linux +++ b/docker/Dockerfile.release-linux @@ -153,6 +153,8 @@ COPY . /work COPY --from=sentencepiece-dependency /work/.deps/sentencepiece /work/.deps/sentencepiece COPY --from=itn-dependency /work/.deps/itn /work/.deps/itn +# Build metadata for the reported version (nightly. for nightly builds). +ARG VERSION_METADATA= RUN case "${BACKEND}" in cpu|vulkan) ;; \ *) echo "BACKEND must be cpu or vulkan" >&2; exit 2 ;; \ esac \ @@ -161,6 +163,7 @@ RUN case "${BACKEND}" in cpu|vulkan) ;; \ -DNEMO_SPEECH_BUILD_GRPC=OFF \ -DNEMO_SPEECH_WITH_GRPC=OFF \ -DNEMO_SPEECH_WITH_NORM=ON \ + "-DNEMO_SPEECH_VERSION_METADATA=${VERSION_METADATA}" \ -DCMAKE_INSTALL_PREFIX=/opt/nemo-speech \ "-DCMAKE_CXX_STANDARD_LIBRARIES:STRING=-lstdc++fs -lanl" \ "-DCMAKE_EXE_LINKER_FLAGS=-Wl,--disable-new-dtags" \ @@ -206,6 +209,8 @@ COPY . /work COPY --from=sentencepiece-dependency /work/.deps/sentencepiece /work/.deps/sentencepiece COPY --from=itn-dependency /work/.deps/itn /work/.deps/itn +# Build metadata for the reported version (nightly. for nightly builds). +ARG VERSION_METADATA= # The default architectures match the default CUDA 12.8 image; CUDA 13 builds # (sm_110, sm_121) pass CUDA_ARCH. RUN set -eux; \ @@ -224,6 +229,7 @@ RUN set -eux; \ -DNEMO_SPEECH_BUILD_GRPC=OFF \ -DNEMO_SPEECH_WITH_GRPC=OFF \ -DNEMO_SPEECH_WITH_NORM=ON \ + "-DNEMO_SPEECH_VERSION_METADATA=${VERSION_METADATA}" \ -DCMAKE_INSTALL_PREFIX=/opt/nemo-speech \ "-DCMAKE_CXX_STANDARD_LIBRARIES:STRING=-lstdc++fs -lanl" \ "-DCMAKE_EXE_LINKER_FLAGS=-Wl,--disable-new-dtags" \ diff --git a/docs/development/releasing.md b/docs/development/releasing.md index 3cad802..4b04961 100644 --- a/docs/development/releasing.md +++ b/docs/development/releasing.md @@ -36,7 +36,10 @@ them to a release and update the tag and digests there. A daily scheduled run publishes the `nightly` prerelease, which `install.sh --channel nightly` installs. It is skipped when `main` has not moved since the current nightly; running the workflow manually with `nightly` -always rebuilds. +always rebuilds. Nightly binaries report the `VERSION` value with build metadata +(for example `0.2.0+nightly.a1b2c3d`, set through the +`NEMO_SPEECH_VERSION_METADATA` CMake option), so they are not mistaken for the +release. Pull requests that change release packaging (the workflow, the release Dockerfile and packagers, the dependency build scripts, the CMake files that diff --git a/docs/install.md b/docs/install.md index 24a94c3..7d4358d 100644 --- a/docs/install.md +++ b/docs/install.md @@ -9,10 +9,11 @@ downloads a model only when explicitly enabled with an indexed name. See [models and cache](cli.md#models-and-cache). > [!IMPORTANT] -> **For the best performance and the latest features, build natively from source** with -> `--source` (`-Source` on Windows) or by following the [source-build guide](build.md). A native -> build is compiled for your machine's CPU and GPU, and release tags can be out of sync with the -> `main` branch. +> **For the best performance, build natively from source** with `--source` (`-Source` on +> Windows) or by following the [source-build guide](build.md). A native build is compiled for +> your machine's CPU and GPU. Tagged releases are cut periodically and can trail the `main` +> branch; for prebuilt binaries of the latest `main`, pass `--channel nightly` +> (`-Channel nightly` on Windows) to install the nightly prerelease, which is rebuilt daily. ## Linux and macOS @@ -43,11 +44,14 @@ The grammars are published with each release as `itn_configs.tar.bz2` and `tn_configs.tar.bz2`. The installer selects CUDA when `nvidia-smi` is available, Metal on Apple -Silicon, and CPU otherwise. Override the backend or force a source build: +Silicon, and CPU otherwise. Override the backend, install the nightly +prerelease, or force a source build: ```bash curl -fsSL https://github.com/NVIDIA/NeMo-Speech.cpp/raw/main/scripts/install.sh | sh -s -- --backend cpu +curl -fsSL https://github.com/NVIDIA/NeMo-Speech.cpp/raw/main/scripts/install.sh | + sh -s -- --channel nightly curl -fsSL https://github.com/NVIDIA/NeMo-Speech.cpp/raw/main/scripts/install.sh | sh -s -- --source ``` @@ -96,25 +100,22 @@ irm https://github.com/NVIDIA/NeMo-Speech.cpp/raw/main/scripts/install.ps1 | iex The installer updates the current user's `PATH`. Open a new PowerShell window, then run `nemo-speech --version`. -Select a backend explicitly when needed: +Pass options after the script block. Install the nightly prerelease, or select +a backend explicitly: ```powershell -irm https://github.com/NVIDIA/NeMo-Speech.cpp/raw/main/scripts/install.ps1 ` - -OutFile .\install-nemo-speech.ps1 -powershell -ExecutionPolicy Bypass -File .\install-nemo-speech.ps1 ` - -Source -Backend cuda +iex "& {$(irm https://github.com/NVIDIA/NeMo-Speech.cpp/raw/main/scripts/install.ps1)} -Channel nightly" +iex "& {$(irm https://github.com/NVIDIA/NeMo-Speech.cpp/raw/main/scripts/install.ps1)} -Source -Backend cuda" ``` Select the components to install: ```powershell # ASR and diarization only -powershell -ExecutionPolicy Bypass -File .\install-nemo-speech.ps1 ` - -Source -Backend cpu -Profile asr +iex "& {$(irm https://github.com/NVIDIA/NeMo-Speech.cpp/raw/main/scripts/install.ps1)} -Source -Backend cpu -Profile asr" # Full runtime profile (add -HttpTls for TLS) -powershell -ExecutionPolicy Bypass -File .\install-nemo-speech.ps1 ` - -Source -Backend cuda -Profile full +iex "& {$(irm https://github.com/NVIDIA/NeMo-Speech.cpp/raw/main/scripts/install.ps1)} -Source -Backend cuda -Profile full" ``` | Profile | Components | diff --git a/greptile.json b/greptile.json new file mode 100644 index 0000000..4b676e0 --- /dev/null +++ b/greptile.json @@ -0,0 +1,100 @@ +{ + "strictness": 2, + "commentTypes": [ + "logic", + "syntax", + "style" + ], + "triggerOnUpdates": true, + "includeBranches": [ + "main" + ], + "excludeAuthors": [ + "dependabot[bot]", + "renovate[bot]", + "pre-commit-ci[bot]" + ], + "ignorePatterns": "greptile.json\nthird_party/**\ntest_files/**\n", + "fixWithAI": false, + "shouldUpdateDescription": false, + "statusCommentsEnabled": false, + "summarySection": { + "included": true, + "collapsible": true, + "defaultOpen": false + }, + "issuesTableSection": { + "included": true, + "collapsible": true, + "defaultOpen": false + }, + "confidenceScoreSection": { + "included": false, + "collapsible": false, + "defaultOpen": false + }, + "sequenceDiagramSection": { + "included": false, + "collapsible": false, + "defaultOpen": false + }, + "instructions": "NeMo-Speech.cpp is a C++17 speech runtime (ASR, diarization, TTS, translation, VoiceChat) built on ggml from a pinned llama.cpp submodule, with CPU, CUDA, Metal and Vulkan backends and portable release archives for Linux, macOS and Windows. patches/ holds the project's own changes to llama.cpp and ggml, including custom GPU kernels; review them like any other code.\n\nPrioritize findings in this order: correctness, performance, portability, maintainability. Report only problems the change introduces or makes reachable, each with the concrete input, backend or platform where it fails and what to do about it. Do not comment on formatting, naming or anything pre-commit already checks.", + "customContext": { + "rules": [ + { + "scope": [ + "src/**", + "app/**", + "server/**", + "include/**", + "patches/**", + "kernels/**", + "conversion/**", + "convert_model.py", + "scripts/**", + ".github/workflows/**" + ], + "rule": "Look for wrong results, crashes, memory errors, races, uninitialized or stale state, unhandled edge cases (empty input, boundary sizes, partial tiles or chunks), numerical instability, and fallback paths that behave differently from the fast path. Streaming results must not depend on chunk size." + }, + { + "scope": [ + "src/**", + "server/**", + "patches/**", + "kernels/**" + ], + "rule": "Look for regressions on hot paths: added host-device synchronization, per-chunk allocations, redundant work, extra copies between backends, ops that fall back to the CPU, and caches or reuse that never take effect." + }, + { + "scope": [ + "src/**", + "app/**", + "server/**", + "include/**", + "patches/**", + "kernels/**", + "scripts/**", + "docker/**", + "CMakeLists.txt", + "**/CMakeLists.txt" + ], + "rule": "Look for assumptions tied to one GPU architecture, backend, operating system or compiler. Hardware limits and features must be checked with a working fallback, and code must build with GCC, Clang, Apple Clang and MSVC. Release archives must stay self-contained and run on every supported target." + }, + { + "scope": [ + "src/**", + "app/**", + "server/**", + "include/**", + "patches/**", + "kernels/**", + "conversion/**", + "scripts/**", + "docs/**", + "README.md" + ], + "rule": "Look for duplicated logic that should reuse an existing helper, platform-specific branches scattered across call sites, dead or unreachable code, unexplained magic constants, and documentation that no longer matches the code." + } + ] + } +} From 1b69b625f547311c365431125018de5a645581e7 Mon Sep 17 00:00:00 2001 From: Prabhsimran Singh Date: Fri, 2 Oct 2026 03:40:39 +0530 Subject: [PATCH 06/15] release: bump version to 0.2.0 --- .greptile/config.json | 58 ------------------------------------------- VERSION | 2 +- greptile.json | 38 ---------------------------- 3 files changed, 1 insertion(+), 97 deletions(-) delete mode 100644 .greptile/config.json diff --git a/.greptile/config.json b/.greptile/config.json deleted file mode 100644 index 419f14f..0000000 --- a/.greptile/config.json +++ /dev/null @@ -1,58 +0,0 @@ -{ - "strictness": 2, - "commentTypes": ["logic", "syntax", "style"], - "triggerOnUpdates": true, - "includeBranches": ["main"], - "excludeAuthors": ["dependabot[bot]", "renovate[bot]", "pre-commit-ci[bot]"], - "ignorePatterns": "third_party/**\ntest_files/**\n", - "fixWithAI": false, - "shouldUpdateDescription": false, - "statusCommentsEnabled": false, - "summarySection": { - "included": true, - "collapsible": true, - "defaultOpen": false - }, - "issuesTableSection": { - "included": true, - "collapsible": true, - "defaultOpen": false - }, - "confidenceScoreSection": { - "included": false, - "collapsible": false, - "defaultOpen": false - }, - "sequenceDiagramSection": { - "included": false, - "collapsible": false, - "defaultOpen": false - }, - "instructions": "NeMo-Speech.cpp is a C++17 speech runtime (ASR, diarization, TTS, translation, VoiceChat) built on ggml from a pinned llama.cpp submodule, with CPU, CUDA, Metal and Vulkan backends and portable release archives for Linux, macOS and Windows. patches/ holds the project's own changes to llama.cpp and ggml, including custom GPU kernels; review them like any other code.\n\nPrioritize findings in this order: correctness, performance, portability, maintainability. Report only problems the change introduces or makes reachable, each with the concrete input, backend or platform where it fails and what to do about it. Do not comment on formatting, naming or anything pre-commit already checks.", - "rules": [ - { - "id": "correctness", - "rule": "Look for wrong results, crashes, memory errors, races, uninitialized or stale state, unhandled edge cases (empty input, boundary sizes, partial tiles or chunks), numerical instability, and fallback paths that behave differently from the fast path. Streaming results must not depend on chunk size.", - "scope": ["src/**", "app/**", "server/**", "include/**", "patches/**", "kernels/**", "conversion/**", "convert_model.py", "scripts/**", ".github/workflows/**"], - "severity": "high" - }, - { - "id": "performance", - "rule": "Look for regressions on hot paths: added host-device synchronization, per-chunk allocations, redundant work, extra copies between backends, ops that fall back to the CPU, and caches or reuse that never take effect.", - "scope": ["src/**", "server/**", "patches/**", "kernels/**"], - "severity": "medium" - }, - { - "id": "portability", - "rule": "Look for assumptions tied to one GPU architecture, backend, operating system or compiler. Hardware limits and features must be checked with a working fallback, and code must build with GCC, Clang, Apple Clang and MSVC. Release archives must stay self-contained and run on every supported target.", - "scope": ["src/**", "app/**", "server/**", "include/**", "patches/**", "kernels/**", "scripts/**", "docker/**", "CMakeLists.txt", "**/CMakeLists.txt"], - "severity": "medium" - }, - { - "id": "maintainability", - "rule": "Look for duplicated logic that should reuse an existing helper, platform-specific branches scattered across call sites, dead or unreachable code, unexplained magic constants, and documentation that no longer matches the code.", - "scope": ["src/**", "app/**", "server/**", "include/**", "patches/**", "kernels/**", "conversion/**", "scripts/**", "docs/**", "README.md"], - "severity": "low" - } - ] -} diff --git a/VERSION b/VERSION index 5e37c63..844ba4a 100644 --- a/VERSION +++ b/VERSION @@ -1 +1 @@ -NEMO_SPEECH_VERSION: 0.1.0 +NEMO_SPEECH_VERSION: 0.2.0 diff --git a/greptile.json b/greptile.json index 4b676e0..5cecc9c 100644 --- a/greptile.json +++ b/greptile.json @@ -1,43 +1,5 @@ { - "strictness": 2, - "commentTypes": [ - "logic", - "syntax", - "style" - ], "triggerOnUpdates": true, - "includeBranches": [ - "main" - ], - "excludeAuthors": [ - "dependabot[bot]", - "renovate[bot]", - "pre-commit-ci[bot]" - ], - "ignorePatterns": "greptile.json\nthird_party/**\ntest_files/**\n", - "fixWithAI": false, - "shouldUpdateDescription": false, - "statusCommentsEnabled": false, - "summarySection": { - "included": true, - "collapsible": true, - "defaultOpen": false - }, - "issuesTableSection": { - "included": true, - "collapsible": true, - "defaultOpen": false - }, - "confidenceScoreSection": { - "included": false, - "collapsible": false, - "defaultOpen": false - }, - "sequenceDiagramSection": { - "included": false, - "collapsible": false, - "defaultOpen": false - }, "instructions": "NeMo-Speech.cpp is a C++17 speech runtime (ASR, diarization, TTS, translation, VoiceChat) built on ggml from a pinned llama.cpp submodule, with CPU, CUDA, Metal and Vulkan backends and portable release archives for Linux, macOS and Windows. patches/ holds the project's own changes to llama.cpp and ggml, including custom GPU kernels; review them like any other code.\n\nPrioritize findings in this order: correctness, performance, portability, maintainability. Report only problems the change introduces or makes reachable, each with the concrete input, backend or platform where it fails and what to do about it. Do not comment on formatting, naming or anything pre-commit already checks.", "customContext": { "rules": [ From 7e30fc98586759ef4987d28c29026fb927dcee02 Mon Sep 17 00:00:00 2001 From: Prabhsimran Singh Date: Fri, 2 Oct 2026 04:09:14 +0530 Subject: [PATCH 07/15] ci: address greptile review comments - installers: identify nightly installs by archive digest so new nightlies install - nightly: upload to a draft before replacing the current release - check_release: detect AVX-512 by EVEX encoding; skip data in code sections - check_release: require filtered tar extraction --- .github/workflows/release.yml | 12 ++++++++-- scripts/install.ps1 | 13 +++++++++++ scripts/install.sh | 6 +++++ scripts/release/check_release.py | 40 +++++++++++++++++++++++++------- 4 files changed, 61 insertions(+), 10 deletions(-) diff --git a/.github/workflows/release.yml b/.github/workflows/release.yml index dfa1ea7..8dd641d 100644 --- a/.github/workflows/release.yml +++ b/.github/workflows/release.yml @@ -702,7 +702,15 @@ jobs: if: needs.prepare.outputs.channel == 'nightly' env: GH_TOKEN: ${{ github.token }} + # Upload to a draft first: if the upload fails, the current nightly stays + # in place. Only a complete draft replaces it. run: | + set -euo pipefail + staging="nightly-staging-$GITHUB_RUN_ID" + if ! gh release create "$staging" --repo "$GITHUB_REPOSITORY" --draft --prerelease \ + --target "$GITHUB_SHA" --title "Nightly" --notes "Built from $GITHUB_SHA." dist/*; then + gh release delete "$staging" --repo "$GITHUB_REPOSITORY" --yes || true + exit 1 + fi gh release delete nightly --repo "$GITHUB_REPOSITORY" --yes --cleanup-tag || true - gh release create nightly --repo "$GITHUB_REPOSITORY" --prerelease --target "$GITHUB_SHA" \ - --title "Nightly" --notes "Built from $GITHUB_SHA." dist/* + gh release edit "$staging" --repo "$GITHUB_REPOSITORY" --tag nightly --draft=false diff --git a/scripts/install.ps1 b/scripts/install.ps1 index e601b6e..513f351 100644 --- a/scripts/install.ps1 +++ b/scripts/install.ps1 @@ -205,6 +205,19 @@ if (-not (Get-Command curl.exe -ErrorAction SilentlyContinue)) { } $installIdentity = "$releaseVersion windows $arch $Backend" +# The nightly tag is rebuilt in place, so identify a nightly install by its archive digest. +if ($releaseVersion -eq 'nightly' -and -not $Source -and $binaryCandidate) { + $digestFile = [IO.Path]::GetTempFileName() + try { + Invoke-DownloadWithRetry -Uri "$url.sha256" -OutFile $digestFile + $nightlySha256 = ((Get-Content $digestFile -Raw).Trim() -split '\s+')[0] + if ($nightlySha256) { $installIdentity += " sha256:$nightlySha256" } + } catch { + # Without the digest the identity never matches, so the archive is downloaded. + } finally { + Remove-Item -Force -ErrorAction SilentlyContinue $digestFile + } +} $extraComponents = [Collections.Generic.List[string]]::new() foreach ($component in @( @{ Name = 'grpc'; Enabled = $Grpc }, @{ Name = 'nmt'; Enabled = $Nmt }, diff --git a/scripts/install.sh b/scripts/install.sh index 6f89230..721c035 100755 --- a/scripts/install.sh +++ b/scripts/install.sh @@ -239,6 +239,12 @@ echo "Prefix: $prefix" [ "$dry_run" -eq 0 ] || exit 0 install_identity="$release_version $os $arch $artifact_backend" +# The nightly tag is rebuilt in place, so identify a nightly install by its archive digest. +if [ "$release_version" = nightly ] && [ "$install_mode" != source ] && [ "$binary_candidate" -eq 1 ] && + nightly_sha256=$(curl -fsSL --retry 3 "$checksum_url" 2>/dev/null | awk 'NR == 1 { print $1 }') && + [ -n "$nightly_sha256" ]; then + install_identity="$install_identity sha256:$nightly_sha256" +fi source_identity="$release_version $os $arch $backend source:$source_ref profile:speech-server" install_metadata=$prefix/.nemo-speech-install if [ "$install_mode" != source ] && [ "$binary_candidate" -eq 1 ] && diff --git a/scripts/release/check_release.py b/scripts/release/check_release.py index 722b0ef..3c4ea7b 100755 --- a/scripts/release/check_release.py +++ b/scripts/release/check_release.py @@ -31,6 +31,13 @@ r"\.(?Ptar\.gz|zip)$" ) +# x86-64-v3 has no EVEX (AVX-512) instructions, and in 64-bit code a leading 0x62 opcode byte +# after legacy prefixes is always EVEX, so the encoding catches every AVX-512 instruction +# whatever its registers. The mnemonics below add the VEX-encoded extensions beyond v3 +# (AVX-VNNI, AMX) and AVX-512 forms reported by name. +INSTRUCTION = re.compile(r"^\s*[0-9a-f]+:\s+((?:[0-9a-f]{2} )+)\s*(\S.*)$") +LEGACY_PREFIXES = {"26", "2e", "36", "3e", "64", "65", "66", "67", "f0", "f2", "f3"} + # AT&T syntax, as printed by objdump and llvm-objdump. BEYOND_X86_64_V3 = re.compile( r"%zmm\d+|%[xy]mm(?:1[6-9]|2\d|3[01])\b|%k[0-7]\b|\{%k[0-7]\}" @@ -74,6 +81,24 @@ def disassembler() -> str: raise SystemExit("error: llvm-objdump or objdump is required") +def beyond_v3(disassembly: str) -> list[str]: + """Return the instructions beyond x86-64-v3 in llvm-objdump or objdump output.""" + lines = [m for m in map(INSTRUCTION.match, disassembly.splitlines()) if m] + # Bytes the disassembler cannot decode are data in a code section, such as a jump table: + # llvm-objdump prints , GNU objdump (bad). Data next to them can decode as a + # valid-looking instruction, so ignore hits adjacent to undecodable bytes; real AVX-512 + # code never appears only there. + data = [m.group(2).startswith(("(bad)", "")) for m in lines] + hits = [] + for i, m in enumerate(lines): + if data[i] or (i > 0 and data[i - 1]) or (i + 1 < len(lines) and data[i + 1]): + continue + opcode = next((b for b in m.group(1).split() if b not in LEGACY_PREFIXES), "") + if opcode == "62" or BEYOND_X86_64_V3.search(m.group(2)): + hits.append(m.group(0).strip()) + return hits + + def scan_isa(paths: list[pathlib.Path]) -> list[str]: tool = disassembler() failures = [] @@ -90,7 +115,7 @@ def scan_isa(paths: list[pathlib.Path]) -> list[str]: if arch != "x86_64": continue result = subprocess.run( - [tool, "-d", "--no-show-raw-insn", str(path)], + [tool, "-d", str(path)], capture_output=True, text=True, errors="replace", @@ -98,9 +123,7 @@ def scan_isa(paths: list[pathlib.Path]) -> list[str]: if result.returncode != 0 or "file format not recognized" in result.stderr: failures.append(f"{path}: {tool} could not disassemble it") continue - hits = [ - line.strip() for line in result.stdout.splitlines() if BEYOND_X86_64_V3.search(line) - ] + hits = beyond_v3(result.stdout) if hits: failures.append( f"{path}: {len(hits)} instructions beyond x86-64-v3, first: {hits[0]}" @@ -113,11 +136,12 @@ def extract(archive: pathlib.Path, dest: pathlib.Path) -> None: with zipfile.ZipFile(archive) as z: z.extractall(dest) else: + if not hasattr(tarfile, "data_filter"): + raise SystemExit( + "error: safe tar extraction needs Python 3.12 or a release with tarfile.data_filter" + ) with tarfile.open(archive) as t: - if hasattr(tarfile, "data_filter"): - t.extractall(dest, filter="data") - else: - t.extractall(dest) + t.extractall(dest, filter="data") def check_archives(directory: pathlib.Path, version: str) -> list[str]: From 789b3580fd440fe758c00a9d7f2b92d1906a622c Mon Sep 17 00:00:00 2001 From: Prabhsimran Singh Date: Fri, 2 Oct 2026 04:33:36 +0530 Subject: [PATCH 08/15] ci: switch coderabbit profile to chill, disable summary in desc --- .coderabbit.yaml | 5 ++++- 1 file changed, 4 insertions(+), 1 deletion(-) diff --git a/.coderabbit.yaml b/.coderabbit.yaml index 970ef7e..7f8de18 100644 --- a/.coderabbit.yaml +++ b/.coderabbit.yaml @@ -4,7 +4,10 @@ # yaml-language-server: $schema=https://coderabbit.ai/integrations/schema.v2.json reviews: - profile: assertive + profile: chill + # Keep the summary in the walkthrough comment instead of editing the PR description. + high_level_summary: false + high_level_summary_in_walkthrough: true auto_review: enabled: true auto_incremental_review: true From b8c29b31e75eef6e19dad65c4ab5011f1c58ffe4 Mon Sep 17 00:00:00 2001 From: Prabhsimran Singh Date: Fri, 2 Oct 2026 08:57:32 +0000 Subject: [PATCH 09/15] ci: fix release scan and windows smoke test --- .github/workflows/release.yml | 18 ++++++++++++++++++ scripts/release/check_release.py | 6 ++++-- 2 files changed, 22 insertions(+), 2 deletions(-) diff --git a/.github/workflows/release.yml b/.github/workflows/release.yml index 8dd641d..5f4f1ef 100644 --- a/.github/workflows/release.yml +++ b/.github/workflows/release.yml @@ -644,6 +644,24 @@ jobs: path: ${{ env.NEMO_SPEECH_MODEL_DIR }} key: models-${{ hashFiles('models/index.json') }} + - name: Install Python + # The ephemeral VM does not necessarily ship Python (see gpu.yml). + shell: pwsh + run: | + $ErrorActionPreference = 'Stop' + if (Get-Command python -ErrorAction SilentlyContinue) { exit 0 } + if (-not (Get-Command choco -ErrorAction SilentlyContinue)) { + Set-ExecutionPolicy Bypass -Scope Process -Force + [Net.ServicePointManager]::SecurityProtocol = [Net.SecurityProtocolType]::Tls12 + Invoke-Expression ((New-Object Net.WebClient).DownloadString('https://community.chocolatey.org/install.ps1')) + $env:Path = "$env:ProgramData\chocolatey\bin;$env:Path" + } + choco install -y --no-progress python + # Later steps start with the job's original PATH; publish the new one. + $machine = [Environment]::GetEnvironmentVariable('Path', 'Machine') + $user = [Environment]::GetEnvironmentVariable('Path', 'User') + "$machine;$user" -split ';' | Where-Object { $_ } | ForEach-Object { Add-Content -Path $env:GITHUB_PATH -Value $_ } + - name: Model smoke test shell: pwsh run: | diff --git a/scripts/release/check_release.py b/scripts/release/check_release.py index 3c4ea7b..50ac2ef 100755 --- a/scripts/release/check_release.py +++ b/scripts/release/check_release.py @@ -34,8 +34,10 @@ # x86-64-v3 has no EVEX (AVX-512) instructions, and in 64-bit code a leading 0x62 opcode byte # after legacy prefixes is always EVEX, so the encoding catches every AVX-512 instruction # whatever its registers. The mnemonics below add the VEX-encoded extensions beyond v3 -# (AVX-VNNI, AMX) and AVX-512 forms reported by name. -INSTRUCTION = re.compile(r"^\s*[0-9a-f]+:\s+((?:[0-9a-f]{2} )+)\s*(\S.*)$") +# (AVX-VNNI, AMX) and AVX-512 forms reported by name. Both disassemblers put a tab before the +# mnemonic; GNU objdump continues a long instruction's bytes on lines without one, which are +# not instructions. +INSTRUCTION = re.compile(r"^\s*[0-9a-f]+:\s+((?:[0-9a-f]{2} )+)\s*\t(\S.*)$") LEGACY_PREFIXES = {"26", "2e", "36", "3e", "64", "65", "66", "67", "f0", "f2", "f3"} # AT&T syntax, as printed by objdump and llvm-objdump. From acfada79592f8262cef6f3e20430c596238a1afa Mon Sep 17 00:00:00 2001 From: Prabhsimran Singh Date: Fri, 2 Oct 2026 10:35:24 +0000 Subject: [PATCH 10/15] fix(cuda): export the cuBLAS calls the llama.cpp pin needs from the shim - add cublasSetWorkspace_v2 (accepted and unused) and cublasSgemmBatched - test pointer-array f32 GEMMs in test_cublas_shim_numerics - fail CUDA release packaging when ggml-cuda imports a cuBLAS function the shim does not export --- kernels/cublas_shim.cu | 19 ++++++ kernels/ver_cublas.map | 2 + scripts/release/package-linux.sh | 11 ++++ scripts/windows/package-release.ps1 | 23 ++++++++ tests/cpp/test_cublas_shim_numerics.cu | 82 ++++++++++++++++++++++++++ 5 files changed, 137 insertions(+) diff --git a/kernels/cublas_shim.cu b/kernels/cublas_shim.cu index 07d1efa..6162912 100644 --- a/kernels/cublas_shim.cu +++ b/kernels/cublas_shim.cu @@ -1414,6 +1414,12 @@ NEMO_SPEECH_CUBLAS_EXPORT cublasStatus_t cublasSetMathMode(cublasHandle_t, cublasMath_t) { return STATUS_SUCCESS; } +// The shim allocates its own split-K workspaces, so a caller-provided +// workspace is accepted and left unused. +NEMO_SPEECH_CUBLAS_EXPORT cublasStatus_t +cublasSetWorkspace_v2(cublasHandle_t, void*, size_t) { + return STATUS_SUCCESS; +} NEMO_SPEECH_CUBLAS_EXPORT const char* cublasGetStatusString(cublasStatus_t) { return "EDGE_SHIM_OK"; @@ -1484,6 +1490,19 @@ cublasSgemmStridedBatched( return STATUS_SUCCESS; } NEMO_SPEECH_CUBLAS_EXPORT cublasStatus_t +cublasSgemmBatched( + cublasHandle_t h, cublasOperation_t opA, cublasOperation_t opB, int m, int n, int k, + const float* alpha, const float* const Aarray[], int lda, const float* const Barray[], int ldb, + const float* beta, float* const Carray[], int ldc, int batch) { + ShimHandle* sh = (ShimHandle*)h; + const cudaStream_t stream = stream_for_handle(sh); + dim3 blk(16, 16, 1), grd((m + 15) / 16, (n + 15) / 16, batch); + k_ptrs<<>>( + m, n, k, opA, opB, (const void* const*)Aarray, lda, R_32F, (const void* const*)Barray, ldb, + R_32F, (void* const*)Carray, ldc, R_32F, *alpha, *beta, batch); + return STATUS_SUCCESS; +} +NEMO_SPEECH_CUBLAS_EXPORT cublasStatus_t cublasStrsmBatched( cublasHandle_t, cublasSideMode_t, cublasFillMode_t, cublasOperation_t, cublasDiagType_t, int, int, const float*, const float* const[], int, float* const[], int, int) { diff --git a/kernels/ver_cublas.map b/kernels/ver_cublas.map index 76f1682..754e58b 100644 --- a/kernels/ver_cublas.map +++ b/kernels/ver_cublas.map @@ -7,11 +7,13 @@ libcublas.so.@NEMO_SPEECH_CUBLAS_SOVERSION@ { cublasDestroy_v2; cublasSetStream_v2; cublasSetMathMode; + cublasSetWorkspace_v2; cublasGetStatusString; cublasGemmEx; cublasGemmStridedBatchedEx; cublasGemmBatchedEx; cublasSgemm_v2; + cublasSgemmBatched; cublasSgemmStridedBatched; cublasStrsmBatched; local: *; diff --git a/scripts/release/package-linux.sh b/scripts/release/package-linux.sh index 45912cf..a93436e 100755 --- a/scripts/release/package-linux.sh +++ b/scripts/release/package-linux.sh @@ -286,6 +286,17 @@ if [[ "$backend" == cuda ]]; then echo "error: libggml-cuda requires '$required_cublas'; expected '$cublas_soname'" >&2 exit 1 } + # The shim implements only the cuBLAS calls ggml makes; a call added by a + # llama.cpp update must be added to kernels/ before it can ship. + missing_cublas="$(comm -23 \ + <(nm -D --undefined-only "$ggml_cuda" | awk '{ sub(/@.*/, "", $2); print $2 }' | + grep '^cublas' | sort -u) \ + <(nm -D --defined-only "$cublas_shim" | awk '{ sub(/@.*/, "", $3); print $3 }' | sort -u))" + [[ -z "$missing_cublas" ]] || { + echo "error: the cuBLAS shim does not export these functions libggml-cuda calls:" >&2 + sed 's/^/ /' <<< "$missing_cublas" >&2 + exit 1 + } cuda_license="/usr/share/doc/cuda-cudart-${cuda_version}/copyright" [[ -f "$cuda_license" ]] || { echo "error: CUDA runtime license was not found: $cuda_license" >&2 diff --git a/scripts/windows/package-release.ps1 b/scripts/windows/package-release.ps1 index 5529ddd..e382870 100644 --- a/scripts/windows/package-release.ps1 +++ b/scripts/windows/package-release.ps1 @@ -95,6 +95,29 @@ try { throw "the package is not self-contained:`n $($missing -join "`n ")" } + # The cuBLAS shim implements only the calls ggml makes; a call added by a + # llama.cpp update must be added to kernels\ before it can ship. + if ($Backend -eq 'cuda') { + $shim = (Get-ChildItem $bin -Filter 'cublas64_*.dll' | Select-Object -First 1) + $exported = @{} + & $dumpbin /nologo /exports $shim.FullName | ForEach-Object { + if ($_ -match '^\s+\d+\s+[0-9A-Fa-f]+\s+[0-9A-Fa-f]+\s+(\S+)') { $exported[$Matches[1]] = $true } + } + $dll = $null + $unexported = @() + & $dumpbin /nologo /imports (Join-Path $bin 'ggml-cuda.dll') | ForEach-Object { + if ($_ -match '^\s{4}(\S+\.dll)\s*$') { + $dll = $Matches[1] + } elseif ($dll -ieq $shim.Name -and $_ -match '^\s+[0-9A-Fa-f]+\s+(\S+)\s*$' -and + -not $exported.ContainsKey($Matches[1])) { + $unexported += $Matches[1] + } + } + if ($unexported) { + throw "the cuBLAS shim does not export these functions ggml-cuda.dll calls:`n $($unexported -join "`n ")" + } + } + $zip = Join-Path (Resolve-Path $OutputDir) "$name.zip" if (Test-Path $zip) { Remove-Item $zip } Add-Type -AssemblyName System.IO.Compression.FileSystem diff --git a/tests/cpp/test_cublas_shim_numerics.cu b/tests/cpp/test_cublas_shim_numerics.cu index 20ad716..22a9990 100644 --- a/tests/cpp/test_cublas_shim_numerics.cu +++ b/tests/cpp/test_cublas_shim_numerics.cu @@ -167,6 +167,87 @@ run_batched_f32_case(cublasHandle_t handle) { return true; } +// Pointer-array f32 GEMM with a transposed B, as ggml's OUT_PROD issues it. +bool +run_pointer_batched_f32_case(cublasHandle_t handle) { + constexpr int m = 37; + constexpr int n = 29; + constexpr int k = 13; + constexpr int batch = 5; + constexpr size_t size_a = (size_t)m * k; // m x k, column-major + constexpr size_t size_b = (size_t)n * k; // n x k, used transposed + constexpr size_t size_c = (size_t)m * n; + + std::vector a(batch * size_a), b(batch * size_b), c(batch * size_c); + for (size_t i = 0; i < a.size(); ++i) a[i] = (float)((i * 7) % 11) - 5.0f; + for (size_t i = 0; i < b.size(); ++i) b[i] = (float)((i * 5) % 13) - 6.0f; + + float* device = nullptr; + const float** device_ptrs = nullptr; + const size_t total = a.size() + b.size() + c.size(); + bool ok = check_cuda(cudaMalloc(&device, total * sizeof(float)), "cudaMalloc(batched)") && + check_cuda(cudaMalloc(&device_ptrs, 3 * batch * sizeof(float*)), "cudaMalloc(ptrs)"); + float* device_a = device; + float* device_b = device_a + a.size(); + float* device_c = device_b + b.size(); + std::vector ptrs(3 * batch); + for (int item = 0; item < batch; ++item) { + ptrs[item] = device_a + item * size_a; + ptrs[batch + item] = device_b + item * size_b; + ptrs[2 * batch + item] = device_c + item * size_c; + } + ok = ok && + check_cuda( + cudaMemcpy(device_a, a.data(), a.size() * sizeof(float), cudaMemcpyHostToDevice), + "cudaMemcpy(A batched)") && + check_cuda( + cudaMemcpy(device_b, b.data(), b.size() * sizeof(float), cudaMemcpyHostToDevice), + "cudaMemcpy(B batched)") && + check_cuda( + cudaMemcpy( + device_ptrs, ptrs.data(), ptrs.size() * sizeof(float*), cudaMemcpyHostToDevice), + "cudaMemcpy(ptrs)"); + + const float alpha = 1.0f; + const float beta = 0.0f; + if (ok) { + ok = check_cublas( + cublasSgemmBatched( + handle, CUBLAS_OP_N, CUBLAS_OP_T, m, n, k, &alpha, device_ptrs, m, + device_ptrs + batch, n, &beta, (float* const*)(device_ptrs + 2 * batch), m, batch), + "cublasSgemmBatched"); + } + if (ok) { + ok = check_cuda( + cudaMemcpy(c.data(), device_c, c.size() * sizeof(float), cudaMemcpyDeviceToHost), + "cudaMemcpy(C batched)"); + } + cudaFree(device); + cudaFree(device_ptrs); + if (!ok) { + return false; + } + for (int item = 0; item < batch; ++item) { + for (int col = 0; col < n; ++col) { + for (int row = 0; row < m; ++row) { + float expected = 0.0f; + for (int i = 0; i < k; ++i) { + expected += a[item * size_a + (size_t)i * m + row] * + b[item * size_b + (size_t)i * n + col]; + } + const float actual = c[item * size_c + (size_t)col * m + row]; + if (std::fabs(actual - expected) > 1.0e-3f) { + std::fprintf( + stderr, "FAIL: pointer-batched f32 [%d,%d,%d]=%g, expected %g\n", item, row, + col, actual, expected); + return false; + } + } + } + } + return true; +} + bool run_stream_churn_case(cublasHandle_t handle) { bool ok = true; @@ -201,6 +282,7 @@ main() { ok &= run_cancellation_case(handle, 24, 432, 1296); ok &= run_cancellation_case(handle, 768, 111, 768); ok &= run_batched_f32_case(handle); + ok &= run_pointer_batched_f32_case(handle); ok &= run_stream_churn_case(handle); ok &= check_cublas(cublasDestroy(handle), "cublasDestroy"); From 0b56aae50df25b52c1e3d66761d8e337bfd9e932 Mon Sep 17 00:00:00 2001 From: Prabhsimran Singh Date: Fri, 2 Oct 2026 10:47:35 +0000 Subject: [PATCH 11/15] fix(cuda): handle large and empty batches in the cuBLAS shim --- kernels/cublas_shim.cu | 76 +++++++++++++---- scripts/release/package-linux.sh | 16 +++- scripts/windows/package-release.ps1 | 13 ++- tests/cpp/test_cublas_shim_numerics.cu | 114 ++++++++++++++++++++++++- 4 files changed, 196 insertions(+), 23 deletions(-) diff --git a/kernels/cublas_shim.cu b/kernels/cublas_shim.cu index 6162912..ada62e0 100644 --- a/kernels/cublas_shim.cu +++ b/kernels/cublas_shim.cu @@ -36,7 +36,7 @@ enum { OP_N = 0, OP_T = 1 }; enum { R_32F = 0, R_16F = 2, R_16BF = 14 }; enum { COMPUTE_16F = 64, COMPUTE_32F = 68 }; -enum { STATUS_SUCCESS = 0 }; +enum { STATUS_SUCCESS = 0, STATUS_EXECUTION_FAILED = 13 }; // cudaDataType is already defined by library_types.h (via cuda_runtime.h). typedef void* cublasHandle_t; @@ -1238,6 +1238,9 @@ launch( int m, int n, int k, int opA, int opB, const void* A, int lda, long long sa, int ta, const void* B, int ldb, long long sb, int tb, void* C, int ldc, long long sc, int tc, float alpha, float beta, int batch, cudaStream_t s, ShimHandle* sh) { + if (m <= 0 || n <= 0) { + return; + } int bb = batch > 0 ? batch : 1; if (use_hgemv_tn(n, opA, opB, ta, tb, tc)) { if (tc == R_16F && alpha == 1.0f && beta == 0.0f) { @@ -1362,6 +1365,51 @@ launch( m, n, k, opA, opB, A, lda, sa, ta, B, ldb, sb, tb, C, ldc, sc, tc, alpha, beta, esz(ta), esz(tb), esz(tc), bb); } + +// The batch index is carried in grid.z, which CUDA limits to 65535; larger +// batches are issued as several launches. +constexpr int kMaxGridZ = 65535; + +inline const void* +offset(const void* p, long long elements, int dt) { + return (const char*)p + elements * (long long)esz(dt); +} + +inline void +launch_strided( + int m, int n, int k, int opA, int opB, const void* A, int lda, long long sa, int ta, + const void* B, int ldb, long long sb, int tb, void* C, int ldc, long long sc, int tc, + float alpha, float beta, int batch, cudaStream_t s, ShimHandle* sh) { + for (int b0 = 0; b0 < batch; b0 += kMaxGridZ) { + const int count = std::min(batch - b0, kMaxGridZ); + launch( + m, n, k, opA, opB, offset(A, b0 * sa, ta), lda, sa, ta, offset(B, b0 * sb, tb), ldb, sb, + tb, (void*)offset(C, b0 * sc, tc), ldc, sc, tc, alpha, beta, count, s, sh); + } +} + +inline void +launch_ptrs( + int m, int n, int k, int opA, int opB, const void* const* A, int lda, int ta, + const void* const* B, int ldb, int tb, void* const* C, int ldc, int tc, float alpha, float beta, + int batch, cudaStream_t s) { + if (m <= 0 || n <= 0) { + return; + } + for (int b0 = 0; b0 < batch; b0 += kMaxGridZ) { + const int count = std::min(batch - b0, kMaxGridZ); + dim3 blk(16, 16, 1), grd((m + 15) / 16, (n + 15) / 16, count); + k_ptrs<<>>( + m, n, k, opA, opB, A + b0, lda, ta, B + b0, ldb, tb, C + b0, ldc, tc, alpha, beta, + count); + } +} + +// Report a failed kernel launch instead of claiming success. +inline cublasStatus_t +launch_status() { + return cudaGetLastError() == cudaSuccess ? STATUS_SUCCESS : STATUS_EXECUTION_FAILED; +} } // namespace extern "C" { @@ -1436,7 +1484,7 @@ cublasGemmEx( launch( m, n, k, opA, opB, A, lda, 0, ta, B, ldb, 0, tb, C, ldc, 0, tc, host_scalar(alpha, ct), host_scalar(beta, ct), 1, stream, sh); - return STATUS_SUCCESS; + return launch_status(); } NEMO_SPEECH_CUBLAS_EXPORT cublasStatus_t cublasGemmStridedBatchedEx( @@ -1446,10 +1494,10 @@ cublasGemmStridedBatchedEx( long long sc, int batch, cublasComputeType_t ct, cublasGemmAlgo_t) { ShimHandle* sh = (ShimHandle*)h; const cudaStream_t stream = stream_for_handle(sh); - launch( + launch_strided( m, n, k, opA, opB, A, lda, sa, ta, B, ldb, sb, tb, C, ldc, sc, tc, host_scalar(alpha, ct), host_scalar(beta, ct), batch, stream, sh); - return STATUS_SUCCESS; + return launch_status(); } NEMO_SPEECH_CUBLAS_EXPORT cublasStatus_t cublasGemmBatchedEx( @@ -1459,11 +1507,10 @@ cublasGemmBatchedEx( cudaDataType tc, int ldc, int batch, cublasComputeType_t ct, cublasGemmAlgo_t) { ShimHandle* sh = (ShimHandle*)h; const cudaStream_t stream = stream_for_handle(sh); - dim3 blk(16, 16, 1), grd((m + 15) / 16, (n + 15) / 16, batch); - k_ptrs<<>>( + launch_ptrs( m, n, k, opA, opB, Aarray, lda, ta, Barray, ldb, tb, Carray, ldc, tc, - host_scalar(alpha, ct), host_scalar(beta, ct), batch); - return STATUS_SUCCESS; + host_scalar(alpha, ct), host_scalar(beta, ct), batch, stream); + return launch_status(); } NEMO_SPEECH_CUBLAS_EXPORT cublasStatus_t cublasSgemm_v2( @@ -1475,7 +1522,7 @@ cublasSgemm_v2( launch( m, n, k, opA, opB, A, lda, 0, R_32F, B, ldb, 0, R_32F, C, ldc, 0, R_32F, *alpha, *beta, 1, stream, sh); - return STATUS_SUCCESS; + return launch_status(); } NEMO_SPEECH_CUBLAS_EXPORT cublasStatus_t cublasSgemmStridedBatched( @@ -1484,10 +1531,10 @@ cublasSgemmStridedBatched( long long sb, const float* beta, float* C, int ldc, long long sc, int batch) { ShimHandle* sh = (ShimHandle*)h; const cudaStream_t stream = stream_for_handle(sh); - launch( + launch_strided( m, n, k, opA, opB, A, lda, sa, R_32F, B, ldb, sb, R_32F, C, ldc, sc, R_32F, *alpha, *beta, batch, stream, sh); - return STATUS_SUCCESS; + return launch_status(); } NEMO_SPEECH_CUBLAS_EXPORT cublasStatus_t cublasSgemmBatched( @@ -1496,11 +1543,10 @@ cublasSgemmBatched( const float* beta, float* const Carray[], int ldc, int batch) { ShimHandle* sh = (ShimHandle*)h; const cudaStream_t stream = stream_for_handle(sh); - dim3 blk(16, 16, 1), grd((m + 15) / 16, (n + 15) / 16, batch); - k_ptrs<<>>( + launch_ptrs( m, n, k, opA, opB, (const void* const*)Aarray, lda, R_32F, (const void* const*)Barray, ldb, - R_32F, (void* const*)Carray, ldc, R_32F, *alpha, *beta, batch); - return STATUS_SUCCESS; + R_32F, (void* const*)Carray, ldc, R_32F, *alpha, *beta, batch, stream); + return launch_status(); } NEMO_SPEECH_CUBLAS_EXPORT cublasStatus_t cublasStrsmBatched( diff --git a/scripts/release/package-linux.sh b/scripts/release/package-linux.sh index a93436e..1bedf9d 100755 --- a/scripts/release/package-linux.sh +++ b/scripts/release/package-linux.sh @@ -288,10 +288,18 @@ if [[ "$backend" == cuda ]]; then } # The shim implements only the cuBLAS calls ggml makes; a call added by a # llama.cpp update must be added to kernels/ before it can ship. - missing_cublas="$(comm -23 \ - <(nm -D --undefined-only "$ggml_cuda" | awk '{ sub(/@.*/, "", $2); print $2 }' | - grep '^cublas' | sort -u) \ - <(nm -D --defined-only "$cublas_shim" | awk '{ sub(/@.*/, "", $3); print $3 }' | sort -u))" + imported_cublas="$work_dir/imported-cublas" + exported_cublas="$work_dir/exported-cublas" + nm -D --undefined-only "$ggml_cuda" > "$imported_cublas.raw" + nm -D --defined-only "$cublas_shim" > "$exported_cublas.raw" + awk '{ sub(/@.*/, "", $2); if ($2 ~ /^cublas/) print $2 }' "$imported_cublas.raw" | + sort -u > "$imported_cublas" + awk '{ sub(/@.*/, "", $3); print $3 }' "$exported_cublas.raw" | sort -u > "$exported_cublas" + [[ -s "$imported_cublas" && -s "$exported_cublas" ]] || { + echo "error: could not read the cuBLAS symbols of libggml-cuda or the shim" >&2 + exit 1 + } + missing_cublas="$(comm -23 "$imported_cublas" "$exported_cublas")" [[ -z "$missing_cublas" ]] || { echo "error: the cuBLAS shim does not export these functions libggml-cuda calls:" >&2 sed 's/^/ /' <<< "$missing_cublas" >&2 diff --git a/scripts/windows/package-release.ps1 b/scripts/windows/package-release.ps1 index e382870..23f0764 100644 --- a/scripts/windows/package-release.ps1 +++ b/scripts/windows/package-release.ps1 @@ -74,6 +74,12 @@ try { $vswhere = "${env:ProgramFiles(x86)}\Microsoft Visual Studio\Installer\vswhere.exe" $dumpbin = & $vswhere -latest -products * -find '**\Hostx64\x64\dumpbin.exe' | Select-Object -First 1 if (-not $dumpbin) { throw 'dumpbin.exe was not found (Visual Studio C++ tools are required)' } + # A failed inspection would otherwise read as "no dependencies". + function Invoke-Dumpbin([string]$Option, [string]$Path) { + $output = & $dumpbin /nologo $Option $Path + if ($LASTEXITCODE -ne 0) { throw "dumpbin $Option failed for $Path ($LASTEXITCODE)" } + $output + } $system = '^(api-ms-win-.*|ext-ms-.*|kernel32|kernelbase|user32|gdi32|advapi32|shell32|ole32|oleaut32|' + 'ws2_32|wsock32|mswsock|bcrypt|crypt32|ncrypt|secur32|dbghelp|shlwapi|winmm|ntdll|userenv|' + 'psapi|version|iphlpapi|setupapi|cfgmgr32|comctl32|comdlg32|rpcrt4|dnsapi|powrprof|winhttp|' + @@ -82,7 +88,7 @@ try { Get-ChildItem $bin -Filter '*.dll' | ForEach-Object { $bundled[$_.Name.ToLowerInvariant()] = $true } $missing = @() foreach ($file in Get-ChildItem $bin -Include '*.dll', '*.exe' -Recurse) { - $dependencies = & $dumpbin /nologo /dependents $file.FullName | ForEach-Object { + $dependencies = Invoke-Dumpbin /dependents $file.FullName | ForEach-Object { if ($_ -match '^\s+(\S+\.dll)\s*$') { $Matches[1].ToLowerInvariant() } } foreach ($dependency in $dependencies) { @@ -100,12 +106,12 @@ try { if ($Backend -eq 'cuda') { $shim = (Get-ChildItem $bin -Filter 'cublas64_*.dll' | Select-Object -First 1) $exported = @{} - & $dumpbin /nologo /exports $shim.FullName | ForEach-Object { + Invoke-Dumpbin /exports $shim.FullName | ForEach-Object { if ($_ -match '^\s+\d+\s+[0-9A-Fa-f]+\s+[0-9A-Fa-f]+\s+(\S+)') { $exported[$Matches[1]] = $true } } $dll = $null $unexported = @() - & $dumpbin /nologo /imports (Join-Path $bin 'ggml-cuda.dll') | ForEach-Object { + Invoke-Dumpbin /imports (Join-Path $bin 'ggml-cuda.dll') | ForEach-Object { if ($_ -match '^\s{4}(\S+\.dll)\s*$') { $dll = $Matches[1] } elseif ($dll -ieq $shim.Name -and $_ -match '^\s+[0-9A-Fa-f]+\s+(\S+)\s*$' -and @@ -113,6 +119,7 @@ try { $unexported += $Matches[1] } } + if (-not $exported.Count) { throw "dumpbin listed no exports for $($shim.Name)" } if ($unexported) { throw "the cuBLAS shim does not export these functions ggml-cuda.dll calls:`n $($unexported -join "`n ")" } diff --git a/tests/cpp/test_cublas_shim_numerics.cu b/tests/cpp/test_cublas_shim_numerics.cu index 22a9990..20eef1f 100644 --- a/tests/cpp/test_cublas_shim_numerics.cu +++ b/tests/cpp/test_cublas_shim_numerics.cu @@ -236,7 +236,7 @@ run_pointer_batched_f32_case(cublasHandle_t handle) { b[item * size_b + (size_t)i * n + col]; } const float actual = c[item * size_c + (size_t)col * m + row]; - if (std::fabs(actual - expected) > 1.0e-3f) { + if (!std::isfinite(actual) || std::fabs(actual - expected) > 1.0e-3f) { std::fprintf( stderr, "FAIL: pointer-batched f32 [%d,%d,%d]=%g, expected %g\n", item, row, col, actual, expected); @@ -248,6 +248,117 @@ run_pointer_batched_f32_case(cublasHandle_t handle) { return true; } +// More batches than CUDA's grid.z limit (65535), and a zero-sized no-op. +bool +run_large_batch_case(cublasHandle_t handle) { + constexpr int batch = 70000; + constexpr int dim = 2; + constexpr size_t size = (size_t)dim * dim; + std::vector a(batch * size), b(batch * size), c(batch * size); + for (int item = 0; item < batch; ++item) { + for (size_t i = 0; i < size; ++i) { + a[item * size + i] = (float)(item % 7) + (float)i; + b[item * size + i] = i % 3 == 0 ? 1.0f : 0.5f; + } + } + + float* device = nullptr; + const float** device_ptrs = nullptr; + bool ok = check_cuda(cudaMalloc(&device, 3 * a.size() * sizeof(float)), "cudaMalloc(large)") && + check_cuda(cudaMalloc(&device_ptrs, 3 * batch * sizeof(float*)), "cudaMalloc(ptrs)"); + float* device_a = device; + float* device_b = device_a + a.size(); + float* device_c = device_b + b.size(); + std::vector ptrs(3 * batch); + for (int item = 0; item < batch; ++item) { + ptrs[item] = device_a + item * size; + ptrs[batch + item] = device_b + item * size; + ptrs[2 * batch + item] = device_c + item * size; + } + ok = ok && + check_cuda( + cudaMemcpy(device_a, a.data(), a.size() * sizeof(float), cudaMemcpyHostToDevice), + "cudaMemcpy(A large)") && + check_cuda( + cudaMemcpy(device_b, b.data(), b.size() * sizeof(float), cudaMemcpyHostToDevice), + "cudaMemcpy(B large)") && + check_cuda( + cudaMemcpy( + device_ptrs, ptrs.data(), ptrs.size() * sizeof(float*), cudaMemcpyHostToDevice), + "cudaMemcpy(ptrs large)"); + + const float alpha = 1.0f; + const float beta = 0.0f; + std::vector expected(c.size()); + for (int item = 0; item < batch; ++item) { + for (int col = 0; col < dim; ++col) { + for (int row = 0; row < dim; ++row) { + float sum = 0.0f; + for (int i = 0; i < dim; ++i) { + sum += a[item * size + (size_t)i * dim + row] * + b[item * size + (size_t)col * dim + i]; + } + expected[item * size + (size_t)col * dim + row] = sum; + } + } + } + const auto check_output = [&](const char* label) { + if (!check_cuda( + cudaMemcpy(c.data(), device_c, c.size() * sizeof(float), cudaMemcpyDeviceToHost), + "cudaMemcpy(C large)")) { + return false; + } + for (size_t i = 0; i < c.size(); ++i) { + if (!std::isfinite(c[i]) || std::fabs(c[i] - expected[i]) > 1.0e-4f) { + std::fprintf( + stderr, "FAIL: %s output[%zu]=%g, expected %g\n", label, i, c[i], expected[i]); + return false; + } + } + return true; + }; + + if (ok) { + ok = check_cuda(cudaMemset(device_c, 0xff, c.size() * sizeof(float)), "cudaMemset(C)") && + check_cublas( + cublasSgemmBatched( + handle, CUBLAS_OP_N, CUBLAS_OP_N, dim, dim, dim, &alpha, device_ptrs, dim, + device_ptrs + batch, dim, &beta, (float* const*)(device_ptrs + 2 * batch), dim, + batch), + "cublasSgemmBatched(large)") && + check_output("pointer-batched large"); + } + if (ok) { + ok = check_cuda(cudaMemset(device_c, 0xff, c.size() * sizeof(float)), "cudaMemset(C)") && + check_cublas( + cublasSgemmStridedBatched( + handle, CUBLAS_OP_N, CUBLAS_OP_N, dim, dim, dim, &alpha, device_a, dim, + (long long)size, device_b, dim, (long long)size, &beta, device_c, dim, + (long long)size, batch), + "cublasSgemmStridedBatched(large)") && + check_output("strided-batched large"); + } + // Zero-sized problems complete without touching C. + if (ok) { + ok = check_cublas( + cublasSgemmBatched( + handle, CUBLAS_OP_N, CUBLAS_OP_N, dim, dim, dim, &alpha, device_ptrs, dim, + device_ptrs + batch, dim, &beta, (float* const*)(device_ptrs + 2 * batch), dim, + 0), + "cublasSgemmBatched(batch=0)") && + check_cublas( + cublasSgemmStridedBatched( + handle, CUBLAS_OP_N, CUBLAS_OP_N, 0, dim, dim, &alpha, device_a, dim, 0, + device_b, dim, 0, &beta, device_c, dim, 0, 1), + "cublasSgemmStridedBatched(m=0)") && + check_output("zero-sized no-op"); + } + + cudaFree(device); + cudaFree(device_ptrs); + return ok; +} + bool run_stream_churn_case(cublasHandle_t handle) { bool ok = true; @@ -283,6 +394,7 @@ main() { ok &= run_cancellation_case(handle, 768, 111, 768); ok &= run_batched_f32_case(handle); ok &= run_pointer_batched_f32_case(handle); + ok &= run_large_batch_case(handle); ok &= run_stream_churn_case(handle); ok &= check_cublas(cublasDestroy(handle), "cublasDestroy"); From d12885314b0c9c89e91a72705ae117ab6ac78e44 Mon Sep 17 00:00:00 2001 From: Prabhsimran Singh Date: Fri, 2 Oct 2026 10:52:26 +0000 Subject: [PATCH 12/15] ci: install windows build tools on nv-gha runner --- .github/workflows/release.yml | 28 ++++++++++++++++++++++++++++ 1 file changed, 28 insertions(+) diff --git a/.github/workflows/release.yml b/.github/workflows/release.yml index 5f4f1ef..ff8c360 100644 --- a/.github/workflows/release.yml +++ b/.github/workflows/release.yml @@ -398,6 +398,34 @@ jobs: run: | git submodule update --init --depth 1 llama.cpp third_party/cpp-httplib + - name: Install build prerequisites + # Hosted windows-2022 images have these; self-hosted VMs selected with + # RELEASE_WINDOWS_CUDA_RUNNER may not (see gpu.yml). + shell: pwsh + run: | + $ErrorActionPreference = 'Stop' + $packages = @() + foreach ($tool in @{ cmake = 'cmake'; ninja = 'ninja' }.GetEnumerator()) { + if (-not (Get-Command $tool.Key -ErrorAction SilentlyContinue)) { $packages += $tool.Value } + } + $vswhere = "${env:ProgramFiles(x86)}\Microsoft Visual Studio\Installer\vswhere.exe" + $hasVc = (Test-Path $vswhere) -and + (& $vswhere -latest -products * -requires Microsoft.VisualStudio.Component.VC.Tools.x86.x64 -property installationPath) + if (-not $hasVc) { $packages += 'visualstudio2022-workload-vctools' } + if (-not $packages) { exit 0 } + if (-not (Get-Command choco -ErrorAction SilentlyContinue)) { + Set-ExecutionPolicy Bypass -Scope Process -Force + [Net.ServicePointManager]::SecurityProtocol = [Net.SecurityProtocolType]::Tls12 + Invoke-Expression ((New-Object Net.WebClient).DownloadString('https://community.chocolatey.org/install.ps1')) + $env:Path = "$env:ProgramData\chocolatey\bin;$env:Path" + } + choco install -y --no-progress @packages + if ($LASTEXITCODE -ne 0) { throw "choco install failed ($LASTEXITCODE)" } + # Later steps start with the job's original PATH; publish the new one. + $machine = [Environment]::GetEnvironmentVariable('Path', 'Machine') + $user = [Environment]::GetEnvironmentVariable('Path', 'User') + "$machine;$user" -split ';' | Where-Object { $_ } | ForEach-Object { Add-Content -Path $env:GITHUB_PATH -Value $_ } + - name: Install the Vulkan SDK if: matrix.backend == 'vulkan' shell: pwsh From 709ac33b7dcb8961bcda847ea15603a6bbde4713 Mon Sep 17 00:00:00 2001 From: Prabhsimran Singh Date: Fri, 2 Oct 2026 11:09:08 +0000 Subject: [PATCH 13/15] fix(cuda): validate cuBLAS shim GEMM sizes | docs: note AI review tools --- .coderabbit.yaml | 5 ++- CONTRIBUTING.md | 9 +++++ kernels/cublas_shim.cu | 52 ++++++++++++++++++++------ tests/cpp/test_cublas_shim_numerics.cu | 6 +++ 4 files changed, 58 insertions(+), 14 deletions(-) diff --git a/.coderabbit.yaml b/.coderabbit.yaml index 7f8de18..f9e0c0f 100644 --- a/.coderabbit.yaml +++ b/.coderabbit.yaml @@ -5,8 +5,9 @@ reviews: profile: chill - # Keep the summary in the walkthrough comment instead of editing the PR description. - high_level_summary: false + # Generate the summary, and put it in the walkthrough comment instead of the PR + # description. (high_level_summary: false would disable it entirely.) + high_level_summary: true high_level_summary_in_walkthrough: true auto_review: enabled: true diff --git a/CONTRIBUTING.md b/CONTRIBUTING.md index fa83270..918d0be 100644 --- a/CONTRIBUTING.md +++ b/CONTRIBUTING.md @@ -18,6 +18,15 @@ You may use AI tools as assistants for code, but the contribution must be yours. Pull requests that don't follow these guidelines may be closed without review. +### Automated review + +The project uses AI tools too: CodeRabbit and Greptile review every pull +request, configured in [`.coderabbit.yaml`](.coderabbit.yaml) and +[`greptile.json`](greptile.json). Their comments are suggestions; a maintainer +decides what must change before merging. Fix the findings that are valid, and +reply briefly to the ones that are not so reviewers can see why. The rules above +apply to those replies as well. + ## Development checks Follow the [source-build guide](docs/build.md) for prerequisites and submodules. diff --git a/kernels/cublas_shim.cu b/kernels/cublas_shim.cu index ada62e0..ad82b98 100644 --- a/kernels/cublas_shim.cu +++ b/kernels/cublas_shim.cu @@ -36,7 +36,7 @@ enum { OP_N = 0, OP_T = 1 }; enum { R_32F = 0, R_16F = 2, R_16BF = 14 }; enum { COMPUTE_16F = 64, COMPUTE_32F = 68 }; -enum { STATUS_SUCCESS = 0, STATUS_EXECUTION_FAILED = 13 }; +enum { STATUS_SUCCESS = 0, STATUS_INVALID_VALUE = 7, STATUS_EXECUTION_FAILED = 13 }; // cudaDataType is already defined by library_types.h (via cuda_runtime.h). typedef void* cublasHandle_t; @@ -1238,9 +1238,6 @@ launch( int m, int n, int k, int opA, int opB, const void* A, int lda, long long sa, int ta, const void* B, int ldb, long long sb, int tb, void* C, int ldc, long long sc, int tc, float alpha, float beta, int batch, cudaStream_t s, ShimHandle* sh) { - if (m <= 0 || n <= 0) { - return; - } int bb = batch > 0 ? batch : 1; if (use_hgemv_tn(n, opA, opB, ta, tb, tc)) { if (tc == R_16F && alpha == 1.0f && beta == 0.0f) { @@ -1380,8 +1377,8 @@ launch_strided( int m, int n, int k, int opA, int opB, const void* A, int lda, long long sa, int ta, const void* B, int ldb, long long sb, int tb, void* C, int ldc, long long sc, int tc, float alpha, float beta, int batch, cudaStream_t s, ShimHandle* sh) { - for (int b0 = 0; b0 < batch; b0 += kMaxGridZ) { - const int count = std::min(batch - b0, kMaxGridZ); + for (int b0 = 0, count = 0; b0 < batch; b0 += count) { + count = std::min(batch - b0, kMaxGridZ); launch( m, n, k, opA, opB, offset(A, b0 * sa, ta), lda, sa, ta, offset(B, b0 * sb, tb), ldb, sb, tb, (void*)offset(C, b0 * sc, tc), ldc, sc, tc, alpha, beta, count, s, sh); @@ -1393,11 +1390,8 @@ launch_ptrs( int m, int n, int k, int opA, int opB, const void* const* A, int lda, int ta, const void* const* B, int ldb, int tb, void* const* C, int ldc, int tc, float alpha, float beta, int batch, cudaStream_t s) { - if (m <= 0 || n <= 0) { - return; - } - for (int b0 = 0; b0 < batch; b0 += kMaxGridZ) { - const int count = std::min(batch - b0, kMaxGridZ); + for (int b0 = 0, count = 0; b0 < batch; b0 += count) { + count = std::min(batch - b0, kMaxGridZ); dim3 blk(16, 16, 1), grd((m + 15) / 16, (n + 15) / 16, count); k_ptrs<<>>( m, n, k, opA, opB, A + b0, lda, ta, B + b0, ldb, tb, C + b0, ldc, tc, alpha, beta, @@ -1405,7 +1399,17 @@ launch_ptrs( } } -// Report a failed kernel launch instead of claiming success. +// cuBLAS rejects negative sizes and completes empty problems without work. +// Returns false, with the status to report, when there is nothing to launch. +inline bool +gemm_has_work(int m, int n, int k, int batch, cublasStatus_t* status) { + *status = m < 0 || n < 0 || k < 0 || batch < 0 ? STATUS_INVALID_VALUE : STATUS_SUCCESS; + return *status == STATUS_SUCCESS && m > 0 && n > 0 && batch > 0; +} + +// Report a failed kernel launch instead of claiming success. The shim links the +// CUDA runtime statically, so this error state is its own: it never holds or +// clears an error pending in the caller's runtime. inline cublasStatus_t launch_status() { return cudaGetLastError() == cudaSuccess ? STATUS_SUCCESS : STATUS_EXECUTION_FAILED; @@ -1479,6 +1483,10 @@ cublasGemmEx( const void* alpha, const void* A, cudaDataType ta, int lda, const void* B, cudaDataType tb, int ldb, const void* beta, void* C, cudaDataType tc, int ldc, cublasComputeType_t ct, cublasGemmAlgo_t) { + cublasStatus_t status; + if (!gemm_has_work(m, n, k, 1, &status)) { + return status; + } ShimHandle* sh = (ShimHandle*)h; const cudaStream_t stream = stream_for_handle(sh); launch( @@ -1492,6 +1500,10 @@ cublasGemmStridedBatchedEx( const void* alpha, const void* A, cudaDataType ta, int lda, long long sa, const void* B, cudaDataType tb, int ldb, long long sb, const void* beta, void* C, cudaDataType tc, int ldc, long long sc, int batch, cublasComputeType_t ct, cublasGemmAlgo_t) { + cublasStatus_t status; + if (!gemm_has_work(m, n, k, batch, &status)) { + return status; + } ShimHandle* sh = (ShimHandle*)h; const cudaStream_t stream = stream_for_handle(sh); launch_strided( @@ -1505,6 +1517,10 @@ cublasGemmBatchedEx( const void* alpha, const void* const Aarray[], cudaDataType ta, int lda, const void* const Barray[], cudaDataType tb, int ldb, const void* beta, void* const Carray[], cudaDataType tc, int ldc, int batch, cublasComputeType_t ct, cublasGemmAlgo_t) { + cublasStatus_t status; + if (!gemm_has_work(m, n, k, batch, &status)) { + return status; + } ShimHandle* sh = (ShimHandle*)h; const cudaStream_t stream = stream_for_handle(sh); launch_ptrs( @@ -1517,6 +1533,10 @@ cublasSgemm_v2( cublasHandle_t h, cublasOperation_t opA, cublasOperation_t opB, int m, int n, int k, const float* alpha, const float* A, int lda, const float* B, int ldb, const float* beta, float* C, int ldc) { + cublasStatus_t status; + if (!gemm_has_work(m, n, k, 1, &status)) { + return status; + } ShimHandle* sh = (ShimHandle*)h; const cudaStream_t stream = stream_for_handle(sh); launch( @@ -1529,6 +1549,10 @@ cublasSgemmStridedBatched( cublasHandle_t h, cublasOperation_t opA, cublasOperation_t opB, int m, int n, int k, const float* alpha, const float* A, int lda, long long sa, const float* B, int ldb, long long sb, const float* beta, float* C, int ldc, long long sc, int batch) { + cublasStatus_t status; + if (!gemm_has_work(m, n, k, batch, &status)) { + return status; + } ShimHandle* sh = (ShimHandle*)h; const cudaStream_t stream = stream_for_handle(sh); launch_strided( @@ -1541,6 +1565,10 @@ cublasSgemmBatched( cublasHandle_t h, cublasOperation_t opA, cublasOperation_t opB, int m, int n, int k, const float* alpha, const float* const Aarray[], int lda, const float* const Barray[], int ldb, const float* beta, float* const Carray[], int ldc, int batch) { + cublasStatus_t status; + if (!gemm_has_work(m, n, k, batch, &status)) { + return status; + } ShimHandle* sh = (ShimHandle*)h; const cudaStream_t stream = stream_for_handle(sh); launch_ptrs( diff --git a/tests/cpp/test_cublas_shim_numerics.cu b/tests/cpp/test_cublas_shim_numerics.cu index 20eef1f..7fb2a16 100644 --- a/tests/cpp/test_cublas_shim_numerics.cu +++ b/tests/cpp/test_cublas_shim_numerics.cu @@ -353,6 +353,12 @@ run_large_batch_case(cublasHandle_t handle) { "cublasSgemmStridedBatched(m=0)") && check_output("zero-sized no-op"); } + if (ok && cublasSgemmStridedBatched( + handle, CUBLAS_OP_N, CUBLAS_OP_N, -1, dim, dim, &alpha, device_a, dim, 0, + device_b, dim, 0, &beta, device_c, dim, 0, 1) != CUBLAS_STATUS_INVALID_VALUE) { + std::fprintf(stderr, "FAIL: a negative dimension was not rejected\n"); + ok = false; + } cudaFree(device); cudaFree(device_ptrs); From 999562f826cda7e2b042541f4f6e580031858d5c Mon Sep 17 00:00:00 2001 From: Prabhsimran Singh Date: Fri, 2 Oct 2026 17:27:52 +0530 Subject: [PATCH 14/15] fix(windows): pass -CMakeArgs through to CMake --- scripts/windows/build.ps1 | 52 +++++++++++++++++++-------------------- 1 file changed, 26 insertions(+), 26 deletions(-) diff --git a/scripts/windows/build.ps1 b/scripts/windows/build.ps1 index 053b75b..012f2bb 100644 --- a/scripts/windows/build.ps1 +++ b/scripts/windows/build.ps1 @@ -381,7 +381,7 @@ function ConvertTo-CMakeBool([bool]$Value) { return 'OFF' } -$cmakeArgs = @( +$configureArgs = @( '-S', $RepoRoot, '-B', $BuildDir, '-G', 'Ninja', "-DCMAKE_BUILD_TYPE=$Config", "-DNEMO_SPEECH_BUILD_ASR=$(ConvertTo-CMakeBool $BuildAsr)", "-DNEMO_SPEECH_BUILD_DIAR=$(ConvertTo-CMakeBool $BuildDiar)", @@ -400,56 +400,56 @@ $cmakeArgs = @( "-DNEMO_SPEECH_BUILD_TOOLS=$(ConvertTo-CMakeBool $BuildTools)" ) if ($VcpkgFeatures.Count -gt 0) { - $cmakeArgs += "-DCMAKE_TOOLCHAIN_FILE=$toolchain" - $cmakeArgs += "-DVCPKG_TARGET_TRIPLET=$VcpkgTriplet" - $cmakeArgs += "-DVCPKG_MANIFEST_FEATURES=$($VcpkgFeatures -join ';')" - $cmakeArgs += "-DVCPKG_INSTALLED_DIR=$(Join-Path $BuildDir 'vcpkg_installed')" + $configureArgs += "-DCMAKE_TOOLCHAIN_FILE=$toolchain" + $configureArgs += "-DVCPKG_TARGET_TRIPLET=$VcpkgTriplet" + $configureArgs += "-DVCPKG_MANIFEST_FEATURES=$($VcpkgFeatures -join ';')" + $configureArgs += "-DVCPKG_INSTALLED_DIR=$(Join-Path $BuildDir 'vcpkg_installed')" } if ($Compiler -eq 'clang-cl') { - $cmakeArgs += '-DCMAKE_C_COMPILER=clang-cl' - $cmakeArgs += '-DCMAKE_CXX_COMPILER=clang-cl' + $configureArgs += '-DCMAKE_C_COMPILER=clang-cl' + $configureArgs += '-DCMAKE_CXX_COMPILER=clang-cl' if ($CrossCompiling) { $llvmTarget = if ($TargetArch -eq 'arm64') { 'arm64-pc-windows-msvc' } else { 'x86_64-pc-windows-msvc' } - $cmakeArgs += "-DCMAKE_C_COMPILER_TARGET=$llvmTarget" - $cmakeArgs += "-DCMAKE_CXX_COMPILER_TARGET=$llvmTarget" - $cmakeArgs += '-DCMAKE_SYSTEM_NAME=Windows' - $cmakeArgs += "-DCMAKE_SYSTEM_PROCESSOR=$TargetArch" + $configureArgs += "-DCMAKE_C_COMPILER_TARGET=$llvmTarget" + $configureArgs += "-DCMAKE_CXX_COMPILER_TARGET=$llvmTarget" + $configureArgs += '-DCMAKE_SYSTEM_NAME=Windows' + $configureArgs += "-DCMAKE_SYSTEM_PROCESSOR=$TargetArch" } if ($Backend -eq 'cuda') { # nvcc only supports cl.exe as its host compiler on Windows; pin it # explicitly so it never inherits clang-cl. - $cmakeArgs += '-DCMAKE_CUDA_HOST_COMPILER=cl' + $configureArgs += '-DCMAKE_CUDA_HOST_COMPILER=cl' } } if ($TargetArch -eq 'arm64') { # Avoid an additional OpenMP runtime DLL in ARM64 packages. - $cmakeArgs += '-DGGML_OPENMP=OFF' + $configureArgs += '-DGGML_OPENMP=OFF' } switch ($Backend) { 'cuda' { - $cmakeArgs += '-DGGML_CUDA=ON' - $cmakeArgs += '-DGGML_VULKAN=OFF' - $cmakeArgs += "-DNEMO_SPEECH_CUBLAS_SHIM=$(ConvertTo-CMakeBool $CublasShim.IsPresent)" - $cmakeArgs += "-DCMAKE_CUDA_ARCHITECTURES=$CudaArch" + $configureArgs += '-DGGML_CUDA=ON' + $configureArgs += '-DGGML_VULKAN=OFF' + $configureArgs += "-DNEMO_SPEECH_CUBLAS_SHIM=$(ConvertTo-CMakeBool $CublasShim.IsPresent)" + $configureArgs += "-DCMAKE_CUDA_ARCHITECTURES=$CudaArch" } 'vulkan' { - $cmakeArgs += '-DGGML_CUDA=OFF' - $cmakeArgs += '-DGGML_VULKAN=ON' - $cmakeArgs += '-DNEMO_SPEECH_GGML_PATCHED=OFF' + $configureArgs += '-DGGML_CUDA=OFF' + $configureArgs += '-DGGML_VULKAN=ON' + $configureArgs += '-DNEMO_SPEECH_GGML_PATCHED=OFF' # ggml-vulkan hard-requires the SPIRV-Headers CMake package; the Vulkan # SDK ships its config, but not on CMake's default search path. $spirvDir = Join-Path $env:VULKAN_SDK 'Lib\cmake\SPIRV-Headers' - if (Test-Path $spirvDir) { $cmakeArgs += "-DSPIRV-Headers_DIR=$spirvDir" } + if (Test-Path $spirvDir) { $configureArgs += "-DSPIRV-Headers_DIR=$spirvDir" } } 'cpu' { - $cmakeArgs += '-DGGML_CUDA=OFF' - $cmakeArgs += '-DGGML_VULKAN=OFF' + $configureArgs += '-DGGML_CUDA=OFF' + $configureArgs += '-DGGML_VULKAN=OFF' } } -$cmakeArgs += $CMakeArgs -Write-Host "==> cmake $($cmakeArgs -join ' ')" -& cmake @cmakeArgs +$configureArgs += $CMakeArgs +Write-Host "==> cmake $($configureArgs -join ' ')" +& cmake @configureArgs if ($LASTEXITCODE -ne 0) { throw "CMake configure failed ($LASTEXITCODE)" } Write-Host "==> building" From 47aabd145d46e2b6a63d0ef9b9eed419bf0c8e7b Mon Sep 17 00:00:00 2001 From: Prabhsimran Singh Date: Fri, 2 Oct 2026 18:32:07 +0530 Subject: [PATCH 15/15] ci(release): drop the unreliable x86 scan --- .github/workflows/release.yml | 10 ++- docs/development/releasing.md | 8 +-- scripts/release/check_release.py | 119 +------------------------------ scripts/release/package-linux.sh | 9 +-- 4 files changed, 11 insertions(+), 135 deletions(-) diff --git a/.github/workflows/release.yml b/.github/workflows/release.yml index ff8c360..1dc9862 100644 --- a/.github/workflows/release.yml +++ b/.github/workflows/release.yml @@ -10,8 +10,8 @@ name: Release # branches pushed by copy-pr-bot, as in gpu.yml) # # x86_64 binaries target x86-64-v3 (AVX2): the archives are built with -# GGML_NATIVE=OFF, scanned for AVX-512/AMX instructions, and the Linux CPU -# archive runs under an emulated Haswell CPU before anything is published. +# GGML_NATIVE=OFF, and the Linux CPU archive runs under an emulated Haswell CPU +# before anything is published. # # Linux and macOS archives include text normalization (ITN/TN). The grammars are # not built here: they are taken from the release pinned in the grammars job, @@ -552,12 +552,10 @@ jobs: exit 1 fi - - name: Check checksums, layout, and the x86_64 instruction baseline + - name: Check checksums and layout env: VERSION: ${{ needs.prepare.outputs.version }} - run: | - sudo apt-get update && sudo apt-get install -y --no-install-recommends llvm - python3 scripts/release/check_release.py archives dist --version "$VERSION" + run: python3 scripts/release/check_release.py archives dist --version "$VERSION" baseline: name: x86_64 baseline (emulated Haswell CPU) diff --git a/docs/development/releasing.md b/docs/development/releasing.md index 4b04961..e740382 100644 --- a/docs/development/releasing.md +++ b/docs/development/releasing.md @@ -50,11 +50,9 @@ run the workflow as a dry run once copy-pr-bot mirrors them to a ## What every run checks -- **CPU baseline:** x86_64 archives are built with `GGML_NATIVE=OFF` and must - not contain instructions beyond x86-64-v3 (AVX2, FMA, F16C, BMI2); - `scripts/release/check_release.py` disassembles every x86_64 binary. The - Linux x86_64 CPU archive also transcribes audio under QEMU's Haswell model, - which has no AVX-512. +- **CPU baseline:** x86_64 archives target x86-64-v3 (AVX2, FMA, F16C, BMI2) + and are built with `GGML_NATIVE=OFF`. The Linux x86_64 CPU archive + transcribes audio under QEMU's Haswell model, which has no AVX-512. - **Self-contained packages:** Linux archives need glibc 2.31 or newer and find their bundled libraries through `DT_RPATH`, which `LD_LIBRARY_PATH` cannot override. Vulkan archives use the host's `libstdc++` and `libgcc_s`, which diff --git a/scripts/release/check_release.py b/scripts/release/check_release.py index 50ac2ef..734be46 100755 --- a/scripts/release/check_release.py +++ b/scripts/release/check_release.py @@ -3,13 +3,9 @@ # SPDX-License-Identifier: Apache-2.0 """Check release archives before they are published. - check_release.py isa PATH... Reject AVX-512/AMX code in x86_64 binaries under PATH. check_release.py archives DIR --version VERSION Check every archive in DIR: checksum, name, layout, - required files, and the x86_64 instruction baseline. - -x86_64 archives target x86-64-v3 (AVX2, FMA, F16C, BMI2). ELF, PE, and Mach-O files are -disassembled with llvm-objdump when available, otherwise with objdump. + and required files. """ from __future__ import annotations @@ -17,9 +13,6 @@ import hashlib import pathlib import re -import shutil -import struct -import subprocess import sys import tarfile import tempfile @@ -31,107 +24,6 @@ r"\.(?Ptar\.gz|zip)$" ) -# x86-64-v3 has no EVEX (AVX-512) instructions, and in 64-bit code a leading 0x62 opcode byte -# after legacy prefixes is always EVEX, so the encoding catches every AVX-512 instruction -# whatever its registers. The mnemonics below add the VEX-encoded extensions beyond v3 -# (AVX-VNNI, AMX) and AVX-512 forms reported by name. Both disassemblers put a tab before the -# mnemonic; GNU objdump continues a long instruction's bytes on lines without one, which are -# not instructions. -INSTRUCTION = re.compile(r"^\s*[0-9a-f]+:\s+((?:[0-9a-f]{2} )+)\s*\t(\S.*)$") -LEGACY_PREFIXES = {"26", "2e", "36", "3e", "64", "65", "66", "67", "f0", "f2", "f3"} - -# AT&T syntax, as printed by objdump and llvm-objdump. -BEYOND_X86_64_V3 = re.compile( - r"%zmm\d+|%[xy]mm(?:1[6-9]|2\d|3[01])\b|%k[0-7]\b|\{%k[0-7]\}" - r"|\b(?:vpternlog[dq]|vmovdqu(?:8|16|32|64)|vmovdqa(?:32|64)|vperm[it]2\w*|vpcompress\w*" - r"|vpexpand\w*|vfixupimm\w*|vgetexp\w*|vgetmant\w*|vrndscale\w*|vreduce\w*|vscalef\w*" - r"|vpmovm2\w*|vpbroadcastm\w*|vpdpbusds?|vpdpwssds?|vdpbf16ps|vcvtne2ps2bf16" - r"|kmov[bwdq]|kand\w*|kor\w*|kxor\w*|knot\w*|ktest\w*|kshift\w*|kunpck\w*" - r"|tdp\w+|tileload\w*|tilestored|tilezero|ldtilecfg|sttilecfg|tilerelease)\b" -) - - -def binary_arch(path: pathlib.Path) -> str | None: - """Return 'x86_64', 'aarch64', 'other', or None for files that are not executables.""" - try: - with path.open("rb") as f: - head = f.read(64) - if head[:4] == b"\x7fELF": - machine = struct.unpack_from("= 64: - f.seek(struct.unpack_from(" str: - for tool in ("llvm-objdump", "objdump"): - if shutil.which(tool): - return tool - raise SystemExit("error: llvm-objdump or objdump is required") - - -def beyond_v3(disassembly: str) -> list[str]: - """Return the instructions beyond x86-64-v3 in llvm-objdump or objdump output.""" - lines = [m for m in map(INSTRUCTION.match, disassembly.splitlines()) if m] - # Bytes the disassembler cannot decode are data in a code section, such as a jump table: - # llvm-objdump prints , GNU objdump (bad). Data next to them can decode as a - # valid-looking instruction, so ignore hits adjacent to undecodable bytes; real AVX-512 - # code never appears only there. - data = [m.group(2).startswith(("(bad)", "")) for m in lines] - hits = [] - for i, m in enumerate(lines): - if data[i] or (i > 0 and data[i - 1]) or (i + 1 < len(lines) and data[i + 1]): - continue - opcode = next((b for b in m.group(1).split() if b not in LEGACY_PREFIXES), "") - if opcode == "62" or BEYOND_X86_64_V3.search(m.group(2)): - hits.append(m.group(0).strip()) - return hits - - -def scan_isa(paths: list[pathlib.Path]) -> list[str]: - tool = disassembler() - failures = [] - for root in paths: - files = ( - [root] - if root.is_file() - else sorted(p for p in root.rglob("*") if p.is_file() and not p.is_symlink()) - ) - for path in files: - arch = binary_arch(path) - if arch == "other": - failures.append(f"{path}: unsupported binary format or architecture") - if arch != "x86_64": - continue - result = subprocess.run( - [tool, "-d", str(path)], - capture_output=True, - text=True, - errors="replace", - ) - if result.returncode != 0 or "file format not recognized" in result.stderr: - failures.append(f"{path}: {tool} could not disassemble it") - continue - hits = beyond_v3(result.stdout) - if hits: - failures.append( - f"{path}: {len(hits)} instructions beyond x86-64-v3, first: {hits[0]}" - ) - return failures - def extract(archive: pathlib.Path, dest: pathlib.Path) -> None: if archive.name.endswith(".zip"): @@ -185,8 +77,6 @@ def check_archives(directory: pathlib.Path, version: str) -> list[str]: ): if not (root / required).is_file(): failures.append(f"{archive.name}: missing {required}") - if match["arch"] == "x86_64": - failures += [f"{archive.name}: {f}" for f in scan_isa([root])] print(f"checked {archive.name}", flush=True) return failures @@ -196,17 +86,12 @@ def main() -> None: description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter ) sub = parser.add_subparsers(dest="command", required=True) - isa = sub.add_parser("isa") - isa.add_argument("paths", nargs="+", type=pathlib.Path) archives = sub.add_parser("archives") archives.add_argument("directory", type=pathlib.Path) archives.add_argument("--version", required=True) args = parser.parse_args() - if args.command == "isa": - failures = scan_isa(args.paths) - else: - failures = check_archives(args.directory, args.version) + failures = check_archives(args.directory, args.version) for failure in failures: print(f"error: {failure}", file=sys.stderr) if failures: diff --git a/scripts/release/package-linux.sh b/scripts/release/package-linux.sh index 1bedf9d..00a47f0 100755 --- a/scripts/release/package-linux.sh +++ b/scripts/release/package-linux.sh @@ -29,9 +29,8 @@ The installed project and GCC runtimes are packaged together; the project must be built with text normalization (-DNEMO_SPEECH_WITH_NORM=ON and the static dependencies from scripts/build_itn_deps.sh). CUDA archives also include libcudart. glibc, GPU drivers, and the Vulkan loader remain host -dependencies. x86_64 binaries must not use instructions beyond x86-64-v3 (AVX2), -and project binaries must find bundled libraries through DT_RPATH, which -LD_LIBRARY_PATH cannot override. +dependencies. Project binaries must find bundled libraries through DT_RPATH, +which LD_LIBRARY_PATH cannot override. EOF } @@ -373,10 +372,6 @@ if [[ -s "$runpath_files" ]]; then exit 1 fi -if [[ "$arch" == x86_64 ]]; then - python3 "$ROOT/scripts/release/check_release.py" isa "$package_root" -fi - source_date_epoch="${SOURCE_DATE_EPOCH:-0}" tar --sort=name \ --mtime="@$source_date_epoch" \