diff --git a/.coderabbit.yaml b/.coderabbit.yaml index 970ef7e..f9e0c0f 100644 --- a/.coderabbit.yaml +++ b/.coderabbit.yaml @@ -4,7 +4,11 @@ # yaml-language-server: $schema=https://coderabbit.ai/integrations/schema.v2.json reviews: - profile: assertive + profile: chill + # Generate the summary, and put it in the walkthrough comment instead of the PR + # description. (high_level_summary: false would disable it entirely.) + high_level_summary: true + high_level_summary_in_walkthrough: true auto_review: enabled: true auto_incremental_review: true diff --git a/.github/workflows/pre-commit.yml b/.github/workflows/pre-commit.yml deleted file mode 100644 index e381975..0000000 --- a/.github/workflows/pre-commit.yml +++ /dev/null @@ -1,38 +0,0 @@ -name: Pre-commit Formatting - -on: - pull_request: - types: [opened, synchronize, reopened, ready_for_review] - push: - branches: [main] - workflow_dispatch: - -concurrency: - group: pre-commit-${{ github.workflow }}-${{ github.ref }} - cancel-in-progress: true - -permissions: - contents: read - -jobs: - pre-commit: - name: Formatting and lint checks - if: github.event_name != 'pull_request' || !github.event.pull_request.draft - runs-on: ubuntu-latest - timeout-minutes: 15 - steps: - - name: Checkout repository - uses: actions/checkout@df4cb1c069e1874edd31b4311f1884172cec0e10 # v6.0.3 - with: - persist-credentials: false - - - name: Set up Python - uses: actions/setup-python@ece7cb06caefa5fff74198d8649806c4678c61a1 # v6 - with: - python-version: "3.12" - - - name: Install pre-commit - run: python -m pip install --disable-pip-version-check pre-commit==4.6.2 - - - name: Run pre-commit - run: pre-commit run --all-files --show-diff-on-failure diff --git a/.github/workflows/release.yml b/.github/workflows/release.yml new file mode 100644 index 0000000..1dc9862 --- /dev/null +++ b/.github/workflows/release.yml @@ -0,0 +1,760 @@ +name: Release + +# Builds the portable release archives that scripts/install.sh and +# scripts/install.ps1 download, checks them, and publishes them. +# +# push of a vX.Y.Z tag -> draft release vX.Y.Z; a maintainer reviews and publishes it +# nightly schedule -> prerelease "nightly", rebuilt in place when main has moved +# workflow_dispatch -> dry run (build and check only) or a nightly +# pull request -> dry run when it changes release packaging (pull-request/N +# branches pushed by copy-pr-bot, as in gpu.yml) +# +# x86_64 binaries target x86-64-v3 (AVX2): the archives are built with +# GGML_NATIVE=OFF, and the Linux CPU archive runs under an emulated Haswell CPU +# before anything is published. +# +# Linux and macOS archives include text normalization (ITN/TN). The grammars are +# not built here: they are taken from the release pinned in the grammars job, +# checked against recorded SHA-256 digests, used by the smoke tests, and +# published again with every release. + +on: + push: + tags: ["v*"] + branches: ["pull-request/[0-9]+"] + schedule: + - cron: "0 6 * * *" + workflow_dispatch: + inputs: + channel: + description: "dry-run builds and checks without publishing" + type: choice + options: [dry-run, nightly] + default: dry-run + +concurrency: + group: release-${{ github.ref }} + cancel-in-progress: ${{ startsWith(github.ref, 'refs/heads/pull-request/') }} + +permissions: + contents: read + +env: + # Release jobs only restore the model cache, so they never write entries that + # a later release could read; build.yml and gpu.yml save it from main. + NEMO_SPEECH_MODEL_DIR: ${{ github.workspace }}/.ci-models + JFK_AUDIO: test_files/asr/wav/test/jfk.wav + +jobs: + prepare: + name: Resolve version + if: github.repository == 'NVIDIA/NeMo-Speech.cpp' + runs-on: ubuntu-latest + outputs: + version: ${{ steps.version.outputs.version }} + channel: ${{ steps.version.outputs.channel }} + build: ${{ steps.version.outputs.build }} + version_metadata: ${{ steps.version.outputs.version_metadata }} + steps: + - name: Checkout repository + uses: actions/checkout@df4cb1c069e1874edd31b4311f1884172cec0e10 # v6.0.3 + with: + persist-credentials: false + fetch-depth: 0 + + - name: Resolve version and channel + id: version + env: + EVENT: ${{ github.event_name }} + INPUT_CHANNEL: ${{ inputs.channel }} + RELEASE_PATHS: >- + ^(\.github/workflows/release\.yml|docker/Dockerfile\.release-linux|scripts/build_itn_deps\.sh|scripts/build_sentencepiece_static\.sh|scripts/release/|scripts/windows/(build|package-release)\.ps1|CMakeLists\.txt|src/(asr|common)/CMakeLists\.txt|tests/ci/model_smoke\.py) + run: | + set -euo pipefail + declared="$(sed -n 's/^NEMO_SPEECH_VERSION:[[:space:]]*//p' VERSION)" + build=true + case "$EVENT:$GITHUB_REF_TYPE" in + push:branch) + # Dry run only when the pull request changes release packaging. A + # paths filter cannot tell: copy-pr-bot mirrors commits that + # already exist, so the push lists no changed files. + channel=dry-run + version="$declared" + git fetch --no-tags --quiet origin main + if ! git diff --name-only FETCH_HEAD...HEAD | grep -qE "$RELEASE_PATHS"; then + build=false + echo "::notice::no release packaging changes; skipping" + fi + ;; + push:tag) + if [ "$GITHUB_REF_NAME" != "v$declared" ]; then + echo "::error::tag $GITHUB_REF_NAME does not match VERSION ($declared)" + exit 1 + fi + channel=release + version="$declared" + ;; + schedule:*) channel=nightly; version=nightly ;; + *) + channel="$INPUT_CHANNEL" + if [ "$channel" = nightly ] && [ "$GITHUB_REF" != refs/heads/main ]; then + echo "::error::nightly can only be published from main" + exit 1 + fi + if [ "$channel" = nightly ]; then version=nightly; else version="$declared"; fi + ;; + esac + if [ "$EVENT" = schedule ]; then + # The nightly tag points at the commit it was built from; the + # peeled entry, when present, sorts last. + published="$(git ls-remote origin refs/tags/nightly 'refs/tags/nightly^{}' | + tail -n 1 | cut -f 1)" + if [ "$published" = "$GITHUB_SHA" ]; then + build=false + echo "::notice::nightly is already built from $GITHUB_SHA; skipping" + fi + fi + # Nightly binaries report VERSION plus build metadata, e.g. + # 0.2.0+nightly.a1b2c3d, so they are not mistaken for the release. + metadata="" + if [ "$channel" = nightly ]; then metadata="nightly.${GITHUB_SHA::7}"; fi + echo "channel=$channel" >> "$GITHUB_OUTPUT" + echo "version=$version" >> "$GITHUB_OUTPUT" + echo "build=$build" >> "$GITHUB_OUTPUT" + echo "version_metadata=$metadata" >> "$GITHUB_OUTPUT" + echo "Building $version ($channel): $build" + + grammars: + name: Text-normalization grammars + needs: prepare + if: needs.prepare.outputs.build == 'true' + runs-on: ubuntu-24.04 + steps: + - name: Download the pinned grammars + # To ship new grammars, attach them to a release and update the tag and + # digests here. + env: + GRAMMARS_RELEASE: v0.1.0 + run: | + set -euo pipefail + mkdir -p grammars + for name in itn_configs tn_configs; do + curl -fsSL --retry 3 -o "grammars/$name.tar.bz2" \ + "https://github.com/$GITHUB_REPOSITORY/releases/download/$GRAMMARS_RELEASE/$name.tar.bz2" + done + (cd grammars && sha256sum --check --strict) <<'EOF' + 880c9365d1d52c17450bd930950b0e58eca294421b7be62eb71666fb21b8997f itn_configs.tar.bz2 + 2ca242c6d29f551eba3663d7e508c0d9dad10440e287628234154c6d1a72c7bc tn_configs.tar.bz2 + EOF + + - name: Upload grammars + uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1 + with: + name: grammars + path: grammars/* + if-no-files-found: error + + linux: + name: Linux ${{ matrix.arch }} ${{ matrix.artifact_backend }} + needs: [prepare, grammars] + if: needs.prepare.outputs.build == 'true' + runs-on: ${{ matrix.runner }} + timeout-minutes: 360 + strategy: + fail-fast: false + matrix: + include: + - { arch: x86_64, runner: ubuntu-24.04, platform: linux/amd64, backend: cpu, artifact_backend: cpu, target: artifact } + - { arch: x86_64, runner: ubuntu-24.04, platform: linux/amd64, backend: vulkan, artifact_backend: vulkan, target: artifact } + - arch: x86_64 + runner: ${{ vars.RELEASE_LINUX_X64_CUDA_RUNNER || 'ubuntu-24.04' }} + platform: linux/amd64 + backend: cuda + artifact_backend: cuda + target: cuda-artifact + cuda_image: nvcr.io/nvidia/cuda:12.8.1-devel-ubuntu24.04 + cuda_arch: "75-real;80-virtual;86-real;89-real;120a-real" + - { arch: aarch64, runner: ubuntu-24.04-arm, platform: linux/arm64, backend: cpu, artifact_backend: cpu, target: artifact } + - { arch: aarch64, runner: ubuntu-24.04-arm, platform: linux/arm64, backend: vulkan, artifact_backend: vulkan, target: artifact } + - arch: aarch64 + runner: ${{ vars.RELEASE_LINUX_ARM64_CUDA_RUNNER || 'ubuntu-24.04-arm' }} + platform: linux/arm64 + backend: cuda + artifact_backend: cuda12 + target: cuda-artifact + cuda_image: nvcr.io/nvidia/cuda:12.8.1-devel-ubuntu24.04 + cuda_arch: "87-real;87-virtual" + - arch: aarch64 + runner: ${{ vars.RELEASE_LINUX_ARM64_CUDA_RUNNER || 'ubuntu-24.04-arm' }} + platform: linux/arm64 + backend: cuda + artifact_backend: cuda13 + target: cuda-artifact + cuda_image: nvcr.io/nvidia/cuda:13.0.0-devel-ubuntu24.04 + cuda_arch: "110a-real;121a-real;110-virtual" + steps: + - name: Checkout repository + uses: actions/checkout@df4cb1c069e1874edd31b4311f1884172cec0e10 # v6.0.3 + with: + persist-credentials: false + + - name: Initialize submodules + run: | + git submodule update --init --depth 1 llama.cpp third_party/cpp-httplib + + - name: Free disk space + # Make room for the CUDA image and build tree. + if: matrix.backend == 'cuda' + run: | + sudo rm -rf /usr/share/dotnet /usr/local/lib/android /opt/ghc /opt/hostedtoolcache/CodeQL + docker system prune -af + df -h / + + - name: Build and package + env: + PLATFORM: ${{ matrix.platform }} + TARGET: ${{ matrix.target }} + BACKEND: ${{ matrix.backend }} + ARTIFACT_BACKEND: ${{ matrix.artifact_backend }} + CUDA_IMAGE: ${{ matrix.cuda_image }} + CUDA_ARCH: ${{ matrix.cuda_arch }} + RELEASE_VERSION: ${{ needs.prepare.outputs.version }} + VERSION_METADATA: ${{ needs.prepare.outputs.version_metadata }} + run: | + set -euo pipefail + args=(--build-arg "BACKEND=$BACKEND" --build-arg "ARTIFACT_BACKEND=$ARTIFACT_BACKEND" + --build-arg "RELEASE_VERSION=$RELEASE_VERSION" --build-arg "JOBS=$(nproc)" + --build-arg "VERSION_METADATA=$VERSION_METADATA") + if [ "$BACKEND" = cuda ]; then + args+=(--build-arg "CUDA_IMAGE=$CUDA_IMAGE" --build-arg "CUDA_ARCH=$CUDA_ARCH") + fi + docker buildx build --platform "$PLATFORM" -f docker/Dockerfile.release-linux \ + --target "$TARGET" "${args[@]}" \ + --output type=local,dest=release-artifacts . + test "$(find release-artifacts -maxdepth 1 -name '*.tar.gz' | wc -l)" -eq 1 + (cd release-artifacts && sha256sum --check ./*.sha256) + + - name: Restore cached models + if: matrix.backend == 'cpu' + uses: actions/cache/restore@55cc8345863c7cc4c66a329aec7e433d2d1c52a9 # v6.1.0 + with: + path: ${{ env.NEMO_SPEECH_MODEL_DIR }} + key: models-${{ hashFiles('models/index.json') }} + + - name: Download grammars + if: matrix.backend == 'cpu' + uses: actions/download-artifact@3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c # v8.0.1 + with: + name: grammars + path: ${{ runner.temp }}/grammars + + - name: Smoke test + # CPU archives run the model smoke test; Vulkan archives must start (the + # hosted runner has no GPU); CUDA archives are tested in gpu-smoke. + if: matrix.backend != 'cuda' + env: + BACKEND: ${{ matrix.backend }} + run: | + set -euo pipefail + mkdir -p "$RUNNER_TEMP/extract" + tar -xzf release-artifacts/*.tar.gz -C "$RUNNER_TEMP/extract" + cli="$(echo "$RUNNER_TEMP"/extract/*/bin/nemo-speech)" + if [ "$BACKEND" = cpu ]; then + grammars="$RUNNER_TEMP/grammars" + tar -xjf "$grammars/itn_configs.tar.bz2" -C "$grammars" + tar -xjf "$grammars/tn_configs.tar.bz2" -C "$grammars" + python3 tests/ci/model_smoke.py --binary "$cli" --backend cpu --audio "$JFK_AUDIO" \ + --itn-model-dir "$grammars/itn_configs/en" --tn-model-dir "$grammars/tn_configs" + else + sudo apt-get update && sudo apt-get install -y --no-install-recommends libvulkan1 + "$cli" --version + fi + + - name: Upload archive + uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1 + with: + name: release-linux-${{ matrix.arch }}-${{ matrix.artifact_backend }} + path: release-artifacts/* + if-no-files-found: error + + macos: + name: macOS ${{ matrix.arch }} ${{ matrix.backend }} + needs: [prepare, grammars] + if: needs.prepare.outputs.build == 'true' + runs-on: ${{ matrix.runner }} + timeout-minutes: 180 + strategy: + fail-fast: false + matrix: + include: + - { arch: aarch64, runner: macos-15, backend: cpu, preset: cpu-server, cmake_args: "-DGGML_METAL=OFF" } + - { arch: aarch64, runner: macos-15, backend: metal, preset: metal-server, cmake_args: "" } + - { arch: x86_64, runner: macos-15-intel, backend: cpu, preset: cpu-server, cmake_args: "-DGGML_METAL=OFF" } + env: + MACOSX_DEPLOYMENT_TARGET: "13.3" + steps: + - name: Checkout repository + uses: actions/checkout@df4cb1c069e1874edd31b4311f1884172cec0e10 # v6.0.3 + with: + persist-credentials: false + + - name: Initialize submodules + run: | + git submodule update --init --depth 1 llama.cpp third_party/cpp-httplib + + - name: Install dependencies + run: brew install bash ninja + + - name: Build SentencePiece + run: JOBS="$(sysctl -n hw.ncpu)" scripts/build_sentencepiece_static.sh + + - name: Build the text-normalization dependencies + run: STATIC=1 JOBS="$(sysctl -n hw.ncpu)" scripts/build_itn_deps.sh + + - name: Configure and build + env: + PRESET: ${{ matrix.preset }} + CMAKE_ARGS: ${{ matrix.cmake_args }} + VERSION_METADATA: ${{ needs.prepare.outputs.version_metadata }} + run: | + set -euo pipefail + # configure.sh needs bash 4+; the runner's default bash is 3.2. + # shellcheck disable=SC2086 + "$(brew --prefix)/bin/bash" scripts/configure.sh "$PRESET" \ + -DGGML_NATIVE=OFF \ + -DNEMO_SPEECH_BUILD_GRPC=OFF -DNEMO_SPEECH_WITH_GRPC=OFF \ + -DNEMO_SPEECH_WITH_NORM=ON \ + -DCMAKE_OSX_DEPLOYMENT_TARGET="$MACOSX_DEPLOYMENT_TARGET" \ + -DNEMO_SPEECH_VERSION_METADATA="$VERSION_METADATA" \ + $CMAKE_ARGS + cmake --build --preset "$PRESET" --parallel + cmake --install "build/$PRESET" --prefix "$RUNNER_TEMP/install" + + - name: Package + env: + BACKEND: ${{ matrix.backend }} + ARCH: ${{ matrix.arch }} + RELEASE_VERSION: ${{ needs.prepare.outputs.version }} + run: | + scripts/release/package-macos.sh --install-prefix "$RUNNER_TEMP/install" \ + --backend "$BACKEND" --arch "$ARCH" --version "$RELEASE_VERSION" \ + --min-macos "$MACOSX_DEPLOYMENT_TARGET" --output-dir release-artifacts + + - name: Restore cached models + uses: actions/cache/restore@55cc8345863c7cc4c66a329aec7e433d2d1c52a9 # v6.1.0 + with: + path: ${{ env.NEMO_SPEECH_MODEL_DIR }} + key: models-${{ hashFiles('models/index.json') }} + + - name: Download grammars + uses: actions/download-artifact@3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c # v8.0.1 + with: + name: grammars + path: ${{ runner.temp }}/grammars + + - name: Smoke test + # The hosted VM's paravirtual GPU cannot run ggml Metal inference + # (see build.yml), so every archive runs on the CPU backend. + run: | + set -euo pipefail + mkdir -p "$RUNNER_TEMP/extract" + tar -xzf release-artifacts/*.tar.gz -C "$RUNNER_TEMP/extract" + cli="$(echo "$RUNNER_TEMP"/extract/*/bin/nemo-speech)" + grammars="$RUNNER_TEMP/grammars" + tar -xjf "$grammars/itn_configs.tar.bz2" -C "$grammars" + tar -xjf "$grammars/tn_configs.tar.bz2" -C "$grammars" + python3 tests/ci/model_smoke.py --binary "$cli" --backend cpu --audio "$JFK_AUDIO" \ + --itn-model-dir "$grammars/itn_configs/en" --tn-model-dir "$grammars/tn_configs" + + - name: Upload archive + uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1 + with: + name: release-macos-${{ matrix.arch }}-${{ matrix.backend }} + path: release-artifacts/* + if-no-files-found: error + + windows: + name: Windows x86_64 ${{ matrix.backend }} + needs: prepare + if: needs.prepare.outputs.build == 'true' + runs-on: ${{ matrix.runner }} + timeout-minutes: 360 + strategy: + fail-fast: false + matrix: + include: + - { backend: cpu, runner: windows-2022 } + - { backend: vulkan, runner: windows-2022 } + - backend: cuda + runner: ${{ vars.RELEASE_WINDOWS_CUDA_RUNNER || 'windows-2022' }} + steps: + - name: Checkout repository + uses: actions/checkout@df4cb1c069e1874edd31b4311f1884172cec0e10 # v6.0.3 + with: + persist-credentials: false + + - name: Initialize submodules + shell: bash + run: | + git submodule update --init --depth 1 llama.cpp third_party/cpp-httplib + + - name: Install build prerequisites + # Hosted windows-2022 images have these; self-hosted VMs selected with + # RELEASE_WINDOWS_CUDA_RUNNER may not (see gpu.yml). + shell: pwsh + run: | + $ErrorActionPreference = 'Stop' + $packages = @() + foreach ($tool in @{ cmake = 'cmake'; ninja = 'ninja' }.GetEnumerator()) { + if (-not (Get-Command $tool.Key -ErrorAction SilentlyContinue)) { $packages += $tool.Value } + } + $vswhere = "${env:ProgramFiles(x86)}\Microsoft Visual Studio\Installer\vswhere.exe" + $hasVc = (Test-Path $vswhere) -and + (& $vswhere -latest -products * -requires Microsoft.VisualStudio.Component.VC.Tools.x86.x64 -property installationPath) + if (-not $hasVc) { $packages += 'visualstudio2022-workload-vctools' } + if (-not $packages) { exit 0 } + if (-not (Get-Command choco -ErrorAction SilentlyContinue)) { + Set-ExecutionPolicy Bypass -Scope Process -Force + [Net.ServicePointManager]::SecurityProtocol = [Net.SecurityProtocolType]::Tls12 + Invoke-Expression ((New-Object Net.WebClient).DownloadString('https://community.chocolatey.org/install.ps1')) + $env:Path = "$env:ProgramData\chocolatey\bin;$env:Path" + } + choco install -y --no-progress @packages + if ($LASTEXITCODE -ne 0) { throw "choco install failed ($LASTEXITCODE)" } + # Later steps start with the job's original PATH; publish the new one. + $machine = [Environment]::GetEnvironmentVariable('Path', 'Machine') + $user = [Environment]::GetEnvironmentVariable('Path', 'User') + "$machine;$user" -split ';' | Where-Object { $_ } | ForEach-Object { Add-Content -Path $env:GITHUB_PATH -Value $_ } + + - name: Install the Vulkan SDK + if: matrix.backend == 'vulkan' + shell: pwsh + run: | + $ErrorActionPreference = 'Stop' + choco install -y --no-progress vulkan-sdk + $vulkan = [Environment]::GetEnvironmentVariable('VULKAN_SDK', 'Machine') + if (-not $vulkan) { throw 'VULKAN_SDK was not set by the installer' } + Add-Content -Path $env:GITHUB_ENV -Value "VULKAN_SDK=$vulkan" + Add-Content -Path $env:GITHUB_PATH -Value "$vulkan\Bin" + + - name: Install the CUDA toolkit + if: matrix.backend == 'cuda' + uses: Jimver/cuda-toolkit@b8bf9c6c28f8a92fbb04dcfcaee872e60c57462d # v0.2.36 + with: + cuda: "13.0.0" + method: network + + - name: Restore vcpkg binaries + id: vcpkg-cache + uses: actions/cache/restore@55cc8345863c7cc4c66a329aec7e433d2d1c52a9 # v6.1.0 + with: + path: ~\AppData\Local\vcpkg\archives + key: vcpkg-windows-x64-${{ hashFiles('vcpkg.json', 'scripts/windows/build.ps1') }} + restore-keys: vcpkg-windows-x64- + + - name: Build + id: build + shell: pwsh + env: + BACKEND: ${{ matrix.backend }} + VERSION_METADATA: ${{ needs.prepare.outputs.version_metadata }} + run: | + $arguments = @{ + Backend = $env:BACKEND + Profile = 'server' + BuildDir = "${{ github.workspace }}\build\release-$env:BACKEND" + CMakeArgs = @('-DGGML_NATIVE=OFF', "-DNEMO_SPEECH_VERSION_METADATA=$env:VERSION_METADATA") + } + if ($env:BACKEND -eq 'cuda') { + $arguments.CudaArch = '75-real;80-virtual;86-real;89-real;120a-real' + $arguments.CublasShim = $true + } + scripts\windows\build.ps1 @arguments + + - name: Save vcpkg binaries + if: steps.build.outcome == 'success' && steps.vcpkg-cache.outputs.cache-hit != 'true' + uses: actions/cache/save@55cc8345863c7cc4c66a329aec7e433d2d1c52a9 # v6.1.0 + with: + path: ~\AppData\Local\vcpkg\archives + key: ${{ steps.vcpkg-cache.outputs.cache-primary-key }} + + - name: Package + shell: pwsh + env: + BACKEND: ${{ matrix.backend }} + RELEASE_VERSION: ${{ needs.prepare.outputs.version }} + run: | + scripts\windows\package-release.ps1 -BuildDir "build\release-$env:BACKEND" ` + -Backend $env:BACKEND -Version $env:RELEASE_VERSION -OutputDir release-artifacts + + - name: Restore cached models + if: matrix.backend == 'cpu' + uses: actions/cache/restore@55cc8345863c7cc4c66a329aec7e433d2d1c52a9 # v6.1.0 + with: + path: ${{ env.NEMO_SPEECH_MODEL_DIR }} + key: models-${{ hashFiles('models/index.json') }} + + - name: Smoke test + # GPU archives need a GPU driver at load time; the CUDA archive is tested in gpu-smoke. + if: matrix.backend == 'cpu' + shell: pwsh + run: | + $ErrorActionPreference = 'Stop' + $extract = Join-Path $env:RUNNER_TEMP 'extract' + Expand-Archive -Path (Get-Item release-artifacts\*.zip).FullName -DestinationPath $extract + $cli = (Get-Item "$extract\*\bin\nemo-speech.exe").FullName + python tests\ci\model_smoke.py --binary $cli --backend cpu --audio $env:JFK_AUDIO + if ($LASTEXITCODE -ne 0) { throw "model smoke test failed ($LASTEXITCODE)" } + + - name: Upload archive + uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1 + with: + name: release-windows-x86_64-${{ matrix.backend }} + path: release-artifacts/* + if-no-files-found: error + + verify: + name: Verify archives + needs: [prepare, linux, macos, windows] + runs-on: ubuntu-24.04 + steps: + - name: Checkout repository + uses: actions/checkout@df4cb1c069e1874edd31b4311f1884172cec0e10 # v6.0.3 + with: + persist-credentials: false + + - name: Download archives + uses: actions/download-artifact@3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c # v8.0.1 + with: + pattern: release-* + path: dist + merge-multiple: true + + - name: Check the archive set + env: + VERSION: ${{ needs.prepare.outputs.version }} + run: | + set -euo pipefail + expected="" + for name in linux-x86_64-cpu linux-x86_64-vulkan linux-x86_64-cuda \ + linux-aarch64-cpu linux-aarch64-vulkan linux-aarch64-cuda12 linux-aarch64-cuda13 \ + macos-aarch64-cpu macos-aarch64-metal macos-x86_64-cpu; do + expected="$expected nemo-speech-$VERSION-$name.tar.gz" + done + for name in windows-x86_64-cpu windows-x86_64-vulkan windows-x86_64-cuda; do + expected="$expected nemo-speech-$VERSION-$name.zip" + done + actual="$(cd dist && ls -- *.tar.gz *.zip | sort | tr '\n' ' ')" + wanted="$(printf '%s\n' $expected | sort | tr '\n' ' ')" + if [ "$actual" != "$wanted" ]; then + echo "::error::archive set differs from the expected 13" + diff <(printf '%s\n' $wanted) <(printf '%s\n' $actual) || true + exit 1 + fi + + - name: Check checksums and layout + env: + VERSION: ${{ needs.prepare.outputs.version }} + run: python3 scripts/release/check_release.py archives dist --version "$VERSION" + + baseline: + name: x86_64 baseline (emulated Haswell CPU) + needs: [prepare, linux] + runs-on: ubuntu-24.04 + timeout-minutes: 120 + steps: + - name: Checkout repository + uses: actions/checkout@df4cb1c069e1874edd31b4311f1884172cec0e10 # v6.0.3 + with: + persist-credentials: false + + - name: Download the Linux x86_64 CPU archive + uses: actions/download-artifact@3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c # v8.0.1 + with: + name: release-linux-x86_64-cpu + path: dist + + - name: Restore cached models + uses: actions/cache/restore@55cc8345863c7cc4c66a329aec7e433d2d1c52a9 # v6.1.0 + with: + path: ${{ env.NEMO_SPEECH_MODEL_DIR }} + key: models-${{ hashFiles('models/index.json') }} + + - name: Transcribe on an AVX2-only CPU + # Hosted runners usually have AVX-512, so a native run would not catch + # AVX-512 code. QEMU's Haswell model has AVX2/FMA/F16C and no AVX-512, + # so such an instruction raises SIGILL here. + run: | + set -euo pipefail + sudo apt-get update && sudo apt-get install -y --no-install-recommends qemu-user + mkdir -p "$RUNNER_TEMP/extract" + tar -xzf dist/*.tar.gz -C "$RUNNER_TEMP/extract" + cli="$(echo "$RUNNER_TEMP"/extract/*/bin/nemo-speech)" + "$cli" pull nvidia/nemotron-3.5-asr-streaming-0.6b + text="$(qemu-x86_64 -cpu Haswell "$cli" transcribe "$JFK_AUDIO" --backend cpu)" + echo "$text" + echo "$text" | grep -qi "fellow americans" + + gpu-smoke-linux: + name: Linux x86_64 CUDA archive (L4) + needs: [prepare, grammars, linux] + runs-on: linux-amd64-gpu-l4-latest-1 + timeout-minutes: 60 + container: + image: nvcr.io/nvidia/cuda@sha256:1e8ac7a54c184a1af8ef2167f28fa98281892a835c981ebcddb1fad04bdd452d # 13.0.0-devel-ubuntu24.04 + options: -u root --security-opt seccomp=unconfined --shm-size 16g + env: + NVIDIA_VISIBLE_DEVICES: ${{ env.NVIDIA_VISIBLE_DEVICES }} + steps: + - name: Install prerequisites + # The CLI downloads models with curl. + run: apt-get update && apt-get install -y --no-install-recommends git curl python3 ca-certificates bzip2 + + - name: Checkout repository + uses: actions/checkout@df4cb1c069e1874edd31b4311f1884172cec0e10 # v6.0.3 + with: + persist-credentials: false + + - name: Download the CUDA archive + uses: actions/download-artifact@3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c # v8.0.1 + with: + name: release-linux-x86_64-cuda + path: dist + + - name: Restore cached models + uses: actions/cache/restore@55cc8345863c7cc4c66a329aec7e433d2d1c52a9 # v6.1.0 + with: + path: ${{ env.NEMO_SPEECH_MODEL_DIR }} + key: models-${{ hashFiles('models/index.json') }} + + - name: Download grammars + uses: actions/download-artifact@3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c # v8.0.1 + with: + name: grammars + path: /tmp/grammars + + - name: Model smoke test + run: | + set -eu + mkdir -p /tmp/extract + tar -xzf dist/*.tar.gz -C /tmp/extract + tar -xjf /tmp/grammars/itn_configs.tar.bz2 -C /tmp/grammars + tar -xjf /tmp/grammars/tn_configs.tar.bz2 -C /tmp/grammars + python3 tests/ci/model_smoke.py --binary "$(echo /tmp/extract/*/bin/nemo-speech)" \ + --backend cuda --audio "$JFK_AUDIO" \ + --itn-model-dir /tmp/grammars/itn_configs/en --tn-model-dir /tmp/grammars/tn_configs + + gpu-smoke-windows: + name: Windows x86_64 CUDA archive (L4) + needs: [prepare, windows] + runs-on: windows-amd64-gpu-l4-latest-1 + timeout-minutes: 60 + # Non-blocking, like the other jobs on this runner (see gpu.yml), except + # for tagged releases. + continue-on-error: ${{ needs.prepare.outputs.channel != 'release' }} + steps: + - name: Checkout repository + uses: actions/checkout@df4cb1c069e1874edd31b4311f1884172cec0e10 # v6.0.3 + with: + persist-credentials: false + + - name: Download the CUDA archive + uses: actions/download-artifact@3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c # v8.0.1 + with: + name: release-windows-x86_64-cuda + path: dist + + - name: Restore cached models + uses: actions/cache/restore@55cc8345863c7cc4c66a329aec7e433d2d1c52a9 # v6.1.0 + with: + path: ${{ env.NEMO_SPEECH_MODEL_DIR }} + key: models-${{ hashFiles('models/index.json') }} + + - name: Install Python + # The ephemeral VM does not necessarily ship Python (see gpu.yml). + shell: pwsh + run: | + $ErrorActionPreference = 'Stop' + if (Get-Command python -ErrorAction SilentlyContinue) { exit 0 } + if (-not (Get-Command choco -ErrorAction SilentlyContinue)) { + Set-ExecutionPolicy Bypass -Scope Process -Force + [Net.ServicePointManager]::SecurityProtocol = [Net.SecurityProtocolType]::Tls12 + Invoke-Expression ((New-Object Net.WebClient).DownloadString('https://community.chocolatey.org/install.ps1')) + $env:Path = "$env:ProgramData\chocolatey\bin;$env:Path" + } + choco install -y --no-progress python + # Later steps start with the job's original PATH; publish the new one. + $machine = [Environment]::GetEnvironmentVariable('Path', 'Machine') + $user = [Environment]::GetEnvironmentVariable('Path', 'User') + "$machine;$user" -split ';' | Where-Object { $_ } | ForEach-Object { Add-Content -Path $env:GITHUB_PATH -Value $_ } + + - name: Model smoke test + shell: pwsh + run: | + $ErrorActionPreference = 'Stop' + $extract = Join-Path $env:RUNNER_TEMP 'extract' + Expand-Archive -Path (Get-Item dist\*.zip).FullName -DestinationPath $extract + $cli = (Get-Item "$extract\*\bin\nemo-speech.exe").FullName + python tests\ci\model_smoke.py --binary $cli --backend cuda --audio $env:JFK_AUDIO + if ($LASTEXITCODE -ne 0) { throw "model smoke test failed ($LASTEXITCODE)" } + + publish: + name: Publish (${{ needs.prepare.outputs.channel }}) + needs: [prepare, grammars, verify, baseline, gpu-smoke-linux, gpu-smoke-windows] + if: needs.prepare.outputs.channel != 'dry-run' + runs-on: ubuntu-latest + # Requires approval when the "release" environment has required reviewers. + environment: release + permissions: + contents: write + id-token: write + attestations: write + steps: + - name: Download archives + uses: actions/download-artifact@3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c # v8.0.1 + with: + pattern: release-* + path: dist + merge-multiple: true + + - name: Write SHA256SUMS + run: cd dist && sha256sum -- *.tar.gz *.zip > SHA256SUMS + + - name: Attest build provenance + uses: actions/attest-build-provenance@4d101475d8b20a2381f78447822ac1eab6504dd8 # v4.2.2 + with: + subject-path: | + dist/*.tar.gz + dist/*.zip + + - name: Add the text-normalization grammars + uses: actions/download-artifact@3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c # v8.0.1 + with: + name: grammars + path: dist + + - name: Create the draft release + if: needs.prepare.outputs.channel == 'release' + env: + GH_TOKEN: ${{ github.token }} + VERSION: ${{ needs.prepare.outputs.version }} + run: | + gh release create "v$VERSION" --repo "$GITHUB_REPOSITORY" --draft --verify-tag \ + --title "NeMo-Speech.cpp $VERSION" --generate-notes dist/* + + - name: Replace the nightly prerelease + if: needs.prepare.outputs.channel == 'nightly' + env: + GH_TOKEN: ${{ github.token }} + # Upload to a draft first: if the upload fails, the current nightly stays + # in place. Only a complete draft replaces it. + run: | + set -euo pipefail + staging="nightly-staging-$GITHUB_RUN_ID" + if ! gh release create "$staging" --repo "$GITHUB_REPOSITORY" --draft --prerelease \ + --target "$GITHUB_SHA" --title "Nightly" --notes "Built from $GITHUB_SHA." dist/*; then + gh release delete "$staging" --repo "$GITHUB_REPOSITORY" --yes || true + exit 1 + fi + gh release delete nightly --repo "$GITHUB_REPOSITORY" --yes --cleanup-tag || true + gh release edit "$staging" --repo "$GITHUB_REPOSITORY" --tag nightly --draft=false diff --git a/CMakeLists.txt b/CMakeLists.txt index 685b661..3a1397c 100644 --- a/CMakeLists.txt +++ b/CMakeLists.txt @@ -17,6 +17,21 @@ endforeach() if(NOT NEMO_SPEECH_VERSION OR NOT NEMO_SPEECH_PROJECT_VERSION) message(FATAL_ERROR "could not parse NEMO_SPEECH_VERSION from ${CMAKE_CURRENT_SOURCE_DIR}/VERSION") endif() +# Semver build metadata for the reported version, e.g. nightly. for +# nightly release builds, which then report 0.2.0+nightly.. +set(NEMO_SPEECH_VERSION_METADATA "" CACHE STRING + "Build metadata appended to the reported version (dot-separated [0-9A-Za-z-] identifiers)") +if(NEMO_SPEECH_VERSION_METADATA) + if(NOT NEMO_SPEECH_VERSION_METADATA MATCHES "^[0-9A-Za-z-]+(\\.[0-9A-Za-z-]+)*$") + message(FATAL_ERROR + "NEMO_SPEECH_VERSION_METADATA must be dot-separated [0-9A-Za-z-] identifiers") + endif() + if(NEMO_SPEECH_VERSION MATCHES "\\+") + string(APPEND NEMO_SPEECH_VERSION ".${NEMO_SPEECH_VERSION_METADATA}") + else() + string(APPEND NEMO_SPEECH_VERSION "+${NEMO_SPEECH_VERSION_METADATA}") + endif() +endif() project(nemo_speech VERSION ${NEMO_SPEECH_PROJECT_VERSION} LANGUAGES C CXX) add_compile_definitions(NEMO_SPEECH_VERSION_STR="${NEMO_SPEECH_VERSION}") @@ -24,7 +39,12 @@ add_compile_definitions(NEMO_SPEECH_VERSION_STR="${NEMO_SPEECH_VERSION}") include(GNUInstallDirs) include(CMakePackageConfigHelpers) if(WIN32) - # Install the Visual C++ runtime required by Windows packages. + # Install the Visual C++ runtime required by Windows packages, plus the + # OpenMP runtime (vcomp140.dll) that ggml's CPU threading links against + # unless GGML_OPENMP is OFF. + if(NOT DEFINED GGML_OPENMP OR GGML_OPENMP) + set(CMAKE_INSTALL_OPENMP_LIBRARIES TRUE) + endif() include(InstallRequiredSystemLibraries) endif() set(NEMO_SPEECH_LICENSE_INSTALL_DIR diff --git a/CONTRIBUTING.md b/CONTRIBUTING.md index fa83270..918d0be 100644 --- a/CONTRIBUTING.md +++ b/CONTRIBUTING.md @@ -18,6 +18,15 @@ You may use AI tools as assistants for code, but the contribution must be yours. Pull requests that don't follow these guidelines may be closed without review. +### Automated review + +The project uses AI tools too: CodeRabbit and Greptile review every pull +request, configured in [`.coderabbit.yaml`](.coderabbit.yaml) and +[`greptile.json`](greptile.json). Their comments are suggestions; a maintainer +decides what must change before merging. Fix the findings that are valid, and +reply briefly to the ones that are not so reviewers can see why. The rules above +apply to those replies as well. + ## Development checks Follow the [source-build guide](docs/build.md) for prerequisites and submodules. diff --git a/README.md b/README.md index a119077..d343214 100644 --- a/README.md +++ b/README.md @@ -50,9 +50,11 @@ See [BENCHMARK.md](BENCHMARK.md) for the methodology and more results. ## Installation > [!IMPORTANT] -> **For the best performance and the latest features, build natively from source.** A native -> build is compiled for your machine, and release tags can be out of sync with the -> `main` branch. See [Build from source](#build-from-source). +> **For the best performance, build natively from source.** A native build is compiled for +> your machine. Tagged releases are cut periodically and can trail the `main` branch; for +> prebuilt binaries of the latest `main`, pass `--channel nightly` (`-Channel nightly` on +> Windows) to install the nightly prerelease, which is rebuilt daily. See +> [Build from source](#build-from-source). Install the `nemo-speech` CLI for the detected platform and backend: diff --git a/THIRD_PARTY_NOTICES.md b/THIRD_PARTY_NOTICES.md index 282bd20..6d71c60 100644 --- a/THIRD_PARTY_NOTICES.md +++ b/THIRD_PARTY_NOTICES.md @@ -160,7 +160,7 @@ The command-line microphone capture layer compiles miniaudio directly into ### SentencePiece - Source: [`google/sentencepiece`](https://github.com/google/sentencepiece), - revision `17d7580d6407802f85855d2cc9190634e2c95624` + revision `31646a467d2051eb904e0b45de3a73e91fe1c1e3` - Copyright 2018 Google Inc. - License: Apache License 2.0 @@ -169,6 +169,27 @@ SentencePiece runtime and its bundled Abseil, protobuf-lite, and Darts-clone components. Their Apache 2.0 and BSD license texts are installed under `share/licenses/nemo-speech/third_party/sentencepiece/`. +### Text normalization runtime + +`scripts/build_itn_deps.sh` builds these pinned sources for ITN and TN: + +- OpenFST: [`sarane22/openfst`](https://github.com/sarane22/openfst), revision + `fc23b4cf529429284b874a26f28b15c6cc94f404`; Copyright 2005-2024 Google LLC; + Apache License 2.0 +- Sparrowhawk: [`sarane22/sparrowhawk`](https://github.com/sarane22/sparrowhawk), + revision `8b082acc507312077a096be8398584a13832c490`; Copyright 2015 and + onwards Google, Inc.; Apache License 2.0 +- Protocol Buffers: + [`protocolbuffers/protobuf`](https://github.com/protocolbuffers/protobuf) + v21.12; Copyright 2008 Google Inc.; BSD 3-Clause License +- RE2: [`google/re2`](https://github.com/google/re2) 2023-03-01; Copyright (c) + 2009 The RE2 Authors; BSD 3-Clause License + +Linux and macOS release archives statically link all four into +`libnemo_speech_text_normalization`. Builds that use system Protocol Buffers and +RE2 link those instead. The license texts are installed under +`share/licenses/nemo-speech/third_party/`. + ### whisper.cpp sample audio The ASR quick-start fixture at `test_files/asr/wav/test/jfk.wav` is copied from diff --git a/VERSION b/VERSION index 5e37c63..844ba4a 100644 --- a/VERSION +++ b/VERSION @@ -1 +1 @@ -NEMO_SPEECH_VERSION: 0.1.0 +NEMO_SPEECH_VERSION: 0.2.0 diff --git a/docker/Dockerfile.release-linux b/docker/Dockerfile.release-linux new file mode 100644 index 0000000..58e16d0 --- /dev/null +++ b/docker/Dockerfile.release-linux @@ -0,0 +1,249 @@ +# syntax=docker/dockerfile:1.7 +# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-License-Identifier: Apache-2.0 +# Portable x86_64 and aarch64 CPU/Vulkan/CUDA release archives. +# +# Build an installer-compatible archive into release-artifacts/ from the +# repository root (.github/workflows/release.yml runs the same commands): +# docker build --platform=linux/amd64 -f docker/Dockerfile.release-linux \ +# --build-arg BACKEND=vulkan --target artifact \ +# --output type=local,dest=release-artifacts . +# docker build --platform=linux/amd64 -f docker/Dockerfile.release-linux \ +# --target cuda-artifact \ +# --output type=local,dest=release-artifacts . + +ARG CUDA_IMAGE=nvcr.io/nvidia/cuda:12.8.1-devel-ubuntu24.04 + +FROM ubuntu:24.04 AS vulkan-tools + +ENV DEBIAN_FRONTEND=noninteractive + +RUN apt-get update && apt-get install -y --no-install-recommends \ + ca-certificates \ + glslc \ + libvulkan-dev \ + spirv-headers \ + && rm -rf /var/lib/apt/lists/* \ + && set -eux; \ + mkdir -p /opt/vulkan-tools/bin /opt/vulkan-tools/lib; \ + cp -L "$(command -v glslc)" /opt/vulkan-tools/bin/glslc; \ + ldd "$(command -v glslc)" \ + | awk '$1 ~ /^\// { print $1 } $2 == "=>" && $3 ~ /^\// { print $3 }' \ + | sort -u \ + | while read -r dependency; do \ + cp -L "$dependency" "/opt/vulkan-tools/lib/$(basename "$dependency")"; \ + done; \ + loader="$(ldd "$(command -v glslc)" \ + | awk '$1 ~ /ld-linux/ { print $1 } $2 == "=>" && $3 ~ /ld-linux/ { print $3 }' \ + | head -n 1)"; \ + test -n "$loader"; \ + cp -L "$loader" /opt/vulkan-tools/lib/ld-linux.so + +FROM ${CUDA_IMAGE} AS cuda-tools + +RUN set -eux; \ + cuda_home="$(readlink -f /usr/local/cuda)"; \ + mkdir -p /opt/cuda; \ + cp -a "$cuda_home"/. /opt/cuda/; \ + cuda_license="$(find /usr/share/doc -maxdepth 2 -type f \ + -path '*/cuda-cudart-*/copyright' -print | sort | head -n 1)"; \ + test -n "$cuda_license"; \ + cp "$cuda_license" /opt/cuda-cudart-copyright + +FROM ubuntu:20.04 AS portable-base + +ARG TARGETARCH +ARG CMAKE_VERSION=3.31.6 +ARG CMAKE_SHA256_AMD64=5a1133ff103c71eb5120e2cc3de922733e7d8a26a98ae716397e8676adb367bf +ARG CMAKE_SHA256_ARM64=b4cc788d63112b2749b40627e719eb5d3b8ed8f00c36d77189f4019cfe64bc9e +ARG JOBS=4 + +ENV DEBIAN_FRONTEND=noninteractive \ + CMAKE_BUILD_PARALLEL_LEVEL=${JOBS} \ + PATH=/opt/cmake/bin:/usr/local/sbin:/usr/local/bin:/usr/sbin:/usr/bin:/sbin:/bin + +# Use the target architecture's native toolchain on the portable baseline. +RUN test "$(dpkg --print-architecture)" = "${TARGETARCH}" \ + && apt-get update && apt-get install -y --no-install-recommends \ + binutils \ + ca-certificates \ + curl \ + g++-9 \ + gcc-9 \ + git \ + libvulkan-dev \ + ninja-build \ + pkg-config \ + python3 \ + && update-alternatives --install /usr/bin/gcc gcc /usr/bin/gcc-9 100 \ + && update-alternatives --install /usr/bin/g++ g++ /usr/bin/g++-9 100 \ + && update-alternatives --install /usr/bin/cc cc /usr/bin/gcc-9 100 \ + && update-alternatives --install /usr/bin/c++ c++ /usr/bin/g++-9 100 \ + && rm -rf /var/lib/apt/lists/* + +RUN set -eux; \ + case "${TARGETARCH}" in \ + amd64) \ + cmake_arch=x86_64; \ + cmake_sha256="${CMAKE_SHA256_AMD64}" \ + ;; \ + arm64) \ + cmake_arch=aarch64; \ + cmake_sha256="${CMAKE_SHA256_ARM64}" \ + ;; \ + *) echo "unsupported TARGETARCH: ${TARGETARCH}" >&2; exit 2 ;; \ + esac; \ + curl -fsSL \ + "https://github.com/Kitware/CMake/releases/download/v${CMAKE_VERSION}/cmake-${CMAKE_VERSION}-linux-${cmake_arch}.tar.gz" \ + -o /tmp/cmake.tar.gz; \ + echo "${cmake_sha256} /tmp/cmake.tar.gz" | sha256sum -c -; \ + mkdir -p /opt/cmake; \ + tar -xzf /tmp/cmake.tar.gz --strip-components=1 -C /opt/cmake; \ + rm /tmp/cmake.tar.gz; \ + cmake --version; \ + ninja --version + +FROM portable-base AS sentencepiece-dependency + +WORKDIR /work +COPY scripts/build_sentencepiece_static.sh /work/scripts/build_sentencepiece_static.sh + +# SentencePiece is a core ASR dependency. Build the repository's pinned static +# revision once per target architecture and share it across every release +# backend instead of relying on a package from the portable base image. +RUN scripts/build_sentencepiece_static.sh + +FROM portable-base AS itn-dependency + +RUN apt-get update && apt-get install -y --no-install-recommends make perl \ + && rm -rf /var/lib/apt/lists/* + +WORKDIR /work +COPY scripts/build_itn_deps.sh /work/scripts/build_itn_deps.sh +COPY src/common/text_normalization/compat /work/src/common/text_normalization/compat + +# Text normalization (ITN/TN): OpenFST, Sparrowhawk, protobuf, and RE2 as +# static archives, linked privately into libnemo_speech_text_normalization. +RUN STATIC=1 scripts/build_itn_deps.sh + +FROM portable-base AS builder + +ARG BACKEND +ARG ARTIFACT_BACKEND +# Archive version; defaults to VERSION ("nightly" for the nightly channel). +ARG RELEASE_VERSION= + +# Ubuntu 20.04 supplies the release ABI baseline. Current Vulkan headers and +# glslc are build-only inputs copied from the newer tool stage. +COPY --from=vulkan-tools /opt/vulkan-tools /opt/vulkan-tools +COPY --from=vulkan-tools /usr/include/vulkan /usr/include/vulkan +COPY --from=vulkan-tools /usr/include/vk_video /usr/include/vk_video +COPY --from=vulkan-tools /usr/include/spirv /usr/include/spirv +COPY --from=vulkan-tools /usr/share/cmake/SPIRV-Headers /usr/share/cmake/SPIRV-Headers + +RUN printf '%s\n' \ + '#!/bin/sh' \ + 'exec /opt/vulkan-tools/lib/ld-linux.so --library-path /opt/vulkan-tools/lib /opt/vulkan-tools/bin/glslc "$@"' \ + > /usr/local/bin/glslc \ + && chmod 0755 /usr/local/bin/glslc \ + && glslc --version + +WORKDIR /work +COPY . /work +COPY --from=sentencepiece-dependency /work/.deps/sentencepiece /work/.deps/sentencepiece +COPY --from=itn-dependency /work/.deps/itn /work/.deps/itn + +# Build metadata for the reported version (nightly. for nightly builds). +ARG VERSION_METADATA= +RUN case "${BACKEND}" in cpu|vulkan) ;; \ + *) echo "BACKEND must be cpu or vulkan" >&2; exit 2 ;; \ + esac \ + && scripts/configure.sh "${BACKEND}-server" \ + -DGGML_NATIVE=OFF \ + -DNEMO_SPEECH_BUILD_GRPC=OFF \ + -DNEMO_SPEECH_WITH_GRPC=OFF \ + -DNEMO_SPEECH_WITH_NORM=ON \ + "-DNEMO_SPEECH_VERSION_METADATA=${VERSION_METADATA}" \ + -DCMAKE_INSTALL_PREFIX=/opt/nemo-speech \ + "-DCMAKE_CXX_STANDARD_LIBRARIES:STRING=-lstdc++fs -lanl" \ + "-DCMAKE_EXE_LINKER_FLAGS=-Wl,--disable-new-dtags" \ + "-DCMAKE_SHARED_LINKER_FLAGS=-Wl,--disable-new-dtags" \ + && cmake --build --preset "${BACKEND}-server" \ + && cmake --install "build/${BACKEND}-server" --prefix /opt/nemo-speech \ + && artifact_backend="${ARTIFACT_BACKEND:-${BACKEND}}" \ + && scripts/release/package-linux.sh \ + --install-prefix /opt/nemo-speech \ + --backend "${BACKEND}" \ + --artifact-backend "${artifact_backend}" \ + --output-dir /release \ + --max-glibc 2.31 \ + ${RELEASE_VERSION:+--version "$RELEASE_VERSION"} + +FROM scratch AS artifact +COPY --from=builder /release/ / + +FROM portable-base AS cuda-builder + +ARG TARGETARCH +ARG CUDA_ARCH= +ARG ARTIFACT_BACKEND=cuda +ARG RELEASE_VERSION= + +ENV CUDA_HOME=/usr/local/cuda \ + PATH=/opt/cmake/bin:/usr/local/cuda/bin:/usr/local/sbin:/usr/local/bin:/usr/sbin:/usr/bin:/sbin:/bin + +COPY --from=cuda-tools /opt/cuda /usr/local/cuda +COPY --from=cuda-tools /opt/cuda-cudart-copyright /opt/cuda-cudart-copyright + +RUN set -eux; \ + nvcc --version; \ + cuda_version="$(nvcc --version \ + | sed -n 's/.*release \([0-9]*\)\.\([0-9]*\).*/\1-\2/p' \ + | head -n 1)"; \ + test -n "$cuda_version"; \ + install -Dm0644 /opt/cuda-cudart-copyright \ + "/usr/share/doc/cuda-cudart-${cuda_version}/copyright" + +WORKDIR /work +COPY . /work +COPY --from=sentencepiece-dependency /work/.deps/sentencepiece /work/.deps/sentencepiece +COPY --from=itn-dependency /work/.deps/itn /work/.deps/itn + +# Build metadata for the reported version (nightly. for nightly builds). +ARG VERSION_METADATA= +# The default architectures match the default CUDA 12.8 image; CUDA 13 builds +# (sm_110, sm_121) pass CUDA_ARCH. +RUN set -eux; \ + cuda_arch="${CUDA_ARCH}"; \ + if [ -z "$cuda_arch" ]; then \ + case "${TARGETARCH}" in \ + amd64) cuda_arch='75-real;80-virtual;86-real;89-real;120a-real' ;; \ + arm64) cuda_arch='87-real;87-virtual' ;; \ + *) echo "unsupported TARGETARCH: ${TARGETARCH}" >&2; exit 2 ;; \ + esac; \ + fi; \ + scripts/configure.sh cuda-server \ + -DGGML_NATIVE=OFF \ + -DGGML_CUDA_NCCL=OFF \ + -DNEMO_SPEECH_CUBLAS_SHIM=ON \ + -DNEMO_SPEECH_BUILD_GRPC=OFF \ + -DNEMO_SPEECH_WITH_GRPC=OFF \ + -DNEMO_SPEECH_WITH_NORM=ON \ + "-DNEMO_SPEECH_VERSION_METADATA=${VERSION_METADATA}" \ + -DCMAKE_INSTALL_PREFIX=/opt/nemo-speech \ + "-DCMAKE_CXX_STANDARD_LIBRARIES:STRING=-lstdc++fs -lanl" \ + "-DCMAKE_EXE_LINKER_FLAGS=-Wl,--disable-new-dtags" \ + "-DCMAKE_SHARED_LINKER_FLAGS=-Wl,--disable-new-dtags" \ + "-DCMAKE_CUDA_ARCHITECTURES=${cuda_arch}"; \ + cmake --build --preset cuda-server; \ + cmake --install build/cuda-server --prefix /opt/nemo-speech; \ + scripts/release/package-linux.sh \ + --install-prefix /opt/nemo-speech \ + --backend cuda \ + --artifact-backend "${ARTIFACT_BACKEND}" \ + --output-dir /release \ + --max-glibc 2.31 \ + ${RELEASE_VERSION:+--version "$RELEASE_VERSION"} + +FROM scratch AS cuda-artifact +COPY --from=cuda-builder /release/ / diff --git a/docs/build.md b/docs/build.md index 1c8d91e..42d6e67 100644 --- a/docs/build.md +++ b/docs/build.md @@ -196,6 +196,15 @@ prefix without `sudo`: CC=gcc-12 CXX=g++-12 scripts/build_itn_deps.sh ``` +With `STATIC=1`, the default on macOS, the script also builds pinned Protobuf +and RE2 and installs static archives only. `libnemo_speech_text_normalization` +then carries the whole stack privately, so neither system package is needed at +build or run time. The release archives use this mode: + +```bash +STATIC=1 scripts/build_itn_deps.sh +``` + On Linux, normalization builds also require the static SentencePiece dependency: diff --git a/docs/development/README.md b/docs/development/README.md index bc3896f..d9e3399 100644 --- a/docs/development/README.md +++ b/docs/development/README.md @@ -16,3 +16,5 @@ the server want [ASR configuration](../asr/configuration.md), - [`cublas-shim.md`](cublas-shim.md) - the in-tree drop-in cuBLAS replacement under `kernels/` and where the custom GPU kernels live. - [Windows build notes](windows-build.md) +- [`releasing.md`](releasing.md) - how the release workflow builds, checks, and + publishes the binary archives. diff --git a/docs/development/releasing.md b/docs/development/releasing.md new file mode 100644 index 0000000..e740382 --- /dev/null +++ b/docs/development/releasing.md @@ -0,0 +1,98 @@ +# Releasing + +[`.github/workflows/release.yml`](../../.github/workflows/release.yml) builds +the binary archives that `scripts/install.sh` and `scripts/install.ps1` +download, checks them, and publishes them to GitHub Releases. + +| Platform | Archives | Built on | +|---|---|---| +| Linux x86_64 | `cpu`, `vulkan`, `cuda` | `docker/Dockerfile.release-linux` | +| Linux aarch64 | `cpu`, `vulkan`, `cuda12` (Orin), `cuda13` (Thor, DGX Spark) | `docker/Dockerfile.release-linux` | +| macOS | `aarch64-cpu`, `aarch64-metal`, `x86_64-cpu` | `macos-15`, `macos-15-intel` | +| Windows x86_64 | `cpu`, `vulkan`, `cuda` | `scripts/windows/build.ps1` | + +Archives are named `nemo-speech----.tar.gz` +(`.zip` on Windows), contain a single directory of the same name, and ship +with a `.sha256` file. + +Linux and macOS archives include ITN and TN: `scripts/build_itn_deps.sh` with +`STATIC=1` builds OpenFST, Sparrowhawk, Protobuf, and RE2 as static archives, +and `libnemo_speech_text_normalization` links them privately. The grammars, +`itn_configs.tar.bz2` and `tn_configs.tar.bz2`, are not built by the workflow; +each run downloads them from the release pinned in the `grammars` job, checks +their SHA-256 digests, and publishes them again. To ship new grammars, attach +them to a release and update the tag and digests there. + +## Cutting a release + +1. Set `NEMO_SPEECH_VERSION` in `VERSION` and merge it to `main`. +2. Push the matching tag: `git tag v0.2.0 && git push origin v0.2.0`. The + workflow fails if the tag and `VERSION` differ. +3. Approve the `release` environment when the build and checks finish. The + workflow creates a draft release with the archives, `SHA256SUMS`, build + provenance attestations, and the text-normalization grammars. +4. Review the draft and publish it. + +A daily scheduled run publishes the `nightly` prerelease, which +`install.sh --channel nightly` installs. It is skipped when `main` has not +moved since the current nightly; running the workflow manually with `nightly` +always rebuilds. Nightly binaries report the `VERSION` value with build metadata +(for example `0.2.0+nightly.a1b2c3d`, set through the +`NEMO_SPEECH_VERSION_METADATA` CMake option), so they are not mistaken for the +release. + +Pull requests that change release packaging (the workflow, the release +Dockerfile and packagers, the dependency build scripts, the CMake files that +link the bundled dependencies, or the model smoke test) +run the workflow as a dry run once copy-pr-bot mirrors them to a +`pull-request/` branch. Run the workflow manually with +`dry-run` to build and check everything without publishing. + +## What every run checks + +- **CPU baseline:** x86_64 archives target x86-64-v3 (AVX2, FMA, F16C, BMI2) + and are built with `GGML_NATIVE=OFF`. The Linux x86_64 CPU archive + transcribes audio under QEMU's Haswell model, which has no AVX-512. +- **Self-contained packages:** Linux archives need glibc 2.31 or newer and find + their bundled libraries through `DT_RPATH`, which `LD_LIBRARY_PATH` cannot + override. Vulkan archives use the host's `libstdc++` and `libgcc_s`, which + the host's Vulkan drivers also need. macOS archives need macOS 13.3 or newer, link only system + libraries, and are ad-hoc signed. Windows archives bundle every DLL they + import except Windows and GPU driver libraries. +- **Smoke tests:** CPU archives run `tests/ci/model_smoke.py` on their runner. + The x86_64 Linux and Windows CUDA archives run it on L4 GPUs; the Windows run + blocks only tagged releases. Linux Vulkan archives only start (`--version`); + the aarch64 CUDA and Windows Vulkan archives are not run. On Linux and macOS + the test also synthesizes digits through TN and transcribes them back through + ITN. + +## Runners + +CUDA builds are the slowest jobs. Set these repository variables to run them +on larger runners: + +| Variable | Default | +|---|---| +| `RELEASE_LINUX_X64_CUDA_RUNNER` | `ubuntu-24.04` | +| `RELEASE_LINUX_ARM64_CUDA_RUNNER` | `ubuntu-24.04-arm` | +| `RELEASE_WINDOWS_CUDA_RUNNER` | `windows-2022` | + +## Building an archive locally + +```sh +# Linux, from the repository root (CUDA: --target cuda-artifact) +docker build --platform=linux/amd64 -f docker/Dockerfile.release-linux \ + --build-arg BACKEND=cpu --target artifact \ + --output type=local,dest=release-artifacts . + +# macOS, after scripts/build_sentencepiece_static.sh, scripts/build_itn_deps.sh, +# and installing a preset configured with -DNEMO_SPEECH_WITH_NORM=ON +# -DGGML_NATIVE=OFF -DCMAKE_OSX_DEPLOYMENT_TARGET=13.3, all with +# MACOSX_DEPLOYMENT_TARGET=13.3 in the environment +scripts/release/package-macos.sh --install-prefix --backend metal --arch aarch64 +``` + +```powershell +# Windows, after scripts\windows\build.ps1 -Backend cpu -Profile server -CMakeArgs '-DGGML_NATIVE=OFF' +scripts\windows\package-release.ps1 -BuildDir -Backend cpu +``` diff --git a/docs/install.md b/docs/install.md index c29d4ee..7d4358d 100644 --- a/docs/install.md +++ b/docs/install.md @@ -9,10 +9,11 @@ downloads a model only when explicitly enabled with an indexed name. See [models and cache](cli.md#models-and-cache). > [!IMPORTANT] -> **For the best performance and the latest features, build natively from source** with -> `--source` (`-Source` on Windows) or by following the [source-build guide](build.md). A native -> build is compiled for your machine's CPU and GPU, and release tags can be out of sync with the -> `main` branch. +> **For the best performance, build natively from source** with `--source` (`-Source` on +> Windows) or by following the [source-build guide](build.md). A native build is compiled for +> your machine's CPU and GPU. Tagged releases are cut periodically and can trail the `main` +> branch; for prebuilt binaries of the latest `main`, pass `--channel nightly` +> (`-Channel nightly` on Windows) to install the nightly prerelease, which is rebuilt daily. ## Linux and macOS @@ -37,12 +38,20 @@ x86_64 CUDA archive supports Turing-class GPUs (compute capability 7.5, including RTX 20-series) and newer. On an older GPU, select `--backend cpu` or `--backend vulkan`, or build from source with a compatible CUDA toolkit. +Linux and macOS archives include inverse text normalization for transcripts +(`--itn-model-dir`) and text normalization for synthesis (`--tn-model-dir`). +The grammars are published with each release as `itn_configs.tar.bz2` and +`tn_configs.tar.bz2`. + The installer selects CUDA when `nvidia-smi` is available, Metal on Apple -Silicon, and CPU otherwise. Override the backend or force a source build: +Silicon, and CPU otherwise. Override the backend, install the nightly +prerelease, or force a source build: ```bash curl -fsSL https://github.com/NVIDIA/NeMo-Speech.cpp/raw/main/scripts/install.sh | sh -s -- --backend cpu +curl -fsSL https://github.com/NVIDIA/NeMo-Speech.cpp/raw/main/scripts/install.sh | + sh -s -- --channel nightly curl -fsSL https://github.com/NVIDIA/NeMo-Speech.cpp/raw/main/scripts/install.sh | sh -s -- --source ``` @@ -91,25 +100,22 @@ irm https://github.com/NVIDIA/NeMo-Speech.cpp/raw/main/scripts/install.ps1 | iex The installer updates the current user's `PATH`. Open a new PowerShell window, then run `nemo-speech --version`. -Select a backend explicitly when needed: +Pass options after the script block. Install the nightly prerelease, or select +a backend explicitly: ```powershell -irm https://github.com/NVIDIA/NeMo-Speech.cpp/raw/main/scripts/install.ps1 ` - -OutFile .\install-nemo-speech.ps1 -powershell -ExecutionPolicy Bypass -File .\install-nemo-speech.ps1 ` - -Source -Backend cuda +iex "& {$(irm https://github.com/NVIDIA/NeMo-Speech.cpp/raw/main/scripts/install.ps1)} -Channel nightly" +iex "& {$(irm https://github.com/NVIDIA/NeMo-Speech.cpp/raw/main/scripts/install.ps1)} -Source -Backend cuda" ``` Select the components to install: ```powershell # ASR and diarization only -powershell -ExecutionPolicy Bypass -File .\install-nemo-speech.ps1 ` - -Source -Backend cpu -Profile asr +iex "& {$(irm https://github.com/NVIDIA/NeMo-Speech.cpp/raw/main/scripts/install.ps1)} -Source -Backend cpu -Profile asr" # Full runtime profile (add -HttpTls for TLS) -powershell -ExecutionPolicy Bypass -File .\install-nemo-speech.ps1 ` - -Source -Backend cuda -Profile full +iex "& {$(irm https://github.com/NVIDIA/NeMo-Speech.cpp/raw/main/scripts/install.ps1)} -Source -Backend cuda -Profile full" ``` | Profile | Components | diff --git a/greptile.json b/greptile.json new file mode 100644 index 0000000..5cecc9c --- /dev/null +++ b/greptile.json @@ -0,0 +1,62 @@ +{ + "triggerOnUpdates": true, + "instructions": "NeMo-Speech.cpp is a C++17 speech runtime (ASR, diarization, TTS, translation, VoiceChat) built on ggml from a pinned llama.cpp submodule, with CPU, CUDA, Metal and Vulkan backends and portable release archives for Linux, macOS and Windows. patches/ holds the project's own changes to llama.cpp and ggml, including custom GPU kernels; review them like any other code.\n\nPrioritize findings in this order: correctness, performance, portability, maintainability. Report only problems the change introduces or makes reachable, each with the concrete input, backend or platform where it fails and what to do about it. Do not comment on formatting, naming or anything pre-commit already checks.", + "customContext": { + "rules": [ + { + "scope": [ + "src/**", + "app/**", + "server/**", + "include/**", + "patches/**", + "kernels/**", + "conversion/**", + "convert_model.py", + "scripts/**", + ".github/workflows/**" + ], + "rule": "Look for wrong results, crashes, memory errors, races, uninitialized or stale state, unhandled edge cases (empty input, boundary sizes, partial tiles or chunks), numerical instability, and fallback paths that behave differently from the fast path. Streaming results must not depend on chunk size." + }, + { + "scope": [ + "src/**", + "server/**", + "patches/**", + "kernels/**" + ], + "rule": "Look for regressions on hot paths: added host-device synchronization, per-chunk allocations, redundant work, extra copies between backends, ops that fall back to the CPU, and caches or reuse that never take effect." + }, + { + "scope": [ + "src/**", + "app/**", + "server/**", + "include/**", + "patches/**", + "kernels/**", + "scripts/**", + "docker/**", + "CMakeLists.txt", + "**/CMakeLists.txt" + ], + "rule": "Look for assumptions tied to one GPU architecture, backend, operating system or compiler. Hardware limits and features must be checked with a working fallback, and code must build with GCC, Clang, Apple Clang and MSVC. Release archives must stay self-contained and run on every supported target." + }, + { + "scope": [ + "src/**", + "app/**", + "server/**", + "include/**", + "patches/**", + "kernels/**", + "conversion/**", + "scripts/**", + "docs/**", + "README.md" + ], + "rule": "Look for duplicated logic that should reuse an existing helper, platform-specific branches scattered across call sites, dead or unreachable code, unexplained magic constants, and documentation that no longer matches the code." + } + ] + } +} diff --git a/kernels/cublas_shim.cu b/kernels/cublas_shim.cu index 07d1efa..ad82b98 100644 --- a/kernels/cublas_shim.cu +++ b/kernels/cublas_shim.cu @@ -36,7 +36,7 @@ enum { OP_N = 0, OP_T = 1 }; enum { R_32F = 0, R_16F = 2, R_16BF = 14 }; enum { COMPUTE_16F = 64, COMPUTE_32F = 68 }; -enum { STATUS_SUCCESS = 0 }; +enum { STATUS_SUCCESS = 0, STATUS_INVALID_VALUE = 7, STATUS_EXECUTION_FAILED = 13 }; // cudaDataType is already defined by library_types.h (via cuda_runtime.h). typedef void* cublasHandle_t; @@ -1362,6 +1362,58 @@ launch( m, n, k, opA, opB, A, lda, sa, ta, B, ldb, sb, tb, C, ldc, sc, tc, alpha, beta, esz(ta), esz(tb), esz(tc), bb); } + +// The batch index is carried in grid.z, which CUDA limits to 65535; larger +// batches are issued as several launches. +constexpr int kMaxGridZ = 65535; + +inline const void* +offset(const void* p, long long elements, int dt) { + return (const char*)p + elements * (long long)esz(dt); +} + +inline void +launch_strided( + int m, int n, int k, int opA, int opB, const void* A, int lda, long long sa, int ta, + const void* B, int ldb, long long sb, int tb, void* C, int ldc, long long sc, int tc, + float alpha, float beta, int batch, cudaStream_t s, ShimHandle* sh) { + for (int b0 = 0, count = 0; b0 < batch; b0 += count) { + count = std::min(batch - b0, kMaxGridZ); + launch( + m, n, k, opA, opB, offset(A, b0 * sa, ta), lda, sa, ta, offset(B, b0 * sb, tb), ldb, sb, + tb, (void*)offset(C, b0 * sc, tc), ldc, sc, tc, alpha, beta, count, s, sh); + } +} + +inline void +launch_ptrs( + int m, int n, int k, int opA, int opB, const void* const* A, int lda, int ta, + const void* const* B, int ldb, int tb, void* const* C, int ldc, int tc, float alpha, float beta, + int batch, cudaStream_t s) { + for (int b0 = 0, count = 0; b0 < batch; b0 += count) { + count = std::min(batch - b0, kMaxGridZ); + dim3 blk(16, 16, 1), grd((m + 15) / 16, (n + 15) / 16, count); + k_ptrs<<>>( + m, n, k, opA, opB, A + b0, lda, ta, B + b0, ldb, tb, C + b0, ldc, tc, alpha, beta, + count); + } +} + +// cuBLAS rejects negative sizes and completes empty problems without work. +// Returns false, with the status to report, when there is nothing to launch. +inline bool +gemm_has_work(int m, int n, int k, int batch, cublasStatus_t* status) { + *status = m < 0 || n < 0 || k < 0 || batch < 0 ? STATUS_INVALID_VALUE : STATUS_SUCCESS; + return *status == STATUS_SUCCESS && m > 0 && n > 0 && batch > 0; +} + +// Report a failed kernel launch instead of claiming success. The shim links the +// CUDA runtime statically, so this error state is its own: it never holds or +// clears an error pending in the caller's runtime. +inline cublasStatus_t +launch_status() { + return cudaGetLastError() == cudaSuccess ? STATUS_SUCCESS : STATUS_EXECUTION_FAILED; +} } // namespace extern "C" { @@ -1414,6 +1466,12 @@ NEMO_SPEECH_CUBLAS_EXPORT cublasStatus_t cublasSetMathMode(cublasHandle_t, cublasMath_t) { return STATUS_SUCCESS; } +// The shim allocates its own split-K workspaces, so a caller-provided +// workspace is accepted and left unused. +NEMO_SPEECH_CUBLAS_EXPORT cublasStatus_t +cublasSetWorkspace_v2(cublasHandle_t, void*, size_t) { + return STATUS_SUCCESS; +} NEMO_SPEECH_CUBLAS_EXPORT const char* cublasGetStatusString(cublasStatus_t) { return "EDGE_SHIM_OK"; @@ -1425,12 +1483,16 @@ cublasGemmEx( const void* alpha, const void* A, cudaDataType ta, int lda, const void* B, cudaDataType tb, int ldb, const void* beta, void* C, cudaDataType tc, int ldc, cublasComputeType_t ct, cublasGemmAlgo_t) { + cublasStatus_t status; + if (!gemm_has_work(m, n, k, 1, &status)) { + return status; + } ShimHandle* sh = (ShimHandle*)h; const cudaStream_t stream = stream_for_handle(sh); launch( m, n, k, opA, opB, A, lda, 0, ta, B, ldb, 0, tb, C, ldc, 0, tc, host_scalar(alpha, ct), host_scalar(beta, ct), 1, stream, sh); - return STATUS_SUCCESS; + return launch_status(); } NEMO_SPEECH_CUBLAS_EXPORT cublasStatus_t cublasGemmStridedBatchedEx( @@ -1438,12 +1500,16 @@ cublasGemmStridedBatchedEx( const void* alpha, const void* A, cudaDataType ta, int lda, long long sa, const void* B, cudaDataType tb, int ldb, long long sb, const void* beta, void* C, cudaDataType tc, int ldc, long long sc, int batch, cublasComputeType_t ct, cublasGemmAlgo_t) { + cublasStatus_t status; + if (!gemm_has_work(m, n, k, batch, &status)) { + return status; + } ShimHandle* sh = (ShimHandle*)h; const cudaStream_t stream = stream_for_handle(sh); - launch( + launch_strided( m, n, k, opA, opB, A, lda, sa, ta, B, ldb, sb, tb, C, ldc, sc, tc, host_scalar(alpha, ct), host_scalar(beta, ct), batch, stream, sh); - return STATUS_SUCCESS; + return launch_status(); } NEMO_SPEECH_CUBLAS_EXPORT cublasStatus_t cublasGemmBatchedEx( @@ -1451,37 +1517,64 @@ cublasGemmBatchedEx( const void* alpha, const void* const Aarray[], cudaDataType ta, int lda, const void* const Barray[], cudaDataType tb, int ldb, const void* beta, void* const Carray[], cudaDataType tc, int ldc, int batch, cublasComputeType_t ct, cublasGemmAlgo_t) { + cublasStatus_t status; + if (!gemm_has_work(m, n, k, batch, &status)) { + return status; + } ShimHandle* sh = (ShimHandle*)h; const cudaStream_t stream = stream_for_handle(sh); - dim3 blk(16, 16, 1), grd((m + 15) / 16, (n + 15) / 16, batch); - k_ptrs<<>>( + launch_ptrs( m, n, k, opA, opB, Aarray, lda, ta, Barray, ldb, tb, Carray, ldc, tc, - host_scalar(alpha, ct), host_scalar(beta, ct), batch); - return STATUS_SUCCESS; + host_scalar(alpha, ct), host_scalar(beta, ct), batch, stream); + return launch_status(); } NEMO_SPEECH_CUBLAS_EXPORT cublasStatus_t cublasSgemm_v2( cublasHandle_t h, cublasOperation_t opA, cublasOperation_t opB, int m, int n, int k, const float* alpha, const float* A, int lda, const float* B, int ldb, const float* beta, float* C, int ldc) { + cublasStatus_t status; + if (!gemm_has_work(m, n, k, 1, &status)) { + return status; + } ShimHandle* sh = (ShimHandle*)h; const cudaStream_t stream = stream_for_handle(sh); launch( m, n, k, opA, opB, A, lda, 0, R_32F, B, ldb, 0, R_32F, C, ldc, 0, R_32F, *alpha, *beta, 1, stream, sh); - return STATUS_SUCCESS; + return launch_status(); } NEMO_SPEECH_CUBLAS_EXPORT cublasStatus_t cublasSgemmStridedBatched( cublasHandle_t h, cublasOperation_t opA, cublasOperation_t opB, int m, int n, int k, const float* alpha, const float* A, int lda, long long sa, const float* B, int ldb, long long sb, const float* beta, float* C, int ldc, long long sc, int batch) { + cublasStatus_t status; + if (!gemm_has_work(m, n, k, batch, &status)) { + return status; + } ShimHandle* sh = (ShimHandle*)h; const cudaStream_t stream = stream_for_handle(sh); - launch( + launch_strided( m, n, k, opA, opB, A, lda, sa, R_32F, B, ldb, sb, R_32F, C, ldc, sc, R_32F, *alpha, *beta, batch, stream, sh); - return STATUS_SUCCESS; + return launch_status(); +} +NEMO_SPEECH_CUBLAS_EXPORT cublasStatus_t +cublasSgemmBatched( + cublasHandle_t h, cublasOperation_t opA, cublasOperation_t opB, int m, int n, int k, + const float* alpha, const float* const Aarray[], int lda, const float* const Barray[], int ldb, + const float* beta, float* const Carray[], int ldc, int batch) { + cublasStatus_t status; + if (!gemm_has_work(m, n, k, batch, &status)) { + return status; + } + ShimHandle* sh = (ShimHandle*)h; + const cudaStream_t stream = stream_for_handle(sh); + launch_ptrs( + m, n, k, opA, opB, (const void* const*)Aarray, lda, R_32F, (const void* const*)Barray, ldb, + R_32F, (void* const*)Carray, ldc, R_32F, *alpha, *beta, batch, stream); + return launch_status(); } NEMO_SPEECH_CUBLAS_EXPORT cublasStatus_t cublasStrsmBatched( diff --git a/kernels/ver_cublas.map b/kernels/ver_cublas.map index 76f1682..754e58b 100644 --- a/kernels/ver_cublas.map +++ b/kernels/ver_cublas.map @@ -7,11 +7,13 @@ libcublas.so.@NEMO_SPEECH_CUBLAS_SOVERSION@ { cublasDestroy_v2; cublasSetStream_v2; cublasSetMathMode; + cublasSetWorkspace_v2; cublasGetStatusString; cublasGemmEx; cublasGemmStridedBatchedEx; cublasGemmBatchedEx; cublasSgemm_v2; + cublasSgemmBatched; cublasSgemmStridedBatched; cublasStrsmBatched; local: *; diff --git a/scripts/build_itn_deps.sh b/scripts/build_itn_deps.sh index 3a04c7f..7fe288d 100755 --- a/scripts/build_itn_deps.sh +++ b/scripts/build_itn_deps.sh @@ -3,14 +3,20 @@ # SPDX-License-Identifier: Apache-2.0 # Build the Sparrowhawk ITN stack (OpenFST 1.8 + Sparrowhawk) from pinned # sarane22 forks and install to a user-writable project prefix, enabling -# -DNEMO_SPEECH_WITH_ITN=ON. Shared by the x86_64 and aarch64 images. +# -DNEMO_SPEECH_WITH_NORM=ON. Shared by the x86_64 and aarch64 images. # -# Expects: protobuf headers + protoc, re2, autotools, and gcc-12 as CC/CXX. The -# Dockerfiles set CC/CXX=gcc-12 for this step (gcc-13/14 ICE on OpenFST's heavy -# templates at -O2) while the runtime itself builds with gcc-13. Sparrowhawk uses -# an in-tree, OpenFST-only compatibility implementation for the tiny subset of +# Expects autotools and, on Linux, gcc-12 as CC/CXX. docker/Dockerfile sets +# CC/CXX=gcc-12 for this step (gcc-13/14 ICE on OpenFST's heavy templates at +# -O2) while the runtime itself builds with gcc-13. Sparrowhawk uses an in-tree, +# OpenFST-only compatibility implementation for the tiny subset of # thrax::GrmManager that it calls; no Thrax or fstscript library is built/linked. # +# STATIC=0 (Linux default) builds shared libraries against the system protobuf +# headers, protoc, and RE2. STATIC=1 (macOS default) also builds pinned protobuf +# and RE2, and installs position-independent static archives only, so +# nemo_speech_text_normalization carries the whole stack privately; the release +# archives use this mode. It additionally requires CMake. +# # Usage: scripts/build_itn_deps.sh [WORKDIR] (default: ./.deps/itn-build) set -euo pipefail @@ -20,14 +26,21 @@ PREFIX="${PREFIX:-$REPO/.deps/itn}" JOBS="${JOBS:-8}" # Cap parallelism: OpenFST's template-heavy translation units can OOM cc1plus. JOBS="$(( JOBS < 4 ? JOBS : 4 ))" +if [ "$(uname -s)" = Darwin ]; then + STATIC="${STATIC:-1}" +else + STATIC="${STATIC:-0}" +fi CXXO="-std=c++17 -O2" SHIM="$REPO/src/common/text_normalization/compat/sparrowhawk_compat.h" ITN_COMPAT="$REPO/src/common/text_normalization/compat" +LICENSE_DIR="$PREFIX/share/licenses/nemo-speech/third_party" clone() { # clone if [ ! -d "$2" ]; then - git clone "$1" "$2" - git -C "$2" checkout --quiet "$3" + git init --quiet "$2" + git -C "$2" fetch --quiet --depth 1 "$1" "$3" + git -C "$2" checkout --quiet FETCH_HEAD elif [ "$(git -C "$2" rev-parse HEAD)" != "$3" ]; then echo "$2 exists at the wrong revision; remove it or choose a clean WORKDIR" >&2 echo " expected: $3" >&2 @@ -36,19 +49,75 @@ clone() { # clone fi } +# Stamp the shipped autotools output newer than its inputs so make does not try +# to regenerate it with whichever autoconf/automake the host has. +stamp_autotools() { + touch -t 202001010000 configure.ac acinclude.m4 2>/dev/null || true + [ -d m4 ] && touch -t 202001010000 m4/*.m4 2>/dev/null || true + find . -name 'Makefile.am' -exec touch -t 202001010000 {} + + touch -t 202001020000 aclocal.m4 + touch -t 202001030000 configure + find . -name '*.in' -exec touch -t 202001030000 {} + +} + +install_license() { # install_license + install -d "$LICENSE_DIR/$2" + install -m 0644 "$1" "$LICENSE_DIR/$2/$(basename "$1")" +} + +LIBRARY_KIND=() +if [ "$STATIC" = 1 ]; then + LIBRARY_KIND=(--disable-shared --enable-static --with-pic) +fi + mkdir -p "$WORK" cd "$WORK" # Pinned OpenFST and Sparrowhawk compatibility revisions. clone https://github.com/sarane22/openfst.git openfst fc23b4cf529429284b874a26f28b15c6cc94f404 clone https://github.com/sarane22/sparrowhawk.git sparrowhawk 8b082acc507312077a096be8398584a13832c490 +# --------------------------------------------------------- protobuf and RE2 +# The last releases that do not require Abseil. +if [ "$STATIC" = 1 ]; then + clone https://github.com/protocolbuffers/protobuf.git protobuf f0dc78d7e6e331b8c6bb2d5283e06aa26883ca7c # v21.12 + clone https://github.com/google/re2.git re2 3a8436ac436124a57a4e22d5c8713a2d42b381d7 # 2023-03-01 + # Build outside the source trees: RE2 ships a Bazel BUILD file, which + # "re2/build" resolves to on case-insensitive filesystems (macOS). + cmake -S "$WORK/protobuf" -B "$WORK/protobuf-build" \ + -DCMAKE_BUILD_TYPE=Release \ + -DCMAKE_INSTALL_PREFIX="$PREFIX" \ + -DCMAKE_INSTALL_LIBDIR=lib \ + -DCMAKE_POSITION_INDEPENDENT_CODE=ON \ + -Dprotobuf_BUILD_SHARED_LIBS=OFF \ + -Dprotobuf_BUILD_TESTS=OFF \ + -Dprotobuf_WITH_ZLIB=OFF + cmake --build "$WORK/protobuf-build" --target install -j "$JOBS" + cmake -S "$WORK/re2" -B "$WORK/re2-build" \ + -DCMAKE_BUILD_TYPE=Release \ + -DCMAKE_INSTALL_PREFIX="$PREFIX" \ + -DCMAKE_INSTALL_LIBDIR=lib \ + -DCMAKE_POSITION_INDEPENDENT_CODE=ON \ + -DBUILD_SHARED_LIBS=OFF \ + -DRE2_BUILD_TESTING=OFF + cmake --build "$WORK/re2-build" --target install -j "$JOBS" + # Sparrowhawk's configure and Makefiles run protoc from PATH. + export PATH="$PREFIX/bin:$PATH" + install_license "$WORK/protobuf/LICENSE" protobuf + install_license "$WORK/re2/LICENSE" re2 +fi + # ---------------------------------------------------------------- OpenFST 1.8 cd "$WORK/openfst" # FST_FLAGS_v rename missed by the fork. -sed -i 's/\bFLAGS_v\b/FST_FLAGS_v/g' src/include/fst/label-reachable.h +perl -pi -e 's/\bFLAGS_v\b/FST_FLAGS_v/g' src/include/fst/label-reachable.h +# VectorHashBiTable's copy constructor reads a nonexistent member (fixed +# upstream); Clang 20+ rejects it without instantiation. +perl -pi -e 's/selector_\(table\.s_\)/selector_(table.selector_)/' src/include/fst/bi-table.h # FAR + PDT cover Sparrowhawk's runtime grammar formats. Disable command-line # tools/script wrappers: the runtime calls the typed C++ OpenFST API directly. -./configure --prefix="$PREFIX" --enable-far --enable-pdt --disable-bin CXXFLAGS="$CXXO" +stamp_autotools +./configure --prefix="$PREFIX" --enable-far --enable-pdt --disable-bin \ + ${LIBRARY_KIND[@]+"${LIBRARY_KIND[@]}"} CXXFLAGS="$CXXO" make -j"$JOBS" make install # Stage all core headers plus the FAR/PDT extension templates used by the @@ -67,25 +136,17 @@ fi # -------------------------------------------------------------- Sparrowhawk cd "$WORK/sparrowhawk" # (1) Autoconf tarball pins -std=c++11; OpenFST 1.8 headers need C++17. -sed -i 's/-std=c++11/-std=c++17/g' configure -[ -f configure.ac ] && sed -i 's/-std=c++11/-std=c++17/g' configure.ac || true -./configure --prefix="$PREFIX" --disable-bin \ +perl -pi -e 's/-std=c\+\+11/-std=c++17/g' configure +[ -f configure.ac ] && perl -pi -e 's/-std=c\+\+11/-std=c++17/g' configure.ac || true +# (2) Stamp after the edits so make does not try to regenerate configure. +stamp_autotools +./configure --prefix="$PREFIX" --disable-bin ${LIBRARY_KIND[@]+"${LIBRARY_KIND[@]}"} \ CPPFLAGS="-I$ITN_COMPAT -I$PREFIX/include" \ LDFLAGS="-L$PREFIX/lib" CXXFLAGS="$CXXO" -# (2) Stamp generated autotools files so make does not try to regenerate them. -touch -d '2020-01-01 00:00:00' configure.ac acinclude.m4 2>/dev/null || true -[ -d m4 ] && touch -d '2020-01-01 00:00:00' m4/*.m4 2>/dev/null || true -find . -name 'Makefile.am' -exec touch -d '2020-01-01 00:00:00' {} + -touch -d '2020-01-02 00:00:00' aclocal.m4 -touch -d '2020-01-03 00:00:00' configure -find . -name '*.in' -exec touch -d '2020-01-03 00:00:00' {} + -touch -d '2020-01-04 00:00:00' config.status -find . -name 'Makefile' -exec touch -d '2020-01-05 00:00:00' {} + - # (3) Build + install the proto stubs, library, and headers with the OpenFST 1.8 # compat shim force-included. src/bin (normalizer_main CLI) is skipped: the -# the runtime links libsparrowhawk directly. Put the compatibility include first so +# runtime links libsparrowhawk directly. Put the compatibility include first so # Sparrowhawk resolves without the Thrax project. CPPF="-I$ITN_COMPAT -I$PREFIX/include -include $SHIM -funsigned-char" make -C src/proto CPPFLAGS="$CPPF" @@ -101,13 +162,12 @@ for f in "$PREFIX"/lib/lib{fst,fstfar,sparrowhawk}.so.*; do [ -f "$f" ] && [ ! -L "$f" ] && strip --strip-unneeded "$f" done -LICENSE_DIR="$PREFIX/share/licenses/nemo-speech/third_party" -install -Dm0644 "$WORK/openfst/COPYING" "$LICENSE_DIR/openfst/COPYING" -install -Dm0644 "$WORK/sparrowhawk/LICENSE" "$LICENSE_DIR/sparrowhawk/LICENSE" +install_license "$WORK/openfst/COPYING" openfst +install_license "$WORK/sparrowhawk/LICENSE" sparrowhawk if command -v ldconfig >/dev/null 2>&1 && [ "$(id -u)" -eq 0 ]; then ldconfig fi echo echo "ITN stack installed to $PREFIX:" -ls -1 "$PREFIX"/lib/libsparrowhawk.so "$PREFIX"/lib/libfstfar.so "$PREFIX"/lib/libfst.so +ls -1 "$PREFIX"/lib/lib{sparrowhawk,fstfar,fst}.* diff --git a/scripts/build_sentencepiece_static.sh b/scripts/build_sentencepiece_static.sh index 05f5b20..00bd999 100755 --- a/scripts/build_sentencepiece_static.sh +++ b/scripts/build_sentencepiece_static.sh @@ -12,7 +12,7 @@ JOBS="${JOBS:-8}" JOBS="$(( JOBS < 4 ? JOBS : 4 ))" SOURCE="$WORK/source" BUILD="$WORK/build" -COMMIT=17d7580d6407802f85855d2cc9190634e2c95624 +COMMIT=31646a467d2051eb904e0b45de3a73e91fe1c1e3 if [ ! -d "$SOURCE/.git" ]; then git clone --filter=blob:none --no-checkout https://github.com/google/sentencepiece.git "$SOURCE" @@ -32,7 +32,8 @@ install -m 0644 "$BUILD/src/libsentencepiece.a" "$PREFIX/lib/libsentencepiece.a" install -m 0644 "$SOURCE/src/sentencepiece_processor.h" "$PREFIX/include/sentencepiece_processor.h" LICENSE_DIR="$PREFIX/share/licenses/nemo-speech/third_party/sentencepiece" -install -Dm0644 "$SOURCE/LICENSE" "$LICENSE_DIR/LICENSE" -install -Dm0644 "$SOURCE/third_party/absl/LICENSE" "$LICENSE_DIR/absl-LICENSE" -install -Dm0644 "$SOURCE/third_party/darts_clone/LICENSE" "$LICENSE_DIR/darts-clone-LICENSE" -install -Dm0644 "$SOURCE/third_party/protobuf-lite/LICENSE" "$LICENSE_DIR/protobuf-lite-LICENSE" +install -d "$LICENSE_DIR" +install -m 0644 "$SOURCE/LICENSE" "$LICENSE_DIR/LICENSE" +install -m 0644 "$SOURCE/third_party/absl/LICENSE" "$LICENSE_DIR/absl-LICENSE" +install -m 0644 "$SOURCE/third_party/darts_clone/LICENSE" "$LICENSE_DIR/darts-clone-LICENSE" +install -m 0644 "$SOURCE/third_party/protobuf-lite/LICENSE" "$LICENSE_DIR/protobuf-lite-LICENSE" diff --git a/scripts/install.ps1 b/scripts/install.ps1 index e601b6e..513f351 100644 --- a/scripts/install.ps1 +++ b/scripts/install.ps1 @@ -205,6 +205,19 @@ if (-not (Get-Command curl.exe -ErrorAction SilentlyContinue)) { } $installIdentity = "$releaseVersion windows $arch $Backend" +# The nightly tag is rebuilt in place, so identify a nightly install by its archive digest. +if ($releaseVersion -eq 'nightly' -and -not $Source -and $binaryCandidate) { + $digestFile = [IO.Path]::GetTempFileName() + try { + Invoke-DownloadWithRetry -Uri "$url.sha256" -OutFile $digestFile + $nightlySha256 = ((Get-Content $digestFile -Raw).Trim() -split '\s+')[0] + if ($nightlySha256) { $installIdentity += " sha256:$nightlySha256" } + } catch { + # Without the digest the identity never matches, so the archive is downloaded. + } finally { + Remove-Item -Force -ErrorAction SilentlyContinue $digestFile + } +} $extraComponents = [Collections.Generic.List[string]]::new() foreach ($component in @( @{ Name = 'grpc'; Enabled = $Grpc }, @{ Name = 'nmt'; Enabled = $Nmt }, diff --git a/scripts/install.sh b/scripts/install.sh index 6f89230..721c035 100755 --- a/scripts/install.sh +++ b/scripts/install.sh @@ -239,6 +239,12 @@ echo "Prefix: $prefix" [ "$dry_run" -eq 0 ] || exit 0 install_identity="$release_version $os $arch $artifact_backend" +# The nightly tag is rebuilt in place, so identify a nightly install by its archive digest. +if [ "$release_version" = nightly ] && [ "$install_mode" != source ] && [ "$binary_candidate" -eq 1 ] && + nightly_sha256=$(curl -fsSL --retry 3 "$checksum_url" 2>/dev/null | awk 'NR == 1 { print $1 }') && + [ -n "$nightly_sha256" ]; then + install_identity="$install_identity sha256:$nightly_sha256" +fi source_identity="$release_version $os $arch $backend source:$source_ref profile:speech-server" install_metadata=$prefix/.nemo-speech-install if [ "$install_mode" != source ] && [ "$binary_candidate" -eq 1 ] && diff --git a/scripts/release/check_release.py b/scripts/release/check_release.py new file mode 100755 index 0000000..734be46 --- /dev/null +++ b/scripts/release/check_release.py @@ -0,0 +1,103 @@ +#!/usr/bin/env python3 +# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-License-Identifier: Apache-2.0 +"""Check release archives before they are published. + + check_release.py archives DIR --version VERSION + Check every archive in DIR: checksum, name, layout, + and required files. +""" +from __future__ import annotations + +import argparse +import hashlib +import pathlib +import re +import sys +import tarfile +import tempfile +import zipfile + +ARCHIVE = re.compile( + r"^nemo-speech-(?Pnightly|[0-9][0-9A-Za-z.+-]*)-(?Plinux|macos|windows)-" + r"(?Px86_64|aarch64)-(?Pcpu|vulkan|metal|cuda|cuda12|cuda13)" + r"\.(?Ptar\.gz|zip)$" +) + + +def extract(archive: pathlib.Path, dest: pathlib.Path) -> None: + if archive.name.endswith(".zip"): + with zipfile.ZipFile(archive) as z: + z.extractall(dest) + else: + if not hasattr(tarfile, "data_filter"): + raise SystemExit( + "error: safe tar extraction needs Python 3.12 or a release with tarfile.data_filter" + ) + with tarfile.open(archive) as t: + t.extractall(dest, filter="data") + + +def check_archives(directory: pathlib.Path, version: str) -> list[str]: + failures = [] + archives = sorted(p for p in directory.iterdir() if p.name.endswith((".tar.gz", ".zip"))) + if not archives: + return [f"{directory}: no archives"] + for archive in archives: + match = ARCHIVE.match(archive.name) + if not match: + failures.append(f"{archive.name}: unexpected archive name") + continue + if match["version"] != version: + failures.append(f"{archive.name}: version is not {version}") + expected_ext = "zip" if match["os"] == "windows" else "tar.gz" + if match["ext"] != expected_ext: + failures.append(f"{archive.name}: {match['os']} archives must be .{expected_ext}") + checksum = archive.with_name(archive.name + ".sha256") + digest = hashlib.sha256(archive.read_bytes()).hexdigest() + if not checksum.is_file(): + failures.append(f"{archive.name}: missing .sha256") + elif checksum.read_text().split() != [digest, archive.name]: + failures.append(f"{archive.name}: .sha256 does not match") + + package = archive.name[: -len("." + match["ext"])] + with tempfile.TemporaryDirectory() as tmp: + tmp_path = pathlib.Path(tmp) + extract(archive, tmp_path) + entries = list(tmp_path.iterdir()) + if [e.name for e in entries] != [package] or not entries[0].is_dir(): + failures.append(f"{archive.name}: must contain exactly one directory, {package}/") + continue + root = entries[0] + exe = "nemo-speech.exe" if match["os"] == "windows" else "nemo-speech" + for required in ( + f"bin/{exe}", + "share/licenses/nemo-speech/LICENSE", + "share/licenses/nemo-speech/THIRD_PARTY_NOTICES.md", + ): + if not (root / required).is_file(): + failures.append(f"{archive.name}: missing {required}") + print(f"checked {archive.name}", flush=True) + return failures + + +def main() -> None: + parser = argparse.ArgumentParser( + description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter + ) + sub = parser.add_subparsers(dest="command", required=True) + archives = sub.add_parser("archives") + archives.add_argument("directory", type=pathlib.Path) + archives.add_argument("--version", required=True) + args = parser.parse_args() + + failures = check_archives(args.directory, args.version) + for failure in failures: + print(f"error: {failure}", file=sys.stderr) + if failures: + raise SystemExit(1) + print("OK") + + +if __name__ == "__main__": + main() diff --git a/scripts/release/package-linux.sh b/scripts/release/package-linux.sh new file mode 100755 index 0000000..00a47f0 --- /dev/null +++ b/scripts/release/package-linux.sh @@ -0,0 +1,388 @@ +#!/usr/bin/env bash +# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-License-Identifier: Apache-2.0 +# Package a portable Linux installation for the binary installer. +set -euo pipefail + +ROOT="$(cd "$(dirname "${BASH_SOURCE[0]}")/../.." && pwd)" +install_prefix= +output_dir="$ROOT/release-artifacts" +backend= +artifact_backend= +max_glibc=2.31 +version= + +usage() { + cat <<'EOF' +Usage: scripts/release/package-linux.sh --install-prefix DIR --backend cpu|vulkan|cuda [OPTION ...] + +Options: + --artifact-backend NAME + Backend label used in the archive name + --output-dir DIR Destination for the archive and checksum + --version VERSION Override the version read from VERSION ("nightly" for + the nightly channel) + --max-glibc VERSION Reject binaries requiring a newer glibc (default: 2.31) + -h, --help + +The installed project and GCC runtimes are packaged together; the project must +be built with text normalization (-DNEMO_SPEECH_WITH_NORM=ON and the static +dependencies from scripts/build_itn_deps.sh). CUDA archives also include +libcudart. glibc, GPU drivers, and the Vulkan loader remain host +dependencies. Project binaries must find bundled libraries through DT_RPATH, +which LD_LIBRARY_PATH cannot override. +EOF +} + +while [[ $# -gt 0 ]]; do + case "$1" in + --install-prefix) + [[ $# -ge 2 ]] || { echo "error: --install-prefix requires a value" >&2; exit 2; } + install_prefix=$2 + shift 2 + ;; + --backend) + [[ $# -ge 2 ]] || { echo "error: --backend requires a value" >&2; exit 2; } + backend=$2 + shift 2 + ;; + --artifact-backend) + [[ $# -ge 2 ]] || { echo "error: --artifact-backend requires a value" >&2; exit 2; } + artifact_backend=$2 + shift 2 + ;; + --output-dir) + [[ $# -ge 2 ]] || { echo "error: --output-dir requires a value" >&2; exit 2; } + output_dir=$2 + shift 2 + ;; + --version) + [[ $# -ge 2 ]] || { echo "error: --version requires a value" >&2; exit 2; } + version=$2 + shift 2 + ;; + --max-glibc) + [[ $# -ge 2 ]] || { echo "error: --max-glibc requires a value" >&2; exit 2; } + max_glibc=$2 + shift 2 + ;; + -h|--help) + usage + exit 0 + ;; + *) + echo "error: unknown option '$1'" >&2 + usage >&2 + exit 2 + ;; + esac +done + +[[ -n "$install_prefix" ]] || { echo "error: --install-prefix is required" >&2; exit 2; } +[[ -d "$install_prefix" ]] || { echo "error: install prefix does not exist: $install_prefix" >&2; exit 1; } +[[ -x "$install_prefix/bin/nemo-speech" ]] || { + echo "error: install prefix does not contain bin/nemo-speech" >&2 + exit 1 +} +case "$backend" in + cpu|vulkan|cuda) ;; + *) + echo "error: --backend must be cpu, vulkan, or cuda" >&2 + exit 2 + ;; +esac +if [[ -z "$artifact_backend" ]]; then + artifact_backend=$backend +fi +case "$backend:$artifact_backend" in + cpu:cpu|vulkan:vulkan|cuda:cuda|cuda:cuda12|cuda:cuda13) ;; + *) + echo "error: invalid artifact backend '$artifact_backend' for '$backend'" >&2 + exit 2 + ;; +esac +[[ "$max_glibc" =~ ^[0-9]+(\.[0-9]+)+$ ]] || { + echo "error: --max-glibc must be a dotted version" >&2 + exit 2 +} + +if [[ -z "$version" ]]; then + version="$(sed -n 's/^NEMO_SPEECH_VERSION:[[:space:]]*//p' "$ROOT/VERSION")" +fi +[[ "$version" == nightly || "$version" =~ ^[0-9]+\.[0-9]+\.[0-9]+([-.][0-9A-Za-z.]+)?$ ]] || { + echo "error: invalid release version '$version'" >&2 + exit 1 +} + +case "$(uname -m)" in + x86_64|amd64) arch=x86_64 ;; + aarch64|arm64) arch=aarch64 ;; + *) + echo "error: unsupported architecture: $(uname -m)" >&2 + exit 1 + ;; +esac + +for command_name in cc c++ ldd readelf strip tar gzip sha256sum sort; do + command -v "$command_name" >/dev/null 2>&1 || { + echo "error: required command not found: $command_name" >&2 + exit 1 + } +done + +package_name="nemo-speech-${version}-linux-${arch}-${artifact_backend}" +archive="${output_dir}/${package_name}.tar.gz" +work_dir="$(mktemp -d "${TMPDIR:-/tmp}/nemo-speech-package.XXXXXX")" +trap 'rm -rf "$work_dir"' EXIT HUP INT TERM +package_root="$work_dir/$package_name" + +mkdir -p "$package_root" "$output_dir" +cp -a "$install_prefix"/. "$package_root"/ + +forbidden_payload="$(find "$package_root" \ + \( -name 'riva_server' -o -name 'riva_server.exe' \ + -o -name 'libgrpc*.so*' -o -path '*/riva-common' \) \ + -print -quit)" +[[ -z "$forbidden_payload" ]] || { + echo "error: release contains Riva gRPC payload: $forbidden_payload" >&2 + exit 1 +} + +sentencepiece_license_dir="$package_root/share/licenses/nemo-speech/third_party/sentencepiece" +for license_file in LICENSE absl-LICENSE darts-clone-LICENSE protobuf-lite-LICENSE; do + [[ -f "$sentencepiece_license_dir/$license_file" ]] || { + echo "error: release is missing SentencePiece notice: $license_file" >&2 + exit 1 + } +done + +# Release archives ship ITN/TN with its statically linked dependencies. +[[ -n "$(find "$package_root/lib" -maxdepth 1 -name 'libnemo_speech_text_normalization.so*' -print -quit)" ]] || { + echo "error: release is missing text normalization; configure with -DNEMO_SPEECH_WITH_NORM=ON" >&2 + exit 1 +} +for license_file in openfst/COPYING sparrowhawk/LICENSE protobuf/LICENSE re2/LICENSE; do + [[ -f "$package_root/share/licenses/nemo-speech/third_party/$license_file" ]] || { + echo "error: release is missing text normalization notice: $license_file" >&2 + exit 1 + } +done + +runtime_license_dir="$package_root/share/licenses/nemo-speech/third_party/gcc-runtime" +mkdir -p "$package_root/lib" "$runtime_license_dir" +for runtime in libstdc++.so.6 libgcc_s.so.1 libgomp.so.1 libatomic.so.1; do + # Vulkan drivers (Mesa ICDs) need the host's newer C++ runtime; a bundled + # copy loaded first through DT_RPATH would make them fail to load. + if [[ "$backend" == vulkan ]] && + [[ "$runtime" == libstdc++.so.6 || "$runtime" == libgcc_s.so.1 ]]; then + continue + fi + if [[ "$runtime" == libstdc++* ]]; then + compiler=c++ + else + compiler=cc + fi + runtime_path="$("$compiler" -print-file-name="$runtime")" + [[ "$runtime_path" != "$runtime" && -f "$runtime_path" ]] || { + echo "error: $compiler could not locate $runtime" >&2 + exit 1 + } + cp -L "$runtime_path" "$package_root/lib/$runtime" +done + +compiler_major="$(cc -dumpfullversion -dumpversion | cut -d. -f1)" +runtime_copyright="/usr/share/doc/gcc-${compiler_major}-base/copyright" +if [[ ! -f "$runtime_copyright" ]]; then + runtime_copyright="$(find /usr/share/doc -maxdepth 2 -type f \ + \( -path '*/gcc-*-base/copyright' -o -path '*/libstdc++*/copyright' \) \ + -print | sort | head -n 1)" +fi +[[ -n "$runtime_copyright" && -f "$runtime_copyright" ]] || { + echo "error: could not locate the GCC runtime copyright file" >&2 + exit 1 +} +cp "$runtime_copyright" "$runtime_license_dir/copyright" + +if [[ "$backend" == cuda ]]; then + cuda_home="${CUDA_HOME:-${CUDA_PATH:-}}" + if [[ -z "$cuda_home" ]] && command -v nvcc >/dev/null 2>&1; then + cuda_home="$(cd "$(dirname "$(command -v nvcc)")/.." && pwd)" + fi + [[ -n "$cuda_home" && -d "$cuda_home" ]] || { + echo "error: CUDA_HOME is required when packaging a CUDA build" >&2 + exit 1 + } + cuda_lib_dir= + case "$arch" in + x86_64) + cuda_targets=(x86_64-linux) + ;; + aarch64) + cuda_targets=(aarch64-linux sbsa-linux) + ;; + esac + for cuda_target in "${cuda_targets[@]}"; do + candidate="$cuda_home/targets/$cuda_target/lib" + if [[ -d "$candidate" ]]; then + cuda_lib_dir="$candidate" + break + fi + done + [[ -n "$cuda_lib_dir" ]] || { + echo "error: CUDA target libraries were not found for $arch under $cuda_home/targets" >&2 + exit 1 + } + cudart_path="$(find -L "$cuda_lib_dir" -maxdepth 1 -type f \ + -name 'libcudart.so.*' -print | sort -V | tail -n 1)" + [[ -n "$cudart_path" ]] || { + echo "error: libcudart was not found under $cuda_lib_dir" >&2 + exit 1 + } + cudart_path="$(readlink -f "$cudart_path")" + cudart_soname="$(readelf -d "$cudart_path" | + sed -n 's/.*Library soname: \[\([^]]*\)\].*/\1/p')" + [[ -n "$cudart_soname" ]] || { + echo "error: libcudart does not declare a SONAME: $cudart_path" >&2 + exit 1 + } + cp -L "$cudart_path" "$package_root/lib/$cudart_soname" + + cuda_version="$("$cuda_home/bin/nvcc" --version | + sed -n 's/.*release \([0-9]*\)\.\([0-9]*\).*/\1-\2/p' | head -n 1)" + actual_cuda_major="${cuda_version%%-*}" + if [[ "$artifact_backend" == cuda12 || "$artifact_backend" == cuda13 ]]; then + expected_cuda_major="${artifact_backend#cuda}" + [[ "$actual_cuda_major" == "$expected_cuda_major" ]] || { + echo "error: artifact label '$artifact_backend' does not match CUDA $cuda_version" >&2 + exit 1 + } + fi + cublas_soname="libcublas.so.${actual_cuda_major}" + cublas_shim="$package_root/lib/$cublas_soname" + [[ -f "$cublas_shim" ]] || { + echo "error: CUDA release does not contain the $cublas_soname shim" >&2 + exit 1 + } + actual_cublas_soname="$(readelf -d "$cublas_shim" | + sed -n 's/.*Library soname: \[\([^]]*\)\].*/\1/p')" + [[ "$actual_cublas_soname" == "$cublas_soname" ]] || { + echo "error: cuBLAS shim SONAME is '$actual_cublas_soname'; expected '$cublas_soname'" >&2 + exit 1 + } + readelf --version-info "$cublas_shim" | grep -Fq "$cublas_soname" || { + echo "error: cuBLAS shim does not export the $cublas_soname symbol version" >&2 + exit 1 + } + ggml_cuda="$(find -L "$package_root/lib" -maxdepth 1 -type f \ + -name 'libggml-cuda.so.*' -print | sort -V | tail -n 1)" + [[ -n "$ggml_cuda" ]] || { + echo "error: CUDA release does not contain libggml-cuda" >&2 + exit 1 + } + required_cublas="$(readelf -d "$ggml_cuda" | + sed -n 's/.*Shared library: \[\(libcublas\.so\.[^]]*\)\].*/\1/p')" + [[ "$required_cublas" == "$cublas_soname" ]] || { + echo "error: libggml-cuda requires '$required_cublas'; expected '$cublas_soname'" >&2 + exit 1 + } + # The shim implements only the cuBLAS calls ggml makes; a call added by a + # llama.cpp update must be added to kernels/ before it can ship. + imported_cublas="$work_dir/imported-cublas" + exported_cublas="$work_dir/exported-cublas" + nm -D --undefined-only "$ggml_cuda" > "$imported_cublas.raw" + nm -D --defined-only "$cublas_shim" > "$exported_cublas.raw" + awk '{ sub(/@.*/, "", $2); if ($2 ~ /^cublas/) print $2 }' "$imported_cublas.raw" | + sort -u > "$imported_cublas" + awk '{ sub(/@.*/, "", $3); print $3 }' "$exported_cublas.raw" | sort -u > "$exported_cublas" + [[ -s "$imported_cublas" && -s "$exported_cublas" ]] || { + echo "error: could not read the cuBLAS symbols of libggml-cuda or the shim" >&2 + exit 1 + } + missing_cublas="$(comm -23 "$imported_cublas" "$exported_cublas")" + [[ -z "$missing_cublas" ]] || { + echo "error: the cuBLAS shim does not export these functions libggml-cuda calls:" >&2 + sed 's/^/ /' <<< "$missing_cublas" >&2 + exit 1 + } + cuda_license="/usr/share/doc/cuda-cudart-${cuda_version}/copyright" + [[ -f "$cuda_license" ]] || { + echo "error: CUDA runtime license was not found: $cuda_license" >&2 + exit 1 + } + install -Dm0644 "$cuda_license" \ + "$package_root/share/licenses/nemo-speech/nvidia/cuda-runtime/copyright" +fi + +elf_candidates="$work_dir/elf-candidates" +find "$package_root/bin" "$package_root/lib" -type f -print0 > "$elf_candidates" +while IFS= read -r -d '' file; do + if readelf -h "$file" >/dev/null 2>&1; then + strip --strip-unneeded "$file" + fi +done < "$elf_candidates" + +abi_versions="$work_dir/glibc-versions" +missing_dependencies="$work_dir/missing-dependencies" +: > "$abi_versions" +: > "$missing_dependencies" +while IFS= read -r -d '' file; do + readelf -h "$file" >/dev/null 2>&1 || continue + readelf --version-info "$file" 2>/dev/null | + grep -Eo 'GLIBC_[0-9]+(\.[0-9]+)*' | + sed 's/^GLIBC_//' >> "$abi_versions" || true + LD_LIBRARY_PATH="$package_root/lib${LD_LIBRARY_PATH:+:$LD_LIBRARY_PATH}" \ + ldd "$file" 2>/dev/null | + awk -v file="$file" -v backend="$backend" \ + '$2 == "not" && $3 == "found" { + if (!(backend == "cuda" && $1 == "libcuda.so.1")) { + print file ": " $1 + } + }' \ + >> "$missing_dependencies" || true +done < "$elf_candidates" + +if [[ -s "$missing_dependencies" ]]; then + echo "error: packaged ELF dependencies are unresolved:" >&2 + sed 's/^/ /' "$missing_dependencies" >&2 + exit 1 +fi + +highest_glibc="$(sort -Vu "$abi_versions" | tail -n 1)" +[[ -n "$highest_glibc" ]] || { + echo "error: no glibc requirements were found in the package" >&2 + exit 1 +} +if [[ "$highest_glibc" != "$max_glibc" ]] && + [[ "$(printf '%s\n%s\n' "$highest_glibc" "$max_glibc" | sort -V | tail -n 1)" == "$highest_glibc" ]]; then + echo "error: package requires GLIBC_$highest_glibc; maximum is GLIBC_$max_glibc" >&2 + exit 1 +fi + +runpath_files="$work_dir/runpath-files" +: > "$runpath_files" +while IFS= read -r -d '' file; do + readelf -h "$file" >/dev/null 2>&1 || continue + if readelf -d "$file" 2>/dev/null | grep -q '(RUNPATH)'; then + echo "$file" >> "$runpath_files" + fi +done < "$elf_candidates" +if [[ -s "$runpath_files" ]]; then + echo "error: these binaries use DT_RUNPATH; link with -Wl,--disable-new-dtags:" >&2 + sed 's/^/ /' "$runpath_files" >&2 + exit 1 +fi + +source_date_epoch="${SOURCE_DATE_EPOCH:-0}" +tar --sort=name \ + --mtime="@$source_date_epoch" \ + --owner=0 --group=0 --numeric-owner \ + -C "$work_dir" -cf - "$package_name" | + gzip -n -9 > "$archive" +( + cd "$output_dir" + sha256sum "$(basename "$archive")" > "$(basename "$archive").sha256" +) + +echo "Created: $archive" +echo "SHA-256: $archive.sha256" +echo "Required: GLIBC_$highest_glibc or newer" diff --git a/scripts/release/package-macos.sh b/scripts/release/package-macos.sh new file mode 100755 index 0000000..4a4c08c --- /dev/null +++ b/scripts/release/package-macos.sh @@ -0,0 +1,170 @@ +#!/usr/bin/env bash +# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-License-Identifier: Apache-2.0 +# Package a macOS installation for the binary installer. +set -euo pipefail + +ROOT="$(cd "$(dirname "${BASH_SOURCE[0]}")/../.." && pwd)" +install_prefix= +output_dir="$ROOT/release-artifacts" +backend= +arch= +version= +min_macos=13.3 + +usage() { + cat <<'EOF' +Usage: scripts/release/package-macos.sh --install-prefix DIR --backend cpu|metal --arch aarch64|x86_64 [OPTION ...] + +Options: + --output-dir DIR Destination for the archive and checksum + --version VERSION Override the version read from VERSION ("nightly" for + the nightly channel) + --min-macos VERSION Reject binaries built for a newer macOS (default: 13.3) + -h, --help + +The project must be built with text normalization (-DNEMO_SPEECH_WITH_NORM=ON). +Every Mach-O file must be built for --arch and link only system libraries or +libraries inside the package. Binaries are stripped of local symbols and +ad-hoc signed. +EOF +} + +while [[ $# -gt 0 ]]; do + case "$1" in + --install-prefix|--backend|--arch|--output-dir|--version|--min-macos) + [[ $# -ge 2 ]] || { echo "error: $1 requires a value" >&2; exit 2; } + case "$1" in + --install-prefix) install_prefix=$2 ;; + --backend) backend=$2 ;; + --arch) arch=$2 ;; + --output-dir) output_dir=$2 ;; + --version) version=$2 ;; + --min-macos) min_macos=$2 ;; + esac + shift 2 + ;; + -h|--help) usage; exit 0 ;; + *) echo "error: unknown option: $1" >&2; usage >&2; exit 2 ;; + esac +done + +[[ -n "$install_prefix" && -x "$install_prefix/bin/nemo-speech" ]] || { + echo "error: --install-prefix must contain bin/nemo-speech" >&2 + exit 2 +} +case "$backend" in cpu|metal) ;; *) echo "error: --backend must be cpu or metal" >&2; exit 2 ;; esac +case "$arch" in + aarch64) macho_arch=arm64 ;; + x86_64) macho_arch=x86_64 ;; + *) echo "error: --arch must be aarch64 or x86_64" >&2; exit 2 ;; +esac +if [[ "$backend" == metal && "$arch" != aarch64 ]]; then + echo "error: Metal archives are built for aarch64 only" >&2 + exit 2 +fi +if [[ -z "$version" ]]; then + version="$(sed -n 's/^NEMO_SPEECH_VERSION:[[:space:]]*//p' "$ROOT/VERSION")" +fi +[[ "$version" == nightly || "$version" =~ ^[0-9]+\.[0-9]+\.[0-9]+([-.][0-9A-Za-z.]+)?$ ]] || { + echo "error: invalid release version '$version'" >&2 + exit 1 +} +for command_name in otool lipo codesign strip tar gzip shasum; do + command -v "$command_name" >/dev/null 2>&1 || { + echo "error: required command not found: $command_name" >&2 + exit 1 + } +done + +package_name="nemo-speech-${version}-macos-${arch}-${backend}" +archive="${output_dir}/${package_name}.tar.gz" +work_dir="$(mktemp -d "${TMPDIR:-/tmp}/nemo-speech-package.XXXXXX")" +trap 'rm -rf "$work_dir"' EXIT HUP INT TERM +package_root="$work_dir/$package_name" + +mkdir -p "$package_root" "$output_dir" +cp -R "$install_prefix"/. "$package_root"/ + +if find "$package_root" \( -name 'riva_server' -o -name 'libgrpc*' \) -print | grep -q .; then + echo "error: release contains Riva gRPC payload" >&2 + exit 1 +fi +sentencepiece_license_dir="$package_root/share/licenses/nemo-speech/third_party/sentencepiece" +[[ -f "$sentencepiece_license_dir/LICENSE" ]] || { + echo "error: release is missing the SentencePiece notice" >&2 + exit 1 +} + +# Release archives ship ITN/TN with its statically linked dependencies. +[[ -f "$package_root/lib/libnemo_speech_text_normalization.dylib" ]] || { + echo "error: release is missing text normalization; configure with -DNEMO_SPEECH_WITH_NORM=ON" >&2 + exit 1 +} +for license_file in openfst/COPYING sparrowhawk/LICENSE protobuf/LICENSE re2/LICENSE; do + [[ -f "$package_root/share/licenses/nemo-speech/third_party/$license_file" ]] || { + echo "error: release is missing text normalization notice: $license_file" >&2 + exit 1 + } +done + +version_le() { # version_le A B: A <= B + [[ "$(printf '%s\n%s\n' "$1" "$2" | sort -t. -k1,1n -k2,2n -k3,3n | head -n 1)" == "$1" ]] +} + +failures="$work_dir/failures" +: > "$failures" +macho_files="$work_dir/macho-files" +find "$package_root/bin" "$package_root/lib" -type f -print | LC_ALL=C sort > "$macho_files" +while IFS= read -r file; do + # lipo, not otool: llvm-otool exits 0 for files that are not Mach-O. + archs="$(lipo -archs "$file" 2>/dev/null)" || continue + [[ "$archs" == "$macho_arch" ]] || echo "$file: built for '$archs', expected $macho_arch" >> "$failures" + + minos="$(otool -l "$file" | awk '$1 == "minos" { print $2; exit }')" + if [[ -n "$minos" ]] && ! version_le "$minos" "$min_macos"; then + echo "$file: requires macOS $minos, newer than $min_macos" >> "$failures" + fi + + # Dependencies: system libraries or @rpath/@loader_path entries present in the package. + otool -L "$file" | tail -n +2 | awk '{ print $1 }' | while IFS= read -r dependency; do + case "$dependency" in + /usr/lib/*|/System/Library/*) ;; + @rpath/*|@loader_path/*|@executable_path/*) + name="${dependency##*/}" + [[ -e "$package_root/lib/$name" || -e "$package_root/bin/$name" ]] || + echo "$file: $dependency is not in the package" >> "$failures" + ;; + *) echo "$file: links $dependency, which is outside the package" >> "$failures" ;; + esac + done +done < "$macho_files" + +if [[ -s "$failures" ]]; then + echo "error: the package is not self-contained:" >&2 + sed 's/^/ /' "$failures" >&2 + exit 1 +fi + +# Strip local symbols, then ad-hoc sign: install-time rpath edits invalidate the +# linker's signature, and arm64 macOS refuses to run unsigned code. +while IFS= read -r file; do + lipo -archs "$file" >/dev/null 2>&1 || continue + strip -x "$file" + codesign --force --sign - "$file" +done < "$macho_files" + +source_date_epoch="${SOURCE_DATE_EPOCH:-0}" +stamp="$(date -u -r "$source_date_epoch" +%Y%m%d%H%M.%S)" +find "$package_root" -exec touch -h -t "$stamp" {} + +(cd "$work_dir" && find "$package_name" -print | LC_ALL=C sort > "$work_dir/files") +COPYFILE_DISABLE=1 tar --no-mac-metadata --no-xattrs --uid 0 --gid 0 --uname root --gname wheel \ + -C "$work_dir" -n -T "$work_dir/files" -cf - | gzip -n -9 > "$archive" +( + cd "$output_dir" + shasum -a 256 "$(basename "$archive")" > "$(basename "$archive").sha256" +) + +echo "Created: $archive" +echo "SHA-256: $archive.sha256" +echo "Requires: macOS $min_macos or newer ($macho_arch)" diff --git a/scripts/windows/build.ps1 b/scripts/windows/build.ps1 index 07a7bfa..012f2bb 100644 --- a/scripts/windows/build.ps1 +++ b/scripts/windows/build.ps1 @@ -77,6 +77,9 @@ .PARAMETER Architecture Target architecture: auto (the host), x64, or arm64. +.PARAMETER CMakeArgs + Extra arguments appended to the CMake configure command, for example + -CMakeArgs '-DGGML_NATIVE=OFF'. .EXAMPLE pwsh scripts\windows\build.ps1 -Backend cuda -Profile server @@ -112,7 +115,8 @@ param( [ValidateSet('auto', 'msvc', 'clang-cl')] [string]$Compiler = 'auto', [int]$Jobs = 0, - [switch]$DryRun + [switch]$DryRun, + [string[]]$CMakeArgs = @() ) $ErrorActionPreference = 'Stop' @@ -377,7 +381,7 @@ function ConvertTo-CMakeBool([bool]$Value) { return 'OFF' } -$cmakeArgs = @( +$configureArgs = @( '-S', $RepoRoot, '-B', $BuildDir, '-G', 'Ninja', "-DCMAKE_BUILD_TYPE=$Config", "-DNEMO_SPEECH_BUILD_ASR=$(ConvertTo-CMakeBool $BuildAsr)", "-DNEMO_SPEECH_BUILD_DIAR=$(ConvertTo-CMakeBool $BuildDiar)", @@ -396,55 +400,56 @@ $cmakeArgs = @( "-DNEMO_SPEECH_BUILD_TOOLS=$(ConvertTo-CMakeBool $BuildTools)" ) if ($VcpkgFeatures.Count -gt 0) { - $cmakeArgs += "-DCMAKE_TOOLCHAIN_FILE=$toolchain" - $cmakeArgs += "-DVCPKG_TARGET_TRIPLET=$VcpkgTriplet" - $cmakeArgs += "-DVCPKG_MANIFEST_FEATURES=$($VcpkgFeatures -join ';')" - $cmakeArgs += "-DVCPKG_INSTALLED_DIR=$(Join-Path $BuildDir 'vcpkg_installed')" + $configureArgs += "-DCMAKE_TOOLCHAIN_FILE=$toolchain" + $configureArgs += "-DVCPKG_TARGET_TRIPLET=$VcpkgTriplet" + $configureArgs += "-DVCPKG_MANIFEST_FEATURES=$($VcpkgFeatures -join ';')" + $configureArgs += "-DVCPKG_INSTALLED_DIR=$(Join-Path $BuildDir 'vcpkg_installed')" } if ($Compiler -eq 'clang-cl') { - $cmakeArgs += '-DCMAKE_C_COMPILER=clang-cl' - $cmakeArgs += '-DCMAKE_CXX_COMPILER=clang-cl' + $configureArgs += '-DCMAKE_C_COMPILER=clang-cl' + $configureArgs += '-DCMAKE_CXX_COMPILER=clang-cl' if ($CrossCompiling) { $llvmTarget = if ($TargetArch -eq 'arm64') { 'arm64-pc-windows-msvc' } else { 'x86_64-pc-windows-msvc' } - $cmakeArgs += "-DCMAKE_C_COMPILER_TARGET=$llvmTarget" - $cmakeArgs += "-DCMAKE_CXX_COMPILER_TARGET=$llvmTarget" - $cmakeArgs += '-DCMAKE_SYSTEM_NAME=Windows' - $cmakeArgs += "-DCMAKE_SYSTEM_PROCESSOR=$TargetArch" + $configureArgs += "-DCMAKE_C_COMPILER_TARGET=$llvmTarget" + $configureArgs += "-DCMAKE_CXX_COMPILER_TARGET=$llvmTarget" + $configureArgs += '-DCMAKE_SYSTEM_NAME=Windows' + $configureArgs += "-DCMAKE_SYSTEM_PROCESSOR=$TargetArch" } if ($Backend -eq 'cuda') { # nvcc only supports cl.exe as its host compiler on Windows; pin it # explicitly so it never inherits clang-cl. - $cmakeArgs += '-DCMAKE_CUDA_HOST_COMPILER=cl' + $configureArgs += '-DCMAKE_CUDA_HOST_COMPILER=cl' } } if ($TargetArch -eq 'arm64') { # Avoid an additional OpenMP runtime DLL in ARM64 packages. - $cmakeArgs += '-DGGML_OPENMP=OFF' + $configureArgs += '-DGGML_OPENMP=OFF' } switch ($Backend) { 'cuda' { - $cmakeArgs += '-DGGML_CUDA=ON' - $cmakeArgs += '-DGGML_VULKAN=OFF' - $cmakeArgs += "-DNEMO_SPEECH_CUBLAS_SHIM=$(ConvertTo-CMakeBool $CublasShim.IsPresent)" - $cmakeArgs += "-DCMAKE_CUDA_ARCHITECTURES=$CudaArch" + $configureArgs += '-DGGML_CUDA=ON' + $configureArgs += '-DGGML_VULKAN=OFF' + $configureArgs += "-DNEMO_SPEECH_CUBLAS_SHIM=$(ConvertTo-CMakeBool $CublasShim.IsPresent)" + $configureArgs += "-DCMAKE_CUDA_ARCHITECTURES=$CudaArch" } 'vulkan' { - $cmakeArgs += '-DGGML_CUDA=OFF' - $cmakeArgs += '-DGGML_VULKAN=ON' - $cmakeArgs += '-DNEMO_SPEECH_GGML_PATCHED=OFF' + $configureArgs += '-DGGML_CUDA=OFF' + $configureArgs += '-DGGML_VULKAN=ON' + $configureArgs += '-DNEMO_SPEECH_GGML_PATCHED=OFF' # ggml-vulkan hard-requires the SPIRV-Headers CMake package; the Vulkan # SDK ships its config, but not on CMake's default search path. $spirvDir = Join-Path $env:VULKAN_SDK 'Lib\cmake\SPIRV-Headers' - if (Test-Path $spirvDir) { $cmakeArgs += "-DSPIRV-Headers_DIR=$spirvDir" } + if (Test-Path $spirvDir) { $configureArgs += "-DSPIRV-Headers_DIR=$spirvDir" } } 'cpu' { - $cmakeArgs += '-DGGML_CUDA=OFF' - $cmakeArgs += '-DGGML_VULKAN=OFF' + $configureArgs += '-DGGML_CUDA=OFF' + $configureArgs += '-DGGML_VULKAN=OFF' } } -Write-Host "==> cmake $($cmakeArgs -join ' ')" -& cmake @cmakeArgs +$configureArgs += $CMakeArgs +Write-Host "==> cmake $($configureArgs -join ' ')" +& cmake @configureArgs if ($LASTEXITCODE -ne 0) { throw "CMake configure failed ($LASTEXITCODE)" } Write-Host "==> building" diff --git a/scripts/windows/package-release.ps1 b/scripts/windows/package-release.ps1 new file mode 100644 index 0000000..23f0764 --- /dev/null +++ b/scripts/windows/package-release.ps1 @@ -0,0 +1,139 @@ +# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-License-Identifier: Apache-2.0 +<# +.SYNOPSIS + Package a Windows build for the binary installer. +.DESCRIPTION + Installs -BuildDir into a directory named after the archive, checks that the + package is self-contained, and writes + nemo-speech--windows--.zip and its .sha256 to + -OutputDir. Every DLL a packaged binary imports must ship in bin\ or be a + Windows or GPU driver library. +.PARAMETER BuildDir + Build tree produced by scripts\windows\build.ps1. +.PARAMETER Backend + cpu, vulkan, or cuda. +.PARAMETER Version + Release version, or nightly. Defaults to the one in VERSION. +.PARAMETER OutputDir + Destination for the archive and checksum (default: release-artifacts). +.EXAMPLE + pwsh scripts\windows\package-release.ps1 -BuildDir build\release-cuda -Backend cuda +#> +[CmdletBinding()] +param( + [Parameter(Mandatory)] + [string]$BuildDir, + [Parameter(Mandatory)] + [ValidateSet('cpu', 'vulkan', 'cuda')] + [string]$Backend, + [string]$Version, + [string]$OutputDir +) + +$ErrorActionPreference = 'Stop' +$RepoRoot = Split-Path -Parent (Split-Path -Parent $PSScriptRoot) +if (-not $OutputDir) { $OutputDir = Join-Path $RepoRoot 'release-artifacts' } +if (-not $Version) { + $Version = ((Get-Content (Join-Path $RepoRoot 'VERSION')) -match '^NEMO_SPEECH_VERSION:' | + Select-Object -First 1) -replace '^NEMO_SPEECH_VERSION:\s*', '' +} +if ($Version -ne 'nightly' -and $Version -notmatch '^[0-9]+\.[0-9]+\.[0-9]+([-.][0-9A-Za-z.]+)?$') { + throw "invalid release version '$Version'" +} +$arch = switch ([System.Runtime.InteropServices.RuntimeInformation]::OSArchitecture) { + 'X64' { 'x86_64' } + 'Arm64' { 'aarch64' } + default { throw "unsupported architecture: $_" } +} + +$name = "nemo-speech-$Version-windows-$arch-$Backend" +$staging = Join-Path ([System.IO.Path]::GetTempPath()) "nemo-speech-package-$PID" +$root = Join-Path $staging $name +New-Item -ItemType Directory -Force -Path $root, $OutputDir | Out-Null +try { + & cmake --install $BuildDir --config Release --prefix $root + if ($LASTEXITCODE -ne 0) { throw "cmake --install failed ($LASTEXITCODE)" } + + $bin = Join-Path $root 'bin' + foreach ($required in @('bin\nemo-speech.exe', 'share\licenses\nemo-speech\LICENSE', + 'share\licenses\nemo-speech\THIRD_PARTY_NOTICES.md')) { + if (-not (Test-Path (Join-Path $root $required))) { throw "package is missing $required" } + } + if (Get-ChildItem -Recurse $root -Include 'riva_server.exe', 'grpc*.dll') { + throw 'release contains Riva gRPC payload' + } + if ($Backend -eq 'cuda') { + if (-not (Get-ChildItem $bin -Filter 'ggml-cuda.dll')) { throw 'CUDA package is missing ggml-cuda.dll' } + if (-not (Get-ChildItem $bin -Filter 'cublas64_*.dll')) { + throw 'CUDA package is missing the cuBLAS shim; build with -CublasShim' + } + } + + # Every imported DLL must be bundled or provided by Windows or the GPU driver. + $vswhere = "${env:ProgramFiles(x86)}\Microsoft Visual Studio\Installer\vswhere.exe" + $dumpbin = & $vswhere -latest -products * -find '**\Hostx64\x64\dumpbin.exe' | Select-Object -First 1 + if (-not $dumpbin) { throw 'dumpbin.exe was not found (Visual Studio C++ tools are required)' } + # A failed inspection would otherwise read as "no dependencies". + function Invoke-Dumpbin([string]$Option, [string]$Path) { + $output = & $dumpbin /nologo $Option $Path + if ($LASTEXITCODE -ne 0) { throw "dumpbin $Option failed for $Path ($LASTEXITCODE)" } + $output + } + $system = '^(api-ms-win-.*|ext-ms-.*|kernel32|kernelbase|user32|gdi32|advapi32|shell32|ole32|oleaut32|' + + 'ws2_32|wsock32|mswsock|bcrypt|crypt32|ncrypt|secur32|dbghelp|shlwapi|winmm|ntdll|userenv|' + + 'psapi|version|iphlpapi|setupapi|cfgmgr32|comctl32|comdlg32|rpcrt4|dnsapi|powrprof|winhttp|' + + 'wininet|normaliz|avrt|ucrtbase|nvcuda|nvml|vulkan-1)\.dll$' + $bundled = @{} + Get-ChildItem $bin -Filter '*.dll' | ForEach-Object { $bundled[$_.Name.ToLowerInvariant()] = $true } + $missing = @() + foreach ($file in Get-ChildItem $bin -Include '*.dll', '*.exe' -Recurse) { + $dependencies = Invoke-Dumpbin /dependents $file.FullName | ForEach-Object { + if ($_ -match '^\s+(\S+\.dll)\s*$') { $Matches[1].ToLowerInvariant() } + } + foreach ($dependency in $dependencies) { + if (-not $bundled.ContainsKey($dependency) -and $dependency -notmatch $system) { + $missing += "$($file.Name): $dependency" + } + } + } + if ($missing) { + throw "the package is not self-contained:`n $($missing -join "`n ")" + } + + # The cuBLAS shim implements only the calls ggml makes; a call added by a + # llama.cpp update must be added to kernels\ before it can ship. + if ($Backend -eq 'cuda') { + $shim = (Get-ChildItem $bin -Filter 'cublas64_*.dll' | Select-Object -First 1) + $exported = @{} + Invoke-Dumpbin /exports $shim.FullName | ForEach-Object { + if ($_ -match '^\s+\d+\s+[0-9A-Fa-f]+\s+[0-9A-Fa-f]+\s+(\S+)') { $exported[$Matches[1]] = $true } + } + $dll = $null + $unexported = @() + Invoke-Dumpbin /imports (Join-Path $bin 'ggml-cuda.dll') | ForEach-Object { + if ($_ -match '^\s{4}(\S+\.dll)\s*$') { + $dll = $Matches[1] + } elseif ($dll -ieq $shim.Name -and $_ -match '^\s+[0-9A-Fa-f]+\s+(\S+)\s*$' -and + -not $exported.ContainsKey($Matches[1])) { + $unexported += $Matches[1] + } + } + if (-not $exported.Count) { throw "dumpbin listed no exports for $($shim.Name)" } + if ($unexported) { + throw "the cuBLAS shim does not export these functions ggml-cuda.dll calls:`n $($unexported -join "`n ")" + } + } + + $zip = Join-Path (Resolve-Path $OutputDir) "$name.zip" + if (Test-Path $zip) { Remove-Item $zip } + Add-Type -AssemblyName System.IO.Compression.FileSystem + [System.IO.Compression.ZipFile]::CreateFromDirectory( + $root, $zip, [System.IO.Compression.CompressionLevel]::Optimal, $true) + $hash = (Get-FileHash -Algorithm SHA256 $zip).Hash.ToLowerInvariant() + [System.IO.File]::WriteAllText("$zip.sha256", "$hash $name.zip`n") + Write-Host "Created: $zip" + Write-Host "SHA-256: $zip.sha256" +} finally { + Remove-Item -Recurse -Force -ErrorAction SilentlyContinue $staging +} diff --git a/src/asr/CMakeLists.txt b/src/asr/CMakeLists.txt index e140856..7c09ea0 100644 --- a/src/asr/CMakeLists.txt +++ b/src/asr/CMakeLists.txt @@ -63,8 +63,8 @@ target_link_libraries(nemo_speech_asr PUBLIC nemo_speech_runtime_ggml) # SentencePiece is a core ASR dependency: the RNNT head tokenizes word-boosting # phrases with the model's embedded tokenizer (context biasing in # rnnt_greedy_decoder.cpp / RnntModel::encode_phrase); the flashlight decoder -# also encodes OOV boost phrases with it. Prefer a static archive on ELF -# platforms so its bundled protobuf symbols stay private. +# also encodes OOV boost phrases with it. Prefer a static archive on ELF and +# Apple platforms so its bundled protobuf symbols stay private. if(UNIX AND NOT APPLE) set(_NEMO_SPEECH_LIBRARY_SUFFIXES "${CMAKE_FIND_LIBRARY_SUFFIXES}") set(CMAKE_FIND_LIBRARY_SUFFIXES ".a") @@ -72,6 +72,11 @@ if(UNIX AND NOT APPLE) HINTS "${NEMO_SPEECH_DEPENDENCY_PREFIX}/sentencepiece/lib") set(CMAKE_FIND_LIBRARY_SUFFIXES "${_NEMO_SPEECH_LIBRARY_SUFFIXES}") unset(_NEMO_SPEECH_LIBRARY_SUFFIXES) +elseif(APPLE) + # Only the project-built archive: Homebrew's libsentencepiece.a needs its + # shared abseil, which release archives cannot carry. + find_library(SENTENCEPIECE_STATIC_LIB libsentencepiece.a + PATHS "${NEMO_SPEECH_DEPENDENCY_PREFIX}/sentencepiece/lib" NO_DEFAULT_PATH) endif() if(SENTENCEPIECE_STATIC_LIB) find_path(SENTENCEPIECE_INCLUDE_DIR sentencepiece_processor.h @@ -79,10 +84,15 @@ if(SENTENCEPIECE_STATIC_LIB) if(NOT SENTENCEPIECE_INCLUDE_DIR) message(FATAL_ERROR "SentencePiece archive found without sentencepiece_processor.h") endif() - target_link_libraries(nemo_speech_asr PRIVATE ${SENTENCEPIECE_STATIC_LIB}) target_include_directories(nemo_speech_asr PRIVATE ${SENTENCEPIECE_INCLUDE_DIR}) - target_link_options( - nemo_speech_asr PRIVATE "LINKER:--exclude-libs,libsentencepiece.a") + if(APPLE) + target_link_options( + nemo_speech_asr PRIVATE "LINKER:-load_hidden,${SENTENCEPIECE_STATIC_LIB}") + else() + target_link_libraries(nemo_speech_asr PRIVATE ${SENTENCEPIECE_STATIC_LIB}) + target_link_options( + nemo_speech_asr PRIVATE "LINKER:--exclude-libs,libsentencepiece.a") + endif() elseif(NEMO_SPEECH_WITH_NORM AND UNIX AND NOT APPLE) message(FATAL_ERROR "normalization requires private static SentencePiece; " diff --git a/src/common/CMakeLists.txt b/src/common/CMakeLists.txt index 9b9a18d..2ab30c0 100644 --- a/src/common/CMakeLists.txt +++ b/src/common/CMakeLists.txt @@ -23,9 +23,11 @@ set_target_properties(nemo_speech_common PROPERTIES POSITION_INDEPENDENT_CODE ON # Shared WFST grammar runtime for ASR ITN and TTS TN. Keep this separate from # nemo_speech_common so builds that do not request text normalization retain -# no OpenFST/Sparrowhawk dependency. +# no OpenFST/Sparrowhawk dependency. It is a shared library of its own so that +# its protobuf runtime stays out of the ASR library, which links SentencePiece's +# bundled protobuf, and so ASR and TTS share one copy. if(NEMO_SPEECH_WITH_NORM) - add_library(nemo_speech_text_normalization STATIC + add_library(nemo_speech_text_normalization SHARED text_normalization/fst_normalizer.cpp) target_compile_definitions( nemo_speech_text_normalization PUBLIC NEMO_SPEECH_WITH_NORM=1) @@ -34,46 +36,70 @@ if(NEMO_SPEECH_WITH_NORM) find_library(SPARROWHAWK_LIB sparrowhawk HINTS "${_NEMO_SPEECH_ITN_PREFIX}/lib" REQUIRED) find_library(FSTFAR_LIB fstfar HINTS "${_NEMO_SPEECH_ITN_PREFIX}/lib" REQUIRED) find_library(FST_LIB fst HINTS "${_NEMO_SPEECH_ITN_PREFIX}/lib" REQUIRED) - find_library(RE2_LIB re2 REQUIRED) + find_library(RE2_LIB re2 HINTS "${_NEMO_SPEECH_ITN_PREFIX}/lib" REQUIRED) find_path(SPARROWHAWK_INCLUDE_DIR sparrowhawk/normalizer.h HINTS "${_NEMO_SPEECH_ITN_PREFIX}/include" REQUIRED) find_path(OPENFST_INCLUDE_DIR fst/fst.h HINTS "${_NEMO_SPEECH_ITN_PREFIX}/include" REQUIRED) + find_package(Threads REQUIRED) - find_package(Protobuf REQUIRED) - find_library(ITN_PROTOBUF_LIB - NAMES libprotobuf.so.25.3.0 libprotobuf.so.25 - PATHS /lib /lib/x86_64-linux-gnu /usr/lib /usr/lib/x86_64-linux-gnu - NO_DEFAULT_PATH) - if(NOT ITN_PROTOBUF_LIB) - if(TARGET protobuf::libprotobuf) - set(ITN_PROTOBUF_LIB protobuf::libprotobuf) - else() - set(ITN_PROTOBUF_LIB protobuf) + # scripts/build_itn_deps.sh with STATIC=1 installs protobuf and RE2 into + # the same prefix; otherwise use the system protobuf. + find_library(ITN_PROTOBUF_LIB protobuf + PATHS "${_NEMO_SPEECH_ITN_PREFIX}/lib" NO_DEFAULT_PATH) + set(ABSL_LIBS) + if(ITN_PROTOBUF_LIB) + set(ITN_PROTOBUF_LIBS ${ITN_PROTOBUF_LIB}) + else() + find_package(Protobuf REQUIRED) + find_library(ITN_PROTOBUF_LIB + NAMES libprotobuf.so.25.3.0 libprotobuf.so.25 + PATHS /lib /lib/x86_64-linux-gnu /usr/lib /usr/lib/x86_64-linux-gnu + NO_DEFAULT_PATH) + if(NOT ITN_PROTOBUF_LIB) + if(TARGET protobuf::libprotobuf) + set(ITN_PROTOBUF_LIB protobuf::libprotobuf) + else() + set(ITN_PROTOBUF_LIB protobuf) + endif() + endif() + find_library(UTF8_RANGE_LIB NAMES utf8_range_lib PATHS /usr/lib /lib) + file(GLOB ABSL_LIBS /usr/lib/libabsl_*.so /lib/libabsl_*.so) + set(ITN_PROTOBUF_LIBS ${ITN_PROTOBUF_LIB}) + if(UTF8_RANGE_LIB) + list(APPEND ITN_PROTOBUF_LIBS ${UTF8_RANGE_LIB}) endif() - endif() - find_library(UTF8_RANGE_LIB NAMES utf8_range_lib PATHS /usr/lib /lib) - file(GLOB ABSL_LIBS /usr/lib/libabsl_*.so /lib/libabsl_*.so) - set(ITN_PROTOBUF_LIBS ${ITN_PROTOBUF_LIB}) - if(UTF8_RANGE_LIB) - list(APPEND ITN_PROTOBUF_LIBS ${UTF8_RANGE_LIB}) endif() target_include_directories(nemo_speech_text_normalization PUBLIC ${CMAKE_CURRENT_SOURCE_DIR}/text_normalization + PRIVATE ${CMAKE_CURRENT_SOURCE_DIR}/text_normalization/compat) target_include_directories(nemo_speech_text_normalization SYSTEM - PUBLIC - ${OPENFST_INCLUDE_DIR} PRIVATE + ${OPENFST_INCLUDE_DIR} ${SPARROWHAWK_INCLUDE_DIR}) - target_link_libraries(nemo_speech_text_normalization PUBLIC - ${SPARROWHAWK_LIB} ${FSTFAR_LIB} ${FST_LIB} - ${ITN_PROTOBUF_LIBS} - -Wl,--as-needed ${ABSL_LIBS} -Wl,--no-as-needed - ${RE2_LIB}) - set_target_properties( - nemo_speech_text_normalization PROPERTIES POSITION_INDEPENDENT_CODE ON) + # OpenFST registers its FST types from static initializers; keep them when + # linking the static archive. + set(_NEMO_SPEECH_FST_LINK ${FST_LIB}) + if(FST_LIB MATCHES "\\${CMAKE_STATIC_LIBRARY_SUFFIX}$") + set(_NEMO_SPEECH_FST_LINK "$") + endif() + target_link_libraries(nemo_speech_text_normalization PRIVATE + ${SPARROWHAWK_LIB} ${FSTFAR_LIB} ${_NEMO_SPEECH_FST_LINK} + ${ITN_PROTOBUF_LIBS}) + if(ABSL_LIBS) + target_link_libraries(nemo_speech_text_normalization PRIVATE + -Wl,--as-needed ${ABSL_LIBS} -Wl,--no-as-needed) + endif() + target_link_libraries(nemo_speech_text_normalization PRIVATE + ${RE2_LIB} Threads::Threads ${CMAKE_DL_LIBS}) + if(UNIX AND NOT APPLE) + # Keep the symbols of statically linked dependencies private. + target_link_options(nemo_speech_text_normalization PRIVATE "LINKER:--exclude-libs,ALL") + endif() + install(TARGETS nemo_speech_text_normalization + LIBRARY DESTINATION ${CMAKE_INSTALL_LIBDIR}) file(GLOB _NEMO_SPEECH_NORM_RUNTIME_LIBS "${_NEMO_SPEECH_ITN_PREFIX}/lib/libsparrowhawk${CMAKE_SHARED_LIBRARY_SUFFIX}*" @@ -84,4 +110,10 @@ if(NEMO_SPEECH_WITH_NORM) "${_NEMO_SPEECH_ITN_PREFIX}/lib/libfstfar.*${CMAKE_SHARED_LIBRARY_SUFFIX}") install(FILES ${_NEMO_SPEECH_NORM_RUNTIME_LIBS} DESTINATION ${CMAKE_INSTALL_LIBDIR}) + set(_NEMO_SPEECH_ITN_LICENSE_DIR + "${_NEMO_SPEECH_ITN_PREFIX}/share/licenses/nemo-speech/third_party") + if(EXISTS "${_NEMO_SPEECH_ITN_LICENSE_DIR}") + install(DIRECTORY "${_NEMO_SPEECH_ITN_LICENSE_DIR}/" + DESTINATION "${NEMO_SPEECH_LICENSE_INSTALL_DIR}/third_party") + endif() endif() diff --git a/src/tts/preproc/text_normalizer.cpp b/src/tts/preproc/text_normalizer.cpp index d2a49a0..1a95367 100644 --- a/src/tts/preproc/text_normalizer.cpp +++ b/src/tts/preproc/text_normalizer.cpp @@ -9,6 +9,8 @@ #include #ifdef NEMO_SPEECH_WITH_NORM +#include // mkdtemp (macOS declares it only here) + #include #include #include diff --git a/tests/ci/model_smoke.py b/tests/ci/model_smoke.py index 6dde2fb..07b2130 100644 --- a/tests/ci/model_smoke.py +++ b/tests/ci/model_smoke.py @@ -3,6 +3,9 @@ # SPDX-License-Identifier: Apache-2.0 """CLI smoke test with real models: ASR, diarization, and a TTS round trip. +With grammar directories, it also checks text normalization: TN must turn digits +into words before synthesis, and ITN must turn them back in the transcript. + Models are pulled by the CLI from the indexed Hugging Face repos into NEMO_SPEECH_MODEL_DIR, so a warm cache makes this cheap. Text comparisons use a similarity ratio rather than exact match so backend numerics cannot flake it. @@ -24,6 +27,8 @@ "ask what you can do for your country" ) TTS_TEXT = "The quick brown fox jumps over the lazy dog." +NORM_TEXT = "I paid 25 dollars for 3 tickets." +NORM_SPOKEN = "i paid twenty five dollars for three tickets" FIXTURES = pathlib.Path(__file__).resolve().parents[2] / "test_files" / "diar" @@ -126,16 +131,36 @@ def main() -> None: parser.add_argument("--backend", default="cpu") parser.add_argument("--audio", required=True) parser.add_argument("--skip-tts", action="store_true") + parser.add_argument("--itn-model-dir", help="ITN grammar directory") + parser.add_argument("--tn-model-dir", help="TN grammar directory") args = parser.parse_args() with tempfile.TemporaryDirectory(prefix="nemo-speech-smoke-") as temporary: work = pathlib.Path(temporary) + if args.itn_model_dir or args.tn_model_dir: + features = json.loads(run(args.binary, "doctor", "--json"))["features"] + check(features.get("text_normalization") is True, "build includes text normalization") + text = run(args.binary, "transcribe", args.audio, "--backend", args.backend) ratio = similarity(text, JFK_TEXT) print(f"asr transcript: {text.strip()!r} (similarity {ratio:.2f})") check(ratio >= 0.9, "ASR offline transcript matches the reference") + if args.itn_model_dir: + text = run( + args.binary, + "transcribe", + args.audio, + "--backend", + args.backend, + "--itn-model-dir", + args.itn_model_dir, + ) + ratio = similarity(text, JFK_TEXT) + print(f"asr transcript with itn: {text.strip()!r} (similarity {ratio:.2f})") + check(ratio >= 0.9, "ASR transcript with ITN matches the reference") + streamed = run( args.binary, "transcribe", @@ -219,6 +244,36 @@ def main() -> None: print(f"tts round trip: {round_trip.strip()!r} (similarity {ratio:.2f})") check(ratio >= 0.8, "TTS output transcribes back to the input text") + if args.tn_model_dir: + wav = work / "tn.wav" + run( + args.binary, + "synthesize", + NORM_TEXT, + "--backend", + args.backend, + "--tn-model-dir", + args.tn_model_dir, + "--output", + str(wav), + ) + spoken = run(args.binary, "transcribe", str(wav), "--backend", args.backend) + ratio = similarity(spoken, NORM_SPOKEN) + print(f"tn round trip: {spoken.strip()!r} (similarity {ratio:.2f})") + check(ratio >= 0.8, "TN spells out numbers before synthesis") + if args.itn_model_dir: + written = run( + args.binary, + "transcribe", + str(wav), + "--backend", + args.backend, + "--itn-model-dir", + args.itn_model_dir, + ) + print(f"itn round trip: {written.strip()!r}") + check("$25" in written, "ITN writes spoken numbers in written form") + if __name__ == "__main__": main() diff --git a/tests/cpp/test_cublas_shim_numerics.cu b/tests/cpp/test_cublas_shim_numerics.cu index 20ad716..7fb2a16 100644 --- a/tests/cpp/test_cublas_shim_numerics.cu +++ b/tests/cpp/test_cublas_shim_numerics.cu @@ -167,6 +167,204 @@ run_batched_f32_case(cublasHandle_t handle) { return true; } +// Pointer-array f32 GEMM with a transposed B, as ggml's OUT_PROD issues it. +bool +run_pointer_batched_f32_case(cublasHandle_t handle) { + constexpr int m = 37; + constexpr int n = 29; + constexpr int k = 13; + constexpr int batch = 5; + constexpr size_t size_a = (size_t)m * k; // m x k, column-major + constexpr size_t size_b = (size_t)n * k; // n x k, used transposed + constexpr size_t size_c = (size_t)m * n; + + std::vector a(batch * size_a), b(batch * size_b), c(batch * size_c); + for (size_t i = 0; i < a.size(); ++i) a[i] = (float)((i * 7) % 11) - 5.0f; + for (size_t i = 0; i < b.size(); ++i) b[i] = (float)((i * 5) % 13) - 6.0f; + + float* device = nullptr; + const float** device_ptrs = nullptr; + const size_t total = a.size() + b.size() + c.size(); + bool ok = check_cuda(cudaMalloc(&device, total * sizeof(float)), "cudaMalloc(batched)") && + check_cuda(cudaMalloc(&device_ptrs, 3 * batch * sizeof(float*)), "cudaMalloc(ptrs)"); + float* device_a = device; + float* device_b = device_a + a.size(); + float* device_c = device_b + b.size(); + std::vector ptrs(3 * batch); + for (int item = 0; item < batch; ++item) { + ptrs[item] = device_a + item * size_a; + ptrs[batch + item] = device_b + item * size_b; + ptrs[2 * batch + item] = device_c + item * size_c; + } + ok = ok && + check_cuda( + cudaMemcpy(device_a, a.data(), a.size() * sizeof(float), cudaMemcpyHostToDevice), + "cudaMemcpy(A batched)") && + check_cuda( + cudaMemcpy(device_b, b.data(), b.size() * sizeof(float), cudaMemcpyHostToDevice), + "cudaMemcpy(B batched)") && + check_cuda( + cudaMemcpy( + device_ptrs, ptrs.data(), ptrs.size() * sizeof(float*), cudaMemcpyHostToDevice), + "cudaMemcpy(ptrs)"); + + const float alpha = 1.0f; + const float beta = 0.0f; + if (ok) { + ok = check_cublas( + cublasSgemmBatched( + handle, CUBLAS_OP_N, CUBLAS_OP_T, m, n, k, &alpha, device_ptrs, m, + device_ptrs + batch, n, &beta, (float* const*)(device_ptrs + 2 * batch), m, batch), + "cublasSgemmBatched"); + } + if (ok) { + ok = check_cuda( + cudaMemcpy(c.data(), device_c, c.size() * sizeof(float), cudaMemcpyDeviceToHost), + "cudaMemcpy(C batched)"); + } + cudaFree(device); + cudaFree(device_ptrs); + if (!ok) { + return false; + } + for (int item = 0; item < batch; ++item) { + for (int col = 0; col < n; ++col) { + for (int row = 0; row < m; ++row) { + float expected = 0.0f; + for (int i = 0; i < k; ++i) { + expected += a[item * size_a + (size_t)i * m + row] * + b[item * size_b + (size_t)i * n + col]; + } + const float actual = c[item * size_c + (size_t)col * m + row]; + if (!std::isfinite(actual) || std::fabs(actual - expected) > 1.0e-3f) { + std::fprintf( + stderr, "FAIL: pointer-batched f32 [%d,%d,%d]=%g, expected %g\n", item, row, + col, actual, expected); + return false; + } + } + } + } + return true; +} + +// More batches than CUDA's grid.z limit (65535), and a zero-sized no-op. +bool +run_large_batch_case(cublasHandle_t handle) { + constexpr int batch = 70000; + constexpr int dim = 2; + constexpr size_t size = (size_t)dim * dim; + std::vector a(batch * size), b(batch * size), c(batch * size); + for (int item = 0; item < batch; ++item) { + for (size_t i = 0; i < size; ++i) { + a[item * size + i] = (float)(item % 7) + (float)i; + b[item * size + i] = i % 3 == 0 ? 1.0f : 0.5f; + } + } + + float* device = nullptr; + const float** device_ptrs = nullptr; + bool ok = check_cuda(cudaMalloc(&device, 3 * a.size() * sizeof(float)), "cudaMalloc(large)") && + check_cuda(cudaMalloc(&device_ptrs, 3 * batch * sizeof(float*)), "cudaMalloc(ptrs)"); + float* device_a = device; + float* device_b = device_a + a.size(); + float* device_c = device_b + b.size(); + std::vector ptrs(3 * batch); + for (int item = 0; item < batch; ++item) { + ptrs[item] = device_a + item * size; + ptrs[batch + item] = device_b + item * size; + ptrs[2 * batch + item] = device_c + item * size; + } + ok = ok && + check_cuda( + cudaMemcpy(device_a, a.data(), a.size() * sizeof(float), cudaMemcpyHostToDevice), + "cudaMemcpy(A large)") && + check_cuda( + cudaMemcpy(device_b, b.data(), b.size() * sizeof(float), cudaMemcpyHostToDevice), + "cudaMemcpy(B large)") && + check_cuda( + cudaMemcpy( + device_ptrs, ptrs.data(), ptrs.size() * sizeof(float*), cudaMemcpyHostToDevice), + "cudaMemcpy(ptrs large)"); + + const float alpha = 1.0f; + const float beta = 0.0f; + std::vector expected(c.size()); + for (int item = 0; item < batch; ++item) { + for (int col = 0; col < dim; ++col) { + for (int row = 0; row < dim; ++row) { + float sum = 0.0f; + for (int i = 0; i < dim; ++i) { + sum += a[item * size + (size_t)i * dim + row] * + b[item * size + (size_t)col * dim + i]; + } + expected[item * size + (size_t)col * dim + row] = sum; + } + } + } + const auto check_output = [&](const char* label) { + if (!check_cuda( + cudaMemcpy(c.data(), device_c, c.size() * sizeof(float), cudaMemcpyDeviceToHost), + "cudaMemcpy(C large)")) { + return false; + } + for (size_t i = 0; i < c.size(); ++i) { + if (!std::isfinite(c[i]) || std::fabs(c[i] - expected[i]) > 1.0e-4f) { + std::fprintf( + stderr, "FAIL: %s output[%zu]=%g, expected %g\n", label, i, c[i], expected[i]); + return false; + } + } + return true; + }; + + if (ok) { + ok = check_cuda(cudaMemset(device_c, 0xff, c.size() * sizeof(float)), "cudaMemset(C)") && + check_cublas( + cublasSgemmBatched( + handle, CUBLAS_OP_N, CUBLAS_OP_N, dim, dim, dim, &alpha, device_ptrs, dim, + device_ptrs + batch, dim, &beta, (float* const*)(device_ptrs + 2 * batch), dim, + batch), + "cublasSgemmBatched(large)") && + check_output("pointer-batched large"); + } + if (ok) { + ok = check_cuda(cudaMemset(device_c, 0xff, c.size() * sizeof(float)), "cudaMemset(C)") && + check_cublas( + cublasSgemmStridedBatched( + handle, CUBLAS_OP_N, CUBLAS_OP_N, dim, dim, dim, &alpha, device_a, dim, + (long long)size, device_b, dim, (long long)size, &beta, device_c, dim, + (long long)size, batch), + "cublasSgemmStridedBatched(large)") && + check_output("strided-batched large"); + } + // Zero-sized problems complete without touching C. + if (ok) { + ok = check_cublas( + cublasSgemmBatched( + handle, CUBLAS_OP_N, CUBLAS_OP_N, dim, dim, dim, &alpha, device_ptrs, dim, + device_ptrs + batch, dim, &beta, (float* const*)(device_ptrs + 2 * batch), dim, + 0), + "cublasSgemmBatched(batch=0)") && + check_cublas( + cublasSgemmStridedBatched( + handle, CUBLAS_OP_N, CUBLAS_OP_N, 0, dim, dim, &alpha, device_a, dim, 0, + device_b, dim, 0, &beta, device_c, dim, 0, 1), + "cublasSgemmStridedBatched(m=0)") && + check_output("zero-sized no-op"); + } + if (ok && cublasSgemmStridedBatched( + handle, CUBLAS_OP_N, CUBLAS_OP_N, -1, dim, dim, &alpha, device_a, dim, 0, + device_b, dim, 0, &beta, device_c, dim, 0, 1) != CUBLAS_STATUS_INVALID_VALUE) { + std::fprintf(stderr, "FAIL: a negative dimension was not rejected\n"); + ok = false; + } + + cudaFree(device); + cudaFree(device_ptrs); + return ok; +} + bool run_stream_churn_case(cublasHandle_t handle) { bool ok = true; @@ -201,6 +399,8 @@ main() { ok &= run_cancellation_case(handle, 24, 432, 1296); ok &= run_cancellation_case(handle, 768, 111, 768); ok &= run_batched_f32_case(handle); + ok &= run_pointer_batched_f32_case(handle); + ok &= run_large_batch_case(handle); ok &= run_stream_churn_case(handle); ok &= check_cublas(cublasDestroy(handle), "cublasDestroy");