Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
4 changes: 2 additions & 2 deletions .coderabbit.yaml
Original file line number Diff line number Diff line change
Expand Up @@ -23,9 +23,9 @@ reviews:
synchronization, and caches or reuse that do not actually take effect.
- Portability: assumptions tied to one GPU, architecture, backend or platform; hardware
limits must be queried or guarded, with a working fallback.
- path: "ggml-patches/**"
- path: "patches/**"
instructions: |
Patches to the pinned ggml submodule. Apply the same correctness, performance and
Patches to the pinned llama.cpp submodule. Apply the same correctness, performance and
portability checks; also check that op preconditions match what the kernels support
and that the patch series stays consistent.
- path: "{BENCHMARK.md,README.md,docs/**,app/bench*}"
Expand Down
6 changes: 3 additions & 3 deletions .dockerignore
Original file line number Diff line number Diff line change
@@ -1,7 +1,7 @@
.git
# Nested submodule .git files (ggml, llama.cpp, third_party/*) become broken
# pointers once copied; exclude them so apply-ggml-patches.sh runs git-apply
# on a plain tree and the build context stays small.
# Nested submodule .git files (llama.cpp, third_party/*) become broken pointers
# once copied; exclude them so the build context stays small. CMake applies
# patches/ to a copy of the plain llama.cpp tree.
**/.git
.gitignore
.gitmodules
Expand Down
9 changes: 4 additions & 5 deletions .gitattributes
Original file line number Diff line number Diff line change
@@ -1,8 +1,7 @@
# Keep ggml patch files LF on every platform. On Windows with core.autocrlf=true
# they would otherwise check out CRLF, and `git apply` rejects some hunks as
# "corrupt patch" on mixed endings (notably 0007-magpietts-nanocodec.patch).
ggml-patches/*.patch text eol=lf -whitespace
llama-patches/*.patch text eol=lf -whitespace
# Keep the llama.cpp patch files LF on every platform. On Windows with
# core.autocrlf=true they would otherwise check out CRLF, and `git apply`
# rejects some hunks as "corrupt patch" on mixed endings.
patches/**/*.patch text eol=lf -whitespace
# Shell scripts must stay LF so they run under bash / Git Bash on Windows.
*.sh text eol=lf
src/tts/tokenizer/mandarin_data/pinyin_phrases.tsv filter=lfs diff=lfs merge=lfs -text
Expand Down
32 changes: 26 additions & 6 deletions .github/workflows/build.yml
Original file line number Diff line number Diff line change
Expand Up @@ -4,6 +4,7 @@ name: Build and Test
# model-free ctest suite, then a CLI smoke test with real models.
#
# Coverage per job:
# Patch series - patches/ applies to the pinned llama.cpp in exported form
# Linux CPU - build, ctest, CPU inference
# macOS Metal - Metal backend COMPILE AND LINK only; inference runs on CPU.
# The hosted macOS VM's paravirtual GPU cannot run ggml Metal.
Expand All @@ -28,6 +29,25 @@ env:
NEMO_SPEECH_MODEL_DIR: ${{ github.workspace }}/.ci-models

jobs:
patch-series:
name: Patch series
if: github.event_name != 'pull_request' || !github.event.pull_request.draft
runs-on: ubuntu-latest
timeout-minutes: 10
steps:
- name: Checkout repository
uses: actions/checkout@df4cb1c069e1874edd31b4311f1884172cec0e10 # v6.0.3
with:
persist-credentials: false

- name: Initialize submodules
run: git submodule update --init --depth 1 llama.cpp

- name: Check patches/
# The series must apply to the pinned llama.cpp and be exactly what
# scripts/llama-patches.sh export produces.
run: scripts/llama-patches.sh check

linux-cpu:
name: Linux CPU
if: github.event_name != 'pull_request' || !github.event.pull_request.draft
Expand All @@ -40,8 +60,8 @@ jobs:
persist-credentials: false

- name: Initialize submodules
# llama.cpp is needed only for the vendored miniaudio header.
run: git submodule update --init --depth 1 ggml llama.cpp
# llama.cpp also provides ggml.
run: git submodule update --init --depth 1 llama.cpp

- name: Install dependencies
run: |
Expand Down Expand Up @@ -86,16 +106,16 @@ jobs:
persist-credentials: false

- name: Initialize submodules
run: git submodule update --init --depth 1 ggml llama.cpp
run: git submodule update --init --depth 1 llama.cpp

- name: Install dependencies
# Homebrew's sentencepiece header includes abseil, which the formula
# does not pull in as a build-time dependency for consumers.
run: brew install bash ninja sentencepiece abseil

- name: Configure
# The runner's default shell is Apple's bash 3.2; configure.sh and
# apply-ggml-patches.sh need bash 4+, so run them under Homebrew bash.
# The runner's default shell is Apple's bash 3.2; configure.sh needs
# bash 4+, so run it under Homebrew bash.
shell: /opt/homebrew/bin/bash -e {0}
run: |
scripts/configure.sh metal-speech \
Expand Down Expand Up @@ -148,7 +168,7 @@ jobs:
persist-credentials: false

- name: Initialize submodules
run: git submodule update --init --depth 1 ggml llama.cpp
run: git submodule update --init --depth 1 llama.cpp

- name: Install Ninja
shell: pwsh
Expand Down
6 changes: 3 additions & 3 deletions .github/workflows/gpu.yml
Original file line number Diff line number Diff line change
Expand Up @@ -74,10 +74,10 @@ jobs:
echo "NEMO_SPEECH_MODEL_DIR=$GITHUB_WORKSPACE/.ci-models" >> "$GITHUB_ENV"

- name: Initialize submodules
run: git submodule update --init --depth 1 ggml llama.cpp
run: git submodule update --init --depth 1 llama.cpp

- name: Configure
# configure.sh applies the ggml patch series for cuda-* presets.
# CMake applies patches/ for cuda-* presets.
# sm_89 is the L4.
run: |
scripts/configure.sh cuda-speech \
Expand Down Expand Up @@ -118,7 +118,7 @@ jobs:
persist-credentials: false

- name: Initialize submodules
run: git submodule update --init --depth 1 ggml llama.cpp
run: git submodule update --init --depth 1 llama.cpp

- name: Install build prerequisites
# The ephemeral VM ships the NVIDIA driver (with the Vulkan ICD) but not
Expand Down
3 changes: 0 additions & 3 deletions .gitmodules
Original file line number Diff line number Diff line change
@@ -1,6 +1,3 @@
[submodule "ggml"]
path = ggml
url = https://github.com/ggml-org/ggml.git
[submodule "proto/riva-common"]
path = proto/riva-common
url = https://github.com/nvidia-riva/common.git
Expand Down
16 changes: 8 additions & 8 deletions .pre-commit-config.yaml
Original file line number Diff line number Diff line change
Expand Up @@ -23,27 +23,27 @@ repos:
- id: check-yaml
- id: check-merge-conflict
- id: end-of-file-fixer
exclude: '^(ggml/|ggml-patches/|proto/riva-common/).*'
exclude: '^(patches/|proto/riva-common/).*'
- id: trailing-whitespace
exclude: '^(ggml/|ggml-patches/|proto/riva-common/).*'
exclude: '^(patches/|proto/riva-common/).*'
- id: mixed-line-ending
args: ['--fix=lf']
exclude: '^(ggml/|ggml-patches/|proto/riva-common/).*'
exclude: '^(patches/|proto/riva-common/).*'

- repo: https://github.com/pre-commit/mirrors-clang-format
rev: v18.1.8
hooks:
- id: clang-format
types_or: [c++, c, cuda]
# ggml + vendored riva protos are upstream code — don't reformat.
exclude: '^(ggml/|ggml-patches/|proto/riva-common/|build/|build-.*/).*'
# llama.cpp patches + vendored riva protos are upstream code — don't reformat.
exclude: '^(patches/|proto/riva-common/|build/|build-.*/).*'

- repo: https://github.com/psf/black
rev: 24.10.0
hooks:
- id: black
args: ['--skip-string-normalization', '--line-length=100']
exclude: '^(ggml/|proto/riva-common/|build/).*'
exclude: '^(proto/riva-common/|build/).*'

- repo: https://github.com/pycqa/isort
rev: 5.13.2
Expand All @@ -53,7 +53,7 @@ repos:
args:
- '--profile=black'
- '--line-length=100'
- '--skip=ggml'
- '--skip=llama.cpp'
- '--skip=proto/riva-common'
- '--skip=build'

Expand All @@ -62,4 +62,4 @@ repos:
hooks:
- id: shellcheck
args: ['--severity=warning']
exclude: '^(ggml/|ggml-patches/|proto/riva-common/).*'
exclude: '^(patches/|proto/riva-common/).*'
2 changes: 1 addition & 1 deletion BENCHMARK.md
Original file line number Diff line number Diff line change
Expand Up @@ -144,7 +144,7 @@ Whisper English normalizer.

| | |
|---|---|
| Models | MagpieTTS Multilingual 357M v2607, NeMo NanoCodec 22 kHz (F16) |
| Models | MagpieTTS Multilingual 357M v2607 (Q8_0, [converted locally](docs/tts/models.md#magpietts-token-generator)), NeMo NanoCodec 22 kHz (F16) |
| Synthesis | `en-US`, default voice, seed 1, 22.05 kHz audio in 186 ms chunks (4 codec frames) |
| Inputs | The 10 LJSpeech sentences of the Riva TTS performance reports ([`ljs_audio_text_test_filelist_small.txt`](test_files/tts/ljs_audio_text_test_filelist_small.txt), 20 requests); by length, [`test_files/tts/bench`](test_files/tts/bench) (5 requests per input) |
| Metrics | Latencies at the client from the streaming audio callback; throughput is audio duration over wall time |
Expand Down
94 changes: 20 additions & 74 deletions CMakeLists.txt
Original file line number Diff line number Diff line change
Expand Up @@ -119,22 +119,16 @@ set(NEMO_SPEECH_DEPENDENCY_PREFIX "${CMAKE_SOURCE_DIR}/.deps" CACHE PATH
"User-writable prefix containing optional project-built dependencies")
# GGML_CUDA is forwarded directly to ggml's CMake via -DGGML_CUDA=ON.

# Keep the legacy WITH_NMT/WITH_GRPC options as downstream-compatible aliases.
if(NEMO_SPEECH_WITH_NMT)
set(NEMO_SPEECH_BUILD_NMT ON CACHE BOOL "Build text translation (links llama.cpp)" FORCE)
endif()
if(NEMO_SPEECH_BUILD_NMT)
set(NEMO_SPEECH_WITH_NMT ON)
endif()
# WITH_NMT/WITH_GRPC only seed the BUILD_* defaults above; an explicit BUILD_*
# value takes precedence.
foreach(_alias NMT GRPC)
if(NEMO_SPEECH_WITH_${_alias})
message(DEPRECATION "NEMO_SPEECH_WITH_${_alias} is deprecated; use NEMO_SPEECH_BUILD_${_alias}")
endif()
endforeach()
if(NEMO_SPEECH_BUILD_S2S)
set(NEMO_SPEECH_BUILD_ASR ON CACHE BOOL "Build automatic speech recognition" FORCE)
endif()
if(NEMO_SPEECH_WITH_GRPC)
set(NEMO_SPEECH_BUILD_GRPC ON CACHE BOOL "Build Riva-compatible gRPC adapters" FORCE)
endif()
if(NEMO_SPEECH_BUILD_GRPC)
set(NEMO_SPEECH_WITH_GRPC ON)
endif()

# Reject invalid component combinations early.
if(WIN32 AND NEMO_SPEECH_WITH_NORM)
Expand Down Expand Up @@ -176,57 +170,14 @@ endif()
# no-op for Metal, Vulkan, and CPU builds.
option(NEMO_SPEECH_CUBLAS_SHIM "Build the in-tree drop-in cuBLAS shim (native GEMM, no cuBLASLt)" OFF)

# Whether the linked ggml has the project ASR patches applied (ggml-patches/:
# the fused rel-pos attention op and the F16 depthwise-conv kernel). The ASR
# directly references these (a new op symbol + F16 CONV_2D_DW behaviour), so a
# build against STANDARD upstream ggml must set -DNEMO_SPEECH_GGML_PATCHED=OFF
# - the encoder then uses only stock ggml ops (unfused rel-pos, ggml_conv_1d_dw)
# at some latency cost. The remaining low-level patches (NVFP4, skinny-q8,
# norm/BF16 fusions, CUDA graph and launch fixes) are transparent ggml-cuda internals.
option(NEMO_SPEECH_GGML_PATCHED "Linked ggml has the project ASR patches applied (fused rel-pos op, F16 dw-conv)" ON)

# Relative-position mode of the fused CUDA attention op. Replaces the unfused
# encoder attention sequence with one kernel. Defaults ON with CUDA and the
# patched ggml; the encoder otherwise uses stock ggml ops.
if(GGML_CUDA AND NEMO_SPEECH_GGML_PATCHED)
option(NEMO_SPEECH_FUSED_RELPOS_ATTN "Use the fused rel-pos attention CUDA op in the encoder" ON)
else()
set(NEMO_SPEECH_FUSED_RELPOS_ATTN OFF CACHE BOOL
"Use the fused rel-pos attention CUDA op in the encoder" FORCE)
message(STATUS
"NEMO_SPEECH_FUSED_RELPOS_ATTN forced OFF: requires GGML_CUDA=ON and "
"NEMO_SPEECH_GGML_PATCHED=ON (patched ggml). Using the unfused path.")
endif()

# Direct depthwise-conv kernel (GGML_OP_CONV_2D_DW) for the conformer/subsample
# convs. Faster than ggml_conv_1d_dw's im2col + matmul, but the conv weights are
# F16 and the direct CONV_2D_DW op only reads F16 kernels correctly on the
# patched CUDA backend (ggml patch 0004); stock ggml (CPU or unpatched CUDA)
# reads them as F32. Defaults ON whenever GGML_CUDA + a patched ggml are present;
# forced OFF otherwise, where nn.cpp falls back to the portable ggml_conv_1d_dw
# lowering (im2col + mul_mat, F16-safe on every backend and on stock ggml). Kept
# as an explicit option so it can be toggled OFF for debugging.
if(GGML_CUDA AND NEMO_SPEECH_GGML_PATCHED)
option(NEMO_SPEECH_DIRECT_DW_CONV "Use the direct CUDA depthwise-conv kernel in the encoder" ON)
else()
set(NEMO_SPEECH_DIRECT_DW_CONV OFF CACHE BOOL
"Use the direct CUDA depthwise-conv kernel in the encoder" FORCE)
message(STATUS
"NEMO_SPEECH_DIRECT_DW_CONV forced OFF: requires GGML_CUDA=ON and "
"NEMO_SPEECH_GGML_PATCHED=ON (patched ggml). Using ggml_conv_1d_dw.")
endif()

# FastConformer graph rewrites that target CUDA-only ggml ops/fusions (sigmoid
# GLU plus BF16 projection epilogues). Keep the symbols out of standard-ggml,
# CPU, Metal, and Vulkan builds. The CUDA backend performs the finer runtime
# architecture checks: native BF16 epilogues require NVIDIA SM80+, and the
# 1024-wide LayerNorm launch specialization requires SM90+.
if(GGML_CUDA AND NEMO_SPEECH_GGML_PATCHED)
option(NEMO_SPEECH_FASTCONFORMER_CUDA_FUSIONS
"Use patched CUDA FastConformer graph fusions" ON)
else()
set(NEMO_SPEECH_FASTCONFORMER_CUDA_FUSIONS OFF CACHE BOOL
"Use patched CUDA FastConformer graph fusions" FORCE)
# Build ggml and llama.cpp with the patches/ series applied (see patches/README.md
# and cmake/llama_cpp.cmake). OFF builds the pristine llama.cpp submodule; the code
# then uses only stock ggml operations, at some latency cost on CUDA.
option(NEMO_SPEECH_GGML_PATCHED "Apply patches/ to ggml and llama.cpp and use the patched operations" ON)
if(NEMO_SPEECH_BUILD_S2S AND NOT NEMO_SPEECH_GGML_PATCHED AND NOT NEMO_SPEECH_LLAMA_CPP_SOURCE_DIR)
# EarTTS stores its Gemma 3 attention scale in GGUF metadata that stock
# llama.cpp ignores, which would silently change the model's output.
message(FATAL_ERROR "NEMO_SPEECH_BUILD_S2S requires NEMO_SPEECH_GGML_PATCHED=ON")
endif()

set(GGML_DEPENDENCIES ggml ggml-base ggml-cpu)
Expand All @@ -245,12 +196,6 @@ if(GGML_CUDA)
# on by default. ggml-cuda gates it at runtime and disables it for graphs it
# can't capture (e.g. shapes that change between calls).
set(GGML_CUDA_GRAPHS ON CACHE BOOL "Enable ggml-cuda CUDA Graphs")
file(READ "${CMAKE_SOURCE_DIR}/ggml/src/ggml-cuda/conv-transpose-1d.cu" GGML_CUDA_CONV_TRANSPOSE_1D)
if(NOT GGML_CUDA_CONV_TRANSPOSE_1D MATCHES "conv_transpose_1d_use_nanocodec_grouped2")
message(WARNING
"MagpieTTS/NanoCodec CUDA ggml patch is not applied. "
"Run scripts/apply-ggml-patches.sh before compiling CUDA TTS targets.")
endif()
endif()
if(GGML_VULKAN)
list(APPEND GGML_DEPENDENCIES ggml-vulkan)
Expand All @@ -263,7 +208,8 @@ endif()
# it CPU matmuls run one dot product per output.
set(GGML_LLAMAFILE ON CACHE BOOL "ggml: use llamafile SGEMM")

add_subdirectory(ggml EXCLUDE_FROM_ALL)
include(cmake/llama_cpp.cmake)
add_subdirectory("${NEMO_SPEECH_LLAMA_CPP_DIR}/ggml" "${CMAKE_BINARY_DIR}/ggml" EXCLUDE_FROM_ALL)

# ggml is an implementation dependency, so install only the runtime libraries
# required by our shared libraries. Its C++ headers, CMake package, pkg-config
Expand Down Expand Up @@ -367,14 +313,14 @@ endif()
# llama.cpp for the NMT decoder. It reuses the ggml target added above (its
# CMake builds its own ggml only when the target is absent), so one ggml backend
# is shared across asr/tts/nmt. Build only libllama.
if(NEMO_SPEECH_WITH_NMT OR NEMO_SPEECH_BUILD_S2S)
if(NEMO_SPEECH_BUILD_NMT OR NEMO_SPEECH_BUILD_S2S)
set(LLAMA_BUILD_COMMON OFF CACHE BOOL "" FORCE)
set(LLAMA_BUILD_TESTS OFF CACHE BOOL "" FORCE)
set(LLAMA_BUILD_EXAMPLES OFF CACHE BOOL "" FORCE)
set(LLAMA_BUILD_TOOLS OFF CACHE BOOL "" FORCE)
set(LLAMA_BUILD_SERVER OFF CACHE BOOL "" FORCE)
set(LLAMA_CURL OFF CACHE BOOL "" FORCE)
add_subdirectory(llama.cpp EXCLUDE_FROM_ALL)
add_subdirectory("${NEMO_SPEECH_LLAMA_CPP_DIR}" "${CMAKE_BINARY_DIR}/llama.cpp" EXCLUDE_FROM_ALL)
set_property(TARGET llama PROPERTY PUBLIC_HEADER "")
install(TARGETS llama
RUNTIME DESTINATION ${CMAKE_INSTALL_BINDIR}
Expand Down Expand Up @@ -470,7 +416,7 @@ if(DEFINED VCPKG_INSTALLED_DIR AND DEFINED VCPKG_TARGET_TRIPLET)
RENAME LICENSE)
endforeach()
endif()
install(FILES ggml/LICENSE
install(FILES llama.cpp/LICENSE
DESTINATION "${NEMO_SPEECH_THIRD_PARTY_LICENSE_DIR}/ggml")
if(NEMO_SPEECH_BUILD_ASR AND NEMO_SPEECH_BUILD_CLI AND NEMO_SPEECH_BUILD_MIC_CAPTURE)
install(FILES third_party/miniaudio/LICENSE
Expand Down
Loading
Loading