diff --git a/.coderabbit.yaml b/.coderabbit.yaml index 200ce48..970ef7e 100644 --- a/.coderabbit.yaml +++ b/.coderabbit.yaml @@ -23,9 +23,9 @@ reviews: synchronization, and caches or reuse that do not actually take effect. - Portability: assumptions tied to one GPU, architecture, backend or platform; hardware limits must be queried or guarded, with a working fallback. - - path: "ggml-patches/**" + - path: "patches/**" instructions: | - Patches to the pinned ggml submodule. Apply the same correctness, performance and + Patches to the pinned llama.cpp submodule. Apply the same correctness, performance and portability checks; also check that op preconditions match what the kernels support and that the patch series stays consistent. - path: "{BENCHMARK.md,README.md,docs/**,app/bench*}" diff --git a/.dockerignore b/.dockerignore index 601e7c9..270352a 100644 --- a/.dockerignore +++ b/.dockerignore @@ -1,7 +1,7 @@ .git -# Nested submodule .git files (ggml, llama.cpp, third_party/*) become broken -# pointers once copied; exclude them so apply-ggml-patches.sh runs git-apply -# on a plain tree and the build context stays small. +# Nested submodule .git files (llama.cpp, third_party/*) become broken pointers +# once copied; exclude them so the build context stays small. CMake applies +# patches/ to a copy of the plain llama.cpp tree. **/.git .gitignore .gitmodules diff --git a/.gitattributes b/.gitattributes index c5d6456..7853d7f 100644 --- a/.gitattributes +++ b/.gitattributes @@ -1,8 +1,7 @@ -# Keep ggml patch files LF on every platform. On Windows with core.autocrlf=true -# they would otherwise check out CRLF, and `git apply` rejects some hunks as -# "corrupt patch" on mixed endings (notably 0007-magpietts-nanocodec.patch). -ggml-patches/*.patch text eol=lf -whitespace -llama-patches/*.patch text eol=lf -whitespace +# Keep the llama.cpp patch files LF on every platform. On Windows with +# core.autocrlf=true they would otherwise check out CRLF, and `git apply` +# rejects some hunks as "corrupt patch" on mixed endings. +patches/**/*.patch text eol=lf -whitespace # Shell scripts must stay LF so they run under bash / Git Bash on Windows. *.sh text eol=lf src/tts/tokenizer/mandarin_data/pinyin_phrases.tsv filter=lfs diff=lfs merge=lfs -text diff --git a/.github/workflows/build.yml b/.github/workflows/build.yml index 6792e96..664c3bc 100644 --- a/.github/workflows/build.yml +++ b/.github/workflows/build.yml @@ -4,6 +4,7 @@ name: Build and Test # model-free ctest suite, then a CLI smoke test with real models. # # Coverage per job: +# Patch series - patches/ applies to the pinned llama.cpp in exported form # Linux CPU - build, ctest, CPU inference # macOS Metal - Metal backend COMPILE AND LINK only; inference runs on CPU. # The hosted macOS VM's paravirtual GPU cannot run ggml Metal. @@ -28,6 +29,25 @@ env: NEMO_SPEECH_MODEL_DIR: ${{ github.workspace }}/.ci-models jobs: + patch-series: + name: Patch series + if: github.event_name != 'pull_request' || !github.event.pull_request.draft + runs-on: ubuntu-latest + timeout-minutes: 10 + steps: + - name: Checkout repository + uses: actions/checkout@df4cb1c069e1874edd31b4311f1884172cec0e10 # v6.0.3 + with: + persist-credentials: false + + - name: Initialize submodules + run: git submodule update --init --depth 1 llama.cpp + + - name: Check patches/ + # The series must apply to the pinned llama.cpp and be exactly what + # scripts/llama-patches.sh export produces. + run: scripts/llama-patches.sh check + linux-cpu: name: Linux CPU if: github.event_name != 'pull_request' || !github.event.pull_request.draft @@ -40,8 +60,8 @@ jobs: persist-credentials: false - name: Initialize submodules - # llama.cpp is needed only for the vendored miniaudio header. - run: git submodule update --init --depth 1 ggml llama.cpp + # llama.cpp also provides ggml. + run: git submodule update --init --depth 1 llama.cpp - name: Install dependencies run: | @@ -86,7 +106,7 @@ jobs: persist-credentials: false - name: Initialize submodules - run: git submodule update --init --depth 1 ggml llama.cpp + run: git submodule update --init --depth 1 llama.cpp - name: Install dependencies # Homebrew's sentencepiece header includes abseil, which the formula @@ -94,8 +114,8 @@ jobs: run: brew install bash ninja sentencepiece abseil - name: Configure - # The runner's default shell is Apple's bash 3.2; configure.sh and - # apply-ggml-patches.sh need bash 4+, so run them under Homebrew bash. + # The runner's default shell is Apple's bash 3.2; configure.sh needs + # bash 4+, so run it under Homebrew bash. shell: /opt/homebrew/bin/bash -e {0} run: | scripts/configure.sh metal-speech \ @@ -148,7 +168,7 @@ jobs: persist-credentials: false - name: Initialize submodules - run: git submodule update --init --depth 1 ggml llama.cpp + run: git submodule update --init --depth 1 llama.cpp - name: Install Ninja shell: pwsh diff --git a/.github/workflows/gpu.yml b/.github/workflows/gpu.yml index f5e80c9..0ddfa8a 100644 --- a/.github/workflows/gpu.yml +++ b/.github/workflows/gpu.yml @@ -74,10 +74,10 @@ jobs: echo "NEMO_SPEECH_MODEL_DIR=$GITHUB_WORKSPACE/.ci-models" >> "$GITHUB_ENV" - name: Initialize submodules - run: git submodule update --init --depth 1 ggml llama.cpp + run: git submodule update --init --depth 1 llama.cpp - name: Configure - # configure.sh applies the ggml patch series for cuda-* presets. + # CMake applies patches/ for cuda-* presets. # sm_89 is the L4. run: | scripts/configure.sh cuda-speech \ @@ -118,7 +118,7 @@ jobs: persist-credentials: false - name: Initialize submodules - run: git submodule update --init --depth 1 ggml llama.cpp + run: git submodule update --init --depth 1 llama.cpp - name: Install build prerequisites # The ephemeral VM ships the NVIDIA driver (with the Vulkan ICD) but not diff --git a/.gitmodules b/.gitmodules index 1736af9..7f26d70 100644 --- a/.gitmodules +++ b/.gitmodules @@ -1,6 +1,3 @@ -[submodule "ggml"] - path = ggml - url = https://github.com/ggml-org/ggml.git [submodule "proto/riva-common"] path = proto/riva-common url = https://github.com/nvidia-riva/common.git diff --git a/.pre-commit-config.yaml b/.pre-commit-config.yaml index ff55267..bd9bd1c 100644 --- a/.pre-commit-config.yaml +++ b/.pre-commit-config.yaml @@ -23,27 +23,27 @@ repos: - id: check-yaml - id: check-merge-conflict - id: end-of-file-fixer - exclude: '^(ggml/|ggml-patches/|proto/riva-common/).*' + exclude: '^(patches/|proto/riva-common/).*' - id: trailing-whitespace - exclude: '^(ggml/|ggml-patches/|proto/riva-common/).*' + exclude: '^(patches/|proto/riva-common/).*' - id: mixed-line-ending args: ['--fix=lf'] - exclude: '^(ggml/|ggml-patches/|proto/riva-common/).*' + exclude: '^(patches/|proto/riva-common/).*' - repo: https://github.com/pre-commit/mirrors-clang-format rev: v18.1.8 hooks: - id: clang-format types_or: [c++, c, cuda] - # ggml + vendored riva protos are upstream code — don't reformat. - exclude: '^(ggml/|ggml-patches/|proto/riva-common/|build/|build-.*/).*' + # llama.cpp patches + vendored riva protos are upstream code — don't reformat. + exclude: '^(patches/|proto/riva-common/|build/|build-.*/).*' - repo: https://github.com/psf/black rev: 24.10.0 hooks: - id: black args: ['--skip-string-normalization', '--line-length=100'] - exclude: '^(ggml/|proto/riva-common/|build/).*' + exclude: '^(proto/riva-common/|build/).*' - repo: https://github.com/pycqa/isort rev: 5.13.2 @@ -53,7 +53,7 @@ repos: args: - '--profile=black' - '--line-length=100' - - '--skip=ggml' + - '--skip=llama.cpp' - '--skip=proto/riva-common' - '--skip=build' @@ -62,4 +62,4 @@ repos: hooks: - id: shellcheck args: ['--severity=warning'] - exclude: '^(ggml/|ggml-patches/|proto/riva-common/).*' + exclude: '^(patches/|proto/riva-common/).*' diff --git a/BENCHMARK.md b/BENCHMARK.md index 1d0f3b4..ecb19d7 100644 --- a/BENCHMARK.md +++ b/BENCHMARK.md @@ -144,7 +144,7 @@ Whisper English normalizer. | | | |---|---| -| Models | MagpieTTS Multilingual 357M v2607, NeMo NanoCodec 22 kHz (F16) | +| Models | MagpieTTS Multilingual 357M v2607 (Q8_0, [converted locally](docs/tts/models.md#magpietts-token-generator)), NeMo NanoCodec 22 kHz (F16) | | Synthesis | `en-US`, default voice, seed 1, 22.05 kHz audio in 186 ms chunks (4 codec frames) | | Inputs | The 10 LJSpeech sentences of the Riva TTS performance reports ([`ljs_audio_text_test_filelist_small.txt`](test_files/tts/ljs_audio_text_test_filelist_small.txt), 20 requests); by length, [`test_files/tts/bench`](test_files/tts/bench) (5 requests per input) | | Metrics | Latencies at the client from the streaming audio callback; throughput is audio duration over wall time | diff --git a/CMakeLists.txt b/CMakeLists.txt index 50fd2fe..685b661 100644 --- a/CMakeLists.txt +++ b/CMakeLists.txt @@ -119,22 +119,16 @@ set(NEMO_SPEECH_DEPENDENCY_PREFIX "${CMAKE_SOURCE_DIR}/.deps" CACHE PATH "User-writable prefix containing optional project-built dependencies") # GGML_CUDA is forwarded directly to ggml's CMake via -DGGML_CUDA=ON. -# Keep the legacy WITH_NMT/WITH_GRPC options as downstream-compatible aliases. -if(NEMO_SPEECH_WITH_NMT) - set(NEMO_SPEECH_BUILD_NMT ON CACHE BOOL "Build text translation (links llama.cpp)" FORCE) -endif() -if(NEMO_SPEECH_BUILD_NMT) - set(NEMO_SPEECH_WITH_NMT ON) -endif() +# WITH_NMT/WITH_GRPC only seed the BUILD_* defaults above; an explicit BUILD_* +# value takes precedence. +foreach(_alias NMT GRPC) + if(NEMO_SPEECH_WITH_${_alias}) + message(DEPRECATION "NEMO_SPEECH_WITH_${_alias} is deprecated; use NEMO_SPEECH_BUILD_${_alias}") + endif() +endforeach() if(NEMO_SPEECH_BUILD_S2S) set(NEMO_SPEECH_BUILD_ASR ON CACHE BOOL "Build automatic speech recognition" FORCE) endif() -if(NEMO_SPEECH_WITH_GRPC) - set(NEMO_SPEECH_BUILD_GRPC ON CACHE BOOL "Build Riva-compatible gRPC adapters" FORCE) -endif() -if(NEMO_SPEECH_BUILD_GRPC) - set(NEMO_SPEECH_WITH_GRPC ON) -endif() # Reject invalid component combinations early. if(WIN32 AND NEMO_SPEECH_WITH_NORM) @@ -176,57 +170,14 @@ endif() # no-op for Metal, Vulkan, and CPU builds. option(NEMO_SPEECH_CUBLAS_SHIM "Build the in-tree drop-in cuBLAS shim (native GEMM, no cuBLASLt)" OFF) -# Whether the linked ggml has the project ASR patches applied (ggml-patches/: -# the fused rel-pos attention op and the F16 depthwise-conv kernel). The ASR -# directly references these (a new op symbol + F16 CONV_2D_DW behaviour), so a -# build against STANDARD upstream ggml must set -DNEMO_SPEECH_GGML_PATCHED=OFF -# - the encoder then uses only stock ggml ops (unfused rel-pos, ggml_conv_1d_dw) -# at some latency cost. The remaining low-level patches (NVFP4, skinny-q8, -# norm/BF16 fusions, CUDA graph and launch fixes) are transparent ggml-cuda internals. -option(NEMO_SPEECH_GGML_PATCHED "Linked ggml has the project ASR patches applied (fused rel-pos op, F16 dw-conv)" ON) - -# Relative-position mode of the fused CUDA attention op. Replaces the unfused -# encoder attention sequence with one kernel. Defaults ON with CUDA and the -# patched ggml; the encoder otherwise uses stock ggml ops. -if(GGML_CUDA AND NEMO_SPEECH_GGML_PATCHED) - option(NEMO_SPEECH_FUSED_RELPOS_ATTN "Use the fused rel-pos attention CUDA op in the encoder" ON) -else() - set(NEMO_SPEECH_FUSED_RELPOS_ATTN OFF CACHE BOOL - "Use the fused rel-pos attention CUDA op in the encoder" FORCE) - message(STATUS - "NEMO_SPEECH_FUSED_RELPOS_ATTN forced OFF: requires GGML_CUDA=ON and " - "NEMO_SPEECH_GGML_PATCHED=ON (patched ggml). Using the unfused path.") -endif() - -# Direct depthwise-conv kernel (GGML_OP_CONV_2D_DW) for the conformer/subsample -# convs. Faster than ggml_conv_1d_dw's im2col + matmul, but the conv weights are -# F16 and the direct CONV_2D_DW op only reads F16 kernels correctly on the -# patched CUDA backend (ggml patch 0004); stock ggml (CPU or unpatched CUDA) -# reads them as F32. Defaults ON whenever GGML_CUDA + a patched ggml are present; -# forced OFF otherwise, where nn.cpp falls back to the portable ggml_conv_1d_dw -# lowering (im2col + mul_mat, F16-safe on every backend and on stock ggml). Kept -# as an explicit option so it can be toggled OFF for debugging. -if(GGML_CUDA AND NEMO_SPEECH_GGML_PATCHED) - option(NEMO_SPEECH_DIRECT_DW_CONV "Use the direct CUDA depthwise-conv kernel in the encoder" ON) -else() - set(NEMO_SPEECH_DIRECT_DW_CONV OFF CACHE BOOL - "Use the direct CUDA depthwise-conv kernel in the encoder" FORCE) - message(STATUS - "NEMO_SPEECH_DIRECT_DW_CONV forced OFF: requires GGML_CUDA=ON and " - "NEMO_SPEECH_GGML_PATCHED=ON (patched ggml). Using ggml_conv_1d_dw.") -endif() - -# FastConformer graph rewrites that target CUDA-only ggml ops/fusions (sigmoid -# GLU plus BF16 projection epilogues). Keep the symbols out of standard-ggml, -# CPU, Metal, and Vulkan builds. The CUDA backend performs the finer runtime -# architecture checks: native BF16 epilogues require NVIDIA SM80+, and the -# 1024-wide LayerNorm launch specialization requires SM90+. -if(GGML_CUDA AND NEMO_SPEECH_GGML_PATCHED) - option(NEMO_SPEECH_FASTCONFORMER_CUDA_FUSIONS - "Use patched CUDA FastConformer graph fusions" ON) -else() - set(NEMO_SPEECH_FASTCONFORMER_CUDA_FUSIONS OFF CACHE BOOL - "Use patched CUDA FastConformer graph fusions" FORCE) +# Build ggml and llama.cpp with the patches/ series applied (see patches/README.md +# and cmake/llama_cpp.cmake). OFF builds the pristine llama.cpp submodule; the code +# then uses only stock ggml operations, at some latency cost on CUDA. +option(NEMO_SPEECH_GGML_PATCHED "Apply patches/ to ggml and llama.cpp and use the patched operations" ON) +if(NEMO_SPEECH_BUILD_S2S AND NOT NEMO_SPEECH_GGML_PATCHED AND NOT NEMO_SPEECH_LLAMA_CPP_SOURCE_DIR) + # EarTTS stores its Gemma 3 attention scale in GGUF metadata that stock + # llama.cpp ignores, which would silently change the model's output. + message(FATAL_ERROR "NEMO_SPEECH_BUILD_S2S requires NEMO_SPEECH_GGML_PATCHED=ON") endif() set(GGML_DEPENDENCIES ggml ggml-base ggml-cpu) @@ -245,12 +196,6 @@ if(GGML_CUDA) # on by default. ggml-cuda gates it at runtime and disables it for graphs it # can't capture (e.g. shapes that change between calls). set(GGML_CUDA_GRAPHS ON CACHE BOOL "Enable ggml-cuda CUDA Graphs") - file(READ "${CMAKE_SOURCE_DIR}/ggml/src/ggml-cuda/conv-transpose-1d.cu" GGML_CUDA_CONV_TRANSPOSE_1D) - if(NOT GGML_CUDA_CONV_TRANSPOSE_1D MATCHES "conv_transpose_1d_use_nanocodec_grouped2") - message(WARNING - "MagpieTTS/NanoCodec CUDA ggml patch is not applied. " - "Run scripts/apply-ggml-patches.sh before compiling CUDA TTS targets.") - endif() endif() if(GGML_VULKAN) list(APPEND GGML_DEPENDENCIES ggml-vulkan) @@ -263,7 +208,8 @@ endif() # it CPU matmuls run one dot product per output. set(GGML_LLAMAFILE ON CACHE BOOL "ggml: use llamafile SGEMM") -add_subdirectory(ggml EXCLUDE_FROM_ALL) +include(cmake/llama_cpp.cmake) +add_subdirectory("${NEMO_SPEECH_LLAMA_CPP_DIR}/ggml" "${CMAKE_BINARY_DIR}/ggml" EXCLUDE_FROM_ALL) # ggml is an implementation dependency, so install only the runtime libraries # required by our shared libraries. Its C++ headers, CMake package, pkg-config @@ -367,14 +313,14 @@ endif() # llama.cpp for the NMT decoder. It reuses the ggml target added above (its # CMake builds its own ggml only when the target is absent), so one ggml backend # is shared across asr/tts/nmt. Build only libllama. -if(NEMO_SPEECH_WITH_NMT OR NEMO_SPEECH_BUILD_S2S) +if(NEMO_SPEECH_BUILD_NMT OR NEMO_SPEECH_BUILD_S2S) set(LLAMA_BUILD_COMMON OFF CACHE BOOL "" FORCE) set(LLAMA_BUILD_TESTS OFF CACHE BOOL "" FORCE) set(LLAMA_BUILD_EXAMPLES OFF CACHE BOOL "" FORCE) set(LLAMA_BUILD_TOOLS OFF CACHE BOOL "" FORCE) set(LLAMA_BUILD_SERVER OFF CACHE BOOL "" FORCE) set(LLAMA_CURL OFF CACHE BOOL "" FORCE) - add_subdirectory(llama.cpp EXCLUDE_FROM_ALL) + add_subdirectory("${NEMO_SPEECH_LLAMA_CPP_DIR}" "${CMAKE_BINARY_DIR}/llama.cpp" EXCLUDE_FROM_ALL) set_property(TARGET llama PROPERTY PUBLIC_HEADER "") install(TARGETS llama RUNTIME DESTINATION ${CMAKE_INSTALL_BINDIR} @@ -470,7 +416,7 @@ if(DEFINED VCPKG_INSTALLED_DIR AND DEFINED VCPKG_TARGET_TRIPLET) RENAME LICENSE) endforeach() endif() -install(FILES ggml/LICENSE +install(FILES llama.cpp/LICENSE DESTINATION "${NEMO_SPEECH_THIRD_PARTY_LICENSE_DIR}/ggml") if(NEMO_SPEECH_BUILD_ASR AND NEMO_SPEECH_BUILD_CLI AND NEMO_SPEECH_BUILD_MIC_CAPTURE) install(FILES third_party/miniaudio/LICENSE diff --git a/CMakePresets.json b/CMakePresets.json index 0d79eb5..dc6cbbe 100644 --- a/CMakePresets.json +++ b/CMakePresets.json @@ -15,7 +15,8 @@ "CMAKE_BUILD_TYPE": "Release", "NEMO_SPEECH_BUILD_EXAMPLES": "OFF", "NEMO_SPEECH_BUILD_TESTS": "OFF", - "NEMO_SPEECH_BUILD_TOOLS": "OFF" + "NEMO_SPEECH_BUILD_TOOLS": "OFF", + "GGML_LLAMAFILE": "ON" } }, { @@ -23,7 +24,7 @@ "inherits": "base", "displayName": "CPU ASR and diarization", "cacheVariables": { - "NEMO_SPEECH_GGML_PATCHED": "OFF", + "NEMO_SPEECH_GGML_PATCHED": "ON", "NEMO_SPEECH_BUILD_ASR": "ON", "NEMO_SPEECH_BUILD_DIAR": "ON", "NEMO_SPEECH_BUILD_TTS": "OFF", @@ -35,7 +36,7 @@ "inherits": "base", "displayName": "CPU standalone diarization", "cacheVariables": { - "NEMO_SPEECH_GGML_PATCHED": "OFF", + "NEMO_SPEECH_GGML_PATCHED": "ON", "NEMO_SPEECH_BUILD_ASR": "OFF", "NEMO_SPEECH_BUILD_DIAR": "ON", "NEMO_SPEECH_BUILD_TTS": "OFF", @@ -47,7 +48,7 @@ "inherits": "base", "displayName": "CPU text-to-speech", "cacheVariables": { - "NEMO_SPEECH_GGML_PATCHED": "OFF", + "NEMO_SPEECH_GGML_PATCHED": "ON", "NEMO_SPEECH_BUILD_ASR": "OFF", "NEMO_SPEECH_BUILD_DIAR": "OFF", "NEMO_SPEECH_BUILD_TTS": "ON", @@ -59,7 +60,7 @@ "inherits": "base", "displayName": "CPU text translation", "cacheVariables": { - "NEMO_SPEECH_GGML_PATCHED": "OFF", + "NEMO_SPEECH_GGML_PATCHED": "ON", "NEMO_SPEECH_BUILD_ASR": "OFF", "NEMO_SPEECH_BUILD_DIAR": "OFF", "NEMO_SPEECH_BUILD_TTS": "OFF", @@ -71,7 +72,7 @@ "inherits": "base", "displayName": "CPU ASR, diarization, NMT, and TTS", "cacheVariables": { - "NEMO_SPEECH_GGML_PATCHED": "OFF", + "NEMO_SPEECH_GGML_PATCHED": "ON", "NEMO_SPEECH_BUILD_ASR": "ON", "NEMO_SPEECH_BUILD_DIAR": "ON", "NEMO_SPEECH_BUILD_TTS": "ON", @@ -128,7 +129,8 @@ "inherits": "cpu-asr", "displayName": "Metal ASR and diarization", "cacheVariables": { - "GGML_METAL": "ON" + "GGML_METAL": "ON", + "NEMO_SPEECH_GGML_PATCHED": "OFF" } }, { @@ -136,7 +138,8 @@ "inherits": "cpu-diar", "displayName": "Metal standalone diarization", "cacheVariables": { - "GGML_METAL": "ON" + "GGML_METAL": "ON", + "NEMO_SPEECH_GGML_PATCHED": "OFF" } }, { @@ -144,7 +147,8 @@ "inherits": "cpu-tts", "displayName": "Metal text-to-speech", "cacheVariables": { - "GGML_METAL": "ON" + "GGML_METAL": "ON", + "NEMO_SPEECH_GGML_PATCHED": "OFF" } }, { @@ -152,7 +156,8 @@ "inherits": "cpu-nmt", "displayName": "Metal text translation", "cacheVariables": { - "GGML_METAL": "ON" + "GGML_METAL": "ON", + "NEMO_SPEECH_GGML_PATCHED": "OFF" } }, { @@ -160,7 +165,8 @@ "inherits": "cpu-speech", "displayName": "Metal ASR, diarization, NMT, and TTS", "cacheVariables": { - "GGML_METAL": "ON" + "GGML_METAL": "ON", + "NEMO_SPEECH_GGML_PATCHED": "OFF" } }, { @@ -168,7 +174,8 @@ "inherits": "cpu-asr", "displayName": "Vulkan ASR and diarization", "cacheVariables": { - "GGML_VULKAN": "ON" + "GGML_VULKAN": "ON", + "NEMO_SPEECH_GGML_PATCHED": "OFF" } }, { @@ -176,7 +183,8 @@ "inherits": "cpu-diar", "displayName": "Vulkan standalone diarization", "cacheVariables": { - "GGML_VULKAN": "ON" + "GGML_VULKAN": "ON", + "NEMO_SPEECH_GGML_PATCHED": "OFF" } }, { @@ -184,7 +192,8 @@ "inherits": "cpu-tts", "displayName": "Vulkan text-to-speech", "cacheVariables": { - "GGML_VULKAN": "ON" + "GGML_VULKAN": "ON", + "NEMO_SPEECH_GGML_PATCHED": "OFF" } }, { @@ -192,7 +201,8 @@ "inherits": "cpu-nmt", "displayName": "Vulkan text translation", "cacheVariables": { - "GGML_VULKAN": "ON" + "GGML_VULKAN": "ON", + "NEMO_SPEECH_GGML_PATCHED": "OFF" } }, { @@ -200,7 +210,8 @@ "inherits": "cpu-speech", "displayName": "Vulkan ASR, diarization, NMT, and TTS", "cacheVariables": { - "GGML_VULKAN": "ON" + "GGML_VULKAN": "ON", + "NEMO_SPEECH_GGML_PATCHED": "OFF" } }, { @@ -211,9 +222,7 @@ "NEMO_SPEECH_BUILD_TTS": "ON", "NEMO_SPEECH_BUILD_NMT": "ON", "NEMO_SPEECH_BUILD_HTTP": "ON", - "NEMO_SPEECH_BUILD_GRPC": "OFF", - "NEMO_SPEECH_WITH_GRPC": "OFF", - "NEMO_SPEECH_WITH_NMT": "ON" + "NEMO_SPEECH_BUILD_GRPC": "OFF" } }, { @@ -224,9 +233,7 @@ "NEMO_SPEECH_BUILD_TTS": "ON", "NEMO_SPEECH_BUILD_NMT": "ON", "NEMO_SPEECH_BUILD_HTTP": "ON", - "NEMO_SPEECH_BUILD_GRPC": "OFF", - "NEMO_SPEECH_WITH_GRPC": "OFF", - "NEMO_SPEECH_WITH_NMT": "ON" + "NEMO_SPEECH_BUILD_GRPC": "OFF" } }, { @@ -239,9 +246,7 @@ "NEMO_SPEECH_BUILD_NMT": "OFF", "NEMO_SPEECH_BUILD_S2S": "ON", "NEMO_SPEECH_BUILD_HTTP": "ON", - "NEMO_SPEECH_BUILD_GRPC": "OFF", - "NEMO_SPEECH_WITH_GRPC": "OFF", - "NEMO_SPEECH_WITH_NMT": "OFF" + "NEMO_SPEECH_BUILD_GRPC": "OFF" } }, { @@ -252,9 +257,7 @@ "NEMO_SPEECH_BUILD_TTS": "ON", "NEMO_SPEECH_BUILD_NMT": "ON", "NEMO_SPEECH_BUILD_HTTP": "ON", - "NEMO_SPEECH_BUILD_GRPC": "OFF", - "NEMO_SPEECH_WITH_GRPC": "OFF", - "NEMO_SPEECH_WITH_NMT": "ON" + "NEMO_SPEECH_BUILD_GRPC": "OFF" } }, { @@ -265,9 +268,7 @@ "NEMO_SPEECH_BUILD_TTS": "ON", "NEMO_SPEECH_BUILD_NMT": "ON", "NEMO_SPEECH_BUILD_HTTP": "ON", - "NEMO_SPEECH_BUILD_GRPC": "OFF", - "NEMO_SPEECH_WITH_GRPC": "OFF", - "NEMO_SPEECH_WITH_NMT": "ON" + "NEMO_SPEECH_BUILD_GRPC": "OFF" } }, { @@ -278,8 +279,6 @@ "NEMO_SPEECH_BUILD_TTS": "ON", "NEMO_SPEECH_BUILD_NMT": "ON", "NEMO_SPEECH_BUILD_GRPC": "ON", - "NEMO_SPEECH_WITH_NMT": "ON", - "NEMO_SPEECH_WITH_GRPC": "ON", "NEMO_SPEECH_WITH_FLASHLIGHT": "ON", "NEMO_SPEECH_WITH_NORM": "ON", "NEMO_SPEECH_TTS_WITH_JA": "ON", diff --git a/CONTRIBUTING.md b/CONTRIBUTING.md index f288c62..fa83270 100644 --- a/CONTRIBUTING.md +++ b/CONTRIBUTING.md @@ -24,7 +24,7 @@ Follow the [source-build guide](docs/build.md) for prerequisites and submodules. For a model-independent CPU ASR test build: ```bash -git submodule update --init ggml llama.cpp +git submodule update --init llama.cpp scripts/configure.sh cpu-asr -DNEMO_SPEECH_BUILD_TESTS=ON cmake --build --preset cpu-asr ctest --test-dir build/cpu-asr --output-on-failure @@ -41,6 +41,12 @@ Use the closest matching CUDA, Metal, Vulkan, server, or component preset when the change affects code outside the CPU ASR path. Include the commands and results relevant to the change in the pull request. +## Changing llama.cpp or ggml + +The `llama.cpp` submodule stays at its pinned upstream commit. Changes to it, +including ggml, live as patches in [`patches/`](patches/README.md), which +explains the patch format and the `scripts/llama-patches.sh` workflow. + ## Contribution license and provenance Unless a file states otherwise, contributions are submitted under the diff --git a/README.md b/README.md index 498c37a..a119077 100644 --- a/README.md +++ b/README.md @@ -156,7 +156,7 @@ development files, and the toolchain required by the selected backend, if any. For a CUDA ASR and TTS server with the playground: ```bash -git submodule update --init ggml llama.cpp third_party/cpp-httplib +git submodule update --init llama.cpp third_party/cpp-httplib scripts/configure.sh cuda-server cmake --build --preset cuda-server ``` diff --git a/THIRD_PARTY_NOTICES.md b/THIRD_PARTY_NOTICES.md index e15abb7..282bd20 100644 --- a/THIRD_PARTY_NOTICES.md +++ b/THIRD_PARTY_NOTICES.md @@ -9,10 +9,11 @@ checkouts and are summarized here. ### ggml -- Source: [`ggml-org/ggml`](https://github.com/ggml-org/ggml) -- Path: `ggml` +- Source: [`ggml-org/llama.cpp`](https://github.com/ggml-org/llama.cpp) (`ggml/`, + developed and vendored in llama.cpp) +- Path: `llama.cpp/ggml` - Copyright (c) 2023-2026 The ggml authors -- License: MIT; upstream text: [`ggml/LICENSE`](ggml/LICENSE) +- License: MIT; upstream text: [`llama.cpp/LICENSE`](llama.cpp/LICENSE) ### llama.cpp @@ -220,9 +221,9 @@ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE. -## NVIDIA ggml patches +## NVIDIA llama.cpp and ggml patches -The `ggml-patches/` directory contains NVIDIA-authored changes applied to the -pinned MIT-licensed ggml source. New source files created by those patches -carry the NVIDIA Apache-2.0 header; existing ggml files retain their upstream -notices. The resulting combined source and binaries retain ggml's MIT notice. +The `patches/` directory contains NVIDIA-authored changes applied to the pinned +MIT-licensed llama.cpp and ggml source. New source files created by those patches +carry the NVIDIA Apache-2.0 header; existing files retain their upstream +notices. The resulting combined source and binaries retain the MIT notice. diff --git a/app/CMakeLists.txt b/app/CMakeLists.txt index 9b54fda..be7cfbc 100644 --- a/app/CMakeLists.txt +++ b/app/CMakeLists.txt @@ -31,18 +31,13 @@ if(NEMO_SPEECH_BUILD_ASR) target_compile_definitions(nemo_speech_cli PRIVATE NEMO_SPEECH_CLI_ASR=1) target_link_libraries(nemo_speech_cli PRIVATE nemo_speech_asr) if(NEMO_SPEECH_BUILD_MIC_CAPTURE) - if(NOT EXISTS "${CMAKE_SOURCE_DIR}/llama.cpp/vendor/miniaudio/miniaudio.h") - message(FATAL_ERROR - "ASR CLI microphone capture requires the vendored miniaudio header; " - "run: git submodule update --init llama.cpp") - endif() target_sources(nemo_speech_cli PRIVATE live_terminal.cpp live_transcript.cpp microphone_capture.cpp) target_compile_definitions(nemo_speech_cli PRIVATE NEMO_SPEECH_CLI_LIVE=1) target_include_directories(nemo_speech_cli PRIVATE - ${CMAKE_SOURCE_DIR}/llama.cpp/vendor/miniaudio) + ${NEMO_SPEECH_LLAMA_CPP_DIR}/vendor/miniaudio) target_link_libraries(nemo_speech_cli PRIVATE ${CMAKE_DL_LIBS}) if(APPLE) target_link_libraries(nemo_speech_cli PRIVATE diff --git a/cmake/llama_cpp.cmake b/cmake/llama_cpp.cmake new file mode 100644 index 0000000..050db26 --- /dev/null +++ b/cmake/llama_cpp.cmake @@ -0,0 +1,199 @@ +# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-License-Identifier: Apache-2.0 +# Resolve the llama.cpp source tree, which also provides ggml. +# +# The llama.cpp submodule is kept pristine. When NEMO_SPEECH_GGML_PATCHED is ON +# the patches/ series is applied to a copy of its pinned commit in the build +# directory, and the copy is refreshed only when the series or the pin changes. +# A refresh rewrites only the files whose content changed, so editing one patch +# rebuilds only what that edit affects. +# +# Also runs as a script, for tools that need the patched tree outside a build: +# cmake -DSOURCE_DIR=llama.cpp -DPATCH_DIR=patches -DDEST_DIR= -P cmake/llama_cpp.cmake + +# The patches listed in PATCH_DIR/series, in apply order. Every *.patch file in +# PATCH_DIR must be listed, so a new patch cannot be silently left out. +function(nemo_speech_llama_cpp_series patch_dir out_var) + if(NOT EXISTS "${patch_dir}/series") + message(FATAL_ERROR "missing ${patch_dir}/series") + endif() + file(STRINGS "${patch_dir}/series" lines) + set(patches "") + foreach(line IN LISTS lines) + string(REGEX REPLACE "#.*$" "" line "${line}") + string(STRIP "${line}" line) + if(line STREQUAL "") + continue() + endif() + if(NOT EXISTS "${patch_dir}/${line}") + message(FATAL_ERROR "${patch_dir}/series lists ${line}, which does not exist") + endif() + list(APPEND patches "${patch_dir}/${line}") + endforeach() + file(GLOB present LIST_DIRECTORIES false "${patch_dir}/*.patch") + foreach(patch IN LISTS present) + list(FIND patches "${patch}" index) + if(index EQUAL -1) + message(FATAL_ERROR "${patch} is not listed in ${patch_dir}/series") + endif() + endforeach() + set(${out_var} "${patches}" PARENT_SCOPE) +endfunction() + +function(nemo_speech_materialize_llama_cpp source_dir patch_dir dest_dir) + find_package(Git QUIET) + if(NOT GIT_EXECUTABLE) + message(FATAL_ERROR "git is required to apply ${patch_dir} to llama.cpp") + endif() + + nemo_speech_llama_cpp_series("${patch_dir}" patches) + + # Key the copy by the pinned commit and the series content. Source archives + # without git metadata are keyed by the content of every file the copy + # takes, so replacing the tree with another llama.cpp version refreshes it. + execute_process( + COMMAND "${GIT_EXECUTABLE}" -C "${source_dir}" rev-parse HEAD + OUTPUT_VARIABLE base RESULT_VARIABLE rc OUTPUT_STRIP_TRAILING_WHITESPACE ERROR_QUIET) + set(have_git_metadata OFF) + if(rc EQUAL 0) + set(have_git_metadata ON) + else() + file(GLOB_RECURSE tree_files RELATIVE "${source_dir}" LIST_DIRECTORIES false "${source_dir}/*") + list(FILTER tree_files EXCLUDE REGEX "^(\\.git|models|docs|media)(/|$)") + list(SORT tree_files) + set(tree_listing "") + foreach(path IN LISTS tree_files) + file(SHA256 "${source_dir}/${path}" digest) + string(APPEND tree_listing "${path} ${digest}\n") + endforeach() + string(SHA256 base "${tree_listing}") + endif() + set(stamp_input "${base}") + foreach(patch IN LISTS patches) + file(SHA256 "${patch}" digest) + string(APPEND stamp_input "-${digest}") + endforeach() + string(SHA256 stamp "${stamp_input}") + + set(stamp_file "${dest_dir}.stamp") + if(EXISTS "${stamp_file}" AND EXISTS "${dest_dir}") + file(READ "${stamp_file}" previous_stamp) + if(previous_stamp STREQUAL stamp) + return() + endif() + endif() + file(REMOVE "${stamp_file}") + + list(LENGTH patches n_patches) + message(STATUS "Applying ${n_patches} patches from ${patch_dir} to llama.cpp in ${dest_dir}") + # Build the new tree in a staging directory, then copy over only what changed. + set(staging "${dest_dir}.staging") + file(REMOVE_RECURSE "${staging}") + file(MAKE_DIRECTORY "${staging}") + # Model and documentation assets are not needed to build. + if(have_git_metadata) + # Only tracked files at the pinned commit: local edits and build + # directories inside the submodule stay out of the copy. + execute_process( + COMMAND "${GIT_EXECUTABLE}" -C "${source_dir}" archive --format=tar -o "${staging}.tar" HEAD + -- . ":(exclude)models" ":(exclude)docs" ":(exclude)media" + RESULT_VARIABLE rc) + if(rc EQUAL 0) + execute_process(COMMAND "${CMAKE_COMMAND}" -E tar xf "${staging}.tar" + WORKING_DIRECTORY "${staging}" RESULT_VARIABLE rc) + endif() + file(REMOVE "${staging}.tar") + if(NOT rc EQUAL 0) + message(FATAL_ERROR "could not export llama.cpp ${base} from ${source_dir}") + endif() + else() + file(GLOB entries RELATIVE "${source_dir}" "${source_dir}/*") + foreach(entry IN LISTS entries) + if(NOT entry MATCHES "^(\\.git|models|docs|media)$") + file(COPY "${source_dir}/${entry}" DESTINATION "${staging}") + endif() + endforeach() + endif() + + # A build directory usually sits inside another git work tree; stop git from + # discovering it so paths apply relative to the copy. + get_filename_component(dest_parent "${dest_dir}" DIRECTORY) + set(normalized "${staging}.patch") + foreach(patch IN LISTS patches) + # Windows checkouts may carry CRLF line endings. + file(READ "${patch}" content) + string(REPLACE "\r\n" "\n" content "${content}") + file(WRITE "${normalized}" "${content}") + execute_process( + COMMAND "${CMAKE_COMMAND}" -E env "GIT_CEILING_DIRECTORIES=${dest_parent}" + "${GIT_EXECUTABLE}" apply --whitespace=nowarn "${normalized}" + WORKING_DIRECTORY "${staging}" + RESULT_VARIABLE rc ERROR_VARIABLE error) + if(NOT rc EQUAL 0) + get_filename_component(name "${patch}" NAME) + message(FATAL_ERROR + "${name} does not apply to llama.cpp ${base}:\n${error}\n" + "Restore the pinned submodule (git submodule update llama.cpp) or rebase the " + "series with scripts/llama-patches.sh rebase.") + endif() + endforeach() + file(REMOVE "${normalized}") + + # Unchanged files keep their timestamps, so the build recompiles only what changed. + file(GLOB_RECURSE new_files RELATIVE "${staging}" LIST_DIRECTORIES false "${staging}/*") + foreach(path IN LISTS new_files) + get_filename_component(dir "${dest_dir}/${path}" DIRECTORY) + file(MAKE_DIRECTORY "${dir}") + file(COPY_FILE "${staging}/${path}" "${dest_dir}/${path}" ONLY_IF_DIFFERENT) + endforeach() + file(GLOB_RECURSE old_files RELATIVE "${dest_dir}" LIST_DIRECTORIES false "${dest_dir}/*") + foreach(path IN LISTS old_files) + if(NOT EXISTS "${staging}/${path}") + file(REMOVE "${dest_dir}/${path}") + endif() + endforeach() + file(REMOVE_RECURSE "${staging}") + file(WRITE "${stamp_file}" "${stamp}") +endfunction() + +if(CMAKE_SCRIPT_MODE_FILE) + foreach(var SOURCE_DIR PATCH_DIR DEST_DIR) + if(NOT DEFINED ${var}) + message(FATAL_ERROR "usage: cmake -DSOURCE_DIR=... -DPATCH_DIR=... -DDEST_DIR=... -P ${CMAKE_SCRIPT_MODE_FILE}") + endif() + get_filename_component(${var} "${${var}}" ABSOLUTE) + endforeach() + nemo_speech_materialize_llama_cpp("${SOURCE_DIR}" "${PATCH_DIR}" "${DEST_DIR}") + return() +endif() + +set(NEMO_SPEECH_LLAMA_CPP_SOURCE_DIR "" CACHE PATH + "Build ggml and llama.cpp from this tree as-is (for example a scripts/llama-patches.sh edit worktree)") +set(_nemo_speech_llama_cpp_submodule "${CMAKE_SOURCE_DIR}/llama.cpp") +if(NEMO_SPEECH_LLAMA_CPP_SOURCE_DIR) + set(NEMO_SPEECH_LLAMA_CPP_DIR "${NEMO_SPEECH_LLAMA_CPP_SOURCE_DIR}") +elseif(NOT EXISTS "${_nemo_speech_llama_cpp_submodule}/ggml/CMakeLists.txt") + message(FATAL_ERROR + "The llama.cpp submodule (which also provides ggml) is not initialized.\n" + "Run: git submodule update --init llama.cpp") +elseif(NEMO_SPEECH_GGML_PATCHED) + set(NEMO_SPEECH_LLAMA_CPP_DIR "${CMAKE_BINARY_DIR}/_deps/llama.cpp") + nemo_speech_materialize_llama_cpp( + "${_nemo_speech_llama_cpp_submodule}" "${CMAKE_SOURCE_DIR}/patches" "${NEMO_SPEECH_LLAMA_CPP_DIR}") + # Re-run configuration when patches are edited, added, removed, or reordered, + # or when the submodule moves to another commit. + nemo_speech_llama_cpp_series("${CMAKE_SOURCE_DIR}/patches" _nemo_speech_patches) + set_property(DIRECTORY APPEND PROPERTY CMAKE_CONFIGURE_DEPENDS + "${CMAKE_SOURCE_DIR}/patches" "${CMAKE_SOURCE_DIR}/patches/series" ${_nemo_speech_patches}) + execute_process( + COMMAND "${GIT_EXECUTABLE}" -C "${_nemo_speech_llama_cpp_submodule}" rev-parse --absolute-git-dir + OUTPUT_VARIABLE _nemo_speech_llama_cpp_git_dir RESULT_VARIABLE _nemo_speech_rc + OUTPUT_STRIP_TRAILING_WHITESPACE ERROR_QUIET) + if(_nemo_speech_rc EQUAL 0 AND EXISTS "${_nemo_speech_llama_cpp_git_dir}/HEAD") + set_property(DIRECTORY APPEND PROPERTY CMAKE_CONFIGURE_DEPENDS + "${_nemo_speech_llama_cpp_git_dir}/HEAD") + endif() +else() + set(NEMO_SPEECH_LLAMA_CPP_DIR "${_nemo_speech_llama_cpp_submodule}") +endif() +message(STATUS "ggml and llama.cpp source: ${NEMO_SPEECH_LLAMA_CPP_DIR}") diff --git a/conversion/s2s_components/voicechat_source.py b/conversion/s2s_components/voicechat_source.py index 7bf4c85..b5e1e41 100644 --- a/conversion/s2s_components/voicechat_source.py +++ b/conversion/s2s_components/voicechat_source.py @@ -265,18 +265,36 @@ def ensure_quantizer( "C++ compiler; install the build prerequisites or pass --llama-quantize PATH" ) + source = checkout if checkout == (root / "llama.cpp").resolve(): - patch_script = root / "scripts" / "apply-llama-patches.sh" - if not patch_script.is_file(): - raise RuntimeError(f"missing llama.cpp patch helper: {patch_script}") - print("[convert-s2s] applying pinned llama.cpp compatibility patches") - subprocess.run(["bash", str(patch_script)], cwd=root, check=True) + # Build from a patched copy; the submodule itself stays pristine. + materialize = root / "cmake" / "llama_cpp.cmake" + if not materialize.is_file(): + raise RuntimeError(f"missing llama.cpp patch helper: {materialize}") + source = root / ".deps" / "llama.cpp-patched" + print("[convert-s2s] applying patches/ to a copy of the pinned llama.cpp") + subprocess.run( + [ + cmake, + f"-DSOURCE_DIR={checkout}", + f"-DPATCH_DIR={root / 'patches'}", + f"-DDEST_DIR={source}", + "-P", + str(materialize), + ], + check=True, + ) build_dir = checkout / "build-quantize" + cache = build_dir / "CMakeCache.txt" + if cache.is_file() and f"CMAKE_HOME_DIRECTORY:INTERNAL={source}\n" not in cache.read_text(): + # Configured from a different source tree; CMake cannot switch sources + # in place. + shutil.rmtree(build_dir) configure = [ cmake, "-S", - str(checkout), + str(source), "-B", str(build_dir), "-DCMAKE_BUILD_TYPE=Release", diff --git a/docker/Dockerfile b/docker/Dockerfile index cfcb151..7a7fdde 100644 --- a/docker/Dockerfile +++ b/docker/Dockerfile @@ -39,9 +39,8 @@ ARG ENABLE_TESTS=OFF ARG ENABLE_NMT=OFF # Full-duplex VoiceChat via llama.cpp. Set ON to build realtime server support. ARG ENABLE_S2S=OFF -# Apply the project ggml patches and build the ASR against them. Set OFF to -# build the ASR against STANDARD upstream ggml (stock ops only, no patches): -# skips apply-ggml-patches.sh and sets -DNEMO_SPEECH_GGML_PATCHED=OFF. +# Build ggml and llama.cpp with patches/ applied (CMake applies them). Set OFF to +# build against pristine upstream llama.cpp (stock ggml ops only). ARG ENABLE_GGML_PATCHES=ON ARG JOBS= # GPU arch(es) for ggml-cuda. Empty = ggml's portable default (sm_86/89/120/121 @@ -120,12 +119,9 @@ COPY README.md CONTRIBUTING.md /work/ COPY docs /work/docs COPY config /work/config COPY models/index.json /work/models/index.json -# Core submodules: ggml (every backend) and llama.cpp (NMT and VoiceChat; -# copied unconditionally so the layer cache is stable). -COPY ggml /work/ggml +# llama.cpp provides ggml for every backend (and libllama for NMT and VoiceChat). COPY llama.cpp /work/llama.cpp -COPY ggml-patches /work/ggml-patches -COPY llama-patches /work/llama-patches +COPY patches /work/patches COPY proto /work/proto COPY include /work/include COPY src /work/src @@ -161,19 +157,6 @@ RUN if [ "${ENABLE_FLASHLIGHT}" = "ON" ]; then \ && rm -rf /work/.deps/sentencepiece-build; \ fi -# Apply the project ggml patches (fused rel-pos attention, NVFP4 quantization, -# norm fusion, dw-conv F16, skinny-q8, BF16 fusions, and CUDA correctness) onto the clean -# vendored ggml. Idempotent: a working tree that already has them applied is -# detected and skipped. Skipped entirely when ENABLE_GGML_PATCHES=OFF — the ASR -# then builds against standard upstream ggml (see NEMO_SPEECH_GGML_PATCHED). -RUN if [ "${ENABLE_GGML_PATCHES}" = "ON" ]; then bash /work/scripts/apply-ggml-patches.sh; fi - -# NMT and VoiceChat rely on the project's llama.cpp compatibility and -# quantization patches. The patch script is idempotent for prepatched trees. -RUN if [ "${ENABLE_NMT}" = "ON" ] || [ "${ENABLE_S2S}" = "ON" ]; then \ - bash /work/scripts/apply-llama-patches.sh; \ - fi - # NCCL disabled: single-GPU inference does not use multi-GPU collectives, # and the linked libnccl is otherwise dead runtime weight. # NEMO_SPEECH_CUBLAS_SHIM=ON builds the drop-in libcublas.so. @@ -229,7 +212,7 @@ RUN mkdir -p /out/bin /out/lib /out/share/nemo-speech \ && install -Dm0644 /work/build/share/nemo-speech/model-index.json \ /out/share/nemo-speech/model-index.json \ && license_dir=/out/share/licenses/nemo-speech/third_party \ - && install -Dm0644 /work/ggml/LICENSE "$license_dir/ggml/LICENSE" \ + && install -Dm0644 /work/llama.cpp/LICENSE "$license_dir/ggml/LICENSE" \ && install -Dm0644 /work/llama.cpp/LICENSE "$license_dir/llama.cpp/LICENSE" \ && if [ "${ENABLE_GRPC}" = "ON" ]; then \ install -Dm0644 /work/proto/riva-common/LICENSE \ diff --git a/docs/README.md b/docs/README.md index b98cee9..2b4b492 100644 --- a/docs/README.md +++ b/docs/README.md @@ -50,9 +50,10 @@ Start with: ## Developer guide - [Overview](development/README.md) - implementation and performance internals. -- [Diagnostics](development/diagnostics.md) - `check_backend_coverage`. +- [Diagnostics](development/diagnostics.md) - build switches, runtime knobs, and + `check_backend_coverage`. - [ASR batching](development/asr-batching.md) - neural microbatching and streaming-state arenas. -- [ggml patches](development/ggml-patches.md) - the project-specific ggml changes. +- [llama.cpp and ggml patches](../patches/README.md) - the project-specific changes. - [cuBLAS shim](development/cublas-shim.md) - the in-tree drop-in cuBLAS replacement. diff --git a/docs/build.md b/docs/build.md index e313d72..1c8d91e 100644 --- a/docs/build.md +++ b/docs/build.md @@ -72,9 +72,8 @@ only when selecting those features; see the platform sections below. Initialize the submodules needed by the selected components: ```bash -git submodule update --init ggml +git submodule update --init llama.cpp # required (also provides ggml) git submodule update --init third_party/cpp-httplib # HTTP server only -git submodule update --init llama.cpp # ASR live capture, NMT, or VoiceChat git submodule update --init proto/riva-common # gRPC only git submodule update --init third_party/flashlight-text third_party/kenlm # Flashlight only git submodule update --init third_party/open_jtalk # Japanese TTS only @@ -82,8 +81,8 @@ git submodule update --init --recursive third_party/cppjieba # Mandarin TTS onl ``` Always configure through `scripts/configure.sh`. It checks the required -submodules and applies the patches needed by the selected preset. CUDA presets -apply the pinned patches from `ggml-patches/` in order. Mandarin TTS also +submodules and runs the preset. CMake applies the project's llama.cpp and ggml +changes from `patches/` itself, so no separate patch step is needed. Mandarin TTS also requires the Git LFS files under `src/tts/tokenizer/mandarin_data/`; the helper reports any files that are still LFS pointers. Materialize them with `git lfs pull --include='src/tts/tokenizer/mandarin_data/*'`. @@ -126,8 +125,8 @@ scripts/configure.sh cuda-server -DNEMO_SPEECH_WITH_NORM=ON cmake --build --preset cuda-server ``` -For a manual configuration, explicitly disable the patched CUDA paths when -building against stock ggml: +A raw CMake configuration also applies `patches/` automatically. To build +against pristine upstream llama.cpp and ggml instead: ```bash cmake -S . -B build -G Ninja \ @@ -136,8 +135,8 @@ cmake -S . -B build -G Ninja \ cmake --build build -j"$(nproc)" ``` -See [ggml patches](development/ggml-patches.md) for the patched and stock -runtime tradeoffs. +See [`patches/README.md`](../patches/README.md) for what the patches change and +how to edit them. ## Components diff --git a/docs/development/README.md b/docs/development/README.md index c6598c2..bc3896f 100644 --- a/docs/development/README.md +++ b/docs/development/README.md @@ -7,12 +7,12 @@ the server want [ASR configuration](../asr/configuration.md), ## Contents -- [`diagnostics.md`](diagnostics.md) - `check_backend_coverage`, catching silent - CPU fallbacks on a new backend. +- [`diagnostics.md`](diagnostics.md) - build switches, runtime knobs, and + `check_backend_coverage` for catching silent CPU fallbacks on a new backend. - [`asr-batching.md`](asr-batching.md) - exact-shape neural microbatching and indexed streaming-state arenas. -- [`ggml-patches.md`](ggml-patches.md) - the project-specific ggml patches and how - they are applied at build setup. +- [`patches/README.md`](../../patches/README.md) - the project's llama.cpp and ggml + patches, how builds apply them, and how to edit them. - [`cublas-shim.md`](cublas-shim.md) - the in-tree drop-in cuBLAS replacement under `kernels/` and where the custom GPU kernels live. - [Windows build notes](windows-build.md) diff --git a/docs/development/asr-batching.md b/docs/development/asr-batching.md index a6b9141..d422d4d 100644 --- a/docs/development/asr-batching.md +++ b/docs/development/asr-batching.md @@ -134,6 +134,6 @@ summaries. submissions rather than concurrent access to ggml's shared scheduler. - Disabling batching keeps the scalar path and avoids its queue-delay cost. -For planar-Q8 kernel behavior, diagnostic environment switches, and patched -versus stock ggml builds, see [ggml patches](ggml-patches.md). The complete +For planar-Q8 kernel behavior and patched versus stock ggml builds, see +[`patches/README.md`](../../patches/README.md) and the patch descriptions there. The complete runtime key reference is in [ASR configuration](../asr/configuration.md). diff --git a/docs/development/cublas-shim.md b/docs/development/cublas-shim.md index 4e6a02a..491129f 100644 --- a/docs/development/cublas-shim.md +++ b/docs/development/cublas-shim.md @@ -49,5 +49,5 @@ On Windows: The heavier project-specific CUDA kernels (fused rel-pos attention, skinny-Q8 GEMM, NVFP4 quantization, BF16 FastConformer epilogues, fused LayerNorm, and F16 depthwise conv2d) live as ggml patches rather than in `kernels/` - see -[ggml patches](ggml-patches.md). `kernels/` holds only the cuBLAS shim and its +[`patches/`](../../patches/README.md). `kernels/` holds only the cuBLAS shim and its version-map template. diff --git a/docs/development/diagnostics.md b/docs/development/diagnostics.md index d4c378a..8bd484f 100644 --- a/docs/development/diagnostics.md +++ b/docs/development/diagnostics.md @@ -1,13 +1,52 @@ -# Backend coverage diagnostic +# Build switches, runtime knobs, and diagnostics + +Every switch that changes which code path runs, in one place. Defaults are what +you want; the rest exist for bisection and debugging. + +## Build + +| CMake option | Default | Effect | +|---|---|---| +| `NEMO_SPEECH_GGML_PATCHED` | `ON` (`cuda-*`, `cpu-*` presets), `OFF` (`metal-*`, `vulkan-*`) | Apply [`patches/`](../../patches/README.md) to llama.cpp and ggml and, on CUDA, use the fused kernels. `OFF` builds pristine upstream llama.cpp with stock ggml operations. | +| `NEMO_SPEECH_LLAMA_CPP_SOURCE_DIR` | unset | Build ggml and llama.cpp from this tree as-is (for example a `scripts/llama-patches.sh edit` worktree). | +| `GGML_CUDA_GRAPHS` | `ON` | Capture and replay CUDA graphs (upstream option; this project turns it on). | +| `GGML_LLAMAFILE` | `ON` | Tiled CPU matrix multiplication (upstream option; this project turns it on). | +| `NEMO_SPEECH_CUBLAS_SHIM` | `OFF` | Build the drop-in cuBLAS replacement ([cuBLAS shim](cublas-shim.md)). | + +Component options (`NEMO_SPEECH_BUILD_*`, `NEMO_SPEECH_WITH_*`) are listed in +[the build guide](../build.md). `NEMO_SPEECH_WITH_NMT` and `NEMO_SPEECH_WITH_GRPC` +are deprecated spellings of `NEMO_SPEECH_BUILD_NMT` and `NEMO_SPEECH_BUILD_GRPC`. + +## Runtime (patched CUDA backend) + +| Variable | Default | Effect | +|---|---|---| +| `GGML_CUDA_GRAPH_EVICT_AFTER_MS` | `10000`; `0` once a server loads TTS, and for VoiceChat | Drop CUDA graphs idle for this long; `0` keeps them. Read when a CUDA backend is created. | +| `GGML_SKINNY_Q8=0` | enabled | Disable the skinny Q8_0 GEMM for block Q8_0 weights. Planar Q8 weights always use it. | +| `GGML_SKINNY_Q8_INPLACE=0` | in place; `0` when ASR and NMT share a process | Keep repacked skinny-Q8 weights in a separate buffer. | + +Upstream switches that are useful for bisecting a CUDA problem: +`GGML_CUDA_DISABLE_GRAPHS=1`, `GGML_CUDA_DISABLE_FUSION=1`, and +`GGML_SCHED_DEBUG=2`. + +## Other variables + +- `NEMO_SPEECH_MODEL_DIR`, `NEMO_SPEECH_MODEL_INDEX`, `NEMO_SPEECH_HF_BASE_URL`: the CLI model store. +- `NEMO_SPEECH_`: overrides one configuration key. +- `S2S_*`: VoiceChat tuning; see [VoiceChat configuration](../s2s/configuration.md). +- `MAGPIETTS_LOGIT_DUMP` and `MAGPIETTS_FORCE_CODES`: file paths used for MagpieTTS parity testing. +- `EDGE_SHIM_TRACE_SHAPES`: logs every call in the cuBLAS shim. + +## Backend coverage `check_backend_coverage` loads an ASR GGUF and exercises the frontend and encoder Sessions used by its CTC or streaming-transducer path, including the compact CTC head, RNNT/TDT predictor and joint, and cache-aware encoder when applicable. It then prints their per-op backend assignment. -Use it to catch **silent CPU fallbacks** when enabling a new GPU backend - a -single fallback op mid-graph adds a GPU↔CPU roundtrip per audio chunk and can -significantly increase streaming latency. The lazy offline transducer path is -outside this diagnostic's coverage. +Use it to catch **silent CPU fallbacks** when enabling a new GPU backend or +updating llama.cpp - a single fallback op mid-graph adds a GPU↔CPU roundtrip per +audio chunk and can significantly increase streaming latency. The lazy offline +transducer path is outside this diagnostic's coverage. ```bash scripts/configure.sh cuda-asr -DNEMO_SPEECH_BUILD_TOOLS=ON diff --git a/docs/development/ggml-patches.md b/docs/development/ggml-patches.md deleted file mode 100644 index f408919..0000000 --- a/docs/development/ggml-patches.md +++ /dev/null @@ -1,7 +0,0 @@ -# ggml patches - -The canonical documentation for the patch series, build modes, runtime gates, -and patch-regeneration workflow is -[`ggml-patches/README.md`](../../ggml-patches/README.md). - -This page intentionally contains no duplicated patch details. diff --git a/docs/development/windows-build.md b/docs/development/windows-build.md index 8993ca9..48c8eae 100644 --- a/docs/development/windows-build.md +++ b/docs/development/windows-build.md @@ -49,9 +49,8 @@ defaults. ## Get the sources ```powershell -git submodule update --init ggml # required (all backends) +git submodule update --init llama.cpp # required (also provides ggml) git submodule update --init proto/riva-common # gRPC server -git submodule update --init llama.cpp # ASR live capture or NMT git submodule update --init third_party/flashlight-text third_party/kenlm # only for LM-fused CTC decoding git submodule update --init third_party/open_jtalk # optional TTS JA tokenizer (-TtsJa) git submodule update --init --recursive third_party/cppjieba # optional TTS ZH tokenizer (-TtsZh) @@ -111,9 +110,7 @@ If you prefer to drive CMake yourself, run from an **x64 Native Tools** prompt (or after `vcvars64.bat`), with CMake/Ninja/CUDA/Vulkan on `PATH`: ```powershell -# CUDA: apply the CUDA-only ggml patches first -powershell -ExecutionPolicy Bypass -File scripts\windows\apply-ggml-patches.ps1 - +# CUDA: CMake applies patches/ itself (git must be on PATH). cmake -S . -B build-cuda -G Ninja -DCMAKE_BUILD_TYPE=Release ` -DGGML_CUDA=ON -DCMAKE_CUDA_ARCHITECTURES=native ` -DNEMO_SPEECH_BUILD_GRPC=ON ` @@ -121,7 +118,7 @@ cmake -S . -B build-cuda -G Ninja -DCMAKE_BUILD_TYPE=Release ` -DVCPKG_TARGET_TRIPLET=x64-windows cmake --build build-cuda --parallel -# Vulkan: stock ggml (the project's ggml patches are CUDA-only). ggml-vulkan requires the +# Vulkan: stock ggml (the patches are applied for CUDA and CPU builds). ggml-vulkan requires the # SPIRV-Headers CMake package; the Vulkan SDK ships it under Lib\cmake. cmake -S . -B build-vulkan -G Ninja -DCMAKE_BUILD_TYPE=Release ` -DGGML_VULKAN=ON -DNEMO_SPEECH_GGML_PATCHED=OFF ` @@ -136,25 +133,13 @@ cmake --build build-vulkan --parallel - **The cuBLAS shim is optional.** Pass `-CublasShim` to the build driver for an app-local `cublas64_.dll` that avoids shipping cuBLAS and cuBLASLt. -- **ggml patches are CUDA-only.** A Vulkan/CPU build uses stock ggml; pass - `-DNEMO_SPEECH_GGML_PATCHED=OFF` (the encoder uses the portable op path). +- **CUDA and CPU builds apply the patches.** A Vulkan build uses stock ggml; + pass `-DNEMO_SPEECH_GGML_PATCHED=OFF` (the build driver does this). - Dependent DLLs must be next to the executable or on `PATH`. Ninja places them together in `build-\bin`. - Flashlight builds install the replaceable `kenlm.dll` alongside the runtime libraries. -### Reset a partially patched ggml checkout - -If `apply-ggml-patches` reports that a patch does not apply cleanly, reset the -submodule and re-apply it: - -```powershell -git -C ggml reset -q -git -C ggml checkout -- . -git -C ggml clean -fd src # removes patch-created files -powershell -ExecutionPolicy Bypass -File scripts\windows\apply-ggml-patches.ps1 -``` - ### Backend status | Backend | ASR | TTS | NMT | diff --git a/docs/s2s/README.md b/docs/s2s/README.md index 23bd664..baebfec 100644 --- a/docs/s2s/README.md +++ b/docs/s2s/README.md @@ -27,7 +27,7 @@ See [Build from source](../build.md) for other platforms and toolchains. ```bash git clone https://github.com/NVIDIA/NeMo-Speech.cpp.git cd NeMo-Speech.cpp -git submodule update --init ggml llama.cpp third_party/cpp-httplib +git submodule update --init llama.cpp third_party/cpp-httplib python3 -m venv .venv . .venv/bin/activate diff --git a/docs/tts/models.md b/docs/tts/models.md index 36daab7..3a0dc27 100644 --- a/docs/tts/models.md +++ b/docs/tts/models.md @@ -16,7 +16,7 @@ options are omitted. Hugging Face: [nvidia/magpie_tts_multilingual_357m](https://huggingface.co/nvidia/magpie_tts_multilingual_357m) -Use `nemo-speech pull magpie` to download the GGUF and matching tokenizer +Use `nemo-speech pull magpie` to download the F16 GGUF and matching tokenizer assets pinned by the model index. For manually managed checkpoints, download the GGUF and `.nemo` archive from the same Hugging Face revision and pass their local paths explicitly. @@ -26,8 +26,9 @@ frame-stacking factor of 2. This is independent of `tts.chunk-frames`, which groups generated frames for NanoCodec streaming. Both versions use the same NanoCodec decoder. -The fused CUDA decode path needs a Q8_0 GGUF. Convert the `.nemo` locally -(Q8_0 is the converter default): +The fused CUDA decode path, which the published TTS benchmarks use, needs a +Q8_0 GGUF; the pulled F16 GGUF runs the unfused path. Convert the `.nemo` +locally (Q8_0 is the converter default): ```bash python3 convert_model.py magpie_tts_multilingual_357m.nemo \ diff --git a/ggml b/ggml deleted file mode 160000 index c03b4e2..0000000 --- a/ggml +++ /dev/null @@ -1 +0,0 @@ -Subproject commit c03b4e2bcece5134827881af90242086daf75be5 diff --git a/ggml-patches/0001-fused-relpos-attn.patch b/ggml-patches/0001-fused-relpos-attn.patch deleted file mode 100644 index bd4c7cb..0000000 --- a/ggml-patches/0001-fused-relpos-attn.patch +++ /dev/null @@ -1,811 +0,0 @@ -diff --git a/include/ggml.h b/include/ggml.h -index f6725265..82670deb 100644 ---- a/include/ggml.h -+++ b/include/ggml.h -@@ -583,6 +583,8 @@ extern "C" { - - GGML_OP_GLU, - -+ GGML_OP_FUSED_RELPOS_ATTN, -+ - GGML_OP_COUNT, - }; - -@@ -2416,6 +2418,43 @@ extern "C" { - struct ggml_tensor * a, - struct ggml_tensor * sinks); - -+ // Fused FastConformer relative-position multi-head attention. -+ // Replaces the content (K*Qu) + position (P*Qv with rel-shift) + softmax + -+ // context (attn*V) op sequence with one kernel. CUDA-only (CPU/other -+ // backends report unsupported and the unfused graph runs instead). -+ // -+ // Logical shapes, head dim ne[0]=d_k fastest: -+ // q [d_k, q_len, n_head, batch] pre-bias query (Qu/Qv added inside) -+ // k [d_k, kv_len, n_head, batch] -+ // v [d_k, kv_len, n_head, batch] -+ // p [d_k, pos_len, n_head] positional encoding projection, -+ // pos_len >= kv_len + q_len - 1 -+ // bias_u [d_k, n_head] pos_bias_u (content term) -+ // bias_v [d_k, n_head] pos_bias_v (position term) -+ // mask [kv_len] or [kv_len, batch] additive (0 / -inf) key mask, -+ // or NULL shared or per-stream columns -+ // Q/K/V/P may be non-contiguous views as long as each d_k row is -+ // contiguous — e.g. Q sliced from a fused-QKV projection and K/V read -+ // head-split from a feat-major [n_feat, kv] window (the CUDA op derives -+ // all addressing from the tensors' nb[]). bias_u/bias_v/mask must be -+ // contiguous. -+ // Output: [d_k, q_len, n_head, batch] (attention context, pre-output-proj). -+ // With merge_heads=true the output keeps that logical shape but uses a -+ // head-merged memory layout: permute(out, 0, 2, 1, 3) is a contiguous -+ // (n_feat, q_len, batch) matrix, consumable by the output projection -+ // without a copy. -+ GGML_API struct ggml_tensor * ggml_fused_relpos_attn( -+ struct ggml_context * ctx, -+ struct ggml_tensor * q, -+ struct ggml_tensor * k, -+ struct ggml_tensor * v, -+ struct ggml_tensor * p, -+ struct ggml_tensor * bias_u, -+ struct ggml_tensor * bias_v, -+ struct ggml_tensor * mask, -+ float scale, -+ bool merge_heads); -+ - // TODO: needs to be adapted to ggml_flash_attn_ext - GGML_API struct ggml_tensor * ggml_flash_attn_back( - struct ggml_context * ctx, -diff --git a/src/ggml-cpu/ggml-cpu.c b/src/ggml-cpu/ggml-cpu.c -index cd5c61a8..cce0e5a8 100644 ---- a/src/ggml-cpu/ggml-cpu.c -+++ b/src/ggml-cpu/ggml-cpu.c -@@ -1988,6 +1988,10 @@ static void ggml_compute_forward(struct ggml_compute_params * params, struct ggm - { - ggml_compute_forward_flash_attn_ext(params, tensor); - } break; -+ case GGML_OP_FUSED_RELPOS_ATTN: -+ { -+ GGML_ABORT("FUSED_RELPOS_ATTN has no CPU path (CUDA-only)"); -+ } break; - case GGML_OP_FLASH_ATTN_BACK: - { - int32_t t = ggml_get_op_params_i32(tensor, 0); -diff --git a/src/ggml-cpu/ggml-cpu.cpp b/src/ggml-cpu/ggml-cpu.cpp -index 128883b4..7429cc45 100644 ---- a/src/ggml-cpu/ggml-cpu.cpp -+++ b/src/ggml-cpu/ggml-cpu.cpp -@@ -439,6 +439,8 @@ static bool ggml_backend_cpu_device_supports_op(ggml_backend_dev_t dev, const st - } - - switch (op->op) { -+ case GGML_OP_FUSED_RELPOS_ATTN: -+ return false; // CUDA-only; CPU falls back to the unfused graph - case GGML_OP_CPY: - case GGML_OP_SET_ROWS: - return -diff --git a/src/ggml-cuda/fused-relpos-attn.cu b/src/ggml-cuda/fused-relpos-attn.cu -new file mode 100644 -index 00000000..f3c4836a ---- /dev/null -+++ b/src/ggml-cuda/fused-relpos-attn.cu -@@ -0,0 +1,600 @@ -+// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. -+// SPDX-License-Identifier: Apache-2.0 -+#include "fused-relpos-attn.cuh" -+ -+#include -+ -+// Fused FastConformer relative-position multi-head attention. -+// -+// One block per (head, query, batch); blockDim.x = d_k threads (one per head -+// dim). Scores, rel-shifted position term, scale+mask, two-pass softmax, and -+// the attn*V context are all computed in-kernel, so the rel-shift matrix and -+// the score matrix are never materialized in global memory. -+// -+// Operand addressing is fully stride-driven (strides read from each tensor's -+// nb[] by the host wrapper, in elements): Q/K/V may be non-contiguous views — -+// e.g. the Q slice of a fused-QKV projection, or a feat-major [n_feat, kv] -+// K/V window — as long as d_k stays innermost-contiguous (asserted by the op -+// constructor; the vectorized loads rely on it). P is [d_k, pos_len, n_head] -+// with pos_len = kv + q - 1 rows addressable; bu/bv are [d_k, n_head]; -+// mask is [kv] (shared) or [kv, batch] (per-stream) additive (0 / -inf), or NULL. -+// Output ctx is [d_k, q, n_head, batch] logical; with the merge_heads op flag -+// its memory layout is head-merged ([d_k+h*d_k] innermost, i.e. a plain -+// (n_feat, q, batch) matrix), so the output projection consumes it without a -+// permute copy. -+// -+// Requires d_k to be a power of two (the softmax reduction halves blockDim.x). -+ -+// K/V/P are templated: F16 operands halve the dominant re-read traffic when -+// the caller stages them; F32 operands skip the staging casts entirely. All -+// math stays in F32 either way. -+ -+static constexpr int RELPOS_ATTN_DK_128 = 128; -+static constexpr int RELPOS_ATTN_WARPS_128 = RELPOS_ATTN_DK_128 / 32; -+static constexpr int RELPOS_ATTN_CC_SM100 = 1000; -+ -+static __device__ __forceinline__ float relpos_warp_sum(float value) { -+#pragma unroll -+ for (int offset = 16; offset > 0; offset >>= 1) { -+ value += __shfl_down_sync(0xffffffff, value, offset); -+ } -+ return value; -+} -+ -+static __device__ __forceinline__ float4 relpos_load4(const float * ptr) { -+ return *reinterpret_cast(ptr); -+} -+ -+static __device__ __forceinline__ float4 relpos_load4(const half * ptr) { -+ const int2 packed = *reinterpret_cast(ptr); -+ const half2 * values = reinterpret_cast(&packed); -+ return make_float4( -+ __low2float(values[0]), __high2float(values[0]), -+ __low2float(values[1]), __high2float(values[1])); -+} -+ -+// SM100 streaming specialization for d_k=128 and q=2. Keeping one block per -+// query preserves the two rows' parallelism while the complete grid fits in a -+// resident wave. Compared with the generic shared-memory reduction tree, the -+// score-producing warps retain their local maxima and max/sum use only four -+// warp partials. This removes fourteen block barriers and 124 scratch floats. -+template -+static __global__ void fused_relpos_attn_warp_128_kernel( -+ const float * __restrict__ Q, const T * __restrict__ K, -+ const T * __restrict__ V, const T * __restrict__ Ppos, -+ const float * __restrict__ bu, const float * __restrict__ bv, -+ const float * __restrict__ mask, float * __restrict__ ctx, -+ int kv, float scale, -+ long q_sq, long q_sh, long q_sb, -+ long k_sj, long k_sh, long k_sb, -+ long v_sj, long v_sh, long v_sb, -+ long p_sr, long p_sh, -+ long o_si, long o_sh, long o_sb, long m_sb) { -+ extern __shared__ float sh[]; -+ float * Qu = sh; -+ float * Qv = Qu + RELPOS_ATTN_DK_128; -+ float * sc = Qv + RELPOS_ATTN_DK_128; -+ float * red = sc + kv; -+ -+ const int h = blockIdx.x; -+ const int i = blockIdx.y; -+ const int b = blockIdx.z; -+ const int d = threadIdx.x; -+ const int warp = d >> 5; -+ const int lane = d & 31; -+ -+ const float * Qhi = Q + (size_t) b * q_sb + (size_t) h * q_sh + (size_t) i * q_sq; -+ const T * Kh = K + (size_t) b * k_sb + (size_t) h * k_sh; -+ const T * Vh = V + (size_t) b * v_sb + (size_t) h * v_sh; -+ const T * Ph = Ppos + (size_t) h * p_sh; -+ const float * Mb = mask ? mask + (size_t) b * m_sb : nullptr; -+ -+ Qu[d] = Qhi[d] + bu[h * RELPOS_ATTN_DK_128 + d]; -+ Qv[d] = Qhi[d] + bv[h * RELPOS_ATTN_DK_128 + d]; -+ __syncthreads(); -+ -+ const int d4 = lane * 4; -+ const float4 qu4 = *reinterpret_cast(Qu + d4); -+ const float4 qv4 = *reinterpret_cast(Qv + d4); -+ float produced_max = -INFINITY; -+ for (int j = warp; j < kv; j += RELPOS_ATTN_WARPS_128) { -+ const float4 k4 = relpos_load4(Kh + (size_t) j * k_sj + d4); -+ const int row = 1 + j - i; -+ const float4 p4 = relpos_load4(Ph + (size_t) row * p_sr + d4); -+ float score = -+ k4.x * qu4.x + p4.x * qv4.x + -+ k4.y * qu4.y + p4.y * qv4.y + -+ k4.z * qu4.z + p4.z * qv4.z + -+ k4.w * qu4.w + p4.w * qv4.w; -+ score = relpos_warp_sum(score); -+ if (lane == 0) { -+ sc[j] = score * scale + (Mb ? Mb[j] : 0.0f); -+ produced_max = fmaxf(produced_max, sc[j]); -+ } -+ } -+ if (lane == 0) { -+ red[warp] = produced_max; -+ } -+ __syncthreads(); -+ -+ if (d == 0) { -+ float maximum = red[0]; -+#pragma unroll -+ for (int w = 1; w < RELPOS_ATTN_WARPS_128; ++w) { -+ maximum = fmaxf(maximum, red[w]); -+ } -+ red[0] = maximum; -+ } -+ __syncthreads(); -+ const float maximum = red[0]; -+ -+ // red[] is reused below. Ensure every warp has captured red[0] first; -+ // otherwise warp 0 can overwrite the maximum while another warp loads it. -+ __syncthreads(); -+ float local_sum = 0.0f; -+ for (int j = d; j < kv; j += RELPOS_ATTN_DK_128) { -+ const float weight = __expf(sc[j] - maximum); -+ sc[j] = weight; -+ local_sum += weight; -+ } -+ local_sum = relpos_warp_sum(local_sum); -+ if (lane == 0) { -+ red[warp] = local_sum; -+ } -+ __syncthreads(); -+ if (d == 0) { -+ float sum = red[0]; -+#pragma unroll -+ for (int w = 1; w < RELPOS_ATTN_WARPS_128; ++w) { -+ sum += red[w]; -+ } -+ red[0] = sum; -+ } -+ __syncthreads(); -+ -+ float context = 0.0f; -+ for (int j = 0; j < kv; ++j) { -+ context += sc[j] * (float) Vh[(size_t) j * v_sj + d]; -+ } -+ ctx[(size_t) b * o_sb + (size_t) i * o_si + (size_t) h * o_sh + d] = context / red[0]; -+} -+ -+// Once the one-block-per-query grid spills into another occupancy wave, one -+// block computes both query rows. K and V are then loaded once and reused; -+// only the two relative-position rows differ. The reduction order matches the -+// one-query specialization so changing batch size does not change numerics. -+template -+static __global__ void fused_relpos_attn_q2_warp_128_kernel( -+ const float * __restrict__ Q, const T * __restrict__ K, -+ const T * __restrict__ V, const T * __restrict__ Ppos, -+ const float * __restrict__ bu, const float * __restrict__ bv, -+ const float * __restrict__ mask, float * __restrict__ ctx, -+ int kv, float scale, -+ long q_sq, long q_sh, long q_sb, -+ long k_sj, long k_sh, long k_sb, -+ long v_sj, long v_sh, long v_sb, -+ long p_sr, long p_sh, -+ long o_si, long o_sh, long o_sb, long m_sb) { -+ extern __shared__ float sh[]; -+ float * Qu0 = sh; -+ float * Qv0 = Qu0 + RELPOS_ATTN_DK_128; -+ float * Qu1 = Qv0 + RELPOS_ATTN_DK_128; -+ float * Qv1 = Qu1 + RELPOS_ATTN_DK_128; -+ float * sc0 = Qv1 + RELPOS_ATTN_DK_128; -+ float * sc1 = sc0 + kv; -+ float * red0 = sc1 + kv; -+ float * red1 = red0 + RELPOS_ATTN_WARPS_128; -+ -+ const int h = blockIdx.x; -+ const int b = blockIdx.z; -+ const int d = threadIdx.x; -+ const int warp = d >> 5; -+ const int lane = d & 31; -+ -+ const float * Qh0 = Q + (size_t) b * q_sb + (size_t) h * q_sh; -+ const float * Qh1 = Qh0 + q_sq; -+ const T * Kh = K + (size_t) b * k_sb + (size_t) h * k_sh; -+ const T * Vh = V + (size_t) b * v_sb + (size_t) h * v_sh; -+ const T * Ph = Ppos + (size_t) h * p_sh; -+ const float * Mb = mask ? mask + (size_t) b * m_sb : nullptr; -+ -+ const float bias_u = bu[h * RELPOS_ATTN_DK_128 + d]; -+ const float bias_v = bv[h * RELPOS_ATTN_DK_128 + d]; -+ Qu0[d] = Qh0[d] + bias_u; -+ Qv0[d] = Qh0[d] + bias_v; -+ Qu1[d] = Qh1[d] + bias_u; -+ Qv1[d] = Qh1[d] + bias_v; -+ __syncthreads(); -+ -+ const int d4 = lane * 4; -+ const float4 qu04 = *reinterpret_cast(Qu0 + d4); -+ const float4 qv04 = *reinterpret_cast(Qv0 + d4); -+ const float4 qu14 = *reinterpret_cast(Qu1 + d4); -+ const float4 qv14 = *reinterpret_cast(Qv1 + d4); -+ float produced_max0 = -INFINITY; -+ float produced_max1 = -INFINITY; -+ for (int j = warp; j < kv; j += RELPOS_ATTN_WARPS_128) { -+ const float4 k4 = relpos_load4(Kh + (size_t) j * k_sj + d4); -+ const float4 p04 = relpos_load4(Ph + (size_t) (j + 1) * p_sr + d4); -+ const float4 p14 = relpos_load4(Ph + (size_t) j * p_sr + d4); -+ float score0 = -+ k4.x * qu04.x + p04.x * qv04.x + -+ k4.y * qu04.y + p04.y * qv04.y + -+ k4.z * qu04.z + p04.z * qv04.z + -+ k4.w * qu04.w + p04.w * qv04.w; -+ float score1 = -+ k4.x * qu14.x + p14.x * qv14.x + -+ k4.y * qu14.y + p14.y * qv14.y + -+ k4.z * qu14.z + p14.z * qv14.z + -+ k4.w * qu14.w + p14.w * qv14.w; -+ score0 = relpos_warp_sum(score0); -+ score1 = relpos_warp_sum(score1); -+ if (lane == 0) { -+ const float additive_mask = Mb ? Mb[j] : 0.0f; -+ sc0[j] = score0 * scale + additive_mask; -+ sc1[j] = score1 * scale + additive_mask; -+ produced_max0 = fmaxf(produced_max0, sc0[j]); -+ produced_max1 = fmaxf(produced_max1, sc1[j]); -+ } -+ } -+ if (lane == 0) { -+ red0[warp] = produced_max0; -+ red1[warp] = produced_max1; -+ } -+ __syncthreads(); -+ -+ if (d == 0) { -+ float maximum0 = red0[0]; -+ float maximum1 = red1[0]; -+#pragma unroll -+ for (int w = 1; w < RELPOS_ATTN_WARPS_128; ++w) { -+ maximum0 = fmaxf(maximum0, red0[w]); -+ maximum1 = fmaxf(maximum1, red1[w]); -+ } -+ red0[0] = maximum0; -+ red1[0] = maximum1; -+ } -+ __syncthreads(); -+ const float maximum0 = red0[0]; -+ const float maximum1 = red1[0]; -+ __syncthreads(); -+ -+ float local_sum0 = 0.0f; -+ float local_sum1 = 0.0f; -+ for (int j = d; j < kv; j += RELPOS_ATTN_DK_128) { -+ const float weight0 = __expf(sc0[j] - maximum0); -+ const float weight1 = __expf(sc1[j] - maximum1); -+ sc0[j] = weight0; -+ sc1[j] = weight1; -+ local_sum0 += weight0; -+ local_sum1 += weight1; -+ } -+ local_sum0 = relpos_warp_sum(local_sum0); -+ local_sum1 = relpos_warp_sum(local_sum1); -+ if (lane == 0) { -+ red0[warp] = local_sum0; -+ red1[warp] = local_sum1; -+ } -+ __syncthreads(); -+ if (d == 0) { -+ float sum0 = red0[0]; -+ float sum1 = red1[0]; -+#pragma unroll -+ for (int w = 1; w < RELPOS_ATTN_WARPS_128; ++w) { -+ sum0 += red0[w]; -+ sum1 += red1[w]; -+ } -+ red0[0] = sum0; -+ red1[0] = sum1; -+ } -+ __syncthreads(); -+ -+ float context0 = 0.0f; -+ float context1 = 0.0f; -+ for (int j = 0; j < kv; ++j) { -+ const float value = (float) Vh[(size_t) j * v_sj + d]; -+ context0 += sc0[j] * value; -+ context1 += sc1[j] * value; -+ } -+ const size_t out = (size_t) b * o_sb + (size_t) h * o_sh + d; -+ ctx[out] = context0 / red0[0]; -+ ctx[out + o_si] = context1 / red1[0]; -+} -+ -+template -+static int relpos_attn_warp_128_max_blocks_per_sm(int device, size_t shmem) { -+ // The target shape fixes dynamic shared memory, so occupancy is invariant -+ // for a given compiled kernel and device. Avoid repeating the CUDA runtime -+ // query in every attention layer and graph execution. -+ static std::atomic cached[GGML_CUDA_MAX_DEVICES] = {}; -+ int blocks = cached[device].load(std::memory_order_relaxed); -+ if (blocks == 0) { -+ CUDA_CHECK(cudaOccupancyMaxActiveBlocksPerMultiprocessor( -+ &blocks, fused_relpos_attn_warp_128_kernel, RELPOS_ATTN_DK_128, shmem)); -+ GGML_ASSERT(blocks > 0); -+ cached[device].store(blocks, std::memory_order_relaxed); -+ } -+ return blocks; -+} -+ -+template -+static __global__ void fused_relpos_attn_kernel( -+ const float * __restrict__ Q, const T * __restrict__ K, -+ const T * __restrict__ V, const T * __restrict__ Ppos, -+ const float * __restrict__ bu, const float * __restrict__ bv, -+ const float * __restrict__ mask, float * __restrict__ ctx, -+ int q, int kv, int n_head, float scale, -+ // element strides: x_sq = between queries/keys, x_sh = between heads, -+ // x_sb = between batch items -+ long q_sq, long q_sh, long q_sb, -+ long k_sj, long k_sh, long k_sb, -+ long v_sj, long v_sh, long v_sb, -+ long p_sr, long p_sh, -+ long o_si, long o_sh, long o_sb, long m_sb) { -+ extern __shared__ float sh[]; -+ const int dk = blockDim.x; -+ float * Qu = sh; // [dk] -+ float * Qv = sh + dk; // [dk] -+ float * sc = sh + 2 * dk; // [kv] -+ float * red = sh + 2 * dk + kv; // [dk] reduction scratch -+ -+ const int h = blockIdx.x; // head -+ const int i = blockIdx.y; // query -+ const int b = blockIdx.z; // batch -+ const int d = threadIdx.x; // head dim 0..dk-1 -+ -+ const float * Qhi = Q + (size_t) b * q_sb + (size_t) h * q_sh + (size_t) i * q_sq; -+ const T * Kh = K + (size_t) b * k_sb + (size_t) h * k_sh; -+ const T * Vh = V + (size_t) b * v_sb + (size_t) h * v_sh; -+ const T * Ph = Ppos + (size_t) h * p_sh; -+ const float * Mb = mask ? mask + (size_t) b * m_sb : nullptr; -+ -+ Qu[d] = Qhi[d] + bu[h * dk + d]; -+ Qv[d] = Qhi[d] + bv[h * dk + d]; -+ __syncthreads(); -+ -+ // scores. Two layouts: -+ // * dk == 128 (the FastConformer case): WARP-COOPERATIVE — each warp owns -+ // a key j and the 32 lanes split the 128 dims 4-a-piece with one -+ // vectorized row load + shuffle reduction. The original -+ // thread-per-key loop left dk-kv threads idle (kv ~= 50 < 128) and -+ // issued dk scalar loads per row, which made the kernel -+ // load-issue-bound (F16 operands alone changed nothing). -+ // * otherwise: legacy thread-per-key scalar loop. -+ if (dk == 128) { -+ const int warp = d >> 5, lane = d & 31, nw = dk >> 5; -+ for (int j = warp; j < kv; j += nw) { -+ const T * Kj = Kh + (size_t) j * k_sj + lane * 4; -+ const int row = (q - 1) + j - i; // rel-shift index -+ const T * Pr = Ph + (size_t) row * p_sr + lane * 4; -+ float k4[4], p4[4]; -+ if (sizeof(T) == 2) { -+ const int2 kr = *(const int2 *) Kj; -+ const int2 pr = *(const int2 *) Pr; -+ const half2 * kh = (const half2 *) &kr; -+ const half2 * ph = (const half2 *) ≺ -+ k4[0] = __low2float(kh[0]); k4[1] = __high2float(kh[0]); -+ k4[2] = __low2float(kh[1]); k4[3] = __high2float(kh[1]); -+ p4[0] = __low2float(ph[0]); p4[1] = __high2float(ph[0]); -+ p4[2] = __low2float(ph[1]); p4[3] = __high2float(ph[1]); -+ } else { -+ const float4 kr = *(const float4 *) Kj; -+ const float4 pr = *(const float4 *) Pr; -+ k4[0] = ((const float *) &kr)[0]; k4[1] = ((const float *) &kr)[1]; -+ k4[2] = ((const float *) &kr)[2]; k4[3] = ((const float *) &kr)[3]; -+ p4[0] = ((const float *) &pr)[0]; p4[1] = ((const float *) &pr)[1]; -+ p4[2] = ((const float *) &pr)[2]; p4[3] = ((const float *) &pr)[3]; -+ } -+ float s = 0.0f; -+#pragma unroll -+ for (int e = 0; e < 4; e++) { -+ s += k4[e] * Qu[lane * 4 + e] + p4[e] * Qv[lane * 4 + e]; -+ } -+#pragma unroll -+ for (int off = 16; off > 0; off >>= 1) { -+ s += __shfl_xor_sync(0xffffffff, s, off); -+ } -+ if (lane == 0) { -+ sc[j] = s * scale + (Mb ? Mb[j] : 0.0f); -+ } -+ } -+ } else { -+ for (int j = d; j < kv; j += dk) { -+ const T * Kj = Kh + (size_t) j * k_sj; -+ const int row = (q - 1) + j - i; // rel-shift index -+ const T * Pr = Ph + (size_t) row * p_sr; -+ float ac = 0.0f, bd = 0.0f; -+ for (int dd = 0; dd < dk; dd++) { -+ ac += (float) Kj[dd] * Qu[dd]; -+ bd += (float) Pr[dd] * Qv[dd]; -+ } -+ sc[j] = (ac + bd) * scale + (Mb ? Mb[j] : 0.0f); -+ } -+ } -+ __syncthreads(); -+ -+ // block max over sc[0..kv) -+ float lm = -INFINITY; -+ for (int j = d; j < kv; j += dk) lm = fmaxf(lm, sc[j]); -+ red[d] = lm; -+ __syncthreads(); -+ for (int s = dk / 2; s > 0; s >>= 1) { -+ if (d < s) red[d] = fmaxf(red[d], red[d + s]); -+ __syncthreads(); -+ } -+ const float m = red[0]; -+ __syncthreads(); -+ -+ // exp + block sum -+ float ls = 0.0f; -+ for (int j = d; j < kv; j += dk) { -+ const float e = __expf(sc[j] - m); -+ sc[j] = e; -+ ls += e; -+ } -+ red[d] = ls; -+ __syncthreads(); -+ for (int s = dk / 2; s > 0; s >>= 1) { -+ if (d < s) red[d] += red[d + s]; -+ __syncthreads(); -+ } -+ const float inv = 1.0f / red[0]; -+ __syncthreads(); -+ -+ // ctx[d] = inv * sum_j softmax(sc[j]) * V[j,d] (thread d owns output dim d) -+ float c = 0.0f; -+ for (int j = 0; j < kv; j++) c += sc[j] * (float) Vh[(size_t) j * v_sj + d]; -+ ctx[(size_t) b * o_sb + (size_t) i * o_si + (size_t) h * o_sh + d] = c * inv; -+} -+ -+void ggml_cuda_op_fused_relpos_attn(ggml_backend_cuda_context & ctx, ggml_tensor * dst) { -+ const ggml_tensor * q = dst->src[0]; -+ const ggml_tensor * k = dst->src[1]; -+ const ggml_tensor * v = dst->src[2]; -+ const ggml_tensor * p = dst->src[3]; -+ const ggml_tensor * bias_u = dst->src[4]; -+ const ggml_tensor * bias_v = dst->src[5]; -+ const ggml_tensor * mask = dst->src[6]; // may be null -+ -+ GGML_ASSERT(q->type == GGML_TYPE_F32); -+ // K/V/P may be F32 or (all together) F16 — see the kernel comment. -+ GGML_ASSERT(k->type == v->type && k->type == p->type); -+ GGML_ASSERT(k->type == GGML_TYPE_F32 || k->type == GGML_TYPE_F16); -+ GGML_ASSERT(bias_u->type == GGML_TYPE_F32 && bias_v->type == GGML_TYPE_F32); -+ GGML_ASSERT(dst->type == GGML_TYPE_F32); -+ -+ const int d_k = q->ne[0]; -+ const int q_len = q->ne[1]; -+ const int n_head = q->ne[2]; -+ const int batch = q->ne[3]; -+ const int kv_len = k->ne[1]; -+ -+ GGML_ASSERT((d_k & (d_k - 1)) == 0 && "fused_relpos_attn: d_k must be a power of two"); -+ GGML_ASSERT(k->ne[0] == d_k && v->ne[0] == d_k && p->ne[0] == d_k); -+ GGML_ASSERT(k->ne[1] == kv_len && v->ne[1] == kv_len); -+ GGML_ASSERT(k->ne[2] == n_head && v->ne[2] == n_head && p->ne[2] == n_head); -+ GGML_ASSERT(k->ne[3] == batch && v->ne[3] == batch); -+ GGML_ASSERT(bias_u->ne[0] == d_k && bias_v->ne[0] == d_k); -+ GGML_ASSERT(bias_u->ne[1] == n_head && bias_v->ne[1] == n_head); -+ if (mask != nullptr) { -+ GGML_ASSERT(mask->type == GGML_TYPE_F32); -+ GGML_ASSERT(ggml_is_contiguous(mask)); -+ GGML_ASSERT(mask->ne[0] == kv_len); -+ // Shared across the batch (ne[1]==1) or one key-mask column per -+ // stream (ne[1]==batch, the cache-aware layout). -+ GGML_ASSERT(mask->ne[1] == 1 || mask->ne[1] == batch); -+ GGML_ASSERT(mask->ne[2] == 1 && mask->ne[3] == 1); -+ } -+ const long m_sb = -+ (mask != nullptr && mask->ne[1] == batch && batch > 1) ? (long) (mask->nb[1] / sizeof(float)) : 0; -+ -+ float scale; -+ memcpy(&scale, dst->op_params, sizeof(scale)); -+ -+ // Element strides from tensor byte strides. d_k rows must be contiguous -+ // (constructor invariant), everything else is free-form. -+ const size_t qe = ggml_type_size(q->type); -+ const size_t ke = ggml_type_size(k->type); -+ const long q_sq = (long)(q->nb[1] / qe), q_sh = (long)(q->nb[2] / qe), -+ q_sb = (long)(q->nb[3] / qe); -+ const long k_sj = (long)(k->nb[1] / ke), k_sh = (long)(k->nb[2] / ke), -+ k_sb = (long)(k->nb[3] / ke); -+ const long v_sj = (long)(v->nb[1] / ke), v_sh = (long)(v->nb[2] / ke), -+ v_sb = (long)(v->nb[3] / ke); -+ const long p_sr = (long)(p->nb[1] / ke), p_sh = (long)(p->nb[2] / ke); -+ const size_t oe = ggml_type_size(dst->type); -+ const long o_si = (long)(dst->nb[1] / oe), o_sh = (long)(dst->nb[2] / oe), -+ o_sb = (long)(dst->nb[3] / oe); -+ -+ const int device = ggml_cuda_get_device(); -+ const auto & device_info = ggml_cuda_info().devices[device]; -+ const size_t shmem = ((size_t) 3 * d_k + kv_len) * sizeof(float); -+ const size_t max_shmem = device_info.smpb; -+ GGML_ASSERT(shmem <= max_shmem && "fused_relpos_attn: kv window too large for shared memory"); -+ -+ cudaStream_t stream = ctx.stream(); -+ -+ // The cache-aware Nemotron streaming geometry is Q=2, KV=72, H=8, -+ // d_k=128. On SM100 the one-query warp kernel is fastest while its grid -+ // fits in one resident wave. If that grid exceeds its measured occupancy, -+ // fuse both query rows: halving the block count and reusing K/V then wins. -+ // cudaOccupancyMaxActiveBlocksPerMultiprocessor uses the compiled kernel's -+ // actual register count, avoiding a hard-coded batch-size threshold. -+ const bool use_sm100_q2 = -+ device_info.cc == RELPOS_ATTN_CC_SM100 && device_info.warp_size == 32 && -+ d_k == RELPOS_ATTN_DK_128 && q_len == 2 && kv_len == 72; -+ if (use_sm100_q2) { -+ const size_t warp_shmem = -+ ((size_t) 2 * RELPOS_ATTN_DK_128 + kv_len + RELPOS_ATTN_WARPS_128) * sizeof(float); -+ const size_t q2_shmem = -+ ((size_t) 4 * RELPOS_ATTN_DK_128 + (size_t) 2 * kv_len + -+ (size_t) 2 * RELPOS_ATTN_WARPS_128) * sizeof(float); -+ GGML_ASSERT(q2_shmem <= max_shmem); -+ -+ const int max_single_blocks_per_sm = k->type == GGML_TYPE_F16 -+ ? relpos_attn_warp_128_max_blocks_per_sm(device, warp_shmem) -+ : relpos_attn_warp_128_max_blocks_per_sm(device, warp_shmem); -+ const int64_t single_query_blocks = (int64_t) n_head * q_len * batch; -+ const int64_t single_wave_blocks = (int64_t) device_info.nsm * max_single_blocks_per_sm; -+ const bool fuse_queries = single_query_blocks > single_wave_blocks; -+ const dim3 tuned_grid(n_head, fuse_queries ? 1 : q_len, batch); -+ -+ if (k->type == GGML_TYPE_F16) { -+ if (fuse_queries) { -+ fused_relpos_attn_q2_warp_128_kernel<<< -+ tuned_grid, RELPOS_ATTN_DK_128, q2_shmem, stream>>>( -+ (const float *) q->data, (const half *) k->data, (const half *) v->data, -+ (const half *) p->data, (const float *) bias_u->data, (const float *) bias_v->data, -+ mask ? (const float *) mask->data : nullptr, (float *) dst->data, -+ kv_len, scale, q_sq, q_sh, q_sb, k_sj, k_sh, k_sb, v_sj, v_sh, v_sb, -+ p_sr, p_sh, o_si, o_sh, o_sb, m_sb); -+ } else { -+ fused_relpos_attn_warp_128_kernel<<< -+ tuned_grid, RELPOS_ATTN_DK_128, warp_shmem, stream>>>( -+ (const float *) q->data, (const half *) k->data, (const half *) v->data, -+ (const half *) p->data, (const float *) bias_u->data, (const float *) bias_v->data, -+ mask ? (const float *) mask->data : nullptr, (float *) dst->data, -+ kv_len, scale, q_sq, q_sh, q_sb, k_sj, k_sh, k_sb, v_sj, v_sh, v_sb, -+ p_sr, p_sh, o_si, o_sh, o_sb, m_sb); -+ } -+ } else { -+ if (fuse_queries) { -+ fused_relpos_attn_q2_warp_128_kernel<<< -+ tuned_grid, RELPOS_ATTN_DK_128, q2_shmem, stream>>>( -+ (const float *) q->data, (const float *) k->data, (const float *) v->data, -+ (const float *) p->data, (const float *) bias_u->data, (const float *) bias_v->data, -+ mask ? (const float *) mask->data : nullptr, (float *) dst->data, -+ kv_len, scale, q_sq, q_sh, q_sb, k_sj, k_sh, k_sb, v_sj, v_sh, v_sb, -+ p_sr, p_sh, o_si, o_sh, o_sb, m_sb); -+ } else { -+ fused_relpos_attn_warp_128_kernel<<< -+ tuned_grid, RELPOS_ATTN_DK_128, warp_shmem, stream>>>( -+ (const float *) q->data, (const float *) k->data, (const float *) v->data, -+ (const float *) p->data, (const float *) bias_u->data, (const float *) bias_v->data, -+ mask ? (const float *) mask->data : nullptr, (float *) dst->data, -+ kv_len, scale, q_sq, q_sh, q_sb, k_sj, k_sh, k_sb, v_sj, v_sh, v_sb, -+ p_sr, p_sh, o_si, o_sh, o_sb, m_sb); -+ } -+ } -+ return; -+ } -+ -+ const dim3 grid(n_head, q_len, batch); -+ if (k->type == GGML_TYPE_F16) { -+ fused_relpos_attn_kernel<<>>( -+ (const float *) q->data, (const half *) k->data, (const half *) v->data, -+ (const half *) p->data, (const float *) bias_u->data, (const float *) bias_v->data, -+ mask ? (const float *) mask->data : nullptr, (float *) dst->data, -+ q_len, kv_len, n_head, scale, -+ q_sq, q_sh, q_sb, k_sj, k_sh, k_sb, v_sj, v_sh, v_sb, p_sr, p_sh, o_si, o_sh, o_sb, -+ m_sb); -+ } else { -+ fused_relpos_attn_kernel<<>>( -+ (const float *) q->data, (const float *) k->data, (const float *) v->data, -+ (const float *) p->data, (const float *) bias_u->data, (const float *) bias_v->data, -+ mask ? (const float *) mask->data : nullptr, (float *) dst->data, -+ q_len, kv_len, n_head, scale, -+ q_sq, q_sh, q_sb, k_sj, k_sh, k_sb, v_sj, v_sh, v_sb, p_sr, p_sh, o_si, o_sh, o_sb, -+ m_sb); -+ } -+} -diff --git a/src/ggml-cuda/fused-relpos-attn.cuh b/src/ggml-cuda/fused-relpos-attn.cuh -new file mode 100644 -index 00000000..b647d6b1 ---- /dev/null -+++ b/src/ggml-cuda/fused-relpos-attn.cuh -@@ -0,0 +1,5 @@ -+// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. -+// SPDX-License-Identifier: Apache-2.0 -+#include "common.cuh" -+ -+void ggml_cuda_op_fused_relpos_attn(ggml_backend_cuda_context & ctx, ggml_tensor * dst); -diff --git a/src/ggml.c b/src/ggml.c -index 476c3079..88b39537 100644 ---- a/src/ggml.c -+++ b/src/ggml.c -@@ -1078,9 +1078,11 @@ static const char * GGML_OP_NAME[GGML_OP_COUNT] = { - "OPT_STEP_SGD", - - "GLU", -+ -+ "FUSED_RELPOS_ATTN", - }; - --static_assert(GGML_OP_COUNT == 96, "GGML_OP_COUNT != 96"); -+static_assert(GGML_OP_COUNT == 97, "GGML_OP_COUNT != 97"); - - static const char * GGML_OP_SYMBOL[GGML_OP_COUNT] = { - "none", -@@ -1188,9 +1190,11 @@ static const char * GGML_OP_SYMBOL[GGML_OP_COUNT] = { - "sgd(x)", - - "glu(x)", -+ -+ "fused_relpos_attn(q,k,v,p)", - }; - --static_assert(GGML_OP_COUNT == 96, "GGML_OP_COUNT != 96"); -+static_assert(GGML_OP_COUNT == 97, "GGML_OP_COUNT != 97"); - - static_assert(GGML_OP_POOL_COUNT == 2, "GGML_OP_POOL_COUNT != 2"); - -@@ -5370,6 +5374,78 @@ struct ggml_tensor * ggml_flash_attn_ext( - return result; - } - -+// ggml_fused_relpos_attn -+ -+struct ggml_tensor * ggml_fused_relpos_attn( -+ struct ggml_context * ctx, -+ struct ggml_tensor * q, -+ struct ggml_tensor * k, -+ struct ggml_tensor * v, -+ struct ggml_tensor * p, -+ struct ggml_tensor * bias_u, -+ struct ggml_tensor * bias_v, -+ struct ggml_tensor * mask, -+ float scale, -+ bool merge_heads) { -+ // Q/K/V/P may be arbitrary-strided views; the CUDA op derives addressing -+ // from their nb[]. Only each d_k row must be contiguous (vectorized row -+ // loads). -+ GGML_ASSERT(q->nb[0] == ggml_type_size(q->type)); -+ GGML_ASSERT(k->nb[0] == ggml_type_size(k->type)); -+ GGML_ASSERT(v->nb[0] == ggml_type_size(v->type)); -+ GGML_ASSERT(p->nb[0] == ggml_type_size(p->type)); -+ GGML_ASSERT(ggml_is_contiguous(bias_u)); -+ GGML_ASSERT(ggml_is_contiguous(bias_v)); -+ -+ const int64_t d_k = q->ne[0]; -+ const int64_t kv_len = k->ne[1]; -+ const int64_t q_len = q->ne[1]; -+ const int64_t n_head = q->ne[2]; -+ -+ GGML_ASSERT(k->ne[0] == d_k && v->ne[0] == d_k && p->ne[0] == d_k); -+ GGML_ASSERT(bias_u->ne[0] == d_k && bias_v->ne[0] == d_k); -+ // The kernel reads rel-pos rows (q_len-1)+j-i for j in [0,kv_len), i in -+ // [0,q_len) — i.e. rows [0, kv_len+q_len-1) — and takes its head stride -+ // from p->nb, so a longer table (e.g. precomputed for the full chunk -+ // length and reused by shorter tail chunks) is safe. -+ GGML_ASSERT(p->ne[1] >= kv_len + q_len - 1); // rel-pos length -+ GGML_ASSERT(v->ne[1] == kv_len); -+ if (mask) { -+ GGML_ASSERT(ggml_is_contiguous(mask)); -+ GGML_ASSERT(mask->ne[0] == kv_len); -+ // One shared key mask, or one column per batch item (the cache-aware -+ // streaming layout, where each stream's history has its own validity). -+ GGML_ASSERT(mask->ne[1] == 1 || mask->ne[1] == q->ne[3]); -+ GGML_ASSERT(mask->ne[2] == 1 && mask->ne[3] == 1); -+ } -+ -+ // Output mirrors q logically: [d_k, q_len, n_head, batch]. With -+ // merge_heads the memory layout interleaves heads inside each query -+ // column (nb[2] = d_k, nb[1] = d_k*n_head) so that permute(0,2,1,3) of -+ // the result is a contiguous (d_k*n_head, q_len, batch) matrix. -+ struct ggml_tensor * result = ggml_new_tensor(ctx, GGML_TYPE_F32, 4, q->ne); -+ if (merge_heads) { -+ const size_t ts = ggml_type_size(result->type); -+ result->nb[0] = ts; -+ result->nb[2] = (size_t) d_k * ts; -+ result->nb[1] = (size_t) d_k * n_head * ts; -+ result->nb[3] = (size_t) d_k * n_head * q_len * ts; -+ } -+ -+ ggml_set_op_params(result, &scale, sizeof(scale)); -+ -+ result->op = GGML_OP_FUSED_RELPOS_ATTN; -+ result->src[0] = q; -+ result->src[1] = k; -+ result->src[2] = v; -+ result->src[3] = p; -+ result->src[4] = bias_u; -+ result->src[5] = bias_v; -+ result->src[6] = mask; -+ -+ return result; -+} -+ - void ggml_flash_attn_ext_set_prec( - struct ggml_tensor * a, - enum ggml_prec prec) { diff --git a/ggml-patches/0002-nvfp4-residual-activations.patch b/ggml-patches/0002-nvfp4-residual-activations.patch deleted file mode 100644 index 5f17141..0000000 --- a/ggml-patches/0002-nvfp4-residual-activations.patch +++ /dev/null @@ -1,400 +0,0 @@ -diff --git a/src/ggml-cuda/mmq.cu b/src/ggml-cuda/mmq.cu -index e1add5e0..9c98e3c3 100644 ---- a/src/ggml-cuda/mmq.cu -+++ b/src/ggml-cuda/mmq.cu -@@ -127,7 +127,9 @@ void ggml_cuda_mul_mat_q( - if (!ids) { - const size_t nbytes_src1_q8_1 = ne13*ne12 * ne11*ne10_padded * sizeof(block_q8_1)/QK8_1 + - get_mmq_x_max_host(cc)*sizeof(block_q8_1_mmq); -- ggml_cuda_pool_alloc src1_q8_1(ctx.pool(), nbytes_src1_q8_1); -+ const bool use_nvfp4_residual = use_native_fp4 && src0->type == GGML_TYPE_NVFP4; -+ ggml_cuda_pool_alloc src1_q8_1(ctx.pool(), nbytes_src1_q8_1 * (use_nvfp4_residual ? 2 : 1)); -+ char * src1_residual = use_nvfp4_residual ? src1_q8_1.get() + nbytes_src1_q8_1 : nullptr; - - { - const int64_t s11 = src1->nb[1] / ts_src1; -@@ -135,7 +137,7 @@ void ggml_cuda_mul_mat_q( - const int64_t s13 = src1->nb[3] / ts_src1; - if (use_native_fp4) { - static_assert(sizeof(block_fp4_mmq) == 4 * sizeof(block_q8_1)); -- quantize_mmq_fp4_cuda(src1_d, nullptr, src1_q8_1.get(), src0->type, ne10, s11, s12, s13, ne10_padded, -+ quantize_mmq_fp4_cuda(src1_d, nullptr, src1_q8_1.get(), src1_residual, src0->type, ne10, s11, s12, s13, ne10_padded, - ne11, ne12, ne13, stream); - - } else { -@@ -152,7 +154,7 @@ void ggml_cuda_mul_mat_q( - const int64_t s13 = ne12*s12; - - const mmq_args args = { -- src0_d, src0->type, (const int *) src1_q8_1.ptr, nullptr, nullptr, dst_d, -+ src0_d, src0->type, (const int *) src1_q8_1.ptr, (const int *) src1_residual, nullptr, nullptr, dst_d, - ne00, ne01, ne1, s01, ne11, s1, - ne02, ne12, s02, s12, s2, - ne03, ne13, s03, s13, s3, -@@ -185,7 +187,9 @@ void ggml_cuda_mul_mat_q( - - const size_t nbytes_src1_q8_1 = ne12*n_expert_used*ne10_padded * sizeof(block_q8_1)/QK8_1 + - get_mmq_x_max_host(cc)*sizeof(block_q8_1_mmq); -- ggml_cuda_pool_alloc src1_q8_1(ctx.pool(), nbytes_src1_q8_1); -+ const bool use_nvfp4_residual = use_native_fp4 && src0->type == GGML_TYPE_NVFP4; -+ ggml_cuda_pool_alloc src1_q8_1(ctx.pool(), nbytes_src1_q8_1 * (use_nvfp4_residual ? 2 : 1)); -+ char * src1_residual = use_nvfp4_residual ? src1_q8_1.get() + nbytes_src1_q8_1 : nullptr; - - const int64_t ne11_flat = ne12*n_expert_used; - const int64_t ne12_flat = 1; -@@ -197,7 +201,7 @@ void ggml_cuda_mul_mat_q( - const int64_t s13 = src1->nb[3] / ts_src1; - - if (use_native_fp4) { -- quantize_mmq_fp4_cuda(src1_d, ids_src1.get(), src1_q8_1.get(), src0->type, ne10, s11, s12, s13, -+ quantize_mmq_fp4_cuda(src1_d, ids_src1.get(), src1_q8_1.get(), src1_residual, src0->type, ne10, s11, s12, s13, - ne10_padded, ne11_flat, ne12_flat, ne13_flat, stream); - } else { - quantize_mmq_q8_1_cuda(src1_d, ids_src1.get(), src1_q8_1.get(), src0->type, ne10, s11, s12, s13, -@@ -213,7 +217,7 @@ void ggml_cuda_mul_mat_q( - - // Note that ne02 is used instead of ne12 because the number of y channels determines the z dimension of the CUDA grid. - const mmq_args args = { -- src0_d, src0->type, (const int *) src1_q8_1.get(), ids_dst.get(), expert_bounds.get(), dst_d, -+ src0_d, src0->type, (const int *) src1_q8_1.get(), (const int *) src1_residual, ids_dst.get(), expert_bounds.get(), dst_d, - ne00, ne01, ne_get_rows, s01, ne_get_rows, s1, - ne02, ne02, s02, s12, s2, - ne03, ne13, s03, s13, s3, -@@ -253,7 +257,7 @@ void ggml_cuda_op_mul_mat_q( - || GGML_CUDA_CC_IS_CDNA(cc)) - && src1_ncols == ne11; - const mmq_args args = { -- src0_dd_i, src0->type, (const int *) src1_ddq_i, nullptr, nullptr, dst_dd_i, -+ src0_dd_i, src0->type, (const int *) src1_ddq_i, nullptr, nullptr, nullptr, dst_dd_i, - ne00, row_diff, src1_ncols, stride01, ne11, nrows_dst, - 1, 1, 0, 0, 0, - 1, 1, 0, 0, 0, -diff --git a/src/ggml-cuda/mmq.cuh b/src/ggml-cuda/mmq.cuh -index edf546d8..1881b7f3 100644 ---- a/src/ggml-cuda/mmq.cuh -+++ b/src/ggml-cuda/mmq.cuh -@@ -3446,6 +3446,7 @@ struct mmq_type_traits { - template - static __device__ __forceinline__ void mul_mat_q_process_tile( - const char * __restrict__ x, const int offset_x, const int * __restrict__ y, -+ const int * __restrict__ y_residual, - const int * __restrict__ ids_dst, float * __restrict__ dst, float * __restrict__ tmp_fixup, - const int stride_row_x, const int ncols_y, const int stride_col_dst, - const int tile_x_max_i, const int tile_y_max_j, const int kb0_start, const int kb0_stop) { -@@ -3500,6 +3501,22 @@ static __device__ __forceinline__ void mul_mat_q_process_tile( - - __syncthreads(); - -+#if defined(BLACKWELL_MMA_AVAILABLE) -+ if constexpr (type == GGML_TYPE_NVFP4) { -+ if (y_residual) { -+ const int * by0 = y_residual + ncols_y * (kb0 * qk / ne_block) * sz; -+#pragma unroll -+ for (int l0 = 0; l0 < mmq_x * MMQ_TILE_Y_K; l0 += nwarps * warp_size) { -+ const int l = l0 + threadIdx.y * warp_size + threadIdx.x; -+ tile_y[l] = by0[l]; -+ } -+ __syncthreads(); -+ vec_dot(tile_x, tile_y, sum, 0); -+ __syncthreads(); -+ } -+ } -+#endif -+ - { - const int * by0 = y + ncols_y * ((kb0 * qk / ne_block) * sz + sz); - #pragma unroll -@@ -3515,6 +3532,22 @@ static __device__ __forceinline__ void mul_mat_q_process_tile( - vec_dot(tile_x, tile_y, sum, MMQ_TILE_NE_K); - - __syncthreads(); -+ -+#if defined(BLACKWELL_MMA_AVAILABLE) -+ if constexpr (type == GGML_TYPE_NVFP4) { -+ if (y_residual) { -+ const int * by0 = y_residual + ncols_y * ((kb0 * qk / ne_block) * sz + sz); -+#pragma unroll -+ for (int l0 = 0; l0 < mmq_x * MMQ_TILE_Y_K; l0 += nwarps * warp_size) { -+ const int l = l0 + threadIdx.y * warp_size + threadIdx.x; -+ tile_y[l] = by0[l]; -+ } -+ __syncthreads(); -+ vec_dot(tile_x, tile_y, sum, MMQ_TILE_NE_K); -+ __syncthreads(); -+ } -+ } -+#endif - } - - if (fixup) { -@@ -3540,7 +3573,8 @@ template - #endif // __CUDA_ARCH__ >= GGML_CUDA_CC_VOLTA - #endif // defined(GGML_USE_HIP) - static __global__ void mul_mat_q( -- const char * __restrict__ x, const int * __restrict__ y, const int32_t * __restrict__ ids_dst, -+ const char * __restrict__ x, const int * __restrict__ y, const int * __restrict__ y_residual, -+ const int32_t * __restrict__ ids_dst, - const int32_t * __restrict__ expert_bounds, float * __restrict__ dst, float * __restrict__ tmp_fixup, - const uint3 blocks_per_ne00, const int nrows_x, const int ncols_dst, const int stride_row_x, const int ncols_y, const int stride_col_dst, - const uint3 channel_ratio, const uint3 nchannels_y, const int stride_channel_x, const int stride_channel_y, const int stride_channel_dst, -@@ -3629,7 +3663,8 @@ static __global__ void mul_mat_q( - - constexpr bool fixup = false; - mul_mat_q_process_tile -- (x, offset_x, y + offset_y, ids_dst_shared, dst + offset_dst, tmp_fixup, stride_row_x, ncols_y, stride_col_dst, -+ (x, offset_x, y + offset_y, y_residual ? y_residual + offset_y : nullptr, -+ ids_dst_shared, dst + offset_dst, tmp_fixup, stride_row_x, ncols_y, stride_col_dst, - tile_x_max_i, tile_y_max_j, 0, blocks_per_ne00.z); - return; - } -@@ -3709,7 +3744,8 @@ static __global__ void mul_mat_q( - - constexpr bool fixup = false; // All but (potentially) the last iterations write their data to dst rather than the fixup buffer. - mul_mat_q_process_tile -- (x, offset_x, y + offset_y, ids_dst_shared, dst + offset_dst, tmp_fixup, stride_row_x, ncols_y, stride_col_dst, -+ (x, offset_x, y + offset_y, y_residual ? y_residual + offset_y : nullptr, -+ ids_dst_shared, dst + offset_dst, tmp_fixup, stride_row_x, ncols_y, stride_col_dst, - tile_x_max_i, tile_y_max_j, kb0_start, kb0_stop); - - kbc += blocks_per_ne00.z; -@@ -3778,7 +3814,8 @@ static __global__ void mul_mat_q( - - constexpr bool fixup = true; // Last index writes its data to fixup buffer to avoid data races with other blocks. - mul_mat_q_process_tile -- (x, offset_x, y + offset_y, ids_dst_shared, dst + offset_dst, tmp_fixup, stride_row_x, ncols_y, stride_col_dst, -+ (x, offset_x, y + offset_y, y_residual ? y_residual + offset_y : nullptr, -+ ids_dst_shared, dst + offset_dst, tmp_fixup, stride_row_x, ncols_y, stride_col_dst, - tile_x_max_i, tile_y_max_j, kb0_start, kb0_stop); - } - -@@ -3922,7 +3959,8 @@ static __global__ void mul_mat_q_stream_k_fixup( - } - - struct mmq_args { -- const char * x; ggml_type type_x; const int * y; const int32_t * ids_dst; const int32_t * expert_bounds; float * dst; -+ const char * x; ggml_type type_x; const int * y; const int * y_residual; -+ const int32_t * ids_dst; const int32_t * expert_bounds; float * dst; - int64_t ncols_x; int64_t nrows_x; int64_t ncols_dst; int64_t stride_row_x; int64_t ncols_y; int64_t nrows_dst; - int64_t nchannels_x; int64_t nchannels_y; int64_t stride_channel_x; int64_t stride_channel_y; int64_t stride_channel_dst; - int64_t nsamples_x; int64_t nsamples_y; int64_t stride_sample_x; int64_t stride_sample_y; int64_t stride_sample_dst; -@@ -3976,7 +4014,7 @@ static void launch_mul_mat_q(ggml_backend_cuda_context & ctx, const mmq_args & a - if (args.nrows_x % mmq_y == 0) { - constexpr bool need_check = false; - mul_mat_q<<>> -- (args.x, args.y, args.ids_dst, args.expert_bounds, args.dst, nullptr, -+ (args.x, args.y, args.y_residual, args.ids_dst, args.expert_bounds, args.dst, nullptr, - blocks_per_ne00_fd, args.nrows_x, args.ncols_dst, args.stride_row_x, args.ncols_y, args.nrows_dst, - channel_ratio_fd, nchannels_y_fd, args.stride_channel_x, args.stride_channel_y, args.stride_channel_dst, - sample_ratio_fd, nsamples_y_fd, args.stride_sample_x, args.stride_sample_y, args.stride_sample_dst, -@@ -3984,7 +4022,7 @@ static void launch_mul_mat_q(ggml_backend_cuda_context & ctx, const mmq_args & a - } else { - constexpr bool need_check = true; - mul_mat_q<<>> -- (args.x, args.y, args.ids_dst, args.expert_bounds, args.dst, nullptr, -+ (args.x, args.y, args.y_residual, args.ids_dst, args.expert_bounds, args.dst, nullptr, - blocks_per_ne00_fd, args.nrows_x, args.ncols_dst, args.stride_row_x, args.ncols_y, args.nrows_dst, - channel_ratio_fd, nchannels_y_fd, args.stride_channel_x, args.stride_channel_y, args.stride_channel_dst, - sample_ratio_fd, nsamples_y_fd, args.stride_sample_x, args.stride_sample_y, args.stride_sample_dst, -@@ -4016,7 +4054,7 @@ static void launch_mul_mat_q(ggml_backend_cuda_context & ctx, const mmq_args & a - if (args.nrows_x % mmq_y == 0) { - constexpr bool need_check = false; - mul_mat_q<<>> -- (args.x, args.y, args.ids_dst, args.expert_bounds, args.dst, tmp_fixup.ptr, -+ (args.x, args.y, args.y_residual, args.ids_dst, args.expert_bounds, args.dst, tmp_fixup.ptr, - blocks_per_ne00_fd, args.nrows_x, args.ncols_dst, args.stride_row_x, args.ncols_y, args.nrows_dst, - channel_ratio_fd, nchannels_y_fd, args.stride_channel_x, args.stride_channel_y, args.stride_channel_dst, - sample_ratio_fd, nsamples_y_fd, args.stride_sample_x, args.stride_sample_y, args.stride_sample_dst, -@@ -4034,7 +4072,7 @@ static void launch_mul_mat_q(ggml_backend_cuda_context & ctx, const mmq_args & a - } else { - constexpr bool need_check = true; - mul_mat_q<<>> -- (args.x, args.y, args.ids_dst, args.expert_bounds, args.dst, tmp_fixup.ptr, -+ (args.x, args.y, args.y_residual, args.ids_dst, args.expert_bounds, args.dst, tmp_fixup.ptr, - blocks_per_ne00_fd, args.nrows_x, args.ncols_dst, args.stride_row_x, args.ncols_y, args.nrows_dst, - channel_ratio_fd, nchannels_y_fd, args.stride_channel_x, args.stride_channel_y, args.stride_channel_dst, - sample_ratio_fd, nsamples_y_fd, args.stride_sample_x, args.stride_sample_y, args.stride_sample_dst, -diff --git a/src/ggml-cuda/quantize.cu b/src/ggml-cuda/quantize.cu -index 52f66471..2a878c73 100644 ---- a/src/ggml-cuda/quantize.cu -+++ b/src/ggml-cuda/quantize.cu -@@ -72,7 +72,8 @@ __device__ __forceinline__ uint8_t compute_e8m0_scale(float amax) { - - - static __global__ void quantize_mmq_nvfp4( -- const float * __restrict__ x, const int32_t * __restrict__ ids, void * __restrict__ vy, -+ const float * __restrict__ x, const int32_t * __restrict__ ids, -+ void * __restrict__ vy, void * __restrict__ vy_residual, - const int64_t ne00, const int64_t s01, const int64_t s02, const int64_t s03, - const int64_t ne0, const int64_t ne1, const int64_t ne2) { - #if defined(BLACKWELL_MMA_AVAILABLE) -@@ -95,6 +96,7 @@ static __global__ void quantize_mmq_nvfp4( - const int64_t ib = blockIdx.z * ((int64_t) blocks_per_col * ne1) + k_block * ne1 + blockIdx.x; - block_fp4_mmq * y = (block_fp4_mmq *) vy; - block_fp4_mmq * yb = y + ib; -+ block_fp4_mmq * yrb = (block_fp4_mmq *) vy_residual + ib; - - const int sub = (i0_base % QK_K) / QK_NVFP4_SUB; - -@@ -113,57 +115,62 @@ static __global__ void quantize_mmq_nvfp4( - } - } - -- static constexpr int test_offsets[5] = { 0, -1, 1, -2, 2}; -- const int first_fp8_code = (int) ggml_cuda_fp32_to_ue4m3(amax_raw / 6.0f); -- -- float best_err = FLT_MAX; -- uint8_t fp8_code = 0; -- float subblock_scale = 0.0f; -- --#pragma unroll // Check +/- 2 to find best code to reduce NVFP4 activation loss. Negligible overhead on Blackwell. -- for (int i = 0; i < 5; i++) { -- const int test_code = first_fp8_code + test_offsets[i]; -- if (test_code < 0 || test_code > 0x7e) { -- continue; -- } -- const uint8_t code = (uint8_t) test_code; -- const float test_scale = ggml_cuda_ue4m3_to_fp32(code); -- const float test_inv_scale = test_scale > 0.0f ? 0.5f / test_scale : 0.0f; -- float cur_err = 0.0f; --#pragma unroll -- for (int k = 0; k < QK_NVFP4_SUB; ++k) { -- const float v = vals_raw[k]; -- const uint8_t q = ggml_cuda_float_to_fp4_e2m1(v, test_inv_scale); -- const float err_diff = fabsf(v) - fabsf(kvalues_mxfp4[q & 0x7]) * test_scale; -- cur_err = fmaf(err_diff, err_diff, cur_err); -- } -- -- if (cur_err < best_err) { -- best_err = cur_err; -- fp8_code = test_code; -- subblock_scale = test_scale; -- } -- } -- -+ const uint8_t fp8_code = ggml_cuda_fp32_to_ue4m3(amax_raw / 6.0f); -+ const float subblock_scale = ggml_cuda_ue4m3_to_fp32(fp8_code); - const float inv_scale = subblock_scale > 0.0f ? 0.5f / subblock_scale : 0.0f; - uint32_t q0 = 0; - uint32_t q1 = 0; --#pragma unroll // this is faster than the previous __nv_fp4x4_e2m1 -+ float residual[QK_NVFP4_SUB]; -+ float residual_amax = 0.0f; -+#pragma unroll - for (int k = 0; k < QK_NVFP4_SUB / 4; ++k) { -- q0 |= (uint32_t) ggml_cuda_float_to_fp4_e2m1(vals_raw[k + 0], inv_scale) << (8 * k); -- q0 |= (uint32_t) ggml_cuda_float_to_fp4_e2m1(vals_raw[k + 8], inv_scale) << (8 * k + 4); -- q1 |= (uint32_t) ggml_cuda_float_to_fp4_e2m1(vals_raw[k + 4], inv_scale) << (8 * k); -- q1 |= (uint32_t) ggml_cuda_float_to_fp4_e2m1(vals_raw[k + 12], inv_scale) << (8 * k + 4); -+ const int i0 = k + 0; -+ const int i1 = k + 8; -+ const int i2 = k + 4; -+ const int i3 = k + 12; -+ const uint8_t p0 = ggml_cuda_float_to_fp4_e2m1(vals_raw[i0], inv_scale); -+ const uint8_t p1 = ggml_cuda_float_to_fp4_e2m1(vals_raw[i1], inv_scale); -+ const uint8_t p2 = ggml_cuda_float_to_fp4_e2m1(vals_raw[i2], inv_scale); -+ const uint8_t p3 = ggml_cuda_float_to_fp4_e2m1(vals_raw[i3], inv_scale); -+ q0 |= (uint32_t) p0 << (8 * k); -+ q0 |= (uint32_t) p1 << (8 * k + 4); -+ q1 |= (uint32_t) p2 << (8 * k); -+ q1 |= (uint32_t) p3 << (8 * k + 4); -+ residual[i0] = vals_raw[i0] - (float) kvalues_mxfp4[p0] * subblock_scale; -+ residual[i1] = vals_raw[i1] - (float) kvalues_mxfp4[p1] * subblock_scale; -+ residual[i2] = vals_raw[i2] - (float) kvalues_mxfp4[p2] * subblock_scale; -+ residual[i3] = vals_raw[i3] - (float) kvalues_mxfp4[p3] * subblock_scale; -+ residual_amax = fmaxf(residual_amax, fabsf(residual[i0])); -+ residual_amax = fmaxf(residual_amax, fabsf(residual[i1])); -+ residual_amax = fmaxf(residual_amax, fabsf(residual[i2])); -+ residual_amax = fmaxf(residual_amax, fabsf(residual[i3])); - } - - uint32_t * yqs = reinterpret_cast(yb->qs); - yqs[2 * sub + 0] = q0; - yqs[2 * sub + 1] = q1; - reinterpret_cast(yb->d4)[sub] = fp8_code; -+ -+ const uint8_t residual_code = ggml_cuda_fp32_to_ue4m3(residual_amax / 6.0f); -+ const float residual_scale = ggml_cuda_ue4m3_to_fp32(residual_code); -+ const float residual_inv_scale = residual_scale > 0.0f ? 0.5f / residual_scale : 0.0f; -+ uint32_t rq0 = 0; -+ uint32_t rq1 = 0; -+#pragma unroll -+ for (int k = 0; k < QK_NVFP4_SUB / 4; ++k) { -+ rq0 |= (uint32_t) ggml_cuda_float_to_fp4_e2m1(residual[k + 0], residual_inv_scale) << (8 * k); -+ rq0 |= (uint32_t) ggml_cuda_float_to_fp4_e2m1(residual[k + 8], residual_inv_scale) << (8 * k + 4); -+ rq1 |= (uint32_t) ggml_cuda_float_to_fp4_e2m1(residual[k + 4], residual_inv_scale) << (8 * k); -+ rq1 |= (uint32_t) ggml_cuda_float_to_fp4_e2m1(residual[k + 12], residual_inv_scale) << (8 * k + 4); -+ } -+ -+ uint32_t * yrqs = reinterpret_cast(yrb->qs); -+ yrqs[2 * sub + 0] = rq0; -+ yrqs[2 * sub + 1] = rq1; -+ reinterpret_cast(yrb->d4)[sub] = residual_code; - #else - NO_DEVICE_CODE; // This is for Blackwell NVFP4 activations only. - #endif // defined(BLACKWELL_MMA_AVAILABLE) -- - } - - // quantize values in the format mxfp4 is stored which is interleaved nibbles -@@ -413,7 +420,7 @@ void quantize_mmq_q8_1_cuda( - } - - void quantize_mmq_fp4_cuda( -- const float * x, const int32_t * ids, void * vy, const ggml_type type_src0, -+ const float * x, const int32_t * ids, void * vy, void * vy_residual, const ggml_type type_src0, - const int64_t ne00, const int64_t s01, const int64_t s02, const int64_t s03, - const int64_t ne0, const int64_t ne1, const int64_t ne2, const int64_t ne3, cudaStream_t stream) { - GGML_ASSERT(type_src0 == GGML_TYPE_MXFP4 || type_src0 == GGML_TYPE_NVFP4); -@@ -421,13 +428,16 @@ void quantize_mmq_fp4_cuda( - - if (type_src0 == GGML_TYPE_NVFP4) { - GGML_ASSERT(ne00 % QK_NVFP4 == 0); -+ GGML_ASSERT(vy_residual != nullptr); - constexpr int nvfp4_block_size = 128; -- const int64_t block_num_y = (ne0 + QK_NVFP4_SUB * nvfp4_block_size - 1) / (QK_NVFP4_SUB * nvfp4_block_size); -+ const int64_t block_num_y = -+ (ne0 + QK_NVFP4_SUB * nvfp4_block_size - 1) / (QK_NVFP4_SUB * nvfp4_block_size); - const dim3 block_size(nvfp4_block_size, 1, 1); - const dim3 num_blocks(ne1, block_num_y, ne2 * ne3); - quantize_mmq_nvfp4<<>>( -- x, ids, vy, ne00, s01, s02, s03, ne0, ne1, ne2); -+ x, ids, vy, vy_residual, ne00, s01, s02, s03, ne0, ne1, ne2); - } else { -+ GGML_ASSERT(vy_residual == nullptr); - GGML_ASSERT(ne0 % (2 * QK_MXFP4) == 0); - - constexpr int nwarps = 8; -diff --git a/src/ggml-cuda/quantize.cuh b/src/ggml-cuda/quantize.cuh -index 768a3ae6..84c708ba 100644 ---- a/src/ggml-cuda/quantize.cuh -+++ b/src/ggml-cuda/quantize.cuh -@@ -29,6 +29,7 @@ void quantize_mmq_q8_1_cuda( - void quantize_mmq_fp4_cuda(const float * x, - const int32_t * ids, - void * vy, -+ void * vy_residual, - ggml_type type_src0, - int64_t ne00, - int64_t s01, -diff --git a/tests/test-backend-ops.cpp b/tests/test-backend-ops.cpp -index f54ab41c..e325787f 100644 ---- a/tests/test-backend-ops.cpp -+++ b/tests/test-backend-ops.cpp -@@ -3977,7 +3977,7 @@ struct test_mul_mat : public test_case { - - double max_nmse_err(ggml_backend_t backend) override { - // for blackwell we quantize activations to mxfp4 instead of q8_1 so we add higher tolerance -- if ((type_a == GGML_TYPE_MXFP4 || type_a == GGML_TYPE_NVFP4) && backend_has_feature(backend, "BLACKWELL_NATIVE_FP4")) { -+ if (type_a == GGML_TYPE_MXFP4 && backend_has_feature(backend, "BLACKWELL_NATIVE_FP4")) { - return 2e-2; - } - return max_nmse_err(); -@@ -4166,7 +4166,7 @@ struct test_mul_mat_id : public test_case { - - double max_nmse_err(ggml_backend_t backend) override { - // for blackwell we quantize activations to mxfp4 instead of q8_1 so we add higher tolerance -- if ((type_a == GGML_TYPE_MXFP4 || type_a == GGML_TYPE_NVFP4) && backend_has_feature(backend, "BLACKWELL_NATIVE_FP4")) { -+ if (type_a == GGML_TYPE_MXFP4 && backend_has_feature(backend, "BLACKWELL_NATIVE_FP4")) { - return 2e-2; - } - return max_nmse_err(); diff --git a/ggml-patches/0005-skinny-q8-gemm.patch b/ggml-patches/0005-skinny-q8-gemm.patch deleted file mode 100644 index 2b7de6b..0000000 --- a/ggml-patches/0005-skinny-q8-gemm.patch +++ /dev/null @@ -1,646 +0,0 @@ -diff --git a/src/ggml-cuda/skinny-q8.cu b/src/ggml-cuda/skinny-q8.cu -new file mode 100644 -index 00000000..60147ff1 ---- /dev/null -+++ b/src/ggml-cuda/skinny-q8.cu -@@ -0,0 +1,616 @@ -+// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. -+// SPDX-License-Identifier: Apache-2.0 -+// Skinny-N Q8_0 GEMM for streaming ASR encoders (nemo-speech). -+// -+// This kernel is specialized for 9 <= N <= 64, where the activation tile fits -+// on chip and weight reads dominate memory traffic. -+// -+// It uses int8 tensor cores and a cp.async pipeline: -+// * mma.sync.m16n8k32.s8 — one MMA spans exactly one q8 block of k, so the -+// per-block scales (weight d x activation d) apply directly to the s32 -+// MMA result, accumulated in fp32. -+// * Block tile = 32 rows x 64 cols: 8 warps as 2 row-halves x 4 col-quarters -+// (each warp: 16x16 = 2 MMAs per k-block). All N tiles are launched in -+// grid.z so a large outer batch fills the GPU instead of issuing a serial -+// stream of sub-SM grids. -+// * A 2-buffer pipeline stages W (32x128B), A (64x128B), and their scales per -+// 128-k step. The next buffer is issued before the current buffer computes, -+// keeping one cp.async group in flight without paying for a third buffer. -+// * Weights use aligned planes qs[M][K] + d[M][K/32]. Models can serialize -+// those planes directly; standard block_q8_0 weights are repacked once on -+// first use and cached. Activations are quantized to a zero-padded -+// 64-column int8 buffer so staging needs no bounds checks. -+// -+// Bit-accuracy: same q8 activation round-trip as mul_mat_q, but a different -+// summation order — results are not bit-identical to mmq. Dispatch must -+// therefore depend only on the per-sequence matrix shape, never on the outer -+// batch size, so singleton and batched execution cannot select different math. -+ -+#include "skinny-q8.cuh" -+#include "mmvq.cuh" // MMVQ_MAX_BATCH_SIZE: the N range mmvq already covers -+ -+#include -+#include -+#include -+#include -+ -+#define SKQ8_WARPS 8 -+#define SKQ8_ROWS 32 // rows per block (2 warp row-halves x 16); M % 32 == 0 -+#define SKQ8_NPAD 64 // max padded cols (4 warp col-quarters x 2 n-tiles x 8) -+#define SKQ8_KSTEP 128 // k bytes staged per pipeline stage; K % KSTEP == 0 -+#define SKQ8_KTB (SKQ8_KSTEP / 32) // q8 blocks per stage -+#define SKQ8_STAGES 2 -+#define SKQ8_NMAX SKQ8_NPAD -+ -+// Shared-memory strides (bytes). KSTEP + 16 keeps the 16B cp.async stores -+// aligned while breaking the power-of-two bank pattern on fragment loads. -+#define SKQ8_SW (SKQ8_KSTEP + 16) -+#define SKQ8_SA (SKQ8_KSTEP + 16) -+// Per-stage layout: W tile | A tile | W scales (half) | A scales (float). -+// The A tile + A scales are sized for NPAD but only ncols (= n-tiles actually -+// used, host-padded to 8) are staged/consumed. -+#define SKQ8_STAGE_W (SKQ8_ROWS * SKQ8_SW) -+#define SKQ8_STAGE_A (SKQ8_NPAD * SKQ8_SA) -+#define SKQ8_STAGE_DW (SKQ8_ROWS * SKQ8_KTB * 2) -+#define SKQ8_STAGE_DA (SKQ8_NPAD * SKQ8_KTB * 4) -+#define SKQ8_STAGE_BYTES (SKQ8_STAGE_W + SKQ8_STAGE_A + SKQ8_STAGE_DW + SKQ8_STAGE_DA) -+ -+// --------------------------------------------------------------------------- -+// Repack block_q8_0[M][K/32] -> qs plane int8[M][K] + d plane half[M][K/32]. -+static __global__ void skq8_repack( -+ const void * __restrict__ src, int8_t * __restrict__ qs, half * __restrict__ d, -+ const int64_t M, const int64_t KB) { -+ const int64_t i = (int64_t) blockIdx.x * blockDim.x + threadIdx.x; -+ if (i >= M * KB) { -+ return; -+ } -+ const block_q8_0 * b = (const block_q8_0 *) src + i; -+ d[i] = b->d; -+ int8_t * out = qs + i * 32; -+#pragma unroll -+ for (int j = 0; j < 32; j++) { -+ out[j] = b->qs[j]; -+ } -+} -+ -+// --------------------------------------------------------------------------- -+// Quantize F32 activations [K, N] (col-contiguous) into a zero-padded -+// SKQ8_NPAD-column buffer: int8[NPAD][K] + d float[NPAD][K/32]. -+static __global__ void skq8_quantize( -+ const float * __restrict__ X, int8_t * __restrict__ Aq, float * __restrict__ Ad, -+ const int K, const int N) { -+ // One WARP per (col, q8-block): lane j owns element j, so the 32-float -+ // read is one coalesced 128B transaction and the int8 store one 32B -+ // transaction (a thread-per-block version did 32 scalar strided loads and -+ // cost 5x the GEMM quantize budget). -+ const int KB = K / 32; -+ const int wid = (blockIdx.x * blockDim.x + threadIdx.x) / 32; -+ const int lane = threadIdx.x % 32; -+ const int ncols = (N + 7) / 8 * 8; // active column tiles only -+ if (wid >= ncols * KB) { -+ return; -+ } -+ const int n = wid / KB, kb = wid % KB; -+ int8_t * q = Aq + (size_t) n * K + kb * 32; -+ if (n >= N) { -+ // Pad columns: only the scale must be zeroed — the GEMM multiplies the -+ // integer MMA result by da, and int math on stale qs bytes is finite, -+ // so da == 0 makes the contribution exactly 0. -+ if (lane == 0) { -+ Ad[(size_t) n * KB + kb] = 0.0f; -+ } -+ return; -+ } -+ const float x = X[(size_t) n * K + kb * 32 + lane]; -+ float amax = fabsf(x); -+#pragma unroll -+ for (int off = 16; off > 0; off >>= 1) { -+ amax = fmaxf(amax, __shfl_xor_sync(0xffffffff, amax, off)); -+ } -+ const float dv = amax / 127.0f; -+ const float id = dv > 0.0f ? 1.0f / dv : 0.0f; -+ q[lane] = (int8_t) roundf(x * id); -+ if (lane == 0) { -+ Ad[(size_t) n * KB + kb] = dv; -+ } -+} -+ -+// --------------------------------------------------------------------------- -+#if defined(__CUDA_ARCH__) && __CUDA_ARCH__ >= 800 -+static __device__ __forceinline__ void skq8_cp16(void * dst, const void * src) { -+ const unsigned s = (unsigned) __cvta_generic_to_shared(dst); -+ asm volatile("cp.async.cg.shared.global [%0], [%1], 16;" ::"r"(s), "l"(src)); -+} -+static __device__ __forceinline__ void skq8_cp8(void * dst, const void * src) { -+ const unsigned s = (unsigned) __cvta_generic_to_shared(dst); -+ asm volatile("cp.async.ca.shared.global [%0], [%1], 8;" ::"r"(s), "l"(src)); -+} -+static __device__ __forceinline__ void skq8_cp4(void * dst, const void * src) { -+ const unsigned s = (unsigned) __cvta_generic_to_shared(dst); -+ asm volatile("cp.async.ca.shared.global [%0], [%1], 4;" ::"r"(s), "l"(src)); -+} -+static __device__ __forceinline__ void skq8_commit() { -+ asm volatile("cp.async.commit_group;"); -+} -+template static __device__ __forceinline__ void skq8_wait() { -+ asm volatile("cp.async.wait_group %0;" ::"n"(n)); -+} -+static __device__ __forceinline__ void skq8_mma( -+ int & d0, int & d1, int & d2, int & d3, int a0, int a1, int a2, int a3, int b0, int b1) { -+ asm volatile( -+ "mma.sync.aligned.m16n8k32.row.col.s32.s8.s8.s32 " -+ "{%0,%1,%2,%3}, {%4,%5,%6,%7}, {%8,%9}, {%0,%1,%2,%3};" -+ : "+r"(d0), "+r"(d1), "+r"(d2), "+r"(d3) -+ : "r"(a0), "r"(a1), "r"(a2), "r"(a3), "r"(b0), "r"(b1)); -+} -+#endif -+ -+// C[M,N] (F32, col stride M) = Wq8 x A. Small-M GEMMs split K over -+// blockIdx.y (gridDim.y k-ranges, scales already folded per range); each -+// split writes a private plane for deterministic reduction afterward. -+static __global__ __launch_bounds__(SKQ8_WARPS * 32, 5) void skq8_gemm( -+ const int8_t * __restrict__ Wq, const half * __restrict__ Wd, const int8_t * __restrict__ Aq, -+ const float * __restrict__ Ad, const float * __restrict__ bias, float * __restrict__ C, -+ const int M, const int N_total, const int K, const int ncols_total /* N padded to 8 */) { -+#if defined(__CUDA_ARCH__) && __CUDA_ARCH__ >= 800 -+ const int KB = K / 32; -+ const int warp = threadIdx.x / 32; -+ const int lane = threadIdx.x % 32; -+ const int gid = lane >> 2; // mma group id (0..7) -+ const int tid4 = lane & 3; // mma thread-in-group (0..3) -+ -+ const int k_len = K / gridDim.y; // multiple of SKQ8_KSTEP (host-enforced) -+ const int k0 = blockIdx.y * k_len; -+ -+ // grid.z carries independent 64-column output tiles. Keeping them in one -+ // launch is essential on high-SM-count GPUs: a typical encoder projection -+ // has only 128 row/K-split CTAs and can underfill the device. The old host -+ // loop repeated that underfilled launch per tile and serialized the -+ // entire outer batch. Pointer rebasing keeps all inner indexing unchanged, -+ // including the per-q8-block FP32 accumulation order. -+ const int cbase = (int) blockIdx.z * SKQ8_NPAD; -+ const int ncols = min(SKQ8_NPAD, ncols_total - cbase); -+ const int N = min(SKQ8_NPAD, N_total - cbase); -+ Aq += (size_t) cbase * K; -+ Ad += (size_t) cbase * KB; -+ C += (size_t) cbase * M; -+ -+ const int row_blk = blockIdx.x * SKQ8_ROWS; -+ const int mhalf = warp / 4; // which 16-row half -+ const int nt0 = (warp % 4) * 2; // first of this warp's two 8-col n-tiles -+ -+ extern __shared__ char smem[]; -+ char * stage[SKQ8_STAGES]; -+#pragma unroll -+ for (int s = 0; s < SKQ8_STAGES; s++) { -+ stage[s] = smem + (size_t) s * SKQ8_STAGE_BYTES; -+ } -+ -+ const int nsteps = k_len / SKQ8_KSTEP; -+ -+ // Stage `ks` of this block's k-range into buffer ks % STAGES. -+ auto issue = [&](int ks) { -+ char * buf = stage[ks % SKQ8_STAGES]; -+ char * s_w = buf; -+ char * s_a = buf + SKQ8_STAGE_W; -+ char * s_dw = s_a + SKQ8_STAGE_A; -+ char * s_da = s_dw + SKQ8_STAGE_DW; -+ const int kbyte = k0 + ks * SKQ8_KSTEP; // k offset in elements (== bytes for s8) -+ const int kbb = kbyte / 32; // k offset in q8 blocks -+ constexpr int vw = SKQ8_KSTEP / 16; // 16B vectors per row per stage -+ // W tile: ROWS x vw x 16B -+ for (int it = threadIdx.x; it < SKQ8_ROWS * vw; it += blockDim.x) { -+ const int r = it / vw, v = it % vw; -+ skq8_cp16( -+ s_w + r * SKQ8_SW + v * 16, Wq + (size_t) (row_blk + r) * K + kbyte + v * 16); -+ } -+ // A tile: ncols x vw x 16B (only the active column tiles) -+ for (int it = threadIdx.x; it < ncols * vw; it += blockDim.x) { -+ const int c = it / vw, v = it % vw; -+ skq8_cp16(s_a + c * SKQ8_SA + v * 16, Aq + (size_t) c * K + kbyte + v * 16); -+ } -+ // scales: dw ROWS x KTB half (KTB*2 bytes/row), da ncols x KTB float -+ // (KTB*4 bytes/col); copy width follows KTB. -+ if (threadIdx.x < SKQ8_ROWS) { -+ if (SKQ8_KTB * 2 == 16) { -+ skq8_cp16( -+ s_dw + threadIdx.x * (SKQ8_KTB * 2), -+ Wd + (size_t) (row_blk + threadIdx.x) * KB + kbb); -+ } else if (SKQ8_KTB * 2 == 8) { -+ skq8_cp8( -+ s_dw + threadIdx.x * (SKQ8_KTB * 2), -+ Wd + (size_t) (row_blk + threadIdx.x) * KB + kbb); -+ } else { -+ skq8_cp4( -+ s_dw + threadIdx.x * (SKQ8_KTB * 2), -+ Wd + (size_t) (row_blk + threadIdx.x) * KB + kbb); -+ } -+ } -+ constexpr int da_vec = SKQ8_KTB * 4 / 16; // 16B vectors per col -+ if constexpr (da_vec > 0) { -+ for (int it = threadIdx.x; it < ncols * da_vec; it += blockDim.x) { -+ const int c = it / da_vec, h = it % da_vec; -+ skq8_cp16( -+ s_da + c * (SKQ8_KTB * 4) + h * 16, Ad + (size_t) c * KB + kbb + h * 4); -+ } -+ } else if (threadIdx.x < ncols) { -+ skq8_cp8( -+ s_da + threadIdx.x * (SKQ8_KTB * 4), -+ Ad + (size_t) threadIdx.x * KB + kbb); -+ } -+ skq8_commit(); -+ }; -+ -+ float facc[2][4] = { { 0.0f } }; -+ -+ issue(0); -+ -+ for (int ks = 0; ks < nsteps; ks++) { -+ // The next stage is issued after the current buffer is ready and before -+ // its MMA loop. This preserves compute/copy overlap with two buffers; -+ // the trailing barrier makes reusing the just-consumed buffer safe. -+ skq8_wait<0>(); -+ __syncthreads(); -+ -+ if (ks + 1 < nsteps) { -+ issue(ks + 1); -+ } -+ -+ const char * buf = stage[ks % SKQ8_STAGES]; -+ const char * s_w = buf; -+ const char * s_a = buf + SKQ8_STAGE_W; -+ const half * s_dw = (const half *) (buf + SKQ8_STAGE_W + SKQ8_STAGE_A); -+ const float * s_da = -+ (const float *) (buf + SKQ8_STAGE_W + SKQ8_STAGE_A + SKQ8_STAGE_DW); -+ -+#pragma unroll -+ for (int kb = 0; kb < SKQ8_KSTEP / 32; kb++) { -+ // W fragment for this warp's 16 rows (rows mhalf*16 + gid, +8). -+ const char * wbase = s_w + (mhalf * 16 + gid) * SKQ8_SW + kb * 32 + tid4 * 4; -+ const int a0 = *(const int *) (wbase); -+ const int a1 = *(const int *) (wbase + 8 * SKQ8_SW); -+ const int a2 = *(const int *) (wbase + 16); -+ const int a3 = *(const int *) (wbase + 8 * SKQ8_SW + 16); -+ const float dw0 = __half2float(s_dw[(mhalf * 16 + gid) * SKQ8_KTB + kb]); -+ const float dw1 = __half2float(s_dw[(mhalf * 16 + gid + 8) * SKQ8_KTB + kb]); -+#pragma unroll -+ for (int t = 0; t < 2; t++) { -+ const int nt = nt0 + t; -+ if (nt * 8 >= ncols) { -+ continue; // dead column tile (N padded to ncols < NPAD) -+ } -+ const char * bbase = s_a + (nt * 8 + gid) * SKQ8_SA + kb * 32 + tid4 * 4; -+ const int b0 = *(const int *) (bbase); -+ const int b1 = *(const int *) (bbase + 16); -+ int d0 = 0, d1 = 0, d2 = 0, d3 = 0; -+ skq8_mma(d0, d1, d2, d3, a0, a1, a2, a3, b0, b1); -+ const int c0 = nt * 8 + tid4 * 2; -+ const float da0 = s_da[c0 * SKQ8_KTB + kb]; -+ const float da1 = s_da[(c0 + 1) * SKQ8_KTB + kb]; -+ facc[t][0] += dw0 * da0 * (float) d0; -+ facc[t][1] += dw0 * da1 * (float) d1; -+ facc[t][2] += dw1 * da0 * (float) d2; -+ facc[t][3] += dw1 * da1 * (float) d3; -+ } -+ } -+ __syncthreads(); -+ } -+ -+ // Write back: C rows mhalf*16 + gid (+8), cols nt*8 + tid4*2 (+1). -+ // With a K-split grid, each split writes a private output plane. A -+ // follow-up kernel reduces those planes in a fixed order, avoiding the -+ // run-to-run numerical drift caused by unordered FP32 atomic additions. -+ const int row0 = row_blk + mhalf * 16 + gid; -+ const float b0 = (bias != nullptr && blockIdx.y == 0) ? bias[row0] : 0.0f; -+ const float b1 = (bias != nullptr && blockIdx.y == 0) ? bias[row0 + 8] : 0.0f; -+#pragma unroll -+ for (int t = 0; t < 2; t++) { -+ facc[t][0] += b0; -+ facc[t][1] += b0; -+ facc[t][2] += b1; -+ facc[t][3] += b1; -+ } -+#pragma unroll -+ for (int t = 0; t < 2; t++) { -+ const int c0 = (nt0 + t) * 8 + tid4 * 2; -+ if (gridDim.y > 1) { -+ // Plane stride covers the full logical output. `N` is only this -+ // grid.z tile's width (<= SKQ8_NPAD), so using it here aliases the -+ // next tile whenever N_total > SKQ8_NPAD. -+ const size_t split_base = (size_t) blockIdx.y * M * N_total; -+ if (c0 < N) { -+ C[split_base + row0 + (size_t) c0 * M] = facc[t][0]; -+ C[split_base + row0 + 8 + (size_t) c0 * M] = facc[t][2]; -+ } -+ if (c0 + 1 < N) { -+ C[split_base + row0 + (size_t) (c0 + 1) * M] = facc[t][1]; -+ C[split_base + row0 + 8 + (size_t) (c0 + 1) * M] = facc[t][3]; -+ } -+ } else { -+ if (c0 < N) { -+ C[row0 + (size_t) c0 * M] = facc[t][0]; -+ C[row0 + 8 + (size_t) c0 * M] = facc[t][2]; -+ } -+ if (c0 + 1 < N) { -+ C[row0 + (size_t) (c0 + 1) * M] = facc[t][1]; -+ C[row0 + 8 + (size_t) (c0 + 1) * M] = facc[t][3]; -+ } -+ } -+ } -+#else -+ GGML_UNUSED_VARS(Wq, Wd, Aq, Ad, bias, C, M, N_total, K, ncols_total); -+ NO_DEVICE_CODE; -+#endif -+} -+ -+static __global__ void skq8_reduce_splitk( -+ const float * __restrict__ partials, float * __restrict__ dst, -+ const int64_t count, const int ksplit) { -+ const int64_t i = (int64_t) blockIdx.x * blockDim.x + threadIdx.x; -+ if (i >= count) { -+ return; -+ } -+ float sum = partials[i]; -+ for (int split = 1; split < ksplit; ++split) { -+ sum += partials[(int64_t) split * count + i]; -+ } -+ dst[i] = sum; -+} -+ -+// Repacked-weight cache. Keyed by the weight tensor's device pointer; entries -+// live for the process lifetime (model weights are loaded once). Repacking -+// happens on first (eager/warmup) use, before any CUDA-graph capture. -+namespace { -+struct skq8_planes { -+ int8_t * qs; -+ half * d; -+}; -+std::unordered_map g_skq8_cache; -+std::mutex g_skq8_mutex; -+} // namespace -+ -+bool ggml_cuda_skinny_q8_supported( -+ const ggml_tensor * src0, const ggml_tensor * src1, const ggml_tensor * dst) { -+ static const bool disabled = []() { -+ const char * e = getenv("GGML_SKINNY_Q8"); -+ return e != nullptr && e[0] == '0'; -+ }(); -+ if (disabled) { -+ return false; -+ } -+ // Resolve views before checking the persistent model-weight marker. A -+ // serialized planar tensor is accepted only through its full allocation, -+ // since a normal ggml byte-offset view cannot describe two separate -+ // tensor-wide planes. -+ const ggml_tensor * root = src0; -+ while (root->view_src != nullptr) { -+ root = root->view_src; -+ } -+ const bool planar_q8 = (root->flags & GGML_TENSOR_FLAG_Q8_PLANAR) != 0; -+ // This specialization is for FastConformer encoder matrices. Keeping the -+ // domain explicit also prevents a decoder's request batch (which commonly -+ // occupies ne[1]) from being mistaken for a skinny time dimension. -+ if (strncmp(root->name, "encoder.", 8) != 0) { -+ return false; -+ } -+ if (!planar_q8 && strstr(root->name, ".linear_qkv.weight") != nullptr) { -+ return false; -+ } -+ // Tensor-wide planes cannot be addressed through a stock block-row view. -+ // Current encoder graphs use the full fused QKV tensor, never a weight -+ // view; reject defensively if that changes. -+ if (planar_q8 && src0 != root) { -+ return false; -+ } -+ -+ const int cc = ggml_cuda_info().devices[ggml_cuda_get_device()].cc; -+ const bool skinny_q8_available = ampere_mma_available(cc); -+ if (!planar_q8 && !skinny_q8_available) { -+ return false; -+ } -+ // The repack cache is keyed by src0->data and assumes the bytes are -+ // immutable from the outside: only accept long-lived weight buffers, never -+ // transient compute-pool tensors (whose addresses get reused). -+ if (src0->buffer == nullptr || -+ ggml_backend_buffer_get_usage(src0->buffer) == GGML_BACKEND_BUFFER_USAGE_COMPUTE) { -+ return false; -+ } -+ -+ // Weight-side (per-tensor, call-invariant) conditions. -+ const bool tensor_ok = src0->type == GGML_TYPE_Q8_0 && ggml_is_contiguous(src0) && -+ src0->ne[2] == 1 && src0->ne[3] == 1 && -+ src0->ne[0] % SKQ8_KSTEP == 0 && src0->ne[1] % SKQ8_ROWS == 0; -+ // Call-side conditions. -+ // A dense batched RHS [K,N,B...] is byte-identical to [K,N*B...]. Treat -+ // all outer columns as one logical N so a weight repacked by a scalar -+ // graph remains usable by a later true-batch graph. -+ const int64_t total_n = ggml_nelements(src1) / src1->ne[0]; -+ const bool call_ok = src1->type == GGML_TYPE_F32 && dst->type == GGML_TYPE_F32 && -+ ggml_is_contiguous(src1) && ggml_is_contiguous(dst) && -+ src1->ne[0] == src0->ne[0] && -+ ggml_nelements(dst) == src0->ne[1] * total_n; -+ -+ if (planar_q8) { -+ // N<=8 is handled by planar MMVQ. Wider calls must stay on this path -+ // because stock MMQ cannot interpret tensor-wide planes; skq8_gemm -+ // tiles arbitrary N in 64-column chunks. -+ GGML_ASSERT(tensor_ok && call_ok && "planar Q8 weight used in unsupported mul_mat shape"); -+ GGML_ASSERT((skinny_q8_available || total_n <= MMVQ_MAX_BATCH_SIZE) && -+ "wide planar Q8 requires an SM80+ CUDA kernel; use block Q8 on older GPUs"); -+ return skinny_q8_available && total_n > MMVQ_MAX_BATCH_SIZE; -+ } -+ -+ const bool repacked = [&]() { -+ std::lock_guard lock(g_skq8_mutex); -+ return g_skq8_cache.find(src0->data) != g_skq8_cache.end(); -+ }(); -+ if (repacked) { -+ // The weight was converted IN PLACE to the plane layout — the -+ // block_q8_0 bytes no longer exist, so every mul_mat on this tensor -+ // must come through here (the >64-col path tiles, small N pads). -+ // Falling back to mmq/mmvq would silently read garbage: abort loudly -+ // instead if a call shape we cannot serve ever appears. -+ GGML_ASSERT(call_ok && "skinny-q8: repacked weight used in an unsupported mul_mat shape"); -+ return true; -+ } -+ // By default, select only on logical width so outer batch size cannot -+ // change the accumulation path. The opt-in mode is an end-to-end ASR -+ // experiment for streaming shapes such as [K,2,B]: it flattens the dense -+ // outer batch and lets the tensor-core kernel tile N beyond 64. It is not -+ // the default because skinny-Q8 has a different accumulation order from -+ // MMVQ; callers must validate transcript/accuracy parity. Use a separate -+ // repack allocation (GGML_SKINNY_Q8_INPLACE=0) with multi-stream schedulers. -+ static const bool outer_batch_dispatch = []() { -+ const char * e = getenv("GGML_SKINNY_Q8_OUTER_BATCH"); -+ return e != nullptr && e[0] != '0'; -+ }(); -+ const int64_t logical_n = src1->ne[1]; -+ const bool logical_eligible = logical_n > MMVQ_MAX_BATCH_SIZE && logical_n <= SKQ8_NMAX; -+ const bool outer_eligible = total_n > MMVQ_MAX_BATCH_SIZE; -+ const bool eligible = -+ tensor_ok && call_ok && (outer_batch_dispatch ? outer_eligible : logical_eligible); -+ return eligible; -+} -+ -+static void skq8_run( -+ ggml_backend_cuda_context & ctx, const ggml_tensor * src0, const ggml_tensor * src1, -+ const ggml_tensor * bias, ggml_tensor * dst) { -+ const int64_t M = src0->ne[1]; -+ const int64_t K = src0->ne[0]; -+ const int64_t N = ggml_nelements(src1) / src1->ne[0]; -+ const int64_t KB = K / 32; -+ -+ cudaStream_t stream = ctx.stream(); -+ -+ // Repacked weight planes (create on first use). Default: IN PLACE. The plane -+ // layout (qs M*K + d M*KB*2) is byte-for-byte the same total size as -+ // block_q8_0 (M*KB*34), so it repacks through a transient pool staging buffer -+ // and copies back over the original allocation: zero extra weight memory (the -+ // cudaMalloc'd duplicate cost ~1.07 GB on parakeet-xxl). After this the -+ // tensor's block_q8_0 layout is GONE; the dispatch in -+ // ggml_cuda_skinny_q8_supported() therefore claims every mul_mat on a -+ // repacked tensor. -+ // -+ // GGML_SKINNY_Q8_INPLACE=0 forces the separate-buffer (cudaMalloc) layout. -+ // The in-place D2D memcpy is a stream-ordering hazard under multi-stream -+ // graph-split scheduling (llama.cpp's NMT decoder): it corrupts the GEMM even -+ // though the repacked bytes are correct. Callers that share the process with -+ // such a scheduler set GGML_SKINNY_Q8_INPLACE=0 (the NMT pipeline does this -+ // when enabled). The streaming-ASR encoder runtime has no such hazard. -+ skq8_planes planes; -+ { -+ std::lock_guard lock(g_skq8_mutex); -+ auto it = g_skq8_cache.find(src0->data); -+ if (it != g_skq8_cache.end()) { -+ planes = it->second; -+ } else { -+ const size_t qs_bytes = (size_t) M * K; -+ const size_t d_bytes = (size_t) M * KB * sizeof(half); -+ GGML_ASSERT(qs_bytes + d_bytes == ggml_nbytes(src0)); -+ if ((src0->flags & GGML_TENSOR_FLAG_Q8_PLANAR) != 0) { -+ planes.qs = (int8_t *) src0->data; -+ planes.d = (half *) ((char *) src0->data + qs_bytes); -+ g_skq8_cache.emplace(src0->data, planes); -+ } else { -+ static const bool inplace = []() { -+ const char * e = getenv("GGML_SKINNY_Q8_INPLACE"); -+ return e == nullptr || e[0] != '0'; -+ }(); -+ const int64_t total = M * KB; -+ const int blocks = (int) ((total + 255) / 256); -+ size_t extra = 0; -+ if (inplace) { -+ ggml_cuda_pool_alloc staging(ctx.pool(), qs_bytes + d_bytes); -+ skq8_repack<<>>( -+ src0->data, staging.get(), (half *) (staging.get() + qs_bytes), M, KB); -+ CUDA_CHECK(cudaMemcpyAsync( -+ src0->data, staging.get(), qs_bytes + d_bytes, cudaMemcpyDeviceToDevice, -+ stream)); -+ planes.qs = (int8_t *) src0->data; -+ planes.d = (half *) ((char *) src0->data + qs_bytes); -+ // staging returns to the pool at scope exit; the copy is -+ // stream-ordered before any later reuse on this stream. -+ } else { -+ CUDA_CHECK(cudaMalloc(&planes.qs, qs_bytes)); -+ CUDA_CHECK(cudaMalloc(&planes.d, d_bytes)); -+ skq8_repack<<>>( -+ src0->data, planes.qs, planes.d, M, KB); -+ extra = qs_bytes + d_bytes; -+ } -+ g_skq8_cache.emplace(src0->data, planes); -+ static const bool memstats = getenv("NEMO_SPEECH_MEMSTATS") != nullptr; -+ if (memstats) { -+ static size_t g_repack_total = 0; -+ g_repack_total += extra; -+ fprintf( -+ stderr, -+ "[memstats] skinny-q8 repack %s +%.1f MB (cache extra total %.1f MB, " -+ "%zu tensors)\n", -+ inplace ? "in-place" : "alloc", extra / 1048576.0, -+ g_repack_total / 1048576.0, g_skq8_cache.size()); -+ } -+ } -+ } -+ } -+ // Quantize activations into the zero-padded buffer (all columns, once — -+ // ntot may exceed NPAD; the GEMM below tiles over 64-col chunks). -+ const int ntot = (int) ((N + SKQ8_NPAD - 1) / SKQ8_NPAD * SKQ8_NPAD); -+ ggml_cuda_pool_alloc aq_alloc(ctx.pool(), (size_t) ntot * K); -+ ggml_cuda_pool_alloc ad_alloc(ctx.pool(), (size_t) ntot * KB); -+ { -+ const int64_t warps = (int64_t) ntot * KB; // one warp per (col, q8 block) -+ const int blocks = (int) ((warps * 32 + 255) / 256); -+ skq8_quantize<<>>( -+ (const float *) src1->data, aq_alloc.get(), ad_alloc.get(), (int) K, (int) N); -+ } -+ -+ // Split K for small-M shapes until there is enough block-level parallelism. -+ // Each split writes a private plane and a deterministic reduction combines -+ // the planes afterward; unordered atomics can amplify into visible -+ // B=1-versus-batched drift across a deep encoder. -+ // 1-step splits are allowed: no pipeline, but block-level K-parallelism. -+ // N > NPAD is mapped into grid.z. Weights are still read once per 64-column -+ // tile, but all tiles are visible to the scheduler in a single launch. -+ { -+ const int row_blocks = (int) (M / SKQ8_ROWS); -+ int ksplit = 1; -+ while (ksplit < 4 && row_blocks * ksplit < 128 && -+ K % ((int64_t) SKQ8_KSTEP * ksplit * 2) == 0) { -+ ksplit *= 2; -+ } -+ const size_t smem = (size_t) SKQ8_STAGES * SKQ8_STAGE_BYTES; -+ static std::once_flag attr_set_flag; -+ std::call_once(attr_set_flag, [&]() { -+ CUDA_CHECK(cudaFuncSetAttribute( -+ skq8_gemm, cudaFuncAttributeMaxDynamicSharedMemorySize, (int) smem)); -+ }); -+ const int ntiles = (ntot + SKQ8_NPAD - 1) / SKQ8_NPAD; -+ const dim3 grid(row_blocks, ksplit, ntiles); -+ const float * bias_data = bias != nullptr ? (const float *) bias->data : nullptr; -+ if (ksplit > 1) { -+ const int64_t count = M * N; -+ ggml_cuda_pool_alloc partials(ctx.pool(), (size_t) ksplit * count); -+ skq8_gemm<<>>( -+ planes.qs, planes.d, aq_alloc.get(), ad_alloc.get(), bias_data, partials.get(), -+ (int) M, (int) N, (int) K, ntot); -+ skq8_reduce_splitk<<<(count + 255) / 256, 256, 0, stream>>>( -+ partials.get(), (float *) dst->data, count, ksplit); -+ } else { -+ skq8_gemm<<>>( -+ planes.qs, planes.d, aq_alloc.get(), ad_alloc.get(), bias_data, -+ (float *) dst->data, (int) M, (int) N, (int) K, ntot); -+ } -+ } -+} -+ -+void ggml_cuda_mul_mat_skinny_q8( -+ ggml_backend_cuda_context & ctx, const ggml_tensor * src0, const ggml_tensor * src1, -+ ggml_tensor * dst) { -+ skq8_run(ctx, src0, src1, nullptr, dst); -+} -+ -+void ggml_cuda_mul_mat_skinny_q8_bias( -+ ggml_backend_cuda_context & ctx, const ggml_tensor * src0, const ggml_tensor * src1, -+ const ggml_tensor * bias, ggml_tensor * dst) { -+ skq8_run(ctx, src0, src1, bias, dst); -+} -diff --git a/src/ggml-cuda/skinny-q8.cuh b/src/ggml-cuda/skinny-q8.cuh -new file mode 100644 -index 00000000..17175bbb ---- /dev/null -+++ b/src/ggml-cuda/skinny-q8.cuh -@@ -0,0 +1,18 @@ -+// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. -+// SPDX-License-Identifier: Apache-2.0 -+#include "common.cuh" -+ -+// Skinny-N (9..64 cols) Q8_0 x F32 GEMM specialized for streaming encoders. -+// See skinny-q8.cu for the design notes. -+bool ggml_cuda_skinny_q8_supported( -+ const ggml_tensor * src0, const ggml_tensor * src1, const ggml_tensor * dst); -+ -+void ggml_cuda_mul_mat_skinny_q8( -+ ggml_backend_cuda_context & ctx, const ggml_tensor * src0, const ggml_tensor * src1, -+ ggml_tensor * dst); -+ -+// Fused variant: adds a row-vector bias ([M], F32) in the GEMM epilogue and -+// writes the result to `dst` (the bias-add node's buffer). -+void ggml_cuda_mul_mat_skinny_q8_bias( -+ ggml_backend_cuda_context & ctx, const ggml_tensor * src0, const ggml_tensor * src1, -+ const ggml_tensor * bias, ggml_tensor * dst); diff --git a/ggml-patches/0006-cuda-dispatch-wiring.patch b/ggml-patches/0006-cuda-dispatch-wiring.patch deleted file mode 100644 index 94110ba..0000000 --- a/ggml-patches/0006-cuda-dispatch-wiring.patch +++ /dev/null @@ -1,671 +0,0 @@ -diff --git a/include/ggml.h b/include/ggml.h -index 0b6945ac..b32d0432 100644 ---- a/include/ggml.h -+++ b/include/ggml.h -@@ -648,6 +648,10 @@ extern "C" { - GGML_TENSOR_FLAG_PARAM = 4, // ...contains trainable parameters - GGML_TENSOR_FLAG_LOSS = 8, // ...defines loss for numerical optimization (multiple loss tensors add up) - GGML_TENSOR_FLAG_COMPUTE = 16, // ...must be computed -+ // Model weight uses Q8_0 bytes serialized as one tensor-wide int8 -+ // plane followed by one FP16 scale plane. CUDA-only storage hint; -+ // logical type, element count, and allocation size remain Q8_0. -+ GGML_TENSOR_FLAG_Q8_PLANAR = 32, - }; - - enum ggml_tri_type { -diff --git a/src/ggml-cuda/common.cuh b/src/ggml-cuda/common.cuh -index 10817505..9bd40faa 100644 ---- a/src/ggml-cuda/common.cuh -+++ b/src/ggml-cuda/common.cuh -@@ -1479,11 +1479,18 @@ struct ggml_cuda_mm_fusion_args_host { - const ggml_tensor * x_bias = nullptr; - const ggml_tensor * gate = nullptr; - const ggml_tensor * gate_bias = nullptr; -+ bool silu = false; - ggml_glu_op glu_op; - }; - struct ggml_cuda_mm_fusion_args_device { - const void * x_bias = nullptr; - const void * gate = nullptr; - const void * gate_bias = nullptr; -+ // A Linear bias has shape [M, 1, 1, 1] and is broadcast across the -+ // destination columns/channels. Keep this explicit because the legacy -+ // fusion path also accepts a full-shape elementwise bias. -+ bool x_bias_broadcast = false; -+ bool gate_bias_broadcast = false; -+ bool silu = false; - ggml_glu_op glu_op; - }; -diff --git a/src/ggml-cuda/ggml-cuda.cu b/src/ggml-cuda/ggml-cuda.cu -index e25be359..6f3e1c42 100644 ---- a/src/ggml-cuda/ggml-cuda.cu -+++ b/src/ggml-cuda/ggml-cuda.cu -@@ -16,6 +16,7 @@ - #include "ggml-cuda/conv2d.cuh" - #include "ggml-cuda/conv2d-dw.cuh" - #include "ggml-cuda/conv2d-transpose.cuh" -+#include "ggml-cuda/skinny-q8.cuh" - #include "ggml-cuda/convert.cuh" - #include "ggml-cuda/count-equal.cuh" - #include "ggml-cuda/cpy.cuh" -@@ -24,6 +25,7 @@ - #include "ggml-cuda/diagmask.cuh" - #include "ggml-cuda/diag.cuh" - #include "ggml-cuda/fattn.cuh" -+#include "ggml-cuda/fused-relpos-attn.cuh" - #include "ggml-cuda/getrows.cuh" - #include "ggml-cuda/im2col.cuh" - #include "ggml-cuda/mmf.cuh" -@@ -2505,16 +2507,20 @@ static bool ggml_cuda_should_fuse_mul_mat_vec_q(const ggml_tensor * tensor) { - ggml_nbytes(src0) != ggml_backend_buffer_get_alloc_size(src0->buffer, src0) && - src0->view_src; - -+ const int64_t total_n = ggml_nelements(src1) / src1->ne[0]; - bool use_mul_mat_vec_q = ggml_is_quantized(src0->type) && !bad_padding_clear && src1->type == GGML_TYPE_F32 && -- dst->type == GGML_TYPE_F32 && src1->ne[1] <= MMVQ_MAX_BATCH_SIZE; -+ dst->type == GGML_TYPE_F32 && total_n <= MMVQ_MAX_BATCH_SIZE; - - // fusion is not universally faster on Pascal - const int cc = ggml_cuda_info().devices[ggml_cuda_get_device()].cc; - if (cc <= GGML_CUDA_CC_PASCAL) { - return false; - } -- //we only support fusion for ncols_dst = 1 -- if (tensor->op == GGML_OP_MUL_MAT && dst->ne[1] != 1) { -+ // MMVQ's narrow epilogue supports the two-frame streaming chunk used by -+ // NeMo-Speech.cpp. Outer request-batch columns are included in -+ // total_n above so [K,2,B] stays on skinny-Q8 when B makes it wider than -+ // MMVQ_MAX_BATCH_SIZE. -+ if (tensor->op == GGML_OP_MUL_MAT && dst->ne[1] > 2) { - return false; - } - -@@ -2534,6 +2540,19 @@ static bool ggml_cuda_should_fuse_mul_mat_vec_q(const ggml_tensor * tensor) { - return use_mul_mat_vec_q; - } - -+static bool ggml_cuda_q8_narrow_epilogue_enabled() { -+ static const bool enabled = [] { -+ const char * value = getenv("GGML_CUDA_Q8_NARROW_EPILOGUE"); -+ // Keep the former variable name as an alias for existing -+ // deployments. -+ if (value == nullptr) { -+ value = getenv("GGML_CUDA_Q8_NARROW_BIAS_FUSION"); -+ } -+ return value == nullptr || value[0] != '0'; -+ }(); -+ return enabled; -+} -+ - static void ggml_cuda_mul_mat(ggml_backend_cuda_context & ctx, const ggml_tensor * src0, const ggml_tensor * src1, ggml_tensor * dst) { - const bool split = ggml_backend_buft_is_cuda_split(src0->buffer->buft); - -@@ -2594,7 +2613,11 @@ static void ggml_cuda_mul_mat(ggml_backend_cuda_context & ctx, const ggml_tensor - bool use_batched_cublas_bf16 = src0->type == GGML_TYPE_BF16 && bf16_mma_hardware_available(cc); - bool use_batched_cublas_f32 = src0->type == GGML_TYPE_F32; - -- if (!split && use_mul_mat_vec_f) { -+ if (!split && !bad_padding_clear && ggml_cuda_skinny_q8_supported(src0, src1, dst)) { -+ // streaming-ASR specialization: Q8_0 weights x skinny (9..64 col) -+ // activations — beats mul_mat_q's LLM-batch tiling at these shapes -+ ggml_cuda_mul_mat_skinny_q8(ctx, src0, src1, dst); -+ } else if (!split && use_mul_mat_vec_f) { - // the custom F16 vector kernel can be used over batched cuBLAS GEMM - // but this is only faster for GPUs without tensor cores or with a thin src0 matrix (particularly KQV in attention) - ggml_cuda_mul_mat_vec_f(ctx, src0, src1, nullptr, dst); -@@ -3071,6 +3094,9 @@ static bool ggml_cuda_compute_forward(ggml_backend_cuda_context & ctx, struct gg - case GGML_OP_FLASH_ATTN_EXT: - ggml_cuda_flash_attn_ext(ctx, dst); - break; -+ case GGML_OP_FUSED_RELPOS_ATTN: -+ ggml_cuda_op_fused_relpos_attn(ctx, dst); -+ break; - case GGML_OP_CROSS_ENTROPY_LOSS: - ggml_cuda_cross_entropy_loss(ctx, dst); - break; -@@ -3662,6 +3688,51 @@ static bool ggml_cuda_can_fuse(const struct ggml_cgraph * cgraph, - return false; - } - -+ // Standard LayerNorm (GGML_OP_NORM) + row-vector gamma (+ row-vector beta). -+ // Restricted to the classic affine pattern: the mul/add operands must be -+ // contiguous ne0-length vectors broadcast over rows — that is what the -+ // fused norm_mul_add_f32 kernel implements (see norm.cu). -+ if ((ops.size() == 2 || ops.size() == 3) && ops.begin()[0] == GGML_OP_NORM && ops.begin()[1] == GGML_OP_MUL) { -+ const ggml_tensor * norm = cgraph->nodes[node_idx]; -+ const ggml_tensor * mul = cgraph->nodes[node_idx+1]; -+ const ggml_tensor * add = nullptr; -+ -+ if (ops.size() == 3) { -+ if (ops.begin()[2] != GGML_OP_ADD) { -+ return false; -+ } -+ add = cgraph->nodes[node_idx+2]; -+ } -+ -+ if (norm->src[0]->type != GGML_TYPE_F32 || norm->type != GGML_TYPE_F32 || -+ mul->type != GGML_TYPE_F32 || (add && add->type != GGML_TYPE_F32)) { -+ return false; -+ } -+ -+ const ggml_tensor * gamma = -+ mul->src[0] == norm ? mul->src[1] : (mul->src[1] == norm ? mul->src[0] : nullptr); -+ if (gamma == nullptr || gamma->type != GGML_TYPE_F32 || -+ ggml_nelements(gamma) != norm->ne[0] || !ggml_is_contiguous(gamma)) { -+ return false; -+ } -+ -+ if (add) { -+ const ggml_tensor * beta = -+ add->src[0] == mul ? add->src[1] : (add->src[1] == mul ? add->src[0] : nullptr); -+ if (beta == nullptr || beta->type != GGML_TYPE_F32 || -+ ggml_nelements(beta) != norm->ne[0] || !ggml_is_contiguous(beta)) { -+ return false; -+ } -+ } -+ -+ // the fused kernel writes dst rows contiguously -+ if (!ggml_is_contiguous(mul) || (add && !ggml_is_contiguous(add))) { -+ return false; -+ } -+ -+ return true; -+ } -+ - if ((ops.size() == 2 || ops.size() == 3) && ops.begin()[0] == GGML_OP_RMS_NORM && ops.begin()[1] == GGML_OP_MUL) { - const ggml_tensor *rms_norm = cgraph->nodes[node_idx]; - const ggml_tensor *mul = cgraph->nodes[node_idx+1]; -@@ -4119,6 +4190,66 @@ static int ggml_cuda_try_fuse(ggml_backend_cuda_context * cuda_ctx, ggml_cgraph - fused_mul_mat_vec = false; - fused_node_count = 0; - -+ // Q8 narrow projection + SiLU. NeMo-Speech.cpp's Conformer FF1 -+ // layers are bias-free, so this is their actual hot graph sequence. -+ if (ggml_cuda_q8_narrow_epilogue_enabled() && -+ ggml_can_fuse(cgraph, i, { GGML_OP_MUL_MAT, GGML_OP_UNARY })) { -+ ggml_tensor * mm_node = cgraph->nodes[i]; -+ ggml_tensor * silu_node = cgraph->nodes[i + 1]; -+ if (ggml_get_unary_op(silu_node) == GGML_UNARY_OP_SILU && -+ silu_node->src[0] == mm_node && ggml_cuda_should_fuse_mul_mat_vec_q(mm_node)) { -+ ggml_cuda_mm_fusion_args_host fusion_data{}; -+ fusion_data.silu = true; -+ ggml_cuda_mul_mat_vec_q( -+ *cuda_ctx, mm_node->src[0], mm_node->src[1], nullptr, silu_node, &fusion_data); -+ return 1; -+ } -+ } -+ -+ // Q8 narrow projection + broadcast Linear bias + SiLU. This is the hot -+ // Conformer FF1 sequence at the two-frame streaming chunk size. Folding -+ // both epilogues avoids two launches and two full output round trips. -+ if (ggml_cuda_q8_narrow_epilogue_enabled() && -+ ggml_can_fuse(cgraph, i, { GGML_OP_MUL_MAT, GGML_OP_ADD, GGML_OP_UNARY })) { -+ ggml_tensor * mm_node = cgraph->nodes[i]; -+ ggml_tensor * bias_node = cgraph->nodes[i + 1]; -+ ggml_tensor * silu_node = cgraph->nodes[i + 2]; -+ const ggml_tensor * bias = bias_node->src[0] == mm_node ? bias_node->src[1] : -+ bias_node->src[1] == mm_node ? bias_node->src[0] : -+ nullptr; -+ if (ggml_get_unary_op(silu_node) == GGML_UNARY_OP_SILU && -+ silu_node->src[0] == bias_node && bias != nullptr && -+ bias->type == GGML_TYPE_F32 && ggml_is_contiguous(bias) && -+ ggml_nelements(bias) == mm_node->ne[0] && -+ ggml_cuda_should_fuse_mul_mat_vec_q(mm_node)) { -+ ggml_cuda_mm_fusion_args_host fusion_data{}; -+ fusion_data.x_bias = bias; -+ fusion_data.silu = true; -+ ggml_cuda_mul_mat_vec_q( -+ *cuda_ctx, mm_node->src[0], mm_node->src[1], nullptr, silu_node, &fusion_data); -+ return 2; -+ } -+ } -+ -+ // skinny-q8 GEMM + row-vector bias: the upstream MUL_MAT+ADD fusion below -+ // requires a same-shape add, so the classic broadcast Linear bias -+ // ([M] over [M,N]) never qualifies — fold it into the skinny GEMM's -+ // epilogue instead. -+ if (ggml_can_fuse(cgraph, i, { GGML_OP_MUL_MAT, GGML_OP_ADD })) { -+ ggml_tensor * mm_node = cgraph->nodes[i]; -+ ggml_tensor * bias_node = cgraph->nodes[i + 1]; -+ const ggml_tensor * bias = bias_node->src[0] == mm_node ? bias_node->src[1] : -+ bias_node->src[1] == mm_node ? bias_node->src[0] : -+ nullptr; -+ if (bias != nullptr && bias->type == GGML_TYPE_F32 && ggml_is_contiguous(bias) && -+ ggml_nelements(bias) == mm_node->ne[0] && ggml_is_contiguous(bias_node) && -+ ggml_cuda_skinny_q8_supported(mm_node->src[0], mm_node->src[1], mm_node)) { -+ ggml_cuda_mul_mat_skinny_q8_bias( -+ *cuda_ctx, mm_node->src[0], mm_node->src[1], bias, bias_node); -+ return 1; -+ } -+ } -+ - // gate + add + glu + up + add - for (ggml_op op : { GGML_OP_MUL_MAT, GGML_OP_MUL_MAT_ID }) { - const ggml_op bias_op = op == GGML_OP_MUL_MAT ? GGML_OP_ADD : GGML_OP_ADD_ID; -@@ -4154,14 +4285,21 @@ static int ggml_cuda_try_fuse(ggml_backend_cuda_context * cuda_ctx, ggml_cgraph - continue; - } - -- if (bias_op == GGML_OP_ADD && !ggml_are_same_shape(bias_node->src[0], bias_node->src[1])) { -+ const bool same_shape_bias = bias_op != GGML_OP_ADD || -+ ggml_are_same_shape(bias_node->src[0], bias_node->src[1]); -+ const bool broadcast_q_bias = ggml_cuda_q8_narrow_epilogue_enabled() && -+ bias_op == GGML_OP_ADD && -+ bias_tensor->type == GGML_TYPE_F32 && -+ ggml_is_contiguous(bias_tensor) && -+ ggml_nelements(bias_tensor) == mm_node->ne[0]; -+ if (!same_shape_bias && !broadcast_q_bias) { - continue; - } - - ggml_cuda_mm_fusion_args_host fusion_data{}; - fusion_data.x_bias = bias_tensor; - -- if (ggml_cuda_should_fuse_mul_mat_vec_f(mm_node)) { -+ if (same_shape_bias && ggml_cuda_should_fuse_mul_mat_vec_f(mm_node)) { - ggml_cuda_mul_mat_vec_f(*cuda_ctx, src0, src1, ids, bias_node, &fusion_data); - fused_mul_mat_vec = true; - fused_node_count = 2; -@@ -4190,6 +4328,16 @@ static int ggml_cuda_try_fuse(ggml_backend_cuda_context * cuda_ctx, ggml_cgraph - return 1; - } - -+ if (ggml_cuda_can_fuse(cgraph, i, { GGML_OP_NORM, GGML_OP_MUL, GGML_OP_ADD }, {})) { -+ ggml_cuda_op_norm_fused(*cuda_ctx, node, cgraph->nodes[i + 1], cgraph->nodes[i + 2]); -+ return 2; -+ } -+ -+ if (ggml_cuda_can_fuse(cgraph, i, { GGML_OP_NORM, GGML_OP_MUL }, {})) { -+ ggml_cuda_op_norm_fused(*cuda_ctx, node, cgraph->nodes[i + 1], nullptr); -+ return 1; -+ } -+ - if (ggml_cuda_can_fuse(cgraph, i, { GGML_OP_SSM_CONV, GGML_OP_ADD, GGML_OP_UNARY }, { GGML_UNARY_OP_SILU })) { - ggml_cuda_op_ssm_conv(*cuda_ctx, node, cgraph->nodes[i + 1], cgraph->nodes[i + 2]); - return 2; -@@ -5402,6 +5550,8 @@ static bool ggml_backend_cuda_device_supports_op(ggml_backend_dev_t dev, const g - #endif // GGML_USE_MUSA - case GGML_OP_FLASH_ATTN_EXT: - return ggml_cuda_flash_attn_ext_supported(dev_ctx->device, op); -+ case GGML_OP_FUSED_RELPOS_ATTN: -+ return (op->src[0]->ne[0] & (op->src[0]->ne[0] - 1)) == 0; // d_k power of two - case GGML_OP_CROSS_ENTROPY_LOSS: - case GGML_OP_CROSS_ENTROPY_LOSS_BACK: - case GGML_OP_OPT_STEP_ADAMW: -diff --git a/src/ggml-cuda/mmvq.cu b/src/ggml-cuda/mmvq.cu -index da48f313..03668cb8 100644 ---- a/src/ggml-cuda/mmvq.cu -+++ b/src/ggml-cuda/mmvq.cu -@@ -391,7 +391,17 @@ static constexpr __host__ __device__ int calc_rows_per_block(int ncols_dst, int - return 1; - } - --template -+template -+static __device__ __forceinline__ float vec_dot_q8_0_q8_1_planar( -+ const void * __restrict__ vx, const block_q8_1 * __restrict__ y, -+ int block_idx, int iqs, size_t scale_plane_offset) { -+ const int * v = (const int *) vx + (size_t) block_idx * (QK8_0 / 4) + iqs; -+ const int * u = (const int *) y->qs + iqs; -+ const half * d = (const half *) ((const char *) vx + scale_plane_offset); -+ return vec_dot_q8_0_q8_1_impl(v, u, d[block_idx], __low2half(y->ds)); -+} -+ -+template - __launch_bounds__(calc_nwarps(type, ncols_dst, get_device_table_id())*ggml_cuda_get_physical_warp_size(), 1) - static __global__ void mul_mat_vec_q( - const void * __restrict__ vx, const void * __restrict__ vy, const int32_t * __restrict__ ids, const ggml_cuda_mm_fusion_args_device fusion, float * __restrict__ dst, -@@ -415,6 +425,10 @@ static __global__ void mul_mat_vec_q( - const int row0 = rows_per_cuda_block*blockIdx.x; - const int blocks_per_row_x = ncols_x / qk; - constexpr int blocks_per_iter = vdr * nwarps*warp_size / qi; -+ // Serialized planar weights are restricted to contiguous 2-D matrices. -+ // stride_col_dst is therefore the source row count, and one int8 byte is -+ // stored per logical weight before the FP16 scale plane. -+ const size_t q8_scale_plane_offset = (size_t) ncols_x * stride_col_dst; - - const uint32_t channel_dst = blockIdx.y; - -@@ -432,6 +446,7 @@ static __global__ void mul_mat_vec_q( - bool use_gate = false; - bool use_bias = false; - bool use_gate_bias = false; -+ bool use_silu = false; - const void * vgate = nullptr; - const float * x_bias = nullptr; - const float * gate_bias = nullptr; -@@ -441,6 +456,7 @@ static __global__ void mul_mat_vec_q( - use_gate = fusion.gate != nullptr; - use_bias = fusion.x_bias != nullptr; - use_gate_bias = fusion.gate_bias != nullptr && use_gate; -+ use_silu = fusion.silu; - vgate = fusion.gate; - x_bias = (const float *) fusion.x_bias; - gate_bias = (const float *) fusion.gate_bias; -@@ -453,24 +469,26 @@ static __global__ void mul_mat_vec_q( - if constexpr (has_fusion) { - const uint32_t channel_bias = ids ? channel_x : channel_dst; - if (use_bias) { -- x_bias = x_bias + sample_dst*stride_sample_dst + channel_bias*stride_channel_dst + row0; -+ x_bias = x_bias + (fusion.x_bias_broadcast ? row0 : -+ sample_dst*stride_sample_dst + channel_bias*stride_channel_dst + row0); - // 1. Hide latency by prefetching bias and gate here - // 2. load only on threads that won't die after partial sum calculation - if (threadIdx.x < rows_per_cuda_block && threadIdx.y == 0 && - (rows_per_cuda_block == 1 || uint32_t(row0 + threadIdx.x) < stride_col_dst)) { - #pragma unroll - for (int j = 0; j < ncols_dst; ++j) { -- x_biases[j] = x_bias[j * stride_col_dst + threadIdx.x]; -+ x_biases[j] = x_bias[(fusion.x_bias_broadcast ? 0 : j * stride_col_dst) + threadIdx.x]; - } - } - } - if (use_gate_bias) { -- gate_bias = gate_bias + sample_dst*stride_sample_dst + channel_bias*stride_channel_dst + row0; -+ gate_bias = gate_bias + (fusion.gate_bias_broadcast ? row0 : -+ sample_dst*stride_sample_dst + channel_bias*stride_channel_dst + row0); - if (threadIdx.x < rows_per_cuda_block && threadIdx.y == 0 && - (rows_per_cuda_block == 1 || uint32_t(row0 + threadIdx.x) < stride_col_dst)) { - #pragma unroll - for (int j = 0; j < ncols_dst; ++j) { -- gate_biases[j] = gate_bias[j * stride_col_dst + threadIdx.x]; -+ gate_biases[j] = gate_bias[(fusion.gate_bias_broadcast ? 0 : j * stride_col_dst) + threadIdx.x]; - } - } - } -@@ -493,12 +511,26 @@ static __global__ void mul_mat_vec_q( - for (int j = 0; j < ncols_dst; ++j) { - #pragma unroll - for (int i = 0; i < rows_per_cuda_block; ++i) { -- tmp[j][i] += vec_dot_q_cuda( -- vx, &y[j*stride_col_y + kby], kbx_offset + i*stride_row_x + kbx, kqs); -+ if constexpr (q8_planar) { -+ static_assert(type == GGML_TYPE_Q8_0, "planar MMVQ is Q8_0-only"); -+ tmp[j][i] += vec_dot_q8_0_q8_1_planar( -+ vx, &y[j*stride_col_y + kby], -+ kbx_offset + i*stride_row_x + kbx, kqs, q8_scale_plane_offset); -+ } else { -+ tmp[j][i] += vec_dot_q_cuda( -+ vx, &y[j*stride_col_y + kby], kbx_offset + i*stride_row_x + kbx, kqs); -+ } - if constexpr (has_fusion) { - if (use_gate) { -- tmp_gate[j][i] += vec_dot_q_cuda( -- vgate, &y[j*stride_col_y + kby], kbx_offset + i*stride_row_x + kbx, kqs); -+ if constexpr (q8_planar) { -+ tmp_gate[j][i] += vec_dot_q8_0_q8_1_planar( -+ vgate, &y[j*stride_col_y + kby], -+ kbx_offset + i*stride_row_x + kbx, kqs, q8_scale_plane_offset); -+ } else { -+ tmp_gate[j][i] += vec_dot_q_cuda( -+ vgate, &y[j*stride_col_y + kby], -+ kbx_offset + i*stride_row_x + kbx, kqs); -+ } - } - } - } -@@ -583,13 +615,16 @@ static __global__ void mul_mat_vec_q( - break; - } - } -+ if (use_silu) { -+ result = ggml_cuda_op_silu_single(result); -+ } - } - dst[j*stride_col_dst + threadIdx.x] = result; - } - } - - if constexpr (!has_fusion) { -- GGML_UNUSED_VARS(use_gate, use_bias, use_gate_bias, active_glu, gate_bias, x_bias, tmp_gate); -+ GGML_UNUSED_VARS(use_gate, use_bias, use_gate_bias, use_silu, active_glu, gate_bias, x_bias, tmp_gate); - } - } - -@@ -668,7 +703,7 @@ static std::pair calc_launch_params( - return {block_nums, block_dims}; - } - --template -+template - static void mul_mat_vec_q_switch_fusion( - const void * vx, const void * vy, const int32_t * ids, const ggml_cuda_mm_fusion_args_device fusion, float * dst, - const uint32_t ncols_x, const uint3 nchannels_y, const uint32_t stride_row_x, const uint32_t stride_col_y, -@@ -678,10 +713,11 @@ static void mul_mat_vec_q_switch_fusion( - const dim3 & block_nums, const dim3 & block_dims, const int nbytes_shared, - const uint32_t ids_stride, cudaStream_t stream) { - -- const bool has_fusion = fusion.gate != nullptr || fusion.x_bias != nullptr || fusion.gate_bias != nullptr; -- if constexpr (c_ncols_dst == 1) { -+ const bool has_fusion = fusion.gate != nullptr || fusion.x_bias != nullptr || -+ fusion.gate_bias != nullptr || fusion.silu; -+ if constexpr (c_ncols_dst <= 2) { - if (has_fusion) { -- mul_mat_vec_q<<>> -+ mul_mat_vec_q<<>> - (vx, vy, ids, fusion, dst, ncols_x, nchannels_y, stride_row_x, stride_col_y, stride_col_dst, - channel_ratio, stride_channel_x, stride_channel_y, stride_channel_dst, - sample_ratio, stride_sample_x, stride_sample_y, stride_sample_dst, ids_stride); -@@ -689,9 +725,9 @@ static void mul_mat_vec_q_switch_fusion( - } - } - -- GGML_ASSERT(!has_fusion && "fusion only supported for ncols_dst=1"); -+ GGML_ASSERT(!has_fusion && "fusion only supported for ncols_dst<=2"); - -- mul_mat_vec_q<<>> -+ mul_mat_vec_q<<>> - (vx, vy, ids, fusion, dst, ncols_x, nchannels_y, stride_row_x, stride_col_y, stride_col_dst, - channel_ratio, stride_channel_x, stride_channel_y, stride_channel_dst, - sample_ratio, stride_sample_x, stride_sample_y, stride_sample_dst, ids_stride); -@@ -718,7 +754,7 @@ static void mul_mat_vec_q_moe_launch( - ncols_dst, ids_stride); - } - --template -+template - static void mul_mat_vec_q_switch_ncols_dst( - const void * vx, const void * vy, const int32_t * ids, const ggml_cuda_mm_fusion_args_device fusion, float * dst, - const int ncols_x, const int nrows_x, const int ncols_dst, -@@ -730,6 +766,7 @@ static void mul_mat_vec_q_switch_ncols_dst( - - GGML_ASSERT(ncols_x % ggml_blck_size(type) == 0); - GGML_ASSERT(ncols_dst <= MMVQ_MAX_BATCH_SIZE); -+ static_assert(!q8_planar || type == GGML_TYPE_Q8_0, "planar MMVQ is Q8_0-only"); - - const uint3 nchannels_y_fd = ids ? init_fastdiv_values(nchannels_y) : make_uint3(0, 0, 0); - const uint3 channel_ratio_fd = ids ? make_uint3(0, 0, 0) : init_fastdiv_values(nchannels_dst / nchannels_x); -@@ -786,6 +823,7 @@ static void mul_mat_vec_q_switch_ncols_dst( - }; - - if (has_ids && ncols_dst > 1) { -+ GGML_ASSERT(!q8_planar && "planar Q8 does not support MUL_MAT_ID"); - // Multi-token MUL_MAT_ID path - dedicated MoE kernel - mul_mat_vec_q_moe_launch( - vx, vy, ids, dst, ncols_x, nchannels_y_fd, nrows_x, -@@ -804,7 +842,7 @@ static void mul_mat_vec_q_switch_ncols_dst( - if (use_small_k) { - std::pair dims = calc_launch_params(c_ncols_dst, nrows_x, nchannels_dst, - nsamples_dst, warp_size, table_id, true); -- mul_mat_vec_q_switch_fusion( -+ mul_mat_vec_q_switch_fusion( - vx, vy, ids, fusion, dst, ncols_x, nchannels_y_fd, stride_row_x, stride_col_y, stride_col_dst, - channel_ratio_fd, stride_channel_x, stride_channel_y, stride_channel_dst, sample_ratio_fd, - stride_sample_x, stride_sample_y, stride_sample_dst, dims.first, dims.second, 0, ids_stride, -@@ -812,7 +850,7 @@ static void mul_mat_vec_q_switch_ncols_dst( - } else { - std::pair dims = calc_launch_params(c_ncols_dst, nrows_x, nchannels_dst, - nsamples_dst, warp_size, table_id); -- mul_mat_vec_q_switch_fusion( -+ mul_mat_vec_q_switch_fusion( - vx, vy, ids, fusion, dst, ncols_x, nchannels_y_fd, stride_row_x, stride_col_y, stride_col_dst, - channel_ratio_fd, stride_channel_x, stride_channel_y, stride_channel_dst, sample_ratio_fd, - stride_sample_x, stride_sample_y, stride_sample_dst, dims.first, dims.second, 0, ids_stride, -@@ -822,7 +860,7 @@ static void mul_mat_vec_q_switch_ncols_dst( - case 2: { - constexpr int c_ncols_dst = 2; - std::pair dims = calc_launch_params(c_ncols_dst, nrows_x, nchannels_dst, nsamples_dst, warp_size, table_id); -- mul_mat_vec_q_switch_fusion(vx, vy, ids, fusion, dst, ncols_x, nchannels_y_fd, stride_row_x, stride_col_y, stride_col_dst, -+ mul_mat_vec_q_switch_fusion(vx, vy, ids, fusion, dst, ncols_x, nchannels_y_fd, stride_row_x, stride_col_y, stride_col_dst, - channel_ratio_fd, stride_channel_x, stride_channel_y, stride_channel_dst, - sample_ratio_fd, stride_sample_x, stride_sample_y, stride_sample_dst, - dims.first, dims.second, 0, ids_stride, stream); -@@ -830,7 +868,7 @@ static void mul_mat_vec_q_switch_ncols_dst( - case 3: { - constexpr int c_ncols_dst = 3; - std::pair dims = calc_launch_params(c_ncols_dst, nrows_x, nchannels_dst, nsamples_dst, warp_size, table_id); -- mul_mat_vec_q_switch_fusion(vx, vy, ids, fusion, dst, ncols_x, nchannels_y_fd, stride_row_x, stride_col_y, stride_col_dst, -+ mul_mat_vec_q_switch_fusion(vx, vy, ids, fusion, dst, ncols_x, nchannels_y_fd, stride_row_x, stride_col_y, stride_col_dst, - channel_ratio_fd, stride_channel_x, stride_channel_y, stride_channel_dst, - sample_ratio_fd, stride_sample_x, stride_sample_y, stride_sample_dst, - dims.first, dims.second, 0, ids_stride, stream); -@@ -838,7 +876,7 @@ static void mul_mat_vec_q_switch_ncols_dst( - case 4: { - constexpr int c_ncols_dst = 4; - std::pair dims = calc_launch_params(c_ncols_dst, nrows_x, nchannels_dst, nsamples_dst, warp_size, table_id); -- mul_mat_vec_q_switch_fusion(vx, vy, ids, fusion, dst, ncols_x, nchannels_y_fd, stride_row_x, stride_col_y, stride_col_dst, -+ mul_mat_vec_q_switch_fusion(vx, vy, ids, fusion, dst, ncols_x, nchannels_y_fd, stride_row_x, stride_col_y, stride_col_dst, - channel_ratio_fd, stride_channel_x, stride_channel_y, stride_channel_dst, - sample_ratio_fd, stride_sample_x, stride_sample_y, stride_sample_dst, - dims.first, dims.second, 0, ids_stride, stream); -@@ -846,7 +884,7 @@ static void mul_mat_vec_q_switch_ncols_dst( - case 5: { - constexpr int c_ncols_dst = 5; - std::pair dims = calc_launch_params(c_ncols_dst, nrows_x, nchannels_dst, nsamples_dst, warp_size, table_id); -- mul_mat_vec_q_switch_fusion(vx, vy, ids, fusion, dst, ncols_x, nchannels_y_fd, stride_row_x, stride_col_y, stride_col_dst, -+ mul_mat_vec_q_switch_fusion(vx, vy, ids, fusion, dst, ncols_x, nchannels_y_fd, stride_row_x, stride_col_y, stride_col_dst, - channel_ratio_fd, stride_channel_x, stride_channel_y, stride_channel_dst, - sample_ratio_fd, stride_sample_x, stride_sample_y, stride_sample_dst, - dims.first, dims.second, 0, ids_stride, stream); -@@ -854,7 +892,7 @@ static void mul_mat_vec_q_switch_ncols_dst( - case 6: { - constexpr int c_ncols_dst = 6; - std::pair dims = calc_launch_params(c_ncols_dst, nrows_x, nchannels_dst, nsamples_dst, warp_size, table_id); -- mul_mat_vec_q_switch_fusion(vx, vy, ids, fusion, dst, ncols_x, nchannels_y_fd, stride_row_x, stride_col_y, stride_col_dst, -+ mul_mat_vec_q_switch_fusion(vx, vy, ids, fusion, dst, ncols_x, nchannels_y_fd, stride_row_x, stride_col_y, stride_col_dst, - channel_ratio_fd, stride_channel_x, stride_channel_y, stride_channel_dst, - sample_ratio_fd, stride_sample_x, stride_sample_y, stride_sample_dst, - dims.first, dims.second, 0, ids_stride, stream); -@@ -862,7 +900,7 @@ static void mul_mat_vec_q_switch_ncols_dst( - case 7: { - constexpr int c_ncols_dst = 7; - std::pair dims = calc_launch_params(c_ncols_dst, nrows_x, nchannels_dst, nsamples_dst, warp_size, table_id); -- mul_mat_vec_q_switch_fusion(vx, vy, ids, fusion, dst, ncols_x, nchannels_y_fd, stride_row_x, stride_col_y, stride_col_dst, -+ mul_mat_vec_q_switch_fusion(vx, vy, ids, fusion, dst, ncols_x, nchannels_y_fd, stride_row_x, stride_col_y, stride_col_dst, - channel_ratio_fd, stride_channel_x, stride_channel_y, stride_channel_dst, - sample_ratio_fd, stride_sample_x, stride_sample_y, stride_sample_dst, - dims.first, dims.second, 0, ids_stride, stream); -@@ -870,7 +908,7 @@ static void mul_mat_vec_q_switch_ncols_dst( - case 8: { - constexpr int c_ncols_dst = 8; - std::pair dims = calc_launch_params(c_ncols_dst, nrows_x, nchannels_dst, nsamples_dst, warp_size, table_id); -- mul_mat_vec_q_switch_fusion(vx, vy, ids, fusion, dst, ncols_x, nchannels_y_fd, stride_row_x, stride_col_y, stride_col_dst, -+ mul_mat_vec_q_switch_fusion(vx, vy, ids, fusion, dst, ncols_x, nchannels_y_fd, stride_row_x, stride_col_y, stride_col_dst, - channel_ratio_fd, stride_channel_x, stride_channel_y, stride_channel_dst, - sample_ratio_fd, stride_sample_x, stride_sample_y, stride_sample_dst, - dims.first, dims.second, 0, ids_stride, stream); -@@ -889,7 +927,8 @@ static void mul_mat_vec_q_switch_type( - const int nchannels_x, const int nchannels_y, const int nchannels_dst, - const int stride_channel_x, const int stride_channel_y, const int stride_channel_dst, - const int nsamples_x, const int nsamples_dst, const int stride_sample_x, const int stride_sample_y, const int stride_sample_dst, -- const int ids_stride, cudaStream_t stream) { -+ const int ids_stride, const bool q8_planar, cudaStream_t stream) { -+ GGML_ASSERT(!q8_planar || type_x == GGML_TYPE_Q8_0); - switch (type_x) { - case GGML_TYPE_Q1_0: - mul_mat_vec_q_switch_ncols_dst -@@ -922,10 +961,17 @@ static void mul_mat_vec_q_switch_type( - nsamples_x, nsamples_dst, stride_sample_x, stride_sample_y, stride_sample_dst, ids_stride, stream); - break; - case GGML_TYPE_Q8_0: -- mul_mat_vec_q_switch_ncols_dst -- (vx, vy, ids, fusion, dst, ncols_x, nrows_x, ncols_dst, stride_row_x, stride_col_y, stride_col_dst, -- nchannels_x, nchannels_y, nchannels_dst, stride_channel_x, stride_channel_y, stride_channel_dst, -- nsamples_x, nsamples_dst, stride_sample_x, stride_sample_y, stride_sample_dst, ids_stride, stream); -+ if (q8_planar) { -+ mul_mat_vec_q_switch_ncols_dst -+ (vx, vy, ids, fusion, dst, ncols_x, nrows_x, ncols_dst, stride_row_x, stride_col_y, stride_col_dst, -+ nchannels_x, nchannels_y, nchannels_dst, stride_channel_x, stride_channel_y, stride_channel_dst, -+ nsamples_x, nsamples_dst, stride_sample_x, stride_sample_y, stride_sample_dst, ids_stride, stream); -+ } else { -+ mul_mat_vec_q_switch_ncols_dst -+ (vx, vy, ids, fusion, dst, ncols_x, nrows_x, ncols_dst, stride_row_x, stride_col_y, stride_col_dst, -+ nchannels_x, nchannels_y, nchannels_dst, stride_channel_x, stride_channel_y, stride_channel_dst, -+ nsamples_x, nsamples_dst, stride_sample_x, stride_sample_y, stride_sample_dst, ids_stride, stream); -+ } - break; - case GGML_TYPE_MXFP4: - mul_mat_vec_q_switch_ncols_dst -@@ -1059,13 +1105,14 @@ void ggml_cuda_mul_mat_vec_q( - - if (fusion) { - GGML_ASSERT( !ids || dst->ne[2] == 1); -- GGML_ASSERT( ids || dst->ne[1] == 1); -+ GGML_ASSERT( ids || dst->ne[1] <= 2); - - if (fusion->x_bias) { - GGML_ASSERT(fusion->x_bias->type == GGML_TYPE_F32); - GGML_ASSERT(fusion->x_bias->ne[0] == dst->ne[0]); - GGML_ASSERT(!ids || fusion->x_bias->ne[1] == src0->ne[2]); - fusion_local.x_bias = fusion->x_bias->data; -+ fusion_local.x_bias_broadcast = ggml_nelements(fusion->x_bias) == dst->ne[0]; - } - if (fusion->gate) { - GGML_ASSERT(fusion->gate->type == src0->type && ggml_are_same_stride(fusion->gate, src0)); -@@ -1076,8 +1123,10 @@ void ggml_cuda_mul_mat_vec_q( - GGML_ASSERT(fusion->gate_bias->ne[0] == dst->ne[0]); - GGML_ASSERT(!ids || fusion->gate_bias->ne[1] == src0->ne[2]); - fusion_local.gate_bias = fusion->gate_bias->data; -+ fusion_local.gate_bias_broadcast = ggml_nelements(fusion->gate_bias) == dst->ne[0]; - } - fusion_local.glu_op = fusion->glu_op; -+ fusion_local.silu = fusion->silu; - } - - // If src0 is a temporary compute buffer, clear any potential padding. -@@ -1121,12 +1170,18 @@ void ggml_cuda_mul_mat_vec_q( - const int64_t stride_channel_y = ids ? s11 : s12; - - const int64_t ids_stride = ids ? ids->nb[1] / ggml_type_size(ids->type) : 0; -+ const bool q8_planar = (src0->flags & GGML_TENSOR_FLAG_Q8_PLANAR) != 0; -+ if (q8_planar) { -+ GGML_ASSERT(src0->type == GGML_TYPE_Q8_0 && ids == nullptr); -+ GGML_ASSERT(src0->ne[2] == 1 && src0->ne[3] == 1 && ggml_is_contiguous(src0)); -+ } - - mul_mat_vec_q_switch_type( - src0->data, src0->type, src1_q8_1.get(), ids_d, fusion_local, dst_d, ne00, - ne01, ncols_dst, s01, stride_col_y, stride_col_dst, - ne02, nchannels_y, nchannels_dst, s02, stride_channel_y, stride_channel_dst, -- ne03, ne3, s03, s13, s3, ids_stride, stream); -+ ne03, ne3, s03, s13, s3, ids_stride, -+ q8_planar, stream); - } - - void ggml_cuda_op_mul_mat_vec_q( -@@ -1153,9 +1208,11 @@ void ggml_cuda_op_mul_mat_vec_q( - const int stride_col_y = src1_padded_row_size / QK8_1; - - ggml_cuda_mm_fusion_args_device fusion_local{}; -+ GGML_ASSERT((src0->flags & GGML_TENSOR_FLAG_Q8_PLANAR) == 0 && -+ "planar Q8 does not support split-buffer MMVQ"); - mul_mat_vec_q_switch_type( - src0_dd_i, src0->type, src1_ddq_i, nullptr, fusion_local, dst_dd_i, ne00, row_diff, src1_ncols, stride_row_x, stride_col_y, nrows_dst, -- 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 0, stream); -+ 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 0, false, stream); - - GGML_UNUSED_VARS(src1, dst, src1_ddf_i, src1_ncols, src1_padded_row_size); - } -diff --git a/src/ggml-cuda/vecdotq.cuh b/src/ggml-cuda/vecdotq.cuh -index d1741cc8..33b54ab3 100644 ---- a/src/ggml-cuda/vecdotq.cuh -+++ b/src/ggml-cuda/vecdotq.cuh -@@ -237,7 +237,7 @@ template static __device__ __forceinline__ float vec_dot_q5_1_q8_1_imp - return sumi*d5d8 + m5s8 / (QI5_1 / vdr); - } - --#define VDR_Q8_0_Q8_1_MMVQ 2 -+#define VDR_Q8_0_Q8_1_MMVQ 4 - #define VDR_Q8_0_Q8_1_MMQ 8 - - template static __device__ __forceinline__ T vec_dot_q8_0_q8_1_impl( diff --git a/ggml-patches/0007-magpietts-nanocodec.patch b/ggml-patches/0007-magpietts-nanocodec.patch deleted file mode 100644 index b43bea0..0000000 --- a/ggml-patches/0007-magpietts-nanocodec.patch +++ /dev/null @@ -1,732 +0,0 @@ -diff --git a/src/ggml-cuda/CMakeLists.txt b/src/ggml-cuda/CMakeLists.txt -index b54d4a6b..07a94052 100644 ---- a/src/ggml-cuda/CMakeLists.txt -+++ b/src/ggml-cuda/CMakeLists.txt -@@ -15,7 +15,8 @@ if (CUDAToolkit_FOUND) - # 80 == Ampere, asynchronous data loading, faster tensor core instructions - # 86 == RTX 3000, needs CUDA v11.1 - # 89 == RTX 4000, needs CUDA v11.8 -- # 120 == Blackwell, needs CUDA v12.8, FP4 tensor cores -+ # 110 == Jetson Thor, needs CUDA v13.0, Blackwell tensor cores -+ # 120 == Blackwell, needs CUDA v12.8, SM120 FP4 tensor cores - # - # XX-virtual == compile CUDA code as PTX, do JIT compilation to binary code on first run - # XX-real == compile CUDA code as device code for this specific architecture -@@ -36,6 +37,10 @@ if (CUDAToolkit_FOUND) - list(APPEND CMAKE_CUDA_ARCHITECTURES 89-real) - endif() - -+ if (CUDAToolkit_VERSION VERSION_GREATER_EQUAL "13.0") -+ list(APPEND CMAKE_CUDA_ARCHITECTURES 110a-real) -+ endif() -+ - if (CUDAToolkit_VERSION VERSION_GREATER_EQUAL "12.8") - # The CUDA architecture 120f-virtual would in principle work for Blackwell support - # but the newly added "f" suffix conflicted with a preexising regex for validating CUDA architectures in CMake. -@@ -71,16 +76,16 @@ if (CUDAToolkit_FOUND) - FetchContent_MakeAvailable(CCCL) - endif() - -- # Replace any plain 12X CUDA architectures with their "architecture-specific" equivalents 12Xa. -- # 12X is forwards-compatible, 12Xa is not. -- # Notably the Blackwell FP4 tensor core instructions are not forwards compatible and therefore need 12Xa. -+ # Replace plain Blackwell CUDA architectures with their "architecture-specific" equivalents. -+ # 11X/12X are forwards-compatible, 11Xa/12Xa are not. -+ # Notably the Blackwell tensor core instructions are not forwards compatible and therefore need architecture-specific targets. - # But while 12X vs. 12Xa can be checked in device code there is (to my knowledge) no easy way to do the same check in host code. -- # So for now just replace all instances of 12X with 12Xa, this should be fine until Rubin is released. -+ # So for now just replace the supported plain Blackwell targets with architecture-specific targets. - foreach(ARCHS IN ITEMS CMAKE_CUDA_ARCHITECTURES CMAKE_CUDA_ARCHITECTURES_NATIVE) - set(FIXED_ARCHS "") - foreach(ARCH IN LISTS ${ARCHS}) -- if (ARCH MATCHES "^12[0-9](-real|-virtual)?$") -- string(REGEX REPLACE "^(12[0-9])((-real|-virtual)?)$" "\\1a\\2" FIXED_ARCH ${ARCH}) -+ if (ARCH MATCHES "^(110|12[0-9])(-real|-virtual)?$") -+ string(REGEX REPLACE "^(110|12[0-9])((-real|-virtual)?)$" "\\1a\\2" FIXED_ARCH ${ARCH}) - message(STATUS "Replacing ${ARCH} in ${ARCHS} with ${FIXED_ARCH}") - list(APPEND FIXED_ARCHS "${FIXED_ARCH}") - else() -@@ -90,8 +95,8 @@ if (CUDAToolkit_FOUND) - set(${ARCHS} ${FIXED_ARCHS}) - endforeach() - -- # If we try to compile a "native" build it will use the 12X architectures and fail. -- # So we should instead use the native architectures as determined by CMake after replacing 12X with 12Xa. -+ # If we try to compile a "native" build it may use plain Blackwell architectures and fail. -+ # So we should instead use the native architectures as determined by CMake after replacing them with architecture-specific forms. - # But if at the time of the build no GPUs are connected at all CMAKE_CUDA_ARCHITECTURES will contain garbage that we should not use. - if (CMAKE_CUDA_ARCHITECTURES STREQUAL "native" AND CMAKE_CUDA_ARCHITECTURES_NATIVE MATCHES "^[0-9]+(a|f)?(-real|-virtual)?(;[0-9]+(a|f)?(-real|-virtual)?|;)*$") - set(CMAKE_CUDA_ARCHITECTURES ${CMAKE_CUDA_ARCHITECTURES_NATIVE}) -diff --git a/src/ggml-cuda/common.cuh b/src/ggml-cuda/common.cuh -index 9bd40faa..56fe1bb7 100644 ---- a/src/ggml-cuda/common.cuh -+++ b/src/ggml-cuda/common.cuh -@@ -25,6 +25,7 @@ - #include - #include - #include -+#include - #include - #include - #include -@@ -50,7 +51,10 @@ - #define GGML_CUDA_CC_TURING 750 - #define GGML_CUDA_CC_AMPERE 800 - #define GGML_CUDA_CC_ADA_LOVELACE 890 --// While BW spans CC 1000, 1100 & 1200, we are integrating Tensor Core instructions available to 1200 family, see -+// Jetson Thor is Blackwell SM110. The hand-written FP4 block-scale PTX below is currently SM120-only, but SM110 -+// should still be detected so ggml can favor CUDA library kernels that may use Thor tcgen05 tensor cores. -+#define GGML_CUDA_CC_THOR 1100 -+// While BW spans CC 1000, 1100 & 1200, the hand-written FP4 path integrates Tensor Core instructions available to 1200 family, see - // https://docs.nvidia.com/cutlass/media/docs/cpp/blackwell_functionality.html#blackwell-sm120-gemms - #define GGML_CUDA_CC_BLACKWELL 1200 - #define GGML_CUDA_CC_DGX_SPARK 1210 -@@ -315,6 +319,10 @@ static bool amd_wmma_available(const int cc) { - return (GGML_CUDA_CC_IS_RDNA4(cc) || GGML_CUDA_CC_IS_RDNA3(cc)); - } - -+static bool thor_mma_available(const int cc) { -+ return GGML_CUDA_CC_IS_NVIDIA(cc) && cc == GGML_CUDA_CC_THOR; -+} -+ - static bool volta_mma_available(const int cc) { - return GGML_CUDA_CC_IS_NVIDIA(cc) && ggml_cuda_highest_compiled_arch(cc) == GGML_CUDA_CC_VOLTA; - } -@@ -1379,14 +1387,34 @@ struct ggml_backend_cuda_context { - - int64_t last_graph_eviction_sweep = 0; - -+ static int64_t cuda_graph_env_ms_to_us(const char * name, int64_t default_ms) { -+ const char * value = getenv(name); -+ if (value == nullptr || value[0] == '\0') { -+ return default_ms * 1000; -+ } -+ -+ char * end = nullptr; -+ const long long parsed_ms = std::strtoll(value, &end, 10); -+ if (end == value || *end != '\0' || parsed_ms < 0) { -+ return default_ms * 1000; -+ } -+ return (int64_t) parsed_ms * 1000; -+ } -+ - ggml_cuda_graph * cuda_graph(const void * first_node_ptr) { - const int64_t time_now = ggml_time_us(); -- -- // sweep every 5s, evicting cuda graphs unused for >=10s -- if (time_now - last_graph_eviction_sweep >= 5'000'000) { -+ static const int64_t sweep_interval_us = -+ cuda_graph_env_ms_to_us("GGML_CUDA_GRAPH_SWEEP_MS", 5000); -+ static const int64_t evict_after_us = -+ cuda_graph_env_ms_to_us("GGML_CUDA_GRAPH_EVICT_AFTER_MS", 10000); -+ -+ // By default sweep every 5s, evicting CUDA graphs unused for >=10s. -+ // Set GGML_CUDA_GRAPH_EVICT_AFTER_MS=0 to keep captured graphs resident. -+ if (evict_after_us > 0 && sweep_interval_us > 0 -+ && time_now - last_graph_eviction_sweep >= sweep_interval_us) { - last_graph_eviction_sweep = time_now; - for (auto it = cuda_graphs.begin(); it != cuda_graphs.end(); ) { -- if (time_now - it->second->last_used_time >= 10'000'000) { -+ if (time_now - it->second->last_used_time >= evict_after_us) { - it = cuda_graphs.erase(it); - } else { - ++it; -diff --git a/src/ggml-cuda/conv-transpose-1d.cu b/src/ggml-cuda/conv-transpose-1d.cu -index 8418ba66..20a84516 100644 ---- a/src/ggml-cuda/conv-transpose-1d.cu -+++ b/src/ggml-cuda/conv-transpose-1d.cu -@@ -1,34 +1,38 @@ - #include "conv-transpose-1d.cuh" -+#include "convert.cuh" - --static __global__ void conv_transpose_1d_kernel( -+#include -+ -+template -+static __global__ void conv_transpose_1d_kernel( - const int s0, const int p0, const int d0, const int output_size, - const int src0_ne0, const int src0_ne1, const int src0_ne2, const int src0_ne3, - const int src1_ne0, const int src1_ne1, const int src1_ne2, const int src1_ne3, - const int dst_ne0, const int dst_ne1, const int dst_ne2, const int dst_ne3, -- const float * src0, const float * src1, float * dst) { -+ const T * src0, const float * src1, float * dst) { - int global_index = threadIdx.x + blockIdx.x * blockDim.x; - if (global_index >= output_size) { - return; - } - -- int out_index = global_index / dst_ne0; -+ const int out_t = global_index % dst_ne0; -+ const int out_c = (global_index / dst_ne0) % dst_ne1; -+ const int out_b = global_index / (dst_ne0 * dst_ne1); - - float accumulator = 0; - -- for (int c = 0; c < src0_ne2; c++) { -- int idx = global_index % dst_ne0; -+ const int in_end = min(src1_ne0 - 1, out_t / s0); -+ const int in_start = max(0, (out_t - src0_ne0 + s0) / s0); - -- int kernel_offset = (src0_ne0 * src0_ne1 * c) + (out_index * src0_ne0); -- int input_offset = src1_ne0 * c; -+ for (int c = 0; c < src0_ne2; c++) { -+ const int kernel_offset = src0_ne0 * (out_c + src0_ne1 * c); -+ const int input_offset = src1_ne0 * (c + src1_ne1 * out_b); - -- for (int i = 0; i < src1_ne0; i++) { -- if (!(idx >= i*s0 && idx < i*s0 + src0_ne0)) { -- continue; -- } -- int weight_idx = idx - i*s0; -+ for (int i = in_start; i <= in_end; i++) { -+ const int weight_idx = out_t - i*s0; - -- float kernel_weight = src0[kernel_offset + weight_idx]; -- float input_value = src1[input_offset+i]; -+ const float kernel_weight = ggml_cuda_cast(src0[kernel_offset + weight_idx]); -+ const float input_value = src1[input_offset+i]; - - accumulator += kernel_weight * input_value; - } -@@ -37,26 +41,96 @@ static __global__ void conv_transpose_1d_kernel( - GGML_UNUSED_VARS(p0, d0, src0_ne3, src1_ne3, dst_ne3, src1_ne1, dst_ne1, src1_ne2, dst_ne2); - } - --static void conv_transpose_1d_f32_f32_cuda( -+template -+static __global__ void conv_transpose_1d_grouped2_kernel( - const int s0, const int p0, const int d0, const int output_size, - const int src0_ne0, const int src0_ne1, const int src0_ne2, const int src0_ne3, - const int src1_ne0, const int src1_ne1, const int src1_ne2, const int src1_ne3, - const int dst_ne0, const int dst_ne1, const int dst_ne2, const int dst_ne3, -- const float * src0, const float * src1, float * dst, -+ const T * src0, const float * src1, float * dst) { -+ int global_index = threadIdx.x + blockIdx.x * blockDim.x; -+ if (global_index >= output_size) { -+ return; -+ } -+ -+ const int out_t = global_index % dst_ne0; -+ const int out_c = (global_index / dst_ne0) % dst_ne1; -+ const int out_b = global_index / (dst_ne0 * dst_ne1); -+ -+ const int in_end = min(src1_ne0 - 1, out_t / s0); -+ const int in_start = max(0, (out_t - src0_ne0 + s0) / s0); -+ -+ float accumulator = 0; -+ -+ const int c0 = out_c * 2; -+ const int c1 = c0 + 1; -+ -+ for (int i = in_start; i <= in_end; i++) { -+ const int weight_idx = out_t - i*s0; -+ -+ const int input_offset0 = src1_ne0 * (c0 + src1_ne1 * out_b); -+ const int kernel_offset0 = src0_ne0 * (out_c + src0_ne1 * c0); -+ accumulator += ggml_cuda_cast(src0[kernel_offset0 + weight_idx]) * src1[input_offset0 + i]; -+ -+ const int input_offset1 = src1_ne0 * (c1 + src1_ne1 * out_b); -+ const int kernel_offset1 = src0_ne0 * (out_c + src0_ne1 * c1); -+ accumulator += ggml_cuda_cast(src0[kernel_offset1 + weight_idx]) * src1[input_offset1 + i]; -+ } -+ -+ dst[global_index] = accumulator; -+ GGML_UNUSED_VARS(p0, d0, src0_ne2, src0_ne3, src1_ne2, src1_ne3, dst_ne2, dst_ne3); -+} -+ -+template -+static void conv_transpose_1d_cuda( -+ const int s0, const int p0, const int d0, const int output_size, -+ const int src0_ne0, const int src0_ne1, const int src0_ne2, const int src0_ne3, -+ const int src1_ne0, const int src1_ne1, const int src1_ne2, const int src1_ne3, -+ const int dst_ne0, const int dst_ne1, const int dst_ne2, const int dst_ne3, -+ const T * src0, const float * src1, float * dst, -+ const bool use_grouped2, - cudaStream_t stream) { - - const int num_blocks = (output_size + CUDA_CONV_TRANPOSE_1D_BLOCK_SIZE - 1) / CUDA_CONV_TRANPOSE_1D_BLOCK_SIZE; -- conv_transpose_1d_kernel<<>>( -- s0,p0,d0,output_size, -- src0_ne0, src0_ne1, src0_ne2, src0_ne3, -- src1_ne0, src1_ne1, src1_ne2, src1_ne3, -- dst_ne0, dst_ne1, dst_ne2, dst_ne3, -- src0,src1, dst); -+ if (use_grouped2) { -+ conv_transpose_1d_grouped2_kernel<<>>( -+ s0,p0,d0,output_size, -+ src0_ne0, src0_ne1, src0_ne2, src0_ne3, -+ src1_ne0, src1_ne1, src1_ne2, src1_ne3, -+ dst_ne0, dst_ne1, dst_ne2, dst_ne3, -+ src0,src1, dst); -+ } else { -+ conv_transpose_1d_kernel<<>>( -+ s0,p0,d0,output_size, -+ src0_ne0, src0_ne1, src0_ne2, src0_ne3, -+ src1_ne0, src1_ne1, src1_ne2, src1_ne3, -+ dst_ne0, dst_ne1, dst_ne2, dst_ne3, -+ src0,src1, dst); -+ } -+} -+ -+static bool conv_transpose_1d_use_nanocodec_grouped2( -+ const ggml_tensor * src0, const ggml_tensor * src1, const ggml_tensor * dst, const int s0) { -+ const bool shape_matches = -+ src0->ne[0] == 2*s0 && -+ src0->ne[2] == 2*src0->ne[1] && -+ src1->ne[1] == src0->ne[2] && -+ dst->ne[1] == src0->ne[1] && -+ src1->ne[2] == dst->ne[2] && -+ src1->ne[3] == dst->ne[3]; -+ -+ const char * name = src0->name; -+ const bool is_nanocodec_up_weight = -+ name != nullptr && -+ std::strncmp(name, "dec.up.", 7) == 0 && -+ std::strstr(name + 7, ".w") != nullptr; -+ -+ return shape_matches && is_nanocodec_up_weight; - } - - void ggml_cuda_op_conv_transpose_1d(ggml_backend_cuda_context & ctx, ggml_tensor * dst) { - const ggml_tensor * src0 = dst->src[0]; -- const float * src0_d = (const float *)src0->data; -+ const void * src0_d = src0->data; - - const ggml_tensor * src1 = dst->src[1]; - const float * src1_d = (const float *)src1->data; -@@ -64,7 +138,8 @@ void ggml_cuda_op_conv_transpose_1d(ggml_backend_cuda_context & ctx, ggml_tensor - float * dst_d = (float *)dst->data; - cudaStream_t stream = ctx.stream(); - -- GGML_ASSERT(src0->type == GGML_TYPE_F32); -+ GGML_ASSERT(src0->type == GGML_TYPE_F32 || src0->type == GGML_TYPE_F16); -+ GGML_ASSERT(src1->type == GGML_TYPE_F32); - GGML_ASSERT( dst->type == GGML_TYPE_F32); - - GGML_ASSERT(ggml_is_contiguous(src0)); -@@ -77,10 +152,19 @@ void ggml_cuda_op_conv_transpose_1d(ggml_backend_cuda_context & ctx, ggml_tensor - const int d0 = 1;//opts[4]; - - const int64_t output_size = ggml_nelements(dst); -- -- conv_transpose_1d_f32_f32_cuda(s0, p0, d0, output_size, -- src0->ne[0], src0->ne[1], src0->ne[2], src0->ne[3], -- src1->ne[0], src1->ne[1], src1->ne[2], src1->ne[3], -- dst->ne[0], dst->ne[1], dst->ne[2], dst->ne[3], -- src0_d, src1_d, dst_d, stream); -+ const bool use_grouped2 = conv_transpose_1d_use_nanocodec_grouped2(src0, src1, dst, s0); -+ -+ if (src0->type == GGML_TYPE_F16) { -+ conv_transpose_1d_cuda(s0, p0, d0, output_size, -+ src0->ne[0], src0->ne[1], src0->ne[2], src0->ne[3], -+ src1->ne[0], src1->ne[1], src1->ne[2], src1->ne[3], -+ dst->ne[0], dst->ne[1], dst->ne[2], dst->ne[3], -+ (const half *) src0_d, src1_d, dst_d, use_grouped2, stream); -+ } else { -+ conv_transpose_1d_cuda(s0, p0, d0, output_size, -+ src0->ne[0], src0->ne[1], src0->ne[2], src0->ne[3], -+ src1->ne[0], src1->ne[1], src1->ne[2], src1->ne[3], -+ dst->ne[0], dst->ne[1], dst->ne[2], dst->ne[3], -+ (const float *) src0_d, src1_d, dst_d, use_grouped2, stream); -+ } - } -diff --git a/src/ggml-cuda/ggml-cuda.cu b/src/ggml-cuda/ggml-cuda.cu -index a531ce07..4b4a8488 100644 ---- a/src/ggml-cuda/ggml-cuda.cu -+++ b/src/ggml-cuda/ggml-cuda.cu -@@ -2485,8 +2485,8 @@ static bool ggml_cuda_should_fuse_mul_mat_vec_f(const ggml_tensor * tensor) { - return false; - } - -- //we only support fusion for ncols_dst = 1 -- if (tensor->op == GGML_OP_MUL_MAT && dst->ne[1] != 1) { -+ // MMVF supports a two-column epilogue so paired CFG lanes can share each weight-row load. -+ if (tensor->op == GGML_OP_MUL_MAT && dst->ne[1] > 2) { - return false; - } - -@@ -3310,6 +3310,23 @@ static const void * ggml_cuda_graph_get_key(ggml_cgraph * cgraph) { - return cgraph->nodes[0]; - } - -+static const char * ggml_cuda_graph_tensor_name(const ggml_tensor * tensor) { -+ return tensor && tensor->name[0] ? tensor->name : "(unnamed)"; -+} -+ -+static void ggml_cuda_graph_log_event( -+ const char * func, -+ const char * event, -+ const ggml_cgraph * cgraph, -+ const void * graph_key) { -+ const int n_nodes = cgraph ? cgraph->n_nodes : 0; -+ const ggml_tensor * first = n_nodes > 0 ? cgraph->nodes[0] : nullptr; -+ const ggml_tensor * last = n_nodes > 0 ? cgraph->nodes[n_nodes - 1] : nullptr; -+ GGML_LOG_DEBUG("%s: CUDA graph %s key=%p uid=%" PRIu64 " nodes=%d first=%s last=%s\n", -+ func, event, graph_key, cgraph ? cgraph->uid : 0, n_nodes, -+ ggml_cuda_graph_tensor_name(first), ggml_cuda_graph_tensor_name(last)); -+} -+ - static bool ggml_cuda_graph_update_required(ggml_backend_cuda_context * cuda_ctx, ggml_cgraph * cgraph) { - bool res = false; - -@@ -3318,7 +3335,6 @@ static bool ggml_cuda_graph_update_required(ggml_backend_cuda_context * cuda_ctx - - if (cgraph->uid != 0 && - cgraph->uid == graph->uid) { -- GGML_LOG_DEBUG("CUDA Graph id %zu reused\n", cgraph->uid); - GGML_ASSERT((int)graph->node_props.size() == cgraph->n_nodes); - return false; - } -@@ -3891,6 +3907,89 @@ static bool ggml_cuda_can_fuse(const struct ggml_cgraph * cgraph, - return false; - } - -+static bool ggml_cuda_node_can_be_elided(const struct ggml_cgraph * cgraph, int node_idx, int32_t expected_uses) { -+ const ggml_tensor * node = cgraph->nodes[node_idx]; -+ return (node->flags & GGML_TENSOR_FLAG_COMPUTE) != 0 && -+ (node->flags & GGML_TENSOR_FLAG_OUTPUT) == 0 && -+ ggml_node_get_use_count(cgraph, node_idx) == expected_uses; -+} -+ -+static int ggml_cuda_try_fuse_half_snake(ggml_backend_cuda_context * cuda_ctx, ggml_cgraph * cgraph, int i) { -+ if (i <= 0 || i + 7 >= cgraph->n_nodes) { -+ return 0; -+ } -+ -+ ggml_tensor * mul0 = cgraph->nodes[i + 0]; -+ ggml_tensor * sin = cgraph->nodes[i + 1]; -+ ggml_tensor * sqr = cgraph->nodes[i + 2]; -+ ggml_tensor * mul1 = cgraph->nodes[i + 3]; -+ ggml_tensor * add = cgraph->nodes[i + 4]; -+ ggml_tensor * view_l = cgraph->nodes[i + 5]; -+ ggml_tensor * lrelu = cgraph->nodes[i + 6]; -+ ggml_tensor * concat = cgraph->nodes[i + 7]; -+ -+ if (mul0->op != GGML_OP_MUL || sin->op != GGML_OP_SIN || sqr->op != GGML_OP_SQR || -+ mul1->op != GGML_OP_MUL || add->op != GGML_OP_ADD || view_l->op != GGML_OP_VIEW || -+ lrelu->op != GGML_OP_LEAKY_RELU || concat->op != GGML_OP_CONCAT) { -+ return 0; -+ } -+ -+ if (!ggml_cuda_node_can_be_elided(cgraph, i + 0, 1) || -+ !ggml_cuda_node_can_be_elided(cgraph, i + 1, 1) || -+ !ggml_cuda_node_can_be_elided(cgraph, i + 2, 1) || -+ !ggml_cuda_node_can_be_elided(cgraph, i + 3, 1) || -+ !ggml_cuda_node_can_be_elided(cgraph, i + 4, 1) || -+ !ggml_cuda_node_can_be_elided(cgraph, i + 5, 1) || -+ !ggml_cuda_node_can_be_elided(cgraph, i + 6, 1)) { -+ return 0; -+ } -+ -+ const ggml_tensor * x_snake = ggml_are_same_shape(mul0, mul0->src[0]) ? mul0->src[0] : mul0->src[1]; -+ const ggml_tensor * alpha = (x_snake == mul0->src[0]) ? mul0->src[1] : mul0->src[0]; -+ if (x_snake->op != GGML_OP_VIEW || x_snake != cgraph->nodes[i - 1]) { -+ return 0; -+ } -+ if (!ggml_cuda_node_can_be_elided(cgraph, i - 1, 2)) { -+ return 0; -+ } -+ -+ const ggml_tensor * inv_b = (mul1->src[0] == sqr) ? mul1->src[1] : mul1->src[0]; -+ const ggml_tensor * x_in_add = (add->src[0] == mul1) ? add->src[1] : add->src[0]; -+ if (sin->src[0] != mul0 || sqr->src[0] != sin || (mul1->src[0] != sqr && mul1->src[1] != sqr) || -+ x_in_add != x_snake || lrelu->src[0] != view_l || concat->src[0] != add || concat->src[1] != lrelu) { -+ return 0; -+ } -+ -+ const int32_t concat_dim = ((const int32_t *) concat->op_params)[0]; -+ const bool type_ok = (x_snake->type == GGML_TYPE_F32 || x_snake->type == GGML_TYPE_F16 || x_snake->type == GGML_TYPE_BF16) && -+ x_snake->type == view_l->type && x_snake->type == concat->type; -+ const bool shape_ok = x_snake->view_src == view_l->view_src && -+ concat_dim == 1 && -+ x_snake->ne[0] == view_l->ne[0] && -+ x_snake->ne[2] == view_l->ne[2] && -+ x_snake->ne[3] == view_l->ne[3] && -+ concat->ne[0] == x_snake->ne[0] && -+ concat->ne[1] == x_snake->ne[1] + view_l->ne[1] && -+ concat->ne[2] == x_snake->ne[2] && -+ concat->ne[3] == x_snake->ne[3] && -+ alpha->type == GGML_TYPE_F32 && -+ inv_b->type == GGML_TYPE_F32 && -+ ggml_is_contiguous(alpha) && -+ ggml_is_contiguous(inv_b) && -+ ggml_nelements(alpha) == x_snake->ne[1] && -+ ggml_nelements(inv_b) == x_snake->ne[1] && -+ ggml_is_contiguous(x_snake) && -+ ggml_is_contiguous(view_l) && -+ ggml_is_contiguous(concat); -+ -+ if (!type_ok || !shape_ok) { -+ return 0; -+ } -+ -+ ggml_cuda_op_half_snake_fused(*cuda_ctx, x_snake, view_l, alpha, inv_b, lrelu, concat); -+ return 7; -+} -+ - // try and fuse nodes and return the number of nodes to skip - static int ggml_cuda_try_fuse(ggml_backend_cuda_context * cuda_ctx, ggml_cgraph * cgraph, int i) { - -@@ -3901,6 +3999,10 @@ static int ggml_cuda_try_fuse(ggml_backend_cuda_context * cuda_ctx, ggml_cgraph - - ggml_tensor * node = cgraph->nodes[i]; - -+ if (int n_fused = ggml_cuda_try_fuse_half_snake(cuda_ctx, cgraph, i)) { -+ return n_fused; -+ } -+ - //topk-moe - if (cgraph->nodes[i]->op == GGML_OP_UNARY || cgraph->nodes[i]->op == GGML_OP_SOFT_MAX || - cgraph->nodes[i]->op == GGML_OP_ARGSORT) { -@@ -4623,7 +4725,7 @@ static enum ggml_status ggml_backend_cuda_graph_compute(ggml_backend_t backend, - // Warmup: need at least 2 calls with no property change on the 2nd call - if (!properties_changed) { - graph->warmup_complete = true; -- GGML_LOG_DEBUG("%s: CUDA graph warmup complete\n", __func__); -+ ggml_cuda_graph_log_event(__func__, "warmup complete", cgraph, graph_key); - use_cuda_graph = true; - cuda_graph_update_required = true; - } -@@ -4633,7 +4735,7 @@ static enum ggml_status ggml_backend_cuda_graph_compute(ggml_backend_t backend, - if (properties_changed) { - // Properties changed - reset warmup, execute directly until stable again - graph->warmup_complete = false; -- GGML_LOG_DEBUG("%s: CUDA graph warmup reset\n", __func__); -+ ggml_cuda_graph_log_event(__func__, "warmup reset", cgraph, graph_key); - } else { - use_cuda_graph = true; - cuda_graph_update_required = graph->instance == nullptr; -@@ -5434,7 +5536,7 @@ static bool ggml_backend_cuda_device_supports_op(ggml_backend_dev_t dev, const g - { - ggml_type src0_type = op->src[0]->type; - ggml_type src1_type = op->src[1]->type; -- if (src0_type == GGML_TYPE_F32 && src1_type == GGML_TYPE_F32) { -+ if ((src0_type == GGML_TYPE_F32 || src0_type == GGML_TYPE_F16) && src1_type == GGML_TYPE_F32) { - return true; - } - return false; -@@ -5705,6 +5807,12 @@ static ggml_backend_feature * ggml_backend_cuda_get_features(ggml_backend_reg_t - - { - const auto & info = ggml_cuda_info(); -+ for (int id = 0; id < info.device_count; ++id) { -+ if (thor_mma_available(info.devices[id].cc)) { -+ features.push_back({ "BLACKWELL_THOR_SM110", "1"}); -+ break; -+ } -+ } - for (int id = 0; id < info.device_count; ++id) { - if (blackwell_mma_available(info.devices[id].cc)) { - features.push_back({ "BLACKWELL_NATIVE_FP4", "1"}); -diff --git a/src/ggml-cuda/mmf.cu b/src/ggml-cuda/mmf.cu -index aad4c34a..628d8718 100644 ---- a/src/ggml-cuda/mmf.cu -+++ b/src/ggml-cuda/mmf.cu -@@ -159,6 +159,11 @@ bool ggml_cuda_should_use_mmf(enum ggml_type type, int cc, int warp_size, const - return false; - } - -+ // Thor's tcgen05 paths are provided by CUDA libraries today; avoid ggml's warp-level matrix kernels there. -+ if (thor_mma_available(cc) && (type == GGML_TYPE_F32 || type == GGML_TYPE_F16 || type == GGML_TYPE_BF16)) { -+ return false; -+ } -+ - if (mul_mat_id) { - if (src0_ne[1] <= 1024 && src1_ncols > 512) { - return false; -diff --git a/src/ggml-cuda/mmvf.cu b/src/ggml-cuda/mmvf.cu -index d9147202..a347447c 100644 ---- a/src/ggml-cuda/mmvf.cu -+++ b/src/ggml-cuda/mmvf.cu -@@ -383,7 +383,10 @@ static void mul_mat_vec_f_switch_fusion( - const dim3 & block_dims, const dim3 & block_nums, const int nbytes_shared, const int ids_stride, const cudaStream_t stream) { - - const bool has_fusion = fusion.gate != nullptr || fusion.x_bias != nullptr || fusion.gate_bias != nullptr; -- if constexpr (ncols_dst == 1) { -+ // The same epilogue addressing works for two adjacent columns. This is especially useful -+ // for classifier-free guidance: both lanes share the weight-row load and still fold the -+ // residual/bias writeback into the projection kernel. -+ if constexpr (ncols_dst <= 2) { - if (has_fusion) { - mul_mat_vec_f<<>> - (x, y, ids, fusion, dst, ncols, nchannels_y, stride_row, stride_col_y, stride_col_dst, -@@ -393,7 +396,7 @@ static void mul_mat_vec_f_switch_fusion( - } - } - -- GGML_ASSERT(!has_fusion && "fusion only supported for ncols_dst=1"); -+ GGML_ASSERT(!has_fusion && "fusion only supported for ncols_dst<=2"); - - mul_mat_vec_f<<>> - (x, y, ids, fusion, dst, ncols, nchannels_y, stride_row, stride_col_y, stride_col_dst, -@@ -436,6 +439,11 @@ void launch_mul_mat_vec_f_cuda( - block_size_best = block_size; - } - } -+ if constexpr (std::is_same_v) { -+ if (warp_size == 32 && ncols >= 768 && ncols_dst <= 2) { -+ block_size_best = 96; -+ } -+ } - - const bool has_fusion = fusion.gate != nullptr || fusion.x_bias != nullptr || fusion.gate_bias != nullptr; - -@@ -650,7 +658,7 @@ void ggml_cuda_mul_mat_vec_f(ggml_backend_cuda_context & ctx, const ggml_tensor - - if (fusion) { - GGML_ASSERT( !ids || dst->ne[2] == 1); -- GGML_ASSERT( ids || dst->ne[1] == 1); -+ GGML_ASSERT( ids || dst->ne[1] <= 2); - if (fusion->x_bias) { - GGML_ASSERT(fusion->x_bias->type == GGML_TYPE_F32); - GGML_ASSERT(fusion->x_bias->ne[0] == dst->ne[0]); -@@ -793,9 +801,15 @@ bool ggml_cuda_should_use_mmvf(enum ggml_type type, int cc, const int64_t * src0 - } - } - -+ // Thor's tcgen05 paths are provided by CUDA libraries today; avoid ggml's warp-level vector kernels there. -+ const bool prefer_cublas_tcgen05 = thor_mma_available(cc); -+ - switch (type) { - case GGML_TYPE_F32: - if (GGML_CUDA_CC_IS_NVIDIA(cc)) { -+ if (prefer_cublas_tcgen05) { -+ return false; -+ } - if (ampere_mma_available(cc)) { - return ne11 <= 3; - } -@@ -812,9 +826,12 @@ bool ggml_cuda_should_use_mmvf(enum ggml_type type, int cc, const int64_t * src0 - return ne11 <= 8; - case GGML_TYPE_F16: - if (GGML_CUDA_CC_IS_NVIDIA(cc)) { -+ if (prefer_cublas_tcgen05) { -+ return false; -+ } - const bool src0_small = (src0_ne[1] <= 512 || src0_ne[2]*src0_ne[3] == 1); - if (ampere_mma_available(cc)) { -- return src0_small && ne11 == 1; -+ return src0_small && ne11 <= 2; - } - if (cc >= GGML_CUDA_CC_ADA_LOVELACE) { - return src0_small && ne11 <= 4; -@@ -838,6 +855,9 @@ bool ggml_cuda_should_use_mmvf(enum ggml_type type, int cc, const int64_t * src0 - return ne11 <= 8; - case GGML_TYPE_BF16: - if (GGML_CUDA_CC_IS_NVIDIA(cc)) { -+ if (prefer_cublas_tcgen05) { -+ return false; -+ } - const bool src0_small = (src0_ne[1] <= 512 || src0_ne[2]*src0_ne[3] == 1); - if (ampere_mma_available(cc)) { - return src0_small && ne11 == 1; -diff --git a/src/ggml-cuda/snake.cu b/src/ggml-cuda/snake.cu -index 384638c1..6f4b219c 100644 ---- a/src/ggml-cuda/snake.cu -+++ b/src/ggml-cuda/snake.cu -@@ -1,6 +1,8 @@ - #include "snake.cuh" - #include "convert.cuh" - -+#include -+ - // Fused Snake activation: y = x + sin^2(a * x) * inv_b - // x: [T, C] (T contiguous), a: [1, C], inv_b: [1, C] - // Supports F32, F16, BF16 data with F32 compute. -@@ -70,3 +72,80 @@ void ggml_cuda_op_snake_fused(ggml_backend_cuda_context & ctx, - ggml_tensor * dst) { - launch_snake(ctx, x, a, inv_b, dst); - } -+ -+template -+static __global__ void half_snake_kernel( -+ const T * __restrict__ x_snake, -+ const T * __restrict__ x_lrelu, -+ const float * __restrict__ a, -+ const float * __restrict__ inv_b, -+ T * __restrict__ dst, -+ const int total, -+ const int T_len, -+ const int snake_channels, -+ const int lrelu_channels, -+ const int channels, -+ const float negative_slope) { -+ const int idx = blockIdx.x * blockDim.x + threadIdx.x; -+ if (idx >= total) return; -+ -+ const int t = idx % T_len; -+ const int c = (idx / T_len) % channels; -+ const int plane = idx / (T_len * channels); -+ -+ if (c < snake_channels) { -+ const int src_idx = plane * T_len * snake_channels + c * T_len + t; -+ const float xi = ggml_cuda_cast(x_snake[src_idx]); -+ const float s = sinf(a[c] * xi); -+ dst[idx] = ggml_cuda_cast(xi + s * s * inv_b[c]); -+ } else { -+ const int lc = c - snake_channels; -+ const int src_idx = plane * T_len * lrelu_channels + lc * T_len + t; -+ const float xi = ggml_cuda_cast(x_lrelu[src_idx]); -+ dst[idx] = ggml_cuda_cast(fmaxf(xi, 0.0f) + fminf(xi, 0.0f) * negative_slope); -+ } -+} -+ -+void ggml_cuda_op_half_snake_fused(ggml_backend_cuda_context & ctx, -+ const ggml_tensor * x_snake, -+ const ggml_tensor * x_lrelu, -+ const ggml_tensor * a, -+ const ggml_tensor * inv_b, -+ const ggml_tensor * lrelu, -+ ggml_tensor * dst) { -+ float negative_slope; -+ std::memcpy(&negative_slope, lrelu->op_params, sizeof(float)); -+ -+ const int T = (int) dst->ne[0]; -+ const int snake_channels = (int) x_snake->ne[1]; -+ const int lrelu_channels = (int) x_lrelu->ne[1]; -+ const int channels = (int) dst->ne[1]; -+ const int total = (int) ggml_nelements(dst); -+ -+ const int block_size = 256; -+ const int grid_size = (total + block_size - 1) / block_size; -+ cudaStream_t stream = ctx.stream(); -+ -+ const float * a_d = (const float *) a->data; -+ const float * inv_b_d = (const float *) inv_b->data; -+ -+ switch (dst->type) { -+ case GGML_TYPE_F32: { -+ half_snake_kernel<<>>( -+ (const float *) x_snake->data, (const float *) x_lrelu->data, a_d, inv_b_d, (float *) dst->data, -+ total, T, snake_channels, lrelu_channels, channels, negative_slope); -+ } break; -+ case GGML_TYPE_F16: { -+ half_snake_kernel<<>>( -+ (const half *) x_snake->data, (const half *) x_lrelu->data, a_d, inv_b_d, (half *) dst->data, -+ total, T, snake_channels, lrelu_channels, channels, negative_slope); -+ } break; -+ case GGML_TYPE_BF16: { -+ half_snake_kernel<<>>( -+ (const nv_bfloat16 *) x_snake->data, (const nv_bfloat16 *) x_lrelu->data, a_d, inv_b_d, -+ (nv_bfloat16 *) dst->data, total, T, snake_channels, lrelu_channels, channels, negative_slope); -+ } break; -+ default: -+ GGML_ABORT("half_snake: unsupported type"); -+ } -+} -diff --git a/src/ggml-cuda/snake.cuh b/src/ggml-cuda/snake.cuh -index 7f6f1cb3..7a3e7aa6 100644 ---- a/src/ggml-cuda/snake.cuh -+++ b/src/ggml-cuda/snake.cuh -@@ -6,3 +6,11 @@ void ggml_cuda_op_snake_fused(ggml_backend_cuda_context & ctx, - const ggml_tensor * a, - const ggml_tensor * inv_b, - ggml_tensor * dst); -+ -+void ggml_cuda_op_half_snake_fused(ggml_backend_cuda_context & ctx, -+ const ggml_tensor * x_snake, -+ const ggml_tensor * x_lrelu, -+ const ggml_tensor * a, -+ const ggml_tensor * inv_b, -+ const ggml_tensor * lrelu, -+ ggml_tensor * dst); diff --git a/ggml-patches/0008-cublas-bf16-projections.patch b/ggml-patches/0008-cublas-bf16-projections.patch deleted file mode 100644 index c31c810..0000000 --- a/ggml-patches/0008-cublas-bf16-projections.patch +++ /dev/null @@ -1,326 +0,0 @@ -diff --git a/src/ggml-cuda/ggml-cuda.cu b/src/ggml-cuda/ggml-cuda.cu -index 58a2330d..1a1b58fb 100644 ---- a/src/ggml-cuda/ggml-cuda.cu -+++ b/src/ggml-cuda/ggml-cuda.cu -@@ -2187,8 +2187,65 @@ struct batched_mul_mat_traits { - static inline auto get_nc_converter(ggml_type src_type) { return ggml_get_to_fp16_nc_cuda(src_type); } - }; - -+static __global__ void bf16_to_f32_add_row_bias( -+ const nv_bfloat16 * __restrict__ src, const float * __restrict__ bias, -+ float * __restrict__ dst, int64_t rows) { -+#if defined(__CUDA_ARCH__) && __CUDA_ARCH__ >= GGML_CUDA_CC_AMPERE -+ const int64_t row = int64_t(blockIdx.x) * blockDim.x + threadIdx.x; -+ if (row >= rows) { -+ return; -+ } -+ const int64_t i = int64_t(blockIdx.y) * rows + row; -+ dst[i] = __bfloat162float(src[i]) + bias[row]; -+#else -+ GGML_UNUSED_VARS(src, bias, dst, rows); -+ NO_DEVICE_CODE; -+#endif -+} -+ -+static void launch_bf16_to_f32_add_row_bias( -+ const nv_bfloat16 * src, const float * bias, float * dst, -+ int64_t rows, int64_t cols, cudaStream_t stream) { -+ constexpr int block_size = 256; -+ GGML_ASSERT(cols <= 65535); -+ const dim3 block(block_size, 1, 1); -+ const dim3 grid((rows + block_size - 1) / block_size, cols, 1); -+ bf16_to_f32_add_row_bias<<>>(src, bias, dst, rows); -+} -+ -+static __global__ void bf16_add_row_bias_silu_to_bf16( -+ const nv_bfloat16 * __restrict__ src, const float * __restrict__ bias, -+ nv_bfloat16 * __restrict__ dst, int64_t rows) { -+#if defined(__CUDA_ARCH__) && __CUDA_ARCH__ >= GGML_CUDA_CC_AMPERE -+ const int64_t row = int64_t(blockIdx.x) * blockDim.x + threadIdx.x; -+ if (row >= rows) { -+ return; -+ } -+ const int64_t i = int64_t(blockIdx.y) * rows + row; -+ const float x = __bfloat162float(src[i]) + bias[row]; -+ dst[i] = __float2bfloat16(ggml_cuda_op_silu_single(x)); -+#else -+ GGML_UNUSED_VARS(src, bias, dst, rows); -+ NO_DEVICE_CODE; -+#endif -+} -+ -+static void launch_bf16_add_row_bias_silu_to_bf16( -+ const nv_bfloat16 * src, const float * bias, nv_bfloat16 * dst, -+ int64_t rows, int64_t cols, cudaStream_t stream) { -+ constexpr int block_size = 256; -+ GGML_ASSERT(cols <= 65535); -+ const dim3 block(block_size, 1, 1); -+ const dim3 grid((rows + block_size - 1) / block_size, cols, 1); -+ bf16_add_row_bias_silu_to_bf16<<>>(src, bias, dst, rows); -+} -+ - template --static void ggml_cuda_mul_mat_batched_cublas_impl(ggml_backend_cuda_context & ctx, const ggml_tensor * src0, const ggml_tensor * src1, ggml_tensor * dst) { -+static void ggml_cuda_mul_mat_batched_cublas_impl( -+ ggml_backend_cuda_context & ctx, const ggml_tensor * src0, -+ const ggml_tensor * src1, ggml_tensor * dst, -+ const ggml_tensor * row_bias = nullptr, -+ ggml_tensor * bf16_silu_dst = nullptr) { - using traits = batched_mul_mat_traits; - using cuda_t = typename traits::cuda_type; - -@@ -2296,7 +2353,27 @@ static void ggml_cuda_mul_mat_batched_cublas_impl(ggml_backend_cuda_context & ct - const int64_t r2 = ne12/ne02; - const int64_t r3 = ne13/ne03; - -- if (r2 == 1 && r3 == 1 && is_src0_cont_2 && is_src1_cont_2) { -+ // A single shared weight matrix broadcast over contiguous outer activation -+ // dimensions is one large GEMM, not a collection of independent skinny -+ // GEMMs. [K,T,B,D] and [K,T*B*D] have identical storage order; likewise -+ // for the [M,T,B,D] output. Flattening only the cuBLAS geometry therefore -+ // preserves the graph-visible tensor layout while improving weight reuse -+ // and tensor-core occupancy and avoiding pointer-array setup. -+ const bool src1_outer_contiguous = -+ src1->type != src0_type || ggml_is_contiguous(src1); -+ const bool flatten_shared_weight = -+ ne02 == 1 && ne03 == 1 && ne12 * ne13 > 1 && is_src0_cont_2 && -+ is_src1_cont_2 && src1_outer_contiguous; -+ if (flatten_shared_weight) { -+ CUBLAS_CHECK( -+ cublasGemmEx(ctx.cublas_handle(), CUBLAS_OP_T, CUBLAS_OP_N, -+ ne01, ne11*ne12*ne13, ne10, -+ alpha, src0_ptr, cu_data_type_a, nb01/nb00, -+ src1_ptr, cu_data_type_b, s11, -+ beta, dst_t, cu_data_type, ne0, -+ cu_compute_type, -+ CUBLAS_GEMM_DEFAULT_TENSOR_OP)); -+ } else if (r2 == 1 && r3 == 1 && is_src0_cont_2 && is_src1_cont_2) { - // with a [0, 2, 1, 3] perm. and ne02==1 the matrix strides need to be determined from dim 3: - const int64_t sma = ne02 == 1 ? nb03/nb00 : nb02/nb00; - const int64_t smb = ne12 == 1 ? s13 : s12; -@@ -2355,8 +2432,31 @@ static void ggml_cuda_mul_mat_batched_cublas_impl(ggml_backend_cuda_context & ct - - // Convert output back to F32 if needed - if (dst->op_params[0] == GGML_PREC_DEFAULT && cu_data_type != CUDA_R_32F) { -- const to_fp32_cuda_t to_fp32_cuda = ggml_get_to_fp32_cuda(traits::ggml_type_val); -- to_fp32_cuda(dst_temp.get(), dst_ddf, ne_dst, main_stream); -+ if (row_bias != nullptr) { -+ if constexpr (src0_type == GGML_TYPE_BF16) { -+ GGML_ASSERT(row_bias->type == GGML_TYPE_F32); -+ GGML_ASSERT(ggml_is_contiguous(row_bias)); -+ GGML_ASSERT(ggml_nelements(row_bias) == ne0); -+ if (bf16_silu_dst != nullptr) { -+ GGML_ASSERT(bf16_silu_dst->type == GGML_TYPE_BF16); -+ GGML_ASSERT(ggml_is_contiguous(bf16_silu_dst)); -+ GGML_ASSERT(ggml_are_same_shape(dst, bf16_silu_dst)); -+ launch_bf16_add_row_bias_silu_to_bf16( -+ dst_temp.get(), static_cast(row_bias->data), -+ static_cast(bf16_silu_dst->data), -+ ne0, ne_dst / ne0, main_stream); -+ } else { -+ launch_bf16_to_f32_add_row_bias( -+ dst_temp.get(), static_cast(row_bias->data), dst_ddf, -+ ne0, ne_dst / ne0, main_stream); -+ } -+ } else { -+ GGML_ABORT("row-bias conversion fusion is BF16-only"); -+ } -+ } else { -+ const to_fp32_cuda_t to_fp32_cuda = ggml_get_to_fp32_cuda(traits::ggml_type_val); -+ to_fp32_cuda(dst_temp.get(), dst_ddf, ne_dst, main_stream); -+ } - } - } - -@@ -2378,6 +2478,23 @@ static void ggml_cuda_mul_mat_batched_cublas(ggml_backend_cuda_context & ctx, co - } - } - -+static void ggml_cuda_mul_mat_bf16_row_bias( -+ ggml_backend_cuda_context & ctx, const ggml_tensor * src0, -+ const ggml_tensor * src1, const ggml_tensor * row_bias, ggml_tensor * dst) { -+ GGML_ASSERT(src0->type == GGML_TYPE_BF16); -+ ggml_cuda_mul_mat_batched_cublas_impl( -+ ctx, src0, src1, dst, row_bias); -+} -+ -+static void ggml_cuda_mul_mat_bf16_row_bias_silu( -+ ggml_backend_cuda_context & ctx, const ggml_tensor * src0, -+ const ggml_tensor * src1, const ggml_tensor * row_bias, -+ ggml_tensor * dst, ggml_tensor * bf16_silu_dst) { -+ GGML_ASSERT(src0->type == GGML_TYPE_BF16); -+ ggml_cuda_mul_mat_batched_cublas_impl( -+ ctx, src0, src1, dst, row_bias, bf16_silu_dst); -+} -+ - static bool ggml_cuda_should_fuse_mul_mat(const ggml_tensor * ffn_up, - const ggml_tensor * ffn_gate, - const ggml_tensor * glu, -@@ -3998,6 +4115,9 @@ static int ggml_cuda_try_fuse(ggml_backend_cuda_context * cuda_ctx, ggml_cgraph - } - - ggml_tensor * node = cgraph->nodes[i]; -+ const int cc = ggml_cuda_info().devices[cuda_ctx->device].cc; -+ const bool native_bf16 = -+ GGML_CUDA_CC_IS_NVIDIA(cc) && cc >= GGML_CUDA_CC_AMPERE; - - if (int n_fused = ggml_cuda_try_fuse_half_snake(cuda_ctx, cgraph, i)) { - return n_fused; -@@ -4126,6 +4246,75 @@ static int ggml_cuda_try_fuse(ggml_backend_cuda_context * cuda_ctx, ggml_cgraph - } - } - -+ // Shared BF16 projection + broadcast F32 row bias + SiLU + BF16 cast. -+ // Feed-forward linear2 consumes the rounded BF16 activation directly, so -+ // avoid materializing the intermediate F32 biased activation entirely. -+ if (ggml_can_fuse(cgraph, i, -+ { GGML_OP_MUL_MAT, GGML_OP_ADD, GGML_OP_UNARY, GGML_OP_CPY })) { -+ ggml_tensor * mm_node = cgraph->nodes[i]; -+ ggml_tensor * bias_node = cgraph->nodes[i + 1]; -+ ggml_tensor * silu_node = cgraph->nodes[i + 2]; -+ ggml_tensor * cast_node = cgraph->nodes[i + 3]; -+ const ggml_tensor * row_bias = -+ bias_node->src[0] == mm_node ? bias_node->src[1] : -+ bias_node->src[1] == mm_node ? bias_node->src[0] : nullptr; -+ const ggml_tensor * weight = mm_node->src[0]; -+ const ggml_tensor * input = mm_node->src[1]; -+ const int64_t cols = mm_node->ne[1] * mm_node->ne[2] * mm_node->ne[3]; -+ const bool eligible = -+ native_bf16 && row_bias != nullptr && silu_node->src[0] == bias_node && -+ cast_node->src[0] == silu_node && -+ ggml_get_unary_op(silu_node) == GGML_UNARY_OP_SILU && -+ weight->type == GGML_TYPE_BF16 && -+ (input->type == GGML_TYPE_F32 || input->type == GGML_TYPE_BF16) && -+ mm_node->type == GGML_TYPE_F32 && bias_node->type == GGML_TYPE_F32 && -+ silu_node->type == GGML_TYPE_F32 && cast_node->type == GGML_TYPE_BF16 && -+ row_bias->type == GGML_TYPE_F32 && ggml_is_contiguous(row_bias) && -+ ggml_is_contiguous(bias_node) && ggml_is_contiguous(cast_node) && -+ ggml_nelements(row_bias) == mm_node->ne[0] && -+ ggml_are_same_shape(mm_node, cast_node) && -+ weight->ne[2] == 1 && weight->ne[3] == 1 && -+ input->ne[2] * input->ne[3] > 1 && cols <= 65535 && -+ !ggml_is_transposed(weight) && !ggml_is_transposed(input) && -+ !ggml_backend_buft_is_cuda_split(weight->buffer->buft); -+ if (eligible) { -+ ggml_cuda_mul_mat_bf16_row_bias_silu( -+ *cuda_ctx, weight, input, row_bias, bias_node, cast_node); -+ return 3; -+ } -+ } -+ -+ // Shared BF16 projection + broadcast F32 row bias. The batched-cuBLAS -+ // path already materializes BF16 GEMM output before converting it to the -+ // graph's F32 activation. Add the bias during that conversion rather than -+ // launching another full-tensor read/modify/write pass. -+ if (ggml_can_fuse(cgraph, i, { GGML_OP_MUL_MAT, GGML_OP_ADD })) { -+ ggml_tensor * mm_node = cgraph->nodes[i]; -+ ggml_tensor * bias_node = cgraph->nodes[i + 1]; -+ const ggml_tensor * row_bias = -+ bias_node->src[0] == mm_node ? bias_node->src[1] : -+ bias_node->src[1] == mm_node ? bias_node->src[0] : nullptr; -+ const ggml_tensor * weight = mm_node->src[0]; -+ const ggml_tensor * input = mm_node->src[1]; -+ const int64_t cols = mm_node->ne[1] * mm_node->ne[2] * mm_node->ne[3]; -+ const bool eligible = -+ native_bf16 && row_bias != nullptr && weight->type == GGML_TYPE_BF16 && -+ (input->type == GGML_TYPE_F32 || input->type == GGML_TYPE_BF16) && -+ mm_node->type == GGML_TYPE_F32 && -+ bias_node->type == GGML_TYPE_F32 && row_bias->type == GGML_TYPE_F32 && -+ ggml_is_contiguous(row_bias) && ggml_is_contiguous(bias_node) && -+ ggml_nelements(row_bias) == mm_node->ne[0] && -+ weight->ne[2] == 1 && weight->ne[3] == 1 && -+ input->ne[2] * input->ne[3] > 1 && cols <= 65535 && -+ !ggml_is_transposed(weight) && !ggml_is_transposed(input) && -+ !ggml_backend_buft_is_cuda_split(weight->buffer->buft); -+ if (eligible) { -+ ggml_cuda_mul_mat_bf16_row_bias( -+ *cuda_ctx, weight, input, row_bias, bias_node); -+ return 1; -+ } -+ } -+ - // multi-(add or mul) - if (node->op == GGML_OP_ADD || node->op == GGML_OP_MUL) { - int n_fuse = 0; -@@ -4292,6 +4481,25 @@ static int ggml_cuda_try_fuse(ggml_backend_cuda_context * cuda_ctx, ggml_cgraph - fused_mul_mat_vec = false; - fused_node_count = 0; - -+ // A BF16 projection immediately following SiLU would otherwise launch an -+ // F32 SiLU kernel and then a separate F32-to-BF16 input conversion. Emit -+ // the same rounded BF16 activation directly in one pass. -+ if (ggml_can_fuse(cgraph, i, { GGML_OP_UNARY, GGML_OP_CPY })) { -+ const ggml_tensor * silu_node = cgraph->nodes[i]; -+ ggml_tensor * cast_node = cgraph->nodes[i + 1]; -+ const bool eligible = -+ native_bf16 && ggml_get_unary_op(silu_node) == GGML_UNARY_OP_SILU && -+ cast_node->src[0] == silu_node && -+ silu_node->src[0]->type == GGML_TYPE_F32 && -+ silu_node->type == GGML_TYPE_F32 && cast_node->type == GGML_TYPE_BF16 && -+ ggml_are_same_shape(silu_node->src[0], cast_node) && -+ ggml_is_contiguous(silu_node->src[0]) && ggml_is_contiguous(cast_node); -+ if (eligible) { -+ ggml_cuda_op_silu_f32_to_bf16(*cuda_ctx, silu_node, cast_node); -+ return 1; -+ } -+ } -+ - // Q8 narrow projection + SiLU. NeMo-Speech.cpp's Conformer FF1 - // layers are bias-free, so this is their actual hot graph sequence. - if (ggml_cuda_q8_narrow_epilogue_enabled() && -diff --git a/src/ggml-cuda/unary.cu b/src/ggml-cuda/unary.cu -index 2aeba26f..36c608e1 100644 ---- a/src/ggml-cuda/unary.cu -+++ b/src/ggml-cuda/unary.cu -@@ -183,6 +183,37 @@ void ggml_cuda_op_silu(ggml_backend_cuda_context & ctx, ggml_tensor * dst) { - ggml_cuda_op_unary(ctx, dst); - } - -+static __global__ void silu_f32_to_bf16( -+ const float * __restrict__ x, nv_bfloat16 * __restrict__ dst, -+ const int64_t nelements) { -+#if defined(__CUDA_ARCH__) && __CUDA_ARCH__ >= GGML_CUDA_CC_AMPERE -+ const int64_t i = (int64_t) blockIdx.x * blockDim.x + threadIdx.x; -+ if (i < nelements) { -+ dst[i] = __float2bfloat16(op_silu(x[i])); -+ } -+#else -+ GGML_UNUSED_VARS(x, dst, nelements); -+ NO_DEVICE_CODE; -+#endif -+} -+ -+void ggml_cuda_op_silu_f32_to_bf16( -+ ggml_backend_cuda_context & ctx, const ggml_tensor * silu_node, -+ ggml_tensor * dst) { -+ const ggml_tensor * src = silu_node->src[0]; -+ GGML_ASSERT(src->type == GGML_TYPE_F32); -+ GGML_ASSERT(silu_node->type == GGML_TYPE_F32); -+ GGML_ASSERT(dst->type == GGML_TYPE_BF16); -+ GGML_ASSERT(ggml_are_same_shape(src, dst)); -+ GGML_ASSERT(ggml_is_contiguous(src)); -+ GGML_ASSERT(ggml_is_contiguous(dst)); -+ -+ const int64_t nelements = ggml_nelements(src); -+ const int64_t num_blocks = (nelements + CUDA_SILU_BLOCK_SIZE - 1) / CUDA_SILU_BLOCK_SIZE; -+ silu_f32_to_bf16<<>>( -+ (const float *) src->data, (nv_bfloat16 *) dst->data, nelements); -+} -+ - void ggml_cuda_op_tanh(ggml_backend_cuda_context & ctx, ggml_tensor * dst) { - ggml_cuda_op_unary(ctx, dst); - } -diff --git a/src/ggml-cuda/unary.cuh b/src/ggml-cuda/unary.cuh -index 81ed873e..592a0dec 100644 ---- a/src/ggml-cuda/unary.cuh -+++ b/src/ggml-cuda/unary.cuh -@@ -31,6 +31,9 @@ void ggml_cuda_op_gelu(ggml_backend_cuda_context & ctx, ggml_tensor * dst); - - void ggml_cuda_op_silu(ggml_backend_cuda_context & ctx, ggml_tensor * dst); - -+void ggml_cuda_op_silu_f32_to_bf16( -+ ggml_backend_cuda_context & ctx, const ggml_tensor * silu_node, ggml_tensor * dst); -+ - void ggml_cuda_op_silu_back(ggml_backend_cuda_context & ctx, ggml_tensor * dst); - - void ggml_cuda_op_gelu_erf(ggml_backend_cuda_context & ctx, ggml_tensor * dst); diff --git a/ggml-patches/0009-fastconformer-cuda-fusions.patch b/ggml-patches/0009-fastconformer-cuda-fusions.patch deleted file mode 100644 index 204d203..0000000 --- a/ggml-patches/0009-fastconformer-cuda-fusions.patch +++ /dev/null @@ -1,330 +0,0 @@ -diff --git a/include/ggml.h b/include/ggml.h -index a4727087..fb823571 100644 ---- a/include/ggml.h -+++ b/include/ggml.h -@@ -622,6 +622,7 @@ extern "C" { - GGML_GLU_OP_SWIGLU_OAI, - GGML_GLU_OP_GEGLU_ERF, - GGML_GLU_OP_GEGLU_QUICK, -+ GGML_GLU_OP_SIGMOID, - - GGML_GLU_OP_COUNT, - }; -diff --git a/src/ggml-cuda/common.cuh b/src/ggml-cuda/common.cuh -index 56fe1bb7..01b239f1 100644 ---- a/src/ggml-cuda/common.cuh -+++ b/src/ggml-cuda/common.cuh -@@ -51,6 +51,7 @@ - #define GGML_CUDA_CC_TURING 750 - #define GGML_CUDA_CC_AMPERE 800 - #define GGML_CUDA_CC_ADA_LOVELACE 890 -+#define GGML_CUDA_CC_HOPPER 900 - // Jetson Thor is Blackwell SM110. The hand-written FP4 block-scale PTX below is currently SM120-only, but SM110 - // should still be detected so ggml can favor CUDA library kernels that may use Thor tcgen05 tensor cores. - #define GGML_CUDA_CC_THOR 1100 -diff --git a/src/ggml-cuda/ggml-cuda.cu b/src/ggml-cuda/ggml-cuda.cu -index 1a1b58fb..41605f45 100644 ---- a/src/ggml-cuda/ggml-cuda.cu -+++ b/src/ggml-cuda/ggml-cuda.cu -@@ -3063,6 +3063,9 @@ static bool ggml_cuda_compute_forward(ggml_backend_cuda_context & ctx, struct gg - case GGML_GLU_OP_GEGLU_QUICK: - ggml_cuda_op_geglu_quick(ctx, dst); - break; -+ case GGML_GLU_OP_SIGMOID: -+ ggml_cuda_op_sigmoid_glu(ctx, dst); -+ break; - default: - return false; - } -@@ -4284,6 +4287,31 @@ static int ggml_cuda_try_fuse(ggml_backend_cuda_context * cuda_ctx, ggml_cgraph - } - } - -+ // Macaron feed-forward residual: residual + scale * ff. Both tensors are -+ // contiguous F32 and have identical shapes, so materializing the scaled -+ // activation only creates an avoidable full-tensor round trip. -+ if (ggml_can_fuse(cgraph, i, { GGML_OP_SCALE, GGML_OP_ADD })) { -+ const ggml_tensor * scale_node = cgraph->nodes[i]; -+ ggml_tensor * add_node = cgraph->nodes[i + 1]; -+ const ggml_tensor * residual = -+ add_node->src[0] == scale_node ? add_node->src[1] : -+ add_node->src[1] == scale_node ? add_node->src[0] : nullptr; -+ -+ const float bias = ggml_get_op_params_f32(scale_node, 1); -+ const bool eligible = -+ residual != nullptr && bias == 0.0f && -+ scale_node->src[0]->type == GGML_TYPE_F32 && -+ residual->type == GGML_TYPE_F32 && add_node->type == GGML_TYPE_F32 && -+ ggml_are_same_shape(scale_node->src[0], residual) && -+ ggml_are_same_shape(scale_node->src[0], add_node) && -+ ggml_is_contiguous(scale_node->src[0]) && -+ ggml_is_contiguous(residual) && ggml_is_contiguous(add_node); -+ if (eligible) { -+ ggml_cuda_op_scale_add(*cuda_ctx, scale_node, residual, add_node); -+ return 1; -+ } -+ } -+ - // Shared BF16 projection + broadcast F32 row bias. The batched-cuBLAS - // path already materializes BF16 GEMM output before converting it to the - // graph's F32 activation. Add the bias during that conversion rather than -@@ -4628,6 +4656,35 @@ static int ggml_cuda_try_fuse(ggml_backend_cuda_context * cuda_ctx, ggml_cgraph - return fused_node_count - 1; - } - -+ // LayerNorm affine followed by a BF16 projection. Store the affine result -+ // directly in the projection's input precision instead of writing F32 and -+ // launching a second full-tensor conversion. -+ if (ggml_can_fuse(cgraph, i, -+ { GGML_OP_NORM, GGML_OP_MUL, GGML_OP_ADD, GGML_OP_CPY })) { -+ ggml_tensor * norm = cgraph->nodes[i]; -+ ggml_tensor * mul = cgraph->nodes[i + 1]; -+ ggml_tensor * add = cgraph->nodes[i + 2]; -+ ggml_tensor * cast = cgraph->nodes[i + 3]; -+ const ggml_tensor * gamma = -+ mul->src[0] == norm ? mul->src[1] : mul->src[0]; -+ const ggml_tensor * beta = -+ add->src[0] == mul ? add->src[1] : add->src[0]; -+ const bool eligible = -+ native_bf16 && cast->src[0] == add && norm->src[0]->type == GGML_TYPE_F32 && -+ norm->type == GGML_TYPE_F32 && mul->type == GGML_TYPE_F32 && -+ add->type == GGML_TYPE_F32 && cast->type == GGML_TYPE_BF16 && -+ gamma->type == GGML_TYPE_F32 && beta->type == GGML_TYPE_F32 && -+ ggml_nelements(gamma) == norm->ne[0] && -+ ggml_nelements(beta) == norm->ne[0] && -+ ggml_is_contiguous(gamma) && ggml_is_contiguous(beta) && -+ ggml_is_contiguous(mul) && ggml_is_contiguous(add) && -+ ggml_is_contiguous(cast) && ggml_are_same_shape(norm, cast); -+ if (eligible) { -+ ggml_cuda_op_norm_fused(*cuda_ctx, norm, mul, add, cast); -+ return 3; -+ } -+ } -+ - if (ggml_cuda_can_fuse(cgraph, i, { GGML_OP_RMS_NORM, GGML_OP_MUL, GGML_OP_ADD }, {})) { - ggml_cuda_op_rms_norm_fused_add(*cuda_ctx, node, cgraph->nodes[i + 1], cgraph->nodes[i + 2]); - return 2; -@@ -5551,6 +5608,7 @@ static bool ggml_backend_cuda_device_supports_op(ggml_backend_dev_t dev, const g - case GGML_GLU_OP_SWIGLU_OAI: - case GGML_GLU_OP_GEGLU_ERF: - case GGML_GLU_OP_GEGLU_QUICK: -+ case GGML_GLU_OP_SIGMOID: - return ggml_is_contiguous_1(op->src[0]); - default: - return false; -diff --git a/src/ggml-cuda/norm.cu b/src/ggml-cuda/norm.cu -index d77e1a61..555c9514 100644 ---- a/src/ggml-cuda/norm.cu -+++ b/src/ggml-cuda/norm.cu -@@ -42,9 +42,9 @@ static __global__ void norm_f32( - // to ne0-length vectors broadcast over rows/channels/samples (the standard - // LayerNorm affine), enforced by ggml_cuda_can_fuse — this keeps indexing - // trivial instead of replicating rms_norm's general broadcast machinery. --template -+template - static __global__ void norm_mul_add_f32( -- const float * x, const float * mul, const float * add, float * dst, const int ncols, -+ const float * x, const float * mul, const float * add, T * dst, const int ncols, - const int64_t stride_row, const int64_t stride_channel, const int64_t stride_sample, const float eps) { - const int nrows = gridDim.x; - const int nchannels = gridDim.y; -@@ -77,7 +77,7 @@ static __global__ void norm_mul_add_f32( - if constexpr (do_add) { - v += add[col]; - } -- dst[col] = v; -+ dst[col] = (T) v; - } - } - -@@ -327,27 +327,41 @@ static void norm_f32_cuda( - } - } - --static void norm_mul_add_f32_cuda( -- const float * x, const float * mul, const float * add, float * dst, const int ncols, const int nrows, -+template -+static void norm_mul_add_cuda( -+ const float * x, const float * mul, const float * add, T * dst, const int ncols, const int nrows, - const int nchannels, const int nsamples, const int64_t stride_row, const int64_t stride_channel, - const int64_t stride_sample, const float eps, cudaStream_t stream) { - const dim3 blocks_num(nrows, nchannels, nsamples); -+ const int cc = ggml_cuda_info().devices[ggml_cuda_get_device()].cc; - if (ncols < 1024) { - const dim3 block_dims(WARP_SIZE, 1, 1); - if (add) { -- norm_mul_add_f32<<>>( -+ norm_mul_add_f32<<>>( - x, mul, add, dst, ncols, stride_row, stride_channel, stride_sample, eps); - } else { -- norm_mul_add_f32<<>>( -+ norm_mul_add_f32<<>>( -+ x, mul, add, dst, ncols, stride_row, stride_channel, stride_sample, eps); -+ } -+ } else if (ncols == 1024 && GGML_CUDA_CC_IS_NVIDIA(cc) && cc >= GGML_CUDA_CC_HOPPER) { -+ // Hopper and newer: four elements per thread keeps high occupancy -+ // while reducing the affine LayerNorm reduction from 32 warps/block -+ // to 8. Retain the established 1024-thread path on older devices. -+ const dim3 block_dims(256, 1, 1); -+ if (add) { -+ norm_mul_add_f32<256, true, T><<>>( -+ x, mul, add, dst, ncols, stride_row, stride_channel, stride_sample, eps); -+ } else { -+ norm_mul_add_f32<256, false, T><<>>( - x, mul, add, dst, ncols, stride_row, stride_channel, stride_sample, eps); - } - } else { - const dim3 block_dims(1024, 1, 1); - if (add) { -- norm_mul_add_f32<1024, true><<>>( -+ norm_mul_add_f32<1024, true, T><<>>( - x, mul, add, dst, ncols, stride_row, stride_channel, stride_sample, eps); - } else { -- norm_mul_add_f32<1024, false><<>>( -+ norm_mul_add_f32<1024, false, T><<>>( - x, mul, add, dst, ncols, stride_row, stride_channel, stride_sample, eps); - } - } -@@ -505,7 +519,7 @@ void ggml_cuda_op_norm(ggml_backend_cuda_context & ctx, ggml_tensor * dst) { - // or add_tensor), mirroring ggml_cuda_op_rms_norm_fused(_add). Eligibility - // (row-vector operands, F32, contiguity) is enforced in ggml_cuda_can_fuse. - void ggml_cuda_op_norm_fused(ggml_backend_cuda_context & ctx, ggml_tensor * dst, ggml_tensor * mul_tensor, -- ggml_tensor * add_tensor) { -+ ggml_tensor * add_tensor, ggml_tensor * bf16_dst) { - const ggml_tensor * norm_src = (ggml_tensor *) dst->src[0]; - float eps = 0.0f; - memcpy(&eps, dst->op_params, sizeof(float)); -@@ -549,8 +563,18 @@ void ggml_cuda_op_norm_fused(ggml_backend_cuda_context & ctx, ggml_tensor * dst, - const int64_t s02 = norm_src->nb[2] / ts0; - const int64_t s03 = norm_src->nb[3] / ts0; - -- norm_mul_add_f32_cuda( -- src0_d, mul_d, add_d, dst_d, ne00, ne01, ne02, ne03, s01, s02, s03, eps, ctx.stream()); -+ if (bf16_dst != nullptr) { -+ GGML_ASSERT(bf16_dst->type == GGML_TYPE_BF16); -+ GGML_ASSERT(ggml_is_contiguous(bf16_dst)); -+ GGML_ASSERT(ggml_are_same_shape(dst, bf16_dst)); -+ norm_mul_add_cuda( -+ src0_d, mul_d, add_d, (nv_bfloat16 *) bf16_dst->data, -+ ne00, ne01, ne02, ne03, s01, s02, s03, eps, ctx.stream()); -+ } else { -+ norm_mul_add_cuda( -+ src0_d, mul_d, add_d, dst_d, -+ ne00, ne01, ne02, ne03, s01, s02, s03, eps, ctx.stream()); -+ } - } - - void ggml_cuda_op_group_norm(ggml_backend_cuda_context & ctx, ggml_tensor * dst) { -diff --git a/src/ggml-cuda/norm.cuh b/src/ggml-cuda/norm.cuh -index 20ecaa2d..8f4287c3 100644 ---- a/src/ggml-cuda/norm.cuh -+++ b/src/ggml-cuda/norm.cuh -@@ -5,7 +5,7 @@ void ggml_cuda_op_norm(ggml_backend_cuda_context & ctx, ggml_tensor * dst); - // Fused LayerNorm + row-vector mul (gamma) + optional row-vector add (beta). - // add_tensor may be nullptr (norm+mul only). - void ggml_cuda_op_norm_fused(ggml_backend_cuda_context & ctx, ggml_tensor * dst, ggml_tensor * mul_tensor, -- ggml_tensor * add_tensor); -+ ggml_tensor * add_tensor, ggml_tensor * bf16_dst = nullptr); - - void ggml_cuda_op_group_norm(ggml_backend_cuda_context & ctx, ggml_tensor * dst); - -diff --git a/src/ggml-cuda/scale.cu b/src/ggml-cuda/scale.cu -index 0ddeff6a..bdd1eb4f 100644 ---- a/src/ggml-cuda/scale.cu -+++ b/src/ggml-cuda/scale.cu -@@ -32,3 +32,39 @@ void ggml_cuda_op_scale(ggml_backend_cuda_context & ctx, ggml_tensor * dst) { - - scale_f32_cuda(src0_d, dst_d, scale, bias, ggml_nelements(src0), stream); - } -+ -+static __global__ void scale_add_f32( -+ const float * __restrict__ x, const float * __restrict__ residual, -+ float * __restrict__ dst, const float scale, const int64_t nelements) { -+ const int64_t tid = (int64_t) blockIdx.x * blockDim.x + threadIdx.x; -+ const int64_t stride = (int64_t) blockDim.x * gridDim.x; -+ -+ for (int64_t i = tid; i < nelements; i += stride) { -+ dst[i] = residual[i] + scale * x[i]; -+ } -+} -+ -+void ggml_cuda_op_scale_add( -+ ggml_backend_cuda_context & ctx, const ggml_tensor * scale_node, -+ const ggml_tensor * residual, ggml_tensor * dst) { -+ const ggml_tensor * x = scale_node->src[0]; -+ -+ GGML_ASSERT(x->type == GGML_TYPE_F32); -+ GGML_ASSERT(residual->type == GGML_TYPE_F32); -+ GGML_ASSERT(dst->type == GGML_TYPE_F32); -+ GGML_ASSERT(ggml_is_contiguous(x)); -+ GGML_ASSERT(ggml_is_contiguous(residual)); -+ GGML_ASSERT(ggml_is_contiguous(dst)); -+ GGML_ASSERT(ggml_are_same_shape(x, residual)); -+ GGML_ASSERT(ggml_are_same_shape(x, dst)); -+ -+ const float scale = ggml_get_op_params_f32(scale_node, 0); -+ const float bias = ggml_get_op_params_f32(scale_node, 1); -+ GGML_ASSERT(bias == 0.0f); -+ -+ const int64_t nelements = ggml_nelements(x); -+ const int64_t num_blocks = (nelements + CUDA_SCALE_BLOCK_SIZE - 1) / CUDA_SCALE_BLOCK_SIZE; -+ scale_add_f32<<>>( -+ (const float *) x->data, (const float *) residual->data, -+ (float *) dst->data, scale, nelements); -+} -diff --git a/src/ggml-cuda/scale.cuh b/src/ggml-cuda/scale.cuh -index 8ff75c82..16fe10ab 100644 ---- a/src/ggml-cuda/scale.cuh -+++ b/src/ggml-cuda/scale.cuh -@@ -3,3 +3,7 @@ - #define CUDA_SCALE_BLOCK_SIZE 256 - - void ggml_cuda_op_scale(ggml_backend_cuda_context & ctx, ggml_tensor * dst); -+ -+void ggml_cuda_op_scale_add( -+ ggml_backend_cuda_context & ctx, const ggml_tensor * scale_node, -+ const ggml_tensor * residual, ggml_tensor * dst); -diff --git a/src/ggml-cuda/unary.cu b/src/ggml-cuda/unary.cu -index 36c608e1..4c68411f 100644 ---- a/src/ggml-cuda/unary.cu -+++ b/src/ggml-cuda/unary.cu -@@ -374,6 +374,10 @@ void ggml_cuda_op_swiglu(ggml_backend_cuda_context & ctx, ggml_tensor * dst) { - ggml_cuda_op_unary_gated(ctx, dst); - } - -+void ggml_cuda_op_sigmoid_glu(ggml_backend_cuda_context & ctx, ggml_tensor * dst) { -+ ggml_cuda_op_unary_gated(ctx, dst); -+} -+ - void ggml_cuda_op_geglu_erf(ggml_backend_cuda_context & ctx, ggml_tensor * dst) { - ggml_cuda_op_unary_gated(ctx, dst); - } -diff --git a/src/ggml-cuda/unary.cuh b/src/ggml-cuda/unary.cuh -index 592a0dec..53073275 100644 ---- a/src/ggml-cuda/unary.cuh -+++ b/src/ggml-cuda/unary.cuh -@@ -84,6 +84,8 @@ void ggml_cuda_op_geglu(ggml_backend_cuda_context & ctx, ggml_tensor * dst); - - void ggml_cuda_op_swiglu(ggml_backend_cuda_context & ctx, ggml_tensor * dst); - -+void ggml_cuda_op_sigmoid_glu(ggml_backend_cuda_context & ctx, ggml_tensor * dst); -+ - void ggml_cuda_op_swiglu_oai(ggml_backend_cuda_context & ctx, ggml_tensor * dst); - - void ggml_cuda_op_geglu_erf(ggml_backend_cuda_context & ctx, ggml_tensor * dst); -diff --git a/src/ggml.c b/src/ggml.c -index 88b39537..80b5802f 100644 ---- a/src/ggml.c -+++ b/src/ggml.c -@@ -1232,9 +1232,10 @@ static const char * GGML_GLU_OP_NAME[GGML_GLU_OP_COUNT] = { - "SWIGLU_OAI", - "GEGLU_ERF", - "GEGLU_QUICK", -+ "SIGMOID_GLU", - }; - --static_assert(GGML_GLU_OP_COUNT == 6, "GGML_GLU_OP_COUNT != 6"); -+static_assert(GGML_GLU_OP_COUNT == 7, "GGML_GLU_OP_COUNT != 7"); - - - static_assert(sizeof(struct ggml_object)%GGML_MEM_ALIGN == 0, "ggml_object size must be a multiple of GGML_MEM_ALIGN"); diff --git a/ggml-patches/0012-cuda-streaming-cache-copies.patch b/ggml-patches/0012-cuda-streaming-cache-copies.patch deleted file mode 100644 index 453a40b..0000000 --- a/ggml-patches/0012-cuda-streaming-cache-copies.patch +++ /dev/null @@ -1,151 +0,0 @@ -diff --git a/src/ggml-cuda/cpy.cu b/src/ggml-cuda/cpy.cu -index d208acf2..79861b4d 100644 ---- a/src/ggml-cuda/cpy.cu -+++ b/src/ggml-cuda/cpy.cu -@@ -407,7 +407,24 @@ void ggml_cuda_cpy(ggml_backend_cuda_context & ctx, const ggml_tensor * src0, gg - const bool can_be_transposed = nb01 == (int64_t)ggml_element_size(src0) && - src0->ne[3] == 1 && nb02 == ne00 * ne01 * (int64_t)ggml_element_size(src0); - -- if (src0->type == src1->type && contiguous_srcs) { -+ // Cache-aware ASR keeps the last time rows of a [C,T,B] tensor. The view -+ // is contiguous within each batch plane but has the original T stride -+ // between planes; materializing it with cpy_scalar performs six 64-bit -+ // div/mod operations per float. Express this common layout as one pitched -+ // device copy instead (for C512 K/V this is 512 x ~224 KiB rows). -+ const size_t elem_size = ggml_element_size(src0); -+ const size_t plane_bytes = (size_t) ne00 * ne01 * elem_size; -+ const bool pitched_f32_to_contiguous = -+ src0->type == GGML_TYPE_F32 && src1->type == GGML_TYPE_F32 && -+ src0->ne[3] == 1 && ne == ne00 * ne01 * ne02 && -+ nb00 == (int64_t) elem_size && nb01 == ne00 * (int64_t) elem_size && -+ nb02 >= (int64_t) plane_bytes && ggml_is_contiguous(src1); -+ -+ if (pitched_f32_to_contiguous && !ggml_is_contiguous(src0)) { -+ CUDA_CHECK(cudaMemcpy2DAsync( -+ src1_ddc, plane_bytes, src0_ddc, (size_t) nb02, -+ plane_bytes, (size_t) ne02, cudaMemcpyDeviceToDevice, main_stream)); -+ } else if (src0->type == src1->type && contiguous_srcs) { - GGML_ASSERT(ggml_nbytes(src0) == ggml_nbytes(src1)); - #if defined(GGML_USE_MUSA) && defined(GGML_MUSA_MUDNN_COPY) - if (src0->type == GGML_TYPE_F32 || src0->type == GGML_TYPE_F16) { -diff --git a/src/ggml-cuda/getrows.cu b/src/ggml-cuda/getrows.cu -index 36b840e8..c91ac35a 100644 ---- a/src/ggml-cuda/getrows.cu -+++ b/src/ggml-cuda/getrows.cu -@@ -70,6 +70,28 @@ static __global__ void k_get_rows_float( - } - } - -+// Fast path for large F32 cache rows. One CTA owns an output row and moves -+// float4 vectors from the indexed arena row. The generic kernel creates one -+// CTA per 256 scalar columns and reloads/recomputes the same row metadata in -+// every CTA; a 57,344-element ASR K/V row therefore used 224 CTAs. -+static __global__ void k_get_rows_contiguous_f32x4( -+ const float * __restrict__ src0, const int32_t * __restrict__ rows, -+ float * __restrict__ dst, const int row_elements, const size_t src_row_stride, -+ const size_t src_plane_stride, const size_t dst_plane_stride, -+ const size_t row_index_plane_stride) { -+ const int output_row = (int) blockIdx.x; -+ const int plane = (int) blockIdx.y; -+ const int source_row = rows[output_row + (size_t) plane * row_index_plane_stride]; -+ const float4 * src = (const float4 *) ( -+ src0 + (size_t) plane * src_plane_stride + (size_t) source_row * src_row_stride); -+ float4 * out = (float4 *) ( -+ dst + (size_t) plane * dst_plane_stride + (size_t) output_row * row_elements); -+ const int vectors = row_elements / 4; -+ for (int i = threadIdx.x; i < vectors; i += blockDim.x) { -+ out[i] = src[i]; -+ } -+} -+ - template - static __global__ void k_get_rows_back_float( - const grad_t * __restrict__ grad, const int32_t * __restrict__ rows, dst_t * __restrict__ dst, const int64_t ncols, const int64_t nrows_grad) { -@@ -264,6 +286,24 @@ void ggml_cuda_op_get_rows(ggml_backend_cuda_context & ctx, ggml_tensor * dst) { - GGML_ASSERT(src1->nb[0] == ggml_type_size(src1->type)); - GGML_ASSERT(dst->nb[0] == ggml_type_size(dst->type)); - -+ const bool contiguous_cache_rows = -+ src0->type == GGML_TYPE_F32 && dst->type == GGML_TYPE_F32 && -+ src1->type == GGML_TYPE_I32 && src0->ne[3] == 1 && -+ src1->ne[1] == src0->ne[2] && src1->ne[2] == 1 && src1->ne[3] == 1 && -+ dst->ne[2] == src0->ne[2] && dst->ne[3] == 1 && dst->ne[0] == src0->ne[0] && -+ dst->ne[1] == src1->ne[0] && ggml_is_contiguous(dst) && -+ src0->nb[0] == sizeof(float) && src0->nb[1] % sizeof(float4) == 0 && -+ src0->ne[0] >= 1024 && src0->ne[0] % 4 == 0; -+ if (contiguous_cache_rows) { -+ const dim3 grid((unsigned) src1->ne[0], (unsigned) src0->ne[2]); -+ k_get_rows_contiguous_f32x4<<>>( -+ (const float *) src0->data, (const int32_t *) src1->data, -+ (float *) dst->data, (int) src0->ne[0], src0->nb[1] / sizeof(float), -+ src0->nb[2] / sizeof(float), dst->nb[2] / sizeof(float), -+ src1->nb[1] / sizeof(int32_t)); -+ return; -+ } -+ - get_rows_cuda(src0->data, src0->type, (const int32_t *) src1->data, dst->data, dst->type, - ne00, nb01, nb02, nb03, ne10, ne11, ne12, nb10, nb11, nb12, nb1, nb2, nb3, stream); - } -diff --git a/src/ggml-cuda/set-rows.cu b/src/ggml-cuda/set-rows.cu -index 631de7e8..0c5e5abc 100644 ---- a/src/ggml-cuda/set-rows.cu -+++ b/src/ggml-cuda/set-rows.cu -@@ -170,6 +170,26 @@ static __global__ void k_set_rows(const src_t * __restrict__ src0, - GGML_UNUSED(ne13); - } - -+template -+static __global__ void k_set_rows_contiguous_f32x4( -+ const float * __restrict__ src0, const idx_t * __restrict__ rows, -+ float * __restrict__ dst, const int row_elements, const size_t dst_row_stride, -+ const size_t src_plane_stride, const size_t dst_plane_stride, -+ const size_t row_index_plane_stride) { -+ const int source_row = (int) blockIdx.x; -+ const int plane = (int) blockIdx.y; -+ const int64_t destination_row = -+ (int64_t) rows[source_row + (size_t) plane * row_index_plane_stride]; -+ const float4 * src = (const float4 *) ( -+ src0 + (size_t) plane * src_plane_stride + (size_t) source_row * row_elements); -+ float4 * out = (float4 *) ( -+ dst + (size_t) plane * dst_plane_stride + destination_row * dst_row_stride); -+ const int vectors = row_elements / 4; -+ for (int i = threadIdx.x; i < vectors; i += blockDim.x) { -+ out[i] = src[i]; -+ } -+} -+ - template - static void set_rows_cuda( - const src_t * src0_d, const idx_t * src1_d, dst_t * dst_d, -@@ -322,6 +342,31 @@ void ggml_cuda_op_set_rows(ggml_backend_cuda_context & ctx, ggml_tensor * dst) { - GGML_ASSERT(src0->type == GGML_TYPE_F32); - GGML_ASSERT(src1->type == GGML_TYPE_I64 || src1->type == GGML_TYPE_I32); - -+ const bool contiguous_cache_rows = -+ dst->type == GGML_TYPE_F32 && src0->ne[3] == 1 && -+ src1->ne[1] == src0->ne[2] && src1->ne[2] == 1 && src1->ne[3] == 1 && -+ dst->ne[2] == src0->ne[2] && dst->ne[3] == 1 && src0->ne[0] == dst->ne[0] && -+ src0->ne[1] == src1->ne[0] && ggml_is_contiguous(src0) && -+ dst->nb[0] == sizeof(float) && dst->nb[1] % sizeof(float4) == 0 && -+ src0->ne[0] >= 1024 && src0->ne[0] % 4 == 0; -+ if (contiguous_cache_rows) { -+ const dim3 grid((unsigned) src0->ne[1], (unsigned) src0->ne[2]); -+ if (src1->type == GGML_TYPE_I64) { -+ k_set_rows_contiguous_f32x4<<>>( -+ (const float *) src0->data, (const int64_t *) src1->data, -+ (float *) dst->data, (int) src0->ne[0], dst->nb[1] / sizeof(float), -+ src0->nb[2] / sizeof(float), dst->nb[2] / sizeof(float), -+ src1->nb[1] / sizeof(int64_t)); -+ } else { -+ k_set_rows_contiguous_f32x4<<>>( -+ (const float *) src0->data, (const int32_t *) src1->data, -+ (float *) dst->data, (int) src0->ne[0], dst->nb[1] / sizeof(float), -+ src0->nb[2] / sizeof(float), dst->nb[2] / sizeof(float), -+ src1->nb[1] / sizeof(int32_t)); -+ } -+ return; -+ } -+ - if (src1->type == GGML_TYPE_I64) { - set_rows_cuda(ctx, src0, src1, dst); - } else { diff --git a/ggml-patches/0013-cuda-cached-f16-cublas.patch b/ggml-patches/0013-cuda-cached-f16-cublas.patch deleted file mode 100644 index aec9e90..0000000 --- a/ggml-patches/0013-cuda-cached-f16-cublas.patch +++ /dev/null @@ -1,184 +0,0 @@ -diff --git a/src/ggml-cuda/CMakeLists.txt b/src/ggml-cuda/CMakeLists.txt -index 07a94052..f43ee5c4 100644 ---- a/src/ggml-cuda/CMakeLists.txt -+++ b/src/ggml-cuda/CMakeLists.txt -@@ -77,15 +77,15 @@ if (CUDAToolkit_FOUND) - endif() - - # Replace plain Blackwell CUDA architectures with their "architecture-specific" equivalents. -- # 11X/12X are forwards-compatible, 11Xa/12Xa are not. -+ # 10X/11X/12X are forwards-compatible, 10Xa/11Xa/12Xa are not. - # Notably the Blackwell tensor core instructions are not forwards compatible and therefore need architecture-specific targets. - # But while 12X vs. 12Xa can be checked in device code there is (to my knowledge) no easy way to do the same check in host code. - # So for now just replace the supported plain Blackwell targets with architecture-specific targets. - foreach(ARCHS IN ITEMS CMAKE_CUDA_ARCHITECTURES CMAKE_CUDA_ARCHITECTURES_NATIVE) - set(FIXED_ARCHS "") - foreach(ARCH IN LISTS ${ARCHS}) -- if (ARCH MATCHES "^(110|12[0-9])(-real|-virtual)?$") -- string(REGEX REPLACE "^(110|12[0-9])((-real|-virtual)?)$" "\\1a\\2" FIXED_ARCH ${ARCH}) -+ if (ARCH MATCHES "^(100|110|12[0-9])(-real|-virtual)?$") -+ string(REGEX REPLACE "^(100|110|12[0-9])((-real|-virtual)?)$" "\\1a\\2" FIXED_ARCH ${ARCH}) - message(STATUS "Replacing ${ARCH} in ${ARCHS} with ${FIXED_ARCH}") - list(APPEND FIXED_ARCHS "${FIXED_ARCH}") - else() -@@ -133,6 +133,9 @@ if (CUDAToolkit_FOUND) - ${GGML_SOURCES_CUDA} - ) - -+ # The cached-F16 path uses portable CUDA conversion and cuBLAS kernels. -+ target_compile_definitions(ggml-cuda PRIVATE GGML_CUDA_SKINNY_Q8_CUBLAS_F16=1) -+ - add_compile_definitions(GGML_CUDA_PEER_MAX_BATCH_SIZE=${GGML_CUDA_PEER_MAX_BATCH_SIZE}) - - if (GGML_CUDA_GRAPHS) -diff --git a/src/ggml-cuda/skinny-q8.cu b/src/ggml-cuda/skinny-q8.cu -index 88e1e1ec..1a7b4f3f 100644 ---- a/src/ggml-cuda/skinny-q8.cu -+++ b/src/ggml-cuda/skinny-q8.cu -@@ -27,6 +27,9 @@ - // batch size, so singleton and batched execution cannot select different math. - - #include "skinny-q8.cuh" -+#if defined(GGML_CUDA_SKINNY_Q8_CUBLAS_F16) -+#include "convert.cuh" -+#endif - #include "mmvq.cuh" // MMVQ_MAX_BATCH_SIZE: the N range mmvq already covers - - #include -@@ -41,7 +44,6 @@ - #define SKQ8_KTB (SKQ8_KSTEP / 32) // q8 blocks per stage - #define SKQ8_STAGES 2 - #define SKQ8_NMAX SKQ8_NPAD -- - // Shared-memory strides (bytes). KSTEP + 16 keeps the 16B cp.async stores - // aligned while breaking the power-of-two bank pattern on fragment loads. - #define SKQ8_SW (SKQ8_KSTEP + 16) -@@ -73,6 +75,35 @@ static __global__ void skq8_repack( - } - } - -+#if defined(GGML_CUDA_SKINNY_Q8_CUBLAS_F16) -+// One-time expansion of a planar Q8 weight into F16. Caching removes weight -+// dequantization from every invocation while retaining the compact Q8 model -+// on disk. -+static __global__ void skq8_dequantize_weight_f16( -+ const int8_t * __restrict__ qs, const half * __restrict__ d, -+ half * __restrict__ out, const int M, const int K) { -+ const size_t i2 = (size_t) blockIdx.x * blockDim.x + threadIdx.x; -+ const size_t i = i2 * 2; -+ if (i >= (size_t) M * K) { -+ return; -+ } -+ const int col = (int) (i % K); -+ const int row = (int) (i / K); -+ const float scale = __half2float(d[(size_t) row * (K / 32) + col / 32]); -+ const char2 q = *(const char2 *) (qs + i); -+ *(half2 *) (out + i) = __floats2half2_rn((float) q.x * scale, (float) q.y * scale); -+} -+ -+static __global__ void skq8_add_bias_f32( -+ float * __restrict__ dst, const float * __restrict__ bias, -+ const int M, const int64_t count) { -+ const int64_t i = (int64_t) blockIdx.x * blockDim.x + threadIdx.x; -+ if (i < count) { -+ dst[i] += bias[i % M]; -+ } -+} -+#endif -+ - // --------------------------------------------------------------------------- - // Quantize F32 activations [K, N] (col-contiguous) into a zero-padded - // SKQ8_NPAD-column buffer: int8[NPAD][K] + d float[NPAD][K/32]. -@@ -345,6 +376,9 @@ namespace { - struct skq8_planes { - int8_t * qs; - half * d; -+#if defined(GGML_CUDA_SKINNY_Q8_CUBLAS_F16) -+ half * f16; -+#endif - }; - std::unordered_map g_skq8_cache; - std::mutex g_skq8_mutex; -@@ -454,6 +488,18 @@ static void skq8_run( - const int64_t KB = K / 32; - - cudaStream_t stream = ctx.stream(); -+#if defined(GGML_CUDA_SKINNY_Q8_CUBLAS_F16) -+ static const bool cublas_f16_enabled = []() { -+ const char * e = getenv("GGML_SKINNY_Q8_CUBLAS_F16"); -+ return e != nullptr && e[0] != '0'; -+ }(); -+ static const int cublas_f16_min_n = []() { -+ const char * e = getenv("GGML_SKINNY_Q8_CUBLAS_F16_MIN_N"); -+ const int value = e != nullptr ? atoi(e) : 128; -+ return value > 0 ? value : 1; -+ }(); -+ const bool use_cublas_f16 = cublas_f16_enabled && N >= cublas_f16_min_n; -+#endif - - // Repacked weight planes (create on first use). Default: IN PLACE. The plane - // layout (qs M*K + d M*KB*2) is byte-for-byte the same total size as -@@ -470,7 +516,7 @@ static void skq8_run( - // though the repacked bytes are correct. Callers that share the process with - // such a scheduler set GGML_SKINNY_Q8_INPLACE=0 (the NMT pipeline does this - // when enabled). The streaming-ASR encoder runtime has no such hazard. -- skq8_planes planes; -+ skq8_planes planes = {}; - { - std::lock_guard lock(g_skq8_mutex); - auto it = g_skq8_cache.find(src0->data); -@@ -524,7 +570,54 @@ static void skq8_run( - } - } - } -+#if defined(GGML_CUDA_SKINNY_Q8_CUBLAS_F16) -+ if (use_cublas_f16 && planes.f16 == nullptr) { -+ const size_t elements = (size_t) M * K; -+ CUDA_CHECK(cudaMalloc(&planes.f16, elements * sizeof(half))); -+ const int blocks = (int) (((elements + 1) / 2 + 255) / 256); -+ skq8_dequantize_weight_f16<<>>( -+ planes.qs, planes.d, planes.f16, (int) M, (int) K); -+ g_skq8_cache.at(src0->data).f16 = planes.f16; -+ static const bool memstats = getenv("NEMO_SPEECH_MEMSTATS") != nullptr; -+ if (memstats) { -+ fprintf(stderr, -+ "[memstats] skinny-q8 F16 cache +%.1f MB (%s)\n", -+ elements * sizeof(half) / 1048576.0, src0->name); -+ } -+ } -+#endif -+ } -+ -+#if defined(GGML_CUDA_SKINNY_Q8_CUBLAS_F16) -+ // Keep the model artifact Q8, expand immutable weights once, convert only -+ // the live activation, and retain FP32 accumulation/output. -+ if (use_cublas_f16) { -+ GGML_ASSERT(planes.f16 != nullptr); -+ ggml_cuda_pool_alloc a_f16(ctx.pool(), (size_t) N * K); -+ const to_fp16_cuda_t to_fp16 = ggml_get_to_fp16_cuda(GGML_TYPE_F32); -+ GGML_ASSERT(to_fp16 != nullptr); -+ to_fp16(src1->data, a_f16.get(), N * K, stream); -+ -+ const float alpha = 1.0f; -+ const float beta = 0.0f; -+ cublasHandle_t handle = ctx.cublas_handle(); -+ CUBLAS_CHECK(cublasSetStream(handle, stream)); -+ CUBLAS_CHECK(cublasGemmEx( -+ handle, CUBLAS_OP_T, CUBLAS_OP_N, -+ (int) M, (int) N, (int) K, -+ &alpha, planes.f16, CUDA_R_16F, (int) K, -+ a_f16.get(), CUDA_R_16F, (int) K, -+ &beta, dst->data, CUDA_R_32F, (int) M, -+ CUBLAS_COMPUTE_32F, CUBLAS_GEMM_DEFAULT_TENSOR_OP)); -+ if (bias != nullptr) { -+ const int64_t count = M * N; -+ skq8_add_bias_f32<<<(count + 255) / 256, 256, 0, stream>>>( -+ (float *) dst->data, (const float *) bias->data, (int) M, count); -+ } -+ return; - } -+#endif -+ - // Quantize activations into the zero-padded buffer (all columns, once — - // ntot may exceed NPAD; the GEMM below tiles over 64-col chunks). - const int ntot = (int) ((N + SKQ8_NPAD - 1) / SKQ8_NPAD * SKQ8_NPAD); diff --git a/ggml-patches/0015-cuda-ctc-batch-fusions.patch b/ggml-patches/0015-cuda-ctc-batch-fusions.patch deleted file mode 100644 index 1f33287..0000000 --- a/ggml-patches/0015-cuda-ctc-batch-fusions.patch +++ /dev/null @@ -1,752 +0,0 @@ -diff --git a/src/ggml-cuda/ggml-cuda.cu b/src/ggml-cuda/ggml-cuda.cu -index 873ac0e..b92b14f 100644 ---- a/src/ggml-cuda/ggml-cuda.cu -+++ b/src/ggml-cuda/ggml-cuda.cu -@@ -4183,6 +4183,111 @@ static int ggml_cuda_try_fuse(ggml_backend_cuda_context * cuda_ctx, ggml_cgraph - return n_fused; - } - -+ const std::initializer_list batch_norm_ops = { -+ GGML_OP_SUB, GGML_OP_ADD, GGML_OP_SQRT, GGML_OP_DIV, GGML_OP_MUL, GGML_OP_ADD -+ }; -+ const bool batch_norm_ops_match = -+ i + (int) batch_norm_ops.size() <= cgraph->n_nodes && -+ std::equal( -+ batch_norm_ops.begin(), batch_norm_ops.end(), cgraph->nodes + i, -+ [](ggml_op op, const ggml_tensor * tensor) { return op == tensor->op; }); -+ const bool batch_norm_subgraph = -+ batch_norm_ops_match && -+ ggml_can_fuse_subgraph(cgraph, i, batch_norm_ops, { i + 5 }); -+ if (batch_norm_subgraph) { -+ ggml_tensor * sub = cgraph->nodes[i]; -+ ggml_tensor * var_add = cgraph->nodes[i + 1]; -+ ggml_tensor * sqrt = cgraph->nodes[i + 2]; -+ ggml_tensor * div = cgraph->nodes[i + 3]; -+ ggml_tensor * mul = cgraph->nodes[i + 4]; -+ ggml_tensor * out = cgraph->nodes[i + 5]; -+ -+ const ggml_tensor * input = sub->src[0]; -+ const ggml_tensor * mean = sub->src[1]; -+ const ggml_tensor * variance = nullptr; -+ const ggml_tensor * epsilon = nullptr; -+ if (ggml_nelements(var_add->src[0]) == 1) { -+ epsilon = var_add->src[0]; -+ variance = var_add->src[1]; -+ } else if (ggml_nelements(var_add->src[1]) == 1) { -+ variance = var_add->src[0]; -+ epsilon = var_add->src[1]; -+ } -+ -+ const ggml_tensor * weight = -+ mul->src[0] == div ? mul->src[1] : -+ mul->src[1] == div ? mul->src[0] : nullptr; -+ const ggml_tensor * bias = -+ out->src[0] == mul ? out->src[1] : -+ out->src[1] == mul ? out->src[0] : nullptr; -+ -+ const int64_t channels = input->ne[1]; -+ const bool graph_ok = -+ sqrt->src[0] == var_add && -+ div->src[0] == sub && div->src[1] == sqrt; -+ const bool params_ok = -+ variance != nullptr && epsilon != nullptr && weight != nullptr && bias != nullptr && -+ ggml_nelements(mean) == channels && -+ ggml_nelements(variance) == channels && -+ ggml_nelements(weight) == channels && -+ ggml_nelements(bias) == channels; -+ const bool types_ok = -+ input->type == GGML_TYPE_F32 && mean->type == GGML_TYPE_F32 && -+ variance != nullptr && variance->type == GGML_TYPE_F32 && -+ epsilon != nullptr && epsilon->type == GGML_TYPE_F32 && -+ weight != nullptr && weight->type == GGML_TYPE_F32 && -+ bias != nullptr && bias->type == GGML_TYPE_F32 && -+ sub->type == GGML_TYPE_F32 && var_add->type == GGML_TYPE_F32 && -+ sqrt->type == GGML_TYPE_F32 && div->type == GGML_TYPE_F32 && -+ mul->type == GGML_TYPE_F32 && out->type == GGML_TYPE_F32; -+ const bool layout_ok = -+ ggml_is_contiguous(input) && ggml_is_contiguous(mean) && -+ variance != nullptr && ggml_is_contiguous(variance) && -+ epsilon != nullptr && ggml_is_contiguous(epsilon) && -+ weight != nullptr && ggml_is_contiguous(weight) && -+ bias != nullptr && ggml_is_contiguous(bias) && -+ ggml_is_contiguous(out) && ggml_are_same_shape(input, out); -+ const int out_node = i + 5; -+ -+ const bool common_eligible = -+ graph_ok && params_ok && types_ok && layout_ok; -+ if (common_eligible && -+ ggml_can_fuse_subgraph( -+ cgraph, i, -+ { GGML_OP_SUB, GGML_OP_ADD, GGML_OP_SQRT, GGML_OP_DIV, GGML_OP_MUL, -+ GGML_OP_ADD, GGML_OP_PERMUTE, GGML_OP_CONT, GGML_OP_UNARY }, -+ { i + 8 })) { -+ ggml_tensor * permute = cgraph->nodes[i + 6]; -+ ggml_tensor * cont = cgraph->nodes[i + 7]; -+ ggml_tensor * silu = cgraph->nodes[i + 8]; -+ const bool transpose_ok = -+ permute->src[0] == out && cont->src[0] == permute && silu->src[0] == cont && -+ ggml_get_unary_op(silu) == GGML_UNARY_OP_SILU && -+ permute->type == GGML_TYPE_F32 && cont->type == GGML_TYPE_F32 && -+ silu->type == GGML_TYPE_F32 && -+ permute->ne[0] == input->ne[1] && permute->ne[1] == input->ne[0] && -+ permute->ne[2] == input->ne[2] && permute->ne[3] == input->ne[3] && -+ permute->nb[0] == out->nb[1] && permute->nb[1] == out->nb[0] && -+ permute->nb[2] == out->nb[2] && permute->nb[3] == out->nb[3] && -+ ggml_is_contiguous(cont) && ggml_is_contiguous(silu) && -+ ggml_are_same_shape(cont, silu); -+ const int transpose_out_node = i + 8; -+ if (transpose_ok && -+ ggml_cuda_check_fusion_memory_ranges( -+ cgraph, i, 9, &transpose_out_node, 1)) { -+ ggml_cuda_op_batch_norm_silu_transpose_fused( -+ *cuda_ctx, input, mean, variance, epsilon, weight, bias, silu); -+ return 8; -+ } -+ } -+ if (common_eligible && -+ ggml_cuda_check_fusion_memory_ranges(cgraph, i, 6, &out_node, 1)) { -+ ggml_cuda_op_batch_norm_fused( -+ *cuda_ctx, input, mean, variance, epsilon, weight, bias, out); -+ return 5; -+ } -+ } -+ - //topk-moe - if (cgraph->nodes[i]->op == GGML_OP_UNARY || cgraph->nodes[i]->op == GGML_OP_SOFT_MAX || - cgraph->nodes[i]->op == GGML_OP_ARGSORT) { -@@ -4566,21 +4671,23 @@ static int ggml_cuda_try_fuse(ggml_backend_cuda_context * cuda_ctx, ggml_cgraph - fused_mul_mat_vec = false; - fused_node_count = 0; - -- // A BF16 projection immediately following SiLU would otherwise launch an -- // F32 SiLU kernel and then a separate F32-to-BF16 input conversion. Emit -- // the same rounded BF16 activation directly in one pass. -+ // A 16-bit projection immediately following SiLU would otherwise launch an -+ // F32 SiLU kernel and then a separate input conversion. - if (ggml_can_fuse(cgraph, i, { GGML_OP_UNARY, GGML_OP_CPY })) { - const ggml_tensor * silu_node = cgraph->nodes[i]; - ggml_tensor * cast_node = cgraph->nodes[i + 1]; -+ const bool cast_supported = -+ cast_node->type == GGML_TYPE_F16 || -+ (cast_node->type == GGML_TYPE_BF16 && native_bf16); - const bool eligible = -- native_bf16 && ggml_get_unary_op(silu_node) == GGML_UNARY_OP_SILU && -+ cast_supported && ggml_get_unary_op(silu_node) == GGML_UNARY_OP_SILU && - cast_node->src[0] == silu_node && - silu_node->src[0]->type == GGML_TYPE_F32 && -- silu_node->type == GGML_TYPE_F32 && cast_node->type == GGML_TYPE_BF16 && -+ silu_node->type == GGML_TYPE_F32 && - ggml_are_same_shape(silu_node->src[0], cast_node) && - ggml_is_contiguous(silu_node->src[0]) && ggml_is_contiguous(cast_node); - if (eligible) { -- ggml_cuda_op_silu_f32_to_bf16(*cuda_ctx, silu_node, cast_node); -+ ggml_cuda_op_silu_f32_to_16(*cuda_ctx, silu_node, cast_node); - return 1; - } - } -@@ -4626,6 +4733,31 @@ static int ggml_cuda_try_fuse(ggml_backend_cuda_context * cuda_ctx, ggml_cgraph - } - } - -+ // Fold a following residual into the same cuBLASLt projection epilogue. -+ if (ggml_can_fuse(cgraph, i, { GGML_OP_MUL_MAT, GGML_OP_ADD, GGML_OP_ADD })) { -+ ggml_tensor * mm_node = cgraph->nodes[i]; -+ ggml_tensor * bias_node = cgraph->nodes[i + 1]; -+ ggml_tensor * residual_add = cgraph->nodes[i + 2]; -+ const ggml_tensor * bias = -+ bias_node->src[0] == mm_node ? bias_node->src[1] : -+ bias_node->src[1] == mm_node ? bias_node->src[0] : nullptr; -+ const ggml_tensor * residual = -+ residual_add->src[0] == bias_node ? residual_add->src[1] : -+ residual_add->src[1] == bias_node ? residual_add->src[0] : nullptr; -+ if (bias != nullptr && residual != nullptr && -+ bias->type == GGML_TYPE_F32 && residual->type == GGML_TYPE_F32 && -+ ggml_is_contiguous(bias) && ggml_is_contiguous(residual) && -+ ggml_nelements(bias) == mm_node->ne[0] && -+ ggml_are_same_shape(mm_node, residual) && -+ ggml_is_contiguous(residual_add) && -+ ggml_cuda_skinny_q8_residual_supported( -+ mm_node->src[0], mm_node->src[1], mm_node)) { -+ ggml_cuda_mul_mat_skinny_q8_bias_residual( -+ *cuda_ctx, mm_node->src[0], mm_node->src[1], bias, residual, residual_add); -+ return 2; -+ } -+ } -+ - // skinny-q8 GEMM + row-vector bias: the upstream MUL_MAT+ADD fusion below - // requires a same-shape add, so the classic broadcast Linear bias - // ([M] over [M,N]) never qualifies — fold it into the skinny GEMM's -@@ -4713,7 +4845,7 @@ static int ggml_cuda_try_fuse(ggml_backend_cuda_context * cuda_ctx, ggml_cgraph - return fused_node_count - 1; - } - -- // LayerNorm affine followed by a BF16 projection. Store the affine result -+ // LayerNorm affine followed by a 16-bit projection. Store the affine result - // directly in the projection's input precision instead of writing F32 and - // launching a second full-tensor conversion. - if (ggml_can_fuse(cgraph, i, -@@ -4726,10 +4858,13 @@ static int ggml_cuda_try_fuse(ggml_backend_cuda_context * cuda_ctx, ggml_cgraph - mul->src[0] == norm ? mul->src[1] : mul->src[0]; - const ggml_tensor * beta = - add->src[0] == mul ? add->src[1] : add->src[0]; -+ const bool cast_supported = -+ cast->type == GGML_TYPE_F16 || -+ (cast->type == GGML_TYPE_BF16 && native_bf16); - const bool eligible = -- native_bf16 && cast->src[0] == add && norm->src[0]->type == GGML_TYPE_F32 && -+ cast_supported && cast->src[0] == add && norm->src[0]->type == GGML_TYPE_F32 && - norm->type == GGML_TYPE_F32 && mul->type == GGML_TYPE_F32 && -- add->type == GGML_TYPE_F32 && cast->type == GGML_TYPE_BF16 && -+ add->type == GGML_TYPE_F32 && - gamma->type == GGML_TYPE_F32 && beta->type == GGML_TYPE_F32 && - ggml_nelements(gamma) == norm->ne[0] && - ggml_nelements(beta) == norm->ne[0] && -@@ -5690,7 +5825,9 @@ static bool ggml_backend_cuda_device_supports_op(ggml_backend_dev_t dev, const g - return false; - } - } -- if (b->type == GGML_TYPE_F16 && a->type != GGML_TYPE_F16) { -+ if (b->type == GGML_TYPE_F16 && a->type != GGML_TYPE_F16 && -+ !(op->op == GGML_OP_MUL_MAT && -+ ggml_cuda_skinny_q8_supported(a, b, op))) { - return false; - } - #ifdef GGML_USE_MUSA -diff --git a/src/ggml-cuda/norm.cu b/src/ggml-cuda/norm.cu -index 555c951..e30c025 100644 ---- a/src/ggml-cuda/norm.cu -+++ b/src/ggml-cuda/norm.cu -@@ -1,4 +1,5 @@ - #include "norm.cuh" -+#include "unary.cuh" - #include - - template -@@ -81,6 +82,61 @@ static __global__ void norm_mul_add_f32( - } - } - -+static __global__ void batch_norm_f32( -+ const float * x, const float * mean, const float * variance, const float * epsilon, -+ const float * weight, const float * bias, float * dst, int64_t nelements, -+ int64_t ntime, int64_t nchannels) { -+ const int64_t i = int64_t(blockIdx.x) * blockDim.x + threadIdx.x; -+ if (i >= nelements) { -+ return; -+ } -+ -+ const int64_t channel = (i / ntime) % nchannels; -+ const float centered = __fsub_rn(x[i], mean[channel]); -+ const float denominator = sqrtf(__fadd_rn(variance[channel], epsilon[0])); -+ const float normalized = __fdiv_rn(centered, denominator); -+ dst[i] = __fadd_rn(__fmul_rn(normalized, weight[channel]), bias[channel]); -+} -+ -+static __global__ void batch_norm_silu_transpose_f32( -+ const float * x, const float * mean, const float * variance, const float * epsilon, -+ const float * weight, const float * bias, float * dst, int64_t nelements, -+ int64_t ntime, int64_t nchannels) { -+ const int64_t i = int64_t(blockIdx.x) * blockDim.x + threadIdx.x; -+ if (i >= nelements) { -+ return; -+ } -+ -+ const int64_t channel = (i / ntime) % nchannels; -+ const int64_t time = i % ntime; -+ const int64_t sample = i / (ntime * nchannels); -+ const float centered = __fsub_rn(x[i], mean[channel]); -+ const float denominator = sqrtf(__fadd_rn(variance[channel], epsilon[0])); -+ const float normalized = __fdiv_rn(centered, denominator); -+ const float affine = __fadd_rn(__fmul_rn(normalized, weight[channel]), bias[channel]); -+ dst[channel + nchannels * (time + ntime * sample)] = ggml_cuda_op_silu_single(affine); -+} -+ -+static void batch_norm_f32_cuda( -+ const float * x, const float * mean, const float * variance, const float * epsilon, -+ const float * weight, const float * bias, float * dst, int64_t nelements, -+ int64_t ntime, int64_t nchannels, cudaStream_t stream) { -+ constexpr int block_size = 256; -+ const int64_t block_count = (nelements + block_size - 1) / block_size; -+ batch_norm_f32<<>>( -+ x, mean, variance, epsilon, weight, bias, dst, nelements, ntime, nchannels); -+} -+ -+static void batch_norm_silu_transpose_f32_cuda( -+ const float * x, const float * mean, const float * variance, const float * epsilon, -+ const float * weight, const float * bias, float * dst, int64_t nelements, -+ int64_t ntime, int64_t nchannels, cudaStream_t stream) { -+ constexpr int block_size = 256; -+ const int64_t block_count = (nelements + block_size - 1) / block_size; -+ batch_norm_silu_transpose_f32<<>>( -+ x, mean, variance, epsilon, weight, bias, dst, nelements, ntime, nchannels); -+} -+ - template - static __global__ void group_norm_f32(const float * x, float * dst, const int group_size, const int ne_elements, const float eps) { - // blockIdx.x: num_groups idx -@@ -334,7 +390,18 @@ static void norm_mul_add_cuda( - const int64_t stride_sample, const float eps, cudaStream_t stream) { - const dim3 blocks_num(nrows, nchannels, nsamples); - const int cc = ggml_cuda_info().devices[ggml_cuda_get_device()].cc; -- if (ncols < 1024) { -+ if (ncols == 768 && WARP_SIZE == 32) { -+ constexpr int block_size = 384; -+ if (add) { -+ norm_mul_add_f32 -+ <<>>( -+ x, mul, add, dst, ncols, stride_row, stride_channel, stride_sample, eps); -+ } else { -+ norm_mul_add_f32 -+ <<>>( -+ x, mul, add, dst, ncols, stride_row, stride_channel, stride_sample, eps); -+ } -+ } else if (ncols < 1024) { - const dim3 block_dims(WARP_SIZE, 1, 1); - if (add) { - norm_mul_add_f32<<>>( -@@ -343,10 +410,8 @@ static void norm_mul_add_cuda( - norm_mul_add_f32<<>>( - x, mul, add, dst, ncols, stride_row, stride_channel, stride_sample, eps); - } -- } else if (ncols == 1024 && GGML_CUDA_CC_IS_NVIDIA(cc) && cc >= GGML_CUDA_CC_HOPPER) { -- // Hopper and newer: four elements per thread keeps high occupancy -- // while reducing the affine LayerNorm reduction from 32 warps/block -- // to 8. Retain the established 1024-thread path on older devices. -+ } else if (ncols == 1024 && GGML_CUDA_CC_IS_NVIDIA(cc) && cc >= GGML_CUDA_CC_AMPERE) { -+ // Four elements per thread balances reduction work and occupancy. - const dim3 block_dims(256, 1, 1); - if (add) { - norm_mul_add_f32<256, true, T><<>>( -@@ -519,7 +584,7 @@ void ggml_cuda_op_norm(ggml_backend_cuda_context & ctx, ggml_tensor * dst) { - // or add_tensor), mirroring ggml_cuda_op_rms_norm_fused(_add). Eligibility - // (row-vector operands, F32, contiguity) is enforced in ggml_cuda_can_fuse. - void ggml_cuda_op_norm_fused(ggml_backend_cuda_context & ctx, ggml_tensor * dst, ggml_tensor * mul_tensor, -- ggml_tensor * add_tensor, ggml_tensor * bf16_dst) { -+ ggml_tensor * add_tensor, ggml_tensor * cast_dst) { - const ggml_tensor * norm_src = (ggml_tensor *) dst->src[0]; - float eps = 0.0f; - memcpy(&eps, dst->op_params, sizeof(float)); -@@ -563,13 +628,19 @@ void ggml_cuda_op_norm_fused(ggml_backend_cuda_context & ctx, ggml_tensor * dst, - const int64_t s02 = norm_src->nb[2] / ts0; - const int64_t s03 = norm_src->nb[3] / ts0; - -- if (bf16_dst != nullptr) { -- GGML_ASSERT(bf16_dst->type == GGML_TYPE_BF16); -- GGML_ASSERT(ggml_is_contiguous(bf16_dst)); -- GGML_ASSERT(ggml_are_same_shape(dst, bf16_dst)); -- norm_mul_add_cuda( -- src0_d, mul_d, add_d, (nv_bfloat16 *) bf16_dst->data, -- ne00, ne01, ne02, ne03, s01, s02, s03, eps, ctx.stream()); -+ if (cast_dst != nullptr) { -+ GGML_ASSERT(cast_dst->type == GGML_TYPE_BF16 || cast_dst->type == GGML_TYPE_F16); -+ GGML_ASSERT(ggml_is_contiguous(cast_dst)); -+ GGML_ASSERT(ggml_are_same_shape(dst, cast_dst)); -+ if (cast_dst->type == GGML_TYPE_BF16) { -+ norm_mul_add_cuda( -+ src0_d, mul_d, add_d, (nv_bfloat16 *) cast_dst->data, -+ ne00, ne01, ne02, ne03, s01, s02, s03, eps, ctx.stream()); -+ } else { -+ norm_mul_add_cuda( -+ src0_d, mul_d, add_d, (half *) cast_dst->data, -+ ne00, ne01, ne02, ne03, s01, s02, s03, eps, ctx.stream()); -+ } - } else { - norm_mul_add_cuda( - src0_d, mul_d, add_d, dst_d, -@@ -577,6 +648,50 @@ void ggml_cuda_op_norm_fused(ggml_backend_cuda_context & ctx, ggml_tensor * dst, - } - } - -+void ggml_cuda_op_batch_norm_fused(ggml_backend_cuda_context & ctx, -+ const ggml_tensor * input, -+ const ggml_tensor * mean, -+ const ggml_tensor * variance, -+ const ggml_tensor * epsilon, -+ const ggml_tensor * weight, -+ const ggml_tensor * bias, -+ ggml_tensor * dst) { -+ batch_norm_f32_cuda( -+ (const float *) input->data, -+ (const float *) mean->data, -+ (const float *) variance->data, -+ (const float *) epsilon->data, -+ (const float *) weight->data, -+ (const float *) bias->data, -+ (float *) dst->data, -+ ggml_nelements(input), -+ input->ne[0], -+ input->ne[1], -+ ctx.stream()); -+} -+ -+void ggml_cuda_op_batch_norm_silu_transpose_fused(ggml_backend_cuda_context & ctx, -+ const ggml_tensor * input, -+ const ggml_tensor * mean, -+ const ggml_tensor * variance, -+ const ggml_tensor * epsilon, -+ const ggml_tensor * weight, -+ const ggml_tensor * bias, -+ ggml_tensor * dst) { -+ batch_norm_silu_transpose_f32_cuda( -+ (const float *) input->data, -+ (const float *) mean->data, -+ (const float *) variance->data, -+ (const float *) epsilon->data, -+ (const float *) weight->data, -+ (const float *) bias->data, -+ (float *) dst->data, -+ ggml_nelements(input), -+ input->ne[0], -+ input->ne[1], -+ ctx.stream()); -+} -+ - void ggml_cuda_op_group_norm(ggml_backend_cuda_context & ctx, ggml_tensor * dst) { - const ggml_tensor * src0 = dst->src[0]; - const float * src0_d = (const float *)src0->data; -diff --git a/src/ggml-cuda/norm.cuh b/src/ggml-cuda/norm.cuh -index 8f4287c..f580ae9 100644 ---- a/src/ggml-cuda/norm.cuh -+++ b/src/ggml-cuda/norm.cuh -@@ -5,7 +5,25 @@ void ggml_cuda_op_norm(ggml_backend_cuda_context & ctx, ggml_tensor * dst); - // Fused LayerNorm + row-vector mul (gamma) + optional row-vector add (beta). - // add_tensor may be nullptr (norm+mul only). - void ggml_cuda_op_norm_fused(ggml_backend_cuda_context & ctx, ggml_tensor * dst, ggml_tensor * mul_tensor, -- ggml_tensor * add_tensor, ggml_tensor * bf16_dst = nullptr); -+ ggml_tensor * add_tensor, ggml_tensor * cast_dst = nullptr); -+ -+void ggml_cuda_op_batch_norm_fused(ggml_backend_cuda_context & ctx, -+ const ggml_tensor * input, -+ const ggml_tensor * mean, -+ const ggml_tensor * variance, -+ const ggml_tensor * epsilon, -+ const ggml_tensor * weight, -+ const ggml_tensor * bias, -+ ggml_tensor * dst); -+ -+void ggml_cuda_op_batch_norm_silu_transpose_fused(ggml_backend_cuda_context & ctx, -+ const ggml_tensor * input, -+ const ggml_tensor * mean, -+ const ggml_tensor * variance, -+ const ggml_tensor * epsilon, -+ const ggml_tensor * weight, -+ const ggml_tensor * bias, -+ ggml_tensor * dst); - - void ggml_cuda_op_group_norm(ggml_backend_cuda_context & ctx, ggml_tensor * dst); - -diff --git a/src/ggml-cuda/skinny-q8.cu b/src/ggml-cuda/skinny-q8.cu -index fe329bb..1122e05 100644 ---- a/src/ggml-cuda/skinny-q8.cu -+++ b/src/ggml-cuda/skinny-q8.cu -@@ -94,14 +94,19 @@ static __global__ void skq8_dequantize_weight_f16( - *(half2 *) (out + i) = __floats2half2_rn((float) q.x * scale, (float) q.y * scale); - } - --static __global__ void skq8_add_bias_f32( -- float * __restrict__ dst, const float * __restrict__ bias, -- const int M, const int64_t count) { -- const int64_t i = (int64_t) blockIdx.x * blockDim.x + threadIdx.x; -- if (i < count) { -- dst[i] += bias[i % M]; -- } -+static bool skq8_cublas_f16_enabled_for_n(int64_t n) { -+ static const bool enabled = []() { -+ const char * e = getenv("GGML_SKINNY_Q8_CUBLAS_F16"); -+ return e != nullptr && e[0] != '0'; -+ }(); -+ static const int min_n = []() { -+ const char * e = getenv("GGML_SKINNY_Q8_CUBLAS_F16_MIN_N"); -+ const int value = e != nullptr ? atoi(e) : 128; -+ return value > 0 ? value : 1; -+ }(); -+ return enabled && n >= min_n; - } -+ - #endif - - // --------------------------------------------------------------------------- -@@ -388,6 +393,25 @@ static __global__ void skq8_reduce_splitk( - dst[i] = sum; - } - -+#if defined(GGML_CUDA_SKINNY_Q8_CUBLAS_F16) -+static __global__ void skq8_add_epilogue_f32( -+ float * __restrict__ dst, const float * __restrict__ bias, -+ const float * __restrict__ residual, const int m, const int64_t count) { -+ const int64_t i = (int64_t) blockIdx.x * blockDim.x + threadIdx.x; -+ if (i >= count) { -+ return; -+ } -+ float value = dst[i]; -+ if (residual != nullptr) { -+ value += residual[i]; -+ } -+ if (bias != nullptr) { -+ value += bias[i % m]; -+ } -+ dst[i] = value; -+} -+#endif -+ - // Repacked-weight cache. Keyed by the weight tensor's device pointer; entries - // live for the process lifetime (model weights are loaded once). Repacking - // happens on first (eager/warmup) use, before any CUDA-graph capture. -@@ -453,10 +477,18 @@ bool ggml_cuda_skinny_q8_supported( - // all outer columns as one logical N so a weight repacked by a scalar - // graph remains usable by a later true-batch graph. - const int64_t total_n = ggml_nelements(src1) / src1->ne[0]; -- const bool call_ok = src1->type == GGML_TYPE_F32 && dst->type == GGML_TYPE_F32 && -- ggml_is_contiguous(src1) && ggml_is_contiguous(dst) && -- src1->ne[0] == src0->ne[0] && -- ggml_nelements(dst) == src0->ne[1] * total_n; -+ const bool f16_input_ok = -+#if defined(GGML_CUDA_SKINNY_Q8_CUBLAS_F16) -+ src1->type == GGML_TYPE_F16 && skq8_cublas_f16_enabled_for_n(total_n); -+#else -+ false; -+#endif -+ const bool call_ok = -+ (src1->type == GGML_TYPE_F32 || f16_input_ok) && -+ dst->type == GGML_TYPE_F32 && -+ ggml_is_contiguous(src1) && ggml_is_contiguous(dst) && -+ src1->ne[0] == src0->ne[0] && -+ ggml_nelements(dst) == src0->ne[1] * total_n; - - if (planar_q8) { - // N<=8 is handled by planar MMVQ. Wider calls must stay on this path -@@ -479,13 +511,9 @@ bool ggml_cuda_skinny_q8_supported( - GGML_ASSERT(call_ok && "skinny-q8: repacked weight used in an unsupported mul_mat shape"); - return true; - } -- // By default, select only on logical width so outer batch size cannot -- // change the accumulation path. The opt-in mode is an end-to-end ASR -- // experiment for streaming shapes such as [K,2,B]: it flattens the dense -- // outer batch and lets the tensor-core kernel tile N beyond 64. It is not -- // the default because skinny-Q8 has a different accumulation order from -- // MMVQ; callers must validate transcript/accuracy parity. Use a separate -- // repack allocation (GGML_SKINNY_Q8_INPLACE=0) with multi-stream schedulers. -+ // Keep the accumulation path independent of outer batch size by default. -+ // The opt-in mode flattens [K,N,B...] for the tensor-core kernel; use a -+ // separate repack allocation with multi-stream schedulers. - static const bool outer_batch_dispatch = []() { - const char * e = getenv("GGML_SKINNY_Q8_OUTER_BATCH"); - return e != nullptr && e[0] != '0'; -@@ -498,9 +526,21 @@ bool ggml_cuda_skinny_q8_supported( - return eligible; - } - -+bool ggml_cuda_skinny_q8_residual_supported( -+ const ggml_tensor * src0, const ggml_tensor * src1, const ggml_tensor * dst) { -+#if defined(GGML_CUDA_SKINNY_Q8_CUBLAS_F16) -+ const int64_t total_n = ggml_nelements(src1) / src1->ne[0]; -+ return ggml_cuda_skinny_q8_supported(src0, src1, dst) && -+ skq8_cublas_f16_enabled_for_n(total_n); -+#else -+ GGML_UNUSED_VARS(src0, src1, dst); -+ return false; -+#endif -+} -+ - static void skq8_run( - ggml_backend_cuda_context & ctx, const ggml_tensor * src0, const ggml_tensor * src1, -- const ggml_tensor * bias, ggml_tensor * dst) { -+ const ggml_tensor * bias, const ggml_tensor * residual, ggml_tensor * dst) { - const int64_t M = src0->ne[1]; - const int64_t K = src0->ne[0]; - const int64_t N = ggml_nelements(src1) / src1->ne[0]; -@@ -508,16 +548,7 @@ static void skq8_run( - - cudaStream_t stream = ctx.stream(); - #if defined(GGML_CUDA_SKINNY_Q8_CUBLAS_F16) -- static const bool cublas_f16_enabled = []() { -- const char * e = getenv("GGML_SKINNY_Q8_CUBLAS_F16"); -- return e != nullptr && e[0] != '0'; -- }(); -- static const int cublas_f16_min_n = []() { -- const char * e = getenv("GGML_SKINNY_Q8_CUBLAS_F16_MIN_N"); -- const int value = e != nullptr ? atoi(e) : 128; -- return value > 0 ? value : 1; -- }(); -- const bool use_cublas_f16 = cublas_f16_enabled && N >= cublas_f16_min_n; -+ const bool use_cublas_f16 = skq8_cublas_f16_enabled_for_n(N); - #endif - - // Repacked weight planes (create on first use). Default: IN PLACE. The plane -@@ -612,29 +643,44 @@ static void skq8_run( - // the live activation, and retain FP32 accumulation/output. - if (use_cublas_f16) { - GGML_ASSERT(planes.f16 != nullptr); -- ggml_cuda_pool_alloc a_f16(ctx.pool(), (size_t) N * K); -- const to_fp16_cuda_t to_fp16 = ggml_get_to_fp16_cuda(GGML_TYPE_F32); -- GGML_ASSERT(to_fp16 != nullptr); -- to_fp16(src1->data, a_f16.get(), N * K, stream); -- -- const float alpha = 1.0f; -- const float beta = 0.0f; -- cublasHandle_t handle = ctx.cublas_handle(); -- CUBLAS_CHECK(cublasSetStream(handle, stream)); -- CUBLAS_CHECK(cublasGemmEx( -- handle, CUBLAS_OP_T, CUBLAS_OP_N, -- (int) M, (int) N, (int) K, -- &alpha, planes.f16, CUDA_R_16F, (int) K, -- a_f16.get(), CUDA_R_16F, (int) K, -- &beta, dst->data, CUDA_R_32F, (int) M, -- CUBLAS_COMPUTE_32F, CUBLAS_GEMM_DEFAULT_TENSOR_OP)); -- if (bias != nullptr) { -- const int64_t count = M * N; -- skq8_add_bias_f32<<<(count + 255) / 256, 256, 0, stream>>>( -- (float *) dst->data, (const float *) bias->data, (int) M, count); -+ auto gemm = [&](const half* activation) { -+ const float alpha = 1.0f; -+ const bool residual_in_place = -+ residual != nullptr && residual->data == dst->data; -+ const float beta = residual_in_place ? 1.0f : 0.0f; -+ cublasHandle_t handle = ctx.cublas_handle(); -+ CUBLAS_CHECK(cublasSetStream(handle, stream)); -+ CUBLAS_CHECK(cublasGemmEx( -+ handle, CUBLAS_OP_T, CUBLAS_OP_N, -+ (int) M, (int) N, (int) K, -+ &alpha, planes.f16, CUDA_R_16F, (int) K, -+ activation, CUDA_R_16F, (int) K, -+ &beta, dst->data, CUDA_R_32F, (int) M, -+ CUBLAS_COMPUTE_32F, CUBLAS_GEMM_DEFAULT_TENSOR_OP)); -+ if (bias != nullptr || (residual != nullptr && !residual_in_place)) { -+ const int64_t count = M * N; -+ skq8_add_epilogue_f32<<<(count + 255) / 256, 256, 0, stream>>>( -+ (float *) dst->data, -+ bias != nullptr ? (const float *) bias->data : nullptr, -+ residual != nullptr && !residual_in_place -+ ? (const float *) residual->data -+ : nullptr, -+ (int) M, count); -+ } -+ }; -+ if (src1->type == GGML_TYPE_F16) { -+ gemm((const half*) src1->data); -+ } else { -+ ggml_cuda_pool_alloc a_f16(ctx.pool(), (size_t) N * K); -+ const to_fp16_cuda_t to_fp16 = ggml_get_to_fp16_cuda(GGML_TYPE_F32); -+ GGML_ASSERT(to_fp16 != nullptr); -+ to_fp16(src1->data, a_f16.get(), N * K, stream); -+ gemm(a_f16.get()); - } - return; - } -+ GGML_ASSERT(src1->type == GGML_TYPE_F32); -+ GGML_ASSERT(residual == nullptr); - #endif - - // Quantize activations into the zero-padded buffer (all columns, once — -@@ -691,11 +737,17 @@ static void skq8_run( - void ggml_cuda_mul_mat_skinny_q8( - ggml_backend_cuda_context & ctx, const ggml_tensor * src0, const ggml_tensor * src1, - ggml_tensor * dst) { -- skq8_run(ctx, src0, src1, nullptr, dst); -+ skq8_run(ctx, src0, src1, nullptr, nullptr, dst); - } - - void ggml_cuda_mul_mat_skinny_q8_bias( - ggml_backend_cuda_context & ctx, const ggml_tensor * src0, const ggml_tensor * src1, - const ggml_tensor * bias, ggml_tensor * dst) { -- skq8_run(ctx, src0, src1, bias, dst); -+ skq8_run(ctx, src0, src1, bias, nullptr, dst); -+} -+ -+void ggml_cuda_mul_mat_skinny_q8_bias_residual( -+ ggml_backend_cuda_context & ctx, const ggml_tensor * src0, const ggml_tensor * src1, -+ const ggml_tensor * bias, const ggml_tensor * residual, ggml_tensor * dst) { -+ skq8_run(ctx, src0, src1, bias, residual, dst); - } -diff --git a/src/ggml-cuda/skinny-q8.cuh b/src/ggml-cuda/skinny-q8.cuh -index 17175bb..30238fe 100644 ---- a/src/ggml-cuda/skinny-q8.cuh -+++ b/src/ggml-cuda/skinny-q8.cuh -@@ -7,6 +7,9 @@ - bool ggml_cuda_skinny_q8_supported( - const ggml_tensor * src0, const ggml_tensor * src1, const ggml_tensor * dst); - -+bool ggml_cuda_skinny_q8_residual_supported( -+ const ggml_tensor * src0, const ggml_tensor * src1, const ggml_tensor * dst); -+ - void ggml_cuda_mul_mat_skinny_q8( - ggml_backend_cuda_context & ctx, const ggml_tensor * src0, const ggml_tensor * src1, - ggml_tensor * dst); -@@ -16,3 +19,7 @@ void ggml_cuda_mul_mat_skinny_q8( - void ggml_cuda_mul_mat_skinny_q8_bias( - ggml_backend_cuda_context & ctx, const ggml_tensor * src0, const ggml_tensor * src1, - const ggml_tensor * bias, ggml_tensor * dst); -+ -+void ggml_cuda_mul_mat_skinny_q8_bias_residual( -+ ggml_backend_cuda_context & ctx, const ggml_tensor * src0, const ggml_tensor * src1, -+ const ggml_tensor * bias, const ggml_tensor * residual, ggml_tensor * dst); -diff --git a/src/ggml-cuda/unary.cu b/src/ggml-cuda/unary.cu -index 4c68411..24b3ff7 100644 ---- a/src/ggml-cuda/unary.cu -+++ b/src/ggml-cuda/unary.cu -@@ -183,11 +183,20 @@ void ggml_cuda_op_silu(ggml_backend_cuda_context & ctx, ggml_tensor * dst) { - ggml_cuda_op_unary(ctx, dst); - } - -+static __global__ void silu_f32_to_f16( -+ const float * __restrict__ x, half * __restrict__ dst, -+ const int64_t nelements) { -+ const int64_t i = (int64_t) blockIdx.x * blockDim.x + threadIdx.x; -+ if (i < nelements) { -+ dst[i] = __float2half(op_silu(x[i])); -+ } -+} -+ - static __global__ void silu_f32_to_bf16( - const float * __restrict__ x, nv_bfloat16 * __restrict__ dst, - const int64_t nelements) { - #if defined(__CUDA_ARCH__) && __CUDA_ARCH__ >= GGML_CUDA_CC_AMPERE -- const int64_t i = (int64_t) blockIdx.x * blockDim.x + threadIdx.x; -+ const int64_t i = int64_t(blockIdx.x) * blockDim.x + threadIdx.x; - if (i < nelements) { - dst[i] = __float2bfloat16(op_silu(x[i])); - } -@@ -197,21 +206,26 @@ static __global__ void silu_f32_to_bf16( - #endif - } - --void ggml_cuda_op_silu_f32_to_bf16( -+void ggml_cuda_op_silu_f32_to_16( - ggml_backend_cuda_context & ctx, const ggml_tensor * silu_node, - ggml_tensor * dst) { - const ggml_tensor * src = silu_node->src[0]; - GGML_ASSERT(src->type == GGML_TYPE_F32); - GGML_ASSERT(silu_node->type == GGML_TYPE_F32); -- GGML_ASSERT(dst->type == GGML_TYPE_BF16); -+ GGML_ASSERT(dst->type == GGML_TYPE_BF16 || dst->type == GGML_TYPE_F16); - GGML_ASSERT(ggml_are_same_shape(src, dst)); - GGML_ASSERT(ggml_is_contiguous(src)); - GGML_ASSERT(ggml_is_contiguous(dst)); - - const int64_t nelements = ggml_nelements(src); - const int64_t num_blocks = (nelements + CUDA_SILU_BLOCK_SIZE - 1) / CUDA_SILU_BLOCK_SIZE; -- silu_f32_to_bf16<<>>( -- (const float *) src->data, (nv_bfloat16 *) dst->data, nelements); -+ if (dst->type == GGML_TYPE_BF16) { -+ silu_f32_to_bf16<<>>( -+ (const float *) src->data, (nv_bfloat16 *) dst->data, nelements); -+ } else { -+ silu_f32_to_f16<<>>( -+ (const float *) src->data, (half *) dst->data, nelements); -+ } - } - - void ggml_cuda_op_tanh(ggml_backend_cuda_context & ctx, ggml_tensor * dst) { -diff --git a/src/ggml-cuda/unary.cuh b/src/ggml-cuda/unary.cuh -index 5307327..5bfc74f 100644 ---- a/src/ggml-cuda/unary.cuh -+++ b/src/ggml-cuda/unary.cuh -@@ -31,7 +31,7 @@ void ggml_cuda_op_gelu(ggml_backend_cuda_context & ctx, ggml_tensor * dst); - - void ggml_cuda_op_silu(ggml_backend_cuda_context & ctx, ggml_tensor * dst); - --void ggml_cuda_op_silu_f32_to_bf16( -+void ggml_cuda_op_silu_f32_to_16( - ggml_backend_cuda_context & ctx, const ggml_tensor * silu_node, ggml_tensor * dst); - - void ggml_cuda_op_silu_back(ggml_backend_cuda_context & ctx, ggml_tensor * dst); diff --git a/ggml-patches/0016-fix-batched-conv1d-layout.patch b/ggml-patches/0016-fix-batched-conv1d-layout.patch deleted file mode 100644 index b1ded39..0000000 --- a/ggml-patches/0016-fix-batched-conv1d-layout.patch +++ /dev/null @@ -1,21 +0,0 @@ -diff --git a/src/ggml.c b/src/ggml.c ---- a/src/ggml.c -+++ b/src/ggml.c -@@ -4494,7 +4494,16 @@ struct ggml_tensor * ggml_conv_1d( - ggml_reshape_2d(ctx, im2col, im2col->ne[0], (im2col->ne[2] * im2col->ne[1])), // [N, OL, IC * K] => [N*OL, IC * K] - ggml_reshape_2d(ctx, a, (a->ne[0] * a->ne[1]), a->ne[2])); // [OC,IC, K] => [OC, IC * K] - -- result = ggml_reshape_3d(ctx, result, im2col->ne[1], a->ne[2], im2col->ne[2]); // [N, OC, OL] -+ if (im2col->ne[2] == 1) { -+ result = ggml_reshape_3d(ctx, result, im2col->ne[1], a->ne[2], 1); -+ } else { -+ // mul_mat produces [N*OL, OC]. Restore the flattened axes before moving -+ // the batch axis behind the output channels; a direct [OL, OC, N] -+ // reshape interleaves OC and N when N > 1. -+ result = -+ ggml_reshape_3d(ctx, result, im2col->ne[1], im2col->ne[2], a->ne[2]); -+ result = ggml_cont(ctx, ggml_permute(ctx, result, 0, 2, 1, 3)); -+ } - - return result; - } diff --git a/ggml-patches/0018-metal-tensor-api-dynamic-k.patch b/ggml-patches/0018-metal-tensor-api-dynamic-k.patch deleted file mode 100644 index 0a51dad..0000000 --- a/ggml-patches/0018-metal-tensor-api-dynamic-k.patch +++ /dev/null @@ -1,60 +0,0 @@ -diff --git a/src/ggml-metal/ggml-metal.metal b/src/ggml-metal/ggml-metal.metal -index f6ffb2b..55ad5ea 100644 ---- a/src/ggml-metal/ggml-metal.metal -+++ b/src/ggml-metal/ggml-metal.metal -@@ -9412,9 +9412,12 @@ kernel void kernel_mul_mm( - auto tB = tensor(ptrB, dextents(K, N), array({1, strideB})); - - // Configure matmul operation -+ // note: K is dynamic_extent (clamped to the valid range in PHASE 2), since a static -+ // N_MM_NK_TOTAL K tile would read src1 out of bounds when K % N_MM_NK_TOTAL != 0 -+ // ref: https://github.com/ggml-org/llama.cpp/pull/27064 - mpp::tensor_ops::matmul2d< - mpp::tensor_ops::matmul2d_descriptor( -- NRB, NRA, N_MM_NK_TOTAL, false, true, true, -+ NRB, NRA, static_cast(dynamic_extent), false, true, true, - mpp::tensor_ops::matmul2d_descriptor::mode::multiply_accumulate), - execution_simdgroups> mm; - -@@ -9466,10 +9469,14 @@ kernel void kernel_mul_mm( - threadgroup_barrier(mem_flags::mem_threadgroup); - - // === PHASE 2: Tensor matmul === -- auto mA = tA.slice(0, 0); -- auto mB = tB.slice(loop_k, rb); -+ // Clamp the K extent of both operand tensors to the remaining valid K range so -+ // the dynamic-K op never reads past the K extent of src1 (or the staged A tile). -+ const int kExt = min(N_MM_NK_TOTAL, K - loop_k); - -- mm.run(mB, mA, cT); -+ auto tAv = tensor(sa, dextents(kExt, NRA), array({1, N_MM_NK_TOTAL})); -+ auto tBv = tensor(ptrB + loop_k + rb * strideB, dextents(kExt, N - rb), array({1, strideB})); -+ -+ mm.run(tBv, tAv, cT); - - threadgroup_barrier(mem_flags::mem_threadgroup); - } -diff --git a/tests/test-backend-ops.cpp b/tests/test-backend-ops.cpp -index f54ab41..85e0e92 100644 ---- a/tests/test-backend-ops.cpp -+++ b/tests/test-backend-ops.cpp -@@ -8401,6 +8401,19 @@ static std::vector> make_test_cases_eval() { - test_cases.emplace_back(new test_mul_mat(GGML_TYPE_F32, GGML_TYPE_F32, 16, 32, 32, { 1, 1}, {1, 1}, {0, 1, 2, 3}, 64, 3)); - test_cases.emplace_back(new test_mul_mat(GGML_TYPE_F32, GGML_TYPE_F32, 64, 77, 77, {12,1}, {1,1})); - -+ // K not a multiple of 32 - exercises the partial-K tile of the Metal tensor API -+ // mat-mat kernel, which previously read past the K extent of src1 (ggml 33c9ea5) -+ test_cases.emplace_back(new test_mul_mat(GGML_TYPE_F16, GGML_TYPE_F16, 64, 32, 65, {1, 1}, {1, 1})); -+ test_cases.emplace_back(new test_mul_mat(GGML_TYPE_F16, GGML_TYPE_F16, 64, 32, 80, {1, 1}, {1, 1})); -+ test_cases.emplace_back(new test_mul_mat(GGML_TYPE_F16, GGML_TYPE_F32, 64, 32, 80, {1, 1}, {1, 1})); -+ test_cases.emplace_back(new test_mul_mat(GGML_TYPE_F32, GGML_TYPE_F32, 64, 32, 80, {1, 1}, {1, 1})); -+ test_cases.emplace_back(new test_mul_mat(GGML_TYPE_F16, GGML_TYPE_F16, 64, 32, 588, {1, 1}, {1, 1})); // 14*14*3, e.g. conv_2d im2col -+ test_cases.emplace_back(new test_mul_mat(GGML_TYPE_F16, GGML_TYPE_F16, 64, 32, 80, {4, 1}, {1, 1})); -+ // NanoCodec residual input convolution: K = 1296 (1296 % 32 == 16), plus an aligned control -+ test_cases.emplace_back(new test_mul_mat(GGML_TYPE_F16, GGML_TYPE_F16, 8, 432, 1296, {1, 1}, {1, 1})); -+ test_cases.emplace_back(new test_mul_mat(GGML_TYPE_F16, GGML_TYPE_F16, 32, 432, 1296, {1, 1}, {1, 1})); -+ test_cases.emplace_back(new test_mul_mat(GGML_TYPE_F16, GGML_TYPE_F16, 32, 432, 1280, {1, 1}, {1, 1})); -+ - test_cases.emplace_back(new test_mul_mat(GGML_TYPE_Q4_0, GGML_TYPE_F32, 576, 512, 576, {1,1}, {1,1})); - test_cases.emplace_back(new test_mul_mat(GGML_TYPE_Q4_0, GGML_TYPE_F32, 1, 2048, 8192, {1, 1}, {1, 1})); - for (ggml_type type_a : all_types) { diff --git a/ggml-patches/0019-cuda-graph-dynamic-update.patch b/ggml-patches/0019-cuda-graph-dynamic-update.patch deleted file mode 100644 index ea38ca0..0000000 --- a/ggml-patches/0019-cuda-graph-dynamic-update.patch +++ /dev/null @@ -1,59 +0,0 @@ -diff --git a/src/ggml-cuda/common.cuh b/src/ggml-cuda/common.cuh -index 610fdf37..474eb5c7 100644 ---- a/src/ggml-cuda/common.cuh -+++ b/src/ggml-cuda/common.cuh -@@ -1210,5 +1210,6 @@ struct ggml_cuda_graph { - std::vector nodes; - bool disable_due_to_gpu_arch = false; -+ bool warmup_started = false; - bool warmup_complete = false; - uint64_t uid = 0; - int64_t last_used_time = 0; -diff --git a/src/ggml-cuda/ggml-cuda.cu b/src/ggml-cuda/ggml-cuda.cu -index e8f6c10f..fd97f206 100644 ---- a/src/ggml-cuda/ggml-cuda.cu -+++ b/src/ggml-cuda/ggml-cuda.cu -@@ -5044,25 +5044,25 @@ static enum ggml_status ggml_backend_cuda_graph_compute(ggml_backend_t backend, - if (graph_compatible) { - const bool properties_changed = ggml_cuda_graph_update_required(cuda_ctx, cgraph, graph_key); - -- if (!graph->warmup_complete) { -- // Warmup: need at least 2 calls with no property change on the 2nd call -- if (!properties_changed) { -- graph->warmup_complete = true; -- ggml_cuda_graph_log_event(__func__, "warmup complete", cgraph, graph_key); -- use_cuda_graph = true; -- cuda_graph_update_required = true; -- } -- // else: properties changed or first call - execute directly (use_cuda_graph stays false) -+ if (!graph->warmup_started) { -+ // Execute one call directly so lazy kernel/library setup happens -+ // outside stream capture. Cache/state addresses are expected to -+ // change between streaming calls, so stability is not a valid -+ // prerequisite for completing warmup. -+ graph->warmup_started = true; -+ } else if (!graph->warmup_complete) { -+ graph->warmup_complete = true; -+ ggml_cuda_graph_log_event(__func__, "warmup complete", cgraph, graph_key); -+ use_cuda_graph = true; -+ cuda_graph_update_required = true; - } else { -- // Post-warmup: normal CUDA graph operation -- if (properties_changed) { -- // Properties changed - reset warmup, execute directly until stable again -- graph->warmup_complete = false; -- ggml_cuda_graph_log_event(__func__, "warmup reset", cgraph, graph_key); -- } else { -- use_cuda_graph = true; -- cuda_graph_update_required = graph->instance == nullptr; -- } -+ // CUDA graph topology is stable even when tensor/cache pointers -+ // rotate. Re-capture the current parameters and update the -+ // executable instead of invalidating it and falling back to -+ // repeated direct launches. cudaGraphExecUpdate() below safely -+ // re-instantiates if CUDA reports a topology incompatibility. -+ use_cuda_graph = true; -+ cuda_graph_update_required = properties_changed || graph->instance == nullptr; - } - } - } diff --git a/ggml-patches/0020-bf16-convolution.patch b/ggml-patches/0020-bf16-convolution.patch deleted file mode 100644 index 3cb975d..0000000 --- a/ggml-patches/0020-bf16-convolution.patch +++ /dev/null @@ -1,337 +0,0 @@ -diff --git a/src/ggml-cuda/conv2d-dw.cu b/src/ggml-cuda/conv2d-dw.cu -index db7ee6bf..572eb6bf 100644 ---- a/src/ggml-cuda/conv2d-dw.cu -+++ b/src/ggml-cuda/conv2d-dw.cu -@@ -121,7 +121,8 @@ void ggml_cuda_op_conv2d_dw(ggml_backend_cuda_context & ctx, ggml_tensor * dst) - const ggml_tensor * input = dst->src[1]; - - GGML_ASSERT(input->type == GGML_TYPE_F32 && dst->type == GGML_TYPE_F32); -- GGML_ASSERT(kernel->type == GGML_TYPE_F32 || kernel->type == GGML_TYPE_F16); -+ GGML_ASSERT(kernel->type == GGML_TYPE_F32 || kernel->type == GGML_TYPE_F16 || -+ kernel->type == GGML_TYPE_BF16); - const void * w_d = kernel->data; - const float * x_d = (const float *) input->data; - float * y_d = (float *) dst->data; -@@ -153,6 +154,10 @@ void ggml_cuda_op_conv2d_dw(ggml_backend_cuda_context & ctx, ggml_tensor * dst) - conv2d_dw_kernel<<>>( - x_d, (const half *) w_d, y_d, in_w, in_h, out_w, out_h, kernel_w, kernel_h, stride_x, stride_y, - padding_x, padding_y, dilation_x, dilation_y, channels, batches); -+ } else if (kernel->type == GGML_TYPE_BF16) { -+ conv2d_dw_kernel<<>>( -+ x_d, (const nv_bfloat16 *) w_d, y_d, in_w, in_h, out_w, out_h, kernel_w, kernel_h, stride_x, -+ stride_y, padding_x, padding_y, dilation_x, dilation_y, channels, batches); - } else { - conv2d_dw_kernel<<>>( - x_d, (const float *) w_d, y_d, in_w, in_h, out_w, out_h, kernel_w, kernel_h, stride_x, stride_y, -@@ -163,6 +168,10 @@ void ggml_cuda_op_conv2d_dw(ggml_backend_cuda_context & ctx, ggml_tensor * dst) - conv2d_dw_kernel<<>>( - x_d, (const half *) w_d, y_d, in_w, in_h, out_w, out_h, kernel_w, kernel_h, stride_x, stride_y, - padding_x, padding_y, dilation_x, dilation_y, channels, batches); -+ } else if (kernel->type == GGML_TYPE_BF16) { -+ conv2d_dw_kernel<<>>( -+ x_d, (const nv_bfloat16 *) w_d, y_d, in_w, in_h, out_w, out_h, kernel_w, kernel_h, stride_x, -+ stride_y, padding_x, padding_y, dilation_x, dilation_y, channels, batches); - } else { - conv2d_dw_kernel<<>>( - x_d, (const float *) w_d, y_d, in_w, in_h, out_w, out_h, kernel_w, kernel_h, stride_x, stride_y, -diff --git a/src/ggml-cuda/ggml-cuda.cu b/src/ggml-cuda/ggml-cuda.cu -index d974948c..3b4185c0 100644 ---- a/src/ggml-cuda/ggml-cuda.cu -+++ b/src/ggml-cuda/ggml-cuda.cu -@@ -2213,6 +2213,90 @@ static void launch_bf16_to_f32_add_row_bias( - bf16_to_f32_add_row_bias<<>>(src, bias, dst, rows); - } - -+static __global__ void bf16_add_row_bias_round_to_f32( -+ const nv_bfloat16 * __restrict__ src, const float * __restrict__ bias, -+ float * __restrict__ dst, int64_t rows) { -+#if defined(__CUDA_ARCH__) && __CUDA_ARCH__ >= GGML_CUDA_CC_AMPERE -+ const int64_t row = int64_t(blockIdx.x) * blockDim.x + threadIdx.x; -+ if (row >= rows) { -+ return; -+ } -+ const int64_t i = int64_t(blockIdx.y) * rows + row; -+ dst[i] = __bfloat162float(__float2bfloat16(__bfloat162float(src[i]) + bias[row])); -+#else -+ GGML_UNUSED_VARS(src, bias, dst, rows); -+ NO_DEVICE_CODE; -+#endif -+} -+ -+static void launch_bf16_add_row_bias_round_to_f32( -+ const nv_bfloat16 * src, const float * bias, float * dst, -+ int64_t rows, int64_t cols, cudaStream_t stream) { -+ constexpr int block_size = 256; -+ GGML_ASSERT(cols <= 65535); -+ const dim3 block(block_size, 1, 1); -+ const dim3 grid((rows + block_size - 1) / block_size, cols, 1); -+ bf16_add_row_bias_round_to_f32<<>>(src, bias, dst, rows); -+} -+ -+template -+static __global__ void f32_add_bias_round_bf16_to_f32( -+ const float * __restrict__ src, const float * __restrict__ bias, -+ float * __restrict__ dst, int64_t count, -+ int64_t ne0, int64_t ne1, int64_t ne2, -+ int64_t bne0, int64_t bne1, int64_t bne2, int64_t bne3) { -+#if defined(__CUDA_ARCH__) && __CUDA_ARCH__ >= GGML_CUDA_CC_AMPERE -+ const int64_t i = int64_t(blockIdx.x) * blockDim.x + threadIdx.x; -+ if (i >= count) { -+ return; -+ } -+ -+ int64_t coordinate = i; -+ const int64_t i0 = coordinate % ne0; -+ coordinate /= ne0; -+ const int64_t i1 = coordinate % ne1; -+ coordinate /= ne1; -+ const int64_t i2 = coordinate % ne2; -+ const int64_t i3 = coordinate / ne2; -+ const int64_t bias_index = -+ (bne0 == 1 ? 0 : i0) + bne0 * ( -+ (bne1 == 1 ? 0 : i1) + bne1 * ( -+ (bne2 == 1 ? 0 : i2) + bne2 * (bne3 == 1 ? 0 : i3))); -+ -+ float value = __bfloat162float(__float2bfloat16(src[i] + bias[bias_index])); -+ if constexpr (apply_relu) { -+ value = fmaxf(value, 0.0f); -+ } -+ dst[i] = value; -+#else -+ GGML_UNUSED_VARS(src, bias, dst, count, ne0, ne1, ne2, bne0, bne1, bne2, bne3); -+ NO_DEVICE_CODE; -+#endif -+} -+ -+static void launch_f32_add_bias_round_bf16_to_f32( -+ const ggml_tensor * src, const ggml_tensor * bias, ggml_tensor * dst, -+ bool apply_relu, cudaStream_t stream) { -+ constexpr int block_size = 256; -+ const int64_t count = ggml_nelements(src); -+ const int64_t blocks = (count + block_size - 1) / block_size; -+ if (apply_relu) { -+ f32_add_bias_round_bf16_to_f32<<>>( -+ static_cast(src->data), -+ static_cast(bias->data), -+ static_cast(dst->data), count, -+ src->ne[0], src->ne[1], src->ne[2], -+ bias->ne[0], bias->ne[1], bias->ne[2], bias->ne[3]); -+ } else { -+ f32_add_bias_round_bf16_to_f32<<>>( -+ static_cast(src->data), -+ static_cast(bias->data), -+ static_cast(dst->data), count, -+ src->ne[0], src->ne[1], src->ne[2], -+ bias->ne[0], bias->ne[1], bias->ne[2], bias->ne[3]); -+ } -+} -+ - static __global__ void bf16_add_row_bias_silu_to_bf16( - const nv_bfloat16 * __restrict__ src, const float * __restrict__ bias, - nv_bfloat16 * __restrict__ dst, int64_t rows) { -@@ -2245,7 +2329,8 @@ static void ggml_cuda_mul_mat_batched_cublas_impl( - ggml_backend_cuda_context & ctx, const ggml_tensor * src0, - const ggml_tensor * src1, ggml_tensor * dst, - const ggml_tensor * row_bias = nullptr, -- ggml_tensor * bf16_silu_dst = nullptr) { -+ ggml_tensor * bf16_silu_dst = nullptr, -+ ggml_tensor * bf16_rounded_f32_dst = nullptr) { - using traits = batched_mul_mat_traits; - using cuda_t = typename traits::cuda_type; - -@@ -2445,6 +2530,14 @@ static void ggml_cuda_mul_mat_batched_cublas_impl( - dst_temp.get(), static_cast(row_bias->data), - static_cast(bf16_silu_dst->data), - ne0, ne_dst / ne0, main_stream); -+ } else if (bf16_rounded_f32_dst != nullptr) { -+ GGML_ASSERT(bf16_rounded_f32_dst->type == GGML_TYPE_F32); -+ GGML_ASSERT(ggml_is_contiguous(bf16_rounded_f32_dst)); -+ GGML_ASSERT(ggml_are_same_shape(dst, bf16_rounded_f32_dst)); -+ launch_bf16_add_row_bias_round_to_f32( -+ dst_temp.get(), static_cast(row_bias->data), -+ static_cast(bf16_rounded_f32_dst->data), -+ ne0, ne_dst / ne0, main_stream); - } else { - launch_bf16_to_f32_add_row_bias( - dst_temp.get(), static_cast(row_bias->data), dst_ddf, -@@ -2495,6 +2588,15 @@ static void ggml_cuda_mul_mat_bf16_row_bias_silu( - ctx, src0, src1, dst, row_bias, bf16_silu_dst); - } - -+static void ggml_cuda_mul_mat_bf16_row_bias_round_to_f32( -+ ggml_backend_cuda_context & ctx, const ggml_tensor * src0, -+ const ggml_tensor * src1, const ggml_tensor * row_bias, -+ ggml_tensor * dst, ggml_tensor * bf16_rounded_f32_dst) { -+ GGML_ASSERT(src0->type == GGML_TYPE_BF16); -+ ggml_cuda_mul_mat_batched_cublas_impl( -+ ctx, src0, src1, dst, row_bias, nullptr, bf16_rounded_f32_dst); -+} -+ - static bool ggml_cuda_should_fuse_mul_mat(const ggml_tensor * ffn_up, - const ggml_tensor * ffn_gate, - const ggml_tensor * glu, -@@ -4411,6 +4513,46 @@ static int ggml_cuda_try_fuse(ggml_backend_cuda_context * cuda_ctx, ggml_cgraph - } - } - -+ // Shared BF16 projection + broadcast F32 row bias + BF16 output -+ // semantics. cuBLAS already materializes a BF16 GEMM result; add the -+ // bias and write the rounded value directly to the graph's F32 carrier. -+ if (ggml_can_fuse(cgraph, i, -+ { GGML_OP_MUL_MAT, GGML_OP_ADD, GGML_OP_CPY, GGML_OP_CPY })) { -+ ggml_tensor * mm_node = cgraph->nodes[i]; -+ ggml_tensor * bias_node = cgraph->nodes[i + 1]; -+ ggml_tensor * bf16_node = cgraph->nodes[i + 2]; -+ ggml_tensor * rounded_node = cgraph->nodes[i + 3]; -+ const ggml_tensor * row_bias = -+ bias_node->src[0] == mm_node ? bias_node->src[1] : -+ bias_node->src[1] == mm_node ? bias_node->src[0] : nullptr; -+ const ggml_tensor * weight = mm_node->src[0]; -+ const ggml_tensor * input = mm_node->src[1]; -+ const int64_t cols = mm_node->ne[1] * mm_node->ne[2] * mm_node->ne[3]; -+ const bool eligible = -+ native_bf16 && row_bias != nullptr && bf16_node->src[0] == bias_node && -+ rounded_node->src[0] == bf16_node && -+ weight->type == GGML_TYPE_BF16 && -+ (input->type == GGML_TYPE_F32 || input->type == GGML_TYPE_BF16) && -+ mm_node->type == GGML_TYPE_F32 && bias_node->type == GGML_TYPE_F32 && -+ bf16_node->type == GGML_TYPE_BF16 && rounded_node->type == GGML_TYPE_F32 && -+ row_bias->type == GGML_TYPE_F32 && ggml_is_contiguous(row_bias) && -+ ggml_is_contiguous(bias_node) && ggml_is_contiguous(bf16_node) && -+ ggml_is_contiguous(rounded_node) && -+ ggml_nelements(row_bias) == mm_node->ne[0] && -+ ggml_are_same_shape(mm_node, rounded_node) && -+ weight->ne[2] == 1 && weight->ne[3] == 1 && -+ input->ne[2] * input->ne[3] > 1 && cols <= 65535 && -+ !ggml_is_transposed(weight) && !ggml_is_transposed(input) && -+ !ggml_backend_buft_is_cuda_split(weight->buffer->buft); -+ const int output_index = i + 3; -+ if (eligible && ggml_cuda_check_fusion_memory_ranges( -+ cgraph, i, 4, &output_index, 1)) { -+ ggml_cuda_mul_mat_bf16_row_bias_round_to_f32( -+ *cuda_ctx, weight, input, row_bias, bias_node, rounded_node); -+ return 3; -+ } -+ } -+ - // Shared BF16 projection + broadcast F32 row bias + SiLU + BF16 cast. - // Feed-forward linear2 consumes the rounded BF16 activation directly, so - // avoid materializing the intermediate F32 biased activation entirely. -@@ -4474,6 +4616,58 @@ static int ggml_cuda_try_fuse(ggml_backend_cuda_context * cuda_ctx, ggml_cgraph - } - } - -+ // Convolution epilogue: add a broadcast bias and apply the model's BF16 -+ // output boundary in one pass. ReLU is folded in when it immediately -+ // consumes that rounded output. The graph retains F32 carrier tensors for -+ // operators that do not natively accept BF16, but every stored value is -+ // exactly BF16-representable. -+ bool round_with_relu = false; -+ bool round_pattern = false; -+ if (ggml_can_fuse(cgraph, i, -+ { GGML_OP_ADD, GGML_OP_CPY, GGML_OP_CPY, GGML_OP_UNARY })) { -+ ggml_tensor * relu_node = cgraph->nodes[i + 3]; -+ round_with_relu = ggml_get_unary_op(relu_node) == GGML_UNARY_OP_RELU; -+ round_pattern = round_with_relu; -+ } -+ if (!round_pattern && -+ ggml_can_fuse(cgraph, i, { GGML_OP_ADD, GGML_OP_CPY, GGML_OP_CPY })) { -+ round_pattern = true; -+ } -+ if (round_pattern) { -+ ggml_tensor * add_node = cgraph->nodes[i]; -+ ggml_tensor * bf16_node = cgraph->nodes[i + 1]; -+ ggml_tensor * rounded_node = cgraph->nodes[i + 2]; -+ ggml_tensor * output_node = round_with_relu ? cgraph->nodes[i + 3] : rounded_node; -+ const ggml_tensor * activation = -+ ggml_are_same_shape(add_node, add_node->src[0]) ? add_node->src[0] : -+ ggml_are_same_shape(add_node, add_node->src[1]) ? add_node->src[1] : nullptr; -+ const ggml_tensor * bias = -+ activation == add_node->src[0] ? add_node->src[1] : -+ activation == add_node->src[1] ? add_node->src[0] : nullptr; -+ bool broadcast_ok = bias != nullptr; -+ for (int d = 0; broadcast_ok && d < GGML_MAX_DIMS; ++d) { -+ broadcast_ok = bias->ne[d] == 1 || bias->ne[d] == activation->ne[d]; -+ } -+ const bool eligible = -+ native_bf16 && activation != nullptr && bias != nullptr && -+ bf16_node->src[0] == add_node && rounded_node->src[0] == bf16_node && -+ (!round_with_relu || output_node->src[0] == rounded_node) && -+ activation->type == GGML_TYPE_F32 && bias->type == GGML_TYPE_F32 && -+ add_node->type == GGML_TYPE_F32 && bf16_node->type == GGML_TYPE_BF16 && -+ rounded_node->type == GGML_TYPE_F32 && output_node->type == GGML_TYPE_F32 && -+ broadcast_ok && ggml_is_contiguous(activation) && ggml_is_contiguous(bias) && -+ ggml_is_contiguous(add_node) && ggml_is_contiguous(bf16_node) && -+ ggml_is_contiguous(rounded_node) && ggml_is_contiguous(output_node) && -+ ggml_are_same_shape(activation, output_node); -+ const int output_index = i + (round_with_relu ? 3 : 2); -+ if (eligible && ggml_cuda_check_fusion_memory_ranges( -+ cgraph, i, round_with_relu ? 4 : 3, &output_index, 1)) { -+ launch_f32_add_bias_round_bf16_to_f32( -+ activation, bias, output_node, round_with_relu, cuda_ctx->stream()); -+ return round_with_relu ? 3 : 2; -+ } -+ } -+ - // Shared BF16 projection + broadcast F32 row bias. The batched-cuBLAS - // path already materializes BF16 GEMM output before converting it to the - // graph's F32 activation. Add the bias during that conversion rather than -diff --git a/src/ggml-cuda/im2col.cu b/src/ggml-cuda/im2col.cu -index 28c79ab4..c2cc16fe 100644 ---- a/src/ggml-cuda/im2col.cu -+++ b/src/ggml-cuda/im2col.cu -@@ -1,4 +1,5 @@ - #include "im2col.cuh" -+#include "convert.cuh" - - #define MAX_GRIDDIM_Y 65535 - #define MAX_GRIDDIM_Z 65535 -@@ -31,10 +32,10 @@ static __global__ void im2col_kernel( - ((in * OH + ioh) * OW + iow) * IC_KH_KW + iic * KH_KW + ikh * KW + ikw; - - if (iih < 0 || iih >= IH || iiw < 0 || iiw >= IW) { -- dst[offset_dst] = 0.0f; -+ dst[offset_dst] = ggml_cuda_cast(0.0f); - } else { - const int64_t offset_src = iic * IC_IH_IW + in * IH_IW; -- dst[offset_dst] = x[offset_src + iih * IW + iiw]; -+ dst[offset_dst] = ggml_cuda_cast(x[offset_src + iih * IW + iiw]); - } - } - } -@@ -75,6 +76,14 @@ static void im2col_cuda_f32(const float * x, float * dst, - im2col_cuda(x, dst, IW, IH, OW, OH, KW, KH, IC, N, IC_IH_IW, IH_IW, s0, s1, p0, p1, d0, d1, stream); - } - -+static void im2col_cuda_bf16(const float * x, nv_bfloat16 * dst, -+ int64_t IW, int64_t IH, int64_t OW, int64_t OH, int64_t KW, int64_t KH, int64_t IC, -+ int64_t N, int64_t IC_IH_IW, int64_t IH_IW, -+ int s0,int s1,int p0,int p1,int d0,int d1, cudaStream_t stream) { -+ -+ im2col_cuda(x, dst, IW, IH, OW, OH, KW, KH, IC, N, IC_IH_IW, IH_IW, s0, s1, p0, p1, d0, d1, stream); -+} -+ - void ggml_cuda_op_im2col(ggml_backend_cuda_context & ctx, ggml_tensor * dst) { - const ggml_tensor * src0 = dst->src[0]; - const ggml_tensor * src1 = dst->src[1]; -@@ -83,7 +92,7 @@ void ggml_cuda_op_im2col(ggml_backend_cuda_context & ctx, ggml_tensor * dst) { - cudaStream_t stream = ctx.stream(); - - GGML_ASSERT(src1->type == GGML_TYPE_F32); -- GGML_ASSERT( dst->type == GGML_TYPE_F16 || dst->type == GGML_TYPE_F32); -+ GGML_ASSERT(dst->type == GGML_TYPE_F16 || dst->type == GGML_TYPE_BF16 || dst->type == GGML_TYPE_F32); - - const int32_t s0 = ((const int32_t*)(dst->op_params))[0]; - const int32_t s1 = ((const int32_t*)(dst->op_params))[1]; -@@ -108,8 +117,10 @@ void ggml_cuda_op_im2col(ggml_backend_cuda_context & ctx, ggml_tensor * dst) { - const int64_t N = src1->ne[is_2D ? 3 : 2]; - const int64_t IH_IW = src1->nb[is_2D ? 3 : 2] / 4; // nb is byte offset, src is type float32 - -- if(dst->type == GGML_TYPE_F16) { -+ if (dst->type == GGML_TYPE_F16) { - im2col_cuda_f16(src1_d, (half *) dst_d, IW, IH, OW, OH, KW, KH, IC, N, IC_IH_IW, IH_IW, s0, s1, p0, p1, d0, d1, stream); -+ } else if (dst->type == GGML_TYPE_BF16) { -+ im2col_cuda_bf16(src1_d, (nv_bfloat16 *) dst_d, IW, IH, OW, OH, KW, KH, IC, N, IC_IH_IW, IH_IW, s0, s1, p0, p1, d0, d1, stream); - } else { - im2col_cuda_f32(src1_d, (float *) dst_d, IW, IH, OW, OH, KW, KH, IC, N, IC_IH_IW, IH_IW, s0, s1, p0, p1, d0, d1, stream); - } diff --git a/ggml-patches/0021-half-snake-fusion-aliasing-guard.patch b/ggml-patches/0021-half-snake-fusion-aliasing-guard.patch deleted file mode 100644 index 8be046d..0000000 --- a/ggml-patches/0021-half-snake-fusion-aliasing-guard.patch +++ /dev/null @@ -1,32 +0,0 @@ -diff --git a/src/ggml-cuda/ggml-cuda.cu b/src/ggml-cuda/ggml-cuda.cu ---- a/src/ggml-cuda/ggml-cuda.cu -+++ b/src/ggml-cuda/ggml-cuda.cu -@@ -4162,6 +4162,28 @@ - return 0; - } - -+ // The graph allocator planned these buffers for the *unfused* sequence, in which the -+ // parent activation dies at the leaky_relu and its memory may be recycled for the -+ // concat output -- the unfused concat reads the materialised add/leaky_relu, not the -+ // parent. The fused kernel still reads the parent while writing the concat, so an -+ // output overlapping an input at a different offset races. Fuse only when the output -+ // misses both inputs entirely, or aliases them exactly (each thread then reads and -+ // writes one and the same address). -+ const char * dst_beg = (const char *) concat->data; -+ const char * dst_end = dst_beg + ggml_nbytes(concat); -+ const char * snk_beg = (const char *) x_snake->data; -+ const char * snk_end = snk_beg + ggml_nbytes(x_snake); -+ const char * lrl_beg = (const char *) view_l->data; -+ const char * lrl_end = lrl_beg + ggml_nbytes(view_l); -+ -+ const bool disjoint = (snk_end <= dst_beg || dst_end <= snk_beg) && -+ (lrl_end <= dst_beg || dst_end <= lrl_beg); -+ const bool exact_inplace = dst_beg == snk_beg && lrl_beg == snk_end && dst_end == lrl_end; -+ -+ if (!disjoint && !exact_inplace) { -+ return 0; -+ } -+ - ggml_cuda_op_half_snake_fused(*cuda_ctx, x_snake, view_l, alpha, inv_b, lrelu, concat); - return 7; - } diff --git a/ggml-patches/0022-cuda-q8-gelu-fusion.patch b/ggml-patches/0022-cuda-q8-gelu-fusion.patch deleted file mode 100644 index 2035469..0000000 --- a/ggml-patches/0022-cuda-q8-gelu-fusion.patch +++ /dev/null @@ -1,294 +0,0 @@ ---- a/src/ggml-cuda/ggml-cuda.cu -+++ b/src/ggml-cuda/ggml-cuda.cu -@@ -4758,6 +4758,68 @@ - } - } - -+ // Q8 projection + broadcast bias + exact GELU_ERF, optionally cast to F16. -+ // cuBLASLt's GELU epilogue is approximate, so combine bias, erf, and cast -+ // in one post-GEMM kernel. Linear's bias add is an -+ // in-place view of the matmul output, so use subgraph validation rather -+ // than the linear-chain helper that rejects intermediate views. -+ static const bool disable_q8_gelu = []() { -+ const char * value = getenv("GGML_SKINNY_Q8_GELU"); -+ return value != nullptr && value[0] == '0'; -+ }(); -+ if (!disable_q8_gelu && ggml_can_fuse_subgraph( -+ cgraph, i, -+ { GGML_OP_MUL_MAT, GGML_OP_ADD, GGML_OP_UNARY, GGML_OP_CPY }, -+ { i + 3 })) { -+ ggml_tensor * mm_node = cgraph->nodes[i]; -+ ggml_tensor * bias_node = cgraph->nodes[i + 1]; -+ ggml_tensor * gelu_node = cgraph->nodes[i + 2]; -+ ggml_tensor * cast_node = cgraph->nodes[i + 3]; -+ const ggml_tensor * bias = -+ bias_node->src[0] == mm_node ? bias_node->src[1] : -+ bias_node->src[1] == mm_node ? bias_node->src[0] : nullptr; -+ if (bias != nullptr && gelu_node->src[0] == bias_node && -+ cast_node->src[0] == gelu_node && -+ ggml_get_unary_op(gelu_node) == GGML_UNARY_OP_GELU_ERF && -+ mm_node->type == GGML_TYPE_F32 && bias_node->type == GGML_TYPE_F32 && -+ gelu_node->type == GGML_TYPE_F32 && cast_node->type == GGML_TYPE_F16 && -+ bias->type == GGML_TYPE_F32 && ggml_is_contiguous(bias) && -+ ggml_nelements(bias) == mm_node->ne[0] && -+ bias->ne[0] == mm_node->ne[0] && -+ ggml_is_contiguous(cast_node) && -+ ggml_are_same_shape(mm_node, cast_node) && -+ ggml_cuda_skinny_q8_gelu_supported( -+ mm_node->src[0], mm_node->src[1], mm_node)) { -+ ggml_cuda_mul_mat_skinny_q8_bias_gelu( -+ *cuda_ctx, mm_node->src[0], mm_node->src[1], bias, cast_node); -+ return 3; -+ } -+ } -+ if (!disable_q8_gelu && ggml_can_fuse_subgraph( -+ cgraph, i, { GGML_OP_MUL_MAT, GGML_OP_ADD, GGML_OP_UNARY }, { i + 2 })) { -+ ggml_tensor * mm_node = cgraph->nodes[i]; -+ ggml_tensor * bias_node = cgraph->nodes[i + 1]; -+ ggml_tensor * gelu_node = cgraph->nodes[i + 2]; -+ const ggml_tensor * bias = -+ bias_node->src[0] == mm_node ? bias_node->src[1] : -+ bias_node->src[1] == mm_node ? bias_node->src[0] : nullptr; -+ if (bias != nullptr && gelu_node->src[0] == bias_node && -+ ggml_get_unary_op(gelu_node) == GGML_UNARY_OP_GELU_ERF && -+ mm_node->type == GGML_TYPE_F32 && bias_node->type == GGML_TYPE_F32 && -+ gelu_node->type == GGML_TYPE_F32 && -+ bias->type == GGML_TYPE_F32 && ggml_is_contiguous(bias) && -+ ggml_nelements(bias) == mm_node->ne[0] && -+ bias->ne[0] == mm_node->ne[0] && -+ ggml_is_contiguous(gelu_node) && -+ ggml_are_same_shape(mm_node, gelu_node) && -+ ggml_cuda_skinny_q8_gelu_supported( -+ mm_node->src[0], mm_node->src[1], mm_node)) { -+ ggml_cuda_mul_mat_skinny_q8_bias_gelu( -+ *cuda_ctx, mm_node->src[0], mm_node->src[1], bias, gelu_node); -+ return 2; -+ } -+ } -+ - // skinny-q8 GEMM + row-vector bias: the upstream MUL_MAT+ADD fusion below - // requires a same-shape add, so the classic broadcast Linear bias - // ([M] over [M,N]) never qualifies — fold it into the skinny GEMM's ---- a/src/ggml-cuda/skinny-q8.cu -+++ b/src/ggml-cuda/skinny-q8.cu -@@ -410,6 +410,17 @@ - } - dst[i] = value; - } -+ -+template -+static __global__ void skq8_bias_gelu_erf( -+ const float * src, const float * bias, T * dst, int m, int64_t count) { -+ const int64_t i = int64_t(blockIdx.x) * blockDim.x + threadIdx.x; -+ if (i >= count) { -+ return; -+ } -+ const float x = src[i] + bias[i % m]; -+ dst[i] = (T) (0.5f * x * (1.0f + erff(x * 0.7071067811865475f))); -+} - #endif - - // Repacked-weight cache. Keyed by the weight tensor's device pointer; entries -@@ -546,9 +557,21 @@ - #endif - } - -+bool ggml_cuda_skinny_q8_gelu_supported( -+ const ggml_tensor * src0, const ggml_tensor * src1, const ggml_tensor * dst) { -+#if defined(GGML_CUDA_SKINNY_Q8_CUBLAS_F16) -+ const int64_t total_n = ggml_nelements(src1) / src1->ne[0]; -+ return ggml_cuda_skinny_q8_supported(src0, src1, dst) && -+ skq8_cublas_f16_enabled_for_n(total_n); -+#else -+ GGML_UNUSED_VARS(src0, src1, dst); -+ return false; -+#endif -+} -+ - static void skq8_run( - ggml_backend_cuda_context & ctx, const ggml_tensor * src0, const ggml_tensor * src1, -- const ggml_tensor * bias, const ggml_tensor * residual, ggml_tensor * dst) { -+ const ggml_tensor * bias, const ggml_tensor * residual, bool gelu, ggml_tensor * dst) { - const int64_t M = src0->ne[1]; - const int64_t K = src0->ne[0]; - const int64_t N = ggml_nelements(src1) / src1->ne[0]; -@@ -656,6 +679,11 @@ - const bool residual_in_place = - residual != nullptr && residual->data == dst->data; - const float beta = residual_in_place ? 1.0f : 0.0f; -+ ggml_cuda_pool_alloc gelu_input(ctx.pool()); -+ float * gemm_dst = (float *) dst->data; -+ if (gelu && dst->type != GGML_TYPE_F32) { -+ gemm_dst = gelu_input.alloc((size_t) M * N); -+ } - cublasHandle_t handle = ctx.cublas_handle(); - CUBLAS_CHECK(cublasSetStream(handle, stream)); - CUBLAS_CHECK(cublasGemmEx( -@@ -663,8 +691,21 @@ - (int) M, (int) N, (int) K, - &alpha, planes.f16, CUDA_R_16F, (int) K, - activation, CUDA_R_16F, (int) K, -- &beta, dst->data, CUDA_R_32F, (int) M, -+ &beta, gemm_dst, CUDA_R_32F, (int) M, - CUBLAS_COMPUTE_32F, CUBLAS_GEMM_DEFAULT_TENSOR_OP)); -+ if (gelu) { -+ GGML_ASSERT(bias != nullptr && residual == nullptr); -+ const int64_t count = M * N; -+ if (dst->type == GGML_TYPE_F16) { -+ skq8_bias_gelu_erf<<<(count + 255) / 256, 256, 0, stream>>>( -+ gemm_dst, (const float *) bias->data, (half *) dst->data, (int) M, count); -+ } else { -+ GGML_ASSERT(dst->type == GGML_TYPE_F32); -+ skq8_bias_gelu_erf<<<(count + 255) / 256, 256, 0, stream>>>( -+ gemm_dst, (const float *) bias->data, (float *) dst->data, (int) M, count); -+ } -+ return; -+ } - if (bias != nullptr || (residual != nullptr && !residual_in_place)) { - const int64_t count = M * N; - skq8_add_epilogue_f32<<<(count + 255) / 256, 256, 0, stream>>>( -@@ -689,6 +730,7 @@ - } - GGML_ASSERT(src1->type == GGML_TYPE_F32); - GGML_ASSERT(residual == nullptr); -+ GGML_ASSERT(!gelu); - #endif - - // Quantize activations into the zero-padded buffer (all columns, once — -@@ -745,17 +787,23 @@ - void ggml_cuda_mul_mat_skinny_q8( - ggml_backend_cuda_context & ctx, const ggml_tensor * src0, const ggml_tensor * src1, - ggml_tensor * dst) { -- skq8_run(ctx, src0, src1, nullptr, nullptr, dst); -+ skq8_run(ctx, src0, src1, nullptr, nullptr, false, dst); - } - - void ggml_cuda_mul_mat_skinny_q8_bias( - ggml_backend_cuda_context & ctx, const ggml_tensor * src0, const ggml_tensor * src1, - const ggml_tensor * bias, ggml_tensor * dst) { -- skq8_run(ctx, src0, src1, bias, nullptr, dst); -+ skq8_run(ctx, src0, src1, bias, nullptr, false, dst); - } - - void ggml_cuda_mul_mat_skinny_q8_bias_residual( - ggml_backend_cuda_context & ctx, const ggml_tensor * src0, const ggml_tensor * src1, - const ggml_tensor * bias, const ggml_tensor * residual, ggml_tensor * dst) { -- skq8_run(ctx, src0, src1, bias, residual, dst); -+ skq8_run(ctx, src0, src1, bias, residual, false, dst); -+} -+ -+void ggml_cuda_mul_mat_skinny_q8_bias_gelu( -+ ggml_backend_cuda_context & ctx, const ggml_tensor * src0, const ggml_tensor * src1, -+ const ggml_tensor * bias, ggml_tensor * dst) { -+ skq8_run(ctx, src0, src1, bias, nullptr, true, dst); - } ---- a/src/ggml-cuda/skinny-q8.cuh -+++ b/src/ggml-cuda/skinny-q8.cuh -@@ -10,6 +10,9 @@ - bool ggml_cuda_skinny_q8_residual_supported( - const ggml_tensor * src0, const ggml_tensor * src1, const ggml_tensor * dst); - -+bool ggml_cuda_skinny_q8_gelu_supported( -+ const ggml_tensor * src0, const ggml_tensor * src1, const ggml_tensor * dst); -+ - void ggml_cuda_mul_mat_skinny_q8( - ggml_backend_cuda_context & ctx, const ggml_tensor * src0, const ggml_tensor * src1, - ggml_tensor * dst); -@@ -23,3 +26,7 @@ - void ggml_cuda_mul_mat_skinny_q8_bias_residual( - ggml_backend_cuda_context & ctx, const ggml_tensor * src0, const ggml_tensor * src1, - const ggml_tensor * bias, const ggml_tensor * residual, ggml_tensor * dst); -+ -+void ggml_cuda_mul_mat_skinny_q8_bias_gelu( -+ ggml_backend_cuda_context & ctx, const ggml_tensor * src0, const ggml_tensor * src1, -+ const ggml_tensor * bias, ggml_tensor * dst); ---- a/tests/test-backend-ops.cpp -+++ b/tests/test-backend-ops.cpp -@@ -5744,6 +5744,67 @@ - } - }; - -+// Zero projection weights isolate GELU_ERF semantics from quantization error. -+// Biases span the nonlinear region, where a tanh GELU substitution is detectable. -+// Uniform weights check that the fused epilogue reads the GEMM result. -+struct test_q8_gelu_erf_fusion : public test_case { -+ const ggml_type input_type; -+ const ggml_type output_type; -+ const int64_t batch; -+ const bool preserve_intermediate; -+ const bool zero_weights; -+ -+ test_q8_gelu_erf_fusion(ggml_type input_type, ggml_type output_type, -+ int64_t batch, bool preserve_intermediate, bool zero_weights = true) -+ : input_type(input_type), output_type(output_type), batch(batch), -+ preserve_intermediate(preserve_intermediate), zero_weights(zero_weights) {} -+ -+ std::string vars() override { -+ return VARS_TO_STR5(input_type, output_type, batch, preserve_intermediate, zero_weights); -+ } -+ -+ std::string op_desc(ggml_tensor * t) override { -+ GGML_UNUSED(t); -+ return "Q8_GELU_ERF_FUSION"; -+ } -+ -+ bool run_whole_graph() override { return true; } -+ double max_nmse_err() override { return zero_weights ? 1e-12 : 5e-4; } -+ -+ ggml_tensor * build_graph(ggml_context * ctx) override { -+ auto * weight = ggml_new_tensor_2d(ctx, GGML_TYPE_Q8_0, 128, 128); -+ ggml_set_name(weight, "encoder.gelu_test.weight"); -+ auto * input = ggml_new_tensor_3d(ctx, input_type, 128, 16, batch); -+ auto * bias = ggml_new_tensor_1d(ctx, GGML_TYPE_F32, 128); -+ ggml_set_name(bias, "gelu_test.bias"); -+ auto * biased = ggml_add_inplace(ctx, ggml_mul_mat(ctx, weight, input), bias); -+ if (preserve_intermediate) { -+ ggml_set_output(biased); -+ } -+ auto * out = ggml_gelu_erf(ctx, biased); -+ return output_type == GGML_TYPE_F16 ? ggml_cast(ctx, out, output_type) : out; -+ } -+ -+ void initialize_tensors(ggml_context * ctx) override { -+ for (auto * t = ggml_get_first_tensor(ctx); t; t = ggml_get_next_tensor(ctx, t)) { -+ if (t->view_src != nullptr) { -+ continue; -+ } -+ if (t->type == GGML_TYPE_Q8_0 && zero_weights) { -+ ggml_backend_tensor_memset(t, 0, 0, ggml_nbytes(t)); -+ } else if (strcmp(t->name, "gelu_test.bias") == 0) { -+ std::vector values(128); -+ for (int i = 0; i < 128; ++i) { -+ values[i] = -3.0f + 6.0f * i / 127; -+ } -+ ggml_backend_tensor_set(t, values.data(), 0, ggml_nbytes(t)); -+ } else { -+ init_tensor_uniform(t); -+ } -+ } -+ } -+}; -+ - struct test_mul_mat_vec_fusion : public test_case { - const ggml_type type; - const ggml_glu_op glu_op; -@@ -8537,6 +8598,19 @@ - 256, 16, 16, {ne2, 1}, {1, 1})); - } - -+ for (auto output_type : {GGML_TYPE_F32, GGML_TYPE_F16}) { -+ for (int64_t batch : {1, 8, 64, 128, 256}) { -+ for (bool preserve : {false, true}) { -+ test_cases.emplace_back(new test_q8_gelu_erf_fusion( -+ GGML_TYPE_F32, output_type, batch, preserve)); -+ } -+ } -+ for (int64_t batch : {1, 64}) { -+ test_cases.emplace_back(new test_q8_gelu_erf_fusion( -+ GGML_TYPE_F32, output_type, batch, false, /*zero_weights=*/false)); -+ } -+ } -+ - // add_id - for (ggml_type type_a : {GGML_TYPE_F32}) { - for (ggml_type type_b : {GGML_TYPE_F32}) { diff --git a/ggml-patches/0023-cuda-skinny-q8-cache-lifetime.patch b/ggml-patches/0023-cuda-skinny-q8-cache-lifetime.patch deleted file mode 100644 index 616fb25..0000000 --- a/ggml-patches/0023-cuda-skinny-q8-cache-lifetime.patch +++ /dev/null @@ -1,82 +0,0 @@ -diff --git a/src/ggml-cuda/ggml-cuda.cu b/src/ggml-cuda/ggml-cuda.cu ---- a/src/ggml-cuda/ggml-cuda.cu -+++ b/src/ggml-cuda/ggml-cuda.cu -@@ -640,6 +640,8 @@ - - static void ggml_backend_cuda_buffer_free_buffer(ggml_backend_buffer_t buffer) { - ggml_backend_cuda_buffer_context * ctx = (ggml_backend_cuda_buffer_context *)buffer->context; -+ ggml_cuda_set_device(ctx->device); -+ ggml_cuda_skinny_q8_forget(ctx->dev_ptr, buffer->size); - delete ctx; - } - -@@ -744,6 +746,7 @@ - ggml_cuda_set_device(ctx->device); - CUDA_CHECK(cudaMemsetAsync(ctx->dev_ptr, value, buffer->size, cudaStreamPerThread)); - CUDA_CHECK(cudaStreamSynchronize(cudaStreamPerThread)); -+ ggml_cuda_skinny_q8_forget(ctx->dev_ptr, buffer->size); - } - - static const ggml_backend_buffer_i ggml_backend_cuda_buffer_interface = { -diff --git a/src/ggml-cuda/skinny-q8.cu b/src/ggml-cuda/skinny-q8.cu ---- a/src/ggml-cuda/skinny-q8.cu -+++ b/src/ggml-cuda/skinny-q8.cu -@@ -424,8 +424,9 @@ - #endif - - // Repacked-weight cache. Keyed by the weight tensor's device pointer; entries --// live for the process lifetime (model weights are loaded once). Repacking --// happens on first (eager/warmup) use, before any CUDA-graph capture. -+// are dropped when the device buffer holding the weight is freed or cleared, so -+// new weights at the same address are repacked again. Repacking happens on -+// first (eager/warmup) use, before any CUDA-graph capture. - namespace { - struct skq8_planes { - int8_t * qs; -@@ -438,6 +439,32 @@ - std::mutex g_skq8_mutex; - } // namespace - -+void ggml_cuda_skinny_q8_forget(const void * base, size_t size) { -+ const uintptr_t begin = (uintptr_t) base; -+ std::lock_guard lock(g_skq8_mutex); -+ for (auto it = g_skq8_cache.begin(); it != g_skq8_cache.end();) { -+ // Compare as integers: the cache holds pointers into unrelated allocations. -+ const uintptr_t key = (uintptr_t) it->first; -+ if (key < begin || key - begin >= size) { -+ ++it; -+ continue; -+ } -+ const skq8_planes & planes = it->second; -+ // In-place and planar weights alias the tensor; only a separate -+ // (GGML_SKINNY_Q8_INPLACE=0) repack owns its planes. -+ if ((const void *) planes.qs != it->first) { -+ CUDA_CHECK(cudaFree(planes.qs)); -+ CUDA_CHECK(cudaFree(planes.d)); -+ } -+#if defined(GGML_CUDA_SKINNY_Q8_CUBLAS_F16) -+ if (planes.f16 != nullptr) { -+ CUDA_CHECK(cudaFree(planes.f16)); -+ } -+#endif -+ it = g_skq8_cache.erase(it); -+ } -+} -+ - bool ggml_cuda_skinny_q8_supported( - const ggml_tensor * src0, const ggml_tensor * src1, const ggml_tensor * dst) { - static const bool disabled = []() { -diff --git a/src/ggml-cuda/skinny-q8.cuh b/src/ggml-cuda/skinny-q8.cuh ---- a/src/ggml-cuda/skinny-q8.cuh -+++ b/src/ggml-cuda/skinny-q8.cuh -@@ -4,6 +4,10 @@ - - // Skinny-N (9..64 cols) Q8_0 x F32 GEMM specialized for streaming encoders. - // See skinny-q8.cu for the design notes. -+ -+// Drop cached planes for weights in a device allocation being freed or cleared. -+void ggml_cuda_skinny_q8_forget(const void * base, size_t size); -+ - bool ggml_cuda_skinny_q8_supported( - const ggml_tensor * src0, const ggml_tensor * src1, const ggml_tensor * dst); - diff --git a/ggml-patches/0026-cuda-backend-graphs-toggle.patch b/ggml-patches/0026-cuda-backend-graphs-toggle.patch deleted file mode 100644 index 9040084..0000000 --- a/ggml-patches/0026-cuda-backend-graphs-toggle.patch +++ /dev/null @@ -1,67 +0,0 @@ -diff --git a/include/ggml-cuda.h b/include/ggml-cuda.h -index 166ef8c1..5bf7c32e 100644 ---- a/include/ggml-cuda.h -+++ b/include/ggml-cuda.h -@@ -30,6 +30,10 @@ GGML_BACKEND_API void * ggml_backend_cuda_get_stream(ggml_backend_t backend); - // Set the CUDA stream priority used by this backend's streams (0 default; negative = higher, - // clamped to the device range). Streams already created are recreated. - GGML_BACKEND_API void ggml_backend_cuda_set_stream_priority(ggml_backend_t backend, int priority); -+// Enable/disable CUDA graph capture and replay for this backend's graph_compute (default on). -+// Side backends running one-shot graphs should turn it off: capture/instantiate hold the driver -+// lock and stall kernel launches issued by other threads. -+GGML_BACKEND_API void ggml_backend_cuda_set_graphs_enabled(ggml_backend_t backend, bool enabled); - - // Returns the backend-owned native CUDA/HIP graph template for a stable GGML graph after its - // normal warm-up/capture has completed. The opaque handle is borrowed and is intended for runtime -diff --git a/src/ggml-cuda/common.cuh b/src/ggml-cuda/common.cuh -index fa09cd6b..895ee058 100644 ---- a/src/ggml-cuda/common.cuh -+++ b/src/ggml-cuda/common.cuh -@@ -1398,6 +1398,10 @@ struct ggml_backend_cuda_context { - cublasHandle_t cublas_handles[GGML_CUDA_MAX_DEVICES] = {nullptr}; - - int curr_stream_no = 0; -+ // Per-backend opt-out of CUDA graph capture/replay (ggml_backend_cuda_set_graphs_enabled): -+ // one-shot graphs computed on a side backend gain nothing from capture, and the capture and -+ // instantiate calls hold the driver lock long enough to stall launches on other threads. -+ bool graphs_enabled = true; - - #ifdef USE_CUDA_GRAPH - // The structural signature separates batch shapes and graph topologies even -diff --git a/src/ggml-cuda/ggml-cuda.cu b/src/ggml-cuda/ggml-cuda.cu -index 0a9d67fe..e2cd19f0 100644 ---- a/src/ggml-cuda/ggml-cuda.cu -+++ b/src/ggml-cuda/ggml-cuda.cu -@@ -5372,7 +5372,7 @@ static enum ggml_status ggml_backend_cuda_graph_compute(ggml_backend_t backend, - ggml_cuda_graph_set_enabled(cuda_ctx, graph_key); - - ggml_cuda_graph * graph = cuda_ctx->cuda_graph(graph_key); -- if (graph->is_enabled()) { -+ if (graph->is_enabled() && cuda_ctx->graphs_enabled) { - const bool graph_compatible = ggml_cuda_graph_check_compability(cgraph); - if (graph_compatible) { - const bool properties_changed = ggml_cuda_graph_update_required(cuda_ctx, cgraph, graph_key); -@@ -5446,7 +5446,7 @@ static void ggml_backend_cuda_graph_optimize(ggml_backend_t backend, ggml_cgraph - - #ifdef USE_CUDA_GRAPH - const ggml_cuda_graph_key graph_key = ggml_cuda_graph_get_key(cgraph); -- const bool use_cuda_graph = ggml_cuda_graph_set_enabled(cuda_ctx, graph_key); -+ const bool use_cuda_graph = ggml_cuda_graph_set_enabled(cuda_ctx, graph_key) && cuda_ctx->graphs_enabled; - #else - const bool use_cuda_graph = false; - GGML_UNUSED(cuda_ctx); -@@ -5739,6 +5739,14 @@ void ggml_backend_cuda_set_stream_priority(ggml_backend_t backend, int priority) - } - } - -+void ggml_backend_cuda_set_graphs_enabled(ggml_backend_t backend, bool enabled) { -+ if (!ggml_backend_is_cuda(backend)) { -+ return; -+ } -+ ggml_backend_cuda_context * cuda_ctx = (ggml_backend_cuda_context *) backend->context; -+ cuda_ctx->graphs_enabled = enabled; -+} -+ - void * ggml_backend_cuda_get_graph_template( - ggml_backend_t backend, const struct ggml_cgraph * cgraph) { - if (!ggml_backend_is_cuda(backend) || cgraph == nullptr) { diff --git a/ggml-patches/0027-cuda-block-reduce-barrier.patch b/ggml-patches/0027-cuda-block-reduce-barrier.patch deleted file mode 100644 index 48c7250..0000000 --- a/ggml-patches/0027-cuda-block-reduce-barrier.patch +++ /dev/null @@ -1,16 +0,0 @@ -diff --git a/src/ggml-cuda/common.cuh b/src/ggml-cuda/common.cuh -index 895ee058..5d731cba 100644 ---- a/src/ggml-cuda/common.cuh -+++ b/src/ggml-cuda/common.cuh -@@ -608,6 +608,11 @@ static __device__ T block_reduce(T val, T * shared_vals) { - assert((block_size <= 1024) && (block_size % WARP_SIZE) == 0); - const int warp_id = threadIdx.x / WARP_SIZE; - const int lane_id = threadIdx.x % WARP_SIZE; -+ // Callers reuse shared_vals across consecutive reductions (e.g. soft_max: max then sum). -+ // Without this barrier a warp that finished the previous reduction early overwrites a -+ // slot another warp is still reading -> racy, scheduling-dependent results (visible when -+ // the SM is shared with a co-resident kernel). -+ __syncthreads(); - if (lane_id == 0) { - shared_vals[warp_id] = val; - } diff --git a/ggml-patches/0028-cuda-conv1d-preactivation.patch b/ggml-patches/0028-cuda-conv1d-preactivation.patch deleted file mode 100644 index 2294f1b..0000000 --- a/ggml-patches/0028-cuda-conv1d-preactivation.patch +++ /dev/null @@ -1,139 +0,0 @@ -diff --git a/src/ggml-cuda/conv1d-fused.cu b/src/ggml-cuda/conv1d-fused.cu -index e66928b3..62306312 100644 ---- a/src/ggml-cuda/conv1d-fused.cu -+++ b/src/ggml-cuda/conv1d-fused.cu -@@ -27,6 +27,7 @@ constexpr int CF_PF = 16; // window elements per thread: CF_BK * 128 / 12 - constexpr int CF_STAGES = 2; - - struct conv1d_fused_params { -+ const half * activated; // optional [group][cache_len+T][cin_pad], prepared once - const float * x; - const float * cache; - const half * w; -@@ -62,6 +63,29 @@ __device__ __forceinline__ void mma_16816(float (&c)[4], const unsigned (&a)[4], - : "r"(a[0]), "r"(a[1]), "r"(a[2]), "r"(a[3]), "r"(b[0]), "r"(b[1])); - } - -+// Pack and activate once per element, instead of once per output-channel tile. -+__global__ void conv1d_activate_pack(conv1d_fused_params p, half * dst) { -+ const int ci = blockIdx.x * blockDim.x + threadIdx.x; -+ const int t = blockIdx.y; -+ const int grp = blockIdx.z; -+ if (ci >= p.cin_pad) return; -+ float v = 0.0f; -+ if (ci < p.cin) { -+ const int cb = (p.shared_input ? 0 : grp * p.cin) + ci; -+ v = t < p.cache_len ? p.cache[(size_t)cb * p.cs + t] -+ : p.x[(size_t)cb * p.xs + t - p.cache_len]; -+ if (p.alpha) { -+ if (ci < p.snake_ch) { -+ const float sn = __sinf(p.alpha[grp * p.snake_ch + ci] * v); -+ v = v + sn * sn * p.inv_b[grp * p.snake_ch + ci]; -+ } else { -+ v = v > 0.f ? v : v * p.slope; -+ } -+ } -+ } -+ dst[((size_t)grp * (p.cache_len + p.T) + t) * p.cin_pad + ci] = __float2half(v); -+} -+ - constexpr int CF_MAX_REGS = 168; - // __maxnreg__ is a CUDA 12.4+ toolkit macro; older toolkits and HIP get plain launch bounds. - #if defined(__maxnreg__) && !defined(GGML_USE_HIP) && !defined(GGML_USE_MUSA) -@@ -70,7 +94,7 @@ constexpr int CF_MAX_REGS = 168; - #define CF_LAUNCH_BOUNDS __launch_bounds__(CF_THREADS) - #endif - --template -+template - // 128 threads x <= 168 registers (21.5K) fit beside a resident MagpieTTS persistent kernel block - // (168 x 256 = 43K) in the 64K-register file, so codec blocks co-schedule instead of serializing. - __global__ void CF_LAUNCH_BOUNDS conv1d_fused_kernel(const conv1d_fused_params p) { -@@ -215,12 +239,32 @@ __global__ void CF_LAUNCH_BOUNDS conv1d_fused_kernel(const conv1d_fused_params p - } - } - }; -+ auto stage_packed = [&](int c, int buf) { -+ if (c < c_end) { -+ for (int e = tid; e < win * 2; e += CF_THREADS) { -+ const int u = e >> 1, seg = e & 1; -+ half * dst = s_win + buf * W_CHUNK + u * CF_LDW + seg * 8; -+ const int t = cache_col0 + t0 + u; -+ if (t < p.cache_len + p.T) { -+ const half * src = p.activated + ((size_t)grp * (p.cache_len + p.T) + t) * p.cin_pad + c * CF_BK + seg * 8; -+ cp_async16(dst, src); -+ } else { -+ *reinterpret_cast(dst) = make_uint4(0,0,0,0); -+ } -+ } -+ } -+ cp_async_commit(); -+ }; - float pf0[CF_PF], pf1[CF_PF]; - for (int q = 0; q < CF_STAGES - 1; ++q) stage_b(c_begin + q, q); -- fetch_window(c_begin + 0, pf0); -- fetch_window(c_begin + 1, pf1); -- stage_act(c_begin + 0, 0); -- stage_act(c_begin + 1, 1); -+ if constexpr (PACKED) { -+ stage_packed(c_begin, 0); -+ } else { -+ fetch_window(c_begin + 0, pf0); -+ fetch_window(c_begin + 1, pf1); -+ stage_act(c_begin + 0, 0); -+ stage_act(c_begin + 1, 1); -+ } - __syncthreads(); - - const int a_row = lane >> 2; -@@ -229,12 +273,18 @@ __global__ void CF_LAUNCH_BOUNDS conv1d_fused_kernel(const conv1d_fused_params p - const int c = c_begin + it; - const int slot = it % CF_STAGES; - const int wbuf = it & 1; -- if (it & 1) store_window(c, pf1, wbuf); else store_window(c, pf0, wbuf); -+ if constexpr (!PACKED) { -+ if (it & 1) store_window(c, pf1, wbuf); else store_window(c, pf0, wbuf); -+ } - cp_async_wait(); - __syncthreads(); - stage_b(c + CF_STAGES - 1, (it + CF_STAGES - 1) % CF_STAGES); -- if (it & 1) fetch_window(c + 2, pf1); else fetch_window(c + 2, pf0); -- stage_act(c + 2, wbuf); // chunk it's params were consumed by store_window above (before the sync) -+ if constexpr (PACKED) { -+ stage_packed(c + 1, wbuf ^ 1); -+ } else { -+ if (it & 1) fetch_window(c + 2, pf1); else fetch_window(c + 2, pf0); -+ stage_act(c + 2, wbuf); -+ } // chunk it's params were consumed by store_window above (before the sync) - const half * bcur = s_b + slot * B_CHUNK; - const half * wcur = s_win + wbuf * W_CHUNK; - for (int k = 0; k < K; ++k) { -@@ -393,6 +443,8 @@ static conv1d_fused_scratch & conv1d_fused_get_scratch(int device) { - CUDA_CHECK(cudaMemset(sc.counters, 0, CF_MAX_TILES * sizeof(int))); - CUDA_CHECK(cudaFuncSetAttribute(conv1d_fused_kernel<64>, cudaFuncAttributeMaxDynamicSharedMemorySize, 96 * 1024)); - CUDA_CHECK(cudaFuncSetAttribute(conv1d_fused_kernel<32>, cudaFuncAttributeMaxDynamicSharedMemorySize, 96 * 1024)); -+ CUDA_CHECK(cudaFuncSetAttribute(conv1d_fused_kernel<64, true>, cudaFuncAttributeMaxDynamicSharedMemorySize, 96 * 1024)); -+ CUDA_CHECK(cudaFuncSetAttribute(conv1d_fused_kernel<32, true>, cudaFuncAttributeMaxDynamicSharedMemorySize, 96 * 1024)); - } - return sc; - } -@@ -471,7 +523,17 @@ void ggml_cuda_op_conv1d_fused(ggml_backend_cuda_context & ctx, ggml_tensor * ds - p.counters = sc.counters; - cudaStream_t stream = ctx.stream(); - dim3 grid(g.tiles_m, g.tiles_n, g.splits); -- if (g.BM == 64) conv1d_fused_kernel<64><<>>(p); -- else conv1d_fused_kernel<32><<>>(p); -+ // Grouped problems with several cout tiles re-activate the same input per tile: pack and -+ // activate it once instead (bit-identical results; grid.y bounds the time extent). -+ if (p.groups > 1 && p.tiles_n > 1 && p.cache_len + p.T <= 65535) { -+ ggml_cuda_pool_alloc activated(ctx.pool(), (size_t)p.groups * (p.cache_len + p.T) * p.cin_pad); -+ p.activated = activated.get(); -+ conv1d_activate_pack<<>>(p,activated.get()); -+ if (g.BM == 64) conv1d_fused_kernel<64, true><<>>(p); -+ else conv1d_fused_kernel<32, true><<>>(p); -+ } else { -+ if (g.BM == 64) conv1d_fused_kernel<64><<>>(p); -+ else conv1d_fused_kernel<32><<>>(p); -+ } - CUDA_CHECK(cudaGetLastError()); - } diff --git a/ggml-patches/0029-cuda-skinny-q8-history-independent-dispatch.patch b/ggml-patches/0029-cuda-skinny-q8-history-independent-dispatch.patch deleted file mode 100644 index a7e6697..0000000 --- a/ggml-patches/0029-cuda-skinny-q8-history-independent-dispatch.patch +++ /dev/null @@ -1,166 +0,0 @@ -diff --git a/src/ggml-cuda/ggml-cuda.cu b/src/ggml-cuda/ggml-cuda.cu -index 77521087..7382e03e 100644 ---- a/src/ggml-cuda/ggml-cuda.cu -+++ b/src/ggml-cuda/ggml-cuda.cu -@@ -2837,8 +2837,8 @@ static void ggml_cuda_mul_mat(ggml_backend_cuda_context & ctx, const ggml_tensor - bool use_batched_cublas_f32 = src0->type == GGML_TYPE_F32; - - if (!split && !bad_padding_clear && ggml_cuda_skinny_q8_supported(src0, src1, dst)) { -- // streaming-ASR specialization: Q8_0 weights x skinny (9..64 col) -- // activations — beats mul_mat_q's LLM-batch tiling at these shapes -+ // streaming-encoder specialization: Q8_0 weights x activations wider -+ // than MMVQ's range; beats mul_mat_q's LLM-batch tiling at skinny N - ggml_cuda_mul_mat_skinny_q8(ctx, src0, src1, dst); - } else if (!split && use_mul_mat_vec_f) { - // the custom F16 vector kernel can be used over batched cuBLAS GEMM -diff --git a/src/ggml-cuda/skinny-q8.cu b/src/ggml-cuda/skinny-q8.cu -index 61a433b9..60dcaddc 100644 ---- a/src/ggml-cuda/skinny-q8.cu -+++ b/src/ggml-cuda/skinny-q8.cu -@@ -23,8 +23,11 @@ - // - // Bit-accuracy: same q8 activation round-trip as mul_mat_q, but a different - // summation order — results are not bit-identical to mmq. Dispatch must --// therefore depend only on the per-sequence matrix shape, never on the outer --// batch size, so singleton and batched execution cannot select different math. -+// therefore depend only on the call's shape, never on process history: a -+// weight takes MMVQ for N <= 8 and this kernel for every wider call, from its -+// first call on, whether or not an earlier call already repacked it. Each -+// output column's sum is independent of N (tiles only add columns), so -+// singleton and batched calls wider than 8 columns compute the same bits. - - #include "skinny-q8.cuh" - #if defined(GGML_CUDA_SKINNY_Q8_CUBLAS_F16) -@@ -43,7 +46,6 @@ - #define SKQ8_KSTEP 128 // k bytes staged per pipeline stage; K % KSTEP == 0 - #define SKQ8_KTB (SKQ8_KSTEP / 32) // q8 blocks per stage - #define SKQ8_STAGES 2 --#define SKQ8_NMAX SKQ8_NPAD - // Shared-memory strides (bytes). KSTEP + 16 keeps the 16B cp.async stores - // aligned while breaking the power-of-two bank pattern on fragment loads. - #define SKQ8_SW (SKQ8_KSTEP + 16) -@@ -393,7 +395,6 @@ static __global__ void skq8_reduce_splitk( - dst[i] = sum; - } - --#if defined(GGML_CUDA_SKINNY_Q8_CUBLAS_F16) - static __global__ void skq8_add_epilogue_f32( - float * __restrict__ dst, const float * __restrict__ bias, - const float * __restrict__ residual, const int m, const int64_t count) { -@@ -411,6 +412,7 @@ static __global__ void skq8_add_epilogue_f32( - dst[i] = value; - } - -+#if defined(GGML_CUDA_SKINNY_Q8_CUBLAS_F16) - template - static __global__ void skq8_bias_gelu_erf( - const float * src, const float * bias, T * dst, int m, int64_t count) { -@@ -551,32 +553,28 @@ bool ggml_cuda_skinny_q8_supported( - if (repacked) { - // The weight was converted IN PLACE to the plane layout — the - // block_q8_0 bytes no longer exist, so every mul_mat on this tensor -- // must come through here (the >64-col path tiles, small N pads). -- // Falling back to mmq/mmvq would silently read garbage: abort loudly -- // instead if a call shape we cannot serve ever appears. -+ // must come through here (skq8_run serves N <= MMVQ_MAX_BATCH_SIZE -+ // with planar MMVQ and tiles wider N). Falling back to mmq/mmvq would -+ // silently read garbage: abort loudly instead if a call shape we -+ // cannot serve ever appears. - GGML_ASSERT(call_ok && "skinny-q8: repacked weight used in an unsupported mul_mat shape"); - return true; - } -- // Keep the accumulation path independent of outer batch size by default. -- // The opt-in mode flattens [K,N,B...] for the tensor-core kernel; use a -- // separate repack allocation with multi-stream schedulers. -- static const bool outer_batch_dispatch = []() { -- const char * e = getenv("GGML_SKINNY_Q8_OUTER_BATCH"); -- return e != nullptr && e[0] != '0'; -- }(); -- const int64_t logical_n = src1->ne[1]; -- const bool logical_eligible = logical_n > MMVQ_MAX_BATCH_SIZE && logical_n <= SKQ8_NMAX; -- const bool outer_eligible = total_n > MMVQ_MAX_BATCH_SIZE; -- const bool eligible = -- tensor_ok && call_ok && (outer_batch_dispatch ? outer_eligible : logical_eligible); -- return eligible; -+ // Same rule as serialized planar weights above, applied from the first -+ // call: the kernel depends only on the call's column count, never on -+ // whether an earlier call already repacked the weight. N <= 8 stays on -+ // MMVQ (planar MMVQ after a repack computes the same bits); every wider -+ // call runs skq8, which repacks on first use. An N cap here would send -+ // wide calls to mmq before the first repack and to skq8 after it, making -+ // outputs depend on which requests the process served earlier. -+ return tensor_ok && call_ok && total_n > MMVQ_MAX_BATCH_SIZE; - } - - bool ggml_cuda_skinny_q8_residual_supported( - const ggml_tensor * src0, const ggml_tensor * src1, const ggml_tensor * dst) { - #if defined(GGML_CUDA_SKINNY_Q8_CUBLAS_F16) - const int64_t total_n = ggml_nelements(src1) / src1->ne[0]; -- return ggml_cuda_skinny_q8_supported(src0, src1, dst) && -+ return total_n > MMVQ_MAX_BATCH_SIZE && ggml_cuda_skinny_q8_supported(src0, src1, dst) && - skq8_cublas_f16_enabled_for_n(total_n); - #else - GGML_UNUSED_VARS(src0, src1, dst); -@@ -588,7 +586,7 @@ bool ggml_cuda_skinny_q8_gelu_supported( - const ggml_tensor * src0, const ggml_tensor * src1, const ggml_tensor * dst) { - #if defined(GGML_CUDA_SKINNY_Q8_CUBLAS_F16) - const int64_t total_n = ggml_nelements(src1) / src1->ne[0]; -- return ggml_cuda_skinny_q8_supported(src0, src1, dst) && -+ return total_n > MMVQ_MAX_BATCH_SIZE && ggml_cuda_skinny_q8_supported(src0, src1, dst) && - skq8_cublas_f16_enabled_for_n(total_n); - #else - GGML_UNUSED_VARS(src0, src1, dst); -@@ -605,8 +603,11 @@ static void skq8_run( - const int64_t KB = K / 32; - - cudaStream_t stream = ctx.stream(); -+ // Only a repacked weight reaches here with N <= MMVQ_MAX_BATCH_SIZE (see -+ // ggml_cuda_skinny_q8_supported); it keeps the MMVQ math it had before. -+ const bool mmvq_cols = N <= MMVQ_MAX_BATCH_SIZE; - #if defined(GGML_CUDA_SKINNY_Q8_CUBLAS_F16) -- const bool use_cublas_f16 = skq8_cublas_f16_enabled_for_n(N); -+ const bool use_cublas_f16 = !mmvq_cols && skq8_cublas_f16_enabled_for_n(N); - #endif - - // Repacked weight planes (create on first use). Default: IN PLACE. The plane -@@ -696,6 +697,26 @@ static void skq8_run( - #endif - } - -+ if (mmvq_cols) { -+ // Serve MMVQ's column range with MMVQ, as before the repack. In-place -+ // (and serialized planar) planes alias the tensor bytes, so read them -+ // through planar MMVQ, which computes the same bits as block MMVQ; a -+ // separate (GGML_SKINNY_Q8_INPLACE=0) repack left the block bytes intact. -+ GGML_ASSERT(src1->type == GGML_TYPE_F32 && residual == nullptr && !gelu); -+ ggml_tensor weight = *src0; -+ if ((const void *) planes.qs == src0->data) { -+ weight.flags |= GGML_TENSOR_FLAG_Q8_PLANAR; -+ } -+ ggml_cuda_mul_mat_vec_q(ctx, &weight, src1, nullptr, dst); -+ if (bias != nullptr) { -+ // separate add, like the unfused mul_mat + add graph -+ const int64_t count = M * N; -+ skq8_add_epilogue_f32<<<(count + 255) / 256, 256, 0, stream>>>( -+ (float *) dst->data, (const float *) bias->data, nullptr, (int) M, count); -+ } -+ return; -+ } -+ - #if defined(GGML_CUDA_SKINNY_Q8_CUBLAS_F16) - // Keep the model artifact Q8, expand immutable weights once, convert only - // the live activation, and retain FP32 accumulation/output. -diff --git a/src/ggml-cuda/skinny-q8.cuh b/src/ggml-cuda/skinny-q8.cuh -index bb3dac78..fa5a9b35 100644 ---- a/src/ggml-cuda/skinny-q8.cuh -+++ b/src/ggml-cuda/skinny-q8.cuh -@@ -2,7 +2,8 @@ - // SPDX-License-Identifier: Apache-2.0 - #include "common.cuh" - --// Skinny-N (9..64 cols) Q8_0 x F32 GEMM specialized for streaming encoders. -+// Skinny-N Q8_0 x F32 GEMM specialized for streaming encoders (claims every -+// call wider than MMVQ's 8 columns on an eligible weight). - // See skinny-q8.cu for the design notes. - - // Drop cached planes for weights in a device allocation being freed or cleared. diff --git a/ggml-patches/README.md b/ggml-patches/README.md deleted file mode 100644 index a204992..0000000 --- a/ggml-patches/README.md +++ /dev/null @@ -1,289 +0,0 @@ -# ggml-patches - -Edge-only changes to the vendored `ggml` submodule, kept as patches so the -submodule stays pinned to clean upstream (`ggml-org/ggml`). Apply them after -checking out / updating the submodule: - -```sh -git submodule update --init ggml -scripts/apply-ggml-patches.sh # applies patches in filename order -``` - -Patched CUDA and Metal builds expect the patches before CMake configuration. -CPU, Vulkan, and stock-CUDA builds do not. `apply-ggml-patches.sh` uses `git -apply`, skips patches that are already applied, and applies new patches in -filename order. Later patches may build on files changed by earlier patches; -0006 carries the dispatch wiring for the ops/kernels introduced by -0001/0003/0005. Docker builds and `scripts/configure.sh` apply the series -automatically; apply it explicitly before a raw CUDA or Metal CMake -configuration. - -Metal requires only `0018-metal-tensor-api-dynamic-k.patch`. The `metal-*` -presets apply the whole series because `apply-ggml-patches.sh` requires all of -them or it fails. Once the submodule moves past upstream ggml `33c9ea5`, drop -0018 as described below and remove `metal-*` from the `case` in -`scripts/configure.sh`. - -## Building against patched vs stock ggml - -`NEMO_SPEECH_GGML_PATCHED` tells the build whether the vendored ggml contains -this patch series. It defaults to `ON` because the ASR encoder directly uses -the fused relative-position attention op from 0001 and the F16 depthwise-conv -behavior from 0004. - -```sh -# Patched ggml (default): apply the patches before configuring CMake. -scripts/apply-ggml-patches.sh -cmake -S . -B build -DGGML_CUDA=ON - -# Pristine upstream ggml: do not apply the patches and opt out explicitly. -cmake -S . -B build -DGGML_CUDA=ON -DNEMO_SPEECH_GGML_PATCHED=OFF -``` - -With `NEMO_SPEECH_GGML_PATCHED=OFF`, the encoder uses stock ggml operations: -unfused relative-position attention and `ggml_conv_1d_dw` im2col lowering. -These paths remain correct across backends but cost latency on CUDA. The -dependent options below default to `ON` only when both `GGML_CUDA` and -`NEMO_SPEECH_GGML_PATCHED` are enabled, and are otherwise forced `OFF`: - -| option | effect when enabled | -|---|---| -| `NEMO_SPEECH_FUSED_RELPOS_ATTN` | emit the fused relative-position attention CUDA op | -| `NEMO_SPEECH_DIRECT_DW_CONV` | use the direct CUDA depthwise-convolution kernel | -| `NEMO_SPEECH_FASTCONFORMER_CUDA_FUSIONS` | emit patched sigmoid-GLU and BF16-fusion graph patterns | - -The options can be disabled independently for correctness or performance -bisection even in a patched build. Other patches optimize ordinary ggml -operations through their own eligibility checks and may still activate when -`NEMO_SPEECH_GGML_PATCHED=OFF` if the patched sources are present. A genuine -stock comparison therefore requires both a pristine ggml checkout and -`NEMO_SPEECH_GGML_PATCHED=OFF`. - -## Patches - -- **0001-fused-relpos-attn.patch** - adds relative-position fused attention for - CUDA, including stride-aware inputs, F16 K/V/P storage, head-merged output, - and the `NEMO_SPEECH_FUSED_RELPOS_ATTN` encoder path. - -- **0002-nvfp4-residual-activations.patch** - keeps the native NVFP4 weight - path while reducing activation-quantization error. Each activation - sub-block is quantized once to FP4, its reconstruction residual is quantized - to a second FP4 block, and both contributions are accumulated by the same - MMQ tile. The correction is restricted to NVFP4; MXFP4 and non-native paths - retain their upstream behavior. Backend correctness tests use the standard - quantized-matmul tolerance rather than the previous relaxed NVFP4 threshold. - -- **0003-norm-mul-add-fusion.patch** - fuses affine LayerNorm with row-vector - scale and optional bias. - -- **0004-conv2d-dw-f16-kernel.patch** - supports F16 weights in the direct - depthwise-convolution CUDA kernel with F32 input and output. - -- **0005-skinny-q8-gemm.patch** - adds a Q8_0 x F32 GEMM for skinny streaming - activations, including planar weights, deterministic K-split reduction, and - an optional bias epilogue. `GGML_SKINNY_Q8` controls dispatch; - `GGML_SKINNY_Q8_INPLACE=0` uses separate repack storage when required by a - multi-stream scheduler. - -- **0006-cuda-dispatch-wiring.patch** - wires fused attention, affine - LayerNorm, skinny-Q8, planar-Q8, and narrow bias/SiLU epilogues into the CUDA - backend. - -- **0007-magpietts-nanocodec.patch** - adds grouped transposed convolution and - Snake for MagpieTTS and NanoCodec, two-column MMVF epilogues for paired CFG, - bounded CUDA graph caching, and CUDA architecture handling. - -- **0008-cublas-bf16-projections.patch** - flattens contiguous outer activation - dimensions into shared-weight cuBLAS GEMMs and folds supported BF16 projection - epilogues into output conversion. - -- **0009-fastconformer-cuda-fusions.patch** - adds sigmoid GLU, Macaron - residual, affine LayerNorm conversion, and BF16 projection fusions for - FastConformer. - -- **0010-cuda-pad-large-batch-grid.patch** - flattens CUDA PAD launches into - `grid.x` so large batch dimensions do not exceed the `grid.z` limit. - -- **0011-cuda-graph-shape-key.patch** - keys cached CUDA graph executables by - graph identity and structural shape so allocator reuse cannot select an - incompatible executable. - -- **0012-cuda-streaming-cache-copies.patch** - adds pitched cache-tail copies - and vectorized indexed state-arena transfers for streaming K/V state. - -- **0013-cuda-cached-f16-cublas.patch** - adds an optional cached-F16/cuBLAS - path for skinny Q8 projections while retaining F32 accumulation and output. - -- **0014-cuda-fused-attention-extensions.patch** - generalizes fused attention - to standard and relative-position modes, persistent circular K/V caches, - active-length state, streaming FastConformer shapes, and Magpie's cached - single-query shape. The `q_len == 1`, `d_k == 64` cached kernel splits the - key range across blocks (about 96 keys per block, up to 8 blocks per head) - with a last-arriving-block merge, so an autoregressive decoder step uses - several SMs per head instead of one. - -- **0015-cuda-ctc-batch-fusions.patch** - adds BatchNorm, transpose/SiLU, - affine LayerNorm, and cached-F16 projection epilogues used by batched - FastConformer and Magpie. - -- **0016-fix-batched-conv1d-layout.patch** - restores batch and output-channel - axes after flattened Conv1D matrix multiplication. - -- **0017-cuda-stream-interop.patch** - exposes borrowed access to the active - CUDA stream and stable graph templates for external graph composition, and - adds `ggml_backend_cuda_set_stream_priority()` so a latency-critical backend - (the MagpieTTS decoder) can run its streams at a higher CUDA stream priority - than a concurrent worker backend (the NanoCodec vocoder). - -- **0018-metal-tensor-api-dynamic-k.patch** - backports upstream ggml `33c9ea5` - (llama/27450), which stops the Metal tensor-API matmul reading `src1` out of - bounds when `K % 32 != 0`. Drop it once the submodule passes that commit, - keeping the `K = 1296` test case. - -- **0019-cuda-graph-dynamic-update.patch** - refreshes CUDA graph node - parameters when a cached graph is replayed so dynamic pointers and launch - geometry do not retain values from an earlier execution. - -- **0020-bf16-convolution.patch** - adds BF16 im2col and direct depthwise - convolution support, then fuses bias, BF16 output rounding, and optional - ReLU epilogues. This preserves the VoiceChat perception stem's native BF16 - behavior without adding standalone conversion kernels. - -- **0021-half-snake-fusion-aliasing-guard.patch** - fixes a read/write race in - the `half_snake` CUDA fusion added by 0007. The graph allocator plans buffers - for the *unfused* node sequence, in which the parent activation's last reader - is the `leaky_relu`, so its memory is free to be recycled for the `concat` - output -- the unfused `concat` reads the materialised add/leaky_relu, never - the parent. The fused kernel still reads that parent while writing the concat, - so wherever the allocator overlapped the two at different offsets, threads - clobbered input that other threads had not yet read. - - Fusion is now skipped unless the output is disjoint from both inputs, or - aliases them exactly -- in which case every thread reads and writes a single - address and is safe. Because it depends on where the allocator happens to - place buffers, this corrupted some graphs and left others alone, which is why - the arithmetic always checked out in isolation. - - On a GB300, a 3-frame NanoCodec decode against its own CPU reference. Before - the guard the corruption depends on allocator placement, so it is a - distribution, not a figure: 15 runs spanned 11.8 to 14.6 dB SNR, median 13.9. - With the guard the decode is stable at 39.5 dB, against 40.2 dB with fusion - disabled outright -- and the guard costs 0.8% throughput where disabling - fusion costs 26%. Covered by - `tests/cpp/tts/test_nanocodec_half_snake_fusion.cpp`. - -- **0022-cuda-q8-gelu-fusion.patch** - fuses the bias add, exact `GELU_ERF`, - and optional F16 cast after the cached-F16 Q8 cuBLAS GEMM into one CUDA - kernel. Active only with `GGML_SKINNY_Q8_CUBLAS_F16=1` on SM80+ GPUs; - `GGML_SKINNY_Q8_GELU=0` disables it. Covered by `test-backend-ops -o - Q8_GELU_ERF_FUSION`. - -- **0023-cuda-skinny-q8-cache-lifetime.patch** - drops skinny-Q8 cache entries, - and any planes they own, when their CUDA weight buffer is freed or cleared. - The cache was keyed only by device address, so weights later allocated at a - reused address were treated as already repacked. - -- **0024-cuda-im2col-1d-tiled.patch** - a tiled, coalesced fast path for 1D - `im2col` (`IH == KH == OH == 1`) that stages a channel-by-time input tile - through shared memory and writes consecutive output rows with consecutive - threads; the generic kernel strides across channels and re-reads every input - `KW` times. Layered on 0020, which also edits `im2col.cu`. - -- **0025-cuda-conv1d-fused.patch** - adds the `ggml_conv1d_fused` and - `ggml_conv1d_fused_grouped` ops: a causal 1D convolution over an F32 - streaming-cache prefix plus the current input, with the snake/leaky-ReLU - activation applied on load, computed as an implicit GEMM with `mma.sync` - tensor-core fragments (F16 operands, F32 accumulation, split-K, fused bias and - residual epilogue) from F16 weights pre-packed as `[K][cout][cin]`. The - grouped form runs several kernel-size branches in one launch. CUDA only: the - op is reported supported only for NVIDIA devices with sm_80+ device code in - the build, the CPU backend reports it unsupported, and other backends reject - it, so callers keep their unfused path. Replaces the im2col + cuBLAS - decomposition in the NanoCodec decoder. - -- **0026-cuda-backend-graphs-toggle.patch** - adds - `ggml_backend_cuda_set_graphs_enabled()`: a per-backend opt-out of CUDA graph - capture/replay for backends that compute one-shot graphs (capture and - instantiate hold the driver lock and stall launches on other threads). -- **0027-cuda-block-reduce-barrier.patch** - `block_reduce()` synchronizes the - block before writing its shared-memory slot: consecutive reductions on the same - buffer (soft_max: max then sum) otherwise race and give scheduling-dependent - results when the SM is shared with a co-resident kernel. - -- **0028-cuda-conv1d-preactivation.patch** - for grouped `ggml_conv1d_fused` - problems with several output-channel tiles, activates and packs the input once - per element into an F16 buffer instead of once per tile, then runs the - convolution on the packed input. Results are bit-identical; the NanoCodec - decoder runs about twice as fast. - -- **0029-cuda-skinny-q8-history-independent-dispatch.patch** - skinny-Q8 - dispatch depends only on the column count (N <= 8: MMVQ, planar after a - repack; wider: skinny-Q8), not on whether an earlier call repacked the - weight, so outputs no longer depend on previously served requests. Removes - `GGML_SKINNY_Q8_OUTER_BATCH`. - -- **0030-cpu-fp16-conversion-im2col.patch** - on x86 with F16C, scalar - FP32 -> FP16 conversions use `vcvtps2ph` instead of the portable - bit-manipulating routine, and `im2col` splits threads by output row instead - of interleaved channels, so threads no longer write interleaved segments of - the same output rows. The conversion matches the portable routine for every - non-NaN FP32 input (checked exhaustively; NaN payloads may differ), and - `im2col` computes the same values in a different thread order. Both speed up - F16 convolutions such as the NanoCodec decoder on CPU. - -## Regenerating after editing ggml - -Several patches touch the same ggml files, so regenerating a patch from the -fully patched submodule can accidentally fold later changes into it. Edit and -diff at the patch's actual point in the series: -Most base files belong to one patch; 0013 and 0014 are explicit layered -exceptions. Do not regenerate 0001 from a fully patched live tree without first -removing the 0014 delta, or the circular-cache extension will be folded into it. - -```sh -# Create a disposable worktree at the pinned upstream commit. -git -C ggml worktree add "$PWD/ggml-patch-work" HEAD -cd ggml-patch-work - -# Apply every patch before the one being edited, then stage that baseline. -target=0007-magpietts-nanocodec.patch -for patch in ../ggml-patches/*.patch; do - [ "$(basename "$patch")" = "$target" ] && break - git apply "$patch" -done -git add -A - -# Apply the target patch, edit it, and capture only its delta. -git apply "../ggml-patches/$target" -# Edit and test the affected files. -git add -N src/ggml-cuda/ # only when the patch adds a new file -git diff --binary > "../ggml-patches/$target" - -# Return to the repository root, check the patch, and remove the worktree. -cd .. -git diff --check -- "ggml-patches/$target" -git -C ggml worktree remove --force "$PWD/ggml-patch-work" -``` - -Adjust paths when the disposable worktree is placed elsewhere. If an edited -patch changes context used by later patches, rebase those later patches in the -same way. - -Patch 0013 is intentionally layered on top of 0005. To regenerate it without -folding the generic skinny-Q8 implementation into the cached-F16 patch, use a -temporary ggml worktree, apply and stage patches 0001 through 0012 as the -baseline, then copy in only the cached-F16 changes and the CUDA architecture -target correction, and run `git diff` against that staged baseline for -`CMakeLists.txt` and `skinny-q8.cu`. - -Patch 0014 is intentionally layered on top of 0001 and 0013. Apply and stage -patches 0001 through 0013 in a temporary worktree, copy the edited attention -API, dispatch, and CUDA implementation into that worktree, then generate 0014 -with `git diff` against the staged baseline. - -Generate patches with `git diff` only (GNU `diff`/editors can strip the -leading space on blank context lines, which `git apply` rejects as corrupt). - -To verify the complete series, apply every patch in order to a fresh worktree at -the pinned ggml commit, then recursively compare that tree with the live patched -submodule and confirm that all files match. diff --git a/kernels/cublas_shim.cu b/kernels/cublas_shim.cu index 4024426..07d1efa 100644 --- a/kernels/cublas_shim.cu +++ b/kernels/cublas_shim.cu @@ -1077,7 +1077,7 @@ k_hgemm_tn_smalln_warp( // large subsampling-conv GEMMs (k-reuse bound) but LOSES ~2.5x on the small // streaming GEMMs (which need parallelism, not reuse) — so launch() picks per // shape: tiled iff the aggregate batched output is large enough to fill the -// GPU (see EDGE_SHIM_TILE_MIN). +// GPU (see use_tiled). #define EDGE_TM 64 #define EDGE_TN 64 #define EDGE_TK 16 @@ -1167,26 +1167,13 @@ k_strided_tiled( inline bool use_tiled(int m, int n, int batch) { // Crossover between the naive (parallelism-bound, small outputs) and the - // tiled (k-reuse-bound, large batched outputs). Overridable for tuning. - static const size_t min_mn = [] { - const char* e = getenv("EDGE_SHIM_TILE_MIN"); - return e ? (size_t)atoll(e) : (size_t)64 * 1024; - }(); - return (size_t)m * (size_t)n * (size_t)batch >= min_mn; -} -inline size_t -wmma_min_mn() { - static const size_t min_mn = [] { - const char* e = getenv("EDGE_SHIM_WMMA_MIN_MN"); - return e ? (size_t)atoll(e) : (size_t)1; - }(); - return min_mn; + // tiled (k-reuse-bound, large batched outputs). + return (size_t)m * (size_t)n * (size_t)batch >= (size_t)64 * 1024; } inline bool use_wmma_tn(int m, int n, int opA, int opB, int ta, int tb, int tc) { return opA == OP_T && opB == OP_N && ta == R_16F && tb == R_16F && - (tc == R_16F || tc == R_32F || tc == R_16BF) && n > 1 && - (size_t)m * (size_t)n >= wmma_min_mn(); + (tc == R_16F || tc == R_32F || tc == R_16BF) && n > 1 && m > 0; } inline bool use_hgemv_tn(int n, int opA, int opB, int ta, int tb, int tc) { diff --git a/llama-patches/0004-mamba2-flat-projections.patch b/llama-patches/0004-mamba2-flat-projections.patch deleted file mode 100644 index 1a581a5..0000000 --- a/llama-patches/0004-mamba2-flat-projections.patch +++ /dev/null @@ -1,41 +0,0 @@ -diff --git a/src/models/mamba-base.cpp b/src/models/mamba-base.cpp -index c37f29c48..2760a28b7 100644 ---- a/src/models/mamba-base.cpp -+++ b/src/models/mamba-base.cpp -@@ -178,13 +178,14 @@ ggml_tensor * llm_build_mamba_base::build_mamba2_layer(llm_graph_input_rs * inp, - ggml_tensor * conv = build_rs(inp, conv_states_all, hparams.n_embd_r(), n_seqs); - conv = ggml_reshape_3d(ctx0, conv, d_conv - 1, d_inner + 2 * n_group * d_state, n_seqs); - -- // {n_embd, n_tokens} => {n_embd, n_seq_tokens, n_seqs} -- cur = ggml_reshape_3d(ctx0, cur, cur->ne[0], n_seq_tokens, n_seqs); -- - // d_in_proj = 2 * self.d_inner + 2 * self.ngroups * self.d_state + self.nheads - -- // {n_embd, d_in_proj} @ {n_embd, n_seq_tokens, n_seqs} => {d_in_proj, n_seq_tokens, n_seqs} -+ // Keep the projection 2D: with a {n_embd, 1, n_seqs} batch the CUDA backend -+ // dispatches a column-batched GEMV for what is a large dense GEMM. -+ // {n_embd, d_in_proj} @ {n_embd, n_tokens} => {d_in_proj, n_tokens} - ggml_tensor * zxBCdt = build_lora_mm(model.layers[il].ssm_in, cur, model.layers[il].ssm_in_s); -+ // {d_in_proj, n_tokens} => {d_in_proj, n_seq_tokens, n_seqs} -+ zxBCdt = ggml_reshape_3d(ctx0, zxBCdt, zxBCdt->ne[0], n_seq_tokens, n_seqs); - - // split the above in three - ggml_tensor * z = ggml_view_4d(ctx0, zxBCdt, head_dim, n_head, n_seq_tokens, n_seqs, head_dim * zxBCdt->nb[0], -@@ -275,15 +276,12 @@ ggml_tensor * llm_build_mamba_base::build_mamba2_layer(llm_graph_input_rs * inp, - y = build_norm(y, model.layers[il].ssm_norm, NULL, LLM_NORM_RMS, il); - } - -- y = ggml_reshape_3d(ctx0, y, d_inner, n_seq_tokens, n_seqs); -+ y = ggml_reshape_2d(ctx0, y, d_inner, n_seq_tokens * n_seqs); - -- // {d_inner, n_embd} @ {d_inner, n_seq_tokens, n_seqs} => {n_embd, n_seq_tokens, n_seqs} -+ // {d_inner, n_embd} @ {d_inner, n_tokens} => {n_embd, n_tokens} - cur = build_lora_mm(model.layers[il].ssm_out, y, model.layers[il].ssm_out_s); - } - -- // {n_embd, n_seq_tokens, n_seqs} => {n_embd, n_tokens} -- cur = ggml_reshape_2d(ctx0, cur, cur->ne[0], n_seq_tokens * n_seqs); - cb(cur, "mamba_out", il); -- - return cur; - } diff --git a/llama-patches/README.md b/llama-patches/README.md deleted file mode 100644 index 1604775..0000000 --- a/llama-patches/README.md +++ /dev/null @@ -1,24 +0,0 @@ -# llama.cpp patches - -Project-specific changes to the pinned `llama.cpp` submodule. Apply them in -filename order after initializing the submodule: - -```sh -git submodule update --init llama.cpp -scripts/apply-llama-patches.sh -``` - -`scripts/configure.sh` and S2S/NMT Docker builds apply the series -automatically. - -## Patches - -- **0001-batch-all-recurrent-outputs.patch** - keeps equal-length recurrent - sequences in one microbatch when all token outputs are requested. -- **0002-enable-nvfp4-quantization.patch** - enables NVFP4 model quantization - and its Q8 fallback path. -- **0003-gemma3-attention-scale.patch** - honors the attention scale stored in - Gemma 3 model metadata, with the upstream default as a fallback. -- **0004-mamba2-flat-projections.patch** - keeps Mamba2 input and output - projections flat so batched CUDA decode dispatches GEMM instead of GEMV. - Backports [llama.cpp PR #27513](https://github.com/ggml-org/llama.cpp/pull/27513). diff --git a/llama.cpp b/llama.cpp index 560445b..bd4f514 160000 --- a/llama.cpp +++ b/llama.cpp @@ -1 +1 @@ -Subproject commit 560445bf34c87356ad0f8d80fb03ec5488850b65 +Subproject commit bd4f514db14d87fded667787a7a963bfbaa98e89 diff --git a/patches/README.md b/patches/README.md new file mode 100644 index 0000000..62be99c --- /dev/null +++ b/patches/README.md @@ -0,0 +1,116 @@ +# llama.cpp and ggml patches + +ggml comes from the `llama.cpp` submodule, which is pinned to an upstream commit +and never modified. This directory holds the project's changes to it: one patch +file per change, applied in the order `series` lists them. + +## Builds + +CMake applies the series automatically. With `NEMO_SPEECH_GGML_PATCHED=ON` (the +default, and the `cuda-*` and `cpu-*` presets) it copies the submodule to +`/_deps/llama.cpp`, applies these patches, and builds ggml and llama.cpp +from the copy. It refreshes the copy only when a patch or the pinned commit +changes, and rewrites only the files whose content changed, so an edit to one +patch recompiles only what that edit affects. + +- `-DNEMO_SPEECH_GGML_PATCHED=OFF` builds the pristine submodule (the `metal-*` + and `vulkan-*` presets do this). +- `-DNEMO_SPEECH_LLAMA_CPP_SOURCE_DIR=` builds `` as-is. + +The environment variables the patches read are listed in +[docs/development/diagnostics.md](../docs/development/diagnostics.md). + +## Patch files + +Each patch is a commit exported with `git format-patch`, trimmed to the author, +the subject, the description and the diff: + +``` +From: Prabhsimran Singh +Subject: ggml-cuda: flatten PAD launches for large batches + +PAD launched ne1 x (ne2*ne3) blocks, so large batches or long inputs exceeded +the 65535 grid.y/grid.z limit. Flatten the launch into grid.x. + +Upstream: candidate (add test_pad cases beyond 65535). + +diff --git a/ggml/src/ggml-cuda/pad.cu b/ggml/src/ggml-cuda/pad.cu +index 31cd00f77816fe6f5ffdda0428289e2aa1a25f9b..40b1b496b011eb4f85c35af3d1c29d223e2639e4 100644 +... +``` + +- `git am` reads the header; `git apply`, which builds use, skips everything + before the first `diff`. `export` writes the same bytes for the same commits. +- The `index` lines carry full blob IDs. Abbreviated IDs lengthen as a + repository grows, so full ones keep the export identical in every clone. + They also let `git am -3` or `git apply -3` fall back to a three-way merge + when a patch file no longer applies cleanly, for example when bringing in + patches from another branch. +- The file name is the subject in lowercase, with every character other than + a letter, digit, `.` or `_` replaced by `-`, cut at 100 characters. +- The subject starts with the area: `ggml:`, `ggml-cuda:`, `ggml-cpu:`, or + `llama:`. The description says what the change does and why this project + needs it, and ends with an `Upstream:` line: a pull request link, + `candidate` (with what it still needs), or why it stays here. Credit merged + work with `Co-authored-by:` lines after it. +- `series` lists the files in apply order. A patch file that `series` does not + list, or a listed file that does not exist, fails the build and the check. + +Don't edit patch files by hand. Change the commits and run `export`: CI runs +`scripts/llama-patches.sh check`, which fails when the files differ from what +`export` writes. + +## Working on the patches + +`scripts/llama-patches.sh edit` creates `.deps/llama.cpp-patches`, a worktree of +the `llama.cpp` submodule at the pinned commit with one commit per patch. Build +against it and use ordinary git there; its commits never leave the worktree. + +```sh +scripts/llama-patches.sh edit +cmake --preset cuda-asr -DNEMO_SPEECH_LLAMA_CPP_SOURCE_DIR=$PWD/.deps/llama.cpp-patches +cd .deps/llama.cpp-patches +``` + +On Windows, run the script from Git Bash, which Git for Windows installs; it +prints the worktree path in the `C:/...` form CMake expects. Builds need no +script on any platform. + +Then, in the worktree: + +- **Change a patch:** edit, build, test, then + `git commit -a --fixup=` and + `git rebase -i --autosquash `. +- **Add a patch:** commit on top (or move it with `git rebase -i`). Its commit + message becomes the patch description; follow the conventions above. +- **Reword, reorder, split, merge or drop patches:** `git rebase -i `. +- **Finish:** from the repository root, `scripts/llama-patches.sh export` + rewrites `patches/` and `series`, and `scripts/llama-patches.sh done` removes + the worktree (it refuses while the worktree has unexported changes). Commit + the `patches/` changes together with the source changes that need them. + +`scripts/llama-patches.sh diff` prints a `git range-diff` between the committed +series and the working tree, patch by patch, which is easier to review than a +diff of patch files. + +When two branches both add a patch, both edit the end of `series`, so the merge +conflicts there on purpose: keep both lines in the order that applies, then run +`scripts/llama-patches.sh check`. + +## Updating llama.cpp + +```sh +scripts/llama-patches.sh rebase # pins the submodule there +# resolve conflicts with git in .deps/llama.cpp-patches, build, test +scripts/llama-patches.sh export +``` + +Patches that upstream has absorbed become empty and are dropped; `export` lists +what was removed. Update at least monthly: the longer the pin sits, the more the +series conflicts. + +A clean rebase can still change behavior: when upstream narrows a backend's +`supports_op`, a patched op silently falls back to the CPU. After an update, run +`test-backend-ops` and +[`check_backend_coverage`](../docs/development/diagnostics.md#backend-coverage), +and compare outputs against the previous pin. diff --git a/ggml-patches/0014-cuda-fused-attention-extensions.patch b/patches/ggml-add-fused-attention-op-for-fastconformer-and-magpietts.patch similarity index 72% rename from ggml-patches/0014-cuda-fused-attention-extensions.patch rename to patches/ggml-add-fused-attention-op-for-fastconformer-and-magpietts.patch index 9b89f3f..fa87551 100644 --- a/ggml-patches/0014-cuda-fused-attention-extensions.patch +++ b/patches/ggml-add-fused-attention-op-for-fastconformer-and-magpietts.patch @@ -1,49 +1,79 @@ -diff --git a/include/ggml.h b/include/ggml.h -index fb823571..d403756e 100644 ---- a/include/ggml.h -+++ b/include/ggml.h -@@ -583,7 +583,7 @@ extern "C" { +From: Prabhsimran Singh +Subject: ggml: add fused attention op for FastConformer and MagpieTTS + +Adds GGML_OP_FUSED_ATTN with three CUDA modes: + +- relative-position (Transformer-XL) attention for the FastConformer encoder: + scale*((q+u).k + (q+v).p[rel]) + mask -> softmax -> .v, with strided + Q/K/V/P views, F16 K/V/P storage and head-merged output; +- the same with a persistent per-stream K/V ring arena (slot ids, ring head, + active length) for cache-aware streaming ASR; +- standard cached single-query attention for the MagpieTTS decoder. The + q_len == 1, d_k == 64 kernel splits the key range across blocks (about 96 + keys per block, up to 8 blocks per head) with a last-arriving-block merge, + so a decoder step uses several SMs per head instead of one. + +Used by the FastConformer encoder in CUDA builds with NEMO_SPEECH_GGML_PATCHED +and by the MagpieTTS persistent decoder. There is no CPU implementation. + +Upstream: not planned (model-specific). + +Co-authored-by: anand-nv <105917641+anand-nv@users.noreply.github.com> + +diff --git a/ggml/include/ggml.h b/ggml/include/ggml.h +index 224bdef927ac90bfa07ee39db3247b7bac4f14ec..fd31b70ca3a05c4814c62a0be599ec0a56766e8e 100644 +--- a/ggml/include/ggml.h ++++ b/ggml/include/ggml.h +@@ -601,6 +601,8 @@ extern "C" { GGML_OP_GLU, -- GGML_OP_FUSED_RELPOS_ATTN, + GGML_OP_FUSED_ATTN, - ++ GGML_OP_COUNT, }; -@@ -2423,12 +2423,11 @@ extern "C" { + +@@ -2512,6 +2514,81 @@ extern "C" { struct ggml_tensor * a, struct ggml_tensor * sinks); -- // Fused FastConformer relative-position multi-head attention. -- // Replaces the content (K*Qu) + position (P*Qv with rel-shift) + softmax + -- // context (attn*V) op sequence with one kernel. CUDA-only (CPU/other -- // backends report unsupported and the unfused graph runs instead). + // Fused CUDA multi-head attention with optional relative-position terms + // and an optional persistent circular K/V cache. CPU and other backends + // report this op as unsupported. - // -- // Logical shapes, head dim ne[0]=d_k fastest: ++ // + // Logical shapes, with head dimension d_k as ne[0]: - // q [d_k, q_len, n_head, batch] pre-bias query (Qu/Qv added inside) - // k [d_k, kv_len, n_head, batch] - // v [d_k, kv_len, n_head, batch] -@@ -2436,8 +2435,9 @@ extern "C" { - // pos_len >= kv_len + q_len - 1 - // bias_u [d_k, n_head] pos_bias_u (content term) - // bias_v [d_k, n_head] pos_bias_v (position term) -- // mask [kv_len] or [kv_len, batch] additive (0 / -inf) key mask, -- // or NULL shared or per-stream columns ++ // q [d_k, q_len, n_head, batch] pre-bias query (Qu/Qv added inside) ++ // k [d_k, kv_len, n_head, batch] ++ // v [d_k, kv_len, n_head, batch] ++ // p [d_k, pos_len, n_head] positional encoding projection, ++ // pos_len >= kv_len + q_len - 1 ++ // bias_u [d_k, n_head] pos_bias_u (content term) ++ // bias_v [d_k, n_head] pos_bias_v (position term) + // mask [kv_len], [kv_len, q_len], additive (0 / -inf) mask, + // or [kv_len, batch], or NULL offline per-query or streaming + // per-stream columns - // Q/K/V/P may be non-contiguous views as long as each d_k row is - // contiguous — e.g. Q sliced from a fused-QKV projection and K/V read - // head-split from a feat-major [n_feat, kv] window (the CUDA op derives -@@ -2460,6 +2460,44 @@ extern "C" { - float scale, - bool merge_heads); - ++ // Q/K/V/P may be non-contiguous views as long as each d_k row is ++ // contiguous — e.g. Q sliced from a fused-QKV projection and K/V read ++ // head-split from a feat-major [n_feat, kv] window (the CUDA op derives ++ // all addressing from the tensors' nb[]). bias_u/bias_v/mask must be ++ // contiguous. ++ // Output: [d_k, q_len, n_head, batch] (attention context, pre-output-proj). ++ // With merge_heads=true the output keeps that logical shape but uses a ++ // head-merged memory layout: permute(out, 0, 2, 1, 3) is a contiguous ++ // (n_feat, q_len, batch) matrix, consumable by the output projection ++ // without a copy. ++ GGML_API struct ggml_tensor * ggml_fused_relpos_attn( ++ struct ggml_context * ctx, ++ struct ggml_tensor * q, ++ struct ggml_tensor * k, ++ struct ggml_tensor * v, ++ struct ggml_tensor * p, ++ struct ggml_tensor * bias_u, ++ struct ggml_tensor * bias_v, ++ struct ggml_tensor * mask, ++ float scale, ++ bool merge_heads); ++ + // Streaming CUDA variant. K/V contain only the current chunk; cached K/V + // are read directly from a persistent [n_feat*cache_len, slots, 2] F32 + // arena using one I32 slot id and circular-cache head per batch item. @@ -85,40 +115,39 @@ index fb823571..d403756e 100644 // TODO: needs to be adapted to ggml_flash_attn_ext GGML_API struct ggml_tensor * ggml_flash_attn_back( struct ggml_context * ctx, -diff --git a/src/ggml-cpu/ggml-cpu.c b/src/ggml-cpu/ggml-cpu.c -index cce0e5a8..43de6346 100644 ---- a/src/ggml-cpu/ggml-cpu.c -+++ b/src/ggml-cpu/ggml-cpu.c -@@ -1988,9 +1988,9 @@ static void ggml_compute_forward(struct ggml_compute_params * params, struct ggm +diff --git a/ggml/src/ggml-cpu/ggml-cpu.c b/ggml/src/ggml-cpu/ggml-cpu.c +index 8bb0ff7bc3366be957dd20aba3fbe0f50dd249d7..4fb2e8b6f7236c95e66a0be167d531f154cfd4d1 100644 +--- a/ggml/src/ggml-cpu/ggml-cpu.c ++++ b/ggml/src/ggml-cpu/ggml-cpu.c +@@ -2034,6 +2034,10 @@ static void ggml_compute_forward(struct ggml_compute_params * params, struct ggm { ggml_compute_forward_flash_attn_ext(params, tensor); } break; -- case GGML_OP_FUSED_RELPOS_ATTN: + case GGML_OP_FUSED_ATTN: - { -- GGML_ABORT("FUSED_RELPOS_ATTN has no CPU path (CUDA-only)"); ++ { + GGML_ABORT("FUSED_ATTN has no CPU path (CUDA-only)"); - } break; ++ } break; case GGML_OP_FLASH_ATTN_BACK: { -diff --git a/src/ggml-cpu/ggml-cpu.cpp b/src/ggml-cpu/ggml-cpu.cpp -index 7429cc45..0c570639 100644 ---- a/src/ggml-cpu/ggml-cpu.cpp -+++ b/src/ggml-cpu/ggml-cpu.cpp -@@ -439,7 +439,7 @@ static bool ggml_backend_cpu_device_supports_op(ggml_backend_dev_t dev, const st + int32_t t = ggml_get_op_params_i32(tensor, 0); +diff --git a/ggml/src/ggml-cpu/ggml-cpu.cpp b/ggml/src/ggml-cpu/ggml-cpu.cpp +index 1df0f2bb926894eba96beef691ff2bf520ab70b3..38b5dbf1205c43d33de6d225f6da8c3fc915e2c8 100644 +--- a/ggml/src/ggml-cpu/ggml-cpu.cpp ++++ b/ggml/src/ggml-cpu/ggml-cpu.cpp +@@ -440,6 +440,8 @@ static bool ggml_backend_cpu_device_supports_op(ggml_backend_dev_t dev, const st } switch (op->op) { -- case GGML_OP_FUSED_RELPOS_ATTN: + case GGML_OP_FUSED_ATTN: - return false; // CUDA-only; CPU falls back to the unfused graph ++ return false; // CUDA-only; CPU falls back to the unfused graph case GGML_OP_CPY: case GGML_OP_SET_ROWS: -diff --git a/src/ggml-cuda/fused-attention.cu b/src/ggml-cuda/fused-attention.cu + return +diff --git a/ggml/src/ggml-cuda/fused-attention.cu b/ggml/src/ggml-cuda/fused-attention.cu new file mode 100644 -index 00000000..84e1c025 +index 0000000000000000000000000000000000000000..84e1c0255ad34c985cd6e4699e535ba7b7123d63 --- /dev/null -+++ b/src/ggml-cuda/fused-attention.cu ++++ b/ggml/src/ggml-cuda/fused-attention.cu @@ -0,0 +1,1720 @@ +// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +// SPDX-License-Identifier: Apache-2.0 @@ -1840,718 +1869,105 @@ index 00000000..84e1c025 + } + update_cache(); +} -diff --git a/src/ggml-cuda/fused-relpos-attn.cuh b/src/ggml-cuda/fused-attention.cuh -similarity index 64% -rename from src/ggml-cuda/fused-relpos-attn.cuh -rename to src/ggml-cuda/fused-attention.cuh -index 1543fa18..48e23c67 100644 ---- a/src/ggml-cuda/fused-relpos-attn.cuh -+++ b/src/ggml-cuda/fused-attention.cuh -@@ -2,4 +2,4 @@ - // SPDX-License-Identifier: Apache-2.0 - #include "common.cuh" - --void ggml_cuda_op_fused_relpos_attn(ggml_backend_cuda_context & ctx, ggml_tensor * dst); +diff --git a/ggml/src/ggml-cuda/fused-attention.cuh b/ggml/src/ggml-cuda/fused-attention.cuh +new file mode 100644 +index 0000000000000000000000000000000000000000..48e23c673ca7cb6b27b238fc44eaf038660e3a99 +--- /dev/null ++++ b/ggml/src/ggml-cuda/fused-attention.cuh +@@ -0,0 +1,5 @@ ++// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. ++// SPDX-License-Identifier: Apache-2.0 ++#include "common.cuh" ++ +void ggml_cuda_op_fused_attention(ggml_backend_cuda_context & ctx, ggml_tensor * dst); -diff --git a/src/ggml-cuda/fused-relpos-attn.cu b/src/ggml-cuda/fused-relpos-attn.cu -deleted file mode 100644 -index f3c4836a..00000000 ---- a/src/ggml-cuda/fused-relpos-attn.cu -+++ /dev/null -@@ -1,600 +0,0 @@ --// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. --// SPDX-License-Identifier: Apache-2.0 --#include "fused-relpos-attn.cuh" -- --#include -- --// Fused FastConformer relative-position multi-head attention. --// --// One block per (head, query, batch); blockDim.x = d_k threads (one per head --// dim). Scores, rel-shifted position term, scale+mask, two-pass softmax, and --// the attn*V context are all computed in-kernel, so the rel-shift matrix and --// the score matrix are never materialized in global memory. --// --// Operand addressing is fully stride-driven (strides read from each tensor's --// nb[] by the host wrapper, in elements): Q/K/V may be non-contiguous views — --// e.g. the Q slice of a fused-QKV projection, or a feat-major [n_feat, kv] --// K/V window — as long as d_k stays innermost-contiguous (asserted by the op --// constructor; the vectorized loads rely on it). P is [d_k, pos_len, n_head] --// with pos_len = kv + q - 1 rows addressable; bu/bv are [d_k, n_head]; --// mask is [kv] (shared) or [kv, batch] (per-stream) additive (0 / -inf), or NULL. --// Output ctx is [d_k, q, n_head, batch] logical; with the merge_heads op flag --// its memory layout is head-merged ([d_k+h*d_k] innermost, i.e. a plain --// (n_feat, q, batch) matrix), so the output projection consumes it without a --// permute copy. --// --// Requires d_k to be a power of two (the softmax reduction halves blockDim.x). -- --// K/V/P are templated: F16 operands halve the dominant re-read traffic when --// the caller stages them; F32 operands skip the staging casts entirely. All --// math stays in F32 either way. -- --static constexpr int RELPOS_ATTN_DK_128 = 128; --static constexpr int RELPOS_ATTN_WARPS_128 = RELPOS_ATTN_DK_128 / 32; --static constexpr int RELPOS_ATTN_CC_SM100 = 1000; -- --static __device__ __forceinline__ float relpos_warp_sum(float value) { --#pragma unroll -- for (int offset = 16; offset > 0; offset >>= 1) { -- value += __shfl_down_sync(0xffffffff, value, offset); -- } -- return value; --} -- --static __device__ __forceinline__ float4 relpos_load4(const float * ptr) { -- return *reinterpret_cast(ptr); --} -- --static __device__ __forceinline__ float4 relpos_load4(const half * ptr) { -- const int2 packed = *reinterpret_cast(ptr); -- const half2 * values = reinterpret_cast(&packed); -- return make_float4( -- __low2float(values[0]), __high2float(values[0]), -- __low2float(values[1]), __high2float(values[1])); --} -- --// SM100 streaming specialization for d_k=128 and q=2. Keeping one block per --// query preserves the two rows' parallelism while the complete grid fits in a --// resident wave. Compared with the generic shared-memory reduction tree, the --// score-producing warps retain their local maxima and max/sum use only four --// warp partials. This removes fourteen block barriers and 124 scratch floats. --template --static __global__ void fused_relpos_attn_warp_128_kernel( -- const float * __restrict__ Q, const T * __restrict__ K, -- const T * __restrict__ V, const T * __restrict__ Ppos, -- const float * __restrict__ bu, const float * __restrict__ bv, -- const float * __restrict__ mask, float * __restrict__ ctx, -- int kv, float scale, -- long q_sq, long q_sh, long q_sb, -- long k_sj, long k_sh, long k_sb, -- long v_sj, long v_sh, long v_sb, -- long p_sr, long p_sh, -- long o_si, long o_sh, long o_sb, long m_sb) { -- extern __shared__ float sh[]; -- float * Qu = sh; -- float * Qv = Qu + RELPOS_ATTN_DK_128; -- float * sc = Qv + RELPOS_ATTN_DK_128; -- float * red = sc + kv; -- -- const int h = blockIdx.x; -- const int i = blockIdx.y; -- const int b = blockIdx.z; -- const int d = threadIdx.x; -- const int warp = d >> 5; -- const int lane = d & 31; -- -- const float * Qhi = Q + (size_t) b * q_sb + (size_t) h * q_sh + (size_t) i * q_sq; -- const T * Kh = K + (size_t) b * k_sb + (size_t) h * k_sh; -- const T * Vh = V + (size_t) b * v_sb + (size_t) h * v_sh; -- const T * Ph = Ppos + (size_t) h * p_sh; -- const float * Mb = mask ? mask + (size_t) b * m_sb : nullptr; -- -- Qu[d] = Qhi[d] + bu[h * RELPOS_ATTN_DK_128 + d]; -- Qv[d] = Qhi[d] + bv[h * RELPOS_ATTN_DK_128 + d]; -- __syncthreads(); -- -- const int d4 = lane * 4; -- const float4 qu4 = *reinterpret_cast(Qu + d4); -- const float4 qv4 = *reinterpret_cast(Qv + d4); -- float produced_max = -INFINITY; -- for (int j = warp; j < kv; j += RELPOS_ATTN_WARPS_128) { -- const float4 k4 = relpos_load4(Kh + (size_t) j * k_sj + d4); -- const int row = 1 + j - i; -- const float4 p4 = relpos_load4(Ph + (size_t) row * p_sr + d4); -- float score = -- k4.x * qu4.x + p4.x * qv4.x + -- k4.y * qu4.y + p4.y * qv4.y + -- k4.z * qu4.z + p4.z * qv4.z + -- k4.w * qu4.w + p4.w * qv4.w; -- score = relpos_warp_sum(score); -- if (lane == 0) { -- sc[j] = score * scale + (Mb ? Mb[j] : 0.0f); -- produced_max = fmaxf(produced_max, sc[j]); -- } -- } -- if (lane == 0) { -- red[warp] = produced_max; -- } -- __syncthreads(); -- -- if (d == 0) { -- float maximum = red[0]; --#pragma unroll -- for (int w = 1; w < RELPOS_ATTN_WARPS_128; ++w) { -- maximum = fmaxf(maximum, red[w]); -- } -- red[0] = maximum; -- } -- __syncthreads(); -- const float maximum = red[0]; -- -- // red[] is reused below. Ensure every warp has captured red[0] first; -- // otherwise warp 0 can overwrite the maximum while another warp loads it. -- __syncthreads(); -- float local_sum = 0.0f; -- for (int j = d; j < kv; j += RELPOS_ATTN_DK_128) { -- const float weight = __expf(sc[j] - maximum); -- sc[j] = weight; -- local_sum += weight; -- } -- local_sum = relpos_warp_sum(local_sum); -- if (lane == 0) { -- red[warp] = local_sum; -- } -- __syncthreads(); -- if (d == 0) { -- float sum = red[0]; --#pragma unroll -- for (int w = 1; w < RELPOS_ATTN_WARPS_128; ++w) { -- sum += red[w]; -- } -- red[0] = sum; -- } -- __syncthreads(); -- -- float context = 0.0f; -- for (int j = 0; j < kv; ++j) { -- context += sc[j] * (float) Vh[(size_t) j * v_sj + d]; -- } -- ctx[(size_t) b * o_sb + (size_t) i * o_si + (size_t) h * o_sh + d] = context / red[0]; --} -- --// Once the one-block-per-query grid spills into another occupancy wave, one --// block computes both query rows. K and V are then loaded once and reused; --// only the two relative-position rows differ. The reduction order matches the --// one-query specialization so changing batch size does not change numerics. --template --static __global__ void fused_relpos_attn_q2_warp_128_kernel( -- const float * __restrict__ Q, const T * __restrict__ K, -- const T * __restrict__ V, const T * __restrict__ Ppos, -- const float * __restrict__ bu, const float * __restrict__ bv, -- const float * __restrict__ mask, float * __restrict__ ctx, -- int kv, float scale, -- long q_sq, long q_sh, long q_sb, -- long k_sj, long k_sh, long k_sb, -- long v_sj, long v_sh, long v_sb, -- long p_sr, long p_sh, -- long o_si, long o_sh, long o_sb, long m_sb) { -- extern __shared__ float sh[]; -- float * Qu0 = sh; -- float * Qv0 = Qu0 + RELPOS_ATTN_DK_128; -- float * Qu1 = Qv0 + RELPOS_ATTN_DK_128; -- float * Qv1 = Qu1 + RELPOS_ATTN_DK_128; -- float * sc0 = Qv1 + RELPOS_ATTN_DK_128; -- float * sc1 = sc0 + kv; -- float * red0 = sc1 + kv; -- float * red1 = red0 + RELPOS_ATTN_WARPS_128; -- -- const int h = blockIdx.x; -- const int b = blockIdx.z; -- const int d = threadIdx.x; -- const int warp = d >> 5; -- const int lane = d & 31; -- -- const float * Qh0 = Q + (size_t) b * q_sb + (size_t) h * q_sh; -- const float * Qh1 = Qh0 + q_sq; -- const T * Kh = K + (size_t) b * k_sb + (size_t) h * k_sh; -- const T * Vh = V + (size_t) b * v_sb + (size_t) h * v_sh; -- const T * Ph = Ppos + (size_t) h * p_sh; -- const float * Mb = mask ? mask + (size_t) b * m_sb : nullptr; -- -- const float bias_u = bu[h * RELPOS_ATTN_DK_128 + d]; -- const float bias_v = bv[h * RELPOS_ATTN_DK_128 + d]; -- Qu0[d] = Qh0[d] + bias_u; -- Qv0[d] = Qh0[d] + bias_v; -- Qu1[d] = Qh1[d] + bias_u; -- Qv1[d] = Qh1[d] + bias_v; -- __syncthreads(); -- -- const int d4 = lane * 4; -- const float4 qu04 = *reinterpret_cast(Qu0 + d4); -- const float4 qv04 = *reinterpret_cast(Qv0 + d4); -- const float4 qu14 = *reinterpret_cast(Qu1 + d4); -- const float4 qv14 = *reinterpret_cast(Qv1 + d4); -- float produced_max0 = -INFINITY; -- float produced_max1 = -INFINITY; -- for (int j = warp; j < kv; j += RELPOS_ATTN_WARPS_128) { -- const float4 k4 = relpos_load4(Kh + (size_t) j * k_sj + d4); -- const float4 p04 = relpos_load4(Ph + (size_t) (j + 1) * p_sr + d4); -- const float4 p14 = relpos_load4(Ph + (size_t) j * p_sr + d4); -- float score0 = -- k4.x * qu04.x + p04.x * qv04.x + -- k4.y * qu04.y + p04.y * qv04.y + -- k4.z * qu04.z + p04.z * qv04.z + -- k4.w * qu04.w + p04.w * qv04.w; -- float score1 = -- k4.x * qu14.x + p14.x * qv14.x + -- k4.y * qu14.y + p14.y * qv14.y + -- k4.z * qu14.z + p14.z * qv14.z + -- k4.w * qu14.w + p14.w * qv14.w; -- score0 = relpos_warp_sum(score0); -- score1 = relpos_warp_sum(score1); -- if (lane == 0) { -- const float additive_mask = Mb ? Mb[j] : 0.0f; -- sc0[j] = score0 * scale + additive_mask; -- sc1[j] = score1 * scale + additive_mask; -- produced_max0 = fmaxf(produced_max0, sc0[j]); -- produced_max1 = fmaxf(produced_max1, sc1[j]); -- } -- } -- if (lane == 0) { -- red0[warp] = produced_max0; -- red1[warp] = produced_max1; -- } -- __syncthreads(); -- -- if (d == 0) { -- float maximum0 = red0[0]; -- float maximum1 = red1[0]; --#pragma unroll -- for (int w = 1; w < RELPOS_ATTN_WARPS_128; ++w) { -- maximum0 = fmaxf(maximum0, red0[w]); -- maximum1 = fmaxf(maximum1, red1[w]); -- } -- red0[0] = maximum0; -- red1[0] = maximum1; -- } -- __syncthreads(); -- const float maximum0 = red0[0]; -- const float maximum1 = red1[0]; -- __syncthreads(); -- -- float local_sum0 = 0.0f; -- float local_sum1 = 0.0f; -- for (int j = d; j < kv; j += RELPOS_ATTN_DK_128) { -- const float weight0 = __expf(sc0[j] - maximum0); -- const float weight1 = __expf(sc1[j] - maximum1); -- sc0[j] = weight0; -- sc1[j] = weight1; -- local_sum0 += weight0; -- local_sum1 += weight1; -- } -- local_sum0 = relpos_warp_sum(local_sum0); -- local_sum1 = relpos_warp_sum(local_sum1); -- if (lane == 0) { -- red0[warp] = local_sum0; -- red1[warp] = local_sum1; -- } -- __syncthreads(); -- if (d == 0) { -- float sum0 = red0[0]; -- float sum1 = red1[0]; --#pragma unroll -- for (int w = 1; w < RELPOS_ATTN_WARPS_128; ++w) { -- sum0 += red0[w]; -- sum1 += red1[w]; -- } -- red0[0] = sum0; -- red1[0] = sum1; -- } -- __syncthreads(); -- -- float context0 = 0.0f; -- float context1 = 0.0f; -- for (int j = 0; j < kv; ++j) { -- const float value = (float) Vh[(size_t) j * v_sj + d]; -- context0 += sc0[j] * value; -- context1 += sc1[j] * value; -- } -- const size_t out = (size_t) b * o_sb + (size_t) h * o_sh + d; -- ctx[out] = context0 / red0[0]; -- ctx[out + o_si] = context1 / red1[0]; --} -- --template --static int relpos_attn_warp_128_max_blocks_per_sm(int device, size_t shmem) { -- // The target shape fixes dynamic shared memory, so occupancy is invariant -- // for a given compiled kernel and device. Avoid repeating the CUDA runtime -- // query in every attention layer and graph execution. -- static std::atomic cached[GGML_CUDA_MAX_DEVICES] = {}; -- int blocks = cached[device].load(std::memory_order_relaxed); -- if (blocks == 0) { -- CUDA_CHECK(cudaOccupancyMaxActiveBlocksPerMultiprocessor( -- &blocks, fused_relpos_attn_warp_128_kernel, RELPOS_ATTN_DK_128, shmem)); -- GGML_ASSERT(blocks > 0); -- cached[device].store(blocks, std::memory_order_relaxed); -- } -- return blocks; --} -- --template --static __global__ void fused_relpos_attn_kernel( -- const float * __restrict__ Q, const T * __restrict__ K, -- const T * __restrict__ V, const T * __restrict__ Ppos, -- const float * __restrict__ bu, const float * __restrict__ bv, -- const float * __restrict__ mask, float * __restrict__ ctx, -- int q, int kv, int n_head, float scale, -- // element strides: x_sq = between queries/keys, x_sh = between heads, -- // x_sb = between batch items -- long q_sq, long q_sh, long q_sb, -- long k_sj, long k_sh, long k_sb, -- long v_sj, long v_sh, long v_sb, -- long p_sr, long p_sh, -- long o_si, long o_sh, long o_sb, long m_sb) { -- extern __shared__ float sh[]; -- const int dk = blockDim.x; -- float * Qu = sh; // [dk] -- float * Qv = sh + dk; // [dk] -- float * sc = sh + 2 * dk; // [kv] -- float * red = sh + 2 * dk + kv; // [dk] reduction scratch -- -- const int h = blockIdx.x; // head -- const int i = blockIdx.y; // query -- const int b = blockIdx.z; // batch -- const int d = threadIdx.x; // head dim 0..dk-1 -- -- const float * Qhi = Q + (size_t) b * q_sb + (size_t) h * q_sh + (size_t) i * q_sq; -- const T * Kh = K + (size_t) b * k_sb + (size_t) h * k_sh; -- const T * Vh = V + (size_t) b * v_sb + (size_t) h * v_sh; -- const T * Ph = Ppos + (size_t) h * p_sh; -- const float * Mb = mask ? mask + (size_t) b * m_sb : nullptr; -- -- Qu[d] = Qhi[d] + bu[h * dk + d]; -- Qv[d] = Qhi[d] + bv[h * dk + d]; -- __syncthreads(); -- -- // scores. Two layouts: -- // * dk == 128 (the FastConformer case): WARP-COOPERATIVE — each warp owns -- // a key j and the 32 lanes split the 128 dims 4-a-piece with one -- // vectorized row load + shuffle reduction. The original -- // thread-per-key loop left dk-kv threads idle (kv ~= 50 < 128) and -- // issued dk scalar loads per row, which made the kernel -- // load-issue-bound (F16 operands alone changed nothing). -- // * otherwise: legacy thread-per-key scalar loop. -- if (dk == 128) { -- const int warp = d >> 5, lane = d & 31, nw = dk >> 5; -- for (int j = warp; j < kv; j += nw) { -- const T * Kj = Kh + (size_t) j * k_sj + lane * 4; -- const int row = (q - 1) + j - i; // rel-shift index -- const T * Pr = Ph + (size_t) row * p_sr + lane * 4; -- float k4[4], p4[4]; -- if (sizeof(T) == 2) { -- const int2 kr = *(const int2 *) Kj; -- const int2 pr = *(const int2 *) Pr; -- const half2 * kh = (const half2 *) &kr; -- const half2 * ph = (const half2 *) ≺ -- k4[0] = __low2float(kh[0]); k4[1] = __high2float(kh[0]); -- k4[2] = __low2float(kh[1]); k4[3] = __high2float(kh[1]); -- p4[0] = __low2float(ph[0]); p4[1] = __high2float(ph[0]); -- p4[2] = __low2float(ph[1]); p4[3] = __high2float(ph[1]); -- } else { -- const float4 kr = *(const float4 *) Kj; -- const float4 pr = *(const float4 *) Pr; -- k4[0] = ((const float *) &kr)[0]; k4[1] = ((const float *) &kr)[1]; -- k4[2] = ((const float *) &kr)[2]; k4[3] = ((const float *) &kr)[3]; -- p4[0] = ((const float *) &pr)[0]; p4[1] = ((const float *) &pr)[1]; -- p4[2] = ((const float *) &pr)[2]; p4[3] = ((const float *) &pr)[3]; -- } -- float s = 0.0f; --#pragma unroll -- for (int e = 0; e < 4; e++) { -- s += k4[e] * Qu[lane * 4 + e] + p4[e] * Qv[lane * 4 + e]; -- } --#pragma unroll -- for (int off = 16; off > 0; off >>= 1) { -- s += __shfl_xor_sync(0xffffffff, s, off); -- } -- if (lane == 0) { -- sc[j] = s * scale + (Mb ? Mb[j] : 0.0f); -- } -- } -- } else { -- for (int j = d; j < kv; j += dk) { -- const T * Kj = Kh + (size_t) j * k_sj; -- const int row = (q - 1) + j - i; // rel-shift index -- const T * Pr = Ph + (size_t) row * p_sr; -- float ac = 0.0f, bd = 0.0f; -- for (int dd = 0; dd < dk; dd++) { -- ac += (float) Kj[dd] * Qu[dd]; -- bd += (float) Pr[dd] * Qv[dd]; -- } -- sc[j] = (ac + bd) * scale + (Mb ? Mb[j] : 0.0f); -- } -- } -- __syncthreads(); -- -- // block max over sc[0..kv) -- float lm = -INFINITY; -- for (int j = d; j < kv; j += dk) lm = fmaxf(lm, sc[j]); -- red[d] = lm; -- __syncthreads(); -- for (int s = dk / 2; s > 0; s >>= 1) { -- if (d < s) red[d] = fmaxf(red[d], red[d + s]); -- __syncthreads(); -- } -- const float m = red[0]; -- __syncthreads(); -- -- // exp + block sum -- float ls = 0.0f; -- for (int j = d; j < kv; j += dk) { -- const float e = __expf(sc[j] - m); -- sc[j] = e; -- ls += e; -- } -- red[d] = ls; -- __syncthreads(); -- for (int s = dk / 2; s > 0; s >>= 1) { -- if (d < s) red[d] += red[d + s]; -- __syncthreads(); -- } -- const float inv = 1.0f / red[0]; -- __syncthreads(); -- -- // ctx[d] = inv * sum_j softmax(sc[j]) * V[j,d] (thread d owns output dim d) -- float c = 0.0f; -- for (int j = 0; j < kv; j++) c += sc[j] * (float) Vh[(size_t) j * v_sj + d]; -- ctx[(size_t) b * o_sb + (size_t) i * o_si + (size_t) h * o_sh + d] = c * inv; --} -- --void ggml_cuda_op_fused_relpos_attn(ggml_backend_cuda_context & ctx, ggml_tensor * dst) { -- const ggml_tensor * q = dst->src[0]; -- const ggml_tensor * k = dst->src[1]; -- const ggml_tensor * v = dst->src[2]; -- const ggml_tensor * p = dst->src[3]; -- const ggml_tensor * bias_u = dst->src[4]; -- const ggml_tensor * bias_v = dst->src[5]; -- const ggml_tensor * mask = dst->src[6]; // may be null -- -- GGML_ASSERT(q->type == GGML_TYPE_F32); -- // K/V/P may be F32 or (all together) F16 — see the kernel comment. -- GGML_ASSERT(k->type == v->type && k->type == p->type); -- GGML_ASSERT(k->type == GGML_TYPE_F32 || k->type == GGML_TYPE_F16); -- GGML_ASSERT(bias_u->type == GGML_TYPE_F32 && bias_v->type == GGML_TYPE_F32); -- GGML_ASSERT(dst->type == GGML_TYPE_F32); -- -- const int d_k = q->ne[0]; -- const int q_len = q->ne[1]; -- const int n_head = q->ne[2]; -- const int batch = q->ne[3]; -- const int kv_len = k->ne[1]; -- -- GGML_ASSERT((d_k & (d_k - 1)) == 0 && "fused_relpos_attn: d_k must be a power of two"); -- GGML_ASSERT(k->ne[0] == d_k && v->ne[0] == d_k && p->ne[0] == d_k); -- GGML_ASSERT(k->ne[1] == kv_len && v->ne[1] == kv_len); -- GGML_ASSERT(k->ne[2] == n_head && v->ne[2] == n_head && p->ne[2] == n_head); -- GGML_ASSERT(k->ne[3] == batch && v->ne[3] == batch); -- GGML_ASSERT(bias_u->ne[0] == d_k && bias_v->ne[0] == d_k); -- GGML_ASSERT(bias_u->ne[1] == n_head && bias_v->ne[1] == n_head); -- if (mask != nullptr) { -- GGML_ASSERT(mask->type == GGML_TYPE_F32); -- GGML_ASSERT(ggml_is_contiguous(mask)); -- GGML_ASSERT(mask->ne[0] == kv_len); -- // Shared across the batch (ne[1]==1) or one key-mask column per -- // stream (ne[1]==batch, the cache-aware layout). -- GGML_ASSERT(mask->ne[1] == 1 || mask->ne[1] == batch); -- GGML_ASSERT(mask->ne[2] == 1 && mask->ne[3] == 1); -- } -- const long m_sb = -- (mask != nullptr && mask->ne[1] == batch && batch > 1) ? (long) (mask->nb[1] / sizeof(float)) : 0; -- -- float scale; -- memcpy(&scale, dst->op_params, sizeof(scale)); -- -- // Element strides from tensor byte strides. d_k rows must be contiguous -- // (constructor invariant), everything else is free-form. -- const size_t qe = ggml_type_size(q->type); -- const size_t ke = ggml_type_size(k->type); -- const long q_sq = (long)(q->nb[1] / qe), q_sh = (long)(q->nb[2] / qe), -- q_sb = (long)(q->nb[3] / qe); -- const long k_sj = (long)(k->nb[1] / ke), k_sh = (long)(k->nb[2] / ke), -- k_sb = (long)(k->nb[3] / ke); -- const long v_sj = (long)(v->nb[1] / ke), v_sh = (long)(v->nb[2] / ke), -- v_sb = (long)(v->nb[3] / ke); -- const long p_sr = (long)(p->nb[1] / ke), p_sh = (long)(p->nb[2] / ke); -- const size_t oe = ggml_type_size(dst->type); -- const long o_si = (long)(dst->nb[1] / oe), o_sh = (long)(dst->nb[2] / oe), -- o_sb = (long)(dst->nb[3] / oe); -- -- const int device = ggml_cuda_get_device(); -- const auto & device_info = ggml_cuda_info().devices[device]; -- const size_t shmem = ((size_t) 3 * d_k + kv_len) * sizeof(float); -- const size_t max_shmem = device_info.smpb; -- GGML_ASSERT(shmem <= max_shmem && "fused_relpos_attn: kv window too large for shared memory"); -- -- cudaStream_t stream = ctx.stream(); -- -- // The cache-aware Nemotron streaming geometry is Q=2, KV=72, H=8, -- // d_k=128. On SM100 the one-query warp kernel is fastest while its grid -- // fits in one resident wave. If that grid exceeds its measured occupancy, -- // fuse both query rows: halving the block count and reusing K/V then wins. -- // cudaOccupancyMaxActiveBlocksPerMultiprocessor uses the compiled kernel's -- // actual register count, avoiding a hard-coded batch-size threshold. -- const bool use_sm100_q2 = -- device_info.cc == RELPOS_ATTN_CC_SM100 && device_info.warp_size == 32 && -- d_k == RELPOS_ATTN_DK_128 && q_len == 2 && kv_len == 72; -- if (use_sm100_q2) { -- const size_t warp_shmem = -- ((size_t) 2 * RELPOS_ATTN_DK_128 + kv_len + RELPOS_ATTN_WARPS_128) * sizeof(float); -- const size_t q2_shmem = -- ((size_t) 4 * RELPOS_ATTN_DK_128 + (size_t) 2 * kv_len + -- (size_t) 2 * RELPOS_ATTN_WARPS_128) * sizeof(float); -- GGML_ASSERT(q2_shmem <= max_shmem); -- -- const int max_single_blocks_per_sm = k->type == GGML_TYPE_F16 -- ? relpos_attn_warp_128_max_blocks_per_sm(device, warp_shmem) -- : relpos_attn_warp_128_max_blocks_per_sm(device, warp_shmem); -- const int64_t single_query_blocks = (int64_t) n_head * q_len * batch; -- const int64_t single_wave_blocks = (int64_t) device_info.nsm * max_single_blocks_per_sm; -- const bool fuse_queries = single_query_blocks > single_wave_blocks; -- const dim3 tuned_grid(n_head, fuse_queries ? 1 : q_len, batch); -- -- if (k->type == GGML_TYPE_F16) { -- if (fuse_queries) { -- fused_relpos_attn_q2_warp_128_kernel<<< -- tuned_grid, RELPOS_ATTN_DK_128, q2_shmem, stream>>>( -- (const float *) q->data, (const half *) k->data, (const half *) v->data, -- (const half *) p->data, (const float *) bias_u->data, (const float *) bias_v->data, -- mask ? (const float *) mask->data : nullptr, (float *) dst->data, -- kv_len, scale, q_sq, q_sh, q_sb, k_sj, k_sh, k_sb, v_sj, v_sh, v_sb, -- p_sr, p_sh, o_si, o_sh, o_sb, m_sb); -- } else { -- fused_relpos_attn_warp_128_kernel<<< -- tuned_grid, RELPOS_ATTN_DK_128, warp_shmem, stream>>>( -- (const float *) q->data, (const half *) k->data, (const half *) v->data, -- (const half *) p->data, (const float *) bias_u->data, (const float *) bias_v->data, -- mask ? (const float *) mask->data : nullptr, (float *) dst->data, -- kv_len, scale, q_sq, q_sh, q_sb, k_sj, k_sh, k_sb, v_sj, v_sh, v_sb, -- p_sr, p_sh, o_si, o_sh, o_sb, m_sb); -- } -- } else { -- if (fuse_queries) { -- fused_relpos_attn_q2_warp_128_kernel<<< -- tuned_grid, RELPOS_ATTN_DK_128, q2_shmem, stream>>>( -- (const float *) q->data, (const float *) k->data, (const float *) v->data, -- (const float *) p->data, (const float *) bias_u->data, (const float *) bias_v->data, -- mask ? (const float *) mask->data : nullptr, (float *) dst->data, -- kv_len, scale, q_sq, q_sh, q_sb, k_sj, k_sh, k_sb, v_sj, v_sh, v_sb, -- p_sr, p_sh, o_si, o_sh, o_sb, m_sb); -- } else { -- fused_relpos_attn_warp_128_kernel<<< -- tuned_grid, RELPOS_ATTN_DK_128, warp_shmem, stream>>>( -- (const float *) q->data, (const float *) k->data, (const float *) v->data, -- (const float *) p->data, (const float *) bias_u->data, (const float *) bias_v->data, -- mask ? (const float *) mask->data : nullptr, (float *) dst->data, -- kv_len, scale, q_sq, q_sh, q_sb, k_sj, k_sh, k_sb, v_sj, v_sh, v_sb, -- p_sr, p_sh, o_si, o_sh, o_sb, m_sb); -- } -- } -- return; -- } -- -- const dim3 grid(n_head, q_len, batch); -- if (k->type == GGML_TYPE_F16) { -- fused_relpos_attn_kernel<<>>( -- (const float *) q->data, (const half *) k->data, (const half *) v->data, -- (const half *) p->data, (const float *) bias_u->data, (const float *) bias_v->data, -- mask ? (const float *) mask->data : nullptr, (float *) dst->data, -- q_len, kv_len, n_head, scale, -- q_sq, q_sh, q_sb, k_sj, k_sh, k_sb, v_sj, v_sh, v_sb, p_sr, p_sh, o_si, o_sh, o_sb, -- m_sb); -- } else { -- fused_relpos_attn_kernel<<>>( -- (const float *) q->data, (const float *) k->data, (const float *) v->data, -- (const float *) p->data, (const float *) bias_u->data, (const float *) bias_v->data, -- mask ? (const float *) mask->data : nullptr, (float *) dst->data, -- q_len, kv_len, n_head, scale, -- q_sq, q_sh, q_sb, k_sj, k_sh, k_sb, v_sj, v_sh, v_sb, p_sr, p_sh, o_si, o_sh, o_sb, -- m_sb); -- } --} -diff --git a/src/ggml-cuda/ggml-cuda.cu b/src/ggml-cuda/ggml-cuda.cu -index 6b934a1e..826d4a59 100644 ---- a/src/ggml-cuda/ggml-cuda.cu -+++ b/src/ggml-cuda/ggml-cuda.cu -@@ -25,7 +25,7 @@ - #include "ggml-cuda/diagmask.cuh" +diff --git a/ggml/src/ggml-cuda/ggml-cuda.cu b/ggml/src/ggml-cuda/ggml-cuda.cu +index 27b83503d28b00e0ac831af0fe2a68f9b1288875..95c0b1bf1e9d91249aab7b56be251dd4bd6fa288 100644 +--- a/ggml/src/ggml-cuda/ggml-cuda.cu ++++ b/ggml/src/ggml-cuda/ggml-cuda.cu +@@ -26,6 +26,7 @@ #include "ggml-cuda/diag.cuh" #include "ggml-cuda/fattn.cuh" --#include "ggml-cuda/fused-relpos-attn.cuh" + #include "ggml-cuda/fwht.cuh" +#include "ggml-cuda/fused-attention.cuh" #include "ggml-cuda/getrows.cuh" #include "ggml-cuda/im2col.cuh" #include "ggml-cuda/mmf.cuh" -@@ -3214,8 +3214,8 @@ static bool ggml_cuda_compute_forward(ggml_backend_cuda_context & ctx, struct gg +@@ -2367,6 +2368,9 @@ static bool ggml_cuda_compute_forward(ggml_backend_cuda_context & ctx, struct gg case GGML_OP_FLASH_ATTN_EXT: ggml_cuda_flash_attn_ext(ctx, dst); break; -- case GGML_OP_FUSED_RELPOS_ATTN: -- ggml_cuda_op_fused_relpos_attn(ctx, dst); + case GGML_OP_FUSED_ATTN: + ggml_cuda_op_fused_attention(ctx, dst); - break; ++ break; case GGML_OP_CROSS_ENTROPY_LOSS: ggml_cuda_cross_entropy_loss(ctx, dst); -@@ -5976,7 +5976,7 @@ static bool ggml_backend_cuda_device_supports_op(ggml_backend_dev_t dev, const g - #endif // GGML_USE_MUSA + break; +@@ -5572,6 +5576,8 @@ static bool ggml_backend_cuda_device_supports_op(ggml_backend_dev_t dev, const g + op->type == GGML_TYPE_F32; case GGML_OP_FLASH_ATTN_EXT: return ggml_cuda_flash_attn_ext_supported(dev_ctx->device, op); -- case GGML_OP_FUSED_RELPOS_ATTN: + case GGML_OP_FUSED_ATTN: - return (op->src[0]->ne[0] & (op->src[0]->ne[0] - 1)) == 0; // d_k power of two ++ return (op->src[0]->ne[0] & (op->src[0]->ne[0] - 1)) == 0; // d_k power of two case GGML_OP_CROSS_ENTROPY_LOSS: case GGML_OP_CROSS_ENTROPY_LOSS_BACK: -diff --git a/src/ggml.c b/src/ggml.c -index 80b5802f..4f5aeb09 100644 ---- a/src/ggml.c -+++ b/src/ggml.c -@@ -1079,7 +1079,7 @@ static const char * GGML_OP_NAME[GGML_OP_COUNT] = { + case GGML_OP_OPT_STEP_ADAMW: +diff --git a/ggml/src/ggml.c b/ggml/src/ggml.c +index a286083b9320acbe9654b34c1cec68930a8d39ee..81ecb230f6f88ff0f58292f6b4283e552f7ada39 100644 +--- a/ggml/src/ggml.c ++++ b/ggml/src/ggml.c +@@ -1099,9 +1099,11 @@ static const char * GGML_OP_NAME[GGML_OP_COUNT] = { + "OPT_STEP_SGD", "GLU", - -- "FUSED_RELPOS_ATTN", ++ + "FUSED_ATTN", }; - static_assert(GGML_OP_COUNT == 97, "GGML_OP_COUNT != 97"); -@@ -1191,7 +1191,7 @@ static const char * GGML_OP_SYMBOL[GGML_OP_COUNT] = { +-static_assert(GGML_OP_COUNT == 101, "GGML_OP_COUNT != 101"); ++static_assert(GGML_OP_COUNT == 102, "GGML_OP_COUNT != 102"); - "glu(x)", + static const char * GGML_OP_SYMBOL[GGML_OP_COUNT] = { + "none", +@@ -1214,9 +1216,11 @@ static const char * GGML_OP_SYMBOL[GGML_OP_COUNT] = { + "sgd(x)", -- "fused_relpos_attn(q,k,v,p)", + "glu(x)", ++ + "fused_attn(q,k,v,p)", }; - static_assert(GGML_OP_COUNT == 97, "GGML_OP_COUNT != 97"); -@@ -5375,9 +5375,9 @@ struct ggml_tensor * ggml_flash_attn_ext( +-static_assert(GGML_OP_COUNT == 101, "GGML_OP_COUNT != 101"); ++static_assert(GGML_OP_COUNT == 102, "GGML_OP_COUNT != 102"); + + static_assert(GGML_OP_POOL_COUNT == 2, "GGML_OP_POOL_COUNT != 2"); + +@@ -5538,6 +5542,169 @@ struct ggml_tensor * ggml_flash_attn_ext( return result; } --// ggml_fused_relpos_attn +// ggml_fused_attention - --struct ggml_tensor * ggml_fused_relpos_attn( ++ +static struct ggml_tensor * ggml_fused_attention_impl( - struct ggml_context * ctx, - struct ggml_tensor * q, - struct ggml_tensor * k, -@@ -5386,6 +5386,10 @@ struct ggml_tensor * ggml_fused_relpos_attn( - struct ggml_tensor * bias_u, - struct ggml_tensor * bias_v, - struct ggml_tensor * mask, ++ struct ggml_context * ctx, ++ struct ggml_tensor * q, ++ struct ggml_tensor * k, ++ struct ggml_tensor * v, ++ struct ggml_tensor * p, ++ struct ggml_tensor * bias_u, ++ struct ggml_tensor * bias_v, ++ struct ggml_tensor * mask, + struct ggml_tensor * kv_cache, + struct ggml_tensor * slot_ids, + struct ggml_tensor * cache_state, + int64_t cache_len, - float scale, - bool merge_heads) { - // Q/K/V/P may be arbitrary-strided views; the CUDA op derives addressing -@@ -5394,30 +5398,62 @@ struct ggml_tensor * ggml_fused_relpos_attn( - GGML_ASSERT(q->nb[0] == ggml_type_size(q->type)); - GGML_ASSERT(k->nb[0] == ggml_type_size(k->type)); - GGML_ASSERT(v->nb[0] == ggml_type_size(v->type)); -- GGML_ASSERT(p->nb[0] == ggml_type_size(p->type)); -- GGML_ASSERT(ggml_is_contiguous(bias_u)); -- GGML_ASSERT(ggml_is_contiguous(bias_v)); -- -- const int64_t d_k = q->ne[0]; -- const int64_t kv_len = k->ne[1]; -- const int64_t q_len = q->ne[1]; -- const int64_t n_head = q->ne[2]; -- -- GGML_ASSERT(k->ne[0] == d_k && v->ne[0] == d_k && p->ne[0] == d_k); -- GGML_ASSERT(bias_u->ne[0] == d_k && bias_v->ne[0] == d_k); ++ float scale, ++ bool merge_heads) { ++ // Q/K/V/P may be arbitrary-strided views; the CUDA op derives addressing ++ // from their nb[]. Only each d_k row must be contiguous (vectorized row ++ // loads). ++ GGML_ASSERT(q->nb[0] == ggml_type_size(q->type)); ++ GGML_ASSERT(k->nb[0] == ggml_type_size(k->type)); ++ GGML_ASSERT(v->nb[0] == ggml_type_size(v->type)); + const bool relative = p != NULL; + GGML_ASSERT(relative == (bias_u != NULL)); + GGML_ASSERT(relative == (bias_v != NULL)); @@ -2568,12 +1984,10 @@ index 80b5802f..4f5aeb09 100644 + const int64_t n_head = q->ne[2]; + + GGML_ASSERT(k->ne[0] == d_k && v->ne[0] == d_k); - // The kernel reads rel-pos rows (q_len-1)+j-i for j in [0,kv_len), i in - // [0,q_len) — i.e. rows [0, kv_len+q_len-1) — and takes its head stride - // from p->nb, so a longer table (e.g. precomputed for the full chunk - // length and reused by shorter tail chunks) is safe. -- GGML_ASSERT(p->ne[1] >= kv_len + q_len - 1); // rel-pos length -- GGML_ASSERT(v->ne[1] == kv_len); ++ // The kernel reads rel-pos rows (q_len-1)+j-i for j in [0,kv_len), i in ++ // [0,q_len) — i.e. rows [0, kv_len+q_len-1) — and takes its head stride ++ // from p->nb, so a longer table (e.g. precomputed for the full chunk ++ // length and reused by shorter tail chunks) is safe. + if (relative) { + GGML_ASSERT(p->ne[0] == d_k); + GGML_ASSERT(bias_u->ne[0] == d_k && bias_v->ne[0] == d_k); @@ -2596,13 +2010,9 @@ index 80b5802f..4f5aeb09 100644 + } else { + GGML_ASSERT(cache_len == 0); + } - if (mask) { - GGML_ASSERT(ggml_is_contiguous(mask)); - GGML_ASSERT(mask->ne[0] == kv_len); -- // One shared key mask, or one column per batch item (the cache-aware -- // streaming layout, where each stream's history has its own validity). -- GGML_ASSERT(mask->ne[1] == 1 || mask->ne[1] == q->ne[3]); -- GGML_ASSERT(mask->ne[2] == 1 && mask->ne[3] == 1); ++ if (mask) { ++ GGML_ASSERT(ggml_is_contiguous(mask)); ++ GGML_ASSERT(mask->ne[0] == kv_len); + if (kv_cache) { + // Cache-aware streaming: one shared key mask, or one column per + // batch item whose history has its own validity. @@ -2614,31 +2024,39 @@ index 80b5802f..4f5aeb09 100644 + GGML_ASSERT(mask->ne[1] == 1 || mask->ne[1] == q_len); + GGML_ASSERT(mask->ne[2] == 1 && (mask->ne[3] == 1 || mask->ne[3] == q->ne[3])); + } - } - - // Output mirrors q logically: [d_k, q_len, n_head, batch]. With -@@ -5434,8 +5470,9 @@ struct ggml_tensor * ggml_fused_relpos_attn( - } - - ggml_set_op_params(result, &scale, sizeof(scale)); ++ } ++ ++ // Output mirrors q logically: [d_k, q_len, n_head, batch]. With ++ // merge_heads the memory layout interleaves heads inside each query ++ // column (nb[2] = d_k, nb[1] = d_k*n_head) so that permute(0,2,1,3) of ++ // the result is a contiguous (d_k*n_head, q_len, batch) matrix. ++ struct ggml_tensor * result = ggml_new_tensor(ctx, GGML_TYPE_F32, 4, q->ne); ++ if (merge_heads) { ++ const size_t ts = ggml_type_size(result->type); ++ result->nb[0] = ts; ++ result->nb[2] = (size_t) d_k * ts; ++ result->nb[1] = (size_t) d_k * n_head * ts; ++ result->nb[3] = (size_t) d_k * n_head * q_len * ts; ++ } ++ ++ ggml_set_op_params(result, &scale, sizeof(scale)); + ggml_set_op_params_i32(result, 1, (int32_t) cache_len); - -- result->op = GGML_OP_FUSED_RELPOS_ATTN; ++ + result->op = GGML_OP_FUSED_ATTN; - result->src[0] = q; - result->src[1] = k; - result->src[2] = v; -@@ -5443,10 +5480,65 @@ struct ggml_tensor * ggml_fused_relpos_attn( - result->src[4] = bias_u; - result->src[5] = bias_v; - result->src[6] = mask; ++ result->src[0] = q; ++ result->src[1] = k; ++ result->src[2] = v; ++ result->src[3] = p; ++ result->src[4] = bias_u; ++ result->src[5] = bias_v; ++ result->src[6] = mask; + result->src[7] = kv_cache; + result->src[8] = slot_ids; + result->src[9] = cache_state; - - return result; - } - ++ ++ return result; ++} ++ +struct ggml_tensor * ggml_fused_relpos_attn( + struct ggml_context * ctx, + struct ggml_tensor * q, @@ -2690,7 +2108,6 @@ index 80b5802f..4f5aeb09 100644 + ctx, q, k, v, NULL, NULL, NULL, mask, kv_cache, slot_ids, cache_state, cache_len, scale, + merge_heads); +} -+ + void ggml_flash_attn_ext_set_prec( struct ggml_tensor * a, - enum ggml_prec prec) { diff --git a/ggml-patches/0030-cpu-fp16-conversion-im2col.patch b/patches/ggml-cpu-f16c-scalar-fp32-fp16-and-row-split-f16-im2col.patch similarity index 79% rename from ggml-patches/0030-cpu-fp16-conversion-im2col.patch rename to patches/ggml-cpu-f16c-scalar-fp32-fp16-and-row-split-f16-im2col.patch index 723a669..5b50a79 100644 --- a/ggml-patches/0030-cpu-fp16-conversion-im2col.patch +++ b/patches/ggml-cpu-f16c-scalar-fp32-fp16-and-row-split-f16-im2col.patch @@ -1,8 +1,20 @@ -diff --git a/src/ggml-cpu/ops.cpp b/src/ggml-cpu/ops.cpp -index 7485ba4f..af250c21 100644 ---- a/src/ggml-cpu/ops.cpp -+++ b/src/ggml-cpu/ops.cpp -@@ -6292,38 +6292,42 @@ static void ggml_compute_forward_im2col_f16( +From: Prabhsimran Singh +Subject: ggml-cpu: F16C scalar FP32->FP16 and row-split F16 im2col + +On x86 with F16C, scalar FP32 -> FP16 conversions use vcvtps2ph instead of the +portable bit-manipulating routine. The result matches the portable routine for +every non-NaN input (checked exhaustively; NaN payloads may differ). The F16 +im2col splits threads by output row instead of interleaving channels, so +threads no longer write interleaved segments of the same output rows. Both +speed up F16 convolutions such as the NanoCodec decoder on CPU. + +Upstream: candidate (needs CPU benchmarks). + +diff --git a/ggml/src/ggml-cpu/ops.cpp b/ggml/src/ggml-cpu/ops.cpp +index b05475768dfe92655a33bf876895b2becbf30488..53555e3cede08cef22792219657ba5bc00731d42 100644 +--- a/ggml/src/ggml-cpu/ops.cpp ++++ b/ggml/src/ggml-cpu/ops.cpp +@@ -6746,38 +6746,42 @@ static void ggml_compute_forward_im2col_f16( GGML_ASSERT(nb10 == ggml_type_size(src1->type)); // im2col: [N, IC, IH, IW] => [N, OH, OW, IC*KH*KW] @@ -74,11 +86,11 @@ index 7485ba4f..af250c21 100644 } } } -diff --git a/src/ggml-cpu/simd-mappings.h b/src/ggml-cpu/simd-mappings.h -index 0deda930..b698056b 100644 ---- a/src/ggml-cpu/simd-mappings.h -+++ b/src/ggml-cpu/simd-mappings.h -@@ -61,6 +61,9 @@ extern "C" { +diff --git a/ggml/src/ggml-cpu/simd-mappings.h b/ggml/src/ggml-cpu/simd-mappings.h +index 10ce4bfc593b3bee0fe05515af3b45fd70f3a108..ee3a30ac0f15ac8135feeb06366fae94b27237fc 100644 +--- a/ggml/src/ggml-cpu/simd-mappings.h ++++ b/ggml/src/ggml-cpu/simd-mappings.h +@@ -63,6 +63,9 @@ extern "C" { #define GGML_CPU_COMPUTE_FP16_TO_FP32(x) _cvtsh_ss(x) #define GGML_CPU_COMPUTE_FP32_TO_FP16(x) _cvtss_sh(x, 0) #endif diff --git a/patches/ggml-cuda-bf16-depthwise-conv-and-bias-round-epilogue.patch b/patches/ggml-cuda-bf16-depthwise-conv-and-bias-round-epilogue.patch new file mode 100644 index 0000000..aeb1175 --- /dev/null +++ b/patches/ggml-cuda-bf16-depthwise-conv-and-bias-round-epilogue.patch @@ -0,0 +1,191 @@ +From: Prabhsimran Singh +Subject: ggml-cuda: BF16 depthwise conv and bias/round epilogue + +- CONV_2D_DW reads BF16 kernels. +- Fuses bias add + BF16 rounding (+ ReLU) for the VoiceChat perception stem, + which keeps F32 carrier tensors with BF16-representable values. + +Upstream: not planned (model-specific BF16 emulation). + +diff --git a/ggml/src/ggml-cuda/conv2d-dw.cu b/ggml/src/ggml-cuda/conv2d-dw.cu +index db7ee6bf62077dc9ed518ad2d4d13fbe1a8f3926..572eb6bf3abb8812133c51456fac6564235a9b4b 100644 +--- a/ggml/src/ggml-cuda/conv2d-dw.cu ++++ b/ggml/src/ggml-cuda/conv2d-dw.cu +@@ -121,7 +121,8 @@ void ggml_cuda_op_conv2d_dw(ggml_backend_cuda_context & ctx, ggml_tensor * dst) + const ggml_tensor * input = dst->src[1]; + + GGML_ASSERT(input->type == GGML_TYPE_F32 && dst->type == GGML_TYPE_F32); +- GGML_ASSERT(kernel->type == GGML_TYPE_F32 || kernel->type == GGML_TYPE_F16); ++ GGML_ASSERT(kernel->type == GGML_TYPE_F32 || kernel->type == GGML_TYPE_F16 || ++ kernel->type == GGML_TYPE_BF16); + const void * w_d = kernel->data; + const float * x_d = (const float *) input->data; + float * y_d = (float *) dst->data; +@@ -153,6 +154,10 @@ void ggml_cuda_op_conv2d_dw(ggml_backend_cuda_context & ctx, ggml_tensor * dst) + conv2d_dw_kernel<<>>( + x_d, (const half *) w_d, y_d, in_w, in_h, out_w, out_h, kernel_w, kernel_h, stride_x, stride_y, + padding_x, padding_y, dilation_x, dilation_y, channels, batches); ++ } else if (kernel->type == GGML_TYPE_BF16) { ++ conv2d_dw_kernel<<>>( ++ x_d, (const nv_bfloat16 *) w_d, y_d, in_w, in_h, out_w, out_h, kernel_w, kernel_h, stride_x, ++ stride_y, padding_x, padding_y, dilation_x, dilation_y, channels, batches); + } else { + conv2d_dw_kernel<<>>( + x_d, (const float *) w_d, y_d, in_w, in_h, out_w, out_h, kernel_w, kernel_h, stride_x, stride_y, +@@ -163,6 +168,10 @@ void ggml_cuda_op_conv2d_dw(ggml_backend_cuda_context & ctx, ggml_tensor * dst) + conv2d_dw_kernel<<>>( + x_d, (const half *) w_d, y_d, in_w, in_h, out_w, out_h, kernel_w, kernel_h, stride_x, stride_y, + padding_x, padding_y, dilation_x, dilation_y, channels, batches); ++ } else if (kernel->type == GGML_TYPE_BF16) { ++ conv2d_dw_kernel<<>>( ++ x_d, (const nv_bfloat16 *) w_d, y_d, in_w, in_h, out_w, out_h, kernel_w, kernel_h, stride_x, ++ stride_y, padding_x, padding_y, dilation_x, dilation_y, channels, batches); + } else { + conv2d_dw_kernel<<>>( + x_d, (const float *) w_d, y_d, in_w, in_h, out_w, out_h, kernel_w, kernel_h, stride_x, stride_y, +diff --git a/ggml/src/ggml-cuda/ggml-cuda.cu b/ggml/src/ggml-cuda/ggml-cuda.cu +index 1d31dfa2f5a30488726b907873b3c66450af9cea..1924347f82ae913405ebdab956056c7d6f1a32f1 100644 +--- a/ggml/src/ggml-cuda/ggml-cuda.cu ++++ b/ggml/src/ggml-cuda/ggml-cuda.cu +@@ -1410,6 +1410,64 @@ struct batched_mul_mat_traits { + static inline auto convert_nc(ggml_type src_type) { return ggml_get_to_fp16_nc_cuda(src_type); } + }; + ++template ++static __global__ void f32_add_bias_round_bf16_to_f32( ++ const float * __restrict__ src, const float * __restrict__ bias, ++ float * __restrict__ dst, int64_t count, ++ int64_t ne0, int64_t ne1, int64_t ne2, ++ int64_t bne0, int64_t bne1, int64_t bne2, int64_t bne3) { ++#if defined(__CUDA_ARCH__) && __CUDA_ARCH__ >= GGML_CUDA_CC_AMPERE ++ const int64_t i = int64_t(blockIdx.x) * blockDim.x + threadIdx.x; ++ if (i >= count) { ++ return; ++ } ++ ++ int64_t coordinate = i; ++ const int64_t i0 = coordinate % ne0; ++ coordinate /= ne0; ++ const int64_t i1 = coordinate % ne1; ++ coordinate /= ne1; ++ const int64_t i2 = coordinate % ne2; ++ const int64_t i3 = coordinate / ne2; ++ const int64_t bias_index = ++ (bne0 == 1 ? 0 : i0) + bne0 * ( ++ (bne1 == 1 ? 0 : i1) + bne1 * ( ++ (bne2 == 1 ? 0 : i2) + bne2 * (bne3 == 1 ? 0 : i3))); ++ ++ float value = __bfloat162float(__float2bfloat16(src[i] + bias[bias_index])); ++ if constexpr (apply_relu) { ++ value = fmaxf(value, 0.0f); ++ } ++ dst[i] = value; ++#else ++ GGML_UNUSED_VARS(src, bias, dst, count, ne0, ne1, ne2, bne0, bne1, bne2, bne3); ++ NO_DEVICE_CODE; ++#endif ++} ++ ++static void launch_f32_add_bias_round_bf16_to_f32( ++ const ggml_tensor * src, const ggml_tensor * bias, ggml_tensor * dst, ++ bool apply_relu, cudaStream_t stream) { ++ constexpr int block_size = 256; ++ const int64_t count = ggml_nelements(src); ++ const int64_t blocks = (count + block_size - 1) / block_size; ++ if (apply_relu) { ++ f32_add_bias_round_bf16_to_f32<<>>( ++ static_cast(src->data), ++ static_cast(bias->data), ++ static_cast(dst->data), count, ++ src->ne[0], src->ne[1], src->ne[2], ++ bias->ne[0], bias->ne[1], bias->ne[2], bias->ne[3]); ++ } else { ++ f32_add_bias_round_bf16_to_f32<<>>( ++ static_cast(src->data), ++ static_cast(bias->data), ++ static_cast(dst->data), count, ++ src->ne[0], src->ne[1], src->ne[2], ++ bias->ne[0], bias->ne[1], bias->ne[2], bias->ne[3]); ++ } ++} ++ + template + static void ggml_cuda_mul_mat_cublas_impl(ggml_backend_cuda_context & ctx, const ggml_tensor * src0, const ggml_tensor * src1, ggml_tensor * dst) { + using traits = batched_mul_mat_traits; +@@ -1694,6 +1752,7 @@ static void ggml_cuda_mul_mat_cublas(ggml_backend_cuda_context & ctx, const ggml + } + + ++ + static bool ggml_cuda_should_fuse_mul_mat(const ggml_tensor * ffn_up, + const ggml_tensor * ffn_gate, + const ggml_tensor * glu, +@@ -3889,6 +3948,58 @@ static int ggml_cuda_try_fuse(ggml_backend_cuda_context * cuda_ctx, ggml_cgraph + } + } + ++ // Convolution epilogue: add a broadcast bias and apply the model's BF16 ++ // output boundary in one pass. ReLU is folded in when it immediately ++ // consumes that rounded output. The graph retains F32 carrier tensors for ++ // operators that do not natively accept BF16, but every stored value is ++ // exactly BF16-representable. ++ bool round_with_relu = false; ++ bool round_pattern = false; ++ if (ggml_can_fuse(cgraph, i, ++ { GGML_OP_ADD, GGML_OP_CPY, GGML_OP_CPY, GGML_OP_UNARY })) { ++ ggml_tensor * relu_node = cgraph->nodes[i + 3]; ++ round_with_relu = ggml_get_unary_op(relu_node) == GGML_UNARY_OP_RELU; ++ round_pattern = round_with_relu; ++ } ++ if (!round_pattern && ++ ggml_can_fuse(cgraph, i, { GGML_OP_ADD, GGML_OP_CPY, GGML_OP_CPY })) { ++ round_pattern = true; ++ } ++ if (round_pattern) { ++ ggml_tensor * add_node = cgraph->nodes[i]; ++ ggml_tensor * bf16_node = cgraph->nodes[i + 1]; ++ ggml_tensor * rounded_node = cgraph->nodes[i + 2]; ++ ggml_tensor * output_node = round_with_relu ? cgraph->nodes[i + 3] : rounded_node; ++ const ggml_tensor * activation = ++ ggml_are_same_shape(add_node, add_node->src[0]) ? add_node->src[0] : ++ ggml_are_same_shape(add_node, add_node->src[1]) ? add_node->src[1] : nullptr; ++ const ggml_tensor * bias = ++ activation == add_node->src[0] ? add_node->src[1] : ++ activation == add_node->src[1] ? add_node->src[0] : nullptr; ++ bool broadcast_ok = bias != nullptr; ++ for (int d = 0; broadcast_ok && d < GGML_MAX_DIMS; ++d) { ++ broadcast_ok = bias->ne[d] == 1 || bias->ne[d] == activation->ne[d]; ++ } ++ const bool eligible = ++ native_bf16 && activation != nullptr && bias != nullptr && ++ bf16_node->src[0] == add_node && rounded_node->src[0] == bf16_node && ++ (!round_with_relu || output_node->src[0] == rounded_node) && ++ activation->type == GGML_TYPE_F32 && bias->type == GGML_TYPE_F32 && ++ add_node->type == GGML_TYPE_F32 && bf16_node->type == GGML_TYPE_BF16 && ++ rounded_node->type == GGML_TYPE_F32 && output_node->type == GGML_TYPE_F32 && ++ broadcast_ok && ggml_is_contiguous(activation) && ggml_is_contiguous(bias) && ++ ggml_is_contiguous(add_node) && ggml_is_contiguous(bf16_node) && ++ ggml_is_contiguous(rounded_node) && ggml_is_contiguous(output_node) && ++ ggml_are_same_shape(activation, output_node); ++ const int output_index = i + (round_with_relu ? 3 : 2); ++ if (eligible && ggml_cuda_check_fusion_memory_ranges( ++ cgraph, i, round_with_relu ? 4 : 3, &output_index, 1)) { ++ launch_f32_add_bias_round_bf16_to_f32( ++ activation, bias, output_node, round_with_relu, cuda_ctx->stream()); ++ return round_with_relu ? 3 : 2; ++ } ++ } ++ + // multi-(add or mul) + if (node->op == GGML_OP_ADD || node->op == GGML_OP_MUL) { + int n_fuse = 0; +@@ -5987,7 +6098,8 @@ static bool ggml_backend_cuda_device_supports_op(ggml_backend_dev_t dev, const g + case GGML_OP_CONV_2D: + return (ggml_is_contiguous(op->src[0]) && ggml_is_contiguous(op->src[1])); + case GGML_OP_CONV_2D_DW: +- return (op->src[0]->type == GGML_TYPE_F32 || op->src[0]->type == GGML_TYPE_F16) && ++ return (op->src[0]->type == GGML_TYPE_F32 || op->src[0]->type == GGML_TYPE_F16 || ++ op->src[0]->type == GGML_TYPE_BF16) && + op->src[1]->type == GGML_TYPE_F32 && op->type == GGML_TYPE_F32; + case GGML_OP_CONV_TRANSPOSE_2D: + case GGML_OP_POOL_1D: diff --git a/ggml-patches/0011-cuda-graph-shape-key.patch b/patches/ggml-cuda-cuda-graph-cache-keys-replay-updates-and-eviction-knobs.patch similarity index 55% rename from ggml-patches/0011-cuda-graph-shape-key.patch rename to patches/ggml-cuda-cuda-graph-cache-keys-replay-updates-and-eviction-knobs.patch index 963fe9a..3067a7a 100644 --- a/ggml-patches/0011-cuda-graph-shape-key.patch +++ b/patches/ggml-cuda-cuda-graph-cache-keys-replay-updates-and-eviction-knobs.patch @@ -1,8 +1,38 @@ -diff --git a/src/ggml-cuda/common.cuh b/src/ggml-cuda/common.cuh -index 01b239f1..610fdf37 100644 ---- a/src/ggml-cuda/common.cuh -+++ b/src/ggml-cuda/common.cuh -@@ -1220,6 +1220,23 @@ struct ggml_cuda_graph { +From: Prabhsimran Singh +Subject: ggml-cuda: CUDA graph cache keys, replay updates and eviction knobs + +- Key cached graph executables by graph identity and structural shape, so + allocator reuse cannot select an executable for a different shape. +- Refresh node parameters when a cached graph is replayed, so per-stream + state pointers that change every call keep the graph resident. +- GGML_CUDA_GRAPH_EVICT_AFTER_MS, read when a backend is created, sets how + long an idle graph is kept (default 10 s; 0 keeps graphs resident). + +Upstream: the shape key is the follow-up requested in +ggml-org/llama.cpp#28243 and the eviction knobs are a small candidate; replay +updates are NeMo-specific. + +diff --git a/ggml/src/ggml-cuda/common.cuh b/ggml/src/ggml-cuda/common.cuh +index f5f4465b5e0cade7888d67728b71d8d745eb1939..fbd855dd5ca483874abba3e17db15b77e1f15821 100644 +--- a/ggml/src/ggml-cuda/common.cuh ++++ b/ggml/src/ggml-cuda/common.cuh +@@ -27,6 +27,7 @@ + #include + #include + #include ++#include + #include + #include + #include +@@ -1290,6 +1291,7 @@ struct ggml_cuda_graph { + size_t num_nodes = 0; + std::vector nodes; + bool disable_due_to_gpu_arch = false; ++ bool warmup_started = false; + bool warmup_complete = false; + uint64_t uid = 0; + int64_t last_used_time = 0; +@@ -1308,6 +1310,23 @@ struct ggml_cuda_graph { #endif }; @@ -26,7 +56,7 @@ index 01b239f1..610fdf37 100644 struct ggml_cuda_concurrent_event { std::vector join_events; cudaEvent_t fork_event = nullptr; -@@ -1382,9 +1399,9 @@ struct ggml_backend_cuda_context { +@@ -1472,20 +1491,34 @@ struct ggml_backend_cuda_context { int curr_stream_no = 0; #ifdef USE_CUDA_GRAPH @@ -39,16 +69,36 @@ index 01b239f1..610fdf37 100644 int64_t last_graph_eviction_sweep = 0; -@@ -1402,7 +1419,7 @@ struct ggml_backend_cuda_context { - return (int64_t) parsed_ms * 1000; - } - - ggml_cuda_graph * cuda_graph(const void * first_node_ptr) { ++ // Read when the backend is created, so an application can set it first. ++ // GGML_CUDA_GRAPH_EVICT_AFTER_MS=0 keeps captured graphs resident. ++ const int64_t graph_evict_after_us = graph_evict_after_us_from_env(); ++ ++ static int64_t graph_evict_after_us_from_env() { ++ const char * value = getenv("GGML_CUDA_GRAPH_EVICT_AFTER_MS"); ++ if (value == nullptr || value[0] == '\0') { ++ return 10'000'000; ++ } ++ char * end = nullptr; ++ const long long ms = std::strtoll(value, &end, 10); ++ return end == value || *end != '\0' || ms < 0 ? 10'000'000 : (int64_t) ms * 1000; ++ } ++ + ggml_cuda_graph * cuda_graph(const ggml_cuda_graph_key & graph_key) { const int64_t time_now = ggml_time_us(); - static const int64_t sweep_interval_us = - cuda_graph_env_ms_to_us("GGML_CUDA_GRAPH_SWEEP_MS", 5000); -@@ -1423,9 +1440,9 @@ struct ggml_backend_cuda_context { + +- // sweep every 5s, evicting cuda graphs unused for >=10s +- if (time_now - last_graph_eviction_sweep >= 5'000'000) { ++ // sweep every 5s, evicting cuda graphs unused for graph_evict_after_us ++ if (graph_evict_after_us > 0 && time_now - last_graph_eviction_sweep >= 5'000'000) { + last_graph_eviction_sweep = time_now; + for (auto it = cuda_graphs.begin(); it != cuda_graphs.end(); ) { +- if (time_now - it->second->last_used_time >= 10'000'000) { ++ if (time_now - it->second->last_used_time >= graph_evict_after_us) { + it = cuda_graphs.erase(it); + } else { + ++it; +@@ -1493,9 +1526,9 @@ struct ggml_backend_cuda_context { } } @@ -60,11 +110,11 @@ index 01b239f1..610fdf37 100644 } it->second->last_used_time = time_now; return it->second.get(); -diff --git a/src/ggml-cuda/ggml-cuda.cu b/src/ggml-cuda/ggml-cuda.cu -index 6ee36c12..e8f6c10f 100644 ---- a/src/ggml-cuda/ggml-cuda.cu -+++ b/src/ggml-cuda/ggml-cuda.cu -@@ -3426,8 +3426,65 @@ static bool ggml_cuda_graph_check_compability(ggml_cgraph * cgraph) { +diff --git a/ggml/src/ggml-cuda/ggml-cuda.cu b/ggml/src/ggml-cuda/ggml-cuda.cu +index 1c0c25329d832ca97c327df4ab2329dddc2d5013..ee47bc09e20c299d8813080a48415a2e958d83d6 100644 +--- a/ggml/src/ggml-cuda/ggml-cuda.cu ++++ b/ggml/src/ggml-cuda/ggml-cuda.cu +@@ -2606,19 +2606,92 @@ static bool ggml_cuda_graph_check_compability(ggml_cgraph * cgraph) { return use_cuda_graph; } @@ -75,8 +125,9 @@ index 6ee36c12..e8f6c10f 100644 + // serialized hundreds of thousands of dependent multiplies for large + // encoder graphs, and get_key runs in both graph_optimize and compute. + hash ^= value + 0x9e3779b97f4a7c15ULL + (hash << 6) + (hash >> 2); -+} -+ + } + +-static bool ggml_cuda_graph_update_required(ggml_backend_cuda_context * cuda_ctx, ggml_cgraph * cgraph) { +static void ggml_cuda_graph_hash_tensor_descriptor(uint64_t & hash, const ggml_tensor * tensor) { + ggml_cuda_graph_hash_mix(hash, reinterpret_cast(tensor)); + if (tensor == nullptr) { @@ -129,27 +180,26 @@ index 6ee36c12..e8f6c10f 100644 + } + + return { cgraph->nodes[0], signature }; - } - - static const char * ggml_cuda_graph_tensor_name(const ggml_tensor * tensor) { -@@ -3438,19 +3495,19 @@ static void ggml_cuda_graph_log_event( - const char * func, - const char * event, - const ggml_cgraph * cgraph, -- const void * graph_key) { ++} ++ ++static const char * ggml_cuda_graph_tensor_name(const ggml_tensor * tensor) { ++ return tensor && tensor->name[0] ? tensor->name : "(unnamed)"; ++} ++ ++static void ggml_cuda_graph_log_event( ++ const char * func, ++ const char * event, ++ const ggml_cgraph * cgraph, + const ggml_cuda_graph_key & graph_key) { - const int n_nodes = cgraph ? cgraph->n_nodes : 0; - const ggml_tensor * first = n_nodes > 0 ? cgraph->nodes[0] : nullptr; - const ggml_tensor * last = n_nodes > 0 ? cgraph->nodes[n_nodes - 1] : nullptr; -- GGML_LOG_DEBUG("%s: CUDA graph %s key=%p uid=%" PRIu64 " nodes=%d first=%s last=%s\n", -- func, event, graph_key, cgraph ? cgraph->uid : 0, n_nodes, ++ const int n_nodes = cgraph ? cgraph->n_nodes : 0; ++ const ggml_tensor * first = n_nodes > 0 ? cgraph->nodes[0] : nullptr; ++ const ggml_tensor * last = n_nodes > 0 ? cgraph->nodes[n_nodes - 1] : nullptr; + GGML_LOG_DEBUG("%s: CUDA graph %s key=%p signature=%016" PRIx64 " uid=%" PRIu64 + " nodes=%d first=%s last=%s\n", + func, event, graph_key.first_node_ptr, graph_key.signature, cgraph ? cgraph->uid : 0, n_nodes, - ggml_cuda_graph_tensor_name(first), ggml_cuda_graph_tensor_name(last)); - } - --static bool ggml_cuda_graph_update_required(ggml_backend_cuda_context * cuda_ctx, ggml_cgraph * cgraph) { ++ ggml_cuda_graph_tensor_name(first), ggml_cuda_graph_tensor_name(last)); ++} ++ +static bool ggml_cuda_graph_update_required(ggml_backend_cuda_context * cuda_ctx, ggml_cgraph * cgraph, const ggml_cuda_graph_key & graph_key) { bool res = false; @@ -157,7 +207,12 @@ index 6ee36c12..e8f6c10f 100644 ggml_cuda_graph * graph = cuda_ctx->cuda_graph(graph_key); if (cgraph->uid != 0 && -@@ -3489,7 +3546,7 @@ static bool ggml_cuda_graph_update_required(ggml_backend_cuda_context * cuda_ctx + cgraph->uid == graph->uid) { +- GGML_LOG_DEBUG("CUDA Graph id %zu reused\n", cgraph->uid); + GGML_ASSERT((int)graph->node_props.size() == cgraph->n_nodes); + return false; + } +@@ -2652,7 +2725,7 @@ static bool ggml_cuda_graph_update_required(ggml_backend_cuda_context * cuda_ctx return res; } @@ -166,7 +221,7 @@ index 6ee36c12..e8f6c10f 100644 ggml_cuda_graph * graph = cuda_ctx->cuda_graph(graph_key); #if CUDART_VERSION >= 12000 -@@ -4736,7 +4793,7 @@ static int ggml_cuda_try_fuse(ggml_backend_cuda_context * cuda_ctx, ggml_cgraph +@@ -4321,7 +4394,7 @@ static int ggml_cuda_try_fuse(ggml_backend_cuda_context * cuda_ctx, ggml_cgraph return 0; } @@ -175,7 +230,7 @@ index 6ee36c12..e8f6c10f 100644 bool graph_evaluated_or_captured = false; // flag used to determine whether it is an integrated_gpu -@@ -4951,7 +5008,7 @@ static void ggml_cuda_graph_evaluate_and_capture(ggml_backend_cuda_context * cud +@@ -4540,7 +4613,7 @@ static void ggml_cuda_graph_evaluate_and_capture(ggml_backend_cuda_context * cud } #ifdef USE_CUDA_GRAPH @@ -184,7 +239,7 @@ index 6ee36c12..e8f6c10f 100644 ggml_cuda_graph * graph = cuda_ctx->cuda_graph(graph_key); if (graph->graph == nullptr) { -@@ -4974,7 +5031,7 @@ static enum ggml_status ggml_backend_cuda_graph_compute(ggml_backend_t backend, +@@ -4563,7 +4636,7 @@ static enum ggml_status ggml_backend_cuda_graph_compute(ggml_backend_t backend, bool use_cuda_graph = false; bool cuda_graph_update_required = false; @@ -193,17 +248,56 @@ index 6ee36c12..e8f6c10f 100644 #ifdef USE_CUDA_GRAPH graph_key = ggml_cuda_graph_get_key(cgraph); -@@ -4985,7 +5042,7 @@ static enum ggml_status ggml_backend_cuda_graph_compute(ggml_backend_t backend, +@@ -4574,27 +4647,27 @@ static enum ggml_status ggml_backend_cuda_graph_compute(ggml_backend_t backend, if (graph->is_enabled()) { const bool graph_compatible = ggml_cuda_graph_check_compability(cgraph); if (graph_compatible) { - const bool properties_changed = ggml_cuda_graph_update_required(cuda_ctx, cgraph); +- +- if (!graph->warmup_complete) { +- // Warmup: need at least 2 calls with no property change on the 2nd call +- if (!properties_changed) { +- graph->warmup_complete = true; +- GGML_LOG_DEBUG("%s: CUDA graph warmup complete\n", __func__); +- use_cuda_graph = true; +- cuda_graph_update_required = true; +- } +- // else: properties changed or first call - execute directly (use_cuda_graph stays false) + const bool properties_changed = ggml_cuda_graph_update_required(cuda_ctx, cgraph, graph_key); - - if (!graph->warmup_complete) { - // Warmup: need at least 2 calls with no property change on the 2nd call -@@ -5055,7 +5112,7 @@ static void ggml_backend_cuda_graph_optimize(ggml_backend_t backend, ggml_cgraph - ggml_backend_cuda_context * cuda_ctx = (ggml_backend_cuda_context *) backend->context; ++ ++ if (!graph->warmup_started) { ++ // Execute one call directly so lazy kernel/library setup happens ++ // outside stream capture. Cache/state addresses are expected to ++ // change between streaming calls, so stability is not a valid ++ // prerequisite for completing warmup. ++ graph->warmup_started = true; ++ } else if (!graph->warmup_complete) { ++ graph->warmup_complete = true; ++ ggml_cuda_graph_log_event(__func__, "warmup complete", cgraph, graph_key); ++ use_cuda_graph = true; ++ cuda_graph_update_required = true; + } else { +- // Post-warmup: normal CUDA graph operation +- if (properties_changed) { +- // Properties changed - reset warmup, execute directly until stable again +- graph->warmup_complete = false; +- GGML_LOG_DEBUG("%s: CUDA graph warmup reset\n", __func__); +- } else { +- use_cuda_graph = true; +- cuda_graph_update_required = graph->instance == nullptr; +- } ++ // CUDA graph topology is stable even when tensor/cache pointers ++ // rotate. Re-capture the current parameters and update the ++ // executable instead of invalidating it and falling back to ++ // repeated direct launches. cudaGraphExecUpdate() below safely ++ // re-instantiates if CUDA reports a topology incompatibility. ++ use_cuda_graph = true; ++ cuda_graph_update_required = properties_changed || graph->instance == nullptr; + } + } + } +@@ -4733,7 +4806,7 @@ static void ggml_backend_cuda_graph_optimize(ggml_backend_t backend, ggml_cgraph + } #ifdef USE_CUDA_GRAPH - const void * graph_key = ggml_cuda_graph_get_key(cgraph); diff --git a/ggml-patches/0017-cuda-stream-interop.patch b/patches/ggml-cuda-expose-the-backend-stream-and-cuda-graph-controls.patch similarity index 50% rename from ggml-patches/0017-cuda-stream-interop.patch rename to patches/ggml-cuda-expose-the-backend-stream-and-cuda-graph-controls.patch index f321f12..2361811 100644 --- a/ggml-patches/0017-cuda-stream-interop.patch +++ b/patches/ggml-cuda-expose-the-backend-stream-and-cuda-graph-controls.patch @@ -1,8 +1,27 @@ -diff --git a/include/ggml-cuda.h b/include/ggml-cuda.h -index 5436c7ef..166ef8c1 100644 ---- a/include/ggml-cuda.h -+++ b/include/ggml-cuda.h -@@ -24,6 +24,20 @@ GGML_BACKEND_API ggml_backend_t ggml_backend_cuda_init(int device); +From: anand-nv <105917641+anand-nv@users.noreply.github.com> +Subject: ggml-cuda: expose the backend stream and CUDA graph controls + +ggml_backend_cuda_get_stream() and ggml_backend_cuda_get_graph_template() let +the MagpieTTS CUDA sampler run on ggml's stream and compose ggml's captured +graph into its own per-frame graph. ggml_backend_cuda_set_stream_priority() +lets a latency-critical backend (the MagpieTTS decoder) run its streams at a +higher CUDA priority than a concurrent worker backend (the NanoCodec vocoder). +ggml_backend_cuda_set_graphs_enabled() opts one backend out of CUDA graph +capture and replay: a side backend that computes one-shot graphs (the MagpieTTS +prefetch backend) gains nothing from capture, and capture and instantiate hold +the driver lock long enough to stall launches issued by other threads. +GGML_CUDA_DISABLE_GRAPHS can only turn graphs off process-wide. + +Upstream: get_stream, set_stream_priority and set_graphs_enabled are +candidates; the graph template accessor is NeMo-specific. + +Co-authored-by: Prabhsimran Singh + +diff --git a/ggml/include/ggml-cuda.h b/ggml/include/ggml-cuda.h +index 1cd81eeaebcdf4abcd46c87ba1a9a46e275aa12b..fbbfaf56abd8938910ea71a649baf02de7aed087 100644 +--- a/ggml/include/ggml-cuda.h ++++ b/ggml/include/ggml-cuda.h +@@ -24,6 +24,24 @@ GGML_BACKEND_API ggml_backend_t ggml_backend_cuda_init(int device); GGML_BACKEND_API bool ggml_backend_is_cuda(ggml_backend_t backend); @@ -12,6 +31,10 @@ index 5436c7ef..166ef8c1 100644 +// Set the CUDA stream priority used by this backend's streams (0 default; negative = higher, +// clamped to the device range). Streams already created are recreated. +GGML_BACKEND_API void ggml_backend_cuda_set_stream_priority(ggml_backend_t backend, int priority); ++// Enable/disable CUDA graph capture and replay for this backend's graph_compute (default on). ++// Side backends running one-shot graphs should turn it off: capture/instantiate hold the driver ++// lock and stall kernel launches issued by other threads. ++GGML_BACKEND_API void ggml_backend_cuda_set_graphs_enabled(ggml_backend_t backend, bool enabled); + +// Returns the backend-owned native CUDA/HIP graph template for a stable GGML graph after its +// normal warm-up/capture has completed. The opaque handle is borrowed and is intended for runtime @@ -23,11 +46,22 @@ index 5436c7ef..166ef8c1 100644 // device buffer GGML_BACKEND_API ggml_backend_buffer_type_t ggml_backend_cuda_buffer_type(int device); -diff --git a/src/ggml-cuda/common.cuh b/src/ggml-cuda/common.cuh -index 610fdf37..cdb17991 100644 ---- a/src/ggml-cuda/common.cuh -+++ b/src/ggml-cuda/common.cuh -@@ -1479,10 +1479,22 @@ struct ggml_backend_cuda_context { +diff --git a/ggml/src/ggml-cuda/common.cuh b/ggml/src/ggml-cuda/common.cuh +index fbd855dd5ca483874abba3e17db15b77e1f15821..a8bbc07b63d7057a54c7a01a7470e8f8a447b904 100644 +--- a/ggml/src/ggml-cuda/common.cuh ++++ b/ggml/src/ggml-cuda/common.cuh +@@ -1489,6 +1489,10 @@ struct ggml_backend_cuda_context { + size_t cublas_workspace_sizes[GGML_CUDA_MAX_DEVICES] = {0}; + + int curr_stream_no = 0; ++ // Per-backend opt-out of CUDA graph capture/replay (ggml_backend_cuda_set_graphs_enabled): ++ // one-shot graphs computed on a side backend gain nothing from capture, and the capture and ++ // instantiate calls hold the driver lock long enough to stall launches on other threads. ++ bool graphs_enabled = true; + + #ifdef USE_CUDA_GRAPH + // The structural signature separates batch shapes and graph topologies even +@@ -1565,10 +1569,22 @@ struct ggml_backend_cuda_context { ~ggml_backend_cuda_context(); @@ -51,11 +85,29 @@ index 610fdf37..cdb17991 100644 } return streams[device][stream]; } -diff --git a/src/ggml-cuda/ggml-cuda.cu b/src/ggml-cuda/ggml-cuda.cu -index c83735a6..ad0f5e25 100644 ---- a/src/ggml-cuda/ggml-cuda.cu -+++ b/src/ggml-cuda/ggml-cuda.cu -@@ -5515,6 +5515,51 @@ bool ggml_backend_is_cuda(ggml_backend_t backend) { +diff --git a/ggml/src/ggml-cuda/ggml-cuda.cu b/ggml/src/ggml-cuda/ggml-cuda.cu +index 29bddaff6da1ee9f4f6aec73a89b4078641845ee..1d31dfa2f5a30488726b907873b3c66450af9cea 100644 +--- a/ggml/src/ggml-cuda/ggml-cuda.cu ++++ b/ggml/src/ggml-cuda/ggml-cuda.cu +@@ -4853,7 +4853,7 @@ static enum ggml_status ggml_backend_cuda_graph_compute(ggml_backend_t backend, + ggml_cuda_graph_set_enabled(cuda_ctx, graph_key); + + ggml_cuda_graph * graph = cuda_ctx->cuda_graph(graph_key); +- if (graph->is_enabled()) { ++ if (graph->is_enabled() && cuda_ctx->graphs_enabled) { + const bool graph_compatible = ggml_cuda_graph_check_compability(cgraph); + if (graph_compatible) { + const bool properties_changed = ggml_cuda_graph_update_required(cuda_ctx, cgraph, graph_key); +@@ -5016,7 +5016,7 @@ static void ggml_backend_cuda_graph_optimize(ggml_backend_t backend, ggml_cgraph + + #ifdef USE_CUDA_GRAPH + const ggml_cuda_graph_key graph_key = ggml_cuda_graph_get_key(cgraph); +- const bool use_cuda_graph = ggml_cuda_graph_set_enabled(cuda_ctx, graph_key); ++ const bool use_cuda_graph = ggml_cuda_graph_set_enabled(cuda_ctx, graph_key) && cuda_ctx->graphs_enabled; + #else + const bool use_cuda_graph = false; + GGML_UNUSED(cuda_ctx); +@@ -5285,6 +5285,59 @@ bool ggml_backend_is_cuda(ggml_backend_t backend) { return backend != NULL && ggml_guid_matches(backend->guid, ggml_backend_cuda_guid()); } @@ -85,6 +137,14 @@ index c83735a6..ad0f5e25 100644 + } +} + ++void ggml_backend_cuda_set_graphs_enabled(ggml_backend_t backend, bool enabled) { ++ if (!ggml_backend_is_cuda(backend)) { ++ return; ++ } ++ ggml_backend_cuda_context * cuda_ctx = (ggml_backend_cuda_context *) backend->context; ++ cuda_ctx->graphs_enabled = enabled; ++} ++ +void * ggml_backend_cuda_get_graph_template( + ggml_backend_t backend, const struct ggml_cgraph * cgraph) { + if (!ggml_backend_is_cuda(backend) || cgraph == nullptr) { diff --git a/ggml-patches/0010-cuda-pad-large-batch-grid.patch b/patches/ggml-cuda-flatten-pad-launches-for-large-batches.patch similarity index 78% rename from ggml-patches/0010-cuda-pad-large-batch-grid.patch rename to patches/ggml-cuda-flatten-pad-launches-for-large-batches.patch index d02244e..b188e2a 100644 --- a/ggml-patches/0010-cuda-pad-large-batch-grid.patch +++ b/patches/ggml-cuda-flatten-pad-launches-for-large-batches.patch @@ -1,7 +1,17 @@ -diff --git a/src/ggml-cuda/pad.cu b/src/ggml-cuda/pad.cu ---- a/src/ggml-cuda/pad.cu -+++ b/src/ggml-cuda/pad.cu +From: Prabhsimran Singh +Subject: ggml-cuda: flatten PAD launches for large batches + +PAD launched ne1 x (ne2*ne3) blocks, so large batches or long inputs exceeded +the 65535 grid.y/grid.z limit. Flatten the launch into grid.x. + +Upstream: candidate (add test_pad cases beyond 65535). + +diff --git a/ggml/src/ggml-cuda/pad.cu b/ggml/src/ggml-cuda/pad.cu +index 31cd00f77816fe6f5ffdda0428289e2aa1a25f9b..40b1b496b011eb4f85c35af3d1c29d223e2639e4 100644 +--- a/ggml/src/ggml-cuda/pad.cu ++++ b/ggml/src/ggml-cuda/pad.cu @@ -12,14 +12,19 @@ static __global__ void pad_f32(const float * src, size_t s00, size_t s01, size_t + const int lp2, const int rp2, const int lp3, const int rp3, const int ne0, const int ne1, const int ne2, const int ne3, const bool circular) { - // blockIdx.z: i3*ne2+i2 @@ -28,7 +38,6 @@ diff --git a/src/ggml-cuda/pad.cu b/src/ggml-cuda/pad.cu if (i0 >= ne0 || i1 >= ne1 || i2 >= ne2 || i3 >= ne3) { return; - } @@ -66,9 +71,11 @@ static void pad_f32_cuda(const float * src, size_t s00, size_t s01, size_t s02, const int lp2, const int rp2, const int lp3, const int rp3, const int ne0, const int ne1, const int ne2, const int ne3, diff --git a/patches/ggml-cuda-flatten-shared-weight-cublas-gemms-fuse-silu-to-bf16.patch b/patches/ggml-cuda-flatten-shared-weight-cublas-gemms-fuse-silu-to-bf16.patch new file mode 100644 index 0000000..6225d7c --- /dev/null +++ b/patches/ggml-cuda-flatten-shared-weight-cublas-gemms-fuse-silu-to-bf16.patch @@ -0,0 +1,146 @@ +From: Prabhsimran Singh +Subject: ggml-cuda: flatten shared-weight cuBLAS GEMMs, fuse SiLU to BF16 + +- A 2-D weight applied to contiguous [K,T,B,...] activations is one GEMM; + issue it as such instead of a pointer-array batched GEMM. +- Fuse SILU followed by a cast to BF16 into one kernel. + +Upstream: the GEMM flatten is a candidate; the SiLU/BF16 fusion is NeMo-specific. + +diff --git a/ggml/src/ggml-cuda/ggml-cuda.cu b/ggml/src/ggml-cuda/ggml-cuda.cu +index 141caeca47a748ccb3916f98d980d220ee63217f..4c0c0633dd66cc6f717d69cd9f74a1f5d0acc404 100644 +--- a/ggml/src/ggml-cuda/ggml-cuda.cu ++++ b/ggml/src/ggml-cuda/ggml-cuda.cu +@@ -1542,6 +1542,9 @@ static void ggml_cuda_mul_mat_cublas_impl(ggml_backend_cuda_context & ctx, const + const int64_t r2 = ne12/ne02; + const int64_t r3 = ne13/ne03; + ++ const bool flatten_shared_weight = ++ ne02 == 1 && ne03 == 1 && ne12*ne13 > 1 && s12 == ne11*s11 && s13 == ne12*s12; ++ + // Theoretically cublasGemmStridedBatchedEx would always work, even for a single matrix. + // However, for some old NVIDIA and AMD GPUs the strided/Ex GEMM is much slower, + // probably because the internal kernel selection logic is suboptimal. +@@ -1561,6 +1564,21 @@ static void ggml_cuda_mul_mat_cublas_impl(ggml_backend_cuda_context & ctx, const + beta, dst_ptr, cu_data_type, ne0, + cu_compute_type, + CUBLAS_GEMM_DEFAULT_TENSOR_OP)); ++ } else if (flatten_shared_weight) { ++ // A single shared weight matrix broadcast over contiguous outer activation ++ // dimensions is one large GEMM, not a collection of independent skinny ++ // GEMMs. [K,T,B,D] and [K,T*B*D] have identical storage order; likewise ++ // for the [M,T,B,D] output. Flattening only the cuBLAS geometry therefore ++ // preserves the graph-visible tensor layout while improving weight reuse ++ // and tensor-core occupancy and avoiding pointer-array setup. ++ CUBLAS_CHECK( ++ cublasGemmEx(cublas_h, CUBLAS_OP_T, CUBLAS_OP_N, ++ ne01, ne11*ne12*ne13, ne10, ++ alpha, src0_ptr, cu_data_type_a, s01, ++ src1_ptr, cu_data_type_b, s11, ++ beta, dst_ptr, cu_data_type, ne0, ++ cu_compute_type, ++ CUBLAS_GEMM_DEFAULT_TENSOR_OP)); + } else if (r2 == 1 && r3 == 1 && is_src0_cont_2 && is_src1_cont_2) { + // with a [0, 2, 1, 3] perm. and ne02==1 the matrix strides need to be determined from dim 3: + const int64_t sma = ne02 == 1 ? s03 : s02; +@@ -1675,6 +1693,7 @@ static void ggml_cuda_mul_mat_cublas(ggml_backend_cuda_context & ctx, const ggml + } + } + ++ + static bool ggml_cuda_should_fuse_mul_mat(const ggml_tensor * ffn_up, + const ggml_tensor * ffn_gate, + const ggml_tensor * glu, +@@ -3573,6 +3592,10 @@ static int ggml_cuda_try_fuse(ggml_backend_cuda_context * cuda_ctx, ggml_cgraph + } + + ggml_tensor * node = cgraph->nodes[i]; ++ const int cc = ggml_cuda_info().devices[cuda_ctx->device].cc; ++ const bool native_bf16 = ++ GGML_CUDA_CC_IS_NVIDIA(cc) && cc >= GGML_CUDA_CC_AMPERE && ++ ggml_cuda_highest_compiled_arch(cc) >= GGML_CUDA_CC_AMPERE; + + if (node->op == GGML_OP_MUL) { + ggml_cuda_moe_weighted_reduction_match match; +@@ -4208,6 +4231,25 @@ static int ggml_cuda_try_fuse(ggml_backend_cuda_context * cuda_ctx, ggml_cgraph + return fused_node_count - 1; + } + ++ // A BF16 projection immediately following SiLU would otherwise launch an ++ // F32 SiLU kernel and then a separate F32-to-BF16 input conversion. Emit ++ // the same rounded BF16 activation directly in one pass. ++ if (ggml_can_fuse(cgraph, i, { GGML_OP_UNARY, GGML_OP_CPY })) { ++ const ggml_tensor * silu_node = cgraph->nodes[i]; ++ ggml_tensor * cast_node = cgraph->nodes[i + 1]; ++ const bool eligible = ++ native_bf16 && ggml_get_unary_op(silu_node) == GGML_UNARY_OP_SILU && ++ cast_node->src[0] == silu_node && ++ silu_node->src[0]->type == GGML_TYPE_F32 && ++ silu_node->type == GGML_TYPE_F32 && cast_node->type == GGML_TYPE_BF16 && ++ ggml_are_same_shape(silu_node->src[0], cast_node) && ++ ggml_is_contiguous(silu_node->src[0]) && ggml_is_contiguous(cast_node); ++ if (eligible) { ++ ggml_cuda_op_silu_f32_to_bf16(*cuda_ctx, silu_node, cast_node); ++ return 1; ++ } ++ } ++ + // Q8 narrow projection + SiLU. NeMo-Speech.cpp's Conformer FF1 + // layers are bias-free, so this is their actual hot graph sequence. + if (ggml_can_fuse(cgraph, i, { GGML_OP_MUL_MAT, GGML_OP_UNARY })) { +diff --git a/ggml/src/ggml-cuda/unary.cu b/ggml/src/ggml-cuda/unary.cu +index d3e594878fcc9335e127a3743d8f2a8aa32e8417..0b908345a1a5ba0424979d0dde4c0b94751de898 100644 +--- a/ggml/src/ggml-cuda/unary.cu ++++ b/ggml/src/ggml-cuda/unary.cu +@@ -186,6 +186,37 @@ void ggml_cuda_op_silu(ggml_backend_cuda_context & ctx, ggml_tensor * dst) { + ggml_cuda_op_unary(ctx, dst); + } + ++static __global__ void silu_f32_to_bf16( ++ const float * __restrict__ x, nv_bfloat16 * __restrict__ dst, ++ const int64_t nelements) { ++#if defined(__CUDA_ARCH__) && __CUDA_ARCH__ >= GGML_CUDA_CC_AMPERE ++ const int64_t i = (int64_t) blockIdx.x * blockDim.x + threadIdx.x; ++ if (i < nelements) { ++ dst[i] = __float2bfloat16(op_silu(x[i])); ++ } ++#else ++ GGML_UNUSED_VARS(x, dst, nelements); ++ NO_DEVICE_CODE; ++#endif ++} ++ ++void ggml_cuda_op_silu_f32_to_bf16( ++ ggml_backend_cuda_context & ctx, const ggml_tensor * silu_node, ++ ggml_tensor * dst) { ++ const ggml_tensor * src = silu_node->src[0]; ++ GGML_ASSERT(src->type == GGML_TYPE_F32); ++ GGML_ASSERT(silu_node->type == GGML_TYPE_F32); ++ GGML_ASSERT(dst->type == GGML_TYPE_BF16); ++ GGML_ASSERT(ggml_are_same_shape(src, dst)); ++ GGML_ASSERT(ggml_is_contiguous(src)); ++ GGML_ASSERT(ggml_is_contiguous(dst)); ++ ++ const int64_t nelements = ggml_nelements(src); ++ const int64_t num_blocks = (nelements + CUDA_SILU_BLOCK_SIZE - 1) / CUDA_SILU_BLOCK_SIZE; ++ silu_f32_to_bf16<<>>( ++ (const float *) src->data, (nv_bfloat16 *) dst->data, nelements); ++} ++ + void ggml_cuda_op_tanh(ggml_backend_cuda_context & ctx, ggml_tensor * dst) { + ggml_cuda_op_unary(ctx, dst); + } +diff --git a/ggml/src/ggml-cuda/unary.cuh b/ggml/src/ggml-cuda/unary.cuh +index 04f3af6443a74283a9994dfcd3be3e45eb95c685..5c39ccb2a05d57a82b70ffb445c0e7977e5f2f47 100644 +--- a/ggml/src/ggml-cuda/unary.cuh ++++ b/ggml/src/ggml-cuda/unary.cuh +@@ -31,6 +31,9 @@ void ggml_cuda_op_gelu(ggml_backend_cuda_context & ctx, ggml_tensor * dst); + + void ggml_cuda_op_silu(ggml_backend_cuda_context & ctx, ggml_tensor * dst); + ++void ggml_cuda_op_silu_f32_to_bf16( ++ ggml_backend_cuda_context & ctx, const ggml_tensor * silu_node, ggml_tensor * dst); ++ + void ggml_cuda_op_silu_back(ggml_backend_cuda_context & ctx, ggml_tensor * dst); + + void ggml_cuda_op_gelu_erf(ggml_backend_cuda_context & ctx, ggml_tensor * dst); diff --git a/ggml-patches/0003-norm-mul-add-fusion.patch b/patches/ggml-cuda-fuse-layernorm-with-row-vector-scale-and-bias.patch similarity index 60% rename from ggml-patches/0003-norm-mul-add-fusion.patch rename to patches/ggml-cuda-fuse-layernorm-with-row-vector-scale-and-bias.patch index f5afa6f..8364b33 100644 --- a/ggml-patches/0003-norm-mul-add-fusion.patch +++ b/patches/ggml-cuda-fuse-layernorm-with-row-vector-scale-and-bias.patch @@ -1,8 +1,91 @@ -diff --git a/src/ggml-cuda/norm.cu b/src/ggml-cuda/norm.cu -index ef98f675..d77e1a61 100644 ---- a/src/ggml-cuda/norm.cu -+++ b/src/ggml-cuda/norm.cu -@@ -37,6 +37,50 @@ static __global__ void norm_f32( +From: Prabhsimran Singh +Subject: ggml-cuda: fuse LayerNorm with row-vector scale and bias + +Fuses GGML_OP_NORM + MUL (+ ADD) when gamma/beta are contiguous ne0-length +row vectors, the affine LayerNorm used throughout FastConformer. + +Upstream: candidate. CUDA lacks NORM+MUL+ADD fusion while Metal and OpenCL +have it; an upstream version should reuse the rms_norm broadcast machinery +and add row-vector test cases. + +diff --git a/ggml/src/ggml-cuda/ggml-cuda.cu b/ggml/src/ggml-cuda/ggml-cuda.cu +index 95c0b1bf1e9d91249aab7b56be251dd4bd6fa288..ecf831b6446e1ddb1f0efb28c9c01d18e3591d6f 100644 +--- a/ggml/src/ggml-cuda/ggml-cuda.cu ++++ b/ggml/src/ggml-cuda/ggml-cuda.cu +@@ -3274,6 +3274,51 @@ static bool ggml_cuda_can_fuse(const struct ggml_cgraph * cgraph, + return false; + } + ++ // Standard LayerNorm (GGML_OP_NORM) + row-vector gamma (+ row-vector beta). ++ // Restricted to the classic affine pattern: the mul/add operands must be ++ // contiguous ne0-length vectors broadcast over rows — that is what the ++ // fused norm_mul_add_f32 kernel implements (see norm.cu). ++ if ((ops.size() == 2 || ops.size() == 3) && ops.begin()[0] == GGML_OP_NORM && ops.begin()[1] == GGML_OP_MUL) { ++ const ggml_tensor * norm = cgraph->nodes[node_idx]; ++ const ggml_tensor * mul = cgraph->nodes[node_idx+1]; ++ const ggml_tensor * add = nullptr; ++ ++ if (ops.size() == 3) { ++ if (ops.begin()[2] != GGML_OP_ADD) { ++ return false; ++ } ++ add = cgraph->nodes[node_idx+2]; ++ } ++ ++ if (norm->src[0]->type != GGML_TYPE_F32 || norm->type != GGML_TYPE_F32 || ++ mul->type != GGML_TYPE_F32 || (add && add->type != GGML_TYPE_F32)) { ++ return false; ++ } ++ ++ const ggml_tensor * gamma = ++ mul->src[0] == norm ? mul->src[1] : (mul->src[1] == norm ? mul->src[0] : nullptr); ++ if (gamma == nullptr || gamma->type != GGML_TYPE_F32 || ++ ggml_nelements(gamma) != norm->ne[0] || !ggml_is_contiguous(gamma)) { ++ return false; ++ } ++ ++ if (add) { ++ const ggml_tensor * beta = ++ add->src[0] == mul ? add->src[1] : (add->src[1] == mul ? add->src[0] : nullptr); ++ if (beta == nullptr || beta->type != GGML_TYPE_F32 || ++ ggml_nelements(beta) != norm->ne[0] || !ggml_is_contiguous(beta)) { ++ return false; ++ } ++ } ++ ++ // the fused kernel writes dst rows contiguously ++ if (!ggml_is_contiguous(mul) || (add && !ggml_is_contiguous(add))) { ++ return false; ++ } ++ ++ return true; ++ } ++ + if ((ops.size() == 2 || ops.size() == 3) && ops.begin()[0] == GGML_OP_RMS_NORM && ops.begin()[1] == GGML_OP_MUL) { + const ggml_tensor *rms_norm = cgraph->nodes[node_idx]; + const ggml_tensor *mul = cgraph->nodes[node_idx+1]; +@@ -4157,6 +4202,16 @@ static int ggml_cuda_try_fuse(ggml_backend_cuda_context * cuda_ctx, ggml_cgraph + return 1; + } + ++ if (ggml_cuda_can_fuse(cgraph, i, { GGML_OP_NORM, GGML_OP_MUL, GGML_OP_ADD }, {})) { ++ ggml_cuda_op_norm_fused(*cuda_ctx, node, cgraph->nodes[i + 1], cgraph->nodes[i + 2]); ++ return 2; ++ } ++ ++ if (ggml_cuda_can_fuse(cgraph, i, { GGML_OP_NORM, GGML_OP_MUL }, {})) { ++ ggml_cuda_op_norm_fused(*cuda_ctx, node, cgraph->nodes[i + 1], nullptr); ++ return 1; ++ } ++ + if (ggml_cuda_can_fuse(cgraph, i, { GGML_OP_SSM_CONV, GGML_OP_ADD, GGML_OP_UNARY }, { GGML_UNARY_OP_SILU })) { + ggml_cuda_op_ssm_conv(*cuda_ctx, node, cgraph->nodes[i + 1], cgraph->nodes[i + 2]); + return 2; +diff --git a/ggml/src/ggml-cuda/norm.cu b/ggml/src/ggml-cuda/norm.cu +index c3758cd50cfeae5d94621ada986d18e268faf8a4..8d2c036b6b414d8ba4d05b61382855b3da4a99b0 100644 +--- a/ggml/src/ggml-cuda/norm.cu ++++ b/ggml/src/ggml-cuda/norm.cu +@@ -38,6 +38,50 @@ static __global__ void norm_f32( } } @@ -53,7 +136,7 @@ index ef98f675..d77e1a61 100644 template static __global__ void group_norm_f32(const float * x, float * dst, const int group_size, const int ne_elements, const float eps) { // blockIdx.x: num_groups idx -@@ -283,6 +327,32 @@ static void norm_f32_cuda( +@@ -290,6 +334,32 @@ static void norm_f32_cuda( } } @@ -86,7 +169,7 @@ index ef98f675..d77e1a61 100644 static void group_norm_f32_cuda( const float * x, float * dst, const int num_groups, const float eps, const int group_size, const int ne_elements, cudaStream_t stream) { if (group_size < 1024) { -@@ -430,6 +500,59 @@ void ggml_cuda_op_norm(ggml_backend_cuda_context & ctx, ggml_tensor * dst) { +@@ -456,6 +526,59 @@ void ggml_cuda_op_norm(ggml_backend_cuda_context & ctx, ggml_tensor * dst) { norm_f32_cuda(src0_d, dst_d, ne00, ne01, ne02, ne03, s01, s02, s03, eps, stream); } @@ -146,10 +229,10 @@ index ef98f675..d77e1a61 100644 void ggml_cuda_op_group_norm(ggml_backend_cuda_context & ctx, ggml_tensor * dst) { const ggml_tensor * src0 = dst->src[0]; const float * src0_d = (const float *)src0->data; -diff --git a/src/ggml-cuda/norm.cuh b/src/ggml-cuda/norm.cuh -index a74f6376..20ecaa2d 100644 ---- a/src/ggml-cuda/norm.cuh -+++ b/src/ggml-cuda/norm.cuh +diff --git a/ggml/src/ggml-cuda/norm.cuh b/ggml/src/ggml-cuda/norm.cuh +index a74f6376720ab60b07b6a036042bdedca61f4e4a..20ecaa2d9e7536c3f19d111ade42fa1f2497b38d 100644 +--- a/ggml/src/ggml-cuda/norm.cuh ++++ b/ggml/src/ggml-cuda/norm.cuh @@ -2,6 +2,11 @@ void ggml_cuda_op_norm(ggml_backend_cuda_context & ctx, ggml_tensor * dst); diff --git a/patches/ggml-cuda-skinny-q8_0-gemm-with-planar-weights.patch b/patches/ggml-cuda-skinny-q8_0-gemm-with-planar-weights.patch new file mode 100644 index 0000000..c2b778a --- /dev/null +++ b/patches/ggml-cuda-skinny-q8_0-gemm-with-planar-weights.patch @@ -0,0 +1,1305 @@ +From: Prabhsimran Singh +Subject: ggml-cuda: skinny Q8_0 GEMM with planar weights + +Adds a Q8_0 x F32 GEMM for streaming activations wider than MMVQ's 8 columns +with planar weight storage (GGML_TENSOR_FLAG_Q8_PLANAR), deterministic +K-split, and bias/SiLU epilogues on MMVQ for narrow (<= 2 column) +projections. Dispatch is limited to encoder.* Q8_0 weights and depends only on +the call's column count (N <= 8: MMVQ, planar MMVQ after a repack; wider: +this kernel), never on whether an earlier call repacked the weight, so outputs +do not depend on previously served requests. Weights are repacked in place on +first use; GGML_SKINNY_Q8_INPLACE=0 keeps a separate copy instead, which is +needed when llama.cpp's multi-stream scheduler shares the process. Cache +entries are dropped when their weight buffer is freed or cleared. +GGML_SKINNY_Q8=0 disables the kernel for block (not planar) Q8_0 weights. + +Upstream: not planned (NeMo-specific encoder shapes). The MMVQ +broadcast-bias/SiLU epilogue is a candidate on its own. + +diff --git a/ggml/include/ggml.h b/ggml/include/ggml.h +index fd31b70ca3a05c4814c62a0be599ec0a56766e8e..19ade01735e6c072fcb7752ef7955ac5d2afe19e 100644 +--- a/ggml/include/ggml.h ++++ b/ggml/include/ggml.h +@@ -667,6 +667,10 @@ extern "C" { + GGML_TENSOR_FLAG_PARAM = 4, // ...contains trainable parameters + GGML_TENSOR_FLAG_LOSS = 8, // ...defines loss for numerical optimization (multiple loss tensors add up) + GGML_TENSOR_FLAG_COMPUTE = 16, // ...must be computed ++ // Model weight uses Q8_0 bytes serialized as one tensor-wide int8 ++ // plane followed by one FP16 scale plane. CUDA-only storage hint; ++ // logical type, element count, and allocation size remain Q8_0. ++ GGML_TENSOR_FLAG_Q8_PLANAR = 32, + }; + + enum ggml_tri_type { +diff --git a/ggml/src/ggml-cuda/common.cuh b/ggml/src/ggml-cuda/common.cuh +index 2e78ae4facb9d2fffdfe0e3352d68baf7dc4a847..25893da48941518b5e0a8cfa20c1538fd2a5d305 100644 +--- a/ggml/src/ggml-cuda/common.cuh ++++ b/ggml/src/ggml-cuda/common.cuh +@@ -1578,6 +1578,7 @@ struct ggml_cuda_mm_fusion_args_host { + const ggml_tensor * gate_bias = nullptr; + const ggml_tensor * x_scale = nullptr; + const ggml_tensor * gate_scale = nullptr; ++ bool silu = false; + ggml_glu_op glu_op; + float glu_limit = 0.0f; + }; +@@ -1587,6 +1588,12 @@ struct ggml_cuda_mm_fusion_args_device { + const void * gate_bias = nullptr; + const void * x_scale = nullptr; + const void * gate_scale = nullptr; ++ // A Linear bias has shape [M, 1, 1, 1] and is broadcast across the ++ // destination columns/channels. Keep this explicit because the legacy ++ // fusion path also accepts a full-shape elementwise bias. ++ bool x_bias_broadcast = false; ++ bool gate_bias_broadcast = false; ++ bool silu = false; + ggml_glu_op glu_op; + float glu_limit = 0.0f; + }; +diff --git a/ggml/src/ggml-cuda/ggml-cuda.cu b/ggml/src/ggml-cuda/ggml-cuda.cu +index 6730a6e28ef9ab02c74cf4e164ff2e8ac5cfb0bf..32b96b5e7fdb48a02ad255dfab688520d86ea884 100644 +--- a/ggml/src/ggml-cuda/ggml-cuda.cu ++++ b/ggml/src/ggml-cuda/ggml-cuda.cu +@@ -17,6 +17,7 @@ + #include "ggml-cuda/conv2d.cuh" + #include "ggml-cuda/conv2d-dw.cuh" + #include "ggml-cuda/conv2d-transpose.cuh" ++#include "ggml-cuda/skinny-q8.cuh" + #include "ggml-cuda/convert.cuh" + #include "ggml-cuda/count-equal.cuh" + #include "ggml-cuda/cpy.cuh" +@@ -741,6 +742,8 @@ struct ggml_backend_cuda_buffer_context { + + static void ggml_backend_cuda_buffer_free_buffer(ggml_backend_buffer_t buffer) { + ggml_backend_cuda_buffer_context * ctx = (ggml_backend_cuda_buffer_context *)buffer->context; ++ ggml_cuda_set_device(ctx->device); ++ ggml_cuda_skinny_q8_forget(ctx->dev_ptr, buffer->size); + delete ctx; + } + +@@ -849,6 +852,7 @@ static void ggml_backend_cuda_buffer_clear(ggml_backend_buffer_t buffer, uint8_t + ggml_cuda_set_device(ctx->device); + CUDA_CHECK(cudaMemsetAsync(ctx->dev_ptr, value, buffer->size, cudaStreamPerThread)); + CUDA_CHECK(cudaStreamSynchronize(cudaStreamPerThread)); ++ ggml_cuda_skinny_q8_forget(ctx->dev_ptr, buffer->size); + } + + static const ggml_backend_buffer_i ggml_backend_cuda_buffer_interface = { +@@ -1801,16 +1805,20 @@ static bool ggml_cuda_should_fuse_mul_mat_vec_q(const ggml_tensor * tensor) { + ggml_nbytes(src0) != ggml_backend_buffer_get_alloc_size(src0->buffer, src0) && + src0->view_src; + ++ const int64_t total_n = ggml_nelements(src1) / src1->ne[0]; + bool use_mul_mat_vec_q = ggml_is_quantized(src0->type) && !bad_padding_clear && src1->type == GGML_TYPE_F32 && +- dst->type == GGML_TYPE_F32 && src1->ne[1] <= MMVQ_MAX_BATCH_SIZE; ++ dst->type == GGML_TYPE_F32 && total_n <= MMVQ_MAX_BATCH_SIZE; + + // fusion is not universally faster on Pascal + const int cc = ggml_cuda_info().devices[ggml_cuda_get_device()].cc; + if (cc <= GGML_CUDA_CC_PASCAL) { + return false; + } +- //we only support fusion for ncols_dst = 1 +- if (tensor->op == GGML_OP_MUL_MAT && dst->ne[1] != 1) { ++ // MMVQ's narrow epilogue supports the two-frame streaming chunk used by ++ // NeMo-Speech.cpp. Outer request-batch columns are included in ++ // total_n above so [K,2,B] stays on skinny-Q8 when B makes it wider than ++ // MMVQ_MAX_BATCH_SIZE. ++ if (tensor->op == GGML_OP_MUL_MAT && dst->ne[1] > 2) { + return false; + } + +@@ -1842,6 +1850,12 @@ static void ggml_cuda_mul_mat(ggml_backend_cuda_context & ctx, const ggml_tensor + const int cc = ggml_cuda_info().devices[ctx.device].cc; + const int warp_size = ggml_cuda_info().devices[ctx.device].warp_size; + ++ if (ggml_cuda_skinny_q8_supported(src0, src1, dst)) { ++ // streaming-encoder specialization: Q8_0 weights x activations wider ++ // than MMVQ's range; beats mul_mat_q's LLM-batch tiling at skinny N ++ ggml_cuda_mul_mat_skinny_q8(ctx, src0, src1, dst); ++ return; ++ } + if (ggml_cuda_should_use_mmvf(src0->type, cc, src0->ne, src0->nb, ne11)) { + // The custom F16 vector kernel can be used over batched cuBLAS GEMM. + // But this is only faster for GPUs without tensor cores or with a thin src0 matrix (particularly KQV in attention) +@@ -4121,6 +4135,64 @@ static int ggml_cuda_try_fuse(ggml_backend_cuda_context * cuda_ctx, ggml_cgraph + return fused_node_count - 1; + } + ++ // Q8 narrow projection + SiLU. NeMo-Speech.cpp's Conformer FF1 ++ // layers are bias-free, so this is their actual hot graph sequence. ++ if (ggml_can_fuse(cgraph, i, { GGML_OP_MUL_MAT, GGML_OP_UNARY })) { ++ ggml_tensor * mm_node = cgraph->nodes[i]; ++ ggml_tensor * silu_node = cgraph->nodes[i + 1]; ++ if (ggml_get_unary_op(silu_node) == GGML_UNARY_OP_SILU && ++ silu_node->src[0] == mm_node && ggml_cuda_should_fuse_mul_mat_vec_q(mm_node)) { ++ ggml_cuda_mm_fusion_args_host fusion_data{}; ++ fusion_data.silu = true; ++ ggml_cuda_mul_mat_vec_q( ++ *cuda_ctx, mm_node->src[0], mm_node->src[1], nullptr, silu_node, &fusion_data); ++ return 1; ++ } ++ } ++ ++ // Q8 narrow projection + broadcast Linear bias + SiLU. This is the hot ++ // Conformer FF1 sequence at the two-frame streaming chunk size. Folding ++ // both epilogues avoids two launches and two full output round trips. ++ if (ggml_can_fuse(cgraph, i, { GGML_OP_MUL_MAT, GGML_OP_ADD, GGML_OP_UNARY })) { ++ ggml_tensor * mm_node = cgraph->nodes[i]; ++ ggml_tensor * bias_node = cgraph->nodes[i + 1]; ++ ggml_tensor * silu_node = cgraph->nodes[i + 2]; ++ const ggml_tensor * bias = bias_node->src[0] == mm_node ? bias_node->src[1] : ++ bias_node->src[1] == mm_node ? bias_node->src[0] : ++ nullptr; ++ if (ggml_get_unary_op(silu_node) == GGML_UNARY_OP_SILU && ++ silu_node->src[0] == bias_node && bias != nullptr && ++ bias->type == GGML_TYPE_F32 && ggml_is_contiguous(bias) && ++ ggml_nelements(bias) == mm_node->ne[0] && ++ ggml_cuda_should_fuse_mul_mat_vec_q(mm_node)) { ++ ggml_cuda_mm_fusion_args_host fusion_data{}; ++ fusion_data.x_bias = bias; ++ fusion_data.silu = true; ++ ggml_cuda_mul_mat_vec_q( ++ *cuda_ctx, mm_node->src[0], mm_node->src[1], nullptr, silu_node, &fusion_data); ++ return 2; ++ } ++ } ++ ++ // skinny-q8 GEMM + row-vector bias: the upstream MUL_MAT+ADD fusion below ++ // requires a same-shape add, so the classic broadcast Linear bias ++ // ([M] over [M,N]) never qualifies — fold it into the skinny GEMM's ++ // epilogue instead. ++ if (ggml_can_fuse(cgraph, i, { GGML_OP_MUL_MAT, GGML_OP_ADD })) { ++ ggml_tensor * mm_node = cgraph->nodes[i]; ++ ggml_tensor * bias_node = cgraph->nodes[i + 1]; ++ const ggml_tensor * bias = bias_node->src[0] == mm_node ? bias_node->src[1] : ++ bias_node->src[1] == mm_node ? bias_node->src[0] : ++ nullptr; ++ if (bias != nullptr && bias->type == GGML_TYPE_F32 && ggml_is_contiguous(bias) && ++ ggml_nelements(bias) == mm_node->ne[0] && ggml_is_contiguous(bias_node) && ++ ggml_cuda_skinny_q8_supported(mm_node->src[0], mm_node->src[1], mm_node)) { ++ ggml_cuda_mul_mat_skinny_q8_bias( ++ *cuda_ctx, mm_node->src[0], mm_node->src[1], bias, bias_node); ++ return 1; ++ } ++ } ++ + // mul_mat + add + for (ggml_op op : { GGML_OP_MUL_MAT, GGML_OP_MUL_MAT_ID }) { + const ggml_op bias_op = op == GGML_OP_MUL_MAT ? GGML_OP_ADD : GGML_OP_ADD_ID; +@@ -4156,14 +4228,21 @@ static int ggml_cuda_try_fuse(ggml_backend_cuda_context * cuda_ctx, ggml_cgraph + continue; + } + +- if (bias_op == GGML_OP_ADD && !ggml_are_same_shape(bias_node->src[0], bias_node->src[1])) { ++ const bool same_shape_bias = bias_op != GGML_OP_ADD || ++ ggml_are_same_shape(bias_node->src[0], bias_node->src[1]); ++ const bool broadcast_q_bias = ++ bias_op == GGML_OP_ADD && ++ bias_tensor->type == GGML_TYPE_F32 && ++ ggml_is_contiguous(bias_tensor) && ++ ggml_nelements(bias_tensor) == mm_node->ne[0]; ++ if (!same_shape_bias && !broadcast_q_bias) { + continue; + } + + ggml_cuda_mm_fusion_args_host fusion_data{}; + fusion_data.x_bias = bias_tensor; + +- if (ggml_cuda_should_fuse_mul_mat_vec_f(mm_node)) { ++ if (same_shape_bias && ggml_cuda_should_fuse_mul_mat_vec_f(mm_node)) { + ggml_cuda_mul_mat_vec_f(*cuda_ctx, src0, src1, ids, bias_node, &fusion_data); + fused_mul_mat_vec = true; + fused_node_count = 2; +diff --git a/ggml/src/ggml-cuda/mmvq.cu b/ggml/src/ggml-cuda/mmvq.cu +index a85155360a7ba48014bf8a450fc3a35bafd2aa1d..8deb33650513227e81eea584c6fb1b744550e76b 100644 +--- a/ggml/src/ggml-cuda/mmvq.cu ++++ b/ggml/src/ggml-cuda/mmvq.cu +@@ -1,5 +1,6 @@ + #include "mmvq.cuh" + #include "quantize.cuh" ++#include "skinny-q8.cuh" + #include "unary.cuh" + #include "vecdotq.cuh" + +@@ -596,7 +597,17 @@ static constexpr __host__ __device__ int calc_rows_per_block(int ncols_dst, int + return 1; + } + +-template ++template ++static __device__ __forceinline__ float vec_dot_q8_0_q8_1_planar( ++ const void * __restrict__ vx, const block_q8_1 * __restrict__ y, ++ int block_idx, int iqs, size_t scale_plane_offset) { ++ const int * v = (const int *) vx + (size_t) block_idx * (QK8_0 / 4) + iqs; ++ const int * u = (const int *) y->qs + iqs; ++ const half * d = (const half *) ((const char *) vx + scale_plane_offset); ++ return vec_dot_q8_0_q8_1_impl(v, u, d[block_idx], __low2half(y->ds)); ++} ++ ++template + __launch_bounds__(calc_nwarps(type, ncols_dst, get_device_table_id(), small_k, halve_iters)*ggml_cuda_get_physical_warp_size(), 1) + static __global__ void mul_mat_vec_q( + const void * vx_ptr, const void * vy_ptr, const int32_t * ids_ptr, const ggml_cuda_mm_fusion_args_device fusion, float * dst_ptr, +@@ -624,6 +635,10 @@ static __global__ void mul_mat_vec_q( + const int row0 = rows_per_cuda_block*blockIdx.x; + const int blocks_per_row_x = ncols_x / qk; + constexpr int blocks_per_iter = vdr * nwarps*warp_size / qi; ++ // Serialized planar weights are restricted to contiguous 2-D matrices. ++ // stride_col_dst is therefore the source row count, and one int8 byte is ++ // stored per logical weight before the FP16 scale plane. ++ const size_t q8_scale_plane_offset = (size_t) ncols_x * stride_col_dst; + + const uint32_t channel_dst = blockIdx.y; + +@@ -644,6 +659,7 @@ static __global__ void mul_mat_vec_q( + bool use_gate_bias = false; + bool use_scale = false; + bool use_gate_scale = false; ++ bool use_silu = false; + [[maybe_unused]] const void * vgate = nullptr; + const float * x_bias = nullptr; + const float * gate_bias = nullptr; +@@ -656,6 +672,7 @@ static __global__ void mul_mat_vec_q( + use_gate = fusion.gate != nullptr; + use_bias = fusion.x_bias != nullptr; + use_gate_bias = fusion.gate_bias != nullptr && use_gate; ++ use_silu = fusion.silu; + vgate = fusion.gate; + x_bias = (const float *) fusion.x_bias; + gate_bias = (const float *) fusion.gate_bias; +@@ -681,17 +698,19 @@ static __global__ void mul_mat_vec_q( + if (threadIdx.x < rows_per_cuda_block && threadIdx.y == 0 && + (rows_per_cuda_block == 1 || uint32_t(row0 + threadIdx.x) < stride_col_dst)) { + if (use_bias) { +- x_bias = x_bias + sample_dst * stride_sample_dst + channel_bias * stride_channel_dst + row0; ++ x_bias = x_bias + (fusion.x_bias_broadcast ? row0 : ++ sample_dst * stride_sample_dst + channel_bias * stride_channel_dst + row0); + #pragma unroll + for (int j = 0; j < ncols_dst; ++j) { +- x_biases[j] = x_bias[j * stride_col_dst + threadIdx.x]; ++ x_biases[j] = x_bias[(fusion.x_bias_broadcast ? 0 : j * stride_col_dst) + threadIdx.x]; + } + } + if (use_gate_bias) { +- gate_bias = gate_bias + sample_dst * stride_sample_dst + channel_bias * stride_channel_dst + row0; ++ gate_bias = gate_bias + (fusion.gate_bias_broadcast ? row0 : ++ sample_dst * stride_sample_dst + channel_bias * stride_channel_dst + row0); + #pragma unroll + for (int j = 0; j < ncols_dst; ++j) { +- gate_biases[j] = gate_bias[j * stride_col_dst + threadIdx.x]; ++ gate_biases[j] = gate_bias[(fusion.gate_bias_broadcast ? 0 : j * stride_col_dst) + threadIdx.x]; + } + } + if constexpr (type == GGML_TYPE_NVFP4) { +@@ -742,12 +761,26 @@ static __global__ void mul_mat_vec_q( + for (int j = 0; j < ncols_dst; ++j) { + #pragma unroll + for (int i = 0; i < rows_per_cuda_block; ++i) { +- tmp[j][i] += vec_dot_q_cuda( +- vx, &y[j*stride_col_y + kby], kbx_offset + i*stride_row_x + kbx, kqs); ++ if constexpr (q8_planar) { ++ static_assert(type == GGML_TYPE_Q8_0, "planar MMVQ is Q8_0-only"); ++ tmp[j][i] += vec_dot_q8_0_q8_1_planar( ++ vx, &y[j*stride_col_y + kby], ++ kbx_offset + i*stride_row_x + kbx, kqs, q8_scale_plane_offset); ++ } else { ++ tmp[j][i] += vec_dot_q_cuda( ++ vx, &y[j*stride_col_y + kby], kbx_offset + i*stride_row_x + kbx, kqs); ++ } + if constexpr (has_fusion) { + if (use_gate) { +- tmp_gate[j][i] += vec_dot_q_cuda( +- vgate, &y[j*stride_col_y + kby], kbx_offset + i*stride_row_x + kbx, kqs); ++ if constexpr (q8_planar) { ++ tmp_gate[j][i] += vec_dot_q8_0_q8_1_planar( ++ vgate, &y[j*stride_col_y + kby], ++ kbx_offset + i*stride_row_x + kbx, kqs, q8_scale_plane_offset); ++ } else { ++ tmp_gate[j][i] += vec_dot_q_cuda( ++ vgate, &y[j*stride_col_y + kby], ++ kbx_offset + i*stride_row_x + kbx, kqs); ++ } + } + } + } +@@ -831,13 +864,16 @@ static __global__ void mul_mat_vec_q( + } + } + } ++ if (use_silu) { ++ result = ggml_cuda_op_silu_single(result); ++ } + dst[j*stride_col_dst + i] = result; + } + } + } + + if constexpr (!has_fusion) { +- GGML_UNUSED_VARS(use_gate, use_bias, use_gate_bias, use_scale, use_gate_scale, active_glu, glu_limit, gate_bias, x_bias, x_scale, gate_scale, tmp_gate); ++ GGML_UNUSED_VARS(use_gate, use_bias, use_gate_bias, use_scale, use_gate_scale, use_silu, active_glu, glu_limit, gate_bias, x_bias, x_scale, gate_scale, tmp_gate); + } + if constexpr (type != GGML_TYPE_NVFP4) { + GGML_UNUSED_VARS(use_scale, use_gate_scale, x_scale, gate_scale, x_scales, gate_scales); +@@ -1007,7 +1043,7 @@ static std::pair calc_launch_params( + return {block_nums, block_dims}; + } + +-template ++template + static void mul_mat_vec_q_switch_fusion( + const void * vx, const void * vy, const int32_t * ids, const ggml_cuda_mm_fusion_args_device fusion, float * dst, + const uint32_t ncols_x, const uint3 nchannels_y, const uint32_t stride_row_x, const uint32_t stride_col_y, +@@ -1018,11 +1054,11 @@ static void mul_mat_vec_q_switch_fusion( + const uint32_t ids_stride, cudaStream_t stream) { + + const bool has_fusion = fusion.gate != nullptr || fusion.x_bias != nullptr || fusion.gate_bias != nullptr || +- fusion.x_scale != nullptr || fusion.gate_scale != nullptr; +- if constexpr (c_ncols_dst == 1) { ++ fusion.x_scale != nullptr || fusion.gate_scale != nullptr || fusion.silu; ++ if constexpr (c_ncols_dst <= 2) { + if (has_fusion) { + const ggml_cuda_kernel_launch_params launch_params = ggml_cuda_kernel_launch_params(block_nums, block_dims, nbytes_shared, stream); +- ggml_cuda_kernel_launch(mul_mat_vec_q, launch_params, ++ ggml_cuda_kernel_launch(mul_mat_vec_q, launch_params, + vx, vy, ids, fusion, dst, ncols_x, nchannels_y, stride_row_x, stride_col_y, stride_col_dst, + channel_ratio, stride_channel_x, stride_channel_y, stride_channel_dst, + sample_ratio, stride_sample_x, stride_sample_y, stride_sample_dst, ids_stride); +@@ -1030,10 +1066,10 @@ static void mul_mat_vec_q_switch_fusion( + } + } + +- GGML_ASSERT(!has_fusion && "fusion only supported for ncols_dst=1"); ++ GGML_ASSERT(!has_fusion && "fusion only supported for ncols_dst<=2"); + + const ggml_cuda_kernel_launch_params launch_params = ggml_cuda_kernel_launch_params(block_nums, block_dims, nbytes_shared, stream); +- ggml_cuda_kernel_launch(mul_mat_vec_q, launch_params, ++ ggml_cuda_kernel_launch(mul_mat_vec_q, launch_params, + vx, vy, ids, fusion, dst, ncols_x, nchannels_y, stride_row_x, stride_col_y, stride_col_dst, + channel_ratio, stride_channel_x, stride_channel_y, stride_channel_dst, + sample_ratio, stride_sample_x, stride_sample_y, stride_sample_dst, ids_stride); +@@ -1072,7 +1108,7 @@ static void mul_mat_vec_q_moe_launch( + } + } + +-template ++template + static void mul_mat_vec_q_switch_ncols_dst( + const void * vx, const void * vy, const int32_t * ids, const ggml_cuda_mm_fusion_args_device fusion, float * dst, + const int ncols_x, const int nrows_x, const int ncols_dst, +@@ -1084,6 +1120,7 @@ static void mul_mat_vec_q_switch_ncols_dst( + + GGML_ASSERT(ncols_x % ggml_blck_size(type) == 0); + GGML_ASSERT(ncols_dst <= MMVQ_MAX_BATCH_SIZE); ++ static_assert(!q8_planar || type == GGML_TYPE_Q8_0, "planar MMVQ is Q8_0-only"); + + const uint3 nchannels_y_fd = ids ? init_fastdiv_values(nchannels_y) : make_uint3(0, 0, 0); + const uint3 channel_ratio_fd = ids ? make_uint3(0, 0, 0) : init_fastdiv_values(nchannels_dst / nchannels_x); +@@ -1163,6 +1200,7 @@ static void mul_mat_vec_q_switch_ncols_dst( + }; + + if (has_ids && ncols_dst > 1) { ++ GGML_ASSERT(!q8_planar && "planar Q8 does not support MUL_MAT_ID"); + // Multi-token MUL_MAT_ID path - dedicated MoE kernel + mul_mat_vec_q_moe_launch( + vx, vy, ids, fusion, dst, ncols_x, nchannels_y_fd, nrows_x, +@@ -1189,7 +1227,7 @@ static void mul_mat_vec_q_switch_ncols_dst( + + const std::pair dims = calc_launch_params(c_ncols_dst, nrows_x, nchannels_dst, + nsamples_dst, warp_size, table_id, c_small_k, c_halve_iters); +- mul_mat_vec_q_switch_fusion( ++ mul_mat_vec_q_switch_fusion( + vx, vy, ids, fusion, dst, ncols_x, nchannels_y_fd, stride_row_x, stride_col_y, stride_col_dst, + channel_ratio_fd, stride_channel_x, stride_channel_y, stride_channel_dst, sample_ratio_fd, + stride_sample_x, stride_sample_y, stride_sample_dst, dims.first, dims.second, 0, ids_stride, +@@ -1207,7 +1245,7 @@ static void mul_mat_vec_q_switch_ncols_dst( + case 2: { + constexpr int c_ncols_dst = 2; + std::pair dims = calc_launch_params(c_ncols_dst, nrows_x, nchannels_dst, nsamples_dst, warp_size, table_id); +- mul_mat_vec_q_switch_fusion(vx, vy, ids, fusion, dst, ncols_x, nchannels_y_fd, stride_row_x, stride_col_y, stride_col_dst, ++ mul_mat_vec_q_switch_fusion(vx, vy, ids, fusion, dst, ncols_x, nchannels_y_fd, stride_row_x, stride_col_y, stride_col_dst, + channel_ratio_fd, stride_channel_x, stride_channel_y, stride_channel_dst, + sample_ratio_fd, stride_sample_x, stride_sample_y, stride_sample_dst, + dims.first, dims.second, 0, ids_stride, stream); +@@ -1215,7 +1253,7 @@ static void mul_mat_vec_q_switch_ncols_dst( + case 3: { + constexpr int c_ncols_dst = 3; + std::pair dims = calc_launch_params(c_ncols_dst, nrows_x, nchannels_dst, nsamples_dst, warp_size, table_id); +- mul_mat_vec_q_switch_fusion(vx, vy, ids, fusion, dst, ncols_x, nchannels_y_fd, stride_row_x, stride_col_y, stride_col_dst, ++ mul_mat_vec_q_switch_fusion(vx, vy, ids, fusion, dst, ncols_x, nchannels_y_fd, stride_row_x, stride_col_y, stride_col_dst, + channel_ratio_fd, stride_channel_x, stride_channel_y, stride_channel_dst, + sample_ratio_fd, stride_sample_x, stride_sample_y, stride_sample_dst, + dims.first, dims.second, 0, ids_stride, stream); +@@ -1223,7 +1261,7 @@ static void mul_mat_vec_q_switch_ncols_dst( + case 4: { + constexpr int c_ncols_dst = 4; + std::pair dims = calc_launch_params(c_ncols_dst, nrows_x, nchannels_dst, nsamples_dst, warp_size, table_id); +- mul_mat_vec_q_switch_fusion(vx, vy, ids, fusion, dst, ncols_x, nchannels_y_fd, stride_row_x, stride_col_y, stride_col_dst, ++ mul_mat_vec_q_switch_fusion(vx, vy, ids, fusion, dst, ncols_x, nchannels_y_fd, stride_row_x, stride_col_y, stride_col_dst, + channel_ratio_fd, stride_channel_x, stride_channel_y, stride_channel_dst, + sample_ratio_fd, stride_sample_x, stride_sample_y, stride_sample_dst, + dims.first, dims.second, 0, ids_stride, stream); +@@ -1231,7 +1269,7 @@ static void mul_mat_vec_q_switch_ncols_dst( + case 5: { + constexpr int c_ncols_dst = 5; + std::pair dims = calc_launch_params(c_ncols_dst, nrows_x, nchannels_dst, nsamples_dst, warp_size, table_id); +- mul_mat_vec_q_switch_fusion(vx, vy, ids, fusion, dst, ncols_x, nchannels_y_fd, stride_row_x, stride_col_y, stride_col_dst, ++ mul_mat_vec_q_switch_fusion(vx, vy, ids, fusion, dst, ncols_x, nchannels_y_fd, stride_row_x, stride_col_y, stride_col_dst, + channel_ratio_fd, stride_channel_x, stride_channel_y, stride_channel_dst, + sample_ratio_fd, stride_sample_x, stride_sample_y, stride_sample_dst, + dims.first, dims.second, 0, ids_stride, stream); +@@ -1239,7 +1277,7 @@ static void mul_mat_vec_q_switch_ncols_dst( + case 6: { + constexpr int c_ncols_dst = 6; + std::pair dims = calc_launch_params(c_ncols_dst, nrows_x, nchannels_dst, nsamples_dst, warp_size, table_id); +- mul_mat_vec_q_switch_fusion(vx, vy, ids, fusion, dst, ncols_x, nchannels_y_fd, stride_row_x, stride_col_y, stride_col_dst, ++ mul_mat_vec_q_switch_fusion(vx, vy, ids, fusion, dst, ncols_x, nchannels_y_fd, stride_row_x, stride_col_y, stride_col_dst, + channel_ratio_fd, stride_channel_x, stride_channel_y, stride_channel_dst, + sample_ratio_fd, stride_sample_x, stride_sample_y, stride_sample_dst, + dims.first, dims.second, 0, ids_stride, stream); +@@ -1247,7 +1285,7 @@ static void mul_mat_vec_q_switch_ncols_dst( + case 7: { + constexpr int c_ncols_dst = 7; + std::pair dims = calc_launch_params(c_ncols_dst, nrows_x, nchannels_dst, nsamples_dst, warp_size, table_id); +- mul_mat_vec_q_switch_fusion(vx, vy, ids, fusion, dst, ncols_x, nchannels_y_fd, stride_row_x, stride_col_y, stride_col_dst, ++ mul_mat_vec_q_switch_fusion(vx, vy, ids, fusion, dst, ncols_x, nchannels_y_fd, stride_row_x, stride_col_y, stride_col_dst, + channel_ratio_fd, stride_channel_x, stride_channel_y, stride_channel_dst, + sample_ratio_fd, stride_sample_x, stride_sample_y, stride_sample_dst, + dims.first, dims.second, 0, ids_stride, stream); +@@ -1255,7 +1293,7 @@ static void mul_mat_vec_q_switch_ncols_dst( + case 8: { + constexpr int c_ncols_dst = 8; + std::pair dims = calc_launch_params(c_ncols_dst, nrows_x, nchannels_dst, nsamples_dst, warp_size, table_id); +- mul_mat_vec_q_switch_fusion(vx, vy, ids, fusion, dst, ncols_x, nchannels_y_fd, stride_row_x, stride_col_y, stride_col_dst, ++ mul_mat_vec_q_switch_fusion(vx, vy, ids, fusion, dst, ncols_x, nchannels_y_fd, stride_row_x, stride_col_y, stride_col_dst, + channel_ratio_fd, stride_channel_x, stride_channel_y, stride_channel_dst, + sample_ratio_fd, stride_sample_x, stride_sample_y, stride_sample_dst, + dims.first, dims.second, 0, ids_stride, stream); +@@ -1272,7 +1310,8 @@ static void mul_mat_vec_q_switch_type( + const int nchannels_x, const int nchannels_y, const int nchannels_dst, + const int stride_channel_x, const int stride_channel_y, const int stride_channel_dst, + const int nsamples_x, const int nsamples_dst, const int stride_sample_x, const int stride_sample_y, const int stride_sample_dst, +- const int ids_stride, cudaStream_t stream) { ++ const int ids_stride, const bool q8_planar, cudaStream_t stream) { ++ GGML_ASSERT(!q8_planar || type_x == GGML_TYPE_Q8_0); + switch (type_x) { + case GGML_TYPE_Q1_0: + mul_mat_vec_q_switch_ncols_dst +@@ -1311,10 +1350,17 @@ static void mul_mat_vec_q_switch_type( + nsamples_x, nsamples_dst, stride_sample_x, stride_sample_y, stride_sample_dst, ids_stride, stream); + break; + case GGML_TYPE_Q8_0: +- mul_mat_vec_q_switch_ncols_dst +- (vx, vy, ids, fusion, dst, ncols_x, nrows_x, ncols_dst, stride_row_x, stride_col_y, stride_col_dst, +- nchannels_x, nchannels_y, nchannels_dst, stride_channel_x, stride_channel_y, stride_channel_dst, +- nsamples_x, nsamples_dst, stride_sample_x, stride_sample_y, stride_sample_dst, ids_stride, stream); ++ if (q8_planar) { ++ mul_mat_vec_q_switch_ncols_dst ++ (vx, vy, ids, fusion, dst, ncols_x, nrows_x, ncols_dst, stride_row_x, stride_col_y, stride_col_dst, ++ nchannels_x, nchannels_y, nchannels_dst, stride_channel_x, stride_channel_y, stride_channel_dst, ++ nsamples_x, nsamples_dst, stride_sample_x, stride_sample_y, stride_sample_dst, ids_stride, stream); ++ } else { ++ mul_mat_vec_q_switch_ncols_dst ++ (vx, vy, ids, fusion, dst, ncols_x, nrows_x, ncols_dst, stride_row_x, stride_col_y, stride_col_dst, ++ nchannels_x, nchannels_y, nchannels_dst, stride_channel_x, stride_channel_y, stride_channel_dst, ++ nsamples_x, nsamples_dst, stride_sample_x, stride_sample_y, stride_sample_dst, ids_stride, stream); ++ } + break; + case GGML_TYPE_MXFP4: + mul_mat_vec_q_switch_ncols_dst +@@ -1449,7 +1495,7 @@ void ggml_cuda_mul_mat_vec_q( + if (fusion) { + const int cc = ggml_cuda_info().devices[ggml_cuda_get_device()].cc; + GGML_ASSERT( !ids || dst->ne[2] <= get_mmvq_mmid_max_batch(src0->type, cc)); +- GGML_ASSERT( ids || dst->ne[1] == 1); ++ GGML_ASSERT( ids || dst->ne[1] <= 2); + // Scale fusion is only allowed for NVFP4 currently as the cost of checking this at run-time in the prologue is + // non-negligible for some models such as gpt-oss-20b + GGML_ASSERT((fusion->x_scale == nullptr && fusion->gate_scale == nullptr) || src0->type == GGML_TYPE_NVFP4); +@@ -1459,6 +1505,7 @@ void ggml_cuda_mul_mat_vec_q( + GGML_ASSERT(fusion->x_bias->ne[0] == dst->ne[0]); + GGML_ASSERT(!ids || fusion->x_bias->ne[1] == src0->ne[2]); + fusion_local.x_bias = fusion->x_bias->data; ++ fusion_local.x_bias_broadcast = ggml_nelements(fusion->x_bias) == dst->ne[0]; + } + if (fusion->gate) { + GGML_ASSERT(fusion->gate->type == src0->type && ggml_are_same_stride(fusion->gate, src0)); +@@ -1469,6 +1516,7 @@ void ggml_cuda_mul_mat_vec_q( + GGML_ASSERT(fusion->gate_bias->ne[0] == dst->ne[0]); + GGML_ASSERT(!ids || fusion->gate_bias->ne[1] == src0->ne[2]); + fusion_local.gate_bias = fusion->gate_bias->data; ++ fusion_local.gate_bias_broadcast = ggml_nelements(fusion->gate_bias) == dst->ne[0]; + } + if (fusion->x_scale) { + GGML_ASSERT(fusion->x_scale->type == GGML_TYPE_F32); +@@ -1484,6 +1532,7 @@ void ggml_cuda_mul_mat_vec_q( + } + fusion_local.glu_op = fusion->glu_op; + fusion_local.glu_limit = fusion->glu_limit; ++ fusion_local.silu = fusion->silu; + } + + // If src0 is a temporary compute buffer, clear any potential padding. +@@ -1527,12 +1576,19 @@ void ggml_cuda_mul_mat_vec_q( + const int64_t stride_channel_y = ids ? s11 : s12; + + const int64_t ids_stride = ids ? ids->nb[1] / ggml_type_size(ids->type) : 0; ++ const bool q8_planar = (src0->flags & GGML_TENSOR_FLAG_Q8_PLANAR) != 0 || ++ ggml_cuda_skinny_q8_repacked_in_place(src0); ++ if (q8_planar) { ++ GGML_ASSERT(src0->type == GGML_TYPE_Q8_0 && ids == nullptr); ++ GGML_ASSERT(src0->ne[2] == 1 && src0->ne[3] == 1 && ggml_is_contiguous(src0)); ++ } + + mul_mat_vec_q_switch_type( + src0->data, src0->type, src1_q8_1.get(), ids_d, fusion_local, dst_d, ne00, + ne01, ncols_dst, s01, stride_col_y, stride_col_dst, + ne02, nchannels_y, nchannels_dst, s02, stride_channel_y, stride_channel_dst, +- ne03, ne3, s03, s13, s3, ids_stride, stream); ++ ne03, ne3, s03, s13, s3, ids_stride, ++ q8_planar, stream); + } + + void ggml_cuda_op_mul_mat_vec_q( +@@ -1559,9 +1615,12 @@ void ggml_cuda_op_mul_mat_vec_q( + const int stride_col_y = src1_padded_row_size / QK8_1; + + ggml_cuda_mm_fusion_args_device fusion_local{}; ++ GGML_ASSERT((src0->flags & GGML_TENSOR_FLAG_Q8_PLANAR) == 0 && ++ !ggml_cuda_skinny_q8_repacked_in_place(src0) && ++ "planar Q8 does not support split-buffer MMVQ"); + mul_mat_vec_q_switch_type( + src0_dd_i, src0->type, src1_ddq_i, nullptr, fusion_local, dst_dd_i, ne00, row_diff, src1_ncols, stride_row_x, stride_col_y, nrows_dst, +- 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 0, stream); ++ 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 0, false, stream); + + GGML_UNUSED_VARS(src1, dst, src1_ddf_i, src1_ncols, src1_padded_row_size); + } +diff --git a/ggml/src/ggml-cuda/skinny-q8.cu b/ggml/src/ggml-cuda/skinny-q8.cu +new file mode 100644 +index 0000000000000000000000000000000000000000..a39355ea5cb3640b6244c73a246bd46876d98fdb +--- /dev/null ++++ b/ggml/src/ggml-cuda/skinny-q8.cu +@@ -0,0 +1,676 @@ ++// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. ++// SPDX-License-Identifier: Apache-2.0 ++// Skinny-N Q8_0 GEMM for streaming ASR encoders (nemo-speech). ++// ++// This kernel is specialized for 9 <= N <= 64, where the activation tile fits ++// on chip and weight reads dominate memory traffic. ++// ++// It uses int8 tensor cores and a cp.async pipeline: ++// * mma.sync.m16n8k32.s8 — one MMA spans exactly one q8 block of k, so the ++// per-block scales (weight d x activation d) apply directly to the s32 ++// MMA result, accumulated in fp32. ++// * Block tile = 32 rows x 64 cols: 8 warps as 2 row-halves x 4 col-quarters ++// (each warp: 16x16 = 2 MMAs per k-block). All N tiles are launched in ++// grid.z so a large outer batch fills the GPU instead of issuing a serial ++// stream of sub-SM grids. ++// * A 2-buffer pipeline stages W (32x128B), A (64x128B), and their scales per ++// 128-k step. The next buffer is issued before the current buffer computes, ++// keeping one cp.async group in flight without paying for a third buffer. ++// * Weights use aligned planes qs[M][K] + d[M][K/32]. Models can serialize ++// those planes directly; standard block_q8_0 weights are repacked once on ++// first use and cached. Activations are quantized to a zero-padded ++// 64-column int8 buffer so staging needs no bounds checks. ++// ++// Bit-accuracy: same q8 activation round-trip as mul_mat_q, but a different ++// summation order — results are not bit-identical to mmq. Dispatch must ++// therefore depend only on the call's shape, never on process history: a ++// weight takes MMVQ for N <= 8 and this kernel for every wider call, from its ++// first call on, whether or not an earlier call already repacked it. Each ++// output column's sum is independent of N (tiles only add columns), so ++// singleton and batched calls wider than 8 columns compute the same bits. ++ ++#include "skinny-q8.cuh" ++#include "mmvq.cuh" // MMVQ_MAX_BATCH_SIZE: the N range mmvq already covers ++ ++#include ++#include ++#include ++#include ++ ++#define SKQ8_WARPS 8 ++#define SKQ8_ROWS 32 // rows per block (2 warp row-halves x 16); M % 32 == 0 ++#define SKQ8_NPAD 64 // max padded cols (4 warp col-quarters x 2 n-tiles x 8) ++#define SKQ8_KSTEP 128 // k bytes staged per pipeline stage; K % KSTEP == 0 ++#define SKQ8_KTB (SKQ8_KSTEP / 32) // q8 blocks per stage ++#define SKQ8_STAGES 2 ++// Shared-memory strides (bytes). KSTEP + 16 keeps the 16B cp.async stores ++// aligned while breaking the power-of-two bank pattern on fragment loads. ++#define SKQ8_SW (SKQ8_KSTEP + 16) ++#define SKQ8_SA (SKQ8_KSTEP + 16) ++// Per-stage layout: W tile | A tile | W scales (half) | A scales (float). ++// The A tile + A scales are sized for NPAD but only ncols (= n-tiles actually ++// used, host-padded to 8) are staged/consumed. ++#define SKQ8_STAGE_W (SKQ8_ROWS * SKQ8_SW) ++#define SKQ8_STAGE_A (SKQ8_NPAD * SKQ8_SA) ++#define SKQ8_STAGE_DW (SKQ8_ROWS * SKQ8_KTB * 2) ++#define SKQ8_STAGE_DA (SKQ8_NPAD * SKQ8_KTB * 4) ++#define SKQ8_STAGE_BYTES (SKQ8_STAGE_W + SKQ8_STAGE_A + SKQ8_STAGE_DW + SKQ8_STAGE_DA) ++ ++// --------------------------------------------------------------------------- ++// Repack block_q8_0[M][K/32] -> qs plane int8[M][K] + d plane half[M][K/32]. ++static __global__ void skq8_repack( ++ const void * __restrict__ src, int8_t * __restrict__ qs, half * __restrict__ d, ++ const int64_t M, const int64_t KB) { ++ const int64_t i = (int64_t) blockIdx.x * blockDim.x + threadIdx.x; ++ if (i >= M * KB) { ++ return; ++ } ++ const block_q8_0 * b = (const block_q8_0 *) src + i; ++ d[i] = b->d; ++ int8_t * out = qs + i * 32; ++#pragma unroll ++ for (int j = 0; j < 32; j++) { ++ out[j] = b->qs[j]; ++ } ++} ++ ++ ++// --------------------------------------------------------------------------- ++// Quantize F32 activations [K, N] (col-contiguous) into a zero-padded ++// SKQ8_NPAD-column buffer: int8[NPAD][K] + d float[NPAD][K/32]. ++static __global__ void skq8_quantize( ++ const float * __restrict__ X, int8_t * __restrict__ Aq, float * __restrict__ Ad, ++ const int K, const int N) { ++ // One WARP per (col, q8-block): lane j owns element j, so the 32-float ++ // read is one coalesced 128B transaction and the int8 store one 32B ++ // transaction (a thread-per-block version did 32 scalar strided loads and ++ // cost 5x the GEMM quantize budget). ++ const int KB = K / 32; ++ const int wid = (blockIdx.x * blockDim.x + threadIdx.x) / 32; ++ const int lane = threadIdx.x % 32; ++ const int ncols = (N + 7) / 8 * 8; // active column tiles only ++ if (wid >= ncols * KB) { ++ return; ++ } ++ const int n = wid / KB, kb = wid % KB; ++ int8_t * q = Aq + (size_t) n * K + kb * 32; ++ if (n >= N) { ++ // Pad columns: only the scale must be zeroed — the GEMM multiplies the ++ // integer MMA result by da, and int math on stale qs bytes is finite, ++ // so da == 0 makes the contribution exactly 0. ++ if (lane == 0) { ++ Ad[(size_t) n * KB + kb] = 0.0f; ++ } ++ return; ++ } ++ const float x = X[(size_t) n * K + kb * 32 + lane]; ++ float amax = fabsf(x); ++#pragma unroll ++ for (int off = 16; off > 0; off >>= 1) { ++ amax = fmaxf(amax, __shfl_xor_sync(0xffffffff, amax, off)); ++ } ++ const float dv = amax / 127.0f; ++ const float id = dv > 0.0f ? 1.0f / dv : 0.0f; ++ q[lane] = (int8_t) roundf(x * id); ++ if (lane == 0) { ++ Ad[(size_t) n * KB + kb] = dv; ++ } ++} ++ ++// --------------------------------------------------------------------------- ++#if defined(__CUDA_ARCH__) && __CUDA_ARCH__ >= 800 ++static __device__ __forceinline__ void skq8_cp16(void * dst, const void * src) { ++ const unsigned s = (unsigned) __cvta_generic_to_shared(dst); ++ asm volatile("cp.async.cg.shared.global [%0], [%1], 16;" ::"r"(s), "l"(src)); ++} ++static __device__ __forceinline__ void skq8_cp8(void * dst, const void * src) { ++ const unsigned s = (unsigned) __cvta_generic_to_shared(dst); ++ asm volatile("cp.async.ca.shared.global [%0], [%1], 8;" ::"r"(s), "l"(src)); ++} ++static __device__ __forceinline__ void skq8_cp4(void * dst, const void * src) { ++ const unsigned s = (unsigned) __cvta_generic_to_shared(dst); ++ asm volatile("cp.async.ca.shared.global [%0], [%1], 4;" ::"r"(s), "l"(src)); ++} ++static __device__ __forceinline__ void skq8_commit() { ++ asm volatile("cp.async.commit_group;"); ++} ++template static __device__ __forceinline__ void skq8_wait() { ++ asm volatile("cp.async.wait_group %0;" ::"n"(n)); ++} ++static __device__ __forceinline__ void skq8_mma( ++ int & d0, int & d1, int & d2, int & d3, int a0, int a1, int a2, int a3, int b0, int b1) { ++ asm volatile( ++ "mma.sync.aligned.m16n8k32.row.col.s32.s8.s8.s32 " ++ "{%0,%1,%2,%3}, {%4,%5,%6,%7}, {%8,%9}, {%0,%1,%2,%3};" ++ : "+r"(d0), "+r"(d1), "+r"(d2), "+r"(d3) ++ : "r"(a0), "r"(a1), "r"(a2), "r"(a3), "r"(b0), "r"(b1)); ++} ++#endif ++ ++// C[M,N] (F32, col stride M) = Wq8 x A. Small-M GEMMs split K over ++// blockIdx.y (gridDim.y k-ranges, scales already folded per range); each ++// split writes a private plane for deterministic reduction afterward. ++static __global__ __launch_bounds__(SKQ8_WARPS * 32, 5) void skq8_gemm( ++ const int8_t * __restrict__ Wq, const half * __restrict__ Wd, const int8_t * __restrict__ Aq, ++ const float * __restrict__ Ad, const float * __restrict__ bias, float * __restrict__ C, ++ const int M, const int N_total, const int K, const int ncols_total /* N padded to 8 */) { ++#if defined(__CUDA_ARCH__) && __CUDA_ARCH__ >= 800 ++ const int KB = K / 32; ++ const int warp = threadIdx.x / 32; ++ const int lane = threadIdx.x % 32; ++ const int gid = lane >> 2; // mma group id (0..7) ++ const int tid4 = lane & 3; // mma thread-in-group (0..3) ++ ++ const int k_len = K / gridDim.y; // multiple of SKQ8_KSTEP (host-enforced) ++ const int k0 = blockIdx.y * k_len; ++ ++ // grid.z carries independent 64-column output tiles. Keeping them in one ++ // launch is essential on high-SM-count GPUs: a typical encoder projection ++ // has only 128 row/K-split CTAs and can underfill the device. The old host ++ // loop repeated that underfilled launch per tile and serialized the ++ // entire outer batch. Pointer rebasing keeps all inner indexing unchanged, ++ // including the per-q8-block FP32 accumulation order. ++ const int cbase = (int) blockIdx.z * SKQ8_NPAD; ++ const int ncols = min(SKQ8_NPAD, ncols_total - cbase); ++ const int N = min(SKQ8_NPAD, N_total - cbase); ++ Aq += (size_t) cbase * K; ++ Ad += (size_t) cbase * KB; ++ C += (size_t) cbase * M; ++ ++ const int row_blk = blockIdx.x * SKQ8_ROWS; ++ const int mhalf = warp / 4; // which 16-row half ++ const int nt0 = (warp % 4) * 2; // first of this warp's two 8-col n-tiles ++ ++ extern __shared__ char smem[]; ++ char * stage[SKQ8_STAGES]; ++#pragma unroll ++ for (int s = 0; s < SKQ8_STAGES; s++) { ++ stage[s] = smem + (size_t) s * SKQ8_STAGE_BYTES; ++ } ++ ++ const int nsteps = k_len / SKQ8_KSTEP; ++ ++ // Stage `ks` of this block's k-range into buffer ks % STAGES. ++ auto issue = [&](int ks) { ++ char * buf = stage[ks % SKQ8_STAGES]; ++ char * s_w = buf; ++ char * s_a = buf + SKQ8_STAGE_W; ++ char * s_dw = s_a + SKQ8_STAGE_A; ++ char * s_da = s_dw + SKQ8_STAGE_DW; ++ const int kbyte = k0 + ks * SKQ8_KSTEP; // k offset in elements (== bytes for s8) ++ const int kbb = kbyte / 32; // k offset in q8 blocks ++ constexpr int vw = SKQ8_KSTEP / 16; // 16B vectors per row per stage ++ // W tile: ROWS x vw x 16B ++ for (int it = threadIdx.x; it < SKQ8_ROWS * vw; it += blockDim.x) { ++ const int r = it / vw, v = it % vw; ++ skq8_cp16( ++ s_w + r * SKQ8_SW + v * 16, Wq + (size_t) (row_blk + r) * K + kbyte + v * 16); ++ } ++ // A tile: ncols x vw x 16B (only the active column tiles) ++ for (int it = threadIdx.x; it < ncols * vw; it += blockDim.x) { ++ const int c = it / vw, v = it % vw; ++ skq8_cp16(s_a + c * SKQ8_SA + v * 16, Aq + (size_t) c * K + kbyte + v * 16); ++ } ++ // scales: dw ROWS x KTB half (KTB*2 bytes/row), da ncols x KTB float ++ // (KTB*4 bytes/col); copy width follows KTB. ++ if (threadIdx.x < SKQ8_ROWS) { ++ if (SKQ8_KTB * 2 == 16) { ++ skq8_cp16( ++ s_dw + threadIdx.x * (SKQ8_KTB * 2), ++ Wd + (size_t) (row_blk + threadIdx.x) * KB + kbb); ++ } else if (SKQ8_KTB * 2 == 8) { ++ skq8_cp8( ++ s_dw + threadIdx.x * (SKQ8_KTB * 2), ++ Wd + (size_t) (row_blk + threadIdx.x) * KB + kbb); ++ } else { ++ skq8_cp4( ++ s_dw + threadIdx.x * (SKQ8_KTB * 2), ++ Wd + (size_t) (row_blk + threadIdx.x) * KB + kbb); ++ } ++ } ++ constexpr int da_vec = SKQ8_KTB * 4 / 16; // 16B vectors per col ++ if constexpr (da_vec > 0) { ++ for (int it = threadIdx.x; it < ncols * da_vec; it += blockDim.x) { ++ const int c = it / da_vec, h = it % da_vec; ++ skq8_cp16( ++ s_da + c * (SKQ8_KTB * 4) + h * 16, Ad + (size_t) c * KB + kbb + h * 4); ++ } ++ } else if (threadIdx.x < ncols) { ++ skq8_cp8( ++ s_da + threadIdx.x * (SKQ8_KTB * 4), ++ Ad + (size_t) threadIdx.x * KB + kbb); ++ } ++ skq8_commit(); ++ }; ++ ++ float facc[2][4] = { { 0.0f } }; ++ ++ issue(0); ++ ++ for (int ks = 0; ks < nsteps; ks++) { ++ // The next stage is issued after the current buffer is ready and before ++ // its MMA loop. This preserves compute/copy overlap with two buffers; ++ // the trailing barrier makes reusing the just-consumed buffer safe. ++ skq8_wait<0>(); ++ __syncthreads(); ++ ++ if (ks + 1 < nsteps) { ++ issue(ks + 1); ++ } ++ ++ const char * buf = stage[ks % SKQ8_STAGES]; ++ const char * s_w = buf; ++ const char * s_a = buf + SKQ8_STAGE_W; ++ const half * s_dw = (const half *) (buf + SKQ8_STAGE_W + SKQ8_STAGE_A); ++ const float * s_da = ++ (const float *) (buf + SKQ8_STAGE_W + SKQ8_STAGE_A + SKQ8_STAGE_DW); ++ ++#pragma unroll ++ for (int kb = 0; kb < SKQ8_KSTEP / 32; kb++) { ++ // W fragment for this warp's 16 rows (rows mhalf*16 + gid, +8). ++ const char * wbase = s_w + (mhalf * 16 + gid) * SKQ8_SW + kb * 32 + tid4 * 4; ++ const int a0 = *(const int *) (wbase); ++ const int a1 = *(const int *) (wbase + 8 * SKQ8_SW); ++ const int a2 = *(const int *) (wbase + 16); ++ const int a3 = *(const int *) (wbase + 8 * SKQ8_SW + 16); ++ const float dw0 = __half2float(s_dw[(mhalf * 16 + gid) * SKQ8_KTB + kb]); ++ const float dw1 = __half2float(s_dw[(mhalf * 16 + gid + 8) * SKQ8_KTB + kb]); ++#pragma unroll ++ for (int t = 0; t < 2; t++) { ++ const int nt = nt0 + t; ++ if (nt * 8 >= ncols) { ++ continue; // dead column tile (N padded to ncols < NPAD) ++ } ++ const char * bbase = s_a + (nt * 8 + gid) * SKQ8_SA + kb * 32 + tid4 * 4; ++ const int b0 = *(const int *) (bbase); ++ const int b1 = *(const int *) (bbase + 16); ++ int d0 = 0, d1 = 0, d2 = 0, d3 = 0; ++ skq8_mma(d0, d1, d2, d3, a0, a1, a2, a3, b0, b1); ++ const int c0 = nt * 8 + tid4 * 2; ++ const float da0 = s_da[c0 * SKQ8_KTB + kb]; ++ const float da1 = s_da[(c0 + 1) * SKQ8_KTB + kb]; ++ facc[t][0] += dw0 * da0 * (float) d0; ++ facc[t][1] += dw0 * da1 * (float) d1; ++ facc[t][2] += dw1 * da0 * (float) d2; ++ facc[t][3] += dw1 * da1 * (float) d3; ++ } ++ } ++ __syncthreads(); ++ } ++ ++ // Write back: C rows mhalf*16 + gid (+8), cols nt*8 + tid4*2 (+1). ++ // With a K-split grid, each split writes a private output plane. A ++ // follow-up kernel reduces those planes in a fixed order, avoiding the ++ // run-to-run numerical drift caused by unordered FP32 atomic additions. ++ const int row0 = row_blk + mhalf * 16 + gid; ++ const float b0 = (bias != nullptr && blockIdx.y == 0) ? bias[row0] : 0.0f; ++ const float b1 = (bias != nullptr && blockIdx.y == 0) ? bias[row0 + 8] : 0.0f; ++#pragma unroll ++ for (int t = 0; t < 2; t++) { ++ facc[t][0] += b0; ++ facc[t][1] += b0; ++ facc[t][2] += b1; ++ facc[t][3] += b1; ++ } ++#pragma unroll ++ for (int t = 0; t < 2; t++) { ++ const int c0 = (nt0 + t) * 8 + tid4 * 2; ++ if (gridDim.y > 1) { ++ // Plane stride covers the full logical output. `N` is only this ++ // grid.z tile's width (<= SKQ8_NPAD), so using it here aliases the ++ // next tile whenever N_total > SKQ8_NPAD. ++ const size_t split_base = (size_t) blockIdx.y * M * N_total; ++ if (c0 < N) { ++ C[split_base + row0 + (size_t) c0 * M] = facc[t][0]; ++ C[split_base + row0 + 8 + (size_t) c0 * M] = facc[t][2]; ++ } ++ if (c0 + 1 < N) { ++ C[split_base + row0 + (size_t) (c0 + 1) * M] = facc[t][1]; ++ C[split_base + row0 + 8 + (size_t) (c0 + 1) * M] = facc[t][3]; ++ } ++ } else { ++ if (c0 < N) { ++ C[row0 + (size_t) c0 * M] = facc[t][0]; ++ C[row0 + 8 + (size_t) c0 * M] = facc[t][2]; ++ } ++ if (c0 + 1 < N) { ++ C[row0 + (size_t) (c0 + 1) * M] = facc[t][1]; ++ C[row0 + 8 + (size_t) (c0 + 1) * M] = facc[t][3]; ++ } ++ } ++ } ++#else ++ GGML_UNUSED_VARS(Wq, Wd, Aq, Ad, bias, C, M, N_total, K, ncols_total); ++ NO_DEVICE_CODE; ++#endif ++} ++ ++static __global__ void skq8_reduce_splitk( ++ const float * __restrict__ partials, float * __restrict__ dst, ++ const int64_t count, const int ksplit) { ++ const int64_t i = (int64_t) blockIdx.x * blockDim.x + threadIdx.x; ++ if (i >= count) { ++ return; ++ } ++ float sum = partials[i]; ++ for (int split = 1; split < ksplit; ++split) { ++ sum += partials[(int64_t) split * count + i]; ++ } ++ dst[i] = sum; ++} ++ ++static __global__ void skq8_add_bias_f32( ++ float * __restrict__ dst, const float * __restrict__ bias, const int m, const int64_t count) { ++ const int64_t i = (int64_t) blockIdx.x * blockDim.x + threadIdx.x; ++ if (i >= count) { ++ return; ++ } ++ dst[i] += bias[i % m]; ++} ++ ++// Repacked-weight cache. Keyed by the weight tensor's device pointer; entries ++// are dropped when the device buffer holding the weight is freed or cleared, so ++// new weights at the same address are repacked again. Repacking happens on ++// first (eager/warmup) use, before any CUDA-graph capture. ++namespace { ++struct skq8_planes { ++ int8_t * qs; ++ half * d; ++}; ++std::unordered_map g_skq8_cache; ++std::mutex g_skq8_mutex; ++} // namespace ++ ++static const ggml_tensor * skq8_root(const ggml_tensor * t) { ++ while (t->view_src != nullptr) { ++ t = t->view_src; ++ } ++ return t; ++} ++ ++// This specialization targets encoder matrices, whose ne[1] is a time or token ++// dimension: the FastConformer encoders and the MagpieTTS text encoder, which ++// shares the "encoder." prefix. Keeping the domain explicit also prevents a ++// decoder's request batch (which commonly occupies ne[1]) from being mistaken ++// for a skinny time dimension. ++static bool skq8_encoder_weight(const ggml_tensor * root) { ++ return strncmp(root->name, "encoder.", 8) == 0; ++} ++ ++bool ggml_cuda_skinny_q8_repacked_in_place(const ggml_tensor * src0) { ++ // Only encoder weights are ever repacked, so every other Q8_0 weight skips ++ // the lock on MMVQ's per-call check. ++ if (src0->type != GGML_TYPE_Q8_0 || !skq8_encoder_weight(skq8_root(src0))) { ++ return false; ++ } ++ std::lock_guard lock(g_skq8_mutex); ++ const auto it = g_skq8_cache.find(src0->data); ++ return it != g_skq8_cache.end() && (const void *) it->second.qs == src0->data; ++} ++ ++void ggml_cuda_skinny_q8_forget(const void * base, size_t size) { ++ const uintptr_t begin = (uintptr_t) base; ++ std::lock_guard lock(g_skq8_mutex); ++ for (auto it = g_skq8_cache.begin(); it != g_skq8_cache.end();) { ++ // Compare as integers: the cache holds pointers into unrelated allocations. ++ const uintptr_t key = (uintptr_t) it->first; ++ if (key < begin || key - begin >= size) { ++ ++it; ++ continue; ++ } ++ const skq8_planes & planes = it->second; ++ // In-place and planar weights alias the tensor; only a separate ++ // (GGML_SKINNY_Q8_INPLACE=0) repack owns its planes. ++ if ((const void *) planes.qs != it->first) { ++ CUDA_CHECK(cudaFree(planes.qs)); ++ CUDA_CHECK(cudaFree(planes.d)); ++ } ++ it = g_skq8_cache.erase(it); ++ } ++} ++ ++bool ggml_cuda_skinny_q8_supported( ++ const ggml_tensor * src0, const ggml_tensor * src1, const ggml_tensor * dst) { ++ // GGML_SKINNY_Q8=0 disables the kernel for block Q8_0 weights. Planar weights ++ // cannot be read by any other kernel, so they ignore it. ++ static const bool disabled = []() { ++ const char * e = getenv("GGML_SKINNY_Q8"); ++ return e != nullptr && e[0] == '0'; ++ }(); ++ // Resolve views before checking the persistent model-weight marker. A ++ // serialized planar tensor is accepted only through its full allocation, ++ // since a normal ggml byte-offset view cannot describe two separate ++ // tensor-wide planes. ++ const ggml_tensor * root = skq8_root(src0); ++ const bool planar_q8 = (root->flags & GGML_TENSOR_FLAG_Q8_PLANAR) != 0; ++ if (disabled && !planar_q8) { ++ return false; ++ } ++ if (!skq8_encoder_weight(root)) { ++ return false; ++ } ++ if (!planar_q8 && strstr(root->name, ".linear_qkv.weight") != nullptr) { ++ return false; ++ } ++ // Tensor-wide planes cannot be addressed through a stock block-row view. ++ // Current encoder graphs use the full fused QKV tensor, never a weight ++ // view; reject defensively if that changes. ++ if (planar_q8 && src0 != root) { ++ return false; ++ } ++ ++ const int cc = ggml_cuda_info().devices[ggml_cuda_get_device()].cc; ++ const bool skinny_q8_available = ampere_mma_available(cc); ++ if (!planar_q8 && !skinny_q8_available) { ++ return false; ++ } ++ // The repack cache is keyed by src0->data and assumes the bytes are ++ // immutable from the outside: only accept long-lived weight buffers, never ++ // transient compute-pool tensors (whose addresses get reused). ++ if (src0->buffer == nullptr || ++ ggml_backend_buffer_get_usage(src0->buffer) == GGML_BACKEND_BUFFER_USAGE_COMPUTE) { ++ return false; ++ } ++ ++ // Weight-side (per-tensor, call-invariant) conditions. ++ const bool tensor_ok = src0->type == GGML_TYPE_Q8_0 && ggml_is_contiguous(src0) && ++ src0->ne[2] == 1 && src0->ne[3] == 1 && ++ src0->ne[0] % SKQ8_KSTEP == 0 && src0->ne[1] % SKQ8_ROWS == 0; ++ // Call-side conditions. ++ // A dense batched RHS [K,N,B...] is byte-identical to [K,N*B...]. Treat ++ // all outer columns as one logical N so a weight repacked by a scalar ++ // graph remains usable by a later true-batch graph. ++ const int64_t total_n = ggml_nelements(src1) / src1->ne[0]; ++ const bool call_ok = src1->type == GGML_TYPE_F32 && dst->type == GGML_TYPE_F32 && ++ ggml_is_contiguous(src1) && ggml_is_contiguous(dst) && ++ src1->ne[0] == src0->ne[0] && ++ ggml_nelements(dst) == src0->ne[1] * total_n; ++ ++ if (planar_q8) { ++ // N<=8 is handled by planar MMVQ. Wider calls must stay on this path ++ // because stock MMQ cannot interpret tensor-wide planes; skq8_gemm ++ // tiles arbitrary N in 64-column chunks. ++ GGML_ASSERT(tensor_ok && call_ok && "planar Q8 weight used in unsupported mul_mat shape"); ++ GGML_ASSERT((skinny_q8_available || total_n <= MMVQ_MAX_BATCH_SIZE) && ++ "wide planar Q8 requires an SM80+ CUDA kernel; use block Q8 on older GPUs"); ++ return skinny_q8_available && total_n > MMVQ_MAX_BATCH_SIZE; ++ } ++ ++ const bool repacked = [&]() { ++ std::lock_guard lock(g_skq8_mutex); ++ return g_skq8_cache.find(src0->data) != g_skq8_cache.end(); ++ }(); ++ if (repacked) { ++ // The weight was converted IN PLACE to the plane layout — the ++ // block_q8_0 bytes no longer exist, so every mul_mat on this tensor ++ // must come through here (skq8_run serves N <= MMVQ_MAX_BATCH_SIZE ++ // with planar MMVQ and tiles wider N). Falling back to mmq/mmvq would ++ // silently read garbage: abort loudly instead if a call shape we ++ // cannot serve ever appears. ++ GGML_ASSERT(call_ok && "skinny-q8: repacked weight used in an unsupported mul_mat shape"); ++ return true; ++ } ++ // Same rule as serialized planar weights above, applied from the first ++ // call: the kernel depends only on the call's column count, never on ++ // whether an earlier call already repacked the weight. N <= 8 stays on ++ // MMVQ (planar MMVQ after a repack computes the same bits); every wider ++ // call runs skq8, which repacks on first use. An N cap here would send ++ // wide calls to mmq before the first repack and to skq8 after it, making ++ // outputs depend on which requests the process served earlier. ++ return tensor_ok && call_ok && total_n > MMVQ_MAX_BATCH_SIZE; ++} ++ ++static void skq8_run( ++ ggml_backend_cuda_context & ctx, const ggml_tensor * src0, const ggml_tensor * src1, ++ const ggml_tensor * bias, ggml_tensor * dst) { ++ const int64_t M = src0->ne[1]; ++ const int64_t K = src0->ne[0]; ++ const int64_t N = ggml_nelements(src1) / src1->ne[0]; ++ const int64_t KB = K / 32; ++ ++ cudaStream_t stream = ctx.stream(); ++ // Only a repacked weight reaches here with N <= MMVQ_MAX_BATCH_SIZE (see ++ // ggml_cuda_skinny_q8_supported); it keeps the MMVQ math it had before. ++ const bool mmvq_cols = N <= MMVQ_MAX_BATCH_SIZE; ++ ++ // Repacked weight planes (create on first use). Default: IN PLACE. The plane ++ // layout (qs M*K + d M*KB*2) is byte-for-byte the same total size as ++ // block_q8_0 (M*KB*34), so it repacks through a transient pool staging buffer ++ // and copies back over the original allocation: zero extra weight memory (the ++ // cudaMalloc'd duplicate cost ~1.07 GB on parakeet-xxl). After this the ++ // tensor's block_q8_0 layout is GONE; the dispatch in ++ // ggml_cuda_skinny_q8_supported() therefore claims every mul_mat on a ++ // repacked tensor. ++ // ++ // GGML_SKINNY_Q8_INPLACE=0 forces the separate-buffer (cudaMalloc) layout. ++ // The in-place D2D memcpy is a stream-ordering hazard under multi-stream ++ // graph-split scheduling (llama.cpp's NMT decoder): it corrupts the GEMM even ++ // though the repacked bytes are correct. Callers that share the process with ++ // such a scheduler set GGML_SKINNY_Q8_INPLACE=0 (the NMT pipeline does this ++ // when enabled). The streaming-ASR encoder runtime has no such hazard. ++ skq8_planes planes = {}; ++ { ++ std::lock_guard lock(g_skq8_mutex); ++ auto it = g_skq8_cache.find(src0->data); ++ if (it != g_skq8_cache.end()) { ++ planes = it->second; ++ } else { ++ const size_t qs_bytes = (size_t) M * K; ++ const size_t d_bytes = (size_t) M * KB * sizeof(half); ++ GGML_ASSERT(qs_bytes + d_bytes == ggml_nbytes(src0)); ++ if ((src0->flags & GGML_TENSOR_FLAG_Q8_PLANAR) != 0) { ++ planes.qs = (int8_t *) src0->data; ++ planes.d = (half *) ((char *) src0->data + qs_bytes); ++ g_skq8_cache.emplace(src0->data, planes); ++ } else { ++ static const bool inplace = []() { ++ const char * e = getenv("GGML_SKINNY_Q8_INPLACE"); ++ return e == nullptr || e[0] != '0'; ++ }(); ++ const int64_t total = M * KB; ++ const int blocks = (int) ((total + 255) / 256); ++ if (inplace) { ++ ggml_cuda_pool_alloc staging(ctx.pool(), qs_bytes + d_bytes); ++ skq8_repack<<>>( ++ src0->data, staging.get(), (half *) (staging.get() + qs_bytes), M, KB); ++ CUDA_CHECK(cudaMemcpyAsync( ++ src0->data, staging.get(), qs_bytes + d_bytes, cudaMemcpyDeviceToDevice, ++ stream)); ++ planes.qs = (int8_t *) src0->data; ++ planes.d = (half *) ((char *) src0->data + qs_bytes); ++ // staging returns to the pool at scope exit; the copy is ++ // stream-ordered before any later reuse on this stream. ++ } else { ++ CUDA_CHECK(cudaMalloc(&planes.qs, qs_bytes)); ++ CUDA_CHECK(cudaMalloc(&planes.d, d_bytes)); ++ skq8_repack<<>>( ++ src0->data, planes.qs, planes.d, M, KB); ++ } ++ g_skq8_cache.emplace(src0->data, planes); ++ } ++ } ++ } ++ ++ if (mmvq_cols) { ++ // Serve MMVQ's column range with MMVQ, as before the repack. In-place ++ // (and serialized planar) planes alias the tensor bytes, so read them ++ // through planar MMVQ, which computes the same bits as block MMVQ; a ++ // separate (GGML_SKINNY_Q8_INPLACE=0) repack left the block bytes intact. ++ GGML_ASSERT(src1->type == GGML_TYPE_F32); ++ ggml_tensor weight = *src0; ++ if ((const void *) planes.qs == src0->data) { ++ weight.flags |= GGML_TENSOR_FLAG_Q8_PLANAR; ++ } ++ ggml_cuda_mul_mat_vec_q(ctx, &weight, src1, nullptr, dst); ++ if (bias != nullptr) { ++ // separate add, like the unfused mul_mat + add graph ++ const int64_t count = M * N; ++ skq8_add_bias_f32<<<(count + 255) / 256, 256, 0, stream>>>( ++ (float *) dst->data, (const float *) bias->data, (int) M, count); ++ } ++ return; ++ } ++ ++ ++ // Quantize activations into the zero-padded buffer (all columns, once — ++ // ntot may exceed NPAD; the GEMM below tiles over 64-col chunks). ++ const int ntot = (int) ((N + SKQ8_NPAD - 1) / SKQ8_NPAD * SKQ8_NPAD); ++ ggml_cuda_pool_alloc aq_alloc(ctx.pool(), (size_t) ntot * K); ++ ggml_cuda_pool_alloc ad_alloc(ctx.pool(), (size_t) ntot * KB); ++ { ++ const int64_t warps = (int64_t) ntot * KB; // one warp per (col, q8 block) ++ const int blocks = (int) ((warps * 32 + 255) / 256); ++ skq8_quantize<<>>( ++ (const float *) src1->data, aq_alloc.get(), ad_alloc.get(), (int) K, (int) N); ++ } ++ ++ // Split K for small-M shapes until there is enough block-level parallelism. ++ // Each split writes a private plane and a deterministic reduction combines ++ // the planes afterward; unordered atomics can amplify into visible ++ // B=1-versus-batched drift across a deep encoder. ++ // 1-step splits are allowed: no pipeline, but block-level K-parallelism. ++ // N > NPAD is mapped into grid.z. Weights are still read once per 64-column ++ // tile, but all tiles are visible to the scheduler in a single launch. ++ { ++ const int row_blocks = (int) (M / SKQ8_ROWS); ++ int ksplit = 1; ++ while (ksplit < 4 && row_blocks * ksplit < 128 && ++ K % ((int64_t) SKQ8_KSTEP * ksplit * 2) == 0) { ++ ksplit *= 2; ++ } ++ const size_t smem = (size_t) SKQ8_STAGES * SKQ8_STAGE_BYTES; ++ static std::once_flag attr_set_flag; ++ std::call_once(attr_set_flag, [&]() { ++ CUDA_CHECK(cudaFuncSetAttribute( ++ skq8_gemm, cudaFuncAttributeMaxDynamicSharedMemorySize, (int) smem)); ++ }); ++ const int ntiles = (ntot + SKQ8_NPAD - 1) / SKQ8_NPAD; ++ const dim3 grid(row_blocks, ksplit, ntiles); ++ const float * bias_data = bias != nullptr ? (const float *) bias->data : nullptr; ++ if (ksplit > 1) { ++ const int64_t count = M * N; ++ ggml_cuda_pool_alloc partials(ctx.pool(), (size_t) ksplit * count); ++ skq8_gemm<<>>( ++ planes.qs, planes.d, aq_alloc.get(), ad_alloc.get(), bias_data, partials.get(), ++ (int) M, (int) N, (int) K, ntot); ++ skq8_reduce_splitk<<<(count + 255) / 256, 256, 0, stream>>>( ++ partials.get(), (float *) dst->data, count, ksplit); ++ } else { ++ skq8_gemm<<>>( ++ planes.qs, planes.d, aq_alloc.get(), ad_alloc.get(), bias_data, ++ (float *) dst->data, (int) M, (int) N, (int) K, ntot); ++ } ++ } ++} ++ ++void ggml_cuda_mul_mat_skinny_q8( ++ ggml_backend_cuda_context & ctx, const ggml_tensor * src0, const ggml_tensor * src1, ++ ggml_tensor * dst) { ++ skq8_run(ctx, src0, src1, nullptr, dst); ++} ++ ++void ggml_cuda_mul_mat_skinny_q8_bias( ++ ggml_backend_cuda_context & ctx, const ggml_tensor * src0, const ggml_tensor * src1, ++ const ggml_tensor * bias, ggml_tensor * dst) { ++ skq8_run(ctx, src0, src1, bias, dst); ++} +diff --git a/ggml/src/ggml-cuda/skinny-q8.cuh b/ggml/src/ggml-cuda/skinny-q8.cuh +new file mode 100644 +index 0000000000000000000000000000000000000000..1847c2b1391a58f78937c1ef6f4edfccd1dea8d3 +--- /dev/null ++++ b/ggml/src/ggml-cuda/skinny-q8.cuh +@@ -0,0 +1,27 @@ ++// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. ++// SPDX-License-Identifier: Apache-2.0 ++#include "common.cuh" ++ ++// Skinny-N Q8_0 x F32 GEMM specialized for streaming encoders (claims every ++// call wider than MMVQ's 8 columns on an eligible weight). ++// See skinny-q8.cu for the design notes. ++ ++// Drop cached planes for weights in a device allocation being freed or cleared. ++void ggml_cuda_skinny_q8_forget(const void * base, size_t size); ++ ++// True when the Q8_0 weight src0 was repacked to the planar layout in place, so its ++// bytes are no longer block_q8_0 even though the tensor is not flagged planar. ++bool ggml_cuda_skinny_q8_repacked_in_place(const ggml_tensor * src0); ++ ++bool ggml_cuda_skinny_q8_supported( ++ const ggml_tensor * src0, const ggml_tensor * src1, const ggml_tensor * dst); ++ ++void ggml_cuda_mul_mat_skinny_q8( ++ ggml_backend_cuda_context & ctx, const ggml_tensor * src0, const ggml_tensor * src1, ++ ggml_tensor * dst); ++ ++// Fused variant: adds a row-vector bias ([M], F32) in the GEMM epilogue and ++// writes the result to `dst` (the bias-add node's buffer). ++void ggml_cuda_mul_mat_skinny_q8_bias( ++ ggml_backend_cuda_context & ctx, const ggml_tensor * src0, const ggml_tensor * src1, ++ const ggml_tensor * bias, ggml_tensor * dst); +diff --git a/ggml/src/ggml-cuda/vecdotq.cuh b/ggml/src/ggml-cuda/vecdotq.cuh +index f2a6f2009c9ac907b6002667a49dcb6c2e75be56..ce3ddbc5f8b2e58385e8017af6d5be4d66d601e1 100644 +--- a/ggml/src/ggml-cuda/vecdotq.cuh ++++ b/ggml/src/ggml-cuda/vecdotq.cuh +@@ -240,7 +240,7 @@ template static __device__ __forceinline__ float vec_dot_q5_1_q8_1_imp + return sumi*d5d8 + m5s8 / (QI5_1 / vdr); + } + +-#define VDR_Q8_0_Q8_1_MMVQ 2 ++#define VDR_Q8_0_Q8_1_MMVQ 4 + #define VDR_Q8_0_Q8_1_MMQ 8 + + template static __device__ __forceinline__ T vec_dot_q8_0_q8_1_impl( diff --git a/ggml-patches/0004-conv2d-dw-f16-kernel.patch b/patches/ggml-cuda-support-f16-weights-in-direct-depthwise-conv.patch similarity index 74% rename from ggml-patches/0004-conv2d-dw-f16-kernel.patch rename to patches/ggml-cuda-support-f16-weights-in-direct-depthwise-conv.patch index 3536674..7860ee0 100644 --- a/ggml-patches/0004-conv2d-dw-f16-kernel.patch +++ b/patches/ggml-cuda-support-f16-weights-in-direct-depthwise-conv.patch @@ -1,7 +1,16 @@ -diff --git a/src/ggml-cuda/conv2d-dw.cu b/src/ggml-cuda/conv2d-dw.cu -index 7583233b..db7ee6bf 100644 ---- a/src/ggml-cuda/conv2d-dw.cu -+++ b/src/ggml-cuda/conv2d-dw.cu +From: Prabhsimran Singh +Subject: ggml-cuda: support F16 weights in direct depthwise conv + +CONV_2D_DW reads F16 kernels with F32 input and output, so the FastConformer +depthwise convolutions can run the direct kernel without an im2col lowering. + +Upstream: ggml-org/llama.cpp#29064 (open) makes the same change. Drop this +patch once it is merged. + +diff --git a/ggml/src/ggml-cuda/conv2d-dw.cu b/ggml/src/ggml-cuda/conv2d-dw.cu +index 7583233b1b7cd4bee67cf7b84434ac7b41f17165..db7ee6bf62077dc9ed518ad2d4d13fbe1a8f3926 100644 +--- a/ggml/src/ggml-cuda/conv2d-dw.cu ++++ b/ggml/src/ggml-cuda/conv2d-dw.cu @@ -78,8 +78,8 @@ struct cwhn_layout { } }; @@ -66,3 +75,17 @@ index 7583233b..db7ee6bf 100644 } else { GGML_ABORT("Unsupported memory layout for conv_2d_dw"); } +diff --git a/ggml/src/ggml-cuda/ggml-cuda.cu b/ggml/src/ggml-cuda/ggml-cuda.cu +index ecf831b6446e1ddb1f0efb28c9c01d18e3591d6f..6730a6e28ef9ab02c74cf4e164ff2e8ac5cfb0bf 100644 +--- a/ggml/src/ggml-cuda/ggml-cuda.cu ++++ b/ggml/src/ggml-cuda/ggml-cuda.cu +@@ -5572,7 +5572,8 @@ static bool ggml_backend_cuda_device_supports_op(ggml_backend_dev_t dev, const g + case GGML_OP_CONV_2D: + return (ggml_is_contiguous(op->src[0]) && ggml_is_contiguous(op->src[1])); + case GGML_OP_CONV_2D_DW: +- return op->src[0]->type == GGML_TYPE_F32; ++ return (op->src[0]->type == GGML_TYPE_F32 || op->src[0]->type == GGML_TYPE_F16) && ++ op->src[1]->type == GGML_TYPE_F32 && op->type == GGML_TYPE_F32; + case GGML_OP_CONV_TRANSPOSE_2D: + case GGML_OP_POOL_1D: + case GGML_OP_POOL_2D: diff --git a/patches/ggml-cuda-target-jetson-thor-sm110.patch b/patches/ggml-cuda-target-jetson-thor-sm110.patch new file mode 100644 index 0000000..32b6c05 --- /dev/null +++ b/patches/ggml-cuda-target-jetson-thor-sm110.patch @@ -0,0 +1,168 @@ +From: Prabhsimran Singh +Subject: ggml-cuda: target Jetson Thor (SM110) + +Adds 110a to the default architectures with CUDA 13, maps plain 100/110/12X +to their architecture-specific targets, detects Thor, and prefers cuBLAS +over the MMF/MMVF warp kernels there. + +Upstream: candidate (needs Thor performance data). + +diff --git a/ggml/src/ggml-cuda/CMakeLists.txt b/ggml/src/ggml-cuda/CMakeLists.txt +index 2254090cbab05c04d8a70e784484d7f5151a1c22..4ec8f415b9ab058975fe7b69716e7c522fabd8e9 100644 +--- a/ggml/src/ggml-cuda/CMakeLists.txt ++++ b/ggml/src/ggml-cuda/CMakeLists.txt +@@ -16,7 +16,8 @@ if (CUDAToolkit_FOUND) + # 86 == RTX 3000, needs CUDA v11.1 + # 89 == RTX 4000, needs CUDA v11.8 + # 90 == Hopper H100/200, needs CUDA v11.8 +- # 120 == Blackwell, needs CUDA v12.8, FP4 tensor cores ++ # 110 == Jetson Thor, needs CUDA v13.0, Blackwell tensor cores ++ # 120 == Blackwell, needs CUDA v12.8, SM120 FP4 tensor cores + # + # XX-virtual == compile CUDA code as PTX, do JIT compilation to binary code on first run + # XX-real == compile CUDA code as device code for this specific architecture +@@ -37,6 +38,10 @@ if (CUDAToolkit_FOUND) + list(APPEND CMAKE_CUDA_ARCHITECTURES 89-real 90-virtual) + endif() + ++ if (CUDAToolkit_VERSION VERSION_GREATER_EQUAL "13.0") ++ list(APPEND CMAKE_CUDA_ARCHITECTURES 110a-real) ++ endif() ++ + if (CUDAToolkit_VERSION VERSION_GREATER_EQUAL "12.8") + # The CUDA architecture 120f-virtual would in principle work for Blackwell support + # but the newly added "f" suffix conflicted with a preexising regex for validating CUDA architectures in CMake. +@@ -72,16 +77,16 @@ if (CUDAToolkit_FOUND) + FetchContent_MakeAvailable(CCCL) + endif() + +- # Replace any plain 12X CUDA architectures with their "architecture-specific" equivalents 12Xa. +- # 12X is forwards-compatible, 12Xa is not. +- # Notably the Blackwell FP4 tensor core instructions are not forwards compatible and therefore need 12Xa. ++ # Replace plain Blackwell CUDA architectures with their "architecture-specific" equivalents. ++ # 10X/11X/12X are forwards-compatible, 10Xa/11Xa/12Xa are not. ++ # Notably the Blackwell tensor core instructions are not forwards compatible and therefore need architecture-specific targets. + # But while 12X vs. 12Xa can be checked in device code there is (to my knowledge) no easy way to do the same check in host code. +- # So for now just replace all instances of 12X with 12Xa, this should be fine until Rubin is released. ++ # So for now just replace the supported plain Blackwell targets with architecture-specific targets. + foreach(ARCHS IN ITEMS CMAKE_CUDA_ARCHITECTURES CMAKE_CUDA_ARCHITECTURES_NATIVE) + set(FIXED_ARCHS "") + foreach(ARCH IN LISTS ${ARCHS}) +- if (ARCH MATCHES "^12[0-9](-real|-virtual)?$") +- string(REGEX REPLACE "^(12[0-9])((-real|-virtual)?)$" "\\1a\\2" FIXED_ARCH ${ARCH}) ++ if (ARCH MATCHES "^(100|110|12[0-9])(-real|-virtual)?$") ++ string(REGEX REPLACE "^(100|110|12[0-9])((-real|-virtual)?)$" "\\1a\\2" FIXED_ARCH ${ARCH}) + message(STATUS "Replacing ${ARCH} in ${ARCHS} with ${FIXED_ARCH}") + list(APPEND FIXED_ARCHS "${FIXED_ARCH}") + else() +@@ -91,8 +96,8 @@ if (CUDAToolkit_FOUND) + set(${ARCHS} ${FIXED_ARCHS}) + endforeach() + +- # If we try to compile a "native" build it will use the 12X architectures and fail. +- # So we should instead use the native architectures as determined by CMake after replacing 12X with 12Xa. ++ # If we try to compile a "native" build it may use plain Blackwell architectures and fail. ++ # So we should instead use the native architectures as determined by CMake after replacing them with architecture-specific forms. + # But if at the time of the build no GPUs are connected at all CMAKE_CUDA_ARCHITECTURES will contain garbage that we should not use. + if (CMAKE_CUDA_ARCHITECTURES STREQUAL "native" AND CMAKE_CUDA_ARCHITECTURES_NATIVE MATCHES "^[0-9]+(a|f)?(-real|-virtual)?(;[0-9]+(a|f)?(-real|-virtual)?|;)*$") + set(CMAKE_CUDA_ARCHITECTURES ${CMAKE_CUDA_ARCHITECTURES_NATIVE}) +diff --git a/ggml/src/ggml-cuda/common.cuh b/ggml/src/ggml-cuda/common.cuh +index 25893da48941518b5e0a8cfa20c1538fd2a5d305..f5f4465b5e0cade7888d67728b71d8d745eb1939 100644 +--- a/ggml/src/ggml-cuda/common.cuh ++++ b/ggml/src/ggml-cuda/common.cuh +@@ -55,7 +55,10 @@ + #define GGML_CUDA_CC_ORIN 870 + #define GGML_CUDA_CC_ADA_LOVELACE 890 + #define GGML_CUDA_CC_HOPPER 900 +-// While BW spans CC 1000, 1100 & 1200, we are integrating Tensor Core instructions available to 1200 family, see ++// Jetson Thor is Blackwell SM110. The hand-written FP4 block-scale PTX below is currently SM120-only, but SM110 ++// should still be detected so ggml can favor CUDA library kernels that may use Thor tcgen05 tensor cores. ++#define GGML_CUDA_CC_THOR 1100 ++// While BW spans CC 1000, 1100 & 1200, the hand-written FP4 path integrates Tensor Core instructions available to 1200 family, see + // https://docs.nvidia.com/cutlass/media/docs/cpp/blackwell_functionality.html#blackwell-sm120-gemms + #define GGML_CUDA_CC_BLACKWELL 1200 + #define GGML_CUDA_CC_DGX_SPARK 1210 +@@ -357,6 +360,10 @@ static bool amd_wmma_available(const int cc) { + return (GGML_CUDA_CC_IS_RDNA4(cc) || GGML_CUDA_CC_IS_RDNA3(cc)); + } + ++static bool thor_mma_available(const int cc) { ++ return GGML_CUDA_CC_IS_NVIDIA(cc) && cc == GGML_CUDA_CC_THOR; ++} ++ + static bool volta_mma_available(const int cc) { + return GGML_CUDA_CC_IS_NVIDIA(cc) && ggml_cuda_highest_compiled_arch(cc) == GGML_CUDA_CC_VOLTA; + } +diff --git a/ggml/src/ggml-cuda/ggml-cuda.cu b/ggml/src/ggml-cuda/ggml-cuda.cu +index 32b96b5e7fdb48a02ad255dfab688520d86ea884..1c0c25329d832ca97c327df4ab2329dddc2d5013 100644 +--- a/ggml/src/ggml-cuda/ggml-cuda.cu ++++ b/ggml/src/ggml-cuda/ggml-cuda.cu +@@ -5865,6 +5865,12 @@ static ggml_backend_feature * ggml_backend_cuda_get_features(ggml_backend_reg_t + + { + const auto & info = ggml_cuda_info(); ++ for (int id = 0; id < info.device_count; ++id) { ++ if (thor_mma_available(info.devices[id].cc)) { ++ features.push_back({ "BLACKWELL_THOR_SM110", "1"}); ++ break; ++ } ++ } + for (int id = 0; id < info.device_count; ++id) { + if (blackwell_mma_available(info.devices[id].cc)) { + features.push_back({ "BLACKWELL_NATIVE_FP4", "1"}); +diff --git a/ggml/src/ggml-cuda/mmf.cu b/ggml/src/ggml-cuda/mmf.cu +index 646a5899c803bd337f0e5c05d239af02898da865..741ecbf1da0547ff94f5bf69f4db3c28003d68fd 100644 +--- a/ggml/src/ggml-cuda/mmf.cu ++++ b/ggml/src/ggml-cuda/mmf.cu +@@ -159,6 +159,11 @@ bool ggml_cuda_should_use_mmf(enum ggml_type type, int cc, int warp_size, const + return false; + } + ++ // Thor's tcgen05 paths are provided by CUDA libraries today; avoid ggml's warp-level matrix kernels there. ++ if (thor_mma_available(cc) && (type == GGML_TYPE_F32 || type == GGML_TYPE_F16 || type == GGML_TYPE_BF16)) { ++ return false; ++ } ++ + if (mul_mat_id) { + if (src0_ne[1] <= 1024 && src1_ncols > 512) { + return false; +diff --git a/ggml/src/ggml-cuda/mmvf.cu b/ggml/src/ggml-cuda/mmvf.cu +index bd5c5d421a4c9acef77dff6070c5463b1a319b01..1db6b47b406c2615ca76f8adb6befddbe1698ba6 100644 +--- a/ggml/src/ggml-cuda/mmvf.cu ++++ b/ggml/src/ggml-cuda/mmvf.cu +@@ -806,9 +806,15 @@ bool ggml_cuda_should_use_mmvf(enum ggml_type type, int cc, const int64_t * src0 + } + } + ++ // Thor's tcgen05 paths are provided by CUDA libraries today; avoid ggml's warp-level vector kernels there. ++ const bool prefer_cublas_tcgen05 = thor_mma_available(cc); ++ + switch (type) { + case GGML_TYPE_F32: + if (GGML_CUDA_CC_IS_NVIDIA(cc)) { ++ if (prefer_cublas_tcgen05) { ++ return false; ++ } + if (ampere_mma_available(cc)) { + return ne11 <= 3; + } +@@ -825,6 +831,9 @@ bool ggml_cuda_should_use_mmvf(enum ggml_type type, int cc, const int64_t * src0 + return ne11 <= 8; + case GGML_TYPE_F16: + if (GGML_CUDA_CC_IS_NVIDIA(cc)) { ++ if (prefer_cublas_tcgen05) { ++ return false; ++ } + const bool src0_small = (src0_ne[1] <= 512 || src0_ne[2]*src0_ne[3] == 1); + if (ampere_mma_available(cc)) { + return src0_small && ne11 == 1; +@@ -851,6 +860,9 @@ bool ggml_cuda_should_use_mmvf(enum ggml_type type, int cc, const int64_t * src0 + return ne11 <= 8; + case GGML_TYPE_BF16: + if (GGML_CUDA_CC_IS_NVIDIA(cc)) { ++ if (prefer_cublas_tcgen05) { ++ return false; ++ } + const bool src0_small = (src0_ne[1] <= 512 || src0_ne[2]*src0_ne[3] == 1); + if (ampere_mma_available(cc)) { + return src0_small && ne11 == 1; diff --git a/ggml-patches/0024-cuda-im2col-1d-tiled.patch b/patches/ggml-cuda-tiled-1d-im2col.patch similarity index 78% rename from ggml-patches/0024-cuda-im2col-1d-tiled.patch rename to patches/ggml-cuda-tiled-1d-im2col.patch index 69b7f38..2bcf5ca 100644 --- a/ggml-patches/0024-cuda-im2col-1d-tiled.patch +++ b/patches/ggml-cuda-tiled-1d-im2col.patch @@ -1,9 +1,26 @@ -diff --git a/src/ggml-cuda/im2col.cu b/src/ggml-cuda/im2col.cu -index c2cc16fe..8e243e4f 100644 ---- a/src/ggml-cuda/im2col.cu -+++ b/src/ggml-cuda/im2col.cu -@@ -44,12 +44,72 @@ static __global__ void im2col_kernel( - GGML_UNUSED(KH); +From: Prabhsimran Singh +Subject: ggml-cuda: tiled 1D im2col + +A tiled, coalesced fast path for 1D im2col (IH == KH == OH == 1). A block +stages a channel-by-time input tile through shared memory, then consecutive +threads write consecutive output columns for a run of output rows. The generic +kernel strides across channels and re-reads every input element KW times. +Speeds up the NanoCodec 1D convolutions that go through im2col. + +Upstream: candidate (needs a benchmark). + +diff --git a/ggml/src/ggml-cuda/im2col.cu b/ggml/src/ggml-cuda/im2col.cu +index d377f28564393bbf70631f15ac866fc224090593..883b721d7305883bf60534e556a512291bf1af36 100644 +--- a/ggml/src/ggml-cuda/im2col.cu ++++ b/ggml/src/ggml-cuda/im2col.cu +@@ -1,4 +1,5 @@ + #include "im2col.cuh" ++#include "convert.cuh" + + #define MAX_GRIDDIM_Y 65535 + #define MAX_GRIDDIM_Z 65535 +@@ -44,12 +45,72 @@ static __global__ void im2col_kernel( + } } +// 1D fast path: [N, IC, IW] => [N, OW, IC*KW]. The generic kernel assigns consecutive threads @@ -73,5 +90,5 @@ index c2cc16fe..8e243e4f 100644 + } + } const int64_t IC_KH_KW = IC * KH * KW; - const int64_t num_blocks = (IC_KH_KW + CUDA_IM2COL_BLOCK_SIZE - 1) / CUDA_IM2COL_BLOCK_SIZE; const int64_t N_OH = N * OH; + const int64_t KH_KW = KW*KH; diff --git a/patches/ggml-cuda-two-column-mmvf-epilogue.patch b/patches/ggml-cuda-two-column-mmvf-epilogue.patch new file mode 100644 index 0000000..d744798 --- /dev/null +++ b/patches/ggml-cuda-two-column-mmvf-epilogue.patch @@ -0,0 +1,78 @@ +From: Prabhsimran Singh +Subject: ggml-cuda: two-column MMVF epilogue + +Allows MMVF bias/GLU fusion for up to two destination columns, so MagpieTTS +paired CFG lanes share every weight-row load. + +Upstream: candidate (generalize to ncols <= N, add m=2 fusion tests, perf data). + +diff --git a/ggml/src/ggml-cuda/ggml-cuda.cu b/ggml/src/ggml-cuda/ggml-cuda.cu +index ee47bc09e20c299d8813080a48415a2e958d83d6..141caeca47a748ccb3916f98d980d220ee63217f 100644 +--- a/ggml/src/ggml-cuda/ggml-cuda.cu ++++ b/ggml/src/ggml-cuda/ggml-cuda.cu +@@ -1783,8 +1783,8 @@ static bool ggml_cuda_should_fuse_mul_mat_vec_f(const ggml_tensor * tensor) { + const int cc = ggml_cuda_info().devices[ggml_cuda_get_device()].cc; + use_mul_mat_vec_f = use_mul_mat_vec_f && ggml_cuda_should_use_mmvf(src0->type, cc, src0->ne, src0->nb, is_mul_mat_id ? src1->ne[2] : src1->ne[1]); + +- //we only support fusion for ncols_dst = 1 +- if (tensor->op == GGML_OP_MUL_MAT && dst->ne[1] != 1) { ++ // MMVF supports a two-column epilogue so paired CFG lanes can share each weight-row load. ++ if (tensor->op == GGML_OP_MUL_MAT && dst->ne[1] > 2) { + return false; + } + +diff --git a/ggml/src/ggml-cuda/mmvf.cu b/ggml/src/ggml-cuda/mmvf.cu +index 1db6b47b406c2615ca76f8adb6befddbe1698ba6..ddb4895df364f97100de2e508f7bb512f0aad215 100644 +--- a/ggml/src/ggml-cuda/mmvf.cu ++++ b/ggml/src/ggml-cuda/mmvf.cu +@@ -395,7 +395,10 @@ static void mul_mat_vec_f_switch_fusion( + const ggml_cuda_kernel_launch_params launch_params = {block_nums, block_dims, nbytes_shared, stream}; + + const bool has_fusion = fusion.gate != nullptr || fusion.x_bias != nullptr || fusion.gate_bias != nullptr; +- if constexpr (ncols_dst == 1) { ++ // The same epilogue addressing works for two adjacent columns. This is especially useful ++ // for classifier-free guidance: both lanes share the weight-row load and still fold the ++ // residual/bias writeback into the projection kernel. ++ if constexpr (ncols_dst <= 2) { + if (has_fusion) { + ggml_cuda_kernel_launch(mul_mat_vec_f, launch_params, + x, y, ids, fusion, dst, ncols, nchannels_y, stride_row, stride_col_y, stride_col_dst, +@@ -405,7 +408,7 @@ static void mul_mat_vec_f_switch_fusion( + } + } + +- GGML_ASSERT(!has_fusion && "fusion only supported for ncols_dst=1"); ++ GGML_ASSERT(!has_fusion && "fusion only supported for ncols_dst<=2"); + + ggml_cuda_kernel_launch(mul_mat_vec_f, launch_params, + x, y, ids, fusion, dst, ncols, nchannels_y, stride_row, stride_col_y, stride_col_dst, +@@ -448,6 +451,11 @@ void launch_mul_mat_vec_f_cuda( + block_size_best = block_size; + } + } ++ if constexpr (std::is_same_v) { ++ if (warp_size == 32 && ncols >= 768 && ncols_dst <= 2) { ++ block_size_best = 96; ++ } ++ } + + const bool has_fusion = fusion.gate != nullptr || fusion.x_bias != nullptr || fusion.gate_bias != nullptr; + +@@ -662,7 +670,7 @@ void ggml_cuda_mul_mat_vec_f(ggml_backend_cuda_context & ctx, const ggml_tensor + + if (fusion) { + GGML_ASSERT( !ids || dst->ne[2] == 1); +- GGML_ASSERT( ids || dst->ne[1] == 1); ++ GGML_ASSERT( ids || dst->ne[1] <= 2); + if (fusion->x_bias) { + GGML_ASSERT(fusion->x_bias->type == GGML_TYPE_F32); + GGML_ASSERT(fusion->x_bias->ne[0] == dst->ne[0]); +@@ -836,7 +844,7 @@ bool ggml_cuda_should_use_mmvf(enum ggml_type type, int cc, const int64_t * src0 + } + const bool src0_small = (src0_ne[1] <= 512 || src0_ne[2]*src0_ne[3] == 1); + if (ampere_mma_available(cc)) { +- return src0_small && ne11 == 1; ++ return src0_small && ne11 <= 2; + } + if (cc >= GGML_CUDA_CC_ADA_LOVELACE) { + return src0_small && ne11 <= 4; diff --git a/patches/ggml-cuda-vectorized-contiguous-set_rows.patch b/patches/ggml-cuda-vectorized-contiguous-set_rows.patch new file mode 100644 index 0000000..0ea08b6 --- /dev/null +++ b/patches/ggml-cuda-vectorized-contiguous-set_rows.patch @@ -0,0 +1,74 @@ +From: Prabhsimran Singh +Subject: ggml-cuda: vectorized contiguous set_rows + +Writes contiguous F32 rows (ne0 >= 1024) with float4 stores, one block per +row, for the streaming K/V and convolution state arenas. + +Upstream: candidate (needs F16, the PDL launch helper and perf cases). + +diff --git a/ggml/src/ggml-cuda/set-rows.cu b/ggml/src/ggml-cuda/set-rows.cu +index 4659970651e59538842d72cfb4d30815c1fd7f25..8563338c0ec74c2aee3b00e0c8e1e9fd18983b1e 100644 +--- a/ggml/src/ggml-cuda/set-rows.cu ++++ b/ggml/src/ggml-cuda/set-rows.cu +@@ -176,6 +176,26 @@ static __global__ void k_set_rows(const src_t * src0_ptr, + GGML_UNUSED(ne13); + } + ++template ++static __global__ void k_set_rows_contiguous_f32x4( ++ const float * __restrict__ src0, const idx_t * __restrict__ rows, ++ float * __restrict__ dst, const int row_elements, const size_t dst_row_stride, ++ const size_t src_plane_stride, const size_t dst_plane_stride, ++ const size_t row_index_plane_stride) { ++ const int source_row = (int) blockIdx.x; ++ const int plane = (int) blockIdx.y; ++ const int64_t destination_row = ++ (int64_t) rows[source_row + (size_t) plane * row_index_plane_stride]; ++ const float4 * src = (const float4 *) ( ++ src0 + (size_t) plane * src_plane_stride + (size_t) source_row * row_elements); ++ float4 * out = (float4 *) ( ++ dst + (size_t) plane * dst_plane_stride + destination_row * dst_row_stride); ++ const int vectors = row_elements / 4; ++ for (int i = threadIdx.x; i < vectors; i += blockDim.x) { ++ out[i] = src[i]; ++ } ++} ++ + template + static void set_rows_cuda( + const src_t * src0_d, const idx_t * src1_d, dst_t * dst_d, +@@ -380,6 +400,34 @@ void ggml_cuda_op_set_rows(ggml_backend_cuda_context & ctx, ggml_tensor * dst) { + GGML_ASSERT(src0->type == GGML_TYPE_F32 || (src0->type == GGML_TYPE_F16 && dst->type == GGML_TYPE_F16)); + GGML_ASSERT(src1->type == GGML_TYPE_I64 || src1->type == GGML_TYPE_I32); + ++ const bool contiguous_cache_rows = ++ src0->type == GGML_TYPE_F32 && dst->type == GGML_TYPE_F32 && src0->ne[3] == 1 && ++ src1->ne[1] == src0->ne[2] && src1->ne[2] == 1 && src1->ne[3] == 1 && ++ dst->ne[2] == src0->ne[2] && dst->ne[3] == 1 && src0->ne[0] == dst->ne[0] && ++ src0->ne[1] == src1->ne[0] && ggml_is_contiguous(src0) && ++ dst->nb[0] == sizeof(float) && dst->nb[1] % sizeof(float4) == 0 && ++ dst->nb[2] % sizeof(float4) == 0 && (uintptr_t) dst->data % sizeof(float4) == 0 && ++ (uintptr_t) src0->data % sizeof(float4) == 0 && ++ src1->nb[0] == ggml_type_size(src1->type) && src0->ne[2] <= 65535 && ++ src0->ne[0] >= 1024 && src0->ne[0] % 4 == 0; ++ if (contiguous_cache_rows) { ++ const dim3 grid((unsigned) src0->ne[1], (unsigned) src0->ne[2]); ++ if (src1->type == GGML_TYPE_I64) { ++ k_set_rows_contiguous_f32x4<<>>( ++ (const float *) src0->data, (const int64_t *) src1->data, ++ (float *) dst->data, (int) src0->ne[0], dst->nb[1] / sizeof(float), ++ src0->nb[2] / sizeof(float), dst->nb[2] / sizeof(float), ++ src1->nb[1] / sizeof(int64_t)); ++ } else { ++ k_set_rows_contiguous_f32x4<<>>( ++ (const float *) src0->data, (const int32_t *) src1->data, ++ (float *) dst->data, (int) src0->ne[0], dst->nb[1] / sizeof(float), ++ src0->nb[2] / sizeof(float), dst->nb[2] / sizeof(float), ++ src1->nb[1] / sizeof(int32_t)); ++ } ++ return; ++ } ++ + if (src0->type == GGML_TYPE_F32) { + if (src1->type == GGML_TYPE_I64) { + set_rows_cuda(ctx, src0, src1, dst); diff --git a/patches/ggml-fastconformer-cuda-fusions-and-sigmoid-glu.patch b/patches/ggml-fastconformer-cuda-fusions-and-sigmoid-glu.patch new file mode 100644 index 0000000..8c04f96 --- /dev/null +++ b/patches/ggml-fastconformer-cuda-fusions-and-sigmoid-glu.patch @@ -0,0 +1,882 @@ +From: Prabhsimran Singh +Subject: ggml: FastConformer CUDA fusions and sigmoid GLU + +- GGML_GLU_OP_SIGMOID (PyTorch nn.GLU) with CPU and CUDA implementations. +- Macaron residual (SCALE + ADD), BatchNorm (+ transpose + SiLU) and + LayerNorm -> BF16/F16 cast epilogues, plus LayerNorm launch tuning. + +Used by the FastConformer encoder in CUDA builds with NEMO_SPEECH_GGML_PATCHED. + +Upstream: GGML_GLU_OP_SIGMOID is a candidate; the other fusions are +NeMo-specific. + +Co-authored-by: anand-nv <105917641+anand-nv@users.noreply.github.com> + +diff --git a/ggml/include/ggml.h b/ggml/include/ggml.h +index 19ade01735e6c072fcb7752ef7955ac5d2afe19e..6197891dbb20ad7bfd322abe731cf6fed5162eb3 100644 +--- a/ggml/include/ggml.h ++++ b/ggml/include/ggml.h +@@ -641,6 +641,7 @@ extern "C" { + GGML_GLU_OP_GEGLU_ERF, + GGML_GLU_OP_GEGLU_QUICK, + GGML_GLU_OP_SWIGLU_CLAMP, ++ GGML_GLU_OP_SIGMOID, + + GGML_GLU_OP_COUNT, + }; +diff --git a/ggml/src/ggml-cpu/ggml-cpu.c b/ggml/src/ggml-cpu/ggml-cpu.c +index 4fb2e8b6f7236c95e66a0be167d531f154cfd4d1..01893e3bd07a2b98b110865a364216b56cd652d7 100644 +--- a/ggml/src/ggml-cpu/ggml-cpu.c ++++ b/ggml/src/ggml-cpu/ggml-cpu.c +@@ -2349,6 +2349,7 @@ static int ggml_get_n_tasks(struct ggml_tensor * node, int n_threads) { + case GGML_GLU_OP_GEGLU_ERF: + case GGML_GLU_OP_GEGLU_QUICK: + case GGML_GLU_OP_SWIGLU_CLAMP: ++ case GGML_GLU_OP_SIGMOID: + { + n_tasks = n_threads; + } break; +diff --git a/ggml/src/ggml-cpu/ops.cpp b/ggml/src/ggml-cpu/ops.cpp +index ba00a0a73ed8d3b42fc1e840d686918a05e9c327..b05475768dfe92655a33bf876895b2becbf30488 100644 +--- a/ggml/src/ggml-cpu/ops.cpp ++++ b/ggml/src/ggml-cpu/ops.cpp +@@ -3030,6 +3030,149 @@ static void ggml_compute_forward_reglu( + } + } + ++// ggml_compute_forward_sigmoid_glu ++ ++static void ggml_compute_forward_sigmoid_glu_f32( ++ const ggml_compute_params * params, ++ ggml_tensor * dst) { ++ ++ const ggml_tensor * src0 = dst->src[0]; ++ const ggml_tensor * src1 = dst->src[1]; ++ char * src0_d = (char *) src0->data; ++ char * src1_d = (char *) (src1 ? src1->data : src0->data); ++ const size_t src0_o = src0->nb[1]; ++ const size_t src1_o = src1 ? src1->nb[1] : src0->nb[1]; ++ ++ GGML_ASSERT(ggml_is_contiguous_1(src0)); ++ GGML_ASSERT(ggml_is_contiguous_1(dst)); ++ ++ if (src1) { ++ GGML_ASSERT(ggml_is_contiguous_1(src1)); ++ GGML_ASSERT(src0->type == src1->type); ++ } ++ ++ const int ith = params->ith; ++ const int nth = params->nth; ++ ++ const int nc = src1 ? src0->ne[0] : src0->ne[0] / 2; ++ const int nr = ggml_nrows(src0); ++ ++ GGML_ASSERT(dst->ne[0] == nc); ++ GGML_ASSERT(ggml_nrows(dst) == nr); ++ ++ const int32_t swapped = ggml_get_op_params_i32(dst, 1); ++ ++ // rows per thread ++ const int dr = (nr + nth - 1)/nth; ++ ++ // row range for this thread ++ const int ir0 = dr*ith; ++ const int ir1 = MIN(ir0 + dr, nr); ++ ++ for (int i1 = ir0; i1 < ir1; i1++) { ++ float * src0_p = (float *) (src0_d + i1*src0_o); ++ float * src1_p = (float *) (src1_d + i1*src1_o); ++ ++ if (!src1) { ++ src0_p += swapped ? nc : 0; ++ src1_p += swapped ? 0 : nc; ++ } ++ ++ ggml_vec_sigmoid_glu_f32(nc, (float *) ((char *) dst->data + i1*(dst->nb[1])), src0_p, src1_p); ++ ++#ifndef NDEBUG ++ for (int k = 0; k < nc; k++) { ++ const float x = ((float *) ((char *) dst->data + i1*( dst->nb[1])))[k]; ++ GGML_UNUSED(x); ++ assert(!isnan(x)); ++ assert(!isinf(x)); ++ } ++#endif // NDEBUG ++ } ++} ++ ++static void ggml_compute_forward_sigmoid_glu_f16( ++ const ggml_compute_params * params, ++ ggml_tensor * dst) { ++ ++ const ggml_tensor * src0 = dst->src[0]; ++ const ggml_tensor * src1 = dst->src[1]; ++ char * src0_d = (char *) src0->data; ++ char * src1_d = (char *) (src1 ? src1->data : src0->data); ++ const size_t src0_o = src0->nb[1]; ++ const size_t src1_o = src1 ? src1->nb[1] : src0->nb[1]; ++ ++ GGML_ASSERT(ggml_is_contiguous_1(src0)); ++ GGML_ASSERT(ggml_is_contiguous_1(dst)); ++ ++ if (src1) { ++ GGML_ASSERT(ggml_is_contiguous_1(src1)); ++ GGML_ASSERT(src0->type == src1->type); ++ } ++ ++ const int ith = params->ith; ++ const int nth = params->nth; ++ ++ const int nc = src1 ? src0->ne[0] : src0->ne[0] / 2; ++ const int nr = ggml_nrows(src0); ++ ++ GGML_ASSERT(dst->ne[0] == nc); ++ GGML_ASSERT(ggml_nrows(dst) == nr); ++ ++ const int32_t swapped = ggml_get_op_params_i32(dst, 1); ++ ++ // rows per thread ++ const int dr = (nr + nth - 1)/nth; ++ ++ // row range for this thread ++ const int ir0 = dr*ith; ++ const int ir1 = MIN(ir0 + dr, nr); ++ ++ for (int i1 = ir0; i1 < ir1; i1++) { ++ ggml_fp16_t * src0_p = (ggml_fp16_t *) (src0_d + i1*src0_o); ++ ggml_fp16_t * src1_p = (ggml_fp16_t *) (src1_d + i1*src1_o); ++ ++ if (!src1) { ++ src0_p += swapped ? nc : 0; ++ src1_p += swapped ? 0 : nc; ++ } ++ ++ ggml_vec_sigmoid_glu_f16(nc, (ggml_fp16_t *) ((char *) dst->data + i1*(dst->nb[1])), src0_p, src1_p); ++ ++#ifndef NDEBUG ++ for (int k = 0; k < nc; k++) { ++ const ggml_fp16_t x = ((ggml_fp16_t *) ((char *) dst->data + i1*( dst->nb[1])))[k]; ++ const float v = GGML_FP16_TO_FP32(x); ++ GGML_UNUSED(v); ++ assert(!isnan(v)); ++ assert(!isinf(v)); ++ } ++#endif // NDEBUG ++ } ++} ++ ++static void ggml_compute_forward_sigmoid_glu( ++ const ggml_compute_params * params, ++ ggml_tensor * dst) { ++ ++ const ggml_tensor * src0 = dst->src[0]; ++ ++ switch (src0->type) { ++ case GGML_TYPE_F32: ++ { ++ ggml_compute_forward_sigmoid_glu_f32(params, dst); ++ } break; ++ case GGML_TYPE_F16: ++ { ++ ggml_compute_forward_sigmoid_glu_f16(params, dst); ++ } break; ++ default: ++ { ++ GGML_ABORT("fatal error"); ++ } ++ } ++} ++ + // ggml_compute_forward_geglu + + static void ggml_compute_forward_geglu_f32( +@@ -10273,6 +10416,10 @@ void ggml_compute_forward_glu( + { + ggml_compute_forward_swiglu_clamp(params, dst); + } break; ++ case GGML_GLU_OP_SIGMOID: ++ { ++ ggml_compute_forward_sigmoid_glu(params, dst); ++ } break; + default: + { + GGML_ABORT("fatal error"); +diff --git a/ggml/src/ggml-cpu/vec.h b/ggml/src/ggml-cpu/vec.h +index 5de9cb5b7e0969bc10a99e40675ed35b0561637b..2a380b7d7345c79780c63ad708f1815f8ccf313c 100644 +--- a/ggml/src/ggml-cpu/vec.h ++++ b/ggml/src/ggml-cpu/vec.h +@@ -1411,6 +1411,19 @@ inline static void ggml_vec_reglu_f16 (const int n, ggml_fp16_t * y, const ggml_ + } + } + ++inline static void ggml_vec_sigmoid_glu_f32(const int n, float * y, const float * x, const float * g) { ++ for (int i = 0; i < n; ++i) { ++ y[i] = g[i] / (1.0f + expf(-x[i])); ++ } ++} ++ ++inline static void ggml_vec_sigmoid_glu_f16(const int n, ggml_fp16_t * y, const ggml_fp16_t * x, const ggml_fp16_t * g) { ++ for (int i = 0; i < n; ++i) { ++ const float v = GGML_CPU_FP16_TO_FP32(x[i]); ++ y[i] = GGML_CPU_FP32_TO_FP16(GGML_CPU_FP16_TO_FP32(g[i]) / (1.0f + expf(-v))); ++ } ++} ++ + #ifdef GGML_GELU_FP16 + inline static void ggml_vec_geglu_f32(const int n, float * y, const float * x, const float * g) { + uint16_t t; +diff --git a/ggml/src/ggml-cuda/ggml-cuda.cu b/ggml/src/ggml-cuda/ggml-cuda.cu +index 4c0c0633dd66cc6f717d69cd9f74a1f5d0acc404..29bddaff6da1ee9f4f6aec73a89b4078641845ee 100644 +--- a/ggml/src/ggml-cuda/ggml-cuda.cu ++++ b/ggml/src/ggml-cuda/ggml-cuda.cu +@@ -2247,6 +2247,9 @@ static bool ggml_cuda_compute_forward(ggml_backend_cuda_context & ctx, struct gg + case GGML_GLU_OP_SWIGLU_CLAMP: + ggml_cuda_op_swiglu_clamp(ctx, dst); + break; ++ case GGML_GLU_OP_SIGMOID: ++ ggml_cuda_op_sigmoid_glu(ctx, dst); ++ break; + default: + return false; + } +@@ -3623,6 +3626,111 @@ static int ggml_cuda_try_fuse(ggml_backend_cuda_context * cuda_ctx, ggml_cgraph + } + } + ++ const std::initializer_list batch_norm_ops = { ++ GGML_OP_SUB, GGML_OP_ADD, GGML_OP_SQRT, GGML_OP_DIV, GGML_OP_MUL, GGML_OP_ADD ++ }; ++ const bool batch_norm_ops_match = ++ i + (int) batch_norm_ops.size() <= cgraph->n_nodes && ++ std::equal( ++ batch_norm_ops.begin(), batch_norm_ops.end(), cgraph->nodes + i, ++ [](ggml_op op, const ggml_tensor * tensor) { return op == tensor->op; }); ++ const bool batch_norm_subgraph = ++ batch_norm_ops_match && ++ ggml_can_fuse_subgraph(cgraph, i, batch_norm_ops, { i + 5 }); ++ if (batch_norm_subgraph) { ++ ggml_tensor * sub = cgraph->nodes[i]; ++ ggml_tensor * var_add = cgraph->nodes[i + 1]; ++ ggml_tensor * sqrt = cgraph->nodes[i + 2]; ++ ggml_tensor * div = cgraph->nodes[i + 3]; ++ ggml_tensor * mul = cgraph->nodes[i + 4]; ++ ggml_tensor * out = cgraph->nodes[i + 5]; ++ ++ const ggml_tensor * input = sub->src[0]; ++ const ggml_tensor * mean = sub->src[1]; ++ const ggml_tensor * variance = nullptr; ++ const ggml_tensor * epsilon = nullptr; ++ if (ggml_nelements(var_add->src[0]) == 1) { ++ epsilon = var_add->src[0]; ++ variance = var_add->src[1]; ++ } else if (ggml_nelements(var_add->src[1]) == 1) { ++ variance = var_add->src[0]; ++ epsilon = var_add->src[1]; ++ } ++ ++ const ggml_tensor * weight = ++ mul->src[0] == div ? mul->src[1] : ++ mul->src[1] == div ? mul->src[0] : nullptr; ++ const ggml_tensor * bias = ++ out->src[0] == mul ? out->src[1] : ++ out->src[1] == mul ? out->src[0] : nullptr; ++ ++ const int64_t channels = input->ne[1]; ++ const bool graph_ok = ++ sqrt->src[0] == var_add && ++ div->src[0] == sub && div->src[1] == sqrt; ++ const bool params_ok = ++ variance != nullptr && epsilon != nullptr && weight != nullptr && bias != nullptr && ++ ggml_nelements(mean) == channels && ++ ggml_nelements(variance) == channels && ++ ggml_nelements(weight) == channels && ++ ggml_nelements(bias) == channels; ++ const bool types_ok = ++ input->type == GGML_TYPE_F32 && mean->type == GGML_TYPE_F32 && ++ variance != nullptr && variance->type == GGML_TYPE_F32 && ++ epsilon != nullptr && epsilon->type == GGML_TYPE_F32 && ++ weight != nullptr && weight->type == GGML_TYPE_F32 && ++ bias != nullptr && bias->type == GGML_TYPE_F32 && ++ sub->type == GGML_TYPE_F32 && var_add->type == GGML_TYPE_F32 && ++ sqrt->type == GGML_TYPE_F32 && div->type == GGML_TYPE_F32 && ++ mul->type == GGML_TYPE_F32 && out->type == GGML_TYPE_F32; ++ const bool layout_ok = ++ ggml_is_contiguous(input) && ggml_is_contiguous(mean) && ++ variance != nullptr && ggml_is_contiguous(variance) && ++ epsilon != nullptr && ggml_is_contiguous(epsilon) && ++ weight != nullptr && ggml_is_contiguous(weight) && ++ bias != nullptr && ggml_is_contiguous(bias) && ++ ggml_is_contiguous(out) && ggml_are_same_shape(input, out); ++ const int out_node = i + 5; ++ ++ const bool common_eligible = ++ graph_ok && params_ok && types_ok && layout_ok; ++ if (common_eligible && ++ ggml_can_fuse_subgraph( ++ cgraph, i, ++ { GGML_OP_SUB, GGML_OP_ADD, GGML_OP_SQRT, GGML_OP_DIV, GGML_OP_MUL, ++ GGML_OP_ADD, GGML_OP_PERMUTE, GGML_OP_CONT, GGML_OP_UNARY }, ++ { i + 8 })) { ++ ggml_tensor * permute = cgraph->nodes[i + 6]; ++ ggml_tensor * cont = cgraph->nodes[i + 7]; ++ ggml_tensor * silu = cgraph->nodes[i + 8]; ++ const bool transpose_ok = ++ permute->src[0] == out && cont->src[0] == permute && silu->src[0] == cont && ++ ggml_get_unary_op(silu) == GGML_UNARY_OP_SILU && ++ permute->type == GGML_TYPE_F32 && cont->type == GGML_TYPE_F32 && ++ silu->type == GGML_TYPE_F32 && ++ permute->ne[0] == input->ne[1] && permute->ne[1] == input->ne[0] && ++ permute->ne[2] == input->ne[2] && permute->ne[3] == input->ne[3] && ++ permute->nb[0] == out->nb[1] && permute->nb[1] == out->nb[0] && ++ permute->nb[2] == out->nb[2] && permute->nb[3] == out->nb[3] && ++ ggml_is_contiguous(cont) && ggml_is_contiguous(silu) && ++ ggml_are_same_shape(cont, silu); ++ const int transpose_out_node = i + 8; ++ if (transpose_ok && ++ ggml_cuda_check_fusion_memory_ranges( ++ cgraph, i, 9, &transpose_out_node, 1)) { ++ ggml_cuda_op_batch_norm_silu_transpose_fused( ++ *cuda_ctx, input, mean, variance, epsilon, weight, bias, silu); ++ return 8; ++ } ++ } ++ if (common_eligible && ++ ggml_cuda_check_fusion_memory_ranges(cgraph, i, 6, &out_node, 1)) { ++ ggml_cuda_op_batch_norm_fused( ++ *cuda_ctx, input, mean, variance, epsilon, weight, bias, out); ++ return 5; ++ } ++ } ++ + //topk-moe + if (cgraph->nodes[i]->op == GGML_OP_UNARY || cgraph->nodes[i]->op == GGML_OP_SOFT_MAX || + cgraph->nodes[i]->op == GGML_OP_ARGSORT) { +@@ -3756,6 +3864,31 @@ static int ggml_cuda_try_fuse(ggml_backend_cuda_context * cuda_ctx, ggml_cgraph + } + } + ++ // Macaron feed-forward residual: residual + scale * ff. Both tensors are ++ // contiguous F32 and have identical shapes, so materializing the scaled ++ // activation only creates an avoidable full-tensor round trip. ++ if (ggml_can_fuse(cgraph, i, { GGML_OP_SCALE, GGML_OP_ADD })) { ++ const ggml_tensor * scale_node = cgraph->nodes[i]; ++ ggml_tensor * add_node = cgraph->nodes[i + 1]; ++ const ggml_tensor * residual = ++ add_node->src[0] == scale_node ? add_node->src[1] : ++ add_node->src[1] == scale_node ? add_node->src[0] : nullptr; ++ ++ const float bias = ggml_get_op_params_f32(scale_node, 1); ++ const bool eligible = ++ residual != nullptr && bias == 0.0f && ++ scale_node->src[0]->type == GGML_TYPE_F32 && ++ residual->type == GGML_TYPE_F32 && add_node->type == GGML_TYPE_F32 && ++ ggml_are_same_shape(scale_node->src[0], residual) && ++ ggml_are_same_shape(scale_node->src[0], add_node) && ++ ggml_is_contiguous(scale_node->src[0]) && ++ ggml_is_contiguous(residual) && ggml_is_contiguous(add_node); ++ if (eligible) { ++ ggml_cuda_op_scale_add(*cuda_ctx, scale_node, residual, add_node); ++ return 1; ++ } ++ } ++ + // multi-(add or mul) + if (node->op == GGML_OP_ADD || node->op == GGML_OP_MUL) { + int n_fuse = 0; +@@ -4231,21 +4364,23 @@ static int ggml_cuda_try_fuse(ggml_backend_cuda_context * cuda_ctx, ggml_cgraph + return fused_node_count - 1; + } + +- // A BF16 projection immediately following SiLU would otherwise launch an +- // F32 SiLU kernel and then a separate F32-to-BF16 input conversion. Emit +- // the same rounded BF16 activation directly in one pass. ++ // A 16-bit projection immediately following SiLU would otherwise launch an ++ // F32 SiLU kernel and then a separate input conversion. + if (ggml_can_fuse(cgraph, i, { GGML_OP_UNARY, GGML_OP_CPY })) { + const ggml_tensor * silu_node = cgraph->nodes[i]; + ggml_tensor * cast_node = cgraph->nodes[i + 1]; ++ const bool cast_supported = ++ cast_node->type == GGML_TYPE_F16 || ++ (cast_node->type == GGML_TYPE_BF16 && native_bf16); + const bool eligible = +- native_bf16 && ggml_get_unary_op(silu_node) == GGML_UNARY_OP_SILU && ++ cast_supported && ggml_get_unary_op(silu_node) == GGML_UNARY_OP_SILU && + cast_node->src[0] == silu_node && + silu_node->src[0]->type == GGML_TYPE_F32 && +- silu_node->type == GGML_TYPE_F32 && cast_node->type == GGML_TYPE_BF16 && ++ silu_node->type == GGML_TYPE_F32 && + ggml_are_same_shape(silu_node->src[0], cast_node) && + ggml_is_contiguous(silu_node->src[0]) && ggml_is_contiguous(cast_node); + if (eligible) { +- ggml_cuda_op_silu_f32_to_bf16(*cuda_ctx, silu_node, cast_node); ++ ggml_cuda_op_silu_f32_to_16(*cuda_ctx, silu_node, cast_node); + return 1; + } + } +@@ -4386,6 +4521,38 @@ static int ggml_cuda_try_fuse(ggml_backend_cuda_context * cuda_ctx, ggml_cgraph + return 2; + } + ++ // LayerNorm affine followed by a 16-bit projection. Store the affine result ++ // directly in the projection's input precision instead of writing F32 and ++ // launching a second full-tensor conversion. ++ if (ggml_can_fuse(cgraph, i, ++ { GGML_OP_NORM, GGML_OP_MUL, GGML_OP_ADD, GGML_OP_CPY })) { ++ ggml_tensor * norm = cgraph->nodes[i]; ++ ggml_tensor * mul = cgraph->nodes[i + 1]; ++ ggml_tensor * add = cgraph->nodes[i + 2]; ++ ggml_tensor * cast = cgraph->nodes[i + 3]; ++ const ggml_tensor * gamma = ++ mul->src[0] == norm ? mul->src[1] : mul->src[0]; ++ const ggml_tensor * beta = ++ add->src[0] == mul ? add->src[1] : add->src[0]; ++ const bool cast_supported = ++ cast->type == GGML_TYPE_F16 || ++ (cast->type == GGML_TYPE_BF16 && native_bf16); ++ const bool eligible = ++ cast_supported && cast->src[0] == add && norm->src[0]->type == GGML_TYPE_F32 && ++ norm->type == GGML_TYPE_F32 && mul->type == GGML_TYPE_F32 && ++ add->type == GGML_TYPE_F32 && ++ gamma->type == GGML_TYPE_F32 && beta->type == GGML_TYPE_F32 && ++ ggml_nelements(gamma) == norm->ne[0] && ++ ggml_nelements(beta) == norm->ne[0] && ++ ggml_is_contiguous(gamma) && ggml_is_contiguous(beta) && ++ ggml_is_contiguous(mul) && ggml_is_contiguous(add) && ++ ggml_is_contiguous(cast) && ggml_are_same_shape(norm, cast); ++ if (eligible) { ++ ggml_cuda_op_norm_fused(*cuda_ctx, norm, mul, add, cast); ++ return 3; ++ } ++ } ++ + if (ggml_cuda_can_fuse(cgraph, i, { GGML_OP_RMS_NORM, GGML_OP_MUL, GGML_OP_ADD }, {})) { + ggml_cuda_op_rms_norm_fused_add(*cuda_ctx, node, cgraph->nodes[i + 1], cgraph->nodes[i + 2]); + return 2; +@@ -5436,6 +5603,7 @@ static bool ggml_backend_cuda_device_supports_op(ggml_backend_dev_t dev, const g + case GGML_GLU_OP_GEGLU_ERF: + case GGML_GLU_OP_GEGLU_QUICK: + case GGML_GLU_OP_SWIGLU_CLAMP: ++ case GGML_GLU_OP_SIGMOID: + return ggml_is_contiguous_1(op->src[0]); + default: + return false; +diff --git a/ggml/src/ggml-cuda/norm.cu b/ggml/src/ggml-cuda/norm.cu +index 8d2c036b6b414d8ba4d05b61382855b3da4a99b0..142650af5e8b4b7cd3c38235016bb4541c19fbcc 100644 +--- a/ggml/src/ggml-cuda/norm.cu ++++ b/ggml/src/ggml-cuda/norm.cu +@@ -1,4 +1,5 @@ + #include "norm.cuh" ++#include "unary.cuh" + #include + + template +@@ -43,9 +44,9 @@ static __global__ void norm_f32( + // to ne0-length vectors broadcast over rows/channels/samples (the standard + // LayerNorm affine), enforced by ggml_cuda_can_fuse — this keeps indexing + // trivial instead of replicating rms_norm's general broadcast machinery. +-template ++template + static __global__ void norm_mul_add_f32( +- const float * x, const float * mul, const float * add, float * dst, const int ncols, ++ const float * x, const float * mul, const float * add, T * dst, const int ncols, + const int64_t stride_row, const int64_t stride_channel, const int64_t stride_sample, const float eps) { + const int nrows = gridDim.x; + const int nchannels = gridDim.y; +@@ -78,10 +79,65 @@ static __global__ void norm_mul_add_f32( + if constexpr (do_add) { + v += add[col]; + } +- dst[col] = v; ++ dst[col] = (T) v; + } + } + ++static __global__ void batch_norm_f32( ++ const float * x, const float * mean, const float * variance, const float * epsilon, ++ const float * weight, const float * bias, float * dst, int64_t nelements, ++ int64_t ntime, int64_t nchannels) { ++ const int64_t i = int64_t(blockIdx.x) * blockDim.x + threadIdx.x; ++ if (i >= nelements) { ++ return; ++ } ++ ++ const int64_t channel = (i / ntime) % nchannels; ++ const float centered = __fsub_rn(x[i], mean[channel]); ++ const float denominator = sqrtf(__fadd_rn(variance[channel], epsilon[0])); ++ const float normalized = __fdiv_rn(centered, denominator); ++ dst[i] = __fadd_rn(__fmul_rn(normalized, weight[channel]), bias[channel]); ++} ++ ++static __global__ void batch_norm_silu_transpose_f32( ++ const float * x, const float * mean, const float * variance, const float * epsilon, ++ const float * weight, const float * bias, float * dst, int64_t nelements, ++ int64_t ntime, int64_t nchannels) { ++ const int64_t i = int64_t(blockIdx.x) * blockDim.x + threadIdx.x; ++ if (i >= nelements) { ++ return; ++ } ++ ++ const int64_t channel = (i / ntime) % nchannels; ++ const int64_t time = i % ntime; ++ const int64_t sample = i / (ntime * nchannels); ++ const float centered = __fsub_rn(x[i], mean[channel]); ++ const float denominator = sqrtf(__fadd_rn(variance[channel], epsilon[0])); ++ const float normalized = __fdiv_rn(centered, denominator); ++ const float affine = __fadd_rn(__fmul_rn(normalized, weight[channel]), bias[channel]); ++ dst[channel + nchannels * (time + ntime * sample)] = ggml_cuda_op_silu_single(affine); ++} ++ ++static void batch_norm_f32_cuda( ++ const float * x, const float * mean, const float * variance, const float * epsilon, ++ const float * weight, const float * bias, float * dst, int64_t nelements, ++ int64_t ntime, int64_t nchannels, cudaStream_t stream) { ++ constexpr int block_size = 256; ++ const int64_t block_count = (nelements + block_size - 1) / block_size; ++ batch_norm_f32<<>>( ++ x, mean, variance, epsilon, weight, bias, dst, nelements, ntime, nchannels); ++} ++ ++static void batch_norm_silu_transpose_f32_cuda( ++ const float * x, const float * mean, const float * variance, const float * epsilon, ++ const float * weight, const float * bias, float * dst, int64_t nelements, ++ int64_t ntime, int64_t nchannels, cudaStream_t stream) { ++ constexpr int block_size = 256; ++ const int64_t block_count = (nelements + block_size - 1) / block_size; ++ batch_norm_silu_transpose_f32<<>>( ++ x, mean, variance, epsilon, weight, bias, dst, nelements, ntime, nchannels); ++} ++ + template + static __global__ void group_norm_f32(const float * x, float * dst, const int group_size, const int ne_elements, const float eps) { + // blockIdx.x: num_groups idx +@@ -334,27 +390,50 @@ static void norm_f32_cuda( + } + } + +-static void norm_mul_add_f32_cuda( +- const float * x, const float * mul, const float * add, float * dst, const int ncols, const int nrows, ++template ++static void norm_mul_add_cuda( ++ const float * x, const float * mul, const float * add, T * dst, const int ncols, const int nrows, + const int nchannels, const int nsamples, const int64_t stride_row, const int64_t stride_channel, + const int64_t stride_sample, const float eps, cudaStream_t stream) { + const dim3 blocks_num(nrows, nchannels, nsamples); +- if (ncols < 1024) { ++ const int cc = ggml_cuda_info().devices[ggml_cuda_get_device()].cc; ++ if (ncols == 768 && WARP_SIZE == 32) { ++ constexpr int block_size = 384; ++ if (add) { ++ norm_mul_add_f32 ++ <<>>( ++ x, mul, add, dst, ncols, stride_row, stride_channel, stride_sample, eps); ++ } else { ++ norm_mul_add_f32 ++ <<>>( ++ x, mul, add, dst, ncols, stride_row, stride_channel, stride_sample, eps); ++ } ++ } else if (ncols < 1024) { + const dim3 block_dims(WARP_SIZE, 1, 1); + if (add) { +- norm_mul_add_f32<<>>( ++ norm_mul_add_f32<<>>( ++ x, mul, add, dst, ncols, stride_row, stride_channel, stride_sample, eps); ++ } else { ++ norm_mul_add_f32<<>>( ++ x, mul, add, dst, ncols, stride_row, stride_channel, stride_sample, eps); ++ } ++ } else if (ncols == 1024 && GGML_CUDA_CC_IS_NVIDIA(cc) && cc >= GGML_CUDA_CC_AMPERE) { ++ // Four elements per thread balances reduction work and occupancy. ++ const dim3 block_dims(256, 1, 1); ++ if (add) { ++ norm_mul_add_f32<256, true, T><<>>( + x, mul, add, dst, ncols, stride_row, stride_channel, stride_sample, eps); + } else { +- norm_mul_add_f32<<>>( ++ norm_mul_add_f32<256, false, T><<>>( + x, mul, add, dst, ncols, stride_row, stride_channel, stride_sample, eps); + } + } else { + const dim3 block_dims(1024, 1, 1); + if (add) { +- norm_mul_add_f32<1024, true><<>>( ++ norm_mul_add_f32<1024, true, T><<>>( + x, mul, add, dst, ncols, stride_row, stride_channel, stride_sample, eps); + } else { +- norm_mul_add_f32<1024, false><<>>( ++ norm_mul_add_f32<1024, false, T><<>>( + x, mul, add, dst, ncols, stride_row, stride_channel, stride_sample, eps); + } + } +@@ -531,7 +610,7 @@ void ggml_cuda_op_norm(ggml_backend_cuda_context & ctx, ggml_tensor * dst) { + // or add_tensor), mirroring ggml_cuda_op_rms_norm_fused(_add). Eligibility + // (row-vector operands, F32, contiguity) is enforced in ggml_cuda_can_fuse. + void ggml_cuda_op_norm_fused(ggml_backend_cuda_context & ctx, ggml_tensor * dst, ggml_tensor * mul_tensor, +- ggml_tensor * add_tensor) { ++ ggml_tensor * add_tensor, ggml_tensor * cast_dst) { + const ggml_tensor * norm_src = (ggml_tensor *) dst->src[0]; + float eps = 0.0f; + memcpy(&eps, dst->op_params, sizeof(float)); +@@ -575,8 +654,68 @@ void ggml_cuda_op_norm_fused(ggml_backend_cuda_context & ctx, ggml_tensor * dst, + const int64_t s02 = norm_src->nb[2] / ts0; + const int64_t s03 = norm_src->nb[3] / ts0; + +- norm_mul_add_f32_cuda( +- src0_d, mul_d, add_d, dst_d, ne00, ne01, ne02, ne03, s01, s02, s03, eps, ctx.stream()); ++ if (cast_dst != nullptr) { ++ GGML_ASSERT(cast_dst->type == GGML_TYPE_BF16 || cast_dst->type == GGML_TYPE_F16); ++ GGML_ASSERT(ggml_is_contiguous(cast_dst)); ++ GGML_ASSERT(ggml_are_same_shape(dst, cast_dst)); ++ if (cast_dst->type == GGML_TYPE_BF16) { ++ norm_mul_add_cuda( ++ src0_d, mul_d, add_d, (nv_bfloat16 *) cast_dst->data, ++ ne00, ne01, ne02, ne03, s01, s02, s03, eps, ctx.stream()); ++ } else { ++ norm_mul_add_cuda( ++ src0_d, mul_d, add_d, (half *) cast_dst->data, ++ ne00, ne01, ne02, ne03, s01, s02, s03, eps, ctx.stream()); ++ } ++ } else { ++ norm_mul_add_cuda( ++ src0_d, mul_d, add_d, dst_d, ++ ne00, ne01, ne02, ne03, s01, s02, s03, eps, ctx.stream()); ++ } ++} ++ ++void ggml_cuda_op_batch_norm_fused(ggml_backend_cuda_context & ctx, ++ const ggml_tensor * input, ++ const ggml_tensor * mean, ++ const ggml_tensor * variance, ++ const ggml_tensor * epsilon, ++ const ggml_tensor * weight, ++ const ggml_tensor * bias, ++ ggml_tensor * dst) { ++ batch_norm_f32_cuda( ++ (const float *) input->data, ++ (const float *) mean->data, ++ (const float *) variance->data, ++ (const float *) epsilon->data, ++ (const float *) weight->data, ++ (const float *) bias->data, ++ (float *) dst->data, ++ ggml_nelements(input), ++ input->ne[0], ++ input->ne[1], ++ ctx.stream()); ++} ++ ++void ggml_cuda_op_batch_norm_silu_transpose_fused(ggml_backend_cuda_context & ctx, ++ const ggml_tensor * input, ++ const ggml_tensor * mean, ++ const ggml_tensor * variance, ++ const ggml_tensor * epsilon, ++ const ggml_tensor * weight, ++ const ggml_tensor * bias, ++ ggml_tensor * dst) { ++ batch_norm_silu_transpose_f32_cuda( ++ (const float *) input->data, ++ (const float *) mean->data, ++ (const float *) variance->data, ++ (const float *) epsilon->data, ++ (const float *) weight->data, ++ (const float *) bias->data, ++ (float *) dst->data, ++ ggml_nelements(input), ++ input->ne[0], ++ input->ne[1], ++ ctx.stream()); + } + + void ggml_cuda_op_group_norm(ggml_backend_cuda_context & ctx, ggml_tensor * dst) { +diff --git a/ggml/src/ggml-cuda/norm.cuh b/ggml/src/ggml-cuda/norm.cuh +index 20ecaa2d9e7536c3f19d111ade42fa1f2497b38d..f580ae90bf032cfdddb13ba999265be80ca8bb0e 100644 +--- a/ggml/src/ggml-cuda/norm.cuh ++++ b/ggml/src/ggml-cuda/norm.cuh +@@ -5,7 +5,25 @@ void ggml_cuda_op_norm(ggml_backend_cuda_context & ctx, ggml_tensor * dst); + // Fused LayerNorm + row-vector mul (gamma) + optional row-vector add (beta). + // add_tensor may be nullptr (norm+mul only). + void ggml_cuda_op_norm_fused(ggml_backend_cuda_context & ctx, ggml_tensor * dst, ggml_tensor * mul_tensor, +- ggml_tensor * add_tensor); ++ ggml_tensor * add_tensor, ggml_tensor * cast_dst = nullptr); ++ ++void ggml_cuda_op_batch_norm_fused(ggml_backend_cuda_context & ctx, ++ const ggml_tensor * input, ++ const ggml_tensor * mean, ++ const ggml_tensor * variance, ++ const ggml_tensor * epsilon, ++ const ggml_tensor * weight, ++ const ggml_tensor * bias, ++ ggml_tensor * dst); ++ ++void ggml_cuda_op_batch_norm_silu_transpose_fused(ggml_backend_cuda_context & ctx, ++ const ggml_tensor * input, ++ const ggml_tensor * mean, ++ const ggml_tensor * variance, ++ const ggml_tensor * epsilon, ++ const ggml_tensor * weight, ++ const ggml_tensor * bias, ++ ggml_tensor * dst); + + void ggml_cuda_op_group_norm(ggml_backend_cuda_context & ctx, ggml_tensor * dst); + +diff --git a/ggml/src/ggml-cuda/scale.cu b/ggml/src/ggml-cuda/scale.cu +index 7b2e59a4383f2bb2c3c63f8f132193cbfb9d8122..d9f1af8c3b6ffba577fc24ac9a0f30253961ad7b 100644 +--- a/ggml/src/ggml-cuda/scale.cu ++++ b/ggml/src/ggml-cuda/scale.cu +@@ -35,3 +35,39 @@ void ggml_cuda_op_scale(ggml_backend_cuda_context & ctx, ggml_tensor * dst) { + + scale_f32_cuda(src0_d, dst_d, scale, bias, ggml_nelements(src0), stream); + } ++ ++static __global__ void scale_add_f32( ++ const float * __restrict__ x, const float * __restrict__ residual, ++ float * __restrict__ dst, const float scale, const int64_t nelements) { ++ const int64_t tid = (int64_t) blockIdx.x * blockDim.x + threadIdx.x; ++ const int64_t stride = (int64_t) blockDim.x * gridDim.x; ++ ++ for (int64_t i = tid; i < nelements; i += stride) { ++ dst[i] = residual[i] + scale * x[i]; ++ } ++} ++ ++void ggml_cuda_op_scale_add( ++ ggml_backend_cuda_context & ctx, const ggml_tensor * scale_node, ++ const ggml_tensor * residual, ggml_tensor * dst) { ++ const ggml_tensor * x = scale_node->src[0]; ++ ++ GGML_ASSERT(x->type == GGML_TYPE_F32); ++ GGML_ASSERT(residual->type == GGML_TYPE_F32); ++ GGML_ASSERT(dst->type == GGML_TYPE_F32); ++ GGML_ASSERT(ggml_is_contiguous(x)); ++ GGML_ASSERT(ggml_is_contiguous(residual)); ++ GGML_ASSERT(ggml_is_contiguous(dst)); ++ GGML_ASSERT(ggml_are_same_shape(x, residual)); ++ GGML_ASSERT(ggml_are_same_shape(x, dst)); ++ ++ const float scale = ggml_get_op_params_f32(scale_node, 0); ++ const float bias = ggml_get_op_params_f32(scale_node, 1); ++ GGML_ASSERT(bias == 0.0f); ++ ++ const int64_t nelements = ggml_nelements(x); ++ const int64_t num_blocks = (nelements + CUDA_SCALE_BLOCK_SIZE - 1) / CUDA_SCALE_BLOCK_SIZE; ++ scale_add_f32<<>>( ++ (const float *) x->data, (const float *) residual->data, ++ (float *) dst->data, scale, nelements); ++} +diff --git a/ggml/src/ggml-cuda/scale.cuh b/ggml/src/ggml-cuda/scale.cuh +index 8ff75c8298b0207e81fff8736b39c0260d246ee1..16fe10ab7359f843b7946a2bdada339d4d540fe9 100644 +--- a/ggml/src/ggml-cuda/scale.cuh ++++ b/ggml/src/ggml-cuda/scale.cuh +@@ -3,3 +3,7 @@ + #define CUDA_SCALE_BLOCK_SIZE 256 + + void ggml_cuda_op_scale(ggml_backend_cuda_context & ctx, ggml_tensor * dst); ++ ++void ggml_cuda_op_scale_add( ++ ggml_backend_cuda_context & ctx, const ggml_tensor * scale_node, ++ const ggml_tensor * residual, ggml_tensor * dst); +diff --git a/ggml/src/ggml-cuda/unary.cu b/ggml/src/ggml-cuda/unary.cu +index 0b908345a1a5ba0424979d0dde4c0b94751de898..4ed4dc65e937d3bb8edba77f7fa25d652cf036e7 100644 +--- a/ggml/src/ggml-cuda/unary.cu ++++ b/ggml/src/ggml-cuda/unary.cu +@@ -186,11 +186,20 @@ void ggml_cuda_op_silu(ggml_backend_cuda_context & ctx, ggml_tensor * dst) { + ggml_cuda_op_unary(ctx, dst); + } + ++static __global__ void silu_f32_to_f16( ++ const float * __restrict__ x, half * __restrict__ dst, ++ const int64_t nelements) { ++ const int64_t i = (int64_t) blockIdx.x * blockDim.x + threadIdx.x; ++ if (i < nelements) { ++ dst[i] = __float2half(op_silu(x[i])); ++ } ++} ++ + static __global__ void silu_f32_to_bf16( + const float * __restrict__ x, nv_bfloat16 * __restrict__ dst, + const int64_t nelements) { + #if defined(__CUDA_ARCH__) && __CUDA_ARCH__ >= GGML_CUDA_CC_AMPERE +- const int64_t i = (int64_t) blockIdx.x * blockDim.x + threadIdx.x; ++ const int64_t i = int64_t(blockIdx.x) * blockDim.x + threadIdx.x; + if (i < nelements) { + dst[i] = __float2bfloat16(op_silu(x[i])); + } +@@ -200,21 +209,26 @@ static __global__ void silu_f32_to_bf16( + #endif + } + +-void ggml_cuda_op_silu_f32_to_bf16( ++void ggml_cuda_op_silu_f32_to_16( + ggml_backend_cuda_context & ctx, const ggml_tensor * silu_node, + ggml_tensor * dst) { + const ggml_tensor * src = silu_node->src[0]; + GGML_ASSERT(src->type == GGML_TYPE_F32); + GGML_ASSERT(silu_node->type == GGML_TYPE_F32); +- GGML_ASSERT(dst->type == GGML_TYPE_BF16); ++ GGML_ASSERT(dst->type == GGML_TYPE_BF16 || dst->type == GGML_TYPE_F16); + GGML_ASSERT(ggml_are_same_shape(src, dst)); + GGML_ASSERT(ggml_is_contiguous(src)); + GGML_ASSERT(ggml_is_contiguous(dst)); + + const int64_t nelements = ggml_nelements(src); + const int64_t num_blocks = (nelements + CUDA_SILU_BLOCK_SIZE - 1) / CUDA_SILU_BLOCK_SIZE; +- silu_f32_to_bf16<<>>( +- (const float *) src->data, (nv_bfloat16 *) dst->data, nelements); ++ if (dst->type == GGML_TYPE_BF16) { ++ silu_f32_to_bf16<<>>( ++ (const float *) src->data, (nv_bfloat16 *) dst->data, nelements); ++ } else { ++ silu_f32_to_f16<<>>( ++ (const float *) src->data, (half *) dst->data, nelements); ++ } + } + + void ggml_cuda_op_tanh(ggml_backend_cuda_context & ctx, ggml_tensor * dst) { +@@ -380,6 +394,10 @@ void ggml_cuda_op_swiglu(ggml_backend_cuda_context & ctx, ggml_tensor * dst) { + ggml_cuda_op_unary_gated(ctx, dst); + } + ++void ggml_cuda_op_sigmoid_glu(ggml_backend_cuda_context & ctx, ggml_tensor * dst) { ++ ggml_cuda_op_unary_gated(ctx, dst); ++} ++ + void ggml_cuda_op_geglu_erf(ggml_backend_cuda_context & ctx, ggml_tensor * dst) { + ggml_cuda_op_unary_gated(ctx, dst); + } +diff --git a/ggml/src/ggml-cuda/unary.cuh b/ggml/src/ggml-cuda/unary.cuh +index 5c39ccb2a05d57a82b70ffb445c0e7977e5f2f47..7b9148e42741d11a05fe452d0df55089d3a5b4f0 100644 +--- a/ggml/src/ggml-cuda/unary.cuh ++++ b/ggml/src/ggml-cuda/unary.cuh +@@ -31,7 +31,7 @@ void ggml_cuda_op_gelu(ggml_backend_cuda_context & ctx, ggml_tensor * dst); + + void ggml_cuda_op_silu(ggml_backend_cuda_context & ctx, ggml_tensor * dst); + +-void ggml_cuda_op_silu_f32_to_bf16( ++void ggml_cuda_op_silu_f32_to_16( + ggml_backend_cuda_context & ctx, const ggml_tensor * silu_node, ggml_tensor * dst); + + void ggml_cuda_op_silu_back(ggml_backend_cuda_context & ctx, ggml_tensor * dst); +@@ -84,6 +84,8 @@ void ggml_cuda_op_geglu(ggml_backend_cuda_context & ctx, ggml_tensor * dst); + + void ggml_cuda_op_swiglu(ggml_backend_cuda_context & ctx, ggml_tensor * dst); + ++void ggml_cuda_op_sigmoid_glu(ggml_backend_cuda_context & ctx, ggml_tensor * dst); ++ + void ggml_cuda_op_swiglu_oai(ggml_backend_cuda_context & ctx, ggml_tensor * dst); + + void ggml_cuda_op_swiglu_clamp(ggml_backend_cuda_context & ctx, ggml_tensor * dst); +diff --git a/ggml/src/ggml.c b/ggml/src/ggml.c +index 81ecb230f6f88ff0f58292f6b4283e552f7ada39..3aaa06603030c43ee8dfe72a2170272327e3f708 100644 +--- a/ggml/src/ggml.c ++++ b/ggml/src/ggml.c +@@ -1259,9 +1259,10 @@ static const char * GGML_GLU_OP_NAME[GGML_GLU_OP_COUNT] = { + "GEGLU_ERF", + "GEGLU_QUICK", + "SWIGLU_CLAMP", ++ "SIGMOID_GLU", + }; + +-static_assert(GGML_GLU_OP_COUNT == 7, "GGML_GLU_OP_COUNT != 7"); ++static_assert(GGML_GLU_OP_COUNT == 8, "GGML_GLU_OP_COUNT != 8"); + + static_assert(sizeof(struct ggml_object)%GGML_MEM_ALIGN == 0, "ggml_object size must be a multiple of GGML_MEM_ALIGN"); + static_assert(sizeof(struct ggml_tensor)%GGML_MEM_ALIGN == 0, "ggml_tensor size must be a multiple of GGML_MEM_ALIGN"); diff --git a/ggml-patches/0025-cuda-conv1d-fused.patch b/patches/ggml-nanocodec-convolution-kernels.patch similarity index 58% rename from ggml-patches/0025-cuda-conv1d-fused.patch rename to patches/ggml-nanocodec-convolution-kernels.patch index 615ce08..4c8c73a 100644 --- a/ggml-patches/0025-cuda-conv1d-fused.patch +++ b/patches/ggml-nanocodec-convolution-kernels.patch @@ -1,8 +1,35 @@ -diff --git a/include/ggml.h b/include/ggml.h -index d403756e..e40cf62c 100644 ---- a/include/ggml.h -+++ b/include/ggml.h -@@ -584,6 +584,7 @@ extern "C" { +From: Prabhsimran Singh +Subject: ggml: NanoCodec convolution kernels + +- Adds ggml_conv1d_fused and ggml_conv1d_fused_grouped: a causal 1D + convolution over an F32 streaming-cache prefix plus the current input, with + the half-snake or leaky-ReLU activation applied on load. The CUDA kernel is + an implicit GEMM on mma.sync tensor-core fragments (F16 operands, F32 + accumulation, split-K, fused bias and residual epilogue) over F16 weights + pre-packed as [K][cout][cin]. The grouped form runs several kernel-size + branches in one launch; for problems with several output-channel tiles it + activates and packs the input once into an F16 buffer instead of once per + tile (bit-identical, about 2x faster NanoCodec decoding). CUDA only: + supported for NVIDIA devices with sm_80+ device code in the build; the CPU + backend reports it unsupported and other backends reject it, so callers keep + their unfused path. +- conv_transpose_1d reads only the two non-zero input channels of NanoCodec's + grouped upsampling weights (selected by the dec.up.* tensor name). +- Fuses NanoCodec's half-snake activation (snake on one half of the channels, + leaky ReLU on the other, then concat). Fusion is skipped unless the output is + disjoint from both inputs or aliases them exactly, since the graph allocator + may otherwise reuse the parent activation's memory for the concat. + +Upstream: not planned (NanoCodec-specific). + +Co-authored-by: anand-nv <105917641+anand-nv@users.noreply.github.com> +Co-authored-by: Ryan Leary + +diff --git a/ggml/include/ggml.h b/ggml/include/ggml.h +index 6197891dbb20ad7bfd322abe731cf6fed5162eb3..21015f8a9fac02038874027836b9cb02827989e9 100644 +--- a/ggml/include/ggml.h ++++ b/ggml/include/ggml.h +@@ -602,6 +602,7 @@ extern "C" { GGML_OP_GLU, GGML_OP_FUSED_ATTN, @@ -10,7 +37,7 @@ index d403756e..e40cf62c 100644 GGML_OP_COUNT, }; -@@ -2498,6 +2499,54 @@ extern "C" { +@@ -2594,6 +2595,54 @@ extern "C" { float scale, bool merge_heads); @@ -65,11 +92,11 @@ index d403756e..e40cf62c 100644 // TODO: needs to be adapted to ggml_flash_attn_ext GGML_API struct ggml_tensor * ggml_flash_attn_back( struct ggml_context * ctx, -diff --git a/src/ggml-cpu/ggml-cpu.c b/src/ggml-cpu/ggml-cpu.c -index 43de6346..21af9c52 100644 ---- a/src/ggml-cpu/ggml-cpu.c -+++ b/src/ggml-cpu/ggml-cpu.c -@@ -1992,6 +1992,10 @@ static void ggml_compute_forward(struct ggml_compute_params * params, struct ggm +diff --git a/ggml/src/ggml-cpu/ggml-cpu.c b/ggml/src/ggml-cpu/ggml-cpu.c +index 01893e3bd07a2b98b110865a364216b56cd652d7..2ff83dfeeb81b633c2fd0fc5afee81c1fec1efa5 100644 +--- a/ggml/src/ggml-cpu/ggml-cpu.c ++++ b/ggml/src/ggml-cpu/ggml-cpu.c +@@ -2038,6 +2038,10 @@ static void ggml_compute_forward(struct ggml_compute_params * params, struct ggm { GGML_ABORT("FUSED_ATTN has no CPU path (CUDA-only)"); } break; @@ -80,11 +107,11 @@ index 43de6346..21af9c52 100644 case GGML_OP_FLASH_ATTN_BACK: { int32_t t = ggml_get_op_params_i32(tensor, 0); -diff --git a/src/ggml-cpu/ggml-cpu.cpp b/src/ggml-cpu/ggml-cpu.cpp -index 0c570639..f20b8bcb 100644 ---- a/src/ggml-cpu/ggml-cpu.cpp -+++ b/src/ggml-cpu/ggml-cpu.cpp -@@ -439,6 +439,7 @@ static bool ggml_backend_cpu_device_supports_op(ggml_backend_dev_t dev, const st +diff --git a/ggml/src/ggml-cpu/ggml-cpu.cpp b/ggml/src/ggml-cpu/ggml-cpu.cpp +index 38b5dbf1205c43d33de6d225f6da8c3fc915e2c8..194c87e2f5306b08c5431f2f1216e1d09100b0fc 100644 +--- a/ggml/src/ggml-cpu/ggml-cpu.cpp ++++ b/ggml/src/ggml-cpu/ggml-cpu.cpp +@@ -440,6 +440,7 @@ static bool ggml_backend_cpu_device_supports_op(ggml_backend_dev_t dev, const st } switch (op->op) { @@ -92,12 +119,102 @@ index 0c570639..f20b8bcb 100644 case GGML_OP_FUSED_ATTN: return false; // CUDA-only; CPU falls back to the unfused graph case GGML_OP_CPY: -diff --git a/src/ggml-cuda/conv1d-fused.cu b/src/ggml-cuda/conv1d-fused.cu +diff --git a/ggml/src/ggml-cuda/conv-transpose-1d.cu b/ggml/src/ggml-cuda/conv-transpose-1d.cu +index ebf2aa8045eab22120d1fca4a6357f6188680d88..69ef05593937693692ea6942d8b767ef44827dda 100644 +--- a/ggml/src/ggml-cuda/conv-transpose-1d.cu ++++ b/ggml/src/ggml-cuda/conv-transpose-1d.cu +@@ -1,5 +1,7 @@ + #include "conv-transpose-1d.cuh" + ++#include ++ + static __global__ void conv_transpose_1d_kernel( + const int s0, const int p0, const int d0, const int output_size, + const int src0_ne0, const int src0_ne1, const int src0_ne2, const int src0_ne3, +@@ -56,6 +58,62 @@ static void conv_transpose_1d_f32_f32_cuda( + src0,src1, dst); + } + ++// NanoCodec upsampling convolutions are grouped (2 input channels per output ++// channel) but converted as dense, mostly-zero weights. This kernel reads only ++// the two non-zero input channels of each output channel. ++static __global__ void conv_transpose_1d_grouped2_kernel( ++ const int s0, const int output_size, ++ const int src0_ne0, const int src0_ne1, ++ const int src1_ne0, const int src1_ne1, ++ const int dst_ne0, const int dst_ne1, ++ const float * src0, const float * src1, float * dst) { ++ const int global_index = threadIdx.x + blockIdx.x * blockDim.x; ++ if (global_index >= output_size) { ++ return; ++ } ++ ++ const int out_t = global_index % dst_ne0; ++ const int out_c = (global_index / dst_ne0) % dst_ne1; ++ const int out_b = global_index / (dst_ne0 * dst_ne1); ++ ++ const int in_end = min(src1_ne0 - 1, out_t / s0); ++ const int in_start = max(0, (out_t - src0_ne0 + s0) / s0); ++ ++ const int c0 = out_c * 2; ++ const int c1 = c0 + 1; ++ const int input_offset0 = src1_ne0 * (c0 + src1_ne1 * out_b); ++ const int input_offset1 = src1_ne0 * (c1 + src1_ne1 * out_b); ++ const int kernel_offset0 = src0_ne0 * (out_c + src0_ne1 * c0); ++ const int kernel_offset1 = src0_ne0 * (out_c + src0_ne1 * c1); ++ ++ float accumulator = 0; ++ for (int i = in_start; i <= in_end; i++) { ++ const int weight_idx = out_t - i*s0; ++ accumulator += src0[kernel_offset0 + weight_idx] * src1[input_offset0 + i]; ++ accumulator += src0[kernel_offset1 + weight_idx] * src1[input_offset1 + i]; ++ } ++ dst[global_index] = accumulator; ++} ++ ++static bool conv_transpose_1d_use_nanocodec_grouped2( ++ const ggml_tensor * src0, const ggml_tensor * src1, const ggml_tensor * dst, const int s0) { ++ const bool shape_matches = ++ src0->ne[0] == 2*s0 && ++ src0->ne[2] == 2*src0->ne[1] && ++ src1->ne[1] == src0->ne[2] && ++ dst->ne[1] == src0->ne[1] && ++ src1->ne[2] == dst->ne[2] && ++ src1->ne[3] == dst->ne[3]; ++ ++ const char * name = src0->name; ++ const bool is_nanocodec_up_weight = ++ name != nullptr && ++ std::strncmp(name, "dec.up.", 7) == 0 && ++ std::strstr(name + 7, ".w") != nullptr; ++ ++ return shape_matches && is_nanocodec_up_weight; ++} ++ + void ggml_cuda_op_conv_transpose_1d(ggml_backend_cuda_context & ctx, ggml_tensor * dst) { + const ggml_tensor * src0 = dst->src[0]; + const float * src0_d = (const float *)src0->data; +@@ -80,6 +138,14 @@ void ggml_cuda_op_conv_transpose_1d(ggml_backend_cuda_context & ctx, ggml_tensor + + const int64_t output_size = ggml_nelements(dst); + ++ if (conv_transpose_1d_use_nanocodec_grouped2(src0, src1, dst, s0)) { ++ const int num_blocks = (output_size + CUDA_CONV_TRANPOSE_1D_BLOCK_SIZE - 1) / CUDA_CONV_TRANPOSE_1D_BLOCK_SIZE; ++ conv_transpose_1d_grouped2_kernel<<>>( ++ s0, output_size, src0->ne[0], src0->ne[1], src1->ne[0], src1->ne[1], dst->ne[0], dst->ne[1], ++ src0_d, src1_d, dst_d); ++ return; ++ } ++ + conv_transpose_1d_f32_f32_cuda(s0, p0, d0, output_size, + src0->ne[0], src0->ne[1], src0->ne[2], src0->ne[3], + src1->ne[0], src1->ne[1], src1->ne[2], src1->ne[3], +diff --git a/ggml/src/ggml-cuda/conv1d-fused.cu b/ggml/src/ggml-cuda/conv1d-fused.cu new file mode 100644 -index 00000000..336ba959 +index 0000000000000000000000000000000000000000..849f7d34f836d1c5b36d1a505663e3a7d5b86045 --- /dev/null -+++ b/src/ggml-cuda/conv1d-fused.cu -@@ -0,0 +1,483 @@ ++++ b/ggml/src/ggml-cuda/conv1d-fused.cu +@@ -0,0 +1,545 @@ +#include "conv1d-fused.cuh" + +#include @@ -130,6 +247,7 @@ index 00000000..336ba959 +constexpr int CF_STAGES = 2; + +struct conv1d_fused_params { ++ const half * activated; // optional [group][cache_len+T][cin_pad], prepared once + const float * x; + const float * cache; + const half * w; @@ -165,6 +283,29 @@ index 00000000..336ba959 + : "r"(a[0]), "r"(a[1]), "r"(a[2]), "r"(a[3]), "r"(b[0]), "r"(b[1])); +} + ++// Pack and activate once per element, instead of once per output-channel tile. ++__global__ void conv1d_activate_pack(conv1d_fused_params p, half * dst) { ++ const int ci = blockIdx.x * blockDim.x + threadIdx.x; ++ const int t = blockIdx.y; ++ const int grp = blockIdx.z; ++ if (ci >= p.cin_pad) return; ++ float v = 0.0f; ++ if (ci < p.cin) { ++ const int cb = (p.shared_input ? 0 : grp * p.cin) + ci; ++ v = t < p.cache_len ? p.cache[(size_t)cb * p.cs + t] ++ : p.x[(size_t)cb * p.xs + t - p.cache_len]; ++ if (p.alpha) { ++ if (ci < p.snake_ch) { ++ const float sn = __sinf(p.alpha[grp * p.snake_ch + ci] * v); ++ v = v + sn * sn * p.inv_b[grp * p.snake_ch + ci]; ++ } else { ++ v = v > 0.f ? v : v * p.slope; ++ } ++ } ++ } ++ dst[((size_t)grp * (p.cache_len + p.T) + t) * p.cin_pad + ci] = __float2half(v); ++} ++ +constexpr int CF_MAX_REGS = 168; +// __maxnreg__ is a CUDA 12.4+ toolkit macro; older toolkits and HIP get plain launch bounds. +#if defined(__maxnreg__) && !defined(GGML_USE_HIP) && !defined(GGML_USE_MUSA) @@ -173,7 +314,7 @@ index 00000000..336ba959 +#define CF_LAUNCH_BOUNDS __launch_bounds__(CF_THREADS) +#endif + -+template ++template +// 128 threads x <= 168 registers (21.5K) fit beside a resident MagpieTTS persistent kernel block +// (168 x 256 = 43K) in the 64K-register file, so codec blocks co-schedule instead of serializing. +__global__ void CF_LAUNCH_BOUNDS conv1d_fused_kernel(const conv1d_fused_params p) { @@ -318,12 +459,32 @@ index 00000000..336ba959 + } + } + }; ++ auto stage_packed = [&](int c, int buf) { ++ if (c < c_end) { ++ for (int e = tid; e < win * 2; e += CF_THREADS) { ++ const int u = e >> 1, seg = e & 1; ++ half * dst = s_win + buf * W_CHUNK + u * CF_LDW + seg * 8; ++ const int t = cache_col0 + t0 + u; ++ if (t < p.cache_len + p.T) { ++ const half * src = p.activated + ((size_t)grp * (p.cache_len + p.T) + t) * p.cin_pad + c * CF_BK + seg * 8; ++ cp_async16(dst, src); ++ } else { ++ *reinterpret_cast(dst) = make_uint4(0,0,0,0); ++ } ++ } ++ } ++ cp_async_commit(); ++ }; + float pf0[CF_PF], pf1[CF_PF]; + for (int q = 0; q < CF_STAGES - 1; ++q) stage_b(c_begin + q, q); -+ fetch_window(c_begin + 0, pf0); -+ fetch_window(c_begin + 1, pf1); -+ stage_act(c_begin + 0, 0); -+ stage_act(c_begin + 1, 1); ++ if constexpr (PACKED) { ++ stage_packed(c_begin, 0); ++ } else { ++ fetch_window(c_begin + 0, pf0); ++ fetch_window(c_begin + 1, pf1); ++ stage_act(c_begin + 0, 0); ++ stage_act(c_begin + 1, 1); ++ } + __syncthreads(); + + const int a_row = lane >> 2; @@ -332,12 +493,18 @@ index 00000000..336ba959 + const int c = c_begin + it; + const int slot = it % CF_STAGES; + const int wbuf = it & 1; -+ if (it & 1) store_window(c, pf1, wbuf); else store_window(c, pf0, wbuf); ++ if constexpr (!PACKED) { ++ if (it & 1) store_window(c, pf1, wbuf); else store_window(c, pf0, wbuf); ++ } + cp_async_wait(); + __syncthreads(); + stage_b(c + CF_STAGES - 1, (it + CF_STAGES - 1) % CF_STAGES); -+ if (it & 1) fetch_window(c + 2, pf1); else fetch_window(c + 2, pf0); -+ stage_act(c + 2, wbuf); // chunk it's params were consumed by store_window above (before the sync) ++ if constexpr (PACKED) { ++ stage_packed(c + 1, wbuf ^ 1); ++ } else { ++ if (it & 1) fetch_window(c + 2, pf1); else fetch_window(c + 2, pf0); ++ stage_act(c + 2, wbuf); ++ } // chunk it's params were consumed by store_window above (before the sync) + const half * bcur = s_b + slot * B_CHUNK; + const half * wcur = s_win + wbuf * W_CHUNK; + for (int k = 0; k < K; ++k) { @@ -499,6 +666,8 @@ index 00000000..336ba959 + CUDA_CHECK(cudaMemset(sc.counters, 0, CF_MAX_TILES * sizeof(int))); + CUDA_CHECK(cudaFuncSetAttribute(conv1d_fused_kernel<64>, cudaFuncAttributeMaxDynamicSharedMemorySize, 96 * 1024)); + CUDA_CHECK(cudaFuncSetAttribute(conv1d_fused_kernel<32>, cudaFuncAttributeMaxDynamicSharedMemorySize, 96 * 1024)); ++ CUDA_CHECK(cudaFuncSetAttribute(conv1d_fused_kernel<64, true>, cudaFuncAttributeMaxDynamicSharedMemorySize, 96 * 1024)); ++ CUDA_CHECK(cudaFuncSetAttribute(conv1d_fused_kernel<32, true>, cudaFuncAttributeMaxDynamicSharedMemorySize, 96 * 1024)); + } + return sc; +} @@ -577,33 +746,43 @@ index 00000000..336ba959 + p.counters = sc.counters; + cudaStream_t stream = ctx.stream(); + dim3 grid(g.tiles_m, g.tiles_n, g.splits); -+ if (g.BM == 64) conv1d_fused_kernel<64><<>>(p); -+ else conv1d_fused_kernel<32><<>>(p); ++ // Grouped problems with several cout tiles re-activate the same input per tile: pack and ++ // activate it once instead (bit-identical results; grid.y bounds the time extent). ++ if (p.groups > 1 && p.tiles_n > 1 && p.cache_len + p.T <= 65535) { ++ ggml_cuda_pool_alloc activated(ctx.pool(), (size_t)p.groups * (p.cache_len + p.T) * p.cin_pad); ++ p.activated = activated.get(); ++ conv1d_activate_pack<<>>(p,activated.get()); ++ if (g.BM == 64) conv1d_fused_kernel<64, true><<>>(p); ++ else conv1d_fused_kernel<32, true><<>>(p); ++ } else { ++ if (g.BM == 64) conv1d_fused_kernel<64><<>>(p); ++ else conv1d_fused_kernel<32><<>>(p); ++ } + CUDA_CHECK(cudaGetLastError()); +} -diff --git a/src/ggml-cuda/conv1d-fused.cuh b/src/ggml-cuda/conv1d-fused.cuh +diff --git a/ggml/src/ggml-cuda/conv1d-fused.cuh b/ggml/src/ggml-cuda/conv1d-fused.cuh new file mode 100644 -index 00000000..97bb27e1 +index 0000000000000000000000000000000000000000..97bb27e1fced299b96fc57c7bfc4a88d66195d12 --- /dev/null -+++ b/src/ggml-cuda/conv1d-fused.cuh ++++ b/ggml/src/ggml-cuda/conv1d-fused.cuh @@ -0,0 +1,4 @@ +#include "common.cuh" + +bool ggml_cuda_conv1d_fused_supported(const ggml_tensor * op, int cc); +void ggml_cuda_op_conv1d_fused(ggml_backend_cuda_context & ctx, ggml_tensor * dst); -diff --git a/src/ggml-cuda/ggml-cuda.cu b/src/ggml-cuda/ggml-cuda.cu -index e9a7c373..20841c2f 100644 ---- a/src/ggml-cuda/ggml-cuda.cu -+++ b/src/ggml-cuda/ggml-cuda.cu -@@ -26,6 +26,7 @@ - #include "ggml-cuda/diag.cuh" +diff --git a/ggml/src/ggml-cuda/ggml-cuda.cu b/ggml/src/ggml-cuda/ggml-cuda.cu +index 1924347f82ae913405ebdab956056c7d6f1a32f1..538da52595bdd669aa9a601240e9b3bd4ee40e89 100644 +--- a/ggml/src/ggml-cuda/ggml-cuda.cu ++++ b/ggml/src/ggml-cuda/ggml-cuda.cu +@@ -28,6 +28,7 @@ #include "ggml-cuda/fattn.cuh" + #include "ggml-cuda/fwht.cuh" #include "ggml-cuda/fused-attention.cuh" +#include "ggml-cuda/conv1d-fused.cuh" #include "ggml-cuda/getrows.cuh" #include "ggml-cuda/im2col.cuh" #include "ggml-cuda/mmf.cuh" -@@ -3322,6 +3323,9 @@ static bool ggml_cuda_compute_forward(ggml_backend_cuda_context & ctx, struct gg +@@ -2466,6 +2467,9 @@ static bool ggml_cuda_compute_forward(ggml_backend_cuda_context & ctx, struct gg case GGML_OP_FUSED_ATTN: ggml_cuda_op_fused_attention(ctx, dst); break; @@ -613,7 +792,130 @@ index e9a7c373..20841c2f 100644 case GGML_OP_CROSS_ENTROPY_LOSS: ggml_cuda_cross_entropy_loss(ctx, dst); break; -@@ -6441,6 +6445,8 @@ static bool ggml_backend_cuda_device_supports_op(ggml_backend_dev_t dev, const g +@@ -3645,6 +3649,111 @@ static bool ggml_cuda_can_fuse(const struct ggml_cgraph * cgraph, + return false; + } + ++static bool ggml_cuda_node_can_be_elided(const struct ggml_cgraph * cgraph, int node_idx, int32_t expected_uses) { ++ const ggml_tensor * node = cgraph->nodes[node_idx]; ++ return (node->flags & GGML_TENSOR_FLAG_COMPUTE) != 0 && ++ (node->flags & GGML_TENSOR_FLAG_OUTPUT) == 0 && ++ ggml_node_get_use_count(cgraph, node_idx) == expected_uses; ++} ++ ++static int ggml_cuda_try_fuse_half_snake(ggml_backend_cuda_context * cuda_ctx, ggml_cgraph * cgraph, int i) { ++ if (i <= 0 || i + 7 >= cgraph->n_nodes) { ++ return 0; ++ } ++ ++ ggml_tensor * mul0 = cgraph->nodes[i + 0]; ++ ggml_tensor * sin = cgraph->nodes[i + 1]; ++ ggml_tensor * sqr = cgraph->nodes[i + 2]; ++ ggml_tensor * mul1 = cgraph->nodes[i + 3]; ++ ggml_tensor * add = cgraph->nodes[i + 4]; ++ ggml_tensor * view_l = cgraph->nodes[i + 5]; ++ ggml_tensor * lrelu = cgraph->nodes[i + 6]; ++ ggml_tensor * concat = cgraph->nodes[i + 7]; ++ ++ if (mul0->op != GGML_OP_MUL || sin->op != GGML_OP_SIN || sqr->op != GGML_OP_SQR || ++ mul1->op != GGML_OP_MUL || add->op != GGML_OP_ADD || view_l->op != GGML_OP_VIEW || ++ lrelu->op != GGML_OP_LEAKY_RELU || concat->op != GGML_OP_CONCAT) { ++ return 0; ++ } ++ ++ if (!ggml_cuda_node_can_be_elided(cgraph, i + 0, 1) || ++ !ggml_cuda_node_can_be_elided(cgraph, i + 1, 1) || ++ !ggml_cuda_node_can_be_elided(cgraph, i + 2, 1) || ++ !ggml_cuda_node_can_be_elided(cgraph, i + 3, 1) || ++ !ggml_cuda_node_can_be_elided(cgraph, i + 4, 1) || ++ !ggml_cuda_node_can_be_elided(cgraph, i + 5, 1) || ++ !ggml_cuda_node_can_be_elided(cgraph, i + 6, 1)) { ++ return 0; ++ } ++ ++ const ggml_tensor * x_snake = ggml_are_same_shape(mul0, mul0->src[0]) ? mul0->src[0] : mul0->src[1]; ++ const ggml_tensor * alpha = (x_snake == mul0->src[0]) ? mul0->src[1] : mul0->src[0]; ++ if (x_snake->op != GGML_OP_VIEW || x_snake != cgraph->nodes[i - 1]) { ++ return 0; ++ } ++ if (!ggml_cuda_node_can_be_elided(cgraph, i - 1, 2)) { ++ return 0; ++ } ++ ++ const ggml_tensor * inv_b = (mul1->src[0] == sqr) ? mul1->src[1] : mul1->src[0]; ++ const ggml_tensor * x_in_add = (add->src[0] == mul1) ? add->src[1] : add->src[0]; ++ if (sin->src[0] != mul0 || sqr->src[0] != sin || (mul1->src[0] != sqr && mul1->src[1] != sqr) || ++ x_in_add != x_snake || lrelu->src[0] != view_l || concat->src[0] != add || concat->src[1] != lrelu) { ++ return 0; ++ } ++ ++ const int32_t concat_dim = ((const int32_t *) concat->op_params)[0]; ++ const bool type_ok = (x_snake->type == GGML_TYPE_F32 || x_snake->type == GGML_TYPE_F16 || x_snake->type == GGML_TYPE_BF16) && ++ x_snake->type == view_l->type && x_snake->type == concat->type; ++ const bool shape_ok = x_snake->view_src == view_l->view_src && ++ concat_dim == 1 && ++ x_snake->ne[0] == view_l->ne[0] && ++ x_snake->ne[2] == view_l->ne[2] && ++ x_snake->ne[3] == view_l->ne[3] && ++ concat->ne[0] == x_snake->ne[0] && ++ concat->ne[1] == x_snake->ne[1] + view_l->ne[1] && ++ concat->ne[2] == x_snake->ne[2] && ++ concat->ne[3] == x_snake->ne[3] && ++ alpha->type == GGML_TYPE_F32 && ++ inv_b->type == GGML_TYPE_F32 && ++ ggml_is_contiguous(alpha) && ++ ggml_is_contiguous(inv_b) && ++ ggml_nelements(alpha) == x_snake->ne[1] && ++ ggml_nelements(inv_b) == x_snake->ne[1] && ++ ggml_is_contiguous(x_snake) && ++ ggml_is_contiguous(view_l) && ++ ggml_is_contiguous(concat); ++ ++ if (!type_ok || !shape_ok) { ++ return 0; ++ } ++ ++ // The graph allocator planned these buffers for the *unfused* sequence, in which the ++ // parent activation dies at the leaky_relu and its memory may be recycled for the ++ // concat output -- the unfused concat reads the materialised add/leaky_relu, not the ++ // parent. The fused kernel still reads the parent while writing the concat, so an ++ // output overlapping an input at a different offset races. Fuse only when the output ++ // misses both inputs entirely, or aliases them exactly (each thread then reads and ++ // writes one and the same address). ++ const char * dst_beg = (const char *) concat->data; ++ const char * dst_end = dst_beg + ggml_nbytes(concat); ++ const char * snk_beg = (const char *) x_snake->data; ++ const char * snk_end = snk_beg + ggml_nbytes(x_snake); ++ const char * lrl_beg = (const char *) view_l->data; ++ const char * lrl_end = lrl_beg + ggml_nbytes(view_l); ++ ++ const bool disjoint = (snk_end <= dst_beg || dst_end <= snk_beg) && ++ (lrl_end <= dst_beg || dst_end <= lrl_beg); ++ const bool exact_inplace = dst_beg == snk_beg && lrl_beg == snk_end && dst_end == lrl_end; ++ ++ if (!disjoint && !exact_inplace) { ++ return 0; ++ } ++ ++ ggml_cuda_op_half_snake_fused(*cuda_ctx, x_snake, view_l, alpha, inv_b, lrelu, concat); ++ return 7; ++} ++ + // try and fuse nodes and return the number of nodes to skip + static int ggml_cuda_try_fuse(ggml_backend_cuda_context * cuda_ctx, ggml_cgraph * cgraph, int i) { + +@@ -3685,6 +3794,10 @@ static int ggml_cuda_try_fuse(ggml_backend_cuda_context * cuda_ctx, ggml_cgraph + } + } + ++ if (int n_fused = ggml_cuda_try_fuse_half_snake(cuda_ctx, cgraph, i)) { ++ return n_fused; ++ } ++ + const std::initializer_list batch_norm_ops = { + GGML_OP_SUB, GGML_OP_ADD, GGML_OP_SQRT, GGML_OP_DIV, GGML_OP_MUL, GGML_OP_ADD + }; +@@ -6161,6 +6274,8 @@ static bool ggml_backend_cuda_device_supports_op(ggml_backend_dev_t dev, const g return ggml_cuda_flash_attn_ext_supported(dev_ctx->device, op); case GGML_OP_FUSED_ATTN: return (op->src[0]->ne[0] & (op->src[0]->ne[0] - 1)) == 0; // d_k power of two @@ -622,35 +924,145 @@ index e9a7c373..20841c2f 100644 case GGML_OP_CROSS_ENTROPY_LOSS: case GGML_OP_CROSS_ENTROPY_LOSS_BACK: case GGML_OP_OPT_STEP_ADAMW: -diff --git a/src/ggml.c b/src/ggml.c -index cc2e1858..043c2efb 100644 ---- a/src/ggml.c -+++ b/src/ggml.c -@@ -1080,9 +1080,10 @@ static const char * GGML_OP_NAME[GGML_OP_COUNT] = { +diff --git a/ggml/src/ggml-cuda/snake.cu b/ggml/src/ggml-cuda/snake.cu +index 384638c1f475093aec13a08361a9ee96df75e280..6f4b219c3520fa954e2f5c72150a0a1f3ed1f412 100644 +--- a/ggml/src/ggml-cuda/snake.cu ++++ b/ggml/src/ggml-cuda/snake.cu +@@ -1,6 +1,8 @@ + #include "snake.cuh" + #include "convert.cuh" + ++#include ++ + // Fused Snake activation: y = x + sin^2(a * x) * inv_b + // x: [T, C] (T contiguous), a: [1, C], inv_b: [1, C] + // Supports F32, F16, BF16 data with F32 compute. +@@ -70,3 +72,80 @@ void ggml_cuda_op_snake_fused(ggml_backend_cuda_context & ctx, + ggml_tensor * dst) { + launch_snake(ctx, x, a, inv_b, dst); + } ++ ++template ++static __global__ void half_snake_kernel( ++ const T * __restrict__ x_snake, ++ const T * __restrict__ x_lrelu, ++ const float * __restrict__ a, ++ const float * __restrict__ inv_b, ++ T * __restrict__ dst, ++ const int total, ++ const int T_len, ++ const int snake_channels, ++ const int lrelu_channels, ++ const int channels, ++ const float negative_slope) { ++ const int idx = blockIdx.x * blockDim.x + threadIdx.x; ++ if (idx >= total) return; ++ ++ const int t = idx % T_len; ++ const int c = (idx / T_len) % channels; ++ const int plane = idx / (T_len * channels); ++ ++ if (c < snake_channels) { ++ const int src_idx = plane * T_len * snake_channels + c * T_len + t; ++ const float xi = ggml_cuda_cast(x_snake[src_idx]); ++ const float s = sinf(a[c] * xi); ++ dst[idx] = ggml_cuda_cast(xi + s * s * inv_b[c]); ++ } else { ++ const int lc = c - snake_channels; ++ const int src_idx = plane * T_len * lrelu_channels + lc * T_len + t; ++ const float xi = ggml_cuda_cast(x_lrelu[src_idx]); ++ dst[idx] = ggml_cuda_cast(fmaxf(xi, 0.0f) + fminf(xi, 0.0f) * negative_slope); ++ } ++} ++ ++void ggml_cuda_op_half_snake_fused(ggml_backend_cuda_context & ctx, ++ const ggml_tensor * x_snake, ++ const ggml_tensor * x_lrelu, ++ const ggml_tensor * a, ++ const ggml_tensor * inv_b, ++ const ggml_tensor * lrelu, ++ ggml_tensor * dst) { ++ float negative_slope; ++ std::memcpy(&negative_slope, lrelu->op_params, sizeof(float)); ++ ++ const int T = (int) dst->ne[0]; ++ const int snake_channels = (int) x_snake->ne[1]; ++ const int lrelu_channels = (int) x_lrelu->ne[1]; ++ const int channels = (int) dst->ne[1]; ++ const int total = (int) ggml_nelements(dst); ++ ++ const int block_size = 256; ++ const int grid_size = (total + block_size - 1) / block_size; ++ cudaStream_t stream = ctx.stream(); ++ ++ const float * a_d = (const float *) a->data; ++ const float * inv_b_d = (const float *) inv_b->data; ++ ++ switch (dst->type) { ++ case GGML_TYPE_F32: { ++ half_snake_kernel<<>>( ++ (const float *) x_snake->data, (const float *) x_lrelu->data, a_d, inv_b_d, (float *) dst->data, ++ total, T, snake_channels, lrelu_channels, channels, negative_slope); ++ } break; ++ case GGML_TYPE_F16: { ++ half_snake_kernel<<>>( ++ (const half *) x_snake->data, (const half *) x_lrelu->data, a_d, inv_b_d, (half *) dst->data, ++ total, T, snake_channels, lrelu_channels, channels, negative_slope); ++ } break; ++ case GGML_TYPE_BF16: { ++ half_snake_kernel<<>>( ++ (const nv_bfloat16 *) x_snake->data, (const nv_bfloat16 *) x_lrelu->data, a_d, inv_b_d, ++ (nv_bfloat16 *) dst->data, total, T, snake_channels, lrelu_channels, channels, negative_slope); ++ } break; ++ default: ++ GGML_ABORT("half_snake: unsupported type"); ++ } ++} +diff --git a/ggml/src/ggml-cuda/snake.cuh b/ggml/src/ggml-cuda/snake.cuh +index 7f6f1cb3b41bcd7d98876c9f8acf6a1489e6229b..7a3e7aa675bffefdd7910c4d24d6a26f38e1dcdb 100644 +--- a/ggml/src/ggml-cuda/snake.cuh ++++ b/ggml/src/ggml-cuda/snake.cuh +@@ -6,3 +6,11 @@ void ggml_cuda_op_snake_fused(ggml_backend_cuda_context & ctx, + const ggml_tensor * a, + const ggml_tensor * inv_b, + ggml_tensor * dst); ++ ++void ggml_cuda_op_half_snake_fused(ggml_backend_cuda_context & ctx, ++ const ggml_tensor * x_snake, ++ const ggml_tensor * x_lrelu, ++ const ggml_tensor * a, ++ const ggml_tensor * inv_b, ++ const ggml_tensor * lrelu, ++ ggml_tensor * dst); +diff --git a/ggml/src/ggml.c b/ggml/src/ggml.c +index 3aaa06603030c43ee8dfe72a2170272327e3f708..178d2bc275e4c1efce9aeffd62f3fc5565385f39 100644 +--- a/ggml/src/ggml.c ++++ b/ggml/src/ggml.c +@@ -1101,9 +1101,10 @@ static const char * GGML_OP_NAME[GGML_OP_COUNT] = { "GLU", "FUSED_ATTN", + "CONV1D_FUSED", }; --static_assert(GGML_OP_COUNT == 97, "GGML_OP_COUNT != 97"); -+static_assert(GGML_OP_COUNT == 98, "GGML_OP_COUNT != 98"); +-static_assert(GGML_OP_COUNT == 102, "GGML_OP_COUNT != 102"); ++static_assert(GGML_OP_COUNT == 103, "GGML_OP_COUNT != 103"); static const char * GGML_OP_SYMBOL[GGML_OP_COUNT] = { "none", -@@ -1192,9 +1193,10 @@ static const char * GGML_OP_SYMBOL[GGML_OP_COUNT] = { +@@ -1218,9 +1219,10 @@ static const char * GGML_OP_SYMBOL[GGML_OP_COUNT] = { "glu(x)", "fused_attn(q,k,v,p)", + "conv1d_fused(x,w)", }; --static_assert(GGML_OP_COUNT == 97, "GGML_OP_COUNT != 97"); -+static_assert(GGML_OP_COUNT == 98, "GGML_OP_COUNT != 98"); +-static_assert(GGML_OP_COUNT == 102, "GGML_OP_COUNT != 102"); ++static_assert(GGML_OP_COUNT == 103, "GGML_OP_COUNT != 103"); static_assert(GGML_OP_POOL_COUNT == 2, "GGML_OP_POOL_COUNT != 2"); -@@ -5548,6 +5550,100 @@ struct ggml_tensor * ggml_fused_attn_cached( +@@ -5707,6 +5709,100 @@ struct ggml_tensor * ggml_fused_attn_cached( merge_heads); } diff --git a/llama-patches/0002-enable-nvfp4-quantization.patch b/patches/llama-enable-nvfp4-in-llama-quantize.patch similarity index 60% rename from llama-patches/0002-enable-nvfp4-quantization.patch rename to patches/llama-enable-nvfp4-in-llama-quantize.patch index d1b3fbd..0ddce23 100644 --- a/llama-patches/0002-enable-nvfp4-quantization.patch +++ b/patches/llama-enable-nvfp4-in-llama-quantize.patch @@ -1,28 +1,36 @@ +From: Prabhsimran Singh +Subject: llama: enable NVFP4 in llama-quantize + +Maps LLAMA_FTYPE_MOSTLY_NVFP4 to GGML_TYPE_NVFP4 and exposes it in +llama-quantize for the S2S nvfp4 conversion profile. + +Upstream: candidate (needs PPL/KLD data). + diff --git a/src/llama-quant.cpp b/src/llama-quant.cpp -index 03c0e292c..9619df75b 100644 +index 34ff25db57e661ceceabc1579983ff908dbbb3ed..9bbb61b366b98970e88a8199e090c86050c18afe 100644 --- a/src/llama-quant.cpp +++ b/src/llama-quant.cpp -@@ -387,6 +387,7 @@ static ggml_type get_fallback_type(ggml_type target_type, const ggml_tensor * te +@@ -398,6 +398,7 @@ static ggml_type tensor_type_fallback(quantize_state_impl & qs, const ggml_tenso case GGML_TYPE_Q4_K: return_type = GGML_TYPE_Q5_0; break; case GGML_TYPE_Q5_K: return_type = GGML_TYPE_Q5_1; break; case GGML_TYPE_Q6_K: return_type = GGML_TYPE_Q8_0; break; + case GGML_TYPE_NVFP4: return_type = GGML_TYPE_Q8_0; break; default: - throw std::runtime_error(format("no tensor type fallback is defined for type %s", - ggml_type_name(target_type))); -@@ -802,6 +803,7 @@ ggml_type llama_ftype_get_default_type(llama_ftype ftype) { - case LLAMA_FTYPE_MOSTLY_Q1_0: return GGML_TYPE_Q1_0; - + if (qk_k <= 32) { + // the target is already a 32-block type, so there is no smaller block to demote to +@@ -858,6 +859,7 @@ ggml_type llama_ftype_get_default_type(llama_ftype ftype) { + case LLAMA_FTYPE_MOSTLY_Q2_0: return GGML_TYPE_Q2_0; + case LLAMA_FTYPE_MOSTLY_MXFP4_MOE: return GGML_TYPE_MXFP4; + case LLAMA_FTYPE_MOSTLY_NVFP4: return GGML_TYPE_NVFP4; - + // K-quants case LLAMA_FTYPE_MOSTLY_Q2_K_S: diff --git a/tools/quantize/quantize.cpp b/tools/quantize/quantize.cpp -index 3b072966f..c7e5b3d7e 100644 +index 38950036cd820e41f49fe28714a03ea261ce9df3..daa0a819b60afafe3ba1a2b313fec1303b6e460f 100644 --- a/tools/quantize/quantize.cpp +++ b/tools/quantize/quantize.cpp -@@ -36,6 +36,7 @@ static const std::vector QUANT_OPTIONS = { +@@ -37,6 +37,7 @@ static const std::vector QUANT_OPTIONS = { { "Q4_0", LLAMA_FTYPE_MOSTLY_Q4_0, " 4.34G, +0.4685 ppl @ Llama-3-8B", }, { "Q4_1", LLAMA_FTYPE_MOSTLY_Q4_1, " 4.78G, +0.4511 ppl @ Llama-3-8B", }, { "MXFP4_MOE",LLAMA_FTYPE_MOSTLY_MXFP4_MOE," MXFP4 MoE", }, diff --git a/llama-patches/0001-batch-all-recurrent-outputs.patch b/patches/llama-keep-equal-length-recurrent-sequences-in-one-ubatch.patch similarity index 57% rename from llama-patches/0001-batch-all-recurrent-outputs.patch rename to patches/llama-keep-equal-length-recurrent-sequences-in-one-ubatch.patch index a63d010..d9e5dbd 100644 --- a/llama-patches/0001-batch-all-recurrent-outputs.patch +++ b/patches/llama-keep-equal-length-recurrent-sequences-in-one-ubatch.patch @@ -1,18 +1,34 @@ +From: Prabhsimran Singh +Subject: llama: keep equal-length recurrent sequences in one ubatch + +When every token's output is requested (embeddings), recurrent and hybrid +memories split by sequence, turning an N-stream decode into N graph +executions and invalidating the CUDA graph. Use the equal split instead; +decode() restores output order. Used by the S2S NemotronH backbone. + +Upstream: candidate once split_seq is kept for pooled embeddings. + diff --git a/src/llama-memory-hybrid-iswa.cpp b/src/llama-memory-hybrid-iswa.cpp -index a59561ea5..0ba969e7d 100644 +index 06f7fd5428c4e92c445b5c0170e92e557b2f62df..eaaa5a9fa87227f1f9f7e8ea3ef32b4e042328de 100644 --- a/src/llama-memory-hybrid-iswa.cpp +++ b/src/llama-memory-hybrid-iswa.cpp -@@ -71,14 +71,14 @@ llama_memory_context_ptr llama_memory_hybrid_iswa::init_batch(llama_batch_allocr +@@ -73,20 +73,20 @@ llama_memory_context_ptr llama_memory_hybrid_iswa::init_batch(llama_batch_allocr while (true) { llama_ubatch ubatch; - + - if (embd_all) { - // if all tokens are output, split by sequence - ubatch = balloc.split_seq(n_ubatch); - } else { - // Use non-sequential split when KV cache is unified (needed for hellaswag/winogrande/multiple-choice) - const bool unified = (mem_attn->get_base()->get_n_stream() == 1); -- ubatch = balloc.split_equal(n_ubatch, !unified); +- +- // [TAG_RECURRENT_ROLLBACK_SPLITS] +- // the trailing (1 + n_rs_seq) tokens of each seq must stay in the same ubatch +- // so that the rollback snapshots remain valid +- const uint32_t n_rs_seq = mem_recr->n_rs_seq; +- +- ubatch = balloc.split_equal(n_ubatch, !unified, n_rs_seq > 0 ? n_rs_seq + 1 : 0); - } + // Requesting every token's output does not require serializing + // independent equal-length sequences. Keep them in one recurrent @@ -20,26 +36,38 @@ index a59561ea5..0ba969e7d 100644 + // Use non-sequential split when KV cache is unified (needed for + // hellaswag/winogrande/multiple-choice). + const bool unified = (mem_attn->get_base()->get_n_stream() == 1); ++ ++ // [TAG_RECURRENT_ROLLBACK_SPLITS] ++ // the trailing (1 + n_rs_seq) tokens of each seq must stay in the same ubatch ++ // so that the rollback snapshots remain valid ++ const uint32_t n_rs_seq = mem_recr->n_rs_seq; ++ + GGML_UNUSED(embd_all); -+ ubatch = balloc.split_equal(n_ubatch, !unified); - ++ ubatch = balloc.split_equal(n_ubatch, !unified, n_rs_seq > 0 ? n_rs_seq + 1 : 0); + if (ubatch.n_tokens == 0) { break; diff --git a/src/llama-memory-hybrid.cpp b/src/llama-memory-hybrid.cpp -index fd305cab7..951dd28df 100644 +index 42c7381a9e6fa507c925ba66a35a7e433c4447b8..577844e070989eb6a541ef773bddc57b53dcfc22 100644 --- a/src/llama-memory-hybrid.cpp +++ b/src/llama-memory-hybrid.cpp -@@ -71,14 +71,16 @@ llama_memory_context_ptr llama_memory_hybrid::init_batch(llama_batch_allocr & ba +@@ -74,20 +74,22 @@ llama_memory_context_ptr llama_memory_hybrid::init_batch(llama_batch_allocr & ba while (true) { llama_ubatch ubatch; - + - if (embd_all) { - // if all tokens are output, split by sequence - ubatch = balloc.split_seq(n_ubatch); - } else { - // Use non-sequential split when KV cache is unified (needed for hellaswag/winogrande/multiple-choice) - const bool unified = (mem_attn->get_n_stream() == 1); -- ubatch = balloc.split_equal(n_ubatch, !unified); +- +- // [TAG_RECURRENT_ROLLBACK_SPLITS] +- // the trailing (1 + n_rs_seq) tokens of each seq must stay in the same ubatch +- // so that the rollback snapshots remain valid +- const uint32_t n_rs_seq = mem_recr->n_rs_seq; +- +- ubatch = balloc.split_equal(n_ubatch, !unified, n_rs_seq > 0 ? n_rs_seq + 1 : 0); - } + // Requesting every token's output does not require serializing + // independent equal-length sequences. Keeping them together is @@ -49,26 +77,35 @@ index fd305cab7..951dd28df 100644 + // Use non-sequential split when KV cache is unified (needed for + // hellaswag/winogrande/multiple-choice). + const bool unified = (mem_attn->get_n_stream() == 1); ++ ++ // [TAG_RECURRENT_ROLLBACK_SPLITS] ++ // the trailing (1 + n_rs_seq) tokens of each seq must stay in the same ubatch ++ // so that the rollback snapshots remain valid ++ const uint32_t n_rs_seq = mem_recr->n_rs_seq; ++ + GGML_UNUSED(embd_all); -+ ubatch = balloc.split_equal(n_ubatch, !unified); - ++ ubatch = balloc.split_equal(n_ubatch, !unified, n_rs_seq > 0 ? n_rs_seq + 1 : 0); + if (ubatch.n_tokens == 0) { break; diff --git a/src/llama-memory-recurrent.cpp b/src/llama-memory-recurrent.cpp -index aeb866657..9991e89cf 100644 +index 57919accf09569e129ab35b5b2bd4dab7dbe8a3a..954b86fe52425658959c07fb0c0ef8543355b550 100644 --- a/src/llama-memory-recurrent.cpp +++ b/src/llama-memory-recurrent.cpp -@@ -412,14 +412,20 @@ llama_memory_context_ptr llama_memory_recurrent::init_batch(llama_batch_allocr & +@@ -433,17 +433,23 @@ llama_memory_context_ptr llama_memory_recurrent::init_batch(llama_batch_allocr & while (true) { llama_ubatch ubatch; - + - if (embd_all) { - // if all tokens are output, split by sequence - ubatch = balloc.split_seq(n_ubatch); - } else { - // TODO: non-sequential equal split can be done if using unified KV cache - // for simplicity, we always use sequential equal split for now -- ubatch = balloc.split_equal(n_ubatch, true); +- // [TAG_RECURRENT_ROLLBACK_SPLITS] +- // the trailing (1 + n_rs_seq) tokens of each seq must stay in the same ubatch +- // so that the rollback snapshots remain valid +- ubatch = balloc.split_equal(n_ubatch, true, n_rs_seq > 0 ? n_rs_seq + 1 : 0); - } + // Recurrent models require equal-length independent sequences in a + // microbatch, but producing an embedding for every token does not @@ -82,8 +119,11 @@ index aeb866657..9991e89cf 100644 + // retains the existing behavior for a single multi-token prompt. + // TODO: non-sequential equal split can be done if using unified KV cache + // for simplicity, we always use sequential equal split for now ++ // [TAG_RECURRENT_ROLLBACK_SPLITS] ++ // the trailing (1 + n_rs_seq) tokens of each seq must stay in the same ubatch ++ // so that the rollback snapshots remain valid + GGML_UNUSED(embd_all); -+ ubatch = balloc.split_equal(n_ubatch, true); - ++ ubatch = balloc.split_equal(n_ubatch, true, n_rs_seq > 0 ? n_rs_seq + 1 : 0); + if (ubatch.n_tokens == 0) { break; diff --git a/llama-patches/0003-gemma3-attention-scale.patch b/patches/llama-read-the-gemma-3-attention-scale-from-gguf.patch similarity index 73% rename from llama-patches/0003-gemma3-attention-scale.patch rename to patches/llama-read-the-gemma-3-attention-scale-from-gguf.patch index 4a53ea9..e11ccf1 100644 --- a/llama-patches/0003-gemma3-attention-scale.patch +++ b/patches/llama-read-the-gemma-3-attention-scale-from-gguf.patch @@ -1,19 +1,27 @@ +From: Prabhsimran Singh +Subject: llama: read the Gemma 3 attention scale from GGUF + +Honors gemma3.attention.scale (written by the EarTTS converter), falling back +to the upstream default. + +Upstream: candidate (needs converter support and a round-trip test). + diff --git a/src/models/gemma3.cpp b/src/models/gemma3.cpp -index 63a2b380e..b982bd9c6 100644 +index f99bbaacd8adcdd34810064eb4b2320ff28e2d71..48da85e596f97852622636fc6925287ea5684f85 100644 --- a/src/models/gemma3.cpp +++ b/src/models/gemma3.cpp @@ -1,6 +1,8 @@ #include "models.h" - + void llama_model_gemma3::load_arch_hparams(llama_model_loader & ml) { + const bool found_attention_scale = + ml.get_key(LLM_KV_ATTENTION_SCALE, hparams.f_attention_scale, false); const bool found_swa = ml.get_key(LLM_KV_ATTENTION_SLIDING_WINDOW, hparams.n_swa, false); if (found_swa && hparams.n_swa > 0) { hparams.swa_type = LLAMA_SWA_TYPE_STANDARD; -@@ -28,9 +30,11 @@ void llama_model_gemma3::load_arch_hparams(llama_model_loader & ml) { +@@ -26,9 +28,11 @@ void llama_model_gemma3::load_arch_hparams(llama_model_loader & ml) { } - + // ref: https://github.com/google/gemma_pytorch/blob/014acb7ac4563a5f77c76d7ff98f31b568c16508/gemma/config.py#L289 - hparams.f_attention_scale = type == LLM_TYPE_27B - ? 1.0f / std::sqrt(float(hparams.n_embd / hparams.n_head(0))) @@ -24,5 +32,5 @@ index 63a2b380e..b982bd9c6 100644 + : 1.0f / std::sqrt(float(hparams.n_embd_head_k())); + } } - + void llama_model_gemma3::load_arch_tensors(llama_model_loader &) { diff --git a/patches/series b/patches/series new file mode 100644 index 0000000..b0e418a --- /dev/null +++ b/patches/series @@ -0,0 +1,20 @@ +# Apply order for the patches in this directory (scripts/llama-patches.sh export). +ggml-add-fused-attention-op-for-fastconformer-and-magpietts.patch +ggml-cuda-fuse-layernorm-with-row-vector-scale-and-bias.patch +ggml-cuda-support-f16-weights-in-direct-depthwise-conv.patch +ggml-cuda-skinny-q8_0-gemm-with-planar-weights.patch +ggml-cuda-target-jetson-thor-sm110.patch +ggml-cuda-cuda-graph-cache-keys-replay-updates-and-eviction-knobs.patch +ggml-cuda-two-column-mmvf-epilogue.patch +ggml-cuda-flatten-shared-weight-cublas-gemms-fuse-silu-to-bf16.patch +ggml-fastconformer-cuda-fusions-and-sigmoid-glu.patch +ggml-cuda-flatten-pad-launches-for-large-batches.patch +ggml-cuda-vectorized-contiguous-set_rows.patch +ggml-cuda-expose-the-backend-stream-and-cuda-graph-controls.patch +ggml-cuda-bf16-depthwise-conv-and-bias-round-epilogue.patch +ggml-cuda-tiled-1d-im2col.patch +ggml-nanocodec-convolution-kernels.patch +ggml-cpu-f16c-scalar-fp32-fp16-and-row-split-f16-im2col.patch +llama-keep-equal-length-recurrent-sequences-in-one-ubatch.patch +llama-enable-nvfp4-in-llama-quantize.patch +llama-read-the-gemma-3-attention-scale-from-gguf.patch diff --git a/scripts/apply-ggml-patches.sh b/scripts/apply-ggml-patches.sh deleted file mode 100755 index a69a5d0..0000000 --- a/scripts/apply-ggml-patches.sh +++ /dev/null @@ -1,121 +0,0 @@ -#!/usr/bin/env bash -# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. -# SPDX-License-Identifier: Apache-2.0 -# Apply the in-tree ggml patches (ggml-patches/*.patch, in filename order) onto -# the vendored ggml submodule. This keeps the submodule pinned to clean -# upstream. Project-specific kernels and runtime extensions are applied during -# build setup. -# -# Patches that still reverse-apply cleanly are skipped. Apply the complete -# series to a clean submodule for deterministic setup; later patches may refine -# lines introduced by earlier patches, making reverse detection ambiguous on a -# fully patched tree. -# Usage: scripts/apply-ggml-patches.sh -set -euo pipefail - -ROOT="$(cd "$(dirname "${BASH_SOURCE[0]}")/.." && pwd)" -GGML="${ROOT}/ggml" -PATCHES="${ROOT}/ggml-patches" -source "${ROOT}/scripts/patch-series-common.sh" - -if [ ! -d "${GGML}/src" ]; then - echo "error: ggml submodule not initialized at ${GGML}" >&2 - echo " run: git submodule update --init ggml" >&2 - exit 1 -fi - -shopt -s nullglob -patch_files=("${PATCHES}"/*.patch) -if [ "${#patch_files[@]}" -eq 0 ]; then - echo "error: no .patch files found in ${PATCHES}" >&2 - exit 1 -fi - -# A per-patch reverse check is not sufficient once a later patch changes a -# hunk introduced by an earlier patch. Compare the complete worktree state -# with the result of applying the full series to the pinned submodule commit. -# Temporary indexes keep both this check and the caller's real index untouched. -if git -c "safe.directory=${GGML}" -C "${GGML}" rev-parse --git-dir >/dev/null 2>&1; then - tmp_dir="$(mktemp -d)" - expected_index="${tmp_dir}/expected.index" - current_index="${tmp_dir}/current.index" - cleanup_indexes() { - rm -rf "${tmp_dir}" - } - trap cleanup_indexes EXIT - - GIT_INDEX_FILE="${expected_index}" git -c "safe.directory=${GGML}" -C "${GGML}" read-tree HEAD - for p in "${patch_files[@]}"; do - if ! GIT_INDEX_FILE="${expected_index}" git -c "safe.directory=${GGML}" -C "${GGML}" apply --cached "${p}"; then - echo "error: $(basename "${p}") does not apply to the pinned ggml commit" >&2 - exit 1 - fi - done - - # note: not mapfile -d '' - that is bash 4+, and macOS ships bash 3.2 - patched_paths=() - while IFS= read -r -d '' path; do - patched_paths+=("${path}") - done < <(GIT_INDEX_FILE="${expected_index}" git -c "safe.directory=${GGML}" -C "${GGML}" \ - diff --cached --name-only -z HEAD) - GIT_INDEX_FILE="${current_index}" git -c "safe.directory=${GGML}" -C "${GGML}" read-tree HEAD - current_paths=() - for path in "${patched_paths[@]}"; do - if [ -e "${GGML}/${path}" ] || [ -L "${GGML}/${path}" ] \ - || git -c "safe.directory=${GGML}" -C "${GGML}" cat-file -e "HEAD:${path}" 2>/dev/null; then - current_paths+=("${path}") - fi - done - if [ "${#current_paths[@]}" -ne 0 ]; then - GIT_INDEX_FILE="${current_index}" git -c "safe.directory=${GGML}" -C "${GGML}" \ - add -A -- "${current_paths[@]}" - fi - - expected_tree="$(GIT_INDEX_FILE="${expected_index}" git -c "safe.directory=${GGML}" -C "${GGML}" write-tree)" - current_tree="$(GIT_INDEX_FILE="${current_index}" git -c "safe.directory=${GGML}" -C "${GGML}" write-tree)" - if [ "${current_tree}" = "${expected_tree}" ]; then - echo "[ggml-patch] current series already applied" - echo "[ggml-patch] done" - exit 0 - fi -else - # Docker build contexts intentionally omit the submodule's .git file. Test - # the complete applied series in reverse order against a temporary index; - # this handles later patches that refine hunks introduced by earlier ones - # without changing the copied source tree. - if patch_series_applied_without_git "${GGML}" "${PATCHES}" "[ggml-patch]"; then - echo "[ggml-patch] done" - exit 0 - fi -fi - -for p in "${patch_files[@]}"; do - name="$(basename "${p}")" - # Already applied? (the reverse patch applies cleanly) -> skip. - if git -c "safe.directory=${GGML}" -C "${GGML}" apply --reverse --check "${p}" >/dev/null 2>&1; then - echo "[ggml-patch] ${name}: already applied" - continue - fi - if ! git -c "safe.directory=${GGML}" -C "${GGML}" apply --check "${p}" >/dev/null 2>&1; then - echo "[ggml-patch] ${name}: does NOT apply cleanly to current ggml" >&2 - echo " (the tree is modified or contains a stale patch series)" >&2 - echo " restore the pinned ggml submodule, then retry" >&2 - exit 1 - fi - git -c "safe.directory=${GGML}" -C "${GGML}" apply "${p}" - echo "[ggml-patch] ${name}: applied" -done - -# Intent-add any files the patches created so a later `git diff`-based patch -# regeneration includes them. Without this, regenerating a patch that owns a -# NEW file silently produces an empty diff (untracked files are invisible to -# `git diff`). Only meaningful on a dev host where ggml is a git repo; the -# docker build copies ggml without its .git, so skip there. -if git -c "safe.directory=${GGML}" -C "${GGML}" rev-parse --git-dir >/dev/null 2>&1; then - git -c "safe.directory=${GGML}" -C "${GGML}" status --porcelain \ - | awk '$1 == "??" { print $2 }' \ - | while read -r f; do - git -c "safe.directory=${GGML}" -C "${GGML}" add -N "${f}" - done -fi -echo "[ggml-patch] done" diff --git a/scripts/apply-llama-patches.sh b/scripts/apply-llama-patches.sh deleted file mode 100755 index acffb39..0000000 --- a/scripts/apply-llama-patches.sh +++ /dev/null @@ -1,84 +0,0 @@ -#!/usr/bin/env bash -# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. -# SPDX-License-Identifier: Apache-2.0 -# Apply the NeMo-Speech.cpp llama.cpp patch series to the pinned submodule. -set -euo pipefail - -ROOT="$(cd "$(dirname "${BASH_SOURCE[0]}")/.." && pwd)" -LLAMA="${ROOT}/llama.cpp" -PATCHES="${ROOT}/llama-patches" -source "${ROOT}/scripts/patch-series-common.sh" - -if [ ! -d "${LLAMA}/src" ]; then - echo "error: llama.cpp submodule not initialized at ${LLAMA}" >&2 - echo " run: git submodule update --init llama.cpp" >&2 - exit 1 -fi - -shopt -s nullglob -patch_files=("${PATCHES}"/*.patch) -if [ "${#patch_files[@]}" -eq 0 ]; then - echo "error: no .patch files found in ${PATCHES}" >&2 - exit 1 -fi - -if git -c "safe.directory=${LLAMA}" -C "${LLAMA}" rev-parse --git-dir >/dev/null 2>&1; then - tmp_dir="$(mktemp -d)" - expected_index="${tmp_dir}/expected.index" - current_index="${tmp_dir}/current.index" - cleanup_indexes() { - rm -rf "${tmp_dir}" - } - trap cleanup_indexes EXIT - - GIT_INDEX_FILE="${expected_index}" git -c "safe.directory=${LLAMA}" -C "${LLAMA}" read-tree HEAD - for patch in "${patch_files[@]}"; do - if ! GIT_INDEX_FILE="${expected_index}" git -c "safe.directory=${LLAMA}" -C "${LLAMA}" apply --cached "${patch}"; then - echo "error: $(basename "${patch}") does not apply to the pinned llama.cpp commit" >&2 - exit 1 - fi - done - - patched_paths=() - while IFS= read -r -d '' path; do - patched_paths+=("${path}") - done < <(GIT_INDEX_FILE="${expected_index}" git -c "safe.directory=${LLAMA}" -C "${LLAMA}" \ - diff --cached --name-only -z HEAD) - GIT_INDEX_FILE="${current_index}" git -c "safe.directory=${LLAMA}" -C "${LLAMA}" read-tree HEAD - if [ "${#patched_paths[@]}" -ne 0 ]; then - GIT_INDEX_FILE="${current_index}" git -c "safe.directory=${LLAMA}" -C "${LLAMA}" add -A -- "${patched_paths[@]}" - fi - expected_tree="$(GIT_INDEX_FILE="${expected_index}" git -c "safe.directory=${LLAMA}" -C "${LLAMA}" write-tree)" - current_tree="$(GIT_INDEX_FILE="${current_index}" git -c "safe.directory=${LLAMA}" -C "${LLAMA}" write-tree)" - if [ "${current_tree}" = "${expected_tree}" ]; then - echo "[llama-patch] current series already applied" - echo "[llama-patch] done" - exit 0 - fi -else - # Docker build contexts intentionally omit the submodule's .git file. A - # per-patch reverse check is insufficient when a later patch changes a - # hunk introduced by an earlier one, so validate the complete applied - # series against a temporary index, in reverse order. This never changes - # the copied source tree. - if patch_series_applied_without_git "${LLAMA}" "${PATCHES}" "[llama-patch]"; then - echo "[llama-patch] done" - exit 0 - fi -fi - -for patch in "${patch_files[@]}"; do - name="$(basename "${patch}")" - if git -c "safe.directory=${LLAMA}" -C "${LLAMA}" apply --reverse --check "${patch}" >/dev/null 2>&1; then - echo "[llama-patch] ${name}: already applied" - continue - fi - if ! git -c "safe.directory=${LLAMA}" -C "${LLAMA}" apply --check "${patch}" >/dev/null 2>&1; then - echo "[llama-patch] ${name}: does NOT apply cleanly to current llama.cpp" >&2 - echo " restore the pinned submodule, then retry" >&2 - exit 1 - fi - git -c "safe.directory=${LLAMA}" -C "${LLAMA}" apply "${patch}" - echo "[llama-patch] ${name}: applied" -done -echo "[llama-patch] done" diff --git a/scripts/configure.sh b/scripts/configure.sh index ad056d4..627f8a8 100755 --- a/scripts/configure.sh +++ b/scripts/configure.sh @@ -1,8 +1,7 @@ #!/usr/bin/env bash # SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. # SPDX-License-Identifier: Apache-2.0 -# Validate a source checkout, apply the pinned ggml CUDA and llama.cpp patches -# when needed, and configure one of the supported CMake presets. +# Validate a source checkout and configure one of the supported CMake presets. set -euo pipefail ROOT="$(cd "$(dirname "${BASH_SOURCE[0]}")/.." && pwd)" @@ -21,9 +20,8 @@ Examples: scripts/configure.sh cpu-server scripts/configure.sh cuda-server -DNEMO_SPEECH_WITH_NORM=ON -The script checks required submodules and optional feature assets, applies the -pinned ggml patch series for CUDA and Metal presets and the llama.cpp batching -patch series when required, and then runs cmake --preset PRESET. +The script checks required submodules and optional feature assets, then runs +cmake --preset PRESET. CMake applies patches/ to llama.cpp (and ggml) itself. EOF exit 0 fi @@ -50,12 +48,6 @@ esac cd "$ROOT" -if [ ! -f ggml/CMakeLists.txt ]; then - echo "error: ggml submodule is not initialized" >&2 - echo " run: git submodule update --init ggml" >&2 - exit 1 -fi - cmake_bool_override() { # cmake_bool_override VARIABLE DEFAULT ARGS... local variable="$1" value="$2" arg definition key setting shift 2 @@ -76,23 +68,11 @@ cmake_bool_override() { # cmake_bool_override VARIABLE DEFAULT ARGS... printf '%s' "$value" } -need_nmt=OFF -need_asr=OFF -need_s2s=OFF need_grpc=OFF need_http=OFF need_flashlight=OFF need_ja=OFF need_zh=OFF -case "$PRESET" in - *-nmt|*-speech|*-server|cuda-full|developer) need_nmt=ON ;; -esac -case "$PRESET" in - *-asr|*-speech|*-server|cuda-full|developer) need_asr=ON ;; -esac -case "$PRESET" in - cuda-s2s) need_s2s=ON ;; -esac case "$PRESET" in cuda-full|developer) need_grpc=ON ;; esac @@ -105,12 +85,7 @@ if [ "$PRESET" = cuda-full ]; then need_zh=ON fi -need_nmt="$(cmake_bool_override NEMO_SPEECH_BUILD_NMT "$need_nmt" "$@")" -need_nmt="$(cmake_bool_override NEMO_SPEECH_WITH_NMT "$need_nmt" "$@")" -need_asr="$(cmake_bool_override NEMO_SPEECH_BUILD_ASR "$need_asr" "$@")" -need_s2s="$(cmake_bool_override NEMO_SPEECH_BUILD_S2S "$need_s2s" "$@")" need_grpc="$(cmake_bool_override NEMO_SPEECH_BUILD_GRPC "$need_grpc" "$@")" -need_grpc="$(cmake_bool_override NEMO_SPEECH_WITH_GRPC "$need_grpc" "$@")" need_http="$(cmake_bool_override NEMO_SPEECH_BUILD_HTTP "$need_http" "$@")" need_flashlight="$(cmake_bool_override NEMO_SPEECH_WITH_FLASHLIGHT "$need_flashlight" "$@")" need_ja="$(cmake_bool_override NEMO_SPEECH_TTS_WITH_JA "$need_ja" "$@")" @@ -138,11 +113,8 @@ require_submodule() { # require_submodule PATH SENTINEL if [ "$need_http" = ON ]; then require_submodule third_party/cpp-httplib httplib.h fi -if [ "$need_nmt" = ON ] || [ "$need_s2s" = ON ]; then - require_submodule llama.cpp CMakeLists.txt -elif [ "$need_asr" = ON ]; then - require_submodule llama.cpp vendor/miniaudio/miniaudio.h -fi +# llama.cpp also provides ggml, so every build needs it. +require_submodule llama.cpp ggml/CMakeLists.txt if [ "$need_grpc" = ON ]; then require_submodule proto/riva-common LICENSE fi @@ -166,14 +138,4 @@ if [ "${#missing[@]}" -ne 0 ]; then exit 1 fi -case "$PRESET" in - cuda-*|metal-*) - scripts/apply-ggml-patches.sh - ;; -esac - -if [ "$need_nmt" = ON ] || [ "$need_s2s" = ON ]; then - scripts/apply-llama-patches.sh -fi - cmake --preset "$PRESET" "$@" diff --git a/scripts/llama-patches.sh b/scripts/llama-patches.sh new file mode 100755 index 0000000..28f932e --- /dev/null +++ b/scripts/llama-patches.sh @@ -0,0 +1,257 @@ +#!/usr/bin/env bash +# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-License-Identifier: Apache-2.0 +# Maintain the llama.cpp/ggml patch series in patches/. +# +# Builds apply the series automatically (cmake/llama_cpp.cmake); this script is +# only for changing it. It turns the series into one commit per patch inside a +# scratch worktree of the llama.cpp submodule, where it can be edited with plain +# git, and writes it back to patches/. Those scratch commits never leave the +# worktree; what you commit in this repository is patches/series and the +# patches/*.patch files it lists, in apply order. +set -euo pipefail + +usage() { + cat <<'EOF' +Usage: scripts/llama-patches.sh COMMAND + + edit Create .deps/llama.cpp-patches: the pinned llama.cpp with one commit + per patch. Edit and build against it with + -DNEMO_SPEECH_LLAMA_CPP_SOURCE_DIR=$PWD/.deps/llama.cpp-patches + and fold changes into a patch with + git commit --fixup= && git rebase -i --autosquash + export Write the worktree's commits back to patches/. + rebase REF Move the series to another llama.cpp commit or tag and pin the + submodule there. Resolve any conflicts with git in the worktree, + then run export. + check Verify that the series applies to the pinned commit and that + exporting it reproduces patches/ byte for byte. + diff [REV] Show a range-diff of patches/ against the series at REV (default + HEAD), for reviewing a change to the series. + done Remove the scratch worktree (refuses if it has unexported changes). +EOF +} + +ROOT="$(cd "$(dirname "${BASH_SOURCE[0]}")/.." && pwd)" +LLAMA="${ROOT}/llama.cpp" +PATCHES="${ROOT}/patches" +WORK="${ROOT}/.deps/llama.cpp-patches" + +# Identity for the scratch commits. It never appears in patches/. +export GIT_COMMITTER_NAME="llama-patches" GIT_COMMITTER_EMAIL="llama-patches@localhost" + +die() { + echo "llama-patches: $*" >&2 + exit 1 +} + +llama() { + git -C "${LLAMA}" "$@" +} + +pin() { + llama rev-parse HEAD +} + +# series_files DIR: the patches DIR/series lists, in apply order. Every *.patch in +# DIR must be listed, so a new patch cannot be silently left out. +series_files() { + local dir="$1" name path listed=" " + [ -f "${dir}/series" ] || die "missing ${dir}/series" + while IFS= read -r name || [ -n "${name}" ]; do + name="${name%%#*}" + name="$(printf '%s' "${name}" | tr -d '[:space:]')" + [ -n "${name}" ] || continue + [ -f "${dir}/${name}" ] || die "${dir}/series lists ${name}, which does not exist" + listed="${listed}${name} " + printf '%s\n' "${dir}/${name}" + done < "${dir}/series" + for path in "${dir}"/*.patch; do + [ -e "${path}" ] || continue + case "${listed}" in + *" $(basename "${path}") "*) ;; + *) die "${path} is not listed in ${dir}/series" ;; + esac + done +} + +# add_series DIR BASE [PATCH_DIR]: a detached worktree at BASE with the series as commits. +add_series() { + local dir="$1" base="$2" patch_dir="${3:-${PATCHES}}" path listing + local -a series=() + listing="$(series_files "${patch_dir}")" + while IFS= read -r path; do + [ -n "${path}" ] && series+=("${path}") + done <<< "${listing}" + llama worktree add --quiet --detach "${dir}" "${base}" + if [ "${#series[@]}" -ne 0 ] && ! git -C "${dir}" am --quiet --keep "${series[@]}"; then + die "the series does not apply to ${base}; see ${dir} (git am --show-current-patch)" + fi +} + +# write_series DIR BASE OUT: one .patch per commit plus OUT/series, with +# every format-patch option that could vary pinned. Each header keeps only From: and +# Subject: (unfolded onto one line). +write_series() { + local dir="$1" base="$2" out="$3" tmp path name listed=" " + git -C "${dir}" merge-base --is-ancestor "${base}" HEAD \ + || die "${dir} is not based on the pinned llama.cpp commit ${base}" + tmp="$(mktemp -d)" + git -C "${dir}" \ + -c diff.noprefix=false -c diff.mnemonicPrefix=false -c diff.relative=false \ + -c diff.algorithm=myers -c diff.indentHeuristic=true -c diff.renames=true \ + -c format.signature= -c format.thread=false -c format.notes=false \ + -c format.coverLetter=false -c format.from=false -c format.useAutoBase=false \ + -c format.suffix=.patch -c format.filenameMaxLength=100 \ + format-patch --quiet --zero-commit --no-signature --keep-subject --no-stat \ + --full-index --no-base -o "${tmp}" "${base}..HEAD" + mkdir -p "${out}" + printf '%s\n' "# Apply order for the patches in this directory (scripts/llama-patches.sh export)." \ + > "${tmp}/series" + for path in "${tmp}"/*.patch; do + [ -e "${path}" ] || continue + name="$(basename "${path}")" + name="$(printf '%s' "${name#[0-9][0-9][0-9][0-9]-}" | tr '[:upper:]' '[:lower:]')" + case "${listed}" in + *" ${name} "*) rm -rf "${tmp}"; die "two patches would be named ${name}; change one subject" ;; + esac + listed="${listed}${name} " + trim_header < "${path}" > "${tmp}/${name}" + rm "${path}" + printf '%s\n' "${name}" >> "${tmp}/series" + done + for path in "${out}"/*.patch; do + [ -e "${path}" ] || continue + case "${listed}" in + *" $(basename "${path}") "*) ;; + *) rm -f "${path}" ;; + esac + done + # Rewrite only what changed, so unchanged patch files keep their history. + for path in "${tmp}"/*.patch "${tmp}/series"; do + [ -e "${path}" ] || continue + cmp -s "${path}" "${out}/$(basename "${path}")" || cp "${path}" "${out}/" + done + rm -rf "${tmp}" +} + +# trim_header: drop the mbox separator and Date: from a format-patch header and +# unfold its continuation lines. +trim_header() { + awk ' + NR == 1 && /^From [0-9a-f]+ Mon Sep 17 00:00:00 2001$/ { next } + body { print; next } + /^[ \t]/ { held = held $0; next } + { if (held != "") print held; held = "" } + /^$/ { body = 1; print; next } + /^Date: / { next } + { held = $0 } + END { if (held != "") print held } + ' +} + +subjects() { + awk 'FNR == 1 { header = 1 } /^$/ { header = 0 } header && sub(/^Subject: /, "")' \ + "$1"/*.patch 2>/dev/null | sort +} + +scratch_trees=() +cleanup() { + local tree + for tree in "${scratch_trees[@]+"${scratch_trees[@]}"}"; do + llama worktree remove --force "${tree}" >/dev/null 2>&1 || true + rm -rf "$(dirname "${tree}")" + done +} +trap cleanup EXIT + +# scratch_series VAR BASE [PATCH_DIR]: a temporary series worktree, removed on exit. +scratch_series() { + local scratch_tree + scratch_tree="$(mktemp -d)/tree" + scratch_trees+=("${scratch_tree}") + add_series "${scratch_tree}" "$2" "${3:-${PATCHES}}" + printf -v "$1" '%s' "${scratch_tree}" +} + +[ -e "${LLAMA}/.git" ] || die "llama.cpp submodule is not initialized; run: git submodule update --init llama.cpp" + +case "${1:-}" in + edit) + [ ! -e "${WORK}" ] || die "${WORK} already exists; run export or done first" + mkdir -p "$(dirname "${WORK}")" + add_series "${WORK}" "$(pin)" + echo "Series applied as commits in ${WORK}." + # Git Bash on Windows: native CMake needs C:/... rather than /c/... + work_native="${WORK}" + if command -v cygpath >/dev/null 2>&1; then + work_native="$(cygpath -m "${WORK}")" + fi + echo "Build against it: cmake ... -DNEMO_SPEECH_LLAMA_CPP_SOURCE_DIR=${work_native}" + echo "When done: scripts/llama-patches.sh export" + ;; + export) + [ -d "${WORK}" ] || die "no worktree at ${WORK}; run edit first" + before="$(subjects "${PATCHES}")" + write_series "${WORK}" "$(pin)" "${PATCHES}" + after="$(subjects "${PATCHES}")" + echo "Wrote $(ls "${PATCHES}"/*.patch | wc -l | tr -d ' ') patches and patches/series." + comm -23 <(printf '%s\n' "${before}") <(printf '%s\n' "${after}") | sed 's/^/ removed: /' + comm -13 <(printf '%s\n' "${before}") <(printf '%s\n' "${after}") | sed 's/^/ added: /' + ;; + rebase) + [ -n "${2:-}" ] || die "usage: scripts/llama-patches.sh rebase REF" + old="$(pin)" + if ! new="$(llama rev-parse --verify --quiet "${2}^{commit}")"; then + llama fetch --quiet origin "${2}" + new="$(llama rev-parse FETCH_HEAD)" + fi + [ -d "${WORK}" ] || add_series "${WORK}" "${old}" + # Pin first, so that export uses the new base once conflicts are resolved. + llama checkout --quiet --detach "${new}" + echo "llama.cpp pinned at ${new}; rebasing the series in ${WORK}." + if ! git -C "${WORK}" rebase --quiet --empty=drop --onto "${new}" "${old}"; then + echo "Resolve the conflicts in ${WORK} with git (rebase --continue), then run export." >&2 + exit 1 + fi + echo "Rebased cleanly. Build and test, then run export." + ;; + check) + scratch_series tree "$(pin)" + write_series "${tree}" "$(pin)" "${tree}.out" + if ! diff -r "${PATCHES}" "${tree}.out" --exclude='*.md'; then + die "patches/ is not in exported form; run edit, export and commit the result" + fi + echo "patches/ applies to llama.cpp $(pin) and round-trips byte for byte." + ;; + diff) + rev="${2:-HEAD}" + git -C "${ROOT}" cat-file -e "${rev}:patches" 2>/dev/null || die "${rev} has no patches/ directory" + old_base="$(git -C "${ROOT}" rev-parse "${rev}:llama.cpp")" + llama cat-file -e "${old_base}^{commit}" 2>/dev/null || llama fetch --quiet origin "${old_base}" + old_patches="$(mktemp -d)" + scratch_trees+=("${old_patches}/none") + git -C "${ROOT}" archive "${rev}" patches | tar -x -C "${old_patches}" --strip-components=1 + scratch_series old "${old_base}" "${old_patches}" + scratch_series new "$(pin)" + llama range-diff "${old_base}..$(git -C "${old}" rev-parse HEAD)" "$(pin)..$(git -C "${new}" rev-parse HEAD)" + ;; + done) + [ -d "${WORK}" ] || exit 0 + if [ "${2:-}" != "--force" ]; then + out="$(mktemp -d)" + scratch_trees+=("${out}/none") + write_series "${WORK}" "$(pin)" "${out}" + diff -rq "${PATCHES}" "${out}" --exclude='*.md' >/dev/null \ + || die "${WORK} has changes that are not exported; run export, or done --force" + fi + llama worktree remove --force "${WORK}" + ;; + -h|--help|help|"") + usage + ;; + *) + usage >&2 + exit 2 + ;; +esac diff --git a/scripts/patch-series-common.sh b/scripts/patch-series-common.sh deleted file mode 100644 index 35b8989..0000000 --- a/scripts/patch-series-common.sh +++ /dev/null @@ -1,54 +0,0 @@ -#!/usr/bin/env bash -# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. -# SPDX-License-Identifier: Apache-2.0 - -# Return success when a source tree without .git contains the complete patch -# series. The reverse applications operate only on a temporary index. -patch_series_applied_without_git() ( - local source_tree="$1" - local patch_dir="$2" - local log_prefix="$3" - local plain_git_dir plain_index series_applied path i - local -a patch_files patched_paths current_paths - - shopt -s nullglob - patch_files=("${patch_dir}"/*.patch) - PATCH_SERIES_TEMP_DIR="$(mktemp -d)" - plain_git_dir="${PATCH_SERIES_TEMP_DIR}/plain.git" - plain_index="${PATCH_SERIES_TEMP_DIR}/plain.index" - trap 'rm -rf -- "${PATCH_SERIES_TEMP_DIR}"' EXIT - - git init --bare --quiet "${plain_git_dir}" - GIT_INDEX_FILE="${plain_index}" \ - git --git-dir="${plain_git_dir}" --work-tree="${source_tree}" read-tree --empty - patched_paths=() - while IFS= read -r path; do - patched_paths+=("${path}") - done < <(sed -n -e 's|^--- a/||p' -e 's|^+++ b/||p' "${patch_files[@]}" | sort -u) - current_paths=() - for path in "${patched_paths[@]}"; do - if [ -e "${source_tree}/${path}" ] || [ -L "${source_tree}/${path}" ]; then - current_paths+=("${path}") - fi - done - if [ "${#current_paths[@]}" -ne 0 ]; then - GIT_INDEX_FILE="${plain_index}" \ - git --git-dir="${plain_git_dir}" --work-tree="${source_tree}" \ - add -- "${current_paths[@]}" - fi - - series_applied=ON - for ((i = ${#patch_files[@]} - 1; i >= 0; --i)); do - if ! GIT_INDEX_FILE="${plain_index}" \ - git --git-dir="${plain_git_dir}" --work-tree="${source_tree}" \ - apply --cached --reverse "${patch_files[i]}" >/dev/null 2>&1; then - series_applied=OFF - break - fi - done - if [ "${series_applied}" = ON ]; then - echo "${log_prefix} current series already applied" - return 0 - fi - return 1 -) diff --git a/scripts/windows/apply-ggml-patches.ps1 b/scripts/windows/apply-ggml-patches.ps1 deleted file mode 100644 index 8501100..0000000 --- a/scripts/windows/apply-ggml-patches.ps1 +++ /dev/null @@ -1,157 +0,0 @@ -# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. -# SPDX-License-Identifier: Apache-2.0 -<# -.SYNOPSIS - Apply the in-tree ggml patches onto the vendored ggml submodule (Windows). - -.DESCRIPTION - PowerShell port of scripts/apply-ggml-patches.sh. Applies ggml-patches/*.patch - (filename order) via `git apply`, keeping the submodule pinned to clean - upstream. The patches are CUDA-only, so they are only needed for a CUDA build - (-DGGML_CUDA=ON with the default -DNEMO_SPEECH_GGML_PATCHED=ON). - - Patches that still reverse-apply cleanly are skipped. Apply the complete - series to a clean submodule for deterministic setup; later patches may - refine lines introduced by earlier patches, making reverse detection - ambiguous on a fully patched tree. Each patch is normalized to LF before - `git apply` - on Windows with core.autocrlf=true the .patch files - check out CRLF, and git apply rejects some hunks ("corrupt patch") on mixed - endings (notably 0007). Normalizing makes it work regardless of checkout EOL. - - Control flow is driven off $LASTEXITCODE rather than $ErrorActionPreference: - `git apply --reverse --check` writes to stderr on the expected "not applied - yet" path, which under -ErrorAction Stop would otherwise abort the script. -#> -[CmdletBinding()] -param() - -$ErrorActionPreference = 'Continue' - -$Root = Split-Path -Parent (Split-Path -Parent $PSScriptRoot) # scripts\windows\..\.. -$Ggml = Join-Path $Root 'ggml' -$Patches = Join-Path $Root 'ggml-patches' - -if (-not (Test-Path (Join-Path $Ggml 'src'))) { - throw "ggml submodule not initialized at $Ggml. Run: git submodule update --init ggml" -} -if (-not (Test-Path -LiteralPath $Patches -PathType Container)) { - throw "ggml-patches directory not found: $Patches" -} - -# Run `git apply ` against an LF-normalized copy of the patch. -# Returns git's exit code; stderr is swallowed (callers decide off the code). -function Invoke-GitApply { - param([string[]] $ApplyArgs, [string] $PatchFile) - $tmp = [System.IO.Path]::GetTempFileName() - try { - # Strip ALL CR bytes (same as `tr -d '\r'`), not just CRLF pairs: a - # worktree checked out before .gitattributes landed can carry stray CRs - # that git apply rejects as "corrupt patch". - $content = [System.IO.File]::ReadAllText($PatchFile) -replace "`r", "" - [System.IO.File]::WriteAllText($tmp, $content) # UTF-8, no BOM - & git -C $Ggml apply @ApplyArgs $tmp 2>&1 | Out-Null - return $LASTEXITCODE - } - finally { - Remove-Item $tmp -ErrorAction SilentlyContinue - } -} - -# Enumerate up front with -ErrorAction Stop (so a read failure isn't swallowed by -# the 'Continue' preference above) and fail if there are no patches to apply. -$patchFiles = @(Get-ChildItem -LiteralPath $Patches -Filter '*.patch' -File -ErrorAction Stop | - Sort-Object Name) -if ($patchFiles.Count -eq 0) { - throw "no .patch files found in $Patches" -} - -# Later patches may modify hunks introduced by earlier patches, so individual -# reverse checks cannot always recognize a fully applied series. Build the -# expected tree in a temporary index and compare it with the patched paths in -# the worktree. The caller's real Git index is never modified. -function Test-PatchSeriesApplied { - param([System.IO.FileInfo[]] $PatchFiles) - - & git -C $Ggml rev-parse --git-dir 2>&1 | Out-Null - if ($LASTEXITCODE -ne 0) { - return $false - } - - $tmpDir = Join-Path ([System.IO.Path]::GetTempPath()) ([Guid]::NewGuid().ToString('N')) - [System.IO.Directory]::CreateDirectory($tmpDir) | Out-Null - $expectedIndex = Join-Path $tmpDir 'expected.index' - $currentIndex = Join-Path $tmpDir 'current.index' - $previousIndex = $env:GIT_INDEX_FILE - - try { - $env:GIT_INDEX_FILE = $expectedIndex - & git -C $Ggml read-tree HEAD 2>&1 | Out-Null - if ($LASTEXITCODE -ne 0) { - throw 'failed to initialize temporary ggml index' - } - foreach ($patch in $PatchFiles) { - if ((Invoke-GitApply @('--cached') $patch.FullName) -ne 0) { - throw "$($patch.Name) does not apply to the pinned ggml commit" - } - } - $paths = @(& git -C $Ggml diff --cached --name-only HEAD) - $expectedTree = (& git -C $Ggml write-tree).Trim() - - $env:GIT_INDEX_FILE = $currentIndex - & git -C $Ggml read-tree HEAD 2>&1 | Out-Null - if ($LASTEXITCODE -ne 0) { - throw 'failed to initialize temporary current ggml index' - } - $currentPaths = @() - foreach ($path in $paths) { - & git -C $Ggml cat-file -e "HEAD:$path" 2>&1 | Out-Null - if ((Test-Path -LiteralPath (Join-Path $Ggml $path)) -or $LASTEXITCODE -eq 0) { - $currentPaths += $path - } - } - if ($currentPaths.Count -ne 0) { - & git -C $Ggml add -A -- @currentPaths 2>&1 | Out-Null - if ($LASTEXITCODE -ne 0) { - throw 'failed to populate temporary current ggml index' - } - } - $currentTree = (& git -C $Ggml write-tree).Trim() - return $currentTree -eq $expectedTree - } - finally { - if ($null -eq $previousIndex) { - Remove-Item Env:GIT_INDEX_FILE -ErrorAction SilentlyContinue - } - else { - $env:GIT_INDEX_FILE = $previousIndex - } - Remove-Item -LiteralPath $tmpDir -Recurse -Force -ErrorAction SilentlyContinue - } -} - -if (Test-PatchSeriesApplied $patchFiles) { - Write-Host '[ggml-patch] current series already applied' - Write-Host '[ggml-patch] done' - exit 0 -} - -$applied = 0 -$skipped = 0 -foreach ($patch in $patchFiles) { - $name = $patch.Name - if ((Invoke-GitApply @('--reverse', '--check') $patch.FullName) -eq 0) { - Write-Host "[ggml-patch] ${name}: already applied" - $skipped++ - continue - } - if ((Invoke-GitApply @('--check') $patch.FullName) -ne 0) { - throw "[ggml-patch] ${name}: does NOT apply cleanly to current ggml (the tree is modified or contains a stale patch series; restore the pinned ggml submodule, then retry)" - } - if ((Invoke-GitApply @() $patch.FullName) -ne 0) { - throw "[ggml-patch] ${name}: git apply failed" - } - Write-Host "[ggml-patch] ${name}: applied" - $applied++ -} - -Write-Host "[ggml-patch] done (applied $applied, skipped $skipped)" diff --git a/scripts/windows/build.ps1 b/scripts/windows/build.ps1 index dd01216..07a7bfa 100644 --- a/scripts/windows/build.ps1 +++ b/scripts/windows/build.ps1 @@ -12,8 +12,8 @@ find the toolset. MSVC is required: nvcc on Windows only supports cl.exe as the CUDA host compiler. 3. Provisions required C++ dependencies. - 4. For a CUDA build, applies the CUDA-only ggml patches. - 5. Configures with CMake (Ninja) and builds. + 4. Initializes the required submodules. + 5. Configures with CMake (Ninja), which applies patches/ for CUDA and CPU builds, and builds. .PARAMETER Backend cuda | vulkan | cpu. CUDA and Vulkan are separate build trees (different ggml @@ -342,12 +342,8 @@ function Initialize-RequiredSubmodule { } } -Initialize-RequiredSubmodule 'ggml' 'CMakeLists.txt' -if ($BuildNmt) { - Initialize-RequiredSubmodule 'llama.cpp' 'CMakeLists.txt' -} elseif ($BuildAsr) { - Initialize-RequiredSubmodule 'llama.cpp' 'vendor\miniaudio\miniaudio.h' -} +# llama.cpp also provides ggml, so every build needs it. +Initialize-RequiredSubmodule 'llama.cpp' 'ggml\CMakeLists.txt' if ($BuildHttp) { Initialize-RequiredSubmodule 'third_party\cpp-httplib' 'httplib.h' } if ($BuildGrpc) { Initialize-RequiredSubmodule 'proto\riva-common' 'LICENSE' } if ($BuildFlashlight) { @@ -375,13 +371,7 @@ if ($BuildTtsZh) { } } -# --- 5. CUDA-only: apply the ggml patches --------------------------------------- -if ($Backend -eq 'cuda') { - Write-Host "==> applying ggml patches (CUDA)" - & (Join-Path $PSScriptRoot 'apply-ggml-patches.ps1') -} - -# --- 6. Configure + build ------------------------------------------------------- +# --- 5. Configure + build ------------------------------------------------------- function ConvertTo-CMakeBool([bool]$Value) { if ($Value) { return 'ON' } return 'OFF' @@ -393,11 +383,9 @@ $cmakeArgs = @( "-DNEMO_SPEECH_BUILD_DIAR=$(ConvertTo-CMakeBool $BuildDiar)", "-DNEMO_SPEECH_BUILD_TTS=$(ConvertTo-CMakeBool $BuildTts)", "-DNEMO_SPEECH_BUILD_NMT=$(ConvertTo-CMakeBool $BuildNmt)", - "-DNEMO_SPEECH_WITH_NMT=$(ConvertTo-CMakeBool $BuildNmt)", "-DNEMO_SPEECH_BUILD_HTTP=$(ConvertTo-CMakeBool $BuildHttp)", "-DNEMO_SPEECH_HTTP_TLS=$(ConvertTo-CMakeBool $BuildHttpTls)", "-DNEMO_SPEECH_BUILD_GRPC=$(ConvertTo-CMakeBool $BuildGrpc)", - "-DNEMO_SPEECH_WITH_GRPC=$(ConvertTo-CMakeBool $BuildGrpc)", "-DNEMO_SPEECH_WITH_FLASHLIGHT=$(ConvertTo-CMakeBool $BuildFlashlight)", '-DNEMO_SPEECH_WITH_NORM=OFF', "-DNEMO_SPEECH_TTS_WITH_JA=$(ConvertTo-CMakeBool $BuildTtsJa)", @@ -452,7 +440,6 @@ switch ($Backend) { 'cpu' { $cmakeArgs += '-DGGML_CUDA=OFF' $cmakeArgs += '-DGGML_VULKAN=OFF' - $cmakeArgs += '-DNEMO_SPEECH_GGML_PATCHED=OFF' } } diff --git a/src/asr/CMakeLists.txt b/src/asr/CMakeLists.txt index 5026575..e140856 100644 --- a/src/asr/CMakeLists.txt +++ b/src/asr/CMakeLists.txt @@ -20,10 +20,6 @@ list(REMOVE_ITEM ASR_SOURCES "${CMAKE_CURRENT_SOURCE_DIR}/c_api.cpp") add_library(nemo_speech_asr SHARED ${ASR_SOURCES}) -if(NEMO_SPEECH_GGML_PATCHED) - target_compile_definitions(nemo_speech_asr PRIVATE NEMO_SPEECH_GGML_PATCHED=1) -endif() - # Windows portability: this lib relies on Linux default symbol visibility to # export its public API (AsrModel, runners, read_wav_mono_16k). MSVC/clang-cl # export nothing from a DLL unless annotated with __declspec(dllexport), so the @@ -147,22 +143,6 @@ if(NEMO_SPEECH_WITH_FLASHLIGHT) target_link_libraries(nemo_speech_asr PRIVATE flashlight-text flashlight-text-kenlm) endif() -if(NEMO_SPEECH_FUSED_RELPOS_ATTN) - target_compile_definitions(nemo_speech_asr PRIVATE NEMO_SPEECH_FUSED_RELPOS_ATTN=1) -endif() - -if(NEMO_SPEECH_DIRECT_DW_CONV) - # The encoder's channels-inner conv-module path uses the cwhn direct - # depthwise kernel (F16-safe only on the patched CUDA backend, like the - # nn.cpp Conv1D fast path this flag already gates). - target_compile_definitions(nemo_speech_asr PRIVATE NEMO_SPEECH_DIRECT_DW_CONV=1) -endif() - -if(NEMO_SPEECH_FASTCONFORMER_CUDA_FUSIONS) - target_compile_definitions( - nemo_speech_asr PRIVATE NEMO_SPEECH_FASTCONFORMER_CUDA_FUSIONS=1) -endif() - # --------------------------------------------------------------------------- # Stable C ABI shared library (nemo_speech_asr_*). Hidden visibility + a version script # so only the nemo_speech_asr_* symbols are exported; the C++ implementation is linked diff --git a/src/asr/batching.h b/src/asr/batching.h index be2dde9..50a724e 100644 --- a/src/asr/batching.h +++ b/src/asr/batching.h @@ -7,7 +7,6 @@ #include #include -#include #include #include #include @@ -21,7 +20,6 @@ #include #include #include -#include #include #include #include @@ -133,27 +131,7 @@ class IngressBatchCoordinator { public: explicit IngressBatchCoordinator(const BatchingConfig& config) : enabled_(config.enabled), delay_us_(std::max(0, config.ingress_cohort_delay_us)), - max_batch_size_(std::max(1, config.max_batch_size)) { - auto parse_env_int = [](const char* name, int& parsed) { - const char* value = std::getenv(name); - if (value == nullptr) - return false; - const std::string_view text(value); - if (text.empty()) - return false; - int candidate = 0; - const auto result = std::from_chars(text.data(), text.data() + text.size(), candidate); - if (result.ec != std::errc{} || result.ptr != text.data() + text.size()) - return false; - parsed = candidate; - return true; - }; - int value = 0; - if (parse_env_int("NEMO_SPEECH_INGRESS_COHORT", value)) - enabled_ = enabled_ && value != 0; - if (parse_env_int("NEMO_SPEECH_INGRESS_COHORT_US", value)) - delay_us_ = std::max(0, value); - } + max_batch_size_(std::max(1, config.max_batch_size)) {} int arrive(int expected_participants = 0) { if (!enabled_) diff --git a/src/asr/encoder/cache_aware_encoder.cpp b/src/asr/encoder/cache_aware_encoder.cpp index 6ab8f00..dc49f03 100644 --- a/src/asr/encoder/cache_aware_encoder.cpp +++ b/src/asr/encoder/cache_aware_encoder.cpp @@ -265,7 +265,7 @@ CacheAwareEncoder::ensure_session() { session_->setup(); // declares the indexed K/V/conv cache arenas slots_used_.assign(static_cast(arena_slots_), false); slots_need_reset_.assign(static_cast(arena_slots_), true); -#ifdef NEMO_SPEECH_FUSED_RELPOS_ATTN +#ifdef NEMO_SPEECH_CUDA_FAST_PATHS const int d_k = cfg_.d_model / cfg_.n_heads; ring_cache_enabled_ = session_->params.use_gpu && cfg_.cache_left_ctx > 0 && cfg_.cache_chunk_frames > 0 && (d_k & (d_k - 1)) == 0; diff --git a/src/asr/encoder/fastconformer.cpp b/src/asr/encoder/fastconformer.cpp index 6aeb1a5..08e9497 100644 --- a/src/asr/encoder/fastconformer.cpp +++ b/src/asr/encoder/fastconformer.cpp @@ -22,7 +22,7 @@ static inline ggml_tensor* pad_ext_backend( ggml_context* ctx, ggml_tensor* t, int lp0, int rp0, int lp1, int rp1, int lp2, int rp2, int lp3, int rp3) { -#ifdef NEMO_SPEECH_FASTCONFORMER_CUDA_FUSIONS +#ifdef NEMO_SPEECH_CUDA_FAST_PATHS return ggml_pad_ext(ctx, t, lp0, rp0, lp1, rp1, lp2, rp2, lp3, rp3); #else ggml_tensor* right_padded = @@ -452,7 +452,7 @@ ConformerConv::build_graph( : ggml_add_inplace(bf_ctx.ctx, x_ct, b1.tensor); } ggml_runtime::TensorBag out_bag; -#ifdef NEMO_SPEECH_FASTCONFORMER_CUDA_FUSIONS +#ifdef NEMO_SPEECH_CUDA_FAST_PATHS if (session->params.use_gpu) { // Split contiguous (2*C,T,B) in the CUDA GLU kernel and materialize // only the final (T,C,B) depthwise-convolution layout. @@ -919,7 +919,7 @@ ConformerLayer::build_mha_cached( const float scale = 1.0f / std::sqrt(static_cast(d_k)); ggml_tensor* ctx_attn = nullptr; -#ifdef NEMO_SPEECH_FUSED_RELPOS_ATTN +#ifdef NEMO_SPEECH_CUDA_FAST_PATHS // Single-kernel path: bias adds, content+position scores, the rel-shift, // scale+mask, softmax and the attn@V context all happen inside // ggml_fused_relpos_attn (the same op the offline encoder uses; its @@ -1439,7 +1439,7 @@ FastConformerEncoder::build_graph( // gathers degenerate to arena views and feedback uses in-place copies. // Multi-slot arenas retain indexed gather/scatter. const bool single_slot = cfg_.cache_state_slots == 1 && batch == 1; -#ifdef NEMO_SPEECH_FUSED_RELPOS_ATTN +#ifdef NEMO_SPEECH_CUDA_FAST_PATHS const int d_k = cfg_.d_model / cfg_.n_heads; const bool direct_kv_arena = session->params.use_gpu && mask_t.tensor != nullptr && (d_k & (d_k - 1)) == 0; @@ -1493,9 +1493,9 @@ FastConformerEncoder::build_graph( cache.attn_mask = mask_t.tensor; cache.pos_proj = session->model_tensor_container->get_tensor_by_name(pos_proj_name(l)).tensor; -#ifdef NEMO_SPEECH_DIRECT_DW_CONV +#ifdef NEMO_SPEECH_CUDA_FAST_PATHS // Repacked channel-inner dw weight enables the (d_model, T)-layout - // conv module; the cwhn direct kernel is CUDA-only (patch 0004), so a + // conv module; the cwhn direct kernel is CUDA-only (depthwise-conv patch), so a // CPU session in a CUDA build leaves it null and the layer takes the // portable transpose-based conv path (runtime-gated in nn.cpp). if (session->params.use_gpu) { diff --git a/src/asr/encoder/rel_pos_attention.cpp b/src/asr/encoder/rel_pos_attention.cpp index f7fceda..ada27be 100644 --- a/src/asr/encoder/rel_pos_attention.cpp +++ b/src/asr/encoder/rel_pos_attention.cpp @@ -230,7 +230,7 @@ RelPositionMultiHeadAttention::build_graph_masked( // the shared merge-heads reshape below. ggml_tensor* attn = nullptr; bool attn_heads_merged = false; -#ifdef NEMO_SPEECH_FUSED_RELPOS_ATTN +#ifdef NEMO_SPEECH_CUDA_FAST_PATHS // The fused CUDA op accepts both the streaming per-key mask and the // offline per-(key,query) band mask. This entire branch is compiled ONLY // with a patched ggml: ggml_fused_relpos_attn is a patch-only symbol, so a diff --git a/src/asr/model.cpp b/src/asr/model.cpp index a82c06c..e2bcbcd 100644 --- a/src/asr/model.cpp +++ b/src/asr/model.cpp @@ -1292,13 +1292,7 @@ AsrModel::AsrModel(ggml_runtime::BackendManager& bm, Common&& c, const BatchingC : bm_(&bm), loader_(std::move(c.loader)), ns_(std::move(c.ns)), model_name_(std::move(c.model_name)), enc_cfg_(std::move(c.enc_cfg)), fe_cfg_(std::move(c.fe_cfg)), vocab_(std::move(c.vocab)) { - // Use the GPU frontend by default; NEMO_SPEECH_STREAM_GPU_FE=0 disables it. - static const bool stream_gpu_fe = [] { - const char* e = std::getenv("NEMO_SPEECH_STREAM_GPU_FE"); - return e == nullptr || e[0] != '0'; - }(); - const bool gpu_fe = (batching.enabled && batching.max_batch_size > 1) || stream_gpu_fe; - fe_ = std::make_unique(fe_cfg_, gpu_fe ? &bm : nullptr, batching); + fe_ = std::make_unique(fe_cfg_, &bm, batching); apply_model_mel_basis(*fe_); diff --git a/src/asr/vad/silero_vad.cpp b/src/asr/vad/silero_vad.cpp index d83d161..7797454 100644 --- a/src/asr/vad/silero_vad.cpp +++ b/src/asr/vad/silero_vad.cpp @@ -13,23 +13,6 @@ namespace nemo_speech::asr { -namespace { -ggml_tensor* -conv_1d_frames( - ggml_context* ctx, ggml_tensor* weight, ggml_tensor* input, int stride, int padding) { - // Keep frame batches separate: columns are (kernel*channels, time, frame). - // Flatten only time/frame for the GEMM, then restore (time, channels, frame). - // A direct reshape of (time*frame, channels) interleaves frames and channels. - auto* columns = - ggml_im2col(ctx, weight, input, stride, 0, padding, 0, 1, 0, false, GGML_TYPE_F16); - auto* product = ggml_mul_mat( - ctx, ggml_reshape_2d(ctx, weight, weight->ne[0] * weight->ne[1], weight->ne[2]), - ggml_reshape_2d(ctx, columns, columns->ne[0], columns->ne[1] * columns->ne[2])); - product = ggml_reshape_3d(ctx, product, weight->ne[2], columns->ne[1], columns->ne[2]); - return ggml_cont(ctx, ggml_permute(ctx, product, 1, 0, 2, 3)); -} -} // namespace - // SileroVadModule - the ggml graph. Consecutive windows and streams share one // graph run, with one probability emitted per window_size frame. // @@ -118,7 +101,8 @@ class SileroVadModule : public ggml_runtime::Module { const int stft_hop = cfg_.stft_filter_length / 2; // = 128 ggml_tensor* padded = ggml_pad_reflect_1d(g, frame.tensor, cfg_.context_size, cfg_.context_size); - ggml_tensor* stft = conv_1d_frames(g, basis, padded, stft_hop, /*padding=*/0); + ggml_tensor* stft = + ggml_runtime::conv_1d(g, basis, padded, stft_hop, /*padding=*/0, /*d0=*/1); const int w = static_cast(stft->ne[0]); // STFT frames (=4) const int cutoff = cfg_.stft_n_basis / 2; // n_freqs (=129) @@ -140,7 +124,7 @@ class SileroVadModule : public ggml_runtime::Module { for (int i = 0; i < cfg_.n_encoder_layers; i++) { ggml_tensor* wt = mtc->get_tensor_by_name(enc_w(i)).tensor; ggml_tensor* b = mtc->get_tensor_by_name(enc_b(i)).tensor; - cur = conv_1d_frames(g, wt, cur, cfg_.enc_strides[i], /*padding=*/1); + cur = ggml_runtime::conv_1d(g, wt, cur, cfg_.enc_strides[i], /*padding=*/1, /*d0=*/1); cur = ggml_add(g, cur, ggml_reshape_3d(g, b, 1, cfg_.enc_out_channels[i], 1)); cur = ggml_relu(g, cur); } diff --git a/src/core/CMakeLists.txt b/src/core/CMakeLists.txt index cd40a3f..7bcb585 100644 --- a/src/core/CMakeLists.txt +++ b/src/core/CMakeLists.txt @@ -3,7 +3,7 @@ add_library(nemo_speech_engine_registry STATIC engine_registry.cpp) target_include_directories(nemo_speech_engine_registry PUBLIC ${CMAKE_CURRENT_SOURCE_DIR}) -target_link_libraries(nemo_speech_engine_registry PUBLIC nemo_speech_common) +target_link_libraries(nemo_speech_engine_registry PUBLIC nemo_speech_common nemo_speech_runtime_ggml) if(NEMO_SPEECH_BUILD_ASR OR NEMO_SPEECH_BUILD_DIAR) target_link_libraries(nemo_speech_engine_registry PUBLIC nemo_speech_asr) diff --git a/src/core/engine_registry.cpp b/src/core/engine_registry.cpp index 58e7525..7de5907 100644 --- a/src/core/engine_registry.cpp +++ b/src/core/engine_registry.cpp @@ -3,34 +3,17 @@ #include "engine_registry.h" -#include #include #include -namespace nemo_speech { -namespace { - -#if defined(NEMO_SPEECH_REGISTRY_TTS) || defined(NEMO_SPEECH_REGISTRY_NMT) -void -set_environment_default(const char* name, const char* value) { - if (std::getenv(name)) - return; -#if defined(_WIN32) - if (_putenv_s(name, value) != 0) - throw std::runtime_error(std::string("could not set process default ") + name); -#else - if (setenv(name, value, 0) != 0) - throw std::runtime_error(std::string("could not set process default ") + name); -#endif -} -#endif +#include "runtime.h" -} // namespace +namespace nemo_speech { EngineRegistry::EngineRegistry(EngineRegistryConfig config) { #if defined(NEMO_SPEECH_REGISTRY_NMT) - if (config.nmt) - set_environment_default(config.asr ? "GGML_SKINNY_Q8_INPLACE" : "GGML_SKINNY_Q8", "0"); + if (config.nmt && config.asr) + ggml_runtime::keep_skinny_q8_weights_separate(); #else (void)config; #endif @@ -91,7 +74,7 @@ EngineRegistry::asr() const { #if defined(NEMO_SPEECH_REGISTRY_TTS) std::shared_ptr EngineRegistry::load_tts(tts::SynthesizerConfig config) { - set_environment_default("GGML_CUDA_GRAPH_EVICT_AFTER_MS", "0"); + ggml_runtime::keep_cuda_graphs_resident(); auto engine = std::make_shared(std::move(config)); std::lock_guard lock(mutex_); tts_ = engine; @@ -111,17 +94,6 @@ EngineRegistry::tts() const { #if defined(NEMO_SPEECH_REGISTRY_NMT) std::shared_ptr EngineRegistry::load_nmt(nmt::TranslatorConfig config) { -#if defined(NEMO_SPEECH_REGISTRY_ASR) - { - std::lock_guard lock(mutex_); - if (asr_) - set_environment_default("GGML_SKINNY_Q8_INPLACE", "0"); - else - set_environment_default("GGML_SKINNY_Q8", "0"); - } -#else - set_environment_default("GGML_SKINNY_Q8", "0"); -#endif auto engine = std::make_shared(std::move(config)); std::lock_guard lock(mutex_); nmt_ = engine; diff --git a/src/nmt/translator.cpp b/src/nmt/translator.cpp index 8ece70f..8234d32 100644 --- a/src/nmt/translator.cpp +++ b/src/nmt/translator.cpp @@ -29,24 +29,6 @@ ensure_backend() { std::call_once(once, [] { llama_backend_init(); }); } -// NMT-only entry points (test_nmt, the nemo_speech_nmt_* C ABI, direct lib use). The -// skinny-q8 ggml kernel (see ggml-patches/0005) is an ASR-encoder optimization -// whose in-place repack is unsafe for the decoder and whose padded single-token -// decode is slower than the stock path; with no ASR in the process, disable it -// entirely. The flag is read once, on the first skinny repack, so this runs in -// the Translator ctor, before any decode. putenv (not setenv) keeps it portable -// with no _WIN32 branch; the string must outlive the call, hence static. Defer -// to the server (it sets one of these for the combined ASR+NMT case): act only -// when neither is already set. -void -force_skinny_q8_safe_for_nmt() { - if (std::getenv("GGML_SKINNY_Q8") == nullptr && - std::getenv("GGML_SKINNY_Q8_INPLACE") == nullptr) { - static char kv[] = "GGML_SKINNY_Q8=0"; - putenv(kv); - } -} - std::string path_stem(const std::string& path) { const std::size_t slash = path.find_last_of("/\\"); @@ -169,7 +151,6 @@ struct Translator::Impl { }; Translator::Translator(TranslatorConfig cfg) : impl_(std::make_unique()) { - force_skinny_q8_safe_for_nmt(); configure_llama_logging(cfg.verbose); ensure_backend(); impl_->cfg = std::move(cfg); diff --git a/src/runtime/ggml/CMakeLists.txt b/src/runtime/ggml/CMakeLists.txt index 5f04650..9bcb4d7 100644 --- a/src/runtime/ggml/CMakeLists.txt +++ b/src/runtime/ggml/CMakeLists.txt @@ -20,8 +20,6 @@ set_target_properties(nemo_speech_runtime_ggml PROPERTIES target_include_directories(nemo_speech_runtime_ggml PUBLIC ${CMAKE_CURRENT_SOURCE_DIR} - ${CMAKE_SOURCE_DIR}/ggml/include - ${CMAKE_SOURCE_DIR}/ggml/src ) target_include_directories(nemo_speech_runtime_ggml PRIVATE ${CMAKE_SOURCE_DIR}/src/common @@ -29,14 +27,10 @@ target_include_directories(nemo_speech_runtime_ggml PRIVATE target_link_libraries(nemo_speech_runtime_ggml PUBLIC ${GGML_DEPENDENCIES}) +# Consumers use the patched operations only when patches/ was applied; runtime.h +# derives NEMO_SPEECH_CUDA_FAST_PATHS from this and GGML_USE_CUDA. if(NEMO_SPEECH_GGML_PATCHED) - target_compile_definitions(nemo_speech_runtime_ggml PRIVATE NEMO_SPEECH_GGML_PATCHED=1) -endif() - -# CUDA-only direct depthwise-conv kernel (see CMakeLists.txt option). When off, -# nn.cpp uses the portable ggml_conv_1d_dw lowering (F16-safe on all backends). -if(NEMO_SPEECH_DIRECT_DW_CONV) - target_compile_definitions(nemo_speech_runtime_ggml PRIVATE NEMO_SPEECH_DIRECT_DW_CONV=1) + target_compile_definitions(nemo_speech_runtime_ggml PUBLIC NEMO_SPEECH_GGML_PATCHED=1) endif() # ggml-cuda links CUDA::cuda_driver PRIVATE, so the stub libcuda.so on the diff --git a/src/runtime/ggml/backend.cpp b/src/runtime/ggml/backend.cpp index 8d899bd..c57fad9 100644 --- a/src/runtime/ggml/backend.cpp +++ b/src/runtime/ggml/backend.cpp @@ -13,6 +13,30 @@ namespace ggml_runtime { +namespace { +void +set_environment_default(const char* name, const char* value) { + if (std::getenv(name) != nullptr) { + return; + } +#ifdef _WIN32 + _putenv_s(name, value); +#else + setenv(name, value, 0); +#endif +} +} // namespace + +void +keep_cuda_graphs_resident() { + set_environment_default("GGML_CUDA_GRAPH_EVICT_AFTER_MS", "0"); +} + +void +keep_skinny_q8_weights_separate() { + set_environment_default("GGML_SKINNY_Q8_INPLACE", "0"); +} + BackendManager::BackendManager(Params params) { nemo_speech::common::ensure_ggml_logging(); this->params = params; @@ -30,12 +54,8 @@ void BackendManager::init_backends() { #if defined(GGML_USE_VULKAN) // Graph optimization is incompatible with in-place persistent cache tensors. - // Vulkan reads this setting during device initialization; respect user overrides. - // putenv retains the supplied storage, so the buffer must have static lifetime. - if (std::getenv("GGML_VK_DISABLE_GRAPH_OPTIMIZE") == nullptr) { - static char kv[] = "GGML_VK_DISABLE_GRAPH_OPTIMIZE=1"; - putenv(kv); - } + // Vulkan reads this setting during device initialization. + set_environment_default("GGML_VK_DISABLE_GRAPH_OPTIMIZE", "1"); #endif // GGML_USE_VULKAN ggml_time_init(); diff --git a/src/runtime/ggml/nn.cpp b/src/runtime/ggml/nn.cpp index 3336c52..c407b14 100644 --- a/src/runtime/ggml/nn.cpp +++ b/src/runtime/ggml/nn.cpp @@ -20,43 +20,6 @@ round_bf16_output(ggml_context* ctx, ggml_tensor* tensor) { return ggml_cast(ctx, ggml_cast(ctx, tensor, GGML_TYPE_BF16), GGML_TYPE_F32); } -ggml_tensor* -cached_q8_input( - Session* session, TensorContainer* session_tensor_container, const ggml_bf_tensor& weight, - ggml_tensor* input) { -#ifdef NEMO_SPEECH_GGML_PATCHED - static const bool enabled = [] { - const char* value = std::getenv("GGML_SKINNY_Q8_CUBLAS_F16"); - return value != nullptr && value[0] != '0'; - }(); -#else - // Stock ggml has no Q8_0 x F16 matmul; the cast would leave no backend. - constexpr bool enabled = false; -#endif - static const int min_columns = [] { - const char* value = std::getenv("GGML_SKINNY_Q8_CUBLAS_F16_MIN_N"); - const int parsed = value != nullptr ? std::atoi(value) : 128; - return parsed > 0 ? parsed : 1; - }(); - const int64_t columns = ggml_nelements(input) / input->ne[0]; - // Mirrors ggml_cuda_skinny_q8_supported: skinny Q8 claims every call wider than MMVQ's - // 8 columns. - const bool eligible_columns = columns > 8; - const bool eligible_block_q8 = - std::string(weight.tensor->name).rfind("encoder.", 0) == 0 && eligible_columns; -#ifdef NEMO_SPEECH_GGML_PATCHED - const bool planar = (weight.tensor->flags & GGML_TENSOR_FLAG_Q8_PLANAR) != 0; -#else - const bool planar = false; -#endif - if (enabled && session->params.use_gpu && weight.tensor->type == GGML_TYPE_Q8_0 && - (planar || eligible_block_q8) && input->type == GGML_TYPE_F32 && columns >= min_columns) { - const auto bf_ctx = session_tensor_container->get_ctx_of_buffer_type(weight.buft); - return ggml_cast(bf_ctx.ctx, input, GGML_TYPE_F16); - } - return input; -} - void Conv1D::define_tensors(Session* session) { // The converter squeezes pointwise kernels to 2D so they can be quantized. @@ -99,7 +62,6 @@ Conv1D::build_graph( // from ggml_conv_1d. auto x_in = ggml_cont(bf_ctx.ctx, ggml_permute(bf_ctx.ctx, input_tensor.tensor, 1, 0, 2, 3)); - x_in = cached_q8_input(session, session_tensor_container, weight_tensor, x_in); // Older CTC GGUFs keep the k=1 conv as [1,in,out]; newer quantized // pointwise tensors are [in,out]. Both are byte-identical after // dropping the unit kernel dimension. @@ -107,8 +69,8 @@ Conv1D::build_graph( out_tensor = ggml_cont(bf_ctx.ctx, ggml_permute(bf_ctx.ctx, matmul, 1, 0, 2, 3)); } else if (is_dw) { bool direct_dw = false; -#ifdef NEMO_SPEECH_DIRECT_DW_CONV - // Patch 0004 adds the F16 direct depthwise kernel only for CUDA; other +#ifdef NEMO_SPEECH_CUDA_FAST_PATHS + // The F16 depthwise-conv patch adds the direct kernel only for CUDA; other // backends require the portable path below. direct_dw = session->params.use_gpu; #endif @@ -149,7 +111,7 @@ Conv1D::build_graph( bf_ctx.ctx, weight_tensor.tensor, input_tensor.tensor, stride, padding, dilation); } } else { - out_tensor = ggml_conv_1d( + out_tensor = conv_1d( bf_ctx.ctx, weight_tensor.tensor, input_tensor.tensor, stride, padding, dilation); } if (use_bias) { @@ -256,7 +218,7 @@ Conv2DDW::build_graph( ggml_bf_context bf_ctx = session_tensor_container->get_ctx_of_buffer_type(weight_tensor.buft); ggml_tensor* conv2d_ret = nullptr; bool direct_dw = false; -#ifdef NEMO_SPEECH_DIRECT_DW_CONV +#ifdef NEMO_SPEECH_CUDA_FAST_PATHS // The direct kernel preserves the explicit N dimension (the stock im2col // lowering does not support this operator with ne[3] > 1) and reads the // F16 weights correctly only on the patched CUDA backend. Same runtime @@ -340,9 +302,7 @@ Linear::build_graph( ggml_bf_tensor weight_tensor = session->model_tensor_container->get_tensor_by_name(weight_name); ggml_bf_context bf_ctx = session_tensor_container->get_ctx_of_buffer_type(weight_tensor.buft); - ggml_tensor* matmul_input = - cached_q8_input(session, session_tensor_container, weight_tensor, input_tensor.tensor); - ggml_tensor* matmul_ret = ggml_mul_mat(bf_ctx.ctx, weight_tensor.tensor, matmul_input); + ggml_tensor* matmul_ret = ggml_mul_mat(bf_ctx.ctx, weight_tensor.tensor, input_tensor.tensor); ggml_tensor* output_tensor = nullptr; if (use_bias) { diff --git a/src/runtime/ggml/nn.h b/src/runtime/ggml/nn.h index aad4839..578fe2b 100644 --- a/src/runtime/ggml/nn.h +++ b/src/runtime/ggml/nn.h @@ -12,10 +12,6 @@ namespace ggml_runtime { -ggml_tensor* cached_q8_input( - Session* session, TensorContainer* session_tensor_container, const ggml_bf_tensor& weight, - ggml_tensor* input); - class Conv1D : public Module { public: Conv1D( diff --git a/src/runtime/ggml/runtime.h b/src/runtime/ggml/runtime.h index 05dded7..fe736c8 100644 --- a/src/runtime/ggml/runtime.h +++ b/src/runtime/ggml/runtime.h @@ -11,6 +11,11 @@ #include #include +// CUDA paths that use operations added by patches/ (see patches/README.md). +#if defined(NEMO_SPEECH_GGML_PATCHED) && defined(GGML_USE_CUDA) +#define NEMO_SPEECH_CUDA_FAST_PATHS 1 +#endif + #include #include #include @@ -58,6 +63,24 @@ struct llama_file; namespace ggml_runtime { +// ggml_conv_1d with the batch axis kept apart from the output channels for a batch +// (b->ne[2]) > 1, which ggml_conv_1d mixes up (ggml-org/llama.cpp#28738). +// Use ggml_conv_1d again once that fix is in the pinned llama.cpp. +inline ggml_tensor* +conv_1d(ggml_context* ctx, ggml_tensor* a, ggml_tensor* b, int s0, int p0, int d0) { + ggml_tensor* im2col = ggml_im2col( + ctx, a, b, s0, 0, p0, 0, d0, 0, false, + a->type == GGML_TYPE_BF16 ? GGML_TYPE_F32 : GGML_TYPE_F16); // [N, OL, IC * K] + ggml_tensor* result = ggml_mul_mat( + ctx, ggml_reshape_2d(ctx, im2col, im2col->ne[0], im2col->ne[2] * im2col->ne[1]), + ggml_reshape_2d(ctx, a, a->ne[0] * a->ne[1], a->ne[2])); // [N * OL, OC] + if (im2col->ne[2] == 1) { + return ggml_reshape_3d(ctx, result, im2col->ne[1], a->ne[2], 1); + } + result = ggml_reshape_3d(ctx, result, im2col->ne[1], im2col->ne[2], a->ne[2]); + return ggml_cont(ctx, ggml_permute(ctx, result, 0, 2, 1, 3)); // [OL, OC, N] +} + class GGUFLoader { public: explicit GGUFLoader(const std::string& path); @@ -165,6 +188,17 @@ class Module { virtual void set_data(Session* session) = 0; }; +// Process-wide settings for the patched CUDA backend (see patches/README.md). They +// are read when a CUDA backend is created, so call them before loading models, and +// they leave a value the user already set in the environment alone. +// +// Keep captured CUDA graphs instead of evicting ones idle for 10 s. For TTS and +// VoiceChat, whose per-frame graphs must stay resident between requests. +void keep_cuda_graphs_resident(); +// Repack skinny-Q8 encoder weights into a separate buffer instead of in place. The +// in-place copy races with llama.cpp's multi-stream scheduler in the same process. +void keep_skinny_q8_weights_separate(); + struct Params { bool use_gpu = false; int gpu_device_idx = 0; diff --git a/src/s2s/CMakeLists.txt b/src/s2s/CMakeLists.txt index e692ab0..6387b6d 100644 --- a/src/s2s/CMakeLists.txt +++ b/src/s2s/CMakeLists.txt @@ -14,13 +14,13 @@ target_include_directories(nemo_speech_s2s PUBLIC ${CMAKE_SOURCE_DIR}/src ${CMAKE_CURRENT_SOURCE_DIR} - ${CMAKE_SOURCE_DIR}/llama.cpp/include + ${NEMO_SPEECH_LLAMA_CPP_DIR}/include PRIVATE ${CMAKE_CURRENT_SOURCE_DIR}/codec ${CMAKE_CURRENT_SOURCE_DIR}/eartts ${CMAKE_CURRENT_SOURCE_DIR}/llm ${CMAKE_CURRENT_SOURCE_DIR}/perception - ${CMAKE_SOURCE_DIR}/llama.cpp/vendor + ${NEMO_SPEECH_LLAMA_CPP_DIR}/vendor ) # The perception/RNN-T side reuses the ASR model implementation and its diff --git a/src/s2s/codec/decoder.cpp b/src/s2s/codec/decoder.cpp index 9d27ff2..c10d848 100644 --- a/src/s2s/codec/decoder.cpp +++ b/src/s2s/codec/decoder.cpp @@ -188,7 +188,7 @@ CodecDecodeWavModule::build_graph(gr::Session* s, gr::TensorBag in, gr::TensorCo } } - x = ggml_conv_1d(g, W(s, "dec.proj_out.weight"), x, 1, 0, 1); + x = ggml_runtime::conv_1d(g, W(s, "dec.proj_out.weight"), x, 1, 0, 1); x = add_channel_bias(g, x, W_opt(s, "dec.proj_out.bias")); ggml_tensor* out_spec = x; // (out_t, spec_channels, B) diff --git a/src/s2s/codec/encoder.cpp b/src/s2s/codec/encoder.cpp index c24d09c..0a90195 100644 --- a/src/s2s/codec/encoder.cpp +++ b/src/s2s/codec/encoder.cpp @@ -87,7 +87,7 @@ CodecEncodeModule::build_graph(gr::Session* s, gr::TensorBag in, gr::TensorConta auto ctx0 = tc->get_ctx_of_buffer_type(spec.buft); ggml_context* g = ctx0.ctx; - ggml_tensor* x = ggml_conv_1d(g, W(s, "enc.proj_in.weight"), spec.tensor, 1, 0, 1); + ggml_tensor* x = gr::conv_1d(g, W(s, "enc.proj_in.weight"), spec.tensor, 1, 0, 1); x = add_channel_bias(g, x, W_opt(s, "enc.proj_in.bias")); int block_idx = 0; @@ -101,7 +101,7 @@ CodecEncodeModule::build_graph(gr::Session* s, gr::TensorBag in, gr::TensorConta const int rate = cfg_.rates[st]; ggml_tensor* w = (st < n_stages - 1) ? W(s, fmt_name("enc.ds%d.weight", st)) : W(s, "enc.bottleneck.weight"); - x = ggml_conv_1d(g, w, x, rate, 0, 1); + x = gr::conv_1d(g, w, x, rate, 0, 1); } // x: (n_frames, L, B) -> latent (L, BT) ggml_tensor* r = ggml_cont(g, ggml_permute(g, x, 1, 0, 2, 3)); diff --git a/src/s2s/voicechat.cpp b/src/s2s/voicechat.cpp index badab25..b95be86 100644 --- a/src/s2s/voicechat.cpp +++ b/src/s2s/voicechat.cpp @@ -13,22 +13,6 @@ #include "runtime.h" namespace nemo_speech::s2s { -namespace { - -void -set_environment_default(const char* name, const char* value) { - if (std::getenv(name)) - return; -#if defined(_WIN32) - if (_putenv_s(name, value) != 0) - throw std::runtime_error(std::string("could not set process default ") + name); -#else - if (setenv(name, value, 0) != 0) - throw std::runtime_error(std::string("could not set process default ") + name); -#endif -} - -} // namespace VoiceChatConfig voicechat_config_from_model_dir(const std::string& model_dir, int gpu, int max_streams) { @@ -54,7 +38,7 @@ voicechat_config_from_model_dir(const std::string& model_dir, int gpu, int max_s struct VoiceChat::Impl { explicit Impl(VoiceChatConfig value) : voicechat_config(std::move(value)) { configure_llama_logging(voicechat_config.verbose); - set_environment_default("GGML_CUDA_GRAPH_EVICT_AFTER_MS", "0"); + ggml_runtime::keep_cuda_graphs_resident(); ggml_runtime::Params params; params.use_gpu = voicechat_config.gpu >= 0; params.gpu_device_idx = voicechat_config.gpu >= 0 ? voicechat_config.gpu : 0; diff --git a/src/server/riva_server.cc b/src/server/riva_server.cc index 87ec6b2..92ba72b 100644 --- a/src/server/riva_server.cc +++ b/src/server/riva_server.cc @@ -24,6 +24,7 @@ #include "model_logging.h" #include "parameter_parser.h" #include "recognizer.h" +#include "runtime.h" #if defined(NEMO_SPEECH_BUILD_TTS) #include "grpc_tts.h" #include "tts/magpietts/config.h" @@ -121,20 +122,6 @@ struct RivaServerConfig { } }; -#if defined(NEMO_SPEECH_BUILD_TTS) -void -configure_cuda_graph_cache_defaults() { - if (std::getenv("GGML_CUDA_GRAPH_EVICT_AFTER_MS") != nullptr) { - return; - } - // putenv (not setenv) so there is no _WIN32 branch; the string must outlive - // the call, hence static. - static char kv[] = "GGML_CUDA_GRAPH_EVICT_AFTER_MS=0"; - putenv(kv); - std::cerr << "[riva_server] defaulting GGML_CUDA_GRAPH_EVICT_AFTER_MS=0" - << " to keep CUDA graphs resident across TTS requests\n"; -} -#endif bool asr_configured(const AsrServerConfig& cfg) { @@ -347,27 +334,8 @@ main(int argc, char** argv) { return 1; } - // The skinny-q8 ggml kernel (see ggml-patches/0005) is an ASR-encoder - // optimization; its in-place repack is unsafe for the NMT decoder and its - // padded single-token decode is slower than the stock path. The flag is read - // once, on the first skinny repack, so set the right env process-wide before - // any service warms up: - // * NMT alone: disable skinny (GGML_SKINNY_Q8=0). TTS is fp16 and does not - // use skinny. - // * NMT + ASR: keep skinny for the ASR encoder, force the safe non-in-place - // repack (GGML_SKINNY_Q8_INPLACE=0); NMT then runs skinny. - if (enable_nmt) { - if (enable_asr) { - static char kv[] = "GGML_SKINNY_Q8_INPLACE=0"; - putenv(kv); - std::cerr << "[riva_server] NMT+ASR: forcing GGML_SKINNY_Q8_INPLACE=0 " - "(keep ASR skinny-q8, non-in-place)\n"; - } else { - static char kv[] = "GGML_SKINNY_Q8=0"; - putenv(kv); - std::cerr << "[riva_server] NMT without ASR: forcing GGML_SKINNY_Q8=0 " - "(disable skinny-q8; faster decode)\n"; - } + if (enable_nmt && enable_asr) { + ggml_runtime::keep_skinny_q8_weights_separate(); } // Core capabilities outlive the protocol adapters. @@ -405,7 +373,7 @@ main(int argc, char** argv) { #if defined(NEMO_SPEECH_BUILD_TTS) if (enable_tts) { - configure_cuda_graph_cache_defaults(); + ggml_runtime::keep_cuda_graphs_resident(); std::cerr << "[riva_server] loading MagpieTTS model: " << cfg.tts.server.runtime.magpie_model << "\n"; std::cerr << "[riva_server] loading NanoCodec model: " << cfg.tts.server.runtime.codec_model diff --git a/src/tts/magpietts/CMakeLists.txt b/src/tts/magpietts/CMakeLists.txt index 8089b56..640ecdc 100644 --- a/src/tts/magpietts/CMakeLists.txt +++ b/src/tts/magpietts/CMakeLists.txt @@ -45,9 +45,6 @@ target_link_libraries(nemo_speech_tts PUBLIC # Stable C ABI (nemo_speech_tts_*) and the in-tree C++ API share one implementation # library. Aliases preserve existing target names without producing extra DSOs. target_compile_definitions(nemo_speech_tts PRIVATE NEMO_SPEECH_TTS_BUILD) -if(NEMO_SPEECH_GGML_PATCHED) - target_compile_definitions(nemo_speech_tts PRIVATE NEMO_SPEECH_GGML_PATCHED=1) -endif() set_target_properties(nemo_speech_tts PROPERTIES SOVERSION 1) if(UNIX AND NOT APPLE) target_link_options(nemo_speech_tts PRIVATE @@ -75,7 +72,8 @@ foreach(target ${MAGPIETTS_TARGETS}) ) endforeach() -if(GGML_CUDA) +# The CUDA sampler runs on ggml's stream, which only the patched ggml exposes. +if(GGML_CUDA AND NEMO_SPEECH_GGML_PATCHED) find_package(CUDAToolkit QUIET) if(CUDAToolkit_FOUND AND CUDAToolkit_INCLUDE_DIRS) foreach(target ${MAGPIETTS_TARGETS}) @@ -86,7 +84,7 @@ if(GGML_CUDA) enable_language(CUDA) foreach(target ${MAGPIETTS_TARGETS}) target_sources(${target} PRIVATE magpietts_cuda_sampling.cu magpietts_lt_fused.cu magpietts_decoder_fused.cu) - target_compile_definitions(${target} PRIVATE GGML_USE_CUDA MAGPIETTS_CUDA_SAMPLING) + target_compile_definitions(${target} PRIVATE MAGPIETTS_CUDA_SAMPLING) target_link_libraries(${target} PRIVATE CUDA::cudart CUDA::cuda_driver) set_target_properties(${target} PROPERTIES CUDA_STANDARD 17 @@ -106,9 +104,3 @@ if(GGML_CUDA) endif() endif() endif() - -if(GGML_METAL) - foreach(target ${MAGPIETTS_TARGETS}) - target_compile_definitions(${target} PRIVATE GGML_USE_METAL) - endforeach() -endif() diff --git a/src/tts/magpietts/decoder.cpp b/src/tts/magpietts/decoder.cpp index 5a39cbd..55a6d99 100644 --- a/src/tts/magpietts/decoder.cpp +++ b/src/tts/magpietts/decoder.cpp @@ -15,7 +15,7 @@ #include "../../runtime/ggml/runtime.h" #include "graph.h" #include "nvtx_utils.h" -#if defined(GGML_USE_CUDA) +#if defined(MAGPIETTS_CUDA_SAMPLING) #include #include @@ -267,7 +267,7 @@ DecoderCrossKvCache::init(const magpietts_model& model, int requested_text_len) constexpr int kTextCapacityGranule = 128; int wanted_capacity = (requested_text_len + kTextCapacityGranule - 1) / kTextCapacityGranule * kTextCapacityGranule; -#if defined(GGML_USE_CUDA) +#if defined(MAGPIETTS_CUDA_SAMPLING) // The persistent decoder runtime is built on this buffer; allocate the fused step's full // text window once so a longer sentence never forces a rebuild. if (model.backend && ggml_backend_is_cuda(model.backend)) @@ -718,7 +718,7 @@ class MagpieDecoder::PersistentDecoderRuntime { } ~PersistentDecoderRuntime() { -#if defined(GGML_USE_CUDA) +#if defined(MAGPIETTS_CUDA_SAMPLING) if (fused_) magpietts_decoder_fused_free(fused_); if (fused_dev_) @@ -847,7 +847,7 @@ class MagpieDecoder::PersistentDecoderRuntime { {"magpietts.decoder.runtime.mask", GGML_TYPE_F32, pad_mask.data(), {text_capacity_}}}; const bool has_alignment = module_.alignment_count() > 0; -#if defined(GGML_USE_CUDA) +#if defined(MAGPIETTS_CUDA_SAMPLING) if (fused_) { std::vector fused_alignment; const bool want_alignment = has_alignment && attention && attention->alignment_scores; @@ -906,7 +906,7 @@ class MagpieDecoder::PersistentDecoderRuntime { // Single-launch CUDA decoder step (magpietts_decoder_fused.cu) replacing the ggml graph // whenever the projections are Q8_0, the shapes fit and the GPU supports it. void init_fused() { -#if defined(GGML_USE_CUDA) +#if defined(MAGPIETTS_CUDA_SAMPLING) if (!model_.backend || !ggml_backend_is_cuda(model_.backend)) return; const magpietts_hparams& h = model_.hparams; @@ -1000,7 +1000,7 @@ class MagpieDecoder::PersistentDecoderRuntime { #endif } -#if defined(GGML_USE_CUDA) +#if defined(MAGPIETTS_CUDA_SAMPLING) // Runs the fused step. hidden_*: device outputs; alignment (host, text_len_) filled when // requested. public: @@ -1152,7 +1152,7 @@ MagpieDecoder::evalCachedPair( magpietts_cuda_sample_request* cuda_sample, const magpietts_backend_tensor* text_cond_device, magpietts_backend_tensor* cond_hidden_out, magpietts_backend_tensor* uncond_hidden_out, DecoderCrossKvCache* cond_cross_kv, const magpietts_decoder_attention* attention) const { -#if defined(GGML_USE_CUDA) +#if defined(MAGPIETTS_CUDA_SAMPLING) if (persistent_runtime_) { persistent_runtime_->discardPendingAlignment(); } @@ -1894,7 +1894,7 @@ MagpieDecoder::prefillPair( bool MagpieDecoder::completeAlignment(const magpietts_decoder_attention* attention) const { -#if defined(GGML_USE_CUDA) +#if defined(MAGPIETTS_CUDA_SAMPLING) if (persistent_runtime_ && attention && attention->alignment_scores) return persistent_runtime_->completeAlignment(attention->alignment_scores); #else @@ -1905,7 +1905,7 @@ MagpieDecoder::completeAlignment(const magpietts_decoder_attention* attention) c bool MagpieDecoder::fusedStepExclusive() const { -#if defined(GGML_USE_CUDA) +#if defined(MAGPIETTS_CUDA_SAMPLING) return persistent_runtime_ && persistent_runtime_->fusedStepExclusive(); #else return false; @@ -1919,7 +1919,7 @@ MagpieDecoder::adoptPrefill( DecoderKvCache& uncond_kv, DecoderCrossKvCache& cond_cross_kv, int text_len, int stacked_position_budget) const { const ggml_nvtx::range nvtx_range("magpietts_decoder_adopt_prefill"); -#if defined(GGML_USE_CUDA) +#if defined(MAGPIETTS_CUDA_SAMPLING) if (persistent_runtime_) { persistent_runtime_->discardPendingAlignment(); } diff --git a/src/tts/magpietts/magpietts.cpp b/src/tts/magpietts/magpietts.cpp index a131773..de6dda0 100644 --- a/src/tts/magpietts/magpietts.cpp +++ b/src/tts/magpietts/magpietts.cpp @@ -32,7 +32,7 @@ #include "nvtx_utils.h" #include "token_utils.h" #include "tts/nanocodec/model.h" -#if defined(GGML_USE_CUDA) +#if defined(MAGPIETTS_CUDA_SAMPLING) #include "magpietts_decoder_fused.h" #include "magpietts_lt_fused.h" #endif @@ -2017,7 +2017,7 @@ stream_magpie_to_audio( metrics.generated_frames = frames_generated; metrics.finish(outputs.samples_written, codec.sampleRate(), codec_fps); -#if defined(GGML_USE_CUDA) +#if defined(MAGPIETTS_CUDA_SAMPLING) // -DLTF_CHAIN_TIMING builds only magpietts_decoder_fused_timing_report(); magpietts_lt_fused_timing_report(); diff --git a/src/tts/magpietts/model.h b/src/tts/magpietts/model.h index 306ba15..55a0c9c 100644 --- a/src/tts/magpietts/model.h +++ b/src/tts/magpietts/model.h @@ -421,7 +421,8 @@ const char* magpietts_uma_mode_name(magpietts_uma_mode mode); bool parse_uma_mode(const std::string& value, magpietts_uma_mode& mode); bool magpietts_backend_is_cuda(ggml_backend_t backend); -// ggml_fused_attn_cached comes from ggml-patches/0014 and has a CUDA kernel only. +// ggml_fused_attn_cached comes from the "fused attention op" patch in patches/ and has a CUDA +// kernel only. bool magpietts_fused_cached_attention_available(ggml_backend_t backend); std::vector parse_token_list(const std::string& text); diff --git a/src/tts/nanocodec/CMakeLists.txt b/src/tts/nanocodec/CMakeLists.txt index 50f0e7e..dae9329 100644 --- a/src/tts/nanocodec/CMakeLists.txt +++ b/src/tts/nanocodec/CMakeLists.txt @@ -31,15 +31,6 @@ if(GGML_CUDA) endforeach() endif() if(CUDAToolkit_FOUND) - foreach(target ${NANOCODEC_TARGETS}) - target_compile_definitions(${target} PRIVATE GGML_USE_CUDA) - endforeach() target_link_libraries(nemo_speech_tts_nanocodec_obj PUBLIC CUDA::cuda_driver) endif() endif() - -if(GGML_METAL) - foreach(target ${NANOCODEC_TARGETS}) - target_compile_definitions(${target} PRIVATE GGML_USE_METAL) - endforeach() -endif() diff --git a/tests/conversion/converter_contract_test.py b/tests/conversion/converter_contract_test.py index ffbcc35..60a21ad 100644 --- a/tests/conversion/converter_contract_test.py +++ b/tests/conversion/converter_contract_test.py @@ -363,9 +363,9 @@ def test_s2s_quantizer_is_built_automatically_when_missing(self) -> None: llama_cpp.mkdir() (llama_cpp / "CMakeLists.txt").touch() (llama_cpp / "convert_hf_to_gguf.py").touch() - patch_script = root / "scripts" / "apply-llama-patches.sh" - patch_script.parent.mkdir() - patch_script.touch() + materialize = root / "cmake" / "llama_cpp.cmake" + materialize.parent.mkdir() + materialize.touch() built = default_quantizer_path(llama_cpp) def fake_run(command, **_kwargs): @@ -390,7 +390,11 @@ def fake_which(name): self.assertEqual(actual, built) commands = [call.args[0] for call in run.call_args_list] - self.assertEqual(commands[0], ["bash", str(patch_script)]) + self.assertEqual(commands[0][-2:], ["-P", str(materialize)]) + self.assertIn(f"-DDEST_DIR={root / '.deps' / 'llama.cpp-patched'}", commands[0]) + self.assertEqual( + commands[1][commands[1].index("-S") + 1], str(root / ".deps" / "llama.cpp-patched") + ) self.assertIn("-DLLAMA_BUILD_TOOLS=ON", commands[1]) self.assertEqual(commands[2][-3:], ["--target", "llama-quantize", "--parallel"]) diff --git a/tests/cpp/asr/CMakeLists.txt b/tests/cpp/asr/CMakeLists.txt index d1f6542..2a7b054 100644 --- a/tests/cpp/asr/CMakeLists.txt +++ b/tests/cpp/asr/CMakeLists.txt @@ -83,7 +83,7 @@ endif() add_executable(asr_c_dlopen ${CMAKE_SOURCE_DIR}/tests/c/asr_c_dlopen.c) target_link_libraries(asr_c_dlopen PRIVATE ${CMAKE_DL_LIBS}) -# Skinny-Q8 dispatch (ggml patch 0029) must give the same bits before and after a weight is +# Skinny-Q8 dispatch (the skinny-Q8 ggml patch) must give the same bits before and after a weight is # repacked. The kernel exists only in the patched CUDA ggml. Skips itself (exit 77) without a GPU. if(GGML_CUDA AND NEMO_SPEECH_GGML_PATCHED) add_executable(test_skinny_q8_dispatch test_skinny_q8_dispatch.cpp) diff --git a/tests/cpp/asr/test_skinny_q8_dispatch.cpp b/tests/cpp/asr/test_skinny_q8_dispatch.cpp index 70844df..7bb7559 100644 --- a/tests/cpp/asr/test_skinny_q8_dispatch.cpp +++ b/tests/cpp/asr/test_skinny_q8_dispatch.cpp @@ -1,10 +1,11 @@ // SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. // SPDX-License-Identifier: Apache-2.0 // -// Skinny-Q8 dispatch must not depend on call history (ggml patch 0029). The first 9..64-column -// mul_mat on an eligible Q8_0 weight repacks it in place; every column count must produce the same -// bits before and after that repack, including MMVQ's range (N <= 8, planar MMVQ after the repack) -// and widths above 64. Skips itself (exit 77) without a GPU backend. +// Skinny-Q8 dispatch must not depend on call history (the skinny-Q8 ggml patch). The first +// 9..64-column mul_mat on an eligible Q8_0 weight repacks it in place; every column count must +// produce the same bits before and after that repack, including MMVQ's range (N <= 8, planar MMVQ +// after the repack), the fused MMVQ bias/SiLU epilogues, and widths above 64. Skips itself +// (exit 77) without a GPU backend. #include #include #include @@ -20,7 +21,7 @@ constexpr int64_t kK = 512; // multiple of the skinny kernel's 128-byte k step constexpr int64_t kM = 256; // multiple of its 32-row block std::vector -run(ggml_backend_t backend, ggml_tensor* w, ggml_tensor* bias, int64_t n) { +run(ggml_backend_t backend, ggml_tensor* w, ggml_tensor* bias, int64_t n, bool silu = false) { ggml_init_params params{ggml_tensor_overhead() * 8 + ggml_graph_overhead(), nullptr, true}; ggml_context* ctx = ggml_init(params); ggml_tensor* x = ggml_new_tensor_2d(ctx, GGML_TYPE_F32, kK, n); @@ -28,6 +29,9 @@ run(ggml_backend_t backend, ggml_tensor* w, ggml_tensor* bias, int64_t n) { if (bias) { y = ggml_add(ctx, y, bias); // mul_mat + add is fused into the skinny-Q8 bias epilogue } + if (silu) { + y = ggml_silu(ctx, y); // narrow mul_mat (+ add) + silu is fused into MMVQ + } ggml_cgraph* gf = ggml_new_graph(ctx); ggml_build_forward_expand(gf, y); ggml_gallocr_t allocr = ggml_gallocr_new(ggml_backend_get_default_buffer_type(backend)); @@ -81,26 +85,32 @@ main() { // was already repacked: MMVQ's range and wider than the old 64-column cap. A skinny-range call // here would repack the weight before the reference outputs are recorded. const int64_t widths[] = {1, 2, 4, 8, 65, 69, 128}; - std::vector> before, before_bias; + std::vector> before, before_bias, before_silu, before_bias_silu; for (int64_t n : widths) { before.push_back(run(backend, w, nullptr, n)); before_bias.push_back(run(backend, w, bias, n)); + before_silu.push_back(run(backend, w, nullptr, n, true)); + before_bias_silu.push_back(run(backend, w, bias, n, true)); } run(backend, w, nullptr, 16); // a skinny-range call repacks the weight in place int failures = 0; for (size_t i = 0; i < sizeof(widths) / sizeof(widths[0]); ++i) { - const std::vector after = run(backend, w, nullptr, widths[i]); - const std::vector after_bias = run(backend, w, bias, widths[i]); - const bool same = - std::memcmp(after.data(), before[i].data(), after.size() * sizeof(float)) == 0; - const bool same_bias = - std::memcmp( - after_bias.data(), before_bias[i].data(), after_bias.size() * sizeof(float)) == 0; - if (!same || !same_bias) { + const auto same = [](const std::vector& a, const std::vector& b) { + return std::memcmp(a.data(), b.data(), a.size() * sizeof(float)) == 0; + }; + const bool plain = same(run(backend, w, nullptr, widths[i]), before[i]); + const bool with_bias = same(run(backend, w, bias, widths[i]), before_bias[i]); + const bool with_silu = same(run(backend, w, nullptr, widths[i], true), before_silu[i]); + const bool with_bias_silu = + same(run(backend, w, bias, widths[i], true), before_bias_silu[i]); + if (!plain || !with_bias || !with_silu || !with_bias_silu) { std::fprintf( - stderr, "FAIL: N=%lld output changed after the repack (plain %s, bias %s)\n", - (long long)widths[i], same ? "same" : "differs", same_bias ? "same" : "differs"); + stderr, + "FAIL: N=%lld output changed after the repack (plain %s, bias %s, silu %s, " + "bias+silu %s)\n", + (long long)widths[i], plain ? "same" : "differs", with_bias ? "same" : "differs", + with_silu ? "same" : "differs", with_bias_silu ? "same" : "differs"); ++failures; } } diff --git a/tools/CMakeLists.txt b/tools/CMakeLists.txt index a9cea83..cf695a5 100644 --- a/tools/CMakeLists.txt +++ b/tools/CMakeLists.txt @@ -10,7 +10,6 @@ if(NEMO_SPEECH_BUILD_ASR OR NEMO_SPEECH_BUILD_DIAR) add_executable(bench_asr_batching bench_asr_batching.cpp) target_link_libraries(bench_asr_batching PRIVATE nemo_speech_asr Threads::Threads) if(GGML_CUDA) - target_compile_definitions(bench_asr_batching PRIVATE GGML_USE_CUDA) target_link_libraries(bench_asr_batching PRIVATE CUDA::cudart) endif() endif()