diff --git a/.coderabbit.yaml b/.coderabbit.yaml
index 200ce48..970ef7e 100644
--- a/.coderabbit.yaml
+++ b/.coderabbit.yaml
@@ -23,9 +23,9 @@ reviews:
synchronization, and caches or reuse that do not actually take effect.
- Portability: assumptions tied to one GPU, architecture, backend or platform; hardware
limits must be queried or guarded, with a working fallback.
- - path: "ggml-patches/**"
+ - path: "patches/**"
instructions: |
- Patches to the pinned ggml submodule. Apply the same correctness, performance and
+ Patches to the pinned llama.cpp submodule. Apply the same correctness, performance and
portability checks; also check that op preconditions match what the kernels support
and that the patch series stays consistent.
- path: "{BENCHMARK.md,README.md,docs/**,app/bench*}"
diff --git a/.dockerignore b/.dockerignore
index 601e7c9..270352a 100644
--- a/.dockerignore
+++ b/.dockerignore
@@ -1,7 +1,7 @@
.git
-# Nested submodule .git files (ggml, llama.cpp, third_party/*) become broken
-# pointers once copied; exclude them so apply-ggml-patches.sh runs git-apply
-# on a plain tree and the build context stays small.
+# Nested submodule .git files (llama.cpp, third_party/*) become broken pointers
+# once copied; exclude them so the build context stays small. CMake applies
+# patches/ to a copy of the plain llama.cpp tree.
**/.git
.gitignore
.gitmodules
diff --git a/.gitattributes b/.gitattributes
index c5d6456..7853d7f 100644
--- a/.gitattributes
+++ b/.gitattributes
@@ -1,8 +1,7 @@
-# Keep ggml patch files LF on every platform. On Windows with core.autocrlf=true
-# they would otherwise check out CRLF, and `git apply` rejects some hunks as
-# "corrupt patch" on mixed endings (notably 0007-magpietts-nanocodec.patch).
-ggml-patches/*.patch text eol=lf -whitespace
-llama-patches/*.patch text eol=lf -whitespace
+# Keep the llama.cpp patch files LF on every platform. On Windows with
+# core.autocrlf=true they would otherwise check out CRLF, and `git apply`
+# rejects some hunks as "corrupt patch" on mixed endings.
+patches/**/*.patch text eol=lf -whitespace
# Shell scripts must stay LF so they run under bash / Git Bash on Windows.
*.sh text eol=lf
src/tts/tokenizer/mandarin_data/pinyin_phrases.tsv filter=lfs diff=lfs merge=lfs -text
diff --git a/.github/workflows/build.yml b/.github/workflows/build.yml
index 6792e96..664c3bc 100644
--- a/.github/workflows/build.yml
+++ b/.github/workflows/build.yml
@@ -4,6 +4,7 @@ name: Build and Test
# model-free ctest suite, then a CLI smoke test with real models.
#
# Coverage per job:
+# Patch series - patches/ applies to the pinned llama.cpp in exported form
# Linux CPU - build, ctest, CPU inference
# macOS Metal - Metal backend COMPILE AND LINK only; inference runs on CPU.
# The hosted macOS VM's paravirtual GPU cannot run ggml Metal.
@@ -28,6 +29,25 @@ env:
NEMO_SPEECH_MODEL_DIR: ${{ github.workspace }}/.ci-models
jobs:
+ patch-series:
+ name: Patch series
+ if: github.event_name != 'pull_request' || !github.event.pull_request.draft
+ runs-on: ubuntu-latest
+ timeout-minutes: 10
+ steps:
+ - name: Checkout repository
+ uses: actions/checkout@df4cb1c069e1874edd31b4311f1884172cec0e10 # v6.0.3
+ with:
+ persist-credentials: false
+
+ - name: Initialize submodules
+ run: git submodule update --init --depth 1 llama.cpp
+
+ - name: Check patches/
+ # The series must apply to the pinned llama.cpp and be exactly what
+ # scripts/llama-patches.sh export produces.
+ run: scripts/llama-patches.sh check
+
linux-cpu:
name: Linux CPU
if: github.event_name != 'pull_request' || !github.event.pull_request.draft
@@ -40,8 +60,8 @@ jobs:
persist-credentials: false
- name: Initialize submodules
- # llama.cpp is needed only for the vendored miniaudio header.
- run: git submodule update --init --depth 1 ggml llama.cpp
+ # llama.cpp also provides ggml.
+ run: git submodule update --init --depth 1 llama.cpp
- name: Install dependencies
run: |
@@ -86,7 +106,7 @@ jobs:
persist-credentials: false
- name: Initialize submodules
- run: git submodule update --init --depth 1 ggml llama.cpp
+ run: git submodule update --init --depth 1 llama.cpp
- name: Install dependencies
# Homebrew's sentencepiece header includes abseil, which the formula
@@ -94,8 +114,8 @@ jobs:
run: brew install bash ninja sentencepiece abseil
- name: Configure
- # The runner's default shell is Apple's bash 3.2; configure.sh and
- # apply-ggml-patches.sh need bash 4+, so run them under Homebrew bash.
+ # The runner's default shell is Apple's bash 3.2; configure.sh needs
+ # bash 4+, so run it under Homebrew bash.
shell: /opt/homebrew/bin/bash -e {0}
run: |
scripts/configure.sh metal-speech \
@@ -148,7 +168,7 @@ jobs:
persist-credentials: false
- name: Initialize submodules
- run: git submodule update --init --depth 1 ggml llama.cpp
+ run: git submodule update --init --depth 1 llama.cpp
- name: Install Ninja
shell: pwsh
diff --git a/.github/workflows/gpu.yml b/.github/workflows/gpu.yml
index f5e80c9..0ddfa8a 100644
--- a/.github/workflows/gpu.yml
+++ b/.github/workflows/gpu.yml
@@ -74,10 +74,10 @@ jobs:
echo "NEMO_SPEECH_MODEL_DIR=$GITHUB_WORKSPACE/.ci-models" >> "$GITHUB_ENV"
- name: Initialize submodules
- run: git submodule update --init --depth 1 ggml llama.cpp
+ run: git submodule update --init --depth 1 llama.cpp
- name: Configure
- # configure.sh applies the ggml patch series for cuda-* presets.
+ # CMake applies patches/ for cuda-* presets.
# sm_89 is the L4.
run: |
scripts/configure.sh cuda-speech \
@@ -118,7 +118,7 @@ jobs:
persist-credentials: false
- name: Initialize submodules
- run: git submodule update --init --depth 1 ggml llama.cpp
+ run: git submodule update --init --depth 1 llama.cpp
- name: Install build prerequisites
# The ephemeral VM ships the NVIDIA driver (with the Vulkan ICD) but not
diff --git a/.gitmodules b/.gitmodules
index 1736af9..7f26d70 100644
--- a/.gitmodules
+++ b/.gitmodules
@@ -1,6 +1,3 @@
-[submodule "ggml"]
- path = ggml
- url = https://github.com/ggml-org/ggml.git
[submodule "proto/riva-common"]
path = proto/riva-common
url = https://github.com/nvidia-riva/common.git
diff --git a/.pre-commit-config.yaml b/.pre-commit-config.yaml
index ff55267..bd9bd1c 100644
--- a/.pre-commit-config.yaml
+++ b/.pre-commit-config.yaml
@@ -23,27 +23,27 @@ repos:
- id: check-yaml
- id: check-merge-conflict
- id: end-of-file-fixer
- exclude: '^(ggml/|ggml-patches/|proto/riva-common/).*'
+ exclude: '^(patches/|proto/riva-common/).*'
- id: trailing-whitespace
- exclude: '^(ggml/|ggml-patches/|proto/riva-common/).*'
+ exclude: '^(patches/|proto/riva-common/).*'
- id: mixed-line-ending
args: ['--fix=lf']
- exclude: '^(ggml/|ggml-patches/|proto/riva-common/).*'
+ exclude: '^(patches/|proto/riva-common/).*'
- repo: https://github.com/pre-commit/mirrors-clang-format
rev: v18.1.8
hooks:
- id: clang-format
types_or: [c++, c, cuda]
- # ggml + vendored riva protos are upstream code — don't reformat.
- exclude: '^(ggml/|ggml-patches/|proto/riva-common/|build/|build-.*/).*'
+ # llama.cpp patches + vendored riva protos are upstream code — don't reformat.
+ exclude: '^(patches/|proto/riva-common/|build/|build-.*/).*'
- repo: https://github.com/psf/black
rev: 24.10.0
hooks:
- id: black
args: ['--skip-string-normalization', '--line-length=100']
- exclude: '^(ggml/|proto/riva-common/|build/).*'
+ exclude: '^(proto/riva-common/|build/).*'
- repo: https://github.com/pycqa/isort
rev: 5.13.2
@@ -53,7 +53,7 @@ repos:
args:
- '--profile=black'
- '--line-length=100'
- - '--skip=ggml'
+ - '--skip=llama.cpp'
- '--skip=proto/riva-common'
- '--skip=build'
@@ -62,4 +62,4 @@ repos:
hooks:
- id: shellcheck
args: ['--severity=warning']
- exclude: '^(ggml/|ggml-patches/|proto/riva-common/).*'
+ exclude: '^(patches/|proto/riva-common/).*'
diff --git a/BENCHMARK.md b/BENCHMARK.md
index 1d0f3b4..ecb19d7 100644
--- a/BENCHMARK.md
+++ b/BENCHMARK.md
@@ -144,7 +144,7 @@ Whisper English normalizer.
| | |
|---|---|
-| Models | MagpieTTS Multilingual 357M v2607, NeMo NanoCodec 22 kHz (F16) |
+| Models | MagpieTTS Multilingual 357M v2607 (Q8_0, [converted locally](docs/tts/models.md#magpietts-token-generator)), NeMo NanoCodec 22 kHz (F16) |
| Synthesis | `en-US`, default voice, seed 1, 22.05 kHz audio in 186 ms chunks (4 codec frames) |
| Inputs | The 10 LJSpeech sentences of the Riva TTS performance reports ([`ljs_audio_text_test_filelist_small.txt`](test_files/tts/ljs_audio_text_test_filelist_small.txt), 20 requests); by length, [`test_files/tts/bench`](test_files/tts/bench) (5 requests per input) |
| Metrics | Latencies at the client from the streaming audio callback; throughput is audio duration over wall time |
diff --git a/CMakeLists.txt b/CMakeLists.txt
index 50fd2fe..685b661 100644
--- a/CMakeLists.txt
+++ b/CMakeLists.txt
@@ -119,22 +119,16 @@ set(NEMO_SPEECH_DEPENDENCY_PREFIX "${CMAKE_SOURCE_DIR}/.deps" CACHE PATH
"User-writable prefix containing optional project-built dependencies")
# GGML_CUDA is forwarded directly to ggml's CMake via -DGGML_CUDA=ON.
-# Keep the legacy WITH_NMT/WITH_GRPC options as downstream-compatible aliases.
-if(NEMO_SPEECH_WITH_NMT)
- set(NEMO_SPEECH_BUILD_NMT ON CACHE BOOL "Build text translation (links llama.cpp)" FORCE)
-endif()
-if(NEMO_SPEECH_BUILD_NMT)
- set(NEMO_SPEECH_WITH_NMT ON)
-endif()
+# WITH_NMT/WITH_GRPC only seed the BUILD_* defaults above; an explicit BUILD_*
+# value takes precedence.
+foreach(_alias NMT GRPC)
+ if(NEMO_SPEECH_WITH_${_alias})
+ message(DEPRECATION "NEMO_SPEECH_WITH_${_alias} is deprecated; use NEMO_SPEECH_BUILD_${_alias}")
+ endif()
+endforeach()
if(NEMO_SPEECH_BUILD_S2S)
set(NEMO_SPEECH_BUILD_ASR ON CACHE BOOL "Build automatic speech recognition" FORCE)
endif()
-if(NEMO_SPEECH_WITH_GRPC)
- set(NEMO_SPEECH_BUILD_GRPC ON CACHE BOOL "Build Riva-compatible gRPC adapters" FORCE)
-endif()
-if(NEMO_SPEECH_BUILD_GRPC)
- set(NEMO_SPEECH_WITH_GRPC ON)
-endif()
# Reject invalid component combinations early.
if(WIN32 AND NEMO_SPEECH_WITH_NORM)
@@ -176,57 +170,14 @@ endif()
# no-op for Metal, Vulkan, and CPU builds.
option(NEMO_SPEECH_CUBLAS_SHIM "Build the in-tree drop-in cuBLAS shim (native GEMM, no cuBLASLt)" OFF)
-# Whether the linked ggml has the project ASR patches applied (ggml-patches/:
-# the fused rel-pos attention op and the F16 depthwise-conv kernel). The ASR
-# directly references these (a new op symbol + F16 CONV_2D_DW behaviour), so a
-# build against STANDARD upstream ggml must set -DNEMO_SPEECH_GGML_PATCHED=OFF
-# - the encoder then uses only stock ggml ops (unfused rel-pos, ggml_conv_1d_dw)
-# at some latency cost. The remaining low-level patches (NVFP4, skinny-q8,
-# norm/BF16 fusions, CUDA graph and launch fixes) are transparent ggml-cuda internals.
-option(NEMO_SPEECH_GGML_PATCHED "Linked ggml has the project ASR patches applied (fused rel-pos op, F16 dw-conv)" ON)
-
-# Relative-position mode of the fused CUDA attention op. Replaces the unfused
-# encoder attention sequence with one kernel. Defaults ON with CUDA and the
-# patched ggml; the encoder otherwise uses stock ggml ops.
-if(GGML_CUDA AND NEMO_SPEECH_GGML_PATCHED)
- option(NEMO_SPEECH_FUSED_RELPOS_ATTN "Use the fused rel-pos attention CUDA op in the encoder" ON)
-else()
- set(NEMO_SPEECH_FUSED_RELPOS_ATTN OFF CACHE BOOL
- "Use the fused rel-pos attention CUDA op in the encoder" FORCE)
- message(STATUS
- "NEMO_SPEECH_FUSED_RELPOS_ATTN forced OFF: requires GGML_CUDA=ON and "
- "NEMO_SPEECH_GGML_PATCHED=ON (patched ggml). Using the unfused path.")
-endif()
-
-# Direct depthwise-conv kernel (GGML_OP_CONV_2D_DW) for the conformer/subsample
-# convs. Faster than ggml_conv_1d_dw's im2col + matmul, but the conv weights are
-# F16 and the direct CONV_2D_DW op only reads F16 kernels correctly on the
-# patched CUDA backend (ggml patch 0004); stock ggml (CPU or unpatched CUDA)
-# reads them as F32. Defaults ON whenever GGML_CUDA + a patched ggml are present;
-# forced OFF otherwise, where nn.cpp falls back to the portable ggml_conv_1d_dw
-# lowering (im2col + mul_mat, F16-safe on every backend and on stock ggml). Kept
-# as an explicit option so it can be toggled OFF for debugging.
-if(GGML_CUDA AND NEMO_SPEECH_GGML_PATCHED)
- option(NEMO_SPEECH_DIRECT_DW_CONV "Use the direct CUDA depthwise-conv kernel in the encoder" ON)
-else()
- set(NEMO_SPEECH_DIRECT_DW_CONV OFF CACHE BOOL
- "Use the direct CUDA depthwise-conv kernel in the encoder" FORCE)
- message(STATUS
- "NEMO_SPEECH_DIRECT_DW_CONV forced OFF: requires GGML_CUDA=ON and "
- "NEMO_SPEECH_GGML_PATCHED=ON (patched ggml). Using ggml_conv_1d_dw.")
-endif()
-
-# FastConformer graph rewrites that target CUDA-only ggml ops/fusions (sigmoid
-# GLU plus BF16 projection epilogues). Keep the symbols out of standard-ggml,
-# CPU, Metal, and Vulkan builds. The CUDA backend performs the finer runtime
-# architecture checks: native BF16 epilogues require NVIDIA SM80+, and the
-# 1024-wide LayerNorm launch specialization requires SM90+.
-if(GGML_CUDA AND NEMO_SPEECH_GGML_PATCHED)
- option(NEMO_SPEECH_FASTCONFORMER_CUDA_FUSIONS
- "Use patched CUDA FastConformer graph fusions" ON)
-else()
- set(NEMO_SPEECH_FASTCONFORMER_CUDA_FUSIONS OFF CACHE BOOL
- "Use patched CUDA FastConformer graph fusions" FORCE)
+# Build ggml and llama.cpp with the patches/ series applied (see patches/README.md
+# and cmake/llama_cpp.cmake). OFF builds the pristine llama.cpp submodule; the code
+# then uses only stock ggml operations, at some latency cost on CUDA.
+option(NEMO_SPEECH_GGML_PATCHED "Apply patches/ to ggml and llama.cpp and use the patched operations" ON)
+if(NEMO_SPEECH_BUILD_S2S AND NOT NEMO_SPEECH_GGML_PATCHED AND NOT NEMO_SPEECH_LLAMA_CPP_SOURCE_DIR)
+ # EarTTS stores its Gemma 3 attention scale in GGUF metadata that stock
+ # llama.cpp ignores, which would silently change the model's output.
+ message(FATAL_ERROR "NEMO_SPEECH_BUILD_S2S requires NEMO_SPEECH_GGML_PATCHED=ON")
endif()
set(GGML_DEPENDENCIES ggml ggml-base ggml-cpu)
@@ -245,12 +196,6 @@ if(GGML_CUDA)
# on by default. ggml-cuda gates it at runtime and disables it for graphs it
# can't capture (e.g. shapes that change between calls).
set(GGML_CUDA_GRAPHS ON CACHE BOOL "Enable ggml-cuda CUDA Graphs")
- file(READ "${CMAKE_SOURCE_DIR}/ggml/src/ggml-cuda/conv-transpose-1d.cu" GGML_CUDA_CONV_TRANSPOSE_1D)
- if(NOT GGML_CUDA_CONV_TRANSPOSE_1D MATCHES "conv_transpose_1d_use_nanocodec_grouped2")
- message(WARNING
- "MagpieTTS/NanoCodec CUDA ggml patch is not applied. "
- "Run scripts/apply-ggml-patches.sh before compiling CUDA TTS targets.")
- endif()
endif()
if(GGML_VULKAN)
list(APPEND GGML_DEPENDENCIES ggml-vulkan)
@@ -263,7 +208,8 @@ endif()
# it CPU matmuls run one dot product per output.
set(GGML_LLAMAFILE ON CACHE BOOL "ggml: use llamafile SGEMM")
-add_subdirectory(ggml EXCLUDE_FROM_ALL)
+include(cmake/llama_cpp.cmake)
+add_subdirectory("${NEMO_SPEECH_LLAMA_CPP_DIR}/ggml" "${CMAKE_BINARY_DIR}/ggml" EXCLUDE_FROM_ALL)
# ggml is an implementation dependency, so install only the runtime libraries
# required by our shared libraries. Its C++ headers, CMake package, pkg-config
@@ -367,14 +313,14 @@ endif()
# llama.cpp for the NMT decoder. It reuses the ggml target added above (its
# CMake builds its own ggml only when the target is absent), so one ggml backend
# is shared across asr/tts/nmt. Build only libllama.
-if(NEMO_SPEECH_WITH_NMT OR NEMO_SPEECH_BUILD_S2S)
+if(NEMO_SPEECH_BUILD_NMT OR NEMO_SPEECH_BUILD_S2S)
set(LLAMA_BUILD_COMMON OFF CACHE BOOL "" FORCE)
set(LLAMA_BUILD_TESTS OFF CACHE BOOL "" FORCE)
set(LLAMA_BUILD_EXAMPLES OFF CACHE BOOL "" FORCE)
set(LLAMA_BUILD_TOOLS OFF CACHE BOOL "" FORCE)
set(LLAMA_BUILD_SERVER OFF CACHE BOOL "" FORCE)
set(LLAMA_CURL OFF CACHE BOOL "" FORCE)
- add_subdirectory(llama.cpp EXCLUDE_FROM_ALL)
+ add_subdirectory("${NEMO_SPEECH_LLAMA_CPP_DIR}" "${CMAKE_BINARY_DIR}/llama.cpp" EXCLUDE_FROM_ALL)
set_property(TARGET llama PROPERTY PUBLIC_HEADER "")
install(TARGETS llama
RUNTIME DESTINATION ${CMAKE_INSTALL_BINDIR}
@@ -470,7 +416,7 @@ if(DEFINED VCPKG_INSTALLED_DIR AND DEFINED VCPKG_TARGET_TRIPLET)
RENAME LICENSE)
endforeach()
endif()
-install(FILES ggml/LICENSE
+install(FILES llama.cpp/LICENSE
DESTINATION "${NEMO_SPEECH_THIRD_PARTY_LICENSE_DIR}/ggml")
if(NEMO_SPEECH_BUILD_ASR AND NEMO_SPEECH_BUILD_CLI AND NEMO_SPEECH_BUILD_MIC_CAPTURE)
install(FILES third_party/miniaudio/LICENSE
diff --git a/CMakePresets.json b/CMakePresets.json
index 0d79eb5..dc6cbbe 100644
--- a/CMakePresets.json
+++ b/CMakePresets.json
@@ -15,7 +15,8 @@
"CMAKE_BUILD_TYPE": "Release",
"NEMO_SPEECH_BUILD_EXAMPLES": "OFF",
"NEMO_SPEECH_BUILD_TESTS": "OFF",
- "NEMO_SPEECH_BUILD_TOOLS": "OFF"
+ "NEMO_SPEECH_BUILD_TOOLS": "OFF",
+ "GGML_LLAMAFILE": "ON"
}
},
{
@@ -23,7 +24,7 @@
"inherits": "base",
"displayName": "CPU ASR and diarization",
"cacheVariables": {
- "NEMO_SPEECH_GGML_PATCHED": "OFF",
+ "NEMO_SPEECH_GGML_PATCHED": "ON",
"NEMO_SPEECH_BUILD_ASR": "ON",
"NEMO_SPEECH_BUILD_DIAR": "ON",
"NEMO_SPEECH_BUILD_TTS": "OFF",
@@ -35,7 +36,7 @@
"inherits": "base",
"displayName": "CPU standalone diarization",
"cacheVariables": {
- "NEMO_SPEECH_GGML_PATCHED": "OFF",
+ "NEMO_SPEECH_GGML_PATCHED": "ON",
"NEMO_SPEECH_BUILD_ASR": "OFF",
"NEMO_SPEECH_BUILD_DIAR": "ON",
"NEMO_SPEECH_BUILD_TTS": "OFF",
@@ -47,7 +48,7 @@
"inherits": "base",
"displayName": "CPU text-to-speech",
"cacheVariables": {
- "NEMO_SPEECH_GGML_PATCHED": "OFF",
+ "NEMO_SPEECH_GGML_PATCHED": "ON",
"NEMO_SPEECH_BUILD_ASR": "OFF",
"NEMO_SPEECH_BUILD_DIAR": "OFF",
"NEMO_SPEECH_BUILD_TTS": "ON",
@@ -59,7 +60,7 @@
"inherits": "base",
"displayName": "CPU text translation",
"cacheVariables": {
- "NEMO_SPEECH_GGML_PATCHED": "OFF",
+ "NEMO_SPEECH_GGML_PATCHED": "ON",
"NEMO_SPEECH_BUILD_ASR": "OFF",
"NEMO_SPEECH_BUILD_DIAR": "OFF",
"NEMO_SPEECH_BUILD_TTS": "OFF",
@@ -71,7 +72,7 @@
"inherits": "base",
"displayName": "CPU ASR, diarization, NMT, and TTS",
"cacheVariables": {
- "NEMO_SPEECH_GGML_PATCHED": "OFF",
+ "NEMO_SPEECH_GGML_PATCHED": "ON",
"NEMO_SPEECH_BUILD_ASR": "ON",
"NEMO_SPEECH_BUILD_DIAR": "ON",
"NEMO_SPEECH_BUILD_TTS": "ON",
@@ -128,7 +129,8 @@
"inherits": "cpu-asr",
"displayName": "Metal ASR and diarization",
"cacheVariables": {
- "GGML_METAL": "ON"
+ "GGML_METAL": "ON",
+ "NEMO_SPEECH_GGML_PATCHED": "OFF"
}
},
{
@@ -136,7 +138,8 @@
"inherits": "cpu-diar",
"displayName": "Metal standalone diarization",
"cacheVariables": {
- "GGML_METAL": "ON"
+ "GGML_METAL": "ON",
+ "NEMO_SPEECH_GGML_PATCHED": "OFF"
}
},
{
@@ -144,7 +147,8 @@
"inherits": "cpu-tts",
"displayName": "Metal text-to-speech",
"cacheVariables": {
- "GGML_METAL": "ON"
+ "GGML_METAL": "ON",
+ "NEMO_SPEECH_GGML_PATCHED": "OFF"
}
},
{
@@ -152,7 +156,8 @@
"inherits": "cpu-nmt",
"displayName": "Metal text translation",
"cacheVariables": {
- "GGML_METAL": "ON"
+ "GGML_METAL": "ON",
+ "NEMO_SPEECH_GGML_PATCHED": "OFF"
}
},
{
@@ -160,7 +165,8 @@
"inherits": "cpu-speech",
"displayName": "Metal ASR, diarization, NMT, and TTS",
"cacheVariables": {
- "GGML_METAL": "ON"
+ "GGML_METAL": "ON",
+ "NEMO_SPEECH_GGML_PATCHED": "OFF"
}
},
{
@@ -168,7 +174,8 @@
"inherits": "cpu-asr",
"displayName": "Vulkan ASR and diarization",
"cacheVariables": {
- "GGML_VULKAN": "ON"
+ "GGML_VULKAN": "ON",
+ "NEMO_SPEECH_GGML_PATCHED": "OFF"
}
},
{
@@ -176,7 +183,8 @@
"inherits": "cpu-diar",
"displayName": "Vulkan standalone diarization",
"cacheVariables": {
- "GGML_VULKAN": "ON"
+ "GGML_VULKAN": "ON",
+ "NEMO_SPEECH_GGML_PATCHED": "OFF"
}
},
{
@@ -184,7 +192,8 @@
"inherits": "cpu-tts",
"displayName": "Vulkan text-to-speech",
"cacheVariables": {
- "GGML_VULKAN": "ON"
+ "GGML_VULKAN": "ON",
+ "NEMO_SPEECH_GGML_PATCHED": "OFF"
}
},
{
@@ -192,7 +201,8 @@
"inherits": "cpu-nmt",
"displayName": "Vulkan text translation",
"cacheVariables": {
- "GGML_VULKAN": "ON"
+ "GGML_VULKAN": "ON",
+ "NEMO_SPEECH_GGML_PATCHED": "OFF"
}
},
{
@@ -200,7 +210,8 @@
"inherits": "cpu-speech",
"displayName": "Vulkan ASR, diarization, NMT, and TTS",
"cacheVariables": {
- "GGML_VULKAN": "ON"
+ "GGML_VULKAN": "ON",
+ "NEMO_SPEECH_GGML_PATCHED": "OFF"
}
},
{
@@ -211,9 +222,7 @@
"NEMO_SPEECH_BUILD_TTS": "ON",
"NEMO_SPEECH_BUILD_NMT": "ON",
"NEMO_SPEECH_BUILD_HTTP": "ON",
- "NEMO_SPEECH_BUILD_GRPC": "OFF",
- "NEMO_SPEECH_WITH_GRPC": "OFF",
- "NEMO_SPEECH_WITH_NMT": "ON"
+ "NEMO_SPEECH_BUILD_GRPC": "OFF"
}
},
{
@@ -224,9 +233,7 @@
"NEMO_SPEECH_BUILD_TTS": "ON",
"NEMO_SPEECH_BUILD_NMT": "ON",
"NEMO_SPEECH_BUILD_HTTP": "ON",
- "NEMO_SPEECH_BUILD_GRPC": "OFF",
- "NEMO_SPEECH_WITH_GRPC": "OFF",
- "NEMO_SPEECH_WITH_NMT": "ON"
+ "NEMO_SPEECH_BUILD_GRPC": "OFF"
}
},
{
@@ -239,9 +246,7 @@
"NEMO_SPEECH_BUILD_NMT": "OFF",
"NEMO_SPEECH_BUILD_S2S": "ON",
"NEMO_SPEECH_BUILD_HTTP": "ON",
- "NEMO_SPEECH_BUILD_GRPC": "OFF",
- "NEMO_SPEECH_WITH_GRPC": "OFF",
- "NEMO_SPEECH_WITH_NMT": "OFF"
+ "NEMO_SPEECH_BUILD_GRPC": "OFF"
}
},
{
@@ -252,9 +257,7 @@
"NEMO_SPEECH_BUILD_TTS": "ON",
"NEMO_SPEECH_BUILD_NMT": "ON",
"NEMO_SPEECH_BUILD_HTTP": "ON",
- "NEMO_SPEECH_BUILD_GRPC": "OFF",
- "NEMO_SPEECH_WITH_GRPC": "OFF",
- "NEMO_SPEECH_WITH_NMT": "ON"
+ "NEMO_SPEECH_BUILD_GRPC": "OFF"
}
},
{
@@ -265,9 +268,7 @@
"NEMO_SPEECH_BUILD_TTS": "ON",
"NEMO_SPEECH_BUILD_NMT": "ON",
"NEMO_SPEECH_BUILD_HTTP": "ON",
- "NEMO_SPEECH_BUILD_GRPC": "OFF",
- "NEMO_SPEECH_WITH_GRPC": "OFF",
- "NEMO_SPEECH_WITH_NMT": "ON"
+ "NEMO_SPEECH_BUILD_GRPC": "OFF"
}
},
{
@@ -278,8 +279,6 @@
"NEMO_SPEECH_BUILD_TTS": "ON",
"NEMO_SPEECH_BUILD_NMT": "ON",
"NEMO_SPEECH_BUILD_GRPC": "ON",
- "NEMO_SPEECH_WITH_NMT": "ON",
- "NEMO_SPEECH_WITH_GRPC": "ON",
"NEMO_SPEECH_WITH_FLASHLIGHT": "ON",
"NEMO_SPEECH_WITH_NORM": "ON",
"NEMO_SPEECH_TTS_WITH_JA": "ON",
diff --git a/CONTRIBUTING.md b/CONTRIBUTING.md
index f288c62..fa83270 100644
--- a/CONTRIBUTING.md
+++ b/CONTRIBUTING.md
@@ -24,7 +24,7 @@ Follow the [source-build guide](docs/build.md) for prerequisites and submodules.
For a model-independent CPU ASR test build:
```bash
-git submodule update --init ggml llama.cpp
+git submodule update --init llama.cpp
scripts/configure.sh cpu-asr -DNEMO_SPEECH_BUILD_TESTS=ON
cmake --build --preset cpu-asr
ctest --test-dir build/cpu-asr --output-on-failure
@@ -41,6 +41,12 @@ Use the closest matching CUDA, Metal, Vulkan, server, or component preset when
the change affects code outside the CPU ASR path. Include the commands and
results relevant to the change in the pull request.
+## Changing llama.cpp or ggml
+
+The `llama.cpp` submodule stays at its pinned upstream commit. Changes to it,
+including ggml, live as patches in [`patches/`](patches/README.md), which
+explains the patch format and the `scripts/llama-patches.sh` workflow.
+
## Contribution license and provenance
Unless a file states otherwise, contributions are submitted under the
diff --git a/README.md b/README.md
index 498c37a..a119077 100644
--- a/README.md
+++ b/README.md
@@ -156,7 +156,7 @@ development files, and the toolchain required by the selected backend, if any.
For a CUDA ASR and TTS server with the playground:
```bash
-git submodule update --init ggml llama.cpp third_party/cpp-httplib
+git submodule update --init llama.cpp third_party/cpp-httplib
scripts/configure.sh cuda-server
cmake --build --preset cuda-server
```
diff --git a/THIRD_PARTY_NOTICES.md b/THIRD_PARTY_NOTICES.md
index e15abb7..282bd20 100644
--- a/THIRD_PARTY_NOTICES.md
+++ b/THIRD_PARTY_NOTICES.md
@@ -9,10 +9,11 @@ checkouts and are summarized here.
### ggml
-- Source: [`ggml-org/ggml`](https://github.com/ggml-org/ggml)
-- Path: `ggml`
+- Source: [`ggml-org/llama.cpp`](https://github.com/ggml-org/llama.cpp) (`ggml/`,
+ developed and vendored in llama.cpp)
+- Path: `llama.cpp/ggml`
- Copyright (c) 2023-2026 The ggml authors
-- License: MIT; upstream text: [`ggml/LICENSE`](ggml/LICENSE)
+- License: MIT; upstream text: [`llama.cpp/LICENSE`](llama.cpp/LICENSE)
### llama.cpp
@@ -220,9 +221,9 @@ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
SOFTWARE.
-## NVIDIA ggml patches
+## NVIDIA llama.cpp and ggml patches
-The `ggml-patches/` directory contains NVIDIA-authored changes applied to the
-pinned MIT-licensed ggml source. New source files created by those patches
-carry the NVIDIA Apache-2.0 header; existing ggml files retain their upstream
-notices. The resulting combined source and binaries retain ggml's MIT notice.
+The `patches/` directory contains NVIDIA-authored changes applied to the pinned
+MIT-licensed llama.cpp and ggml source. New source files created by those patches
+carry the NVIDIA Apache-2.0 header; existing files retain their upstream
+notices. The resulting combined source and binaries retain the MIT notice.
diff --git a/app/CMakeLists.txt b/app/CMakeLists.txt
index 9b54fda..be7cfbc 100644
--- a/app/CMakeLists.txt
+++ b/app/CMakeLists.txt
@@ -31,18 +31,13 @@ if(NEMO_SPEECH_BUILD_ASR)
target_compile_definitions(nemo_speech_cli PRIVATE NEMO_SPEECH_CLI_ASR=1)
target_link_libraries(nemo_speech_cli PRIVATE nemo_speech_asr)
if(NEMO_SPEECH_BUILD_MIC_CAPTURE)
- if(NOT EXISTS "${CMAKE_SOURCE_DIR}/llama.cpp/vendor/miniaudio/miniaudio.h")
- message(FATAL_ERROR
- "ASR CLI microphone capture requires the vendored miniaudio header; "
- "run: git submodule update --init llama.cpp")
- endif()
target_sources(nemo_speech_cli PRIVATE
live_terminal.cpp
live_transcript.cpp
microphone_capture.cpp)
target_compile_definitions(nemo_speech_cli PRIVATE NEMO_SPEECH_CLI_LIVE=1)
target_include_directories(nemo_speech_cli PRIVATE
- ${CMAKE_SOURCE_DIR}/llama.cpp/vendor/miniaudio)
+ ${NEMO_SPEECH_LLAMA_CPP_DIR}/vendor/miniaudio)
target_link_libraries(nemo_speech_cli PRIVATE ${CMAKE_DL_LIBS})
if(APPLE)
target_link_libraries(nemo_speech_cli PRIVATE
diff --git a/cmake/llama_cpp.cmake b/cmake/llama_cpp.cmake
new file mode 100644
index 0000000..050db26
--- /dev/null
+++ b/cmake/llama_cpp.cmake
@@ -0,0 +1,199 @@
+# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved.
+# SPDX-License-Identifier: Apache-2.0
+# Resolve the llama.cpp source tree, which also provides ggml.
+#
+# The llama.cpp submodule is kept pristine. When NEMO_SPEECH_GGML_PATCHED is ON
+# the patches/ series is applied to a copy of its pinned commit in the build
+# directory, and the copy is refreshed only when the series or the pin changes.
+# A refresh rewrites only the files whose content changed, so editing one patch
+# rebuilds only what that edit affects.
+#
+# Also runs as a script, for tools that need the patched tree outside a build:
+# cmake -DSOURCE_DIR=llama.cpp -DPATCH_DIR=patches -DDEST_DIR=
-P cmake/llama_cpp.cmake
+
+# The patches listed in PATCH_DIR/series, in apply order. Every *.patch file in
+# PATCH_DIR must be listed, so a new patch cannot be silently left out.
+function(nemo_speech_llama_cpp_series patch_dir out_var)
+ if(NOT EXISTS "${patch_dir}/series")
+ message(FATAL_ERROR "missing ${patch_dir}/series")
+ endif()
+ file(STRINGS "${patch_dir}/series" lines)
+ set(patches "")
+ foreach(line IN LISTS lines)
+ string(REGEX REPLACE "#.*$" "" line "${line}")
+ string(STRIP "${line}" line)
+ if(line STREQUAL "")
+ continue()
+ endif()
+ if(NOT EXISTS "${patch_dir}/${line}")
+ message(FATAL_ERROR "${patch_dir}/series lists ${line}, which does not exist")
+ endif()
+ list(APPEND patches "${patch_dir}/${line}")
+ endforeach()
+ file(GLOB present LIST_DIRECTORIES false "${patch_dir}/*.patch")
+ foreach(patch IN LISTS present)
+ list(FIND patches "${patch}" index)
+ if(index EQUAL -1)
+ message(FATAL_ERROR "${patch} is not listed in ${patch_dir}/series")
+ endif()
+ endforeach()
+ set(${out_var} "${patches}" PARENT_SCOPE)
+endfunction()
+
+function(nemo_speech_materialize_llama_cpp source_dir patch_dir dest_dir)
+ find_package(Git QUIET)
+ if(NOT GIT_EXECUTABLE)
+ message(FATAL_ERROR "git is required to apply ${patch_dir} to llama.cpp")
+ endif()
+
+ nemo_speech_llama_cpp_series("${patch_dir}" patches)
+
+ # Key the copy by the pinned commit and the series content. Source archives
+ # without git metadata are keyed by the content of every file the copy
+ # takes, so replacing the tree with another llama.cpp version refreshes it.
+ execute_process(
+ COMMAND "${GIT_EXECUTABLE}" -C "${source_dir}" rev-parse HEAD
+ OUTPUT_VARIABLE base RESULT_VARIABLE rc OUTPUT_STRIP_TRAILING_WHITESPACE ERROR_QUIET)
+ set(have_git_metadata OFF)
+ if(rc EQUAL 0)
+ set(have_git_metadata ON)
+ else()
+ file(GLOB_RECURSE tree_files RELATIVE "${source_dir}" LIST_DIRECTORIES false "${source_dir}/*")
+ list(FILTER tree_files EXCLUDE REGEX "^(\\.git|models|docs|media)(/|$)")
+ list(SORT tree_files)
+ set(tree_listing "")
+ foreach(path IN LISTS tree_files)
+ file(SHA256 "${source_dir}/${path}" digest)
+ string(APPEND tree_listing "${path} ${digest}\n")
+ endforeach()
+ string(SHA256 base "${tree_listing}")
+ endif()
+ set(stamp_input "${base}")
+ foreach(patch IN LISTS patches)
+ file(SHA256 "${patch}" digest)
+ string(APPEND stamp_input "-${digest}")
+ endforeach()
+ string(SHA256 stamp "${stamp_input}")
+
+ set(stamp_file "${dest_dir}.stamp")
+ if(EXISTS "${stamp_file}" AND EXISTS "${dest_dir}")
+ file(READ "${stamp_file}" previous_stamp)
+ if(previous_stamp STREQUAL stamp)
+ return()
+ endif()
+ endif()
+ file(REMOVE "${stamp_file}")
+
+ list(LENGTH patches n_patches)
+ message(STATUS "Applying ${n_patches} patches from ${patch_dir} to llama.cpp in ${dest_dir}")
+ # Build the new tree in a staging directory, then copy over only what changed.
+ set(staging "${dest_dir}.staging")
+ file(REMOVE_RECURSE "${staging}")
+ file(MAKE_DIRECTORY "${staging}")
+ # Model and documentation assets are not needed to build.
+ if(have_git_metadata)
+ # Only tracked files at the pinned commit: local edits and build
+ # directories inside the submodule stay out of the copy.
+ execute_process(
+ COMMAND "${GIT_EXECUTABLE}" -C "${source_dir}" archive --format=tar -o "${staging}.tar" HEAD
+ -- . ":(exclude)models" ":(exclude)docs" ":(exclude)media"
+ RESULT_VARIABLE rc)
+ if(rc EQUAL 0)
+ execute_process(COMMAND "${CMAKE_COMMAND}" -E tar xf "${staging}.tar"
+ WORKING_DIRECTORY "${staging}" RESULT_VARIABLE rc)
+ endif()
+ file(REMOVE "${staging}.tar")
+ if(NOT rc EQUAL 0)
+ message(FATAL_ERROR "could not export llama.cpp ${base} from ${source_dir}")
+ endif()
+ else()
+ file(GLOB entries RELATIVE "${source_dir}" "${source_dir}/*")
+ foreach(entry IN LISTS entries)
+ if(NOT entry MATCHES "^(\\.git|models|docs|media)$")
+ file(COPY "${source_dir}/${entry}" DESTINATION "${staging}")
+ endif()
+ endforeach()
+ endif()
+
+ # A build directory usually sits inside another git work tree; stop git from
+ # discovering it so paths apply relative to the copy.
+ get_filename_component(dest_parent "${dest_dir}" DIRECTORY)
+ set(normalized "${staging}.patch")
+ foreach(patch IN LISTS patches)
+ # Windows checkouts may carry CRLF line endings.
+ file(READ "${patch}" content)
+ string(REPLACE "\r\n" "\n" content "${content}")
+ file(WRITE "${normalized}" "${content}")
+ execute_process(
+ COMMAND "${CMAKE_COMMAND}" -E env "GIT_CEILING_DIRECTORIES=${dest_parent}"
+ "${GIT_EXECUTABLE}" apply --whitespace=nowarn "${normalized}"
+ WORKING_DIRECTORY "${staging}"
+ RESULT_VARIABLE rc ERROR_VARIABLE error)
+ if(NOT rc EQUAL 0)
+ get_filename_component(name "${patch}" NAME)
+ message(FATAL_ERROR
+ "${name} does not apply to llama.cpp ${base}:\n${error}\n"
+ "Restore the pinned submodule (git submodule update llama.cpp) or rebase the "
+ "series with scripts/llama-patches.sh rebase.")
+ endif()
+ endforeach()
+ file(REMOVE "${normalized}")
+
+ # Unchanged files keep their timestamps, so the build recompiles only what changed.
+ file(GLOB_RECURSE new_files RELATIVE "${staging}" LIST_DIRECTORIES false "${staging}/*")
+ foreach(path IN LISTS new_files)
+ get_filename_component(dir "${dest_dir}/${path}" DIRECTORY)
+ file(MAKE_DIRECTORY "${dir}")
+ file(COPY_FILE "${staging}/${path}" "${dest_dir}/${path}" ONLY_IF_DIFFERENT)
+ endforeach()
+ file(GLOB_RECURSE old_files RELATIVE "${dest_dir}" LIST_DIRECTORIES false "${dest_dir}/*")
+ foreach(path IN LISTS old_files)
+ if(NOT EXISTS "${staging}/${path}")
+ file(REMOVE "${dest_dir}/${path}")
+ endif()
+ endforeach()
+ file(REMOVE_RECURSE "${staging}")
+ file(WRITE "${stamp_file}" "${stamp}")
+endfunction()
+
+if(CMAKE_SCRIPT_MODE_FILE)
+ foreach(var SOURCE_DIR PATCH_DIR DEST_DIR)
+ if(NOT DEFINED ${var})
+ message(FATAL_ERROR "usage: cmake -DSOURCE_DIR=... -DPATCH_DIR=... -DDEST_DIR=... -P ${CMAKE_SCRIPT_MODE_FILE}")
+ endif()
+ get_filename_component(${var} "${${var}}" ABSOLUTE)
+ endforeach()
+ nemo_speech_materialize_llama_cpp("${SOURCE_DIR}" "${PATCH_DIR}" "${DEST_DIR}")
+ return()
+endif()
+
+set(NEMO_SPEECH_LLAMA_CPP_SOURCE_DIR "" CACHE PATH
+ "Build ggml and llama.cpp from this tree as-is (for example a scripts/llama-patches.sh edit worktree)")
+set(_nemo_speech_llama_cpp_submodule "${CMAKE_SOURCE_DIR}/llama.cpp")
+if(NEMO_SPEECH_LLAMA_CPP_SOURCE_DIR)
+ set(NEMO_SPEECH_LLAMA_CPP_DIR "${NEMO_SPEECH_LLAMA_CPP_SOURCE_DIR}")
+elseif(NOT EXISTS "${_nemo_speech_llama_cpp_submodule}/ggml/CMakeLists.txt")
+ message(FATAL_ERROR
+ "The llama.cpp submodule (which also provides ggml) is not initialized.\n"
+ "Run: git submodule update --init llama.cpp")
+elseif(NEMO_SPEECH_GGML_PATCHED)
+ set(NEMO_SPEECH_LLAMA_CPP_DIR "${CMAKE_BINARY_DIR}/_deps/llama.cpp")
+ nemo_speech_materialize_llama_cpp(
+ "${_nemo_speech_llama_cpp_submodule}" "${CMAKE_SOURCE_DIR}/patches" "${NEMO_SPEECH_LLAMA_CPP_DIR}")
+ # Re-run configuration when patches are edited, added, removed, or reordered,
+ # or when the submodule moves to another commit.
+ nemo_speech_llama_cpp_series("${CMAKE_SOURCE_DIR}/patches" _nemo_speech_patches)
+ set_property(DIRECTORY APPEND PROPERTY CMAKE_CONFIGURE_DEPENDS
+ "${CMAKE_SOURCE_DIR}/patches" "${CMAKE_SOURCE_DIR}/patches/series" ${_nemo_speech_patches})
+ execute_process(
+ COMMAND "${GIT_EXECUTABLE}" -C "${_nemo_speech_llama_cpp_submodule}" rev-parse --absolute-git-dir
+ OUTPUT_VARIABLE _nemo_speech_llama_cpp_git_dir RESULT_VARIABLE _nemo_speech_rc
+ OUTPUT_STRIP_TRAILING_WHITESPACE ERROR_QUIET)
+ if(_nemo_speech_rc EQUAL 0 AND EXISTS "${_nemo_speech_llama_cpp_git_dir}/HEAD")
+ set_property(DIRECTORY APPEND PROPERTY CMAKE_CONFIGURE_DEPENDS
+ "${_nemo_speech_llama_cpp_git_dir}/HEAD")
+ endif()
+else()
+ set(NEMO_SPEECH_LLAMA_CPP_DIR "${_nemo_speech_llama_cpp_submodule}")
+endif()
+message(STATUS "ggml and llama.cpp source: ${NEMO_SPEECH_LLAMA_CPP_DIR}")
diff --git a/conversion/s2s_components/voicechat_source.py b/conversion/s2s_components/voicechat_source.py
index 7bf4c85..b5e1e41 100644
--- a/conversion/s2s_components/voicechat_source.py
+++ b/conversion/s2s_components/voicechat_source.py
@@ -265,18 +265,36 @@ def ensure_quantizer(
"C++ compiler; install the build prerequisites or pass --llama-quantize PATH"
)
+ source = checkout
if checkout == (root / "llama.cpp").resolve():
- patch_script = root / "scripts" / "apply-llama-patches.sh"
- if not patch_script.is_file():
- raise RuntimeError(f"missing llama.cpp patch helper: {patch_script}")
- print("[convert-s2s] applying pinned llama.cpp compatibility patches")
- subprocess.run(["bash", str(patch_script)], cwd=root, check=True)
+ # Build from a patched copy; the submodule itself stays pristine.
+ materialize = root / "cmake" / "llama_cpp.cmake"
+ if not materialize.is_file():
+ raise RuntimeError(f"missing llama.cpp patch helper: {materialize}")
+ source = root / ".deps" / "llama.cpp-patched"
+ print("[convert-s2s] applying patches/ to a copy of the pinned llama.cpp")
+ subprocess.run(
+ [
+ cmake,
+ f"-DSOURCE_DIR={checkout}",
+ f"-DPATCH_DIR={root / 'patches'}",
+ f"-DDEST_DIR={source}",
+ "-P",
+ str(materialize),
+ ],
+ check=True,
+ )
build_dir = checkout / "build-quantize"
+ cache = build_dir / "CMakeCache.txt"
+ if cache.is_file() and f"CMAKE_HOME_DIRECTORY:INTERNAL={source}\n" not in cache.read_text():
+ # Configured from a different source tree; CMake cannot switch sources
+ # in place.
+ shutil.rmtree(build_dir)
configure = [
cmake,
"-S",
- str(checkout),
+ str(source),
"-B",
str(build_dir),
"-DCMAKE_BUILD_TYPE=Release",
diff --git a/docker/Dockerfile b/docker/Dockerfile
index cfcb151..7a7fdde 100644
--- a/docker/Dockerfile
+++ b/docker/Dockerfile
@@ -39,9 +39,8 @@ ARG ENABLE_TESTS=OFF
ARG ENABLE_NMT=OFF
# Full-duplex VoiceChat via llama.cpp. Set ON to build realtime server support.
ARG ENABLE_S2S=OFF
-# Apply the project ggml patches and build the ASR against them. Set OFF to
-# build the ASR against STANDARD upstream ggml (stock ops only, no patches):
-# skips apply-ggml-patches.sh and sets -DNEMO_SPEECH_GGML_PATCHED=OFF.
+# Build ggml and llama.cpp with patches/ applied (CMake applies them). Set OFF to
+# build against pristine upstream llama.cpp (stock ggml ops only).
ARG ENABLE_GGML_PATCHES=ON
ARG JOBS=
# GPU arch(es) for ggml-cuda. Empty = ggml's portable default (sm_86/89/120/121
@@ -120,12 +119,9 @@ COPY README.md CONTRIBUTING.md /work/
COPY docs /work/docs
COPY config /work/config
COPY models/index.json /work/models/index.json
-# Core submodules: ggml (every backend) and llama.cpp (NMT and VoiceChat;
-# copied unconditionally so the layer cache is stable).
-COPY ggml /work/ggml
+# llama.cpp provides ggml for every backend (and libllama for NMT and VoiceChat).
COPY llama.cpp /work/llama.cpp
-COPY ggml-patches /work/ggml-patches
-COPY llama-patches /work/llama-patches
+COPY patches /work/patches
COPY proto /work/proto
COPY include /work/include
COPY src /work/src
@@ -161,19 +157,6 @@ RUN if [ "${ENABLE_FLASHLIGHT}" = "ON" ]; then \
&& rm -rf /work/.deps/sentencepiece-build; \
fi
-# Apply the project ggml patches (fused rel-pos attention, NVFP4 quantization,
-# norm fusion, dw-conv F16, skinny-q8, BF16 fusions, and CUDA correctness) onto the clean
-# vendored ggml. Idempotent: a working tree that already has them applied is
-# detected and skipped. Skipped entirely when ENABLE_GGML_PATCHES=OFF — the ASR
-# then builds against standard upstream ggml (see NEMO_SPEECH_GGML_PATCHED).
-RUN if [ "${ENABLE_GGML_PATCHES}" = "ON" ]; then bash /work/scripts/apply-ggml-patches.sh; fi
-
-# NMT and VoiceChat rely on the project's llama.cpp compatibility and
-# quantization patches. The patch script is idempotent for prepatched trees.
-RUN if [ "${ENABLE_NMT}" = "ON" ] || [ "${ENABLE_S2S}" = "ON" ]; then \
- bash /work/scripts/apply-llama-patches.sh; \
- fi
-
# NCCL disabled: single-GPU inference does not use multi-GPU collectives,
# and the linked libnccl is otherwise dead runtime weight.
# NEMO_SPEECH_CUBLAS_SHIM=ON builds the drop-in libcublas.so.
@@ -229,7 +212,7 @@ RUN mkdir -p /out/bin /out/lib /out/share/nemo-speech \
&& install -Dm0644 /work/build/share/nemo-speech/model-index.json \
/out/share/nemo-speech/model-index.json \
&& license_dir=/out/share/licenses/nemo-speech/third_party \
- && install -Dm0644 /work/ggml/LICENSE "$license_dir/ggml/LICENSE" \
+ && install -Dm0644 /work/llama.cpp/LICENSE "$license_dir/ggml/LICENSE" \
&& install -Dm0644 /work/llama.cpp/LICENSE "$license_dir/llama.cpp/LICENSE" \
&& if [ "${ENABLE_GRPC}" = "ON" ]; then \
install -Dm0644 /work/proto/riva-common/LICENSE \
diff --git a/docs/README.md b/docs/README.md
index b98cee9..2b4b492 100644
--- a/docs/README.md
+++ b/docs/README.md
@@ -50,9 +50,10 @@ Start with:
## Developer guide
- [Overview](development/README.md) - implementation and performance internals.
-- [Diagnostics](development/diagnostics.md) - `check_backend_coverage`.
+- [Diagnostics](development/diagnostics.md) - build switches, runtime knobs, and
+ `check_backend_coverage`.
- [ASR batching](development/asr-batching.md) - neural microbatching and
streaming-state arenas.
-- [ggml patches](development/ggml-patches.md) - the project-specific ggml changes.
+- [llama.cpp and ggml patches](../patches/README.md) - the project-specific changes.
- [cuBLAS shim](development/cublas-shim.md) - the in-tree drop-in cuBLAS
replacement.
diff --git a/docs/build.md b/docs/build.md
index e313d72..1c8d91e 100644
--- a/docs/build.md
+++ b/docs/build.md
@@ -72,9 +72,8 @@ only when selecting those features; see the platform sections below.
Initialize the submodules needed by the selected components:
```bash
-git submodule update --init ggml
+git submodule update --init llama.cpp # required (also provides ggml)
git submodule update --init third_party/cpp-httplib # HTTP server only
-git submodule update --init llama.cpp # ASR live capture, NMT, or VoiceChat
git submodule update --init proto/riva-common # gRPC only
git submodule update --init third_party/flashlight-text third_party/kenlm # Flashlight only
git submodule update --init third_party/open_jtalk # Japanese TTS only
@@ -82,8 +81,8 @@ git submodule update --init --recursive third_party/cppjieba # Mandarin TTS onl
```
Always configure through `scripts/configure.sh`. It checks the required
-submodules and applies the patches needed by the selected preset. CUDA presets
-apply the pinned patches from `ggml-patches/` in order. Mandarin TTS also
+submodules and runs the preset. CMake applies the project's llama.cpp and ggml
+changes from `patches/` itself, so no separate patch step is needed. Mandarin TTS also
requires the Git LFS files under `src/tts/tokenizer/mandarin_data/`; the helper
reports any files that are still LFS pointers. Materialize them with
`git lfs pull --include='src/tts/tokenizer/mandarin_data/*'`.
@@ -126,8 +125,8 @@ scripts/configure.sh cuda-server -DNEMO_SPEECH_WITH_NORM=ON
cmake --build --preset cuda-server
```
-For a manual configuration, explicitly disable the patched CUDA paths when
-building against stock ggml:
+A raw CMake configuration also applies `patches/` automatically. To build
+against pristine upstream llama.cpp and ggml instead:
```bash
cmake -S . -B build -G Ninja \
@@ -136,8 +135,8 @@ cmake -S . -B build -G Ninja \
cmake --build build -j"$(nproc)"
```
-See [ggml patches](development/ggml-patches.md) for the patched and stock
-runtime tradeoffs.
+See [`patches/README.md`](../patches/README.md) for what the patches change and
+how to edit them.
## Components
diff --git a/docs/development/README.md b/docs/development/README.md
index c6598c2..bc3896f 100644
--- a/docs/development/README.md
+++ b/docs/development/README.md
@@ -7,12 +7,12 @@ the server want [ASR configuration](../asr/configuration.md),
## Contents
-- [`diagnostics.md`](diagnostics.md) - `check_backend_coverage`, catching silent
- CPU fallbacks on a new backend.
+- [`diagnostics.md`](diagnostics.md) - build switches, runtime knobs, and
+ `check_backend_coverage` for catching silent CPU fallbacks on a new backend.
- [`asr-batching.md`](asr-batching.md) - exact-shape neural microbatching and
indexed streaming-state arenas.
-- [`ggml-patches.md`](ggml-patches.md) - the project-specific ggml patches and how
- they are applied at build setup.
+- [`patches/README.md`](../../patches/README.md) - the project's llama.cpp and ggml
+ patches, how builds apply them, and how to edit them.
- [`cublas-shim.md`](cublas-shim.md) - the in-tree drop-in cuBLAS replacement
under `kernels/` and where the custom GPU kernels live.
- [Windows build notes](windows-build.md)
diff --git a/docs/development/asr-batching.md b/docs/development/asr-batching.md
index a6b9141..d422d4d 100644
--- a/docs/development/asr-batching.md
+++ b/docs/development/asr-batching.md
@@ -134,6 +134,6 @@ summaries.
submissions rather than concurrent access to ggml's shared scheduler.
- Disabling batching keeps the scalar path and avoids its queue-delay cost.
-For planar-Q8 kernel behavior, diagnostic environment switches, and patched
-versus stock ggml builds, see [ggml patches](ggml-patches.md). The complete
+For planar-Q8 kernel behavior and patched versus stock ggml builds, see
+[`patches/README.md`](../../patches/README.md) and the patch descriptions there. The complete
runtime key reference is in [ASR configuration](../asr/configuration.md).
diff --git a/docs/development/cublas-shim.md b/docs/development/cublas-shim.md
index 4e6a02a..491129f 100644
--- a/docs/development/cublas-shim.md
+++ b/docs/development/cublas-shim.md
@@ -49,5 +49,5 @@ On Windows:
The heavier project-specific CUDA kernels (fused rel-pos attention, skinny-Q8 GEMM,
NVFP4 quantization, BF16 FastConformer epilogues, fused LayerNorm, and F16
depthwise conv2d) live as ggml patches rather than in `kernels/` - see
-[ggml patches](ggml-patches.md). `kernels/` holds only the cuBLAS shim and its
+[`patches/`](../../patches/README.md). `kernels/` holds only the cuBLAS shim and its
version-map template.
diff --git a/docs/development/diagnostics.md b/docs/development/diagnostics.md
index d4c378a..8bd484f 100644
--- a/docs/development/diagnostics.md
+++ b/docs/development/diagnostics.md
@@ -1,13 +1,52 @@
-# Backend coverage diagnostic
+# Build switches, runtime knobs, and diagnostics
+
+Every switch that changes which code path runs, in one place. Defaults are what
+you want; the rest exist for bisection and debugging.
+
+## Build
+
+| CMake option | Default | Effect |
+|---|---|---|
+| `NEMO_SPEECH_GGML_PATCHED` | `ON` (`cuda-*`, `cpu-*` presets), `OFF` (`metal-*`, `vulkan-*`) | Apply [`patches/`](../../patches/README.md) to llama.cpp and ggml and, on CUDA, use the fused kernels. `OFF` builds pristine upstream llama.cpp with stock ggml operations. |
+| `NEMO_SPEECH_LLAMA_CPP_SOURCE_DIR` | unset | Build ggml and llama.cpp from this tree as-is (for example a `scripts/llama-patches.sh edit` worktree). |
+| `GGML_CUDA_GRAPHS` | `ON` | Capture and replay CUDA graphs (upstream option; this project turns it on). |
+| `GGML_LLAMAFILE` | `ON` | Tiled CPU matrix multiplication (upstream option; this project turns it on). |
+| `NEMO_SPEECH_CUBLAS_SHIM` | `OFF` | Build the drop-in cuBLAS replacement ([cuBLAS shim](cublas-shim.md)). |
+
+Component options (`NEMO_SPEECH_BUILD_*`, `NEMO_SPEECH_WITH_*`) are listed in
+[the build guide](../build.md). `NEMO_SPEECH_WITH_NMT` and `NEMO_SPEECH_WITH_GRPC`
+are deprecated spellings of `NEMO_SPEECH_BUILD_NMT` and `NEMO_SPEECH_BUILD_GRPC`.
+
+## Runtime (patched CUDA backend)
+
+| Variable | Default | Effect |
+|---|---|---|
+| `GGML_CUDA_GRAPH_EVICT_AFTER_MS` | `10000`; `0` once a server loads TTS, and for VoiceChat | Drop CUDA graphs idle for this long; `0` keeps them. Read when a CUDA backend is created. |
+| `GGML_SKINNY_Q8=0` | enabled | Disable the skinny Q8_0 GEMM for block Q8_0 weights. Planar Q8 weights always use it. |
+| `GGML_SKINNY_Q8_INPLACE=0` | in place; `0` when ASR and NMT share a process | Keep repacked skinny-Q8 weights in a separate buffer. |
+
+Upstream switches that are useful for bisecting a CUDA problem:
+`GGML_CUDA_DISABLE_GRAPHS=1`, `GGML_CUDA_DISABLE_FUSION=1`, and
+`GGML_SCHED_DEBUG=2`.
+
+## Other variables
+
+- `NEMO_SPEECH_MODEL_DIR`, `NEMO_SPEECH_MODEL_INDEX`, `NEMO_SPEECH_HF_BASE_URL`: the CLI model store.
+- `NEMO_SPEECH_`: overrides one configuration key.
+- `S2S_*`: VoiceChat tuning; see [VoiceChat configuration](../s2s/configuration.md).
+- `MAGPIETTS_LOGIT_DUMP` and `MAGPIETTS_FORCE_CODES`: file paths used for MagpieTTS parity testing.
+- `EDGE_SHIM_TRACE_SHAPES`: logs every call in the cuBLAS shim.
+
+## Backend coverage
`check_backend_coverage` loads an ASR GGUF and exercises the frontend and
encoder Sessions used by its CTC or streaming-transducer path, including the
compact CTC head, RNNT/TDT predictor and joint, and cache-aware encoder when
applicable. It then prints their per-op backend assignment.
-Use it to catch **silent CPU fallbacks** when enabling a new GPU backend - a
-single fallback op mid-graph adds a GPU↔CPU roundtrip per audio chunk and can
-significantly increase streaming latency. The lazy offline transducer path is
-outside this diagnostic's coverage.
+Use it to catch **silent CPU fallbacks** when enabling a new GPU backend or
+updating llama.cpp - a single fallback op mid-graph adds a GPU↔CPU roundtrip per
+audio chunk and can significantly increase streaming latency. The lazy offline
+transducer path is outside this diagnostic's coverage.
```bash
scripts/configure.sh cuda-asr -DNEMO_SPEECH_BUILD_TOOLS=ON
diff --git a/docs/development/ggml-patches.md b/docs/development/ggml-patches.md
deleted file mode 100644
index f408919..0000000
--- a/docs/development/ggml-patches.md
+++ /dev/null
@@ -1,7 +0,0 @@
-# ggml patches
-
-The canonical documentation for the patch series, build modes, runtime gates,
-and patch-regeneration workflow is
-[`ggml-patches/README.md`](../../ggml-patches/README.md).
-
-This page intentionally contains no duplicated patch details.
diff --git a/docs/development/windows-build.md b/docs/development/windows-build.md
index 8993ca9..48c8eae 100644
--- a/docs/development/windows-build.md
+++ b/docs/development/windows-build.md
@@ -49,9 +49,8 @@ defaults.
## Get the sources
```powershell
-git submodule update --init ggml # required (all backends)
+git submodule update --init llama.cpp # required (also provides ggml)
git submodule update --init proto/riva-common # gRPC server
-git submodule update --init llama.cpp # ASR live capture or NMT
git submodule update --init third_party/flashlight-text third_party/kenlm # only for LM-fused CTC decoding
git submodule update --init third_party/open_jtalk # optional TTS JA tokenizer (-TtsJa)
git submodule update --init --recursive third_party/cppjieba # optional TTS ZH tokenizer (-TtsZh)
@@ -111,9 +110,7 @@ If you prefer to drive CMake yourself, run from an **x64 Native Tools** prompt
(or after `vcvars64.bat`), with CMake/Ninja/CUDA/Vulkan on `PATH`:
```powershell
-# CUDA: apply the CUDA-only ggml patches first
-powershell -ExecutionPolicy Bypass -File scripts\windows\apply-ggml-patches.ps1
-
+# CUDA: CMake applies patches/ itself (git must be on PATH).
cmake -S . -B build-cuda -G Ninja -DCMAKE_BUILD_TYPE=Release `
-DGGML_CUDA=ON -DCMAKE_CUDA_ARCHITECTURES=native `
-DNEMO_SPEECH_BUILD_GRPC=ON `
@@ -121,7 +118,7 @@ cmake -S . -B build-cuda -G Ninja -DCMAKE_BUILD_TYPE=Release `
-DVCPKG_TARGET_TRIPLET=x64-windows
cmake --build build-cuda --parallel
-# Vulkan: stock ggml (the project's ggml patches are CUDA-only). ggml-vulkan requires the
+# Vulkan: stock ggml (the patches are applied for CUDA and CPU builds). ggml-vulkan requires the
# SPIRV-Headers CMake package; the Vulkan SDK ships it under Lib\cmake.
cmake -S . -B build-vulkan -G Ninja -DCMAKE_BUILD_TYPE=Release `
-DGGML_VULKAN=ON -DNEMO_SPEECH_GGML_PATCHED=OFF `
@@ -136,25 +133,13 @@ cmake --build build-vulkan --parallel
- **The cuBLAS shim is optional.** Pass `-CublasShim` to the build driver for an
app-local `cublas64_.dll` that avoids shipping cuBLAS and cuBLASLt.
-- **ggml patches are CUDA-only.** A Vulkan/CPU build uses stock ggml; pass
- `-DNEMO_SPEECH_GGML_PATCHED=OFF` (the encoder uses the portable op path).
+- **CUDA and CPU builds apply the patches.** A Vulkan build uses stock ggml;
+ pass `-DNEMO_SPEECH_GGML_PATCHED=OFF` (the build driver does this).
- Dependent DLLs must be next to the executable or on `PATH`. Ninja places them
together in `build-\bin`.
- Flashlight builds install the replaceable `kenlm.dll` alongside the runtime
libraries.
-### Reset a partially patched ggml checkout
-
-If `apply-ggml-patches` reports that a patch does not apply cleanly, reset the
-submodule and re-apply it:
-
-```powershell
-git -C ggml reset -q
-git -C ggml checkout -- .
-git -C ggml clean -fd src # removes patch-created files
-powershell -ExecutionPolicy Bypass -File scripts\windows\apply-ggml-patches.ps1
-```
-
### Backend status
| Backend | ASR | TTS | NMT |
diff --git a/docs/s2s/README.md b/docs/s2s/README.md
index 23bd664..baebfec 100644
--- a/docs/s2s/README.md
+++ b/docs/s2s/README.md
@@ -27,7 +27,7 @@ See [Build from source](../build.md) for other platforms and toolchains.
```bash
git clone https://github.com/NVIDIA/NeMo-Speech.cpp.git
cd NeMo-Speech.cpp
-git submodule update --init ggml llama.cpp third_party/cpp-httplib
+git submodule update --init llama.cpp third_party/cpp-httplib
python3 -m venv .venv
. .venv/bin/activate
diff --git a/docs/tts/models.md b/docs/tts/models.md
index 36daab7..3a0dc27 100644
--- a/docs/tts/models.md
+++ b/docs/tts/models.md
@@ -16,7 +16,7 @@ options are omitted.
Hugging Face: [nvidia/magpie_tts_multilingual_357m](https://huggingface.co/nvidia/magpie_tts_multilingual_357m)
-Use `nemo-speech pull magpie` to download the GGUF and matching tokenizer
+Use `nemo-speech pull magpie` to download the F16 GGUF and matching tokenizer
assets pinned by the model index. For manually managed checkpoints, download
the GGUF and `.nemo` archive from the same Hugging Face revision and pass their
local paths explicitly.
@@ -26,8 +26,9 @@ frame-stacking factor of 2. This is independent of `tts.chunk-frames`, which
groups generated frames for NanoCodec streaming. Both versions use the same
NanoCodec decoder.
-The fused CUDA decode path needs a Q8_0 GGUF. Convert the `.nemo` locally
-(Q8_0 is the converter default):
+The fused CUDA decode path, which the published TTS benchmarks use, needs a
+Q8_0 GGUF; the pulled F16 GGUF runs the unfused path. Convert the `.nemo`
+locally (Q8_0 is the converter default):
```bash
python3 convert_model.py magpie_tts_multilingual_357m.nemo \
diff --git a/ggml b/ggml
deleted file mode 160000
index c03b4e2..0000000
--- a/ggml
+++ /dev/null
@@ -1 +0,0 @@
-Subproject commit c03b4e2bcece5134827881af90242086daf75be5
diff --git a/ggml-patches/0001-fused-relpos-attn.patch b/ggml-patches/0001-fused-relpos-attn.patch
deleted file mode 100644
index bd4c7cb..0000000
--- a/ggml-patches/0001-fused-relpos-attn.patch
+++ /dev/null
@@ -1,811 +0,0 @@
-diff --git a/include/ggml.h b/include/ggml.h
-index f6725265..82670deb 100644
---- a/include/ggml.h
-+++ b/include/ggml.h
-@@ -583,6 +583,8 @@ extern "C" {
-
- GGML_OP_GLU,
-
-+ GGML_OP_FUSED_RELPOS_ATTN,
-+
- GGML_OP_COUNT,
- };
-
-@@ -2416,6 +2418,43 @@ extern "C" {
- struct ggml_tensor * a,
- struct ggml_tensor * sinks);
-
-+ // Fused FastConformer relative-position multi-head attention.
-+ // Replaces the content (K*Qu) + position (P*Qv with rel-shift) + softmax +
-+ // context (attn*V) op sequence with one kernel. CUDA-only (CPU/other
-+ // backends report unsupported and the unfused graph runs instead).
-+ //
-+ // Logical shapes, head dim ne[0]=d_k fastest:
-+ // q [d_k, q_len, n_head, batch] pre-bias query (Qu/Qv added inside)
-+ // k [d_k, kv_len, n_head, batch]
-+ // v [d_k, kv_len, n_head, batch]
-+ // p [d_k, pos_len, n_head] positional encoding projection,
-+ // pos_len >= kv_len + q_len - 1
-+ // bias_u [d_k, n_head] pos_bias_u (content term)
-+ // bias_v [d_k, n_head] pos_bias_v (position term)
-+ // mask [kv_len] or [kv_len, batch] additive (0 / -inf) key mask,
-+ // or NULL shared or per-stream columns
-+ // Q/K/V/P may be non-contiguous views as long as each d_k row is
-+ // contiguous — e.g. Q sliced from a fused-QKV projection and K/V read
-+ // head-split from a feat-major [n_feat, kv] window (the CUDA op derives
-+ // all addressing from the tensors' nb[]). bias_u/bias_v/mask must be
-+ // contiguous.
-+ // Output: [d_k, q_len, n_head, batch] (attention context, pre-output-proj).
-+ // With merge_heads=true the output keeps that logical shape but uses a
-+ // head-merged memory layout: permute(out, 0, 2, 1, 3) is a contiguous
-+ // (n_feat, q_len, batch) matrix, consumable by the output projection
-+ // without a copy.
-+ GGML_API struct ggml_tensor * ggml_fused_relpos_attn(
-+ struct ggml_context * ctx,
-+ struct ggml_tensor * q,
-+ struct ggml_tensor * k,
-+ struct ggml_tensor * v,
-+ struct ggml_tensor * p,
-+ struct ggml_tensor * bias_u,
-+ struct ggml_tensor * bias_v,
-+ struct ggml_tensor * mask,
-+ float scale,
-+ bool merge_heads);
-+
- // TODO: needs to be adapted to ggml_flash_attn_ext
- GGML_API struct ggml_tensor * ggml_flash_attn_back(
- struct ggml_context * ctx,
-diff --git a/src/ggml-cpu/ggml-cpu.c b/src/ggml-cpu/ggml-cpu.c
-index cd5c61a8..cce0e5a8 100644
---- a/src/ggml-cpu/ggml-cpu.c
-+++ b/src/ggml-cpu/ggml-cpu.c
-@@ -1988,6 +1988,10 @@ static void ggml_compute_forward(struct ggml_compute_params * params, struct ggm
- {
- ggml_compute_forward_flash_attn_ext(params, tensor);
- } break;
-+ case GGML_OP_FUSED_RELPOS_ATTN:
-+ {
-+ GGML_ABORT("FUSED_RELPOS_ATTN has no CPU path (CUDA-only)");
-+ } break;
- case GGML_OP_FLASH_ATTN_BACK:
- {
- int32_t t = ggml_get_op_params_i32(tensor, 0);
-diff --git a/src/ggml-cpu/ggml-cpu.cpp b/src/ggml-cpu/ggml-cpu.cpp
-index 128883b4..7429cc45 100644
---- a/src/ggml-cpu/ggml-cpu.cpp
-+++ b/src/ggml-cpu/ggml-cpu.cpp
-@@ -439,6 +439,8 @@ static bool ggml_backend_cpu_device_supports_op(ggml_backend_dev_t dev, const st
- }
-
- switch (op->op) {
-+ case GGML_OP_FUSED_RELPOS_ATTN:
-+ return false; // CUDA-only; CPU falls back to the unfused graph
- case GGML_OP_CPY:
- case GGML_OP_SET_ROWS:
- return
-diff --git a/src/ggml-cuda/fused-relpos-attn.cu b/src/ggml-cuda/fused-relpos-attn.cu
-new file mode 100644
-index 00000000..f3c4836a
---- /dev/null
-+++ b/src/ggml-cuda/fused-relpos-attn.cu
-@@ -0,0 +1,600 @@
-+// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved.
-+// SPDX-License-Identifier: Apache-2.0
-+#include "fused-relpos-attn.cuh"
-+
-+#include
-+
-+// Fused FastConformer relative-position multi-head attention.
-+//
-+// One block per (head, query, batch); blockDim.x = d_k threads (one per head
-+// dim). Scores, rel-shifted position term, scale+mask, two-pass softmax, and
-+// the attn*V context are all computed in-kernel, so the rel-shift matrix and
-+// the score matrix are never materialized in global memory.
-+//
-+// Operand addressing is fully stride-driven (strides read from each tensor's
-+// nb[] by the host wrapper, in elements): Q/K/V may be non-contiguous views —
-+// e.g. the Q slice of a fused-QKV projection, or a feat-major [n_feat, kv]
-+// K/V window — as long as d_k stays innermost-contiguous (asserted by the op
-+// constructor; the vectorized loads rely on it). P is [d_k, pos_len, n_head]
-+// with pos_len = kv + q - 1 rows addressable; bu/bv are [d_k, n_head];
-+// mask is [kv] (shared) or [kv, batch] (per-stream) additive (0 / -inf), or NULL.
-+// Output ctx is [d_k, q, n_head, batch] logical; with the merge_heads op flag
-+// its memory layout is head-merged ([d_k+h*d_k] innermost, i.e. a plain
-+// (n_feat, q, batch) matrix), so the output projection consumes it without a
-+// permute copy.
-+//
-+// Requires d_k to be a power of two (the softmax reduction halves blockDim.x).
-+
-+// K/V/P are templated: F16 operands halve the dominant re-read traffic when
-+// the caller stages them; F32 operands skip the staging casts entirely. All
-+// math stays in F32 either way.
-+
-+static constexpr int RELPOS_ATTN_DK_128 = 128;
-+static constexpr int RELPOS_ATTN_WARPS_128 = RELPOS_ATTN_DK_128 / 32;
-+static constexpr int RELPOS_ATTN_CC_SM100 = 1000;
-+
-+static __device__ __forceinline__ float relpos_warp_sum(float value) {
-+#pragma unroll
-+ for (int offset = 16; offset > 0; offset >>= 1) {
-+ value += __shfl_down_sync(0xffffffff, value, offset);
-+ }
-+ return value;
-+}
-+
-+static __device__ __forceinline__ float4 relpos_load4(const float * ptr) {
-+ return *reinterpret_cast(ptr);
-+}
-+
-+static __device__ __forceinline__ float4 relpos_load4(const half * ptr) {
-+ const int2 packed = *reinterpret_cast(ptr);
-+ const half2 * values = reinterpret_cast(&packed);
-+ return make_float4(
-+ __low2float(values[0]), __high2float(values[0]),
-+ __low2float(values[1]), __high2float(values[1]));
-+}
-+
-+// SM100 streaming specialization for d_k=128 and q=2. Keeping one block per
-+// query preserves the two rows' parallelism while the complete grid fits in a
-+// resident wave. Compared with the generic shared-memory reduction tree, the
-+// score-producing warps retain their local maxima and max/sum use only four
-+// warp partials. This removes fourteen block barriers and 124 scratch floats.
-+template
-+static __global__ void fused_relpos_attn_warp_128_kernel(
-+ const float * __restrict__ Q, const T * __restrict__ K,
-+ const T * __restrict__ V, const T * __restrict__ Ppos,
-+ const float * __restrict__ bu, const float * __restrict__ bv,
-+ const float * __restrict__ mask, float * __restrict__ ctx,
-+ int kv, float scale,
-+ long q_sq, long q_sh, long q_sb,
-+ long k_sj, long k_sh, long k_sb,
-+ long v_sj, long v_sh, long v_sb,
-+ long p_sr, long p_sh,
-+ long o_si, long o_sh, long o_sb, long m_sb) {
-+ extern __shared__ float sh[];
-+ float * Qu = sh;
-+ float * Qv = Qu + RELPOS_ATTN_DK_128;
-+ float * sc = Qv + RELPOS_ATTN_DK_128;
-+ float * red = sc + kv;
-+
-+ const int h = blockIdx.x;
-+ const int i = blockIdx.y;
-+ const int b = blockIdx.z;
-+ const int d = threadIdx.x;
-+ const int warp = d >> 5;
-+ const int lane = d & 31;
-+
-+ const float * Qhi = Q + (size_t) b * q_sb + (size_t) h * q_sh + (size_t) i * q_sq;
-+ const T * Kh = K + (size_t) b * k_sb + (size_t) h * k_sh;
-+ const T * Vh = V + (size_t) b * v_sb + (size_t) h * v_sh;
-+ const T * Ph = Ppos + (size_t) h * p_sh;
-+ const float * Mb = mask ? mask + (size_t) b * m_sb : nullptr;
-+
-+ Qu[d] = Qhi[d] + bu[h * RELPOS_ATTN_DK_128 + d];
-+ Qv[d] = Qhi[d] + bv[h * RELPOS_ATTN_DK_128 + d];
-+ __syncthreads();
-+
-+ const int d4 = lane * 4;
-+ const float4 qu4 = *reinterpret_cast(Qu + d4);
-+ const float4 qv4 = *reinterpret_cast(Qv + d4);
-+ float produced_max = -INFINITY;
-+ for (int j = warp; j < kv; j += RELPOS_ATTN_WARPS_128) {
-+ const float4 k4 = relpos_load4(Kh + (size_t) j * k_sj + d4);
-+ const int row = 1 + j - i;
-+ const float4 p4 = relpos_load4(Ph + (size_t) row * p_sr + d4);
-+ float score =
-+ k4.x * qu4.x + p4.x * qv4.x +
-+ k4.y * qu4.y + p4.y * qv4.y +
-+ k4.z * qu4.z + p4.z * qv4.z +
-+ k4.w * qu4.w + p4.w * qv4.w;
-+ score = relpos_warp_sum(score);
-+ if (lane == 0) {
-+ sc[j] = score * scale + (Mb ? Mb[j] : 0.0f);
-+ produced_max = fmaxf(produced_max, sc[j]);
-+ }
-+ }
-+ if (lane == 0) {
-+ red[warp] = produced_max;
-+ }
-+ __syncthreads();
-+
-+ if (d == 0) {
-+ float maximum = red[0];
-+#pragma unroll
-+ for (int w = 1; w < RELPOS_ATTN_WARPS_128; ++w) {
-+ maximum = fmaxf(maximum, red[w]);
-+ }
-+ red[0] = maximum;
-+ }
-+ __syncthreads();
-+ const float maximum = red[0];
-+
-+ // red[] is reused below. Ensure every warp has captured red[0] first;
-+ // otherwise warp 0 can overwrite the maximum while another warp loads it.
-+ __syncthreads();
-+ float local_sum = 0.0f;
-+ for (int j = d; j < kv; j += RELPOS_ATTN_DK_128) {
-+ const float weight = __expf(sc[j] - maximum);
-+ sc[j] = weight;
-+ local_sum += weight;
-+ }
-+ local_sum = relpos_warp_sum(local_sum);
-+ if (lane == 0) {
-+ red[warp] = local_sum;
-+ }
-+ __syncthreads();
-+ if (d == 0) {
-+ float sum = red[0];
-+#pragma unroll
-+ for (int w = 1; w < RELPOS_ATTN_WARPS_128; ++w) {
-+ sum += red[w];
-+ }
-+ red[0] = sum;
-+ }
-+ __syncthreads();
-+
-+ float context = 0.0f;
-+ for (int j = 0; j < kv; ++j) {
-+ context += sc[j] * (float) Vh[(size_t) j * v_sj + d];
-+ }
-+ ctx[(size_t) b * o_sb + (size_t) i * o_si + (size_t) h * o_sh + d] = context / red[0];
-+}
-+
-+// Once the one-block-per-query grid spills into another occupancy wave, one
-+// block computes both query rows. K and V are then loaded once and reused;
-+// only the two relative-position rows differ. The reduction order matches the
-+// one-query specialization so changing batch size does not change numerics.
-+template
-+static __global__ void fused_relpos_attn_q2_warp_128_kernel(
-+ const float * __restrict__ Q, const T * __restrict__ K,
-+ const T * __restrict__ V, const T * __restrict__ Ppos,
-+ const float * __restrict__ bu, const float * __restrict__ bv,
-+ const float * __restrict__ mask, float * __restrict__ ctx,
-+ int kv, float scale,
-+ long q_sq, long q_sh, long q_sb,
-+ long k_sj, long k_sh, long k_sb,
-+ long v_sj, long v_sh, long v_sb,
-+ long p_sr, long p_sh,
-+ long o_si, long o_sh, long o_sb, long m_sb) {
-+ extern __shared__ float sh[];
-+ float * Qu0 = sh;
-+ float * Qv0 = Qu0 + RELPOS_ATTN_DK_128;
-+ float * Qu1 = Qv0 + RELPOS_ATTN_DK_128;
-+ float * Qv1 = Qu1 + RELPOS_ATTN_DK_128;
-+ float * sc0 = Qv1 + RELPOS_ATTN_DK_128;
-+ float * sc1 = sc0 + kv;
-+ float * red0 = sc1 + kv;
-+ float * red1 = red0 + RELPOS_ATTN_WARPS_128;
-+
-+ const int h = blockIdx.x;
-+ const int b = blockIdx.z;
-+ const int d = threadIdx.x;
-+ const int warp = d >> 5;
-+ const int lane = d & 31;
-+
-+ const float * Qh0 = Q + (size_t) b * q_sb + (size_t) h * q_sh;
-+ const float * Qh1 = Qh0 + q_sq;
-+ const T * Kh = K + (size_t) b * k_sb + (size_t) h * k_sh;
-+ const T * Vh = V + (size_t) b * v_sb + (size_t) h * v_sh;
-+ const T * Ph = Ppos + (size_t) h * p_sh;
-+ const float * Mb = mask ? mask + (size_t) b * m_sb : nullptr;
-+
-+ const float bias_u = bu[h * RELPOS_ATTN_DK_128 + d];
-+ const float bias_v = bv[h * RELPOS_ATTN_DK_128 + d];
-+ Qu0[d] = Qh0[d] + bias_u;
-+ Qv0[d] = Qh0[d] + bias_v;
-+ Qu1[d] = Qh1[d] + bias_u;
-+ Qv1[d] = Qh1[d] + bias_v;
-+ __syncthreads();
-+
-+ const int d4 = lane * 4;
-+ const float4 qu04 = *reinterpret_cast(Qu0 + d4);
-+ const float4 qv04 = *reinterpret_cast(Qv0 + d4);
-+ const float4 qu14 = *reinterpret_cast(Qu1 + d4);
-+ const float4 qv14 = *reinterpret_cast(Qv1 + d4);
-+ float produced_max0 = -INFINITY;
-+ float produced_max1 = -INFINITY;
-+ for (int j = warp; j < kv; j += RELPOS_ATTN_WARPS_128) {
-+ const float4 k4 = relpos_load4(Kh + (size_t) j * k_sj + d4);
-+ const float4 p04 = relpos_load4(Ph + (size_t) (j + 1) * p_sr + d4);
-+ const float4 p14 = relpos_load4(Ph + (size_t) j * p_sr + d4);
-+ float score0 =
-+ k4.x * qu04.x + p04.x * qv04.x +
-+ k4.y * qu04.y + p04.y * qv04.y +
-+ k4.z * qu04.z + p04.z * qv04.z +
-+ k4.w * qu04.w + p04.w * qv04.w;
-+ float score1 =
-+ k4.x * qu14.x + p14.x * qv14.x +
-+ k4.y * qu14.y + p14.y * qv14.y +
-+ k4.z * qu14.z + p14.z * qv14.z +
-+ k4.w * qu14.w + p14.w * qv14.w;
-+ score0 = relpos_warp_sum(score0);
-+ score1 = relpos_warp_sum(score1);
-+ if (lane == 0) {
-+ const float additive_mask = Mb ? Mb[j] : 0.0f;
-+ sc0[j] = score0 * scale + additive_mask;
-+ sc1[j] = score1 * scale + additive_mask;
-+ produced_max0 = fmaxf(produced_max0, sc0[j]);
-+ produced_max1 = fmaxf(produced_max1, sc1[j]);
-+ }
-+ }
-+ if (lane == 0) {
-+ red0[warp] = produced_max0;
-+ red1[warp] = produced_max1;
-+ }
-+ __syncthreads();
-+
-+ if (d == 0) {
-+ float maximum0 = red0[0];
-+ float maximum1 = red1[0];
-+#pragma unroll
-+ for (int w = 1; w < RELPOS_ATTN_WARPS_128; ++w) {
-+ maximum0 = fmaxf(maximum0, red0[w]);
-+ maximum1 = fmaxf(maximum1, red1[w]);
-+ }
-+ red0[0] = maximum0;
-+ red1[0] = maximum1;
-+ }
-+ __syncthreads();
-+ const float maximum0 = red0[0];
-+ const float maximum1 = red1[0];
-+ __syncthreads();
-+
-+ float local_sum0 = 0.0f;
-+ float local_sum1 = 0.0f;
-+ for (int j = d; j < kv; j += RELPOS_ATTN_DK_128) {
-+ const float weight0 = __expf(sc0[j] - maximum0);
-+ const float weight1 = __expf(sc1[j] - maximum1);
-+ sc0[j] = weight0;
-+ sc1[j] = weight1;
-+ local_sum0 += weight0;
-+ local_sum1 += weight1;
-+ }
-+ local_sum0 = relpos_warp_sum(local_sum0);
-+ local_sum1 = relpos_warp_sum(local_sum1);
-+ if (lane == 0) {
-+ red0[warp] = local_sum0;
-+ red1[warp] = local_sum1;
-+ }
-+ __syncthreads();
-+ if (d == 0) {
-+ float sum0 = red0[0];
-+ float sum1 = red1[0];
-+#pragma unroll
-+ for (int w = 1; w < RELPOS_ATTN_WARPS_128; ++w) {
-+ sum0 += red0[w];
-+ sum1 += red1[w];
-+ }
-+ red0[0] = sum0;
-+ red1[0] = sum1;
-+ }
-+ __syncthreads();
-+
-+ float context0 = 0.0f;
-+ float context1 = 0.0f;
-+ for (int j = 0; j < kv; ++j) {
-+ const float value = (float) Vh[(size_t) j * v_sj + d];
-+ context0 += sc0[j] * value;
-+ context1 += sc1[j] * value;
-+ }
-+ const size_t out = (size_t) b * o_sb + (size_t) h * o_sh + d;
-+ ctx[out] = context0 / red0[0];
-+ ctx[out + o_si] = context1 / red1[0];
-+}
-+
-+template
-+static int relpos_attn_warp_128_max_blocks_per_sm(int device, size_t shmem) {
-+ // The target shape fixes dynamic shared memory, so occupancy is invariant
-+ // for a given compiled kernel and device. Avoid repeating the CUDA runtime
-+ // query in every attention layer and graph execution.
-+ static std::atomic cached[GGML_CUDA_MAX_DEVICES] = {};
-+ int blocks = cached[device].load(std::memory_order_relaxed);
-+ if (blocks == 0) {
-+ CUDA_CHECK(cudaOccupancyMaxActiveBlocksPerMultiprocessor(
-+ &blocks, fused_relpos_attn_warp_128_kernel, RELPOS_ATTN_DK_128, shmem));
-+ GGML_ASSERT(blocks > 0);
-+ cached[device].store(blocks, std::memory_order_relaxed);
-+ }
-+ return blocks;
-+}
-+
-+template
-+static __global__ void fused_relpos_attn_kernel(
-+ const float * __restrict__ Q, const T * __restrict__ K,
-+ const T * __restrict__ V, const T * __restrict__ Ppos,
-+ const float * __restrict__ bu, const float * __restrict__ bv,
-+ const float * __restrict__ mask, float * __restrict__ ctx,
-+ int q, int kv, int n_head, float scale,
-+ // element strides: x_sq = between queries/keys, x_sh = between heads,
-+ // x_sb = between batch items
-+ long q_sq, long q_sh, long q_sb,
-+ long k_sj, long k_sh, long k_sb,
-+ long v_sj, long v_sh, long v_sb,
-+ long p_sr, long p_sh,
-+ long o_si, long o_sh, long o_sb, long m_sb) {
-+ extern __shared__ float sh[];
-+ const int dk = blockDim.x;
-+ float * Qu = sh; // [dk]
-+ float * Qv = sh + dk; // [dk]
-+ float * sc = sh + 2 * dk; // [kv]
-+ float * red = sh + 2 * dk + kv; // [dk] reduction scratch
-+
-+ const int h = blockIdx.x; // head
-+ const int i = blockIdx.y; // query
-+ const int b = blockIdx.z; // batch
-+ const int d = threadIdx.x; // head dim 0..dk-1
-+
-+ const float * Qhi = Q + (size_t) b * q_sb + (size_t) h * q_sh + (size_t) i * q_sq;
-+ const T * Kh = K + (size_t) b * k_sb + (size_t) h * k_sh;
-+ const T * Vh = V + (size_t) b * v_sb + (size_t) h * v_sh;
-+ const T * Ph = Ppos + (size_t) h * p_sh;
-+ const float * Mb = mask ? mask + (size_t) b * m_sb : nullptr;
-+
-+ Qu[d] = Qhi[d] + bu[h * dk + d];
-+ Qv[d] = Qhi[d] + bv[h * dk + d];
-+ __syncthreads();
-+
-+ // scores. Two layouts:
-+ // * dk == 128 (the FastConformer case): WARP-COOPERATIVE — each warp owns
-+ // a key j and the 32 lanes split the 128 dims 4-a-piece with one
-+ // vectorized row load + shuffle reduction. The original
-+ // thread-per-key loop left dk-kv threads idle (kv ~= 50 < 128) and
-+ // issued dk scalar loads per row, which made the kernel
-+ // load-issue-bound (F16 operands alone changed nothing).
-+ // * otherwise: legacy thread-per-key scalar loop.
-+ if (dk == 128) {
-+ const int warp = d >> 5, lane = d & 31, nw = dk >> 5;
-+ for (int j = warp; j < kv; j += nw) {
-+ const T * Kj = Kh + (size_t) j * k_sj + lane * 4;
-+ const int row = (q - 1) + j - i; // rel-shift index
-+ const T * Pr = Ph + (size_t) row * p_sr + lane * 4;
-+ float k4[4], p4[4];
-+ if (sizeof(T) == 2) {
-+ const int2 kr = *(const int2 *) Kj;
-+ const int2 pr = *(const int2 *) Pr;
-+ const half2 * kh = (const half2 *) &kr;
-+ const half2 * ph = (const half2 *) ≺
-+ k4[0] = __low2float(kh[0]); k4[1] = __high2float(kh[0]);
-+ k4[2] = __low2float(kh[1]); k4[3] = __high2float(kh[1]);
-+ p4[0] = __low2float(ph[0]); p4[1] = __high2float(ph[0]);
-+ p4[2] = __low2float(ph[1]); p4[3] = __high2float(ph[1]);
-+ } else {
-+ const float4 kr = *(const float4 *) Kj;
-+ const float4 pr = *(const float4 *) Pr;
-+ k4[0] = ((const float *) &kr)[0]; k4[1] = ((const float *) &kr)[1];
-+ k4[2] = ((const float *) &kr)[2]; k4[3] = ((const float *) &kr)[3];
-+ p4[0] = ((const float *) &pr)[0]; p4[1] = ((const float *) &pr)[1];
-+ p4[2] = ((const float *) &pr)[2]; p4[3] = ((const float *) &pr)[3];
-+ }
-+ float s = 0.0f;
-+#pragma unroll
-+ for (int e = 0; e < 4; e++) {
-+ s += k4[e] * Qu[lane * 4 + e] + p4[e] * Qv[lane * 4 + e];
-+ }
-+#pragma unroll
-+ for (int off = 16; off > 0; off >>= 1) {
-+ s += __shfl_xor_sync(0xffffffff, s, off);
-+ }
-+ if (lane == 0) {
-+ sc[j] = s * scale + (Mb ? Mb[j] : 0.0f);
-+ }
-+ }
-+ } else {
-+ for (int j = d; j < kv; j += dk) {
-+ const T * Kj = Kh + (size_t) j * k_sj;
-+ const int row = (q - 1) + j - i; // rel-shift index
-+ const T * Pr = Ph + (size_t) row * p_sr;
-+ float ac = 0.0f, bd = 0.0f;
-+ for (int dd = 0; dd < dk; dd++) {
-+ ac += (float) Kj[dd] * Qu[dd];
-+ bd += (float) Pr[dd] * Qv[dd];
-+ }
-+ sc[j] = (ac + bd) * scale + (Mb ? Mb[j] : 0.0f);
-+ }
-+ }
-+ __syncthreads();
-+
-+ // block max over sc[0..kv)
-+ float lm = -INFINITY;
-+ for (int j = d; j < kv; j += dk) lm = fmaxf(lm, sc[j]);
-+ red[d] = lm;
-+ __syncthreads();
-+ for (int s = dk / 2; s > 0; s >>= 1) {
-+ if (d < s) red[d] = fmaxf(red[d], red[d + s]);
-+ __syncthreads();
-+ }
-+ const float m = red[0];
-+ __syncthreads();
-+
-+ // exp + block sum
-+ float ls = 0.0f;
-+ for (int j = d; j < kv; j += dk) {
-+ const float e = __expf(sc[j] - m);
-+ sc[j] = e;
-+ ls += e;
-+ }
-+ red[d] = ls;
-+ __syncthreads();
-+ for (int s = dk / 2; s > 0; s >>= 1) {
-+ if (d < s) red[d] += red[d + s];
-+ __syncthreads();
-+ }
-+ const float inv = 1.0f / red[0];
-+ __syncthreads();
-+
-+ // ctx[d] = inv * sum_j softmax(sc[j]) * V[j,d] (thread d owns output dim d)
-+ float c = 0.0f;
-+ for (int j = 0; j < kv; j++) c += sc[j] * (float) Vh[(size_t) j * v_sj + d];
-+ ctx[(size_t) b * o_sb + (size_t) i * o_si + (size_t) h * o_sh + d] = c * inv;
-+}
-+
-+void ggml_cuda_op_fused_relpos_attn(ggml_backend_cuda_context & ctx, ggml_tensor * dst) {
-+ const ggml_tensor * q = dst->src[0];
-+ const ggml_tensor * k = dst->src[1];
-+ const ggml_tensor * v = dst->src[2];
-+ const ggml_tensor * p = dst->src[3];
-+ const ggml_tensor * bias_u = dst->src[4];
-+ const ggml_tensor * bias_v = dst->src[5];
-+ const ggml_tensor * mask = dst->src[6]; // may be null
-+
-+ GGML_ASSERT(q->type == GGML_TYPE_F32);
-+ // K/V/P may be F32 or (all together) F16 — see the kernel comment.
-+ GGML_ASSERT(k->type == v->type && k->type == p->type);
-+ GGML_ASSERT(k->type == GGML_TYPE_F32 || k->type == GGML_TYPE_F16);
-+ GGML_ASSERT(bias_u->type == GGML_TYPE_F32 && bias_v->type == GGML_TYPE_F32);
-+ GGML_ASSERT(dst->type == GGML_TYPE_F32);
-+
-+ const int d_k = q->ne[0];
-+ const int q_len = q->ne[1];
-+ const int n_head = q->ne[2];
-+ const int batch = q->ne[3];
-+ const int kv_len = k->ne[1];
-+
-+ GGML_ASSERT((d_k & (d_k - 1)) == 0 && "fused_relpos_attn: d_k must be a power of two");
-+ GGML_ASSERT(k->ne[0] == d_k && v->ne[0] == d_k && p->ne[0] == d_k);
-+ GGML_ASSERT(k->ne[1] == kv_len && v->ne[1] == kv_len);
-+ GGML_ASSERT(k->ne[2] == n_head && v->ne[2] == n_head && p->ne[2] == n_head);
-+ GGML_ASSERT(k->ne[3] == batch && v->ne[3] == batch);
-+ GGML_ASSERT(bias_u->ne[0] == d_k && bias_v->ne[0] == d_k);
-+ GGML_ASSERT(bias_u->ne[1] == n_head && bias_v->ne[1] == n_head);
-+ if (mask != nullptr) {
-+ GGML_ASSERT(mask->type == GGML_TYPE_F32);
-+ GGML_ASSERT(ggml_is_contiguous(mask));
-+ GGML_ASSERT(mask->ne[0] == kv_len);
-+ // Shared across the batch (ne[1]==1) or one key-mask column per
-+ // stream (ne[1]==batch, the cache-aware layout).
-+ GGML_ASSERT(mask->ne[1] == 1 || mask->ne[1] == batch);
-+ GGML_ASSERT(mask->ne[2] == 1 && mask->ne[3] == 1);
-+ }
-+ const long m_sb =
-+ (mask != nullptr && mask->ne[1] == batch && batch > 1) ? (long) (mask->nb[1] / sizeof(float)) : 0;
-+
-+ float scale;
-+ memcpy(&scale, dst->op_params, sizeof(scale));
-+
-+ // Element strides from tensor byte strides. d_k rows must be contiguous
-+ // (constructor invariant), everything else is free-form.
-+ const size_t qe = ggml_type_size(q->type);
-+ const size_t ke = ggml_type_size(k->type);
-+ const long q_sq = (long)(q->nb[1] / qe), q_sh = (long)(q->nb[2] / qe),
-+ q_sb = (long)(q->nb[3] / qe);
-+ const long k_sj = (long)(k->nb[1] / ke), k_sh = (long)(k->nb[2] / ke),
-+ k_sb = (long)(k->nb[3] / ke);
-+ const long v_sj = (long)(v->nb[1] / ke), v_sh = (long)(v->nb[2] / ke),
-+ v_sb = (long)(v->nb[3] / ke);
-+ const long p_sr = (long)(p->nb[1] / ke), p_sh = (long)(p->nb[2] / ke);
-+ const size_t oe = ggml_type_size(dst->type);
-+ const long o_si = (long)(dst->nb[1] / oe), o_sh = (long)(dst->nb[2] / oe),
-+ o_sb = (long)(dst->nb[3] / oe);
-+
-+ const int device = ggml_cuda_get_device();
-+ const auto & device_info = ggml_cuda_info().devices[device];
-+ const size_t shmem = ((size_t) 3 * d_k + kv_len) * sizeof(float);
-+ const size_t max_shmem = device_info.smpb;
-+ GGML_ASSERT(shmem <= max_shmem && "fused_relpos_attn: kv window too large for shared memory");
-+
-+ cudaStream_t stream = ctx.stream();
-+
-+ // The cache-aware Nemotron streaming geometry is Q=2, KV=72, H=8,
-+ // d_k=128. On SM100 the one-query warp kernel is fastest while its grid
-+ // fits in one resident wave. If that grid exceeds its measured occupancy,
-+ // fuse both query rows: halving the block count and reusing K/V then wins.
-+ // cudaOccupancyMaxActiveBlocksPerMultiprocessor uses the compiled kernel's
-+ // actual register count, avoiding a hard-coded batch-size threshold.
-+ const bool use_sm100_q2 =
-+ device_info.cc == RELPOS_ATTN_CC_SM100 && device_info.warp_size == 32 &&
-+ d_k == RELPOS_ATTN_DK_128 && q_len == 2 && kv_len == 72;
-+ if (use_sm100_q2) {
-+ const size_t warp_shmem =
-+ ((size_t) 2 * RELPOS_ATTN_DK_128 + kv_len + RELPOS_ATTN_WARPS_128) * sizeof(float);
-+ const size_t q2_shmem =
-+ ((size_t) 4 * RELPOS_ATTN_DK_128 + (size_t) 2 * kv_len +
-+ (size_t) 2 * RELPOS_ATTN_WARPS_128) * sizeof(float);
-+ GGML_ASSERT(q2_shmem <= max_shmem);
-+
-+ const int max_single_blocks_per_sm = k->type == GGML_TYPE_F16
-+ ? relpos_attn_warp_128_max_blocks_per_sm(device, warp_shmem)
-+ : relpos_attn_warp_128_max_blocks_per_sm(device, warp_shmem);
-+ const int64_t single_query_blocks = (int64_t) n_head * q_len * batch;
-+ const int64_t single_wave_blocks = (int64_t) device_info.nsm * max_single_blocks_per_sm;
-+ const bool fuse_queries = single_query_blocks > single_wave_blocks;
-+ const dim3 tuned_grid(n_head, fuse_queries ? 1 : q_len, batch);
-+
-+ if (k->type == GGML_TYPE_F16) {
-+ if (fuse_queries) {
-+ fused_relpos_attn_q2_warp_128_kernel<<<
-+ tuned_grid, RELPOS_ATTN_DK_128, q2_shmem, stream>>>(
-+ (const float *) q->data, (const half *) k->data, (const half *) v->data,
-+ (const half *) p->data, (const float *) bias_u->data, (const float *) bias_v->data,
-+ mask ? (const float *) mask->data : nullptr, (float *) dst->data,
-+ kv_len, scale, q_sq, q_sh, q_sb, k_sj, k_sh, k_sb, v_sj, v_sh, v_sb,
-+ p_sr, p_sh, o_si, o_sh, o_sb, m_sb);
-+ } else {
-+ fused_relpos_attn_warp_128_kernel<<<
-+ tuned_grid, RELPOS_ATTN_DK_128, warp_shmem, stream>>>(
-+ (const float *) q->data, (const half *) k->data, (const half *) v->data,
-+ (const half *) p->data, (const float *) bias_u->data, (const float *) bias_v->data,
-+ mask ? (const float *) mask->data : nullptr, (float *) dst->data,
-+ kv_len, scale, q_sq, q_sh, q_sb, k_sj, k_sh, k_sb, v_sj, v_sh, v_sb,
-+ p_sr, p_sh, o_si, o_sh, o_sb, m_sb);
-+ }
-+ } else {
-+ if (fuse_queries) {
-+ fused_relpos_attn_q2_warp_128_kernel<<<
-+ tuned_grid, RELPOS_ATTN_DK_128, q2_shmem, stream>>>(
-+ (const float *) q->data, (const float *) k->data, (const float *) v->data,
-+ (const float *) p->data, (const float *) bias_u->data, (const float *) bias_v->data,
-+ mask ? (const float *) mask->data : nullptr, (float *) dst->data,
-+ kv_len, scale, q_sq, q_sh, q_sb, k_sj, k_sh, k_sb, v_sj, v_sh, v_sb,
-+ p_sr, p_sh, o_si, o_sh, o_sb, m_sb);
-+ } else {
-+ fused_relpos_attn_warp_128_kernel<<<
-+ tuned_grid, RELPOS_ATTN_DK_128, warp_shmem, stream>>>(
-+ (const float *) q->data, (const float *) k->data, (const float *) v->data,
-+ (const float *) p->data, (const float *) bias_u->data, (const float *) bias_v->data,
-+ mask ? (const float *) mask->data : nullptr, (float *) dst->data,
-+ kv_len, scale, q_sq, q_sh, q_sb, k_sj, k_sh, k_sb, v_sj, v_sh, v_sb,
-+ p_sr, p_sh, o_si, o_sh, o_sb, m_sb);
-+ }
-+ }
-+ return;
-+ }
-+
-+ const dim3 grid(n_head, q_len, batch);
-+ if (k->type == GGML_TYPE_F16) {
-+ fused_relpos_attn_kernel<<>>(
-+ (const float *) q->data, (const half *) k->data, (const half *) v->data,
-+ (const half *) p->data, (const float *) bias_u->data, (const float *) bias_v->data,
-+ mask ? (const float *) mask->data : nullptr, (float *) dst->data,
-+ q_len, kv_len, n_head, scale,
-+ q_sq, q_sh, q_sb, k_sj, k_sh, k_sb, v_sj, v_sh, v_sb, p_sr, p_sh, o_si, o_sh, o_sb,
-+ m_sb);
-+ } else {
-+ fused_relpos_attn_kernel<<>>(
-+ (const float *) q->data, (const float *) k->data, (const float *) v->data,
-+ (const float *) p->data, (const float *) bias_u->data, (const float *) bias_v->data,
-+ mask ? (const float *) mask->data : nullptr, (float *) dst->data,
-+ q_len, kv_len, n_head, scale,
-+ q_sq, q_sh, q_sb, k_sj, k_sh, k_sb, v_sj, v_sh, v_sb, p_sr, p_sh, o_si, o_sh, o_sb,
-+ m_sb);
-+ }
-+}
-diff --git a/src/ggml-cuda/fused-relpos-attn.cuh b/src/ggml-cuda/fused-relpos-attn.cuh
-new file mode 100644
-index 00000000..b647d6b1
---- /dev/null
-+++ b/src/ggml-cuda/fused-relpos-attn.cuh
-@@ -0,0 +1,5 @@
-+// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved.
-+// SPDX-License-Identifier: Apache-2.0
-+#include "common.cuh"
-+
-+void ggml_cuda_op_fused_relpos_attn(ggml_backend_cuda_context & ctx, ggml_tensor * dst);
-diff --git a/src/ggml.c b/src/ggml.c
-index 476c3079..88b39537 100644
---- a/src/ggml.c
-+++ b/src/ggml.c
-@@ -1078,9 +1078,11 @@ static const char * GGML_OP_NAME[GGML_OP_COUNT] = {
- "OPT_STEP_SGD",
-
- "GLU",
-+
-+ "FUSED_RELPOS_ATTN",
- };
-
--static_assert(GGML_OP_COUNT == 96, "GGML_OP_COUNT != 96");
-+static_assert(GGML_OP_COUNT == 97, "GGML_OP_COUNT != 97");
-
- static const char * GGML_OP_SYMBOL[GGML_OP_COUNT] = {
- "none",
-@@ -1188,9 +1190,11 @@ static const char * GGML_OP_SYMBOL[GGML_OP_COUNT] = {
- "sgd(x)",
-
- "glu(x)",
-+
-+ "fused_relpos_attn(q,k,v,p)",
- };
-
--static_assert(GGML_OP_COUNT == 96, "GGML_OP_COUNT != 96");
-+static_assert(GGML_OP_COUNT == 97, "GGML_OP_COUNT != 97");
-
- static_assert(GGML_OP_POOL_COUNT == 2, "GGML_OP_POOL_COUNT != 2");
-
-@@ -5370,6 +5374,78 @@ struct ggml_tensor * ggml_flash_attn_ext(
- return result;
- }
-
-+// ggml_fused_relpos_attn
-+
-+struct ggml_tensor * ggml_fused_relpos_attn(
-+ struct ggml_context * ctx,
-+ struct ggml_tensor * q,
-+ struct ggml_tensor * k,
-+ struct ggml_tensor * v,
-+ struct ggml_tensor * p,
-+ struct ggml_tensor * bias_u,
-+ struct ggml_tensor * bias_v,
-+ struct ggml_tensor * mask,
-+ float scale,
-+ bool merge_heads) {
-+ // Q/K/V/P may be arbitrary-strided views; the CUDA op derives addressing
-+ // from their nb[]. Only each d_k row must be contiguous (vectorized row
-+ // loads).
-+ GGML_ASSERT(q->nb[0] == ggml_type_size(q->type));
-+ GGML_ASSERT(k->nb[0] == ggml_type_size(k->type));
-+ GGML_ASSERT(v->nb[0] == ggml_type_size(v->type));
-+ GGML_ASSERT(p->nb[0] == ggml_type_size(p->type));
-+ GGML_ASSERT(ggml_is_contiguous(bias_u));
-+ GGML_ASSERT(ggml_is_contiguous(bias_v));
-+
-+ const int64_t d_k = q->ne[0];
-+ const int64_t kv_len = k->ne[1];
-+ const int64_t q_len = q->ne[1];
-+ const int64_t n_head = q->ne[2];
-+
-+ GGML_ASSERT(k->ne[0] == d_k && v->ne[0] == d_k && p->ne[0] == d_k);
-+ GGML_ASSERT(bias_u->ne[0] == d_k && bias_v->ne[0] == d_k);
-+ // The kernel reads rel-pos rows (q_len-1)+j-i for j in [0,kv_len), i in
-+ // [0,q_len) — i.e. rows [0, kv_len+q_len-1) — and takes its head stride
-+ // from p->nb, so a longer table (e.g. precomputed for the full chunk
-+ // length and reused by shorter tail chunks) is safe.
-+ GGML_ASSERT(p->ne[1] >= kv_len + q_len - 1); // rel-pos length
-+ GGML_ASSERT(v->ne[1] == kv_len);
-+ if (mask) {
-+ GGML_ASSERT(ggml_is_contiguous(mask));
-+ GGML_ASSERT(mask->ne[0] == kv_len);
-+ // One shared key mask, or one column per batch item (the cache-aware
-+ // streaming layout, where each stream's history has its own validity).
-+ GGML_ASSERT(mask->ne[1] == 1 || mask->ne[1] == q->ne[3]);
-+ GGML_ASSERT(mask->ne[2] == 1 && mask->ne[3] == 1);
-+ }
-+
-+ // Output mirrors q logically: [d_k, q_len, n_head, batch]. With
-+ // merge_heads the memory layout interleaves heads inside each query
-+ // column (nb[2] = d_k, nb[1] = d_k*n_head) so that permute(0,2,1,3) of
-+ // the result is a contiguous (d_k*n_head, q_len, batch) matrix.
-+ struct ggml_tensor * result = ggml_new_tensor(ctx, GGML_TYPE_F32, 4, q->ne);
-+ if (merge_heads) {
-+ const size_t ts = ggml_type_size(result->type);
-+ result->nb[0] = ts;
-+ result->nb[2] = (size_t) d_k * ts;
-+ result->nb[1] = (size_t) d_k * n_head * ts;
-+ result->nb[3] = (size_t) d_k * n_head * q_len * ts;
-+ }
-+
-+ ggml_set_op_params(result, &scale, sizeof(scale));
-+
-+ result->op = GGML_OP_FUSED_RELPOS_ATTN;
-+ result->src[0] = q;
-+ result->src[1] = k;
-+ result->src[2] = v;
-+ result->src[3] = p;
-+ result->src[4] = bias_u;
-+ result->src[5] = bias_v;
-+ result->src[6] = mask;
-+
-+ return result;
-+}
-+
- void ggml_flash_attn_ext_set_prec(
- struct ggml_tensor * a,
- enum ggml_prec prec) {
diff --git a/ggml-patches/0002-nvfp4-residual-activations.patch b/ggml-patches/0002-nvfp4-residual-activations.patch
deleted file mode 100644
index 5f17141..0000000
--- a/ggml-patches/0002-nvfp4-residual-activations.patch
+++ /dev/null
@@ -1,400 +0,0 @@
-diff --git a/src/ggml-cuda/mmq.cu b/src/ggml-cuda/mmq.cu
-index e1add5e0..9c98e3c3 100644
---- a/src/ggml-cuda/mmq.cu
-+++ b/src/ggml-cuda/mmq.cu
-@@ -127,7 +127,9 @@ void ggml_cuda_mul_mat_q(
- if (!ids) {
- const size_t nbytes_src1_q8_1 = ne13*ne12 * ne11*ne10_padded * sizeof(block_q8_1)/QK8_1 +
- get_mmq_x_max_host(cc)*sizeof(block_q8_1_mmq);
-- ggml_cuda_pool_alloc src1_q8_1(ctx.pool(), nbytes_src1_q8_1);
-+ const bool use_nvfp4_residual = use_native_fp4 && src0->type == GGML_TYPE_NVFP4;
-+ ggml_cuda_pool_alloc src1_q8_1(ctx.pool(), nbytes_src1_q8_1 * (use_nvfp4_residual ? 2 : 1));
-+ char * src1_residual = use_nvfp4_residual ? src1_q8_1.get() + nbytes_src1_q8_1 : nullptr;
-
- {
- const int64_t s11 = src1->nb[1] / ts_src1;
-@@ -135,7 +137,7 @@ void ggml_cuda_mul_mat_q(
- const int64_t s13 = src1->nb[3] / ts_src1;
- if (use_native_fp4) {
- static_assert(sizeof(block_fp4_mmq) == 4 * sizeof(block_q8_1));
-- quantize_mmq_fp4_cuda(src1_d, nullptr, src1_q8_1.get(), src0->type, ne10, s11, s12, s13, ne10_padded,
-+ quantize_mmq_fp4_cuda(src1_d, nullptr, src1_q8_1.get(), src1_residual, src0->type, ne10, s11, s12, s13, ne10_padded,
- ne11, ne12, ne13, stream);
-
- } else {
-@@ -152,7 +154,7 @@ void ggml_cuda_mul_mat_q(
- const int64_t s13 = ne12*s12;
-
- const mmq_args args = {
-- src0_d, src0->type, (const int *) src1_q8_1.ptr, nullptr, nullptr, dst_d,
-+ src0_d, src0->type, (const int *) src1_q8_1.ptr, (const int *) src1_residual, nullptr, nullptr, dst_d,
- ne00, ne01, ne1, s01, ne11, s1,
- ne02, ne12, s02, s12, s2,
- ne03, ne13, s03, s13, s3,
-@@ -185,7 +187,9 @@ void ggml_cuda_mul_mat_q(
-
- const size_t nbytes_src1_q8_1 = ne12*n_expert_used*ne10_padded * sizeof(block_q8_1)/QK8_1 +
- get_mmq_x_max_host(cc)*sizeof(block_q8_1_mmq);
-- ggml_cuda_pool_alloc src1_q8_1(ctx.pool(), nbytes_src1_q8_1);
-+ const bool use_nvfp4_residual = use_native_fp4 && src0->type == GGML_TYPE_NVFP4;
-+ ggml_cuda_pool_alloc src1_q8_1(ctx.pool(), nbytes_src1_q8_1 * (use_nvfp4_residual ? 2 : 1));
-+ char * src1_residual = use_nvfp4_residual ? src1_q8_1.get() + nbytes_src1_q8_1 : nullptr;
-
- const int64_t ne11_flat = ne12*n_expert_used;
- const int64_t ne12_flat = 1;
-@@ -197,7 +201,7 @@ void ggml_cuda_mul_mat_q(
- const int64_t s13 = src1->nb[3] / ts_src1;
-
- if (use_native_fp4) {
-- quantize_mmq_fp4_cuda(src1_d, ids_src1.get(), src1_q8_1.get(), src0->type, ne10, s11, s12, s13,
-+ quantize_mmq_fp4_cuda(src1_d, ids_src1.get(), src1_q8_1.get(), src1_residual, src0->type, ne10, s11, s12, s13,
- ne10_padded, ne11_flat, ne12_flat, ne13_flat, stream);
- } else {
- quantize_mmq_q8_1_cuda(src1_d, ids_src1.get(), src1_q8_1.get(), src0->type, ne10, s11, s12, s13,
-@@ -213,7 +217,7 @@ void ggml_cuda_mul_mat_q(
-
- // Note that ne02 is used instead of ne12 because the number of y channels determines the z dimension of the CUDA grid.
- const mmq_args args = {
-- src0_d, src0->type, (const int *) src1_q8_1.get(), ids_dst.get(), expert_bounds.get(), dst_d,
-+ src0_d, src0->type, (const int *) src1_q8_1.get(), (const int *) src1_residual, ids_dst.get(), expert_bounds.get(), dst_d,
- ne00, ne01, ne_get_rows, s01, ne_get_rows, s1,
- ne02, ne02, s02, s12, s2,
- ne03, ne13, s03, s13, s3,
-@@ -253,7 +257,7 @@ void ggml_cuda_op_mul_mat_q(
- || GGML_CUDA_CC_IS_CDNA(cc))
- && src1_ncols == ne11;
- const mmq_args args = {
-- src0_dd_i, src0->type, (const int *) src1_ddq_i, nullptr, nullptr, dst_dd_i,
-+ src0_dd_i, src0->type, (const int *) src1_ddq_i, nullptr, nullptr, nullptr, dst_dd_i,
- ne00, row_diff, src1_ncols, stride01, ne11, nrows_dst,
- 1, 1, 0, 0, 0,
- 1, 1, 0, 0, 0,
-diff --git a/src/ggml-cuda/mmq.cuh b/src/ggml-cuda/mmq.cuh
-index edf546d8..1881b7f3 100644
---- a/src/ggml-cuda/mmq.cuh
-+++ b/src/ggml-cuda/mmq.cuh
-@@ -3446,6 +3446,7 @@ struct mmq_type_traits {
- template
- static __device__ __forceinline__ void mul_mat_q_process_tile(
- const char * __restrict__ x, const int offset_x, const int * __restrict__ y,
-+ const int * __restrict__ y_residual,
- const int * __restrict__ ids_dst, float * __restrict__ dst, float * __restrict__ tmp_fixup,
- const int stride_row_x, const int ncols_y, const int stride_col_dst,
- const int tile_x_max_i, const int tile_y_max_j, const int kb0_start, const int kb0_stop) {
-@@ -3500,6 +3501,22 @@ static __device__ __forceinline__ void mul_mat_q_process_tile(
-
- __syncthreads();
-
-+#if defined(BLACKWELL_MMA_AVAILABLE)
-+ if constexpr (type == GGML_TYPE_NVFP4) {
-+ if (y_residual) {
-+ const int * by0 = y_residual + ncols_y * (kb0 * qk / ne_block) * sz;
-+#pragma unroll
-+ for (int l0 = 0; l0 < mmq_x * MMQ_TILE_Y_K; l0 += nwarps * warp_size) {
-+ const int l = l0 + threadIdx.y * warp_size + threadIdx.x;
-+ tile_y[l] = by0[l];
-+ }
-+ __syncthreads();
-+ vec_dot(tile_x, tile_y, sum, 0);
-+ __syncthreads();
-+ }
-+ }
-+#endif
-+
- {
- const int * by0 = y + ncols_y * ((kb0 * qk / ne_block) * sz + sz);
- #pragma unroll
-@@ -3515,6 +3532,22 @@ static __device__ __forceinline__ void mul_mat_q_process_tile(
- vec_dot(tile_x, tile_y, sum, MMQ_TILE_NE_K);
-
- __syncthreads();
-+
-+#if defined(BLACKWELL_MMA_AVAILABLE)
-+ if constexpr (type == GGML_TYPE_NVFP4) {
-+ if (y_residual) {
-+ const int * by0 = y_residual + ncols_y * ((kb0 * qk / ne_block) * sz + sz);
-+#pragma unroll
-+ for (int l0 = 0; l0 < mmq_x * MMQ_TILE_Y_K; l0 += nwarps * warp_size) {
-+ const int l = l0 + threadIdx.y * warp_size + threadIdx.x;
-+ tile_y[l] = by0[l];
-+ }
-+ __syncthreads();
-+ vec_dot(tile_x, tile_y, sum, MMQ_TILE_NE_K);
-+ __syncthreads();
-+ }
-+ }
-+#endif
- }
-
- if (fixup) {
-@@ -3540,7 +3573,8 @@ template
- #endif // __CUDA_ARCH__ >= GGML_CUDA_CC_VOLTA
- #endif // defined(GGML_USE_HIP)
- static __global__ void mul_mat_q(
-- const char * __restrict__ x, const int * __restrict__ y, const int32_t * __restrict__ ids_dst,
-+ const char * __restrict__ x, const int * __restrict__ y, const int * __restrict__ y_residual,
-+ const int32_t * __restrict__ ids_dst,
- const int32_t * __restrict__ expert_bounds, float * __restrict__ dst, float * __restrict__ tmp_fixup,
- const uint3 blocks_per_ne00, const int nrows_x, const int ncols_dst, const int stride_row_x, const int ncols_y, const int stride_col_dst,
- const uint3 channel_ratio, const uint3 nchannels_y, const int stride_channel_x, const int stride_channel_y, const int stride_channel_dst,
-@@ -3629,7 +3663,8 @@ static __global__ void mul_mat_q(
-
- constexpr bool fixup = false;
- mul_mat_q_process_tile
-- (x, offset_x, y + offset_y, ids_dst_shared, dst + offset_dst, tmp_fixup, stride_row_x, ncols_y, stride_col_dst,
-+ (x, offset_x, y + offset_y, y_residual ? y_residual + offset_y : nullptr,
-+ ids_dst_shared, dst + offset_dst, tmp_fixup, stride_row_x, ncols_y, stride_col_dst,
- tile_x_max_i, tile_y_max_j, 0, blocks_per_ne00.z);
- return;
- }
-@@ -3709,7 +3744,8 @@ static __global__ void mul_mat_q(
-
- constexpr bool fixup = false; // All but (potentially) the last iterations write their data to dst rather than the fixup buffer.
- mul_mat_q_process_tile
-- (x, offset_x, y + offset_y, ids_dst_shared, dst + offset_dst, tmp_fixup, stride_row_x, ncols_y, stride_col_dst,
-+ (x, offset_x, y + offset_y, y_residual ? y_residual + offset_y : nullptr,
-+ ids_dst_shared, dst + offset_dst, tmp_fixup, stride_row_x, ncols_y, stride_col_dst,
- tile_x_max_i, tile_y_max_j, kb0_start, kb0_stop);
-
- kbc += blocks_per_ne00.z;
-@@ -3778,7 +3814,8 @@ static __global__ void mul_mat_q(
-
- constexpr bool fixup = true; // Last index writes its data to fixup buffer to avoid data races with other blocks.
- mul_mat_q_process_tile
-- (x, offset_x, y + offset_y, ids_dst_shared, dst + offset_dst, tmp_fixup, stride_row_x, ncols_y, stride_col_dst,
-+ (x, offset_x, y + offset_y, y_residual ? y_residual + offset_y : nullptr,
-+ ids_dst_shared, dst + offset_dst, tmp_fixup, stride_row_x, ncols_y, stride_col_dst,
- tile_x_max_i, tile_y_max_j, kb0_start, kb0_stop);
- }
-
-@@ -3922,7 +3959,8 @@ static __global__ void mul_mat_q_stream_k_fixup(
- }
-
- struct mmq_args {
-- const char * x; ggml_type type_x; const int * y; const int32_t * ids_dst; const int32_t * expert_bounds; float * dst;
-+ const char * x; ggml_type type_x; const int * y; const int * y_residual;
-+ const int32_t * ids_dst; const int32_t * expert_bounds; float * dst;
- int64_t ncols_x; int64_t nrows_x; int64_t ncols_dst; int64_t stride_row_x; int64_t ncols_y; int64_t nrows_dst;
- int64_t nchannels_x; int64_t nchannels_y; int64_t stride_channel_x; int64_t stride_channel_y; int64_t stride_channel_dst;
- int64_t nsamples_x; int64_t nsamples_y; int64_t stride_sample_x; int64_t stride_sample_y; int64_t stride_sample_dst;
-@@ -3976,7 +4014,7 @@ static void launch_mul_mat_q(ggml_backend_cuda_context & ctx, const mmq_args & a
- if (args.nrows_x % mmq_y == 0) {
- constexpr bool need_check = false;
- mul_mat_q<<>>
-- (args.x, args.y, args.ids_dst, args.expert_bounds, args.dst, nullptr,
-+ (args.x, args.y, args.y_residual, args.ids_dst, args.expert_bounds, args.dst, nullptr,
- blocks_per_ne00_fd, args.nrows_x, args.ncols_dst, args.stride_row_x, args.ncols_y, args.nrows_dst,
- channel_ratio_fd, nchannels_y_fd, args.stride_channel_x, args.stride_channel_y, args.stride_channel_dst,
- sample_ratio_fd, nsamples_y_fd, args.stride_sample_x, args.stride_sample_y, args.stride_sample_dst,
-@@ -3984,7 +4022,7 @@ static void launch_mul_mat_q(ggml_backend_cuda_context & ctx, const mmq_args & a
- } else {
- constexpr bool need_check = true;
- mul_mat_q<<>>
-- (args.x, args.y, args.ids_dst, args.expert_bounds, args.dst, nullptr,
-+ (args.x, args.y, args.y_residual, args.ids_dst, args.expert_bounds, args.dst, nullptr,
- blocks_per_ne00_fd, args.nrows_x, args.ncols_dst, args.stride_row_x, args.ncols_y, args.nrows_dst,
- channel_ratio_fd, nchannels_y_fd, args.stride_channel_x, args.stride_channel_y, args.stride_channel_dst,
- sample_ratio_fd, nsamples_y_fd, args.stride_sample_x, args.stride_sample_y, args.stride_sample_dst,
-@@ -4016,7 +4054,7 @@ static void launch_mul_mat_q(ggml_backend_cuda_context & ctx, const mmq_args & a
- if (args.nrows_x % mmq_y == 0) {
- constexpr bool need_check = false;
- mul_mat_q<<>>
-- (args.x, args.y, args.ids_dst, args.expert_bounds, args.dst, tmp_fixup.ptr,
-+ (args.x, args.y, args.y_residual, args.ids_dst, args.expert_bounds, args.dst, tmp_fixup.ptr,
- blocks_per_ne00_fd, args.nrows_x, args.ncols_dst, args.stride_row_x, args.ncols_y, args.nrows_dst,
- channel_ratio_fd, nchannels_y_fd, args.stride_channel_x, args.stride_channel_y, args.stride_channel_dst,
- sample_ratio_fd, nsamples_y_fd, args.stride_sample_x, args.stride_sample_y, args.stride_sample_dst,
-@@ -4034,7 +4072,7 @@ static void launch_mul_mat_q(ggml_backend_cuda_context & ctx, const mmq_args & a
- } else {
- constexpr bool need_check = true;
- mul_mat_q<<>>
-- (args.x, args.y, args.ids_dst, args.expert_bounds, args.dst, tmp_fixup.ptr,
-+ (args.x, args.y, args.y_residual, args.ids_dst, args.expert_bounds, args.dst, tmp_fixup.ptr,
- blocks_per_ne00_fd, args.nrows_x, args.ncols_dst, args.stride_row_x, args.ncols_y, args.nrows_dst,
- channel_ratio_fd, nchannels_y_fd, args.stride_channel_x, args.stride_channel_y, args.stride_channel_dst,
- sample_ratio_fd, nsamples_y_fd, args.stride_sample_x, args.stride_sample_y, args.stride_sample_dst,
-diff --git a/src/ggml-cuda/quantize.cu b/src/ggml-cuda/quantize.cu
-index 52f66471..2a878c73 100644
---- a/src/ggml-cuda/quantize.cu
-+++ b/src/ggml-cuda/quantize.cu
-@@ -72,7 +72,8 @@ __device__ __forceinline__ uint8_t compute_e8m0_scale(float amax) {
-
-
- static __global__ void quantize_mmq_nvfp4(
-- const float * __restrict__ x, const int32_t * __restrict__ ids, void * __restrict__ vy,
-+ const float * __restrict__ x, const int32_t * __restrict__ ids,
-+ void * __restrict__ vy, void * __restrict__ vy_residual,
- const int64_t ne00, const int64_t s01, const int64_t s02, const int64_t s03,
- const int64_t ne0, const int64_t ne1, const int64_t ne2) {
- #if defined(BLACKWELL_MMA_AVAILABLE)
-@@ -95,6 +96,7 @@ static __global__ void quantize_mmq_nvfp4(
- const int64_t ib = blockIdx.z * ((int64_t) blocks_per_col * ne1) + k_block * ne1 + blockIdx.x;
- block_fp4_mmq * y = (block_fp4_mmq *) vy;
- block_fp4_mmq * yb = y + ib;
-+ block_fp4_mmq * yrb = (block_fp4_mmq *) vy_residual + ib;
-
- const int sub = (i0_base % QK_K) / QK_NVFP4_SUB;
-
-@@ -113,57 +115,62 @@ static __global__ void quantize_mmq_nvfp4(
- }
- }
-
-- static constexpr int test_offsets[5] = { 0, -1, 1, -2, 2};
-- const int first_fp8_code = (int) ggml_cuda_fp32_to_ue4m3(amax_raw / 6.0f);
--
-- float best_err = FLT_MAX;
-- uint8_t fp8_code = 0;
-- float subblock_scale = 0.0f;
--
--#pragma unroll // Check +/- 2 to find best code to reduce NVFP4 activation loss. Negligible overhead on Blackwell.
-- for (int i = 0; i < 5; i++) {
-- const int test_code = first_fp8_code + test_offsets[i];
-- if (test_code < 0 || test_code > 0x7e) {
-- continue;
-- }
-- const uint8_t code = (uint8_t) test_code;
-- const float test_scale = ggml_cuda_ue4m3_to_fp32(code);
-- const float test_inv_scale = test_scale > 0.0f ? 0.5f / test_scale : 0.0f;
-- float cur_err = 0.0f;
--#pragma unroll
-- for (int k = 0; k < QK_NVFP4_SUB; ++k) {
-- const float v = vals_raw[k];
-- const uint8_t q = ggml_cuda_float_to_fp4_e2m1(v, test_inv_scale);
-- const float err_diff = fabsf(v) - fabsf(kvalues_mxfp4[q & 0x7]) * test_scale;
-- cur_err = fmaf(err_diff, err_diff, cur_err);
-- }
--
-- if (cur_err < best_err) {
-- best_err = cur_err;
-- fp8_code = test_code;
-- subblock_scale = test_scale;
-- }
-- }
--
-+ const uint8_t fp8_code = ggml_cuda_fp32_to_ue4m3(amax_raw / 6.0f);
-+ const float subblock_scale = ggml_cuda_ue4m3_to_fp32(fp8_code);
- const float inv_scale = subblock_scale > 0.0f ? 0.5f / subblock_scale : 0.0f;
- uint32_t q0 = 0;
- uint32_t q1 = 0;
--#pragma unroll // this is faster than the previous __nv_fp4x4_e2m1
-+ float residual[QK_NVFP4_SUB];
-+ float residual_amax = 0.0f;
-+#pragma unroll
- for (int k = 0; k < QK_NVFP4_SUB / 4; ++k) {
-- q0 |= (uint32_t) ggml_cuda_float_to_fp4_e2m1(vals_raw[k + 0], inv_scale) << (8 * k);
-- q0 |= (uint32_t) ggml_cuda_float_to_fp4_e2m1(vals_raw[k + 8], inv_scale) << (8 * k + 4);
-- q1 |= (uint32_t) ggml_cuda_float_to_fp4_e2m1(vals_raw[k + 4], inv_scale) << (8 * k);
-- q1 |= (uint32_t) ggml_cuda_float_to_fp4_e2m1(vals_raw[k + 12], inv_scale) << (8 * k + 4);
-+ const int i0 = k + 0;
-+ const int i1 = k + 8;
-+ const int i2 = k + 4;
-+ const int i3 = k + 12;
-+ const uint8_t p0 = ggml_cuda_float_to_fp4_e2m1(vals_raw[i0], inv_scale);
-+ const uint8_t p1 = ggml_cuda_float_to_fp4_e2m1(vals_raw[i1], inv_scale);
-+ const uint8_t p2 = ggml_cuda_float_to_fp4_e2m1(vals_raw[i2], inv_scale);
-+ const uint8_t p3 = ggml_cuda_float_to_fp4_e2m1(vals_raw[i3], inv_scale);
-+ q0 |= (uint32_t) p0 << (8 * k);
-+ q0 |= (uint32_t) p1 << (8 * k + 4);
-+ q1 |= (uint32_t) p2 << (8 * k);
-+ q1 |= (uint32_t) p3 << (8 * k + 4);
-+ residual[i0] = vals_raw[i0] - (float) kvalues_mxfp4[p0] * subblock_scale;
-+ residual[i1] = vals_raw[i1] - (float) kvalues_mxfp4[p1] * subblock_scale;
-+ residual[i2] = vals_raw[i2] - (float) kvalues_mxfp4[p2] * subblock_scale;
-+ residual[i3] = vals_raw[i3] - (float) kvalues_mxfp4[p3] * subblock_scale;
-+ residual_amax = fmaxf(residual_amax, fabsf(residual[i0]));
-+ residual_amax = fmaxf(residual_amax, fabsf(residual[i1]));
-+ residual_amax = fmaxf(residual_amax, fabsf(residual[i2]));
-+ residual_amax = fmaxf(residual_amax, fabsf(residual[i3]));
- }
-
- uint32_t * yqs = reinterpret_cast(yb->qs);
- yqs[2 * sub + 0] = q0;
- yqs[2 * sub + 1] = q1;
- reinterpret_cast(yb->d4)[sub] = fp8_code;
-+
-+ const uint8_t residual_code = ggml_cuda_fp32_to_ue4m3(residual_amax / 6.0f);
-+ const float residual_scale = ggml_cuda_ue4m3_to_fp32(residual_code);
-+ const float residual_inv_scale = residual_scale > 0.0f ? 0.5f / residual_scale : 0.0f;
-+ uint32_t rq0 = 0;
-+ uint32_t rq1 = 0;
-+#pragma unroll
-+ for (int k = 0; k < QK_NVFP4_SUB / 4; ++k) {
-+ rq0 |= (uint32_t) ggml_cuda_float_to_fp4_e2m1(residual[k + 0], residual_inv_scale) << (8 * k);
-+ rq0 |= (uint32_t) ggml_cuda_float_to_fp4_e2m1(residual[k + 8], residual_inv_scale) << (8 * k + 4);
-+ rq1 |= (uint32_t) ggml_cuda_float_to_fp4_e2m1(residual[k + 4], residual_inv_scale) << (8 * k);
-+ rq1 |= (uint32_t) ggml_cuda_float_to_fp4_e2m1(residual[k + 12], residual_inv_scale) << (8 * k + 4);
-+ }
-+
-+ uint32_t * yrqs = reinterpret_cast(yrb->qs);
-+ yrqs[2 * sub + 0] = rq0;
-+ yrqs[2 * sub + 1] = rq1;
-+ reinterpret_cast(yrb->d4)[sub] = residual_code;
- #else
- NO_DEVICE_CODE; // This is for Blackwell NVFP4 activations only.
- #endif // defined(BLACKWELL_MMA_AVAILABLE)
--
- }
-
- // quantize values in the format mxfp4 is stored which is interleaved nibbles
-@@ -413,7 +420,7 @@ void quantize_mmq_q8_1_cuda(
- }
-
- void quantize_mmq_fp4_cuda(
-- const float * x, const int32_t * ids, void * vy, const ggml_type type_src0,
-+ const float * x, const int32_t * ids, void * vy, void * vy_residual, const ggml_type type_src0,
- const int64_t ne00, const int64_t s01, const int64_t s02, const int64_t s03,
- const int64_t ne0, const int64_t ne1, const int64_t ne2, const int64_t ne3, cudaStream_t stream) {
- GGML_ASSERT(type_src0 == GGML_TYPE_MXFP4 || type_src0 == GGML_TYPE_NVFP4);
-@@ -421,13 +428,16 @@ void quantize_mmq_fp4_cuda(
-
- if (type_src0 == GGML_TYPE_NVFP4) {
- GGML_ASSERT(ne00 % QK_NVFP4 == 0);
-+ GGML_ASSERT(vy_residual != nullptr);
- constexpr int nvfp4_block_size = 128;
-- const int64_t block_num_y = (ne0 + QK_NVFP4_SUB * nvfp4_block_size - 1) / (QK_NVFP4_SUB * nvfp4_block_size);
-+ const int64_t block_num_y =
-+ (ne0 + QK_NVFP4_SUB * nvfp4_block_size - 1) / (QK_NVFP4_SUB * nvfp4_block_size);
- const dim3 block_size(nvfp4_block_size, 1, 1);
- const dim3 num_blocks(ne1, block_num_y, ne2 * ne3);
- quantize_mmq_nvfp4<<>>(
-- x, ids, vy, ne00, s01, s02, s03, ne0, ne1, ne2);
-+ x, ids, vy, vy_residual, ne00, s01, s02, s03, ne0, ne1, ne2);
- } else {
-+ GGML_ASSERT(vy_residual == nullptr);
- GGML_ASSERT(ne0 % (2 * QK_MXFP4) == 0);
-
- constexpr int nwarps = 8;
-diff --git a/src/ggml-cuda/quantize.cuh b/src/ggml-cuda/quantize.cuh
-index 768a3ae6..84c708ba 100644
---- a/src/ggml-cuda/quantize.cuh
-+++ b/src/ggml-cuda/quantize.cuh
-@@ -29,6 +29,7 @@ void quantize_mmq_q8_1_cuda(
- void quantize_mmq_fp4_cuda(const float * x,
- const int32_t * ids,
- void * vy,
-+ void * vy_residual,
- ggml_type type_src0,
- int64_t ne00,
- int64_t s01,
-diff --git a/tests/test-backend-ops.cpp b/tests/test-backend-ops.cpp
-index f54ab41c..e325787f 100644
---- a/tests/test-backend-ops.cpp
-+++ b/tests/test-backend-ops.cpp
-@@ -3977,7 +3977,7 @@ struct test_mul_mat : public test_case {
-
- double max_nmse_err(ggml_backend_t backend) override {
- // for blackwell we quantize activations to mxfp4 instead of q8_1 so we add higher tolerance
-- if ((type_a == GGML_TYPE_MXFP4 || type_a == GGML_TYPE_NVFP4) && backend_has_feature(backend, "BLACKWELL_NATIVE_FP4")) {
-+ if (type_a == GGML_TYPE_MXFP4 && backend_has_feature(backend, "BLACKWELL_NATIVE_FP4")) {
- return 2e-2;
- }
- return max_nmse_err();
-@@ -4166,7 +4166,7 @@ struct test_mul_mat_id : public test_case {
-
- double max_nmse_err(ggml_backend_t backend) override {
- // for blackwell we quantize activations to mxfp4 instead of q8_1 so we add higher tolerance
-- if ((type_a == GGML_TYPE_MXFP4 || type_a == GGML_TYPE_NVFP4) && backend_has_feature(backend, "BLACKWELL_NATIVE_FP4")) {
-+ if (type_a == GGML_TYPE_MXFP4 && backend_has_feature(backend, "BLACKWELL_NATIVE_FP4")) {
- return 2e-2;
- }
- return max_nmse_err();
diff --git a/ggml-patches/0005-skinny-q8-gemm.patch b/ggml-patches/0005-skinny-q8-gemm.patch
deleted file mode 100644
index 2b7de6b..0000000
--- a/ggml-patches/0005-skinny-q8-gemm.patch
+++ /dev/null
@@ -1,646 +0,0 @@
-diff --git a/src/ggml-cuda/skinny-q8.cu b/src/ggml-cuda/skinny-q8.cu
-new file mode 100644
-index 00000000..60147ff1
---- /dev/null
-+++ b/src/ggml-cuda/skinny-q8.cu
-@@ -0,0 +1,616 @@
-+// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved.
-+// SPDX-License-Identifier: Apache-2.0
-+// Skinny-N Q8_0 GEMM for streaming ASR encoders (nemo-speech).
-+//
-+// This kernel is specialized for 9 <= N <= 64, where the activation tile fits
-+// on chip and weight reads dominate memory traffic.
-+//
-+// It uses int8 tensor cores and a cp.async pipeline:
-+// * mma.sync.m16n8k32.s8 — one MMA spans exactly one q8 block of k, so the
-+// per-block scales (weight d x activation d) apply directly to the s32
-+// MMA result, accumulated in fp32.
-+// * Block tile = 32 rows x 64 cols: 8 warps as 2 row-halves x 4 col-quarters
-+// (each warp: 16x16 = 2 MMAs per k-block). All N tiles are launched in
-+// grid.z so a large outer batch fills the GPU instead of issuing a serial
-+// stream of sub-SM grids.
-+// * A 2-buffer pipeline stages W (32x128B), A (64x128B), and their scales per
-+// 128-k step. The next buffer is issued before the current buffer computes,
-+// keeping one cp.async group in flight without paying for a third buffer.
-+// * Weights use aligned planes qs[M][K] + d[M][K/32]. Models can serialize
-+// those planes directly; standard block_q8_0 weights are repacked once on
-+// first use and cached. Activations are quantized to a zero-padded
-+// 64-column int8 buffer so staging needs no bounds checks.
-+//
-+// Bit-accuracy: same q8 activation round-trip as mul_mat_q, but a different
-+// summation order — results are not bit-identical to mmq. Dispatch must
-+// therefore depend only on the per-sequence matrix shape, never on the outer
-+// batch size, so singleton and batched execution cannot select different math.
-+
-+#include "skinny-q8.cuh"
-+#include "mmvq.cuh" // MMVQ_MAX_BATCH_SIZE: the N range mmvq already covers
-+
-+#include
-+#include
-+#include
-+#include
-+
-+#define SKQ8_WARPS 8
-+#define SKQ8_ROWS 32 // rows per block (2 warp row-halves x 16); M % 32 == 0
-+#define SKQ8_NPAD 64 // max padded cols (4 warp col-quarters x 2 n-tiles x 8)
-+#define SKQ8_KSTEP 128 // k bytes staged per pipeline stage; K % KSTEP == 0
-+#define SKQ8_KTB (SKQ8_KSTEP / 32) // q8 blocks per stage
-+#define SKQ8_STAGES 2
-+#define SKQ8_NMAX SKQ8_NPAD
-+
-+// Shared-memory strides (bytes). KSTEP + 16 keeps the 16B cp.async stores
-+// aligned while breaking the power-of-two bank pattern on fragment loads.
-+#define SKQ8_SW (SKQ8_KSTEP + 16)
-+#define SKQ8_SA (SKQ8_KSTEP + 16)
-+// Per-stage layout: W tile | A tile | W scales (half) | A scales (float).
-+// The A tile + A scales are sized for NPAD but only ncols (= n-tiles actually
-+// used, host-padded to 8) are staged/consumed.
-+#define SKQ8_STAGE_W (SKQ8_ROWS * SKQ8_SW)
-+#define SKQ8_STAGE_A (SKQ8_NPAD * SKQ8_SA)
-+#define SKQ8_STAGE_DW (SKQ8_ROWS * SKQ8_KTB * 2)
-+#define SKQ8_STAGE_DA (SKQ8_NPAD * SKQ8_KTB * 4)
-+#define SKQ8_STAGE_BYTES (SKQ8_STAGE_W + SKQ8_STAGE_A + SKQ8_STAGE_DW + SKQ8_STAGE_DA)
-+
-+// ---------------------------------------------------------------------------
-+// Repack block_q8_0[M][K/32] -> qs plane int8[M][K] + d plane half[M][K/32].
-+static __global__ void skq8_repack(
-+ const void * __restrict__ src, int8_t * __restrict__ qs, half * __restrict__ d,
-+ const int64_t M, const int64_t KB) {
-+ const int64_t i = (int64_t) blockIdx.x * blockDim.x + threadIdx.x;
-+ if (i >= M * KB) {
-+ return;
-+ }
-+ const block_q8_0 * b = (const block_q8_0 *) src + i;
-+ d[i] = b->d;
-+ int8_t * out = qs + i * 32;
-+#pragma unroll
-+ for (int j = 0; j < 32; j++) {
-+ out[j] = b->qs[j];
-+ }
-+}
-+
-+// ---------------------------------------------------------------------------
-+// Quantize F32 activations [K, N] (col-contiguous) into a zero-padded
-+// SKQ8_NPAD-column buffer: int8[NPAD][K] + d float[NPAD][K/32].
-+static __global__ void skq8_quantize(
-+ const float * __restrict__ X, int8_t * __restrict__ Aq, float * __restrict__ Ad,
-+ const int K, const int N) {
-+ // One WARP per (col, q8-block): lane j owns element j, so the 32-float
-+ // read is one coalesced 128B transaction and the int8 store one 32B
-+ // transaction (a thread-per-block version did 32 scalar strided loads and
-+ // cost 5x the GEMM quantize budget).
-+ const int KB = K / 32;
-+ const int wid = (blockIdx.x * blockDim.x + threadIdx.x) / 32;
-+ const int lane = threadIdx.x % 32;
-+ const int ncols = (N + 7) / 8 * 8; // active column tiles only
-+ if (wid >= ncols * KB) {
-+ return;
-+ }
-+ const int n = wid / KB, kb = wid % KB;
-+ int8_t * q = Aq + (size_t) n * K + kb * 32;
-+ if (n >= N) {
-+ // Pad columns: only the scale must be zeroed — the GEMM multiplies the
-+ // integer MMA result by da, and int math on stale qs bytes is finite,
-+ // so da == 0 makes the contribution exactly 0.
-+ if (lane == 0) {
-+ Ad[(size_t) n * KB + kb] = 0.0f;
-+ }
-+ return;
-+ }
-+ const float x = X[(size_t) n * K + kb * 32 + lane];
-+ float amax = fabsf(x);
-+#pragma unroll
-+ for (int off = 16; off > 0; off >>= 1) {
-+ amax = fmaxf(amax, __shfl_xor_sync(0xffffffff, amax, off));
-+ }
-+ const float dv = amax / 127.0f;
-+ const float id = dv > 0.0f ? 1.0f / dv : 0.0f;
-+ q[lane] = (int8_t) roundf(x * id);
-+ if (lane == 0) {
-+ Ad[(size_t) n * KB + kb] = dv;
-+ }
-+}
-+
-+// ---------------------------------------------------------------------------
-+#if defined(__CUDA_ARCH__) && __CUDA_ARCH__ >= 800
-+static __device__ __forceinline__ void skq8_cp16(void * dst, const void * src) {
-+ const unsigned s = (unsigned) __cvta_generic_to_shared(dst);
-+ asm volatile("cp.async.cg.shared.global [%0], [%1], 16;" ::"r"(s), "l"(src));
-+}
-+static __device__ __forceinline__ void skq8_cp8(void * dst, const void * src) {
-+ const unsigned s = (unsigned) __cvta_generic_to_shared(dst);
-+ asm volatile("cp.async.ca.shared.global [%0], [%1], 8;" ::"r"(s), "l"(src));
-+}
-+static __device__ __forceinline__ void skq8_cp4(void * dst, const void * src) {
-+ const unsigned s = (unsigned) __cvta_generic_to_shared(dst);
-+ asm volatile("cp.async.ca.shared.global [%0], [%1], 4;" ::"r"(s), "l"(src));
-+}
-+static __device__ __forceinline__ void skq8_commit() {
-+ asm volatile("cp.async.commit_group;");
-+}
-+template static __device__ __forceinline__ void skq8_wait() {
-+ asm volatile("cp.async.wait_group %0;" ::"n"(n));
-+}
-+static __device__ __forceinline__ void skq8_mma(
-+ int & d0, int & d1, int & d2, int & d3, int a0, int a1, int a2, int a3, int b0, int b1) {
-+ asm volatile(
-+ "mma.sync.aligned.m16n8k32.row.col.s32.s8.s8.s32 "
-+ "{%0,%1,%2,%3}, {%4,%5,%6,%7}, {%8,%9}, {%0,%1,%2,%3};"
-+ : "+r"(d0), "+r"(d1), "+r"(d2), "+r"(d3)
-+ : "r"(a0), "r"(a1), "r"(a2), "r"(a3), "r"(b0), "r"(b1));
-+}
-+#endif
-+
-+// C[M,N] (F32, col stride M) = Wq8 x A. Small-M GEMMs split K over
-+// blockIdx.y (gridDim.y k-ranges, scales already folded per range); each
-+// split writes a private plane for deterministic reduction afterward.
-+static __global__ __launch_bounds__(SKQ8_WARPS * 32, 5) void skq8_gemm(
-+ const int8_t * __restrict__ Wq, const half * __restrict__ Wd, const int8_t * __restrict__ Aq,
-+ const float * __restrict__ Ad, const float * __restrict__ bias, float * __restrict__ C,
-+ const int M, const int N_total, const int K, const int ncols_total /* N padded to 8 */) {
-+#if defined(__CUDA_ARCH__) && __CUDA_ARCH__ >= 800
-+ const int KB = K / 32;
-+ const int warp = threadIdx.x / 32;
-+ const int lane = threadIdx.x % 32;
-+ const int gid = lane >> 2; // mma group id (0..7)
-+ const int tid4 = lane & 3; // mma thread-in-group (0..3)
-+
-+ const int k_len = K / gridDim.y; // multiple of SKQ8_KSTEP (host-enforced)
-+ const int k0 = blockIdx.y * k_len;
-+
-+ // grid.z carries independent 64-column output tiles. Keeping them in one
-+ // launch is essential on high-SM-count GPUs: a typical encoder projection
-+ // has only 128 row/K-split CTAs and can underfill the device. The old host
-+ // loop repeated that underfilled launch per tile and serialized the
-+ // entire outer batch. Pointer rebasing keeps all inner indexing unchanged,
-+ // including the per-q8-block FP32 accumulation order.
-+ const int cbase = (int) blockIdx.z * SKQ8_NPAD;
-+ const int ncols = min(SKQ8_NPAD, ncols_total - cbase);
-+ const int N = min(SKQ8_NPAD, N_total - cbase);
-+ Aq += (size_t) cbase * K;
-+ Ad += (size_t) cbase * KB;
-+ C += (size_t) cbase * M;
-+
-+ const int row_blk = blockIdx.x * SKQ8_ROWS;
-+ const int mhalf = warp / 4; // which 16-row half
-+ const int nt0 = (warp % 4) * 2; // first of this warp's two 8-col n-tiles
-+
-+ extern __shared__ char smem[];
-+ char * stage[SKQ8_STAGES];
-+#pragma unroll
-+ for (int s = 0; s < SKQ8_STAGES; s++) {
-+ stage[s] = smem + (size_t) s * SKQ8_STAGE_BYTES;
-+ }
-+
-+ const int nsteps = k_len / SKQ8_KSTEP;
-+
-+ // Stage `ks` of this block's k-range into buffer ks % STAGES.
-+ auto issue = [&](int ks) {
-+ char * buf = stage[ks % SKQ8_STAGES];
-+ char * s_w = buf;
-+ char * s_a = buf + SKQ8_STAGE_W;
-+ char * s_dw = s_a + SKQ8_STAGE_A;
-+ char * s_da = s_dw + SKQ8_STAGE_DW;
-+ const int kbyte = k0 + ks * SKQ8_KSTEP; // k offset in elements (== bytes for s8)
-+ const int kbb = kbyte / 32; // k offset in q8 blocks
-+ constexpr int vw = SKQ8_KSTEP / 16; // 16B vectors per row per stage
-+ // W tile: ROWS x vw x 16B
-+ for (int it = threadIdx.x; it < SKQ8_ROWS * vw; it += blockDim.x) {
-+ const int r = it / vw, v = it % vw;
-+ skq8_cp16(
-+ s_w + r * SKQ8_SW + v * 16, Wq + (size_t) (row_blk + r) * K + kbyte + v * 16);
-+ }
-+ // A tile: ncols x vw x 16B (only the active column tiles)
-+ for (int it = threadIdx.x; it < ncols * vw; it += blockDim.x) {
-+ const int c = it / vw, v = it % vw;
-+ skq8_cp16(s_a + c * SKQ8_SA + v * 16, Aq + (size_t) c * K + kbyte + v * 16);
-+ }
-+ // scales: dw ROWS x KTB half (KTB*2 bytes/row), da ncols x KTB float
-+ // (KTB*4 bytes/col); copy width follows KTB.
-+ if (threadIdx.x < SKQ8_ROWS) {
-+ if (SKQ8_KTB * 2 == 16) {
-+ skq8_cp16(
-+ s_dw + threadIdx.x * (SKQ8_KTB * 2),
-+ Wd + (size_t) (row_blk + threadIdx.x) * KB + kbb);
-+ } else if (SKQ8_KTB * 2 == 8) {
-+ skq8_cp8(
-+ s_dw + threadIdx.x * (SKQ8_KTB * 2),
-+ Wd + (size_t) (row_blk + threadIdx.x) * KB + kbb);
-+ } else {
-+ skq8_cp4(
-+ s_dw + threadIdx.x * (SKQ8_KTB * 2),
-+ Wd + (size_t) (row_blk + threadIdx.x) * KB + kbb);
-+ }
-+ }
-+ constexpr int da_vec = SKQ8_KTB * 4 / 16; // 16B vectors per col
-+ if constexpr (da_vec > 0) {
-+ for (int it = threadIdx.x; it < ncols * da_vec; it += blockDim.x) {
-+ const int c = it / da_vec, h = it % da_vec;
-+ skq8_cp16(
-+ s_da + c * (SKQ8_KTB * 4) + h * 16, Ad + (size_t) c * KB + kbb + h * 4);
-+ }
-+ } else if (threadIdx.x < ncols) {
-+ skq8_cp8(
-+ s_da + threadIdx.x * (SKQ8_KTB * 4),
-+ Ad + (size_t) threadIdx.x * KB + kbb);
-+ }
-+ skq8_commit();
-+ };
-+
-+ float facc[2][4] = { { 0.0f } };
-+
-+ issue(0);
-+
-+ for (int ks = 0; ks < nsteps; ks++) {
-+ // The next stage is issued after the current buffer is ready and before
-+ // its MMA loop. This preserves compute/copy overlap with two buffers;
-+ // the trailing barrier makes reusing the just-consumed buffer safe.
-+ skq8_wait<0>();
-+ __syncthreads();
-+
-+ if (ks + 1 < nsteps) {
-+ issue(ks + 1);
-+ }
-+
-+ const char * buf = stage[ks % SKQ8_STAGES];
-+ const char * s_w = buf;
-+ const char * s_a = buf + SKQ8_STAGE_W;
-+ const half * s_dw = (const half *) (buf + SKQ8_STAGE_W + SKQ8_STAGE_A);
-+ const float * s_da =
-+ (const float *) (buf + SKQ8_STAGE_W + SKQ8_STAGE_A + SKQ8_STAGE_DW);
-+
-+#pragma unroll
-+ for (int kb = 0; kb < SKQ8_KSTEP / 32; kb++) {
-+ // W fragment for this warp's 16 rows (rows mhalf*16 + gid, +8).
-+ const char * wbase = s_w + (mhalf * 16 + gid) * SKQ8_SW + kb * 32 + tid4 * 4;
-+ const int a0 = *(const int *) (wbase);
-+ const int a1 = *(const int *) (wbase + 8 * SKQ8_SW);
-+ const int a2 = *(const int *) (wbase + 16);
-+ const int a3 = *(const int *) (wbase + 8 * SKQ8_SW + 16);
-+ const float dw0 = __half2float(s_dw[(mhalf * 16 + gid) * SKQ8_KTB + kb]);
-+ const float dw1 = __half2float(s_dw[(mhalf * 16 + gid + 8) * SKQ8_KTB + kb]);
-+#pragma unroll
-+ for (int t = 0; t < 2; t++) {
-+ const int nt = nt0 + t;
-+ if (nt * 8 >= ncols) {
-+ continue; // dead column tile (N padded to ncols < NPAD)
-+ }
-+ const char * bbase = s_a + (nt * 8 + gid) * SKQ8_SA + kb * 32 + tid4 * 4;
-+ const int b0 = *(const int *) (bbase);
-+ const int b1 = *(const int *) (bbase + 16);
-+ int d0 = 0, d1 = 0, d2 = 0, d3 = 0;
-+ skq8_mma(d0, d1, d2, d3, a0, a1, a2, a3, b0, b1);
-+ const int c0 = nt * 8 + tid4 * 2;
-+ const float da0 = s_da[c0 * SKQ8_KTB + kb];
-+ const float da1 = s_da[(c0 + 1) * SKQ8_KTB + kb];
-+ facc[t][0] += dw0 * da0 * (float) d0;
-+ facc[t][1] += dw0 * da1 * (float) d1;
-+ facc[t][2] += dw1 * da0 * (float) d2;
-+ facc[t][3] += dw1 * da1 * (float) d3;
-+ }
-+ }
-+ __syncthreads();
-+ }
-+
-+ // Write back: C rows mhalf*16 + gid (+8), cols nt*8 + tid4*2 (+1).
-+ // With a K-split grid, each split writes a private output plane. A
-+ // follow-up kernel reduces those planes in a fixed order, avoiding the
-+ // run-to-run numerical drift caused by unordered FP32 atomic additions.
-+ const int row0 = row_blk + mhalf * 16 + gid;
-+ const float b0 = (bias != nullptr && blockIdx.y == 0) ? bias[row0] : 0.0f;
-+ const float b1 = (bias != nullptr && blockIdx.y == 0) ? bias[row0 + 8] : 0.0f;
-+#pragma unroll
-+ for (int t = 0; t < 2; t++) {
-+ facc[t][0] += b0;
-+ facc[t][1] += b0;
-+ facc[t][2] += b1;
-+ facc[t][3] += b1;
-+ }
-+#pragma unroll
-+ for (int t = 0; t < 2; t++) {
-+ const int c0 = (nt0 + t) * 8 + tid4 * 2;
-+ if (gridDim.y > 1) {
-+ // Plane stride covers the full logical output. `N` is only this
-+ // grid.z tile's width (<= SKQ8_NPAD), so using it here aliases the
-+ // next tile whenever N_total > SKQ8_NPAD.
-+ const size_t split_base = (size_t) blockIdx.y * M * N_total;
-+ if (c0 < N) {
-+ C[split_base + row0 + (size_t) c0 * M] = facc[t][0];
-+ C[split_base + row0 + 8 + (size_t) c0 * M] = facc[t][2];
-+ }
-+ if (c0 + 1 < N) {
-+ C[split_base + row0 + (size_t) (c0 + 1) * M] = facc[t][1];
-+ C[split_base + row0 + 8 + (size_t) (c0 + 1) * M] = facc[t][3];
-+ }
-+ } else {
-+ if (c0 < N) {
-+ C[row0 + (size_t) c0 * M] = facc[t][0];
-+ C[row0 + 8 + (size_t) c0 * M] = facc[t][2];
-+ }
-+ if (c0 + 1 < N) {
-+ C[row0 + (size_t) (c0 + 1) * M] = facc[t][1];
-+ C[row0 + 8 + (size_t) (c0 + 1) * M] = facc[t][3];
-+ }
-+ }
-+ }
-+#else
-+ GGML_UNUSED_VARS(Wq, Wd, Aq, Ad, bias, C, M, N_total, K, ncols_total);
-+ NO_DEVICE_CODE;
-+#endif
-+}
-+
-+static __global__ void skq8_reduce_splitk(
-+ const float * __restrict__ partials, float * __restrict__ dst,
-+ const int64_t count, const int ksplit) {
-+ const int64_t i = (int64_t) blockIdx.x * blockDim.x + threadIdx.x;
-+ if (i >= count) {
-+ return;
-+ }
-+ float sum = partials[i];
-+ for (int split = 1; split < ksplit; ++split) {
-+ sum += partials[(int64_t) split * count + i];
-+ }
-+ dst[i] = sum;
-+}
-+
-+// Repacked-weight cache. Keyed by the weight tensor's device pointer; entries
-+// live for the process lifetime (model weights are loaded once). Repacking
-+// happens on first (eager/warmup) use, before any CUDA-graph capture.
-+namespace {
-+struct skq8_planes {
-+ int8_t * qs;
-+ half * d;
-+};
-+std::unordered_map g_skq8_cache;
-+std::mutex g_skq8_mutex;
-+} // namespace
-+
-+bool ggml_cuda_skinny_q8_supported(
-+ const ggml_tensor * src0, const ggml_tensor * src1, const ggml_tensor * dst) {
-+ static const bool disabled = []() {
-+ const char * e = getenv("GGML_SKINNY_Q8");
-+ return e != nullptr && e[0] == '0';
-+ }();
-+ if (disabled) {
-+ return false;
-+ }
-+ // Resolve views before checking the persistent model-weight marker. A
-+ // serialized planar tensor is accepted only through its full allocation,
-+ // since a normal ggml byte-offset view cannot describe two separate
-+ // tensor-wide planes.
-+ const ggml_tensor * root = src0;
-+ while (root->view_src != nullptr) {
-+ root = root->view_src;
-+ }
-+ const bool planar_q8 = (root->flags & GGML_TENSOR_FLAG_Q8_PLANAR) != 0;
-+ // This specialization is for FastConformer encoder matrices. Keeping the
-+ // domain explicit also prevents a decoder's request batch (which commonly
-+ // occupies ne[1]) from being mistaken for a skinny time dimension.
-+ if (strncmp(root->name, "encoder.", 8) != 0) {
-+ return false;
-+ }
-+ if (!planar_q8 && strstr(root->name, ".linear_qkv.weight") != nullptr) {
-+ return false;
-+ }
-+ // Tensor-wide planes cannot be addressed through a stock block-row view.
-+ // Current encoder graphs use the full fused QKV tensor, never a weight
-+ // view; reject defensively if that changes.
-+ if (planar_q8 && src0 != root) {
-+ return false;
-+ }
-+
-+ const int cc = ggml_cuda_info().devices[ggml_cuda_get_device()].cc;
-+ const bool skinny_q8_available = ampere_mma_available(cc);
-+ if (!planar_q8 && !skinny_q8_available) {
-+ return false;
-+ }
-+ // The repack cache is keyed by src0->data and assumes the bytes are
-+ // immutable from the outside: only accept long-lived weight buffers, never
-+ // transient compute-pool tensors (whose addresses get reused).
-+ if (src0->buffer == nullptr ||
-+ ggml_backend_buffer_get_usage(src0->buffer) == GGML_BACKEND_BUFFER_USAGE_COMPUTE) {
-+ return false;
-+ }
-+
-+ // Weight-side (per-tensor, call-invariant) conditions.
-+ const bool tensor_ok = src0->type == GGML_TYPE_Q8_0 && ggml_is_contiguous(src0) &&
-+ src0->ne[2] == 1 && src0->ne[3] == 1 &&
-+ src0->ne[0] % SKQ8_KSTEP == 0 && src0->ne[1] % SKQ8_ROWS == 0;
-+ // Call-side conditions.
-+ // A dense batched RHS [K,N,B...] is byte-identical to [K,N*B...]. Treat
-+ // all outer columns as one logical N so a weight repacked by a scalar
-+ // graph remains usable by a later true-batch graph.
-+ const int64_t total_n = ggml_nelements(src1) / src1->ne[0];
-+ const bool call_ok = src1->type == GGML_TYPE_F32 && dst->type == GGML_TYPE_F32 &&
-+ ggml_is_contiguous(src1) && ggml_is_contiguous(dst) &&
-+ src1->ne[0] == src0->ne[0] &&
-+ ggml_nelements(dst) == src0->ne[1] * total_n;
-+
-+ if (planar_q8) {
-+ // N<=8 is handled by planar MMVQ. Wider calls must stay on this path
-+ // because stock MMQ cannot interpret tensor-wide planes; skq8_gemm
-+ // tiles arbitrary N in 64-column chunks.
-+ GGML_ASSERT(tensor_ok && call_ok && "planar Q8 weight used in unsupported mul_mat shape");
-+ GGML_ASSERT((skinny_q8_available || total_n <= MMVQ_MAX_BATCH_SIZE) &&
-+ "wide planar Q8 requires an SM80+ CUDA kernel; use block Q8 on older GPUs");
-+ return skinny_q8_available && total_n > MMVQ_MAX_BATCH_SIZE;
-+ }
-+
-+ const bool repacked = [&]() {
-+ std::lock_guard lock(g_skq8_mutex);
-+ return g_skq8_cache.find(src0->data) != g_skq8_cache.end();
-+ }();
-+ if (repacked) {
-+ // The weight was converted IN PLACE to the plane layout — the
-+ // block_q8_0 bytes no longer exist, so every mul_mat on this tensor
-+ // must come through here (the >64-col path tiles, small N pads).
-+ // Falling back to mmq/mmvq would silently read garbage: abort loudly
-+ // instead if a call shape we cannot serve ever appears.
-+ GGML_ASSERT(call_ok && "skinny-q8: repacked weight used in an unsupported mul_mat shape");
-+ return true;
-+ }
-+ // By default, select only on logical width so outer batch size cannot
-+ // change the accumulation path. The opt-in mode is an end-to-end ASR
-+ // experiment for streaming shapes such as [K,2,B]: it flattens the dense
-+ // outer batch and lets the tensor-core kernel tile N beyond 64. It is not
-+ // the default because skinny-Q8 has a different accumulation order from
-+ // MMVQ; callers must validate transcript/accuracy parity. Use a separate
-+ // repack allocation (GGML_SKINNY_Q8_INPLACE=0) with multi-stream schedulers.
-+ static const bool outer_batch_dispatch = []() {
-+ const char * e = getenv("GGML_SKINNY_Q8_OUTER_BATCH");
-+ return e != nullptr && e[0] != '0';
-+ }();
-+ const int64_t logical_n = src1->ne[1];
-+ const bool logical_eligible = logical_n > MMVQ_MAX_BATCH_SIZE && logical_n <= SKQ8_NMAX;
-+ const bool outer_eligible = total_n > MMVQ_MAX_BATCH_SIZE;
-+ const bool eligible =
-+ tensor_ok && call_ok && (outer_batch_dispatch ? outer_eligible : logical_eligible);
-+ return eligible;
-+}
-+
-+static void skq8_run(
-+ ggml_backend_cuda_context & ctx, const ggml_tensor * src0, const ggml_tensor * src1,
-+ const ggml_tensor * bias, ggml_tensor * dst) {
-+ const int64_t M = src0->ne[1];
-+ const int64_t K = src0->ne[0];
-+ const int64_t N = ggml_nelements(src1) / src1->ne[0];
-+ const int64_t KB = K / 32;
-+
-+ cudaStream_t stream = ctx.stream();
-+
-+ // Repacked weight planes (create on first use). Default: IN PLACE. The plane
-+ // layout (qs M*K + d M*KB*2) is byte-for-byte the same total size as
-+ // block_q8_0 (M*KB*34), so it repacks through a transient pool staging buffer
-+ // and copies back over the original allocation: zero extra weight memory (the
-+ // cudaMalloc'd duplicate cost ~1.07 GB on parakeet-xxl). After this the
-+ // tensor's block_q8_0 layout is GONE; the dispatch in
-+ // ggml_cuda_skinny_q8_supported() therefore claims every mul_mat on a
-+ // repacked tensor.
-+ //
-+ // GGML_SKINNY_Q8_INPLACE=0 forces the separate-buffer (cudaMalloc) layout.
-+ // The in-place D2D memcpy is a stream-ordering hazard under multi-stream
-+ // graph-split scheduling (llama.cpp's NMT decoder): it corrupts the GEMM even
-+ // though the repacked bytes are correct. Callers that share the process with
-+ // such a scheduler set GGML_SKINNY_Q8_INPLACE=0 (the NMT pipeline does this
-+ // when enabled). The streaming-ASR encoder runtime has no such hazard.
-+ skq8_planes planes;
-+ {
-+ std::lock_guard lock(g_skq8_mutex);
-+ auto it = g_skq8_cache.find(src0->data);
-+ if (it != g_skq8_cache.end()) {
-+ planes = it->second;
-+ } else {
-+ const size_t qs_bytes = (size_t) M * K;
-+ const size_t d_bytes = (size_t) M * KB * sizeof(half);
-+ GGML_ASSERT(qs_bytes + d_bytes == ggml_nbytes(src0));
-+ if ((src0->flags & GGML_TENSOR_FLAG_Q8_PLANAR) != 0) {
-+ planes.qs = (int8_t *) src0->data;
-+ planes.d = (half *) ((char *) src0->data + qs_bytes);
-+ g_skq8_cache.emplace(src0->data, planes);
-+ } else {
-+ static const bool inplace = []() {
-+ const char * e = getenv("GGML_SKINNY_Q8_INPLACE");
-+ return e == nullptr || e[0] != '0';
-+ }();
-+ const int64_t total = M * KB;
-+ const int blocks = (int) ((total + 255) / 256);
-+ size_t extra = 0;
-+ if (inplace) {
-+ ggml_cuda_pool_alloc staging(ctx.pool(), qs_bytes + d_bytes);
-+ skq8_repack<<>>(
-+ src0->data, staging.get(), (half *) (staging.get() + qs_bytes), M, KB);
-+ CUDA_CHECK(cudaMemcpyAsync(
-+ src0->data, staging.get(), qs_bytes + d_bytes, cudaMemcpyDeviceToDevice,
-+ stream));
-+ planes.qs = (int8_t *) src0->data;
-+ planes.d = (half *) ((char *) src0->data + qs_bytes);
-+ // staging returns to the pool at scope exit; the copy is
-+ // stream-ordered before any later reuse on this stream.
-+ } else {
-+ CUDA_CHECK(cudaMalloc(&planes.qs, qs_bytes));
-+ CUDA_CHECK(cudaMalloc(&planes.d, d_bytes));
-+ skq8_repack<<>>(
-+ src0->data, planes.qs, planes.d, M, KB);
-+ extra = qs_bytes + d_bytes;
-+ }
-+ g_skq8_cache.emplace(src0->data, planes);
-+ static const bool memstats = getenv("NEMO_SPEECH_MEMSTATS") != nullptr;
-+ if (memstats) {
-+ static size_t g_repack_total = 0;
-+ g_repack_total += extra;
-+ fprintf(
-+ stderr,
-+ "[memstats] skinny-q8 repack %s +%.1f MB (cache extra total %.1f MB, "
-+ "%zu tensors)\n",
-+ inplace ? "in-place" : "alloc", extra / 1048576.0,
-+ g_repack_total / 1048576.0, g_skq8_cache.size());
-+ }
-+ }
-+ }
-+ }
-+ // Quantize activations into the zero-padded buffer (all columns, once —
-+ // ntot may exceed NPAD; the GEMM below tiles over 64-col chunks).
-+ const int ntot = (int) ((N + SKQ8_NPAD - 1) / SKQ8_NPAD * SKQ8_NPAD);
-+ ggml_cuda_pool_alloc aq_alloc(ctx.pool(), (size_t) ntot * K);
-+ ggml_cuda_pool_alloc ad_alloc(ctx.pool(), (size_t) ntot * KB);
-+ {
-+ const int64_t warps = (int64_t) ntot * KB; // one warp per (col, q8 block)
-+ const int blocks = (int) ((warps * 32 + 255) / 256);
-+ skq8_quantize<<>>(
-+ (const float *) src1->data, aq_alloc.get(), ad_alloc.get(), (int) K, (int) N);
-+ }
-+
-+ // Split K for small-M shapes until there is enough block-level parallelism.
-+ // Each split writes a private plane and a deterministic reduction combines
-+ // the planes afterward; unordered atomics can amplify into visible
-+ // B=1-versus-batched drift across a deep encoder.
-+ // 1-step splits are allowed: no pipeline, but block-level K-parallelism.
-+ // N > NPAD is mapped into grid.z. Weights are still read once per 64-column
-+ // tile, but all tiles are visible to the scheduler in a single launch.
-+ {
-+ const int row_blocks = (int) (M / SKQ8_ROWS);
-+ int ksplit = 1;
-+ while (ksplit < 4 && row_blocks * ksplit < 128 &&
-+ K % ((int64_t) SKQ8_KSTEP * ksplit * 2) == 0) {
-+ ksplit *= 2;
-+ }
-+ const size_t smem = (size_t) SKQ8_STAGES * SKQ8_STAGE_BYTES;
-+ static std::once_flag attr_set_flag;
-+ std::call_once(attr_set_flag, [&]() {
-+ CUDA_CHECK(cudaFuncSetAttribute(
-+ skq8_gemm, cudaFuncAttributeMaxDynamicSharedMemorySize, (int) smem));
-+ });
-+ const int ntiles = (ntot + SKQ8_NPAD - 1) / SKQ8_NPAD;
-+ const dim3 grid(row_blocks, ksplit, ntiles);
-+ const float * bias_data = bias != nullptr ? (const float *) bias->data : nullptr;
-+ if (ksplit > 1) {
-+ const int64_t count = M * N;
-+ ggml_cuda_pool_alloc partials(ctx.pool(), (size_t) ksplit * count);
-+ skq8_gemm<<>>(
-+ planes.qs, planes.d, aq_alloc.get(), ad_alloc.get(), bias_data, partials.get(),
-+ (int) M, (int) N, (int) K, ntot);
-+ skq8_reduce_splitk<<<(count + 255) / 256, 256, 0, stream>>>(
-+ partials.get(), (float *) dst->data, count, ksplit);
-+ } else {
-+ skq8_gemm<<>>(
-+ planes.qs, planes.d, aq_alloc.get(), ad_alloc.get(), bias_data,
-+ (float *) dst->data, (int) M, (int) N, (int) K, ntot);
-+ }
-+ }
-+}
-+
-+void ggml_cuda_mul_mat_skinny_q8(
-+ ggml_backend_cuda_context & ctx, const ggml_tensor * src0, const ggml_tensor * src1,
-+ ggml_tensor * dst) {
-+ skq8_run(ctx, src0, src1, nullptr, dst);
-+}
-+
-+void ggml_cuda_mul_mat_skinny_q8_bias(
-+ ggml_backend_cuda_context & ctx, const ggml_tensor * src0, const ggml_tensor * src1,
-+ const ggml_tensor * bias, ggml_tensor * dst) {
-+ skq8_run(ctx, src0, src1, bias, dst);
-+}
-diff --git a/src/ggml-cuda/skinny-q8.cuh b/src/ggml-cuda/skinny-q8.cuh
-new file mode 100644
-index 00000000..17175bbb
---- /dev/null
-+++ b/src/ggml-cuda/skinny-q8.cuh
-@@ -0,0 +1,18 @@
-+// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved.
-+// SPDX-License-Identifier: Apache-2.0
-+#include "common.cuh"
-+
-+// Skinny-N (9..64 cols) Q8_0 x F32 GEMM specialized for streaming encoders.
-+// See skinny-q8.cu for the design notes.
-+bool ggml_cuda_skinny_q8_supported(
-+ const ggml_tensor * src0, const ggml_tensor * src1, const ggml_tensor * dst);
-+
-+void ggml_cuda_mul_mat_skinny_q8(
-+ ggml_backend_cuda_context & ctx, const ggml_tensor * src0, const ggml_tensor * src1,
-+ ggml_tensor * dst);
-+
-+// Fused variant: adds a row-vector bias ([M], F32) in the GEMM epilogue and
-+// writes the result to `dst` (the bias-add node's buffer).
-+void ggml_cuda_mul_mat_skinny_q8_bias(
-+ ggml_backend_cuda_context & ctx, const ggml_tensor * src0, const ggml_tensor * src1,
-+ const ggml_tensor * bias, ggml_tensor * dst);
diff --git a/ggml-patches/0006-cuda-dispatch-wiring.patch b/ggml-patches/0006-cuda-dispatch-wiring.patch
deleted file mode 100644
index 94110ba..0000000
--- a/ggml-patches/0006-cuda-dispatch-wiring.patch
+++ /dev/null
@@ -1,671 +0,0 @@
-diff --git a/include/ggml.h b/include/ggml.h
-index 0b6945ac..b32d0432 100644
---- a/include/ggml.h
-+++ b/include/ggml.h
-@@ -648,6 +648,10 @@ extern "C" {
- GGML_TENSOR_FLAG_PARAM = 4, // ...contains trainable parameters
- GGML_TENSOR_FLAG_LOSS = 8, // ...defines loss for numerical optimization (multiple loss tensors add up)
- GGML_TENSOR_FLAG_COMPUTE = 16, // ...must be computed
-+ // Model weight uses Q8_0 bytes serialized as one tensor-wide int8
-+ // plane followed by one FP16 scale plane. CUDA-only storage hint;
-+ // logical type, element count, and allocation size remain Q8_0.
-+ GGML_TENSOR_FLAG_Q8_PLANAR = 32,
- };
-
- enum ggml_tri_type {
-diff --git a/src/ggml-cuda/common.cuh b/src/ggml-cuda/common.cuh
-index 10817505..9bd40faa 100644
---- a/src/ggml-cuda/common.cuh
-+++ b/src/ggml-cuda/common.cuh
-@@ -1479,11 +1479,18 @@ struct ggml_cuda_mm_fusion_args_host {
- const ggml_tensor * x_bias = nullptr;
- const ggml_tensor * gate = nullptr;
- const ggml_tensor * gate_bias = nullptr;
-+ bool silu = false;
- ggml_glu_op glu_op;
- };
- struct ggml_cuda_mm_fusion_args_device {
- const void * x_bias = nullptr;
- const void * gate = nullptr;
- const void * gate_bias = nullptr;
-+ // A Linear bias has shape [M, 1, 1, 1] and is broadcast across the
-+ // destination columns/channels. Keep this explicit because the legacy
-+ // fusion path also accepts a full-shape elementwise bias.
-+ bool x_bias_broadcast = false;
-+ bool gate_bias_broadcast = false;
-+ bool silu = false;
- ggml_glu_op glu_op;
- };
-diff --git a/src/ggml-cuda/ggml-cuda.cu b/src/ggml-cuda/ggml-cuda.cu
-index e25be359..6f3e1c42 100644
---- a/src/ggml-cuda/ggml-cuda.cu
-+++ b/src/ggml-cuda/ggml-cuda.cu
-@@ -16,6 +16,7 @@
- #include "ggml-cuda/conv2d.cuh"
- #include "ggml-cuda/conv2d-dw.cuh"
- #include "ggml-cuda/conv2d-transpose.cuh"
-+#include "ggml-cuda/skinny-q8.cuh"
- #include "ggml-cuda/convert.cuh"
- #include "ggml-cuda/count-equal.cuh"
- #include "ggml-cuda/cpy.cuh"
-@@ -24,6 +25,7 @@
- #include "ggml-cuda/diagmask.cuh"
- #include "ggml-cuda/diag.cuh"
- #include "ggml-cuda/fattn.cuh"
-+#include "ggml-cuda/fused-relpos-attn.cuh"
- #include "ggml-cuda/getrows.cuh"
- #include "ggml-cuda/im2col.cuh"
- #include "ggml-cuda/mmf.cuh"
-@@ -2505,16 +2507,20 @@ static bool ggml_cuda_should_fuse_mul_mat_vec_q(const ggml_tensor * tensor) {
- ggml_nbytes(src0) != ggml_backend_buffer_get_alloc_size(src0->buffer, src0) &&
- src0->view_src;
-
-+ const int64_t total_n = ggml_nelements(src1) / src1->ne[0];
- bool use_mul_mat_vec_q = ggml_is_quantized(src0->type) && !bad_padding_clear && src1->type == GGML_TYPE_F32 &&
-- dst->type == GGML_TYPE_F32 && src1->ne[1] <= MMVQ_MAX_BATCH_SIZE;
-+ dst->type == GGML_TYPE_F32 && total_n <= MMVQ_MAX_BATCH_SIZE;
-
- // fusion is not universally faster on Pascal
- const int cc = ggml_cuda_info().devices[ggml_cuda_get_device()].cc;
- if (cc <= GGML_CUDA_CC_PASCAL) {
- return false;
- }
-- //we only support fusion for ncols_dst = 1
-- if (tensor->op == GGML_OP_MUL_MAT && dst->ne[1] != 1) {
-+ // MMVQ's narrow epilogue supports the two-frame streaming chunk used by
-+ // NeMo-Speech.cpp. Outer request-batch columns are included in
-+ // total_n above so [K,2,B] stays on skinny-Q8 when B makes it wider than
-+ // MMVQ_MAX_BATCH_SIZE.
-+ if (tensor->op == GGML_OP_MUL_MAT && dst->ne[1] > 2) {
- return false;
- }
-
-@@ -2534,6 +2540,19 @@ static bool ggml_cuda_should_fuse_mul_mat_vec_q(const ggml_tensor * tensor) {
- return use_mul_mat_vec_q;
- }
-
-+static bool ggml_cuda_q8_narrow_epilogue_enabled() {
-+ static const bool enabled = [] {
-+ const char * value = getenv("GGML_CUDA_Q8_NARROW_EPILOGUE");
-+ // Keep the former variable name as an alias for existing
-+ // deployments.
-+ if (value == nullptr) {
-+ value = getenv("GGML_CUDA_Q8_NARROW_BIAS_FUSION");
-+ }
-+ return value == nullptr || value[0] != '0';
-+ }();
-+ return enabled;
-+}
-+
- static void ggml_cuda_mul_mat(ggml_backend_cuda_context & ctx, const ggml_tensor * src0, const ggml_tensor * src1, ggml_tensor * dst) {
- const bool split = ggml_backend_buft_is_cuda_split(src0->buffer->buft);
-
-@@ -2594,7 +2613,11 @@ static void ggml_cuda_mul_mat(ggml_backend_cuda_context & ctx, const ggml_tensor
- bool use_batched_cublas_bf16 = src0->type == GGML_TYPE_BF16 && bf16_mma_hardware_available(cc);
- bool use_batched_cublas_f32 = src0->type == GGML_TYPE_F32;
-
-- if (!split && use_mul_mat_vec_f) {
-+ if (!split && !bad_padding_clear && ggml_cuda_skinny_q8_supported(src0, src1, dst)) {
-+ // streaming-ASR specialization: Q8_0 weights x skinny (9..64 col)
-+ // activations — beats mul_mat_q's LLM-batch tiling at these shapes
-+ ggml_cuda_mul_mat_skinny_q8(ctx, src0, src1, dst);
-+ } else if (!split && use_mul_mat_vec_f) {
- // the custom F16 vector kernel can be used over batched cuBLAS GEMM
- // but this is only faster for GPUs without tensor cores or with a thin src0 matrix (particularly KQV in attention)
- ggml_cuda_mul_mat_vec_f(ctx, src0, src1, nullptr, dst);
-@@ -3071,6 +3094,9 @@ static bool ggml_cuda_compute_forward(ggml_backend_cuda_context & ctx, struct gg
- case GGML_OP_FLASH_ATTN_EXT:
- ggml_cuda_flash_attn_ext(ctx, dst);
- break;
-+ case GGML_OP_FUSED_RELPOS_ATTN:
-+ ggml_cuda_op_fused_relpos_attn(ctx, dst);
-+ break;
- case GGML_OP_CROSS_ENTROPY_LOSS:
- ggml_cuda_cross_entropy_loss(ctx, dst);
- break;
-@@ -3662,6 +3688,51 @@ static bool ggml_cuda_can_fuse(const struct ggml_cgraph * cgraph,
- return false;
- }
-
-+ // Standard LayerNorm (GGML_OP_NORM) + row-vector gamma (+ row-vector beta).
-+ // Restricted to the classic affine pattern: the mul/add operands must be
-+ // contiguous ne0-length vectors broadcast over rows — that is what the
-+ // fused norm_mul_add_f32 kernel implements (see norm.cu).
-+ if ((ops.size() == 2 || ops.size() == 3) && ops.begin()[0] == GGML_OP_NORM && ops.begin()[1] == GGML_OP_MUL) {
-+ const ggml_tensor * norm = cgraph->nodes[node_idx];
-+ const ggml_tensor * mul = cgraph->nodes[node_idx+1];
-+ const ggml_tensor * add = nullptr;
-+
-+ if (ops.size() == 3) {
-+ if (ops.begin()[2] != GGML_OP_ADD) {
-+ return false;
-+ }
-+ add = cgraph->nodes[node_idx+2];
-+ }
-+
-+ if (norm->src[0]->type != GGML_TYPE_F32 || norm->type != GGML_TYPE_F32 ||
-+ mul->type != GGML_TYPE_F32 || (add && add->type != GGML_TYPE_F32)) {
-+ return false;
-+ }
-+
-+ const ggml_tensor * gamma =
-+ mul->src[0] == norm ? mul->src[1] : (mul->src[1] == norm ? mul->src[0] : nullptr);
-+ if (gamma == nullptr || gamma->type != GGML_TYPE_F32 ||
-+ ggml_nelements(gamma) != norm->ne[0] || !ggml_is_contiguous(gamma)) {
-+ return false;
-+ }
-+
-+ if (add) {
-+ const ggml_tensor * beta =
-+ add->src[0] == mul ? add->src[1] : (add->src[1] == mul ? add->src[0] : nullptr);
-+ if (beta == nullptr || beta->type != GGML_TYPE_F32 ||
-+ ggml_nelements(beta) != norm->ne[0] || !ggml_is_contiguous(beta)) {
-+ return false;
-+ }
-+ }
-+
-+ // the fused kernel writes dst rows contiguously
-+ if (!ggml_is_contiguous(mul) || (add && !ggml_is_contiguous(add))) {
-+ return false;
-+ }
-+
-+ return true;
-+ }
-+
- if ((ops.size() == 2 || ops.size() == 3) && ops.begin()[0] == GGML_OP_RMS_NORM && ops.begin()[1] == GGML_OP_MUL) {
- const ggml_tensor *rms_norm = cgraph->nodes[node_idx];
- const ggml_tensor *mul = cgraph->nodes[node_idx+1];
-@@ -4119,6 +4190,66 @@ static int ggml_cuda_try_fuse(ggml_backend_cuda_context * cuda_ctx, ggml_cgraph
- fused_mul_mat_vec = false;
- fused_node_count = 0;
-
-+ // Q8 narrow projection + SiLU. NeMo-Speech.cpp's Conformer FF1
-+ // layers are bias-free, so this is their actual hot graph sequence.
-+ if (ggml_cuda_q8_narrow_epilogue_enabled() &&
-+ ggml_can_fuse(cgraph, i, { GGML_OP_MUL_MAT, GGML_OP_UNARY })) {
-+ ggml_tensor * mm_node = cgraph->nodes[i];
-+ ggml_tensor * silu_node = cgraph->nodes[i + 1];
-+ if (ggml_get_unary_op(silu_node) == GGML_UNARY_OP_SILU &&
-+ silu_node->src[0] == mm_node && ggml_cuda_should_fuse_mul_mat_vec_q(mm_node)) {
-+ ggml_cuda_mm_fusion_args_host fusion_data{};
-+ fusion_data.silu = true;
-+ ggml_cuda_mul_mat_vec_q(
-+ *cuda_ctx, mm_node->src[0], mm_node->src[1], nullptr, silu_node, &fusion_data);
-+ return 1;
-+ }
-+ }
-+
-+ // Q8 narrow projection + broadcast Linear bias + SiLU. This is the hot
-+ // Conformer FF1 sequence at the two-frame streaming chunk size. Folding
-+ // both epilogues avoids two launches and two full output round trips.
-+ if (ggml_cuda_q8_narrow_epilogue_enabled() &&
-+ ggml_can_fuse(cgraph, i, { GGML_OP_MUL_MAT, GGML_OP_ADD, GGML_OP_UNARY })) {
-+ ggml_tensor * mm_node = cgraph->nodes[i];
-+ ggml_tensor * bias_node = cgraph->nodes[i + 1];
-+ ggml_tensor * silu_node = cgraph->nodes[i + 2];
-+ const ggml_tensor * bias = bias_node->src[0] == mm_node ? bias_node->src[1] :
-+ bias_node->src[1] == mm_node ? bias_node->src[0] :
-+ nullptr;
-+ if (ggml_get_unary_op(silu_node) == GGML_UNARY_OP_SILU &&
-+ silu_node->src[0] == bias_node && bias != nullptr &&
-+ bias->type == GGML_TYPE_F32 && ggml_is_contiguous(bias) &&
-+ ggml_nelements(bias) == mm_node->ne[0] &&
-+ ggml_cuda_should_fuse_mul_mat_vec_q(mm_node)) {
-+ ggml_cuda_mm_fusion_args_host fusion_data{};
-+ fusion_data.x_bias = bias;
-+ fusion_data.silu = true;
-+ ggml_cuda_mul_mat_vec_q(
-+ *cuda_ctx, mm_node->src[0], mm_node->src[1], nullptr, silu_node, &fusion_data);
-+ return 2;
-+ }
-+ }
-+
-+ // skinny-q8 GEMM + row-vector bias: the upstream MUL_MAT+ADD fusion below
-+ // requires a same-shape add, so the classic broadcast Linear bias
-+ // ([M] over [M,N]) never qualifies — fold it into the skinny GEMM's
-+ // epilogue instead.
-+ if (ggml_can_fuse(cgraph, i, { GGML_OP_MUL_MAT, GGML_OP_ADD })) {
-+ ggml_tensor * mm_node = cgraph->nodes[i];
-+ ggml_tensor * bias_node = cgraph->nodes[i + 1];
-+ const ggml_tensor * bias = bias_node->src[0] == mm_node ? bias_node->src[1] :
-+ bias_node->src[1] == mm_node ? bias_node->src[0] :
-+ nullptr;
-+ if (bias != nullptr && bias->type == GGML_TYPE_F32 && ggml_is_contiguous(bias) &&
-+ ggml_nelements(bias) == mm_node->ne[0] && ggml_is_contiguous(bias_node) &&
-+ ggml_cuda_skinny_q8_supported(mm_node->src[0], mm_node->src[1], mm_node)) {
-+ ggml_cuda_mul_mat_skinny_q8_bias(
-+ *cuda_ctx, mm_node->src[0], mm_node->src[1], bias, bias_node);
-+ return 1;
-+ }
-+ }
-+
- // gate + add + glu + up + add
- for (ggml_op op : { GGML_OP_MUL_MAT, GGML_OP_MUL_MAT_ID }) {
- const ggml_op bias_op = op == GGML_OP_MUL_MAT ? GGML_OP_ADD : GGML_OP_ADD_ID;
-@@ -4154,14 +4285,21 @@ static int ggml_cuda_try_fuse(ggml_backend_cuda_context * cuda_ctx, ggml_cgraph
- continue;
- }
-
-- if (bias_op == GGML_OP_ADD && !ggml_are_same_shape(bias_node->src[0], bias_node->src[1])) {
-+ const bool same_shape_bias = bias_op != GGML_OP_ADD ||
-+ ggml_are_same_shape(bias_node->src[0], bias_node->src[1]);
-+ const bool broadcast_q_bias = ggml_cuda_q8_narrow_epilogue_enabled() &&
-+ bias_op == GGML_OP_ADD &&
-+ bias_tensor->type == GGML_TYPE_F32 &&
-+ ggml_is_contiguous(bias_tensor) &&
-+ ggml_nelements(bias_tensor) == mm_node->ne[0];
-+ if (!same_shape_bias && !broadcast_q_bias) {
- continue;
- }
-
- ggml_cuda_mm_fusion_args_host fusion_data{};
- fusion_data.x_bias = bias_tensor;
-
-- if (ggml_cuda_should_fuse_mul_mat_vec_f(mm_node)) {
-+ if (same_shape_bias && ggml_cuda_should_fuse_mul_mat_vec_f(mm_node)) {
- ggml_cuda_mul_mat_vec_f(*cuda_ctx, src0, src1, ids, bias_node, &fusion_data);
- fused_mul_mat_vec = true;
- fused_node_count = 2;
-@@ -4190,6 +4328,16 @@ static int ggml_cuda_try_fuse(ggml_backend_cuda_context * cuda_ctx, ggml_cgraph
- return 1;
- }
-
-+ if (ggml_cuda_can_fuse(cgraph, i, { GGML_OP_NORM, GGML_OP_MUL, GGML_OP_ADD }, {})) {
-+ ggml_cuda_op_norm_fused(*cuda_ctx, node, cgraph->nodes[i + 1], cgraph->nodes[i + 2]);
-+ return 2;
-+ }
-+
-+ if (ggml_cuda_can_fuse(cgraph, i, { GGML_OP_NORM, GGML_OP_MUL }, {})) {
-+ ggml_cuda_op_norm_fused(*cuda_ctx, node, cgraph->nodes[i + 1], nullptr);
-+ return 1;
-+ }
-+
- if (ggml_cuda_can_fuse(cgraph, i, { GGML_OP_SSM_CONV, GGML_OP_ADD, GGML_OP_UNARY }, { GGML_UNARY_OP_SILU })) {
- ggml_cuda_op_ssm_conv(*cuda_ctx, node, cgraph->nodes[i + 1], cgraph->nodes[i + 2]);
- return 2;
-@@ -5402,6 +5550,8 @@ static bool ggml_backend_cuda_device_supports_op(ggml_backend_dev_t dev, const g
- #endif // GGML_USE_MUSA
- case GGML_OP_FLASH_ATTN_EXT:
- return ggml_cuda_flash_attn_ext_supported(dev_ctx->device, op);
-+ case GGML_OP_FUSED_RELPOS_ATTN:
-+ return (op->src[0]->ne[0] & (op->src[0]->ne[0] - 1)) == 0; // d_k power of two
- case GGML_OP_CROSS_ENTROPY_LOSS:
- case GGML_OP_CROSS_ENTROPY_LOSS_BACK:
- case GGML_OP_OPT_STEP_ADAMW:
-diff --git a/src/ggml-cuda/mmvq.cu b/src/ggml-cuda/mmvq.cu
-index da48f313..03668cb8 100644
---- a/src/ggml-cuda/mmvq.cu
-+++ b/src/ggml-cuda/mmvq.cu
-@@ -391,7 +391,17 @@ static constexpr __host__ __device__ int calc_rows_per_block(int ncols_dst, int
- return 1;
- }
-
--template
-+template
-+static __device__ __forceinline__ float vec_dot_q8_0_q8_1_planar(
-+ const void * __restrict__ vx, const block_q8_1 * __restrict__ y,
-+ int block_idx, int iqs, size_t scale_plane_offset) {
-+ const int * v = (const int *) vx + (size_t) block_idx * (QK8_0 / 4) + iqs;
-+ const int * u = (const int *) y->qs + iqs;
-+ const half * d = (const half *) ((const char *) vx + scale_plane_offset);
-+ return vec_dot_q8_0_q8_1_impl(v, u, d[block_idx], __low2half(y->ds));
-+}
-+
-+template
- __launch_bounds__(calc_nwarps(type, ncols_dst, get_device_table_id())*ggml_cuda_get_physical_warp_size(), 1)
- static __global__ void mul_mat_vec_q(
- const void * __restrict__ vx, const void * __restrict__ vy, const int32_t * __restrict__ ids, const ggml_cuda_mm_fusion_args_device fusion, float * __restrict__ dst,
-@@ -415,6 +425,10 @@ static __global__ void mul_mat_vec_q(
- const int row0 = rows_per_cuda_block*blockIdx.x;
- const int blocks_per_row_x = ncols_x / qk;
- constexpr int blocks_per_iter = vdr * nwarps*warp_size / qi;
-+ // Serialized planar weights are restricted to contiguous 2-D matrices.
-+ // stride_col_dst is therefore the source row count, and one int8 byte is
-+ // stored per logical weight before the FP16 scale plane.
-+ const size_t q8_scale_plane_offset = (size_t) ncols_x * stride_col_dst;
-
- const uint32_t channel_dst = blockIdx.y;
-
-@@ -432,6 +446,7 @@ static __global__ void mul_mat_vec_q(
- bool use_gate = false;
- bool use_bias = false;
- bool use_gate_bias = false;
-+ bool use_silu = false;
- const void * vgate = nullptr;
- const float * x_bias = nullptr;
- const float * gate_bias = nullptr;
-@@ -441,6 +456,7 @@ static __global__ void mul_mat_vec_q(
- use_gate = fusion.gate != nullptr;
- use_bias = fusion.x_bias != nullptr;
- use_gate_bias = fusion.gate_bias != nullptr && use_gate;
-+ use_silu = fusion.silu;
- vgate = fusion.gate;
- x_bias = (const float *) fusion.x_bias;
- gate_bias = (const float *) fusion.gate_bias;
-@@ -453,24 +469,26 @@ static __global__ void mul_mat_vec_q(
- if constexpr (has_fusion) {
- const uint32_t channel_bias = ids ? channel_x : channel_dst;
- if (use_bias) {
-- x_bias = x_bias + sample_dst*stride_sample_dst + channel_bias*stride_channel_dst + row0;
-+ x_bias = x_bias + (fusion.x_bias_broadcast ? row0 :
-+ sample_dst*stride_sample_dst + channel_bias*stride_channel_dst + row0);
- // 1. Hide latency by prefetching bias and gate here
- // 2. load only on threads that won't die after partial sum calculation
- if (threadIdx.x < rows_per_cuda_block && threadIdx.y == 0 &&
- (rows_per_cuda_block == 1 || uint32_t(row0 + threadIdx.x) < stride_col_dst)) {
- #pragma unroll
- for (int j = 0; j < ncols_dst; ++j) {
-- x_biases[j] = x_bias[j * stride_col_dst + threadIdx.x];
-+ x_biases[j] = x_bias[(fusion.x_bias_broadcast ? 0 : j * stride_col_dst) + threadIdx.x];
- }
- }
- }
- if (use_gate_bias) {
-- gate_bias = gate_bias + sample_dst*stride_sample_dst + channel_bias*stride_channel_dst + row0;
-+ gate_bias = gate_bias + (fusion.gate_bias_broadcast ? row0 :
-+ sample_dst*stride_sample_dst + channel_bias*stride_channel_dst + row0);
- if (threadIdx.x < rows_per_cuda_block && threadIdx.y == 0 &&
- (rows_per_cuda_block == 1 || uint32_t(row0 + threadIdx.x) < stride_col_dst)) {
- #pragma unroll
- for (int j = 0; j < ncols_dst; ++j) {
-- gate_biases[j] = gate_bias[j * stride_col_dst + threadIdx.x];
-+ gate_biases[j] = gate_bias[(fusion.gate_bias_broadcast ? 0 : j * stride_col_dst) + threadIdx.x];
- }
- }
- }
-@@ -493,12 +511,26 @@ static __global__ void mul_mat_vec_q(
- for (int j = 0; j < ncols_dst; ++j) {
- #pragma unroll
- for (int i = 0; i < rows_per_cuda_block; ++i) {
-- tmp[j][i] += vec_dot_q_cuda(
-- vx, &y[j*stride_col_y + kby], kbx_offset + i*stride_row_x + kbx, kqs);
-+ if constexpr (q8_planar) {
-+ static_assert(type == GGML_TYPE_Q8_0, "planar MMVQ is Q8_0-only");
-+ tmp[j][i] += vec_dot_q8_0_q8_1_planar(
-+ vx, &y[j*stride_col_y + kby],
-+ kbx_offset + i*stride_row_x + kbx, kqs, q8_scale_plane_offset);
-+ } else {
-+ tmp[j][i] += vec_dot_q_cuda(
-+ vx, &y[j*stride_col_y + kby], kbx_offset + i*stride_row_x + kbx, kqs);
-+ }
- if constexpr (has_fusion) {
- if (use_gate) {
-- tmp_gate[j][i] += vec_dot_q_cuda(
-- vgate, &y[j*stride_col_y + kby], kbx_offset + i*stride_row_x + kbx, kqs);
-+ if constexpr (q8_planar) {
-+ tmp_gate[j][i] += vec_dot_q8_0_q8_1_planar(
-+ vgate, &y[j*stride_col_y + kby],
-+ kbx_offset + i*stride_row_x + kbx, kqs, q8_scale_plane_offset);
-+ } else {
-+ tmp_gate[j][i] += vec_dot_q_cuda(
-+ vgate, &y[j*stride_col_y + kby],
-+ kbx_offset + i*stride_row_x + kbx, kqs);
-+ }
- }
- }
- }
-@@ -583,13 +615,16 @@ static __global__ void mul_mat_vec_q(
- break;
- }
- }
-+ if (use_silu) {
-+ result = ggml_cuda_op_silu_single(result);
-+ }
- }
- dst[j*stride_col_dst + threadIdx.x] = result;
- }
- }
-
- if constexpr (!has_fusion) {
-- GGML_UNUSED_VARS(use_gate, use_bias, use_gate_bias, active_glu, gate_bias, x_bias, tmp_gate);
-+ GGML_UNUSED_VARS(use_gate, use_bias, use_gate_bias, use_silu, active_glu, gate_bias, x_bias, tmp_gate);
- }
- }
-
-@@ -668,7 +703,7 @@ static std::pair calc_launch_params(
- return {block_nums, block_dims};
- }
-
--template
-+template
- static void mul_mat_vec_q_switch_fusion(
- const void * vx, const void * vy, const int32_t * ids, const ggml_cuda_mm_fusion_args_device fusion, float * dst,
- const uint32_t ncols_x, const uint3 nchannels_y, const uint32_t stride_row_x, const uint32_t stride_col_y,
-@@ -678,10 +713,11 @@ static void mul_mat_vec_q_switch_fusion(
- const dim3 & block_nums, const dim3 & block_dims, const int nbytes_shared,
- const uint32_t ids_stride, cudaStream_t stream) {
-
-- const bool has_fusion = fusion.gate != nullptr || fusion.x_bias != nullptr || fusion.gate_bias != nullptr;
-- if constexpr (c_ncols_dst == 1) {
-+ const bool has_fusion = fusion.gate != nullptr || fusion.x_bias != nullptr ||
-+ fusion.gate_bias != nullptr || fusion.silu;
-+ if constexpr (c_ncols_dst <= 2) {
- if (has_fusion) {
-- mul_mat_vec_q<<>>
-+ mul_mat_vec_q<<>>
- (vx, vy, ids, fusion, dst, ncols_x, nchannels_y, stride_row_x, stride_col_y, stride_col_dst,
- channel_ratio, stride_channel_x, stride_channel_y, stride_channel_dst,
- sample_ratio, stride_sample_x, stride_sample_y, stride_sample_dst, ids_stride);
-@@ -689,9 +725,9 @@ static void mul_mat_vec_q_switch_fusion(
- }
- }
-
-- GGML_ASSERT(!has_fusion && "fusion only supported for ncols_dst=1");
-+ GGML_ASSERT(!has_fusion && "fusion only supported for ncols_dst<=2");
-
-- mul_mat_vec_q<<>>
-+ mul_mat_vec_q<<>>
- (vx, vy, ids, fusion, dst, ncols_x, nchannels_y, stride_row_x, stride_col_y, stride_col_dst,
- channel_ratio, stride_channel_x, stride_channel_y, stride_channel_dst,
- sample_ratio, stride_sample_x, stride_sample_y, stride_sample_dst, ids_stride);
-@@ -718,7 +754,7 @@ static void mul_mat_vec_q_moe_launch(
- ncols_dst, ids_stride);
- }
-
--template
-+template
- static void mul_mat_vec_q_switch_ncols_dst(
- const void * vx, const void * vy, const int32_t * ids, const ggml_cuda_mm_fusion_args_device fusion, float * dst,
- const int ncols_x, const int nrows_x, const int ncols_dst,
-@@ -730,6 +766,7 @@ static void mul_mat_vec_q_switch_ncols_dst(
-
- GGML_ASSERT(ncols_x % ggml_blck_size(type) == 0);
- GGML_ASSERT(ncols_dst <= MMVQ_MAX_BATCH_SIZE);
-+ static_assert(!q8_planar || type == GGML_TYPE_Q8_0, "planar MMVQ is Q8_0-only");
-
- const uint3 nchannels_y_fd = ids ? init_fastdiv_values(nchannels_y) : make_uint3(0, 0, 0);
- const uint3 channel_ratio_fd = ids ? make_uint3(0, 0, 0) : init_fastdiv_values(nchannels_dst / nchannels_x);
-@@ -786,6 +823,7 @@ static void mul_mat_vec_q_switch_ncols_dst(
- };
-
- if (has_ids && ncols_dst > 1) {
-+ GGML_ASSERT(!q8_planar && "planar Q8 does not support MUL_MAT_ID");
- // Multi-token MUL_MAT_ID path - dedicated MoE kernel
- mul_mat_vec_q_moe_launch(
- vx, vy, ids, dst, ncols_x, nchannels_y_fd, nrows_x,
-@@ -804,7 +842,7 @@ static void mul_mat_vec_q_switch_ncols_dst(
- if (use_small_k) {
- std::pair dims = calc_launch_params(c_ncols_dst, nrows_x, nchannels_dst,
- nsamples_dst, warp_size, table_id, true);
-- mul_mat_vec_q_switch_fusion(
-+ mul_mat_vec_q_switch_fusion(
- vx, vy, ids, fusion, dst, ncols_x, nchannels_y_fd, stride_row_x, stride_col_y, stride_col_dst,
- channel_ratio_fd, stride_channel_x, stride_channel_y, stride_channel_dst, sample_ratio_fd,
- stride_sample_x, stride_sample_y, stride_sample_dst, dims.first, dims.second, 0, ids_stride,
-@@ -812,7 +850,7 @@ static void mul_mat_vec_q_switch_ncols_dst(
- } else {
- std::pair dims = calc_launch_params(c_ncols_dst, nrows_x, nchannels_dst,
- nsamples_dst, warp_size, table_id);
-- mul_mat_vec_q_switch_fusion(
-+ mul_mat_vec_q_switch_fusion(
- vx, vy, ids, fusion, dst, ncols_x, nchannels_y_fd, stride_row_x, stride_col_y, stride_col_dst,
- channel_ratio_fd, stride_channel_x, stride_channel_y, stride_channel_dst, sample_ratio_fd,
- stride_sample_x, stride_sample_y, stride_sample_dst, dims.first, dims.second, 0, ids_stride,
-@@ -822,7 +860,7 @@ static void mul_mat_vec_q_switch_ncols_dst(
- case 2: {
- constexpr int c_ncols_dst = 2;
- std::pair dims = calc_launch_params(c_ncols_dst, nrows_x, nchannels_dst, nsamples_dst, warp_size, table_id);
-- mul_mat_vec_q_switch_fusion(vx, vy, ids, fusion, dst, ncols_x, nchannels_y_fd, stride_row_x, stride_col_y, stride_col_dst,
-+ mul_mat_vec_q_switch_fusion(vx, vy, ids, fusion, dst, ncols_x, nchannels_y_fd, stride_row_x, stride_col_y, stride_col_dst,
- channel_ratio_fd, stride_channel_x, stride_channel_y, stride_channel_dst,
- sample_ratio_fd, stride_sample_x, stride_sample_y, stride_sample_dst,
- dims.first, dims.second, 0, ids_stride, stream);
-@@ -830,7 +868,7 @@ static void mul_mat_vec_q_switch_ncols_dst(
- case 3: {
- constexpr int c_ncols_dst = 3;
- std::pair dims = calc_launch_params(c_ncols_dst, nrows_x, nchannels_dst, nsamples_dst, warp_size, table_id);
-- mul_mat_vec_q_switch_fusion(vx, vy, ids, fusion, dst, ncols_x, nchannels_y_fd, stride_row_x, stride_col_y, stride_col_dst,
-+ mul_mat_vec_q_switch_fusion(vx, vy, ids, fusion, dst, ncols_x, nchannels_y_fd, stride_row_x, stride_col_y, stride_col_dst,
- channel_ratio_fd, stride_channel_x, stride_channel_y, stride_channel_dst,
- sample_ratio_fd, stride_sample_x, stride_sample_y, stride_sample_dst,
- dims.first, dims.second, 0, ids_stride, stream);
-@@ -838,7 +876,7 @@ static void mul_mat_vec_q_switch_ncols_dst(
- case 4: {
- constexpr int c_ncols_dst = 4;
- std::pair dims = calc_launch_params(c_ncols_dst, nrows_x, nchannels_dst, nsamples_dst, warp_size, table_id);
-- mul_mat_vec_q_switch_fusion(vx, vy, ids, fusion, dst, ncols_x, nchannels_y_fd, stride_row_x, stride_col_y, stride_col_dst,
-+ mul_mat_vec_q_switch_fusion(vx, vy, ids, fusion, dst, ncols_x, nchannels_y_fd, stride_row_x, stride_col_y, stride_col_dst,
- channel_ratio_fd, stride_channel_x, stride_channel_y, stride_channel_dst,
- sample_ratio_fd, stride_sample_x, stride_sample_y, stride_sample_dst,
- dims.first, dims.second, 0, ids_stride, stream);
-@@ -846,7 +884,7 @@ static void mul_mat_vec_q_switch_ncols_dst(
- case 5: {
- constexpr int c_ncols_dst = 5;
- std::pair dims = calc_launch_params(c_ncols_dst, nrows_x, nchannels_dst, nsamples_dst, warp_size, table_id);
-- mul_mat_vec_q_switch_fusion(vx, vy, ids, fusion, dst, ncols_x, nchannels_y_fd, stride_row_x, stride_col_y, stride_col_dst,
-+ mul_mat_vec_q_switch_fusion(vx, vy, ids, fusion, dst, ncols_x, nchannels_y_fd, stride_row_x, stride_col_y, stride_col_dst,
- channel_ratio_fd, stride_channel_x, stride_channel_y, stride_channel_dst,
- sample_ratio_fd, stride_sample_x, stride_sample_y, stride_sample_dst,
- dims.first, dims.second, 0, ids_stride, stream);
-@@ -854,7 +892,7 @@ static void mul_mat_vec_q_switch_ncols_dst(
- case 6: {
- constexpr int c_ncols_dst = 6;
- std::pair dims = calc_launch_params(c_ncols_dst, nrows_x, nchannels_dst, nsamples_dst, warp_size, table_id);
-- mul_mat_vec_q_switch_fusion(vx, vy, ids, fusion, dst, ncols_x, nchannels_y_fd, stride_row_x, stride_col_y, stride_col_dst,
-+ mul_mat_vec_q_switch_fusion(vx, vy, ids, fusion, dst, ncols_x, nchannels_y_fd, stride_row_x, stride_col_y, stride_col_dst,
- channel_ratio_fd, stride_channel_x, stride_channel_y, stride_channel_dst,
- sample_ratio_fd, stride_sample_x, stride_sample_y, stride_sample_dst,
- dims.first, dims.second, 0, ids_stride, stream);
-@@ -862,7 +900,7 @@ static void mul_mat_vec_q_switch_ncols_dst(
- case 7: {
- constexpr int c_ncols_dst = 7;
- std::pair dims = calc_launch_params(c_ncols_dst, nrows_x, nchannels_dst, nsamples_dst, warp_size, table_id);
-- mul_mat_vec_q_switch_fusion(vx, vy, ids, fusion, dst, ncols_x, nchannels_y_fd, stride_row_x, stride_col_y, stride_col_dst,
-+ mul_mat_vec_q_switch_fusion(vx, vy, ids, fusion, dst, ncols_x, nchannels_y_fd, stride_row_x, stride_col_y, stride_col_dst,
- channel_ratio_fd, stride_channel_x, stride_channel_y, stride_channel_dst,
- sample_ratio_fd, stride_sample_x, stride_sample_y, stride_sample_dst,
- dims.first, dims.second, 0, ids_stride, stream);
-@@ -870,7 +908,7 @@ static void mul_mat_vec_q_switch_ncols_dst(
- case 8: {
- constexpr int c_ncols_dst = 8;
- std::pair dims = calc_launch_params(c_ncols_dst, nrows_x, nchannels_dst, nsamples_dst, warp_size, table_id);
-- mul_mat_vec_q_switch_fusion(vx, vy, ids, fusion, dst, ncols_x, nchannels_y_fd, stride_row_x, stride_col_y, stride_col_dst,
-+ mul_mat_vec_q_switch_fusion(vx, vy, ids, fusion, dst, ncols_x, nchannels_y_fd, stride_row_x, stride_col_y, stride_col_dst,
- channel_ratio_fd, stride_channel_x, stride_channel_y, stride_channel_dst,
- sample_ratio_fd, stride_sample_x, stride_sample_y, stride_sample_dst,
- dims.first, dims.second, 0, ids_stride, stream);
-@@ -889,7 +927,8 @@ static void mul_mat_vec_q_switch_type(
- const int nchannels_x, const int nchannels_y, const int nchannels_dst,
- const int stride_channel_x, const int stride_channel_y, const int stride_channel_dst,
- const int nsamples_x, const int nsamples_dst, const int stride_sample_x, const int stride_sample_y, const int stride_sample_dst,
-- const int ids_stride, cudaStream_t stream) {
-+ const int ids_stride, const bool q8_planar, cudaStream_t stream) {
-+ GGML_ASSERT(!q8_planar || type_x == GGML_TYPE_Q8_0);
- switch (type_x) {
- case GGML_TYPE_Q1_0:
- mul_mat_vec_q_switch_ncols_dst
-@@ -922,10 +961,17 @@ static void mul_mat_vec_q_switch_type(
- nsamples_x, nsamples_dst, stride_sample_x, stride_sample_y, stride_sample_dst, ids_stride, stream);
- break;
- case GGML_TYPE_Q8_0:
-- mul_mat_vec_q_switch_ncols_dst
-- (vx, vy, ids, fusion, dst, ncols_x, nrows_x, ncols_dst, stride_row_x, stride_col_y, stride_col_dst,
-- nchannels_x, nchannels_y, nchannels_dst, stride_channel_x, stride_channel_y, stride_channel_dst,
-- nsamples_x, nsamples_dst, stride_sample_x, stride_sample_y, stride_sample_dst, ids_stride, stream);
-+ if (q8_planar) {
-+ mul_mat_vec_q_switch_ncols_dst
-+ (vx, vy, ids, fusion, dst, ncols_x, nrows_x, ncols_dst, stride_row_x, stride_col_y, stride_col_dst,
-+ nchannels_x, nchannels_y, nchannels_dst, stride_channel_x, stride_channel_y, stride_channel_dst,
-+ nsamples_x, nsamples_dst, stride_sample_x, stride_sample_y, stride_sample_dst, ids_stride, stream);
-+ } else {
-+ mul_mat_vec_q_switch_ncols_dst
-+ (vx, vy, ids, fusion, dst, ncols_x, nrows_x, ncols_dst, stride_row_x, stride_col_y, stride_col_dst,
-+ nchannels_x, nchannels_y, nchannels_dst, stride_channel_x, stride_channel_y, stride_channel_dst,
-+ nsamples_x, nsamples_dst, stride_sample_x, stride_sample_y, stride_sample_dst, ids_stride, stream);
-+ }
- break;
- case GGML_TYPE_MXFP4:
- mul_mat_vec_q_switch_ncols_dst
-@@ -1059,13 +1105,14 @@ void ggml_cuda_mul_mat_vec_q(
-
- if (fusion) {
- GGML_ASSERT( !ids || dst->ne[2] == 1);
-- GGML_ASSERT( ids || dst->ne[1] == 1);
-+ GGML_ASSERT( ids || dst->ne[1] <= 2);
-
- if (fusion->x_bias) {
- GGML_ASSERT(fusion->x_bias->type == GGML_TYPE_F32);
- GGML_ASSERT(fusion->x_bias->ne[0] == dst->ne[0]);
- GGML_ASSERT(!ids || fusion->x_bias->ne[1] == src0->ne[2]);
- fusion_local.x_bias = fusion->x_bias->data;
-+ fusion_local.x_bias_broadcast = ggml_nelements(fusion->x_bias) == dst->ne[0];
- }
- if (fusion->gate) {
- GGML_ASSERT(fusion->gate->type == src0->type && ggml_are_same_stride(fusion->gate, src0));
-@@ -1076,8 +1123,10 @@ void ggml_cuda_mul_mat_vec_q(
- GGML_ASSERT(fusion->gate_bias->ne[0] == dst->ne[0]);
- GGML_ASSERT(!ids || fusion->gate_bias->ne[1] == src0->ne[2]);
- fusion_local.gate_bias = fusion->gate_bias->data;
-+ fusion_local.gate_bias_broadcast = ggml_nelements(fusion->gate_bias) == dst->ne[0];
- }
- fusion_local.glu_op = fusion->glu_op;
-+ fusion_local.silu = fusion->silu;
- }
-
- // If src0 is a temporary compute buffer, clear any potential padding.
-@@ -1121,12 +1170,18 @@ void ggml_cuda_mul_mat_vec_q(
- const int64_t stride_channel_y = ids ? s11 : s12;
-
- const int64_t ids_stride = ids ? ids->nb[1] / ggml_type_size(ids->type) : 0;
-+ const bool q8_planar = (src0->flags & GGML_TENSOR_FLAG_Q8_PLANAR) != 0;
-+ if (q8_planar) {
-+ GGML_ASSERT(src0->type == GGML_TYPE_Q8_0 && ids == nullptr);
-+ GGML_ASSERT(src0->ne[2] == 1 && src0->ne[3] == 1 && ggml_is_contiguous(src0));
-+ }
-
- mul_mat_vec_q_switch_type(
- src0->data, src0->type, src1_q8_1.get(), ids_d, fusion_local, dst_d, ne00,
- ne01, ncols_dst, s01, stride_col_y, stride_col_dst,
- ne02, nchannels_y, nchannels_dst, s02, stride_channel_y, stride_channel_dst,
-- ne03, ne3, s03, s13, s3, ids_stride, stream);
-+ ne03, ne3, s03, s13, s3, ids_stride,
-+ q8_planar, stream);
- }
-
- void ggml_cuda_op_mul_mat_vec_q(
-@@ -1153,9 +1208,11 @@ void ggml_cuda_op_mul_mat_vec_q(
- const int stride_col_y = src1_padded_row_size / QK8_1;
-
- ggml_cuda_mm_fusion_args_device fusion_local{};
-+ GGML_ASSERT((src0->flags & GGML_TENSOR_FLAG_Q8_PLANAR) == 0 &&
-+ "planar Q8 does not support split-buffer MMVQ");
- mul_mat_vec_q_switch_type(
- src0_dd_i, src0->type, src1_ddq_i, nullptr, fusion_local, dst_dd_i, ne00, row_diff, src1_ncols, stride_row_x, stride_col_y, nrows_dst,
-- 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 0, stream);
-+ 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 0, false, stream);
-
- GGML_UNUSED_VARS(src1, dst, src1_ddf_i, src1_ncols, src1_padded_row_size);
- }
-diff --git a/src/ggml-cuda/vecdotq.cuh b/src/ggml-cuda/vecdotq.cuh
-index d1741cc8..33b54ab3 100644
---- a/src/ggml-cuda/vecdotq.cuh
-+++ b/src/ggml-cuda/vecdotq.cuh
-@@ -237,7 +237,7 @@ template static __device__ __forceinline__ float vec_dot_q5_1_q8_1_imp
- return sumi*d5d8 + m5s8 / (QI5_1 / vdr);
- }
-
--#define VDR_Q8_0_Q8_1_MMVQ 2
-+#define VDR_Q8_0_Q8_1_MMVQ 4
- #define VDR_Q8_0_Q8_1_MMQ 8
-
- template static __device__ __forceinline__ T vec_dot_q8_0_q8_1_impl(
diff --git a/ggml-patches/0007-magpietts-nanocodec.patch b/ggml-patches/0007-magpietts-nanocodec.patch
deleted file mode 100644
index b43bea0..0000000
--- a/ggml-patches/0007-magpietts-nanocodec.patch
+++ /dev/null
@@ -1,732 +0,0 @@
-diff --git a/src/ggml-cuda/CMakeLists.txt b/src/ggml-cuda/CMakeLists.txt
-index b54d4a6b..07a94052 100644
---- a/src/ggml-cuda/CMakeLists.txt
-+++ b/src/ggml-cuda/CMakeLists.txt
-@@ -15,7 +15,8 @@ if (CUDAToolkit_FOUND)
- # 80 == Ampere, asynchronous data loading, faster tensor core instructions
- # 86 == RTX 3000, needs CUDA v11.1
- # 89 == RTX 4000, needs CUDA v11.8
-- # 120 == Blackwell, needs CUDA v12.8, FP4 tensor cores
-+ # 110 == Jetson Thor, needs CUDA v13.0, Blackwell tensor cores
-+ # 120 == Blackwell, needs CUDA v12.8, SM120 FP4 tensor cores
- #
- # XX-virtual == compile CUDA code as PTX, do JIT compilation to binary code on first run
- # XX-real == compile CUDA code as device code for this specific architecture
-@@ -36,6 +37,10 @@ if (CUDAToolkit_FOUND)
- list(APPEND CMAKE_CUDA_ARCHITECTURES 89-real)
- endif()
-
-+ if (CUDAToolkit_VERSION VERSION_GREATER_EQUAL "13.0")
-+ list(APPEND CMAKE_CUDA_ARCHITECTURES 110a-real)
-+ endif()
-+
- if (CUDAToolkit_VERSION VERSION_GREATER_EQUAL "12.8")
- # The CUDA architecture 120f-virtual would in principle work for Blackwell support
- # but the newly added "f" suffix conflicted with a preexising regex for validating CUDA architectures in CMake.
-@@ -71,16 +76,16 @@ if (CUDAToolkit_FOUND)
- FetchContent_MakeAvailable(CCCL)
- endif()
-
-- # Replace any plain 12X CUDA architectures with their "architecture-specific" equivalents 12Xa.
-- # 12X is forwards-compatible, 12Xa is not.
-- # Notably the Blackwell FP4 tensor core instructions are not forwards compatible and therefore need 12Xa.
-+ # Replace plain Blackwell CUDA architectures with their "architecture-specific" equivalents.
-+ # 11X/12X are forwards-compatible, 11Xa/12Xa are not.
-+ # Notably the Blackwell tensor core instructions are not forwards compatible and therefore need architecture-specific targets.
- # But while 12X vs. 12Xa can be checked in device code there is (to my knowledge) no easy way to do the same check in host code.
-- # So for now just replace all instances of 12X with 12Xa, this should be fine until Rubin is released.
-+ # So for now just replace the supported plain Blackwell targets with architecture-specific targets.
- foreach(ARCHS IN ITEMS CMAKE_CUDA_ARCHITECTURES CMAKE_CUDA_ARCHITECTURES_NATIVE)
- set(FIXED_ARCHS "")
- foreach(ARCH IN LISTS ${ARCHS})
-- if (ARCH MATCHES "^12[0-9](-real|-virtual)?$")
-- string(REGEX REPLACE "^(12[0-9])((-real|-virtual)?)$" "\\1a\\2" FIXED_ARCH ${ARCH})
-+ if (ARCH MATCHES "^(110|12[0-9])(-real|-virtual)?$")
-+ string(REGEX REPLACE "^(110|12[0-9])((-real|-virtual)?)$" "\\1a\\2" FIXED_ARCH ${ARCH})
- message(STATUS "Replacing ${ARCH} in ${ARCHS} with ${FIXED_ARCH}")
- list(APPEND FIXED_ARCHS "${FIXED_ARCH}")
- else()
-@@ -90,8 +95,8 @@ if (CUDAToolkit_FOUND)
- set(${ARCHS} ${FIXED_ARCHS})
- endforeach()
-
-- # If we try to compile a "native" build it will use the 12X architectures and fail.
-- # So we should instead use the native architectures as determined by CMake after replacing 12X with 12Xa.
-+ # If we try to compile a "native" build it may use plain Blackwell architectures and fail.
-+ # So we should instead use the native architectures as determined by CMake after replacing them with architecture-specific forms.
- # But if at the time of the build no GPUs are connected at all CMAKE_CUDA_ARCHITECTURES will contain garbage that we should not use.
- if (CMAKE_CUDA_ARCHITECTURES STREQUAL "native" AND CMAKE_CUDA_ARCHITECTURES_NATIVE MATCHES "^[0-9]+(a|f)?(-real|-virtual)?(;[0-9]+(a|f)?(-real|-virtual)?|;)*$")
- set(CMAKE_CUDA_ARCHITECTURES ${CMAKE_CUDA_ARCHITECTURES_NATIVE})
-diff --git a/src/ggml-cuda/common.cuh b/src/ggml-cuda/common.cuh
-index 9bd40faa..56fe1bb7 100644
---- a/src/ggml-cuda/common.cuh
-+++ b/src/ggml-cuda/common.cuh
-@@ -25,6 +25,7 @@
- #include
- #include
- #include
-+#include
- #include
- #include
- #include
-@@ -50,7 +51,10 @@
- #define GGML_CUDA_CC_TURING 750
- #define GGML_CUDA_CC_AMPERE 800
- #define GGML_CUDA_CC_ADA_LOVELACE 890
--// While BW spans CC 1000, 1100 & 1200, we are integrating Tensor Core instructions available to 1200 family, see
-+// Jetson Thor is Blackwell SM110. The hand-written FP4 block-scale PTX below is currently SM120-only, but SM110
-+// should still be detected so ggml can favor CUDA library kernels that may use Thor tcgen05 tensor cores.
-+#define GGML_CUDA_CC_THOR 1100
-+// While BW spans CC 1000, 1100 & 1200, the hand-written FP4 path integrates Tensor Core instructions available to 1200 family, see
- // https://docs.nvidia.com/cutlass/media/docs/cpp/blackwell_functionality.html#blackwell-sm120-gemms
- #define GGML_CUDA_CC_BLACKWELL 1200
- #define GGML_CUDA_CC_DGX_SPARK 1210
-@@ -315,6 +319,10 @@ static bool amd_wmma_available(const int cc) {
- return (GGML_CUDA_CC_IS_RDNA4(cc) || GGML_CUDA_CC_IS_RDNA3(cc));
- }
-
-+static bool thor_mma_available(const int cc) {
-+ return GGML_CUDA_CC_IS_NVIDIA(cc) && cc == GGML_CUDA_CC_THOR;
-+}
-+
- static bool volta_mma_available(const int cc) {
- return GGML_CUDA_CC_IS_NVIDIA(cc) && ggml_cuda_highest_compiled_arch(cc) == GGML_CUDA_CC_VOLTA;
- }
-@@ -1379,14 +1387,34 @@ struct ggml_backend_cuda_context {
-
- int64_t last_graph_eviction_sweep = 0;
-
-+ static int64_t cuda_graph_env_ms_to_us(const char * name, int64_t default_ms) {
-+ const char * value = getenv(name);
-+ if (value == nullptr || value[0] == '\0') {
-+ return default_ms * 1000;
-+ }
-+
-+ char * end = nullptr;
-+ const long long parsed_ms = std::strtoll(value, &end, 10);
-+ if (end == value || *end != '\0' || parsed_ms < 0) {
-+ return default_ms * 1000;
-+ }
-+ return (int64_t) parsed_ms * 1000;
-+ }
-+
- ggml_cuda_graph * cuda_graph(const void * first_node_ptr) {
- const int64_t time_now = ggml_time_us();
--
-- // sweep every 5s, evicting cuda graphs unused for >=10s
-- if (time_now - last_graph_eviction_sweep >= 5'000'000) {
-+ static const int64_t sweep_interval_us =
-+ cuda_graph_env_ms_to_us("GGML_CUDA_GRAPH_SWEEP_MS", 5000);
-+ static const int64_t evict_after_us =
-+ cuda_graph_env_ms_to_us("GGML_CUDA_GRAPH_EVICT_AFTER_MS", 10000);
-+
-+ // By default sweep every 5s, evicting CUDA graphs unused for >=10s.
-+ // Set GGML_CUDA_GRAPH_EVICT_AFTER_MS=0 to keep captured graphs resident.
-+ if (evict_after_us > 0 && sweep_interval_us > 0
-+ && time_now - last_graph_eviction_sweep >= sweep_interval_us) {
- last_graph_eviction_sweep = time_now;
- for (auto it = cuda_graphs.begin(); it != cuda_graphs.end(); ) {
-- if (time_now - it->second->last_used_time >= 10'000'000) {
-+ if (time_now - it->second->last_used_time >= evict_after_us) {
- it = cuda_graphs.erase(it);
- } else {
- ++it;
-diff --git a/src/ggml-cuda/conv-transpose-1d.cu b/src/ggml-cuda/conv-transpose-1d.cu
-index 8418ba66..20a84516 100644
---- a/src/ggml-cuda/conv-transpose-1d.cu
-+++ b/src/ggml-cuda/conv-transpose-1d.cu
-@@ -1,34 +1,38 @@
- #include "conv-transpose-1d.cuh"
-+#include "convert.cuh"
-
--static __global__ void conv_transpose_1d_kernel(
-+#include
-+
-+template
-+static __global__ void conv_transpose_1d_kernel(
- const int s0, const int p0, const int d0, const int output_size,
- const int src0_ne0, const int src0_ne1, const int src0_ne2, const int src0_ne3,
- const int src1_ne0, const int src1_ne1, const int src1_ne2, const int src1_ne3,
- const int dst_ne0, const int dst_ne1, const int dst_ne2, const int dst_ne3,
-- const float * src0, const float * src1, float * dst) {
-+ const T * src0, const float * src1, float * dst) {
- int global_index = threadIdx.x + blockIdx.x * blockDim.x;
- if (global_index >= output_size) {
- return;
- }
-
-- int out_index = global_index / dst_ne0;
-+ const int out_t = global_index % dst_ne0;
-+ const int out_c = (global_index / dst_ne0) % dst_ne1;
-+ const int out_b = global_index / (dst_ne0 * dst_ne1);
-
- float accumulator = 0;
-
-- for (int c = 0; c < src0_ne2; c++) {
-- int idx = global_index % dst_ne0;
-+ const int in_end = min(src1_ne0 - 1, out_t / s0);
-+ const int in_start = max(0, (out_t - src0_ne0 + s0) / s0);
-
-- int kernel_offset = (src0_ne0 * src0_ne1 * c) + (out_index * src0_ne0);
-- int input_offset = src1_ne0 * c;
-+ for (int c = 0; c < src0_ne2; c++) {
-+ const int kernel_offset = src0_ne0 * (out_c + src0_ne1 * c);
-+ const int input_offset = src1_ne0 * (c + src1_ne1 * out_b);
-
-- for (int i = 0; i < src1_ne0; i++) {
-- if (!(idx >= i*s0 && idx < i*s0 + src0_ne0)) {
-- continue;
-- }
-- int weight_idx = idx - i*s0;
-+ for (int i = in_start; i <= in_end; i++) {
-+ const int weight_idx = out_t - i*s0;
-
-- float kernel_weight = src0[kernel_offset + weight_idx];
-- float input_value = src1[input_offset+i];
-+ const float kernel_weight = ggml_cuda_cast(src0[kernel_offset + weight_idx]);
-+ const float input_value = src1[input_offset+i];
-
- accumulator += kernel_weight * input_value;
- }
-@@ -37,26 +41,96 @@ static __global__ void conv_transpose_1d_kernel(
- GGML_UNUSED_VARS(p0, d0, src0_ne3, src1_ne3, dst_ne3, src1_ne1, dst_ne1, src1_ne2, dst_ne2);
- }
-
--static void conv_transpose_1d_f32_f32_cuda(
-+template
-+static __global__ void conv_transpose_1d_grouped2_kernel(
- const int s0, const int p0, const int d0, const int output_size,
- const int src0_ne0, const int src0_ne1, const int src0_ne2, const int src0_ne3,
- const int src1_ne0, const int src1_ne1, const int src1_ne2, const int src1_ne3,
- const int dst_ne0, const int dst_ne1, const int dst_ne2, const int dst_ne3,
-- const float * src0, const float * src1, float * dst,
-+ const T * src0, const float * src1, float * dst) {
-+ int global_index = threadIdx.x + blockIdx.x * blockDim.x;
-+ if (global_index >= output_size) {
-+ return;
-+ }
-+
-+ const int out_t = global_index % dst_ne0;
-+ const int out_c = (global_index / dst_ne0) % dst_ne1;
-+ const int out_b = global_index / (dst_ne0 * dst_ne1);
-+
-+ const int in_end = min(src1_ne0 - 1, out_t / s0);
-+ const int in_start = max(0, (out_t - src0_ne0 + s0) / s0);
-+
-+ float accumulator = 0;
-+
-+ const int c0 = out_c * 2;
-+ const int c1 = c0 + 1;
-+
-+ for (int i = in_start; i <= in_end; i++) {
-+ const int weight_idx = out_t - i*s0;
-+
-+ const int input_offset0 = src1_ne0 * (c0 + src1_ne1 * out_b);
-+ const int kernel_offset0 = src0_ne0 * (out_c + src0_ne1 * c0);
-+ accumulator += ggml_cuda_cast(src0[kernel_offset0 + weight_idx]) * src1[input_offset0 + i];
-+
-+ const int input_offset1 = src1_ne0 * (c1 + src1_ne1 * out_b);
-+ const int kernel_offset1 = src0_ne0 * (out_c + src0_ne1 * c1);
-+ accumulator += ggml_cuda_cast(src0[kernel_offset1 + weight_idx]) * src1[input_offset1 + i];
-+ }
-+
-+ dst[global_index] = accumulator;
-+ GGML_UNUSED_VARS(p0, d0, src0_ne2, src0_ne3, src1_ne2, src1_ne3, dst_ne2, dst_ne3);
-+}
-+
-+template
-+static void conv_transpose_1d_cuda(
-+ const int s0, const int p0, const int d0, const int output_size,
-+ const int src0_ne0, const int src0_ne1, const int src0_ne2, const int src0_ne3,
-+ const int src1_ne0, const int src1_ne1, const int src1_ne2, const int src1_ne3,
-+ const int dst_ne0, const int dst_ne1, const int dst_ne2, const int dst_ne3,
-+ const T * src0, const float * src1, float * dst,
-+ const bool use_grouped2,
- cudaStream_t stream) {
-
- const int num_blocks = (output_size + CUDA_CONV_TRANPOSE_1D_BLOCK_SIZE - 1) / CUDA_CONV_TRANPOSE_1D_BLOCK_SIZE;
-- conv_transpose_1d_kernel<<>>(
-- s0,p0,d0,output_size,
-- src0_ne0, src0_ne1, src0_ne2, src0_ne3,
-- src1_ne0, src1_ne1, src1_ne2, src1_ne3,
-- dst_ne0, dst_ne1, dst_ne2, dst_ne3,
-- src0,src1, dst);
-+ if (use_grouped2) {
-+ conv_transpose_1d_grouped2_kernel<<>>(
-+ s0,p0,d0,output_size,
-+ src0_ne0, src0_ne1, src0_ne2, src0_ne3,
-+ src1_ne0, src1_ne1, src1_ne2, src1_ne3,
-+ dst_ne0, dst_ne1, dst_ne2, dst_ne3,
-+ src0,src1, dst);
-+ } else {
-+ conv_transpose_1d_kernel<<>>(
-+ s0,p0,d0,output_size,
-+ src0_ne0, src0_ne1, src0_ne2, src0_ne3,
-+ src1_ne0, src1_ne1, src1_ne2, src1_ne3,
-+ dst_ne0, dst_ne1, dst_ne2, dst_ne3,
-+ src0,src1, dst);
-+ }
-+}
-+
-+static bool conv_transpose_1d_use_nanocodec_grouped2(
-+ const ggml_tensor * src0, const ggml_tensor * src1, const ggml_tensor * dst, const int s0) {
-+ const bool shape_matches =
-+ src0->ne[0] == 2*s0 &&
-+ src0->ne[2] == 2*src0->ne[1] &&
-+ src1->ne[1] == src0->ne[2] &&
-+ dst->ne[1] == src0->ne[1] &&
-+ src1->ne[2] == dst->ne[2] &&
-+ src1->ne[3] == dst->ne[3];
-+
-+ const char * name = src0->name;
-+ const bool is_nanocodec_up_weight =
-+ name != nullptr &&
-+ std::strncmp(name, "dec.up.", 7) == 0 &&
-+ std::strstr(name + 7, ".w") != nullptr;
-+
-+ return shape_matches && is_nanocodec_up_weight;
- }
-
- void ggml_cuda_op_conv_transpose_1d(ggml_backend_cuda_context & ctx, ggml_tensor * dst) {
- const ggml_tensor * src0 = dst->src[0];
-- const float * src0_d = (const float *)src0->data;
-+ const void * src0_d = src0->data;
-
- const ggml_tensor * src1 = dst->src[1];
- const float * src1_d = (const float *)src1->data;
-@@ -64,7 +138,8 @@ void ggml_cuda_op_conv_transpose_1d(ggml_backend_cuda_context & ctx, ggml_tensor
- float * dst_d = (float *)dst->data;
- cudaStream_t stream = ctx.stream();
-
-- GGML_ASSERT(src0->type == GGML_TYPE_F32);
-+ GGML_ASSERT(src0->type == GGML_TYPE_F32 || src0->type == GGML_TYPE_F16);
-+ GGML_ASSERT(src1->type == GGML_TYPE_F32);
- GGML_ASSERT( dst->type == GGML_TYPE_F32);
-
- GGML_ASSERT(ggml_is_contiguous(src0));
-@@ -77,10 +152,19 @@ void ggml_cuda_op_conv_transpose_1d(ggml_backend_cuda_context & ctx, ggml_tensor
- const int d0 = 1;//opts[4];
-
- const int64_t output_size = ggml_nelements(dst);
--
-- conv_transpose_1d_f32_f32_cuda(s0, p0, d0, output_size,
-- src0->ne[0], src0->ne[1], src0->ne[2], src0->ne[3],
-- src1->ne[0], src1->ne[1], src1->ne[2], src1->ne[3],
-- dst->ne[0], dst->ne[1], dst->ne[2], dst->ne[3],
-- src0_d, src1_d, dst_d, stream);
-+ const bool use_grouped2 = conv_transpose_1d_use_nanocodec_grouped2(src0, src1, dst, s0);
-+
-+ if (src0->type == GGML_TYPE_F16) {
-+ conv_transpose_1d_cuda(s0, p0, d0, output_size,
-+ src0->ne[0], src0->ne[1], src0->ne[2], src0->ne[3],
-+ src1->ne[0], src1->ne[1], src1->ne[2], src1->ne[3],
-+ dst->ne[0], dst->ne[1], dst->ne[2], dst->ne[3],
-+ (const half *) src0_d, src1_d, dst_d, use_grouped2, stream);
-+ } else {
-+ conv_transpose_1d_cuda(s0, p0, d0, output_size,
-+ src0->ne[0], src0->ne[1], src0->ne[2], src0->ne[3],
-+ src1->ne[0], src1->ne[1], src1->ne[2], src1->ne[3],
-+ dst->ne[0], dst->ne[1], dst->ne[2], dst->ne[3],
-+ (const float *) src0_d, src1_d, dst_d, use_grouped2, stream);
-+ }
- }
-diff --git a/src/ggml-cuda/ggml-cuda.cu b/src/ggml-cuda/ggml-cuda.cu
-index a531ce07..4b4a8488 100644
---- a/src/ggml-cuda/ggml-cuda.cu
-+++ b/src/ggml-cuda/ggml-cuda.cu
-@@ -2485,8 +2485,8 @@ static bool ggml_cuda_should_fuse_mul_mat_vec_f(const ggml_tensor * tensor) {
- return false;
- }
-
-- //we only support fusion for ncols_dst = 1
-- if (tensor->op == GGML_OP_MUL_MAT && dst->ne[1] != 1) {
-+ // MMVF supports a two-column epilogue so paired CFG lanes can share each weight-row load.
-+ if (tensor->op == GGML_OP_MUL_MAT && dst->ne[1] > 2) {
- return false;
- }
-
-@@ -3310,6 +3310,23 @@ static const void * ggml_cuda_graph_get_key(ggml_cgraph * cgraph) {
- return cgraph->nodes[0];
- }
-
-+static const char * ggml_cuda_graph_tensor_name(const ggml_tensor * tensor) {
-+ return tensor && tensor->name[0] ? tensor->name : "(unnamed)";
-+}
-+
-+static void ggml_cuda_graph_log_event(
-+ const char * func,
-+ const char * event,
-+ const ggml_cgraph * cgraph,
-+ const void * graph_key) {
-+ const int n_nodes = cgraph ? cgraph->n_nodes : 0;
-+ const ggml_tensor * first = n_nodes > 0 ? cgraph->nodes[0] : nullptr;
-+ const ggml_tensor * last = n_nodes > 0 ? cgraph->nodes[n_nodes - 1] : nullptr;
-+ GGML_LOG_DEBUG("%s: CUDA graph %s key=%p uid=%" PRIu64 " nodes=%d first=%s last=%s\n",
-+ func, event, graph_key, cgraph ? cgraph->uid : 0, n_nodes,
-+ ggml_cuda_graph_tensor_name(first), ggml_cuda_graph_tensor_name(last));
-+}
-+
- static bool ggml_cuda_graph_update_required(ggml_backend_cuda_context * cuda_ctx, ggml_cgraph * cgraph) {
- bool res = false;
-
-@@ -3318,7 +3335,6 @@ static bool ggml_cuda_graph_update_required(ggml_backend_cuda_context * cuda_ctx
-
- if (cgraph->uid != 0 &&
- cgraph->uid == graph->uid) {
-- GGML_LOG_DEBUG("CUDA Graph id %zu reused\n", cgraph->uid);
- GGML_ASSERT((int)graph->node_props.size() == cgraph->n_nodes);
- return false;
- }
-@@ -3891,6 +3907,89 @@ static bool ggml_cuda_can_fuse(const struct ggml_cgraph * cgraph,
- return false;
- }
-
-+static bool ggml_cuda_node_can_be_elided(const struct ggml_cgraph * cgraph, int node_idx, int32_t expected_uses) {
-+ const ggml_tensor * node = cgraph->nodes[node_idx];
-+ return (node->flags & GGML_TENSOR_FLAG_COMPUTE) != 0 &&
-+ (node->flags & GGML_TENSOR_FLAG_OUTPUT) == 0 &&
-+ ggml_node_get_use_count(cgraph, node_idx) == expected_uses;
-+}
-+
-+static int ggml_cuda_try_fuse_half_snake(ggml_backend_cuda_context * cuda_ctx, ggml_cgraph * cgraph, int i) {
-+ if (i <= 0 || i + 7 >= cgraph->n_nodes) {
-+ return 0;
-+ }
-+
-+ ggml_tensor * mul0 = cgraph->nodes[i + 0];
-+ ggml_tensor * sin = cgraph->nodes[i + 1];
-+ ggml_tensor * sqr = cgraph->nodes[i + 2];
-+ ggml_tensor * mul1 = cgraph->nodes[i + 3];
-+ ggml_tensor * add = cgraph->nodes[i + 4];
-+ ggml_tensor * view_l = cgraph->nodes[i + 5];
-+ ggml_tensor * lrelu = cgraph->nodes[i + 6];
-+ ggml_tensor * concat = cgraph->nodes[i + 7];
-+
-+ if (mul0->op != GGML_OP_MUL || sin->op != GGML_OP_SIN || sqr->op != GGML_OP_SQR ||
-+ mul1->op != GGML_OP_MUL || add->op != GGML_OP_ADD || view_l->op != GGML_OP_VIEW ||
-+ lrelu->op != GGML_OP_LEAKY_RELU || concat->op != GGML_OP_CONCAT) {
-+ return 0;
-+ }
-+
-+ if (!ggml_cuda_node_can_be_elided(cgraph, i + 0, 1) ||
-+ !ggml_cuda_node_can_be_elided(cgraph, i + 1, 1) ||
-+ !ggml_cuda_node_can_be_elided(cgraph, i + 2, 1) ||
-+ !ggml_cuda_node_can_be_elided(cgraph, i + 3, 1) ||
-+ !ggml_cuda_node_can_be_elided(cgraph, i + 4, 1) ||
-+ !ggml_cuda_node_can_be_elided(cgraph, i + 5, 1) ||
-+ !ggml_cuda_node_can_be_elided(cgraph, i + 6, 1)) {
-+ return 0;
-+ }
-+
-+ const ggml_tensor * x_snake = ggml_are_same_shape(mul0, mul0->src[0]) ? mul0->src[0] : mul0->src[1];
-+ const ggml_tensor * alpha = (x_snake == mul0->src[0]) ? mul0->src[1] : mul0->src[0];
-+ if (x_snake->op != GGML_OP_VIEW || x_snake != cgraph->nodes[i - 1]) {
-+ return 0;
-+ }
-+ if (!ggml_cuda_node_can_be_elided(cgraph, i - 1, 2)) {
-+ return 0;
-+ }
-+
-+ const ggml_tensor * inv_b = (mul1->src[0] == sqr) ? mul1->src[1] : mul1->src[0];
-+ const ggml_tensor * x_in_add = (add->src[0] == mul1) ? add->src[1] : add->src[0];
-+ if (sin->src[0] != mul0 || sqr->src[0] != sin || (mul1->src[0] != sqr && mul1->src[1] != sqr) ||
-+ x_in_add != x_snake || lrelu->src[0] != view_l || concat->src[0] != add || concat->src[1] != lrelu) {
-+ return 0;
-+ }
-+
-+ const int32_t concat_dim = ((const int32_t *) concat->op_params)[0];
-+ const bool type_ok = (x_snake->type == GGML_TYPE_F32 || x_snake->type == GGML_TYPE_F16 || x_snake->type == GGML_TYPE_BF16) &&
-+ x_snake->type == view_l->type && x_snake->type == concat->type;
-+ const bool shape_ok = x_snake->view_src == view_l->view_src &&
-+ concat_dim == 1 &&
-+ x_snake->ne[0] == view_l->ne[0] &&
-+ x_snake->ne[2] == view_l->ne[2] &&
-+ x_snake->ne[3] == view_l->ne[3] &&
-+ concat->ne[0] == x_snake->ne[0] &&
-+ concat->ne[1] == x_snake->ne[1] + view_l->ne[1] &&
-+ concat->ne[2] == x_snake->ne[2] &&
-+ concat->ne[3] == x_snake->ne[3] &&
-+ alpha->type == GGML_TYPE_F32 &&
-+ inv_b->type == GGML_TYPE_F32 &&
-+ ggml_is_contiguous(alpha) &&
-+ ggml_is_contiguous(inv_b) &&
-+ ggml_nelements(alpha) == x_snake->ne[1] &&
-+ ggml_nelements(inv_b) == x_snake->ne[1] &&
-+ ggml_is_contiguous(x_snake) &&
-+ ggml_is_contiguous(view_l) &&
-+ ggml_is_contiguous(concat);
-+
-+ if (!type_ok || !shape_ok) {
-+ return 0;
-+ }
-+
-+ ggml_cuda_op_half_snake_fused(*cuda_ctx, x_snake, view_l, alpha, inv_b, lrelu, concat);
-+ return 7;
-+}
-+
- // try and fuse nodes and return the number of nodes to skip
- static int ggml_cuda_try_fuse(ggml_backend_cuda_context * cuda_ctx, ggml_cgraph * cgraph, int i) {
-
-@@ -3901,6 +3999,10 @@ static int ggml_cuda_try_fuse(ggml_backend_cuda_context * cuda_ctx, ggml_cgraph
-
- ggml_tensor * node = cgraph->nodes[i];
-
-+ if (int n_fused = ggml_cuda_try_fuse_half_snake(cuda_ctx, cgraph, i)) {
-+ return n_fused;
-+ }
-+
- //topk-moe
- if (cgraph->nodes[i]->op == GGML_OP_UNARY || cgraph->nodes[i]->op == GGML_OP_SOFT_MAX ||
- cgraph->nodes[i]->op == GGML_OP_ARGSORT) {
-@@ -4623,7 +4725,7 @@ static enum ggml_status ggml_backend_cuda_graph_compute(ggml_backend_t backend,
- // Warmup: need at least 2 calls with no property change on the 2nd call
- if (!properties_changed) {
- graph->warmup_complete = true;
-- GGML_LOG_DEBUG("%s: CUDA graph warmup complete\n", __func__);
-+ ggml_cuda_graph_log_event(__func__, "warmup complete", cgraph, graph_key);
- use_cuda_graph = true;
- cuda_graph_update_required = true;
- }
-@@ -4633,7 +4735,7 @@ static enum ggml_status ggml_backend_cuda_graph_compute(ggml_backend_t backend,
- if (properties_changed) {
- // Properties changed - reset warmup, execute directly until stable again
- graph->warmup_complete = false;
-- GGML_LOG_DEBUG("%s: CUDA graph warmup reset\n", __func__);
-+ ggml_cuda_graph_log_event(__func__, "warmup reset", cgraph, graph_key);
- } else {
- use_cuda_graph = true;
- cuda_graph_update_required = graph->instance == nullptr;
-@@ -5434,7 +5536,7 @@ static bool ggml_backend_cuda_device_supports_op(ggml_backend_dev_t dev, const g
- {
- ggml_type src0_type = op->src[0]->type;
- ggml_type src1_type = op->src[1]->type;
-- if (src0_type == GGML_TYPE_F32 && src1_type == GGML_TYPE_F32) {
-+ if ((src0_type == GGML_TYPE_F32 || src0_type == GGML_TYPE_F16) && src1_type == GGML_TYPE_F32) {
- return true;
- }
- return false;
-@@ -5705,6 +5807,12 @@ static ggml_backend_feature * ggml_backend_cuda_get_features(ggml_backend_reg_t
-
- {
- const auto & info = ggml_cuda_info();
-+ for (int id = 0; id < info.device_count; ++id) {
-+ if (thor_mma_available(info.devices[id].cc)) {
-+ features.push_back({ "BLACKWELL_THOR_SM110", "1"});
-+ break;
-+ }
-+ }
- for (int id = 0; id < info.device_count; ++id) {
- if (blackwell_mma_available(info.devices[id].cc)) {
- features.push_back({ "BLACKWELL_NATIVE_FP4", "1"});
-diff --git a/src/ggml-cuda/mmf.cu b/src/ggml-cuda/mmf.cu
-index aad4c34a..628d8718 100644
---- a/src/ggml-cuda/mmf.cu
-+++ b/src/ggml-cuda/mmf.cu
-@@ -159,6 +159,11 @@ bool ggml_cuda_should_use_mmf(enum ggml_type type, int cc, int warp_size, const
- return false;
- }
-
-+ // Thor's tcgen05 paths are provided by CUDA libraries today; avoid ggml's warp-level matrix kernels there.
-+ if (thor_mma_available(cc) && (type == GGML_TYPE_F32 || type == GGML_TYPE_F16 || type == GGML_TYPE_BF16)) {
-+ return false;
-+ }
-+
- if (mul_mat_id) {
- if (src0_ne[1] <= 1024 && src1_ncols > 512) {
- return false;
-diff --git a/src/ggml-cuda/mmvf.cu b/src/ggml-cuda/mmvf.cu
-index d9147202..a347447c 100644
---- a/src/ggml-cuda/mmvf.cu
-+++ b/src/ggml-cuda/mmvf.cu
-@@ -383,7 +383,10 @@ static void mul_mat_vec_f_switch_fusion(
- const dim3 & block_dims, const dim3 & block_nums, const int nbytes_shared, const int ids_stride, const cudaStream_t stream) {
-
- const bool has_fusion = fusion.gate != nullptr || fusion.x_bias != nullptr || fusion.gate_bias != nullptr;
-- if constexpr (ncols_dst == 1) {
-+ // The same epilogue addressing works for two adjacent columns. This is especially useful
-+ // for classifier-free guidance: both lanes share the weight-row load and still fold the
-+ // residual/bias writeback into the projection kernel.
-+ if constexpr (ncols_dst <= 2) {
- if (has_fusion) {
- mul_mat_vec_f<<>>
- (x, y, ids, fusion, dst, ncols, nchannels_y, stride_row, stride_col_y, stride_col_dst,
-@@ -393,7 +396,7 @@ static void mul_mat_vec_f_switch_fusion(
- }
- }
-
-- GGML_ASSERT(!has_fusion && "fusion only supported for ncols_dst=1");
-+ GGML_ASSERT(!has_fusion && "fusion only supported for ncols_dst<=2");
-
- mul_mat_vec_f<<>>
- (x, y, ids, fusion, dst, ncols, nchannels_y, stride_row, stride_col_y, stride_col_dst,
-@@ -436,6 +439,11 @@ void launch_mul_mat_vec_f_cuda(
- block_size_best = block_size;
- }
- }
-+ if constexpr (std::is_same_v) {
-+ if (warp_size == 32 && ncols >= 768 && ncols_dst <= 2) {
-+ block_size_best = 96;
-+ }
-+ }
-
- const bool has_fusion = fusion.gate != nullptr || fusion.x_bias != nullptr || fusion.gate_bias != nullptr;
-
-@@ -650,7 +658,7 @@ void ggml_cuda_mul_mat_vec_f(ggml_backend_cuda_context & ctx, const ggml_tensor
-
- if (fusion) {
- GGML_ASSERT( !ids || dst->ne[2] == 1);
-- GGML_ASSERT( ids || dst->ne[1] == 1);
-+ GGML_ASSERT( ids || dst->ne[1] <= 2);
- if (fusion->x_bias) {
- GGML_ASSERT(fusion->x_bias->type == GGML_TYPE_F32);
- GGML_ASSERT(fusion->x_bias->ne[0] == dst->ne[0]);
-@@ -793,9 +801,15 @@ bool ggml_cuda_should_use_mmvf(enum ggml_type type, int cc, const int64_t * src0
- }
- }
-
-+ // Thor's tcgen05 paths are provided by CUDA libraries today; avoid ggml's warp-level vector kernels there.
-+ const bool prefer_cublas_tcgen05 = thor_mma_available(cc);
-+
- switch (type) {
- case GGML_TYPE_F32:
- if (GGML_CUDA_CC_IS_NVIDIA(cc)) {
-+ if (prefer_cublas_tcgen05) {
-+ return false;
-+ }
- if (ampere_mma_available(cc)) {
- return ne11 <= 3;
- }
-@@ -812,9 +826,12 @@ bool ggml_cuda_should_use_mmvf(enum ggml_type type, int cc, const int64_t * src0
- return ne11 <= 8;
- case GGML_TYPE_F16:
- if (GGML_CUDA_CC_IS_NVIDIA(cc)) {
-+ if (prefer_cublas_tcgen05) {
-+ return false;
-+ }
- const bool src0_small = (src0_ne[1] <= 512 || src0_ne[2]*src0_ne[3] == 1);
- if (ampere_mma_available(cc)) {
-- return src0_small && ne11 == 1;
-+ return src0_small && ne11 <= 2;
- }
- if (cc >= GGML_CUDA_CC_ADA_LOVELACE) {
- return src0_small && ne11 <= 4;
-@@ -838,6 +855,9 @@ bool ggml_cuda_should_use_mmvf(enum ggml_type type, int cc, const int64_t * src0
- return ne11 <= 8;
- case GGML_TYPE_BF16:
- if (GGML_CUDA_CC_IS_NVIDIA(cc)) {
-+ if (prefer_cublas_tcgen05) {
-+ return false;
-+ }
- const bool src0_small = (src0_ne[1] <= 512 || src0_ne[2]*src0_ne[3] == 1);
- if (ampere_mma_available(cc)) {
- return src0_small && ne11 == 1;
-diff --git a/src/ggml-cuda/snake.cu b/src/ggml-cuda/snake.cu
-index 384638c1..6f4b219c 100644
---- a/src/ggml-cuda/snake.cu
-+++ b/src/ggml-cuda/snake.cu
-@@ -1,6 +1,8 @@
- #include "snake.cuh"
- #include "convert.cuh"
-
-+#include
-+
- // Fused Snake activation: y = x + sin^2(a * x) * inv_b
- // x: [T, C] (T contiguous), a: [1, C], inv_b: [1, C]
- // Supports F32, F16, BF16 data with F32 compute.
-@@ -70,3 +72,80 @@ void ggml_cuda_op_snake_fused(ggml_backend_cuda_context & ctx,
- ggml_tensor * dst) {
- launch_snake(ctx, x, a, inv_b, dst);
- }
-+
-+template
-+static __global__ void half_snake_kernel(
-+ const T * __restrict__ x_snake,
-+ const T * __restrict__ x_lrelu,
-+ const float * __restrict__ a,
-+ const float * __restrict__ inv_b,
-+ T * __restrict__ dst,
-+ const int total,
-+ const int T_len,
-+ const int snake_channels,
-+ const int lrelu_channels,
-+ const int channels,
-+ const float negative_slope) {
-+ const int idx = blockIdx.x * blockDim.x + threadIdx.x;
-+ if (idx >= total) return;
-+
-+ const int t = idx % T_len;
-+ const int c = (idx / T_len) % channels;
-+ const int plane = idx / (T_len * channels);
-+
-+ if (c < snake_channels) {
-+ const int src_idx = plane * T_len * snake_channels + c * T_len + t;
-+ const float xi = ggml_cuda_cast(x_snake[src_idx]);
-+ const float s = sinf(a[c] * xi);
-+ dst[idx] = ggml_cuda_cast(xi + s * s * inv_b[c]);
-+ } else {
-+ const int lc = c - snake_channels;
-+ const int src_idx = plane * T_len * lrelu_channels + lc * T_len + t;
-+ const float xi = ggml_cuda_cast(x_lrelu[src_idx]);
-+ dst[idx] = ggml_cuda_cast(fmaxf(xi, 0.0f) + fminf(xi, 0.0f) * negative_slope);
-+ }
-+}
-+
-+void ggml_cuda_op_half_snake_fused(ggml_backend_cuda_context & ctx,
-+ const ggml_tensor * x_snake,
-+ const ggml_tensor * x_lrelu,
-+ const ggml_tensor * a,
-+ const ggml_tensor * inv_b,
-+ const ggml_tensor * lrelu,
-+ ggml_tensor * dst) {
-+ float negative_slope;
-+ std::memcpy(&negative_slope, lrelu->op_params, sizeof(float));
-+
-+ const int T = (int) dst->ne[0];
-+ const int snake_channels = (int) x_snake->ne[1];
-+ const int lrelu_channels = (int) x_lrelu->ne[1];
-+ const int channels = (int) dst->ne[1];
-+ const int total = (int) ggml_nelements(dst);
-+
-+ const int block_size = 256;
-+ const int grid_size = (total + block_size - 1) / block_size;
-+ cudaStream_t stream = ctx.stream();
-+
-+ const float * a_d = (const float *) a->data;
-+ const float * inv_b_d = (const float *) inv_b->data;
-+
-+ switch (dst->type) {
-+ case GGML_TYPE_F32: {
-+ half_snake_kernel<<>>(
-+ (const float *) x_snake->data, (const float *) x_lrelu->data, a_d, inv_b_d, (float *) dst->data,
-+ total, T, snake_channels, lrelu_channels, channels, negative_slope);
-+ } break;
-+ case GGML_TYPE_F16: {
-+ half_snake_kernel<<>>(
-+ (const half *) x_snake->data, (const half *) x_lrelu->data, a_d, inv_b_d, (half *) dst->data,
-+ total, T, snake_channels, lrelu_channels, channels, negative_slope);
-+ } break;
-+ case GGML_TYPE_BF16: {
-+ half_snake_kernel<<>>(
-+ (const nv_bfloat16 *) x_snake->data, (const nv_bfloat16 *) x_lrelu->data, a_d, inv_b_d,
-+ (nv_bfloat16 *) dst->data, total, T, snake_channels, lrelu_channels, channels, negative_slope);
-+ } break;
-+ default:
-+ GGML_ABORT("half_snake: unsupported type");
-+ }
-+}
-diff --git a/src/ggml-cuda/snake.cuh b/src/ggml-cuda/snake.cuh
-index 7f6f1cb3..7a3e7aa6 100644
---- a/src/ggml-cuda/snake.cuh
-+++ b/src/ggml-cuda/snake.cuh
-@@ -6,3 +6,11 @@ void ggml_cuda_op_snake_fused(ggml_backend_cuda_context & ctx,
- const ggml_tensor * a,
- const ggml_tensor * inv_b,
- ggml_tensor * dst);
-+
-+void ggml_cuda_op_half_snake_fused(ggml_backend_cuda_context & ctx,
-+ const ggml_tensor * x_snake,
-+ const ggml_tensor * x_lrelu,
-+ const ggml_tensor * a,
-+ const ggml_tensor * inv_b,
-+ const ggml_tensor * lrelu,
-+ ggml_tensor * dst);
diff --git a/ggml-patches/0008-cublas-bf16-projections.patch b/ggml-patches/0008-cublas-bf16-projections.patch
deleted file mode 100644
index c31c810..0000000
--- a/ggml-patches/0008-cublas-bf16-projections.patch
+++ /dev/null
@@ -1,326 +0,0 @@
-diff --git a/src/ggml-cuda/ggml-cuda.cu b/src/ggml-cuda/ggml-cuda.cu
-index 58a2330d..1a1b58fb 100644
---- a/src/ggml-cuda/ggml-cuda.cu
-+++ b/src/ggml-cuda/ggml-cuda.cu
-@@ -2187,8 +2187,65 @@ struct batched_mul_mat_traits {
- static inline auto get_nc_converter(ggml_type src_type) { return ggml_get_to_fp16_nc_cuda(src_type); }
- };
-
-+static __global__ void bf16_to_f32_add_row_bias(
-+ const nv_bfloat16 * __restrict__ src, const float * __restrict__ bias,
-+ float * __restrict__ dst, int64_t rows) {
-+#if defined(__CUDA_ARCH__) && __CUDA_ARCH__ >= GGML_CUDA_CC_AMPERE
-+ const int64_t row = int64_t(blockIdx.x) * blockDim.x + threadIdx.x;
-+ if (row >= rows) {
-+ return;
-+ }
-+ const int64_t i = int64_t(blockIdx.y) * rows + row;
-+ dst[i] = __bfloat162float(src[i]) + bias[row];
-+#else
-+ GGML_UNUSED_VARS(src, bias, dst, rows);
-+ NO_DEVICE_CODE;
-+#endif
-+}
-+
-+static void launch_bf16_to_f32_add_row_bias(
-+ const nv_bfloat16 * src, const float * bias, float * dst,
-+ int64_t rows, int64_t cols, cudaStream_t stream) {
-+ constexpr int block_size = 256;
-+ GGML_ASSERT(cols <= 65535);
-+ const dim3 block(block_size, 1, 1);
-+ const dim3 grid((rows + block_size - 1) / block_size, cols, 1);
-+ bf16_to_f32_add_row_bias<<>>(src, bias, dst, rows);
-+}
-+
-+static __global__ void bf16_add_row_bias_silu_to_bf16(
-+ const nv_bfloat16 * __restrict__ src, const float * __restrict__ bias,
-+ nv_bfloat16 * __restrict__ dst, int64_t rows) {
-+#if defined(__CUDA_ARCH__) && __CUDA_ARCH__ >= GGML_CUDA_CC_AMPERE
-+ const int64_t row = int64_t(blockIdx.x) * blockDim.x + threadIdx.x;
-+ if (row >= rows) {
-+ return;
-+ }
-+ const int64_t i = int64_t(blockIdx.y) * rows + row;
-+ const float x = __bfloat162float(src[i]) + bias[row];
-+ dst[i] = __float2bfloat16(ggml_cuda_op_silu_single(x));
-+#else
-+ GGML_UNUSED_VARS(src, bias, dst, rows);
-+ NO_DEVICE_CODE;
-+#endif
-+}
-+
-+static void launch_bf16_add_row_bias_silu_to_bf16(
-+ const nv_bfloat16 * src, const float * bias, nv_bfloat16 * dst,
-+ int64_t rows, int64_t cols, cudaStream_t stream) {
-+ constexpr int block_size = 256;
-+ GGML_ASSERT(cols <= 65535);
-+ const dim3 block(block_size, 1, 1);
-+ const dim3 grid((rows + block_size - 1) / block_size, cols, 1);
-+ bf16_add_row_bias_silu_to_bf16<<>>(src, bias, dst, rows);
-+}
-+
- template
--static void ggml_cuda_mul_mat_batched_cublas_impl(ggml_backend_cuda_context & ctx, const ggml_tensor * src0, const ggml_tensor * src1, ggml_tensor * dst) {
-+static void ggml_cuda_mul_mat_batched_cublas_impl(
-+ ggml_backend_cuda_context & ctx, const ggml_tensor * src0,
-+ const ggml_tensor * src1, ggml_tensor * dst,
-+ const ggml_tensor * row_bias = nullptr,
-+ ggml_tensor * bf16_silu_dst = nullptr) {
- using traits = batched_mul_mat_traits;
- using cuda_t = typename traits::cuda_type;
-
-@@ -2296,7 +2353,27 @@ static void ggml_cuda_mul_mat_batched_cublas_impl(ggml_backend_cuda_context & ct
- const int64_t r2 = ne12/ne02;
- const int64_t r3 = ne13/ne03;
-
-- if (r2 == 1 && r3 == 1 && is_src0_cont_2 && is_src1_cont_2) {
-+ // A single shared weight matrix broadcast over contiguous outer activation
-+ // dimensions is one large GEMM, not a collection of independent skinny
-+ // GEMMs. [K,T,B,D] and [K,T*B*D] have identical storage order; likewise
-+ // for the [M,T,B,D] output. Flattening only the cuBLAS geometry therefore
-+ // preserves the graph-visible tensor layout while improving weight reuse
-+ // and tensor-core occupancy and avoiding pointer-array setup.
-+ const bool src1_outer_contiguous =
-+ src1->type != src0_type || ggml_is_contiguous(src1);
-+ const bool flatten_shared_weight =
-+ ne02 == 1 && ne03 == 1 && ne12 * ne13 > 1 && is_src0_cont_2 &&
-+ is_src1_cont_2 && src1_outer_contiguous;
-+ if (flatten_shared_weight) {
-+ CUBLAS_CHECK(
-+ cublasGemmEx(ctx.cublas_handle(), CUBLAS_OP_T, CUBLAS_OP_N,
-+ ne01, ne11*ne12*ne13, ne10,
-+ alpha, src0_ptr, cu_data_type_a, nb01/nb00,
-+ src1_ptr, cu_data_type_b, s11,
-+ beta, dst_t, cu_data_type, ne0,
-+ cu_compute_type,
-+ CUBLAS_GEMM_DEFAULT_TENSOR_OP));
-+ } else if (r2 == 1 && r3 == 1 && is_src0_cont_2 && is_src1_cont_2) {
- // with a [0, 2, 1, 3] perm. and ne02==1 the matrix strides need to be determined from dim 3:
- const int64_t sma = ne02 == 1 ? nb03/nb00 : nb02/nb00;
- const int64_t smb = ne12 == 1 ? s13 : s12;
-@@ -2355,8 +2432,31 @@ static void ggml_cuda_mul_mat_batched_cublas_impl(ggml_backend_cuda_context & ct
-
- // Convert output back to F32 if needed
- if (dst->op_params[0] == GGML_PREC_DEFAULT && cu_data_type != CUDA_R_32F) {
-- const to_fp32_cuda_t to_fp32_cuda = ggml_get_to_fp32_cuda(traits::ggml_type_val);
-- to_fp32_cuda(dst_temp.get(), dst_ddf, ne_dst, main_stream);
-+ if (row_bias != nullptr) {
-+ if constexpr (src0_type == GGML_TYPE_BF16) {
-+ GGML_ASSERT(row_bias->type == GGML_TYPE_F32);
-+ GGML_ASSERT(ggml_is_contiguous(row_bias));
-+ GGML_ASSERT(ggml_nelements(row_bias) == ne0);
-+ if (bf16_silu_dst != nullptr) {
-+ GGML_ASSERT(bf16_silu_dst->type == GGML_TYPE_BF16);
-+ GGML_ASSERT(ggml_is_contiguous(bf16_silu_dst));
-+ GGML_ASSERT(ggml_are_same_shape(dst, bf16_silu_dst));
-+ launch_bf16_add_row_bias_silu_to_bf16(
-+ dst_temp.get(), static_cast(row_bias->data),
-+ static_cast(bf16_silu_dst->data),
-+ ne0, ne_dst / ne0, main_stream);
-+ } else {
-+ launch_bf16_to_f32_add_row_bias(
-+ dst_temp.get(), static_cast(row_bias->data), dst_ddf,
-+ ne0, ne_dst / ne0, main_stream);
-+ }
-+ } else {
-+ GGML_ABORT("row-bias conversion fusion is BF16-only");
-+ }
-+ } else {
-+ const to_fp32_cuda_t to_fp32_cuda = ggml_get_to_fp32_cuda(traits::ggml_type_val);
-+ to_fp32_cuda(dst_temp.get(), dst_ddf, ne_dst, main_stream);
-+ }
- }
- }
-
-@@ -2378,6 +2478,23 @@ static void ggml_cuda_mul_mat_batched_cublas(ggml_backend_cuda_context & ctx, co
- }
- }
-
-+static void ggml_cuda_mul_mat_bf16_row_bias(
-+ ggml_backend_cuda_context & ctx, const ggml_tensor * src0,
-+ const ggml_tensor * src1, const ggml_tensor * row_bias, ggml_tensor * dst) {
-+ GGML_ASSERT(src0->type == GGML_TYPE_BF16);
-+ ggml_cuda_mul_mat_batched_cublas_impl(
-+ ctx, src0, src1, dst, row_bias);
-+}
-+
-+static void ggml_cuda_mul_mat_bf16_row_bias_silu(
-+ ggml_backend_cuda_context & ctx, const ggml_tensor * src0,
-+ const ggml_tensor * src1, const ggml_tensor * row_bias,
-+ ggml_tensor * dst, ggml_tensor * bf16_silu_dst) {
-+ GGML_ASSERT(src0->type == GGML_TYPE_BF16);
-+ ggml_cuda_mul_mat_batched_cublas_impl(
-+ ctx, src0, src1, dst, row_bias, bf16_silu_dst);
-+}
-+
- static bool ggml_cuda_should_fuse_mul_mat(const ggml_tensor * ffn_up,
- const ggml_tensor * ffn_gate,
- const ggml_tensor * glu,
-@@ -3998,6 +4115,9 @@ static int ggml_cuda_try_fuse(ggml_backend_cuda_context * cuda_ctx, ggml_cgraph
- }
-
- ggml_tensor * node = cgraph->nodes[i];
-+ const int cc = ggml_cuda_info().devices[cuda_ctx->device].cc;
-+ const bool native_bf16 =
-+ GGML_CUDA_CC_IS_NVIDIA(cc) && cc >= GGML_CUDA_CC_AMPERE;
-
- if (int n_fused = ggml_cuda_try_fuse_half_snake(cuda_ctx, cgraph, i)) {
- return n_fused;
-@@ -4126,6 +4246,75 @@ static int ggml_cuda_try_fuse(ggml_backend_cuda_context * cuda_ctx, ggml_cgraph
- }
- }
-
-+ // Shared BF16 projection + broadcast F32 row bias + SiLU + BF16 cast.
-+ // Feed-forward linear2 consumes the rounded BF16 activation directly, so
-+ // avoid materializing the intermediate F32 biased activation entirely.
-+ if (ggml_can_fuse(cgraph, i,
-+ { GGML_OP_MUL_MAT, GGML_OP_ADD, GGML_OP_UNARY, GGML_OP_CPY })) {
-+ ggml_tensor * mm_node = cgraph->nodes[i];
-+ ggml_tensor * bias_node = cgraph->nodes[i + 1];
-+ ggml_tensor * silu_node = cgraph->nodes[i + 2];
-+ ggml_tensor * cast_node = cgraph->nodes[i + 3];
-+ const ggml_tensor * row_bias =
-+ bias_node->src[0] == mm_node ? bias_node->src[1] :
-+ bias_node->src[1] == mm_node ? bias_node->src[0] : nullptr;
-+ const ggml_tensor * weight = mm_node->src[0];
-+ const ggml_tensor * input = mm_node->src[1];
-+ const int64_t cols = mm_node->ne[1] * mm_node->ne[2] * mm_node->ne[3];
-+ const bool eligible =
-+ native_bf16 && row_bias != nullptr && silu_node->src[0] == bias_node &&
-+ cast_node->src[0] == silu_node &&
-+ ggml_get_unary_op(silu_node) == GGML_UNARY_OP_SILU &&
-+ weight->type == GGML_TYPE_BF16 &&
-+ (input->type == GGML_TYPE_F32 || input->type == GGML_TYPE_BF16) &&
-+ mm_node->type == GGML_TYPE_F32 && bias_node->type == GGML_TYPE_F32 &&
-+ silu_node->type == GGML_TYPE_F32 && cast_node->type == GGML_TYPE_BF16 &&
-+ row_bias->type == GGML_TYPE_F32 && ggml_is_contiguous(row_bias) &&
-+ ggml_is_contiguous(bias_node) && ggml_is_contiguous(cast_node) &&
-+ ggml_nelements(row_bias) == mm_node->ne[0] &&
-+ ggml_are_same_shape(mm_node, cast_node) &&
-+ weight->ne[2] == 1 && weight->ne[3] == 1 &&
-+ input->ne[2] * input->ne[3] > 1 && cols <= 65535 &&
-+ !ggml_is_transposed(weight) && !ggml_is_transposed(input) &&
-+ !ggml_backend_buft_is_cuda_split(weight->buffer->buft);
-+ if (eligible) {
-+ ggml_cuda_mul_mat_bf16_row_bias_silu(
-+ *cuda_ctx, weight, input, row_bias, bias_node, cast_node);
-+ return 3;
-+ }
-+ }
-+
-+ // Shared BF16 projection + broadcast F32 row bias. The batched-cuBLAS
-+ // path already materializes BF16 GEMM output before converting it to the
-+ // graph's F32 activation. Add the bias during that conversion rather than
-+ // launching another full-tensor read/modify/write pass.
-+ if (ggml_can_fuse(cgraph, i, { GGML_OP_MUL_MAT, GGML_OP_ADD })) {
-+ ggml_tensor * mm_node = cgraph->nodes[i];
-+ ggml_tensor * bias_node = cgraph->nodes[i + 1];
-+ const ggml_tensor * row_bias =
-+ bias_node->src[0] == mm_node ? bias_node->src[1] :
-+ bias_node->src[1] == mm_node ? bias_node->src[0] : nullptr;
-+ const ggml_tensor * weight = mm_node->src[0];
-+ const ggml_tensor * input = mm_node->src[1];
-+ const int64_t cols = mm_node->ne[1] * mm_node->ne[2] * mm_node->ne[3];
-+ const bool eligible =
-+ native_bf16 && row_bias != nullptr && weight->type == GGML_TYPE_BF16 &&
-+ (input->type == GGML_TYPE_F32 || input->type == GGML_TYPE_BF16) &&
-+ mm_node->type == GGML_TYPE_F32 &&
-+ bias_node->type == GGML_TYPE_F32 && row_bias->type == GGML_TYPE_F32 &&
-+ ggml_is_contiguous(row_bias) && ggml_is_contiguous(bias_node) &&
-+ ggml_nelements(row_bias) == mm_node->ne[0] &&
-+ weight->ne[2] == 1 && weight->ne[3] == 1 &&
-+ input->ne[2] * input->ne[3] > 1 && cols <= 65535 &&
-+ !ggml_is_transposed(weight) && !ggml_is_transposed(input) &&
-+ !ggml_backend_buft_is_cuda_split(weight->buffer->buft);
-+ if (eligible) {
-+ ggml_cuda_mul_mat_bf16_row_bias(
-+ *cuda_ctx, weight, input, row_bias, bias_node);
-+ return 1;
-+ }
-+ }
-+
- // multi-(add or mul)
- if (node->op == GGML_OP_ADD || node->op == GGML_OP_MUL) {
- int n_fuse = 0;
-@@ -4292,6 +4481,25 @@ static int ggml_cuda_try_fuse(ggml_backend_cuda_context * cuda_ctx, ggml_cgraph
- fused_mul_mat_vec = false;
- fused_node_count = 0;
-
-+ // A BF16 projection immediately following SiLU would otherwise launch an
-+ // F32 SiLU kernel and then a separate F32-to-BF16 input conversion. Emit
-+ // the same rounded BF16 activation directly in one pass.
-+ if (ggml_can_fuse(cgraph, i, { GGML_OP_UNARY, GGML_OP_CPY })) {
-+ const ggml_tensor * silu_node = cgraph->nodes[i];
-+ ggml_tensor * cast_node = cgraph->nodes[i + 1];
-+ const bool eligible =
-+ native_bf16 && ggml_get_unary_op(silu_node) == GGML_UNARY_OP_SILU &&
-+ cast_node->src[0] == silu_node &&
-+ silu_node->src[0]->type == GGML_TYPE_F32 &&
-+ silu_node->type == GGML_TYPE_F32 && cast_node->type == GGML_TYPE_BF16 &&
-+ ggml_are_same_shape(silu_node->src[0], cast_node) &&
-+ ggml_is_contiguous(silu_node->src[0]) && ggml_is_contiguous(cast_node);
-+ if (eligible) {
-+ ggml_cuda_op_silu_f32_to_bf16(*cuda_ctx, silu_node, cast_node);
-+ return 1;
-+ }
-+ }
-+
- // Q8 narrow projection + SiLU. NeMo-Speech.cpp's Conformer FF1
- // layers are bias-free, so this is their actual hot graph sequence.
- if (ggml_cuda_q8_narrow_epilogue_enabled() &&
-diff --git a/src/ggml-cuda/unary.cu b/src/ggml-cuda/unary.cu
-index 2aeba26f..36c608e1 100644
---- a/src/ggml-cuda/unary.cu
-+++ b/src/ggml-cuda/unary.cu
-@@ -183,6 +183,37 @@ void ggml_cuda_op_silu(ggml_backend_cuda_context & ctx, ggml_tensor * dst) {
- ggml_cuda_op_unary(ctx, dst);
- }
-
-+static __global__ void silu_f32_to_bf16(
-+ const float * __restrict__ x, nv_bfloat16 * __restrict__ dst,
-+ const int64_t nelements) {
-+#if defined(__CUDA_ARCH__) && __CUDA_ARCH__ >= GGML_CUDA_CC_AMPERE
-+ const int64_t i = (int64_t) blockIdx.x * blockDim.x + threadIdx.x;
-+ if (i < nelements) {
-+ dst[i] = __float2bfloat16(op_silu(x[i]));
-+ }
-+#else
-+ GGML_UNUSED_VARS(x, dst, nelements);
-+ NO_DEVICE_CODE;
-+#endif
-+}
-+
-+void ggml_cuda_op_silu_f32_to_bf16(
-+ ggml_backend_cuda_context & ctx, const ggml_tensor * silu_node,
-+ ggml_tensor * dst) {
-+ const ggml_tensor * src = silu_node->src[0];
-+ GGML_ASSERT(src->type == GGML_TYPE_F32);
-+ GGML_ASSERT(silu_node->type == GGML_TYPE_F32);
-+ GGML_ASSERT(dst->type == GGML_TYPE_BF16);
-+ GGML_ASSERT(ggml_are_same_shape(src, dst));
-+ GGML_ASSERT(ggml_is_contiguous(src));
-+ GGML_ASSERT(ggml_is_contiguous(dst));
-+
-+ const int64_t nelements = ggml_nelements(src);
-+ const int64_t num_blocks = (nelements + CUDA_SILU_BLOCK_SIZE - 1) / CUDA_SILU_BLOCK_SIZE;
-+ silu_f32_to_bf16<<>>(
-+ (const float *) src->data, (nv_bfloat16 *) dst->data, nelements);
-+}
-+
- void ggml_cuda_op_tanh(ggml_backend_cuda_context & ctx, ggml_tensor * dst) {
- ggml_cuda_op_unary(ctx, dst);
- }
-diff --git a/src/ggml-cuda/unary.cuh b/src/ggml-cuda/unary.cuh
-index 81ed873e..592a0dec 100644
---- a/src/ggml-cuda/unary.cuh
-+++ b/src/ggml-cuda/unary.cuh
-@@ -31,6 +31,9 @@ void ggml_cuda_op_gelu(ggml_backend_cuda_context & ctx, ggml_tensor * dst);
-
- void ggml_cuda_op_silu(ggml_backend_cuda_context & ctx, ggml_tensor * dst);
-
-+void ggml_cuda_op_silu_f32_to_bf16(
-+ ggml_backend_cuda_context & ctx, const ggml_tensor * silu_node, ggml_tensor * dst);
-+
- void ggml_cuda_op_silu_back(ggml_backend_cuda_context & ctx, ggml_tensor * dst);
-
- void ggml_cuda_op_gelu_erf(ggml_backend_cuda_context & ctx, ggml_tensor * dst);
diff --git a/ggml-patches/0009-fastconformer-cuda-fusions.patch b/ggml-patches/0009-fastconformer-cuda-fusions.patch
deleted file mode 100644
index 204d203..0000000
--- a/ggml-patches/0009-fastconformer-cuda-fusions.patch
+++ /dev/null
@@ -1,330 +0,0 @@
-diff --git a/include/ggml.h b/include/ggml.h
-index a4727087..fb823571 100644
---- a/include/ggml.h
-+++ b/include/ggml.h
-@@ -622,6 +622,7 @@ extern "C" {
- GGML_GLU_OP_SWIGLU_OAI,
- GGML_GLU_OP_GEGLU_ERF,
- GGML_GLU_OP_GEGLU_QUICK,
-+ GGML_GLU_OP_SIGMOID,
-
- GGML_GLU_OP_COUNT,
- };
-diff --git a/src/ggml-cuda/common.cuh b/src/ggml-cuda/common.cuh
-index 56fe1bb7..01b239f1 100644
---- a/src/ggml-cuda/common.cuh
-+++ b/src/ggml-cuda/common.cuh
-@@ -51,6 +51,7 @@
- #define GGML_CUDA_CC_TURING 750
- #define GGML_CUDA_CC_AMPERE 800
- #define GGML_CUDA_CC_ADA_LOVELACE 890
-+#define GGML_CUDA_CC_HOPPER 900
- // Jetson Thor is Blackwell SM110. The hand-written FP4 block-scale PTX below is currently SM120-only, but SM110
- // should still be detected so ggml can favor CUDA library kernels that may use Thor tcgen05 tensor cores.
- #define GGML_CUDA_CC_THOR 1100
-diff --git a/src/ggml-cuda/ggml-cuda.cu b/src/ggml-cuda/ggml-cuda.cu
-index 1a1b58fb..41605f45 100644
---- a/src/ggml-cuda/ggml-cuda.cu
-+++ b/src/ggml-cuda/ggml-cuda.cu
-@@ -3063,6 +3063,9 @@ static bool ggml_cuda_compute_forward(ggml_backend_cuda_context & ctx, struct gg
- case GGML_GLU_OP_GEGLU_QUICK:
- ggml_cuda_op_geglu_quick(ctx, dst);
- break;
-+ case GGML_GLU_OP_SIGMOID:
-+ ggml_cuda_op_sigmoid_glu(ctx, dst);
-+ break;
- default:
- return false;
- }
-@@ -4284,6 +4287,31 @@ static int ggml_cuda_try_fuse(ggml_backend_cuda_context * cuda_ctx, ggml_cgraph
- }
- }
-
-+ // Macaron feed-forward residual: residual + scale * ff. Both tensors are
-+ // contiguous F32 and have identical shapes, so materializing the scaled
-+ // activation only creates an avoidable full-tensor round trip.
-+ if (ggml_can_fuse(cgraph, i, { GGML_OP_SCALE, GGML_OP_ADD })) {
-+ const ggml_tensor * scale_node = cgraph->nodes[i];
-+ ggml_tensor * add_node = cgraph->nodes[i + 1];
-+ const ggml_tensor * residual =
-+ add_node->src[0] == scale_node ? add_node->src[1] :
-+ add_node->src[1] == scale_node ? add_node->src[0] : nullptr;
-+
-+ const float bias = ggml_get_op_params_f32(scale_node, 1);
-+ const bool eligible =
-+ residual != nullptr && bias == 0.0f &&
-+ scale_node->src[0]->type == GGML_TYPE_F32 &&
-+ residual->type == GGML_TYPE_F32 && add_node->type == GGML_TYPE_F32 &&
-+ ggml_are_same_shape(scale_node->src[0], residual) &&
-+ ggml_are_same_shape(scale_node->src[0], add_node) &&
-+ ggml_is_contiguous(scale_node->src[0]) &&
-+ ggml_is_contiguous(residual) && ggml_is_contiguous(add_node);
-+ if (eligible) {
-+ ggml_cuda_op_scale_add(*cuda_ctx, scale_node, residual, add_node);
-+ return 1;
-+ }
-+ }
-+
- // Shared BF16 projection + broadcast F32 row bias. The batched-cuBLAS
- // path already materializes BF16 GEMM output before converting it to the
- // graph's F32 activation. Add the bias during that conversion rather than
-@@ -4628,6 +4656,35 @@ static int ggml_cuda_try_fuse(ggml_backend_cuda_context * cuda_ctx, ggml_cgraph
- return fused_node_count - 1;
- }
-
-+ // LayerNorm affine followed by a BF16 projection. Store the affine result
-+ // directly in the projection's input precision instead of writing F32 and
-+ // launching a second full-tensor conversion.
-+ if (ggml_can_fuse(cgraph, i,
-+ { GGML_OP_NORM, GGML_OP_MUL, GGML_OP_ADD, GGML_OP_CPY })) {
-+ ggml_tensor * norm = cgraph->nodes[i];
-+ ggml_tensor * mul = cgraph->nodes[i + 1];
-+ ggml_tensor * add = cgraph->nodes[i + 2];
-+ ggml_tensor * cast = cgraph->nodes[i + 3];
-+ const ggml_tensor * gamma =
-+ mul->src[0] == norm ? mul->src[1] : mul->src[0];
-+ const ggml_tensor * beta =
-+ add->src[0] == mul ? add->src[1] : add->src[0];
-+ const bool eligible =
-+ native_bf16 && cast->src[0] == add && norm->src[0]->type == GGML_TYPE_F32 &&
-+ norm->type == GGML_TYPE_F32 && mul->type == GGML_TYPE_F32 &&
-+ add->type == GGML_TYPE_F32 && cast->type == GGML_TYPE_BF16 &&
-+ gamma->type == GGML_TYPE_F32 && beta->type == GGML_TYPE_F32 &&
-+ ggml_nelements(gamma) == norm->ne[0] &&
-+ ggml_nelements(beta) == norm->ne[0] &&
-+ ggml_is_contiguous(gamma) && ggml_is_contiguous(beta) &&
-+ ggml_is_contiguous(mul) && ggml_is_contiguous(add) &&
-+ ggml_is_contiguous(cast) && ggml_are_same_shape(norm, cast);
-+ if (eligible) {
-+ ggml_cuda_op_norm_fused(*cuda_ctx, norm, mul, add, cast);
-+ return 3;
-+ }
-+ }
-+
- if (ggml_cuda_can_fuse(cgraph, i, { GGML_OP_RMS_NORM, GGML_OP_MUL, GGML_OP_ADD }, {})) {
- ggml_cuda_op_rms_norm_fused_add(*cuda_ctx, node, cgraph->nodes[i + 1], cgraph->nodes[i + 2]);
- return 2;
-@@ -5551,6 +5608,7 @@ static bool ggml_backend_cuda_device_supports_op(ggml_backend_dev_t dev, const g
- case GGML_GLU_OP_SWIGLU_OAI:
- case GGML_GLU_OP_GEGLU_ERF:
- case GGML_GLU_OP_GEGLU_QUICK:
-+ case GGML_GLU_OP_SIGMOID:
- return ggml_is_contiguous_1(op->src[0]);
- default:
- return false;
-diff --git a/src/ggml-cuda/norm.cu b/src/ggml-cuda/norm.cu
-index d77e1a61..555c9514 100644
---- a/src/ggml-cuda/norm.cu
-+++ b/src/ggml-cuda/norm.cu
-@@ -42,9 +42,9 @@ static __global__ void norm_f32(
- // to ne0-length vectors broadcast over rows/channels/samples (the standard
- // LayerNorm affine), enforced by ggml_cuda_can_fuse — this keeps indexing
- // trivial instead of replicating rms_norm's general broadcast machinery.
--template
-+template
- static __global__ void norm_mul_add_f32(
-- const float * x, const float * mul, const float * add, float * dst, const int ncols,
-+ const float * x, const float * mul, const float * add, T * dst, const int ncols,
- const int64_t stride_row, const int64_t stride_channel, const int64_t stride_sample, const float eps) {
- const int nrows = gridDim.x;
- const int nchannels = gridDim.y;
-@@ -77,7 +77,7 @@ static __global__ void norm_mul_add_f32(
- if constexpr (do_add) {
- v += add[col];
- }
-- dst[col] = v;
-+ dst[col] = (T) v;
- }
- }
-
-@@ -327,27 +327,41 @@ static void norm_f32_cuda(
- }
- }
-
--static void norm_mul_add_f32_cuda(
-- const float * x, const float * mul, const float * add, float * dst, const int ncols, const int nrows,
-+template
-+static void norm_mul_add_cuda(
-+ const float * x, const float * mul, const float * add, T * dst, const int ncols, const int nrows,
- const int nchannels, const int nsamples, const int64_t stride_row, const int64_t stride_channel,
- const int64_t stride_sample, const float eps, cudaStream_t stream) {
- const dim3 blocks_num(nrows, nchannels, nsamples);
-+ const int cc = ggml_cuda_info().devices[ggml_cuda_get_device()].cc;
- if (ncols < 1024) {
- const dim3 block_dims(WARP_SIZE, 1, 1);
- if (add) {
-- norm_mul_add_f32<<>>(
-+ norm_mul_add_f32<<>>(
- x, mul, add, dst, ncols, stride_row, stride_channel, stride_sample, eps);
- } else {
-- norm_mul_add_f32<<>>(
-+ norm_mul_add_f32<<>>(
-+ x, mul, add, dst, ncols, stride_row, stride_channel, stride_sample, eps);
-+ }
-+ } else if (ncols == 1024 && GGML_CUDA_CC_IS_NVIDIA(cc) && cc >= GGML_CUDA_CC_HOPPER) {
-+ // Hopper and newer: four elements per thread keeps high occupancy
-+ // while reducing the affine LayerNorm reduction from 32 warps/block
-+ // to 8. Retain the established 1024-thread path on older devices.
-+ const dim3 block_dims(256, 1, 1);
-+ if (add) {
-+ norm_mul_add_f32<256, true, T><<>>(
-+ x, mul, add, dst, ncols, stride_row, stride_channel, stride_sample, eps);
-+ } else {
-+ norm_mul_add_f32<256, false, T><<>>(
- x, mul, add, dst, ncols, stride_row, stride_channel, stride_sample, eps);
- }
- } else {
- const dim3 block_dims(1024, 1, 1);
- if (add) {
-- norm_mul_add_f32<1024, true><<>>(
-+ norm_mul_add_f32<1024, true, T><<>>(
- x, mul, add, dst, ncols, stride_row, stride_channel, stride_sample, eps);
- } else {
-- norm_mul_add_f32<1024, false><<>>(
-+ norm_mul_add_f32<1024, false, T><<>>(
- x, mul, add, dst, ncols, stride_row, stride_channel, stride_sample, eps);
- }
- }
-@@ -505,7 +519,7 @@ void ggml_cuda_op_norm(ggml_backend_cuda_context & ctx, ggml_tensor * dst) {
- // or add_tensor), mirroring ggml_cuda_op_rms_norm_fused(_add). Eligibility
- // (row-vector operands, F32, contiguity) is enforced in ggml_cuda_can_fuse.
- void ggml_cuda_op_norm_fused(ggml_backend_cuda_context & ctx, ggml_tensor * dst, ggml_tensor * mul_tensor,
-- ggml_tensor * add_tensor) {
-+ ggml_tensor * add_tensor, ggml_tensor * bf16_dst) {
- const ggml_tensor * norm_src = (ggml_tensor *) dst->src[0];
- float eps = 0.0f;
- memcpy(&eps, dst->op_params, sizeof(float));
-@@ -549,8 +563,18 @@ void ggml_cuda_op_norm_fused(ggml_backend_cuda_context & ctx, ggml_tensor * dst,
- const int64_t s02 = norm_src->nb[2] / ts0;
- const int64_t s03 = norm_src->nb[3] / ts0;
-
-- norm_mul_add_f32_cuda(
-- src0_d, mul_d, add_d, dst_d, ne00, ne01, ne02, ne03, s01, s02, s03, eps, ctx.stream());
-+ if (bf16_dst != nullptr) {
-+ GGML_ASSERT(bf16_dst->type == GGML_TYPE_BF16);
-+ GGML_ASSERT(ggml_is_contiguous(bf16_dst));
-+ GGML_ASSERT(ggml_are_same_shape(dst, bf16_dst));
-+ norm_mul_add_cuda(
-+ src0_d, mul_d, add_d, (nv_bfloat16 *) bf16_dst->data,
-+ ne00, ne01, ne02, ne03, s01, s02, s03, eps, ctx.stream());
-+ } else {
-+ norm_mul_add_cuda(
-+ src0_d, mul_d, add_d, dst_d,
-+ ne00, ne01, ne02, ne03, s01, s02, s03, eps, ctx.stream());
-+ }
- }
-
- void ggml_cuda_op_group_norm(ggml_backend_cuda_context & ctx, ggml_tensor * dst) {
-diff --git a/src/ggml-cuda/norm.cuh b/src/ggml-cuda/norm.cuh
-index 20ecaa2d..8f4287c3 100644
---- a/src/ggml-cuda/norm.cuh
-+++ b/src/ggml-cuda/norm.cuh
-@@ -5,7 +5,7 @@ void ggml_cuda_op_norm(ggml_backend_cuda_context & ctx, ggml_tensor * dst);
- // Fused LayerNorm + row-vector mul (gamma) + optional row-vector add (beta).
- // add_tensor may be nullptr (norm+mul only).
- void ggml_cuda_op_norm_fused(ggml_backend_cuda_context & ctx, ggml_tensor * dst, ggml_tensor * mul_tensor,
-- ggml_tensor * add_tensor);
-+ ggml_tensor * add_tensor, ggml_tensor * bf16_dst = nullptr);
-
- void ggml_cuda_op_group_norm(ggml_backend_cuda_context & ctx, ggml_tensor * dst);
-
-diff --git a/src/ggml-cuda/scale.cu b/src/ggml-cuda/scale.cu
-index 0ddeff6a..bdd1eb4f 100644
---- a/src/ggml-cuda/scale.cu
-+++ b/src/ggml-cuda/scale.cu
-@@ -32,3 +32,39 @@ void ggml_cuda_op_scale(ggml_backend_cuda_context & ctx, ggml_tensor * dst) {
-
- scale_f32_cuda(src0_d, dst_d, scale, bias, ggml_nelements(src0), stream);
- }
-+
-+static __global__ void scale_add_f32(
-+ const float * __restrict__ x, const float * __restrict__ residual,
-+ float * __restrict__ dst, const float scale, const int64_t nelements) {
-+ const int64_t tid = (int64_t) blockIdx.x * blockDim.x + threadIdx.x;
-+ const int64_t stride = (int64_t) blockDim.x * gridDim.x;
-+
-+ for (int64_t i = tid; i < nelements; i += stride) {
-+ dst[i] = residual[i] + scale * x[i];
-+ }
-+}
-+
-+void ggml_cuda_op_scale_add(
-+ ggml_backend_cuda_context & ctx, const ggml_tensor * scale_node,
-+ const ggml_tensor * residual, ggml_tensor * dst) {
-+ const ggml_tensor * x = scale_node->src[0];
-+
-+ GGML_ASSERT(x->type == GGML_TYPE_F32);
-+ GGML_ASSERT(residual->type == GGML_TYPE_F32);
-+ GGML_ASSERT(dst->type == GGML_TYPE_F32);
-+ GGML_ASSERT(ggml_is_contiguous(x));
-+ GGML_ASSERT(ggml_is_contiguous(residual));
-+ GGML_ASSERT(ggml_is_contiguous(dst));
-+ GGML_ASSERT(ggml_are_same_shape(x, residual));
-+ GGML_ASSERT(ggml_are_same_shape(x, dst));
-+
-+ const float scale = ggml_get_op_params_f32(scale_node, 0);
-+ const float bias = ggml_get_op_params_f32(scale_node, 1);
-+ GGML_ASSERT(bias == 0.0f);
-+
-+ const int64_t nelements = ggml_nelements(x);
-+ const int64_t num_blocks = (nelements + CUDA_SCALE_BLOCK_SIZE - 1) / CUDA_SCALE_BLOCK_SIZE;
-+ scale_add_f32<<>>(
-+ (const float *) x->data, (const float *) residual->data,
-+ (float *) dst->data, scale, nelements);
-+}
-diff --git a/src/ggml-cuda/scale.cuh b/src/ggml-cuda/scale.cuh
-index 8ff75c82..16fe10ab 100644
---- a/src/ggml-cuda/scale.cuh
-+++ b/src/ggml-cuda/scale.cuh
-@@ -3,3 +3,7 @@
- #define CUDA_SCALE_BLOCK_SIZE 256
-
- void ggml_cuda_op_scale(ggml_backend_cuda_context & ctx, ggml_tensor * dst);
-+
-+void ggml_cuda_op_scale_add(
-+ ggml_backend_cuda_context & ctx, const ggml_tensor * scale_node,
-+ const ggml_tensor * residual, ggml_tensor * dst);
-diff --git a/src/ggml-cuda/unary.cu b/src/ggml-cuda/unary.cu
-index 36c608e1..4c68411f 100644
---- a/src/ggml-cuda/unary.cu
-+++ b/src/ggml-cuda/unary.cu
-@@ -374,6 +374,10 @@ void ggml_cuda_op_swiglu(ggml_backend_cuda_context & ctx, ggml_tensor * dst) {
- ggml_cuda_op_unary_gated(ctx, dst);
- }
-
-+void ggml_cuda_op_sigmoid_glu(ggml_backend_cuda_context & ctx, ggml_tensor * dst) {
-+ ggml_cuda_op_unary_gated(ctx, dst);
-+}
-+
- void ggml_cuda_op_geglu_erf(ggml_backend_cuda_context & ctx, ggml_tensor * dst) {
- ggml_cuda_op_unary_gated(ctx, dst);
- }
-diff --git a/src/ggml-cuda/unary.cuh b/src/ggml-cuda/unary.cuh
-index 592a0dec..53073275 100644
---- a/src/ggml-cuda/unary.cuh
-+++ b/src/ggml-cuda/unary.cuh
-@@ -84,6 +84,8 @@ void ggml_cuda_op_geglu(ggml_backend_cuda_context & ctx, ggml_tensor * dst);
-
- void ggml_cuda_op_swiglu(ggml_backend_cuda_context & ctx, ggml_tensor * dst);
-
-+void ggml_cuda_op_sigmoid_glu(ggml_backend_cuda_context & ctx, ggml_tensor * dst);
-+
- void ggml_cuda_op_swiglu_oai(ggml_backend_cuda_context & ctx, ggml_tensor * dst);
-
- void ggml_cuda_op_geglu_erf(ggml_backend_cuda_context & ctx, ggml_tensor * dst);
-diff --git a/src/ggml.c b/src/ggml.c
-index 88b39537..80b5802f 100644
---- a/src/ggml.c
-+++ b/src/ggml.c
-@@ -1232,9 +1232,10 @@ static const char * GGML_GLU_OP_NAME[GGML_GLU_OP_COUNT] = {
- "SWIGLU_OAI",
- "GEGLU_ERF",
- "GEGLU_QUICK",
-+ "SIGMOID_GLU",
- };
-
--static_assert(GGML_GLU_OP_COUNT == 6, "GGML_GLU_OP_COUNT != 6");
-+static_assert(GGML_GLU_OP_COUNT == 7, "GGML_GLU_OP_COUNT != 7");
-
-
- static_assert(sizeof(struct ggml_object)%GGML_MEM_ALIGN == 0, "ggml_object size must be a multiple of GGML_MEM_ALIGN");
diff --git a/ggml-patches/0012-cuda-streaming-cache-copies.patch b/ggml-patches/0012-cuda-streaming-cache-copies.patch
deleted file mode 100644
index 453a40b..0000000
--- a/ggml-patches/0012-cuda-streaming-cache-copies.patch
+++ /dev/null
@@ -1,151 +0,0 @@
-diff --git a/src/ggml-cuda/cpy.cu b/src/ggml-cuda/cpy.cu
-index d208acf2..79861b4d 100644
---- a/src/ggml-cuda/cpy.cu
-+++ b/src/ggml-cuda/cpy.cu
-@@ -407,7 +407,24 @@ void ggml_cuda_cpy(ggml_backend_cuda_context & ctx, const ggml_tensor * src0, gg
- const bool can_be_transposed = nb01 == (int64_t)ggml_element_size(src0) &&
- src0->ne[3] == 1 && nb02 == ne00 * ne01 * (int64_t)ggml_element_size(src0);
-
-- if (src0->type == src1->type && contiguous_srcs) {
-+ // Cache-aware ASR keeps the last time rows of a [C,T,B] tensor. The view
-+ // is contiguous within each batch plane but has the original T stride
-+ // between planes; materializing it with cpy_scalar performs six 64-bit
-+ // div/mod operations per float. Express this common layout as one pitched
-+ // device copy instead (for C512 K/V this is 512 x ~224 KiB rows).
-+ const size_t elem_size = ggml_element_size(src0);
-+ const size_t plane_bytes = (size_t) ne00 * ne01 * elem_size;
-+ const bool pitched_f32_to_contiguous =
-+ src0->type == GGML_TYPE_F32 && src1->type == GGML_TYPE_F32 &&
-+ src0->ne[3] == 1 && ne == ne00 * ne01 * ne02 &&
-+ nb00 == (int64_t) elem_size && nb01 == ne00 * (int64_t) elem_size &&
-+ nb02 >= (int64_t) plane_bytes && ggml_is_contiguous(src1);
-+
-+ if (pitched_f32_to_contiguous && !ggml_is_contiguous(src0)) {
-+ CUDA_CHECK(cudaMemcpy2DAsync(
-+ src1_ddc, plane_bytes, src0_ddc, (size_t) nb02,
-+ plane_bytes, (size_t) ne02, cudaMemcpyDeviceToDevice, main_stream));
-+ } else if (src0->type == src1->type && contiguous_srcs) {
- GGML_ASSERT(ggml_nbytes(src0) == ggml_nbytes(src1));
- #if defined(GGML_USE_MUSA) && defined(GGML_MUSA_MUDNN_COPY)
- if (src0->type == GGML_TYPE_F32 || src0->type == GGML_TYPE_F16) {
-diff --git a/src/ggml-cuda/getrows.cu b/src/ggml-cuda/getrows.cu
-index 36b840e8..c91ac35a 100644
---- a/src/ggml-cuda/getrows.cu
-+++ b/src/ggml-cuda/getrows.cu
-@@ -70,6 +70,28 @@ static __global__ void k_get_rows_float(
- }
- }
-
-+// Fast path for large F32 cache rows. One CTA owns an output row and moves
-+// float4 vectors from the indexed arena row. The generic kernel creates one
-+// CTA per 256 scalar columns and reloads/recomputes the same row metadata in
-+// every CTA; a 57,344-element ASR K/V row therefore used 224 CTAs.
-+static __global__ void k_get_rows_contiguous_f32x4(
-+ const float * __restrict__ src0, const int32_t * __restrict__ rows,
-+ float * __restrict__ dst, const int row_elements, const size_t src_row_stride,
-+ const size_t src_plane_stride, const size_t dst_plane_stride,
-+ const size_t row_index_plane_stride) {
-+ const int output_row = (int) blockIdx.x;
-+ const int plane = (int) blockIdx.y;
-+ const int source_row = rows[output_row + (size_t) plane * row_index_plane_stride];
-+ const float4 * src = (const float4 *) (
-+ src0 + (size_t) plane * src_plane_stride + (size_t) source_row * src_row_stride);
-+ float4 * out = (float4 *) (
-+ dst + (size_t) plane * dst_plane_stride + (size_t) output_row * row_elements);
-+ const int vectors = row_elements / 4;
-+ for (int i = threadIdx.x; i < vectors; i += blockDim.x) {
-+ out[i] = src[i];
-+ }
-+}
-+
- template
- static __global__ void k_get_rows_back_float(
- const grad_t * __restrict__ grad, const int32_t * __restrict__ rows, dst_t * __restrict__ dst, const int64_t ncols, const int64_t nrows_grad) {
-@@ -264,6 +286,24 @@ void ggml_cuda_op_get_rows(ggml_backend_cuda_context & ctx, ggml_tensor * dst) {
- GGML_ASSERT(src1->nb[0] == ggml_type_size(src1->type));
- GGML_ASSERT(dst->nb[0] == ggml_type_size(dst->type));
-
-+ const bool contiguous_cache_rows =
-+ src0->type == GGML_TYPE_F32 && dst->type == GGML_TYPE_F32 &&
-+ src1->type == GGML_TYPE_I32 && src0->ne[3] == 1 &&
-+ src1->ne[1] == src0->ne[2] && src1->ne[2] == 1 && src1->ne[3] == 1 &&
-+ dst->ne[2] == src0->ne[2] && dst->ne[3] == 1 && dst->ne[0] == src0->ne[0] &&
-+ dst->ne[1] == src1->ne[0] && ggml_is_contiguous(dst) &&
-+ src0->nb[0] == sizeof(float) && src0->nb[1] % sizeof(float4) == 0 &&
-+ src0->ne[0] >= 1024 && src0->ne[0] % 4 == 0;
-+ if (contiguous_cache_rows) {
-+ const dim3 grid((unsigned) src1->ne[0], (unsigned) src0->ne[2]);
-+ k_get_rows_contiguous_f32x4<<>>(
-+ (const float *) src0->data, (const int32_t *) src1->data,
-+ (float *) dst->data, (int) src0->ne[0], src0->nb[1] / sizeof(float),
-+ src0->nb[2] / sizeof(float), dst->nb[2] / sizeof(float),
-+ src1->nb[1] / sizeof(int32_t));
-+ return;
-+ }
-+
- get_rows_cuda(src0->data, src0->type, (const int32_t *) src1->data, dst->data, dst->type,
- ne00, nb01, nb02, nb03, ne10, ne11, ne12, nb10, nb11, nb12, nb1, nb2, nb3, stream);
- }
-diff --git a/src/ggml-cuda/set-rows.cu b/src/ggml-cuda/set-rows.cu
-index 631de7e8..0c5e5abc 100644
---- a/src/ggml-cuda/set-rows.cu
-+++ b/src/ggml-cuda/set-rows.cu
-@@ -170,6 +170,26 @@ static __global__ void k_set_rows(const src_t * __restrict__ src0,
- GGML_UNUSED(ne13);
- }
-
-+template
-+static __global__ void k_set_rows_contiguous_f32x4(
-+ const float * __restrict__ src0, const idx_t * __restrict__ rows,
-+ float * __restrict__ dst, const int row_elements, const size_t dst_row_stride,
-+ const size_t src_plane_stride, const size_t dst_plane_stride,
-+ const size_t row_index_plane_stride) {
-+ const int source_row = (int) blockIdx.x;
-+ const int plane = (int) blockIdx.y;
-+ const int64_t destination_row =
-+ (int64_t) rows[source_row + (size_t) plane * row_index_plane_stride];
-+ const float4 * src = (const float4 *) (
-+ src0 + (size_t) plane * src_plane_stride + (size_t) source_row * row_elements);
-+ float4 * out = (float4 *) (
-+ dst + (size_t) plane * dst_plane_stride + destination_row * dst_row_stride);
-+ const int vectors = row_elements / 4;
-+ for (int i = threadIdx.x; i < vectors; i += blockDim.x) {
-+ out[i] = src[i];
-+ }
-+}
-+
- template
- static void set_rows_cuda(
- const src_t * src0_d, const idx_t * src1_d, dst_t * dst_d,
-@@ -322,6 +342,31 @@ void ggml_cuda_op_set_rows(ggml_backend_cuda_context & ctx, ggml_tensor * dst) {
- GGML_ASSERT(src0->type == GGML_TYPE_F32);
- GGML_ASSERT(src1->type == GGML_TYPE_I64 || src1->type == GGML_TYPE_I32);
-
-+ const bool contiguous_cache_rows =
-+ dst->type == GGML_TYPE_F32 && src0->ne[3] == 1 &&
-+ src1->ne[1] == src0->ne[2] && src1->ne[2] == 1 && src1->ne[3] == 1 &&
-+ dst->ne[2] == src0->ne[2] && dst->ne[3] == 1 && src0->ne[0] == dst->ne[0] &&
-+ src0->ne[1] == src1->ne[0] && ggml_is_contiguous(src0) &&
-+ dst->nb[0] == sizeof(float) && dst->nb[1] % sizeof(float4) == 0 &&
-+ src0->ne[0] >= 1024 && src0->ne[0] % 4 == 0;
-+ if (contiguous_cache_rows) {
-+ const dim3 grid((unsigned) src0->ne[1], (unsigned) src0->ne[2]);
-+ if (src1->type == GGML_TYPE_I64) {
-+ k_set_rows_contiguous_f32x4<<>>(
-+ (const float *) src0->data, (const int64_t *) src1->data,
-+ (float *) dst->data, (int) src0->ne[0], dst->nb[1] / sizeof(float),
-+ src0->nb[2] / sizeof(float), dst->nb[2] / sizeof(float),
-+ src1->nb[1] / sizeof(int64_t));
-+ } else {
-+ k_set_rows_contiguous_f32x4<<>>(
-+ (const float *) src0->data, (const int32_t *) src1->data,
-+ (float *) dst->data, (int) src0->ne[0], dst->nb[1] / sizeof(float),
-+ src0->nb[2] / sizeof(float), dst->nb[2] / sizeof(float),
-+ src1->nb[1] / sizeof(int32_t));
-+ }
-+ return;
-+ }
-+
- if (src1->type == GGML_TYPE_I64) {
- set_rows_cuda(ctx, src0, src1, dst);
- } else {
diff --git a/ggml-patches/0013-cuda-cached-f16-cublas.patch b/ggml-patches/0013-cuda-cached-f16-cublas.patch
deleted file mode 100644
index aec9e90..0000000
--- a/ggml-patches/0013-cuda-cached-f16-cublas.patch
+++ /dev/null
@@ -1,184 +0,0 @@
-diff --git a/src/ggml-cuda/CMakeLists.txt b/src/ggml-cuda/CMakeLists.txt
-index 07a94052..f43ee5c4 100644
---- a/src/ggml-cuda/CMakeLists.txt
-+++ b/src/ggml-cuda/CMakeLists.txt
-@@ -77,15 +77,15 @@ if (CUDAToolkit_FOUND)
- endif()
-
- # Replace plain Blackwell CUDA architectures with their "architecture-specific" equivalents.
-- # 11X/12X are forwards-compatible, 11Xa/12Xa are not.
-+ # 10X/11X/12X are forwards-compatible, 10Xa/11Xa/12Xa are not.
- # Notably the Blackwell tensor core instructions are not forwards compatible and therefore need architecture-specific targets.
- # But while 12X vs. 12Xa can be checked in device code there is (to my knowledge) no easy way to do the same check in host code.
- # So for now just replace the supported plain Blackwell targets with architecture-specific targets.
- foreach(ARCHS IN ITEMS CMAKE_CUDA_ARCHITECTURES CMAKE_CUDA_ARCHITECTURES_NATIVE)
- set(FIXED_ARCHS "")
- foreach(ARCH IN LISTS ${ARCHS})
-- if (ARCH MATCHES "^(110|12[0-9])(-real|-virtual)?$")
-- string(REGEX REPLACE "^(110|12[0-9])((-real|-virtual)?)$" "\\1a\\2" FIXED_ARCH ${ARCH})
-+ if (ARCH MATCHES "^(100|110|12[0-9])(-real|-virtual)?$")
-+ string(REGEX REPLACE "^(100|110|12[0-9])((-real|-virtual)?)$" "\\1a\\2" FIXED_ARCH ${ARCH})
- message(STATUS "Replacing ${ARCH} in ${ARCHS} with ${FIXED_ARCH}")
- list(APPEND FIXED_ARCHS "${FIXED_ARCH}")
- else()
-@@ -133,6 +133,9 @@ if (CUDAToolkit_FOUND)
- ${GGML_SOURCES_CUDA}
- )
-
-+ # The cached-F16 path uses portable CUDA conversion and cuBLAS kernels.
-+ target_compile_definitions(ggml-cuda PRIVATE GGML_CUDA_SKINNY_Q8_CUBLAS_F16=1)
-+
- add_compile_definitions(GGML_CUDA_PEER_MAX_BATCH_SIZE=${GGML_CUDA_PEER_MAX_BATCH_SIZE})
-
- if (GGML_CUDA_GRAPHS)
-diff --git a/src/ggml-cuda/skinny-q8.cu b/src/ggml-cuda/skinny-q8.cu
-index 88e1e1ec..1a7b4f3f 100644
---- a/src/ggml-cuda/skinny-q8.cu
-+++ b/src/ggml-cuda/skinny-q8.cu
-@@ -27,6 +27,9 @@
- // batch size, so singleton and batched execution cannot select different math.
-
- #include "skinny-q8.cuh"
-+#if defined(GGML_CUDA_SKINNY_Q8_CUBLAS_F16)
-+#include "convert.cuh"
-+#endif
- #include "mmvq.cuh" // MMVQ_MAX_BATCH_SIZE: the N range mmvq already covers
-
- #include
-@@ -41,7 +44,6 @@
- #define SKQ8_KTB (SKQ8_KSTEP / 32) // q8 blocks per stage
- #define SKQ8_STAGES 2
- #define SKQ8_NMAX SKQ8_NPAD
--
- // Shared-memory strides (bytes). KSTEP + 16 keeps the 16B cp.async stores
- // aligned while breaking the power-of-two bank pattern on fragment loads.
- #define SKQ8_SW (SKQ8_KSTEP + 16)
-@@ -73,6 +75,35 @@ static __global__ void skq8_repack(
- }
- }
-
-+#if defined(GGML_CUDA_SKINNY_Q8_CUBLAS_F16)
-+// One-time expansion of a planar Q8 weight into F16. Caching removes weight
-+// dequantization from every invocation while retaining the compact Q8 model
-+// on disk.
-+static __global__ void skq8_dequantize_weight_f16(
-+ const int8_t * __restrict__ qs, const half * __restrict__ d,
-+ half * __restrict__ out, const int M, const int K) {
-+ const size_t i2 = (size_t) blockIdx.x * blockDim.x + threadIdx.x;
-+ const size_t i = i2 * 2;
-+ if (i >= (size_t) M * K) {
-+ return;
-+ }
-+ const int col = (int) (i % K);
-+ const int row = (int) (i / K);
-+ const float scale = __half2float(d[(size_t) row * (K / 32) + col / 32]);
-+ const char2 q = *(const char2 *) (qs + i);
-+ *(half2 *) (out + i) = __floats2half2_rn((float) q.x * scale, (float) q.y * scale);
-+}
-+
-+static __global__ void skq8_add_bias_f32(
-+ float * __restrict__ dst, const float * __restrict__ bias,
-+ const int M, const int64_t count) {
-+ const int64_t i = (int64_t) blockIdx.x * blockDim.x + threadIdx.x;
-+ if (i < count) {
-+ dst[i] += bias[i % M];
-+ }
-+}
-+#endif
-+
- // ---------------------------------------------------------------------------
- // Quantize F32 activations [K, N] (col-contiguous) into a zero-padded
- // SKQ8_NPAD-column buffer: int8[NPAD][K] + d float[NPAD][K/32].
-@@ -345,6 +376,9 @@ namespace {
- struct skq8_planes {
- int8_t * qs;
- half * d;
-+#if defined(GGML_CUDA_SKINNY_Q8_CUBLAS_F16)
-+ half * f16;
-+#endif
- };
- std::unordered_map g_skq8_cache;
- std::mutex g_skq8_mutex;
-@@ -454,6 +488,18 @@ static void skq8_run(
- const int64_t KB = K / 32;
-
- cudaStream_t stream = ctx.stream();
-+#if defined(GGML_CUDA_SKINNY_Q8_CUBLAS_F16)
-+ static const bool cublas_f16_enabled = []() {
-+ const char * e = getenv("GGML_SKINNY_Q8_CUBLAS_F16");
-+ return e != nullptr && e[0] != '0';
-+ }();
-+ static const int cublas_f16_min_n = []() {
-+ const char * e = getenv("GGML_SKINNY_Q8_CUBLAS_F16_MIN_N");
-+ const int value = e != nullptr ? atoi(e) : 128;
-+ return value > 0 ? value : 1;
-+ }();
-+ const bool use_cublas_f16 = cublas_f16_enabled && N >= cublas_f16_min_n;
-+#endif
-
- // Repacked weight planes (create on first use). Default: IN PLACE. The plane
- // layout (qs M*K + d M*KB*2) is byte-for-byte the same total size as
-@@ -470,7 +516,7 @@ static void skq8_run(
- // though the repacked bytes are correct. Callers that share the process with
- // such a scheduler set GGML_SKINNY_Q8_INPLACE=0 (the NMT pipeline does this
- // when enabled). The streaming-ASR encoder runtime has no such hazard.
-- skq8_planes planes;
-+ skq8_planes planes = {};
- {
- std::lock_guard lock(g_skq8_mutex);
- auto it = g_skq8_cache.find(src0->data);
-@@ -524,7 +570,54 @@ static void skq8_run(
- }
- }
- }
-+#if defined(GGML_CUDA_SKINNY_Q8_CUBLAS_F16)
-+ if (use_cublas_f16 && planes.f16 == nullptr) {
-+ const size_t elements = (size_t) M * K;
-+ CUDA_CHECK(cudaMalloc(&planes.f16, elements * sizeof(half)));
-+ const int blocks = (int) (((elements + 1) / 2 + 255) / 256);
-+ skq8_dequantize_weight_f16<<>>(
-+ planes.qs, planes.d, planes.f16, (int) M, (int) K);
-+ g_skq8_cache.at(src0->data).f16 = planes.f16;
-+ static const bool memstats = getenv("NEMO_SPEECH_MEMSTATS") != nullptr;
-+ if (memstats) {
-+ fprintf(stderr,
-+ "[memstats] skinny-q8 F16 cache +%.1f MB (%s)\n",
-+ elements * sizeof(half) / 1048576.0, src0->name);
-+ }
-+ }
-+#endif
-+ }
-+
-+#if defined(GGML_CUDA_SKINNY_Q8_CUBLAS_F16)
-+ // Keep the model artifact Q8, expand immutable weights once, convert only
-+ // the live activation, and retain FP32 accumulation/output.
-+ if (use_cublas_f16) {
-+ GGML_ASSERT(planes.f16 != nullptr);
-+ ggml_cuda_pool_alloc a_f16(ctx.pool(), (size_t) N * K);
-+ const to_fp16_cuda_t to_fp16 = ggml_get_to_fp16_cuda(GGML_TYPE_F32);
-+ GGML_ASSERT(to_fp16 != nullptr);
-+ to_fp16(src1->data, a_f16.get(), N * K, stream);
-+
-+ const float alpha = 1.0f;
-+ const float beta = 0.0f;
-+ cublasHandle_t handle = ctx.cublas_handle();
-+ CUBLAS_CHECK(cublasSetStream(handle, stream));
-+ CUBLAS_CHECK(cublasGemmEx(
-+ handle, CUBLAS_OP_T, CUBLAS_OP_N,
-+ (int) M, (int) N, (int) K,
-+ &alpha, planes.f16, CUDA_R_16F, (int) K,
-+ a_f16.get(), CUDA_R_16F, (int) K,
-+ &beta, dst->data, CUDA_R_32F, (int) M,
-+ CUBLAS_COMPUTE_32F, CUBLAS_GEMM_DEFAULT_TENSOR_OP));
-+ if (bias != nullptr) {
-+ const int64_t count = M * N;
-+ skq8_add_bias_f32<<<(count + 255) / 256, 256, 0, stream>>>(
-+ (float *) dst->data, (const float *) bias->data, (int) M, count);
-+ }
-+ return;
- }
-+#endif
-+
- // Quantize activations into the zero-padded buffer (all columns, once —
- // ntot may exceed NPAD; the GEMM below tiles over 64-col chunks).
- const int ntot = (int) ((N + SKQ8_NPAD - 1) / SKQ8_NPAD * SKQ8_NPAD);
diff --git a/ggml-patches/0015-cuda-ctc-batch-fusions.patch b/ggml-patches/0015-cuda-ctc-batch-fusions.patch
deleted file mode 100644
index 1f33287..0000000
--- a/ggml-patches/0015-cuda-ctc-batch-fusions.patch
+++ /dev/null
@@ -1,752 +0,0 @@
-diff --git a/src/ggml-cuda/ggml-cuda.cu b/src/ggml-cuda/ggml-cuda.cu
-index 873ac0e..b92b14f 100644
---- a/src/ggml-cuda/ggml-cuda.cu
-+++ b/src/ggml-cuda/ggml-cuda.cu
-@@ -4183,6 +4183,111 @@ static int ggml_cuda_try_fuse(ggml_backend_cuda_context * cuda_ctx, ggml_cgraph
- return n_fused;
- }
-
-+ const std::initializer_list batch_norm_ops = {
-+ GGML_OP_SUB, GGML_OP_ADD, GGML_OP_SQRT, GGML_OP_DIV, GGML_OP_MUL, GGML_OP_ADD
-+ };
-+ const bool batch_norm_ops_match =
-+ i + (int) batch_norm_ops.size() <= cgraph->n_nodes &&
-+ std::equal(
-+ batch_norm_ops.begin(), batch_norm_ops.end(), cgraph->nodes + i,
-+ [](ggml_op op, const ggml_tensor * tensor) { return op == tensor->op; });
-+ const bool batch_norm_subgraph =
-+ batch_norm_ops_match &&
-+ ggml_can_fuse_subgraph(cgraph, i, batch_norm_ops, { i + 5 });
-+ if (batch_norm_subgraph) {
-+ ggml_tensor * sub = cgraph->nodes[i];
-+ ggml_tensor * var_add = cgraph->nodes[i + 1];
-+ ggml_tensor * sqrt = cgraph->nodes[i + 2];
-+ ggml_tensor * div = cgraph->nodes[i + 3];
-+ ggml_tensor * mul = cgraph->nodes[i + 4];
-+ ggml_tensor * out = cgraph->nodes[i + 5];
-+
-+ const ggml_tensor * input = sub->src[0];
-+ const ggml_tensor * mean = sub->src[1];
-+ const ggml_tensor * variance = nullptr;
-+ const ggml_tensor * epsilon = nullptr;
-+ if (ggml_nelements(var_add->src[0]) == 1) {
-+ epsilon = var_add->src[0];
-+ variance = var_add->src[1];
-+ } else if (ggml_nelements(var_add->src[1]) == 1) {
-+ variance = var_add->src[0];
-+ epsilon = var_add->src[1];
-+ }
-+
-+ const ggml_tensor * weight =
-+ mul->src[0] == div ? mul->src[1] :
-+ mul->src[1] == div ? mul->src[0] : nullptr;
-+ const ggml_tensor * bias =
-+ out->src[0] == mul ? out->src[1] :
-+ out->src[1] == mul ? out->src[0] : nullptr;
-+
-+ const int64_t channels = input->ne[1];
-+ const bool graph_ok =
-+ sqrt->src[0] == var_add &&
-+ div->src[0] == sub && div->src[1] == sqrt;
-+ const bool params_ok =
-+ variance != nullptr && epsilon != nullptr && weight != nullptr && bias != nullptr &&
-+ ggml_nelements(mean) == channels &&
-+ ggml_nelements(variance) == channels &&
-+ ggml_nelements(weight) == channels &&
-+ ggml_nelements(bias) == channels;
-+ const bool types_ok =
-+ input->type == GGML_TYPE_F32 && mean->type == GGML_TYPE_F32 &&
-+ variance != nullptr && variance->type == GGML_TYPE_F32 &&
-+ epsilon != nullptr && epsilon->type == GGML_TYPE_F32 &&
-+ weight != nullptr && weight->type == GGML_TYPE_F32 &&
-+ bias != nullptr && bias->type == GGML_TYPE_F32 &&
-+ sub->type == GGML_TYPE_F32 && var_add->type == GGML_TYPE_F32 &&
-+ sqrt->type == GGML_TYPE_F32 && div->type == GGML_TYPE_F32 &&
-+ mul->type == GGML_TYPE_F32 && out->type == GGML_TYPE_F32;
-+ const bool layout_ok =
-+ ggml_is_contiguous(input) && ggml_is_contiguous(mean) &&
-+ variance != nullptr && ggml_is_contiguous(variance) &&
-+ epsilon != nullptr && ggml_is_contiguous(epsilon) &&
-+ weight != nullptr && ggml_is_contiguous(weight) &&
-+ bias != nullptr && ggml_is_contiguous(bias) &&
-+ ggml_is_contiguous(out) && ggml_are_same_shape(input, out);
-+ const int out_node = i + 5;
-+
-+ const bool common_eligible =
-+ graph_ok && params_ok && types_ok && layout_ok;
-+ if (common_eligible &&
-+ ggml_can_fuse_subgraph(
-+ cgraph, i,
-+ { GGML_OP_SUB, GGML_OP_ADD, GGML_OP_SQRT, GGML_OP_DIV, GGML_OP_MUL,
-+ GGML_OP_ADD, GGML_OP_PERMUTE, GGML_OP_CONT, GGML_OP_UNARY },
-+ { i + 8 })) {
-+ ggml_tensor * permute = cgraph->nodes[i + 6];
-+ ggml_tensor * cont = cgraph->nodes[i + 7];
-+ ggml_tensor * silu = cgraph->nodes[i + 8];
-+ const bool transpose_ok =
-+ permute->src[0] == out && cont->src[0] == permute && silu->src[0] == cont &&
-+ ggml_get_unary_op(silu) == GGML_UNARY_OP_SILU &&
-+ permute->type == GGML_TYPE_F32 && cont->type == GGML_TYPE_F32 &&
-+ silu->type == GGML_TYPE_F32 &&
-+ permute->ne[0] == input->ne[1] && permute->ne[1] == input->ne[0] &&
-+ permute->ne[2] == input->ne[2] && permute->ne[3] == input->ne[3] &&
-+ permute->nb[0] == out->nb[1] && permute->nb[1] == out->nb[0] &&
-+ permute->nb[2] == out->nb[2] && permute->nb[3] == out->nb[3] &&
-+ ggml_is_contiguous(cont) && ggml_is_contiguous(silu) &&
-+ ggml_are_same_shape(cont, silu);
-+ const int transpose_out_node = i + 8;
-+ if (transpose_ok &&
-+ ggml_cuda_check_fusion_memory_ranges(
-+ cgraph, i, 9, &transpose_out_node, 1)) {
-+ ggml_cuda_op_batch_norm_silu_transpose_fused(
-+ *cuda_ctx, input, mean, variance, epsilon, weight, bias, silu);
-+ return 8;
-+ }
-+ }
-+ if (common_eligible &&
-+ ggml_cuda_check_fusion_memory_ranges(cgraph, i, 6, &out_node, 1)) {
-+ ggml_cuda_op_batch_norm_fused(
-+ *cuda_ctx, input, mean, variance, epsilon, weight, bias, out);
-+ return 5;
-+ }
-+ }
-+
- //topk-moe
- if (cgraph->nodes[i]->op == GGML_OP_UNARY || cgraph->nodes[i]->op == GGML_OP_SOFT_MAX ||
- cgraph->nodes[i]->op == GGML_OP_ARGSORT) {
-@@ -4566,21 +4671,23 @@ static int ggml_cuda_try_fuse(ggml_backend_cuda_context * cuda_ctx, ggml_cgraph
- fused_mul_mat_vec = false;
- fused_node_count = 0;
-
-- // A BF16 projection immediately following SiLU would otherwise launch an
-- // F32 SiLU kernel and then a separate F32-to-BF16 input conversion. Emit
-- // the same rounded BF16 activation directly in one pass.
-+ // A 16-bit projection immediately following SiLU would otherwise launch an
-+ // F32 SiLU kernel and then a separate input conversion.
- if (ggml_can_fuse(cgraph, i, { GGML_OP_UNARY, GGML_OP_CPY })) {
- const ggml_tensor * silu_node = cgraph->nodes[i];
- ggml_tensor * cast_node = cgraph->nodes[i + 1];
-+ const bool cast_supported =
-+ cast_node->type == GGML_TYPE_F16 ||
-+ (cast_node->type == GGML_TYPE_BF16 && native_bf16);
- const bool eligible =
-- native_bf16 && ggml_get_unary_op(silu_node) == GGML_UNARY_OP_SILU &&
-+ cast_supported && ggml_get_unary_op(silu_node) == GGML_UNARY_OP_SILU &&
- cast_node->src[0] == silu_node &&
- silu_node->src[0]->type == GGML_TYPE_F32 &&
-- silu_node->type == GGML_TYPE_F32 && cast_node->type == GGML_TYPE_BF16 &&
-+ silu_node->type == GGML_TYPE_F32 &&
- ggml_are_same_shape(silu_node->src[0], cast_node) &&
- ggml_is_contiguous(silu_node->src[0]) && ggml_is_contiguous(cast_node);
- if (eligible) {
-- ggml_cuda_op_silu_f32_to_bf16(*cuda_ctx, silu_node, cast_node);
-+ ggml_cuda_op_silu_f32_to_16(*cuda_ctx, silu_node, cast_node);
- return 1;
- }
- }
-@@ -4626,6 +4733,31 @@ static int ggml_cuda_try_fuse(ggml_backend_cuda_context * cuda_ctx, ggml_cgraph
- }
- }
-
-+ // Fold a following residual into the same cuBLASLt projection epilogue.
-+ if (ggml_can_fuse(cgraph, i, { GGML_OP_MUL_MAT, GGML_OP_ADD, GGML_OP_ADD })) {
-+ ggml_tensor * mm_node = cgraph->nodes[i];
-+ ggml_tensor * bias_node = cgraph->nodes[i + 1];
-+ ggml_tensor * residual_add = cgraph->nodes[i + 2];
-+ const ggml_tensor * bias =
-+ bias_node->src[0] == mm_node ? bias_node->src[1] :
-+ bias_node->src[1] == mm_node ? bias_node->src[0] : nullptr;
-+ const ggml_tensor * residual =
-+ residual_add->src[0] == bias_node ? residual_add->src[1] :
-+ residual_add->src[1] == bias_node ? residual_add->src[0] : nullptr;
-+ if (bias != nullptr && residual != nullptr &&
-+ bias->type == GGML_TYPE_F32 && residual->type == GGML_TYPE_F32 &&
-+ ggml_is_contiguous(bias) && ggml_is_contiguous(residual) &&
-+ ggml_nelements(bias) == mm_node->ne[0] &&
-+ ggml_are_same_shape(mm_node, residual) &&
-+ ggml_is_contiguous(residual_add) &&
-+ ggml_cuda_skinny_q8_residual_supported(
-+ mm_node->src[0], mm_node->src[1], mm_node)) {
-+ ggml_cuda_mul_mat_skinny_q8_bias_residual(
-+ *cuda_ctx, mm_node->src[0], mm_node->src[1], bias, residual, residual_add);
-+ return 2;
-+ }
-+ }
-+
- // skinny-q8 GEMM + row-vector bias: the upstream MUL_MAT+ADD fusion below
- // requires a same-shape add, so the classic broadcast Linear bias
- // ([M] over [M,N]) never qualifies — fold it into the skinny GEMM's
-@@ -4713,7 +4845,7 @@ static int ggml_cuda_try_fuse(ggml_backend_cuda_context * cuda_ctx, ggml_cgraph
- return fused_node_count - 1;
- }
-
-- // LayerNorm affine followed by a BF16 projection. Store the affine result
-+ // LayerNorm affine followed by a 16-bit projection. Store the affine result
- // directly in the projection's input precision instead of writing F32 and
- // launching a second full-tensor conversion.
- if (ggml_can_fuse(cgraph, i,
-@@ -4726,10 +4858,13 @@ static int ggml_cuda_try_fuse(ggml_backend_cuda_context * cuda_ctx, ggml_cgraph
- mul->src[0] == norm ? mul->src[1] : mul->src[0];
- const ggml_tensor * beta =
- add->src[0] == mul ? add->src[1] : add->src[0];
-+ const bool cast_supported =
-+ cast->type == GGML_TYPE_F16 ||
-+ (cast->type == GGML_TYPE_BF16 && native_bf16);
- const bool eligible =
-- native_bf16 && cast->src[0] == add && norm->src[0]->type == GGML_TYPE_F32 &&
-+ cast_supported && cast->src[0] == add && norm->src[0]->type == GGML_TYPE_F32 &&
- norm->type == GGML_TYPE_F32 && mul->type == GGML_TYPE_F32 &&
-- add->type == GGML_TYPE_F32 && cast->type == GGML_TYPE_BF16 &&
-+ add->type == GGML_TYPE_F32 &&
- gamma->type == GGML_TYPE_F32 && beta->type == GGML_TYPE_F32 &&
- ggml_nelements(gamma) == norm->ne[0] &&
- ggml_nelements(beta) == norm->ne[0] &&
-@@ -5690,7 +5825,9 @@ static bool ggml_backend_cuda_device_supports_op(ggml_backend_dev_t dev, const g
- return false;
- }
- }
-- if (b->type == GGML_TYPE_F16 && a->type != GGML_TYPE_F16) {
-+ if (b->type == GGML_TYPE_F16 && a->type != GGML_TYPE_F16 &&
-+ !(op->op == GGML_OP_MUL_MAT &&
-+ ggml_cuda_skinny_q8_supported(a, b, op))) {
- return false;
- }
- #ifdef GGML_USE_MUSA
-diff --git a/src/ggml-cuda/norm.cu b/src/ggml-cuda/norm.cu
-index 555c951..e30c025 100644
---- a/src/ggml-cuda/norm.cu
-+++ b/src/ggml-cuda/norm.cu
-@@ -1,4 +1,5 @@
- #include "norm.cuh"
-+#include "unary.cuh"
- #include
-
- template
-@@ -81,6 +82,61 @@ static __global__ void norm_mul_add_f32(
- }
- }
-
-+static __global__ void batch_norm_f32(
-+ const float * x, const float * mean, const float * variance, const float * epsilon,
-+ const float * weight, const float * bias, float * dst, int64_t nelements,
-+ int64_t ntime, int64_t nchannels) {
-+ const int64_t i = int64_t(blockIdx.x) * blockDim.x + threadIdx.x;
-+ if (i >= nelements) {
-+ return;
-+ }
-+
-+ const int64_t channel = (i / ntime) % nchannels;
-+ const float centered = __fsub_rn(x[i], mean[channel]);
-+ const float denominator = sqrtf(__fadd_rn(variance[channel], epsilon[0]));
-+ const float normalized = __fdiv_rn(centered, denominator);
-+ dst[i] = __fadd_rn(__fmul_rn(normalized, weight[channel]), bias[channel]);
-+}
-+
-+static __global__ void batch_norm_silu_transpose_f32(
-+ const float * x, const float * mean, const float * variance, const float * epsilon,
-+ const float * weight, const float * bias, float * dst, int64_t nelements,
-+ int64_t ntime, int64_t nchannels) {
-+ const int64_t i = int64_t(blockIdx.x) * blockDim.x + threadIdx.x;
-+ if (i >= nelements) {
-+ return;
-+ }
-+
-+ const int64_t channel = (i / ntime) % nchannels;
-+ const int64_t time = i % ntime;
-+ const int64_t sample = i / (ntime * nchannels);
-+ const float centered = __fsub_rn(x[i], mean[channel]);
-+ const float denominator = sqrtf(__fadd_rn(variance[channel], epsilon[0]));
-+ const float normalized = __fdiv_rn(centered, denominator);
-+ const float affine = __fadd_rn(__fmul_rn(normalized, weight[channel]), bias[channel]);
-+ dst[channel + nchannels * (time + ntime * sample)] = ggml_cuda_op_silu_single(affine);
-+}
-+
-+static void batch_norm_f32_cuda(
-+ const float * x, const float * mean, const float * variance, const float * epsilon,
-+ const float * weight, const float * bias, float * dst, int64_t nelements,
-+ int64_t ntime, int64_t nchannels, cudaStream_t stream) {
-+ constexpr int block_size = 256;
-+ const int64_t block_count = (nelements + block_size - 1) / block_size;
-+ batch_norm_f32<<>>(
-+ x, mean, variance, epsilon, weight, bias, dst, nelements, ntime, nchannels);
-+}
-+
-+static void batch_norm_silu_transpose_f32_cuda(
-+ const float * x, const float * mean, const float * variance, const float * epsilon,
-+ const float * weight, const float * bias, float * dst, int64_t nelements,
-+ int64_t ntime, int64_t nchannels, cudaStream_t stream) {
-+ constexpr int block_size = 256;
-+ const int64_t block_count = (nelements + block_size - 1) / block_size;
-+ batch_norm_silu_transpose_f32<<>>(
-+ x, mean, variance, epsilon, weight, bias, dst, nelements, ntime, nchannels);
-+}
-+
- template
- static __global__ void group_norm_f32(const float * x, float * dst, const int group_size, const int ne_elements, const float eps) {
- // blockIdx.x: num_groups idx
-@@ -334,7 +390,18 @@ static void norm_mul_add_cuda(
- const int64_t stride_sample, const float eps, cudaStream_t stream) {
- const dim3 blocks_num(nrows, nchannels, nsamples);
- const int cc = ggml_cuda_info().devices[ggml_cuda_get_device()].cc;
-- if (ncols < 1024) {
-+ if (ncols == 768 && WARP_SIZE == 32) {
-+ constexpr int block_size = 384;
-+ if (add) {
-+ norm_mul_add_f32
-+ <<>>(
-+ x, mul, add, dst, ncols, stride_row, stride_channel, stride_sample, eps);
-+ } else {
-+ norm_mul_add_f32
-+ <<>>(
-+ x, mul, add, dst, ncols, stride_row, stride_channel, stride_sample, eps);
-+ }
-+ } else if (ncols < 1024) {
- const dim3 block_dims(WARP_SIZE, 1, 1);
- if (add) {
- norm_mul_add_f32<<>>(
-@@ -343,10 +410,8 @@ static void norm_mul_add_cuda(
- norm_mul_add_f32<<>>(
- x, mul, add, dst, ncols, stride_row, stride_channel, stride_sample, eps);
- }
-- } else if (ncols == 1024 && GGML_CUDA_CC_IS_NVIDIA(cc) && cc >= GGML_CUDA_CC_HOPPER) {
-- // Hopper and newer: four elements per thread keeps high occupancy
-- // while reducing the affine LayerNorm reduction from 32 warps/block
-- // to 8. Retain the established 1024-thread path on older devices.
-+ } else if (ncols == 1024 && GGML_CUDA_CC_IS_NVIDIA(cc) && cc >= GGML_CUDA_CC_AMPERE) {
-+ // Four elements per thread balances reduction work and occupancy.
- const dim3 block_dims(256, 1, 1);
- if (add) {
- norm_mul_add_f32<256, true, T><<>>(
-@@ -519,7 +584,7 @@ void ggml_cuda_op_norm(ggml_backend_cuda_context & ctx, ggml_tensor * dst) {
- // or add_tensor), mirroring ggml_cuda_op_rms_norm_fused(_add). Eligibility
- // (row-vector operands, F32, contiguity) is enforced in ggml_cuda_can_fuse.
- void ggml_cuda_op_norm_fused(ggml_backend_cuda_context & ctx, ggml_tensor * dst, ggml_tensor * mul_tensor,
-- ggml_tensor * add_tensor, ggml_tensor * bf16_dst) {
-+ ggml_tensor * add_tensor, ggml_tensor * cast_dst) {
- const ggml_tensor * norm_src = (ggml_tensor *) dst->src[0];
- float eps = 0.0f;
- memcpy(&eps, dst->op_params, sizeof(float));
-@@ -563,13 +628,19 @@ void ggml_cuda_op_norm_fused(ggml_backend_cuda_context & ctx, ggml_tensor * dst,
- const int64_t s02 = norm_src->nb[2] / ts0;
- const int64_t s03 = norm_src->nb[3] / ts0;
-
-- if (bf16_dst != nullptr) {
-- GGML_ASSERT(bf16_dst->type == GGML_TYPE_BF16);
-- GGML_ASSERT(ggml_is_contiguous(bf16_dst));
-- GGML_ASSERT(ggml_are_same_shape(dst, bf16_dst));
-- norm_mul_add_cuda(
-- src0_d, mul_d, add_d, (nv_bfloat16 *) bf16_dst->data,
-- ne00, ne01, ne02, ne03, s01, s02, s03, eps, ctx.stream());
-+ if (cast_dst != nullptr) {
-+ GGML_ASSERT(cast_dst->type == GGML_TYPE_BF16 || cast_dst->type == GGML_TYPE_F16);
-+ GGML_ASSERT(ggml_is_contiguous(cast_dst));
-+ GGML_ASSERT(ggml_are_same_shape(dst, cast_dst));
-+ if (cast_dst->type == GGML_TYPE_BF16) {
-+ norm_mul_add_cuda(
-+ src0_d, mul_d, add_d, (nv_bfloat16 *) cast_dst->data,
-+ ne00, ne01, ne02, ne03, s01, s02, s03, eps, ctx.stream());
-+ } else {
-+ norm_mul_add_cuda(
-+ src0_d, mul_d, add_d, (half *) cast_dst->data,
-+ ne00, ne01, ne02, ne03, s01, s02, s03, eps, ctx.stream());
-+ }
- } else {
- norm_mul_add_cuda(
- src0_d, mul_d, add_d, dst_d,
-@@ -577,6 +648,50 @@ void ggml_cuda_op_norm_fused(ggml_backend_cuda_context & ctx, ggml_tensor * dst,
- }
- }
-
-+void ggml_cuda_op_batch_norm_fused(ggml_backend_cuda_context & ctx,
-+ const ggml_tensor * input,
-+ const ggml_tensor * mean,
-+ const ggml_tensor * variance,
-+ const ggml_tensor * epsilon,
-+ const ggml_tensor * weight,
-+ const ggml_tensor * bias,
-+ ggml_tensor * dst) {
-+ batch_norm_f32_cuda(
-+ (const float *) input->data,
-+ (const float *) mean->data,
-+ (const float *) variance->data,
-+ (const float *) epsilon->data,
-+ (const float *) weight->data,
-+ (const float *) bias->data,
-+ (float *) dst->data,
-+ ggml_nelements(input),
-+ input->ne[0],
-+ input->ne[1],
-+ ctx.stream());
-+}
-+
-+void ggml_cuda_op_batch_norm_silu_transpose_fused(ggml_backend_cuda_context & ctx,
-+ const ggml_tensor * input,
-+ const ggml_tensor * mean,
-+ const ggml_tensor * variance,
-+ const ggml_tensor * epsilon,
-+ const ggml_tensor * weight,
-+ const ggml_tensor * bias,
-+ ggml_tensor * dst) {
-+ batch_norm_silu_transpose_f32_cuda(
-+ (const float *) input->data,
-+ (const float *) mean->data,
-+ (const float *) variance->data,
-+ (const float *) epsilon->data,
-+ (const float *) weight->data,
-+ (const float *) bias->data,
-+ (float *) dst->data,
-+ ggml_nelements(input),
-+ input->ne[0],
-+ input->ne[1],
-+ ctx.stream());
-+}
-+
- void ggml_cuda_op_group_norm(ggml_backend_cuda_context & ctx, ggml_tensor * dst) {
- const ggml_tensor * src0 = dst->src[0];
- const float * src0_d = (const float *)src0->data;
-diff --git a/src/ggml-cuda/norm.cuh b/src/ggml-cuda/norm.cuh
-index 8f4287c..f580ae9 100644
---- a/src/ggml-cuda/norm.cuh
-+++ b/src/ggml-cuda/norm.cuh
-@@ -5,7 +5,25 @@ void ggml_cuda_op_norm(ggml_backend_cuda_context & ctx, ggml_tensor * dst);
- // Fused LayerNorm + row-vector mul (gamma) + optional row-vector add (beta).
- // add_tensor may be nullptr (norm+mul only).
- void ggml_cuda_op_norm_fused(ggml_backend_cuda_context & ctx, ggml_tensor * dst, ggml_tensor * mul_tensor,
-- ggml_tensor * add_tensor, ggml_tensor * bf16_dst = nullptr);
-+ ggml_tensor * add_tensor, ggml_tensor * cast_dst = nullptr);
-+
-+void ggml_cuda_op_batch_norm_fused(ggml_backend_cuda_context & ctx,
-+ const ggml_tensor * input,
-+ const ggml_tensor * mean,
-+ const ggml_tensor * variance,
-+ const ggml_tensor * epsilon,
-+ const ggml_tensor * weight,
-+ const ggml_tensor * bias,
-+ ggml_tensor * dst);
-+
-+void ggml_cuda_op_batch_norm_silu_transpose_fused(ggml_backend_cuda_context & ctx,
-+ const ggml_tensor * input,
-+ const ggml_tensor * mean,
-+ const ggml_tensor * variance,
-+ const ggml_tensor * epsilon,
-+ const ggml_tensor * weight,
-+ const ggml_tensor * bias,
-+ ggml_tensor * dst);
-
- void ggml_cuda_op_group_norm(ggml_backend_cuda_context & ctx, ggml_tensor * dst);
-
-diff --git a/src/ggml-cuda/skinny-q8.cu b/src/ggml-cuda/skinny-q8.cu
-index fe329bb..1122e05 100644
---- a/src/ggml-cuda/skinny-q8.cu
-+++ b/src/ggml-cuda/skinny-q8.cu
-@@ -94,14 +94,19 @@ static __global__ void skq8_dequantize_weight_f16(
- *(half2 *) (out + i) = __floats2half2_rn((float) q.x * scale, (float) q.y * scale);
- }
-
--static __global__ void skq8_add_bias_f32(
-- float * __restrict__ dst, const float * __restrict__ bias,
-- const int M, const int64_t count) {
-- const int64_t i = (int64_t) blockIdx.x * blockDim.x + threadIdx.x;
-- if (i < count) {
-- dst[i] += bias[i % M];
-- }
-+static bool skq8_cublas_f16_enabled_for_n(int64_t n) {
-+ static const bool enabled = []() {
-+ const char * e = getenv("GGML_SKINNY_Q8_CUBLAS_F16");
-+ return e != nullptr && e[0] != '0';
-+ }();
-+ static const int min_n = []() {
-+ const char * e = getenv("GGML_SKINNY_Q8_CUBLAS_F16_MIN_N");
-+ const int value = e != nullptr ? atoi(e) : 128;
-+ return value > 0 ? value : 1;
-+ }();
-+ return enabled && n >= min_n;
- }
-+
- #endif
-
- // ---------------------------------------------------------------------------
-@@ -388,6 +393,25 @@ static __global__ void skq8_reduce_splitk(
- dst[i] = sum;
- }
-
-+#if defined(GGML_CUDA_SKINNY_Q8_CUBLAS_F16)
-+static __global__ void skq8_add_epilogue_f32(
-+ float * __restrict__ dst, const float * __restrict__ bias,
-+ const float * __restrict__ residual, const int m, const int64_t count) {
-+ const int64_t i = (int64_t) blockIdx.x * blockDim.x + threadIdx.x;
-+ if (i >= count) {
-+ return;
-+ }
-+ float value = dst[i];
-+ if (residual != nullptr) {
-+ value += residual[i];
-+ }
-+ if (bias != nullptr) {
-+ value += bias[i % m];
-+ }
-+ dst[i] = value;
-+}
-+#endif
-+
- // Repacked-weight cache. Keyed by the weight tensor's device pointer; entries
- // live for the process lifetime (model weights are loaded once). Repacking
- // happens on first (eager/warmup) use, before any CUDA-graph capture.
-@@ -453,10 +477,18 @@ bool ggml_cuda_skinny_q8_supported(
- // all outer columns as one logical N so a weight repacked by a scalar
- // graph remains usable by a later true-batch graph.
- const int64_t total_n = ggml_nelements(src1) / src1->ne[0];
-- const bool call_ok = src1->type == GGML_TYPE_F32 && dst->type == GGML_TYPE_F32 &&
-- ggml_is_contiguous(src1) && ggml_is_contiguous(dst) &&
-- src1->ne[0] == src0->ne[0] &&
-- ggml_nelements(dst) == src0->ne[1] * total_n;
-+ const bool f16_input_ok =
-+#if defined(GGML_CUDA_SKINNY_Q8_CUBLAS_F16)
-+ src1->type == GGML_TYPE_F16 && skq8_cublas_f16_enabled_for_n(total_n);
-+#else
-+ false;
-+#endif
-+ const bool call_ok =
-+ (src1->type == GGML_TYPE_F32 || f16_input_ok) &&
-+ dst->type == GGML_TYPE_F32 &&
-+ ggml_is_contiguous(src1) && ggml_is_contiguous(dst) &&
-+ src1->ne[0] == src0->ne[0] &&
-+ ggml_nelements(dst) == src0->ne[1] * total_n;
-
- if (planar_q8) {
- // N<=8 is handled by planar MMVQ. Wider calls must stay on this path
-@@ -479,13 +511,9 @@ bool ggml_cuda_skinny_q8_supported(
- GGML_ASSERT(call_ok && "skinny-q8: repacked weight used in an unsupported mul_mat shape");
- return true;
- }
-- // By default, select only on logical width so outer batch size cannot
-- // change the accumulation path. The opt-in mode is an end-to-end ASR
-- // experiment for streaming shapes such as [K,2,B]: it flattens the dense
-- // outer batch and lets the tensor-core kernel tile N beyond 64. It is not
-- // the default because skinny-Q8 has a different accumulation order from
-- // MMVQ; callers must validate transcript/accuracy parity. Use a separate
-- // repack allocation (GGML_SKINNY_Q8_INPLACE=0) with multi-stream schedulers.
-+ // Keep the accumulation path independent of outer batch size by default.
-+ // The opt-in mode flattens [K,N,B...] for the tensor-core kernel; use a
-+ // separate repack allocation with multi-stream schedulers.
- static const bool outer_batch_dispatch = []() {
- const char * e = getenv("GGML_SKINNY_Q8_OUTER_BATCH");
- return e != nullptr && e[0] != '0';
-@@ -498,9 +526,21 @@ bool ggml_cuda_skinny_q8_supported(
- return eligible;
- }
-
-+bool ggml_cuda_skinny_q8_residual_supported(
-+ const ggml_tensor * src0, const ggml_tensor * src1, const ggml_tensor * dst) {
-+#if defined(GGML_CUDA_SKINNY_Q8_CUBLAS_F16)
-+ const int64_t total_n = ggml_nelements(src1) / src1->ne[0];
-+ return ggml_cuda_skinny_q8_supported(src0, src1, dst) &&
-+ skq8_cublas_f16_enabled_for_n(total_n);
-+#else
-+ GGML_UNUSED_VARS(src0, src1, dst);
-+ return false;
-+#endif
-+}
-+
- static void skq8_run(
- ggml_backend_cuda_context & ctx, const ggml_tensor * src0, const ggml_tensor * src1,
-- const ggml_tensor * bias, ggml_tensor * dst) {
-+ const ggml_tensor * bias, const ggml_tensor * residual, ggml_tensor * dst) {
- const int64_t M = src0->ne[1];
- const int64_t K = src0->ne[0];
- const int64_t N = ggml_nelements(src1) / src1->ne[0];
-@@ -508,16 +548,7 @@ static void skq8_run(
-
- cudaStream_t stream = ctx.stream();
- #if defined(GGML_CUDA_SKINNY_Q8_CUBLAS_F16)
-- static const bool cublas_f16_enabled = []() {
-- const char * e = getenv("GGML_SKINNY_Q8_CUBLAS_F16");
-- return e != nullptr && e[0] != '0';
-- }();
-- static const int cublas_f16_min_n = []() {
-- const char * e = getenv("GGML_SKINNY_Q8_CUBLAS_F16_MIN_N");
-- const int value = e != nullptr ? atoi(e) : 128;
-- return value > 0 ? value : 1;
-- }();
-- const bool use_cublas_f16 = cublas_f16_enabled && N >= cublas_f16_min_n;
-+ const bool use_cublas_f16 = skq8_cublas_f16_enabled_for_n(N);
- #endif
-
- // Repacked weight planes (create on first use). Default: IN PLACE. The plane
-@@ -612,29 +643,44 @@ static void skq8_run(
- // the live activation, and retain FP32 accumulation/output.
- if (use_cublas_f16) {
- GGML_ASSERT(planes.f16 != nullptr);
-- ggml_cuda_pool_alloc a_f16(ctx.pool(), (size_t) N * K);
-- const to_fp16_cuda_t to_fp16 = ggml_get_to_fp16_cuda(GGML_TYPE_F32);
-- GGML_ASSERT(to_fp16 != nullptr);
-- to_fp16(src1->data, a_f16.get(), N * K, stream);
--
-- const float alpha = 1.0f;
-- const float beta = 0.0f;
-- cublasHandle_t handle = ctx.cublas_handle();
-- CUBLAS_CHECK(cublasSetStream(handle, stream));
-- CUBLAS_CHECK(cublasGemmEx(
-- handle, CUBLAS_OP_T, CUBLAS_OP_N,
-- (int) M, (int) N, (int) K,
-- &alpha, planes.f16, CUDA_R_16F, (int) K,
-- a_f16.get(), CUDA_R_16F, (int) K,
-- &beta, dst->data, CUDA_R_32F, (int) M,
-- CUBLAS_COMPUTE_32F, CUBLAS_GEMM_DEFAULT_TENSOR_OP));
-- if (bias != nullptr) {
-- const int64_t count = M * N;
-- skq8_add_bias_f32<<<(count + 255) / 256, 256, 0, stream>>>(
-- (float *) dst->data, (const float *) bias->data, (int) M, count);
-+ auto gemm = [&](const half* activation) {
-+ const float alpha = 1.0f;
-+ const bool residual_in_place =
-+ residual != nullptr && residual->data == dst->data;
-+ const float beta = residual_in_place ? 1.0f : 0.0f;
-+ cublasHandle_t handle = ctx.cublas_handle();
-+ CUBLAS_CHECK(cublasSetStream(handle, stream));
-+ CUBLAS_CHECK(cublasGemmEx(
-+ handle, CUBLAS_OP_T, CUBLAS_OP_N,
-+ (int) M, (int) N, (int) K,
-+ &alpha, planes.f16, CUDA_R_16F, (int) K,
-+ activation, CUDA_R_16F, (int) K,
-+ &beta, dst->data, CUDA_R_32F, (int) M,
-+ CUBLAS_COMPUTE_32F, CUBLAS_GEMM_DEFAULT_TENSOR_OP));
-+ if (bias != nullptr || (residual != nullptr && !residual_in_place)) {
-+ const int64_t count = M * N;
-+ skq8_add_epilogue_f32<<<(count + 255) / 256, 256, 0, stream>>>(
-+ (float *) dst->data,
-+ bias != nullptr ? (const float *) bias->data : nullptr,
-+ residual != nullptr && !residual_in_place
-+ ? (const float *) residual->data
-+ : nullptr,
-+ (int) M, count);
-+ }
-+ };
-+ if (src1->type == GGML_TYPE_F16) {
-+ gemm((const half*) src1->data);
-+ } else {
-+ ggml_cuda_pool_alloc a_f16(ctx.pool(), (size_t) N * K);
-+ const to_fp16_cuda_t to_fp16 = ggml_get_to_fp16_cuda(GGML_TYPE_F32);
-+ GGML_ASSERT(to_fp16 != nullptr);
-+ to_fp16(src1->data, a_f16.get(), N * K, stream);
-+ gemm(a_f16.get());
- }
- return;
- }
-+ GGML_ASSERT(src1->type == GGML_TYPE_F32);
-+ GGML_ASSERT(residual == nullptr);
- #endif
-
- // Quantize activations into the zero-padded buffer (all columns, once —
-@@ -691,11 +737,17 @@ static void skq8_run(
- void ggml_cuda_mul_mat_skinny_q8(
- ggml_backend_cuda_context & ctx, const ggml_tensor * src0, const ggml_tensor * src1,
- ggml_tensor * dst) {
-- skq8_run(ctx, src0, src1, nullptr, dst);
-+ skq8_run(ctx, src0, src1, nullptr, nullptr, dst);
- }
-
- void ggml_cuda_mul_mat_skinny_q8_bias(
- ggml_backend_cuda_context & ctx, const ggml_tensor * src0, const ggml_tensor * src1,
- const ggml_tensor * bias, ggml_tensor * dst) {
-- skq8_run(ctx, src0, src1, bias, dst);
-+ skq8_run(ctx, src0, src1, bias, nullptr, dst);
-+}
-+
-+void ggml_cuda_mul_mat_skinny_q8_bias_residual(
-+ ggml_backend_cuda_context & ctx, const ggml_tensor * src0, const ggml_tensor * src1,
-+ const ggml_tensor * bias, const ggml_tensor * residual, ggml_tensor * dst) {
-+ skq8_run(ctx, src0, src1, bias, residual, dst);
- }
-diff --git a/src/ggml-cuda/skinny-q8.cuh b/src/ggml-cuda/skinny-q8.cuh
-index 17175bb..30238fe 100644
---- a/src/ggml-cuda/skinny-q8.cuh
-+++ b/src/ggml-cuda/skinny-q8.cuh
-@@ -7,6 +7,9 @@
- bool ggml_cuda_skinny_q8_supported(
- const ggml_tensor * src0, const ggml_tensor * src1, const ggml_tensor * dst);
-
-+bool ggml_cuda_skinny_q8_residual_supported(
-+ const ggml_tensor * src0, const ggml_tensor * src1, const ggml_tensor * dst);
-+
- void ggml_cuda_mul_mat_skinny_q8(
- ggml_backend_cuda_context & ctx, const ggml_tensor * src0, const ggml_tensor * src1,
- ggml_tensor * dst);
-@@ -16,3 +19,7 @@ void ggml_cuda_mul_mat_skinny_q8(
- void ggml_cuda_mul_mat_skinny_q8_bias(
- ggml_backend_cuda_context & ctx, const ggml_tensor * src0, const ggml_tensor * src1,
- const ggml_tensor * bias, ggml_tensor * dst);
-+
-+void ggml_cuda_mul_mat_skinny_q8_bias_residual(
-+ ggml_backend_cuda_context & ctx, const ggml_tensor * src0, const ggml_tensor * src1,
-+ const ggml_tensor * bias, const ggml_tensor * residual, ggml_tensor * dst);
-diff --git a/src/ggml-cuda/unary.cu b/src/ggml-cuda/unary.cu
-index 4c68411..24b3ff7 100644
---- a/src/ggml-cuda/unary.cu
-+++ b/src/ggml-cuda/unary.cu
-@@ -183,11 +183,20 @@ void ggml_cuda_op_silu(ggml_backend_cuda_context & ctx, ggml_tensor * dst) {
- ggml_cuda_op_unary(ctx, dst);
- }
-
-+static __global__ void silu_f32_to_f16(
-+ const float * __restrict__ x, half * __restrict__ dst,
-+ const int64_t nelements) {
-+ const int64_t i = (int64_t) blockIdx.x * blockDim.x + threadIdx.x;
-+ if (i < nelements) {
-+ dst[i] = __float2half(op_silu(x[i]));
-+ }
-+}
-+
- static __global__ void silu_f32_to_bf16(
- const float * __restrict__ x, nv_bfloat16 * __restrict__ dst,
- const int64_t nelements) {
- #if defined(__CUDA_ARCH__) && __CUDA_ARCH__ >= GGML_CUDA_CC_AMPERE
-- const int64_t i = (int64_t) blockIdx.x * blockDim.x + threadIdx.x;
-+ const int64_t i = int64_t(blockIdx.x) * blockDim.x + threadIdx.x;
- if (i < nelements) {
- dst[i] = __float2bfloat16(op_silu(x[i]));
- }
-@@ -197,21 +206,26 @@ static __global__ void silu_f32_to_bf16(
- #endif
- }
-
--void ggml_cuda_op_silu_f32_to_bf16(
-+void ggml_cuda_op_silu_f32_to_16(
- ggml_backend_cuda_context & ctx, const ggml_tensor * silu_node,
- ggml_tensor * dst) {
- const ggml_tensor * src = silu_node->src[0];
- GGML_ASSERT(src->type == GGML_TYPE_F32);
- GGML_ASSERT(silu_node->type == GGML_TYPE_F32);
-- GGML_ASSERT(dst->type == GGML_TYPE_BF16);
-+ GGML_ASSERT(dst->type == GGML_TYPE_BF16 || dst->type == GGML_TYPE_F16);
- GGML_ASSERT(ggml_are_same_shape(src, dst));
- GGML_ASSERT(ggml_is_contiguous(src));
- GGML_ASSERT(ggml_is_contiguous(dst));
-
- const int64_t nelements = ggml_nelements(src);
- const int64_t num_blocks = (nelements + CUDA_SILU_BLOCK_SIZE - 1) / CUDA_SILU_BLOCK_SIZE;
-- silu_f32_to_bf16<<>>(
-- (const float *) src->data, (nv_bfloat16 *) dst->data, nelements);
-+ if (dst->type == GGML_TYPE_BF16) {
-+ silu_f32_to_bf16<<>>(
-+ (const float *) src->data, (nv_bfloat16 *) dst->data, nelements);
-+ } else {
-+ silu_f32_to_f16<<>>(
-+ (const float *) src->data, (half *) dst->data, nelements);
-+ }
- }
-
- void ggml_cuda_op_tanh(ggml_backend_cuda_context & ctx, ggml_tensor * dst) {
-diff --git a/src/ggml-cuda/unary.cuh b/src/ggml-cuda/unary.cuh
-index 5307327..5bfc74f 100644
---- a/src/ggml-cuda/unary.cuh
-+++ b/src/ggml-cuda/unary.cuh
-@@ -31,7 +31,7 @@ void ggml_cuda_op_gelu(ggml_backend_cuda_context & ctx, ggml_tensor * dst);
-
- void ggml_cuda_op_silu(ggml_backend_cuda_context & ctx, ggml_tensor * dst);
-
--void ggml_cuda_op_silu_f32_to_bf16(
-+void ggml_cuda_op_silu_f32_to_16(
- ggml_backend_cuda_context & ctx, const ggml_tensor * silu_node, ggml_tensor * dst);
-
- void ggml_cuda_op_silu_back(ggml_backend_cuda_context & ctx, ggml_tensor * dst);
diff --git a/ggml-patches/0016-fix-batched-conv1d-layout.patch b/ggml-patches/0016-fix-batched-conv1d-layout.patch
deleted file mode 100644
index b1ded39..0000000
--- a/ggml-patches/0016-fix-batched-conv1d-layout.patch
+++ /dev/null
@@ -1,21 +0,0 @@
-diff --git a/src/ggml.c b/src/ggml.c
---- a/src/ggml.c
-+++ b/src/ggml.c
-@@ -4494,7 +4494,16 @@ struct ggml_tensor * ggml_conv_1d(
- ggml_reshape_2d(ctx, im2col, im2col->ne[0], (im2col->ne[2] * im2col->ne[1])), // [N, OL, IC * K] => [N*OL, IC * K]
- ggml_reshape_2d(ctx, a, (a->ne[0] * a->ne[1]), a->ne[2])); // [OC,IC, K] => [OC, IC * K]
-
-- result = ggml_reshape_3d(ctx, result, im2col->ne[1], a->ne[2], im2col->ne[2]); // [N, OC, OL]
-+ if (im2col->ne[2] == 1) {
-+ result = ggml_reshape_3d(ctx, result, im2col->ne[1], a->ne[2], 1);
-+ } else {
-+ // mul_mat produces [N*OL, OC]. Restore the flattened axes before moving
-+ // the batch axis behind the output channels; a direct [OL, OC, N]
-+ // reshape interleaves OC and N when N > 1.
-+ result =
-+ ggml_reshape_3d(ctx, result, im2col->ne[1], im2col->ne[2], a->ne[2]);
-+ result = ggml_cont(ctx, ggml_permute(ctx, result, 0, 2, 1, 3));
-+ }
-
- return result;
- }
diff --git a/ggml-patches/0018-metal-tensor-api-dynamic-k.patch b/ggml-patches/0018-metal-tensor-api-dynamic-k.patch
deleted file mode 100644
index 0a51dad..0000000
--- a/ggml-patches/0018-metal-tensor-api-dynamic-k.patch
+++ /dev/null
@@ -1,60 +0,0 @@
-diff --git a/src/ggml-metal/ggml-metal.metal b/src/ggml-metal/ggml-metal.metal
-index f6ffb2b..55ad5ea 100644
---- a/src/ggml-metal/ggml-metal.metal
-+++ b/src/ggml-metal/ggml-metal.metal
-@@ -9412,9 +9412,12 @@ kernel void kernel_mul_mm(
- auto tB = tensor(ptrB, dextents(K, N), array({1, strideB}));
-
- // Configure matmul operation
-+ // note: K is dynamic_extent (clamped to the valid range in PHASE 2), since a static
-+ // N_MM_NK_TOTAL K tile would read src1 out of bounds when K % N_MM_NK_TOTAL != 0
-+ // ref: https://github.com/ggml-org/llama.cpp/pull/27064
- mpp::tensor_ops::matmul2d<
- mpp::tensor_ops::matmul2d_descriptor(
-- NRB, NRA, N_MM_NK_TOTAL, false, true, true,
-+ NRB, NRA, static_cast(dynamic_extent), false, true, true,
- mpp::tensor_ops::matmul2d_descriptor::mode::multiply_accumulate),
- execution_simdgroups> mm;
-
-@@ -9466,10 +9469,14 @@ kernel void kernel_mul_mm(
- threadgroup_barrier(mem_flags::mem_threadgroup);
-
- // === PHASE 2: Tensor matmul ===
-- auto mA = tA.slice(0, 0);
-- auto mB = tB.slice(loop_k, rb);
-+ // Clamp the K extent of both operand tensors to the remaining valid K range so
-+ // the dynamic-K op never reads past the K extent of src1 (or the staged A tile).
-+ const int kExt = min(N_MM_NK_TOTAL, K - loop_k);
-
-- mm.run(mB, mA, cT);
-+ auto tAv = tensor(sa, dextents(kExt, NRA), array({1, N_MM_NK_TOTAL}));
-+ auto tBv = tensor(ptrB + loop_k + rb * strideB, dextents(kExt, N - rb), array({1, strideB}));
-+
-+ mm.run(tBv, tAv, cT);
-
- threadgroup_barrier(mem_flags::mem_threadgroup);
- }
-diff --git a/tests/test-backend-ops.cpp b/tests/test-backend-ops.cpp
-index f54ab41..85e0e92 100644
---- a/tests/test-backend-ops.cpp
-+++ b/tests/test-backend-ops.cpp
-@@ -8401,6 +8401,19 @@ static std::vector> make_test_cases_eval() {
- test_cases.emplace_back(new test_mul_mat(GGML_TYPE_F32, GGML_TYPE_F32, 16, 32, 32, { 1, 1}, {1, 1}, {0, 1, 2, 3}, 64, 3));
- test_cases.emplace_back(new test_mul_mat(GGML_TYPE_F32, GGML_TYPE_F32, 64, 77, 77, {12,1}, {1,1}));
-
-+ // K not a multiple of 32 - exercises the partial-K tile of the Metal tensor API
-+ // mat-mat kernel, which previously read past the K extent of src1 (ggml 33c9ea5)
-+ test_cases.emplace_back(new test_mul_mat(GGML_TYPE_F16, GGML_TYPE_F16, 64, 32, 65, {1, 1}, {1, 1}));
-+ test_cases.emplace_back(new test_mul_mat(GGML_TYPE_F16, GGML_TYPE_F16, 64, 32, 80, {1, 1}, {1, 1}));
-+ test_cases.emplace_back(new test_mul_mat(GGML_TYPE_F16, GGML_TYPE_F32, 64, 32, 80, {1, 1}, {1, 1}));
-+ test_cases.emplace_back(new test_mul_mat(GGML_TYPE_F32, GGML_TYPE_F32, 64, 32, 80, {1, 1}, {1, 1}));
-+ test_cases.emplace_back(new test_mul_mat(GGML_TYPE_F16, GGML_TYPE_F16, 64, 32, 588, {1, 1}, {1, 1})); // 14*14*3, e.g. conv_2d im2col
-+ test_cases.emplace_back(new test_mul_mat(GGML_TYPE_F16, GGML_TYPE_F16, 64, 32, 80, {4, 1}, {1, 1}));
-+ // NanoCodec residual input convolution: K = 1296 (1296 % 32 == 16), plus an aligned control
-+ test_cases.emplace_back(new test_mul_mat(GGML_TYPE_F16, GGML_TYPE_F16, 8, 432, 1296, {1, 1}, {1, 1}));
-+ test_cases.emplace_back(new test_mul_mat(GGML_TYPE_F16, GGML_TYPE_F16, 32, 432, 1296, {1, 1}, {1, 1}));
-+ test_cases.emplace_back(new test_mul_mat(GGML_TYPE_F16, GGML_TYPE_F16, 32, 432, 1280, {1, 1}, {1, 1}));
-+
- test_cases.emplace_back(new test_mul_mat(GGML_TYPE_Q4_0, GGML_TYPE_F32, 576, 512, 576, {1,1}, {1,1}));
- test_cases.emplace_back(new test_mul_mat(GGML_TYPE_Q4_0, GGML_TYPE_F32, 1, 2048, 8192, {1, 1}, {1, 1}));
- for (ggml_type type_a : all_types) {
diff --git a/ggml-patches/0019-cuda-graph-dynamic-update.patch b/ggml-patches/0019-cuda-graph-dynamic-update.patch
deleted file mode 100644
index ea38ca0..0000000
--- a/ggml-patches/0019-cuda-graph-dynamic-update.patch
+++ /dev/null
@@ -1,59 +0,0 @@
-diff --git a/src/ggml-cuda/common.cuh b/src/ggml-cuda/common.cuh
-index 610fdf37..474eb5c7 100644
---- a/src/ggml-cuda/common.cuh
-+++ b/src/ggml-cuda/common.cuh
-@@ -1210,5 +1210,6 @@ struct ggml_cuda_graph {
- std::vector nodes;
- bool disable_due_to_gpu_arch = false;
-+ bool warmup_started = false;
- bool warmup_complete = false;
- uint64_t uid = 0;
- int64_t last_used_time = 0;
-diff --git a/src/ggml-cuda/ggml-cuda.cu b/src/ggml-cuda/ggml-cuda.cu
-index e8f6c10f..fd97f206 100644
---- a/src/ggml-cuda/ggml-cuda.cu
-+++ b/src/ggml-cuda/ggml-cuda.cu
-@@ -5044,25 +5044,25 @@ static enum ggml_status ggml_backend_cuda_graph_compute(ggml_backend_t backend,
- if (graph_compatible) {
- const bool properties_changed = ggml_cuda_graph_update_required(cuda_ctx, cgraph, graph_key);
-
-- if (!graph->warmup_complete) {
-- // Warmup: need at least 2 calls with no property change on the 2nd call
-- if (!properties_changed) {
-- graph->warmup_complete = true;
-- ggml_cuda_graph_log_event(__func__, "warmup complete", cgraph, graph_key);
-- use_cuda_graph = true;
-- cuda_graph_update_required = true;
-- }
-- // else: properties changed or first call - execute directly (use_cuda_graph stays false)
-+ if (!graph->warmup_started) {
-+ // Execute one call directly so lazy kernel/library setup happens
-+ // outside stream capture. Cache/state addresses are expected to
-+ // change between streaming calls, so stability is not a valid
-+ // prerequisite for completing warmup.
-+ graph->warmup_started = true;
-+ } else if (!graph->warmup_complete) {
-+ graph->warmup_complete = true;
-+ ggml_cuda_graph_log_event(__func__, "warmup complete", cgraph, graph_key);
-+ use_cuda_graph = true;
-+ cuda_graph_update_required = true;
- } else {
-- // Post-warmup: normal CUDA graph operation
-- if (properties_changed) {
-- // Properties changed - reset warmup, execute directly until stable again
-- graph->warmup_complete = false;
-- ggml_cuda_graph_log_event(__func__, "warmup reset", cgraph, graph_key);
-- } else {
-- use_cuda_graph = true;
-- cuda_graph_update_required = graph->instance == nullptr;
-- }
-+ // CUDA graph topology is stable even when tensor/cache pointers
-+ // rotate. Re-capture the current parameters and update the
-+ // executable instead of invalidating it and falling back to
-+ // repeated direct launches. cudaGraphExecUpdate() below safely
-+ // re-instantiates if CUDA reports a topology incompatibility.
-+ use_cuda_graph = true;
-+ cuda_graph_update_required = properties_changed || graph->instance == nullptr;
- }
- }
- }
diff --git a/ggml-patches/0020-bf16-convolution.patch b/ggml-patches/0020-bf16-convolution.patch
deleted file mode 100644
index 3cb975d..0000000
--- a/ggml-patches/0020-bf16-convolution.patch
+++ /dev/null
@@ -1,337 +0,0 @@
-diff --git a/src/ggml-cuda/conv2d-dw.cu b/src/ggml-cuda/conv2d-dw.cu
-index db7ee6bf..572eb6bf 100644
---- a/src/ggml-cuda/conv2d-dw.cu
-+++ b/src/ggml-cuda/conv2d-dw.cu
-@@ -121,7 +121,8 @@ void ggml_cuda_op_conv2d_dw(ggml_backend_cuda_context & ctx, ggml_tensor * dst)
- const ggml_tensor * input = dst->src[1];
-
- GGML_ASSERT(input->type == GGML_TYPE_F32 && dst->type == GGML_TYPE_F32);
-- GGML_ASSERT(kernel->type == GGML_TYPE_F32 || kernel->type == GGML_TYPE_F16);
-+ GGML_ASSERT(kernel->type == GGML_TYPE_F32 || kernel->type == GGML_TYPE_F16 ||
-+ kernel->type == GGML_TYPE_BF16);
- const void * w_d = kernel->data;
- const float * x_d = (const float *) input->data;
- float * y_d = (float *) dst->data;
-@@ -153,6 +154,10 @@ void ggml_cuda_op_conv2d_dw(ggml_backend_cuda_context & ctx, ggml_tensor * dst)
- conv2d_dw_kernel<<>>(
- x_d, (const half *) w_d, y_d, in_w, in_h, out_w, out_h, kernel_w, kernel_h, stride_x, stride_y,
- padding_x, padding_y, dilation_x, dilation_y, channels, batches);
-+ } else if (kernel->type == GGML_TYPE_BF16) {
-+ conv2d_dw_kernel<<>>(
-+ x_d, (const nv_bfloat16 *) w_d, y_d, in_w, in_h, out_w, out_h, kernel_w, kernel_h, stride_x,
-+ stride_y, padding_x, padding_y, dilation_x, dilation_y, channels, batches);
- } else {
- conv2d_dw_kernel<<>>(
- x_d, (const float *) w_d, y_d, in_w, in_h, out_w, out_h, kernel_w, kernel_h, stride_x, stride_y,
-@@ -163,6 +168,10 @@ void ggml_cuda_op_conv2d_dw(ggml_backend_cuda_context & ctx, ggml_tensor * dst)
- conv2d_dw_kernel<<>>(
- x_d, (const half *) w_d, y_d, in_w, in_h, out_w, out_h, kernel_w, kernel_h, stride_x, stride_y,
- padding_x, padding_y, dilation_x, dilation_y, channels, batches);
-+ } else if (kernel->type == GGML_TYPE_BF16) {
-+ conv2d_dw_kernel<<>>(
-+ x_d, (const nv_bfloat16 *) w_d, y_d, in_w, in_h, out_w, out_h, kernel_w, kernel_h, stride_x,
-+ stride_y, padding_x, padding_y, dilation_x, dilation_y, channels, batches);
- } else {
- conv2d_dw_kernel<<