From dbb0d10ff702873a0498afe793a8a02a1d691e57 Mon Sep 17 00:00:00 2001 From: Prabhsimran Singh Date: Thu, 10 Sep 2026 21:57:10 +0530 Subject: [PATCH 1/2] feat(tts): update Magpie in index to latest v2607 gguf --- app/model_store.cpp | 179 +++++++++++++++++++++++++++------- config/server.example.yaml | 2 +- config/tts.example.yaml | 2 +- docs/clients.md | 4 +- docs/server.md | 2 +- docs/tts/configuration.md | 8 +- docs/tts/models.md | 42 ++------ models/index.json | 73 +++++++++----- src/tts/magpietts/README.md | 15 ++- tests/cli/model_store_test.py | 73 ++++++++++++++ 10 files changed, 292 insertions(+), 108 deletions(-) diff --git a/app/model_store.cpp b/app/model_store.cpp index ce6874c..e62e2d2 100644 --- a/app/model_store.cpp +++ b/app/model_store.cpp @@ -53,6 +53,12 @@ struct ArchiveMember { uint64_t size = 0; }; +struct ArchiveRange { + uint64_t start = 0; + uint64_t end = 0; + std::string stop_before; +}; + struct Artifact { std::string role; std::string type; @@ -64,6 +70,7 @@ struct Artifact { uint64_t range_end = 0; std::string stop_before; std::vector members; + std::vector ranges; }; struct Model { @@ -370,18 +377,55 @@ load_index() { artifact.members.push_back(std::move(member)); } } - if (artifact.type != "file" && artifact.type != "tar-prefix") + if (const Value* ranges = artifact_value.find("ranges")) { + for (const auto& range_value : ranges->array()) { + ArchiveRange range; + range.start = integer(range_value, "start"); + range.end = integer(range_value, "end"); + range.stop_before = range_value.string_or("stop_before"); + if (!range.stop_before.empty()) + validate_component(range.stop_before, "archive stop member"); + artifact.ranges.push_back(std::move(range)); + } + } + if (artifact.type != "file" && artifact.type != "tar-prefix" && + artifact.type != "tar-ranges") throw std::runtime_error("unsupported artifact type in model index"); if (artifact.size == 0) throw std::runtime_error("model index artifact size must be positive"); if (artifact.type == "file" && (!artifact.directory.empty() || !artifact.stop_before.empty() || - !artifact.members.empty())) + !artifact.members.empty() || !artifact.ranges.empty())) throw std::runtime_error("regular model artifact contains archive-only fields"); if (artifact.type == "tar-prefix" && (artifact.directory.empty() || artifact.stop_before.empty() || - artifact.members.empty() || artifact.range_end != artifact.size - 1)) + artifact.members.empty() || !artifact.ranges.empty() || + artifact.range_end != artifact.size - 1)) throw std::runtime_error("invalid tokenizer archive metadata in model index"); + if (artifact.type == "tar-ranges") { + if (artifact.directory.empty() || !artifact.stop_before.empty() || + artifact.members.empty() || artifact.ranges.empty() || artifact.range_end != 0) + throw std::runtime_error( + "invalid ranged tokenizer archive metadata in model index"); + uint64_t total_size = 0; + uint64_t previous_end = 0; + for (size_t i = 0; i < artifact.ranges.size(); ++i) { + const auto& range = artifact.ranges[i]; + if (range.end < range.start || range.end == UINT64_MAX || + range.start % 512 != 0 || (range.end + 1) % 512 != 0 || + (i > 0 && range.start <= previous_end)) + throw std::runtime_error("invalid tokenizer archive range in model index"); + const uint64_t range_size = range.end - range.start + 1; + if (range_size > UINT64_MAX - total_size) + throw std::runtime_error( + "tokenizer archive range size overflows in model index"); + total_size += range_size; + previous_end = range.end; + } + if (total_size != artifact.size) + throw std::runtime_error( + "tokenizer archive ranges do not match artifact size in model index"); + } model.artifacts.push_back(std::move(artifact)); } if (model.artifacts.empty()) @@ -434,7 +478,7 @@ find_model(const Index& index, const std::string& repo) { const Artifact& find_artifact(const Model& model, const std::string& role, bool directory) { for (const auto& artifact : model.artifacts) { - if (artifact.role == role && (artifact.type == "tar-prefix") == directory) + if (artifact.role == role && (artifact.type != "file") == directory) return artifact; } throw MissingModelError( @@ -695,7 +739,7 @@ download(const Model& model, const Artifact& artifact, const fs::path& output) { const std::string protocols = loopback ? "=http,https" : "=https"; fs::path curl_errors = output; curl_errors += ".curl-errors"; - auto invoke = [&](bool resume) { + auto invoke = [&](bool resume, const fs::path& destination, const std::string& range) { std::error_code remove_error; fs::remove(curl_errors, remove_error); std::vector arguments = { @@ -718,11 +762,11 @@ download(const Model& model, const Artifact& artifact, const fs::path& output) { "--speed-time", "30", "--output", - output.u8string()}; - if (artifact.type == "tar-prefix") { + destination.u8string()}; + if (!range.empty()) { arguments.push_back("--range"); - arguments.push_back("0-" + std::to_string(artifact.range_end)); - } else if (resume && fs::exists(output)) { + arguments.push_back(range); + } else if (resume && fs::exists(destination)) { arguments.push_back("--continue-at"); arguments.push_back("-"); } @@ -737,11 +781,50 @@ download(const Model& model, const Artifact& artifact, const fs::path& output) { arguments.push_back(download_url(model, artifact)); return run_curl(arguments); }; - int status = invoke(true); + int status = 0; + if (artifact.type == "tar-ranges") { + std::ofstream combined(output, std::ios::binary | std::ios::trunc); + if (!combined) + throw std::runtime_error("cannot create ranged artifact " + path_utf8(output)); + for (size_t i = 0; i < artifact.ranges.size(); ++i) { + const auto& range = artifact.ranges[i]; + fs::path segment = output; + segment += ".range-" + std::to_string(i); + std::error_code error; + fs::remove(segment, error); + status = invoke( + false, segment, std::to_string(range.start) + "-" + std::to_string(range.end)); + if (status != 0) { + fs::remove(segment, error); + break; + } + const uint64_t expected_size = range.end - range.start + 1; + const uintmax_t actual_size = fs::file_size(segment, error); + if (error || actual_size != expected_size) { + fs::remove(segment, error); + throw std::runtime_error("downloaded tokenizer archive range has the wrong size"); + } + std::ifstream input(segment, std::ios::binary); + combined << input.rdbuf(); + if (input.bad() || !combined) { + fs::remove(segment, error); + throw std::runtime_error("cannot assemble ranged tokenizer artifact"); + } + fs::remove(segment, error); + } + combined.close(); + if (!combined) + throw std::runtime_error("cannot assemble ranged tokenizer artifact"); + } else { + const std::string range = artifact.type == "tar-prefix" + ? "0-" + std::to_string(artifact.range_end) + : std::string{}; + status = invoke(true, output, range); + } if (status == 33 && artifact.type == "file") { std::error_code error; fs::remove(output, error); - status = invoke(false); + status = invoke(false, output, {}); } if (status == 127) throw std::runtime_error(curl_missing_message()); @@ -980,16 +1063,32 @@ validate_pax_metadata(const std::string& data) { } void -extract_tar_prefix(const fs::path& archive, const fs::path& destination, const Artifact& artifact) { - std::ifstream input(archive, std::ios::binary); - if (!input) - throw std::runtime_error("cannot read tokenizer archive " + path_utf8(archive)); - fs::create_directories(destination); +extract_tar_segment( + std::ifstream& input, const fs::path& destination, uint64_t segment_size, + const std::string& stop_before) { std::array header{}; + uint64_t segment_remaining = segment_size; bool reached_stop = false; - while (input.read(header.data(), header.size())) { - if (std::all_of(header.begin(), header.end(), [](char c) { return c == '\0'; })) + bool reached_end = false; + auto read = [&](char* output, size_t size) { + if (size > segment_remaining || !input.read(output, static_cast(size))) + throw std::runtime_error("truncated tokenizer artifact"); + segment_remaining -= size; + }; + auto skip = [&](uint64_t size) { + if (size > segment_remaining) + throw std::runtime_error("truncated tokenizer artifact"); + input.seekg(static_cast(size), std::ios::cur); + if (!input) + throw std::runtime_error("truncated tokenizer artifact"); + segment_remaining -= size; + }; + while (segment_remaining >= header.size()) { + read(header.data(), header.size()); + if (std::all_of(header.begin(), header.end(), [](char c) { return c == '\0'; })) { + reached_end = true; break; + } uint64_t checksum = 0; for (size_t i = 0; i < header.size(); ++i) checksum += static_cast(i >= 148 && i < 156 ? ' ' : header[i]); @@ -1005,7 +1104,7 @@ extract_tar_prefix(const fs::path& archive, const fs::path& destination, const A for (const auto& component : relative) if (component == "..") throw std::runtime_error("unsafe path in tokenizer artifact"); - if (relative.filename() == artifact.stop_before) { + if (!stop_before.empty() && relative.filename() == stop_before) { reached_stop = true; break; } @@ -1024,8 +1123,7 @@ extract_tar_prefix(const fs::path& archive, const fs::path& destination, const A while (remaining > 0) { const size_t count = static_cast(std::min(remaining, buffer.size())); - if (!input.read(buffer.data(), static_cast(count))) - throw std::runtime_error("truncated tokenizer artifact"); + read(buffer.data(), count); file.write(buffer.data(), static_cast(count)); remaining -= count; } @@ -1033,8 +1131,7 @@ extract_tar_prefix(const fs::path& archive, const fs::path& destination, const A throw std::runtime_error("cannot extract " + path_utf8(output)); } else if ((type == 'x' || type == 'g') && size <= 64 * 1024) { std::string metadata(static_cast(size), '\0'); - if (!input.read(metadata.data(), static_cast(metadata.size()))) - throw std::runtime_error("truncated tokenizer artifact"); + read(metadata.data(), metadata.size()); validate_pax_metadata(metadata); } else { throw std::runtime_error("unsupported TAR entry in tokenizer artifact"); @@ -1042,16 +1139,32 @@ extract_tar_prefix(const fs::path& archive, const fs::path& destination, const A const uint64_t padding = (512 - (size % 512)) % 512; if (type == '5') { if (size != 0) - input.seekg(static_cast(size + padding), std::ios::cur); + skip(size + padding); } else if (padding != 0) { - input.seekg(static_cast(padding), std::ios::cur); + skip(padding); + } + } + if (!stop_before.empty() && !reached_stop) + throw std::runtime_error("tokenizer archive range did not reach the expected stop member"); + if (stop_before.empty() && !reached_end && segment_remaining != 0) + throw std::runtime_error("tokenizer archive range ends inside a TAR header"); + skip(segment_remaining); +} + +void +extract_tar_artifact( + const fs::path& archive, const fs::path& destination, const Artifact& artifact) { + std::ifstream input(archive, std::ios::binary); + if (!input) + throw std::runtime_error("cannot read tokenizer archive " + path_utf8(archive)); + fs::create_directories(destination); + if (artifact.type == "tar-prefix") { + extract_tar_segment(input, destination, artifact.size, artifact.stop_before); + } else { + for (const auto& range : artifact.ranges) { + extract_tar_segment(input, destination, range.end - range.start + 1, range.stop_before); } - if (!input) - throw std::runtime_error("truncated tokenizer artifact"); } - if (!reached_stop) - throw std::runtime_error( - "tokenizer archive prefix did not reach the expected model weights"); for (const auto& member : artifact.members) { if (!valid_member(destination / member.name, member)) throw std::runtime_error( @@ -1074,7 +1187,7 @@ materialize(const Model& model, const Artifact& artifact) { std::fprintf(stderr, "[model] cached: %s\n", path_utf8(destination).c_str()); return {model.repo, artifact.role, destination, true}; } - if (artifact.type == "tar-prefix") { + if (artifact.type != "file") { bool valid = fs::is_directory(destination); for (const auto& member : artifact.members) valid = valid && valid_member(destination / member.name, member); @@ -1102,7 +1215,7 @@ materialize(const Model& model, const Artifact& artifact) { fs::remove(partial_revision_path(partial), partial_error); } if (!valid_file(partial, artifact)) { - if (artifact.type == "tar-prefix") { + if (artifact.type != "file") { discard_partial_download(partial); } else { std::error_code error; @@ -1153,7 +1266,7 @@ materialize(const Model& model, const Artifact& artifact) { } else { const fs::path extracting = directory / (artifact.directory + ".extracting"); fs::remove_all(extracting, error); - extract_tar_prefix(partial, extracting, artifact); + extract_tar_artifact(partial, extracting, artifact); fs::remove_all(destination, error); error.clear(); fs::rename(extracting, destination, error); diff --git a/config/server.example.yaml b/config/server.example.yaml index be6667c..c51d7fa 100644 --- a/config/server.example.yaml +++ b/config/server.example.yaml @@ -56,7 +56,7 @@ diar: tts: enabled: true # true | false | auto - magpie-model: /models/magpie-tts/magpie_tts_multilingual_357m.v2602.f16.gguf + magpie-model: /models/magpie-tts/magpie.gguf codec-model: /models/nano-codec/nemo_nano_codec_22khz_1.89kbps_21.5fps.decoder.f16.gguf tokenizer-model-dir: /models/magpie-tts/extracted # tn-model-dir: /models/en_tn_grammars_cased # written-form -> spoken-form TN diff --git a/config/tts.example.yaml b/config/tts.example.yaml index 3726661..604ea62 100644 --- a/config/tts.example.yaml +++ b/config/tts.example.yaml @@ -10,7 +10,7 @@ # Every key is optional, and omitted keys keep their built-in defaults. Unknown # keys are a hard error so typos are not silently ignored. tts: - magpie-model: /models/magpie-tts/magpie_tts_multilingual_357m.v2602.f16.gguf + magpie-model: /models/magpie-tts/magpie.gguf codec-model: /models/nano-codec/nemo_nano_codec_22khz_1.89kbps_21.5fps.decoder.f16.gguf tokenizer-model-dir: /models/magpie-tts/extracted # tn-model-dir: /models/en_tn_grammars_cased # written-form -> spoken-form TN diff --git a/docs/clients.md b/docs/clients.md index 2ef84ff..38d7adf 100644 --- a/docs/clients.md +++ b/docs/clients.md @@ -57,8 +57,8 @@ than placing an API key in a public page. ## curl The speech example requires a TTS model. Start a TTS-only server with -`nemo-speech serve --tts-model magpie`, or add `--tts-model magpie` to the ASR -server command above. +`nemo-speech serve --tts-model models/magpie-tts/magpie.gguf`, or add that +option to the ASR server command above. ```bash curl -s http://127.0.0.1:8080/v1/audio/transcriptions \ diff --git a/docs/server.md b/docs/server.md index 025f162..530bfb8 100644 --- a/docs/server.md +++ b/docs/server.md @@ -21,7 +21,7 @@ examples, tests, and tools. Individual features can be selected explicitly ```bash nemo-speech serve \ --asr-model models/asr.q8_0.gguf \ - --tts-model models/magpie-tts/magpie_tts_multilingual_357m.v2602.f16.gguf \ + --tts-model models/magpie-tts/magpie.gguf \ --codec-model models/nano-codec/nemo_nano_codec_22khz_1.89kbps_21.5fps.decoder.f16.gguf \ --tokenizer-dir models/magpie-tts/extracted # HTTP API and playground: http://127.0.0.1:8080/ diff --git a/docs/tts/configuration.md b/docs/tts/configuration.md index 38482cc..6d61485 100644 --- a/docs/tts/configuration.md +++ b/docs/tts/configuration.md @@ -17,7 +17,7 @@ or with flags for HTTP: ```bash nemo-speech serve \ - --tts.magpie-model models/magpie-tts/magpie_tts_multilingual_357m.v2602.f16.gguf \ + --tts.magpie-model models/magpie-tts/magpie.gguf \ --tts.codec-model models/nano-codec/nemo_nano_codec_22khz_1.89kbps_21.5fps.decoder.f16.gguf \ --tts.tokenizer-model-dir models/magpie-tts/extracted \ --host 127.0.0.1 --port 8080 \ @@ -89,7 +89,7 @@ Pass the grammar directory to the CLI or server: ```bash nemo-speech synthesize "I have 2 apples." \ - --magpie-model models/magpie-tts/magpie_tts_multilingual_357m.v2602.f16.gguf \ + --magpie-model models/magpie-tts/magpie.gguf \ --codec-model models/nano-codec/nemo_nano_codec_22khz_1.89kbps_21.5fps.decoder.f16.gguf \ --tokenizer-dir models/magpie-tts/extracted \ --tn-model-dir models/tn_configs \ @@ -100,7 +100,7 @@ The equivalent YAML setting is: ```yaml tts: - magpie-model: /models/magpie-tts/magpie_tts_multilingual_357m.v2602.f16.gguf + magpie-model: /models/magpie-tts/magpie.gguf codec-model: /models/nano-codec/nemo_nano_codec_22khz_1.89kbps_21.5fps.decoder.f16.gguf tokenizer-model-dir: /models/magpie-tts/extracted tn-model-dir: /models/tn_configs @@ -122,7 +122,7 @@ Use the unified CLI for synthesis without a server: ```bash nemo-speech synthesize "Hello from Magpie." \ - --magpie-model models/magpie-tts/magpie_tts_multilingual_357m.v2602.f16.gguf \ + --magpie-model models/magpie-tts/magpie.gguf \ --codec-model models/nano-codec/nemo_nano_codec_22khz_1.89kbps_21.5fps.decoder.f16.gguf \ --tokenizer-dir models/magpie-tts/extracted \ --speaker 0 --output magpie.wav diff --git a/docs/tts/models.md b/docs/tts/models.md index 0b4f0d9..fbe7ca2 100644 --- a/docs/tts/models.md +++ b/docs/tts/models.md @@ -16,39 +16,15 @@ options are omitted. Hugging Face: [nvidia/magpie_tts_multilingual_357m](https://huggingface.co/nvidia/magpie_tts_multilingual_357m) -```bash -# Download the v2602 GGUF and its matching tokenizer archive from their -# immutable revisions. -hf download nvidia/magpie_tts_multilingual_357m \ - --include magpie_tts_multilingual_357m.v2602.f16.gguf \ - --revision 452ef560f972c38d5fc16476259aac9456453547 \ - --local-dir models/magpie-tts -hf download nvidia/magpie_tts_multilingual_357m \ - --include magpie_tts_multilingual_357m.nemo \ - --revision 34d7e40da85cabc97f92198889b65cea27bc7fd1 \ - --local-dir models/magpie-tts - -# Extract the tokenizer assets loaded by the runtime. -mkdir -p models/magpie-tts/extracted -tar -xf models/magpie-tts/magpie_tts_multilingual_357m.nemo \ - -C models/magpie-tts/extracted -``` - -MagpieTTS v2607 uses factor-2 frame stacking and must currently be converted -locally before use: - -```bash -hf download nvidia/magpie_tts_multilingual_357m \ - magpie_tts_multilingual_357m.nemo \ - --revision v2607 --local-dir models/magpie-tts-v2607 -python3 convert_model.py models/magpie-tts-v2607/magpie_tts_multilingual_357m.nemo \ - --outfile models/magpie-tts-v2607/magpie_tts_multilingual_357m.v2607.f16.gguf -mkdir -p models/magpie-tts-v2607/extracted -tar -xf models/magpie-tts-v2607/magpie_tts_multilingual_357m.nemo \ - -C models/magpie-tts-v2607/extracted -``` - -Both v2602 (factor 1) and v2607 (factor 2) use the same NanoCodec decoder. +Use `nemo-speech pull magpie` to download the GGUF and matching tokenizer +assets pinned by the model index. For manually managed checkpoints, download +the GGUF and `.nemo` archive from the same Hugging Face revision and pass their +local paths explicitly. + +v2602 generates one codec frame per autoregressive step, while v2607 uses a +frame-stacking factor of 2. This is independent of `tts.chunk-frames`, which +groups generated frames for NanoCodec streaming. Both versions use the same +NanoCodec decoder. **Tokenizer.** MagpieTTS's tokenizer assets live *inside* the `.nemo` archive - they are not part of the GGUF. The built-in pull extracts only the required, diff --git a/models/index.json b/models/index.json index 8b01556..a1f7e9d 100644 --- a/models/index.json +++ b/models/index.json @@ -107,7 +107,7 @@ { "repo": "nvidia/magpie_tts_multilingual_357m", "aliases": ["magpie", "magpie-tts"], - "revision": "452ef560f972c38d5fc16476259aac9456453547", + "revision": "19806879b16d3f2ccf28fb112b1bcd16a3c7923e", "license": "NVIDIA Open Model License", "license_url": "https://huggingface.co/nvidia/magpie_tts_multilingual_357m", "companions": [ @@ -117,55 +117,78 @@ { "role": "tts", "type": "file", - "filename": "magpie_tts_multilingual_357m.v2602.f16.gguf", - "size": 448604832, - "sha256": "901d299a8b1df016cf81cae0089a7a7c15627b9633d033357e15a47d9a219a75" + "filename": "magpie_tts_multilingual_357m.v2607.f16.gguf", + "size": 568663328, + "sha256": "30d27551fc4a050e8095139f0185796024493373e83d49443c8ddba33c514e68" }, { "role": "tokenizer", - "type": "tar-prefix", + "type": "tar-ranges", "filename": "magpie_tts_multilingual_357m.nemo", - "revision": "34d7e40da85cabc97f92198889b65cea27bc7fd1", "directory": "tokenizer", - "size": 33554432, - "sha256": "9efde603de69818ea8dc449ab4165c39bae661c8aa7f798a2c9891961e8b0dd0", - "range_end": 33554431, - "stop_before": "model_weights.ckpt", + "size": 31076864, + "sha256": "169144a64f6d775178696914e5aad8282db44820e73341d5f9a4e7865fac1f26", + "ranges": [ + { + "start": 0, + "end": 20619263, + "stop_before": "model_weights.ckpt" + }, + { + "start": 1459750400, + "end": 1470207999 + } + ], "members": [ { - "name": "1d848e67f9454a73af772cc26f21d0db_heteronyms-052722", + "name": "05adc40366e149b69319acd4b28a4919_pt_br_prondict-v1.0.dict", + "size": 3580284, + "sha256": "6492abc404db16bbad83dd9d7f9a60eb617699f0a3ff54b3be6184e14507b745" + }, + { + "name": "339da71c54b046f98cbcf38ef6d4ff67_hindi_phoneme_merged_phoneme_dict.dict", + "size": 7319864, + "sha256": "3b50cc2c21cd8d34983d8492c91308e9db4a58c1194d6506fdbb2b97aead75fc" + }, + { + "name": "41913ebaa70342058574da74293b7630_magpie_tts_multilingual_357m.nemo.speakers.json", + "size": 68, + "sha256": "36cdcf01ebc0afb506660ac71f2d0a211236374ad5c872d7d8d985a3f9f6ccaf" + }, + { + "name": "61d57a4ccb064d1e8e3d9f871c571932_heteronyms-052722", "size": 1606, "sha256": "b701909aedf753172eff223950f8859cd4b9b4c80199cf0a6e9ac4a307c8f8ec" }, { - "name": "938dc98b70524b23b19f72a19276ca22_de_nv230119.heteronym", - "size": 44566, - "sha256": "771cc585a574fd35bd14f4ce6108edf1e0e512a8dbc810db7de4c1165ba9d0ef" + "name": "74b3e832fe914a569fcd42b51d06b2f1_ipa_dict_nv23.05.txt", + "size": 5419, + "sha256": "252e8eaf60dfd891520759913dc8534393eebfa8d57192365f76f9229f294e9e" }, { - "name": "9a6b090bd4a14621b340b39ba65e23b0_es_ES_nv230301.dict", + "name": "7dbc31751f224f2486090d59dc95b9f7_es_ES_nv230301.dict", "size": 2230910, "sha256": "94c9a25bb359f733cd863c887d0112e7ae19a744b0333ef64b6a88e576428669" }, { - "name": "bafa5b4ca36942ad883c5c228474ca7e_de_nv230119.dict", - "size": 4313907, - "sha256": "5c7bbf3346ebd6dc57769b5cc805124215cb1bcc3b8174c4254e9e928c5d6094" + "name": "c5e4ec2af5a14ce294f4b9edcc936535_de_nv230119.heteronym", + "size": 44566, + "sha256": "771cc585a574fd35bd14f4ce6108edf1e0e512a8dbc810db7de4c1165ba9d0ef" }, { - "name": "c799361ecf2540709dbb931cf0e08ece_ipa_dict_nv23.05.txt", - "size": 5419, - "sha256": "252e8eaf60dfd891520759913dc8534393eebfa8d57192365f76f9229f294e9e" + "name": "cf01ab5c48c84f3282ef7888263361e5_de_nv230119.dict", + "size": 4313907, + "sha256": "5c7bbf3346ebd6dc57769b5cc805124215cb1bcc3b8174c4254e9e928c5d6094" }, { - "name": "cdd41953cfe7479cbb74ba0f9fd42d2b_ipa_cmudict-0.7b_nv23.01.txt", + "name": "dc7d60d6b15a4651b21c9ca2932b62c6_ipa_cmudict-0.7b_nv23.01.txt", "size": 3093097, - "sha256": "dd0927fffc89e8539ea0a26ccbc164a908f4b3de9613d924c90afa5300e00f72" + "sha256": "7ea5c4c2c59cc780748b24d4ab60bd689779c4ac4ea7e27c0a8c3e20f300c0fa" }, { "name": "model_config.yaml", - "size": 5372, - "sha256": "ccf353af2b3433282a6cf38b96f25f54a022266f99cbfc046f9122d509e12ea0" + "size": 8358, + "sha256": "9577bfa4769981849feccd6a54d38425d81fff41ac16c282f60be97faf46ebe4" } ] } diff --git a/src/tts/magpietts/README.md b/src/tts/magpietts/README.md index 2ea5918..41cff9c 100644 --- a/src/tts/magpietts/README.md +++ b/src/tts/magpietts/README.md @@ -18,15 +18,14 @@ into 22050 Hz mono PCM audio. The examples below assume these files are available: ```text -models/magpie-tts/magpie_tts_multilingual_357m.v2602.f16.gguf +models/magpie-tts/magpie.gguf models/magpie-tts/extracted models/nano-codec/nemo_nano_codec_22khz_1.89kbps_21.5fps.decoder.f16.gguf ``` -`magpie_tts_multilingual_357m.v2602.f16.gguf` is the MagpieTTS autoregressive -model. The `extracted` directory is the unpacked MagpieTTS `.nemo` checkpoint -and is needed by the tokenizer. The NanoCodec GGUF is the token-to-audio -decoder. +`magpie.gguf` is an example path for the MagpieTTS autoregressive model. The +`extracted` directory is the unpacked MagpieTTS `.nemo` checkpoint and is +needed by the tokenizer. The NanoCodec GGUF is the token-to-audio decoder. ## Build @@ -49,7 +48,7 @@ Convert the MagpieTTS `.nemo` checkpoint or extracted checkpoint directory: ```bash python convert_model.py models/magpie-tts/extracted \ - --outfile models/magpie-tts/magpie_tts_multilingual_357m.v2602.f16.gguf \ + --outfile models/magpie-tts/magpie.gguf \ --outtype f16 \ --metadata-json models/magpie-tts/magpie_tts_multilingual_357m.gguf.json ``` @@ -81,7 +80,7 @@ It accepts text by default and also supports pre-tokenized IDs for diagnostics: ```bash build/cuda-tts/bin/synthesize_text \ - --tts.magpie-model models/magpie-tts/magpie_tts_multilingual_357m.v2602.f16.gguf \ + --tts.magpie-model models/magpie-tts/magpie.gguf \ --tts.codec-model models/nano-codec/nemo_nano_codec_22khz_1.89kbps_21.5fps.decoder.f16.gguf \ --tts.tokenizer-model-dir models/magpie-tts/extracted \ --tts.text "Hello world." \ @@ -103,7 +102,7 @@ Launch the server: ```bash build/cuda-full/bin/riva_server \ - --tts.magpie-model models/magpie-tts/magpie_tts_multilingual_357m.v2602.f16.gguf \ + --tts.magpie-model models/magpie-tts/magpie.gguf \ --tts.codec-model models/nano-codec/nemo_nano_codec_22khz_1.89kbps_21.5fps.decoder.f16.gguf \ --tts.tokenizer-model-dir models/magpie-tts/extracted \ --bind 0.0.0.0:50051 \ diff --git a/tests/cli/model_store_test.py b/tests/cli/model_store_test.py index 58a828b..ec0d24a 100644 --- a/tests/cli/model_store_test.py +++ b/tests/cli/model_store_test.py @@ -20,6 +20,7 @@ CODEC_PAYLOAD = b"NeMo-Speech.cpp codec fixture\n" TTS_PAYLOAD = b"NeMo-Speech.cpp TTS fixture\n" TOKENIZER_PAYLOAD = b"tokenizer configuration\n" +UPDATED_TOKENIZER_PAYLOAD = b"updated tokenizer configuration\n" TTS_REVISION = "1" * 40 TOKENIZER_REVISION = "3" * 40 CODEC_REVISION = "2" * 40 @@ -31,6 +32,7 @@ class ArtifactHandler(http.server.BaseHTTPRequestHandler): requests = 0 request_counts: dict[str, int] = {} request_paths: list[str] = [] + range_headers: list[str | None] = [] payloads: dict[str, bytes] = {} def do_GET(self) -> None: @@ -46,6 +48,7 @@ def do_GET(self) -> None: start = 0 end = len(payload) - 1 range_header = self.headers.get("Range") + type(self).range_headers.append(range_header) if range_header: assert range_header.startswith("bytes=") bounds = range_header.removeprefix("bytes=").split("-", 1) @@ -89,6 +92,38 @@ def tokenizer_archive() -> bytes: return output.getvalue() +def ranged_tokenizer_archive() -> tuple[bytes, list[dict[str, int | str]], bytes]: + output = io.BytesIO() + with tarfile.open(fileobj=output, mode="w", format=tarfile.PAX_FORMAT) as bundle: + tokenizer = tarfile.TarInfo("tokenizer.txt") + tokenizer.size = len(TOKENIZER_PAYLOAD) + tokenizer.pax_headers = {"mtime": "0.0"} + bundle.addfile(tokenizer, io.BytesIO(TOKENIZER_PAYLOAD)) + weights = tarfile.TarInfo("model_weights.ckpt") + weights.size = 4096 + weights.pax_headers = {"mtime": "0.0"} + bundle.addfile(weights, io.BytesIO(b"x" * weights.size)) + updated = tarfile.TarInfo("tokenizer.txt") + updated.size = len(UPDATED_TOKENIZER_PAYLOAD) + updated.pax_headers = {"mtime": "0.0"} + bundle.addfile(updated, io.BytesIO(UPDATED_TOKENIZER_PAYLOAD)) + archive = output.getvalue() + with tarfile.open(fileobj=io.BytesIO(archive), mode="r") as bundle: + members = bundle.getmembers() + weights = next(member for member in members if member.name == "model_weights.ckpt") + updated = [member for member in members if member.name == "tokenizer.txt"][-1] + ranges: list[dict[str, int | str]] = [ + { + "start": 0, + "end": weights.offset_data - 1, + "stop_before": "model_weights.ckpt", + }, + {"start": updated.offset, "end": len(archive) - 1}, + ] + selected = b"".join(archive[int(item["start"]) : int(item["end"]) + 1] for item in ranges) + return archive, ranges, selected + + def file_artifact(role: str, filename: str, payload: bytes, sha256: str | None = None) -> dict: return { "role": role, @@ -187,6 +222,7 @@ def main() -> None: ArtifactHandler.requests = 0 ArtifactHandler.request_counts = {} ArtifactHandler.request_paths = [] + ArtifactHandler.range_headers = [] ArtifactHandler.payloads = { "tiny.gguf": PAYLOAD, "tiny-tts.gguf": TTS_PAYLOAD, @@ -293,6 +329,43 @@ def main() -> None: assert ArtifactHandler.request_counts["tiny-tts.nemo"] == 2 assert ArtifactHandler.request_counts["tiny-codec.gguf"] == 1 + ranged_archive, ranges, selected = ranged_tokenizer_archive() + ranged_index = json.loads(index.read_text(encoding="utf-8")) + ranged_artifact = ranged_index["models"][1]["artifacts"][1] + ranged_artifact["type"] = "tar-ranges" + ranged_artifact["size"] = len(selected) + ranged_artifact["sha256"] = hashlib.sha256(selected).hexdigest() + ranged_artifact["ranges"] = ranges + ranged_artifact["members"][0] = { + "name": "tokenizer.txt", + "size": len(UPDATED_TOKENIZER_PAYLOAD), + "sha256": hashlib.sha256(UPDATED_TOKENIZER_PAYLOAD).hexdigest(), + } + del ranged_artifact["range_end"] + del ranged_artifact["stop_before"] + index.write_text(json.dumps(ranged_index), encoding="utf-8") + ArtifactHandler.payloads["tiny-tts.nemo"] = ranged_archive + ranged_cache = root / "ranged-cache" + ranged_environment = { + **environment, + "NEMO_SPEECH_MODEL_DIR": str(ranged_cache), + } + range_header_start = len(ArtifactHandler.range_headers) + ranged_pull = run(binary, ranged_environment, "--json", "pull", "tiny-tts") + assert ranged_pull.returncode == 0, ranged_pull.stderr + ranged_tokenizer = pathlib.Path(json.loads(ranged_pull.stdout)["artifacts"][1]["path"]) + assert (ranged_tokenizer / "tokenizer.txt").read_bytes() == UPDATED_TOKENIZER_PAYLOAD + assert not (ranged_tokenizer / "model_weights.ckpt").exists() + expected_ranges = [f'bytes={item["start"]}-{item["end"]}' for item in ranges] + actual_ranges = [ + value + for value in ArtifactHandler.range_headers[range_header_start:] + if value is not None + ] + assert actual_ranges == expected_ranges + ArtifactHandler.payloads["tiny-tts.nemo"] = tokenizer_tar + write_index(index, hashlib.sha256(PAYLOAD).hexdigest(), tokenizer_tar) + previous_mtime = destination.stat().st_mtime_ns destination.write_bytes(b"x" * len(PAYLOAD)) os.utime(destination, ns=(previous_mtime + 2_000_000_000,) * 2) From 14e26e0a3b9b0d3aedc5b83dc28b2008d1ae09ae Mon Sep 17 00:00:00 2001 From: Prabhsimran Singh Date: Tue, 29 Sep 2026 13:42:04 +0530 Subject: [PATCH 2/2] fix(tts): address review comments --- app/model_store.cpp | 12 +++++++----- docs/clients.md | 4 ++-- tests/cli/model_store_test.py | 1 + 3 files changed, 10 insertions(+), 7 deletions(-) diff --git a/app/model_store.cpp b/app/model_store.cpp index e62e2d2..744ba8f 100644 --- a/app/model_store.cpp +++ b/app/model_store.cpp @@ -804,13 +804,15 @@ download(const Model& model, const Artifact& artifact, const fs::path& output) { fs::remove(segment, error); throw std::runtime_error("downloaded tokenizer archive range has the wrong size"); } - std::ifstream input(segment, std::ios::binary); - combined << input.rdbuf(); - if (input.bad() || !combined) { - fs::remove(segment, error); - throw std::runtime_error("cannot assemble ranged tokenizer artifact"); + bool assembled = false; + { + std::ifstream input(segment, std::ios::binary); + assembled = input && (combined << input.rdbuf()); } + // Close the segment before removing it; Windows cannot delete open files. fs::remove(segment, error); + if (!assembled) + throw std::runtime_error("cannot assemble ranged tokenizer artifact"); } combined.close(); if (!combined) diff --git a/docs/clients.md b/docs/clients.md index 38d7adf..2ef84ff 100644 --- a/docs/clients.md +++ b/docs/clients.md @@ -57,8 +57,8 @@ than placing an API key in a public page. ## curl The speech example requires a TTS model. Start a TTS-only server with -`nemo-speech serve --tts-model models/magpie-tts/magpie.gguf`, or add that -option to the ASR server command above. +`nemo-speech serve --tts-model magpie`, or add `--tts-model magpie` to the ASR +server command above. ```bash curl -s http://127.0.0.1:8080/v1/audio/transcriptions \ diff --git a/tests/cli/model_store_test.py b/tests/cli/model_store_test.py index ec0d24a..cd66fac 100644 --- a/tests/cli/model_store_test.py +++ b/tests/cli/model_store_test.py @@ -363,6 +363,7 @@ def main() -> None: if value is not None ] assert actual_ranges == expected_ranges + assert not list(ranged_cache.rglob("*.range-*")) ArtifactHandler.payloads["tiny-tts.nemo"] = tokenizer_tar write_index(index, hashlib.sha256(PAYLOAD).hexdigest(), tokenizer_tar)