From 3cbeea8022c5447df5fd925d51c781176ac1e52c Mon Sep 17 00:00:00 2001 From: Fabian Meyer <44942030+dinooo13@users.noreply.github.com> Date: Sat, 26 Sep 2026 01:11:34 +0200 Subject: [PATCH 1/6] Fillers: a filler that opens a later sentence keeps it capitalised MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit "Ich habe einen Test geschrieben. Ähm, beim Timeout …" came out as "… geschrieben. beim Timeout …": only a filler that opened the whole transcript handed its capital on. Any filler after a full stop, question or exclamation mark now does too; an ellipsis trails off and does not. Co-Authored-By: Claude Opus 5.5 (1M context) --- .../Processors/FillerRemover.swift | 45 ++++++++++++++----- .../PladderCoreTests/FillerRemoverTests.swift | 14 ++++++ 2 files changed, 49 insertions(+), 10 deletions(-) diff --git a/Sources/PladderCore/Processors/FillerRemover.swift b/Sources/PladderCore/Processors/FillerRemover.swift index 85d5294..be7c9c5 100644 --- a/Sources/PladderCore/Processors/FillerRemover.swift +++ b/Sources/PladderCore/Processors/FillerRemover.swift @@ -120,26 +120,51 @@ public struct FillerRemover: TextProcessor { var out = "" out.reserveCapacity(text.count) var cursor = 0 + // Set by a filler that opened a sentence, and spent on the first + // letter that follows it: "Test. Ähm, beim Timeout" keeps "Beim" + // a sentence start, as a filler opening the transcript always did. + var capitalizeNext = false + func append(_ segment: String) { + guard capitalizeNext, let letter = segment.firstIndex(where: { !$0.isWhitespace }) else { + out += segment + return + } + capitalizeNext = false + out += segment[.. Bool { + // From the end, since this runs once per filler on the text so far. + var end = text.endIndex + while end > text.startIndex, text[text.index(before: end)].isWhitespace { + end = text.index(before: end) + } + guard end > text.startIndex else { return true } + let head = text[.. Date: Sat, 26 Sep 2026 01:11:34 +0200 Subject: [PATCH 2/6] Processors: spoken punctuation in English, German and Spanish Parakeet already writes numbers as digits and takes most spoken marks, but "question mark", "Fragezeichen" and the like reach the paste as words, usually with the model's own mark stuck to them ("verschieben? Fragezeichen."). A deterministic step now turns the unambiguous ones into the mark and absorbs the stray punctuation: comma, question and exclamation marks, full stop, colon, semicolon and new paragraph. "period", "Punkt" and "punto" stay words, and Spanish "coma" needs Spanish language evidence. Between digits a spoken comma is the decimal comma. It runs last, after the whitespace step, which would otherwise fold its paragraph breaks away. On the fixtures' text through the whole pipeline (M1, median of 31): a typical dictation costs the same as on main up to 2 min and 1 ms more at 10 min; a spoken mark in every sentence of a 10 min dictation, 2.7 ms more. Co-Authored-By: Claude Opus 5.5 (1M context) --- Sources/Pladder/AppModel.swift | 10 +- .../Pladder/Resources/Localizable.xcstrings | 20 ++ .../Processors/SpokenPunctuation.swift | 178 ++++++++++++++++++ .../SpokenPunctuationTests.swift | 80 ++++++++ 4 files changed, 284 insertions(+), 4 deletions(-) create mode 100644 Sources/PladderCore/Processors/SpokenPunctuation.swift create mode 100644 Tests/PladderCoreTests/SpokenPunctuationTests.swift diff --git a/Sources/Pladder/AppModel.swift b/Sources/Pladder/AppModel.swift index 3667f7e..08b649d 100644 --- a/Sources/Pladder/AppModel.swift +++ b/Sources/Pladder/AppModel.swift @@ -198,15 +198,17 @@ final class AppModel { // Processors, in pipeline order: fillers go first so the dictionary sees // cleaned text, the fuzzy custom-word corrector runs after the exact - // replacer so it only sees what the replacer could not fix, and - // whitespace is tidied last. Each entry is a factory so a processor that - // needs settings builds itself from them; nothing here knows which - // processor that is. + // replacer so it only sees what the replacer could not fix, whitespace + // is tidied next, and spoken punctuation comes last, because the + // whitespace step would fold its paragraph breaks back into spaces. + // Each entry is a factory so a processor that needs settings builds + // itself from them; nothing here knows which processor that is. let processorFactories: [@Sendable (Settings) -> any TextProcessor] = [ { _ in FillerRemover(languageHint: { TranscriptLanguage.hint(for: $0) }) }, { DictionaryReplacer(entries: $0.dictionary) }, { CustomWordCorrector(entries: $0.dictionary) }, { _ in WhitespaceNormalizer() }, + { _ in SpokenPunctuation(languageHint: { TranscriptLanguage.hint(for: $0) }) }, ] self.processors = processorFactories.map { $0(initial) } diff --git a/Sources/Pladder/Resources/Localizable.xcstrings b/Sources/Pladder/Resources/Localizable.xcstrings index 6d4724c..67c1f78 100644 --- a/Sources/Pladder/Resources/Localizable.xcstrings +++ b/Sources/Pladder/Resources/Localizable.xcstrings @@ -1430,6 +1430,26 @@ } } } + }, + "Spoken punctuation": { + "localizations": { + "de": { + "stringUnit": { + "state": "translated", + "value": "Gesprochene Satzzeichen" + } + } + } + }, + "Turns spoken marks such as “comma”, “question mark” and “new paragraph” into the marks themselves, in English, German and Spanish.": { + "localizations": { + "de": { + "stringUnit": { + "state": "translated", + "value": "Macht aus gesprochenen Zeichen wie „Komma“, „Fragezeichen“ und „neuer Absatz“ die Zeichen selbst, auf Englisch, Deutsch und Spanisch." + } + } + } } }, "version": "1.0" diff --git a/Sources/PladderCore/Processors/SpokenPunctuation.swift b/Sources/PladderCore/Processors/SpokenPunctuation.swift new file mode 100644 index 0000000..8bd326c --- /dev/null +++ b/Sources/PladderCore/Processors/SpokenPunctuation.swift @@ -0,0 +1,178 @@ +import Foundation + +/// Turns punctuation the speaker said out loud into the mark: "question +/// mark", "Fragezeichen", "signo de interrogación" become "?", and so on in +/// English, German and Spanish. +/// +/// The speech model already turns most spoken marks into punctuation and +/// numbers into digits; what reaches this step is the rest, usually with the +/// model's own punctuation stuck to it ("verschieben? Fragezeichen."). That +/// punctuation is absorbed into the mark, so the result carries it once. +/// +/// Only phrases that are never an ordinary word are taken. "period", +/// "Punkt" and "punto" are left alone, since "a trial period", "ein +/// wichtiger Punkt" and "el punto" are far more common than a dictated full +/// stop, and so are "colon" and "dos puntos". The one ambiguous phrase kept, +/// Spanish "coma", is a word in English too, so it needs the same language +/// evidence the filler remover uses. Between two digits "Komma" and "coma" +/// are the decimal comma: "3 Komma 5" becomes "3,5". +/// +/// Runs after the whitespace step, which would otherwise fold the line +/// breaks of "new paragraph" back into spaces. +public struct SpokenPunctuation: TextProcessor { + public static let processorID = "spoken-punctuation" + + public let id = SpokenPunctuation.processorID + public let displayName = "Spoken punctuation" + public let detail = "Turns spoken marks such as “comma”, “question mark” and “new paragraph” into the marks themselves, in English, German and Spanish." + + private enum Mark { + case comma, fullStop, question, exclamation, colon, semicolon, paragraph + + var text: String { + switch self { + case .comma: "," + case .fullStop: "." + case .question: "?" + case .exclamation: "!" + case .colon: ":" + case .semicolon: ";" + case .paragraph: "\n\n" + } + } + + /// The next word starts a sentence. + var endsSentence: Bool { + switch self { + case .fullStop, .question, .exclamation, .paragraph: true + case .comma, .colon, .semicolon: false + } + } + } + + /// Longer phrases first, so "punto y coma" is never read as "coma". + private static let phrases: [(pattern: String, mark: Mark)] = [ + ("punto y coma", .semicolon), + ("semicolon|semikolon", .semicolon), + ("question mark|fragezeichen|signo de interrogaci[oó]n", .question), + ("exclamation (?:mark|point)|ausrufezeichen|signo de exclamaci[oó]n", .exclamation), + ("full stop|punto final", .fullStop), + ("doppelpunkt", .colon), + ("new paragraph|neuer absatz|nuevo p[aá]rrafo", .paragraph), + ("comma|komma", .comma), + ] + + /// Needs the language evidence; see the type's comment. + private static let spanishComma = "coma" + + + private let universal: [(regex: NSRegularExpression, mark: Mark)] + private let spanish: NSRegularExpression + /// Every phrase in one pass. Almost no dictation holds one, and this is + /// the only scan such a dictation pays. + private let candidate: NSRegularExpression + private let languageHint: (@Sendable (String) -> String?)? + + /// - Parameter languageHint: The same closure the filler remover gets: + /// a confident ISO 639-1 code for the text, or nil. Called only when + /// the text holds "coma". + public init(languageHint: (@Sendable (String) -> String?)? = nil) { + self.languageHint = languageHint + universal = Self.phrases.map { (try! NSRegularExpression(pattern: Self.pattern($0.pattern), options: .caseInsensitive), $0.mark) } + spanish = try! NSRegularExpression(pattern: Self.pattern(Self.spanishComma), options: .caseInsensitive) + let every = (Self.phrases.map(\.pattern) + [Self.spanishComma]).joined(separator: "|") + candidate = try! NSRegularExpression(pattern: Self.pattern(every), options: .caseInsensitive) + } + + /// The phrase as a whole word. The punctuation around it is taken by + /// hand in `apply`: a pattern that starts with optional spaces cannot be + /// scanned for quickly. + private static func pattern(_ phrase: String) -> String { + "\\b(?:\(phrase))\\b" + } + + public func process(_ text: String) async throws -> String { + guard candidate.firstMatch(in: text, range: NSRange(location: 0, length: (text as NSString).length)) != nil + else { return text } + var result = text + for (regex, mark) in universal { + result = Self.apply(regex, mark: mark, to: result) + } + if result.range(of: Self.spanishComma, options: .caseInsensitive) != nil, + let languageHint, languageHint(result) == "es" + { + result = Self.apply(spanish, mark: .comma, to: result) + } + return result + } + + private static func apply(_ regex: NSRegularExpression, mark: Mark, to text: String) -> String { + let ns = text as NSString + let matches = regex.matches(in: text, range: NSRange(location: 0, length: ns.length)) + guard !matches.isEmpty else { return text } + + var out = "" + var cursor = 0 + var capitalizeNext = false + for match in matches { + // The speech model's own punctuation around the phrase, and the + // spaces on either side: "verschieben? Fragezeichen." is one mark. + var start = match.range.location + while start > cursor, isSpace(ns.character(at: start - 1)) { start -= 1 } + while start > cursor, isStray(ns.character(at: start - 1)) { start -= 1 } + while start > cursor, isSpace(ns.character(at: start - 1)) { start -= 1 } + let lead = ns.substring(with: NSRange(location: start, length: match.range.location - start)) + var before = ns.substring(with: NSRange(location: cursor, length: start - cursor)) + if capitalizeNext { before = capitalized(before); capitalizeNext = false } + out += before + cursor = match.range.upperBound + let trailStart = cursor + while cursor < ns.length, isStray(ns.character(at: cursor)) { cursor += 1 } + let trailed = cursor > trailStart + // Spaces after the phrase, up to the next word or line break. + while cursor < ns.length, isSpace(ns.character(at: cursor)) { cursor += 1 } + let next = cursor < ns.length ? ns.character(at: cursor) : nil + + if mark == .comma, out.last?.isNumber == true, lead.allSatisfy(\.isWhitespace), + !trailed, let next, isDigit(next) + { + // "3 Komma 5": the decimal comma, no spaces. + out += "," + continue + } + if mark == .paragraph { + // A paragraph keeps the sentence mark before it; only the + // spaces go. + out += lead.trimmingCharacters(in: .whitespaces) + while out.last?.isWhitespace == true { out.removeLast() } + out += out.isEmpty ? "" : mark.text + } else { + while out.last?.isWhitespace == true { out.removeLast() } + out += mark.text + // One space before the next word; none before a line break + // or at the end. + if let next, !isNewline(next) { out += " " } + } + capitalizeNext = mark.endsSentence + } + var tail = ns.substring(from: cursor) + if capitalizeNext { tail = capitalized(tail) } + out += tail + return out.trimmingCharacters(in: .whitespaces) + } + + private static func isSpace(_ c: unichar) -> Bool { c == 0x20 || c == 0x09 || c == 0xA0 } + private static func isNewline(_ c: unichar) -> Bool { c == 0x0A || c == 0x0D } + private static func isDigit(_ c: unichar) -> Bool { c >= 0x30 && c <= 0x39 } + /// Punctuation the speech model put around the spoken word. + private static func isStray(_ c: unichar) -> Bool { ",.;:!?".utf16.contains(c) } + + /// The first letter upper-cased, if there is one before anything else. + /// A word with a capital inside ("iPhone") is left as it is. + private static func capitalized(_ text: String) -> String { + guard let index = text.firstIndex(where: { !$0.isWhitespace }), text[index].isLetter else { return text } + let next = text.index(after: index) + if next < text.endIndex, text[next].isUppercase { return text } + return text[.. String?)? = nil) async throws -> String { + try await SpokenPunctuation(languageHint: hint).process(text) + } + + // MARK: What the speech model leaves behind + + @Test func absorbsTheSpeechModelsOwnPunctuation() async throws { + // Parakeet's output for "…verschieben Fragezeichen" and "…to three + // question mark": its mark and the spoken one become one. + #expect(try await run("Können wir den Termin auf drei verschieben? Fragezeichen.") + == "Können wir den Termin auf drei verschieben?") + #expect(try await run("Can we push the call to three question mark?") == "Can we push the call to three?") + } + + @Test func turnsEachLanguagesMarksIntoPunctuation() async throws { + #expect(try await run("Hi Sarah comma thanks") == "Hi Sarah, thanks") + #expect(try await run("Hallo Sarah Komma danke") == "Hallo Sarah, danke") + #expect(try await run("Wow exclamation mark that worked") == "Wow! That worked") + #expect(try await run("Das klappt Ausrufezeichen") == "Das klappt!") + #expect(try await run("Es gibt drei Dinge Doppelpunkt Milch, Eier, Brot") == "Es gibt drei Dinge: Milch, Eier, Brot") + #expect(try await run("first part semicolon second part") == "first part; second part") + #expect(try await run("That is all full stop") == "That is all.") + #expect(try await run("Podemos mover la llamada signo de interrogación") == "Podemos mover la llamada?") + #expect(try await run("primero punto y coma segundo") == "primero; segundo") + } + + @Test func capitalisesTheWordAfterASentenceMark() async throws { + #expect(try await run("done question mark what next") == "done? What next") + #expect(try await run("fertig Fragezeichen dann los") == "fertig? Dann los") + #expect(try await run("send it question mark iPhone users too") == "send it? iPhone users too") + } + + @Test func startsAParagraphAndKeepsTheSentenceMarkBeforeIt() async throws { + #expect(try await run("That is the summary. New paragraph. next we ship.") + == "That is the summary.\n\nNext we ship.") + #expect(try await run("Das war es. Neuer Absatz jetzt weiter") == "Das war es.\n\nJetzt weiter") + } + + @Test func aCommaBetweenDigitsIsTheDecimalComma() async throws { + #expect(try await run("Es waren 3 Komma 5 Prozent") == "Es waren 3,5 Prozent") + #expect(try await run("fueron 3 coma 5 grados", hint: { _ in "es" }) == "fueron 3,5 grados") + } + + // MARK: What stays a word + + @Test func leavesAmbiguousWordsAlone() async throws { + let texts = [ + "The trial period ends on Friday.", + "Das ist ein wichtiger Punkt. Punkt acht Uhr geht es los.", + "Ese es el punto clave.", + "The colon is part of the large intestine.", + "Tenemos dos puntos de vista.", + ] + for text in texts { + #expect(try await run(text, hint: { _ in nil }) == text) + } + } + + @Test func spanishComaNeedsSpanishEvidence() async throws { + #expect(try await run("The patient was in a coma for a week.", hint: { _ in "en" }) + == "The patient was in a coma for a week.") + #expect(try await run("The patient was in a coma for a week.") == "The patient was in a coma for a week.") + #expect(try await run("Hola Sara coma gracias", hint: { _ in "es" }) == "Hola Sara, gracias") + } + + @Test func leavesTextWithoutSpokenMarksUnchanged() async throws { + let text = "Der Build läuft auf der CI durch, und die Tests brauchen weniger als eine Sekunde." + #expect(try await run(text) == text) + #expect(try await run("") == "") + } + + @Test func partsOfLongerWordsAreNotMarks() async throws { + #expect(try await run("Kommandozeile und Kommata") == "Kommandozeile und Kommata") + #expect(try await run("commas and commander") == "commas and commander") + } +} From 7a3f6688406e0d2bb4d92c241dc080f9bec73c9f Mon Sep 17 00:00:00 2001 From: Fabian Meyer <44942030+dinooo13@users.noreply.github.com> Date: Sat, 26 Sep 2026 01:37:56 +0200 Subject: [PATCH 3/6] Polish: pick the model, Apple Intelligence or S1-mini by Superwhisper A Model picker under "Polish dictations": Apple Intelligence as before, or S1-mini by Superwhisper at full precision (f16, 1.5 GB) or 8-bit (Q8_0, 805 MB). S1-mini is Qwen3-0.6B fine-tuned for dictation cleanup and runs through llama.cpp on the GPU, leaving the Neural Engine to Parakeet. On the polish set it is more accurate than Apple's model and three to five times faster: 0.49 s and 0.34 s median against 1.66 s. - PolishRouter is the coordinator's one refiner and forwards to the chosen model, so switching needs no new coordinator. - ModelFiles downloads a file once from Hugging Face, only while polish is on and that model is picked, pinned to a commit and moved into place only after its SHA-256 matched. The picker shows progress, and a failed download offers Try Again. - S1MiniPolisher uses the prompt S1-mini was trained on, decodes its fixed prefix once at load, chunks long dictations to its 2,048-token context, and keeps the 8 s cap and the paste-as-dictated fallback. It loads at the first key-down and is freed on a switch or with polish off; llama.cpp's Metal setup is warmed as soon as S1-mini is chosen, since its first run after an install compiles shaders for seconds. - llama.cpp is its prebuilt XCFramework, pinned by release and checksum. bundle.sh embeds it in Contents/Frameworks, drops the Intel slice and signs it with the app's identity; an ad-hoc build goes without the hardened runtime, which would otherwise refuse the framework. Co-Authored-By: Claude Opus 5.5 (1M context) --- Package.swift | 17 +- Sources/Pladder/AppModel.swift | 93 +++++++- Sources/Pladder/PolishRouter.swift | 26 +++ .../Pladder/Resources/Localizable.xcstrings | 122 ++++++++++- Sources/Pladder/Settings/SettingsView.swift | 52 ++++- Sources/Pladder/StatusText.swift | 26 +++ Sources/PladderCore/Models/Settings.swift | 22 +- Sources/PladderRefine/LlamaModel.swift | 173 +++++++++++++++ Sources/PladderRefine/ModelFiles.swift | 206 ++++++++++++++++++ Sources/PladderRefine/S1MiniPolisher.swift | 155 +++++++++++++ .../PladderRefine/TranscriptPolisher.swift | 14 +- .../DictationCoordinatorTests.swift | 16 ++ Tests/PladderRefineTests/S1MiniTests.swift | 120 ++++++++++ scripts/bundle.sh | 27 ++- 14 files changed, 1044 insertions(+), 25 deletions(-) create mode 100644 Sources/Pladder/PolishRouter.swift create mode 100644 Sources/PladderRefine/LlamaModel.swift create mode 100644 Sources/PladderRefine/ModelFiles.swift create mode 100644 Sources/PladderRefine/S1MiniPolisher.swift create mode 100644 Tests/PladderRefineTests/S1MiniTests.swift diff --git a/Package.swift b/Package.swift index 7687d9a..e17303a 100644 --- a/Package.swift +++ b/Package.swift @@ -45,9 +45,18 @@ let package = Package( ] ), - // Apple's on-device model behind the polish hotkey. The only target - // that imports FoundationModels. - .target(name: "PladderRefine", dependencies: ["PladderCore"]), + // The polish models: Apple's on-device model, the only target that + // imports FoundationModels, and S1-mini through llama.cpp. + .target(name: "PladderRefine", dependencies: ["PladderCore", "llama"]), + + // llama.cpp's own prebuilt release, Metal included: a dynamic + // framework that scripts/bundle.sh embeds in the app. Pinned by + // release and checksum; bump both together. + .binaryTarget( + name: "llama", + url: "https://github.com/ggml-org/llama.cpp/releases/download/b11191/llama-b11191-xcframework.zip", + checksum: "c8f9af07555a15b00a87334e13a21320596178c58bbafa2d7c4915c574e7086e" + ), // The menu bar app. .executableTarget( @@ -73,7 +82,7 @@ let package = Package( // engines, or run the benchmark (see docs/BENCHMARKS.md). .executableTarget( name: "PladderCLI", - dependencies: ["PladderCore", "PladderEngines", "PladderAudio", "PladderBench", "PladderRefine"] + dependencies: ["PladderCore", "PladderEngines", "PladderAudio", "PladderBench", "PladderRefine", "PladderSystem"] ), .testTarget(name: "PladderCoreTests", dependencies: ["PladderCore"]), diff --git a/Sources/Pladder/AppModel.swift b/Sources/Pladder/AppModel.swift index 08b649d..d25acbb 100644 --- a/Sources/Pladder/AppModel.swift +++ b/Sources/Pladder/AppModel.swift @@ -51,6 +51,19 @@ final class AppModel { /// no-op that pastes as dictated. private(set) var polishAvailability: OnDeviceModelAvailability = TranscriptPolisher.availability + /// Where the chosen polish model's file stands; nil for Apple's, which + /// has none. Shown under the model picker. + private(set) var polishModelStatus: ModelFileStatus? + /// The coordinator's refiner; `applyPolishModel` points it at the model + /// the settings name. + private let polishRouter: PolishRouter + private let applePolisher = TranscriptPolisher() + /// The S1-mini polisher for the chosen file, nil while Apple's is chosen. + /// Only one is ever held, so switching frees the other's memory. + private var s1MiniPolisher: S1MiniPolisher? + private let modelFiles: ModelFiles + private let modelStatusRelay = ModelStatusRelay() + /// The default chord, standing in for a stored chord Carbon cannot /// register while Accessibility is missing. The stored chord is never /// rewritten and returns with the grant. @@ -135,6 +148,9 @@ final class AppModel { // modifier-only one does. updateEffectiveHotkey() } + if newValue.polishModel != old.polishModel || newValue.polishDictations != old.polishDictations { + applyPolishModel() + } try? store.save(newValue) } } @@ -212,6 +228,13 @@ final class AppModel { ] self.processors = processorFactories.map { $0(initial) } + let router = PolishRouter(applePolisher) + polishRouter = router + let modelStatusRelay = self.modelStatusRelay + modelFiles = ModelFiles(directory: ModelFiles.defaultDirectory) { file, status in + modelStatusRelay.send(file, status) + } + let events = self.events let trusted = Permissions.isAccessibilityTrusted hotkeyUsesTap = trusted @@ -223,7 +246,7 @@ final class AppModel { outputMuter: OutputMuteController( control: CoreAudioOutputMute(), log: { Self.muteLog.info("\($0, privacy: .public)") }), - refiner: TranscriptPolisher(), + refiner: router, hotkeyMonitor: trusted ? tapHotkey : carbonHotkey, makePipeline: { s in ProcessorPipeline(processorFactories.map { $0(s) }, onFailure: { id, error in @@ -252,6 +275,10 @@ final class AppModel { overlay = OverlayController(coordinator: coordinator) events.handler = { [weak self] event, at in self?.handle(event, at: at) } proposalRelay.handler = { [weak self] proposal in self?.propose(proposal) } + modelStatusRelay.handler = { [weak self] file, status in + guard let self, ModelFile(for: self.settings.polishModel) == file else { return } + self.polishModelStatus = status + } } /// `PLADDER_SETTINGS_PATH` points a copy launched for testing at a file @@ -302,9 +329,59 @@ final class AppModel { } overlay.start() coordinator.start() + applyPolishModel() startPermissionMirroring() } + // MARK: Polish model + + /// Points the coordinator's refiner at the chosen model. An S1-mini file + /// is downloaded only while polish is on and that model is chosen, the + /// one network use besides the speech model's; it is loaded at the first + /// key-down, not here, and freed when polish goes off or another model + /// is picked. + private func applyPolishModel() { + guard let file = ModelFile(for: settings.polishModel) else { + releaseS1Mini() + polishRouter.use(applePolisher) + polishModelStatus = nil + return + } + if s1MiniPolisher?.file != file { + releaseS1Mini() + let polisher = S1MiniPolisher(file: file, location: modelFiles.location(of: file)) + s1MiniPolisher = polisher + polishRouter.use(polisher) + } + let polishing = settings.polishDictations + if !polishing, let polisher = s1MiniPolisher { + Task { await polisher.unload() } + } + if polishing { + Task.detached(priority: .utility) { await S1MiniPolisher.warmUpRuntime() } + } + let files = modelFiles + Task { [weak self] in + if polishing { await files.ensure(file) } + let status = await files.status(of: file) + guard let self, ModelFile(for: self.settings.polishModel) == file else { return } + self.polishModelStatus = status + } + } + + /// Tries a failed download again; the picker's "Try Again". + func retryPolishModelDownload() { + guard let file = ModelFile(for: settings.polishModel) else { return } + let files = modelFiles + Task { await files.ensure(file) } + } + + private func releaseS1Mini() { + guard let polisher = s1MiniPolisher else { return } + s1MiniPolisher = nil + Task { await polisher.unload() } + } + func stop() { permissionTask?.cancel() overlay.stop() @@ -335,7 +412,7 @@ final class AppModel { let total = Self.seconds(instant - released) // A polish cycle gets its own line, so the plain one stays the // number the benchmark rule watches; one predicate finds both. - let polish = timing.polish.map { "polish \(fmt($0)), " } ?? "" + let polish = timing.polish.map { "polish \(fmt($0)) (\(settings.polishModel.rawValue)), " } ?? "" let label = timing.polish == nil ? "release-to-paste" : "polished release-to-paste" let stages = "stop \(fmt(timing.captureStop)), engine \(fmt(timing.engine)), " + "process \(fmt(timing.processing)), \(polish)paste \(fmt(timing.insert))" @@ -618,6 +695,18 @@ final class AppModel { } } +/// Bridges `ModelFiles`' status changes onto the main actor, and lets the +/// files be set up before `self` exists, as `EventRelay` does for the +/// coordinator. +@MainActor +final class ModelStatusRelay { + var handler: ((ModelFile, ModelFileStatus) -> Void)? + + nonisolated func send(_ file: ModelFile, _ status: ModelFileStatus) { + Task { @MainActor in self.handler?(file, status) } + } +} + /// Bridges the learner's proposals onto the main actor, and lets the learner /// be built before `self` exists, as `EventRelay` does for the coordinator. @MainActor diff --git a/Sources/Pladder/PolishRouter.swift b/Sources/Pladder/PolishRouter.swift new file mode 100644 index 0000000..330e160 --- /dev/null +++ b/Sources/Pladder/PolishRouter.swift @@ -0,0 +1,26 @@ +import PladderCore +import Synchronization + +/// The coordinator's one refiner. The coordinator is built once and keeps +/// its refiner for life, while the polish model can be switched at any time, +/// so this hands each call to the model the settings name. The switch is a +/// lock around one reference; the call itself runs outside it. +final class PolishRouter: TranscriptRefiner { + private let current: Mutex + + init(_ initial: any TranscriptRefiner) { + current = Mutex(initial) + } + + func use(_ refiner: any TranscriptRefiner) { + current.withLock { $0 = refiner } + } + + func prepare() async { + await current.withLock { $0 }.prepare() + } + + func refine(_ text: String) async -> String? { + await current.withLock { $0 }.refine(text) + } +} diff --git a/Sources/Pladder/Resources/Localizable.xcstrings b/Sources/Pladder/Resources/Localizable.xcstrings index 67c1f78..2e3b773 100644 --- a/Sources/Pladder/Resources/Localizable.xcstrings +++ b/Sources/Pladder/Resources/Localizable.xcstrings @@ -1421,32 +1421,142 @@ } } }, - "Before the text is pasted, Apple Intelligence cleans it up on this Mac: self-corrections, spoken punctuation and numbers, lists. This adds a second or two to every dictation, and anything the model cannot fix is pasted as dictated.": { + "Spoken punctuation": { "localizations": { "de": { "stringUnit": { "state": "translated", - "value": "Bevor der Text eingefügt wird, bereinigt Apple Intelligence ihn auf diesem Mac: Selbstkorrekturen, gesprochene Satzzeichen und Zahlen, Listen. Das fügt jedem Diktat ein bis zwei Sekunden hinzu, und alles, was das Modell nicht beheben kann, wird so eingefügt, wie es diktiert wurde." + "value": "Gesprochene Satzzeichen" } } } }, - "Spoken punctuation": { + "Turns spoken marks such as “comma”, “question mark” and “new paragraph” into the marks themselves, in English, German and Spanish.": { "localizations": { "de": { "stringUnit": { "state": "translated", - "value": "Gesprochene Satzzeichen" + "value": "Macht aus gesprochenen Zeichen wie „Komma“, „Fragezeichen“ und „neuer Absatz“ die Zeichen selbst, auf Englisch, Deutsch und Spanisch." } } } }, - "Turns spoken marks such as “comma”, “question mark” and “new paragraph” into the marks themselves, in English, German and Spanish.": { + "Model": { "localizations": { "de": { "stringUnit": { "state": "translated", - "value": "Macht aus gesprochenen Zeichen wie „Komma“, „Fragezeichen“ und „neuer Absatz“ die Zeichen selbst, auf Englisch, Deutsch und Spanisch." + "value": "Modell" + } + } + } + }, + "Apple Intelligence": { + "localizations": { + "de": { + "stringUnit": { + "state": "translated", + "value": "Apple Intelligence" + } + } + } + }, + "S1-mini by Superwhisper (1.5 GB)": { + "localizations": { + "de": { + "stringUnit": { + "state": "translated", + "value": "S1-mini von Superwhisper (1,5 GB)" + } + } + } + }, + "S1-mini by Superwhisper, 8-bit (805 MB)": { + "localizations": { + "de": { + "stringUnit": { + "state": "translated", + "value": "S1-mini von Superwhisper, 8 Bit (805 MB)" + } + } + } + }, + "Downloads when polish is on.": { + "localizations": { + "de": { + "stringUnit": { + "state": "translated", + "value": "Wird geladen, sobald „Diktate polieren“ an ist." + } + } + } + }, + "Downloading %@…": { + "localizations": { + "de": { + "stringUnit": { + "state": "translated", + "value": "%@ wird geladen…" + } + } + } + }, + "Checking the download…": { + "localizations": { + "de": { + "stringUnit": { + "state": "translated", + "value": "Download wird geprüft…" + } + } + } + }, + "Try Again": { + "localizations": { + "de": { + "stringUnit": { + "state": "translated", + "value": "Erneut versuchen" + } + } + } + }, + "The download stopped, so dictations paste as dictated.": { + "localizations": { + "de": { + "stringUnit": { + "state": "translated", + "value": "Der Download wurde abgebrochen, Diktate werden deshalb unverändert eingefügt." + } + } + } + }, + "The download did not match the published model and was deleted, so dictations paste as dictated.": { + "localizations": { + "de": { + "stringUnit": { + "state": "translated", + "value": "Der Download passte nicht zum veröffentlichten Modell und wurde gelöscht, Diktate werden deshalb unverändert eingefügt." + } + } + } + }, + "The model could not be saved, probably for lack of disk space, so dictations paste as dictated.": { + "localizations": { + "de": { + "stringUnit": { + "state": "translated", + "value": "Das Modell konnte nicht gespeichert werden, vermutlich fehlt Speicherplatz. Diktate werden deshalb unverändert eingefügt." + } + } + } + }, + "Before the text is pasted, the model cleans it up on this Mac: self-corrections, spoken punctuation, lists. Apple Intelligence adds one to two seconds to every dictation, S1-mini by Superwhisper about half a second. S1-mini is trained on English, also handles German and Spanish, and is downloaded once from Hugging Face when you pick it. Anything a model cannot fix is pasted as dictated.": { + "localizations": { + "de": { + "stringUnit": { + "state": "translated", + "value": "Bevor der Text eingefügt wird, bereinigt ihn das Modell auf diesem Mac: Selbstkorrekturen, gesprochene Satzzeichen, Listen. Apple Intelligence braucht dafür ein bis zwei Sekunden pro Diktat, S1-mini von Superwhisper etwa eine halbe. S1-mini ist auf Englisch trainiert, kann auch Deutsch und Spanisch und wird einmalig von Hugging Face geladen, wenn du es auswählst. Was ein Modell nicht bereinigen kann, wird unverändert eingefügt." } } } diff --git a/Sources/Pladder/Settings/SettingsView.swift b/Sources/Pladder/Settings/SettingsView.swift index 3b23619..3587678 100644 --- a/Sources/Pladder/Settings/SettingsView.swift +++ b/Sources/Pladder/Settings/SettingsView.swift @@ -1,6 +1,7 @@ import AVFoundation import SwiftUI import PladderCore +import PladderRefine import PladderSystem @@ -352,21 +353,60 @@ private struct ProcessingSettingsView: View { Section { Toggle("Polish dictations", isOn: $model.settings.polishDictations) - if let warning = model.polishAvailability.polishKeyText { - Label(warning, systemImage: "exclamationmark.triangle") - .font(.callout) - .foregroundStyle(.orange) - .fixedSize(horizontal: false, vertical: true) + Picker("Model", selection: $model.settings.polishModel) { + ForEach(PolishModel.allCases, id: \.self) { choice in + Text(choice.displayName).tag(choice) + } } + polishModelStatus } header: { Text("Experimental") } footer: { - FootnoteText("Before the text is pasted, Apple Intelligence cleans it up on this Mac: self-corrections, spoken punctuation and numbers, lists. This adds a second or two to every dictation, and anything the model cannot fix is pasted as dictated.") + FootnoteText("Before the text is pasted, the model cleans it up on this Mac: self-corrections, spoken punctuation, lists. Apple Intelligence adds one to two seconds to every dictation, S1-mini by Superwhisper about half a second. S1-mini is trained on English, also handles German and Spanish, and is downloaded once from Hugging Face when you pick it. Anything a model cannot fix is pasted as dictated.") } } .formStyle(.grouped) } + /// What stands between the chosen model and a polished dictation: + /// Apple Intelligence switched off, or an S1-mini file still to come. + @ViewBuilder + private var polishModelStatus: some View { + if let status = model.polishModelStatus { + switch status { + case .ready: + EmptyView() + case .missing: + FootnoteText("Downloads when polish is on.") + case .downloading(let fraction): + ProgressView(value: fraction) { + Text("Downloading \(model.settings.polishModel.displayName)…") + .font(.callout) + } currentValueLabel: { + Text(fraction, format: .percent.precision(.fractionLength(0))) + } + case .verifying: + ProgressView { + Text("Checking the download…").font(.callout) + } + case .failed(let failure): + HStack(alignment: .firstTextBaseline) { + Label(failure.text, systemImage: "exclamationmark.triangle") + .font(.callout) + .foregroundStyle(.orange) + .fixedSize(horizontal: false, vertical: true) + Spacer() + Button("Try Again") { model.retryPolishModelDownload() } + } + } + } else if let warning = model.polishAvailability.polishKeyText { + Label(warning, systemImage: "exclamationmark.triangle") + .font(.callout) + .foregroundStyle(.orange) + .fixedSize(horizontal: false, vertical: true) + } + } + /// Settings stores the *disabled* IDs, so absence means on. private func binding(for id: String) -> Binding { Binding( diff --git a/Sources/Pladder/StatusText.swift b/Sources/Pladder/StatusText.swift index c43b69d..2c564cf 100644 --- a/Sources/Pladder/StatusText.swift +++ b/Sources/Pladder/StatusText.swift @@ -84,3 +84,29 @@ extension OnDeviceModelAvailability { } } } + +extension PolishModel { + /// The picker's wording. S1-mini's licence asks for its name exactly + /// so: "S1-mini" by "Superwhisper". + var displayName: String { + switch self { + case .appleIntelligence: String(localized: "Apple Intelligence") + case .s1Mini: String(localized: "S1-mini by Superwhisper (1.5 GB)") + case .s1Mini8Bit: String(localized: "S1-mini by Superwhisper, 8-bit (805 MB)") + } + } +} + +extension ModelFileFailure { + /// Each says what dictations do meanwhile, like the Apple sentences. + var text: String { + switch self { + case .download: + String(localized: "The download stopped, so dictations paste as dictated.") + case .checksum: + String(localized: "The download did not match the published model and was deleted, so dictations paste as dictated.") + case .disk: + String(localized: "The model could not be saved, probably for lack of disk space, so dictations paste as dictated.") + } + } +} diff --git a/Sources/PladderCore/Models/Settings.swift b/Sources/PladderCore/Models/Settings.swift index 5746868..3915fd1 100644 --- a/Sources/PladderCore/Models/Settings.swift +++ b/Sources/PladderCore/Models/Settings.swift @@ -32,6 +32,18 @@ public enum OverlayAnimationSpeed: String, Codable, Sendable, CaseIterable, Equa } } +/// Which model the polish runs through. Pure data; PladderRefine maps each +/// case to a model and the app words it. +public enum PolishModel: String, Codable, Sendable, CaseIterable, Equatable { + /// Apple's on-device model, part of macOS: nothing to download. + case appleIntelligence + /// S1-mini by Superwhisper at full precision (16-bit), downloaded once. + case s1Mini + /// The same model at 8-bit: about half the download and memory, and + /// faster, for a little accuracy (docs/BENCHMARKS.md). + case s1Mini8Bit +} + /// Everything the user can change. Persisted as JSON by `SettingsStore`. public struct Settings: Codable, Sendable, Equatable { public var engineID: EngineID @@ -44,6 +56,8 @@ public struct Settings: Codable, Sendable, Equatable { /// Experimental and off by default: the model costs one to three seconds, /// so this sits on the normal hotkey's release path. public var polishDictations: Bool + /// What `polishDictations` runs the text through. + public var polishModel: PolishModel /// A chord that starts a recording on one press and ends it on the next. /// The same chord as `hotkey` makes that key hybrid: a tap latches, a hold /// stops at release. Empty, the default, turns it off. @@ -74,6 +88,7 @@ public struct Settings: Codable, Sendable, Equatable { hotkey: Hotkey = .optionSpace, submitKey: Hotkey = .rightOption, polishDictations: Bool = false, + polishModel: PolishModel = .appleIntelligence, toggleHotkey: Hotkey = Hotkey(keyCodes: []), disabledProcessors: Set = [], dictionary: [DictionaryEntry] = [], @@ -90,6 +105,7 @@ public struct Settings: Codable, Sendable, Equatable { self.hotkey = hotkey self.submitKey = submitKey self.polishDictations = polishDictations + self.polishModel = polishModel self.toggleHotkey = toggleHotkey self.disabledProcessors = disabledProcessors self.dictionary = dictionary @@ -106,7 +122,7 @@ public struct Settings: Codable, Sendable, Equatable { // Decoding tolerates missing keys so adding a field in a later version // never makes an existing settings file unreadable. private enum CodingKeys: String, CodingKey { - case engineID, hotkey, submitKey, polishDictations, disabledProcessors, dictionary, appendTrailingSpace, launchAtLogin, playSounds, appearance + case engineID, hotkey, submitKey, polishDictations, polishModel, disabledProcessors, dictionary, appendTrailingSpace, launchAtLogin, playSounds, appearance case overlayStyle, overlayGlass, overlayAnimationSpeed, muteOutputWhileDictating case toggleHotkey // Read once for the migration, never written. @@ -128,6 +144,9 @@ public struct Settings: Codable, Sendable, Equatable { let legacyPolish = try c.decodeIfPresent(Hotkey.self, forKey: .polishHotkey) ?? Hotkey(keyCodes: []) polishDictations = try c.decodeIfPresent(Bool.self, forKey: .polishDictations) ?? (!legacyPolish.isEmpty) + // A model this build does not know, say from a newer version, falls + // back to Apple's rather than making the whole file unreadable. + polishModel = (try? c.decodeIfPresent(PolishModel.self, forKey: .polishModel)) ?? .appleIntelligence // Like the submit key, empty is meaningful: off. toggleHotkey = try c.decodeIfPresent(Hotkey.self, forKey: .toggleHotkey) ?? Hotkey(keyCodes: []) disabledProcessors = try c.decodeIfPresent(Set.self, forKey: .disabledProcessors) ?? [] @@ -150,6 +169,7 @@ public struct Settings: Codable, Sendable, Equatable { try c.encode(hotkey, forKey: .hotkey) try c.encode(submitKey, forKey: .submitKey) try c.encode(polishDictations, forKey: .polishDictations) + try c.encode(polishModel, forKey: .polishModel) try c.encode(toggleHotkey, forKey: .toggleHotkey) try c.encode(disabledProcessors, forKey: .disabledProcessors) try c.encode(dictionary, forKey: .dictionary) diff --git a/Sources/PladderRefine/LlamaModel.swift b/Sources/PladderRefine/LlamaModel.swift new file mode 100644 index 0000000..11f3d92 --- /dev/null +++ b/Sources/PladderRefine/LlamaModel.swift @@ -0,0 +1,173 @@ +import Foundation +import llama + +/// One GGUF model and one context on llama.cpp, generating greedily. +/// +/// Every llama.cpp call runs on the model's own serial queue: a decode blocks +/// its thread for hundreds of milliseconds, which must not happen on Swift's +/// cooperative pool, and the context is not thread-safe. Callers await. +/// +/// The prompt is split into a fixed prefix (the system prompt and whatever +/// never changes) and the part that does. The prefix is decoded once, at +/// load, and its key-value cache kept, so a dictation only pays for its own +/// tokens. +final class LlamaModel: @unchecked Sendable { + enum Failure: Error, Equatable { + case load + case tokenize + case decode(Int32) + case contextFull + case timedOut + } + + /// Context length in tokens. The key-value cache is allocated for all of + /// it up front (about 114 KB a token for Qwen3-0.6B), so it is sized for + /// one chunk of a long dictation, in and out, not for the model's limit. + static let contextLength: UInt32 = 2048 + + private let queue = DispatchQueue(label: "de.dinooo13.pladder.llama", qos: .userInitiated) + // Touched on `queue` only. + private let model: OpaquePointer + private let context: OpaquePointer + private let vocab: OpaquePointer + private let sampler: UnsafeMutablePointer + private let prefix: [llama_token] + + /// Loads the model onto the GPU and decodes `prefix`. Blocking; call from + /// `load(path:prefix:)`. + private init(path: String, prefix: String) throws { + _ = Self.backend + var modelParams = llama_model_default_params() + modelParams.n_gpu_layers = -1 + guard let model = llama_model_load_from_file(path, modelParams) else { throw Failure.load } + var contextParams = llama_context_default_params() + contextParams.n_ctx = Self.contextLength + contextParams.n_batch = Self.contextLength + contextParams.no_perf = true + guard let context = llama_init_from_model(model, contextParams) else { + llama_model_free(model) + throw Failure.load + } + self.model = model + self.context = context + vocab = llama_model_get_vocab(model) + sampler = llama_sampler_chain_init(llama_sampler_chain_default_params()) + llama_sampler_chain_add(sampler, llama_sampler_init_greedy()) + self.prefix = try Self.tokenize(prefix, vocab: vocab) + try decode(self.prefix) + } + + deinit { + llama_sampler_free(sampler) + llama_free(context) + llama_model_free(model) + } + + /// Loads `path` and decodes `prefix` off the caller's thread. + static func load(path: String, prefix: String) async throws -> LlamaModel { + try await withCheckedThrowingContinuation { continuation in + loadQueue.async { + continuation.resume(with: Result { try LlamaModel(path: path, prefix: prefix) }) + } + } + } + + /// Greedy continuation of the prefix with `suffix`, up to an end-of-turn + /// token, `maxTokens` or `deadline`, whichever comes first. The deadline + /// is checked between tokens, so an answer can overrun it by one token. + func complete(suffix: String, maxTokens: Int, deadline: ContinuousClock.Instant) async throws -> String { + try await withCheckedThrowingContinuation { continuation in + queue.async { + continuation.resume(with: Result { + try self.completeNow(suffix: suffix, maxTokens: maxTokens, deadline: deadline) + }) + } + } + } + + // MARK: On the queue + + private func completeNow(suffix: String, maxTokens: Int, deadline: ContinuousClock.Instant) throws -> String { + // Back to the prefix: the previous dictation's tokens go, the + // prefix's cache stays. + let memory = llama_get_memory(context) + if !llama_memory_seq_rm(memory, 0, llama_pos(prefix.count), -1) { + llama_memory_clear(memory, true) + try decode(prefix) + } + llama_sampler_reset(sampler) + + let input = try Self.tokenize(suffix, vocab: vocab) + let room = Int(Self.contextLength) - prefix.count - input.count + guard room > 0 else { throw Failure.contextFull } + guard ContinuousClock.now < deadline else { throw Failure.timedOut } + try decode(input) + + var bytes: [CChar] = [] + var piece = [CChar](repeating: 0, count: 256) + for _ in 0.. 0 { bytes.append(contentsOf: piece[0..` are parsed, since the chat + /// format is written out by hand; no beginning-of-text token is added, + /// which the Qwen family does not use. + static func tokenize(_ text: String, vocab: OpaquePointer) throws -> [llama_token] { + let utf8 = Array(text.utf8CString.dropLast()) + var tokens = [llama_token](repeating: 0, count: utf8.count + 8) + var count = llama_tokenize(vocab, utf8, Int32(utf8.count), &tokens, Int32(tokens.count), false, true) + if count < 0 { + tokens = [llama_token](repeating: 0, count: Int(-count)) + count = llama_tokenize(vocab, utf8, Int32(utf8.count), &tokens, Int32(tokens.count), false, true) + } + guard count >= 0 else { throw Failure.tokenize } + return Array(tokens.prefix(Int(count))) + } + + // MARK: Process-wide + + /// Sets up the backends ahead of the first load. The first time a build + /// of llama.cpp runs, Metal compiles its shaders, about seven seconds on + /// an M1; macOS caches them after that, so only the first launch after + /// an install or update pays, and with this it pays in the background. + static func warmUp() async { + await withCheckedContinuation { continuation in + loadQueue.async { + _ = backend + continuation.resume() + } + } + } + + /// Loads are rare and slow; they get a queue of their own so a load never + /// waits behind another model's generation. + private static let loadQueue = DispatchQueue(label: "de.dinooo13.pladder.llama.load", qos: .userInitiated) + + /// Once per process, before the first model: the backends, and llama.cpp's + /// own logging, which would otherwise print every tensor to stderr. + private static let backend: Void = { + llama_log_set({ _, _, _ in }, nil) + llama_backend_init() + }() +} diff --git a/Sources/PladderRefine/ModelFiles.swift b/Sources/PladderRefine/ModelFiles.swift new file mode 100644 index 0000000..8843511 --- /dev/null +++ b/Sources/PladderRefine/ModelFiles.swift @@ -0,0 +1,206 @@ +import CryptoKit +import Foundation +import PladderCore + +/// A model file the polish can download: where it comes from, pinned to a +/// commit so it can never change under us, and the checksum it must match +/// before it is used. +public struct ModelFile: Sendable, Equatable { + public let fileName: String + public let url: URL + public let sha256: String + public let byteCount: Int64 + + /// S1-mini by Superwhisper at 16-bit, from Superwhisper's own repository. + public static let s1MiniFullPrecision = ModelFile( + fileName: "s1-mini-f16.gguf", + url: URL(string: "https://huggingface.co/superwhisper/s1-mini-GGUF/resolve/34add00a48a2e5d24e5a4ee5405a99620a3a240c/s1-mini-f16.gguf")!, + sha256: "0370da4f1bae19e3150bcafa33c5d396c15f97bf25519540a3e013db5cc00af4", + byteCount: 1_509_347_232) + + /// The same weights at Q8_0. Superwhisper publishes only 16-bit and + /// 4-bit files, and 4-bit lost German and Spanish on the polish set, so + /// this is mradermacher's quantisation of the same release. + public static let s1Mini8Bit = ModelFile( + fileName: "s1-mini.Q8_0.gguf", + url: URL(string: "https://huggingface.co/mradermacher/s1-mini-GGUF/resolve/f46488282bc2417789271ea5dab6b85c423f9439/s1-mini.Q8_0.gguf")!, + sha256: "19ddecf5dd46cb37ea13ce78d013c197b39c0bc69b880ea0b3f7c16528ddb52e", + byteCount: 804_754_240) + + /// The file a polish model needs; nil for Apple's, which ships with macOS. + public init?(for model: PolishModel) { + switch model { + case .appleIntelligence: return nil + case .s1Mini: self = .s1MiniFullPrecision + case .s1Mini8Bit: self = .s1Mini8Bit + } + } + + /// Internal for the tests, which download a local file. + init(fileName: String, url: URL, sha256: String, byteCount: Int64) { + self.fileName = fileName + self.url = url + self.sha256 = sha256 + self.byteCount = byteCount + } +} + +/// Where a model file stands. The app words it. +public enum ModelFileStatus: Sendable, Equatable { + case missing + case downloading(fraction: Double) + /// Downloaded, the checksum is being computed. + case verifying + case ready + case failed(ModelFileFailure) +} + +public enum ModelFileFailure: Error, Sendable, Equatable { + /// No connection, a server error, or the download stopped. + case download + /// The file arrived but is not the one pinned: never used, deleted. + case checksum + /// It could not be written, usually a full disk. + case disk +} + +/// The downloaded model files, one directory, one download at a time per +/// file. This is the polish's one network use: the one-time download from +/// Hugging Face, started only when the user picks a model that needs it. +public actor ModelFiles { + public let directory: URL + private let onChange: @Sendable (ModelFile, ModelFileStatus) -> Void + private var statuses: [String: ModelFileStatus] = [:] + private var downloads: [String: Task] = [:] + + /// `onChange` is called on every status change, off the main actor. + public init(directory: URL, onChange: @escaping @Sendable (ModelFile, ModelFileStatus) -> Void = { _, _ in }) { + self.directory = directory + self.onChange = onChange + } + + /// `~/Library/Application Support/Pladder/Models`, beside the settings. + public static var defaultDirectory: URL { + FileManager.default.homeDirectoryForCurrentUser + .appending(path: "Library/Application Support/Pladder/Models") + } + + public nonisolated func location(of file: ModelFile) -> URL { + directory.appending(path: file.fileName) + } + + /// A file on disk is ready: it is only ever moved into place after its + /// checksum matched. + public func status(of file: ModelFile) -> ModelFileStatus { + if let status = statuses[file.fileName] { return status } + return FileManager.default.fileExists(atPath: location(of: file).path) ? .ready : .missing + } + + /// Starts the download unless the file is there or on its way. A failed + /// download is tried again. + public func ensure(_ file: ModelFile) { + switch status(of: file) { + case .ready, .downloading, .verifying: return + case .missing, .failed: break + } + set(file, .downloading(fraction: 0)) + downloads[file.fileName] = Task { await self.download(file) } + } + + /// Waits for a download `ensure` started; returns the final status. + public func finished(_ file: ModelFile) async -> ModelFileStatus { + await downloads[file.fileName]?.value + return status(of: file) + } + + private func set(_ file: ModelFile, _ status: ModelFileStatus) { + statuses[file.fileName] = status + onChange(file, status) + } + + private func download(_ file: ModelFile) async { + defer { downloads[file.fileName] = nil } + let staging = directory.appending(path: file.fileName + ".download") + do { + try FileManager.default.createDirectory(at: directory, withIntermediateDirectories: true) + } catch { + return set(file, .failed(.disk)) + } + do { + try await fetch(file, to: staging) + } catch let failure as ModelFileFailure { + return set(file, .failed(failure)) + } catch { + return set(file, .failed(.download)) + } + set(file, .verifying) + let digest = await Task.detached(priority: .utility) { Self.sha256(of: staging) }.value + guard digest == file.sha256 else { + try? FileManager.default.removeItem(at: staging) + return set(file, .failed(.checksum)) + } + do { + try? FileManager.default.removeItem(at: location(of: file)) + try FileManager.default.moveItem(at: staging, to: location(of: file)) + } catch { + return set(file, .failed(.disk)) + } + statuses[file.fileName] = nil + onChange(file, .ready) + } + + /// A plain download task, polled for progress. The file lands in + /// `staging` before the completion handler returns, since URLSession + /// deletes its temporary file right after. + private func fetch(_ file: ModelFile, to staging: URL) async throws { + let session = URLSession(configuration: .ephemeral) + defer { session.finishTasksAndInvalidate() } + let task: URLSessionDownloadTask + let result = AsyncStream>.makeStream() + task = session.downloadTask(with: file.url) { temporary, response, error in + // A file URL (the tests) has no status code. + let succeeded = (response as? HTTPURLResponse).map { (200..<300).contains($0.statusCode) } ?? true + guard error == nil, let temporary, succeeded else { + result.continuation.yield(.failure(.download)) + return + } + do { + try? FileManager.default.removeItem(at: staging) + try FileManager.default.moveItem(at: temporary, to: staging) + result.continuation.yield(.success(())) + } catch { + result.continuation.yield(.failure(.disk)) + } + } + task.resume() + let progress = Task { + while !Task.isCancelled { + try? await Task.sleep(for: .milliseconds(250)) + // Woken by the cancel: the download is over, and a late + // report would overwrite what came after it. + guard !Task.isCancelled else { return } + let received = task.countOfBytesReceived + if received > 0 { + set(file, .downloading(fraction: min(1, Double(received) / Double(file.byteCount)))) + } + } + } + defer { progress.cancel() } + for await outcome in result.stream { + try outcome.get() + return + } + throw ModelFileFailure.download + } + + /// Streamed, so a 1.5 GB file never sits in memory. + private static func sha256(of url: URL) -> String? { + guard let handle = try? FileHandle(forReadingFrom: url) else { return nil } + defer { try? handle.close() } + var hasher = SHA256() + while let chunk = try? handle.read(upToCount: 16 << 20), !chunk.isEmpty { + hasher.update(data: chunk) + } + return hasher.finalize().map { String(format: "%02x", $0) }.joined() + } +} diff --git a/Sources/PladderRefine/S1MiniPolisher.swift b/Sources/PladderRefine/S1MiniPolisher.swift new file mode 100644 index 0000000..f23acd4 --- /dev/null +++ b/Sources/PladderRefine/S1MiniPolisher.swift @@ -0,0 +1,155 @@ +import Foundation +import PladderCore +import os + +/// The polish on S1-mini by Superwhisper: Qwen3-0.6B fine-tuned to clean +/// dictation (fillers, self-corrections, spoken punctuation, numbers, lists), +/// run by llama.cpp on the GPU. +/// +/// Trained on English only, it still resolved German and Spanish +/// self-corrections and lists on the polish set, and translated nothing, +/// where Apple's model turned two of them into English (docs/BENCHMARKS.md). +/// Its prompt is the one it was trained on, not `TranscriptPolisher`'s: a +/// fixed system line and a control line, then the transcript. +/// +/// Best effort like the Apple polisher: no file yet, a load that fails, the +/// time budget, an empty answer all return nil and the coordinator pastes +/// the text as dictated. The model stays loaded from the first `prepare()` +/// until `unload()`, as the speech engine does. +public actor S1MiniPolisher: TranscriptRefiner { + /// S1-mini's system prompt, word for word from its model card; it is + /// part of the input format the model was trained on. + static let systemPrompt = "You are a text normalizer for speech-to-text transcripts. The input begins with a control line specifying the styling, structure, and context settings; clean the transcript to match those settings and output only the cleaned text." + + /// Semi-formal keeps the capitals a dictation into any app wants + /// (semi-casual lower-cases sentence starts, formal rewrites "let's" as + /// "let us"), and lists lets spoken enumerations become lines without + /// forcing a list where there is none. Chosen on the polish set. + static let controlLine = "[Styling: semi-formal] [Structure: lists] [Context: general]" + + /// The part of the prompt that never changes, decoded once at load. + /// Qwen's chat format, written out: the model's template would add the + /// same, and a thinking block must stay empty (`enable_thinking=False`). + static let promptPrefix = "<|im_start|>system\n\(systemPrompt)<|im_end|>\n<|im_start|>user\n\(controlLine)\n" + + static func promptSuffix(for transcript: String) -> String { + "\(transcript)<|im_end|>\n<|im_start|>assistant\n\n\n\n\n" + } + + /// The context holds 2,048 tokens for the prompt, the chunk and its + /// answer, so chunks are smaller than the Apple polisher's. + static let chunkThreshold = 400 + static let chunkSize = 250 + + public nonisolated let file: ModelFile + private let location: URL + private let timeout: Duration + private var model: LlamaModel? + private var loading: Task? + private static let log = Logger(subsystem: "de.dinooo13.pladder", category: "polish") + + /// `location` is where `ModelFiles` keeps `file`. + public init(file: ModelFile, location: URL, timeout: Duration = .seconds(8)) { + self.file = file + self.location = location + self.timeout = timeout + } + + /// Loads the model if its file is there. Called at key-down, so a first + /// dictation's load happens while the user is still speaking. + public func prepare() async { + _ = await loaded() + } + + public func refine(_ text: String) async -> String? { + await polish(text).text + } + + /// Sets up llama.cpp without loading a model: cheap except on the first + /// launch after an install, when Metal compiles its shaders for seconds. + /// The app calls it as soon as S1-mini is chosen, so no dictation waits. + public static func warmUpRuntime() async { + await LlamaModel.warmUp() + } + + /// Frees the model's memory; the next `prepare()` loads it again. + public func unload() { + loading?.cancel() + loading = nil + model = nil + } + + /// `refine` with the numbers kept, for the log and the CLI. One log line + /// per call, numbers only: the transcript never goes in the log. + public func polish(_ text: String) async -> TranscriptPolisher.Report { + let started = ContinuousClock.now + let deadline = started + timeout + let pieces = TranscriptPolisher.chunks(of: text, threshold: Self.chunkThreshold, size: Self.chunkSize) + var report = TranscriptPolisher.Report( + text: nil, elapsed: .zero, wordsIn: TranscriptPolisher.wordCount(text), wordsOut: 0, + chunks: pieces.count, mode: .completion, failure: nil) + if let model = await loaded() { + do { + var cleaned: [String] = [] + for piece in pieces { + // Room for an answer somewhat longer than the input: a + // list adds line breaks and dashes. + let maxTokens = TranscriptPolisher.wordCount(piece) * 3 + 64 + let answer = try await model.complete( + suffix: Self.promptSuffix(for: piece), maxTokens: maxTokens, deadline: deadline) + let filtered = PolishPostFilter.clean(answer) + guard !filtered.isEmpty else { throw EmptyAnswer() } + cleaned.append(filtered) + } + let joined = cleaned.joined(separator: " ") + report.text = joined + report.wordsOut = TranscriptPolisher.wordCount(joined) + } catch LlamaModel.Failure.timedOut { + report.failure = "timed out" + } catch is EmptyAnswer { + report.failure = "empty answer" + } catch { + report.failure = "model error \(error)" + } + } else { + report.failure = FileManager.default.fileExists(atPath: location.path) ? "model did not load" : "not downloaded" + } + report.elapsed = ContinuousClock.now - started + Self.log(report, file: file) + return report + } + + private struct EmptyAnswer: Error {} + + /// The loaded model, loading it first if its file is there; nil + /// otherwise. Concurrent callers share one load. + private func loaded() async -> LlamaModel? { + if let model { return model } + if let loading { return await loading.value } + guard FileManager.default.fileExists(atPath: location.path) else { return nil } + let path = location.path + let task = Task { try? await LlamaModel.load(path: path, prefix: Self.promptPrefix) } + loading = task + let model = await task.value + // `unload()` during the load cancelled this task: drop the result. + if loading == task, !task.isCancelled { self.model = model } + if loading == task { loading = nil } + return self.model + } + + private static func log(_ report: TranscriptPolisher.Report, file: ModelFile) { + let secs = String(format: "%.3f", Double(report.elapsed.components.seconds) + + Double(report.elapsed.components.attoseconds) / 1e18) + if let failure = report.failure { + log.notice( + "polish (\(file.fileName, privacy: .public)) failed after \(secs, privacy: .public) s, \(report.wordsIn, privacy: .public) words in: \(failure, privacy: .public)") + } else { + log.notice( + """ + polish (\(file.fileName, privacy: .public)) \(secs, privacy: .public) s, \ + \(report.wordsIn, privacy: .public) words in, \(report.wordsOut, privacy: .public) out\ + \(report.chunks > 1 ? ", \(report.chunks) chunks" : "", privacy: .public) + """) + } + } +} diff --git a/Sources/PladderRefine/TranscriptPolisher.swift b/Sources/PladderRefine/TranscriptPolisher.swift index 8cfdfd1..092ec30 100644 --- a/Sources/PladderRefine/TranscriptPolisher.swift +++ b/Sources/PladderRefine/TranscriptPolisher.swift @@ -61,6 +61,9 @@ public actor TranscriptPolisher: TranscriptRefiner { case guided /// Plain `respond(to:)`, after guided generation could not decode. case plain + /// A local model completing the prompt format it was trained on + /// (`S1MiniPolisher`). + case completion } /// Nil when the model could not help; paste the input as it is. @@ -175,17 +178,18 @@ public actor TranscriptPolisher: TranscriptRefiner { // MARK: Chunking /// The transcript as it is when it is short, which is most dictations; - /// otherwise windows of about `chunkSize` words, - /// cut after a sentence end so no window starts mid-sentence. - static func chunks(of text: String) -> [String] { + /// otherwise windows of about `size` words (`chunkSize` by default), + /// cut after a sentence end so no window starts mid-sentence. The + /// S1-mini polisher passes its own, smaller numbers. + static func chunks(of text: String, threshold: Int = chunkThreshold, size: Int = chunkSize) -> [String] { let words = text.split(whereSeparator: \.isWhitespace) - guard words.count > chunkThreshold else { return [text] } + guard words.count > threshold else { return [text] } var windows: [String] = [] var current: [Substring] = [] for word in words { current.append(word) let endsSentence = word.last.map { ".?!".contains($0) } ?? false - if current.count >= chunkSize, endsSentence { + if current.count >= size, endsSentence { windows.append(current.joined(separator: " ")) current = [] } diff --git a/Tests/PladderCoreTests/DictationCoordinatorTests.swift b/Tests/PladderCoreTests/DictationCoordinatorTests.swift index 5f23c7f..5c74ca2 100644 --- a/Tests/PladderCoreTests/DictationCoordinatorTests.swift +++ b/Tests/PladderCoreTests/DictationCoordinatorTests.swift @@ -1840,6 +1840,7 @@ final class EventLog: @unchecked Sendable { #expect(decoded.hotkey == .optionSpace) #expect(decoded.submitKey == .rightOption) #expect(!decoded.polishDictations) + #expect(decoded.polishModel == .appleIntelligence) #expect(decoded.appendTrailingSpace == true) #expect(decoded.appearance == .system) #expect(decoded.overlayStyle == .compact) @@ -1856,6 +1857,21 @@ final class EventLog: @unchecked Sendable { #expect(!off.polishDictations) } + @Test func polishModelPersists() throws { + var settings = Settings(engineID: EchoEngine.engineID) + settings.polishModel = .s1Mini8Bit + let decoded = try JSONDecoder().decode(Settings.self, from: JSONEncoder().encode(settings)) + #expect(decoded.polishModel == .s1Mini8Bit) + } + + @Test func anUnknownPolishModelFallsBackToApples() throws { + // Written by a newer build that knows a model this one does not. + let json = #"{"engineID":"echo","polishDictations":true,"polishModel":"someFutureModel"}"# + let decoded = try JSONDecoder().decode(Settings.self, from: Data(json.utf8)) + #expect(decoded.polishModel == .appleIntelligence) + #expect(decoded.polishDictations) + } + @Test func appearancePersists() throws { let dir = FileManager.default.temporaryDirectory.appendingPathComponent(UUID().uuidString) let url = dir.appendingPathComponent("settings.json") diff --git a/Tests/PladderRefineTests/S1MiniTests.swift b/Tests/PladderRefineTests/S1MiniTests.swift new file mode 100644 index 0000000..ce1fe6a --- /dev/null +++ b/Tests/PladderRefineTests/S1MiniTests.swift @@ -0,0 +1,120 @@ +import Foundation +import PladderCore +import Testing +@testable import PladderRefine + +// Nothing here loads a model or touches the network: the downloads come from +// a file URL, and the polisher is asked about a file that is not there. + +@Suite struct S1MiniPromptTests { + @Test func thePromptIsQwensChatFormatWithAnEmptyThinkBlock() { + // What S1-mini's own chat template renders with enable_thinking=False, + // taken from its tokenizer: the model was trained on exactly this. + let prompt = S1MiniPolisher.promptPrefix + S1MiniPolisher.promptSuffix(for: "hello there") + #expect(prompt == """ + <|im_start|>system + \(S1MiniPolisher.systemPrompt)<|im_end|> + <|im_start|>user + [Styling: semi-formal] [Structure: lists] [Context: general] + hello there<|im_end|> + <|im_start|>assistant + + + + + + """) + } + + @Test func theSystemPromptIsTheModelCardsWordForWord() { + #expect(S1MiniPolisher.systemPrompt.hasPrefix("You are a text normalizer for speech-to-text transcripts.")) + #expect(S1MiniPolisher.systemPrompt.hasSuffix("output only the cleaned text.")) + } + + @Test func longDictationsAreChunkedSmallerThanForApple() { + let sentence = "One two three four five six seven eight nine ten." + let text = Array(repeating: sentence, count: 50).joined(separator: " ") // 500 words + #expect(TranscriptPolisher.chunks(of: text).count == 1) + let chunks = TranscriptPolisher.chunks( + of: text, threshold: S1MiniPolisher.chunkThreshold, size: S1MiniPolisher.chunkSize) + #expect(chunks.count == 2) + #expect(chunks.joined(separator: " ") == text) + } + + @Test func eachModelNamesItsFile() { + #expect(ModelFile(for: .appleIntelligence) == nil) + #expect(ModelFile(for: .s1Mini) == .s1MiniFullPrecision) + #expect(ModelFile(for: .s1Mini8Bit) == .s1Mini8Bit) + // Pinned to a commit, never a branch. + for file in [ModelFile.s1MiniFullPrecision, .s1Mini8Bit] { + #expect(file.url.host() == "huggingface.co") + #expect(!file.url.path().contains("/resolve/main/")) + #expect(file.sha256.count == 64) + } + } + + @Test func withoutItsFileThePolisherPastesAsDictated() async { + let missing = FileManager.default.temporaryDirectory.appending(path: UUID().uuidString + ".gguf") + let polisher = S1MiniPolisher(file: .s1Mini8Bit, location: missing) + await polisher.prepare() + let report = await polisher.polish("hello there") + #expect(report.text == nil) + #expect(report.failure == "not downloaded") + #expect(await polisher.refine("hello there") == nil) + } +} + +@Suite struct ModelFilesTests { + private func scratch() throws -> URL { + let dir = FileManager.default.temporaryDirectory.appending(path: "ModelFilesTests-" + UUID().uuidString) + try FileManager.default.createDirectory(at: dir, withIntermediateDirectories: true) + return dir + } + + private func source(in dir: URL, contents: String) throws -> URL { + let url = dir.appending(path: "source.bin") + try Data(contents.utf8).write(to: url) + return url + } + + @Test func aDownloadThatMatchesItsChecksumBecomesReady() async throws { + let dir = try scratch() + defer { try? FileManager.default.removeItem(at: dir) } + let file = ModelFile( + fileName: "model.gguf", url: try source(in: dir, contents: "weights"), + // SHA-256 of "weights". + sha256: "9a129038d9a00aed0cf6a7ea059ca50a813449061ab87848cf1a13eafdf33b2c", + byteCount: 7) + let files = ModelFiles(directory: dir.appending(path: "models")) + #expect(await files.status(of: file) == .missing) + await files.ensure(file) + #expect(await files.finished(file) == .ready) + #expect(try String(contentsOf: files.location(of: file), encoding: .utf8) == "weights") + } + + @Test func aDownloadThatDoesNotMatchIsDeletedAndNeverUsed() async throws { + let dir = try scratch() + defer { try? FileManager.default.removeItem(at: dir) } + let file = ModelFile( + fileName: "model.gguf", url: try source(in: dir, contents: "tampered"), + sha256: "9a129038d9a00aed0cf6a7ea059ca50a813449061ab87848cf1a13eafdf33b2c", + byteCount: 8) + let models = dir.appending(path: "models") + let files = ModelFiles(directory: models) + await files.ensure(file) + #expect(await files.finished(file) == .failed(.checksum)) + #expect(!FileManager.default.fileExists(atPath: files.location(of: file).path)) + #expect(try FileManager.default.contentsOfDirectory(atPath: models.path).isEmpty) + } + + @Test func aMissingSourceFailsAsADownload() async throws { + let dir = try scratch() + defer { try? FileManager.default.removeItem(at: dir) } + let file = ModelFile( + fileName: "model.gguf", url: dir.appending(path: "nowhere.bin"), + sha256: String(repeating: "0", count: 64), byteCount: 1) + let files = ModelFiles(directory: dir.appending(path: "models")) + await files.ensure(file) + #expect(await files.finished(file) == .failed(.download)) + } +} diff --git a/scripts/bundle.sh b/scripts/bundle.sh index fb28097..fc28627 100755 --- a/scripts/bundle.sh +++ b/scripts/bundle.sh @@ -43,6 +43,18 @@ rm -rf "$APP" mkdir -p "$CONTENTS/MacOS" "$CONTENTS/Resources" cp "$BIN_DIR/Pladder" "$CONTENTS/MacOS/Pladder" +# llama.cpp, which runs the S1-mini polish, is a dynamic framework (the +# binary target in Package.swift). SwiftPM leaves it beside the binary and +# links with @loader_path; the app keeps it in Contents/Frameworks, where +# the added search path finds it. Apple Silicon only, so the Intel slice +# goes. +mkdir -p "$CONTENTS/Frameworks" +ditto "$BIN_DIR/llama.framework" "$CONTENTS/Frameworks/llama.framework" +LLAMA_BIN="$CONTENTS/Frameworks/llama.framework/Versions/A/llama" +if lipo -archs "$LLAMA_BIN" | grep -q x86_64; then + lipo -remove x86_64 "$LLAMA_BIN" -output "$LLAMA_BIN" +fi +install_name_tool -add_rpath "@executable_path/../Frameworks" "$CONTENTS/MacOS/Pladder" cp "$PLIST" "$CONTENTS/Info.plist" printf 'APPL????' > "$CONTENTS/PkgInfo" # App icon, rendered by scripts/make-icon.swift and packed with iconutil. @@ -78,9 +90,22 @@ if [[ -z "${CODESIGN_IDENTITY:-}" ]]; then CODESIGN_IDENTITY=${CODESIGN_IDENTITY:--} fi echo "Signing with: $CODESIGN_IDENTITY" +# The hardened runtime only loads libraries signed by the app's own team, and +# two ad-hoc signatures count as different teams, so an ad-hoc build would +# refuse its own llama.framework at launch. It goes without the hardened +# runtime; a certificate build keeps it. +RUNTIME=(--options runtime) +if [[ "$CODESIGN_IDENTITY" == "-" ]]; then + RUNTIME=() +fi +# Inside out: the framework first, with the same identity. +codesign --force --sign "$CODESIGN_IDENTITY" \ + ${RUNTIME[@]+"${RUNTIME[@]}"} \ + --timestamp=none \ + "$CONTENTS/Frameworks/llama.framework" codesign --force --sign "$CODESIGN_IDENTITY" \ --entitlements "$ROOT/scripts/Pladder.entitlements" \ - --options runtime \ + ${RUNTIME[@]+"${RUNTIME[@]}"} \ --timestamp=none \ "$APP" From 12c0b25cfa77fb667ec8f86a87a4efc9fed15490 Mon Sep 17 00:00:00 2001 From: Fabian Meyer <44942030+dinooo13@users.noreply.github.com> Date: Sat, 26 Sep 2026 01:37:56 +0200 Subject: [PATCH 4/6] CLI: polish --model, and polish-set over a test set pladder-cli polish takes --model apple|s1-mini|s1-mini-8bit and fetches an S1-mini file into the app's own model directory if needed. polish-set runs a model over docs/polish-set.json, 42 dictations in English, German and Spanish with the text that should be pasted, after the app's processors, warm, and prints every answer, the word error rate per language, exact matches and the polish time. It is how the models in the picker were chosen. Co-Authored-By: Claude Opus 5.5 (1M context) --- Sources/PladderCLI/main.swift | 148 +++++++++++++++++++++++++++++----- docs/polish-set.json | 44 ++++++++++ 2 files changed, 172 insertions(+), 20 deletions(-) create mode 100644 docs/polish-set.json diff --git a/Sources/PladderCLI/main.swift b/Sources/PladderCLI/main.swift index e56f2c8..e968764 100644 --- a/Sources/PladderCLI/main.swift +++ b/Sources/PladderCLI/main.swift @@ -6,6 +6,7 @@ import PladderBench import PladderCore import PladderEngines import PladderRefine +import PladderSystem // Developer tool. // @@ -28,11 +29,17 @@ import PladderRefine // how many there were and what they cost. The // `identical:` column then also proves the live // passes leave the release's windows alone. -// pladder-cli polish run the polisher over a transcript over a transcript -// with Apple's on-device model: once cold, once -// after prepare() and a two-second wait, the way -// a real press warms it. Prints both timings. -// [--instructions ] try another system prompt before committing it. +// pladder-cli polish run the polisher over a transcript: once cold, +// once after prepare() and a two-second wait, the +// way a real press warms it. Prints both timings. +// [--model ] apple (default), s1-mini or s1-mini-8bit. An S1-mini +// file is downloaded first if the app has not yet. +// [--instructions ] Apple only: try another system prompt before +// committing it. +// pladder-cli polish-set run a polish model over a test set (docs/polish-set.json) +// [--model ] after the app's processors, warm, and print each +// answer, the word error rate against the expected +// text per language, exact matches and timings. // // Fixtures are audio files with a sibling .txt holding the spoken script, as // produced by scripts/make-fixtures.sh. @@ -42,7 +49,8 @@ func usage() -> Never { usage: pladder-cli