@@ -93,6 +93,8 @@ pub struct StreamingAsrEngine {
9393 /// 可选标点恢复器(ADR-012 F4-2:重打分未通过的 final 补语义标点;
9494 /// 模型缺失 → None 零开销降级,不阻断 ASR)
9595 punctuator : Option < OfflinePunctuation > ,
96+ /// tokens.txt 单字集合(热词过滤;读取失败 → None 不阻断——仅失去过滤能力)
97+ token_chars : Option < std:: collections:: HashSet < char > > ,
9698}
9799
98100impl StreamingAsrEngine {
@@ -154,6 +156,11 @@ impl StreamingAsrEngine {
154156 sentence_pcm : Vec :: new ( ) ,
155157 rescorer,
156158 punctuator,
159+ // 2026-08-21 热词崩溃修复:tokens.txt 单字集合——领域热词(心理成长
160+ // 种子词等)含 tokens 表外字(焦/冥/哲)时 sherpa-onnx EncodeBase
161+ // 失败仍创建 ContextGraph,greedy_search 解码断言 abort(exit
162+ // 0xffffffff,用户真机日志实证);读取失败 → None(不阻断加载)。
163+ token_chars : load_token_chars ( & models. tokens ) ,
157164 } )
158165 }
159166
@@ -250,7 +257,18 @@ impl StreamingAsrEngine {
250257 . and_then ( |v| v. hotwords_string ( ) ) ;
251258 match hotwords. as_deref ( ) {
252259 Some ( h) if !h. trim ( ) . is_empty ( ) => {
253- self . recognizer . create_stream_with_hotwords ( h)
260+ // TD-032 延伸修复(2026-08-21):热词含 tokens.txt 外字符时
261+ // sherpa-onnx 编码失败仍创建 ContextGraph → greedy_search 解码
262+ // 断言 abort(exit 0xffffffff);过滤后重建,空则回退普通流。
263+ let filtered = match & self . token_chars {
264+ Some ( chars) => filter_hotwords_by_tokens ( h, chars) ,
265+ None => h. to_string ( ) ,
266+ } ;
267+ if filtered. trim ( ) . is_empty ( ) {
268+ self . recognizer . create_stream ( )
269+ } else {
270+ self . recognizer . create_stream_with_hotwords ( & filtered)
271+ }
254272 }
255273 _ => self . recognizer . create_stream ( ) ,
256274 }
@@ -307,6 +325,36 @@ impl StreamingAsrEngine {
307325 }
308326}
309327
328+ /// 读取 tokens.txt 构建单字集合(纯函数;失败 → None)。
329+ ///
330+ /// @ai-context: sherpa-onnx 的 EncodeBase 按字符查 token ID(日志实证
331+ /// "Cannot find ID for token 焦")——只收集单字符 token;
332+ /// 多字符 token(▁/标点/英文词)不参与单字覆盖判断。
333+ fn load_token_chars ( tokens_path : & str ) -> Option < std:: collections:: HashSet < char > > {
334+ std:: fs:: read_to_string ( tokens_path) . ok ( ) . map ( |raw| {
335+ raw. lines ( )
336+ . filter_map ( |l| l. split_whitespace ( ) . next ( ) )
337+ . filter ( |t| t. chars ( ) . count ( ) == 1 )
338+ . flat_map ( |t| t. chars ( ) )
339+ . collect ( )
340+ } )
341+ }
342+
343+ /// 热词 tokens 过滤(纯函数):仅保留所有字符都在 token 集合中的词。
344+ ///
345+ /// @ai-context: 词级剔除("冥想"含非法字"冥" → 整词剔除,语义完整);
346+ /// 全部被剔 → 空串(调用方回退普通流,防 ContextGraph 崩溃)。
347+ fn filter_hotwords_by_tokens (
348+ hotwords : & str ,
349+ token_chars : & std:: collections:: HashSet < char > ,
350+ ) -> String {
351+ hotwords
352+ . split_whitespace ( )
353+ . filter ( |w| w. chars ( ) . all ( |c| token_chars. contains ( & c) ) )
354+ . collect :: < Vec < _ > > ( )
355+ . join ( " " )
356+ }
357+
310358// 兼容 re-export:编辑距离(asr_rescore.rs 实现;dtw_align/subtitle_ocr/fusion 引用此路径)。
311359pub use crate :: asr_rescore:: levenshtein;
312360
0 commit comments