Skip to content

Commit ad788fe

Browse files
committed
feat: 本地 ASR 引擎 + 留存系统 + 奖赏组件 + 主题闪烁修复 + 3D 导航优化
- feat(ai): 集成 Sherpa-ONNX 本地离线语音识别引擎(模型管理+配置+服务) - feat(retention): 新增留存模块(组件/hooks/lib/store/types 完整架构) - feat: 学习奖赏微交互(ConceptInternalized/MemoryStrengthPulse/CompletionCelebration) - fix(theme): 修复多实例竞态主题闪烁,useTheme 委托 Zustand 单一状态源 - perf(3d): 轨道导航阻尼惯性优化 + 相机控制器 + 场景渲染精简 - feat(classroom): ASR 转写器增强(本地引擎适配+降噪管线) - feat(sync-service): 新增统计接口 handlers/stats.go
1 parent c4cd821 commit ad788fe

44 files changed

Lines changed: 4110 additions & 157 deletions

Some content is hidden

Large Commits have some content hidden by default. Use the searchbox below for content that may be hidden.

‎client/electron/ai/index.ts‎

Lines changed: 6 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -11,6 +11,8 @@ import { logger } from '../logger.js';
1111
import type { AIFeatureDef } from './utils.js';
1212
import { registerOllamaHandlers, initOllama } from './ollama/index.js';
1313
import { registerStreamHandler } from './streamHandler.js';
14+
import { registerLocalAsrHandlers } from './local-asr/index.js';
15+
import { loadLocalAsrConfig } from './local-asr/config.js';
1416

1517
// 导入所有 AI 功能模块
1618
import { feature as summarizeFeature } from './handlers/summarizeHandler.js';
@@ -73,6 +75,9 @@ export function registerAIHandlers(): void {
7375
// 注册 Ollama 本地推理 IPC handler
7476
registerOllamaHandlers();
7577

78+
// 注册本地 ASR(whisper.cpp)IPC handler
79+
registerLocalAsrHandlers();
80+
7681
// 注册流式输出 IPC handler
7782
registerStreamHandler();
7883

@@ -85,4 +90,5 @@ export function registerAIHandlers(): void {
8590
*/
8691
export async function initAIModule(): Promise<void> {
8792
await initOllama();
93+
await loadLocalAsrConfig();
8894
}
Lines changed: 288 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,288 @@
1+
/**
2+
* 本地 ASR — sherpa-onnx 语音识别服务
3+
*
4+
* @ai-context: 基于 sherpa-onnx-node 原生 addon 实现本地 ASR。
5+
* 双引擎架构:
6+
* - offline(SenseVoice):非流式,50ms/段极速推理,中文准确率最高,课后精修
7+
* - streaming(Paraformer):流式,边说边出 < 200ms 延迟,实时字幕
8+
* @ai-context: sherpa-onnx-node 为可选依赖(optionalDependencies),
9+
* 加载失败时 isAvailable() 返回 false,上层自动降级到云端 ASR。
10+
* 符合项目"可选增强"设计原则。
11+
* @ai-context: 音频输入要求:16kHz 单声道 Float32 PCM。
12+
* 渲染进程发来的音频块已满足此格式(process-audio native addon 输出)。
13+
*/
14+
15+
import * as path from 'path';
16+
import * as os from 'os';
17+
import { logger } from '../../logger.js';
18+
import {
19+
getLocalAsrConfig,
20+
getModelDir,
21+
isModelReady,
22+
type AsrEngine,
23+
} from './config.js';
24+
25+
// ================================================================
26+
// sherpa-onnx 动态加载(可选依赖,加载失败不崩溃)
27+
// ================================================================
28+
29+
interface SherpaOnnx {
30+
createOfflineRecognizer(config: Record<string, unknown>): OfflineRecognizer;
31+
createOnlineRecognizer(config: Record<string, unknown>): OnlineRecognizer;
32+
}
33+
34+
interface OfflineRecognizer {
35+
createStream(): OfflineStream;
36+
decode(stream: OfflineStream): void;
37+
getResult(stream: OfflineStream): { text: string };
38+
}
39+
40+
interface OfflineStream {
41+
acceptWaveform(sampleRate: number, samples: Float32Array): void;
42+
inputFinished(): void;
43+
free(): void;
44+
}
45+
46+
interface OnlineRecognizer {
47+
createStream(): OnlineStream;
48+
decode(stream: OnlineStream): void;
49+
isReady(stream: OnlineStream): boolean;
50+
getResult(stream: OnlineStream): { text: string };
51+
isEndpoint(stream: OnlineStream): boolean;
52+
reset(stream: OnlineStream): void;
53+
}
54+
55+
interface OnlineStream {
56+
acceptWaveform(sampleRate: number, samples: Float32Array): void;
57+
inputFinished(): void;
58+
free(): void;
59+
}
60+
61+
let _sherpa: SherpaOnnx | null = null;
62+
let _loadAttempted = false;
63+
64+
/** 尝试加载 sherpa-onnx-node(仅尝试一次) */
65+
function loadSherpa(): SherpaOnnx | null {
66+
if (_loadAttempted) return _sherpa;
67+
_loadAttempted = true;
68+
69+
try {
70+
// eslint-disable-next-line @typescript-eslint/no-require-imports
71+
_sherpa = require('sherpa-onnx-node') as SherpaOnnx;
72+
logger.info('[LocalASR] sherpa-onnx-node loaded successfully');
73+
} catch (err) {
74+
logger.warn(`[LocalASR] sherpa-onnx-node not available (optional dependency): ${err}`);
75+
_sherpa = null;
76+
}
77+
return _sherpa;
78+
}
79+
80+
// ================================================================
81+
// Recognizer 单例缓存
82+
// ================================================================
83+
84+
let _offlineRecognizer: OfflineRecognizer | null = null;
85+
let _onlineRecognizer: OnlineRecognizer | null = null;
86+
87+
/** 获取/创建 SenseVoice 非流式识别器(单例) */
88+
function getOfflineRecognizer(): OfflineRecognizer | null {
89+
if (_offlineRecognizer) return _offlineRecognizer;
90+
91+
const sherpa = loadSherpa();
92+
if (!sherpa) return null;
93+
if (!isModelReady('offline')) return null;
94+
95+
const modelDir = getModelDir('offline');
96+
const config = getLocalAsrConfig();
97+
const threads = config.threads > 0 ? config.threads : Math.max(1, os.cpus().length - 1);
98+
99+
try {
100+
_offlineRecognizer = sherpa.createOfflineRecognizer({
101+
featConfig: { sampleRate: 16000, featureDim: 80 },
102+
modelConfig: {
103+
senseVoice: {
104+
model: path.join(modelDir, 'model.int8.onnx'),
105+
language: config.language === 'auto' ? 'auto' : config.language,
106+
useItn: true,
107+
},
108+
tokens: path.join(modelDir, 'tokens.txt'),
109+
numThreads: threads,
110+
provider: 'cpu',
111+
},
112+
});
113+
logger.info(`[LocalASR] SenseVoice offline recognizer created (threads=${threads})`);
114+
return _offlineRecognizer;
115+
} catch (err) {
116+
logger.error(`[LocalASR] Failed to create offline recognizer: ${err}`);
117+
return null;
118+
}
119+
}
120+
121+
/** 获取/创建 Paraformer 流式识别器(单例) */
122+
function getOnlineRecognizer(): OnlineRecognizer | null {
123+
if (_onlineRecognizer) return _onlineRecognizer;
124+
125+
const sherpa = loadSherpa();
126+
if (!sherpa) return null;
127+
if (!isModelReady('streaming')) return null;
128+
129+
const modelDir = getModelDir('streaming');
130+
const config = getLocalAsrConfig();
131+
const threads = config.threads > 0 ? config.threads : Math.max(1, os.cpus().length - 1);
132+
133+
try {
134+
_onlineRecognizer = sherpa.createOnlineRecognizer({
135+
featConfig: { sampleRate: 16000, featureDim: 80 },
136+
modelConfig: {
137+
paraformer: {
138+
encoder: path.join(modelDir, 'encoder.onnx'),
139+
decoder: path.join(modelDir, 'decoder.onnx'),
140+
},
141+
tokens: path.join(modelDir, 'tokens.txt'),
142+
numThreads: threads,
143+
provider: 'cpu',
144+
},
145+
endpointConfig: {
146+
rule1: { minTrailingSilence: 2.4 },
147+
rule2: { minTrailingSilence: 1.2 },
148+
rule3: { minUtteranceLength: 20 },
149+
},
150+
});
151+
logger.info(`[LocalASR] Paraformer streaming recognizer created (threads=${threads})`);
152+
return _onlineRecognizer;
153+
} catch (err) {
154+
logger.error(`[LocalASR] Failed to create streaming recognizer: ${err}`);
155+
return null;
156+
}
157+
}
158+
159+
// ================================================================
160+
// 公共 API
161+
// ================================================================
162+
163+
/**
164+
* 检测本地 ASR 是否可用(sherpa-onnx 已加载 + 模型已下载)
165+
*/
166+
export async function checkLocalAsrAvailable(): Promise<boolean> {
167+
const sherpa = loadSherpa();
168+
if (!sherpa) return false;
169+
170+
const config = getLocalAsrConfig();
171+
return isModelReady(config.engine);
172+
}
173+
174+
/** 重置可用性缓存(模型下载完成后调用) */
175+
export function resetAvailabilityCache(): void {
176+
_offlineRecognizer = null;
177+
_onlineRecognizer = null;
178+
}
179+
180+
/**
181+
* 非流式转写(SenseVoice)— 适合课后全量分析
182+
*
183+
* @param pcmData - Float32 PCM 音频数据(16kHz 单声道)
184+
* @param options - 转写选项
185+
* @returns 转写结果
186+
*/
187+
export async function transcribeOffline(
188+
pcmData: Float32Array,
189+
options?: { language?: string },
190+
): Promise<{ text: string; engine: 'offline'; durationMs: number }> {
191+
const startTime = Date.now();
192+
193+
const recognizer = getOfflineRecognizer();
194+
if (!recognizer) {
195+
throw new Error('SenseVoice 识别器不可用(模型未下载或 sherpa-onnx 未安装)');
196+
}
197+
198+
const stream = recognizer.createStream();
199+
try {
200+
stream.acceptWaveform(16000, pcmData);
201+
stream.inputFinished();
202+
recognizer.decode(stream);
203+
const result = recognizer.getResult(stream);
204+
const text = result.text?.trim() ?? '';
205+
const durationMs = Date.now() - startTime;
206+
207+
logger.debug(`[LocalASR] Offline transcribe: ${text.length} chars, ${durationMs}ms`);
208+
return { text, engine: 'offline', durationMs };
209+
} finally {
210+
stream.free();
211+
}
212+
}
213+
214+
/**
215+
* 流式转写(Paraformer)— 适合实时字幕
216+
*
217+
* 将完整音频段一次性喂入流式识别器,逐帧解码后返回最终文本。
218+
* 真正的"边录边出"需要渲染进程持续推送音频块(后续版本支持)。
219+
*
220+
* @param pcmData - Float32 PCM 音频数据(16kHz 单声道)
221+
* @returns 转写结果
222+
*/
223+
export async function transcribeStreaming(
224+
pcmData: Float32Array,
225+
): Promise<{ text: string; engine: 'streaming'; durationMs: number }> {
226+
const startTime = Date.now();
227+
228+
const recognizer = getOnlineRecognizer();
229+
if (!recognizer) {
230+
throw new Error('Paraformer 识别器不可用(模型未下载或 sherpa-onnx 未安装)');
231+
}
232+
233+
const stream = recognizer.createStream();
234+
try {
235+
// 分块喂入(模拟流式,每块 1600 样本 = 100ms)
236+
const chunkSize = 1600;
237+
for (let offset = 0; offset < pcmData.length; offset += chunkSize) {
238+
const chunk = pcmData.subarray(offset, Math.min(offset + chunkSize, pcmData.length));
239+
stream.acceptWaveform(16000, chunk);
240+
while (recognizer.isReady(stream)) {
241+
recognizer.decode(stream);
242+
}
243+
}
244+
245+
// 尾部解码
246+
stream.inputFinished();
247+
while (recognizer.isReady(stream)) {
248+
recognizer.decode(stream);
249+
}
250+
251+
const result = recognizer.getResult(stream);
252+
const text = result.text?.trim() ?? '';
253+
const durationMs = Date.now() - startTime;
254+
255+
logger.debug(`[LocalASR] Streaming transcribe: ${text.length} chars, ${durationMs}ms`);
256+
return { text, engine: 'streaming', durationMs };
257+
} finally {
258+
stream.free();
259+
}
260+
}
261+
262+
/**
263+
* 统一转写入口(根据配置选择引擎)
264+
*
265+
* @param audioBase64 - base64 编码的 Float32 PCM 音频(16kHz 单声道)
266+
* @param options - 转写选项
267+
*/
268+
export async function transcribeLocal(
269+
audioBase64: string,
270+
options?: { language?: string; sampleRate?: number; channels?: number; engine?: AsrEngine },
271+
): Promise<{ text: string; language: string; durationMs: number }> {
272+
const config = getLocalAsrConfig();
273+
const engine = options?.engine ?? config.engine;
274+
const language = options?.language ?? config.language;
275+
276+
// base64 → Float32Array
277+
const rawBytes = Buffer.from(audioBase64, 'base64');
278+
const pcmData = new Float32Array(rawBytes.buffer, rawBytes.byteOffset, rawBytes.byteLength / 4);
279+
280+
if (engine === 'streaming' && isModelReady('streaming')) {
281+
const result = await transcribeStreaming(pcmData);
282+
return { text: result.text, language, durationMs: result.durationMs };
283+
}
284+
285+
// 默认走 offline(SenseVoice)
286+
const result = await transcribeOffline(pcmData, { language });
287+
return { text: result.text, language, durationMs: result.durationMs };
288+
}

0 commit comments

Comments
 (0)