Skip to content

Commit cd26df0

Browse files
committed
fix(classroom): 智能采集语音识别流式化+merge降级+去重复传输
- 流式ASR:VADMarker添加onSegmentReady回调,语音段完成后即时转写并回填audioText,UI显示'已转写N段语音'进度 - ASR语言修复:ASRWorker支持language构造参数,captureManager传入用户配置,消除硬编码'zh' - merge降级:mergeNotes失败时本地拼接片段笔记(零网络),避免显示错误+触发全量重发 - 去重复传输:local-concat结果不显示重试按钮,防止增量已处理帧被全量重发 - sessionAnalyzer:优先使用流式已转写文本(seg.audioText),仅对未转写段补充ASR
1 parent d3ecec9 commit cd26df0

8 files changed

Lines changed: 143 additions & 20 deletions

File tree

‎client/src/features/classroom/hooks/useClassroomCapture.ts‎

Lines changed: 54 additions & 4 deletions
Original file line numberDiff line numberDiff line change
@@ -23,6 +23,7 @@ import type {
2323
} from '@/lib/capture';
2424
import { analyzeSession, analyzeVideo, analyzePartial, mergeNotes } from '@/lib/ai/sessionAnalyzer';
2525
import type { AnalyzeResult } from '@/lib/ai/sessionAnalyzer';
26+
import { aiClient } from '@/lib/http/apiClient';
2627

2728
interface IPCAudioStartResult {
2829
success: boolean;
@@ -136,6 +137,7 @@ export function useClassroomCapture() {
136137
const pendingKeyframesRef = useRef<KeyFrame[]>([]);
137138
const isPartialAnalyzingRef = useRef(false);
138139
const [partialCount, setPartialCount] = useState(0);
140+
const [transcribedCount, setTranscribedCount] = useState(0);
139141
const INCREMENTAL_BATCH_SIZE = 5;
140142

141143
useEffect(() => {
@@ -173,6 +175,46 @@ export function useClassroomCapture() {
173175
return () => { offKeyframe(); offBundleReady(); };
174176
}, [config.language]);
175177

178+
// ── Path B:流式 ASR — 语音段完成后立即转写 ──
179+
useEffect(() => {
180+
const offSegmentReady = captureEventBus.on<{ sessionId: string; segment: import('@/lib/capture').AudioSegment }>(
181+
'smart:audio_segment_ready',
182+
(data) => {
183+
const seg = data.segment;
184+
// 先将音频段加入 bundle
185+
setSmartBundle((prev) => ({
186+
...prev,
187+
audioSegments: [...(prev.audioSegments ?? []), seg],
188+
}));
189+
190+
// 后台流式 ASR 转写(不阻塞采集)
191+
if (!seg.audioBase64) return;
192+
const lang = config.language === 'en' ? 'en' : config.language === 'mixed' ? 'auto' : 'zh';
193+
aiClient.post<{ text: string }>('/api/v1/asr/transcribe', {
194+
audio_base64: seg.audioBase64,
195+
language: lang,
196+
sample_rate: 16000,
197+
channels: 1,
198+
})
199+
.then((resp) => {
200+
const text = resp.text?.trim() || null;
201+
// 将转写结果回填到对应的音频段
202+
setSmartBundle((prev) => ({
203+
...prev,
204+
audioSegments: (prev.audioSegments ?? []).map((s) =>
205+
s.id === seg.id ? { ...s, audioText: text } : s,
206+
),
207+
}));
208+
if (text) setTranscribedCount((c) => c + 1);
209+
})
210+
.catch((err) => {
211+
console.warn('[useClassroomCapture] 流式 ASR 转写失败:', err);
212+
});
213+
},
214+
);
215+
return () => { offSegmentReady(); };
216+
}, [config.language]);
217+
176218
// ── Path C:监听录制视频就绪 ──
177219
useEffect(() => {
178220
const offVideoReady = captureEventBus.on<{ sessionId: string; videoRecording: VideoRecording }>(
@@ -477,14 +519,22 @@ export function useClassroomCapture() {
477519
if (partialNotesRef.current.length > 0) {
478520
setIsAnalyzing(true);
479521
setAnalysisError(null);
522+
const partials = [...partialNotesRef.current];
480523
try {
481-
const result = await mergeNotes(partialNotesRef.current, {
524+
const result = await mergeNotes(partials, {
482525
duration: (smartBundle.duration ?? 0) / 1000,
483526
language: config.language,
484527
});
485528
setAnalysisResult(result);
486-
} catch (err) {
487-
setAnalysisError(err instanceof Error ? err.message : '笔记合并失败');
529+
} catch {
530+
// 降级:本地拼接片段笔记(无需 AI,零网络,避免全量重发)
531+
const fallbackContent = partials.join('\n\n---\n\n');
532+
setAnalysisResult({
533+
content: fallbackContent,
534+
keyframesAnalyzed: smartBundle.keyframes?.length ?? 0,
535+
modelUsed: 'local-concat',
536+
});
537+
toast({ type: 'warning', message: 'AI 合并不可用,已直接拼接片段笔记' });
488538
} finally {
489539
setIsAnalyzing(false);
490540
}
@@ -597,7 +647,7 @@ export function useClassroomCapture() {
597647
// 路径
598648
capturePath, setCapturePath, smartBundle,
599649
// 分析
600-
isAnalyzing, analysisResult, analysisError, partialCount,
650+
isAnalyzing, analysisResult, analysisError, partialCount, transcribedCount,
601651
// 录制
602652
recordingStatus, videoFilePath,
603653
// 操作

‎client/src/features/classroom/pages/ClassroomPage.tsx‎

Lines changed: 7 additions & 1 deletion
Original file line numberDiff line numberDiff line change
@@ -346,6 +346,12 @@ export default function ClassroomPage() {
346346
<span className="text-b3 text-brand-600">已增量分析 {capture.partialCount} 段,课后将快速合并生成笔记</span>
347347
</div>
348348
)}
349+
{capture.transcribedCount > 0 && (
350+
<div className="mx-4 mt-2 flex items-center gap-2 px-3 py-2 rounded-kb-md bg-emerald-50/50 border border-emerald-100/50">
351+
<Mic className="w-4 h-4 text-emerald-500" strokeWidth={1.5} />
352+
<span className="text-b3 text-emerald-600">已转写 {capture.transcribedCount} 段语音</span>
353+
</div>
354+
)}
349355
</>
350356
)}
351357

@@ -361,7 +367,7 @@ export default function ClassroomPage() {
361367
error={capture.analysisError}
362368
onInsert={() => capture.handleDismissAnalysis()}
363369
onDismiss={capture.handleDismissAnalysis}
364-
onRetry={capture.handleAnalyze}
370+
onRetry={capture.analysisResult?.modelUsed === 'local-concat' ? undefined : capture.handleAnalyze}
365371
/>
366372
)}
367373
</div>

‎client/src/lib/ai/asrWorker.ts‎

Lines changed: 7 additions & 2 deletions
Original file line numberDiff line numberDiff line change
@@ -38,6 +38,11 @@ interface TranscribeApiResponse {
3838

3939
export class ASRWorker implements PipelineWorker {
4040
name = 'asr-worker';
41+
private readonly language: string;
42+
43+
constructor(language: string = 'zh') {
44+
this.language = language;
45+
}
4146

4247
canProcess(message: PipelineMessage): boolean {
4348
return message.type === 'audio_chunk';
@@ -49,12 +54,12 @@ export class ASRWorker implements PipelineWorker {
4954
// ArrayBuffer → base64
5055
const base64 = arrayBufferToBase64(audioData.audioBuffer);
5156

52-
// 调用后端 ASR API
57+
// 调用后端 ASR API(使用配置的语言)
5358
const response = await aiClient.post<TranscribeApiResponse>(
5459
'/api/v1/asr/transcribe',
5560
{
5661
audio_base64: base64,
57-
language: 'zh',
62+
language: this.language,
5863
sample_rate: audioData.sampleRate,
5964
channels: audioData.channels,
6065
},

‎client/src/lib/ai/sessionAnalyzer.ts‎

Lines changed: 48 additions & 6 deletions
Original file line numberDiff line numberDiff line change
@@ -6,7 +6,8 @@
66
import { supabase } from '@/lib/auth/supabaseClient';
77
import { getActiveUserKey } from '@/lib/ai/apiKeyManager';
88
import { classroomNoteStore } from '@/lib/storage/classroomNoteStore';
9-
import type { SessionBundle, KeyFrame } from '@/lib/capture/captureTypes';
9+
import { aiClient } from '@/lib/http/apiClient';
10+
import type { SessionBundle, KeyFrame, AudioSegment } from '@/lib/capture/captureTypes';
1011

1112
// ================================================================
1213
// 分析结果类型
@@ -20,6 +21,42 @@ export interface AnalyzeResult {
2021
source?: 'local' | 'remote';
2122
}
2223

24+
// ================================================================
25+
// 音频段转写工具
26+
// ================================================================
27+
28+
interface TranscribeResponse {
29+
text: string;
30+
confidence: number;
31+
model_used: string;
32+
}
33+
34+
/**
35+
* 将音频段的 audioBase64 通过 ASR API 转写为文本
36+
* 失败时静默返回 null,不阻塞整体分析流程
37+
*/
38+
async function transcribeSegment(
39+
seg: AudioSegment,
40+
language: string,
41+
): Promise<string | null> {
42+
if (!seg.audioBase64) return null;
43+
try {
44+
const resp = await aiClient.post<TranscribeResponse>(
45+
'/api/v1/asr/transcribe',
46+
{
47+
audio_base64: seg.audioBase64,
48+
language,
49+
sample_rate: 16000,
50+
channels: 1,
51+
},
52+
);
53+
return resp.text?.trim() || null;
54+
} catch (e) {
55+
console.warn('[sessionAnalyzer] 音频段转写失败:', e);
56+
return null;
57+
}
58+
}
59+
2360
// ================================================================
2461
// 分析函数
2562
// ================================================================
@@ -42,11 +79,16 @@ export async function analyzeSession(
4279
imageBase64: kf.imageBase64,
4380
changeType: kf.changeType,
4481
}));
45-
const audioSegments = bundle.audioSegments.map((seg) => ({
46-
timestampStart: seg.timestampStart / 1000,
47-
timestampEnd: seg.timestampEnd / 1000,
48-
audioText: null,
49-
}));
82+
83+
// 转写音频段:优先使用流式 ASR 已转写的文本,仅对未转写的段进行补充转写
84+
const lang = options?.language === 'en' ? 'en' : options?.language === 'mixed' ? 'auto' : 'zh';
85+
const audioSegments = await Promise.all(
86+
bundle.audioSegments.map(async (seg) => ({
87+
timestampStart: seg.timestampStart / 1000,
88+
timestampEnd: seg.timestampEnd / 1000,
89+
audioText: seg.audioText ?? await transcribeSegment(seg, lang),
90+
})),
91+
);
5092

5193
const result = await window.electronAPI!.invoke('ai_session_analyze', {
5294
keyframes,

‎client/src/lib/capture/captureManager.ts‎

Lines changed: 12 additions & 4 deletions
Original file line numberDiff line numberDiff line change
@@ -110,7 +110,7 @@ export class CaptureManager {
110110
this.capturePath = config.path ?? 'fine';
111111

112112
// ================================================================
113-
// Path B 智能模式:跳过 Pipeline/Worker,用轻量采样器替代
113+
// Path B 智能模式:跳过 Pipeline/Worker,用轻量采样器 + 流式 ASR 替代
114114
// ================================================================
115115
if (this.capturePath === 'smart') {
116116
const session = await captureStore.createSession({
@@ -129,6 +129,14 @@ export class CaptureManager {
129129
this.smartSampler = new SmartSampler();
130130
this.vadMarker = new VADMarker();
131131

132+
// 流式 ASR:语音段完成后立即发射事件,由上层 Hook 触发转写
133+
this.vadMarker.onSegmentReady = (segment) => {
134+
captureEventBus.emit('smart:audio_segment_ready', {
135+
sessionId: this.sessionId,
136+
segment,
137+
});
138+
};
139+
132140
captureEventBus.emit('session:started', {
133141
sessionId: this.sessionId,
134142
config,
@@ -176,9 +184,9 @@ export class CaptureManager {
176184
uiAutomationAvailable: false, // Electron 环境下后续检测
177185
});
178186

179-
// 根据决策动态注册 ASR Worker
187+
// 根据决策动态注册 ASR Worker(传入用户配置的语言)
180188
if (this.lastDecision.audioEnabled && !this.asrWorker) {
181-
this.asrWorker = new ASRWorker();
189+
this.asrWorker = new ASRWorker(config.language || 'zh');
182190
this.pipeline.registerWorker(this.asrWorker);
183191
this.dispatcher.registerWorker(this.asrWorker);
184192
} else if (!this.lastDecision.audioEnabled && this.asrWorker) {
@@ -411,7 +419,7 @@ export class CaptureManager {
411419
}
412420

413421
// ================================================================
414-
// Path B 智能模式:VADMarker 检测语音段,不送入 ASR
422+
// Path B 智能模式:VADMarker 检测语音段,流式触发 ASR 转写
415423
// ================================================================
416424
if (this.capturePath === 'smart' && this.vadMarker) {
417425
this.vadMarker.processChunk(audioData);

‎client/src/lib/capture/captureTypes.ts‎

Lines changed: 3 additions & 1 deletion
Original file line numberDiff line numberDiff line change
@@ -135,13 +135,15 @@ export interface KeyFrame {
135135
changeType: 'slide_change' | 'writing' | 'scene_change' | 'periodic';
136136
}
137137

138-
/** @ai-context VAD 标记器切出的语音段,含编码后的音频数据 */
138+
/** @ai-context VAD 标记器切出的语音段,含编码后的音频数据,支持流式 ASR 转写 */
139139
export interface AudioSegment {
140140
id: string;
141141
timestampStart: number;
142142
timestampEnd: number;
143143
audioBase64: string;
144144
energy: number;
145+
/** 流式 ASR 转写结果(课堂进行中即时填充,无需课后批量转写) */
146+
audioText?: string | null;
145147
}
146148

147149
/** @ai-context 全局时间轴条目,串联关键帧和语音段供后续分析回放 */

‎client/src/lib/capture/vadMarker.ts‎

Lines changed: 11 additions & 2 deletions
Original file line numberDiff line numberDiff line change
@@ -2,8 +2,8 @@
22
* VAD 音频标记器 — Path B 语音段检测与分段
33
*
44
* @ai-context
5-
* Path B 不做实时 ASR,而是通过 RMS 能量检测将连续语音切段,
6-
* 每段打包为 WAV base64 供后续按需转写,降低 AI 调用次数。
5+
* Path B 通过 RMS 能量检测将连续语音切段,
6+
* 每段完成后立即触发 onSegmentReady 回调,支持流式 ASR 转写。
77
*/
88

99
import type { AudioChunkData, AudioSegment, TimelineEntry } from './captureTypes';
@@ -36,6 +36,9 @@ export class VADMarker {
3636
private segments: AudioSegment[] = [];
3737
private timeline: TimelineEntry[] = [];
3838

39+
/** 语音段完成回调(流式 ASR 触发点) */
40+
onSegmentReady: ((segment: AudioSegment) => void) | null = null;
41+
3942
// 当前语音段状态
4043
private isSpeaking = false;
4144
private speechStartTime = 0;
@@ -172,6 +175,12 @@ export class VADMarker {
172175
audioBase64,
173176
energy: Math.round(avgEnergy * 10000) / 10000,
174177
});
178+
179+
// 流式触发:语音段完成后立即通知外部进行 ASR 转写
180+
const newSegment = this.segments[this.segments.length - 1];
181+
if (this.onSegmentReady) {
182+
this.onSegmentReady(newSegment);
183+
}
175184
}
176185
}
177186

‎server/nginx/nginx.conf‎

Lines changed: 1 addition & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -1,3 +1,4 @@
1+
12
worker_processes auto;
23

34
events {

0 commit comments

Comments
 (0)