Skip to content

Commit f7ddf96

Browse files
committed
fix(ai-gateway): 修复 ASR 降级链断裂与 Redis 配置未生效
FallbackProvider.transcribe 缺少 **kwargs,call_with_fallback 透传的 _feature 撞到签名抛 TypeError,导致降级链末端失败、ASR 请求返回 503 而非友好降级响应(已补两条回归测试锁定)。 get_cache 以无参构造 RedisCache,忽略 REDIS_URL 环境变量,容器部署下 恒连 localhost 失败,限流与响应缓存自部署起从未生效;main.py 原先无论 成败都打印'连接已建立'掩盖了该故障。 其余:transcribe 限流路径由 /api/v1/ai/ 更正为 /api/v1/asr/,配额 30->600 并豁免全局每日总量(段级高频调用);开启 Qwen ASR 的 ITN; 上游 429 不再误报为本地配额耗尽。
1 parent 8733c87 commit f7ddf96

8 files changed

Lines changed: 118 additions & 9 deletions

File tree

‎server/ai-gateway/cache/redis_cache.py‎

Lines changed: 9 additions & 2 deletions
Original file line numberDiff line numberDiff line change
@@ -156,8 +156,15 @@ async def set_ai_cache(
156156

157157

158158
def get_cache() -> RedisCache:
159-
"""获取全局缓存实例"""
159+
"""
160+
获取全局缓存实例
161+
162+
@ai-context: 必须用 APP_CONFIG["redis_url"](来自环境变量 REDIS_URL)构造,
163+
容器部署时 Redis 位于独立容器(redis:6379,带密码),若用无参默认值
164+
redis://localhost:6379/0 会永远连接失败,导致限流与响应缓存静默失效。
165+
"""
160166
global _cache_instance
161167
if _cache_instance is None:
162-
_cache_instance = RedisCache()
168+
from config import APP_CONFIG
169+
_cache_instance = RedisCache(APP_CONFIG["redis_url"])
163170
return _cache_instance

‎server/ai-gateway/config/limits.py‎

Lines changed: 3 additions & 1 deletion
Original file line numberDiff line numberDiff line change
@@ -49,7 +49,9 @@
4949
"evaluate": 10,
5050
"recommend": 15,
5151
"vision_extract": 20,
52-
"transcribe": 30,
52+
# 课堂实时转录为段级高频调用(VAD 每 5-30s 产生一段,一节课数百段),
53+
# 对齐主流 ASR 按时长计费模式放宽次数限制;中间件对其豁免全局每日总量
54+
"transcribe": 600,
5355
"tag_content": 30,
5456
"optimize_card": 15,
5557
"feynman_question": 15,

‎server/ai-gateway/errors.py‎

Lines changed: 12 additions & 2 deletions
Original file line numberDiff line numberDiff line change
@@ -29,11 +29,21 @@ def __init__(self, provider: str, reason: str = ""):
2929

3030

3131
class RateLimitExceededError(AIError):
32-
"""频率限制超限"""
32+
"""
33+
频率限制超限
34+
35+
@ai-context: limit=0 表示上游服务商返回 429(其配额/并发受限),
36+
而非本项目的每日配额耗尽——两者文案必须区分,否则日志会出现
37+
"已达上限(0 次)"这类自相矛盾的误导信息。
38+
"""
3339

3440
def __init__(self, feature: str, limit: int):
41+
if limit > 0:
42+
message = f"今日 {feature} 功能使用次数已达上限({limit} 次),请明天再试"
43+
else:
44+
message = f"{feature} 服务商当前访问量过大或配额受限(上游 429),请稍后重试"
3545
super().__init__(
36-
message=f"今日 {feature} 功能使用次数已达上限({limit} 次),请明天再试",
46+
message=message,
3747
status_code=429,
3848
detail={"feature": feature, "limit": limit},
3949
)

‎server/ai-gateway/main.py‎

Lines changed: 9 additions & 1 deletion
Original file line numberDiff line numberDiff line change
@@ -69,7 +69,15 @@ async def lifespan(app: FastAPI):
6969
# 初始化 Redis 连接
7070
cache = get_cache()
7171
await cache.connect()
72-
logger.info("Redis 连接已建立")
72+
# connect 内部失败时会降级(_client=None),此处按实际结果记录,
73+
# 避免"连接已建立"的误导日志掩盖限流/缓存静默失效
74+
if cache._client is not None:
75+
logger.info("Redis 连接已建立")
76+
else:
77+
logger.error(
78+
"Redis 连接失败,限流与响应缓存将全部降级失效(限流放行)。"
79+
"请检查 REDIS_URL 环境变量与 redis 容器状态"
80+
)
7381

7482
# 初始化各 Provider 并检查 API Key 配置
7583
init_providers(app)

‎server/ai-gateway/middleware/rate_limit.py‎

Lines changed: 13 additions & 1 deletion
Original file line numberDiff line numberDiff line change
@@ -35,13 +35,17 @@
3535
"/api/v1/ai/evaluate-explanation": "evaluate",
3636
"/api/v1/ai/recommend-duration": "recommend",
3737
"/api/v1/ai/vision": "vision_extract",
38-
"/api/v1/ai/transcribe": "transcribe",
38+
"/api/v1/asr/transcribe": "transcribe",
3939
"/api/v1/ai/tag-content": "tag_content",
4040
"/api/v1/ai/optimize-card": "optimize_card",
4141
"/api/v1/ai/feynman-question": "feynman_question",
4242
"/api/v1/ai/feynman-evaluate-answers": "feynman_evaluate",
4343
}
4444

45+
# 豁免全局每日总量的功能:段级高频调用(如课堂实时转录一节课数百段),
46+
# 若计入 daily_total 会在几分钟内耗尽全部 AI 配额,仅受各自功能级上限约束
47+
GLOBAL_EXEMPT_FEATURES: frozenset[str] = frozenset({"transcribe"})
48+
4549

4650
class RateLimitMiddleware(BaseHTTPMiddleware):
4751
"""频率限制中间件 — 基于 Redis 滑动窗口"""
@@ -183,6 +187,10 @@ async def _check_rate_limit(
183187
"请明天再试,或升级套餐获取更多配额。"
184188
)
185189

190+
# 段级高频功能豁免全局每日总量(仅受功能级上限约束)
191+
if feature in GLOBAL_EXEMPT_FEATURES:
192+
return True, ""
193+
186194
# ---- 第二层:全局每日总量限制(预检查,仅读取) ----
187195
global_key = f"rate_limit:{user_id}:global:{today}"
188196
try:
@@ -236,6 +244,10 @@ async def _increment_rate_limit(self, user_id: str, feature: str) -> None:
236244
except Exception as exc:
237245
logger.warning("频率限制计数失败(功能级): %s", exc)
238246

247+
# 段级高频功能不计入全局每日总量
248+
if feature in GLOBAL_EXEMPT_FEATURES:
249+
return
250+
239251
# ---- 第二层:全局每日总量计数 ----
240252
global_key = f"rate_limit:{user_id}:global:{today}"
241253
try:

‎server/ai-gateway/providers/fallback_provider.py‎

Lines changed: 9 additions & 1 deletion
Original file line numberDiff line numberDiff line change
@@ -178,8 +178,16 @@ async def transcribe(
178178
sample_rate: int = 16000,
179179
channels: int = 1,
180180
model: str = "",
181+
**kwargs: Any,
181182
) -> dict[str, Any]:
182-
"""Fallback 不支持 ASR,返回友好降级结果"""
183+
"""
184+
Fallback 不支持 ASR,返回友好降级结果
185+
186+
@ai-context: 必须接受 **kwargs —— 云端 Provider 的 transcribe 由
187+
with_retry_and_timeout 装饰并在内部吞掉 _feature,而本方法无装饰器,
188+
call_with_fallback 传入的 _feature 会直接透传到签名上。缺少 **kwargs
189+
会抛 TypeError 使 fallback 链最后一环也失败,请求最终返回 503 而非降级响应。
190+
"""
183191
import time as _time
184192
start = _time.monotonic()
185193
logger.warning("使用 FallbackProvider 降级 ASR 响应")

‎server/ai-gateway/providers/qwen_provider.py‎

Lines changed: 3 additions & 1 deletion
Original file line numberDiff line numberDiff line change
@@ -120,7 +120,9 @@ async def transcribe(
120120
try:
121121
# 音频以 Data URL 内嵌(客户端上送 WAV/PCM base64)
122122
data_uri = f"data:audio/wav;base64,{audio_base64}"
123-
asr_options: dict[str, Any] = {"enable_itn": False}
123+
# ITN 开启:数字/单位规范化("三点一四"→"3.14"),对齐主流 ASR 默认行为,
124+
# 课堂场景公式/数据密集,规范化文本对笔记质量至关重要
125+
asr_options: dict[str, Any] = {"enable_itn": True}
124126
if language != "auto":
125127
asr_options["language"] = language
126128

‎server/ai-gateway/tests/test_providers.py‎

Lines changed: 60 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -183,6 +183,66 @@ def test_recommend_duration_fallback_no_history(self):
183183
assert result["recommended_minutes"] == 25
184184
assert result["source"] == "local_rule"
185185

186+
@pytest.mark.asyncio
187+
async def test_transcribe_accepts_feature_kwarg(self):
188+
"""
189+
回归:transcribe 必须容忍 call_with_fallback 透传的 _feature 关键字。
190+
191+
云端 Provider 的 transcribe 由 with_retry_and_timeout 装饰并吞掉 _feature,
192+
但 FallbackProvider 无装饰器,若签名不接受 **kwargs 会抛 TypeError,
193+
导致 fallback 链最后一环失败、ASR 请求返回 503 而非降级响应。
194+
"""
195+
# Arrange
196+
provider = FallbackProvider()
197+
198+
# Act
199+
result = await provider.transcribe(
200+
audio_base64="ZmFrZQ==", language="zh", _feature="transcribe",
201+
)
202+
203+
# Assert
204+
assert result["model"] == "fallback"
205+
assert result["text"] == ""
206+
assert result["warning"] # 必须透传降级提示供客户端识别失败
207+
208+
@pytest.mark.asyncio
209+
async def test_transcribe_fallback_reachable_when_cloud_asr_fails(self):
210+
"""
211+
回归:云端 ASR 全失败时,fallback 链末端必须能返回降级响应。
212+
213+
真实故障场景——Qwen 404 + GLM 400 后,若 FallbackProvider.transcribe
214+
因签名不接受 _feature 抛 TypeError,整个请求会变成 503。
215+
"""
216+
# Arrange:模拟 qwen/glm 的 transcribe 均失败,链末端为真实 FallbackProvider
217+
class FailingASRProvider:
218+
def __init__(self, name):
219+
self.provider_name = name
220+
221+
async def transcribe(self, *args, **kwargs):
222+
raise RuntimeError(f"{self.provider_name} ASR 上游故障")
223+
224+
app = MagicMock()
225+
app.state.providers = {
226+
"qwen": FailingASRProvider("qwen"),
227+
"glm": FailingASRProvider("glm"),
228+
"fallback": FallbackProvider(),
229+
}
230+
231+
async def _run_chain(provider, model_name):
232+
# 与 transcribe 路由一致:显式透传 _feature
233+
return await provider.transcribe(
234+
audio_base64="ZmFrZQ==", language="zh", model=model_name,
235+
_feature="transcribe",
236+
)
237+
238+
# Act
239+
result, provider_key = await call_with_fallback(app, "transcribe", _run_chain)
240+
241+
# Assert
242+
assert provider_key == "fallback"
243+
assert result["model"] == "fallback"
244+
assert result["warning"]
245+
186246
def test_recommend_duration_fallback_short_history(self):
187247
history = [
188248
{"duration_minutes": 10, "completed": True},

0 commit comments

Comments
 (0)