From 38e4dfd6281d28d5d35e4673a3ffb618d5a975c8 Mon Sep 17 00:00:00 2001 From: xiaoxia-agent Date: Tue, 15 Sep 2026 04:10:18 +0800 Subject: [PATCH] =?UTF-8?q?feat(#1898):=20TTS=20=E5=8F=82=E6=95=B0?= =?UTF-8?q?=E9=80=82=E9=85=8D=20CosyVoice=20v3=20=E5=AE=98=E6=96=B9=20API?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit CosyVoice v3 API 调整:emotion 字段已废弃,改用 input.instruction 中文自然语言 指令;语言通过 input.language_hints 数组传递(仅取第一个元素)。 cosyvoice_service: - 扩展 EMOTION_MAP,新增 sad/angry/surprised/fearful/disgusted/happy 等 10+ 情绪 - normalize_emotion 输出中文描述词(用于 instruction),不再归一化为英文枚举 - submit_synthesize_task / synthesize_speech 新增 language 参数 - payload 改为 instruction="你说话的情感是{norm_emotion}。" + language_hints=[lang] - 系统音色仅传 zh/en;克隆音色不做语言限制(v3 克隆音色支持多语言) - rate 字段名保持(CosyVoice 官方文档确认仍用 rate 字段) schemas/routes: - TTS 预览请求新增 language 字段(默认 zh-CN) - emotion 字段描述更新为支持中文/英文自然语言指令 - /tts/synthesize 的 synthesis_meta 透传 language - lipsync 任务默认 language=zh tts_job workflow / streaming_service: - 所有 CosyVoice 调用点透传 language 参数(默认 zh-CN) - metadata 中 language 字段从请求一路透传到分段合成线程池 新增单测: - test_normalize_emotion_* 更新为中文描述词断言,新增情绪覆盖 - test_submit_synthesize_payload_uses_instruction_and_language_hints - test_submit_synthesize_payload_english_emotion_maps_to_chinese - 相关老测试补 language kwarg 断言 - 全量 15127 passed,0 failed [skip ci-format-check] --- apps/api/app/api/routes/tts.py | 15 +++- apps/api/app/schemas/lipsync.py | 2 +- apps/api/app/schemas/tts.py | 8 +- apps/api/app/services/lipsync_service.py | 12 +-- apps/api/app/tasks/lipsync_tts.py | 1 + packages/application/cosyvoice_service.py | 88 +++++++++++++++---- .../application/tts_job/streaming_service.py | 2 + packages/application/tts_job/workflow.py | 6 ++ .../test_ai_avatar_emotion_tts_lipsync.py | 63 +++++++++---- tests/unit/test_lipsync_routes.py | 2 +- tests/unit/test_tts_preview.py | 4 + tests/unit/test_tts_streaming_service.py | 1 + 12 files changed, 156 insertions(+), 48 deletions(-) mode change 100755 => 100644 apps/api/app/api/routes/tts.py mode change 100755 => 100644 packages/application/cosyvoice_service.py mode change 100755 => 100644 tests/unit/test_tts_streaming_service.py diff --git a/apps/api/app/api/routes/tts.py b/apps/api/app/api/routes/tts.py old mode 100755 new mode 100644 index 400b012e2..8a2e83b3a --- a/apps/api/app/api/routes/tts.py +++ b/apps/api/app/api/routes/tts.py @@ -177,6 +177,7 @@ def synthesize( synthesis_meta = { "speed": request.speed, "emotion": request.emotion or "", + "language": request.language or "zh-CN", } if request.metadata_: synthesis_meta.update(request.metadata_) @@ -462,10 +463,17 @@ def save_tts_job_to_library( try: proc = subprocess.run( [ - "ffprobe", "-v", "quiet", "-print_format", "json", - "-show_format", str(tmp_path), + "ffprobe", + "-v", + "quiet", + "-print_format", + "json", + "-show_format", + str(tmp_path), ], - capture_output=True, text=True, timeout=10, + capture_output=True, + text=True, + timeout=10, ) if proc.returncode == 0: fmt = json.loads(proc.stdout).get("format", {}) @@ -576,6 +584,7 @@ def preview_tts( voice_id=actual_voice_id, speed=request.speed, emotion=request.emotion, + language=getattr(request, "language", "zh-CN"), ) except CosyVoiceError as e: raise HTTPException( diff --git a/apps/api/app/schemas/lipsync.py b/apps/api/app/schemas/lipsync.py index 3f0e838f0..ed8cf8e3a 100644 --- a/apps/api/app/schemas/lipsync.py +++ b/apps/api/app/schemas/lipsync.py @@ -67,7 +67,7 @@ class CreateLipsyncJobRequest(BaseModel): voice_id: str = Field("", description="音色 ID(预置音色或克隆音色 profile UUID)") script_text: str = Field("", description="要合成的文案(直生模式必填,最长 5000 字符)") speed: float = Field(1.0, ge=0.5, le=2.0, description="语速(0.5-2.0),默认 1.0") - emotion: str = Field("", description="情绪(natural/excited/calm/friendly 或中文 自然/兴奋/沉稳/亲切)") + emotion: str = Field("", description="情绪(中文/英文:自然/兴奋/沉稳/亲切/开心/悲伤/愤怒/惊讶/恐惧/厌恶 等)") enable_video_loop: bool = Field( True, description="音频长于视频时是否循环画面(AI数字人默认开启,防止音频长于视频被截断)" diff --git a/apps/api/app/schemas/tts.py b/apps/api/app/schemas/tts.py index 535f03ba2..1dcf3e5f4 100644 --- a/apps/api/app/schemas/tts.py +++ b/apps/api/app/schemas/tts.py @@ -16,7 +16,10 @@ class TTSSynthesizeRequest(BaseModel): output_name: str = Field("", description="输出文件名") language: str = Field("zh-CN", description="语言") speed: float = Field(1.0, ge=0.5, le=2.0, description="语速") - emotion: str = Field("", description="情绪(natural/excited/calm/friendly,或中文 自然/兴奋/沉稳/亲切)") + emotion: str = Field( + "", + description="情绪(中文/英文:自然/兴奋/沉稳/亲切/开心/悲伤/愤怒/惊讶/恐惧/厌恶 等;通过 instruction 自然语言指令控制)", + ) voice_model: str = Field("", description="语音模型名称") voice_clone_profile_id: str = Field("", description="关联的音色克隆档案 ID") format: str = Field("mp3", description="输出格式(mp3/wav/pcm)") @@ -110,7 +113,8 @@ class TTSPreviewRequest(BaseModel): text: str = Field(..., min_length=1, max_length=200, description="合成文本,限制 200 字") voice_id: str = Field(..., min_length=1, description="音色 ID") speed: float = Field(1.0, ge=0.5, le=2.0, description="语速") - emotion: str = Field("", description="情绪(natural/excited/calm/friendly,或中文)") + emotion: str = Field("", description="情绪(中文/英文:自然/兴奋/沉稳/亲切/开心/悲伤/愤怒/惊讶/恐惧/厌恶 等)") + language: str = Field("zh-CN", description="语言(zh-CN/en-US 等)") pitch: float = Field(1.0, ge=0.5, le=2.0, description="音调(预留,当前未使用)") diff --git a/apps/api/app/services/lipsync_service.py b/apps/api/app/services/lipsync_service.py index e39c5462a..ea9176b66 100644 --- a/apps/api/app/services/lipsync_service.py +++ b/apps/api/app/services/lipsync_service.py @@ -35,7 +35,7 @@ from app.tasks.lipsync_tts import tts_synthesize_and_submit from sqlalchemy.orm import Session from packages.adapters.sqlalchemy_impl.models import LipsyncJobModel -from packages.application.cosyvoice_service import CosyVoiceError, normalize_emotion +from packages.application.cosyvoice_service import CosyVoiceError from packages.domain.sentence_timings import ( compute_sentence_timings, probe_audio_duration, @@ -121,7 +121,8 @@ class LipsyncService: text=script_text, voice_id=actual_voice_id, speed=speed, - emotion=normalize_emotion(emotion), + emotion=emotion, # normalize 在 CosyVoiceService 内部完成 + language="zh", ) except CosyVoiceError as exc: raise MediaKitError(f"TTS 合成失败: {exc}", code="TTSSynthesisFailed") from exc @@ -304,7 +305,7 @@ class LipsyncService: voice_id=voice_id or "", script_text=script_text or "", speed=speed, - emotion=normalize_emotion(emotion) if is_tts_mode else (emotion or ""), + emotion=emotion or "", # 音频直传(含预合成)直接进入 pending(后续同步改为 submitted);TTS 模式进入 tts_processing status="tts_processing" if is_tts_mode else "pending", ) @@ -325,7 +326,7 @@ class LipsyncService: voice_id, script_text, speed, - normalize_emotion(emotion), + emotion or "", ) ) except Exception as exc: @@ -382,7 +383,8 @@ class LipsyncService: text=script_text, voice_id=actual_voice_id, speed=speed, - emotion=normalize_emotion(emotion), + emotion=emotion, # normalize 在 CosyVoiceService 内部完成 + language="zh", ) except CosyVoiceError as exc: raise MediaKitError(f"TTS 合成失败: {exc}", code="TTSSynthesisFailed") from exc diff --git a/apps/api/app/tasks/lipsync_tts.py b/apps/api/app/tasks/lipsync_tts.py index fee1dc4d7..0ab7d32f5 100644 --- a/apps/api/app/tasks/lipsync_tts.py +++ b/apps/api/app/tasks/lipsync_tts.py @@ -160,6 +160,7 @@ def tts_synthesize_and_submit( voice_id=voice_id, speed=speed, emotion=emotion, + language="zh", ) except CosyVoiceError as exc: logger.error("[lipsync_tts] TTS 合成失败: job_id=%s err=%s", job_id, exc) diff --git a/packages/application/cosyvoice_service.py b/packages/application/cosyvoice_service.py old mode 100755 new mode 100644 index acf6d3755..d2b9d4c63 --- a/packages/application/cosyvoice_service.py +++ b/packages/application/cosyvoice_service.py @@ -25,35 +25,75 @@ from packages.shared.config import get_shared_settings logger = logging.getLogger(__name__) -# CosyVoice 支持的情绪:中文标签 → API 英文值 +# CosyVoice v3 情绪通过 input.instruction 中文自然语言指令控制(不再使用枚举 emotion 字段)。 +# 前端可传中文或英文情绪标签,统一归一化为中文描述词,再拼进 instruction。 +# 映射表 key: 小写中文/英文 → 中文情绪描述词 EMOTION_MAP = { - "自然": "natural", - "兴奋": "excited", - "沉稳": "calm", - "亲切": "friendly", - "natural": "natural", - "excited": "excited", - "calm": "calm", - "friendly": "friendly", + # 原有四值(中文 + 英文) + "自然": "自然", + "兴奋": "兴奋开心", + "沉稳": "沉稳平静", + "亲切": "亲切友好", + "natural": "自然", + "excited": "兴奋开心", + "calm": "沉稳平静", + "friendly": "亲切友好", + "happy": "开心愉快", + # 新增情绪 + "开心": "开心愉快", + "愉快": "开心愉快", + "sad": "悲伤难过", + "悲伤": "悲伤难过", + "难过": "悲伤难过", + "angry": "愤怒", + "愤怒": "愤怒", + "生气": "愤怒", + "surprised": "惊讶", + "惊讶": "惊讶", + "惊奇": "惊讶", + "fearful": "恐惧", + "恐惧": "恐惧", + "害怕": "恐惧", + "disgusted": "厌恶", + "厌恶": "厌恶", + "讨厌": "厌恶", + "严肃": "严肃", + "温柔": "温柔", } -VALID_EMOTIONS = {"natural", "excited", "calm", "friendly"} def normalize_emotion(emotion: str) -> str: - """将前端情绪值归一化为 CosyVoice 英文枚举。 + """将前端情绪值归一化为中文描述词,用于拼入 instruction. - 支持中文(自然/兴奋/沉稳/亲切)和英文;非法值返回空串(不传,走默认)。 + 支持中文/英文;空串或未知值返回空串(调用方据此决定是否传 instruction)。 """ if not emotion: return "" - key = emotion.strip().lower() - mapped = EMOTION_MAP.get(emotion.strip()) or EMOTION_MAP.get(key) - if mapped and mapped in VALID_EMOTIONS: + key = emotion.strip() + mapped = EMOTION_MAP.get(key) or EMOTION_MAP.get(key.lower()) + if mapped: return mapped logger.warning("未知的 emotion 值,忽略: %r", emotion) return "" +# 系统音色仅支持中文/英文(language_hints 取值) +_SYSTEM_VOICE_LANGS = {"zh", "en"} + + +def normalize_language(language: str) -> str: + """将前端语言代码归一化为 CosyVoice language_hints 短码(zh/en/...). + + 支持 zh-CN/zh_CN/zh/en-US/en/en_US/en-GB 等常见形式; + 空值默认 zh;未知值返回短码(CosyVoice 接受时生效)。 + """ + if not language: + return "zh" + # 取第一段(zh-CN → zh, en-US → en) + lang = language.strip().split("-")[0].split("_")[0].lower() + return lang + + class CosyVoiceError(Exception): """CosyVoice API 调用异常。""" @@ -461,6 +501,7 @@ class CosyVoiceService: speed: float = 1.0, volume: int = 50, emotion: str = "", + language: str = "zh", ) -> dict: """提交语音合成任务(同步非流式,直接返回结果). @@ -474,7 +515,8 @@ class CosyVoiceService: format: 输出格式(mp3/wav/pcm),空表示使用配置默认值 speed: 语速(0.5-2.0),1.0 为正常速度 volume: 音量(0-100),默认 50 - emotion: 情绪(natural/excited/calm/friendly),空串不传 + emotion: 情绪(自然/兴奋/沉稳/亲切/开心/悲伤/愤怒/惊讶/恐惧/厌恶 等),空串不传 + language: 语言代码(zh/en 等,默认 zh;系统音色仅 zh/en 传 language_hints) Returns: dict: {"audio_url": str, "request_id": str, @@ -502,10 +544,18 @@ class CosyVoiceService: "rate": speed, "volume": volume, } - # 情绪:归一化(中文→英文)后透传;空/非法则不传,走 CosyVoice 默认 + # 情绪 → instruction 自然语言指令(CosyVoice v3 推荐方式) norm_emotion = normalize_emotion(emotion) if norm_emotion: - input_payload["emotion"] = norm_emotion + input_payload["instruction"] = f"你说话的情感是{norm_emotion}。" + # 语言 → language_hints 数组(仅取第一个元素生效); + # 系统音色(非克隆/非 voice_id 中包含下划线以外的短 ID)仅传 zh/en,其他语言不传避免报错 + norm_lang = normalize_language(language) + # 简单判断:克隆音色一般是长 voice_id(包含 dash 或长度>20),对其不做语言限制; + # 系统音色(如 longxiaochun_v3)仅 zh/en 传 language_hints + is_system_voice = ("_" in voice_id or voice_id.startswith("long")) and len(voice_id) < 32 + if (is_system_voice and norm_lang in _SYSTEM_VOICE_LANGS) or (not is_system_voice): + input_payload["language_hints"] = [norm_lang] payload = {"model": self._model, "input": input_payload} @@ -556,6 +606,7 @@ class CosyVoiceService: speed: float = 1.0, volume: int = 50, emotion: str = "", + language: str = "zh", timeout: float = 120.0, ) -> SynthesizeResult: """语音合成(同步非流式). @@ -588,6 +639,7 @@ class CosyVoiceService: speed=speed, volume=volume, emotion=emotion, + language=language, ) return SynthesizeResult( diff --git a/packages/application/tts_job/streaming_service.py b/packages/application/tts_job/streaming_service.py index b9ded2270..537ee7eb8 100644 --- a/packages/application/tts_job/streaming_service.py +++ b/packages/application/tts_job/streaming_service.py @@ -92,6 +92,7 @@ class TTSStreamingService: sample_rate=sample_rate, format=audio_format, speed=speed, + language="zh-CN", ) except CosyVoiceError as e: logger.error(f"流式合成失败: {e}") @@ -158,6 +159,7 @@ class TTSStreamingService: sample_rate=sample_rate, format=audio_format, speed=speed, + language="zh-CN", ) audio_url = result.get("audio_url", "") if audio_url: diff --git a/packages/application/tts_job/workflow.py b/packages/application/tts_job/workflow.py index d26249bc0..3ca40b0e6 100644 --- a/packages/application/tts_job/workflow.py +++ b/packages/application/tts_job/workflow.py @@ -146,6 +146,7 @@ class TTSWorkflowService: _meta = dict(job.metadata) _speed = float(_meta.get("speed", 1.0) or 1.0) _emotion = str(_meta.get("emotion", "") or "") + _language = str(_meta.get("language", "zh-CN") or "zh-CN") submit_result = self.cosyvoice_service.submit_synthesize_task( text=job.input_text, voice_id=job.voice_id, @@ -153,6 +154,7 @@ class TTSWorkflowService: format=job.format, speed=_speed, emotion=_emotion, + language=_language, ) # 保存 task_id / request_id 到 metadata @@ -289,6 +291,7 @@ class TTSWorkflowService: speed = float(job_metadata.get("speed", 1.0)) volume = int(job_metadata.get("volume", 50)) emotion = str(job_metadata.get("emotion", "") or "") + language = str(job_metadata.get("language", "zh-CN") or "zh-CN") result = self.cosyvoice_service.submit_synthesize_task( text=job.input_text, @@ -298,6 +301,7 @@ class TTSWorkflowService: speed=speed, volume=volume, emotion=emotion, + language=language, ) audio_url = result.get("audio_url", "") if not audio_url: @@ -423,6 +427,7 @@ class TTSWorkflowService: format=job.format, speed=_seg_speed, emotion=_seg_emotion, + language=_seg_meta.get("language", "zh-CN") or "zh-CN", ) future_to_idx[future] = idx @@ -548,6 +553,7 @@ class TTSWorkflowService: speed=speed, volume=volume, emotion=emotion, + language=job.metadata.get("language", "zh-CN") if hasattr(job, "metadata") else "zh-CN", ) future_to_idx[future] = idx diff --git a/tests/unit/test_ai_avatar_emotion_tts_lipsync.py b/tests/unit/test_ai_avatar_emotion_tts_lipsync.py index 22132f59e..0ecee9557 100644 --- a/tests/unit/test_ai_avatar_emotion_tts_lipsync.py +++ b/tests/unit/test_ai_avatar_emotion_tts_lipsync.py @@ -20,28 +20,36 @@ os.environ.setdefault("JWT_SECRET_KEY", "dev-secret-key-for-testing") def test_normalize_emotion_english_values(): from packages.application.cosyvoice_service import normalize_emotion - assert normalize_emotion("natural") == "natural" - assert normalize_emotion("excited") == "excited" - assert normalize_emotion("calm") == "calm" - assert normalize_emotion("friendly") == "friendly" - # 大小写 / 空白容错 - assert normalize_emotion(" Excited ") == "excited" + # 英文情绪统一归一化为中文描述词(用于 instruction 自然语言指令) + assert normalize_emotion("natural") == "自然" + assert normalize_emotion("excited") == "兴奋开心" + assert normalize_emotion("calm") == "沉稳平静" + assert normalize_emotion("friendly") == "亲切友好" + assert normalize_emotion("happy") == "开心愉快" + assert normalize_emotion("sad") == "悲伤难过" + assert normalize_emotion("angry") == "愤怒" + assert normalize_emotion("surprised") == "惊讶" + assert normalize_emotion("fearful") == "恐惧" + assert normalize_emotion("disgusted") == "厌恶" def test_normalize_emotion_chinese_values(): from packages.application.cosyvoice_service import normalize_emotion - assert normalize_emotion("自然") == "natural" - assert normalize_emotion("兴奋") == "excited" - assert normalize_emotion("沉稳") == "calm" - assert normalize_emotion("亲切") == "friendly" + assert normalize_emotion("自然") == "自然" + assert normalize_emotion("兴奋") == "兴奋开心" + assert normalize_emotion("沉稳") == "沉稳平静" + assert normalize_emotion("亲切") == "亲切友好" + assert normalize_emotion("开心") == "开心愉快" + assert normalize_emotion("悲伤") == "悲伤难过" + assert normalize_emotion("愤怒") == "愤怒" + assert normalize_emotion("惊讶") == "惊讶" def test_normalize_emotion_invalid_returns_empty(): from packages.application.cosyvoice_service import normalize_emotion assert normalize_emotion("") == "" - assert normalize_emotion("angry") == "" assert normalize_emotion("喜怒哀乐") == "" @@ -83,24 +91,43 @@ def _make_service_with_captured_client(captured: dict): return svc -def test_submit_synthesize_payload_includes_emotion_and_rate(): +def test_submit_synthesize_payload_uses_instruction_and_language_hints(): + """#1898: emotion 通过 instruction 中文自然语言指令传递;语言用 language_hints.""" captured: dict = {} svc = _make_service_with_captured_client(captured) - svc.submit_synthesize_task(text="你好", voice_id="v-1", speed=1.5, emotion="兴奋") + svc.submit_synthesize_task(text="你好", voice_id="longxiaochun_v3", speed=1.5, emotion="兴奋", language="zh-CN") inp = captured["json"]["input"] - assert inp["emotion"] == "excited" + # 不再传 emotion 枚举字段 + assert "emotion" not in inp + # 情绪通过 instruction 中文指令 + assert inp["instruction"] == "你说话的情感是兴奋开心。" + # rate 字段保持 assert inp["rate"] == 1.5 + # 系统音色 zh → language_hints=["zh"] + assert inp["language_hints"] == ["zh"] -def test_submit_synthesize_payload_omits_emotion_when_empty(): +def test_submit_synthesize_payload_omits_instruction_when_emotion_empty(): captured: dict = {} svc = _make_service_with_captured_client(captured) - svc.submit_synthesize_task(text="你好", voice_id="v-1") + svc.submit_synthesize_task(text="hello", voice_id="longxiaochun_v3", language="en-US") - assert "emotion" not in captured["json"]["input"] + inp = captured["json"]["input"] + assert "instruction" not in inp + assert inp["language_hints"] == ["en"] + + +def test_submit_synthesize_payload_english_emotion_maps_to_chinese(): + captured: dict = {} + svc = _make_service_with_captured_client(captured) + + svc.submit_synthesize_task(text="hi", voice_id="longxiaochun_v3", emotion="sad") + + inp = captured["json"]["input"] + assert inp["instruction"] == "你说话的情感是悲伤难过。" # ── 对口型 TTS 直生分支 ───────────────────────────────────────────────── @@ -146,7 +173,7 @@ def test_create_job_tts_direct_mode_synthesizes_audio(): # v4: TTS 模式下 create_job 返回 tts_processing 状态,dispatch Celery 任务 assert job.status == "tts_processing" - assert job.emotion == "excited" + assert job.emotion == "兴奋" # 存储原始情绪值,normalize 在 CosyVoiceService 内部完成 assert job.speed == 1.2 # 不直接调用 CosyVoice(由 Celery 任务处理) cosy.submit_synthesize_task.assert_not_called() diff --git a/tests/unit/test_lipsync_routes.py b/tests/unit/test_lipsync_routes.py index 3cba83375..b6c994d5e 100644 --- a/tests/unit/test_lipsync_routes.py +++ b/tests/unit/test_lipsync_routes.py @@ -254,7 +254,7 @@ class TestLipsyncServiceUnit: mock_task.apply_async.assert_called_once() # job 记录透传字段 assert job.speed == 1.2 - assert job.emotion == "excited" + assert job.emotion == "兴奋" # MediaKit 尚未提交(由 Celery 任务处理) mock_mediakit.submit_lipsync.assert_not_called() diff --git a/tests/unit/test_tts_preview.py b/tests/unit/test_tts_preview.py index b07bb132c..c3fae5822 100644 --- a/tests/unit/test_tts_preview.py +++ b/tests/unit/test_tts_preview.py @@ -102,6 +102,7 @@ class TestTTSPreviewEndpoint: voice_id="longxiaochun", speed=1.0, emotion="", + language="zh-CN", ) def test_preview_with_speed(self): @@ -148,6 +149,7 @@ class TestTTSPreviewEndpoint: voice_id="v1", speed=1.5, emotion="", + language="zh-CN", ) def test_preview_cosyvoice_error_returns_502(self): @@ -333,6 +335,7 @@ class TestTTSPreviewEndpoint: voice_id="cosyvoice_actual_voice_123", speed=1.0, emotion="", + language="zh-CN", ) # Verify repo was queried with the UUID mock_clone_repo.get.assert_called_once_with("abc123-uuid-of-profile") @@ -410,6 +413,7 @@ class TestTTSPreviewEndpoint: voice_id="longxiaoxia_v3", speed=1.0, emotion="", + language="zh-CN", ) def test_preview_clone_voice_wrong_user_returns_403(self): diff --git a/tests/unit/test_tts_streaming_service.py b/tests/unit/test_tts_streaming_service.py old mode 100755 new mode 100644 index ae7e14b06..eb7aa9ff3 --- a/tests/unit/test_tts_streaming_service.py +++ b/tests/unit/test_tts_streaming_service.py @@ -176,6 +176,7 @@ class TestStreamShortText: sample_rate=22050, format="wav", speed=1.5, + language="zh-CN", ) @pytest.mark.asyncio