From 19eeb1b475dca30c1adb050b4783c77a0701be2b Mon Sep 17 00:00:00 2001 From: xiaoxia Date: Tue, 15 Sep 2026 14:02:49 +0800 Subject: [PATCH] =?UTF-8?q?fix(#1898):=20TTS=20=E6=83=85=E7=BB=AA=E6=9E=9A?= =?UTF-8?q?=E4=B8=BE=E7=BB=9F=E4=B8=80=E4=B8=BA7=E7=A7=8D=E6=A0=87?= =?UTF-8?q?=E5=87=86=E8=8B=B1=E6=96=87=E6=A0=87=E7=AD=BE=EF=BC=88P1?= =?UTF-8?q?=EF=BC=89=20(#1932)?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Co-authored-by: xiaoxia Co-committed-by: xiaoxia --- apps/api/app/api/routes/voice_clones.py | 24 +++- apps/api/app/schemas/lipsync.py | 10 +- apps/api/app/services/lipsync_service.py | 2 +- packages/application/cosyvoice_service.py | 41 +++--- tests/unit/test_cosyvoice_emotion.py | 145 ++++++++++++++++++++++ tests/unit/test_voice_clone_preview.py | 2 +- 6 files changed, 202 insertions(+), 22 deletions(-) create mode 100644 tests/unit/test_cosyvoice_emotion.py diff --git a/apps/api/app/api/routes/voice_clones.py b/apps/api/app/api/routes/voice_clones.py index 0a2a5afc5..2112abbbe 100755 --- a/apps/api/app/api/routes/voice_clones.py +++ b/apps/api/app/api/routes/voice_clones.py @@ -287,7 +287,22 @@ def retry_voice_clone( return _to_response(profile) -_ALLOWED_PREVIEW_EMOTIONS = {"", "natural", "excited", "calm", "friendly"} +_ALLOWED_PREVIEW_EMOTIONS = { + "", + # 7 种标准英文枚举 + "neutral", + "happy", + "sad", + "angry", + "surprised", + "fearful", + "disgusted", + # 旧英文 4 枚举兼容 + "natural", + "excited", + "calm", + "friendly", +} @router.get("/{clone_id}/preview", response_model=VoiceClonePreviewResponse) @@ -295,7 +310,10 @@ def get_voice_clone_preview( clone_id: str, text: str = Query("", description="自定义试听文本,为空则使用默认示例"), speed: float = Query(1.0, ge=0.5, le=2.0, description="语速,0.5-2.0,默认 1.0"), - emotion: str = Query("", description="情绪:natural/excited/calm/friendly,空字符串为默认自然"), + emotion: str = Query( + "", + description="情绪:neutral/happy/sad/angry/surprised/fearful/disgusted,兼容旧值 natural/excited/calm/friendly,空为默认自然", + ), authenticated_user: AuthenticatedUser = Depends(get_current_user), repository: SQLAlchemyVoiceCloneProfileRepository = Depends(get_voice_clone_profile_repository), cosyvoice: CosyVoiceService = Depends(get_cosyvoice_service), @@ -311,7 +329,7 @@ def get_voice_clone_preview( if emotion not in _ALLOWED_PREVIEW_EMOTIONS: raise HTTPException( status_code=status.HTTP_400_BAD_REQUEST, - detail=f"不支持的 emotion 值: {emotion},可选: natural/excited/calm/friendly 或留空", + detail=f"不支持的 emotion 值: {emotion},可选: neutral/happy/sad/angry/surprised/fearful/disgusted(兼容 natural/excited/calm/friendly)或留空", ) use_case = GetVoiceCloneUseCase(repository) diff --git a/apps/api/app/schemas/lipsync.py b/apps/api/app/schemas/lipsync.py index ed8cf8e3a..c3680ac6c 100644 --- a/apps/api/app/schemas/lipsync.py +++ b/apps/api/app/schemas/lipsync.py @@ -67,7 +67,9 @@ class CreateLipsyncJobRequest(BaseModel): voice_id: str = Field("", description="音色 ID(预置音色或克隆音色 profile UUID)") script_text: str = Field("", description="要合成的文案(直生模式必填,最长 5000 字符)") speed: float = Field(1.0, ge=0.5, le=2.0, description="语速(0.5-2.0),默认 1.0") - emotion: str = Field("", description="情绪(中文/英文:自然/兴奋/沉稳/亲切/开心/悲伤/愤怒/惊讶/恐惧/厌恶 等)") + emotion: str = Field( + "", description="情绪(英文枚举 neutral/happy/sad/angry/surprised/fearful/disgusted,兼容中文/旧值;空为默认)" + ) enable_video_loop: bool = Field( True, description="音频长于视频时是否循环画面(AI数字人默认开启,防止音频长于视频被截断)" @@ -120,7 +122,11 @@ class AiAvatarTtsPreviewRequest(BaseModel): voice_id: str = Field(..., min_length=1, max_length=128, description="音色 ID") script_text: str = Field(..., min_length=1, max_length=5000, description="要合成的文案") speed: float = Field(1.0, ge=0.5, le=2.0, description="语速(0.5-2.0),默认 1.0") - emotion: str = Field("natural", max_length=32, description="情绪") + emotion: str = Field( + "neutral", + max_length=32, + description="情绪(英文枚举 neutral/happy/sad/angry/surprised/fearful/disgusted,或中文/旧值)", + ) class AiAvatarTtsPreviewResponse(BaseModel): diff --git a/apps/api/app/services/lipsync_service.py b/apps/api/app/services/lipsync_service.py index 9ea698354..10110d27d 100644 --- a/apps/api/app/services/lipsync_service.py +++ b/apps/api/app/services/lipsync_service.py @@ -360,7 +360,7 @@ class LipsyncService: voice_id: str, script_text: str, speed: float = 1.0, - emotion: str = "natural", + emotion: str = "neutral", ) -> dict: """同步做 TTS 合成 + 下载 + ffprobe + 句子时间戳计算. diff --git a/packages/application/cosyvoice_service.py b/packages/application/cosyvoice_service.py index c3ee745b4..fd70f8afa 100644 --- a/packages/application/cosyvoice_service.py +++ b/packages/application/cosyvoice_service.py @@ -29,35 +29,40 @@ logger = logging.getLogger(__name__) # CosyVoice v3 情绪通过 input.instruction 中文自然语言指令控制(不再使用枚举 emotion 字段)。 # 前端可传中文或英文情绪标签,统一归一化为中文描述词,再拼进 instruction。 # 映射表 key: 小写中文/英文 → 中文情绪描述词 -EMOTION_MAP = { - # 原有四值(中文 + 英文) - "自然": "自然", - "兴奋": "兴奋开心", - "沉稳": "沉稳平静", - "亲切": "亲切友好", +# 第一层:7 种标准英文枚举(与前端 emotion enum 对齐,必须存在且作为第一组) +# 第二层:旧英文 4 枚举(natural/excited/calm/friendly),保持向后兼容 +# 第三层:中文标签别名 +EMOTION_MAP: dict[str, str] = { + # 7 种标准英文枚举(P1 修复:以英文枚举为标准 key,neutral 为默认) + "neutral": "自然", + "happy": "开心愉快", + "sad": "悲伤难过", + "angry": "愤怒", + "surprised": "惊讶", + "fearful": "恐惧", + "disgusted": "厌恶", + # 旧英文 4 枚举兼容(自然映射到对应中文) "natural": "自然", "excited": "兴奋开心", "calm": "沉稳平静", "friendly": "亲切友好", - "happy": "开心愉快", - # 新增情绪 + # 中文标签 + "自然": "自然", "开心": "开心愉快", "愉快": "开心愉快", - "sad": "悲伤难过", + "兴奋": "兴奋开心", "悲伤": "悲伤难过", "难过": "悲伤难过", - "angry": "愤怒", "愤怒": "愤怒", "生气": "愤怒", - "surprised": "惊讶", "惊讶": "惊讶", "惊奇": "惊讶", - "fearful": "恐惧", "恐惧": "恐惧", "害怕": "恐惧", - "disgusted": "厌恶", "厌恶": "厌恶", "讨厌": "厌恶", + "沉稳": "沉稳平静", + "亲切": "亲切友好", "严肃": "严肃", "温柔": "温柔", } @@ -66,11 +71,16 @@ EMOTION_MAP = { def normalize_emotion(emotion: str) -> str: """将前端情绪值归一化为中文描述词,用于拼入 instruction. - 支持中文/英文;空串或未知值返回空串(调用方据此决定是否传 instruction)。 + 支持 7 种标准英文枚举(neutral/happy/sad/angry/surprised/fearful/disgusted)、 + 旧英文兼容值(natural/excited/calm/friendly)以及中文标签;大小写不敏感。 + 空串/空白/未知值返回空串(调用方据此决定是否传 instruction)。 """ if not emotion: return "" key = emotion.strip() + if not key: + return "" + # 大小写不敏感:先按原 key 查,再按 lower 查 mapped = EMOTION_MAP.get(key) or EMOTION_MAP.get(key.lower()) if mapped: return mapped @@ -516,7 +526,8 @@ class CosyVoiceService: format: 输出格式(mp3/wav/pcm),空表示使用配置默认值 speed: 语速(0.5-2.0),1.0 为正常速度 volume: 音量(0-100),默认 50 - emotion: 情绪(自然/兴奋/沉稳/亲切/开心/悲伤/愤怒/惊讶/恐惧/厌恶 等),空串不传 + emotion: 情绪,英文枚举 neutral/happy/sad/angry/surprised/fearful/disgusted, + 兼容旧值 natural/excited/calm/friendly 及中文标签;空串不传 language: 语言代码(zh/en 等,默认 zh;系统音色仅 zh/en 传 language_hints) Returns: diff --git a/tests/unit/test_cosyvoice_emotion.py b/tests/unit/test_cosyvoice_emotion.py new file mode 100644 index 000000000..ed5cbcecc --- /dev/null +++ b/tests/unit/test_cosyvoice_emotion.py @@ -0,0 +1,145 @@ +"""CosyVoice EMOTION_MAP 与 normalize_emotion 单测(P1 修复 #1898). + +覆盖: +- 7 种标准英文枚举 neutral/happy/sad/angry/surprised/fearful/disgusted +- 大小写不敏感 +- 中文标签别名 +- 旧英文 4 枚举兼容(natural/excited/calm/friendly) +- 空串/空白/未知值边界 +- instruction 拼接格式("你说话的情感是{emotion}。" 单句号) +""" + +from __future__ import annotations + +import logging + +import pytest + +from packages.application.cosyvoice_service import ( + EMOTION_MAP, + normalize_emotion, +) + +SEVEN_STANDARD_ENUMS = [ + ("neutral", "自然"), + ("happy", "开心愉快"), + ("sad", "悲伤难过"), + ("angry", "愤怒"), + ("surprised", "惊讶"), + ("fearful", "恐惧"), + ("disgusted", "厌恶"), +] + +OLD_FOUR_ENUMS = [ + ("natural", "自然"), + ("excited", "兴奋开心"), + ("calm", "沉稳平静"), + ("friendly", "亲切友好"), +] + +CHINESE_ALIASES = [ + ("自然", "自然"), + ("开心", "开心愉快"), + ("愉快", "开心愉快"), + ("兴奋", "兴奋开心"), + ("悲伤", "悲伤难过"), + ("难过", "悲伤难过"), + ("愤怒", "愤怒"), + ("生气", "愤怒"), + ("惊讶", "惊讶"), + ("惊奇", "惊讶"), + ("恐惧", "恐惧"), + ("害怕", "恐惧"), + ("厌恶", "厌恶"), + ("讨厌", "厌恶"), + ("沉稳", "沉稳平静"), + ("亲切", "亲切友好"), + ("严肃", "严肃"), + ("温柔", "温柔"), +] + + +class TestEmotionMapSevenStandard: + """7 种标准英文枚举必须作为 key 存在并映射到正确中文描述词。""" + + @pytest.mark.parametrize("enum_key,expected", SEVEN_STANDARD_ENUMS) + def test_standard_enum_present(self, enum_key: str, expected: str) -> None: + assert enum_key in EMOTION_MAP + assert EMOTION_MAP[enum_key] == expected + + @pytest.mark.parametrize("enum_key,expected", SEVEN_STANDARD_ENUMS) + def test_normalize_standard_enum(self, enum_key: str, expected: str) -> None: + assert normalize_emotion(enum_key) == expected + + @pytest.mark.parametrize("enum_key,expected", SEVEN_STANDARD_ENUMS) + def test_normalize_case_insensitive(self, enum_key: str, expected: str) -> None: + """大小写不敏感:Neutral/HAPPY/Angry 等都能匹配。""" + assert normalize_emotion(enum_key.upper()) == expected + assert normalize_emotion(enum_key.capitalize()) == expected + assert normalize_emotion(f" {enum_key} ") == expected + + def test_neutral_is_first_standard_key(self) -> None: + """neutral 必须是标准枚举第一 key(作为默认值语义)。""" + en_keys = [k for k in EMOTION_MAP if all(ord(c) < 128 for c in k)] + assert en_keys[0] == "neutral" + + +class TestEmotionMapBackwardCompat: + """旧英文 4 枚举与中文标签必须继续兼容。""" + + @pytest.mark.parametrize("old_key,expected", OLD_FOUR_ENUMS) + def test_old_four_enums(self, old_key: str, expected: str) -> None: + assert normalize_emotion(old_key) == expected + + @pytest.mark.parametrize("cn_key,expected", CHINESE_ALIASES) + def test_chinese_aliases(self, cn_key: str, expected: str) -> None: + assert normalize_emotion(cn_key) == expected + + +class TestNormalizeEmotionEdgeCases: + """空串、空白、未知值等边界。""" + + @pytest.mark.parametrize("empty_val", ["", None]) + def test_empty_returns_empty(self, empty_val) -> None: + assert normalize_emotion(empty_val) == "" + + @pytest.mark.parametrize("ws", [" ", "\t", "\n", " \n "]) + def test_whitespace_only_returns_empty(self, ws: str) -> None: + assert normalize_emotion(ws) == "" + + def test_unknown_value_returns_empty_and_warns(self, caplog) -> None: + caplog.set_level(logging.WARNING) + assert normalize_emotion("not_a_real_emotion") == "" + assert any("未知的 emotion" in r.message for r in caplog.records) + + def test_strips_leading_trailing_whitespace(self) -> None: + assert normalize_emotion(" happy ") == "开心愉快" + + +class TestEmotionInstructionFormat: + """7 种 emotion 生成的 instruction 必须符合 CosyVoice v3 格式要求。""" + + @pytest.mark.parametrize("enum_key,expected_desc", SEVEN_STANDARD_ENUMS) + def test_instruction_starts_with_prefix(self, enum_key: str, expected_desc: str) -> None: + norm = normalize_emotion(enum_key) + instruction = f"你说话的情感是{norm}。" + assert instruction.startswith("你说话的情感是") + assert instruction.endswith("。") + + @pytest.mark.parametrize("enum_key,expected_desc", SEVEN_STANDARD_ENUMS) + def test_instruction_single_period(self, enum_key: str, expected_desc: str) -> None: + """instruction 只能有一个中文句号(结尾),防止误注入多个句子。""" + norm = normalize_emotion(enum_key) + instruction = f"你说话的情感是{norm}。" + assert instruction.count("。") == 1 + + @pytest.mark.parametrize("enum_key,expected_desc", SEVEN_STANDARD_ENUMS) + def test_instruction_contains_expected_description(self, enum_key: str, expected_desc: str) -> None: + norm = normalize_emotion(enum_key) + instruction = f"你说话的情感是{norm}。" + assert expected_desc in instruction + + def test_empty_emotion_produces_no_instruction(self) -> None: + """空 emotion 不应拼 instruction(调用方据此跳过字段)。""" + assert normalize_emotion("") == "" + assert normalize_emotion(" ") == "" diff --git a/tests/unit/test_voice_clone_preview.py b/tests/unit/test_voice_clone_preview.py index a88d459eb..27c7c5091 100755 --- a/tests/unit/test_voice_clone_preview.py +++ b/tests/unit/test_voice_clone_preview.py @@ -196,7 +196,7 @@ class TestVoiceClonePreview: cosyvoice = MagicMock() with pytest.raises(HTTPException) as exc_info: - self._call_preview(profile, cosyvoice, emotion="angry") + self._call_preview(profile, cosyvoice, emotion="invalid_emotion_xyz") assert exc_info.value.status_code == 400 cosyvoice.synthesize_speech.assert_not_called()