diff --git a/apps/api/app/api/routes/voice_clones.py b/apps/api/app/api/routes/voice_clones.py index 2112abbbe..d56a8fd92 100755 --- a/apps/api/app/api/routes/voice_clones.py +++ b/apps/api/app/api/routes/voice_clones.py @@ -289,7 +289,7 @@ def retry_voice_clone( _ALLOWED_PREVIEW_EMOTIONS = { "", - # 7 种标准英文枚举 + # 7 种标准英文枚举(CosyVoice v3 官方值) "neutral", "happy", "sad", @@ -297,11 +297,31 @@ _ALLOWED_PREVIEW_EMOTIONS = { "surprised", "fearful", "disgusted", - # 旧英文 4 枚举兼容 + # 前端中文 7 标签 + "中立", + "开心", + "难过", + "生气", + "惊讶", + "恐惧", + "厌恶", + # 旧英文 4 枚举 + 常见中文别名兼容 "natural", "excited", "calm", "friendly", + "自然", + "愉快", + "高兴", + "快乐", + "兴奋", + "悲伤", + "愤怒", + "惊奇", + "害怕", + "讨厌", + "沉稳", + "亲切", } @@ -329,7 +349,7 @@ def get_voice_clone_preview( if emotion not in _ALLOWED_PREVIEW_EMOTIONS: raise HTTPException( status_code=status.HTTP_400_BAD_REQUEST, - detail=f"不支持的 emotion 值: {emotion},可选: neutral/happy/sad/angry/surprised/fearful/disgusted(兼容 natural/excited/calm/friendly)或留空", + detail=f"不支持的 emotion 值: {emotion},可选: neutral/happy/sad/angry/surprised/fearful/disgusted 或中文 中立/开心/难过/生气/惊讶/恐惧/厌恶,或留空", ) use_case = GetVoiceCloneUseCase(repository) diff --git a/apps/api/app/schemas/lipsync.py b/apps/api/app/schemas/lipsync.py index c3680ac6c..8bd9ef9bc 100644 --- a/apps/api/app/schemas/lipsync.py +++ b/apps/api/app/schemas/lipsync.py @@ -68,7 +68,8 @@ class CreateLipsyncJobRequest(BaseModel): script_text: str = Field("", description="要合成的文案(直生模式必填,最长 5000 字符)") speed: float = Field(1.0, ge=0.5, le=2.0, description="语速(0.5-2.0),默认 1.0") emotion: str = Field( - "", description="情绪(英文枚举 neutral/happy/sad/angry/surprised/fearful/disgusted,兼容中文/旧值;空为默认)" + "", + description="情绪(英文枚举 neutral/happy/sad/angry/surprised/fearful/disgusted,或中文 中立/开心/难过/生气/惊讶/恐惧/厌恶;空为默认自然)", ) enable_video_loop: bool = Field( @@ -125,7 +126,7 @@ class AiAvatarTtsPreviewRequest(BaseModel): emotion: str = Field( "neutral", max_length=32, - description="情绪(英文枚举 neutral/happy/sad/angry/surprised/fearful/disgusted,或中文/旧值)", + description="情绪(英文枚举 neutral/happy/sad/angry/surprised/fearful/disgusted,或中文 中立/开心/难过/生气/惊讶/恐惧/厌恶;默认 neutral)", ) diff --git a/packages/application/cosyvoice_service.py b/packages/application/cosyvoice_service.py index fd70f8afa..334c06e94 100644 --- a/packages/application/cosyvoice_service.py +++ b/packages/application/cosyvoice_service.py @@ -27,53 +27,58 @@ logger = logging.getLogger(__name__) # CosyVoice v3 情绪通过 input.instruction 中文自然语言指令控制(不再使用枚举 emotion 字段)。 -# 前端可传中文或英文情绪标签,统一归一化为中文描述词,再拼进 instruction。 -# 映射表 key: 小写中文/英文 → 中文情绪描述词 -# 第一层:7 种标准英文枚举(与前端 emotion enum 对齐,必须存在且作为第一组) -# 第二层:旧英文 4 枚举(natural/excited/calm/friendly),保持向后兼容 -# 第三层:中文标签别名 +# 官方文档:instruction 格式严格为 "你说话的情感是<情感值>。",结尾中文句号不可省略; +# 情感值必须是 7 种英文枚举之一:neutral/happy/sad/angry/surprised/fearful/disgusted。 +# 参考:https://help.aliyun.com/zh/model-studio/cosyvoice-voice-list +# 前端可传英文枚举或中文标签(中立/开心/难过/生气/惊讶/恐惧/厌恶),统一归一化为英文枚举。 +# 映射表 key(不区分大小写): 英文枚举/旧英文/中文标签 → 7 种标准英文枚举 EMOTION_MAP: dict[str, str] = { - # 7 种标准英文枚举(P1 修复:以英文枚举为标准 key,neutral 为默认) - "neutral": "自然", - "happy": "开心愉快", - "sad": "悲伤难过", - "angry": "愤怒", - "surprised": "惊讶", - "fearful": "恐惧", - "disgusted": "厌恶", - # 旧英文 4 枚举兼容(自然映射到对应中文) - "natural": "自然", - "excited": "兴奋开心", - "calm": "沉稳平静", - "friendly": "亲切友好", - # 中文标签 - "自然": "自然", - "开心": "开心愉快", - "愉快": "开心愉快", - "兴奋": "兴奋开心", - "悲伤": "悲伤难过", - "难过": "悲伤难过", - "愤怒": "愤怒", - "生气": "愤怒", - "惊讶": "惊讶", - "惊奇": "惊讶", - "恐惧": "恐惧", - "害怕": "恐惧", - "厌恶": "厌恶", - "讨厌": "厌恶", - "沉稳": "沉稳平静", - "亲切": "亲切友好", - "严肃": "严肃", - "温柔": "温柔", + # ── 7 种标准英文枚举(CosyVoice v3 官方支持的情感值)── + "neutral": "neutral", + "happy": "happy", + "sad": "sad", + "angry": "angry", + "surprised": "surprised", + "fearful": "fearful", + "disgusted": "disgusted", + # ── 前端中文 7 标签(P1:前端已扩展为这 7 个中文选项)── + "中立": "neutral", + "开心": "happy", + "难过": "sad", + "生气": "angry", + "惊讶": "surprised", + "恐惧": "fearful", + "厌恶": "disgusted", + # ── 常见中文别名 ── + "自然": "neutral", + "愉快": "happy", + "高兴": "happy", + "快乐": "happy", + "兴奋": "happy", # 旧 excited → 映射到最接近的 happy + "悲伤": "sad", + "愤怒": "angry", + "惊奇": "surprised", + "害怕": "fearful", + "讨厌": "disgusted", + # ── 旧英文 4 枚举兼容(natural/excited/calm/friendly 归并到最接近的标准值)── + "natural": "neutral", + "excited": "happy", + "calm": "neutral", + "friendly": "happy", } def normalize_emotion(emotion: str) -> str: - """将前端情绪值归一化为中文描述词,用于拼入 instruction. + """将前端情绪值归一化为 CosyVoice v3 官方英文枚举,用于拼入 instruction. - 支持 7 种标准英文枚举(neutral/happy/sad/angry/surprised/fearful/disgusted)、 - 旧英文兼容值(natural/excited/calm/friendly)以及中文标签;大小写不敏感。 - 空串/空白/未知值返回空串(调用方据此决定是否传 instruction)。 + 支持: + - 7 种标准英文枚举(neutral/happy/sad/angry/surprised/fearful/disgusted); + - 前端中文 7 标签(中立/开心/难过/生气/惊讶/恐惧/厌恶); + - 常见中文别名与旧英文 4 枚举兼容值(natural/excited/calm/friendly); + - 大小写不敏感。 + + 返回:始终返回 7 种英文枚举之一;空串/None 返回空串(调用方据此不传 instruction, + CosyVoice 按默认自然情绪合成);未知值记录 warning 并默认 "neutral"。 """ if not emotion: return "" @@ -84,8 +89,9 @@ def normalize_emotion(emotion: str) -> str: mapped = EMOTION_MAP.get(key) or EMOTION_MAP.get(key.lower()) if mapped: return mapped - logger.warning("未知的 emotion 值,忽略: %r", emotion) - return "" + # 未识别的情绪值:默认 neutral,保证合成能正常进行 + logger.warning("未知的 emotion 值 %r,默认使用 neutral", emotion) + return "neutral" # 系统音色仅支持中文/英文(language_hints 取值) @@ -526,8 +532,9 @@ class CosyVoiceService: format: 输出格式(mp3/wav/pcm),空表示使用配置默认值 speed: 语速(0.5-2.0),1.0 为正常速度 volume: 音量(0-100),默认 50 - emotion: 情绪,英文枚举 neutral/happy/sad/angry/surprised/fearful/disgusted, - 兼容旧值 natural/excited/calm/friendly 及中文标签;空串不传 + emotion: 情绪,英文枚举 neutral/happy/sad/angry/surprised/fearful/disgusted + 或前端中文标签(中立/开心/难过/生气/惊讶/恐惧/厌恶),兼容旧值 + natural/excited/calm/friendly;空串不传,未知值默认 neutral language: 语言代码(zh/en 等,默认 zh;系统音色仅 zh/en 传 language_hints) Returns: @@ -556,7 +563,7 @@ class CosyVoiceService: "rate": speed, "volume": volume, } - # 情绪 → instruction 自然语言指令(CosyVoice v3 推荐方式) + # 情绪 → instruction 严格按官方格式: 你说话的情感是{emotion_enum}。(结尾中文句号) norm_emotion = normalize_emotion(emotion) if norm_emotion: input_payload["instruction"] = f"你说话的情感是{norm_emotion}。" diff --git a/tests/unit/test_ai_avatar_emotion_tts_lipsync.py b/tests/unit/test_ai_avatar_emotion_tts_lipsync.py index 0ecee9557..c3b01c33c 100644 --- a/tests/unit/test_ai_avatar_emotion_tts_lipsync.py +++ b/tests/unit/test_ai_avatar_emotion_tts_lipsync.py @@ -20,37 +20,45 @@ os.environ.setdefault("JWT_SECRET_KEY", "dev-secret-key-for-testing") def test_normalize_emotion_english_values(): from packages.application.cosyvoice_service import normalize_emotion - # 英文情绪统一归一化为中文描述词(用于 instruction 自然语言指令) - assert normalize_emotion("natural") == "自然" - assert normalize_emotion("excited") == "兴奋开心" - assert normalize_emotion("calm") == "沉稳平静" - assert normalize_emotion("friendly") == "亲切友好" - assert normalize_emotion("happy") == "开心愉快" - assert normalize_emotion("sad") == "悲伤难过" - assert normalize_emotion("angry") == "愤怒" - assert normalize_emotion("surprised") == "惊讶" - assert normalize_emotion("fearful") == "恐惧" - assert normalize_emotion("disgusted") == "厌恶" + # 英文情绪统一归一化为 CosyVoice v3 官方 7 种英文枚举 + assert normalize_emotion("natural") == "neutral" + assert normalize_emotion("excited") == "happy" + assert normalize_emotion("calm") == "neutral" + assert normalize_emotion("friendly") == "happy" + assert normalize_emotion("happy") == "happy" + assert normalize_emotion("sad") == "sad" + assert normalize_emotion("angry") == "angry" + assert normalize_emotion("surprised") == "surprised" + assert normalize_emotion("fearful") == "fearful" + assert normalize_emotion("disgusted") == "disgusted" def test_normalize_emotion_chinese_values(): from packages.application.cosyvoice_service import normalize_emotion - assert normalize_emotion("自然") == "自然" - assert normalize_emotion("兴奋") == "兴奋开心" - assert normalize_emotion("沉稳") == "沉稳平静" - assert normalize_emotion("亲切") == "亲切友好" - assert normalize_emotion("开心") == "开心愉快" - assert normalize_emotion("悲伤") == "悲伤难过" - assert normalize_emotion("愤怒") == "愤怒" - assert normalize_emotion("惊讶") == "惊讶" + # 中文标签归一化为对应英文枚举(CosyVoice v3 官方值) + assert normalize_emotion("自然") == "neutral" + assert normalize_emotion("兴奋") == "happy" + assert normalize_emotion("开心") == "happy" + assert normalize_emotion("悲伤") == "sad" + assert normalize_emotion("愤怒") == "angry" + assert normalize_emotion("惊讶") == "surprised" + assert normalize_emotion("恐惧") == "fearful" + assert normalize_emotion("厌恶") == "disgusted" + assert normalize_emotion("中立") == "neutral" + assert normalize_emotion("难过") == "sad" + assert normalize_emotion("生气") == "angry" -def test_normalize_emotion_invalid_returns_empty(): +def test_normalize_emotion_invalid_defaults_neutral(): from packages.application.cosyvoice_service import normalize_emotion + # 空串/None 返回空(调用方不传 instruction,走默认自然情绪) assert normalize_emotion("") == "" - assert normalize_emotion("喜怒哀乐") == "" + assert normalize_emotion(None) == "" + # 未知情绪默认 neutral(不中断合成,warning 日志) + assert normalize_emotion("喜怒哀乐") == "neutral" + assert normalize_emotion("unknown_xyz") == "neutral" # ── CosyVoice payload 携带 emotion + rate ────────────────────────────── @@ -92,7 +100,7 @@ def _make_service_with_captured_client(captured: dict): def test_submit_synthesize_payload_uses_instruction_and_language_hints(): - """#1898: emotion 通过 instruction 中文自然语言指令传递;语言用 language_hints.""" + """#1898: emotion 通过 instruction 按官方格式传递(你说话的情感是{英文枚举}。);语言用 language_hints.""" captured: dict = {} svc = _make_service_with_captured_client(captured) @@ -102,7 +110,7 @@ def test_submit_synthesize_payload_uses_instruction_and_language_hints(): # 不再传 emotion 枚举字段 assert "emotion" not in inp # 情绪通过 instruction 中文指令 - assert inp["instruction"] == "你说话的情感是兴奋开心。" + assert inp["instruction"] == "你说话的情感是happy。" # rate 字段保持 assert inp["rate"] == 1.5 # 系统音色 zh → language_hints=["zh"] @@ -127,7 +135,7 @@ def test_submit_synthesize_payload_english_emotion_maps_to_chinese(): svc.submit_synthesize_task(text="hi", voice_id="longxiaochun_v3", emotion="sad") inp = captured["json"]["input"] - assert inp["instruction"] == "你说话的情感是悲伤难过。" + assert inp["instruction"] == "你说话的情感是sad。" # ── 对口型 TTS 直生分支 ───────────────────────────────────────────────── diff --git a/tests/unit/test_cosyvoice_emotion.py b/tests/unit/test_cosyvoice_emotion.py index ed5cbcecc..4a486da50 100644 --- a/tests/unit/test_cosyvoice_emotion.py +++ b/tests/unit/test_cosyvoice_emotion.py @@ -2,11 +2,14 @@ 覆盖: - 7 种标准英文枚举 neutral/happy/sad/angry/surprised/fearful/disgusted + (CosyVoice v3 官方 instruction 情感值,必须原样拼入 "你说话的情感是<值>。") - 大小写不敏感 -- 中文标签别名 -- 旧英文 4 枚举兼容(natural/excited/calm/friendly) -- 空串/空白/未知值边界 -- instruction 拼接格式("你说话的情感是{emotion}。" 单句号) +- 前端中文 7 标签(中立/开心/难过/生气/惊讶/恐惧/厌恶)→ 英文枚举 +- 常见中文别名 +- 旧英文 4 枚举兼容(natural/excited/calm/friendly)→ 归并到最接近的标准值 +- 空串/空白/None 边界 +- 未知值默认 neutral(warning 日志) +- instruction 拼接格式严格符合官方要求("你说话的情感是{emotion_enum}。",单中文句号) """ from __future__ import annotations @@ -20,126 +23,190 @@ from packages.application.cosyvoice_service import ( normalize_emotion, ) +# 7 种官方英文枚举及其期望归一化结果 SEVEN_STANDARD_ENUMS = [ - ("neutral", "自然"), - ("happy", "开心愉快"), - ("sad", "悲伤难过"), - ("angry", "愤怒"), - ("surprised", "惊讶"), - ("fearful", "恐惧"), - ("disgusted", "厌恶"), + "neutral", + "happy", + "sad", + "angry", + "surprised", + "fearful", + "disgusted", ] +# 前端中文 7 标签 → 期望英文枚举 +FRONTEND_CN_LABELS = [ + ("中立", "neutral"), + ("开心", "happy"), + ("难过", "sad"), + ("生气", "angry"), + ("惊讶", "surprised"), + ("恐惧", "fearful"), + ("厌恶", "disgusted"), +] + +# 常见中文别名 → 期望英文枚举 +CN_ALIASES = [ + ("自然", "neutral"), + ("愉快", "happy"), + ("高兴", "happy"), + ("快乐", "happy"), + ("兴奋", "happy"), + ("悲伤", "sad"), + ("愤怒", "angry"), + ("惊奇", "surprised"), + ("害怕", "fearful"), + ("讨厌", "disgusted"), +] + +# 旧英文 4 枚举 → 最接近的标准枚举 OLD_FOUR_ENUMS = [ - ("natural", "自然"), - ("excited", "兴奋开心"), - ("calm", "沉稳平静"), - ("friendly", "亲切友好"), -] - -CHINESE_ALIASES = [ - ("自然", "自然"), - ("开心", "开心愉快"), - ("愉快", "开心愉快"), - ("兴奋", "兴奋开心"), - ("悲伤", "悲伤难过"), - ("难过", "悲伤难过"), - ("愤怒", "愤怒"), - ("生气", "愤怒"), - ("惊讶", "惊讶"), - ("惊奇", "惊讶"), - ("恐惧", "恐惧"), - ("害怕", "恐惧"), - ("厌恶", "厌恶"), - ("讨厌", "厌恶"), - ("沉稳", "沉稳平静"), - ("亲切", "亲切友好"), - ("严肃", "严肃"), - ("温柔", "温柔"), + ("natural", "neutral"), + ("excited", "happy"), + ("calm", "neutral"), + ("friendly", "happy"), ] class TestEmotionMapSevenStandard: - """7 种标准英文枚举必须作为 key 存在并映射到正确中文描述词。""" + """7 种标准英文枚举必须映射到自身(CosyVoice 官方值)。""" - @pytest.mark.parametrize("enum_key,expected", SEVEN_STANDARD_ENUMS) - def test_standard_enum_present(self, enum_key: str, expected: str) -> None: - assert enum_key in EMOTION_MAP - assert EMOTION_MAP[enum_key] == expected + @pytest.mark.parametrize("enum_val", SEVEN_STANDARD_ENUMS) + def test_standard_enum_maps_to_self(self, enum_val: str) -> None: + assert enum_val in EMOTION_MAP + assert EMOTION_MAP[enum_val] == enum_val - @pytest.mark.parametrize("enum_key,expected", SEVEN_STANDARD_ENUMS) - def test_normalize_standard_enum(self, enum_key: str, expected: str) -> None: - assert normalize_emotion(enum_key) == expected + @pytest.mark.parametrize("enum_val", SEVEN_STANDARD_ENUMS) + def test_normalize_standard_enum(self, enum_val: str) -> None: + assert normalize_emotion(enum_val) == enum_val - @pytest.mark.parametrize("enum_key,expected", SEVEN_STANDARD_ENUMS) - def test_normalize_case_insensitive(self, enum_key: str, expected: str) -> None: - """大小写不敏感:Neutral/HAPPY/Angry 等都能匹配。""" - assert normalize_emotion(enum_key.upper()) == expected - assert normalize_emotion(enum_key.capitalize()) == expected - assert normalize_emotion(f" {enum_key} ") == expected + @pytest.mark.parametrize("enum_val", SEVEN_STANDARD_ENUMS) + def test_normalize_case_insensitive(self, enum_val: str) -> None: + """大小写不敏感:NEUTRAL/Happy/Angry 都能正确归一。""" + assert normalize_emotion(enum_val.upper()) == enum_val + assert normalize_emotion(enum_val.capitalize()) == enum_val + assert normalize_emotion(f" {enum_val} ") == enum_val - def test_neutral_is_first_standard_key(self) -> None: - """neutral 必须是标准枚举第一 key(作为默认值语义)。""" - en_keys = [k for k in EMOTION_MAP if all(ord(c) < 128 for c in k)] - assert en_keys[0] == "neutral" + def test_neutral_maps_to_neutral(self) -> None: + """neutral 必须映射到 neutral(默认情绪)。""" + assert normalize_emotion("neutral") == "neutral" -class TestEmotionMapBackwardCompat: - """旧英文 4 枚举与中文标签必须继续兼容。""" +class TestFrontendCnLabels: + """前端中文 7 标签必须映射到对应英文枚举。""" + + @pytest.mark.parametrize("cn,expected", FRONTEND_CN_LABELS) + def test_cn_label_normalizes_to_enum(self, cn: str, expected: str) -> None: + assert normalize_emotion(cn) == expected + + +class TestBackwardCompatAliases: + """旧英文 4 枚举和中文别名必须兼容映射到标准值。""" @pytest.mark.parametrize("old_key,expected", OLD_FOUR_ENUMS) def test_old_four_enums(self, old_key: str, expected: str) -> None: assert normalize_emotion(old_key) == expected - @pytest.mark.parametrize("cn_key,expected", CHINESE_ALIASES) + @pytest.mark.parametrize("cn_key,expected", CN_ALIASES) def test_chinese_aliases(self, cn_key: str, expected: str) -> None: assert normalize_emotion(cn_key) == expected class TestNormalizeEmotionEdgeCases: - """空串、空白、未知值等边界。""" + """空串、空白、None、未知值等边界。""" @pytest.mark.parametrize("empty_val", ["", None]) - def test_empty_returns_empty(self, empty_val) -> None: + def test_empty_or_none_returns_empty(self, empty_val) -> None: + """空串/None 返回空串——调用方据此不传 instruction(CosyVoice 走默认自然情绪)。""" assert normalize_emotion(empty_val) == "" @pytest.mark.parametrize("ws", [" ", "\t", "\n", " \n "]) def test_whitespace_only_returns_empty(self, ws: str) -> None: assert normalize_emotion(ws) == "" - def test_unknown_value_returns_empty_and_warns(self, caplog) -> None: + def test_unknown_value_defaults_to_neutral_with_warning(self, caplog) -> None: + """未知值:不能失败,默认返回 neutral,并打 warning 日志。""" caplog.set_level(logging.WARNING) - assert normalize_emotion("not_a_real_emotion") == "" + result = normalize_emotion("not_a_real_emotion_xyz") + assert result == "neutral" assert any("未知的 emotion" in r.message for r in caplog.records) def test_strips_leading_trailing_whitespace(self) -> None: - assert normalize_emotion(" happy ") == "开心愉快" + assert normalize_emotion(" happy ") == "happy" + assert normalize_emotion(" 生气 ") == "angry" class TestEmotionInstructionFormat: - """7 种 emotion 生成的 instruction 必须符合 CosyVoice v3 格式要求。""" + """instruction 严格符合 CosyVoice v3 官方要求: + 格式:"你说话的情感是。" + - 必须以"你说话的情感是"开头 + - 必须以中文句号"。"结尾 + - 情感值必须是 7 种英文枚举之一 + - 只允许一个中文句号(结尾),防止多句注入 + """ - @pytest.mark.parametrize("enum_key,expected_desc", SEVEN_STANDARD_ENUMS) - def test_instruction_starts_with_prefix(self, enum_key: str, expected_desc: str) -> None: - norm = normalize_emotion(enum_key) - instruction = f"你说话的情感是{norm}。" - assert instruction.startswith("你说话的情感是") - assert instruction.endswith("。") + PREFIX = "你说话的情感是" + SUFFIX = "。" - @pytest.mark.parametrize("enum_key,expected_desc", SEVEN_STANDARD_ENUMS) - def test_instruction_single_period(self, enum_key: str, expected_desc: str) -> None: - """instruction 只能有一个中文句号(结尾),防止误注入多个句子。""" - norm = normalize_emotion(enum_key) - instruction = f"你说话的情感是{norm}。" - assert instruction.count("。") == 1 + def _build_instruction(self, emotion: str) -> str: + return f"{self.PREFIX}{normalize_emotion(emotion)}{self.SUFFIX}" - @pytest.mark.parametrize("enum_key,expected_desc", SEVEN_STANDARD_ENUMS) - def test_instruction_contains_expected_description(self, enum_key: str, expected_desc: str) -> None: - norm = normalize_emotion(enum_key) - instruction = f"你说话的情感是{norm}。" - assert expected_desc in instruction + @pytest.mark.parametrize("enum_val", SEVEN_STANDARD_ENUMS) + def test_instruction_starts_with_prefix(self, enum_val: str) -> None: + assert self._build_instruction(enum_val).startswith(self.PREFIX) + + @pytest.mark.parametrize("enum_val", SEVEN_STANDARD_ENUMS) + def test_instruction_ends_with_single_cn_period(self, enum_val: str) -> None: + inst = self._build_instruction(enum_val) + assert inst.endswith(self.SUFFIX) + assert inst.count("。") == 1, f"instruction 只能有一个中文句号,实际: {inst!r}" + + @pytest.mark.parametrize("enum_val", SEVEN_STANDARD_ENUMS) + def test_instruction_contains_english_enum(self, enum_val: str) -> None: + """instruction 中情感值必须是英文枚举,不能是中文描述词。""" + inst = self._build_instruction(enum_val) + # 取出情感值部分(去掉前缀和句号) + emotion_part = inst[len(self.PREFIX) : -len(self.SUFFIX)] + assert emotion_part == enum_val + # 必须是纯 ASCII 英文(英文枚举值) + assert emotion_part.isascii() + + @pytest.mark.parametrize("cn,expected", FRONTEND_CN_LABELS) + def test_cn_label_produces_correct_instruction(self, cn: str, expected: str) -> None: + """中文标签拼出的 instruction 情感值必须是英文枚举。""" + inst = self._build_instruction(cn) + assert inst == f"{self.PREFIX}{expected}{self.SUFFIX}" + + @pytest.mark.parametrize( + "example", + [ + "neutral", + "happy", + "sad", + "angry", + "中立", + "开心", + "难过", + "生气", + ], + ) + def test_example_matches_official_doc_format(self, example: str) -> None: + """对照官方示例 "你说话的情感是neutral。" 格式。""" + inst = self._build_instruction(example) + # 官方格式示例:你说话的情感是neutral。 + assert inst.startswith(self.PREFIX) + assert inst.endswith(self.SUFFIX) + # 中间必须是英文枚举 + mid = inst[len(self.PREFIX) : -len(self.SUFFIX)] + assert mid in SEVEN_STANDARD_ENUMS def test_empty_emotion_produces_no_instruction(self) -> None: - """空 emotion 不应拼 instruction(调用方据此跳过字段)。""" + """空 emotion 不应拼 instruction(调用方据此跳过字段,CosyVoice 走默认)。""" assert normalize_emotion("") == "" assert normalize_emotion(" ") == "" + assert normalize_emotion(None) == "" + + def test_unknown_emotion_still_produces_valid_instruction(self) -> None: + """未知 emotion 默认 neutral,仍能产生合法 instruction,不会导致合成失败。""" + inst = self._build_instruction("unknown_xyz") + assert inst == f"{self.PREFIX}neutral{self.SUFFIX}"