59fbbd8e26
CI/CD Pipeline / Dedup Check - skip PR tests when covered by push pipeline (push) Successful in 1s
CI/CD Pipeline / Check push changed paths (push) Successful in 12s
CI/CD Pipeline / Build Staging API Image (push) Successful in 25s
CI/CD Pipeline / Build Staging Worker Image (push) Successful in 33s
CI/CD Pipeline / Build Staging Web Image (push) Successful in 1m54s
CI/CD Pipeline / Frontend Unit Tests (push) Failing after 2m40s
CI/CD Pipeline / Deploy Staging (Watchtower auto-deploy) (push) Successful in 2m22s
CI/CD Pipeline / Integration Tests (push) Successful in 4m52s
CI/CD Pipeline / Validate - Python (mypy + alembic) (push) Successful in 5m47s
CI/CD Pipeline / Validate - Style (push) Successful in 5m48s
CI/CD Pipeline / ACR Image Cleanup (push) Successful in 2m2s
CI/CD Pipeline / Staging API Integration Tests (push) Successful in 3m24s
CI/CD Pipeline / Staging E2E Tests (push) Failing after 3m32s
CI/CD Pipeline / Unit Tests (push) Successful in 10m38s
CI/CD Pipeline / Validate - Security (push) Successful in 15m27s
CI/CD Pipeline / Production Browser E2E (push) Has been skipped
CI/CD Pipeline / Build Production Web Image (push) Failing after 23h7m13s
CI/CD Pipeline / Retag skipped Staging Web Image (push) Failing after 23h20m31s
CI/CD Pipeline / PR Build Worker Image (push) Failing after 23h22m4s
CI/CD Pipeline / PR Build Web Image (push) Failing after 23h22m4s
CI/CD Pipeline / PR Build API Image (push) Failing after 23h22m4s
CI/CD Pipeline / Deploy Production (push) Failing after 23h6m24s
CI/CD Pipeline / Build Production Worker Image (push) Failing after 23h6m33s
CI/CD Pipeline / Build Production API Image (push) Failing after 23h6m33s
CI/CD Pipeline / CI Gate (push) Failing after 23h6m33s
CI/CD Pipeline / Retag skipped Staging API Image (push) Failing after 23h19m54s
CI/CD Pipeline / Frontend Lint (push) Failing after 23h22m3s
CI/CD Pipeline / Check if frontend-only change (push) Failing after 23h22m5s
CI/CD Pipeline / Canary Release to Production (push) Failing after 23h6m24s
CI/CD Pipeline / Retag skipped Staging Worker Image (push) Failing after 23h19m51s
fix(tts): P1 TTS emotion instruction format per voice type + missing CN aliases
- Add CN aliases 中性→neutral/伤心→sad/吃惊→surprised per 灵应 spec
- build_emotion_instruction(voice_id, enum): cloned voices use English "Speak in a {emotion} tone.", system voices with emotion Instruct (longanyang/longanhuan/longhuhu_v3) use strict Chinese format "你说话的情感是{emotion}。", default voice longxiaochun_v3 (no Instruct support) skips instruction entirely
- Expand preview emotion whitelist; 210 emotion tests passing, 15438 total unit tests passing
- E2E templates fix already in PR#1931 (0e5127df), no additional change needed
244 lines
8.7 KiB
Python
244 lines
8.7 KiB
Python
"""CosyVoice EMOTION_MAP / normalize_emotion / build_emotion_instruction 单测(P1 修复 #1898).
|
||
|
||
覆盖:
|
||
- 7 种标准英文枚举 neutral/happy/sad/angry/surprised/fearful/disgusted
|
||
(CosyVoice v3 官方 emotion 值)
|
||
- 大小写不敏感
|
||
- 前端中文 7 标签(中立/开心/难过/生气/惊讶/恐惧/厌恶)→ 英文枚举
|
||
- 灵应指定别名(中性/伤心/愤怒/吃惊)→ 英文枚举
|
||
- 常见中文别名与旧英文 4 枚举兼容
|
||
- 空串/空白/None 边界
|
||
- 未知值默认 neutral(warning 日志)
|
||
- build_emotion_instruction 三路分支:
|
||
· 克隆/设计音色 → "Speak in a {emotion} tone."
|
||
· 支持 emotion Instruct 的系统音色(longanyang/longanhuan/longhuhu_v3)→ "你说话的情感是{emotion}。"
|
||
· 默认系统音色(含 longxiaochun_v3)→ 返回空串(不传 instruction)
|
||
"""
|
||
|
||
from __future__ import annotations
|
||
|
||
import logging
|
||
|
||
import pytest
|
||
|
||
from packages.application.cosyvoice_service import (
|
||
EMOTION_MAP,
|
||
build_emotion_instruction,
|
||
normalize_emotion,
|
||
)
|
||
|
||
# 7 种官方英文枚举
|
||
SEVEN_STANDARD_ENUMS = [
|
||
"neutral",
|
||
"happy",
|
||
"sad",
|
||
"angry",
|
||
"surprised",
|
||
"fearful",
|
||
"disgusted",
|
||
]
|
||
|
||
# 前端中文 7 标签 → 期望英文枚举
|
||
FRONTEND_CN_LABELS = [
|
||
("中立", "neutral"),
|
||
("开心", "happy"),
|
||
("难过", "sad"),
|
||
("生气", "angry"),
|
||
("惊讶", "surprised"),
|
||
("恐惧", "fearful"),
|
||
("厌恶", "disgusted"),
|
||
]
|
||
|
||
# 灵应派任务补充的中文别名
|
||
LINGYING_CN_ALIASES = [
|
||
("中性", "neutral"),
|
||
("伤心", "sad"),
|
||
("愤怒", "angry"),
|
||
("吃惊", "surprised"),
|
||
]
|
||
|
||
# 其他常见中文别名
|
||
CN_ALIASES = [
|
||
("自然", "neutral"),
|
||
("愉快", "happy"),
|
||
("高兴", "happy"),
|
||
("快乐", "happy"),
|
||
("兴奋", "happy"),
|
||
("悲伤", "sad"),
|
||
("惊奇", "surprised"),
|
||
("害怕", "fearful"),
|
||
("讨厌", "disgusted"),
|
||
]
|
||
|
||
# 旧英文 4 枚举 → 最接近的标准枚举
|
||
OLD_FOUR_ENUMS = [
|
||
("natural", "neutral"),
|
||
("excited", "happy"),
|
||
("calm", "neutral"),
|
||
("friendly", "happy"),
|
||
]
|
||
|
||
# 支持 emotion Instruct 的系统音色(白名单)
|
||
SYSTEM_EMOTION_VOICES = ["longanyang", "longanhuan", "longhuhu_v3"]
|
||
|
||
# 不支持 Instruct 的典型系统音色(含默认音色 longxiaochun_v3)
|
||
NON_INSTRUCT_SYSTEM_VOICES = [
|
||
"longxiaochun_v3",
|
||
"longxiaoxia_v3",
|
||
"longsanshu_v3",
|
||
"longyue_v3",
|
||
"longyingjing_v3",
|
||
"loongabby_v3",
|
||
"loongandy_v3",
|
||
"longfei_v3",
|
||
]
|
||
|
||
# 克隆/设计音色(非 long/loong 前缀)
|
||
CLONED_VOICE_IDS = [
|
||
"myclone_abc123",
|
||
"xiaoming_20260915",
|
||
"clone_voice_42",
|
||
"custom_voice_test",
|
||
]
|
||
|
||
|
||
class TestEmotionMapSevenStandard:
|
||
@pytest.mark.parametrize("enum_val", SEVEN_STANDARD_ENUMS)
|
||
def test_standard_enum_maps_to_self(self, enum_val: str) -> None:
|
||
assert enum_val in EMOTION_MAP
|
||
assert EMOTION_MAP[enum_val] == enum_val
|
||
|
||
@pytest.mark.parametrize("enum_val", SEVEN_STANDARD_ENUMS)
|
||
def test_normalize_standard_enum(self, enum_val: str) -> None:
|
||
assert normalize_emotion(enum_val) == enum_val
|
||
|
||
@pytest.mark.parametrize("enum_val", SEVEN_STANDARD_ENUMS)
|
||
def test_normalize_case_insensitive(self, enum_val: str) -> None:
|
||
assert normalize_emotion(enum_val.upper()) == enum_val
|
||
assert normalize_emotion(enum_val.capitalize()) == enum_val
|
||
assert normalize_emotion(f" {enum_val} ") == enum_val
|
||
|
||
def test_neutral_maps_to_neutral(self) -> None:
|
||
assert normalize_emotion("neutral") == "neutral"
|
||
|
||
|
||
class TestFrontendCnLabels:
|
||
@pytest.mark.parametrize("cn,expected", FRONTEND_CN_LABELS)
|
||
def test_cn_label_normalizes_to_enum(self, cn: str, expected: str) -> None:
|
||
assert normalize_emotion(cn) == expected
|
||
|
||
|
||
class TestLingyingSpecAliases:
|
||
@pytest.mark.parametrize("cn,expected", LINGYING_CN_ALIASES)
|
||
def test_lingying_aliases(self, cn: str, expected: str) -> None:
|
||
assert normalize_emotion(cn) == expected
|
||
|
||
|
||
class TestBackwardCompatAliases:
|
||
@pytest.mark.parametrize("old_key,expected", OLD_FOUR_ENUMS)
|
||
def test_old_four_enums(self, old_key: str, expected: str) -> None:
|
||
assert normalize_emotion(old_key) == expected
|
||
|
||
@pytest.mark.parametrize("cn_key,expected", CN_ALIASES)
|
||
def test_chinese_aliases(self, cn_key: str, expected: str) -> None:
|
||
assert normalize_emotion(cn_key) == expected
|
||
|
||
|
||
class TestNormalizeEmotionEdgeCases:
|
||
@pytest.mark.parametrize("empty_val", ["", None])
|
||
def test_empty_or_none_returns_empty(self, empty_val) -> None:
|
||
assert normalize_emotion(empty_val) == ""
|
||
|
||
@pytest.mark.parametrize("ws", [" ", "\t", "\n", " \n "])
|
||
def test_whitespace_only_returns_empty(self, ws: str) -> None:
|
||
assert normalize_emotion(ws) == ""
|
||
|
||
def test_unknown_value_defaults_to_neutral_with_warning(self, caplog) -> None:
|
||
caplog.set_level(logging.WARNING)
|
||
result = normalize_emotion("not_a_real_emotion_xyz")
|
||
assert result == "neutral"
|
||
assert any("未知的 emotion" in r.message for r in caplog.records)
|
||
|
||
def test_strips_leading_trailing_whitespace(self) -> None:
|
||
assert normalize_emotion(" happy ") == "happy"
|
||
assert normalize_emotion(" 生气 ") == "angry"
|
||
assert normalize_emotion(" 中性 ") == "neutral"
|
||
|
||
|
||
class TestBuildEmotionInstructionClonedVoice:
|
||
@pytest.mark.parametrize("voice_id", CLONED_VOICE_IDS)
|
||
@pytest.mark.parametrize("enum_val", SEVEN_STANDARD_ENUMS)
|
||
def test_cloned_voice_uses_english_tone_format(self, voice_id: str, enum_val: str) -> None:
|
||
inst = build_emotion_instruction(voice_id, enum_val)
|
||
assert inst == f"Speak in a {enum_val} tone."
|
||
assert inst.isascii(), f"克隆音色 instruction 必须是纯 ASCII 英文: {inst!r}"
|
||
|
||
@pytest.mark.parametrize(
|
||
"cn,expected",
|
||
FRONTEND_CN_LABELS + LINGYING_CN_ALIASES + CN_ALIASES,
|
||
)
|
||
def test_cloned_voice_chinese_input_english_output(self, cn: str, expected: str) -> None:
|
||
norm = normalize_emotion(cn)
|
||
inst = build_emotion_instruction("myclone_voice", norm)
|
||
assert inst == f"Speak in a {expected} tone."
|
||
assert inst.isascii()
|
||
|
||
def test_cloned_voice_empty_emotion_returns_empty(self) -> None:
|
||
assert build_emotion_instruction("myclone", "") == ""
|
||
|
||
@pytest.mark.parametrize("voice_id", CLONED_VOICE_IDS)
|
||
def test_cloned_voice_unknown_emotion_falls_back_neutral(self, voice_id: str) -> None:
|
||
norm = normalize_emotion("unknown_xyz")
|
||
assert norm == "neutral"
|
||
inst = build_emotion_instruction(voice_id, norm)
|
||
assert inst == "Speak in a neutral tone."
|
||
|
||
|
||
class TestBuildEmotionInstructionSystemVoiceEmotion:
|
||
CN_PREFIX = "你说话的情感是"
|
||
CN_SUFFIX = "。"
|
||
|
||
@pytest.mark.parametrize("voice_id", SYSTEM_EMOTION_VOICES)
|
||
@pytest.mark.parametrize("enum_val", SEVEN_STANDARD_ENUMS)
|
||
def test_system_emotion_voice_cn_fixed_format(self, voice_id: str, enum_val: str) -> None:
|
||
inst = build_emotion_instruction(voice_id, enum_val)
|
||
assert inst == f"{self.CN_PREFIX}{enum_val}{self.CN_SUFFIX}"
|
||
assert inst.count("。") == 1
|
||
mid = inst[len(self.CN_PREFIX) : -len(self.CN_SUFFIX)]
|
||
assert mid == enum_val
|
||
assert mid.isascii(), f"系统音色 emotion 值必须是纯 ASCII 英文枚举: {inst!r}"
|
||
|
||
@pytest.mark.parametrize("voice_id", SYSTEM_EMOTION_VOICES)
|
||
def test_system_emotion_voice_empty_returns_empty(self, voice_id: str) -> None:
|
||
assert build_emotion_instruction(voice_id, "") == ""
|
||
|
||
|
||
class TestBuildEmotionInstructionNonInstructSystemVoice:
|
||
@pytest.mark.parametrize("voice_id", NON_INSTRUCT_SYSTEM_VOICES)
|
||
@pytest.mark.parametrize("enum_val", SEVEN_STANDARD_ENUMS)
|
||
def test_non_instruct_voice_returns_empty(self, voice_id: str, enum_val: str) -> None:
|
||
assert build_emotion_instruction(voice_id, enum_val) == ""
|
||
|
||
def test_default_voice_longxiaochun_v3_no_instruction(self) -> None:
|
||
assert build_emotion_instruction("longxiaochun_v3", "happy") == ""
|
||
assert build_emotion_instruction("longxiaochun_v3", "neutral") == ""
|
||
|
||
|
||
class TestBuildEmotionInstructionEdgeCases:
|
||
def test_empty_voice_id_treated_as_system(self) -> None:
|
||
assert build_emotion_instruction("", "happy") == ""
|
||
|
||
@pytest.mark.parametrize(
|
||
"voice_id,enum_val,expected",
|
||
[
|
||
("MYCLONE_VOICE", "happy", "Speak in a happy tone."),
|
||
("CloneVoice", "sad", "Speak in a sad tone."),
|
||
],
|
||
)
|
||
def test_voice_id_case_handling(self, voice_id: str, enum_val: str, expected: str) -> None:
|
||
assert build_emotion_instruction(voice_id, enum_val) == expected
|
||
|
||
def test_loong_prefix_is_system_voice(self) -> None:
|
||
assert build_emotion_instruction("loongandy_v3", "happy") == ""
|
||
assert build_emotion_instruction("loongabby_v3", "angry") == ""
|