fix(#1898): TTS 情绪枚举统一为7种标准英文标签(P1) (#1932)
CI/CD Pipeline / Dedup Check - skip PR tests when covered by push pipeline (push) Successful in 1s
CI/CD Pipeline / Check push changed paths (push) Successful in 13s
CI/CD Pipeline / Build Staging Web Image (push) Successful in 26s
CI/CD Pipeline / Build Staging API Image (push) Successful in 30s
CI/CD Pipeline / Build Staging Worker Image (push) Successful in 31s
CI/CD Pipeline / Deploy Staging (Watchtower auto-deploy) (push) Successful in 35s
CI/CD Pipeline / Integration Tests (push) Successful in 2m47s
CI/CD Pipeline / Validate - Python (mypy + alembic) (push) Successful in 3m6s
CI/CD Pipeline / ACR Image Cleanup (push) Successful in 1m59s
CI/CD Pipeline / Staging E2E Tests (push) Failing after 2m1s
CI/CD Pipeline / Validate - Style (push) Successful in 3m44s
CI/CD Pipeline / Staging API Integration Tests (push) Successful in 3m3s
CI/CD Pipeline / Frontend Unit Tests (push) Failing after 5m15s
CI/CD Pipeline / Validate - Security (push) Successful in 6m32s
CI/CD Pipeline / Unit Tests (push) Successful in 8m16s
CI/CD Pipeline / Production Browser E2E (push) Has been skipped
CI/CD Pipeline / Deploy Production (push) Failing after 25h12m28s
CI/CD Pipeline / Build Production Web Image (push) Failing after 25h12m30s
CI/CD Pipeline / Retag skipped Staging Web Image (push) Failing after 25h20m3s
CI/CD Pipeline / PR Build API Image (push) Failing after 25h20m52s
CI/CD Pipeline / PR Build Worker Image (push) Failing after 25h20m13s
CI/CD Pipeline / PR Build Web Image (push) Failing after 25h20m13s
CI/CD Pipeline / Retag skipped Staging API Image (push) Failing after 25h19m24s
CI/CD Pipeline / CI Gate (push) Failing after 25h11m51s
CI/CD Pipeline / Canary Release to Production (push) Failing after 25h11m49s
CI/CD Pipeline / Build Production Worker Image (push) Failing after 25h11m51s
CI/CD Pipeline / Build Production API Image (push) Failing after 25h11m51s
CI/CD Pipeline / Retag skipped Staging Worker Image (push) Failing after 25h19m24s
CI/CD Pipeline / Frontend Lint (push) Failing after 25h20m8s
CI/CD Pipeline / Check if frontend-only change (push) Failing after 25h20m14s

Co-authored-by: xiaoxia <dev@xiaoxiajianji.com>
Co-committed-by: xiaoxia <dev@xiaoxiajianji.com>
This commit was merged in pull request #1932.
This commit is contained in:
2026-09-15 14:02:49 +08:00
committed by auto-approve-bot
parent 0e5127df05
commit 19eeb1b475
6 changed files with 202 additions and 22 deletions
+21 -3
View File
@@ -287,7 +287,22 @@ def retry_voice_clone(
return _to_response(profile)
_ALLOWED_PREVIEW_EMOTIONS = {"", "natural", "excited", "calm", "friendly"}
_ALLOWED_PREVIEW_EMOTIONS = {
"",
# 7 种标准英文枚举
"neutral",
"happy",
"sad",
"angry",
"surprised",
"fearful",
"disgusted",
# 旧英文 4 枚举兼容
"natural",
"excited",
"calm",
"friendly",
}
@router.get("/{clone_id}/preview", response_model=VoiceClonePreviewResponse)
@@ -295,7 +310,10 @@ def get_voice_clone_preview(
clone_id: str,
text: str = Query("", description="自定义试听文本,为空则使用默认示例"),
speed: float = Query(1.0, ge=0.5, le=2.0, description="语速,0.5-2.0,默认 1.0"),
emotion: str = Query("", description="情绪:natural/excited/calm/friendly,空字符串为默认自然"),
emotion: str = Query(
"",
description="情绪:neutral/happy/sad/angry/surprised/fearful/disgusted,兼容旧值 natural/excited/calm/friendly,空为默认自然",
),
authenticated_user: AuthenticatedUser = Depends(get_current_user),
repository: SQLAlchemyVoiceCloneProfileRepository = Depends(get_voice_clone_profile_repository),
cosyvoice: CosyVoiceService = Depends(get_cosyvoice_service),
@@ -311,7 +329,7 @@ def get_voice_clone_preview(
if emotion not in _ALLOWED_PREVIEW_EMOTIONS:
raise HTTPException(
status_code=status.HTTP_400_BAD_REQUEST,
detail=f"不支持的 emotion 值: {emotion},可选: natural/excited/calm/friendly 或留空",
detail=f"不支持的 emotion 值: {emotion},可选: neutral/happy/sad/angry/surprised/fearful/disgusted(兼容 natural/excited/calm/friendly)或留空",
)
use_case = GetVoiceCloneUseCase(repository)
+8 -2
View File
@@ -67,7 +67,9 @@ class CreateLipsyncJobRequest(BaseModel):
voice_id: str = Field("", description="音色 ID(预置音色或克隆音色 profile UUID)")
script_text: str = Field("", description="要合成的文案(直生模式必填,最长 5000 字符)")
speed: float = Field(1.0, ge=0.5, le=2.0, description="语速(0.5-2.0),默认 1.0")
emotion: str = Field("", description="情绪(中文/英文:自然/兴奋/沉稳/亲切/开心/悲伤/愤怒/惊讶/恐惧/厌恶 等)")
emotion: str = Field(
"", description="情绪(英文枚举 neutral/happy/sad/angry/surprised/fearful/disgusted,兼容中文/旧值;空为默认)"
)
enable_video_loop: bool = Field(
True, description="音频长于视频时是否循环画面(AI数字人默认开启,防止音频长于视频被截断)"
@@ -120,7 +122,11 @@ class AiAvatarTtsPreviewRequest(BaseModel):
voice_id: str = Field(..., min_length=1, max_length=128, description="音色 ID")
script_text: str = Field(..., min_length=1, max_length=5000, description="要合成的文案")
speed: float = Field(1.0, ge=0.5, le=2.0, description="语速(0.5-2.0),默认 1.0")
emotion: str = Field("natural", max_length=32, description="情绪")
emotion: str = Field(
"neutral",
max_length=32,
description="情绪(英文枚举 neutral/happy/sad/angry/surprised/fearful/disgusted,或中文/旧值)",
)
class AiAvatarTtsPreviewResponse(BaseModel):
+1 -1
View File
@@ -360,7 +360,7 @@ class LipsyncService:
voice_id: str,
script_text: str,
speed: float = 1.0,
emotion: str = "natural",
emotion: str = "neutral",
) -> dict:
"""同步做 TTS 合成 + 下载 + ffprobe + 句子时间戳计算.
+26 -15
View File
@@ -29,35 +29,40 @@ logger = logging.getLogger(__name__)
# CosyVoice v3 情绪通过 input.instruction 中文自然语言指令控制(不再使用枚举 emotion 字段)。
# 前端可传中文或英文情绪标签,统一归一化为中文描述词,再拼进 instruction。
# 映射表 key: 小写中文/英文 → 中文情绪描述词
EMOTION_MAP = {
# 原有四值(中文 + 英文)
"自然": "自然",
"兴奋": "兴奋开心",
"沉稳": "沉稳平静",
"亲切": "亲切友好",
# 第一层:7 种标准英文枚举(与前端 emotion enum 对齐,必须存在且作为第一组)
# 第二层:旧英文 4 枚举(natural/excited/calm/friendly),保持向后兼容
# 第三层:中文标签别名
EMOTION_MAP: dict[str, str] = {
# 7 种标准英文枚举(P1 修复:以英文枚举为标准 key,neutral 为默认)
"neutral": "自然",
"happy": "开心愉快",
"sad": "悲伤难过",
"angry": "愤怒",
"surprised": "惊讶",
"fearful": "恐惧",
"disgusted": "厌恶",
# 旧英文 4 枚举兼容(自然映射到对应中文)
"natural": "自然",
"excited": "兴奋开心",
"calm": "沉稳平静",
"friendly": "亲切友好",
"happy": "开心愉快",
# 新增情绪
# 中文标签
"自然": "自然",
"开心": "开心愉快",
"愉快": "开心愉快",
"sad": "悲伤难过",
"兴奋": "兴奋开心",
"悲伤": "悲伤难过",
"难过": "悲伤难过",
"angry": "愤怒",
"愤怒": "愤怒",
"生气": "愤怒",
"surprised": "惊讶",
"惊讶": "惊讶",
"惊奇": "惊讶",
"fearful": "恐惧",
"恐惧": "恐惧",
"害怕": "恐惧",
"disgusted": "厌恶",
"厌恶": "厌恶",
"讨厌": "厌恶",
"沉稳": "沉稳平静",
"亲切": "亲切友好",
"严肃": "严肃",
"温柔": "温柔",
}
@@ -66,11 +71,16 @@ EMOTION_MAP = {
def normalize_emotion(emotion: str) -> str:
"""将前端情绪值归一化为中文描述词,用于拼入 instruction.
支持中文/英文;空串或未知值返回空串(调用方据此决定是否传 instruction)。
支持 7 种标准英文枚举(neutral/happy/sad/angry/surprised/fearful/disgusted)、
旧英文兼容值(natural/excited/calm/friendly)以及中文标签;大小写不敏感。
空串/空白/未知值返回空串(调用方据此决定是否传 instruction)。
"""
if not emotion:
return ""
key = emotion.strip()
if not key:
return ""
# 大小写不敏感:先按原 key 查,再按 lower 查
mapped = EMOTION_MAP.get(key) or EMOTION_MAP.get(key.lower())
if mapped:
return mapped
@@ -516,7 +526,8 @@ class CosyVoiceService:
format: 输出格式(mp3/wav/pcm),空表示使用配置默认值
speed: 语速(0.5-2.0),1.0 为正常速度
volume: 音量(0-100),默认 50
emotion: 情绪(自然/兴奋/沉稳/亲切/开心/悲伤/愤怒/惊讶/恐惧/厌恶 等),空串不传
emotion: 情绪,英文枚举 neutral/happy/sad/angry/surprised/fearful/disgusted,
兼容旧值 natural/excited/calm/friendly 及中文标签;空串不传
language: 语言代码(zh/en 等,默认 zh;系统音色仅 zh/en 传 language_hints)
Returns:
+145
View File
@@ -0,0 +1,145 @@
"""CosyVoice EMOTION_MAP 与 normalize_emotion 单测(P1 修复 #1898).
覆盖:
- 7 种标准英文枚举 neutral/happy/sad/angry/surprised/fearful/disgusted
- 大小写不敏感
- 中文标签别名
- 旧英文 4 枚举兼容(natural/excited/calm/friendly)
- 空串/空白/未知值边界
- instruction 拼接格式("你说话的情感是{emotion}。" 单句号)
"""
from __future__ import annotations
import logging
import pytest
from packages.application.cosyvoice_service import (
EMOTION_MAP,
normalize_emotion,
)
SEVEN_STANDARD_ENUMS = [
("neutral", "自然"),
("happy", "开心愉快"),
("sad", "悲伤难过"),
("angry", "愤怒"),
("surprised", "惊讶"),
("fearful", "恐惧"),
("disgusted", "厌恶"),
]
OLD_FOUR_ENUMS = [
("natural", "自然"),
("excited", "兴奋开心"),
("calm", "沉稳平静"),
("friendly", "亲切友好"),
]
CHINESE_ALIASES = [
("自然", "自然"),
("开心", "开心愉快"),
("愉快", "开心愉快"),
("兴奋", "兴奋开心"),
("悲伤", "悲伤难过"),
("难过", "悲伤难过"),
("愤怒", "愤怒"),
("生气", "愤怒"),
("惊讶", "惊讶"),
("惊奇", "惊讶"),
("恐惧", "恐惧"),
("害怕", "恐惧"),
("厌恶", "厌恶"),
("讨厌", "厌恶"),
("沉稳", "沉稳平静"),
("亲切", "亲切友好"),
("严肃", "严肃"),
("温柔", "温柔"),
]
class TestEmotionMapSevenStandard:
"""7 种标准英文枚举必须作为 key 存在并映射到正确中文描述词。"""
@pytest.mark.parametrize("enum_key,expected", SEVEN_STANDARD_ENUMS)
def test_standard_enum_present(self, enum_key: str, expected: str) -> None:
assert enum_key in EMOTION_MAP
assert EMOTION_MAP[enum_key] == expected
@pytest.mark.parametrize("enum_key,expected", SEVEN_STANDARD_ENUMS)
def test_normalize_standard_enum(self, enum_key: str, expected: str) -> None:
assert normalize_emotion(enum_key) == expected
@pytest.mark.parametrize("enum_key,expected", SEVEN_STANDARD_ENUMS)
def test_normalize_case_insensitive(self, enum_key: str, expected: str) -> None:
"""大小写不敏感:Neutral/HAPPY/Angry 等都能匹配。"""
assert normalize_emotion(enum_key.upper()) == expected
assert normalize_emotion(enum_key.capitalize()) == expected
assert normalize_emotion(f" {enum_key} ") == expected
def test_neutral_is_first_standard_key(self) -> None:
"""neutral 必须是标准枚举第一 key(作为默认值语义)。"""
en_keys = [k for k in EMOTION_MAP if all(ord(c) < 128 for c in k)]
assert en_keys[0] == "neutral"
class TestEmotionMapBackwardCompat:
"""旧英文 4 枚举与中文标签必须继续兼容。"""
@pytest.mark.parametrize("old_key,expected", OLD_FOUR_ENUMS)
def test_old_four_enums(self, old_key: str, expected: str) -> None:
assert normalize_emotion(old_key) == expected
@pytest.mark.parametrize("cn_key,expected", CHINESE_ALIASES)
def test_chinese_aliases(self, cn_key: str, expected: str) -> None:
assert normalize_emotion(cn_key) == expected
class TestNormalizeEmotionEdgeCases:
"""空串、空白、未知值等边界。"""
@pytest.mark.parametrize("empty_val", ["", None])
def test_empty_returns_empty(self, empty_val) -> None:
assert normalize_emotion(empty_val) == ""
@pytest.mark.parametrize("ws", [" ", "\t", "\n", " \n "])
def test_whitespace_only_returns_empty(self, ws: str) -> None:
assert normalize_emotion(ws) == ""
def test_unknown_value_returns_empty_and_warns(self, caplog) -> None:
caplog.set_level(logging.WARNING)
assert normalize_emotion("not_a_real_emotion") == ""
assert any("未知的 emotion" in r.message for r in caplog.records)
def test_strips_leading_trailing_whitespace(self) -> None:
assert normalize_emotion(" happy ") == "开心愉快"
class TestEmotionInstructionFormat:
"""7 种 emotion 生成的 instruction 必须符合 CosyVoice v3 格式要求。"""
@pytest.mark.parametrize("enum_key,expected_desc", SEVEN_STANDARD_ENUMS)
def test_instruction_starts_with_prefix(self, enum_key: str, expected_desc: str) -> None:
norm = normalize_emotion(enum_key)
instruction = f"你说话的情感是{norm}。"
assert instruction.startswith("你说话的情感是")
assert instruction.endswith("。")
@pytest.mark.parametrize("enum_key,expected_desc", SEVEN_STANDARD_ENUMS)
def test_instruction_single_period(self, enum_key: str, expected_desc: str) -> None:
"""instruction 只能有一个中文句号(结尾),防止误注入多个句子。"""
norm = normalize_emotion(enum_key)
instruction = f"你说话的情感是{norm}。"
assert instruction.count("。") == 1
@pytest.mark.parametrize("enum_key,expected_desc", SEVEN_STANDARD_ENUMS)
def test_instruction_contains_expected_description(self, enum_key: str, expected_desc: str) -> None:
norm = normalize_emotion(enum_key)
instruction = f"你说话的情感是{norm}。"
assert expected_desc in instruction
def test_empty_emotion_produces_no_instruction(self) -> None:
"""空 emotion 不应拼 instruction(调用方据此跳过字段)。"""
assert normalize_emotion("") == ""
assert normalize_emotion(" ") == ""
+1 -1
View File
@@ -196,7 +196,7 @@ class TestVoiceClonePreview:
cosyvoice = MagicMock()
with pytest.raises(HTTPException) as exc_info:
self._call_preview(profile, cosyvoice, emotion="angry")
self._call_preview(profile, cosyvoice, emotion="invalid_emotion_xyz")
assert exc_info.value.status_code == 400
cosyvoice.synthesize_speech.assert_not_called()