feat(#1898): TTS 参数适配 CosyVoice v3 官方 API (#1917)
CI/CD Pipeline / Dedup Check - skip PR tests when covered by push pipeline (push) Successful in 2s
CI/CD Pipeline / Check push changed paths (push) Successful in 5s
CI/CD Pipeline / Build Staging Web Image (push) Successful in 31s
CI/CD Pipeline / Build Staging API Image (push) Successful in 33s
CI/CD Pipeline / Build Staging Worker Image (push) Successful in 33s
CI/CD Pipeline / Deploy Staging (Watchtower auto-deploy) (push) Successful in 44s
CI/CD Pipeline / ACR Image Cleanup (push) Successful in 1m55s
CI/CD Pipeline / Frontend Unit Tests (push) Successful in 3m27s
CI/CD Pipeline / Validate - Python (mypy + alembic) (push) Successful in 3m57s
CI/CD Pipeline / Integration Tests (push) Successful in 3m59s
CI/CD Pipeline / Validate - Style (push) Successful in 4m4s
CI/CD Pipeline / Staging E2E Tests (push) Failing after 3m45s
CI/CD Pipeline / Staging API Integration Tests (push) Successful in 4m46s
CI/CD Pipeline / Validate - Security (push) Successful in 8m21s
CI/CD Pipeline / Unit Tests (push) Successful in 8m51s
CI/CD Pipeline / Production Browser E2E (push) Has been skipped
CI/CD Pipeline / Deploy Production (push) Failing after 34h54m41s
CI/CD Pipeline / Retag skipped Staging Worker Image (push) Failing after 35h2m58s
CI/CD Pipeline / PR Build Web Image (push) Failing after 35h3m44s
CI/CD Pipeline / Build Production API Image (push) Failing after 34h54m14s
CI/CD Pipeline / Retag skipped Staging API Image (push) Failing after 35h2m25s
CI/CD Pipeline / PR Build Worker Image (push) Failing after 35h3m11s
CI/CD Pipeline / PR Build API Image (push) Failing after 35h3m11s
CI/CD Pipeline / Build Production Worker Image (push) Failing after 34h54m14s
CI/CD Pipeline / Build Production Web Image (push) Failing after 34h54m14s
CI/CD Pipeline / CI Gate (push) Failing after 34h54m14s
CI/CD Pipeline / Canary Release to Production (push) Failing after 34h54m8s
CI/CD Pipeline / Retag skipped Staging Web Image (push) Failing after 35h2m25s
CI/CD Pipeline / Frontend Lint (push) Failing after 35h3m7s
CI/CD Pipeline / Check if frontend-only change (push) Failing after 35h3m12s

Co-authored-by: xiaoxia <dev@xiaoxiajianji.com>
Co-committed-by: xiaoxia <dev@xiaoxiajianji.com>
This commit was merged in pull request #1917.
This commit is contained in:
2026-09-15 04:19:58 +08:00
committed by auto-approve-bot
parent 695a491c5d
commit 201a3f0af5
12 changed files with 156 additions and 48 deletions
Executable → Regular
+12 -3
View File
@@ -177,6 +177,7 @@ def synthesize(
synthesis_meta = {
"speed": request.speed,
"emotion": request.emotion or "",
"language": request.language or "zh-CN",
}
if request.metadata_:
synthesis_meta.update(request.metadata_)
@@ -462,10 +463,17 @@ def save_tts_job_to_library(
try:
proc = subprocess.run(
[
"ffprobe", "-v", "quiet", "-print_format", "json",
"-show_format", str(tmp_path),
"ffprobe",
"-v",
"quiet",
"-print_format",
"json",
"-show_format",
str(tmp_path),
],
capture_output=True, text=True, timeout=10,
capture_output=True,
text=True,
timeout=10,
)
if proc.returncode == 0:
fmt = json.loads(proc.stdout).get("format", {})
@@ -576,6 +584,7 @@ def preview_tts(
voice_id=actual_voice_id,
speed=request.speed,
emotion=request.emotion,
language=getattr(request, "language", "zh-CN"),
)
except CosyVoiceError as e:
raise HTTPException(
+1 -1
View File
@@ -67,7 +67,7 @@ class CreateLipsyncJobRequest(BaseModel):
voice_id: str = Field("", description="音色 ID(预置音色或克隆音色 profile UUID)")
script_text: str = Field("", description="要合成的文案(直生模式必填,最长 5000 字符)")
speed: float = Field(1.0, ge=0.5, le=2.0, description="语速(0.5-2.0),默认 1.0")
emotion: str = Field("", description="情绪(natural/excited/calm/friendly 或中文 自然/兴奋/沉稳/亲切)")
emotion: str = Field("", description="情绪(中文/英文:自然/兴奋/沉稳/亲切/开心/悲伤/愤怒/惊讶/恐惧/厌恶 等)")
enable_video_loop: bool = Field(
True, description="音频长于视频时是否循环画面(AI数字人默认开启,防止音频长于视频被截断)"
+6 -2
View File
@@ -16,7 +16,10 @@ class TTSSynthesizeRequest(BaseModel):
output_name: str = Field("", description="输出文件名")
language: str = Field("zh-CN", description="语言")
speed: float = Field(1.0, ge=0.5, le=2.0, description="语速")
emotion: str = Field("", description="情绪(natural/excited/calm/friendly,或中文 自然/兴奋/沉稳/亲切)")
emotion: str = Field(
"",
description="情绪(中文/英文:自然/兴奋/沉稳/亲切/开心/悲伤/愤怒/惊讶/恐惧/厌恶 等;通过 instruction 自然语言指令控制)",
)
voice_model: str = Field("", description="语音模型名称")
voice_clone_profile_id: str = Field("", description="关联的音色克隆档案 ID")
format: str = Field("mp3", description="输出格式(mp3/wav/pcm)")
@@ -110,7 +113,8 @@ class TTSPreviewRequest(BaseModel):
text: str = Field(..., min_length=1, max_length=200, description="合成文本,限制 200 字")
voice_id: str = Field(..., min_length=1, description="音色 ID")
speed: float = Field(1.0, ge=0.5, le=2.0, description="语速")
emotion: str = Field("", description="情绪(natural/excited/calm/friendly,或中文)")
emotion: str = Field("", description="情绪(中文/英文:自然/兴奋/沉稳/亲切/开心/悲伤/愤怒/惊讶/恐惧/厌恶 等)")
language: str = Field("zh-CN", description="语言(zh-CN/en-US 等)")
pitch: float = Field(1.0, ge=0.5, le=2.0, description="音调(预留,当前未使用)")
+7 -5
View File
@@ -35,7 +35,7 @@ from app.tasks.lipsync_tts import tts_synthesize_and_submit
from sqlalchemy.orm import Session
from packages.adapters.sqlalchemy_impl.models import LipsyncJobModel
from packages.application.cosyvoice_service import CosyVoiceError, normalize_emotion
from packages.application.cosyvoice_service import CosyVoiceError
from packages.domain.sentence_timings import (
compute_sentence_timings,
probe_audio_duration,
@@ -121,7 +121,8 @@ class LipsyncService:
text=script_text,
voice_id=actual_voice_id,
speed=speed,
emotion=normalize_emotion(emotion),
emotion=emotion, # normalize 在 CosyVoiceService 内部完成
language="zh",
)
except CosyVoiceError as exc:
raise MediaKitError(f"TTS 合成失败: {exc}", code="TTSSynthesisFailed") from exc
@@ -304,7 +305,7 @@ class LipsyncService:
voice_id=voice_id or "",
script_text=script_text or "",
speed=speed,
emotion=normalize_emotion(emotion) if is_tts_mode else (emotion or ""),
emotion=emotion or "",
# 音频直传(含预合成)直接进入 pending(后续同步改为 submitted);TTS 模式进入 tts_processing
status="tts_processing" if is_tts_mode else "pending",
)
@@ -325,7 +326,7 @@ class LipsyncService:
voice_id,
script_text,
speed,
normalize_emotion(emotion),
emotion or "",
)
)
except Exception as exc:
@@ -382,7 +383,8 @@ class LipsyncService:
text=script_text,
voice_id=actual_voice_id,
speed=speed,
emotion=normalize_emotion(emotion),
emotion=emotion, # normalize 在 CosyVoiceService 内部完成
language="zh",
)
except CosyVoiceError as exc:
raise MediaKitError(f"TTS 合成失败: {exc}", code="TTSSynthesisFailed") from exc
+1
View File
@@ -160,6 +160,7 @@ def tts_synthesize_and_submit(
voice_id=voice_id,
speed=speed,
emotion=emotion,
language="zh",
)
except CosyVoiceError as exc:
logger.error("[lipsync_tts] TTS 合成失败: job_id=%s err=%s", job_id, exc)
+70 -18
View File
@@ -25,35 +25,75 @@ from packages.shared.config import get_shared_settings
logger = logging.getLogger(__name__)
# CosyVoice 支持的情绪:中文标签 → API 英文值
# CosyVoice v3 情绪通过 input.instruction 中文自然语言指令控制(不再使用枚举 emotion 字段)。
# 前端可传中文或英文情绪标签,统一归一化为中文描述词,再拼进 instruction。
# 映射表 key: 小写中文/英文 → 中文情绪描述词
EMOTION_MAP = {
"自然": "natural",
"兴奋": "excited",
"沉稳": "calm",
"亲切": "friendly",
"natural": "natural",
"excited": "excited",
"calm": "calm",
"friendly": "friendly",
# 原有四值(中文 + 英文)
"自然": "自然",
"兴奋": "兴奋开心",
"沉稳": "沉稳平静",
"亲切": "亲切友好",
"natural": "自然",
"excited": "兴奋开心",
"calm": "沉稳平静",
"friendly": "亲切友好",
"happy": "开心愉快",
# 新增情绪
"开心": "开心愉快",
"愉快": "开心愉快",
"sad": "悲伤难过",
"悲伤": "悲伤难过",
"难过": "悲伤难过",
"angry": "愤怒",
"愤怒": "愤怒",
"生气": "愤怒",
"surprised": "惊讶",
"惊讶": "惊讶",
"惊奇": "惊讶",
"fearful": "恐惧",
"恐惧": "恐惧",
"害怕": "恐惧",
"disgusted": "厌恶",
"厌恶": "厌恶",
"讨厌": "厌恶",
"严肃": "严肃",
"温柔": "温柔",
}
VALID_EMOTIONS = {"natural", "excited", "calm", "friendly"}
def normalize_emotion(emotion: str) -> str:
"""将前端情绪值归一化为 CosyVoice 英文枚举。
"""将前端情绪值归一化为中文描述词,用于拼入 instruction.
支持中文(自然/兴奋/沉稳/亲切)和英文;非法值返回空串(不传,走默认)。
支持中文/英文;空串或未知值返回空串(调用方据此决定是否传 instruction)。
"""
if not emotion:
return ""
key = emotion.strip().lower()
mapped = EMOTION_MAP.get(emotion.strip()) or EMOTION_MAP.get(key)
if mapped and mapped in VALID_EMOTIONS:
key = emotion.strip()
mapped = EMOTION_MAP.get(key) or EMOTION_MAP.get(key.lower())
if mapped:
return mapped
logger.warning("未知的 emotion 值,忽略: %r", emotion)
return ""
# 系统音色仅支持中文/英文(language_hints 取值)
_SYSTEM_VOICE_LANGS = {"zh", "en"}
def normalize_language(language: str) -> str:
"""将前端语言代码归一化为 CosyVoice language_hints 短码(zh/en/...).
支持 zh-CN/zh_CN/zh/en-US/en/en_US/en-GB 等常见形式;
空值默认 zh;未知值返回短码(CosyVoice 接受时生效)。
"""
if not language:
return "zh"
# 取第一段(zh-CN → zh, en-US → en)
lang = language.strip().split("-")[0].split("_")[0].lower()
return lang
class CosyVoiceError(Exception):
"""CosyVoice API 调用异常。"""
@@ -461,6 +501,7 @@ class CosyVoiceService:
speed: float = 1.0,
volume: int = 50,
emotion: str = "",
language: str = "zh",
) -> dict:
"""提交语音合成任务(同步非流式,直接返回结果).
@@ -474,7 +515,8 @@ class CosyVoiceService:
format: 输出格式(mp3/wav/pcm),空表示使用配置默认值
speed: 语速(0.5-2.0),1.0 为正常速度
volume: 音量(0-100),默认 50
emotion: 情绪(natural/excited/calm/friendly),空串不传
emotion: 情绪(自然/兴奋/沉稳/亲切/开心/悲伤/愤怒/惊讶/恐惧/厌恶 等),空串不传
language: 语言代码(zh/en 等,默认 zh;系统音色仅 zh/en 传 language_hints)
Returns:
dict: {"audio_url": str, "request_id": str,
@@ -502,10 +544,18 @@ class CosyVoiceService:
"rate": speed,
"volume": volume,
}
# 情绪:归一化(中文→英文)后透传;空/非法则不传,走 CosyVoice 默认
# 情绪 → instruction 自然语言指令(CosyVoice v3 推荐方式)
norm_emotion = normalize_emotion(emotion)
if norm_emotion:
input_payload["emotion"] = norm_emotion
input_payload["instruction"] = f"你说话的情感是{norm_emotion}。"
# 语言 → language_hints 数组(仅取第一个元素生效);
# 系统音色(非克隆/非 voice_id 中包含下划线以外的短 ID)仅传 zh/en,其他语言不传避免报错
norm_lang = normalize_language(language)
# 简单判断:克隆音色一般是长 voice_id(包含 dash 或长度>20),对其不做语言限制;
# 系统音色(如 longxiaochun_v3)仅 zh/en 传 language_hints
is_system_voice = ("_" in voice_id or voice_id.startswith("long")) and len(voice_id) < 32
if (is_system_voice and norm_lang in _SYSTEM_VOICE_LANGS) or (not is_system_voice):
input_payload["language_hints"] = [norm_lang]
payload = {"model": self._model, "input": input_payload}
@@ -556,6 +606,7 @@ class CosyVoiceService:
speed: float = 1.0,
volume: int = 50,
emotion: str = "",
language: str = "zh",
timeout: float = 120.0,
) -> SynthesizeResult:
"""语音合成(同步非流式).
@@ -588,6 +639,7 @@ class CosyVoiceService:
speed=speed,
volume=volume,
emotion=emotion,
language=language,
)
return SynthesizeResult(
@@ -92,6 +92,7 @@ class TTSStreamingService:
sample_rate=sample_rate,
format=audio_format,
speed=speed,
language="zh-CN",
)
except CosyVoiceError as e:
logger.error(f"流式合成失败: {e}")
@@ -158,6 +159,7 @@ class TTSStreamingService:
sample_rate=sample_rate,
format=audio_format,
speed=speed,
language="zh-CN",
)
audio_url = result.get("audio_url", "")
if audio_url:
+6
View File
@@ -146,6 +146,7 @@ class TTSWorkflowService:
_meta = dict(job.metadata)
_speed = float(_meta.get("speed", 1.0) or 1.0)
_emotion = str(_meta.get("emotion", "") or "")
_language = str(_meta.get("language", "zh-CN") or "zh-CN")
submit_result = self.cosyvoice_service.submit_synthesize_task(
text=job.input_text,
voice_id=job.voice_id,
@@ -153,6 +154,7 @@ class TTSWorkflowService:
format=job.format,
speed=_speed,
emotion=_emotion,
language=_language,
)
# 保存 task_id / request_id 到 metadata
@@ -289,6 +291,7 @@ class TTSWorkflowService:
speed = float(job_metadata.get("speed", 1.0))
volume = int(job_metadata.get("volume", 50))
emotion = str(job_metadata.get("emotion", "") or "")
language = str(job_metadata.get("language", "zh-CN") or "zh-CN")
result = self.cosyvoice_service.submit_synthesize_task(
text=job.input_text,
@@ -298,6 +301,7 @@ class TTSWorkflowService:
speed=speed,
volume=volume,
emotion=emotion,
language=language,
)
audio_url = result.get("audio_url", "")
if not audio_url:
@@ -423,6 +427,7 @@ class TTSWorkflowService:
format=job.format,
speed=_seg_speed,
emotion=_seg_emotion,
language=_seg_meta.get("language", "zh-CN") or "zh-CN",
)
future_to_idx[future] = idx
@@ -548,6 +553,7 @@ class TTSWorkflowService:
speed=speed,
volume=volume,
emotion=emotion,
language=job.metadata.get("language", "zh-CN") if hasattr(job, "metadata") else "zh-CN",
)
future_to_idx[future] = idx
@@ -20,28 +20,36 @@ os.environ.setdefault("JWT_SECRET_KEY", "dev-secret-key-for-testing")
def test_normalize_emotion_english_values():
from packages.application.cosyvoice_service import normalize_emotion
assert normalize_emotion("natural") == "natural"
assert normalize_emotion("excited") == "excited"
assert normalize_emotion("calm") == "calm"
assert normalize_emotion("friendly") == "friendly"
# 大小写 / 空白容错
assert normalize_emotion(" Excited ") == "excited"
# 英文情绪统一归一化为中文描述词(用于 instruction 自然语言指令)
assert normalize_emotion("natural") == "自然"
assert normalize_emotion("excited") == "兴奋开心"
assert normalize_emotion("calm") == "沉稳平静"
assert normalize_emotion("friendly") == "亲切友好"
assert normalize_emotion("happy") == "开心愉快"
assert normalize_emotion("sad") == "悲伤难过"
assert normalize_emotion("angry") == "愤怒"
assert normalize_emotion("surprised") == "惊讶"
assert normalize_emotion("fearful") == "恐惧"
assert normalize_emotion("disgusted") == "厌恶"
def test_normalize_emotion_chinese_values():
from packages.application.cosyvoice_service import normalize_emotion
assert normalize_emotion("自然") == "natural"
assert normalize_emotion("兴奋") == "excited"
assert normalize_emotion("沉稳") == "calm"
assert normalize_emotion("亲切") == "friendly"
assert normalize_emotion("自然") == "自然"
assert normalize_emotion("兴奋") == "兴奋开心"
assert normalize_emotion("沉稳") == "沉稳平静"
assert normalize_emotion("亲切") == "亲切友好"
assert normalize_emotion("开心") == "开心愉快"
assert normalize_emotion("悲伤") == "悲伤难过"
assert normalize_emotion("愤怒") == "愤怒"
assert normalize_emotion("惊讶") == "惊讶"
def test_normalize_emotion_invalid_returns_empty():
from packages.application.cosyvoice_service import normalize_emotion
assert normalize_emotion("") == ""
assert normalize_emotion("angry") == ""
assert normalize_emotion("喜怒哀乐") == ""
@@ -83,24 +91,43 @@ def _make_service_with_captured_client(captured: dict):
return svc
def test_submit_synthesize_payload_includes_emotion_and_rate():
def test_submit_synthesize_payload_uses_instruction_and_language_hints():
"""#1898: emotion 通过 instruction 中文自然语言指令传递;语言用 language_hints."""
captured: dict = {}
svc = _make_service_with_captured_client(captured)
svc.submit_synthesize_task(text="你好", voice_id="v-1", speed=1.5, emotion="兴奋")
svc.submit_synthesize_task(text="你好", voice_id="longxiaochun_v3", speed=1.5, emotion="兴奋", language="zh-CN")
inp = captured["json"]["input"]
assert inp["emotion"] == "excited"
# 不再传 emotion 枚举字段
assert "emotion" not in inp
# 情绪通过 instruction 中文指令
assert inp["instruction"] == "你说话的情感是兴奋开心。"
# rate 字段保持
assert inp["rate"] == 1.5
# 系统音色 zh → language_hints=["zh"]
assert inp["language_hints"] == ["zh"]
def test_submit_synthesize_payload_omits_emotion_when_empty():
def test_submit_synthesize_payload_omits_instruction_when_emotion_empty():
captured: dict = {}
svc = _make_service_with_captured_client(captured)
svc.submit_synthesize_task(text="你好", voice_id="v-1")
svc.submit_synthesize_task(text="hello", voice_id="longxiaochun_v3", language="en-US")
assert "emotion" not in captured["json"]["input"]
inp = captured["json"]["input"]
assert "instruction" not in inp
assert inp["language_hints"] == ["en"]
def test_submit_synthesize_payload_english_emotion_maps_to_chinese():
captured: dict = {}
svc = _make_service_with_captured_client(captured)
svc.submit_synthesize_task(text="hi", voice_id="longxiaochun_v3", emotion="sad")
inp = captured["json"]["input"]
assert inp["instruction"] == "你说话的情感是悲伤难过。"
# ── 对口型 TTS 直生分支 ─────────────────────────────────────────────────
@@ -146,7 +173,7 @@ def test_create_job_tts_direct_mode_synthesizes_audio():
# v4: TTS 模式下 create_job 返回 tts_processing 状态,dispatch Celery 任务
assert job.status == "tts_processing"
assert job.emotion == "excited"
assert job.emotion == "兴奋" # 存储原始情绪值,normalize 在 CosyVoiceService 内部完成
assert job.speed == 1.2
# 不直接调用 CosyVoice(由 Celery 任务处理)
cosy.submit_synthesize_task.assert_not_called()
+1 -1
View File
@@ -254,7 +254,7 @@ class TestLipsyncServiceUnit:
mock_task.apply_async.assert_called_once()
# job 记录透传字段
assert job.speed == 1.2
assert job.emotion == "excited"
assert job.emotion == "兴奋"
# MediaKit 尚未提交(由 Celery 任务处理)
mock_mediakit.submit_lipsync.assert_not_called()
+4
View File
@@ -102,6 +102,7 @@ class TestTTSPreviewEndpoint:
voice_id="longxiaochun",
speed=1.0,
emotion="",
language="zh-CN",
)
def test_preview_with_speed(self):
@@ -148,6 +149,7 @@ class TestTTSPreviewEndpoint:
voice_id="v1",
speed=1.5,
emotion="",
language="zh-CN",
)
def test_preview_cosyvoice_error_returns_502(self):
@@ -333,6 +335,7 @@ class TestTTSPreviewEndpoint:
voice_id="cosyvoice_actual_voice_123",
speed=1.0,
emotion="",
language="zh-CN",
)
# Verify repo was queried with the UUID
mock_clone_repo.get.assert_called_once_with("abc123-uuid-of-profile")
@@ -410,6 +413,7 @@ class TestTTSPreviewEndpoint:
voice_id="longxiaoxia_v3",
speed=1.0,
emotion="",
language="zh-CN",
)
def test_preview_clone_voice_wrong_user_returns_403(self):
+1
View File
@@ -176,6 +176,7 @@ class TestStreamShortText:
sample_rate=22050,
format="wav",
speed=1.5,
language="zh-CN",
)
@pytest.mark.asyncio