diff --git a/alembic/versions/084_lipsync_jobs_style.py b/alembic/versions/084_lipsync_jobs_style.py new file mode 100644 index 000000000..08575bdf3 --- /dev/null +++ b/alembic/versions/084_lipsync_jobs_style.py @@ -0,0 +1,26 @@ +"""lipsync_jobs 新增 style 字段(TTS 语气风格) + +Revision ID: 084_lipsync_jobs_style +Revises: 083_cover_title_config +Create Date: 2026-09-21 +""" + +import sqlalchemy as sa + +from alembic import op + +revision = "084_lipsync_jobs_style" +down_revision = "083_cover_title_config" +branch_labels = None +depends_on = None + + +def upgrade() -> None: + op.add_column( + "lipsync_jobs", + sa.Column("style", sa.String(length=32), nullable=False, server_default=""), + ) + + +def downgrade() -> None: + op.drop_column("lipsync_jobs", "style") diff --git a/apps/api/app/api/routes/lipsync.py b/apps/api/app/api/routes/lipsync.py index ef8359f6a..12184cf85 100644 --- a/apps/api/app/api/routes/lipsync.py +++ b/apps/api/app/api/routes/lipsync.py @@ -111,6 +111,8 @@ def create_lipsync_job( voice_id=body.voice_id, script_text=body.script_text, speed=body.speed, + style=body.style or "", + volume=body.volume if body.volume is not None else 50, emotion=body.emotion, enable_video_loop=body.enable_video_loop, project_id=body.project_id, @@ -211,6 +213,8 @@ def preview_tts( voice_id=body.voice_id, script_text=body.script_text, speed=body.speed, + style=body.style or "", + volume=body.volume if body.volume is not None else 50, emotion=body.emotion, ) except MediaKitError as exc: diff --git a/apps/api/app/api/routes/tts.py b/apps/api/app/api/routes/tts.py index a01c33d12..6eca8baea 100644 --- a/apps/api/app/api/routes/tts.py +++ b/apps/api/app/api/routes/tts.py @@ -207,6 +207,9 @@ def synthesize( synthesis_meta = { "speed": request.speed, "emotion": request.emotion or "", + "style": request.style or "", + "volume": request.volume if request.volume is not None else 50, + "pitch": request.pitch if request.pitch is not None else 1.0, "language": request.language or "zh-CN", } if request.metadata_: @@ -654,6 +657,9 @@ def preview_tts( text=request.text, voice_id=actual_voice_id, speed=request.speed, + style=request.style or "", + volume=request.volume if request.volume is not None else 50, + pitch=request.pitch, emotion=request.emotion, language=getattr(request, "language", "zh-CN"), ) diff --git a/apps/api/app/schemas/lipsync.py b/apps/api/app/schemas/lipsync.py index 8bd9ef9bc..d73611913 100644 --- a/apps/api/app/schemas/lipsync.py +++ b/apps/api/app/schemas/lipsync.py @@ -29,6 +29,7 @@ class LipsyncJobResponse(BaseModel): voice_id: str = "" script_text: str = "" speed: float = 1.0 + style: str = "" emotion: str = "" mediakit_task_id: str status: str @@ -67,9 +68,14 @@ class CreateLipsyncJobRequest(BaseModel): voice_id: str = Field("", description="音色 ID(预置音色或克隆音色 profile UUID)") script_text: str = Field("", description="要合成的文案(直生模式必填,最长 5000 字符)") speed: float = Field(1.0, ge=0.5, le=2.0, description="语速(0.5-2.0),默认 1.0") + style: Optional[str] = Field( + None, + description="语气风格(natural/sweet/excited/professional/news/livestream),可选;优先级高于 emotion", + ) + volume: Optional[int] = Field(None, ge=0, le=100, description="音量(0-100),默认 50") emotion: str = Field( "", - description="情绪(英文枚举 neutral/happy/sad/angry/surprised/fearful/disgusted,或中文 中立/开心/难过/生气/惊讶/恐惧/厌恶;空为默认自然)", + description="[deprecated] 旧情绪参数,内部映射为 style", ) enable_video_loop: bool = Field( @@ -123,10 +129,15 @@ class AiAvatarTtsPreviewRequest(BaseModel): voice_id: str = Field(..., min_length=1, max_length=128, description="音色 ID") script_text: str = Field(..., min_length=1, max_length=5000, description="要合成的文案") speed: float = Field(1.0, ge=0.5, le=2.0, description="语速(0.5-2.0),默认 1.0") + style: Optional[str] = Field( + None, + description="语气风格(natural/sweet/excited/professional/news/livestream),可选;优先级高于 emotion", + ) + volume: Optional[int] = Field(None, ge=0, le=100, description="音量(0-100),默认 50") emotion: str = Field( "neutral", max_length=32, - description="情绪(英文枚举 neutral/happy/sad/angry/surprised/fearful/disgusted,或中文 中立/开心/难过/生气/惊讶/恐惧/厌恶;默认 neutral)", + description="[deprecated] 旧情绪参数,内部映射为 style;默认 neutral", ) diff --git a/apps/api/app/schemas/tts.py b/apps/api/app/schemas/tts.py index 39dd4f380..8cb92351a 100644 --- a/apps/api/app/schemas/tts.py +++ b/apps/api/app/schemas/tts.py @@ -16,9 +16,15 @@ class TTSSynthesizeRequest(BaseModel): output_name: str = Field("", description="输出文件名") language: str = Field("zh-CN", description="语言") speed: float = Field(1.0, ge=0.5, le=2.0, description="语速") + style: Optional[str] = Field( + None, + description="语气风格(natural/sweet/excited/professional/news/livestream),可选;优先级高于 emotion", + ) + volume: Optional[int] = Field(None, ge=0, le=100, description="音量(0-100),默认 50") + pitch: Optional[float] = Field(None, ge=0.5, le=2.0, description="音调(0.5-2.0),默认 1.0") emotion: str = Field( "", - description="情绪(中文/英文:自然/兴奋/沉稳/亲切/开心/悲伤/愤怒/惊讶/恐惧/厌恶 等;通过 instruction 自然语言指令控制)", + description="[deprecated] 旧情绪参数,内部映射为 style;新接入请使用 style", ) voice_model: str = Field("", description="语音模型名称") voice_clone_profile_id: str = Field("", description="关联的音色克隆档案 ID") @@ -113,9 +119,14 @@ class TTSPreviewRequest(BaseModel): text: str = Field(..., min_length=1, max_length=200, description="合成文本,限制 200 字") voice_id: str = Field(..., min_length=1, description="音色 ID") speed: float = Field(1.0, ge=0.5, le=2.0, description="语速") - emotion: str = Field("", description="情绪(中文/英文:自然/兴奋/沉稳/亲切/开心/悲伤/愤怒/惊讶/恐惧/厌恶 等)") + style: Optional[str] = Field( + None, + description="语气风格(natural/sweet/excited/professional/news/livestream),可选;优先级高于 emotion", + ) + volume: Optional[int] = Field(None, ge=0, le=100, description="音量(0-100),默认 50") + emotion: str = Field("", description="[deprecated] 旧情绪参数,内部映射为 style") language: str = Field("zh-CN", description="语言(zh-CN/en-US 等)") - pitch: float = Field(1.0, ge=0.5, le=2.0, description="音调(预留,当前未使用)") + pitch: float = Field(1.0, ge=0.5, le=2.0, description="音调(0.5-2.0),默认 1.0") class TTSPreviewResponse(BaseModel): diff --git a/apps/api/app/services/lipsync_service.py b/apps/api/app/services/lipsync_service.py index 517587e38..362e12f59 100644 --- a/apps/api/app/services/lipsync_service.py +++ b/apps/api/app/services/lipsync_service.py @@ -111,6 +111,8 @@ class LipsyncService: script_text: str, speed: float, emotion: str, + style: str = "", + volume: int = 50, ) -> str: """TTS 直生:调 CosyVoice 合成音频并转存 OSS,返回可公网访问的音频 URL. @@ -124,7 +126,9 @@ class LipsyncService: text=script_text, voice_id=actual_voice_id, speed=speed, - emotion=emotion, # normalize 在 CosyVoiceService 内部完成 + style=style, + volume=volume, + emotion=emotion, language="zh", ) except CosyVoiceError as exc: @@ -424,6 +428,8 @@ class LipsyncService: voice_id: str = "", script_text: str = "", speed: float = 1.0, + style: str = "", + volume: int = 50, emotion: str = "", enable_video_loop: bool = True, project_id: str = "", @@ -472,6 +478,7 @@ class LipsyncService: voice_id=voice_id or "", script_text=script_text or "", speed=speed, + style=style or "", emotion=emotion or "", # 音频直传(含预合成)直接进入 pending(后续同步改为 submitted);TTS 模式进入 tts_processing status="tts_processing" if is_tts_mode else "pending", @@ -493,6 +500,8 @@ class LipsyncService: voice_id, script_text, speed, + style or "", + volume, emotion or "", ) ) @@ -527,6 +536,8 @@ class LipsyncService: voice_id: str, script_text: str, speed: float = 1.0, + style: str = "", + volume: int = 50, emotion: str = "neutral", ) -> dict: """同步做 TTS 合成 + 下载 + ffprobe + 句子时间戳计算. diff --git a/apps/api/app/tasks/lipsync_tts.py b/apps/api/app/tasks/lipsync_tts.py index dce535485..bc58127ae 100644 --- a/apps/api/app/tasks/lipsync_tts.py +++ b/apps/api/app/tasks/lipsync_tts.py @@ -82,7 +82,9 @@ def tts_synthesize_and_submit( voice_id: str, script_text: str, speed: float, - emotion: str, + style: str = "", + volume: int = 50, + emotion: str = "", ): """异步执行 TTS 合成 + OSS 转存 + MediaKit 提交. @@ -159,6 +161,8 @@ def tts_synthesize_and_submit( text=script_text, voice_id=voice_id, speed=speed, + style=style, + volume=volume, emotion=emotion, language="zh", ) diff --git a/packages/adapters/sqlalchemy_impl/models.py b/packages/adapters/sqlalchemy_impl/models.py index ee7f37ce7..131e311c8 100755 --- a/packages/adapters/sqlalchemy_impl/models.py +++ b/packages/adapters/sqlalchemy_impl/models.py @@ -711,6 +711,7 @@ class LipsyncJobModel(Base): voice_id = Column(String(200), nullable=False, default="") script_text = Column(Text, nullable=False, default="") speed = Column(Float, nullable=False, default=1.0) + style = Column(String(32), nullable=False, default="") emotion = Column(String(20), nullable=False, default="") # MediaKit 任务状态 diff --git a/packages/application/cosyvoice_service.py b/packages/application/cosyvoice_service.py index 38df07560..73606dfbe 100644 --- a/packages/application/cosyvoice_service.py +++ b/packages/application/cosyvoice_service.py @@ -26,12 +26,43 @@ from packages.shared.config import get_shared_settings logger = logging.getLogger(__name__) -# CosyVoice v3 情绪通过 input.instruction 中文自然语言指令控制(不再使用枚举 emotion 字段)。 -# 官方文档:instruction 格式严格为 "你说话的情感是<情感值>。",结尾中文句号不可省略; -# 情感值必须是 7 种英文枚举之一:neutral/happy/sad/angry/surprised/fearful/disgusted。 +# ── Style(语气风格)→ CosyVoice instruct 自然语言指令 ── +# 前端 PR#2002 传 6 种 style:natural/sweet/excited/professional/news/livestream。 +# style 是新的统一参数;emotion 为 deprecated 兼容别名,内部映射为 style。 # 参考:https://help.aliyun.com/zh/model-studio/cosyvoice-voice-list -# 前端可传英文枚举或中文标签(中立/开心/难过/生气/惊讶/恐惧/厌恶),统一归一化为英文枚举。 -# 映射表 key(不区分大小写): 英文枚举/旧英文/中文标签 → 7 种标准英文枚举 +STYLE_INSTRUCTION_MAP: dict[str, str] = { + "natural": "用自然、平和的语气说话。", + "sweet": "用温柔甜美、亲切柔和的语气说话。", + "excited": "用兴奋、激动的语气说话。", + "professional": "用专业、正式的语气说话。", + "news": "用新闻播报的语气说话。", + "livestream": "用直播解说的语气说话。", +} + +# 有效 style 值集合(供 schema / 校验使用) +VALID_STYLES: frozenset[str] = frozenset(STYLE_INSTRUCTION_MAP.keys()) + +# ── 旧 emotion → 新 style 兼容映射(方案 B:统一 style,emotion deprecated)── +_EMOTION_TO_STYLE: dict[str, str] = { + "neutral": "natural", + "happy": "excited", + "sad": "sweet", + "angry": "excited", + "surprised": "excited", + "fearful": "sweet", + "disgusted": "natural", +} + +# 严格格式系统音色:style → emotion 回退(用于无法使用自由文本指令的音色) +_STYLE_TO_EMOTION: dict[str, str] = { + "natural": "neutral", + "sweet": "sad", + "excited": "happy", + "professional": "neutral", + # news / livestream 无直接对应 emotion,特殊处理 +} + +# ── 旧 emotion 映射表(deprecated,保留以兼容历史数据)── EMOTION_MAP: dict[str, str] = { # ── 7 种标准英文枚举(CosyVoice v3 官方支持的情感值)── "neutral": "neutral", @@ -122,6 +153,70 @@ def build_emotion_instruction(voice_id: str, emotion_enum: str) -> str: return "" +def resolve_style(style: str = "", emotion: str = "") -> str: + """统一解析 style 参数(方案 B). + + - style 有值且合法:直接使用(style 优先级最高)。 + - style 为空但 emotion 有值:将旧 emotion 归一化后映射为 style。 + - 两者皆空:返回空串(调用方不传 instruction)。 + + Args: + style: 新的语气风格(natural/sweet/excited/professional/news/livestream) + emotion: 旧的情绪参数(deprecated,内部映射为 style) + + Returns: + 解析后的 style 字符串;无需 instruct 时返回空串 + """ + s = (style or "").strip().lower() + if s: + if s in VALID_STYLES: + return s + logger.warning("未知的 style 值 %r,忽略 style 参数", style) + # 回退:emotion → style + norm = normalize_emotion(emotion) + if not norm: + return "" + mapped = _EMOTION_TO_STYLE.get(norm) + if not mapped: + logger.warning("emotion %r 无法映射到 style,跳过 instruct", norm) + return mapped or "" + + +def build_style_instruction(voice_id: str, style: str) -> str: + """根据 voice 类型构造 style instruction. + + - natural:返回空串(不额外加 instruct,使用 CosyVoice 默认自然语气)。 + - 克隆/设计音色:使用中文自然语言指令(DashScope 允许任意自然语言)。 + - 系统音色中支持 emotion instruct 的白名单音色: + 若 style 可映射到 emotion,用严格格式 "你说话的情感是。"; + news/livestream 尝试直接用中文 instruct(部分音色支持自由文本)。 + - 其他系统音色:返回空串。 + + Args: + voice_id: CosyVoice voice 参数 + style: 已通过 resolve_style() 解析的 style 值 + + Returns: + 拼接好的 instruction 字符串;无需 instruct 时返回空串 + """ + if not style or style == "natural": + return "" + # 克隆音色:直接使用中文自然语言指令 + if _is_cloned_voice(voice_id): + return STYLE_INSTRUCTION_MAP.get(style, "") + # 系统音色白名单:优先映射到严格 emotion 格式 + if voice_id in _SYSTEM_VOICES_WITH_EMOTION_INSTRUCT: + emotion_val = _STYLE_TO_EMOTION.get(style) + if emotion_val: + return f"你说话的情感是{emotion_val}。" + # news/livestream 无 emotion 对应,尝试自由中文 instruct + desc = STYLE_INSTRUCTION_MAP.get(style, "") + if desc: + logger.info("音色 %s 使用自由文本 style instruct: %s", voice_id, desc) + return desc + return "" + + def normalize_emotion(emotion: str) -> str: """将前端情绪值归一化为 CosyVoice v3 官方英文枚举,用于拼入 instruction. @@ -573,6 +668,8 @@ class CosyVoiceService: volume: int = 50, emotion: str = "", language: str = "zh", + style: str = "", + pitch: float = 1.0, ) -> dict: """提交语音合成任务(同步非流式,直接返回结果). @@ -586,9 +683,10 @@ class CosyVoiceService: format: 输出格式(mp3/wav/pcm),空表示使用配置默认值 speed: 语速(0.5-2.0),1.0 为正常速度 volume: 音量(0-100),默认 50 - emotion: 情绪,英文枚举 neutral/happy/sad/angry/surprised/fearful/disgusted - 或前端中文标签(中立/开心/难过/生气/惊讶/恐惧/厌恶),兼容旧值 - natural/excited/calm/friendly;空串不传,未知值默认 neutral + style: 语气风格(natural/sweet/excited/professional/news/livestream), + 新的统一参数;优先级高于 emotion + pitch: 音调(0.5-2.0),1.0 为默认值 + emotion: 【deprecated】旧情绪参数,内部通过 resolve_style() 映射为 style language: 语言代码(zh/en 等,默认 zh;系统音色仅 zh/en 传 language_hints) Returns: @@ -617,11 +715,20 @@ class CosyVoiceService: "rate": speed, "volume": volume, } - # 情绪 → instruction(按 voice 类型选择格式) - norm_emotion = normalize_emotion(emotion) - emotion_instruction = build_emotion_instruction(voice_id, norm_emotion) - if emotion_instruction: - input_payload["instruction"] = emotion_instruction + # pitch: CosyVoice API 支持 [0.5, 2.0],非默认值时才传 + if pitch and pitch != 1.0: + input_payload["pitch"] = pitch + # style 优先:显式传 style 时走 build_style_instruction(); + # 无 style 时回退到旧的 emotion → build_emotion_instruction() 逻辑(向后兼容) + s = (style or "").strip().lower() + instruction = "" + if s: + instruction = build_style_instruction(voice_id, s) + if not instruction: + norm_emotion = normalize_emotion(emotion) + instruction = build_emotion_instruction(voice_id, norm_emotion) + if instruction: + input_payload["instruction"] = instruction # 语言 → language_hints 数组(仅取第一个元素生效); # 系统音色(非克隆/非 voice_id 中包含下划线以外的短 ID)仅传 zh/en,其他语言不传避免报错 norm_lang = normalize_language(language) @@ -682,6 +789,8 @@ class CosyVoiceService: emotion: str = "", language: str = "zh", timeout: float = 120.0, + style: str = "", + pitch: float = 1.0, ) -> SynthesizeResult: """语音合成(同步非流式). @@ -695,6 +804,8 @@ class CosyVoiceService: format: 输出格式(mp3/wav/pcm),空表示使用配置默认值 speed: 语速(0.5-2.0),1.0 为正常速度 volume: 音量(0-100),默认 50 + style: 语气风格(natural/sweet/excited/professional/news/livestream) + pitch: 音调(0.5-2.0),1.0 为默认值 timeout: 超时时间(秒),保留参数兼容 Returns: @@ -714,6 +825,8 @@ class CosyVoiceService: volume=volume, emotion=emotion, language=language, + style=style, + pitch=pitch, ) return SynthesizeResult( diff --git a/packages/application/tts_job/workflow.py b/packages/application/tts_job/workflow.py index 3ca40b0e6..17cfcd204 100644 --- a/packages/application/tts_job/workflow.py +++ b/packages/application/tts_job/workflow.py @@ -146,6 +146,9 @@ class TTSWorkflowService: _meta = dict(job.metadata) _speed = float(_meta.get("speed", 1.0) or 1.0) _emotion = str(_meta.get("emotion", "") or "") + _style = str(_meta.get("style", "") or "") + _volume = int(_meta.get("volume", 50) or 50) + _pitch = float(_meta.get("pitch", 1.0) or 1.0) _language = str(_meta.get("language", "zh-CN") or "zh-CN") submit_result = self.cosyvoice_service.submit_synthesize_task( text=job.input_text, @@ -153,6 +156,9 @@ class TTSWorkflowService: sample_rate=job.sample_rate, format=job.format, speed=_speed, + style=_style, + volume=_volume, + pitch=_pitch, emotion=_emotion, language=_language, ) @@ -290,6 +296,8 @@ class TTSWorkflowService: job_metadata = job.metadata or {} speed = float(job_metadata.get("speed", 1.0)) volume = int(job_metadata.get("volume", 50)) + style = str(job_metadata.get("style", "") or "") + pitch = float(job_metadata.get("pitch", 1.0)) emotion = str(job_metadata.get("emotion", "") or "") language = str(job_metadata.get("language", "zh-CN") or "zh-CN") @@ -299,7 +307,9 @@ class TTSWorkflowService: sample_rate=job.sample_rate, format=job.format, speed=speed, + style=style, volume=volume, + pitch=pitch, emotion=emotion, language=language, ) @@ -415,6 +425,9 @@ class TTSWorkflowService: _seg_meta = job.metadata or {} _seg_speed = float(_seg_meta.get("speed", 1.0) or 1.0) _seg_emotion = str(_seg_meta.get("emotion", "") or "") + _seg_style = str(_seg_meta.get("style", "") or "") + _seg_volume = int(_seg_meta.get("volume", 50) or 50) + _seg_pitch = float(_seg_meta.get("pitch", 1.0) or 1.0) with ThreadPoolExecutor(max_workers=max_workers) as executor: future_to_idx = {} @@ -426,6 +439,9 @@ class TTSWorkflowService: sample_rate=job.sample_rate, format=job.format, speed=_seg_speed, + style=_seg_style, + volume=_seg_volume, + pitch=_seg_pitch, emotion=_seg_emotion, language=_seg_meta.get("language", "zh-CN") or "zh-CN", ) @@ -517,6 +533,8 @@ class TTSWorkflowService: job_metadata = job.metadata or {} speed = float(job_metadata.get("speed", 1.0)) volume = int(job_metadata.get("volume", 50)) + style = str(job_metadata.get("style", "") or "") + pitch = float(job_metadata.get("pitch", 1.0)) emotion = str(job_metadata.get("emotion", "") or "") # 分段文本(用于缺失段重新合成) @@ -551,7 +569,9 @@ class TTSWorkflowService: sample_rate=job.sample_rate, format=job.format, speed=speed, + style=style, volume=volume, + pitch=pitch, emotion=emotion, language=job.metadata.get("language", "zh-CN") if hasattr(job, "metadata") else "zh-CN", ) diff --git a/tests/unit/test_ai_avatar_emotion_tts_lipsync.py b/tests/unit/test_ai_avatar_emotion_tts_lipsync.py index 635555873..c7c658e70 100644 --- a/tests/unit/test_ai_avatar_emotion_tts_lipsync.py +++ b/tests/unit/test_ai_avatar_emotion_tts_lipsync.py @@ -6,6 +6,7 @@ CI 增量映射: ai_avatar_cover_service 智能选帧 """ +import importlib import os from unittest.mock import MagicMock, patch @@ -166,6 +167,116 @@ def test_submit_synthesize_payload_cloned_voice_english_emotion(): assert inp["instruction"] == "Speak in a sad tone." +# ── style(语气风格,#2002)──────────────────────────────────────────── + + +def test_style_natural_omits_instruction(): + """style=natural 不加 instruct,使用 CosyVoice 默认自然语气。""" + captured: dict = {} + svc = _make_service_with_captured_client(captured) + + svc.submit_synthesize_task(text="你好", voice_id="myclone_voice", style="natural") + + inp = captured["json"]["input"] + assert "instruction" not in inp + + +def test_style_sweet_cloned_voice_uses_chinese_instruction(): + """克隆音色 + style=sweet → 中文自然语言指令。""" + captured: dict = {} + svc = _make_service_with_captured_client(captured) + + svc.submit_synthesize_task(text="你好", voice_id="myclone_voice", style="sweet") + + inp = captured["json"]["input"] + assert inp["instruction"] == "用温柔甜美、亲切柔和的语气说话。" + + +@pytest.mark.parametrize( + ("style", "expected_fragment"), + [ + ("excited", "兴奋"), + ("professional", "专业"), + ("news", "新闻"), + ("livestream", "直播"), + ], +) +def test_style_values_cloned_voice(style, expected_fragment): + captured: dict = {} + svc = _make_service_with_captured_client(captured) + + svc.submit_synthesize_task(text="你好", voice_id="myclone_voice", style=style) + + inp = captured["json"]["input"] + assert expected_fragment in inp["instruction"] + + +def test_style_system_voice_uses_emotion_mapping(): + """系统白名单音色 + style=sweet → 严格中文 emotion 格式(映射到 sad)。""" + captured: dict = {} + svc = _make_service_with_captured_client(captured) + + svc.submit_synthesize_task(text="你好", voice_id="longanyang", style="sweet") + + inp = captured["json"]["input"] + assert inp["instruction"] == "你说话的情感是sad。" + + +def test_style_takes_priority_over_emotion(): + """同时传 style 和 emotion 时以 style 为准。""" + captured: dict = {} + svc = _make_service_with_captured_client(captured) + + svc.submit_synthesize_task(text="你好", voice_id="myclone_voice", style="excited", emotion="sad") + + inp = captured["json"]["input"] + assert "兴奋" in inp["instruction"] + + +def test_unknown_style_ignored_falls_back_to_emotion(): + """未知 style 值被忽略,回退到 emotion 逻辑。""" + captured: dict = {} + svc = _make_service_with_captured_client(captured) + + svc.submit_synthesize_task(text="hi", voice_id="myclone_voice", style="nonexistent", emotion="sad") + + inp = captured["json"]["input"] + assert inp["instruction"] == "Speak in a sad tone." + + +def test_pitch_passed_only_when_non_default(): + captured: dict = {} + svc = _make_service_with_captured_client(captured) + + svc.submit_synthesize_task(text="hi", voice_id="myclone_voice", pitch=1.5) + + inp = captured["json"]["input"] + assert inp["pitch"] == 1.5 + + +def test_pitch_omitted_at_default(): + captured: dict = {} + svc = _make_service_with_captured_client(captured) + + svc.submit_synthesize_task(text="hi", voice_id="myclone_voice", pitch=1.0) + + inp = captured["json"]["input"] + assert "pitch" not in inp + + +def test_resolve_style_maps_emotion_to_style(): + """旧 emotion 值通过 resolve_style 映射为 style。""" + mod = importlib.import_module("packages.application.cosyvoice_service") + + assert mod.resolve_style(emotion="happy") == "excited" + assert mod.resolve_style(emotion="sad") == "sweet" + assert mod.resolve_style(emotion="neutral") == "natural" + assert mod.resolve_style(style="news") == "news" + # style 优先 + assert mod.resolve_style(style="news", emotion="happy") == "news" + assert mod.resolve_style() == "" + + # ── 对口型 TTS 直生分支 ───────────────────────────────────────────────── diff --git a/tests/unit/test_tts_preview.py b/tests/unit/test_tts_preview.py index c3fae5822..efc4b1e09 100644 --- a/tests/unit/test_tts_preview.py +++ b/tests/unit/test_tts_preview.py @@ -101,6 +101,9 @@ class TestTTSPreviewEndpoint: text="你好世界", voice_id="longxiaochun", speed=1.0, + style="", + volume=50, + pitch=1.0, emotion="", language="zh-CN", ) @@ -148,6 +151,9 @@ class TestTTSPreviewEndpoint: text="测试", voice_id="v1", speed=1.5, + style="", + volume=50, + pitch=1.0, emotion="", language="zh-CN", ) @@ -334,6 +340,9 @@ class TestTTSPreviewEndpoint: text="克隆音色测试", voice_id="cosyvoice_actual_voice_123", speed=1.0, + style="", + volume=50, + pitch=1.0, emotion="", language="zh-CN", ) @@ -412,6 +421,9 @@ class TestTTSPreviewEndpoint: text="预设音色测试", voice_id="longxiaoxia_v3", speed=1.0, + style="", + volume=50, + pitch=1.0, emotion="", language="zh-CN", )