feat(tts): style语气风格参数全链路接入 + 语速/音量/音调透传修复 #2005
@@ -0,0 +1,26 @@
|
||||
"""lipsync_jobs 新增 style 字段(TTS 语气风格)
|
||||
|
||||
Revision ID: 084_lipsync_jobs_style
|
||||
Revises: 083_cover_title_config
|
||||
Create Date: 2026-09-21
|
||||
"""
|
||||
|
||||
import sqlalchemy as sa
|
||||
|
||||
from alembic import op
|
||||
|
||||
revision = "084_lipsync_jobs_style"
|
||||
down_revision = "083_cover_title_config"
|
||||
branch_labels = None
|
||||
depends_on = None
|
||||
|
||||
|
||||
def upgrade() -> None:
|
||||
op.add_column(
|
||||
"lipsync_jobs",
|
||||
sa.Column("style", sa.String(length=32), nullable=False, server_default=""),
|
||||
)
|
||||
|
||||
|
||||
def downgrade() -> None:
|
||||
op.drop_column("lipsync_jobs", "style")
|
||||
@@ -111,6 +111,8 @@ def create_lipsync_job(
|
||||
voice_id=body.voice_id,
|
||||
script_text=body.script_text,
|
||||
speed=body.speed,
|
||||
style=body.style or "",
|
||||
volume=body.volume if body.volume is not None else 50,
|
||||
emotion=body.emotion,
|
||||
enable_video_loop=body.enable_video_loop,
|
||||
project_id=body.project_id,
|
||||
@@ -211,6 +213,8 @@ def preview_tts(
|
||||
voice_id=body.voice_id,
|
||||
script_text=body.script_text,
|
||||
speed=body.speed,
|
||||
style=body.style or "",
|
||||
volume=body.volume if body.volume is not None else 50,
|
||||
emotion=body.emotion,
|
||||
)
|
||||
except MediaKitError as exc:
|
||||
|
||||
@@ -207,6 +207,9 @@ def synthesize(
|
||||
synthesis_meta = {
|
||||
"speed": request.speed,
|
||||
"emotion": request.emotion or "",
|
||||
"style": request.style or "",
|
||||
"volume": request.volume if request.volume is not None else 50,
|
||||
"pitch": request.pitch if request.pitch is not None else 1.0,
|
||||
"language": request.language or "zh-CN",
|
||||
}
|
||||
if request.metadata_:
|
||||
@@ -654,6 +657,9 @@ def preview_tts(
|
||||
text=request.text,
|
||||
voice_id=actual_voice_id,
|
||||
speed=request.speed,
|
||||
style=request.style or "",
|
||||
volume=request.volume if request.volume is not None else 50,
|
||||
pitch=request.pitch,
|
||||
emotion=request.emotion,
|
||||
language=getattr(request, "language", "zh-CN"),
|
||||
)
|
||||
|
||||
@@ -29,6 +29,7 @@ class LipsyncJobResponse(BaseModel):
|
||||
voice_id: str = ""
|
||||
script_text: str = ""
|
||||
speed: float = 1.0
|
||||
style: str = ""
|
||||
emotion: str = ""
|
||||
mediakit_task_id: str
|
||||
status: str
|
||||
@@ -67,9 +68,14 @@ class CreateLipsyncJobRequest(BaseModel):
|
||||
voice_id: str = Field("", description="音色 ID(预置音色或克隆音色 profile UUID)")
|
||||
script_text: str = Field("", description="要合成的文案(直生模式必填,最长 5000 字符)")
|
||||
speed: float = Field(1.0, ge=0.5, le=2.0, description="语速(0.5-2.0),默认 1.0")
|
||||
style: Optional[str] = Field(
|
||||
None,
|
||||
description="语气风格(natural/sweet/excited/professional/news/livestream),可选;优先级高于 emotion",
|
||||
)
|
||||
volume: Optional[int] = Field(None, ge=0, le=100, description="音量(0-100),默认 50")
|
||||
emotion: str = Field(
|
||||
"",
|
||||
description="情绪(英文枚举 neutral/happy/sad/angry/surprised/fearful/disgusted,或中文 中立/开心/难过/生气/惊讶/恐惧/厌恶;空为默认自然)",
|
||||
description="[deprecated] 旧情绪参数,内部映射为 style",
|
||||
)
|
||||
|
||||
enable_video_loop: bool = Field(
|
||||
@@ -123,10 +129,15 @@ class AiAvatarTtsPreviewRequest(BaseModel):
|
||||
voice_id: str = Field(..., min_length=1, max_length=128, description="音色 ID")
|
||||
script_text: str = Field(..., min_length=1, max_length=5000, description="要合成的文案")
|
||||
speed: float = Field(1.0, ge=0.5, le=2.0, description="语速(0.5-2.0),默认 1.0")
|
||||
style: Optional[str] = Field(
|
||||
None,
|
||||
description="语气风格(natural/sweet/excited/professional/news/livestream),可选;优先级高于 emotion",
|
||||
)
|
||||
volume: Optional[int] = Field(None, ge=0, le=100, description="音量(0-100),默认 50")
|
||||
emotion: str = Field(
|
||||
"neutral",
|
||||
max_length=32,
|
||||
description="情绪(英文枚举 neutral/happy/sad/angry/surprised/fearful/disgusted,或中文 中立/开心/难过/生气/惊讶/恐惧/厌恶;默认 neutral)",
|
||||
description="[deprecated] 旧情绪参数,内部映射为 style;默认 neutral",
|
||||
)
|
||||
|
||||
|
||||
|
||||
@@ -16,9 +16,15 @@ class TTSSynthesizeRequest(BaseModel):
|
||||
output_name: str = Field("", description="输出文件名")
|
||||
language: str = Field("zh-CN", description="语言")
|
||||
speed: float = Field(1.0, ge=0.5, le=2.0, description="语速")
|
||||
style: Optional[str] = Field(
|
||||
None,
|
||||
description="语气风格(natural/sweet/excited/professional/news/livestream),可选;优先级高于 emotion",
|
||||
)
|
||||
volume: Optional[int] = Field(None, ge=0, le=100, description="音量(0-100),默认 50")
|
||||
pitch: Optional[float] = Field(None, ge=0.5, le=2.0, description="音调(0.5-2.0),默认 1.0")
|
||||
emotion: str = Field(
|
||||
"",
|
||||
description="情绪(中文/英文:自然/兴奋/沉稳/亲切/开心/悲伤/愤怒/惊讶/恐惧/厌恶 等;通过 instruction 自然语言指令控制)",
|
||||
description="[deprecated] 旧情绪参数,内部映射为 style;新接入请使用 style",
|
||||
)
|
||||
voice_model: str = Field("", description="语音模型名称")
|
||||
voice_clone_profile_id: str = Field("", description="关联的音色克隆档案 ID")
|
||||
@@ -113,9 +119,14 @@ class TTSPreviewRequest(BaseModel):
|
||||
text: str = Field(..., min_length=1, max_length=200, description="合成文本,限制 200 字")
|
||||
voice_id: str = Field(..., min_length=1, description="音色 ID")
|
||||
speed: float = Field(1.0, ge=0.5, le=2.0, description="语速")
|
||||
emotion: str = Field("", description="情绪(中文/英文:自然/兴奋/沉稳/亲切/开心/悲伤/愤怒/惊讶/恐惧/厌恶 等)")
|
||||
style: Optional[str] = Field(
|
||||
None,
|
||||
description="语气风格(natural/sweet/excited/professional/news/livestream),可选;优先级高于 emotion",
|
||||
)
|
||||
volume: Optional[int] = Field(None, ge=0, le=100, description="音量(0-100),默认 50")
|
||||
emotion: str = Field("", description="[deprecated] 旧情绪参数,内部映射为 style")
|
||||
language: str = Field("zh-CN", description="语言(zh-CN/en-US 等)")
|
||||
pitch: float = Field(1.0, ge=0.5, le=2.0, description="音调(预留,当前未使用)")
|
||||
pitch: float = Field(1.0, ge=0.5, le=2.0, description="音调(0.5-2.0),默认 1.0")
|
||||
|
||||
|
||||
class TTSPreviewResponse(BaseModel):
|
||||
|
||||
@@ -111,6 +111,8 @@ class LipsyncService:
|
||||
script_text: str,
|
||||
speed: float,
|
||||
emotion: str,
|
||||
style: str = "",
|
||||
volume: int = 50,
|
||||
) -> str:
|
||||
"""TTS 直生:调 CosyVoice 合成音频并转存 OSS,返回可公网访问的音频 URL.
|
||||
|
||||
@@ -124,7 +126,9 @@ class LipsyncService:
|
||||
text=script_text,
|
||||
voice_id=actual_voice_id,
|
||||
speed=speed,
|
||||
emotion=emotion, # normalize 在 CosyVoiceService 内部完成
|
||||
style=style,
|
||||
volume=volume,
|
||||
emotion=emotion,
|
||||
language="zh",
|
||||
)
|
||||
except CosyVoiceError as exc:
|
||||
@@ -424,6 +428,8 @@ class LipsyncService:
|
||||
voice_id: str = "",
|
||||
script_text: str = "",
|
||||
speed: float = 1.0,
|
||||
style: str = "",
|
||||
volume: int = 50,
|
||||
emotion: str = "",
|
||||
enable_video_loop: bool = True,
|
||||
project_id: str = "",
|
||||
@@ -472,6 +478,7 @@ class LipsyncService:
|
||||
voice_id=voice_id or "",
|
||||
script_text=script_text or "",
|
||||
speed=speed,
|
||||
style=style or "",
|
||||
emotion=emotion or "",
|
||||
# 音频直传(含预合成)直接进入 pending(后续同步改为 submitted);TTS 模式进入 tts_processing
|
||||
status="tts_processing" if is_tts_mode else "pending",
|
||||
@@ -493,6 +500,8 @@ class LipsyncService:
|
||||
voice_id,
|
||||
script_text,
|
||||
speed,
|
||||
style or "",
|
||||
volume,
|
||||
emotion or "",
|
||||
)
|
||||
)
|
||||
@@ -527,6 +536,8 @@ class LipsyncService:
|
||||
voice_id: str,
|
||||
script_text: str,
|
||||
speed: float = 1.0,
|
||||
style: str = "",
|
||||
volume: int = 50,
|
||||
emotion: str = "neutral",
|
||||
) -> dict:
|
||||
"""同步做 TTS 合成 + 下载 + ffprobe + 句子时间戳计算.
|
||||
|
||||
@@ -82,7 +82,9 @@ def tts_synthesize_and_submit(
|
||||
voice_id: str,
|
||||
script_text: str,
|
||||
speed: float,
|
||||
emotion: str,
|
||||
style: str = "",
|
||||
volume: int = 50,
|
||||
emotion: str = "",
|
||||
):
|
||||
"""异步执行 TTS 合成 + OSS 转存 + MediaKit 提交.
|
||||
|
||||
@@ -159,6 +161,8 @@ def tts_synthesize_and_submit(
|
||||
text=script_text,
|
||||
voice_id=voice_id,
|
||||
speed=speed,
|
||||
style=style,
|
||||
volume=volume,
|
||||
emotion=emotion,
|
||||
language="zh",
|
||||
)
|
||||
|
||||
@@ -711,6 +711,7 @@ class LipsyncJobModel(Base):
|
||||
voice_id = Column(String(200), nullable=False, default="")
|
||||
script_text = Column(Text, nullable=False, default="")
|
||||
speed = Column(Float, nullable=False, default=1.0)
|
||||
style = Column(String(32), nullable=False, default="")
|
||||
emotion = Column(String(20), nullable=False, default="")
|
||||
|
||||
# MediaKit 任务状态
|
||||
|
||||
@@ -26,12 +26,43 @@ from packages.shared.config import get_shared_settings
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
# CosyVoice v3 情绪通过 input.instruction 中文自然语言指令控制(不再使用枚举 emotion 字段)。
|
||||
# 官方文档:instruction 格式严格为 "你说话的情感是<情感值>。",结尾中文句号不可省略;
|
||||
# 情感值必须是 7 种英文枚举之一:neutral/happy/sad/angry/surprised/fearful/disgusted。
|
||||
# ── Style(语气风格)→ CosyVoice instruct 自然语言指令 ──
|
||||
# 前端 PR#2002 传 6 种 style:natural/sweet/excited/professional/news/livestream。
|
||||
# style 是新的统一参数;emotion 为 deprecated 兼容别名,内部映射为 style。
|
||||
# 参考:https://help.aliyun.com/zh/model-studio/cosyvoice-voice-list
|
||||
# 前端可传英文枚举或中文标签(中立/开心/难过/生气/惊讶/恐惧/厌恶),统一归一化为英文枚举。
|
||||
# 映射表 key(不区分大小写): 英文枚举/旧英文/中文标签 → 7 种标准英文枚举
|
||||
STYLE_INSTRUCTION_MAP: dict[str, str] = {
|
||||
"natural": "用自然、平和的语气说话。",
|
||||
"sweet": "用温柔甜美、亲切柔和的语气说话。",
|
||||
"excited": "用兴奋、激动的语气说话。",
|
||||
"professional": "用专业、正式的语气说话。",
|
||||
"news": "用新闻播报的语气说话。",
|
||||
"livestream": "用直播解说的语气说话。",
|
||||
}
|
||||
|
||||
# 有效 style 值集合(供 schema / 校验使用)
|
||||
VALID_STYLES: frozenset[str] = frozenset(STYLE_INSTRUCTION_MAP.keys())
|
||||
|
||||
# ── 旧 emotion → 新 style 兼容映射(方案 B:统一 style,emotion deprecated)──
|
||||
_EMOTION_TO_STYLE: dict[str, str] = {
|
||||
"neutral": "natural",
|
||||
"happy": "excited",
|
||||
"sad": "sweet",
|
||||
"angry": "excited",
|
||||
"surprised": "excited",
|
||||
"fearful": "sweet",
|
||||
"disgusted": "natural",
|
||||
}
|
||||
|
||||
# 严格格式系统音色:style → emotion 回退(用于无法使用自由文本指令的音色)
|
||||
_STYLE_TO_EMOTION: dict[str, str] = {
|
||||
"natural": "neutral",
|
||||
"sweet": "sad",
|
||||
"excited": "happy",
|
||||
"professional": "neutral",
|
||||
# news / livestream 无直接对应 emotion,特殊处理
|
||||
}
|
||||
|
||||
# ── 旧 emotion 映射表(deprecated,保留以兼容历史数据)──
|
||||
EMOTION_MAP: dict[str, str] = {
|
||||
# ── 7 种标准英文枚举(CosyVoice v3 官方支持的情感值)──
|
||||
"neutral": "neutral",
|
||||
@@ -122,6 +153,70 @@ def build_emotion_instruction(voice_id: str, emotion_enum: str) -> str:
|
||||
return ""
|
||||
|
||||
|
||||
def resolve_style(style: str = "", emotion: str = "") -> str:
|
||||
"""统一解析 style 参数(方案 B).
|
||||
|
||||
- style 有值且合法:直接使用(style 优先级最高)。
|
||||
- style 为空但 emotion 有值:将旧 emotion 归一化后映射为 style。
|
||||
- 两者皆空:返回空串(调用方不传 instruction)。
|
||||
|
||||
Args:
|
||||
style: 新的语气风格(natural/sweet/excited/professional/news/livestream)
|
||||
emotion: 旧的情绪参数(deprecated,内部映射为 style)
|
||||
|
||||
Returns:
|
||||
解析后的 style 字符串;无需 instruct 时返回空串
|
||||
"""
|
||||
s = (style or "").strip().lower()
|
||||
if s:
|
||||
if s in VALID_STYLES:
|
||||
return s
|
||||
logger.warning("未知的 style 值 %r,忽略 style 参数", style)
|
||||
# 回退:emotion → style
|
||||
norm = normalize_emotion(emotion)
|
||||
if not norm:
|
||||
return ""
|
||||
mapped = _EMOTION_TO_STYLE.get(norm)
|
||||
if not mapped:
|
||||
logger.warning("emotion %r 无法映射到 style,跳过 instruct", norm)
|
||||
return mapped or ""
|
||||
|
||||
|
||||
def build_style_instruction(voice_id: str, style: str) -> str:
|
||||
"""根据 voice 类型构造 style instruction.
|
||||
|
||||
- natural:返回空串(不额外加 instruct,使用 CosyVoice 默认自然语气)。
|
||||
- 克隆/设计音色:使用中文自然语言指令(DashScope 允许任意自然语言)。
|
||||
- 系统音色中支持 emotion instruct 的白名单音色:
|
||||
若 style 可映射到 emotion,用严格格式 "你说话的情感是<emotion>。";
|
||||
news/livestream 尝试直接用中文 instruct(部分音色支持自由文本)。
|
||||
- 其他系统音色:返回空串。
|
||||
|
||||
Args:
|
||||
voice_id: CosyVoice voice 参数
|
||||
style: 已通过 resolve_style() 解析的 style 值
|
||||
|
||||
Returns:
|
||||
拼接好的 instruction 字符串;无需 instruct 时返回空串
|
||||
"""
|
||||
if not style or style == "natural":
|
||||
return ""
|
||||
# 克隆音色:直接使用中文自然语言指令
|
||||
if _is_cloned_voice(voice_id):
|
||||
return STYLE_INSTRUCTION_MAP.get(style, "")
|
||||
# 系统音色白名单:优先映射到严格 emotion 格式
|
||||
if voice_id in _SYSTEM_VOICES_WITH_EMOTION_INSTRUCT:
|
||||
emotion_val = _STYLE_TO_EMOTION.get(style)
|
||||
if emotion_val:
|
||||
return f"你说话的情感是{emotion_val}。"
|
||||
# news/livestream 无 emotion 对应,尝试自由中文 instruct
|
||||
desc = STYLE_INSTRUCTION_MAP.get(style, "")
|
||||
if desc:
|
||||
logger.info("音色 %s 使用自由文本 style instruct: %s", voice_id, desc)
|
||||
return desc
|
||||
return ""
|
||||
|
||||
|
||||
def normalize_emotion(emotion: str) -> str:
|
||||
"""将前端情绪值归一化为 CosyVoice v3 官方英文枚举,用于拼入 instruction.
|
||||
|
||||
@@ -573,6 +668,8 @@ class CosyVoiceService:
|
||||
volume: int = 50,
|
||||
emotion: str = "",
|
||||
language: str = "zh",
|
||||
style: str = "",
|
||||
pitch: float = 1.0,
|
||||
) -> dict:
|
||||
"""提交语音合成任务(同步非流式,直接返回结果).
|
||||
|
||||
@@ -586,9 +683,10 @@ class CosyVoiceService:
|
||||
format: 输出格式(mp3/wav/pcm),空表示使用配置默认值
|
||||
speed: 语速(0.5-2.0),1.0 为正常速度
|
||||
volume: 音量(0-100),默认 50
|
||||
emotion: 情绪,英文枚举 neutral/happy/sad/angry/surprised/fearful/disgusted
|
||||
或前端中文标签(中立/开心/难过/生气/惊讶/恐惧/厌恶),兼容旧值
|
||||
natural/excited/calm/friendly;空串不传,未知值默认 neutral
|
||||
style: 语气风格(natural/sweet/excited/professional/news/livestream),
|
||||
新的统一参数;优先级高于 emotion
|
||||
pitch: 音调(0.5-2.0),1.0 为默认值
|
||||
emotion: 【deprecated】旧情绪参数,内部通过 resolve_style() 映射为 style
|
||||
language: 语言代码(zh/en 等,默认 zh;系统音色仅 zh/en 传 language_hints)
|
||||
|
||||
Returns:
|
||||
@@ -617,11 +715,20 @@ class CosyVoiceService:
|
||||
"rate": speed,
|
||||
"volume": volume,
|
||||
}
|
||||
# 情绪 → instruction(按 voice 类型选择格式)
|
||||
norm_emotion = normalize_emotion(emotion)
|
||||
emotion_instruction = build_emotion_instruction(voice_id, norm_emotion)
|
||||
if emotion_instruction:
|
||||
input_payload["instruction"] = emotion_instruction
|
||||
# pitch: CosyVoice API 支持 [0.5, 2.0],非默认值时才传
|
||||
if pitch and pitch != 1.0:
|
||||
input_payload["pitch"] = pitch
|
||||
# style 优先:显式传 style 时走 build_style_instruction();
|
||||
# 无 style 时回退到旧的 emotion → build_emotion_instruction() 逻辑(向后兼容)
|
||||
s = (style or "").strip().lower()
|
||||
instruction = ""
|
||||
if s:
|
||||
instruction = build_style_instruction(voice_id, s)
|
||||
if not instruction:
|
||||
norm_emotion = normalize_emotion(emotion)
|
||||
instruction = build_emotion_instruction(voice_id, norm_emotion)
|
||||
if instruction:
|
||||
input_payload["instruction"] = instruction
|
||||
# 语言 → language_hints 数组(仅取第一个元素生效);
|
||||
# 系统音色(非克隆/非 voice_id 中包含下划线以外的短 ID)仅传 zh/en,其他语言不传避免报错
|
||||
norm_lang = normalize_language(language)
|
||||
@@ -682,6 +789,8 @@ class CosyVoiceService:
|
||||
emotion: str = "",
|
||||
language: str = "zh",
|
||||
timeout: float = 120.0,
|
||||
style: str = "",
|
||||
pitch: float = 1.0,
|
||||
) -> SynthesizeResult:
|
||||
"""语音合成(同步非流式).
|
||||
|
||||
@@ -695,6 +804,8 @@ class CosyVoiceService:
|
||||
format: 输出格式(mp3/wav/pcm),空表示使用配置默认值
|
||||
speed: 语速(0.5-2.0),1.0 为正常速度
|
||||
volume: 音量(0-100),默认 50
|
||||
style: 语气风格(natural/sweet/excited/professional/news/livestream)
|
||||
pitch: 音调(0.5-2.0),1.0 为默认值
|
||||
timeout: 超时时间(秒),保留参数兼容
|
||||
|
||||
Returns:
|
||||
@@ -714,6 +825,8 @@ class CosyVoiceService:
|
||||
volume=volume,
|
||||
emotion=emotion,
|
||||
language=language,
|
||||
style=style,
|
||||
pitch=pitch,
|
||||
)
|
||||
|
||||
return SynthesizeResult(
|
||||
|
||||
@@ -146,6 +146,9 @@ class TTSWorkflowService:
|
||||
_meta = dict(job.metadata)
|
||||
_speed = float(_meta.get("speed", 1.0) or 1.0)
|
||||
_emotion = str(_meta.get("emotion", "") or "")
|
||||
_style = str(_meta.get("style", "") or "")
|
||||
_volume = int(_meta.get("volume", 50) or 50)
|
||||
_pitch = float(_meta.get("pitch", 1.0) or 1.0)
|
||||
_language = str(_meta.get("language", "zh-CN") or "zh-CN")
|
||||
submit_result = self.cosyvoice_service.submit_synthesize_task(
|
||||
text=job.input_text,
|
||||
@@ -153,6 +156,9 @@ class TTSWorkflowService:
|
||||
sample_rate=job.sample_rate,
|
||||
format=job.format,
|
||||
speed=_speed,
|
||||
style=_style,
|
||||
volume=_volume,
|
||||
pitch=_pitch,
|
||||
emotion=_emotion,
|
||||
language=_language,
|
||||
)
|
||||
@@ -290,6 +296,8 @@ class TTSWorkflowService:
|
||||
job_metadata = job.metadata or {}
|
||||
speed = float(job_metadata.get("speed", 1.0))
|
||||
volume = int(job_metadata.get("volume", 50))
|
||||
style = str(job_metadata.get("style", "") or "")
|
||||
pitch = float(job_metadata.get("pitch", 1.0))
|
||||
emotion = str(job_metadata.get("emotion", "") or "")
|
||||
language = str(job_metadata.get("language", "zh-CN") or "zh-CN")
|
||||
|
||||
@@ -299,7 +307,9 @@ class TTSWorkflowService:
|
||||
sample_rate=job.sample_rate,
|
||||
format=job.format,
|
||||
speed=speed,
|
||||
style=style,
|
||||
volume=volume,
|
||||
pitch=pitch,
|
||||
emotion=emotion,
|
||||
language=language,
|
||||
)
|
||||
@@ -415,6 +425,9 @@ class TTSWorkflowService:
|
||||
_seg_meta = job.metadata or {}
|
||||
_seg_speed = float(_seg_meta.get("speed", 1.0) or 1.0)
|
||||
_seg_emotion = str(_seg_meta.get("emotion", "") or "")
|
||||
_seg_style = str(_seg_meta.get("style", "") or "")
|
||||
_seg_volume = int(_seg_meta.get("volume", 50) or 50)
|
||||
_seg_pitch = float(_seg_meta.get("pitch", 1.0) or 1.0)
|
||||
|
||||
with ThreadPoolExecutor(max_workers=max_workers) as executor:
|
||||
future_to_idx = {}
|
||||
@@ -426,6 +439,9 @@ class TTSWorkflowService:
|
||||
sample_rate=job.sample_rate,
|
||||
format=job.format,
|
||||
speed=_seg_speed,
|
||||
style=_seg_style,
|
||||
volume=_seg_volume,
|
||||
pitch=_seg_pitch,
|
||||
emotion=_seg_emotion,
|
||||
language=_seg_meta.get("language", "zh-CN") or "zh-CN",
|
||||
)
|
||||
@@ -517,6 +533,8 @@ class TTSWorkflowService:
|
||||
job_metadata = job.metadata or {}
|
||||
speed = float(job_metadata.get("speed", 1.0))
|
||||
volume = int(job_metadata.get("volume", 50))
|
||||
style = str(job_metadata.get("style", "") or "")
|
||||
pitch = float(job_metadata.get("pitch", 1.0))
|
||||
emotion = str(job_metadata.get("emotion", "") or "")
|
||||
|
||||
# 分段文本(用于缺失段重新合成)
|
||||
@@ -551,7 +569,9 @@ class TTSWorkflowService:
|
||||
sample_rate=job.sample_rate,
|
||||
format=job.format,
|
||||
speed=speed,
|
||||
style=style,
|
||||
volume=volume,
|
||||
pitch=pitch,
|
||||
emotion=emotion,
|
||||
language=job.metadata.get("language", "zh-CN") if hasattr(job, "metadata") else "zh-CN",
|
||||
)
|
||||
|
||||
@@ -6,6 +6,7 @@ CI 增量映射:
|
||||
ai_avatar_cover_service 智能选帧
|
||||
"""
|
||||
|
||||
import importlib
|
||||
import os
|
||||
from unittest.mock import MagicMock, patch
|
||||
|
||||
@@ -166,6 +167,116 @@ def test_submit_synthesize_payload_cloned_voice_english_emotion():
|
||||
assert inp["instruction"] == "Speak in a sad tone."
|
||||
|
||||
|
||||
# ── style(语气风格,#2002)────────────────────────────────────────────
|
||||
|
||||
|
||||
def test_style_natural_omits_instruction():
|
||||
"""style=natural 不加 instruct,使用 CosyVoice 默认自然语气。"""
|
||||
captured: dict = {}
|
||||
svc = _make_service_with_captured_client(captured)
|
||||
|
||||
svc.submit_synthesize_task(text="你好", voice_id="myclone_voice", style="natural")
|
||||
|
||||
inp = captured["json"]["input"]
|
||||
assert "instruction" not in inp
|
||||
|
||||
|
||||
def test_style_sweet_cloned_voice_uses_chinese_instruction():
|
||||
"""克隆音色 + style=sweet → 中文自然语言指令。"""
|
||||
captured: dict = {}
|
||||
svc = _make_service_with_captured_client(captured)
|
||||
|
||||
svc.submit_synthesize_task(text="你好", voice_id="myclone_voice", style="sweet")
|
||||
|
||||
inp = captured["json"]["input"]
|
||||
assert inp["instruction"] == "用温柔甜美、亲切柔和的语气说话。"
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
("style", "expected_fragment"),
|
||||
[
|
||||
("excited", "兴奋"),
|
||||
("professional", "专业"),
|
||||
("news", "新闻"),
|
||||
("livestream", "直播"),
|
||||
],
|
||||
)
|
||||
def test_style_values_cloned_voice(style, expected_fragment):
|
||||
captured: dict = {}
|
||||
svc = _make_service_with_captured_client(captured)
|
||||
|
||||
svc.submit_synthesize_task(text="你好", voice_id="myclone_voice", style=style)
|
||||
|
||||
inp = captured["json"]["input"]
|
||||
assert expected_fragment in inp["instruction"]
|
||||
|
||||
|
||||
def test_style_system_voice_uses_emotion_mapping():
|
||||
"""系统白名单音色 + style=sweet → 严格中文 emotion 格式(映射到 sad)。"""
|
||||
captured: dict = {}
|
||||
svc = _make_service_with_captured_client(captured)
|
||||
|
||||
svc.submit_synthesize_task(text="你好", voice_id="longanyang", style="sweet")
|
||||
|
||||
inp = captured["json"]["input"]
|
||||
assert inp["instruction"] == "你说话的情感是sad。"
|
||||
|
||||
|
||||
def test_style_takes_priority_over_emotion():
|
||||
"""同时传 style 和 emotion 时以 style 为准。"""
|
||||
captured: dict = {}
|
||||
svc = _make_service_with_captured_client(captured)
|
||||
|
||||
svc.submit_synthesize_task(text="你好", voice_id="myclone_voice", style="excited", emotion="sad")
|
||||
|
||||
inp = captured["json"]["input"]
|
||||
assert "兴奋" in inp["instruction"]
|
||||
|
||||
|
||||
def test_unknown_style_ignored_falls_back_to_emotion():
|
||||
"""未知 style 值被忽略,回退到 emotion 逻辑。"""
|
||||
captured: dict = {}
|
||||
svc = _make_service_with_captured_client(captured)
|
||||
|
||||
svc.submit_synthesize_task(text="hi", voice_id="myclone_voice", style="nonexistent", emotion="sad")
|
||||
|
||||
inp = captured["json"]["input"]
|
||||
assert inp["instruction"] == "Speak in a sad tone."
|
||||
|
||||
|
||||
def test_pitch_passed_only_when_non_default():
|
||||
captured: dict = {}
|
||||
svc = _make_service_with_captured_client(captured)
|
||||
|
||||
svc.submit_synthesize_task(text="hi", voice_id="myclone_voice", pitch=1.5)
|
||||
|
||||
inp = captured["json"]["input"]
|
||||
assert inp["pitch"] == 1.5
|
||||
|
||||
|
||||
def test_pitch_omitted_at_default():
|
||||
captured: dict = {}
|
||||
svc = _make_service_with_captured_client(captured)
|
||||
|
||||
svc.submit_synthesize_task(text="hi", voice_id="myclone_voice", pitch=1.0)
|
||||
|
||||
inp = captured["json"]["input"]
|
||||
assert "pitch" not in inp
|
||||
|
||||
|
||||
def test_resolve_style_maps_emotion_to_style():
|
||||
"""旧 emotion 值通过 resolve_style 映射为 style。"""
|
||||
mod = importlib.import_module("packages.application.cosyvoice_service")
|
||||
|
||||
assert mod.resolve_style(emotion="happy") == "excited"
|
||||
assert mod.resolve_style(emotion="sad") == "sweet"
|
||||
assert mod.resolve_style(emotion="neutral") == "natural"
|
||||
assert mod.resolve_style(style="news") == "news"
|
||||
# style 优先
|
||||
assert mod.resolve_style(style="news", emotion="happy") == "news"
|
||||
assert mod.resolve_style() == ""
|
||||
|
||||
|
||||
# ── 对口型 TTS 直生分支 ─────────────────────────────────────────────────
|
||||
|
||||
|
||||
|
||||
@@ -101,6 +101,9 @@ class TestTTSPreviewEndpoint:
|
||||
text="你好世界",
|
||||
voice_id="longxiaochun",
|
||||
speed=1.0,
|
||||
style="",
|
||||
volume=50,
|
||||
pitch=1.0,
|
||||
emotion="",
|
||||
language="zh-CN",
|
||||
)
|
||||
@@ -148,6 +151,9 @@ class TestTTSPreviewEndpoint:
|
||||
text="测试",
|
||||
voice_id="v1",
|
||||
speed=1.5,
|
||||
style="",
|
||||
volume=50,
|
||||
pitch=1.0,
|
||||
emotion="",
|
||||
language="zh-CN",
|
||||
)
|
||||
@@ -334,6 +340,9 @@ class TestTTSPreviewEndpoint:
|
||||
text="克隆音色测试",
|
||||
voice_id="cosyvoice_actual_voice_123",
|
||||
speed=1.0,
|
||||
style="",
|
||||
volume=50,
|
||||
pitch=1.0,
|
||||
emotion="",
|
||||
language="zh-CN",
|
||||
)
|
||||
@@ -412,6 +421,9 @@ class TestTTSPreviewEndpoint:
|
||||
text="预设音色测试",
|
||||
voice_id="longxiaoxia_v3",
|
||||
speed=1.0,
|
||||
style="",
|
||||
volume=50,
|
||||
pitch=1.0,
|
||||
emotion="",
|
||||
language="zh-CN",
|
||||
)
|
||||
|
||||
Reference in New Issue
Block a user