Files
xiaoxia-saas/packages/application/cosyvoice_service.py
T
xiaoxia a9f596fca3
CI/CD Pipeline / Check push changed paths (pull_request) Has been skipped
CI/CD Pipeline / Build Staging API Image (pull_request) Has been skipped
CI/CD Pipeline / Build Staging Web Image (pull_request) Has been skipped
CI/CD Pipeline / Dedup Check - skip PR tests when covered by push pipeline (pull_request) Successful in 3s
CI/CD Pipeline / Check if frontend-only change (pull_request) Successful in 3s
CI/CD Pipeline / Build Staging Worker Image (pull_request) Has been skipped
CI/CD Pipeline / Validate - Style (pull_request) Has been skipped
CI/CD Pipeline / Validate - Security (pull_request) Has been skipped
CI/CD Pipeline / Validate - Python (mypy + alembic) (pull_request) Has been skipped
CI/CD Pipeline / Integration Tests (pull_request) Has been skipped
CI/CD Pipeline / Unit Tests (pull_request) Has been skipped
CI/CD Pipeline / Frontend Lint (pull_request) Has been skipped
CI/CD Pipeline / Frontend Unit Tests (pull_request) Has been skipped
CI/CD Pipeline / Retag skipped Staging API Image (pull_request) Has been skipped
CI/CD Pipeline / Retag skipped Staging Web Image (pull_request) Has been skipped
CI/CD Pipeline / Retag skipped Staging Worker Image (pull_request) Has been skipped
CI/CD Pipeline / PR Build API Image (pull_request) Successful in 1m2s
Preview Deploy / Deploy Preview Environment (pull_request) Successful in 2m5s
PR Automation / Auto Approve on CI Green (pull_request) Successful in 3m19s
CI/CD Pipeline / Build Production API Image (pull_request) Has been skipped
CI/CD Pipeline / Build Production Web Image (pull_request) Has been skipped
CI/CD Pipeline / Deploy Staging (Watchtower auto-deploy) (pull_request) Has been skipped
CI/CD Pipeline / Build Production Worker Image (pull_request) Has been skipped
PR Automation / Auto Merge on CI Green + Approved (pull_request) Has been skipped
CI/CD Pipeline / Deploy Production (pull_request) Has been skipped
CI/CD Pipeline / Staging E2E Tests (pull_request) Has been skipped
CI/CD Pipeline / Staging API Integration Tests (pull_request) Has been skipped
CI/CD Pipeline / ACR Image Cleanup (pull_request) Has been skipped
CI/CD Pipeline / Production Browser E2E (pull_request) Has been skipped
CI/CD Pipeline / Canary Release to Production (pull_request) Has been skipped
CI/CD Pipeline / PR Build Web Image (pull_request) Successful in 5m58s
AI Code Review / AI Code Review (pull_request) Successful in 6m48s
CI/CD Pipeline / PR Build Worker Image (pull_request) Successful in 8m10s
CI/CD Pipeline / CI Gate (pull_request) Successful in 1s
CI/CD Pipeline / Dedup Check - skip PR tests when covered by push pipeline (push) Successful in 0s
CI/CD Pipeline / Check if frontend-only change (push) Has been skipped
CI/CD Pipeline / Check push changed paths (push) Successful in 2s
CI/CD Pipeline / Frontend Lint (push) Has been skipped
CI/CD Pipeline / PR Build API Image (push) Has been skipped
CI/CD Pipeline / PR Build Web Image (push) Has been skipped
CI/CD Pipeline / PR Build Worker Image (push) Has been skipped
CI/CD Pipeline / Build Staging Web Image (push) Successful in 2m33s
CI/CD Pipeline / Frontend Unit Tests (push) Successful in 3m43s
CI/CD Pipeline / Build Staging API Image (push) Successful in 7m32s
CI/CD Pipeline / Validate - Style (push) Successful in 36m8s
CI/CD Pipeline / Unit Tests (push) Successful in 41m59s
CI/CD Pipeline / Integration Tests (push) Successful in 42m40s
CI/CD Pipeline / Build Staging Worker Image (push) Successful in 46m4s
CI/CD Pipeline / Retag skipped Staging API Image (push) Has been skipped
CI/CD Pipeline / Retag skipped Staging Web Image (push) Has been skipped
CI/CD Pipeline / Retag skipped Staging Worker Image (push) Has been skipped
CI/CD Pipeline / Validate - Python (mypy + alembic) (push) Successful in 49m6s
CI/CD Pipeline / Deploy Staging (Watchtower auto-deploy) (push) Successful in 7m56s
CI/CD Pipeline / Staging API Integration Tests (push) Successful in 3m36s
CI/CD Pipeline / Staging E2E Tests (push) Failing after 4m47s
CI/CD Pipeline / ACR Image Cleanup (push) Successful in 20m52s
CI/CD Pipeline / Validate - Security (push) Successful in 2h17m39s
CI/CD Pipeline / Build Production API Image (push) Has been skipped
CI/CD Pipeline / Build Production Web Image (push) Has been skipped
CI/CD Pipeline / Build Production Worker Image (push) Has been skipped
CI/CD Pipeline / CI Gate (push) Has been skipped
CI/CD Pipeline / Deploy Production (push) Has been skipped
CI/CD Pipeline / Production Browser E2E (push) Has been skipped
CI/CD Pipeline / Canary Release to Production (push) Has been skipped
feat(tts): style语气风格参数全链路接入 + 语速/音量/音调透传修复 (#2005)
Co-authored-by: xiaoxia <dev@xiaoxiajianji.com>
Co-committed-by: xiaoxia <dev@xiaoxiajianji.com>
2026-09-21 04:23:07 +08:00

982 lines
36 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
"""CosyVoice 语音服务 — 适配阿里云百炼 DashScope API.
封装阿里云百炼 CosyVoice 语音合成 API,提供:
- 预置音色列表查询
- 音色克隆(提交 + 轮询状态)
- 语音合成(同步非流式调用)
API 文档:
- 音色克隆: https://help.aliyun.com/document_detail/3027318.html
- 语音合成: https://help.aliyun.com/zh/model-studio/cosyvoice-tts-http-api
"""
from __future__ import annotations
import logging
import time
from collections.abc import Callable
from dataclasses import dataclass
from typing import Any, Optional
import httpx
from packages.domain.preset_voices import PresetVoice, get_preset_voices
from packages.shared.config import get_shared_settings
logger = logging.getLogger(__name__)
# ── Style(语气风格)→ CosyVoice instruct 自然语言指令 ──
# 前端 PR#2002 传 6 种 style:natural/sweet/excited/professional/news/livestream。
# style 是新的统一参数;emotion 为 deprecated 兼容别名,内部映射为 style。
# 参考:https://help.aliyun.com/zh/model-studio/cosyvoice-voice-list
STYLE_INSTRUCTION_MAP: dict[str, str] = {
"natural": "用自然、平和的语气说话。",
"sweet": "用温柔甜美、亲切柔和的语气说话。",
"excited": "用兴奋、激动的语气说话。",
"professional": "用专业、正式的语气说话。",
"news": "用新闻播报的语气说话。",
"livestream": "用直播解说的语气说话。",
}
# 有效 style 值集合(供 schema / 校验使用)
VALID_STYLES: frozenset[str] = frozenset(STYLE_INSTRUCTION_MAP.keys())
# ── 旧 emotion → 新 style 兼容映射(方案 B:统一 style,emotion deprecated)──
_EMOTION_TO_STYLE: dict[str, str] = {
"neutral": "natural",
"happy": "excited",
"sad": "sweet",
"angry": "excited",
"surprised": "excited",
"fearful": "sweet",
"disgusted": "natural",
}
# 严格格式系统音色:style → emotion 回退(用于无法使用自由文本指令的音色)
_STYLE_TO_EMOTION: dict[str, str] = {
"natural": "neutral",
"sweet": "sad",
"excited": "happy",
"professional": "neutral",
# news / livestream 无直接对应 emotion,特殊处理
}
# ── 旧 emotion 映射表(deprecated,保留以兼容历史数据)──
EMOTION_MAP: dict[str, str] = {
# ── 7 种标准英文枚举(CosyVoice v3 官方支持的情感值)──
"neutral": "neutral",
"happy": "happy",
"sad": "sad",
"angry": "angry",
"surprised": "surprised",
"fearful": "fearful",
"disgusted": "disgusted",
# ── 前端中文 7 标签(P1:前端已扩展为这 7 个中文选项)──
"中立": "neutral",
"开心": "happy",
"难过": "sad",
"生气": "angry",
"惊讶": "surprised",
"恐惧": "fearful",
"厌恶": "disgusted",
# ── 常见中文别名 ──
"自然": "neutral",
"愉快": "happy",
"高兴": "happy",
"快乐": "happy",
"兴奋": "happy", # 旧 excited → 映射到最接近的 happy
"悲伤": "sad",
"愤怒": "angry",
"惊奇": "surprised",
"吃惊": "surprised",
"害怕": "fearful",
"讨厌": "disgusted",
# ── 灵应派任务指定的中文别名(中性/伤心/愤怒 等)──
"中性": "neutral",
"伤心": "sad",
# ── 旧英文 4 枚举兼容(natural/excited/calm/friendly 归并到最接近的标准值)──
"natural": "neutral",
"excited": "happy",
"calm": "neutral",
"friendly": "happy",
}
# ── 支持 emotion Instruct 的 v3-flash 系统音色白名单(官方音色列表标注"Instruct:支持"且支持情感值)──
# 这些音色的 instruction 必须使用中文固定格式 "你说话的情感是<emotion>。";
# longanhuan_v3 虽然 Instruct 支持,但只支持方言 instruct(请用<方言>表达。),不支持 emotion,故不列入。
_SYSTEM_VOICES_WITH_EMOTION_INSTRUCT: frozenset[str] = frozenset(
{
"longanyang", # 龙安洋(标杆音色)
"longanhuan", # 龙安欢
"longhuhu_v3", # 龙呼呼
}
)
def _is_cloned_voice(voice_id: str) -> bool:
"""判断一个 voice_id 是否为克隆/设计音色(非系统预置音色)。
所有以 "long"/"loong" 开头的是系统预置音色(longxiaochun_v3/longanyang/loongabby_v3 等),
其余视为用户克隆音色/设计音色,支持任意中英文自然语言 instruction。
"""
if not voice_id:
return False
v = voice_id.lower()
return not (v.startswith("long") or v.startswith("loong"))
def build_emotion_instruction(voice_id: str, emotion_enum: str) -> str:
"""根据 voice 类型构造符合官方规范的 emotion instruction.
- 克隆/设计音色(非 long*/loong* 前缀):英文自然语言 "Speak in a {emotion} tone.",
DashScope 对克隆音色允许任意自然语言指令。
- 系统音色中 emotion-instruct 可用的(longanyang/longanhuan/longhuhu_v3):
严格按官方中文固定格式 "你说话的情感是{emotion}。",结尾中文句号不可省。
- 其他系统音色(含默认 longxiaochun_v3 等绝大多数 v3 系统音色):官方不支持 Instruct,
返回空串(调用方据此不传 instruction,避免被 API 报错或忽略)。
Args:
voice_id: CosyVoice voice 参数
emotion_enum: 已归一化的 7 种英文枚举之一(neutral/happy/sad/...)
Returns:
拼接好的 instruction 字符串;不支持时返回空串
"""
if not emotion_enum:
return ""
if _is_cloned_voice(voice_id):
return f"Speak in a {emotion_enum} tone."
if voice_id in _SYSTEM_VOICES_WITH_EMOTION_INSTRUCT:
return f"你说话的情感是{emotion_enum}。"
return ""
def resolve_style(style: str = "", emotion: str = "") -> str:
"""统一解析 style 参数(方案 B).
- style 有值且合法:直接使用(style 优先级最高)。
- style 为空但 emotion 有值:将旧 emotion 归一化后映射为 style。
- 两者皆空:返回空串(调用方不传 instruction)。
Args:
style: 新的语气风格(natural/sweet/excited/professional/news/livestream)
emotion: 旧的情绪参数(deprecated,内部映射为 style)
Returns:
解析后的 style 字符串;无需 instruct 时返回空串
"""
s = (style or "").strip().lower()
if s:
if s in VALID_STYLES:
return s
logger.warning("未知的 style 值 %r,忽略 style 参数", style)
# 回退:emotion → style
norm = normalize_emotion(emotion)
if not norm:
return ""
mapped = _EMOTION_TO_STYLE.get(norm)
if not mapped:
logger.warning("emotion %r 无法映射到 style,跳过 instruct", norm)
return mapped or ""
def build_style_instruction(voice_id: str, style: str) -> str:
"""根据 voice 类型构造 style instruction.
- natural:返回空串(不额外加 instruct,使用 CosyVoice 默认自然语气)。
- 克隆/设计音色:使用中文自然语言指令(DashScope 允许任意自然语言)。
- 系统音色中支持 emotion instruct 的白名单音色:
若 style 可映射到 emotion,用严格格式 "你说话的情感是<emotion>。";
news/livestream 尝试直接用中文 instruct(部分音色支持自由文本)。
- 其他系统音色:返回空串。
Args:
voice_id: CosyVoice voice 参数
style: 已通过 resolve_style() 解析的 style 值
Returns:
拼接好的 instruction 字符串;无需 instruct 时返回空串
"""
if not style or style == "natural":
return ""
# 克隆音色:直接使用中文自然语言指令
if _is_cloned_voice(voice_id):
return STYLE_INSTRUCTION_MAP.get(style, "")
# 系统音色白名单:优先映射到严格 emotion 格式
if voice_id in _SYSTEM_VOICES_WITH_EMOTION_INSTRUCT:
emotion_val = _STYLE_TO_EMOTION.get(style)
if emotion_val:
return f"你说话的情感是{emotion_val}。"
# news/livestream 无 emotion 对应,尝试自由中文 instruct
desc = STYLE_INSTRUCTION_MAP.get(style, "")
if desc:
logger.info("音色 %s 使用自由文本 style instruct: %s", voice_id, desc)
return desc
return ""
def normalize_emotion(emotion: str) -> str:
"""将前端情绪值归一化为 CosyVoice v3 官方英文枚举,用于拼入 instruction.
支持:
- 7 种标准英文枚举(neutral/happy/sad/angry/surprised/fearful/disgusted);
- 前端中文 7 标签(中立/开心/难过/生气/惊讶/恐惧/厌恶);
- 常见中文别名与旧英文 4 枚举兼容值(natural/excited/calm/friendly);
- 大小写不敏感。
返回:始终返回 7 种英文枚举之一;空串/None 返回空串(调用方据此不传 instruction,
CosyVoice 按默认自然情绪合成);未知值记录 warning 并默认 "neutral"。
"""
if not emotion:
return ""
key = emotion.strip()
if not key:
return ""
# 大小写不敏感:先按原 key 查,再按 lower 查
mapped = EMOTION_MAP.get(key) or EMOTION_MAP.get(key.lower())
if mapped:
return mapped
# 未识别的情绪值:默认 neutral,保证合成能正常进行
logger.warning("未知的 emotion 值 %r,默认使用 neutral", emotion)
return "neutral"
# 系统音色仅支持中文/英文(language_hints 取值)
_SYSTEM_VOICE_LANGS = {"zh", "en"}
def normalize_language(language: str) -> str:
"""将前端语言代码归一化为 CosyVoice language_hints 短码(zh/en/...).
支持 zh-CN/zh_CN/zh/en-US/en/en_US/en-GB 等常见形式;
空值默认 zh;未知值返回短码(CosyVoice 接受时生效)。
"""
if not language:
return "zh"
# 取第一段(zh-CN → zh, en-US → en)
lang = language.strip().split("-")[0].split("_")[0].lower()
return lang
class CosyVoiceError(Exception):
"""CosyVoice API 调用异常。"""
pass
class CosyVoiceTimeoutError(CosyVoiceError):
"""CosyVoice API 超时。"""
pass
class CosyVoiceAuthError(CosyVoiceError):
"""CosyVoice API 认证失败。"""
pass
@dataclass
class CloneResult:
"""音色克隆结果。"""
voice_id: str
request_id: str = ""
@dataclass
class SynthesizeResult:
"""语音合成结果。"""
audio_url: str
duration: float = 0.0
file_size: int = 0
request_id: str = ""
class CosyVoiceService:
"""CosyVoice 语音服务.
封装阿里云百炼 CosyVoice API,提供音色克隆和语音合成功能.
接口总览:
- 音色克隆: POST /services/audio/tts/customization (model=voice-enrollment)
- action=create_voice: 创建克隆音色,返回 voice_id(状态 DEPLOYING)
- action=query_voice: 查询音色状态(DEPLOYING / OK / UNDEPLOYED)
- 语音合成: POST /services/audio/tts/SpeechSynthesizer (model=cosyvoice-v3-flash)
- 非流式: 同步返回音频 URL
使用示例:
service = CosyVoiceService(
api_key="your-api-key",
base_url="https://dashscope.aliyuncs.com/api/v1",
model="cosyvoice-v3-flash",
)
# 音色克隆
result = service.clone_voice(audio_url="https://example.com/audio.mp3")
# 语音合成
result = service.synthesize_speech(text="你好世界", voice_id="longxiaochun_v3")
"""
# 音色状态轮询配置
CLONE_POLL_INTERVAL = 5.0 # 秒
CLONE_MAX_POLL_ATTEMPTS = 60 # 最多轮询 60 次(5分钟)
# 重试配置
MAX_RETRIES = 3
RETRY_BACKOFF = 1.0 # 秒,指数退避基数
def __init__(
self,
api_key: str = "",
base_url: str = "",
model: str = "",
clone_model: str = "",
http_client: Optional[httpx.Client] = None,
audio_url_signer: Optional[Callable[[str], str]] = None,
) -> None:
"""初始化 CosyVoice 服务.
Args:
api_key: DashScope API Key,为空时从配置读取
base_url: DashScope API Base URL,为空时从配置读取
model: 语音合成模型名称,为空时从配置读取
clone_model: 音色克隆模型名称,为空时从配置读取
http_client: 可选的 HTTP 客户端(用于测试注入)
audio_url_signer: 可选的音频URL预签名函数,签名式 fn(url) -> str.
用于私有 bucket 下,将裸 URL 转为预签名 URL,
确保 CosyVoice 服务器能下载参考音频.
"""
settings = get_shared_settings()
self._api_key = api_key or settings.cosyvoice_api_key
self._base_url = base_url or settings.cosyvoice_base_url
self._model = model or settings.cosyvoice_model
self._clone_model = clone_model or getattr(settings, "cosyvoice_clone_model", "voice-enrollment")
self._audio_url_signer = audio_url_signer
# base_url 规范化:去掉末尾的路径残留(兼容旧版配置)
# 旧版 .env 模板中 base_url 包含 /services/aigc/text2audio 完整路径,
# 新版只需 /api/v1,具体路径由代码拼接。这里自动修正,避免配置滞后导致418。
if "/services/aigc/text2audio" in self._base_url:
old_url = self._base_url
# 截取到 /api/v1 为止
idx = self._base_url.find("/api/v1")
if idx >= 0:
self._base_url = self._base_url[: idx + len("/api/v1")]
logger.warning(
"[CosyVoice Config] base_url包含旧版text2audio路径,已自动修正: " "%s -> %s",
old_url,
self._base_url,
)
self._client = http_client or httpx.Client(
timeout=httpx.Timeout(60.0, connect=10.0),
)
self._owns_client = http_client is None
# 启动时打印配置(脱敏),方便排查环境变量覆盖问题
if self._owns_client:
masked_key = ""
if self._api_key:
if len(self._api_key) > 8:
masked_key = f"{self._api_key[:4]}...{self._api_key[-4:]}"
else:
masked_key = "***"
logger.info(
"[CosyVoice Config] 初始化配置: "
"model=%s, base_url=%s, default_voice=%s, "
"sample_rate=%d, format=%s, api_key=%s",
self._model,
self._base_url,
getattr(settings, "cosyvoice_voice", "(unset)"),
settings.cosyvoice_sample_rate,
settings.cosyvoice_format,
masked_key or "(empty)",
)
def __enter__(self) -> CosyVoiceService:
return self
def __exit__(self, *args: Any) -> None:
self.close()
def close(self) -> None:
"""关闭 HTTP 客户端。"""
if self._owns_client and self._client:
self._client.close()
# ── 预置音色 ─────────────────────────────────────────
def list_preset_voices(self) -> list[PresetVoice]:
"""获取预置音色列表.
Returns:
预置音色列表
"""
return get_preset_voices()
# ── 音色克隆 ─────────────────────────────────────────
def submit_clone_task(
self,
audio_url: str,
voice_name: str = "",
language: str = "zh-CN",
target_model: str = "",
) -> dict:
"""提交音色克隆任务(非阻塞).
调用百炼 voice-enrollment API 创建克隆音色.
创建后音色状态为 DEPLOYING,需通过 query_voice_status 轮询直到 OK.
Args:
audio_url: 参考音频 URL(必须公网可访问)
voice_name: 音色名称前缀(字母数字,最多10字符)
language: 语言代码(zh-CN 会转换为 zh)
target_model: 目标合成模型,默认使用当前 model
Returns:
dict: {"voice_id": str, "status": str, "request_id": str}
voice_id 非空,status 通常为 DEPLOYING
Raises:
CosyVoiceError: API 调用失败
CosyVoiceAuthError: 认证失败
ValueError: 参数无效
"""
if not audio_url:
raise ValueError("audio_url 不能为空")
if not self._api_key:
raise CosyVoiceAuthError("CosyVoice API Key 未配置")
# voice_name 作为 prefix,限制字母数字,最多10字符
# 不符合要求的做清洗
prefix = self._sanitize_prefix(voice_name) if voice_name else "clone"
# 语言转换:zh-CN → zh,保留 ISO 639-1 格式
lang_code = language.split("-")[0].lower() if language else "zh"
target = target_model or self._model
# 如果配置了 audio_url_signer,对音频URL做预签名
# (私有 bucket 下 CosyVoice 服务器无法直接访问裸 URL)
signed_audio_url = audio_url
if self._audio_url_signer:
try:
signed_audio_url = self._audio_url_signer(audio_url)
logger.info("音频URL已预签名: original=%s signed_prefix=%s", audio_url[:80], signed_audio_url[:80])
except Exception as e:
logger.warning("音频URL预签名失败,使用原始URL: %s", e)
payload = {
"model": self._clone_model,
"input": {
"action": "create_voice",
"target_model": target,
"prefix": prefix,
"url": signed_audio_url,
"language_hints": [lang_code],
},
}
response = self._call_api(
method="POST",
path="/services/audio/tts/customization",
json=payload,
timeout=60.0,
)
output = response.get("output", {})
voice_id = output.get("voice_id", "")
status = output.get("status", "DEPLOYING")
request_id = response.get("request_id", "")
if not voice_id:
raise CosyVoiceError(f"CosyVoice API 未返回 voice_id: {response}")
return {
"voice_id": voice_id,
"status": status,
"request_id": request_id,
}
def query_voice_status(self, voice_id: str) -> dict:
"""查询音色状态(单次查询,不轮询).
Args:
voice_id: 音色 ID
Returns:
dict: {"status": str, "target_model": str, "gmt_create": str,
"gmt_modified": str, "resource_link": str}
status 为 DEPLOYING / OK / UNDEPLOYED
Raises:
CosyVoiceError: API 调用失败
CosyVoiceAuthError: 认证失败
"""
if not self._api_key:
raise CosyVoiceAuthError("CosyVoice API Key 未配置")
if not voice_id:
raise ValueError("voice_id 不能为空")
payload = {
"model": self._clone_model,
"input": {
"action": "query_voice",
"voice_id": voice_id,
},
}
response = self._call_api(
method="POST",
path="/services/audio/tts/customization",
json=payload,
timeout=30.0,
)
output = response.get("output", {})
return {
"status": output.get("status", ""),
"target_model": output.get("target_model", ""),
"gmt_create": output.get("gmt_create", ""),
"gmt_modified": output.get("gmt_modified", ""),
"resource_link": output.get("resource_link", ""),
}
def check_task_status(self, task_id: str) -> dict:
"""查询克隆任务状态(兼容旧接口,实际用 voice_id 查询).
为了兼容旧代码,task_id 参数名保留,但实际传的是 voice_id.
Args:
task_id: 音色 ID(兼容旧接口名)
Returns:
dict: {"status": str, "voice_id": str, "message": str}
"""
result = self.query_voice_status(task_id)
return {
"status": result["status"],
"voice_id": task_id,
"message": "",
}
def poll_clone_task(self, voice_id: str, timeout: float = 300.0) -> dict:
"""轮询音色克隆状态直到完成或超时.
供 Celery 后台任务调用,轮询直到状态变为 OK 或 UNDEPLOYED.
Args:
voice_id: 音色 ID
timeout: 超时时间(秒),默认 300
Returns:
dict: {"voice_id": str}
Raises:
CosyVoiceError: 任务失败(状态 UNDEPLOYED)
CosyVoiceTimeoutError: 超时
"""
start_time = time.time()
attempts = 0
while attempts < self.CLONE_MAX_POLL_ATTEMPTS:
elapsed = time.time() - start_time
if elapsed > timeout:
raise CosyVoiceTimeoutError(f"音色克隆任务超时({timeout}秒): voice_id={voice_id}")
result = self.query_voice_status(voice_id)
status = result.get("status", "").upper()
if status == "OK":
return {"voice_id": voice_id}
elif status == "UNDEPLOYED":
raise CosyVoiceError(f"音色克隆任务失败(审核未通过): voice_id={voice_id}")
elif status in ("DEPLOYING", "PENDING", "PROCESSING", ""):
# 继续轮询
time.sleep(self.CLONE_POLL_INTERVAL)
attempts += 1
else:
logger.warning("未知的音色状态: %s (voice_id=%s)", status, voice_id)
time.sleep(self.CLONE_POLL_INTERVAL)
attempts += 1
raise CosyVoiceTimeoutError(f"音色克隆任务轮询次数超限: voice_id={voice_id}")
def clone_voice(
self,
audio_url: str,
voice_name: str = "",
language: str = "zh-CN",
timeout: float = 300.0,
target_model: str = "",
) -> CloneResult:
"""克隆音色(阻塞,直到完成或超时).
提交音色克隆到百炼 API,并轮询直到状态变为 OK 或超时.
Args:
audio_url: 参考音频 URL(必须公网可访问)
voice_name: 音色名称前缀
language: 语言代码
timeout: 超时时间(秒)
target_model: 目标合成模型
Returns:
CloneResult: 克隆结果,包含 voice_id
Raises:
CosyVoiceError: API 调用失败或克隆失败
CosyVoiceTimeoutError: 超时
CosyVoiceAuthError: 认证失败
ValueError: 参数无效
"""
submit_result = self.submit_clone_task(
audio_url=audio_url,
voice_name=voice_name,
language=language,
target_model=target_model,
)
voice_id = submit_result["voice_id"]
request_id = submit_result["request_id"]
# 如果创建时已经是 OK 状态,直接返回
if submit_result.get("status", "").upper() == "OK":
return CloneResult(voice_id=voice_id, request_id=request_id)
# 否则轮询
result = self.poll_clone_task(voice_id, timeout=timeout)
return CloneResult(voice_id=result["voice_id"], request_id=request_id)
# ── 语音合成 ─────────────────────────────────────────
def submit_synthesize_task(
self,
text: str,
voice_id: str = "",
sample_rate: int = 0,
format: str = "",
speed: float = 1.0,
volume: int = 50,
emotion: str = "",
language: str = "zh",
style: str = "",
pitch: float = 1.0,
) -> dict:
"""提交语音合成任务(同步非流式,直接返回结果).
CosyVoice SpeechSynthesizer 非流式接口是同步的,
调用后直接返回音频 URL. 此方法保持与旧接口兼容.
Args:
text: 要合成的文本
voice_id: 音色 ID(预置音色或克隆音色)
sample_rate: 采样率(Hz),0 表示使用配置默认值
format: 输出格式(mp3/wav/pcm),空表示使用配置默认值
speed: 语速(0.5-2.0),1.0 为正常速度
volume: 音量(0-100),默认 50
style: 语气风格(natural/sweet/excited/professional/news/livestream),
新的统一参数;优先级高于 emotion
pitch: 音调(0.5-2.0),1.0 为默认值
emotion: 【deprecated】旧情绪参数,内部通过 resolve_style() 映射为 style
language: 语言代码(zh/en 等,默认 zh;系统音色仅 zh/en 传 language_hints)
Returns:
dict: {"audio_url": str, "request_id": str,
"duration": float, "file_size": int}
Raises:
CosyVoiceError: API 调用失败
CosyVoiceAuthError: 认证失败
ValueError: 参数无效
"""
if not text:
raise ValueError("text 不能为空")
if not voice_id:
raise ValueError("voice_id 不能为空")
if not self._api_key:
raise CosyVoiceAuthError("CosyVoice API Key 未配置")
settings = get_shared_settings()
input_payload: dict[str, Any] = {
"text": text,
"voice": voice_id,
"format": format or settings.cosyvoice_format,
"sample_rate": sample_rate or settings.cosyvoice_sample_rate,
"rate": speed,
"volume": volume,
}
# pitch: CosyVoice API 支持 [0.5, 2.0],非默认值时才传
if pitch and pitch != 1.0:
input_payload["pitch"] = pitch
# style 优先:显式传 style 时走 build_style_instruction();
# 无 style 时回退到旧的 emotion → build_emotion_instruction() 逻辑(向后兼容)
s = (style or "").strip().lower()
instruction = ""
if s:
instruction = build_style_instruction(voice_id, s)
if not instruction:
norm_emotion = normalize_emotion(emotion)
instruction = build_emotion_instruction(voice_id, norm_emotion)
if instruction:
input_payload["instruction"] = instruction
# 语言 → language_hints 数组(仅取第一个元素生效);
# 系统音色(非克隆/非 voice_id 中包含下划线以外的短 ID)仅传 zh/en,其他语言不传避免报错
norm_lang = normalize_language(language)
# 简单判断:克隆音色一般是长 voice_id(包含 dash 或长度>20),对其不做语言限制;
# 系统音色(如 longxiaochun_v3)仅 zh/en 传 language_hints
is_system_voice = ("_" in voice_id or voice_id.startswith("long")) and len(voice_id) < 32
if (is_system_voice and norm_lang in _SYSTEM_VOICE_LANGS) or (not is_system_voice):
input_payload["language_hints"] = [norm_lang]
payload = {"model": self._model, "input": input_payload}
response = self._call_api(
method="POST",
path="/services/audio/tts/SpeechSynthesizer",
json=payload,
timeout=120.0,
)
output = response.get("output", {})
audio = output.get("audio", {})
audio_url = audio.get("url", "")
request_id = response.get("request_id", "")
if not audio_url:
raise CosyVoiceError(f"CosyVoice API 未返回 audio_url: {response}")
# DashScope 返回 http://,统一升级为 https://
if audio_url.startswith("http://"):
audio_url = audio_url.replace("http://", "https://", 1)
return {
"task_id": "", # 同步接口无 task_id,兼容旧接口
"audio_url": audio_url,
"duration": 0.0, # 同步接口不返回 duration
"file_size": 0, # 同步接口不返回 file_size
"request_id": request_id,
}
def poll_synthesize_task(self, task_id: str, timeout: float = 120.0) -> dict:
"""轮询合成任务(同步接口无需轮询,保留兼容).
CosyVoice SpeechSynthesizer 非流式接口是同步的,
此方法仅为保持接口兼容,实际调用时 task_id 应该为空.
Raises:
CosyVoiceError: 同步接口无需轮询
"""
raise CosyVoiceError("CosyVoice 非流式合成接口是同步的,无需轮询. " "请直接使用 submit_synthesize_task().")
def synthesize_speech(
self,
text: str,
voice_id: str = "",
sample_rate: int = 0,
format: str = "",
speed: float = 1.0,
volume: int = 50,
emotion: str = "",
language: str = "zh",
timeout: float = 120.0,
style: str = "",
pitch: float = 1.0,
) -> SynthesizeResult:
"""语音合成(同步非流式).
调用百炼 CosyVoice SpeechSynthesizer 非流式接口,
直接返回合成音频 URL.
Args:
text: 要合成的文本
voice_id: 音色 ID(预置音色或克隆音色)
sample_rate: 采样率(Hz),0 表示使用配置默认值
format: 输出格式(mp3/wav/pcm),空表示使用配置默认值
speed: 语速(0.5-2.0),1.0 为正常速度
volume: 音量(0-100),默认 50
style: 语气风格(natural/sweet/excited/professional/news/livestream)
pitch: 音调(0.5-2.0),1.0 为默认值
timeout: 超时时间(秒),保留参数兼容
Returns:
SynthesizeResult: 合成结果,包含 audio_url
Raises:
CosyVoiceError: API 调用失败
CosyVoiceAuthError: 认证失败
ValueError: 参数无效
"""
result = self.submit_synthesize_task(
text=text,
voice_id=voice_id,
sample_rate=sample_rate,
format=format,
speed=speed,
volume=volume,
emotion=emotion,
language=language,
style=style,
pitch=pitch,
)
return SynthesizeResult(
audio_url=result["audio_url"],
duration=result.get("duration", 0.0),
file_size=result.get("file_size", 0),
request_id=result.get("request_id", ""),
)
# ── 内部方法 ─────────────────────────────────────────
def _sanitize_prefix(self, name: str) -> str:
"""清洗音色名称为合法的 prefix(字母数字,最多10字符).
Args:
name: 原始音色名称
Returns:
清洗后的 prefix
"""
# 只保留字母和数字
import re
cleaned = re.sub(r"[^a-zA-Z0-9]", "", name)
# 最多10字符
cleaned = cleaned[:10]
# 如果清洗后为空,用默认值
if not cleaned:
cleaned = "clone"
return cleaned
def _call_api(
self,
method: str,
path: str,
json: Optional[dict] = None,
timeout: float = 30.0,
) -> dict:
"""调用 DashScope API.
支持重试和错误处理.
Args:
method: HTTP 方法(GET/POST)
path: API 路径(以 / 开头)
json: 请求体
timeout: 超时时间(秒)
Returns:
API 响应字典
Raises:
CosyVoiceError: API 调用失败
CosyVoiceAuthError: 认证失败
CosyVoiceTimeoutError: 超时
"""
url = f"{self._base_url}{path}"
headers = {
"Authorization": f"Bearer {self._api_key}",
"Content-Type": "application/json",
}
# DEBUG: 打印完整请求信息,用于排查418错误
import json as json_lib
safe_headers = {k: v for k, v in headers.items()}
if "Authorization" in safe_headers:
token = safe_headers["Authorization"]
if len(token) > 20:
safe_headers["Authorization"] = token[:13] + "..." + token[-4:]
logger.info(
"[CosyVoice Debug] 请求详情: " "method=%s, url=%s, headers=%s, body=%s",
method,
url,
safe_headers,
json_lib.dumps(json, ensure_ascii=False) if json else "None",
)
last_error: Optional[Exception] = None
for attempt in range(self.MAX_RETRIES):
try:
response = self._client.request(
method=method,
url=url,
headers=headers,
json=json,
timeout=timeout,
)
# DEBUG: 打印响应状态和完整响应体
logger.info(
"[CosyVoice Debug] 响应详情: " "status=%d, body=%s",
response.status_code,
response.text[:2000], # 最多2000字符,避免日志过大
)
# 处理响应
if response.status_code == 200:
return response.json()
elif response.status_code in (401, 403):
raise CosyVoiceAuthError(f"CosyVoice API 认证失败: HTTP {response.status_code}")
elif response.status_code == 400:
# 客户端错误,不重试
body_text = response.text
try:
body = response.json()
code = body.get("code", "")
message = body.get("message", "")
raise CosyVoiceError(f"CosyVoice API 参数错误: HTTP 400, " f"code={code}, message={message}")
except ValueError as _e:
raise CosyVoiceError(f"CosyVoice API 调用失败: HTTP 400, body={body_text}") from _e
elif response.status_code >= 500:
# 服务端错误,可重试
last_error = CosyVoiceError(f"CosyVoice API 服务端错误: HTTP {response.status_code}")
logger.warning(
"CosyVoice API 失败 (尝试 %d/%d): HTTP %d",
attempt + 1,
self.MAX_RETRIES,
response.status_code,
)
else:
# 其他客户端错误,不重试
raise CosyVoiceError(
f"CosyVoice API 调用失败: HTTP {response.status_code}, " f"body={response.text}"
)
except httpx.TimeoutException as e:
last_error = CosyVoiceTimeoutError(f"请求超时: {e}")
logger.warning(
"CosyVoice API 超时 (尝试 %d/%d)",
attempt + 1,
self.MAX_RETRIES,
)
except httpx.RequestError as e:
last_error = CosyVoiceError(f"请求错误: {e}")
logger.warning(
"CosyVoice API 请求错误 (尝试 %d/%d): %s",
attempt + 1,
self.MAX_RETRIES,
e,
)
# 指数退避
if attempt < self.MAX_RETRIES - 1:
sleep_time = self.RETRY_BACKOFF * (2**attempt)
time.sleep(sleep_time)
# 所有重试都失败
if last_error:
raise last_error
raise CosyVoiceError("CosyVoice API 调用失败,未知错误")