Files
xiaoxia-saas/packages/adapters/tts/cosyvoice_tts_service.py
xiaoxia ef8b747c7b
CI Build & Deploy Pipeline / Production Browser E2E (push) Failing after 1424h0m10s
CI Build & Deploy Pipeline / Build Production Worker Image (push) Failing after 1424h0m14s
CI Build & Deploy Pipeline / Build Production Web Image (push) Failing after 1424h0m14s
CI/CD Pipeline / Check if frontend-only change (push) Failing after 1424h0m37s
CI Build & Deploy Pipeline / Build Production API Image (push) Failing after 1424h0m14s
CI Build & Deploy Pipeline / Build Staging API Image (push) Has been skipped
CI Build & Deploy Pipeline / Build Staging Web Image (push) Has been skipped
CI Build & Deploy Pipeline / Build Staging Worker Image (push) Has been skipped
CI Build & Deploy Pipeline / Deploy Staging (Watchtower auto-deploy) (push) Has been skipped
CI Build & Deploy Pipeline / Staging E2E Tests (push) Failing after 2m24s
CI Build & Deploy Pipeline / Staging API Integration Tests (push) Has been skipped
CI Build & Deploy Pipeline / Deploy Production (push) Failing after 1424h32m16s
CI/CD Pipeline / Frontend Unit Tests (push) Has been skipped
CI/CD Pipeline / Validate Code Quality And Tests (push) Has been skipped
CI/CD Pipeline / Unit Tests (push) Has been skipped
CI/CD Pipeline / Integration Tests (push) Has been skipped
CI/CD Pipeline / Frontend Lint (push) Has been skipped
fix(P1): 修复预设配音没有真实声音 - 接入CosyVoice TTS适配器 + legacy引擎配音补齐 (#559)
Co-authored-by: xiaoxia <dev@xiaoxiajianji.com>
Co-committed-by: xiaoxia <dev@xiaoxiajianji.com>
2026-07-19 07:28:56 +08:00

250 lines
7.6 KiB
Python
Executable File
Raw Permalink Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
"""CosyVoice TTS 服务适配器.
将 CosyVoiceService 包装为 TtsService 接口,供统一渲染管道的 TtsEngine 使用。
支持阿里云百炼 CosyVoice 真实音色合成。
"""
from __future__ import annotations
import logging
import tempfile
from pathlib import Path
from urllib.parse import urlparse
from packages.ports.tts_service import TtsError, TtsService
logger = logging.getLogger(__name__)
class CosyVoiceTtsService(TtsService):
"""CosyVoice TTS 服务适配器.
包装 CosyVoiceService,实现 TtsService 接口。
合成流程:调用 CosyVoice API → 获取音频 URL → 下载到本地 → (可选)转码为目标格式
"""
def __init__(
self,
api_key: str = "",
base_url: str = "",
model: str = "",
sample_rate: int = 24000,
format: str = "mp3",
ffmpeg_bin: str = "ffmpeg",
) -> None:
"""初始化 CosyVoice TTS 服务.
Args:
api_key: DashScope API Key,为空时从配置读取
base_url: DashScope API Base URL
model: 语音合成模型
sample_rate: 默认采样率
format: 默认输出格式
ffmpeg_bin: ffmpeg 可执行文件路径(用于转码)
"""
# 延迟导入避免循环依赖
from packages.application.cosyvoice_service import CosyVoiceService
self._service = CosyVoiceService(
api_key=api_key,
base_url=base_url,
model=model,
)
self._default_sample_rate = sample_rate
self._default_format = format
self._ffmpeg_bin = ffmpeg_bin
@property
def provider_name(self) -> str:
return "cosyvoice"
def synthesize(
self,
text: str,
*,
voice_id: str = "",
speed: float = 1.0,
pitch: float = 0.0,
output_path: Path | None = None,
sample_rate: int = 22050,
format: str = "wav",
) -> Path:
"""调用 CosyVoice 合成语音并下载到本地.
Args:
text: 输入文本
voice_id: 音色 ID(CosyVoice 音色名,如 longxiaochun_v3)
speed: 语速 (0.5 ~ 2.0)
pitch: 语调(半音,-12 ~ 12)— CosyVoice 原生不支持,用 ffmpeg 后处理实现
output_path: 输出文件路径(None 则自动生成)
sample_rate: 采样率
format: 输出格式 (wav/mp3)
Returns:
输出音频文件路径
"""
if not text.strip():
raise TtsError("文本不能为空")
if not voice_id:
voice_id = "longxiaochun_v3" # 默认音色
# 语速边界
speed = max(0.5, min(2.0, speed))
# 输出路径
if output_path is None:
suffix = f".{format}"
tmp = tempfile.NamedTemporaryFile(suffix=suffix, delete=False)
tmp.close()
output_path = Path(tmp.name)
output_path.parent.mkdir(parents=True, exist_ok=True)
try:
# 1. 调用 CosyVoice 合成(默认 mp3 格式,兼容性最好)
result = self._service.synthesize_speech(
text=text,
voice_id=voice_id,
sample_rate=sample_rate or self._default_sample_rate,
format="mp3", # 先下mp3,后面按需转码
speed=speed,
)
if not result.audio_url:
raise TtsError("CosyVoice 未返回音频 URL")
# 2. 下载音频文件
downloaded = self._download_audio(result.audio_url, output_path.parent / "_cosyvoice_tmp.mp3")
if not downloaded.exists() or downloaded.stat().st_size == 0:
raise TtsError("音频下载失败或文件为空")
# 3. 如需转码(wav)或 pitch 调整,用 ffmpeg 处理
need_transcode = (format != "mp3") or abs(pitch) > 0.01
if need_transcode:
self._post_process(downloaded, output_path, format=format, pitch=pitch, sample_rate=sample_rate)
else:
# 直接移动文件
downloaded.rename(output_path)
if not output_path.exists() or output_path.stat().st_size == 0:
raise TtsError("输出文件为空或不存在")
return output_path
except TtsError:
raise
except Exception as e:
logger.error("CosyVoice TTS 合成失败: %s", e)
raise TtsError(f"CosyVoice TTS 合成失败: {e}") from e
def estimate_duration(self, text: str, *, speed: float = 1.0) -> float:
"""估算音频时长(秒).
CosyVoice 不返回预估时长,按中文语速经验值估算:
- 正常语速约 4 字/秒
"""
if not text:
return 0.0
char_count = len([c for c in text if not c.isspace()])
if char_count == 0:
return 0.0
base_duration = char_count / 4.0 # 4 字/秒
return base_duration / max(0.1, speed)
def available_voices(self) -> list[str]:
"""支持的音色列表."""
from packages.domain.preset_voices import get_preset_voices
return [v.voice_id for v in get_preset_voices()]
def _download_audio(self, url: str, save_path: Path) -> Path:
"""下载音频文件.
Args:
url: 音频 URL
save_path: 保存路径
Returns:
保存路径
"""
import httpx
parsed = urlparse(url)
if parsed.scheme not in ("http", "https"):
raise TtsError(f"不支持的音频 URL scheme: {parsed.scheme}")
save_path.parent.mkdir(parents=True, exist_ok=True)
with httpx.Client(timeout=120.0) as client:
with client.stream("GET", url) as response:
response.raise_for_status()
with open(save_path, "wb") as f:
for chunk in response.iter_bytes():
f.write(chunk)
return save_path
def _post_process(
self,
input_path: Path,
output_path: Path,
*,
format: str = "wav",
pitch: float = 0.0,
sample_rate: int = 22050,
) -> None:
"""后处理:转码 + pitch 调整.
Args:
input_path: 输入文件路径
output_path: 输出文件路径
format: 输出格式
pitch: 语调偏移(半音)
sample_rate: 输出采样率
"""
import subprocess
# 构建滤镜
filter_parts = []
# pitch 调整:通过 asetrate 实现
if abs(pitch) > 0.01:
pitch_factor = 2 ** (pitch / 12)
new_rate = int(sample_rate * pitch_factor)
filter_parts.append(f"asetrate={new_rate}")
filter_parts.append(f"aresample={sample_rate}")
filter_str = ",".join(filter_parts) if filter_parts else None
# 编码参数
if format == "mp3":
codec_args = ["-acodec", "libmp3lame", "-b:a", "128k"]
else: # wav
codec_args = ["-acodec", "pcm_s16le"]
command = [
self._ffmpeg_bin,
"-y",
"-i",
str(input_path),
]
if filter_str:
command.extend(["-af", filter_str])
command.extend(codec_args)
command.extend(["-ar", str(sample_rate), "-ac", "1", str(output_path)])
result = subprocess.run(
command,
capture_output=True,
text=True,
timeout=60,
)
if result.returncode != 0:
raise TtsError(f"音频后处理失败: {result.stderr[-500:]}")