ef8b747c7b
CI Build & Deploy Pipeline / Production Browser E2E (push) Failing after 1424h0m10s
CI Build & Deploy Pipeline / Build Production Worker Image (push) Failing after 1424h0m14s
CI Build & Deploy Pipeline / Build Production Web Image (push) Failing after 1424h0m14s
CI/CD Pipeline / Check if frontend-only change (push) Failing after 1424h0m37s
CI Build & Deploy Pipeline / Build Production API Image (push) Failing after 1424h0m14s
CI Build & Deploy Pipeline / Build Staging API Image (push) Has been skipped
CI Build & Deploy Pipeline / Build Staging Web Image (push) Has been skipped
CI Build & Deploy Pipeline / Build Staging Worker Image (push) Has been skipped
CI Build & Deploy Pipeline / Deploy Staging (Watchtower auto-deploy) (push) Has been skipped
CI Build & Deploy Pipeline / Staging E2E Tests (push) Failing after 2m24s
CI Build & Deploy Pipeline / Staging API Integration Tests (push) Has been skipped
CI Build & Deploy Pipeline / Deploy Production (push) Failing after 1424h32m16s
CI/CD Pipeline / Frontend Unit Tests (push) Has been skipped
CI/CD Pipeline / Validate Code Quality And Tests (push) Has been skipped
CI/CD Pipeline / Unit Tests (push) Has been skipped
CI/CD Pipeline / Integration Tests (push) Has been skipped
CI/CD Pipeline / Frontend Lint (push) Has been skipped
Co-authored-by: xiaoxia <dev@xiaoxiajianji.com> Co-committed-by: xiaoxia <dev@xiaoxiajianji.com>
250 lines
7.6 KiB
Python
Executable File
250 lines
7.6 KiB
Python
Executable File
"""CosyVoice TTS 服务适配器.
|
||
|
||
将 CosyVoiceService 包装为 TtsService 接口,供统一渲染管道的 TtsEngine 使用。
|
||
支持阿里云百炼 CosyVoice 真实音色合成。
|
||
"""
|
||
|
||
from __future__ import annotations
|
||
|
||
import logging
|
||
import tempfile
|
||
from pathlib import Path
|
||
from urllib.parse import urlparse
|
||
|
||
from packages.ports.tts_service import TtsError, TtsService
|
||
|
||
logger = logging.getLogger(__name__)
|
||
|
||
|
||
class CosyVoiceTtsService(TtsService):
|
||
"""CosyVoice TTS 服务适配器.
|
||
|
||
包装 CosyVoiceService,实现 TtsService 接口。
|
||
合成流程:调用 CosyVoice API → 获取音频 URL → 下载到本地 → (可选)转码为目标格式
|
||
"""
|
||
|
||
def __init__(
|
||
self,
|
||
api_key: str = "",
|
||
base_url: str = "",
|
||
model: str = "",
|
||
sample_rate: int = 24000,
|
||
format: str = "mp3",
|
||
ffmpeg_bin: str = "ffmpeg",
|
||
) -> None:
|
||
"""初始化 CosyVoice TTS 服务.
|
||
|
||
Args:
|
||
api_key: DashScope API Key,为空时从配置读取
|
||
base_url: DashScope API Base URL
|
||
model: 语音合成模型
|
||
sample_rate: 默认采样率
|
||
format: 默认输出格式
|
||
ffmpeg_bin: ffmpeg 可执行文件路径(用于转码)
|
||
"""
|
||
# 延迟导入避免循环依赖
|
||
from packages.application.cosyvoice_service import CosyVoiceService
|
||
|
||
self._service = CosyVoiceService(
|
||
api_key=api_key,
|
||
base_url=base_url,
|
||
model=model,
|
||
)
|
||
self._default_sample_rate = sample_rate
|
||
self._default_format = format
|
||
self._ffmpeg_bin = ffmpeg_bin
|
||
|
||
@property
|
||
def provider_name(self) -> str:
|
||
return "cosyvoice"
|
||
|
||
def synthesize(
|
||
self,
|
||
text: str,
|
||
*,
|
||
voice_id: str = "",
|
||
speed: float = 1.0,
|
||
pitch: float = 0.0,
|
||
output_path: Path | None = None,
|
||
sample_rate: int = 22050,
|
||
format: str = "wav",
|
||
) -> Path:
|
||
"""调用 CosyVoice 合成语音并下载到本地.
|
||
|
||
Args:
|
||
text: 输入文本
|
||
voice_id: 音色 ID(CosyVoice 音色名,如 longxiaochun_v3)
|
||
speed: 语速 (0.5 ~ 2.0)
|
||
pitch: 语调(半音,-12 ~ 12)— CosyVoice 原生不支持,用 ffmpeg 后处理实现
|
||
output_path: 输出文件路径(None 则自动生成)
|
||
sample_rate: 采样率
|
||
format: 输出格式 (wav/mp3)
|
||
|
||
Returns:
|
||
输出音频文件路径
|
||
"""
|
||
if not text.strip():
|
||
raise TtsError("文本不能为空")
|
||
|
||
if not voice_id:
|
||
voice_id = "longxiaochun_v3" # 默认音色
|
||
|
||
# 语速边界
|
||
speed = max(0.5, min(2.0, speed))
|
||
|
||
# 输出路径
|
||
if output_path is None:
|
||
suffix = f".{format}"
|
||
tmp = tempfile.NamedTemporaryFile(suffix=suffix, delete=False)
|
||
tmp.close()
|
||
output_path = Path(tmp.name)
|
||
|
||
output_path.parent.mkdir(parents=True, exist_ok=True)
|
||
|
||
try:
|
||
# 1. 调用 CosyVoice 合成(默认 mp3 格式,兼容性最好)
|
||
result = self._service.synthesize_speech(
|
||
text=text,
|
||
voice_id=voice_id,
|
||
sample_rate=sample_rate or self._default_sample_rate,
|
||
format="mp3", # 先下mp3,后面按需转码
|
||
speed=speed,
|
||
)
|
||
|
||
if not result.audio_url:
|
||
raise TtsError("CosyVoice 未返回音频 URL")
|
||
|
||
# 2. 下载音频文件
|
||
downloaded = self._download_audio(result.audio_url, output_path.parent / "_cosyvoice_tmp.mp3")
|
||
|
||
if not downloaded.exists() or downloaded.stat().st_size == 0:
|
||
raise TtsError("音频下载失败或文件为空")
|
||
|
||
# 3. 如需转码(wav)或 pitch 调整,用 ffmpeg 处理
|
||
need_transcode = (format != "mp3") or abs(pitch) > 0.01
|
||
|
||
if need_transcode:
|
||
self._post_process(downloaded, output_path, format=format, pitch=pitch, sample_rate=sample_rate)
|
||
else:
|
||
# 直接移动文件
|
||
downloaded.rename(output_path)
|
||
|
||
if not output_path.exists() or output_path.stat().st_size == 0:
|
||
raise TtsError("输出文件为空或不存在")
|
||
|
||
return output_path
|
||
|
||
except TtsError:
|
||
raise
|
||
except Exception as e:
|
||
logger.error("CosyVoice TTS 合成失败: %s", e)
|
||
raise TtsError(f"CosyVoice TTS 合成失败: {e}") from e
|
||
|
||
def estimate_duration(self, text: str, *, speed: float = 1.0) -> float:
|
||
"""估算音频时长(秒).
|
||
|
||
CosyVoice 不返回预估时长,按中文语速经验值估算:
|
||
- 正常语速约 4 字/秒
|
||
"""
|
||
if not text:
|
||
return 0.0
|
||
char_count = len([c for c in text if not c.isspace()])
|
||
if char_count == 0:
|
||
return 0.0
|
||
base_duration = char_count / 4.0 # 4 字/秒
|
||
return base_duration / max(0.1, speed)
|
||
|
||
def available_voices(self) -> list[str]:
|
||
"""支持的音色列表."""
|
||
from packages.domain.preset_voices import get_preset_voices
|
||
|
||
return [v.voice_id for v in get_preset_voices()]
|
||
|
||
def _download_audio(self, url: str, save_path: Path) -> Path:
|
||
"""下载音频文件.
|
||
|
||
Args:
|
||
url: 音频 URL
|
||
save_path: 保存路径
|
||
|
||
Returns:
|
||
保存路径
|
||
"""
|
||
import httpx
|
||
|
||
parsed = urlparse(url)
|
||
if parsed.scheme not in ("http", "https"):
|
||
raise TtsError(f"不支持的音频 URL scheme: {parsed.scheme}")
|
||
|
||
save_path.parent.mkdir(parents=True, exist_ok=True)
|
||
|
||
with httpx.Client(timeout=120.0) as client:
|
||
with client.stream("GET", url) as response:
|
||
response.raise_for_status()
|
||
with open(save_path, "wb") as f:
|
||
for chunk in response.iter_bytes():
|
||
f.write(chunk)
|
||
|
||
return save_path
|
||
|
||
def _post_process(
|
||
self,
|
||
input_path: Path,
|
||
output_path: Path,
|
||
*,
|
||
format: str = "wav",
|
||
pitch: float = 0.0,
|
||
sample_rate: int = 22050,
|
||
) -> None:
|
||
"""后处理:转码 + pitch 调整.
|
||
|
||
Args:
|
||
input_path: 输入文件路径
|
||
output_path: 输出文件路径
|
||
format: 输出格式
|
||
pitch: 语调偏移(半音)
|
||
sample_rate: 输出采样率
|
||
"""
|
||
import subprocess
|
||
|
||
# 构建滤镜
|
||
filter_parts = []
|
||
|
||
# pitch 调整:通过 asetrate 实现
|
||
if abs(pitch) > 0.01:
|
||
pitch_factor = 2 ** (pitch / 12)
|
||
new_rate = int(sample_rate * pitch_factor)
|
||
filter_parts.append(f"asetrate={new_rate}")
|
||
filter_parts.append(f"aresample={sample_rate}")
|
||
|
||
filter_str = ",".join(filter_parts) if filter_parts else None
|
||
|
||
# 编码参数
|
||
if format == "mp3":
|
||
codec_args = ["-acodec", "libmp3lame", "-b:a", "128k"]
|
||
else: # wav
|
||
codec_args = ["-acodec", "pcm_s16le"]
|
||
|
||
command = [
|
||
self._ffmpeg_bin,
|
||
"-y",
|
||
"-i",
|
||
str(input_path),
|
||
]
|
||
|
||
if filter_str:
|
||
command.extend(["-af", filter_str])
|
||
|
||
command.extend(codec_args)
|
||
command.extend(["-ar", str(sample_rate), "-ac", "1", str(output_path)])
|
||
|
||
result = subprocess.run(
|
||
command,
|
||
capture_output=True,
|
||
text=True,
|
||
timeout=60,
|
||
)
|
||
|
||
if result.returncode != 0:
|
||
raise TtsError(f"音频后处理失败: {result.stderr[-500:]}")
|