a27ed596b4
- 对口型支持 TTS 直生模式:POST /lipsync/jobs 传 voice_id+script_text(+speed/emotion), 后端内部调 CosyVoice 合成音频→转存OSS→提交 MediaKit;保留 audio_url 直接音频模式 - cosyvoice_service: 新增 emotion 参数 + normalize_emotion(中文 自然/兴奋/沉稳/亲切 映射 natural/excited/calm/friendly),空值不透传 - TTS 链路 speed/emotion 全链路透传:schema→use_case(metadata)→workflow(单段/分段/重合成) →submit_synthesize_task(rate/emotion);preview 即时试听同步 - lipsync refresh: 中间状态(running/processing)同步DB;completed 输出视频转存自家OSS防过期 - 修复 lipsync/ai-avatar 路由 current_user.id → current_user.user.id(AuthenticatedUser 无 .id) - 新增 POST /ai-avatar/render/smart-cover 独立封面接口:复用 MediaKit extract_frames + cover_frame_scorer 评分选最佳帧(非 FFmpeg 首帧),渲染管线封面同样优先智能选帧 - 迁移 073: lipsync_jobs 增加 voice_id/script_text/speed/emotion 列,audio_url 改可空 - docs/ai-avatar-api-contract-1822.md: 前后端接口契约 + title_config 字段清单 - 新增 11 个单测(情绪归一化/payload透传/TTS直生/中间状态/智能封面),相关 82 测试全绿 Refs #1797 #1822
99 lines
3.6 KiB
Python
99 lines
3.6 KiB
Python
"""对口型 API Schema 定义 — #1796 / #1822.
|
||
|
||
支持两种输入模式(二选一):
|
||
1. TTS 直生模式(推荐):传 voice_id + script_text(+ speed/emotion),
|
||
后端内部先调 CosyVoice 合成音频,再提交 MediaKit 对口型。
|
||
2. 直接音频模式:传 video_url + audio_url(音频已由调用方准备好)。
|
||
"""
|
||
|
||
from __future__ import annotations
|
||
|
||
from datetime import datetime
|
||
from typing import Optional
|
||
|
||
from pydantic import BaseModel, Field, model_validator
|
||
|
||
|
||
class LipsyncJobResponse(BaseModel):
|
||
"""对口型任务响应."""
|
||
|
||
id: str
|
||
user_id: str
|
||
project_id: str
|
||
video_url: str
|
||
audio_url: str
|
||
enable_video_loop: bool
|
||
voice_id: str = ""
|
||
script_text: str = ""
|
||
speed: float = 1.0
|
||
emotion: str = ""
|
||
mediakit_task_id: str
|
||
status: str
|
||
output_video_url: str
|
||
output_duration: float
|
||
error_message: str
|
||
error_code: str
|
||
submitted_at: Optional[datetime] = None
|
||
completed_at: Optional[datetime] = None
|
||
created_at: datetime
|
||
updated_at: datetime
|
||
|
||
class Config:
|
||
from_attributes = True
|
||
|
||
|
||
class CreateLipsyncJobRequest(BaseModel):
|
||
"""创建对口型任务请求.
|
||
|
||
两种模式:
|
||
- TTS 直生:voice_id + script_text 必填;video_url 必填(人物视频);
|
||
audio_url 留空(后端合成)。
|
||
- 直接音频:video_url + audio_url 必填。
|
||
"""
|
||
|
||
video_url: str = Field(..., description="人物视频 URL(MP4,≤30min,单人真人)")
|
||
|
||
# 模式 2:直接音频
|
||
audio_url: str = Field("", description="驱动音频 URL(mp3/aac/wav/m4a/flac);直生模式留空")
|
||
|
||
# 模式 1:TTS 直生
|
||
voice_id: str = Field("", description="音色 ID(预置音色或克隆音色 profile UUID)")
|
||
script_text: str = Field("", description="要合成的文案(直生模式必填)")
|
||
speed: float = Field(1.0, ge=0.5, le=2.0, description="语速(0.5-2.0),默认 1.0")
|
||
emotion: str = Field("", description="情绪(natural/excited/calm/friendly 或中文 自然/兴奋/沉稳/亲切)")
|
||
|
||
enable_video_loop: bool = Field(False, description="音频长于视频时是否循环画面")
|
||
project_id: str = Field("", description="项目 ID(可选)")
|
||
|
||
@model_validator(mode="after")
|
||
def _validate_input_mode(self) -> "CreateLipsyncJobRequest":
|
||
video = (self.video_url or "").strip()
|
||
if not video:
|
||
raise ValueError("video_url 不能为空")
|
||
if not video.startswith(("http://", "https://")):
|
||
raise ValueError("video_url 必须是 HTTP/HTTPS URL")
|
||
lower = video.lower().split("?")[0]
|
||
if not lower.endswith(".mp4"):
|
||
raise ValueError("video_url 仅支持 MP4 格式")
|
||
|
||
has_audio = bool((self.audio_url or "").strip())
|
||
has_tts = bool((self.voice_id or "").strip()) and bool((self.script_text or "").strip())
|
||
|
||
if not has_audio and not has_tts:
|
||
raise ValueError(
|
||
"必须提供驱动音频:要么传 audio_url(直接音频模式),"
|
||
"要么同时传 voice_id + script_text(TTS 直生模式)"
|
||
)
|
||
|
||
if has_audio:
|
||
au = self.audio_url.strip()
|
||
if not au.startswith(("http://", "https://")):
|
||
raise ValueError("audio_url 必须是 HTTP/HTTPS URL")
|
||
au_lower = au.lower().split("?")[0]
|
||
allowed = (".mp3", ".aac", ".wav", ".m4a", ".flac")
|
||
if not any(au_lower.endswith(ext) for ext in allowed):
|
||
raise ValueError(f"audio_url 格式不支持,仅支持: {', '.join(allowed)}")
|
||
self.audio_url = au
|
||
|
||
return self
|