feat(ai-avatar): TTS直生对口型 + 语速/情绪透传 + MediaKit智能封面 + 契约文档
- 对口型支持 TTS 直生模式:POST /lipsync/jobs 传 voice_id+script_text(+speed/emotion), 后端内部调 CosyVoice 合成音频→转存OSS→提交 MediaKit;保留 audio_url 直接音频模式 - cosyvoice_service: 新增 emotion 参数 + normalize_emotion(中文 自然/兴奋/沉稳/亲切 映射 natural/excited/calm/friendly),空值不透传 - TTS 链路 speed/emotion 全链路透传:schema→use_case(metadata)→workflow(单段/分段/重合成) →submit_synthesize_task(rate/emotion);preview 即时试听同步 - lipsync refresh: 中间状态(running/processing)同步DB;completed 输出视频转存自家OSS防过期 - 修复 lipsync/ai-avatar 路由 current_user.id → current_user.user.id(AuthenticatedUser 无 .id) - 新增 POST /ai-avatar/render/smart-cover 独立封面接口:复用 MediaKit extract_frames + cover_frame_scorer 评分选最佳帧(非 FFmpeg 首帧),渲染管线封面同样优先智能选帧 - 迁移 073: lipsync_jobs 增加 voice_id/script_text/speed/emotion 列,audio_url 改可空 - docs/ai-avatar-api-contract-1822.md: 前后端接口契约 + title_config 字段清单 - 新增 11 个单测(情绪归一化/payload透传/TTS直生/中间状态/智能封面),相关 82 测试全绿 Refs #1797 #1822
This commit is contained in:
@@ -17,7 +17,10 @@ from app.dependencies import get_db_session
|
||||
from app.schemas.ai_avatar_render import (
|
||||
AiAvatarRenderJobResponse,
|
||||
CreateAiAvatarRenderRequest,
|
||||
SmartCoverRequest,
|
||||
SmartCoverResponse,
|
||||
)
|
||||
from app.services.ai_avatar_cover_service import generate_smart_cover
|
||||
from app.services.ai_avatar_render_service import (
|
||||
AiAvatarRenderError,
|
||||
AiAvatarRenderService,
|
||||
@@ -49,7 +52,7 @@ def create_render_job(
|
||||
"""
|
||||
try:
|
||||
job = svc.create_render_job(
|
||||
user_id=current_user.id,
|
||||
user_id=current_user.user.id,
|
||||
lipsync_job_id=body.lipsync_job_id,
|
||||
script_id=body.script_id,
|
||||
b_roll_segments=[s.model_dump() for s in body.b_roll_segments],
|
||||
@@ -94,7 +97,7 @@ def list_render_jobs(
|
||||
):
|
||||
"""获取 AI 数字人渲染任务列表."""
|
||||
items, total = svc.list_render_jobs(
|
||||
user_id=current_user.id,
|
||||
user_id=current_user.user.id,
|
||||
project_id=project_id,
|
||||
status=status,
|
||||
offset=offset,
|
||||
@@ -118,7 +121,7 @@ def get_render_job(
|
||||
svc: AiAvatarRenderService = Depends(_get_service),
|
||||
):
|
||||
"""获取渲染任务详情."""
|
||||
job = svc.get_render_job(job_id, current_user.id)
|
||||
job = svc.get_render_job(job_id, current_user.user.id)
|
||||
if job is None:
|
||||
raise HTTPException(status_code=404, detail="渲染任务不存在")
|
||||
return job
|
||||
@@ -134,7 +137,7 @@ def cancel_render_job(
|
||||
svc: AiAvatarRenderService = Depends(_get_service),
|
||||
):
|
||||
"""取消渲染任务(仅 pending 状态可取消)."""
|
||||
job = svc.cancel_render_job(job_id, current_user.id)
|
||||
job = svc.cancel_render_job(job_id, current_user.user.id)
|
||||
if job is None:
|
||||
raise HTTPException(status_code=404, detail="渲染任务不存在")
|
||||
if job.status != "cancelled":
|
||||
@@ -155,7 +158,7 @@ def retry_render_job(
|
||||
svc: AiAvatarRenderService = Depends(_get_service),
|
||||
):
|
||||
"""重试失败的渲染任务."""
|
||||
job = svc.retry_render_job(job_id, current_user.id)
|
||||
job = svc.retry_render_job(job_id, current_user.user.id)
|
||||
if job is None:
|
||||
raise HTTPException(status_code=404, detail="渲染任务不存在")
|
||||
if job.status != "pending":
|
||||
@@ -173,3 +176,34 @@ def retry_render_job(
|
||||
logger.warning("Celery 任务提交失败,重试任务已重置但未触发执行: %s", job.id)
|
||||
|
||||
return job
|
||||
|
||||
|
||||
|
||||
# ── POST /smart-cover — 智能获取封面(MediaKit 抽帧 + 评分选帧)────────
|
||||
|
||||
|
||||
@router.post("/smart-cover", response_model=SmartCoverResponse)
|
||||
def generate_avatar_smart_cover(
|
||||
body: SmartCoverRequest,
|
||||
current_user: AuthenticatedUser = Depends(get_current_user),
|
||||
) -> SmartCoverResponse:
|
||||
"""智能获取数字人视频封面.
|
||||
|
||||
复用智能剪辑的 MediaKit 抽帧 + 质量评分选最佳帧逻辑(非 FFmpeg 简单截帧),
|
||||
并将选中帧转存到自家 OSS,返回非临时的封面公网 URL。
|
||||
|
||||
前端「智能获取封面」按钮可直接调用本接口;不依赖渲染任务完成。
|
||||
"""
|
||||
video_url = (body.video_url or "").strip()
|
||||
if not video_url.startswith(("http://", "https://")):
|
||||
raise HTTPException(status_code=400, detail="video_url 必须是合法的 HTTP/HTTPS URL")
|
||||
|
||||
cover_url = generate_smart_cover(video_url, max_frames=body.max_frames)
|
||||
if not cover_url:
|
||||
return SmartCoverResponse(
|
||||
cover_url="",
|
||||
status="fallback_failed",
|
||||
message="智能抽帧失败(MediaKit 不可用或抽帧异常),请稍后重试",
|
||||
)
|
||||
logger.info("智能封面生成成功: user=%s", current_user.user.id)
|
||||
return SmartCoverResponse(cover_url=cover_url, status="completed")
|
||||
|
||||
@@ -13,7 +13,7 @@ from __future__ import annotations
|
||||
import logging
|
||||
|
||||
from app.auth import AuthenticatedUser, get_current_user
|
||||
from app.dependencies import get_db_session
|
||||
from app.dependencies import get_db_session, get_voice_clone_profile_repository
|
||||
from app.schemas.lipsync import CreateLipsyncJobRequest, LipsyncJobResponse
|
||||
from app.services.lipsync_service import LipsyncService
|
||||
from app.services.mediakit_client import MediaKitError
|
||||
@@ -25,8 +25,11 @@ logger = logging.getLogger(__name__)
|
||||
router = APIRouter()
|
||||
|
||||
|
||||
def _get_service(db: Session = Depends(get_db_session)) -> LipsyncService:
|
||||
return LipsyncService(db)
|
||||
def _get_service(
|
||||
db: Session = Depends(get_db_session),
|
||||
voice_clone_repo=Depends(get_voice_clone_profile_repository),
|
||||
) -> LipsyncService:
|
||||
return LipsyncService(db, voice_clone_repo=voice_clone_repo)
|
||||
|
||||
|
||||
# ── POST /jobs — 提交对口型任务 ───────────────────────────────────────────
|
||||
@@ -44,20 +47,31 @@ def create_lipsync_job(
|
||||
"""
|
||||
try:
|
||||
job = svc.create_job(
|
||||
user_id=current_user.id,
|
||||
user_id=current_user.user.id,
|
||||
video_url=body.video_url,
|
||||
audio_url=body.audio_url,
|
||||
voice_id=body.voice_id,
|
||||
script_text=body.script_text,
|
||||
speed=body.speed,
|
||||
emotion=body.emotion,
|
||||
enable_video_loop=body.enable_video_loop,
|
||||
project_id=body.project_id,
|
||||
)
|
||||
except MediaKitError as exc:
|
||||
# 创建失败(job 已记录 error),返回 502
|
||||
# TTS 合成失败 / 音色无权访问 → 400/403;MediaKit 提交失败 → 502
|
||||
status_code = 502
|
||||
if exc.code in ("VoiceForbidden",):
|
||||
status_code = 403
|
||||
elif exc.code in ("InvalidInput", "TTSInvalidParam", "VoiceNotReady"):
|
||||
status_code = 400
|
||||
elif exc.code == "TTSSynthesisFailed":
|
||||
status_code = 502
|
||||
raise HTTPException(
|
||||
status_code=502,
|
||||
status_code=status_code,
|
||||
detail={
|
||||
"code": exc.code,
|
||||
"message": str(exc),
|
||||
"request_id": exc.request_id,
|
||||
"request_id": getattr(exc, "request_id", ""),
|
||||
},
|
||||
) from exc
|
||||
|
||||
@@ -78,7 +92,7 @@ def list_lipsync_jobs(
|
||||
):
|
||||
"""获取对口型任务列表."""
|
||||
items, total = svc.list_jobs(
|
||||
user_id=current_user.id,
|
||||
user_id=current_user.user.id,
|
||||
project_id=project_id,
|
||||
status=status,
|
||||
offset=offset,
|
||||
@@ -102,7 +116,7 @@ def get_lipsync_job(
|
||||
svc: LipsyncService = Depends(_get_service),
|
||||
):
|
||||
"""获取对口型任务详情."""
|
||||
job = svc.get_job(job_id, current_user.id)
|
||||
job = svc.get_job(job_id, current_user.user.id)
|
||||
if job is None:
|
||||
raise HTTPException(status_code=404, detail="任务不存在")
|
||||
return job
|
||||
@@ -118,7 +132,7 @@ def refresh_lipsync_job(
|
||||
svc: LipsyncService = Depends(_get_service),
|
||||
):
|
||||
"""从 MediaKit 拉取最新状态并更新."""
|
||||
job = svc.refresh_job_status(job_id, current_user.id)
|
||||
job = svc.refresh_job_status(job_id, current_user.user.id)
|
||||
if job is None:
|
||||
raise HTTPException(status_code=404, detail="任务不存在")
|
||||
return job
|
||||
@@ -134,7 +148,7 @@ def cancel_lipsync_job(
|
||||
svc: LipsyncService = Depends(_get_service),
|
||||
):
|
||||
"""取消对口型任务(仅 pending/submitted 状态可取消)."""
|
||||
job = svc.cancel_job(job_id, current_user.id)
|
||||
job = svc.cancel_job(job_id, current_user.user.id)
|
||||
if job is None:
|
||||
raise HTTPException(status_code=404, detail="任务不存在")
|
||||
if job.status != "cancelled":
|
||||
|
||||
@@ -173,6 +173,14 @@ def synthesize(
|
||||
# job.voice_id 统一存解析后的 CosyVoice voice_id
|
||||
actual_voice_id = resolved_profile.voice_id
|
||||
|
||||
# 语速/情绪等合成参数随 metadata 落库,workflow 提交 CosyVoice 时读取透传
|
||||
synthesis_meta = {
|
||||
"speed": request.speed,
|
||||
"emotion": request.emotion or "",
|
||||
}
|
||||
if request.metadata_:
|
||||
synthesis_meta.update(request.metadata_)
|
||||
|
||||
use_case = CreateTTSJobUseCase(repository)
|
||||
job = use_case.execute(
|
||||
user_id=user_id,
|
||||
@@ -180,7 +188,7 @@ def synthesize(
|
||||
voice_id=actual_voice_id,
|
||||
voice_model=request.voice_model,
|
||||
voice_clone_profile_id=voice_clone_profile_id,
|
||||
metadata=request.metadata_,
|
||||
metadata=synthesis_meta,
|
||||
)
|
||||
|
||||
# 提交 CosyVoice 合成任务
|
||||
@@ -567,6 +575,7 @@ def preview_tts(
|
||||
text=request.text,
|
||||
voice_id=actual_voice_id,
|
||||
speed=request.speed,
|
||||
emotion=request.emotion,
|
||||
)
|
||||
except CosyVoiceError as e:
|
||||
raise HTTPException(
|
||||
|
||||
Reference in New Issue
Block a user