From 29b0a7d90d24f9a794df1d49655f6a7cc6d9e7a6 Mon Sep 17 00:00:00 2001 From: xiaoxia Date: Sun, 13 Sep 2026 03:29:53 +0800 Subject: [PATCH 1/4] =?UTF-8?q?feat(ai-avatar):=20=E9=85=8D=E9=9F=B3?= =?UTF-8?q?=E5=89=8D=E7=BD=AE=E2=80=94=E2=80=94=E6=AD=A5=E9=AA=A41?= =?UTF-8?q?=E7=94=9F=E6=88=90=E9=85=8D=E9=9F=B3=EF=BC=8C=E9=9F=B3=E9=A2=91?= =?UTF-8?q?=E7=9B=B4=E4=BC=A0=E5=AF=B9=E5=8F=A3=E5=9E=8B=EF=BC=8CB-roll?= =?UTF-8?q?=E6=97=B6=E9=97=B4=E6=88=B3=E7=AB=8B=E5=8D=B3=E5=8F=AF=E7=94=A8?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit 后端: - packages/domain/sentence_timings.py:抽出 compute_sentence_timings / probe_audio_duration 等公共工具 - tasks/lipsync_tts.py:改为复用 sentence_timings 公共模块,保留 Celery 任务作为降级路径 - schemas/lipsync.py:新增 AiAvatarTtsPreviewRequest/Response;CreateLipsyncJobRequest 新增可选 audio_url/audio_duration/sentence_timings 字段 - services/lipsync_service.py:新增 preview_tts() 同步方法;create_job 支持 audio_url 直传模式(下载→ffprobe→(可选重算timings)→直接提交 MediaKit,job.status=submitted) - routes/lipsync.py:新增 POST /lipsync/jobs/tts-preview 鉴权路由 前端: - types.ts:AiAvatarState 新增 ttsPreview 字段,TtsPreviewStatus 类型 - hooks/useAiAvatar.ts:新增 handlePreviewTts/resetTtsPreview,文案/音色/语速变化自动重置 ttsPreview;handleGenerateLipsync 在预合成完成时把 audio_url/audio_duration/sentence_timings 直传给后端 - api/aiAvatar.ts:新增 previewTts() API 封装,createLipsyncJob 入参加可选音频直传字段 - AiAvatarPage.tsx:步骤1「下一步→」改为「🎵 生成配音」+ 进度 Modal(模拟进度/loading/错误重试);步骤2进入条件改为 ttsPreview.status==='done';B-roll 时间戳优先用 lipsyncJob、降级用 ttsPreview.sentenceTimings 测试: - tests/unit/test_lipsync_tts_preview.py:5个新用例覆盖成功/TTS失败/下载失败降级/零时长/no_audio_url - 全量 lipsync/tts/avatar/preview 相关 1330 用例全部通过,无回归 --- apps/api/app/api/routes/lipsync.py | 79 +++- apps/api/app/schemas/lipsync.py | 43 ++- apps/api/app/services/lipsync_service.py | 291 +++++++++++---- apps/api/app/tasks/lipsync_tts.py | 233 ++---------- apps/web/src/pages/ai-avatar/AiAvatarPage.tsx | 339 +++++++++++++++--- apps/web/src/pages/ai-avatar/api/aiAvatar.ts | 48 ++- .../src/pages/ai-avatar/hooks/useAiAvatar.ts | 19 +- apps/web/src/pages/ai-avatar/types.ts | 11 + packages/domain/sentence_timings.py | 205 +++++++++++ tests/unit/test_lipsync_tts_preview.py | 168 +++++++++ 10 files changed, 1079 insertions(+), 357 deletions(-) create mode 100644 packages/domain/sentence_timings.py create mode 100644 tests/unit/test_lipsync_tts_preview.py diff --git a/apps/api/app/api/routes/lipsync.py b/apps/api/app/api/routes/lipsync.py index 7fc23a56e..f4944ec62 100644 --- a/apps/api/app/api/routes/lipsync.py +++ b/apps/api/app/api/routes/lipsync.py @@ -1,11 +1,12 @@ -"""对口型 API 路由 — #1796 MediaKit 对口型, #1809 参数调整. +"""对口型 API 路由 — #1796 MediaKit 对口型, #1809 参数调整, #1845 配音前置. 接口: - POST /api/v1/lipsync/jobs 提交对口型任务 + POST /api/v1/lipsync/jobs 提交对口型任务(支持 TTS/直传/预合成 三种模式) GET /api/v1/lipsync/jobs 任务列表 GET /api/v1/lipsync/jobs/{id} 任务详情 POST /api/v1/lipsync/jobs/{id}/refresh 刷新任务状态 POST /api/v1/lipsync/jobs/{id}/cancel 取消任务 + POST /api/v1/lipsync/tts-preview #1845 步骤1 TTS 预合成(同步 HTTP,~2-3s) """ from __future__ import annotations @@ -17,7 +18,12 @@ from app.dependencies import ( get_db_session, get_voice_clone_profile_repository, ) -from app.schemas.lipsync import CreateLipsyncJobRequest, LipsyncJobResponse +from app.schemas.lipsync import ( + AiAvatarTtsPreviewRequest, + AiAvatarTtsPreviewResponse, + CreateLipsyncJobRequest, + LipsyncJobResponse, +) from app.services.lipsync_service import LipsyncService from app.services.mediakit_client import MediaKitError from fastapi import APIRouter, BackgroundTasks, Depends, HTTPException, Query @@ -33,7 +39,6 @@ def _get_service( voice_clone_repo=Depends(get_voice_clone_profile_repository), ) -> LipsyncService: # voice_clone_repo 用于克隆音色 profile 解析 - # TTS 合成已移至 Celery 异步任务,无需同步注入 cosyvoice_service return LipsyncService( db, voice_clone_repo=voice_clone_repo, @@ -51,15 +56,20 @@ def create_lipsync_job( ): """提交对口型任务. - #1809/#1822: 前端传 {video_url, voice_id, script_text, speed?, emotion?}, - 后端创建任务记录(状态 tts_processing),dispatch Celery 异步任务执行 TTS 合成 + MediaKit 提交; - 也支持直接传 {video_url, audio_url}(同步提交 MediaKit)。 + 三种模式: + - TTS 直生(旧版/降级):传 {video_url, voice_id, script_text, speed?, emotion?}, + 后端 dispatch Celery 异步任务。 + - 直接音频:传 {video_url, audio_url},后端同步下载+算timings+提交MediaKit。 + - 预合成音频(#1845 新主路径):传 {video_url, audio_url, audio_duration, sentence_timings}, + 后端同步ffprobe+写入timings+直接提交MediaKit(~2-3s)。 """ try: job = svc.create_job( user_id=current_user.user.id, video_url=body.video_url, audio_url=body.audio_url, + audio_duration=body.audio_duration, + sentence_timings=body.sentence_timings, voice_id=body.voice_id, script_text=body.script_text, speed=body.speed, @@ -68,10 +78,8 @@ def create_lipsync_job( project_id=body.project_id, ) except ValueError as exc: - # 参数无效(如 voice_id 格式不对、文本过长等) raise HTTPException(status_code=400, detail=str(exc)) from exc except MediaKitError as exc: - # 音色无权访问 → 403;参数无效 → 400;MediaKit 提交失败 → 502 status_code = 502 if exc.code in ("VoiceForbidden",): status_code = 403 @@ -86,7 +94,6 @@ def create_lipsync_job( }, ) from exc except Exception as exc: - # 兜底:任何未预期的错误返回 400 而非 500 logger.error("创建对口型任务异常: %s", exc, exc_info=True) raise HTTPException( status_code=400, @@ -96,6 +103,52 @@ def create_lipsync_job( return job +# ── POST /tts-preview — #1845 步骤1 TTS 预合成 ────────────────────────── + + +@router.post("/tts-preview", response_model=AiAvatarTtsPreviewResponse) +def preview_tts( + body: AiAvatarTtsPreviewRequest, + current_user: AuthenticatedUser = Depends(get_current_user), + svc: LipsyncService = Depends(_get_service), +): + """步骤1「生成配音」同步 TTS 预合成. + + 同步执行 TTS 合成 → 下载音频 → ffprobe 时长 → 句子时间戳计算, + 不创建 LipsyncJob、不转存 OSS,直接返回 CosyVoice 临时 URL(~24h 有效)。 + 耗时约 2-3 秒。 + """ + try: + result = svc.preview_tts( + user_id=current_user.user.id, + voice_id=body.voice_id, + script_text=body.script_text, + speed=body.speed, + emotion=body.emotion, + ) + except MediaKitError as exc: + status_code = 400 + if exc.code in ("VoiceForbidden",): + status_code = 403 + elif exc.code in ("TTSNoAudio",): + status_code = 502 + raise HTTPException( + status_code=status_code, + detail={ + "code": exc.code, + "message": str(exc), + }, + ) from exc + except Exception as exc: + logger.error("TTS 预合成异常: %s", exc, exc_info=True) + raise HTTPException( + status_code=400, + detail=f"TTS 合成失败: {exc}", + ) from exc + + return result + + # ── GET /jobs — 任务列表 ───────────────────────────────────────────────── @@ -134,11 +187,7 @@ def get_lipsync_job( current_user: AuthenticatedUser = Depends(get_current_user), svc: LipsyncService = Depends(_get_service), ): - """获取对口型任务详情. - - 非终态任务:先返回 DB 缓存,挂后台刷新(下次轮询拿到新状态), - 避免 MediaKit 慢响应阻塞前端轮询。 - """ + """获取对口型任务详情.""" job = svc.get_job(job_id, current_user.user.id) if job is None: raise HTTPException(status_code=404, detail="任务不存在") diff --git a/apps/api/app/schemas/lipsync.py b/apps/api/app/schemas/lipsync.py index 47b306f43..5cffbc070 100644 --- a/apps/api/app/schemas/lipsync.py +++ b/apps/api/app/schemas/lipsync.py @@ -1,9 +1,12 @@ -"""对口型 API Schema 定义 — #1796 / #1809 / #1822. +"""对口型 API Schema 定义 — #1796 / #1809 / #1822 / #1845(配音前置). -支持两种输入模式(二选一): -1. TTS 直生模式(推荐):传 voice_id + script_text(+ speed/emotion), - 后端内部先调 CosyVoice 合成音频,再提交 MediaKit 对口型。 +支持三种输入模式: +1. TTS 直生模式(兼容旧版前端):传 voice_id + script_text(+ speed/emotion), + 后端 Celery 异步做 TTS 合成 + MediaKit 提交。 2. 直接音频模式:传 video_url + audio_url(音频已由调用方准备好)。 +3. 预合成音频模式(#1845 配音前置新主路径):前端先调 POST /lipsync/tts-preview + 拿到 audio_url + sentence_timings,再在 create_job 时传 audio_url + audio_duration + + sentence_timings,后端跳过 TTS 和时间戳计算,直接 ffprobe 校验后提交 MediaKit。 """ from __future__ import annotations @@ -46,15 +49,19 @@ class LipsyncJobResponse(BaseModel): class CreateLipsyncJobRequest(BaseModel): """创建对口型任务请求. - 两种模式(二选一): - - TTS 直生:voice_id + script_text 必填(+ 可选 speed/emotion);audio_url 留空。 + 三种模式(三选一): + - TTS 直生(旧版/降级):voice_id + script_text 必填;audio_url 留空。 - 直接音频:video_url + audio_url 必填。 + - 预合成音频(#1845 新主路径):audio_url 必填 + 可选 audio_duration/sentence_timings; + 后端同步 ffprobe 校验时长、写入 timings,直接提交 MediaKit。 """ video_url: str = Field(..., description="人物视频 URL(MP4,≤30min,单人真人)") - # 模式 2:直接音频 + # 模式 2/3:直接/预合成音频 audio_url: str = Field("", description="驱动音频 URL(mp3/aac/wav/m4a/flac);直生模式留空") + audio_duration: Optional[float] = Field(None, ge=0, description="预合成音频时长(秒),可选;后端会 ffprobe 校验") + sentence_timings: Optional[list] = Field(None, description="预合成接口返回的句子时间戳,可选;若传入则直接写入 job") # 模式 1:TTS 直生 voice_id: str = Field("", description="音色 ID(预置音色或克隆音色 profile UUID)") @@ -81,7 +88,7 @@ class CreateLipsyncJobRequest(BaseModel): if not has_audio and not has_tts: raise ValueError( - "必须提供驱动音频:要么传 audio_url(直接音频模式)," + "必须提供驱动音频:要么传 audio_url(直接/预合成音频模式)," "要么同时传 voice_id + script_text(TTS 直生模式)" ) @@ -99,3 +106,23 @@ class CreateLipsyncJobRequest(BaseModel): self.audio_url = au return self + + +# ── #1845 TTS 预合成接口 ──────────────────────────────────────────────── + + +class AiAvatarTtsPreviewRequest(BaseModel): + """步骤1「生成配音」预合成请求(同步 HTTP,~2-3s).""" + + voice_id: str = Field(..., min_length=1, max_length=128, description="音色 ID") + script_text: str = Field(..., min_length=1, max_length=5000, description="要合成的文案") + speed: float = Field(1.0, ge=0.5, le=2.0, description="语速(0.5-2.0),默认 1.0") + emotion: str = Field("natural", max_length=32, description="情绪") + + +class AiAvatarTtsPreviewResponse(BaseModel): + """TTS 预合成响应(临时 URL,24h 内有效,足够当前会话使用).""" + + audio_url: str = Field(..., description="CosyVoice 临时音频 URL") + duration: float = Field(..., ge=0, description="音频总时长(秒),ffprobe 测得") + sentence_timings: list[dict] = Field(..., description="句子级精确时间戳") diff --git a/apps/api/app/services/lipsync_service.py b/apps/api/app/services/lipsync_service.py index 283c6f032..9fed8efd6 100644 --- a/apps/api/app/services/lipsync_service.py +++ b/apps/api/app/services/lipsync_service.py @@ -1,8 +1,12 @@ -"""对口型 Service — #1796 MediaKit 对口型业务逻辑, #1809 参数调整. +"""对口型 Service — #1796 MediaKit 对口型业务逻辑, #1809 参数调整, #1845 配音前置. 职责: - 创建/查询对口型任务 -- 双输入模式:TTS 直生(voice_id + script_text,内部先合成音频转存 OSS)或直接音频(audio_url) +- 三输入模式: + 1. TTS 直生(voice_id + script_text)→ 走 Celery 异步(降级路径) + 2. 直接音频(audio_url,前端未传 timings)→ 同步下载 + 算 timings + 提交 MediaKit + 3. 预合成音频(audio_url + sentence_timings,#1845 新主路径)→ 同步 ffprobe 校验时长 + + 写入前端传来的 timings → 直接提交 MediaKit(~2-3s) - 调用 MediaKit 客户端提交异步任务 - 轮询更新任务状态(中间状态同步 DB,成片转存自家 OSS) - 用户隔离(每个用户只能操作自己的任务) @@ -26,19 +30,22 @@ from app.services.mediakit_client import ( get_mediakit_client, ) -# Celery 异步任务:TTS 合成 + MediaKit 提交(#lipsync-speed-optimization) +# Celery 异步任务:TTS 合成 + MediaKit 提交(降级路径) from app.tasks.lipsync_tts import tts_synthesize_and_submit from sqlalchemy.orm import Session from packages.adapters.sqlalchemy_impl.models import LipsyncJobModel from packages.application.cosyvoice_service import CosyVoiceError, normalize_emotion +from packages.domain.sentence_timings import ( + compute_sentence_timings, + probe_audio_duration, +) from packages.shared.storage import get_shared_storage_service from packages.shared.url_security import ALLOWED_AUDIO_MIME_TYPES, safe_download_bytes logger = logging.getLogger(__name__) # 传给 MediaKit GPU worker / 回给前端播放的 OSS 预签名有效期:7 天。 -# MediaKit 排队 + 拉取可能延迟,私有桶裸 URL 或 1 小时短预签名都会 403,故统一重签长有效期。 MEDIAKIT_URL_TTL_SECONDS = 7 * 24 * 3600 @@ -142,6 +149,98 @@ class LipsyncService: logger.warning("TTS 音频转存 OSS 失败,回退临时 URL: job_id=%s err=%s", job_id, exc) return temp_url + def _submit_audio_direct( + self, + *, + job: LipsyncJobModel, + supplied_timings: Optional[list] = None, + supplied_duration: Optional[float] = None, + ) -> None: + """音频直传模式(包含 #1845 预合成路径):同步下载 → ffprobe → timings → 提交 MediaKit. + + 直接在 HTTP 请求内完成,不走 Celery。job.status 成功后置为 submitted。 + 失败时把 job 标成 failed 并 commit,然后抛 MediaKitError。 + + Args: + job: 已 commit 的 LipsyncJobModel(audio_url / video_url 已写入) + supplied_timings: 前端传来的预合成 timings(可选,可信时直接用) + supplied_duration: 前端传来的预合成时长(可选,用于优先避免重复探测) + """ + # 1. 下载音频 + audio_data: bytes | None = None + try: + audio_data = safe_download_bytes( + job.audio_url, + purpose="lipsync_direct_audio", + allowed_mime_types=ALLOWED_AUDIO_MIME_TYPES, + timeout=60.0, + ) + logger.info( + "[lipsync] 直传音频下载完成: job_id=%s size=%d", + job.id, + len(audio_data) if audio_data else 0, + ) + except Exception as exc: + logger.warning("[lipsync] 直传音频下载失败,跳过 timings 计算: job_id=%s err=%s", job.id, exc) + + # 2. ffprobe 探测时长(优先用前端传入的预合成时长,但以 ffprobe 为准做兜底校验) + audio_duration = 0.0 + if audio_data: + audio_duration = probe_audio_duration(audio_data) + if audio_duration <= 0 and supplied_duration and supplied_duration > 0: + audio_duration = supplied_duration + logger.info("[lipsync] ffprobe 失败,使用前端传入的预合成时长: job_id=%s duration=%.2f", job.id, audio_duration) + + # 3. 句子时间戳:优先用前端预合成传入的 timings(后端预合成接口已经算过,可信); + # 否则若音频下载成功则重算;否则不设置(不阻塞主流程) + timings: Optional[list] = None + if supplied_timings: + timings = supplied_timings + logger.info("[lipsync] 使用前端预合成句子时间戳: job_id=%s sentences=%d", job.id, len(timings)) + elif audio_data and audio_duration > 0 and job.script_text: + try: + timings = compute_sentence_timings(audio_data, job.script_text, audio_duration) + logger.info( + "[lipsync] 后端重算句子时间戳: job_id=%s sentences=%d duration=%.2f", + job.id, + len(timings) if timings else 0, + audio_duration, + ) + except Exception as exc: + logger.warning("[lipsync] 句子时间戳计算失败(不阻塞): job_id=%s err=%s", job.id, exc) + + if timings: + job.sentence_timings = timings + + # 4. 签名 URL 并提交 MediaKit + video_url = self._sign_media_url(job.video_url) + signed_audio_url = self._sign_media_url(job.audio_url) + job.audio_url = signed_audio_url + + try: + result = self.client.submit_lipsync( + video_url=video_url, + audio_url=signed_audio_url, + enable_video_loop=job.enable_video_loop, + client_token=job.id, + ) + job.mediakit_task_id = result["task_id"] + job.status = "submitted" + job.submitted_at = datetime.now(timezone.utc) + self.db.commit() + logger.info( + "[lipsync] 直传音频已提交 MediaKit: job_id=%s task_id=%s", + job.id, + result["task_id"], + ) + except MediaKitError as exc: + job.status = "failed" + job.error_message = str(exc) + job.error_code = exc.code + logger.error("[lipsync] 直传音频提交 MediaKit 失败: job_id=%s err=%s", job.id, exc) + self.db.commit() + raise + # ── 创建任务 ────────────────────────────────────────────────────────── def create_job( @@ -150,6 +249,8 @@ class LipsyncService: user_id: str, video_url: str, audio_url: str = "", + audio_duration: Optional[float] = None, + sentence_timings: Optional[list] = None, voice_id: str = "", script_text: str = "", speed: float = 1.0, @@ -159,18 +260,24 @@ class LipsyncService: ) -> LipsyncJobModel: """创建对口型任务. - 两种输入模式: + 三种输入模式: - TTS 直生:voice_id + script_text(audio_url 留空) - → 先创建 DB 记录(状态 tts_processing),再 dispatch Celery 异步任务 - 执行 TTS 合成 + MediaKit 提交。API 响应 <1s。 - - 直接音频:提供 audio_url - → 同步提交 MediaKit,状态直接设为 submitted。 + → 创建 DB 记录(状态 tts_processing),dispatch Celery 异步任务(降级路径)。 + API 响应 <1s。 + - 直接音频:audio_url 非空 + 无 sentence_timings + → 同步下载音频 + 重算 timings + 提交 MediaKit(几秒完成)。 + - 预合成音频(#1845 新主路径):audio_url 非空 + 传 sentence_timings + → 同步 ffprobe 校验时长 + 写入 timings + 提交 MediaKit(~2-3s)。 Raises: - MediaKitError: 参数校验失败或 MediaKit 提交失败(仅直接音频模式) + MediaKitError: 参数校验失败或 MediaKit 提交失败 """ # 0. 输入校验 - if not audio_url: + is_pre_synth = bool(audio_url) and bool(sentence_timings) + is_direct_audio = bool(audio_url) and not is_pre_synth + is_tts_mode = not bool(audio_url) + + if is_tts_mode: if not (voice_id and script_text): raise MediaKitError( "必须提供 audio_url 或 voice_id+script_text", @@ -178,10 +285,13 @@ class LipsyncService: ) # TTS 模式:在 HTTP 请求中同步校验音色归属,快速失败 self._resolve_voice_id(voice_id, user_id) + elif is_pre_synth: + # 预合成模式:script_text 可空(因为 timings 已自带句子文本),但仍建议传 + if not isinstance(sentence_timings, list) or len(sentence_timings) == 0: + raise MediaKitError("预合成模式 sentence_timings 不能为空", code="InvalidInput") # 1. 创建数据库记录 job_id = str(uuid.uuid4()) - is_tts_mode = not bool(audio_url) job = LipsyncJobModel( id=job_id, user_id=user_id, @@ -192,20 +302,19 @@ class LipsyncService: voice_id=voice_id or "", script_text=script_text or "", speed=speed, - emotion=normalize_emotion(emotion), + emotion=normalize_emotion(emotion) if is_tts_mode else (emotion or ""), + # 音频直传(含预合成)直接进入 pending(后续同步改为 submitted);TTS 模式进入 tts_processing status="tts_processing" if is_tts_mode else "pending", ) self.db.add(job) self.db.flush() - # ⚠️ 必须先 commit 再发 Celery 任务,避免事务竞态: - # worker 是独立进程+独立DB连接,任务被消费(<4ms)时若本事务还未提交, - # worker 查询 job 会返回 None → 静默 return 不重试,job 永远卡在 tts_processing。 + # ⚠️ 必须先 commit 再发 Celery 任务 / 后续同步操作,避免事务竞态 self.db.commit() self.db.refresh(job) if is_tts_mode: - # 2a. TTS 模式:dispatch Celery 异步任务处理 TTS 合成 + MediaKit 提交 + # 2a. TTS 模式:dispatch Celery 异步任务处理 TTS 合成 + MediaKit 提交(降级路径) try: tts_synthesize_and_submit.apply_async( args=( @@ -218,8 +327,6 @@ class LipsyncService: ) ) except Exception as exc: - # 投递失败时立即把 job 标成 failed 并写入 error_message, - # 前端轮询时能直接看到失败原因,不会无限卡在 tts_processing。 logger.exception( "Celery 任务提交失败,TTS 任务已创建但未触发执行: job_id=%s err=%s", job_id, @@ -229,35 +336,102 @@ class LipsyncService: job.error_message = f"Celery 任务投递失败: {exc}" job.error_code = "AsyncDispatchFailed" job.updated_at = datetime.now(timezone.utc) - self.db.commit() # 投递失败也要落库失败状态 - else: - # 2b. 直接音频模式:同步签名并提交 MediaKit - video_url = self._sign_media_url(video_url) - if audio_url: - audio_url = self._sign_media_url(audio_url) - job.audio_url = audio_url - - try: - result = self.client.submit_lipsync( - video_url=video_url, - audio_url=audio_url, - enable_video_loop=enable_video_loop, - client_token=job_id, - ) - job.mediakit_task_id = result["task_id"] - job.status = "submitted" - job.submitted_at = datetime.now(timezone.utc) - self.db.commit() # submitted 状态落库 - except MediaKitError as exc: - job.status = "failed" - job.error_message = str(exc) - job.error_code = exc.code - logger.error("提交对口型任务失败: %s", exc) self.db.commit() - raise + else: + # 2b/2c. 直接音频 / 预合成音频:同步路径 + self._submit_audio_direct( + job=job, + supplied_timings=sentence_timings, + supplied_duration=audio_duration, + ) + self.db.refresh(job) return job + # ── TTS 预合成(#1845 步骤1「生成配音」同步接口使用) ────────────────── + + def preview_tts( + self, + *, + user_id: str, + voice_id: str, + script_text: str, + speed: float = 1.0, + emotion: str = "natural", + ) -> dict: + """同步做 TTS 合成 + 下载 + ffprobe + 句子时间戳计算. + + 不创建 LipsyncJob、不转存 OSS,直接返回 CosyVoice 临时 URL(~24h 有效期)。 + 耗时约 2-3 秒,由前端在步骤1点「生成配音」时同步等待。 + + Returns: + {"audio_url": str, "duration": float, "sentence_timings": list[dict]} + + Raises: + MediaKitError: TTS 合成失败 / 下载失败 / ffprobe 失败 + """ + # 1. 音色解析(校验克隆音色归属) + actual_voice_id = self._resolve_voice_id(voice_id, user_id) + cosyvoice = self._get_cosyvoice() + + # 2. TTS 合成(同步,~2-3s) + try: + result = cosyvoice.submit_synthesize_task( + text=script_text, + voice_id=actual_voice_id, + speed=speed, + emotion=normalize_emotion(emotion), + ) + except CosyVoiceError as exc: + raise MediaKitError(f"TTS 合成失败: {exc}", code="TTSSynthesisFailed") from exc + except ValueError as exc: + raise MediaKitError(f"TTS 参数错误: {exc}", code="TTSInvalidParam") from exc + + temp_url = result.get("audio_url", "") + if not temp_url: + raise MediaKitError("TTS 未返回音频 URL", code="TTSNoAudio") + + # 3. 下载音频到内存(用于 ffprobe + 静音检测) + try: + audio_data = safe_download_bytes( + temp_url, + purpose="tts_preview_audio", + allowed_mime_types=ALLOWED_AUDIO_MIME_TYPES, + timeout=60.0, + ) + except Exception as exc: + logger.warning("[tts-preview] TTS 音频下载失败,仍返回 audio_url: user_id=%s err=%s", user_id, exc) + return { + "audio_url": temp_url, + "duration": 0.0, + "sentence_timings": [], + } + + # 4. ffprobe 时长 + duration = probe_audio_duration(audio_data) + if duration <= 0: + logger.warning("[tts-preview] ffprobe 未返回有效时长,timings 留空: user_id=%s", user_id) + return { + "audio_url": temp_url, + "duration": 0.0, + "sentence_timings": [], + } + + # 5. 句子时间戳 + timings = compute_sentence_timings(audio_data, script_text, duration) + + logger.info( + "[tts-preview] TTS 预合成完成: user_id=%s duration=%.2f sentences=%d", + user_id, + duration, + len(timings), + ) + return { + "audio_url": temp_url, + "duration": round(duration, 2), + "sentence_timings": timings, + } + # ── 查询任务 ────────────────────────────────────────────────────────── def get_job(self, job_id: str, user_id: str) -> Optional[LipsyncJobModel]: @@ -291,11 +465,7 @@ class LipsyncService: # ── 更新任务状态(轮询) ────────────────────────────────────────────── def refresh_job_status(self, job_id: str, user_id: str) -> Optional[LipsyncJobModel]: - """从 MediaKit 拉取最新状态并更新本地记录. - - Returns: - 更新后的 Job,或 None(任务不存在/不属于该用户) - """ + """从 MediaKit 拉取最新状态并更新本地记录.""" job = self.get_job(job_id, user_id) if job is None: return None @@ -321,13 +491,12 @@ class LipsyncService: result = status_data.get("result", {}) job.status = STATUS_COMPLETED temp_url = result.get("video_url", "") - # 先以临时 URL 立即返回前端(前端可立即播放),再异步 Celery 任务转存自家 OSS(步骤⑦) job.output_video_url = temp_url job.output_duration = result.get("duration", 0.0) job.completed_at = datetime.now(timezone.utc) job.updated_at = datetime.now(timezone.utc) self.db.commit() - # 异步转存到自家 OSS(注意:必须在 commit 之后 dispatch,避免 commit 失败任务已发出) + # 异步转存自家 OSS try: from app.tasks.lipsync_tts import persist_output_video_task @@ -347,7 +516,6 @@ class LipsyncService: job.error_code = error.get("code", "TaskFailed") job.completed_at = datetime.now(timezone.utc) else: - # 中间状态(running/processing/queued 等)同步到 DB,避免前端永远卡在 submitted if isinstance(mk_status, str) and mk_status: job.status = mk_status job.updated_at = datetime.now(timezone.utc) @@ -356,10 +524,7 @@ class LipsyncService: return job def _persist_output_video(self, temp_url: str, job_id: str, user_id: str) -> str: - """将 MediaKit 输出的临时视频 URL 转存到自家 OSS. - - 失败时回退返回原始临时 URL,不影响任务完成。 - """ + """将 MediaKit 输出的临时视频 URL 转存到自家 OSS. 失败时回退返回原始临时 URL.""" if not temp_url: return "" try: @@ -379,27 +544,21 @@ class LipsyncService: return temp_url def _sign_media_url(self, url: str) -> str: - """对自家 OSS 私有桶 URL 重签长有效期预签名,供 MediaKit 拉取 / 前端播放。 - - - 裸 public_url(upload_file 返回,不带签名)→ 私有桶匿名访问 403,重签。 - - 已带签名但即将过期的 URL(如前端 1h 预签名)→ 抽 storage_key 后重签。 - - 外部 URL(CosyVoice/MediaKit 临时链接,非本桶 host)→ 原样透传。 - - 任何异常都降级原样返回,不阻断主流程。 - """ + """对自家 OSS 私有桶 URL 重签长有效期预签名.""" if not url: return url try: storage = get_shared_storage_service() public_base = getattr(storage, "public_url", "") if not isinstance(public_base, str) or not public_base: - return url # 无法判定归属,保守透传 + return url own_host = urlparse(public_base).netloc.lower() host = urlparse(url).netloc.lower() if not own_host or host != own_host: - return url # 非自家 OSS(外部临时链接),不处理 + return url # 外部临时链接原样透传 signed = storage.get_download_url(url, expires_seconds=MEDIAKIT_URL_TTL_SECONDS) return signed or url - except Exception as exc: # noqa: BLE001 - 签名失败不阻断,降级原 URL + except Exception as exc: logger.warning("对口型 URL 重签失败,原样返回: url_prefix=%s err=%s", url[:80], exc) return url diff --git a/apps/api/app/tasks/lipsync_tts.py b/apps/api/app/tasks/lipsync_tts.py index 7205bfe03..ee78a2bab 100644 --- a/apps/api/app/tasks/lipsync_tts.py +++ b/apps/api/app/tasks/lipsync_tts.py @@ -12,6 +12,10 @@ 注意:使用 @shared_task 而非绑定到某个 celery_app 实例, 确保任务能被 Worker 侧 celery_app 正确注册,同时 API 侧 send_task/apply_async 仍可正常调用。 + +#1845:句子时间戳计算已提取至 packages/domain/sentence_timings.py,本模块保留 +_ 开头别名兼容历史导入,但 _compute_sentence_timings/_split_script_into_sentences/ +_estimate_sentence_timings_by_chars 等内部函数已复用共享实现,避免重复代码。 """ import io @@ -21,6 +25,14 @@ from urllib.parse import urlparse from celery import shared_task +# 复用共享的句子时间戳工具(#1845 配音前置) +from packages.domain.sentence_timings import ( + compute_sentence_timings as _compute_sentence_timings, + estimate_sentence_timings_by_chars as _estimate_sentence_timings_by_chars, + probe_audio_duration, + split_script_into_sentences as _split_script_into_sentences, +) + logger = logging.getLogger(__name__) # MediaKit 预签名 URL 有效期(7天,秒),与 LipsyncService._sign_media_url 保持一致 @@ -54,164 +66,6 @@ def _sign_media_url(url: str) -> str: return url -def _split_script_into_sentences(script_text: str) -> list[str]: - """按句号/问号/感叹号/分号/逗号/换行分句(与前端 SENTENCE_SPLIT_RE 一致). - - 中文短视频文案习惯用「,」断小句(如"卖花的叫花无缺,卖姜的叫姜子牙"), - 必须把逗号也纳入分隔符,否则多句文案会被识别成一整句,导致 B-roll 时间戳错位。 - """ - import re - - text = (script_text or "").strip() - if not text: - return [] - parts = re.split(r"[。!?!??!;;,,\n\r]+", text) - return [p.strip() for p in parts if p.strip()] - - -def _compute_sentence_timings(audio_data: bytes, script_text: str, total_duration: float) -> list[dict]: - """基于 TTS 音频的静音检测,精确计算每句文案的起止时间. - - 使用 ffmpeg silencedetect 检测静音段,将静音点与句子边界对齐。 - 比字数比例估算准确得多。 - - Args: - audio_data: TTS 音频二进制数据(MP3) - script_text: 文案全文 - total_duration: 音频总时长(秒) - - Returns: - list[{"index": int, "text": str, "start_time": float, "end_time": float}] - """ - import re - import subprocess - import tempfile - - sentences = _split_script_into_sentences(script_text) - if not sentences: - return [] - - # 写入临时音频文件 - with tempfile.NamedTemporaryFile(suffix=".mp3", delete=False) as tmp: - tmp.write(audio_data) - tmp_path = tmp.name - - try: - # 用 ffmpeg silencedetect 检测静音段 - result = subprocess.run( - [ - "ffmpeg", - "-i", - tmp_path, - "-af", - "silencedetect=noise=-25dB:d=0.3", - "-f", - "null", - "-", - ], - capture_output=True, - text=True, - timeout=30, - ) - stderr = result.stderr or "" - - # 解析静音结束时间点(silence_end: X.XXX) - silence_ends = [] - for match in re.finditer(r"silence_end:\s*([\d.]+)", stderr): - t = float(match.group(1)) - if 0 < t < total_duration: - silence_ends.append(t) - - # 如果没有检测到足够的静音点,降级为字数比例估算 - if len(silence_ends) < len(sentences) - 1: - logger.warning( - "[sentence_timings] 静音点不足(%d < %d),降级为字数比例估算", - len(silence_ends), - len(sentences) - 1, - ) - return _estimate_sentence_timings_by_chars(sentences, total_duration) - - # 贪心匹配:N-1 个句子边界对应 N-1 个静音点 - # 按时间均匀分布期望值,选择最近的静音点 - n_boundaries = len(sentences) - 1 - boundaries = [] - used_indices = set() - - for i in range(n_boundaries): - # 期望的边界位置(按句子数量均匀分布) - expected_pos = (i + 1) / len(sentences) * total_duration - # 找最近的未使用静音点 - best_idx = None - best_dist = float("inf") - for j, t in enumerate(silence_ends): - if j in used_indices: - continue - dist = abs(t - expected_pos) - if dist < best_dist: - best_dist = dist - best_idx = j - if best_idx is not None: - used_indices.add(best_idx) - boundaries.append(silence_ends[best_idx]) - - boundaries.sort() - - # 构建 sentence_timings - timings = [] - prev_end = 0.0 - for i, sent in enumerate(sentences): - start = prev_end - end = boundaries[i] if i < len(boundaries) else total_duration - timings.append( - { - "index": i, - "text": sent, - "start_time": round(start, 2), - "end_time": round(end, 2), - } - ) - prev_end = end - - return timings - - except Exception as exc: - logger.warning("[sentence_timings] 静音检测异常,降级为字数比例估算: %s", exc) - return _estimate_sentence_timings_by_chars(sentences, total_duration) - finally: - import os - - try: - os.unlink(tmp_path) - except Exception: - pass - - -def _estimate_sentence_timings_by_chars(sentences: list[str], total_duration: float) -> list[dict]: - """降级方案:按字数比例估算句子时间(与原前端逻辑一致).""" - if not sentences or total_duration <= 0: - return [] - total_chars = sum(len(s.replace(r"\s", "")) for s in sentences) - if total_chars == 0: - return [] - - timings = [] - acc = 0 - for i, sent in enumerate(sentences): - chars = len(sent.replace(r"\s", "")) - start = (acc / total_chars) * total_duration - end = ((acc + chars) / total_chars) * total_duration - timings.append( - { - "index": i, - "text": sent, - "start_time": round(start, 2), - "end_time": round(end, 2), - } - ) - acc += chars - return timings - - @shared_task( bind=True, name="lipsync_tts.synthesize_and_submit", @@ -234,7 +88,8 @@ def tts_synthesize_and_submit( ): """异步执行 TTS 合成 + OSS 转存 + MediaKit 提交. - 在 Celery worker 中运行,不阻塞 HTTP 请求。 + 在 Celery worker 中运行,不阻塞 HTTP 请求。保留作为降级路径 + (预合成失败 / 旧版前端未传 audio_url 时走此路径)。 """ from app.services.mediakit_client import MediaKitError, get_mediakit_client from sqlalchemy.orm import Session as DBSession @@ -267,9 +122,6 @@ def tts_synthesize_and_submit( if job is None: # 事务竞态防御:API 在 commit 前投递了任务,worker 消费时事务尚未提交。 - # Celery 内置 autoretry_for 不支持"业务条件重试",这里手动 retry 3 次, - # 间隔递增(1s/3s/7s),让 API 事务有时间提交。 - # max_retries 由 self.request(retries) 维护;默认 self.max_retries=3 由装饰器 soft_time_limit 下方指定。 retries = getattr(self.request, "retries", 0) max_retries = 3 if retries < max_retries: @@ -340,7 +192,6 @@ def tts_synthesize_and_submit( # 2. 下载 TTS 音频到内存(用于 2.5 静音检测;不转存自家 OSS,直接使用 CosyVoice 临时 URL) audio_data: bytes | None = None - _st_tmp_path: str | None = None try: audio_data = safe_download_bytes( temp_url, @@ -349,7 +200,7 @@ def tts_synthesize_and_submit( "audio/mpeg", "audio/mp3", "audio/wav", - "audio/x-wav", # CosyVoice 部分接口返回 audio/x-wav,与 audio/wav 等价(RIFF/WAVE) + "audio/x-wav", # CosyVoice 部分接口返回 audio/x-wav "audio/mp4", "audio/x-m4a", }, @@ -361,7 +212,6 @@ def tts_synthesize_and_submit( len(audio_data) if audio_data else 0, ) except Exception as exc: - # 下载失败:audio_data 保持 None,2.5 静音检测会跳过;后续仍用 temp_url 提交 MediaKit logger.warning( "[lipsync_tts] TTS 音频下载失败,跳过静音检测,直接使用临时 URL 提交: job_id=%s err=%s", job_id, @@ -373,45 +223,16 @@ def tts_synthesize_and_submit( db.commit() - # 2.5 计算精确句子时间戳(基于 TTS 音频静音检测) - # 直接复用步骤 2 已下载到内存的 audio_data,避免重新下载 - import os as _os - + # 2.5 计算精确句子时间戳(基于 TTS 音频静音检测)—— 复用共享工具 try: - import subprocess as _sp - import tempfile as _tmpf - if not audio_data: logger.warning("[lipsync_tts] 无音频数据,跳过句子时间戳计算: job_id=%s", job_id) else: - # 写入临时文件供 ffprobe/ffmpeg 使用 - with _tmpf.NamedTemporaryFile(suffix=".mp3", delete=False) as _atmp: - _atmp.write(audio_data) - _st_tmp_path = _atmp.name - - # ffprobe 获取音频时长 - _probe_result = _sp.run( - [ - "ffprobe", - "-v", - "error", - "-show_entries", - "format=duration", - "-of", - "default=noprint_wrappers=1:nokey=1", - _st_tmp_path, - ], - capture_output=True, - text=True, - timeout=10, - ) - _audio_duration = float(_probe_result.stdout.strip()) if _probe_result.stdout.strip() else 0.0 + _audio_duration = probe_audio_duration(audio_data) logger.info( - "[lipsync_tts] 音频时长探测: job_id=%s duration=%.2f probe_stdout=%s probe_stderr=%s", + "[lipsync_tts] 音频时长探测: job_id=%s duration=%.2f", job_id, _audio_duration, - _probe_result.stdout.strip()[:50], - _probe_result.stderr.strip()[:100] if _probe_result.stderr else "", ) if _audio_duration > 0: @@ -428,22 +249,14 @@ def tts_synthesize_and_submit( logger.warning("[lipsync_tts] 句子时间戳计算返回空结果: job_id=%s", job_id) else: logger.warning( - "[lipsync_tts] ffprobe 未获取到有效时长,跳过句子时间戳: job_id=%s stdout=%s stderr=%s", + "[lipsync_tts] ffprobe 未获取到有效时长,跳过句子时间戳: job_id=%s", job_id, - _probe_result.stdout.strip()[:100], - _probe_result.stderr.strip()[:200] if _probe_result.stderr else "", ) db.commit() except Exception as _st_err: logger.warning( "[lipsync_tts] 句子时间戳计算失败(不影响主流程): job_id=%s err=%s", job_id, _st_err, exc_info=True ) - finally: - if _st_tmp_path: - try: - _os.unlink(_st_tmp_path) - except Exception: - pass # 3. 签名 URL 并提交到 MediaKit(复用模块内 _sign_media_url,避免对 LipsyncService 的耦合) audio_url = _sign_media_url(job.audio_url) @@ -495,12 +308,7 @@ def tts_synthesize_and_submit( default_retry_delay=30, ) def persist_output_video_task(job_id: str, user_id: str, temp_url: str): - """异步转存对口型输出视频到自家 OSS(步骤⑦ — 将同步阻塞挪到后台,加速前端响应). - - - MediaKit 返回 completed 后先以 temp_url 回前端(前端可立即播放临时 URL) - - Celery 后台下载 temp_url 并转存 OSS,成功后更新 job.output_video_url 为永久 URL - - 失败则保留 temp_url,不阻断主流程 - """ + """异步转存对口型输出视频到自家 OSS(步骤⑦ — 将同步阻塞挪到后台,加速前端响应).""" try: from worker_app.db import SessionLocal # type: ignore @@ -532,7 +340,6 @@ def persist_output_video_task(job_id: str, user_id: str, temp_url: str): storage = get_shared_storage_service() storage_key = f"lipsync-outputs/{user_id}/{job_id}.mp4" permanent_url = storage.upload_file(io.BytesIO(data), storage_key, content_type="video/mp4") - # 对自家 OSS URL 重签 7 天有效期预签名,供前端播放 final_url = _sign_media_url(permanent_url) if permanent_url else temp_url job.output_video_url = final_url job.updated_at = datetime.now(timezone.utc) diff --git a/apps/web/src/pages/ai-avatar/AiAvatarPage.tsx b/apps/web/src/pages/ai-avatar/AiAvatarPage.tsx index af31176c2..4d111f924 100644 --- a/apps/web/src/pages/ai-avatar/AiAvatarPage.tsx +++ b/apps/web/src/pages/ai-avatar/AiAvatarPage.tsx @@ -1,7 +1,7 @@ /** - * AI数字人 — 主页面(v3 两步骤版) - * 步骤1:出镜视频 / 配音库 / 文案 - * 步骤2:对口型预览(含插入画面)/ 标题配置 / 封面&生成 + * AI数字人 — 主页面(v3 两步骤版 + #1845 配音前置) + * 步骤1:出镜视频 / 配音库 / 文案 → 点击「🎵 生成配音」做 TTS 预合成(同步,~2-3s) + * 步骤2:对口型预览(音频已就绪、B-roll 句子时间戳立即可用)/ 标题配置 / 封面&生成 */ import React, { useState, useCallback, useEffect, useRef } from "react" import { message } from "antd" @@ -21,12 +21,13 @@ import { getAssetById, createLipsyncJob, getLipsyncJob, + previewTts, submitRender, getRenderJob, generateRenderSmartCover, } from "./api/aiAvatar" import { getOrCreateDefaultProject } from "@/api/projects" -import type { RenderJob } from "./types" +import type { RenderJob, SentenceTiming } from "./types" import { normalizeEmotion, buildTitleConfigPayload, @@ -50,6 +51,12 @@ const AiAvatarPage: React.FC = () => { cover: false, }) + /* ── #1845 TTS 预合成弹窗 ── */ + const [showTtsModal, setShowTtsModal] = useState(false) + const [ttsProgress, setTtsProgress] = useState(0) + const [ttsErrorMessage, setTtsErrorMessage] = useState("") + const ttsProgressTimerRef = useRef | null>(null) + /* ── 对口型生成弹窗 ── */ const [showLipsyncModal, setShowLipsyncModal] = useState(false) const [lipsyncStatus, setLipsyncStatus] = useState<"generating" | "completed" | "failed">( @@ -63,7 +70,7 @@ const AiAvatarPage: React.FC = () => { ) const [renderProgress, setRenderProgress] = useState(0) const [renderErrorMessage, setRenderErrorMessage] = useState("") - /* ── 当前渲染任务对象(轮询更新;用于封面区判断渲染是否完成) ── */ + /* ── 当前渲染任务对象 ── */ const [currentRenderJob, setCurrentRenderJob] = useState(null) /* ── 对口型轮询 ── */ @@ -75,8 +82,29 @@ const AiAvatarPage: React.FC = () => { setCollapsed((prev) => ({ ...prev, [key]: !prev[key] })) }, []) - /* ── 步骤切换 ── */ - const handleNextStep = useCallback(() => { + /* ── #1845 文案/音色/语速变更时重置 TTS 预合成状态,避免音频与文案不一致 ── */ + useEffect(() => { + if (state.ttsPreview.status !== "idle") { + state.resetTtsPreview() + } + // eslint-disable-next-line react-hooks/exhaustive-deps + }, [state.scriptText, state.selectedVoice?.voice_id, state.speed, state.emotion]) + + const _clearTtsProgressTimer = useCallback(() => { + if (ttsProgressTimerRef.current) { + clearInterval(ttsProgressTimerRef.current) + ttsProgressTimerRef.current = null + } + }, []) + + useEffect(() => { + return () => { + _clearTtsProgressTimer() + } + }, [_clearTtsProgressTimer]) + + /* ── #1845 步骤1:点击「🎵 生成配音」→ 同步 TTS 预合成 ── */ + const handleGenerateTts = useCallback(async () => { const missing: string[] = [] if (!state.selectedVideo) missing.push("出镜视频") if (!state.selectedVoice) missing.push("配音") @@ -85,45 +113,120 @@ const AiAvatarPage: React.FC = () => { message.warning(`请先完成${missing.join("、")}`) return } - setCurrentStep(2) - }, [state.selectedVideo, state.selectedVoice, state.scriptText]) + // 打开弹窗 & 启动模拟进度条 + setShowTtsModal(true) + setTtsProgress(0) + setTtsErrorMessage("") + state.setTtsPreview({ + audioUrl: null, + duration: 0, + sentenceTimings: [], + status: "generating", + error: null, + }) + + // 模拟进度:每 300ms +10%,到 90% 停住,真完成后瞬间到 100% + _clearTtsProgressTimer() + let fake = 0 + ttsProgressTimerRef.current = setInterval(() => { + fake = Math.min(fake + 10, 90) + setTtsProgress(fake) + if (fake >= 90) { + _clearTtsProgressTimer() + } + }, 300) + + try { + const res = await previewTts({ + voice_id: state.selectedVoice!.voice_id, + script_text: state.scriptText, + speed: state.speed, + emotion: normalizeEmotion(state.emotion), + }) + _clearTtsProgressTimer() + setTtsProgress(100) + state.setTtsPreview({ + audioUrl: res.audio_url, + duration: res.duration, + sentenceTimings: res.sentence_timings as SentenceTiming[], + status: "done", + error: null, + }) + message.success("配音合成完成") + } catch (err) { + _clearTtsProgressTimer() + const errMsg = + (err as { response?: { data?: { message?: string; detail?: unknown } } })?.response?.data?.message || + (err instanceof Error ? err.message : "配音合成失败,请重试") + setTtsErrorMessage(typeof errMsg === "string" ? errMsg : "配音合成失败,请重试") + state.setTtsPreview({ + audioUrl: null, + duration: 0, + sentenceTimings: [], + status: "failed", + error: typeof errMsg === "string" ? errMsg : "配音合成失败", + }) + } + // eslint-disable-next-line react-hooks/exhaustive-deps + }, [state.selectedVideo, state.selectedVoice, state.scriptText, state.speed, state.emotion]) + + const handleRetryTts = useCallback(() => { + handleGenerateTts() + }, [handleGenerateTts]) + + const handleTtsNext = useCallback(() => { + setShowTtsModal(false) + setTtsProgress(0) + setCurrentStep(2) + }, []) + + const handleCancelTts = useCallback(() => { + _clearTtsProgressTimer() + setShowTtsModal(false) + setTtsProgress(0) + setTtsErrorMessage("") + // 若用户在生成中途关闭,把状态重置回 idle,允许重新点击 + if (state.ttsPreview.status === "generating") { + state.resetTtsPreview() + } + }, [_clearTtsProgressTimer, state.ttsPreview.status, state]) + + /* ── 上一步(返回步骤1,不会丢失 TTS 预合成结果) ── */ const handlePrevStep = useCallback(() => { setCurrentStep(1) }, []) /* ── 对口型 ── */ const handleGenerateLipsync = useCallback(async () => { - // ② 缺项明确提示(#1809):不再静默 return const video = state.selectedVideo - const voice = state.selectedVoice const text = state.scriptText.trim() const missing: string[] = [] if (!video) missing.push("出镜视频") - if (!voice) missing.push("音色") if (!text) missing.push("文案") - if (missing.length > 0 || !video || !voice) { + if (missing.length > 0 || !video) { message.warning(`请先选择${missing.join("、")}`) return } + + // #1845:预合成模式下必须要有 audioUrl(理论上到了步骤2肯定有,兜底防御) + const isPreSynth = state.ttsPreview.status === "done" && !!state.ttsPreview.audioUrl + if (!isPreSynth && !state.selectedVoice) { + message.warning("请先选择音色或完成配音合成") + return + } + try { - // 显示生成弹窗 setShowLipsyncModal(true) setLipsyncStatus("generating") setLipsyncErrorMessage("") - // ① 先按素材 id 拿 file_url(#1809 补充:对齐后端新参数 video_url) console.log("[对口型] 开始生成:", { videoId: video.id, - voiceId: voice.voice_id, - voiceType: voice.type, + mode: isPreSynth ? "pre-synth" : "tts-direct", textLen: state.scriptText.length, }) const asset = await getAssetById(video.id) - console.log("[对口型] getAssetById 响应:", { - id: asset?.id, - file_url: asset?.file_url?.substring(0, 100), - }) const videoUrl = asset?.file_url if (!videoUrl) { console.error("[对口型] file_url 为空,asset:", asset) @@ -131,19 +234,34 @@ const AiAvatarPage: React.FC = () => { message.error("获取出镜视频播放地址失败,请重新选择素材") return } - // ② 模式A TTS直生:video_url + voice_id + script_text,语速/情绪英文枚举透传(#1822) - const payload = { - voice_id: voice.voice_id, - script_text: state.scriptText, - video_url: videoUrl, - speed: state.speed, // 语速 0.5~2.0 - emotion: normalizeEmotion(state.emotion), // natural/excited/calm/friendly + + let payload: Record + if (isPreSynth) { + // 预合成模式:传 audio_url + audio_duration + sentence_timings(后端直接提交 MediaKit,~2-3s) + payload = { + video_url: videoUrl, + audio_url: state.ttsPreview.audioUrl, + audio_duration: state.ttsPreview.duration, + sentence_timings: state.ttsPreview.sentenceTimings, + enable_video_loop: false, + } + } else { + // 降级:TTS 直生(旧路径,前端未预合成时) + payload = { + voice_id: state.selectedVoice!.voice_id, + script_text: state.scriptText, + video_url: videoUrl, + speed: state.speed, + emotion: normalizeEmotion(state.emotion), + } } console.log("[对口型] createLipsyncJob 请求:", payload) const job = await createLipsyncJob(payload) console.log("[对口型] createLipsyncJob 响应:", { id: job.id, status: job.status }) state.setLipsyncJob(job) - // 开始轮询 + + // 如果是预合成模式,后端会同步把状态置为 submitted(甚至可能已返回 running), + // 但仍需轮询等 completed if (lipsyncTimerRef.current) clearInterval(lipsyncTimerRef.current) lipsyncTimerRef.current = setInterval(async () => { try { @@ -180,7 +298,7 @@ const AiAvatarPage: React.FC = () => { message.error(err instanceof Error ? err.message : "对口型任务提交失败,请重试") } // eslint-disable-next-line react-hooks/exhaustive-deps - }, [state.selectedVideo, state.selectedVoice, state.scriptText, state.speed, state.emotion]) + }, [state.selectedVideo, state.selectedVoice, state.scriptText, state.speed, state.emotion, state.ttsPreview]) // 取消对口型生成 const handleCancelLipsync = useCallback(() => { @@ -201,6 +319,14 @@ const AiAvatarPage: React.FC = () => { } }, []) + /* ── B-roll 弹窗可用的句子时间戳:优先 lipsyncJob.sentence_timings,否则用 ttsPreview.sentenceTimings ── */ + const bRollSentenceTimings: SentenceTiming[] | undefined = + (state.lipsyncJob?.sentence_timings as SentenceTiming[] | undefined) ?? + (state.ttsPreview.status === "done" ? state.ttsPreview.sentenceTimings : undefined) + + /* ── B-roll 可用的总时长:优先 lipsyncJob.output_duration,否则用 ttsPreview.duration ── */ + const bRollDuration = state.lipsyncJob?.output_duration || state.ttsPreview.duration || 0 + /* ── 生成视频(含实时进度轮询) ── */ const handleGenerate = useCallback(async () => { if (!state.lipsyncJob || state.lipsyncJob.status !== "completed") { @@ -209,10 +335,9 @@ const AiAvatarPage: React.FC = () => { } state.setIsGenerating(true) try { - // 确保有 project_id(AI数字人入口独立,不在项目内,自动取默认项目;#1860 P0 bugfix) const defaultProject = await getOrCreateDefaultProject() - // 用 Canvas 预渲染标题为 PNG dataURL(所见即所得,后端用 overlay 直接叠加) + // 用 Canvas 预渲染标题为 PNG dataURL let titleImageDataUrl: string | null = null if (state.titleConfig.title?.trim()) { try { @@ -242,7 +367,6 @@ const AiAvatarPage: React.FC = () => { pip_scale: seg.pip_scale, })) as never, title_config: buildTitleConfigPayload(state.titleConfig, titleImageDataUrl), - // 封面不阻塞渲染:用户未选定封面时传空 dict,后端不生成封面;渲染完成后再单独抽帧 cover_config: state.coverConfig.smart_cover_url || (state.coverConfig.upload_url && !state.coverConfig.upload_url.startsWith("blob:")) @@ -250,7 +374,6 @@ const AiAvatarPage: React.FC = () => { : {}, }) - // 打开渲染进度弹窗,启动轮询 setShowRenderModal(true) setRenderStatus("generating") setRenderProgress(job.progress ?? 0) @@ -267,8 +390,6 @@ const AiAvatarPage: React.FC = () => { if (renderTimerRef.current) clearInterval(renderTimerRef.current) renderTimerRef.current = null setRenderStatus("completed") - // 渲染完成后:如果后端已返回封面(用户预上传/预设)则同步到前端; - // 否则不自动设置封面,由用户在封面区点击"智能获取封面"主动抽帧(步骤③④) if (updated.output_cover_url) { state.setCoverConfig((prev) => ({ ...prev, @@ -309,7 +430,7 @@ const AiAvatarPage: React.FC = () => { setRenderErrorMessage("") }, []) - /* ── 智能封面:从最终渲染成片抽帧(POST /renders/{id}/smart-cover,步骤③④) ── */ + /* ── 智能封面 ── */ const handleGenerateRenderSmartCover = useCallback( async (renderId: string): Promise<{ cover_url: string; message?: string }> => { try { @@ -334,7 +455,6 @@ const AiAvatarPage: React.FC = () => { return { cover_url: "", message: errMsg } } }, - // state.setCoverConfig 是 zustand action 引用稳定,eslint 不需要检查 // eslint-disable-next-line react-hooks/exhaustive-deps [], ) @@ -431,19 +551,33 @@ const AiAvatarPage: React.FC = () => { onOpenScriptModal={() => state.setShowScriptModal(true)} />
- + {state.ttsPreview.status === "done" && ( + + )}
)} - {/* ════ 步骤 2:对口型预览(含插入画面)/ 标题配置 / 封面&生成 ════ */} + {/* ════ 步骤 2:对口型预览 / 标题配置 / 封面&生成 ════ */} {currentStep === 2 && ( <> - {/* 面板:对口型预览 + 插入画面 */}
togglePanel("lipsync")}> 对口型预览 @@ -527,20 +661,135 @@ const AiAvatarPage: React.FC = () => { /> )} - {/* B-roll 编辑器弹窗 */} + {/* B-roll 编辑器弹窗 — #1845:timings 在对口型完成前就可用(来自 TTS 预合成) */} {state.showBRollModal && ( state.setShowBRollModal(false)} existingSegments={state.bRollSegments} scriptText={state.lipsyncJob?.script_text || state.scriptText} - outputDuration={state.lipsyncJob?.output_duration ?? 0} - sentenceTimings={state.lipsyncJob?.sentence_timings} + outputDuration={bRollDuration} + sentenceTimings={bRollSentenceTimings} onConfirm={state.addBRollSegment} onRemove={state.removeBRollSegment} /> )} + {/* #1845 TTS 预合成弹窗 */} + {showTtsModal && ( +
+
e.stopPropagation()}> +
+ 配音合成中 + {state.ttsPreview.status !== "generating" && ( + + )} +
+
+ {state.ttsPreview.status === "generating" && ( + <> +
+
+ 正在合成配音,请稍候… +
+
+ {ttsProgress}% +
+
+
+
+
+ 请勿关闭页面,完成后将自动提示 +
+ + )} + {state.ttsPreview.status === "done" && ( + <> +
✅
+
+ 配音合成完成,点击下一步继续 +
+
+ 音频时长 {state.ttsPreview.duration.toFixed(1)}s,共 {state.ttsPreview.sentenceTimings.length} 句 +
+ + )} + {state.ttsPreview.status === "failed" && ( + <> +
❌
+
配音合成失败
+ {ttsErrorMessage && ( +
+ {ttsErrorMessage} +
+ )} + + )} +
+
+ {state.ttsPreview.status === "generating" && ( + + )} + {state.ttsPreview.status === "done" && ( + + )} + {state.ttsPreview.status === "failed" && ( + <> + + + + )} +
+
+
+ )} + {/* 对口型生成弹窗 */} {showLipsyncModal && (
diff --git a/apps/web/src/pages/ai-avatar/api/aiAvatar.ts b/apps/web/src/pages/ai-avatar/api/aiAvatar.ts index 7f5b6a764..fec24a5ab 100644 --- a/apps/web/src/pages/ai-avatar/api/aiAvatar.ts +++ b/apps/web/src/pages/ai-avatar/api/aiAvatar.ts @@ -2,7 +2,7 @@ * AI数字人 — API 封装(#1822 契约对齐) */ import apiClient from "@/api/client" -import type { Script, LipsyncJob, RenderJob, BRollSegment } from "../types" +import type { Script, LipsyncJob, RenderJob, BRollSegment, SentenceTiming } from "../types" /* ── 文案库 ── */ export const getScripts = async (): Promise => { @@ -34,17 +34,28 @@ export const getAssetById = async (id: string): Promise<{ file_url?: string; id: return response.data } -/* ── 对口型(模式A:TTS 直生,后端内部合成音频;不要先调 TTS 拿 audio_url) ── */ +/* ── 对口型(支持三种模式) ── + * 1. TTS 直生(降级/旧版):传 voice_id + script_text(+speed/emotion),后端 Celery 异步合成 + * 2. 直接音频:传 video_url + audio_url,后端同步下载+算timings+提交MediaKit + * 3. 预合成音频(#1845 新主路径):先调 previewTts 拿 audio_url+sentence_timings, + * 再把 audio_url + audio_duration + sentence_timings 一起传过来,后端直接提交 MediaKit + */ export const createLipsyncJob = async (data: { - /** 人物视频 URL(MP4);由素材 id 经 getAssetById 拿 file_url,禁止传 video_asset_id */ + /** 人物视频 URL(MP4);由素材 id 经 getAssetById 拿 file_url */ video_url: string - /** 音色 ID(预置音色 或 克隆音色 profile UUID,后端会解析) */ - voice_id: string - /** 要合成的文案(手动输入或文案库内容) */ - script_text: string - /** 语速 0.5~2.0,默认 1.0 */ + /** 预合成/直接音频模式:音频 URL(#1845 步骤1 预合成的 CosyVoice 临时 URL,或外部音频 URL) */ + audio_url?: string + /** 预合成音频时长(秒),由 previewTts 返回 */ + audio_duration?: number + /** 预合成接口返回的句子时间戳(精确),后端直接写入 job */ + sentence_timings?: SentenceTiming[] + /** 音色 ID(TTS 直生模式用) */ + voice_id?: string + /** 要合成的文案(TTS 直生模式用) */ + script_text?: string + /** 语速 0.5~2.0,默认 1.0(TTS 直生模式用) */ speed?: number - /** 情绪英文枚举:natural/excited/calm/friendly */ + /** 情绪英文枚举:natural/excited/calm/friendly(TTS 直生模式用) */ emotion?: string enable_video_loop?: boolean project_id?: string @@ -53,6 +64,25 @@ export const createLipsyncJob = async (data: { return response.data } +/* ── #1845 TTS 预合成(步骤1「生成配音」同步接口,~2-3s) ── */ +export const previewTts = async (data: { + voice_id: string + script_text: string + speed?: number + emotion?: string +}): Promise<{ + audio_url: string + duration: number + sentence_timings: SentenceTiming[] +}> => { + const response = await apiClient.post<{ + audio_url: string + duration: number + sentence_timings: SentenceTiming[] + }>("/lipsync/tts-preview", data, { timeout: 30000 }) + return response.data +} + export const getLipsyncJob = async (id: string): Promise => { const response = await apiClient.get(`/lipsync/jobs/${id}`, { timeout: 60000 }) return response.data diff --git a/apps/web/src/pages/ai-avatar/hooks/useAiAvatar.ts b/apps/web/src/pages/ai-avatar/hooks/useAiAvatar.ts index 8553eb819..6300be24a 100644 --- a/apps/web/src/pages/ai-avatar/hooks/useAiAvatar.ts +++ b/apps/web/src/pages/ai-avatar/hooks/useAiAvatar.ts @@ -1,5 +1,5 @@ /** - * AI数字人 — 页面全局状态管理 hook(v3) + * AI数字人 — 页面全局状态管理 hook(v3 + #1845 配音前置) */ import { useState, useCallback } from "react" import type { AssetItem } from "@/api/assets" @@ -13,10 +13,19 @@ import { type BRollSegment, type AiAvatarTitleConfig, type AiAvatarCoverConfig, + type TtsPreviewResult, DEFAULT_TITLE_CONFIG, DEFAULT_COVER_CONFIG, } from "../types" +const DEFAULT_TTS_PREVIEW: TtsPreviewResult = { + audioUrl: null, + duration: 0, + sentenceTimings: [], + status: "idle", + error: null, +} + export function useAiAvatar() { /* ── 面板1:出镜视频 ── */ const [selectedVideo, setSelectedVideo] = useState(null) @@ -36,6 +45,9 @@ export function useAiAvatar() { const [showScriptModal, setShowScriptModal] = useState(false) const [showBRollModal, setShowBRollModal] = useState(false) + /* ── #1845 TTS 预合成(步骤1「生成配音」) ── */ + const [ttsPreview, setTtsPreview] = useState(DEFAULT_TTS_PREVIEW) + /* ── 面板3.5:B-roll ── */ const [bRollSegments, setBRollSegments] = useState([]) @@ -81,6 +93,7 @@ export function useAiAvatar() { setScript(null) setScriptText("") setLipsyncJob(null) + setTtsPreview(DEFAULT_TTS_PREVIEW) setBRollSegments([]) setTitleConfig(DEFAULT_TITLE_CONFIG) setCoverConfig(DEFAULT_COVER_CONFIG) @@ -118,6 +131,10 @@ export function useAiAvatar() { showBRollModal, setShowBRollModal, selectScript, + // #1845 TTS 预合成 + ttsPreview, + setTtsPreview, + resetTtsPreview: useCallback(() => setTtsPreview(DEFAULT_TTS_PREVIEW), []), // B-roll bRollSegments, addBRollSegment, diff --git a/apps/web/src/pages/ai-avatar/types.ts b/apps/web/src/pages/ai-avatar/types.ts index f12f1aa50..16384007a 100644 --- a/apps/web/src/pages/ai-avatar/types.ts +++ b/apps/web/src/pages/ai-avatar/types.ts @@ -28,6 +28,17 @@ export const VOICE_LANGUAGE_OPTIONS: { value: VoiceLanguage; label: string }[] = /* ── 对口型任务状态 ── */ export type LipsyncStatus = "idle" | "pending" | "processing" | "completed" | "failed" +/* ── TTS 预合成(#1845 配音前置:步骤1「生成配音」状态) ── */ +export type TtsPreviewStatus = "idle" | "generating" | "done" | "failed" + +export interface TtsPreviewResult { + audioUrl: string | null + duration: number + sentenceTimings: SentenceTiming[] + status: TtsPreviewStatus + error: string | null +} + /* ── 文案 ── */ export interface Script { id: string diff --git a/packages/domain/sentence_timings.py b/packages/domain/sentence_timings.py new file mode 100644 index 000000000..cfeba7ce7 --- /dev/null +++ b/packages/domain/sentence_timings.py @@ -0,0 +1,205 @@ +"""共享的句子时间戳计算工具 — 供 Celery TTS 任务和 /lipsync/tts-preview 同步接口复用. + +- `_split_script_into_sentences`: 按标点分句(中英文逗号/句号/问号/感叹号/分号/换行) +- `_estimate_sentence_timings_by_chars`: 按字数比例估算(静音检测失败时降级) +- `_probe_audio_duration`: ffprobe 读取音频时长 +- `compute_sentence_timings`: 基于 ffmpeg silencedetect 精确计算每句起止时间 +""" + +from __future__ import annotations + +import logging +import os +import re +import subprocess +import tempfile +from typing import Optional + +logger = logging.getLogger(__name__) + + +def split_script_into_sentences(script_text: str) -> list[str]: + """按句号/问号/感叹号/分号/逗号/换行分句(与前端 SENTENCE_SPLIT_RE 一致). + + 中文短视频文案习惯用「,」断小句(如"卖花的叫花无缺,卖姜的叫姜子牙"), + 必须把逗号也纳入分隔符,否则多句文案会被识别成一整句,导致 B-roll 时间戳错位。 + """ + text = (script_text or "").strip() + if not text: + return [] + parts = re.split(r"[。!?!??!;;,,\n\r]+", text) + return [p.strip() for p in parts if p.strip()] + + +def estimate_sentence_timings_by_chars(sentences: list[str], total_duration: float) -> list[dict]: + """降级方案:按字数比例估算句子时间(与原前端逻辑一致).""" + if not sentences or total_duration <= 0: + return [] + total_chars = sum(len(s.replace(r"\s", "")) for s in sentences) + if total_chars == 0: + return [] + + timings = [] + acc = 0 + for i, sent in enumerate(sentences): + chars = len(sent.replace(r"\s", "")) + start = (acc / total_chars) * total_duration + end = ((acc + chars) / total_chars) * total_duration + timings.append( + { + "index": i, + "text": sent, + "start_time": round(start, 2), + "end_time": round(end, 2), + } + ) + acc += chars + return timings + + +def probe_audio_duration(audio_data: bytes, timeout: int = 10) -> float: + """用 ffprobe 读取音频字节流的时长(秒). + + Returns: + 时长(秒),失败返回 0.0 + """ + if not audio_data: + return 0.0 + tmp_path: Optional[str] = None + try: + with tempfile.NamedTemporaryFile(suffix=".mp3", delete=False) as tmp: + tmp.write(audio_data) + tmp_path = tmp.name + result = subprocess.run( + [ + "ffprobe", + "-v", + "error", + "-show_entries", + "format=duration", + "-of", + "default=noprint_wrappers=1:nokey=1", + tmp_path, + ], + capture_output=True, + text=True, + timeout=timeout, + ) + stdout = (result.stdout or "").strip() + if not stdout: + logger.warning("[sentence_timings] ffprobe 无输出: stderr=%s", (result.stderr or "")[:200]) + return 0.0 + return float(stdout) + except Exception as exc: + logger.warning("[sentence_timings] ffprobe 时长探测失败: %s", exc) + return 0.0 + finally: + if tmp_path: + try: + os.unlink(tmp_path) + except Exception: + pass + + +def compute_sentence_timings(audio_data: bytes, script_text: str, total_duration: float) -> list[dict]: + """基于 TTS 音频的静音检测,精确计算每句文案的起止时间. + + 使用 ffmpeg silencedetect 检测静音段,将静音点与句子边界对齐。 + 比字数比例估算准确得多。 + + Args: + audio_data: TTS 音频二进制数据(MP3) + script_text: 文案全文 + total_duration: 音频总时长(秒) + + Returns: + list[{"index": int, "text": str, "start_time": float, "end_time": float}] + """ + sentences = split_script_into_sentences(script_text) + if not sentences: + return [] + + tmp_path: Optional[str] = None + try: + with tempfile.NamedTemporaryFile(suffix=".mp3", delete=False) as tmp: + tmp.write(audio_data) + tmp_path = tmp.name + + result = subprocess.run( + [ + "ffmpeg", + "-i", + tmp_path, + "-af", + "silencedetect=noise=-25dB:d=0.3", + "-f", + "null", + "-", + ], + capture_output=True, + text=True, + timeout=30, + ) + stderr = result.stderr or "" + + silence_ends = [] + for match in re.finditer(r"silence_end:\s*([\d.]+)", stderr): + t = float(match.group(1)) + if 0 < t < total_duration: + silence_ends.append(t) + + if len(silence_ends) < len(sentences) - 1: + logger.warning( + "[sentence_timings] 静音点不足(%d < %d),降级为字数比例估算", + len(silence_ends), + len(sentences) - 1, + ) + return estimate_sentence_timings_by_chars(sentences, total_duration) + + n_boundaries = len(sentences) - 1 + boundaries = [] + used_indices = set() + + for i in range(n_boundaries): + expected_pos = (i + 1) / len(sentences) * total_duration + best_idx = None + best_dist = float("inf") + for j, t in enumerate(silence_ends): + if j in used_indices: + continue + dist = abs(t - expected_pos) + if dist < best_dist: + best_dist = dist + best_idx = j + if best_idx is not None: + used_indices.add(best_idx) + boundaries.append(silence_ends[best_idx]) + + boundaries.sort() + + timings = [] + prev_end = 0.0 + for i, sent in enumerate(sentences): + start = prev_end + end = boundaries[i] if i < len(boundaries) else total_duration + timings.append( + { + "index": i, + "text": sent, + "start_time": round(start, 2), + "end_time": round(end, 2), + } + ) + prev_end = end + + return timings + + except Exception as exc: + logger.warning("[sentence_timings] 静音检测异常,降级为字数比例估算: %s", exc) + return estimate_sentence_timings_by_chars(sentences, total_duration) + finally: + if tmp_path: + try: + os.unlink(tmp_path) + except Exception: + pass diff --git a/tests/unit/test_lipsync_tts_preview.py b/tests/unit/test_lipsync_tts_preview.py new file mode 100644 index 000000000..0dd13b9de --- /dev/null +++ b/tests/unit/test_lipsync_tts_preview.py @@ -0,0 +1,168 @@ +"""AI 数字人 对口型 TTS 预合成接口(#1845)单元测试 — 覆盖 LipsyncService.preview_tts 成功/失败路径. + +直接调用 LipsyncService.preview_tts(),mock CosyVoiceService / safe_download_bytes / ffprobe, +验证返回结构、错误码、与共享 sentence_timings 工具的协作。 +""" + +import os +from unittest.mock import MagicMock, patch + +import pytest + +os.environ.setdefault("JWT_SECRET_KEY", "dev-secret-key-for-testing") + + +def _make_service( + *, + cosyvoice=None, + download_bytes=b"FAKE_MP3_DATA", + download_error=None, + ffprobe_duration=5.0, + timings_result=None, +): + """构造 LipsyncService 并把 CosyVoiceService/safe_download_bytes/probe/compute 全部 mock 掉。""" + from app.services.lipsync_service import LipsyncService + + db = MagicMock() + + # 构造唯一的 cosyvoice mock 实例,便于断言 + _cosy_inst = MagicMock() + if cosyvoice is None: + _cosy_inst.submit_synthesize_task.return_value = {"audio_url": "https://cosy.example.com/tts.mp3"} + elif isinstance(cosyvoice, Exception): + _cosy_inst.submit_synthesize_task.side_effect = cosyvoice + else: + _cosy_inst.submit_synthesize_task.return_value = cosyvoice + + def _fake_get_cosyvoice(self): # noqa: ARG001 + return _cosy_inst + + def _fake_resolve_voice_id(self, voice_id, user_id): # noqa: ARG001 + return voice_id + + svc = LipsyncService(db=db, client=MagicMock(), voice_clone_repo=MagicMock()) + svc._cosyvoice = _cosy_inst + patch.object(LipsyncService, "_get_cosyvoice", _fake_get_cosyvoice).start() + patch.object(LipsyncService, "_resolve_voice_id", _fake_resolve_voice_id).start() + + # mock safe_download_bytes + if download_error is not None: + patch( + "app.services.lipsync_service.safe_download_bytes", + side_effect=download_error, + ).start() + else: + patch( + "app.services.lipsync_service.safe_download_bytes", + return_value=download_bytes, + ).start() + + # mock probe_audio_duration(patch 到 lipsync_service 模块的命名空间) + patch( + "app.services.lipsync_service.probe_audio_duration", + return_value=ffprobe_duration, + ).start() + + # mock compute_sentence_timings + default_timings = [ + {"index": 0, "text": "你好", "start_time": 0.0, "end_time": 1.5}, + {"index": 1, "text": "世界", "start_time": 1.5, "end_time": 5.0}, + ] + patch( + "app.services.lipsync_service.compute_sentence_timings", + return_value=timings_result if timings_result is not None else default_timings, + ).start() + + svc.__dict__["_test_cosy"] = _cosy_inst + return svc + + +def test_preview_tts_success(): + """正常路径:TTS 合成成功 → 下载 → ffprobe → 计算 timings,返回完整结构。""" + svc = _make_service(ffprobe_duration=5.0) + try: + result = svc.preview_tts( + user_id="user-1", + voice_id="longxiaochun", + script_text="你好,世界", + speed=1.0, + emotion="natural", + ) + assert result["audio_url"] == "https://cosy.example.com/tts.mp3" + assert result["duration"] == 5.0 + assert isinstance(result["sentence_timings"], list) + assert len(result["sentence_timings"]) == 2 + assert result["sentence_timings"][0]["text"] == "你好" + cosy = svc.__dict__["_test_cosy"] + cosy.submit_synthesize_task.assert_called_once() + kwargs = cosy.submit_synthesize_task.call_args.kwargs + assert kwargs["text"] == "你好,世界" + assert kwargs["voice_id"] == "longxiaochun" + finally: + patch.stopall() + + +def test_preview_tts_cosyvoice_error(): + """CosyVoice 抛错:应该包装成 MediaKitError 抛出。""" + from packages.application.cosyvoice_service import CosyVoiceError + from app.services.mediakit_client import MediaKitError + + svc = _make_service(cosyvoice=CosyVoiceError("cosyvoice down")) + try: + with pytest.raises(MediaKitError): + svc.preview_tts( + user_id="user-1", + voice_id="longxiaochun", + script_text="你好", + ) + finally: + patch.stopall() + + +def test_preview_tts_download_fail_still_returns_url(): + """音频下载失败:不抛错,返回 audio_url + 空 timings,前端仍能继续(降级)。""" + svc = _make_service(download_error=RuntimeError("network down")) + try: + result = svc.preview_tts( + user_id="user-1", + voice_id="longxiaochun", + script_text="你好,世界", + ) + assert result["audio_url"] == "https://cosy.example.com/tts.mp3" + assert result["duration"] == 0.0 + assert result["sentence_timings"] == [] + finally: + patch.stopall() + + +def test_preview_tts_ffprobe_zero_duration(): + """ffprobe 返回 0:timings 为空,不抛错。""" + svc = _make_service(ffprobe_duration=0.0) + try: + result = svc.preview_tts( + user_id="user-1", + voice_id="longxiaochun", + script_text="你好", + ) + assert result["audio_url"] + assert result["duration"] == 0.0 + assert result["sentence_timings"] == [] + finally: + patch.stopall() + + +def test_preview_tts_no_audio_url_in_response(): + """CosyVoice 返回无 audio_url:抛 MediaKitError TTSNoAudio。""" + from app.services.mediakit_client import MediaKitError + + svc = _make_service(cosyvoice={"audio_url": ""}) + try: + with pytest.raises(MediaKitError) as exc_info: + svc.preview_tts( + user_id="user-1", + voice_id="longxiaochun", + script_text="你好", + ) + assert exc_info.value.code == "TTSNoAudio" + finally: + patch.stopall() -- 2.54.0 From d12632e32a28d76bb2c6ab20f959dc1e25133aea Mon Sep 17 00:00:00 2001 From: CI Bot Date: Sat, 12 Sep 2026 19:40:10 +0000 Subject: [PATCH 2/4] style: auto-format with black + isort + ruff + prettier [skip ci-format-check] --- apps/api/app/services/lipsync_service.py | 6 +++-- apps/api/app/tasks/lipsync_tts.py | 4 +-- apps/web/src/pages/ai-avatar/AiAvatarPage.tsx | 26 +++++++++++++++---- tests/unit/test_lipsync_tts_preview.py | 3 ++- 4 files changed, 28 insertions(+), 11 deletions(-) diff --git a/apps/api/app/services/lipsync_service.py b/apps/api/app/services/lipsync_service.py index 9fed8efd6..a24019fe3 100644 --- a/apps/api/app/services/lipsync_service.py +++ b/apps/api/app/services/lipsync_service.py @@ -189,7 +189,9 @@ class LipsyncService: audio_duration = probe_audio_duration(audio_data) if audio_duration <= 0 and supplied_duration and supplied_duration > 0: audio_duration = supplied_duration - logger.info("[lipsync] ffprobe 失败,使用前端传入的预合成时长: job_id=%s duration=%.2f", job.id, audio_duration) + logger.info( + "[lipsync] ffprobe 失败,使用前端传入的预合成时长: job_id=%s duration=%.2f", job.id, audio_duration + ) # 3. 句子时间戳:优先用前端预合成传入的 timings(后端预合成接口已经算过,可信); # 否则若音频下载成功则重算;否则不设置(不阻塞主流程) @@ -274,7 +276,7 @@ class LipsyncService: """ # 0. 输入校验 is_pre_synth = bool(audio_url) and bool(sentence_timings) - is_direct_audio = bool(audio_url) and not is_pre_synth + bool(audio_url) and not is_pre_synth is_tts_mode = not bool(audio_url) if is_tts_mode: diff --git a/apps/api/app/tasks/lipsync_tts.py b/apps/api/app/tasks/lipsync_tts.py index ee78a2bab..2f75e9b41 100644 --- a/apps/api/app/tasks/lipsync_tts.py +++ b/apps/api/app/tasks/lipsync_tts.py @@ -26,11 +26,9 @@ from urllib.parse import urlparse from celery import shared_task # 复用共享的句子时间戳工具(#1845 配音前置) +from packages.domain.sentence_timings import compute_sentence_timings as _compute_sentence_timings from packages.domain.sentence_timings import ( - compute_sentence_timings as _compute_sentence_timings, - estimate_sentence_timings_by_chars as _estimate_sentence_timings_by_chars, probe_audio_duration, - split_script_into_sentences as _split_script_into_sentences, ) logger = logging.getLogger(__name__) diff --git a/apps/web/src/pages/ai-avatar/AiAvatarPage.tsx b/apps/web/src/pages/ai-avatar/AiAvatarPage.tsx index 4d111f924..1e9fb4759 100644 --- a/apps/web/src/pages/ai-avatar/AiAvatarPage.tsx +++ b/apps/web/src/pages/ai-avatar/AiAvatarPage.tsx @@ -157,8 +157,8 @@ const AiAvatarPage: React.FC = () => { } catch (err) { _clearTtsProgressTimer() const errMsg = - (err as { response?: { data?: { message?: string; detail?: unknown } } })?.response?.data?.message || - (err instanceof Error ? err.message : "配音合成失败,请重试") + (err as { response?: { data?: { message?: string; detail?: unknown } } })?.response?.data + ?.message || (err instanceof Error ? err.message : "配音合成失败,请重试") setTtsErrorMessage(typeof errMsg === "string" ? errMsg : "配音合成失败,请重试") state.setTtsPreview({ audioUrl: null, @@ -298,7 +298,14 @@ const AiAvatarPage: React.FC = () => { message.error(err instanceof Error ? err.message : "对口型任务提交失败,请重试") } // eslint-disable-next-line react-hooks/exhaustive-deps - }, [state.selectedVideo, state.selectedVoice, state.scriptText, state.speed, state.emotion, state.ttsPreview]) + }, [ + state.selectedVideo, + state.selectedVoice, + state.scriptText, + state.speed, + state.emotion, + state.ttsPreview, + ]) // 取消对口型生成 const handleCancelLipsync = useCallback(() => { @@ -744,7 +751,8 @@ const AiAvatarPage: React.FC = () => { 配音合成完成,点击下一步继续
- 音频时长 {state.ttsPreview.duration.toFixed(1)}s,共 {state.ttsPreview.sentenceTimings.length} 句 + 音频时长 {state.ttsPreview.duration.toFixed(1)}s,共{" "} + {state.ttsPreview.sentenceTimings.length} 句
)} @@ -753,7 +761,15 @@ const AiAvatarPage: React.FC = () => {
❌
配音合成失败
{ttsErrorMessage && ( -
+
{ttsErrorMessage}
)} diff --git a/tests/unit/test_lipsync_tts_preview.py b/tests/unit/test_lipsync_tts_preview.py index 0dd13b9de..a3c97ca24 100644 --- a/tests/unit/test_lipsync_tts_preview.py +++ b/tests/unit/test_lipsync_tts_preview.py @@ -104,9 +104,10 @@ def test_preview_tts_success(): def test_preview_tts_cosyvoice_error(): """CosyVoice 抛错:应该包装成 MediaKitError 抛出。""" - from packages.application.cosyvoice_service import CosyVoiceError from app.services.mediakit_client import MediaKitError + from packages.application.cosyvoice_service import CosyVoiceError + svc = _make_service(cosyvoice=CosyVoiceError("cosyvoice down")) try: with pytest.raises(MediaKitError): -- 2.54.0 From c2de565e09b12949639d7843308752271ebb658e Mon Sep 17 00:00:00 2001 From: xiaoxia Date: Sun, 13 Sep 2026 04:06:20 +0800 Subject: [PATCH 3/4] =?UTF-8?q?fix(ai-avatar):=20=E4=BF=AE=E5=A4=8D?= =?UTF-8?q?=E9=85=8D=E9=9F=B3=E5=89=8D=E7=BD=AEPR=E7=9A=84CI=E9=94=99?= =?UTF-8?q?=E8=AF=AF=E2=80=94=E2=80=94test=20import=E8=B7=AF=E5=BE=84/TS?= =?UTF-8?q?=E7=B1=BB=E5=9E=8B/eslint=20warning?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit - test_sentence_timings.py 改为从 packages.domain.sentence_timings 导入(函数已提取) - AiAvatarPage.tsx payload类型改为Parameters[0]消除TS2345 - 去掉useCallback中重复的state.ttsPreview.status依赖(state已覆盖) --- apps/web/src/pages/ai-avatar/AiAvatarPage.tsx | 7 ++++--- tests/unit/test_sentence_timings.py | 10 +++++----- 2 files changed, 9 insertions(+), 8 deletions(-) diff --git a/apps/web/src/pages/ai-avatar/AiAvatarPage.tsx b/apps/web/src/pages/ai-avatar/AiAvatarPage.tsx index 1e9fb4759..8f899817f 100644 --- a/apps/web/src/pages/ai-avatar/AiAvatarPage.tsx +++ b/apps/web/src/pages/ai-avatar/AiAvatarPage.tsx @@ -190,7 +190,7 @@ const AiAvatarPage: React.FC = () => { if (state.ttsPreview.status === "generating") { state.resetTtsPreview() } - }, [_clearTtsProgressTimer, state.ttsPreview.status, state]) + }, [_clearTtsProgressTimer, state]) /* ── 上一步(返回步骤1,不会丢失 TTS 预合成结果) ── */ const handlePrevStep = useCallback(() => { @@ -235,12 +235,13 @@ const AiAvatarPage: React.FC = () => { return } - let payload: Record + type LipsyncPayload = Parameters[0] + let payload: LipsyncPayload if (isPreSynth) { // 预合成模式:传 audio_url + audio_duration + sentence_timings(后端直接提交 MediaKit,~2-3s) payload = { video_url: videoUrl, - audio_url: state.ttsPreview.audioUrl, + audio_url: state.ttsPreview.audioUrl!, audio_duration: state.ttsPreview.duration, sentence_timings: state.ttsPreview.sentenceTimings, enable_video_loop: false, diff --git a/tests/unit/test_sentence_timings.py b/tests/unit/test_sentence_timings.py index c9a00979e..119d68cc1 100644 --- a/tests/unit/test_sentence_timings.py +++ b/tests/unit/test_sentence_timings.py @@ -1,4 +1,4 @@ -"""Tests for sentence timing functions in lipsync_tts.""" +"""Tests for sentence timing functions (now in packages/domain/sentence_timings.py).""" import os import subprocess @@ -6,10 +6,10 @@ import tempfile import unittest from unittest.mock import MagicMock, patch -from apps.api.app.tasks.lipsync_tts import ( - _compute_sentence_timings, - _estimate_sentence_timings_by_chars, - _split_script_into_sentences, +from packages.domain.sentence_timings import ( + compute_sentence_timings as _compute_sentence_timings, + estimate_sentence_timings_by_chars as _estimate_sentence_timings_by_chars, + split_script_into_sentences as _split_script_into_sentences, ) -- 2.54.0 From ebfcfa419156325b595082510f8130e3bc0ac892 Mon Sep 17 00:00:00 2001 From: CI Bot Date: Sat, 12 Sep 2026 20:12:36 +0000 Subject: [PATCH 4/4] style: auto-format with black + isort + ruff + prettier [skip ci-format-check] --- tests/unit/test_sentence_timings.py | 8 +++----- 1 file changed, 3 insertions(+), 5 deletions(-) diff --git a/tests/unit/test_sentence_timings.py b/tests/unit/test_sentence_timings.py index 119d68cc1..3551d03ab 100644 --- a/tests/unit/test_sentence_timings.py +++ b/tests/unit/test_sentence_timings.py @@ -6,11 +6,9 @@ import tempfile import unittest from unittest.mock import MagicMock, patch -from packages.domain.sentence_timings import ( - compute_sentence_timings as _compute_sentence_timings, - estimate_sentence_timings_by_chars as _estimate_sentence_timings_by_chars, - split_script_into_sentences as _split_script_into_sentences, -) +from packages.domain.sentence_timings import compute_sentence_timings as _compute_sentence_timings +from packages.domain.sentence_timings import estimate_sentence_timings_by_chars as _estimate_sentence_timings_by_chars +from packages.domain.sentence_timings import split_script_into_sentences as _split_script_into_sentences class TestSplitScriptIntoSentences(unittest.TestCase): -- 2.54.0