diff --git a/apps/api/app/api/routes/ai_avatar_render.py b/apps/api/app/api/routes/ai_avatar_render.py index b05818856..a031fb8d2 100644 --- a/apps/api/app/api/routes/ai_avatar_render.py +++ b/apps/api/app/api/routes/ai_avatar_render.py @@ -18,7 +18,6 @@ from app.dependencies import get_db_session from app.schemas.ai_avatar_render import ( AiAvatarRenderJobResponse, CreateAiAvatarRenderRequest, - SmartCoverRequest, SmartCoverResponse, ) from app.services.ai_avatar_cover_service import generate_smart_cover @@ -191,57 +190,6 @@ def retry_render_job( return AiAvatarRenderJobResponse.model_validate(job) -# ── POST /smart-cover — 智能获取封面(MediaKit 抽帧 + 评分选帧)──────── - - -@router.post("/smart-cover", response_model=SmartCoverResponse) -def generate_avatar_smart_cover( - body: SmartCoverRequest, - current_user: AuthenticatedUser = Depends(get_current_user), -) -> SmartCoverResponse: - """智能获取数字人视频封面. - - 复用智能剪辑的 MediaKit 抽帧 + 质量评分选最佳帧逻辑(非 FFmpeg 简单截帧), - 并将选中帧转存到自家 OSS,返回非临时的封面公网 URL。 - - 前端「智能获取封面」按钮可直接调用本接口;不依赖渲染任务完成。 - """ - video_url = (body.video_url or "").strip() - if not video_url.startswith(("http://", "https://")): - raise HTTPException(status_code=400, detail="video_url 必须是合法的 HTTP/HTTPS URL") - - # 兼容:顶层 title_image_dataurl 透传进 title_config(前端可以放在任一位置) - title_config = getattr(body, "title_config", None) - top_dataurl = getattr(body, "title_image_dataurl", None) - if top_dataurl and isinstance(title_config, dict) and not title_config.get("title_image_dataurl"): - title_config = {**title_config, "title_image_dataurl": top_dataurl} - - try: - cover_url = generate_smart_cover( - video_url, - max_frames=body.max_frames, - title_config=title_config, - ) - except Exception as exc: - logger.error( - "智能封面生成异常: user=%s video_url=%s err=%s", - current_user.user.id, - video_url[:80], - exc, - exc_info=True, - ) - cover_url = "" - - if not cover_url: - return SmartCoverResponse( - cover_url="", - status="fallback_failed", - message="智能抽帧失败(MediaKit 不可用或抽帧异常),请稍后重试", - ) - logger.info("智能封面生成成功: user=%s cover_url=%s", current_user.user.id, cover_url[:120]) - return SmartCoverResponse(cover_url=cover_url, status="completed") - - # ── POST /{job_id}/smart-cover — 从最终成片智能抽封面(步骤②)──────── @@ -269,7 +217,7 @@ def generate_render_smart_cover( raise HTTPException(status_code=400, detail="渲染成片视频 URL 为空") try: - # 成片已叠加标题,不传 title_config 避免双重叠加 + # 从最终成片抽帧,帧本身已含标题/B-roll,直接转存 OSS cover_url = generate_smart_cover(video_url, job_id=job_id, max_frames=5) except Exception as exc: logger.error( diff --git a/apps/api/app/schemas/ai_avatar_render.py b/apps/api/app/schemas/ai_avatar_render.py index a574b4012..876363c7b 100644 --- a/apps/api/app/schemas/ai_avatar_render.py +++ b/apps/api/app/schemas/ai_avatar_render.py @@ -52,7 +52,7 @@ class CreateAiAvatarRenderRequest(BaseModel): lipsync_job_id: str = Field(..., description="对口型任务 ID") script_id: str = Field("", description="文案 ID(选自文案库时传;手动输入文案直生场景可留空)") b_roll_segments: list[BRollSegment] = Field(default_factory=list, description="B-roll 片段列表") - title_config: dict[str, Any] = Field(default_factory=dict, description="标题配置") + title_config: dict[str, Any] = Field(default_factory=dict, description="标题配置(可含 title_image_dataurl:前端 Canvas 渲染的标题 PNG dataURL)") cover_config: dict[str, Any] = Field(default_factory=dict, description="封面配置") project_id: str = Field("", description="项目 ID") @@ -67,7 +67,6 @@ class CreateAiAvatarRenderRequest(BaseModel): @field_validator("script_id") @classmethod def validate_script_id(cls, v: str) -> str: - # script_id 可选:手动输入文案(TTS 直生)场景不关联文案库条目 return (v or "").strip() @@ -109,27 +108,8 @@ class AiAvatarRenderProgressResponse(BaseModel): error_message: str -class SmartCoverRequest(BaseModel): - """智能封面请求 — MediaKit 抽帧 + 质量评分选最佳帧 + 可选标题叠加.""" - - video_url: str = Field(..., description="数字人视频 URL(对口型/渲染成片)") - max_frames: int = Field(5, ge=1, le=10, description="抽帧数量(默认 5)") - title_config: Optional[dict[str, Any]] = Field( - None, - description="标题配置;传入时在封面上叠加标题。" - "若 title_config 含 title_image_dataurl(前端 Canvas 渲染的 PNG dataURL)," - "后端用 overlay 叠加图片图层实现所见即所得;否则降级用 drawtext 重画文字。", - ) - title_image_dataurl: Optional[str] = Field( - None, - description="前端 Canvas 渲染的标题 PNG dataURL(data:image/png;base64,...);" - "传入时后端用 overlay 叠加图片图层,不再用 drawtext 重画文字。" - "推荐方式:直接放在 title_config.title_image_dataurl 里透传,本字段为兼容保留。", - ) - - class SmartCoverResponse(BaseModel): - """智能封面响应.""" + """智能封面响应(封面从最终成片抽帧,不再叠加标题).""" cover_url: str = Field("", description="封面图公网 URL(OSS,非临时);失败为空") status: str = Field("completed", description="completed / fallback_failed") diff --git a/apps/api/app/services/ai_avatar_cover_service.py b/apps/api/app/services/ai_avatar_cover_service.py index 50f8b994d..dca585d1e 100644 --- a/apps/api/app/services/ai_avatar_cover_service.py +++ b/apps/api/app/services/ai_avatar_cover_service.py @@ -1,20 +1,19 @@ -"""AI 数字人封面服务 — 复用智能剪辑的 MediaKit 抽帧 + 质量评分选最佳帧. +"""AI 数字人封面服务 — MediaKit 抽帧 + 质量评分选最佳帧 + 转存 OSS. 与 generation_cover.py 的智能选帧能力对齐(不再用 FFmpeg 简单截帧): 1. MediaKit extract_frames 抽取多帧(默认 5 帧,SpecifiedFrames 策略) 2. cover_frame_scorer.score_frames 按清晰度/亮度/色彩评分选最佳 3. 下载最佳帧并转存 OSS,返回公网封面 URL +设计原则:封面一律从最终成片(已叠加标题/B-roll)抽帧,帧本身已含标题, +本服务**不再叠加标题**。对口型阶段的裸视频封面入口已删除(废弃)。 + 降级:MediaKit 不可用或抽帧失败时返回空字符串,由调用方决定回退策略。 """ from __future__ import annotations -import base64 -import binascii import logging -import os -import subprocess import tempfile import uuid from pathlib import Path @@ -53,7 +52,6 @@ def _sign_video_url_for_mediakit(video_url: str) -> str: own_host = urlparse(public_base).netloc.lower() url_host = urlparse(video_url).netloc.lower() if own_host and url_host == own_host: - # 是自家 OSS URL,重签 7 天有效期供 MediaKit 拉取 signed = storage.get_download_url(video_url, expires_seconds=MEDIAKIT_URL_TTL_SECONDS) if signed: logger.info("[数字人封面] video_url 已重签(自家 OSS 私有桶)") @@ -64,19 +62,10 @@ def _sign_video_url_for_mediakit(video_url: str) -> str: def select_best_cover_frame(video_url: str, *, max_frames: int = 5) -> str: - """从视频抽取多帧并评分选最佳帧,返回最佳帧的临时 URL. - - Args: - video_url: 可公网访问的视频 URL - max_frames: 抽帧数量 - - Returns: - 最佳帧图片 URL;失败返回空字符串 - """ + """从视频抽取多帧并评分选最佳帧,返回最佳帧的临时 URL.""" if not video_url: return "" - # 确保 MediaKit 能访问 video_url(自家 OSS 私有桶需重签) video_url = _sign_video_url_for_mediakit(video_url) try: @@ -89,11 +78,8 @@ def select_best_cover_frame(video_url: str, *, max_frames: int = 5) -> str: return "" logger.info( - "[数字人封面] 开始抽帧: video_url=%s max_frames=%d poll_interval=%.1f max_poll=%d", - video_url[:80], - max_frames, - COVER_POLL_INTERVAL, - COVER_MAX_POLL_ATTEMPTS, + "[数字人封面] 开始抽帧: video_url=%s max_frames=%d", + video_url[:80], max_frames, ) snapshots = mk.extract_frames( @@ -111,7 +97,6 @@ def select_best_cover_frame(video_url: str, *, max_frames: int = 5) -> str: if len(snapshots) == 1: return snapshots[0].get("image_url") or snapshots[0].get("url") or "" - # 使用连接池下载各帧(复用 TCP 连接,减少延迟) import httpx candidates = [] @@ -139,7 +124,6 @@ def select_best_cover_frame(video_url: str, *, max_frames: int = 5) -> str: best = scored[0] if scored else None best_url = best.get("url", "") if best else "" - # 清理临时文件 for c in candidates: p = c.get("image_path") if p: @@ -150,8 +134,7 @@ def select_best_cover_frame(video_url: str, *, max_frames: int = 5) -> str: logger.info( "[数字人封面] 智能选帧完成: candidates=%d best_score=%s", - len(candidates), - best.get("score") if best else "n/a", + len(candidates), best.get("score") if best else "n/a", ) return best_url @@ -160,161 +143,19 @@ def select_best_cover_frame(video_url: str, *, max_frames: int = 5) -> str: return "" -def apply_title_to_cover(local_frame: str, *, title_config: dict | None) -> str: - """在封面图上叠加标题,返回叠加后图片的本地路径. - - 优先使用前端 Canvas 渲染的 PNG 图层(overlay=0:0,所见即所得); - 无 title_image_dataurl 时降级到 drawtext 重画文字。 - ffmpeg 失败时回退返回原始 local_frame。竖屏封面按 720x1280 计算位置(drawtext 降级路径)。 - """ - if not title_config or not isinstance(title_config, dict): - return local_frame - text = ( - title_config.get("text") - or title_config.get("content") - or title_config.get("title") - or "" - ).strip() - if not text: - return local_frame - enabled = title_config.get("enabled", True) - if not enabled: - return local_frame - - # 优先:前端 Canvas 渲染的 PNG 图层 - title_dataurl = title_config.get("title_image_dataurl") - if isinstance(title_dataurl, str) and title_dataurl.startswith("data:image/"): - out_path = _overlay_title_png_on_image(local_frame, title_dataurl) - if out_path: - return out_path - logger.warning("[数字人封面] PNG 叠加失败,回退 drawtext") - - # 降级:drawtext 重画文字 - try: - from packages.domain.video_filter_builder import build_title_drawtext_filter - - drawtext_filter = build_title_drawtext_filter( - title_config, - output_width=720, - output_height=1280, - ) - if not drawtext_filter: - return local_frame - - base, ext = os.path.splitext(local_frame) - titled_path = f"{base}_titled{ext or '.jpg'}" - cmd = [ - "ffmpeg", - "-i", - local_frame, - "-vf", - drawtext_filter, - "-y", - titled_path, - ] - logger.info("[数字人封面] drawtext 叠加标题: text=%s", text[:30]) - result = subprocess.run( - cmd, - capture_output=True, - text=True, - timeout=30, - ) - if result.returncode != 0: - logger.warning( - "[数字人封面] drawtext 失败,回退无标题: exit=%s stderr=%s", - result.returncode, - (result.stderr or "")[-300:], - ) - return local_frame - if not os.path.exists(titled_path) or os.path.getsize(titled_path) == 0: - logger.warning("[数字人封面] drawtext 输出为空,回退无标题") - return local_frame - return titled_path - except Exception as exc: - logger.warning("[数字人封面] 标题叠加异常,回退无标题: %s", exc, exc_info=True) - return local_frame - - -def _overlay_title_png_on_image(local_frame: str, dataurl: str) -> str | None: - """解码 title PNG dataURL 并用 ffmpeg overlay 叠加到封面图上。 - - 成功返回新文件路径;失败返回 None。 - """ - try: - header, b64 = dataurl.split(",", 1) - if "base64" not in header: - return None - png_bytes = base64.b64decode(b64, validate=True) - if not png_bytes: - return None - - base_dir = os.path.dirname(local_frame) - title_png = os.path.join(base_dir, f"title_{uuid.uuid4().hex[:8]}.png") - with open(title_png, "wb") as f: - f.write(png_bytes) - - base, ext = os.path.splitext(local_frame) - titled_path = f"{base}_titled{ext or '.jpg'}" - cmd = [ - "ffmpeg", - "-i", - local_frame, - "-i", - title_png, - "-filter_complex", - "[0:v][1:v]overlay=0:0", - "-y", - titled_path, - ] - logger.info("[数字人封面] PNG overlay 叠加标题") - result = subprocess.run( - cmd, - capture_output=True, - text=True, - timeout=30, - ) - # 清理临时 PNG - try: - os.unlink(title_png) - except OSError: - pass - if result.returncode != 0: - logger.warning( - "[数字人封面] PNG overlay 失败: exit=%s stderr=%s", - result.returncode, - (result.stderr or "")[-300:], - ) - return None - if not os.path.exists(titled_path) or os.path.getsize(titled_path) == 0: - return None - return titled_path - except (binascii.Error, ValueError, OSError) as exc: - logger.warning("[数字人封面] PNG 解码/overlay 异常: %s", exc, exc_info=True) - return None - - def persist_cover_to_oss( frame_url: str, *, job_id: str = "", prefix: str = "ai-avatar/covers", - title_config: dict | None = None, ) -> str: - """下载帧图并转存到 OSS,返回公网封面 URL. + """下载最佳帧图并转存到 OSS,返回公网封面 URL(预签名). - Args: - frame_url: MediaKit 返回的临时帧图 URL - job_id: 关联任务 ID(用于 OSS key 命名) - prefix: OSS key 前缀 - title_config: 可选标题配置;传入时用 drawtext 叠加标题(竖屏 720x1280) - - Returns: - OSS 公网 URL;失败回退原始 frame_url + 封面来自最终成片抽帧,帧本身已含标题,本函数不再做任何文字/图片叠加。 """ if not frame_url: return "" tmp_path: Optional[str] = None - titled_path: Optional[str] = None try: import httpx @@ -335,21 +176,12 @@ def persist_cover_to_oss( token = job_id or uuid.uuid4().hex[:12] cover_key = f"{prefix}/{token}/cover_{uuid.uuid4().hex[:8]}.jpg" - upload_path = apply_title_to_cover(tmp_path, title_config=title_config) - if upload_path != tmp_path: - titled_path = upload_path - public_url = storage.upload_file( - file_or_path=upload_path, + file_or_path=tmp_path, storage_key=cover_key, content_type="image/jpeg", ) - logger.info( - "[数字人封面] 封面已转存 OSS: key=%s titled=%s", - cover_key, - bool(titled_path), - ) - # 私有桶:返回预签名 URL(前端才能加载) + logger.info("[数字人封面] 封面已转存 OSS: key=%s", cover_key) if public_url: signed = storage.get_download_url(cover_key, expires_seconds=86400) return signed @@ -358,12 +190,11 @@ def persist_cover_to_oss( logger.warning("[数字人封面] 封面转存 OSS 失败,返回原始 URL", exc_info=True) return frame_url finally: - for p in (tmp_path, titled_path): - if p: - try: - Path(p).unlink(missing_ok=True) - except Exception: - pass + if tmp_path: + try: + Path(tmp_path).unlink(missing_ok=True) + except Exception: + pass def generate_smart_cover( @@ -371,19 +202,12 @@ def generate_smart_cover( *, job_id: str = "", max_frames: int = 5, - title_config: dict | None = None, ) -> str: - """一站式:MediaKit 智能抽帧选最佳 → (可选)drawtext 叠加标题 → 转存 OSS. + """一站式:MediaKit 智能抽帧选最佳 → 转存 OSS。失败返回空字符串。 - 供独立封面接口与渲染管线复用。失败返回空字符串。 - - Args: - video_url: 可公网访问的视频 URL - job_id: 关联任务 ID - max_frames: 抽帧数量 - title_config: 可选标题配置;传入时在封面上叠加 drawtext 标题(竖屏 720x1280) + 封面从最终成片抽帧,不再叠加任何标题(帧本身已含)。 """ best_frame = select_best_cover_frame(video_url, max_frames=max_frames) if not best_frame: return "" - return persist_cover_to_oss(best_frame, job_id=job_id, title_config=title_config) + return persist_cover_to_oss(best_frame, job_id=job_id) diff --git a/apps/web/src/pages/ai-avatar/api/aiAvatar.ts b/apps/web/src/pages/ai-avatar/api/aiAvatar.ts index 813415916..7f5b6a764 100644 --- a/apps/web/src/pages/ai-avatar/api/aiAvatar.ts +++ b/apps/web/src/pages/ai-avatar/api/aiAvatar.ts @@ -58,21 +58,6 @@ export const getLipsyncJob = async (id: string): Promise => { return response.data } -/* ── 智能封面(MediaKit 抽帧 + 质量评分选最佳帧 + 可选标题 overlay/drawtext 叠加) ── */ -export const generateSmartCover = async ( - video_url: string, - title_config?: Record | null, - max_frames = 5, -): Promise<{ cover_url: string; status: string; message: string }> => { - const response = await apiClient.post<{ cover_url: string; status: string; message: string }>( - "/ai-avatar/render/smart-cover", - { video_url, max_frames, title_config: title_config ?? null }, - // smart-cover 链路:下载视频+抽帧+overlay/drawtext 加标题+上传 OSS,需要较长时间,120s 超时 - { timeout: 120000 }, - ) - return response.data -} - /* ── 渲染 ── */ export const submitRender = async (data: { lipsync_job_id: string