diff --git a/apps/worker/video_processing/thumbnail_generator.py b/apps/worker/video_processing/thumbnail_generator.py index a33be8d51..375aa9f9c 100755 --- a/apps/worker/video_processing/thumbnail_generator.py +++ b/apps/worker/video_processing/thumbnail_generator.py @@ -1,7 +1,10 @@ """视频封面抽帧工具 — 从视频中抽取帧作为封面,支持标题文字叠加。 -统一封面管道(P1 优化后默认本地路径): -- 默认路径:本地 ffmpeg -ss 单帧 seek 抽取 + cv2 质量评分(清晰度/亮度/色彩),1-2s 完成 +封面管道(P2 优化后): +- 黑屏检测:ffmpeg blackdetect 扫描黑屏区间,抽帧点自动避开黑屏 +- 单次 ffmpeg select 抽多帧:一次 ffmpeg 进程用 select 滤镜输出 5 帧,避免 5 次起停进程 +- 并发上传:5 帧用 ThreadPoolExecutor 并行上传 OSS,目标封面阶段 <1.5s +- 质量评分:cv2 清晰度/亮度/色彩三维评分选最佳帧 - 可选 MediaKit 路径:配置 MEDIAKIT_COVER_ENABLED=true 时启用火山 MediaKit SceneChange 抽帧 - 从已渲染视频抽帧:标题已通过 ASS 字幕烧进视频,帧天然带标题,无需再叠加。 - 从源素材抽帧(API E2 兜底):源素材无标题,通过 Pillow 在帧上绘制标题文字。 @@ -10,7 +13,9 @@ from __future__ import annotations import logging +import re import tempfile +from concurrent.futures import ThreadPoolExecutor, as_completed from pathlib import Path from typing import Optional @@ -173,6 +178,225 @@ def generate_and_upload_thumbnail( Path(tmp.name).unlink(missing_ok=True) +def _detect_black_intervals( + video_path: str, + duration: float, + *, + black_min_duration: float = 0.3, + picture_black_ratio_th: float = 0.98, + pixel_black_th: float = 0.10, + timeout: int = 30, +) -> list[tuple[float, float]]: + """用 ffmpeg blackdetect 扫描黑屏区间,返回 [(start, end), ...]。""" + from video_processing.ffmpeg_utils import FFMPEG_BIN, run_ffmpeg + + if duration <= 0: + return [] + cmd = [ + FFMPEG_BIN, + "-nostdin", + "-i", + video_path, + "-vf", + (f"blackdetect=d={black_min_duration:.2f}:pic_th={picture_black_ratio_th:.2f}:pix_th={pixel_black_th:.2f}"), + "-an", + "-f", + "null", + "-", + ] + try: + _, stderr = run_ffmpeg(cmd, capture_output=True, timeout=timeout) + except Exception as e: + logger.warning("[thumbnail] blackdetect 失败,忽略黑屏规避: %s", e) + return [] + + intervals: list[tuple[float, float]] = [] + pattern = re.compile( + r"black_start:(\d+(?:\.\d+)?)\s+black_end:(\d+(?:\.\d+)?)\s+black_duration:(\d+(?:\.\d+)?)", + ) + for m in pattern.finditer(stderr or ""): + try: + bs = float(m.group(1)) + be = float(m.group(2)) + intervals.append((bs, be)) + except ValueError: + continue + intervals.sort() + if intervals: + logger.info("[thumbnail] blackdetect 发现 %d 段黑屏: %s", len(intervals), intervals[:5]) + return intervals + + +def _adjust_seek_points_avoid_black( + seek_points: list[float], + black_intervals: list[tuple[float, float]], + duration: float, + *, + tolerance: float = 0.25, +) -> list[float]: + """把落在黑屏区间的 seek 点偏移到最近的非黑屏位置。 + + 策略: + - 若点在黑屏内,先尝试向前偏移到黑屏起点 - tolerance,再尝试向后偏移到黑屏终点 + tolerance; + - 若整个视频全黑(偏移后 <0 或 >duration),保留原点但日志标记警告; + - 偏移后若点与已有点重合(误差 <0.3s),做微调去重。 + """ + if not black_intervals or not seek_points: + return list(seek_points) + + def in_black(t: float) -> tuple[float, float] | None: + for bs, be in black_intervals: + if bs <= t <= be: + return (bs, be) + return None + + adjusted: list[float] = [] + for t in seek_points: + seg = in_black(t) + if seg is None: + adjusted.append(max(0.0, min(duration, t))) + continue + bs, be = seg + # 先尝试向前 + forward_t = bs - tolerance + if forward_t >= 0.0 and in_black(forward_t) is None: + adjusted.append(forward_t) + continue + # 再尝试向后 + backward_t = be + tolerance + if backward_t <= duration and in_black(backward_t) is None: + adjusted.append(backward_t) + continue + # 整段 clip 全黑?保留中点但标记 + logger.warning( + "[thumbnail] seek 点 %.2fs 落在黑屏区间 [%.2f,%.2f] 且无法偏移,保留原位置(可能是全黑片段)", + t, + bs, + be, + ) + adjusted.append(max(0.0, min(duration, t))) + + # 去重:相邻点若 <0.3s 则拉开 + adjusted.sort() + deduped: list[float] = [] + for t in adjusted: + if not deduped or abs(t - deduped[-1]) >= 0.3: + deduped.append(t) + else: + # 往后挪 0.5s + nt = t + 0.5 + if nt <= duration and in_black(nt) is None: + deduped.append(nt) + else: + deduped.append(t) + return [round(max(0.0, min(duration, t)), 3) for t in deduped[: len(seek_points)]] + + +def _extract_frames_single_pass( + video_path: str, + seek_points: list[float], + out_dir: str, + *, + prefix: str = "frame", + width: int = -1, + height: int = -1, + q: int = 2, + timeout: int = 30, +) -> list[tuple[float, str]]: + """单次 ffmpeg 用 select 滤镜抽出 seek_points 对应的多帧。 + + ffmpeg -i input -vf "select='between(t,t1-0.03,t1+0.03)+between(t,t2-0.03,t2+0.03)+...',scale=...,format=yuvj420p" + -vsync vfr -q:v 2 out_dir/prefix_%02d.jpg + + 返回 [(seek_t, output_path), ...],按输出帧序号升序。若输出帧数 < seek_points 数量, + 不足部分用 extract_first_frame 兜底(保证返回数量 == len(seek_points))。 + """ + from video_processing.ffmpeg_utils import FFMPEG_BIN, run_ffmpeg + + out_dir_p = Path(out_dir) + out_dir_p.mkdir(parents=True, exist_ok=True) + + # 构造 select 表达式:每个 seek 点用 ±30ms 窗口命中 + # between(t, a, b) 返回 1 表示 t 在 [a,b] 内;多个 between 相加即为"任一命中" + select_terms = [] + for t in seek_points: + a = max(0.0, t - 0.03) + b = t + 0.04 + select_terms.append(f"between(t,{a:.3f},{b:.3f})") + select_expr = "+".join(select_terms) + + if width > 0 or height > 0: + w_str = str(width) if width > 0 else "-1" + h_str = str(height) if height > 0 else "-1" + scale_filter = f"scale={w_str}:{h_str}:force_original_aspect_ratio=decrease" + vf = f"select='{select_expr}',{scale_filter},format=yuvj420p" + else: + vf = f"select='{select_expr}',format=yuvj420p" + + out_pattern = str(out_dir_p / f"{prefix}_%02d.jpg") + cmd = [ + FFMPEG_BIN, + "-y", + "-i", + video_path, + "-vf", + vf, + "-vsync", + "vfr", + "-q:v", + str(q), + out_pattern, + ] + + results: list[tuple[float, str]] = [] + single_pass_ok = False + try: + run_ffmpeg(cmd, capture_output=True, timeout=timeout) + # 读取输出文件 + for i in range(1, len(seek_points) + 1): + fp = out_dir_p / f"{prefix}_{i:02d}.jpg" + if fp.exists() and fp.stat().st_size > 0: + results.append((seek_points[i - 1] if i - 1 < len(seek_points) else 0.0, str(fp))) + if len(results) >= len(seek_points): + single_pass_ok = True + else: + logger.warning( + "[thumbnail] 单次 ffmpeg 抽帧仅命中 %d/%d 帧,不足部分用单帧 seek 兜底", + len(results), + len(seek_points), + ) + except Exception as e: + logger.warning("[thumbnail] 单次 ffmpeg select 抽帧失败,回退到单帧 seek: %s", e) + + # 兜底:对缺失/失败的帧用 extract_first_frame 补抽 + if not single_pass_ok: + # 清理不完整结果 + for _, fp in results: + try: + Path(fp).unlink(missing_ok=True) + except Exception: + pass + results = [] + for i, st in enumerate(seek_points): + fp = out_dir_p / f"{prefix}_fallback_{i:02d}.jpg" + try: + extract_first_frame( + video_path, + output_path=str(fp), + seek_seconds=st, + min_seek_seconds=0.5, + timeout=timeout, + ) + if fp.exists() and fp.stat().st_size > 0: + results.append((st, str(fp))) + else: + logger.warning("[thumbnail] 兜底单帧抽帧也失败 idx=%d t=%.2f", i, st) + except Exception as e: + logger.warning("[thumbnail] 兜底单帧抽帧异常 idx=%d t=%.2f: %s", i, st, e) + + return results[: len(seek_points)] + + def _compute_clip_boundary_seek_points( duration: float, clip_boundaries: Optional[list[tuple[float, float]]] = None, @@ -310,10 +534,11 @@ def extract_and_upload_cover_frames( ) -> list[dict]: """从视频中抽取多帧作为封面候选,通过质量评分选出最佳帧,上传到 OSS。 - 默认路径(P1优化):本地 ffmpeg 单帧 seek 抽帧 + cv2 评分,预期 <2s 完成。 - - 基于 clip 分段边界取各段中间帧(clip_boundaries 参数),效果优于均匀抽帧 - - 无边界信息时均匀分布(10%~90% 之间) - - 所有帧本地 cv2 清晰度/亮度/色彩三维评分,最高分自动选出 + P2 优化: + - 先用 ffmpeg blackdetect 扫描黑屏区间,seek 点自动避开黑屏 + - 单次 ffmpeg select 抽 num_frames 帧(避免 5 次起停 ffmpeg 进程) + - 多帧 OSS 上传用 ThreadPoolExecutor 并发,目标封面阶段 <1.5s + - cv2 清晰度/亮度/色彩三维评分选最佳帧 Fallback(MEDIAKIT_COVER_ENABLED=true):火山 MediaKit SceneChange 抽帧(~60-90s)。 @@ -381,46 +606,81 @@ def extract_and_upload_cover_frames( if len(candidates) >= num_frames: logger.info("[thumbnail] MediaKit 抽帧完成: %d 帧", len(candidates)) - # ── 默认路径:本地 ffmpeg 单帧 seek ─────────────────────────── + # ── 默认路径:本地 ffmpeg 单次 select 抽帧 + 并发上传 ────────────── if len(candidates) < num_frames: if candidates: logger.info("[thumbnail] MediaKit 不足 %d 帧,本地 ffmpeg 补充", num_frames) else: - logger.info("[thumbnail] 使用本地 ffmpeg 抽帧(num=%d, duration=%.1fs)", num_frames, duration) + logger.info( + "[thumbnail] 使用本地 ffmpeg 抽帧(num=%d, duration=%.1fs)", + num_frames, + duration, + ) + # 1) 计算 seek 点 seek_points = _compute_clip_boundary_seek_points(duration, clip_boundaries, num_frames) - for i, seek_t in enumerate(seek_points): - tmp = tempfile.NamedTemporaryFile(suffix=".jpg", delete=False) - tmp.close() - _temp_paths.append(tmp.name) - try: - frame_path = extract_first_frame( - video_path, - output_path=tmp.name, - seek_seconds=seek_t, - min_seek_seconds=0.5, - ) + # 2) 黑屏检测 + 偏移 seek 点 + black_intervals = _detect_black_intervals(video_path, duration) if duration > 0 else [] + if black_intervals: + seek_points = _adjust_seek_points_avoid_black(seek_points, black_intervals, duration) + logger.info("[thumbnail] 黑屏规避后 seek 点: %s", seek_points) + + # 3) 单次 ffmpeg select 抽出所有帧(带失败兜底到单帧 seek) + with tempfile.TemporaryDirectory(prefix="thumb_") as frame_dir: + t1 = time.monotonic() + frame_results = _extract_frames_single_pass( + video_path, + seek_points, + frame_dir, + prefix="frame", + ) + logger.info("[thumbnail] 抽帧耗时: %.2fs (%d 帧)", time.monotonic() - t1, len(frame_results)) + + # 4) 标题叠加(本地,CPU 很快) + for _st, fp in frame_results: if title_text and title_text.strip(): - apply_title_overlay( - frame_path, - title_text, - color=title_color, - position=title_position, - font_size=title_font_size, - ) - storage_key = f"covers/{plan_id}/{task_id}/frame_{i}.jpg" - url = upload_to_oss(frame_path, storage_key) - if url: - candidates.append( - { - "url": url, - "position": seek_t, - "image_path": tmp.name, - } - ) - except Exception as e: - logger.warning("[thumbnail] 封面候选帧 %d 提取失败: %s", i, e) + try: + apply_title_overlay( + fp, + title_text, + color=title_color, + position=title_position, + font_size=title_font_size, + ) + except Exception as e: + logger.warning("[thumbnail] 标题叠加失败 %s: %s", fp, e) + + # 5) 并发上传 OSS(线程池并发) + t2 = time.monotonic() + + def _upload_one(idx: int, st: float, fp: str) -> dict | None: + try: + storage_key = f"covers/{plan_id}/{task_id}/frame_{idx}.jpg" + url = upload_to_oss(fp, storage_key) + if url: + return {"url": url, "position": st, "image_path": fp} + logger.warning("[thumbnail] 上传失败 idx=%d", idx) + except Exception as e: + logger.warning("[thumbnail] 上传异常 idx=%d t=%.2f: %s", idx, st, e) + return None + + upload_results: list[dict | None] = [None] * len(frame_results) + max_workers = min(8, max(2, len(frame_results))) + with ThreadPoolExecutor(max_workers=max_workers) as pool: + future_map = {pool.submit(_upload_one, i, st, fp): i for i, (st, fp) in enumerate(frame_results)} + for fut in as_completed(future_map): + i = future_map[fut] + try: + upload_results[i] = fut.result() + except Exception as e: + logger.warning("[thumbnail] 上传 feature 异常 idx=%d: %s", i, e) + logger.info("[thumbnail] 并发上传耗时: %.2fs", time.monotonic() - t2) + + for r in upload_results: + if r is not None: + _temp_paths.append(r["image_path"]) + candidates.append(r) # ── 阶段 2:质量评分 ──────────────────────────────────────────── if len(candidates) > 1: diff --git a/infra/docker/worker.Dockerfile b/infra/docker/worker.Dockerfile index 83475b7cd..a6ec11baf 100755 --- a/infra/docker/worker.Dockerfile +++ b/infra/docker/worker.Dockerfile @@ -10,6 +10,20 @@ FROM xiaoxia-registry.cn-hangzhou.cr.aliyuncs.com/xiaoxiakeji/saas-worker-base:l # 构建参数:版本号(CI 传入 commit hash) ARG APP_VERSION=dev +# 使用阿里云镜像加速 +RUN sed -i 's|deb.debian.org|mirrors.aliyun.com|g' /etc/apt/sources.list.d/debian.sources 2>/dev/null || \ + sed -i 's|deb.debian.org|mirrors.aliyun.com|g' /etc/apt/sources.list 2>/dev/null || true + +# CJK 字体保障:确保 fonts-noto-cjk 已安装(base 镜像漂移兜底)+ 重建字体缓存 +# 验证 fc-match 能正确解析 Noto Sans CJK SC,避免 PIL 直读 ttf 兜底 +RUN apt-get update && apt-get install -y --no-install-recommends \ + fonts-noto-cjk \ + fontconfig \ + && rm -rf /var/lib/apt/lists/* \ + && fc-cache -fv \ + && fc-match 'Noto Sans CJK SC' | grep -qi 'noto' \ + && echo "[font] fc-match Noto Sans CJK SC: $(fc-match 'Noto Sans CJK SC' | head -1)" + # 创建非 root 用户 RUN groupadd -r celery \ && useradd -r -g celery -d /app -s /sbin/nologin celery \