"""视频封面抽帧工具 — 从视频中抽取帧作为封面,支持标题文字叠加。 封面管道(P2 优化后): - 黑屏检测:ffmpeg blackdetect 扫描黑屏区间,抽帧点自动避开黑屏 - 单次 ffmpeg select 抽多帧:一次 ffmpeg 进程用 select 滤镜输出 5 帧,避免 5 次起停进程 - 并发上传:5 帧用 ThreadPoolExecutor 并行上传 OSS,目标封面阶段 <1.5s - 质量评分:cv2 清晰度/亮度/色彩三维评分选最佳帧 - 可选 MediaKit 路径:配置 MEDIAKIT_COVER_ENABLED=true 时启用火山 MediaKit SceneChange 抽帧 - 从已渲染视频抽帧:标题已通过 ASS 字幕烧进视频,帧天然带标题,无需再叠加。 - 从源素材抽帧(API E2 兜底):源素材无标题,通过 Pillow 在帧上绘制标题文字。 """ from __future__ import annotations import logging import re import tempfile from concurrent.futures import ThreadPoolExecutor, as_completed from pathlib import Path from typing import Optional logger = logging.getLogger(__name__) def apply_title_overlay( image_path: str, title_text: str, *, color: str = "#ffffff", position: str = "bottom", font_size: int | None = None, margin_ratio: float = 0.06, stroke_width_ratio: float = 0.04, ) -> str: """在图片上绘制标题文字(指定颜色 + 黑色描边/阴影)。""" from packages.shared.title_overlay import apply_title_to_image if not title_text or not title_text.strip(): return image_path result = apply_title_to_image( image_path, title_text, color=color, position=position, font_size=font_size, margin_ratio=margin_ratio, stroke_width_ratio=stroke_width_ratio, ) return result or image_path def extract_first_frame( video_path: str, output_path: str | None = None, *, width: int = -1, height: int = -1, timeout: int = 30, seek_ratio: float = 0.15, seek_seconds: float | None = None, min_seek_seconds: float = 1.0, ) -> str: """抽取视频封面帧(ffmpeg -ss 单帧 seek,<100ms/帧)。 Args: video_path: 视频文件路径 output_path: 输出图片路径,不传则用临时文件 width/height: 输出宽高(默认保持原始分辨率) timeout: 超时(秒) seek_ratio: 抽帧位置占视频时长的比例 seek_seconds: 指定具体抽帧时间点(秒),优先于 seek_ratio min_seek_seconds: 最小抽帧时间 """ from video_processing.ffmpeg_utils import FFMPEG_BIN, probe_duration, run_ffmpeg _is_temp_output = False if output_path is None: tmp = tempfile.NamedTemporaryFile(suffix=".jpg", delete=False) tmp.close() output_path = tmp.name _is_temp_output = True try: if seek_seconds is not None: seek_time = max(0.0, float(seek_seconds)) else: try: duration = probe_duration(video_path) seek_time = max(min_seek_seconds, duration * seek_ratio) except Exception: seek_time = min_seek_seconds seek_str = _format_seek_time(seek_time) if width > 0 or height > 0: w_str = str(width) if width > 0 else "-1" h_str = str(height) if height > 0 else "-1" scale_filter = f"scale={w_str}:{h_str}:force_original_aspect_ratio=decrease,format=yuvj420p" else: scale_filter = "format=yuvj420p" # -ss 放在 -i 前面(input seeking,极快),-vframes 1 只取一帧 cmd = [ FFMPEG_BIN, "-y", "-ss", seek_str, "-i", video_path, "-vframes", "1", "-vf", scale_filter, "-q:v", "2", output_path, ] try: run_ffmpeg(cmd, capture_output=True, timeout=timeout) except Exception: # 失败时退回到第0帧兜底 cmd2 = [ FFMPEG_BIN, "-y", "-i", video_path, "-ss", "00:00:00", "-vframes", "1", "-vf", scale_filter, "-q:v", "2", output_path, ] run_ffmpeg(cmd2, capture_output=True, timeout=timeout) if not Path(output_path).exists() or Path(output_path).stat().st_size == 0: raise RuntimeError(f"Cover frame extraction failed: {output_path}") return output_path except Exception: if _is_temp_output and output_path: try: Path(output_path).unlink(missing_ok=True) except Exception: pass raise def _format_seek_time(seconds: float) -> str: h = int(seconds // 3600) m = int((seconds % 3600) // 60) s = seconds % 60 return f"{h:02d}:{m:02d}:{s:05.2f}" def generate_and_upload_thumbnail( video_path: str, storage_key: str, *, seek_ratio: float = 0.15, ) -> str: """从视频中提取一帧缩略图并上传到 OSS。""" from video_processing.oss_helpers import upload_to_oss tmp = tempfile.NamedTemporaryFile(suffix=".jpg", delete=False) tmp.close() try: frame_path = extract_first_frame(video_path, output_path=tmp.name, seek_ratio=seek_ratio) url = upload_to_oss(frame_path, storage_key) if not url: raise RuntimeError(f"上传缩略图到 OSS 失败: {storage_key}") return url finally: Path(tmp.name).unlink(missing_ok=True) def _detect_black_intervals( video_path: str, duration: float, *, black_min_duration: float = 0.3, picture_black_ratio_th: float = 0.98, pixel_black_th: float = 0.10, timeout: int = 30, ) -> list[tuple[float, float]]: """用 ffmpeg blackdetect 扫描黑屏区间,返回 [(start, end), ...]。""" from video_processing.ffmpeg_utils import FFMPEG_BIN, run_ffmpeg if duration <= 0: return [] cmd = [ FFMPEG_BIN, "-nostdin", "-i", video_path, "-vf", (f"blackdetect=d={black_min_duration:.2f}:pic_th={picture_black_ratio_th:.2f}:pix_th={pixel_black_th:.2f}"), "-an", "-f", "null", "-", ] try: _, stderr = run_ffmpeg(cmd, capture_output=True, timeout=timeout) except Exception as e: logger.warning("[thumbnail] blackdetect 失败,忽略黑屏规避: %s", e) return [] intervals: list[tuple[float, float]] = [] pattern = re.compile( r"black_start:(\d+(?:\.\d+)?)\s+black_end:(\d+(?:\.\d+)?)\s+black_duration:(\d+(?:\.\d+)?)", ) for m in pattern.finditer(stderr or ""): try: bs = float(m.group(1)) be = float(m.group(2)) intervals.append((bs, be)) except ValueError: continue intervals.sort() if intervals: logger.info("[thumbnail] blackdetect 发现 %d 段黑屏: %s", len(intervals), intervals[:5]) return intervals def _adjust_seek_points_avoid_black( seek_points: list[float], black_intervals: list[tuple[float, float]], duration: float, *, tolerance: float = 0.25, ) -> list[float]: """把落在黑屏区间的 seek 点偏移到最近的非黑屏位置。 策略: - 若点在黑屏内,先尝试向前偏移到黑屏起点 - tolerance,再尝试向后偏移到黑屏终点 + tolerance; - 若整个视频全黑(偏移后 <0 或 >duration),保留原点但日志标记警告; - 偏移后若点与已有点重合(误差 <0.3s),做微调去重。 """ if not black_intervals or not seek_points: return list(seek_points) def in_black(t: float) -> tuple[float, float] | None: for bs, be in black_intervals: if bs <= t <= be: return (bs, be) return None adjusted: list[float] = [] for t in seek_points: seg = in_black(t) if seg is None: adjusted.append(max(0.0, min(duration, t))) continue bs, be = seg # 先尝试向前 forward_t = bs - tolerance if forward_t >= 0.0 and in_black(forward_t) is None: adjusted.append(forward_t) continue # 再尝试向后 backward_t = be + tolerance if backward_t <= duration and in_black(backward_t) is None: adjusted.append(backward_t) continue # 整段 clip 全黑?保留中点但标记 logger.warning( "[thumbnail] seek 点 %.2fs 落在黑屏区间 [%.2f,%.2f] 且无法偏移,保留原位置(可能是全黑片段)", t, bs, be, ) adjusted.append(max(0.0, min(duration, t))) # 去重:相邻点若 <0.3s 则拉开 adjusted.sort() deduped: list[float] = [] for t in adjusted: if not deduped or abs(t - deduped[-1]) >= 0.3: deduped.append(t) else: # 往后挪 0.5s nt = t + 0.5 if nt <= duration and in_black(nt) is None: deduped.append(nt) else: deduped.append(t) return [round(max(0.0, min(duration, t)), 3) for t in deduped[: len(seek_points)]] def _extract_frames_single_pass( video_path: str, seek_points: list[float], out_dir: str, *, prefix: str = "frame", width: int = -1, height: int = -1, q: int = 2, timeout: int = 30, ) -> list[tuple[float, str]]: """单次 ffmpeg 用 select 滤镜抽出 seek_points 对应的多帧。 ffmpeg -i input -vf "select='between(t,t1-0.03,t1+0.03)+between(t,t2-0.03,t2+0.03)+...',scale=...,format=yuvj420p" -vsync vfr -q:v 2 out_dir/prefix_%02d.jpg 返回 [(seek_t, output_path), ...],按输出帧序号升序。若输出帧数 < seek_points 数量, 不足部分用 extract_first_frame 兜底(保证返回数量 == len(seek_points))。 """ from video_processing.ffmpeg_utils import FFMPEG_BIN, run_ffmpeg out_dir_p = Path(out_dir) out_dir_p.mkdir(parents=True, exist_ok=True) # 构造 select 表达式:每个 seek 点用 ±30ms 窗口命中 # between(t, a, b) 返回 1 表示 t 在 [a,b] 内;多个 between 相加即为"任一命中" select_terms = [] for t in seek_points: a = max(0.0, t - 0.03) b = t + 0.04 select_terms.append(f"between(t,{a:.3f},{b:.3f})") select_expr = "+".join(select_terms) if width > 0 or height > 0: w_str = str(width) if width > 0 else "-1" h_str = str(height) if height > 0 else "-1" scale_filter = f"scale={w_str}:{h_str}:force_original_aspect_ratio=decrease" vf = f"select='{select_expr}',{scale_filter},format=yuvj420p" else: vf = f"select='{select_expr}',format=yuvj420p" out_pattern = str(out_dir_p / f"{prefix}_%02d.jpg") cmd = [ FFMPEG_BIN, "-y", "-i", video_path, "-vf", vf, "-vsync", "vfr", "-q:v", str(q), out_pattern, ] results: list[tuple[float, str]] = [] single_pass_ok = False try: run_ffmpeg(cmd, capture_output=True, timeout=timeout) # 读取输出文件 for i in range(1, len(seek_points) + 1): fp = out_dir_p / f"{prefix}_{i:02d}.jpg" if fp.exists() and fp.stat().st_size > 0: results.append((seek_points[i - 1] if i - 1 < len(seek_points) else 0.0, str(fp))) if len(results) >= len(seek_points): single_pass_ok = True else: logger.warning( "[thumbnail] 单次 ffmpeg 抽帧仅命中 %d/%d 帧,不足部分用单帧 seek 兜底", len(results), len(seek_points), ) except Exception as e: logger.warning("[thumbnail] 单次 ffmpeg select 抽帧失败,回退到单帧 seek: %s", e) # 兜底:对缺失/失败的帧用 extract_first_frame 补抽 if not single_pass_ok: # 清理不完整结果 for _, fp in results: try: Path(fp).unlink(missing_ok=True) except Exception: pass results = [] for i, st in enumerate(seek_points): fp = out_dir_p / f"{prefix}_fallback_{i:02d}.jpg" try: extract_first_frame( video_path, output_path=str(fp), seek_seconds=st, min_seek_seconds=0.5, timeout=timeout, ) if fp.exists() and fp.stat().st_size > 0: results.append((st, str(fp))) else: logger.warning("[thumbnail] 兜底单帧抽帧也失败 idx=%d t=%.2f", i, st) except Exception as e: logger.warning("[thumbnail] 兜底单帧抽帧异常 idx=%d t=%.2f: %s", i, st, e) return results[: len(seek_points)] def _compute_clip_boundary_seek_points( duration: float, clip_boundaries: Optional[list[tuple[float, float]]] = None, num_frames: int = 5, head_skip_ratio: float = 0.08, tail_skip_ratio: float = 0.08, ) -> list[float]: """基于clip分段边界计算抽帧时间点(取每段中间帧,效果比均匀抽更好)。 策略: - 如果传入 clip_boundaries(每个元素是 (clip_start_in_timeline, clip_duration)), 取每个片段的中点作为抽帧候选点 - 候选点不足 num_frames 时,均匀补充 - 跳过片头 head_skip_ratio(8%,避免片头黑屏/开场标题)和片尾 tail_skip_ratio(8%) - 返回按时间排序的 num_frames 个抽帧点(秒) """ if duration <= 0: # 无法probe,均匀分布兜底 return [max(1.0, duration * (0.1 + 0.8 * i / max(num_frames - 1, 1))) for i in range(num_frames)] head_skip = duration * head_skip_ratio tail_skip = duration * tail_skip_ratio valid_start = head_skip valid_end = max(valid_start + 1.0, duration - tail_skip) candidates: list[float] = [] if clip_boundaries: # 累加timeline start,取每clip中点 cur = 0.0 for _clip_start, clip_dur in clip_boundaries: if clip_dur <= 0: continue mid = cur + clip_dur / 2.0 if valid_start <= mid <= valid_end: candidates.append(mid) cur += clip_dur # 去重+排序 candidates = sorted(set(round(c, 3) for c in candidates)) # 如果候选点不足,均匀补充 if len(candidates) < num_frames: needed = num_frames - len(candidates) existing = set(round(c, 1) for c in candidates) for i in range(needed * 3): ratio = 0.1 + 0.8 * (i + 0.5) / (needed * 3) t = valid_start + (valid_end - valid_start) * ratio if round(t, 1) not in existing: candidates.append(t) existing.add(round(t, 1)) if len(candidates) >= num_frames: break # 如果还不够,强制均匀 while len(candidates) < num_frames: idx = len(candidates) ratio = 0.1 + 0.8 * idx / max(num_frames - 1, 1) candidates.append(valid_start + (valid_end - valid_start) * ratio) candidates.sort() # 如果超过num_frames,均匀选取 if len(candidates) > num_frames: step = len(candidates) / num_frames candidates = [candidates[int(i * step)] for i in range(num_frames)] return [round(t, 3) for t in candidates[:num_frames]] def _extract_frames_via_mediakit( video_path: str, plan_id: str, num_frames: int, ) -> list[dict] | None: """使用 MediaKit 智能抽帧 API 提取封面帧(fallback 路径,默认不启用)。""" import uuid from video_processing.oss_helpers import delete_from_oss, get_signed_download_url, upload_to_oss from packages.shared.mediakit_client import get_mediakit_client client = get_mediakit_client() if not client.is_available: logger.info("[thumbnail] MediaKit 未配置,跳过智能抽帧") return None video_storage_key: str = "" try: video_storage_key = f"temp/{plan_id}/{uuid.uuid4().hex[:8]}_{Path(video_path).name}" public_url = upload_to_oss(video_path, video_storage_key) if not public_url: logger.warning("[thumbnail] 视频上传 OSS 失败,无法使用 MediaKit") return None video_url = get_signed_download_url(video_storage_key, expires_seconds=3600) or public_url logger.info("[thumbnail] 视频已上传 OSS 并生成签名 URL: key=%s", video_storage_key[:80]) except Exception as e: logger.warning("[thumbnail] 视频上传 OSS 异常: %s,降级到本地 ffmpeg", e) return None try: frames = client.extract_frames( video_url=video_url, strategy="SceneChange", max_frames=num_frames * 2, ) if not frames: logger.warning("[thumbnail] MediaKit 抽帧返回空") return None if len(frames) > num_frames: step = len(frames) // num_frames frames = [frames[i * step] for i in range(num_frames)] logger.info("[thumbnail] MediaKit 抽帧成功: %d 帧", len(frames)) return frames except Exception as e: logger.warning("[thumbnail] MediaKit 抽帧异常: %s", e) return None finally: try: delete_from_oss(video_storage_key) except Exception: pass def extract_and_upload_cover_frames( video_path: str, plan_id: str, *, task_id: str = "", num_frames: int = 5, title_text: str = "", title_color: str = "#ffffff", title_position: str = "bottom", title_font_size: int | None = None, clip_boundaries: Optional[list[tuple[float, float]]] = None, ) -> list[dict]: """从视频中抽取多帧作为封面候选,通过质量评分选出最佳帧,上传到 OSS。 P2 优化: - 先用 ffmpeg blackdetect 扫描黑屏区间,seek 点自动避开黑屏 - 单次 ffmpeg select 抽 num_frames 帧(避免 5 次起停 ffmpeg 进程) - 多帧 OSS 上传用 ThreadPoolExecutor 并发,目标封面阶段 <1.5s - cv2 清晰度/亮度/色彩三维评分选最佳帧 Fallback(MEDIAKIT_COVER_ENABLED=true):火山 MediaKit SceneChange 抽帧(~60-90s)。 Args: clip_boundaries: 片段边界列表 [(clip_start, clip_duration), ...],用于智能取点 """ import time import httpx from video_processing.ffmpeg_utils import probe_duration from video_processing.oss_helpers import upload_to_oss from packages.shared.config import get_shared_settings t0 = time.monotonic() try: duration = probe_duration(video_path) except Exception: duration = 0.0 candidates: list[dict] = [] _temp_paths: list[str] = [] try: settings = get_shared_settings() use_mediakit = getattr(settings, "mediakit_cover_enabled", False) if use_mediakit: logger.info("[thumbnail] MEDIAKIT_COVER_ENABLED=true,走 MediaKit 路径") mediakit_frames = _extract_frames_via_mediakit(video_path, plan_id, num_frames) if mediakit_frames: for i, frame in enumerate(mediakit_frames): frame_url = frame.get("image_url") if not frame_url: continue tmp = tempfile.NamedTemporaryFile(suffix=".jpg", delete=False) tmp.close() _temp_paths.append(tmp.name) try: resp = httpx.get(frame_url, timeout=30, follow_redirects=True) resp.raise_for_status() with open(tmp.name, "wb") as f: f.write(resp.content) if title_text and title_text.strip(): apply_title_overlay( tmp.name, title_text, color=title_color, position=title_position, font_size=title_font_size, ) storage_key = f"covers/{plan_id}/{task_id}/mediakit_frame_{i}.jpg" url = upload_to_oss(tmp.name, storage_key) if url: candidates.append( { "url": url, "position": round(frame.get("timestamp", 0.0), 2), "image_path": tmp.name, } ) except Exception as e: logger.warning("[thumbnail] MediaKit 帧 %d 处理失败: %s", i, e) if len(candidates) >= num_frames: logger.info("[thumbnail] MediaKit 抽帧完成: %d 帧", len(candidates)) # MediaKit 路径帧在 NamedTemporaryFile 中持久存在(finally 清理),在进入本地 ffmpeg 前评分 if len(candidates) > 1: try: from packages.shared.cover_frame_scorer import score_frames candidates = score_frames(candidates) logger.info( "[thumbnail] MediaKit 封面帧评分完成: count=%d best_score=%.1f", len(candidates), candidates[0].get("score", 0.0) if candidates else 0.0, ) except Exception: logger.warning("[thumbnail] MediaKit 封面帧质量评分失败,保持原始顺序", exc_info=True) # ── 默认路径:本地 ffmpeg 单次 select 抽帧 + 并发上传 ────────────── if len(candidates) < num_frames: if candidates: logger.info("[thumbnail] MediaKit 不足 %d 帧,本地 ffmpeg 补充", num_frames) else: logger.info( "[thumbnail] 使用本地 ffmpeg 抽帧(num=%d, duration=%.1fs)", num_frames, duration, ) # 1) 计算 seek 点 seek_points = _compute_clip_boundary_seek_points(duration, clip_boundaries, num_frames) # 2) 黑屏检测 + 偏移 seek 点 black_intervals = _detect_black_intervals(video_path, duration) if duration > 0 else [] if black_intervals: seek_points = _adjust_seek_points_avoid_black(seek_points, black_intervals, duration) logger.info("[thumbnail] 黑屏规避后 seek 点: %s", seek_points) # 3) 单次 ffmpeg select 抽出所有帧(带失败兜底到单帧 seek) with tempfile.TemporaryDirectory(prefix="thumb_") as frame_dir: t1 = time.monotonic() frame_results = _extract_frames_single_pass( video_path, seek_points, frame_dir, prefix="frame", ) logger.info("[thumbnail] 抽帧耗时: %.2fs (%d 帧)", time.monotonic() - t1, len(frame_results)) # 4) 标题叠加(本地,CPU 很快) for _st, fp in frame_results: if title_text and title_text.strip(): try: apply_title_overlay( fp, title_text, color=title_color, position=title_position, font_size=title_font_size, ) except Exception as e: logger.warning("[thumbnail] 标题叠加失败 %s: %s", fp, e) # 5) 质量评分(必须在 TemporaryDirectory 内,帧文件还在磁盘上) t_score = time.monotonic() local_candidates: list[dict] = [{"position": st, "image_path": fp} for (st, fp) in frame_results] scored: list[dict] = local_candidates if len(local_candidates) > 1: try: from packages.shared.cover_frame_scorer import score_frames scored = score_frames(local_candidates) logger.info( "[thumbnail] 封面评分耗时: %.2fs (best_score=%.1f, count=%d)", time.monotonic() - t_score, scored[0].get("score", 0.0) if scored else 0.0, len(scored), ) except Exception: logger.warning( "[thumbnail] 封面帧质量评分失败,保持 seek 点原始顺序", exc_info=True, ) scored = local_candidates # 6) 按评分顺序并发上传 OSS(best 帧先上传;best 已是 scored[0]) t2 = time.monotonic() def _upload_one(rank: int, st: float, fp: str, score: float) -> dict | None: try: storage_key = f"covers/{plan_id}/{task_id}/frame_{rank}.jpg" url = upload_to_oss(fp, storage_key) if url: return { "url": url, "position": st, "image_path": fp, "score": score, "is_best": rank == 0, } logger.warning("[thumbnail] 上传失败 rank=%d t=%.2f", rank, st) except Exception as e: logger.warning("[thumbnail] 上传异常 rank=%d t=%.2f: %s", rank, st, e) return None upload_results: list[dict | None] = [None] * len(scored) max_workers = min(8, max(2, len(scored))) with ThreadPoolExecutor(max_workers=max_workers) as pool: future_map = { pool.submit( _upload_one, i, float(c.get("position", 0.0)), str(c["image_path"]), float(c.get("score", 0.0)), ): i for i, c in enumerate(scored) } for fut in as_completed(future_map): i = future_map[fut] try: upload_results[i] = fut.result() except Exception as e: logger.warning("[thumbnail] 上传 future 异常 rank=%d: %s", i, e) logger.info("[thumbnail] 并发上传耗时: %.2fs", time.monotonic() - t2) for r in upload_results: if r is not None: # 本地帧在 TemporaryDirectory 内,with 退出自动删除,无需进 _temp_paths candidates.append(r) # 如果本地 ffmpeg 路径产生了候选(已评分)但未经过 MediaKit 路径,candidates 已按评分顺序排好。 # 混合场景下(MediaKit + 本地 ffmpeg 都产出),统一按 score 降序排列;缺失 score 的(理论上不应出现)排末尾。 if len(candidates) > 1: candidates.sort(key=lambda c: c.get("score", -1.0), reverse=True) if candidates: candidates[0]["is_best"] = True elapsed = time.monotonic() - t0 logger.info( "[thumbnail] 封面完成: plan_id=%s count=%d best=t%.2fs score=%.1f elapsed=%.2fs", plan_id, len(candidates), candidates[0].get("position", 0.0), candidates[0].get("score", 0.0), elapsed, ) for c in candidates: c.pop("image_path", None) return candidates finally: for path in _temp_paths: try: Path(path).unlink(missing_ok=True) except Exception: pass