"""统一渲染引擎 — 输入 EditPlan + EditPlanClips,按时间线+图层渲染视频. 核心原则(灵应):渲染引擎是统一的,不判断模式,只按 clip_type/config.role 分组为图层再合成。 图层分组: main (无 config.role) → main (z=0) main + config.role=b_roll → broll (z=0,与 main 同层替换) overlay → overlay (z=1,画中画叠加) background → background (z=0,全屏底图) corner_voice → corner_voice (z=1,右上角小窗) b_roll → broll (z=0) intro / outro → main (z=0,按 order 排在首/尾) 合成流程: 1. 每个 clip 先 trim + scale + setpts 预处理 2. 同层 clips 按 order 用 xfade 串联 3. overlay/corner_voice 层 overlay 到主层 4. 如有独立音频轨,amix 混入 """ from __future__ import annotations import logging import subprocess import time from dataclasses import dataclass, field from pathlib import Path from typing import Any from video_processing.ffmpeg_utils import ( DEFAULT_FPS, DEFAULT_OUTPUT_HEIGHT, DEFAULT_OUTPUT_WIDTH, DEFAULT_TRANSITION_DURATION, FFMPEG_BIN, build_xfade_filter_chain, probe_duration, probe_video_info, run_ffmpeg, ) from video_processing.render_audio import RenderContext, merge_audio_video, mix_audio from video_processing.render_subtitles import generate_ass_subtitles logger = logging.getLogger(__name__) # ── 数据结构 ────────────────────────────────────────────────────────────────── @dataclass class ResolvedClip: """已解析到本地路径的片段。""" clip_id: str asset_id: str local_path: Path clip_type: str order: int start_time: float = 0.0 duration: float = 0.0 # 0 表示使用素材完整时长 transition_effect: str = "cut" config: dict[str, Any] = field(default_factory=dict) # 运行时填充 actual_duration: float = 0.0 # 素材实际时长(probe 后填充) @dataclass class RenderLayer: """渲染图层。""" role: str # "main" | "overlay" | "pip" | "background" | "corner_voice" | "broll" | "audio" clips: list[ResolvedClip] = field(default_factory=list) z_index: int = 0 opacity: float = 1.0 position: tuple[int, int] | None = None # (x, y) 偏移,None 表示全屏 @dataclass class RenderResult: """渲染结果。""" output_path: Path duration: float file_size: int width: int height: int # ── clip_type → layer role 映射 ────────────────────────────────────────────── def _resolve_layer_role(clip_type: str, config: dict[str, Any]) -> str: """根据 clip_type 和 config.role 确定图层角色。 映射规则: intro / outro → "main"(按 order 排在首/尾) overlay → "overlay"(画中画叠加,z=1) corner_voice → "corner_voice"(右上角小窗,z=1) background → "background"(全屏底图,z=0) b_roll → "broll"(z=0) main + config.role=b_roll → "broll" main (default) → "main" """ role = config.get("role", "") if clip_type in ("intro", "outro"): return "main" if clip_type == "overlay": return "overlay" if clip_type == "corner_voice": return "corner_voice" if clip_type == "background": return "background" if clip_type == "b_roll": return "broll" # main type if role == "b_roll": return "broll" if role == "audio": return "audio" return "main" # ── 图层默认 z_index ───────────────────────────────────────────────────────── _LAYER_Z_INDEX: dict[str, int] = { "background": -1, "broll": 0, "main": 0, "overlay": 1, "corner_voice": 1, "audio": 2, } # 图层默认 PiP 位置(相对输出画布的偏移) _PIP_SCALE = 0.25 # PiP 占主画面的比例 # ── 统一渲染引擎 ───────────────────────────────────────────────────────────── class UnifiedRenderService: """统一渲染引擎。 输入 EditPlan + EditPlanClips + 素材路径映射,按时间线+图层执行渲染。 """ def __init__( self, plan: Any, # EditPlan clips: list[Any], # list[EditPlanClip] asset_path_map: dict[str, Path], # asset_id → local_path work_dir: Path, *, output_width: int = DEFAULT_OUTPUT_WIDTH, output_height: int = DEFAULT_OUTPUT_HEIGHT, output_fps: int = DEFAULT_FPS, transition_duration: float = DEFAULT_TRANSITION_DURATION, ): self.plan = plan self.clips = clips self.asset_path_map = asset_path_map self.work_dir = work_dir self.output_width = output_width self.output_height = output_height self.output_fps = output_fps self.transition_duration = transition_duration def render(self) -> RenderResult: """执行渲染,返回 RenderResult. 优化路径: - 单图层单 clip → 直通模式(-vf),性能最优 - 其他情况 → 完整 filter_complex 渲染 字幕渲染流程: 1. 视频主渲染(直通或完整链路) 2. 如有 title/subtitle,叠加 ASS 字幕 音频后处理: 1. 主图层音频 concat 拼接 2. 独立音频轨 amix 混入 3. 合并到输出视频 Raises: ValueError: 没有可渲染的片段时抛出 """ t_start = time.time() # 1. 解析 clips → ResolvedClips(跳过无素材的 clip) resolved = self._resolve_clips() if not resolved: raise ValueError("没有可渲染的片段(所有片段素材缺失或下载失败)") # 2. 分组为 RenderLayers layers = self._group_clips_into_layers(resolved) # 3. 计算视频总时长(用于字幕显示时长) video_duration = self._estimate_total_duration(layers) # 4. 生成 ASS 字幕文件(如果有 title/subtitle 配置) ass_path = self._maybe_generate_ass(video_duration) # 灰度埋点:开始渲染 layer_roles = [layer.role for layer in layers] clip_counts = {layer.role: len(layer.clips) for layer in layers} logger.info( "[unified-render] start render: plan_id=%s clip_count=%d layers=%s clip_counts=%s", self.plan.id, len(resolved), layer_roles, clip_counts, ) # 5. 视频主渲染 t_video_start = time.time() video_only_path = self.work_dir / f"rendered_{self.plan.id}_video.mp4" output_path = self.work_dir / f"rendered_{self.plan.id}.mp4" is_pass_through = self._can_use_pass_through(layers) pass_through_has_audio = False used_stream_copy = False if is_pass_through: # 先尝试 stream copy 优化(无重编码,性能提升 10 倍+) # 条件不满足或失败时回退到带滤镜的直通渲染 stream_copy_ok = self._try_render_stream_copy( layers, output_path, ass_path=ass_path, video_duration=video_duration ) if stream_copy_ok: used_stream_copy = True # stream copy 模式下,直接探测输出是否有音频 clip = layers[0].clips[0] info = probe_video_info(str(clip.local_path)) pass_through_has_audio = info.get("has_audio", True) else: # 回退到带滤镜的直通渲染 pass_through_has_audio = self._render_pass_through( layers, output_path, ass_path=ass_path, video_duration=video_duration, ) else: filter_complex, input_args = self._build_filter_complex(layers, ass_path=ass_path) self._execute_ffmpeg(filter_complex, input_args, video_only_path) t_video_end = time.time() video_render_ms = int((t_video_end - t_video_start) * 1000) logger.info( "[unified-render] video render done: plan_id=%s duration_ms=%d pass_through=%s stream_copy=%s", self.plan.id, video_render_ms, is_pass_through, used_stream_copy, ) # 6. 音频后处理混音(直通场景已合并处理,跳过) t_audio_start = time.time() audio_mix_ms = 0 has_audio = False if is_pass_through: # 直通场景已在一次调用中完成视频+音频 has_audio = pass_through_has_audio else: ctx = RenderContext(work_dir=self.work_dir, plan_id=self.plan.id) audio_path = mix_audio(ctx, layers, video_duration) t_audio_end = time.time() audio_mix_ms = int((t_audio_end - t_audio_start) * 1000) has_audio = audio_path is not None if has_audio: logger.info( "[unified-render] audio mix done: plan_id=%s duration_ms=%d", self.plan.id, audio_mix_ms, ) # 7. 合并音视频 merge_audio_video(ctx, video_only_path, audio_path, output_path) else: # 无音频,直接用无声视频 import shutil shutil.copy2(video_only_path, output_path) # 8. 探测输出 duration, file_size, width, height = self._probe_output(output_path) t_total = int((time.time() - t_start) * 1000) logger.info( "[unified-render] render done: plan_id=%s total_ms=%d video_ms=%d audio_ms=%d " "output_duration=%.2fs output_size=%d resolution=%dx%d has_audio=%s", self.plan.id, t_total, video_render_ms, audio_mix_ms if has_audio else 0, duration, file_size, width, height, has_audio, ) return RenderResult( output_path=output_path, duration=duration, file_size=file_size, width=width, height=height, ) def _estimate_total_duration(self, layers: list[RenderLayer]) -> float: """估算视频总时长(用于字幕等需要)。 取主图层(main/broll/background)的总时长,转场重叠按 transition_duration 估算。 """ # 找主图层(第一个有视频内容的图层) main_layer = None for role in ("main", "broll", "background"): for layer in layers: if layer.role == role: main_layer = layer break if main_layer: break if not main_layer or not main_layer.clips: return 0.0 total = sum(UnifiedRenderService._clip_effective_duration(c) for c in main_layer.clips) # 减去转场重叠时间(粗略估算) n_clips = len(main_layer.clips) if n_clips > 1: total -= (n_clips - 1) * self.transition_duration return max(0.1, total) def _maybe_generate_ass(self, video_duration: float) -> Path | None: """根据 plan.config 生成 ASS 字幕文件。 Returns: ASS 文件路径,没有字幕时返回 None """ config = self.plan.config or {} title_cfg = config.get("title", {}) or {} subtitle_cfg = config.get("subtitle", {}) or {} title_enabled = title_cfg.get("enabled", True) subtitle_enabled = subtitle_cfg.get("enabled", True) title_text = title_cfg.get("text", "") or "" subtitle_text = subtitle_cfg.get("text", "") or "" has_title = title_enabled and bool(title_text.strip()) has_subtitle = subtitle_enabled and bool(subtitle_text.strip()) if not has_title and not has_subtitle: return None ass_path = self.work_dir / f"subtitles_{self.plan.id}.ass" generate_ass_subtitles( ass_path, video_width=self.output_width, video_height=self.output_height, video_duration=video_duration, title_text=title_text, title_config=title_cfg, subtitle_text=subtitle_text, subtitle_config=subtitle_cfg, ) logger.info( "生成字幕: plan_id=%s title=%s subtitle=%s ass=%s", self.plan.id, has_title, has_subtitle, ass_path, ) return ass_path def _can_use_pass_through(self, layers: list[RenderLayer]) -> bool: """判断是否可以走直通优化路径。 条件: 1. 只有 1 个图层 2. 该图层是视频图层(main/broll/background),不是 overlay/corner_voice/audio 3. 该图层只有 1 个 clip(无转场需求) """ if len(layers) != 1: return False layer = layers[0] if layer.role not in ("main", "broll", "background"): return False if len(layer.clips) != 1: return False return True def _can_use_stream_copy( self, clip: ResolvedClip, *, ass_path: Path | None = None, video_duration: float = 0.0, ) -> tuple[bool, str]: """判断是否可以走 stream copy(流拷贝,不重编码)。 性能提升:10 倍以上(典型场景从 20s → 1-2s)。 条件: 1. 视频编码为 h264(输出目标也是 h264) 2. 像素格式为 yuv420p 3. 分辨率与输出一致(不需要 scale/crop) 4. 帧率与输出一致(误差 < 0.1fps) 5. 无字幕叠加(字幕需要滤镜) 6. 无 trim 需求(或 trim 后恰好等于原时长) 7. 无转场、无特效(单 clip 直通已保证) Returns: (是否可以 copy, 原因说明) """ # 有字幕 → 需要滤镜 → 不能 copy if ass_path is not None: return False, "有字幕叠加" # 探测输入视频参数 info = probe_video_info(str(clip.local_path)) # 编码必须是 h264 if info.get("video_codec", "") != "h264": return False, f"视频编码不是h264: {info.get('video_codec', 'unknown')}" # 像素格式必须是 yuv420p if info.get("pix_fmt", "") != "yuv420p": return False, f"像素格式不是yuv420p: {info.get('pix_fmt', 'unknown')}" # 分辨率必须一致 if info.get("width", 0) != self.output_width or info.get("height", 0) != self.output_height: return False, ( f"分辨率不匹配: " f"{info.get('width', 0)}x{info.get('height', 0)} " f"vs {self.output_width}x{self.output_height}" ) # 帧率必须一致(误差 < 0.1fps) fps_diff = abs(info.get("fps", 0) - self.output_fps) if fps_diff > 0.1: return False, f"帧率不匹配: {info.get('fps', 0)} vs {self.output_fps}" # 检查是否需要 trim effective_duration = UnifiedRenderService._clip_effective_duration(clip) if effective_duration > 0: # 有 trim 需求但视频时长足够,可用 -ss/-t 实现 copy trim input_duration = info.get("duration", 0) if input_duration <= 0: return False, "无法探测输入时长" # trim 起始点 + 目标时长 <= 输入时长 start_time = getattr(clip, "start_time", 0) or 0 if start_time + effective_duration > input_duration + 0.1: return False, "trim 超出输入时长" # video_duration 截断 if video_duration > 0 and effective_duration > 0: final_duration = min(effective_duration, video_duration) if final_duration != effective_duration: # 也需要截断,但 -t 可以 copy 模式下用 pass return True, "所有条件满足" def _try_render_stream_copy( self, layers: list[RenderLayer], output_path: Path, *, ass_path: Path | None = None, video_duration: float = 0.0, ) -> bool: """尝试 stream copy 渲染,成功返回 True,失败返回 False(调用方回退到重编码)。 stream copy 模式:不重编码,直接拷贝视频/音频流,性能提升 10 倍+。 仅用于单 clip 直通场景且满足 copy 条件。 """ clip = layers[0].clips[0] role = layers[0].role # 判断是否满足 copy 条件 can_copy, reason = self._can_use_stream_copy(clip, ass_path=ass_path, video_duration=video_duration) if not can_copy: logger.info( "[unified-render] stream_copy 跳过: plan_id=%s reason=%s", self.plan.id, reason, ) return False # 构建 copy 命令 command = [ FFMPEG_BIN, "-y", ] # trim 支持(-ss 放在 -i 前 = input seeking,速度更快但精度稍差; # 放在 -i 后 = output seeking,精度高但慢) # 这里用 output seeking 保证精度,反正 copy 模式已经很快了 start_time = getattr(clip, "start_time", 0) or 0 effective_duration = UnifiedRenderService._clip_effective_duration(clip) command.extend(["-i", str(clip.local_path)]) if start_time > 0: command.extend(["-ss", f"{start_time:.3f}"]) # 计算最终时长 final_duration = effective_duration if video_duration > 0 and (final_duration <= 0 or final_duration > video_duration): final_duration = video_duration if final_duration > 0: command.extend(["-t", f"{final_duration:.3f}"]) # 流拷贝 command.extend( [ "-c:v", "copy", "-c:a", "copy", "-movflags", "+faststart", str(output_path), ] ) logger.info( "[unified-render] stream_copy 渲染: plan_id=%s clip=%s role=%s duration=%.2fs", self.plan.id, clip.clip_id, role, final_duration, ) try: run_ffmpeg(command) # 验证输出文件存在且有大小 if output_path.exists() and output_path.stat().st_size > 0: logger.info( "[unified-render] stream_copy 成功: plan_id=%s size=%d", self.plan.id, output_path.stat().st_size, ) return True else: logger.warning("[unified-render] stream_copy 输出为空: plan_id=%s", self.plan.id) return False except (subprocess.CalledProcessError, subprocess.TimeoutExpired) as e: logger.warning( "[unified-render] stream_copy 失败,回退到重编码: plan_id=%s error=%s", self.plan.id, str(e)[:200], ) # 清理可能的损坏输出文件 if output_path.exists(): try: output_path.unlink() except OSError as unlink_err: logger.warning( "[unified-render] 损坏输出文件清理失败: path=%s error=%s", output_path, unlink_err, ) return False def _render_pass_through( self, layers: list[RenderLayer], output_path: Path, *, ass_path: Path | None = None, video_duration: float = 0.0, ) -> bool: """单图层单 clip 直通渲染(使用 -vf 而非 -filter_complex),一次性输出带音频的最终视频。 性能优化: - 避免 filter_complex 的解析和调度开销,单clip场景性能提升 ~30% - 视频+音频一次FFmpeg调用完成,省去后续音频提取+音视频合并两次调用 Args: layers: 图层列表(只有1个图层1个clip) output_path: 输出文件路径 ass_path: ASS 字幕文件路径,有则叠加字幕 video_duration: 视频总时长(用于截断音频,0表示不额外截断) Returns: True 表示输出包含音频(近似判断,实际以输出文件为准) """ clip = layers[0].clips[0] role = layers[0].role # 构建视频滤镜链(与 _build_filter_complex 中预处理逻辑一致) filters: list[str] = [] # trim effective_duration = UnifiedRenderService._clip_effective_duration(clip) if effective_duration > 0: filters.append(f"trim=duration={effective_duration}") filters.append("setpts=PTS-STARTPTS") # scale + crop(铺满裁剪) if role in ("overlay", "corner_voice"): pip_w = int(self.output_width * _PIP_SCALE) pip_h = int(self.output_height * _PIP_SCALE) filters.append(f"scale={pip_w}:{pip_h}") else: # main / broll / background: 铺满裁剪 filters.append(f"scale={self.output_width}:{self.output_height}" ":force_original_aspect_ratio=increase") filters.append(f"crop={self.output_width}:{self.output_height}") filters.append("setpts=PTS-STARTPTS") filters.append(f"fps={self.output_fps}") filters.append("format=yuv420p") # 字幕叠加 if ass_path is not None: ass_filter_path = str(ass_path).replace("\\", "/").replace(":", "\\:") filters.append(f"subtitles='{ass_filter_path}'") vf_str = ",".join(filters) # 最终输出时长:取 clip 有效时长和 video_duration 的较小值 final_duration = effective_duration if video_duration > 0 and (final_duration <= 0 or final_duration > video_duration): final_duration = video_duration command = [ FFMPEG_BIN, "-y", "-i", str(clip.local_path), "-vf", vf_str, "-c:v", "libx264", "-crf", "23", "-preset", "medium", "-pix_fmt", "yuv420p", "-movflags", "+faststart", ] # 音频处理:background 通常是图片无音频,跳过;其他编码为 aac # background 以外的视频素材,默认带音频 has_audio = role != "background" if has_audio: command.extend(["-c:a", "aac", "-b:a", "128k"]) # 统一截断时长(同时作用于视频和音频) if final_duration > 0: command.extend(["-t", f"{final_duration:.3f}"]) command.append(str(output_path)) logger.info( "直通渲染: plan_id=%s clip=%s role=%s duration=%.2fs has_audio=%s", self.plan.id, clip.clip_id, role, effective_duration, has_audio, ) try: run_ffmpeg(command) except subprocess.CalledProcessError as e: logger.error( "直通渲染失败: plan_id=%s clip=%s exit_code=%d\nvf=%s", self.plan.id, clip.clip_id, e.returncode, vf_str[:2000], ) raise return has_audio # ── 内部方法 ────────────────────────────────────────────────────────────── def _resolve_clips(self) -> list[ResolvedClip]: """将 EditPlanClip 列表解析为 ResolvedClip 列表。 跳过 asset_id 为空或在 asset_path_map 中找不到的片段。 """ resolved: list[ResolvedClip] = [] for clip in self.clips: asset_id = clip.asset_id if not asset_id: logger.warning("片段无素材: clip_id=%s", clip.id) continue local_path = self.asset_path_map.get(asset_id) if local_path is None or not local_path.exists(): logger.warning("素材不存在: clip_id=%s asset_id=%s", clip.id, asset_id) continue # 探测实际时长 try: actual_duration = probe_duration(local_path) except Exception: actual_duration = clip.duration or 5.0 rc = ResolvedClip( clip_id=clip.id, asset_id=asset_id, local_path=local_path, clip_type=clip.clip_type, order=clip.order, start_time=clip.start_time, duration=clip.duration, transition_effect=clip.transition_effect or "cut", config=clip.config or {}, actual_duration=actual_duration, ) resolved.append(rc) # 按 order 排序 resolved.sort(key=lambda c: c.order) return resolved def _group_clips_into_layers(self, resolved_clips: list[ResolvedClip]) -> list[RenderLayer]: """将 ResolvedClips 分组为 RenderLayers。 分组规则见 _resolve_layer_role 函数文档。 """ layer_map: dict[str, RenderLayer] = {} for clip in resolved_clips: role = _resolve_layer_role(clip.clip_type, clip.config) if role not in layer_map: z = _LAYER_Z_INDEX.get(role, 0) layer_map[role] = RenderLayer(role=role, z_index=z) layer_map[role].clips.append(clip) # 每个 layer 内的 clips 按 order 排序 for layer in layer_map.values(): layer.clips.sort(key=lambda c: c.order) # 计算 PiP 位置 pip_width = int(self.output_width * _PIP_SCALE) margin = 20 # 边距 if "overlay" in layer_map: layer_map["overlay"].position = ( self.output_width - pip_width - margin, margin, ) if "corner_voice" in layer_map: layer_map["corner_voice"].position = ( self.output_width - pip_width - margin, margin, ) # 按 z_index 排序返回 layers = sorted(layer_map.values(), key=lambda lyr: lyr.z_index) return layers def _build_filter_complex( self, layers: list[RenderLayer], *, ass_path: Path | None = None ) -> tuple[str, list[str]]: """构建 FFmpeg filter_complex 字符串和输入参数列表。 Args: layers: 图层列表 ass_path: ASS 字幕文件路径,有则在最后叠加字幕 Returns: (filter_complex_str, input_args_list) input_args_list 是 ["-i", path1, "-i", path2, ...] 格式 """ if not layers: raise ValueError("没有可渲染的图层") # 收集所有 clips(按图层顺序,同层按 order) all_clips: list[ResolvedClip] = [] for layer in layers: all_clips.extend(layer.clips) # 构建输入参数 input_args: list[str] = [] clip_to_input_idx: dict[str, int] = {} for i, clip in enumerate(all_clips): input_args.extend(["-i", str(clip.local_path)]) clip_to_input_idx[clip.clip_id] = i filter_parts: list[str] = [] # Step 1: 预处理每个 clip — scale + setpts # 为每个 clip 生成预处理后的标签 [v0], [v1], ... preprocessed_labels: list[str] = [] for i, clip in enumerate(all_clips): label = f"v{i}" role = _resolve_layer_role(clip.clip_type, clip.config) filters: list[str] = [] # trim — 始终将输出截断到有效时长,防止 xfade offset 与实际时长不匹配 effective_duration = UnifiedRenderService._clip_effective_duration(clip) if effective_duration > 0: filters.append(f"trim=duration={effective_duration}") filters.append("setpts=PTS-STARTPTS") # scale if role in ("overlay", "corner_voice"): pip_w = int(self.output_width * _PIP_SCALE) pip_h = int(self.output_height * _PIP_SCALE) filters.append(f"scale={pip_w}:{pip_h}") elif role == "background": filters.append( f"scale={self.output_width}:{self.output_height}" ":force_original_aspect_ratio=increase" ) filters.append(f"crop={self.output_width}:{self.output_height}") else: # main / broll: 铺满裁剪(scale to cover + center crop) # 对齐链路A编辑器合成行为,与主流短视频平台一致 filters.append( f"scale={self.output_width}:{self.output_height}" ":force_original_aspect_ratio=increase" ) filters.append(f"crop={self.output_width}:{self.output_height}") filters.append("setpts=PTS-STARTPTS") filters.append(f"fps={self.output_fps}") filter_str = f"[{i}:v]{','.join(filters)}[{label}]" filter_parts.append(filter_str) preprocessed_labels.append(label) # Step 2: 同层 clips 用 xfade 串联 layer_output_labels: dict[str, str] = {} for layer in layers: layer_clip_indices = [all_clips.index(c) for c in layer.clips] layer_labels = [preprocessed_labels[i] for i in layer_clip_indices] # 使用 trim 后的有效时长,与 Step 1 的 trim=duration 保持一致 layer_durations = [UnifiedRenderService._clip_effective_duration(all_clips[i]) for i in layer_clip_indices] layer_transitions = [all_clips[i].transition_effect for i in layer_clip_indices] if len(layer_labels) == 1: # 单 clip 层,直接使用预处理标签 layer_output_labels[layer.role] = layer_labels[0] else: # 多 clip 层,用 xfade 串联 out_label = f"{layer.role}_merged" xfade_filter, _ = build_xfade_filter_chain( clip_durations=layer_durations, clip_video_labels=layer_labels, transitions=layer_transitions, transition_duration=self.transition_duration, output_label=out_label, ) if xfade_filter: filter_parts.append(xfade_filter) layer_output_labels[layer.role] = out_label # Step 3: 合成各层 # 找到主层 — background 优先作为底图,其次 broll / main final_video_label = None if "background" in layer_output_labels: final_video_label = layer_output_labels["background"] # b_roll / main 叠加到 background 上 for role in ("broll", "main"): if role in layer_output_labels: base_label = layer_output_labels[role] combined_label = f"combined_{role}" filter_parts.append( f"[{final_video_label}][{base_label}]" f"overlay=(W-w)/2:(H-h)/2[{combined_label}]" ) final_video_label = combined_label else: # 无 background 时,取 broll 或 main 作为基础 for role in ("broll", "main"): if role in layer_output_labels: final_video_label = layer_output_labels[role] break if final_video_label is None: # 没有任何主层,使用第一个层 final_video_label = layer_output_labels[layers[0].role] # 叠加 overlay 层 for layer in layers: if layer.role in ("overlay", "corner_voice"): if layer.role not in layer_output_labels: continue overlay_label = layer_output_labels[layer.role] x, y = layer.position or ( self.output_width - int(self.output_width * _PIP_SCALE) - 20, 20, ) combined_label = f"combined_{layer.role}" filter_parts.append(f"[{final_video_label}][{overlay_label}]" f"overlay={x}:{y}[{combined_label}]") final_video_label = combined_label # 叠加字幕(如有)+ 最终像素格式 if ass_path is not None: ass_filter_path = str(ass_path).replace("\\", "/").replace(":", "\\:") filter_parts.append(f"[{final_video_label}]subtitles='{ass_filter_path}',format=yuv420p[final_video]") else: filter_parts.append(f"[{final_video_label}]format=yuv420p[final_video]") filter_complex = ";".join(filter_parts) return filter_complex, input_args def _execute_ffmpeg( self, filter_complex: str, input_args: list[str], output_path: Path, ) -> None: """执行 FFmpeg 渲染命令。 失败时记录完整 filter_complex 以便排查(如 exit code 183)。 """ command = [ FFMPEG_BIN, "-y", *input_args, "-filter_complex", filter_complex, "-map", "[final_video]", "-c:v", "libx264", "-crf", "23", "-preset", "medium", "-pix_fmt", "yuv420p", "-movflags", "+faststart", str(output_path), ] logger.info( "执行渲染: plan_id=%s inputs=%d output=%s", self.plan.id, input_args.count("-i"), output_path, ) try: run_ffmpeg(command) except subprocess.CalledProcessError as e: # 额外记录 filter_complex,方便排查滤镜链构建问题 logger.error( "渲染失败: plan_id=%s exit_code=%d\nfilter_complex:\n%s", self.plan.id, e.returncode, filter_complex[:5000], ) raise def _probe_output(self, output_path: Path) -> tuple[float, int, int, int]: """探测输出文件的时长、大小、宽高. Returns: (duration, file_size, width, height) """ info = probe_video_info(str(output_path)) file_size = output_path.stat().st_size if output_path.exists() else 0 return ( info["duration"], file_size, info["width"], info["height"], ) @staticmethod def _clip_effective_duration(clip: ResolvedClip) -> float: """计算 clip 的有效时长.""" if clip.duration > 0: return min(clip.duration, clip.actual_duration) if clip.actual_duration > 0 else clip.duration return clip.actual_duration if clip.actual_duration > 0 else 0.0