diff --git a/apps/worker/video_processing/gpu_direct_pipeline.py b/apps/worker/video_processing/gpu_direct_pipeline.py index ea4e3bc20..43b48f23d 100644 --- a/apps/worker/video_processing/gpu_direct_pipeline.py +++ b/apps/worker/video_processing/gpu_direct_pipeline.py @@ -4,7 +4,7 @@ 上传后再由 P4000 NVENC 编码,渲染后还要单独跑一次随机边缘裁剪重编码(约 26s)。 本管线取消 mezzanine:把原始素材签名 URL 作为多输入直接交给 P4000,filter_complex 内 一步完成 trim/scale/pad/concat/边缘随机裁剪/drawtext 字幕,末端 h264_nvenc 只编码一次; -原素材音轨 concat + TTS/BGM 混音也在同一命令里完成。 +原素材音轨 concat + TTS/配音/BGM 混音也在同一命令里完成。 约束(P1): - 仅覆盖智能剪辑主流场景:单一主视频轨、全硬切、无 PiP/overlay/水印/贴纸/片头片尾/绿幕。 @@ -138,7 +138,16 @@ def build_direct_render( cq: int = 23, edge_crop_pct: float = 0.0, total_duration: float = 0.0, + clip_has_audio: Optional[list[bool]] = None, + clip_volumes: Optional[list[float]] = None, + extra_audio_tracks: Optional[list[tuple[Any, float]]] = None, ) -> DirectRenderPlan: + """构造 P4000 直连渲染所需的 inputs 与 ffmpeg_args。 + + 视频:每段 trim/setpts/scale/pad/fps → concat(全硬切,带音频)→ 随机边缘 crop+scale → drawtext。 + 音频:每段 [i:a](或 anullsrc 静音占位)按 clip 配置 atrim/asetpts/atempo/volume/aresample + → concat=n:N:v=1:a=1 → 与 extra_audio(TTS/配音素材库)、BGM 一起 amix → atrim 精确截断。 + """ if not resolved_clips: raise ValueError("build_direct_render: no resolved clips") @@ -148,6 +157,18 @@ def build_direct_render( fc: list[str] = [] n = len(resolved_clips) + # 规范化每段参数 + if clip_has_audio is None: + clip_has_audio = [True] * n + else: + clip_has_audio = list(clip_has_audio) + [True] * max(0, n - len(clip_has_audio)) + clip_has_audio = clip_has_audio[:n] + if clip_volumes is None: + clip_volumes = [1.0] * n + else: + clip_volumes = list(clip_volumes) + [1.0] * max(0, n - len(clip_volumes)) + clip_volumes = clip_volumes[:n] + clip_starts: list[float] = [] clip_effs: list[float] = [] clip_speeds: list[float] = [] @@ -161,6 +182,7 @@ def build_direct_render( clip_effs.append(eff) clip_speeds.append(speed) + # 1. 视频输入(原始素材签名 URL) for i, clip in enumerate(resolved_clips): sk = (getattr(clip, "config", None) or {}).get("_storage_key") if not sk: @@ -169,6 +191,7 @@ def build_direct_render( inputs[fname] = sign_asset_url(sk) input_args.extend(["-i", fname]) + # 2. 视频段预处理 pre_labels: list[str] = [] for i in range(n): vf: list[str] = [] @@ -189,10 +212,28 @@ def build_direct_render( fc.append(f"[{i}:v]{','.join(vf)}[{label}]") pre_labels.append(label) + # 2b. 音频段预处理(无声源用 anullsrc 占位;volume=0 的段也用 anullsrc 静音占位保持时间轴) + anullsrc_counter = 0 audio_pre_labels: list[str] = [] for i in range(n): - af: list[str] = [] start, eff, speed = clip_starts[i], clip_effs[i], clip_speeds[i] + vol = float(clip_volumes[i] if i < len(clip_volumes) else 1.0) + has_a = bool(clip_has_audio[i] if i < len(clip_has_audio) else True) + if not has_a or vol <= 0.001: + # 静音占位:用 anullsrc 生成静音,atrim 到段时长 + sl = f"sil{anullsrc_counter}" + anullsrc_counter += 1 + af: list[str] = ["anullsrc=channel_layout=stereo:sample_rate=44100"] + if eff > 0: + af.append(f"atrim=duration={eff:.3f}") + af.append("asetpts=PTS-STARTPTS") + af.append("aformat=sample_fmts=fltp:channel_layouts=stereo") + fc.append(f"{','.join(af)}[{sl}]") + # anullsrc 作为 filter 源不需要 -i 输入,直接给 label + audio_pre_labels.append(sl) + continue + + af = [] if eff > 0: if start > 0: af.append(f"atrim=start={start:.3f}:duration={eff:.3f}") @@ -203,17 +244,21 @@ def build_direct_render( atempo = _build_atempo_chain(speed) if atempo: af.append(atempo) + if abs(vol - 1.0) >= 1e-3: + af.append(f"volume={vol:.3f}") af.append("aresample=44100") af.append("aformat=sample_fmts=fltp:channel_layouts=stereo") alabel = f"ac{i}" fc.append(f"[{i}:a]{','.join(af)}[{alabel}]") audio_pre_labels.append(alabel) + # 3. concat(全硬切;v=1:a=1,视频音频一起拼接) concat_in = "".join(f"[{v}][{a}]" for v, a in zip(pre_labels, audio_pre_labels, strict=True)) fc.append(f"{concat_in}concat=n={n}:v=1:a=1[vcat][acat]") cur_v = "vcat" cur_a = "acat" + # 4. 随机边缘裁剪降重(四边独立随机 2%~5%,与 ffmpeg_utils.random_edge_crop 一致) if edge_crop_pct and edge_crop_pct > 0: _r = random.Random() p_min = EDGE_CROP_MIN_PCT @@ -232,6 +277,7 @@ def build_direct_render( ) cur_v = "vcrop" + # 5. drawtext 字幕 draw_filters: list[str] = [] if title_text.strip(): title_size = max(int(output_height * 0.05), 24) @@ -274,9 +320,35 @@ def build_direct_render( fc.append(f"[{cur_v}]format=yuv420p[vfinal]") vfinal_label = "vfinal" + # 6. 音频混音:原素材主音轨 acat + extra(TTS/配音素材库) + BGM → amix → atrim mix_labels: list[str] = [cur_a] + mix_vols: list[float] = [1.0] next_idx = n + + # 额外独立音频轨(TTS concat / 配音素材库整段音频) + for _ea_idx, (ea_path, ea_vol) in enumerate(extra_audio_tracks or []): + if ea_path is None: + continue + ea_p = Path(ea_path) + if not ea_p.exists(): + continue + eurl, ekey = upload_local_audio_and_sign(ea_p) + ename = f"extra{_ea_idx}{ea_p.suffix or '.mp3'}" + inputs[ename] = eurl + oss_keys.append(ekey) + input_args.extend(["-i", ename]) + elabel = f"aex{_ea_idx}" + fc.append( + f"[{next_idx}:a]aresample=44100,volume={float(ea_vol):.2f}," + f"aformat=sample_fmts=fltp:channel_layouts=stereo[{elabel}]" + ) + mix_labels.append(elabel) + mix_vols.append(float(ea_vol)) + next_idx += 1 + if tts_audio and Path(tts_audio).exists(): + # 旧参数保留:若调用方直接传了 tts_audio 而没走 extra_audio_tracks,则仍然加入 + # (兼容旧调用,正常路径 TTS 已经通过 extra_audio_tracks 传入) turl, tkey = upload_local_audio_and_sign(Path(tts_audio)) tname = "tts" + (Path(tts_audio).suffix or ".mp3") inputs[tname] = turl @@ -287,6 +359,7 @@ def build_direct_render( f"[{next_idx}:a]aresample=44100,volume=1.00,aformat=sample_fmts=fltp:channel_layouts=stereo[{alabel}]" ) mix_labels.append(alabel) + mix_vols.append(1.0) next_idx += 1 if bgm_audio and Path(bgm_audio).exists(): burl, bkey = upload_local_audio_and_sign(Path(bgm_audio)) @@ -299,6 +372,7 @@ def build_direct_render( f"[{next_idx}:a]aresample=44100,volume=0.35,aformat=sample_fmts=fltp:channel_layouts=stereo[{alabel}]" ) mix_labels.append(alabel) + mix_vols.append(0.35) next_idx += 1 maps: list[str] = ["-map", f"[{vfinal_label}]"] @@ -309,6 +383,7 @@ def build_direct_render( f"amix=inputs={n_mix}:duration=longest:dropout_transition=2:normalize=0", "aresample=44100", ] + # Bug2 修复:atrim 到视频精确时长 if total_duration and total_duration > 0: mix_parts.append(f"atrim=0:{total_duration:.3f}") mix_parts.append("asetpts=PTS-STARTPTS") @@ -317,6 +392,7 @@ def build_direct_render( else: logger.info("[gpu-direct] no audio tracks; output silent video") + # 7. 组装 ffmpeg_args + NVENC 编码 ffmpeg_args = ["-y", *input_args, "-filter_complex", ";".join(fc), *maps] ffmpeg_args.extend(["-c:v", vcodec, "-preset", preset, "-pix_fmt", "yuv420p"]) if video_bitrate: diff --git a/apps/worker/video_processing/unified_render_service.py b/apps/worker/video_processing/unified_render_service.py index 4001acf17..9ea73c30a 100755 --- a/apps/worker/video_processing/unified_render_service.py +++ b/apps/worker/video_processing/unified_render_service.py @@ -2279,13 +2279,34 @@ class UnifiedRenderService: video_layer = next(_lyr for _lyr in layers if _lyr.role not in ("audio",)) video_clips = [c for c in video_layer.clips if c.clip_type != "audio"] - # TTS:把 audio 层的 TTS 片段合并成一个文件给 P4000 - tts_merged: Path | None = None + # 音频层处理:收集 TTS 分段与配音素材库整段音频 + # - TTS 分段(带 tts 标记)→ 无间隙 concat 成单文件 + # - 配音素材库(voice_library=True)→ 单独作为整段音轨(不走分段 concat,已从 0 覆盖整段) audio_layer = next((_lyr for _lyr in layers if _lyr.role == "audio"), None) + tts_merged: Path | None = None + voiceover_track: Path | None = None if audio_layer: tts_clips = [c for c in audio_layer.clips if (c.config or {}).get("tts") and c.local_path.exists()] if tts_clips: tts_merged = self._concat_audio_clips(tts_clips, tag="tts_direct") + # 配音素材库整段音频(按 _maybe_add_voice_library_layer 约定只有一个 clip_id=voice_library_main) + vo_clips = [ + c for c in audio_layer.clips if (c.config or {}).get("voice_library") and c.local_path.exists() + ] + if vo_clips: + voiceover_track = vo_clips[-1].local_path # 理论上只有一个,取最后一个 + logger.info( + "[gpu-direct] 配音素材库音轨: plan_id=%s path=%s", + self.plan.id, + voiceover_track, + ) + + # 额外独立音轨(TTS concat、配音素材库)→ gpu_direct_pipeline 会与主音轨/BGM 一起 amix + extra_audio_tracks: list[tuple[Path, float]] = [] + if tts_merged: + extra_audio_tracks.append((tts_merged, 1.0)) + if voiceover_track: + extra_audio_tracks.append((voiceover_track, 1.0)) # BGM 本地文件 bgm_path = Path(self.bgm_path) if self.bgm_path else None @@ -2304,21 +2325,39 @@ class UnifiedRenderService: if sub_cfg.get("auto_generated") and self._asr_timeline_cache is not None: subtitle_segments = list(self._asr_timeline_cache.segments) - # 边缘裁剪比例(与 random_edge_crop 默认 2~5% 同口径,取固定 3%) + # 边缘裁剪:dedup 开启时在 GPU 内做四边随机 2~5% 裁剪(gpu_direct_pipeline 内部随机) dedup = self._dedup_enabled() - edge_pct = 0.03 if dedup else 0.0 + edge_pct = 0.03 if dedup else 0.0 # >0 表示启用;实际区间 [2%,5%] 在 pipeline 内随机 + + # 探测每个视频素材是否含音轨、读取 volume 配置 + clip_has_audio_list: list[bool] = [] + clip_volumes_list: list[float] = [] + for c in video_clips: + lp = getattr(c, "local_path", None) + _ha = False + if lp and Path(lp).exists(): + try: + _ha = probe_has_audio(str(lp)) + except Exception as _pe: # noqa: BLE001 + logger.warning("[gpu-direct] probe_has_audio 失败按有声处理: %s", _pe) + _ha = True + clip_has_audio_list.append(_ha) + _vol = float((c.config or {}).get("volume", 1.0)) + clip_volumes_list.append(_vol if _vol > 0 else 0.0) plan = gdp.build_direct_render( resolved_clips=video_clips, output_width=self.output_width, output_height=self.output_height, output_fps=self.output_fps, - tts_audio=tts_merged, bgm_audio=bgm_path, title_text=title_text, subtitle_segments=subtitle_segments, edge_crop_pct=edge_pct, total_duration=video_duration, + clip_has_audio=clip_has_audio_list, + clip_volumes=clip_volumes_list, + extra_audio_tracks=extra_audio_tracks, ) client = get_gpu_encoder()