fix(worker): GPU直连支持配音素材库音轨+无声片段anullsrc+per-clip音量 #2090

Merged
auto-approve-bot merged 1 commits from fix/gpu-direct-voice-library-and-silent into develop 2026-09-28 21:09:11 +08:00
2 changed files with 122 additions and 7 deletions
@@ -4,7 +4,7 @@
上传后再由 P4000 NVENC 编码,渲染后还要单独跑一次随机边缘裁剪重编码(约 26s)。
本管线取消 mezzanine:把原始素材签名 URL 作为多输入直接交给 P4000,filter_complex 内
一步完成 trim/scale/pad/concat/边缘随机裁剪/drawtext 字幕,末端 h264_nvenc 只编码一次;
原素材音轨 concat + TTS/BGM 混音也在同一命令里完成。
原素材音轨 concat + TTS/配音/BGM 混音也在同一命令里完成。
约束(P1):
- 仅覆盖智能剪辑主流场景:单一主视频轨、全硬切、无 PiP/overlay/水印/贴纸/片头片尾/绿幕。
@@ -138,7 +138,16 @@ def build_direct_render(
cq: int = 23,
edge_crop_pct: float = 0.0,
total_duration: float = 0.0,
clip_has_audio: Optional[list[bool]] = None,
clip_volumes: Optional[list[float]] = None,
extra_audio_tracks: Optional[list[tuple[Any, float]]] = None,
) -> DirectRenderPlan:
"""构造 P4000 直连渲染所需的 inputs 与 ffmpeg_args。
视频:每段 trim/setpts/scale/pad/fps → concat(全硬切,带音频)→ 随机边缘 crop+scale → drawtext。
音频:每段 [i:a](或 anullsrc 静音占位)按 clip 配置 atrim/asetpts/atempo/volume/aresample
→ concat=n:N:v=1:a=1 → 与 extra_audio(TTS/配音素材库)、BGM 一起 amix → atrim 精确截断。
"""
if not resolved_clips:
raise ValueError("build_direct_render: no resolved clips")
@@ -148,6 +157,18 @@ def build_direct_render(
fc: list[str] = []
n = len(resolved_clips)
# 规范化每段参数
if clip_has_audio is None:
clip_has_audio = [True] * n
else:
clip_has_audio = list(clip_has_audio) + [True] * max(0, n - len(clip_has_audio))
clip_has_audio = clip_has_audio[:n]
if clip_volumes is None:
clip_volumes = [1.0] * n
else:
clip_volumes = list(clip_volumes) + [1.0] * max(0, n - len(clip_volumes))
clip_volumes = clip_volumes[:n]
clip_starts: list[float] = []
clip_effs: list[float] = []
clip_speeds: list[float] = []
@@ -161,6 +182,7 @@ def build_direct_render(
clip_effs.append(eff)
clip_speeds.append(speed)
# 1. 视频输入(原始素材签名 URL)
for i, clip in enumerate(resolved_clips):
sk = (getattr(clip, "config", None) or {}).get("_storage_key")
if not sk:
@@ -169,6 +191,7 @@ def build_direct_render(
inputs[fname] = sign_asset_url(sk)
input_args.extend(["-i", fname])
# 2. 视频段预处理
pre_labels: list[str] = []
for i in range(n):
vf: list[str] = []
@@ -189,10 +212,28 @@ def build_direct_render(
fc.append(f"[{i}:v]{','.join(vf)}[{label}]")
pre_labels.append(label)
# 2b. 音频段预处理(无声源用 anullsrc 占位;volume=0 的段也用 anullsrc 静音占位保持时间轴)
anullsrc_counter = 0
audio_pre_labels: list[str] = []
for i in range(n):
af: list[str] = []
start, eff, speed = clip_starts[i], clip_effs[i], clip_speeds[i]
vol = float(clip_volumes[i] if i < len(clip_volumes) else 1.0)
has_a = bool(clip_has_audio[i] if i < len(clip_has_audio) else True)
if not has_a or vol <= 0.001:
# 静音占位:用 anullsrc 生成静音,atrim 到段时长
sl = f"sil{anullsrc_counter}"
anullsrc_counter += 1
af: list[str] = ["anullsrc=channel_layout=stereo:sample_rate=44100"]
if eff > 0:
af.append(f"atrim=duration={eff:.3f}")
af.append("asetpts=PTS-STARTPTS")
af.append("aformat=sample_fmts=fltp:channel_layouts=stereo")
fc.append(f"{','.join(af)}[{sl}]")
# anullsrc 作为 filter 源不需要 -i 输入,直接给 label
audio_pre_labels.append(sl)
continue
af = []
if eff > 0:
if start > 0:
af.append(f"atrim=start={start:.3f}:duration={eff:.3f}")
@@ -203,17 +244,21 @@ def build_direct_render(
atempo = _build_atempo_chain(speed)
if atempo:
af.append(atempo)
if abs(vol - 1.0) >= 1e-3:
af.append(f"volume={vol:.3f}")
af.append("aresample=44100")
af.append("aformat=sample_fmts=fltp:channel_layouts=stereo")
alabel = f"ac{i}"
fc.append(f"[{i}:a]{','.join(af)}[{alabel}]")
audio_pre_labels.append(alabel)
# 3. concat(全硬切;v=1:a=1,视频音频一起拼接)
concat_in = "".join(f"[{v}][{a}]" for v, a in zip(pre_labels, audio_pre_labels, strict=True))
fc.append(f"{concat_in}concat=n={n}:v=1:a=1[vcat][acat]")
cur_v = "vcat"
cur_a = "acat"
# 4. 随机边缘裁剪降重(四边独立随机 2%~5%,与 ffmpeg_utils.random_edge_crop 一致)
if edge_crop_pct and edge_crop_pct > 0:
_r = random.Random()
p_min = EDGE_CROP_MIN_PCT
@@ -232,6 +277,7 @@ def build_direct_render(
)
cur_v = "vcrop"
# 5. drawtext 字幕
draw_filters: list[str] = []
if title_text.strip():
title_size = max(int(output_height * 0.05), 24)
@@ -274,9 +320,35 @@ def build_direct_render(
fc.append(f"[{cur_v}]format=yuv420p[vfinal]")
vfinal_label = "vfinal"
# 6. 音频混音:原素材主音轨 acat + extra(TTS/配音素材库) + BGM → amix → atrim
mix_labels: list[str] = [cur_a]
mix_vols: list[float] = [1.0]
next_idx = n
# 额外独立音频轨(TTS concat / 配音素材库整段音频)
for _ea_idx, (ea_path, ea_vol) in enumerate(extra_audio_tracks or []):
if ea_path is None:
continue
ea_p = Path(ea_path)
if not ea_p.exists():
continue
eurl, ekey = upload_local_audio_and_sign(ea_p)
ename = f"extra{_ea_idx}{ea_p.suffix or '.mp3'}"
inputs[ename] = eurl
oss_keys.append(ekey)
input_args.extend(["-i", ename])
elabel = f"aex{_ea_idx}"
fc.append(
f"[{next_idx}:a]aresample=44100,volume={float(ea_vol):.2f},"
f"aformat=sample_fmts=fltp:channel_layouts=stereo[{elabel}]"
)
mix_labels.append(elabel)
mix_vols.append(float(ea_vol))
next_idx += 1
if tts_audio and Path(tts_audio).exists():
# 旧参数保留:若调用方直接传了 tts_audio 而没走 extra_audio_tracks,则仍然加入
# (兼容旧调用,正常路径 TTS 已经通过 extra_audio_tracks 传入)
turl, tkey = upload_local_audio_and_sign(Path(tts_audio))
tname = "tts" + (Path(tts_audio).suffix or ".mp3")
inputs[tname] = turl
@@ -287,6 +359,7 @@ def build_direct_render(
f"[{next_idx}:a]aresample=44100,volume=1.00,aformat=sample_fmts=fltp:channel_layouts=stereo[{alabel}]"
)
mix_labels.append(alabel)
mix_vols.append(1.0)
next_idx += 1
if bgm_audio and Path(bgm_audio).exists():
burl, bkey = upload_local_audio_and_sign(Path(bgm_audio))
@@ -299,6 +372,7 @@ def build_direct_render(
f"[{next_idx}:a]aresample=44100,volume=0.35,aformat=sample_fmts=fltp:channel_layouts=stereo[{alabel}]"
)
mix_labels.append(alabel)
mix_vols.append(0.35)
next_idx += 1
maps: list[str] = ["-map", f"[{vfinal_label}]"]
@@ -309,6 +383,7 @@ def build_direct_render(
f"amix=inputs={n_mix}:duration=longest:dropout_transition=2:normalize=0",
"aresample=44100",
]
# Bug2 修复:atrim 到视频精确时长
if total_duration and total_duration > 0:
mix_parts.append(f"atrim=0:{total_duration:.3f}")
mix_parts.append("asetpts=PTS-STARTPTS")
@@ -317,6 +392,7 @@ def build_direct_render(
else:
logger.info("[gpu-direct] no audio tracks; output silent video")
# 7. 组装 ffmpeg_args + NVENC 编码
ffmpeg_args = ["-y", *input_args, "-filter_complex", ";".join(fc), *maps]
ffmpeg_args.extend(["-c:v", vcodec, "-preset", preset, "-pix_fmt", "yuv420p"])
if video_bitrate:
@@ -2279,13 +2279,34 @@ class UnifiedRenderService:
video_layer = next(_lyr for _lyr in layers if _lyr.role not in ("audio",))
video_clips = [c for c in video_layer.clips if c.clip_type != "audio"]
# TTS:把 audio 层的 TTS 片段合并成一个文件给 P4000
tts_merged: Path | None = None
# 音频层处理:收集 TTS 分段与配音素材库整段音频
# - TTS 分段(带 tts 标记)→ 无间隙 concat 成单文件
# - 配音素材库(voice_library=True)→ 单独作为整段音轨(不走分段 concat,已从 0 覆盖整段)
audio_layer = next((_lyr for _lyr in layers if _lyr.role == "audio"), None)
tts_merged: Path | None = None
voiceover_track: Path | None = None
if audio_layer:
tts_clips = [c for c in audio_layer.clips if (c.config or {}).get("tts") and c.local_path.exists()]
if tts_clips:
tts_merged = self._concat_audio_clips(tts_clips, tag="tts_direct")
# 配音素材库整段音频(按 _maybe_add_voice_library_layer 约定只有一个 clip_id=voice_library_main)
vo_clips = [
c for c in audio_layer.clips if (c.config or {}).get("voice_library") and c.local_path.exists()
]
if vo_clips:
voiceover_track = vo_clips[-1].local_path # 理论上只有一个,取最后一个
logger.info(
"[gpu-direct] 配音素材库音轨: plan_id=%s path=%s",
self.plan.id,
voiceover_track,
)
# 额外独立音轨(TTS concat、配音素材库)→ gpu_direct_pipeline 会与主音轨/BGM 一起 amix
extra_audio_tracks: list[tuple[Path, float]] = []
if tts_merged:
extra_audio_tracks.append((tts_merged, 1.0))
if voiceover_track:
extra_audio_tracks.append((voiceover_track, 1.0))
# BGM 本地文件
bgm_path = Path(self.bgm_path) if self.bgm_path else None
@@ -2304,21 +2325,39 @@ class UnifiedRenderService:
if sub_cfg.get("auto_generated") and self._asr_timeline_cache is not None:
subtitle_segments = list(self._asr_timeline_cache.segments)
# 边缘裁剪比例(与 random_edge_crop 默认 2~5% 同口径,取固定 3%)
# 边缘裁剪:dedup 开启时在 GPU 内做四边随机 2~5% 裁剪(gpu_direct_pipeline 内部随机)
dedup = self._dedup_enabled()
edge_pct = 0.03 if dedup else 0.0
edge_pct = 0.03 if dedup else 0.0 # >0 表示启用;实际区间 [2%,5%] 在 pipeline 内随机
# 探测每个视频素材是否含音轨、读取 volume 配置
clip_has_audio_list: list[bool] = []
clip_volumes_list: list[float] = []
for c in video_clips:
lp = getattr(c, "local_path", None)
_ha = False
if lp and Path(lp).exists():
try:
_ha = probe_has_audio(str(lp))
except Exception as _pe: # noqa: BLE001
logger.warning("[gpu-direct] probe_has_audio 失败按有声处理: %s", _pe)
_ha = True
clip_has_audio_list.append(_ha)
_vol = float((c.config or {}).get("volume", 1.0))
clip_volumes_list.append(_vol if _vol > 0 else 0.0)
plan = gdp.build_direct_render(
resolved_clips=video_clips,
output_width=self.output_width,
output_height=self.output_height,
output_fps=self.output_fps,
tts_audio=tts_merged,
bgm_audio=bgm_path,
title_text=title_text,
subtitle_segments=subtitle_segments,
edge_crop_pct=edge_pct,
total_duration=video_duration,
clip_has_audio=clip_has_audio_list,
clip_volumes=clip_volumes_list,
extra_audio_tracks=extra_audio_tracks,
)
client = get_gpu_encoder()