fix(worker): render_plan 保留视频原声并尊重每clip音量 (#1474)
CI/CD Pipeline / Check if frontend-only change (pull_request) Successful in 1m12s
CI/CD Pipeline / Build Staging Worker Image (push) Successful in 1m40s
CI/CD Pipeline / Validate - Migration (alembic) (pull_request) Successful in 3m4s
CI/CD Pipeline / Validate - Migration (alembic) (push) Successful in 3m8s
CI/CD Pipeline / Validate - Type Check (mypy) (push) Successful in 3m9s
CI/CD Pipeline / Validate - Type Check (mypy) (pull_request) Successful in 3m10s
CI/CD Pipeline / Build Staging API Image (push) Successful in 3m28s
AI Code Review / AI Code Review (pull_request) Failing after 2m17s
CI/CD Pipeline / PR Build API Image (pull_request) Successful in 35s
CI/CD Pipeline / Build Staging Web Image (push) Successful in 4m34s
CI/CD Pipeline / PR Build Worker Image (pull_request) Successful in 38s
Preview Deploy / Deploy Preview Environment (pull_request) Successful in 2m31s
PR Automation / Auto Approve on CI Green (pull_request) Successful in 4m3s
CI/CD Pipeline / Deploy Staging (Watchtower auto-deploy) (push) Successful in 1m11s
CI/CD Pipeline / Validate - Code Quality (push) Successful in 6m47s
CI/CD Pipeline / ACR Image Cleanup (push) Successful in 1m15s
CI/CD Pipeline / Validate - Code Quality (pull_request) Successful in 7m47s
CI/CD Pipeline / Frontend Unit Tests (push) Successful in 6m38s
CI/CD Pipeline / Staging E2E Tests (push) Successful in 2m41s
CI/CD Pipeline / Staging API Integration Tests (push) Successful in 3m46s
CI/CD Pipeline / Integration Tests (push) Successful in 3m4s
CI/CD Pipeline / Integration Tests (pull_request) Successful in 3m12s
CI/CD Pipeline / Unit Tests (push) Failing after 13m33s
CI/CD Pipeline / Unit Tests (pull_request) Failing after 11m30s
CI/CD Pipeline / CI Gate (pull_request) Failing after 10s
CI/CD Pipeline / Production Browser E2E (pull_request) Failing after 565h55m23s
CI/CD Pipeline / Canary Release to Production (pull_request) Failing after 565h55m24s
CI/CD Pipeline / Build Production Web Image (pull_request) Failing after 565h55m26s
CI/CD Pipeline / Build Production API Image (pull_request) Failing after 565h55m26s
CI/CD Pipeline / Deploy Production (pull_request) Failing after 565h55m24s
CI/CD Pipeline / Canary Release to Production (push) Failing after 565h55m58s
CI/CD Pipeline / Production Browser E2E (push) Failing after 565h55m58s
CI/CD Pipeline / Deploy Production (push) Failing after 565h55m59s
CI/CD Pipeline / Build Production Web Image (push) Failing after 565h56m0s
CI/CD Pipeline / Build Production API Image (push) Failing after 565h56m1s
CI/CD Pipeline / ACR Image Cleanup (pull_request) Failing after 566h5m57s
CI/CD Pipeline / CI Gate (push) Failing after 565h56m2s
CI/CD Pipeline / Staging E2E Tests (pull_request) Failing after 566h6m1s
CI/CD Pipeline / Deploy Staging (Watchtower auto-deploy) (pull_request) Failing after 566h6m3s
CI/CD Pipeline / Frontend Unit Tests (pull_request) Failing after 566h6m56s
CI/CD Pipeline / Frontend Lint (pull_request) Failing after 566h6m57s
PR Automation / Auto Merge on CI Green + Approved (pull_request) Failing after 566h7m1s
CI/CD Pipeline / PR Build Worker Image (push) Failing after 566h8m28s
CI/CD Pipeline / PR Build Web Image (push) Failing after 566h8m30s
CI/CD Pipeline / Build Staging Worker Image (pull_request) Failing after 566h8m34s
CI/CD Pipeline / PR Build API Image (push) Failing after 566h8m32s
CI/CD Pipeline / Frontend Lint (push) Failing after 566h8m54s
CI/CD Pipeline / Build Staging Web Image (pull_request) Failing after 566h8m36s
CI/CD Pipeline / Check if frontend-only change (push) Failing after 566h10m7s
CI/CD Pipeline / Build Staging API Image (pull_request) Failing after 566h8m38s
CI/CD Pipeline / Build Production Worker Image (pull_request) Failing after 566h29m17s
CI/CD Pipeline / Build Production Worker Image (push) Failing after 566h29m51s
CI/CD Pipeline / Staging API Integration Tests (pull_request) Failing after 566h39m50s
CI/CD Pipeline / PR Build Web Image (pull_request) Failing after 566h40m0s

Co-authored-by: xiaoxia <dev@xiaoxiajianji.com>
Co-committed-by: xiaoxia <dev@xiaoxiajianji.com>
This commit was merged in pull request #1474.
This commit is contained in:
2026-08-24 01:16:00 +08:00
committed by auto-approve-bot
parent 024aca3557
commit 61770cd4e4
4 changed files with 443 additions and 118 deletions
+94 -32
View File
@@ -67,6 +67,16 @@ def clip_has_audio(ctx: RenderContext, clip: ResolvedClip) -> bool:
return ctx._audio_cache[key]
def _clip_volume(clip: ResolvedClip) -> float:
"""读取 clip 的音量配置(0.0~1.0,>1 放大)。缺省 1.0 原声。"""
cfg = getattr(clip, "config", None) or {}
try:
vol = float(cfg.get("volume", 1.0))
except (TypeError, ValueError):
return 1.0
return max(0.0, vol)
# ── 音频混音 ──────────────────────────────────────────────────────────────────
@@ -82,12 +92,13 @@ def mix_audio(
"""音频后处理混音.
处理逻辑:
1. 丢弃主图层(main/broll/overlay/corner_voice)的原始音频,避免录入源视频杂音
2. 仅使用独立音频轨(audio role,TTS/配音)作为主音频
3. 如果提供了 bgm_path,则额外混入 BGM(支持淡入淡出、循环、人声闪避)
4. 如果配置了 audio_tracks,则混入多轨道音频(配音、音效等)
5. 输出时长截断到 video_duration
6. 如果配置了降噪,最后应用降噪
1. 保留主图层(main/broll/overlay/corner_voice)视频素材的原声,按顺序 concat 拼接
2. 每个 clip 按 config.volume 应用音量(volume=0 静音,=1 原声)
3. 独立音频轨(audio role,TTS/配音)通过 amix 混入
4. 如果提供了 bgm_path,则额外混入 BGM(支持淡入淡出、循环、人声闪避)
5. 如果配置了 audio_tracks,则混入多轨道音频(配音、音效等)
6. 输出时长截断到 video_duration
7. 如果配置了降噪,最后应用降噪
Args:
ctx: 渲染上下文
@@ -123,10 +134,12 @@ def mix_audio(
if "audio" in layer_map:
audio_clips = layer_map["audio"].clips
# ── 丢弃源视频的原始音频(避免录入杂音),成片仅保留 TTS 配音 + BGM ──
main_clips = []
# ── 保留源视频原声:过滤掉无音频流的 main clip(图片/无声素材) ──
# 注意:volume=0 的 clip 不能移除——移除会导致后续 clip 音频时间轴前移、音画不同步。
# volume=0 通过滤镜链生成静音流,保持时间轴对齐。
main_clips = [c for c in main_clips if clip_has_audio(ctx, c)]
# ── 防御:过滤掉无音频流的 clip ──
# ── 防御:过滤掉无音频流的独立音频轨 ──
audio_clips = [c for c in audio_clips if clip_has_audio(ctx, c)]
if not main_clips and not audio_clips:
@@ -144,7 +157,7 @@ def mix_audio(
# 构建音频处理命令
output_path = ctx.work_dir / f"audio_{ctx.plan_id}.aac"
# 源视频原始音频已被丢弃(main_clips = []),最终音频完全由独立音频轨 + BGM + 多轨配置组成。
# 主音频为视频素材原声 concat;独立音频轨(TTS/配音)通过 amix 混入。
# 当无 main_clips 时,将独立音频轨作为主音频走 concat 拼接;当二者均有则走 amix 混音。
if main_clips:
effective_main = main_clips
@@ -266,28 +279,67 @@ def concat_main_audio(
has_speed = abs(speed - 1.0) >= 1e-6
if not has_speed and not has_reverse:
# 无调速无倒放:简单命令行,-ss 裁剪更高效
command = [
FFMPEG_BIN,
"-y",
"-i",
str(clip.local_path),
"-vn",
"-acodec",
"aac",
"-b:a",
"128k",
"-ar",
"48000",
"-ac",
"2",
]
if trim_start > 0:
command.extend(["-ss", f"{trim_start:.3f}"])
if final_duration > 0:
command.extend(["-t", f"{final_duration:.3f}"])
command.append(str(output_path))
run_ffmpeg(command)
# 无调速无倒放:根据是否需要裁剪/音量选择最高效的路径。
vol = _clip_volume(clip)
need_trim = trim_start > 0 or (effective_duration > 0 and final_duration < adjusted_duration)
need_volume = abs(vol - 1.0) >= 1e-6
if need_trim:
# 需要裁剪:用 atrim 滤镜在滤镜链中精确裁剪(采样点级精度,不浪费解码)。
# 滤镜顺序:atrim → asetpts → volume(先裁剪再调音量,避免处理被丢弃的数据)。
af_parts: list[str] = []
if trim_start > 0 and effective_duration > 0:
af_parts.append(f"atrim=start={trim_start:.3f}:duration={final_duration:.3f}")
elif trim_start > 0:
af_parts.append(f"atrim=start={trim_start:.3f}")
elif final_duration > 0:
af_parts.append(f"atrim=duration={final_duration:.3f}")
af_parts.append("asetpts=PTS-STARTPTS")
if need_volume:
af_parts.append(f"volume={vol:.4f}")
command = [
FFMPEG_BIN,
"-y",
"-i",
str(clip.local_path),
"-vn",
"-af",
",".join(af_parts),
"-acodec",
"aac",
"-b:a",
"128k",
"-ar",
"48000",
"-ac",
"2",
]
# atrim 已精确控制时长,无需额外 -t
command.append(str(output_path))
run_ffmpeg(command)
else:
# 无需裁剪:直接提取,最高效。音量用单个 -af(如有)。
command = [
FFMPEG_BIN,
"-y",
"-i",
str(clip.local_path),
"-vn",
"-acodec",
"aac",
"-b:a",
"128k",
"-ar",
"48000",
"-ac",
"2",
]
if need_volume:
command.extend(["-af", f"volume={vol:.4f}"])
if final_duration > 0:
command.extend(["-t", f"{final_duration:.3f}"])
command.append(str(output_path))
run_ffmpeg(command)
else:
# 有调速或倒放:用 filter_complex
speed_engine = SpeedEngine()
@@ -312,6 +364,11 @@ def concat_main_audio(
if reverse_filter:
audio_filters.append(reverse_filter)
# 音量
vol = _clip_volume(clip)
if abs(vol - 1.0) >= 1e-6:
audio_filters.append(f"volume={vol:.4f}")
# aformat 归一化:统一输出格式为 48000Hz + stereo + fltp
audio_filters.append("aformat=sample_rates=48000:channel_layouts=stereo:sample_fmts=fltp")
@@ -378,6 +435,11 @@ def concat_main_audio(
if reverse_filter:
audio_filters.append(reverse_filter)
# 音量(0=静音,1=原声)
vol = _clip_volume(clip)
if abs(vol - 1.0) >= 1e-6:
audio_filters.append(f"volume={vol:.4f}")
# aformat 归一化:统一采样率48000Hz + 双声道stereo + fltp采样格式
# concat filter 要求所有输入音频参数完全一致,否则 exit=234 失败
audio_filters.append("aformat=sample_rates=48000:channel_layouts=stereo:sample_fmts=fltp")