feat: BGM音轨混音能力(音量/淡入淡出/人声闪避/预设BGM库) (#291)
CI/CD Pipeline / Integration Tests (push) Has been cancelled
CI/CD Pipeline / Build & Push Staging (Watchtower auto-deploy) (push) Has been cancelled
CI/CD Pipeline / Staging E2E Tests (push) Has been cancelled
CI/CD Pipeline / Staging API Integration Tests (push) Has been cancelled
CI/CD Pipeline / Production Browser E2E (push) Failing after 1540h49m57s
CI/CD Pipeline / Build Production Runtime Images (push) Failing after 1540h50m2s
CI/CD Pipeline / Validate Code Quality And Tests (push) Has been skipped
CI/CD Pipeline / Unit Tests (push) Has been skipped
CI/CD Pipeline / Deploy Production (push) Failing after 1541h21m36s
CI/CD Pipeline / Frontend Lint (push) Has been skipped
CI/CD Pipeline / Integration Tests (push) Has been cancelled
CI/CD Pipeline / Build & Push Staging (Watchtower auto-deploy) (push) Has been cancelled
CI/CD Pipeline / Staging E2E Tests (push) Has been cancelled
CI/CD Pipeline / Staging API Integration Tests (push) Has been cancelled
CI/CD Pipeline / Production Browser E2E (push) Failing after 1540h49m57s
CI/CD Pipeline / Build Production Runtime Images (push) Failing after 1540h50m2s
CI/CD Pipeline / Validate Code Quality And Tests (push) Has been skipped
CI/CD Pipeline / Unit Tests (push) Has been skipped
CI/CD Pipeline / Deploy Production (push) Failing after 1541h21m36s
CI/CD Pipeline / Frontend Lint (push) Has been skipped
This commit was merged in pull request #291.
This commit is contained in:
Executable
+313
@@ -0,0 +1,313 @@
|
||||
"""BGM 混音模块 — 背景音乐与主音频混合.
|
||||
|
||||
基于 FFmpeg 实现:
|
||||
- BGM 音量调节
|
||||
- 淡入淡出(afade)
|
||||
- 循环播放(aloop,短 BGM 铺长视频)
|
||||
- 人声闪避(sidechaincompress,有人声时BGM自动降低音量)
|
||||
- amix 混音
|
||||
|
||||
作为 render_audio.py 的增强模块,在 mix_audio 后处理阶段被调用。
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
from dataclasses import dataclass
|
||||
from pathlib import Path
|
||||
from typing import TYPE_CHECKING
|
||||
|
||||
from video_processing.ffmpeg_utils import FFMPEG_BIN, probe_duration, run_ffmpeg
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from video_processing.render_audio import RenderContext
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
@dataclass
|
||||
class BGMConfig:
|
||||
"""BGM 混音配置(内部使用,从 plan.config.bgm 转换而来)"""
|
||||
|
||||
bgm_path: str # BGM 本地文件路径
|
||||
volume: float = 0.3 # 0.0 ~ 1.0
|
||||
fade_in: float = 0.0 # 淡入时长(秒)
|
||||
fade_out: float = 0.0 # 淡出时长(秒)
|
||||
loop_enabled: bool = True # 是否循环铺满
|
||||
sidechain_enabled: bool = False # 人声闪避
|
||||
sidechain_ratio: float = 0.3 # 闪避时音量降低比例
|
||||
sidechain_attack: float = 0.02 # 攻击时间
|
||||
sidechain_release: float = 0.5 # 释放时间
|
||||
sidechain_threshold: float = -25.0 # 触发阈值(dB)
|
||||
|
||||
@classmethod
|
||||
def from_config_dict(cls, bgm_path: str, config: dict) -> "BGMConfig":
|
||||
"""从 plan.config.bgm 字典创建 BGMConfig。"""
|
||||
return cls(
|
||||
bgm_path=bgm_path,
|
||||
volume=float(config.get("volume", 0.3)),
|
||||
fade_in=float(config.get("fade_in", 0.0)),
|
||||
fade_out=float(config.get("fade_out", 0.0)),
|
||||
loop_enabled=bool(config.get("loop_enabled", True)),
|
||||
sidechain_enabled=bool(config.get("sidechain_enabled", False)),
|
||||
sidechain_ratio=float(config.get("sidechain_ratio", 0.3)),
|
||||
sidechain_attack=float(config.get("sidechain_attack", 0.02)),
|
||||
sidechain_release=float(config.get("sidechain_release", 0.5)),
|
||||
sidechain_threshold=float(config.get("sidechain_threshold", -25.0)),
|
||||
)
|
||||
|
||||
|
||||
# ── BGM 预处理 ────────────────────────────────────────────────────────────────
|
||||
|
||||
|
||||
def prepare_bgm_track(
|
||||
ctx: "RenderContext",
|
||||
bgm: BGMConfig,
|
||||
target_duration: float,
|
||||
) -> Path:
|
||||
"""预处理 BGM 轨道:循环/截断 + 音量 + 淡入淡出.
|
||||
|
||||
生成一个时长精确等于 target_duration 的 BGM 音频文件。
|
||||
后续再与主音频混音。
|
||||
|
||||
Args:
|
||||
ctx: 渲染上下文
|
||||
bgm: BGM 配置
|
||||
target_duration: 目标时长(秒),通常等于视频总时长
|
||||
|
||||
Returns:
|
||||
处理后的 BGM 音频文件路径
|
||||
"""
|
||||
output_path = ctx.work_dir / f"bgm_processed_{ctx.plan_id}.aac"
|
||||
|
||||
if target_duration <= 0:
|
||||
target_duration = 5.0 # 兜底
|
||||
|
||||
bgm_dur = probe_duration(bgm.bgm_path)
|
||||
needs_loop = bgm.loop_enabled and bgm_dur > 0 and bgm_dur < target_duration * 0.9
|
||||
|
||||
# 构建滤镜链
|
||||
filter_parts: list[str] = []
|
||||
input_looped: bool = False
|
||||
|
||||
if needs_loop:
|
||||
# 计算需要循环多少次才能铺满
|
||||
loop_count = max(1, int(target_duration / bgm_dur) + 2)
|
||||
# aloop 滤镜:循环指定次数
|
||||
filter_parts.append(f"aloop=loop={loop_count}:size=0")
|
||||
input_looped = True
|
||||
|
||||
# 音量调节
|
||||
volume = max(0.0, min(1.0, bgm.volume))
|
||||
if abs(volume - 1.0) > 0.001:
|
||||
filter_parts.append(f"volume={volume:.3f}")
|
||||
|
||||
# 淡入
|
||||
if bgm.fade_in > 0:
|
||||
filter_parts.append(f"afade=t=in:st=0:d={bgm.fade_in:.3f}")
|
||||
|
||||
# 淡出(从 target_duration - fade_out 开始)
|
||||
if bgm.fade_out > 0 and target_duration > bgm.fade_out:
|
||||
fade_start = target_duration - bgm.fade_out
|
||||
filter_parts.append(f"afade=t=out:st={fade_start:.3f}:d={bgm.fade_out:.3f}")
|
||||
|
||||
# 最终截断到目标时长
|
||||
filter_parts.append(f"atrim=0:{target_duration:.3f}")
|
||||
filter_parts.append("asetpts=N/SR/TB") # 重置时间戳
|
||||
|
||||
filter_str = ",".join(filter_parts)
|
||||
|
||||
command = [
|
||||
FFMPEG_BIN,
|
||||
"-y",
|
||||
"-i",
|
||||
bgm.bgm_path,
|
||||
"-filter:a",
|
||||
filter_str,
|
||||
"-c:a",
|
||||
"aac",
|
||||
"-b:a",
|
||||
"128k",
|
||||
str(output_path),
|
||||
]
|
||||
|
||||
logger.info(
|
||||
"[bgm] prepare BGM track: path=%s dur=%.2f target=%.2f loop=%s fade_in=%.2f fade_out=%.2f",
|
||||
bgm.bgm_path[-40:],
|
||||
bgm_dur,
|
||||
target_duration,
|
||||
needs_loop,
|
||||
bgm.fade_in,
|
||||
bgm.fade_out,
|
||||
)
|
||||
|
||||
run_ffmpeg(command)
|
||||
return output_path
|
||||
|
||||
|
||||
# ── BGM + 主音频混音 ──────────────────────────────────────────────────────────
|
||||
|
||||
|
||||
def mix_bgm_with_main(
|
||||
ctx: "RenderContext",
|
||||
main_audio_path: Path,
|
||||
bgm: BGMConfig,
|
||||
target_duration: float,
|
||||
) -> Path:
|
||||
"""将 BGM 与主音频混合.
|
||||
|
||||
两种模式:
|
||||
1. 普通混音(sidechain 关闭):amix 两路音频
|
||||
2. 人声闪避(sidechain 开启):用 sidechaincompress 让 BGM 跟随主音频音量自动调整
|
||||
|
||||
Args:
|
||||
ctx: 渲染上下文
|
||||
main_audio_path: 主音频文件路径(人声/原始音频)
|
||||
bgm: BGM 配置
|
||||
target_duration: 目标时长
|
||||
|
||||
Returns:
|
||||
混音后的音频文件路径
|
||||
"""
|
||||
output_path = ctx.work_dir / f"audio_with_bgm_{ctx.plan_id}.aac"
|
||||
|
||||
# 先预处理 BGM 轨道(循环/音量/淡入淡出/截断)
|
||||
bgm_processed = prepare_bgm_track(ctx, bgm, target_duration)
|
||||
|
||||
if not bgm.sidechain_enabled:
|
||||
# 普通 amix 混音
|
||||
_mix_simple(main_audio_path, bgm_processed, output_path)
|
||||
else:
|
||||
# sidechain 人声闪避混音
|
||||
_mix_sidechain(main_audio_path, bgm_processed, output_path, bgm)
|
||||
|
||||
return output_path
|
||||
|
||||
|
||||
def _mix_simple(main_path: Path, bgm_path: Path, output_path: Path) -> None:
|
||||
"""简单 amix 混音:主音频 + BGM = 输出.
|
||||
|
||||
主音频权重 1.0,BGM 已经在预处理阶段调好了音量。
|
||||
amix 会自动归一化,需要用 volume 补偿。
|
||||
"""
|
||||
# 使用 amix:inputs=2,duration=first(以主音频时长为准)
|
||||
# 然后用 volume=2 补偿 amix 的衰减(2路输入每路平均乘0.5)
|
||||
filter_complex = "[0:a][1:a]amix=inputs=2:duration=first:dropout_transition=0[outa];" "[outa]volume=2[final]"
|
||||
|
||||
command = [
|
||||
FFMPEG_BIN,
|
||||
"-y",
|
||||
"-i",
|
||||
str(main_path),
|
||||
"-i",
|
||||
str(bgm_path),
|
||||
"-filter_complex",
|
||||
filter_complex,
|
||||
"-map",
|
||||
"[final]",
|
||||
"-c:a",
|
||||
"aac",
|
||||
"-b:a",
|
||||
"128k",
|
||||
str(output_path),
|
||||
]
|
||||
|
||||
logger.info("[bgm] simple amix mix")
|
||||
run_ffmpeg(command)
|
||||
|
||||
|
||||
def _mix_sidechain(
|
||||
main_path: Path,
|
||||
bgm_path: Path,
|
||||
output_path: Path,
|
||||
bgm: BGMConfig,
|
||||
) -> None:
|
||||
"""sidechain 人声闪避混音.
|
||||
|
||||
原理:
|
||||
- 主音频作为 sidechain 信号源
|
||||
- BGM 轨道经过 sidechaincompress,根据主音频音量动态调整 BGM 音量
|
||||
- 最后 amix 混音
|
||||
|
||||
FFmpeg sidechaincompress 参数:
|
||||
- threshold: 触发阈值(dB),主音频超过此值时开始压缩
|
||||
- ratio: 压缩比,越高压缩越狠
|
||||
- attack: 攻击时间(秒)
|
||||
- release: 释放时间(秒)
|
||||
"""
|
||||
# sidechain_ratio 表示闪避时 BGM 音量降低比例
|
||||
# ratio = 1 / (1 - sidechain_ratio),但实际压缩比需要更精细调整
|
||||
# 简化处理:把 ratio 映射到 2:1 ~ 10:1 范围
|
||||
ratio = max(2.0, min(10.0, 1.0 / (1.0 - bgm.sidechain_ratio)))
|
||||
|
||||
filter_complex = (
|
||||
# BGM 经过 sidechain 压缩,用主音频做触发
|
||||
f"[1:a][0:a]sidechaincompress="
|
||||
f"threshold={bgm.sidechain_threshold}dB:"
|
||||
f"ratio={ratio:.1f}:"
|
||||
f"attack={bgm.sidechain_attack:.3f}:"
|
||||
f"release={bgm.sidechain_release:.3f}:"
|
||||
f"knee=6[bgm_comp];"
|
||||
# 主音频 + 压缩后的 BGM 混音
|
||||
f"[0:a][bgm_comp]amix=inputs=2:duration=first:dropout_transition=0[outa];"
|
||||
f"[outa]volume=1.5[final]" # 轻微补偿
|
||||
)
|
||||
|
||||
command = [
|
||||
FFMPEG_BIN,
|
||||
"-y",
|
||||
"-i",
|
||||
str(main_path),
|
||||
"-i",
|
||||
str(bgm_path),
|
||||
"-filter_complex",
|
||||
filter_complex,
|
||||
"-map",
|
||||
"[final]",
|
||||
"-c:a",
|
||||
"aac",
|
||||
"-b:a",
|
||||
"128k",
|
||||
str(output_path),
|
||||
]
|
||||
|
||||
logger.info(
|
||||
"[bgm] sidechain mix: threshold=%.1fdB ratio=%.1f attack=%.3f release=%.3f",
|
||||
bgm.sidechain_threshold,
|
||||
ratio,
|
||||
bgm.sidechain_attack,
|
||||
bgm.sidechain_release,
|
||||
)
|
||||
run_ffmpeg(command)
|
||||
|
||||
|
||||
# ── 纯 BGM 模式(无主音频) ──────────────────────────────────────────────────
|
||||
|
||||
|
||||
def build_bgm_only(
|
||||
ctx: "RenderContext",
|
||||
bgm: BGMConfig,
|
||||
target_duration: float,
|
||||
) -> Path:
|
||||
"""只有 BGM、没有主音频时,直接生成 BGM 音频.
|
||||
|
||||
Args:
|
||||
ctx: 渲染上下文
|
||||
bgm: BGM 配置
|
||||
target_duration: 目标时长
|
||||
|
||||
Returns:
|
||||
BGM 音频文件路径
|
||||
"""
|
||||
output_path = ctx.work_dir / f"bgm_only_{ctx.plan_id}.aac"
|
||||
|
||||
if target_duration <= 0:
|
||||
target_duration = 5.0
|
||||
|
||||
bgm_processed = prepare_bgm_track(ctx, bgm, target_duration)
|
||||
|
||||
# 直接复制
|
||||
import shutil
|
||||
|
||||
shutil.copy2(bgm_processed, output_path)
|
||||
return output_path
|
||||
Regular → Executable
+33
-3
@@ -70,6 +70,9 @@ def mix_audio(
|
||||
ctx: RenderContext,
|
||||
layers: list[RenderLayer],
|
||||
video_duration: float,
|
||||
*,
|
||||
bgm_path: str | None = None,
|
||||
bgm_config: dict | None = None,
|
||||
) -> Path | None:
|
||||
"""音频后处理混音.
|
||||
|
||||
@@ -79,11 +82,14 @@ def mix_audio(
|
||||
3. 独立音频轨(audio role)用 amix 混入
|
||||
4. 输出时长截断到 video_duration
|
||||
5. 无音频流的 clip 会被自动跳过,避免 FFmpeg 引用 [i:a] 失败
|
||||
6. 如果提供了 bgm_path,则额外混入 BGM(支持淡入淡出、循环、人声闪避)
|
||||
|
||||
Args:
|
||||
ctx: 渲染上下文
|
||||
layers: 图层列表
|
||||
video_duration: 视频总时长(用于截断音频)
|
||||
bgm_path: BGM 音频本地路径,为 None 时不混入 BGM
|
||||
bgm_config: BGM 配置字典(volume/fade_in/fade_out/sidechain 等)
|
||||
|
||||
Returns:
|
||||
混音后的音频文件路径,无音频时返回 None
|
||||
@@ -116,6 +122,15 @@ def mix_audio(
|
||||
audio_clips = [c for c in audio_clips if clip_has_audio(ctx, c)]
|
||||
|
||||
if not main_clips and not audio_clips:
|
||||
# 没有主音频也没有独立音频 → 检查是否有 BGM
|
||||
if bgm_path and bgm_config and bgm_config.get("enabled", False):
|
||||
from video_processing.bgm_mixer import BGMConfig, build_bgm_only
|
||||
|
||||
bgm_cfg = BGMConfig.from_config_dict(bgm_path, bgm_config)
|
||||
try:
|
||||
return build_bgm_only(ctx, bgm_cfg, video_duration)
|
||||
except Exception:
|
||||
logger.exception("[bgm] 纯BGM生成失败: plan_id=%s", ctx.plan_id)
|
||||
return None
|
||||
|
||||
# 构建音频处理命令
|
||||
@@ -124,10 +139,25 @@ def mix_audio(
|
||||
# 简单场景:只有主图层 + 无独立音频 → 直接从视频提取音频并拼接
|
||||
if main_clips and not audio_clips:
|
||||
concat_main_audio(ctx, main_clips, output_path, video_duration)
|
||||
return output_path
|
||||
else:
|
||||
# 有独立音频轨 → amix 混音
|
||||
mix_with_independent_audio(ctx, main_clips, audio_clips, output_path, video_duration)
|
||||
|
||||
# ── BGM 混音 ──
|
||||
if bgm_path and bgm_config and bgm_config.get("enabled", False):
|
||||
from video_processing.bgm_mixer import BGMConfig, mix_bgm_with_main
|
||||
|
||||
bgm_cfg = BGMConfig.from_config_dict(bgm_path, bgm_config)
|
||||
bgm_output = ctx.work_dir / f"audio_with_bgm_{ctx.plan_id}.aac"
|
||||
|
||||
try:
|
||||
# 这里 main_audio 就是 output_path,先有主音频再混 BGM
|
||||
final_path = mix_bgm_with_main(ctx, output_path, bgm_cfg, video_duration)
|
||||
return final_path
|
||||
except Exception:
|
||||
logger.exception("[bgm] BGM 混音失败,回退到无 BGM 音频: plan_id=%s", ctx.plan_id)
|
||||
return output_path
|
||||
|
||||
# 有独立音频轨 → amix 混音
|
||||
mix_with_independent_audio(ctx, main_clips, audio_clips, output_path, video_duration)
|
||||
return output_path
|
||||
|
||||
|
||||
|
||||
@@ -165,6 +165,7 @@ class UnifiedRenderService:
|
||||
output_fps: int = DEFAULT_FPS,
|
||||
transition_duration: float = DEFAULT_TRANSITION_DURATION,
|
||||
asr_service: Any = None, # ASRService 实例,用于自动生成字幕
|
||||
bgm_path: str | None = None, # BGM 本地文件路径
|
||||
):
|
||||
self.plan = plan
|
||||
self.clips = clips
|
||||
@@ -175,6 +176,7 @@ class UnifiedRenderService:
|
||||
self.output_fps = output_fps
|
||||
self.transition_duration = transition_duration
|
||||
self.asr_service = asr_service
|
||||
self.bgm_path = bgm_path
|
||||
|
||||
def render(self) -> RenderResult:
|
||||
"""执行渲染,返回 RenderResult.
|
||||
@@ -288,9 +290,55 @@ class UnifiedRenderService:
|
||||
if is_pass_through:
|
||||
# 直通场景已在一次调用中完成视频+音频
|
||||
has_audio = pass_through_has_audio
|
||||
# 直通模式下也支持 BGM 混音:提取音频 → 混 BGM → 合并回视频
|
||||
if self.bgm_path and pass_through_has_audio:
|
||||
config = self.plan.config or {}
|
||||
bgm_config = config.get("bgm", {}) or {}
|
||||
if bgm_config.get("enabled", False):
|
||||
ctx = RenderContext(work_dir=self.work_dir, plan_id=self.plan.id)
|
||||
from video_processing.bgm_mixer import BGMConfig, mix_bgm_with_main
|
||||
|
||||
bgm_cfg = BGMConfig.from_config_dict(self.bgm_path, bgm_config)
|
||||
# 从直通输出中提取音频
|
||||
main_audio_path = self.work_dir / f"pass_through_audio_{self.plan.id}.aac"
|
||||
extract_cmd = [
|
||||
FFMPEG_BIN,
|
||||
"-y",
|
||||
"-i",
|
||||
str(output_path),
|
||||
"-vn",
|
||||
"-acodec",
|
||||
"aac",
|
||||
"-b:a",
|
||||
"128k",
|
||||
str(main_audio_path),
|
||||
]
|
||||
try:
|
||||
from video_processing.ffmpeg_utils import run_ffmpeg
|
||||
|
||||
run_ffmpeg(extract_cmd)
|
||||
final_audio = mix_bgm_with_main(ctx, main_audio_path, bgm_cfg, video_duration)
|
||||
# 合并回视频
|
||||
|
||||
bgm_output = self.work_dir / f"rendered_{self.plan.id}_bgm.mp4"
|
||||
merge_audio_video(ctx, output_path, final_audio, bgm_output)
|
||||
output_path = bgm_output
|
||||
logger.info("[unified-render] pass-through BGM mix done: plan_id=%s", self.plan.id)
|
||||
except Exception:
|
||||
logger.exception(
|
||||
"[unified-render] pass-through BGM mix failed, skipping: plan_id=%s", self.plan.id
|
||||
)
|
||||
else:
|
||||
ctx = RenderContext(work_dir=self.work_dir, plan_id=self.plan.id)
|
||||
audio_path = mix_audio(ctx, layers, video_duration)
|
||||
config = self.plan.config or {}
|
||||
bgm_config = config.get("bgm", {}) or {}
|
||||
audio_path = mix_audio(
|
||||
ctx,
|
||||
layers,
|
||||
video_duration,
|
||||
bgm_path=self.bgm_path,
|
||||
bgm_config=bgm_config,
|
||||
)
|
||||
t_audio_end = time.time()
|
||||
audio_mix_ms = int((t_audio_end - t_audio_start) * 1000)
|
||||
has_audio = audio_path is not None
|
||||
|
||||
Reference in New Issue
Block a user