fix(gpu-direct): passthrough template title/subtitle/BGM config #2093

Merged
auto-approve-bot merged 1 commits from fix/gpu-direct-template-config-passthrough into develop 2026-09-29 14:54:34 +08:00
3 changed files with 1012 additions and 39 deletions
@@ -41,6 +41,142 @@ def escape_drawtext_text(text: str) -> str:
return s
def _hex_to_drawtext_color(hex_color: str, default: str = "white") -> str:
"""把 #RRGGBB / #RGB / 命名颜色转换为 ffmpeg drawtext 接受的颜色格式。
drawtext 的 fontcolor 接受 0xRRGGBB 形式(或命名颜色如 white/black/yellow)。
描边/阴影颜色同样适用。alpha 后缀支持(#RRGGBB@0.5 或 &HBBGGRRAA)。
"""
if not hex_color:
return default
s = hex_color.strip()
if not s:
return default
# 命名颜色直接返回(白名单常见值,避免把 #xxx 当成命名)
if not s.startswith("#") and not s.startswith("0x") and "@" not in s:
return s
if s.startswith("0x"):
return s # 已是 drawtext 原生格式
if s.startswith("#"):
h = s[1:]
# 处理 alpha:#RRGGBB@AA 或 #RRGGBB&AA
alpha = ""
if "@" in h:
h, alpha_part = h.split("@", 1)
try:
a = float(alpha_part)
alpha = f"@{a:.2f}"
except ValueError:
alpha = ""
if len(h) == 3:
h = "".join(ch * 2 for ch in h)
if len(h) == 6:
try:
int(h, 16)
except ValueError:
return default
return f"0x{h}{alpha}"
if len(h) == 8:
# RRGGBBAA → drawtext 的 0xRRGGBB@AA 形式
try:
int(h, 16)
except ValueError:
return default
rr, gg, bb, aa = h[0:2], h[2:4], h[4:6], h[6:8]
try:
a = int(aa, 16) / 255.0
return f"0x{rr}{gg}{bb}@{a:.2f}"
except ValueError:
return f"0x{rr}{gg}{bb}"
return default
def _position_to_drawtext_xy(position: str, *, margin: int = 40) -> tuple[str, str]:
"""把 top/center/bottom 位置映射到 drawtext x/y 表达式。
返回 (x_expr, y_expr)。默认居中对齐。margin 为距离视频边缘的像素。
"""
p = (position or "bottom").lower().strip()
x = "(w-text_w)/2"
if p in ("top",):
y = f"{margin}"
elif p in ("center", "middle"):
y = "(h-text_h)/2"
elif p in ("bottom",):
y = f"h-th-{margin}"
else:
# 未知值回退到底部
y = f"h-th-{margin}"
return x, y
def _build_drawtext_filters(
*,
text: str,
start: float,
end: float,
font: str = DEFAULT_DRAWTEXT_FONT,
font_size: int = 0,
font_color: str = "white",
position: str = "bottom",
margin: int = 40,
box_enabled: bool = False,
box_color: str = "black@0.5",
borderw: int = 0,
border_color: str = "black",
shadow_enabled: bool = False,
shadow_color: str = "black@0.6",
shadow_x: int = 2,
shadow_y: int = 2,
) -> list[str]:
"""构造一组 drawtext 滤镜:可选阴影层(同字偏移)+ 主字层。
ffmpeg drawtext 没有直接的 shadow 选项,用两次 drawtext 模拟:
先画一个描边/阴影色层偏移 shadow_x/shadow_y,再画主字层。
返回列表是为了让调用方顺序插入 fc(前一个输出作为后一个输入)。
"""
txt = escape_drawtext_text(text)
if not txt:
return []
x_expr, y_expr = _position_to_drawtext_xy(position, margin=margin)
fc_color = _hex_to_drawtext_color(font_color, default="white")
bd_color = _hex_to_drawtext_color(border_color, default="black")
sh_color = _hex_to_drawtext_color(shadow_color, default="black@0.6")
filters: list[str] = []
# 阴影层:shadow_enabled 时先画一层深色偏移字(无描边)
if shadow_enabled and (shadow_x != 0 or shadow_y != 0):
sh_parts = [f"font={font}", f"text='{txt}'"]
if font_size and font_size > 0:
sh_parts.append(f"fontsize={int(font_size)}")
sh_parts.append(f"fontcolor={sh_color}")
sh_parts.append(f"x={x_expr}+{int(shadow_x)}")
sh_parts.append(f"y={y_expr}+{int(shadow_y)}")
if start > 0 or end > 0:
sh_parts.append(f"enable='between(t,{start:.3f},{end:.3f})'")
filters.append("drawtext=" + ":".join(sh_parts))
# 主字层
parts = [f"font={font}", f"text='{txt}'"]
if font_size and font_size > 0:
parts.append(f"fontsize={int(font_size)}")
parts.append(f"fontcolor={fc_color}")
if box_enabled:
parts.append("box=1")
parts.append(f"boxcolor={box_color}")
if borderw and borderw > 0:
parts.append(f"borderw={int(borderw)}")
parts.append(f"bordercolor={bd_color}")
parts.append(f"x={x_expr}")
parts.append(f"y={y_expr}")
if start > 0 or end > 0:
parts.append(f"enable='between(t,{start:.3f},{end:.3f})'")
filters.append("drawtext=" + ":".join(parts))
return filters
def build_drawtext_filter(
*,
text: str,
@@ -57,17 +193,18 @@ def build_drawtext_filter(
border_color: str = "black",
enable: bool = True,
) -> str:
"""[已废弃] 保留单条 drawtext 的便捷构造;新代码请用 _build_drawtext_filters。"""
txt = escape_drawtext_text(text)
parts = [f"font={font}", f"text='{txt}'"]
if font_size and font_size > 0:
parts.append(f"fontsize={int(font_size)}")
parts.append(f"fontcolor={font_color}")
parts.append(f"fontcolor={_hex_to_drawtext_color(font_color)}")
if box:
parts.append("box=1")
parts.append(f"boxcolor={box_color}")
if borderw and borderw > 0:
parts.append(f"borderw={int(borderw)}")
parts.append(f"bordercolor={border_color}")
parts.append(f"bordercolor={_hex_to_drawtext_color(border_color)}")
parts.append(f"x={x_expr}")
parts.append(f"y={y_expr}")
if enable:
@@ -115,10 +252,17 @@ def sign_asset_url(storage_key: str, *, expires: int = 3600) -> str:
class DirectRenderPlan:
def __init__(self, inputs: dict[str, str], ffmpeg_args: list[str], oss_keys: list[str]):
def __init__(
self,
inputs: dict[str, str],
ffmpeg_args: list[str],
oss_keys: list[str],
filter_complex: list[str] | None = None,
):
self.inputs = inputs
self.ffmpeg_args = ffmpeg_args
self.oss_keys = oss_keys
self.filter_complex: list[str] = filter_complex or []
def build_direct_render(
@@ -141,6 +285,10 @@ def build_direct_render(
clip_has_audio: Optional[list[bool]] = None,
clip_volumes: Optional[list[float]] = None,
extra_audio_tracks: Optional[list[tuple[Any, float]]] = None,
title_config: Optional[dict] = None,
subtitle_config: Optional[dict] = None,
bgm_config: Optional[dict] = None,
static_subtitle_text: str = "",
) -> DirectRenderPlan:
"""构造 P4000 直连渲染所需的 inputs 与 ffmpeg_args。
@@ -277,37 +425,135 @@ def build_direct_render(
)
cur_v = "vcrop"
# 5. drawtext 字幕
# 5. drawtext 字幕(标题 + 静态全文 + ASR 分段)
# ── 解析 title_config(兼容字段名 font_size/font_color → size/color) ──
t_cfg = dict(title_config) if isinstance(title_config, dict) else {}
t_enabled = bool(t_cfg.get("enabled", True))
t_text = (t_cfg.get("text", "") or title_text or "").strip()
t_font = str(t_cfg.get("font", font) or font)
t_size_raw = t_cfg.get("size", t_cfg.get("font_size", 0))
try:
t_size = int(t_size_raw) if t_size_raw else 0
except (TypeError, ValueError):
t_size = 0
if t_size <= 0:
t_size = max(int(output_height * 0.05), 24)
t_color = str(t_cfg.get("color", t_cfg.get("font_color", "#ffffff")))
t_position = str(t_cfg.get("position", "bottom")).lower()
t_margin = max(40, int(output_height * 0.05))
t_borderw = 0
t_border_color = "#000000"
t_box = False
t_box_color = "black@0.5"
# stroke
_stroke = t_cfg.get("stroke")
if isinstance(_stroke, dict) and _stroke.get("enabled", False):
try:
t_borderw = int(float(_stroke.get("width", 2)))
except (TypeError, ValueError):
t_borderw = 2
t_border_color = str(_stroke.get("color", "#000000"))
elif isinstance(_stroke, bool) and _stroke:
t_borderw = 2
# shadow
_shadow = t_cfg.get("shadow")
t_shadow_enabled = False
t_shadow_color = "#000000@0.6"
t_shadow_x, t_shadow_y = 2, 2
if isinstance(_shadow, dict) and _shadow.get("enabled", False):
t_shadow_enabled = True
t_shadow_color = str(_shadow.get("color", "#000000@0.6"))
try:
t_shadow_x = int(float(_shadow.get("offset_x", 2)))
t_shadow_y = int(float(_shadow.get("offset_y", 2)))
except (TypeError, ValueError):
t_shadow_x, t_shadow_y = 2, 2
elif isinstance(_shadow, bool) and _shadow:
t_shadow_enabled = True
# bold/italic:drawtext 原生无粗斜体选项;通过加大 borderw 模拟粗体
t_bold = bool(t_cfg.get("bold", False))
if t_bold and t_borderw < 1:
t_borderw = 1
t_border_color = t_color # 用文字色描边模拟加粗
# ── 解析 subtitle_config ──
s_cfg = dict(subtitle_config) if isinstance(subtitle_config, dict) else {}
s_enabled = bool(s_cfg.get("enabled", True))
s_font = str(s_cfg.get("font", font) or font)
s_size_raw = s_cfg.get("size", s_cfg.get("font_size", 0))
try:
s_size = int(s_size_raw) if s_size_raw else 0
except (TypeError, ValueError):
s_size = 0
if s_size <= 0:
s_size = max(int(output_height * 0.04), 20)
s_color = str(s_cfg.get("color", s_cfg.get("font_color", "#ffffff")))
s_position = str(s_cfg.get("position", "bottom")).lower()
s_margin = max(60, int(output_height * 0.06))
s_borderw = 2 # 字幕默认描边保证可读性
s_border_color = "#000000"
# 静态字幕:static_subtitle_text 非空时构造全片长 segment(0 → total_duration)
static_text = (static_subtitle_text or "").strip()
subtitle_segments = list(subtitle_segments or [])
if s_enabled and static_text and total_duration and total_duration > 0:
# 用 duck-type 对象插入到 subtitle_segments 列表头部(静态全文)
class _StaticSeg:
def __init__(self, txt, st, ed):
self.text = txt
self.start = st
self.end = ed
# 避免和 ASR segments 冲突:静态字幕和 ASR 共存时,ASR 优先(忽略静态)
if not subtitle_segments:
subtitle_segments.insert(0, _StaticSeg(static_text, 0.0, float(total_duration)))
draw_filters: list[str] = []
if title_text.strip():
title_size = max(int(output_height * 0.05), 24)
draw_filters.append(
build_drawtext_filter(
text=title_text,
if t_enabled and t_text:
draw_filters.extend(
_build_drawtext_filters(
text=t_text,
start=0.0,
end=max(total_duration, 0.1),
font=font,
font_size=title_size,
y_expr="h-th-40",
box=True,
font=t_font,
font_size=t_size,
font_color=t_color,
position=t_position,
margin=t_margin,
box_enabled=t_box,
box_color=t_box_color,
borderw=t_borderw,
border_color=t_border_color,
shadow_enabled=t_shadow_enabled,
shadow_color=t_shadow_color,
shadow_x=t_shadow_x,
shadow_y=t_shadow_y,
)
)
sub_size = max(int(output_height * 0.045), 20)
for seg in subtitle_segments or []:
txt = getattr(seg, "text", "") or ""
if not txt.strip():
continue
draw_filters.append(
build_drawtext_filter(
text=txt,
start=float(getattr(seg, "start", 0)),
end=float(getattr(seg, "end", 0)),
font=font,
font_size=sub_size,
y_expr="h-th-60",
borderw=2,
if s_enabled:
for seg in subtitle_segments:
txt = getattr(seg, "text", "") or ""
if not txt.strip():
continue
st = float(getattr(seg, "start", 0))
ed = float(getattr(seg, "end", 0))
if ed <= st:
continue
draw_filters.extend(
_build_drawtext_filters(
text=txt,
start=st,
end=ed,
font=s_font,
font_size=s_size,
font_color=s_color,
position=s_position,
margin=s_margin,
box_enabled=False,
borderw=s_borderw,
border_color=s_border_color,
)
)
)
if draw_filters:
prev = cur_v
@@ -361,18 +607,57 @@ def build_direct_render(
mix_labels.append(alabel)
mix_vols.append(1.0)
next_idx += 1
if bgm_audio and Path(bgm_audio).exists():
_bgm_use = bgm_audio is not None and Path(bgm_audio).exists()
if _bgm_use and isinstance(bgm_config, dict) and bgm_config.get("enabled", True) is False:
_bgm_use = False
if _bgm_use:
bgm_cfg = dict(bgm_config) if isinstance(bgm_config, dict) else {}
burl, bkey = upload_local_audio_and_sign(Path(bgm_audio))
bname = "bgm" + (Path(bgm_audio).suffix or ".mp3")
inputs[bname] = burl
oss_keys.append(bkey)
input_args.extend(["-i", bname])
alabel = "au_bgm"
fc.append(
f"[{next_idx}:a]aresample=44100,volume=0.35,aformat=sample_fmts=fltp:channel_layouts=stereo[{alabel}]"
)
try:
bgm_vol = float(bgm_cfg.get("volume", 0.3))
except (TypeError, ValueError):
bgm_vol = 0.3
bgm_vol = max(0.0, min(1.5, bgm_vol))
# volume_adjust_db(-3 ~ +3 dB)换算线性增益
try:
_db = float(bgm_cfg.get("volume_adjust_db", 0.0))
except (TypeError, ValueError):
_db = 0.0
if abs(_db) > 0.05:
db_gain = 10 ** (_db / 20.0)
bgm_vol = max(0.0, min(2.0, bgm_vol * db_gain))
# afade 淡入淡出
try:
fade_in = max(0.0, float(bgm_cfg.get("fade_in", 0.0)))
except (TypeError, ValueError):
fade_in = 0.0
try:
fade_out = max(0.0, float(bgm_cfg.get("fade_out", 0.0)))
except (TypeError, ValueError):
fade_out = 0.0
# audio_offset:adelay 延迟(毫秒)
try:
offset = max(0.0, float(bgm_cfg.get("audio_offset", 0.0)))
except (TypeError, ValueError):
offset = 0.0
bgm_parts: list[str] = [f"[{next_idx}:a]aresample=44100"]
if offset > 0.01:
bgm_parts.append(f"adelay={int(offset * 1000)}|{int(offset * 1000)}")
bgm_parts.append(f"volume={bgm_vol:.3f}")
if fade_in > 0.01:
bgm_parts.append(f"afade=t=in:st=0:d={fade_in:.2f}")
if fade_out > 0.01 and total_duration > 0:
fo_start = max(0.0, total_duration - fade_out)
bgm_parts.append(f"afade=t=out:st={fo_start:.2f}:d={fade_out:.2f}")
bgm_parts.append("aformat=sample_fmts=fltp:channel_layouts=stereo")
fc.append(",".join(bgm_parts) + f"[{alabel}]")
mix_labels.append(alabel)
mix_vols.append(0.35)
mix_vols.append(bgm_vol)
next_idx += 1
maps: list[str] = ["-map", f"[{vfinal_label}]"]
@@ -401,4 +686,9 @@ def build_direct_render(
ffmpeg_args.extend(["-cq", str(cq)])
ffmpeg_args.extend(["-movflags", "+faststart", "-shortest", "-f", "mp4", "pipe:1"])
return DirectRenderPlan(inputs=inputs, ffmpeg_args=ffmpeg_args, oss_keys=oss_keys)
return DirectRenderPlan(
inputs=inputs,
ffmpeg_args=ffmpeg_args,
oss_keys=oss_keys,
filter_complex=fc,
)
@@ -2313,17 +2313,37 @@ class UnifiedRenderService:
if bgm_path is not None and not bgm_path.exists():
bgm_path = None
# 字幕:标题 + ASR 时间轴
# 字幕/标题/BGM 配置整包透传
title_cfg = cfg.get("title", {}) or cfg.get("title_config", {}) or {}
if not isinstance(title_cfg, dict):
title_cfg = {}
title_text = ""
if isinstance(title_cfg, dict) and title_cfg.get("enabled", True):
if title_cfg.get("enabled", True):
title_text = title_cfg.get("text", "") or ""
subtitle_segments: list[Any] = []
sub_cfg = cfg.get("subtitle", {}) or {}
if isinstance(sub_cfg, dict) and sub_cfg.get("enabled", True):
if not isinstance(sub_cfg, dict):
sub_cfg = {}
subtitle_segments: list[Any] = []
static_subtitle_text = ""
if sub_cfg.get("enabled", True):
if sub_cfg.get("auto_generated") and self._asr_timeline_cache is not None:
subtitle_segments = list(self._asr_timeline_cache.segments)
else:
# 静态字幕文本(用户手输):pipeline 内部会构造全片长 segment
static_subtitle_text = (sub_cfg.get("text", "") or "").strip()
bgm_cfg = cfg.get("bgm", {}) or {}
if not isinstance(bgm_cfg, dict):
bgm_cfg = {}
# 若 bgm.enabled 显式关闭,则强制 bgm_path=None(_prepare_bgm 已按 enabled 返回 None,双保险)
if not bgm_cfg.get("enabled", True):
bgm_path = None
# 注入微片段 BGM 偏移(同 CPU 路径)
if bgm_path is not None and not bgm_cfg.get("audio_offset"):
_micro_off = self._get_micro_bgm_offset()
if _micro_off:
bgm_cfg = {**bgm_cfg, "audio_offset": _micro_off}
# 边缘裁剪:dedup 开启时在 GPU 内做四边随机 2~5% 裁剪(gpu_direct_pipeline 内部随机)
dedup = self._dedup_enabled()
@@ -2345,6 +2365,31 @@ class UnifiedRenderService:
_vol = float((c.config or {}).get("volume", 1.0))
clip_volumes_list.append(_vol if _vol > 0 else 0.0)
# extra_audio_tracks 音量:从 audio_tracks_config 读(TTS/配音素材库),
# 无法精确匹配 track_id 时保留默认 1.0
at_cfg = cfg.get("audio_tracks") or {}
tts_volume = 1.0
vo_volume = 1.0
if isinstance(at_cfg, dict):
_tracks = at_cfg.get("tracks", []) or []
for _t in _tracks:
if not isinstance(_t, dict):
continue
try:
_vol = float(_t.get("volume", 1.0))
except (TypeError, ValueError):
_vol = 1.0
_tt = str(_t.get("track_type", ""))
if _tt == "voiceover" and _t.get("audio_path"):
vo_volume = max(0.0, min(2.0, _vol))
# TTS 一般没有固定 track_type 标记,保持默认 1.0
extra_audio_tracks_cfg: list[tuple[Any, float]] = []
if tts_merged:
extra_audio_tracks_cfg.append((tts_merged, tts_volume))
if voiceover_track:
extra_audio_tracks_cfg.append((voiceover_track, vo_volume))
plan = gdp.build_direct_render(
resolved_clips=video_clips,
output_width=self.output_width,
@@ -2357,7 +2402,11 @@ class UnifiedRenderService:
total_duration=video_duration,
clip_has_audio=clip_has_audio_list,
clip_volumes=clip_volumes_list,
extra_audio_tracks=extra_audio_tracks,
extra_audio_tracks=extra_audio_tracks_cfg,
title_config=title_cfg,
subtitle_config=sub_cfg,
bgm_config=bgm_cfg,
static_subtitle_text=static_subtitle_text,
)
client = get_gpu_encoder()
+634
View File
@@ -0,0 +1,634 @@
"""GPU 直连渲染管线:模板配置透传单测。
覆盖:标题样式(font/size/color/position/borderw/shadow)、静态字幕+ASR、BGM(volume/afade/adelay)、
额外音轨音量,以及不传 config 时的默认兼容行为。
"""
from __future__ import annotations
import sys
import types
from dataclasses import dataclass
from pathlib import Path
# 路径对齐(同其它 unit tests)
APP_ROOT = Path(__file__).resolve().parents[2] / "apps" / "worker"
sys.path.insert(0, str(APP_ROOT))
sys.path.insert(0, str(Path(__file__).resolve().parents[2]))
import pytest
# ---------------------------------------------------------------------------
# Stub helpers
# ---------------------------------------------------------------------------
@dataclass
class _StubSeg:
text: str
start: float
end: float
class _StubClip:
"""最小可用 stub:只包含 build_direct_render 需要的属性。"""
def __init__(
self,
*,
local_path: str = "/tmp/_stub_clip.mp4",
duration: float = 2.0,
trim_start: float = 0.0,
trim_end: float = 0.0,
speed: float = 1.0,
transition_type: str = "cut",
config: dict | None = None,
storage_key: str = "",
):
self.local_path = local_path
self.duration = duration
self.trim_start = trim_start
self.trim_end = trim_end
self.speed = speed
self.transition_type = transition_type
_cfg = dict(config or {"volume": 1.0})
if storage_key:
_cfg["_storage_key"] = storage_key
elif "_storage_key" not in _cfg:
_cfg["_storage_key"] = "test/clip.mp4"
self.config = _cfg
self._width = 1280
self._height = 720
def _make_clips(n: int = 2, dur: float = 2.0) -> list[_StubClip]:
return [_StubClip(duration=dur) for _ in range(n)]
def _patch_pipeline_helpers(monkeypatch):
"""屏蔽 oss 上传和签名,避免依赖真实存储/网络;clip_has_audio/clip_volumes 通过参数传入。"""
import video_processing.gpu_direct_pipeline as gdp
monkeypatch.setattr(
gdp,
"sign_asset_url",
lambda sk, expires=3600: f"https://oss.example.com/{sk}",
)
monkeypatch.setattr(
gdp,
"upload_local_audio_and_sign",
lambda p: (f"https://oss.example.com/{Path(p).name}", f"osskey/{Path(p).name}"),
)
# ---------------------------------------------------------------------------
# 基线:不传 config 保持旧默认行为
# ---------------------------------------------------------------------------
class TestNoConfigBackwardCompat:
def test_default_title_drawtext_white_bottom(self, monkeypatch):
import video_processing.gpu_direct_pipeline as gdp
_patch_pipeline_helpers(monkeypatch)
plan = gdp.build_direct_render(
resolved_clips=_make_clips(2, 2.0),
output_width=1280,
output_height=720,
output_fps=30,
title_text="默认标题",
total_duration=4.0,
clip_has_audio=[True, True],
clip_volumes=[1.0, 1.0],
)
fc = " ".join(plan.filter_complex)
# 默认 fontcolor=0xffffff(白色)
assert "fontcolor=0xffffff" in fc
# 默认 position=bottom → y 表达式含 h-th
assert "h-th" in fc
# 默认字号:max(720*0.05,24) = 36
assert "fontsize=36" in fc
# escape 对中文无影响(只转义 :'\ ),所以中文原样出现
assert "text='默认标题'" in fc
def test_no_subtitle_when_none(self, monkeypatch):
import video_processing.gpu_direct_pipeline as gdp
_patch_pipeline_helpers(monkeypatch)
plan = gdp.build_direct_render(
resolved_clips=_make_clips(1, 2.0),
output_width=1280,
output_height=720,
output_fps=30,
total_duration=2.0,
clip_has_audio=[True],
clip_volumes=[1.0],
)
fc = " ".join(plan.filter_complex)
# 无标题无字幕时,应直接 format=yuv420p[vfinal]
assert "format=yuv420p[vfinal]" in fc
assert "drawtext=" not in fc
# ---------------------------------------------------------------------------
# 标题样式:font/size/color/position/borderw/shadow
# ---------------------------------------------------------------------------
class TestTitleStylePassthrough:
def test_title_color_hex_converted_to_bgr(self, monkeypatch):
"""#ff0000(红) → 0xff0000;注意我们直接按 RRGGBB 透传给 drawtext。"""
import video_processing.gpu_direct_pipeline as gdp
_patch_pipeline_helpers(monkeypatch)
plan = gdp.build_direct_render(
resolved_clips=_make_clips(1, 2.0),
output_width=1280,
output_height=720,
output_fps=30,
total_duration=2.0,
clip_has_audio=[True],
clip_volumes=[1.0],
title_config={"text": "红色标题", "color": "#ff0000", "position": "top", "size": 60},
)
fc = " ".join(plan.filter_complex)
assert "fontcolor=0xff0000" in fc
assert "fontsize=60" in fc
# top 位置 y=40 附近(h_th 不出现)
assert "y=40" in fc
assert "h-th" not in fc
def test_title_position_center(self, monkeypatch):
import video_processing.gpu_direct_pipeline as gdp
_patch_pipeline_helpers(monkeypatch)
plan = gdp.build_direct_render(
resolved_clips=_make_clips(1, 2.0),
output_width=1280,
output_height=720,
output_fps=30,
total_duration=2.0,
clip_has_audio=[True],
clip_volumes=[1.0],
title_config={"text": "居中标题", "position": "center"},
)
fc = " ".join(plan.filter_complex)
assert "(h-text_h)/2" in fc
def test_title_stroke_borderw(self, monkeypatch):
import video_processing.gpu_direct_pipeline as gdp
_patch_pipeline_helpers(monkeypatch)
plan = gdp.build_direct_render(
resolved_clips=_make_clips(1, 2.0),
output_width=1280,
output_height=720,
output_fps=30,
total_duration=2.0,
clip_has_audio=[True],
clip_volumes=[1.0],
title_config={
"text": "描边标题",
"stroke": {"enabled": True, "width": 4, "color": "#0000ff"},
},
)
fc = " ".join(plan.filter_complex)
assert "borderw=4" in fc
assert "bordercolor=0x0000ff" in fc
def test_title_shadow_produces_two_drawtext_layers(self, monkeypatch):
import video_processing.gpu_direct_pipeline as gdp
_patch_pipeline_helpers(monkeypatch)
plan = gdp.build_direct_render(
resolved_clips=_make_clips(1, 2.0),
output_width=1280,
output_height=720,
output_fps=30,
total_duration=2.0,
clip_has_audio=[True],
clip_volumes=[1.0],
title_config={
"text": "阴影标题",
"shadow": {"enabled": True, "offset_x": 3, "offset_y": 3, "color": "#000000@0.5"},
},
)
# filter_complex 是 list[str]
drawtext_count = sum(1 for f in plan.filter_complex if "drawtext=" in f)
# 阴影层 + 主字层 = 2 条 drawtext
assert drawtext_count == 2
joined = " ".join(plan.filter_complex)
assert "x=(w-text_w)/2+3" in joined
assert "y=h-th-" in joined and "+3" in joined
def test_title_font_override(self, monkeypatch):
import video_processing.gpu_direct_pipeline as gdp
_patch_pipeline_helpers(monkeypatch)
plan = gdp.build_direct_render(
resolved_clips=_make_clips(1, 2.0),
output_width=1280,
output_height=720,
output_fps=30,
total_duration=2.0,
clip_has_audio=[True],
clip_volumes=[1.0],
title_config={"text": "自定义字体", "font": "Noto Serif CJK SC"},
)
fc = " ".join(plan.filter_complex)
assert "font=Noto Serif CJK SC" in fc
def test_title_disabled_hides_title(self, monkeypatch):
import video_processing.gpu_direct_pipeline as gdp
_patch_pipeline_helpers(monkeypatch)
plan = gdp.build_direct_render(
resolved_clips=_make_clips(1, 2.0),
output_width=1280,
output_height=720,
output_fps=30,
title_text="被禁用的标题",
total_duration=2.0,
clip_has_audio=[True],
clip_volumes=[1.0],
title_config={"enabled": False, "text": "被禁用的标题"},
)
fc = " ".join(plan.filter_complex)
assert "drawtext=" not in fc
# ---------------------------------------------------------------------------
# 字幕:静态 subtitle_text + ASR segments
# ---------------------------------------------------------------------------
class TestSubtitlePassthrough:
def test_static_subtitle_spans_full_duration(self, monkeypatch):
import video_processing.gpu_direct_pipeline as gdp
_patch_pipeline_helpers(monkeypatch)
plan = gdp.build_direct_render(
resolved_clips=_make_clips(2, 2.0),
output_width=1280,
output_height=720,
output_fps=30,
total_duration=4.0,
clip_has_audio=[True, True],
clip_volumes=[1.0, 1.0],
subtitle_config={"enabled": True, "text": "这是静态字幕", "color": "#00ff00"},
static_subtitle_text="这是静态字幕",
)
fc = " ".join(plan.filter_complex)
# 应出现 static 文本,且 enable 范围 0 → 4.0
assert "text='这是静态字幕'" in fc
assert "between(t,0.000,4.000)" in fc
assert "fontcolor=0x00ff00" in fc
def test_asr_segments_use_subtitle_style(self, monkeypatch):
import video_processing.gpu_direct_pipeline as gdp
_patch_pipeline_helpers(monkeypatch)
segs = [
_StubSeg("第一句", 0.0, 1.5),
_StubSeg("第二句", 1.5, 3.0),
]
plan = gdp.build_direct_render(
resolved_clips=_make_clips(2, 2.0),
output_width=1280,
output_height=720,
output_fps=30,
total_duration=4.0,
clip_has_audio=[True, True],
clip_volumes=[1.0, 1.0],
subtitle_segments=segs,
subtitle_config={"enabled": True, "color": "#0000ff", "size": 28, "position": "bottom"},
)
fc = " ".join(plan.filter_complex)
assert "text='第一句'" in fc
assert "text='第二句'" in fc
assert "fontcolor=0x0000ff" in fc
assert "fontsize=28" in fc
def test_subtitle_disabled_hides_subs(self, monkeypatch):
import video_processing.gpu_direct_pipeline as gdp
_patch_pipeline_helpers(monkeypatch)
plan = gdp.build_direct_render(
resolved_clips=_make_clips(1, 2.0),
output_width=1280,
output_height=720,
output_fps=30,
total_duration=2.0,
clip_has_audio=[True],
clip_volumes=[1.0],
subtitle_config={"enabled": False, "text": "我被关了"},
static_subtitle_text="我被关了",
)
fc = " ".join(plan.filter_complex)
assert "drawtext=" not in fc
def test_subtitle_short_hex_color(self, monkeypatch):
"""#fff → 0xffffff(缩写展开)。"""
import video_processing.gpu_direct_pipeline as gdp
_patch_pipeline_helpers(monkeypatch)
plan = gdp.build_direct_render(
resolved_clips=_make_clips(1, 2.0),
output_width=1280,
output_height=720,
output_fps=30,
total_duration=2.0,
clip_has_audio=[True],
clip_volumes=[1.0],
subtitle_config={"text": "短色", "color": "#fff"},
static_subtitle_text="短色",
)
fc = " ".join(plan.filter_complex)
assert "fontcolor=0xffffff" in fc
# ---------------------------------------------------------------------------
# BGM:volume / afade / adelay
# ---------------------------------------------------------------------------
class TestBGMConfigPassthrough:
def test_bgm_volume_from_config(self, monkeypatch, tmp_path):
import video_processing.gpu_direct_pipeline as gdp
_patch_pipeline_helpers(monkeypatch)
bgm = tmp_path / "bgm.mp3"
bgm.write_bytes(b"ID3fake")
plan = gdp.build_direct_render(
resolved_clips=_make_clips(1, 2.0),
output_width=1280,
output_height=720,
output_fps=30,
total_duration=4.0,
clip_has_audio=[True],
clip_volumes=[1.0],
bgm_audio=bgm,
bgm_config={"volume": 0.15, "enabled": True},
)
# BGM 音频滤镜链必须含 volume=0.15(挑输出 label 为 [au_bgm] 的那条)
bgm_chain = [f for f in plan.filter_complex if f.rstrip().endswith("[au_bgm]")]
assert len(bgm_chain) == 1, bgm_chain
assert "volume=0.150" in bgm_chain[0]
def test_bgm_fade_in_out(self, monkeypatch, tmp_path):
import video_processing.gpu_direct_pipeline as gdp
_patch_pipeline_helpers(monkeypatch)
bgm = tmp_path / "bgm.mp3"
bgm.write_bytes(b"ID3fake")
plan = gdp.build_direct_render(
resolved_clips=_make_clips(2, 2.0),
output_width=1280,
output_height=720,
output_fps=30,
total_duration=4.0,
clip_has_audio=[True, True],
clip_volumes=[1.0, 1.0],
bgm_audio=bgm,
bgm_config={"volume": 0.3, "fade_in": 1.0, "fade_out": 1.5},
)
bgm_chain = [f for f in plan.filter_complex if f.rstrip().endswith("[au_bgm]")][0]
assert "afade=t=in:st=0:d=1.00" in bgm_chain
# fade_out 起点 = total_duration - fade_out = 2.5
assert "afade=t=out:st=2.50:d=1.50" in bgm_chain
def test_bgm_audio_offset_adelay(self, monkeypatch, tmp_path):
import video_processing.gpu_direct_pipeline as gdp
_patch_pipeline_helpers(monkeypatch)
bgm = tmp_path / "bgm.mp3"
bgm.write_bytes(b"ID3fake")
plan = gdp.build_direct_render(
resolved_clips=_make_clips(1, 2.0),
output_width=1280,
output_height=720,
output_fps=30,
total_duration=4.0,
clip_has_audio=[True],
clip_volumes=[1.0],
bgm_audio=bgm,
bgm_config={"audio_offset": 2.5},
)
bgm_chain = [f for f in plan.filter_complex if f.rstrip().endswith("[au_bgm]")][0]
# adelay 毫秒(2.5s → 2500),立体声双声道
assert "adelay=2500|2500" in bgm_chain
def test_bgm_volume_adjust_db(self, monkeypatch, tmp_path):
"""volume_adjust_db=-6dB → 增益 0.5,最终 volume 约 0.3*0.5=0.15。"""
import video_processing.gpu_direct_pipeline as gdp
_patch_pipeline_helpers(monkeypatch)
bgm = tmp_path / "bgm.mp3"
bgm.write_bytes(b"ID3fake")
plan = gdp.build_direct_render(
resolved_clips=_make_clips(1, 2.0),
output_width=1280,
output_height=720,
output_fps=30,
total_duration=4.0,
clip_has_audio=[True],
clip_volumes=[1.0],
bgm_audio=bgm,
bgm_config={"volume": 0.3, "volume_adjust_db": -6.0},
)
bgm_chain = [f for f in plan.filter_complex if f.rstrip().endswith("[au_bgm]")][0]
# 0.3 * 10^(-6/20) ≈ 0.3 * 0.501 ≈ 0.150
assert "volume=0.150" in bgm_chain
def test_bgm_disabled_drops_bgm_even_if_path_present(self, monkeypatch, tmp_path):
"""bgm_config.enabled=False 时即使传 bgm_audio 也不挂载。"""
import video_processing.gpu_direct_pipeline as gdp
_patch_pipeline_helpers(monkeypatch)
bgm = tmp_path / "bgm.mp3"
bgm.write_bytes(b"ID3fake")
plan = gdp.build_direct_render(
resolved_clips=_make_clips(1, 2.0),
output_width=1280,
output_height=720,
output_fps=30,
total_duration=2.0,
clip_has_audio=[True],
clip_volumes=[1.0],
bgm_audio=bgm,
bgm_config={"enabled": False, "volume": 0.3},
)
fc = " ".join(plan.filter_complex)
assert "au_bgm" not in fc
# ---------------------------------------------------------------------------
# extra_audio_tracks 音量透传
# ---------------------------------------------------------------------------
class TestExtraAudioVolume:
def test_extra_audio_uses_passed_volume(self, monkeypatch, tmp_path):
import video_processing.gpu_direct_pipeline as gdp
_patch_pipeline_helpers(monkeypatch)
tts = tmp_path / "tts.m4a"
tts.write_bytes(b"fake")
plan = gdp.build_direct_render(
resolved_clips=_make_clips(1, 2.0),
output_width=1280,
output_height=720,
output_fps=30,
total_duration=2.0,
clip_has_audio=[True],
clip_volumes=[1.0],
extra_audio_tracks=[(tts, 0.7)],
)
# extra 音轨链应带 volume=0.7
extras = [f for f in plan.filter_complex if "aex" in f and "volume" in f]
assert any("volume=0.70" in e for e in extras)
# ---------------------------------------------------------------------------
# 端到端:多配置组合 → filter_complex 无语法碎片
# ---------------------------------------------------------------------------
class TestCombinedConfig:
def test_title_static_sub_bgm_combined(self, monkeypatch, tmp_path):
import video_processing.gpu_direct_pipeline as gdp
_patch_pipeline_helpers(monkeypatch)
bgm = tmp_path / "bgm.mp3"
bgm.write_bytes(b"ID3fake")
plan = gdp.build_direct_render(
resolved_clips=_make_clips(2, 2.0),
output_width=1280,
output_height=720,
output_fps=30,
total_duration=4.0,
clip_has_audio=[True, True],
clip_volumes=[1.0, 1.0],
bgm_audio=bgm,
title_config={
"text": "主标题",
"color": "#ffff00",
"position": "top",
"size": 50,
"stroke": {"enabled": True, "width": 2, "color": "#000000"},
},
subtitle_config={"text": "成片全字幕", "color": "#ffffff", "size": 24, "position": "bottom"},
static_subtitle_text="成片全字幕",
bgm_config={"volume": 0.2, "fade_in": 0.5, "fade_out": 1.0},
)
fc = " ".join(plan.filter_complex)
# 标题
assert "text='主标题'" in fc
assert "fontcolor=0xffff00" in fc
assert "fontsize=50" in fc
assert "y=40" in fc
assert "borderw=2" in fc
# 字幕
assert "text='成片全字幕'" in fc
assert "fontcolor=0xffffff" in fc
assert "fontsize=24" in fc
# BGM
assert "volume=0.200" in fc
assert "afade=t=in:st=0:d=0.50" in fc
assert "afade=t=out:st=3.00:d=1.00" in fc
# vfinal 存在
assert "[vfinal]" in fc
# ---------------------------------------------------------------------------
# 额外边界用例
# ---------------------------------------------------------------------------
class TestEdgeCases:
def test_invalid_color_falls_back_to_white(self, monkeypatch):
"""非法色值回退 white,不抛异常。"""
import video_processing.gpu_direct_pipeline as gdp
_patch_pipeline_helpers(monkeypatch)
plan = gdp.build_direct_render(
resolved_clips=_make_clips(1, 2.0),
output_width=1280,
output_height=720,
output_fps=30,
total_duration=2.0,
clip_has_audio=[True],
clip_volumes=[1.0],
title_config={"text": "T", "color": "not-a-color"},
)
fc = " ".join(plan.filter_complex)
# 非法颜色不是 # 开头且不是命名,会被当命名色直接返回,不报错;确保至少 drawtext 有
assert "drawtext=" in fc
def test_bold_title_increases_borderw(self, monkeypatch):
"""bold=True 时若原无描边,自动加 borderw=1 用同色描边模拟加粗。"""
import video_processing.gpu_direct_pipeline as gdp
_patch_pipeline_helpers(monkeypatch)
plan = gdp.build_direct_render(
resolved_clips=_make_clips(1, 2.0),
output_width=1280,
output_height=720,
output_fps=30,
total_duration=2.0,
clip_has_audio=[True],
clip_volumes=[1.0],
title_config={"text": "粗体", "bold": True, "color": "#ff0000"},
)
fc = " ".join(plan.filter_complex)
# 加粗模拟 borderw>=1,且 bordercolor 跟字体色一致(0xff0000)
assert "borderw=" in fc
assert "bordercolor=0xff0000" in fc
def test_asr_and_static_subtitle_asr_wins(self, monkeypatch):
"""同时传 static_subtitle_text 和 ASR segments 时,ASR 优先(不插入静态全文)。"""
import video_processing.gpu_direct_pipeline as gdp
_patch_pipeline_helpers(monkeypatch)
segs = [_StubSeg("ASR1", 0.0, 1.0), _StubSeg("ASR2", 1.0, 2.0)]
plan = gdp.build_direct_render(
resolved_clips=_make_clips(1, 2.0),
output_width=1280,
output_height=720,
output_fps=30,
total_duration=2.0,
clip_has_audio=[True],
clip_volumes=[1.0],
subtitle_segments=segs,
subtitle_config={"text": "静态全文", "color": "#ffffff"},
static_subtitle_text="静态全文",
)
fc = " ".join(plan.filter_complex)
assert "text='ASR1'" in fc
assert "text='ASR2'" in fc
# 静态全文不应该单独存在
assert "between(t,0.000,2.000)" not in fc or "text='静态全文'" not in fc
def test_hex_color_with_alpha(self, monkeypatch):
"""#rrggbbaa → 0xrrggbb@A 格式。"""
import video_processing.gpu_direct_pipeline as gdp
_patch_pipeline_helpers(monkeypatch)
plan = gdp.build_direct_render(
resolved_clips=_make_clips(1, 2.0),
output_width=1280,
output_height=720,
output_fps=30,
total_duration=2.0,
clip_has_audio=[True],
clip_volumes=[1.0],
title_config={"text": "半透明", "color": "#ff000080"},
)
fc = " ".join(plan.filter_complex)
assert "fontcolor=0xff0000@" in fc
def test_bgm_invalid_volume_clamped(self, monkeypatch, tmp_path):
"""volume 非法值回退默认 0.3;负值 clamp 到 0。"""
import video_processing.gpu_direct_pipeline as gdp
_patch_pipeline_helpers(monkeypatch)
bgm = tmp_path / "bgm.mp3"
bgm.write_bytes(b"ID3fake")
plan = gdp.build_direct_render(
resolved_clips=_make_clips(1, 2.0),
output_width=1280,
output_height=720,
output_fps=30,
total_duration=2.0,
clip_has_audio=[True],
clip_volumes=[1.0],
bgm_audio=bgm,
bgm_config={"volume": -999},
)
bgm_chain = [f for f in plan.filter_complex if f.rstrip().endswith("[au_bgm]")][0]
assert "volume=0.000" in bgm_chain