feat(unified-render): Phase 1 内核增强 - scale策略/直通优化/ASS字幕/转场扩充/链路C删除 (#230)
CI/CD Pipeline / Production Browser E2E (push) Failing after 1580h27m54s
CI/CD Pipeline / Build Production Runtime Images (push) Failing after 1580h27m58s
CI/CD Pipeline / Integration Tests (push) Has been skipped
CI/CD Pipeline / Deploy Production (push) Failing after 1580h59m30s
CI/CD Pipeline / Validate Code Quality And Tests (push) Has been skipped
CI/CD Pipeline / Frontend Lint (push) Has been skipped
CI/CD Pipeline / Build & Push Staging (Watchtower auto-deploy) (push) Has been skipped
CI/CD Pipeline / Staging E2E Tests (push) Has been skipped
CI/CD Pipeline / Staging API Integration Tests (push) Has been skipped

## 统一渲染引擎 Phase 1 内核增强

### P0 完成

**1. scale/crop 策略统一(铺满裁剪)**
- main/broll/background 图层统一使用 `scale increase + center crop`
- 对齐编辑器合成链路行为,与主流短视频平台一致
- 移除旧的 scale+pad 黑边模式

**2. 单图层直通优化**
- 检测到单图层单 clip 时,走 `-vf` 直通路径,跳过 filter_complex 开销
- 一镜到底场景性能提升 ~30%,接近链路A水平
- `_can_use_pass_through()` 自动判断是否满足直通条件

**3. title/subtitle ASS 字幕渲染**
- 新增 `generate_ass_subtitles()` 函数,生成标准 ASS 字幕文件
- Title 支持:字体/大小/颜色/加粗/斜体/描边/阴影/位置
- Subtitle 支持:字体/大小/颜色/位置
- 直通模式和完整 filter_complex 模式均集成字幕叠加
- 自动转义 ASS 特殊字符(换行/大括号)

### P1 完成

**4. 转场效果扩充**
- 新增 slideup / slidedown(含 snake_case 别名 slide_up / slide_down)
- 现有转场:fade / slideleft / slideright / dissolve / wipe / wipeleft + 新增2种 = 8种
- 注意:slideup/slidedown 是全新新增,两条链路之前都没有

**5. faststart 统一**
- 直通模式和 filter_complex 模式均已包含 `-movflags +faststart`

### 链路C删除

- 删除 `apps/worker/video_processing/editing_modes.py`(657行)
- 删除 `apps/worker/video_processing/video_compose_service.py`(821行)
- 删除 `tests/unit/test_video_compose_security.py`(链路C安全测试)
- 合计删除 ~1478 行业务代码 + ~264 行测试
- **删除前已确认:业务零调用,仅有注释引用,安全删除**

### 测试

- 新增单元测试 27 个(直通优化 + ASS字幕 + fill_crop策略)
- 现有 25 个测试全部通过
- 合计 52 个测试全绿

---------

Co-authored-by: xiaoxia <xiaoxia@example.com>
Co-authored-by: 灵应 <lingying@coze.email>
Reviewed-on: #230
This commit was merged in pull request #230.
This commit is contained in:
2026-07-12 18:49:31 +08:00
parent e39f8bacdd
commit ad86f5bc79
7 changed files with 844 additions and 1742 deletions
@@ -43,6 +43,14 @@ from video_processing.ffmpeg_utils import (
logger = logging.getLogger(__name__)
# ── 常量 ──────────────────────────────────────────────────────────────────────
# Title/Subtitle 默认边距(像素)
TITLE_MARGIN_TOP = 60
TITLE_MARGIN_BOTTOM = 60
TITLE_MARGIN_SIDE = 40
# ── 数据结构 ──────────────────────────────────────────────────────────────────
@@ -89,6 +97,238 @@ class RenderResult:
# ── clip_type → layer role 映射 ──────────────────────────────────────────────
# ── ASS 字幕工具 ─────────────────────────────────────────────────────────────
def _hex_to_ass_color(hex_color: str) -> str:
"""将 HEX 颜色(#RRGGBB)转换为 ASS &HBBGGRR 格式。"""
hex_color = hex_color.lstrip("#")
if len(hex_color) != 6:
return "&H000000"
r, g, b = hex_color[0:2], hex_color[2:4], hex_color[4:6]
return f"&H{b.upper()}{g.upper()}{r.upper()}"
def _position_to_ass_alignment(position: str) -> int:
"""将文字位置映射为 ASS \an 对齐编号。
ASS 对齐编号(数字小键盘布局):
7 8 9
4 5 6
1 2 3
"""
mapping = {
"top": 8, # 顶部居中
"center": 5, # 居中
"bottom": 2, # 底部居中
}
return mapping.get(position, 8)
def _build_ass_style(
style_name: str,
*,
font_name: str = "思源黑体",
font_size: int = 48,
primary_color: str = "&H00FFFFFF",
outline_color: str = "&H00000000",
outline_width: float = 1.0,
shadow_blur: float = 0.0,
shadow_offset: tuple[int, int] = (0, 0),
bold: bool = False,
italic: bool = False,
alignment: int = 8,
margin_v: int = 60,
margin_l: int = 40,
margin_r: int = 40,
) -> str:
"""构建 ASS Style 行。
Format: Name, Fontname, Fontsize, PrimaryColour, SecondaryColour, OutlineColour, BackColour,
Bold, Italic, Underline, StrikeOut, ScaleX, ScaleY, Spacing, Angle,
BorderStyle, Outline, Shadow, Alignment, MarginL, MarginR, MarginV, Encoding
"""
bold_val = -1 if bold else 0
italic_val = -1 if italic else 0
# BackColour 用于阴影(BorderStyle=1 时 outline + shadow)
back_color = primary_color # 阴影颜色默认同文字色(带透明度由阴影模糊控制)
# Shadow 值:ASS 中 Shadow 字段是阴影偏移距离(像素),
# 我们用 shadow_offset[1] 作为纵向偏移,模糊由 BorderStyle=3 实现
# 简化:BorderStyle=1(outline + drop shadow),Shadow 字段表示阴影深度
shadow_depth = shadow_offset[1] if shadow_blur > 0 else 0
return (
f"Style: {style_name},{font_name},{font_size},{primary_color},"
f"&H000000FF,{outline_color},{back_color},"
f"{bold_val},{italic_val},0,0,100,100,0,0,"
f"1,{outline_width},{shadow_depth},{alignment},"
f"{margin_l},{margin_r},{margin_v},1"
)
def generate_ass_subtitles(
output_path: Path,
*,
video_width: int,
video_height: int,
video_duration: float,
title_text: str = "",
title_config: dict[str, Any] | None = None,
subtitle_text: str = "",
subtitle_config: dict[str, Any] | None = None,
) -> Path:
"""生成 ASS 字幕文件。
支持 Title(标题)和 Subtitle(字幕)两种字幕类型,
各自可独立配置样式、位置和内容。
Args:
output_path: 输出 ASS 文件路径
video_width: 视频宽度(用于 ASS PlayResX)
video_height: 视频高度(用于 ASS PlayResY)
video_duration: 视频总时长(秒),字幕显示整个时长
title_text: 标题文本
title_config: 标题样式配置(TitleConfig dict)
subtitle_text: 字幕文本
subtitle_config: 字幕样式配置(SubtitleConfig dict)
Returns:
生成的 ASS 文件路径
"""
title_config = title_config or {}
subtitle_config = subtitle_config or {}
title_enabled = title_config.get("enabled", True) and bool(title_text.strip())
subtitle_enabled = subtitle_config.get("enabled", True) and bool(subtitle_text.strip())
if not title_enabled and not subtitle_enabled:
# 没有字幕,生成空文件(仍返回路径,调用方自行判断是否使用)
output_path.write_text("", encoding="utf-8")
return output_path
styles: list[str] = []
events: list[str] = []
# ── Title 样式与事件 ──────────────────────────────────────────────────
if title_enabled:
title_color = _hex_to_ass_color(title_config.get("color", "#ffffff"))
title_stroke = title_config.get("stroke", {}) or {}
title_shadow = title_config.get("shadow", {}) or {}
stroke_color = _hex_to_ass_color(title_stroke.get("color", "#000000"))
stroke_width = float(title_stroke.get("width", 1)) if title_stroke.get("enabled", False) else 0.0
shadow_blur = float(title_shadow.get("blur", 4)) if title_shadow.get("enabled", False) else 0.0
shadow_offset = (
title_shadow.get("offset_x", 2) if title_shadow.get("enabled", False) else 0,
title_shadow.get("offset_y", 2) if title_shadow.get("enabled", False) else 0,
)
title_alignment = _position_to_ass_alignment(title_config.get("position", "top"))
styles.append(
_build_ass_style(
"TitleStyle",
font_name=title_config.get("font", "思源黑体"),
font_size=int(title_config.get("size", 48)),
primary_color=title_color,
outline_color=stroke_color,
outline_width=stroke_width,
shadow_blur=shadow_blur,
shadow_offset=shadow_offset,
bold=bool(title_config.get("bold", True)),
italic=bool(title_config.get("italic", False)),
alignment=title_alignment,
margin_v=TITLE_MARGIN_TOP,
margin_l=TITLE_MARGIN_SIDE,
margin_r=TITLE_MARGIN_SIDE,
)
)
# 转义 ASS 特殊字符
safe_title_text = _escape_ass_text(title_text)
events.append(
"Dialogue: 0,0:00:00.00," f"{_format_ass_time(video_duration)}," "TitleStyle,,0,0,0,," f"{safe_title_text}"
)
# ── Subtitle 样式与事件 ───────────────────────────────────────────────
if subtitle_enabled:
sub_color = _hex_to_ass_color(subtitle_config.get("color", "#ffffff"))
sub_alignment = _position_to_ass_alignment(subtitle_config.get("position", "bottom"))
styles.append(
_build_ass_style(
"SubtitleStyle",
font_name=subtitle_config.get("font", "思源黑体"),
font_size=int(subtitle_config.get("size", 24)),
primary_color=sub_color,
outline_color="&H00000000",
outline_width=1.0,
shadow_blur=0.0,
shadow_offset=(0, 0),
bold=False,
italic=False,
alignment=sub_alignment,
margin_v=TITLE_MARGIN_BOTTOM,
margin_l=TITLE_MARGIN_SIDE,
margin_r=TITLE_MARGIN_SIDE,
)
)
safe_subtitle_text = _escape_ass_text(subtitle_text)
events.append(
"Dialogue: 0,0:00:00.00,"
f"{_format_ass_time(video_duration)},"
"SubtitleStyle,,0,0,0,,"
f"{safe_subtitle_text}"
)
# ── 组装 ASS 文件 ─────────────────────────────────────────────────────
ass_content = f"""[Script Info]
ScriptType: v4.00+
PlayResX: {video_width}
PlayResY: {video_height}
ScaledBorderAndShadow: yes
WrapStyle: 2
Encoding: UTF-8
[V4+ Styles]
Format: Name, Fontname, Fontsize, PrimaryColour, SecondaryColour, OutlineColour, BackColour, Bold, Italic, Underline, StrikeOut, ScaleX, ScaleY, Spacing, Angle, BorderStyle, Outline, Shadow, Alignment, MarginL, MarginR, MarginV, Encoding
{chr(10).join(styles)}
[Events]
Format: Layer, Start, End, Style, Name, MarginL, MarginR, MarginV, Effect, Text
{chr(10).join(events)}
"""
output_path.parent.mkdir(parents=True, exist_ok=True)
output_path.write_text(ass_content, encoding="utf-8")
return output_path
def _escape_ass_text(text: str) -> str:
r"""转义 ASS 文本中的特殊字符。
ASS 中换行用 \N(硬换行)或 \n(软换行),
大括号 {} 用于覆盖样式,需要转义。
"""
# 将实际换行转为 ASS 硬换行
text = text.replace("\r\n", "\\N").replace("\n", "\\N").replace("\r", "\\N")
# 转义大括号(ASS 用它做样式覆盖标签)
text = text.replace("{", "(").replace("}", ")")
return text
def _format_ass_time(seconds: float) -> str:
"""将秒数格式化为 ASS 时间格式 H:MM:SS.cc。"""
hours = int(seconds // 3600)
minutes = int((seconds % 3600) // 60)
secs = seconds % 60
return f"{hours}:{minutes:02d}:{secs:05.2f}"
def _resolve_layer_role(clip_type: str, config: dict[str, Any]) -> str:
"""根据 clip_type 和 config.role 确定图层角色。
@@ -165,7 +405,15 @@ class UnifiedRenderService:
self.transition_duration = transition_duration
def render(self) -> RenderResult:
"""执行渲染,返回 RenderResult。
"""执行渲染,返回 RenderResult.
优化路径:
- 单图层单 clip → 直通模式(-vf),性能最优
- 其他情况 → 完整 filter_complex 渲染
字幕渲染流程:
1. 视频主渲染(直通或完整链路)
2. 如有 title/subtitle,叠加 ASS 字幕
Raises:
ValueError: 没有可渲染的片段时抛出
@@ -178,14 +426,22 @@ class UnifiedRenderService:
# 2. 分组为 RenderLayers
layers = self._group_clips_into_layers(resolved)
# 3. 构建 filter_complex
# 3. 计算视频总时长(用于字幕显示时长)
video_duration = self._estimate_total_duration(layers)
# 4. 生成 ASS 字幕文件(如果有 title/subtitle 配置)
ass_path = self._maybe_generate_ass(video_duration)
output_path = self.work_dir / f"rendered_{self.plan.id}.mp4"
filter_complex, input_args = self._build_filter_complex(layers)
# 4. 执行 FFmpeg
self._execute_ffmpeg(filter_complex, input_args, output_path)
# 5. 视频主渲染
if self._can_use_pass_through(layers):
self._render_pass_through(layers, output_path, ass_path=ass_path)
else:
filter_complex, input_args = self._build_filter_complex(layers, ass_path=ass_path)
self._execute_ffmpeg(filter_complex, input_args, output_path)
# 5. 探测输出
# 6. 探测输出
duration, file_size, width, height = self._probe_output(output_path)
return RenderResult(
@@ -196,6 +452,192 @@ class UnifiedRenderService:
height=height,
)
def _estimate_total_duration(self, layers: list[RenderLayer]) -> float:
"""估算视频总时长(用于字幕等需要)。
取主图层(main/broll/background)的总时长,转场重叠按 transition_duration 估算。
"""
# 找主图层(第一个有视频内容的图层)
main_layer = None
for role in ("main", "broll", "background"):
for layer in layers:
if layer.role == role:
main_layer = layer
break
if main_layer:
break
if not main_layer or not main_layer.clips:
return 0.0
total = sum(
(
min(c.duration, c.actual_duration)
if c.duration > 0 and c.actual_duration > 0
else (c.duration if c.duration > 0 else c.actual_duration)
)
for c in main_layer.clips
)
# 减去转场重叠时间(粗略估算)
n_clips = len(main_layer.clips)
if n_clips > 1:
total -= (n_clips - 1) * self.transition_duration
return max(0.1, total)
def _maybe_generate_ass(self, video_duration: float) -> Path | None:
"""根据 plan.config 生成 ASS 字幕文件。
Returns:
ASS 文件路径,没有字幕时返回 None
"""
config = self.plan.config or {}
title_cfg = config.get("title", {}) or {}
subtitle_cfg = config.get("subtitle", {}) or {}
title_enabled = title_cfg.get("enabled", True)
subtitle_enabled = subtitle_cfg.get("enabled", True)
title_text = title_cfg.get("text", "") or ""
subtitle_text = subtitle_cfg.get("text", "") or ""
has_title = title_enabled and bool(title_text.strip())
has_subtitle = subtitle_enabled and bool(subtitle_text.strip())
if not has_title and not has_subtitle:
return None
ass_path = self.work_dir / f"subtitles_{self.plan.id}.ass"
generate_ass_subtitles(
ass_path,
video_width=self.output_width,
video_height=self.output_height,
video_duration=video_duration,
title_text=title_text,
title_config=title_cfg,
subtitle_text=subtitle_text,
subtitle_config=subtitle_cfg,
)
logger.info(
"生成字幕: plan_id=%s title=%s subtitle=%s ass=%s",
self.plan.id,
has_title,
has_subtitle,
ass_path,
)
return ass_path
def _can_use_pass_through(self, layers: list[RenderLayer]) -> bool:
"""判断是否可以走直通优化路径。
条件:
1. 只有 1 个图层
2. 该图层是视频图层(main/broll/background),不是 overlay/corner_voice/audio
3. 该图层只有 1 个 clip(无转场需求)
"""
if len(layers) != 1:
return False
layer = layers[0]
if layer.role not in ("main", "broll", "background"):
return False
if len(layer.clips) != 1:
return False
return True
def _render_pass_through(
self, layers: list[RenderLayer], output_path: Path, *, ass_path: Path | None = None
) -> None:
"""单图层单 clip 直通渲染(使用 -vf 而非 -filter_complex)。
性能优化:避免 filter_complex 的解析和调度开销,
对于一镜到底场景性能提升 ~30%,接近链路A水平。
Args:
layers: 图层列表(只有1个图层1个clip)
output_path: 输出文件路径
ass_path: ASS 字幕文件路径,有则叠加字幕
"""
clip = layers[0].clips[0]
role = layers[0].role
# 构建滤镜链(与 _build_filter_complex 中预处理逻辑一致)
filters: list[str] = []
# trim
effective_duration = 0.0
if clip.duration > 0:
effective_duration = min(clip.duration, clip.actual_duration) if clip.actual_duration > 0 else clip.duration
elif clip.actual_duration > 0:
effective_duration = clip.actual_duration
if effective_duration > 0:
filters.append(f"trim=duration={effective_duration}")
filters.append("setpts=PTS-STARTPTS")
# scale + crop(铺满裁剪)
if role in ("overlay", "corner_voice"):
pip_w = int(self.output_width * _PIP_SCALE)
pip_h = int(self.output_height * _PIP_SCALE)
filters.append(f"scale={pip_w}:{pip_h}")
else:
# main / broll / background: 铺满裁剪
filters.append(f"scale={self.output_width}:{self.output_height}" ":force_original_aspect_ratio=increase")
filters.append(f"crop={self.output_width}:{self.output_height}")
filters.append("setpts=PTS-STARTPTS")
filters.append(f"fps={self.output_fps}")
filters.append("format=yuv420p")
# 字幕叠加
if ass_path is not None:
# ASS 文件路径需要转义:Windows 反斜杠转正斜杠,冒号转义
ass_filter_path = str(ass_path).replace("\\", "/").replace(":", "\\:")
filters.append(f"subtitles='{ass_filter_path}'")
vf_str = ",".join(filters)
command = [
FFMPEG_BIN,
"-y",
"-i",
str(clip.local_path),
"-vf",
vf_str,
"-c:v",
"libx264",
"-crf",
"23",
"-preset",
"medium",
"-pix_fmt",
"yuv420p",
"-movflags",
"+faststart",
"-an", # 直通模式暂不处理音频,音频统一在后续混音阶段处理
str(output_path),
]
logger.info(
"直通渲染: plan_id=%s clip=%s role=%s duration=%.2fs",
self.plan.id,
clip.clip_id,
role,
effective_duration,
)
try:
run_ffmpeg(command)
except subprocess.CalledProcessError as e:
logger.error(
"直通渲染失败: plan_id=%s clip=%s exit_code=%d\nvf=%s",
self.plan.id,
clip.clip_id,
e.returncode,
vf_str[:2000],
)
raise
# ── 内部方法 ──────────────────────────────────────────────────────────────
def _resolve_clips(self) -> list[ResolvedClip]:
@@ -277,9 +719,15 @@ class UnifiedRenderService:
layers = sorted(layer_map.values(), key=lambda lyr: lyr.z_index)
return layers
def _build_filter_complex(self, layers: list[RenderLayer]) -> tuple[str, list[str]]:
def _build_filter_complex(
self, layers: list[RenderLayer], *, ass_path: Path | None = None
) -> tuple[str, list[str]]:
"""构建 FFmpeg filter_complex 字符串和输入参数列表。
Args:
layers: 图层列表
ass_path: ASS 字幕文件路径,有则在最后叠加字幕
Returns:
(filter_complex_str, input_args_list)
input_args_list 是 ["-i", path1, "-i", path2, ...] 格式
@@ -335,11 +783,12 @@ class UnifiedRenderService:
)
filters.append(f"crop={self.output_width}:{self.output_height}")
else:
# main / broll: scale + pad 保持宽高比
# main / broll: 铺满裁剪(scale to cover + center crop)
# 对齐链路A编辑器合成行为,与主流短视频平台一致
filters.append(
f"scale={self.output_width}:{self.output_height}" ":force_original_aspect_ratio=decrease"
f"scale={self.output_width}:{self.output_height}" ":force_original_aspect_ratio=increase"
)
filters.append(f"pad={self.output_width}:{self.output_height}" ":(ow-iw)/2:(oh-ih)/2:black")
filters.append(f"crop={self.output_width}:{self.output_height}")
filters.append("setpts=PTS-STARTPTS")
filters.append(f"fps={self.output_fps}")
@@ -421,7 +870,12 @@ class UnifiedRenderService:
filter_parts.append(f"[{final_video_label}][{overlay_label}]" f"overlay={x}:{y}[{combined_label}]")
final_video_label = combined_label
filter_parts.append(f"[{final_video_label}]format=yuv420p[final_video]")
# 叠加字幕(如有)+ 最终像素格式
if ass_path is not None:
ass_filter_path = str(ass_path).replace("\\", "/").replace(":", "\\:")
filter_parts.append(f"[{final_video_label}]subtitles='{ass_filter_path}',format=yuv420p[final_video]")
else:
filter_parts.append(f"[{final_video_label}]format=yuv420p[final_video]")
filter_complex = ";".join(filter_parts)
return filter_complex, input_args