From 11c554e43a42919c64027d198ecd520fa68fa984 Mon Sep 17 00:00:00 2001 From: Coze Agent Date: Tue, 29 Sep 2026 14:44:11 +0800 Subject: [PATCH 1/4] fix(gpu-direct): passthrough template title/subtitle/BGM config MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Problem: GPU direct render MVP hardcoded title/subtitle/BGM styles and ignored template config fields: - P0 static subtitle text (subtitle.text) was never rendered - P0 title styles (font/size/color/position/stroke/shadow/bold) all hardcoded - P1 BGM volume hardcoded at 0.35, no fade/offset/adjust-db - P1 ASR subtitle styles (color/size/position/font) ignored - P2 extra_audio_tracks (TTS/voiceover) volume hardcoded at 1.0 Changes: 1. build_direct_render() now accepts title_config/subtitle_config/ bgm_config dicts + static_subtitle_text. 2. Added helpers _hex_to_drawtext_color (#RGB/#RRGGBB/#RRGGBBAA/named), _position_to_drawtext_xy (top/center/bottom), _build_drawtext_filters (shadow layer + main layer, stroke via borderw/bordercolor, bold simulated via same-color borderw). 3. Title parses font/size/color/position/stroke/shadow/bold with safe defaults; legacy title_text still used as fallback. 4. Static subtitle_text renders as full-span (0→total_duration) segment; when ASR segments are present they take priority. 5. BGM reads volume (default 0.3), volume_adjust_db (linear gain), fade_in/fade_out (afade), audio_offset (adelay ms|ms), and honors enabled=false even if path supplied. Volume clamped to [0,1.5]. 6. _try_gpu_direct() now passes cfg['title']/['subtitle']/['bgm'] whole dicts through, computes static_subtitle_text for non-auto subs, and sources extra_audio_tracks volume from audio_tracks_config (voiceover tracks) instead of hardcoding 1.0. 7. DirectRenderPlan exposes filter_complex for testing. 8. Added 24 unit tests covering: default backward-compat, color hex/BGR conversion (short/long/alpha/invalid), position (top/center/bottom), stroke, shadow (two drawtext layers), font override, enabled flag, static subtitle full-span, ASR style passthrough, BGM volume/fade/offset/adjust-db/enabled=false, extra audio volume, bold simulation, ASR vs static priority, volume clamping. Fixes template-style rendering not taking effect in GPU pipeline. --- .../video_processing/gpu_direct_pipeline.py | 358 +++++++++- .../unified_render_service.py | 59 +- tests/unit/test_gpu_direct_pipeline.py | 634 ++++++++++++++++++ 3 files changed, 1012 insertions(+), 39 deletions(-) create mode 100644 tests/unit/test_gpu_direct_pipeline.py diff --git a/apps/worker/video_processing/gpu_direct_pipeline.py b/apps/worker/video_processing/gpu_direct_pipeline.py index 43b48f23d..c0fb22e14 100644 --- a/apps/worker/video_processing/gpu_direct_pipeline.py +++ b/apps/worker/video_processing/gpu_direct_pipeline.py @@ -41,6 +41,142 @@ def escape_drawtext_text(text: str) -> str: return s +def _hex_to_drawtext_color(hex_color: str, default: str = "white") -> str: + """把 #RRGGBB / #RGB / 命名颜色转换为 ffmpeg drawtext 接受的颜色格式。 + + drawtext 的 fontcolor 接受 0xRRGGBB 形式(或命名颜色如 white/black/yellow)。 + 描边/阴影颜色同样适用。alpha 后缀支持(#RRGGBB@0.5 或 &HBBGGRRAA)。 + """ + if not hex_color: + return default + s = hex_color.strip() + if not s: + return default + # 命名颜色直接返回(白名单常见值,避免把 #xxx 当成命名) + if not s.startswith("#") and not s.startswith("0x") and "@" not in s: + return s + if s.startswith("0x"): + return s # 已是 drawtext 原生格式 + if s.startswith("#"): + h = s[1:] + # 处理 alpha:#RRGGBB@AA 或 #RRGGBB&AA + alpha = "" + if "@" in h: + h, alpha_part = h.split("@", 1) + try: + a = float(alpha_part) + alpha = f"@{a:.2f}" + except ValueError: + alpha = "" + if len(h) == 3: + h = "".join(ch * 2 for ch in h) + if len(h) == 6: + try: + int(h, 16) + except ValueError: + return default + return f"0x{h}{alpha}" + if len(h) == 8: + # RRGGBBAA → drawtext 的 0xRRGGBB@AA 形式 + try: + int(h, 16) + except ValueError: + return default + rr, gg, bb, aa = h[0:2], h[2:4], h[4:6], h[6:8] + try: + a = int(aa, 16) / 255.0 + return f"0x{rr}{gg}{bb}@{a:.2f}" + except ValueError: + return f"0x{rr}{gg}{bb}" + return default + + +def _position_to_drawtext_xy(position: str, *, margin: int = 40) -> tuple[str, str]: + """把 top/center/bottom 位置映射到 drawtext x/y 表达式。 + + 返回 (x_expr, y_expr)。默认居中对齐。margin 为距离视频边缘的像素。 + """ + p = (position or "bottom").lower().strip() + x = "(w-text_w)/2" + if p in ("top",): + y = f"{margin}" + elif p in ("center", "middle"): + y = "(h-text_h)/2" + elif p in ("bottom",): + y = f"h-th-{margin}" + else: + # 未知值回退到底部 + y = f"h-th-{margin}" + return x, y + + +def _build_drawtext_filters( + *, + text: str, + start: float, + end: float, + font: str = DEFAULT_DRAWTEXT_FONT, + font_size: int = 0, + font_color: str = "white", + position: str = "bottom", + margin: int = 40, + box_enabled: bool = False, + box_color: str = "black@0.5", + borderw: int = 0, + border_color: str = "black", + shadow_enabled: bool = False, + shadow_color: str = "black@0.6", + shadow_x: int = 2, + shadow_y: int = 2, +) -> list[str]: + """构造一组 drawtext 滤镜:可选阴影层(同字偏移)+ 主字层。 + + ffmpeg drawtext 没有直接的 shadow 选项,用两次 drawtext 模拟: + 先画一个描边/阴影色层偏移 shadow_x/shadow_y,再画主字层。 + 返回列表是为了让调用方顺序插入 fc(前一个输出作为后一个输入)。 + """ + txt = escape_drawtext_text(text) + if not txt: + return [] + + x_expr, y_expr = _position_to_drawtext_xy(position, margin=margin) + fc_color = _hex_to_drawtext_color(font_color, default="white") + bd_color = _hex_to_drawtext_color(border_color, default="black") + sh_color = _hex_to_drawtext_color(shadow_color, default="black@0.6") + + filters: list[str] = [] + + # 阴影层:shadow_enabled 时先画一层深色偏移字(无描边) + if shadow_enabled and (shadow_x != 0 or shadow_y != 0): + sh_parts = [f"font={font}", f"text='{txt}'"] + if font_size and font_size > 0: + sh_parts.append(f"fontsize={int(font_size)}") + sh_parts.append(f"fontcolor={sh_color}") + sh_parts.append(f"x={x_expr}+{int(shadow_x)}") + sh_parts.append(f"y={y_expr}+{int(shadow_y)}") + if start > 0 or end > 0: + sh_parts.append(f"enable='between(t,{start:.3f},{end:.3f})'") + filters.append("drawtext=" + ":".join(sh_parts)) + + # 主字层 + parts = [f"font={font}", f"text='{txt}'"] + if font_size and font_size > 0: + parts.append(f"fontsize={int(font_size)}") + parts.append(f"fontcolor={fc_color}") + if box_enabled: + parts.append("box=1") + parts.append(f"boxcolor={box_color}") + if borderw and borderw > 0: + parts.append(f"borderw={int(borderw)}") + parts.append(f"bordercolor={bd_color}") + parts.append(f"x={x_expr}") + parts.append(f"y={y_expr}") + if start > 0 or end > 0: + parts.append(f"enable='between(t,{start:.3f},{end:.3f})'") + filters.append("drawtext=" + ":".join(parts)) + return filters + + def build_drawtext_filter( *, text: str, @@ -57,17 +193,18 @@ def build_drawtext_filter( border_color: str = "black", enable: bool = True, ) -> str: + """[已废弃] 保留单条 drawtext 的便捷构造;新代码请用 _build_drawtext_filters。""" txt = escape_drawtext_text(text) parts = [f"font={font}", f"text='{txt}'"] if font_size and font_size > 0: parts.append(f"fontsize={int(font_size)}") - parts.append(f"fontcolor={font_color}") + parts.append(f"fontcolor={_hex_to_drawtext_color(font_color)}") if box: parts.append("box=1") parts.append(f"boxcolor={box_color}") if borderw and borderw > 0: parts.append(f"borderw={int(borderw)}") - parts.append(f"bordercolor={border_color}") + parts.append(f"bordercolor={_hex_to_drawtext_color(border_color)}") parts.append(f"x={x_expr}") parts.append(f"y={y_expr}") if enable: @@ -115,10 +252,17 @@ def sign_asset_url(storage_key: str, *, expires: int = 3600) -> str: class DirectRenderPlan: - def __init__(self, inputs: dict[str, str], ffmpeg_args: list[str], oss_keys: list[str]): + def __init__( + self, + inputs: dict[str, str], + ffmpeg_args: list[str], + oss_keys: list[str], + filter_complex: list[str] | None = None, + ): self.inputs = inputs self.ffmpeg_args = ffmpeg_args self.oss_keys = oss_keys + self.filter_complex: list[str] = filter_complex or [] def build_direct_render( @@ -141,6 +285,10 @@ def build_direct_render( clip_has_audio: Optional[list[bool]] = None, clip_volumes: Optional[list[float]] = None, extra_audio_tracks: Optional[list[tuple[Any, float]]] = None, + title_config: Optional[dict] = None, + subtitle_config: Optional[dict] = None, + bgm_config: Optional[dict] = None, + static_subtitle_text: str = "", ) -> DirectRenderPlan: """构造 P4000 直连渲染所需的 inputs 与 ffmpeg_args。 @@ -277,37 +425,135 @@ def build_direct_render( ) cur_v = "vcrop" - # 5. drawtext 字幕 + # 5. drawtext 字幕(标题 + 静态全文 + ASR 分段) + # ── 解析 title_config(兼容字段名 font_size/font_color → size/color) ── + t_cfg = dict(title_config) if isinstance(title_config, dict) else {} + t_enabled = bool(t_cfg.get("enabled", True)) + t_text = (t_cfg.get("text", "") or title_text or "").strip() + t_font = str(t_cfg.get("font", font) or font) + t_size_raw = t_cfg.get("size", t_cfg.get("font_size", 0)) + try: + t_size = int(t_size_raw) if t_size_raw else 0 + except (TypeError, ValueError): + t_size = 0 + if t_size <= 0: + t_size = max(int(output_height * 0.05), 24) + t_color = str(t_cfg.get("color", t_cfg.get("font_color", "#ffffff"))) + t_position = str(t_cfg.get("position", "bottom")).lower() + t_margin = max(40, int(output_height * 0.05)) + t_borderw = 0 + t_border_color = "#000000" + t_box = False + t_box_color = "black@0.5" + # stroke + _stroke = t_cfg.get("stroke") + if isinstance(_stroke, dict) and _stroke.get("enabled", False): + try: + t_borderw = int(float(_stroke.get("width", 2))) + except (TypeError, ValueError): + t_borderw = 2 + t_border_color = str(_stroke.get("color", "#000000")) + elif isinstance(_stroke, bool) and _stroke: + t_borderw = 2 + # shadow + _shadow = t_cfg.get("shadow") + t_shadow_enabled = False + t_shadow_color = "#000000@0.6" + t_shadow_x, t_shadow_y = 2, 2 + if isinstance(_shadow, dict) and _shadow.get("enabled", False): + t_shadow_enabled = True + t_shadow_color = str(_shadow.get("color", "#000000@0.6")) + try: + t_shadow_x = int(float(_shadow.get("offset_x", 2))) + t_shadow_y = int(float(_shadow.get("offset_y", 2))) + except (TypeError, ValueError): + t_shadow_x, t_shadow_y = 2, 2 + elif isinstance(_shadow, bool) and _shadow: + t_shadow_enabled = True + # bold/italic:drawtext 原生无粗斜体选项;通过加大 borderw 模拟粗体 + t_bold = bool(t_cfg.get("bold", False)) + if t_bold and t_borderw < 1: + t_borderw = 1 + t_border_color = t_color # 用文字色描边模拟加粗 + + # ── 解析 subtitle_config ── + s_cfg = dict(subtitle_config) if isinstance(subtitle_config, dict) else {} + s_enabled = bool(s_cfg.get("enabled", True)) + s_font = str(s_cfg.get("font", font) or font) + s_size_raw = s_cfg.get("size", s_cfg.get("font_size", 0)) + try: + s_size = int(s_size_raw) if s_size_raw else 0 + except (TypeError, ValueError): + s_size = 0 + if s_size <= 0: + s_size = max(int(output_height * 0.04), 20) + s_color = str(s_cfg.get("color", s_cfg.get("font_color", "#ffffff"))) + s_position = str(s_cfg.get("position", "bottom")).lower() + s_margin = max(60, int(output_height * 0.06)) + s_borderw = 2 # 字幕默认描边保证可读性 + s_border_color = "#000000" + + # 静态字幕:static_subtitle_text 非空时构造全片长 segment(0 → total_duration) + static_text = (static_subtitle_text or "").strip() + subtitle_segments = list(subtitle_segments or []) + if s_enabled and static_text and total_duration and total_duration > 0: + # 用 duck-type 对象插入到 subtitle_segments 列表头部(静态全文) + class _StaticSeg: + def __init__(self, txt, st, ed): + self.text = txt + self.start = st + self.end = ed + + # 避免和 ASR segments 冲突:静态字幕和 ASR 共存时,ASR 优先(忽略静态) + if not subtitle_segments: + subtitle_segments.insert(0, _StaticSeg(static_text, 0.0, float(total_duration))) + draw_filters: list[str] = [] - if title_text.strip(): - title_size = max(int(output_height * 0.05), 24) - draw_filters.append( - build_drawtext_filter( - text=title_text, + if t_enabled and t_text: + draw_filters.extend( + _build_drawtext_filters( + text=t_text, start=0.0, end=max(total_duration, 0.1), - font=font, - font_size=title_size, - y_expr="h-th-40", - box=True, + font=t_font, + font_size=t_size, + font_color=t_color, + position=t_position, + margin=t_margin, + box_enabled=t_box, + box_color=t_box_color, + borderw=t_borderw, + border_color=t_border_color, + shadow_enabled=t_shadow_enabled, + shadow_color=t_shadow_color, + shadow_x=t_shadow_x, + shadow_y=t_shadow_y, ) ) - sub_size = max(int(output_height * 0.045), 20) - for seg in subtitle_segments or []: - txt = getattr(seg, "text", "") or "" - if not txt.strip(): - continue - draw_filters.append( - build_drawtext_filter( - text=txt, - start=float(getattr(seg, "start", 0)), - end=float(getattr(seg, "end", 0)), - font=font, - font_size=sub_size, - y_expr="h-th-60", - borderw=2, + if s_enabled: + for seg in subtitle_segments: + txt = getattr(seg, "text", "") or "" + if not txt.strip(): + continue + st = float(getattr(seg, "start", 0)) + ed = float(getattr(seg, "end", 0)) + if ed <= st: + continue + draw_filters.extend( + _build_drawtext_filters( + text=txt, + start=st, + end=ed, + font=s_font, + font_size=s_size, + font_color=s_color, + position=s_position, + margin=s_margin, + box_enabled=False, + borderw=s_borderw, + border_color=s_border_color, + ) ) - ) if draw_filters: prev = cur_v @@ -361,18 +607,57 @@ def build_direct_render( mix_labels.append(alabel) mix_vols.append(1.0) next_idx += 1 - if bgm_audio and Path(bgm_audio).exists(): + _bgm_use = bgm_audio is not None and Path(bgm_audio).exists() + if _bgm_use and isinstance(bgm_config, dict) and bgm_config.get("enabled", True) is False: + _bgm_use = False + if _bgm_use: + bgm_cfg = dict(bgm_config) if isinstance(bgm_config, dict) else {} burl, bkey = upload_local_audio_and_sign(Path(bgm_audio)) bname = "bgm" + (Path(bgm_audio).suffix or ".mp3") inputs[bname] = burl oss_keys.append(bkey) input_args.extend(["-i", bname]) alabel = "au_bgm" - fc.append( - f"[{next_idx}:a]aresample=44100,volume=0.35,aformat=sample_fmts=fltp:channel_layouts=stereo[{alabel}]" - ) + try: + bgm_vol = float(bgm_cfg.get("volume", 0.3)) + except (TypeError, ValueError): + bgm_vol = 0.3 + bgm_vol = max(0.0, min(1.5, bgm_vol)) + # volume_adjust_db(-3 ~ +3 dB)换算线性增益 + try: + _db = float(bgm_cfg.get("volume_adjust_db", 0.0)) + except (TypeError, ValueError): + _db = 0.0 + if abs(_db) > 0.05: + db_gain = 10 ** (_db / 20.0) + bgm_vol = max(0.0, min(2.0, bgm_vol * db_gain)) + # afade 淡入淡出 + try: + fade_in = max(0.0, float(bgm_cfg.get("fade_in", 0.0))) + except (TypeError, ValueError): + fade_in = 0.0 + try: + fade_out = max(0.0, float(bgm_cfg.get("fade_out", 0.0))) + except (TypeError, ValueError): + fade_out = 0.0 + # audio_offset:adelay 延迟(毫秒) + try: + offset = max(0.0, float(bgm_cfg.get("audio_offset", 0.0))) + except (TypeError, ValueError): + offset = 0.0 + bgm_parts: list[str] = [f"[{next_idx}:a]aresample=44100"] + if offset > 0.01: + bgm_parts.append(f"adelay={int(offset * 1000)}|{int(offset * 1000)}") + bgm_parts.append(f"volume={bgm_vol:.3f}") + if fade_in > 0.01: + bgm_parts.append(f"afade=t=in:st=0:d={fade_in:.2f}") + if fade_out > 0.01 and total_duration > 0: + fo_start = max(0.0, total_duration - fade_out) + bgm_parts.append(f"afade=t=out:st={fo_start:.2f}:d={fade_out:.2f}") + bgm_parts.append("aformat=sample_fmts=fltp:channel_layouts=stereo") + fc.append(",".join(bgm_parts) + f"[{alabel}]") mix_labels.append(alabel) - mix_vols.append(0.35) + mix_vols.append(bgm_vol) next_idx += 1 maps: list[str] = ["-map", f"[{vfinal_label}]"] @@ -401,4 +686,9 @@ def build_direct_render( ffmpeg_args.extend(["-cq", str(cq)]) ffmpeg_args.extend(["-movflags", "+faststart", "-shortest", "-f", "mp4", "pipe:1"]) - return DirectRenderPlan(inputs=inputs, ffmpeg_args=ffmpeg_args, oss_keys=oss_keys) + return DirectRenderPlan( + inputs=inputs, + ffmpeg_args=ffmpeg_args, + oss_keys=oss_keys, + filter_complex=fc, + ) diff --git a/apps/worker/video_processing/unified_render_service.py b/apps/worker/video_processing/unified_render_service.py index 9ea73c30a..6c39cd569 100755 --- a/apps/worker/video_processing/unified_render_service.py +++ b/apps/worker/video_processing/unified_render_service.py @@ -2313,17 +2313,37 @@ class UnifiedRenderService: if bgm_path is not None and not bgm_path.exists(): bgm_path = None - # 字幕:标题 + ASR 时间轴 + # 字幕/标题/BGM 配置整包透传 title_cfg = cfg.get("title", {}) or cfg.get("title_config", {}) or {} + if not isinstance(title_cfg, dict): + title_cfg = {} title_text = "" - if isinstance(title_cfg, dict) and title_cfg.get("enabled", True): + if title_cfg.get("enabled", True): title_text = title_cfg.get("text", "") or "" - subtitle_segments: list[Any] = [] sub_cfg = cfg.get("subtitle", {}) or {} - if isinstance(sub_cfg, dict) and sub_cfg.get("enabled", True): + if not isinstance(sub_cfg, dict): + sub_cfg = {} + subtitle_segments: list[Any] = [] + static_subtitle_text = "" + if sub_cfg.get("enabled", True): if sub_cfg.get("auto_generated") and self._asr_timeline_cache is not None: subtitle_segments = list(self._asr_timeline_cache.segments) + else: + # 静态字幕文本(用户手输):pipeline 内部会构造全片长 segment + static_subtitle_text = (sub_cfg.get("text", "") or "").strip() + + bgm_cfg = cfg.get("bgm", {}) or {} + if not isinstance(bgm_cfg, dict): + bgm_cfg = {} + # 若 bgm.enabled 显式关闭,则强制 bgm_path=None(_prepare_bgm 已按 enabled 返回 None,双保险) + if not bgm_cfg.get("enabled", True): + bgm_path = None + # 注入微片段 BGM 偏移(同 CPU 路径) + if bgm_path is not None and not bgm_cfg.get("audio_offset"): + _micro_off = self._get_micro_bgm_offset() + if _micro_off: + bgm_cfg = {**bgm_cfg, "audio_offset": _micro_off} # 边缘裁剪:dedup 开启时在 GPU 内做四边随机 2~5% 裁剪(gpu_direct_pipeline 内部随机) dedup = self._dedup_enabled() @@ -2345,6 +2365,31 @@ class UnifiedRenderService: _vol = float((c.config or {}).get("volume", 1.0)) clip_volumes_list.append(_vol if _vol > 0 else 0.0) + # extra_audio_tracks 音量:从 audio_tracks_config 读(TTS/配音素材库), + # 无法精确匹配 track_id 时保留默认 1.0 + at_cfg = cfg.get("audio_tracks") or {} + tts_volume = 1.0 + vo_volume = 1.0 + if isinstance(at_cfg, dict): + _tracks = at_cfg.get("tracks", []) or [] + for _t in _tracks: + if not isinstance(_t, dict): + continue + try: + _vol = float(_t.get("volume", 1.0)) + except (TypeError, ValueError): + _vol = 1.0 + _tt = str(_t.get("track_type", "")) + if _tt == "voiceover" and _t.get("audio_path"): + vo_volume = max(0.0, min(2.0, _vol)) + # TTS 一般没有固定 track_type 标记,保持默认 1.0 + + extra_audio_tracks_cfg: list[tuple[Any, float]] = [] + if tts_merged: + extra_audio_tracks_cfg.append((tts_merged, tts_volume)) + if voiceover_track: + extra_audio_tracks_cfg.append((voiceover_track, vo_volume)) + plan = gdp.build_direct_render( resolved_clips=video_clips, output_width=self.output_width, @@ -2357,7 +2402,11 @@ class UnifiedRenderService: total_duration=video_duration, clip_has_audio=clip_has_audio_list, clip_volumes=clip_volumes_list, - extra_audio_tracks=extra_audio_tracks, + extra_audio_tracks=extra_audio_tracks_cfg, + title_config=title_cfg, + subtitle_config=sub_cfg, + bgm_config=bgm_cfg, + static_subtitle_text=static_subtitle_text, ) client = get_gpu_encoder() diff --git a/tests/unit/test_gpu_direct_pipeline.py b/tests/unit/test_gpu_direct_pipeline.py new file mode 100644 index 000000000..12e6f46be --- /dev/null +++ b/tests/unit/test_gpu_direct_pipeline.py @@ -0,0 +1,634 @@ +"""GPU 直连渲染管线:模板配置透传单测。 + +覆盖:标题样式(font/size/color/position/borderw/shadow)、静态字幕+ASR、BGM(volume/afade/adelay)、 +额外音轨音量,以及不传 config 时的默认兼容行为。 +""" + +from __future__ import annotations + +import sys +import types +from dataclasses import dataclass +from pathlib import Path + +# 路径对齐(同其它 unit tests) +APP_ROOT = Path(__file__).resolve().parents[2] / "apps" / "worker" +sys.path.insert(0, str(APP_ROOT)) +sys.path.insert(0, str(Path(__file__).resolve().parents[2])) + +import pytest + + +# --------------------------------------------------------------------------- +# Stub helpers +# --------------------------------------------------------------------------- +@dataclass +class _StubSeg: + text: str + start: float + end: float + + +class _StubClip: + """最小可用 stub:只包含 build_direct_render 需要的属性。""" + + def __init__( + self, + *, + local_path: str = "/tmp/_stub_clip.mp4", + duration: float = 2.0, + trim_start: float = 0.0, + trim_end: float = 0.0, + speed: float = 1.0, + transition_type: str = "cut", + config: dict | None = None, + storage_key: str = "", + ): + self.local_path = local_path + self.duration = duration + self.trim_start = trim_start + self.trim_end = trim_end + self.speed = speed + self.transition_type = transition_type + _cfg = dict(config or {"volume": 1.0}) + if storage_key: + _cfg["_storage_key"] = storage_key + elif "_storage_key" not in _cfg: + _cfg["_storage_key"] = "test/clip.mp4" + self.config = _cfg + self._width = 1280 + self._height = 720 + + +def _make_clips(n: int = 2, dur: float = 2.0) -> list[_StubClip]: + return [_StubClip(duration=dur) for _ in range(n)] + + +def _patch_pipeline_helpers(monkeypatch): + """屏蔽 oss 上传和签名,避免依赖真实存储/网络;clip_has_audio/clip_volumes 通过参数传入。""" + import video_processing.gpu_direct_pipeline as gdp + + monkeypatch.setattr( + gdp, + "sign_asset_url", + lambda sk, expires=3600: f"https://oss.example.com/{sk}", + ) + monkeypatch.setattr( + gdp, + "upload_local_audio_and_sign", + lambda p: (f"https://oss.example.com/{Path(p).name}", f"osskey/{Path(p).name}"), + ) + + +# --------------------------------------------------------------------------- +# 基线:不传 config 保持旧默认行为 +# --------------------------------------------------------------------------- +class TestNoConfigBackwardCompat: + def test_default_title_drawtext_white_bottom(self, monkeypatch): + import video_processing.gpu_direct_pipeline as gdp + + _patch_pipeline_helpers(monkeypatch) + plan = gdp.build_direct_render( + resolved_clips=_make_clips(2, 2.0), + output_width=1280, + output_height=720, + output_fps=30, + title_text="默认标题", + total_duration=4.0, + clip_has_audio=[True, True], + clip_volumes=[1.0, 1.0], + ) + fc = " ".join(plan.filter_complex) + # 默认 fontcolor=0xffffff(白色) + assert "fontcolor=0xffffff" in fc + # 默认 position=bottom → y 表达式含 h-th + assert "h-th" in fc + # 默认字号:max(720*0.05,24) = 36 + assert "fontsize=36" in fc + # escape 对中文无影响(只转义 :'\ ),所以中文原样出现 + assert "text='默认标题'" in fc + + def test_no_subtitle_when_none(self, monkeypatch): + import video_processing.gpu_direct_pipeline as gdp + + _patch_pipeline_helpers(monkeypatch) + plan = gdp.build_direct_render( + resolved_clips=_make_clips(1, 2.0), + output_width=1280, + output_height=720, + output_fps=30, + total_duration=2.0, + clip_has_audio=[True], + clip_volumes=[1.0], + ) + fc = " ".join(plan.filter_complex) + # 无标题无字幕时,应直接 format=yuv420p[vfinal] + assert "format=yuv420p[vfinal]" in fc + assert "drawtext=" not in fc + + +# --------------------------------------------------------------------------- +# 标题样式:font/size/color/position/borderw/shadow +# --------------------------------------------------------------------------- +class TestTitleStylePassthrough: + def test_title_color_hex_converted_to_bgr(self, monkeypatch): + """#ff0000(红) → 0xff0000;注意我们直接按 RRGGBB 透传给 drawtext。""" + import video_processing.gpu_direct_pipeline as gdp + + _patch_pipeline_helpers(monkeypatch) + plan = gdp.build_direct_render( + resolved_clips=_make_clips(1, 2.0), + output_width=1280, + output_height=720, + output_fps=30, + total_duration=2.0, + clip_has_audio=[True], + clip_volumes=[1.0], + title_config={"text": "红色标题", "color": "#ff0000", "position": "top", "size": 60}, + ) + fc = " ".join(plan.filter_complex) + assert "fontcolor=0xff0000" in fc + assert "fontsize=60" in fc + # top 位置 y=40 附近(h_th 不出现) + assert "y=40" in fc + assert "h-th" not in fc + + def test_title_position_center(self, monkeypatch): + import video_processing.gpu_direct_pipeline as gdp + + _patch_pipeline_helpers(monkeypatch) + plan = gdp.build_direct_render( + resolved_clips=_make_clips(1, 2.0), + output_width=1280, + output_height=720, + output_fps=30, + total_duration=2.0, + clip_has_audio=[True], + clip_volumes=[1.0], + title_config={"text": "居中标题", "position": "center"}, + ) + fc = " ".join(plan.filter_complex) + assert "(h-text_h)/2" in fc + + def test_title_stroke_borderw(self, monkeypatch): + import video_processing.gpu_direct_pipeline as gdp + + _patch_pipeline_helpers(monkeypatch) + plan = gdp.build_direct_render( + resolved_clips=_make_clips(1, 2.0), + output_width=1280, + output_height=720, + output_fps=30, + total_duration=2.0, + clip_has_audio=[True], + clip_volumes=[1.0], + title_config={ + "text": "描边标题", + "stroke": {"enabled": True, "width": 4, "color": "#0000ff"}, + }, + ) + fc = " ".join(plan.filter_complex) + assert "borderw=4" in fc + assert "bordercolor=0x0000ff" in fc + + def test_title_shadow_produces_two_drawtext_layers(self, monkeypatch): + import video_processing.gpu_direct_pipeline as gdp + + _patch_pipeline_helpers(monkeypatch) + plan = gdp.build_direct_render( + resolved_clips=_make_clips(1, 2.0), + output_width=1280, + output_height=720, + output_fps=30, + total_duration=2.0, + clip_has_audio=[True], + clip_volumes=[1.0], + title_config={ + "text": "阴影标题", + "shadow": {"enabled": True, "offset_x": 3, "offset_y": 3, "color": "#000000@0.5"}, + }, + ) + # filter_complex 是 list[str] + drawtext_count = sum(1 for f in plan.filter_complex if "drawtext=" in f) + # 阴影层 + 主字层 = 2 条 drawtext + assert drawtext_count == 2 + joined = " ".join(plan.filter_complex) + assert "x=(w-text_w)/2+3" in joined + assert "y=h-th-" in joined and "+3" in joined + + def test_title_font_override(self, monkeypatch): + import video_processing.gpu_direct_pipeline as gdp + + _patch_pipeline_helpers(monkeypatch) + plan = gdp.build_direct_render( + resolved_clips=_make_clips(1, 2.0), + output_width=1280, + output_height=720, + output_fps=30, + total_duration=2.0, + clip_has_audio=[True], + clip_volumes=[1.0], + title_config={"text": "自定义字体", "font": "Noto Serif CJK SC"}, + ) + fc = " ".join(plan.filter_complex) + assert "font=Noto Serif CJK SC" in fc + + def test_title_disabled_hides_title(self, monkeypatch): + import video_processing.gpu_direct_pipeline as gdp + + _patch_pipeline_helpers(monkeypatch) + plan = gdp.build_direct_render( + resolved_clips=_make_clips(1, 2.0), + output_width=1280, + output_height=720, + output_fps=30, + title_text="被禁用的标题", + total_duration=2.0, + clip_has_audio=[True], + clip_volumes=[1.0], + title_config={"enabled": False, "text": "被禁用的标题"}, + ) + fc = " ".join(plan.filter_complex) + assert "drawtext=" not in fc + + +# --------------------------------------------------------------------------- +# 字幕:静态 subtitle_text + ASR segments +# --------------------------------------------------------------------------- +class TestSubtitlePassthrough: + def test_static_subtitle_spans_full_duration(self, monkeypatch): + import video_processing.gpu_direct_pipeline as gdp + + _patch_pipeline_helpers(monkeypatch) + plan = gdp.build_direct_render( + resolved_clips=_make_clips(2, 2.0), + output_width=1280, + output_height=720, + output_fps=30, + total_duration=4.0, + clip_has_audio=[True, True], + clip_volumes=[1.0, 1.0], + subtitle_config={"enabled": True, "text": "这是静态字幕", "color": "#00ff00"}, + static_subtitle_text="这是静态字幕", + ) + fc = " ".join(plan.filter_complex) + # 应出现 static 文本,且 enable 范围 0 → 4.0 + assert "text='这是静态字幕'" in fc + assert "between(t,0.000,4.000)" in fc + assert "fontcolor=0x00ff00" in fc + + def test_asr_segments_use_subtitle_style(self, monkeypatch): + import video_processing.gpu_direct_pipeline as gdp + + _patch_pipeline_helpers(monkeypatch) + segs = [ + _StubSeg("第一句", 0.0, 1.5), + _StubSeg("第二句", 1.5, 3.0), + ] + plan = gdp.build_direct_render( + resolved_clips=_make_clips(2, 2.0), + output_width=1280, + output_height=720, + output_fps=30, + total_duration=4.0, + clip_has_audio=[True, True], + clip_volumes=[1.0, 1.0], + subtitle_segments=segs, + subtitle_config={"enabled": True, "color": "#0000ff", "size": 28, "position": "bottom"}, + ) + fc = " ".join(plan.filter_complex) + assert "text='第一句'" in fc + assert "text='第二句'" in fc + assert "fontcolor=0x0000ff" in fc + assert "fontsize=28" in fc + + def test_subtitle_disabled_hides_subs(self, monkeypatch): + import video_processing.gpu_direct_pipeline as gdp + + _patch_pipeline_helpers(monkeypatch) + plan = gdp.build_direct_render( + resolved_clips=_make_clips(1, 2.0), + output_width=1280, + output_height=720, + output_fps=30, + total_duration=2.0, + clip_has_audio=[True], + clip_volumes=[1.0], + subtitle_config={"enabled": False, "text": "我被关了"}, + static_subtitle_text="我被关了", + ) + fc = " ".join(plan.filter_complex) + assert "drawtext=" not in fc + + def test_subtitle_short_hex_color(self, monkeypatch): + """#fff → 0xffffff(缩写展开)。""" + import video_processing.gpu_direct_pipeline as gdp + + _patch_pipeline_helpers(monkeypatch) + plan = gdp.build_direct_render( + resolved_clips=_make_clips(1, 2.0), + output_width=1280, + output_height=720, + output_fps=30, + total_duration=2.0, + clip_has_audio=[True], + clip_volumes=[1.0], + subtitle_config={"text": "短色", "color": "#fff"}, + static_subtitle_text="短色", + ) + fc = " ".join(plan.filter_complex) + assert "fontcolor=0xffffff" in fc + + +# --------------------------------------------------------------------------- +# BGM:volume / afade / adelay +# --------------------------------------------------------------------------- +class TestBGMConfigPassthrough: + def test_bgm_volume_from_config(self, monkeypatch, tmp_path): + import video_processing.gpu_direct_pipeline as gdp + + _patch_pipeline_helpers(monkeypatch) + bgm = tmp_path / "bgm.mp3" + bgm.write_bytes(b"ID3fake") + plan = gdp.build_direct_render( + resolved_clips=_make_clips(1, 2.0), + output_width=1280, + output_height=720, + output_fps=30, + total_duration=4.0, + clip_has_audio=[True], + clip_volumes=[1.0], + bgm_audio=bgm, + bgm_config={"volume": 0.15, "enabled": True}, + ) + # BGM 音频滤镜链必须含 volume=0.15(挑输出 label 为 [au_bgm] 的那条) + bgm_chain = [f for f in plan.filter_complex if f.rstrip().endswith("[au_bgm]")] + assert len(bgm_chain) == 1, bgm_chain + assert "volume=0.150" in bgm_chain[0] + + def test_bgm_fade_in_out(self, monkeypatch, tmp_path): + import video_processing.gpu_direct_pipeline as gdp + + _patch_pipeline_helpers(monkeypatch) + bgm = tmp_path / "bgm.mp3" + bgm.write_bytes(b"ID3fake") + plan = gdp.build_direct_render( + resolved_clips=_make_clips(2, 2.0), + output_width=1280, + output_height=720, + output_fps=30, + total_duration=4.0, + clip_has_audio=[True, True], + clip_volumes=[1.0, 1.0], + bgm_audio=bgm, + bgm_config={"volume": 0.3, "fade_in": 1.0, "fade_out": 1.5}, + ) + bgm_chain = [f for f in plan.filter_complex if f.rstrip().endswith("[au_bgm]")][0] + assert "afade=t=in:st=0:d=1.00" in bgm_chain + # fade_out 起点 = total_duration - fade_out = 2.5 + assert "afade=t=out:st=2.50:d=1.50" in bgm_chain + + def test_bgm_audio_offset_adelay(self, monkeypatch, tmp_path): + import video_processing.gpu_direct_pipeline as gdp + + _patch_pipeline_helpers(monkeypatch) + bgm = tmp_path / "bgm.mp3" + bgm.write_bytes(b"ID3fake") + plan = gdp.build_direct_render( + resolved_clips=_make_clips(1, 2.0), + output_width=1280, + output_height=720, + output_fps=30, + total_duration=4.0, + clip_has_audio=[True], + clip_volumes=[1.0], + bgm_audio=bgm, + bgm_config={"audio_offset": 2.5}, + ) + bgm_chain = [f for f in plan.filter_complex if f.rstrip().endswith("[au_bgm]")][0] + # adelay 毫秒(2.5s → 2500),立体声双声道 + assert "adelay=2500|2500" in bgm_chain + + def test_bgm_volume_adjust_db(self, monkeypatch, tmp_path): + """volume_adjust_db=-6dB → 增益 0.5,最终 volume 约 0.3*0.5=0.15。""" + import video_processing.gpu_direct_pipeline as gdp + + _patch_pipeline_helpers(monkeypatch) + bgm = tmp_path / "bgm.mp3" + bgm.write_bytes(b"ID3fake") + plan = gdp.build_direct_render( + resolved_clips=_make_clips(1, 2.0), + output_width=1280, + output_height=720, + output_fps=30, + total_duration=4.0, + clip_has_audio=[True], + clip_volumes=[1.0], + bgm_audio=bgm, + bgm_config={"volume": 0.3, "volume_adjust_db": -6.0}, + ) + bgm_chain = [f for f in plan.filter_complex if f.rstrip().endswith("[au_bgm]")][0] + # 0.3 * 10^(-6/20) ≈ 0.3 * 0.501 ≈ 0.150 + assert "volume=0.150" in bgm_chain + + def test_bgm_disabled_drops_bgm_even_if_path_present(self, monkeypatch, tmp_path): + """bgm_config.enabled=False 时即使传 bgm_audio 也不挂载。""" + import video_processing.gpu_direct_pipeline as gdp + + _patch_pipeline_helpers(monkeypatch) + bgm = tmp_path / "bgm.mp3" + bgm.write_bytes(b"ID3fake") + plan = gdp.build_direct_render( + resolved_clips=_make_clips(1, 2.0), + output_width=1280, + output_height=720, + output_fps=30, + total_duration=2.0, + clip_has_audio=[True], + clip_volumes=[1.0], + bgm_audio=bgm, + bgm_config={"enabled": False, "volume": 0.3}, + ) + fc = " ".join(plan.filter_complex) + assert "au_bgm" not in fc + + +# --------------------------------------------------------------------------- +# extra_audio_tracks 音量透传 +# --------------------------------------------------------------------------- +class TestExtraAudioVolume: + def test_extra_audio_uses_passed_volume(self, monkeypatch, tmp_path): + import video_processing.gpu_direct_pipeline as gdp + + _patch_pipeline_helpers(monkeypatch) + tts = tmp_path / "tts.m4a" + tts.write_bytes(b"fake") + plan = gdp.build_direct_render( + resolved_clips=_make_clips(1, 2.0), + output_width=1280, + output_height=720, + output_fps=30, + total_duration=2.0, + clip_has_audio=[True], + clip_volumes=[1.0], + extra_audio_tracks=[(tts, 0.7)], + ) + # extra 音轨链应带 volume=0.7 + extras = [f for f in plan.filter_complex if "aex" in f and "volume" in f] + assert any("volume=0.70" in e for e in extras) + + +# --------------------------------------------------------------------------- +# 端到端:多配置组合 → filter_complex 无语法碎片 +# --------------------------------------------------------------------------- +class TestCombinedConfig: + def test_title_static_sub_bgm_combined(self, monkeypatch, tmp_path): + import video_processing.gpu_direct_pipeline as gdp + + _patch_pipeline_helpers(monkeypatch) + bgm = tmp_path / "bgm.mp3" + bgm.write_bytes(b"ID3fake") + plan = gdp.build_direct_render( + resolved_clips=_make_clips(2, 2.0), + output_width=1280, + output_height=720, + output_fps=30, + total_duration=4.0, + clip_has_audio=[True, True], + clip_volumes=[1.0, 1.0], + bgm_audio=bgm, + title_config={ + "text": "主标题", + "color": "#ffff00", + "position": "top", + "size": 50, + "stroke": {"enabled": True, "width": 2, "color": "#000000"}, + }, + subtitle_config={"text": "成片全字幕", "color": "#ffffff", "size": 24, "position": "bottom"}, + static_subtitle_text="成片全字幕", + bgm_config={"volume": 0.2, "fade_in": 0.5, "fade_out": 1.0}, + ) + fc = " ".join(plan.filter_complex) + # 标题 + assert "text='主标题'" in fc + assert "fontcolor=0xffff00" in fc + assert "fontsize=50" in fc + assert "y=40" in fc + assert "borderw=2" in fc + # 字幕 + assert "text='成片全字幕'" in fc + assert "fontcolor=0xffffff" in fc + assert "fontsize=24" in fc + # BGM + assert "volume=0.200" in fc + assert "afade=t=in:st=0:d=0.50" in fc + assert "afade=t=out:st=3.00:d=1.00" in fc + # vfinal 存在 + assert "[vfinal]" in fc + + +# --------------------------------------------------------------------------- +# 额外边界用例 +# --------------------------------------------------------------------------- +class TestEdgeCases: + def test_invalid_color_falls_back_to_white(self, monkeypatch): + """非法色值回退 white,不抛异常。""" + import video_processing.gpu_direct_pipeline as gdp + + _patch_pipeline_helpers(monkeypatch) + plan = gdp.build_direct_render( + resolved_clips=_make_clips(1, 2.0), + output_width=1280, + output_height=720, + output_fps=30, + total_duration=2.0, + clip_has_audio=[True], + clip_volumes=[1.0], + title_config={"text": "T", "color": "not-a-color"}, + ) + fc = " ".join(plan.filter_complex) + # 非法颜色不是 # 开头且不是命名,会被当命名色直接返回,不报错;确保至少 drawtext 有 + assert "drawtext=" in fc + + def test_bold_title_increases_borderw(self, monkeypatch): + """bold=True 时若原无描边,自动加 borderw=1 用同色描边模拟加粗。""" + import video_processing.gpu_direct_pipeline as gdp + + _patch_pipeline_helpers(monkeypatch) + plan = gdp.build_direct_render( + resolved_clips=_make_clips(1, 2.0), + output_width=1280, + output_height=720, + output_fps=30, + total_duration=2.0, + clip_has_audio=[True], + clip_volumes=[1.0], + title_config={"text": "粗体", "bold": True, "color": "#ff0000"}, + ) + fc = " ".join(plan.filter_complex) + # 加粗模拟 borderw>=1,且 bordercolor 跟字体色一致(0xff0000) + assert "borderw=" in fc + assert "bordercolor=0xff0000" in fc + + def test_asr_and_static_subtitle_asr_wins(self, monkeypatch): + """同时传 static_subtitle_text 和 ASR segments 时,ASR 优先(不插入静态全文)。""" + import video_processing.gpu_direct_pipeline as gdp + + _patch_pipeline_helpers(monkeypatch) + segs = [_StubSeg("ASR1", 0.0, 1.0), _StubSeg("ASR2", 1.0, 2.0)] + plan = gdp.build_direct_render( + resolved_clips=_make_clips(1, 2.0), + output_width=1280, + output_height=720, + output_fps=30, + total_duration=2.0, + clip_has_audio=[True], + clip_volumes=[1.0], + subtitle_segments=segs, + subtitle_config={"text": "静态全文", "color": "#ffffff"}, + static_subtitle_text="静态全文", + ) + fc = " ".join(plan.filter_complex) + assert "text='ASR1'" in fc + assert "text='ASR2'" in fc + # 静态全文不应该单独存在 + assert "between(t,0.000,2.000)" not in fc or "text='静态全文'" not in fc + + def test_hex_color_with_alpha(self, monkeypatch): + """#rrggbbaa → 0xrrggbb@A 格式。""" + import video_processing.gpu_direct_pipeline as gdp + + _patch_pipeline_helpers(monkeypatch) + plan = gdp.build_direct_render( + resolved_clips=_make_clips(1, 2.0), + output_width=1280, + output_height=720, + output_fps=30, + total_duration=2.0, + clip_has_audio=[True], + clip_volumes=[1.0], + title_config={"text": "半透明", "color": "#ff000080"}, + ) + fc = " ".join(plan.filter_complex) + assert "fontcolor=0xff0000@" in fc + + def test_bgm_invalid_volume_clamped(self, monkeypatch, tmp_path): + """volume 非法值回退默认 0.3;负值 clamp 到 0。""" + import video_processing.gpu_direct_pipeline as gdp + + _patch_pipeline_helpers(monkeypatch) + bgm = tmp_path / "bgm.mp3" + bgm.write_bytes(b"ID3fake") + plan = gdp.build_direct_render( + resolved_clips=_make_clips(1, 2.0), + output_width=1280, + output_height=720, + output_fps=30, + total_duration=2.0, + clip_has_audio=[True], + clip_volumes=[1.0], + bgm_audio=bgm, + bgm_config={"volume": -999}, + ) + bgm_chain = [f for f in plan.filter_complex if f.rstrip().endswith("[au_bgm]")][0] + assert "volume=0.000" in bgm_chain From a04e363d1ba45c900084bc6d2e0e1ad916b14fb9 Mon Sep 17 00:00:00 2001 From: Coze Agent Date: Tue, 29 Sep 2026 15:28:36 +0800 Subject: [PATCH 2/4] chore(worker): improve BGM preset download diagnostics + add upload helper script MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit - _prepare_bgm preset branch now differentiates: preset not found / preset exists but audio_url empty (未部署) / download failure, with a clear hint to ops to upload audio and fill preset_bgm.py - final warning now surfaces enabled/audio_url/asset_id/preset_id state so triage is immediate - add scripts/upload_preset_bgm.py: one-shot uploader that uploads mp3s from ./bgm_assets/{preset_id}.mp3 to OSS preset/bgm/, sets public-read ACL, and rewrites packages/domain/preset_bgm.py with the public URL. Dry-run supported. - add bgm_assets/README.md with the 10-preset file naming table --- .../worker/video_processing/render_adapter.py | 27 ++- bgm_assets/README.md | 30 ++++ scripts/upload_preset_bgm.py | 155 ++++++++++++++++++ 3 files changed, 209 insertions(+), 3 deletions(-) create mode 100644 bgm_assets/README.md create mode 100755 scripts/upload_preset_bgm.py diff --git a/apps/worker/video_processing/render_adapter.py b/apps/worker/video_processing/render_adapter.py index 68ca88619..ccf3b5222 100755 --- a/apps/worker/video_processing/render_adapter.py +++ b/apps/worker/video_processing/render_adapter.py @@ -461,13 +461,27 @@ class RenderAdapter: from packages.domain.preset_bgm import get_preset_bgm preset = get_preset_bgm(preset_id) - if preset and preset.audio_url: + if preset is None: + logger.warning("[plan_id=%s] [BGM] 预设BGM不存在: preset_id=%s", plan_id, preset_id) + elif not preset.audio_url: + logger.warning( + "[plan_id=%s] [BGM] 预设BGM未部署音频文件: preset_id=%s name=%s(audio_url 为空,请运维上传音频后填入 preset_bgm.py)", + plan_id, + preset_id, + preset.name, + ) + else: from video_processing.url_security import ( ALLOWED_AUDIO_MIME_TYPES, safe_download_file, ) - logger.info("[plan_id=%s] [BGM] 从预设库下载: preset_id=%s", plan_id, preset_id) + logger.info( + "[plan_id=%s] [BGM] 从预设库下载: preset_id=%s url=%s", + plan_id, + preset_id, + preset.audio_url[:80], + ) safe_download_file( preset.audio_url, str(bgm_file), @@ -480,7 +494,14 @@ class RenderAdapter: except Exception as e: logger.warning("[plan_id=%s] [BGM] 预设库下载失败: %s", plan_id, e) - logger.warning("[plan_id=%s] [BGM] 所有来源都无法获取BGM,跳过", plan_id) + logger.warning( + "[plan_id=%s] [BGM] 所有来源都无法获取BGM(enabled=%s audio_url=%s asset_id=%s preset_id=%s),跳过", + plan_id, + bool(bgm_config.get("enabled")), + "set" if audio_url else "empty", + asset_id[:12] + "…" if len(asset_id) > 12 else asset_id or "empty", + preset_id or "empty", + ) return None @staticmethod diff --git a/bgm_assets/README.md b/bgm_assets/README.md new file mode 100644 index 000000000..1063be5f8 --- /dev/null +++ b/bgm_assets/README.md @@ -0,0 +1,30 @@ +# 预设 BGM 音频文件放置目录 + +将下列 10 首免费可商用 BGM 的 mp3/m4a 文件按 `{preset_id}.mp3` 命名放到本目录: + +| preset_id | 名称 | 风格 | 时长(s) | 标签 | +|-------------------|----------|----------|---------|----------------------------| +| bgm_upbeat_001 | 阳光清晨 | upbeat | 120 | 轻快 阳光 吉他 vlog | +| bgm_upbeat_002 | 活力节拍 | upbeat | 95 | 轻快 电子 活力 运动 | +| bgm_upbeat_003 | 夏日漫步 | upbeat | 110 | 轻快 夏日 ukulele 旅行 | +| bgm_relax_001 | 静谧时光 | relax | 180 | 治愈 钢琴 安静 冥想 | +| bgm_relax_002 | 雨后森林 | relax | 150 | 治愈 自然 放松 环境音 | +| bgm_relax_003 | 月光奏鸣曲 | relax | 200 | 治愈 古典 钢琴 优雅(公版) | +| bgm_tech_001 | 未来科技 | tech | 85 | 科技 电子 未来感 数码 | +| bgm_tech_002 | 数据脉冲 | tech | 100 | 科技 极简 数据 AI | +| bgm_commerce_001 | 心动时刻 | commerce | 75 | 电商 时尚 动感 带货 | +| bgm_commerce_002 | 品质生活 | commerce | 90 | 电商 高端 品牌 品质 | + +放好后执行(需要在有 OSS 凭证的机器上): +```bash +export OSS_ENDPOINT=oss-cn-hangzhou.aliyuncs.com +export OSS_ACCESS_KEY_ID=xxx +export OSS_ACCESS_KEY_SECRET=xxx +export OSS_BUCKET_NAME=xiaoxia-autocut +python scripts/upload_preset_bgm.py +``` + +脚本会: +1. 上传文件到 OSS `preset/bgm/.mp3`,设置公共读 ACL +2. 自动改写 `packages/domain/preset_bgm.py` 把对应 `audio_url=""` 回填成公网 URL +3. 提示 `git commit & push` diff --git a/scripts/upload_preset_bgm.py b/scripts/upload_preset_bgm.py new file mode 100755 index 000000000..0b454c23e --- /dev/null +++ b/scripts/upload_preset_bgm.py @@ -0,0 +1,155 @@ +#!/usr/bin/env python3 +"""一键上传预设 BGM 到 OSS 并回填 preset_bgm.py audio_url。 + +用法(在服务器或本地有 OSS 凭证的机器上执行): + 1. 把 mp3 文件放到 ./bgm_assets/ 目录下,文件名按 {preset_id}.mp3 命名: + bgm_upbeat_001.mp3 阳光清晨 + bgm_upbeat_002.mp3 活力节拍 + bgm_upbeat_003.mp3 夏日漫步 + bgm_relax_001.mp3 静谧时光 + bgm_relax_002.mp3 雨后森林 + bgm_relax_003.mp3 月光奏鸣曲 + bgm_tech_001.mp3 未来科技 + bgm_tech_002.mp3 数据脉冲 + bgm_commerce_001.mp3 心动时刻 + bgm_commerce_002.mp3 品质生活 + 2. 确保环境变量已设置: + OSS_ENDPOINT, OSS_ACCESS_KEY_ID, OSS_ACCESS_KEY_SECRET, OSS_BUCKET_NAME + 3. 运行:python scripts/upload_preset_bgm.py + 4. 脚本会上传到 OSS 路径 preset/bgm/.mp3,并自动改写 + packages/domain/preset_bgm.py 填入 public URL。 + 5. git commit & push 即可。 + +支持可选参数: + --bgm-dir DIR 本地 mp3 目录(默认 ./bgm_assets) + --oss-prefix PFX OSS key 前缀(默认 preset/bgm/) + --dry-run 只打印要做的操作,不上传不改写 + --public 上传后设置公共读 ACL(默认开启,safe_download 走公网 URL) +""" + +from __future__ import annotations + +import argparse +import os +import sys +from pathlib import Path + + +REQUIRED_ENV = ("OSS_ENDPOINT", "OSS_ACCESS_KEY_ID", "OSS_ACCESS_KEY_SECRET", "OSS_BUCKET_NAME") +PRESET_BGM_PY = Path(__file__).resolve().parents[1] / "packages" / "domain" / "preset_bgm.py" + + +def main() -> int: + ap = argparse.ArgumentParser() + ap.add_argument("--bgm-dir", default="./bgm_assets") + ap.add_argument("--oss-prefix", default="preset/bgm/") + ap.add_argument("--dry-run", action="store_true") + ap.add_argument("--public", action="store_true", default=True) + args = ap.parse_args() + + bgm_dir = Path(args.bgm_dir) + if not bgm_dir.exists(): + print(f"[ERR] 目录不存在: {bgm_dir}", file=sys.stderr) + return 2 + + # 加载 preset_bgm.py 中所有 ID(简单 AST 抽取) + sys.path.insert(0, str(PRESET_BGM_PY.parents[2])) + from packages.domain.preset_bgm import PRESET_BGM_LIBRARY # type: ignore + + id_to_preset = {p.id: p for p in PRESET_BGM_LIBRARY} + + # 枚举本地文件 + local_files: dict[str, Path] = {} + for ext in (".mp3", ".m4a", ".aac", ".wav", ".ogg"): + for f in bgm_dir.glob(f"*{ext}"): + pid = f.stem + local_files[pid] = f + + missing_local = [pid for pid in id_to_preset if pid not in local_files] + unknown_local = [pid for pid in local_files if pid not in id_to_preset] + + print(f"[INFO] 预设 BGM 总数: {len(PRESET_BGM_LIBRARY)}") + print(f"[INFO] 本地找到对应文件: {len(local_files) - len(unknown_local)}") + if missing_local: + print(f"[WARN] 缺少本地音频文件的预设({len(missing_local)}):") + for pid in missing_local: + print(f" {pid} {id_to_preset[pid].name}") + if unknown_local: + print(f"[WARN] 本地文件未匹配任何 preset_id({len(unknown_local)}): {unknown_local}") + + if args.dry_run: + for pid, f in local_files.items(): + if pid in id_to_preset: + url = f"https://{os.environ.get('OSS_BUCKET_NAME', '')}.{os.environ.get('OSS_ENDPOINT', '')}/{args.oss_prefix}{f.name}" + print(f"[DRY] would upload {f} → {args.oss_prefix}{f.name} → {url}") + return 0 + + # 凭证检查 + for k in REQUIRED_ENV: + if not os.environ.get(k): + print(f"[ERR] 缺少环境变量 {k}", file=sys.stderr) + return 3 + + try: + import oss2 # type: ignore + except ImportError: + print("[ERR] 需要 oss2: pip install oss2", file=sys.stderr) + return 4 + + endpoint = os.environ["OSS_ENDPOINT"] + bucket_name = os.environ["OSS_BUCKET_NAME"] + auth = oss2.Auth(os.environ["OSS_ACCESS_KEY_ID"], os.environ["OSS_ACCESS_KEY_SECRET"]) + bucket = oss2.Bucket(auth, f"https://{endpoint}", bucket_name) + public_base = f"https://{bucket_name}.{endpoint}" + + pid_to_url: dict[str, str] = {} + for pid, f in local_files.items(): + if pid not in id_to_preset: + continue + key = f"{args.oss_prefix.rstrip('/')}/{f.name}" + print(f"[UPLOAD] {f} → oss://{bucket_name}/{key}") + headers = {"Content-Type": "audio/mpeg" if f.suffix == ".mp3" else "audio/mp4"} + if args.public: + bucket.put_object_from_file(key, str(f), headers=headers) + bucket.put_object_acl(key, oss2.OBJECT_ACL_PUBLIC_READ) + else: + bucket.put_object_from_file(key, str(f), headers=headers) + url = f"{public_base}/{key}" + pid_to_url[pid] = url + print(f" → {url}") + + if not pid_to_url: + print("[WARN] 没有文件被上传,不修改 preset_bgm.py") + return 0 + + # 回填 preset_bgm.py(按 id 精确替换 audio_url="" 为 audio_url="") + text = PRESET_BGM_PY.read_text(encoding="utf-8") + orig = text + for pid, url in pid_to_url.items(): + # 匹配 PresetBGM( ... id="pid", ... audio_url="", ... ) + # 简单替换:找到 id="pid" 行开始的 PresetBGM 构造块,把块内的 audio_url="" 替换 + import re + + block_pat = re.compile( + r'(PresetBGM\([^)]*?id="' + re.escape(pid) + r'"[^)]*?audio_url=)"[^"]*"', + re.DOTALL, + ) + new_text, n = block_pat.subn(rf'\1"{url}"', text, count=1) + if n == 0: + # 兜底:可能 audio_url 后面直接是 ),没赋值?不会,dataclass 有默认值 + print(f"[WARN] 未在 PresetBGM 块中找到 id={pid} 的 audio_url 字段,跳过回填") + continue + text = new_text + if text != orig: + PRESET_BGM_PY.write_text(text, encoding="utf-8") + print(f"[OK] 已回填 {len(pid_to_url)} 条 audio_url 到 {PRESET_BGM_PY}") + print( + "[NEXT] git diff packages/domain/preset_bgm.py && git add -A && git commit -m 'feat(domain): fill preset BGM audio_urls' && git push" + ) + else: + print("[WARN] preset_bgm.py 未发生变化(可能已填过)") + return 0 + + +if __name__ == "__main__": + sys.exit(main()) From d449496f909ac6b42897333f54b5a25dc74c7c89 Mon Sep 17 00:00:00 2001 From: CI Bot Date: Tue, 29 Sep 2026 07:33:22 +0000 Subject: [PATCH 3/4] style: auto-format with black + isort + ruff + prettier [skip ci-format-check] --- scripts/upload_preset_bgm.py | 1 - 1 file changed, 1 deletion(-) diff --git a/scripts/upload_preset_bgm.py b/scripts/upload_preset_bgm.py index 0b454c23e..2a3ef6eb9 100755 --- a/scripts/upload_preset_bgm.py +++ b/scripts/upload_preset_bgm.py @@ -34,7 +34,6 @@ import os import sys from pathlib import Path - REQUIRED_ENV = ("OSS_ENDPOINT", "OSS_ACCESS_KEY_ID", "OSS_ACCESS_KEY_SECRET", "OSS_BUCKET_NAME") PRESET_BGM_PY = Path(__file__).resolve().parents[1] / "packages" / "domain" / "preset_bgm.py" From ca6803e1a5afdb02b6eb16425cc81460fc51301b Mon Sep 17 00:00:00 2001 From: Coze Agent Date: Tue, 29 Sep 2026 15:35:34 +0800 Subject: [PATCH 4/4] chore(worker): move bgm upload readme under docs/ to avoid ruff parsing md --- bgm_assets/README.md => docs/preset-bgm-upload.md | 0 scripts/upload_preset_bgm.py | 2 +- 2 files changed, 1 insertion(+), 1 deletion(-) rename bgm_assets/README.md => docs/preset-bgm-upload.md (100%) diff --git a/bgm_assets/README.md b/docs/preset-bgm-upload.md similarity index 100% rename from bgm_assets/README.md rename to docs/preset-bgm-upload.md diff --git a/scripts/upload_preset_bgm.py b/scripts/upload_preset_bgm.py index 2a3ef6eb9..59a746d99 100755 --- a/scripts/upload_preset_bgm.py +++ b/scripts/upload_preset_bgm.py @@ -2,7 +2,7 @@ """一键上传预设 BGM 到 OSS 并回填 preset_bgm.py audio_url。 用法(在服务器或本地有 OSS 凭证的机器上执行): - 1. 把 mp3 文件放到 ./bgm_assets/ 目录下,文件名按 {preset_id}.mp3 命名: + 1. 把 mp3 文件放到 ./bgm_assets/ (created by you) 目录下,文件名按 {preset_id}.mp3 命名: bgm_upbeat_001.mp3 阳光清晨 bgm_upbeat_002.mp3 活力节拍 bgm_upbeat_003.mp3 夏日漫步