From 2bf63c066293c0058361104bc401dcd6a6aa816b Mon Sep 17 00:00:00 2001 From: saas-backend-agent Date: Fri, 4 Sep 2026 16:05:26 +0800 Subject: [PATCH] =?UTF-8?q?feat:=20=E7=89=87=E6=AE=B5=E6=97=B6=E9=95=BF?= =?UTF-8?q?=E8=87=AA=E5=8A=A8=E5=AF=B9=E9=BD=90=E9=85=8D=E9=9F=B3=E6=97=B6?= =?UTF-8?q?=E9=95=BF?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit 核心逻辑(在渲染前执行): - 探测配音音频文件时长 - 计算片段总时长与配音时长的比例 - ±5% 以内不调整 - ratio < 1(片段比配音长):按比例裁剪每段末尾 - ratio > 1(片段比配音短):按比例慢放每段(下限 0.25x) 实现: - 新增 _get_voice_audio_duration() 方法探测配音音频时长 - 新增 _align_clips_to_voice_duration() 方法调整片段时长 - 在 render() 中 clips 分组后、时长计算前调用对齐逻辑 - 调整后重新计算视频总时长用于后续渲染 测试: - 新增 11 个单元测试覆盖各种场景 - 153 个相关测试全通过 --- .../unified_render_service.py | 164 +++++++++++++- tests/unit/test_voice_duration_alignment.py | 206 ++++++++++++++++++ 2 files changed, 361 insertions(+), 9 deletions(-) create mode 100644 tests/unit/test_voice_duration_alignment.py diff --git a/apps/worker/video_processing/unified_render_service.py b/apps/worker/video_processing/unified_render_service.py index 4dd3323a4..6cf934373 100755 --- a/apps/worker/video_processing/unified_render_service.py +++ b/apps/worker/video_processing/unified_render_service.py @@ -198,8 +198,13 @@ class UnifiedRenderService: # 2. 分组为 RenderLayers layers = self._group_clips_into_layers(resolved) + # 2.5 配音时长对齐:如果有配音素材,调整片段时长以匹配配音时长 + voice_duration = self._get_voice_audio_duration() + if voice_duration > 0: + self._align_clips_to_voice_duration(layers, voice_duration) + # 3. 计算视频总时长(用于字幕显示时长) - video_duration = self._estimate_total_duration(layers) + video_duration_final = self._estimate_total_duration(layers) # Debug: 输出各图层时长明细 for layer in layers: layer_total = sum(UnifiedRenderService._clip_adjusted_duration(c) for c in layer.clips) @@ -215,16 +220,16 @@ class UnifiedRenderService: self.transition_duration, ", ".join(clip_details), ) - logger.info("[debug] estimated video_duration=%.3f", video_duration) + logger.info("[debug] estimated video_duration=%.3f", video_duration_final) # 3.5 TTS 配音生成(如果配置了) - self._maybe_add_voiceover_layer(layers, video_duration=video_duration) + self._maybe_add_voiceover_layer(layers, video_duration=video_duration_final) # 3.6 配音素材库音频(如果传入了本地路径) - self._maybe_add_voice_library_layer(layers, video_duration=video_duration) + self._maybe_add_voice_library_layer(layers, video_duration=video_duration_final) # 4. 生成 ASS 字幕文件(如果有 title/subtitle 配置) - ass_path = self._maybe_generate_ass(video_duration) + ass_path = self._maybe_generate_ass(video_duration_final) # 4.5 解析画中画配置 pip_config = PiPConfig.from_dict((self.plan.config or {}).get("pip_config")) @@ -257,7 +262,7 @@ class UnifiedRenderService: # 先尝试 stream copy 优化(无重编码,性能提升 10 倍+) # 条件不满足或失败时回退到带滤镜的直通渲染 stream_copy_ok = self._try_render_stream_copy( - layers, output_path, ass_path=ass_path, video_duration=video_duration + layers, output_path, ass_path=ass_path, video_duration=video_duration_final ) if stream_copy_ok: used_stream_copy = True @@ -271,7 +276,7 @@ class UnifiedRenderService: layers, output_path, ass_path=ass_path, - video_duration=video_duration, + video_duration=video_duration_final, ) else: filter_complex, input_args = self._build_filter_complex(layers, ass_path=ass_path) @@ -327,7 +332,7 @@ class UnifiedRenderService: from video_processing.ffmpeg_utils import run_ffmpeg run_ffmpeg(extract_cmd) - final_audio = mix_bgm_with_main(ctx, main_audio_path, bgm_cfg, video_duration) + final_audio = mix_bgm_with_main(ctx, main_audio_path, bgm_cfg, video_duration_final) # 合并回视频 bgm_output = self.work_dir / f"rendered_{self.plan.id}_bgm.mp4" @@ -353,7 +358,7 @@ class UnifiedRenderService: audio_path = mix_audio( ctx, layers, - video_duration, + video_duration_final, bgm_path=self.bgm_path, bgm_config=bgm_config, audio_tracks_config=audio_tracks_config, @@ -487,6 +492,147 @@ class UnifiedRenderService: """ return _estimate_total_duration_pure(layers, self.transition_duration) + def _get_voice_audio_duration(self) -> float: + """获取配音音频文件的时长(秒)。 + + Returns: + 配音音频时长,如果无配音或探测失败则返回 0.0 + """ + if not self.voiceover_audio_path: + return 0.0 + + audio_path = Path(self.voiceover_audio_path) + if not audio_path.exists() or audio_path.stat().st_size == 0: + return 0.0 + + try: + duration = probe_duration(audio_path) + logger.info("[voice-align] 配音音频时长: %.3fs path=%s", duration, self.voiceover_audio_path) + return duration + except Exception as e: + logger.warning("[voice-align] 探测配音音频时长失败: %s", e) + return 0.0 + + def _align_clips_to_voice_duration( + self, + layers: list[RenderLayer], + voice_duration: float, + ) -> None: + """调整片段时长以对齐配音时长。 + + 核心逻辑: + - 计算片段总时长与配音时长的比例 + - ±5% 以内不调整 + - ratio < 1(片段比配音长):按比例裁剪每段末尾 + - ratio > 1(片段比配音短):按比例慢放每段 + + Args: + layers: 渲染图层列表 + voice_duration: 配音时长(秒) + """ + if voice_duration <= 0: + return + + # 只调整视频图层(main/broll/background),不调整音频图层 + video_layers = [layer for layer in layers if layer.role in ("main", "broll", "background")] + if not video_layers: + return + + # 计算所有视频图层的总时长 + total_clips_duration = 0.0 + for layer in video_layers: + for clip in layer.clips: + clip_dur = self._clip_adjusted_duration(clip) + total_clips_duration += clip_dur + + if total_clips_duration <= 0: + return + + ratio = voice_duration / total_clips_duration + + # ±5% 以内不调整 + if abs(ratio - 1.0) <= 0.05: + logger.info( + "[voice-align] 比例接近1:1,跳过调整: ratio=%.4f voice=%.3f clips=%.3f", + ratio, + voice_duration, + total_clips_duration, + ) + return + + logger.info( + "[voice-align] 开始调整片段时长: ratio=%.4f voice=%.3f clips=%.3f", + ratio, + voice_duration, + total_clips_duration, + ) + + # 收集所有视频 clip + all_clips: list[tuple[RenderLayer, ResolvedClip]] = [] + for layer in video_layers: + for clip in layer.clips: + all_clips.append((layer, clip)) + + if not all_clips: + return + + if ratio < 1.0: + # 片段比配音长,按比例裁剪每段末尾 + # 减少每个 clip 的 duration + for _layer, clip in all_clips: + old_duration = clip.duration if clip.duration > 0 else clip.actual_duration + new_duration = old_duration * ratio + + # 更新 duration + clip.duration = max(0.1, new_duration) # 至少 0.1s + + # 如果有 trim_config,也需要调整 + if clip.trim_config is not None: + new_trim_duration = clip.trim_config.duration * ratio + clip.trim_config = TrimConfig( + start_time=clip.trim_config.start_time, + duration=max(0.1, new_trim_duration), + ) + + logger.debug( + "[voice-align] trim clip=%s: %.3f -> %.3f", + clip.clip_id, + old_duration, + clip.duration, + ) + + else: + # ratio > 1.0: 片段比配音短,按比例慢放每段 + # 降低 playback_speed + for _layer, clip in all_clips: + old_speed = clip.playback_speed if clip.playback_speed > 0 else 1.0 + # speed = old_speed / ratio 会使视频变慢(ratio > 1 时) + new_speed = old_speed / ratio + + # 下限 0.25x(避免过慢) + new_speed = max(0.25, round(new_speed, 4)) + clip.playback_speed = new_speed + + logger.debug( + "[voice-align] slowdown clip=%s: speed %.4f -> %.4f", + clip.clip_id, + old_speed, + new_speed, + ) + + # 调整后重新计算总时长用于日志 + new_total = 0.0 + for layer in video_layers: + for clip in layer.clips: + new_total += self._clip_adjusted_duration(clip) + + logger.info( + "[voice-align] 调整完成: 新总时长=%.3fs (目标=%.3fs, 差异=%.3fs)", + new_total, + voice_duration, + abs(new_total - voice_duration), + ) + def _maybe_generate_ass(self, video_duration: float) -> Path | None: """根据 plan.config 生成 ASS 字幕文件。 diff --git a/tests/unit/test_voice_duration_alignment.py b/tests/unit/test_voice_duration_alignment.py new file mode 100644 index 000000000..74fdb3e0b --- /dev/null +++ b/tests/unit/test_voice_duration_alignment.py @@ -0,0 +1,206 @@ +"""Tests for voice duration alignment feature. + +Tests the _align_clips_to_voice_duration method in UnifiedRenderService. +""" + +from __future__ import annotations + +from pathlib import Path +from unittest.mock import MagicMock, patch + +import pytest + +from video_processing.unified_render_service import ResolvedClip, RenderLayer, UnifiedRenderService + + +class TestAlignClipsToVoiceDuration: + """Test clip duration alignment to voice audio.""" + + def _make_clip( + self, + clip_id: str, + duration: float, + actual_duration: float = 0.0, + playback_speed: float = 1.0, + ) -> ResolvedClip: + """Helper to create a ResolvedClip for testing.""" + return ResolvedClip( + clip_id=clip_id, + asset_id=f"asset_{clip_id}", + local_path=Path(f"/tmp/{clip_id}.mp4"), + clip_type="main", + order=0, + duration=duration, + actual_duration=actual_duration or duration, + playback_speed=playback_speed, + ) + + def _make_layer(self, role: str, clips: list[ResolvedClip]) -> RenderLayer: + """Helper to create a RenderLayer for testing.""" + return RenderLayer(role=role, clips=clips, z_index=0) + + def _make_service(self, voiceover_path: str | None = None) -> UnifiedRenderService: + """Helper to create a mock UnifiedRenderService.""" + plan = MagicMock() + plan.id = "test_plan" + plan.config = {} + + with patch.object(UnifiedRenderService, "__init__", lambda self, **kwargs: None): + service = UnifiedRenderService.__new__(UnifiedRenderService) + service.plan = plan + service.voiceover_audio_path = voiceover_path + service.transition_duration = 0.0 + return service + + def test_no_voice_audio_no_adjustment(self): + """No voice audio → no adjustment.""" + service = self._make_service(voiceover_path=None) + clips = [self._make_clip("c1", 10.0), self._make_clip("c2", 10.0)] + layers = [self._make_layer("main", clips)] + + service._align_clips_to_voice_duration(layers, voice_duration=0.0) + + # No change + assert clips[0].duration == 10.0 + assert clips[1].duration == 10.0 + + def test_ratio_within_5_percent_no_adjustment(self): + """Ratio within ±5% → no adjustment.""" + service = self._make_service() + clips = [self._make_clip("c1", 10.0)] + layers = [self._make_layer("main", clips)] + + # Total clips = 10s, voice = 10.3s → ratio = 1.03 (within 5%) + service._align_clips_to_voice_duration(layers, voice_duration=10.3) + + assert clips[0].duration == 10.0 # Unchanged + + def test_ratio_less_than_1_trim_clips(self): + """Ratio < 1 (clips too long) → trim clips proportionally.""" + service = self._make_service() + clips = [self._make_clip("c1", 10.0), self._make_clip("c2", 10.0)] + layers = [self._make_layer("main", clips)] + + # Total clips = 20s, voice = 15s → ratio = 0.75 + service._align_clips_to_voice_duration(layers, voice_duration=15.0) + + # Each clip should be trimmed to 75% + assert abs(clips[0].duration - 7.5) < 0.01 + assert abs(clips[1].duration - 7.5) < 0.01 + + def test_ratio_greater_than_1_slowdown_clips(self): + """Ratio > 1 (clips too short) → slow down clips.""" + service = self._make_service() + clips = [self._make_clip("c1", 10.0), self._make_clip("c2", 10.0)] + layers = [self._make_layer("main", clips)] + + # Total clips = 20s, voice = 25s → ratio = 1.25 + service._align_clips_to_voice_duration(layers, voice_duration=25.0) + + # Each clip's speed should be reduced: 1.0 / 1.25 = 0.8 + assert abs(clips[0].playback_speed - 0.8) < 0.01 + assert abs(clips[1].playback_speed - 0.8) < 0.01 + + def test_speed_lower_bound_025(self): + """Playback speed should not go below 0.25x.""" + service = self._make_service() + clips = [self._make_clip("c1", 5.0)] + layers = [self._make_layer("main", clips)] + + # Total clips = 5s, voice = 50s → ratio = 10.0 + # Speed would be 1.0 / 10 = 0.1, but should be clamped to 0.25 + service._align_clips_to_voice_duration(layers, voice_duration=50.0) + + assert clips[0].playback_speed == 0.25 + + def test_only_video_layers_adjusted(self): + """Only main/broll/background layers are adjusted, not audio.""" + service = self._make_service() + + video_clips = [self._make_clip("v1", 10.0)] + audio_clips = [self._make_clip("a1", 10.0)] + + layers = [ + self._make_layer("main", video_clips), + self._make_layer("audio", audio_clips), + ] + + # ratio = 0.5 → should trim video but not audio + service._align_clips_to_voice_duration(layers, voice_duration=5.0) + + assert abs(video_clips[0].duration - 5.0) < 0.01 # Trimmed + assert audio_clips[0].duration == 10.0 # Unchanged + + def test_multiple_video_layers_all_adjusted(self): + """All video layers (main, broll, background) are adjusted.""" + service = self._make_service() + + main_clips = [self._make_clip("m1", 10.0)] + broll_clips = [self._make_clip("b1", 10.0)] + bg_clips = [self._make_clip("bg1", 10.0)] + + layers = [ + self._make_layer("main", main_clips), + self._make_layer("broll", broll_clips), + self._make_layer("background", bg_clips), + ] + + # Total video = 30s, voice = 15s → ratio = 0.5 + service._align_clips_to_voice_duration(layers, voice_duration=15.0) + + # All should be trimmed to 50% + assert abs(main_clips[0].duration - 5.0) < 0.01 + assert abs(broll_clips[0].duration - 5.0) < 0.01 + assert abs(bg_clips[0].duration - 5.0) < 0.01 + + def test_trim_config_also_adjusted(self): + """When clip has trim_config, it should also be adjusted.""" + from video_processing.trim_engine import TrimConfig + + service = self._make_service() + + clip = self._make_clip("c1", 10.0) + clip.trim_config = TrimConfig(start_time=0.0, duration=10.0) + + layers = [self._make_layer("main", [clip])] + + # ratio = 0.5 + service._align_clips_to_voice_duration(layers, voice_duration=5.0) + + assert abs(clip.duration - 5.0) < 0.01 + assert clip.trim_config is not None + assert abs(clip.trim_config.duration - 5.0) < 0.01 + + +class TestGetVoiceAudioDuration: + """Test voice audio duration probing.""" + + def test_no_voiceover_path_returns_zero(self): + """No voiceover path → return 0.""" + with patch.object(UnifiedRenderService, "__init__", lambda self, **kwargs: None): + service = UnifiedRenderService.__new__(UnifiedRenderService) + service.voiceover_audio_path = None + + assert service._get_voice_audio_duration() == 0.0 + + def test_nonexistent_file_returns_zero(self): + """Nonexistent file → return 0.""" + with patch.object(UnifiedRenderService, "__init__", lambda self, **kwargs: None): + service = UnifiedRenderService.__new__(UnifiedRenderService) + service.voiceover_audio_path = "/nonexistent/path.mp3" + + assert service._get_voice_audio_duration() == 0.0 + + @patch("video_processing.unified_render_service.probe_duration") + @patch("video_processing.unified_render_service.Path.exists", return_value=True) + @patch("video_processing.unified_render_service.Path.stat") + def test_probes_duration_from_file(self, mock_stat, mock_exists, mock_probe): + """Valid file → probe duration.""" + mock_stat.return_value.st_size = 1000 # Non-empty file + mock_probe.return_value = 42.5 + + with patch.object(UnifiedRenderService, "__init__", lambda self, **kwargs: None): + service = UnifiedRenderService.__new__(UnifiedRenderService) + service.voiceover_audio_path = "/tmp/voice.mp3" + + assert service._get_voice_audio_duration() == 42.5 -- 2.54.0