fix(#549): 预设配音无声 - 字幕对齐TTS配音模式 + ASR缓存 (#628)
CI/CD Pipeline / Unit Tests (push) Has been cancelled
CI/CD Pipeline / Integration Tests (push) Has been cancelled
CI/CD Pipeline / Frontend Unit Tests (push) Has been cancelled
CI/CD Pipeline / Build Staging Worker Image (push) Has been cancelled
CI/CD Pipeline / Deploy Staging (Watchtower auto-deploy) (push) Has been cancelled
CI/CD Pipeline / Staging E2E Tests (push) Has been cancelled
CI/CD Pipeline / Staging API Integration Tests (push) Has been cancelled
CI/CD Pipeline / ACR Image Cleanup (push) Has been cancelled
CI/CD Pipeline / Production Browser E2E (push) Failing after 1394h39m22s
CI/CD Pipeline / Build Production API Image (push) Failing after 1394h39m26s
CI/CD Pipeline / Deploy Production (push) Failing after 1394h39m24s
CI/CD Pipeline / Check if frontend-only change (push) Failing after 1394h39m27s
CI/CD Pipeline / Build Production Worker Image (push) Failing after 1394h39m26s
CI/CD Pipeline / Validate Code Quality And Tests (push) Has been skipped
CI/CD Pipeline / Frontend Lint (push) Has been skipped
CI/CD Pipeline / Build Staging API Image (push) Has been skipped
CI/CD Pipeline / Build Staging Web Image (push) Has been skipped
CI/CD Pipeline / Build Production Web Image (push) Failing after 1395h11m33s

Co-authored-by: xiaoxia <dev@xiaoxiajianji.com>
Co-committed-by: xiaoxia <dev@xiaoxiajianji.com>
This commit was merged in pull request #628.
This commit is contained in:
2026-07-20 12:49:54 +08:00
committed by auto-approve-bot
parent e5fffe40c1
commit f01f4803e4
3 changed files with 511 additions and 3 deletions
@@ -188,6 +188,8 @@ class UnifiedRenderService:
self.bgm_path = bgm_path
self._transition_engine = TransitionEngine(default_duration=transition_duration)
self._speed_engine = SpeedEngine()
self._asr_timeline_cache: Any = None # ASR 字幕结果缓存,避免重复调用
self._asr_timeline_cached = False
def render(self) -> RenderResult:
"""执行渲染,返回 RenderResult.
@@ -589,7 +591,13 @@ class UnifiedRenderService:
MVP 版本:使用第一个有音频的素材做ASR,然后按比例映射到整个视频时长。
后续优化:支持多片段拼接后的完整音频ASR。
带缓存:同一 plan 只做一次 ASR,TTS 配音和字幕共用结果。
"""
# 检查缓存
if self._asr_timeline_cached:
return self._asr_timeline_cache
from packages.domain.subtitle import SubtitleTimeline
# 找第一个有本地路径的素材
@@ -602,7 +610,10 @@ class UnifiedRenderService:
if first_asset_path is None:
logger.warning("ASR字幕生成失败:找不到可用素材音频")
return SubtitleTimeline(segments=[], total_duration=video_duration)
result = SubtitleTimeline(segments=[], total_duration=video_duration)
self._asr_timeline_cache = result
self._asr_timeline_cached = True
return result
# 提取素材音频为 wav(16kHz单声道,ASR友好格式)
audio_path = self.work_dir / f"asr_audio_{self.plan.id}.wav"
@@ -637,6 +648,9 @@ class UnifiedRenderService:
except Exception:
pass
# 存入缓存
self._asr_timeline_cache = timeline
self._asr_timeline_cached = True
return timeline
def _extract_audio(self, video_path: Path, output_path: Path) -> None:
@@ -669,11 +683,54 @@ class UnifiedRenderService:
) -> bool:
"""根据 plan.config 生成 TTS 配音,加到 audio 图层.
支持三种触发方式:
1. config.tts.enabled = true → 标准 TTS 配置
2. 顶层 voice_id + custom_text → 桥接模式(自定义文案配音)
3. 顶层 voice_id + subtitle.auto_generated=true → ASR 字幕对齐配音(预设配音)
Returns:
是否成功添加了配音音轨
"""
config = self.plan.config or {}
tts_cfg = config.get("tts", {}) or {}
subtitle_cfg = config.get("subtitle", {}) or {}
use_subtitle_align = False # 是否使用字幕对齐模式
# 兼容前端顶层字段:voice_id / custom_text / voice_clone_profile_id
if not tts_cfg.get("enabled"):
top_voice_id = config.get("voice_id", "") or ""
top_text = config.get("custom_text", "") or ""
# 方式A:voice_id + custom_text → 整段配音
if top_voice_id and top_text:
tts_cfg = {
"enabled": True,
"voice_id": top_voice_id,
"text": top_text,
"align_mode": "full",
"overlap_mode": "replace",
}
logger.info(
"[unified-render] 检测到顶层 voice_id+custom_text,桥接到 tts 配置(整段): plan_id=%s voice_id=%s text_len=%d",
self.plan.id,
top_voice_id,
len(top_text),
)
# 方式B:voice_id + 自动字幕 → 字幕对齐配音(预设配音模式)
elif top_voice_id and subtitle_cfg.get("auto_generated", False) and self.asr_service is not None:
tts_cfg = {
"enabled": True,
"voice_id": top_voice_id,
"text": "",
"align_mode": "subtitle",
"overlap_mode": "replace",
}
use_subtitle_align = True
logger.info(
"[unified-render] 检测到预设配音+自动字幕,使用字幕对齐模式: plan_id=%s voice_id=%s",
self.plan.id,
top_voice_id,
)
# 兼容前端顶层字段:voice_id / custom_text / voice_clone_profile_id
# 前端一键生成页面传 config.voice_id + config.custom_text,
@@ -706,8 +763,34 @@ class UnifiedRenderService:
tts_service = get_tts_service()
tts_engine = TtsEngine(tts_service, self.work_dir / "tts")
# 整段配音模式
result = tts_engine.generate_full_voiceover(tts_config, total_duration=video_duration)
# 根据对齐模式选择生成方式
if use_subtitle_align or tts_config.align_mode == "subtitle":
# 字幕对齐模式:先做 ASR,再按字幕生成配音
if not self._asr_timeline_cached:
self._generate_asr_subtitles(video_duration, subtitle_cfg)
timeline = self._asr_timeline_cache
if timeline is None or not timeline.segments:
logger.warning("TTS 字幕对齐配音:ASR 无识别结果,跳过配音")
return False
# 转换为 TtsEngine 需要的字幕格式
subtitles = [
{
"text": seg.text,
"start_time": seg.start,
"end_time": seg.end,
}
for seg in timeline.segments
if getattr(seg, "text", "").strip()
]
if not subtitles:
logger.warning("TTS 字幕对齐配音:字幕文本为空,跳过配音")
return False
result = tts_engine.generate_subtitle_voiceover(tts_config, subtitles)
else:
# 整段配音模式
result = tts_engine.generate_full_voiceover(tts_config, total_duration=video_duration)
if not result.success or not result.segments:
logger.warning("TTS 配音生成失败,跳过: %s", result.error_message)