feat: ASR自动字幕能力(领域模型+渲染管道接入+可扩展ASR后端) (#292)
CI/CD Pipeline / Validate Code Quality And Tests (push) Has been cancelled
CI/CD Pipeline / Unit Tests (push) Has been cancelled
CI/CD Pipeline / Integration Tests (push) Has been cancelled
CI/CD Pipeline / Frontend Lint (push) Has been cancelled
CI/CD Pipeline / Build & Push Staging (Watchtower auto-deploy) (push) Has been cancelled
CI/CD Pipeline / Staging E2E Tests (push) Has been cancelled
CI/CD Pipeline / Staging API Integration Tests (push) Has been cancelled
CI/CD Pipeline / Build Production Runtime Images (push) Has been cancelled
CI/CD Pipeline / Deploy Production (push) Has been cancelled
CI/CD Pipeline / Production Browser E2E (push) Has been cancelled

feat: ASR自动字幕能力
This commit was merged in pull request #292.
This commit is contained in:
2026-07-14 09:50:57 +08:00
parent 58ff565c48
commit f4b4f1fc4f
11 changed files with 1393 additions and 3 deletions
@@ -41,6 +41,7 @@ from video_processing.ffmpeg_utils import (
)
from video_processing.render_audio import RenderContext, merge_audio_video, mix_audio
from video_processing.render_subtitles import generate_ass_subtitles
from video_processing.subtitle_generator import generate_ass_from_timeline
logger = logging.getLogger(__name__)
@@ -158,6 +159,7 @@ class UnifiedRenderService:
output_height: int = DEFAULT_OUTPUT_HEIGHT,
output_fps: int = DEFAULT_FPS,
transition_duration: float = DEFAULT_TRANSITION_DURATION,
asr_service: Any = None, # ASRService 实例,用于自动生成字幕
):
self.plan = plan
self.clips = clips
@@ -167,6 +169,7 @@ class UnifiedRenderService:
self.output_height = output_height
self.output_fps = output_fps
self.transition_duration = transition_duration
self.asr_service = asr_service
def render(self) -> RenderResult:
"""执行渲染,返回 RenderResult.
@@ -341,6 +344,10 @@ class UnifiedRenderService:
def _maybe_generate_ass(self, video_duration: float) -> Path | None:
"""根据 plan.config 生成 ASS 字幕文件。
支持两种字幕模式:
1. 静态字幕 — title/subtitle 配置了 text 时,生成整段静态字幕
2. ASR 自动字幕 — subtitle.auto_generated=true 时,从音频自动识别生成时间轴字幕
Returns:
ASS 文件路径,没有字幕时返回 None
"""
@@ -352,15 +359,46 @@ class UnifiedRenderService:
subtitle_enabled = subtitle_cfg.get("enabled", True)
title_text = title_cfg.get("text", "") or ""
subtitle_text = subtitle_cfg.get("text", "") or ""
auto_generated = subtitle_cfg.get("auto_generated", False)
has_title = title_enabled and bool(title_text.strip())
has_subtitle = subtitle_enabled and bool(subtitle_text.strip())
has_static_subtitle = subtitle_enabled and bool(subtitle_text.strip())
has_auto_subtitle = subtitle_enabled and auto_generated and self.asr_service is not None
if not has_title and not has_subtitle:
if not has_title and not has_static_subtitle and not has_auto_subtitle:
return None
ass_path = self.work_dir / f"subtitles_{self.plan.id}.ass"
# ASR 自动字幕模式
if has_auto_subtitle:
try:
timeline = self._generate_asr_subtitles(video_duration, subtitle_cfg)
if timeline and timeline.segments:
generate_ass_from_timeline(
ass_path,
timeline,
video_width=self.output_width,
video_height=self.output_height,
subtitle_config=subtitle_cfg,
)
logger.info(
"ASR自动字幕生成完成: plan_id=%s segments=%d duration=%.1fs",
self.plan.id,
timeline.segment_count,
video_duration,
)
return ass_path
else:
# ASR 无结果,不生成字幕
logger.info("ASR自动字幕无识别结果,跳过字幕: plan_id=%s", self.plan.id)
return None
except Exception:
# ASR 失败降级:不生成字幕,不阻断主流程
logger.warning("ASR自动字幕生成失败,跳过字幕", exc_info=True)
return None
# 静态字幕模式(原有逻辑)
generate_ass_subtitles(
ass_path,
video_width=self.output_width,
@@ -376,11 +414,95 @@ class UnifiedRenderService:
"生成字幕: plan_id=%s title=%s subtitle=%s ass=%s",
self.plan.id,
has_title,
has_subtitle,
has_static_subtitle,
ass_path,
)
return ass_path
def _generate_asr_subtitles(self, video_duration: float, subtitle_cfg: dict) -> Any: # SubtitleTimeline
"""从视频素材音频中自动识别生成字幕时间轴。
MVP 版本:使用第一个有音频的素材做ASR,然后按比例映射到整个视频时长。
后续优化:支持多片段拼接后的完整音频ASR。
"""
from packages.domain.subtitle import SubtitleTimeline
# 找第一个有本地路径的素材
first_asset_path = None
for clip in self.clips:
asset_id = getattr(clip, "asset_id", None)
if asset_id and asset_id in self.asset_path_map:
first_asset_path = self.asset_path_map[asset_id]
break
if first_asset_path is None:
logger.warning("ASR字幕生成失败:找不到可用素材音频")
return SubtitleTimeline(segments=[], total_duration=video_duration)
# 提取素材音频为 wav(16kHz单声道,ASR友好格式)
audio_path = self.work_dir / f"asr_audio_{self.plan.id}.wav"
try:
self._extract_audio(first_asset_path, audio_path)
except Exception:
logger.warning("ASR音频提取失败", exc_info=True)
return SubtitleTimeline(segments=[], total_duration=video_duration)
if not audio_path.exists():
return SubtitleTimeline(segments=[], total_duration=video_duration)
# 调用 ASR 服务
language = subtitle_cfg.get("language", "") or None
timeline = self.asr_service.transcribe(
audio_path,
language=language,
with_word_timestamps=True,
)
# 字幕后处理:合并短片段 + 拆分长片段
min_chars = int(subtitle_cfg.get("min_chars_per_segment", 8))
max_chars = int(subtitle_cfg.get("max_chars_per_line", 20))
if timeline.segments:
timeline = timeline.merge_short_segments(min_chars=min_chars)
timeline = timeline.split_long_segments(max_chars=max_chars)
# 清理临时音频文件
try:
audio_path.unlink(missing_ok=True)
except Exception:
pass
return timeline
def _extract_audio(self, video_path: Path, output_path: Path) -> None:
"""从视频中提取音频为16kHz单声道wav(ASR友好格式)。"""
import subprocess
cmd = [
"ffmpeg",
"-y",
"-i",
str(video_path),
"-vn",
"-acodec",
"pcm_s16le",
"-ar",
"16000",
"-ac",
"1",
str(output_path),
]
result = subprocess.run(
cmd,
capture_output=True,
text=True,
timeout=120,
)
if result.returncode != 0:
raise RuntimeError(f"音频提取失败: {result.stderr[:200]}")
def _can_use_pass_through(self, layers: list[RenderLayer]) -> bool:
"""判断是否可以走直通优化路径。