feat: ASR自动字幕能力(领域模型+渲染管道接入+可扩展ASR后端) (#292)
CI/CD Pipeline / Validate Code Quality And Tests (push) Has been cancelled
CI/CD Pipeline / Unit Tests (push) Has been cancelled
CI/CD Pipeline / Integration Tests (push) Has been cancelled
CI/CD Pipeline / Frontend Lint (push) Has been cancelled
CI/CD Pipeline / Build & Push Staging (Watchtower auto-deploy) (push) Has been cancelled
CI/CD Pipeline / Staging E2E Tests (push) Has been cancelled
CI/CD Pipeline / Staging API Integration Tests (push) Has been cancelled
CI/CD Pipeline / Build Production Runtime Images (push) Has been cancelled
CI/CD Pipeline / Deploy Production (push) Has been cancelled
CI/CD Pipeline / Production Browser E2E (push) Has been cancelled
CI/CD Pipeline / Validate Code Quality And Tests (push) Has been cancelled
CI/CD Pipeline / Unit Tests (push) Has been cancelled
CI/CD Pipeline / Integration Tests (push) Has been cancelled
CI/CD Pipeline / Frontend Lint (push) Has been cancelled
CI/CD Pipeline / Build & Push Staging (Watchtower auto-deploy) (push) Has been cancelled
CI/CD Pipeline / Staging E2E Tests (push) Has been cancelled
CI/CD Pipeline / Staging API Integration Tests (push) Has been cancelled
CI/CD Pipeline / Build Production Runtime Images (push) Has been cancelled
CI/CD Pipeline / Deploy Production (push) Has been cancelled
CI/CD Pipeline / Production Browser E2E (push) Has been cancelled
feat: ASR自动字幕能力
This commit was merged in pull request #292.
This commit is contained in:
@@ -41,6 +41,7 @@ from video_processing.ffmpeg_utils import (
|
||||
)
|
||||
from video_processing.render_audio import RenderContext, merge_audio_video, mix_audio
|
||||
from video_processing.render_subtitles import generate_ass_subtitles
|
||||
from video_processing.subtitle_generator import generate_ass_from_timeline
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
@@ -158,6 +159,7 @@ class UnifiedRenderService:
|
||||
output_height: int = DEFAULT_OUTPUT_HEIGHT,
|
||||
output_fps: int = DEFAULT_FPS,
|
||||
transition_duration: float = DEFAULT_TRANSITION_DURATION,
|
||||
asr_service: Any = None, # ASRService 实例,用于自动生成字幕
|
||||
):
|
||||
self.plan = plan
|
||||
self.clips = clips
|
||||
@@ -167,6 +169,7 @@ class UnifiedRenderService:
|
||||
self.output_height = output_height
|
||||
self.output_fps = output_fps
|
||||
self.transition_duration = transition_duration
|
||||
self.asr_service = asr_service
|
||||
|
||||
def render(self) -> RenderResult:
|
||||
"""执行渲染,返回 RenderResult.
|
||||
@@ -341,6 +344,10 @@ class UnifiedRenderService:
|
||||
def _maybe_generate_ass(self, video_duration: float) -> Path | None:
|
||||
"""根据 plan.config 生成 ASS 字幕文件。
|
||||
|
||||
支持两种字幕模式:
|
||||
1. 静态字幕 — title/subtitle 配置了 text 时,生成整段静态字幕
|
||||
2. ASR 自动字幕 — subtitle.auto_generated=true 时,从音频自动识别生成时间轴字幕
|
||||
|
||||
Returns:
|
||||
ASS 文件路径,没有字幕时返回 None
|
||||
"""
|
||||
@@ -352,15 +359,46 @@ class UnifiedRenderService:
|
||||
subtitle_enabled = subtitle_cfg.get("enabled", True)
|
||||
title_text = title_cfg.get("text", "") or ""
|
||||
subtitle_text = subtitle_cfg.get("text", "") or ""
|
||||
auto_generated = subtitle_cfg.get("auto_generated", False)
|
||||
|
||||
has_title = title_enabled and bool(title_text.strip())
|
||||
has_subtitle = subtitle_enabled and bool(subtitle_text.strip())
|
||||
has_static_subtitle = subtitle_enabled and bool(subtitle_text.strip())
|
||||
has_auto_subtitle = subtitle_enabled and auto_generated and self.asr_service is not None
|
||||
|
||||
if not has_title and not has_subtitle:
|
||||
if not has_title and not has_static_subtitle and not has_auto_subtitle:
|
||||
return None
|
||||
|
||||
ass_path = self.work_dir / f"subtitles_{self.plan.id}.ass"
|
||||
|
||||
# ASR 自动字幕模式
|
||||
if has_auto_subtitle:
|
||||
try:
|
||||
timeline = self._generate_asr_subtitles(video_duration, subtitle_cfg)
|
||||
if timeline and timeline.segments:
|
||||
generate_ass_from_timeline(
|
||||
ass_path,
|
||||
timeline,
|
||||
video_width=self.output_width,
|
||||
video_height=self.output_height,
|
||||
subtitle_config=subtitle_cfg,
|
||||
)
|
||||
logger.info(
|
||||
"ASR自动字幕生成完成: plan_id=%s segments=%d duration=%.1fs",
|
||||
self.plan.id,
|
||||
timeline.segment_count,
|
||||
video_duration,
|
||||
)
|
||||
return ass_path
|
||||
else:
|
||||
# ASR 无结果,不生成字幕
|
||||
logger.info("ASR自动字幕无识别结果,跳过字幕: plan_id=%s", self.plan.id)
|
||||
return None
|
||||
except Exception:
|
||||
# ASR 失败降级:不生成字幕,不阻断主流程
|
||||
logger.warning("ASR自动字幕生成失败,跳过字幕", exc_info=True)
|
||||
return None
|
||||
|
||||
# 静态字幕模式(原有逻辑)
|
||||
generate_ass_subtitles(
|
||||
ass_path,
|
||||
video_width=self.output_width,
|
||||
@@ -376,11 +414,95 @@ class UnifiedRenderService:
|
||||
"生成字幕: plan_id=%s title=%s subtitle=%s ass=%s",
|
||||
self.plan.id,
|
||||
has_title,
|
||||
has_subtitle,
|
||||
has_static_subtitle,
|
||||
ass_path,
|
||||
)
|
||||
return ass_path
|
||||
|
||||
def _generate_asr_subtitles(self, video_duration: float, subtitle_cfg: dict) -> Any: # SubtitleTimeline
|
||||
"""从视频素材音频中自动识别生成字幕时间轴。
|
||||
|
||||
MVP 版本:使用第一个有音频的素材做ASR,然后按比例映射到整个视频时长。
|
||||
后续优化:支持多片段拼接后的完整音频ASR。
|
||||
"""
|
||||
from packages.domain.subtitle import SubtitleTimeline
|
||||
|
||||
# 找第一个有本地路径的素材
|
||||
first_asset_path = None
|
||||
for clip in self.clips:
|
||||
asset_id = getattr(clip, "asset_id", None)
|
||||
if asset_id and asset_id in self.asset_path_map:
|
||||
first_asset_path = self.asset_path_map[asset_id]
|
||||
break
|
||||
|
||||
if first_asset_path is None:
|
||||
logger.warning("ASR字幕生成失败:找不到可用素材音频")
|
||||
return SubtitleTimeline(segments=[], total_duration=video_duration)
|
||||
|
||||
# 提取素材音频为 wav(16kHz单声道,ASR友好格式)
|
||||
audio_path = self.work_dir / f"asr_audio_{self.plan.id}.wav"
|
||||
try:
|
||||
self._extract_audio(first_asset_path, audio_path)
|
||||
except Exception:
|
||||
logger.warning("ASR音频提取失败", exc_info=True)
|
||||
return SubtitleTimeline(segments=[], total_duration=video_duration)
|
||||
|
||||
if not audio_path.exists():
|
||||
return SubtitleTimeline(segments=[], total_duration=video_duration)
|
||||
|
||||
# 调用 ASR 服务
|
||||
language = subtitle_cfg.get("language", "") or None
|
||||
timeline = self.asr_service.transcribe(
|
||||
audio_path,
|
||||
language=language,
|
||||
with_word_timestamps=True,
|
||||
)
|
||||
|
||||
# 字幕后处理:合并短片段 + 拆分长片段
|
||||
min_chars = int(subtitle_cfg.get("min_chars_per_segment", 8))
|
||||
max_chars = int(subtitle_cfg.get("max_chars_per_line", 20))
|
||||
|
||||
if timeline.segments:
|
||||
timeline = timeline.merge_short_segments(min_chars=min_chars)
|
||||
timeline = timeline.split_long_segments(max_chars=max_chars)
|
||||
|
||||
# 清理临时音频文件
|
||||
try:
|
||||
audio_path.unlink(missing_ok=True)
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
return timeline
|
||||
|
||||
def _extract_audio(self, video_path: Path, output_path: Path) -> None:
|
||||
"""从视频中提取音频为16kHz单声道wav(ASR友好格式)。"""
|
||||
import subprocess
|
||||
|
||||
cmd = [
|
||||
"ffmpeg",
|
||||
"-y",
|
||||
"-i",
|
||||
str(video_path),
|
||||
"-vn",
|
||||
"-acodec",
|
||||
"pcm_s16le",
|
||||
"-ar",
|
||||
"16000",
|
||||
"-ac",
|
||||
"1",
|
||||
str(output_path),
|
||||
]
|
||||
|
||||
result = subprocess.run(
|
||||
cmd,
|
||||
capture_output=True,
|
||||
text=True,
|
||||
timeout=120,
|
||||
)
|
||||
|
||||
if result.returncode != 0:
|
||||
raise RuntimeError(f"音频提取失败: {result.stderr[:200]}")
|
||||
|
||||
def _can_use_pass_through(self, layers: list[RenderLayer]) -> bool:
|
||||
"""判断是否可以走直通优化路径。
|
||||
|
||||
|
||||
Reference in New Issue
Block a user