From a83b53ae5804abef56890d517b3ca4f288e55ffe Mon Sep 17 00:00:00 2001 From: xiaoxia Date: Sat, 12 Sep 2026 21:07:34 +0800 Subject: [PATCH] =?UTF-8?q?fix(ai-avatar):=20B-roll=E6=97=B6=E9=97=B4?= =?UTF-8?q?=E6=88=B3=E3=80=81=E6=A0=87=E9=A2=98=E7=B2=97=E4=BD=93=E9=87=8D?= =?UTF-8?q?=E5=BD=B1=E3=80=81TTS=E5=8D=A1=E6=AD=BB=E9=98=B2=E6=8A=A4?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit 1. B-roll时间戳全0修复: - sentences.ts 新增 sentenceTimings 参数优先使用后端精确时间戳, 降级才按字数比例估算 - 修正参数顺序,与 ModalBRollEditor 现有调用 (scriptText, timings, duration) 对齐 - 前端 AiAvatarPage 已通过 props 传 sentenceTimings(旧版已传但因顺序错位被忽略) 2. 标题粗体重影修复: - 之前 bold=true 时使用 borderw=3 + font_color 同色描边模拟粗体, 会在小字号/竖屏视频上造成字形边缘偏移,视觉上文字像被打印了两次 (用户截图中的'曝光曝光…'重影) - 改为黑色细描边(borderw=2, 黑色),既保留清晰加粗效果又不重影 - 用户显式开启 stroke 时仍按用户配置走 - 新增 _resolve_font_path(bold=True) 预留粗体字体查找能力(当前镜像 无独立 Bold 字体文件,沿用 VF 常规字重) 3. TTS 任务防卡死: - 给 tts_synthesize_and_submit 加 soft_time_limit=180s / time_limit=200s, 避免因网络/上游问题导致 Celery 任务永久挂起(用户之前卡10+分钟 tts_processing 不失败) - 任务开头加 INFO 日志(job_id/voice_id/text_len),方便排查 worker 是否真的收到任务 相关:staging 用户反馈 5 问题中的 1/2/3 项(B-roll时间、标题重影、对口型卡死) --- apps/api/app/tasks/lipsync_tts.py | 9 ++ .../src/pages/ai-avatar/utils/sentences.ts | 111 ++++++++++-------- packages/domain/video_filter_builder.py | 49 +++++--- 3 files changed, 106 insertions(+), 63 deletions(-) diff --git a/apps/api/app/tasks/lipsync_tts.py b/apps/api/app/tasks/lipsync_tts.py index fb771f702..1b3ef0dfc 100644 --- a/apps/api/app/tasks/lipsync_tts.py +++ b/apps/api/app/tasks/lipsync_tts.py @@ -213,6 +213,8 @@ def _estimate_sentence_timings_by_chars(sentences: list[str], total_duration: fl name="lipsync_tts.synthesize_and_submit", max_retries=2, default_retry_delay=30, + soft_time_limit=180, + time_limit=200, ) def tts_synthesize_and_submit( self, @@ -266,6 +268,13 @@ def tts_synthesize_and_submit( return # 1. TTS 合成 + logger.info( + "[lipsync_tts] 开始 TTS 合成: job_id=%s voice_id=%s text_len=%d speed=%.2f", + job_id, + voice_id, + len(script_text), + speed, + ) try: cosyvoice = CosyVoiceService() result = cosyvoice.submit_synthesize_task( diff --git a/apps/web/src/pages/ai-avatar/utils/sentences.ts b/apps/web/src/pages/ai-avatar/utils/sentences.ts index 6c3ad70e1..0c4f1d74b 100644 --- a/apps/web/src/pages/ai-avatar/utils/sentences.ts +++ b/apps/web/src/pages/ai-avatar/utils/sentences.ts @@ -1,10 +1,9 @@ /** - * AI数字人 — 文案分句工具 + * AI数字人 — 文案分句 & B-roll 时间计算 * - * 分句规则与后端 _split_script_into_sentences 保持一致。 - * 时间戳由后端基于 TTS 音频静音检测精确计算,前端不再做字数比例估算。 + * 优先使用后端基于 TTS 音频静音检测计算的精确 sentence_timings; + * 后端未返回(如对口型还在生成中)时,降级为前端按字数比例估算。 */ -import type { SentenceTiming } from "../types" export interface ScriptSentence { /** 句子序号(从 0 开始,对应提交给后端的 script_segment_index) */ @@ -13,75 +12,89 @@ export interface ScriptSentence { text: string /** 句子字数(按中文/字符计,去除空白) */ charCount: number - /** 累计起始字数 */ + /** 累计起始字数(用于时间估算) */ startChar: number - /** 精确起始时间(秒),来自后端 sentence_timings;无数据时为 0 */ + /** 对口型视频内起始时间(秒)——后端精确值或前端估算 */ startTime: number - /** 精确结束时间(秒),来自后端 sentence_timings;无数据时为 0 */ + /** 对口型视频内结束时间(秒)——后端精确值或前端估算 */ endTime: number } /** * 按句号/问号/感叹号/分号/换行分句(兼容中英文标点)。 - * 时间戳从后端 sentence_timings 获取(精确);若无则返回 0(由调用方降级处理)。 + * 空文案返回空数组。时间优先使用后端 sentence_timings;否则按字数线性估算。 + * + * @param sentenceTimings 后端返回的精确句子时间戳(来自 lipsync_job.sentence_timings)。 + * 非空且有效时优先采用,跳过前端估算。 */ export function splitScriptIntoSentences( scriptText: string, - sentenceTimings?: SentenceTiming[] | null, + sentenceTimings?: { index: number; text: string; start_time: number; end_time: number }[] | null, outputDuration: number = 0, ): ScriptSentence[] { const text = (scriptText || "").trim() if (!text) return [] + // 1. 先做基础分句(仅用于降级估算 / 没有 sentenceTimings 时) const rawParts = text .split(/[。!?!?;;\n\r]+/) .map((part) => part.trim()) .filter((part) => part.length > 0) - const sentences: ScriptSentence[] = [] - let accChar = 0 - const totalChars = rawParts.reduce((sum, p) => sum + p.replace(/\s/g, "").length, 0) - // 如果后端返回了完整的时间戳(至少有一个有效结束时间),使用精确时间 - // 需要满足:数组长度与分句数一致,且至少有一个结束时间 > 0(防御全0的异常数据) - const hasBackendTimings = - sentenceTimings && - sentenceTimings.length >= rawParts.length && - sentenceTimings.some((t) => (t.end_time ?? 0) > 0) - - // 若有后端时间戳,取音频总时长;否则用外部传入的 outputDuration 做字数比例降级 - const audioDuration = hasBackendTimings - ? (sentenceTimings![sentenceTimings!.length - 1]?.end_time ?? 0) - : outputDuration - - rawParts.forEach((part, i) => { - const charCount = part.replace(/\s/g, "").length - if (hasBackendTimings) { - // 使用后端精确时间戳 - const timing = sentenceTimings![i] - sentences.push({ - index: i, - text: part, - charCount, - startChar: accChar, - startTime: timing?.start_time ?? 0, - endTime: timing?.end_time ?? 0, - }) - } else { - // 降级方案:按字数比例估算(必须保证有时间,否则B-roll无法定位) - const startTime = totalChars > 0 ? (accChar / totalChars) * (audioDuration || 0) : 0 - const endTime = - totalChars > 0 ? ((accChar + charCount) / totalChars) * (audioDuration || 0) : 0 - sentences.push({ - index: i, - text: part, - charCount, - startChar: accChar, - startTime: Math.round(startTime * 100) / 100, - endTime: Math.round(endTime * 100) / 100, + // 2. 优先使用后端精确时间戳 + // 校验:必须是数组、条数一致、每条都有 start_time/end_time,否则降级估算 + if (Array.isArray(sentenceTimings) && sentenceTimings.length === rawParts.length) { + const valid = sentenceTimings.every( + (t) => + t && + typeof t.start_time === "number" && + typeof t.end_time === "number" && + t.end_time >= t.start_time, + ) + if (valid) { + let accChar = 0 + return sentenceTimings.map((t, i) => { + const part = rawParts[i] ?? t.text ?? "" + const charCount = part.replace(/\s/g, "").length + const sentence: ScriptSentence = { + index: t.index ?? i, + text: part, + charCount, + startChar: accChar, + startTime: round1(t.start_time), + endTime: round1(t.end_time), + } + accChar += charCount + return sentence }) } + } + + // 3. 降级:按字数比例线性估算 + const totalChars = rawParts.reduce((sum, part) => sum + part.replace(/\s/g, "").length, 0) + const duration = outputDuration > 0 ? outputDuration : 0 + + const sentences: ScriptSentence[] = [] + let accChar = 0 + rawParts.forEach((part, i) => { + const charCount = part.replace(/\s/g, "").length + const startTime = duration > 0 && totalChars > 0 ? (accChar / totalChars) * duration : 0 + const endTime = + duration > 0 && totalChars > 0 ? ((accChar + charCount) / totalChars) * duration : 0 + sentences.push({ + index: i, + text: part, + charCount, + startChar: accChar, + startTime: round1(startTime), + endTime: round1(endTime), + }) accChar += charCount }) return sentences } + +function round1(n: number): number { + return Math.round(n * 10) / 10 +} diff --git a/packages/domain/video_filter_builder.py b/packages/domain/video_filter_builder.py index 5a0ddd49f..496e68d5a 100755 --- a/packages/domain/video_filter_builder.py +++ b/packages/domain/video_filter_builder.py @@ -420,21 +420,44 @@ def _escape_drawtext_text(text: str) -> str: return result -def _resolve_font_path(font_name: str) -> str: +# 粗体字体文件映射:服务器镜像只保留了 NotoSansSC-VF.ttf(可变字体,已删除 +# NotoSansCJK-Bold.ttc 以避免 Mono 变体问题,见 worker-base.Dockerfile), +# 因此无法通过 fontfile 切换到 Bold 字重。这里保留路径列表作为未来扩展, +# 实际加粗通过 borderw 黑色描边实现(见下)。 +DRAWTEXT_BOLD_FONT_SEARCH_PATHS: list[str] = [ + "/usr/share/fonts/opentype/noto/NotoSansCJK-Bold.ttc", + "/usr/share/fonts/noto-cjk/NotoSansCJK-Bold.ttc", + "/usr/share/fonts/google-noto-cjk/NotoSansCJK-Bold.ttc", + "/usr/share/fonts/truetype/noto/NotoSansSC-Bold.ttf", + "/usr/share/fonts/noto/NotoSansSC-Bold.ttf", +] + + +def _resolve_font_path(font_name: str, bold: bool = False) -> str: """解析字体名到服务器实际字体文件路径。 查找策略: 1. 通过 DRAWTEXT_FONT_MAP 映射前端字体名到服务器关键字 - 2. 在 DRAWTEXT_FONT_SEARCH_PATHS 中查找匹配路径 - 3. 未找到则返回空字符串(drawtext 使用内置默认字体) + 2. bold=True 时优先查找粗体变体;找不到回退常规字重 + 3. 在 DRAWTEXT_FONT_SEARCH_PATHS 中查找匹配路径 + 4. 未找到则返回空字符串(drawtext 使用内置默认字体) """ keyword = DRAWTEXT_FONT_MAP.get(font_name, font_name) import os + if bold: + for path in DRAWTEXT_BOLD_FONT_SEARCH_PATHS: + if keyword.lower() in path.lower() and os.path.isfile(path): + return path + # 粗体文件找不到时,再查常规字重(后面会用描边兜底加粗) for path in DRAWTEXT_FONT_SEARCH_PATHS: if keyword.lower() in path.lower() and os.path.isfile(path): return path # fallback:遍历搜索任意可用字体 + if bold: + for path in DRAWTEXT_BOLD_FONT_SEARCH_PATHS: + if os.path.isfile(path): + return path for path in DRAWTEXT_FONT_SEARCH_PATHS: if os.path.isfile(path): return path @@ -495,8 +518,8 @@ def build_title_drawtext_filter( # ── 构建 drawtext 参数 ── params: list[str] = [] - # 字体文件 - font_path = _resolve_font_path(font_name) + # 字体文件:粗体优先使用 Bold 字体文件,避免同色描边造成字形偏移/重影 + font_path = _resolve_font_path(font_name, bold=bold) if font_path: escaped_path = font_path.replace("\\", "\\\\").replace(":", "\\\\:").replace("'", "\\\\'") params.append(f"fontfile='{escaped_path}'") @@ -508,13 +531,11 @@ def build_title_drawtext_filter( params.append(f"fontsize={font_size}") params.append(f"fontcolor={font_color}") - # 粗体:drawtext 没有独立的 bold 参数,通过加大 borderw 模拟视觉粗体效果。 - # 注意:不能使用 `font=bold`——FFmpeg drawtext 的 font 参数需要 fontconfig 能解析的 - # 字体族名,而 "bold" 不是合法族名,会导致整个 filter_complex 解析失败(exit code 234)。 - # 当用户未显式配置描边宽度时,bold 模式自动将 borderw 提升到 3 以模拟粗体。 - # 描边(borderw 需要 libfreetype 支持) - # 粗体无显式描边时,自动用 borderw=3 + 近色描边模拟粗体;显式 stroke 按用户配置走 + # 之前用 borderw=3 + font_color 同色描边模拟粗体,会在小字号/竖屏视频上造成 + # 字形偏移、边缘重影,看起来像文字被打印了两次(用户截图中的标题"曝光曝光…")。 + # 修复:粗体改用黑色细描边(borderw=2, 黑色),视觉上清晰加粗且不产生偏移。 + # 用户显式开启 stroke 时按用户配置走;粗体+无stroke 默认黑色细描边。 border_width = 0 border_color = "000000" if stroke: @@ -526,9 +547,9 @@ def build_title_drawtext_filter( border_width = int(stroke.get("width", 2)) border_color = (stroke.get("color") or "#000000").lstrip("#") elif bold: - # 粗体模式且未配描边:加大描边宽度模拟粗体效果 - border_width = 3 - border_color = font_color # 用字体同色描边,视觉上加粗字形而非黑边 + # 粗体模式且未配描边:黑色细描边,模拟粗体同时保证不重影 + border_width = 2 + border_color = "000000" if border_width > 0: params.append(f"borderw={border_width}") params.append(f"bordercolor={border_color}")