diff --git a/apps/api/app/tasks/lipsync_tts.py b/apps/api/app/tasks/lipsync_tts.py index 187e0be07..7205bfe03 100644 --- a/apps/api/app/tasks/lipsync_tts.py +++ b/apps/api/app/tasks/lipsync_tts.py @@ -55,13 +55,17 @@ def _sign_media_url(url: str) -> str: def _split_script_into_sentences(script_text: str) -> list[str]: - """按句号/问号/感叹号/分号/换行分句(与前端 splitScriptIntoSentences 一致).""" + """按句号/问号/感叹号/分号/逗号/换行分句(与前端 SENTENCE_SPLIT_RE 一致). + + 中文短视频文案习惯用「,」断小句(如"卖花的叫花无缺,卖姜的叫姜子牙"), + 必须把逗号也纳入分隔符,否则多句文案会被识别成一整句,导致 B-roll 时间戳错位。 + """ import re text = (script_text or "").strip() if not text: return [] - parts = re.split(r"[。!?!?;;\n\r]+", text) + parts = re.split(r"[。!?!??!;;,,\n\r]+", text) return [p.strip() for p in parts if p.strip()] diff --git a/apps/web/src/pages/ai-avatar/utils/sentences.ts b/apps/web/src/pages/ai-avatar/utils/sentences.ts index 0c4f1d74b..b8ffd43bc 100644 --- a/apps/web/src/pages/ai-avatar/utils/sentences.ts +++ b/apps/web/src/pages/ai-avatar/utils/sentences.ts @@ -1,8 +1,10 @@ /** * AI数字人 — 文案分句 & B-roll 时间计算 * - * 优先使用后端基于 TTS 音频静音检测计算的精确 sentence_timings; - * 后端未返回(如对口型还在生成中)时,降级为前端按字数比例估算。 + * 数据来源优先级: + * 1. 后端 sentence_timings(基于 TTS 音频静音检测,精确到句子边界)—— 直接使用,不重新分句 + * 2. 后端 output_duration(最终渲染视频时长) + 本地分句 —— 按字数比例估算 + * 3. 两者都没有(对口型还在生成中)—— 返回分句文本但 startTime/endTime 全部 0,等数据到位重算 */ export interface ScriptSentence { @@ -20,45 +22,46 @@ export interface ScriptSentence { endTime: number } +/** 句子分隔符:中英文句号/问号/感叹号/分号/逗号/换行(覆盖中文短视频常用断句) */ +const SENTENCE_SPLIT_RE = /[。!?!??!;;,,\n\r]+/ + /** - * 按句号/问号/感叹号/分号/换行分句(兼容中英文标点)。 - * 空文案返回空数组。时间优先使用后端 sentence_timings;否则按字数线性估算。 + * 分句并计算每句的起止时间。 * * @param sentenceTimings 后端返回的精确句子时间戳(来自 lipsync_job.sentence_timings)。 - * 非空且有效时优先采用,跳过前端估算。 + * 非空时直接按后端返回的句子列表渲染,不再本地分句(避免前后端分句不一致导致时间错位)。 + * @param outputDuration 最终视频时长(秒)。对口型预览阶段可能为 0,此时降级估算只能给 0。 */ export function splitScriptIntoSentences( scriptText: string, - sentenceTimings?: { index: number; text: string; start_time: number; end_time: number }[] | null, + sentenceTimings?: { index?: number; text?: string; start_time: number; end_time: number }[] | null, outputDuration: number = 0, ): ScriptSentence[] { const text = (scriptText || "").trim() if (!text) return [] - // 1. 先做基础分句(仅用于降级估算 / 没有 sentenceTimings 时) - const rawParts = text - .split(/[。!?!?;;\n\r]+/) - .map((part) => part.trim()) - .filter((part) => part.length > 0) - - // 2. 优先使用后端精确时间戳 - // 校验:必须是数组、条数一致、每条都有 start_time/end_time,否则降级估算 - if (Array.isArray(sentenceTimings) && sentenceTimings.length === rawParts.length) { + // 1. 后端返回了 sentence_timings:校验通过就直接用,跳过本地分句 + // 校验条件放宽:只要是数组、至少1条、每条 start_time/end_time 是数字即可 + // (不再强制要求条数相等——后端静音检测可能按停顿切出更多/更少边界, + // 比如文案用逗号连写时本地只分1句、后端按停顿切4句,后端的切法才是对的) + if (Array.isArray(sentenceTimings) && sentenceTimings.length > 0) { const valid = sentenceTimings.every( (t) => t && typeof t.start_time === "number" && typeof t.end_time === "number" && + isFinite(t.start_time) && + isFinite(t.end_time) && t.end_time >= t.start_time, ) if (valid) { let accChar = 0 return sentenceTimings.map((t, i) => { - const part = rawParts[i] ?? t.text ?? "" - const charCount = part.replace(/\s/g, "").length + const sentenceText = (t.text || "").trim() || `句子${i + 1}` + const charCount = sentenceText.replace(/\s/g, "").length const sentence: ScriptSentence = { - index: t.index ?? i, - text: part, + index: typeof t.index === "number" ? t.index : i, + text: sentenceText, charCount, startChar: accChar, startTime: round1(t.start_time), @@ -70,7 +73,12 @@ export function splitScriptIntoSentences( } } - // 3. 降级:按字数比例线性估算 + // 2. 本地分句 + 按字数比例估算(降级路径) + const rawParts = text + .split(SENTENCE_SPLIT_RE) + .map((part) => part.trim()) + .filter((part) => part.length > 0) + const totalChars = rawParts.reduce((sum, part) => sum + part.replace(/\s/g, "").length, 0) const duration = outputDuration > 0 ? outputDuration : 0