diff --git a/apps/api/app/tasks/lipsync_tts.py b/apps/api/app/tasks/lipsync_tts.py index 4495f8397..b097d1beb 100644 --- a/apps/api/app/tasks/lipsync_tts.py +++ b/apps/api/app/tasks/lipsync_tts.py @@ -334,6 +334,7 @@ def tts_synthesize_and_submit( db.commit() # 2.5 计算精确句子时间戳(基于 TTS 音频静音检测) + # 直接复用步骤 2 已下载到内存的 audio_data,避免重新从 OSS 下载(私有桶未签名会失败) import os as _os _st_tmp_path = None @@ -341,49 +342,63 @@ def tts_synthesize_and_submit( import subprocess as _sp import tempfile as _tmpf - # 下载音频用于探测时长和静音检测 - if isinstance(job.audio_url, str) and job.audio_url: - from packages.shared.url_security import safe_download_bytes as _sdl - - _audio_bytes = _sdl(job.audio_url, purpose="sentence_timings", timeout=30.0) + if not audio_data: + logger.warning("[lipsync_tts] 无音频数据,跳过句子时间戳计算: job_id=%s", job_id) else: - _audio_bytes = audio_data + # 写入临时文件供 ffprobe/ffmpeg 使用 + with _tmpf.NamedTemporaryFile(suffix=".mp3", delete=False) as _atmp: + _atmp.write(audio_data) + _st_tmp_path = _atmp.name - # ffprobe 获取音频时长 - with _tmpf.NamedTemporaryFile(suffix=".mp3", delete=False) as _atmp: - _atmp.write(_audio_bytes) - _st_tmp_path = _atmp.name + # ffprobe 获取音频时长 + _probe_result = _sp.run( + [ + "ffprobe", + "-v", + "error", + "-show_entries", + "format=duration", + "-of", + "default=noprint_wrappers=1:nokey=1", + _st_tmp_path, + ], + capture_output=True, + text=True, + timeout=10, + ) + _audio_duration = float(_probe_result.stdout.strip()) if _probe_result.stdout.strip() else 0.0 + logger.info( + "[lipsync_tts] 音频时长探测: job_id=%s duration=%.2f probe_stdout=%s probe_stderr=%s", + job_id, + _audio_duration, + _probe_result.stdout.strip()[:50], + _probe_result.stderr.strip()[:100] if _probe_result.stderr else "", + ) - _probe_result = _sp.run( - [ - "ffprobe", - "-v", - "error", - "-show_entries", - "format=duration", - "-of", - "default=noprint_wrappers=1:nokey=1", - _st_tmp_path, - ], - capture_output=True, - text=True, - timeout=10, - ) - _audio_duration = float(_probe_result.stdout.strip()) if _probe_result.stdout.strip() else 0.0 - - if _audio_duration > 0: - _timings = _compute_sentence_timings(_audio_bytes, script_text, _audio_duration) - if _timings: - job.sentence_timings = _timings - logger.info( - "[lipsync_tts] 句子时间戳已计算: job_id=%s sentences=%d duration=%.1f", + if _audio_duration > 0: + _timings = _compute_sentence_timings(audio_data, script_text, _audio_duration) + if _timings: + job.sentence_timings = _timings + logger.info( + "[lipsync_tts] 句子时间戳已计算: job_id=%s sentences=%d duration=%.1f", + job_id, + len(_timings), + _audio_duration, + ) + else: + logger.warning("[lipsync_tts] 句子时间戳计算返回空结果: job_id=%s", job_id) + else: + logger.warning( + "[lipsync_tts] ffprobe 未获取到有效时长,跳过句子时间戳: job_id=%s stdout=%s stderr=%s", job_id, - len(_timings), - _audio_duration, + _probe_result.stdout.strip()[:100], + _probe_result.stderr.strip()[:200] if _probe_result.stderr else "", ) - db.commit() + db.commit() except Exception as _st_err: - logger.warning("[lipsync_tts] 句子时间戳计算失败(不影响主流程): job_id=%s err=%s", job_id, _st_err) + logger.warning( + "[lipsync_tts] 句子时间戳计算失败(不影响主流程): job_id=%s err=%s", job_id, _st_err, exc_info=True + ) finally: if _st_tmp_path: try: diff --git a/apps/web/src/pages/ai-avatar/utils/sentences.ts b/apps/web/src/pages/ai-avatar/utils/sentences.ts index 6f8c5b86c..6c3ad70e1 100644 --- a/apps/web/src/pages/ai-avatar/utils/sentences.ts +++ b/apps/web/src/pages/ai-avatar/utils/sentences.ts @@ -42,9 +42,10 @@ export function splitScriptIntoSentences( let accChar = 0 const totalChars = rawParts.reduce((sum, p) => sum + p.replace(/\s/g, "").length, 0) // 如果后端返回了完整的时间戳(至少有一个有效结束时间),使用精确时间 + // 需要满足:数组长度与分句数一致,且至少有一个结束时间 > 0(防御全0的异常数据) const hasBackendTimings = sentenceTimings && - sentenceTimings.length > 0 && + sentenceTimings.length >= rawParts.length && sentenceTimings.some((t) => (t.end_time ?? 0) > 0) // 若有后端时间戳,取音频总时长;否则用外部传入的 outputDuration 做字数比例降级