From af0216fd3283d0775fdedc2e3fbc11dd5e48ca77 Mon Sep 17 00:00:00 2001 From: LingYing Agent Date: Sat, 12 Sep 2026 01:36:33 +0800 Subject: [PATCH] fix: CI lint/type fixes - Fix unused variable 'outputDuration' in ModalBRollEditor (ESLint) - Fix CSS duplicate 'left'/'transform' in PanelLipsyncPreview (TS2783) - Add script_text to LipsyncJob frontend type (TS2339) - Fix ruff W605 invalid escape sequences in lipsync_tts.py - Apply black formatting to lipsync_tts.py and video_filter_builder.py - Align default font_size (28) between frontend and backend --- apps/api/app/tasks/lipsync_tts.py | 73 +++++++++++++------ .../ai-avatar/components/ModalBRollEditor.tsx | 2 +- .../components/PanelLipsyncPreview.tsx | 10 +-- apps/web/src/pages/ai-avatar/types.ts | 1 + packages/domain/video_filter_builder.py | 5 +- 5 files changed, 58 insertions(+), 33 deletions(-) diff --git a/apps/api/app/tasks/lipsync_tts.py b/apps/api/app/tasks/lipsync_tts.py index 7e28b928e..4dd094488 100644 --- a/apps/api/app/tasks/lipsync_tts.py +++ b/apps/api/app/tasks/lipsync_tts.py @@ -54,11 +54,10 @@ def _sign_media_url(url: str) -> str: return url - - def _split_script_into_sentences(script_text: str) -> list[str]: """按句号/问号/感叹号/分号/换行分句(与前端 splitScriptIntoSentences 一致).""" import re + text = (script_text or "").strip() if not text: return [] @@ -97,9 +96,14 @@ def _compute_sentence_timings(audio_data: bytes, script_text: str, total_duratio # 用 ffmpeg silencedetect 检测静音段 result = subprocess.run( [ - "ffmpeg", "-i", tmp_path, - "-af", "silencedetect=noise=-25dB:d=0.3", - "-f", "null", "-", + "ffmpeg", + "-i", + tmp_path, + "-af", + "silencedetect=noise=-25dB:d=0.3", + "-f", + "null", + "-", ], capture_output=True, text=True, @@ -118,7 +122,8 @@ def _compute_sentence_timings(audio_data: bytes, script_text: str, total_duratio if len(silence_ends) < len(sentences) - 1: logger.warning( "[sentence_timings] 静音点不足(%d < %d),降级为字数比例估算", - len(silence_ends), len(sentences) - 1, + len(silence_ends), + len(sentences) - 1, ) return _estimate_sentence_timings_by_chars(sentences, total_duration) @@ -153,12 +158,14 @@ def _compute_sentence_timings(audio_data: bytes, script_text: str, total_duratio for i, sent in enumerate(sentences): start = prev_end end = boundaries[i] if i < len(boundaries) else total_duration - timings.append({ - "index": i, - "text": sent, - "start_time": round(start, 2), - "end_time": round(end, 2), - }) + timings.append( + { + "index": i, + "text": sent, + "start_time": round(start, 2), + "end_time": round(end, 2), + } + ) prev_end = end return timings @@ -168,6 +175,7 @@ def _compute_sentence_timings(audio_data: bytes, script_text: str, total_duratio return _estimate_sentence_timings_by_chars(sentences, total_duration) finally: import os + try: os.unlink(tmp_path) except Exception: @@ -178,25 +186,28 @@ def _estimate_sentence_timings_by_chars(sentences: list[str], total_duration: fl """降级方案:按字数比例估算句子时间(与原前端逻辑一致).""" if not sentences or total_duration <= 0: return [] - total_chars = sum(len(s.replace("\s", "")) for s in sentences) + total_chars = sum(len(s.replace(r"\s", "")) for s in sentences) if total_chars == 0: return [] timings = [] acc = 0 for i, sent in enumerate(sentences): - chars = len(sent.replace("\s", "")) + chars = len(sent.replace(r"\s", "")) start = (acc / total_chars) * total_duration end = ((acc + chars) / total_chars) * total_duration - timings.append({ - "index": i, - "text": sent, - "start_time": round(start, 2), - "end_time": round(end, 2), - }) + timings.append( + { + "index": i, + "text": sent, + "start_time": round(start, 2), + "end_time": round(end, 2), + } + ) acc += chars return timings + @shared_task( bind=True, name="lipsync_tts.synthesize_and_submit", @@ -324,6 +335,7 @@ def tts_synthesize_and_submit( # 2.5 计算精确句子时间戳(基于 TTS 音频静音检测) import os as _os + _st_tmp_path = None try: import subprocess as _sp @@ -332,6 +344,7 @@ def tts_synthesize_and_submit( # 下载音频用于探测时长和静音检测 if isinstance(job.audio_url, str) and job.audio_url: from packages.shared.url_security import safe_download_bytes as _sdl + _audio_bytes = _sdl(job.audio_url, purpose="sentence_timings", timeout=30.0) else: _audio_bytes = audio_data @@ -342,9 +355,19 @@ def tts_synthesize_and_submit( _st_tmp_path = _atmp.name _probe_result = _sp.run( - ["ffprobe", "-v", "error", "-show_entries", "format=duration", - "-of", "default=noprint_wrappers=1:nokey=1", _st_tmp_path], - capture_output=True, text=True, timeout=10, + [ + "ffprobe", + "-v", + "error", + "-show_entries", + "format=duration", + "-of", + "default=noprint_wrappers=1:nokey=1", + _st_tmp_path, + ], + capture_output=True, + text=True, + timeout=10, ) _audio_duration = float(_probe_result.stdout.strip()) if _probe_result.stdout.strip() else 0.0 @@ -354,7 +377,9 @@ def tts_synthesize_and_submit( job.sentence_timings = _timings logger.info( "[lipsync_tts] 句子时间戳已计算: job_id=%s sentences=%d duration=%.1f", - job_id, len(_timings), _audio_duration, + job_id, + len(_timings), + _audio_duration, ) db.commit() except Exception as _st_err: diff --git a/apps/web/src/pages/ai-avatar/components/ModalBRollEditor.tsx b/apps/web/src/pages/ai-avatar/components/ModalBRollEditor.tsx index 336540973..fbda50381 100644 --- a/apps/web/src/pages/ai-avatar/components/ModalBRollEditor.tsx +++ b/apps/web/src/pages/ai-avatar/components/ModalBRollEditor.tsx @@ -45,7 +45,7 @@ const ModalBRollEditor: React.FC = ({ onClose, existingSegments, scriptText, - outputDuration, + outputDuration: _outputDuration, sentenceTimings, onConfirm, onRemove, diff --git a/apps/web/src/pages/ai-avatar/components/PanelLipsyncPreview.tsx b/apps/web/src/pages/ai-avatar/components/PanelLipsyncPreview.tsx index e9a0b4b38..f78c951aa 100755 --- a/apps/web/src/pages/ai-avatar/components/PanelLipsyncPreview.tsx +++ b/apps/web/src/pages/ai-avatar/components/PanelLipsyncPreview.tsx @@ -56,11 +56,9 @@ export function PanelLipsyncPreview({ const titleOverlayStyle: React.CSSProperties | null = titleConfig?.title ? { position: "absolute", - left: "50%", - transform: "translateX(-50%)", color: titleConfig.color || "#ffffff", fontFamily: titleConfig.font || "思源黑体", - fontSize: `${(titleConfig.size || 36) * 0.55}px`, // 预览等比缩 + fontSize: `${(titleConfig.size || 28) * 0.55}px`, fontWeight: titleConfig.bold ? 700 : 400, fontStyle: titleConfig.italic ? "italic" : "normal", textAlign: "center", @@ -77,10 +75,10 @@ export function PanelLipsyncPreview({ transform: "translateX(-50%) translateY(-50%)", } : titleConfig.position === "top" - ? { top: 8 } + ? { left: "50%", top: 8, transform: "translateX(-50%)" } : titleConfig.position === "bottom" - ? { bottom: 8 } - : { top: "50%", transform: "translateX(-50%) translateY(-50%)" }), + ? { left: "50%", bottom: 8, transform: "translateX(-50%)" } + : { left: "50%", top: "50%", transform: "translateX(-50%) translateY(-50%)" }), } : null diff --git a/apps/web/src/pages/ai-avatar/types.ts b/apps/web/src/pages/ai-avatar/types.ts index a06be59d7..d149a3d3c 100644 --- a/apps/web/src/pages/ai-avatar/types.ts +++ b/apps/web/src/pages/ai-avatar/types.ts @@ -45,6 +45,7 @@ export interface LipsyncJob { progress: number output_video_url: string | null /** 对口型成片总时长(秒),后端返回 */ + script_text: string output_duration?: number /** 精确句子时间戳(后端基于 TTS 音频静音检测计算) */ sentence_timings?: SentenceTiming[] | null diff --git a/packages/domain/video_filter_builder.py b/packages/domain/video_filter_builder.py index e44f80dd9..6b3062cd5 100755 --- a/packages/domain/video_filter_builder.py +++ b/packages/domain/video_filter_builder.py @@ -700,7 +700,9 @@ def _build_fullscreen_filters( for idx, seg in enumerate(sorted_fs_segments): start = seg.get("start_time", 0) # 每段 B-roll 之前是否有主视频片段? - has_main_before = (idx == 0 and start > 0) or (idx > 0 and sorted_fs_segments[idx - 1].get("end_time", 0) < start) + has_main_before = (idx == 0 and start > 0) or ( + idx > 0 and sorted_fs_segments[idx - 1].get("end_time", 0) < start + ) if has_main_before: segment_labels.append(f"[main{idx}]") segment_labels.append(f"[br{idx}]") @@ -770,7 +772,6 @@ def _build_pip_filters( return "".join(parts), cur_label or "vout" - def build_cover_extract_command( cover_config: dict[str, Any], output_path: str,