From 79e6bd52b814fcaae5653bdc6f1c368d0143544a Mon Sep 17 00:00:00 2001 From: LingYing Agent Date: Sat, 12 Sep 2026 01:23:22 +0800 Subject: [PATCH] =?UTF-8?q?fix(ai-avatar):=20=E7=AB=AF=E5=88=B0=E7=AB=AF?= =?UTF-8?q?=E4=B8=80=E8=87=B4=E6=80=A7=E4=BF=AE=E5=A4=8D=EF=BC=88=E6=A0=87?= =?UTF-8?q?=E9=A2=98/=E5=B0=81=E9=9D=A2/B-roll=E4=BD=8D=E7=BD=AE/B-roll?= =?UTF-8?q?=E6=97=B6=E9=95=BF=EF=BC=89?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit 修复5个问题: 1. 最终视频标题一致性 - 后端 drawtext 默认 font_size 从 36 改为 28(对齐前端 DEFAULT_TITLE_CONFIG.size) - 后端 drawtext 默认 position 从 'top' 改为 'bottom'(对齐前端默认值) - 前端 contract.ts font_size fallback 从 36 改为 28 2. 封面与视频一致性 - 渲染完成后自动用 output_cover_url 更新前端封面配置 - 前端封面标题预览 fontSize 改用 size*0.55 缩放(与对口型预览对齐) - 封面默认阴影改为 none(与后端 drawtext 无 shadow 行为对齐) 3. B-roll 位置一致性 - ModalBRollEditor 使用 lipsyncJob.script_text(TTS 时锁定的文案) 而非 state.scriptText(用户可能已修改的文案) 4. B-roll 时长精确化 - 后端 lipsync_tts.py 新增 _compute_sentence_timings(): TTS 音频用 ffmpeg silencedetect 检测静音点,与句子边界对齐 - 新增 sentence_timings JSON 字段到 LipsyncJobModel + Alembic migration - 前端 sentences.ts 不再做字数比例估算,改为接收后端精确时间戳 - ModalBRollEditor 接收 sentenceTimings prop 5. 标题拖拽位置修复 - PanelLipsyncPreview 拖拽标题发送百分比坐标(0-100) + position:'custom' - 后端 build_title_drawtext_filter 的 custom 位置改用 drawtext 表达式 x=(w-text_w)*{pct_x} y=(h-text_h)*{pct_y}(百分比,与视频尺寸解耦) 文件变更: - apps/api/app/tasks/lipsync_tts.py: 句子时间戳计算 - apps/api/app/schemas/lipsync.py: sentence_timings 字段 - packages/adapters/sqlalchemy_impl/models.py: sentence_timings 列 - packages/domain/video_filter_builder.py: 百分比坐标 + 默认值对齐 - alembic/versions/075_add_sentence_timings.py: 数据库迁移 - apps/web/src/pages/ai-avatar/types.ts: SentenceTiming + RenderJob.output_cover_url - apps/web/src/pages/ai-avatar/utils/sentences.ts: 移除字数估算 - apps/web/src/pages/ai-avatar/utils/contract.ts: font_size 默认值 - apps/web/src/pages/ai-avatar/components/PanelLipsyncPreview.tsx: 拖拽百分比 - apps/web/src/pages/ai-avatar/components/PanelCoverAndGenerate.tsx: 标题缩放 - apps/web/src/pages/ai-avatar/components/ModalBRollEditor.tsx: 后端时间戳 - apps/web/src/pages/ai-avatar/AiAvatarPage.tsx: 封面更新 + 传参 - tests/unit/test_video_filter_builder.py: 测试更新 --- alembic/versions/075_add_sentence_timings.py | 27 +++ apps/api/app/schemas/lipsync.py | 1 + apps/api/app/tasks/lipsync_tts.py | 187 ++++++++++++++++++ apps/web/src/pages/ai-avatar/AiAvatarPage.tsx | 12 +- .../ai-avatar/components/ModalBRollEditor.tsx | 21 +- .../components/PanelCoverAndGenerate.tsx | 6 +- .../components/PanelLipsyncPreview.tsx | 21 +- apps/web/src/pages/ai-avatar/types.ts | 15 +- .../web/src/pages/ai-avatar/utils/contract.ts | 2 +- .../src/pages/ai-avatar/utils/sentences.ts | 35 ++-- packages/adapters/sqlalchemy_impl/models.py | 3 + packages/domain/video_filter_builder.py | 13 +- tests/unit/test_video_filter_builder.py | 17 +- 13 files changed, 311 insertions(+), 49 deletions(-) create mode 100644 alembic/versions/075_add_sentence_timings.py diff --git a/alembic/versions/075_add_sentence_timings.py b/alembic/versions/075_add_sentence_timings.py new file mode 100644 index 000000000..09b4246ff --- /dev/null +++ b/alembic/versions/075_add_sentence_timings.py @@ -0,0 +1,27 @@ +"""add sentence_timings to lipsync_jobs + +Revision ID: 075_add_sentence_timings +Revises: 074_ai_avatar_render_script_id_optional +Create Date: 2026-09-12 +""" + +import sqlalchemy as sa + +from alembic import op + +revision = "075_add_sentence_timings" +down_revision = "074_ai_avatar_render_script_id_optional" +branch_labels = None +depends_on = None + + +def upgrade() -> None: + with op.batch_alter_table("lipsync_jobs") as batch: + batch.add_column( + sa.Column("sentence_timings", sa.JSON(), nullable=True), + ) + + +def downgrade() -> None: + with op.batch_alter_table("lipsync_jobs") as batch: + batch.drop_column("sentence_timings") diff --git a/apps/api/app/schemas/lipsync.py b/apps/api/app/schemas/lipsync.py index 77fe9fa5b..47b306f43 100644 --- a/apps/api/app/schemas/lipsync.py +++ b/apps/api/app/schemas/lipsync.py @@ -33,6 +33,7 @@ class LipsyncJobResponse(BaseModel): output_duration: float error_message: str error_code: str + sentence_timings: Optional[list] = None submitted_at: Optional[datetime] = None completed_at: Optional[datetime] = None created_at: datetime diff --git a/apps/api/app/tasks/lipsync_tts.py b/apps/api/app/tasks/lipsync_tts.py index d241c0b89..7e28b928e 100644 --- a/apps/api/app/tasks/lipsync_tts.py +++ b/apps/api/app/tasks/lipsync_tts.py @@ -54,6 +54,149 @@ def _sign_media_url(url: str) -> str: return url + + +def _split_script_into_sentences(script_text: str) -> list[str]: + """按句号/问号/感叹号/分号/换行分句(与前端 splitScriptIntoSentences 一致).""" + import re + text = (script_text or "").strip() + if not text: + return [] + parts = re.split(r"[。!?!?;;\n\r]+", text) + return [p.strip() for p in parts if p.strip()] + + +def _compute_sentence_timings(audio_data: bytes, script_text: str, total_duration: float) -> list[dict]: + """基于 TTS 音频的静音检测,精确计算每句文案的起止时间. + + 使用 ffmpeg silencedetect 检测静音段,将静音点与句子边界对齐。 + 比字数比例估算准确得多。 + + Args: + audio_data: TTS 音频二进制数据(MP3) + script_text: 文案全文 + total_duration: 音频总时长(秒) + + Returns: + list[{"index": int, "text": str, "start_time": float, "end_time": float}] + """ + import re + import subprocess + import tempfile + + sentences = _split_script_into_sentences(script_text) + if not sentences: + return [] + + # 写入临时音频文件 + with tempfile.NamedTemporaryFile(suffix=".mp3", delete=False) as tmp: + tmp.write(audio_data) + tmp_path = tmp.name + + try: + # 用 ffmpeg silencedetect 检测静音段 + result = subprocess.run( + [ + "ffmpeg", "-i", tmp_path, + "-af", "silencedetect=noise=-25dB:d=0.3", + "-f", "null", "-", + ], + capture_output=True, + text=True, + timeout=30, + ) + stderr = result.stderr or "" + + # 解析静音结束时间点(silence_end: X.XXX) + silence_ends = [] + for match in re.finditer(r"silence_end:\s*([\d.]+)", stderr): + t = float(match.group(1)) + if 0 < t < total_duration: + silence_ends.append(t) + + # 如果没有检测到足够的静音点,降级为字数比例估算 + if len(silence_ends) < len(sentences) - 1: + logger.warning( + "[sentence_timings] 静音点不足(%d < %d),降级为字数比例估算", + len(silence_ends), len(sentences) - 1, + ) + return _estimate_sentence_timings_by_chars(sentences, total_duration) + + # 贪心匹配:N-1 个句子边界对应 N-1 个静音点 + # 按时间均匀分布期望值,选择最近的静音点 + n_boundaries = len(sentences) - 1 + boundaries = [] + used_indices = set() + + for i in range(n_boundaries): + # 期望的边界位置(按句子数量均匀分布) + expected_pos = (i + 1) / len(sentences) * total_duration + # 找最近的未使用静音点 + best_idx = None + best_dist = float("inf") + for j, t in enumerate(silence_ends): + if j in used_indices: + continue + dist = abs(t - expected_pos) + if dist < best_dist: + best_dist = dist + best_idx = j + if best_idx is not None: + used_indices.add(best_idx) + boundaries.append(silence_ends[best_idx]) + + boundaries.sort() + + # 构建 sentence_timings + timings = [] + prev_end = 0.0 + for i, sent in enumerate(sentences): + start = prev_end + end = boundaries[i] if i < len(boundaries) else total_duration + timings.append({ + "index": i, + "text": sent, + "start_time": round(start, 2), + "end_time": round(end, 2), + }) + prev_end = end + + return timings + + except Exception as exc: + logger.warning("[sentence_timings] 静音检测异常,降级为字数比例估算: %s", exc) + return _estimate_sentence_timings_by_chars(sentences, total_duration) + finally: + import os + try: + os.unlink(tmp_path) + except Exception: + pass + + +def _estimate_sentence_timings_by_chars(sentences: list[str], total_duration: float) -> list[dict]: + """降级方案:按字数比例估算句子时间(与原前端逻辑一致).""" + if not sentences or total_duration <= 0: + return [] + total_chars = sum(len(s.replace("\s", "")) for s in sentences) + if total_chars == 0: + return [] + + timings = [] + acc = 0 + for i, sent in enumerate(sentences): + chars = len(sent.replace("\s", "")) + start = (acc / total_chars) * total_duration + end = ((acc + chars) / total_chars) * total_duration + timings.append({ + "index": i, + "text": sent, + "start_time": round(start, 2), + "end_time": round(end, 2), + }) + acc += chars + return timings + @shared_task( bind=True, name="lipsync_tts.synthesize_and_submit", @@ -179,6 +322,50 @@ def tts_synthesize_and_submit( db.commit() + # 2.5 计算精确句子时间戳(基于 TTS 音频静音检测) + import os as _os + _st_tmp_path = None + try: + import subprocess as _sp + import tempfile as _tmpf + + # 下载音频用于探测时长和静音检测 + if isinstance(job.audio_url, str) and job.audio_url: + from packages.shared.url_security import safe_download_bytes as _sdl + _audio_bytes = _sdl(job.audio_url, purpose="sentence_timings", timeout=30.0) + else: + _audio_bytes = audio_data + + # ffprobe 获取音频时长 + with _tmpf.NamedTemporaryFile(suffix=".mp3", delete=False) as _atmp: + _atmp.write(_audio_bytes) + _st_tmp_path = _atmp.name + + _probe_result = _sp.run( + ["ffprobe", "-v", "error", "-show_entries", "format=duration", + "-of", "default=noprint_wrappers=1:nokey=1", _st_tmp_path], + capture_output=True, text=True, timeout=10, + ) + _audio_duration = float(_probe_result.stdout.strip()) if _probe_result.stdout.strip() else 0.0 + + if _audio_duration > 0: + _timings = _compute_sentence_timings(_audio_bytes, script_text, _audio_duration) + if _timings: + job.sentence_timings = _timings + logger.info( + "[lipsync_tts] 句子时间戳已计算: job_id=%s sentences=%d duration=%.1f", + job_id, len(_timings), _audio_duration, + ) + db.commit() + except Exception as _st_err: + logger.warning("[lipsync_tts] 句子时间戳计算失败(不影响主流程): job_id=%s err=%s", job_id, _st_err) + finally: + if _st_tmp_path: + try: + _os.unlink(_st_tmp_path) + except Exception: + pass + # 3. 签名 URL 并提交到 MediaKit(复用模块内 _sign_media_url,避免对 LipsyncService 的耦合) audio_url = _sign_media_url(job.audio_url) video_url = _sign_media_url(job.video_url) diff --git a/apps/web/src/pages/ai-avatar/AiAvatarPage.tsx b/apps/web/src/pages/ai-avatar/AiAvatarPage.tsx index e9b244335..05fc5bff0 100644 --- a/apps/web/src/pages/ai-avatar/AiAvatarPage.tsx +++ b/apps/web/src/pages/ai-avatar/AiAvatarPage.tsx @@ -241,6 +241,15 @@ const AiAvatarPage: React.FC = () => { if (renderTimerRef.current) clearInterval(renderTimerRef.current) renderTimerRef.current = null setRenderStatus("completed") + // 渲染完成后,用最终视频的封面更新前端封面配置 + if (updated.output_cover_url) { + state.setCoverConfig((prev) => ({ + ...prev, + mode: "auto_frame", + smart_cover_url: updated.output_cover_url, + thumbnail_url: updated.output_cover_url, + })) + } message.success("视频已生成并保存到成片库") } else if (updated.status === "failed") { if (renderTimerRef.current) clearInterval(renderTimerRef.current) @@ -488,8 +497,9 @@ const AiAvatarPage: React.FC = () => { open={state.showBRollModal} onClose={() => state.setShowBRollModal(false)} existingSegments={state.bRollSegments} - scriptText={state.scriptText} + scriptText={state.lipsyncJob?.script_text || state.scriptText} outputDuration={state.lipsyncJob?.output_duration ?? 0} + sentenceTimings={state.lipsyncJob?.sentence_timings} onConfirm={state.addBRollSegment} onRemove={state.removeBRollSegment} /> diff --git a/apps/web/src/pages/ai-avatar/components/ModalBRollEditor.tsx b/apps/web/src/pages/ai-avatar/components/ModalBRollEditor.tsx index 613bed94f..336540973 100644 --- a/apps/web/src/pages/ai-avatar/components/ModalBRollEditor.tsx +++ b/apps/web/src/pages/ai-avatar/components/ModalBRollEditor.tsx @@ -5,12 +5,12 @@ * - 左侧:先选素材库(video 库)→ 再选该库视频素材(已被其他 segment 使用的素材 * 标灰 + "已选择" 遮罩,pointer-events:none 防重复选择) * - 右侧:文案句子列表(点选对应段落,替代原数字索引框)/ 全屏 or 画中画 / 四角位置+大小 - * (开始/结束时间已删除,按句子字数占比 × 口播总时长自动估算) + * (开始/结束时间来自后端精确句子时间戳,基于 TTS 音频静音检测) * - 底部:已配置的画面插入列表(可删除) */ import React, { useEffect, useMemo, useState } from "react" import { getAssets, getAssetLibraries, type AssetItem, type AssetLibraryItem } from "@/api/assets" -import type { BRollSegment, BRollInsertMode, PipPosition } from "../types" +import type { BRollSegment, BRollInsertMode, PipPosition, SentenceTiming } from "../types" import { splitScriptIntoSentences, type ScriptSentence } from "../utils/sentences" interface ModalBRollEditorProps { @@ -18,10 +18,12 @@ interface ModalBRollEditorProps { onClose: () => void /** 当前已有的 B-roll segments(用于标灰已选素材) */ existingSegments: BRollSegment[] - /** 当前文案全文(用于分句) */ + /** 文案全文(优先使用对口型时锁定的 scriptText) */ scriptText: string - /** 对口型成片总时长(秒),用于时间自动估算 */ + /** 对口型成片总时长(秒) */ outputDuration: number + /** 后端精确句子时间戳(来自 lipsyncJob.sentence_timings) */ + sentenceTimings?: SentenceTiming[] | null onConfirm: (segment: BRollSegment) => void onRemove: (id: string) => void } @@ -44,6 +46,7 @@ const ModalBRollEditor: React.FC = ({ existingSegments, scriptText, outputDuration, + sentenceTimings, onConfirm, onRemove, }) => { @@ -62,10 +65,10 @@ const ModalBRollEditor: React.FC = ({ const [pipPosition, setPipPosition] = useState("top-right") const [pipScale, setPipScale] = useState(0.3) - /** 文案分句(⑤) */ + /** 文案分句(使用后端精确时间戳) */ const sentences = useMemo( - () => splitScriptIntoSentences(scriptText, outputDuration), - [scriptText, outputDuration], + () => splitScriptIntoSentences(scriptText, sentenceTimings), + [scriptText, sentenceTimings], ) /** 已被现有 segments 占用的素材 id 集合(标灰、禁止重复选择) */ @@ -264,7 +267,7 @@ const ModalBRollEditor: React.FC = ({ > {sent.index + 1} {sent.text} - {outputDuration > 0 && ( + {sent.endTime > 0 && ( {sent.startTime.toFixed(1)}-{sent.endTime.toFixed(1)}s @@ -349,7 +352,7 @@ const ModalBRollEditor: React.FC = ({ selectedSentence.endTime, selectedSentence.startTime + 0.5, ).toFixed(1)} - s (按字数自动估算) + s ) : ( diff --git a/apps/web/src/pages/ai-avatar/components/PanelCoverAndGenerate.tsx b/apps/web/src/pages/ai-avatar/components/PanelCoverAndGenerate.tsx index 278e01400..2c7d48893 100644 --- a/apps/web/src/pages/ai-avatar/components/PanelCoverAndGenerate.tsx +++ b/apps/web/src/pages/ai-avatar/components/PanelCoverAndGenerate.tsx @@ -120,7 +120,7 @@ const PanelCoverAndGenerate: React.FC = ({ wordBreak: "break-word", whiteSpace: "pre-wrap", color: titleConfig.color || "#ffffff", - fontSize: `${titleConfig.size}px`, + fontSize: `${(titleConfig.size || 28) * 0.55}px`, // 预览容器缩放,与 PanelLipsyncPreview 对齐 fontFamily: getFontFamily(titleConfig.font), fontWeight: titleConfig.bold ? "bold" : "normal", fontStyle: titleConfig.italic ? "italic" : "normal", @@ -153,8 +153,8 @@ const PanelCoverAndGenerate: React.FC = ({ } else if (titleConfig.shadow) { style.textShadow = "0 2px 8px rgba(0,0,0,0.7), 0 0 2px rgba(0,0,0,0.5)" } else { - // 默认给轻微阴影保证白字在亮背景可读 - style.textShadow = "0 2px 6px rgba(0,0,0,0.6)" + // 无描边无阴影时,不加额外效果(与后端 drawtext 对齐:无 stroke/shadow 则不加) + style.textShadow = "none" } return style diff --git a/apps/web/src/pages/ai-avatar/components/PanelLipsyncPreview.tsx b/apps/web/src/pages/ai-avatar/components/PanelLipsyncPreview.tsx index 4e01e8f08..e690cda97 100755 --- a/apps/web/src/pages/ai-avatar/components/PanelLipsyncPreview.tsx +++ b/apps/web/src/pages/ai-avatar/components/PanelLipsyncPreview.tsx @@ -14,8 +14,8 @@ interface PanelLipsyncPreviewProps { onRemoveBRoll: (id: string) => void /** 标题配置(实时叠加预览用) */ titleConfig?: AiAvatarTitleConfig - /** 标题位置变更回调(拖拽结束时调用) */ - onTitlePositionChange?: (pos: { pos_x: number; pos_y: number }) => void + /** 标题位置变更回调(拖拽结束时调用,发送百分比坐标 + position:"custom") */ + onTitlePositionChange?: (pos: { pos_x: number; pos_y: number; position: string }) => void } const BROLL_MODE_LABEL: Record = { @@ -68,11 +68,13 @@ export function PanelLipsyncPreview({ padding: "4px 8px", textShadow: titleConfig.shadow ? "0 2px 4px rgba(0,0,0,0.8)" : undefined, WebkitTextStroke: titleConfig.stroke ? "1.5px #000" : undefined, - ...(titleConfig.position === "top" - ? { top: 8 } - : titleConfig.position === "bottom" - ? { bottom: 8 } - : { top: "50%", transform: "translateX(-50%) translateY(-50%)" }), + ...(titleConfig.position === "custom" && titleConfig.pos_x != null && titleConfig.pos_y != null + ? { left: `${titleConfig.pos_x}%`, top: `${titleConfig.pos_y}%`, transform: "translateX(-50%) translateY(-50%)" } + : titleConfig.position === "top" + ? { top: 8 } + : titleConfig.position === "bottom" + ? { bottom: 8 } + : { top: "50%", transform: "translateX(-50%) translateY(-50%)" }), } : null @@ -105,7 +107,10 @@ export function PanelLipsyncPreview({ const rect = previewContainerRef.current.getBoundingClientRect() const relX = Math.max(0, Math.min(rect.width, e.clientX - rect.left)) const relY = Math.max(0, Math.min(rect.height, e.clientY - rect.top)) - onTitlePositionChange({ pos_x: relX, pos_y: relY }) + // 发送百分比坐标(0-100),与后端 drawtext 百分比表达式对齐 + const xpct = Math.round((relX / rect.width) * 1000) / 10 + const ypct = Math.round((relY / rect.height) * 1000) / 10 + onTitlePositionChange({ pos_x: xpct, pos_y: ypct, position: "custom" }) } ;(e.currentTarget as HTMLDivElement).style.cursor = "grab" } diff --git a/apps/web/src/pages/ai-avatar/types.ts b/apps/web/src/pages/ai-avatar/types.ts index e2a118c73..a06be59d7 100644 --- a/apps/web/src/pages/ai-avatar/types.ts +++ b/apps/web/src/pages/ai-avatar/types.ts @@ -44,12 +44,22 @@ export interface LipsyncJob { status: LipsyncStatus progress: number output_video_url: string | null - /** 对口型成片总时长(秒),后端返回;用于 B-roll 时间自动估算(#1809 ⑥) */ + /** 对口型成片总时长(秒),后端返回 */ output_duration?: number + /** 精确句子时间戳(后端基于 TTS 音频静音检测计算) */ + sentence_timings?: SentenceTiming[] | null error_message: string | null created_at: string } +/* ── 句子时间戳(后端精确计算) ── */ +export interface SentenceTiming { + index: number + text: string + start_time: number + end_time: number +} + /* ── B-roll 画面插入 ── */ export type BRollInsertMode = "fullscreen" | "pip" export type PipPosition = "top-left" | "top-right" | "bottom-left" | "bottom-right" @@ -77,7 +87,7 @@ export interface AiAvatarTitleConfig { shadow: boolean color: string auto_subtitle: boolean - /** 自定义位置坐标(position=custom 时生效,像素) */ + /** 自定义位置坐标(position=custom 时生效,百分比 0-100) */ pos_x?: number pos_y?: number } @@ -101,6 +111,7 @@ export interface RenderJob { status: RenderStatus progress: number output_video_url: string | null + output_cover_url: string | null error_message: string | null created_at: string } diff --git a/apps/web/src/pages/ai-avatar/utils/contract.ts b/apps/web/src/pages/ai-avatar/utils/contract.ts index 70e5c0fc5..a8ce5b78d 100644 --- a/apps/web/src/pages/ai-avatar/utils/contract.ts +++ b/apps/web/src/pages/ai-avatar/utils/contract.ts @@ -39,7 +39,7 @@ export function buildTitleConfigPayload(cfg: AiAvatarTitleConfig): Record part.trim()) .filter((part) => part.length > 0) - const totalChars = rawParts.reduce((sum, part) => sum + part.replace(/\s/g, "").length, 0) - const duration = outputDuration > 0 ? outputDuration : 0 - const sentences: ScriptSentence[] = [] let accChar = 0 rawParts.forEach((part, i) => { const charCount = part.replace(/\s/g, "").length - const startTime = duration > 0 && totalChars > 0 ? (accChar / totalChars) * duration : 0 - const endTime = - duration > 0 && totalChars > 0 ? ((accChar + charCount) / totalChars) * duration : 0 + // 从后端精确时间戳获取;无数据时返回 0 + const timing = sentenceTimings?.[i] + const startTime = timing?.start_time ?? 0 + const endTime = timing?.end_time ?? 0 + sentences.push({ index: i, text: part, charCount, startChar: accChar, - startTime: round1(startTime), - endTime: round1(endTime), + startTime, + endTime, }) accChar += charCount }) return sentences } - -function round1(n: number): number { - return Math.round(n * 10) / 10 -} diff --git a/packages/adapters/sqlalchemy_impl/models.py b/packages/adapters/sqlalchemy_impl/models.py index 135c54ad9..270f0891e 100755 --- a/packages/adapters/sqlalchemy_impl/models.py +++ b/packages/adapters/sqlalchemy_impl/models.py @@ -703,6 +703,9 @@ class LipsyncJobModel(Base): error_message = Column(Text, nullable=False, default="") error_code = Column(String(100), nullable=False, default="") + # 精确句子时间戳(TTS 合成后由 silencedetect 计算,用于 B-roll 精确定位) + sentence_timings = Column(JSON, nullable=True) # list[{index,text,start_time,end_time}] + # 时间戳 submitted_at = Column(DateTime, nullable=True) completed_at = Column(DateTime, nullable=True) diff --git a/packages/domain/video_filter_builder.py b/packages/domain/video_filter_builder.py index 57523a545..e44f80dd9 100755 --- a/packages/domain/video_filter_builder.py +++ b/packages/domain/video_filter_builder.py @@ -481,13 +481,13 @@ def build_title_drawtext_filter( # ── 样式参数 ── font_name = title_config.get("font") or title_config.get("font_preset") or "思源黑体" - font_size = int(title_config.get("font_size") or title_config.get("size") or 36) + font_size = int(title_config.get("font_size") or title_config.get("size") or 28) font_color = title_config.get("font_color") or title_config.get("color") or "#ffffff" # 去掉 # 前缀(drawtext 用纯 hex 或颜色名) if font_color.startswith("#"): font_color = font_color[1:] - position = title_config.get("position", "top") + position = title_config.get("position") or "bottom" bold = bool(title_config.get("bold", True)) stroke = title_config.get("stroke") shadow = title_config.get("shadow") @@ -556,8 +556,13 @@ def build_title_drawtext_filter( and not isinstance(pos_x, bool) and not isinstance(pos_y, bool) ): - params.append(f"x={int(pos_x)}") - params.append(f"y={int(pos_y)}") + # pos_x/pos_y 为百分比坐标(0-100),转换为 drawtext 表达式 + # 例如 pos_x=50 → x=(w-text_w)*0.50(水平居中偏50%) + # pos_y=30 → y=(h-text_h)*0.30 + pct_x = max(0.0, min(100.0, float(pos_x))) / 100.0 + pct_y = max(0.0, min(100.0, float(pos_y))) / 100.0 + params.append(f"x=(w-text_w)*{pct_x:.4f}") + params.append(f"y=(h-text_h)*{pct_y:.4f}") else: # 三档预设位置:top / center / bottom # x 始终水平居中:(w-text_w)/2 diff --git a/tests/unit/test_video_filter_builder.py b/tests/unit/test_video_filter_builder.py index 45ef53c72..7cb800e25 100755 --- a/tests/unit/test_video_filter_builder.py +++ b/tests/unit/test_video_filter_builder.py @@ -1065,12 +1065,23 @@ class TestDrawtextPositionBranches(unittest.TestCase): self.assertIn("y=h-text_h-50", result) @patch("packages.domain.video_filter_builder._resolve_font_path") - def test_position_custom_with_float_coords(self, mock_font): + def test_position_custom_with_percentage_coords(self, mock_font): + """自定义位置:百分比坐标转换为 drawtext 表达式.""" + mock_font.return_value = "" + # pos_x=50, pos_y=30 → x=(w-text_w)*0.5000, y=(h-text_h)*0.3000 + result = build_title_drawtext_filter({"text": "标题", "position": "custom", "pos_x": 50, "pos_y": 30}) + self.assertIsNotNone(result) + self.assertIn("x=(w-text_w)*0.5000", result) + self.assertIn("y=(h-text_h)*0.3000", result) + + @patch("packages.domain.video_filter_builder._resolve_font_path") + def test_position_custom_clamped_to_100(self, mock_font): + """自定义位置:超过100的坐标被截断到100%.""" mock_font.return_value = "" result = build_title_drawtext_filter({"text": "标题", "position": "custom", "pos_x": 100.7, "pos_y": 200.3}) self.assertIsNotNone(result) - self.assertIn("x=100", result) - self.assertIn("y=200", result) + self.assertIn("x=(w-text_w)*1.0000", result) + self.assertIn("y=(h-text_h)*1.0000", result) @patch("packages.domain.video_filter_builder._resolve_font_path") def test_position_custom_bool_coords_fallback(self, mock_font):