fix(ai-avatar): 修复三个bug——封面重影/标题字号缩放/对口型音频截断 #1876

Merged
xiaoxia merged 2 commits from fix/ai-avatar-three-bugs-0913 into develop 2026-09-13 14:15:31 +08:00
8 changed files with 124 additions and 126 deletions
+3 -1
View File
@@ -69,7 +69,9 @@ class CreateLipsyncJobRequest(BaseModel):
speed: float = Field(1.0, ge=0.5, le=2.0, description="语速(0.5-2.0),默认 1.0")
emotion: str = Field("", description="情绪(natural/excited/calm/friendly 或中文 自然/兴奋/沉稳/亲切)")
enable_video_loop: bool = Field(False, description="音频长于视频时是否循环画面")
enable_video_loop: bool = Field(
True, description="音频长于视频时是否循环画面(AI数字人默认开启,防止音频长于视频被截断)"
)
project_id: str = Field("", description="项目 ID(可选)")
@model_validator(mode="after")
+1 -1
View File
@@ -257,7 +257,7 @@ class LipsyncService:
script_text: str = "",
speed: float = 1.0,
emotion: str = "",
enable_video_loop: bool = False,
enable_video_loop: bool = True,
project_id: str = "",
) -> LipsyncJobModel:
"""创建对口型任务.
+2 -3
View File
@@ -75,7 +75,7 @@ class MediaKitClient:
*,
video_url: str,
audio_url: str,
enable_video_loop: bool = False,
enable_video_loop: bool = True,
callback_url: Optional[str] = None,
callback_args: Optional[str] = None,
client_token: Optional[str] = None,
@@ -103,8 +103,7 @@ class MediaKitClient:
"video_url": video_url,
"audio_url": audio_url,
}
if enable_video_loop:
payload["enable_video_loop"] = True
payload["enable_video_loop"] = bool(enable_video_loop)
if callback_url:
payload["callback_url"] = callback_url
if callback_args:
@@ -244,7 +244,7 @@ const AiAvatarPage: React.FC = () => {
audio_url: state.ttsPreview.audioUrl!,
audio_duration: state.ttsPreview.duration,
sentence_timings: state.ttsPreview.sentenceTimings,
enable_video_loop: false,
enable_video_loop: true,
}
} else {
// 降级:TTS 直生(旧路径,前端未预合成时)
@@ -635,7 +635,6 @@ const AiAvatarPage: React.FC = () => {
onCoverConfigChange={(partial) =>
state.setCoverConfig((prev) => ({ ...prev, ...partial }))
}
titleConfig={state.titleConfig}
renderJob={currentRenderJob}
onGenerateRenderSmartCover={handleGenerateRenderSmartCover}
resolution={state.resolution}
@@ -8,12 +8,11 @@
*
* 注意:v3 已删除"画面插入模式",本面板不包含该选项。
*/
import React, { useMemo, useRef, useState } from "react"
import type { AiAvatarCoverConfig, AiAvatarTitleConfig, RenderJob } from "../types"
import React, { useRef, useState } from "react"
import type { AiAvatarCoverConfig, RenderJob } from "../types"
interface PanelCoverAndGenerateProps {
coverConfig: AiAvatarCoverConfig
titleConfig: AiAvatarTitleConfig
onCoverConfigChange: (partial: Partial<AiAvatarCoverConfig>) => void
resolution: string
onResolutionChange: (r: string) => void
@@ -52,19 +51,8 @@ const LIPSYNC_STATUS_LABEL: Record<string, { text: string; cls: string }> = {
failed: { text: "失败", cls: "aa-status-badge--failed" },
}
/** 字体名 → CSS font-family 映射(与后端 drawtext 对齐) */
const FONT_FAMILY_MAP: Record<string, string> = {
思源黑体: "'Noto Sans SC', 'Source Han Sans SC', 'PingFang SC', 'Microsoft YaHei', sans-serif",
思源宋体: "'Noto Serif SC', 'Source Han Serif SC', 'SimSun', serif",
楷体: "KaiTi, 'STKaiti', serif",
黑体: "'Heiti SC', 'SimHei', 'Microsoft YaHei', sans-serif",
}
const getFontFamily = (font: string): string => FONT_FAMILY_MAP[font] || FONT_FAMILY_MAP["思源黑体"]
const PanelCoverAndGenerate: React.FC<PanelCoverAndGenerateProps> = ({
coverConfig,
titleConfig,
onCoverConfigChange,
resolution,
onResolutionChange,
@@ -131,58 +119,6 @@ const PanelCoverAndGenerate: React.FC<PanelCoverAndGenerateProps> = ({
coverConfig.smart_cover_url || coverConfig.thumbnail_url || coverConfig.upload_url
const hasCoverImage = Boolean(coverUrl)
/** 是否显示标题叠加层:有图、有文字、非加载中 */
const showTitleOverlay =
hasCoverImage && !smartCoverLoading && titleConfig.title.trim().length > 0
/** 计算标题叠加层的 inline 样式 */
const titleOverlayStyle = useMemo<React.CSSProperties>(() => {
const style: React.CSSProperties = {
position: "absolute",
left: "50%",
width: "90%",
transform: "translateX(-50%)",
textAlign: "center",
boxSizing: "border-box",
padding: "0 4px",
wordBreak: "break-word",
whiteSpace: "pre-wrap",
color: titleConfig.color || "#ffffff",
fontSize: `${(titleConfig.size || 48) * 0.35}px`,
fontFamily: getFontFamily(titleConfig.font),
fontWeight: titleConfig.bold ? "bold" : "normal",
fontStyle: titleConfig.italic ? "italic" : "normal",
lineHeight: 1.3,
pointerEvents: "none",
}
const pos = titleConfig.position || "bottom"
if (pos === "top") {
style.top = "40px"
} else if (pos === "center") {
style.top = "50%"
style.transform = "translate(-50%, -50%)"
} else if (pos === "custom" && titleConfig.pos_x != null && titleConfig.pos_y != null) {
style.left = `${titleConfig.pos_x}%`
style.top = `${titleConfig.pos_y}%`
style.transform = "translate(-50%, -50%)"
} else {
style.bottom = "40px"
}
if (titleConfig.stroke) {
const strokeWidth = Math.max(1, Math.round(titleConfig.size / 18))
;(style as React.CSSProperties)["WebkitTextStroke"] = `${strokeWidth}px rgba(0,0,0,0.75)`
style.textShadow = "none"
} else if (titleConfig.shadow) {
style.textShadow = "0 2px 8px rgba(0,0,0,0.7), 0 0 2px rgba(0,0,0,0.5)"
} else {
style.textShadow = "none"
}
return style
}, [titleConfig])
/** 封面区占位文字 */
const coverPlaceholder = isRenderCompleted ? "暂无封面" : "视频生成后可选择封面"
@@ -297,7 +233,7 @@ const PanelCoverAndGenerate: React.FC<PanelCoverAndGenerateProps> = ({
<div className="aa-label" style={{ marginBottom: 8 }}>
封面
</div>
{/* 封面预览(竖屏 9:16) */}
{/* 封面预览(竖屏 9:16)——成片帧已经通过 Canvas PNG overlay 带有标题,直接展示原图即可 */}
<div className="aa-cover-preview" style={{ opacity: isRenderCompleted ? 1 : 0.5 }}>
{hasCoverImage ? (
<img src={coverUrl!} alt="封面预览" draggable={false} />
@@ -305,11 +241,6 @@ const PanelCoverAndGenerate: React.FC<PanelCoverAndGenerateProps> = ({
<span className="aa-cover-preview__placeholder">{coverPlaceholder}</span>
)}
{smartCoverLoading && <div className="aa-cover-preview__loading">⏳ 智能选帧中…</div>}
{showTitleOverlay && (
<div style={titleOverlayStyle} aria-hidden="true">
{titleConfig.title}
</div>
)}
</div>
<div className="aa-cover-actions">
@@ -1,9 +1,9 @@
/**
* AI数字人 — 对口型预览面板(步骤2用)
* B-roll 画面插入 + 对口型视频预览 + 生成/重新生成按钮
* v3.1: 预览容器按 1/2 缩放、标题实时叠加预览
* v3.1: 标题字号按预览容器实际宽度动态计算 previewScale(基准 720p),与成片一致
*/
import React, { useRef } from "react"
import React, { useCallback, useEffect, useRef, useState } from "react"
import type { LipsyncJob, BRollSegment, AiAvatarTitleConfig } from "../types"
interface PanelLipsyncPreviewProps {
@@ -29,6 +29,16 @@ function formatTime(seconds: number): string {
return `${m}:${s.toString().padStart(2, "0")}`
}
/** 字体名 → CSS font-family 映射(与 titleCanvas 字体链对齐) */
const FONT_FAMILY_MAP: Record<string, string> = {
思源黑体: "'Noto Sans CJK SC', 'Source Han Sans CN', 'PingFang SC', 'Microsoft YaHei', sans-serif",
思源宋体: "'Noto Serif SC', 'Source Han Serif SC', 'SimSun', serif",
楷体: "KaiTi, 'STKaiti', serif",
黑体: "'Heiti SC', 'SimHei', 'Microsoft YaHei', sans-serif",
}
const getFontFamily = (font: string): string =>
FONT_FAMILY_MAP[font] || FONT_FAMILY_MAP["思源黑体"]
export function PanelLipsyncPreview({
lipsyncJob,
onGenerateLipsync,
@@ -41,6 +51,8 @@ export function PanelLipsyncPreview({
const titleDragRef = useRef<HTMLDivElement>(null)
const draggingTitleRef = useRef(false)
const previewContainerRef = useRef<HTMLDivElement>(null)
// 预览容器实际宽度(通过 ResizeObserver 监听),用于动态计算 previewScale
const [containerWidth, setContainerWidth] = useState(0)
const isGenerating = lipsyncJob?.status === "pending" || lipsyncJob?.status === "processing"
const isDone = lipsyncJob?.status === "completed"
const isFailed = lipsyncJob?.status === "failed"
@@ -52,35 +64,84 @@ export function PanelLipsyncPreview({
? "排队中…"
: "对口型生成中…"
/** 标题叠加样式 */
const titleOverlayStyle: React.CSSProperties | null = titleConfig?.title
? {
position: "absolute",
color: titleConfig.color || "#ffffff",
fontFamily: titleConfig.font || "思源黑体",
fontSize: `${(titleConfig.size || 48) * 0.35}px`,
fontWeight: titleConfig.bold ? 700 : 400,
fontStyle: titleConfig.italic ? "italic" : "normal",
textAlign: "center",
width: "90%",
padding: "4px 8px",
textShadow: titleConfig.shadow ? "0 2px 4px rgba(0,0,0,0.8)" : undefined,
WebkitTextStroke: titleConfig.stroke ? "1.5px #000" : undefined,
...(titleConfig.position === "custom" &&
titleConfig.pos_x != null &&
titleConfig.pos_y != null
? {
left: `${titleConfig.pos_x}%`,
top: `${titleConfig.pos_y}%`,
transform: "translateX(-50%) translateY(-50%)",
}
: titleConfig.position === "top"
? { left: "50%", top: 8, transform: "translateX(-50%)" }
: titleConfig.position === "bottom"
? { left: "50%", bottom: 8, transform: "translateX(-50%)" }
: { left: "50%", top: "50%", transform: "translateX(-50%) translateY(-50%)" }),
}
: null
// 监听预览容器尺寸变化,动态测量宽度以计算 previewScale(基准 720p)
useEffect(() => {
const el = previewContainerRef.current
if (!el) return
const update = () => setContainerWidth(el.clientWidth || 0)
update()
if (typeof ResizeObserver !== "undefined") {
const ro = new ResizeObserver(update)
ro.observe(el)
return () => ro.disconnect()
}
window.addEventListener("resize", update)
return () => window.removeEventListener("resize", update)
}, [])
// 预览缩放比:预览宽度 / 720(基准宽度)
const previewScale = containerWidth > 0 ? containerWidth / 720 : 0.35
const ps = useCallback((v: number) => Math.round(v * previewScale * 100) / 100, [previewScale])
/** 标题叠加样式(字号/padding/描边/阴影均按 previewScale 缩放,保持与成片视觉一致) */
const titleOverlayStyle: React.CSSProperties | null =
titleConfig?.title && containerWidth > 0
? (() => {
const baseSize = titleConfig.size || 48
const fontSize = ps(baseSize)
// 描边宽度基准 ≈ size * 0.06,最小 1.5px @720p
const strokeW = Math.max(ps(1.5), +(baseSize * 0.06 * previewScale).toFixed(2))
// 阴影按比例缩放
const shadowBlur = ps(4)
const shadowOffsetY = ps(2)
// padding / top 边距按比例(基准 8px 对应预览小窗,成片基准 16px,这里 8px 对应约 0.33 缩放)
const padV = ps(16) * 0.5 // ≈ 8px in ~240px container
const padH = ps(24) * 0.5
const style: React.CSSProperties = {
position: "absolute",
color: titleConfig.color || "#ffffff",
fontFamily: getFontFamily(titleConfig.font || "思源黑体"),
fontSize: `${fontSize}px`,
fontWeight: titleConfig.bold ? 700 : 400,
fontStyle: titleConfig.italic ? "italic" : "normal",
textAlign: "center",
width: "90%",
lineHeight: 1.2,
padding: `${ps(4)}px ${padH}px`,
textShadow: titleConfig.shadow
? `0 ${shadowOffsetY}px ${shadowBlur}px rgba(0,0,0,0.8), 0 0 ${ps(2)}px rgba(0,0,0,0.5)`
: undefined,
WebkitTextStroke: titleConfig.stroke ? `${strokeW}px #000` : undefined,
boxSizing: "border-box",
wordBreak: "break-word",
whiteSpace: "pre-wrap",
}
if (
titleConfig.position === "custom" &&
titleConfig.pos_x != null &&
titleConfig.pos_y != null
) {
style.left = `${titleConfig.pos_x}%`
style.top = `${titleConfig.pos_y}%`
style.transform = "translateX(-50%) translateY(-50%)"
} else if (titleConfig.position === "top") {
style.left = "50%"
style.top = padV
style.transform = "translateX(-50%)"
} else if (titleConfig.position === "bottom") {
style.left = "50%"
style.bottom = padV
style.transform = "translateX(-50%)"
} else {
style.left = "50%"
style.top = "50%"
style.transform = "translateX(-50%) translateY(-50%)"
}
return style
})()
: null
const handleTitlePointerDown = (e: React.PointerEvent<HTMLDivElement>) => {
if (!onTitlePositionChange || !previewContainerRef.current) return
@@ -183,7 +244,7 @@ export function PanelLipsyncPreview({
)}
</div>
{/* ── 对口型预览(v3.1: 缩放1/2 + 标题叠加) ─ */}
{/* ── 对口型预览(标题字号按 previewScale 动态缩放) ─ */}
<div className="aa-lipsync-section">
<div className="aa-lipsync-section__title">对口型预览</div>
@@ -4,6 +4,10 @@
* 把标题按前端预览的 HTML/CSS 效果画到透明背景 PNG 上(与视频同分辨率),
* 以 dataURL 形式传给后端,后端用 FFmpeg overlay 直接叠加图层,
* 彻底解决前端 HTML/CSS 预览 ≠ FFmpeg drawtext 成片的 WYSIWYG 问题。
*
* 约定:titleConfig.size 的语义是"720p 基准宽度下的字号(px)",
* 按 videoWidth / 720 得到 scale,所有长度类参数乘以 scale,
* 保证 1080p / 4K 成片里标题视觉大小与预览一致。
*/
import type { AiAvatarTitleConfig } from "../types"
@@ -35,13 +39,18 @@ export function renderTitleToPngDataUrl(opts: RenderTitlePngOptions): string | n
.filter((l) => l.length > 0)
if (lines.length === 0) return null
// 分辨率缩放系数:基准 720p,所有长度类参数乘以 scale
const scale = videoWidth / 720
const r = (v: number) => Math.round(v * scale)
const canvas = document.createElement("canvas")
canvas.width = videoWidth
canvas.height = videoHeight
const ctx = canvas.getContext("2d")
if (!ctx) return null
const size = Math.max(12, Math.round(titleConfig.size || 48))
const baseSize = Math.max(12, Math.round(titleConfig.size || 48))
const size = r(baseSize)
const bold = !!titleConfig.bold
const italic = !!titleConfig.italic
const color = titleConfig.color || "#ffffff"
@@ -60,19 +69,16 @@ export function renderTitleToPngDataUrl(opts: RenderTitlePngOptions): string | n
ctx.textAlign = "center"
ctx.textBaseline = "middle"
// 阴影(shadow=true 时开启)
// 阴影(shadow=true 时开启)——按 scale 缩放
if (shadow) {
ctx.shadowColor = "rgba(0,0,0,0.8)"
ctx.shadowBlur = 4
ctx.shadowBlur = r(4)
ctx.shadowOffsetX = 0
ctx.shadowOffsetY = 2
ctx.shadowOffsetY = r(2)
}
// 位置计算:与 PanelLipsyncPreview 的 CSS 对齐
// 预览用 top/bottom 8px padding + transform translateX(-50%) 居中;
// 这里画到整尺寸 canvas,padding 按比例放大到全分辨率(预览缩放 0.35x 时 8px ≈ 23px 全尺寸,
// 为更贴近原 CSS 16px 安全边距,用 16px 作为内边距)。
const PAD = 16
// 位置计算:与 PanelLipsyncPreview 的 CSS 对齐(按 scale 缩放 PAD)
const PAD = r(16)
let centerX = videoWidth / 2
const position = titleConfig.position || "bottom"
const lineGap = size * 1.2
@@ -97,10 +103,9 @@ export function renderTitleToPngDataUrl(opts: RenderTitlePngOptions): string | n
firstLineY = videoHeight - totalTextH - PAD + size / 2
}
// 描边参数(stroke=true 或 bold 默认细描边模拟粗体时都画;
// 注意:浏览器原生 bold 已经是粗体 glyph,Canvas 这里对 stroke=true 才加黑描边,
// 与预览 CSS 的 WebkitTextStroke 保持一致,不对 bold 自动加描边避免双粗)。
// 描边参数:描边 lineWidth 按 scale 缩放(基准 size * 0.06,最小 2px @720p)
const doStroke = stroke
const strokeWidth = Math.max(r(2), Math.round(size * 0.06))
// 逐行绘制
lines.forEach((line, idx) => {
const y = firstLineY + idx * lineGap
@@ -110,14 +115,14 @@ export function renderTitleToPngDataUrl(opts: RenderTitlePngOptions): string | n
// 描边不要带阴影(避免黑色描边发虚)
ctx.shadowColor = "rgba(0,0,0,0)"
ctx.shadowBlur = 0
ctx.lineWidth = Math.max(2, size * 0.06)
ctx.lineWidth = strokeWidth
ctx.strokeStyle = "#000000"
ctx.lineJoin = "round"
ctx.strokeText(line, centerX, y)
// 恢复阴影
if (shadow) {
ctx.shadowColor = "rgba(0,0,0,0.8)"
ctx.shadowBlur = 4
ctx.shadowBlur = r(4)
} else {
ctx.shadowColor = prevShadowColor
ctx.shadowBlur = prevShadowBlur
+2 -1
View File
@@ -159,7 +159,8 @@ class TestSchemaValidation:
voice_id="longxiaochun_v3",
script_text="测试文本",
)
assert req.enable_video_loop is False
# AI数字人场景文案长度不可控,默认开启视频循环,防止音频长于视频时被截断
assert req.enable_video_loop is True
def test_video_url_strip_query_params(self):
"""视频 URL 含查询参数时,扩展名检查应忽略 ? 后面的部分."""