feat(web): TTS 情绪扩展为 7 种 + 系统音色语言限 zh/en + 清理残留代码 (#1915)
CI/CD Pipeline / Dedup Check - skip PR tests when covered by push pipeline (push) Successful in 2s
CI/CD Pipeline / Check push changed paths (push) Successful in 3s
CI/CD Pipeline / Build Staging Web Image (push) Successful in 1m42s
CI/CD Pipeline / Build Staging Worker Image (push) Successful in 3m5s
CI/CD Pipeline / Build Staging API Image (push) Successful in 3m6s
CI/CD Pipeline / Integration Tests (push) Successful in 4m16s
CI/CD Pipeline / Deploy Staging (Watchtower auto-deploy) (push) Successful in 1m28s
CI/CD Pipeline / Validate - Python (mypy + alembic) (push) Successful in 4m43s
CI/CD Pipeline / Frontend Unit Tests (push) Successful in 6m8s
CI/CD Pipeline / Validate - Style (push) Successful in 6m18s
CI/CD Pipeline / ACR Image Cleanup (push) Successful in 1m39s
CI/CD Pipeline / Staging API Integration Tests (push) Successful in 3m10s
CI/CD Pipeline / Staging E2E Tests (push) Failing after 3m15s
CI/CD Pipeline / Unit Tests (push) Successful in 10m1s
CI/CD Pipeline / Validate - Security (push) Successful in 11m26s
CI/CD Pipeline / Production Browser E2E (push) Has been skipped
CI/CD Pipeline / Deploy Production (push) Failing after 37h26m18s
CI/CD Pipeline / Build Production Web Image (push) Failing after 37h26m22s
CI/CD Pipeline / PR Build Worker Image (push) Failing after 37h37m48s
CI/CD Pipeline / Check if frontend-only change (push) Failing after 37h37m21s
CI/CD Pipeline / PR Build Web Image (push) Failing after 37h37m15s
CI/CD Pipeline / PR Build API Image (push) Failing after 37h37m15s
CI/CD Pipeline / Retag skipped Staging Web Image (push) Failing after 37h34m5s
CI/CD Pipeline / Retag skipped Staging API Image (push) Failing after 37h34m6s
CI/CD Pipeline / CI Gate (push) Failing after 37h25m49s
CI/CD Pipeline / Canary Release to Production (push) Failing after 37h25m45s
CI/CD Pipeline / Build Production Worker Image (push) Failing after 37h25m49s
CI/CD Pipeline / Build Production API Image (push) Failing after 37h25m49s
CI/CD Pipeline / Retag skipped Staging Worker Image (push) Failing after 37h34m5s
CI/CD Pipeline / Frontend Lint (push) Failing after 37h37m16s

Co-authored-by: xiaoxia <dev@xiaoxiajianji.com>
Co-committed-by: xiaoxia <dev@xiaoxiajianji.com>
This commit was merged in pull request #1915.
This commit is contained in:
2026-09-15 01:45:51 +08:00
committed by auto-approve-bot
parent a56b3f7b42
commit 5d4e07d4f4
11 changed files with 173 additions and 48 deletions
+1 -1
View File
@@ -105,7 +105,7 @@ export interface TTSPreviewRequest {
speed?: number
pitch?: number
language?: string
emotion?: string // 情绪参数:natural/excited/calm/friendly
emotion?: string // 情绪参数:neutral/happy/sad/angry/surprised/fearful/disgusted(后端 normalize_emotion() 兼容旧 natural/excited/calm/friendly 与中文标签)
}
/** TTS 试听响应 */
+1 -1
View File
@@ -55,7 +55,7 @@ export const createLipsyncJob = async (data: {
script_text?: string
/** 语速 0.5~2.0,默认 1.0(TTS 直生模式用) */
speed?: number
/** 情绪英文枚举:natural/excited/calm/friendly(TTS 直生模式用) */
/** 情绪英文枚举:neutral/happy/sad/angry/surprised/fearful/disgusted(TTS 直生模式用;前端经 normalizeEmotion 归一化) */
emotion?: string
enable_video_loop?: boolean
project_id?: string
@@ -13,7 +13,8 @@ import {
type VoiceEmotion,
type VoiceLanguage,
VOICE_EMOTION_OPTIONS,
VOICE_LANGUAGE_OPTIONS,
PRESET_VOICE_LANGUAGE_OPTIONS,
CLONE_VOICE_LANGUAGE_OPTIONS,
} from "../types"
interface PanelVoiceSelectorProps {
@@ -91,6 +92,12 @@ export function PanelVoiceSelector({
const NO_PREVIEW_TIP = "该音色暂无试听音频,请先用此音色生成一段配音后再试听"
// 系统预置音色仅支持 zh/en;克隆音色支持全语言
const languageOptions =
voiceSource === "clone" ? CLONE_VOICE_LANGUAGE_OPTIONS : PRESET_VOICE_LANGUAGE_OPTIONS
// 当前语言不在可选列表(切回预置时 ja/ko/cantonese/mandarin 失效)→ 自动回退到中文
const effectiveLanguage = languageOptions.some((o) => o.value === language) ? language : "zh"
/** 用指定 URL 真实播放(抽取公共) */
const playAudioUrl = (voiceId: string, url: string) => {
// 临时兼容:后端 /tts/preview 返回 HTTP URL,staging 是 HTTPS,Mixed Content 会阻止加载
@@ -289,10 +296,12 @@ export function PanelVoiceSelector({
<select
id="aa-voice-language"
className="aa-select"
value={language}
onChange={(e) => onLanguageChange(e.target.value as VoiceLanguage)}
value={effectiveLanguage}
onChange={(e) => {
onLanguageChange(e.target.value as VoiceLanguage)
}}
>
{VOICE_LANGUAGE_OPTIONS.map((opt) => (
{languageOptions.map((opt) => (
<option key={opt.value} value={opt.value}>
{opt.label}
</option>
@@ -34,9 +34,9 @@ export function useAiAvatar() {
/* ── 面板2:配音库 ── */
const [voiceSource, setVoiceSource] = useState<VoiceSource>("preset")
const [selectedVoice, setSelectedVoice] = useState<UnifiedVoiceItem | null>(null)
const [emotion, setEmotion] = useState<VoiceEmotion>("natural")
const [emotion, setEmotion] = useState<VoiceEmotion>("neutral")
const [speed, setSpeed] = useState(1.0)
const [language, setLanguage] = useState<VoiceLanguage>("mandarin")
const [language, setLanguage] = useState<VoiceLanguage>("zh")
/* ── 面板3:文案 & 对口型 ── */
const [script, setScript] = useState<Script | null>(null)
+28 -11
View File
@@ -6,25 +6,42 @@ import type { AssetItem } from "@/api/assets"
/* ── 音色来源切换 ── */
export type VoiceSource = "preset" | "clone"
/* ── 情绪 ── */
export type VoiceEmotion = "natural" | "excited" | "calm" | "friendly"
/* ── 情绪(对齐 CosyVoice 7 种情绪) ── */
export type VoiceEmotion =
"neutral" | "happy" | "sad" | "angry" | "surprised" | "fearful" | "disgusted"
export const VOICE_EMOTION_OPTIONS: { value: VoiceEmotion; label: string }[] = [
{ value: "natural", label: "自然" },
{ value: "excited", label: "兴奋" },
{ value: "calm", label: "沉稳" },
{ value: "friendly", label: "亲切" },
{ value: "neutral", label: "自然" },
{ value: "happy", label: "开心" },
{ value: "sad", label: "难过" },
{ value: "angry", label: "生气" },
{ value: "surprised", label: "惊讶" },
{ value: "fearful", label: "恐惧" },
{ value: "disgusted", label: "厌恶" },
]
/* ── 语言 ── */
export type VoiceLanguage = "mandarin" | "english" | "cantonese"
/** 系统预置音色支持的语言(zh/en) */
export type PresetVoiceLanguage = "zh" | "en"
/** 克隆音色支持的完整语言列表 */
export type CloneVoiceLanguage = "zh" | "en" | "ja" | "ko"
export type VoiceLanguage = PresetVoiceLanguage | CloneVoiceLanguage
export const VOICE_LANGUAGE_OPTIONS: { value: VoiceLanguage; label: string }[] = [
{ value: "mandarin", label: "普通话" },
{ value: "english", label: "English" },
{ value: "cantonese", label: "粤语" },
export const PRESET_VOICE_LANGUAGE_OPTIONS: { value: PresetVoiceLanguage; label: string }[] = [
{ value: "zh", label: "中文" },
{ value: "en", label: "English" },
]
export const CLONE_VOICE_LANGUAGE_OPTIONS: { value: CloneVoiceLanguage; label: string }[] = [
{ value: "zh", label: "中文" },
{ value: "en", label: "English" },
{ value: "ja", label: "日本語" },
{ value: "ko", label: "한국어" },
]
/** 默认(预置音色)语言选项 */
export const VOICE_LANGUAGE_OPTIONS = PRESET_VOICE_LANGUAGE_OPTIONS
/* ── 对口型任务状态 ── */
export type LipsyncStatus = "idle" | "pending" | "processing" | "completed" | "failed"
+39 -10
View File
@@ -6,21 +6,50 @@
*/
import type { AiAvatarTitleConfig, AiAvatarCoverConfig, VoiceEmotion } from "../types"
/* ── 情绪:中文 → 英文(防御性映射;state 默认已是英文) ── */
const EMOTION_ZH_TO_EN: Record<string, VoiceEmotion> = {
自然: "natural",
兴奋: "excited",
沉稳: "calm",
亲切: "friendly",
/* ── 情绪:中文/旧枚举 → CosyVoice 7 种英文枚举 ── */
const EMOTION_ALIAS: Record<string, VoiceEmotion> = {
// 新英文枚举
neutral: "neutral",
happy: "happy",
sad: "sad",
angry: "angry",
surprised: "surprised",
fearful: "fearful",
disgusted: "disgusted",
// 旧英文枚举(4 种,向前兼容)
natural: "neutral",
excited: "happy",
calm: "neutral",
friendly: "happy",
// 中文
自然: "neutral",
开心: "happy",
难过: "sad",
生气: "angry",
惊讶: "surprised",
恐惧: "fearful",
厌恶: "disgusted",
// 旧中文
兴奋: "happy",
沉稳: "neutral",
亲切: "happy",
}
const VALID_EMOTIONS: VoiceEmotion[] = ["natural", "excited", "calm", "friendly"]
const VALID_EMOTIONS: VoiceEmotion[] = [
"neutral",
"happy",
"sad",
"angry",
"surprised",
"fearful",
"disgusted",
]
/** 归一化为后端英文枚举 natural/excited/calm/friendly;非法/空值回退 natural。 */
/** 归一化为后端英文枚举 neutral/happy/sad/angry/surprised/fearful/disgusted;非法/空值回退 neutral。 */
export function normalizeEmotion(raw: string | undefined | null): VoiceEmotion {
if (!raw) return "natural"
if (!raw) return "neutral"
const v = raw.trim()
if ((VALID_EMOTIONS as string[]).includes(v)) return v as VoiceEmotion
return EMOTION_ZH_TO_EN[v] ?? "natural"
return EMOTION_ALIAS[v] ?? "neutral"
}
/* ── 标题:前端 state → 后端 build_title_drawtext_filter 字段(单个 title_config dict) ── */
+2 -2
View File
@@ -138,11 +138,11 @@ const VoiceLibrary: React.FC = () => {
ttsAudioUrl,
ttsError,
setTtsText,
setTtsVoiceId,
setTtsSpeed,
setTtsEmotion,
setTtsLanguage,
setTtsOpen,
handleVoiceChange,
handleTtsSynthesize,
handleTtsSave,
handleTtsClose,
@@ -378,7 +378,7 @@ const VoiceLibrary: React.FC = () => {
.map((v) => ({ id: v.id, name: v.name }))}
onTtsClose={handleTtsClose}
onTtsTextChange={setTtsText}
onTtsVoiceChange={setTtsVoiceId}
onTtsVoiceChange={handleVoiceChange}
onTtsSpeedChange={setTtsSpeed}
onTtsEmotionChange={setTtsEmotion}
onTtsLanguageChange={setTtsLanguage}
@@ -1,4 +1,4 @@
import React from "react"
import React, { useMemo } from "react"
import { Modal } from "antd"
import { type TtsModalProps, type TtsStatus } from "./tts-modal/types"
import TextInputSection from "./tts-modal/TextInputSection"
@@ -9,6 +9,7 @@ import LanguageControl from "./tts-modal/LanguageControl"
import SynthesizeButton from "./tts-modal/SynthesizeButton"
import ErrorAlert from "./tts-modal/ErrorAlert"
import ResultPanel from "./tts-modal/ResultPanel"
import { PRESET_TTS_LANGUAGE_OPTIONS, CLONE_TTS_LANGUAGE_OPTIONS } from "./tts-modal/constants"
/** AI 配音弹窗 */
const TtsModal: React.FC<TtsModalProps> = ({
@@ -32,6 +33,13 @@ const TtsModal: React.FC<TtsModalProps> = ({
onSynthesize,
onSave,
}) => {
// 预置音色(含空默认)仅支持 zh/en;克隆音色支持全语言列表
const isCloneVoice = useMemo(
() => !!ttsVoiceId && (clonedVoices ?? []).some((v) => v.id === ttsVoiceId),
[ttsVoiceId, clonedVoices],
)
const languageOptions = isCloneVoice ? CLONE_TTS_LANGUAGE_OPTIONS : PRESET_TTS_LANGUAGE_OPTIONS
return (
<Modal title="AI 配音" open={open} onCancel={onClose} footer={null} width={560}>
<div
@@ -57,7 +65,11 @@ const TtsModal: React.FC<TtsModalProps> = ({
}}
>
<EmotionControl emotion={ttsEmotion} onChange={onEmotionChange} />
<LanguageControl language={ttsLanguage} onChange={onLanguageChange} />
<LanguageControl
language={ttsLanguage}
onChange={onLanguageChange}
options={languageOptions}
/>
</div>
<SpeedControl speed={ttsSpeed} onChange={onSpeedChange} />
<SynthesizeButton status={ttsStatus} text={ttsText} onClick={onSynthesize} />
@@ -1,13 +1,28 @@
import React from "react"
import { TTS_LANGUAGE_OPTIONS, type TtsLanguage } from "./constants"
import type { TtsLanguage } from "./constants"
import { TTS_LANGUAGE_OPTIONS } from "./constants"
interface LanguageOption {
value: string
label: string
}
interface LanguageControlProps {
language: TtsLanguage
onChange: (language: TtsLanguage) => void
/** 根据音色来源传入可选语言列表;不传则使用完整列表 */
options?: readonly LanguageOption[]
}
/** 语言选择下拉 */
const LanguageControl: React.FC<LanguageControlProps> = ({ language, onChange }) => {
const LanguageControl: React.FC<LanguageControlProps> = ({
language,
onChange,
options = TTS_LANGUAGE_OPTIONS,
}) => {
// 若当前语言不在可选列表(如从克隆切到预置时 ja/ko 失效),自动回退到中文
const effectiveValue = options.some((o) => o.value === language) ? language : "zh-CN"
return (
<div>
<div
@@ -20,7 +35,7 @@ const LanguageControl: React.FC<LanguageControlProps> = ({ language, onChange })
语言
</div>
<select
value={language}
value={effectiveValue}
onChange={(e) => onChange(e.target.value as TtsLanguage)}
style={{
width: "100%",
@@ -34,7 +49,7 @@ const LanguageControl: React.FC<LanguageControlProps> = ({ language, onChange })
cursor: "pointer",
}}
>
{TTS_LANGUAGE_OPTIONS.map((opt) => (
{options.map((opt) => (
<option key={opt.value} value={opt.value}>
{opt.label}
</option>
@@ -1,22 +1,40 @@
/** TTS 情绪选项(对齐后端 CosyVoice 支持:natural/excited/calm/friendly) */
/**
* TTS 情绪选项(对齐后端 CosyVoice 支持:neutral/happy/sad/angry/surprised/fearful/disgusted)
* 后端 normalize_emotion() 会将 value 映射为内部英文枚举
*/
export const TTS_EMOTION_OPTIONS = [
{ value: "natural", label: "自然" },
{ value: "excited", label: "兴奋" },
{ value: "calm", label: "沉稳" },
{ value: "friendly", label: "亲切" },
{ value: "neutral", label: "自然" },
{ value: "happy", label: "开心" },
{ value: "sad", label: "难过" },
{ value: "angry", label: "生气" },
{ value: "surprised", label: "惊讶" },
{ value: "fearful", label: "恐惧" },
{ value: "disgusted", label: "厌恶" },
] as const
export type TtsEmotion = (typeof TTS_EMOTION_OPTIONS)[number]["value"]
/** TTS 语言选项 */
export const TTS_LANGUAGE_OPTIONS = [
/**
* TTS 语言选项
* - 系统预置音色:仅支持 zh/en(多语言需使用克隆音色)
* - 克隆音色:完整语言列表
*/
export const PRESET_TTS_LANGUAGE_OPTIONS = [
{ value: "zh-CN", label: "中文" },
{ value: "en", label: "英文" },
] as const
export const CLONE_TTS_LANGUAGE_OPTIONS = [
{ value: "zh-CN", label: "中文" },
{ value: "en", label: "英文" },
{ value: "ja", label: "日文" },
{ value: "ko", label: "韩文" },
] as const
export type TtsLanguage = (typeof TTS_LANGUAGE_OPTIONS)[number]["value"]
export const TTS_LANGUAGE_OPTIONS = CLONE_TTS_LANGUAGE_OPTIONS
export const DEFAULT_TTS_EMOTION: TtsEmotion = "natural"
export type TtsLanguage = (typeof TTS_LANGUAGE_OPTIONS)[number]["value"]
export type PresetTtsLanguage = (typeof PRESET_TTS_LANGUAGE_OPTIONS)[number]["value"]
export const DEFAULT_TTS_EMOTION: TtsEmotion = "neutral"
export const DEFAULT_TTS_LANGUAGE: TtsLanguage = "zh-CN"
@@ -7,12 +7,18 @@ import type { TtsClonedVoiceOption } from "../components/tts-modal/VoiceSelector
import {
DEFAULT_TTS_EMOTION,
DEFAULT_TTS_LANGUAGE,
PRESET_TTS_LANGUAGE_OPTIONS,
type TtsEmotion,
type TtsLanguage,
} from "../components/tts-modal/constants"
export type TtsStatus = "idle" | "synthesizing" | "done" | "error"
/** 预置音色支持的语言 value 集合(允许任意 TtsLanguage 做 Has 检查) */
const PRESET_LANG_SET: ReadonlySet<TtsLanguage> = new Set<TtsLanguage>(
PRESET_TTS_LANGUAGE_OPTIONS.map((o) => o.value as TtsLanguage),
)
/**
* TTS 合成 Hook
* 封装 AI 配音弹窗状态、合成请求、轮询、保存到素材库等逻辑
@@ -43,12 +49,29 @@ export function useTtsSynthesize({
const [ttsError, setTtsError] = useState<string | null>(null)
const ttsTimerRef = useRef<ReturnType<typeof setInterval> | null>(null)
/** 切换音色时:预置音色下自动回退到受支持语言(zh-CN/en) */
const handleVoiceChange = useCallback(
(voiceId: string) => {
setTtsVoiceId(voiceId)
const isClone = !!voiceId && clonedVoices.some((v) => v.id === voiceId)
if (!isClone && !PRESET_LANG_SET.has(ttsLanguage)) {
setTtsLanguage(DEFAULT_TTS_LANGUAGE)
}
},
[clonedVoices, ttsLanguage],
)
/** 开始 AI 配音合成 */
const handleTtsSynthesize = useCallback(async () => {
if (!ttsText.trim()) {
message.warning("请输入要合成的文本")
return
}
// 防御:合成前再次校验预置音色语言
const isClone = !!ttsVoiceId && clonedVoices.some((v) => v.id === ttsVoiceId)
const effectiveLang =
!isClone && !PRESET_LANG_SET.has(ttsLanguage) ? DEFAULT_TTS_LANGUAGE : ttsLanguage
setTtsError(null)
setTtsStatus("synthesizing")
setTtsAudioUrl(null)
@@ -60,7 +83,7 @@ export function useTtsSynthesize({
voice_id: ttsVoiceId || undefined,
speed: ttsSpeed,
emotion: ttsEmotion,
language: ttsLanguage,
language: effectiveLang,
})
setTtsJobId(resp.job_id)
@@ -91,7 +114,7 @@ export function useTtsSynthesize({
setTtsStatus("error")
setTtsError(msg)
}
}, [ttsText, ttsVoiceId, ttsSpeed, ttsEmotion, ttsLanguage])
}, [ttsText, ttsVoiceId, ttsSpeed, ttsEmotion, ttsLanguage, clonedVoices])
/** 保存 TTS 结果到素材库 */
const handleTtsSave = useCallback(async () => {
@@ -160,6 +183,8 @@ export function useTtsSynthesize({
setTtsEmotion,
setTtsLanguage,
setTtsOpen,
// 覆写 onVoiceChange(带语言回退)
handleVoiceChange,
// Actions
handleTtsSynthesize,
handleTtsSave,