Files
xiaoxia-saas/tests/unit/test_ai_avatar_emotion_tts_lipsync.py
T
xiaoxia 59fbbd8e26
CI/CD Pipeline / Dedup Check - skip PR tests when covered by push pipeline (push) Successful in 1s
CI/CD Pipeline / Check push changed paths (push) Successful in 12s
CI/CD Pipeline / Build Staging API Image (push) Successful in 25s
CI/CD Pipeline / Build Staging Worker Image (push) Successful in 33s
CI/CD Pipeline / Build Staging Web Image (push) Successful in 1m54s
CI/CD Pipeline / Frontend Unit Tests (push) Failing after 2m40s
CI/CD Pipeline / Deploy Staging (Watchtower auto-deploy) (push) Successful in 2m22s
CI/CD Pipeline / Integration Tests (push) Successful in 4m52s
CI/CD Pipeline / Validate - Python (mypy + alembic) (push) Successful in 5m47s
CI/CD Pipeline / Validate - Style (push) Successful in 5m48s
CI/CD Pipeline / ACR Image Cleanup (push) Successful in 2m2s
CI/CD Pipeline / Staging API Integration Tests (push) Successful in 3m24s
CI/CD Pipeline / Staging E2E Tests (push) Failing after 3m32s
CI/CD Pipeline / Unit Tests (push) Successful in 10m38s
CI/CD Pipeline / Validate - Security (push) Successful in 15m27s
CI/CD Pipeline / Production Browser E2E (push) Has been skipped
CI/CD Pipeline / Build Production Web Image (push) Failing after 23h7m13s
CI/CD Pipeline / Retag skipped Staging Web Image (push) Failing after 23h20m31s
CI/CD Pipeline / PR Build Worker Image (push) Failing after 23h22m4s
CI/CD Pipeline / PR Build Web Image (push) Failing after 23h22m4s
CI/CD Pipeline / PR Build API Image (push) Failing after 23h22m4s
CI/CD Pipeline / Deploy Production (push) Failing after 23h6m24s
CI/CD Pipeline / Build Production Worker Image (push) Failing after 23h6m33s
CI/CD Pipeline / Build Production API Image (push) Failing after 23h6m33s
CI/CD Pipeline / CI Gate (push) Failing after 23h6m33s
CI/CD Pipeline / Retag skipped Staging API Image (push) Failing after 23h19m54s
CI/CD Pipeline / Frontend Lint (push) Failing after 23h22m3s
CI/CD Pipeline / Check if frontend-only change (push) Failing after 23h22m5s
CI/CD Pipeline / Canary Release to Production (push) Failing after 23h6m24s
CI/CD Pipeline / Retag skipped Staging Worker Image (push) Failing after 23h19m51s
fix(tts): P1 TTS emotion instruction format per voice type + missing CN aliases (#1934)
fix(tts): P1 TTS emotion instruction format per voice type + missing CN aliases

- Add CN aliases 中性→neutral/伤心→sad/吃惊→surprised per 灵应 spec
- build_emotion_instruction(voice_id, enum): cloned voices use English "Speak in a {emotion} tone.", system voices with emotion Instruct (longanyang/longanhuan/longhuhu_v3) use strict Chinese format "你说话的情感是{emotion}。", default voice longxiaochun_v3 (no Instruct support) skips instruction entirely
- Expand preview emotion whitelist; 210 emotion tests passing, 15438 total unit tests passing
- E2E templates fix already in PR#1931 (0e5127df), no additional change needed
2026-09-15 16:00:58 +08:00

429 lines
17 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
"""#1822 情绪/语速透传 + 对口型 TTS 直生 + 智能封面 单元测试.
CI 增量映射:
cosyvoice_service.normalize_emotion / payload emotion
lipsync_service TTS 直生分支(voice_id+script_text)
ai_avatar_cover_service 智能选帧
"""
import os
from unittest.mock import MagicMock, patch
import pytest
os.environ.setdefault("JWT_SECRET_KEY", "dev-secret-key-for-testing")
# ── 情绪归一化 ──────────────────────────────────────────────────────────
def test_normalize_emotion_english_values():
from packages.application.cosyvoice_service import normalize_emotion
# 英文情绪统一归一化为 CosyVoice v3 官方 7 种英文枚举
assert normalize_emotion("natural") == "neutral"
assert normalize_emotion("excited") == "happy"
assert normalize_emotion("calm") == "neutral"
assert normalize_emotion("friendly") == "happy"
assert normalize_emotion("happy") == "happy"
assert normalize_emotion("sad") == "sad"
assert normalize_emotion("angry") == "angry"
assert normalize_emotion("surprised") == "surprised"
assert normalize_emotion("fearful") == "fearful"
assert normalize_emotion("disgusted") == "disgusted"
def test_normalize_emotion_chinese_values():
from packages.application.cosyvoice_service import normalize_emotion
# 中文标签归一化为对应英文枚举(CosyVoice v3 官方值)
assert normalize_emotion("自然") == "neutral"
assert normalize_emotion("兴奋") == "happy"
assert normalize_emotion("开心") == "happy"
assert normalize_emotion("悲伤") == "sad"
assert normalize_emotion("愤怒") == "angry"
assert normalize_emotion("惊讶") == "surprised"
assert normalize_emotion("恐惧") == "fearful"
assert normalize_emotion("厌恶") == "disgusted"
assert normalize_emotion("中立") == "neutral"
assert normalize_emotion("中性") == "neutral"
assert normalize_emotion("难过") == "sad"
assert normalize_emotion("伤心") == "sad"
assert normalize_emotion("生气") == "angry"
assert normalize_emotion("吃惊") == "surprised"
def test_normalize_emotion_invalid_defaults_neutral():
from packages.application.cosyvoice_service import normalize_emotion
# 空串/None 返回空(调用方不传 instruction,走默认自然情绪)
assert normalize_emotion("") == ""
assert normalize_emotion(None) == ""
# 未知情绪默认 neutral(不中断合成,warning 日志)
assert normalize_emotion("喜怒哀乐") == "neutral"
assert normalize_emotion("unknown_xyz") == "neutral"
# ── CosyVoice payload 携带 emotion + rate ──────────────────────────────
def _make_service_with_captured_client(captured: dict):
"""构造 CosyVoiceService,拦截 post 请求体到 captured['json']."""
import httpx as _httpx
from packages.application import cosyvoice_service as mod
mock_client = MagicMock(spec=_httpx.Client)
resp = MagicMock()
resp.status_code = 200
resp.json.return_value = {
"request_id": "req-1",
"output": {"audio": {"url": "https://tts/a.mp3", "duration": 1.0}},
}
resp.raise_for_status = MagicMock()
def fake_request(method, url, headers, json, timeout):
captured["json"] = json
return resp
mock_client.request.side_effect = fake_request
with patch.object(mod, "get_shared_settings") as settings_patch:
s = MagicMock()
s.cosyvoice_api_key = "sk-test"
s.cosyvoice_base_url = "https://x/api/v1"
s.cosyvoice_model = "cosyvoice-v3-flash"
s.cosyvoice_clone_model = "voice-enrollment"
s.cosyvoice_format = "mp3"
s.cosyvoice_sample_rate = 22050
s.cosyvoice_voice = "longxiaochun"
settings_patch.return_value = s
svc = mod.CosyVoiceService(http_client=mock_client)
return svc
def test_submit_synthesize_payload_uses_instruction_for_cloned_voice():
"""#1898: 克隆/设计音色传 emotion 时 instruction 走英文 'Speak in a {emotion} tone.' 格式;
不支持 Instruct 的系统音色(含默认 longxiaochun_v3)不传 instruction。"""
captured: dict = {}
svc = _make_service_with_captured_client(captured)
# 克隆音色(非 long/loong 前缀)→ 英文 tone 格式
svc.submit_synthesize_task(text="你好", voice_id="myclone_voice", speed=1.5, emotion="兴奋", language="zh-CN")
inp = captured["json"]["input"]
assert "emotion" not in inp
assert inp["instruction"] == "Speak in a happy tone."
assert inp["rate"] == 1.5
# 克隆音色不做 language_hints 限制
assert inp["language_hints"] == ["zh"]
def test_submit_synthesize_payload_uses_chinese_instruction_for_emotion_system_voice():
"""longanyang 等支持 emotion instruct 的系统音色 → 中文固定格式 '你说话的情感是{emotion}。'。"""
captured: dict = {}
svc = _make_service_with_captured_client(captured)
svc.submit_synthesize_task(text="你好", voice_id="longanyang", emotion="开心", language="zh-CN")
inp = captured["json"]["input"]
assert inp["instruction"] == "你说话的情感是happy。"
assert inp["language_hints"] == ["zh"]
def test_submit_synthesize_payload_default_voice_omits_instruction_even_with_emotion():
"""默认音色 longxiaochun_v3 官方不支持 Instruct,即使传 emotion 也不应拼 instruction,
避免被 API 忽略或报错。"""
captured: dict = {}
svc = _make_service_with_captured_client(captured)
svc.submit_synthesize_task(text="你好", voice_id="longxiaochun_v3", emotion="开心", language="zh-CN")
inp = captured["json"]["input"]
assert "instruction" not in inp
assert inp["language_hints"] == ["zh"]
def test_submit_synthesize_payload_omits_instruction_when_emotion_empty():
captured: dict = {}
svc = _make_service_with_captured_client(captured)
svc.submit_synthesize_task(text="hello", voice_id="longxiaochun_v3", language="en-US")
inp = captured["json"]["input"]
assert "instruction" not in inp
assert inp["language_hints"] == ["en"]
def test_submit_synthesize_payload_cloned_voice_english_emotion():
"""克隆音色 + 英文 emotion 枚举 → 英文 tone 格式。"""
captured: dict = {}
svc = _make_service_with_captured_client(captured)
svc.submit_synthesize_task(text="hi", voice_id="myclone_voice", emotion="sad")
inp = captured["json"]["input"]
assert inp["instruction"] == "Speak in a sad tone."
# ── 对口型 TTS 直生分支 ─────────────────────────────────────────────────
def _lipsync_service_with_mocks():
from app.services.lipsync_service import LipsyncService
db = MagicMock()
client = MagicMock()
client.is_available = True
client.submit_lipsync.return_value = {
"success": True,
"task_id": "mk-1",
"request_id": "req-1",
}
cosy = MagicMock()
cosy.submit_synthesize_task.return_value = {
"audio_url": "https://tts/raw.mp3",
"request_id": "tts-req",
"audio_duration": 3.0,
}
svc = LipsyncService(db, client=client, cosyvoice_service=cosy, voice_clone_repo=MagicMock())
# _resolve_voice_id 默认原样返回(repo.get 返回 None)
svc._voice_clone_repo.get.return_value = None
return svc, client, cosy
def test_create_job_tts_direct_mode_synthesizes_audio():
svc, client, cosy = _lipsync_service_with_mocks()
with patch("app.services.lipsync_service.tts_synthesize_and_submit") as mock_task:
mock_task.apply_async.return_value = MagicMock(id="celery-task-123")
job = svc.create_job(
user_id="user-1",
video_url="https://oss/person.mp4",
voice_id="cosy-v1",
script_text="你好世界",
speed=1.2,
emotion="兴奋",
)
# v4: TTS 模式下 create_job 返回 tts_processing 状态,dispatch Celery 任务
assert job.status == "tts_processing"
assert job.emotion == "兴奋" # 存储原始情绪值,normalize 在 CosyVoiceService 内部完成
assert job.speed == 1.2
# 不直接调用 CosyVoice(由 Celery 任务处理)
cosy.submit_synthesize_task.assert_not_called()
# 不直接提交 MediaKit(由 Celery 任务处理)
client.submit_lipsync.assert_not_called()
# dispatch 了 Celery 任务
mock_task.apply_async.assert_called_once()
def test_create_job_direct_audio_mode_skips_tts():
svc, client, cosy = _lipsync_service_with_mocks()
job = svc.create_job(
user_id="user-1",
video_url="https://oss/person.mp4",
audio_url="https://oss/ready.mp3",
)
cosy.submit_synthesize_task.assert_not_called()
_, submit_kwargs = client.submit_lipsync.call_args
assert submit_kwargs["audio_url"] == "https://oss/ready.mp3"
def test_create_job_tts_failure_raises():
"""v4: TTS 模式下 create_job 不再同步失败,而是 dispatch Celery 任务。
TTS 合成失败由 Celery 任务内部处理并更新 job 状态。"""
svc, client, cosy = _lipsync_service_with_mocks()
with patch("app.services.lipsync_service.tts_synthesize_and_submit") as mock_task:
mock_task.apply_async.return_value = MagicMock(id="celery-task-456")
job = svc.create_job(
user_id="user-1",
video_url="https://oss/person.mp4",
voice_id="v-1",
script_text="文本",
)
# create_job 成功返回 tts_processing,不直接调用 TTS
assert job.status == "tts_processing"
cosy.submit_synthesize_task.assert_not_called()
client.submit_lipsync.assert_not_called()
mock_task.apply_async.assert_called_once()
# ── refresh 同步中间状态 ────────────────────────────────────────────────
def test_refresh_syncs_running_status():
from app.services.lipsync_service import LipsyncService
db = MagicMock()
client = MagicMock()
client.get_task_status.return_value = {"success": True, "status": "running"}
svc = LipsyncService(db, client=client)
job = MagicMock()
job.status = "submitted"
job.mediakit_task_id = "mk-1"
job.id = "j-1"
svc.get_job = MagicMock(return_value=job)
result = svc.refresh_job_status("j-1", "user-1")
assert result.status == "running"
# ── 智能封面 ────────────────────────────────────────────────────────────
def test_smart_cover_selects_best_frame_and_persists():
from app.services import ai_avatar_cover_service as cov
snapshots = [
{"image_url": "https://mk/f0.jpg"},
{"image_url": "https://mk/f1.jpg"},
]
with (
patch("packages.shared.mediakit_client.get_mediakit_client") as mk_patch,
patch("packages.shared.cover_frame_scorer.score_frames") as score_patch,
patch("httpx.Client") as http_client_cls,
patch("packages.shared.storage.get_shared_storage_service") as storage_patch,
):
mk = MagicMock()
mk.is_available = True
mk.extract_frames.return_value = snapshots
mk_patch.return_value = mk
# score_frames 把 f1 选为最佳
score_patch.side_effect = lambda cands: [
{"url": "https://mk/f1.jpg", "score": 90.0, "image_path": cands[1]["image_path"]},
{"url": "https://mk/f0.jpg", "score": 60.0, "image_path": cands[0]["image_path"]},
]
# httpx.Client 连接池 mock
client_instance = MagicMock()
resp = MagicMock()
resp.content = b"IMGDATA"
resp.raise_for_status = MagicMock()
client_instance.get.return_value = resp
client_instance.__enter__ = MagicMock(return_value=client_instance)
client_instance.__exit__ = MagicMock(return_value=False)
http_client_cls.return_value = client_instance
storage = MagicMock()
storage.public_url = "https://oss.example.com"
# video_url 不是自家 OSS,不重签
storage.get_download_url.side_effect = lambda url, **kw: f"{url}?signed=1"
storage.upload_file.return_value = "https://oss.example.com/cover.jpg"
storage_patch.return_value = storage
url = cov.generate_smart_cover("https://other-host/avatar.mp4", job_id="job-1")
assert "signed=1" in url or url == "https://oss.example.com/cover.jpg"
mk.extract_frames.assert_called_once()
score_patch.assert_called_once()
# 验证使用了增大的轮询参数
call_kwargs = mk.extract_frames.call_args
assert call_kwargs.kwargs.get("poll_interval") == 2.0 or call_kwargs[1].get("poll_interval") == 2.0
assert call_kwargs.kwargs.get("max_poll_attempts") == 30 or call_kwargs[1].get("max_poll_attempts") == 30
def test_smart_cover_returns_empty_when_mediakit_unavailable():
from app.services import ai_avatar_cover_service as cov
with (
patch("packages.shared.mediakit_client.get_mediakit_client") as mk_patch,
patch("packages.shared.storage.get_shared_storage_service") as storage_patch,
):
mk = MagicMock()
mk.is_available = False
mk_patch.return_value = mk
storage = MagicMock()
storage.public_url = "https://oss.example.com"
storage_patch.return_value = storage
url = cov.generate_smart_cover("https://oss.example.com/avatar.mp4")
assert url == ""
def test_sign_video_url_resigns_own_oss_url():
"""自家 OSS 私有桶 URL 应被重签为长有效期预签名 URL"""
from app.services.ai_avatar_cover_service import _sign_video_url_for_mediakit
with patch("packages.shared.storage.get_shared_storage_service") as storage_patch:
storage = MagicMock()
storage.public_url = "https://oss.example.com"
storage.get_download_url.return_value = "https://oss.example.com/file.mp4?Expires=xxx&Signature=yyy"
storage_patch.return_value = storage
result = _sign_video_url_for_mediakit("https://oss.example.com/file.mp4")
assert "Signature=yyy" in result
storage.get_download_url.assert_called_once()
def test_sign_video_url_skips_external_url():
"""外部 URL(非自家 OSS)应原样返回,不做重签"""
from app.services.ai_avatar_cover_service import _sign_video_url_for_mediakit
with patch("packages.shared.storage.get_shared_storage_service") as storage_patch:
storage = MagicMock()
storage.public_url = "https://oss.example.com"
storage_patch.return_value = storage
result = _sign_video_url_for_mediakit("https://external-cdn.com/video.mp4")
assert result == "https://external-cdn.com/video.mp4"
storage.get_download_url.assert_not_called()
def test_extract_frames_uses_extended_poll_params():
"""验证 select_best_cover_frame 使用增大后的轮询参数"""
from app.services import ai_avatar_cover_service as cov
with (
patch("packages.shared.mediakit_client.get_mediakit_client") as mk_patch,
patch("packages.shared.storage.get_shared_storage_service") as storage_patch,
):
mk = MagicMock()
mk.is_available = True
mk.extract_frames.return_value = [{"image_url": "https://mk/f0.jpg"}]
mk_patch.return_value = mk
storage = MagicMock()
storage.public_url = "https://oss.example.com"
storage_patch.return_value = storage
cov.select_best_cover_frame("https://other/avatar.mp4", max_frames=3)
call_kwargs = mk.extract_frames.call_args
assert call_kwargs.kwargs.get("poll_interval") == 2.0 or call_kwargs[1].get("poll_interval") == 2.0
assert call_kwargs.kwargs.get("max_poll_attempts") == 30 or call_kwargs[1].get("max_poll_attempts") == 30
assert call_kwargs.kwargs.get("max_retries") == 1 or call_kwargs[1].get("max_retries") == 1
# ── 渲染 script_id 可选(手动文案直生场景)──────────────────────────────
def test_render_request_script_id_optional():
import os
import sys
sys.path.insert(0, os.path.join(os.path.dirname(__file__), "..", "..", "apps", "api"))
from app.schemas.ai_avatar_render import CreateAiAvatarRenderRequest
# 手动文案直生:不传 script_id 也合法
req = CreateAiAvatarRenderRequest(lipsync_job_id="j-1")
assert req.script_id == ""
# 空白被 strip
req2 = CreateAiAvatarRenderRequest(lipsync_job_id="j-1", script_id=" ")
assert req2.script_id == ""
# title_config 是单个 dict
req3 = CreateAiAvatarRenderRequest(
lipsync_job_id="j-1",
title_config={"text": "标题", "position": "top", "font_size": 40},
)
assert req3.title_config["position"] == "top"