diff --git a/apps/api/app/api/routes/voice_clones.py b/apps/api/app/api/routes/voice_clones.py index 940e156dd..0a2a5afc5 100755 --- a/apps/api/app/api/routes/voice_clones.py +++ b/apps/api/app/api/routes/voice_clones.py @@ -287,10 +287,15 @@ def retry_voice_clone( return _to_response(profile) +_ALLOWED_PREVIEW_EMOTIONS = {"", "natural", "excited", "calm", "friendly"} + + @router.get("/{clone_id}/preview", response_model=VoiceClonePreviewResponse) def get_voice_clone_preview( clone_id: str, text: str = Query("", description="自定义试听文本,为空则使用默认示例"), + speed: float = Query(1.0, ge=0.5, le=2.0, description="语速,0.5-2.0,默认 1.0"), + emotion: str = Query("", description="情绪:natural/excited/calm/friendly,空字符串为默认自然"), authenticated_user: AuthenticatedUser = Depends(get_current_user), repository: SQLAlchemyVoiceCloneProfileRepository = Depends(get_voice_clone_profile_repository), cosyvoice: CosyVoiceService = Depends(get_cosyvoice_service), @@ -298,11 +303,17 @@ def get_voice_clone_preview( """获取克隆音色试听音频(实时 TTS 合成)。 - 克隆音色必须处于 ready 状态 - - 使用默认试听文本时,结果缓存 7 天 - - 可传入自定义 text 参数试听不同文本 + - 使用默认试听文本时,结果缓存 7 天(仅默认 text+speed=1.0+emotion=空 组合缓存) + - 可传入自定义 text/speed/emotion 试听不同效果 """ import time + if emotion not in _ALLOWED_PREVIEW_EMOTIONS: + raise HTTPException( + status_code=status.HTTP_400_BAD_REQUEST, + detail=f"不支持的 emotion 值: {emotion},可选: natural/excited/calm/friendly 或留空", + ) + use_case = GetVoiceCloneUseCase(repository) try: profile = use_case.execute(clone_id, authenticated_user.user.id) @@ -315,8 +326,8 @@ def get_voice_clone_preview( detail=f"Voice clone is not ready (current status: {profile.status})", ) - # 有自定义文本时不缓存 - use_cache = not text.strip() + # 仅默认试听文本 + 默认 speed + 默认 emotion 时使用缓存 + use_cache = (not text.strip()) and abs(speed - 1.0) < 1e-6 and (not emotion) if use_cache and clone_id in _clone_preview_cache: audio_url, duration, file_size, cached_text, cached_at = _clone_preview_cache[clone_id] @@ -337,12 +348,15 @@ def get_voice_clone_preview( text=preview_text, voice_id=profile.voice_id, format="mp3", - speed=1.0, + speed=speed, + emotion=emotion, ) except CosyVoiceError as e: raise HTTPException(status_code=502, detail=f"TTS 合成失败: {e}") from e + except ValueError as e: + raise HTTPException(status_code=status.HTTP_400_BAD_REQUEST, detail=str(e)) from e - # 缓存(仅默认试听文本) + # 缓存(仅默认参数组合) if use_cache: _clone_preview_cache[clone_id] = ( result.audio_url, diff --git a/tests/unit/test_voice_clone_preview.py b/tests/unit/test_voice_clone_preview.py index a47694190..75d33e7fb 100755 --- a/tests/unit/test_voice_clone_preview.py +++ b/tests/unit/test_voice_clone_preview.py @@ -50,7 +50,7 @@ def _make_auth_user(user_id: str = "user_001") -> MagicMock: class TestVoiceClonePreview: """克隆音色试听接口测试。""" - def _call_preview(self, profile, cosyvoice_mock, text="", user_id="user_001"): + def _call_preview(self, profile, cosyvoice_mock, text="", user_id="user_001", speed=1.0, emotion=""): """调用路由函数,模拟 FastAPI 注入依赖。""" from app.api.routes.voice_clones import get_voice_clone_preview @@ -93,6 +93,8 @@ class TestVoiceClonePreview: result = get_voice_clone_preview( clone_id=profile.id if profile else "nonexistent", text=text, + speed=speed, + emotion=emotion, authenticated_user=_make_auth_user(user_id), repository=repo, cosyvoice=cosyvoice_mock, @@ -129,6 +131,8 @@ class TestVoiceClonePreview: call_kwargs = cosyvoice.synthesize_speech.call_args assert call_kwargs.kwargs["voice_id"] == "clone_voice_001" assert call_kwargs.kwargs["format"] == "mp3" + assert call_kwargs.kwargs["speed"] == 1.0 + assert call_kwargs.kwargs.get("emotion", "") == "" def test_preview_custom_text(self) -> None: """自定义试听文本。""" @@ -161,6 +165,76 @@ class TestVoiceClonePreview: assert exc_info.value.status_code == 404 cosyvoice.synthesize_speech.assert_not_called() + def test_preview_custom_speed_emotion_passed_through(self) -> None: + """Issue #1897: speed/emotion 参数透传到 cosyvoice.synthesize_speech。""" + from app.api.routes.voice_clones import _clone_preview_cache + + _clone_preview_cache.clear() + + profile = _make_profile() + cosyvoice = MagicMock() + cosyvoice.synthesize_speech.return_value = SynthesizeResult( + audio_url="https://oss.example.com/preview/fast.mp3", + duration=2.0, + file_size=32000, + ) + + result = self._call_preview(profile, cosyvoice, speed=1.3, emotion="excited") + + assert result.audio_url == "https://oss.example.com/preview/fast.mp3" + call_kwargs = cosyvoice.synthesize_speech.call_args.kwargs + assert call_kwargs["speed"] == 1.3 + assert call_kwargs["emotion"] == "excited" + + def test_preview_invalid_emotion_rejected(self) -> None: + """Issue #1897: 非法 emotion 值返回 400。""" + from app.api.routes.voice_clones import _clone_preview_cache + + _clone_preview_cache.clear() + + profile = _make_profile() + cosyvoice = MagicMock() + + with pytest.raises(HTTPException) as exc_info: + self._call_preview(profile, cosyvoice, emotion="angry") + + assert exc_info.value.status_code == 400 + cosyvoice.synthesize_speech.assert_not_called() + + def test_preview_non_default_speed_no_cache(self) -> None: + """Issue #1897: 自定义 speed/emotion 不走缓存。""" + from app.api.routes.voice_clones import _clone_preview_cache + + _clone_preview_cache.clear() + + profile = _make_profile() + cosyvoice = MagicMock() + cosyvoice.synthesize_speech.side_effect = [ + SynthesizeResult(audio_url="https://example.com/fast.mp3", duration=2.0, file_size=32000), + SynthesizeResult(audio_url="https://example.com/slow.mp3", duration=4.0, file_size=60000), + ] + + # 非默认speed — 不应缓存 + result1 = self._call_preview(profile, cosyvoice, speed=1.5) + result2 = self._call_preview(profile, cosyvoice, speed=0.7, emotion="calm") + assert cosyvoice.synthesize_speech.call_count == 2 + assert result1.audio_url != result2.audio_url + + def test_preview_valueerror_from_cosyvoice_returns_400(self) -> None: + """Issue #1897: cosyvoice 因 speed 非法等抛 ValueError 时返回 400(与 TTS preview 一致)。""" + from app.api.routes.voice_clones import _clone_preview_cache + + _clone_preview_cache.clear() + + profile = _make_profile() + cosyvoice = MagicMock() + cosyvoice.synthesize_speech.side_effect = ValueError("speed out of range") + + with pytest.raises(HTTPException) as exc_info: + self._call_preview(profile, cosyvoice) + + assert exc_info.value.status_code == 400 + def test_preview_not_ready_pending(self) -> None: """pending 状态的克隆音色不能试听。""" from app.api.routes.voice_clones import _clone_preview_cache