diff --git a/deploy/gpu_worker/README.md b/deploy/gpu_worker/README.md index e99923b07..41a252b61 100644 --- a/deploy/gpu_worker/README.md +++ b/deploy/gpu_worker/README.md @@ -48,8 +48,33 @@ vim .env | `MUSE_AUDIO_MAX_MB` | 音频上传大小限制 MB | `20` | | `MUSE_DEFAULT_FPS` | 视频 fps 兜底值 | `25.0` | | `MUSE_TEMP_DIR` | 临时文件目录 | `/tmp/musetalk_$$` | +| `MUSE_VIDEO_ENCODER` | 循环视频时的编码器:`auto`(优先 h264_nvenc,失败回退 libx264)/`h264_nvenc`/`libx264` | `auto` | +| `MUSE_ENABLE_VIDEO_LOOP` | 驱动音频比视频长时循环视频补齐画面,`0` 关闭 | `1` | -### 2.2 启动服务 +### 2.2 更新部署(音轨修复,必做) + +> ⚠️ 2026-09-20 修复严重 bug:旧版封装保留了源视频音轨,结果口型配的是原声而不是 TTS 驱动音频。RTX2060 机器必须重新拉取 `musetalk_server.py` 并重启: + +```bash +# 在 RTX2060 上备份旧文件并拉取新版本(按实际部署路径调整) +cp ~/projects/MuseTalk/musetalk_server.py ~/projects/MuseTalk/musetalk_server.py.bak +# 从仓库 raw 地址下载最新版(替换为你的仓库地址/分支) +wget -O ~/projects/MuseTalk/musetalk_server.py \ + "https://git.xiaoxiajianji.com/xiaoxia/xiaoxia-saas/raw/branch/develop/deploy/gpu_worker/musetalk_server.py" + +# 重启服务 +sudo systemctl restart musetalk-server +sudo systemctl status musetalk-server +curl http://127.0.0.1:7861/health +``` + +修复后封装逻辑: + +- 最终 mux 强制 `-map 0:v -map 1:a`:视频流只取 MuseTalk 无声画面,音轨只取 TTS 驱动音频,杜绝 ffmpeg 默认行为带入源视频音轨 +- 驱动音频不长于视频时:`-c:v copy -c:a aac -shortest`,无损秒封装 +- 驱动音频长于视频时(如 TTS 15s vs 视频 9s):`-stream_loop -1` 循环画面,RTX2060 走 `h264_nvenc` 硬件重编码(NVENC 失败自动回退 libx264),`-t` 精确卡到音频时长 + +### 2.3 启动服务 ```bash # 前台运行(调试用) @@ -60,7 +85,7 @@ sudo systemctl start musetalk-server sudo systemctl enable musetalk-server ``` -### 2.3 验证健康检查 +### 2.4 验证健康检查 ```bash curl http://127.0.0.1:7861/health @@ -184,3 +209,9 @@ MuseTalk 健康检查通过: {...} 新增: - `/cancel` 端点:终止当前推理任务,清理临时文件 - `/health` 端点:返回 GPU 显存信息和当前任务状态 + +2026-09-20 追加修复(音轨正确性,上线阻断级): + +9. **音轨未替换(严重)**:旧最终封装让 ffmpeg 默认选流,结果保留了源视频自带音轨(与画面相关系数 0.9998,与 TTS 无关)。改为 `_mux_video_with_audio()` 统一封装,强制 `-map 0:v:0 -map 1:a:0`,画面取 MuseTalk 无声产物、音轨只取驱动音频 +10. **音视频时长不对齐**:TTS 长于原视频时 `-shortest` 会截短语音。改为探测双方时长,音频更长时 `-stream_loop -1` 循环画面 + `h264_nvenc` 硬件重编码(`MUSE_VIDEO_ENCODER=auto`,失败回退 libx264)+ `-t <音频时长>`;不循环时 `-c:v copy` 秒封装 + - 开关 `MUSE_ENABLE_VIDEO_LOOP=0` 可关闭循环;请求也支持 form 参数 `enable_video_loop` 单任务覆盖 diff --git a/deploy/gpu_worker/musetalk_server.py b/deploy/gpu_worker/musetalk_server.py index 007a360ba..48c288bfd 100644 --- a/deploy/gpu_worker/musetalk_server.py +++ b/deploy/gpu_worker/musetalk_server.py @@ -11,6 +11,8 @@ MUSE_AUDIO_MAX_MB 音频上传大小限制 MB,默认 20 MUSE_DEFAULT_FPS 视频 fps 兜底值,默认 25.0 MUSE_TEMP_DIR 临时文件目录,默认 /tmp/musetalk_$$ + MUSE_VIDEO_ENCODER 循环视频时的编码器:auto(默认,优先 h264_nvenc 兜底 libx264)/h264_nvenc/libx264 + MUSE_ENABLE_VIDEO_LOOP 驱动音频比视频长时是否循环视频补齐,默认 1(开启) 接口: GET /health 健康检查 + GPU 显存信息 @@ -57,6 +59,12 @@ class Config: audio_max_mb: int = int(_env("MUSE_AUDIO_MAX_MB", "20")) default_fps: float = float(_env("MUSE_DEFAULT_FPS", "25.0")) temp_dir: str = _env("MUSE_TEMP_DIR", f"/tmp/musetalk_{os.getpid()}") + # 循环视频时编码器:auto 优先 h264_nvenc(RTX2060 支持),失败兜底 libx264 + video_encoder: str = _env("MUSE_VIDEO_ENCODER", "auto") or "auto" + # 驱动音频比视频长时循环视频补齐画面 + enable_video_loop: bool = _env("MUSE_ENABLE_VIDEO_LOOP", "1") not in ("0", "false", "False", "") + # 判定音视频时长差异的容差(秒),避免 ffprobe 微小误差触发无谓的循环/重编码 + duration_epsilon: float = 0.25 # ── 全局状态 ────────────────────────────────────────────────────────── @@ -159,6 +167,152 @@ def _get_video_fps(video_path: Path) -> float: return Config.default_fps +def _get_media_duration(path: Path) -> float: + """用 ffprobe 读媒体时长(秒),失败返回 0.0.""" + try: + out = subprocess.check_output( + [ + "ffprobe", + "-v", + "error", + "-show_entries", + "format=duration", + "-of", + "default=noprint_wrappers=1:nokey=1", + str(path), + ], + stderr=subprocess.DEVNULL, + timeout=10, + ) + duration = float(out.decode().strip()) + return duration if duration > 0 else 0.0 + except Exception as exc: + logger.warning("ffprobe 读时长失败 %s: %s", path, exc) + return 0.0 + + +def _pick_video_encoder() -> str: + """选择视频编码器:配置指定则用指定值;auto 时探测 NVENC 是否可用,不可用回退 libx264.""" + configured = Config.video_encoder.strip() + if configured in ("h264_nvenc", "libx264"): + return configured + # auto:探测本机 ffmpeg 是否编译了 h264_nvenc + try: + result = subprocess.run( + ["ffmpeg", "-hide_banner", "-encoders"], + stdout=subprocess.PIPE, + stderr=subprocess.DEVNULL, + timeout=10, + check=False, + ) + if b"h264_nvenc" in result.stdout: + return "h264_nvenc" + except Exception as exc: + logger.warning("探测 ffmpeg 编码器失败,回退 libx264: %s", exc) + return "libx264" + + +def _mux_video_with_audio( + video_path: Path, + audio_path: Path, + output_path: Path, + enable_video_loop: Optional[bool] = None, + timeout: float = 300, +) -> None: + """把无声画面视频与驱动音频封装为最终结果. + + 关键正确性要求:必须用 -map 0:v -map 1:a 显式指定取第一个输入(推理画面)的 + 视频流和第二个输入(驱动音频 TTS)的音频流,禁止 ffmpeg 默认流选择行为 + (否则会把源视频自带音轨带进结果,口型与声音错位)。 + + 时长对齐:驱动音频比视频长时(TTS 15s vs 原视频 9s 很常见),用 + -stream_loop -1 循环视频画面到音频长度(NVENC 硬件重编码),-t 卡到音频时长; + 音频不超过视频时直接 -c:v copy 无损快封装,-shortest 以较短流为准。 + """ + video_duration = _get_media_duration(video_path) + audio_duration = _get_media_duration(audio_path) + + loop_enabled = Config.enable_video_loop if enable_video_loop is None else enable_video_loop + need_loop = bool( + loop_enabled + and audio_duration > 0 + and video_duration > 0 + and audio_duration > video_duration + Config.duration_epsilon + ) + + if need_loop: + encoder = _pick_video_encoder() + # preset 随编码器选择:h264_nvenc 用 p1-p7,libx264 用词形 preset + preset = "p4" if encoder == "h264_nvenc" else "veryfast" + logger.info( + "音频(%.2fs)长于视频(%.2fs),循环视频并以 %s(%s) 重编码至音频长度", + audio_duration, + video_duration, + encoder, + preset, + ) + + def build_cmd(enc: str, pre: str) -> list: + return [ + "ffmpeg", + "-y", + "-stream_loop", + "-1", + "-i", + str(video_path), + "-i", + str(audio_path), + "-map", + "0:v:0", + "-map", + "1:a:0", + "-c:v", + enc, + "-preset", + pre, + "-c:a", + "aac", + "-b:a", + "128k", + "-t", + f"{audio_duration:.3f}", + str(output_path), + ] + + try: + _run_ffmpeg(build_cmd(encoder, preset), timeout=timeout) + except RuntimeError: + # NVENC 可能因驱动/占用失败,兜底 libx264 重试一次 + if encoder == "h264_nvenc": + logger.warning("h264_nvenc 封装失败,回退 libx264 重试") + _run_ffmpeg(build_cmd("libx264", "veryfast"), timeout=timeout) + else: + raise + else: + # 视频不短于音频:直接复制视频流,只把音频替换为驱动音频并转 AAC + cmd = [ + "ffmpeg", + "-y", + "-i", + str(video_path), + "-i", + str(audio_path), + "-map", + "0:v:0", + "-map", + "1:a:0", + "-c:v", + "copy", + "-c:a", + "aac", + "-b:a", + "128k", + "-shortest", + str(output_path), + ] + _run_ffmpeg(cmd, timeout=timeout) + + def _check_file_size(file, max_mb: int, label: str) -> Optional[str]: """检查文件大小,超限返回错误信息,否则返回 None.""" file.seek(0, 2) @@ -190,11 +344,18 @@ def _run_ffmpeg(cmd: list, timeout: float = 120) -> subprocess.CompletedProcess: raise RuntimeError(f"ffmpeg 超时(>{timeout}s)") from exc -def _run_inference(video_path: Path, audio_path: Path, output_path: Path) -> None: +def _run_inference( + video_path: Path, + audio_path: Path, + output_path: Path, + enable_video_loop: Optional[bool] = None, +) -> None: """执行 MuseTalk 推理(可被子线程和测试独立调用). 实际部署时替换为 MuseTalk 真实推理逻辑。 - 此处为示例实现:提取帧 → 合并音视频。 + 此处为示例实现:提取帧 → 生成无声画面 → 用驱动音频封装。 + + enable_video_loop: 驱动音频长于视频时是否循环视频;None 走全局配置。 """ fps = _get_video_fps(video_path) logger.info("视频 fps: %.2f", fps) @@ -218,27 +379,32 @@ def _run_inference(video_path: Path, audio_path: Path, output_path: Path) -> Non if not frame_files: raise RuntimeError("未从视频中提取到帧") - # TODO: 替换为 MuseTalk 实际推理逻辑 + # TODO: 替换为 MuseTalk 实际推理逻辑。 + # MuseTalk 真实产物是「无声画面视频」,音轨必须在封装阶段用驱动音频替换。 logger.warning("使用示例推理逻辑,未实际调用 MuseTalk 模型") + # 示例:从源视频生成无声画面(-an 丢弃原音轨),模拟 MuseTalk 推理产物。 + # 真实部署时 silent_video_path 应替换为 MuseTalk 输出的无声视频路径。 + silent_video_path = video_path.parent / "visual_silent.mp4" _run_ffmpeg( [ "ffmpeg", "-y", "-i", str(video_path), - "-i", - str(audio_path), + "-an", "-c:v", "libx264", - "-c:a", - "aac", - "-shortest", - str(output_path), + "-preset", + "veryfast", + str(silent_video_path), ], timeout=300, ) + # 统一封装:显式 -map 取推理画面 + 驱动音频;音频更长时循环视频。 + _mux_video_with_audio(silent_video_path, audio_path, output_path, enable_video_loop=enable_video_loop) + if not output_path.exists() or output_path.stat().st_size < 1024: raise RuntimeError("推理产物不存在或过小") @@ -286,6 +452,13 @@ def inference(): audio_file = request.files["audio"] task_id = request.form.get("task_id", f"task_{int(time.time())}") + # 可选:本次任务是否在音频长于视频时循环视频(缺省走全局配置) + loop_param = request.form.get("enable_video_loop") + if loop_param is not None: + task_enable_loop = loop_param.strip() not in ("0", "false", "False", "") + else: + task_enable_loop = None + # 文件大小检查 err = _check_file_size(video_file, Config.video_max_mb, "视频") if err: @@ -319,7 +492,7 @@ def inference(): def inference_thread(): try: - _run_inference(video_path, audio_path, output_path) + _run_inference(video_path, audio_path, output_path, enable_video_loop=task_enable_loop) except Exception as exc: result_container["error"] = str(exc) @@ -413,6 +586,8 @@ def main(): logger.info(" 视频大小限制: %dMB", Config.video_max_mb) logger.info(" 音频大小限制: %dMB", Config.audio_max_mb) logger.info(" 默认 fps: %.1f", Config.default_fps) + logger.info(" 视频编码器: %s", Config.video_encoder) + logger.info(" 音频长于视频时循环视频: %s", Config.enable_video_loop) logger.info("=" * 60) # 检查 GPU diff --git a/tests/unit/test_1978_musetalk_audio_mux.py b/tests/unit/test_1978_musetalk_audio_mux.py new file mode 100644 index 000000000..764a6169c --- /dev/null +++ b/tests/unit/test_1978_musetalk_audio_mux.py @@ -0,0 +1,374 @@ +"""#1978 MuseTalk 服务端音轨替换 + 视频循环修复单测. + +覆盖 deploy/gpu_worker/musetalk_server.py: +1. 最终封装必须 -map 0:v -map 1:a 取「推理画面 + 驱动音频」,禁止默认流选择带入源视频音轨 +2. 音频不超过视频:-c:v copy + -shortest 快速封装 +3. 音频长于视频:-stream_loop -1 循环视频,NVENC/libx264 重编码,-t 卡到音频时长 +4. enable_video_loop=false 时即使音频更长也不循环 +5. h264_nvenc 失败自动回退 libx264 +6. 真实 ffmpeg 端到端:源视频内置 200Hz 音轨 + 驱动音频 800Hz,结果音轨必须是 800Hz + (过零率估计),证明音轨来自第二个输入而非源视频;音频更长时输出时长对齐音频 +""" + +from __future__ import annotations + +import importlib.util +import os +import shutil +import subprocess +import sys +from pathlib import Path +from unittest import mock + +import pytest + +try: + import flask # noqa: F401 + + HAS_FLASK = True +except ImportError: + HAS_FLASK = False + +pytestmark = pytest.mark.skipif(not HAS_FLASK, reason="Flask 未安装(gpu_worker 独立部署依赖)") + +ROOT = Path(__file__).resolve().parents[2] +SERVER_PATH = ROOT / "deploy" / "gpu_worker" / "musetalk_server.py" +HAS_FFMPEG = shutil.which("ffmpeg") is not None and shutil.which("ffprobe") is not None + + +def _load_server(name: str): + if name in sys.modules: + del sys.modules[name] + spec = importlib.util.spec_from_file_location(name, SERVER_PATH) + mod = importlib.util.module_from_spec(spec) + sys.modules[name] = mod + spec.loader.exec_module(mod) + return mod + + +@pytest.fixture +def server(tmp_path, monkeypatch): + if not HAS_FLASK: + pytest.skip("Flask 未安装") + monkeypatch.setenv("MUSE_TEMP_DIR", str(tmp_path / "musetalk_temp")) + monkeypatch.setenv("MUSE_VIDEO_ENCODER", "libx264") + mod = _load_server(f"musetalk_mux_{os.getpid()}_{id(tmp_path)}") + mod.Config.video_encoder = "libx264" + mod.Config.enable_video_loop = True + return mod + + +# ── 命令构造:非循环路径 ───────────────────────────────────────────── + + +def test_mux_non_loop_maps_video_and_drives_audio(server, tmp_path): + """音频(5s)不长于视频(10s):显式 map 0:v/1:a,视频流 copy,-shortest.""" + video = tmp_path / "visual.mp4" + audio = tmp_path / "tts.mp3" + video.write_bytes(b"v") + audio.write_bytes(b"a") + captured = {} + + def fake_run(cmd, timeout=300): + captured["cmd"] = cmd + + with ( + mock.patch.object(server, "_get_media_duration", side_effect=[10.0, 5.0]), + mock.patch.object(server, "_run_ffmpeg", side_effect=fake_run), + ): + server._mux_video_with_audio(video, audio, tmp_path / "out.mp4") + + cmd = captured["cmd"] + # 输入顺序:0=无声画面,1=驱动音频 + assert cmd.index(str(video)) < cmd.index(str(audio)) + # 关键修复:强制流映射,不能让 ffmpeg 默认选择源视频音轨 + assert "-map" in cmd + assert "0:v:0" in cmd + assert "1:a:0" in cmd + assert "-c:v" in cmd and cmd[cmd.index("-c:v") + 1] == "copy" + assert "-shortest" in cmd + # 非循环不重编码 + assert "-stream_loop" not in cmd + assert "-t" not in cmd + + +def test_mux_non_loop_duration_epsilon(server, tmp_path): + """音频略长于视频但在容差内(0.25s)不触发循环重编码.""" + video = tmp_path / "visual.mp4" + audio = tmp_path / "tts.mp3" + video.write_bytes(b"v") + audio.write_bytes(b"a") + captured = {} + with ( + mock.patch.object(server, "_get_media_duration", side_effect=[9.0, 9.1]), + mock.patch.object(server, "_run_ffmpeg", side_effect=lambda cmd, timeout=300: captured.update(cmd=cmd)), + ): + server._mux_video_with_audio(video, audio, tmp_path / "out.mp4") + assert "-stream_loop" not in captured["cmd"] + + +# ── 命令构造:循环路径 ─────────────────────────────────────────────── + + +def test_mux_loop_when_audio_longer_uses_stream_loop_and_nvenc(server, tmp_path): + """音频(15s)长于视频(9s):-stream_loop -1 循环、NVENC 重编码、-t 音频时长.""" + video = tmp_path / "visual.mp4" + audio = tmp_path / "tts.mp3" + video.write_bytes(b"v") + audio.write_bytes(b"a") + server.Config.video_encoder = "h264_nvenc" + captured = {} + with ( + mock.patch.object(server, "_get_media_duration", side_effect=[9.0, 15.0]), + mock.patch.object(server, "_run_ffmpeg", side_effect=lambda cmd, timeout=300: captured.update(cmd=cmd)), + ): + server._mux_video_with_audio(video, audio, tmp_path / "out.mp4") + + cmd = captured["cmd"] + # -stream_loop 必须位于第一个 -i 之前 + assert "-stream_loop" in cmd + sl_idx = cmd.index("-stream_loop") + assert cmd[sl_idx + 1] == "-1" + assert sl_idx < cmd.index("-i") + # 同样必须显式 map + assert "0:v:0" in cmd and "1:a:0" in cmd + assert cmd[cmd.index("-c:v") + 1] == "h264_nvenc" + # -t 卡到音频时长,且不用 -shortest(避免截短音频) + assert "-shortest" not in cmd + t_idx = cmd.index("-t") + assert abs(float(cmd[t_idx + 1]) - 15.0) < 0.01 + + +def test_mux_loop_disabled_falls_back_to_copy(server, tmp_path): + """enable_video_loop=False:即使音频更长也不循环,走 copy+shortest.""" + video = tmp_path / "visual.mp4" + audio = tmp_path / "tts.mp3" + video.write_bytes(b"v") + audio.write_bytes(b"a") + captured = {} + with ( + mock.patch.object(server, "_get_media_duration", side_effect=[9.0, 15.0]), + mock.patch.object(server, "_run_ffmpeg", side_effect=lambda cmd, timeout=300: captured.update(cmd=cmd)), + ): + server._mux_video_with_audio(video, audio, tmp_path / "out.mp4", enable_video_loop=False) + assert "-stream_loop" not in captured["cmd"] + assert captured["cmd"][captured["cmd"].index("-c:v") + 1] == "copy" + + +def test_mux_nvenc_failure_falls_back_to_libx264(server, tmp_path): + """NVENC 调用失败时自动用 libx264 重试一次.""" + video = tmp_path / "visual.mp4" + audio = tmp_path / "tts.mp3" + video.write_bytes(b"v") + audio.write_bytes(b"a") + server.Config.video_encoder = "h264_nvenc" + cmds = [] + + def runner(cmd, timeout=300): + cmds.append(list(cmd)) + if cmd[cmd.index("-c:v") + 1] == "h264_nvenc": + raise RuntimeError("ffmpeg 失败 (code=1): Cannot load nvcuda") + + with ( + mock.patch.object(server, "_get_media_duration", side_effect=[9.0, 15.0]), + mock.patch.object(server, "_run_ffmpeg", side_effect=runner), + ): + server._mux_video_with_audio(video, audio, tmp_path / "out.mp4") + + assert len(cmds) == 2 + assert cmds[0][cmds[0].index("-c:v") + 1] == "h264_nvenc" + second = cmds[1] + assert second[second.index("-c:v") + 1] == "libx264" + # nvenc 的 preset p4 已替换为 x264 兼容值 + assert "p4" not in second + assert "0:v:0" in second and "1:a:0" in second + + +def test_mux_copy_failure_propagates(server, tmp_path): + """非循环路径 ffmpeg 失败应抛出(不静默吞错).""" + video = tmp_path / "visual.mp4" + audio = tmp_path / "tts.mp3" + video.write_bytes(b"v") + audio.write_bytes(b"a") + with ( + mock.patch.object(server, "_get_media_duration", side_effect=[10.0, 5.0]), + mock.patch.object(server, "_run_ffmpeg", side_effect=RuntimeError("ffmpeg 失败")), + ): + with pytest.raises(RuntimeError): + server._mux_video_with_audio(video, audio, tmp_path / "out.mp4") + + +def test_pick_video_encoder_respects_config(server): + """显式配置的编码器优先,auto 时探测.""" + server.Config.video_encoder = "libx264" + assert server._pick_video_encoder() == "libx264" + server.Config.video_encoder = "h264_nvenc" + assert server._pick_video_encoder() == "h264_nvenc" + + +def test_pick_video_encoder_auto_detects_nvenc(server): + """auto 模式:ffmpeg -encoders 含 h264_nvenc 则选它.""" + server.Config.video_encoder = "auto" + completed = subprocess.CompletedProcess(args=["ffmpeg"], returncode=0, stdout=b"... h264_nvenc ...", stderr=b"") + with mock.patch("subprocess.run", return_value=completed): + assert server._pick_video_encoder() == "h264_nvenc" + + +# ── 真实 ffmpeg 端到端:音轨来源与时长对齐 ──────────────────────────── + + +@pytest.mark.skipif(not HAS_FFMPEG, reason="环境无 ffmpeg/ffprobe") +def _make_media(tmp_path: Path): + """生成:带 200Hz 音轨的 2s 源视频 + 800Hz 的 5s 驱动音频.""" + source_video = tmp_path / "source.mp4" + drive_audio = tmp_path / "drive.wav" + subprocess.run( + [ + "ffmpeg", + "-y", + "-f", + "lavfi", + "-i", + "testsrc=duration=2:size=160x120:rate=25", + "-f", + "lavfi", + "-i", + "sine=frequency=200:duration=2", + "-c:v", + "libx264", + "-preset", + "ultrafast", + "-c:a", + "aac", + str(source_video), + ], + stdout=subprocess.DEVNULL, + stderr=subprocess.DEVNULL, + check=True, + ) + subprocess.run( + [ + "ffmpeg", + "-y", + "-f", + "lavfi", + "-i", + "sine=frequency=800:duration=5", + str(drive_audio), + ], + stdout=subprocess.DEVNULL, + stderr=subprocess.DEVNULL, + check=True, + ) + return source_video, drive_audio + + +def _probe_duration(path: Path) -> float: + out = subprocess.check_output( + [ + "ffprobe", + "-v", + "error", + "-show_entries", + "format=duration", + "-of", + "default=noprint_wrappers=1:nokey=1", + str(path), + ] + ) + return float(out.decode().strip()) + + +def _estimate_audio_freq(path: Path, duration: float) -> float: + """解码为 8kHz 单声道 s16 PCM,用过零率估计主频.""" + raw = subprocess.check_output( + [ + "ffmpeg", + "-i", + str(path), + "-vn", + "-ac", + "1", + "-ar", + "8000", + "-f", + "s16le", + "-", + ], + stderr=subprocess.DEVNULL, + ) + import array + + samples = array.array("h") + samples.frombytes(raw) + if len(samples) < 100: + return 0.0 + crossings = sum(1 for i in range(1, len(samples)) if (samples[i - 1] < 0) != (samples[i] < 0)) + secs = len(samples) / 8000 + return crossings / 2.0 / secs + + +@pytest.mark.skipif(not HAS_FFMPEG, reason="环境无 ffmpeg/ffprobe") +def test_real_mux_replaces_source_audio_with_drive_audio(server, tmp_path): + """端到端:结果音轨必须是驱动音频 800Hz,而不是源视频的 200Hz.""" + source_video, drive_audio = _make_media(tmp_path) + + # 模拟 MuseTalk 无声画面产物 + silent_video = tmp_path / "visual_silent.mp4" + subprocess.run( + [ + "ffmpeg", + "-y", + "-i", + str(source_video), + "-an", + "-c:v", + "libx264", + "-preset", + "ultrafast", + str(silent_video), + ], + stdout=subprocess.DEVNULL, + stderr=subprocess.DEVNULL, + check=True, + ) + + output = tmp_path / "output.mp4" + server._mux_video_with_audio(silent_video, drive_audio, output) + assert output.exists() and output.stat().st_size > 1024 + + # 驱动音频 5s 长于画面 2s → 输出应接近 5s(循环补齐) + out_duration = _probe_duration(output) + assert abs(out_duration - 5.0) < 0.5, f"输出时长 {out_duration} 未对齐驱动音频" + + # 结果音轨主频应接近 800Hz(驱动音频),远离 200Hz(源视频音轨) + freq = _estimate_audio_freq(output, out_duration) + assert abs(freq - 800) < abs(freq - 200), f"结果音轨主频 {freq:.0f}Hz 不是驱动音频" + assert freq > 450, f"结果音轨主频 {freq:.0f}Hz 疑似源视频音轨(200Hz)" + + +@pytest.mark.skipif(not HAS_FFMPEG, reason="环境无 ffmpeg/ffprobe") +def test_real_mux_non_loop_keeps_video_copy_path(server, tmp_path): + """驱动音频(1s)短于视频(2s):输出约 1s,音轨仍是驱动音频.""" + source_video, _ = _make_media(tmp_path) + short_audio = tmp_path / "short.wav" + subprocess.run( + [ + "ffmpeg", + "-y", + "-f", + "lavfi", + "-i", + "sine=frequency=800:duration=1", + str(short_audio), + ], + stdout=subprocess.DEVNULL, + stderr=subprocess.DEVNULL, + check=True, + ) + output = tmp_path / "output_short.mp4" + server._mux_video_with_audio(source_video, short_audio, output) + out_duration = _probe_duration(output) + assert abs(out_duration - 1.0) < 0.4 + freq = _estimate_audio_freq(output, out_duration) + assert freq > 450, f"结果音轨主频 {freq:.0f}Hz 疑似源视频音轨"