feat(gpu): 接入 P4000 NVENC 硬件编码加速

- packages/config/base.py: 新增 GPU_ENCODE_* 配置项(开关/endpoint/relay/secret/编码参数/fallback)
- packages/shared/gpu_encoder.py: 新增 GpuEncoderClient,封装 health 探测 + mezzanine 上传 OSS + P4000 nvenc 编码 + relay 回传下载,失败抛 GpuEncodeError 触发 CPU 降级
- apps/api/app/api/routes/gpu_relay.py: 新增 internal PUT/GET/DELETE /api/v1/internal/gpu-relay/{key}(token 鉴权),P4000 PUT 编码结果,worker GET 下载
- apps/api/app/api/router.py: 注册 gpu_relay_router
- apps/worker/video_processing/unified_render_service.py: _execute_ffmpeg 和 _render_pass_through 尝试 GPU 路径:CPU ultrafast mezzanine → P4000 nvenc → 输出到最终路径;任何失败自动回退到原 CPU libx264 路径
- apps/worker/worker_app/tasks/_startup.py: worker_ready 时探测 P4000 健康并打日志
- tests/unit/test_gpu_encoder.py: GpuEncoderClient 单测(health/sync 调用/失败/fallback/relay URL)

架构:
- 输入:CPU 输出 libx264 ultrafast mezzanine → 上传 OSS 临时前缀 → P4000 签名 URL 下载
- 输出:P4000 PUT → 宿主机 nginx(tailscale:80)→ API /api/ 反代 → gpu_relay 路由落盘到 generated/gpu_relay/
- 回传:worker 通过 docker 网络 http://xiaoxia-api-staging:8000 GET 下载最终 mp4 到 output_path
- 降级:GPU 任何环节异常(health/上传/编码/回传/下载)→ 原 CPU 路径继续执行,不影响成片
This commit is contained in:
saas-backend-agent
2026-09-26 18:39:42 +08:00
committed by saas-backend
parent a5075624f8
commit 105ab54059
7 changed files with 952 additions and 28 deletions
@@ -57,6 +57,7 @@ from video_processing.tts_engine import TtsEngine
from video_processing.watermark_engine import WatermarkConfig, WatermarkEngine
from packages.domain.render_layer_utils import LAYER_Z_INDEX as _IMPORTED_LAYER_Z_INDEX
from packages.shared.gpu_encoder import GpuEncodeError, get_gpu_encoder
from packages.domain.render_layer_utils import clip_adjusted_duration as _clip_adjusted_duration_pure
from packages.domain.render_layer_utils import clip_effective_duration as _clip_effective_duration_pure
from packages.domain.render_layer_utils import clip_playback_speed as _clip_playback_speed_pure
@@ -1621,20 +1622,27 @@ class UnifiedRenderService:
effective_duration,
has_audio,
)
try:
run_ffmpeg(command)
except subprocess.CalledProcessError as e:
stderr_text = (e.stderr or "").strip()
stderr_tail = stderr_text[-1500:] if len(stderr_text) > 1500 else stderr_text
logger.error(
"直通渲染失败: plan_id=%s clip=%s exit_code=%d\nvf=%s\nstderr(last 1500):\n%s",
self.plan.id,
clip.clip_id,
e.returncode,
vf_str[:2000],
stderr_tail,
)
raise
# 尝试 GPU NVENC 加速
gpu_ok = False
if self._gpu_encode_available():
mezz_path = output_path.parent / f".{output_path.stem}.mezz{output_path.suffix}"
gpu_ok = self._ffmpeg_output_to_mezzanine(command, mezz_path, output_path)
if not gpu_ok:
try:
run_ffmpeg(command)
except subprocess.CalledProcessError as e:
stderr_text = (e.stderr or "").strip()
stderr_tail = stderr_text[-1500:] if len(stderr_text) > 1500 else stderr_text
logger.error(
"直通渲染失败: plan_id=%s clip=%s exit_code=%d\nvf=%s\nstderr(last 1500):\n%s",
self.plan.id,
clip.clip_id,
e.returncode,
vf_str[:2000],
stderr_tail,
)
raise
return has_audio
@@ -2170,6 +2178,120 @@ class UnifiedRenderService:
filter_complex = ";".join(filter_parts)
return filter_complex, input_args
# ── GPU NVENC 加速 ────────────────────────────────────────────────────
def _gpu_encode_available(self) -> bool:
"""GPU 编码客户端是否已配置且健康(缓存健康状态,单任务内只探测一次)。"""
if not getattr(self, "_gpu_health_ok", None):
client = get_gpu_encoder()
if client is None:
self._gpu_health_ok = False
return False
try:
health = client.check_health()
if health.ready:
logger.info(
"[gpu-encoder] healthy endpoint=%s gpu=%s",
client.endpoint, health.gpu_name,
)
self._gpu_health_ok = True
else:
logger.warning(
"[gpu-encoder] not ready: %s (endpoint=%s)",
health.error, client.endpoint,
)
self._gpu_health_ok = False
except Exception as e: # noqa: BLE001
logger.warning("[gpu-encoder] health probe error (CPU fallback): %s", e)
self._gpu_health_ok = False
return self._gpu_health_ok
def _ffmpeg_output_to_mezzanine(
self,
base_command: list[str],
mezzanine_path: Path,
output_path: Path,
) -> bool:
"""用 CPU ultrafast 把滤镜链输出到 mezzanine_path,然后调 GPU 做最终编码。
base_command: 原本要执行的完整 ffmpeg 命令(含 -c:v libx264 -crf X -preset Y ... output_path)
我们把最后一个参数(output_path)替换成 mezzanine_path,并把编码参数改成 ultrafast,
成功后调用 gpu_encoder 做 nvenc 编码到 output_path。
任何失败返回 False,调用方走原始 CPU 路径。
"""
client = get_gpu_encoder()
if client is None:
return False
# 构造 mezzanine 命令:替换编码参数和输出路径
mezz_cmd = list(base_command)
# 找到编码参数位置并替换
try:
i_crf = mezz_cmd.index("-crf")
mezz_cmd[i_crf + 1] = "20"
i_preset = mezz_cmd.index("-preset")
mezz_cmd[i_preset + 1] = "ultrafast"
except ValueError:
logger.warning("[gpu-encoder] could not find -crf/-preset in command, skip gpu")
return False
# 如果命令有音频编码 -c:a aac,我们保留音频让 GPU 侧不用单独处理
# (P4000 的 ffmpeg_args 可以直接 copy 音频?这里简单起见:把音频编码留在 mezzanine,
# 然后 GPU 侧直接 -c:a copy,避免重编码损失)
has_audio = "-c:a" in mezz_cmd
# 替换输出路径(最后一个参数)
mezz_cmd[-1] = str(mezzanine_path)
# 1) 跑 mezzanine
mezzanine_path.parent.mkdir(parents=True, exist_ok=True)
t0 = time.time()
try:
run_ffmpeg(mezz_cmd)
except subprocess.CalledProcessError as e:
logger.warning("[gpu-encoder] mezzanine encode failed (CPU fallback): %s", e)
return False
logger.info(
"[gpu-encoder] mezzanine ready: %s (%.1fs, %d bytes), dispatching to P4000 nvenc...",
mezzanine_path.name, time.time() - t0,
mezzanine_path.stat().st_size if mezzanine_path.exists() else 0,
)
# 2) GPU nvenc encode(含上传 mezzanine → OSS → P4000 下载+编码 → relay 回传)
try:
# GPU 侧:-i in.mp4 -c:v h264_nvenc ... 音频 copy(mezzanine 里音频已是 aac)
audio_args = ["-c:a", "copy"] if has_audio else None
client.encode_mezzanine_to_output(
mezzanine_path,
output_path,
audio_args=audio_args,
)
logger.info(
"[gpu-encoder] GPU nvenc encode done: %s (total %.1fs)",
output_path.name, time.time() - t0,
)
return True
except GpuEncodeError as e:
logger.warning("[gpu-encoder] GPU encode failed (CPU fallback): %s", e)
# 删除可能残留的不完整 output
try:
if output_path.exists():
output_path.unlink()
except OSError:
pass
return False
except Exception as e: # noqa: BLE001
logger.warning("[gpu-encoder] GPU encode unexpected error (CPU fallback): %s", e)
return False
finally:
# 清理 mezzanine
try:
if mezzanine_path.exists():
mezzanine_path.unlink()
except OSError:
pass
def _execute_ffmpeg(
self,
filter_complex: str,
@@ -2209,20 +2331,28 @@ class UnifiedRenderService:
input_args.count("-i"),
output_path,
)
try:
run_ffmpeg(command)
except subprocess.CalledProcessError as e:
# 额外记录 filter_complex + stderr,方便排查滤镜链构建问题
stderr_text = (e.stderr or "").strip()
stderr_tail = stderr_text[-1500:] if len(stderr_text) > 1500 else stderr_text
logger.error(
"渲染失败: plan_id=%s exit_code=%d\nfilter_complex:\n%s\nstderr(last 1500):\n%s",
self.plan.id,
e.returncode,
filter_complex[:5000],
stderr_tail,
)
raise
# 尝试 GPU NVENC 加速:先出 ultrafast mezzanine,再交给 P4000 做最终编码
gpu_ok = False
if self._gpu_encode_available():
mezz_path = output_path.parent / f".{output_path.stem}.mezz{output_path.suffix}"
gpu_ok = self._ffmpeg_output_to_mezzanine(command, mezz_path, output_path)
if not gpu_ok:
try:
run_ffmpeg(command)
except subprocess.CalledProcessError as e:
# 额外记录 filter_complex + stderr,方便排查滤镜链构建问题
stderr_text = (e.stderr or "").strip()
stderr_tail = stderr_text[-1500:] if len(stderr_text) > 1500 else stderr_text
logger.error(
"渲染失败: plan_id=%s exit_code=%d\nfilter_complex:\n%s\nstderr(last 1500):\n%s",
self.plan.id,
e.returncode,
filter_complex[:5000],
stderr_tail,
)
raise
def _build_sticker_filters(self, input_label: str, output_label: str) -> tuple[str, list[str]]:
"""构建贴纸叠加滤镜链.