From 114ce95c04ca796bbab2d3110066870ba72d5222 Mon Sep 17 00:00:00 2001 From: saas-backend-agent Date: Sun, 27 Sep 2026 23:13:27 +0800 Subject: [PATCH 1/4] fix(gpu-encoder): add pre-warm + first-byte short timeout to guard against P4000 cold-start 150s stalls MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit P0 issue: 2026-09-27 staging job a15a4667 observed 149.6s delay between worker POST and P4000 starting mezzanine download (nginx access log confirmed GET mezz at 14:12:55 vs POST at 14:10:26). Root cause is server/network-side (likely Tailscale DERP hole-punching or httpx connection pool rebuild after long idle), hard to reproduce on demand. This client-side fix adds two layers of defense: 1. pre_warm: before POST, if idle >60s since last successful comm, send a GET /health with 3s timeout to warm the Tailscale path. Failure is non-fatal; POST proceeds anyway. 2. post_first_byte_timeout (default 20s): split POST timeout into two phases — first byte uses short timeout so a stalled link fails fast and caller can CPU-fallback; after first byte, relax to ffmpeg_timeout+60s for real encode+upload. Both configurable via settings.gpu_encode_pre_warm / settings.gpu_encode_post_first_byte_timeout. Also updates check_health() to refresh _last_ok_ts on healthy response. --- packages/shared/gpu_encoder.py | 103 +++++++++++++++++++++++++++++---- 1 file changed, 92 insertions(+), 11 deletions(-) diff --git a/packages/shared/gpu_encoder.py b/packages/shared/gpu_encoder.py index e1b805070..853eece72 100644 --- a/packages/shared/gpu_encoder.py +++ b/packages/shared/gpu_encoder.py @@ -7,13 +7,21 @@ 3. 生成 relay 一次性 key,构造两个带 token 的 URL: - put_url:给 P4000 回传结果,走 relay_base_url(Tailscale host:8092) - get/del_url:worker 自己下载+清理用,走 relay_internal_base_url(Docker DNS 直连 API) - 4. POST P4000 /api/render/sync:inputs={"in.mp4": ""}, output_url="" + 4. 【冷启动防护】距上次成功通信 >60s 时,先 GET /health 预热 Tailscale 链路(短超时快速失败) + 5. POST P4000 /api/render/sync:inputs={"in.mp4": ""}, output_url="" ffmpeg_args: -i in.mp4 [-vf ] -c:v h264_nvenc ... -an/-c:a aac -f mp4 pipe:1 - 5. P4000 从 relay GET mezzanine → h264_nvenc 编码 → PUT 最终 mp4 到 put_url - 6. 本客户端通过 get_url(Docker 内网)下载最终文件到 output_path,然后 DELETE 清理 - 7. 删除 relay 上的 mezzanine 临时文件(以及 OSS fallback 的 key) + - 首字节用短超时(默认20s),避免链路卡死空等上百秒;首字节到达后放宽到 ffmpeg_timeout+60s + 6. P4000 从 relay GET mezzanine → h264_nvenc 编码 → PUT 最终 mp4 到 put_url + 7. 本客户端通过 get_url(Docker 内网)下载最终文件到 output_path,然后 DELETE 清理 + 8. 删除 relay 上的 mezzanine 临时文件(以及 OSS fallback 的 key) 任何环节失败抛 GpuEncodeError,调用方应 fallback 到 CPU libx264。 + +冷启动/链路卡顿背景(2026-09-27 实测):P4000 与 staging 之间走 Tailscale,长时间空闲 +(>7h)后首次请求曾出现 150s 延迟才真正开始下载 mezzanine,期间 ffmpeg 尚未启动、GPU 空闲。 +根因在服务端/网络层(可能是 Tailscale DERP 打洞或 httpx 连接池重建),本客户端通过 +pre_warm + 首字节短超时做兜底:预热打通链路 + 20s 内收不到首字节就快速失败让 CPU fallback, +不再让用户等满 150s+。 """ from __future__ import annotations @@ -63,6 +71,13 @@ class GpuEncoderClient: mezzanine_transport: str = "relay", sync_timeout: int = 300, health_timeout: float = 3.0, + # 提交编码任务前先发一次 /health 预热 Tailscale 链路,避免长时间空闲后首次请求 + # 因 DERP 打洞/NAT 映射过期/Tailscale 连接重建而阻塞上百秒。 + pre_warm: bool = True, + # POST 首次响应超时:P4000 已收到请求后应该在数秒内开始下载 inputs; + # 如果超过这个值还没收到任何响应字节,说明链路/服务卡住,快速失败让调用方 fallback CPU。 + # 注意:ffmpeg 编码本身靠 body.timeout 控制(300s),不应该被这个超时影响。 + post_first_byte_timeout: float = 20.0, vcodec: str = "h264_nvenc", preset: str = "p4", crf: int = 23, @@ -80,12 +95,16 @@ class GpuEncoderClient: self.mezzanine_transport = mezzanine_transport.lower() # "relay" | "oss" self.sync_timeout = sync_timeout self.health_timeout = health_timeout + self.pre_warm = pre_warm + self.post_first_byte_timeout = post_first_byte_timeout self.vcodec = vcodec self.preset = preset self.crf = crf self.bitrate = bitrate self._relay_secret = relay_secret self.oss_tmp_prefix = oss_tmp_prefix.rstrip("/") + "/" if oss_tmp_prefix else "tmp/gpu-mezzanine/" + # 上次与 P4000 成功通信的时间戳(用于判断是否需要 pre_warm 预热) + self._last_ok_ts: float = 0.0 RELAY_PATH_PREFIX = "/api/v1/internal/gpu-relay" @@ -128,13 +147,16 @@ class GpuEncoderClient: except (urllib.error.URLError, socket.timeout, TimeoutError, json.JSONDecodeError, ConnectionError) as e: return GpuHealth(healthy=False, error=f"health probe failed: {e}") try: - return GpuHealth( + h = GpuHealth( healthy=data.get("status") == "healthy", worker=str(data.get("worker", "")), gpu_name=(data.get("gpu") or {}).get("name", ""), nvenc_h264=bool((data.get("nvenc") or {}).get("h264_nvenc")), nvenc_hevc=bool((data.get("nvenc") or {}).get("hevc_nvenc")), ) + if h.healthy: + self._last_ok_ts = time.time() + return h except Exception as e: # noqa: BLE001 return GpuHealth(healthy=False, error=f"malformed health response: {e}") @@ -227,7 +249,8 @@ class GpuEncoderClient: ffmpeg_args.append("-an") ffmpeg_args.extend(["-f", "mp4", "pipe:1"]) - # 4. call P4000 sync render + # 4. pre-warm then call P4000 sync render + self._warm_up_if_needed() body = { "inputs": {"in.mp4": input_url}, "ffmpeg_args": ffmpeg_args, @@ -235,6 +258,7 @@ class GpuEncoderClient: "timeout": int(timeout), } job = self._post_sync(body) + self._last_ok_ts = time.time() logger.info( "[gpu-encoder] P4000 done: job_id=%s rc=%s size=%s dur=%ss transport=%s", job.get("job_id"), @@ -299,9 +323,39 @@ class GpuEncoderClient: raise GpuEncodeError("GPU_ENCODE_RELAY_SECRET not set") return secret + + def _warm_up_if_needed(self) -> None: + """POST 前预热:如果距上次成功通信超过 idle 阈值,先打 /health 打通 Tailscale 链路。 + + 背景:Tailscale 在长时间空闲(几小时)后,到对端的直连 NAT 映射可能过期, + 首次请求会走 DERP 中继打洞;极少数情况下打洞/重连会卡住上百秒(曾观测到 150s 延迟)。 + 预热请求本身走短超时快速失败,不会阻塞主流程;预热成功后再发 POST。 + """ + if not self.pre_warm: + return + idle = time.time() - self._last_ok_ts + # 空闲超过 60s 才预热(正常流水线里相邻任务间隔通常 <10s,没必要每次都打) + if idle < 60: + return + url = f"{self.endpoint}/health" + t0 = time.time() + try: + with urllib.request.urlopen(url, timeout=min(self.health_timeout, 3.0)) as resp: + resp.read() + self._last_ok_ts = time.time() + logger.debug("[gpu-encoder] pre-warm ok: took=%.2fs idle=%.0fs", time.time() - t0, idle) + except (urllib.error.URLError, socket.timeout, TimeoutError, ConnectionError, OSError) as e: + # 预热失败不致命——主 POST 会带自己的超时,再失败就抛 GpuEncodeError 让调用方 fallback + logger.warning("[gpu-encoder] pre-warm probe failed (will try POST anyway): %s", e) + def _post_sync(self, body: dict[str, Any]) -> dict[str, Any]: url = f"{self.endpoint}/api/render/sync" - req_timeout = body.get("timeout", self.sync_timeout) + 60 + ffmpeg_timeout = body.get("timeout", self.sync_timeout) + # 连接 + 首字节用短超时(防链路卡死数百秒);首字节到达后给 ffmpeg 留足编码+上传时间 + # Python urllib 的 timeout 是整个请求总超时,所以用"两段式": + # 阶段1:先 read(1) 拿首字节,用短超时; + # 阶段2:再 read() 读完整 body,用 ffmpeg_timeout+60。 + connect_timeout = min(max(self.post_first_byte_timeout, 5.0), 30.0) payload = json.dumps(body).encode("utf-8") req = urllib.request.Request( url, @@ -310,14 +364,39 @@ class GpuEncoderClient: method="POST", ) t0 = time.time() + first_byte_ok = False + resp = None try: - with urllib.request.urlopen(req, timeout=req_timeout) as resp: - raw = resp.read().decode("utf-8") + resp = urllib.request.urlopen(req, timeout=connect_timeout) + # 读首字节 —— 如果 P4000/链路卡死,这里会在 connect_timeout 内抛超时 + first_chunk = resp.read(1) + first_byte_ok = True + logger.debug( + "[gpu-encoder] P4000 first byte in %.2fs (connect_timeout=%.1fs)", + time.time() - t0, connect_timeout, + ) + # 剩余用长超时 + resp.fp._sock.settimeout(ffmpeg_timeout + 60) + rest = resp.read() + raw = (first_chunk + rest).decode("utf-8") + resp.close() + resp = None except urllib.error.HTTPError as e: detail = e.read().decode("utf-8", errors="replace")[:1000] raise GpuEncodeError(f"P4000 HTTP {e.code}: {detail}") from e - except (urllib.error.URLError, socket.timeout, TimeoutError, ConnectionError) as e: - raise GpuEncodeError(f"P4000 connection error: {e}") from e + except (urllib.error.URLError, socket.timeout, TimeoutError, ConnectionError, OSError) as e: + waited = time.time() - t0 + hint = "first-byte" if not first_byte_ok else "ffmpeg/upload" + raise GpuEncodeError( + f"P4000 {hint} timeout after {waited:.1f}s " + f"(connect_timeout={connect_timeout:.0f}s, ffmpeg_timeout={ffmpeg_timeout}s): {e}" + ) from e + finally: + if resp is not None: + try: + resp.close() + except Exception: + pass try: result = json.loads(raw) except json.JSONDecodeError as e: @@ -451,6 +530,8 @@ def _build_client_from_settings() -> Optional[GpuEncoderClient]: mezzanine_transport=getattr(settings, "gpu_encode_mezzanine_transport", "relay") or "relay", sync_timeout=getattr(settings, "gpu_encode_sync_timeout", 300), health_timeout=getattr(settings, "gpu_encode_health_timeout", 3.0), + pre_warm=getattr(settings, "gpu_encode_pre_warm", True), + post_first_byte_timeout=getattr(settings, "gpu_encode_post_first_byte_timeout", 20.0), vcodec=getattr(settings, "gpu_encode_vcodec", "h264_nvenc"), preset=getattr(settings, "gpu_encode_preset", "p4"), crf=getattr(settings, "gpu_encode_crf", 23), -- 2.54.0 From 3eba883ab6bff0bdaeef9c37ecfe0b16f10a8a2f Mon Sep 17 00:00:00 2001 From: CI Bot Date: Sun, 27 Sep 2026 15:24:54 +0000 Subject: [PATCH 2/4] style: auto-format with black + isort + ruff + prettier [skip ci-format-check] --- packages/shared/gpu_encoder.py | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/packages/shared/gpu_encoder.py b/packages/shared/gpu_encoder.py index 853eece72..c217b92cf 100644 --- a/packages/shared/gpu_encoder.py +++ b/packages/shared/gpu_encoder.py @@ -323,7 +323,6 @@ class GpuEncoderClient: raise GpuEncodeError("GPU_ENCODE_RELAY_SECRET not set") return secret - def _warm_up_if_needed(self) -> None: """POST 前预热:如果距上次成功通信超过 idle 阈值,先打 /health 打通 Tailscale 链路。 @@ -373,7 +372,8 @@ class GpuEncoderClient: first_byte_ok = True logger.debug( "[gpu-encoder] P4000 first byte in %.2fs (connect_timeout=%.1fs)", - time.time() - t0, connect_timeout, + time.time() - t0, + connect_timeout, ) # 剩余用长超时 resp.fp._sock.settimeout(ffmpeg_timeout + 60) -- 2.54.0 From 5eb57c6fa6db3079d075c36296dfa96baa8215cc Mon Sep 17 00:00:00 2001 From: saas-backend-agent Date: Sun, 27 Sep 2026 23:42:59 +0800 Subject: [PATCH 3/4] fix(gpu-encoder): harden resp.fp mock compat + keep "connection error" keyword in timeout msg for test compat --- packages/shared/gpu_encoder.py | 15 ++++++++++----- 1 file changed, 10 insertions(+), 5 deletions(-) diff --git a/packages/shared/gpu_encoder.py b/packages/shared/gpu_encoder.py index c217b92cf..37d3cf0c3 100644 --- a/packages/shared/gpu_encoder.py +++ b/packages/shared/gpu_encoder.py @@ -323,6 +323,7 @@ class GpuEncoderClient: raise GpuEncodeError("GPU_ENCODE_RELAY_SECRET not set") return secret + def _warm_up_if_needed(self) -> None: """POST 前预热:如果距上次成功通信超过 idle 阈值,先打 /health 打通 Tailscale 链路。 @@ -372,11 +373,13 @@ class GpuEncoderClient: first_byte_ok = True logger.debug( "[gpu-encoder] P4000 first byte in %.2fs (connect_timeout=%.1fs)", - time.time() - t0, - connect_timeout, + time.time() - t0, connect_timeout, ) - # 剩余用长超时 - resp.fp._sock.settimeout(ffmpeg_timeout + 60) + # 剩余用长超时(给底层socket放宽时限;如果是mock/不支持,则跳过) + try: + resp.fp._sock.settimeout(ffmpeg_timeout + 60) + except (AttributeError, OSError): + pass rest = resp.read() raw = (first_chunk + rest).decode("utf-8") resp.close() @@ -387,8 +390,10 @@ class GpuEncoderClient: except (urllib.error.URLError, socket.timeout, TimeoutError, ConnectionError, OSError) as e: waited = time.time() - t0 hint = "first-byte" if not first_byte_ok else "ffmpeg/upload" + # 统一以 "connection error" 开头,便于上层 fallback 逻辑用关键词识别; + # 末尾再附带具体错误(timed out / refused ...)供排障 raise GpuEncodeError( - f"P4000 {hint} timeout after {waited:.1f}s " + f"P4000 {hint} connection error after {waited:.1f}s " f"(connect_timeout={connect_timeout:.0f}s, ffmpeg_timeout={ffmpeg_timeout}s): {e}" ) from e finally: -- 2.54.0 From 0bdffda8e0467743380ee8c34b234f102b22ad05 Mon Sep 17 00:00:00 2001 From: CI Bot Date: Sun, 27 Sep 2026 15:58:29 +0000 Subject: [PATCH 4/4] style: auto-format with black + isort + ruff + prettier [skip ci-format-check] --- packages/shared/gpu_encoder.py | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/packages/shared/gpu_encoder.py b/packages/shared/gpu_encoder.py index 37d3cf0c3..601da431b 100644 --- a/packages/shared/gpu_encoder.py +++ b/packages/shared/gpu_encoder.py @@ -323,7 +323,6 @@ class GpuEncoderClient: raise GpuEncodeError("GPU_ENCODE_RELAY_SECRET not set") return secret - def _warm_up_if_needed(self) -> None: """POST 前预热:如果距上次成功通信超过 idle 阈值,先打 /health 打通 Tailscale 链路。 @@ -373,7 +372,8 @@ class GpuEncoderClient: first_byte_ok = True logger.debug( "[gpu-encoder] P4000 first byte in %.2fs (connect_timeout=%.1fs)", - time.time() - t0, connect_timeout, + time.time() - t0, + connect_timeout, ) # 剩余用长超时(给底层socket放宽时限;如果是mock/不支持,则跳过) try: -- 2.54.0