fix(douyin): 修复抖音真实链接503,增加第三方解析API兜底
CI/CD Pipeline / Check if frontend-only change (push) Has been skipped
CI/CD Pipeline / Dedup Check - skip PR tests when covered by push pipeline (push) Successful in 1s
CI/CD Pipeline / Check push changed paths (push) Successful in 2s
CI/CD Pipeline / Frontend Lint (push) Has been skipped
CI/CD Pipeline / PR Build API Image (push) Has been skipped
CI/CD Pipeline / PR Build Web Image (push) Has been skipped
CI/CD Pipeline / PR Build Worker Image (push) Has been skipped
CI/CD Pipeline / Build Staging Web Image (push) Has been skipped
CI/CD Pipeline / Build Staging Worker Image (push) Successful in 36s
CI/CD Pipeline / Build Staging API Image (push) Successful in 41s
CI/CD Pipeline / Retag skipped Staging API Image (push) Has been skipped
CI/CD Pipeline / Retag skipped Staging Worker Image (push) Has been skipped
CI/CD Pipeline / Retag skipped Staging Web Image (push) Successful in 18s
CI/CD Pipeline / Deploy Staging (Watchtower auto-deploy) (push) Successful in 49s
CI/CD Pipeline / Frontend Unit Tests (push) Successful in 2m45s
CI/CD Pipeline / ACR Image Cleanup (push) Successful in 1m35s
CI/CD Pipeline / Integration Tests (push) Successful in 3m56s
CI/CD Pipeline / Validate - Python (mypy + alembic) (push) Successful in 4m24s
CI/CD Pipeline / Build Production API Image (push) Has been cancelled
CI/CD Pipeline / Build Production Web Image (push) Has been cancelled
CI/CD Pipeline / Build Production Worker Image (push) Has been cancelled
CI/CD Pipeline / Deploy Production (push) Has been cancelled
CI/CD Pipeline / Production Browser E2E (push) Has been cancelled
CI/CD Pipeline / Canary Release to Production (push) Has been cancelled
CI/CD Pipeline / CI Gate (push) Has been cancelled
CI/CD Pipeline / Validate - Security (push) Has been cancelled
CI/CD Pipeline / Unit Tests (push) Has been cancelled
CI/CD Pipeline / Validate - Style (push) Has been cancelled
CI/CD Pipeline / Staging E2E Tests (push) Has been cancelled
CI/CD Pipeline / Staging API Integration Tests (push) Has been cancelled

根因:
- yt-dlp DouyinIE 当前需要 fresh cookies(即使只提取元信息),
  staging 上的 douyin_cookies.txt 是占位文件(3行),导致直链拿不到
- 原代码路径A(MediaKit ASR)因 direct_url=None 完全跳过
- 路径B(下载+本地ASR)同样因 cookies 失败返回友好 503

修复:
1. 修复 MediaKitClient monkey-patch 属性名错误
   (原用 self.base_url/self.api_key/self.timeout,实际是 self._base_url 等私有属性;
    错误添加了 _mk_headers 与类自带 _headers() 重复)
2. yt-dlp 增加 extractor_args 和 Accept-Language header
3. 新增第三方无水印解析 API 兜底(apizero.cn),
   yt-dlp 失败时自动切换,拿到直链后仍走 MediaKit ASR
4. 路径B 支持直接用第三方直链下载 MP4(不依赖 yt-dlp/cookies)
5. MediaKit ASR 轮询增加瞬时网络错误重试
6. 全部路径失败时返回友好 503(不再返回空 text 导致前端异常)

验证:本地通过 apizero 成功解析 https://v.douyin.com/hb-giW8cC1Q/
返回直链可下载(23MB mp4),HTTP 200
This commit is contained in:
saas-backend-agent
2026-09-17 11:21:35 +08:00
parent b2ba78c16c
commit 132ca6bb70
+180 -14
View File
@@ -179,6 +179,10 @@ def _extract_and_validate_douyin_url(raw_input):
# ── MediaKitClient ASR 扩展(monkey patch) ────────────────────────────
#
# app/services/mediakit_client.py 已提供:
# self._api_key / self._base_url / self._timeout / self._headers() / self.is_available
# 这里只补两个通用 JSON helper 和 ASR 提交/轮询方法。
def _mk_post_json(self, path, payload):
@@ -200,9 +204,13 @@ def _mk_post_json(self, path, payload):
) from exc
except httpx.RequestError as exc:
raise MediaKitError("MediaKit API 网络错误: %s" % exc, code="NetworkError") from exc
if not data.get("success", True) and data.get("error"):
err = data["error"]
raise MediaKitError(err.get("message", "请求失败"), code=err.get("code", "RequestFailed"))
# 同步错误(提交参数错误等):success=false 且含 error
if data.get("success") is False and data.get("error"):
err = data["error"] if isinstance(data["error"], dict) else {"message": str(data["error"])}
raise MediaKitError(
err.get("message", "请求失败"),
code=err.get("code", "RequestFailed"),
)
return data
@@ -228,20 +236,29 @@ def _mk_get_json(self, path):
def _mediakit_asr_submit(self, video_url):
"""提交语音转字幕任务(POST /tools/asr-subtitles)。返回 task_id。"""
data = self._post_json(
"/tools/asr-subtitles",
{"video_url": video_url, "language": "cmn-Hans-CN"},
)
task_id = data.get("task_id")
if not task_id:
raise MediaKitError("MediaKit ASR 提交响应缺少 task_id")
raise MediaKitError("MediaKit ASR 提交响应缺少 task_id: %s" % str(data)[:200])
return task_id
def _mediakit_asr_poll(self, task_id, poll_interval=2.0, max_attempts=90):
for _ in range(max_attempts):
"""轮询 ASR 任务直到 completed/failed。返回 (text, duration)。"""
for attempt in range(max_attempts):
time.sleep(poll_interval)
data = self._get_json("/tasks/" + task_id)
try:
data = self._get_json("/tasks/" + task_id)
except MediaKitError as exc:
# 瞬时网络/超时可重试
if attempt < max_attempts - 1 and getattr(exc, "code", "") in ("Timeout", "NetworkError"):
logger.warning("MediaKit ASR 轮询异常(第%d次),将重试: %s", attempt + 1, exc)
continue
raise
st = data.get("status")
if st in ("completed", "success"):
result = data.get("result") or {}
@@ -250,17 +267,23 @@ def _mediakit_asr_poll(self, task_id, poll_interval=2.0, max_attempts=90):
duration = float(result.get("duration") or 0.0)
return text.strip(), duration
if st == "failed":
err = data.get("error") or {}
raise MediaKitError(
"MediaKit ASR 任务失败: %s" % err.get("message", "unknown"),
code=err.get("code", "TaskFailed"),
)
err = data.get("error")
if isinstance(err, dict):
msg = err.get("message") or "unknown"
code = err.get("code") or "TaskFailed"
elif isinstance(err, str):
msg, code = err, "TaskFailed"
else:
msg, code = "unknown", "TaskFailed"
raise MediaKitError("MediaKit ASR 任务失败: %s" % msg, code=code)
# running/pending/queued: continue
raise MediaKitError(
"MediaKit ASR 超时(%ss 未完成)" % int(poll_interval * max_attempts),
code="Timeout",
)
# 绑定到类(零侵入,不修改原 mediakit_client.py)
if not hasattr(MediaKitClient, "_post_json"):
MediaKitClient._post_json = _mk_post_json
if not hasattr(MediaKitClient, "_get_json"):
@@ -285,6 +308,7 @@ def _ytdlp_extract_video_url(page_url, cookiefile=None):
"no_warnings": True,
"noplaylist": True,
"skip_download": True,
"extractor_args": {"douyin": {"webpage_cookie": ""}}, # 强制使用页面 cookie 流程
"http_headers": {
"User-Agent": (
"Mozilla/5.0 (Windows NT 10.0; Win64; x64) "
@@ -292,6 +316,7 @@ def _ytdlp_extract_video_url(page_url, cookiefile=None):
"Chrome/128.0.0.0 Safari/537.36"
),
"Referer": "https://www.douyin.com/",
"Accept-Language": "zh-CN,zh;q=0.9,en;q=0.8",
},
}
if cookiefile:
@@ -318,6 +343,125 @@ def _ytdlp_extract_video_url(page_url, cookiefile=None):
return video_url, duration
_PARSER_APIS = [
# 第三方抖音无水印解析 API(cookies 过期/yt-dlp 反爬升级时的兜底)
("https://v1.apizero.cn/api/video-parse?url={url}&flat=1", "apizero"),
]
def _parser_extract_direct_url(page_url, timeout=10):
"""使用第三方解析 API 获取抖音无水印直链(yt-dlp 失败时的兜底)。
返回 (direct_url, duration) 或 (None, 0.0)。
"""
import urllib.parse
import httpx
for api_tpl, name in _PARSER_APIS:
url = api_tpl.format(url=urllib.parse.quote(page_url, safe=""))
try:
with httpx.Client(timeout=timeout, follow_redirects=True) as http:
resp = http.get(
url,
headers={
"User-Agent": (
"Mozilla/5.0 (Windows NT 10.0; Win64; x64) "
"AppleWebKit/537.36 (KHTML, like Gecko) "
"Chrome/128.0.0.0 Safari/537.36"
),
"Accept": "application/json",
},
)
resp.raise_for_status()
data = resp.json()
except Exception as exc: # noqa: BLE001
logger.warning("第三方解析 %s 请求失败: %s", name, exc)
continue
try:
if data.get("code") != 0:
logger.warning("第三方解析 %s 返回错误: %s", name, data.get("msg"))
continue
vd = data.get("data") or {}
vlist = vd.get("video_list") or []
if not vlist:
continue
chosen = None
for v in vlist:
if not v.get("url"):
continue
if "原画" in str(v.get("quality", "")):
chosen = v
break
if chosen is None:
chosen = vlist[0]
direct_url = chosen.get("url")
duration = float(vd.get("duration") or 0.0)
if direct_url:
logger.info("第三方解析 %s 成功,直链长度=%d", name, len(direct_url))
return direct_url, duration
except Exception as exc: # noqa: BLE001
logger.warning("第三方解析 %s 响应解析失败: %s", name, exc)
continue
logger.warning("所有第三方解析均失败")
return None, 0.0
def _direct_url_download_and_local_asr(direct_url, page_url, temp_dir):
"""通过第三方解析得到的直链直接下载 MP4,再做本地 ASR。
返回 (text, duration)。
"""
import httpx
import os
video_path = os.path.join(temp_dir, "video.mp4")
try:
with httpx.Client(timeout=60, follow_redirects=True) as http:
with http.stream(
"GET",
direct_url,
headers={
"User-Agent": (
"Mozilla/5.0 (Windows NT 10.0; Win64; x64) "
"AppleWebKit/537.36 (KHTML, like Gecko) "
"Chrome/128.0.0.0 Safari/537.36"
),
"Referer": "https://www.douyin.com/",
},
) as resp:
resp.raise_for_status()
downloaded = 0
with open(video_path, "wb") as f:
for chunk in resp.iter_bytes(chunk_size=65536):
if chunk:
f.write(chunk)
downloaded += len(chunk)
if downloaded == 0:
raise HTTPException(status_code=status.HTTP_502_BAD_GATEWAY, detail="直链下载为空")
except HTTPException:
raise
except httpx.TimeoutException:
logger.warning("直链下载超时: %s", page_url)
raise HTTPException(status_code=status.HTTP_504_GATEWAY_TIMEOUT, detail="视频下载超时,请稍后重试")
except Exception as exc: # noqa: BLE001
logger.exception("直链下载失败: url=%s err=%s", page_url, exc)
raise HTTPException(status_code=status.HTTP_502_BAD_GATEWAY, detail="视频下载失败: " + str(exc)[:200]) from exc
try:
text = transcribe_to_text(video_path)
return text.strip(), 0.0
except ASRNotConfiguredError as exc:
raise HTTPException(status_code=status.HTTP_503_SERVICE_UNAVAILABLE, detail=str(exc)) from exc
except ASRTranscriptionError as exc:
raise HTTPException(status_code=status.HTTP_502_BAD_GATEWAY, detail=str(exc)) from exc
except Exception as exc:
logger.exception("直链下载后 ASR 转写异常: path=%s", video_path)
raise HTTPException(
status_code=status.HTTP_502_BAD_GATEWAY,
detail="语音识别失败: " + str(exc)[:200],
) from exc
def _ytdlp_download_and_local_asr(page_url, temp_dir, cookiefile=None):
try:
import yt_dlp
@@ -332,6 +476,7 @@ def _ytdlp_download_and_local_asr(page_url, temp_dir, cookiefile=None):
"quiet": True,
"no_warnings": True,
"noplaylist": True,
"extractor_args": {"douyin": {"webpage_cookie": ""}},
"http_headers": {
"User-Agent": (
"Mozilla/5.0 (Windows NT 10.0; Win64; x64) "
@@ -339,6 +484,7 @@ def _ytdlp_download_and_local_asr(page_url, temp_dir, cookiefile=None):
"Chrome/128.0.0.0 Safari/537.36"
),
"Referer": "https://www.douyin.com/",
"Accept-Language": "zh-CN,zh;q=0.9,en;q=0.8",
},
}
if cookiefile:
@@ -447,6 +593,12 @@ def extract_from_douyin(
# 路径 A:yt-dlp 拿直链 + MediaKit 云端 ASR
direct_url, meta_duration = _ytdlp_extract_video_url(page_url, cookiefile=cookiefile)
if not direct_url:
logger.info("yt-dlp 未拿到直链,尝试第三方无水印解析 API 兜底")
try:
direct_url, meta_duration = _parser_extract_direct_url(page_url)
except Exception as exc: # noqa: BLE001
logger.warning("第三方解析兜底异常: %s", exc)
if meta_duration:
duration = meta_duration
_dbg("direct_url", direct_url or "<none>")
@@ -469,12 +621,26 @@ def extract_from_douyin(
if not text:
_dbg("fallback", "download+local_asr")
with tempfile.TemporaryDirectory(prefix="douyin_extract_") as temp_dir:
text, dl_duration = _ytdlp_download_and_local_asr(
page_url, temp_dir, cookiefile=cookiefile
)
# 若已有第三方解析拿到的直链,优先直接下载 MP4 给本地 ASR
if direct_url:
_dbg("fallback_via", "direct_url_download")
text, dl_duration = _direct_url_download_and_local_asr(
direct_url, page_url, temp_dir
)
else:
text, dl_duration = _ytdlp_download_and_local_asr(
page_url, temp_dir, cookiefile=cookiefile
)
if dl_duration and not duration:
duration = dl_duration
if not text:
logger.warning("抖音文案提取全部失败: url=%s", page_url)
raise HTTPException(
status_code=status.HTTP_503_SERVICE_UNAVAILABLE,
detail="抖音链接解析暂时不可用,请稍后重试或手动输入文案",
)
return ExtractFromDouyinResponse(
text=text,
duration_seconds=duration,