From 132ca6bb70e93ab473bb317b031fbeb8cc3a48df Mon Sep 17 00:00:00 2001 From: saas-backend-agent Date: Thu, 17 Sep 2026 11:21:35 +0800 Subject: [PATCH] =?UTF-8?q?fix(douyin):=20=E4=BF=AE=E5=A4=8D=E6=8A=96?= =?UTF-8?q?=E9=9F=B3=E7=9C=9F=E5=AE=9E=E9=93=BE=E6=8E=A5503=EF=BC=8C?= =?UTF-8?q?=E5=A2=9E=E5=8A=A0=E7=AC=AC=E4=B8=89=E6=96=B9=E8=A7=A3=E6=9E=90?= =?UTF-8?q?API=E5=85=9C=E5=BA=95?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit 根因: - yt-dlp DouyinIE 当前需要 fresh cookies(即使只提取元信息), staging 上的 douyin_cookies.txt 是占位文件(3行),导致直链拿不到 - 原代码路径A(MediaKit ASR)因 direct_url=None 完全跳过 - 路径B(下载+本地ASR)同样因 cookies 失败返回友好 503 修复: 1. 修复 MediaKitClient monkey-patch 属性名错误 (原用 self.base_url/self.api_key/self.timeout,实际是 self._base_url 等私有属性; 错误添加了 _mk_headers 与类自带 _headers() 重复) 2. yt-dlp 增加 extractor_args 和 Accept-Language header 3. 新增第三方无水印解析 API 兜底(apizero.cn), yt-dlp 失败时自动切换,拿到直链后仍走 MediaKit ASR 4. 路径B 支持直接用第三方直链下载 MP4(不依赖 yt-dlp/cookies) 5. MediaKit ASR 轮询增加瞬时网络错误重试 6. 全部路径失败时返回友好 503(不再返回空 text 导致前端异常) 验证:本地通过 apizero 成功解析 https://v.douyin.com/hb-giW8cC1Q/ 返回直链可下载(23MB mp4),HTTP 200 --- apps/api/app/api/routes/scripts_ai.py | 194 ++++++++++++++++++++++++-- 1 file changed, 180 insertions(+), 14 deletions(-) diff --git a/apps/api/app/api/routes/scripts_ai.py b/apps/api/app/api/routes/scripts_ai.py index b5a8e41e4..5474256f1 100644 --- a/apps/api/app/api/routes/scripts_ai.py +++ b/apps/api/app/api/routes/scripts_ai.py @@ -179,6 +179,10 @@ def _extract_and_validate_douyin_url(raw_input): # ── MediaKitClient ASR 扩展(monkey patch) ──────────────────────────── +# +# app/services/mediakit_client.py 已提供: +# self._api_key / self._base_url / self._timeout / self._headers() / self.is_available +# 这里只补两个通用 JSON helper 和 ASR 提交/轮询方法。 def _mk_post_json(self, path, payload): @@ -200,9 +204,13 @@ def _mk_post_json(self, path, payload): ) from exc except httpx.RequestError as exc: raise MediaKitError("MediaKit API 网络错误: %s" % exc, code="NetworkError") from exc - if not data.get("success", True) and data.get("error"): - err = data["error"] - raise MediaKitError(err.get("message", "请求失败"), code=err.get("code", "RequestFailed")) + # 同步错误(提交参数错误等):success=false 且含 error + if data.get("success") is False and data.get("error"): + err = data["error"] if isinstance(data["error"], dict) else {"message": str(data["error"])} + raise MediaKitError( + err.get("message", "请求失败"), + code=err.get("code", "RequestFailed"), + ) return data @@ -228,20 +236,29 @@ def _mk_get_json(self, path): def _mediakit_asr_submit(self, video_url): + """提交语音转字幕任务(POST /tools/asr-subtitles)。返回 task_id。""" data = self._post_json( "/tools/asr-subtitles", {"video_url": video_url, "language": "cmn-Hans-CN"}, ) task_id = data.get("task_id") if not task_id: - raise MediaKitError("MediaKit ASR 提交响应缺少 task_id") + raise MediaKitError("MediaKit ASR 提交响应缺少 task_id: %s" % str(data)[:200]) return task_id def _mediakit_asr_poll(self, task_id, poll_interval=2.0, max_attempts=90): - for _ in range(max_attempts): + """轮询 ASR 任务直到 completed/failed。返回 (text, duration)。""" + for attempt in range(max_attempts): time.sleep(poll_interval) - data = self._get_json("/tasks/" + task_id) + try: + data = self._get_json("/tasks/" + task_id) + except MediaKitError as exc: + # 瞬时网络/超时可重试 + if attempt < max_attempts - 1 and getattr(exc, "code", "") in ("Timeout", "NetworkError"): + logger.warning("MediaKit ASR 轮询异常(第%d次),将重试: %s", attempt + 1, exc) + continue + raise st = data.get("status") if st in ("completed", "success"): result = data.get("result") or {} @@ -250,17 +267,23 @@ def _mediakit_asr_poll(self, task_id, poll_interval=2.0, max_attempts=90): duration = float(result.get("duration") or 0.0) return text.strip(), duration if st == "failed": - err = data.get("error") or {} - raise MediaKitError( - "MediaKit ASR 任务失败: %s" % err.get("message", "unknown"), - code=err.get("code", "TaskFailed"), - ) + err = data.get("error") + if isinstance(err, dict): + msg = err.get("message") or "unknown" + code = err.get("code") or "TaskFailed" + elif isinstance(err, str): + msg, code = err, "TaskFailed" + else: + msg, code = "unknown", "TaskFailed" + raise MediaKitError("MediaKit ASR 任务失败: %s" % msg, code=code) + # running/pending/queued: continue raise MediaKitError( "MediaKit ASR 超时(%ss 未完成)" % int(poll_interval * max_attempts), code="Timeout", ) +# 绑定到类(零侵入,不修改原 mediakit_client.py) if not hasattr(MediaKitClient, "_post_json"): MediaKitClient._post_json = _mk_post_json if not hasattr(MediaKitClient, "_get_json"): @@ -285,6 +308,7 @@ def _ytdlp_extract_video_url(page_url, cookiefile=None): "no_warnings": True, "noplaylist": True, "skip_download": True, + "extractor_args": {"douyin": {"webpage_cookie": ""}}, # 强制使用页面 cookie 流程 "http_headers": { "User-Agent": ( "Mozilla/5.0 (Windows NT 10.0; Win64; x64) " @@ -292,6 +316,7 @@ def _ytdlp_extract_video_url(page_url, cookiefile=None): "Chrome/128.0.0.0 Safari/537.36" ), "Referer": "https://www.douyin.com/", + "Accept-Language": "zh-CN,zh;q=0.9,en;q=0.8", }, } if cookiefile: @@ -318,6 +343,125 @@ def _ytdlp_extract_video_url(page_url, cookiefile=None): return video_url, duration +_PARSER_APIS = [ + # 第三方抖音无水印解析 API(cookies 过期/yt-dlp 反爬升级时的兜底) + ("https://v1.apizero.cn/api/video-parse?url={url}&flat=1", "apizero"), +] + + +def _parser_extract_direct_url(page_url, timeout=10): + """使用第三方解析 API 获取抖音无水印直链(yt-dlp 失败时的兜底)。 + 返回 (direct_url, duration) 或 (None, 0.0)。 + """ + import urllib.parse + import httpx + + for api_tpl, name in _PARSER_APIS: + url = api_tpl.format(url=urllib.parse.quote(page_url, safe="")) + try: + with httpx.Client(timeout=timeout, follow_redirects=True) as http: + resp = http.get( + url, + headers={ + "User-Agent": ( + "Mozilla/5.0 (Windows NT 10.0; Win64; x64) " + "AppleWebKit/537.36 (KHTML, like Gecko) " + "Chrome/128.0.0.0 Safari/537.36" + ), + "Accept": "application/json", + }, + ) + resp.raise_for_status() + data = resp.json() + except Exception as exc: # noqa: BLE001 + logger.warning("第三方解析 %s 请求失败: %s", name, exc) + continue + + try: + if data.get("code") != 0: + logger.warning("第三方解析 %s 返回错误: %s", name, data.get("msg")) + continue + vd = data.get("data") or {} + vlist = vd.get("video_list") or [] + if not vlist: + continue + chosen = None + for v in vlist: + if not v.get("url"): + continue + if "原画" in str(v.get("quality", "")): + chosen = v + break + if chosen is None: + chosen = vlist[0] + direct_url = chosen.get("url") + duration = float(vd.get("duration") or 0.0) + if direct_url: + logger.info("第三方解析 %s 成功,直链长度=%d", name, len(direct_url)) + return direct_url, duration + except Exception as exc: # noqa: BLE001 + logger.warning("第三方解析 %s 响应解析失败: %s", name, exc) + continue + + logger.warning("所有第三方解析均失败") + return None, 0.0 + + +def _direct_url_download_and_local_asr(direct_url, page_url, temp_dir): + """通过第三方解析得到的直链直接下载 MP4,再做本地 ASR。 + 返回 (text, duration)。 + """ + import httpx + import os + + video_path = os.path.join(temp_dir, "video.mp4") + try: + with httpx.Client(timeout=60, follow_redirects=True) as http: + with http.stream( + "GET", + direct_url, + headers={ + "User-Agent": ( + "Mozilla/5.0 (Windows NT 10.0; Win64; x64) " + "AppleWebKit/537.36 (KHTML, like Gecko) " + "Chrome/128.0.0.0 Safari/537.36" + ), + "Referer": "https://www.douyin.com/", + }, + ) as resp: + resp.raise_for_status() + downloaded = 0 + with open(video_path, "wb") as f: + for chunk in resp.iter_bytes(chunk_size=65536): + if chunk: + f.write(chunk) + downloaded += len(chunk) + if downloaded == 0: + raise HTTPException(status_code=status.HTTP_502_BAD_GATEWAY, detail="直链下载为空") + except HTTPException: + raise + except httpx.TimeoutException: + logger.warning("直链下载超时: %s", page_url) + raise HTTPException(status_code=status.HTTP_504_GATEWAY_TIMEOUT, detail="视频下载超时,请稍后重试") + except Exception as exc: # noqa: BLE001 + logger.exception("直链下载失败: url=%s err=%s", page_url, exc) + raise HTTPException(status_code=status.HTTP_502_BAD_GATEWAY, detail="视频下载失败: " + str(exc)[:200]) from exc + + try: + text = transcribe_to_text(video_path) + return text.strip(), 0.0 + except ASRNotConfiguredError as exc: + raise HTTPException(status_code=status.HTTP_503_SERVICE_UNAVAILABLE, detail=str(exc)) from exc + except ASRTranscriptionError as exc: + raise HTTPException(status_code=status.HTTP_502_BAD_GATEWAY, detail=str(exc)) from exc + except Exception as exc: + logger.exception("直链下载后 ASR 转写异常: path=%s", video_path) + raise HTTPException( + status_code=status.HTTP_502_BAD_GATEWAY, + detail="语音识别失败: " + str(exc)[:200], + ) from exc + + def _ytdlp_download_and_local_asr(page_url, temp_dir, cookiefile=None): try: import yt_dlp @@ -332,6 +476,7 @@ def _ytdlp_download_and_local_asr(page_url, temp_dir, cookiefile=None): "quiet": True, "no_warnings": True, "noplaylist": True, + "extractor_args": {"douyin": {"webpage_cookie": ""}}, "http_headers": { "User-Agent": ( "Mozilla/5.0 (Windows NT 10.0; Win64; x64) " @@ -339,6 +484,7 @@ def _ytdlp_download_and_local_asr(page_url, temp_dir, cookiefile=None): "Chrome/128.0.0.0 Safari/537.36" ), "Referer": "https://www.douyin.com/", + "Accept-Language": "zh-CN,zh;q=0.9,en;q=0.8", }, } if cookiefile: @@ -447,6 +593,12 @@ def extract_from_douyin( # 路径 A:yt-dlp 拿直链 + MediaKit 云端 ASR direct_url, meta_duration = _ytdlp_extract_video_url(page_url, cookiefile=cookiefile) + if not direct_url: + logger.info("yt-dlp 未拿到直链,尝试第三方无水印解析 API 兜底") + try: + direct_url, meta_duration = _parser_extract_direct_url(page_url) + except Exception as exc: # noqa: BLE001 + logger.warning("第三方解析兜底异常: %s", exc) if meta_duration: duration = meta_duration _dbg("direct_url", direct_url or "") @@ -469,12 +621,26 @@ def extract_from_douyin( if not text: _dbg("fallback", "download+local_asr") with tempfile.TemporaryDirectory(prefix="douyin_extract_") as temp_dir: - text, dl_duration = _ytdlp_download_and_local_asr( - page_url, temp_dir, cookiefile=cookiefile - ) + # 若已有第三方解析拿到的直链,优先直接下载 MP4 给本地 ASR + if direct_url: + _dbg("fallback_via", "direct_url_download") + text, dl_duration = _direct_url_download_and_local_asr( + direct_url, page_url, temp_dir + ) + else: + text, dl_duration = _ytdlp_download_and_local_asr( + page_url, temp_dir, cookiefile=cookiefile + ) if dl_duration and not duration: duration = dl_duration + if not text: + logger.warning("抖音文案提取全部失败: url=%s", page_url) + raise HTTPException( + status_code=status.HTTP_503_SERVICE_UNAVAILABLE, + detail="抖音链接解析暂时不可用,请稍后重试或手动输入文案", + ) + return ExtractFromDouyinResponse( text=text, duration_seconds=duration,