From ea4b74216feb72cf8e3fa72590f39b6dd99b6a6f Mon Sep 17 00:00:00 2001 From: xiaoxia Date: Thu, 17 Sep 2026 15:06:48 +0800 Subject: [PATCH] =?UTF-8?q?fix(douyin):=20=E7=A7=BB=E9=99=A4=E4=BC=AA?= =?UTF-8?q?=E9=80=A0s=5Fv=5Fweb=5Fid(=E6=A0=B9=E5=9B=A0:=E5=AF=BC=E8=87=B4?= =?UTF-8?q?yt-dlp=20100%=E5=A4=B1=E8=B4=A5)+apizero=E4=BC=98=E5=85=88+stag?= =?UTF-8?q?ing=E9=BB=98=E8=AE=A4debug+verify=3DFalse?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- apps/api/app/api/routes/scripts_ai.py | 51 +++++++++++++++------------ 1 file changed, 29 insertions(+), 22 deletions(-) diff --git a/apps/api/app/api/routes/scripts_ai.py b/apps/api/app/api/routes/scripts_ai.py index 5b790e0e2..30cc36411 100644 --- a/apps/api/app/api/routes/scripts_ai.py +++ b/apps/api/app/api/routes/scripts_ai.py @@ -311,7 +311,7 @@ def _fetch_ttwid(): # 方式一:直接调 ttwid 注册接口 try: with httpx.Client(timeout=8, follow_redirects=True, verify=False) as http: - http.post( + r = http.post( "https://ttwid.bytedance.com/ttwid/union/register/", json={ "region": "cn", @@ -366,22 +366,23 @@ def _fetch_ttwid(): def _generate_synthetic_cookie_file(temp_dir, ttwid=None): - """生成一份合成 cookie 文件(ttwid + 随机 s_v_web_id),返回文件路径。 - s_v_web_id 是抖音反爬验证 cookie,格式 verify_ + 22 位随机串。 + """生成一份合成 cookie 文件(仅 ttwid,不伪造 s_v_web_id),返回文件路径。 + + 重要:不要伪造 s_v_web_id(如 verify_xxx 随机串)!yt-dlp DouyinIE 在检测到 + 无效的 s_v_web_id 时会直接抛致命错误("Fresh cookies needed"),而不是走内部 + 重试路径。只提供 ttwid 时,yt-dlp 会把缺少 s_v_web_id 视为 expected 错误, + 内部走兜底路径拿数据,配合多次重试有一定概率成功。 """ import os import secrets - import string import time if not ttwid: ttwid = _fetch_ttwid() if not ttwid: + # 兜底:生成一个格式合法的假 ttwid(至少不会直接报错) ttwid = "1%7C" + secrets.token_hex(20) + "%7C" + str(int(time.time())) + "%7C" + secrets.token_hex(32) - alphabet = string.ascii_lowercase + string.digits + "_" - rand = "".join(secrets.choice(alphabet) for _ in range(22)) - s_v_web_id = f"verify_{rand}" expiry = int(time.time()) + 86400 * 30 # 30 天 path = os.path.join(temp_dir, f"dy_cookies_{secrets.token_hex(4)}.txt") @@ -389,7 +390,7 @@ def _generate_synthetic_cookie_file(temp_dir, ttwid=None): f.write("# Netscape HTTP Cookie File\n") f.write("# This file is generated by yt-dlp. Do not edit.\n\n") f.write(f".douyin.com\tTRUE\t/\tTRUE\t{expiry}\tttwid\t{ttwid}\n") - f.write(f".douyin.com\tTRUE\t/\tFALSE\t{expiry}\ts_v_web_id\t{s_v_web_id}\n") + # 注意:不写 s_v_web_id!伪造的 verify_xxx 串会导致 yt-dlp 直接失败。 return path @@ -490,8 +491,8 @@ def _build_parser_headers(name): "Accept-Language": "zh-CN,zh;q=0.9,en;q=0.8", } if name == "apizero" and _APIZERO_API_KEY: - # apizero 通过 query 参数传 key - pass + # apizero 推荐用 Authorization: Bearer ,同时兼容 query key + base["Authorization"] = f"Bearer {_APIZERO_API_KEY}" return base @@ -567,7 +568,7 @@ def _parser_extract_direct_url(page_url, timeout=12): headers = _build_parser_headers(name) for attempt in range(_PARSER_MAX_RETRIES): try: - with httpx.Client(timeout=timeout, follow_redirects=True) as http: + with httpx.Client(timeout=timeout, follow_redirects=True, verify=False) as http: resp = http.get(url, headers=headers) resp.raise_for_status() data = resp.json() @@ -723,7 +724,7 @@ def _ytdlp_download_and_local_asr(page_url, temp_dir, cookiefile=None): cookie_files_to_try.append(cookiefile) # 2. 合成 cookie 重试(复用 _ytdlp_extract 的 cookie 生成逻辑,8次重试) - import time as _time + import secrets, time as _time ttwid = _fetch_ttwid() for i in range(8): if i % 3 == 0: @@ -919,28 +920,34 @@ def extract_from_douyin( direct_url = None meta_duration = 0.0 - # 路径 A1:yt-dlp 拿直链(有 fresh cookies 时成功率最高) - direct_url, meta_duration = _ytdlp_extract_video_url(page_url, cookiefile=cookiefile) + # 路径 A1:第三方解析 API(最稳定,apizero 付费时 99.88% 可用) + logger.info("抖音解析第%d轮开始: url=%s", round_idx + 1, page_url) + try: + direct_url, meta_duration = _parser_extract_direct_url(page_url) + if direct_url: + logger.info( + "第三方解析 API 成功(第%d轮): url_domain=%s", + round_idx + 1, direct_url.split("/")[2] if "/" in direct_url else "?", + ) + except Exception as exc: # noqa: BLE001 + logger.warning("第三方解析 API 异常(第%d轮): %s", round_idx + 1, exc) - # 路径 A2:第三方解析 API 兜底 + # 路径 A2:yt-dlp 拿直链(作为兜底,需要 cookie 概率性成功) if not direct_url: logger.info( - "yt-dlp 未拿到直链(第%d轮),尝试第三方无水印解析 API", round_idx + 1 + "第三方解析未拿到直链(第%d轮),尝试 yt-dlp", round_idx + 1 ) - try: - direct_url, meta_duration = _parser_extract_direct_url(page_url) - except Exception as exc: # noqa: BLE001 - logger.warning("第三方解析兜底异常: %s", exc) + direct_url, meta_duration = _ytdlp_extract_video_url(page_url, cookiefile=cookiefile) # 路径 A3:HTML 直抓(最后兜底,不依赖任何 API Key / cookies) if not direct_url: logger.info( - "第三方解析未拿到直链(第%d轮),尝试 HTML 直接抓取", round_idx + 1 + "yt-dlp 未拿到直链(第%d轮),尝试 HTML 直接抓取", round_idx + 1 ) try: direct_url, meta_duration = _html_scrape_direct_url(page_url) except Exception as exc: # noqa: BLE001 - logger.warning("HTML 直抓异常: %s", exc) + logger.warning("HTML 直抓异常(第%d轮): %s", round_idx + 1, exc) if meta_duration: duration = meta_duration