diff --git a/.env.example b/.env.example index 9256a77e4..a9caf4b3d 100755 --- a/.env.example +++ b/.env.example @@ -203,3 +203,16 @@ DOUBAO_MAX_RETRIES=2 # `if settings.points_enabled: ...` # 包裹扣点逻辑;所有路由接入完成并验证通过后再在 staging/prod 打开。 POINTS_ENABLED=false + +# ==================== 抖音解析多源轮询 (#1963) ==================== +# P0 App Feed API(免费,默认启用)→ P1 TikHub → P2 apizero → P3 GoDownloader +# 无需配置 Key 也可使用(P0 免费源可用),配置 Key 可增加兜底能力 + +# TikHub API Key (https://tikhub.io) — $0.001/次起,注册送$0.05 +TIKHUB_API_KEY= + +# apizero.cn API Key (https://v1.apizero.cn) — 国内抖音解析服务 +APIZERO_API_KEY= + +# RapidAPI Key (https://rapidapi.com) — 订阅 GoDownloader 免费层(60次/分) +RAPIDAPI_KEY= diff --git a/apps/api/app/api/routes/scripts_ai.py b/apps/api/app/api/routes/scripts_ai.py index 0ea570d5d..b20b98f4f 100644 --- a/apps/api/app/api/routes/scripts_ai.py +++ b/apps/api/app/api/routes/scripts_ai.py @@ -3,10 +3,10 @@ 三个 AI 工具接口(均挂载在 /api/v1/scripts 前缀下): - POST /extract-from-douyin 从抖音视频提取文案 - 入口自动从分享文本中正则提取 http(s) URL,兼容 "复制链接" 粘贴场景 - - yt-dlp 仅解析视频元信息(download=False)拿无水印直链,避免整段下载 - - 优先走火山 MediaKit ASR(asr-subtitles),配置了 MEDIAKIT_API_KEY 即可用 - - MediaKit 不可用/失败时,回退到本地 ASR(下载视频 + transcribe_to_text) - - cookies/ytdlp 均失败时,返回友好 503 不暴露内部错误 + - 多源轮询解析(douyin_resolver):App Feed API → TikHub → apizero → GoDownloader + - 拿到 MP4 直链后优先走火山 MediaKit ASR,失败回退下载+本地 ASR + - ASR 空结果时使用 Feed desc 兜底,图文视频直接返回 desc + - 所有源均失败时返回具体错误信息(不暴露内部细节) - POST /ai-rewrite AI 文案改写(复用豆包 LLM) - POST /ai-generate-titles AI 标题生成(复用 generate_smart_titles) """ @@ -30,6 +30,7 @@ from app.schemas.scripts_ai import ( ExtractFromDouyinRequest, ExtractFromDouyinResponse, ) +from app.services.douyin_resolver import available_providers, resolve_douyin_video from app.services.mediakit_client import ( MediaKitClient, MediaKitError, @@ -50,63 +51,13 @@ logger = logging.getLogger(__name__) router = APIRouter() -DOUYIN_COOKIES_FILE = os.environ.get( - "DOUYIN_COOKIES_FILE", - "/app/configs/douyin_cookies.txt", -) -DOUYIN_COOKIES_FILE_BAKED = "/app/configs/douyin_cookies_default.txt" - -_COOKIES_ERROR_KEYWORDS = ( - "fresh cookies", - "cookies (not necessarily logged in)", - "cookies are needed", - "need cookies", - "cookie is expired", - "login required", - "sign in to continue", - "未登录", - "需要登录", - "cookies过期", -) - -_TAIL_PUNCT = ".,;:!?,。;:!?))]》" + chr(34) + chr(39) + "<>" - - -def _resolve_cookies_file(): - for p in (DOUYIN_COOKIES_FILE, DOUYIN_COOKIES_FILE_BAKED): - try: - if p and os.path.isfile(p) and os.path.getsize(p) > 200: - return p - except OSError: - continue - return None - - -def _dbg(key, val): - logger.debug("douyin_extract %s=%s", key, str(val)[:200]) - - -def _is_cookies_related_error(msg): - low = msg.lower() - return any(kw in low for kw in _COOKIES_ERROR_KEYWORDS) - - -_cf = _resolve_cookies_file() -if _cf: - logger.info("抖音 cookies 文件已加载: %s (%d bytes)", _cf, os.path.getsize(_cf)) -else: - logger.warning( - "抖音 cookies 文件未找到或无效: path=%s baked=%s", - DOUYIN_COOKIES_FILE, - DOUYIN_COOKIES_FILE_BAKED, - ) - _DOUYIN_DEBUG_ERRORS = os.environ.get("DOUYIN_DEBUG_ERRORS", "").lower() in ( "1", "true", "yes", ) or os.environ.get("APP_ENV", "").lower() in ("staging", "dev", "development", "test") +_TAIL_PUNCT = ".,;:!?,。;:!?))]》" + chr(34) + chr(39) + "<>" _URL_EXTRACT_RE = re.compile(r"https?://\S+", re.IGNORECASE) _DOUYIN_HOST_RE = re.compile( r"(^|\.)(douyin\.com|iesdouyin\.com|amemv\.com)$", @@ -115,6 +66,10 @@ _DOUYIN_HOST_RE = re.compile( _ANY_SCHEME_RE = re.compile(r"^[a-z][a-z0-9+.-]*://\S+", re.IGNORECASE) +def _dbg(key, val): + logger.debug("douyin_extract %s=%s", key, str(val)[:200]) + + def _extract_url_from_text(raw): if not raw: return None @@ -139,13 +94,11 @@ def _extract_and_validate_douyin_url(raw_input): url = _extract_url_from_text(raw) if not url: - # 含非 http(s) 的 scheme 前缀(如 ftp://、file:// 等)→ 协议不支持 if _ANY_SCHEME_RE.search(raw): raise HTTPException( status_code=status.HTTP_400_BAD_REQUEST, detail="无效的抖音链接,仅支持 http(s) 协议", ) - # 裸域名兜底:在去除 scheme 的情况下匹配 douyin 域名 short = re.search( r"(?:^|(? 标准格式。 - yt-dlp DouyinIE 只匹配 douyin.com/video/,短链/iesdouyin会报Unsupported URL。 - """ - import httpx as _httpx - - # 已经是标准 douyin.com/video/ID 或 /note/ID 直接返回 - m = re.search(r"douyin\.com/(?:video|note)/(\d+)", url or "") - if m: - return f"https://www.douyin.com/video/{m.group(1)}" - - # iesdouyin.com/share/video/ID - m = re.search(r"iesdouyin\.com/share/video/(\d+)", url or "") - if m: - return f"https://www.douyin.com/video/{m.group(1)}" - - # v.douyin.com 短链:HEAD/GET 跟随重定向解析 - if re.search(r"v\.douyin\.com/", url or ""): - try: - with _httpx.Client( - timeout=8, - follow_redirects=True, - verify=False, - headers={ - "User-Agent": ( - "Mozilla/5.0 (Windows NT 10.0; Win64; x64) " - "AppleWebKit/537.36 (KHTML, like Gecko) Chrome/128.0.0.0 Safari/537.36" - ), - }, - ) as http: - r = http.get(url) - final = str(r.url) - m = re.search(r"douyin\.com/(?:video|note)/(\d+)", final) or re.search( - r"iesdouyin\.com/share/video/(\d+)", final - ) - if m: - return f"https://www.douyin.com/video/{m.group(1)}" - except Exception as exc: - logger.warning("douyin 短链解析失败 %s: %s", url, exc) - return url - - # ── MediaKitClient ASR 扩展(monkey patch) ──────────────────────────── -# -# app/services/mediakit_client.py 已提供: -# self._api_key / self._base_url / self._timeout / self._headers() / self.is_available -# 这里只补两个通用 JSON helper 和 ASR 提交/轮询方法。 def _mk_post_json(self, path, payload): @@ -253,7 +157,6 @@ def _mk_post_json(self, path, payload): ) from exc except httpx.RequestError as exc: raise MediaKitError("MediaKit API 网络错误: %s" % exc, code="NetworkError") from exc - # 同步错误(提交参数错误等):success=false 且含 error if data.get("success") is False and data.get("error"): err = data["error"] if isinstance(data["error"], dict) else {"message": str(data["error"])} raise MediaKitError( @@ -303,7 +206,6 @@ def _mediakit_asr_poll(self, task_id, poll_interval=2.0, max_attempts=90): try: data = self._get_json("/tasks/" + task_id) except MediaKitError as exc: - # 瞬时网络/超时可重试 if attempt < max_attempts - 1 and getattr(exc, "code", "") in ("Timeout", "NetworkError"): logger.warning("MediaKit ASR 轮询异常(第%d次),将重试: %s", attempt + 1, exc) continue @@ -325,14 +227,13 @@ def _mediakit_asr_poll(self, task_id, poll_interval=2.0, max_attempts=90): else: msg, code = "unknown", "TaskFailed" raise MediaKitError("MediaKit ASR 任务失败: %s" % msg, code=code) - # running/pending/queued: continue raise MediaKitError( "MediaKit ASR 超时(%ss 未完成)" % int(poll_interval * max_attempts), code="Timeout", ) -# 绑定到类(零侵入,不修改原 mediakit_client.py) +# 绑定到类(零侵入) if not hasattr(MediaKitClient, "_post_json"): MediaKitClient._post_json = _mk_post_json if not hasattr(MediaKitClient, "_get_json"): @@ -343,332 +244,11 @@ if not hasattr(MediaKitClient, "asr_poll"): MediaKitClient.asr_poll = _mediakit_asr_poll -# ── yt-dlp 辅助 ────────────────────────────────────────────────────── - - -def _fetch_ttwid(): - """从字节跳动 ttwid 注册接口获取 ttwid cookie。 - 通过 POST https://ttwid.bytedance.com/ttwid/union/register/ 可直接获得,无需浏览器。 - 失败时回退到访问 douyin.com 主页。 - """ - import httpx - - # 方式一:直接调 ttwid 注册接口 - try: - with httpx.Client(timeout=8, follow_redirects=True, verify=False) as http: - http.post( - "https://ttwid.bytedance.com/ttwid/union/register/", - json={ - "region": "cn", - "aid": 6383, - "needFid": False, - "service": "www.douyin.com", - "migrate_info": {"ticket": "", "source": "node"}, - "cbUrlProtocol": "https", - "union": True, - }, - headers={ - "User-Agent": ( - "Mozilla/5.0 (Windows NT 10.0; Win64; x64) " - "AppleWebKit/537.36 (KHTML, like Gecko) " - "Chrome/128.0.0.0 Safari/537.36" - ), - "Content-Type": "application/json", - "Referer": "https://www.douyin.com/", - }, - ) - ttwid = http.cookies.get("ttwid") - if ttwid and len(ttwid) > 20: - _dbg("ttwid_fetch", "register_ok len=%d" % len(ttwid)) - return ttwid - except Exception as exc: - logger.debug("ttwid 注册接口失败: %s", exc) - - # 方式二:回退到访问主页 - try: - with httpx.Client(timeout=8, follow_redirects=True, verify=False) as http: - for _ in range(2): - try: - http.get( - "https://www.douyin.com/", - headers={ - "User-Agent": ( - "Mozilla/5.0 (Windows NT 10.0; Win64; x64) " - "AppleWebKit/537.36 (KHTML, like Gecko) " - "Chrome/128.0.0.0 Safari/537.36" - ), - "Accept-Language": "zh-CN,zh;q=0.9", - }, - ) - ttwid = http.cookies.get("ttwid") - if ttwid: - return ttwid - except Exception: - continue - except Exception as exc: - logger.debug("获取 ttwid 失败: %s", exc) - return None - - -def _generate_synthetic_cookie_file(temp_dir, ttwid=None): - """生成一份合成 cookie 文件(仅 ttwid,不伪造 s_v_web_id),返回文件路径。 - - 重要:不要伪造 s_v_web_id(如 verify_xxx 随机串)!yt-dlp DouyinIE 在检测到 - 无效的 s_v_web_id 时会直接抛致命错误("Fresh cookies needed"),而不是走内部 - 重试路径。只提供 ttwid 时,yt-dlp 会把缺少 s_v_web_id 视为 expected 错误, - 内部走兜底路径拿数据,配合多次重试有一定概率成功。 - """ - import os - import secrets - import time - - if not ttwid: - ttwid = _fetch_ttwid() - if not ttwid: - # 兜底:生成一个格式合法的假 ttwid(至少不会直接报错) - ttwid = "1%7C" + secrets.token_hex(20) + "%7C" + str(int(time.time())) + "%7C" + secrets.token_hex(32) - - expiry = int(time.time()) + 86400 * 30 # 30 天 - - path = os.path.join(temp_dir, f"dy_cookies_{secrets.token_hex(4)}.txt") - with open(path, "w") as f: - f.write("# Netscape HTTP Cookie File\n") - f.write("# This file is generated by yt-dlp. Do not edit.\n\n") - f.write(f".douyin.com\tTRUE\t/\tTRUE\t{expiry}\tttwid\t{ttwid}\n") - # 注意:不写 s_v_web_id!伪造的 verify_xxx 串会导致 yt-dlp 直接失败。 - return path - - -def _ytdlp_try_extract(page_url, cookiefile): - """单次 yt-dlp 提取,返回 (info_dict, None) 或 (None, error_message)。 - 显式选择 h264+aac 的 MP4 格式,避免 h265 编码导致 MediaKit ASR 不兼容。 - """ - import yt_dlp - - opts = { - "quiet": True, - "no_warnings": True, - "noplaylist": True, - "skip_download": True, - "extractor_args": {"douyin": {"webpage_cookie": ""}}, - "format": "best[ext=mp4][vcodec^=avc1]/best[ext=mp4][vcodec^=h264]/best[ext=mp4]/best", - "http_headers": { - "User-Agent": ( - "Mozilla/5.0 (Windows NT 10.0; Win64; x64) " - "AppleWebKit/537.36 (KHTML, like Gecko) " - "Chrome/128.0.0.0 Safari/537.36" - ), - "Referer": "https://www.douyin.com/", - "Accept-Language": "zh-CN,zh;q=0.9,en;q=0.8", - }, - "socket_timeout": 15, - "retries": 1, - } - if cookiefile: - opts["cookiefile"] = cookiefile - try: - with yt_dlp.YoutubeDL(opts) as ydl: - info = ydl.extract_info(page_url, download=False) - return info, None - except Exception as exc: - return None, str(exc) - - -def _ytdlp_extract_video_url(page_url, cookiefile=None, max_retries=10): - """yt-dlp 提取视频直链,支持合成 cookie 多次重试。 - 优先使用传入的 cookiefile;失败后自动生成合成 cookie 重试 max_retries 次。 - 每次重试使用新的 s_v_web_id,每3次刷新 ttwid,实测约 30% 单次成功率,10次≈97%。 - """ - # 优先使用现有 cookie 文件 - if cookiefile and os.path.exists(cookiefile): - info, err = _ytdlp_try_extract(page_url, cookiefile) - if info: - vurl = info.get("url") - return (vurl, float(info.get("duration") or 0.0)) if vurl else (None, 0.0) - logger.info("提供的 cookie 文件解析失败: %s", str(err)[:150]) - - # 合成 cookie 重试 - import tempfile - import time as _time - - ttwid = _fetch_ttwid() - logger.info("ytdlp 直链提取开始: page_url=%s ttwid_ok=%s max_retries=%d", page_url, bool(ttwid), max_retries) - with tempfile.TemporaryDirectory(prefix="dy_ytdlp_cookies_") as tmpdir: - for attempt in range(max_retries): - if attempt % 3 == 0: - ttwid = _fetch_ttwid() or ttwid - gen_cookie = _generate_synthetic_cookie_file(tmpdir, ttwid=ttwid) - info, err = _ytdlp_try_extract(page_url, gen_cookie) - if info: - vurl = info.get("url") - logger.info("yt-dlp 合成 cookie 第%d次尝试成功: vurl_domain=%s", attempt + 1, vurl.split("/")[2] if vurl and "/" in vurl else "?") - return (vurl, float(info.get("duration") or 0.0)) if vurl else (None, 0.0) - logger.info("yt-dlp 合成 cookie 第%d次尝试失败: %s", attempt + 1, str(err)[:150]) - _time.sleep(0.4) - - logger.info("yt-dlp 合成 cookie 全部 %d 次重试均失败", max_retries) - return None, 0.0 - - -_PARSER_APIS = [ - # 第三方抖音无水印解析 API(cookies 过期/yt-dlp 反爬升级时的兜底) - # 每项 (url_template, name, auth_header_or_None) - # url 中 {url} 会被替换为 URL-encoded 的抖音链接 - ("https://v1.apizero.cn/api/video-parse?url={url}&flat=1", "apizero", None), -] - -# 可选:apizero API Key(登录后免费额度 5 次/天,付费更高额度) -_APIZERO_API_KEY = os.environ.get("APIZERO_API_KEY", "").strip() -_PARSER_MAX_RETRIES = 3 -_PARSER_RETRY_DELAY = 0.8 # 秒 -_PARSER_RATE_LIMIT_CODES = {4030, 4290, 429, 10004, 10008} # 各家限流 code - - -def _build_parser_headers(name): - """为解析 API 构造请求头,包含可选 API Key。""" - base = { - "User-Agent": ( - "Mozilla/5.0 (Windows NT 10.0; Win64; x64) " - "AppleWebKit/537.36 (KHTML, like Gecko) " - "Chrome/128.0.0.0 Safari/537.36" - ), - "Accept": "application/json, text/plain, */*", - "Accept-Language": "zh-CN,zh;q=0.9,en;q=0.8", - } - if name == "apizero" and _APIZERO_API_KEY: - # apizero 推荐用 Authorization: Bearer ,同时兼容 query key - base["Authorization"] = f"Bearer {_APIZERO_API_KEY}" - return base - - -def _build_parser_url(api_tpl, name, page_url_encoded): - url = api_tpl.format(url=page_url_encoded) - if name == "apizero" and _APIZERO_API_KEY: - sep = "&" if "?" in url else "?" - url = f"{url}{sep}key={_APIZERO_API_KEY}" - return url - - -def _extract_video_url_from_parser_response(name, data): - """从解析 API 的 JSON 响应中提取 (direct_url, duration)。兼容多种响应结构。""" - if not isinstance(data, dict): - return None, 0.0 - # 统一成功判定:code == 0 或 status == "success" - code = data.get("code") - is_success = code == 0 or code == 200 or data.get("success") is True or data.get("status") == "success" - if not is_success: - msg = data.get("msg") or data.get("message") or str(code) - # 返回 msg 供调用方判断是否限流 - return None, 0.0, str(msg), code - vd = data.get("data") or {} - if not isinstance(vd, dict): - return None, 0.0, "bad data", -1 - vlist = vd.get("video_list") or [] - duration = float(vd.get("duration") or vd.get("duration_ms", 0) or 0.0) - if duration > 1000: # ms → s - duration = duration / 1000.0 - direct_url = None - chosen_quality = -1 - quality_priority = ["原画", "1080p", "720p", "540p", "480p"] - for prio, keyword in enumerate(quality_priority): - for v in vlist: - if not isinstance(v, dict) or not v.get("url"): - continue - if keyword in str(v.get("quality", "")): - if prio < chosen_quality or chosen_quality == -1: - direct_url = v.get("url") - chosen_quality = prio - break - if direct_url: - break - if not direct_url: - # 兜底:取 data.video_url 顶层字段,或 video_list[0].url - if vd.get("video_url"): - direct_url = vd["video_url"] - elif vlist: - direct_url = vlist[0].get("url") - else: - # 兼容 play/play_url 字段 - direct_url = vd.get("play") or vd.get("play_url") or vd.get("url") - if not direct_url: - return None, 0.0, "no video url", -1 - return direct_url, duration, None, code - - -def _parser_extract_direct_url(page_url, timeout=12): - """使用第三方解析 API 获取抖音无水印直链(yt-dlp 失败时的兜底)。 - 每个 API 会重试 _PARSER_MAX_RETRIES 次以应对瞬时限流。 - 返回 (direct_url, duration) 或 (None, 0.0)。 - """ - import random - import urllib.parse - - import httpx - - page_url_encoded = urllib.parse.quote(page_url, safe="") - last_err = None - - for api_tpl, name, _auth in _PARSER_APIS: - url = _build_parser_url(api_tpl, name, page_url_encoded) - headers = _build_parser_headers(name) - for attempt in range(_PARSER_MAX_RETRIES): - try: - with httpx.Client(timeout=timeout, follow_redirects=True, verify=False) as http: - resp = http.get(url, headers=headers) - resp.raise_for_status() - data = resp.json() - except Exception as exc: # noqa: BLE001 - last_err = str(exc)[:200] - logger.warning( - "第三方解析 %s 第%d次请求失败: %s", name, attempt + 1, exc - ) - time.sleep(_PARSER_RETRY_DELAY * (attempt + 1) + random.random() * 0.3) - continue - - try: - result = _extract_video_url_from_parser_response(name, data) - if len(result) == 4: - direct_url, duration, err_msg, resp_code = result - else: - direct_url, duration = result - err_msg, resp_code = None, None - if direct_url: - logger.info( - "第三方解析 %s 第%d次成功,直链长度=%d", - name, attempt + 1, len(direct_url), - ) - return direct_url, duration - # 判断是否限流,可重试 - if resp_code in _PARSER_RATE_LIMIT_CODES or ( - err_msg and any( - kw in str(err_msg) for kw in ("次数", "限流", "频率", "用完", "rate limit", "quota") - ) - ): - logger.info( - "第三方解析 %s 第%d次返回限流: code=%s msg=%s,将重试", - name, attempt + 1, resp_code, err_msg, - ) - time.sleep(_PARSER_RETRY_DELAY * (attempt + 1) + random.random() * 0.5) - continue - # 业务错误(如链接无效),直接换下一个 API - logger.warning( - "第三方解析 %s 第%d次返回业务错误: code=%s msg=%s", - name, attempt + 1, resp_code, err_msg, - ) - break - except Exception as exc: # noqa: BLE001 - last_err = str(exc)[:200] - logger.warning("第三方解析 %s 响应解析失败: %s", name, exc) - break # 响应格式异常,不重试当前 API - - logger.warning("所有第三方解析均失败: last_err=%s", last_err) - return None, 0.0 +# ── 下载 + 本地 ASR 兜底 ────────────────────────────────────────────── def _direct_url_download_and_local_asr(direct_url, page_url, temp_dir): - """通过第三方解析得到的直链直接下载 MP4,再做本地 ASR。 - 返回 (text, duration)。 - """ + """通过直链下载 MP4,再做本地 ASR。返回 (text, duration)。""" import os import httpx @@ -723,385 +303,68 @@ def _direct_url_download_and_local_asr(direct_url, page_url, temp_dir): ) from exc -def _ytdlp_download_and_local_asr(page_url, temp_dir, cookiefile=None): - """下载抖音视频 + 本地 ASR。优先使用传入 cookie,失败时自动合成 cookie 重试。""" - try: - import yt_dlp - except ImportError as exc: - raise HTTPException( - status_code=status.HTTP_503_SERVICE_UNAVAILABLE, - detail="抖音提取功能暂不可用(缺少依赖 yt-dlp)", - ) from exc - - def _download(cfile): - opts = { - "format": "best[ext=mp4][vcodec^=avc1]/best[ext=mp4][vcodec^=h264]/best[ext=mp4]/best", - "outtmpl": temp_dir + "/%(id)s.%(ext)s", - "quiet": True, - "no_warnings": True, - "noplaylist": True, - "extractor_args": {"douyin": {"webpage_cookie": ""}}, - "socket_timeout": 30, - "retries": 2, - "http_headers": { - "User-Agent": ( - "Mozilla/5.0 (Windows NT 10.0; Win64; x64) " - "AppleWebKit/537.36 (KHTML, like Gecko) " - "Chrome/128.0.0.0 Safari/537.36" - ), - "Referer": "https://www.douyin.com/", - "Accept-Language": "zh-CN,zh;q=0.9,en;q=0.8", - }, - } - if cfile: - opts["cookiefile"] = cfile - with yt_dlp.YoutubeDL(opts) as ydl: - info = ydl.extract_info(page_url, download=True) - vpath = ydl.prepare_filename(info) - dur = float(info.get("duration") or 0) - return vpath, dur - - # 1. 先用传入 cookie - vpath = None - last_err = None - cookie_files_to_try = [] - if cookiefile and os.path.exists(cookiefile): - cookie_files_to_try.append(cookiefile) - - # 2. 合成 cookie 重试(复用 _ytdlp_extract 的 cookie 生成逻辑,8次重试) - import time as _time - ttwid = _fetch_ttwid() - for i in range(8): - if i % 3 == 0: - ttwid = _fetch_ttwid() or ttwid - cfile = _generate_synthetic_cookie_file(temp_dir, ttwid=ttwid) - cookie_files_to_try.append(cfile) - _time.sleep(0.3) - - for cfile in cookie_files_to_try: - try: - vpath, duration = _download(cfile) - if vpath and os.path.isfile(vpath) and os.path.getsize(vpath) > 0: - break - logger.warning( - "ytdlp 下载返回路径无效或空文件: vpath=%s size=%s", - vpath, - os.path.getsize(vpath) if vpath and os.path.isfile(vpath) else "N/A", - ) - except Exception as exc: # noqa: BLE001 - last_err = exc - logger.warning("ytdlp 下载重试失败: %s", str(exc)[:200]) - continue - else: - msg = str(last_err) if last_err else "unknown" - logger.warning( - "ytdlp 下载全部 %d 次 cookie 尝试均失败: ttwid_ok=%s last_err=%s", - len(cookie_files_to_try), bool(ttwid), msg[:300], - ) - if _is_cookies_related_error(msg) or "Fresh cookies" in msg: - _detail = "抖音链接解析暂时不可用,请稍后重试或手动输入文案" - if _DOUYIN_DEBUG_ERRORS: - _detail = _detail + " [debug: " + msg[:300] + "]" - raise HTTPException(status_code=status.HTTP_503_SERVICE_UNAVAILABLE, detail=_detail) - _detail = "视频下载失败,请稍后重试" - if _DOUYIN_DEBUG_ERRORS: - _detail = _detail + " [debug: " + msg[:300] + "]" - raise HTTPException(status_code=status.HTTP_502_BAD_GATEWAY, detail=_detail) - - if not vpath or not os.path.isfile(vpath) or os.path.getsize(vpath) == 0: - raise HTTPException(status_code=status.HTTP_502_BAD_GATEWAY, detail="视频下载异常:未获取到有效文件") - try: - duration = float(duration or 0) - except (TypeError, ValueError): - duration = 0.0 - - try: - text = transcribe_to_text(vpath) - except ASRNotConfiguredError as exc: - raise HTTPException(status_code=status.HTTP_503_SERVICE_UNAVAILABLE, detail=str(exc)) from exc - except ASRTranscriptionError as exc: - raise HTTPException(status_code=status.HTTP_502_BAD_GATEWAY, detail=str(exc)) from exc - except Exception as exc: - logger.exception("ASR 转写异常: path=%s", vpath) - raise HTTPException( - status_code=status.HTTP_502_BAD_GATEWAY, - detail="语音识别失败: " + str(exc)[:200], - ) from exc - return text.strip(), duration - - -def _html_scrape_direct_url(page_url, timeout=15): - """通过抖音分享页 HTML 直接抓取视频 CDN 直链(最后兜底,不依赖任何第三方 API)。 - 注意:桌面版页面通常是 SPA 需 JS 渲染,移动分享页 iesdouyin.com 含 SSR 数据。 - 返回 (direct_url, duration) 或 (None, 0.0)。 - """ - import re - - import httpx - - aweme_id = None - m = re.search(r"(?:video|note)/(\d{10,})", page_url) - if not m: - # 先展开短链拿 aweme_id - try: - with httpx.Client(timeout=timeout, follow_redirects=False, verify=False) as http: - r = http.head( - page_url, - headers={ - "User-Agent": ( - "Mozilla/5.0 (iPhone; CPU iPhone OS 16_6 like Mac OS X) " - "AppleWebKit/605.1.15 (KHTML, like Gecko) Version/16.6 " - "Mobile/15E148 Safari/604.1" - ), - }, - ) - loc = r.headers.get("location", "") - m2 = re.search(r"(?:video|note)/(\d{10,})", loc) - if m2: - aweme_id = m2.group(1) - except Exception as exc: # noqa: BLE001 - logger.debug("HTML 解析短链重定向失败: %s", exc) - else: - aweme_id = m.group(1) - - if not aweme_id: - # 直接 GET 短链跟随 - try: - with httpx.Client(timeout=timeout, follow_redirects=True, verify=False) as http: - r = http.get( - page_url, - headers={ - "User-Agent": ( - "Mozilla/5.0 (iPhone; CPU iPhone OS 16_6 like Mac OS X) " - "AppleWebKit/605.1.15 (KHTML, like Gecko) Version/16.6 " - "Mobile/15E148 Safari/604.1" - ), - }, - ) - m2 = re.search(r"(?:video|note)/(\d{10,})", str(r.url)) - if m2: - aweme_id = m2.group(1) - except Exception: # noqa: BLE001 - pass - - if not aweme_id: - return None, 0.0 - - # 尝试移动分享页 - share_url = f"https://www.iesdouyin.com/share/video/{aweme_id}/" - try: - with httpx.Client(timeout=timeout, follow_redirects=True, verify=False) as http: - r = http.get( - share_url, - headers={ - "User-Agent": ( - "Mozilla/5.0 (iPhone; CPU iPhone OS 16_6 like Mac OS X) " - "AppleWebKit/605.1.15 (KHTML, like Gecko) Version/16.6 " - "Mobile/15E148 Safari/604.1" - ), - "Referer": "https://www.douyin.com/", - "Accept-Language": "zh-CN,zh;q=0.9", - }, - ) - if r.status_code != 200: - return None, 0.0 - html = r.text - # 先解码转义 - html_decoded = ( - html.replace("\\u002F", "/") - .replace("\\/", "/") - .replace("&", "&") - ) - # 匹配 douyinvod CDN 链接(可能是 playwm 带水印版本) - urls = re.findall( - r"https?://[^\"'\\<>\s]+?douyinvod\.com[^\"'\\<>\s]*", - html_decoded, - ) - mp4_urls = [u for u in urls if ".mp4" in u or "/video/" in u] - # 也匹配 amemv CDN - amemv_urls = re.findall( - r"https?://[^\"'\\<>\s]+?amemv\.com[^\"'\\<>\s]*?\.mp4[^\"'\\<>\s]*", - html_decoded, - ) - all_urls = mp4_urls + amemv_urls - if all_urls: - # 优先无水印(URL 中不含 playwm) - for u in all_urls: - if "playwm" not in u and len(u) > 40: - logger.info("HTML 抓取到无水印直链: %s", u[:120]) - return u, 0.0 - for u in all_urls: - if len(u) > 40: - logger.info("HTML 抓取到带水印直链(可用于ASR): %s", u[:120]) - return u, 0.0 - except Exception as exc: # noqa: BLE001 - logger.debug("HTML 抓取异常: %s", exc) - return None, 0.0 - - # ── 1. 从抖音视频提取文案 ───────────────────────────────────────────── @router.get("/douyin/__debug_diag") def douyin_diag(): - """[Staging only] 容器内网络/yt-dlp诊断""" - import tempfile as _tf + """[Staging/Dev only] 抖音解析源诊断。""" + import httpx as _httpx import time as _t - import httpx as _httpx - import yt_dlp as _ytdlp + from app.services.douyin_resolver import APIZERO_API_KEY as _api_key_apizero + from app.services.douyin_resolver import TIKHUB_API_KEY as _api_key_tikhub - results = {} + results = { + "providers": available_providers(), + "env": { + "APP_ENV": os.environ.get("APP_ENV", ""), + "MEDIAKIT_CONFIGURED": bool(os.environ.get("MEDIAKIT_API_KEY", "")), + }, + } + + test_url = "https://v.douyin.com/hb-giW8cC1Q/" - # 1. ttwid register t0 = _t.time() try: - with _httpx.Client(timeout=8, verify=False, follow_redirects=True) as c: - r = c.post( - "https://ttwid.bytedance.com/ttwid/union/register/", - json={ - "region": "cn", - "aid": 6383, - "needFid": False, - "service": "www.douyin.com", - "migrate_info": {"ticket": "", "source": "node"}, - "cbUrlProtocol": "https", - "union": True, - }, - headers={ - "User-Agent": "Mozilla/5.0 Chrome/128", - "Content-Type": "application/json", - "Referer": "https://www.douyin.com/", - }, - ) - ttwid = c.cookies.get("ttwid") - results["ttwid"] = { - "status": r.status_code, - "time": round(_t.time() - t0, 2), - "ttwid_len": len(ttwid) if ttwid else 0, - "ttwid_prefix": ttwid[:30] if ttwid else None, - } - except Exception as e: - results["ttwid"] = {"error": str(e)[:200], "time": round(_t.time() - t0, 2)} - - # 2. douyin.com homepage - t0 = _t.time() - try: - with _httpx.Client(timeout=8, verify=False, follow_redirects=True) as c: - r = c.get("https://www.douyin.com/", headers={"User-Agent": "Mozilla/5.0 Chrome/128"}) - results["douyin_home"] = {"status": r.status_code, "len": len(r.text), "time": round(_t.time() - t0, 2)} - except Exception as e: - results["douyin_home"] = {"error": str(e)[:200], "time": round(_t.time() - t0, 2)} - - # 3. v.douyin.com short link resolve - t0 = _t.time() - try: - with _httpx.Client(timeout=8, verify=False, follow_redirects=False) as c: - r = c.get("https://v.douyin.com/hb-giW8cC1Q/", headers={"User-Agent": "Mozilla/5.0 Chrome/128"}) - loc = r.headers.get("location", "") - results["short_link"] = {"status": r.status_code, "location_prefix": loc[:120], "time": round(_t.time() - t0, 2)} - except Exception as e: - results["short_link"] = {"error": str(e)[:200], "time": round(_t.time() - t0, 2)} - - # 4. apizero free API - t0 = _t.time() - try: - with _httpx.Client(timeout=10, verify=False, follow_redirects=True) as c: - r = c.get( - "https://v1.apizero.cn/api/video-parse", - params={"url": "https://v.douyin.com/hb-giW8cC1Q/", "flat": 1}, - headers={"User-Agent": "Mozilla/5.0 Chrome/128"}, - ) - results["apizero"] = {"status": r.status_code, "body_prefix": r.text[:200], "time": round(_t.time() - t0, 2)} - except Exception as e: - results["apizero"] = {"error": str(e)[:200], "time": round(_t.time() - t0, 2)} - - # 5. yt-dlp single attempt with ttwid only - t0 = _t.time() - ttwid_val = None - try: - with _httpx.Client(timeout=8, verify=False) as c: - c.post( - "https://ttwid.bytedance.com/ttwid/union/register/", - json={ - "region": "cn", - "aid": 6383, - "needFid": False, - "service": "www.douyin.com", - "migrate_info": {"ticket": "", "source": "node"}, - "cbUrlProtocol": "https", - "union": True, - }, - headers={ - "User-Agent": "Mozilla/5.0 Chrome/128", - "Content-Type": "application/json", - "Referer": "https://www.douyin.com/", - }, - ) - ttwid_val = c.cookies.get("ttwid") - except Exception: - pass - - ytdlp_result = {"attempts": [], "time": 0} - yt_t0 = _t.time() - page_url = "https://www.douyin.com/video/7661819662322649065" - with _tf.TemporaryDirectory() as tmpdir: - for i in range(5): - cfile = f"{tmpdir}/c{i}.txt" - expiry = str(int(_t.time()) + 86400 * 30) - with open(cfile, "w") as f: - f.write("# Netscape HTTP Cookie File\n# This file is generated by yt-dlp. Do not edit.\n\n") - if ttwid_val: - f.write(f".douyin.com\tTRUE\t/\tTRUE\t{expiry}\tttwid\t{ttwid_val}\n") - opts = { - "quiet": True, - "no_warnings": True, - "noplaylist": True, - "skip_download": True, - "cookiefile": cfile, - "extractor_args": {"douyin": {"webpage_cookie": ""}}, - "format": "best[ext=mp4][vcodec^=avc1]/best[ext=mp4]/best", - "http_headers": { - "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) Chrome/128", - "Referer": "https://www.douyin.com/", - }, - "socket_timeout": 15, - "retries": 1, - } - at0 = _t.time() - try: - with _ytdlp.YoutubeDL(opts) as ydl: - info = ydl.extract_info(page_url, download=False) - vurl = info.get("url", "") - ytdlp_result["attempts"].append( - {"i": i + 1, "ok": True, "vcodec": info.get("vcodec"), "url_domain": vurl.split("/")[2] if vurl else "?", "time": round(_t.time() - at0, 2)} - ) - break - except Exception as e: - ytdlp_result["attempts"].append({"i": i + 1, "ok": False, "err": str(e)[:150], "time": round(_t.time() - at0, 2)}) - _t.sleep(0.4) - ytdlp_result["time"] = round(_t.time() - yt_t0, 2) - results["ytdlp"] = ytdlp_result - - # 6. yt-dlp version - results["ytdlp_version"] = _ytdlp.version.__version__ - - # 7. App Feed API 直连测试(路径 A0) - try: - from packages.douyin_parser import fetch_douyin_video_url as _feed_fetch - ft0 = _t.time() - f_url, f_desc = _feed_fetch(page_url, timeout=10, max_retries=2) - results["feed_api"] = { - "ok": bool(f_url), - "desc": (f_desc or "")[:100], - "url_domain": f_url.split("/")[2] if f_url and "/" in f_url else None, - "url_prefix": f_url[:100] if f_url else None, - "time": round(_t.time() - ft0, 2), + r = resolve_douyin_video(test_url) + results["resolver"] = { + "ok": bool(r), + "source": r.source if r else None, + "desc_len": len(r.desc) if r else 0, + "has_video_url": bool(r.video_url) if r else False, + "url_domain": r.video_url.split("/")[2] if r and r.video_url and "/" in r.video_url else None, + "time": round(_t.time() - t0, 2), } except Exception as e: - results["feed_api"] = {"ok": False, "error": str(e)[:200]} + results["resolver"] = {"ok": False, "error": str(e)[:200], "time": round(_t.time() - t0, 2)} + + if _api_key_apizero: + t0 = _t.time() + try: + with _httpx.Client(timeout=8, verify=False) as c: + r = c.get( + "https://v1.apizero.cn/api/video-parse", + params={"url": test_url, "flat": 2}, + headers={"Authorization": f"Bearer {_api_key_apizero}"}, + ) + results["apizero"] = {"status": r.status_code, "prefix": r.text[:200], "time": round(_t.time() - t0, 2)} + except Exception as e: + results["apizero"] = {"error": str(e)[:200], "time": round(_t.time() - t0, 2)} + + if _api_key_tikhub: + t0 = _t.time() + try: + with _httpx.Client(timeout=8, verify=False) as c: + r = c.get( + "https://api.tikhub.io/api/v1/douyin/web/get_aweme_id", + params={"url": test_url}, + headers={"Authorization": f"Bearer {_api_key_tikhub}"}, + ) + results["tikhub"] = {"status": r.status_code, "prefix": r.text[:200], "time": round(_t.time() - t0, 2)} + except Exception as e: + results["tikhub"] = {"error": str(e)[:200], "time": round(_t.time() - t0, 2)} - results["total_time"] = round(_t.time() - t0, 2) return results @@ -1113,177 +376,96 @@ def extract_from_douyin( db: Session = Depends(get_db_session), ): page_url = _extract_and_validate_douyin_url(request.url) - # 规范化为 www.douyin.com/video/ID 格式(yt-dlp DouyinIE 只认这个格式,v.douyin短链/iesdouyin会报Unsupported URL) - page_url = _canonicalize_douyin_url(page_url) _dbg("page_url", page_url) - # 路径 A0:抖音 App Feed API 直连(零依赖、无需 cookies/签名、最稳定,优先级最高) - # 使用 Android 客户端 aid=1128 接口,直接返回 aweme_list 含无水印直链 - direct_url = None - feed_desc = "" - try: - from packages.douyin_parser import fetch_douyin_video_url as _feed_fetch - direct_url, feed_desc = _feed_fetch(page_url, timeout=15, max_retries=3) - if direct_url: - logger.info("抖音 App Feed API 直连成功: url=%s play_len=%d", page_url, len(direct_url)) - elif feed_desc: - logger.info("抖音 App Feed API 返回图文视频,仅用文案: desc_len=%d", len(feed_desc)) - except Exception as exc: # noqa: BLE001 - logger.warning("抖音 App Feed API 异常: %s", exc) + # ── Phase A:多源轮询解析 MP4 直链 ── + last_err_stage = "parse" + t0 = time.time() + result = resolve_douyin_video(page_url) + resolve_elapsed = time.time() - t0 + logger.info("抖音解析耗时: %.2fs providers=%s", resolve_elapsed, available_providers()) - text = "" - duration = 0.0 - # 图文视频(direct_url 为 None 但 feed_desc 有文案)直接返回文案,跳过 ASR - if not direct_url and feed_desc: - text = feed_desc.strip() - logger.info("图文视频直接返回文案: url=%s text_len=%d", page_url, len(text)) + direct_url = result.video_url if result else None + feed_desc = (result.desc or "").strip() if result else "" + + # 图文视频(无 video_url 但有 desc)直接返回文案,跳过 ASR + if result and not direct_url and feed_desc: + logger.info("图文视频直接返回文案: source=%s desc_len=%d", result.source, len(feed_desc)) return ExtractFromDouyinResponse( - text=text, + text=feed_desc, duration_seconds=0.0, source_url=page_url, ) - cookiefile = _resolve_cookies_file() + if not direct_url: + if _DOUYIN_DEBUG_ERRORS: + detail = f"抖音视频链接解析失败,请检查链接是否正确或稍后重试 [debug: providers={available_providers()}]" + else: + detail = "抖音视频链接解析失败,请检查链接是否正确或稍后重试" + logger.warning("抖音解析全部失败: url=%s providers=%s", page_url, available_providers()) + raise HTTPException(status_code=status.HTTP_503_SERVICE_UNAVAILABLE, detail=detail) + + # ── Phase B:ASR 转文字 ── mk_client = get_mediakit_client() + text = "" + duration = 0.0 - # 整体重试:外层最多 2 轮完整链路,应对第三方 API 瞬时限流/CDN 抖动 - max_rounds = 2 - meta_duration = 0.0 - last_err_stage = "unknown" # 记录失败阶段,用于返回具体错误信息 - for round_idx in range(max_rounds): - # 路径 A0 已成功则跳过 A1/A2/A3;否则走兜底链路 - if not direct_url: - last_err_stage = "parse" - # 路径 A1:第三方解析 API(apizero 付费时可用) - logger.info("抖音解析第%d轮开始: url=%s a0_ok=%s", round_idx + 1, page_url, bool(direct_url)) - try: - direct_url, meta_duration = _parser_extract_direct_url(page_url) - if direct_url: - logger.info( - "第三方解析 API 成功(第%d轮): url_domain=%s", - round_idx + 1, direct_url.split("/")[2] if "/" in direct_url else "?", - ) - except Exception as exc: # noqa: BLE001 - logger.warning("第三方解析 API 异常(第%d轮): %s", round_idx + 1, exc) - - # 路径 A2:yt-dlp 拿直链(作为兜底,需要 cookie 概率性成功) - if not direct_url: - logger.info( - "第三方解析未拿到直链(第%d轮),尝试 yt-dlp", round_idx + 1 - ) - direct_url, meta_duration = _ytdlp_extract_video_url(page_url, cookiefile=cookiefile) - - # 路径 A3:HTML 直抓(最后兜底,不依赖任何 API Key / cookies) - if not direct_url: - logger.info( - "yt-dlp 未拿到直链(第%d轮),尝试 HTML 直接抓取", round_idx + 1 - ) - try: - direct_url, meta_duration = _html_scrape_direct_url(page_url) - except Exception as exc: # noqa: BLE001 - logger.warning("HTML 直抓异常(第%d轮): %s", round_idx + 1, exc) - - if meta_duration: - duration = meta_duration - _dbg("direct_url", direct_url or "") - - # 路径 B1:直链 → MediaKit 云端 ASR(最快,不下载视频) - if direct_url and mk_client.is_available: - last_err_stage = "asr" - try: - task_id = mk_client.asr_submit(direct_url) - text, mk_duration = mk_client.asr_poll(task_id) - if mk_duration: - duration = mk_duration + # B1:MediaKit 云端 ASR(不下载视频,最快) + if mk_client.is_available: + last_err_stage = "asr" + try: + task_id = mk_client.asr_submit(direct_url) + text, duration = mk_client.asr_poll(task_id) + text = text.strip() + if text: logger.info( - "抖音 MediaKit ASR 成功(第%d轮): url=%s text_len=%d duration=%.1f", - round_idx + 1, page_url, len(text), duration, + "抖音 MediaKit ASR 成功: source=%s text_len=%d duration=%.1f total_time=%.1fs", + result.source, len(text), duration, time.time() - t0, ) - break - except MediaKitError as exc: - logger.warning("MediaKit ASR 失败(第%d轮),回退本地 ASR: %s", round_idx + 1, exc) - text = "" + else: + logger.info("抖音 MediaKit ASR 返回空文本(无旁白/BGM视频)") + except MediaKitError as exc: + logger.warning("MediaKit ASR 失败,回退本地 ASR: %s", exc) + text = "" - # 路径 B2:回退下载 + 本地 ASR - if not text: - last_err_stage = "download" if not direct_url else "asr" - _dbg("fallback", f"download+local_asr round={round_idx+1}") - try: - with tempfile.TemporaryDirectory(prefix="douyin_extract_") as temp_dir: - if direct_url: - _dbg("fallback_via", "direct_url_download") - last_err_stage = "download" - text, dl_duration = _direct_url_download_and_local_asr( - direct_url, page_url, temp_dir - ) - if text: - last_err_stage = "asr" # 下载成功后才是asr阶段 - else: - last_err_stage = "parse" - text, dl_duration = _ytdlp_download_and_local_asr( - page_url, temp_dir, cookiefile=cookiefile - ) - if dl_duration and not duration: - duration = dl_duration - if text: - logger.info( - "抖音本地 ASR 成功(第%d轮): url=%s text_len=%d", - round_idx + 1, page_url, len(text), - ) - break - except HTTPException as exc: - # 可重试错误(502/503/504)短暂等待后重试 - if exc.status_code in (502, 503, 504) and round_idx < max_rounds - 1: - logger.warning( - "本地 ASR 链路返回 %d(第%d轮),将重试: %s", - exc.status_code, round_idx + 1, exc.detail, + # B2:回退下载 + 本地 ASR + if not text: + last_err_stage = "download" + try: + with tempfile.TemporaryDirectory(prefix="douyin_extract_") as temp_dir: + text, dl_duration = _direct_url_download_and_local_asr(direct_url, page_url, temp_dir) + text = (text or "").strip() + if dl_duration and not duration: + duration = dl_duration + if text: + logger.info( + "抖音本地 ASR 成功: source=%s text_len=%d total_time=%.1fs", + result.source, len(text), time.time() - t0, ) - time.sleep(1.2 + round_idx) - continue - raise - except Exception as exc: # noqa: BLE001 - logger.warning("本地 ASR 链路异常(第%d轮): %s", round_idx + 1, exc) - if round_idx < max_rounds - 1: - time.sleep(1.2 + round_idx) - continue - raise + last_err_stage = "asr" + except HTTPException: + raise + except Exception as exc: # noqa: BLE001 + logger.warning("本地 ASR 链路异常: %s", exc) - # 如果拿到 text 了就 break - if text: - break + # ── Phase C:结果判定 & 兜底 ── - # ASR 可能返回空(视频无旁白/只有BGM/环境音),此时用 feed_desc 作为文案兜底 - if feed_desc and not text: - text = feed_desc.strip() - logger.info( - "抖音 ASR 未识别到语音(第%d轮),使用 Feed API desc 作为文案: desc_len=%d", - round_idx + 1, len(text), - ) - break - - # 本轮全链路失败,等一下重试(直链未拿到可能是瞬时限流) - if round_idx < max_rounds - 1: - logger.info("第%d轮全链路失败,等待后重试", round_idx + 1) - time.sleep(1.5) + # ASR 空结果(无旁白视频)→ 使用解析源 desc 兜底 + if not text and feed_desc: + text = feed_desc + logger.info("抖音 ASR 空结果,使用解析源 desc 兜底: desc_len=%d", len(text)) if not text: - # 终极兜底:所有 ASR 路径均失败/返回空时,若 Feed API 拿到了 desc 就直接用 - if feed_desc: - text = feed_desc.strip() - logger.info("抖音所有 ASR 路径均失败,最终使用 Feed desc 作为文案: desc_len=%d", len(text)) - else: - # 根据失败阶段返回具体错误信息 - stage_msg = { - "parse": "抖音视频链接解析失败,请检查链接是否正确或稍后重试", - "download": "抖音视频下载失败,请检查网络或稍后重试", - "asr": "抖音语音识别失败,请稍后重试或手动输入文案", - } - user_msg = stage_msg.get(last_err_stage, "抖音链接解析暂时不可用,请稍后重试或手动输入文案") - logger.warning("抖音文案提取全部失败: url=%s stage=%s", page_url, last_err_stage) - raise HTTPException( - status_code=status.HTTP_503_SERVICE_UNAVAILABLE, - detail=user_msg, - ) + stage_msg = { + "parse": "抖音视频链接解析失败,请检查链接是否正确或稍后重试", + "download": "抖音视频下载失败,请检查网络或稍后重试", + "asr": "抖音语音识别失败,请稍后重试或手动输入文案", + } + user_msg = stage_msg.get(last_err_stage, "抖音链接解析暂时不可用,请稍后重试或手动输入文案") + if _DOUYIN_DEBUG_ERRORS: + user_msg = user_msg + f" [debug: stage={last_err_stage} source={result.source}]" + logger.warning("抖音文案提取失败: url=%s stage=%s source=%s", page_url, last_err_stage, result.source) + raise HTTPException(status_code=status.HTTP_503_SERVICE_UNAVAILABLE, detail=user_msg) return ExtractFromDouyinResponse( text=text, diff --git a/apps/api/app/services/douyin_resolver.py b/apps/api/app/services/douyin_resolver.py new file mode 100644 index 000000000..ae58005b1 --- /dev/null +++ b/apps/api/app/services/douyin_resolver.py @@ -0,0 +1,275 @@ +"""抖音视频解析多源轮询服务。 + +优先级(P0 最高): + P0: App Feed API 直连(零成本,不用 API Key,当前最稳定) + P1: TikHub API(付费 $0.001/次起,稳定) + P2: apizero.cn 极数本源(按量付费,国内延迟低) + P3: GoDownloader / RapidAPI(免费层 60 次/分兜底) + +任一源成功即返回 MP4 直链 + 标题/文案;所有源均失败时返回 None。 +每个解析源独立超时(5-8s),总耗时不超过所有源超时之和(实际快速失败时远小于此)。 +未配置 API Key 的源自动跳过;无任何 Key 时 P0 仍可使用。 +""" + +from __future__ import annotations + +import logging +import os +import re +import time +from dataclasses import dataclass +from typing import Optional + +import httpx + +logger = logging.getLogger(__name__) + +# ── API Keys from env ────────────────────────────────────────────────── +TIKHUB_API_KEY = os.environ.get("TIKHUB_API_KEY", "").strip() +APIZERO_API_KEY = os.environ.get("APIZERO_API_KEY", "").strip() +RAPIDAPI_KEY = os.environ.get("RAPIDAPI_KEY", "").strip() + +# ── Timeouts (seconds) ──────────────────────────────────────────────── +_TIMEOUT_APP_FEED = 10 +_TIMEOUT_TIKHUB = 5 +_TIMEOUT_APIZERO = 5 +_TIMEOUT_GODOWNLOADER = 8 + + +@dataclass +class ResolveResult: + video_url: str # MP4 直链;图文视频时为空字符串 + desc: str # 视频标题/描述文案 + source: str # 解析源名称,用于日志/metrics + + +# ── URL preprocessing ───────────────────────────────────────────────── +_AWEME_ID_RE = re.compile( + r"(?:douyin\.com/(?:video|note)/|iesdouyin\.com/share/video/|aweme_id=)(\d{15,25})", + re.IGNORECASE, +) + + +def _extract_url_from_text(text: str) -> str: + """从任意分享文本中提取首个 http(s) URL。""" + if not text: + return "" + m = re.search(r"https?://\S+", text) + return m.group(0).rstrip("。,!?!?,,;;\"'))】") if m else "" + + +def _canonicalize_url(url: str, timeout: int = 8) -> str: + """跟随 v.douyin.com 短链 302 重定向,返回完整 URL。失败时返回原 URL。""" + if "v.douyin.com" not in url and "iesdouyin.com" not in url: + return url + try: + with httpx.Client(timeout=timeout, follow_redirects=True, verify=False, + headers={"User-Agent": "Mozilla/5.0"}) as c: + resp = c.get(url) + return str(resp.url) + except Exception as exc: + logger.debug("短链解析失败: %s (%s)", url, exc) + return url + + +# ── Provider P0: App Feed API (零成本直连) ──────────────────────────── +def _resolve_app_feed(url: str, timeout: int = _TIMEOUT_APP_FEED) -> Optional[ResolveResult]: + """抖音 Android App Feed API 直连 — 零依赖、无需 Key、目前最稳定。""" + from packages.douyin_parser import fetch_douyin_video_url + + video_url, desc = fetch_douyin_video_url(url, timeout=timeout, max_retries=2) + if video_url: + return ResolveResult(video_url=video_url, desc=desc or "", source="app_feed") + if desc: + # 图文视频:video_url 为 None 但 desc 可用 + return ResolveResult(video_url="", desc=desc, source="app_feed_image") + return None + + +# ── Provider P1: TikHub ─────────────────────────────────────────────── +def _resolve_tikhub(url: str, api_key: str, timeout: int = _TIMEOUT_TIKHUB) -> Optional[ResolveResult]: + """TikHub API: https://api.tikhub.io/ + 两步:get_aweme_id → fetch_one_video + """ + if not api_key: + return None + headers = {"Authorization": f"Bearer {api_key}"} + aweme_id = _AWEME_ID_RE.search(url or "") + aweme_id = aweme_id.group(1) if aweme_id else None + + if not aweme_id: + try: + with httpx.Client(timeout=timeout, verify=False) as c: + r = c.get( + "https://api.tikhub.io/api/v1/douyin/web/get_aweme_id", + headers=headers, params={"url": url}, + ) + data = r.json() + aweme_id = (data.get("data") or {}).get("aweme_id") + except Exception as exc: + logger.warning("TikHub get_aweme_id 失败: %s", exc) + return None + if not aweme_id: + return None + + try: + with httpx.Client(timeout=timeout, verify=False) as c: + r = c.get( + "https://api.tikhub.io/api/v1/douyin/app/v3/fetch_one_video", + headers=headers, params={"aweme_id": aweme_id}, + ) + data = r.json() + video = ((data.get("data") or {}).get("video") or {}) + urls = [] + for k in ("download_addr", "play_addr_h264", "play_addr"): + urls = (video.get(k) or {}).get("url_list") or [] + if urls: + break + if not urls: + # bit_rate 兜底 + for br in video.get("bit_rate") or []: + urls = (br.get("play_addr") or {}).get("url_list") or [] + if urls: + break + if not urls: + return None + # 优先 CDN 直链 + video_url = urls[0] + for u in urls: + if any(h in u for h in ("douyinvod.com", "bytecdn.com", "365yg.com")): + video_url = u + break + desc = (data.get("data") or {}).get("desc", "") + # 检测图文 + images = (data.get("data") or {}).get("images") or [] + if images and not any(h in video_url for h in ("douyinvod.com", "bytecdn.com", "amemv.com")): + # 图文且无视频直链 + if desc: + return ResolveResult(video_url="", desc=desc, source="tikhub_image") + return None + return ResolveResult(video_url=video_url, desc=desc or "", source="tikhub") + except Exception as exc: + logger.warning("TikHub fetch_one_video 失败: %s", exc) + return None + + +# ── Provider P2: apizero.cn ────────────────────────────────────────── +def _resolve_apizero(url: str, api_key: str, timeout: int = _TIMEOUT_APIZERO) -> Optional[ResolveResult]: + """apizero.cn 极数本源: https://v1.apizero.cn/api/video-parse?url=...&flat=2""" + if not api_key: + return None + headers = {"Authorization": f"Bearer {api_key}"} + try: + with httpx.Client(timeout=timeout, verify=False) as c: + r = c.get( + "https://v1.apizero.cn/api/video-parse", + headers=headers, params={"url": url, "flat": 2}, + ) + data = r.json() + d = data.get("data") or {} + video_list = d.get("video_list") or [] + if not video_list: + return None + video_url = video_list[0].get("url", "") + desc = d.get("title", "") or d.get("desc", "") or d.get("author", "") + if not video_url: + return None + return ResolveResult(video_url=video_url, desc=desc, source="apizero") + except Exception as exc: + logger.warning("apizero 解析失败: %s", exc) + return None + + +# ── Provider P3: GoDownloader (RapidAPI) ────────────────────────────── +def _resolve_godownloader(url: str, api_key: str, timeout: int = _TIMEOUT_GODOWNLOADER) -> Optional[ResolveResult]: + """GoDownloader via RapidAPI""" + if not api_key: + return None + headers = { + "X-RapidAPI-Key": api_key, + "X-RapidAPI-Host": "tiktok-download-video-no-watermark.p.rapidapi.com", + } + try: + with httpx.Client(timeout=timeout, verify=False) as c: + r = c.get( + "https://tiktok-download-video-no-watermark.p.rapidapi.com/auto", + headers=headers, params={"url": url}, + ) + data = r.json() + video_url = ( + (data.get("video") or {}).get("url") + or data.get("video_no_watermark") + or data.get("nwm_video_url") + or "" + ) + desc = data.get("title", "") or data.get("desc", "") + if not video_url: + return None + return ResolveResult(video_url=video_url, desc=desc, source="godownloader") + except Exception as exc: + logger.warning("GoDownloader 解析失败: %s", exc) + return None + + +# ── Main API ────────────────────────────────────────────────────────── +def resolve_douyin_video(page_url: str) -> Optional[ResolveResult]: + """按 P0→P1→P2→P3 顺序轮询解析抖音视频。 + + Args: + page_url: 抖音 URL 或含 URL 的分享文本。 + + Returns: + ResolveResult 或 None(所有源均失败)。 + 图文视频时 video_url 为空字符串、desc 为文案。 + """ + url = _extract_url_from_text(page_url) or page_url + url = _canonicalize_url(url) + + providers = [ + ("app_feed", lambda: _resolve_app_feed(url)), + ("tikhub", lambda: _resolve_tikhub(url, TIKHUB_API_KEY)), + ("apizero", lambda: _resolve_apizero(url, APIZERO_API_KEY)), + ("godownloader", lambda: _resolve_godownloader(url, RAPIDAPI_KEY)), + ] + + enabled_count = 0 + for name, fn in providers: + if name == "tikhub" and not TIKHUB_API_KEY: + continue + if name == "apizero" and not APIZERO_API_KEY: + continue + if name == "godownloader" and not RAPIDAPI_KEY: + continue + enabled_count += 1 + t0 = time.time() + try: + result = fn() + elapsed = time.time() - t0 + if result: + domain = result.video_url.split("/")[2] if result.video_url and "/" in result.video_url else "(image)" + logger.info( + "抖音解析成功: source=%s url_domain=%s desc_len=%d time=%.2fs", + result.source, domain, len(result.desc), elapsed, + ) + return result + logger.debug("解析源 %s 返回空 (%.2fs)", name, elapsed) + except Exception as exc: + logger.warning("解析源 %s 异常 (%.2fs): %s", name, time.time() - t0, exc) + + if enabled_count == 0: + logger.error("无任何抖音解析源可用:请检查 App Feed API 网络连通性") + else: + logger.warning("所有 %d 个抖音解析源均失败: url=%s", enabled_count, url) + return None + + +def available_providers() -> list[str]: + """返回当前可用的解析源列表(用于诊断)。""" + provs = ["app_feed"] + if TIKHUB_API_KEY: + provs.append("tikhub") + if APIZERO_API_KEY: + provs.append("apizero") + if RAPIDAPI_KEY: + provs.append("godownloader") + return provs