f1621ace9f
CI/CD Pipeline / Check if frontend-only change (push) Has been skipped
CI/CD Pipeline / PR Build API Image (push) Has been skipped
CI/CD Pipeline / PR Build Web Image (push) Has been skipped
CI/CD Pipeline / Dedup Check - skip PR tests when covered by push pipeline (push) Successful in 2s
CI/CD Pipeline / Check push changed paths (push) Successful in 5s
CI/CD Pipeline / Frontend Lint (push) Has been skipped
CI/CD Pipeline / PR Build Worker Image (push) Has been skipped
CI/CD Pipeline / Build Staging Worker Image (push) Successful in 2m17s
CI/CD Pipeline / Build Staging Web Image (push) Successful in 2m26s
CI/CD Pipeline / Integration Tests (push) Successful in 3m48s
CI/CD Pipeline / Frontend Unit Tests (push) Successful in 4m13s
CI/CD Pipeline / Build Staging API Image (push) Successful in 4m44s
CI/CD Pipeline / Retag skipped Staging API Image (push) Has been skipped
CI/CD Pipeline / Retag skipped Staging Web Image (push) Has been skipped
CI/CD Pipeline / Retag skipped Staging Worker Image (push) Has been skipped
CI/CD Pipeline / Validate - Python (mypy + alembic) (push) Successful in 5m19s
CI/CD Pipeline / Deploy Staging (Watchtower auto-deploy) (push) Successful in 59s
CI/CD Pipeline / Validate - Style (push) Successful in 7m51s
CI/CD Pipeline / ACR Image Cleanup (push) Successful in 2m54s
CI/CD Pipeline / Staging API Integration Tests (push) Successful in 3m44s
CI/CD Pipeline / Staging E2E Tests (push) Failing after 3m58s
CI/CD Pipeline / Unit Tests (push) Successful in 10m42s
CI/CD Pipeline / Validate - Security (push) Successful in 12m25s
CI/CD Pipeline / Build Production API Image (push) Has been skipped
CI/CD Pipeline / Build Production Web Image (push) Has been skipped
CI/CD Pipeline / Build Production Worker Image (push) Has been skipped
CI/CD Pipeline / CI Gate (push) Has been skipped
CI/CD Pipeline / Deploy Production (push) Has been skipped
CI/CD Pipeline / Canary Release to Production (push) Has been skipped
CI/CD Pipeline / Production Browser E2E (push) Has been skipped
Co-authored-by: xiaoxia <dev@xiaoxiajianji.com> Co-committed-by: xiaoxia <dev@xiaoxiajianji.com>
244 lines
9.6 KiB
Python
244 lines
9.6 KiB
Python
"""抖音视频解析多源轮询服务。
|
||
|
||
优先级(P0 最高):
|
||
P0: App Feed API 直连(零成本,不用 API Key,当前最稳定)
|
||
P1: TikHub API(付费 $0.001/次起,稳定)
|
||
P2: apizero.cn 极数本源(按量付费,国内延迟低)
|
||
|
||
任一源成功即返回 MP4 直链 + 标题/文案;所有源均失败时返回 None。
|
||
每个解析源独立超时(5-10s),总耗时不超过所有源超时之和(实际快速失败时远小于此)。
|
||
未配置 API Key 的源自动跳过;无任何 Key 时 P0 仍可使用。
|
||
"""
|
||
|
||
from __future__ import annotations
|
||
|
||
import logging
|
||
import os
|
||
import re
|
||
import time
|
||
from dataclasses import dataclass
|
||
from typing import Optional
|
||
|
||
import httpx
|
||
|
||
logger = logging.getLogger(__name__)
|
||
|
||
# ── API Keys from env ──────────────────────────────────────────────────
|
||
TIKHUB_API_KEY = os.environ.get("TIKHUB_API_KEY", "").strip()
|
||
APIZERO_API_KEY = os.environ.get("APIZERO_API_KEY", "").strip()
|
||
|
||
# ── Timeouts (seconds) ────────────────────────────────────────────────
|
||
_TIMEOUT_APP_FEED = 12
|
||
_TIMEOUT_TIKHUB = 6
|
||
_TIMEOUT_APIZERO = 6
|
||
|
||
|
||
@dataclass
|
||
class ResolveResult:
|
||
video_url: str # MP4 直链;图文视频时为空字符串
|
||
desc: str # 视频标题/描述文案
|
||
source: str # 解析源名称,用于日志/metrics
|
||
|
||
|
||
# ── URL preprocessing ─────────────────────────────────────────────────
|
||
_AWEME_ID_RE = re.compile(
|
||
r"(?:douyin\.com/(?:video|note)/|iesdouyin\.com/share/video/|aweme_id=)(\d{15,25})",
|
||
re.IGNORECASE,
|
||
)
|
||
|
||
|
||
def _extract_url_from_text(text: str) -> str:
|
||
"""从任意分享文本中提取首个 http(s) URL。"""
|
||
if not text:
|
||
return ""
|
||
m = re.search(r"https?://\S+", text)
|
||
return m.group(0).rstrip("。,!?!?,,;;\"'))】") if m else "" # noqa: B005
|
||
|
||
|
||
def _canonicalize_url(url: str, timeout: int = 8) -> str:
|
||
"""跟随 v.douyin.com 短链 302 重定向,返回完整 URL。失败时返回原 URL。"""
|
||
if "v.douyin.com" not in url and "iesdouyin.com" not in url:
|
||
return url
|
||
try:
|
||
with httpx.Client(
|
||
timeout=timeout, follow_redirects=True, verify=False, headers={"User-Agent": "Mozilla/5.0"}
|
||
) as c:
|
||
resp = c.get(url)
|
||
return str(resp.url)
|
||
except Exception as exc:
|
||
logger.debug("短链解析失败: %s (%s)", url, exc)
|
||
return url
|
||
|
||
|
||
# ── Provider P0: App Feed API (零成本直连) ────────────────────────────
|
||
def _resolve_app_feed(url: str, timeout: int = _TIMEOUT_APP_FEED) -> Optional[ResolveResult]:
|
||
"""抖音 Android App Feed API 直连 — 零依赖、无需 Key、目前最稳定。"""
|
||
from packages.douyin_parser import fetch_douyin_video_url
|
||
|
||
video_url, desc = fetch_douyin_video_url(url, timeout=timeout, max_retries=2)
|
||
if video_url:
|
||
return ResolveResult(video_url=video_url, desc=desc or "", source="app_feed")
|
||
if desc:
|
||
# 图文视频:video_url 为 None 但 desc 可用
|
||
return ResolveResult(video_url="", desc=desc, source="app_feed_image")
|
||
return None
|
||
|
||
|
||
# ── Provider P1: TikHub ───────────────────────────────────────────────
|
||
def _resolve_tikhub(url: str, api_key: str, timeout: int = _TIMEOUT_TIKHUB) -> Optional[ResolveResult]:
|
||
"""TikHub API: https://api.tikhub.io/
|
||
两步:get_aweme_id → fetch_one_video
|
||
"""
|
||
if not api_key:
|
||
return None
|
||
headers = {"Authorization": f"Bearer {api_key}"}
|
||
aweme_id = _AWEME_ID_RE.search(url or "")
|
||
aweme_id = aweme_id.group(1) if aweme_id else None
|
||
|
||
if not aweme_id:
|
||
try:
|
||
with httpx.Client(timeout=timeout, verify=False) as c:
|
||
r = c.get(
|
||
"https://api.tikhub.io/api/v1/douyin/web/get_aweme_id",
|
||
headers=headers,
|
||
params={"url": url},
|
||
)
|
||
data = r.json()
|
||
aweme_id = (data.get("data") or {}).get("aweme_id")
|
||
except Exception as exc:
|
||
logger.warning("TikHub get_aweme_id 失败: %s", exc)
|
||
return None
|
||
if not aweme_id:
|
||
return None
|
||
|
||
try:
|
||
with httpx.Client(timeout=timeout, verify=False) as c:
|
||
r = c.get(
|
||
"https://api.tikhub.io/api/v1/douyin/app/v3/fetch_one_video",
|
||
headers=headers,
|
||
params={"aweme_id": aweme_id},
|
||
)
|
||
data = r.json()
|
||
video = (data.get("data") or {}).get("video") or {}
|
||
urls = []
|
||
for k in ("download_addr", "play_addr_h264", "play_addr"):
|
||
urls = (video.get(k) or {}).get("url_list") or []
|
||
if urls:
|
||
break
|
||
if not urls:
|
||
# bit_rate 兜底
|
||
for br in video.get("bit_rate") or []:
|
||
urls = (br.get("play_addr") or {}).get("url_list") or []
|
||
if urls:
|
||
break
|
||
if not urls:
|
||
return None
|
||
# 优先 CDN 直链
|
||
video_url = urls[0]
|
||
for u in urls:
|
||
if any(h in u for h in ("douyinvod.com", "bytecdn.com", "365yg.com")):
|
||
video_url = u
|
||
break
|
||
desc = (data.get("data") or {}).get("desc", "")
|
||
# 检测图文
|
||
images = (data.get("data") or {}).get("images") or []
|
||
if images and not any(h in video_url for h in ("douyinvod.com", "bytecdn.com", "amemv.com")):
|
||
# 图文且无视频直链
|
||
if desc:
|
||
return ResolveResult(video_url="", desc=desc, source="tikhub_image")
|
||
return None
|
||
return ResolveResult(video_url=video_url, desc=desc or "", source="tikhub")
|
||
except Exception as exc:
|
||
logger.warning("TikHub fetch_one_video 失败: %s", exc)
|
||
return None
|
||
|
||
|
||
# ── Provider P2: apizero.cn ──────────────────────────────────────────
|
||
def _resolve_apizero(url: str, api_key: str, timeout: int = _TIMEOUT_APIZERO) -> Optional[ResolveResult]:
|
||
"""apizero.cn 极数本源: https://v1.apizero.cn/api/video-parse?url=...&flat=2"""
|
||
if not api_key:
|
||
return None
|
||
headers = {"Authorization": f"Bearer {api_key}"}
|
||
try:
|
||
with httpx.Client(timeout=timeout, verify=False) as c:
|
||
r = c.get(
|
||
"https://v1.apizero.cn/api/video-parse",
|
||
headers=headers,
|
||
params={"url": url, "flat": 2},
|
||
)
|
||
data = r.json()
|
||
d = data.get("data") or {}
|
||
video_list = d.get("video_list") or []
|
||
if not video_list:
|
||
return None
|
||
video_url = video_list[0].get("url", "")
|
||
desc = d.get("title", "") or d.get("desc", "") or d.get("author", "")
|
||
if not video_url:
|
||
return None
|
||
return ResolveResult(video_url=video_url, desc=desc, source="apizero")
|
||
except Exception as exc:
|
||
logger.warning("apizero 解析失败: %s", exc)
|
||
return None
|
||
|
||
|
||
# ── Main API ──────────────────────────────────────────────────────────
|
||
def resolve_douyin_video(page_url: str) -> Optional[ResolveResult]:
|
||
"""按 P0→P1→P2 顺序轮询解析抖音视频。
|
||
|
||
Args:
|
||
page_url: 抖音 URL 或含 URL 的分享文本。
|
||
|
||
Returns:
|
||
ResolveResult 或 None(所有源均失败)。
|
||
图文视频时 video_url 为空字符串、desc 为文案。
|
||
"""
|
||
url = _extract_url_from_text(page_url) or page_url
|
||
url = _canonicalize_url(url)
|
||
|
||
providers = [
|
||
("app_feed", lambda: _resolve_app_feed(url)),
|
||
("tikhub", lambda: _resolve_tikhub(url, TIKHUB_API_KEY)),
|
||
("apizero", lambda: _resolve_apizero(url, APIZERO_API_KEY)),
|
||
]
|
||
|
||
enabled_count = 0
|
||
for name, fn in providers:
|
||
if name == "tikhub" and not TIKHUB_API_KEY:
|
||
continue
|
||
if name == "apizero" and not APIZERO_API_KEY:
|
||
continue
|
||
enabled_count += 1
|
||
t0 = time.time()
|
||
try:
|
||
result = fn()
|
||
elapsed = time.time() - t0
|
||
if result:
|
||
domain = result.video_url.split("/")[2] if result.video_url and "/" in result.video_url else "(image)"
|
||
logger.info(
|
||
"抖音解析成功: source=%s url_domain=%s desc_len=%d time=%.2fs",
|
||
result.source,
|
||
domain,
|
||
len(result.desc),
|
||
elapsed,
|
||
)
|
||
return result
|
||
logger.debug("解析源 %s 返回空 (%.2fs)", name, elapsed)
|
||
except Exception as exc:
|
||
logger.warning("解析源 %s 异常 (%.2fs): %s", name, time.time() - t0, exc)
|
||
|
||
if enabled_count == 0:
|
||
logger.error("无任何抖音解析源可用:请检查 App Feed API 网络连通性")
|
||
else:
|
||
logger.warning("所有 %d 个抖音解析源均失败: url=%s", enabled_count, url)
|
||
return None
|
||
|
||
|
||
def available_providers() -> list[str]:
|
||
"""返回当前可用的解析源列表(用于诊断)。"""
|
||
provs = ["app_feed"]
|
||
if TIKHUB_API_KEY:
|
||
provs.append("tikhub")
|
||
if APIZERO_API_KEY:
|
||
provs.append("apizero")
|
||
return provs
|