Files
xiaoxia-saas/apps/api/app/services/douyin_resolver.py
T
xiaoxia f1621ace9f
CI/CD Pipeline / Check if frontend-only change (push) Has been skipped
CI/CD Pipeline / PR Build API Image (push) Has been skipped
CI/CD Pipeline / PR Build Web Image (push) Has been skipped
CI/CD Pipeline / Dedup Check - skip PR tests when covered by push pipeline (push) Successful in 2s
CI/CD Pipeline / Check push changed paths (push) Successful in 5s
CI/CD Pipeline / Frontend Lint (push) Has been skipped
CI/CD Pipeline / PR Build Worker Image (push) Has been skipped
CI/CD Pipeline / Build Staging Worker Image (push) Successful in 2m17s
CI/CD Pipeline / Build Staging Web Image (push) Successful in 2m26s
CI/CD Pipeline / Integration Tests (push) Successful in 3m48s
CI/CD Pipeline / Frontend Unit Tests (push) Successful in 4m13s
CI/CD Pipeline / Build Staging API Image (push) Successful in 4m44s
CI/CD Pipeline / Retag skipped Staging API Image (push) Has been skipped
CI/CD Pipeline / Retag skipped Staging Web Image (push) Has been skipped
CI/CD Pipeline / Retag skipped Staging Worker Image (push) Has been skipped
CI/CD Pipeline / Validate - Python (mypy + alembic) (push) Successful in 5m19s
CI/CD Pipeline / Deploy Staging (Watchtower auto-deploy) (push) Successful in 59s
CI/CD Pipeline / Validate - Style (push) Successful in 7m51s
CI/CD Pipeline / ACR Image Cleanup (push) Successful in 2m54s
CI/CD Pipeline / Staging API Integration Tests (push) Successful in 3m44s
CI/CD Pipeline / Staging E2E Tests (push) Failing after 3m58s
CI/CD Pipeline / Unit Tests (push) Successful in 10m42s
CI/CD Pipeline / Validate - Security (push) Successful in 12m25s
CI/CD Pipeline / Build Production API Image (push) Has been skipped
CI/CD Pipeline / Build Production Web Image (push) Has been skipped
CI/CD Pipeline / Build Production Worker Image (push) Has been skipped
CI/CD Pipeline / CI Gate (push) Has been skipped
CI/CD Pipeline / Deploy Production (push) Has been skipped
CI/CD Pipeline / Canary Release to Production (push) Has been skipped
CI/CD Pipeline / Production Browser E2E (push) Has been skipped
feat(#1970): 素材原子化切片 P1 - 数据层/切片逻辑/原子片段级选片 (#1974)
Co-authored-by: xiaoxia <dev@xiaoxiajianji.com>
Co-committed-by: xiaoxia <dev@xiaoxiajianji.com>
2026-09-18 03:57:07 +08:00

244 lines
9.6 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
"""抖音视频解析多源轮询服务。
优先级(P0 最高):
P0: App Feed API 直连(零成本,不用 API Key,当前最稳定)
P1: TikHub API(付费 $0.001/次起,稳定)
P2: apizero.cn 极数本源(按量付费,国内延迟低)
任一源成功即返回 MP4 直链 + 标题/文案;所有源均失败时返回 None。
每个解析源独立超时(5-10s),总耗时不超过所有源超时之和(实际快速失败时远小于此)。
未配置 API Key 的源自动跳过;无任何 Key 时 P0 仍可使用。
"""
from __future__ import annotations
import logging
import os
import re
import time
from dataclasses import dataclass
from typing import Optional
import httpx
logger = logging.getLogger(__name__)
# ── API Keys from env ──────────────────────────────────────────────────
TIKHUB_API_KEY = os.environ.get("TIKHUB_API_KEY", "").strip()
APIZERO_API_KEY = os.environ.get("APIZERO_API_KEY", "").strip()
# ── Timeouts (seconds) ────────────────────────────────────────────────
_TIMEOUT_APP_FEED = 12
_TIMEOUT_TIKHUB = 6
_TIMEOUT_APIZERO = 6
@dataclass
class ResolveResult:
video_url: str # MP4 直链;图文视频时为空字符串
desc: str # 视频标题/描述文案
source: str # 解析源名称,用于日志/metrics
# ── URL preprocessing ─────────────────────────────────────────────────
_AWEME_ID_RE = re.compile(
r"(?:douyin\.com/(?:video|note)/|iesdouyin\.com/share/video/|aweme_id=)(\d{15,25})",
re.IGNORECASE,
)
def _extract_url_from_text(text: str) -> str:
"""从任意分享文本中提取首个 http(s) URL。"""
if not text:
return ""
m = re.search(r"https?://\S+", text)
return m.group(0).rstrip("。,!?!?,,;;\"'))】") if m else "" # noqa: B005
def _canonicalize_url(url: str, timeout: int = 8) -> str:
"""跟随 v.douyin.com 短链 302 重定向,返回完整 URL。失败时返回原 URL。"""
if "v.douyin.com" not in url and "iesdouyin.com" not in url:
return url
try:
with httpx.Client(
timeout=timeout, follow_redirects=True, verify=False, headers={"User-Agent": "Mozilla/5.0"}
) as c:
resp = c.get(url)
return str(resp.url)
except Exception as exc:
logger.debug("短链解析失败: %s (%s)", url, exc)
return url
# ── Provider P0: App Feed API (零成本直连) ────────────────────────────
def _resolve_app_feed(url: str, timeout: int = _TIMEOUT_APP_FEED) -> Optional[ResolveResult]:
"""抖音 Android App Feed API 直连 — 零依赖、无需 Key、目前最稳定。"""
from packages.douyin_parser import fetch_douyin_video_url
video_url, desc = fetch_douyin_video_url(url, timeout=timeout, max_retries=2)
if video_url:
return ResolveResult(video_url=video_url, desc=desc or "", source="app_feed")
if desc:
# 图文视频:video_url 为 None 但 desc 可用
return ResolveResult(video_url="", desc=desc, source="app_feed_image")
return None
# ── Provider P1: TikHub ───────────────────────────────────────────────
def _resolve_tikhub(url: str, api_key: str, timeout: int = _TIMEOUT_TIKHUB) -> Optional[ResolveResult]:
"""TikHub API: https://api.tikhub.io/
两步:get_aweme_id → fetch_one_video
"""
if not api_key:
return None
headers = {"Authorization": f"Bearer {api_key}"}
aweme_id = _AWEME_ID_RE.search(url or "")
aweme_id = aweme_id.group(1) if aweme_id else None
if not aweme_id:
try:
with httpx.Client(timeout=timeout, verify=False) as c:
r = c.get(
"https://api.tikhub.io/api/v1/douyin/web/get_aweme_id",
headers=headers,
params={"url": url},
)
data = r.json()
aweme_id = (data.get("data") or {}).get("aweme_id")
except Exception as exc:
logger.warning("TikHub get_aweme_id 失败: %s", exc)
return None
if not aweme_id:
return None
try:
with httpx.Client(timeout=timeout, verify=False) as c:
r = c.get(
"https://api.tikhub.io/api/v1/douyin/app/v3/fetch_one_video",
headers=headers,
params={"aweme_id": aweme_id},
)
data = r.json()
video = (data.get("data") or {}).get("video") or {}
urls = []
for k in ("download_addr", "play_addr_h264", "play_addr"):
urls = (video.get(k) or {}).get("url_list") or []
if urls:
break
if not urls:
# bit_rate 兜底
for br in video.get("bit_rate") or []:
urls = (br.get("play_addr") or {}).get("url_list") or []
if urls:
break
if not urls:
return None
# 优先 CDN 直链
video_url = urls[0]
for u in urls:
if any(h in u for h in ("douyinvod.com", "bytecdn.com", "365yg.com")):
video_url = u
break
desc = (data.get("data") or {}).get("desc", "")
# 检测图文
images = (data.get("data") or {}).get("images") or []
if images and not any(h in video_url for h in ("douyinvod.com", "bytecdn.com", "amemv.com")):
# 图文且无视频直链
if desc:
return ResolveResult(video_url="", desc=desc, source="tikhub_image")
return None
return ResolveResult(video_url=video_url, desc=desc or "", source="tikhub")
except Exception as exc:
logger.warning("TikHub fetch_one_video 失败: %s", exc)
return None
# ── Provider P2: apizero.cn ──────────────────────────────────────────
def _resolve_apizero(url: str, api_key: str, timeout: int = _TIMEOUT_APIZERO) -> Optional[ResolveResult]:
"""apizero.cn 极数本源: https://v1.apizero.cn/api/video-parse?url=...&flat=2"""
if not api_key:
return None
headers = {"Authorization": f"Bearer {api_key}"}
try:
with httpx.Client(timeout=timeout, verify=False) as c:
r = c.get(
"https://v1.apizero.cn/api/video-parse",
headers=headers,
params={"url": url, "flat": 2},
)
data = r.json()
d = data.get("data") or {}
video_list = d.get("video_list") or []
if not video_list:
return None
video_url = video_list[0].get("url", "")
desc = d.get("title", "") or d.get("desc", "") or d.get("author", "")
if not video_url:
return None
return ResolveResult(video_url=video_url, desc=desc, source="apizero")
except Exception as exc:
logger.warning("apizero 解析失败: %s", exc)
return None
# ── Main API ──────────────────────────────────────────────────────────
def resolve_douyin_video(page_url: str) -> Optional[ResolveResult]:
"""按 P0→P1→P2 顺序轮询解析抖音视频。
Args:
page_url: 抖音 URL 或含 URL 的分享文本。
Returns:
ResolveResult 或 None(所有源均失败)。
图文视频时 video_url 为空字符串、desc 为文案。
"""
url = _extract_url_from_text(page_url) or page_url
url = _canonicalize_url(url)
providers = [
("app_feed", lambda: _resolve_app_feed(url)),
("tikhub", lambda: _resolve_tikhub(url, TIKHUB_API_KEY)),
("apizero", lambda: _resolve_apizero(url, APIZERO_API_KEY)),
]
enabled_count = 0
for name, fn in providers:
if name == "tikhub" and not TIKHUB_API_KEY:
continue
if name == "apizero" and not APIZERO_API_KEY:
continue
enabled_count += 1
t0 = time.time()
try:
result = fn()
elapsed = time.time() - t0
if result:
domain = result.video_url.split("/")[2] if result.video_url and "/" in result.video_url else "(image)"
logger.info(
"抖音解析成功: source=%s url_domain=%s desc_len=%d time=%.2fs",
result.source,
domain,
len(result.desc),
elapsed,
)
return result
logger.debug("解析源 %s 返回空 (%.2fs)", name, elapsed)
except Exception as exc:
logger.warning("解析源 %s 异常 (%.2fs): %s", name, time.time() - t0, exc)
if enabled_count == 0:
logger.error("无任何抖音解析源可用:请检查 App Feed API 网络连通性")
else:
logger.warning("所有 %d 个抖音解析源均失败: url=%s", enabled_count, url)
return None
def available_providers() -> list[str]:
"""返回当前可用的解析源列表(用于诊断)。"""
provs = ["app_feed"]
if TIKHUB_API_KEY:
provs.append("tikhub")
if APIZERO_API_KEY:
provs.append("apizero")
return provs