Files
xiaoxia-saas/apps/api/app/api/routes/scripts_ai.py
T
xiaoxia ea4b74216f
CI/CD Pipeline / Check if frontend-only change (push) Has been skipped
CI/CD Pipeline / PR Build API Image (push) Has been skipped
CI/CD Pipeline / PR Build Web Image (push) Has been skipped
CI/CD Pipeline / PR Build Worker Image (push) Has been skipped
CI/CD Pipeline / Validate - Style (push) Has been cancelled
CI/CD Pipeline / Validate - Security (push) Has been cancelled
CI/CD Pipeline / Validate - Python (mypy + alembic) (push) Has been cancelled
CI/CD Pipeline / Unit Tests (push) Has been cancelled
CI/CD Pipeline / Integration Tests (push) Has been cancelled
CI/CD Pipeline / Frontend Lint (push) Has been cancelled
CI/CD Pipeline / Frontend Unit Tests (push) Has been cancelled
CI/CD Pipeline / Build Staging API Image (push) Has been cancelled
CI/CD Pipeline / Build Staging Web Image (push) Has been cancelled
CI/CD Pipeline / Build Staging Worker Image (push) Has been cancelled
CI/CD Pipeline / Retag skipped Staging API Image (push) Has been cancelled
CI/CD Pipeline / Retag skipped Staging Web Image (push) Has been cancelled
CI/CD Pipeline / Retag skipped Staging Worker Image (push) Has been cancelled
CI/CD Pipeline / Deploy Staging (Watchtower auto-deploy) (push) Has been cancelled
CI/CD Pipeline / Staging E2E Tests (push) Has been cancelled
CI/CD Pipeline / Staging API Integration Tests (push) Has been cancelled
CI/CD Pipeline / Build Production API Image (push) Has been cancelled
CI/CD Pipeline / Build Production Web Image (push) Has been cancelled
CI/CD Pipeline / Build Production Worker Image (push) Has been cancelled
CI/CD Pipeline / Deploy Production (push) Has been cancelled
CI/CD Pipeline / Production Browser E2E (push) Has been cancelled
CI/CD Pipeline / ACR Image Cleanup (push) Has been cancelled
CI/CD Pipeline / Canary Release to Production (push) Has been cancelled
CI/CD Pipeline / CI Gate (push) Has been cancelled
CI/CD Pipeline / Check push changed paths (push) Has been cancelled
CI/CD Pipeline / Dedup Check - skip PR tests when covered by push pipeline (push) Has been cancelled
fix(douyin): 移除伪造s_v_web_id(根因:导致yt-dlp 100%失败)+apizero优先+staging默认debug+verify=False
2026-09-17 15:06:48 +08:00

1092 lines
44 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
"""Scripts AI 能力路由 — Issue #1893/#1963.
三个 AI 工具接口(均挂载在 /api/v1/scripts 前缀下):
- POST /extract-from-douyin 从抖音视频提取文案
- 入口自动从分享文本中正则提取 http(s) URL,兼容 "复制链接" 粘贴场景
- yt-dlp 仅解析视频元信息(download=False)拿无水印直链,避免整段下载
- 优先走火山 MediaKit ASR(asr-subtitles),配置了 MEDIAKIT_API_KEY 即可用
- MediaKit 不可用/失败时,回退到本地 ASR(下载视频 + transcribe_to_text)
- cookies/ytdlp 均失败时,返回友好 503 不暴露内部错误
- POST /ai-rewrite AI 文案改写(复用豆包 LLM)
- POST /ai-generate-titles AI 标题生成(复用 generate_smart_titles)
"""
from __future__ import annotations
import logging
import os
import re
import tempfile
import time
from urllib.parse import urlparse
from app.auth import AuthenticatedUser, get_current_user
from app.dependencies import get_db_session
from app.schemas.scripts_ai import (
AiGenerateTitlesRequest,
AiGenerateTitlesResponse,
AiRewriteRequest,
AiRewriteResponse,
ExtractFromDouyinRequest,
ExtractFromDouyinResponse,
)
from app.services.mediakit_client import (
MediaKitClient,
MediaKitError,
get_mediakit_client,
)
from app.services.script_asr_service import (
ASRNotConfiguredError,
ASRTranscriptionError,
transcribe_to_text,
)
from fastapi import APIRouter, Depends, HTTPException, status
from sqlalchemy.orm import Session
from packages.middleware.points_gate import points_gate
from packages.shared.ai_client import get_doubao_client
logger = logging.getLogger(__name__)
router = APIRouter()
DOUYIN_COOKIES_FILE = os.environ.get(
"DOUYIN_COOKIES_FILE",
"/app/configs/douyin_cookies.txt",
)
DOUYIN_COOKIES_FILE_BAKED = "/app/configs/douyin_cookies_default.txt"
_COOKIES_ERROR_KEYWORDS = (
"fresh cookies",
"cookies (not necessarily logged in)",
"cookies are needed",
"need cookies",
"cookie is expired",
"login required",
"sign in to continue",
"未登录",
"需要登录",
"cookies过期",
)
_TAIL_PUNCT = ".,;:!?,。;:!?))]》" + chr(34) + chr(39) + "<>"
def _resolve_cookies_file():
for p in (DOUYIN_COOKIES_FILE, DOUYIN_COOKIES_FILE_BAKED):
try:
if p and os.path.isfile(p) and os.path.getsize(p) > 200:
return p
except OSError:
continue
return None
def _dbg(key, val):
logger.debug("douyin_extract %s=%s", key, str(val)[:200])
def _is_cookies_related_error(msg):
low = msg.lower()
return any(kw in low for kw in _COOKIES_ERROR_KEYWORDS)
_cf = _resolve_cookies_file()
if _cf:
logger.info("抖音 cookies 文件已加载: %s (%d bytes)", _cf, os.path.getsize(_cf))
else:
logger.warning(
"抖音 cookies 文件未找到或无效: path=%s baked=%s",
DOUYIN_COOKIES_FILE,
DOUYIN_COOKIES_FILE_BAKED,
)
_DOUYIN_DEBUG_ERRORS = os.environ.get("DOUYIN_DEBUG_ERRORS", "").lower() in (
"1",
"true",
"yes",
) or os.environ.get("APP_ENV", "").lower() in ("staging", "dev", "development", "test")
_URL_EXTRACT_RE = re.compile(r"https?://\S+", re.IGNORECASE)
_DOUYIN_HOST_RE = re.compile(
r"(^|\.)(douyin\.com|iesdouyin\.com|amemv\.com)$",
re.IGNORECASE,
)
_ANY_SCHEME_RE = re.compile(r"^[a-z][a-z0-9+.-]*://\S+", re.IGNORECASE)
def _extract_url_from_text(raw):
if not raw:
return None
m = _URL_EXTRACT_RE.search(raw)
if m:
return m.group(0).rstrip(_TAIL_PUNCT)
short = re.search(
r"(?:^|(?<![a-z0-9/:]))((?:v|www)\.douyin\.com/\S+|douyin\.com/(?:video|note)/\S+)",
raw,
re.IGNORECASE,
)
if short:
return "https://" + short.group(1).rstrip(_TAIL_PUNCT)
return None
def _extract_and_validate_douyin_url(raw_input):
raw = (raw_input or "").strip()
if not raw:
raise HTTPException(status_code=status.HTTP_400_BAD_REQUEST, detail="链接不能为空")
url = _extract_url_from_text(raw)
if not url:
# 含非 http(s) 的 scheme 前缀(如 ftp://、file:// 等)→ 协议不支持
if _ANY_SCHEME_RE.search(raw):
raise HTTPException(
status_code=status.HTTP_400_BAD_REQUEST,
detail="无效的抖音链接,仅支持 http(s) 协议",
)
# 裸域名兜底:在去除 scheme 的情况下匹配 douyin 域名
short = re.search(
r"(?:^|(?<![a-z0-9]))((?:v|www)\.douyin\.com/\S+|douyin\.com/(?:video|note)/\S+)",
raw,
re.IGNORECASE,
)
if short:
url = "https://" + short.group(1).rstrip(_TAIL_PUNCT)
else:
raise HTTPException(
status_code=status.HTTP_400_BAD_REQUEST,
detail="未在输入中找到有效抖音链接,请粘贴包含 v.douyin.com 或 www.douyin.com 的分享文本",
)
if not re.match(r"^https?://", url, re.IGNORECASE):
url = "https://" + url
try:
parsed = urlparse(url)
host = parsed.hostname or ""
scheme = (parsed.scheme or "").lower()
except Exception:
host = ""
scheme = ""
if scheme not in ("http", "https"):
raise HTTPException(
status_code=status.HTTP_400_BAD_REQUEST,
detail="无效的抖音链接,仅支持 http(s) 协议",
)
if not _DOUYIN_HOST_RE.search(host):
raise HTTPException(
status_code=status.HTTP_400_BAD_REQUEST,
detail="无效的抖音链接,仅支持 douyin.com 域名(v.douyin.com 短链或 www.douyin.com 长链)",
)
return url
# ── MediaKitClient ASR 扩展(monkey patch) ────────────────────────────
#
# app/services/mediakit_client.py 已提供:
# self._api_key / self._base_url / self._timeout / self._headers() / self.is_available
# 这里只补两个通用 JSON helper 和 ASR 提交/轮询方法。
def _mk_post_json(self, path, payload):
import httpx
if not self.is_available:
raise MediaKitError("MediaKit API Key 未配置", code="NotConfigured")
url = self._base_url + path
try:
with httpx.Client(timeout=self._timeout) as http:
resp = http.post(url, headers=self._headers(), json=payload)
resp.raise_for_status()
data = resp.json()
except httpx.TimeoutException as exc:
raise MediaKitError("MediaKit API 超时 (%ss)" % self._timeout, code="Timeout") from exc
except httpx.HTTPStatusError as exc:
raise MediaKitError(
"MediaKit API HTTP %s: %s" % (exc.response.status_code, exc.response.text[:300]),
code="HttpError",
) from exc
except httpx.RequestError as exc:
raise MediaKitError("MediaKit API 网络错误: %s" % exc, code="NetworkError") from exc
# 同步错误(提交参数错误等):success=false 且含 error
if data.get("success") is False and data.get("error"):
err = data["error"] if isinstance(data["error"], dict) else {"message": str(data["error"])}
raise MediaKitError(
err.get("message", "请求失败"),
code=err.get("code", "RequestFailed"),
)
return data
def _mk_get_json(self, path):
import httpx
if not self.is_available:
raise MediaKitError("MediaKit API Key 未配置", code="NotConfigured")
url = self._base_url + path
try:
with httpx.Client(timeout=self._timeout) as http:
resp = http.get(url, headers=self._headers())
resp.raise_for_status()
return resp.json()
except httpx.TimeoutException as exc:
raise MediaKitError("MediaKit API 超时 (%ss)" % self._timeout, code="Timeout") from exc
except httpx.HTTPStatusError as exc:
raise MediaKitError(
"MediaKit API HTTP %s: %s" % (exc.response.status_code, exc.response.text[:300]),
code="HttpError",
) from exc
except httpx.RequestError as exc:
raise MediaKitError("MediaKit API 网络错误: %s" % exc, code="NetworkError") from exc
def _mediakit_asr_submit(self, video_url):
"""提交语音转字幕任务(POST /tools/asr-subtitles)。返回 task_id。"""
data = self._post_json(
"/tools/asr-subtitles",
{"video_url": video_url, "language": "cmn-Hans-CN"},
)
task_id = data.get("task_id")
if not task_id:
raise MediaKitError("MediaKit ASR 提交响应缺少 task_id: %s" % str(data)[:200])
return task_id
def _mediakit_asr_poll(self, task_id, poll_interval=2.0, max_attempts=90):
"""轮询 ASR 任务直到 completed/failed。返回 (text, duration)。"""
for attempt in range(max_attempts):
time.sleep(poll_interval)
try:
data = self._get_json("/tasks/" + task_id)
except MediaKitError as exc:
# 瞬时网络/超时可重试
if attempt < max_attempts - 1 and getattr(exc, "code", "") in ("Timeout", "NetworkError"):
logger.warning("MediaKit ASR 轮询异常(第%d次),将重试: %s", attempt + 1, exc)
continue
raise
st = data.get("status")
if st in ("completed", "success"):
result = data.get("result") or {}
subs = result.get("subtitles") or []
text = "".join(s.get("subtitle_text", "") for s in subs if isinstance(s, dict))
duration = float(result.get("duration") or 0.0)
return text.strip(), duration
if st == "failed":
err = data.get("error")
if isinstance(err, dict):
msg = err.get("message") or "unknown"
code = err.get("code") or "TaskFailed"
elif isinstance(err, str):
msg, code = err, "TaskFailed"
else:
msg, code = "unknown", "TaskFailed"
raise MediaKitError("MediaKit ASR 任务失败: %s" % msg, code=code)
# running/pending/queued: continue
raise MediaKitError(
"MediaKit ASR 超时(%ss 未完成)" % int(poll_interval * max_attempts),
code="Timeout",
)
# 绑定到类(零侵入,不修改原 mediakit_client.py)
if not hasattr(MediaKitClient, "_post_json"):
MediaKitClient._post_json = _mk_post_json
if not hasattr(MediaKitClient, "_get_json"):
MediaKitClient._get_json = _mk_get_json
if not hasattr(MediaKitClient, "asr_submit"):
MediaKitClient.asr_submit = _mediakit_asr_submit
if not hasattr(MediaKitClient, "asr_poll"):
MediaKitClient.asr_poll = _mediakit_asr_poll
# ── yt-dlp 辅助 ──────────────────────────────────────────────────────
def _fetch_ttwid():
"""从字节跳动 ttwid 注册接口获取 ttwid cookie。
通过 POST https://ttwid.bytedance.com/ttwid/union/register/ 可直接获得,无需浏览器。
失败时回退到访问 douyin.com 主页。
"""
import httpx
# 方式一:直接调 ttwid 注册接口
try:
with httpx.Client(timeout=8, follow_redirects=True, verify=False) as http:
r = http.post(
"https://ttwid.bytedance.com/ttwid/union/register/",
json={
"region": "cn",
"aid": 6383,
"needFid": False,
"service": "www.douyin.com",
"migrate_info": {"ticket": "", "source": "node"},
"cbUrlProtocol": "https",
"union": True,
},
headers={
"User-Agent": (
"Mozilla/5.0 (Windows NT 10.0; Win64; x64) "
"AppleWebKit/537.36 (KHTML, like Gecko) "
"Chrome/128.0.0.0 Safari/537.36"
),
"Content-Type": "application/json",
"Referer": "https://www.douyin.com/",
},
)
ttwid = http.cookies.get("ttwid")
if ttwid and len(ttwid) > 20:
_dbg("ttwid_fetch", "register_ok len=%d" % len(ttwid))
return ttwid
except Exception as exc:
logger.debug("ttwid 注册接口失败: %s", exc)
# 方式二:回退到访问主页
try:
with httpx.Client(timeout=8, follow_redirects=True, verify=False) as http:
for _ in range(2):
try:
http.get(
"https://www.douyin.com/",
headers={
"User-Agent": (
"Mozilla/5.0 (Windows NT 10.0; Win64; x64) "
"AppleWebKit/537.36 (KHTML, like Gecko) "
"Chrome/128.0.0.0 Safari/537.36"
),
"Accept-Language": "zh-CN,zh;q=0.9",
},
)
ttwid = http.cookies.get("ttwid")
if ttwid:
return ttwid
except Exception:
continue
except Exception as exc:
logger.debug("获取 ttwid 失败: %s", exc)
return None
def _generate_synthetic_cookie_file(temp_dir, ttwid=None):
"""生成一份合成 cookie 文件(仅 ttwid,不伪造 s_v_web_id),返回文件路径。
重要:不要伪造 s_v_web_id(如 verify_xxx 随机串)!yt-dlp DouyinIE 在检测到
无效的 s_v_web_id 时会直接抛致命错误("Fresh cookies needed"),而不是走内部
重试路径。只提供 ttwid 时,yt-dlp 会把缺少 s_v_web_id 视为 expected 错误,
内部走兜底路径拿数据,配合多次重试有一定概率成功。
"""
import os
import secrets
import time
if not ttwid:
ttwid = _fetch_ttwid()
if not ttwid:
# 兜底:生成一个格式合法的假 ttwid(至少不会直接报错)
ttwid = "1%7C" + secrets.token_hex(20) + "%7C" + str(int(time.time())) + "%7C" + secrets.token_hex(32)
expiry = int(time.time()) + 86400 * 30 # 30 天
path = os.path.join(temp_dir, f"dy_cookies_{secrets.token_hex(4)}.txt")
with open(path, "w") as f:
f.write("# Netscape HTTP Cookie File\n")
f.write("# This file is generated by yt-dlp. Do not edit.\n\n")
f.write(f".douyin.com\tTRUE\t/\tTRUE\t{expiry}\tttwid\t{ttwid}\n")
# 注意:不写 s_v_web_id!伪造的 verify_xxx 串会导致 yt-dlp 直接失败。
return path
def _ytdlp_try_extract(page_url, cookiefile):
"""单次 yt-dlp 提取,返回 (info_dict, None) 或 (None, error_message)。
显式选择 h264+aac 的 MP4 格式,避免 h265 编码导致 MediaKit ASR 不兼容。
"""
import yt_dlp
opts = {
"quiet": True,
"no_warnings": True,
"noplaylist": True,
"skip_download": True,
"extractor_args": {"douyin": {"webpage_cookie": ""}},
"format": "best[ext=mp4][vcodec^=avc1]/best[ext=mp4][vcodec^=h264]/best[ext=mp4]/best",
"http_headers": {
"User-Agent": (
"Mozilla/5.0 (Windows NT 10.0; Win64; x64) "
"AppleWebKit/537.36 (KHTML, like Gecko) "
"Chrome/128.0.0.0 Safari/537.36"
),
"Referer": "https://www.douyin.com/",
"Accept-Language": "zh-CN,zh;q=0.9,en;q=0.8",
},
"socket_timeout": 15,
"retries": 1,
}
if cookiefile:
opts["cookiefile"] = cookiefile
try:
with yt_dlp.YoutubeDL(opts) as ydl:
info = ydl.extract_info(page_url, download=False)
return info, None
except Exception as exc:
return None, str(exc)
def _ytdlp_extract_video_url(page_url, cookiefile=None, max_retries=10):
"""yt-dlp 提取视频直链,支持合成 cookie 多次重试。
优先使用传入的 cookiefile;失败后自动生成合成 cookie 重试 max_retries 次。
每次重试使用新的 s_v_web_id,每3次刷新 ttwid,实测约 30% 单次成功率,10次≈97%。
"""
# 优先使用现有 cookie 文件
if cookiefile and os.path.exists(cookiefile):
info, err = _ytdlp_try_extract(page_url, cookiefile)
if info:
vurl = info.get("url")
return (vurl, float(info.get("duration") or 0.0)) if vurl else (None, 0.0)
logger.info("提供的 cookie 文件解析失败: %s", str(err)[:150])
# 合成 cookie 重试
import tempfile
import time as _time
ttwid = _fetch_ttwid()
logger.info("ytdlp 直链提取开始: page_url=%s ttwid_ok=%s max_retries=%d", page_url, bool(ttwid), max_retries)
with tempfile.TemporaryDirectory(prefix="dy_ytdlp_cookies_") as tmpdir:
for attempt in range(max_retries):
if attempt % 3 == 0:
ttwid = _fetch_ttwid() or ttwid
gen_cookie = _generate_synthetic_cookie_file(tmpdir, ttwid=ttwid)
info, err = _ytdlp_try_extract(page_url, gen_cookie)
if info:
vurl = info.get("url")
logger.info("yt-dlp 合成 cookie 第%d次尝试成功: vurl_domain=%s", attempt + 1, vurl.split("/")[2] if vurl and "/" in vurl else "?")
return (vurl, float(info.get("duration") or 0.0)) if vurl else (None, 0.0)
logger.info("yt-dlp 合成 cookie 第%d次尝试失败: %s", attempt + 1, str(err)[:150])
_time.sleep(0.4)
logger.info("yt-dlp 合成 cookie 全部 %d 次重试均失败", max_retries)
return None, 0.0
_PARSER_APIS = [
# 第三方抖音无水印解析 API(cookies 过期/yt-dlp 反爬升级时的兜底)
# 每项 (url_template, name, auth_header_or_None)
# url 中 {url} 会被替换为 URL-encoded 的抖音链接
("https://v1.apizero.cn/api/video-parse?url={url}&flat=1", "apizero", None),
]
# 可选:apizero API Key(登录后免费额度 5 次/天,付费更高额度)
_APIZERO_API_KEY = os.environ.get("APIZERO_API_KEY", "").strip()
_PARSER_MAX_RETRIES = 3
_PARSER_RETRY_DELAY = 0.8 # 秒
_PARSER_RATE_LIMIT_CODES = {4030, 4290, 429, 10004, 10008} # 各家限流 code
def _build_parser_headers(name):
"""为解析 API 构造请求头,包含可选 API Key。"""
base = {
"User-Agent": (
"Mozilla/5.0 (Windows NT 10.0; Win64; x64) "
"AppleWebKit/537.36 (KHTML, like Gecko) "
"Chrome/128.0.0.0 Safari/537.36"
),
"Accept": "application/json, text/plain, */*",
"Accept-Language": "zh-CN,zh;q=0.9,en;q=0.8",
}
if name == "apizero" and _APIZERO_API_KEY:
# apizero 推荐用 Authorization: Bearer <key>,同时兼容 query key
base["Authorization"] = f"Bearer {_APIZERO_API_KEY}"
return base
def _build_parser_url(api_tpl, name, page_url_encoded):
url = api_tpl.format(url=page_url_encoded)
if name == "apizero" and _APIZERO_API_KEY:
sep = "&" if "?" in url else "?"
url = f"{url}{sep}key={_APIZERO_API_KEY}"
return url
def _extract_video_url_from_parser_response(name, data):
"""从解析 API 的 JSON 响应中提取 (direct_url, duration)。兼容多种响应结构。"""
if not isinstance(data, dict):
return None, 0.0
# 统一成功判定:code == 0 或 status == "success"
code = data.get("code")
is_success = code == 0 or code == 200 or data.get("success") is True or data.get("status") == "success"
if not is_success:
msg = data.get("msg") or data.get("message") or str(code)
# 返回 msg 供调用方判断是否限流
return None, 0.0, str(msg), code
vd = data.get("data") or {}
if not isinstance(vd, dict):
return None, 0.0, "bad data", -1
vlist = vd.get("video_list") or []
duration = float(vd.get("duration") or vd.get("duration_ms", 0) or 0.0)
if duration > 1000: # ms → s
duration = duration / 1000.0
direct_url = None
chosen_quality = -1
quality_priority = ["原画", "1080p", "720p", "540p", "480p"]
for prio, keyword in enumerate(quality_priority):
for v in vlist:
if not isinstance(v, dict) or not v.get("url"):
continue
if keyword in str(v.get("quality", "")):
if prio < chosen_quality or chosen_quality == -1:
direct_url = v.get("url")
chosen_quality = prio
break
if direct_url:
break
if not direct_url:
# 兜底:取 data.video_url 顶层字段,或 video_list[0].url
if vd.get("video_url"):
direct_url = vd["video_url"]
elif vlist:
direct_url = vlist[0].get("url")
else:
# 兼容 play/play_url 字段
direct_url = vd.get("play") or vd.get("play_url") or vd.get("url")
if not direct_url:
return None, 0.0, "no video url", -1
return direct_url, duration, None, code
def _parser_extract_direct_url(page_url, timeout=12):
"""使用第三方解析 API 获取抖音无水印直链(yt-dlp 失败时的兜底)。
每个 API 会重试 _PARSER_MAX_RETRIES 次以应对瞬时限流。
返回 (direct_url, duration) 或 (None, 0.0)。
"""
import random
import urllib.parse
import httpx
page_url_encoded = urllib.parse.quote(page_url, safe="")
last_err = None
for api_tpl, name, _auth in _PARSER_APIS:
url = _build_parser_url(api_tpl, name, page_url_encoded)
headers = _build_parser_headers(name)
for attempt in range(_PARSER_MAX_RETRIES):
try:
with httpx.Client(timeout=timeout, follow_redirects=True, verify=False) as http:
resp = http.get(url, headers=headers)
resp.raise_for_status()
data = resp.json()
except Exception as exc: # noqa: BLE001
last_err = str(exc)[:200]
logger.warning(
"第三方解析 %s 第%d次请求失败: %s", name, attempt + 1, exc
)
time.sleep(_PARSER_RETRY_DELAY * (attempt + 1) + random.random() * 0.3)
continue
try:
result = _extract_video_url_from_parser_response(name, data)
if len(result) == 4:
direct_url, duration, err_msg, resp_code = result
else:
direct_url, duration = result
err_msg, resp_code = None, None
if direct_url:
logger.info(
"第三方解析 %s 第%d次成功,直链长度=%d",
name, attempt + 1, len(direct_url),
)
return direct_url, duration
# 判断是否限流,可重试
if resp_code in _PARSER_RATE_LIMIT_CODES or (
err_msg and any(
kw in str(err_msg) for kw in ("次数", "限流", "频率", "用完", "rate limit", "quota")
)
):
logger.info(
"第三方解析 %s 第%d次返回限流: code=%s msg=%s,将重试",
name, attempt + 1, resp_code, err_msg,
)
time.sleep(_PARSER_RETRY_DELAY * (attempt + 1) + random.random() * 0.5)
continue
# 业务错误(如链接无效),直接换下一个 API
logger.warning(
"第三方解析 %s 第%d次返回业务错误: code=%s msg=%s",
name, attempt + 1, resp_code, err_msg,
)
break
except Exception as exc: # noqa: BLE001
last_err = str(exc)[:200]
logger.warning("第三方解析 %s 响应解析失败: %s", name, exc)
break # 响应格式异常,不重试当前 API
logger.warning("所有第三方解析均失败: last_err=%s", last_err)
return None, 0.0
def _direct_url_download_and_local_asr(direct_url, page_url, temp_dir):
"""通过第三方解析得到的直链直接下载 MP4,再做本地 ASR。
返回 (text, duration)。
"""
import os
import httpx
video_path = os.path.join(temp_dir, "video.mp4")
try:
with httpx.Client(timeout=90, follow_redirects=True, verify=False) as http:
with http.stream(
"GET",
direct_url,
headers={
"User-Agent": (
"Mozilla/5.0 (Windows NT 10.0; Win64; x64) "
"AppleWebKit/537.36 (KHTML, like Gecko) "
"Chrome/128.0.0.0 Safari/537.36"
),
"Referer": "https://www.douyin.com/",
"Accept": "*/*",
"Accept-Language": "zh-CN,zh;q=0.9",
},
) as resp:
resp.raise_for_status()
downloaded = 0
with open(video_path, "wb") as f:
for chunk in resp.iter_bytes(chunk_size=65536):
if chunk:
f.write(chunk)
downloaded += len(chunk)
if downloaded == 0:
raise HTTPException(status_code=status.HTTP_502_BAD_GATEWAY, detail="直链下载为空")
except HTTPException:
raise
except httpx.TimeoutException:
logger.warning("直链下载超时: %s", page_url)
raise HTTPException(status_code=status.HTTP_504_GATEWAY_TIMEOUT, detail="视频下载超时,请稍后重试")
except Exception as exc: # noqa: BLE001
logger.exception("直链下载失败: url=%s err=%s", page_url, exc)
raise HTTPException(status_code=status.HTTP_502_BAD_GATEWAY, detail="视频下载失败: " + str(exc)[:200]) from exc
try:
text = transcribe_to_text(video_path)
return text.strip(), 0.0
except ASRNotConfiguredError as exc:
raise HTTPException(status_code=status.HTTP_503_SERVICE_UNAVAILABLE, detail=str(exc)) from exc
except ASRTranscriptionError as exc:
raise HTTPException(status_code=status.HTTP_502_BAD_GATEWAY, detail=str(exc)) from exc
except Exception as exc:
logger.exception("直链下载后 ASR 转写异常: path=%s", video_path)
raise HTTPException(
status_code=status.HTTP_502_BAD_GATEWAY,
detail="语音识别失败: " + str(exc)[:200],
) from exc
def _ytdlp_download_and_local_asr(page_url, temp_dir, cookiefile=None):
"""下载抖音视频 + 本地 ASR。优先使用传入 cookie,失败时自动合成 cookie 重试。"""
try:
import yt_dlp
except ImportError as exc:
raise HTTPException(
status_code=status.HTTP_503_SERVICE_UNAVAILABLE,
detail="抖音提取功能暂不可用(缺少依赖 yt-dlp)",
) from exc
def _download(cfile):
opts = {
"format": "best[ext=mp4][vcodec^=avc1]/best[ext=mp4][vcodec^=h264]/best[ext=mp4]/best",
"outtmpl": temp_dir + "/%(id)s.%(ext)s",
"quiet": True,
"no_warnings": True,
"noplaylist": True,
"extractor_args": {"douyin": {"webpage_cookie": ""}},
"socket_timeout": 30,
"retries": 2,
"http_headers": {
"User-Agent": (
"Mozilla/5.0 (Windows NT 10.0; Win64; x64) "
"AppleWebKit/537.36 (KHTML, like Gecko) "
"Chrome/128.0.0.0 Safari/537.36"
),
"Referer": "https://www.douyin.com/",
"Accept-Language": "zh-CN,zh;q=0.9,en;q=0.8",
},
}
if cfile:
opts["cookiefile"] = cfile
with yt_dlp.YoutubeDL(opts) as ydl:
info = ydl.extract_info(page_url, download=True)
vpath = ydl.prepare_filename(info)
dur = float(info.get("duration") or 0)
return vpath, dur
# 1. 先用传入 cookie
vpath = None
last_err = None
cookie_files_to_try = []
if cookiefile and os.path.exists(cookiefile):
cookie_files_to_try.append(cookiefile)
# 2. 合成 cookie 重试(复用 _ytdlp_extract 的 cookie 生成逻辑,8次重试)
import secrets, time as _time
ttwid = _fetch_ttwid()
for i in range(8):
if i % 3 == 0:
ttwid = _fetch_ttwid() or ttwid
cfile = _generate_synthetic_cookie_file(temp_dir, ttwid=ttwid)
cookie_files_to_try.append(cfile)
_time.sleep(0.3)
for cfile in cookie_files_to_try:
try:
vpath, duration = _download(cfile)
if vpath and os.path.isfile(vpath) and os.path.getsize(vpath) > 0:
break
logger.warning(
"ytdlp 下载返回路径无效或空文件: vpath=%s size=%s",
vpath,
os.path.getsize(vpath) if vpath and os.path.isfile(vpath) else "N/A",
)
except Exception as exc: # noqa: BLE001
last_err = exc
logger.warning("ytdlp 下载重试失败: %s", str(exc)[:200])
continue
else:
msg = str(last_err) if last_err else "unknown"
logger.warning(
"ytdlp 下载全部 %d 次 cookie 尝试均失败: ttwid_ok=%s last_err=%s",
len(cookie_files_to_try), bool(ttwid), msg[:300],
)
if _is_cookies_related_error(msg) or "Fresh cookies" in msg:
_detail = "抖音链接解析暂时不可用,请稍后重试或手动输入文案"
if _DOUYIN_DEBUG_ERRORS:
_detail = _detail + " [debug: " + msg[:300] + "]"
raise HTTPException(status_code=status.HTTP_503_SERVICE_UNAVAILABLE, detail=_detail)
_detail = "视频下载失败,请稍后重试"
if _DOUYIN_DEBUG_ERRORS:
_detail = _detail + " [debug: " + msg[:300] + "]"
raise HTTPException(status_code=status.HTTP_502_BAD_GATEWAY, detail=_detail)
if not vpath or not os.path.isfile(vpath) or os.path.getsize(vpath) == 0:
raise HTTPException(status_code=status.HTTP_502_BAD_GATEWAY, detail="视频下载异常:未获取到有效文件")
try:
duration = float(duration or 0)
except (TypeError, ValueError):
duration = 0.0
try:
text = transcribe_to_text(vpath)
except ASRNotConfiguredError as exc:
raise HTTPException(status_code=status.HTTP_503_SERVICE_UNAVAILABLE, detail=str(exc)) from exc
except ASRTranscriptionError as exc:
raise HTTPException(status_code=status.HTTP_502_BAD_GATEWAY, detail=str(exc)) from exc
except Exception as exc:
logger.exception("ASR 转写异常: path=%s", vpath)
raise HTTPException(
status_code=status.HTTP_502_BAD_GATEWAY,
detail="语音识别失败: " + str(exc)[:200],
) from exc
return text.strip(), duration
def _html_scrape_direct_url(page_url, timeout=15):
"""通过抖音分享页 HTML 直接抓取视频 CDN 直链(最后兜底,不依赖任何第三方 API)。
注意:桌面版页面通常是 SPA 需 JS 渲染,移动分享页 iesdouyin.com 含 SSR 数据。
返回 (direct_url, duration) 或 (None, 0.0)。
"""
import re
import httpx
aweme_id = None
m = re.search(r"(?:video|note)/(\d{10,})", page_url)
if not m:
# 先展开短链拿 aweme_id
try:
with httpx.Client(timeout=timeout, follow_redirects=False, verify=False) as http:
r = http.head(
page_url,
headers={
"User-Agent": (
"Mozilla/5.0 (iPhone; CPU iPhone OS 16_6 like Mac OS X) "
"AppleWebKit/605.1.15 (KHTML, like Gecko) Version/16.6 "
"Mobile/15E148 Safari/604.1"
),
},
)
loc = r.headers.get("location", "")
m2 = re.search(r"(?:video|note)/(\d{10,})", loc)
if m2:
aweme_id = m2.group(1)
except Exception as exc: # noqa: BLE001
logger.debug("HTML 解析短链重定向失败: %s", exc)
else:
aweme_id = m.group(1)
if not aweme_id:
# 直接 GET 短链跟随
try:
with httpx.Client(timeout=timeout, follow_redirects=True, verify=False) as http:
r = http.get(
page_url,
headers={
"User-Agent": (
"Mozilla/5.0 (iPhone; CPU iPhone OS 16_6 like Mac OS X) "
"AppleWebKit/605.1.15 (KHTML, like Gecko) Version/16.6 "
"Mobile/15E148 Safari/604.1"
),
},
)
m2 = re.search(r"(?:video|note)/(\d{10,})", str(r.url))
if m2:
aweme_id = m2.group(1)
except Exception: # noqa: BLE001
pass
if not aweme_id:
return None, 0.0
# 尝试移动分享页
share_url = f"https://www.iesdouyin.com/share/video/{aweme_id}/"
try:
with httpx.Client(timeout=timeout, follow_redirects=True, verify=False) as http:
r = http.get(
share_url,
headers={
"User-Agent": (
"Mozilla/5.0 (iPhone; CPU iPhone OS 16_6 like Mac OS X) "
"AppleWebKit/605.1.15 (KHTML, like Gecko) Version/16.6 "
"Mobile/15E148 Safari/604.1"
),
"Referer": "https://www.douyin.com/",
"Accept-Language": "zh-CN,zh;q=0.9",
},
)
if r.status_code != 200:
return None, 0.0
html = r.text
# 先解码转义
html_decoded = (
html.replace("\\u002F", "/")
.replace("\\/", "/")
.replace("&amp;", "&")
)
# 匹配 douyinvod CDN 链接(可能是 playwm 带水印版本)
urls = re.findall(
r"https?://[^\"'\\<>\s]+?douyinvod\.com[^\"'\\<>\s]*",
html_decoded,
)
mp4_urls = [u for u in urls if ".mp4" in u or "/video/" in u]
# 也匹配 amemv CDN
amemv_urls = re.findall(
r"https?://[^\"'\\<>\s]+?amemv\.com[^\"'\\<>\s]*?\.mp4[^\"'\\<>\s]*",
html_decoded,
)
all_urls = mp4_urls + amemv_urls
if all_urls:
# 优先无水印(URL 中不含 playwm)
for u in all_urls:
if "playwm" not in u and len(u) > 40:
logger.info("HTML 抓取到无水印直链: %s", u[:120])
return u, 0.0
for u in all_urls:
if len(u) > 40:
logger.info("HTML 抓取到带水印直链(可用于ASR): %s", u[:120])
return u, 0.0
except Exception as exc: # noqa: BLE001
logger.debug("HTML 抓取异常: %s", exc)
return None, 0.0
# ── 1. 从抖音视频提取文案 ─────────────────────────────────────────────
@router.post("/extract-from-douyin", response_model=ExtractFromDouyinResponse)
@points_gate("douyin_extract")
def extract_from_douyin(
request: ExtractFromDouyinRequest,
current_user: AuthenticatedUser = Depends(get_current_user),
db: Session = Depends(get_db_session),
):
page_url = _extract_and_validate_douyin_url(request.url)
_dbg("page_url", page_url)
text = ""
duration = 0.0
cookiefile = _resolve_cookies_file()
mk_client = get_mediakit_client()
# 整体重试:外层最多 2 轮完整链路,应对第三方 API 瞬时限流/CDN 抖动
max_rounds = 2
direct_url = None
meta_duration = 0.0
for round_idx in range(max_rounds):
direct_url = None
meta_duration = 0.0
# 路径 A1:第三方解析 API(最稳定,apizero 付费时 99.88% 可用)
logger.info("抖音解析第%d轮开始: url=%s", round_idx + 1, page_url)
try:
direct_url, meta_duration = _parser_extract_direct_url(page_url)
if direct_url:
logger.info(
"第三方解析 API 成功(第%d轮): url_domain=%s",
round_idx + 1, direct_url.split("/")[2] if "/" in direct_url else "?",
)
except Exception as exc: # noqa: BLE001
logger.warning("第三方解析 API 异常(第%d轮): %s", round_idx + 1, exc)
# 路径 A2:yt-dlp 拿直链(作为兜底,需要 cookie 概率性成功)
if not direct_url:
logger.info(
"第三方解析未拿到直链(第%d轮),尝试 yt-dlp", round_idx + 1
)
direct_url, meta_duration = _ytdlp_extract_video_url(page_url, cookiefile=cookiefile)
# 路径 A3:HTML 直抓(最后兜底,不依赖任何 API Key / cookies)
if not direct_url:
logger.info(
"yt-dlp 未拿到直链(第%d轮),尝试 HTML 直接抓取", round_idx + 1
)
try:
direct_url, meta_duration = _html_scrape_direct_url(page_url)
except Exception as exc: # noqa: BLE001
logger.warning("HTML 直抓异常(第%d轮): %s", round_idx + 1, exc)
if meta_duration:
duration = meta_duration
_dbg("direct_url", direct_url or "<none>")
# 路径 B1:直链 → MediaKit 云端 ASR(最快,不下载视频)
if direct_url and mk_client.is_available:
try:
task_id = mk_client.asr_submit(direct_url)
text, mk_duration = mk_client.asr_poll(task_id)
if mk_duration:
duration = mk_duration
logger.info(
"抖音 MediaKit ASR 成功(第%d轮): url=%s text_len=%d duration=%.1f",
round_idx + 1, page_url, len(text), duration,
)
break
except MediaKitError as exc:
logger.warning("MediaKit ASR 失败(第%d轮),回退本地 ASR: %s", round_idx + 1, exc)
text = ""
# 路径 B2:回退下载 + 本地 ASR
if not text:
_dbg("fallback", f"download+local_asr round={round_idx+1}")
try:
with tempfile.TemporaryDirectory(prefix="douyin_extract_") as temp_dir:
if direct_url:
_dbg("fallback_via", "direct_url_download")
text, dl_duration = _direct_url_download_and_local_asr(
direct_url, page_url, temp_dir
)
else:
text, dl_duration = _ytdlp_download_and_local_asr(
page_url, temp_dir, cookiefile=cookiefile
)
if dl_duration and not duration:
duration = dl_duration
if text:
logger.info(
"抖音本地 ASR 成功(第%d轮): url=%s text_len=%d",
round_idx + 1, page_url, len(text),
)
break
except HTTPException as exc:
# 可重试错误(502/503/504)短暂等待后重试
if exc.status_code in (502, 503, 504) and round_idx < max_rounds - 1:
logger.warning(
"本地 ASR 链路返回 %d(第%d轮),将重试: %s",
exc.status_code, round_idx + 1, exc.detail,
)
time.sleep(1.2 + round_idx)
continue
raise
except Exception as exc: # noqa: BLE001
logger.warning("本地 ASR 链路异常(第%d轮): %s", round_idx + 1, exc)
if round_idx < max_rounds - 1:
time.sleep(1.2 + round_idx)
continue
raise
# 如果拿到 text 了就 break
if text:
break
# 本轮全链路失败,等一下重试(直链未拿到可能是瞬时限流)
if round_idx < max_rounds - 1:
logger.info("第%d轮全链路失败,等待后重试", round_idx + 1)
time.sleep(1.5)
if not text:
logger.warning("抖音文案提取全部失败: url=%s", page_url)
raise HTTPException(
status_code=status.HTTP_503_SERVICE_UNAVAILABLE,
detail="抖音链接解析暂时不可用,请稍后重试或手动输入文案",
)
return ExtractFromDouyinResponse(
text=text,
duration_seconds=duration,
source_url=page_url,
)
# ── 2. AI 文案改写 ────────────────────────────────────────────────────
@router.post("/ai-rewrite", response_model=AiRewriteResponse)
@points_gate("ai_rewrite")
def ai_rewrite(
request: AiRewriteRequest,
current_user: AuthenticatedUser = Depends(get_current_user),
db: Session = Depends(get_db_session),
):
content = (request.content or "").strip()
if not content:
raise HTTPException(status_code=status.HTTP_400_BAD_REQUEST, detail="文案内容不能为空")
style = request.style or "口语化"
client = get_doubao_client()
if not client.is_available:
raise HTTPException(
status_code=status.HTTP_502_BAD_GATEWAY,
detail="AI 服务不可用,请联系管理员配置豆包大模型 API Key",
)
system_prompt = (
"你是一个专业的短视频文案改写专家。请对以下文案进行改写,"
"要求:保留原意、口语化、适合短视频口播、调整语序避免查重。"
)
if style:
system_prompt = system_prompt + "\n风格要求:" + style
messages = [
{"role": "system", "content": system_prompt},
{"role": "user", "content": "请改写以下文案:\n\n" + content},
]
try:
rewritten = client.chat_completion(messages=messages, temperature=0.8, max_tokens=2048)
except Exception as exc:
logger.error("AI 改写调用失败: %s", exc)
raise HTTPException(status_code=status.HTTP_502_BAD_GATEWAY, detail="AI 改写失败: " + str(exc)) from exc
if not rewritten:
raise HTTPException(status_code=status.HTTP_502_BAD_GATEWAY, detail="AI 改写未返回有效结果")
return AiRewriteResponse(original=content, rewritten=rewritten.strip(), style=style)
# ── 3. AI 标题生成 ────────────────────────────────────────────────────
@router.post("/ai-generate-titles", response_model=AiGenerateTitlesResponse)
@points_gate("ai_title")
def ai_generate_titles(
request: AiGenerateTitlesRequest,
current_user: AuthenticatedUser = Depends(get_current_user),
db: Session = Depends(get_db_session),
):
content = (request.content or "").strip()
if not content:
raise HTTPException(status_code=status.HTTP_400_BAD_REQUEST, detail="文案内容不能为空")
count = max(1, min(5, request.count))
from app.services.ai_service import generate_smart_titles
result = generate_smart_titles(description=content, style="viral", count=count)
titles = result.get("titles", [])[:count]
return AiGenerateTitlesResponse(titles=titles)