Files
xiaoxia-saas/apps/api/app/api/routes/scripts_ai.py
T
CI Bot d825756c67
CI/CD Pipeline / Check if frontend-only change (push) Has been skipped
CI/CD Pipeline / Dedup Check - skip PR tests when covered by push pipeline (push) Successful in 2s
CI/CD Pipeline / PR Build API Image (push) Has been skipped
CI/CD Pipeline / PR Build Web Image (push) Has been skipped
CI/CD Pipeline / PR Build Worker Image (push) Has been skipped
CI/CD Pipeline / Check push changed paths (push) Successful in 3s
CI/CD Pipeline / Frontend Lint (push) Has been skipped
CI/CD Pipeline / Build Staging API Image (push) Has been cancelled
CI/CD Pipeline / Build Staging Web Image (push) Has been cancelled
CI/CD Pipeline / Retag skipped Staging API Image (push) Has been cancelled
CI/CD Pipeline / Retag skipped Staging Web Image (push) Has been cancelled
CI/CD Pipeline / Validate - Style (push) Has been cancelled
CI/CD Pipeline / Retag skipped Staging Worker Image (push) Has been cancelled
CI/CD Pipeline / Deploy Staging (Watchtower auto-deploy) (push) Has been cancelled
CI/CD Pipeline / Staging E2E Tests (push) Has been cancelled
CI/CD Pipeline / Staging API Integration Tests (push) Has been cancelled
CI/CD Pipeline / Build Production API Image (push) Has been cancelled
CI/CD Pipeline / Build Production Web Image (push) Has been cancelled
CI/CD Pipeline / Build Production Worker Image (push) Has been cancelled
CI/CD Pipeline / Deploy Production (push) Has been cancelled
CI/CD Pipeline / Production Browser E2E (push) Has been cancelled
CI/CD Pipeline / ACR Image Cleanup (push) Has been cancelled
CI/CD Pipeline / Canary Release to Production (push) Has been cancelled
CI/CD Pipeline / CI Gate (push) Has been cancelled
CI/CD Pipeline / Validate - Python (mypy + alembic) (push) Has been cancelled
CI/CD Pipeline / Validate - Security (push) Has been cancelled
CI/CD Pipeline / Unit Tests (push) Has been cancelled
CI/CD Pipeline / Build Staging Worker Image (push) Has been cancelled
CI/CD Pipeline / Integration Tests (push) Has been cancelled
CI/CD Pipeline / Frontend Unit Tests (push) Has been cancelled
style: auto-format with black + isort + ruff + prettier [skip ci-format-check]
2026-09-16 14:28:30 +00:00

400 lines
15 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
"""Scripts AI 能力路由 — Issue #1893.
三个 AI 工具接口(均挂载在 /api/v1/scripts 前缀下):
- POST /extract-from-douyin 从抖音视频提取文案(yt-dlp 下载 + ASR 转写)
- POST /ai-rewrite AI 文案改写(复用豆包 LLM)
- POST /ai-generate-titles AI 标题生成(复用 generate_smart_titles)
"""
from __future__ import annotations
import logging
import os
import re
import tempfile
from app.auth import AuthenticatedUser, get_current_user
from app.dependencies import get_db_session
from app.schemas.scripts_ai import (
AiGenerateTitlesRequest,
AiGenerateTitlesResponse,
AiRewriteRequest,
AiRewriteResponse,
ExtractFromDouyinRequest,
ExtractFromDouyinResponse,
)
from app.services.script_asr_service import (
ASRNotConfiguredError,
ASRTranscriptionError,
transcribe_to_text,
)
from fastapi import APIRouter, Depends, HTTPException, status
from sqlalchemy.orm import Session
from packages.middleware.points_gate import points_gate
from packages.shared.ai_client import get_doubao_client
logger = logging.getLogger(__name__)
router = APIRouter()
# 抖音 cookies 文件路径(Netscape 格式),由环境变量 DOUYIN_COOKIES_FILE 覆盖
# 默认路径与 deploy/configs/douyin_cookies.txt 对应(容器内挂载到 /app/configs/)
DOUYIN_COOKIES_FILE = os.environ.get(
"DOUYIN_COOKIES_FILE",
"/app/configs/douyin_cookies.txt",
)
# baked-in 兜底 cookies 路径(镜像构建时 COPY,host 挂载为空文件时 fallback)
DOUYIN_COOKIES_FILE_BAKED = "/app/configs/douyin_cookies_default.txt"
# cookies 失效/需要刷新的错误关键词
_COOKIES_ERROR_KEYWORDS = (
"fresh cookies",
"cookies (not necessarily logged in)",
"cookies are needed",
"need cookies",
"cookie is expired",
"login required",
"sign in to continue",
"未登录",
"需要登录",
"cookies过期",
)
def _resolve_cookies_file() -> str | None:
"""返回有效的 cookies 文件路径:host 挂载优先 > baked-in 兜底 > None."""
for p in (DOUYIN_COOKIES_FILE, DOUYIN_COOKIES_FILE_BAKED):
try:
if p and os.path.isfile(p) and os.path.getsize(p) > 200:
return p
except OSError:
continue
return None
def _cookies_file_exists() -> bool:
return _resolve_cookies_file() is not None
def _dbg(key: str, val: str) -> None:
"""记录抖音调试信息(debug 日志)。"""
logger.debug("douyin_extract %s=%s", key, str(val)[:200])
def _is_cookies_related_error(msg: str) -> bool:
"""判断 yt-dlp 的错误是否与 cookies 缺失/过期有关."""
low = msg.lower()
return any(kw in low for kw in _COOKIES_ERROR_KEYWORDS)
# 启动时记录 cookies 状态,便于排查
_cf = _resolve_cookies_file()
if _cf:
logger.info("抖音 cookies 文件已加载: %s (%d bytes)", _cf, os.path.getsize(_cf))
else:
logger.warning(
"抖音 cookies 文件未找到或无效: path=%s baked=%s 抖音提取功能可能因 cookies 缺失失败",
DOUYIN_COOKIES_FILE, DOUYIN_COOKIES_FILE_BAKED,
)
# 是否在错误响应中暴露原始 yt-dlp 错误(仅 staging/dev 用于排查,生产默认 False)
_DOUYIN_DEBUG_ERRORS = os.environ.get("DOUYIN_DEBUG_ERRORS", "").lower() in ("1", "true", "yes")
# 抖音 URL 校验:支持短链 v.douyin.com 和长链 www.douyin.com/video/
_DOUYIN_URL_RE = re.compile(
r"^(https?://)?(v\.douyin\.com/\S+|www\.douyin\.com/video/\S+)$",
re.IGNORECASE,
)
def _validate_douyin_url(url: str) -> None:
"""校验抖音 URL 格式,不合法时抛 HTTPException(400)."""
if not url or not url.strip():
raise HTTPException(
status_code=status.HTTP_400_BAD_REQUEST,
detail="链接不能为空",
)
if not _DOUYIN_URL_RE.match(url.strip()):
raise HTTPException(
status_code=status.HTTP_400_BAD_REQUEST,
detail="无效的抖音链接,仅支持 v.douyin.com 短链或 www.douyin.com/video/ 长链",
)
# ── 1. 从抖音视频提取文案 ─────────────────────────────────────────────────────
@router.post(
"/extract-from-douyin",
response_model=ExtractFromDouyinResponse,
)
@points_gate("douyin_extract")
def extract_from_douyin(
request: ExtractFromDouyinRequest,
current_user: AuthenticatedUser = Depends(get_current_user),
db: Session = Depends(get_db_session),
) -> ExtractFromDouyinResponse:
"""从抖音视频下载无水印视频并通过 ASR 提取文案."""
source_url = request.url.strip()
_validate_douyin_url(source_url)
# 确保 URL 有 scheme(yt-dlp 需要完整 URL)
url_for_download = source_url
if not re.match(r"^https?://", url_for_download, re.IGNORECASE):
url_for_download = "https://" + url_for_download
text: str = ""
duration: float = 0.0
try:
with tempfile.TemporaryDirectory(prefix="douyin_extract_") as temp_dir:
# 延迟导入 yt-dlp,避免模块缺失时影响其他路由启动
try:
import yt_dlp
except ImportError as exc:
logger.error("yt-dlp 未安装,抖音提取功能不可用: %s", exc)
raise HTTPException(
status_code=status.HTTP_503_SERVICE_UNAVAILABLE,
detail="抖音提取功能暂不可用(缺少依赖 yt-dlp)",
) from exc
ydl_opts = {
"format": "best[ext=mp4]/best",
"outtmpl": f"{temp_dir}/%(id)s.%(ext)s",
"quiet": True,
"no_warnings": True,
"noplaylist": True,
# 抖音反爬严格,必须用真实桌面浏览器 UA
"http_headers": {
"User-Agent": (
"Mozilla/5.0 (Windows NT 10.0; Win64; x64) "
"AppleWebKit/537.36 (KHTML, like Gecko) "
"Chrome/128.0.0.0 Safari/537.36"
),
"Referer": "https://www.douyin.com/",
},
}
# 如果存在抖音 cookies 文件,传给 yt-dlp 绕过反爬(host 挂载优先,空文件 fallback 到镜像内)
_cookies_path = _resolve_cookies_file()
if _cookies_path:
ydl_opts["cookiefile"] = _cookies_path
logger.debug("使用抖音 cookies 文件: %s (%d bytes)", _cookies_path, os.path.getsize(_cookies_path))
_dbg("cookies", f"{_cookies_path} {os.path.getsize(_cookies_path)}B")
try:
ydl = yt_dlp.YoutubeDL(ydl_opts)
info = ydl.extract_info(url_for_download, download=True)
except yt_dlp.utils.DownloadError as exc:
# yt-dlp 官方异常类型:HTTP 错误、短链失效、视频下架、cookies 过期等
msg = str(exc)
logger.warning("抖音下载失败: url=%s error=%s", source_url, msg)
_dbg("errtype", "DownloadError")
_dbg("errmsg", msg)
# 404/视频不存在/不可下载 → 400
is_bad_url = any(
kw in msg.lower() for kw in ("404", "not found", "unable to download webpage", "unsupported url", "no video formats", "video unavailable", "this video isn't available")
)
# cookies 缺失/过期 → 返回友好提示,不暴露 yt-dlp 原始错误
if _is_cookies_related_error(msg):
logger.error("抖音 cookies 失效或缺失,需要刷新: %s", msg[:300])
_detail = "抖音链接解析暂时不可用,请稍后重试或手动输入文案"
if _DOUYIN_DEBUG_ERRORS:
_detail = f"{_detail} [debug: {msg[:300]}]"
raise HTTPException(
status_code=status.HTTP_503_SERVICE_UNAVAILABLE,
detail=_detail,
) from exc
_detail = "无法解析该抖音链接,请确认链接有效且视频未被下架" if is_bad_url else "视频下载失败,请稍后重试"
if _DOUYIN_DEBUG_ERRORS:
_detail = f"{_detail} [debug: {msg[:300]}]"
raise HTTPException(
status_code=status.HTTP_400_BAD_REQUEST if is_bad_url else status.HTTP_502_BAD_GATEWAY,
detail=_detail,
) from exc
except Exception as exc:
msg = str(exc)
logger.exception("抖音视频下载异常: url=%s error=%s", source_url, msg)
_dbg("errtype", type(exc).__name__)
_dbg("errmsg", msg)
# cookies 相关的未知异常也走友好提示
if _is_cookies_related_error(msg):
_detail = "抖音链接解析暂时不可用,请稍后重试或手动输入文案"
if _DOUYIN_DEBUG_ERRORS:
_detail = f"{_detail} [debug: {msg[:300]}]"
raise HTTPException(
status_code=status.HTTP_503_SERVICE_UNAVAILABLE,
detail=_detail,
) from exc
_detail = "视频下载失败,请稍后重试"
if _DOUYIN_DEBUG_ERRORS:
_detail = f"{_detail} [debug: {msg[:300]}]"
raise HTTPException(
status_code=status.HTTP_502_BAD_GATEWAY,
detail=_detail,
) from exc
if info is None:
raise HTTPException(
status_code=status.HTTP_400_BAD_REQUEST,
detail="无法解析该抖音链接",
)
video_path = ydl.prepare_filename(info)
try:
duration = float(info.get("duration") or 0)
except (TypeError, ValueError):
duration = 0.0
# 校验下载的文件是否真的存在(某些 yt-dlp 版本可能 info 成功但未下载到文件)
if not os.path.isfile(video_path) or os.path.getsize(video_path) == 0:
logger.error("yt-dlp 未产生有效视频文件: path=%s", video_path)
raise HTTPException(
status_code=status.HTTP_502_BAD_GATEWAY,
detail="视频下载异常:未获取到有效文件",
)
# ASR 转写(兜底捕获所有异常,避免 500)
try:
text = transcribe_to_text(video_path)
except ASRNotConfiguredError as exc:
raise HTTPException(
status_code=status.HTTP_503_SERVICE_UNAVAILABLE,
detail=str(exc),
) from exc
except ASRTranscriptionError as exc:
raise HTTPException(
status_code=status.HTTP_502_BAD_GATEWAY,
detail=str(exc),
) from exc
except Exception as exc:
logger.exception("ASR 转写异常: path=%s", video_path)
raise HTTPException(
status_code=status.HTTP_502_BAD_GATEWAY,
detail=f"语音识别失败: {str(exc)[:200]}",
) from exc
except HTTPException:
raise
except Exception as exc:
# 最后兜底:任何未捕获异常都转成 502/400,不允许冒泡成 500
logger.exception("抖音文案提取未预期异常: url=%s", source_url)
raise HTTPException(
status_code=status.HTTP_500_INTERNAL_SERVER_ERROR,
detail=f"抖音文案提取失败: {str(exc)[:200]}",
) from exc
return ExtractFromDouyinResponse(
text=text,
duration_seconds=duration,
source_url=source_url,
)
# ── 2. AI 文案改写 ───────────────────────────────────────────────────────────
@router.post(
"/ai-rewrite",
response_model=AiRewriteResponse,
)
@points_gate("ai_rewrite")
def ai_rewrite(
request: AiRewriteRequest,
current_user: AuthenticatedUser = Depends(get_current_user),
db: Session = Depends(get_db_session),
) -> AiRewriteResponse:
"""使用豆包大模型改写文案."""
content = (request.content or "").strip()
if not content:
raise HTTPException(
status_code=status.HTTP_400_BAD_REQUEST,
detail="文案内容不能为空",
)
style = request.style or "口语化"
client = get_doubao_client()
if not client.is_available:
raise HTTPException(
status_code=status.HTTP_502_BAD_GATEWAY,
detail="AI 服务不可用,请联系管理员配置豆包大模型 API Key",
)
system_prompt = (
"你是一个专业的短视频文案改写专家。请对以下文案进行改写,"
"要求:保留原意、口语化、适合短视频口播、调整语序避免查重。"
)
if style:
system_prompt += f"\n风格要求:{style}"
user_prompt = f"请改写以下文案:\n\n{content}"
messages = [
{"role": "system", "content": system_prompt},
{"role": "user", "content": user_prompt},
]
try:
rewritten = client.chat_completion(
messages=messages,
temperature=0.8,
max_tokens=2048,
)
except Exception as exc:
logger.error("AI 改写调用失败: %s", exc)
raise HTTPException(
status_code=status.HTTP_502_BAD_GATEWAY,
detail=f"AI 改写失败: {exc}",
) from exc
if not rewritten:
raise HTTPException(
status_code=status.HTTP_502_BAD_GATEWAY,
detail="AI 改写未返回有效结果",
)
return AiRewriteResponse(
original=content,
rewritten=rewritten.strip(),
style=style,
)
# ── 3. AI 标题生成 ───────────────────────────────────────────────────────────
@router.post(
"/ai-generate-titles",
response_model=AiGenerateTitlesResponse,
)
@points_gate("ai_title")
def ai_generate_titles(
request: AiGenerateTitlesRequest,
current_user: AuthenticatedUser = Depends(get_current_user),
db: Session = Depends(get_db_session),
) -> AiGenerateTitlesResponse:
"""使用现有 generate_smart_titles 生成标题."""
content = (request.content or "").strip()
if not content:
raise HTTPException(
status_code=status.HTTP_400_BAD_REQUEST,
detail="文案内容不能为空",
)
# count 限制在 1-5(Pydantic ge=1 le=5 已校验),但为兼容直接调用场景截断
count = max(1, min(5, request.count))
from app.services.ai_service import generate_smart_titles
result = generate_smart_titles(
description=content,
style="viral",
count=count,
)
titles = result.get("titles", [])[:count]
return AiGenerateTitlesResponse(titles=titles)