diff --git a/apps/api/app/api/routes/scripts_ai.py b/apps/api/app/api/routes/scripts_ai.py index 5ecb3f5d0..4b80b0840 100644 --- a/apps/api/app/api/routes/scripts_ai.py +++ b/apps/api/app/api/routes/scripts_ai.py @@ -356,7 +356,7 @@ def _fetch_ttwid(): # 方式一:直接调 ttwid 注册接口 try: with httpx.Client(timeout=8, follow_redirects=True, verify=False) as http: - http.post( + r = http.post( "https://ttwid.bytedance.com/ttwid/union/register/", json={ "region": "cn", @@ -769,7 +769,7 @@ def _ytdlp_download_and_local_asr(page_url, temp_dir, cookiefile=None): cookie_files_to_try.append(cookiefile) # 2. 合成 cookie 重试(复用 _ytdlp_extract 的 cookie 生成逻辑,8次重试) - import time as _time + import secrets, time as _time ttwid = _fetch_ttwid() for i in range(8): if i % 3 == 0: @@ -945,10 +945,9 @@ def _html_scrape_direct_url(page_url, timeout=15): @router.get("/douyin/__debug_diag") def douyin_diag(): """[Staging only] 容器内网络/yt-dlp诊断""" - import tempfile as _tf - import time as _t - import httpx as _httpx + import time as _t + import tempfile as _tf import yt_dlp as _ytdlp results = {} @@ -1149,9 +1148,11 @@ def extract_from_douyin( # 整体重试:外层最多 2 轮完整链路,应对第三方 API 瞬时限流/CDN 抖动 max_rounds = 2 meta_duration = 0.0 + last_err_stage = "unknown" # 记录失败阶段,用于返回具体错误信息 for round_idx in range(max_rounds): # 路径 A0 已成功则跳过 A1/A2/A3;否则走兜底链路 if not direct_url: + last_err_stage = "parse" # 路径 A1:第三方解析 API(apizero 付费时可用) logger.info("抖音解析第%d轮开始: url=%s a0_ok=%s", round_idx + 1, page_url, bool(direct_url)) try: @@ -1187,6 +1188,7 @@ def extract_from_douyin( # 路径 B1:直链 → MediaKit 云端 ASR(最快,不下载视频) if direct_url and mk_client.is_available: + last_err_stage = "asr" try: task_id = mk_client.asr_submit(direct_url) text, mk_duration = mk_client.asr_poll(task_id) @@ -1203,15 +1205,20 @@ def extract_from_douyin( # 路径 B2:回退下载 + 本地 ASR if not text: + last_err_stage = "download" if not direct_url else "asr" _dbg("fallback", f"download+local_asr round={round_idx+1}") try: with tempfile.TemporaryDirectory(prefix="douyin_extract_") as temp_dir: if direct_url: _dbg("fallback_via", "direct_url_download") + last_err_stage = "download" text, dl_duration = _direct_url_download_and_local_asr( direct_url, page_url, temp_dir ) + if text: + last_err_stage = "asr" # 下载成功后才是asr阶段 else: + last_err_stage = "parse" text, dl_duration = _ytdlp_download_and_local_asr( page_url, temp_dir, cookiefile=cookiefile ) @@ -1264,10 +1271,17 @@ def extract_from_douyin( text = feed_desc.strip() logger.info("抖音所有 ASR 路径均失败,最终使用 Feed desc 作为文案: desc_len=%d", len(text)) else: - logger.warning("抖音文案提取全部失败: url=%s", page_url) + # 根据失败阶段返回具体错误信息 + stage_msg = { + "parse": "抖音视频链接解析失败,请检查链接是否正确或稍后重试", + "download": "抖音视频下载失败,请检查网络或稍后重试", + "asr": "抖音语音识别失败,请稍后重试或手动输入文案", + } + user_msg = stage_msg.get(last_err_stage, "抖音链接解析暂时不可用,请稍后重试或手动输入文案") + logger.warning("抖音文案提取全部失败: url=%s stage=%s", page_url, last_err_stage) raise HTTPException( status_code=status.HTTP_503_SERVICE_UNAVAILABLE, - detail="抖音链接解析暂时不可用,请稍后重试或手动输入文案", + detail=user_msg, ) return ExtractFromDouyinResponse(