diff --git a/apps/worker/worker_app/tasks/viral_video.py b/apps/worker/worker_app/tasks/viral_video.py index da598d833..d1adf6c58 100644 --- a/apps/worker/worker_app/tasks/viral_video.py +++ b/apps/worker/worker_app/tasks/viral_video.py @@ -422,18 +422,46 @@ def _analyze_single_image( ) def _call(model: str, tmo: int): + # #2194: 单次调用临时关闭 httpx 层重试,超时/失败由外层 pro_fallback 统一兜底, + # 避免底层 max_retries=1 导致 lite timeout × 2 + pro timeout × 2 最坏 240s+ + _t0 = time.time() try: - return call_vision( - image_url=img_url, - prompt=user, - model=model, - max_tokens=1200, # #2188: 结构化 XML 输出 600-900 字足够,2048 翻倍耗时 - temperature=0.3, - timeout=tmo, - system_prompt=system, - ) + from packages.shared.ai_client import get_doubao_client as _gdc + import json as _json + _client = _gdc() + _messages = [ + {"role": "system", "content": system}, + {"role": "user", "content": user}, + ] + _orig_retries = _client.max_retries + _client.max_retries = 0 + try: + raw = _client.vision_completion( + messages=_messages, + images=[img_url], + temperature=0.3, + max_tokens=1200, # #2188: 结构化 XML 输出 600-900 字足够 + timeout=tmo, + model=model, + ) + finally: + _client.max_retries = _orig_retries + _elapsed = time.time() - _t0 + logger.info("[爆款视频] 图片 #%d VLM(%s) 完成 elapsed=%.1fs timeout=%d", idx, model, _elapsed, tmo) + if raw is None: + return None + stripped = raw.strip() + if stripped.startswith("```"): + stripped = stripped.strip("`") + if stripped.startswith("json"): + stripped = stripped[4:].lstrip() + try: + return _json.loads(stripped) + except (_json.JSONDecodeError, TypeError): + return raw except Exception as e: - logger.warning("[爆款视频] 图片 #%d call_vision(%s) 异常 err=%s", idx, model, e) + _elapsed = time.time() - _t0 + logger.warning("[爆款视频] 图片 #%d call_vision(%s) 异常 elapsed=%.1fs err=%s", idx, model, _elapsed, e) return None def _xml_to_product(nodes: list, raw_text: str) -> dict: @@ -731,7 +759,7 @@ def _analyze_single_image( return first_result if pro_fallback_model and pro_fallback_model != vision_model: - pro_raw = _call(pro_fallback_model, 90) # #2188: pro fallback 给 90s 余量 + pro_raw = _call(pro_fallback_model, 45) # #2194: pro 单次 45s 封顶,禁用重试,避免最坏 240s pro_result = _normalize(pro_raw, "pro_fallback") if _is_vision_result_usable(pro_result): pro_result["_fallback_used"] = True @@ -742,10 +770,12 @@ def _analyze_single_image( def _step_image_analysis(job: ViralVideoJob) -> dict: """步骤 1: 图片 VLM 分析 — 识别产品特征(v1.6 优化:并行 + lite 模型提速)。 - #2188: (1) 所有图片 URL 先归一化(storage_key→公网URL+空值报400) + #2188/#2194: (1) 所有图片 URL 先归一化(storage_key→公网URL+空值报400) (2) 爆款视频强制 lite-first,不依赖 .env USE_LITE 开关 (3) max_tokens=1200,max_workers=min(2,n) 防方舟限流 - (4) lite timeout=30s,pro fallback timeout=90s + (4) lite 单次25s封顶、pro单次45s封顶,底层httpx重试关闭, + 单图最坏 25+45=70s,3图2并发最坏约70s(含排队),不超4min + (5) 每张图 VLM 调用结束打印 elapsed 耗时日志便于排查 """ try: from packages.shared.ai_service import call_vision # noqa: F401 @@ -776,7 +806,7 @@ def _step_image_analysis(job: ViralVideoJob) -> dict: lite_model = "doubao-seed-2-1-lite-260915" pro_model = "doubao-seed-2-1-pro-260915" vision_model = lite_model # 永远 lite 主跑 - vision_timeout = 30 # lite 目标 20-30s + vision_timeout = 25 # #2194: lite 单次 25s 封顶,超时立即降级 pro,不让用户等 4min+ results: list[dict] = [None] * len(normalized_urls) # type: ignore max_workers = min(2, max(1, len(normalized_urls))) # #2188: 并发≤2 防方舟限流 @@ -786,7 +816,7 @@ def _step_image_analysis(job: ViralVideoJob) -> dict: vision_model, pro_model, vision_timeout, - 90, + 45, max_workers, ) with ThreadPoolExecutor(max_workers=max_workers) as pool: