diff --git a/apps/worker/worker_app/tasks/viral_video.py b/apps/worker/worker_app/tasks/viral_video.py index 2b05a4748..d4b7bb937 100644 --- a/apps/worker/worker_app/tasks/viral_video.py +++ b/apps/worker/worker_app/tasks/viral_video.py @@ -99,25 +99,104 @@ def _save_job(repo, job, session): # ── 流水线各步骤 ──────────────────────────────────────────────────────── +_IMAGE_ANALYSIS_PROMPT = """请仔细观察这张图片,只基于图片中真实可见的内容进行分析,不要凭空想象。 + +必须输出严格的 JSON(不要 Markdown 代码块,不要额外解释),字段如下: +{ + "category": "产品大类,如护肤品/彩妆/食品/数码/服饰/家居等,若无法识别填『无法判断』", + "name": "产品名称(从包装/品牌/logo/文字推断;没有品牌时描述外观如『粉色包装面霜』)", + "brand": "品牌名(看 logo/包装文字;看不清填『未知』)", + "colors": ["主体颜色"], + "material_or_texture": "材质/质地描述(如玻璃瓶装/塑料软管/哑光质感/金属外壳等;无法判断填『无法判断』)", + "key_features": [ + "3-5 条**图片中确实能看到**的外观特征/卖点描述(如『按压式泵头』『瓶身有金色装饰线』等),不要编图片里没有的功效" + ], + "visual_style": "视觉风格(如简约高端/粉嫩少女/国潮/科技感/生活方式实拍等)", + "scene": "图片中的使用/展示场景(如白底棚拍/浴室场景/户外街拍/桌面静物等;纯白底填『白底产品图』)", + "target_audience_hint": "从视觉推断的目标人群(如年轻女性/男性商务/亲子家庭等;不确定填『通用』)", + "text_on_image": "图片上出现的可读文字(品牌名/Slogan/产品名等,没有则填『无』)" +} + +严格要求: +1. 任何字段无法确认时填『无法判断』或『未知』,不要猜。 +2. key_features 只能描述图片里肉眼可见的物理外观,不要写『补水保湿』『抗衰老』这类功效词(除非包装上明确印了)。 +3. 如果图片完全不是产品图(比如风景/人像/截图),category 填『非产品图』,name 填实际看到的内容。 +""" + + def _step_image_analysis(job: ViralVideoJob) -> dict: - """步骤 1: 图片 VLM 分析 — 识别产品特征、场景、卖点。""" + """步骤 1: 图片 VLM 分析 — 识别产品特征、场景、卖点。 + + Bug #2114 修复: + 1) call_vision 现已走视觉模型 doubao-1-5-vision-pro(之前误走文本模型导致完全没看图); + 2) Prompt 强化为结构化 JSON schema,禁止编造,强制图片可见才写; + 3) 单张失败不影响其他图片,最终至少返回一张占位结果避免后续 NoneType; + 4) 日志打印每张图的 URL 和模型原始返回,方便排查。 + """ try: from packages.shared.ai_service import call_vision except ImportError: logger.warning("[爆款视频] ai_service.call_vision 不可用,使用占位结果") - return {"products": [{"name": "产品", "features": ["特征1", "特征2"], "scene": "通用场景"}]} + return { + "products": [ + { + "name": "产品", + "features": ["特征1", "特征2"], + "scene": "通用场景", + "_source": "fallback_import_error", + } + ] + } + + if not job.images: + logger.warning("[爆款视频] 任务无 images,跳过图片分析") + return {"products": []} results = [] - for img_url in job.images: + for idx, img_url in enumerate(job.images): + logger.info("[爆款视频] 图片分析 #%d img=%s", idx, img_url[:160]) try: - result = call_vision( - image_url=img_url, - prompt="请分析这张产品图片,识别:1)产品名称和类别 2)主要特征和卖点 3)适用场景 4)视觉风格。以JSON格式返回。", - ) - results.append(result) + result = call_vision(image_url=img_url, prompt=_IMAGE_ANALYSIS_PROMPT) + if result is None: + logger.warning("[爆款视频] 图片 #%d call_vision 返回 None(模型超时/Key未配置)", idx) + results.append( + { + "name": "未识别", + "category": "无法判断", + "key_features": [], + "scene": "通用", + "_source": "vision_none", + } + ) + elif isinstance(result, str): + # JSON 解析失败返回的原文,包装一下防止后续 .get 报错 + logger.warning("[爆款视频] 图片 #%d VLM 返回非 JSON 文本,包装为 features: %s", idx, result[:200]) + results.append( + { + "name": "未识别", + "category": "无法判断", + "key_features": [], + "scene": "通用", + "_raw": result[:500], + "_source": "vision_text", + } + ) + else: + # dict 正常 + result.setdefault("_source", "vision") + results.append(result) except Exception as e: - logger.warning("[爆款视频] 图片分析失败 img=%s: %s", img_url, e) - results.append({"name": "未识别", "features": [], "scene": "通用"}) + logger.warning("[爆款视频] 图片分析失败 img=%s err=%s", img_url[:120], e, exc_info=True) + results.append( + { + "name": "未识别", + "category": "无法判断", + "key_features": [], + "scene": "通用", + "_source": "vision_exception", + "_error": str(e)[:200], + } + ) return {"products": results} @@ -157,7 +236,18 @@ def _step_intent_parsing(job: ViralVideoJob, image_analysis: dict) -> dict: products_summary = "" for p in image_analysis.get("products", []): - products_summary += f"- {p.get('name', '产品')}: {', '.join(p.get('features', []))}\n" + feats = p.get("key_features") or p.get("features") or [] + extras = [] + if p.get("brand") and p.get("brand") not in ("未知", "无法判断"): + extras.append(f"品牌={p['brand']}") + if p.get("category") and p.get("category") not in ("无法判断", "非产品图"): + extras.append(f"品类={p['category']}") + if p.get("colors"): + extras.append(f"颜色={','.join(p['colors'])}") + if p.get("scene") and p.get("scene") not in ("通用",): + extras.append(f"场景={p['scene']}") + feat_str = ", ".join([str(x) for x in feats + extras]) + products_summary += f"- {p.get('name', '产品')}: {feat_str}\n" prompt = f"""你是一个营销文案策略师。请分析以下信息,理解用户的营销意图: @@ -192,7 +282,14 @@ def _step_copy_fusion(job: ViralVideoJob, intent: dict, image_analysis: dict) -> products_desc = "" for p in image_analysis.get("products", []): - products_desc += f"{p.get('name', '产品')}({','.join(p.get('features', []))})\n" + feats = p.get("key_features") or p.get("features") or [] + extras = [] + if p.get("brand") and p.get("brand") not in ("未知", "无法判断"): + extras.append(f"品牌={p['brand']}") + if p.get("visual_style"): + extras.append(f"风格={p['visual_style']}") + feat_str = ",".join([str(x) for x in feats + extras]) + products_desc += f"{p.get('name', '产品')}({feat_str})\n" if job.fusion_level == "ai_full": prompt = f"""请为以下产品撰写一段爆款短视频文案({job.duration}秒): @@ -251,8 +348,10 @@ def _step_storyboard(job: ViralVideoJob, copy_text: str, image_analysis: dict) - products = image_analysis.get("products", []) if image_analysis else [] if products: p0 = products[0] if isinstance(products[0], dict) else {} - feats = p0.get("features", []) if isinstance(p0, dict) else [] - products_hint = f"\n首帧参考产品特征:{p0.get('name','')} - {', '.join(feats[:3])}" + feats = (p0.get("key_features") or p0.get("features") or []) if isinstance(p0, dict) else [] + brand = p0.get("brand") if isinstance(p0, dict) else "" + brand_hint = f"(品牌={brand})" if brand and brand not in ("未知", "无法判断") else "" + products_hint = f"\n首帧参考产品特征:{p0.get('name','')}{brand_hint} - {', '.join(feats[:3])}" seg_seconds = 5 n_segments = max(2, min(6, max(1, job.duration // seg_seconds))) diff --git a/packages/shared/ai_service.py b/packages/shared/ai_service.py index d527dd06b..7b557e3b2 100755 --- a/packages/shared/ai_service.py +++ b/packages/shared/ai_service.py @@ -515,26 +515,59 @@ def call_llm(prompt: str, temperature: float = 0.7) -> object: def call_vision(image_url: str, prompt: str) -> object: - """调用豆包视觉大模型分析图片,返回解析后的 JSON 或原文字符串;失败返回 None。""" + """调用豆包视觉大模型分析图片,返回解析后的 JSON 或原文字符串;失败返回 None。 + + Bug #2114 (VLM 牛头不对马嘴根因修复): + 之前误走 client.chat_completion(用文本模型 doubao-seed-1.6),多模态 content list 被当成 + 纯文本发给文本模型 → 模型要么看不到图、要么抛 400,静默被 except 吞掉 → 返回 None → + _step_image_analysis fallback 到 {"name":"未识别"} → 后续文案/分镜完全没图的信息。 + 现改走 vision_completion,走视觉模型 doubao-1-5-vision-pro-250915。 + """ client = get_doubao_client() if not client.is_available: + logger.warning("[call_vision] 豆包客户端未配置 (DOUBAO_API_KEY 缺失)") return None + if not image_url: + logger.warning("[call_vision] 空 image_url,跳过视觉分析") + return None + + system_prompt = ( + "你是资深电商视觉分析师。请严格基于用户提供的图片观察回答," + "图片里没有的信息不要凭空想象或编造;看不清或无法判断时明确说" + "「图片中无法判断」,不要猜测。输出必须是严格 JSON,不要附加 Markdown 或解释文字。" + ) messages = [ - {"role": "system", "content": "你是专业的视觉分析师。需要结构化输出时请严格使用 JSON。"}, - { - "role": "user", - "content": [ - {"type": "text", "text": prompt}, - {"type": "image_url", "image_url": {"url": image_url}}, - ], - }, + {"role": "system", "content": system_prompt}, + {"role": "user", "content": prompt}, ] - raw = client.chat_completion(messages, temperature=0.3, max_tokens=2048) + + logger.info( + "[call_vision] 调用豆包视觉模型 vision_model=%s image_url=%s prompt_len=%d", + getattr(client, "vision_model", "?"), + image_url[:120], + len(prompt), + ) + raw = client.vision_completion( + messages=messages, + images=[image_url], + temperature=0.2, + max_tokens=2048, + timeout=60, + ) if raw is None: + logger.warning("[call_vision] 视觉模型返回 None (image_url=%s)", image_url[:80]) return None + logger.info("[call_vision] 视觉模型原始返回 (前400字): %s", raw[:400]) + # 剥离 ```json ... ``` 包裹 + stripped = raw.strip() + if stripped.startswith("```"): + stripped = stripped.strip("`") + if stripped.startswith("json"): + stripped = stripped[4:].lstrip() try: - return json.loads(raw) - except (json.JSONDecodeError, TypeError): + return json.loads(stripped) + except (json.JSONDecodeError, TypeError) as e: + logger.warning("[call_vision] JSON 解析失败(%s),返回原始文本: %s", e, raw[:200]) return raw