feat(viral-video): VLM提速(lite+并行) + 商品描述增强(6维度+summary) + 爆款结构19选注入 (#2133)
CI/CD Pipeline / Check if frontend-only change (push) Has been skipped
CI/CD Pipeline / Check if frontend-only change (pull_request) Successful in 3s
CI/CD Pipeline / Check push changed paths (pull_request) Has been skipped
CI/CD Pipeline / PR Build API Image (push) Has been skipped
CI/CD Pipeline / PR Build Web Image (push) Has been skipped
CI/CD Pipeline / PR Build Worker Image (push) Has been skipped
CI/CD Pipeline / Dedup Check - skip PR tests when covered by push pipeline (pull_request) Successful in 8s
CI/CD Pipeline / Dedup Check - skip PR tests when covered by push pipeline (push) Successful in 13s
CI/CD Pipeline / Build Staging API Image (pull_request) Has been skipped
CI/CD Pipeline / Build Staging Web Image (pull_request) Has been skipped
CI/CD Pipeline / Build Staging Worker Image (pull_request) Has been skipped
CI/CD Pipeline / Check push changed paths (push) Successful in 13s
CI/CD Pipeline / Validate - Style (pull_request) Has been skipped
CI/CD Pipeline / Validate - Security (pull_request) Has been skipped
CI/CD Pipeline / Validate - Python (mypy + alembic) (pull_request) Has been skipped
CI/CD Pipeline / Unit Tests (pull_request) Has been skipped
CI/CD Pipeline / Integration Tests (pull_request) Has been skipped
CI/CD Pipeline / Frontend Lint (pull_request) Has been skipped
CI/CD Pipeline / Retag skipped Staging API Image (pull_request) Has been skipped
CI/CD Pipeline / Frontend Unit Tests (pull_request) Has been skipped
CI/CD Pipeline / Retag skipped Staging Web Image (pull_request) Has been skipped
CI/CD Pipeline / Retag skipped Staging Worker Image (pull_request) Has been skipped
CI/CD Pipeline / Frontend Lint (push) Has been skipped
CI/CD Pipeline / PR Build API Image (pull_request) Successful in 1m13s
CI/CD Pipeline / PR Build Worker Image (pull_request) Successful in 1m15s
CI/CD Pipeline / Deploy Staging (Watchtower auto-deploy) (pull_request) Has been skipped
CI/CD Pipeline / Build Production API Image (pull_request) Has been skipped
CI/CD Pipeline / Build Production Web Image (pull_request) Has been skipped
CI/CD Pipeline / Build Production Worker Image (pull_request) Has been skipped
CI/CD Pipeline / Staging E2E Tests (pull_request) Has been skipped
CI/CD Pipeline / Staging API Integration Tests (pull_request) Has been skipped
CI/CD Pipeline / Deploy Production (pull_request) Has been skipped
CI/CD Pipeline / Production Browser E2E (pull_request) Has been skipped
CI/CD Pipeline / Build Staging API Image (push) Successful in 59s
CI/CD Pipeline / Canary Release to Production (pull_request) Has been skipped
CI/CD Pipeline / ACR Image Cleanup (pull_request) Has been skipped
CI/CD Pipeline / Build Staging Worker Image (push) Successful in 42s
CI/CD Pipeline / PR Build Web Image (pull_request) Successful in 1m58s
CI/CD Pipeline / CI Gate (pull_request) Successful in 3s
CI/CD Pipeline / Build Staging Web Image (push) Successful in 1m26s
CI/CD Pipeline / Retag skipped Staging API Image (push) Has been skipped
CI/CD Pipeline / Retag skipped Staging Web Image (push) Has been skipped
CI/CD Pipeline / Retag skipped Staging Worker Image (push) Has been skipped
CI/CD Pipeline / Integration Tests (push) Successful in 2m54s
PR Automation / Auto Approve on CI Green (pull_request) Successful in 3m23s
PR Automation / Auto Merge on CI Green + Approved (pull_request) Has been skipped
CI/CD Pipeline / Deploy Staging (Watchtower auto-deploy) (push) Successful in 1m32s
CI/CD Pipeline / Validate - Style (push) Successful in 3m42s
Preview Deploy / Deploy Preview Environment (pull_request) Successful in 4m39s
CI/CD Pipeline / Validate - Security (push) Successful in 4m45s
CI/CD Pipeline / Frontend Unit Tests (push) Successful in 5m7s
CI/CD Pipeline / ACR Image Cleanup (push) Successful in 1m55s
CI/CD Pipeline / Validate - Python (mypy + alembic) (push) Successful in 5m34s
CI/CD Pipeline / Staging API Integration Tests (push) Successful in 3m12s
AI Code Review / AI Code Review (pull_request) Successful in 7m7s
CI/CD Pipeline / Staging E2E Tests (push) Failing after 3m38s
CI/CD Pipeline / Unit Tests (push) Successful in 10m29s
CI/CD Pipeline / Build Production API Image (push) Has been skipped
CI/CD Pipeline / Build Production Web Image (push) Has been skipped
CI/CD Pipeline / Build Production Worker Image (push) Has been skipped
CI/CD Pipeline / CI Gate (push) Has been skipped
CI/CD Pipeline / Deploy Production (push) Has been skipped
CI/CD Pipeline / Canary Release to Production (push) Has been skipped
CI/CD Pipeline / Production Browser E2E (push) Has been skipped
CI/CD Pipeline / Check if frontend-only change (push) Has been skipped
CI/CD Pipeline / Check if frontend-only change (pull_request) Successful in 3s
CI/CD Pipeline / Check push changed paths (pull_request) Has been skipped
CI/CD Pipeline / PR Build API Image (push) Has been skipped
CI/CD Pipeline / PR Build Web Image (push) Has been skipped
CI/CD Pipeline / PR Build Worker Image (push) Has been skipped
CI/CD Pipeline / Dedup Check - skip PR tests when covered by push pipeline (pull_request) Successful in 8s
CI/CD Pipeline / Dedup Check - skip PR tests when covered by push pipeline (push) Successful in 13s
CI/CD Pipeline / Build Staging API Image (pull_request) Has been skipped
CI/CD Pipeline / Build Staging Web Image (pull_request) Has been skipped
CI/CD Pipeline / Build Staging Worker Image (pull_request) Has been skipped
CI/CD Pipeline / Check push changed paths (push) Successful in 13s
CI/CD Pipeline / Validate - Style (pull_request) Has been skipped
CI/CD Pipeline / Validate - Security (pull_request) Has been skipped
CI/CD Pipeline / Validate - Python (mypy + alembic) (pull_request) Has been skipped
CI/CD Pipeline / Unit Tests (pull_request) Has been skipped
CI/CD Pipeline / Integration Tests (pull_request) Has been skipped
CI/CD Pipeline / Frontend Lint (pull_request) Has been skipped
CI/CD Pipeline / Retag skipped Staging API Image (pull_request) Has been skipped
CI/CD Pipeline / Frontend Unit Tests (pull_request) Has been skipped
CI/CD Pipeline / Retag skipped Staging Web Image (pull_request) Has been skipped
CI/CD Pipeline / Retag skipped Staging Worker Image (pull_request) Has been skipped
CI/CD Pipeline / Frontend Lint (push) Has been skipped
CI/CD Pipeline / PR Build API Image (pull_request) Successful in 1m13s
CI/CD Pipeline / PR Build Worker Image (pull_request) Successful in 1m15s
CI/CD Pipeline / Deploy Staging (Watchtower auto-deploy) (pull_request) Has been skipped
CI/CD Pipeline / Build Production API Image (pull_request) Has been skipped
CI/CD Pipeline / Build Production Web Image (pull_request) Has been skipped
CI/CD Pipeline / Build Production Worker Image (pull_request) Has been skipped
CI/CD Pipeline / Staging E2E Tests (pull_request) Has been skipped
CI/CD Pipeline / Staging API Integration Tests (pull_request) Has been skipped
CI/CD Pipeline / Deploy Production (pull_request) Has been skipped
CI/CD Pipeline / Production Browser E2E (pull_request) Has been skipped
CI/CD Pipeline / Build Staging API Image (push) Successful in 59s
CI/CD Pipeline / Canary Release to Production (pull_request) Has been skipped
CI/CD Pipeline / ACR Image Cleanup (pull_request) Has been skipped
CI/CD Pipeline / Build Staging Worker Image (push) Successful in 42s
CI/CD Pipeline / PR Build Web Image (pull_request) Successful in 1m58s
CI/CD Pipeline / CI Gate (pull_request) Successful in 3s
CI/CD Pipeline / Build Staging Web Image (push) Successful in 1m26s
CI/CD Pipeline / Retag skipped Staging API Image (push) Has been skipped
CI/CD Pipeline / Retag skipped Staging Web Image (push) Has been skipped
CI/CD Pipeline / Retag skipped Staging Worker Image (push) Has been skipped
CI/CD Pipeline / Integration Tests (push) Successful in 2m54s
PR Automation / Auto Approve on CI Green (pull_request) Successful in 3m23s
PR Automation / Auto Merge on CI Green + Approved (pull_request) Has been skipped
CI/CD Pipeline / Deploy Staging (Watchtower auto-deploy) (push) Successful in 1m32s
CI/CD Pipeline / Validate - Style (push) Successful in 3m42s
Preview Deploy / Deploy Preview Environment (pull_request) Successful in 4m39s
CI/CD Pipeline / Validate - Security (push) Successful in 4m45s
CI/CD Pipeline / Frontend Unit Tests (push) Successful in 5m7s
CI/CD Pipeline / ACR Image Cleanup (push) Successful in 1m55s
CI/CD Pipeline / Validate - Python (mypy + alembic) (push) Successful in 5m34s
CI/CD Pipeline / Staging API Integration Tests (push) Successful in 3m12s
AI Code Review / AI Code Review (pull_request) Successful in 7m7s
CI/CD Pipeline / Staging E2E Tests (push) Failing after 3m38s
CI/CD Pipeline / Unit Tests (push) Successful in 10m29s
CI/CD Pipeline / Build Production API Image (push) Has been skipped
CI/CD Pipeline / Build Production Web Image (push) Has been skipped
CI/CD Pipeline / Build Production Worker Image (push) Has been skipped
CI/CD Pipeline / CI Gate (push) Has been skipped
CI/CD Pipeline / Deploy Production (push) Has been skipped
CI/CD Pipeline / Canary Release to Production (push) Has been skipped
CI/CD Pipeline / Production Browser E2E (push) Has been skipped
Co-authored-by: xiaoxia <dev@xiaoxiajianji.com> Co-committed-by: xiaoxia <dev@xiaoxiajianji.com>
This commit was merged in pull request #2133.
This commit is contained in:
@@ -215,6 +215,10 @@ DOUBAO_MODEL=doubao-seed-1-6-250615
|
||||
DOUBAO_BASE_URL=https://ark.cn-beijing.volces.com/api/v3
|
||||
DOUBAO_TIMEOUT=30
|
||||
DOUBAO_MAX_RETRIES=2
|
||||
# 视觉模型:pro 精度高,lite 速度快(viral-video 商品识别默认用 lite 提速)
|
||||
DOUBAO_VISION_MODEL=doubao-1-5-vision-pro-250915
|
||||
DOUBAO_VISION_LITE_MODEL=doubao-1-5-vision-lite-250915
|
||||
DOUBAO_VISION_USE_LITE=true
|
||||
|
||||
# ==================== 积分/会员系统 (#1895) ====================
|
||||
# 积分系统总开关:默认 false(暂停积分系统)。
|
||||
|
||||
@@ -23,6 +23,7 @@ import json
|
||||
import logging
|
||||
import os
|
||||
import tempfile
|
||||
from concurrent.futures import ThreadPoolExecutor, as_completed
|
||||
from pathlib import Path
|
||||
|
||||
from celery import Task, shared_task
|
||||
@@ -40,6 +41,7 @@ from packages.domain.viral_video import (
|
||||
ViralVideoStage,
|
||||
ViralVideoStatus,
|
||||
)
|
||||
from packages.shared import get_shared_settings
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
@@ -136,29 +138,41 @@ def _empty_copy_result(duration: int = 15, ratio: str = "9:16") -> dict:
|
||||
# ── 流水线各步骤 ────────────────────────────────────────────────────────
|
||||
|
||||
|
||||
_IMAGE_ANALYSIS_PROMPT = """请仔细观察这张图片,只基于图片中真实可见的内容进行分析,不要凭空想象。
|
||||
_IMAGE_ANALYSIS_SYSTEM_PROMPT = """你是资深电商商品视觉分析师,擅长从商品图片中提取结构化商品信息。严格遵守以下规则:
|
||||
|
||||
必须输出严格的 JSON(不要 Markdown 代码块,不要额外解释),字段如下:
|
||||
1. 只基于图片中真实可见的内容进行分析,严禁编造图片中不存在的品牌、文字、规格、卖点或功效。
|
||||
2. 看不清、包装上没有、无法判断的字段统一填「无法判断」。
|
||||
3. 包装上的文字只 OCR 你能清晰看到的,模糊的不要瞎猜。
|
||||
4. 输出必须是严格 JSON(不要 Markdown 代码块,不要额外解释文字)。
|
||||
|
||||
字段说明:
|
||||
{
|
||||
"category": "产品大类,如护肤品/彩妆/食品/数码/服饰/家居等,若无法识别填『无法判断』",
|
||||
"name": "产品名称(从包装/品牌/logo/文字推断;没有品牌时描述外观如『粉色包装面霜』)",
|
||||
"brand": "品牌名(看 logo/包装文字;看不清填『未知』)",
|
||||
"colors": ["主体颜色"],
|
||||
"material_or_texture": "材质/质地描述(如玻璃瓶装/塑料软管/哑光质感/金属外壳等;无法判断填『无法判断』)",
|
||||
"name": "商品全名(从包装 OCR 读出品牌+产品名+规格/容量/型号,如『李字无烟檀香型蚊香 30单盘装』)",
|
||||
"brand": "品牌名(从 Logo/包装文字读出,如『李字』『奥妙(OMO)』;看不清填『无法判断』)",
|
||||
"category": "商品品类(如『家用清洁/洗衣液』『日化驱蚊/盘式蚊香』『护肤/面霜』;非产品图填『非产品图』)",
|
||||
"spec": "规格/型号/容量/净含量(从包装文字 OCR 读出,如『30单盘装』『3kg』『500ml』;无法判断填『无法判断』)",
|
||||
"appearance": "外观与视觉特征(80-150字自然语言描述整体形状、颜色搭配、材质质感、尺寸感知,让没看到图的人能想象长什么样)",
|
||||
"packaging": "包装设计细节(80-150字自然语言描述标签颜色分区、图案元素、图形标识、封口/塑封状态、瓶盖/泵头/罐型等)",
|
||||
"text_on_package": ["包装上能清晰 OCR 读出的文字逐条列出,按主次:品牌名、产品名、卖点文案、功效描述、规格参数等,看不清的不列"],
|
||||
"colors": ["主体颜色 2-4 个"],
|
||||
"material_or_texture": "材质/质地(如玻璃瓶装/塑料软管/哑光质感/金属外壳/纸塑复合袋等;无法判断填『无法判断』)",
|
||||
"key_features": [
|
||||
"3-5 条**图片中确实能看到**的外观特征/卖点描述(如『按压式泵头』『瓶身有金色装饰线』等),不要编图片里没有的功效"
|
||||
"3-6 条图片中确实能看到的外观特征/视觉卖点(如『按压式泵头设计』『瓶身有金色装饰线条』『罐身带塑封痕迹』等),不要写功效词(除非包装上明确印了)"
|
||||
],
|
||||
"visual_style": "视觉风格(如简约高端/粉嫩少女/国潮/科技感/生活方式实拍等)",
|
||||
"scene": "图片中的使用/展示场景(如白底棚拍/浴室场景/户外街拍/桌面静物等;纯白底填『白底产品图』)",
|
||||
"target_audience_hint": "从视觉推断的目标人群(如年轻女性/男性商务/亲子家庭等;不确定填『通用』)",
|
||||
"text_on_image": "图片上出现的可读文字(品牌名/Slogan/产品名等,没有则填『无』)"
|
||||
}
|
||||
"visual_style": "整体视觉风格(如简约高端/粉嫩少女/国潮/科技感/家庭温馨/生活方式实拍等)",
|
||||
"scene": "图片展示场景(如白底棚拍/浴室场景/户外街拍/桌面静物/手持实拍等;纯白底填『白底产品图』)",
|
||||
"suitable_scenes": ["基于商品类型推断的 2-4 个适用/使用场景,如『家庭日常清洁』『卧室夜间驱蚊』"],
|
||||
"target_audience": "从商品定位/包装风格推断的目标人群(如『家庭主妇/宝妈』『年轻租房群体』『男性商务人士』;不确定填『无法判断』)",
|
||||
"selling_points": ["3-6 条可用于短视频营销的卖点(结合视觉+包装文字推断,尽量贴近电商话术)"],
|
||||
"summary": "识别描述汇总:把上述各维度整合成一段 150-250 字的流畅中文自然段落,像电商详情页的商品介绍,口语化、有画面感,前端会直接展示这段文字"
|
||||
}"""
|
||||
|
||||
严格要求:
|
||||
1. 任何字段无法确认时填『无法判断』或『未知』,不要猜。
|
||||
2. key_features 只能描述图片里肉眼可见的物理外观,不要写『补水保湿』『抗衰老』这类功效词(除非包装上明确印了)。
|
||||
3. 如果图片完全不是产品图(比如风景/人像/截图),category 填『非产品图』,name 填实际看到的内容。
|
||||
"""
|
||||
_IMAGE_ANALYSIS_USER_PROMPT = """请分析这张商品图片,按约定 JSON 字段完整输出。重点:
|
||||
1. name/brand/spec/text_on_package 必须从图片包装上 OCR 读取,不要凭空编造;
|
||||
2. appearance/packaging 两个字段要具体细致,80-150 字自然语言;
|
||||
3. summary 字段务必连贯成一段 150-250 字的中文导购描述,方便前端直接展示;
|
||||
4. 非产品图时 category 填『非产品图』,name 填实际看到的内容,其余字段按需填『无法判断』;
|
||||
5. 所有无法判断的字段统一填「无法判断」。"""
|
||||
|
||||
|
||||
def _vision_fallback(idx: int, reason: str, extra: dict | None = None) -> dict:
|
||||
@@ -174,10 +188,63 @@ def _vision_fallback(idx: int, reason: str, extra: dict | None = None) -> dict:
|
||||
return d
|
||||
|
||||
|
||||
def _step_image_analysis(job: ViralVideoJob) -> dict:
|
||||
"""步骤 1: 图片 VLM 分析 — 识别产品特征、场景、卖点(加 None 防护)。"""
|
||||
def _analyze_single_image(
|
||||
idx: int,
|
||||
img_url: str,
|
||||
vision_model: str,
|
||||
timeout: int,
|
||||
) -> dict:
|
||||
"""单张图片 VLM 分析(线程池并行调用)。失败/None 返回 fallback dict。"""
|
||||
from packages.shared.ai_service import call_vision
|
||||
|
||||
if not img_url or not isinstance(img_url, str):
|
||||
return _vision_fallback(idx, "invalid_url")
|
||||
try:
|
||||
from packages.shared.ai_service import call_vision
|
||||
result = call_vision(
|
||||
image_url=img_url,
|
||||
prompt=_IMAGE_ANALYSIS_USER_PROMPT,
|
||||
model=vision_model,
|
||||
max_tokens=1024,
|
||||
temperature=0.2,
|
||||
timeout=timeout,
|
||||
system_prompt=_IMAGE_ANALYSIS_SYSTEM_PROMPT,
|
||||
)
|
||||
if result is None:
|
||||
logger.warning("[爆款视频] 图片 #%d call_vision 返回 None", idx)
|
||||
return _vision_fallback(idx, "vision_none")
|
||||
if isinstance(result, str):
|
||||
logger.warning("[爆款视频] 图片 #%d VLM 返回非 JSON: %s", idx, result[:200])
|
||||
return _vision_fallback(idx, "vision_text", {"_raw": result[:500]})
|
||||
if isinstance(result, dict):
|
||||
result.setdefault("_source", "vision")
|
||||
result.setdefault("name", "未识别")
|
||||
result.setdefault("brand", "无法判断")
|
||||
result.setdefault("category", "无法判断")
|
||||
result.setdefault("spec", "无法判断")
|
||||
result.setdefault("appearance", "无法判断")
|
||||
result.setdefault("packaging", "无法判断")
|
||||
result.setdefault("text_on_package", [])
|
||||
result.setdefault("colors", [])
|
||||
result.setdefault("material_or_texture", "无法判断")
|
||||
result.setdefault("key_features", [])
|
||||
result.setdefault("visual_style", "通用")
|
||||
result.setdefault("scene", "白底产品图")
|
||||
result.setdefault("suitable_scenes", [])
|
||||
result.setdefault("target_audience", "无法判断")
|
||||
result.setdefault("selling_points", [])
|
||||
result.setdefault("summary", "")
|
||||
return result
|
||||
logger.warning("[爆款视频] 图片 #%d VLM 返回意外类型 %s", idx, type(result))
|
||||
return _vision_fallback(idx, "vision_unexpected_type")
|
||||
except Exception as e:
|
||||
logger.warning("[爆款视频] 图片 #%d 分析失败 err=%s", idx, e, exc_info=True)
|
||||
return _vision_fallback(idx, "vision_exception", {"_error": str(e)[:200]})
|
||||
|
||||
|
||||
def _step_image_analysis(job: ViralVideoJob) -> dict:
|
||||
"""步骤 1: 图片 VLM 分析 — 识别产品特征(v1.6 优化:并行 + lite 模型提速)。"""
|
||||
try:
|
||||
from packages.shared.ai_service import call_vision # noqa: F401
|
||||
except ImportError:
|
||||
logger.warning("[爆款视频] ai_service.call_vision 不可用,使用占位结果")
|
||||
return {"products": [_vision_fallback(0, "fallback_import_error")]}
|
||||
@@ -186,36 +253,36 @@ def _step_image_analysis(job: ViralVideoJob) -> dict:
|
||||
logger.warning("[爆款视频] 任务无 images,跳过图片分析")
|
||||
return {"products": []}
|
||||
|
||||
results = []
|
||||
for idx, img_url in enumerate(job.images):
|
||||
if not img_url or not isinstance(img_url, str):
|
||||
logger.warning("[爆款视频] 图片 #%d URL 非法", idx)
|
||||
results.append(_vision_fallback(idx, "invalid_url"))
|
||||
continue
|
||||
logger.info("[爆款视频] 图片分析 #%d img=%s", idx, img_url[:160])
|
||||
try:
|
||||
result = call_vision(image_url=img_url, prompt=_IMAGE_ANALYSIS_PROMPT)
|
||||
if result is None:
|
||||
logger.warning("[爆款视频] 图片 #%d call_vision 返回 None", idx)
|
||||
results.append(_vision_fallback(idx, "vision_none"))
|
||||
elif isinstance(result, str):
|
||||
# VLM 返回了非 JSON 文本(JSON 解析失败),记录原始文本但不要让 None 传播
|
||||
logger.warning("[爆款视频] 图片 #%d VLM 返回非 JSON 文本: %s", idx, result[:200])
|
||||
results.append(_vision_fallback(idx, "vision_text", {"_raw": result[:500]}))
|
||||
elif isinstance(result, dict):
|
||||
result.setdefault("_source", "vision")
|
||||
# 防御:关键字段缺失则补默认
|
||||
result.setdefault("name", "未识别")
|
||||
result.setdefault("category", "无法判断")
|
||||
result.setdefault("key_features", [])
|
||||
result.setdefault("scene", "通用")
|
||||
results.append(result)
|
||||
else:
|
||||
logger.warning("[爆款视频] 图片 #%d VLM 返回意外类型 %s", idx, type(result))
|
||||
results.append(_vision_fallback(idx, "vision_unexpected_type"))
|
||||
except Exception as e:
|
||||
logger.warning("[爆款视频] 图片分析失败 img=%s err=%s", img_url[:120], e, exc_info=True)
|
||||
results.append(_vision_fallback(idx, "vision_exception", {"_error": str(e)[:200]}))
|
||||
# 选择视觉模型:lite 速度优先(doubao_vision_use_lite=True 默认),pro 备用
|
||||
try:
|
||||
_s = get_shared_settings()
|
||||
vision_model = _s.doubao_vision_lite_model if _s.doubao_vision_use_lite else _s.doubao_vision_model
|
||||
vision_timeout = 30 if _s.doubao_vision_use_lite else 60
|
||||
except Exception:
|
||||
vision_model = "doubao-1-5-vision-lite-250915"
|
||||
vision_timeout = 30
|
||||
|
||||
results: list[dict] = [None] * len(job.images) # type: ignore
|
||||
max_workers = min(4, max(1, len(job.images)))
|
||||
logger.info(
|
||||
"[爆款视频] 开始并行图片分析 n=%d model=%s timeout=%d workers=%d",
|
||||
len(job.images),
|
||||
vision_model,
|
||||
vision_timeout,
|
||||
max_workers,
|
||||
)
|
||||
with ThreadPoolExecutor(max_workers=max_workers) as pool:
|
||||
future_to_idx = {
|
||||
pool.submit(_analyze_single_image, idx, url, vision_model, vision_timeout): idx
|
||||
for idx, url in enumerate(job.images)
|
||||
}
|
||||
for fut in as_completed(future_to_idx):
|
||||
idx = future_to_idx[fut]
|
||||
try:
|
||||
results[idx] = fut.result()
|
||||
except Exception as e:
|
||||
logger.warning("[爆款视频] 图片 #%d future 异常 err=%s", idx, e, exc_info=True)
|
||||
results[idx] = _vision_fallback(idx, "future_exception", {"_error": str(e)[:200]})
|
||||
|
||||
return {"products": results}
|
||||
|
||||
@@ -319,6 +386,7 @@ _SCRIPT_GENERATION_PROMPT = """你是一名资深短视频导演,擅长为 AI
|
||||
- 画幅比例:{ratio}
|
||||
- 产品图片数量:{n_images} 张(将作为 reference_images 传给视频模型,第1张通常作为首帧/主产品图)
|
||||
- 参考风格(可选):{style_hint}
|
||||
- 爆款结构(用户指定,必须严格遵循):{viral_structure_block}
|
||||
|
||||
## 任务
|
||||
请撰写**一段完整的编导分镜脚本**,包含视频总览、场景光线、逐镜头时间轴、硬性约束、负面提示词,以及自然口语化的口播对白。
|
||||
@@ -371,10 +439,13 @@ _SCRIPT_GENERATION_PROMPT = """你是一名资深短视频导演,擅长为 AI
|
||||
5. **时长控制**:所有 shots 的 time_range 加起来必须等于 {duration} 秒,单镜 2-8 秒。
|
||||
6. **硬性约束和负面词必须包含**:不要删减,可根据产品类型追加。
|
||||
7. **voiceover_script 必须是纯口播文本**:不含任何标记、括号、说明,字数按中文每秒 3-4 字估算({duration}秒约{approx_chars}字)。
|
||||
8. **严格遵循爆款结构**:如果上方「爆款结构」字段不为「未指定」,必须严格按该结构的节奏/段落顺序编排文案与镜头,开场钩子、痛点、反转、案例、行动号召等节点要与结构对应,不要打乱顺序。
|
||||
"""
|
||||
|
||||
|
||||
def _build_products_summary(image_analysis: dict) -> str:
|
||||
"""把 VLM 返回的商品分析结果拼给文案/分镜生成 prompt 用。
|
||||
优先用 summary(自然段落);没有时用结构化字段兜底拼一段。"""
|
||||
products = (image_analysis or {}).get("products", []) or []
|
||||
if not products:
|
||||
return "- (无图片信息,请自由创作自然生活化场景)"
|
||||
@@ -383,34 +454,60 @@ def _build_products_summary(image_analysis: dict) -> str:
|
||||
if not isinstance(p, dict):
|
||||
continue
|
||||
name = p.get("name") or "产品"
|
||||
# 优先 VLM 生成的 summary 段(自然语言,给编导模型看效果最好)
|
||||
summary = (p.get("summary") or "").strip()
|
||||
if summary and len(summary) >= 30:
|
||||
lines.append(f"- 图{i+1} {name}:{summary}")
|
||||
continue
|
||||
# 结构化字段兜底
|
||||
brand = p.get("brand") or ""
|
||||
cat = p.get("category") or ""
|
||||
spec = p.get("spec") or ""
|
||||
appearance = p.get("appearance") or ""
|
||||
packaging = p.get("packaging") or ""
|
||||
colors = p.get("colors") or []
|
||||
mat = p.get("material_or_texture") or ""
|
||||
style = p.get("visual_style") or ""
|
||||
scene = p.get("scene") or ""
|
||||
audience = p.get("target_audience_hint") or ""
|
||||
text_on_img = p.get("text_on_image") or ""
|
||||
audience = p.get("target_audience") or p.get("target_audience_hint") or ""
|
||||
# text_on_package 可能是数组(新格式)或字符串(旧格式)
|
||||
text_list = p.get("text_on_package") or []
|
||||
if isinstance(text_list, str):
|
||||
text_on_img = text_list
|
||||
else:
|
||||
text_on_img = ";".join([str(x) for x in text_list[:8]]) if text_list else (p.get("text_on_image") or "")
|
||||
feats = p.get("key_features") or p.get("features") or []
|
||||
sellings = p.get("selling_points") or []
|
||||
scenes = p.get("suitable_scenes") or []
|
||||
parts = [f"图{i+1} {name}"]
|
||||
if brand and brand not in ("未知", "无法判断"):
|
||||
parts.append(f"品牌={brand}")
|
||||
if cat and cat not in ("无法判断", "非产品图"):
|
||||
parts.append(f"品类={cat}")
|
||||
if spec and spec != "无法判断":
|
||||
parts.append(f"规格={spec}")
|
||||
if appearance and appearance != "无法判断":
|
||||
parts.append(f"外观={appearance}")
|
||||
if packaging and packaging != "无法判断":
|
||||
parts.append(f"包装={packaging}")
|
||||
if colors:
|
||||
parts.append(f"颜色={','.join(colors)}")
|
||||
if mat and mat not in ("无法判断",):
|
||||
if mat and mat != "无法判断":
|
||||
parts.append(f"材质={mat}")
|
||||
if style:
|
||||
parts.append(f"风格={style}")
|
||||
if scene and scene not in ("通用",):
|
||||
parts.append(f"场景={scene}")
|
||||
if audience and audience != "通用":
|
||||
parts.append(f"展示场景={scene}")
|
||||
if scenes:
|
||||
parts.append(f"适用场景={','.join([str(x) for x in scenes[:4]])}")
|
||||
if audience and audience not in ("通用", "无法判断"):
|
||||
parts.append(f"目标人群={audience}")
|
||||
if text_on_img and text_on_img not in ("无",):
|
||||
parts.append(f"图片文字={text_on_img}")
|
||||
parts.append(f"包装文字={text_on_img[:300]}")
|
||||
if feats:
|
||||
parts.append("外观特征=" + ";".join([str(x) for x in feats[:6]]))
|
||||
if sellings:
|
||||
parts.append("营销卖点=" + ";".join([str(x) for x in sellings[:5]]))
|
||||
lines.append("- " + ",".join(parts))
|
||||
return "\n".join(lines)
|
||||
|
||||
@@ -594,6 +691,13 @@ def _step_script_generation(job: ViralVideoJob, intent: dict, image_analysis: di
|
||||
intent_str = "推广产品"
|
||||
tone = "亲切自然"
|
||||
|
||||
# 爆款结构:用户在 STEP1/STEP2 选的中文结构名,必须严格注入 prompt 指导 AI 编排
|
||||
_vs = (job.viral_structure or "").strip()
|
||||
if _vs:
|
||||
viral_structure_block = f"【{_vs}】—— 请严格按照这个爆款结构的节奏/段落顺序编排镜头、台词和情绪节点(开场钩子、痛点、反转、案例、行动号召等按结构走),不要打乱顺序"
|
||||
else:
|
||||
viral_structure_block = "未指定(自由编排,但仍需有钩子开头+产品展示+行动号召的基本节奏)"
|
||||
|
||||
prompt = _SCRIPT_GENERATION_PROMPT.format(
|
||||
products_summary=products_summary,
|
||||
intent=intent_str,
|
||||
@@ -606,6 +710,7 @@ def _step_script_generation(job: ViralVideoJob, intent: dict, image_analysis: di
|
||||
n_images=len(job.images or []),
|
||||
style_hint=style_hint,
|
||||
approx_chars=approx_chars,
|
||||
viral_structure_block=viral_structure_block,
|
||||
)
|
||||
|
||||
try:
|
||||
|
||||
@@ -94,7 +94,9 @@ class SharedSettings(BaseSettings):
|
||||
doubao_base_url: str = "https://ark.cn-beijing.volces.com/api/v3"
|
||||
doubao_timeout: int = 30
|
||||
doubao_max_retries: int = 2
|
||||
doubao_vision_model: str = "doubao-1-5-vision-pro-250915"
|
||||
doubao_vision_model: str = "doubao-1-5-vision-pro-250915" # 高精度视觉(备用)
|
||||
doubao_vision_lite_model: str = "doubao-1-5-vision-lite-250915" # 快速视觉(商品识别默认,速度优先)
|
||||
doubao_vision_use_lite: bool = True # viral-video 图片分析默认用 lite 提速
|
||||
doubao_embedding_model: str = "doubao-embedding-large-text-240915"
|
||||
doubao_video_model: str = "doubao-seedance-2-5-260628"
|
||||
doubao_video_timeout: int = 600 # 视频生成轮询总超时(秒)
|
||||
|
||||
@@ -40,6 +40,7 @@ class DoubaoClient:
|
||||
self.timeout: int = settings.doubao_timeout
|
||||
self.max_retries: int = settings.doubao_max_retries
|
||||
self.vision_model: str = settings.doubao_vision_model
|
||||
self.vision_lite_model: str = settings.doubao_vision_lite_model
|
||||
|
||||
def embed_text(self, text: str, timeout: int | None = None) -> list[float] | None:
|
||||
"""调用豆包文本 Embedding API,返回浮点向量;失败返回 None。"""
|
||||
@@ -154,6 +155,7 @@ class DoubaoClient:
|
||||
max_tokens: int = 2048,
|
||||
temperature: float = 0.3,
|
||||
timeout: int | None = None,
|
||||
model: str | None = None,
|
||||
) -> Optional[str]:
|
||||
"""调用豆包视觉理解 API(OpenAI 兼容多模态格式).
|
||||
|
||||
@@ -204,7 +206,7 @@ class DoubaoClient:
|
||||
"Content-Type": "application/json",
|
||||
}
|
||||
payload: dict[str, Any] = {
|
||||
"model": self.vision_model,
|
||||
"model": model or self.vision_model,
|
||||
"messages": vision_messages,
|
||||
"temperature": temperature,
|
||||
"max_tokens": max_tokens,
|
||||
|
||||
@@ -514,14 +514,26 @@ def call_llm(prompt: str, temperature: float = 0.7) -> object:
|
||||
return raw
|
||||
|
||||
|
||||
def call_vision(image_url: str, prompt: str) -> object:
|
||||
def call_vision(
|
||||
image_url: str,
|
||||
prompt: str,
|
||||
*,
|
||||
model: str | None = None,
|
||||
max_tokens: int = 1024,
|
||||
temperature: float = 0.2,
|
||||
timeout: int = 45,
|
||||
system_prompt: str | None = None,
|
||||
) -> object:
|
||||
"""调用豆包视觉大模型分析图片,返回解析后的 JSON 或原文字符串;失败返回 None。
|
||||
|
||||
Bug #2114 (VLM 牛头不对马嘴根因修复):
|
||||
之前误走 client.chat_completion(用文本模型 doubao-seed-1.6),多模态 content list 被当成
|
||||
纯文本发给文本模型 → 模型要么看不到图、要么抛 400,静默被 except 吞掉 → 返回 None →
|
||||
_step_image_analysis fallback 到 {"name":"未识别"} → 后续文案/分镜完全没图的信息。
|
||||
现改走 vision_completion,走视觉模型 doubao-1-5-vision-pro-250915。
|
||||
Args:
|
||||
image_url: 可公网访问的图片 URL(直接传给豆包视觉模型,无需本地下载)。
|
||||
prompt: 用户侧文本提示。
|
||||
model: 覆盖默认视觉模型(如 vision_lite_model 提速用),None 走配置默认。
|
||||
max_tokens: 输出上限,商品识别用 800~1200 足够,避免长输出拖慢首 token。
|
||||
temperature: 温度。
|
||||
timeout: 单次请求超时(秒)。
|
||||
system_prompt: 覆盖默认 system prompt(viral-video 商品分析会传专门的详细 prompt)。
|
||||
"""
|
||||
client = get_doubao_client()
|
||||
if not client.is_available:
|
||||
@@ -531,28 +543,33 @@ def call_vision(image_url: str, prompt: str) -> object:
|
||||
logger.warning("[call_vision] 空 image_url,跳过视觉分析")
|
||||
return None
|
||||
|
||||
system_prompt = (
|
||||
"你是资深电商视觉分析师。请严格基于用户提供的图片观察回答,"
|
||||
"图片里没有的信息不要凭空想象或编造;看不清或无法判断时明确说"
|
||||
"「图片中无法判断」,不要猜测。输出必须是严格 JSON,不要附加 Markdown 或解释文字。"
|
||||
)
|
||||
if system_prompt is None:
|
||||
system_prompt = (
|
||||
"你是资深电商视觉分析师。请严格基于用户提供的图片观察回答,"
|
||||
"图片里没有的信息不要凭空想象或编造;看不清或无法判断时明确说"
|
||||
"「无法判断」,不要猜测。输出必须是严格 JSON,不要附加 Markdown 或解释文字。"
|
||||
)
|
||||
messages = [
|
||||
{"role": "system", "content": system_prompt},
|
||||
{"role": "user", "content": prompt},
|
||||
]
|
||||
|
||||
used_model = model or getattr(client, "vision_model", "?")
|
||||
logger.info(
|
||||
"[call_vision] 调用豆包视觉模型 vision_model=%s image_url=%s prompt_len=%d",
|
||||
getattr(client, "vision_model", "?"),
|
||||
"[call_vision] 调用豆包视觉模型 model=%s image_url=%s prompt_len=%d max_tokens=%d timeout=%d",
|
||||
used_model,
|
||||
image_url[:120],
|
||||
len(prompt),
|
||||
max_tokens,
|
||||
timeout,
|
||||
)
|
||||
raw = client.vision_completion(
|
||||
messages=messages,
|
||||
images=[image_url],
|
||||
temperature=0.2,
|
||||
max_tokens=2048,
|
||||
timeout=60,
|
||||
temperature=temperature,
|
||||
max_tokens=max_tokens,
|
||||
timeout=timeout,
|
||||
model=model,
|
||||
)
|
||||
if raw is None:
|
||||
logger.warning("[call_vision] 视觉模型返回 None (image_url=%s)", image_url[:80])
|
||||
|
||||
@@ -57,7 +57,7 @@ if [ "$TARGET_ENV" = "staging" ]; then
|
||||
fi
|
||||
|
||||
# 共用 secrets 直接导出(如果存在)
|
||||
SHARED_SECRETS="OSS_ACCESS_KEY_ID OSS_ACCESS_KEY_SECRET COSYVOICE_API_KEY DASHSCOPE_API_KEY MEDIAKIT_API_KEY DOUBAO_API_KEY DOUBAO_MODEL DOUBAO_BASE_URL DOUBAO_VISION_MODEL WECHAT_APP_ID WECHAT_APP_SECRET TIKHUB_API_KEY APIZERO_API_KEY GPU_WORKER_TOKEN"
|
||||
SHARED_SECRETS="OSS_ACCESS_KEY_ID OSS_ACCESS_KEY_SECRET COSYVOICE_API_KEY DASHSCOPE_API_KEY MEDIAKIT_API_KEY DOUBAO_API_KEY DOUBAO_MODEL DOUBAO_BASE_URL DOUBAO_VISION_MODEL DOUBAO_VISION_LITE_MODEL DOUBAO_VISION_USE_LITE WECHAT_APP_ID WECHAT_APP_SECRET TIKHUB_API_KEY APIZERO_API_KEY GPU_WORKER_TOKEN"
|
||||
for var in $SHARED_SECRETS; do
|
||||
value="${!var:-}"
|
||||
# 已经在环境中了,无需额外操作
|
||||
|
||||
@@ -18,6 +18,8 @@ class _FakeSettings:
|
||||
doubao_timeout = 10
|
||||
doubao_max_retries = 0
|
||||
doubao_vision_model = "test-vision"
|
||||
doubao_vision_lite_model = "test-vision-lite"
|
||||
doubao_vision_use_lite = False
|
||||
doubao_embedding_model = "test-embedding"
|
||||
|
||||
|
||||
|
||||
@@ -44,7 +44,7 @@ class TestCheckDatabase:
|
||||
assert result["type"] == "postgresql"
|
||||
assert result["message"] == "Database connection successful"
|
||||
mock_psycopg.connect.assert_called_once_with(
|
||||
"postgresql+psycopg://test:test@localhost/test", connect_timeout=3
|
||||
"postgresql://test:test@localhost/test", connect_timeout=3
|
||||
)
|
||||
mock_cur.execute.assert_called_once_with("SELECT 1")
|
||||
mock_conn.close.assert_called_once()
|
||||
@@ -96,7 +96,7 @@ class TestCheckMigrations:
|
||||
assert result["status"] == "healthy"
|
||||
assert result["message"] == "Database migrations applied"
|
||||
mock_psycopg.connect.assert_called_once_with(
|
||||
"postgresql+psycopg://test:test@localhost/test", connect_timeout=3
|
||||
"postgresql://test:test@localhost/test", connect_timeout=3
|
||||
)
|
||||
mock_conn.close.assert_called_once()
|
||||
|
||||
|
||||
Reference in New Issue
Block a user