Files
xiaoxia-saas/apps/worker/worker_app/tasks/vision/assembler.py
T
xiaoxia ae2ddc2af2
CI/CD Pipeline / Check push changed paths (pull_request) Has been cancelled
CI/CD Pipeline / Build Staging API Image (pull_request) Has been cancelled
CI/CD Pipeline / Build Staging Web Image (pull_request) Has been cancelled
CI/CD Pipeline / Build Staging Worker Image (pull_request) Has been cancelled
CI/CD Pipeline / Retag skipped Staging API Image (pull_request) Has been cancelled
CI/CD Pipeline / Retag skipped Staging Web Image (pull_request) Has been cancelled
CI/CD Pipeline / Retag skipped Staging Worker Image (pull_request) Has been cancelled
CI/CD Pipeline / Deploy Staging (Watchtower auto-deploy) (pull_request) Has been cancelled
CI/CD Pipeline / Staging E2E Tests (pull_request) Has been cancelled
CI/CD Pipeline / Staging API Integration Tests (pull_request) Has been cancelled
CI/CD Pipeline / Build Production API Image (pull_request) Has been cancelled
CI/CD Pipeline / Build Production Web Image (pull_request) Has been cancelled
CI/CD Pipeline / Build Production Worker Image (pull_request) Has been cancelled
CI/CD Pipeline / Deploy Production (pull_request) Has been cancelled
CI/CD Pipeline / Dedup Check - skip PR tests when covered by push pipeline (pull_request) Has been cancelled
CI/CD Pipeline / Check if frontend-only change (pull_request) Has been cancelled
CI/CD Pipeline / Validate - Style (pull_request) Has been cancelled
CI/CD Pipeline / Validate - Security (pull_request) Has been cancelled
CI/CD Pipeline / Validate - Python (mypy + alembic) (pull_request) Has been cancelled
CI/CD Pipeline / Unit Tests (pull_request) Has been cancelled
CI/CD Pipeline / Integration Tests (pull_request) Has been cancelled
CI/CD Pipeline / Frontend Lint (pull_request) Has been cancelled
CI/CD Pipeline / Frontend Unit Tests (pull_request) Has been cancelled
CI/CD Pipeline / PR Build API Image (pull_request) Has been cancelled
CI/CD Pipeline / PR Build Web Image (pull_request) Has been cancelled
CI/CD Pipeline / PR Build Worker Image (pull_request) Has been cancelled
CI/CD Pipeline / Production Browser E2E (pull_request) Has been cancelled
CI/CD Pipeline / ACR Image Cleanup (pull_request) Has been cancelled
CI/CD Pipeline / Canary Release to Production (pull_request) Has been cancelled
CI/CD Pipeline / CI Gate (pull_request) Has been cancelled
AI Code Review / AI Code Review (pull_request) Has been cancelled
PR Automation / Auto Approve on CI Green (pull_request) Has been cancelled
PR Automation / Auto Merge on CI Green + Approved (pull_request) Has been cancelled
Preview Deploy / Deploy Preview Environment (pull_request) Has been cancelled
fix(vision): assembler portrait_prompt 拼接自然化 + 清死代码三元
- _AGE_PREFIX 去掉儿童/青少年死代码(已在 _person_subject 里单独按性别处理)
- 风格段(style/mood/scene)之间不用逗号,改为紧凑拼接(休闲阳光街拍风格而非'休闲,阳光,街拍风格')
2026-10-05 17:25:01 +08:00

308 lines
9.7 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
# -*- coding: utf-8 -*-
"""把 fast_json VLM 输出 + OCR 文本组装为与旧 _normalize() 完全一致的 dict。
目标:下游(信任链t2i/intent_parsing/script_generation)零改动。
必出字段:name, brand, category, appearance, packaging, text_on_package,
key_features, scene, mood, portrait_prompt, summary, _source
"""
from __future__ import annotations
from typing import Any
# ---------- portrait_prompt 模板 ----------
# 目标:60-100 字的人物穿搭描述,用于 Seedream 纯文生图。要求具体、风格化、视觉细节丰富。
# 旧 VLM 输出格式参考:"一位25岁左右的亚洲女性,身穿白色V领短袖T恤,黑色高腰阔腿裤,
# 搭配银色项链,长发披肩,表情自信,街拍风格,阳光明媚的城市街头"
def _join_parts(*parts: str | None) -> str:
return "".join(p for p in parts if p)
_AGE_PREFIX = {
"青年": "年轻",
"中年": "中年",
"老年": "老年",
}
# gender 后缀
_GENDER_WORD = {"男": "男性", "女": "女性"}
def _person_subject(fj: dict[str, Any]) -> str:
"""人物主语:年轻女性 / 中年男性 / 少女 / 小男孩 / 人物 等。"""
gender = fj.get("gender") or ""
age = fj.get("age_range") or ""
gw = _GENDER_WORD.get(gender, "")
if age == "儿童":
if gender == "女":
return "小女孩"
if gender == "男":
return "小男孩"
return "儿童"
if age == "青少年":
if gender == "女":
return "少女"
if gender == "男":
return "少年"
return "青少年"
prefix = _AGE_PREFIX.get(age, "")
if gw:
return f"{prefix}{gw}" if prefix else gw
return f"{prefix}人物" if prefix else "人物"
def _build_wear_sentence(fj: dict[str, Any]) -> str:
"""穿搭段:上装+下装/连衣裙,带颜色+材质+图案。"""
upper = fj.get("upper_wear") or ""
upper_color = fj.get("upper_color") or ""
lower = fj.get("lower_wear") or ""
lower_color = fj.get("lower_color") or ""
dress_color = fj.get("dress_color") or ""
material = fj.get("material") or ""
pattern = fj.get("pattern") or ""
is_dress = ("连衣裙" in upper) or ("裙" in upper and not lower)
if is_dress:
c = dress_color or upper_color
wear = f"{c}{upper}" if c else upper
if material and material not in wear:
wear = f"{material}{wear}"
if pattern and pattern not in wear and pattern != "纯色":
wear += f",{pattern}图案"
return f"身穿{wear}"
parts: list[str] = []
if upper:
up = f"{upper_color}{upper}" if upper_color else upper
if material and material not in up:
up = f"{material}{up}"
if pattern and pattern != "纯色" and pattern not in up:
up += f"({pattern})"
parts.append(f"上身{up}" if up else "")
if lower:
lo = f"{lower_color}{lower}" if lower_color else lower
parts.append(f"下身{lo}" if lo else "")
return ",".join(p for p in parts if p)
def _build_portrait_prompt(fj: dict[str, Any]) -> str:
"""组装最终 portrait_prompt(目标 60-100 字,用于 Seedream 纯文生图)。"""
if not fj.get("has_person"):
# 非人像:用商品+场景+mood 拼一段
name = fj.get("product_name") or "商品"
brand = fj.get("brand") or ""
colors = fj.get("colors") or []
style = fj.get("style") or ""
scene = fj.get("scene") or ""
mood = fj.get("mood") or ""
pieces = []
if brand:
pieces.append(brand)
pieces.append(name)
if colors:
pieces.append("、".join(colors[:3]) + "配色")
if style:
pieces.append(style + "风格")
if mood:
pieces.append(mood + "氛围")
if scene and scene not in ("通用",):
pieces.append(scene + "场景")
pieces.append("产品特写")
prompt = ",".join(p for p in pieces if p)
return prompt if len(prompt) >= 10 else "产品展示图,特写镜头"
subject = _person_subject(fj)
wear = _build_wear_sentence(fj)
accessories = fj.get("accessories") or []
if isinstance(accessories, str):
accessories = [accessories]
acc_str = ""
if accessories:
acc_str = ",佩戴" + "、".join(str(a) for a in accessories if a)
hairstyle = fj.get("hairstyle") or ""
expression = fj.get("expression") or ""
pose = fj.get("pose") or ""
style = fj.get("style") or ""
scene = fj.get("scene") or ""
mood = fj.get("mood") or ""
detail_parts: list[str] = []
if hairstyle:
detail_parts.append(hairstyle)
if expression and expression not in ("自然", "平静"):
detail_parts.append(f"神情{expression}")
if pose and pose not in ("站立",):
detail_parts.append(pose)
style_parts: list[str] = []
if style:
style_parts.append(style)
if mood:
style_parts.append(mood)
if scene and scene not in ("通用",):
style_parts.append(scene)
pieces = [f"一位{subject}"]
if wear:
pieces.append(wear)
if acc_str:
pieces.append(acc_str.lstrip(","))
if detail_parts:
pieces.append(",".join(detail_parts))
if style_parts:
# 风格词之间不用逗号,用空格紧凑
pieces.append("".join(style_parts) + "风格")
else:
pieces.append("人像写真")
full = ",".join(p for p in pieces if p)
# 过短补充镜头词
if len(full) < 40:
full += ",自然光线下人像特写,画面清晰"
# 过长截断
if len(full) > 120:
full = full[:120].rstrip(",") + "。"
return full
# ---------- 商品字段 ----------
def _infer_name(fj: dict[str, Any], ocr_texts: list[str]) -> str:
pname = fj.get("product_name")
if pname and pname != "未识别":
return str(pname)
# 人物图 → name 用穿搭主件
if fj.get("has_person"):
up = fj.get("upper_wear") or ""
if "连衣裙" in up:
return up
return up or "人物穿搭"
if ocr_texts:
# 商品名可能是 OCR 最长的一行(品牌/产品名)
return max(ocr_texts, key=len)
return "未识别"
def _infer_brand(fj: dict[str, Any], ocr_texts: list[str]) -> str:
brand = fj.get("brand")
if brand:
return str(brand)
# OCR 里短的、纯字母/汉字短串可能是 brand
for t in ocr_texts:
if 1 < len(t) <= 12:
return t
return "无法判断"
def _infer_category(fj: dict[str, Any]) -> str:
cat = fj.get("category")
if cat:
return str(cat)
if fj.get("has_person"):
return "服饰"
return "非产品图"
def _build_appearance(fj: dict[str, Any]) -> str:
"""外观描述:颜色+款式+材质+图案 拼成一段。"""
parts: list[str] = []
for key, _label in [
("upper_color", "主色"),
("upper_wear", "款式"),
("material", "材质"),
("pattern", "图案"),
]:
v = fj.get(key)
if v and v not in ("无法判断", "未知", "纯色"):
parts.append(str(v))
if not parts:
if fj.get("has_person"):
return "人像穿搭整体造型"
return "无法判断"
return "、".join(parts)
def _build_key_features(fj: dict[str, Any], ocr_texts: list[str]) -> list[str]:
feats: list[str] = []
for key in (
"upper_wear",
"lower_wear",
"upper_color",
"lower_color",
"dress_color",
"material",
"pattern",
"style",
"accessories",
):
v = fj.get(key)
if not v:
continue
if isinstance(v, list):
feats.extend(str(x) for x in v if x)
elif isinstance(v, str) and v not in ("无法判断", "未知", "纯色"):
feats.append(v)
if ocr_texts:
feats.append(f"画面文字: {'/'.join(ocr_texts[:3])}")
# 去重
out: list[str] = []
seen: set[str] = set()
for f in feats:
f = f.strip()
if f and f not in seen and len(f) <= 30:
seen.add(f)
out.append(f)
return out[:6] if out else ["无法判断"]
def assemble_result(
idx: int,
fast_json: dict[str, Any] | None,
ocr_texts: list[str],
) -> dict[str, Any]:
"""把 fast_json 结果 + OCR 文本组装成下游兼容的 product dict。"""
fj = fast_json or {}
ocr_texts = ocr_texts or []
portrait_prompt = _build_portrait_prompt(fj)
name = _infer_name(fj, ocr_texts)
brand = _infer_brand(fj, ocr_texts)
category = _infer_category(fj)
appearance = _build_appearance(fj)
key_features = _build_key_features(fj, ocr_texts)
scene = fj.get("scene") or "通用"
mood = fj.get("mood") or ""
packaging = "无法判断" # 包装细节专用API无,保留占位
text_on_package = ocr_texts[:8]
summary = _build_summary(fj, name, brand, category)
return {
"name": name,
"brand": brand,
"category": category,
"appearance": appearance,
"packaging": packaging,
"text_on_package": text_on_package,
"key_features": key_features,
"scene": scene,
"mood": mood,
"portrait_prompt": portrait_prompt,
"summary": summary,
"_source": "v2_fast_json",
}
def _build_summary(fj: dict, name: str, brand: str, category: str) -> str:
if fj.get("has_person"):
up = fj.get("upper_wear") or "穿搭"
style = fj.get("style") or ""
base = f"{style}{up}" if style and style not in up else up
return base
if brand != "无法判断" and name != brand:
return f"{brand} {name}"
return name