fix(vision): #2205 thinking参数互斥修复——只传thinking=disabled #2205

Closed
xiaoxia wants to merge 4 commits from fix/vision-v2-thinking-param into develop
3 changed files with 50 additions and 42 deletions
@@ -187,10 +187,10 @@ def call_pro_vlm(
return None
text = _strip_code_fence(raw)
l, r = text.find("{"), text.rfind("}")
if l >= 0 and r > l:
lb, rb = text.find("{"), text.rfind("}")
if lb >= 0 and rb > lb:
try:
obj = json.loads(text[l : r + 1])
obj = json.loads(text[lb : rb + 1])
if isinstance(obj, dict):
logger.info("[vision.vlm] 图片 #%d pro VLM JSON 完成 elapsed=%.1fs", idx, elapsed)
return {
@@ -117,14 +117,9 @@ def call_fast_json(
"max_tokens": max_tokens,
"stream": False,
}
# 关键:关闭 thinking(避免产生 reasoning_tokens 拖慢响应)
# 方舟/豆包 2.x 模型支持 thinking.type=disabled
try:
payload["thinking"] = {"type": "disabled"}
except Exception:
pass
# 部分模型用 reasoning_effort 控制思考深度
payload["reasoning_effort"] = "low"
# 关键:关闭 thinking(reasoning_tokens 是延迟主因,单次要10-12s)
# 方舟/豆包 Seed 2.x 支持 thinking={type:"disabled"},且不要和 reasoning_effort 同时传(两者互斥会400)
payload["thinking"] = {"type": "disabled"}
resp = httpx.post(
url,
@@ -133,15 +128,12 @@ def call_fast_json(
timeout=timeout,
)
elapsed = time.time() - t0
if resp.status_code != 200:
logger.warning(
"[vision.v2] fast_json HTTP %d elapsed=%.1fs body=%s", resp.status_code, elapsed, resp.text[:200]
)
# 如果400说明不支持thinking参数,降级重试一次
if resp.status_code == 400 and "thinking" in resp.text.lower():
# 400 说明模型不支持 thinking 参数(极少数旧模型),重试一次不带 thinking
if resp.status_code == 400:
body_preview = resp.text[:300].lower()
logger.warning("[vision.v2] fast_json HTTP 400 elapsed=%.1fs body=%s", elapsed, resp.text[:200])
if "thinking" in body_preview or "reasoning" in body_preview:
payload.pop("thinking", None)
payload.pop("reasoning_effort", None)
time.time()
resp = httpx.post(
url,
headers={"Authorization": f"Bearer {api_key}", "Content-Type": "application/json"},
@@ -150,10 +142,15 @@ def call_fast_json(
)
elapsed = time.time() - t0
if resp.status_code != 200:
logger.warning("[vision.v2] fast_json 降级后 HTTP %d elapsed=%.1fs", resp.status_code, elapsed)
logger.warning("[vision.v2] fast_json 降级重试 HTTP %d elapsed=%.1fs", resp.status_code, elapsed)
return None
else:
return None
elif resp.status_code != 200:
logger.warning(
"[vision.v2] fast_json HTTP %d elapsed=%.1fs body=%s", resp.status_code, elapsed, resp.text[:200]
)
return None
data = resp.json()
raw = (data.get("choices") or [{}])[0].get("message", {}).get("content")
if raw is None:
@@ -175,10 +172,10 @@ def call_fast_json(
text = _strip_code_fence(raw)
# 截到第一个 { 和最后一个 } 之间,容忍前后偶发文字
l = text.find("{")
r = text.rfind("}")
if l >= 0 and r > l:
text = text[l : r + 1]
lb = text.find("{")
rb = text.rfind("}")
if lb >= 0 and rb > lb:
text = text[lb : rb + 1]
try:
obj = json.loads(text)
except json.JSONDecodeError:
+29 -18
View File
@@ -97,21 +97,28 @@ def invalidate_loader_cache():
class TestImageAnalysisWiring:
def test_uses_loader_template_and_xml_parse(self, job):
def test_v2_batch_analysis_returns_products(self, job):
"""V2 路径:_step_image_analysis 批量调用 analyze_images_v2,返回 products。"""
from apps.worker.worker_app.tasks import viral_video as vv
with patch("packages.shared.ai_service.call_vision", return_value=IMAGE_XML) as mock_v:
result = vv._analyze_single_image(0, "https://img/1.jpg", "vlm-lite", 15)
fake_product = {
"name": "lipstick",
"brand": "品牌X",
"category": "唇部彩妆",
"key_features": ["显白", "持久"],
"text_on_package": ["品牌X", "211"],
"_source": "v2_fast_json",
}
with patch("worker_app.tasks.vision.analyze_images_v2", return_value=[fake_product]) as mock_aiv2:
result = vv._step_image_analysis(job)
mock_v.assert_called_once()
# 验证调用时传入了 system_prompt(说明走了 loader 渲染的模板)
call_kwargs = mock_v.call_args.kwargs
assert "system_prompt" in call_kwargs and call_kwargs["system_prompt"]
# 结果包含从 XML 解析出的产品信息
assert result["name"] == "lipstick"
assert result["brand"] == "品牌X"
assert "显白" in result["key_features"]
assert result["text_on_package"] == ["品牌X", "211"]
mock_aiv2.assert_called_once()
products = result["products"]
assert len(products) == 1
assert products[0]["name"] == "lipstick"
assert products[0]["brand"] == "品牌X"
assert "显白" in products[0]["key_features"]
assert products[0]["text_on_package"] == ["品牌X", "211"]
# ── 2) 意图解析走模板 ───────────────────────────────────────────────
@@ -252,18 +259,22 @@ class TestEndToEndLoaderUsed:
called_types.append(prompt_type)
return real_get(prompt_type, **kwargs)
# V2 图片分析不再走 prompt_loader(固定 lite JSON prompt),用 mock 产品代替
img_res = {
"name": "lipstick",
"brand": "品牌X",
"key_features": ["显白", "持久"],
"text_on_package": ["品牌X"],
}
with (
patch.object(pl, "get_template", side_effect=spy_get),
patch("packages.shared.ai_service.call_vision", return_value=IMAGE_XML),
patch("packages.shared.ai_service.call_llm", return_value=INTENT_XML),
):
# 1) image
img_res = vv._analyze_single_image(0, "https://img/1.jpg", "vlm", 15)
# 2) intent
# intent
intent_res = vv._step_intent_parsing(job, {"products": [img_res]})
# 前两步分别调用了 image_analysis 和 intent_parsing
assert "image_analysis" in called_types
# V2 图片分析走固定 prompt(不经 loader);intent 仍走 loader
assert "image_analysis" not in called_types
assert "intent_parsing" in called_types
# script 和 review 单独验证(需要不同的 LLM 返回)