fix(vision): #2205 thinking参数互斥修复——只传thinking=disabled #2205
@@ -187,10 +187,10 @@ def call_pro_vlm(
|
||||
return None
|
||||
|
||||
text = _strip_code_fence(raw)
|
||||
l, r = text.find("{"), text.rfind("}")
|
||||
if l >= 0 and r > l:
|
||||
lb, rb = text.find("{"), text.rfind("}")
|
||||
if lb >= 0 and rb > lb:
|
||||
try:
|
||||
obj = json.loads(text[l : r + 1])
|
||||
obj = json.loads(text[lb : rb + 1])
|
||||
if isinstance(obj, dict):
|
||||
logger.info("[vision.vlm] 图片 #%d pro VLM JSON 完成 elapsed=%.1fs", idx, elapsed)
|
||||
return {
|
||||
|
||||
@@ -117,14 +117,9 @@ def call_fast_json(
|
||||
"max_tokens": max_tokens,
|
||||
"stream": False,
|
||||
}
|
||||
# 关键:关闭 thinking(避免产生 reasoning_tokens 拖慢响应)
|
||||
# 方舟/豆包 2.x 模型支持 thinking.type=disabled
|
||||
try:
|
||||
payload["thinking"] = {"type": "disabled"}
|
||||
except Exception:
|
||||
pass
|
||||
# 部分模型用 reasoning_effort 控制思考深度
|
||||
payload["reasoning_effort"] = "low"
|
||||
# 关键:关闭 thinking(reasoning_tokens 是延迟主因,单次要10-12s)
|
||||
# 方舟/豆包 Seed 2.x 支持 thinking={type:"disabled"},且不要和 reasoning_effort 同时传(两者互斥会400)
|
||||
payload["thinking"] = {"type": "disabled"}
|
||||
|
||||
resp = httpx.post(
|
||||
url,
|
||||
@@ -133,15 +128,12 @@ def call_fast_json(
|
||||
timeout=timeout,
|
||||
)
|
||||
elapsed = time.time() - t0
|
||||
if resp.status_code != 200:
|
||||
logger.warning(
|
||||
"[vision.v2] fast_json HTTP %d elapsed=%.1fs body=%s", resp.status_code, elapsed, resp.text[:200]
|
||||
)
|
||||
# 如果400说明不支持thinking参数,降级重试一次
|
||||
if resp.status_code == 400 and "thinking" in resp.text.lower():
|
||||
# 400 说明模型不支持 thinking 参数(极少数旧模型),重试一次不带 thinking
|
||||
if resp.status_code == 400:
|
||||
body_preview = resp.text[:300].lower()
|
||||
logger.warning("[vision.v2] fast_json HTTP 400 elapsed=%.1fs body=%s", elapsed, resp.text[:200])
|
||||
if "thinking" in body_preview or "reasoning" in body_preview:
|
||||
payload.pop("thinking", None)
|
||||
payload.pop("reasoning_effort", None)
|
||||
time.time()
|
||||
resp = httpx.post(
|
||||
url,
|
||||
headers={"Authorization": f"Bearer {api_key}", "Content-Type": "application/json"},
|
||||
@@ -150,10 +142,15 @@ def call_fast_json(
|
||||
)
|
||||
elapsed = time.time() - t0
|
||||
if resp.status_code != 200:
|
||||
logger.warning("[vision.v2] fast_json 降级后 HTTP %d elapsed=%.1fs", resp.status_code, elapsed)
|
||||
logger.warning("[vision.v2] fast_json 降级重试 HTTP %d elapsed=%.1fs", resp.status_code, elapsed)
|
||||
return None
|
||||
else:
|
||||
return None
|
||||
elif resp.status_code != 200:
|
||||
logger.warning(
|
||||
"[vision.v2] fast_json HTTP %d elapsed=%.1fs body=%s", resp.status_code, elapsed, resp.text[:200]
|
||||
)
|
||||
return None
|
||||
data = resp.json()
|
||||
raw = (data.get("choices") or [{}])[0].get("message", {}).get("content")
|
||||
if raw is None:
|
||||
@@ -175,10 +172,10 @@ def call_fast_json(
|
||||
|
||||
text = _strip_code_fence(raw)
|
||||
# 截到第一个 { 和最后一个 } 之间,容忍前后偶发文字
|
||||
l = text.find("{")
|
||||
r = text.rfind("}")
|
||||
if l >= 0 and r > l:
|
||||
text = text[l : r + 1]
|
||||
lb = text.find("{")
|
||||
rb = text.rfind("}")
|
||||
if lb >= 0 and rb > lb:
|
||||
text = text[lb : rb + 1]
|
||||
try:
|
||||
obj = json.loads(text)
|
||||
except json.JSONDecodeError:
|
||||
|
||||
@@ -97,21 +97,28 @@ def invalidate_loader_cache():
|
||||
|
||||
|
||||
class TestImageAnalysisWiring:
|
||||
def test_uses_loader_template_and_xml_parse(self, job):
|
||||
def test_v2_batch_analysis_returns_products(self, job):
|
||||
"""V2 路径:_step_image_analysis 批量调用 analyze_images_v2,返回 products。"""
|
||||
from apps.worker.worker_app.tasks import viral_video as vv
|
||||
|
||||
with patch("packages.shared.ai_service.call_vision", return_value=IMAGE_XML) as mock_v:
|
||||
result = vv._analyze_single_image(0, "https://img/1.jpg", "vlm-lite", 15)
|
||||
fake_product = {
|
||||
"name": "lipstick",
|
||||
"brand": "品牌X",
|
||||
"category": "唇部彩妆",
|
||||
"key_features": ["显白", "持久"],
|
||||
"text_on_package": ["品牌X", "211"],
|
||||
"_source": "v2_fast_json",
|
||||
}
|
||||
with patch("worker_app.tasks.vision.analyze_images_v2", return_value=[fake_product]) as mock_aiv2:
|
||||
result = vv._step_image_analysis(job)
|
||||
|
||||
mock_v.assert_called_once()
|
||||
# 验证调用时传入了 system_prompt(说明走了 loader 渲染的模板)
|
||||
call_kwargs = mock_v.call_args.kwargs
|
||||
assert "system_prompt" in call_kwargs and call_kwargs["system_prompt"]
|
||||
# 结果包含从 XML 解析出的产品信息
|
||||
assert result["name"] == "lipstick"
|
||||
assert result["brand"] == "品牌X"
|
||||
assert "显白" in result["key_features"]
|
||||
assert result["text_on_package"] == ["品牌X", "211"]
|
||||
mock_aiv2.assert_called_once()
|
||||
products = result["products"]
|
||||
assert len(products) == 1
|
||||
assert products[0]["name"] == "lipstick"
|
||||
assert products[0]["brand"] == "品牌X"
|
||||
assert "显白" in products[0]["key_features"]
|
||||
assert products[0]["text_on_package"] == ["品牌X", "211"]
|
||||
|
||||
|
||||
# ── 2) 意图解析走模板 ───────────────────────────────────────────────
|
||||
@@ -252,18 +259,22 @@ class TestEndToEndLoaderUsed:
|
||||
called_types.append(prompt_type)
|
||||
return real_get(prompt_type, **kwargs)
|
||||
|
||||
# V2 图片分析不再走 prompt_loader(固定 lite JSON prompt),用 mock 产品代替
|
||||
img_res = {
|
||||
"name": "lipstick",
|
||||
"brand": "品牌X",
|
||||
"key_features": ["显白", "持久"],
|
||||
"text_on_package": ["品牌X"],
|
||||
}
|
||||
with (
|
||||
patch.object(pl, "get_template", side_effect=spy_get),
|
||||
patch("packages.shared.ai_service.call_vision", return_value=IMAGE_XML),
|
||||
patch("packages.shared.ai_service.call_llm", return_value=INTENT_XML),
|
||||
):
|
||||
# 1) image
|
||||
img_res = vv._analyze_single_image(0, "https://img/1.jpg", "vlm", 15)
|
||||
# 2) intent
|
||||
# intent
|
||||
intent_res = vv._step_intent_parsing(job, {"products": [img_res]})
|
||||
|
||||
# 前两步分别调用了 image_analysis 和 intent_parsing
|
||||
assert "image_analysis" in called_types
|
||||
# V2 图片分析走固定 prompt(不经 loader);intent 仍走 loader
|
||||
assert "image_analysis" not in called_types
|
||||
assert "intent_parsing" in called_types
|
||||
|
||||
# script 和 review 单独验证(需要不同的 LLM 返回)
|
||||
|
||||
Reference in New Issue
Block a user