From cdce1b2e105ab1e34e4d2a63e1b9745033021f91 Mon Sep 17 00:00:00 2001 From: xiaoxia Date: Thu, 1 Oct 2026 04:00:12 +0800 Subject: [PATCH 1/6] =?UTF-8?q?fix(viral-video):=20fix=206=20E2E=20bugs=20?= =?UTF-8?q?=E2=80=94=20fusion=5Flevel=20alias,=20concat=20silent=20segment?= =?UTF-8?q?s,=20ingest=20lookup,=20duplicated=20URL,=20TTS=20voice/format,?= =?UTF-8?q?=20Seedance=20first-frame=20ratio?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Bug1 (P0): schema accepts 'full_ai' as alias for 'ai_full' (Pydantic field_validator normalizes) Bug2 (P0): concat_video_files probes each segment audio stream; Seedance gen_audio=False segments now marked has_audio=False so concat filter uses aevalsrc silence instead of failing with ffmpeg exit 234 Bug3 (P1): find_by_storage_key now queries (storage_key OR file_url) to cover historical data where the legacy file_url column held assets///... paths Bug4 (P1): duplicated-hit response no longer accesses non-existent domain Asset.file_url; new helper _get_existing_asset_url uses storage_key (fallback file_url) through storage_service.get_url() Bug5 (P1): _step_tts passes job.persona_id as voice_id (default longxiaochun_v3) and forces format='mp3' so downstream ffmpeg -map 1:a:0 works regardless of provider Bug6 (P1): call_video_generation omits ratio param in first-frame (image_url) mode; ai_client.video_generation ratio becomes Optional[str] and is omitted from payload when None, fixing 400 InvalidParameter from Seedance --- apps/api/app/api/routes/upload.py | 19 ++++++++++++++++++- apps/api/app/schemas/viral_video.py | 5 ++++- apps/worker/video_processing/concat_engine.py | 15 ++++++++++++++- apps/worker/worker_app/tasks/viral_video.py | 17 ++++++++++++++--- .../sqlalchemy_impl/asset_repository.py | 15 +++++++++++++-- packages/shared/ai_client.py | 11 +++++++---- packages/shared/ai_service.py | 12 +++++++++--- 7 files changed, 79 insertions(+), 15 deletions(-) diff --git a/apps/api/app/api/routes/upload.py b/apps/api/app/api/routes/upload.py index eb7e5d9f7..135bf8d87 100644 --- a/apps/api/app/api/routes/upload.py +++ b/apps/api/app/api/routes/upload.py @@ -191,6 +191,23 @@ def _find_duplicate_asset( return None + +def _get_existing_asset_url(existing: Any, storage_service: Any) -> str: + """安全获取已存在素材的公网 URL,兼容 domain Asset(无 file_url 字段)和 ORM model。""" + # Domain Asset 只有 storage_key 字段;ORM model 有 file_url 但存的也是 storage_key + key = "" + for attr in ("storage_key", "file_url"): + v = getattr(existing, attr, None) + if v: + key = v + break + if not key: + return "" + try: + return storage_service.get_url(key) or "" + except Exception: + return "" + def _create_pending_asset( asset_repository, project_id, @@ -390,7 +407,7 @@ async def prepare_direct_upload( duplicated=True, skip_transfer=True, asset_id=existing.id, - url=existing.file_url or storage_service.get_url(existing.storage_key) or "", + url=_get_existing_asset_url(existing, storage_service), ) file_id = uuid4().hex[:8] diff --git a/apps/api/app/schemas/viral_video.py b/apps/api/app/schemas/viral_video.py index 1d16bf1fb..359aeae18 100755 --- a/apps/api/app/schemas/viral_video.py +++ b/apps/api/app/schemas/viral_video.py @@ -8,7 +8,7 @@ from pydantic import BaseModel, Field, field_validator # ── 枚举常量 ───────────────────────────────────────────────────────────── -VALID_FUSION_LEVELS = ("ai_full", "ai_polish", "user_primary") +VALID_FUSION_LEVELS = ("ai_full", "full_ai", "ai_polish", "user_primary") VALID_STYLE_STRENGTHS = ("light", "medium", "strict") VALID_STAGES = ( "image_analysis", @@ -50,6 +50,9 @@ class CreateViralVideoRequest(BaseModel): @field_validator("fusion_level") @classmethod def _validate_fusion_level(cls, v: str) -> str: + # 兼容前端历史写法 full_ai(等价 ai_full) + if v == "full_ai": + return "ai_full" if v not in VALID_FUSION_LEVELS: raise ValueError(f"fusion_level 必须是 {VALID_FUSION_LEVELS} 之一") return v diff --git a/apps/worker/video_processing/concat_engine.py b/apps/worker/video_processing/concat_engine.py index 3bf049412..47a21f5df 100755 --- a/apps/worker/video_processing/concat_engine.py +++ b/apps/worker/video_processing/concat_engine.py @@ -553,7 +553,20 @@ def concat_video_files( if work_dir is None: work_dir = output_path.parent - segments = [ConcatSegment(video_path=p) for p in video_paths if p] + # Bug #2110: 探测每段是否真实包含音频流,避免 Seedance 生成的无声片段 + # (gen_audio=False)让 concat filter `a=1` 找不到 [N:a] 而报 exit 234。 + from video_processing.ffmpeg_utils import probe_has_audio as _probe_has_audio + + segments: list[ConcatSegment] = [] + for p in video_paths: + if not p: + continue + try: + has_audio = _probe_has_audio(p) + except Exception: + has_audio = True # 探测失败保守认为有音频 + segments.append(ConcatSegment(video_path=p, has_audio=has_audio)) + config = ConcatConfig(segments=segments, force_reencode=force_reencode) engine = ConcatEngine(work_dir) diff --git a/apps/worker/worker_app/tasks/viral_video.py b/apps/worker/worker_app/tasks/viral_video.py index 9054ead24..cacdd029a 100644 --- a/apps/worker/worker_app/tasks/viral_video.py +++ b/apps/worker/worker_app/tasks/viral_video.py @@ -385,15 +385,26 @@ def _step_tts(job: ViralVideoJob, copy_text: str): from apps.worker.services.tts_service_factory import get_tts_service tts_service = get_tts_service() - # 兼容老接口:部分 provider 只接收 text 参数 + # Bug #2110: persona_id 透传给 voice_id(空则用 CosyVoice 默认 longxiaochun_v3), + # 统一输出 mp3 给后续 ffmpeg 混音(之前默认 wav 导致部分 provider/后处理不兼容)。 + voice_id = (job.persona_id or "").strip() try: - result = tts_service.synthesize(text=copy_text, voice_id=job.persona_id or "default") + result = tts_service.synthesize( + text=copy_text, + voice_id=voice_id or "longxiaochun_v3", + format="mp3", + ) except TypeError: - result = tts_service.synthesize(text=copy_text) + # 老 provider 只支持 text 参数 + try: + result = tts_service.synthesize(text=copy_text, voice_id=voice_id or "longxiaochun_v3") + except TypeError: + result = tts_service.synthesize(text=copy_text) if result is None: return None p = _Path(result) if not isinstance(result, _Path) else result if p.exists(): + logger.info("[爆款视频] TTS 合成完成: voice=%s path=%s size=%d", voice_id or "longxiaochun_v3", p, p.stat().st_size) return p logger.warning("[爆款视频] TTS 返回路径不存在: %s", p) return None diff --git a/packages/adapters/sqlalchemy_impl/asset_repository.py b/packages/adapters/sqlalchemy_impl/asset_repository.py index d1cab4535..c59fe2dd4 100755 --- a/packages/adapters/sqlalchemy_impl/asset_repository.py +++ b/packages/adapters/sqlalchemy_impl/asset_repository.py @@ -502,8 +502,19 @@ class SQLAlchemyAssetRepository: return [self._to_domain(m) for m in models] def find_by_storage_key(self, storage_key: str) -> Asset | None: - """按 storage_key(对应 DB 中的 file_url)查找素材。""" - model = self.session.query(AssetModel).filter(AssetModel.file_url == storage_key).first() + """按 storage_key 查找素材。 + + Bug #2110: 历史数据 file_url 列可能是旧路径(assets/...),新代码统一写入 + storage_key 列。双列 OR 查询,避免占位 asset 因路径错配导致 ingest 兜底新建 + 第二条 READY 记录,原占位卡 PROCESSING → 前端缩略图出现后消失。 + """ + if not storage_key: + return None + model = ( + self.session.query(AssetModel) + .filter((AssetModel.storage_key == storage_key) | (AssetModel.file_url == storage_key)) + .first() + ) if model is None: return None return self._to_domain(model) diff --git a/packages/shared/ai_client.py b/packages/shared/ai_client.py index fe12327e6..936c6f732 100755 --- a/packages/shared/ai_client.py +++ b/packages/shared/ai_client.py @@ -248,7 +248,7 @@ class DoubaoClient: *, image_url: str | None = None, duration: int = 5, - ratio: str = "9:16", + ratio: str | None = "9:16", resolution: str = "720p", generate_audio: bool = False, watermark: bool = False, @@ -287,11 +287,13 @@ class DoubaoClient: "model": video_model, "content": content, "generate_audio": generate_audio, - "ratio": ratio, "duration": int(duration), "resolution": resolution, "watermark": watermark, } + # Bug #2110: ratio=None 时不传(首帧图生视频跟随原图比例,传 ratio 会 400 InvalidParameter) + if ratio: + create_payload["ratio"] = ratio headers = { "Authorization": f"Bearer {self.api_key}", @@ -299,12 +301,13 @@ class DoubaoClient: } create_url = f"{self.base_url}/contents/generations/tasks" logger.info( - "Seedance 创建任务请求: url=%s model=%s duration=%ds ratio=%s gen_audio=%s", + "Seedance 创建任务请求: url=%s model=%s duration=%ds ratio=%s gen_audio=%s image_url=%s", create_url, video_model, duration, - ratio, + ratio or "(follow-image)", generate_audio, + bool(image_url), ) # 1) 创建任务(带重试) diff --git a/packages/shared/ai_service.py b/packages/shared/ai_service.py index d9af4e394..d527dd06b 100755 --- a/packages/shared/ai_service.py +++ b/packages/shared/ai_service.py @@ -543,29 +543,35 @@ def call_video_generation( *, image_url: str | None = None, duration: int = 5, - ratio: str = "9:16", + ratio: str | None = "9:16", resolution: str = "720p", output_dir: str | None = None, ) -> str | None: """调用 Seedance 2.5 生成视频段,返回本地 MP4 路径;失败返回 None。 封装 ai_client.video_generation:提交异步任务→轮询→下载到本地。 + Bug #2110: 首帧参考图模式下不传 ratio(API 要求跟随首帧图比例,传 ratio=9:16 + 会返回 400 InvalidParameter)。 """ client = get_doubao_client() if not client.is_available: logger.warning("[ai_service] 豆包客户端未配置,跳过视频生成") return None + # 首帧模式:不强制 ratio,让模型跟随首帧图比例 + effective_ratio = None if image_url else ratio try: - return client.video_generation( + kwargs: dict = dict( prompt=prompt, image_url=image_url, duration=duration, - ratio=ratio, resolution=resolution, generate_audio=False, # 我们自己混 TTS watermark=False, output_dir=output_dir, ) + if effective_ratio: + kwargs["ratio"] = effective_ratio + return client.video_generation(**kwargs) except Exception as e: logger.error("[ai_service] call_video_generation 异常: %s", e, exc_info=True) return None From 09b8b2990fe21d704d62f6bd44ce0a040a3aa8c2 Mon Sep 17 00:00:00 2001 From: xiaoxia Date: Thu, 1 Oct 2026 04:38:58 +0800 Subject: [PATCH 2/6] test: fix unit tests to match TTS import path and storage_service.get_url mock - test_viral_video_p0: patch new apps.worker.services.tts_service_factory path, assert format='mp3' is passed - test_prepare_dedup: storage_service.get_url.return_value = '' so DirectUploadPrepareResponse.url: str doesn't receive a MagicMock --- tests/unit/test_prepare_dedup_1714.py | 3 +++ tests/unit/test_viral_video_p0.py | 9 ++++++--- 2 files changed, 9 insertions(+), 3 deletions(-) diff --git a/tests/unit/test_prepare_dedup_1714.py b/tests/unit/test_prepare_dedup_1714.py index cb0ad9ff8..689fa618d 100644 --- a/tests/unit/test_prepare_dedup_1714.py +++ b/tests/unit/test_prepare_dedup_1714.py @@ -144,6 +144,9 @@ def _storage(): "expires_at": "2026-01-01T00:00:00Z", "fields": {"key": "uploads/abc/test.mp4"}, } + # Bug #2110: duplicated 命中时 _get_existing_asset_url 调用 get_url 返回公网 URL 字符串, + # Mock 默认返回 MagicMock,会让 DirectUploadPrepareResponse.url: str 校验失败。 + s.get_url.return_value = "" return s diff --git a/tests/unit/test_viral_video_p0.py b/tests/unit/test_viral_video_p0.py index d5d98b8c4..76ccd31b0 100644 --- a/tests/unit/test_viral_video_p0.py +++ b/tests/unit/test_viral_video_p0.py @@ -120,7 +120,7 @@ class TestTTSPath: """get_tts_service 抛 ImportError 时 _step_tts 返回 None。""" from apps.worker.worker_app.tasks import viral_video as vv - with patch("services.tts_service_factory.get_tts_service", side_effect=ImportError("no tts")): + with patch("apps.worker.services.tts_service_factory.get_tts_service", side_effect=ImportError("no tts")): assert vv._step_tts(mock_job, "文案") is None def test_tts_returns_none_when_path_not_exists(self, mock_job, tmp_path): @@ -128,7 +128,7 @@ class TestTTSPath: fake_service = MagicMock() fake_service.synthesize.return_value = str(tmp_path / "not_exist.mp3") - with patch("services.tts_service_factory.get_tts_service", return_value=fake_service): + with patch("apps.worker.services.tts_service_factory.get_tts_service", return_value=fake_service): assert vv._step_tts(mock_job, "文案") is None def test_tts_returns_path_when_exists(self, mock_job, tmp_path): @@ -138,8 +138,11 @@ class TestTTSPath: audio.write_bytes(b"ID3fake") fake_service = MagicMock() fake_service.synthesize.return_value = audio - with patch("services.tts_service_factory.get_tts_service", return_value=fake_service): + with patch("apps.worker.services.tts_service_factory.get_tts_service", return_value=fake_service): result = vv._step_tts(mock_job, "文案") + # Bug #2110: 校验传入了 voice_id+format=mp3 + call_kwargs = fake_service.synthesize.call_args.kwargs + assert call_kwargs.get("format") == "mp3" assert isinstance(result, Path) assert result.exists() From a74be7c7177b67b3a0d9eaa9d170b4a658478094 Mon Sep 17 00:00:00 2001 From: CI Bot Date: Wed, 30 Sep 2026 20:43:44 +0000 Subject: [PATCH 3/6] style: auto-format with black + isort + ruff + prettier [skip ci-format-check] --- apps/worker/worker_app/tasks/viral_video.py | 4 +++- 1 file changed, 3 insertions(+), 1 deletion(-) diff --git a/apps/worker/worker_app/tasks/viral_video.py b/apps/worker/worker_app/tasks/viral_video.py index cacdd029a..2b05a4748 100644 --- a/apps/worker/worker_app/tasks/viral_video.py +++ b/apps/worker/worker_app/tasks/viral_video.py @@ -404,7 +404,9 @@ def _step_tts(job: ViralVideoJob, copy_text: str): return None p = _Path(result) if not isinstance(result, _Path) else result if p.exists(): - logger.info("[爆款视频] TTS 合成完成: voice=%s path=%s size=%d", voice_id or "longxiaochun_v3", p, p.stat().st_size) + logger.info( + "[爆款视频] TTS 合成完成: voice=%s path=%s size=%d", voice_id or "longxiaochun_v3", p, p.stat().st_size + ) return p logger.warning("[爆款视频] TTS 返回路径不存在: %s", p) return None From 3d8882c479ea3e282c5171c410ed2a214109a063 Mon Sep 17 00:00:00 2001 From: xiaoxia Date: Thu, 1 Oct 2026 04:59:11 +0800 Subject: [PATCH 4/6] fix(tests+robustness): mock-safe status_code int-cast in ai_client; seedance test responses default 200 - test_ai_client_video: fake task/poll responses explicitly set status_code=200/text='' to avoid MagicMock()>=int TypeError leaking into retry logic - ai_client.video_generation: wrap resp.status_code comparisons in try/except (int() cast) so any non-int status (MagicMock in tests) degrades to 200 instead of bombing out with TypeError during create/poll --- packages/shared/ai_client.py | 14 ++++++++++++-- tests/unit/test_ai_client_video.py | 4 ++++ 2 files changed, 16 insertions(+), 2 deletions(-) diff --git a/packages/shared/ai_client.py b/packages/shared/ai_client.py index 936c6f732..67102cad5 100755 --- a/packages/shared/ai_client.py +++ b/packages/shared/ai_client.py @@ -316,7 +316,13 @@ class DoubaoClient: for attempt in range(self.max_retries + 1): try: resp = httpx.post(create_url, headers=headers, json=create_payload, timeout=self.timeout) - if resp.status_code >= 400: + # 测试环境下 MagicMock().status_code 是 MagicMock,与 int 比较会抛 TypeError; + # 用显式 int() 转换+类型判断,避免误判。 + try: + _status = int(resp.status_code) + except (TypeError, ValueError): + _status = 200 + if _status >= 400: # 把响应体完整打出来(通常含 error.code/message,能直接定位:模型未开通/Key 无权限/模型 ID 错误) logger.error( "Seedance 创建任务 HTTP %d: body=%s", @@ -360,7 +366,11 @@ class DoubaoClient: while time.time() < deadline: try: resp = httpx.get(poll_url, headers=headers, timeout=self.timeout) - resp.raise_for_status() + try: + if int(getattr(resp, "status_code", 200)) >= 400: + resp.raise_for_status() + except (TypeError, ValueError): + pass data = resp.json() status = data.get("status", "") last_status = status diff --git a/tests/unit/test_ai_client_video.py b/tests/unit/test_ai_client_video.py index 310a6df46..dfdb188bb 100644 --- a/tests/unit/test_ai_client_video.py +++ b/tests/unit/test_ai_client_video.py @@ -46,6 +46,8 @@ class TestVideoGenerationHappyPath: fake_task_resp = MagicMock() fake_task_resp.json.return_value = {"id": "task-001"} fake_task_resp.raise_for_status = MagicMock() + fake_task_resp.status_code = 200 + fake_task_resp.text = "" fake_poll_resp = MagicMock() fake_poll_resp.json.return_value = { @@ -53,6 +55,8 @@ class TestVideoGenerationHappyPath: "content": {"video_url": "https://cdn.example.com/v.mp4"}, } fake_poll_resp.raise_for_status = MagicMock() + fake_poll_resp.status_code = 200 + fake_poll_resp.text = "" class FakeStreamResponse: def __init__(self): From 3a8ef857ac373f171c2682d49563312803098b7a Mon Sep 17 00:00:00 2001 From: xiaoxia Date: Thu, 1 Oct 2026 10:06:59 +0800 Subject: [PATCH 5/6] =?UTF-8?q?fix(vlm):=20call=5Fvision=20=E8=B5=B0?= =?UTF-8?q?=E8=A7=86=E8=A7=89=E6=A8=A1=E5=9E=8B=E8=80=8C=E9=9D=9E=E6=96=87?= =?UTF-8?q?=E6=9C=AC=E6=A8=A1=E5=9E=8B=20+=20=E5=9B=BE=E7=89=87=E5=88=86?= =?UTF-8?q?=E6=9E=90=20prompt=20=E7=BB=93=E6=9E=84=E5=8C=96=E5=BC=BA?= =?UTF-8?q?=E5=8C=96?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit 根因(VLM 牛头不对马嘴):packages/shared/ai_service.call_vision() 之前误用 client.chat_completion(文本模型 doubao-seed-1.6-250615)发送多模态 content list。 文本模型不认图片 content type → 返回 None → _step_image_analysis 的 except 静默吞掉 → fallback 到占位结果 {name:'未识别'} → 后续 intent/copy/storyboard 完全没图信息,自然胡编。 修复: 1) call_vision 改用 client.vision_completion(模型 doubao-1-5-vision-pro-250915), 走标准 OpenAI 多模态 chat/completions + image_url 格式 2) 超时提到 60s,温度降到 0.2,解析 ```json 代码块包裹 3) 增加详细 INFO 日志:打印 vision_model/image_url/prompt_len/原始返回前 400 字, 方便下次直接在 worker 日志排查 4) 重写 _IMAGE_ANALYSIS_PROMPT:强制结构化 JSON schema(category/name/brand/colors/ material_or_texture/key_features/visual_style/scene/target_audience_hint/ text_on_image),明确『无法判断就填无法判断,不许编造』,key_features 只能写外观可见特征、不许编功效 5) _step_image_analysis 健壮化:images 为空/None/文本返回/异常分别落 _source 标记; 每张图独立 try/except,单张失败不影响其他图 6) 下游 intent_parsing/copy_fusion/storyboard 兼容新字段 key_features/brand/ category/colors/visual_style(同时向后兼容旧 features 字段) --- apps/worker/worker_app/tasks/viral_video.py | 92 +++++++++++++++++---- packages/shared/ai_service.py | 57 ++++++++++--- 2 files changed, 123 insertions(+), 26 deletions(-) diff --git a/apps/worker/worker_app/tasks/viral_video.py b/apps/worker/worker_app/tasks/viral_video.py index 2b05a4748..a6b77f3c6 100644 --- a/apps/worker/worker_app/tasks/viral_video.py +++ b/apps/worker/worker_app/tasks/viral_video.py @@ -99,25 +99,69 @@ def _save_job(repo, job, session): # ── 流水线各步骤 ──────────────────────────────────────────────────────── +_IMAGE_ANALYSIS_PROMPT = """请仔细观察这张图片,只基于图片中真实可见的内容进行分析,不要凭空想象。 + +必须输出严格的 JSON(不要 Markdown 代码块,不要额外解释),字段如下: +{ + "category": "产品大类,如护肤品/彩妆/食品/数码/服饰/家居等,若无法识别填『无法判断』", + "name": "产品名称(从包装/品牌/logo/文字推断;没有品牌时描述外观如『粉色包装面霜』)", + "brand": "品牌名(看 logo/包装文字;看不清填『未知』)", + "colors": ["主体颜色"], + "material_or_texture": "材质/质地描述(如玻璃瓶装/塑料软管/哑光质感/金属外壳等;无法判断填『无法判断』)", + "key_features": [ + "3-5 条**图片中确实能看到**的外观特征/卖点描述(如『按压式泵头』『瓶身有金色装饰线』等),不要编图片里没有的功效" + ], + "visual_style": "视觉风格(如简约高端/粉嫩少女/国潮/科技感/生活方式实拍等)", + "scene": "图片中的使用/展示场景(如白底棚拍/浴室场景/户外街拍/桌面静物等;纯白底填『白底产品图』)", + "target_audience_hint": "从视觉推断的目标人群(如年轻女性/男性商务/亲子家庭等;不确定填『通用』)", + "text_on_image": "图片上出现的可读文字(品牌名/Slogan/产品名等,没有则填『无』)" +} + +严格要求: +1. 任何字段无法确认时填『无法判断』或『未知』,不要猜。 +2. key_features 只能描述图片里肉眼可见的物理外观,不要写『补水保湿』『抗衰老』这类功效词(除非包装上明确印了)。 +3. 如果图片完全不是产品图(比如风景/人像/截图),category 填『非产品图』,name 填实际看到的内容。 +""" + + def _step_image_analysis(job: ViralVideoJob) -> dict: - """步骤 1: 图片 VLM 分析 — 识别产品特征、场景、卖点。""" + """步骤 1: 图片 VLM 分析 — 识别产品特征、场景、卖点。 + + Bug #2114 修复: + 1) call_vision 现已走视觉模型 doubao-1-5-vision-pro(之前误走文本模型导致完全没看图); + 2) Prompt 强化为结构化 JSON schema,禁止编造,强制图片可见才写; + 3) 单张失败不影响其他图片,最终至少返回一张占位结果避免后续 NoneType; + 4) 日志打印每张图的 URL 和模型原始返回,方便排查。 + """ try: from packages.shared.ai_service import call_vision except ImportError: logger.warning("[爆款视频] ai_service.call_vision 不可用,使用占位结果") - return {"products": [{"name": "产品", "features": ["特征1", "特征2"], "scene": "通用场景"}]} + return {"products": [{"name": "产品", "features": ["特征1", "特征2"], "scene": "通用场景", "_source": "fallback_import_error"}]} + + if not job.images: + logger.warning("[爆款视频] 任务无 images,跳过图片分析") + return {"products": []} results = [] - for img_url in job.images: + for idx, img_url in enumerate(job.images): + logger.info("[爆款视频] 图片分析 #%d img=%s", idx, img_url[:160]) try: - result = call_vision( - image_url=img_url, - prompt="请分析这张产品图片,识别:1)产品名称和类别 2)主要特征和卖点 3)适用场景 4)视觉风格。以JSON格式返回。", - ) - results.append(result) + result = call_vision(image_url=img_url, prompt=_IMAGE_ANALYSIS_PROMPT) + if result is None: + logger.warning("[爆款视频] 图片 #%d call_vision 返回 None(模型超时/Key未配置)", idx) + results.append({"name": "未识别", "category": "无法判断", "key_features": [], "scene": "通用", "_source": "vision_none"}) + elif isinstance(result, str): + # JSON 解析失败返回的原文,包装一下防止后续 .get 报错 + logger.warning("[爆款视频] 图片 #%d VLM 返回非 JSON 文本,包装为 features: %s", idx, result[:200]) + results.append({"name": "未识别", "category": "无法判断", "key_features": [], "scene": "通用", "_raw": result[:500], "_source": "vision_text"}) + else: + # dict 正常 + result.setdefault("_source", "vision") + results.append(result) except Exception as e: - logger.warning("[爆款视频] 图片分析失败 img=%s: %s", img_url, e) - results.append({"name": "未识别", "features": [], "scene": "通用"}) + logger.warning("[爆款视频] 图片分析失败 img=%s err=%s", img_url[:120], e, exc_info=True) + results.append({"name": "未识别", "category": "无法判断", "key_features": [], "scene": "通用", "_source": "vision_exception", "_error": str(e)[:200]}) return {"products": results} @@ -157,7 +201,18 @@ def _step_intent_parsing(job: ViralVideoJob, image_analysis: dict) -> dict: products_summary = "" for p in image_analysis.get("products", []): - products_summary += f"- {p.get('name', '产品')}: {', '.join(p.get('features', []))}\n" + feats = p.get("key_features") or p.get("features") or [] + extras = [] + if p.get("brand") and p.get("brand") not in ("未知", "无法判断"): + extras.append(f"品牌={p['brand']}") + if p.get("category") and p.get("category") not in ("无法判断", "非产品图"): + extras.append(f"品类={p['category']}") + if p.get("colors"): + extras.append(f"颜色={','.join(p['colors'])}") + if p.get("scene") and p.get("scene") not in ("通用",): + extras.append(f"场景={p['scene']}") + feat_str = ", ".join([str(x) for x in feats + extras]) + products_summary += f"- {p.get('name', '产品')}: {feat_str}\n" prompt = f"""你是一个营销文案策略师。请分析以下信息,理解用户的营销意图: @@ -192,7 +247,14 @@ def _step_copy_fusion(job: ViralVideoJob, intent: dict, image_analysis: dict) -> products_desc = "" for p in image_analysis.get("products", []): - products_desc += f"{p.get('name', '产品')}({','.join(p.get('features', []))})\n" + feats = p.get("key_features") or p.get("features") or [] + extras = [] + if p.get("brand") and p.get("brand") not in ("未知", "无法判断"): + extras.append(f"品牌={p['brand']}") + if p.get("visual_style"): + extras.append(f"风格={p['visual_style']}") + feat_str = ",".join([str(x) for x in feats + extras]) + products_desc += f"{p.get('name', '产品')}({feat_str})\n" if job.fusion_level == "ai_full": prompt = f"""请为以下产品撰写一段爆款短视频文案({job.duration}秒): @@ -251,8 +313,10 @@ def _step_storyboard(job: ViralVideoJob, copy_text: str, image_analysis: dict) - products = image_analysis.get("products", []) if image_analysis else [] if products: p0 = products[0] if isinstance(products[0], dict) else {} - feats = p0.get("features", []) if isinstance(p0, dict) else [] - products_hint = f"\n首帧参考产品特征:{p0.get('name','')} - {', '.join(feats[:3])}" + feats = (p0.get("key_features") or p0.get("features") or []) if isinstance(p0, dict) else [] + brand = p0.get("brand") if isinstance(p0, dict) else "" + brand_hint = f"(品牌={brand})" if brand and brand not in ("未知", "无法判断") else "" + products_hint = f"\n首帧参考产品特征:{p0.get('name','')}{brand_hint} - {', '.join(feats[:3])}" seg_seconds = 5 n_segments = max(2, min(6, max(1, job.duration // seg_seconds))) diff --git a/packages/shared/ai_service.py b/packages/shared/ai_service.py index d527dd06b..7b557e3b2 100755 --- a/packages/shared/ai_service.py +++ b/packages/shared/ai_service.py @@ -515,26 +515,59 @@ def call_llm(prompt: str, temperature: float = 0.7) -> object: def call_vision(image_url: str, prompt: str) -> object: - """调用豆包视觉大模型分析图片,返回解析后的 JSON 或原文字符串;失败返回 None。""" + """调用豆包视觉大模型分析图片,返回解析后的 JSON 或原文字符串;失败返回 None。 + + Bug #2114 (VLM 牛头不对马嘴根因修复): + 之前误走 client.chat_completion(用文本模型 doubao-seed-1.6),多模态 content list 被当成 + 纯文本发给文本模型 → 模型要么看不到图、要么抛 400,静默被 except 吞掉 → 返回 None → + _step_image_analysis fallback 到 {"name":"未识别"} → 后续文案/分镜完全没图的信息。 + 现改走 vision_completion,走视觉模型 doubao-1-5-vision-pro-250915。 + """ client = get_doubao_client() if not client.is_available: + logger.warning("[call_vision] 豆包客户端未配置 (DOUBAO_API_KEY 缺失)") return None + if not image_url: + logger.warning("[call_vision] 空 image_url,跳过视觉分析") + return None + + system_prompt = ( + "你是资深电商视觉分析师。请严格基于用户提供的图片观察回答," + "图片里没有的信息不要凭空想象或编造;看不清或无法判断时明确说" + "「图片中无法判断」,不要猜测。输出必须是严格 JSON,不要附加 Markdown 或解释文字。" + ) messages = [ - {"role": "system", "content": "你是专业的视觉分析师。需要结构化输出时请严格使用 JSON。"}, - { - "role": "user", - "content": [ - {"type": "text", "text": prompt}, - {"type": "image_url", "image_url": {"url": image_url}}, - ], - }, + {"role": "system", "content": system_prompt}, + {"role": "user", "content": prompt}, ] - raw = client.chat_completion(messages, temperature=0.3, max_tokens=2048) + + logger.info( + "[call_vision] 调用豆包视觉模型 vision_model=%s image_url=%s prompt_len=%d", + getattr(client, "vision_model", "?"), + image_url[:120], + len(prompt), + ) + raw = client.vision_completion( + messages=messages, + images=[image_url], + temperature=0.2, + max_tokens=2048, + timeout=60, + ) if raw is None: + logger.warning("[call_vision] 视觉模型返回 None (image_url=%s)", image_url[:80]) return None + logger.info("[call_vision] 视觉模型原始返回 (前400字): %s", raw[:400]) + # 剥离 ```json ... ``` 包裹 + stripped = raw.strip() + if stripped.startswith("```"): + stripped = stripped.strip("`") + if stripped.startswith("json"): + stripped = stripped[4:].lstrip() try: - return json.loads(raw) - except (json.JSONDecodeError, TypeError): + return json.loads(stripped) + except (json.JSONDecodeError, TypeError) as e: + logger.warning("[call_vision] JSON 解析失败(%s),返回原始文本: %s", e, raw[:200]) return raw From a9cbe7d4c9ad8b7be8243c06f0a010570e06e2b7 Mon Sep 17 00:00:00 2001 From: CI Bot Date: Thu, 1 Oct 2026 02:11:07 +0000 Subject: [PATCH 6/6] style: auto-format with black + isort + ruff + prettier [skip ci-format-check] --- apps/worker/worker_app/tasks/viral_video.py | 43 +++++++++++++++++++-- 1 file changed, 39 insertions(+), 4 deletions(-) diff --git a/apps/worker/worker_app/tasks/viral_video.py b/apps/worker/worker_app/tasks/viral_video.py index a6b77f3c6..d4b7bb937 100644 --- a/apps/worker/worker_app/tasks/viral_video.py +++ b/apps/worker/worker_app/tasks/viral_video.py @@ -137,7 +137,16 @@ def _step_image_analysis(job: ViralVideoJob) -> dict: from packages.shared.ai_service import call_vision except ImportError: logger.warning("[爆款视频] ai_service.call_vision 不可用,使用占位结果") - return {"products": [{"name": "产品", "features": ["特征1", "特征2"], "scene": "通用场景", "_source": "fallback_import_error"}]} + return { + "products": [ + { + "name": "产品", + "features": ["特征1", "特征2"], + "scene": "通用场景", + "_source": "fallback_import_error", + } + ] + } if not job.images: logger.warning("[爆款视频] 任务无 images,跳过图片分析") @@ -150,18 +159,44 @@ def _step_image_analysis(job: ViralVideoJob) -> dict: result = call_vision(image_url=img_url, prompt=_IMAGE_ANALYSIS_PROMPT) if result is None: logger.warning("[爆款视频] 图片 #%d call_vision 返回 None(模型超时/Key未配置)", idx) - results.append({"name": "未识别", "category": "无法判断", "key_features": [], "scene": "通用", "_source": "vision_none"}) + results.append( + { + "name": "未识别", + "category": "无法判断", + "key_features": [], + "scene": "通用", + "_source": "vision_none", + } + ) elif isinstance(result, str): # JSON 解析失败返回的原文,包装一下防止后续 .get 报错 logger.warning("[爆款视频] 图片 #%d VLM 返回非 JSON 文本,包装为 features: %s", idx, result[:200]) - results.append({"name": "未识别", "category": "无法判断", "key_features": [], "scene": "通用", "_raw": result[:500], "_source": "vision_text"}) + results.append( + { + "name": "未识别", + "category": "无法判断", + "key_features": [], + "scene": "通用", + "_raw": result[:500], + "_source": "vision_text", + } + ) else: # dict 正常 result.setdefault("_source", "vision") results.append(result) except Exception as e: logger.warning("[爆款视频] 图片分析失败 img=%s err=%s", img_url[:120], e, exc_info=True) - results.append({"name": "未识别", "category": "无法判断", "key_features": [], "scene": "通用", "_source": "vision_exception", "_error": str(e)[:200]}) + results.append( + { + "name": "未识别", + "category": "无法判断", + "key_features": [], + "scene": "通用", + "_source": "vision_exception", + "_error": str(e)[:200], + } + ) return {"products": results}