From d5f6e9499d22b446ceeba376b10592833b2c7ba3 Mon Sep 17 00:00:00 2001 From: xiaoxia-saas-bot Date: Sun, 4 Oct 2026 16:41:05 +0800 Subject: [PATCH] =?UTF-8?q?feat(viral=5Fvideo):=20#2040=20=E6=8E=A5?= =?UTF-8?q?=E5=85=A5Prompt=E6=A8=A1=E6=9D=BF=E7=B3=BB=E7=BB=9F+=E5=88=A0?= =?UTF-8?q?=E9=99=A4=E5=86=97=E4=BD=99=E7=A1=AC=E7=BC=96=E7=A0=81prompt?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- apps/worker/worker_app/tasks/viral_video.py | 588 +++++++++----------- packages/application/viral_video/prompts.py | 46 +- 2 files changed, 313 insertions(+), 321 deletions(-) diff --git a/apps/worker/worker_app/tasks/viral_video.py b/apps/worker/worker_app/tasks/viral_video.py index 4d410c0b9..95b798695 100644 --- a/apps/worker/worker_app/tasks/viral_video.py +++ b/apps/worker/worker_app/tasks/viral_video.py @@ -47,10 +47,8 @@ from packages.shared import get_shared_settings logger = logging.getLogger(__name__) - # ── WS 进度推送 ────────────────────────────────────────────────────────── - def _emit_progress( job_id: str, stage: str, @@ -76,22 +74,18 @@ def _emit_progress( except Exception as e: logger.warning("[爆款视频] WS 进度推送失败: %s", e) - # ── 仓储辅助 ──────────────────────────────────────────────────────────── - def _get_repo_and_job(job_id: str): session = SessionLocal() repo = SQLAlchemyViralVideoJobRepository(session) job = repo.get(job_id) return session, repo, job - def _save_job(repo, job, session): repo.update(job) session.commit() - def _start_trust_chain_preheat(job_id: str, portrait_descriptions: list[str]) -> None: """#2172/#2174 后台启动信任链预热(Seedream t2i 文生图人像),不阻塞调用方。 @@ -157,7 +151,6 @@ def _start_trust_chain_preheat(job_id: str, portrait_descriptions: list[str]) -> t = threading.Thread(target=_run_preheat, name=f"tc-preheat-{job_id[:8]}", daemon=True) t.start() - def _set_stage(job, repo, session, stage: str, message: str, persist: bool = True) -> None: """更新细粒度阶段并持久化到 DB,同时通过 Redis 推送进度事件。 @@ -173,7 +166,6 @@ def _set_stage(job, repo, session, stage: str, message: str, persist: bool = Tru except Exception as e: # 阶段持久化失败不阻塞主流程 logger.warning("[爆款视频] 阶段持久化失败 stage=%s err=%s", stage, e) - # ── worker 心跳(僵尸任务检测) ───────────────────────────────────────── # 心跳间隔(秒);超过此时间未更新 heartbeat_at 视为 worker 异常 @@ -183,7 +175,6 @@ _STALE_RUNNING_TIMEOUT_SEC = 10 * 60 # 10 分钟 # 心跳过期窗口:heartbeat_at 距 now 超过此时长视为失效 _HEARTBEAT_EXPIRE_SEC = 2 * 60 # 2 分钟 - def _heartbeat_once(job_id: str) -> None: """在独立 session 中更新一次 heartbeat_at(不捕获主流程事务状态)。""" ssn = None @@ -208,7 +199,6 @@ def _heartbeat_once(job_id: str) -> None: except Exception: pass - def _start_heartbeat_thread(job_id: str) -> tuple[threading.Event, threading.Thread]: """启动后台心跳线程,每 _HEARTBEAT_INTERVAL_SEC 秒更新一次 heartbeat_at。 返回 (stop_event, thread);任务结束时调用 stop_event.set() 停止心跳。 @@ -225,7 +215,6 @@ def _start_heartbeat_thread(job_id: str) -> tuple[threading.Event, threading.Thr t.start() return stop, t - def _recover_stale_jobs() -> int: """启动/定时扫描:把僵尸任务(running 超时且心跳停止)标记为 failed。 返回本次回收的任务数。可由 celery beat 周期性调用,也可在任务启动前顺带扫一次。 @@ -263,7 +252,6 @@ def _recover_stale_jobs() -> int: except Exception: pass - # ── 默认结构 ───────────────────────────────────────────────────────────── _DEFAULT_HARD_CONSTRAINTS = [ @@ -294,7 +282,6 @@ _DEFAULT_NEGATIVE_PROMPTS = [ "低分辨率", ] - def _empty_copy_result(duration: int = 15, ratio: str = "9:16") -> dict: return { "overview": {"theme": "好物推荐", "total_duration": duration, "aspect_ratio": ratio}, @@ -308,38 +295,8 @@ def _empty_copy_result(duration: int = 15, ratio: str = "9:16") -> dict: "title": "", } - # ── 流水线各步骤 ──────────────────────────────────────────────────────── - -_IMAGE_ANALYSIS_SYSTEM_PROMPT = """你是电商商品视觉分析师,从商品图片中提取关键商品信息。严格规则: -1. 只说图片里真实可见的内容,看不清/没有的填「无法判断」,不要瞎猜。 -2. 输出必须是严格 JSON(不要 Markdown 代码块,不要额外解释文字)。 -3. 字段说明: -{ - "name": "商品全名(品牌+产品名+规格,如『大公鸡头管家 多功能油污净 625ml』,从包装 OCR 读出)", - "brand": "品牌名(从 Logo/包装文字读出,看不清填『无法判断』)", - "category": "商品品类(如『家用清洁/油污清洁剂』『日化/洗衣液』;非产品图填『非产品图』)", - "appearance": "外观特征(50-100字:瓶身形状、颜色、瓶盖、标签颜色、尺寸感)", - "packaging": "包装细节(50-100字:标签分区、图案元素、瓶盖/泵头样式、塑封状态)", - "text_on_package": ["包装上清晰可见的文字列表(品牌、产品名、卖点、规格等,看不清的不列)"], - "key_features": [ - "3-5 条图片中能看到的外观/视觉特征(如『红色瓶盖白色瓶身』『鸡头图案 Logo』等)" - ], - "scene": "图片场景(如白底棚拍/浴室实拍/桌面静物/手持实拍等)", - "portrait_prompt": "如果图片中有清晰人物面部,用60-100字中文描述该人物外貌(性别、年龄段、发型/发色、肤色、脸型、五官特征、当前穿着、表情姿态),用于AI生图参考;没有人物或看不清面部填「无人像」", - "summary": "100-180字中文导购描述,连贯自然段落,像电商详情页介绍,前端直接展示,必须提到品牌/品名/核心外观特征,不能写『无法判断』" -}""" - -_IMAGE_ANALYSIS_USER_PROMPT = """请分析这张商品图片,输出严格 JSON。重点: -1. name/brand/text_on_package 从图片包装 OCR 读取,不编造; -2. appearance/packaging 各写 50-100 字,要具体; -3. portrait_prompt:有人物时详细描述外貌(性别/年龄/发型/肤色/穿着/表情)用于AI人像生成参考,无人像填「无人像」; -4. summary 必须是 100-180 字连贯中文段落,说清商品是什么、长什么样、适合谁用,不要写「无法判断」; -5. 非产品图时 category 填「非产品图」,name 填实际看到的内容; -6. 看不清的字段填「无法判断」。""" - - def _vision_fallback(idx: int, reason: str, extra: dict | None = None) -> dict: d = { "name": "未识别", @@ -357,7 +314,6 @@ def _vision_fallback(idx: int, reason: str, extra: dict | None = None) -> dict: d.update(extra) return d - def _is_vision_result_usable(result: dict) -> bool: """判断 VLM 返回是否有效:name/summary 不能为未识别/无法判断/空,summary 要够长。""" if not isinstance(result, dict): @@ -376,7 +332,6 @@ def _is_vision_result_usable(result: dict) -> bool: return False return True - def _analyze_single_image( idx: int, img_url: str, @@ -385,88 +340,97 @@ def _analyze_single_image( *, pro_fallback_model: str | None = None, ) -> dict: - """单张图片 VLM 分析(线程池并行调用)。 - - lite 失败/结果不可用 时自动用 pro 模型降级重试 1 次。 - - 失败/None/不可用最终返回含默认字段的 dict(不会让用户看到「未识别·无法判断」裸结果)。 + """单张图片 VLM 分析(#2040:改为从 prompt_loader 读模板 + XML 解析)。 + + lite 失败/不可用时用 pro 降级重试 1 次。失败/None 最终返回含默认字段的 dict。 """ - from packages.shared.ai_service import call_vision + try: + from packages.shared.ai_service import call_vision + from packages.application.viral_video.prompt_loader import ( + get_template, render_system_prompt, render_user_prompt, + ) + from packages.application.viral_video import xml_parser as xp + except ImportError as e: + logger.warning("[爆款视频] prompt 模板/解析模块不可用: %s", e) + return _vision_fallback(idx, f"fallback_import_error:{e}") if not img_url or not isinstance(img_url, str): return _vision_fallback(idx, "invalid_url") + template = get_template("image_analysis") + system = render_system_prompt(template) + user = render_user_prompt( + template, image_count=1, industry="通用", + image_urls=f"第1张:{img_url}", + ) + def _call(model: str, tmo: int): try: return call_vision( - image_url=img_url, - prompt=_IMAGE_ANALYSIS_USER_PROMPT, - model=model, - max_tokens=800, - temperature=0.1, - timeout=tmo, - system_prompt=_IMAGE_ANALYSIS_SYSTEM_PROMPT, + image_url=img_url, prompt=user, model=model, + max_tokens=2048, temperature=0.3, timeout=tmo, + system_prompt=system, ) except Exception as e: logger.warning("[爆款视频] 图片 #%d call_vision(%s) 异常 err=%s", idx, model, e) return None + def _xml_to_product(nodes: list, raw_text: str) -> dict: + product_nodes = [n for n in nodes if n["tag"] == "product"] + scene = xp.text_of(raw_text, "scene") or "通用" + mood = xp.text_of(raw_text, "mood") or "" + for p in product_nodes: + a = p["attrs"] + text_on_pkg = a.get("text_on_package", "") + text_list = [x.strip() for x in re.split(r"[,,;;]", text_on_pkg) if x.strip()] if text_on_pkg else [] + features = a.get("features", "") + feat_list = [x.strip() for x in re.split(r"[,,;;]", features) if x.strip()] if features else [] + name = a.get("name", "") or "未识别" + brand = a.get("brand", "") or "无法判断" + category = a.get("category", "") or "无法判断" + appearance = a.get("appearance", "") or "无法判断" + packaging = a.get("packaging", "") or "无法判断" + summary = a.get("summary", "") or f"{brand} {name}" + return { + "name": name, "brand": brand, "category": category, + "appearance": appearance, "packaging": packaging, + "text_on_package": text_list, + "key_features": feat_list or [features] if features else ["无法判断"], + "scene": scene, "mood": mood, + "portrait_prompt": a.get("portrait_prompt", "无人像"), + "summary": summary, "_source": "xml", + } + return _vision_fallback(idx, "no_product_tag") + def _normalize(raw, source: str) -> dict: if raw is None: return _vision_fallback(idx, f"{source}_none") - if isinstance(raw, str): - logger.warning("[爆款视频] 图片 #%d VLM(%s) 返回非 JSON: %s", idx, source, raw[:200]) - return _vision_fallback(idx, f"{source}_text", {"_raw": raw[:500]}) - if not isinstance(raw, dict): + if not isinstance(raw, str): return _vision_fallback(idx, f"{source}_badtype") - raw.setdefault("_source", source) - raw.setdefault("name", "未识别") - raw.setdefault("brand", "无法判断") - raw.setdefault("category", "无法判断") - raw.setdefault("appearance", "无法判断") - raw.setdefault("packaging", "无法判断") - raw.setdefault("text_on_package", []) - raw.setdefault("key_features", []) - raw.setdefault("scene", "通用") - raw.setdefault("portrait_prompt", "无人像") - raw.setdefault("summary", "") - if not isinstance(raw.get("text_on_package"), list): - raw["text_on_package"] = [] - if not isinstance(raw.get("key_features"), list): - raw["key_features"] = [] - return raw + nodes = xp.parse_tags(raw) + if not nodes: + logger.warning("[爆款视频] 图片 #%d XML 解析失败 source=%s", idx, source) + return _vision_fallback(idx, f"{source}_xml_fail", {"_raw": raw[:500]}) + product = _xml_to_product(nodes, raw) + product.setdefault("_source", source) + product["raw"] = raw[:500] + return product - # 第一次:传入模型(通常是 lite) first_raw = _call(vision_model, timeout) tag1 = vision_model.split("/")[-1] if "/" in vision_model else vision_model first_result = _normalize(first_raw, tag1) if _is_vision_result_usable(first_result): return first_result - logger.warning( - "[爆款视频] 图片 #%d VLM(%s) 结果不可用 name=%r summary_len=%d,尝试 pro 降级", - idx, - vision_model, - first_result.get("name"), - len(first_result.get("summary") or ""), - ) - - # 第二次:pro 降级重试 if pro_fallback_model and pro_fallback_model != vision_model: pro_raw = _call(pro_fallback_model, 60) # #2180: pro VLM 实测也需25-38s,原25s太短,提到60s pro_result = _normalize(pro_raw, "pro_fallback") if _is_vision_result_usable(pro_result): pro_result["_fallback_used"] = True return pro_result - logger.warning( - "[爆款视频] 图片 #%d pro 降级仍不可用 name=%r summary_len=%d", - idx, - pro_result.get("name"), - len(pro_result.get("summary") or ""), - ) return pro_result - return first_result - def _step_image_analysis(job: ViralVideoJob) -> dict: """步骤 1: 图片 VLM 分析 — 识别产品特征(v1.6 优化:并行 + lite 模型提速)。""" try: @@ -522,7 +486,6 @@ def _step_image_analysis(job: ViralVideoJob) -> dict: return {"products": results} - def _step_video_analysis(job: ViralVideoJob) -> dict | None: """步骤 1.5: 参考视频风格分析(可选)。""" if not job.reference_video_url: @@ -546,11 +509,14 @@ def _step_video_analysis(job: ViralVideoJob) -> dict | None: logger.error("[爆款视频] 视频风格分析失败: %s", e) return {"error": str(e), "source": "failed"} - def _step_intent_parsing(job: ViralVideoJob, image_analysis: dict) -> dict: - """步骤 2: 用户文案意图解析。""" + """步骤 2: 用户文案意图解析(#2040:改为模板 + XML 解析)。""" try: from packages.shared.ai_service import call_llm + from packages.application.viral_video.prompt_loader import ( + get_template, render_system_prompt, render_user_prompt, + ) + from packages.application.viral_video import xml_parser as xp except ImportError: return {"intent": "推广产品", "key_messages": ["产品亮点"], "tone": "专业", "suggested_title": ""} @@ -572,25 +538,25 @@ def _step_intent_parsing(job: ViralVideoJob, image_analysis: dict) -> dict: feat_str = ", ".join([str(x) for x in feats + extras]) products_summary += f"- {p.get('name', '产品')}: {feat_str}\n" - prompt = f"""你是一个营销编导。请分析以下信息,理解用户的营销意图并给出短视频主题建议: + template = get_template("intent_parsing") + system = render_system_prompt(template) + user = render_user_prompt( + template, + user_copy_text=job.user_copy_text or "(未提供,全由 AI 创作)", + industry=job.industry or "未指定", + image_analysis=products_summary or "- (无图片分析结果)", + ) -用户原始文案:{job.user_copy_text or "(未提供,全由 AI 创作)"} -行业:{job.industry or "未指定"} -目标客户:{job.target_customer or "未指定"} -营销目的:{job.marketing_purpose or "未指定"} -视频时长:{job.duration}秒 -产品信息: -{products_summary or "- (无图片分析结果)"} + def _parse(raw: str) -> dict: + summary = xp.text_of(raw, "intent_summary") + msgs = [n["text"] for n in xp.find_all(raw, "message") if n["text"]] + tone = xp.text_of(raw, "emotion_tone") or "亲切自然" + title = xp.text_of(raw, "suggested_title") or xp.text_of(raw, "title") + return {"intent": summary or "推广产品", "key_messages": msgs or ["产品亮点"], "tone": tone, "suggested_title": title} -请返回严格 JSON(不要 Markdown,不要解释): -{{ - "intent": "核心营销意图(一句话)", - "key_messages": ["要传达的3-5个关键信息"], - "tone": "文案调性(如亲切/专业/高端/活力/治愈/搞笑)", - "target_emotion": "希望触发的用户情感", - "call_to_action": "行动号召短句(口语化,5-10字)", - "suggested_title": "视频主题标题(5-15字)" -}}""" + def _fallback(raw: str) -> dict: + t = (job.user_copy_text or "").strip() + return {"intent": t[:30] or "推广产品", "key_messages": [t[:80]] if t else ["产品亮点"], "tone": "专业", "suggested_title": ""} _s = get_shared_settings() _fast = _s.doubao_fast_model @@ -598,22 +564,17 @@ def _step_intent_parsing(job: ViralVideoJob, image_analysis: dict) -> dict: for _m, _lbl in [(_fast, "fast"), (_pro, "pro-fallback")]: try: logger.info("[爆款视频] 意图解析 model=%s label=%s", _m, _lbl) - result = call_llm(prompt, temperature=0.4, max_tokens=800, model=_m, timeout=45) - return ( - result - if isinstance(result, dict) - else {"intent": str(result)[:200], "key_messages": [], "tone": "专业", "suggested_title": ""} - ) + raw = call_llm([{"role": "system", "content": system}, {"role": "user", "content": user}], + temperature=0.4, max_tokens=1024, model=_m, timeout=45) # #2180: 意图解析 LLM 实测需更长响应,原25s太紧 + if not raw: + continue + parsed = _parse(raw) + if parsed["intent"] or parsed["key_messages"]: + return parsed except Exception as e: logger.warning("[爆款视频] 意图解析失败 label=%s err=%s", _lbl, e) - continue - return {"intent": "推广产品", "key_messages": ["产品亮点"], "tone": "专业", "suggested_title": ""} + return _fallback("") - -# ── 编导分镜脚本生成(核心,v1.6 新 prompt) ────────────────────────────── - - -# 人设 IP 类型 → 文案/出镜风格指导(前端下拉 10 个 IP 类型) _PERSONA_STYLE_GUIDE = { "通用个人IP": "亲切自然、像朋友分享好物,第一人称口语化,不端着", "老板型IP": "沉稳大气、有行业格局感,适度使用『我做了XX年』『我一直坚持』等老板视角,语气自信不夸张", @@ -627,7 +588,6 @@ _PERSONA_STYLE_GUIDE = { "测评种草型": "真实测评感、讲使用体验和优缺点对比,带『亲测』『我用了XX天』『实测下来』真实感词汇", } - def _persona_style_hint(persona_id: str) -> str: """根据 persona_id 查文案风格指导;未命中/空值返回通用提示。""" pid = (persona_id or "").strip() @@ -638,74 +598,6 @@ def _persona_style_hint(persona_id: str) -> str: return f"【人设风格:{pid}】按该人设的口吻、话术习惯组织口播和出镜动作" return "【人设风格:未指定】亲切自然、像朋友分享好物" - -_SCRIPT_GENERATION_PROMPT = """你是资深短视频导演,为 Seedance 2.5(单次生成最多{duration}秒)写编导分镜脚本。脚本将整体作为 prompt 一次性传给视频模型,必须让模型在连贯镜头流中清楚每段时间拍什么、画面如何、人物说什么。 - -## 产品 -{products_summary} - -## 营销参数 -- 主题/意图:{intent} -- 关键信息:{key_messages} -- 调性:{tone} -- 目标客户:{target_customer} -- 用户原始卖点(必须融入口播):{user_copy} -- 时长:{duration}秒 / 画幅:{ratio} / 产品图:{n_images}张(第1张通常是主图/首帧) -- 风格参考:{style_hint} -- 爆款结构(必须严格遵循节奏/段落顺序):{viral_structure_block} -- 人设/出镜口吻(必须贯穿全部对白和动作描写):{persona_hint} - -## 输出格式(必须输出严格 JSON,不要 Markdown,不要解释,字段一个都不能少) - -```json -{{ - "overview": {{ - "theme": "视频主题(一句话概括)", - "total_duration": {duration}, - "aspect_ratio": "{ratio}" - }}, - "scene_and_lighting": "整体场景描述+光线设定(100-200字,要具体:在哪拍、什么光线、什么色调、什么氛围)", - "shots": [ - {{ - "time_range": "0-3秒", - "shot_type_angle_movement": "景别+角度+运镜(例:近景俯拍45度,缓慢推镜;中景平视,固定镜头;特写平视,快速拉镜)", - "scene_and_dialogue": "画面场景描述 + 人物口播台词(对白要自然口语化,像朋友聊天,不要硬广推销腔)", - "action_details": "人物动作、表情、物品操作细节(手怎么动、表情变化、产品怎么展示)", - "audio_bgm": "环境音+BGM提示(例:轻快流行BGM,环境嘈杂咖啡店背景音)", - "transition": "硬切/淡入淡出/叠化(最后一镜写『结束』即可)", - "reference_image_index": 0 - }} - // ... 按时间顺序列出所有镜头,总时长累计 = {duration} 秒 - ], - "hard_constraints": [ - "无字幕、无水印、无任何自动生成文字、无logo", - "同一人物全程五官、发型、服装、身材保持一致,不得换脸变形", - "口播语音在总时长内自然念完,语速自然,口型与语音严格同步", - "画面流畅无闪烁、无多余肢体、无扭曲变形、无穿模", - "色彩自然、曝光正确、电影级质感、高清细节" - ], - "negative_prompts": [ - "字幕","自动字幕","水印","logo","图标","错误文字","乱码文字", - "男女声错配","中途换声","五官崩坏","脸部变形","多余手指", - "肢体扭曲","闪烁","画面抖动","模糊","低分辨率" - ], - "voiceover_script": "完整口播稿(把 shots 里所有对白自然拼接成一段,口语化,不加旁白标注、不加镜头标注、不加'主播:'之类前缀,就是纯念出来的文本,长度适配{duration}秒,约{approx_chars}字)" -}} -``` - -## 关键要求 -1. 每镜写清景别/角度/运镜(特写/近景/中景+平视/俯拍+推/拉/固定)。 -2. 画面具体:主体(性别/年龄/穿着)、场景、动作、光线、镜头运动要可落地。 -3. 对白自然口语化,像朋友分享好物;拒绝"家人们""宝子们""太好用了"等浮夸/硬广腔。 -4. reference_image_index 填 0-based 索引(产品特写用索引0主图),人像/场景可 null。 -5. shots time_range 累计={duration}秒,单镜2-8秒。 -6. hard_constraints/negative_prompts 保留默认项可追加,不要删减。 -7. voiceover_script 为纯口播文本(无标记/括号/前缀),{duration}秒约{approx_chars}字。 -8. 严格按上方「爆款结构」的节奏/段落顺序编排(钩子/痛点/反转/案例/行动号召与结构对齐)。 -9. 输出前自检:口播对白禁止错别字和语病,**严禁使用"很近",正确用词是"最近"**(指"最近一段时间/最近在用",绝不能写成"很近");其他同音字、形近字错误一律修正。 -10. 必须使用产品信息中真实的品牌、品名和外观特征,不要编造与产品无关的内容。""" - - def _build_products_summary(image_analysis: dict) -> str: """把 VLM 返回的商品分析结果拼给文案/分镜生成 prompt 用。 优先用 summary(自然段落);没有时用结构化字段兜底拼一段。""" @@ -774,7 +666,6 @@ def _build_products_summary(image_analysis: dict) -> str: lines.append("- " + ",".join(parts)) return "\n".join(lines) - def _safe_json_loads(raw: str | dict | list | None): if raw is None: return None @@ -801,7 +692,6 @@ def _safe_json_loads(raw: str | dict | list | None): pass return None - def _replace_henjin_everywhere(obj: Any) -> Any: """递归遍历 copy_result 里所有字符串值,把'很近'替换成'最近'。 覆盖 overview.theme、scene_and_lighting、voiceover_script、 @@ -817,7 +707,6 @@ def _replace_henjin_everywhere(obj: Any) -> Any: return {k: _replace_henjin_everywhere(v) for k, v in obj.items()} return obj - def _fallback_script(job: ViralVideoJob) -> dict: """脚本生成失败时的兜底脚本(极简但可用)。""" dur = max(5, min(30, int(getattr(job, "duration", 15) or 15))) @@ -842,7 +731,6 @@ def _fallback_script(job: ViralVideoJob) -> dict: base["title"] = "好物分享" return base - def _validate_and_normalize_script(raw, job: ViralVideoJob) -> dict: """把 LLM 返回的脚本规范化、补默认、校验结构。""" dur = max(5, min(30, int(getattr(job, "duration", 15) or 15))) @@ -939,11 +827,68 @@ def _validate_and_normalize_script(raw, job: ViralVideoJob) -> dict: base["title"] = base["overview"]["theme"] return base +def _script_from_xml(raw: str, job: ViralVideoJob) -> dict | None: + """把 LLM 返回的 XML 分镜规范化为旧 copy_result 结构(供 Seedance 使用)。""" + from packages.application.viral_video import xml_parser as xp + dur = max(5, min(30, int(getattr(job, "duration", 15) or 15))) + ratio = getattr(job, "video_ratio", None) or "9:16" + base = _empty_copy_result(dur, ratio) + if not raw: + return None + base["overview"]["theme"] = xp.text_of(raw, "overview_theme") or xp.text_of(raw, "title") or "好物分享" + est = xp.attr_int(xp.text_of(raw, "estimated_duration"), 0) + if est: + base["overview"]["total_duration"] = est + sl = xp.text_of(raw, "scene_and_lighting") + if sl: + base["scene_and_lighting"] = sl + clips = xp.find_all(raw, "clip") + shots: list[dict] = [] + voice_parts: list[str] = [] + for i, c in enumerate(clips): + a = c["attrs"] + body = c.get("text", "") or "" + ref_idx_raw = a.get("reference_image_index", "") + if ref_idx_raw in (None, "", "null", "None"): + body_ref = xp.text_of(body, "reference_image_index") if body else "" + ref_idx = xp.attr_int(body_ref, 0) if body_ref else None + else: + ref_idx = xp.attr_int(ref_idx_raw, 0) + shot = { + "time_range": a.get("time_range") or f"{i * 3}-{(i + 1) * 3}秒", + "shot_type_angle_movement": (xp.text_of(body, "shot_type_angle_movement") if body else "") or a.get("shot_type_angle_movement", "") or "中景平视,固定镜头", + "scene_and_dialogue": (xp.text_of(body, "scene_and_dialogue") if body else "") or "", + "action_details": (xp.text_of(body, "action_details") if body else "") or "", + "audio_bgm": (xp.text_of(body, "audio_bgm") if body else "") or a.get("bgm_note", "") or "轻快BGM", + "transition": (xp.text_of(body, "transition") if body else "") or a.get("transition", "") or ("硬切" if i < len(clips) - 1 else "结束"), + "reference_image_index": ref_idx, + } + voice = (xp.text_of(body, "voice_text") if body else "") + if voice: + voice_parts.append(voice) + if not shot["scene_and_dialogue"]: + shot["scene_and_dialogue"] = voice + shots.append(shot) + if not shots: + return None + base["shots"] = shots + joined = xp.text_of(raw, "voiceover_script") or "。".join(voice_parts) + base["voiceover_script"] = joined + base["final_copy"] = joined + base["suggested_copy"] = joined + base["title"] = base["overview"]["theme"] + return base def _step_script_generation(job: ViralVideoJob, intent: dict, image_analysis: dict) -> dict: - """步骤 3: 编导分镜脚本生成(v1.6 核心,输出 copy_result 结构)。""" + """步骤 3: 编导分镜脚本生成(#2040:模板 + XML 解析;输出 copy_result 结构)。""" try: from packages.shared.ai_service import call_llm + from packages.application.viral_video.prompt_loader import ( + get_template, render_system_prompt, render_user_prompt, + ) + from packages.application.viral_video.prompts import ( + FUSION_INSTRUCTIONS, GLOBAL_CONSTRAINTS, NEGATIVE_RULES, + ) except ImportError: return _fallback_script(job) @@ -957,20 +902,15 @@ def _step_script_generation(job: ViralVideoJob, intent: dict, image_analysis: di dur = max(5, min(30, int(getattr(job, "duration", 15) or 15))) ratio = getattr(job, "video_ratio", None) or "9:16" - approx_chars = max(20, dur * 4) - intent_str = "" - key_msgs = "" - tone = "" + intent_str = "推广产品" + key_msgs = "产品亮点" + tone = "亲切自然" if isinstance(intent, dict): - intent_str = intent.get("intent") or "推广产品" - key_msgs = "、".join(intent.get("key_messages") or []) - tone = intent.get("tone") or "亲切自然" - else: - intent_str = "推广产品" - tone = "亲切自然" + intent_str = intent.get("intent") or intent_str + key_msgs = "、".join(intent.get("key_messages") or []) or key_msgs + tone = intent.get("tone") or tone - # 爆款结构:用户在 STEP1/STEP2 选的中文结构名,必须严格注入 prompt 指导 AI 编排 _vs = (job.viral_structure or "").strip() if _vs: viral_structure_block = f"【{_vs}】—— 请严格按照这个爆款结构的节奏/段落顺序编排镜头、台词和情绪节点(开场钩子、痛点、反转、案例、行动号召等按结构走),不要打乱顺序" @@ -978,59 +918,56 @@ def _step_script_generation(job: ViralVideoJob, intent: dict, image_analysis: di viral_structure_block = "未指定(自由编排,但仍需有钩子开头+产品展示+行动号召的基本节奏)" persona_hint = _persona_style_hint(getattr(job, "persona_id", "")) - prompt = _SCRIPT_GENERATION_PROMPT.format( - products_summary=products_summary, - intent=intent_str, - key_messages=key_msgs or "产品亮点", - tone=tone, - target_customer=job.target_customer or "通用人群", - user_copy=job.user_copy_text or "(未提供,自由创作)", - duration=dur, - ratio=ratio, - n_images=len(job.images or []), - style_hint=style_hint, - approx_chars=approx_chars, - viral_structure_block=viral_structure_block, - persona_hint=persona_hint, + fusion_level = getattr(job, "fusion_level", "ai_polish") or "ai_polish" + fusion_instruction = FUSION_INSTRUCTIONS.get(fusion_level, FUSION_INSTRUCTIONS["ai_polish"]) + + # 使用 storyboard 模板,注入融合指令/硬约束/反套路词 + template = get_template("storyboard") + system_tpl = template.system_prompt + system_tpl = system_tpl.replace("{fusion_instruction}", fusion_instruction) + system_tpl = system_tpl.replace("{global_constraints}", GLOBAL_CONSTRAINTS) + system_tpl = system_tpl.replace("{negative_rules}", NEGATIVE_RULES) + + fusion_brief = ( + f"意图:{intent_str}\n关键信息:{key_msgs}\n调性:{tone}\n" + f"用户原文:{job.user_copy_text or '(未提供)'}\n创作模式:{fusion_level}" ) + user = render_user_prompt( + template, + duration=dur, image_count=len(job.images or []), + fusion_result=fusion_brief, + image_analysis=products_summary, + ) + + def _try_gen(model: str, temp: float, max_tok: int, label: str, tmo: int = 25): + logger.info("[爆款视频] 编导脚本生成 model=%s label=%s timeout=%d", model, label, tmo) + raw = call_llm([{"role": "system", "content": system_tpl}, {"role": "user", "content": user}], + temperature=temp, max_tokens=max_tok, model=model, timeout=tmo) + if not raw: + return None + normalized = _script_from_xml(raw, job) + if normalized is None: + # 兼容:万一 LLM 仍输出 JSON,走旧规范化 + parsed_json = _safe_json_loads(raw) + if isinstance(parsed_json, dict): + normalized = _validate_and_normalize_script(parsed_json, job) + else: + return None + voiceover = (normalized or {}).get("voiceover_script") or "" + shots_cnt = len((normalized or {}).get("shots") or []) + _before_dump = json.dumps(normalized, ensure_ascii=False) + if "很近" in _before_dump: + normalized = _replace_henjin_everywhere(normalized) + voiceover = (normalized or {}).get("voiceover_script") or "" + fallback_marker = "我最近在用的好物" in voiceover + has_typo_henjin = "很近" in json.dumps(normalized, ensure_ascii=False) + is_fallback = fallback_marker or shots_cnt < 1 or len(voiceover) < 20 or has_typo_henjin + logger.info("[爆款视频] 编导脚本结果 label=%s voiceover_len=%d shots=%d fallback=%s", label, len(voiceover), shots_cnt, is_fallback) + return None if is_fallback else normalized _s = get_shared_settings() _fast = _s.doubao_fast_model _pro = getattr(_s, "doubao_model", None) or _fast - - def _try_gen(model: str, temp: float, max_tok: int, label: str, tmo: int = 25): - logger.info("[爆款视频] 编导脚本生成 model=%s label=%s timeout=%d", model, label, tmo) - r = call_llm(prompt, temperature=temp, max_tokens=max_tok, model=model, timeout=tmo) - if r is None: - logger.warning("[爆款视频] 编导脚本返回None label=%s", label) - return None - parsed = _safe_json_loads(r) - normalized = _validate_and_normalize_script(parsed, job) - voiceover = (normalized or {}).get("voiceover_script") or "" - voiceover_len = len(voiceover) - shots_cnt = len((normalized or {}).get("shots") or []) - # 判定是否"退化到兜底质量":口播过短(<20字)或镜头数<1;正常的短口播(如15s视频~40字)不视为兜底 - # v1.6.1 P1修复:递归替换 copy_result 里所有字符串字段的"很近"→"最近"(覆盖 overview/scene_and_lighting/voiceover/shots.* 全部字段) - _before_dump = json.dumps(normalized, ensure_ascii=False) - if "很近" in _before_dump: - logger.warning("[爆款视频] 编导脚本含错别字'很近',递归替换为'最近' label=%s", label) - normalized = _replace_henjin_everywhere(normalized) - voiceover = (normalized or {}).get("voiceover_script") or "" - fallback_marker = "我最近在用的好物" in voiceover # _fallback_script 的特征串 - has_typo_henjin = "很近" in json.dumps(normalized, ensure_ascii=False) # 递归检查仍有"很近"视为不合格 - is_fallback = fallback_marker or shots_cnt < 1 or voiceover_len < 20 or has_typo_henjin - logger.info( - "[爆款视频] 编导脚本结果 label=%s voiceover_len=%d shots=%d fallback=%s raw_type=%s", - label, - voiceover_len, - shots_cnt, - is_fallback, - type(r).__name__, - ) - if is_fallback: - return None # 触发重试 - return normalized - try: # 第一次:快模型 25s normalized = _try_gen(_fast, 0.8, 2500, "fast-first", tmo=45) @@ -1040,7 +977,6 @@ def _step_script_generation(job: ViralVideoJob, intent: dict, image_analysis: di normalized = _try_gen(_fast, 0.6, 3200, "fast-retry", tmo=45) if normalized is not None: return normalized - # 第三次:用主力模型兜底,给 40s if _pro and _pro != _fast: normalized = _try_gen(_pro, 0.7, 3500, "pro-fallback", tmo=60) if normalized is not None: @@ -1051,41 +987,81 @@ def _step_script_generation(job: ViralVideoJob, intent: dict, image_analysis: di logger.warning("[爆款视频] 编导脚本生成异常: %s,使用兜底脚本", e, exc_info=True) return _fallback_script(job) - def _step_review(job: ViralVideoJob, copy_result: dict) -> dict: - """步骤 4: 合规审核(简化版:基于脚本的 voiceover_script+shots 文本)。""" - dimensions = ["广告法合规", "平台规范", "内容真实性", "版权安全", "价值观", "风格一致性"] + """步骤 4: 合规审核(#2040:使用 Reviewer + review 模板,6 维度 + 自动重写 1 次)。 + + 返回结构与旧版兼容:{passed, score, details, issues, rewritten_copy?} + """ try: - from packages.shared.ai_service import call_llm - except ImportError: - return {"passed": True, "score": 90, "details": {d: "通过" for d in dimensions}} + from packages.application.viral_video.reviewer import Reviewer + from packages.application.viral_video.schemas import ( + FusionResult, IntentResult, CoreMessage, PersonalBrand, ScriptSegment, + ) + except ImportError as e: + logger.warning("[爆款视频] reviewer 模块不可用,跳过审核: %s", e) + return {"passed": True, "score": 80, "details": {}, "issues": []} - voiceover = (copy_result or {}).get("voiceover_script", "") - shots_preview = json.dumps((copy_result or {}).get("shots", [])[:3], ensure_ascii=False) - prompt = f"""请对以下短视频编导脚本进行合规审核,检查6个维度:{", ".join(dimensions)} + voiceover = (copy_result or {}).get("voiceover_script", "") or "" + title = (copy_result or {}).get("title") or (copy_result or {}).get("overview", {}).get("theme", "") -口播文案:{voiceover} -前3个镜头:{shots_preview} -行业:{job.industry} - -请以JSON格式返回: -- passed: bool(是否全部通过) -- score: int(0-100分) -- details: 各维度评分和说明 -- issues: 需要修改的问题列表(如有)""" - _s = get_shared_settings() - _fast = _s.doubao_fast_model - _pro = _s.doubao_model - for _m, _lbl in [(_fast, "fast"), (_pro, "pro-fallback")]: - try: - logger.info("[爆款视频] 合规审核 model=%s label=%s", _m, _lbl) - result = call_llm(prompt, temperature=0.1, max_tokens=500, model=_m, timeout=30) - return result if isinstance(result, dict) else {"passed": True, "score": 80, "details": {}} - except Exception as e: - logger.warning("[爆款视频] 合规审核失败 label=%s err=%s", _lbl, e) - continue - return {"passed": True, "score": 75, "details": {d: "默认通过" for d in dimensions}} + intent_data = job.intent_result or {} + core_msgs = [CoreMessage(text=str(m), must_keep=True, confidence=0.9) for m in (intent_data.get("key_messages") or [])] + brands: list[PersonalBrand] = [] + brand_text = intent_data.get("brand_text") or intent_data.get("suggested_title") or "" + if brand_text: + brands.append(PersonalBrand(text=str(brand_text), category="brand")) + intent_obj = IntentResult( + intent_summary=intent_data.get("intent", "") or "推广产品", + core_messages=core_msgs, personal_brands=brands, + ) + fusion_obj = FusionResult( + title=title or "", + hook=(voiceover[:30] if voiceover else ""), + cta="", + script_segments=[], + word_count=len(voiceover), + estimated_duration=int(getattr(job, "duration", 15) or 15), + ) + for shot in (copy_result or {}).get("shots", []) or []: + if isinstance(shot, dict) and shot.get("scene_and_dialogue"): + fusion_obj.script_segments.append(ScriptSegment(text=shot["scene_and_dialogue"])) + fusion_level = getattr(job, "fusion_level", "ai_polish") or "ai_polish" + try: + reviewer = Reviewer() + review_res = reviewer.review(fusion_obj, intent_obj, fusion_level) + new_copy = copy_result + rewritten_voice = None + if not review_res.passed and review_res.rewrite_suggestions: + try: + rewritten = reviewer.rewrite(fusion_obj, review_res, intent_obj, fusion_level) + if rewritten and (rewritten.title or rewritten.script_segments): + new_voice = rewritten.script_segments[0].text if rewritten.script_segments else (rewritten.hook or voiceover) + new_copy = dict(copy_result) + new_copy["voiceover_script"] = new_voice + new_copy["final_copy"] = new_voice + new_copy["suggested_copy"] = new_voice + if rewritten.title: + new_copy.setdefault("overview", {})["theme"] = rewritten.title + new_copy["title"] = rewritten.title + rewritten_voice = new_voice + review_res = reviewer.review(rewritten, intent_obj, fusion_level) + except Exception as e: + logger.warning("[爆款视频] 自动重写失败: %s", e) + result = { + "passed": review_res.passed, + "score": 90 if review_res.passed else 60, + "details": {i.dimension: i.text for i in review_res.issues}, + "issues": [{"dimension": i.dimension, "severity": i.severity, "location": i.location, "text": i.text} for i in review_res.issues], + } + if rewritten_voice is not None: + result["rewritten_copy"] = new_copy + job.copy_result = new_copy + job.generated_copy_text = rewritten_voice + return result + except Exception as e: + logger.warning("[爆款视频] 审核异常,跳过: %s", e, exc_info=True) + return {"passed": True, "score": 75, "details": {}, "issues": []} def _step_tts(job: ViralVideoJob, voiceover_script: str): """步骤 5: CosyVoice 整段配音 → 返回本地 MP3 Path;失败返回 None。""" @@ -1125,7 +1101,6 @@ def _step_tts(job: ViralVideoJob, voiceover_script: str): logger.warning("[爆款视频] TTS 配音失败: %s", e, exc_info=True) return None - def _upload_tts_to_oss(job: ViralVideoJob, tts_path) -> str | None: """把 TTS 本地 mp3 上传到 OSS,返回公网 URL(供 Seedance 做 reference_audios 口型驱动用)。""" if tts_path is None: @@ -1145,7 +1120,6 @@ def _upload_tts_to_oss(job: ViralVideoJob, tts_path) -> str | None: logger.warning("[爆款视频] TTS 上传 OSS 失败: %s", e, exc_info=True) return None - def _assemble_seedance_prompt(copy_result: dict, job: ViralVideoJob) -> str: """把编导脚本拼成 Seedance 长 prompt。""" if not isinstance(copy_result, dict) or not copy_result: @@ -1196,7 +1170,6 @@ def _assemble_seedance_prompt(copy_result: dict, job: ViralVideoJob) -> str: lines.append(",".join([str(x) for x in np if x])) return "\n".join(lines) - def _step_render(job: ViralVideoJob, copy_result: dict, tts_audio_url: str | None) -> tuple[str, dict | None]: """步骤 6: v1.6 单次 Seedance 生成(不再分段/拼接)。 @@ -1310,7 +1283,6 @@ def _step_render(job: ViralVideoJob, copy_result: dict, tts_audio_url: str | Non logger.info("[爆款视频] 单次生成完成: path=%s size=%d usage=%s", video_path, size, usage) return str(video_path), (usage if isinstance(usage, dict) else None) - def _step_upload(job: ViralVideoJob, video_path: str) -> str: """步骤 7: OSS 上传。""" from video_processing.oss_helpers import upload_to_oss @@ -1323,7 +1295,6 @@ def _step_upload(job: ViralVideoJob, video_path: str) -> str: raise RuntimeError(f"OSS 上传失败: storage_key={storage_key}") return video_url - def _wait_oss_ready(url: str, timeout_sec: int = 10) -> bool: """轮询 OSS 公网 URL,直到 HEAD 返回 200 或超时。 用于缓解 OSS 上传后 1-5s 公网 eventual consistency 导致的 NoSuchKey。 @@ -1344,10 +1315,8 @@ def _wait_oss_ready(url: str, timeout_sec: int = 10) -> bool: logger.warning("[爆款视频] OSS 成片在 %ds 内未就绪 last_status=%s url=%s", timeout_sec, last_status, url[:120]) return False - # ── 主编排器 ──────────────────────────────────────────────────────────── - @shared_task( bind=True, max_retries=2, @@ -1432,7 +1401,6 @@ def run_viral_video_pipeline(self: Task, job_id: str) -> dict: if session: session.close() - @shared_task(bind=True, max_retries=2, name="worker.resume_viral_video_pipeline") def resume_viral_video_pipeline(self: Task, job_id: str) -> dict: """旧 confirm-intent 路径兼容:从 WAIT_USER_CONFIRM 跑完整个渲染。""" @@ -1454,7 +1422,6 @@ def resume_viral_video_pipeline(self: Task, job_id: str) -> dict: if session: session.close() - @shared_task(bind=True, max_retries=1, name="worker.run_video_style_analysis") def run_video_style_analysis(self: Task, job_id: str) -> dict: """独立的视频风格分析任务。""" @@ -1478,10 +1445,8 @@ def run_video_style_analysis(self: Task, job_id: str) -> dict: if session: session.close() - # ── 失败处理 ──────────────────────────────────────────────────────────── - def _mark_failed_and_notify(job_id: str, session, repo, job, err_msg: str, stage: str = "") -> None: """标记任务失败并通知。若传入的 session 已失效(因前面异常导致 rollback 状态), 会自动 fallback 到新建 SessionLocal 重新标记,确保状态一定落库。""" @@ -1522,10 +1487,8 @@ def _mark_failed_and_notify(job_id: str, session, repo, job, err_msg: str, stage event_type="viral_video:failed", ) - # ── v1.5/v1.6 三步分步流水线 Celery 任务 ───────────────────────────────── - @shared_task( bind=True, max_retries=1, @@ -1593,7 +1556,6 @@ def run_viral_video_analyze(self: Task, job_id: str) -> dict: if session: session.close() - @shared_task( bind=True, max_retries=1, @@ -1696,7 +1658,6 @@ def run_viral_video_generate_copy(self: Task, job_id: str) -> dict: if session: session.close() - def _quick_compliance_blacklist_check(copy_result: dict) -> None: """阶段2快速黑名单检查:不调用 LLM,只扫描高风险关键词;命中则在 voiceover 中就地替换。 @@ -1738,7 +1699,6 @@ def _quick_compliance_blacklist_check(copy_result: dict) -> None: for bk, bv in BLACKLIST.items(): copy_result[k] = copy_result[k].replace(bk, bv) - def _try_refund_viral_video(job: ViralVideoJob) -> None: """爆款视频生成失败:若已预扣积分则全额退款。""" try: @@ -1768,7 +1728,6 @@ def _try_refund_viral_video(job: ViralVideoJob) -> None: except Exception: logger.exception("[爆款视频] 失败退款异常 job_id=%s", job.id) - def _settle_viral_video(job: ViralVideoJob, usage: dict | None) -> None: """爆款视频生成成功:按实际 usage 结算,多退少补,写 credits_cost。""" try: @@ -1839,7 +1798,6 @@ def _settle_viral_video(job: ViralVideoJob, usage: dict | None) -> None: job.credits_cost = float(getattr(job, "credits_prepaid", 0) or 0) job.credits_prepaid = 0.0 - def _run_render_pipeline(job_id: str, session, repo, job) -> dict: """v1.6.1 阶段3:出片前合规审核(LLM 深度)→ TTS → Seedance → Upload → Completed。 @@ -1863,9 +1821,14 @@ def _run_render_pipeline(job_id: str, session, repo, job) -> dict: review_result = _step_review(job, copy_result) if not review_result.get("passed", True): _emit_progress(job_id, ViralVideoStage.REVIEW, 67.0, "审核未通过,正在自动重写...") - intent = job.intent_result or _step_intent_parsing(job, image_analysis) - copy_result = _step_script_generation(job, intent, image_analysis) - _step_review(job, copy_result) # 二次审核,不通过也继续出片(避免反复循环) + # #2040: Reviewer 已在 _step_review 内完成 1 次自动重写 + rewritten = review_result.get("rewritten_copy") + if isinstance(rewritten, dict) and rewritten: + copy_result = rewritten + else: + intent = job.intent_result or _step_intent_parsing(job, image_analysis) + copy_result = _step_script_generation(job, intent, image_analysis) + _step_review(job, copy_result) job.copy_result = copy_result job.generated_copy_text = copy_result.get("voiceover_script", "") or "" _save_job(repo, job, session) @@ -1921,7 +1884,6 @@ def _run_render_pipeline(job_id: str, session, repo, job) -> dict: logger.info("[爆款视频] 任务完成: job_id=%s video_url=%s", job_id, video_url) return {"ok": True, "job_id": job_id, "video_url": video_url} - @shared_task(bind=True, max_retries=2, name="worker.run_viral_video_render") def run_viral_video_render(self: Task, job_id: str) -> dict: """v1.6 阶段3:TTS + 单次 Seedance 生成 + 上传。""" diff --git a/packages/application/viral_video/prompts.py b/packages/application/viral_video/prompts.py index 2eab68e7e..520743788 100644 --- a/packages/application/viral_video/prompts.py +++ b/packages/application/viral_video/prompts.py @@ -132,7 +132,10 @@ _FUSION_SYSTEM = """你负责为短视频生成营销文案。请按思维链分 开头3秒钩子,5到15字。 每个要点用一个 标签,属性 elaboration 是展开说明、image_index 是对应第几张图(从0开始),标签内容写要点。 口语化的行动号召。 - 每段配音用一个 标签,属性 duration_sec 是秒数、image_index 是对应图片,标签内容写配音文案。 + 每段配音用一个 标签,属性 duration_sec 是秒数、image_index 是对应图片,标签内容写配音文案(纯口播文本,不加旁白标注、不加镜头标注、不加"主播:"之类前缀)。 + 把所有 segment 的配音文案按顺序自然拼接成一段完整的纯口播文本(无标记、无括号、无前缀),长度要适配 {duration} 秒,约 {approx_chars} 字。 + 视频主题(一句话概括)。 + 整体场景描述+光线设定(100-200字,要具体:在哪拍、什么光线、什么色调、什么氛围)。 配音总字数,只写数字。 预计时长秒数,只写数字。 @@ -161,25 +164,34 @@ _FUSION_EXAMPLE = """厨房重油污,别再用洗洁精硬擦了后来换了这个大公鸡头油污净,喷上等几分钟,一擦就干净 39块钱625ml,厨房重油污的可以试一瓶 +这油污我真的忍很久了,用洗洁精擦半天都没用。后来换了这个大公鸡头油污净,喷上等几分钟,一擦就干净。39块钱625ml,厨房重油污的可以试一瓶。 +厨房油污清洁好物分享 +简洁明亮的厨房台面场景,自然光从窗户洒入,色调温暖柔和,突出产品白色瓶身与去油污对比效果。 58 13""" # ── 模板4:编导级分镜(LLM)──────────────────────────────────────────── -_STORYBOARD_SYSTEM = f"""你是短视频编导,负责把文案拆成可拍摄的分镜。 +_STORYBOARD_SYSTEM = f"""你是短视频编导,负责把文案拆成可拍摄的分镜,为 Seedance 2.5 视频模型写编导分镜脚本。脚本将整体作为 prompt 一次性传给视频模型,必须让模型在连贯镜头流中清楚每段时间拍什么、画面如何、人物说什么。 工作方式: 1. 按文案的 script_segments 顺序分配镜头。 -2. 每个镜头确定画面、运镜、时长、配音和字幕。 +2. 每个镜头确定景别/角度/运镜、画面场景与对白、人物动作细节、音效/BGM、转场。 3. 检查所有镜头时长加起来接近目标时长,误差不超过2秒。 4. image_index 必须在已上传图片范围内,第一张主图必须用在第一个镜头。 {GLOBAL_CONSTRAINTS} 请严格按下面的标签格式输出,不要解释,不要用代码块: - 下面每个镜头用一个 标签,属性 image_index 是图片序号(从0开始)、transition 取 fade、cut、zoom_in、slide_left、dissolve、wipe 之一、zoom 取 in、out 或 null、duration_sec 是该镜头秒数、bgm_note 是该段BGM情绪。每个 里面包含: - 该镜头配音文本; - 字幕文本,可与配音一致或更精简; - 用一个空标签,属性 start、end 写“x,y”坐标、ease 写缓动方式;不需要运镜时坐标相同。""" + 下面每个镜头用一个 标签,属性 image_index 是图片序号(从0开始)、transition 取 fade/cut/zoom_in/slide_left/dissolve/wipe 之一、zoom 取 in/out/null、duration_sec 是该镜头秒数、bgm_note 是该段BGM情绪。每个 里面包含: + 该镜头配音文本(纯口播文本,不加旁白标注); + 字幕文本,可与配音一致或更精简; + 景别+角度+运镜(例:近景俯拍45度,缓慢推镜;中景平视,固定镜头;特写平视,快速拉镜); + 画面场景描述 + 人物口播台词(对白要自然口语化,像朋友聊天,不要硬广推销腔); + 人物动作、表情、物品操作细节(手怎么动、表情变化、产品怎么展示); + 环境音+BGM提示(例:轻快流行BGM,环境嘈杂咖啡店背景音); + 硬切/淡入淡出/叠化(最后一镜写『结束』即可); + 参考图片索引(0-based,对应第几张产品图,无则空); + 用一个空标签,属性 start、end 写"x,y"坐标、ease 写缓动方式;不需要运镜时坐标相同。""" _STORYBOARD_USER = """目标时长:{duration}秒 上传图片数量:{image_count}张(第1张是主图/封面) @@ -194,17 +206,35 @@ _STORYBOARD_EXAMPLE = """ 这油污我真的忍很久了 这油污忍很久了 +近景俯拍45度,缓慢推镜 +厨房台面,主妇皱眉看着灶台油污。对白:这油污我真的忍很久了 +右手拿着脏抹布,无奈摇头 +轻快日常BGM,带一点烦躁感 +硬切 +0 后来换了大公鸡头油污净,喷上等几分钟,一擦就干净 喷上等几分钟,一擦就干净 +特写平视,固定镜头 +手部特写,喷油污净在油污处。对白:后来换了这个大公鸡头油污净,喷上等几分钟,一擦就干净 +左手拿产品瓶身,右手按压喷头,等待片刻后用抹布轻擦 +轻快转折BGM,带清爽感 +淡入淡出 +0 39块钱625ml,厨房重油污的可以试一瓶 39元625ml,可以试一瓶 - +中景平视,缓慢拉镜 +产品正面展示,明亮背景。对白:39块钱625ml,厨房重油污的可以试一瓶 +产品置于画面中央,轻微转动展示瓶身 +温暖收尾BGM +结束 +0 + """