feat(viral_video): #2040 接入Prompt模板系统+删除冗余硬编码prompt
This commit is contained in:
committed by
Xiaoxia Agent
parent
b6243c8ab8
commit
73e6cf2953
@@ -47,10 +47,8 @@ from packages.shared import get_shared_settings
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
# ── WS 进度推送 ──────────────────────────────────────────────────────────
|
||||
|
||||
|
||||
def _emit_progress(
|
||||
job_id: str,
|
||||
stage: str,
|
||||
@@ -76,22 +74,18 @@ def _emit_progress(
|
||||
except Exception as e:
|
||||
logger.warning("[爆款视频] WS 进度推送失败: %s", e)
|
||||
|
||||
|
||||
# ── 仓储辅助 ────────────────────────────────────────────────────────────
|
||||
|
||||
|
||||
def _get_repo_and_job(job_id: str):
|
||||
session = SessionLocal()
|
||||
repo = SQLAlchemyViralVideoJobRepository(session)
|
||||
job = repo.get(job_id)
|
||||
return session, repo, job
|
||||
|
||||
|
||||
def _save_job(repo, job, session):
|
||||
repo.update(job)
|
||||
session.commit()
|
||||
|
||||
|
||||
def _start_trust_chain_preheat(job_id: str, portrait_descriptions: list[str]) -> None:
|
||||
"""#2172/#2174 后台启动信任链预热(Seedream t2i 文生图人像),不阻塞调用方。
|
||||
|
||||
@@ -157,7 +151,6 @@ def _start_trust_chain_preheat(job_id: str, portrait_descriptions: list[str]) ->
|
||||
t = threading.Thread(target=_run_preheat, name=f"tc-preheat-{job_id[:8]}", daemon=True)
|
||||
t.start()
|
||||
|
||||
|
||||
def _set_stage(job, repo, session, stage: str, message: str, persist: bool = True) -> None:
|
||||
"""更新细粒度阶段并持久化到 DB,同时通过 Redis 推送进度事件。
|
||||
|
||||
@@ -173,7 +166,6 @@ def _set_stage(job, repo, session, stage: str, message: str, persist: bool = Tru
|
||||
except Exception as e: # 阶段持久化失败不阻塞主流程
|
||||
logger.warning("[爆款视频] 阶段持久化失败 stage=%s err=%s", stage, e)
|
||||
|
||||
|
||||
# ── worker 心跳(僵尸任务检测) ─────────────────────────────────────────
|
||||
|
||||
# 心跳间隔(秒);超过此时间未更新 heartbeat_at 视为 worker 异常
|
||||
@@ -183,7 +175,6 @@ _STALE_RUNNING_TIMEOUT_SEC = 10 * 60 # 10 分钟
|
||||
# 心跳过期窗口:heartbeat_at 距 now 超过此时长视为失效
|
||||
_HEARTBEAT_EXPIRE_SEC = 2 * 60 # 2 分钟
|
||||
|
||||
|
||||
def _heartbeat_once(job_id: str) -> None:
|
||||
"""在独立 session 中更新一次 heartbeat_at(不捕获主流程事务状态)。"""
|
||||
ssn = None
|
||||
@@ -208,7 +199,6 @@ def _heartbeat_once(job_id: str) -> None:
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
|
||||
def _start_heartbeat_thread(job_id: str) -> tuple[threading.Event, threading.Thread]:
|
||||
"""启动后台心跳线程,每 _HEARTBEAT_INTERVAL_SEC 秒更新一次 heartbeat_at。
|
||||
返回 (stop_event, thread);任务结束时调用 stop_event.set() 停止心跳。
|
||||
@@ -225,7 +215,6 @@ def _start_heartbeat_thread(job_id: str) -> tuple[threading.Event, threading.Thr
|
||||
t.start()
|
||||
return stop, t
|
||||
|
||||
|
||||
def _recover_stale_jobs() -> int:
|
||||
"""启动/定时扫描:把僵尸任务(running 超时且心跳停止)标记为 failed。
|
||||
返回本次回收的任务数。可由 celery beat 周期性调用,也可在任务启动前顺带扫一次。
|
||||
@@ -263,7 +252,6 @@ def _recover_stale_jobs() -> int:
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
|
||||
# ── 默认结构 ─────────────────────────────────────────────────────────────
|
||||
|
||||
_DEFAULT_HARD_CONSTRAINTS = [
|
||||
@@ -294,7 +282,6 @@ _DEFAULT_NEGATIVE_PROMPTS = [
|
||||
"低分辨率",
|
||||
]
|
||||
|
||||
|
||||
def _empty_copy_result(duration: int = 15, ratio: str = "9:16") -> dict:
|
||||
return {
|
||||
"overview": {"theme": "好物推荐", "total_duration": duration, "aspect_ratio": ratio},
|
||||
@@ -308,38 +295,8 @@ def _empty_copy_result(duration: int = 15, ratio: str = "9:16") -> dict:
|
||||
"title": "",
|
||||
}
|
||||
|
||||
|
||||
# ── 流水线各步骤 ────────────────────────────────────────────────────────
|
||||
|
||||
|
||||
_IMAGE_ANALYSIS_SYSTEM_PROMPT = """你是电商商品视觉分析师,从商品图片中提取关键商品信息。严格规则:
|
||||
1. 只说图片里真实可见的内容,看不清/没有的填「无法判断」,不要瞎猜。
|
||||
2. 输出必须是严格 JSON(不要 Markdown 代码块,不要额外解释文字)。
|
||||
3. 字段说明:
|
||||
{
|
||||
"name": "商品全名(品牌+产品名+规格,如『大公鸡头管家 多功能油污净 625ml』,从包装 OCR 读出)",
|
||||
"brand": "品牌名(从 Logo/包装文字读出,看不清填『无法判断』)",
|
||||
"category": "商品品类(如『家用清洁/油污清洁剂』『日化/洗衣液』;非产品图填『非产品图』)",
|
||||
"appearance": "外观特征(50-100字:瓶身形状、颜色、瓶盖、标签颜色、尺寸感)",
|
||||
"packaging": "包装细节(50-100字:标签分区、图案元素、瓶盖/泵头样式、塑封状态)",
|
||||
"text_on_package": ["包装上清晰可见的文字列表(品牌、产品名、卖点、规格等,看不清的不列)"],
|
||||
"key_features": [
|
||||
"3-5 条图片中能看到的外观/视觉特征(如『红色瓶盖白色瓶身』『鸡头图案 Logo』等)"
|
||||
],
|
||||
"scene": "图片场景(如白底棚拍/浴室实拍/桌面静物/手持实拍等)",
|
||||
"portrait_prompt": "如果图片中有清晰人物面部,用60-100字中文描述该人物外貌(性别、年龄段、发型/发色、肤色、脸型、五官特征、当前穿着、表情姿态),用于AI生图参考;没有人物或看不清面部填「无人像」",
|
||||
"summary": "100-180字中文导购描述,连贯自然段落,像电商详情页介绍,前端直接展示,必须提到品牌/品名/核心外观特征,不能写『无法判断』"
|
||||
}"""
|
||||
|
||||
_IMAGE_ANALYSIS_USER_PROMPT = """请分析这张商品图片,输出严格 JSON。重点:
|
||||
1. name/brand/text_on_package 从图片包装 OCR 读取,不编造;
|
||||
2. appearance/packaging 各写 50-100 字,要具体;
|
||||
3. portrait_prompt:有人物时详细描述外貌(性别/年龄/发型/肤色/穿着/表情)用于AI人像生成参考,无人像填「无人像」;
|
||||
4. summary 必须是 100-180 字连贯中文段落,说清商品是什么、长什么样、适合谁用,不要写「无法判断」;
|
||||
5. 非产品图时 category 填「非产品图」,name 填实际看到的内容;
|
||||
6. 看不清的字段填「无法判断」。"""
|
||||
|
||||
|
||||
def _vision_fallback(idx: int, reason: str, extra: dict | None = None) -> dict:
|
||||
d = {
|
||||
"name": "未识别",
|
||||
@@ -357,7 +314,6 @@ def _vision_fallback(idx: int, reason: str, extra: dict | None = None) -> dict:
|
||||
d.update(extra)
|
||||
return d
|
||||
|
||||
|
||||
def _is_vision_result_usable(result: dict) -> bool:
|
||||
"""判断 VLM 返回是否有效:name/summary 不能为未识别/无法判断/空,summary 要够长。"""
|
||||
if not isinstance(result, dict):
|
||||
@@ -376,7 +332,6 @@ def _is_vision_result_usable(result: dict) -> bool:
|
||||
return False
|
||||
return True
|
||||
|
||||
|
||||
def _analyze_single_image(
|
||||
idx: int,
|
||||
img_url: str,
|
||||
@@ -385,88 +340,97 @@ def _analyze_single_image(
|
||||
*,
|
||||
pro_fallback_model: str | None = None,
|
||||
) -> dict:
|
||||
"""单张图片 VLM 分析(线程池并行调用)。
|
||||
- lite 失败/结果不可用 时自动用 pro 模型降级重试 1 次。
|
||||
- 失败/None/不可用最终返回含默认字段的 dict(不会让用户看到「未识别·无法判断」裸结果)。
|
||||
"""单张图片 VLM 分析(#2040:改为从 prompt_loader 读模板 + XML 解析)。
|
||||
|
||||
lite 失败/不可用时用 pro 降级重试 1 次。失败/None 最终返回含默认字段的 dict。
|
||||
"""
|
||||
from packages.shared.ai_service import call_vision
|
||||
try:
|
||||
from packages.shared.ai_service import call_vision
|
||||
from packages.application.viral_video.prompt_loader import (
|
||||
get_template, render_system_prompt, render_user_prompt,
|
||||
)
|
||||
from packages.application.viral_video import xml_parser as xp
|
||||
except ImportError as e:
|
||||
logger.warning("[爆款视频] prompt 模板/解析模块不可用: %s", e)
|
||||
return _vision_fallback(idx, f"fallback_import_error:{e}")
|
||||
|
||||
if not img_url or not isinstance(img_url, str):
|
||||
return _vision_fallback(idx, "invalid_url")
|
||||
|
||||
template = get_template("image_analysis")
|
||||
system = render_system_prompt(template)
|
||||
user = render_user_prompt(
|
||||
template, image_count=1, industry="通用",
|
||||
image_urls=f"第1张:{img_url}",
|
||||
)
|
||||
|
||||
def _call(model: str, tmo: int):
|
||||
try:
|
||||
return call_vision(
|
||||
image_url=img_url,
|
||||
prompt=_IMAGE_ANALYSIS_USER_PROMPT,
|
||||
model=model,
|
||||
max_tokens=800,
|
||||
temperature=0.1,
|
||||
timeout=tmo,
|
||||
system_prompt=_IMAGE_ANALYSIS_SYSTEM_PROMPT,
|
||||
image_url=img_url, prompt=user, model=model,
|
||||
max_tokens=2048, temperature=0.3, timeout=tmo,
|
||||
system_prompt=system,
|
||||
)
|
||||
except Exception as e:
|
||||
logger.warning("[爆款视频] 图片 #%d call_vision(%s) 异常 err=%s", idx, model, e)
|
||||
return None
|
||||
|
||||
def _xml_to_product(nodes: list, raw_text: str) -> dict:
|
||||
product_nodes = [n for n in nodes if n["tag"] == "product"]
|
||||
scene = xp.text_of(raw_text, "scene") or "通用"
|
||||
mood = xp.text_of(raw_text, "mood") or ""
|
||||
for p in product_nodes:
|
||||
a = p["attrs"]
|
||||
text_on_pkg = a.get("text_on_package", "")
|
||||
text_list = [x.strip() for x in re.split(r"[,,;;]", text_on_pkg) if x.strip()] if text_on_pkg else []
|
||||
features = a.get("features", "")
|
||||
feat_list = [x.strip() for x in re.split(r"[,,;;]", features) if x.strip()] if features else []
|
||||
name = a.get("name", "") or "未识别"
|
||||
brand = a.get("brand", "") or "无法判断"
|
||||
category = a.get("category", "") or "无法判断"
|
||||
appearance = a.get("appearance", "") or "无法判断"
|
||||
packaging = a.get("packaging", "") or "无法判断"
|
||||
summary = a.get("summary", "") or f"{brand} {name}"
|
||||
return {
|
||||
"name": name, "brand": brand, "category": category,
|
||||
"appearance": appearance, "packaging": packaging,
|
||||
"text_on_package": text_list,
|
||||
"key_features": feat_list or [features] if features else ["无法判断"],
|
||||
"scene": scene, "mood": mood,
|
||||
"portrait_prompt": a.get("portrait_prompt", "无人像"),
|
||||
"summary": summary, "_source": "xml",
|
||||
}
|
||||
return _vision_fallback(idx, "no_product_tag")
|
||||
|
||||
def _normalize(raw, source: str) -> dict:
|
||||
if raw is None:
|
||||
return _vision_fallback(idx, f"{source}_none")
|
||||
if isinstance(raw, str):
|
||||
logger.warning("[爆款视频] 图片 #%d VLM(%s) 返回非 JSON: %s", idx, source, raw[:200])
|
||||
return _vision_fallback(idx, f"{source}_text", {"_raw": raw[:500]})
|
||||
if not isinstance(raw, dict):
|
||||
if not isinstance(raw, str):
|
||||
return _vision_fallback(idx, f"{source}_badtype")
|
||||
raw.setdefault("_source", source)
|
||||
raw.setdefault("name", "未识别")
|
||||
raw.setdefault("brand", "无法判断")
|
||||
raw.setdefault("category", "无法判断")
|
||||
raw.setdefault("appearance", "无法判断")
|
||||
raw.setdefault("packaging", "无法判断")
|
||||
raw.setdefault("text_on_package", [])
|
||||
raw.setdefault("key_features", [])
|
||||
raw.setdefault("scene", "通用")
|
||||
raw.setdefault("portrait_prompt", "无人像")
|
||||
raw.setdefault("summary", "")
|
||||
if not isinstance(raw.get("text_on_package"), list):
|
||||
raw["text_on_package"] = []
|
||||
if not isinstance(raw.get("key_features"), list):
|
||||
raw["key_features"] = []
|
||||
return raw
|
||||
nodes = xp.parse_tags(raw)
|
||||
if not nodes:
|
||||
logger.warning("[爆款视频] 图片 #%d XML 解析失败 source=%s", idx, source)
|
||||
return _vision_fallback(idx, f"{source}_xml_fail", {"_raw": raw[:500]})
|
||||
product = _xml_to_product(nodes, raw)
|
||||
product.setdefault("_source", source)
|
||||
product["raw"] = raw[:500]
|
||||
return product
|
||||
|
||||
# 第一次:传入模型(通常是 lite)
|
||||
first_raw = _call(vision_model, timeout)
|
||||
tag1 = vision_model.split("/")[-1] if "/" in vision_model else vision_model
|
||||
first_result = _normalize(first_raw, tag1)
|
||||
if _is_vision_result_usable(first_result):
|
||||
return first_result
|
||||
|
||||
logger.warning(
|
||||
"[爆款视频] 图片 #%d VLM(%s) 结果不可用 name=%r summary_len=%d,尝试 pro 降级",
|
||||
idx,
|
||||
vision_model,
|
||||
first_result.get("name"),
|
||||
len(first_result.get("summary") or ""),
|
||||
)
|
||||
|
||||
# 第二次:pro 降级重试
|
||||
if pro_fallback_model and pro_fallback_model != vision_model:
|
||||
pro_raw = _call(pro_fallback_model, 60) # #2180: pro VLM 实测也需25-38s,原25s太短,提到60s
|
||||
pro_result = _normalize(pro_raw, "pro_fallback")
|
||||
if _is_vision_result_usable(pro_result):
|
||||
pro_result["_fallback_used"] = True
|
||||
return pro_result
|
||||
logger.warning(
|
||||
"[爆款视频] 图片 #%d pro 降级仍不可用 name=%r summary_len=%d",
|
||||
idx,
|
||||
pro_result.get("name"),
|
||||
len(pro_result.get("summary") or ""),
|
||||
)
|
||||
return pro_result
|
||||
|
||||
return first_result
|
||||
|
||||
|
||||
def _step_image_analysis(job: ViralVideoJob) -> dict:
|
||||
"""步骤 1: 图片 VLM 分析 — 识别产品特征(v1.6 优化:并行 + lite 模型提速)。"""
|
||||
try:
|
||||
@@ -522,7 +486,6 @@ def _step_image_analysis(job: ViralVideoJob) -> dict:
|
||||
|
||||
return {"products": results}
|
||||
|
||||
|
||||
def _step_video_analysis(job: ViralVideoJob) -> dict | None:
|
||||
"""步骤 1.5: 参考视频风格分析(可选)。"""
|
||||
if not job.reference_video_url:
|
||||
@@ -546,11 +509,14 @@ def _step_video_analysis(job: ViralVideoJob) -> dict | None:
|
||||
logger.error("[爆款视频] 视频风格分析失败: %s", e)
|
||||
return {"error": str(e), "source": "failed"}
|
||||
|
||||
|
||||
def _step_intent_parsing(job: ViralVideoJob, image_analysis: dict) -> dict:
|
||||
"""步骤 2: 用户文案意图解析。"""
|
||||
"""步骤 2: 用户文案意图解析(#2040:改为模板 + XML 解析)。"""
|
||||
try:
|
||||
from packages.shared.ai_service import call_llm
|
||||
from packages.application.viral_video.prompt_loader import (
|
||||
get_template, render_system_prompt, render_user_prompt,
|
||||
)
|
||||
from packages.application.viral_video import xml_parser as xp
|
||||
except ImportError:
|
||||
return {"intent": "推广产品", "key_messages": ["产品亮点"], "tone": "专业", "suggested_title": ""}
|
||||
|
||||
@@ -572,25 +538,25 @@ def _step_intent_parsing(job: ViralVideoJob, image_analysis: dict) -> dict:
|
||||
feat_str = ", ".join([str(x) for x in feats + extras])
|
||||
products_summary += f"- {p.get('name', '产品')}: {feat_str}\n"
|
||||
|
||||
prompt = f"""你是一个营销编导。请分析以下信息,理解用户的营销意图并给出短视频主题建议:
|
||||
template = get_template("intent_parsing")
|
||||
system = render_system_prompt(template)
|
||||
user = render_user_prompt(
|
||||
template,
|
||||
user_copy_text=job.user_copy_text or "(未提供,全由 AI 创作)",
|
||||
industry=job.industry or "未指定",
|
||||
image_analysis=products_summary or "- (无图片分析结果)",
|
||||
)
|
||||
|
||||
用户原始文案:{job.user_copy_text or "(未提供,全由 AI 创作)"}
|
||||
行业:{job.industry or "未指定"}
|
||||
目标客户:{job.target_customer or "未指定"}
|
||||
营销目的:{job.marketing_purpose or "未指定"}
|
||||
视频时长:{job.duration}秒
|
||||
产品信息:
|
||||
{products_summary or "- (无图片分析结果)"}
|
||||
def _parse(raw: str) -> dict:
|
||||
summary = xp.text_of(raw, "intent_summary")
|
||||
msgs = [n["text"] for n in xp.find_all(raw, "message") if n["text"]]
|
||||
tone = xp.text_of(raw, "emotion_tone") or "亲切自然"
|
||||
title = xp.text_of(raw, "suggested_title") or xp.text_of(raw, "title")
|
||||
return {"intent": summary or "推广产品", "key_messages": msgs or ["产品亮点"], "tone": tone, "suggested_title": title}
|
||||
|
||||
请返回严格 JSON(不要 Markdown,不要解释):
|
||||
{{
|
||||
"intent": "核心营销意图(一句话)",
|
||||
"key_messages": ["要传达的3-5个关键信息"],
|
||||
"tone": "文案调性(如亲切/专业/高端/活力/治愈/搞笑)",
|
||||
"target_emotion": "希望触发的用户情感",
|
||||
"call_to_action": "行动号召短句(口语化,5-10字)",
|
||||
"suggested_title": "视频主题标题(5-15字)"
|
||||
}}"""
|
||||
def _fallback(raw: str) -> dict:
|
||||
t = (job.user_copy_text or "").strip()
|
||||
return {"intent": t[:30] or "推广产品", "key_messages": [t[:80]] if t else ["产品亮点"], "tone": "专业", "suggested_title": ""}
|
||||
|
||||
_s = get_shared_settings()
|
||||
_fast = _s.doubao_fast_model
|
||||
@@ -598,22 +564,17 @@ def _step_intent_parsing(job: ViralVideoJob, image_analysis: dict) -> dict:
|
||||
for _m, _lbl in [(_fast, "fast"), (_pro, "pro-fallback")]:
|
||||
try:
|
||||
logger.info("[爆款视频] 意图解析 model=%s label=%s", _m, _lbl)
|
||||
result = call_llm(prompt, temperature=0.4, max_tokens=800, model=_m, timeout=45)
|
||||
return (
|
||||
result
|
||||
if isinstance(result, dict)
|
||||
else {"intent": str(result)[:200], "key_messages": [], "tone": "专业", "suggested_title": ""}
|
||||
)
|
||||
raw = call_llm([{"role": "system", "content": system}, {"role": "user", "content": user}],
|
||||
temperature=0.4, max_tokens=1024, model=_m, timeout=45) # #2180: 意图解析 LLM 实测需更长响应,原25s太紧
|
||||
if not raw:
|
||||
continue
|
||||
parsed = _parse(raw)
|
||||
if parsed["intent"] or parsed["key_messages"]:
|
||||
return parsed
|
||||
except Exception as e:
|
||||
logger.warning("[爆款视频] 意图解析失败 label=%s err=%s", _lbl, e)
|
||||
continue
|
||||
return {"intent": "推广产品", "key_messages": ["产品亮点"], "tone": "专业", "suggested_title": ""}
|
||||
return _fallback("")
|
||||
|
||||
|
||||
# ── 编导分镜脚本生成(核心,v1.6 新 prompt) ──────────────────────────────
|
||||
|
||||
|
||||
# 人设 IP 类型 → 文案/出镜风格指导(前端下拉 10 个 IP 类型)
|
||||
_PERSONA_STYLE_GUIDE = {
|
||||
"通用个人IP": "亲切自然、像朋友分享好物,第一人称口语化,不端着",
|
||||
"老板型IP": "沉稳大气、有行业格局感,适度使用『我做了XX年』『我一直坚持』等老板视角,语气自信不夸张",
|
||||
@@ -627,7 +588,6 @@ _PERSONA_STYLE_GUIDE = {
|
||||
"测评种草型": "真实测评感、讲使用体验和优缺点对比,带『亲测』『我用了XX天』『实测下来』真实感词汇",
|
||||
}
|
||||
|
||||
|
||||
def _persona_style_hint(persona_id: str) -> str:
|
||||
"""根据 persona_id 查文案风格指导;未命中/空值返回通用提示。"""
|
||||
pid = (persona_id or "").strip()
|
||||
@@ -638,74 +598,6 @@ def _persona_style_hint(persona_id: str) -> str:
|
||||
return f"【人设风格:{pid}】按该人设的口吻、话术习惯组织口播和出镜动作"
|
||||
return "【人设风格:未指定】亲切自然、像朋友分享好物"
|
||||
|
||||
|
||||
_SCRIPT_GENERATION_PROMPT = """你是资深短视频导演,为 Seedance 2.5(单次生成最多{duration}秒)写编导分镜脚本。脚本将整体作为 prompt 一次性传给视频模型,必须让模型在连贯镜头流中清楚每段时间拍什么、画面如何、人物说什么。
|
||||
|
||||
## 产品
|
||||
{products_summary}
|
||||
|
||||
## 营销参数
|
||||
- 主题/意图:{intent}
|
||||
- 关键信息:{key_messages}
|
||||
- 调性:{tone}
|
||||
- 目标客户:{target_customer}
|
||||
- 用户原始卖点(必须融入口播):{user_copy}
|
||||
- 时长:{duration}秒 / 画幅:{ratio} / 产品图:{n_images}张(第1张通常是主图/首帧)
|
||||
- 风格参考:{style_hint}
|
||||
- 爆款结构(必须严格遵循节奏/段落顺序):{viral_structure_block}
|
||||
- 人设/出镜口吻(必须贯穿全部对白和动作描写):{persona_hint}
|
||||
|
||||
## 输出格式(必须输出严格 JSON,不要 Markdown,不要解释,字段一个都不能少)
|
||||
|
||||
```json
|
||||
{{
|
||||
"overview": {{
|
||||
"theme": "视频主题(一句话概括)",
|
||||
"total_duration": {duration},
|
||||
"aspect_ratio": "{ratio}"
|
||||
}},
|
||||
"scene_and_lighting": "整体场景描述+光线设定(100-200字,要具体:在哪拍、什么光线、什么色调、什么氛围)",
|
||||
"shots": [
|
||||
{{
|
||||
"time_range": "0-3秒",
|
||||
"shot_type_angle_movement": "景别+角度+运镜(例:近景俯拍45度,缓慢推镜;中景平视,固定镜头;特写平视,快速拉镜)",
|
||||
"scene_and_dialogue": "画面场景描述 + 人物口播台词(对白要自然口语化,像朋友聊天,不要硬广推销腔)",
|
||||
"action_details": "人物动作、表情、物品操作细节(手怎么动、表情变化、产品怎么展示)",
|
||||
"audio_bgm": "环境音+BGM提示(例:轻快流行BGM,环境嘈杂咖啡店背景音)",
|
||||
"transition": "硬切/淡入淡出/叠化(最后一镜写『结束』即可)",
|
||||
"reference_image_index": 0
|
||||
}}
|
||||
// ... 按时间顺序列出所有镜头,总时长累计 = {duration} 秒
|
||||
],
|
||||
"hard_constraints": [
|
||||
"无字幕、无水印、无任何自动生成文字、无logo",
|
||||
"同一人物全程五官、发型、服装、身材保持一致,不得换脸变形",
|
||||
"口播语音在总时长内自然念完,语速自然,口型与语音严格同步",
|
||||
"画面流畅无闪烁、无多余肢体、无扭曲变形、无穿模",
|
||||
"色彩自然、曝光正确、电影级质感、高清细节"
|
||||
],
|
||||
"negative_prompts": [
|
||||
"字幕","自动字幕","水印","logo","图标","错误文字","乱码文字",
|
||||
"男女声错配","中途换声","五官崩坏","脸部变形","多余手指",
|
||||
"肢体扭曲","闪烁","画面抖动","模糊","低分辨率"
|
||||
],
|
||||
"voiceover_script": "完整口播稿(把 shots 里所有对白自然拼接成一段,口语化,不加旁白标注、不加镜头标注、不加'主播:'之类前缀,就是纯念出来的文本,长度适配{duration}秒,约{approx_chars}字)"
|
||||
}}
|
||||
```
|
||||
|
||||
## 关键要求
|
||||
1. 每镜写清景别/角度/运镜(特写/近景/中景+平视/俯拍+推/拉/固定)。
|
||||
2. 画面具体:主体(性别/年龄/穿着)、场景、动作、光线、镜头运动要可落地。
|
||||
3. 对白自然口语化,像朋友分享好物;拒绝"家人们""宝子们""太好用了"等浮夸/硬广腔。
|
||||
4. reference_image_index 填 0-based 索引(产品特写用索引0主图),人像/场景可 null。
|
||||
5. shots time_range 累计={duration}秒,单镜2-8秒。
|
||||
6. hard_constraints/negative_prompts 保留默认项可追加,不要删减。
|
||||
7. voiceover_script 为纯口播文本(无标记/括号/前缀),{duration}秒约{approx_chars}字。
|
||||
8. 严格按上方「爆款结构」的节奏/段落顺序编排(钩子/痛点/反转/案例/行动号召与结构对齐)。
|
||||
9. 输出前自检:口播对白禁止错别字和语病,**严禁使用"很近",正确用词是"最近"**(指"最近一段时间/最近在用",绝不能写成"很近");其他同音字、形近字错误一律修正。
|
||||
10. 必须使用产品信息中真实的品牌、品名和外观特征,不要编造与产品无关的内容。"""
|
||||
|
||||
|
||||
def _build_products_summary(image_analysis: dict) -> str:
|
||||
"""把 VLM 返回的商品分析结果拼给文案/分镜生成 prompt 用。
|
||||
优先用 summary(自然段落);没有时用结构化字段兜底拼一段。"""
|
||||
@@ -774,7 +666,6 @@ def _build_products_summary(image_analysis: dict) -> str:
|
||||
lines.append("- " + ",".join(parts))
|
||||
return "\n".join(lines)
|
||||
|
||||
|
||||
def _safe_json_loads(raw: str | dict | list | None):
|
||||
if raw is None:
|
||||
return None
|
||||
@@ -801,7 +692,6 @@ def _safe_json_loads(raw: str | dict | list | None):
|
||||
pass
|
||||
return None
|
||||
|
||||
|
||||
def _replace_henjin_everywhere(obj: Any) -> Any:
|
||||
"""递归遍历 copy_result 里所有字符串值,把'很近'替换成'最近'。
|
||||
覆盖 overview.theme、scene_and_lighting、voiceover_script、
|
||||
@@ -817,7 +707,6 @@ def _replace_henjin_everywhere(obj: Any) -> Any:
|
||||
return {k: _replace_henjin_everywhere(v) for k, v in obj.items()}
|
||||
return obj
|
||||
|
||||
|
||||
def _fallback_script(job: ViralVideoJob) -> dict:
|
||||
"""脚本生成失败时的兜底脚本(极简但可用)。"""
|
||||
dur = max(5, min(30, int(getattr(job, "duration", 15) or 15)))
|
||||
@@ -842,7 +731,6 @@ def _fallback_script(job: ViralVideoJob) -> dict:
|
||||
base["title"] = "好物分享"
|
||||
return base
|
||||
|
||||
|
||||
def _validate_and_normalize_script(raw, job: ViralVideoJob) -> dict:
|
||||
"""把 LLM 返回的脚本规范化、补默认、校验结构。"""
|
||||
dur = max(5, min(30, int(getattr(job, "duration", 15) or 15)))
|
||||
@@ -939,11 +827,68 @@ def _validate_and_normalize_script(raw, job: ViralVideoJob) -> dict:
|
||||
base["title"] = base["overview"]["theme"]
|
||||
return base
|
||||
|
||||
def _script_from_xml(raw: str, job: ViralVideoJob) -> dict | None:
|
||||
"""把 LLM 返回的 XML 分镜规范化为旧 copy_result 结构(供 Seedance 使用)。"""
|
||||
from packages.application.viral_video import xml_parser as xp
|
||||
dur = max(5, min(30, int(getattr(job, "duration", 15) or 15)))
|
||||
ratio = getattr(job, "video_ratio", None) or "9:16"
|
||||
base = _empty_copy_result(dur, ratio)
|
||||
if not raw:
|
||||
return None
|
||||
base["overview"]["theme"] = xp.text_of(raw, "overview_theme") or xp.text_of(raw, "title") or "好物分享"
|
||||
est = xp.attr_int(xp.text_of(raw, "estimated_duration"), 0)
|
||||
if est:
|
||||
base["overview"]["total_duration"] = est
|
||||
sl = xp.text_of(raw, "scene_and_lighting")
|
||||
if sl:
|
||||
base["scene_and_lighting"] = sl
|
||||
clips = xp.find_all(raw, "clip")
|
||||
shots: list[dict] = []
|
||||
voice_parts: list[str] = []
|
||||
for i, c in enumerate(clips):
|
||||
a = c["attrs"]
|
||||
body = c.get("text", "") or ""
|
||||
ref_idx_raw = a.get("reference_image_index", "")
|
||||
if ref_idx_raw in (None, "", "null", "None"):
|
||||
body_ref = xp.text_of(body, "reference_image_index") if body else ""
|
||||
ref_idx = xp.attr_int(body_ref, 0) if body_ref else None
|
||||
else:
|
||||
ref_idx = xp.attr_int(ref_idx_raw, 0)
|
||||
shot = {
|
||||
"time_range": a.get("time_range") or f"{i * 3}-{(i + 1) * 3}秒",
|
||||
"shot_type_angle_movement": (xp.text_of(body, "shot_type_angle_movement") if body else "") or a.get("shot_type_angle_movement", "") or "中景平视,固定镜头",
|
||||
"scene_and_dialogue": (xp.text_of(body, "scene_and_dialogue") if body else "") or "",
|
||||
"action_details": (xp.text_of(body, "action_details") if body else "") or "",
|
||||
"audio_bgm": (xp.text_of(body, "audio_bgm") if body else "") or a.get("bgm_note", "") or "轻快BGM",
|
||||
"transition": (xp.text_of(body, "transition") if body else "") or a.get("transition", "") or ("硬切" if i < len(clips) - 1 else "结束"),
|
||||
"reference_image_index": ref_idx,
|
||||
}
|
||||
voice = (xp.text_of(body, "voice_text") if body else "")
|
||||
if voice:
|
||||
voice_parts.append(voice)
|
||||
if not shot["scene_and_dialogue"]:
|
||||
shot["scene_and_dialogue"] = voice
|
||||
shots.append(shot)
|
||||
if not shots:
|
||||
return None
|
||||
base["shots"] = shots
|
||||
joined = xp.text_of(raw, "voiceover_script") or "。".join(voice_parts)
|
||||
base["voiceover_script"] = joined
|
||||
base["final_copy"] = joined
|
||||
base["suggested_copy"] = joined
|
||||
base["title"] = base["overview"]["theme"]
|
||||
return base
|
||||
|
||||
def _step_script_generation(job: ViralVideoJob, intent: dict, image_analysis: dict) -> dict:
|
||||
"""步骤 3: 编导分镜脚本生成(v1.6 核心,输出 copy_result 结构)。"""
|
||||
"""步骤 3: 编导分镜脚本生成(#2040:模板 + XML 解析;输出 copy_result 结构)。"""
|
||||
try:
|
||||
from packages.shared.ai_service import call_llm
|
||||
from packages.application.viral_video.prompt_loader import (
|
||||
get_template, render_system_prompt, render_user_prompt,
|
||||
)
|
||||
from packages.application.viral_video.prompts import (
|
||||
FUSION_INSTRUCTIONS, GLOBAL_CONSTRAINTS, NEGATIVE_RULES,
|
||||
)
|
||||
except ImportError:
|
||||
return _fallback_script(job)
|
||||
|
||||
@@ -957,20 +902,15 @@ def _step_script_generation(job: ViralVideoJob, intent: dict, image_analysis: di
|
||||
|
||||
dur = max(5, min(30, int(getattr(job, "duration", 15) or 15)))
|
||||
ratio = getattr(job, "video_ratio", None) or "9:16"
|
||||
approx_chars = max(20, dur * 4)
|
||||
|
||||
intent_str = ""
|
||||
key_msgs = ""
|
||||
tone = ""
|
||||
intent_str = "推广产品"
|
||||
key_msgs = "产品亮点"
|
||||
tone = "亲切自然"
|
||||
if isinstance(intent, dict):
|
||||
intent_str = intent.get("intent") or "推广产品"
|
||||
key_msgs = "、".join(intent.get("key_messages") or [])
|
||||
tone = intent.get("tone") or "亲切自然"
|
||||
else:
|
||||
intent_str = "推广产品"
|
||||
tone = "亲切自然"
|
||||
intent_str = intent.get("intent") or intent_str
|
||||
key_msgs = "、".join(intent.get("key_messages") or []) or key_msgs
|
||||
tone = intent.get("tone") or tone
|
||||
|
||||
# 爆款结构:用户在 STEP1/STEP2 选的中文结构名,必须严格注入 prompt 指导 AI 编排
|
||||
_vs = (job.viral_structure or "").strip()
|
||||
if _vs:
|
||||
viral_structure_block = f"【{_vs}】—— 请严格按照这个爆款结构的节奏/段落顺序编排镜头、台词和情绪节点(开场钩子、痛点、反转、案例、行动号召等按结构走),不要打乱顺序"
|
||||
@@ -978,59 +918,56 @@ def _step_script_generation(job: ViralVideoJob, intent: dict, image_analysis: di
|
||||
viral_structure_block = "未指定(自由编排,但仍需有钩子开头+产品展示+行动号召的基本节奏)"
|
||||
|
||||
persona_hint = _persona_style_hint(getattr(job, "persona_id", ""))
|
||||
prompt = _SCRIPT_GENERATION_PROMPT.format(
|
||||
products_summary=products_summary,
|
||||
intent=intent_str,
|
||||
key_messages=key_msgs or "产品亮点",
|
||||
tone=tone,
|
||||
target_customer=job.target_customer or "通用人群",
|
||||
user_copy=job.user_copy_text or "(未提供,自由创作)",
|
||||
duration=dur,
|
||||
ratio=ratio,
|
||||
n_images=len(job.images or []),
|
||||
style_hint=style_hint,
|
||||
approx_chars=approx_chars,
|
||||
viral_structure_block=viral_structure_block,
|
||||
persona_hint=persona_hint,
|
||||
fusion_level = getattr(job, "fusion_level", "ai_polish") or "ai_polish"
|
||||
fusion_instruction = FUSION_INSTRUCTIONS.get(fusion_level, FUSION_INSTRUCTIONS["ai_polish"])
|
||||
|
||||
# 使用 storyboard 模板,注入融合指令/硬约束/反套路词
|
||||
template = get_template("storyboard")
|
||||
system_tpl = template.system_prompt
|
||||
system_tpl = system_tpl.replace("{fusion_instruction}", fusion_instruction)
|
||||
system_tpl = system_tpl.replace("{global_constraints}", GLOBAL_CONSTRAINTS)
|
||||
system_tpl = system_tpl.replace("{negative_rules}", NEGATIVE_RULES)
|
||||
|
||||
fusion_brief = (
|
||||
f"意图:{intent_str}\n关键信息:{key_msgs}\n调性:{tone}\n"
|
||||
f"用户原文:{job.user_copy_text or '(未提供)'}\n创作模式:{fusion_level}"
|
||||
)
|
||||
user = render_user_prompt(
|
||||
template,
|
||||
duration=dur, image_count=len(job.images or []),
|
||||
fusion_result=fusion_brief,
|
||||
image_analysis=products_summary,
|
||||
)
|
||||
|
||||
def _try_gen(model: str, temp: float, max_tok: int, label: str, tmo: int = 25):
|
||||
logger.info("[爆款视频] 编导脚本生成 model=%s label=%s timeout=%d", model, label, tmo)
|
||||
raw = call_llm([{"role": "system", "content": system_tpl}, {"role": "user", "content": user}],
|
||||
temperature=temp, max_tokens=max_tok, model=model, timeout=tmo)
|
||||
if not raw:
|
||||
return None
|
||||
normalized = _script_from_xml(raw, job)
|
||||
if normalized is None:
|
||||
# 兼容:万一 LLM 仍输出 JSON,走旧规范化
|
||||
parsed_json = _safe_json_loads(raw)
|
||||
if isinstance(parsed_json, dict):
|
||||
normalized = _validate_and_normalize_script(parsed_json, job)
|
||||
else:
|
||||
return None
|
||||
voiceover = (normalized or {}).get("voiceover_script") or ""
|
||||
shots_cnt = len((normalized or {}).get("shots") or [])
|
||||
_before_dump = json.dumps(normalized, ensure_ascii=False)
|
||||
if "很近" in _before_dump:
|
||||
normalized = _replace_henjin_everywhere(normalized)
|
||||
voiceover = (normalized or {}).get("voiceover_script") or ""
|
||||
fallback_marker = "我最近在用的好物" in voiceover
|
||||
has_typo_henjin = "很近" in json.dumps(normalized, ensure_ascii=False)
|
||||
is_fallback = fallback_marker or shots_cnt < 1 or len(voiceover) < 20 or has_typo_henjin
|
||||
logger.info("[爆款视频] 编导脚本结果 label=%s voiceover_len=%d shots=%d fallback=%s", label, len(voiceover), shots_cnt, is_fallback)
|
||||
return None if is_fallback else normalized
|
||||
|
||||
_s = get_shared_settings()
|
||||
_fast = _s.doubao_fast_model
|
||||
_pro = getattr(_s, "doubao_model", None) or _fast
|
||||
|
||||
def _try_gen(model: str, temp: float, max_tok: int, label: str, tmo: int = 25):
|
||||
logger.info("[爆款视频] 编导脚本生成 model=%s label=%s timeout=%d", model, label, tmo)
|
||||
r = call_llm(prompt, temperature=temp, max_tokens=max_tok, model=model, timeout=tmo)
|
||||
if r is None:
|
||||
logger.warning("[爆款视频] 编导脚本返回None label=%s", label)
|
||||
return None
|
||||
parsed = _safe_json_loads(r)
|
||||
normalized = _validate_and_normalize_script(parsed, job)
|
||||
voiceover = (normalized or {}).get("voiceover_script") or ""
|
||||
voiceover_len = len(voiceover)
|
||||
shots_cnt = len((normalized or {}).get("shots") or [])
|
||||
# 判定是否"退化到兜底质量":口播过短(<20字)或镜头数<1;正常的短口播(如15s视频~40字)不视为兜底
|
||||
# v1.6.1 P1修复:递归替换 copy_result 里所有字符串字段的"很近"→"最近"(覆盖 overview/scene_and_lighting/voiceover/shots.* 全部字段)
|
||||
_before_dump = json.dumps(normalized, ensure_ascii=False)
|
||||
if "很近" in _before_dump:
|
||||
logger.warning("[爆款视频] 编导脚本含错别字'很近',递归替换为'最近' label=%s", label)
|
||||
normalized = _replace_henjin_everywhere(normalized)
|
||||
voiceover = (normalized or {}).get("voiceover_script") or ""
|
||||
fallback_marker = "我最近在用的好物" in voiceover # _fallback_script 的特征串
|
||||
has_typo_henjin = "很近" in json.dumps(normalized, ensure_ascii=False) # 递归检查仍有"很近"视为不合格
|
||||
is_fallback = fallback_marker or shots_cnt < 1 or voiceover_len < 20 or has_typo_henjin
|
||||
logger.info(
|
||||
"[爆款视频] 编导脚本结果 label=%s voiceover_len=%d shots=%d fallback=%s raw_type=%s",
|
||||
label,
|
||||
voiceover_len,
|
||||
shots_cnt,
|
||||
is_fallback,
|
||||
type(r).__name__,
|
||||
)
|
||||
if is_fallback:
|
||||
return None # 触发重试
|
||||
return normalized
|
||||
|
||||
try:
|
||||
# 第一次:快模型 25s
|
||||
normalized = _try_gen(_fast, 0.8, 2500, "fast-first", tmo=45)
|
||||
@@ -1040,7 +977,6 @@ def _step_script_generation(job: ViralVideoJob, intent: dict, image_analysis: di
|
||||
normalized = _try_gen(_fast, 0.6, 3200, "fast-retry", tmo=45)
|
||||
if normalized is not None:
|
||||
return normalized
|
||||
# 第三次:用主力模型兜底,给 40s
|
||||
if _pro and _pro != _fast:
|
||||
normalized = _try_gen(_pro, 0.7, 3500, "pro-fallback", tmo=60)
|
||||
if normalized is not None:
|
||||
@@ -1051,41 +987,81 @@ def _step_script_generation(job: ViralVideoJob, intent: dict, image_analysis: di
|
||||
logger.warning("[爆款视频] 编导脚本生成异常: %s,使用兜底脚本", e, exc_info=True)
|
||||
return _fallback_script(job)
|
||||
|
||||
|
||||
def _step_review(job: ViralVideoJob, copy_result: dict) -> dict:
|
||||
"""步骤 4: 合规审核(简化版:基于脚本的 voiceover_script+shots 文本)。"""
|
||||
dimensions = ["广告法合规", "平台规范", "内容真实性", "版权安全", "价值观", "风格一致性"]
|
||||
"""步骤 4: 合规审核(#2040:使用 Reviewer + review 模板,6 维度 + 自动重写 1 次)。
|
||||
|
||||
返回结构与旧版兼容:{passed, score, details, issues, rewritten_copy?}
|
||||
"""
|
||||
try:
|
||||
from packages.shared.ai_service import call_llm
|
||||
except ImportError:
|
||||
return {"passed": True, "score": 90, "details": {d: "通过" for d in dimensions}}
|
||||
from packages.application.viral_video.reviewer import Reviewer
|
||||
from packages.application.viral_video.schemas import (
|
||||
FusionResult, IntentResult, CoreMessage, PersonalBrand, ScriptSegment,
|
||||
)
|
||||
except ImportError as e:
|
||||
logger.warning("[爆款视频] reviewer 模块不可用,跳过审核: %s", e)
|
||||
return {"passed": True, "score": 80, "details": {}, "issues": []}
|
||||
|
||||
voiceover = (copy_result or {}).get("voiceover_script", "")
|
||||
shots_preview = json.dumps((copy_result or {}).get("shots", [])[:3], ensure_ascii=False)
|
||||
prompt = f"""请对以下短视频编导脚本进行合规审核,检查6个维度:{", ".join(dimensions)}
|
||||
voiceover = (copy_result or {}).get("voiceover_script", "") or ""
|
||||
title = (copy_result or {}).get("title") or (copy_result or {}).get("overview", {}).get("theme", "")
|
||||
|
||||
口播文案:{voiceover}
|
||||
前3个镜头:{shots_preview}
|
||||
行业:{job.industry}
|
||||
|
||||
请以JSON格式返回:
|
||||
- passed: bool(是否全部通过)
|
||||
- score: int(0-100分)
|
||||
- details: 各维度评分和说明
|
||||
- issues: 需要修改的问题列表(如有)"""
|
||||
_s = get_shared_settings()
|
||||
_fast = _s.doubao_fast_model
|
||||
_pro = _s.doubao_model
|
||||
for _m, _lbl in [(_fast, "fast"), (_pro, "pro-fallback")]:
|
||||
try:
|
||||
logger.info("[爆款视频] 合规审核 model=%s label=%s", _m, _lbl)
|
||||
result = call_llm(prompt, temperature=0.1, max_tokens=500, model=_m, timeout=30)
|
||||
return result if isinstance(result, dict) else {"passed": True, "score": 80, "details": {}}
|
||||
except Exception as e:
|
||||
logger.warning("[爆款视频] 合规审核失败 label=%s err=%s", _lbl, e)
|
||||
continue
|
||||
return {"passed": True, "score": 75, "details": {d: "默认通过" for d in dimensions}}
|
||||
intent_data = job.intent_result or {}
|
||||
core_msgs = [CoreMessage(text=str(m), must_keep=True, confidence=0.9) for m in (intent_data.get("key_messages") or [])]
|
||||
brands: list[PersonalBrand] = []
|
||||
brand_text = intent_data.get("brand_text") or intent_data.get("suggested_title") or ""
|
||||
if brand_text:
|
||||
brands.append(PersonalBrand(text=str(brand_text), category="brand"))
|
||||
intent_obj = IntentResult(
|
||||
intent_summary=intent_data.get("intent", "") or "推广产品",
|
||||
core_messages=core_msgs, personal_brands=brands,
|
||||
)
|
||||
fusion_obj = FusionResult(
|
||||
title=title or "",
|
||||
hook=(voiceover[:30] if voiceover else ""),
|
||||
cta="",
|
||||
script_segments=[],
|
||||
word_count=len(voiceover),
|
||||
estimated_duration=int(getattr(job, "duration", 15) or 15),
|
||||
)
|
||||
for shot in (copy_result or {}).get("shots", []) or []:
|
||||
if isinstance(shot, dict) and shot.get("scene_and_dialogue"):
|
||||
fusion_obj.script_segments.append(ScriptSegment(text=shot["scene_and_dialogue"]))
|
||||
|
||||
fusion_level = getattr(job, "fusion_level", "ai_polish") or "ai_polish"
|
||||
try:
|
||||
reviewer = Reviewer()
|
||||
review_res = reviewer.review(fusion_obj, intent_obj, fusion_level)
|
||||
new_copy = copy_result
|
||||
rewritten_voice = None
|
||||
if not review_res.passed and review_res.rewrite_suggestions:
|
||||
try:
|
||||
rewritten = reviewer.rewrite(fusion_obj, review_res, intent_obj, fusion_level)
|
||||
if rewritten and (rewritten.title or rewritten.script_segments):
|
||||
new_voice = rewritten.script_segments[0].text if rewritten.script_segments else (rewritten.hook or voiceover)
|
||||
new_copy = dict(copy_result)
|
||||
new_copy["voiceover_script"] = new_voice
|
||||
new_copy["final_copy"] = new_voice
|
||||
new_copy["suggested_copy"] = new_voice
|
||||
if rewritten.title:
|
||||
new_copy.setdefault("overview", {})["theme"] = rewritten.title
|
||||
new_copy["title"] = rewritten.title
|
||||
rewritten_voice = new_voice
|
||||
review_res = reviewer.review(rewritten, intent_obj, fusion_level)
|
||||
except Exception as e:
|
||||
logger.warning("[爆款视频] 自动重写失败: %s", e)
|
||||
result = {
|
||||
"passed": review_res.passed,
|
||||
"score": 90 if review_res.passed else 60,
|
||||
"details": {i.dimension: i.text for i in review_res.issues},
|
||||
"issues": [{"dimension": i.dimension, "severity": i.severity, "location": i.location, "text": i.text} for i in review_res.issues],
|
||||
}
|
||||
if rewritten_voice is not None:
|
||||
result["rewritten_copy"] = new_copy
|
||||
job.copy_result = new_copy
|
||||
job.generated_copy_text = rewritten_voice
|
||||
return result
|
||||
except Exception as e:
|
||||
logger.warning("[爆款视频] 审核异常,跳过: %s", e, exc_info=True)
|
||||
return {"passed": True, "score": 75, "details": {}, "issues": []}
|
||||
|
||||
def _step_tts(job: ViralVideoJob, voiceover_script: str):
|
||||
"""步骤 5: CosyVoice 整段配音 → 返回本地 MP3 Path;失败返回 None。"""
|
||||
@@ -1125,7 +1101,6 @@ def _step_tts(job: ViralVideoJob, voiceover_script: str):
|
||||
logger.warning("[爆款视频] TTS 配音失败: %s", e, exc_info=True)
|
||||
return None
|
||||
|
||||
|
||||
def _upload_tts_to_oss(job: ViralVideoJob, tts_path) -> str | None:
|
||||
"""把 TTS 本地 mp3 上传到 OSS,返回公网 URL(供 Seedance 做 reference_audios 口型驱动用)。"""
|
||||
if tts_path is None:
|
||||
@@ -1145,7 +1120,6 @@ def _upload_tts_to_oss(job: ViralVideoJob, tts_path) -> str | None:
|
||||
logger.warning("[爆款视频] TTS 上传 OSS 失败: %s", e, exc_info=True)
|
||||
return None
|
||||
|
||||
|
||||
def _assemble_seedance_prompt(copy_result: dict, job: ViralVideoJob) -> str:
|
||||
"""把编导脚本拼成 Seedance 长 prompt。"""
|
||||
if not isinstance(copy_result, dict) or not copy_result:
|
||||
@@ -1196,7 +1170,6 @@ def _assemble_seedance_prompt(copy_result: dict, job: ViralVideoJob) -> str:
|
||||
lines.append(",".join([str(x) for x in np if x]))
|
||||
return "\n".join(lines)
|
||||
|
||||
|
||||
def _step_render(job: ViralVideoJob, copy_result: dict, tts_audio_url: str | None) -> tuple[str, dict | None]:
|
||||
"""步骤 6: v1.6 单次 Seedance 生成(不再分段/拼接)。
|
||||
|
||||
@@ -1310,7 +1283,6 @@ def _step_render(job: ViralVideoJob, copy_result: dict, tts_audio_url: str | Non
|
||||
logger.info("[爆款视频] 单次生成完成: path=%s size=%d usage=%s", video_path, size, usage)
|
||||
return str(video_path), (usage if isinstance(usage, dict) else None)
|
||||
|
||||
|
||||
def _step_upload(job: ViralVideoJob, video_path: str) -> str:
|
||||
"""步骤 7: OSS 上传。"""
|
||||
from video_processing.oss_helpers import upload_to_oss
|
||||
@@ -1323,7 +1295,6 @@ def _step_upload(job: ViralVideoJob, video_path: str) -> str:
|
||||
raise RuntimeError(f"OSS 上传失败: storage_key={storage_key}")
|
||||
return video_url
|
||||
|
||||
|
||||
def _wait_oss_ready(url: str, timeout_sec: int = 10) -> bool:
|
||||
"""轮询 OSS 公网 URL,直到 HEAD 返回 200 或超时。
|
||||
用于缓解 OSS 上传后 1-5s 公网 eventual consistency 导致的 NoSuchKey。
|
||||
@@ -1344,10 +1315,8 @@ def _wait_oss_ready(url: str, timeout_sec: int = 10) -> bool:
|
||||
logger.warning("[爆款视频] OSS 成片在 %ds 内未就绪 last_status=%s url=%s", timeout_sec, last_status, url[:120])
|
||||
return False
|
||||
|
||||
|
||||
# ── 主编排器 ────────────────────────────────────────────────────────────
|
||||
|
||||
|
||||
@shared_task(
|
||||
bind=True,
|
||||
max_retries=2,
|
||||
@@ -1432,7 +1401,6 @@ def run_viral_video_pipeline(self: Task, job_id: str) -> dict:
|
||||
if session:
|
||||
session.close()
|
||||
|
||||
|
||||
@shared_task(bind=True, max_retries=2, name="worker.resume_viral_video_pipeline")
|
||||
def resume_viral_video_pipeline(self: Task, job_id: str) -> dict:
|
||||
"""旧 confirm-intent 路径兼容:从 WAIT_USER_CONFIRM 跑完整个渲染。"""
|
||||
@@ -1454,7 +1422,6 @@ def resume_viral_video_pipeline(self: Task, job_id: str) -> dict:
|
||||
if session:
|
||||
session.close()
|
||||
|
||||
|
||||
@shared_task(bind=True, max_retries=1, name="worker.run_video_style_analysis")
|
||||
def run_video_style_analysis(self: Task, job_id: str) -> dict:
|
||||
"""独立的视频风格分析任务。"""
|
||||
@@ -1478,10 +1445,8 @@ def run_video_style_analysis(self: Task, job_id: str) -> dict:
|
||||
if session:
|
||||
session.close()
|
||||
|
||||
|
||||
# ── 失败处理 ────────────────────────────────────────────────────────────
|
||||
|
||||
|
||||
def _mark_failed_and_notify(job_id: str, session, repo, job, err_msg: str, stage: str = "") -> None:
|
||||
"""标记任务失败并通知。若传入的 session 已失效(因前面异常导致 rollback 状态),
|
||||
会自动 fallback 到新建 SessionLocal 重新标记,确保状态一定落库。"""
|
||||
@@ -1522,10 +1487,8 @@ def _mark_failed_and_notify(job_id: str, session, repo, job, err_msg: str, stage
|
||||
event_type="viral_video:failed",
|
||||
)
|
||||
|
||||
|
||||
# ── v1.5/v1.6 三步分步流水线 Celery 任务 ─────────────────────────────────
|
||||
|
||||
|
||||
@shared_task(
|
||||
bind=True,
|
||||
max_retries=1,
|
||||
@@ -1593,7 +1556,6 @@ def run_viral_video_analyze(self: Task, job_id: str) -> dict:
|
||||
if session:
|
||||
session.close()
|
||||
|
||||
|
||||
@shared_task(
|
||||
bind=True,
|
||||
max_retries=1,
|
||||
@@ -1696,7 +1658,6 @@ def run_viral_video_generate_copy(self: Task, job_id: str) -> dict:
|
||||
if session:
|
||||
session.close()
|
||||
|
||||
|
||||
def _quick_compliance_blacklist_check(copy_result: dict) -> None:
|
||||
"""阶段2快速黑名单检查:不调用 LLM,只扫描高风险关键词;命中则在 voiceover 中就地替换。
|
||||
|
||||
@@ -1738,7 +1699,6 @@ def _quick_compliance_blacklist_check(copy_result: dict) -> None:
|
||||
for bk, bv in BLACKLIST.items():
|
||||
copy_result[k] = copy_result[k].replace(bk, bv)
|
||||
|
||||
|
||||
def _try_refund_viral_video(job: ViralVideoJob) -> None:
|
||||
"""爆款视频生成失败:若已预扣积分则全额退款。"""
|
||||
try:
|
||||
@@ -1768,7 +1728,6 @@ def _try_refund_viral_video(job: ViralVideoJob) -> None:
|
||||
except Exception:
|
||||
logger.exception("[爆款视频] 失败退款异常 job_id=%s", job.id)
|
||||
|
||||
|
||||
def _settle_viral_video(job: ViralVideoJob, usage: dict | None) -> None:
|
||||
"""爆款视频生成成功:按实际 usage 结算,多退少补,写 credits_cost。"""
|
||||
try:
|
||||
@@ -1839,7 +1798,6 @@ def _settle_viral_video(job: ViralVideoJob, usage: dict | None) -> None:
|
||||
job.credits_cost = float(getattr(job, "credits_prepaid", 0) or 0)
|
||||
job.credits_prepaid = 0.0
|
||||
|
||||
|
||||
def _run_render_pipeline(job_id: str, session, repo, job) -> dict:
|
||||
"""v1.6.1 阶段3:出片前合规审核(LLM 深度)→ TTS → Seedance → Upload → Completed。
|
||||
|
||||
@@ -1863,9 +1821,14 @@ def _run_render_pipeline(job_id: str, session, repo, job) -> dict:
|
||||
review_result = _step_review(job, copy_result)
|
||||
if not review_result.get("passed", True):
|
||||
_emit_progress(job_id, ViralVideoStage.REVIEW, 67.0, "审核未通过,正在自动重写...")
|
||||
intent = job.intent_result or _step_intent_parsing(job, image_analysis)
|
||||
copy_result = _step_script_generation(job, intent, image_analysis)
|
||||
_step_review(job, copy_result) # 二次审核,不通过也继续出片(避免反复循环)
|
||||
# #2040: Reviewer 已在 _step_review 内完成 1 次自动重写
|
||||
rewritten = review_result.get("rewritten_copy")
|
||||
if isinstance(rewritten, dict) and rewritten:
|
||||
copy_result = rewritten
|
||||
else:
|
||||
intent = job.intent_result or _step_intent_parsing(job, image_analysis)
|
||||
copy_result = _step_script_generation(job, intent, image_analysis)
|
||||
_step_review(job, copy_result)
|
||||
job.copy_result = copy_result
|
||||
job.generated_copy_text = copy_result.get("voiceover_script", "") or ""
|
||||
_save_job(repo, job, session)
|
||||
@@ -1921,7 +1884,6 @@ def _run_render_pipeline(job_id: str, session, repo, job) -> dict:
|
||||
logger.info("[爆款视频] 任务完成: job_id=%s video_url=%s", job_id, video_url)
|
||||
return {"ok": True, "job_id": job_id, "video_url": video_url}
|
||||
|
||||
|
||||
@shared_task(bind=True, max_retries=2, name="worker.run_viral_video_render")
|
||||
def run_viral_video_render(self: Task, job_id: str) -> dict:
|
||||
"""v1.6 阶段3:TTS + 单次 Seedance 生成 + 上传。"""
|
||||
|
||||
@@ -132,7 +132,10 @@ _FUSION_SYSTEM = """你负责为短视频生成营销文案。请按思维链分
|
||||
<hook> 开头3秒钩子,5到15字。
|
||||
<body_points> 每个要点用一个 <point> 标签,属性 elaboration 是展开说明、image_index 是对应第几张图(从0开始),标签内容写要点。
|
||||
<cta> 口语化的行动号召。
|
||||
<script_segments> 每段配音用一个 <segment> 标签,属性 duration_sec 是秒数、image_index 是对应图片,标签内容写配音文案。
|
||||
<script_segments> 每段配音用一个 <segment> 标签,属性 duration_sec 是秒数、image_index 是对应图片,标签内容写配音文案(纯口播文本,不加旁白标注、不加镜头标注、不加"主播:"之类前缀)。
|
||||
<voiceover_script> 把所有 segment 的配音文案按顺序自然拼接成一段完整的纯口播文本(无标记、无括号、无前缀),长度要适配 {duration} 秒,约 {approx_chars} 字。
|
||||
<overview_theme> 视频主题(一句话概括)。
|
||||
<scene_and_lighting> 整体场景描述+光线设定(100-200字,要具体:在哪拍、什么光线、什么色调、什么氛围)。
|
||||
<word_count> 配音总字数,只写数字。
|
||||
<estimated_duration> 预计时长秒数,只写数字。
|
||||
|
||||
@@ -161,25 +164,34 @@ _FUSION_EXAMPLE = """<title>厨房重油污,别再用洗洁精硬擦了</title
|
||||
<segment duration_sec="6" image_index="0">后来换了这个大公鸡头油污净,喷上等几分钟,一擦就干净</segment>
|
||||
<segment duration_sec="4" image_index="0">39块钱625ml,厨房重油污的可以试一瓶</segment>
|
||||
</script_segments>
|
||||
<voiceover_script>这油污我真的忍很久了,用洗洁精擦半天都没用。后来换了这个大公鸡头油污净,喷上等几分钟,一擦就干净。39块钱625ml,厨房重油污的可以试一瓶。</voiceover_script>
|
||||
<overview_theme>厨房油污清洁好物分享</overview_theme>
|
||||
<scene_and_lighting>简洁明亮的厨房台面场景,自然光从窗户洒入,色调温暖柔和,突出产品白色瓶身与去油污对比效果。</scene_and_lighting>
|
||||
<word_count>58</word_count>
|
||||
<estimated_duration>13</estimated_duration>"""
|
||||
|
||||
# ── 模板4:编导级分镜(LLM)────────────────────────────────────────────
|
||||
_STORYBOARD_SYSTEM = f"""你是短视频编导,负责把文案拆成可拍摄的分镜。
|
||||
_STORYBOARD_SYSTEM = f"""你是短视频编导,负责把文案拆成可拍摄的分镜,为 Seedance 2.5 视频模型写编导分镜脚本。脚本将整体作为 prompt 一次性传给视频模型,必须让模型在连贯镜头流中清楚每段时间拍什么、画面如何、人物说什么。
|
||||
|
||||
工作方式:
|
||||
1. 按文案的 script_segments 顺序分配镜头。
|
||||
2. 每个镜头确定画面、运镜、时长、配音和字幕。
|
||||
2. 每个镜头确定景别/角度/运镜、画面场景与对白、人物动作细节、音效/BGM、转场。
|
||||
3. 检查所有镜头时长加起来接近目标时长,误差不超过2秒。
|
||||
4. image_index 必须在已上传图片范围内,第一张主图必须用在第一个镜头。
|
||||
|
||||
{GLOBAL_CONSTRAINTS}
|
||||
|
||||
请严格按下面的标签格式输出,不要解释,不要用代码块:
|
||||
<clips> 下面每个镜头用一个 <clip> 标签,属性 image_index 是图片序号(从0开始)、transition 取 fade、cut、zoom_in、slide_left、dissolve、wipe 之一、zoom 取 in、out 或 null、duration_sec 是该镜头秒数、bgm_note 是该段BGM情绪。每个 <clip> 里面包含:
|
||||
<voice_text> 该镜头配音文本;
|
||||
<subtitle_text> 字幕文本,可与配音一致或更精简;
|
||||
<ken_burns> 用一个空标签,属性 start、end 写“x,y”坐标、ease 写缓动方式;不需要运镜时坐标相同。"""
|
||||
<clips> 下面每个镜头用一个 <clip> 标签,属性 image_index 是图片序号(从0开始)、transition 取 fade/cut/zoom_in/slide_left/dissolve/wipe 之一、zoom 取 in/out/null、duration_sec 是该镜头秒数、bgm_note 是该段BGM情绪。每个 <clip> 里面包含:
|
||||
<voice_text> 该镜头配音文本(纯口播文本,不加旁白标注);
|
||||
<subtitle_text> 字幕文本,可与配音一致或更精简;
|
||||
<shot_type_angle_movement> 景别+角度+运镜(例:近景俯拍45度,缓慢推镜;中景平视,固定镜头;特写平视,快速拉镜);
|
||||
<scene_and_dialogue> 画面场景描述 + 人物口播台词(对白要自然口语化,像朋友聊天,不要硬广推销腔);
|
||||
<action_details> 人物动作、表情、物品操作细节(手怎么动、表情变化、产品怎么展示);
|
||||
<audio_bgm> 环境音+BGM提示(例:轻快流行BGM,环境嘈杂咖啡店背景音);
|
||||
<transition> 硬切/淡入淡出/叠化(最后一镜写『结束』即可);
|
||||
<reference_image_index> 参考图片索引(0-based,对应第几张产品图,无则空);
|
||||
<ken_burns> 用一个空标签,属性 start、end 写"x,y"坐标、ease 写缓动方式;不需要运镜时坐标相同。"""
|
||||
|
||||
_STORYBOARD_USER = """目标时长:{duration}秒
|
||||
上传图片数量:{image_count}张(第1张是主图/封面)
|
||||
@@ -194,17 +206,35 @@ _STORYBOARD_EXAMPLE = """<clips>
|
||||
<clip image_index="0" transition="cut" zoom="null" duration_sec="3" bgm_note="日常、轻微烦躁">
|
||||
<voice_text>这油污我真的忍很久了</voice_text>
|
||||
<subtitle_text>这油污忍很久了</subtitle_text>
|
||||
<shot_type_angle_movement>近景俯拍45度,缓慢推镜</shot_type_angle_movement>
|
||||
<scene_and_dialogue>厨房台面,主妇皱眉看着灶台油污。对白:这油污我真的忍很久了</scene_and_dialogue>
|
||||
<action_details>右手拿着脏抹布,无奈摇头</action_details>
|
||||
<audio_bgm>轻快日常BGM,带一点烦躁感</audio_bgm>
|
||||
<transition>硬切</transition>
|
||||
<reference_image_index>0</reference_image_index>
|
||||
<ken_burns start="0,0" end="0,0" ease="linear"/>
|
||||
</clip>
|
||||
<clip image_index="0" transition="zoom_in" zoom="in" duration_sec="6" bgm_note="轻快、出现转机">
|
||||
<voice_text>后来换了大公鸡头油污净,喷上等几分钟,一擦就干净</voice_text>
|
||||
<subtitle_text>喷上等几分钟,一擦就干净</subtitle_text>
|
||||
<shot_type_angle_movement>特写平视,固定镜头</shot_type_angle_movement>
|
||||
<scene_and_dialogue>手部特写,喷油污净在油污处。对白:后来换了这个大公鸡头油污净,喷上等几分钟,一擦就干净</scene_and_dialogue>
|
||||
<action_details>左手拿产品瓶身,右手按压喷头,等待片刻后用抹布轻擦</action_details>
|
||||
<audio_bgm>轻快转折BGM,带清爽感</audio_bgm>
|
||||
<transition>淡入淡出</transition>
|
||||
<reference_image_index>0</reference_image_index>
|
||||
<ken_burns start="20,20" end="80,80" ease="ease-in-out"/>
|
||||
</clip>
|
||||
<clip image_index="0" transition="fade" zoom="null" duration_sec="4" bgm_note="温暖、推荐">
|
||||
<voice_text>39块钱625ml,厨房重油污的可以试一瓶</voice_text>
|
||||
<subtitle_text>39元625ml,可以试一瓶</subtitle_text>
|
||||
<ken_burns start="0,0" end="0,0" ease="linear"/>
|
||||
<shot_type_angle_movement>中景平视,缓慢拉镜</shot_type_angle_movement>
|
||||
<scene_and_dialogue>产品正面展示,明亮背景。对白:39块钱625ml,厨房重油污的可以试一瓶</scene_and_dialogue>
|
||||
<action_details>产品置于画面中央,轻微转动展示瓶身</action_details>
|
||||
<audio_bgm>温暖收尾BGM</audio_bgm>
|
||||
<transition>结束</transition>
|
||||
<reference_image_index>0</reference_image_index>
|
||||
<ken_burns start="50,50" end="20,20" ease="ease-in-out"/>
|
||||
</clip>
|
||||
</clips>"""
|
||||
|
||||
|
||||
Reference in New Issue
Block a user