diff --git a/alembic/versions/088_viral_video_copy_result.py b/alembic/versions/088_viral_video_copy_result.py new file mode 100644 index 000000000..203b55c17 --- /dev/null +++ b/alembic/versions/088_viral_video_copy_result.py @@ -0,0 +1,51 @@ +"""viral video add copy_result + voice/video columns + +Revision ID: 088_viral_video_copy_result +Revises: 087_viral_video_image_analysis +Create Date: 2026-10-01 + +v1.6 爆款视频字段补齐: +- copy_result JSON: 编导分镜脚本完整结构(overview/scene_and_lighting/shots/hard_constraints/negative_prompts/voiceover_script) +- voice_id/voice_source: TTS 音色参数 +- video_ratio/video_model: Seedance 视频比例/模型 +注意:线上启动也有幂等 ADD COLUMN 补列逻辑 (_ensure_viral_video_columns),本 migration 提供标准 Alembic 路径, +两套机制互不冲突(IF NOT EXISTS 等价行为)。 +""" + +import sqlalchemy as sa + +from alembic import op + +revision = "088_viral_video_copy_result" +down_revision = "087_viral_video_image_analysis" +branch_labels = None +depends_on = None + + +def upgrade() -> None: + # 幂等添加列(通过单独执行 + 异常忽略兼容已由 backfill 补上的环境) + cols = [ + ("voice_id", "VARCHAR(200) NOT NULL DEFAULT ''"), + ("voice_source", "VARCHAR(20) NOT NULL DEFAULT ''"), + ("video_ratio", "VARCHAR(10) NOT NULL DEFAULT '9:16'"), + ("video_model", "VARCHAR(100) NOT NULL DEFAULT ''"), + ("copy_result", "JSON"), + ] + conn = op.get_bind() + for name, ddl in cols: + try: + conn.execute(sa.text(f"ALTER TABLE viral_video_jobs ADD COLUMN IF NOT EXISTS {name} {ddl}")) + except Exception: + # 不支持 IF NOT EXISTS 的库(如老版本 SQLite)直接尝试 ADD COLUMN,失败则忽略 + try: + conn.execute(sa.text(f"ALTER TABLE viral_video_jobs ADD COLUMN {name} {ddl}")) + except Exception: + pass + + +def downgrade() -> None: + for name in ("copy_result", "video_model", "video_ratio", "voice_source", "voice_id"): + try: + op.drop_column("viral_video_jobs", name) + except Exception: + pass diff --git a/apps/api/app/api/routes/viral_video.py b/apps/api/app/api/routes/viral_video.py index 3042d1246..0190ee050 100644 --- a/apps/api/app/api/routes/viral_video.py +++ b/apps/api/app/api/routes/viral_video.py @@ -1,6 +1,6 @@ """爆款视频 API 路由。 -v1.5 三步分步流水线端点(前端新交互): +v1.6 三步分步流水线端点(单次 Seedance 出片版): POST /api/v1/viral-video/analyze-images 阶段1:创建任务 + 仅做图片/视频分析,暂停在 image_analyzed POST /api/v1/viral-video/{job_id}/generate-copy 阶段2:用户填完参数后跑意图+文案+分镜+审核,暂停在 copy_generated POST /api/v1/viral-video/{job_id}/confirm-copy 阶段3:用户确认/编辑文案后跑渲染,直到完成 @@ -10,7 +10,7 @@ v1.5 三步分步流水线端点(前端新交互): POST /api/v1/viral-video/{job_id}/confirm-intent 旧的意图确认后继续渲染 通用: - GET /api/v1/viral-video/{job_id} 查询任务状态(含 image_analysis/storyboard/generated_copy_text) + GET /api/v1/viral-video/{job_id} 查询任务状态(含 image_analysis/copy_result 编导脚本) GET /api/v1/viral-video/history 历史记录 POST /api/v1/viral-video/{job_id}/retry 重试失败任务 POST /api/v1/viral-video/{job_id}/analyze-style 触发风格分析 @@ -56,27 +56,55 @@ router = APIRouter() def _build_copy_result(job) -> dict | None: - """将后端原始字段拼装为前端期望的 CopyResult 结构(final_copy/suggested_copy/title/scenes)。""" + """v1.6: 返回编导分镜脚本 CopyResult 结构(给前端/Seedance 使用)。 + + - 若 job.copy_result 已持久化(v1.6 worker 生成),直接返回(补 final_copy 兜底)。 + - 否则从老字段(generated_copy_text=口播, storyboard=分镜列表, intent_result)拼装兼容结构。 + """ + cr = getattr(job, "copy_result", None) + if isinstance(cr, dict) and cr: + out = dict(cr) + # 向后兼容字段 + voiceover = out.get("voiceover_script", "") or "" + out.setdefault("final_copy", voiceover) + out.setdefault("suggested_copy", voiceover) + out.setdefault("title", "") + return out + # 兼容 v1.5 老数据:storyboard 是老格式 [{order,type,description,text,duration,...}] copy_text = getattr(job, "generated_copy_text", "") or "" sb = getattr(job, "storyboard", None) or [] intent = getattr(job, "intent_result", None) or {} if not copy_text and not sb: return None - scenes = [] + title = "" + if isinstance(intent, dict): + title = intent.get("suggested_title") or intent.get("intent", "") or "" + shots = [] for seg in sb: if isinstance(seg, dict): - scenes.append( + shots.append( { - "shot": seg.get("description", ""), - "narration": seg.get("text", ""), - "duration": seg.get("duration"), + "time_range": "", + "shot_type_angle_movement": seg.get("ken_burns", ""), + "scene_and_dialogue": (seg.get("text") or "") + + (" " + seg.get("description", "") if seg.get("description") else ""), + "action_details": "", + "audio_bgm": "", + "transition": seg.get("transition", "硬切"), + "reference_image_index": None, } ) + ratio = getattr(job, "video_ratio", None) or "9:16" return { - "title": (intent.get("suggested_title") if isinstance(intent, dict) else None) or "", + "overview": {"theme": title, "total_duration": getattr(job, "duration", 15), "aspect_ratio": ratio}, + "scene_and_lighting": "", + "shots": shots, + "hard_constraints": ["无字幕", "无水印", "人物一致性"], + "negative_prompts": ["字幕", "水印", "错误文字", "五官崩坏"], + "voiceover_script": copy_text, "final_copy": copy_text, "suggested_copy": copy_text, - "scenes": scenes, + "title": title, } @@ -91,7 +119,7 @@ def _to_response(job) -> ViralVideoJobResponse: viral_structure=job.viral_structure, marketing_purpose=job.marketing_purpose, bgm_preference=job.bgm_preference, - duration=job.duration, + duration=job.duration or 15, user_copy_text=job.user_copy_text, fusion_level=job.fusion_level, reference_audio_path=job.reference_audio_path, @@ -152,7 +180,7 @@ def create_viral_video( viral_structure=request.viral_structure, marketing_purpose=request.marketing_purpose, bgm_preference=request.bgm_preference, - duration=request.duration, + duration=request.duration or 15, user_copy_text=request.user_copy_text, fusion_level=request.fusion_level, reference_audio_path=request.reference_audio_path, @@ -163,6 +191,7 @@ def create_viral_video( voice_source=getattr(request, "voice_source", "") or "", video_ratio=getattr(request, "video_ratio", "9:16") or "9:16", video_model=getattr(request, "video_model", "") or "", + copy_result=None, ) # 持久化 @@ -204,6 +233,7 @@ def analyze_images( voice_source=request.voice_source or "", video_ratio=request.video_ratio or "9:16", video_model=request.video_model or "", + duration=request.duration or 15, ) repo.save(job) @@ -225,10 +255,11 @@ def generate_copy( authenticated_user: AuthenticatedUser = Depends(get_current_user), session: Session = Depends(get_db_session), ) -> ViralVideoJobResponse: - """v1.5 阶段2:用户填完营销参数后,跑 意图解析 → 文案融合 → 分镜 → 合规审核。 + """v1.6 阶段2:用户填完营销参数后,跑 意图解析 → 编导分镜脚本生成 → 合规审核。 - 跑完后状态=copy_generated,响应包含 generated_copy_text + storyboard, - 前端展示文案供用户编辑;确认/编辑后调 /{id}/confirm-copy 进入阶段3。 + 跑完后状态=copy_generated,响应 copy_result(含 overview/scene_and_lighting/shots/ + hard_constraints/negative_prompts/voiceover_script),前端展示脚本与口播供用户编辑; + 确认/编辑后调 /{id}/confirm-copy 进入阶段3(TTS + 单次 Seedance 出片)。 """ repo = _get_job_repo(session) job = repo.get(job_id) @@ -252,7 +283,7 @@ def generate_copy( job.marketing_purpose = request.marketing_purpose or job.marketing_purpose job.bgm_preference = request.bgm_preference or job.bgm_preference if request.duration: - job.duration = request.duration + job.duration = max(5, min(30, int(request.duration))) job.user_copy_text = request.user_copy_text if request.user_copy_text else job.user_copy_text job.fusion_level = request.fusion_level or job.fusion_level job.reference_audio_path = request.reference_audio_path or job.reference_audio_path @@ -287,7 +318,7 @@ def confirm_copy( authenticated_user: AuthenticatedUser = Depends(get_current_user), session: Session = Depends(get_db_session), ) -> ViralVideoJobResponse: - """v1.5 阶段3:用户确认/编辑文案后开始 TTS+BGM(skip)+渲染+上传。""" + """v1.6 阶段3:用户确认/编辑口播后开始 TTS + 单次 Seedance 生成 + 上传。""" repo = _get_job_repo(session) job = repo.get(job_id) if job is None: diff --git a/apps/api/app/schemas/viral_video.py b/apps/api/app/schemas/viral_video.py index a0a0516d6..f14768788 100755 --- a/apps/api/app/schemas/viral_video.py +++ b/apps/api/app/schemas/viral_video.py @@ -1,4 +1,4 @@ -"""爆款视频 API schemas。""" +"""爆款视频 API schemas (v1.6 单次 Seedance 出片版)。""" from __future__ import annotations @@ -14,46 +14,84 @@ VALID_STAGES = ( "image_analysis", "video_analysis", "intent_parsing", - "copy_fusion", - "storyboard", + "script_generation", "review", "tts", - "bgm_select", "rendering", - "musetalk", "uploading", ) +VALID_VIDEO_RATIOS = ("9:16", "16:9", "1:1", "4:3", "3:4", "21:9") +VALID_DURATIONS = (5, 10, 15, 20, 25, 30) + + +# -- 编导脚本结构(v1.6) -- + + +class ShotScript(BaseModel): + """逐镜头分镜。""" + + time_range: str = Field(default="", description="时间区间,如 0-3秒") + shot_type_angle_movement: str = Field(default="", description="景别/角度/运镜,如『近景俯拍45度,缓慢推镜』") + scene_and_dialogue: str = Field(default="", description="场景描述+口播台词") + action_details: str = Field(default="", description="人物动作、表情、物品操作细节") + audio_bgm: str = Field(default="", description="环境音+BGM提示") + transition: str = Field(default="硬切", description="转场方式:硬切/淡入淡出/叠化") + reference_image_index: int | None = Field( + default=None, description="参考图片索引(0-based,对应上传的第几张产品图)" + ) + + +class CopyResultOverview(BaseModel): + theme: str = "" + total_duration: int = 15 + aspect_ratio: str = "9:16" + + +class CopyResult(BaseModel): + """v1.6 编导分镜脚本结构(给前端 + Seedance 用)。""" + + overview: CopyResultOverview = Field(default_factory=CopyResultOverview) + scene_and_lighting: str = "" + shots: list[ShotScript] = Field(default_factory=list) + hard_constraints: list[str] = Field(default_factory=list) + negative_prompts: list[str] = Field(default_factory=list) + voiceover_script: str = Field( + default="", description="纯口播对白,从各镜 scene_and_dialogue 的对白部分拼接,供 TTS 使用" + ) + # 向后兼容:final_copy = voiceover_script + final_copy: str = "" + suggested_copy: str = "" + title: str = "" # -- Request Schemas -- class CreateViralVideoRequest(BaseModel): - """创建爆款视频任务请求(旧接口:一键跑完前半段到 WAIT_USER_CONFIRM,保留兼容)。""" + """旧接口:一键创建(保留兼容)。""" - images: list[str] = Field(..., min_length=1, max_length=20, description="产品图片 URL 列表") - industry: str = Field(default="", description="行业") - target_customer: str = Field(default="", description="目标客户描述") - persona_id: str = Field(default="", description="人设 ID") - viral_structure: str = Field(default="", description="爆款结构类型") - marketing_purpose: str = Field(default="", description="营销目的") - bgm_preference: str = Field(default="", description="BGM 偏好") - duration: int = Field(default=30, ge=5, le=180, description="视频时长(秒)") - user_copy_text: str = Field(default="", description="用户原始文案(我说你写)") - fusion_level: str = Field(default="ai_polish", description="文案融合级别: ai_full/ai_polish/user_primary") - reference_audio_path: str = Field(default="", description="参考音频路径") - reference_video_url: str = Field(default="", description="参考爆款视频 URL") - style_strength: str = Field(default="medium", description="风格强度: light/medium/strict") - style_template_id: str = Field(default="", description="风格模板 ID") - # v1.5.1 音色/视频参数(旧接口兼容:前端兜底走 /generate 时也能传) - voice_id: str = Field(default="", description="TTS 音色 ID") - voice_source: str = Field(default="", description="音色来源") - video_ratio: str = Field(default="9:16", description="Seedance 视频比例") - video_model: str = Field(default="", description="Seedance 模型 ID") + images: list[str] = Field(..., min_length=1, max_length=20) + industry: str = "" + target_customer: str = "" + persona_id: str = "" + viral_structure: str = "" + marketing_purpose: str = "" + bgm_preference: str = "" + duration: int = Field(default=15, ge=5, le=30, description="视频时长(秒),5-30") + user_copy_text: str = "" + fusion_level: str = "ai_polish" + reference_audio_path: str = "" + reference_video_url: str = "" + style_strength: str = "medium" + style_template_id: str = "" + voice_id: str = "" + voice_source: str = "" + video_ratio: str = "9:16" + video_model: str = "" @field_validator("fusion_level") @classmethod - def _validate_fusion_level(cls, v: str) -> str: + def _v_fl(cls, v: str) -> str: if v == "full_ai": return "ai_full" if v not in VALID_FUSION_LEVELS: @@ -62,48 +100,47 @@ class CreateViralVideoRequest(BaseModel): @field_validator("style_strength") @classmethod - def _validate_style_strength(cls, v: str) -> str: + def _v_ss(cls, v: str) -> str: if v not in VALID_STYLE_STRENGTHS: raise ValueError(f"style_strength must be one of {VALID_STYLE_STRENGTHS}") return v class AnalyzeImagesRequest(BaseModel): - """v1.5 阶段1:创建任务并仅做图片/视频分析。images 必填,其他参数可选(阶段2再传/覆盖)。""" + """v1.5+ 阶段1:创建任务 + 图片/视频分析。""" - images: list[str] = Field(..., min_length=1, max_length=20) - reference_video_url: str = Field(default="", description="参考爆款视频 URL(可选,有则同步做风格分析)") - style_template_id: str = Field(default="", description="风格模板 ID(可选)") - style_strength: str = Field(default="medium") - # v1.5.1 音色/视频参数(STEP1 已选定的音色可先传,阶段2可覆盖) - voice_id: str = Field(default="", description="TTS 音色 ID;空则用 persona_id 兜底") - voice_source: str = Field(default="", description="音色来源:preset/library/clone/upload") - video_ratio: str = Field(default="9:16", description="Seedance 视频比例:9:16/16:9/1:1 等") - video_model: str = Field(default="", description="Seedance 模型 ID;空则使用服务端默认") + images: list[str] = Field(..., min_length=1, max_length=30) + reference_video_url: str = "" + style_template_id: str = "" + style_strength: str = "medium" + voice_id: str = "" + voice_source: str = "" + video_ratio: str = "9:16" + video_model: str = "" + duration: int = Field(default=15, ge=5, le=30) class GenerateCopyRequest(BaseModel): - """v1.5 阶段2:用户填完参数后跑意图+文案+分镜+审核,暂停在 COPY_GENERATED。""" + """v1.5+ 阶段2:填完营销参数,生成编导脚本。""" - industry: str = Field(default="") - target_customer: str = Field(default="") - persona_id: str = Field(default="", description="人设 ID;voice_id 为空时也作为音色 ID 兜底") - viral_structure: str = Field(default="") - marketing_purpose: str = Field(default="") - bgm_preference: str = Field(default="") - duration: int = Field(default=30, ge=5, le=180) - user_copy_text: str = Field(default="") - fusion_level: str = Field(default="ai_polish") - reference_audio_path: str = Field(default="") - reference_video_url: str = Field(default="") - style_strength: str = Field(default="medium") - style_template_id: str = Field(default="") - style_guide: dict | None = Field(default=None) - # v1.5.1 音色/视频参数 - voice_id: str = Field(default="", description="TTS 音色 ID(优先级高于 persona_id)") - voice_source: str = Field(default="", description="音色来源:preset/library/clone/upload") - video_ratio: str = Field(default="9:16", description="Seedance 视频比例") - video_model: str = Field(default="", description="Seedance 模型 ID;空则使用服务端默认") + industry: str = "" + target_customer: str = "" + persona_id: str = "" + viral_structure: str = "" + marketing_purpose: str = "" + bgm_preference: str = "" + duration: int = Field(default=15, ge=5, le=30) + user_copy_text: str = "" + fusion_level: str = "ai_polish" + reference_audio_path: str = "" + reference_video_url: str = "" + style_strength: str = "medium" + style_template_id: str = "" + style_guide: dict | None = None + voice_id: str = "" + voice_source: str = "" + video_ratio: str = "9:16" + video_model: str = "" @field_validator("fusion_level") @classmethod @@ -123,30 +160,28 @@ class GenerateCopyRequest(BaseModel): class ConfirmCopyRequest(BaseModel): - """v1.5 阶段3:用户确认/编辑文案后开始渲染。""" + """v1.5+ 阶段3:用户确认/编辑口播后开始渲染(TTS+单次Seedance)。""" - edited_copy: str = Field(default="", description="用户编辑后的最终文案;为空则使用 AI 生成文案") + edited_copy: str = Field(default="", description="用户编辑后的口播文案;为空则用 AI 生成的 voiceover_script") class ConfirmIntentRequest(BaseModel): - """确认意图请求(旧 confirm-intent,兼容)。""" + """旧 confirm-intent(兼容)。""" - confirmed_copy: str = Field(default="", description="用户确认/修改后的文案") - adjustments: str = Field(default="", description="用户对 AI 文案的调整意见") + confirmed_copy: str = "" + adjustments: str = "" class AnalyzeStyleRequest(BaseModel): - """触发参考视频风格分析请求。""" - reference_video_url: str = Field(..., description="参考视频 URL") - style_template_id: str = Field(default="", description="风格模板 ID(可选覆盖)") + style_template_id: str = "" # -- Response Schemas -- class ViralVideoJobResponse(BaseModel): - """爆款视频任务响应。v1.5 新增 image_analysis/storyboard/generated_copy_text 字段。""" + """爆款视频任务响应(v1.6 包含 copy_result 编导脚本结构)。""" id: str user_id: str @@ -157,7 +192,7 @@ class ViralVideoJobResponse(BaseModel): viral_structure: str = "" marketing_purpose: str = "" bgm_preference: str = "" - duration: int = 30 + duration: int = 15 user_copy_text: str = "" fusion_level: str = "ai_polish" reference_audio_path: str = "" @@ -166,14 +201,13 @@ class ViralVideoJobResponse(BaseModel): style_guide: dict | None = None style_template_id: str = "" status: str - # v1.4 VLM 结果 image_analysis: dict | None = None - # v1.5 三步分步产物(原始字段,保留给后端/老调用方) + # v1.6 编导脚本(推荐前端使用) + copy_result: dict | None = None + # v1.5 兼容字段 storyboard: list | None = None generated_copy_text: str = "" - # v1.5 前端 CopyResult 结构(final_copy/suggested_copy/title/scenes) - copy_result: dict | None = None - # v1.5.1 音色/视频参数 + # 音色/视频参数 voice_id: str = "" voice_source: str = "" video_ratio: str = "9:16" diff --git a/apps/web/src/api/viral-video/types.ts b/apps/web/src/api/viral-video/types.ts index 80307b8da..f8d8ab36b 100644 --- a/apps/web/src/api/viral-video/types.ts +++ b/apps/web/src/api/viral-video/types.ts @@ -12,6 +12,14 @@ export const STYLE_STRENGTHS: { value: StyleStrength; label: string }[] = [ { value: "strict", label: "像素级复刻" }, ] +/** v1.6 前端时长下拉选项(5/10/15/20/25/30秒) */ +export const VALID_DURATIONS = [5, 10, 15, 20, 25, 30] as const +export type VideoDuration = (typeof VALID_DURATIONS)[number] + +/** v1.6 支持的画幅比例 */ +export const VALID_RATIOS = ["9:16", "16:9", "1:1"] as const +export type VideoRatio = (typeof VALID_RATIOS)[number] + export type ViralVideoStatus = | "pending" | "running" @@ -23,39 +31,25 @@ export type ViralVideoStatus = | "cancelled" /** - * 后端流水线阶段字符串。前端不展示逐阶段进度列表,仅保留类型 - * 用于轮询时判断当前在哪个大阶段(分析中 vs 文案生成 vs 视频生成)。 + * v1.6 后端流水线阶段。单次 Seedance 出片版: + * image_analysis → video_analysis(可选) → intent_parsing → script_generation → review → tts → rendering → uploading */ export type ViralVideoStage = | "image_analysis" | "video_analysis" | "intent_parsing" - | "copy_fusion" - | "storyboard" + | "script_generation" | "review" | "tts" - | "bgm_select" | "rendering" - | "musetalk" | "uploading" /** 图片+视频分析阶段:属于「分析图片」按钮的范围 */ const IMAGE_ANALYSIS_STAGES = new Set(["image_analysis", "video_analysis"]) -/** 文案相关阶段:属于「生成文案」按钮的范围 */ -const COPY_STAGES = new Set([ - "intent_parsing", - "copy_fusion", - "storyboard", - "review", -]) -/** 视频相关阶段:属于「开始生成视频」按钮的范围 */ -const VIDEO_STAGES = new Set([ - "tts", - "bgm_select", - "rendering", - "musetalk", - "uploading", -]) +/** 编导脚本阶段:属于「生成文案」按钮的范围 */ +const COPY_STAGES = new Set(["intent_parsing", "script_generation", "review"]) +/** 视频生成阶段:属于「开始生成视频」按钮的范围(v1.6: TTS+单次Seedance+上传) */ +const VIDEO_STAGES = new Set(["tts", "rendering", "uploading"]) export function isImageAnalysisStage(stage: ViralVideoStage | undefined): boolean { return !!stage && IMAGE_ANALYSIS_STAGES.has(stage) @@ -66,7 +60,7 @@ export function isCopyStage(stage: ViralVideoStage | undefined): boolean { export function isVideoStage(stage: ViralVideoStage | undefined): boolean { return !!stage && VIDEO_STAGES.has(stage) } -/** 兼容旧调用:旧的 isAnalysisStage 视为「图片分析+文案」的所有前置阶段 */ +/** 兼容旧调用:分析图片+生成文案 的所有前置阶段 */ export function isAnalysisStage(stage: ViralVideoStage | undefined): boolean { return isImageAnalysisStage(stage) || isCopyStage(stage) } @@ -74,12 +68,20 @@ export function isAnalysisStage(stage: ViralVideoStage | undefined): boolean { /** 单张图片 VLM 识别出的商品信息 */ export interface ImageProductAnalysis { name?: string - spec?: string + category?: string brand?: string + colors?: string[] + material_or_texture?: string + key_features?: string[] + visual_style?: string + scene?: string + target_audience_hint?: string + text_on_image?: string + /** 旧字段兼容 */ + spec?: string features?: string[] | string label_text?: string - selling_points?: string[] - scene?: string + selling_points?: string image_index?: number } @@ -87,10 +89,50 @@ export interface ImageAnalysisResult { products?: ImageProductAnalysis[] } +/** v1.6 编导分镜脚本 - 单镜头 */ +export interface ShotScript { + /** 时间区间,如 "0-3秒" */ + time_range?: string + /** 景别/角度/运镜,如 "近景俯拍45度,缓慢推镜" */ + shot_type_angle_movement?: string + /** 场景描述+对白 */ + scene_and_dialogue?: string + /** 人物动作/表情/物品操作细节 */ + action_details?: string + /** 环境音+BGM提示 */ + audio_bgm?: string + /** 转场方式(硬切/淡入淡出/叠化/结束) */ + transition?: string + /** 参考图片索引(0-based,对应上传产品图数组) */ + reference_image_index?: number | null +} + +/** v1.6 编导分镜脚本 - 总览 */ +export interface CopyResultOverview { + theme?: string + total_duration?: number + aspect_ratio?: string +} + +/** v1.6 编导分镜脚本(核心输出结构,给 Seedance 做 prompt,给 TTS 取 voiceover_script) */ export interface CopyResult { + overview?: CopyResultOverview + /** 整体场景+光线描述 */ + scene_and_lighting?: string + /** 逐镜头时间轴 */ + shots?: ShotScript[] + /** 硬性约束(禁止字幕/水印/变形等) */ + hard_constraints?: string[] + /** 负面提示词 */ + negative_prompts?: string[] + /** 完整口播稿(纯文本,用于 TTS 合成) */ + voiceover_script?: string + /** 向后兼容:= voiceover_script */ final_copy?: string + /** 向后兼容:= voiceover_script */ suggested_copy?: string title?: string + /** v1.5 旧字段兼容(老数据降级时可能出现) */ scenes?: Array<{ shot: string; narration: string; duration?: number }> } @@ -104,13 +146,18 @@ export interface StyleTemplate { } export interface IntentResult { - product: string - selling_points: string[] - target_audience: string - tone: string - structure: string - duration: number + intent?: string + key_messages?: string[] + tone?: string + target_emotion?: string + call_to_action?: string suggested_title?: string + /** v1.5 旧字段兼容 */ + product?: string + selling_points?: string[] + target_audience?: string + structure?: string + duration?: number suggested_copy?: string } @@ -123,7 +170,9 @@ export interface ViralVideoJob { style_template_id?: string style_guide?: string | Record user_copy_text?: string + /** v1.6: = copy_result.voiceover_script(从 copy_result 派生,向后兼容) */ final_copy_text?: string + generated_copy_text?: string fusion_level?: FusionLevel voice_id?: string voice_mode?: "global" | "per_video" @@ -131,12 +180,17 @@ export interface ViralVideoJob { bgm_preference?: string intent_result?: IntentResult intent_text?: string + /** v1.6 编导分镜脚本(核心产物) */ copy_result?: CopyResult + /** 向后兼容:= copy_result.shots */ + storyboard?: ShotScript[] image_analysis?: ImageAnalysisResult - /** 视频比例 */ + /** 视频比例:9:16 / 16:9 / 1:1,默认 9:16 */ video_ratio?: string - /** Seedance 模型 ID */ + /** Seedance 模型 ID(空=后端默认) */ video_model?: string + /** 视频时长(秒,5-30,默认15) */ + duration?: number progress_stage?: ViralVideoStage progress_percent?: number progress_message?: string @@ -166,6 +220,7 @@ export interface GenerateViralVideoRequest { persona_id?: string viral_structure?: string marketing_purpose?: string + /** 视频时长(5-30秒,默认15) */ duration?: number video_model?: string video_ratio?: string @@ -180,23 +235,25 @@ export interface HistoryResponse { page_size: number } -/** v1.5 阶段1请求:仅做图片/视频分析(POST /viral-video/analyze-images) */ +/** v1.6 阶段1请求:图片/视频分析(POST /viral-video/analyze-images) */ export interface AnalyzeImagesRequest { images: string[] reference_video_url?: string style_template_id?: string style_strength?: StyleStrength - /** TTS 音色 ID(STEP1 已选音色时传;阶段2 generate-copy 可覆盖) */ + /** TTS 音色 ID(STEP1 已选音色时传) */ voice_id?: string /** 音色来源:preset | library | clone | upload */ voice_source?: "preset" | "library" | "clone" | "upload" - /** Seedance 视频比例:9:16 | 16:9 | 1:1 等 */ + /** Seedance 视频比例:9:16 | 16:9 | 1:1 */ video_ratio?: string /** Seedance 模型 ID(空则使用服务端默认) */ video_model?: string + /** 视频时长(秒,5-30,默认15) */ + duration?: number } -/** v1.5 阶段2请求:填完营销参数后生成文案+分镜(POST /viral-video/{id}/generate-copy) */ +/** v1.6 阶段2请求:填完营销参数后生成编导分镜脚本(POST /viral-video/{id}/generate-copy) */ export interface GenerateCopyRequest { industry?: string target_customer?: string @@ -204,6 +261,7 @@ export interface GenerateCopyRequest { viral_structure?: string marketing_purpose?: string bgm_preference?: string + /** 视频时长(秒,5-30,默认15) */ duration?: number user_copy_text?: string fusion_level?: FusionLevel @@ -222,13 +280,13 @@ export interface GenerateCopyRequest { video_model?: string } -/** v1.5 阶段3请求:用户确认/编辑文案后开始渲染(POST /viral-video/{id}/confirm-copy) */ +/** v1.6 阶段3请求:用户确认/编辑口播文案后开始单次 Seedance 出片(POST /viral-video/{id}/confirm-copy) */ export interface ConfirmCopyRequest { - /** 用户编辑后的最终文案;为空则使用 AI 生成文案 */ + /** 用户编辑后的口播文案;为空则使用 AI 生成的 voiceover_script */ edited_copy?: string } -/** 分镜片段结构(后端 storyboard 字段的元素形态,保留供调试/进阶使用;主流程请使用 copy_result.scenes) */ +/** 旧分镜片段结构(保留兼容;新代码请使用 ShotScript) */ export interface StoryboardSegment { order: number type: string diff --git a/apps/worker/worker_app/tasks/viral_video.py b/apps/worker/worker_app/tasks/viral_video.py index 798279890..0f7d74aec 100644 --- a/apps/worker/worker_app/tasks/viral_video.py +++ b/apps/worker/worker_app/tasks/viral_video.py @@ -1,16 +1,20 @@ -"""爆款视频 Celery 编排器 — ViralVideoOrchestrator. +"""爆款视频 Celery 编排器 — ViralVideoOrchestrator (v1.6 单次 Seedance 出片版). -9 步流水线(Seedance 2.5 直生口型,不再走 MuseTalk): - 1. 图片 VLM 分析 - 1.5 [v1.3] 视频风格分析(如用户上传参考视频) - 2. 用户文案意图解析 - 3. 文案融合生成 - 4. 分镜脚本生成 - 5. 合规审核(6 维度,不通过自动重写 1 次) - 6. CosyVoice 配音 - 7. BGM 选择(素材未就绪时跳过) - 8. Seedance 逐分镜生成 + ffmpeg concat + 混 TTS - 9. OSS 上传 + 通知 + 扣点 +v1.6 重大简化(Seedance 2.5 单次最长 30 秒,直接出片): + 1. _step_image_analysis 图片 VLM 分析(保留) + 1.5 _step_video_analysis 参考视频风格分析(可选) + 2. _step_intent_parsing 用户文案意图解析 + 3. _step_script_generation 编导分镜脚本生成(融合原 copy_fusion+storyboard+review,输出 copy_result 结构 + voiceover_script) + 4. _step_review 合规审核(6 维度,不通过自动重写 1 次) + 5. _step_tts CosyVoice 整段配音(voiceover_script → 单个 mp3 → 上传 OSS 拿公网 URL) + 6. _step_render 单次 Seedance 生成(prompt=完整编导脚本,reference_audios=[TTS URL],reference_images=产品图,generate_audio=true) + 7. _step_upload OSS 上传单个视频文件 + 通知 + 扣点 + +删除/不再使用: +- 分镜拆分多段生成(storyboard 不再单独驱动分段生成,仅作为 copy_result.shots 存到 DB 给前端/日志参考) +- ffmpeg concat 拼接(concat_engine 保留但 viral video 主流程不再调用) +- placeholder 占位视频、分段重试降级 +- BGM 单独混音(Seedance generate_audio=true 原生生成环境音效/BGM) """ from __future__ import annotations @@ -18,10 +22,12 @@ from __future__ import annotations import json import logging import os +import tempfile +from pathlib import Path from celery import Task, shared_task from celery.exceptions import Retry -from worker_app.celery_app import celery_app # noqa: F401 - 加载 app 以注册任务 +from worker_app.celery_app import celery_app # noqa: F401 from worker_app.db import SessionLocal from packages.adapters.sqlalchemy_impl.viral_video_repository import ( @@ -49,15 +55,6 @@ def _emit_progress( data: dict | None = None, event_type: str = "viral_video:progress", ): - """通过 Redis 发布进度事件,供 WebSocket 消费。 - - event_type 取值: - - viral_video:progress 中间进度(默认) - - viral_video:completed 任务完成 - - viral_video:failed 任务失败 - - viral_video:wait_user 等待用户确认 - 所有事件 payload 均为合法 JSON,前端 JSON.parse 即可。 - """ try: import redis as redis_lib @@ -80,7 +77,6 @@ def _emit_progress( def _get_repo_and_job(job_id: str): - """获取 session, repo, job 三元组。""" session = SessionLocal() repo = SQLAlchemyViralVideoJobRepository(session) job = repo.get(job_id) @@ -88,12 +84,53 @@ def _get_repo_and_job(job_id: str): def _save_job(repo, job, session): - """持久化并关闭 session。""" repo.update(job) session.commit() -# ── 流水线各步骤 ──────────────────────────────────────────────────────── +# ── 默认结构 ───────────────────────────────────────────────────────────── + +_DEFAULT_HARD_CONSTRAINTS = [ + "无字幕、无水印、无任何自动生成文字、无 logo", + "同一人物全程保持一致的五官、发型、服装、身材,不得换脸或变形", + "口播语音必须在指定时长内自然念完,语速自然,口型与语音同步", + "画面流畅无闪烁、无多余肢体、无扭曲变形、无穿模", + "色彩自然、曝光正确、电影级质感、高清细节", +] + +_DEFAULT_NEGATIVE_PROMPTS = [ + "字幕", + "自动字幕", + "水印", + "logo", + "图标", + "错误文字", + "乱码文字", + "男女声错配", + "中途换声", + "五官崩坏", + "脸部变形", + "多余手指", + "肢体扭曲", + "闪烁", + "画面抖动", + "模糊", + "低分辨率", +] + + +def _empty_copy_result(duration: int = 15, ratio: str = "9:16") -> dict: + return { + "overview": {"theme": "好物推荐", "total_duration": duration, "aspect_ratio": ratio}, + "scene_and_lighting": "简洁明亮的室内场景,柔和自然光,产品主体清晰", + "shots": [], + "hard_constraints": list(_DEFAULT_HARD_CONSTRAINTS), + "negative_prompts": list(_DEFAULT_NEGATIVE_PROMPTS), + "voiceover_script": "", + "final_copy": "", + "suggested_copy": "", + "title": "", + } # ── 流水线各步骤 ──────────────────────────────────────────────────────── @@ -124,29 +161,26 @@ _IMAGE_ANALYSIS_PROMPT = """请仔细观察这张图片,只基于图片中真 """ -def _step_image_analysis(job: ViralVideoJob) -> dict: - """步骤 1: 图片 VLM 分析 — 识别产品特征、场景、卖点。 +def _vision_fallback(idx: int, reason: str, extra: dict | None = None) -> dict: + d = { + "name": "未识别", + "category": "无法判断", + "key_features": [], + "scene": "通用", + "_source": reason, + } + if extra: + d.update(extra) + return d - Bug #2114 修复: - 1) call_vision 现已走视觉模型 doubao-1-5-vision-pro(之前误走文本模型导致完全没看图); - 2) Prompt 强化为结构化 JSON schema,禁止编造,强制图片可见才写; - 3) 单张失败不影响其他图片,最终至少返回一张占位结果避免后续 NoneType; - 4) 日志打印每张图的 URL 和模型原始返回,方便排查。 - """ + +def _step_image_analysis(job: ViralVideoJob) -> dict: + """步骤 1: 图片 VLM 分析 — 识别产品特征、场景、卖点(加 None 防护)。""" try: from packages.shared.ai_service import call_vision except ImportError: logger.warning("[爆款视频] ai_service.call_vision 不可用,使用占位结果") - return { - "products": [ - { - "name": "产品", - "features": ["特征1", "特征2"], - "scene": "通用场景", - "_source": "fallback_import_error", - } - ] - } + return {"products": [_vision_fallback(0, "fallback_import_error")]} if not job.images: logger.warning("[爆款视频] 任务无 images,跳过图片分析") @@ -154,64 +188,47 @@ def _step_image_analysis(job: ViralVideoJob) -> dict: results = [] for idx, img_url in enumerate(job.images): + if not img_url or not isinstance(img_url, str): + logger.warning("[爆款视频] 图片 #%d URL 非法", idx) + results.append(_vision_fallback(idx, "invalid_url")) + continue logger.info("[爆款视频] 图片分析 #%d img=%s", idx, img_url[:160]) try: result = call_vision(image_url=img_url, prompt=_IMAGE_ANALYSIS_PROMPT) if result is None: - logger.warning("[爆款视频] 图片 #%d call_vision 返回 None(模型超时/Key未配置)", idx) - results.append( - { - "name": "未识别", - "category": "无法判断", - "key_features": [], - "scene": "通用", - "_source": "vision_none", - } - ) + logger.warning("[爆款视频] 图片 #%d call_vision 返回 None", idx) + results.append(_vision_fallback(idx, "vision_none")) elif isinstance(result, str): - # JSON 解析失败返回的原文,包装一下防止后续 .get 报错 - logger.warning("[爆款视频] 图片 #%d VLM 返回非 JSON 文本,包装为 features: %s", idx, result[:200]) - results.append( - { - "name": "未识别", - "category": "无法判断", - "key_features": [], - "scene": "通用", - "_raw": result[:500], - "_source": "vision_text", - } - ) - else: - # dict 正常 + # VLM 返回了非 JSON 文本(JSON 解析失败),记录原始文本但不要让 None 传播 + logger.warning("[爆款视频] 图片 #%d VLM 返回非 JSON 文本: %s", idx, result[:200]) + results.append(_vision_fallback(idx, "vision_text", {"_raw": result[:500]})) + elif isinstance(result, dict): result.setdefault("_source", "vision") + # 防御:关键字段缺失则补默认 + result.setdefault("name", "未识别") + result.setdefault("category", "无法判断") + result.setdefault("key_features", []) + result.setdefault("scene", "通用") results.append(result) + else: + logger.warning("[爆款视频] 图片 #%d VLM 返回意外类型 %s", idx, type(result)) + results.append(_vision_fallback(idx, "vision_unexpected_type")) except Exception as e: logger.warning("[爆款视频] 图片分析失败 img=%s err=%s", img_url[:120], e, exc_info=True) - results.append( - { - "name": "未识别", - "category": "无法判断", - "key_features": [], - "scene": "通用", - "_source": "vision_exception", - "_error": str(e)[:200], - } - ) + results.append(_vision_fallback(idx, "vision_exception", {"_error": str(e)[:200]})) return {"products": results} def _step_video_analysis(job: ViralVideoJob) -> dict | None: - """步骤 1.5 [v1.3]: 参考视频风格分析。""" + """步骤 1.5: 参考视频风格分析(可选)。""" if not job.reference_video_url: return None - try: - # P0-2: 修正 import 路径(video_analyzer.py 在 apps/worker/viral_video/ 下,worker PYTHONPATH 含 apps/worker) from viral_video.video_analyzer import analyze_video_style style_guide = analyze_video_style(job.reference_video_url) - return style_guide + return style_guide if isinstance(style_guide, dict) else None except ImportError as e: logger.info("[爆款视频] video_analyzer 模块未就绪(%s),使用占位风格分析", e) return { @@ -228,14 +245,17 @@ def _step_video_analysis(job: ViralVideoJob) -> dict | None: def _step_intent_parsing(job: ViralVideoJob, image_analysis: dict) -> dict: - """步骤 2: 用户文案意图解析 — 理解用户想表达什么。""" + """步骤 2: 用户文案意图解析。""" try: from packages.shared.ai_service import call_llm except ImportError: - return {"intent": "推广产品", "key_messages": ["产品亮点"], "tone": "专业"} + return {"intent": "推广产品", "key_messages": ["产品亮点"], "tone": "专业", "suggested_title": ""} products_summary = "" - for p in image_analysis.get("products", []): + products = (image_analysis or {}).get("products", []) or [] + for p in products: + if not isinstance(p, dict): + continue feats = p.get("key_features") or p.get("features") or [] extras = [] if p.get("brand") and p.get("brand") not in ("未知", "无法判断"): @@ -244,221 +264,373 @@ def _step_intent_parsing(job: ViralVideoJob, image_analysis: dict) -> dict: extras.append(f"品类={p['category']}") if p.get("colors"): extras.append(f"颜色={','.join(p['colors'])}") - if p.get("scene") and p.get("scene") not in ("通用",): - extras.append(f"场景={p['scene']}") + if p.get("visual_style"): + extras.append(f"风格={p['visual_style']}") feat_str = ", ".join([str(x) for x in feats + extras]) products_summary += f"- {p.get('name', '产品')}: {feat_str}\n" - prompt = f"""你是一个营销文案策略师。请分析以下信息,理解用户的营销意图: + prompt = f"""你是一个营销编导。请分析以下信息,理解用户的营销意图并给出短视频主题建议: -用户原始文案:{job.user_copy_text or "(未提供)"} +用户原始文案:{job.user_copy_text or "(未提供,全由 AI 创作)"} 行业:{job.industry or "未指定"} 目标客户:{job.target_customer or "未指定"} 营销目的:{job.marketing_purpose or "未指定"} +视频时长:{job.duration}秒 产品信息: -{products_summary} +{products_summary or "- (无图片分析结果)"} -请分析并返回JSON格式: -1. intent: 核心营销意图(一句话) -2. key_messages: 要传达的3-5个关键信息 -3. tone: 文案调性(如:专业/亲切/高端/活力) -4. target_emotion: 希望触发的用户情感 -5. call_to_action: 行动号召建议""" +请返回严格 JSON(不要 Markdown,不要解释): +{{ + "intent": "核心营销意图(一句话)", + "key_messages": ["要传达的3-5个关键信息"], + "tone": "文案调性(如亲切/专业/高端/活力/治愈/搞笑)", + "target_emotion": "希望触发的用户情感", + "call_to_action": "行动号召短句(口语化,5-10字)", + "suggested_title": "视频主题标题(5-15字)" +}}""" try: result = call_llm(prompt) - return result if isinstance(result, dict) else {"raw": result} + return ( + result + if isinstance(result, dict) + else {"intent": str(result)[:200], "key_messages": [], "tone": "专业", "suggested_title": ""} + ) except Exception as e: logger.warning("[爆款视频] 意图解析失败: %s", e) - return {"intent": "推广产品", "key_messages": ["产品亮点"], "tone": "专业"} + return {"intent": "推广产品", "key_messages": ["产品亮点"], "tone": "专业", "suggested_title": ""} -def _step_copy_fusion(job: ViralVideoJob, intent: dict, image_analysis: dict) -> str: - """步骤 3: 文案融合生成 — 根据 fusion_level 融合用户文案和 AI 文案。""" - try: - from packages.shared.ai_service import call_llm - except ImportError: - return f"【{job.industry or '行业'}】优质产品,{job.target_customer or '您'}的不二之选!" +# ── 编导分镜脚本生成(核心,v1.6 新 prompt) ────────────────────────────── - products_desc = "" - for p in image_analysis.get("products", []): + +_SCRIPT_GENERATION_PROMPT = """你是一名资深短视频导演,擅长为 AI 视频生成模型(Seedance 2.5)撰写专业编导分镜脚本。 + +## 产品信息 +{products_summary} + +## 营销参数 +- 视频主题/意图:{intent} +- 关键信息:{key_messages} +- 调性:{tone} +- 目标客户:{target_customer} +- 用户原始文案/卖点(必须融入口播):{user_copy} +- 视频时长:{duration} 秒(单次生成) +- 画幅比例:{ratio} +- 产品图片数量:{n_images} 张(将作为 reference_images 传给视频模型,第1张通常作为首帧/主产品图) +- 参考风格(可选):{style_hint} + +## 任务 +请撰写**一段完整的编导分镜脚本**,包含视频总览、场景光线、逐镜头时间轴、硬性约束、负面提示词,以及自然口语化的口播对白。 + +这段脚本会**整个拼成一个长 prompt**一次性传给 Seedance 2.5(单次生成最多30秒视频),所以你的描述必须让模型在一个长镜头/连贯镜头流里理解每个时间段该拍什么、画面如何、人物说什么做什么。 + +## 输出格式(必须输出严格 JSON,不要 Markdown,不要解释,字段一个都不能少) + +```json +{{ + "overview": {{ + "theme": "视频主题(一句话概括)", + "total_duration": {duration}, + "aspect_ratio": "{ratio}" + }}, + "scene_and_lighting": "整体场景描述+光线设定(100-200字,要具体:在哪拍、什么光线、什么色调、什么氛围)", + "shots": [ + {{ + "time_range": "0-3秒", + "shot_type_angle_movement": "景别+角度+运镜(例:近景俯拍45度,缓慢推镜;中景平视,固定镜头;特写平视,快速拉镜)", + "scene_and_dialogue": "画面场景描述 + 人物口播台词(对白要自然口语化,像朋友聊天,不要硬广推销腔)", + "action_details": "人物动作、表情、物品操作细节(手怎么动、表情变化、产品怎么展示)", + "audio_bgm": "环境音+BGM提示(例:轻快流行BGM,环境嘈杂咖啡店背景音)", + "transition": "硬切/淡入淡出/叠化(最后一镜写『结束』即可)", + "reference_image_index": 0 + }} + // ... 按时间顺序列出所有镜头,总时长累计 = {duration} 秒 + ], + "hard_constraints": [ + "无字幕、无水印、无任何自动生成文字、无logo", + "同一人物全程五官、发型、服装、身材保持一致,不得换脸变形", + "口播语音在总时长内自然念完,语速自然,口型与语音严格同步", + "画面流畅无闪烁、无多余肢体、无扭曲变形、无穿模", + "色彩自然、曝光正确、电影级质感、高清细节" + ], + "negative_prompts": [ + "字幕","自动字幕","水印","logo","图标","错误文字","乱码文字", + "男女声错配","中途换声","五官崩坏","脸部变形","多余手指", + "肢体扭曲","闪烁","画面抖动","模糊","低分辨率" + ], + "voiceover_script": "完整口播稿(把 shots 里所有对白自然拼接成一段,口语化,不加旁白标注、不加镜头标注、不加'主播:'之类前缀,就是纯念出来的文本,长度适配{duration}秒,约{approx_chars}字)" +}} +``` + +## 关键要求 +1. **镜头感**:每镜必须写清景别(特写/近景/中景/全景)、角度(平视/俯拍/仰拍/45度侧拍)、运镜(推/拉/摇/移/跟/固定),不能笼统说"展示产品"。 +2. **画面具体**:描述主体是谁(性别/年龄/穿着风格)、在什么场景、做什么动作、光线从哪来、镜头怎么动,让 AI 能画出来。 +3. **对白自然**:像真人说话,不要"家人们谁懂啊""宝子们"这种浮夸腔,也不要"今天给大家推荐一款XX真的太好用了"这种硬广推销腔。要像朋友自然分享好物。 +4. **参考图片分配**:reference_image_index 填 0-based 索引,产品特写镜头用产品图(索引0通常是主图),人像/场景镜头可留 null。 +5. **时长控制**:所有 shots 的 time_range 加起来必须等于 {duration} 秒,单镜 2-8 秒。 +6. **硬性约束和负面词必须包含**:不要删减,可根据产品类型追加。 +7. **voiceover_script 必须是纯口播文本**:不含任何标记、括号、说明,字数按中文每秒 3-4 字估算({duration}秒约{approx_chars}字)。 +""" + + +def _build_products_summary(image_analysis: dict) -> str: + products = (image_analysis or {}).get("products", []) or [] + if not products: + return "- (无图片信息,请自由创作自然生活化场景)" + lines = [] + for i, p in enumerate(products): + if not isinstance(p, dict): + continue + name = p.get("name") or "产品" + brand = p.get("brand") or "" + cat = p.get("category") or "" + colors = p.get("colors") or [] + mat = p.get("material_or_texture") or "" + style = p.get("visual_style") or "" + scene = p.get("scene") or "" + audience = p.get("target_audience_hint") or "" + text_on_img = p.get("text_on_image") or "" feats = p.get("key_features") or p.get("features") or [] - extras = [] - if p.get("brand") and p.get("brand") not in ("未知", "无法判断"): - extras.append(f"品牌={p['brand']}") - if p.get("visual_style"): - extras.append(f"风格={p['visual_style']}") - feat_str = ",".join([str(x) for x in feats + extras]) - products_desc += f"{p.get('name', '产品')}({feat_str})\n" + parts = [f"图{i+1} {name}"] + if brand and brand not in ("未知", "无法判断"): + parts.append(f"品牌={brand}") + if cat and cat not in ("无法判断", "非产品图"): + parts.append(f"品类={cat}") + if colors: + parts.append(f"颜色={','.join(colors)}") + if mat and mat not in ("无法判断",): + parts.append(f"材质={mat}") + if style: + parts.append(f"风格={style}") + if scene and scene not in ("通用",): + parts.append(f"场景={scene}") + if audience and audience != "通用": + parts.append(f"目标人群={audience}") + if text_on_img and text_on_img not in ("无",): + parts.append(f"图片文字={text_on_img}") + if feats: + parts.append("外观特征=" + ";".join([str(x) for x in feats[:6]])) + lines.append("- " + ",".join(parts)) + return "\n".join(lines) - if job.fusion_level == "ai_full": - prompt = f"""请为以下产品撰写一段爆款短视频文案({job.duration}秒): -产品:{products_desc} -行业:{job.industry} -目标客户:{job.target_customer} -营销目的:{job.marketing_purpose} -调性:{intent.get("tone", "专业")} -关键信息:{", ".join(intent.get("key_messages", []))} - -要求:吸引眼球、节奏紧凑、有行动号召。直接输出文案内容。""" - elif job.fusion_level == "user_primary": - prompt = f"""请基于用户原始文案进行润色优化,保留用户原意和风格: -用户原文:{job.user_copy_text} -产品信息:{products_desc} - -要求:保留用户原意,仅修正表达和节奏。直接输出文案内容。""" - else: # ai_polish (default) - prompt = f"""请将用户文案与AI分析融合,生成一段优化后的爆款短视频文案({job.duration}秒): -用户原文:{job.user_copy_text or "(未提供)"} -产品分析:{products_desc} -行业:{job.industry} -目标客户:{job.target_customer} -营销目的:{job.marketing_purpose} -意图分析:{intent.get("intent", "")} -调性:{intent.get("tone", "专业")} - -要求:融合用户意图和产品卖点,节奏紧凑,适合短视频。直接输出文案内容。""" +def _safe_json_loads(raw: str | dict | list | None): + if raw is None: + return None + if isinstance(raw, (dict, list)): + return raw + if not isinstance(raw, str): + return None + s = raw.strip() + if s.startswith("```"): + s = s.strip("`") + if s.startswith("json"): + s = s[4:].lstrip() try: - result = call_llm(prompt) - return result if isinstance(result, str) else str(result) - except Exception as e: - logger.warning("[爆款视频] 文案融合失败: %s", e) - return job.user_copy_text or f"精选{job.industry or '行业'}好物,值得关注!" + return json.loads(s) + except Exception: + # 尝试截取第一个 { ... } 或 [ ... ] + try: + for open_c, close_c in (("{", "}"), ("[", "]")): + i = s.find(open_c) + j = s.rfind(close_c) + if i >= 0 and j > i: + return json.loads(s[i : j + 1]) + except Exception: + pass + return None -def _step_storyboard(job: ViralVideoJob, copy_text: str, image_analysis: dict) -> list[dict]: - """步骤 4: 分镜脚本生成。每个分镜独立一段视频,段内时长建议 3~6 秒。""" - try: - from packages.shared.ai_service import call_llm - except ImportError: - return [ +def _fallback_script(job: ViralVideoJob) -> dict: + """脚本生成失败时的兜底脚本(极简但可用)。""" + dur = max(5, min(30, int(getattr(job, "duration", 15) or 15))) + ratio = getattr(job, "video_ratio", None) or "9:16" + base = _empty_copy_result(dur, ratio) + voiceover = job.user_copy_text or "你好,给大家分享一款我最近在用的好物,真的很不错,推荐你们也试试。" + shots = [ + { + "time_range": f"0-{dur}秒", + "shot_type_angle_movement": "中景平视,缓慢推镜", + "scene_and_dialogue": "明亮室内,人物自然出镜,微笑着看向镜头。" + voiceover, + "action_details": "人物手持产品自然展示,表情亲切,动作流畅", + "audio_bgm": "轻快流行BGM", + "transition": "结束", + "reference_image_index": 0 if job.images else None, + } + ] + base["shots"] = shots + base["voiceover_script"] = voiceover + base["final_copy"] = voiceover + base["suggested_copy"] = voiceover + base["title"] = "好物分享" + return base + + +def _validate_and_normalize_script(raw, job: ViralVideoJob) -> dict: + """把 LLM 返回的脚本规范化、补默认、校验结构。""" + dur = max(5, min(30, int(getattr(job, "duration", 15) or 15))) + ratio = getattr(job, "video_ratio", None) or "9:16" + base = _empty_copy_result(dur, ratio) + + if not isinstance(raw, dict): + logger.warning("[爆款视频] 脚本返回非 dict,使用兜底") + return _fallback_script(job) + + # overview + ov = raw.get("overview") + if isinstance(ov, dict): + base["overview"] = { + "theme": str(ov.get("theme") or "好物分享"), + "total_duration": int(ov.get("total_duration") or dur), + "aspect_ratio": str(ov.get("aspect_ratio") or ratio), + } + else: + base["overview"]["theme"] = str(raw.get("title") or "好物分享") + + base["scene_and_lighting"] = str(raw.get("scene_and_lighting") or base["scene_and_lighting"]) + + # shots + shots_raw = raw.get("shots") + shots: list[dict] = [] + if isinstance(shots_raw, list): + for i, s in enumerate(shots_raw): + if not isinstance(s, dict): + continue + shots.append( + { + "time_range": str(s.get("time_range") or f"{i*3}-{(i+1)*3}秒"), + "shot_type_angle_movement": str(s.get("shot_type_angle_movement") or "中景平视,固定镜头"), + "scene_and_dialogue": str(s.get("scene_and_dialogue") or ""), + "action_details": str(s.get("action_details") or ""), + "audio_bgm": str(s.get("audio_bgm") or "轻快BGM"), + "transition": str(s.get("transition") or ("硬切" if i < len(shots_raw) - 1 else "结束")), + "reference_image_index": s.get("reference_image_index"), + } + ) + if not shots: + shots = [ { - "order": 0, - "type": "product_shot", - "text": copy_text[:50], - "duration": min(5, job.duration), - "description": "产品展示", - "ken_burns": "zoom_in", - "transition": "cut", + "time_range": f"0-{dur}秒", + "shot_type_angle_movement": "中景平视,缓慢推镜", + "scene_and_dialogue": "明亮室内场景,人物自然出镜。", + "action_details": "自然展示产品", + "audio_bgm": "轻快BGM", + "transition": "结束", + "reference_image_index": 0 if job.images else None, } ] + base["shots"] = shots - products_hint = "" - products = image_analysis.get("products", []) if image_analysis else [] - if products: - p0 = products[0] if isinstance(products[0], dict) else {} - feats = (p0.get("key_features") or p0.get("features") or []) if isinstance(p0, dict) else [] - brand = p0.get("brand") if isinstance(p0, dict) else "" - brand_hint = f"(品牌={brand})" if brand and brand not in ("未知", "无法判断") else "" - products_hint = f"\n首帧参考产品特征:{p0.get('name','')}{brand_hint} - {', '.join(feats[:3])}" + # hard_constraints / negative_prompts + hc = raw.get("hard_constraints") + if isinstance(hc, list) and hc: + merged = list(_DEFAULT_HARD_CONSTRAINTS) + for x in hc: + if isinstance(x, str) and x and x not in merged: + merged.append(x) + base["hard_constraints"] = merged + np = raw.get("negative_prompts") + if isinstance(np, list) and np: + merged = list(_DEFAULT_NEGATIVE_PROMPTS) + for x in np: + if isinstance(x, str) and x and x not in merged: + merged.append(x) + base["negative_prompts"] = merged - seg_seconds = 5 - n_segments = max(2, min(6, max(1, job.duration // seg_seconds))) - ratio = "9:16" + # voiceover_script: 优先从字段取,否则从各镜 scene_and_dialogue 提取(粗暴拼接冒号后部分 / 中文句) + voiceover = str(raw.get("voiceover_script") or "").strip() + if not voiceover: + # 兜底:把所有 scene_and_dialogue 拼接起来,去除镜头描述部分(含"景"、"俯拍"、"平视"等词的前缀) + import re - prompt = f"""请根据以下文案生成爆款短视频分镜脚本,共 {n_segments} 个分镜: + parts = [] + for s in shots: + txt = s.get("scene_and_dialogue", "") + # 去除开头到第一个句号/逗号前的"镜头描述"部分 + # 简单策略:找第一个中文说话片段——按句号切,后半段更像对白 + segs = re.split(r"[。!?]", txt) + for seg in segs: + seg = seg.strip(" ,,。.!?!?::") + if len(seg) >= 4 and not any( + k in seg for k in ("景别", "俯拍", "仰拍", "平视", "镜头", "特写", "中景", "全景", "近景", "运镜") + ): + parts.append(seg) + voiceover = "。".join(parts) if parts else (job.user_copy_text or "你好,给大家分享一款好物。") + base["voiceover_script"] = voiceover + base["final_copy"] = voiceover + base["suggested_copy"] = voiceover + base["title"] = base["overview"]["theme"] + return base -文案内容:{copy_text} -视频总时长:{job.duration}秒(每个分镜 3~6 秒,总和约等于总时长) -风格强度:{job.style_strength} -输出宽高比:{ratio}{products_hint} -请以 JSON 数组格式返回分镜列表,每个分镜包含: -- order: 序号(从0开始) -- type: 镜头类型(product_shot/close_up/scene/action/text_card/closing) -- description: 画面详细描述(中文,含主体、动作、场景、运镜、光影,用于AI视频生成prompt) -- text: 该分镜配音/字幕文本 -- duration: 时长(秒,3~6秒的整数) -- ken_burns: 运镜方式(zoom_in/zoom_out/pan_left/pan_right/static) -- transition: 与下一分镜的转场(cut/dissolve/fade)""" +def _step_script_generation(job: ViralVideoJob, intent: dict, image_analysis: dict) -> dict: + """步骤 3: 编导分镜脚本生成(v1.6 核心,输出 copy_result 结构)。""" + try: + from packages.shared.ai_service import call_llm + except ImportError: + return _fallback_script(job) + + products_summary = _build_products_summary(image_analysis) + style_hint = "无" + if isinstance(job.style_guide, dict): + style_hint = ( + f"节奏{job.style_guide.get('cut_speed','')}、转场{job.style_guide.get('transition','')}、" + f"色调{job.style_guide.get('color_grade','')}、能量{job.style_guide.get('energy','')}" + ) + + dur = max(5, min(30, int(getattr(job, "duration", 15) or 15))) + ratio = getattr(job, "video_ratio", None) or "9:16" + approx_chars = max(20, dur * 4) + + intent_str = "" + key_msgs = "" + tone = "" + if isinstance(intent, dict): + intent_str = intent.get("intent") or "推广产品" + key_msgs = "、".join(intent.get("key_messages") or []) + tone = intent.get("tone") or "亲切自然" + else: + intent_str = "推广产品" + tone = "亲切自然" + + prompt = _SCRIPT_GENERATION_PROMPT.format( + products_summary=products_summary, + intent=intent_str, + key_messages=key_msgs or "产品亮点", + tone=tone, + target_customer=job.target_customer or "通用人群", + user_copy=job.user_copy_text or "(未提供,自由创作)", + duration=dur, + ratio=ratio, + n_images=len(job.images or []), + style_hint=style_hint, + approx_chars=approx_chars, + ) try: result = call_llm(prompt) - if isinstance(result, list): - return _normalize_storyboard(result, job.duration, n_segments, copy_text) - import json as _json - - parsed = _json.loads(result) if isinstance(result, str) else result - if isinstance(parsed, list): - return _normalize_storyboard(parsed, job.duration, n_segments, copy_text) + parsed = _safe_json_loads(result) + return _validate_and_normalize_script(parsed, job) except Exception as e: - logger.warning("[爆款视频] 分镜生成失败: %s", e) - - return _fallback_storyboard(copy_text, job.duration, n_segments) + logger.warning("[爆款视频] 编导脚本生成失败: %s,使用兜底脚本", e, exc_info=True) + return _fallback_script(job) -def _normalize_storyboard(raw: list, total_duration: int, n_segments: int, copy_text: str) -> list[dict]: - """规范化 LLM 输出的分镜:填充缺省字段、保证总时长合理。""" - out: list[dict] = [] - for i, item in enumerate(raw): - if not isinstance(item, dict): - continue - try: - dur = int(item.get("duration") or 5) - except (TypeError, ValueError): - dur = 5 - dur = max(3, min(8, dur)) - out.append( - { - "order": int(item.get("order", i)), - "type": str(item.get("type", "product_shot")), - "description": str(item.get("description", copy_text[:80])), - "text": str(item.get("text", "")), - "duration": dur, - "ken_burns": str(item.get("ken_burns", "zoom_in")), - "transition": str(item.get("transition", "cut")), - } - ) - if not out: - return _fallback_storyboard(copy_text, total_duration, n_segments) - out = out[:n_segments] - total = sum(s["duration"] for s in out) - if total > 0 and total != total_duration: - scale = total_duration / total - acc = 0 - for s in out[:-1]: - s["duration"] = max(3, min(8, round(s["duration"] * scale))) - acc += s["duration"] - out[-1]["duration"] = max(3, total_duration - acc) - return out - - -def _fallback_storyboard(copy_text: str, total_duration: int, n_segments: int) -> list[dict]: - if n_segments <= 0: - n_segments = 1 - dur = total_duration // n_segments - remainder = total_duration - dur * n_segments - out = [] - for i in range(n_segments): - d = dur + (remainder if i == n_segments - 1 else 0) - out.append( - { - "order": i, - "type": "product_shot", - "description": f"产品展示镜头 {i + 1}:{copy_text[:40]}", - "text": copy_text, - "duration": max(3, d), - "ken_burns": "zoom_in" if i % 2 == 0 else "pan_left", - "transition": "cut", - } - ) - return out - - -def _step_review(job: ViralVideoJob, copy_text: str, storyboard: list[dict]) -> dict: - """步骤 5: 合规审核(6 维度)。不通过时自动重写 1 次。""" +def _step_review(job: ViralVideoJob, copy_result: dict) -> dict: + """步骤 4: 合规审核(简化版:基于脚本的 voiceover_script+shots 文本)。""" dimensions = ["广告法合规", "平台规范", "内容真实性", "版权安全", "价值观", "风格一致性"] - try: from packages.shared.ai_service import call_llm except ImportError: return {"passed": True, "score": 90, "details": {d: "通过" for d in dimensions}} - prompt = f"""请对以下短视频内容进行合规审核,检查6个维度:{", ".join(dimensions)} + voiceover = (copy_result or {}).get("voiceover_script", "") + shots_preview = json.dumps((copy_result or {}).get("shots", [])[:3], ensure_ascii=False) + prompt = f"""请对以下短视频编导脚本进行合规审核,检查6个维度:{", ".join(dimensions)} -文案内容:{copy_text} -分镜脚本:{storyboard[:3]}... +口播文案:{voiceover} +前3个镜头:{shots_preview} 行业:{job.industry} 请以JSON格式返回: @@ -466,7 +638,6 @@ def _step_review(job: ViralVideoJob, copy_text: str, storyboard: list[dict]) -> - score: int(0-100分) - details: 各维度评分和说明 - issues: 需要修改的问题列表(如有)""" - try: result = call_llm(prompt) return result if isinstance(result, dict) else {"passed": True, "score": 80, "details": {}} @@ -475,249 +646,171 @@ def _step_review(job: ViralVideoJob, copy_text: str, storyboard: list[dict]) -> return {"passed": True, "score": 75, "details": {d: "默认通过" for d in dimensions}} -def _step_tts(job: ViralVideoJob, copy_text: str): - """步骤 6: CosyVoice 配音。P1:返回 Path;失败返回 None。""" +def _step_tts(job: ViralVideoJob, voiceover_script: str): + """步骤 5: CosyVoice 整段配音 → 返回本地 MP3 Path;失败返回 None。""" try: from pathlib import Path as _Path - # 使用绝对包路径,避免 celery worker 因 cwd/PYTHONPATH 微小差异找不到 services 模块 from apps.worker.services.tts_service_factory import get_tts_service tts_service = get_tts_service() - # v1.5.1: 优先使用 voice_id;voice_id 为空时回退 persona_id(旧字段兼容); - # 再空则用 CosyVoice 默认 longxiaochun_v3。统一输出 mp3 给后续 ffmpeg 混音。 voice_id = (getattr(job, "voice_id", "") or job.persona_id or "").strip() + text = (voiceover_script or "").strip() + if not text: + logger.warning("[爆款视频] voiceover_script 为空,跳过 TTS") + return None try: result = tts_service.synthesize( - text=copy_text, + text=text, voice_id=voice_id or "longxiaochun_v3", format="mp3", ) except TypeError: - # 老 provider 只支持 text 参数 try: - result = tts_service.synthesize(text=copy_text, voice_id=voice_id or "longxiaochun_v3") + result = tts_service.synthesize(text=text, voice_id=voice_id or "longxiaochun_v3") except TypeError: - result = tts_service.synthesize(text=copy_text) + result = tts_service.synthesize(text=text) if result is None: return None p = _Path(result) if not isinstance(result, _Path) else result - if p.exists(): + if p.exists() and p.stat().st_size > 0: logger.info( "[爆款视频] TTS 合成完成: voice=%s path=%s size=%d", voice_id or "longxiaochun_v3", p, p.stat().st_size ) return p - logger.warning("[爆款视频] TTS 返回路径不存在: %s", p) + logger.warning("[爆款视频] TTS 返回路径不存在或空文件: %s", p) return None except Exception as e: - logger.warning("[爆款视频] TTS 配音失败: %s", e) + logger.warning("[爆款视频] TTS 配音失败: %s", e, exc_info=True) return None -def _step_bgm_select(job: ViralVideoJob): - """步骤 7: BGM 选择。P1:素材未就绪前返回 None,跳过 BGM 混音。""" - return None +def _upload_tts_to_oss(job: ViralVideoJob, tts_path) -> str | None: + """把 TTS 本地 mp3 上传到 OSS,返回公网 URL(供 Seedance 做 reference_audios 口型驱动用)。""" + if tts_path is None: + return None + try: + from video_processing.oss_helpers import upload_to_oss + + local = Path(tts_path) if not isinstance(tts_path, Path) else tts_path + if not local.exists(): + return None + storage_key = f"generated/viral-video/{job.user_id}/{job.id}/tts_voiceover.mp3" + url = upload_to_oss(local, storage_key) + if url: + logger.info("[爆款视频] TTS 音频已上传 OSS: %s", url[:160]) + return url + except Exception as e: + logger.warning("[爆款视频] TTS 上传 OSS 失败: %s", e, exc_info=True) + return None -def _build_segment_prompt(seg: dict, job: ViralVideoJob, style_hint: str) -> str: - desc = seg.get("description") or seg.get("text") or "产品展示" - ken_burns = seg.get("ken_burns", "zoom_in") - cam_map = { - "zoom_in": "缓慢推镜放大", - "zoom_out": "缓慢拉镜缩小", - "pan_left": "镜头向左平移", - "pan_right": "镜头向右平移", - "static": "固定镜头", - } - camera = cam_map.get(ken_burns, "缓慢运镜") - parts = [ - f"{desc}。", - f"运镜:{camera}。", - "画面流畅、电影感光影、高清细节,9:16竖屏,适合短视频。", - ] - if style_hint: - parts.append(f"参考风格:{style_hint}") - return " ".join(parts) +def _assemble_seedance_prompt(copy_result: dict, job: ViralVideoJob) -> str: + """把编导脚本拼成 Seedance 长 prompt。""" + if not isinstance(copy_result, dict) or not copy_result: + return "产品展示短视频,清晰明亮,自然讲解" + ov = copy_result.get("overview") or {} + theme = ov.get("theme", "") + total_duration = ov.get("total_duration") or getattr(job, "duration", 15) + aspect_ratio = ov.get("aspect_ratio") or getattr(job, "video_ratio", "9:16") + scene_lighting = copy_result.get("scene_and_lighting", "") + shots = copy_result.get("shots") or [] + hc = copy_result.get("hard_constraints") or _DEFAULT_HARD_CONSTRAINTS + np = copy_result.get("negative_prompts") or _DEFAULT_NEGATIVE_PROMPTS + + lines: list[str] = [] + lines.append("【视频总览】") + lines.append(f"- 整体主题:{theme}") + lines.append(f"- 总时长:{total_duration}秒(单次生成,时长必须严格匹配)") + lines.append(f"- 画幅:{aspect_ratio}") + lines.append("") + lines.append("【场景与光线】") + lines.append(scene_lighting) + lines.append("") + lines.append("【逐镜头时间轴】(按时间顺序连贯拍摄,镜头之间自然衔接)") + for i, s in enumerate(shots): + if not isinstance(s, dict): + continue + tr = s.get("time_range", "") + cam = s.get("shot_type_angle_movement", "") + sd = s.get("scene_and_dialogue", "") + act = s.get("action_details", "") + ab = s.get("audio_bgm", "") + t = s.get("transition", "") + ref = s.get("reference_image_index") + lines.append(f"- 镜头{i+1}({tr}):") + lines.append(f" 景别/运镜:{cam}") + lines.append(f" 画面与对白:{sd}") + lines.append(f" 动作细节:{act}") + lines.append(f" 音效/BGM:{ab}") + lines.append(f" 转场:{t}") + if ref is not None and isinstance(ref, int): + lines.append(f" 参考图片:第{ref+1}张产品图") + lines.append("") + lines.append("【硬性约束】") + for c in hc: + lines.append(f"- {c}") + lines.append("") + lines.append("【负面提示词】(必须避免)") + lines.append(",".join([str(x) for x in np if x])) + return "\n".join(lines) -def _step_render(job, storyboard, tts_path, bgm): - """步骤 8: 渲染(P0-1 核心重写)。 - - 每个 storyboard 分镜 → Seedance 2.5 生成短视频段(无声)→ 下载 → ffmpeg concat → 混入 TTS。 - 返回最终视频本地路径字符串。 - """ - import tempfile - from pathlib import Path - - from video_processing.concat_engine import concat_video_files - +def _step_render(job: ViralVideoJob, copy_result: dict, tts_audio_url: str | None) -> str: + """步骤 6: v1.6 单次 Seedance 生成(不再分段/拼接)。""" from packages.shared.ai_service import call_video_generation - from packages.shared.ffmpeg_utils import run_ffmpeg - if not storyboard: - raise ValueError("storyboard is empty") + prompt = _assemble_seedance_prompt(copy_result, job) + dur = max(5, min(30, int(getattr(job, "duration", 15) or 15))) + ratio = getattr(job, "video_ratio", None) or "9:16" + model = getattr(job, "video_model", "") or None - style_hint = "" - if isinstance(job.style_guide, dict): - style_hint = f"节奏{job.style_guide.get('cut_speed','')}、转场{job.style_guide.get('transition','')}、色调{job.style_guide.get('color_grade','')}" + # reference_audios: TTS 音频驱动口型 + ref_audios = [tts_audio_url] if tts_audio_url else [] + # reference_images: 产品图(除首帧外的其他图作为多参考;首帧通过 image_url 传) + images = list(job.images or []) + first_image = images[0] if images else None + rest_images = images[1:30] if len(images) > 1 else [] + # reference_videos: 参考视频(可选) + ref_videos = [job.reference_video_url] if getattr(job, "reference_video_url", "") else [] tmpdir = Path(tempfile.mkdtemp(prefix=f"viral_{job.id}_")) - logger.info("[爆款视频] 开始渲染,分镜数=%d, tmpdir=%s", len(storyboard), tmpdir) + logger.info( + "[爆款视频] 开始单次 Seedance 生成 dur=%ds ratio=%s model=%s ref_imgs=%d ref_audios=%d ref_videos=%d tmpdir=%s", + dur, + ratio if not first_image else "(follow-image)", + model or "default", + len(rest_images) + (1 if first_image else 0), + len(ref_audios), + len(ref_videos), + tmpdir, + ) + logger.info("[爆款视频] Seedance prompt (前300字): %s", prompt[:300]) - seg_paths: list[str] = [] - first_image = job.images[0] if job.images else None - n_total = len(storyboard) - for i, seg in enumerate(storyboard): - try: - dur = int(seg.get("duration") or 5) - except (TypeError, ValueError): - dur = 5 - dur = max(2, min(12, dur)) - prompt = _build_segment_prompt(seg, job, style_hint) - _emit_progress( - job.id, - ViralVideoStage.RENDERING, - 80.0 + (i + 1) / max(n_total, 1) * 5.0, - f"正在生成分镜 {i + 1}/{n_total} ({dur}s)...", - ) - logger.info("[爆款视频] 分镜 %d/%d dur=%ds prompt=%s", i + 1, n_total, dur, prompt[:80]) - seg_path = call_video_generation( - prompt=prompt, - image_url=first_image if i == 0 else None, - duration=dur, - ratio=(getattr(job, "video_ratio", None) or "9:16"), - resolution="720p", - output_dir=str(tmpdir), - model=getattr(job, "video_model", "") or None, - ) - if not seg_path or not Path(seg_path).exists(): - logger.warning("[爆款视频] 分镜 %d 生成失败,使用占位片段", i + 1) - seg_path = str(_make_placeholder_clip(tmpdir, i, dur)) - seg_paths.append(seg_path) - - _emit_progress(job.id, ViralVideoStage.RENDERING, 86.0, "正在拼接分镜...") - concat_out = tmpdir / "concat_raw.mp4" - try: - concat_video_files(seg_paths, concat_out, work_dir=tmpdir, force_reencode=True) - except Exception as e: - logger.error("[爆款视频] concat 失败: %s,降级过滤无效片段", e, exc_info=True) - valid = [p for p in seg_paths if _probe_ok(p)] - if not valid: - raise RuntimeError(f"所有分镜片段均无效: {e}") from e - concat_video_files(valid, concat_out, work_dir=tmpdir, force_reencode=True) - - final_path = concat_out - - if tts_path is not None: - tts_p = Path(tts_path) if not isinstance(tts_path, Path) else tts_path - if tts_p.exists(): - _emit_progress(job.id, ViralVideoStage.RENDERING, 87.5, "正在合成配音...") - mixed_out = tmpdir / "final_with_audio.mp4" - try: - run_ffmpeg( - [ - "ffmpeg", - "-y", - "-i", - str(concat_out), - "-i", - str(tts_p), - "-c:v", - "copy", - "-c:a", - "aac", - "-b:a", - "192k", - "-map", - "0:v:0", - "-map", - "1:a:0", - "-shortest", - str(mixed_out), - ] - ) - if mixed_out.exists() and mixed_out.stat().st_size > 0: - final_path = mixed_out - except Exception as e: - logger.warning("[爆款视频] TTS 混音失败,使用无声视频: %s", e) - - logger.info("[爆款视频] 渲染完成: %s size=%d", final_path, final_path.stat().st_size if final_path.exists() else 0) - return str(final_path) - - -def _probe_ok(video_path: str) -> bool: - import subprocess - from pathlib import Path as _Path - - try: - if not _Path(video_path).exists(): - return False - r = subprocess.run( - [ - "ffprobe", - "-v", - "error", - "-select_streams", - "v:0", - "-show_entries", - "stream=codec_type", - "-of", - "csv=p=0", - video_path, - ], - capture_output=True, - timeout=10, - ) - return r.returncode == 0 and b"video" in r.stdout - except Exception: - return False - - -def _make_placeholder_clip(tmpdir, idx: int, duration: int): - import subprocess - - out = tmpdir / f"placeholder_{idx}.mp4" - try: - subprocess.run( - [ - "ffmpeg", - "-y", - "-f", - "lavfi", - "-i", - f"color=c=0x202030:s=720x1280:d={max(duration,2)}:r=24", - "-f", - "lavfi", - "-i", - f"anullsrc=r=44100:cl=stereo:d={max(duration,2)}", - "-c:v", - "libx264", - "-pix_fmt", - "yuv420p", - "-preset", - "ultrafast", - "-c:a", - "aac", - "-shortest", - str(out), - ], - capture_output=True, - timeout=60, - check=True, - ) - except Exception as e: - logger.warning("[爆款视频] 占位片段生成失败: %s", e) - return out + video_path = call_video_generation( + prompt=prompt, + image_url=first_image, + duration=dur, + ratio=ratio, + resolution="720p", + output_dir=str(tmpdir), + model=model, + generate_audio=True, # Seedance 原生生成环境音效/BGM;口型由 reference_audios 的 TTS 驱动 + reference_images=rest_images, + reference_audios=ref_audios, + reference_videos=ref_videos, + ) + if not video_path or not Path(video_path).exists() or Path(video_path).stat().st_size == 0: + raise RuntimeError("Seedance 视频生成失败:返回空文件或路径不存在") + logger.info("[爆款视频] Seedance 单次生成完成: %s size=%d", video_path, Path(video_path).stat().st_size) + return str(video_path) def _step_upload(job: ViralVideoJob, video_path: str) -> str: - """步骤 9: OSS 上传。""" - from pathlib import Path - + """步骤 7: OSS 上传。""" from video_processing.oss_helpers import upload_to_oss local = Path(video_path) - # 构造 OSS key,与 generation.py 规则对齐:generated/viral-video/// storage_key = f"generated/viral-video/{job.user_id}/{job.id}/{local.name}" logger.info("[爆款视频] 开始上传成片: local=%s key=%s size=%d", local, storage_key, local.stat().st_size) video_url = upload_to_oss(local, storage_key) @@ -731,42 +824,31 @@ def _step_upload(job: ViralVideoJob, video_path: str) -> str: @shared_task(bind=True, max_retries=2, name="worker.run_viral_video_pipeline") def run_viral_video_pipeline(self: Task, job_id: str) -> dict: - """爆款视频 10 步流水线编排器(前半段:图片分析→风格分析→意图解析,然后 WAIT_USER_CONFIRM)。""" + """旧一键流水线(保留兼容):图片分析→风格分析→意图解析→WAIT_USER_CONFIRM。""" session = None try: session, repo, job = _get_repo_and_job(job_id) if job is None: - logger.error("[爆款视频] 任务不存在: %s", job_id) return {"ok": False, "error": "job not found"} job.mark_running() _save_job(repo, job, session) _emit_progress(job_id, ViralVideoStage.IMAGE_ANALYSIS, 5.0, "开始图片分析") - # ── Step 1: 图片 VLM 分析 ── _emit_progress(job_id, ViralVideoStage.IMAGE_ANALYSIS, 10.0, "正在分析产品图片...") image_analysis = _step_image_analysis(job) - # P0-3: 持久化 image_analysis 到 job,供 resume 阶段使用 job.image_analysis = image_analysis _save_job(repo, job, session) _emit_progress(job_id, ViralVideoStage.IMAGE_ANALYSIS, 15.0, "图片分析完成", {"result": image_analysis}) - # ── Step 1.5: 视频风格分析(v1.3) ── style_guide = None if job.reference_video_url or job.style_template_id: _emit_progress(job_id, ViralVideoStage.VIDEO_ANALYSIS, 20.0, "正在分析参考视频风格...") style_guide = _step_video_analysis(job) job.style_guide = style_guide _save_job(repo, job, session) - _emit_progress( - job_id, - ViralVideoStage.VIDEO_ANALYSIS, - 25.0, - "风格分析完成", - {"style_analyzed": True, "style_guide": style_guide}, - ) + _emit_progress(job_id, ViralVideoStage.VIDEO_ANALYSIS, 25.0, "风格分析完成", {"style_guide": style_guide}) - # ── Step 2: 意图解析 ── _emit_progress(job_id, ViralVideoStage.INTENT_PARSING, 30.0, "正在解析文案意图...") intent_result = _step_intent_parsing(job, image_analysis) @@ -787,37 +869,14 @@ def run_viral_video_pipeline(self: Task, job_id: str) -> dict: {"intent_result": intent_result}, event_type="viral_video:wait_user", ) - return {"ok": True, "job_id": job_id, "status": "wait_user_confirm", "intent_result": intent_result} except Retry: raise except Exception as e: logger.error("[爆款视频] 流水线异常: %s", e, exc_info=True) - err_msg = str(e) - failed_stage = "" - try: - if session is None: - session = SessionLocal() - repo = SQLAlchemyViralVideoJobRepository(session) - job = repo.get(job_id) - else: - _, repo, job = _get_repo_and_job(job_id) - if job is not None and not job.is_terminal: - job.mark_failed(err_msg) - failed_stage = getattr(job, "current_stage", "") or "" - _save_job(repo, job, session) - except Exception as inner: - logger.warning("[爆款视频] 标记失败状态时出错: %s", inner) - _emit_progress( - job_id, - failed_stage, - 0, - f"任务失败: {err_msg}", - {"error": err_msg}, - event_type="viral_video:failed", - ) - return {"ok": False, "job_id": job_id, "error": err_msg} + _mark_failed_and_notify(job_id, session, None, None, str(e), "") + return {"ok": False, "job_id": job_id, "error": str(e)} finally: if session: session.close() @@ -825,47 +884,21 @@ def run_viral_video_pipeline(self: Task, job_id: str) -> dict: @shared_task(bind=True, max_retries=2, name="worker.resume_viral_video_pipeline") def resume_viral_video_pipeline(self: Task, job_id: str) -> dict: - """用户确认意图后,从断点恢复流水线(步骤 3-10)。""" + """旧 confirm-intent 路径兼容:从 WAIT_USER_CONFIRM 跑完整个渲染。""" session = None - job = None try: session, repo, job = _get_repo_and_job(job_id) if job is None: return {"ok": False, "error": "job not found"} - if job.status != ViralVideoStatus.RUNNING: return {"ok": False, "error": f"unexpected status: {job.status}"} - - # 旧 confirm-intent 路径:v1.4 及之前 storyboard/copy_text 未持久化,交给 _run_render_pipeline 兜底重算; - # 新 v1.5 路径(copy_generated -> run_viral_video_render)直接走新 task,不会进入这里。 return _run_render_pipeline(job_id, session, repo, job) - except Retry: raise except Exception as e: logger.error("[爆款视频] 恢复流水线异常: %s", e, exc_info=True) - err_msg = str(e) - failed_stage = "" - try: - if session is None: - session = SessionLocal() - repo = SQLAlchemyViralVideoJobRepository(session) - job = repo.get(job_id) - elif job is not None and not job.is_terminal: - job.mark_failed(err_msg) - failed_stage = getattr(job, "current_stage", "") or "" - _save_job(repo, job, session) - except Exception as inner: - logger.warning("[爆款视频] 标记失败状态时出错: %s", inner) - _emit_progress( - job_id, - failed_stage, - 0, - f"任务失败: {err_msg}", - {"error": err_msg}, - event_type="viral_video:failed", - ) - return {"ok": False, "job_id": job_id, "error": err_msg} + _mark_failed_and_notify(job_id, session, None, None, str(e), "") + return {"ok": False, "job_id": job_id, "error": str(e)} finally: if session: session.close() @@ -873,21 +906,18 @@ def resume_viral_video_pipeline(self: Task, job_id: str) -> dict: @shared_task(bind=True, max_retries=1, name="worker.run_video_style_analysis") def run_video_style_analysis(self: Task, job_id: str) -> dict: - """独立的视频风格分析任务(v1.3)。""" + """独立的视频风格分析任务。""" session = None try: session, repo, job = _get_repo_and_job(job_id) if job is None: return {"ok": False, "error": "job not found"} - _emit_progress(job_id, ViralVideoStage.VIDEO_ANALYSIS, 10.0, "正在分析参考视频风格...") style_guide = _step_video_analysis(job) job.style_guide = style_guide _save_job(repo, job, session) - _emit_progress(job_id, ViralVideoStage.VIDEO_ANALYSIS, 100.0, "风格分析完成", {"style_guide": style_guide}) return {"ok": True, "job_id": job_id, "style_guide": style_guide} - except Retry: raise except Exception as e: @@ -898,11 +928,10 @@ def run_video_style_analysis(self: Task, job_id: str) -> dict: session.close() -# ── v1.5 三步分步流水线 Celery 任务 ────────────────────────────────────── +# ── 失败处理 ──────────────────────────────────────────────────────────── def _mark_failed_and_notify(job_id: str, session, repo, job, err_msg: str, stage: str = "") -> None: - """统一的失败处理:标记 FAILED + 发 failed WS 事件。""" try: if session is None: session = SessionLocal() @@ -923,14 +952,16 @@ def _mark_failed_and_notify(job_id: str, session, repo, job, err_msg: str, stage ) +# ── v1.5/v1.6 三步分步流水线 Celery 任务 ───────────────────────────────── + + @shared_task(bind=True, max_retries=1, name="worker.run_viral_video_analyze") def run_viral_video_analyze(self: Task, job_id: str) -> dict: - """v1.5 阶段1:仅跑图片 VLM 分析(+ 可选视频风格分析),完成后状态=image_analyzed。""" + """v1.5+ 阶段1:图片 VLM 分析 + 可选视频风格分析。""" session = None try: session, repo, job = _get_repo_and_job(job_id) if job is None: - logger.error("[爆款视频][阶段1] 任务不存在: %s", job_id) return {"ok": False, "error": "job not found"} job.mark_running() @@ -963,11 +994,10 @@ def run_viral_video_analyze(self: Task, job_id: str) -> dict: job_id, ViralVideoStage.IMAGE_ANALYSIS, 100.0, - "图片分析完成,请填写营销参数以生成文案", + "图片分析完成,请填写营销参数以生成编导脚本", {"image_analysis": image_analysis, "status": "image_analyzed"}, event_type="viral_video:image_analyzed", ) - logger.info("[爆款视频][阶段1] 图片分析完成 job_id=%s", job_id) return {"ok": True, "job_id": job_id, "status": "image_analyzed", "image_analysis": image_analysis} except Retry: @@ -983,77 +1013,77 @@ def run_viral_video_analyze(self: Task, job_id: str) -> dict: @shared_task(bind=True, max_retries=1, name="worker.run_viral_video_generate_copy") def run_viral_video_generate_copy(self: Task, job_id: str) -> dict: - """v1.5 阶段2:跑 意图解析 → 文案融合 → 分镜 → 合规审核,完成后状态=copy_generated。 - - 入参要求:调用方已 resume_from_image_analyzed() 把状态切到 RUNNING,并把用户填的营销参数写到 job 上。 - """ + """v1.6 阶段2:意图解析 → 编导分镜脚本生成 → 合规审核,完成后状态=copy_generated。""" session = None try: session, repo, job = _get_repo_and_job(job_id) if job is None: return {"ok": False, "error": "job not found"} - if job.status != ViralVideoStatus.RUNNING: return {"ok": False, "error": f"unexpected status: {job.status}"} image_analysis = job.image_analysis or {"products": []} - # Step 2: 意图解析 _emit_progress(job_id, ViralVideoStage.INTENT_PARSING, 20.0, "正在解析文案意图...") intent_result = _step_intent_parsing(job, image_analysis) job.intent_result = intent_result _save_job(repo, job, session) _emit_progress(job_id, ViralVideoStage.INTENT_PARSING, 35.0, "意图解析完成") - # Step 3: 文案融合 - _emit_progress(job_id, ViralVideoStage.COPY_FUSION, 40.0, "正在融合文案...") - copy_text = _step_copy_fusion(job, intent_result, image_analysis) - _emit_progress(job_id, ViralVideoStage.COPY_FUSION, 50.0, "文案融合完成") + _emit_progress(job_id, ViralVideoStage.SCRIPT_GENERATION, 40.0, "正在生成编导分镜脚本...") + copy_result = _step_script_generation(job, intent_result, image_analysis) + _emit_progress( + job_id, + ViralVideoStage.SCRIPT_GENERATION, + 60.0, + "编导脚本生成完成", + {"shots": len(copy_result.get("shots", []))}, + ) - # Step 4: 分镜 - _emit_progress(job_id, ViralVideoStage.STORYBOARD, 55.0, "正在生成分镜脚本...") - storyboard = _step_storyboard(job, copy_text, image_analysis) - _emit_progress(job_id, ViralVideoStage.STORYBOARD, 60.0, "分镜脚本完成", {"segments": len(storyboard)}) - - # Step 5: 合规审核 _emit_progress(job_id, ViralVideoStage.REVIEW, 65.0, "正在进行合规审核...") - review_result = _step_review(job, copy_text, storyboard) + review_result = _step_review(job, copy_result) if not review_result.get("passed", True): _emit_progress(job_id, ViralVideoStage.REVIEW, 67.0, "审核未通过,正在自动重写...") - copy_text = _step_copy_fusion(job, intent_result, image_analysis) - _step_review(job, copy_text, storyboard) + copy_result = _step_script_generation(job, intent_result, image_analysis) + _step_review(job, copy_result) _emit_progress(job_id, ViralVideoStage.REVIEW, 70.0, "合规审核完成") - job.mark_copy_generated(copy_text, storyboard) + job.mark_copy_generated(copy_result) _save_job(repo, job, session) + voiceover = copy_result.get("voiceover_script", "") _emit_progress( job_id, ViralVideoStage.REVIEW, 100.0, - "文案与分镜已生成,请确认或编辑文案", + "编导分镜脚本已生成,请确认或编辑口播文案", { - "generated_copy_text": copy_text, - "storyboard": storyboard, + "copy_result": copy_result, + "generated_copy_text": voiceover, + "storyboard": copy_result.get("shots", []), "status": "copy_generated", }, event_type="viral_video:copy_generated", ) logger.info( - "[爆款视频][阶段2] 文案+分镜生成完成 job_id=%s copy_len=%d segs=%d", job_id, len(copy_text), len(storyboard) + "[爆款视频][阶段2] 编导脚本生成完成 job_id=%s voiceover_len=%d shots=%d", + job_id, + len(voiceover), + len(copy_result.get("shots", [])), ) return { "ok": True, "job_id": job_id, "status": "copy_generated", - "generated_copy_text": copy_text, - "storyboard": storyboard, + "copy_result": copy_result, + "generated_copy_text": voiceover, + "storyboard": copy_result.get("shots", []), } except Retry: raise except Exception as e: logger.error("[爆款视频][阶段2] 异常: %s", e, exc_info=True) - _mark_failed_and_notify(job_id, session, None, None, str(e), ViralVideoStage.COPY_FUSION) + _mark_failed_and_notify(job_id, session, None, None, str(e), ViralVideoStage.SCRIPT_GENERATION) return {"ok": False, "job_id": job_id, "error": str(e)} finally: if session: @@ -1061,46 +1091,33 @@ def run_viral_video_generate_copy(self: Task, job_id: str) -> dict: def _run_render_pipeline(job_id: str, session, repo, job) -> dict: - """v1.5 阶段3 / 旧 resume 共用:TTS → BGM → Render → Upload → Completed。 - - 入参要求:job.status == RUNNING,job.generated_copy_text 或 job.user_copy_text 非空,job.storyboard 已就绪。 - """ + """v1.6 阶段3 / 旧 resume 共用:TTS → 单次 Seedance → Upload → Completed。""" image_analysis = job.image_analysis or {"products": []} - copy_text = job.effective_copy_text - storyboard = job.storyboard or [] - # 兼容旧路径:老 resume_viral_video_pipeline 在 RUNNING 时可能还没 storyboard(v1.4 及之前 job.storyboard 没持久化), - # 这种情况下用 intent_result + image_analysis 现算 copy_text + storyboard。 - if not storyboard: - _emit_progress(job_id, ViralVideoStage.COPY_FUSION, 40.0, "正在融合文案...") - copy_text = _step_copy_fusion(job, job.intent_result or {}, image_analysis) - _emit_progress(job_id, ViralVideoStage.COPY_FUSION, 50.0, "文案融合完成") - _emit_progress(job_id, ViralVideoStage.STORYBOARD, 55.0, "正在生成分镜脚本...") - storyboard = _step_storyboard(job, copy_text, image_analysis) - _emit_progress(job_id, ViralVideoStage.STORYBOARD, 60.0, "分镜脚本完成", {"segments": len(storyboard)}) - _emit_progress(job_id, ViralVideoStage.REVIEW, 65.0, "正在进行合规审核...") - _step_review(job, copy_text, storyboard) - _emit_progress(job_id, ViralVideoStage.REVIEW, 70.0, "合规审核完成") - # 补持久化 - job.storyboard = storyboard - job.generated_copy_text = copy_text + # 如果没有 copy_result(旧数据/失败重试),现场补生成 + copy_result = job.copy_result + if not isinstance(copy_result, dict) or not copy_result: + _emit_progress(job_id, ViralVideoStage.SCRIPT_GENERATION, 40.0, "正在补生成编导脚本...") + intent = job.intent_result or _step_intent_parsing(job, image_analysis) + copy_result = _step_script_generation(job, intent, image_analysis) + _step_review(job, copy_result) + job.mark_copy_generated(copy_result) _save_job(repo, job, session) - # Step 6: TTS - _emit_progress(job_id, ViralVideoStage.TTS, 72.0, "正在生成配音...") - tts_path = _step_tts(job, copy_text) - _emit_progress(job_id, ViralVideoStage.TTS, 75.0, "配音完成", {"has_tts": tts_path is not None}) + voiceover = copy_result.get("voiceover_script", "") or job.effective_copy_text - # Step 7: BGM (skip) - _emit_progress(job_id, ViralVideoStage.BGM_SELECT, 77.0, "BGM 已跳过(素材未就绪)") - bgm = _step_bgm_select(job) + # Step 5: TTS 整段合成 + _emit_progress(job_id, ViralVideoStage.TTS, 72.0, "正在生成AI配音...") + tts_path = _step_tts(job, voiceover) + tts_url = _upload_tts_to_oss(job, tts_path) + _emit_progress(job_id, ViralVideoStage.TTS, 78.0, "配音完成", {"has_tts": tts_url is not None}) - # Step 8: Render - _emit_progress(job_id, ViralVideoStage.RENDERING, 80.0, "正在渲染视频...") - video_path = _step_render(job, storyboard, tts_path, bgm) - _emit_progress(job_id, ViralVideoStage.RENDERING, 88.0, "渲染完成") + # Step 6: 单次 Seedance + _emit_progress(job_id, ViralVideoStage.RENDERING, 80.0, "正在调用AI生成视频(约1-3分钟)...") + video_path = _step_render(job, copy_result, tts_url) + _emit_progress(job_id, ViralVideoStage.RENDERING, 92.0, "视频生成完成") - # Step 9: Upload + # Step 7: Upload _emit_progress(job_id, ViralVideoStage.UPLOADING, 95.0, "正在上传视频...") video_url = _step_upload(job, video_path) @@ -1122,7 +1139,7 @@ def _run_render_pipeline(job_id: str, session, repo, job) -> dict: @shared_task(bind=True, max_retries=2, name="worker.run_viral_video_render") def run_viral_video_render(self: Task, job_id: str) -> dict: - """v1.5 阶段3:用户确认/编辑文案后,跑 TTS+BGM+Render+Upload 直到完成。""" + """v1.6 阶段3:TTS + 单次 Seedance 生成 + 上传。""" session = None try: session, repo, job = _get_repo_and_job(job_id) diff --git a/packages/adapters/sqlalchemy_impl/models.py b/packages/adapters/sqlalchemy_impl/models.py index 4d17bc697..7867fa7e6 100755 --- a/packages/adapters/sqlalchemy_impl/models.py +++ b/packages/adapters/sqlalchemy_impl/models.py @@ -954,6 +954,9 @@ class ViralVideoJobModel(Base): image_analysis = Column(JSON, nullable=True) storyboard = Column(JSON, nullable=True) generated_copy_text = Column(Text, nullable=False, default="") + copy_result = Column( + JSON, nullable=True + ) # v1.6: 编导脚本结构{overview,scene_and_lighting,shots,hard_constraints,negative_prompts,voiceover_script} result_video_url = Column(String(1000), nullable=False, default="") credits_cost = Column(Integer, nullable=False, default=0) error_msg = Column(Text, nullable=False, default="") diff --git a/packages/adapters/sqlalchemy_impl/session.py b/packages/adapters/sqlalchemy_impl/session.py index e67ff22ed..d457abce7 100644 --- a/packages/adapters/sqlalchemy_impl/session.py +++ b/packages/adapters/sqlalchemy_impl/session.py @@ -88,6 +88,7 @@ _VIRAL_VIDEO_BACKFILL_COLS = [ ("voice_source", "VARCHAR(20) NOT NULL DEFAULT ''"), ("video_ratio", "VARCHAR(10) NOT NULL DEFAULT '9:16'"), ("video_model", "VARCHAR(100) NOT NULL DEFAULT ''"), + ("copy_result", "JSON"), ] diff --git a/packages/adapters/sqlalchemy_impl/viral_video_repository.py b/packages/adapters/sqlalchemy_impl/viral_video_repository.py index 36d7935fe..7143985a3 100755 --- a/packages/adapters/sqlalchemy_impl/viral_video_repository.py +++ b/packages/adapters/sqlalchemy_impl/viral_video_repository.py @@ -24,7 +24,7 @@ def _to_domain(model: ViralVideoJobModel) -> ViralVideoJob: viral_structure=model.viral_structure or "", marketing_purpose=model.marketing_purpose or "", bgm_preference=model.bgm_preference or "", - duration=model.duration or 30, + duration=model.duration or 15, user_copy_text=model.user_copy_text or "", fusion_level=model.fusion_level or "ai_polish", reference_audio_path=model.reference_audio_path or "", @@ -41,6 +41,7 @@ def _to_domain(model: ViralVideoJobModel) -> ViralVideoJob: image_analysis=dict(model.image_analysis) if getattr(model, "image_analysis", None) else None, storyboard=list(model.storyboard) if getattr(model, "storyboard", None) else None, generated_copy_text=getattr(model, "generated_copy_text", "") or "", + copy_result=dict(model.copy_result) if getattr(model, "copy_result", None) else None, result_video_url=model.result_video_url or "", credits_cost=model.credits_cost or 0, error_msg=model.error_msg or "", @@ -86,6 +87,7 @@ class SQLAlchemyViralVideoJobRepository: image_analysis=job.image_analysis, storyboard=job.storyboard, generated_copy_text=job.generated_copy_text, + copy_result=job.copy_result, result_video_url=job.result_video_url, credits_cost=job.credits_cost, error_msg=job.error_msg, @@ -108,6 +110,7 @@ class SQLAlchemyViralVideoJobRepository: model.image_analysis = job.image_analysis model.storyboard = job.storyboard model.generated_copy_text = job.generated_copy_text or "" + model.copy_result = job.copy_result model.result_video_url = job.result_video_url model.credits_cost = job.credits_cost model.error_msg = job.error_msg diff --git a/packages/domain/viral_video.py b/packages/domain/viral_video.py index af6708121..b3c74ee5e 100755 --- a/packages/domain/viral_video.py +++ b/packages/domain/viral_video.py @@ -1,6 +1,7 @@ """ViralVideoJob 领域模型 — 爆款视频任务. -状态机(v1.5 三步分步): +v1.6 重大简化:Seedance 2.5 单次最长30秒,单次调用直接出片,不再分段/拼接/ffmpeg concat。 +状态机(三步分步): pending -> running -> image_analyzed -> running -> copy_generated -> running -> completed wait_user_confirm -> running -> completed (旧路径兼容) 任意阶段 fail; 任意非终态 cancel. @@ -26,8 +27,6 @@ from uuid import uuid4 class ViralVideoStatus(StrEnum): - """爆款视频任务状态枚举。""" - PENDING = "pending" RUNNING = "running" IMAGE_ANALYZED = "image_analyzed" @@ -39,44 +38,32 @@ class ViralVideoStatus(StrEnum): class ViralVideoStage(StrEnum): - """编排流水线阶段枚举(用于 WS 进度推送)。""" - IMAGE_ANALYSIS = "image_analysis" VIDEO_ANALYSIS = "video_analysis" INTENT_PARSING = "intent_parsing" - COPY_FUSION = "copy_fusion" - STORYBOARD = "storyboard" + SCRIPT_GENERATION = "script_generation" # v1.6: 编导分镜脚本(融合原 copy_fusion+storyboard+review) REVIEW = "review" TTS = "tts" - BGM_SELECT = "bgm_select" - RENDERING = "rendering" - MUSETALK = "musetalk" + RENDERING = "rendering" # v1.6: 单次 Seedance 生成(BGM/音效/画面一次出片) UPLOADING = "uploading" class FusionLevel(StrEnum): - """文案融合级别。""" - AI_FULL = "ai_full" AI_POLISH = "ai_polish" USER_PRIMARY = "user_primary" class StyleStrength(StrEnum): - """风格强度。""" - LIGHT = "light" MEDIUM = "medium" STRICT = "strict" class PromptType(StrEnum): - """Prompt 模板类型(与 #2040 seed 对齐)。""" - IMAGE_ANALYSIS = "image_analysis" INTENT_PARSING = "intent_parsing" - COPY_FUSION = "copy_fusion" - STORYBOARD = "storyboard" + SCRIPT_GENERATION = "script_generation" REVIEW = "review" VIDEO_STYLE_INTEGRATION = "video_style_integration" STYLE_CONSTRAINT = "style_constraint" @@ -88,20 +75,17 @@ STAGE_LABELS = { ViralVideoStage.IMAGE_ANALYSIS: "图片分析", ViralVideoStage.VIDEO_ANALYSIS: "视频风格分析", ViralVideoStage.INTENT_PARSING: "意图解析", - ViralVideoStage.COPY_FUSION: "文案融合", - ViralVideoStage.STORYBOARD: "分镜脚本", + ViralVideoStage.SCRIPT_GENERATION: "编导脚本生成", ViralVideoStage.REVIEW: "合规审核", ViralVideoStage.TTS: "AI 配音", - ViralVideoStage.BGM_SELECT: "BGM 选择", - ViralVideoStage.RENDERING: "视频渲染", - ViralVideoStage.MUSETALK: "数字人口型", + ViralVideoStage.RENDERING: "视频生成", ViralVideoStage.UPLOADING: "上传发布", } @dataclass class ViralVideoJob: - """爆款视频任务领域实体。""" + """爆款视频任务领域实体(v1.6 单次 Seedance 出片版)。""" user_id: str images: list[str] = field(default_factory=list) @@ -111,29 +95,28 @@ class ViralVideoJob: viral_structure: str = "" marketing_purpose: str = "" bgm_preference: str = "" - duration: int = 30 + duration: int = 15 # v1.6: 默认15秒,上限30秒(Seedance 2.5 单次最大30s) user_copy_text: str = "" fusion_level: str = FusionLevel.AI_POLISH reference_audio_path: str = "" - # v1.3 reference_video_url: str = "" style_strength: str = StyleStrength.MEDIUM style_guide: dict | None = None style_template_id: str = "" - # v1.5 音频/视频参数 - voice_id: str = "" # TTS 音色 ID(CosyVoice voice_id);为空则用 persona_id 兜底 - voice_source: str = "" # preset/library/clone/upload - video_ratio: str = "9:16" # Seedance 视频比例:9:16 / 16:9 / 1:1 / etc. - video_model: str = "" # Seedance 模型 ID;空则用 settings.doubao_video_model 默认值 - # v1.4 图片分析结果(run_pipeline 持久化,resume 时读取给文案/分镜) + # v1.5.1 音频/视频参数 + voice_id: str = "" + voice_source: str = "" + video_ratio: str = "9:16" + video_model: str = "" + # v1.4+ 产物 image_analysis: dict | None = None - # v1.5 三步分步流水线产物(持久化,供前端 GET 读取 + resume 消费) - storyboard: list | None = None - generated_copy_text: str = "" + intent_result: dict | None = None + generated_copy_text: str = "" # v1.6: 存 voiceover_script(纯口播对白),字段名兼容 + storyboard: list | None = None # v1.6: 存 copy_result.shots,字段名兼容 + copy_result: dict | None = None # v1.6: 完整编导脚本结构 # 状态 id: str = field(default_factory=lambda: uuid4().hex) status: ViralVideoStatus = ViralVideoStatus.PENDING - intent_result: dict | None = None result_video_url: str = "" credits_cost: int = 0 error_msg: str = "" @@ -160,7 +143,6 @@ class ViralVideoJob: self.updated_at = datetime.now(timezone.utc) def mark_image_analyzed(self) -> None: - """阶段1完成:图片/视频分析完成,等待用户填参数或直接触发阶段2。""" if self.status not in (ViralVideoStatus.PENDING, ViralVideoStatus.RUNNING): raise ValueError(f"Cannot transition from {self.status} to image_analyzed") self.status = ViralVideoStatus.IMAGE_ANALYZED @@ -168,8 +150,8 @@ class ViralVideoJob: self.started_at = datetime.now(timezone.utc) self.updated_at = datetime.now(timezone.utc) - def mark_copy_generated(self, copy_text: str, storyboard: list) -> None: - """阶段2完成:文案+分镜+审核完成,等待用户编辑后确认。""" + def mark_copy_generated(self, copy_result: dict) -> None: + """v1.6 阶段2完成:编导脚本(含 voiceover_script/shots/硬约束/负面词)已生成。""" if self.status not in ( ViralVideoStatus.IMAGE_ANALYZED, ViralVideoStatus.RUNNING, @@ -177,12 +159,14 @@ class ViralVideoJob: ): raise ValueError(f"Cannot transition from {self.status} to copy_generated") self.status = ViralVideoStatus.COPY_GENERATED - self.generated_copy_text = copy_text or "" - self.storyboard = list(storyboard) if storyboard else [] + self.copy_result = copy_result or {} + if isinstance(copy_result, dict): + self.generated_copy_text = copy_result.get("voiceover_script", "") or "" + shots = copy_result.get("shots") or [] + self.storyboard = list(shots) if isinstance(shots, list) else [] self.updated_at = datetime.now(timezone.utc) def mark_wait_user_confirm(self, intent_result: dict) -> None: - """旧流水线兼容:意图解析完成等待用户确认(老接口)。""" if self.status != ViralVideoStatus.RUNNING: raise ValueError(f"Cannot transition from {self.status} to wait_user_confirm") self.status = ViralVideoStatus.WAIT_USER_CONFIRM @@ -190,7 +174,6 @@ class ViralVideoJob: self.updated_at = datetime.now(timezone.utc) def resume_from_image_analyzed(self, **kwargs) -> None: - """阶段1->阶段2:用户已填参数,开始跑文案/分镜。kwargs 覆盖参数字段。""" if self.status not in (ViralVideoStatus.IMAGE_ANALYZED, ViralVideoStatus.PENDING): raise ValueError(f"Cannot resume from {self.status} to copy-gen") for k, v in kwargs.items(): @@ -200,17 +183,16 @@ class ViralVideoJob: self.updated_at = datetime.now(timezone.utc) def resume_from_copy_generated(self, edited_copy: str | None = None) -> None: - """阶段2->阶段3:用户确认/编辑文案,开始跑渲染。""" + """阶段2->阶段3:用户确认/编辑口播文案,开始跑 TTS+单次Seedance渲染。""" if self.status != ViralVideoStatus.COPY_GENERATED: raise ValueError(f"Cannot resume from {self.status} to render") - if edited_copy: - self.user_copy_text = edited_copy + if edited_copy and isinstance(self.copy_result, dict): + self.copy_result = {**self.copy_result, "voiceover_script": edited_copy} self.generated_copy_text = edited_copy self.status = ViralVideoStatus.RUNNING self.updated_at = datetime.now(timezone.utc) def resume_from_confirm(self) -> None: - """旧流水线兼容:从 WAIT_USER_CONFIRM 恢复。""" if self.status != ViralVideoStatus.WAIT_USER_CONFIRM: raise ValueError(f"Cannot resume from {self.status}") self.status = ViralVideoStatus.RUNNING @@ -245,5 +227,11 @@ class ViralVideoJob: @property def effective_copy_text(self) -> str: - """渲染阶段使用的最终文案:用户编辑 > 生成文案 > 用户原始 > 占位。""" - return self.user_copy_text or self.generated_copy_text or "精选好物推荐" + """TTS 用的最终口播文案:优先 copy_result.voiceover_script,兼容老字段。""" + if isinstance(self.copy_result, dict) and self.copy_result.get("voiceover_script"): + return self.copy_result["voiceover_script"] + return self.generated_copy_text or self.user_copy_text or "你好,给大家推荐一款好物" + + @property + def voiceover_script(self) -> str: + return self.effective_copy_text diff --git a/packages/shared/ai_client.py b/packages/shared/ai_client.py index ded9f4203..4b41d26de 100755 --- a/packages/shared/ai_client.py +++ b/packages/shared/ai_client.py @@ -250,22 +250,31 @@ class DoubaoClient: duration: int = 5, ratio: str | None = "9:16", resolution: str = "720p", - generate_audio: bool = False, + generate_audio: bool = True, watermark: bool = False, output_dir: str | None = None, model: str | None = None, + reference_images: list[str] | None = None, + reference_audios: list[str] | None = None, + reference_videos: list[str] | None = None, ) -> str | None: """调用 Seedance 2.5 文生/图生视频(异步任务→轮询→下载),返回本地 MP4 路径;失败返回 None。 + v1.6: 支持多参考图(产品素材)+ 参考音频(TTS口型驱动)+ 参考视频,单次生成最长 30 秒。 + Args: - prompt: 文本提示词 - image_url: 首帧参考图 URL(可选,提供则走图生视频) - duration: 视频时长 2~30 秒,默认 5 - ratio: 宽高比 16:9/9:16/1:1/4:3/3:4/21:9/adaptive + prompt: 文本提示词(含完整编导脚本:总览+场景光线+逐镜头时间轴+硬约束+负面词) + image_url: 首帧参考图 URL(可选,提供则走图生视频首帧模式,ratio 跟随首帧) + duration: 视频时长 4~30 秒 + ratio: 宽高比 16:9/9:16/1:1/4:3/3:4/21:9/adaptive;image_url 存在时自动忽略 resolution: 480p/720p/1080p - generate_audio: 是否生成模型自带音效(默认 False,我们自己混 TTS) + generate_audio: 是否让模型原生合成音效/BGM(v1.6 默认 True,配合 reference_audios 做口型驱动) watermark: 是否加水印 output_dir: 下载目录,默认 /tmp + model: 指定模型 ID;空则用 settings.doubao_video_model + reference_images: 多参考图 URL 列表(产品素材,最多30张;注意 image_url 为首帧单独传) + reference_audios: 参考音频 URL 列表(TTS口播,驱动口型,最多10段) + reference_videos: 参考视频 URL 列表(风格参考) Returns: 本地 MP4 文件路径,失败返回 None。 @@ -277,24 +286,37 @@ class DoubaoClient: settings = get_shared_settings() poll_interval = getattr(settings, "doubao_video_poll_interval", 10) or 10 - total_timeout = getattr(settings, "doubao_video_timeout", 600) or 600 + total_timeout = getattr(settings, "doubao_video_timeout", 900) or 900 default_video_model = getattr(settings, "doubao_video_model", None) or "doubao-seedance-2-5-260628" video_model = model or default_video_model content: list[dict[str, Any]] = [{"type": "text", "text": prompt.strip()}] if image_url: content.append({"type": "image_url", "image_url": {"url": image_url}}) + # v1.6: 多参考图(产品素材) + if reference_images: + for url in reference_images[:30]: + if url and isinstance(url, str): + content.append({"type": "image_url", "image_url": {"url": url}}) create_payload: dict[str, Any] = { "model": video_model, "content": content, - "generate_audio": generate_audio, + "generate_audio": bool(generate_audio), "duration": int(duration), "resolution": resolution, - "watermark": watermark, + "watermark": bool(watermark), } - # Bug #2110: ratio=None 时不传(首帧图生视频跟随原图比例,传 ratio 会 400 InvalidParameter) - if ratio: + # v1.6: 参考音频(TTS 驱动口型) + if reference_audios: + create_payload["reference_audios"] = [ + {"url": u, "role": "audio_url"} for u in reference_audios[:10] if u and isinstance(u, str) + ] + # v1.6: 参考视频(风格参考) + if reference_videos: + create_payload["reference_videos"] = [{"url": u} for u in reference_videos[:5] if u and isinstance(u, str)] + # Bug #2110 / v1.6: ratio=None 时不传(首帧图生视频跟随原图比例) + if ratio and not image_url: create_payload["ratio"] = ratio headers = { @@ -303,13 +325,16 @@ class DoubaoClient: } create_url = f"{self.base_url}/contents/generations/tasks" logger.info( - "Seedance 创建任务请求: url=%s model=%s duration=%ds ratio=%s gen_audio=%s image_url=%s", + "Seedance 创建任务请求: url=%s model=%s duration=%ds ratio=%s gen_audio=%s image_url=%s ref_imgs=%d ref_audios=%d ref_videos=%d", create_url, video_model, duration, ratio or "(follow-image)", generate_audio, bool(image_url), + len(reference_images or []), + len(reference_audios or []), + len(reference_videos or []), ) # 1) 创建任务(带重试) @@ -358,7 +383,13 @@ class DoubaoClient: ) return None - logger.info("Seedance 任务已创建: task_id=%s model=%s duration=%ds", task_id, video_model, duration) + logger.info( + "Seedance 任务已创建: task_id=%s model=%s duration=%ds gen_audio=%s", + task_id, + video_model, + duration, + generate_audio, + ) # 2) 轮询状态 poll_url = f"{create_url}/{task_id}" diff --git a/packages/shared/ai_service.py b/packages/shared/ai_service.py index 7b557e3b2..65809c641 100755 --- a/packages/shared/ai_service.py +++ b/packages/shared/ai_service.py @@ -575,32 +575,43 @@ def call_video_generation( prompt: str, *, image_url: str | None = None, - duration: int = 5, + duration: int = 15, ratio: str | None = "9:16", resolution: str = "720p", output_dir: str | None = None, + model: str | None = None, + generate_audio: bool = True, + reference_images: list[str] | None = None, + reference_audios: list[str] | None = None, + reference_videos: list[str] | None = None, ) -> str | None: - """调用 Seedance 2.5 生成视频段,返回本地 MP4 路径;失败返回 None。 + """调用 Seedance 2.5 生成视频(v1.6 单次出片版),返回本地 MP4 路径;失败返回 None。 - 封装 ai_client.video_generation:提交异步任务→轮询→下载到本地。 - Bug #2110: 首帧参考图模式下不传 ratio(API 要求跟随首帧图比例,传 ratio=9:16 - 会返回 400 InvalidParameter)。 + v1.6: + - 默认 generate_audio=True,模型原生合成环境音效/BGM; + - reference_audios 传 TTS 音频 URL 数组做口型驱动; + - reference_images 传产品素材 URL 数组做视觉参考; + - 单次最长 30 秒,不分段不拼接; + - image_url 存在时为「首帧图生视频」模式,自动不传 ratio(Bug #2110)。 """ client = get_doubao_client() if not client.is_available: logger.warning("[ai_service] 豆包客户端未配置,跳过视频生成") return None - # 首帧模式:不强制 ratio,让模型跟随首帧图比例 effective_ratio = None if image_url else ratio try: kwargs: dict = dict( prompt=prompt, image_url=image_url, - duration=duration, + duration=int(duration), resolution=resolution, - generate_audio=False, # 我们自己混 TTS + generate_audio=bool(generate_audio), watermark=False, output_dir=output_dir, + model=model, + reference_images=reference_images, + reference_audios=reference_audios, + reference_videos=reference_videos, ) if effective_ratio: kwargs["ratio"] = effective_ratio diff --git a/tests/unit/test_viral_video.py b/tests/unit/test_viral_video.py index a18bf35b7..6487d058e 100755 --- a/tests/unit/test_viral_video.py +++ b/tests/unit/test_viral_video.py @@ -120,7 +120,7 @@ class TestViralVideoJobDefaults: job = ViralVideoJob(user_id="u1") assert job.images == [] assert job.industry == "" - assert job.duration == 30 + assert job.duration == 15 assert job.fusion_level == FusionLevel.AI_POLISH assert job.style_strength == StyleStrength.MEDIUM assert job.status == ViralVideoStatus.PENDING @@ -145,13 +145,10 @@ class TestViralVideoStage: "image_analysis", "video_analysis", "intent_parsing", - "copy_fusion", - "storyboard", + "script_generation", "review", "tts", - "bgm_select", "rendering", - "musetalk", "uploading", ] actual_order = [s.value for s in ViralVideoStage] @@ -171,7 +168,7 @@ class TestViralVideoSchemas: assert req.images == ["https://example.com/img.jpg"] assert req.fusion_level == "ai_polish" assert req.style_strength == "medium" - assert req.duration == 30 + assert req.duration == 15 def test_create_request_empty_images_raises(self): from app.schemas.viral_video import CreateViralVideoRequest @@ -365,9 +362,10 @@ class TestViralVideoPipeline: industry="美妆", target_customer="年轻女性", marketing_purpose="品牌推广", - duration=30, + duration=15, user_copy_text="这款产品超好用", fusion_level="ai_polish", + video_ratio="9:16", ) @patch("packages.shared.ai_service.call_vision") @@ -405,63 +403,100 @@ class TestViralVideoPipeline: assert "intent" in result @patch("packages.shared.ai_service.call_llm") - def test_copy_fusion_ai_polish(self, mock_llm, mock_job): - from apps.worker.worker_app.tasks.viral_video import _step_copy_fusion + def test_script_generation_returns_copy_result(self, mock_llm, mock_job): + """v1.6: _step_script_generation 返回 dict 形式的 CopyResult,含 voiceover_script + shots。""" + from apps.worker.worker_app.tasks.viral_video import _step_script_generation - mock_llm.return_value = "融合后的文案内容" - result = _step_copy_fusion(mock_job, {"intent": "推广"}, {"products": []}) - assert isinstance(result, str) - assert len(result) > 0 + mock_llm.return_value = { + "overview": {"theme": "口红推荐", "total_duration": 15, "aspect_ratio": "9:16"}, + "scene_and_lighting": "明亮化妆台,柔和自然光", + "shots": [ + { + "time_range": "0-5秒", + "shot_type_angle_movement": "近景平视,缓慢推镜", + "scene_and_dialogue": "女主微笑展示口红:大家好,今天分享一款口红", + "action_details": "手持口红特写", + "audio_bgm": "轻快流行BGM", + "transition": "硬切", + "reference_image_index": 0, + }, + { + "time_range": "5-15秒", + "shot_type_angle_movement": "特写,固定镜头", + "scene_and_dialogue": "涂抹口红:颜色特别好看很显白", + "action_details": "嘴唇涂抹特写", + "audio_bgm": "轻快BGM继续", + "transition": "结束", + "reference_image_index": 1, + }, + ], + "hard_constraints": ["无字幕无水印"], + "negative_prompts": ["字幕", "水印"], + "voiceover_script": "大家好,今天分享一款口红,颜色特别好看很显白。", + } + result = _step_script_generation(mock_job, {"intent": "推广口红", "key_messages": [], "tone": "亲切"}, {"products": []}) + assert isinstance(result, dict) + assert "voiceover_script" in result + assert "shots" in result + assert isinstance(result["shots"], list) + assert len(result["shots"]) == 2 + assert result["overview"]["total_duration"] == 15 + # final_copy 必须 = voiceover_script(向后兼容) + assert result.get("final_copy") == result["voiceover_script"] @patch("packages.shared.ai_service.call_llm") - def test_storyboard_generation(self, mock_llm, mock_job): - from apps.worker.worker_app.tasks.viral_video import _step_storyboard + def test_script_generation_fallback(self, mock_llm, mock_job): + """LLM 返回异常时使用兜底脚本(不会抛错)。""" + from apps.worker.worker_app.tasks.viral_video import _fallback_script - mock_llm.return_value = [ - {"order": 0, "type": "product_shot", "duration": 10}, - {"order": 1, "type": "closing", "duration": 5}, - ] - result = _step_storyboard(mock_job, "测试文案", {}) - assert isinstance(result, list) - assert len(result) == 2 + result = _fallback_script(mock_job) + assert isinstance(result, dict) + assert result["voiceover_script"] + assert len(result["shots"]) >= 1 @patch("packages.shared.ai_service.call_llm") - def test_review_pass(self, mock_llm, mock_job): + def test_review_pass_v16(self, mock_llm, mock_job): + """v1.6 _step_review 接收 copy_result dict。""" from apps.worker.worker_app.tasks.viral_video import _step_review mock_llm.return_value = {"passed": True, "score": 90, "details": {}} - result = _step_review(mock_job, "测试文案", []) + cr = {"voiceover_script": "大家好", "shots": []} + result = _step_review(mock_job, cr) assert result["passed"] is True - def test_bgm_select(self, mock_job): - from apps.worker.worker_app.tasks.viral_video import _step_bgm_select + def test_assemble_seedance_prompt(self, mock_job): + """编导脚本必须能拼出完整的 Seedance prompt,含总览/场景/逐镜头/约束。""" + from apps.worker.worker_app.tasks.viral_video import _assemble_seedance_prompt - # P1: BGM 素材未就绪前 _step_bgm_select 统一返回 None(跳过 BGM 混音) - mock_job.bgm_preference = "upbeat" - bgm = _step_bgm_select(mock_job) - assert bgm is None - - def test_bgm_select_default(self, mock_job): - from apps.worker.worker_app.tasks.viral_video import _step_bgm_select - - mock_job.bgm_preference = "" - bgm = _step_bgm_select(mock_job) - assert bgm is None + cr = { + "overview": {"theme": "口红", "total_duration": 15, "aspect_ratio": "9:16"}, + "scene_and_lighting": "明亮化妆台", + "shots": [ + {"time_range": "0-15秒", "shot_type_angle_movement": "中景平视", "scene_and_dialogue": "你好分享", "action_details": "展示", "audio_bgm": "BGM", "transition": "结束", "reference_image_index": 0} + ], + "hard_constraints": ["无字幕"], + "negative_prompts": ["水印"], + } + prompt = _assemble_seedance_prompt(cr, mock_job) + assert "【视频总览】" in prompt + assert "【逐镜头时间轴】" in prompt + assert "【硬性约束】" in prompt + assert "【负面提示词】" in prompt + assert "0-15秒" in prompt # ── 端到端流水线集成测试 ──────────────────────────────────────────────── class TestPipelineIntegration: - """流水线端到端集成测试(mock 外部依赖)。""" + """v1.6 流水线端到端集成测试(mock 外部依赖):TTS+单次 Seedance+上传。""" @patch("apps.worker.worker_app.tasks.viral_video._step_upload") @patch("apps.worker.worker_app.tasks.viral_video._step_render") - @patch("apps.worker.worker_app.tasks.viral_video._step_bgm_select") + @patch("apps.worker.worker_app.tasks.viral_video._upload_tts_to_oss") @patch("apps.worker.worker_app.tasks.viral_video._step_tts") @patch("apps.worker.worker_app.tasks.viral_video._step_review") - @patch("apps.worker.worker_app.tasks.viral_video._step_storyboard") - @patch("apps.worker.worker_app.tasks.viral_video._step_copy_fusion") + @patch("apps.worker.worker_app.tasks.viral_video._step_script_generation") @patch("apps.worker.worker_app.tasks.viral_video._step_intent_parsing") @patch("apps.worker.worker_app.tasks.viral_video._step_video_analysis") @patch("apps.worker.worker_app.tasks.viral_video._step_image_analysis") @@ -474,38 +509,46 @@ class TestPipelineIntegration: mock_img_analysis, mock_video_analysis, mock_intent, - mock_copy_fusion, - mock_storyboard, + mock_script, mock_review, mock_tts, - mock_bgm, + mock_tts_upload, mock_render, mock_upload, ): - """测试 resume 流水线能从确认状态走到完成。""" + """v1.6: TTS整段合成 → 上传TTS到OSS → 单次 Seedance → 上传成片。""" from apps.worker.worker_app.tasks.viral_video import ( resume_viral_video_pipeline, ) - # 构造 mock job job = ViralVideoJob( user_id="user-001", images=["https://img.com/1.jpg"], industry="美妆", status=ViralVideoStatus.RUNNING, intent_result={"intent": "推广"}, + duration=15, + video_ratio="9:16", ) mock_repo = MagicMock() mock_session = MagicMock() mock_get_repo.return_value = (mock_session, mock_repo, job) - # 设置各步骤返回值 - mock_copy_fusion.return_value = "融合文案" - mock_storyboard.return_value = [{"order": 0, "duration": 10}] + # v1.6: 如果没有 copy_result 会现场补生成 + mock_intent.return_value = {"intent": "推广", "key_messages": [], "tone": "亲切"} + mock_script.return_value = { + "overview": {"theme": "口红", "total_duration": 15, "aspect_ratio": "9:16"}, + "scene_and_lighting": "明亮化妆台", + "shots": [], + "hard_constraints": [], + "negative_prompts": [], + "voiceover_script": "大家好,分享一款口红。", + "final_copy": "大家好,分享一款口红。", + } mock_review.return_value = {"passed": True, "score": 90} - mock_tts.return_value = None # P1: TTS 返回 Path|None,mock 用 None 跳过混音 - mock_bgm.return_value = None # P1: BGM 未就绪前返回 None + mock_tts.return_value = None # TTS 失败也能走下去(Seedance generate_audio=True 会自己合成音效) + mock_tts_upload.return_value = None mock_render.return_value = "/tmp/video.mp4" mock_upload.return_value = "https://oss.example.com/final.mp4" diff --git a/tests/unit/test_viral_video_p0.py b/tests/unit/test_viral_video_p0.py index 5c560f684..258225a42 100644 --- a/tests/unit/test_viral_video_p0.py +++ b/tests/unit/test_viral_video_p0.py @@ -67,49 +67,58 @@ class TestImageAnalysisField: # ── P0-1: storyboard 规范化 ──────────────────────────────────────── -class TestStoryboardNormalize: - def test_normalize_fills_defaults(self): - from apps.worker.worker_app.tasks.viral_video import _normalize_storyboard +class TestScriptGenerationV16: + """v1.6 编导分镜脚本生成相关纯函数测试。""" - raw = [{"order": 0, "description": "镜头一"}] - out = _normalize_storyboard(raw, total_duration=10, n_segments=1, copy_text="文案") - assert len(out) == 1 - assert out[0]["duration"] >= 3 - assert out[0]["ken_burns"] in {"zoom_in", "zoom_out", "pan_left", "pan_right", "static"} - assert out[0]["type"] == "product_shot" - assert out[0]["text"] == "" + def test_fallback_script_has_required_fields(self, mock_job): + from apps.worker.worker_app.tasks.viral_video import _fallback_script + out = _fallback_script(mock_job) + assert isinstance(out, dict) + assert "overview" in out + assert "shots" in out + assert "voiceover_script" in out + assert "hard_constraints" in out + assert "negative_prompts" in out + assert out["overview"]["total_duration"] == mock_job.duration + assert out["final_copy"] == out["voiceover_script"] + assert len(out["shots"]) >= 1 - def test_normalize_scales_to_total_duration(self): - from apps.worker.worker_app.tasks.viral_video import _normalize_storyboard + def test_safe_json_loads_parses_fenced_code(self): + from apps.worker.worker_app.tasks.viral_video import _safe_json_loads + fenced = "```json\n{\"voiceover_script\": \"你好\", \"shots\": []}\n```" + out = _safe_json_loads(fenced) + assert out is not None + assert out["voiceover_script"] == "你好" - raw = [ - {"order": 0, "duration": 10, "description": "a"}, - {"order": 1, "duration": 10, "description": "b"}, - ] - out = _normalize_storyboard(raw, total_duration=10, n_segments=2, copy_text="x") - total = sum(s["duration"] for s in out) - assert total == 10 + def test_safe_json_loads_handles_none(self): + from apps.worker.worker_app.tasks.viral_video import _safe_json_loads + assert _safe_json_loads(None) is None + assert _safe_json_loads("not json") is None - def test_fallback_storyboard(self): - from apps.worker.worker_app.tasks.viral_video import _fallback_storyboard + def test_validate_normalize_fills_defaults(self, mock_job): + from apps.worker.worker_app.tasks.viral_video import _validate_and_normalize_script + raw = {"voiceover_script": "你好", "shots": [{"scene_and_dialogue": "测试"}]} + out = _validate_and_normalize_script(raw, mock_job) + assert out["voiceover_script"] == "你好" + assert len(out["shots"]) == 1 + assert out["shots"][0]["shot_type_angle_movement"] + assert out["overview"]["total_duration"] == mock_job.duration - out = _fallback_storyboard("文案", total_duration=15, n_segments=3) - assert len(out) == 3 - assert sum(s["duration"] for s in out) == 15 - assert all(s["duration"] >= 3 for s in out) - - def test_storyboard_llm_list(self, mock_job): - from apps.worker.worker_app.tasks.viral_video import _step_storyboard - - with patch("packages.shared.ai_service.call_llm") as mock_llm: - mock_llm.return_value = [ - {"order": 0, "description": "产品特写", "duration": 5, "text": "t1"}, - {"order": 1, "description": "使用场景", "duration": 5, "text": "t2"}, - {"order": 2, "description": "CTA", "duration": 5, "text": "t3"}, - ] - result = _step_storyboard(mock_job, "文案", {"products": []}) - assert len(result) == 3 - assert all("description" in s for s in result) + def test_assemble_seedance_prompt_contains_sections(self, mock_job): + from apps.worker.worker_app.tasks.viral_video import _assemble_seedance_prompt + cr = { + "overview": {"theme": "测试", "total_duration": 15, "aspect_ratio": "9:16"}, + "scene_and_lighting": "明亮", + "shots": [ + {"time_range": "0-15秒", "shot_type_angle_movement": "中景", "scene_and_dialogue": "你好", + "action_details": "展示", "audio_bgm": "BGM", "transition": "结束", "reference_image_index": 0} + ], + "hard_constraints": ["无字幕"], + "negative_prompts": ["水印"], + } + p = _assemble_seedance_prompt(cr, mock_job) + for key in ("【视频总览】", "【场景与光线】", "【逐镜头时间轴】", "【硬性约束】", "【负面提示词】"): + assert key in p # ── P1: TTS 返回 Path|None ──────────────────────────────────────── @@ -150,12 +159,24 @@ class TestTTSPath: # ── P1: BGM 跳过 / MuseTalk 无 persona 跳过 ─────────────────────── -class TestBGMSkip: - def test_bgm_returns_none(self, mock_job): - from apps.worker.worker_app.tasks.viral_video import _step_bgm_select +class TestDurationClamp: + """v1.6 mark_copy_generated 派生字段 + duration clamp。""" - mock_job.bgm_preference = "upbeat" - assert _step_bgm_select(mock_job) is None + def test_mark_copy_generated_derives_fields(self): + job = ViralVideoJob(user_id="u1", duration=15) + cr = { + "overview": {"theme": "x", "total_duration": 15, "aspect_ratio": "9:16"}, + "scene_and_lighting": "亮", + "shots": [{"time_range": "0-15秒", "scene_and_dialogue": "对白"}], + "voiceover_script": "你好", + "hard_constraints": [], + "negative_prompts": [], + } + job.mark_copy_generated(cr) + assert job.copy_result is cr + assert job.generated_copy_text == "你好" + assert job.storyboard == cr["shots"] + assert job.effective_copy_text == "你好" # ── P0-1: call_video_generation 参数构造 ────────────────────────── @@ -188,23 +209,61 @@ class TestCallVideoGeneration: assert kwargs["prompt"] == "测试" assert kwargs["image_url"] == "https://img/x.jpg" assert kwargs["duration"] == 5 + assert kwargs["generate_audio"] is True # ── P0-1: _step_render 占位片段生成 ────────────────────────────── -class TestPlaceholderClip: - def test_make_placeholder_clip(self, tmp_path): - import shutil +class TestCallVideoGenerationV16: + """v1.6 call_video_generation 透传 reference_audios/reference_images 等参数到 client。""" - from apps.worker.worker_app.tasks.viral_video import _make_placeholder_clip, _probe_ok + def test_passes_reference_params_to_client(self, tmp_path): + from packages.shared.ai_service import call_video_generation - if not shutil.which("ffmpeg"): - pytest.skip("ffmpeg not available") + out = tmp_path / "v.mp4" + out.write_bytes(b"fake") + with patch("packages.shared.ai_service.get_doubao_client") as mock_get: + mock_client = MagicMock() + mock_client.is_available = True + mock_client.video_generation.return_value = str(out) + mock_get.return_value = mock_client + result = call_video_generation( + prompt="测试", + image_url="https://img/x.jpg", + duration=15, + ratio="9:16", + reference_images=["https://img/r1.jpg"], + reference_audios=["https://oss/tts.mp3"], + reference_videos=["https://oss/ref.mp4"], + generate_audio=True, + model="doubao-seedance-2-5-260628", + ) + assert result == str(out) + kwargs = mock_client.video_generation.call_args.kwargs + # 首帧模式不传 ratio(Bug #2110) + assert "ratio" not in kwargs + assert kwargs["image_url"] == "https://img/x.jpg" + assert kwargs["reference_audios"] == ["https://oss/tts.mp3"] + assert kwargs["reference_images"] == ["https://img/r1.jpg"] + assert kwargs["reference_videos"] == ["https://oss/ref.mp4"] + assert kwargs["generate_audio"] is True + assert kwargs["model"] == "doubao-seedance-2-5-260628" - out = _make_placeholder_clip(tmp_path, 0, 3) - assert out.exists() - assert _probe_ok(str(out)) + def test_ratio_passed_when_no_image(self, tmp_path): + from packages.shared.ai_service import call_video_generation + + out = tmp_path / "v.mp4" + out.write_bytes(b"fake") + with patch("packages.shared.ai_service.get_doubao_client") as mock_get: + mock_client = MagicMock() + mock_client.is_available = True + mock_client.video_generation.return_value = str(out) + mock_get.return_value = mock_client + call_video_generation(prompt="测试", duration=10, ratio="16:9") + kwargs = mock_client.video_generation.call_args.kwargs + assert kwargs["ratio"] == "16:9" + assert kwargs["image_url"] is None # ── P0-1: DoubaoClient.video_generation 在不可用时返回 None ─────── diff --git a/tests/unit/test_viral_video_routes.py b/tests/unit/test_viral_video_routes.py index 305e93a6c..2c9bcfcf6 100644 --- a/tests/unit/test_viral_video_routes.py +++ b/tests/unit/test_viral_video_routes.py @@ -34,7 +34,7 @@ def _make_job(job_id: str = "job-1", user_id: str = "u1", status: str = "pending "viral_structure": "", "marketing_purpose": "", "bgm_preference": "", - "duration": 30, + "duration": 15, "user_copy_text": "", "fusion_level": "ai_polish", "reference_audio_path": "", @@ -53,13 +53,18 @@ def _make_job(job_id: str = "job-1", user_id: str = "u1", status: str = "pending "intent_result": None, "image_analysis": None, "storyboard": None, + "copy_result": None, "generated_copy_text": "", "voice_id": "", "voice_source": "", + "voice_mode": "global", "video_ratio": "9:16", "video_model": "", "credits_cost": 0, "updated_at": None, + "is_terminal": False, + "effective_copy_text": "", + "voiceover_script": "", }.items(): setattr(job, k, kwargs.pop(k, v)) return job