Compare commits
41 Commits
| Author | SHA1 | Date | |
|---|---|---|---|
| c1763b995c | |||
| 3a8faeb31d | |||
| d336382f3a | |||
| d947171713 | |||
| 1f4c907bed | |||
| 121820caa9 | |||
| 52f281a66c | |||
| f1bd2d6f1d | |||
| eac05dee30 | |||
| 4263e7f6ca | |||
| db244fe14c | |||
| e86f137c3d | |||
| 8a3115bc54 | |||
| 3fcc65840e | |||
| 4633126bb4 | |||
| ed72a91990 | |||
| 02d226a163 | |||
| 475ee59408 | |||
| 452a484c5b | |||
| d01040cb93 | |||
| f523548eee | |||
| 0542654ca8 | |||
| 2a2dfad137 | |||
| 2205adb8fb | |||
| 4725d94c7e | |||
| df164ddf75 | |||
| f10fd9cd5c | |||
| 1ff81dcd0a | |||
| db9ee89ffa | |||
| a0d4f6e111 | |||
| fac80b1f77 | |||
| 159a62f9a5 | |||
| 109d7afbc7 | |||
| 9d31818222 | |||
| af4dd31dd1 | |||
| 8ecf381a9d | |||
| 244691d335 | |||
| ee4fff42f0 | |||
| b0018e747b | |||
| cbca0c3584 | |||
| c7c30936a9 |
@@ -0,0 +1,46 @@
|
||||
"""add video_fingerprint_chunks table for per-chunk fingerprint storage
|
||||
|
||||
Revision ID: 063_fingerprint_chunks
|
||||
Revises: 062_edit_plan_id
|
||||
Create Date: 2026-09-03
|
||||
"""
|
||||
|
||||
import sqlalchemy as sa
|
||||
|
||||
from alembic import op
|
||||
|
||||
revision = "063_fingerprint_chunks"
|
||||
down_revision = "062_edit_plan_id"
|
||||
branch_labels = None
|
||||
depends_on = None
|
||||
|
||||
|
||||
def upgrade() -> None:
|
||||
op.create_table(
|
||||
"video_fingerprint_chunks",
|
||||
sa.Column("id", sa.String(36), primary_key=True),
|
||||
sa.Column("video_id", sa.String(36), nullable=False),
|
||||
sa.Column("project_id", sa.String(36), nullable=False),
|
||||
sa.Column("user_id", sa.String(36), nullable=False, server_default=""),
|
||||
sa.Column("start_time_ms", sa.Integer, nullable=False),
|
||||
sa.Column("end_time_ms", sa.Integer, nullable=False),
|
||||
sa.Column("phash_binary", sa.String(16), nullable=False),
|
||||
sa.Column("color_histogram", sa.JSON, nullable=False),
|
||||
sa.Column("frame_count", sa.Integer, nullable=False, server_default="1"),
|
||||
sa.Column(
|
||||
"created_at",
|
||||
sa.DateTime,
|
||||
nullable=False,
|
||||
server_default=sa.func.now(),
|
||||
),
|
||||
)
|
||||
op.create_index("ix_vfc_video_id", "video_fingerprint_chunks", ["video_id"])
|
||||
op.create_index("ix_vfc_project_id", "video_fingerprint_chunks", ["project_id"])
|
||||
op.create_index("ix_vfc_user_id", "video_fingerprint_chunks", ["user_id"])
|
||||
|
||||
|
||||
def downgrade() -> None:
|
||||
op.drop_index("ix_vfc_user_id", table_name="video_fingerprint_chunks")
|
||||
op.drop_index("ix_vfc_project_id", table_name="video_fingerprint_chunks")
|
||||
op.drop_index("ix_vfc_video_id", table_name="video_fingerprint_chunks")
|
||||
op.drop_table("video_fingerprint_chunks")
|
||||
@@ -0,0 +1,25 @@
|
||||
"""add match_count and visual_similarity to generated_videos
|
||||
|
||||
Revision ID: 064_match_count_visual_sim
|
||||
Revises: 063_fingerprint_chunks
|
||||
Create Date: 2026-09-03
|
||||
"""
|
||||
|
||||
import sqlalchemy as sa
|
||||
|
||||
from alembic import op
|
||||
|
||||
revision = "064_match_count_visual_sim"
|
||||
down_revision = "063_fingerprint_chunks"
|
||||
branch_labels = None
|
||||
depends_on = None
|
||||
|
||||
|
||||
def upgrade() -> None:
|
||||
op.add_column("generated_videos", sa.Column("match_count", sa.Integer(), nullable=True, server_default="0"))
|
||||
op.add_column("generated_videos", sa.Column("visual_similarity", sa.Float(), nullable=True, server_default="0.0"))
|
||||
|
||||
|
||||
def downgrade() -> None:
|
||||
op.drop_column("generated_videos", "visual_similarity")
|
||||
op.drop_column("generated_videos", "match_count")
|
||||
@@ -0,0 +1,25 @@
|
||||
"""add visual_similarity and match_count to duplication_records
|
||||
|
||||
Revision ID: 065_dup_record_sim_match
|
||||
Revises: 064_match_count_visual_sim
|
||||
Create Date: 2026-09-04
|
||||
"""
|
||||
|
||||
import sqlalchemy as sa
|
||||
|
||||
from alembic import op
|
||||
|
||||
revision = "065_dup_record_sim_match"
|
||||
down_revision = "064_match_count_visual_sim"
|
||||
branch_labels = None
|
||||
depends_on = None
|
||||
|
||||
|
||||
def upgrade() -> None:
|
||||
op.add_column("duplication_records", sa.Column("visual_similarity", sa.Float(), nullable=True))
|
||||
op.add_column("duplication_records", sa.Column("match_count", sa.Integer(), nullable=True))
|
||||
|
||||
|
||||
def downgrade() -> None:
|
||||
op.drop_column("duplication_records", "match_count")
|
||||
op.drop_column("duplication_records", "visual_similarity")
|
||||
@@ -7,6 +7,7 @@ from typing import Any
|
||||
from uuid import uuid4
|
||||
|
||||
from app.auth import AuthenticatedUser, get_current_user
|
||||
from app.core.celery_app import celery_app
|
||||
from app.core.storage import OSSStorageService, get_storage_service
|
||||
from app.dependencies import get_duplication_repository
|
||||
from app.schemas.duplication import (
|
||||
@@ -76,6 +77,8 @@ def _to_record_response(record: DuplicationRecord) -> DuplicationRecordResponse:
|
||||
status=record.status,
|
||||
duplicate_rate=record.duplicate_rate,
|
||||
duplicate_count=record.duplicate_count,
|
||||
visual_similarity=getattr(record, "visual_similarity", None),
|
||||
match_count=getattr(record, "match_count", None),
|
||||
created_at=record.created_at.isoformat(),
|
||||
updated_at=record.updated_at.isoformat(),
|
||||
)
|
||||
@@ -90,6 +93,8 @@ def _to_detail_response(record: DuplicationRecord) -> DuplicationDetailResponse:
|
||||
status=record.status,
|
||||
duplicate_rate=record.duplicate_rate,
|
||||
duplicate_count=record.duplicate_count,
|
||||
visual_similarity=getattr(record, "visual_similarity", None),
|
||||
match_count=getattr(record, "match_count", None),
|
||||
created_at=record.created_at.isoformat(),
|
||||
updated_at=record.updated_at.isoformat(),
|
||||
segments=[
|
||||
@@ -192,6 +197,8 @@ async def upload_for_duplication(
|
||||
authenticated_user.user.id,
|
||||
)
|
||||
|
||||
celery_app.send_task("worker.process_duplication_check", args=[record.id])
|
||||
|
||||
return DuplicationUploadResponse(
|
||||
id=record.id,
|
||||
status=record.status,
|
||||
@@ -296,6 +303,8 @@ def retry_duplication(
|
||||
detail=f"查重记录 {record_id} 不存在",
|
||||
)
|
||||
|
||||
celery_app.send_task("worker.process_duplication_check", args=[updated.id])
|
||||
|
||||
return DuplicationUploadResponse(
|
||||
id=updated.id,
|
||||
status=updated.status,
|
||||
|
||||
@@ -23,6 +23,7 @@ from app.dependencies import (
|
||||
get_generation_task_repository,
|
||||
)
|
||||
from app.schemas.generation_task import (
|
||||
BatchPreviewGenerationTaskResponse,
|
||||
CreatePreviewGenerationTaskRequest,
|
||||
PreviewGenerationTaskResponse,
|
||||
)
|
||||
@@ -193,11 +194,19 @@ def _to_preview_response(task, generated_videos: list | None = None) -> PreviewG
|
||||
if started_at and completed_at:
|
||||
generate_duration = (completed_at - started_at).total_seconds()
|
||||
|
||||
title_cfg = getattr(task, "title_config", None)
|
||||
title_cfg = title_cfg if isinstance(title_cfg, dict) else {}
|
||||
extra_meta = getattr(task, "extra_meta", None)
|
||||
extra_meta = extra_meta if isinstance(extra_meta, dict) else {}
|
||||
voice_library_id = getattr(task, "voice_library_id", "") or ""
|
||||
if not isinstance(voice_library_id, str):
|
||||
voice_library_id = str(voice_library_id) if voice_library_id else ""
|
||||
return PreviewGenerationTaskResponse(
|
||||
task_id=task.id,
|
||||
status=task.status.value if hasattr(task.status, "value") else str(task.status),
|
||||
progress=float(task.progress or 0.0),
|
||||
is_preview=bool(getattr(task, "is_preview", True)),
|
||||
variant_index=int(extra_meta.get("variant_index", 0) or 0),
|
||||
resolution=getattr(task, "resolution", "") or "",
|
||||
video_url=video_url,
|
||||
duration=duration,
|
||||
@@ -206,6 +215,8 @@ def _to_preview_response(task, generated_videos: list | None = None) -> PreviewG
|
||||
transition_count=transition_count,
|
||||
material_usage=material_usage,
|
||||
error_message=task.error_message or "",
|
||||
title_text=str(title_cfg.get("text", "") or ""),
|
||||
voice_library_id=voice_library_id,
|
||||
created_at=task.created_at,
|
||||
started_at=started_at,
|
||||
finished_at=completed_at,
|
||||
@@ -213,45 +224,95 @@ def _to_preview_response(task, generated_videos: list | None = None) -> PreviewG
|
||||
)
|
||||
|
||||
|
||||
@router.post("/preview", response_model=PreviewGenerationTaskResponse, status_code=201)
|
||||
def _resolve_preview_edit_plan_id(
|
||||
*,
|
||||
request: CreatePreviewGenerationTaskRequest,
|
||||
task,
|
||||
db: Session,
|
||||
user_id: str,
|
||||
) -> str:
|
||||
"""确定任务关联的编辑计划ID:优先前端传入,否则按 template_id+user 兜底查找。"""
|
||||
if task.source_edit_plan_id:
|
||||
return task.source_edit_plan_id
|
||||
if not request.template_id:
|
||||
return ""
|
||||
try:
|
||||
from packages.adapters.sqlalchemy_impl.edit_plan_repository import (
|
||||
SQLAlchemyEditPlanRepository,
|
||||
)
|
||||
|
||||
_plan_repo = SQLAlchemyEditPlanRepository(db)
|
||||
_plans = _plan_repo.list_by_template(request.template_id, limit=20)
|
||||
for _p in _plans:
|
||||
if (_p.created_by_user_id or "") == user_id:
|
||||
logger.info(
|
||||
"[预览生成] 自动关联编辑计划: task_id=%s plan_id=%s",
|
||||
task.id,
|
||||
_p.id,
|
||||
)
|
||||
return _p.id
|
||||
except Exception:
|
||||
logger.warning(
|
||||
"[预览生成] 查找关联编辑计划失败(不影响主流程): task_id=%s",
|
||||
task.id,
|
||||
exc_info=True,
|
||||
)
|
||||
return ""
|
||||
|
||||
|
||||
def _variant_value(values: list[str], index: int, fallback: str = "") -> str:
|
||||
"""从变体数组中取值:长度1=共用,长度>N=按索引,空数组=回退 fallback。"""
|
||||
if not values:
|
||||
return fallback
|
||||
if len(values) == 1:
|
||||
return values[0]
|
||||
return values[index] if index < len(values) else fallback
|
||||
|
||||
|
||||
@router.post("/preview", response_model=BatchPreviewGenerationTaskResponse, status_code=201)
|
||||
def create_preview_generation_task(
|
||||
request: CreatePreviewGenerationTaskRequest,
|
||||
authenticated_user: AuthenticatedUser = Depends(get_current_user),
|
||||
generation_task_repository=Depends(get_generation_task_repository),
|
||||
db: Session = Depends(get_db_session),
|
||||
asset_repo=Depends(get_asset_repository),
|
||||
) -> PreviewGenerationTaskResponse:
|
||||
"""创建预览生成任务。
|
||||
) -> BatchPreviewGenerationTaskResponse:
|
||||
"""创建预览生成任务(支持批量)。
|
||||
|
||||
预览渲染品质与正式生成一致(1080p, CRF 23, medium preset),确认生成时可直接复用预览产物。
|
||||
|
||||
Args:
|
||||
request: 预览任务创建请求(template_id + asset_ids 等)
|
||||
preview_count=1 时行为与旧版完全一致(创建 1 个任务);
|
||||
preview_count=N 时一次创建 N 个独立变体任务:
|
||||
- 每个变体克隆独立编辑计划(独立 clips、独立随机素材起点),N 个预览内容互不相同
|
||||
- 每个变体拥有独立 task_id / 状态 / 预览视频 URL,前端按 task_id 分别轮询
|
||||
- 标题样式(font/color/position 等)全局共用;标题文字/配音/封面可按变体独立
|
||||
(titles[] / voice_library_ids[] / cover_urls[],长度1=共用,长度N=独立)
|
||||
|
||||
Returns:
|
||||
201 + 预览任务详情
|
||||
201 + 变体任务数组 {items: [...], total: N}
|
||||
"""
|
||||
user_id = authenticated_user.user.id
|
||||
count = max(1, request.preview_count)
|
||||
logger.info(
|
||||
"[预览生成] 接收请求: user_id=%s, template_id=%s, asset_count=%d, preview_count=%d",
|
||||
user_id,
|
||||
request.template_id,
|
||||
len(request.asset_ids),
|
||||
request.preview_count,
|
||||
count,
|
||||
)
|
||||
|
||||
# 预检查队列限流
|
||||
# 预检查队列限流(按变体总数计)
|
||||
try:
|
||||
user_pending = generation_task_repository.count_pending_by_user(user_id)
|
||||
global_pending = generation_task_repository.count_pending_total()
|
||||
if user_pending + 1 > USER_PENDING_LIMIT:
|
||||
raise UserPendingLimitExceeded(user_id=user_id, pending_count=user_pending + 1, limit=USER_PENDING_LIMIT)
|
||||
if global_pending + 1 > GLOBAL_PENDING_LIMIT:
|
||||
raise GlobalQueueFull(pending_count=global_pending + 1, limit=GLOBAL_PENDING_LIMIT)
|
||||
if user_pending + count > USER_PENDING_LIMIT:
|
||||
raise UserPendingLimitExceeded(
|
||||
user_id=user_id, pending_count=user_pending + count, limit=USER_PENDING_LIMIT
|
||||
)
|
||||
if global_pending + count > GLOBAL_PENDING_LIMIT:
|
||||
raise GlobalQueueFull(pending_count=global_pending + count, limit=GLOBAL_PENDING_LIMIT)
|
||||
except UserPendingLimitExceeded as e:
|
||||
raise HTTPException(
|
||||
status_code=429,
|
||||
detail=f"您的待处理任务过多(当前 {e.pending_count - 1}/{e.limit}),请等待后再提交",
|
||||
detail=f"您的待处理任务过多(当前 {e.pending_count - count}/{e.limit},本次提交 {count} 个),请等待后再提交",
|
||||
) from e
|
||||
except GlobalQueueFull as e:
|
||||
raise HTTPException(
|
||||
@@ -273,14 +334,11 @@ def create_preview_generation_task(
|
||||
w, h = int(parts[0]), int(parts[1])
|
||||
base = 1920
|
||||
if w < h:
|
||||
# 竖屏
|
||||
output_width = round(base * w / h)
|
||||
output_height = base
|
||||
else:
|
||||
# 横屏
|
||||
output_width = base
|
||||
output_height = round(base * h / w)
|
||||
# 对齐到偶数
|
||||
output_width = output_width - output_width % 2
|
||||
output_height = output_height - output_height % 2
|
||||
except (ValueError, ZeroDivisionError):
|
||||
@@ -289,42 +347,71 @@ def create_preview_generation_task(
|
||||
|
||||
logger.info(
|
||||
"[预览生成] 分辨率: video_ratio=%s → %s (%dx%d)",
|
||||
video_ratio, resolution, output_width, output_height,
|
||||
video_ratio,
|
||||
resolution,
|
||||
output_width,
|
||||
output_height,
|
||||
)
|
||||
|
||||
# 从模板读取 editing_mode / mode 作为 strategy_id(渲染 pipeline 的 mode 参数)
|
||||
strategy_id = _resolve_strategy_id_from_template(request.template_id, db, user_id)
|
||||
|
||||
title_config = request.title_config or {}
|
||||
base_title_config = request.title_config or {}
|
||||
|
||||
use_case = CreateGenerationTaskUseCase(generation_task_repository)
|
||||
|
||||
# ── 预创建第一个任务,仅用于解析源编辑计划(不落库为最终任务)──
|
||||
# 先创建一个临时任务拿到 task 对象上下文,实际 N 个任务在循环中统一创建;
|
||||
# 为保持与旧版一致的源 plan 解析逻辑,先创建任务0、解析源 plan,
|
||||
# 再预克隆 N 个变体 plan,最后重建任务关联。
|
||||
# 简化实现:直接创建全部任务,plan 关联在创建后、入队前完成。
|
||||
|
||||
created_tasks: list = []
|
||||
variant_plan_ids: list[str] = [] # 每个变体最终关联的 plan_id(按变体顺序)
|
||||
|
||||
try:
|
||||
task = use_case.execute(
|
||||
CreateGenerationTaskCommand(
|
||||
project_id="",
|
||||
asset_library_id="",
|
||||
strategy_id=strategy_id,
|
||||
voice_library_id=request.voice_library_id,
|
||||
template_id=request.template_id,
|
||||
asset_ids=list(request.asset_ids),
|
||||
title_ids=list(request.title_ids),
|
||||
voice_ids=list(request.voice_ids),
|
||||
created_by_user_id=user_id,
|
||||
source_edit_plan_id=request.source_edit_plan_id,
|
||||
asset_select_mode="",
|
||||
batch_id="",
|
||||
video_title=request.video_title,
|
||||
resolution=resolution,
|
||||
bgm_config=request.bgm_config or {},
|
||||
auto_retry_enabled=False,
|
||||
auto_retry_max=0,
|
||||
is_preview=True,
|
||||
title_config=title_config,
|
||||
output_width=output_width,
|
||||
output_height=output_height,
|
||||
for variant_index in range(count):
|
||||
# 变体独立标题文字:titles[] 覆盖 title_config.text
|
||||
variant_title_text = _variant_value(request.titles, variant_index, "")
|
||||
variant_title_config = dict(base_title_config)
|
||||
if variant_title_text.strip():
|
||||
variant_title_config["text"] = variant_title_text.strip()
|
||||
|
||||
# 变体独立配音
|
||||
variant_voice_library_id = _variant_value(
|
||||
request.voice_library_ids, variant_index, request.voice_library_id
|
||||
)
|
||||
)
|
||||
|
||||
task = use_case.execute(
|
||||
CreateGenerationTaskCommand(
|
||||
project_id="",
|
||||
asset_library_id="",
|
||||
strategy_id=strategy_id,
|
||||
voice_library_id=variant_voice_library_id,
|
||||
template_id=request.template_id,
|
||||
asset_ids=list(request.asset_ids),
|
||||
title_ids=list(request.title_ids),
|
||||
voice_ids=list(request.voice_ids),
|
||||
created_by_user_id=user_id,
|
||||
source_edit_plan_id=request.source_edit_plan_id,
|
||||
asset_select_mode="",
|
||||
batch_id="",
|
||||
video_title=request.video_title,
|
||||
resolution=resolution,
|
||||
bgm_config=request.bgm_config or {},
|
||||
auto_retry_enabled=False,
|
||||
auto_retry_max=0,
|
||||
is_preview=True,
|
||||
title_config=variant_title_config,
|
||||
output_width=output_width,
|
||||
output_height=output_height,
|
||||
)
|
||||
)
|
||||
task.extra_meta["variant_index"] = variant_index
|
||||
|
||||
# 解析源编辑计划(前端传入或按模板兜底查找)
|
||||
source_plan_id = _resolve_preview_edit_plan_id(request=request, task=task, db=db, user_id=user_id)
|
||||
task.source_edit_plan_id = source_plan_id
|
||||
generation_task_repository.update(task)
|
||||
created_tasks.append(task)
|
||||
except ValueError as e:
|
||||
logger.warning("[预览生成] 创建失败: %s", e)
|
||||
raise HTTPException(status_code=400, detail=str(e)) from e
|
||||
@@ -332,93 +419,121 @@ def create_preview_generation_task(
|
||||
logger.error("[预览生成] 创建失败: %s", e, exc_info=True)
|
||||
raise HTTPException(status_code=500, detail="创建预览生成任务失败,请稍后再试") from e
|
||||
|
||||
# 关联编辑计划:如果前端未传 source_edit_plan_id,通过 template_id + user_id 查找
|
||||
if not task.source_edit_plan_id and request.template_id:
|
||||
try:
|
||||
from packages.adapters.sqlalchemy_impl.edit_plan_repository import (
|
||||
SQLAlchemyEditPlanRepository,
|
||||
)
|
||||
|
||||
_plan_repo = SQLAlchemyEditPlanRepository(db)
|
||||
_plans = _plan_repo.list_by_template(request.template_id, limit=20)
|
||||
for _p in _plans:
|
||||
if (_p.created_by_user_id or "") == user_id:
|
||||
task.source_edit_plan_id = _p.id
|
||||
generation_task_repository.update(task)
|
||||
logger.info(
|
||||
"[预览生成] 自动关联编辑计划: task_id=%s plan_id=%s",
|
||||
task.id,
|
||||
_p.id,
|
||||
)
|
||||
break
|
||||
except Exception:
|
||||
logger.warning(
|
||||
"[预览生成] 查找关联编辑计划失败(不影响主流程): task_id=%s",
|
||||
task.id,
|
||||
exc_info=True,
|
||||
)
|
||||
|
||||
# 每条预览都关联独立克隆 plan:多预览前端为 N 次并发调用,若共用同一 plan
|
||||
# 则 N 条预览片段完全相同;克隆时片段起点按持久化历史区间重算(含受控复用),
|
||||
# 保证各预览版本内容不同
|
||||
if task.source_edit_plan_id:
|
||||
# ── 克隆独立变体 plan:N 个预览全部克隆(预览不污染源 plan)──
|
||||
# 源 plan 不存在(无编辑历史)时各任务走自身随机选片流程,不克隆。
|
||||
source_plan_id = created_tasks[0].source_edit_plan_id if created_tasks else ""
|
||||
if source_plan_id:
|
||||
try:
|
||||
from app.services.edit_plan_service import EditPlanService
|
||||
|
||||
_plan_svc = EditPlanService(db)
|
||||
_preview_plan = _plan_svc.clone_plan_for_variant(
|
||||
task.source_edit_plan_id,
|
||||
created_by_user_id=user_id,
|
||||
name_suffix="预览变体",
|
||||
)
|
||||
task.source_edit_plan_id = _preview_plan.id
|
||||
generation_task_repository.update(task)
|
||||
logger.info(
|
||||
"[预览生成] 预览关联独立克隆 plan: task_id=%s clone_plan_id=%s",
|
||||
task.id,
|
||||
_preview_plan.id,
|
||||
)
|
||||
except Exception as clone_err:
|
||||
# 不退回共用原 plan(否则多条预览内容相同,违反去重诉求):
|
||||
# 标记任务失败并中断,前端可重新发起预览
|
||||
logger.error(
|
||||
"[预览生成] 克隆预览变体 plan 失败,任务标记失败: task_id=%s error=%s",
|
||||
task.id,
|
||||
clone_err,
|
||||
exc_info=True,
|
||||
)
|
||||
_mark_task_failed(generation_task_repository, task, "预览变体计划创建失败")
|
||||
for variant_index in range(count):
|
||||
last_err: Exception | None = None
|
||||
variant_plan = None
|
||||
for _attempt in range(2): # 1 次重试,抗 DB 瞬时抖动
|
||||
try:
|
||||
variant_plan = _plan_svc.clone_plan_for_variant(
|
||||
source_plan_id,
|
||||
created_by_user_id=user_id,
|
||||
name_suffix=f"预览变体{variant_index + 1}" if count > 1 else "预览变体",
|
||||
)
|
||||
break
|
||||
except Exception as clone_err: # noqa: PERF203
|
||||
last_err = clone_err
|
||||
logger.warning(
|
||||
"[预览生成] 克隆变体 plan 失败(尝试%d/2): variant=%d error=%s",
|
||||
_attempt + 1,
|
||||
variant_index,
|
||||
clone_err,
|
||||
exc_info=True,
|
||||
)
|
||||
if variant_plan is None:
|
||||
logger.error(
|
||||
"[预览生成] 克隆预览变体 plan 重试仍失败: variant=%d source=%s",
|
||||
variant_index,
|
||||
source_plan_id,
|
||||
exc_info=last_err,
|
||||
)
|
||||
# 标记已创建任务失败
|
||||
for t in created_tasks:
|
||||
_mark_task_failed(generation_task_repository, t, "预览变体计划创建失败")
|
||||
raise HTTPException(
|
||||
status_code=500,
|
||||
detail="创建预览任务失败:无法生成独立剪辑计划,请重试",
|
||||
) from last_err
|
||||
variant_plan_ids.append(variant_plan.id)
|
||||
except HTTPException:
|
||||
raise
|
||||
except Exception as e:
|
||||
logger.error("[预览生成] 克隆变体 plan 异常: %s", e, exc_info=True)
|
||||
for t in created_tasks:
|
||||
_mark_task_failed(generation_task_repository, t, "预览变体计划创建失败")
|
||||
raise HTTPException(
|
||||
status_code=500,
|
||||
detail="创建预览任务失败:无法生成独立剪辑计划,请重试",
|
||||
) from clone_err
|
||||
) from e
|
||||
|
||||
# 入队执行;若入队失败则标记任务为 failed 避免僵尸数据
|
||||
try:
|
||||
if not safe_enqueue_generation_task(
|
||||
task,
|
||||
generation_task_repository,
|
||||
user_id=user_id,
|
||||
log_prefix="[预览生成]",
|
||||
log_task_status=True,
|
||||
):
|
||||
logger.warning("[预览生成] 任务入队失败: task_id=%s", task.id)
|
||||
_mark_task_failed(generation_task_repository, task, "任务入队失败")
|
||||
raise HTTPException(status_code=500, detail="任务入队失败,请稍后重试")
|
||||
except UserPendingLimitExceeded as e:
|
||||
_mark_task_failed(generation_task_repository, task, "待处理任务超限")
|
||||
raise HTTPException(
|
||||
status_code=429,
|
||||
detail=f"您的待处理任务过多(当前 {e.pending_count - 1}/{e.limit}),请等待后再提交",
|
||||
) from None
|
||||
except GlobalQueueFull:
|
||||
_mark_task_failed(generation_task_repository, task, "系统队列已满")
|
||||
raise HTTPException(
|
||||
status_code=503,
|
||||
detail="系统繁忙,请稍后再试",
|
||||
) from None
|
||||
# 关联变体 plan 并回写标题配置
|
||||
for variant_index, task in enumerate(created_tasks):
|
||||
if variant_plan_ids:
|
||||
task.source_edit_plan_id = variant_plan_ids[variant_index]
|
||||
generation_task_repository.update(task)
|
||||
# 回写变体标题到 plan config(worker 渲染时从 plan 读取 title 配置)
|
||||
if task.source_edit_plan_id and (task.title_config or {}).get("text", "").strip():
|
||||
try:
|
||||
from app.api.routes.generation_tasks import _writeback_edit_plan_config
|
||||
|
||||
return _to_preview_response(task)
|
||||
_writeback_edit_plan_config(
|
||||
plan_id=task.source_edit_plan_id,
|
||||
task_id=task.id,
|
||||
title_config=task.title_config,
|
||||
db=db,
|
||||
)
|
||||
except Exception:
|
||||
logger.warning(
|
||||
"[预览生成] 回写标题配置失败(不影响主流程): task_id=%s",
|
||||
task.id,
|
||||
exc_info=True,
|
||||
)
|
||||
|
||||
# ── 入队 ──
|
||||
responses: list[PreviewGenerationTaskResponse] = []
|
||||
for variant_index, task in enumerate(created_tasks):
|
||||
try:
|
||||
enqueued = safe_enqueue_generation_task(
|
||||
task,
|
||||
generation_task_repository,
|
||||
user_id=user_id,
|
||||
log_prefix=f"[预览生成][变体{variant_index + 1}]",
|
||||
log_task_status=True,
|
||||
)
|
||||
if not enqueued:
|
||||
logger.warning("[预览生成] 任务入队失败: task_id=%s", task.id)
|
||||
_mark_task_failed(generation_task_repository, task, "任务入队失败")
|
||||
except UserPendingLimitExceeded:
|
||||
_mark_task_failed(generation_task_repository, task, "待处理任务超限")
|
||||
except GlobalQueueFull:
|
||||
_mark_task_failed(generation_task_repository, task, "系统队列已满")
|
||||
except Exception:
|
||||
logger.exception("[预览生成] 入队异常: task_id=%s", task.id)
|
||||
_mark_task_failed(generation_task_repository, task, "任务入队异常")
|
||||
# enqueue 会原地更新 task 状态/进度,直接用 task 构造响应
|
||||
responses.append(_to_preview_response(task))
|
||||
|
||||
# 队列满/限流时若全部失败,返回明确错误码
|
||||
if all(r.status == "failed" for r in responses):
|
||||
first_err = next((r.error_message for r in responses if r.error_message), "")
|
||||
if "待处理任务" in first_err:
|
||||
raise HTTPException(status_code=429, detail=first_err or "待处理任务超限")
|
||||
if "队列" in first_err:
|
||||
raise HTTPException(status_code=503, detail=first_err or "系统繁忙,请稍后再试")
|
||||
|
||||
logger.info(
|
||||
"[预览生成] 创建完成: %d 个变体任务, task_ids=%s",
|
||||
len(responses),
|
||||
[r.task_id for r in responses],
|
||||
)
|
||||
return BatchPreviewGenerationTaskResponse(items=responses, total=len(responses))
|
||||
|
||||
|
||||
@router.get("/preview/{task_id}", response_model=PreviewGenerationTaskResponse)
|
||||
|
||||
@@ -47,6 +47,15 @@ logger = logging.getLogger(__name__)
|
||||
router = APIRouter()
|
||||
|
||||
|
||||
def _variant_value(values: list[str], index: int, fallback: str = "") -> str:
|
||||
"""从变体数组中取值:长度1=共用,长度>N=按索引,空数组=回退 fallback。"""
|
||||
if not values:
|
||||
return fallback
|
||||
if len(values) == 1:
|
||||
return values[0]
|
||||
return values[index] if index < len(values) else fallback
|
||||
|
||||
|
||||
def _to_generation_task_response(task) -> GenerationTaskResponse:
|
||||
return GenerationTaskResponse(
|
||||
id=task.id,
|
||||
@@ -92,6 +101,9 @@ def _to_generated_video_response(item, download_url: str | None = None) -> Gener
|
||||
height=item.height,
|
||||
fps=item.fps,
|
||||
download_url=download_url,
|
||||
duplicate_rate=getattr(item, "duplicate_rate", None),
|
||||
visual_similarity=getattr(item, "visual_similarity", None),
|
||||
match_count=getattr(item, "match_count", None),
|
||||
)
|
||||
|
||||
|
||||
@@ -137,7 +149,6 @@ def _select_assets_from_library(
|
||||
return [a.id for a in ready_video_assets]
|
||||
|
||||
|
||||
|
||||
def _writeback_edit_plan_config(
|
||||
plan_id: str,
|
||||
task_id: str,
|
||||
@@ -162,7 +173,7 @@ def _writeback_edit_plan_config(
|
||||
current_config = plan_model.config if isinstance(plan_model.config, dict) else {}
|
||||
merged = dict(current_config)
|
||||
merged["generation_task_id"] = task_id
|
||||
|
||||
|
||||
# 检查标题是否发生变化,如果变化则清除 cover 字段强制重新生成封面
|
||||
if title_config:
|
||||
old_title_config = merged.get("title_config", {}) or {}
|
||||
@@ -174,10 +185,12 @@ def _writeback_edit_plan_config(
|
||||
del merged["cover"]
|
||||
logger.info(
|
||||
"[生成任务] 标题变化,清除旧封面: plan_id=%s old_title=%s new_title=%s",
|
||||
plan_id, old_title_text, new_title_text,
|
||||
plan_id,
|
||||
old_title_text,
|
||||
new_title_text,
|
||||
)
|
||||
merged["title_config"] = title_config
|
||||
|
||||
|
||||
plan_model.config = merged
|
||||
db.commit()
|
||||
logger.info(
|
||||
@@ -468,12 +481,21 @@ def create_generation_task(
|
||||
if task_index > 0 and variant_plan_ids:
|
||||
effective_plan_id = variant_plan_ids[task_index - 1]
|
||||
|
||||
# 变体级独立配置:titles[]/voice_library_ids[]/cover_urls[]
|
||||
# 长度1=所有变体共用,长度=count=每个变体独立,空数组=回退单值字段
|
||||
variant_title_text = _variant_value(request.titles, task_index, "")
|
||||
variant_title_config = dict(request.title_config or {})
|
||||
if variant_title_text.strip():
|
||||
variant_title_config["text"] = variant_title_text.strip()
|
||||
variant_voice_library_id = _variant_value(request.voice_library_ids, task_index, request.voice_library_id)
|
||||
variant_cover_url = _variant_value(request.cover_urls, task_index, request.cover_url)
|
||||
|
||||
task = use_case.execute(
|
||||
CreateGenerationTaskCommand(
|
||||
project_id=project_id,
|
||||
asset_library_id=asset_library_id,
|
||||
strategy_id=effective_strategy_id,
|
||||
voice_library_id=request.voice_library_id,
|
||||
voice_library_id=variant_voice_library_id,
|
||||
template_id=request.template_id,
|
||||
asset_ids=resolved_asset_ids,
|
||||
title_ids=request.title_ids,
|
||||
@@ -491,10 +513,12 @@ def create_generation_task(
|
||||
source_task_id=request.source_task_id,
|
||||
output_width=request.output_width,
|
||||
output_height=request.output_height,
|
||||
cover_url=request.cover_url,
|
||||
title_config=request.title_config or {},
|
||||
cover_url=variant_cover_url,
|
||||
title_config=variant_title_config,
|
||||
)
|
||||
)
|
||||
# 变体序号写入 extra_meta(响应/排查时可辨识)
|
||||
task.extra_meta["variant_index"] = task_index
|
||||
try:
|
||||
# 兜底关联编辑计划:前端未传 source_edit_plan_id 时,
|
||||
# 通过 template_id + user_id 在 DB 层直接查找最新的 plan。
|
||||
@@ -529,13 +553,13 @@ def create_generation_task(
|
||||
|
||||
# 回写 plan.config:必须在 enqueue 之前执行,
|
||||
# 确保 worker 读取 plan 时 config 中已包含 generation_task_id。
|
||||
# 只在首个任务时回写一次,避免批量生成时循环覆盖。
|
||||
# 批量场景下每个变体关联独立 plan,需各自回写自己的变体标题配置。
|
||||
_effective_plan_id = task.source_edit_plan_id
|
||||
if _effective_plan_id and len(created_tasks) == 0:
|
||||
if _effective_plan_id:
|
||||
_writeback_edit_plan_config(
|
||||
plan_id=_effective_plan_id,
|
||||
task_id=task.id,
|
||||
title_config=request.title_config,
|
||||
title_config=variant_title_config,
|
||||
db=db,
|
||||
)
|
||||
|
||||
|
||||
@@ -15,6 +15,7 @@ from app.schemas.video_center import (
|
||||
VideoItemResponse,
|
||||
)
|
||||
from fastapi import APIRouter, Depends, HTTPException, Query, Response
|
||||
from pydantic import BaseModel, Field
|
||||
|
||||
from packages.application import (
|
||||
GetGeneratedVideoUseCase,
|
||||
@@ -53,6 +54,8 @@ def _to_video_response(item, storage: OSSStorageService | None = None) -> VideoI
|
||||
download_url=download_url,
|
||||
generated_at=format_utc_datetime(item.generated_at) if hasattr(item, "generated_at") else "",
|
||||
duplicate_rate=getattr(item, "duplicate_rate", None),
|
||||
visual_similarity=getattr(item, "visual_similarity", None),
|
||||
match_count=getattr(item, "match_count", None),
|
||||
)
|
||||
|
||||
|
||||
@@ -237,3 +240,70 @@ def get_batch_download_status(
|
||||
status=api_status,
|
||||
download_url=download_url,
|
||||
)
|
||||
|
||||
|
||||
# ── 重新计算查重率 ─────────────────────────────────────────────────
|
||||
|
||||
|
||||
class RecomputeDedupRequest(BaseModel):
|
||||
"""重新计算查重率请求。"""
|
||||
|
||||
video_ids: list[str] | None = Field(
|
||||
None,
|
||||
description="指定视频 ID 列表。为空则对当前用户所有缺少查重数据的视频重新计算。",
|
||||
)
|
||||
|
||||
|
||||
class RecomputeDedupResponse(BaseModel):
|
||||
"""重新计算查重率响应。"""
|
||||
|
||||
enqueued: int = Field(..., description="已入队的任务数量")
|
||||
total_scanned: int = Field(..., description="扫描的视频总数")
|
||||
skipped: int = Field(..., description="已有查重数据跳过的数量")
|
||||
message: str = ""
|
||||
|
||||
|
||||
@router.post("/videos/recompute-dedup", response_model=RecomputeDedupResponse)
|
||||
def recompute_dedup(
|
||||
request: RecomputeDedupRequest = RecomputeDedupRequest(),
|
||||
repo=Depends(get_generated_video_repository),
|
||||
current_user: AuthenticatedUser = Depends(get_current_user),
|
||||
):
|
||||
"""重新计算视频的查重率/视觉相似度。
|
||||
|
||||
对于已存在但缺少 duplicate_rate / video_fingerprint 的视频,
|
||||
触发异步 Celery 任务重新下载并计算指纹 + 查重率。
|
||||
|
||||
不传 video_ids 时,对当前用户所有视频进行检查。
|
||||
"""
|
||||
user_id = current_user.user.id
|
||||
|
||||
# 获取目标视频列表
|
||||
if request.video_ids:
|
||||
all_videos = repo.get_by_ids(request.video_ids)
|
||||
# 安全校验:只处理当前用户的视频
|
||||
target_videos = [v for v in all_videos if v.user_id == user_id]
|
||||
else:
|
||||
target_videos = repo.list_by_user(user_id)
|
||||
|
||||
total_scanned = len(target_videos)
|
||||
enqueued = 0
|
||||
skipped = 0
|
||||
|
||||
for video in target_videos:
|
||||
# 已有完整查重数据的跳过
|
||||
if video.duplicate_rate is not None and video.video_fingerprint:
|
||||
skipped += 1
|
||||
continue
|
||||
|
||||
# 触发异步查重任务
|
||||
celery_app.send_task("worker.check_duplicate", args=[video.id])
|
||||
enqueued += 1
|
||||
logger.info("Enqueued re-dedup for video %s (user=%s)", video.id, user_id)
|
||||
|
||||
return RecomputeDedupResponse(
|
||||
enqueued=enqueued,
|
||||
total_scanned=total_scanned,
|
||||
skipped=skipped,
|
||||
message=f"已入队 {enqueued} 个查重任务" if enqueued > 0 else "所有视频查重数据已完整",
|
||||
)
|
||||
|
||||
@@ -28,6 +28,9 @@ class DuplicationRecordResponse(BaseModel):
|
||||
status: str = "pending"
|
||||
duplicate_rate: float | None = None
|
||||
duplicate_count: int = 0
|
||||
# #1661 视觉相似度(归一化 0~1)/ 匹配视频数
|
||||
visual_similarity: float | None = None
|
||||
match_count: int | None = None
|
||||
created_at: str
|
||||
updated_at: str
|
||||
|
||||
|
||||
@@ -25,6 +25,10 @@ class GeneratedVideoResponse(BaseModel):
|
||||
review_status: str = "pending_review"
|
||||
generation_params: dict = Field(default_factory=dict)
|
||||
download_url: str | None = None
|
||||
# #1660 查重率(百分比 0~100)/ 视觉相似度(0~1)/ 匹配帧数
|
||||
duplicate_rate: float | None = None
|
||||
visual_similarity: float | None = None
|
||||
match_count: int | None = None
|
||||
|
||||
|
||||
class GeneratedVideoDownloadUrlResponse(BaseModel):
|
||||
|
||||
@@ -25,6 +25,12 @@ class CreateGenerationTaskRequest(BaseModel):
|
||||
asset_library_id: str = ""
|
||||
strategy_id: str = ""
|
||||
voice_library_id: str = ""
|
||||
# ── 多变体独立配音(批量生成)──
|
||||
# 长度 1 = 所有变体共用;长度 = count = 每个变体独立配音;空数组 = 回退 voice_library_id
|
||||
voice_library_ids: list[str] = Field(
|
||||
default_factory=list,
|
||||
description="各变体独立配音素材库ID数组:长度1=共用,长度=count=独立。为空时回退 voice_library_id",
|
||||
)
|
||||
created_by_user_id: str = ""
|
||||
# ── 模板模式新增字段 ──
|
||||
template_id: str = ""
|
||||
@@ -75,6 +81,27 @@ class CreateGenerationTaskRequest(BaseModel):
|
||||
output_width: int = Field(default=1280, description="输出视频宽度")
|
||||
output_height: int = Field(default=720, description="输出视频高度")
|
||||
cover_url: str = Field(default="", description="封面图片 URL")
|
||||
# ── 多变体独立封面(批量生成)──
|
||||
# 长度 1 = 所有变体共用;长度 = count = 每个变体独立封面;空数组 = 回退 cover_url
|
||||
cover_urls: list[str] = Field(
|
||||
default_factory=list,
|
||||
description="各变体独立封面URL数组:长度1=共用,长度=count=独立。为空时回退 cover_url",
|
||||
)
|
||||
# ── 多变体独立标题文字(批量生成)──
|
||||
# 长度 1 = 所有变体共用;长度 = count = 每个变体独立标题文字;空数组 = 使用 title_config.text
|
||||
titles: list[str] = Field(
|
||||
default_factory=list,
|
||||
description="各变体独立标题文字数组:长度1=共用,长度=count=独立。为空时使用 title_config.text",
|
||||
)
|
||||
|
||||
@model_validator(mode="after")
|
||||
def _check_variant_arrays(self) -> "CreateGenerationTaskRequest":
|
||||
"""变体数组字段长度校验:空数组(回退单值)、长度 1(共用)、或长度 = count(独立)。"""
|
||||
for name in ("voice_library_ids", "cover_urls", "titles"):
|
||||
arr = getattr(self, name)
|
||||
if arr and len(arr) != 1 and len(arr) != self.count:
|
||||
raise ValueError(f"{name} 长度必须为 1(共用)或 {self.count}(与 count 一致),当前为 {len(arr)}")
|
||||
return self
|
||||
|
||||
@model_validator(mode="after")
|
||||
def _check_at_least_one_mode(self) -> "CreateGenerationTaskRequest":
|
||||
@@ -185,8 +212,33 @@ class CreatePreviewGenerationTaskRequest(BaseModel):
|
||||
)
|
||||
title_config: dict = Field(
|
||||
default_factory=dict,
|
||||
description="标题配置(可选),渲染时烧录到预览视频中。支持字段: text/font/font_size/font_color/position/bold/stroke/shadow",
|
||||
description="标题配置(可选),渲染时烧录到预览视频中。支持字段: text/font/font_size/font_color/position/bold/stroke/shadow。N个变体时样式全局共用",
|
||||
)
|
||||
# ── 多变体独立配置(preview_count > 1)──
|
||||
# 长度 1 = 所有变体共用;长度 = preview_count = 每个变体独立;空数组 = 回退单值字段
|
||||
titles: list[str] = Field(
|
||||
default_factory=list,
|
||||
description="各变体独立标题文字数组:长度1=共用,长度=preview_count=独立。为空时使用 title_config.text",
|
||||
)
|
||||
voice_library_ids: list[str] = Field(
|
||||
default_factory=list,
|
||||
description="各变体独立配音素材库ID数组:长度1=共用,长度=preview_count=独立。为空时回退 voice_library_id",
|
||||
)
|
||||
cover_urls: list[str] = Field(
|
||||
default_factory=list,
|
||||
description="各变体独立封面URL数组:长度1=共用,长度=preview_count=独立(预览阶段通常为空)",
|
||||
)
|
||||
|
||||
@model_validator(mode="after")
|
||||
def _check_variant_arrays(self) -> "CreatePreviewGenerationTaskRequest":
|
||||
"""变体数组字段长度校验:空数组(回退单值)、长度 1(共用)、或长度 = preview_count(独立)。"""
|
||||
for name in ("titles", "voice_library_ids", "cover_urls"):
|
||||
arr = getattr(self, name)
|
||||
if arr and len(arr) != 1 and len(arr) != self.preview_count:
|
||||
raise ValueError(
|
||||
f"{name} 长度必须为 1(共用)或 {self.preview_count}(与 preview_count 一致),当前为 {len(arr)}"
|
||||
)
|
||||
return self
|
||||
|
||||
@model_validator(mode="after")
|
||||
def _check_template_id(self) -> "CreatePreviewGenerationTaskRequest":
|
||||
@@ -202,7 +254,7 @@ class CreatePreviewGenerationTaskRequest(BaseModel):
|
||||
|
||||
|
||||
class PreviewGenerationTaskResponse(BaseModel):
|
||||
"""预览生成任务响应。
|
||||
"""单个预览变体任务响应。
|
||||
|
||||
包含任务状态、进度、分辨率、生成结果 URL 等关键字段。
|
||||
"""
|
||||
@@ -211,6 +263,7 @@ class PreviewGenerationTaskResponse(BaseModel):
|
||||
status: str
|
||||
progress: float
|
||||
is_preview: bool = True
|
||||
variant_index: int = 0
|
||||
resolution: str = ""
|
||||
video_url: str = ""
|
||||
duration: float = 0.0
|
||||
@@ -219,7 +272,21 @@ class PreviewGenerationTaskResponse(BaseModel):
|
||||
transition_count: int = 0
|
||||
material_usage: dict = Field(default_factory=dict)
|
||||
error_message: str = ""
|
||||
title_text: str = ""
|
||||
voice_library_id: str = ""
|
||||
created_at: datetime | None = None
|
||||
started_at: datetime | None = None
|
||||
finished_at: datetime | None = None
|
||||
generate_duration: float = 0.0
|
||||
|
||||
|
||||
class BatchPreviewGenerationTaskResponse(BaseModel):
|
||||
"""批量预览任务响应:preview_count=N 时返回 N 个独立变体任务。
|
||||
|
||||
- items: 变体任务数组,按 variant_index 顺序排列,每个含独立 task_id/状态/预览视频URL
|
||||
- total: 变体总数(= preview_count)
|
||||
- 前端按 items[i].task_id 分别轮询 GET /preview/{task_id} 获取进度与结果
|
||||
"""
|
||||
|
||||
items: list[PreviewGenerationTaskResponse]
|
||||
total: int
|
||||
|
||||
@@ -22,7 +22,10 @@ class VideoItemResponse(BaseModel):
|
||||
generation_params: dict = Field(default_factory=dict)
|
||||
download_url: str | None = None
|
||||
generated_at: str = ""
|
||||
# #1660 查重率(百分比 0~100)/ 视觉相似度(0~1)/ 匹配帧数
|
||||
duplicate_rate: float | None = None
|
||||
visual_similarity: float | None = None
|
||||
match_count: int | None = None
|
||||
|
||||
|
||||
class ListVideosResponse(BaseModel):
|
||||
|
||||
@@ -131,6 +131,7 @@ class PlanGeneratorService:
|
||||
editing_mode,
|
||||
random_selection=random_preview,
|
||||
asset_durations=asset_durations,
|
||||
user_id=created_by_user_id,
|
||||
)
|
||||
|
||||
# 5. 持久化所有 clips 并计算总时长
|
||||
@@ -218,6 +219,7 @@ class PlanGeneratorService:
|
||||
*,
|
||||
random_selection: bool = False,
|
||||
asset_durations: dict[str, float] | None = None,
|
||||
user_id: str = "",
|
||||
) -> None:
|
||||
"""按 editing_mode 将素材分配到 clips(就地修改,未持久化).
|
||||
|
||||
@@ -239,6 +241,14 @@ class PlanGeneratorService:
|
||||
asset_ids = list(asset_ids) # 复制避免修改调用方原列表
|
||||
random.shuffle(asset_ids)
|
||||
|
||||
# 查询已有视频的已用区间(跨视频避让)
|
||||
external_used_segments = None
|
||||
if user_id and self._clip_repo:
|
||||
try:
|
||||
external_used_segments = self._clip_repo.list_used_segments_by_user(user_id, limit_recent=50)
|
||||
except Exception:
|
||||
logger.warning("跨视频避让查询失败,回退到纯随机", exc_info=True)
|
||||
|
||||
distribute_assets(
|
||||
clips,
|
||||
asset_ids,
|
||||
@@ -246,6 +256,7 @@ class PlanGeneratorService:
|
||||
random_selection=random_selection,
|
||||
asset_durations=asset_durations,
|
||||
asset_scene_points=asset_scene_points,
|
||||
external_used_segments=external_used_segments,
|
||||
)
|
||||
|
||||
def _fetch_asset_scene_points(self, asset_ids: List[str]) -> dict[str, list[float]]:
|
||||
|
||||
@@ -0,0 +1,174 @@
|
||||
#!/usr/bin/env python3
|
||||
"""存量指纹重建脚本 — 为已有视频生成 video_fingerprint_chunks 分片数据。
|
||||
|
||||
功能:
|
||||
- 查询 generated_videos 中 video_fingerprint IS NOT NULL 但尚无分片数据的视频
|
||||
- 从 OSS 下载视频 → 用新的分片算法重新计算指纹 → 写入分片表
|
||||
- 支持 --dry-run(只打印不写入)和 --batch-size(默认 50)
|
||||
- 幂等:已存在分片数据的视频跳过
|
||||
|
||||
用法:
|
||||
# 预览(不写入)
|
||||
python rebuild_fingerprint_chunks.py --dry-run
|
||||
|
||||
# 执行重建
|
||||
python rebuild_fingerprint_chunks.py --batch-size 50
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import logging
|
||||
import os
|
||||
import sys
|
||||
import tempfile
|
||||
|
||||
# 确保可以 import worker_app 和 packages
|
||||
sys.path.insert(0, os.path.join(os.path.dirname(__file__), "..", "..", "..", "worker"))
|
||||
sys.path.insert(0, os.path.join(os.path.dirname(__file__), "..", "..", ".."))
|
||||
|
||||
logging.basicConfig(
|
||||
level=logging.INFO,
|
||||
format="%(asctime)s [%(levelname)s] %(name)s: %(message)s",
|
||||
)
|
||||
logger = logging.getLogger("rebuild_fingerprint_chunks")
|
||||
|
||||
|
||||
def find_videos_needing_rebuild(session, batch_size: int) -> list[dict]:
|
||||
"""查询需要重建分片指纹的视频。"""
|
||||
from sqlalchemy import and_
|
||||
|
||||
from packages.adapters.sqlalchemy_impl.models import GeneratedVideoModel, VideoFingerprintChunkModel
|
||||
|
||||
# 有 video_fingerprint 的视频
|
||||
has_fingerprint = GeneratedVideoModel.video_fingerprint.isnot(None)
|
||||
has_fingerprint = and_(has_fingerprint, GeneratedVideoModel.video_fingerprint != "")
|
||||
|
||||
# 排除已有分片数据的视频
|
||||
subq = session.query(VideoFingerprintChunkModel.video_id).distinct().subquery()
|
||||
no_chunks = ~GeneratedVideoModel.id.in_(subq)
|
||||
|
||||
videos = (
|
||||
session.query(GeneratedVideoModel)
|
||||
.filter(and_(has_fingerprint, no_chunks))
|
||||
.order_by(GeneratedVideoModel.generated_at.desc())
|
||||
.limit(batch_size)
|
||||
.all()
|
||||
)
|
||||
|
||||
return [
|
||||
{
|
||||
"id": v.id,
|
||||
"project_id": v.project_id,
|
||||
"user_id": v.user_id or "",
|
||||
"duration": v.duration,
|
||||
}
|
||||
for v in videos
|
||||
]
|
||||
|
||||
|
||||
def rebuild_one(video_info: dict, dry_run: bool = False) -> int:
|
||||
"""重建单个视频的分片数据。返回写入的 chunk 数量。"""
|
||||
from video_processing.dedup import VideoDeduplicator, _save_fingerprint_chunks
|
||||
from worker_app.db import SessionLocal
|
||||
|
||||
from packages.adapters.sqlalchemy_impl.models import VideoFingerprintChunkModel
|
||||
from packages.shared.storage import get_storage_service
|
||||
|
||||
video_id = video_info["id"]
|
||||
project_id = video_info["project_id"]
|
||||
user_id = video_info["user_id"]
|
||||
|
||||
if dry_run:
|
||||
logger.info("[DRY-RUN] Would rebuild video %s (project=%s)", video_id, project_id)
|
||||
return 0
|
||||
|
||||
session = SessionLocal()
|
||||
temp_dir = tempfile.mkdtemp()
|
||||
|
||||
try:
|
||||
# 再次检查幂等性
|
||||
existing_count = (
|
||||
session.query(VideoFingerprintChunkModel).filter(VideoFingerprintChunkModel.video_id == video_id).count()
|
||||
)
|
||||
if existing_count > 0:
|
||||
logger.info("Video %s already has %d chunks, skipping", video_id, existing_count)
|
||||
return 0
|
||||
|
||||
# 下载视频
|
||||
storage_service = get_storage_service()
|
||||
local_path = os.path.join(temp_dir, f"{video_id}.mp4")
|
||||
storage_key = f"projects/{project_id}/generated/{video_id}/{video_id}.mp4"
|
||||
storage_service.download_file(storage_key, local_path)
|
||||
|
||||
# 重新计算指纹
|
||||
deduplicator = VideoDeduplicator()
|
||||
fingerprint = deduplicator.compute_fingerprint(local_path)
|
||||
|
||||
# 写入分片表
|
||||
_save_fingerprint_chunks(fingerprint, video_id, project_id, user_id, session)
|
||||
session.commit()
|
||||
|
||||
chunk_count = len(fingerprint.chunks)
|
||||
logger.info("Rebuilt %d chunks for video %s", chunk_count, video_id)
|
||||
return chunk_count
|
||||
|
||||
except Exception as e:
|
||||
logger.error("Failed to rebuild video %s: %s", video_id, e)
|
||||
session.rollback()
|
||||
return -1
|
||||
finally:
|
||||
session.close()
|
||||
import shutil
|
||||
|
||||
shutil.rmtree(temp_dir, ignore_errors=True)
|
||||
|
||||
|
||||
def main():
|
||||
parser = argparse.ArgumentParser(description="存量指纹重建脚本")
|
||||
parser.add_argument("--dry-run", action="store_true", help="只打印不写入")
|
||||
parser.add_argument("--batch-size", type=int, default=50, help="每批处理数量(默认 50)")
|
||||
parser.add_argument("--total-limit", type=int, default=0, help="总处理数量限制(0=不限制)")
|
||||
args = parser.parse_args()
|
||||
|
||||
from worker_app.db import SessionLocal
|
||||
|
||||
session = SessionLocal()
|
||||
|
||||
try:
|
||||
videos = find_videos_needing_rebuild(session, args.batch_size)
|
||||
logger.info("Found %d videos needing rebuild", len(videos))
|
||||
|
||||
if args.dry_run:
|
||||
for v in videos:
|
||||
logger.info("[DRY-RUN] Video %s | project=%s | duration=%.1fs", v["id"], v["project_id"], v["duration"])
|
||||
return
|
||||
|
||||
total_chunks = 0
|
||||
processed = 0
|
||||
failed = 0
|
||||
|
||||
for v in videos:
|
||||
if args.total_limit > 0 and processed >= args.total_limit:
|
||||
break
|
||||
|
||||
result = rebuild_one(v, dry_run=False)
|
||||
if result < 0:
|
||||
failed += 1
|
||||
else:
|
||||
total_chunks += result
|
||||
processed += 1
|
||||
|
||||
logger.info(
|
||||
"Rebuild complete: processed=%d, chunks=%d, failed=%d",
|
||||
processed,
|
||||
total_chunks,
|
||||
failed,
|
||||
)
|
||||
|
||||
finally:
|
||||
session.close()
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -20,6 +20,10 @@ export interface DuplicationRecord {
|
||||
duplicate_rate?: number
|
||||
/** 重复片段数 */
|
||||
duplicate_count?: number
|
||||
/** 视觉相似度(0-100),#1660 新增 */
|
||||
visual_similarity?: number
|
||||
/** 匹配帧数,#1660 新增 */
|
||||
match_count?: number
|
||||
/** 创建时间 */
|
||||
created_at: string
|
||||
/** 更新时间 */
|
||||
|
||||
@@ -13,6 +13,8 @@ export type {
|
||||
VideoItem,
|
||||
} from "./types"
|
||||
|
||||
export type { RecomputeDedupResponse } from "./products"
|
||||
|
||||
// 工具函数
|
||||
export { mapVideoToProductItem } from "./utils"
|
||||
|
||||
@@ -25,4 +27,5 @@ export {
|
||||
updateReviewStatus,
|
||||
batchDownload,
|
||||
getBatchDownloadStatus,
|
||||
recomputeDedup,
|
||||
} from "./products"
|
||||
|
||||
@@ -78,3 +78,18 @@ export const getBatchDownloadStatus = async (jobId: string): Promise<BatchDownlo
|
||||
console.warn("[getBatchDownloadStatus] 后端暂无批量下载状态端点", jobId)
|
||||
return { job_id: jobId, status: "processing", progress: 0 }
|
||||
}
|
||||
|
||||
/** 重新计算存量视频查重率(异步) */
|
||||
export interface RecomputeDedupResponse {
|
||||
enqueued: number
|
||||
total_scanned: number
|
||||
skipped: number
|
||||
message: string
|
||||
}
|
||||
|
||||
export const recomputeDedup = async (videoIds?: string[]): Promise<RecomputeDedupResponse> => {
|
||||
const response = await apiClient.post("/videos/recompute-dedup", {
|
||||
video_ids: videoIds,
|
||||
})
|
||||
return response.data
|
||||
}
|
||||
|
||||
@@ -23,6 +23,10 @@ export interface ProductItem {
|
||||
project_name?: string
|
||||
/** 查重率(百分比) */
|
||||
duplicate_rate?: number
|
||||
/** 视觉相似度(0-1),#1660 新增 */
|
||||
visual_similarity?: number
|
||||
/** 匹配帧数,#1660 新增 */
|
||||
match_count?: number
|
||||
created_at?: string
|
||||
updated_at?: string
|
||||
}
|
||||
@@ -72,4 +76,8 @@ export interface VideoItem {
|
||||
download_url: string
|
||||
generated_at: string
|
||||
duplicate_rate?: number
|
||||
/** 视觉相似度(0-1),#1660 新增 */
|
||||
visual_similarity?: number
|
||||
/** 匹配帧数,#1660 新增 */
|
||||
match_count?: number
|
||||
}
|
||||
|
||||
@@ -30,5 +30,7 @@ export function mapVideoToProductItem(video: VideoItem): ProductItem {
|
||||
created_at: video.generated_at,
|
||||
updated_at: video.generated_at,
|
||||
duplicate_rate: video.duplicate_rate,
|
||||
visual_similarity: video.visual_similarity,
|
||||
match_count: video.match_count,
|
||||
}
|
||||
}
|
||||
|
||||
@@ -81,10 +81,11 @@ export const extractVideoVoice = async (
|
||||
): Promise<{ asset_id: string; duration: number }> => {
|
||||
const formData = new FormData()
|
||||
formData.append("file", file)
|
||||
formData.append("project_id", "default")
|
||||
|
||||
return new Promise((resolve, reject) => {
|
||||
const xhr = new XMLHttpRequest()
|
||||
xhr.open("POST", "/api/v1/tts/extract-video-voice")
|
||||
xhr.open("POST", "/api/v1/voices/extract-voice")
|
||||
|
||||
// 携带认证 token(从 localStorage 获取,与 apiClient 拦截器一致)
|
||||
const token = localStorage.getItem("access_token")
|
||||
|
||||
@@ -98,7 +98,7 @@ const DuplicationDetail: React.FC = () => {
|
||||
<div className="dup-detail-grid">
|
||||
<RiskCard riskLevel={riskLevel} similarityPercent={similarityPercent} />
|
||||
<InfoCard detail={detail} />
|
||||
<SegmentsSection segments={detail.segments} />
|
||||
<SegmentsSection segments={detail.segments} totalDuration={detail.duration_seconds} />
|
||||
</div>
|
||||
</div>
|
||||
)
|
||||
|
||||
@@ -1,7 +1,7 @@
|
||||
import React from "react"
|
||||
import { Button, Tag, Tooltip } from "@/components/ui"
|
||||
import type { DuplicationRecord } from "@/api/duplication"
|
||||
import { STATUS_CONFIG } from "../constants"
|
||||
import { STATUS_CONFIG, RISK_TAG_VARIANT, RISK_LABELS } from "../constants"
|
||||
import { getRiskLevel, formatSize, formatDuration } from "../utils"
|
||||
|
||||
interface ResultCardProps {
|
||||
@@ -54,6 +54,9 @@ const ResultCard: React.FC<ResultCardProps> = ({ record, onView, onDelete, onRet
|
||||
/>
|
||||
</div>
|
||||
<span className={`dup-score-value ${riskLevel}`}>{rateValue.toFixed(1)}%</span>
|
||||
<Tag variant={RISK_TAG_VARIANT[riskLevel]} className="dup-score-risk-tag">
|
||||
{RISK_LABELS[riskLevel]}
|
||||
</Tag>
|
||||
</>
|
||||
) : record.status === "failed" ? (
|
||||
<Tooltip title="重新查重">
|
||||
|
||||
@@ -2,34 +2,81 @@ import React from "react"
|
||||
import { Tag } from "@/components/ui"
|
||||
import type { DuplicateSegment } from "@/api/duplication"
|
||||
import { SegmentCard } from "./SegmentCard"
|
||||
import { formatTime } from "../utils"
|
||||
|
||||
interface SegmentsSectionProps {
|
||||
segments?: DuplicateSegment[]
|
||||
/** 视频总时长(秒),用于渲染时间轴 */
|
||||
totalDuration?: number
|
||||
}
|
||||
|
||||
/** 片段相似度 → 风险等级(时间轴配色用) */
|
||||
const getSegmentRisk = (similarity: number): "low" | "medium" | "high" => {
|
||||
if (similarity >= 90) return "high"
|
||||
if (similarity >= 70) return "medium"
|
||||
return "low"
|
||||
}
|
||||
|
||||
/**
|
||||
* 重复片段列表区域
|
||||
* 重复片段列表区域(含时间轴可视化)
|
||||
*/
|
||||
export const SegmentsSection: React.FC<SegmentsSectionProps> = ({ segments = [] }) => (
|
||||
<div className="dup-checks-section">
|
||||
<h3>
|
||||
🔍 重复片段详情
|
||||
<Tag variant="primary" style={{ marginLeft: 8 }}>
|
||||
{segments.length} 个片段
|
||||
</Tag>
|
||||
</h3>
|
||||
export const SegmentsSection: React.FC<SegmentsSectionProps> = ({
|
||||
segments = [],
|
||||
totalDuration,
|
||||
}) => {
|
||||
const showTimeline = segments.length > 0 && totalDuration !== undefined && totalDuration > 0
|
||||
|
||||
{segments.length > 0 ? (
|
||||
<div className="dup-checks-list">
|
||||
{segments.map((segment, index) => (
|
||||
<SegmentCard key={segment.id} segment={segment} index={index} />
|
||||
))}
|
||||
</div>
|
||||
) : (
|
||||
<div className="dup-results-empty" style={{ padding: "32px 0" }}>
|
||||
<div className="dup-results-empty-icon">🎉</div>
|
||||
<p>未发现重复片段,内容原创度很高</p>
|
||||
</div>
|
||||
)}
|
||||
</div>
|
||||
)
|
||||
return (
|
||||
<div className="dup-checks-section">
|
||||
<h3>
|
||||
🔍 重复片段详情
|
||||
<Tag variant="primary" style={{ marginLeft: 8 }}>
|
||||
{segments.length} 个片段
|
||||
</Tag>
|
||||
</h3>
|
||||
|
||||
{showTimeline && (
|
||||
<div className="dup-timeline">
|
||||
<div className="dup-timeline-bar">
|
||||
{segments.map((seg, i) => {
|
||||
const left = (seg.source_start / totalDuration) * 100
|
||||
const width = Math.max(
|
||||
((seg.source_end - seg.source_start) / totalDuration) * 100,
|
||||
0.5,
|
||||
)
|
||||
const segRisk = getSegmentRisk(seg.similarity)
|
||||
return (
|
||||
<div
|
||||
key={seg.id ?? i}
|
||||
className={`dup-timeline-segment ${segRisk}`}
|
||||
style={{
|
||||
left: `${Math.min(left, 100)}%`,
|
||||
width: `${Math.min(width, 100 - Math.min(left, 100))}%`,
|
||||
}}
|
||||
title={`${formatTime(seg.source_start)} - ${formatTime(seg.source_end)} · 相似度 ${seg.similarity.toFixed(0)}% · ${seg.matched_video_name}`}
|
||||
/>
|
||||
)
|
||||
})}
|
||||
</div>
|
||||
<div className="dup-timeline-labels">
|
||||
<span>0s</span>
|
||||
<span>{formatTime(totalDuration ?? 0)}</span>
|
||||
</div>
|
||||
</div>
|
||||
)}
|
||||
|
||||
{segments.length > 0 ? (
|
||||
<div className="dup-checks-list">
|
||||
{segments.map((segment, index) => (
|
||||
<SegmentCard key={segment.id} segment={segment} index={index} />
|
||||
))}
|
||||
</div>
|
||||
) : (
|
||||
<div className="dup-results-empty" style={{ padding: "32px 0" }}>
|
||||
<div className="dup-results-empty-icon">🎉</div>
|
||||
<p>未发现重复片段,内容原创度很高</p>
|
||||
</div>
|
||||
)}
|
||||
</div>
|
||||
)
|
||||
}
|
||||
|
||||
@@ -831,3 +831,61 @@
|
||||
font-size: 16px;
|
||||
}
|
||||
}
|
||||
|
||||
/* ============================================================
|
||||
查重率风险标签(列表卡片)
|
||||
============================================================ */
|
||||
.dup-score-risk-tag {
|
||||
flex-shrink: 0;
|
||||
margin-left: 2px;
|
||||
}
|
||||
|
||||
/* ============================================================
|
||||
重复片段时间轴可视化(#1662)
|
||||
============================================================ */
|
||||
.dup-timeline {
|
||||
margin: 16px 0;
|
||||
padding: 0 8px;
|
||||
}
|
||||
|
||||
.dup-timeline-bar {
|
||||
position: relative;
|
||||
height: 24px;
|
||||
background: var(--bg-secondary, #f1f5f9);
|
||||
border-radius: 4px;
|
||||
overflow: hidden;
|
||||
}
|
||||
|
||||
.dup-timeline-segment {
|
||||
position: absolute;
|
||||
top: 2px;
|
||||
height: 20px;
|
||||
border-radius: 3px;
|
||||
opacity: 0.8;
|
||||
cursor: pointer;
|
||||
transition: opacity 0.2s;
|
||||
}
|
||||
|
||||
.dup-timeline-segment:hover {
|
||||
opacity: 1;
|
||||
}
|
||||
|
||||
.dup-timeline-segment.low {
|
||||
background: #22c55e;
|
||||
}
|
||||
|
||||
.dup-timeline-segment.medium {
|
||||
background: #f59e0b;
|
||||
}
|
||||
|
||||
.dup-timeline-segment.high {
|
||||
background: #ef4444;
|
||||
}
|
||||
|
||||
.dup-timeline-labels {
|
||||
display: flex;
|
||||
justify-content: space-between;
|
||||
font-size: 12px;
|
||||
color: var(--text-secondary);
|
||||
margin-top: 4px;
|
||||
}
|
||||
|
||||
@@ -1,9 +1,9 @@
|
||||
/** 根据查重率获取风险等级 */
|
||||
export const getRiskLevel = (rate?: number): "low" | "medium" | "high" => {
|
||||
if (rate === undefined) return "low"
|
||||
if (rate <= 10) return "low"
|
||||
if (rate <= 30) return "medium"
|
||||
return "high"
|
||||
if (rate < 15) return "low" // <15% 绿色(安全)
|
||||
if (rate <= 30) return "medium" // 15-30% 黄色(注意)
|
||||
return "high" // >30% 红色(危险)
|
||||
}
|
||||
|
||||
/** 格式化时间(秒 → mm:ss) */
|
||||
|
||||
@@ -40,13 +40,6 @@ const clipTypeLabel: Record<ClipType | string, string> = {
|
||||
pip: "混剪",
|
||||
}
|
||||
|
||||
const formatDuration = (sec: number) => {
|
||||
if (sec < 60) return `${sec.toFixed(1)}s`
|
||||
const m = Math.floor(sec / 60)
|
||||
const s = (sec % 60).toFixed(0)
|
||||
return `${m}m${s.padStart(2, "0")}s`
|
||||
}
|
||||
|
||||
const EditorClipList: React.FC<EditorClipListProps> = ({
|
||||
clips,
|
||||
selectedClipId,
|
||||
@@ -102,7 +95,6 @@ const EditorClipList: React.FC<EditorClipListProps> = ({
|
||||
{clipTypeLabel[clip.type] || "片段"}
|
||||
</span>
|
||||
</span>
|
||||
<span className="ep-clip-item-duration">{formatDuration(clip.duration)}</span>
|
||||
</div>
|
||||
|
||||
{/* 文案预览 */}
|
||||
|
||||
@@ -87,7 +87,7 @@ const PreviewPlayer: React.FC<PreviewPlayerProps> = ({
|
||||
WebkitTextStroke: "1px rgba(0,0,0,0.6)",
|
||||
top:
|
||||
titleConfig.position === "top"
|
||||
? "8px"
|
||||
? "6.25%"
|
||||
: titleConfig.position === "center"
|
||||
? "50%"
|
||||
: "auto",
|
||||
|
||||
@@ -7,7 +7,6 @@
|
||||
* - ClipCard - 片段卡片
|
||||
* - ClipTrack - 片段轨道(播放头+片段列表+添加卡片)
|
||||
* - TimelineHeader - 时间线头部(标题+缩放+操作按钮)
|
||||
* - AddClipPicker - 添加片段选择器
|
||||
* - TrimPreview - 裁剪预览 tooltip
|
||||
* - ContextMenu - 右键菜单
|
||||
*
|
||||
@@ -27,7 +26,6 @@ import { usePlayheadDrag } from "./timeline/hooks/usePlayheadDrag"
|
||||
import { TimeRuler } from "./timeline/TimeRuler"
|
||||
import { ClipTrack } from "./timeline/ClipTrack"
|
||||
import { TimelineHeader } from "./timeline/TimelineHeader"
|
||||
import { AddClipPicker } from "./timeline/AddClipPicker"
|
||||
import { TrimPreview } from "./timeline/TrimPreview"
|
||||
import { ContextMenu } from "./timeline/ContextMenu"
|
||||
|
||||
@@ -100,16 +98,7 @@ const TimelinePanel: React.FC<TimelinePanelProps> = ({
|
||||
handleContextSplit,
|
||||
handleContextResetTrim,
|
||||
handleContextDelete,
|
||||
showAddPicker,
|
||||
pickerRef,
|
||||
addCardRef,
|
||||
pickerPos,
|
||||
availableTypes,
|
||||
addType,
|
||||
addDuration,
|
||||
setAddType,
|
||||
setAddDuration,
|
||||
handleTogglePicker,
|
||||
handleConfirmAdd,
|
||||
hoveredClipId,
|
||||
setHoveredClipId,
|
||||
@@ -177,7 +166,7 @@ const TimelinePanel: React.FC<TimelinePanelProps> = ({
|
||||
onClipMouseLeave={() => setHoveredClipId(null)}
|
||||
onTrimHandleMouseDown={handleTrimHandleMouseDown}
|
||||
onClipRemove={onClipRemove}
|
||||
onTogglePicker={handleTogglePicker}
|
||||
onTogglePicker={handleConfirmAdd}
|
||||
/>
|
||||
)}
|
||||
|
||||
@@ -204,20 +193,6 @@ const TimelinePanel: React.FC<TimelinePanelProps> = ({
|
||||
onDelete={handleContextDelete}
|
||||
/>
|
||||
)}
|
||||
|
||||
{/* 类型+时长选择面板 */}
|
||||
{showAddPicker && (
|
||||
<AddClipPicker
|
||||
pickerRef={pickerRef}
|
||||
position={pickerPos}
|
||||
availableTypes={availableTypes}
|
||||
addType={addType}
|
||||
addDuration={addDuration}
|
||||
onTypeChange={setAddType}
|
||||
onDurationChange={setAddDuration}
|
||||
onConfirm={handleConfirmAdd}
|
||||
/>
|
||||
)}
|
||||
</div>
|
||||
)
|
||||
}
|
||||
|
||||
@@ -7,12 +7,8 @@ interface AddClipPickerProps {
|
||||
position: { top: number; right: number }
|
||||
availableTypes: ClipType[]
|
||||
addType: ClipType
|
||||
addDuration: number
|
||||
onTypeChange: (type: ClipType) => void
|
||||
onDurationChange: (duration: number) => void
|
||||
onConfirm: () => void
|
||||
minDuration?: number
|
||||
maxDuration?: number
|
||||
}
|
||||
|
||||
export const AddClipPicker: React.FC<AddClipPickerProps> = ({
|
||||
@@ -20,12 +16,8 @@ export const AddClipPicker: React.FC<AddClipPickerProps> = ({
|
||||
position,
|
||||
availableTypes,
|
||||
addType,
|
||||
addDuration,
|
||||
onTypeChange,
|
||||
onDurationChange,
|
||||
onConfirm,
|
||||
minDuration = 1,
|
||||
maxDuration = 120,
|
||||
}) => {
|
||||
return (
|
||||
<div
|
||||
@@ -53,24 +45,6 @@ export const AddClipPicker: React.FC<AddClipPickerProps> = ({
|
||||
))}
|
||||
</div>
|
||||
|
||||
{/* 时长输入 */}
|
||||
<div className="ep-add-clip-duration-row">
|
||||
<span className="ep-add-clip-type-label">时长:</span>
|
||||
<input
|
||||
type="number"
|
||||
className="ep-duration-input"
|
||||
min={minDuration}
|
||||
max={maxDuration}
|
||||
value={addDuration}
|
||||
onChange={(e) =>
|
||||
onDurationChange(
|
||||
Math.max(minDuration, Math.min(maxDuration, Number(e.target.value) || minDuration)),
|
||||
)
|
||||
}
|
||||
/>
|
||||
<span className="ep-add-clip-duration-unit">秒</span>
|
||||
</div>
|
||||
|
||||
{/* 确认按钮 */}
|
||||
<button className="ep-add-clip-confirm-btn" onClick={onConfirm}>
|
||||
添加
|
||||
|
||||
@@ -105,14 +105,11 @@ export const ClipCard: React.FC<ClipCardProps> = ({
|
||||
<span className="ep-clip-name">
|
||||
{CLIP_TYPE_LABELS[clip.type] || "片段"} {idx + 1}
|
||||
</span>
|
||||
<span className="ep-clip-duration">
|
||||
{clip.duration}s
|
||||
{hasTrim && (
|
||||
<span className="ep-trim-indicator" title="已裁剪">
|
||||
✂
|
||||
</span>
|
||||
)}
|
||||
</span>
|
||||
{hasTrim && (
|
||||
<span className="ep-trim-indicator" title="已裁剪">
|
||||
✂
|
||||
</span>
|
||||
)}
|
||||
</div>
|
||||
|
||||
{/* 速度徽章 */}
|
||||
|
||||
@@ -26,8 +26,6 @@ export function useAddPicker({ currentMode, onAddClip }: UseAddPickerOptions) {
|
||||
}, [currentMode])
|
||||
|
||||
const [addType, setAddType] = useState<ClipType>(defaultAddType)
|
||||
const [addDuration, setAddDuration] = useState<number>(DEFAULT_ADD_DURATION)
|
||||
|
||||
useEffect(() => {
|
||||
if (!availableTypes.includes(addType)) {
|
||||
setAddType(defaultAddType)
|
||||
@@ -95,9 +93,9 @@ export function useAddPicker({ currentMode, onAddClip }: UseAddPickerOptions) {
|
||||
}, [showAddPicker])
|
||||
|
||||
const handleConfirmAdd = useCallback(() => {
|
||||
onAddClip(addType, addDuration)
|
||||
onAddClip(addType, DEFAULT_ADD_DURATION)
|
||||
setShowAddPicker(false)
|
||||
}, [onAddClip, addType, addDuration])
|
||||
}, [onAddClip, addType])
|
||||
|
||||
return {
|
||||
showAddPicker,
|
||||
@@ -107,9 +105,8 @@ export function useAddPicker({ currentMode, onAddClip }: UseAddPickerOptions) {
|
||||
pickerPos,
|
||||
availableTypes,
|
||||
addType,
|
||||
addDuration,
|
||||
addDuration: DEFAULT_ADD_DURATION,
|
||||
setAddType,
|
||||
setAddDuration,
|
||||
handleTogglePicker,
|
||||
handleConfirmAdd,
|
||||
}
|
||||
|
||||
@@ -35,7 +35,6 @@ export const useTimelineMenus = (
|
||||
addType,
|
||||
addDuration,
|
||||
setAddType,
|
||||
setAddDuration,
|
||||
handleTogglePicker,
|
||||
handleConfirmAdd,
|
||||
} = useAddPicker({ currentMode, onAddClip })
|
||||
@@ -57,7 +56,6 @@ export const useTimelineMenus = (
|
||||
addType,
|
||||
addDuration,
|
||||
setAddType,
|
||||
setAddDuration,
|
||||
handleTogglePicker,
|
||||
handleConfirmAdd,
|
||||
// 悬停状态
|
||||
|
||||
@@ -16,10 +16,6 @@ import { useQuery } from "@tanstack/react-query"
|
||||
import { useCloneProgress } from "@/hooks/useCloneProgress"
|
||||
import CloneModal from "@/components/voice/CloneModal"
|
||||
import GenerateHeader from "./components/GenerateHeader"
|
||||
import {
|
||||
calculateTotalVideoDuration,
|
||||
estimateTotalVideoDuration,
|
||||
} from "./utils/calculateTotalVideoDuration"
|
||||
import FrontendPreviewPlayer from "./components/FrontendPreviewPlayer"
|
||||
import GenerateStepsBar from "./components/GenerateStepsBar"
|
||||
import GenerateStepContent from "./components/GenerateStepContent"
|
||||
@@ -128,7 +124,6 @@ const GeneratePage: React.FC = () => {
|
||||
cancelled = true
|
||||
controller.abort()
|
||||
}
|
||||
// eslint-disable-next-line react-hooks/exhaustive-deps
|
||||
}, [selectedVoice, selectedClonedVoice, titleSettings.title, voiceMaterials])
|
||||
|
||||
/* ── 克隆声音 ── */
|
||||
@@ -174,13 +169,6 @@ const GeneratePage: React.FC = () => {
|
||||
[previewAssetsReady, currentTemplate],
|
||||
)
|
||||
|
||||
/* ── 视频总时长计算 ── */
|
||||
const totalVideoDuration = useMemo(() => {
|
||||
const exact = calculateTotalVideoDuration(previewAssets, currentTemplate ?? undefined)
|
||||
if (exact > 0) return exact
|
||||
return estimateTotalVideoDuration(currentTemplate ?? undefined)
|
||||
}, [previewAssets, currentTemplate])
|
||||
|
||||
/* ── 视频生成核心逻辑 ── */
|
||||
const {
|
||||
generating,
|
||||
@@ -291,7 +279,6 @@ const GeneratePage: React.FC = () => {
|
||||
onCoverSettingsChange={setCoverSettings}
|
||||
selectedVoice={selectedVoice}
|
||||
onSelectedVoiceChange={setSelectedVoice}
|
||||
totalVideoDuration={totalVideoDuration}
|
||||
onServerClipsChange={setServerClips}
|
||||
voiceMode={voiceMode}
|
||||
onVoiceModeChange={setVoiceMode}
|
||||
|
||||
@@ -591,6 +591,8 @@ const FrontendPreviewPlayer: React.FC<FrontendPreviewPlayerProps> = ({
|
||||
<div
|
||||
style={{
|
||||
position: "absolute",
|
||||
width: `${100 - 2 * titleSidePct}%`,
|
||||
maxWidth: `${100 - 2 * titleSidePct}%`,
|
||||
...(customTitleXPct != null && customTitleYPct != null
|
||||
? {
|
||||
left: `${customTitleXPct}%`,
|
||||
@@ -599,13 +601,13 @@ const FrontendPreviewPlayer: React.FC<FrontendPreviewPlayerProps> = ({
|
||||
textAlign: "center" as const,
|
||||
}
|
||||
: {
|
||||
left: `${titleSidePct}%`,
|
||||
right: `${titleSidePct}%`,
|
||||
left: "50%",
|
||||
transform: "translateX(-50%)",
|
||||
textAlign: "center" as const,
|
||||
...(titleSettings.position === "top"
|
||||
? { top: `${titleTopPct}%` }
|
||||
: titleSettings.position === "center"
|
||||
? { top: "50%", transform: "translateY(-50%)" }
|
||||
? { top: "50%", transform: "translate(-50%, -50%)" }
|
||||
: { bottom: `${titleBottomPct}%` }),
|
||||
}),
|
||||
pointerEvents: "auto",
|
||||
|
||||
@@ -49,7 +49,6 @@ export interface GenerateStepContentProps {
|
||||
/* 配音 */
|
||||
selectedVoice: string
|
||||
onSelectedVoiceChange: (id: string) => void
|
||||
totalVideoDuration?: number
|
||||
onServerClipsChange: (clips: EditPlanClip[]) => void
|
||||
voiceMode: "preset" | "custom" | "clone"
|
||||
onVoiceModeChange: (mode: "preset" | "custom" | "clone") => void
|
||||
@@ -104,7 +103,6 @@ export const GenerateStepContent: React.FC<GenerateStepContentProps> = (props) =
|
||||
onCoverSettingsChange,
|
||||
selectedVoice,
|
||||
onSelectedVoiceChange,
|
||||
totalVideoDuration,
|
||||
onServerClipsChange,
|
||||
voiceMode,
|
||||
selectedClonedVoice,
|
||||
@@ -151,7 +149,6 @@ export const GenerateStepContent: React.FC<GenerateStepContentProps> = (props) =
|
||||
<Step3VoiceSelect
|
||||
selectedVoice={selectedVoice}
|
||||
onSelectedVoiceChange={onSelectedVoiceChange}
|
||||
totalVideoDuration={totalVideoDuration}
|
||||
/>
|
||||
)
|
||||
case 4:
|
||||
|
||||
@@ -47,9 +47,7 @@ const Step1TemplateSelect: React.FC<Step1TemplateSelectProps> = (props) => {
|
||||
🎬
|
||||
</div>
|
||||
<h4>{tpl.name}</h4>
|
||||
<p>
|
||||
{tpl.estimated_duration}s · {tpl.segments.length}片段
|
||||
</p>
|
||||
<p>{tpl.segments.length}片段</p>
|
||||
{tpl.tags.length > 0 && (
|
||||
<div
|
||||
style={{
|
||||
|
||||
@@ -5,18 +5,15 @@
|
||||
import React, { useState, useRef, useCallback } from "react"
|
||||
import { useNavigate } from "react-router-dom"
|
||||
import { useQuery } from "@tanstack/react-query"
|
||||
import { AudioOutlined, SoundOutlined, WarningOutlined } from "@ant-design/icons"
|
||||
import { Modal } from "antd"
|
||||
import { AudioOutlined, SoundOutlined } from "@ant-design/icons"
|
||||
import { getAssetsByKind } from "@/api/assets"
|
||||
import type { AssetItem } from "@/api/assets"
|
||||
|
||||
interface Step5VoiceSelectProps {
|
||||
selectedVoice: string
|
||||
onSelectedVoiceChange: (id: string) => void
|
||||
totalVideoDuration?: number
|
||||
}
|
||||
|
||||
/** 格式化时长 mm:ss */
|
||||
/** 获取素材实际时长(优先顶层 duration,fallback 到 metadata.duration) */
|
||||
const getDuration = (item: AssetItem): number => {
|
||||
return item.duration ?? (item.metadata?.duration as number) ?? 0
|
||||
@@ -34,13 +31,6 @@ const isAiVoice = (item: AssetItem): boolean => {
|
||||
return (!duration || duration <= 0) && (!size || size <= 0)
|
||||
}
|
||||
|
||||
const formatDuration = (seconds?: number): string => {
|
||||
if (!seconds || seconds <= 0) return "00:00"
|
||||
const m = Math.floor(seconds / 60)
|
||||
const s = Math.floor(seconds % 60)
|
||||
return `${String(m).padStart(2, "0")}:${String(s).padStart(2, "0")}`
|
||||
}
|
||||
|
||||
/** 格式化文件大小 */
|
||||
const formatFileSize = (bytes?: number): string => {
|
||||
if (!bytes || bytes <= 0) return "未知"
|
||||
@@ -53,13 +43,10 @@ const formatFileSize = (bytes?: number): string => {
|
||||
const Step5VoiceSelect: React.FC<Step5VoiceSelectProps> = ({
|
||||
selectedVoice,
|
||||
onSelectedVoiceChange,
|
||||
totalVideoDuration = 0,
|
||||
}) => {
|
||||
const navigate = useNavigate()
|
||||
const [playingId, setPlayingId] = useState<string | null>(null)
|
||||
const audioRef = useRef<HTMLAudioElement | null>(null)
|
||||
const [durationWarningOpen, setDurationWarningOpen] = useState(false)
|
||||
const [pendingVoiceId, setPendingVoiceId] = useState<string | null>(null)
|
||||
|
||||
// 获取用户上传的配音素材
|
||||
const { data: materials = [], isLoading } = useQuery({
|
||||
@@ -100,38 +87,14 @@ const Step5VoiceSelect: React.FC<Step5VoiceSelectProps> = ({
|
||||
[playingId],
|
||||
)
|
||||
|
||||
/** 选中素材(含时长校验) */
|
||||
/** 选中素材(直接选中,不再做时长校验弹窗) */
|
||||
const handleSelect = useCallback(
|
||||
(id: string) => {
|
||||
// 如果启用了时长校验,且配音时长不足(AI 音色按脚本实时合成,不参与时长校验)
|
||||
if (totalVideoDuration > 0) {
|
||||
const material = materials.find((m) => m.id === id)
|
||||
if (material && !isAiVoice(material) && getDuration(material) < totalVideoDuration) {
|
||||
setPendingVoiceId(id)
|
||||
setDurationWarningOpen(true)
|
||||
return
|
||||
}
|
||||
}
|
||||
onSelectedVoiceChange(id)
|
||||
},
|
||||
[onSelectedVoiceChange, totalVideoDuration, materials],
|
||||
[onSelectedVoiceChange],
|
||||
)
|
||||
|
||||
/** 确认使用时长不足的配音 */
|
||||
const handleConfirmUseAnyway = useCallback(() => {
|
||||
if (pendingVoiceId) {
|
||||
onSelectedVoiceChange(pendingVoiceId)
|
||||
}
|
||||
setDurationWarningOpen(false)
|
||||
setPendingVoiceId(null)
|
||||
}, [pendingVoiceId, onSelectedVoiceChange])
|
||||
|
||||
/** 取消选择 */
|
||||
const handleCancelSelection = useCallback(() => {
|
||||
setDurationWarningOpen(false)
|
||||
setPendingVoiceId(null)
|
||||
}, [])
|
||||
|
||||
/** 跳转到配音库上传 */
|
||||
const handleGoToUpload = useCallback(() => {
|
||||
navigate("/app/voices?tab=material&upload=1")
|
||||
@@ -277,7 +240,7 @@ const Step5VoiceSelect: React.FC<Step5VoiceSelectProps> = ({
|
||||
{item.name}
|
||||
</div>
|
||||
|
||||
{/* 时长 + 大小 */}
|
||||
{/* 文件大小 */}
|
||||
<div
|
||||
style={{
|
||||
display: "flex",
|
||||
@@ -289,65 +252,13 @@ const Step5VoiceSelect: React.FC<Step5VoiceSelectProps> = ({
|
||||
>
|
||||
{isAiVoice(item) ? (
|
||||
<span style={{ color: "#1677ff", fontWeight: 500 }}>AI 音色</span>
|
||||
) : (
|
||||
<span style={{ display: "flex", alignItems: "center", gap: 4 }}>
|
||||
{formatDuration(getDuration(item))}
|
||||
{totalVideoDuration > 0 && getDuration(item) < Number(totalVideoDuration) && (
|
||||
<span
|
||||
style={{
|
||||
color: "#ff4d4f",
|
||||
fontSize: 11,
|
||||
fontWeight: 500,
|
||||
display: "inline-flex",
|
||||
alignItems: "center",
|
||||
gap: 2,
|
||||
}}
|
||||
>
|
||||
<WarningOutlined />
|
||||
时长不足
|
||||
</span>
|
||||
)}
|
||||
</span>
|
||||
)}
|
||||
) : null}
|
||||
<span>{isAiVoice(item) ? "按文本合成" : formatFileSize(getFileSize(item))}</span>
|
||||
</div>
|
||||
</div>
|
||||
)
|
||||
})}
|
||||
</div>
|
||||
|
||||
{/* 时长不足警告弹窗 */}
|
||||
<Modal
|
||||
title={
|
||||
<span style={{ display: "flex", alignItems: "center", gap: 8 }}>
|
||||
<WarningOutlined style={{ color: "#faad14" }} />
|
||||
配音时长不足
|
||||
</span>
|
||||
}
|
||||
open={durationWarningOpen}
|
||||
onOk={handleConfirmUseAnyway}
|
||||
onCancel={handleCancelSelection}
|
||||
okText="仍要使用"
|
||||
cancelText="重新选择"
|
||||
okButtonProps={{ danger: true }}
|
||||
>
|
||||
{(() => {
|
||||
const pendingMaterial = pendingVoiceId
|
||||
? materials.find((m) => m.id === pendingVoiceId)
|
||||
: null
|
||||
return (
|
||||
<p>
|
||||
该配音时长(
|
||||
<strong>
|
||||
{pendingMaterial ? formatDuration(getDuration(pendingMaterial)) : "--"}
|
||||
</strong>
|
||||
)短于视频总时长(
|
||||
<strong>{formatDuration(totalVideoDuration)}</strong>
|
||||
),播放时配音可能提前结束,建议选择更长的配音素材。
|
||||
</p>
|
||||
)
|
||||
})()}
|
||||
</Modal>
|
||||
</div>
|
||||
)
|
||||
}
|
||||
|
||||
@@ -35,7 +35,13 @@ export const useTaskHistory = () => {
|
||||
} = useQuery<TaskItem[], Error>({
|
||||
queryKey: ["tasks"],
|
||||
queryFn: getUserTasks,
|
||||
staleTime: 30_000,
|
||||
staleTime: 5_000,
|
||||
// 有进行中任务时每 3 秒自动刷新,全部结束后停止轮询
|
||||
refetchInterval: (query) => {
|
||||
const list = query.state.data ?? []
|
||||
const hasActive = list.some((t) => ["pending", "waiting", "running"].includes(t.status))
|
||||
return hasActive ? 3_000 : false
|
||||
},
|
||||
})
|
||||
|
||||
// 重试 mutation
|
||||
|
||||
@@ -11,7 +11,7 @@
|
||||
* 产品卡片 → components/ProductCard(内联视频播放)
|
||||
*/
|
||||
import React from "react"
|
||||
import { VideoCameraOutlined, DownloadOutlined } from "@ant-design/icons"
|
||||
import { VideoCameraOutlined, DownloadOutlined, ReloadOutlined } from "@ant-design/icons"
|
||||
import { Button } from "@/components/ui"
|
||||
import { ProductCard } from "./components/ProductCard"
|
||||
import { ProductFilterBar } from "./components/ProductFilterBar"
|
||||
@@ -19,6 +19,7 @@ import { ProductBatchBar } from "./components/ProductBatchBar"
|
||||
import { ProductEmptyState } from "./components/ProductEmptyState"
|
||||
import { useProductList } from "./hooks/useProductList"
|
||||
import { useProductActions } from "./hooks/useProductActions"
|
||||
import { useRecomputeDedup } from "./hooks/product-actions/useRecomputeDedup"
|
||||
import "./products.css"
|
||||
|
||||
const ProductLibrary: React.FC = () => {
|
||||
@@ -67,6 +68,8 @@ const ProductLibrary: React.FC = () => {
|
||||
setPlayingProduct: () => {}, // 不再使用弹窗播放
|
||||
})
|
||||
|
||||
const { recomputeDedup, isRecomputing } = useRecomputeDedup()
|
||||
|
||||
// ── Loading 状态 ──
|
||||
if (isLoading) {
|
||||
return <ProductEmptyState type="loading" />
|
||||
@@ -94,6 +97,15 @@ const ProductLibrary: React.FC = () => {
|
||||
<Button buttonType="ghost" buttonSize="sm" icon={<DownloadOutlined />}>
|
||||
批量导出
|
||||
</Button>
|
||||
<Button
|
||||
buttonType="ghost"
|
||||
buttonSize="sm"
|
||||
icon={<ReloadOutlined />}
|
||||
loading={isRecomputing}
|
||||
onClick={recomputeDedup}
|
||||
>
|
||||
重新查重
|
||||
</Button>
|
||||
</div>
|
||||
</div>
|
||||
|
||||
|
||||
@@ -2,6 +2,7 @@ import React from "react"
|
||||
import type { ProductItem } from "../../../api/products"
|
||||
import { STATUS_MAP } from "../constants"
|
||||
import { formatDuration, formatFileSize, formatDate } from "../detailUtils"
|
||||
import { getRiskLevel } from "../../duplication/utils"
|
||||
|
||||
interface ProductInfoPanelProps {
|
||||
product: ProductItem
|
||||
@@ -44,12 +45,28 @@ export const ProductInfoPanel: React.FC<ProductInfoPanelProps> = ({ product }) =
|
||||
</div>
|
||||
<div className="xx-detail-meta-item">
|
||||
<span className="xx-detail-meta-label">查重率</span>
|
||||
<span className="xx-detail-meta-value">
|
||||
<span
|
||||
className={`xx-detail-meta-value dup-risk-text dup-risk-${getRiskLevel(product.duplicate_rate)}`}
|
||||
>
|
||||
{(product.duplicate_rate ?? 0) > 0
|
||||
? `${(product.duplicate_rate ?? 0).toFixed(1)}%`
|
||||
: "-"}
|
||||
</span>
|
||||
</div>
|
||||
{product.visual_similarity != null && (
|
||||
<div className="xx-detail-meta-item">
|
||||
<span className="xx-detail-meta-label">视觉相似度</span>
|
||||
<span className="xx-detail-meta-value">
|
||||
{(product.visual_similarity * 100).toFixed(1)}%
|
||||
</span>
|
||||
</div>
|
||||
)}
|
||||
{product.match_count != null && (
|
||||
<div className="xx-detail-meta-item">
|
||||
<span className="xx-detail-meta-label">匹配帧数</span>
|
||||
<span className="xx-detail-meta-value">{product.match_count}</span>
|
||||
</div>
|
||||
)}
|
||||
<div className="xx-detail-meta-item">
|
||||
<span className="xx-detail-meta-label">创建时间</span>
|
||||
<span className="xx-detail-meta-value">{formatDate(product.created_at ?? "")}</span>
|
||||
|
||||
@@ -0,0 +1,27 @@
|
||||
import { useMutation, useQueryClient } from "@tanstack/react-query"
|
||||
import { message } from "antd"
|
||||
import { recomputeDedup } from "@/api/products"
|
||||
|
||||
export function useRecomputeDedup() {
|
||||
const queryClient = useQueryClient()
|
||||
|
||||
const mutation = useMutation({
|
||||
mutationFn: () => recomputeDedup(),
|
||||
onSuccess: (data) => {
|
||||
queryClient.invalidateQueries({ queryKey: ["products"] })
|
||||
if (data.enqueued > 0) {
|
||||
message.success(`已提交 ${data.enqueued} 个视频的查重任务,后台处理中`)
|
||||
} else {
|
||||
message.info("所有视频查重率已是最新,无需重算")
|
||||
}
|
||||
},
|
||||
onError: () => {
|
||||
message.error("查重任务提交失败,请稍后重试")
|
||||
},
|
||||
})
|
||||
|
||||
return {
|
||||
recomputeDedup: () => mutation.mutate(),
|
||||
isRecomputing: mutation.isPending,
|
||||
}
|
||||
}
|
||||
@@ -1076,3 +1076,19 @@
|
||||
gap: var(--space-sm);
|
||||
}
|
||||
}
|
||||
|
||||
/* 查重率风险颜色(#1662) */
|
||||
.xx-detail-meta-value.dup-risk-low {
|
||||
color: var(--success-color, #22c55e);
|
||||
font-weight: 600;
|
||||
}
|
||||
|
||||
.xx-detail-meta-value.dup-risk-medium {
|
||||
color: var(--warning-color, #f59e0b);
|
||||
font-weight: 600;
|
||||
}
|
||||
|
||||
.xx-detail-meta-value.dup-risk-high {
|
||||
color: var(--error-color, #ef4444);
|
||||
font-weight: 600;
|
||||
}
|
||||
|
||||
@@ -136,8 +136,9 @@ export const MaterialVoiceTab: React.FC<MaterialVoiceTabProps> = ({
|
||||
const material = mapAssetToMaterial(asset)
|
||||
// duration 优先取顶层(后端从 metadata 提取),兜底 metadata
|
||||
const cardDuration = asset.duration || material.duration || 0
|
||||
// AI 生成素材标识(metadata.source === "tts_job")
|
||||
const isAiMaterial = (asset.metadata as Record<string, unknown>)?.source === "tts_job"
|
||||
// AI 生成素材标识:兼容旧素材(无 source 字段但有 tts_job_id)
|
||||
const meta = asset.metadata as Record<string, unknown>
|
||||
const isAiMaterial = meta?.source === "tts_job" || !!meta?.tts_job_id
|
||||
const isPlaying = playingId === asset.id
|
||||
const isSelected = selectedIds.has(asset.id)
|
||||
// 播放中以 audio 真实时长为准,未播放显示卡片时长
|
||||
@@ -184,8 +185,10 @@ export const MaterialVoiceTab: React.FC<MaterialVoiceTabProps> = ({
|
||||
</div>
|
||||
|
||||
<div className="xx-voice-info vmat-info">
|
||||
<div className="xx-voice-name" title={asset.name}>
|
||||
{asset.name}
|
||||
<div className="xx-voice-name-row">
|
||||
<div className="xx-voice-name" title={asset.name}>
|
||||
{asset.name}
|
||||
</div>
|
||||
{isAiMaterial && <span className="vmat-ai-badge">AI</span>}
|
||||
</div>
|
||||
<div className="xx-voice-subtitle">
|
||||
|
||||
@@ -30,6 +30,7 @@ const VideoExtractModal: React.FC<VideoExtractModalProps> = ({
|
||||
title={<span style={{ fontSize: 16, fontWeight: 600 }}>提取视频配音</span>}
|
||||
open={open}
|
||||
onCancel={() => {
|
||||
if (inputRef.current) inputRef.current.value = ""
|
||||
if (isExtracting) return
|
||||
onClose()
|
||||
}}
|
||||
@@ -152,7 +153,7 @@ const VideoExtractModal: React.FC<VideoExtractModalProps> = ({
|
||||
|
||||
{isExtracting && (
|
||||
<p style={{ textAlign: "center", fontSize: 13, color: "#7c3aed", margin: "12px 0 0" }}>
|
||||
{progress === 100 ? "正在提取人声,请稍候..." : "正在上传视频..."}
|
||||
{"正在提取音频,请稍后..."}
|
||||
</p>
|
||||
)}
|
||||
</div>
|
||||
|
||||
@@ -102,6 +102,7 @@ vi.mock("@ant-design/icons", () => ({
|
||||
SearchOutlined: () => <span />,
|
||||
ShareAltOutlined: () => <span />,
|
||||
VideoCameraOutlined: () => <span />,
|
||||
ReloadOutlined: () => <span />,
|
||||
}))
|
||||
|
||||
vi.mock("@/store/authStore", () => ({
|
||||
@@ -116,6 +117,9 @@ vi.mock("@/api/products", () => ({
|
||||
updateReviewStatus: vi.fn().mockResolvedValue({ items: [], total: 0, success: true }),
|
||||
batchDownload: vi.fn().mockResolvedValue({ items: [], total: 0, success: true }),
|
||||
getBatchDownloadStatus: vi.fn().mockResolvedValue({ items: [], total: 0, success: true }),
|
||||
recomputeDedup: vi
|
||||
.fn()
|
||||
.mockResolvedValue({ enqueued: 0, total_scanned: 0, skipped: 0, message: "" }),
|
||||
}))
|
||||
|
||||
vi.mock("@/pages/products/ProductLibrary.css", () => ({}))
|
||||
|
||||
@@ -104,6 +104,9 @@ vi.mock("@ant-design/icons", () => ({
|
||||
SoundOutlined: () => <span />,
|
||||
UploadOutlined: () => <span />,
|
||||
UserOutlined: () => <span />,
|
||||
VideoCameraOutlined: () => <span />,
|
||||
InboxOutlined: () => <span />,
|
||||
CloseOutlined: () => <span />,
|
||||
}))
|
||||
|
||||
vi.mock("@/store/authStore", () => ({
|
||||
|
||||
@@ -0,0 +1,29 @@
|
||||
import { describe, it, expect } from "vitest"
|
||||
import { getRiskLevel } from "@/pages/duplication/utils"
|
||||
|
||||
describe("getRiskLevel (#1662 阈值 <15 / 15-30 / >30)", () => {
|
||||
it("undefined 返回 low(兼容无数据)", () => {
|
||||
expect(getRiskLevel(undefined)).toBe("low")
|
||||
})
|
||||
|
||||
it("<15% 为低风险", () => {
|
||||
expect(getRiskLevel(0)).toBe("low")
|
||||
expect(getRiskLevel(10)).toBe("low")
|
||||
expect(getRiskLevel(14.9)).toBe("low")
|
||||
})
|
||||
|
||||
it("15% 边界为中风险", () => {
|
||||
expect(getRiskLevel(15)).toBe("medium")
|
||||
})
|
||||
|
||||
it("15-30% 为中风险", () => {
|
||||
expect(getRiskLevel(20)).toBe("medium")
|
||||
expect(getRiskLevel(30)).toBe("medium")
|
||||
})
|
||||
|
||||
it(">30% 为高风险", () => {
|
||||
expect(getRiskLevel(30.1)).toBe("high")
|
||||
expect(getRiskLevel(80)).toBe("high")
|
||||
expect(getRiskLevel(100)).toBe("high")
|
||||
})
|
||||
})
|
||||
File diff suppressed because it is too large
Load Diff
@@ -2,6 +2,9 @@
|
||||
|
||||
供 generate_video 共同复用,
|
||||
创建 GeneratedVideo 记录后计算指纹并执行项目级 + 批次内查重。
|
||||
|
||||
v2: 两阶段持久化 — 先计算所有查重数据,再一次性 commit,
|
||||
避免中间异常导致 duplicate_rate 等字段缺失。
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
@@ -34,24 +37,14 @@ def create_video_record_and_dedup(
|
||||
) -> int:
|
||||
"""创建 GeneratedVideo 记录,计算指纹并执行查重(历史 + 批次)。
|
||||
|
||||
Args:
|
||||
generation_task_id: 生成任务 ID
|
||||
project_id: 项目 ID
|
||||
batch_id: 批次 ID(可为空字符串)
|
||||
file_url: 视频文件 URL
|
||||
file_size: 文件大小(字节)
|
||||
duration: 视频时长(秒)
|
||||
video_path: 视频本地路径(用于计算指纹)
|
||||
mode: 剪辑模式名称
|
||||
session: 数据库会话
|
||||
width: 视频宽度
|
||||
height: 视频高度
|
||||
fps: 视频帧率
|
||||
采用两阶段持久化:先计算所有指纹/查重数据(内存),
|
||||
再一次性写入数据库并 commit。若指纹计算失败,
|
||||
视频记录仍会创建(无查重数据),但保证不会出现"写了记录却没 commit"的中间态。
|
||||
|
||||
Returns:
|
||||
创建的视频记录数量(1 表示成功,0 表示失败)
|
||||
"""
|
||||
from video_processing.dedup import VideoDeduplicator
|
||||
from video_processing.dedup import VideoDeduplicator, _save_fingerprint_chunks
|
||||
|
||||
from packages.adapters.sqlalchemy_impl.generated_video_repository import (
|
||||
SQLAlchemyGeneratedVideoRepository,
|
||||
@@ -60,8 +53,9 @@ def create_video_record_and_dedup(
|
||||
|
||||
try:
|
||||
video_id = uuid4().hex
|
||||
# 使用传入的名称,没有则 fallback 到默认命名
|
||||
video_name = name.strip() if name else f"generated-{generation_task_id[:8]}.mp4"
|
||||
|
||||
# ── Phase 1: 构建视频记录(内存,不 commit) ────────────────
|
||||
generated_video = GeneratedVideo(
|
||||
id=video_id,
|
||||
project_id=project_id,
|
||||
@@ -76,73 +70,94 @@ def create_video_record_and_dedup(
|
||||
fps=fps,
|
||||
status="completed",
|
||||
generation_params={"mode": mode},
|
||||
thumbnail_url=thumbnail_url or None,
|
||||
)
|
||||
|
||||
video_repo = SQLAlchemyGeneratedVideoRepository(session)
|
||||
video_repo.create(generated_video)
|
||||
|
||||
# 生成封面缩略图
|
||||
if thumbnail_url:
|
||||
generated_video.thumbnail_url = thumbnail_url
|
||||
video_repo.update_thumbnail(video_id, thumbnail_url)
|
||||
logger.info("Thumbnail set for video %s: %s", video_id, thumbnail_url[:80] if thumbnail_url else "")
|
||||
else:
|
||||
logger.debug("No thumbnail_url provided for video %s, skipping", video_id)
|
||||
|
||||
# 计算视频指纹
|
||||
# ── Phase 2: 计算指纹 & 查重(全部在内存) ────────────────
|
||||
deduplicator = VideoDeduplicator()
|
||||
fingerprint = None
|
||||
|
||||
try:
|
||||
fingerprint = deduplicator.compute_fingerprint(video_path)
|
||||
except Exception as fp_err:
|
||||
logger.warning("Fingerprint computation failed for %s: %s", video_id, fp_err)
|
||||
session.commit()
|
||||
return 1
|
||||
|
||||
generated_video.video_fingerprint = fingerprint.to_dict()
|
||||
if fingerprint is not None:
|
||||
generated_video.video_fingerprint = fingerprint.to_dict()
|
||||
|
||||
# (a) 历史成片查重
|
||||
duplicate_result = deduplicator.check_duplicate(fingerprint, project_id, session)
|
||||
# 写入分片指纹表(失败不阻塞)
|
||||
try:
|
||||
_save_fingerprint_chunks(fingerprint, video_id, project_id, user_id, session)
|
||||
except Exception as chunk_err:
|
||||
logger.warning("Failed to save fingerprint chunks for %s: %s", video_id, chunk_err)
|
||||
|
||||
# (b) 批次内查重(仅当有 batch_id 时)
|
||||
if not duplicate_result and batch_id:
|
||||
duplicate_result = deduplicator.check_batch_duplicate(fingerprint, batch_id, video_id, session)
|
||||
|
||||
if duplicate_result:
|
||||
generated_video.is_duplicate = True
|
||||
generated_video.duplicate_of = duplicate_result["duplicate_of"]
|
||||
logger.info(
|
||||
"Duplicate detected: %s -> %s (reason=%s, similarity=%.3f)",
|
||||
video_id,
|
||||
duplicate_result["duplicate_of"],
|
||||
duplicate_result["reason"],
|
||||
duplicate_result["similarity"],
|
||||
)
|
||||
else:
|
||||
generated_video.is_duplicate = False
|
||||
generated_video.duplicate_of = None
|
||||
|
||||
# 计算重复率百分比(与项目内所有已有视频对比取最高相似度)
|
||||
try:
|
||||
dup_rate = deduplicator.compute_duplicate_rate(
|
||||
# (a) 历史成片查重(跨项目全局 + 时长预过滤)
|
||||
duration_sec = fingerprint.duration / 1000 if fingerprint.duration else 0
|
||||
duplicate_result = deduplicator.check_duplicate(
|
||||
fingerprint,
|
||||
project_id,
|
||||
video_id,
|
||||
session,
|
||||
scope="user",
|
||||
user_id=user_id,
|
||||
duration_sec=duration_sec,
|
||||
)
|
||||
generated_video.duplicate_rate = dup_rate
|
||||
logger.info("Duplicate rate for %s: %.2f%%", video_id, dup_rate)
|
||||
except Exception as rate_err:
|
||||
logger.warning("Failed to compute duplicate_rate for %s: %s", video_id, rate_err)
|
||||
generated_video.duplicate_rate = None
|
||||
|
||||
video_repo.update(generated_video)
|
||||
# (b) 批次内查重(仅当有 batch_id 时)
|
||||
if not duplicate_result and batch_id:
|
||||
duplicate_result = deduplicator.check_batch_duplicate(fingerprint, batch_id, video_id, session)
|
||||
|
||||
if duplicate_result:
|
||||
generated_video.is_duplicate = True
|
||||
generated_video.duplicate_of = duplicate_result["duplicate_of"]
|
||||
logger.info(
|
||||
"Duplicate detected: %s -> %s (reason=%s, similarity=%.3f)",
|
||||
video_id,
|
||||
duplicate_result["duplicate_of"],
|
||||
duplicate_result["reason"],
|
||||
duplicate_result["similarity"],
|
||||
)
|
||||
else:
|
||||
generated_video.is_duplicate = False
|
||||
generated_video.duplicate_of = None
|
||||
|
||||
# 计算重复率百分比(跨项目全局)
|
||||
try:
|
||||
rate_result = deduplicator.compute_duplicate_rate(
|
||||
fingerprint,
|
||||
project_id,
|
||||
video_id,
|
||||
session,
|
||||
scope="user",
|
||||
user_id=user_id,
|
||||
)
|
||||
generated_video.duplicate_rate = rate_result["duplicate_rate"]
|
||||
generated_video.match_count = rate_result["match_count"]
|
||||
generated_video.visual_similarity = rate_result["visual_similarity"]
|
||||
logger.info(
|
||||
"Duplicate rate for %s: %.2f%% (visual_sim=%.3f, matches=%d)",
|
||||
video_id,
|
||||
rate_result["duplicate_rate"],
|
||||
rate_result["visual_similarity"],
|
||||
rate_result["match_count"],
|
||||
)
|
||||
except Exception as rate_err:
|
||||
logger.warning("Failed to compute duplicate_rate for %s: %s", video_id, rate_err)
|
||||
generated_video.duplicate_rate = None
|
||||
|
||||
# ── Phase 3: 一次性持久化 ─────────────────────────────────
|
||||
video_repo = SQLAlchemyGeneratedVideoRepository(session)
|
||||
video_repo.create(generated_video)
|
||||
|
||||
if thumbnail_url:
|
||||
logger.info("Thumbnail set for video %s: %s", video_id, thumbnail_url[:80])
|
||||
|
||||
session.commit()
|
||||
logger.info(
|
||||
"GeneratedVideo record created: %s (task=%s, dup=%s)",
|
||||
"GeneratedVideo record created: %s (task=%s, dup=%s, rate=%s)",
|
||||
video_id,
|
||||
generation_task_id,
|
||||
generated_video.is_duplicate,
|
||||
generated_video.duplicate_rate,
|
||||
)
|
||||
return 1
|
||||
except Exception as e:
|
||||
|
||||
@@ -304,3 +304,127 @@ def normalize_video(
|
||||
]
|
||||
run_ffmpeg(command)
|
||||
return {"width": width, "height": height, "path": output_path}
|
||||
|
||||
|
||||
def random_edge_crop(
|
||||
input_path: str | Path,
|
||||
output_path: str | Path | None = None,
|
||||
*,
|
||||
min_crop_pct: float = 0.02,
|
||||
max_crop_pct: float = 0.05,
|
||||
) -> Path:
|
||||
"""对视频四边做随机裁剪再缩放回原分辨率,用于改变 pHash 指纹。
|
||||
|
||||
Args:
|
||||
input_path: 输入视频路径
|
||||
output_path: 输出路径;为 None 时写入 input_path 同目录的临时文件,
|
||||
成功后覆盖原文件
|
||||
min_crop_pct: 每边最小裁剪比例(默认 2%)
|
||||
max_crop_pct: 每边最大裁剪比例(默认 5%)
|
||||
|
||||
Returns:
|
||||
输出文件路径(Path 对象)
|
||||
|
||||
Raises:
|
||||
subprocess.CalledProcessError: ffmpeg 执行失败时抛出
|
||||
"""
|
||||
import random
|
||||
import shutil
|
||||
import tempfile
|
||||
|
||||
input_path = Path(input_path)
|
||||
|
||||
# 获取原始分辨率
|
||||
info = probe_video_info(str(input_path))
|
||||
W = info["width"]
|
||||
H = info["height"]
|
||||
|
||||
if W <= 0 or H <= 0:
|
||||
logger.warning("无法获取视频分辨率 (W=%d H=%d),跳过裁剪: %s", W, H, input_path)
|
||||
return input_path
|
||||
|
||||
# 四边各自随机裁剪 2%~5%
|
||||
crop_top = int(H * random.uniform(min_crop_pct, max_crop_pct))
|
||||
crop_bottom = int(H * random.uniform(min_crop_pct, max_crop_pct))
|
||||
crop_left = int(W * random.uniform(min_crop_pct, max_crop_pct))
|
||||
crop_right = int(W * random.uniform(min_crop_pct, max_crop_pct))
|
||||
|
||||
# 裁剪后尺寸(确保至少 2 像素)
|
||||
new_w = max(W - crop_left - crop_right, 2)
|
||||
new_h = max(H - crop_top - crop_bottom, 2)
|
||||
x_offset = crop_left
|
||||
y_offset = crop_top
|
||||
|
||||
# 确保裁剪尺寸为偶数(ffmpeg 编码器常要求偶数尺寸)
|
||||
new_w = new_w if new_w % 2 == 0 else new_w - 1
|
||||
new_h = new_h if new_h % 2 == 0 else new_h - 1
|
||||
if new_w < 2:
|
||||
new_w = 2
|
||||
if new_h < 2:
|
||||
new_h = 2
|
||||
|
||||
# 输出分辨率必须与原始一致
|
||||
out_w = W if W % 2 == 0 else W + 1
|
||||
out_h = H if H % 2 == 0 else H + 1
|
||||
|
||||
vf = f"crop={new_w}:{new_h}:{x_offset}:{y_offset},scale={out_w}:{out_h}"
|
||||
|
||||
logger.info(
|
||||
"随机边缘裁剪: %s → crop(%d,%d,%d,%d)=%dx%d scale→%dx%d",
|
||||
input_path.name,
|
||||
crop_top,
|
||||
crop_bottom,
|
||||
crop_left,
|
||||
crop_right,
|
||||
new_w,
|
||||
new_h,
|
||||
out_w,
|
||||
out_h,
|
||||
)
|
||||
|
||||
# 确定输出路径
|
||||
if output_path is None:
|
||||
temp_fd, temp_path = tempfile.mkstemp(suffix=".mp4", dir=input_path.parent)
|
||||
import os
|
||||
|
||||
os.close(temp_fd)
|
||||
temp_output = Path(temp_path)
|
||||
replace_original = True
|
||||
else:
|
||||
temp_output = Path(output_path)
|
||||
replace_original = False
|
||||
|
||||
command = [
|
||||
FFMPEG_BIN,
|
||||
"-y",
|
||||
"-i",
|
||||
str(input_path),
|
||||
"-vf",
|
||||
vf,
|
||||
"-c:v",
|
||||
"libx264",
|
||||
"-preset",
|
||||
"fast",
|
||||
"-crf",
|
||||
"18",
|
||||
"-c:a",
|
||||
"copy",
|
||||
"-movflags",
|
||||
"+faststart",
|
||||
str(temp_output),
|
||||
]
|
||||
|
||||
try:
|
||||
run_ffmpeg(command)
|
||||
except Exception:
|
||||
# 裁剪失败时清理临时文件
|
||||
if temp_output.exists() and replace_original:
|
||||
temp_output.unlink(missing_ok=True)
|
||||
raise
|
||||
|
||||
# 成功 → 覆盖原文件
|
||||
if replace_original:
|
||||
shutil.move(str(temp_output), str(input_path))
|
||||
return input_path
|
||||
|
||||
return temp_output
|
||||
|
||||
@@ -213,7 +213,7 @@ def generate_ass_from_timeline(
|
||||
t_shadow.get("offset_x", 2) if t_shadow.get("enabled", False) else 0,
|
||||
t_shadow.get("offset_y", 2) if t_shadow.get("enabled", False) else 0,
|
||||
)
|
||||
t_alignment = position_to_ass_alignment(title_cfg.get("position", "top"))
|
||||
t_alignment = position_to_ass_alignment(title_cfg.get("position", "bottom"))
|
||||
|
||||
title_style_line = build_ass_style(
|
||||
"TitleStyle",
|
||||
|
||||
@@ -198,8 +198,13 @@ class UnifiedRenderService:
|
||||
# 2. 分组为 RenderLayers
|
||||
layers = self._group_clips_into_layers(resolved)
|
||||
|
||||
# 2.5 配音时长对齐:如果有配音素材,调整片段时长以匹配配音时长
|
||||
voice_duration = self._get_voice_audio_duration()
|
||||
if voice_duration > 0:
|
||||
self._align_clips_to_voice_duration(layers, voice_duration)
|
||||
|
||||
# 3. 计算视频总时长(用于字幕显示时长)
|
||||
video_duration = self._estimate_total_duration(layers)
|
||||
video_duration_final = self._estimate_total_duration(layers)
|
||||
# Debug: 输出各图层时长明细
|
||||
for layer in layers:
|
||||
layer_total = sum(UnifiedRenderService._clip_adjusted_duration(c) for c in layer.clips)
|
||||
@@ -215,16 +220,16 @@ class UnifiedRenderService:
|
||||
self.transition_duration,
|
||||
", ".join(clip_details),
|
||||
)
|
||||
logger.info("[debug] estimated video_duration=%.3f", video_duration)
|
||||
logger.info("[debug] estimated video_duration=%.3f", video_duration_final)
|
||||
|
||||
# 3.5 TTS 配音生成(如果配置了)
|
||||
self._maybe_add_voiceover_layer(layers, video_duration=video_duration)
|
||||
self._maybe_add_voiceover_layer(layers, video_duration=video_duration_final)
|
||||
|
||||
# 3.6 配音素材库音频(如果传入了本地路径)
|
||||
self._maybe_add_voice_library_layer(layers, video_duration=video_duration)
|
||||
self._maybe_add_voice_library_layer(layers, video_duration=video_duration_final)
|
||||
|
||||
# 4. 生成 ASS 字幕文件(如果有 title/subtitle 配置)
|
||||
ass_path = self._maybe_generate_ass(video_duration)
|
||||
ass_path = self._maybe_generate_ass(video_duration_final)
|
||||
|
||||
# 4.5 解析画中画配置
|
||||
pip_config = PiPConfig.from_dict((self.plan.config or {}).get("pip_config"))
|
||||
@@ -257,7 +262,7 @@ class UnifiedRenderService:
|
||||
# 先尝试 stream copy 优化(无重编码,性能提升 10 倍+)
|
||||
# 条件不满足或失败时回退到带滤镜的直通渲染
|
||||
stream_copy_ok = self._try_render_stream_copy(
|
||||
layers, output_path, ass_path=ass_path, video_duration=video_duration
|
||||
layers, output_path, ass_path=ass_path, video_duration=video_duration_final
|
||||
)
|
||||
if stream_copy_ok:
|
||||
used_stream_copy = True
|
||||
@@ -271,7 +276,7 @@ class UnifiedRenderService:
|
||||
layers,
|
||||
output_path,
|
||||
ass_path=ass_path,
|
||||
video_duration=video_duration,
|
||||
video_duration=video_duration_final,
|
||||
)
|
||||
else:
|
||||
filter_complex, input_args = self._build_filter_complex(layers, ass_path=ass_path)
|
||||
@@ -327,7 +332,7 @@ class UnifiedRenderService:
|
||||
from video_processing.ffmpeg_utils import run_ffmpeg
|
||||
|
||||
run_ffmpeg(extract_cmd)
|
||||
final_audio = mix_bgm_with_main(ctx, main_audio_path, bgm_cfg, video_duration)
|
||||
final_audio = mix_bgm_with_main(ctx, main_audio_path, bgm_cfg, video_duration_final)
|
||||
# 合并回视频
|
||||
|
||||
bgm_output = self.work_dir / f"rendered_{self.plan.id}_bgm.mp4"
|
||||
@@ -353,7 +358,7 @@ class UnifiedRenderService:
|
||||
audio_path = mix_audio(
|
||||
ctx,
|
||||
layers,
|
||||
video_duration,
|
||||
video_duration_final,
|
||||
bgm_path=self.bgm_path,
|
||||
bgm_config=bgm_config,
|
||||
audio_tracks_config=audio_tracks_config,
|
||||
@@ -487,6 +492,147 @@ class UnifiedRenderService:
|
||||
"""
|
||||
return _estimate_total_duration_pure(layers, self.transition_duration)
|
||||
|
||||
def _get_voice_audio_duration(self) -> float:
|
||||
"""获取配音音频文件的时长(秒)。
|
||||
|
||||
Returns:
|
||||
配音音频时长,如果无配音或探测失败则返回 0.0
|
||||
"""
|
||||
if not self.voiceover_audio_path:
|
||||
return 0.0
|
||||
|
||||
audio_path = Path(self.voiceover_audio_path)
|
||||
if not audio_path.exists() or audio_path.stat().st_size == 0:
|
||||
return 0.0
|
||||
|
||||
try:
|
||||
duration = probe_duration(audio_path)
|
||||
logger.info("[voice-align] 配音音频时长: %.3fs path=%s", duration, self.voiceover_audio_path)
|
||||
return duration
|
||||
except Exception as e:
|
||||
logger.warning("[voice-align] 探测配音音频时长失败: %s", e)
|
||||
return 0.0
|
||||
|
||||
def _align_clips_to_voice_duration(
|
||||
self,
|
||||
layers: list[RenderLayer],
|
||||
voice_duration: float,
|
||||
) -> None:
|
||||
"""调整片段时长以对齐配音时长。
|
||||
|
||||
核心逻辑:
|
||||
- 计算片段总时长与配音时长的比例
|
||||
- ±5% 以内不调整
|
||||
- ratio < 1(片段比配音长):按比例裁剪每段末尾
|
||||
- ratio > 1(片段比配音短):按比例慢放每段
|
||||
|
||||
Args:
|
||||
layers: 渲染图层列表
|
||||
voice_duration: 配音时长(秒)
|
||||
"""
|
||||
if voice_duration <= 0:
|
||||
return
|
||||
|
||||
# 只调整视频图层(main/broll/background),不调整音频图层
|
||||
video_layers = [layer for layer in layers if layer.role in ("main", "broll", "background")]
|
||||
if not video_layers:
|
||||
return
|
||||
|
||||
# 计算所有视频图层的总时长
|
||||
total_clips_duration = 0.0
|
||||
for layer in video_layers:
|
||||
for clip in layer.clips:
|
||||
clip_dur = self._clip_adjusted_duration(clip)
|
||||
total_clips_duration += clip_dur
|
||||
|
||||
if total_clips_duration <= 0:
|
||||
return
|
||||
|
||||
ratio = voice_duration / total_clips_duration
|
||||
|
||||
# ±5% 以内不调整
|
||||
if abs(ratio - 1.0) <= 0.05:
|
||||
logger.info(
|
||||
"[voice-align] 比例接近1:1,跳过调整: ratio=%.4f voice=%.3f clips=%.3f",
|
||||
ratio,
|
||||
voice_duration,
|
||||
total_clips_duration,
|
||||
)
|
||||
return
|
||||
|
||||
logger.info(
|
||||
"[voice-align] 开始调整片段时长: ratio=%.4f voice=%.3f clips=%.3f",
|
||||
ratio,
|
||||
voice_duration,
|
||||
total_clips_duration,
|
||||
)
|
||||
|
||||
# 收集所有视频 clip
|
||||
all_clips: list[tuple[RenderLayer, ResolvedClip]] = []
|
||||
for layer in video_layers:
|
||||
for clip in layer.clips:
|
||||
all_clips.append((layer, clip))
|
||||
|
||||
if not all_clips:
|
||||
return
|
||||
|
||||
if ratio < 1.0:
|
||||
# 片段比配音长,按比例裁剪每段末尾
|
||||
# 减少每个 clip 的 duration
|
||||
for _layer, clip in all_clips:
|
||||
old_duration = clip.duration if clip.duration > 0 else clip.actual_duration
|
||||
new_duration = old_duration * ratio
|
||||
|
||||
# 更新 duration
|
||||
clip.duration = max(0.1, new_duration) # 至少 0.1s
|
||||
|
||||
# 如果有 trim_config,也需要调整
|
||||
if clip.trim_config is not None:
|
||||
new_trim_duration = clip.trim_config.duration * ratio
|
||||
clip.trim_config = TrimConfig(
|
||||
start_time=clip.trim_config.start_time,
|
||||
duration=max(0.1, new_trim_duration),
|
||||
)
|
||||
|
||||
logger.debug(
|
||||
"[voice-align] trim clip=%s: %.3f -> %.3f",
|
||||
clip.clip_id,
|
||||
old_duration,
|
||||
clip.duration,
|
||||
)
|
||||
|
||||
else:
|
||||
# ratio > 1.0: 片段比配音短,按比例慢放每段
|
||||
# 降低 playback_speed
|
||||
for _layer, clip in all_clips:
|
||||
old_speed = clip.playback_speed if clip.playback_speed > 0 else 1.0
|
||||
# speed = old_speed / ratio 会使视频变慢(ratio > 1 时)
|
||||
new_speed = old_speed / ratio
|
||||
|
||||
# 下限 0.25x(避免过慢)
|
||||
new_speed = max(0.25, round(new_speed, 4))
|
||||
clip.playback_speed = new_speed
|
||||
|
||||
logger.debug(
|
||||
"[voice-align] slowdown clip=%s: speed %.4f -> %.4f",
|
||||
clip.clip_id,
|
||||
old_speed,
|
||||
new_speed,
|
||||
)
|
||||
|
||||
# 调整后重新计算总时长用于日志
|
||||
new_total = 0.0
|
||||
for layer in video_layers:
|
||||
for clip in layer.clips:
|
||||
new_total += self._clip_adjusted_duration(clip)
|
||||
|
||||
logger.info(
|
||||
"[voice-align] 调整完成: 新总时长=%.3fs (目标=%.3fs, 差异=%.3fs)",
|
||||
new_total,
|
||||
voice_duration,
|
||||
abs(new_total - voice_duration),
|
||||
)
|
||||
|
||||
def _maybe_generate_ass(self, video_duration: float) -> Path | None:
|
||||
"""根据 plan.config 生成 ASS 字幕文件。
|
||||
|
||||
|
||||
@@ -15,6 +15,7 @@ celery_app.conf.imports = (
|
||||
"worker_app.tasks.voice_clone",
|
||||
"worker_app.tasks.tts_synthesis",
|
||||
"worker_app.tasks.batch_download",
|
||||
"worker_app.tasks.duplication_check",
|
||||
"worker_app.tasks._startup",
|
||||
"apps.worker.video_processing.dedup",
|
||||
"worker_app.tasks.cleanup",
|
||||
|
||||
@@ -0,0 +1,198 @@
|
||||
"""手动查重任务(Issue #1661)。
|
||||
|
||||
流程:
|
||||
1. 从 OSS 下载用户上传的待查重视频
|
||||
2. 动态抽帧计算指纹(复用 VideoDeduplicator.compute_fingerprint)
|
||||
3. 跨项目与用户所有已有成片比对(compute_duplicate_rate + find_duplicate_segments)
|
||||
4. 更新 DuplicationRecord:status / duplicate_rate / duplicate_count / segments
|
||||
同时写入 visual_similarity / match_count
|
||||
5. 失败重试 3 次、间隔 60 秒,最终失败标记 failed;临时文件始终清理
|
||||
"""
|
||||
|
||||
import logging
|
||||
import os
|
||||
import shutil
|
||||
import tempfile
|
||||
|
||||
from celery import Task
|
||||
from celery.exceptions import Retry
|
||||
from video_processing.dedup import (
|
||||
VideoDeduplicator,
|
||||
find_duplicate_segments,
|
||||
)
|
||||
from worker_app.celery_app import celery_app
|
||||
from worker_app.db import SessionLocal
|
||||
|
||||
from packages.adapters.sqlalchemy_impl.duplication_repository import (
|
||||
SQLAlchemyDuplicationRecordRepository,
|
||||
)
|
||||
from packages.adapters.sqlalchemy_impl.generated_video_repository import (
|
||||
SQLAlchemyGeneratedVideoRepository,
|
||||
)
|
||||
from packages.domain.duplication import DuplicateSegment
|
||||
from packages.shared.storage import get_storage_service
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
def _build_domain_segments(
|
||||
fingerprint,
|
||||
session,
|
||||
deduplicator: VideoDeduplicator,
|
||||
user_id: str,
|
||||
) -> tuple[list[DuplicateSegment], int]:
|
||||
"""对用户所有已有视频做分片级时序匹配,构建领域片段列表。
|
||||
|
||||
Returns:
|
||||
(segments, duplicate_count) — segments 为 query 视频中的重复片段,
|
||||
duplicate_count 为存在重复片段的匹配视频数。
|
||||
"""
|
||||
video_repo = SQLAlchemyGeneratedVideoRepository(session)
|
||||
existing_videos = video_repo.list_by_user(user_id)
|
||||
|
||||
segments_out: list[DuplicateSegment] = []
|
||||
duplicate_count = 0
|
||||
|
||||
for existing in existing_videos:
|
||||
if not existing.video_fingerprint:
|
||||
continue
|
||||
|
||||
chunk_data = deduplicator._get_existing_chunks(existing.id, session)
|
||||
if not chunk_data:
|
||||
# 老视频无分片数据,时序定位不可靠,跳过片段级匹配
|
||||
continue
|
||||
|
||||
raw_segments = find_duplicate_segments(fingerprint.chunks, chunk_data)
|
||||
if not raw_segments:
|
||||
continue
|
||||
|
||||
duplicate_count += 1
|
||||
for raw in raw_segments:
|
||||
avg_sim = 1.0 - raw.avg_distance / 64.0
|
||||
segments_out.append(
|
||||
DuplicateSegment.create(
|
||||
source_start=round(raw.query_start_ms / 1000.0, 2),
|
||||
source_end=round(raw.query_end_ms / 1000.0, 2),
|
||||
matched_video_id=existing.id,
|
||||
matched_video_name=existing.name,
|
||||
matched_start=round(raw.target_start_ms / 1000.0, 2),
|
||||
matched_end=round(raw.target_end_ms / 1000.0, 2),
|
||||
similarity=round(max(0.0, min(1.0, avg_sim)) * 100, 1),
|
||||
)
|
||||
)
|
||||
|
||||
# 按 query 起始时间排序,片段时间轴稳定
|
||||
segments_out.sort(key=lambda s: (s.source_start, s.source_end))
|
||||
return segments_out, duplicate_count
|
||||
|
||||
|
||||
@celery_app.task(bind=True, max_retries=3, name="worker.process_duplication_check")
|
||||
def process_duplication_check(self: Task, record_id: str) -> dict:
|
||||
"""处理一次手动查重请求。
|
||||
|
||||
Args:
|
||||
record_id: DuplicationRecord ID
|
||||
|
||||
Returns:
|
||||
dict: {"ok": True, "record_id": ..., "duplicate_rate": ..., ...}
|
||||
"""
|
||||
session = None
|
||||
temp_dir = None
|
||||
try:
|
||||
session = SessionLocal()
|
||||
repo = SQLAlchemyDuplicationRecordRepository(session)
|
||||
storage_service = get_storage_service()
|
||||
deduplicator = VideoDeduplicator()
|
||||
|
||||
record = repo.get(record_id)
|
||||
if record is None:
|
||||
raise ValueError(f"Duplication record {record_id} not found")
|
||||
|
||||
if record.status not in ("pending", "processing"):
|
||||
logger.info("Duplication record %s already %s, skip", record_id, record.status)
|
||||
return {"ok": True, "record_id": record_id, "status": record.status, "skipped": True}
|
||||
|
||||
record.mark_processing()
|
||||
repo.update(record)
|
||||
session.commit()
|
||||
|
||||
temp_dir = tempfile.mkdtemp(prefix="dup_check_")
|
||||
suffix = os.path.splitext(record.filename)[1] or ".mp4"
|
||||
local_path = os.path.join(temp_dir, f"{record_id}{suffix}")
|
||||
|
||||
storage_service.download_file(record.storage_key, local_path)
|
||||
|
||||
fingerprint = deduplicator.compute_fingerprint(local_path)
|
||||
record.duration_seconds = round(fingerprint.duration, 2) if fingerprint.duration else 0.0
|
||||
record.video_fingerprint = fingerprint.to_dict()
|
||||
|
||||
# 跨项目与用户所有已有视频比对(current_video_id=None:上传视频不在成片表中)
|
||||
rate_result = deduplicator.compute_duplicate_rate(
|
||||
fingerprint,
|
||||
project_id="",
|
||||
current_video_id=None,
|
||||
session=session,
|
||||
scope="user",
|
||||
user_id=record.user_id,
|
||||
)
|
||||
|
||||
# 分片级时序匹配 → 重复片段
|
||||
segments, segment_match_count = _build_domain_segments(fingerprint, session, deduplicator, record.user_id)
|
||||
|
||||
record.mark_completed(
|
||||
duplicate_rate=rate_result["duplicate_rate"],
|
||||
duplicate_count=segment_match_count,
|
||||
segments=segments,
|
||||
visual_similarity=rate_result["visual_similarity"],
|
||||
match_count=rate_result["match_count"],
|
||||
)
|
||||
repo.update(record)
|
||||
session.commit()
|
||||
|
||||
logger.info(
|
||||
"Duplication check completed: record=%s rate=%.2f%% matches=%d segments=%d",
|
||||
record_id,
|
||||
record.duplicate_rate,
|
||||
record.match_count,
|
||||
len(segments),
|
||||
)
|
||||
|
||||
return {
|
||||
"ok": True,
|
||||
"record_id": record_id,
|
||||
"status": "completed",
|
||||
"duplicate_rate": record.duplicate_rate,
|
||||
"duplicate_count": record.duplicate_count,
|
||||
"visual_similarity": record.visual_similarity,
|
||||
"match_count": record.match_count,
|
||||
"segments": len(segments),
|
||||
}
|
||||
|
||||
except Retry:
|
||||
raise
|
||||
|
||||
except Exception as e:
|
||||
logger.error("Duplication check failed for record %s: %s", record_id, e, exc_info=True)
|
||||
if session is not None:
|
||||
session.rollback()
|
||||
# 超过重试上限:标记 failed 并返回失败结果,不再 retry
|
||||
if "repo" in locals() and self.request.retries >= self.max_retries:
|
||||
try:
|
||||
failed_record = repo.get(record_id)
|
||||
if failed_record is not None and failed_record.status != "failed":
|
||||
failed_record.mark_failed(f"查重失败(已重试{self.max_retries}次): {e}")
|
||||
repo.update(failed_record)
|
||||
session.commit()
|
||||
except Exception as inner:
|
||||
logger.error("Failed to mark duplication record %s as failed: %s", record_id, inner)
|
||||
session.rollback()
|
||||
return {"ok": False, "record_id": record_id, "status": "failed", "error": str(e)}
|
||||
# 未达上限:60 秒后重试
|
||||
raise self.retry(exc=e, countdown=60) from e
|
||||
return {"ok": False, "record_id": record_id, "status": "failed", "error": str(e)}
|
||||
|
||||
finally:
|
||||
if session is not None:
|
||||
session.close()
|
||||
if temp_dir and os.path.isdir(temp_dir):
|
||||
shutil.rmtree(temp_dir, ignore_errors=True)
|
||||
@@ -716,6 +716,28 @@ def generate_video(self, task_id: str) -> dict:
|
||||
|
||||
_update_task_progress(task_id, 80, "渲染完成")
|
||||
|
||||
# ── 3.5 随机边缘裁剪降重(#1664) ──────────────────────────
|
||||
from video_processing.ffmpeg_utils import random_edge_crop
|
||||
|
||||
try:
|
||||
cropped_path = random_edge_crop(output_path)
|
||||
if cropped_path != output_path:
|
||||
output_path = cropped_path
|
||||
if gen_task:
|
||||
gen_task.append_log("边缘裁剪", "已应用随机 2-5% 边缘裁剪降重")
|
||||
_flush_logs(task_id, gen_task)
|
||||
logger.info("[task_id=%s] 随机边缘裁剪完成: %s", task_id, output_path)
|
||||
except Exception as crop_err:
|
||||
logger.warning(
|
||||
"[task_id=%s] 随机边缘裁剪失败,使用原始视频继续: %s",
|
||||
task_id,
|
||||
crop_err,
|
||||
exc_info=True,
|
||||
)
|
||||
if gen_task:
|
||||
gen_task.append_log("边缘裁剪", f"裁剪失败,使用原始视频: {crop_err}")
|
||||
_flush_logs(task_id, gen_task)
|
||||
|
||||
# ── 4. 上传 OSS + 查重记录 ───────────────────────────────
|
||||
_update_task_progress(task_id, 85, "开始上传")
|
||||
file_url, duration, file_size, video_count = _upload_and_record(
|
||||
|
||||
@@ -25,6 +25,8 @@ class SQLAlchemyDuplicationRecordRepository:
|
||||
status=record.status,
|
||||
duplicate_rate=record.duplicate_rate,
|
||||
duplicate_count=record.duplicate_count,
|
||||
visual_similarity=record.visual_similarity,
|
||||
match_count=record.match_count,
|
||||
video_fingerprint=json.dumps(record.video_fingerprint) if record.video_fingerprint else None,
|
||||
error_message=record.error_message,
|
||||
created_at=record.created_at,
|
||||
@@ -58,6 +60,8 @@ class SQLAlchemyDuplicationRecordRepository:
|
||||
model.status = record.status
|
||||
model.duplicate_rate = record.duplicate_rate
|
||||
model.duplicate_count = record.duplicate_count
|
||||
model.visual_similarity = record.visual_similarity
|
||||
model.match_count = record.match_count
|
||||
model.video_fingerprint = json.dumps(record.video_fingerprint) if record.video_fingerprint else None
|
||||
model.error_message = record.error_message
|
||||
model.updated_at = record.updated_at
|
||||
@@ -121,6 +125,8 @@ class SQLAlchemyDuplicationRecordRepository:
|
||||
status=model.status,
|
||||
duplicate_rate=model.duplicate_rate,
|
||||
duplicate_count=int(model.duplicate_count or 0),
|
||||
visual_similarity=getattr(model, "visual_similarity", None),
|
||||
match_count=getattr(model, "match_count", None),
|
||||
video_fingerprint=json.loads(fp_raw) if fp_raw else None,
|
||||
error_message=getattr(model, "error_message", ""),
|
||||
segments=segments,
|
||||
|
||||
@@ -131,3 +131,65 @@ class SQLAlchemyEditPlanClipRepository:
|
||||
created_at=model.created_at,
|
||||
updated_at=model.updated_at,
|
||||
)
|
||||
|
||||
def list_used_segments_by_user(
|
||||
self,
|
||||
user_id: str,
|
||||
*,
|
||||
limit_recent: int = 50,
|
||||
) -> dict[str, list[tuple[float, float]]]:
|
||||
"""查询用户已有视频中已使用的素材区间(跨视频避让).
|
||||
|
||||
JOIN edit_plans 表,按 created_by_user_id 过滤,只查 status='completed'
|
||||
的 plan 下 status='rendered' 且 asset_id 非空的 clips。按 plan 的
|
||||
created_at DESC 取最近 limit_recent 个 plan。
|
||||
|
||||
Returns:
|
||||
{asset_id: [(start_time, start_time + duration), ...]}
|
||||
空结果返回空 dict。
|
||||
"""
|
||||
from packages.adapters.sqlalchemy_impl.models import EditPlanModel
|
||||
|
||||
if not user_id:
|
||||
return {}
|
||||
|
||||
# 1. 查出最近 limit_recent 个已完成 plan 的 ID
|
||||
recent_plan_ids = [
|
||||
row[0]
|
||||
for row in self.session.query(EditPlanModel.id)
|
||||
.filter(
|
||||
EditPlanModel.created_by_user_id == user_id,
|
||||
EditPlanModel.status == "completed",
|
||||
)
|
||||
.order_by(EditPlanModel.created_at.desc())
|
||||
.limit(limit_recent)
|
||||
.all()
|
||||
]
|
||||
|
||||
if not recent_plan_ids:
|
||||
return {}
|
||||
|
||||
# 2. 查这些 plan 下已渲染、有素材的 clips
|
||||
clips = (
|
||||
self.session.query(
|
||||
EditPlanClipModel.asset_id,
|
||||
EditPlanClipModel.start_time,
|
||||
EditPlanClipModel.duration,
|
||||
)
|
||||
.filter(
|
||||
EditPlanClipModel.plan_id.in_(recent_plan_ids),
|
||||
EditPlanClipModel.status == "rendered",
|
||||
EditPlanClipModel.asset_id != "",
|
||||
EditPlanClipModel.asset_id.isnot(None),
|
||||
)
|
||||
.all()
|
||||
)
|
||||
|
||||
# 3. 聚合为 {asset_id: [(start, start+duration), ...]}
|
||||
result: dict[str, list[tuple[float, float]]] = {}
|
||||
for asset_id, start_time, duration in clips:
|
||||
if asset_id not in result:
|
||||
result[asset_id] = []
|
||||
result[asset_id].append((start_time or 0.0, (start_time or 0.0) + (duration or 0.0)))
|
||||
|
||||
return result
|
||||
|
||||
@@ -31,6 +31,8 @@ class SQLAlchemyGeneratedVideoRepository:
|
||||
is_duplicate=video.is_duplicate,
|
||||
duplicate_of=video.duplicate_of,
|
||||
duplicate_rate=video.duplicate_rate,
|
||||
match_count=getattr(video, "match_count", None),
|
||||
visual_similarity=getattr(video, "visual_similarity", None),
|
||||
generated_at=video.generated_at,
|
||||
created_at=video.created_at,
|
||||
)
|
||||
@@ -62,6 +64,8 @@ class SQLAlchemyGeneratedVideoRepository:
|
||||
is_duplicate=getattr(model, "is_duplicate", False),
|
||||
duplicate_of=getattr(model, "duplicate_of", None),
|
||||
duplicate_rate=getattr(model, "duplicate_rate", None),
|
||||
match_count=getattr(model, "match_count", None),
|
||||
visual_similarity=getattr(model, "visual_similarity", None),
|
||||
generated_at=model.generated_at,
|
||||
created_at=model.created_at,
|
||||
)
|
||||
@@ -77,6 +81,8 @@ class SQLAlchemyGeneratedVideoRepository:
|
||||
model.is_duplicate = video.is_duplicate
|
||||
model.duplicate_of = video.duplicate_of
|
||||
model.duplicate_rate = video.duplicate_rate
|
||||
model.match_count = getattr(video, "match_count", None)
|
||||
model.visual_similarity = getattr(video, "visual_similarity", None)
|
||||
self.session.add(model)
|
||||
self.session.commit()
|
||||
return video
|
||||
@@ -85,6 +91,24 @@ class SQLAlchemyGeneratedVideoRepository:
|
||||
models = self.session.query(GeneratedVideoModel).filter(GeneratedVideoModel.project_id == project_id).all()
|
||||
return [self._to_domain(model) for model in models]
|
||||
|
||||
def list_by_user(self, user_id: str, *, duration_min: float = 0, duration_max: float = 0) -> list[GeneratedVideo]:
|
||||
"""按 user_id 查询用户所有项目的视频(跨项目查重)。
|
||||
|
||||
Args:
|
||||
user_id: 用户 ID
|
||||
duration_min: 时长下限(秒),0 表示不限
|
||||
duration_max: 时长上限(秒),0 表示不限
|
||||
"""
|
||||
query = self.session.query(GeneratedVideoModel).filter(
|
||||
GeneratedVideoModel.user_id == user_id,
|
||||
)
|
||||
if duration_min > 0:
|
||||
query = query.filter(GeneratedVideoModel.duration >= duration_min)
|
||||
if duration_max > 0:
|
||||
query = query.filter(GeneratedVideoModel.duration <= duration_max)
|
||||
models = query.all()
|
||||
return [self._to_domain(model) for model in models]
|
||||
|
||||
def list_by_generation_task(self, generation_task_id: str) -> list[GeneratedVideo]:
|
||||
models = (
|
||||
self.session.query(GeneratedVideoModel)
|
||||
@@ -208,6 +232,8 @@ class SQLAlchemyGeneratedVideoRepository:
|
||||
is_duplicate=getattr(model, "is_duplicate", False),
|
||||
duplicate_of=getattr(model, "duplicate_of", None),
|
||||
duplicate_rate=getattr(model, "duplicate_rate", None),
|
||||
match_count=getattr(model, "match_count", None),
|
||||
visual_similarity=getattr(model, "visual_similarity", None),
|
||||
generated_at=model.generated_at,
|
||||
created_at=model.created_at,
|
||||
)
|
||||
|
||||
@@ -340,6 +340,8 @@ class GeneratedVideoModel(Base):
|
||||
is_duplicate = Column(Boolean, nullable=False, default=False)
|
||||
duplicate_of = Column(String(36), nullable=True)
|
||||
duplicate_rate = Column(Float, nullable=True)
|
||||
match_count = Column(Integer, nullable=True, default=0)
|
||||
visual_similarity = Column(Float, nullable=True, default=0.0)
|
||||
|
||||
|
||||
class TitleLibraryModel(Base):
|
||||
@@ -415,6 +417,9 @@ class DuplicationRecordModel(Base):
|
||||
status = Column(String(20), nullable=False, default="pending", index=True)
|
||||
duplicate_rate = Column(Float, nullable=True)
|
||||
duplicate_count = Column(Integer, nullable=False, default=0)
|
||||
# #1661 手动查重:视觉相似度(0~1)/ 匹配视频数
|
||||
visual_similarity = Column(Float, nullable=True)
|
||||
match_count = Column(Integer, nullable=True)
|
||||
video_fingerprint = Column(Text, nullable=True)
|
||||
error_message = Column(Text, nullable=False, default="")
|
||||
created_at = Column(DateTime, nullable=False, default=lambda: datetime.now(timezone.utc))
|
||||
@@ -620,3 +625,20 @@ class CoverTemplateModel(Base):
|
||||
config = Column(JSON, nullable=False, default=dict)
|
||||
created_at = Column(DateTime(timezone=True), nullable=False, default=lambda: datetime.now(timezone.utc))
|
||||
updated_at = Column(DateTime(timezone=True), nullable=False, default=lambda: datetime.now(timezone.utc))
|
||||
|
||||
|
||||
class VideoFingerprintChunkModel(Base):
|
||||
"""分片视频指纹 — 每个视频按时间分片存储 pHash + color_histogram."""
|
||||
|
||||
__tablename__ = "video_fingerprint_chunks"
|
||||
|
||||
id = Column(String(36), primary_key=True)
|
||||
video_id = Column(String(36), nullable=False, index=True)
|
||||
project_id = Column(String(36), nullable=False, index=True)
|
||||
user_id = Column(String(36), nullable=False, index=True, default="")
|
||||
start_time_ms = Column(Integer, nullable=False)
|
||||
end_time_ms = Column(Integer, nullable=False)
|
||||
phash_binary = Column(String(16), nullable=False)
|
||||
color_histogram = Column(JSON, nullable=False)
|
||||
frame_count = Column(Integer, nullable=False, default=1)
|
||||
created_at = Column(DateTime, nullable=False, default=lambda: datetime.now(timezone.utc))
|
||||
|
||||
@@ -81,14 +81,14 @@ def position_to_ass_alignment(position: str) -> int:
|
||||
position: 位置字符串 top/center/bottom
|
||||
|
||||
Returns:
|
||||
ASS 对齐编号,默认 8(顶部居中)
|
||||
ASS 对齐编号,默认 2(底部居中,与前端 DEFAULT_TITLE_SETTINGS.position="bottom" 对齐)
|
||||
"""
|
||||
mapping = {
|
||||
"top": 8,
|
||||
"center": 5,
|
||||
"bottom": 2,
|
||||
}
|
||||
return mapping.get(position, 8)
|
||||
return mapping.get(position, 2)
|
||||
|
||||
|
||||
# ── Style 行构建 ──────────────────────────────────────────────────────────────
|
||||
@@ -226,7 +226,6 @@ def _wrap_title_text(
|
||||
|
||||
# 换行计算使用原始 font_size,与 CSS 预览一致;1.35x 补偿仅用于 ASS Fontsize 渲染
|
||||
|
||||
|
||||
# 先按已有 \N 分段,每段独立自动换行,最后用 \N 拼回
|
||||
segments = text.split("\\N")
|
||||
wrapped_segments: list[str] = []
|
||||
@@ -386,8 +385,8 @@ def build_ass_content(
|
||||
# position → alignment 三档逻辑,现有输出保持一字节不变。
|
||||
title_pos = _parse_title_position(title_config, video_width, video_height)
|
||||
|
||||
title_alignment = 5 if title_pos is not None else position_to_ass_alignment(
|
||||
title_config.get("position", "top")
|
||||
title_alignment = (
|
||||
5 if title_pos is not None else position_to_ass_alignment(title_config.get("position", "bottom"))
|
||||
)
|
||||
|
||||
styles.append(
|
||||
|
||||
@@ -63,6 +63,9 @@ class DuplicationRecord:
|
||||
status: str = "pending" # pending / processing / completed / failed
|
||||
duplicate_rate: float | None = None # 0-100
|
||||
duplicate_count: int = 0
|
||||
# #1661 手动查重:视觉相似度(归一化 0~1)/ 匹配视频数
|
||||
visual_similarity: float | None = None
|
||||
match_count: int | None = None
|
||||
video_fingerprint: dict[str, Any] | None = None
|
||||
error_message: str = ""
|
||||
segments: list[DuplicateSegment] = field(default_factory=list)
|
||||
@@ -98,13 +101,23 @@ class DuplicationRecord:
|
||||
self.status = "processing"
|
||||
self.updated_at = datetime.now(timezone.utc)
|
||||
|
||||
def mark_completed(self, duplicate_rate: float, duplicate_count: int, segments: list[DuplicateSegment]) -> None:
|
||||
def mark_completed(
|
||||
self,
|
||||
duplicate_rate: float,
|
||||
duplicate_count: int,
|
||||
segments: list[DuplicateSegment],
|
||||
*,
|
||||
visual_similarity: float | None = None,
|
||||
match_count: int | None = None,
|
||||
) -> None:
|
||||
if not 0 <= duplicate_rate <= 100:
|
||||
raise ValueError("duplicate_rate must be between 0 and 100")
|
||||
self.status = "completed"
|
||||
self.duplicate_rate = duplicate_rate
|
||||
self.duplicate_count = duplicate_count
|
||||
self.segments = segments
|
||||
self.visual_similarity = visual_similarity
|
||||
self.match_count = match_count
|
||||
self.updated_at = datetime.now(timezone.utc)
|
||||
|
||||
def mark_failed(self, error_message: str) -> None:
|
||||
@@ -133,6 +146,8 @@ class DuplicationRecord:
|
||||
self.status = "pending"
|
||||
self.duplicate_rate = None
|
||||
self.duplicate_count = 0
|
||||
self.visual_similarity = None
|
||||
self.match_count = None
|
||||
self.error_message = ""
|
||||
self.segments = []
|
||||
self.video_fingerprint = None
|
||||
|
||||
@@ -27,6 +27,8 @@ class GeneratedVideo:
|
||||
is_duplicate: bool = False
|
||||
duplicate_of: str | None = None
|
||||
duplicate_rate: float | None = None
|
||||
match_count: int | None = None
|
||||
visual_similarity: float | None = None
|
||||
generated_at: datetime = field(default_factory=lambda: datetime.now(timezone.utc))
|
||||
created_at: datetime = field(default_factory=lambda: datetime.now(timezone.utc))
|
||||
|
||||
|
||||
@@ -169,6 +169,7 @@ def distribute_assets(
|
||||
random_selection: bool = False,
|
||||
asset_durations: dict[str, float] | None = None,
|
||||
asset_scene_points: dict[str, list[float]] | None = None,
|
||||
external_used_segments: dict[str, list[tuple[float, float]]] | None = None,
|
||||
) -> None:
|
||||
"""按 editing_mode 将素材分配到 clips(就地修改).
|
||||
|
||||
@@ -188,6 +189,7 @@ def distribute_assets(
|
||||
random_selection: 是否随机选择素材(用于预览生成)
|
||||
asset_durations: 素材 ID -> 时长(秒)映射,用于设置 start_time
|
||||
asset_scene_points: 素材 ID -> 场景切换点列表(metadata 缓存)
|
||||
external_used_segments: 跨视频已用区间(来自其他视频的 clips),注入到分配逻辑中避让
|
||||
"""
|
||||
if not asset_ids or not clips:
|
||||
return
|
||||
@@ -198,16 +200,16 @@ def distribute_assets(
|
||||
random.shuffle(asset_ids)
|
||||
|
||||
if editing_mode == EditingMode.ONE_TAKE.value:
|
||||
_distribute_one_take(clips, asset_ids, asset_durations, asset_scene_points)
|
||||
_distribute_one_take(clips, asset_ids, asset_durations, asset_scene_points, external_used_segments)
|
||||
elif editing_mode == EditingMode.PIP.value:
|
||||
_distribute_pip(clips, asset_ids, asset_durations, asset_scene_points)
|
||||
_distribute_pip(clips, asset_ids, asset_durations, asset_scene_points, external_used_segments)
|
||||
elif editing_mode == EditingMode.VOICE_OVER.value:
|
||||
_distribute_voice_over(clips, asset_ids, asset_durations, asset_scene_points)
|
||||
_distribute_voice_over(clips, asset_ids, asset_durations, asset_scene_points, external_used_segments)
|
||||
elif editing_mode == EditingMode.VOICE_PIP.value:
|
||||
_distribute_voice_pip(clips, asset_ids, asset_durations, asset_scene_points)
|
||||
_distribute_voice_pip(clips, asset_ids, asset_durations, asset_scene_points, external_used_segments)
|
||||
else:
|
||||
# 未知模式,退化为 one_take
|
||||
_distribute_one_take(clips, asset_ids, asset_durations, asset_scene_points)
|
||||
_distribute_one_take(clips, asset_ids, asset_durations, asset_scene_points, external_used_segments)
|
||||
|
||||
|
||||
def _resolve_start_time(
|
||||
@@ -248,9 +250,12 @@ def _distribute_one_take(
|
||||
asset_ids: List[str],
|
||||
asset_durations: dict[str, float] | None = None,
|
||||
asset_scene_points: dict[str, list[float]] | None = None,
|
||||
external_used_segments: dict[str, list[tuple[float, float]]] | None = None,
|
||||
) -> None:
|
||||
"""ONE_TAKE: 素材按顺序依次分配给 main 类型 clips."""
|
||||
used_segments: dict[str, list[tuple[float, float]]] = {}
|
||||
used_segments: dict[str, list[tuple[float, float]]] = (
|
||||
{k: list(v) for k, v in external_used_segments.items()} if external_used_segments else {}
|
||||
)
|
||||
main_clips = [c for c in clips if c.clip_type == ClipType.MAIN.value]
|
||||
for i, clip in enumerate(main_clips):
|
||||
if i < len(asset_ids):
|
||||
@@ -271,9 +276,12 @@ def _distribute_pip(
|
||||
asset_ids: List[str],
|
||||
asset_durations: dict[str, float] | None = None,
|
||||
asset_scene_points: dict[str, list[float]] | None = None,
|
||||
external_used_segments: dict[str, list[tuple[float, float]]] | None = None,
|
||||
) -> None:
|
||||
"""PIP: 第1个素材→main(全屏背景),其余→overlay clips."""
|
||||
used_segments: dict[str, list[tuple[float, float]]] = {}
|
||||
used_segments: dict[str, list[tuple[float, float]]] = (
|
||||
{k: list(v) for k, v in external_used_segments.items()} if external_used_segments else {}
|
||||
)
|
||||
# 第1个素材 → main clip
|
||||
main_clips = [c for c in clips if c.clip_type == ClipType.MAIN.value]
|
||||
if main_clips and asset_ids:
|
||||
@@ -310,9 +318,12 @@ def _distribute_voice_over(
|
||||
asset_ids: List[str],
|
||||
asset_durations: dict[str, float] | None = None,
|
||||
asset_scene_points: dict[str, list[float]] | None = None,
|
||||
external_used_segments: dict[str, list[tuple[float, float]]] | None = None,
|
||||
) -> None:
|
||||
"""VOICE_OVER: 素材→main clips (B-roll)."""
|
||||
used_segments: dict[str, list[tuple[float, float]]] = {}
|
||||
used_segments: dict[str, list[tuple[float, float]]] = (
|
||||
{k: list(v) for k, v in external_used_segments.items()} if external_used_segments else {}
|
||||
)
|
||||
main_clips = [c for c in clips if c.clip_type == ClipType.MAIN.value]
|
||||
for i, clip in enumerate(main_clips):
|
||||
if i < len(asset_ids):
|
||||
@@ -333,9 +344,12 @@ def _distribute_voice_pip(
|
||||
asset_ids: List[str],
|
||||
asset_durations: dict[str, float] | None = None,
|
||||
asset_scene_points: dict[str, list[float]] | None = None,
|
||||
external_used_segments: dict[str, list[tuple[float, float]]] | None = None,
|
||||
) -> None:
|
||||
"""VOICE_PIP: 第1个→background, 第2个→corner_voice, 其余→b_roll."""
|
||||
used_segments: dict[str, list[tuple[float, float]]] = {}
|
||||
used_segments: dict[str, list[tuple[float, float]]] = (
|
||||
{k: list(v) for k, v in external_used_segments.items()} if external_used_segments else {}
|
||||
)
|
||||
bg_clips = [c for c in clips if c.clip_type == "background"]
|
||||
voice_clips = [c for c in clips if c.clip_type == "corner_voice"]
|
||||
broll_clips = [c for c in clips if c.clip_type == "b_roll"]
|
||||
|
||||
@@ -83,11 +83,11 @@ class TestPositionToAssAlignment:
|
||||
def test_bottom(self):
|
||||
assert position_to_ass_alignment("bottom") == 2
|
||||
|
||||
def test_unknown_defaults_top(self):
|
||||
assert position_to_ass_alignment("unknown") == 8
|
||||
def test_unknown_defaults_bottom(self):
|
||||
assert position_to_ass_alignment("unknown") == 2
|
||||
|
||||
def test_empty_defaults_top(self):
|
||||
assert position_to_ass_alignment("") == 8
|
||||
def test_empty_defaults_bottom(self):
|
||||
assert position_to_ass_alignment("") == 2
|
||||
|
||||
|
||||
# ============================================================
|
||||
@@ -581,6 +581,7 @@ class TestConstants:
|
||||
assert isinstance(TITLE_MARGIN_BOTTOM, int)
|
||||
assert isinstance(TITLE_MARGIN_SIDE, int)
|
||||
|
||||
|
||||
# ============================================================
|
||||
# _wrap_title_text 换行逻辑验证
|
||||
# ============================================================
|
||||
|
||||
@@ -0,0 +1,507 @@
|
||||
"""Issue #1677 多视频批量生成 — 变体独立配置与批量预览/批量生成测试。
|
||||
|
||||
覆盖:
|
||||
- 批量预览:preview_count=N 一次创建 N 个独立任务,返回变体数组
|
||||
- 变体克隆链路:N 个预览/正式任务各自关联独立克隆 plan
|
||||
- 变体独立配置:titles[]/voice_library_ids[]/cover_urls[] 按变体注入
|
||||
- 长度校验:数组长度必须为 1 或 N(共用或独立),非法长度报错
|
||||
- N=1 向后兼容:旧字段单值行为不变
|
||||
"""
|
||||
|
||||
from datetime import datetime, timezone
|
||||
from unittest.mock import MagicMock, patch
|
||||
|
||||
import pytest
|
||||
from app.core.task_enqueue import GlobalQueueFull, UserPendingLimitExceeded
|
||||
from app.schemas.generation_task import (
|
||||
BatchPreviewGenerationTaskResponse,
|
||||
CreateGenerationTaskRequest,
|
||||
CreatePreviewGenerationTaskRequest,
|
||||
)
|
||||
|
||||
from packages.domain import GenerationTask
|
||||
from packages.domain.generation_task import GenerationTaskStatus
|
||||
|
||||
# ════════════════════════════════════════════════════════════════════════════
|
||||
# 辅助构造
|
||||
# ════════════════════════════════════════════════════════════════════════════
|
||||
|
||||
|
||||
def _make_user(user_id="test_user_001"):
|
||||
mock_user = MagicMock()
|
||||
mock_user.id = user_id
|
||||
auth = MagicMock()
|
||||
auth.user = mock_user
|
||||
return auth
|
||||
|
||||
|
||||
def _make_task(task_id=None, status=GenerationTaskStatus.PENDING, source_plan_id=None):
|
||||
task = GenerationTask.create(
|
||||
project_id="",
|
||||
asset_library_id="",
|
||||
template_id="tpl_001",
|
||||
asset_ids=["asset_1"],
|
||||
)
|
||||
if task_id:
|
||||
task.id = task_id
|
||||
task.status = status
|
||||
task.is_preview = True
|
||||
task.source_edit_plan_id = source_plan_id or ""
|
||||
task.voice_library_id = ""
|
||||
task.title_config = {}
|
||||
task.cover_url = ""
|
||||
return task
|
||||
|
||||
|
||||
def _make_preview_request(**kwargs):
|
||||
defaults = {
|
||||
"template_id": "tpl_001",
|
||||
"asset_ids": ["asset_1", "asset_2"],
|
||||
}
|
||||
defaults.update(kwargs)
|
||||
return CreatePreviewGenerationTaskRequest(**defaults)
|
||||
|
||||
|
||||
def _repo_mock():
|
||||
repo = MagicMock()
|
||||
repo.count_pending_by_user.return_value = 0
|
||||
repo.count_pending_total.return_value = 0
|
||||
repo.get.side_effect = lambda tid: None
|
||||
return repo
|
||||
|
||||
|
||||
# ════════════════════════════════════════════════════════════════════════════
|
||||
# Schema 校验:变体数组长度
|
||||
# ════════════════════════════════════════════════════════════════════════════
|
||||
|
||||
|
||||
class TestVariantArrayValidation:
|
||||
"""变体数组字段长度校验。"""
|
||||
|
||||
def test_preview_titles_length_matches_count(self):
|
||||
"""titles 长度 = preview_count 合法"""
|
||||
req = _make_preview_request(preview_count=3, titles=["标题A", "标题B", "标题C"])
|
||||
assert len(req.titles) == 3
|
||||
|
||||
def test_preview_titles_single_shared(self):
|
||||
"""titles 长度 1 = 所有变体共用,合法"""
|
||||
req = _make_preview_request(preview_count=3, titles=["共用标题"])
|
||||
assert req.titles == ["共用标题"]
|
||||
|
||||
def test_preview_titles_wrong_length_raises(self):
|
||||
"""titles 长度 2 与 preview_count=3 不匹配 → 报错"""
|
||||
with pytest.raises(ValueError, match="titles"):
|
||||
_make_preview_request(preview_count=3, titles=["A", "B"])
|
||||
|
||||
def test_preview_voice_ids_wrong_length_raises(self):
|
||||
"""voice_library_ids 长度非法 → 报错"""
|
||||
from pydantic import ValidationError
|
||||
|
||||
with pytest.raises(ValidationError, match="voice_library_ids"):
|
||||
_make_preview_request(preview_count=4, voice_library_ids=["v1", "v2"])
|
||||
|
||||
def test_preview_empty_arrays_ok(self):
|
||||
"""空数组(回退单值字段)合法"""
|
||||
req = _make_preview_request(preview_count=3)
|
||||
assert req.titles == []
|
||||
assert req.voice_library_ids == []
|
||||
assert req.cover_urls == []
|
||||
|
||||
def test_generation_titles_length_matches_count(self):
|
||||
"""正式生成 titles 长度 = count 合法"""
|
||||
req = CreateGenerationTaskRequest(
|
||||
template_id="tpl_1",
|
||||
asset_ids=["a1"],
|
||||
count=3,
|
||||
titles=["A", "B", "C"],
|
||||
)
|
||||
assert len(req.titles) == 3
|
||||
|
||||
def test_generation_arrays_wrong_length_raises(self):
|
||||
"""正式生成 cover_urls 长度与 count 不匹配 → 报错"""
|
||||
from pydantic import ValidationError
|
||||
|
||||
with pytest.raises(ValidationError, match="cover_urls"):
|
||||
CreateGenerationTaskRequest(
|
||||
template_id="tpl_1",
|
||||
asset_ids=["a1"],
|
||||
count=3,
|
||||
cover_urls=["c1", "c2"],
|
||||
)
|
||||
|
||||
def test_generation_single_count_no_arrays(self):
|
||||
"""N=1 且不传数组:完全旧行为"""
|
||||
req = CreateGenerationTaskRequest(template_id="tpl_1", asset_ids=["a1"])
|
||||
assert req.count == 1
|
||||
assert req.titles == []
|
||||
assert req.voice_library_ids == []
|
||||
assert req.cover_urls == []
|
||||
|
||||
|
||||
# ════════════════════════════════════════════════════════════════════════════
|
||||
# 批量预览路由
|
||||
# ════════════════════════════════════════════════════════════════════════════
|
||||
|
||||
|
||||
class TestBatchPreviewRoute:
|
||||
"""POST /preview 批量变体。"""
|
||||
|
||||
def test_preview_count_1_returns_single_item_array(self):
|
||||
"""N=1 返回 items 长度 1 的批量响应(结构统一)"""
|
||||
from app.api.routes.generation_preview import create_preview_generation_task
|
||||
|
||||
task = _make_task(task_id="task_1")
|
||||
repo = _repo_mock()
|
||||
with patch("app.api.routes.generation_preview.CreateGenerationTaskUseCase") as MockUC:
|
||||
MockUC.return_value.execute.return_value = task
|
||||
with patch("app.api.routes.generation_preview.safe_enqueue_generation_task", return_value=True):
|
||||
resp = create_preview_generation_task(
|
||||
_make_preview_request(preview_count=1),
|
||||
authenticated_user=_make_user(),
|
||||
generation_task_repository=repo,
|
||||
db=MagicMock(),
|
||||
)
|
||||
assert isinstance(resp, BatchPreviewGenerationTaskResponse)
|
||||
assert resp.total == 1
|
||||
assert len(resp.items) == 1
|
||||
assert resp.items[0].task_id == "task_1"
|
||||
assert resp.items[0].variant_index == 0
|
||||
|
||||
def test_preview_count_3_creates_three_independent_tasks(self):
|
||||
"""N=3 创建 3 个独立任务,返回 3 个变体,task_id 各不相同"""
|
||||
from app.api.routes.generation_preview import create_preview_generation_task
|
||||
|
||||
tasks = [_make_task(task_id=f"task_{i}") for i in range(3)]
|
||||
repo = _repo_mock()
|
||||
with patch("app.api.routes.generation_preview.CreateGenerationTaskUseCase") as MockUC:
|
||||
MockUC.return_value.execute.side_effect = tasks
|
||||
with patch("app.api.routes.generation_preview.safe_enqueue_generation_task", return_value=True):
|
||||
resp = create_preview_generation_task(
|
||||
_make_preview_request(preview_count=3),
|
||||
authenticated_user=_make_user(),
|
||||
generation_task_repository=repo,
|
||||
db=MagicMock(),
|
||||
)
|
||||
assert resp.total == 3
|
||||
task_ids = [item.task_id for item in resp.items]
|
||||
assert task_ids == ["task_0", "task_1", "task_2"]
|
||||
assert len(set(task_ids)) == 3
|
||||
for i, item in enumerate(resp.items):
|
||||
assert item.variant_index == i
|
||||
|
||||
def test_preview_count_3_clones_three_variant_plans(self):
|
||||
"""有源 plan 时,N=3 克隆 3 个独立变体 plan(预览全部克隆,不用源 plan)"""
|
||||
from app.api.routes.generation_preview import create_preview_generation_task
|
||||
|
||||
tasks = [_make_task(task_id=f"task_{i}", source_plan_id="source_plan") for i in range(3)]
|
||||
repo = _repo_mock()
|
||||
cloned_plan_ids = ["clone_1", "clone_2", "clone_3"]
|
||||
with patch("app.api.routes.generation_preview.CreateGenerationTaskUseCase") as MockUC:
|
||||
MockUC.return_value.execute.side_effect = tasks
|
||||
with patch("app.api.routes.generation_preview.safe_enqueue_generation_task", return_value=True):
|
||||
with patch("app.services.edit_plan_service.EditPlanService") as MockPlanSvc:
|
||||
clone_results = [MagicMock(id=pid) for pid in cloned_plan_ids]
|
||||
MockPlanSvc.return_value.clone_plan_for_variant.side_effect = clone_results
|
||||
create_preview_generation_task(
|
||||
_make_preview_request(preview_count=3),
|
||||
authenticated_user=_make_user(),
|
||||
generation_task_repository=repo,
|
||||
db=MagicMock(),
|
||||
)
|
||||
# 克隆被调用 3 次
|
||||
assert MockPlanSvc.return_value.clone_plan_for_variant.call_count == 3
|
||||
# 每个任务关联到不同的克隆 plan
|
||||
for i, task in enumerate(tasks):
|
||||
assert task.source_edit_plan_id == cloned_plan_ids[i]
|
||||
|
||||
def test_preview_variant_titles_injected_per_variant(self):
|
||||
"""titles[] 按变体注入 title_config.text"""
|
||||
from app.api.routes.generation_preview import create_preview_generation_task
|
||||
|
||||
tasks = [_make_task(task_id=f"task_{i}") for i in range(3)]
|
||||
repo = _repo_mock()
|
||||
captured_commands = []
|
||||
with patch("app.api.routes.generation_preview.CreateGenerationTaskUseCase") as MockUC:
|
||||
|
||||
def _execute(cmd):
|
||||
captured_commands.append(cmd)
|
||||
return tasks[len(captured_commands) - 1]
|
||||
|
||||
MockUC.return_value.execute.side_effect = _execute
|
||||
with patch("app.api.routes.generation_preview.safe_enqueue_generation_task", return_value=True):
|
||||
create_preview_generation_task(
|
||||
_make_preview_request(
|
||||
preview_count=3,
|
||||
title_config={"font": "黑体", "position": "bottom"},
|
||||
titles=["标题A", "标题B", "标题C"],
|
||||
),
|
||||
authenticated_user=_make_user(),
|
||||
generation_task_repository=repo,
|
||||
db=MagicMock(),
|
||||
)
|
||||
assert len(captured_commands) == 3
|
||||
assert captured_commands[0].title_config["text"] == "标题A"
|
||||
assert captured_commands[1].title_config["text"] == "标题B"
|
||||
assert captured_commands[2].title_config["text"] == "标题C"
|
||||
# 样式全局共用
|
||||
assert all(c.title_config["font"] == "黑体" for c in captured_commands)
|
||||
|
||||
def test_preview_shared_title_when_single_length(self):
|
||||
"""titles 长度 1 = 所有变体共用同一标题"""
|
||||
from app.api.routes.generation_preview import create_preview_generation_task
|
||||
|
||||
tasks = [_make_task(task_id=f"task_{i}") for i in range(3)]
|
||||
repo = _repo_mock()
|
||||
captured = []
|
||||
with patch("app.api.routes.generation_preview.CreateGenerationTaskUseCase") as MockUC:
|
||||
|
||||
def _execute(cmd):
|
||||
captured.append(cmd)
|
||||
return tasks[len(captured) - 1]
|
||||
|
||||
MockUC.return_value.execute.side_effect = _execute
|
||||
with patch("app.api.routes.generation_preview.safe_enqueue_generation_task", return_value=True):
|
||||
create_preview_generation_task(
|
||||
_make_preview_request(preview_count=3, titles=["共用标题"]),
|
||||
authenticated_user=_make_user(),
|
||||
generation_task_repository=repo,
|
||||
db=MagicMock(),
|
||||
)
|
||||
assert all(c.title_config["text"] == "共用标题" for c in captured)
|
||||
|
||||
def test_preview_independent_voice_per_variant(self):
|
||||
"""voice_library_ids[] 按变体注入独立配音"""
|
||||
from app.api.routes.generation_preview import create_preview_generation_task
|
||||
|
||||
tasks = [_make_task(task_id=f"task_{i}") for i in range(3)]
|
||||
repo = _repo_mock()
|
||||
captured = []
|
||||
with patch("app.api.routes.generation_preview.CreateGenerationTaskUseCase") as MockUC:
|
||||
|
||||
def _execute(cmd):
|
||||
captured.append(cmd)
|
||||
return tasks[len(captured) - 1]
|
||||
|
||||
MockUC.return_value.execute.side_effect = _execute
|
||||
with patch("app.api.routes.generation_preview.safe_enqueue_generation_task", return_value=True):
|
||||
create_preview_generation_task(
|
||||
_make_preview_request(
|
||||
preview_count=3,
|
||||
voice_library_ids=["voice_a", "voice_b", "voice_c"],
|
||||
),
|
||||
authenticated_user=_make_user(),
|
||||
generation_task_repository=repo,
|
||||
db=MagicMock(),
|
||||
)
|
||||
assert [c.voice_library_id for c in captured] == ["voice_a", "voice_b", "voice_c"]
|
||||
|
||||
def test_preview_voice_fallback_to_single_field(self):
|
||||
"""voice_library_ids 为空时回退 voice_library_id 单值字段(向后兼容)"""
|
||||
from app.api.routes.generation_preview import create_preview_generation_task
|
||||
|
||||
task = _make_task(task_id="task_1")
|
||||
repo = _repo_mock()
|
||||
captured = []
|
||||
with patch("app.api.routes.generation_preview.CreateGenerationTaskUseCase") as MockUC:
|
||||
|
||||
def _execute(cmd):
|
||||
captured.append(cmd)
|
||||
return task
|
||||
|
||||
MockUC.return_value.execute.side_effect = _execute
|
||||
with patch("app.api.routes.generation_preview.safe_enqueue_generation_task", return_value=True):
|
||||
create_preview_generation_task(
|
||||
_make_preview_request(voice_library_id="legacy_voice"),
|
||||
authenticated_user=_make_user(),
|
||||
generation_task_repository=repo,
|
||||
db=MagicMock(),
|
||||
)
|
||||
assert captured[0].voice_library_id == "legacy_voice"
|
||||
|
||||
def test_preview_queue_limit_checks_total_count(self):
|
||||
"""限流预检查按变体总数计:用户 pending + N 超限 → 429"""
|
||||
from app.api.routes.generation_preview import create_preview_generation_task
|
||||
from fastapi import HTTPException
|
||||
|
||||
repo = MagicMock()
|
||||
repo.count_pending_by_user.return_value = 3
|
||||
repo.count_pending_total.return_value = 0
|
||||
with pytest.raises(HTTPException) as exc:
|
||||
create_preview_generation_task(
|
||||
_make_preview_request(preview_count=5),
|
||||
authenticated_user=_make_user(),
|
||||
generation_task_repository=repo,
|
||||
db=MagicMock(),
|
||||
)
|
||||
assert exc.value.status_code == 429
|
||||
|
||||
def test_preview_clone_failure_marks_all_failed(self):
|
||||
"""克隆变体 plan 失败 → 已创建任务全部标记 failed 并 500"""
|
||||
from app.api.routes.generation_preview import create_preview_generation_task
|
||||
from fastapi import HTTPException
|
||||
|
||||
tasks = [_make_task(task_id=f"task_{i}", source_plan_id="source_plan") for i in range(3)]
|
||||
repo = _repo_mock()
|
||||
with patch("app.api.routes.generation_preview.CreateGenerationTaskUseCase") as MockUC:
|
||||
MockUC.return_value.execute.side_effect = tasks
|
||||
with patch("app.services.edit_plan_service.EditPlanService") as MockPlanSvc:
|
||||
MockPlanSvc.return_value.clone_plan_for_variant.side_effect = RuntimeError("db down")
|
||||
with pytest.raises(HTTPException) as exc:
|
||||
create_preview_generation_task(
|
||||
_make_preview_request(preview_count=3),
|
||||
authenticated_user=_make_user(),
|
||||
generation_task_repository=repo,
|
||||
db=MagicMock(),
|
||||
)
|
||||
assert exc.value.status_code == 500
|
||||
# 所有已创建任务都被标记 failed
|
||||
assert all(t.status == GenerationTaskStatus.FAILED for t in tasks)
|
||||
|
||||
|
||||
# ════════════════════════════════════════════════════════════════════════════
|
||||
# 批量正式生成:变体配置注入
|
||||
# ════════════════════════════════════════════════════════════════════════════
|
||||
|
||||
|
||||
class TestBatchGenerationVariantConfig:
|
||||
"""POST /tasks count=N 时变体独立配置。"""
|
||||
|
||||
def _call_create_tasks(self, request, repo=None):
|
||||
from app.api.routes.generation_tasks import create_generation_task
|
||||
|
||||
repo = repo or MagicMock()
|
||||
repo.count_pending_by_user.return_value = 0
|
||||
repo.count_pending_total.return_value = 0
|
||||
repo.update.return_value = None
|
||||
|
||||
# 模板模式:asset_repository.find_by_id 返回 None(无 project 关联,
|
||||
# 纯模板模式 project_id/library_id 都为空),避免 MagicMock 属性污染
|
||||
asset_repo = MagicMock()
|
||||
asset_repo.find_by_id.return_value = None
|
||||
|
||||
# db.query().filter()...first() 返回 None:不走兜底关联编辑计划
|
||||
db = MagicMock()
|
||||
db.query.return_value.filter.return_value.order_by.return_value.first.return_value = None
|
||||
|
||||
return create_generation_task(
|
||||
request,
|
||||
authenticated_user=_make_user(),
|
||||
generation_task_repository=repo,
|
||||
project_repository=MagicMock(),
|
||||
asset_library_repository=MagicMock(),
|
||||
asset_repository=asset_repo,
|
||||
db=db,
|
||||
)
|
||||
|
||||
def test_count_3_variant_titles_voices_covers_injected(self):
|
||||
"""count=3:titles/voice_library_ids/cover_urls 按变体注入"""
|
||||
from app.api.routes import generation_tasks as routes
|
||||
|
||||
tasks = [_make_task(task_id=f"gen_{i}") for i in range(3)]
|
||||
captured = []
|
||||
with patch.object(routes, "CreateGenerationTaskUseCase") as MockUC:
|
||||
|
||||
def _execute(cmd):
|
||||
captured.append(cmd)
|
||||
t = tasks[len(captured) - 1]
|
||||
t.title_config = cmd.title_config
|
||||
t.voice_library_id = cmd.voice_library_id
|
||||
t.cover_url = cmd.cover_url
|
||||
return t
|
||||
|
||||
MockUC.return_value.execute.side_effect = _execute
|
||||
with patch.object(routes, "safe_enqueue_generation_task", return_value=True):
|
||||
req = CreateGenerationTaskRequest(
|
||||
template_id="tpl_1",
|
||||
asset_ids=["a1"],
|
||||
count=3,
|
||||
title_config={"font": "宋体"},
|
||||
titles=["成片标题1", "成片标题2", "成片标题3"],
|
||||
voice_library_ids=["v1", "v2", "v3"],
|
||||
cover_urls=["http://c1", "http://c2", "http://c3"],
|
||||
)
|
||||
resp = self._call_create_tasks(req)
|
||||
assert resp.total == 3
|
||||
assert [c.title_config["text"] for c in captured] == ["成片标题1", "成片标题2", "成片标题3"]
|
||||
assert [c.voice_library_id for c in captured] == ["v1", "v2", "v3"]
|
||||
assert [c.cover_url for c in captured] == ["http://c1", "http://c2", "http://c3"]
|
||||
# 样式共用
|
||||
assert all(c.title_config["font"] == "宋体" for c in captured)
|
||||
|
||||
def test_count_1_legacy_fields_unchanged(self):
|
||||
"""N=1 不传数组:旧字段 voice_library_id/cover_url/title_config 行为不变"""
|
||||
from app.api.routes import generation_tasks as routes
|
||||
|
||||
task = _make_task(task_id="gen_1")
|
||||
task.is_preview = False
|
||||
captured = []
|
||||
with patch.object(routes, "CreateGenerationTaskUseCase") as MockUC:
|
||||
|
||||
def _execute(cmd):
|
||||
captured.append(cmd)
|
||||
return task
|
||||
|
||||
MockUC.return_value.execute.side_effect = _execute
|
||||
with patch.object(routes, "safe_enqueue_generation_task", return_value=True):
|
||||
req = CreateGenerationTaskRequest(
|
||||
template_id="tpl_1",
|
||||
asset_ids=["a1"],
|
||||
count=1,
|
||||
voice_library_id="legacy_voice",
|
||||
cover_url="http://legacy-cover",
|
||||
title_config={"text": "旧标题", "font": "黑体"},
|
||||
)
|
||||
resp = self._call_create_tasks(req)
|
||||
assert resp.total == 1
|
||||
assert captured[0].voice_library_id == "legacy_voice"
|
||||
assert captured[0].cover_url == "http://legacy-cover"
|
||||
assert captured[0].title_config["text"] == "旧标题"
|
||||
|
||||
def test_count_3_shared_single_value_arrays(self):
|
||||
"""数组长度 1:3 个变体共用同一配音/封面"""
|
||||
from app.api.routes import generation_tasks as routes
|
||||
|
||||
tasks = [_make_task(task_id=f"gen_{i}") for i in range(3)]
|
||||
captured = []
|
||||
with patch.object(routes, "CreateGenerationTaskUseCase") as MockUC:
|
||||
|
||||
def _execute(cmd):
|
||||
captured.append(cmd)
|
||||
return tasks[len(captured) - 1]
|
||||
|
||||
MockUC.return_value.execute.side_effect = _execute
|
||||
with patch.object(routes, "safe_enqueue_generation_task", return_value=True):
|
||||
req = CreateGenerationTaskRequest(
|
||||
template_id="tpl_1",
|
||||
asset_ids=["a1"],
|
||||
count=3,
|
||||
voice_library_ids=["shared_voice"],
|
||||
cover_urls=["http://shared"],
|
||||
)
|
||||
self._call_create_tasks(req)
|
||||
assert all(c.voice_library_id == "shared_voice" for c in captured)
|
||||
assert all(c.cover_url == "http://shared" for c in captured)
|
||||
|
||||
|
||||
class TestVariantValueHelper:
|
||||
"""_variant_value 取值逻辑。"""
|
||||
|
||||
def test_empty_returns_fallback(self):
|
||||
from app.api.routes.generation_preview import _variant_value
|
||||
|
||||
assert _variant_value([], 0, fallback="fb") == "fb"
|
||||
|
||||
def test_single_length_shared(self):
|
||||
from app.api.routes.generation_preview import _variant_value
|
||||
|
||||
assert _variant_value(["only"], 5) == "only"
|
||||
|
||||
def test_indexed_access(self):
|
||||
from app.api.routes.generation_preview import _variant_value
|
||||
|
||||
assert _variant_value(["a", "b", "c"], 1) == "b"
|
||||
|
||||
def test_index_out_of_range_fallback(self):
|
||||
from app.api.routes.generation_preview import _variant_value
|
||||
|
||||
assert _variant_value(["a", "b"], 9, fallback="x") == "x"
|
||||
@@ -65,11 +65,11 @@ class TestPositionToAssAlignment:
|
||||
def test_bottom(self):
|
||||
assert position_to_ass_alignment("bottom") == 2
|
||||
|
||||
def test_unknown_default_top(self):
|
||||
assert position_to_ass_alignment("unknown") == 8
|
||||
def test_unknown_default_bottom(self):
|
||||
assert position_to_ass_alignment("unknown") == 2
|
||||
|
||||
def test_empty_default_top(self):
|
||||
assert position_to_ass_alignment("") == 8
|
||||
def test_empty_default_bottom(self):
|
||||
assert position_to_ass_alignment("") == 2
|
||||
|
||||
|
||||
# ── Style 行构建 ─────────────────────────────────────────────────────────────
|
||||
@@ -747,3 +747,45 @@ class TestTitleFreePosition:
|
||||
line for line in content.splitlines() if line.startswith("Dialogue:") and "SubtitleStyle" in line
|
||||
][0]
|
||||
assert "\\pos(" not in sub_dialogue
|
||||
|
||||
|
||||
class TestDefaultPositionBottom:
|
||||
"""默认 position 应为 bottom(alignment=2),与前端 DEFAULT_TITLE_SETTINGS 对齐。"""
|
||||
|
||||
def _base_kwargs(self):
|
||||
return dict(
|
||||
video_width=1080,
|
||||
video_height=1920,
|
||||
video_duration=10.0,
|
||||
title_text="测试标题",
|
||||
)
|
||||
|
||||
def test_no_position_defaults_to_bottom_alignment(self):
|
||||
"""不传 position 时,Alignment 应为 2(bottom)。"""
|
||||
content = build_ass_content(
|
||||
**self._base_kwargs(),
|
||||
title_config={"size": 36},
|
||||
)
|
||||
style_line = [line for line in content.splitlines() if line.startswith("Style: TitleStyle")][0]
|
||||
fields = [f.strip() for f in style_line.split(",")]
|
||||
assert fields[18] == "2", f"Expected alignment 2 (bottom), got {fields[18]}"
|
||||
|
||||
def test_no_position_no_coords_defaults_to_bottom(self):
|
||||
"""不传 position 也不传坐标时,走 bottom 三档逻辑。"""
|
||||
content = build_ass_content(
|
||||
**self._base_kwargs(),
|
||||
title_config={},
|
||||
)
|
||||
style_line = [line for line in content.splitlines() if line.startswith("Style: TitleStyle")][0]
|
||||
fields = [f.strip() for f in style_line.split(",")]
|
||||
assert fields[18] == "2"
|
||||
|
||||
def test_explicit_top_still_works(self):
|
||||
"""显式传 position='top' 仍然得到 alignment=8。"""
|
||||
content = build_ass_content(
|
||||
**self._base_kwargs(),
|
||||
title_config={"position": "top", "size": 36},
|
||||
)
|
||||
style_line = [line for line in content.splitlines() if line.startswith("Style: TitleStyle")][0]
|
||||
fields = [f.strip() for f in style_line.split(",")]
|
||||
assert fields[18] == "8"
|
||||
|
||||
@@ -0,0 +1,314 @@
|
||||
"""Tests for bad fingerprint (black screen / uniform color) filtering.
|
||||
|
||||
Issue: 1秒黑屏视频(所有帧phash几乎相同)与任何视频的距离都~30,造成虚假匹配。
|
||||
Fix: _is_bad_fingerprint() 检测并跳过这类低质量指纹。
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import sys
|
||||
from unittest.mock import MagicMock, patch
|
||||
|
||||
import pytest
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Mock heavy deps before importing dedup module (same pattern as test_dedup_engine.py)
|
||||
# ---------------------------------------------------------------------------
|
||||
_ORIGINAL_MODULES = dict(sys.modules)
|
||||
_MOCKED_MODULE_NAMES: list[str] = []
|
||||
|
||||
|
||||
def _mock_if_absent(name: str, mock_obj=None):
|
||||
if name not in sys.modules:
|
||||
sys.modules[name] = mock_obj if mock_obj is not None else MagicMock()
|
||||
_MOCKED_MODULE_NAMES.append(name)
|
||||
|
||||
|
||||
_mock_if_absent("ffmpeg")
|
||||
for mod_name in ["worker_app", "worker_app.celery_app", "worker_app.db"]:
|
||||
_mock_if_absent(mod_name)
|
||||
if "worker_app.celery_app" in sys.modules and isinstance(sys.modules["worker_app.celery_app"], MagicMock):
|
||||
sys.modules["worker_app.celery_app"].celery_app = MagicMock()
|
||||
if "worker_app.db" in sys.modules and isinstance(sys.modules["worker_app.db"], MagicMock):
|
||||
sys.modules["worker_app.db"].SessionLocal = MagicMock()
|
||||
_mock_if_absent("celery", MagicMock())
|
||||
if "celery" in sys.modules and isinstance(sys.modules["celery"], MagicMock):
|
||||
sys.modules["celery"].Task = object
|
||||
_mock_if_absent("packages.shared.storage")
|
||||
_mock_if_absent("packages.adapters.sqlalchemy_impl.generated_video_repository")
|
||||
|
||||
_HAS_CV2 = False
|
||||
try:
|
||||
import cv2 as _cv2
|
||||
|
||||
if not isinstance(_cv2, MagicMock):
|
||||
_HAS_CV2 = True
|
||||
except (ImportError, ModuleNotFoundError):
|
||||
pass
|
||||
|
||||
if not _HAS_CV2:
|
||||
_mock_if_absent("cv2")
|
||||
|
||||
import numpy as np # noqa: E402
|
||||
|
||||
from apps.worker.video_processing.dedup import ( # noqa: E402
|
||||
VideoDeduplicator,
|
||||
VideoFingerprint,
|
||||
)
|
||||
|
||||
# Restore mocked modules
|
||||
for _name in ["worker_app", "worker_app.celery_app", "worker_app.db", "celery"]:
|
||||
if _name in _MOCKED_MODULE_NAMES:
|
||||
sys.modules.pop(_name, None)
|
||||
_MOCKED_MODULE_NAMES.remove(_name)
|
||||
|
||||
|
||||
@pytest.fixture(autouse=True, scope="session")
|
||||
def _cleanup_mocks():
|
||||
yield
|
||||
for name in _MOCKED_MODULE_NAMES:
|
||||
sys.modules.pop(name, None)
|
||||
|
||||
|
||||
# ── _is_bad_fingerprint 单元测试 ─────────────────────────────────
|
||||
|
||||
|
||||
class TestIsBadFingerprint:
|
||||
"""VideoDeduplicator._is_bad_fingerprint() 静态方法测试。"""
|
||||
|
||||
def test_empty_phashes_is_bad(self):
|
||||
"""空 phash 列表视为坏指纹。"""
|
||||
assert VideoDeduplicator._is_bad_fingerprint([]) is True
|
||||
|
||||
def test_single_phash_is_not_bad(self):
|
||||
"""单帧视频不视为坏指纹(短视频或抽帧不足)。"""
|
||||
assert VideoDeduplicator._is_bad_fingerprint(["abcdef0123456789"]) is False
|
||||
|
||||
def test_all_identical_phashes_is_bad(self):
|
||||
"""多帧但所有 phash 完全相同 → 黑屏/纯色视频。"""
|
||||
phashes = ["aaaaaaaaaaaaaaaa"] * 5
|
||||
assert VideoDeduplicator._is_bad_fingerprint(phashes) is True
|
||||
|
||||
def test_two_identical_phashes_is_bad(self):
|
||||
"""两帧完全相同也视为坏指纹。"""
|
||||
assert VideoDeduplicator._is_bad_fingerprint(["bbbbbbbbbbbbbbbb", "bbbbbbbbbbbbbbbb"]) is True
|
||||
|
||||
def test_all_very_similar_phashes_is_bad(self):
|
||||
"""多帧 phash 之间的汉明距离都 < 3 → 近似黑屏。"""
|
||||
phashes = ["0000000000000000", "0000000000000001", "0000000000000002"]
|
||||
assert VideoDeduplicator._is_bad_fingerprint(phashes) is True
|
||||
|
||||
def test_diverse_phashes_is_good(self):
|
||||
"""多样化的 phash 列表是有效指纹。"""
|
||||
phashes = [
|
||||
"abcdef0123456789",
|
||||
"1234567890abcdef",
|
||||
"fedcba9876543210",
|
||||
"0123456789abcdef",
|
||||
]
|
||||
assert VideoDeduplicator._is_bad_fingerprint(phashes) is False
|
||||
|
||||
def test_mixed_similar_and_different_is_good(self):
|
||||
"""有些 phash 相似但有足够多样的 → 有效指纹。"""
|
||||
phashes = [
|
||||
"0000000000000000",
|
||||
"0000000000000001",
|
||||
"0000000000000002",
|
||||
"ffffffffffffffff",
|
||||
]
|
||||
assert VideoDeduplicator._is_bad_fingerprint(phashes) is False
|
||||
|
||||
def test_known_black_screen_phashes(self):
|
||||
"""已知黑屏视频的 phash 特征(全零或均匀分布)。"""
|
||||
assert VideoDeduplicator._is_bad_fingerprint(["0000000000000000"] * 10) is True
|
||||
assert VideoDeduplicator._is_bad_fingerprint(["ffffffffffffffff"] * 8) is True
|
||||
assert VideoDeduplicator._is_bad_fingerprint(["9999999999999966"] * 6) is True
|
||||
|
||||
|
||||
# ── Helper ──────────────────────────────────────────────────────
|
||||
|
||||
|
||||
def _make_existing_video(video_id, md5, phashes):
|
||||
"""创建 mock 视频记录。"""
|
||||
video = MagicMock()
|
||||
video.id = video_id
|
||||
video.video_fingerprint = {
|
||||
"md5": md5,
|
||||
"keyframe_phashes": phashes,
|
||||
"color_histograms": [],
|
||||
}
|
||||
return video
|
||||
|
||||
|
||||
# ── check_duplicate 集成测试 ────────────────────────────────────
|
||||
|
||||
|
||||
class TestCheckDuplicateBadFingerprint:
|
||||
"""check_duplicate 跳过坏指纹视频。"""
|
||||
|
||||
def test_black_screen_existing_video_skipped(self):
|
||||
"""已有视频是黑屏指纹 → 被跳过,不匹配。"""
|
||||
deduplicator = VideoDeduplicator()
|
||||
mock_session = MagicMock()
|
||||
|
||||
black_screen = _make_existing_video("vid-black", "md5_black", ["aaaaaaaaaaaaaaaa"] * 5)
|
||||
mock_repo = MagicMock()
|
||||
mock_repo.list_by_user.return_value = [black_screen]
|
||||
|
||||
fingerprint = VideoFingerprint(
|
||||
md5="md5_normal",
|
||||
keyframe_phashes=["aaaaaaaaaaaaaaaa"] * 5,
|
||||
color_histograms=[],
|
||||
duration=10.0,
|
||||
resolution=(1280, 720),
|
||||
)
|
||||
|
||||
with patch(
|
||||
"apps.worker.video_processing.dedup.SQLAlchemyGeneratedVideoRepository",
|
||||
return_value=mock_repo,
|
||||
):
|
||||
result = deduplicator.check_duplicate(fingerprint, "proj-1", mock_session, scope="user", user_id="user-1")
|
||||
|
||||
assert result is None
|
||||
|
||||
def test_normal_existing_video_not_skipped(self):
|
||||
"""正常视频不会被坏指纹过滤跳过。"""
|
||||
deduplicator = VideoDeduplicator()
|
||||
mock_session = MagicMock()
|
||||
|
||||
normal = _make_existing_video(
|
||||
"vid-normal",
|
||||
"md5_normal_existing",
|
||||
["abcdef0123456789", "1234567890abcdef", "fedcba9876543210"],
|
||||
)
|
||||
mock_repo = MagicMock()
|
||||
mock_repo.list_by_user.return_value = [normal]
|
||||
|
||||
fingerprint = VideoFingerprint(
|
||||
md5="md5_normal_new",
|
||||
keyframe_phashes=["abcdef0123456789", "1234567890abcdef", "fedcba9876543210"],
|
||||
color_histograms=[],
|
||||
duration=10.0,
|
||||
resolution=(1280, 720),
|
||||
)
|
||||
|
||||
with patch(
|
||||
"apps.worker.video_processing.dedup.SQLAlchemyGeneratedVideoRepository",
|
||||
return_value=mock_repo,
|
||||
):
|
||||
result = deduplicator.check_duplicate(fingerprint, "proj-1", mock_session, scope="user", user_id="user-1")
|
||||
|
||||
assert result is not None
|
||||
assert result["duplicate"] is True
|
||||
|
||||
def test_md5_match_overrides_bad_fingerprint(self):
|
||||
"""MD5 精确匹配优先于坏指纹过滤。"""
|
||||
deduplicator = VideoDeduplicator()
|
||||
mock_session = MagicMock()
|
||||
|
||||
black_screen = _make_existing_video("vid-black", "same_md5", ["aaaaaaaaaaaaaaaa"] * 5)
|
||||
mock_repo = MagicMock()
|
||||
mock_repo.list_by_user.return_value = [black_screen]
|
||||
|
||||
fingerprint = VideoFingerprint(
|
||||
md5="same_md5",
|
||||
keyframe_phashes=["bbbbbbbbbbbbbbbb"] * 3,
|
||||
color_histograms=[],
|
||||
duration=10.0,
|
||||
resolution=(1280, 720),
|
||||
)
|
||||
|
||||
with patch(
|
||||
"apps.worker.video_processing.dedup.SQLAlchemyGeneratedVideoRepository",
|
||||
return_value=mock_repo,
|
||||
):
|
||||
result = deduplicator.check_duplicate(fingerprint, "proj-1", mock_session, scope="user", user_id="user-1")
|
||||
|
||||
assert result is not None
|
||||
assert result["reason"] == "exact_md5_match"
|
||||
|
||||
|
||||
# ── compute_duplicate_rate 集成测试 ─────────────────────────────
|
||||
|
||||
|
||||
class TestComputeDuplicateRateBadFingerprint:
|
||||
"""compute_duplicate_rate 跳过坏指纹视频。"""
|
||||
|
||||
def test_black_screen_video_excluded_from_rate(self):
|
||||
"""黑屏视频不参与查重率计算。"""
|
||||
deduplicator = VideoDeduplicator()
|
||||
mock_session = MagicMock()
|
||||
|
||||
videos = [
|
||||
_make_existing_video("vid-b1", "md5_b1", ["cccccccccccccccc"] * 5),
|
||||
_make_existing_video("vid-b2", "md5_b2", ["dddddddddddddddd"] * 5),
|
||||
_make_existing_video("vid-b3", "md5_b3", ["eeeeeeeeeeeeeeee"] * 5),
|
||||
_make_existing_video(
|
||||
"vid-normal",
|
||||
"md5_n",
|
||||
["abcdef0123456789", "1234567890abcdef", "fedcba9876543210"],
|
||||
),
|
||||
]
|
||||
mock_repo = MagicMock()
|
||||
mock_repo.list_by_user.return_value = videos
|
||||
|
||||
fingerprint = VideoFingerprint(
|
||||
md5="md5_new",
|
||||
keyframe_phashes=["abcdef0123456789", "1234567890abcdef", "fedcba9876543210"],
|
||||
color_histograms=[],
|
||||
duration=10.0,
|
||||
resolution=(1280, 720),
|
||||
)
|
||||
|
||||
with patch(
|
||||
"apps.worker.video_processing.dedup.SQLAlchemyGeneratedVideoRepository",
|
||||
return_value=mock_repo,
|
||||
):
|
||||
result = deduplicator.compute_duplicate_rate(
|
||||
fingerprint,
|
||||
"proj-1",
|
||||
"vid-new",
|
||||
mock_session,
|
||||
scope="user",
|
||||
user_id="user-1",
|
||||
)
|
||||
|
||||
assert result is not None
|
||||
assert isinstance(result["duplicate_rate"], float)
|
||||
assert isinstance(result["match_count"], int)
|
||||
|
||||
def test_only_black_screen_videos_zero_rate(self):
|
||||
"""所有已有视频都是黑屏 → 查重率为 0。"""
|
||||
deduplicator = VideoDeduplicator()
|
||||
mock_session = MagicMock()
|
||||
|
||||
videos = [
|
||||
_make_existing_video("vid-b1", "md5_b1", ["aaaaaaaaaaaaaaaa"] * 5),
|
||||
_make_existing_video("vid-b2", "md5_b2", ["bbbbbbbbbbbbbbbb"] * 5),
|
||||
]
|
||||
mock_repo = MagicMock()
|
||||
mock_repo.list_by_user.return_value = videos
|
||||
|
||||
fingerprint = VideoFingerprint(
|
||||
md5="md5_new",
|
||||
keyframe_phashes=["aaaaaaaaaaaaaaaa"] * 5,
|
||||
color_histograms=[],
|
||||
duration=10.0,
|
||||
resolution=(1280, 720),
|
||||
)
|
||||
|
||||
with patch(
|
||||
"apps.worker.video_processing.dedup.SQLAlchemyGeneratedVideoRepository",
|
||||
return_value=mock_repo,
|
||||
):
|
||||
result = deduplicator.compute_duplicate_rate(
|
||||
fingerprint,
|
||||
"proj-1",
|
||||
"vid-new",
|
||||
mock_session,
|
||||
scope="user",
|
||||
user_id="user-1",
|
||||
)
|
||||
|
||||
assert result["duplicate_rate"] == 0.0
|
||||
assert result["match_count"] == 0
|
||||
@@ -0,0 +1,334 @@
|
||||
"""Tests for Issue #1670 — 跨视频片段避让(生成前注入已用区间)."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from datetime import datetime, timezone
|
||||
from unittest.mock import MagicMock, patch
|
||||
|
||||
import pytest
|
||||
|
||||
from packages.adapters.sqlalchemy_impl.edit_plan_clip_repository import (
|
||||
SQLAlchemyEditPlanClipRepository,
|
||||
)
|
||||
from packages.domain.edit_plan_clip import EditPlanClip, EditPlanClipStatus
|
||||
from packages.domain.plan_generator_utils import (
|
||||
_distribute_one_take,
|
||||
distribute_assets,
|
||||
)
|
||||
|
||||
# ── Repository 层测试 ─────────────────────────────────────────────────────────
|
||||
|
||||
|
||||
class TestListUsedSegmentsByUser:
|
||||
"""测试 list_used_segments_by_user 方法."""
|
||||
|
||||
def _make_repo(self, session_mock):
|
||||
return SQLAlchemyEditPlanClipRepository(session_mock)
|
||||
|
||||
def test_empty_user_id_returns_empty_dict(self):
|
||||
"""空 user_id 直接返回空 dict,不查 DB."""
|
||||
session = MagicMock()
|
||||
repo = self._make_repo(session)
|
||||
result = repo.list_used_segments_by_user("")
|
||||
assert result == {}
|
||||
session.query.assert_not_called()
|
||||
|
||||
def test_no_completed_plans_returns_empty_dict(self):
|
||||
"""用户没有已完成的 plan 时返回空 dict."""
|
||||
session = MagicMock()
|
||||
# Mock plan query returns empty
|
||||
plan_query = MagicMock()
|
||||
plan_query.filter.return_value = plan_query
|
||||
plan_query.order_by.return_value = plan_query
|
||||
plan_query.limit.return_value = plan_query
|
||||
plan_query.all.return_value = []
|
||||
session.query.return_value = plan_query
|
||||
|
||||
repo = self._make_repo(session)
|
||||
result = repo.list_used_segments_by_user("user_123")
|
||||
assert result == {}
|
||||
|
||||
def test_aggregates_clips_from_multiple_plans(self):
|
||||
"""从多个已完成 plan 的 clips 聚合已用区间."""
|
||||
session = MagicMock()
|
||||
|
||||
# Mock plan query: 2 completed plans
|
||||
plan_query = MagicMock()
|
||||
plan_query.filter.return_value = plan_query
|
||||
plan_query.order_by.return_value = plan_query
|
||||
plan_query.limit.return_value = plan_query
|
||||
plan_query.all.return_value = [("plan_1",), ("plan_2",)]
|
||||
session.query.return_value = plan_query
|
||||
|
||||
# Mock clip query: clips from both plans
|
||||
clip_query = MagicMock()
|
||||
clip_query.filter.return_value = clip_query
|
||||
clip_query.all.return_value = [
|
||||
("asset_A", 0.0, 5.0), # plan_1, asset A: 0~5s
|
||||
("asset_A", 10.0, 3.0), # plan_1, asset A: 10~13s
|
||||
("asset_B", 2.0, 4.0), # plan_2, asset B: 2~6s
|
||||
]
|
||||
# Second session.query call is for clips
|
||||
session.query.side_effect = [plan_query, clip_query]
|
||||
|
||||
repo = self._make_repo(session)
|
||||
result = repo.list_used_segments_by_user("user_123")
|
||||
|
||||
assert "asset_A" in result
|
||||
assert len(result["asset_A"]) == 2
|
||||
assert result["asset_A"][0] == (0.0, 5.0)
|
||||
assert result["asset_A"][1] == (10.0, 13.0)
|
||||
assert "asset_B" in result
|
||||
assert result["asset_B"][0] == (2.0, 6.0)
|
||||
|
||||
def test_respects_limit_recent_parameter(self):
|
||||
"""limit_recent 参数限制查询的 plan 数量."""
|
||||
session = MagicMock()
|
||||
|
||||
plan_query = MagicMock()
|
||||
plan_query.filter.return_value = plan_query
|
||||
plan_query.order_by.return_value = plan_query
|
||||
plan_query.limit.return_value = plan_query
|
||||
plan_query.all.return_value = [("plan_1",)]
|
||||
session.query.return_value = plan_query
|
||||
|
||||
clip_query = MagicMock()
|
||||
clip_query.filter.return_value = clip_query
|
||||
clip_query.all.return_value = [("asset_X", 1.0, 2.0)]
|
||||
session.query.side_effect = [plan_query, clip_query]
|
||||
|
||||
repo = self._make_repo(session)
|
||||
result = repo.list_used_segments_by_user("user_123", limit_recent=10)
|
||||
|
||||
# Verify limit was called with the parameter
|
||||
plan_query.limit.assert_called_once_with(10)
|
||||
assert "asset_X" in result
|
||||
|
||||
|
||||
# ── Domain 层测试 ─────────────────────────────────────────────────────────────
|
||||
|
||||
|
||||
class TestDistributeAssetsWithExternalSegments:
|
||||
"""测试 distribute_assets 传入 external_used_segments 的行为."""
|
||||
|
||||
def _make_clips(self, count: int, duration: float = 3.0) -> list[EditPlanClip]:
|
||||
"""创建指定数量的 MAIN 类型 clips."""
|
||||
return [
|
||||
EditPlanClip(
|
||||
id=f"clip_{i}",
|
||||
plan_id="plan_1",
|
||||
clip_type="main",
|
||||
order=i,
|
||||
template_clip_config_id="",
|
||||
asset_id="",
|
||||
text_content="",
|
||||
start_time=0.0,
|
||||
duration=duration,
|
||||
status=EditPlanClipStatus.PENDING,
|
||||
)
|
||||
for i in range(count)
|
||||
]
|
||||
|
||||
def test_external_used_segments_none_backward_compatible(self):
|
||||
"""external_used_segments=None 时行为不变(向后兼容)."""
|
||||
clips = self._make_clips(3)
|
||||
asset_ids = ["asset_1", "asset_2", "asset_3"]
|
||||
asset_durations = {aid: 30.0 for aid in asset_ids}
|
||||
|
||||
# Should not raise
|
||||
distribute_assets(
|
||||
clips,
|
||||
asset_ids,
|
||||
"one_take",
|
||||
asset_durations=asset_durations,
|
||||
external_used_segments=None,
|
||||
)
|
||||
|
||||
# All clips should have assets assigned
|
||||
for clip in clips:
|
||||
assert clip.asset_id != ""
|
||||
|
||||
def test_external_used_segments_avoids_existing_ranges(self):
|
||||
"""传入 external_used_segments 后,新分配的 start_time 避开已有区间."""
|
||||
clips = self._make_clips(2, duration=3.0)
|
||||
asset_ids = ["asset_1"]
|
||||
asset_durations = {"asset_1": 30.0}
|
||||
|
||||
# Pretend asset_1 0~10s is already used by another video
|
||||
external = {"asset_1": [(0.0, 10.0)]}
|
||||
|
||||
# Run multiple times to check that start_time always avoids 0~10s
|
||||
# (with some randomness, but the avoidance should be consistent)
|
||||
for _ in range(10):
|
||||
test_clips = self._make_clips(1, duration=3.0)
|
||||
distribute_assets(
|
||||
test_clips,
|
||||
asset_ids,
|
||||
"one_take",
|
||||
asset_durations=asset_durations,
|
||||
external_used_segments=external,
|
||||
)
|
||||
start = test_clips[0].start_time
|
||||
# Start time + duration (3s) should not overlap with 0~10
|
||||
# i.e., start >= 10.0 or start + 3 <= 0.0 (impossible since start >= 0)
|
||||
assert (
|
||||
start >= 10.0 or start + 3.0 <= 0.0 or start >= 10.0
|
||||
), f"start_time {start} overlaps with existing segment 0~10"
|
||||
|
||||
def test_external_used_segments_deep_copy(self):
|
||||
"""external_used_segments 会被深拷贝,不会修改外部数据."""
|
||||
external = {"asset_1": [(0.0, 5.0)]}
|
||||
original = {"asset_1": [(0.0, 5.0)]}
|
||||
|
||||
clips = self._make_clips(1, duration=2.0)
|
||||
asset_ids = ["asset_1"]
|
||||
asset_durations = {"asset_1": 20.0}
|
||||
|
||||
distribute_assets(
|
||||
clips,
|
||||
asset_ids,
|
||||
"one_take",
|
||||
asset_durations=asset_durations,
|
||||
external_used_segments=external,
|
||||
)
|
||||
|
||||
# External dict should be unchanged
|
||||
assert external == original
|
||||
|
||||
def test_empty_external_used_segments_same_as_none(self):
|
||||
"""空 dict 的 external_used_segments 行为与 None 相同."""
|
||||
clips = self._make_clips(2, duration=3.0)
|
||||
asset_ids = ["asset_1", "asset_2"]
|
||||
asset_durations = {aid: 30.0 for aid in asset_ids}
|
||||
|
||||
# Should not raise and should assign assets normally
|
||||
distribute_assets(
|
||||
clips,
|
||||
asset_ids,
|
||||
"one_take",
|
||||
asset_durations=asset_durations,
|
||||
external_used_segments={},
|
||||
)
|
||||
for clip in clips:
|
||||
assert clip.asset_id != ""
|
||||
|
||||
|
||||
# ── Service 层测试 ────────────────────────────────────────────────────────────
|
||||
|
||||
|
||||
class TestServiceLayerIntegration:
|
||||
"""测试 _distribute_assets 在 service 层的查询逻辑."""
|
||||
|
||||
def _make_service(self, clip_repo_mock, asset_repo_mock=None):
|
||||
"""创建 PlanGeneratorService 并注入 mock repos."""
|
||||
|
||||
from apps.api.app.services.plan_generator_service import PlanGeneratorService
|
||||
|
||||
with (
|
||||
patch("apps.api.app.services.plan_generator_service.SQLAlchemyEditPlanRepository"),
|
||||
patch(
|
||||
"apps.api.app.services.plan_generator_service.SQLAlchemyEditPlanClipRepository",
|
||||
return_value=clip_repo_mock,
|
||||
),
|
||||
):
|
||||
db = MagicMock()
|
||||
svc = PlanGeneratorService(db, asset_repo=asset_repo_mock)
|
||||
svc._clip_repo = clip_repo_mock
|
||||
return svc
|
||||
|
||||
def _make_clip(self):
|
||||
return EditPlanClip(
|
||||
id="clip_1",
|
||||
plan_id="plan_1",
|
||||
clip_type="main",
|
||||
order=0,
|
||||
template_clip_config_id="",
|
||||
asset_id="",
|
||||
text_content="",
|
||||
start_time=0.0,
|
||||
duration=3.0,
|
||||
status=EditPlanClipStatus.PENDING,
|
||||
)
|
||||
|
||||
def test_query_called_with_user_id(self):
|
||||
"""有 user_id 时调用 list_used_segments_by_user."""
|
||||
clip_repo = MagicMock()
|
||||
clip_repo.list_used_segments_by_user.return_value = {"asset_A": [(0.0, 5.0)]}
|
||||
asset_repo = MagicMock()
|
||||
asset_repo.get.return_value = None # smart_match fallback
|
||||
|
||||
svc = self._make_service(clip_repo, asset_repo)
|
||||
clips = [self._make_clip()]
|
||||
|
||||
svc._distribute_assets(
|
||||
clips,
|
||||
["asset_A"],
|
||||
"one_take",
|
||||
asset_durations={"asset_A": 30.0},
|
||||
user_id="user_123",
|
||||
)
|
||||
|
||||
clip_repo.list_used_segments_by_user.assert_called_once_with("user_123", limit_recent=50)
|
||||
|
||||
def test_query_not_called_without_user_id(self):
|
||||
"""无 user_id 时不调用查询."""
|
||||
clip_repo = MagicMock()
|
||||
asset_repo = MagicMock()
|
||||
asset_repo.get.return_value = None
|
||||
|
||||
svc = self._make_service(clip_repo, asset_repo)
|
||||
clips = [self._make_clip()]
|
||||
|
||||
svc._distribute_assets(
|
||||
clips,
|
||||
["asset_A"],
|
||||
"one_take",
|
||||
asset_durations={"asset_A": 30.0},
|
||||
user_id="",
|
||||
)
|
||||
|
||||
clip_repo.list_used_segments_by_user.assert_not_called()
|
||||
|
||||
def test_query_failure_does_not_block_generation(self):
|
||||
"""查询失败时不阻塞生成,回退到纯随机."""
|
||||
clip_repo = MagicMock()
|
||||
clip_repo.list_used_segments_by_user.side_effect = Exception("DB error")
|
||||
asset_repo = MagicMock()
|
||||
asset_repo.get.return_value = None
|
||||
|
||||
svc = self._make_service(clip_repo, asset_repo)
|
||||
clips = [self._make_clip()]
|
||||
|
||||
# Should not raise
|
||||
svc._distribute_assets(
|
||||
clips,
|
||||
["asset_A"],
|
||||
"one_take",
|
||||
asset_durations={"asset_A": 30.0},
|
||||
user_id="user_123",
|
||||
)
|
||||
|
||||
# Clip should still get an asset assigned (fallback to random)
|
||||
assert clips[0].asset_id == "asset_A"
|
||||
|
||||
def test_preview_and_final_both_query(self):
|
||||
"""预览和正式生成都触发查询."""
|
||||
for random_selection in [True, False]:
|
||||
clip_repo = MagicMock()
|
||||
clip_repo.list_used_segments_by_user.return_value = {}
|
||||
asset_repo = MagicMock()
|
||||
asset_repo.get.return_value = None
|
||||
|
||||
svc = self._make_service(clip_repo, asset_repo)
|
||||
clips = [self._make_clip()]
|
||||
|
||||
svc._distribute_assets(
|
||||
clips,
|
||||
["asset_A"],
|
||||
"one_take",
|
||||
random_selection=random_selection,
|
||||
asset_durations={"asset_A": 30.0},
|
||||
user_id="user_123",
|
||||
)
|
||||
|
||||
clip_repo.list_used_segments_by_user.assert_called_once()
|
||||
@@ -285,8 +285,10 @@ class TestVideoDeduplicatorCheckDuplicate:
|
||||
result = deduplicator.check_duplicate(fingerprint, "proj-1", mock_session)
|
||||
assert result is not None
|
||||
assert result["duplicate"] is True
|
||||
assert result["similarity"] == 1.0 # distance=0 → 1.0
|
||||
assert result["reason"] == "phash_similar"
|
||||
assert result["similarity"] == pytest.approx(
|
||||
0.85, abs=0.01
|
||||
) # combined: 0.7*1.0 + 0.3*0.5 (no hist fallback)
|
||||
assert result["reason"] == "phash_histogram_fusion"
|
||||
finally:
|
||||
self._restore_repo(mod, orig)
|
||||
|
||||
@@ -425,8 +427,11 @@ class TestVideoDeduplicatorCheckDuplicate:
|
||||
result = deduplicator.check_duplicate(fingerprint, "proj-1", mock_session)
|
||||
assert result is not None
|
||||
assert result["duplicate"] is True
|
||||
# similarity = 1.0 - (1 / 64) = 0.984375
|
||||
assert abs(result["similarity"] - (1.0 - 1.0 / 64)) < 1e-6
|
||||
# 新算法: median_distance=1, phash_sim=1-1/64=0.984375
|
||||
# 无直方图 → hist_sim=0.5(fallback)
|
||||
# combined = 0.7*0.984375 + 0.3*0.5 = 0.839062
|
||||
expected_sim = 0.7 * (1.0 - 1.0 / 64) + 0.3 * 0.5
|
||||
assert abs(result["similarity"] - expected_sim) < 1e-6
|
||||
finally:
|
||||
self._restore_repo(mod, orig)
|
||||
|
||||
@@ -456,7 +461,9 @@ class TestVideoDeduplicatorCheckDuplicate:
|
||||
result = deduplicator.check_duplicate(fingerprint, "proj-1", mock_session)
|
||||
assert result is not None
|
||||
assert result["duplicate"] is True
|
||||
assert result["similarity"] == 1.0 # avg_distance = 0
|
||||
# 新算法: median_distance=0, phash_sim=1.0, hist_sim=0.5(fallback)
|
||||
# combined = 0.7*1.0 + 0.3*0.5 = 0.85
|
||||
assert result["similarity"] == pytest.approx(0.85, abs=0.01)
|
||||
finally:
|
||||
self._restore_repo(mod, orig)
|
||||
|
||||
@@ -539,7 +546,7 @@ class TestVideoDeduplicatorCheckBatchDuplicate:
|
||||
result = deduplicator.check_batch_duplicate(fingerprint, "batch-1", "vid-self", mock_session)
|
||||
assert result is not None
|
||||
assert result["duplicate"] is True
|
||||
assert result["reason"] == "batch_phash_similar"
|
||||
assert result["reason"] == "batch_phash_histogram_fusion"
|
||||
finally:
|
||||
self._restore_repo(mod, orig)
|
||||
|
||||
|
||||
@@ -43,7 +43,11 @@ class TestDedupHelpersUserIdPassthrough:
|
||||
mock_deduplicator = MagicMock()
|
||||
mock_deduplicator.compute_fingerprint.return_value = mock_fingerprint
|
||||
mock_deduplicator.check_duplicate.return_value = None
|
||||
mock_deduplicator.compute_duplicate_rate.return_value = 42.5
|
||||
mock_deduplicator.compute_duplicate_rate.return_value = {
|
||||
"duplicate_rate": 42.5,
|
||||
"visual_similarity": 0.7,
|
||||
"match_count": 2,
|
||||
}
|
||||
|
||||
with (
|
||||
patch(
|
||||
@@ -85,7 +89,11 @@ class TestDedupHelpersUserIdPassthrough:
|
||||
mock_deduplicator = MagicMock()
|
||||
mock_deduplicator.compute_fingerprint.return_value = mock_fingerprint
|
||||
mock_deduplicator.check_duplicate.return_value = None
|
||||
mock_deduplicator.compute_duplicate_rate.return_value = 0.0
|
||||
mock_deduplicator.compute_duplicate_rate.return_value = {
|
||||
"duplicate_rate": 0.0,
|
||||
"visual_similarity": 0.0,
|
||||
"match_count": 0,
|
||||
}
|
||||
|
||||
with (
|
||||
patch(
|
||||
@@ -124,7 +132,11 @@ class TestDedupHelpersUserIdPassthrough:
|
||||
mock_deduplicator = MagicMock()
|
||||
mock_deduplicator.compute_fingerprint.return_value = mock_fingerprint
|
||||
mock_deduplicator.check_duplicate.return_value = None
|
||||
mock_deduplicator.compute_duplicate_rate.return_value = 78.5
|
||||
mock_deduplicator.compute_duplicate_rate.return_value = {
|
||||
"duplicate_rate": 78.5,
|
||||
"visual_similarity": 0.85,
|
||||
"match_count": 3,
|
||||
}
|
||||
|
||||
with (
|
||||
patch(
|
||||
@@ -147,6 +159,6 @@ class TestDedupHelpersUserIdPassthrough:
|
||||
)
|
||||
|
||||
# 验证 update 被调用(包含 duplicate_rate 的记录)
|
||||
mock_video_repo.update.assert_called_once()
|
||||
updated_video = mock_video_repo.update.call_args[0][0]
|
||||
mock_video_repo.create.assert_called_once()
|
||||
updated_video = mock_video_repo.create.call_args[0][0]
|
||||
assert updated_video.duplicate_rate == 78.5
|
||||
|
||||
@@ -181,70 +181,68 @@ class TestVideoFingerprint:
|
||||
assert d["color_histograms"] == []
|
||||
|
||||
|
||||
class TestAverageHistogramSimilarity:
|
||||
"""_average_histogram_similarity 直方图相似度测试."""
|
||||
class TestBhattacharyyaCoefficient:
|
||||
"""_bhattacharyya_coefficient Bhattacharyya 系数测试."""
|
||||
|
||||
def test_identical_histograms(self):
|
||||
"""完全相同的直方图相似度为1.0."""
|
||||
hist = [[0.5, 0.5, 0.0], [0.3, 0.4, 0.3]]
|
||||
sim = VideoDeduplicator._average_histogram_similarity(hist, hist)
|
||||
assert sim == pytest.approx(1.0)
|
||||
"""完全相同的直方图系数为1.0."""
|
||||
hist = [0.5, 0.5, 0.0, 0.3]
|
||||
bc = VideoDeduplicator._bhattacharyya_coefficient(hist, hist)
|
||||
# Σ √(a[i]*a[i]) = Σ a[i] = 1.0 (normalized)
|
||||
assert bc == pytest.approx(sum(h for h in hist))
|
||||
|
||||
def test_empty_first_list(self):
|
||||
def test_zero_histograms(self):
|
||||
"""全零直方图系数为0."""
|
||||
bc = VideoDeduplicator._bhattacharyya_coefficient([0.0, 0.0], [0.0, 0.0])
|
||||
assert bc == 0.0
|
||||
|
||||
def test_orthogonal_histograms(self):
|
||||
"""正交直方图(无重叠)系数为0."""
|
||||
bc = VideoDeduplicator._bhattacharyya_coefficient([1.0, 0.0], [0.0, 1.0])
|
||||
assert bc == pytest.approx(0.0)
|
||||
|
||||
def test_different_lengths(self):
|
||||
"""不同长度直方图取最小长度对齐."""
|
||||
bc = VideoDeduplicator._bhattacharyya_coefficient([1.0, 1.0, 0.0, 0.0], [1.0, 1.0])
|
||||
# 对齐到前2维: √(1*1) + √(1*1) = 2.0
|
||||
assert bc == pytest.approx(2.0)
|
||||
|
||||
def test_known_value(self):
|
||||
"""已知值验证."""
|
||||
# [0.25, 0.25, 0.25, 0.25] vs [0.25, 0.25, 0.25, 0.25]
|
||||
# BC = 4 * √(0.25 * 0.25) = 4 * 0.25 = 1.0
|
||||
hist = [0.25, 0.25, 0.25, 0.25]
|
||||
bc = VideoDeduplicator._bhattacharyya_coefficient(hist, hist)
|
||||
assert bc == pytest.approx(1.0)
|
||||
|
||||
|
||||
class TestComputeHistogramSimilarity:
|
||||
"""_compute_histogram_similarity 多帧直方图相似度测试."""
|
||||
|
||||
def test_identical_histogram_groups(self):
|
||||
"""完全相同的两组直方图."""
|
||||
hist = [[0.5, 0.5], [0.3, 0.4]]
|
||||
sim = VideoDeduplicator._compute_histogram_similarity(hist, hist)
|
||||
# Each hist finds best match = itself
|
||||
assert sim > 0.0
|
||||
|
||||
def test_empty_first(self):
|
||||
"""第一组为空返回0."""
|
||||
sim = VideoDeduplicator._average_histogram_similarity([], [[0.5, 0.5]])
|
||||
assert sim == 0.0
|
||||
assert VideoDeduplicator._compute_histogram_similarity([], [[0.5]]) == 0.0
|
||||
|
||||
def test_empty_second_list(self):
|
||||
def test_empty_second(self):
|
||||
"""第二组为空返回0."""
|
||||
sim = VideoDeduplicator._average_histogram_similarity([[0.5, 0.5]], [])
|
||||
assert sim == 0.0
|
||||
assert VideoDeduplicator._compute_histogram_similarity([[0.5]], []) == 0.0
|
||||
|
||||
def test_both_empty(self):
|
||||
"""两组都为空返回0."""
|
||||
sim = VideoDeduplicator._average_histogram_similarity([], [])
|
||||
assert sim == 0.0
|
||||
assert VideoDeduplicator._compute_histogram_similarity([], []) == 0.0
|
||||
|
||||
def test_orthogonal_histograms(self):
|
||||
"""正交直方图相似度为0."""
|
||||
# [1, 0] 和 [0, 1] 正交
|
||||
sim = VideoDeduplicator._average_histogram_similarity([[1.0, 0.0]], [[0.0, 1.0]])
|
||||
assert sim == pytest.approx(0.0)
|
||||
|
||||
def test_partial_similarity(self):
|
||||
"""部分相似."""
|
||||
# [1, 1] 和 [1, 0] 的余弦相似度 = 1/√2 ≈ 0.707
|
||||
sim = VideoDeduplicator._average_histogram_similarity([[1.0, 1.0]], [[1.0, 0.0]])
|
||||
assert sim == pytest.approx(1.0 / (2**0.5), rel=0.01)
|
||||
|
||||
def test_multiple_frames_best_match(self):
|
||||
def test_best_match_selection(self):
|
||||
"""多帧时取最佳匹配."""
|
||||
# 第一帧完全不同,第二帧完全相同 → 平均 best = (0 + 1) / 2 = 0.5
|
||||
sim = VideoDeduplicator._average_histogram_similarity(
|
||||
[[1.0, 0.0], [0.0, 1.0]],
|
||||
[[0.0, 1.0]], # 只有一帧,和第一帧0相似,和第二帧1相似
|
||||
)
|
||||
# 第一帧最佳匹配=0,第二帧最佳匹配=1,平均=0.5
|
||||
assert sim == pytest.approx(0.5)
|
||||
|
||||
def test_zero_norm_histogram_skipped(self):
|
||||
"""零范数直方图被跳过."""
|
||||
sim = VideoDeduplicator._average_histogram_similarity([[0.0, 0.0]], [[1.0, 1.0]])
|
||||
# 第一组的零范数被跳过,similarities为空,返回0
|
||||
assert sim == 0.0
|
||||
|
||||
def test_different_length_histograms(self):
|
||||
"""不同长度的直方图取最小长度对齐."""
|
||||
sim = VideoDeduplicator._average_histogram_similarity(
|
||||
[[1.0, 1.0, 0.0, 0.0]], # 4维
|
||||
[[1.0, 1.0]], # 2维
|
||||
)
|
||||
# 对齐到前2维,都是[1,1],相似度1.0
|
||||
# ha[0] 与 hb[0] 正交,与 hb[1] 完全相同
|
||||
a = [[1.0, 0.0]]
|
||||
b = [[0.0, 1.0], [1.0, 0.0]]
|
||||
sim = VideoDeduplicator._compute_histogram_similarity(a, b)
|
||||
# Best match for [1,0]: max(BC([1,0],[0,1]), BC([1,0],[1,0])) = max(0, 1) = 1
|
||||
assert sim == pytest.approx(1.0)
|
||||
|
||||
def test_similarity_in_zero_one_range(self):
|
||||
"""相似度在[0, 1]范围内."""
|
||||
hist_a = [np.random.rand(96).tolist() for _ in range(5)]
|
||||
hist_b = [np.random.rand(96).tolist() for _ in range(5)]
|
||||
sim = VideoDeduplicator._average_histogram_similarity(hist_a, hist_b)
|
||||
assert 0.0 <= sim <= 1.0
|
||||
|
||||
@@ -0,0 +1,234 @@
|
||||
"""Tests for two-phase commit pattern in dedup_helpers (#1664 follow-up).
|
||||
|
||||
Verifies that the new dedup_helpers.py:
|
||||
1. Creates video with all dedup fields in a single commit
|
||||
2. Still creates video when fingerprint computation fails
|
||||
3. Creates video with fingerprint but no rate when rate computation fails
|
||||
4. Never does a partial commit (no create + separate update)
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import sys
|
||||
from pathlib import Path
|
||||
from unittest.mock import MagicMock, patch
|
||||
|
||||
# Mock cv2/numpy before imports
|
||||
sys.modules.setdefault("cv2", MagicMock())
|
||||
sys.modules.setdefault("numpy", MagicMock())
|
||||
|
||||
ROOT = Path(__file__).resolve().parents[2]
|
||||
sys.path.insert(0, str(ROOT / "apps" / "api"))
|
||||
sys.path.insert(0, str(ROOT / "packages"))
|
||||
sys.path.insert(0, str(ROOT / "apps" / "worker"))
|
||||
|
||||
import os
|
||||
|
||||
os.environ.setdefault("JWT_SECRET_KEY", "unit-test-secret")
|
||||
os.environ.setdefault("DATABASE_URL", "sqlite:///test.db")
|
||||
|
||||
import pytest
|
||||
from video_processing.dedup_helpers import create_video_record_and_dedup
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def session():
|
||||
s = MagicMock()
|
||||
return s
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def mock_fingerprint():
|
||||
fp = MagicMock()
|
||||
fp.duration = 15000 # 15 seconds in ms
|
||||
fp.to_dict.return_value = {"md5": "abc123", "keyframe_phashes": ["aabb"], "color_histograms": []}
|
||||
fp.chunks = []
|
||||
fp.keyframe_phashes = ["aabb"]
|
||||
fp.color_histograms = []
|
||||
fp.md5 = "abc123"
|
||||
return fp
|
||||
|
||||
|
||||
class TestTwoPhaseCommit:
|
||||
"""Verify that dedup data is computed before commit."""
|
||||
|
||||
def test_video_created_with_all_dedup_fields(self, session, mock_fingerprint):
|
||||
"""When all computations succeed, video is created with all fields in one commit."""
|
||||
mock_repo = MagicMock()
|
||||
mock_deduplicator = MagicMock()
|
||||
mock_deduplicator.compute_fingerprint.return_value = mock_fingerprint
|
||||
mock_deduplicator.check_duplicate.return_value = None
|
||||
mock_deduplicator.compute_duplicate_rate.return_value = {
|
||||
"duplicate_rate": 42.5,
|
||||
"visual_similarity": 0.75,
|
||||
"match_count": 2,
|
||||
}
|
||||
|
||||
with (
|
||||
patch(
|
||||
"packages.adapters.sqlalchemy_impl.generated_video_repository.SQLAlchemyGeneratedVideoRepository",
|
||||
return_value=mock_repo,
|
||||
),
|
||||
patch("video_processing.dedup.VideoDeduplicator", return_value=mock_deduplicator),
|
||||
patch("video_processing.dedup._save_fingerprint_chunks"),
|
||||
):
|
||||
result = create_video_record_and_dedup(
|
||||
generation_task_id="task-001",
|
||||
project_id="proj-001",
|
||||
user_id="user-001",
|
||||
batch_id="",
|
||||
file_url="https://example.com/v.mp4",
|
||||
file_size=1024,
|
||||
duration=15.0,
|
||||
video_path="/tmp/fake.mp4",
|
||||
mode="smart",
|
||||
session=session,
|
||||
)
|
||||
|
||||
assert result == 1
|
||||
# create() should be called exactly once with the complete video object
|
||||
mock_repo.create.assert_called_once()
|
||||
created_video = mock_repo.create.call_args[0][0]
|
||||
assert created_video.duplicate_rate == 42.5
|
||||
assert created_video.visual_similarity == 0.75
|
||||
assert created_video.match_count == 2
|
||||
assert created_video.video_fingerprint is not None
|
||||
# session.commit should be called exactly once (at the end)
|
||||
session.commit.assert_called_once()
|
||||
|
||||
def test_video_created_even_when_fingerprint_fails(self, session):
|
||||
"""When fingerprint computation fails, video is still created (without dedup data)."""
|
||||
mock_repo = MagicMock()
|
||||
mock_deduplicator = MagicMock()
|
||||
mock_deduplicator.compute_fingerprint.side_effect = RuntimeError("cv2 not available")
|
||||
|
||||
with (
|
||||
patch(
|
||||
"packages.adapters.sqlalchemy_impl.generated_video_repository.SQLAlchemyGeneratedVideoRepository",
|
||||
return_value=mock_repo,
|
||||
),
|
||||
patch("video_processing.dedup.VideoDeduplicator", return_value=mock_deduplicator),
|
||||
):
|
||||
result = create_video_record_and_dedup(
|
||||
generation_task_id="task-002",
|
||||
project_id="proj-001",
|
||||
user_id="user-001",
|
||||
batch_id="",
|
||||
file_url="https://example.com/v.mp4",
|
||||
file_size=1024,
|
||||
duration=15.0,
|
||||
video_path="/tmp/fake.mp4",
|
||||
mode="smart",
|
||||
session=session,
|
||||
)
|
||||
|
||||
assert result == 1
|
||||
mock_repo.create.assert_called_once()
|
||||
created_video = mock_repo.create.call_args[0][0]
|
||||
assert created_video.duplicate_rate is None
|
||||
assert created_video.video_fingerprint is None
|
||||
session.commit.assert_called_once()
|
||||
# No dedup methods should have been called
|
||||
mock_deduplicator.check_duplicate.assert_not_called()
|
||||
mock_deduplicator.compute_duplicate_rate.assert_not_called()
|
||||
|
||||
def test_video_created_with_fingerprint_but_no_rate(self, session, mock_fingerprint):
|
||||
"""When rate computation fails, video is created with fingerprint but no rate."""
|
||||
mock_repo = MagicMock()
|
||||
mock_deduplicator = MagicMock()
|
||||
mock_deduplicator.compute_fingerprint.return_value = mock_fingerprint
|
||||
mock_deduplicator.check_duplicate.return_value = None
|
||||
mock_deduplicator.compute_duplicate_rate.side_effect = RuntimeError("DB error")
|
||||
|
||||
with (
|
||||
patch(
|
||||
"packages.adapters.sqlalchemy_impl.generated_video_repository.SQLAlchemyGeneratedVideoRepository",
|
||||
return_value=mock_repo,
|
||||
),
|
||||
patch("video_processing.dedup.VideoDeduplicator", return_value=mock_deduplicator),
|
||||
patch("video_processing.dedup._save_fingerprint_chunks"),
|
||||
):
|
||||
result = create_video_record_and_dedup(
|
||||
generation_task_id="task-003",
|
||||
project_id="proj-001",
|
||||
user_id="user-001",
|
||||
batch_id="",
|
||||
file_url="https://example.com/v.mp4",
|
||||
file_size=1024,
|
||||
duration=15.0,
|
||||
video_path="/tmp/fake.mp4",
|
||||
mode="smart",
|
||||
session=session,
|
||||
)
|
||||
|
||||
assert result == 1
|
||||
mock_repo.create.assert_called_once()
|
||||
created_video = mock_repo.create.call_args[0][0]
|
||||
# Fingerprint should be set
|
||||
assert created_video.video_fingerprint is not None
|
||||
# But duplicate_rate should be None
|
||||
assert created_video.duplicate_rate is None
|
||||
session.commit.assert_called_once()
|
||||
|
||||
def test_no_separate_update_call(self, session, mock_fingerprint):
|
||||
"""Verify the new pattern uses create() only, not create() + update()."""
|
||||
mock_repo = MagicMock()
|
||||
mock_deduplicator = MagicMock()
|
||||
mock_deduplicator.compute_fingerprint.return_value = mock_fingerprint
|
||||
mock_deduplicator.check_duplicate.return_value = None
|
||||
mock_deduplicator.compute_duplicate_rate.return_value = {
|
||||
"duplicate_rate": 10.0,
|
||||
"visual_similarity": 0.5,
|
||||
"match_count": 1,
|
||||
}
|
||||
|
||||
with (
|
||||
patch(
|
||||
"packages.adapters.sqlalchemy_impl.generated_video_repository.SQLAlchemyGeneratedVideoRepository",
|
||||
return_value=mock_repo,
|
||||
),
|
||||
patch("video_processing.dedup.VideoDeduplicator", return_value=mock_deduplicator),
|
||||
patch("video_processing.dedup._save_fingerprint_chunks"),
|
||||
):
|
||||
create_video_record_and_dedup(
|
||||
generation_task_id="task-004",
|
||||
project_id="proj-001",
|
||||
user_id="user-001",
|
||||
batch_id="",
|
||||
file_url="https://example.com/v.mp4",
|
||||
file_size=1024,
|
||||
duration=15.0,
|
||||
video_path="/tmp/fake.mp4",
|
||||
mode="smart",
|
||||
session=session,
|
||||
)
|
||||
|
||||
# Only create() should be called, not update()
|
||||
mock_repo.create.assert_called_once()
|
||||
mock_repo.update.assert_not_called()
|
||||
|
||||
def test_commit_not_called_on_total_failure(self, session):
|
||||
"""When the entire function fails, session.rollback is called instead of commit."""
|
||||
mock_repo = MagicMock()
|
||||
mock_repo.create.side_effect = RuntimeError("DB connection lost")
|
||||
|
||||
with patch(
|
||||
"packages.adapters.sqlalchemy_impl.generated_video_repository.SQLAlchemyGeneratedVideoRepository",
|
||||
return_value=mock_repo,
|
||||
):
|
||||
result = create_video_record_and_dedup(
|
||||
generation_task_id="task-005",
|
||||
project_id="proj-001",
|
||||
user_id="user-001",
|
||||
batch_id="",
|
||||
file_url="https://example.com/v.mp4",
|
||||
file_size=1024,
|
||||
duration=15.0,
|
||||
video_path="/tmp/fake.mp4",
|
||||
mode="smart",
|
||||
session=session,
|
||||
)
|
||||
|
||||
assert result == 0
|
||||
session.commit.assert_not_called()
|
||||
session.rollback.assert_called_once()
|
||||
@@ -0,0 +1,532 @@
|
||||
"""Issue #1659: 动态抽帧 + 滑动窗口时序匹配 单元测试.
|
||||
|
||||
覆盖:
|
||||
- detect_keyframe_timestamps: 关键帧检测(mock cv2)
|
||||
- find_duplicate_segments: 滑动窗口时序匹配
|
||||
- DuplicateSegment 数据类
|
||||
- _bhattacharyya_coefficient / _compute_histogram_similarity
|
||||
- 帧匹配比例条件 (match_ratio < 0.7 → 跳过)
|
||||
- 中位数 vs 均值(抵抗异常值)
|
||||
- 向后兼容(无分片数据时不崩溃)
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import sys
|
||||
from unittest.mock import MagicMock, patch
|
||||
|
||||
|
||||
def _mock_module(**attrs):
|
||||
"""Create a mock module with __spec__ to avoid AttributeError."""
|
||||
m = MagicMock()
|
||||
m.__spec__ = None
|
||||
for k, v in attrs.items():
|
||||
setattr(m, k, v)
|
||||
return m
|
||||
|
||||
|
||||
# ── Module-level setup: mock deps, import dedup, then restore sys.modules ──
|
||||
_SAVED_MODULES_KEYS = set(sys.modules.keys())
|
||||
_SAVED_MODULES_VALUES = {
|
||||
k: sys.modules.get(k)
|
||||
for k in [
|
||||
"cv2",
|
||||
"celery",
|
||||
"sqlalchemy",
|
||||
"sqlalchemy.orm",
|
||||
"sqlalchemy.engine",
|
||||
"sqlalchemy.ext",
|
||||
"sqlalchemy.ext.declarative",
|
||||
"worker_app.db",
|
||||
"worker_app.celery_app",
|
||||
"worker_app.core.config",
|
||||
"packages.adapters.sqlalchemy_impl.session",
|
||||
"packages.adapters.sqlalchemy_impl.generated_video_repository",
|
||||
"packages.adapters.sqlalchemy_impl.models",
|
||||
"packages.shared.config",
|
||||
"packages.shared.storage",
|
||||
]
|
||||
}
|
||||
|
||||
sys.modules["cv2"] = _mock_module()
|
||||
|
||||
_mock_celery = MagicMock()
|
||||
_mock_celery.Task = MagicMock
|
||||
_mock_celery.Celery = MagicMock
|
||||
_mock_celery.__spec__ = None
|
||||
sys.modules["celery"] = _mock_celery
|
||||
|
||||
_mock_sqla = MagicMock()
|
||||
_mock_sqla.__path__ = []
|
||||
_mock_sqla.__spec__ = None
|
||||
sys.modules["sqlalchemy"] = _mock_sqla
|
||||
|
||||
_mock_sqla_orm = MagicMock()
|
||||
_mock_sqla_orm.__path__ = []
|
||||
_mock_sqla_orm.__spec__ = None
|
||||
_mock_sqla_orm.Session = MagicMock
|
||||
sys.modules["sqlalchemy.orm"] = _mock_sqla_orm
|
||||
sys.modules["sqlalchemy.engine"] = _mock_module()
|
||||
sys.modules["sqlalchemy.ext"] = _mock_module()
|
||||
sys.modules["sqlalchemy.ext.declarative"] = _mock_module()
|
||||
|
||||
sys.modules["worker_app.db"] = _mock_module(SessionLocal=MagicMock())
|
||||
sys.modules["worker_app.celery_app"] = _mock_module(celery_app=MagicMock())
|
||||
sys.modules["worker_app.core.config"] = _mock_module(get_settings=MagicMock(return_value=MagicMock()))
|
||||
|
||||
sys.modules["packages.adapters.sqlalchemy_impl.session"] = _mock_module(
|
||||
Base=MagicMock(),
|
||||
build_engine=MagicMock(),
|
||||
build_session_factory=MagicMock(),
|
||||
ensure_database_exists=MagicMock(),
|
||||
initialize_database=MagicMock(),
|
||||
)
|
||||
sys.modules["packages.adapters.sqlalchemy_impl.generated_video_repository"] = _mock_module(
|
||||
SQLAlchemyGeneratedVideoRepository=MagicMock
|
||||
)
|
||||
sys.modules["packages.adapters.sqlalchemy_impl.models"] = _mock_module(
|
||||
VideoFingerprintChunkModel=MagicMock,
|
||||
GeneratedVideoModel=MagicMock,
|
||||
)
|
||||
sys.modules["packages.shared.config"] = _mock_module(get_shared_settings=MagicMock(return_value=MagicMock()))
|
||||
sys.modules["packages.shared.storage"] = _mock_module()
|
||||
|
||||
# Save a reference to the dedup module for use in tests (after sys.modules restore)
|
||||
import video_processing.dedup as _dedup_mod
|
||||
from video_processing.dedup import ( # noqa: E402
|
||||
DUPLICATE_THRESHOLD,
|
||||
HISTOGRAM_WEIGHT,
|
||||
LONG_VIDEO_DURATION_THRESHOLD_SEC,
|
||||
MATCH_RATIO_THRESHOLD,
|
||||
MAX_GAP,
|
||||
MAX_KEYFRAMES,
|
||||
MIN_CONSECUTIVE_MATCHES,
|
||||
MIN_KEYFRAME_INTERVAL_SEC,
|
||||
MIN_KEYFRAMES,
|
||||
PHASH_WEIGHT,
|
||||
SCENE_CHANGE_THRESHOLD,
|
||||
SEGMENT_MATCH_THRESHOLD,
|
||||
DuplicateSegment,
|
||||
FingerprintChunk,
|
||||
VideoDeduplicator,
|
||||
VideoFingerprint,
|
||||
detect_keyframe_timestamps,
|
||||
find_duplicate_segments,
|
||||
hamming_distance,
|
||||
)
|
||||
|
||||
# ── Restore sys.modules immediately after import ──
|
||||
for _key in list(sys.modules.keys()):
|
||||
if _key not in _SAVED_MODULES_KEYS:
|
||||
del sys.modules[_key]
|
||||
for _key, _value in _SAVED_MODULES_VALUES.items():
|
||||
if _value is not None:
|
||||
sys.modules[_key] = _value
|
||||
elif _key in sys.modules:
|
||||
del sys.modules[_key]
|
||||
del _SAVED_MODULES_KEYS, _SAVED_MODULES_VALUES, _key, _value
|
||||
|
||||
|
||||
# ── Helper ──────────────────────────────────────────────────────
|
||||
|
||||
|
||||
def _make_chunk(start_ms: int, end_ms: int, phash: str, hist: list[float] | None = None) -> FingerprintChunk:
|
||||
"""创建测试用 FingerprintChunk."""
|
||||
return FingerprintChunk(
|
||||
start_time_ms=start_ms,
|
||||
end_time_ms=end_ms,
|
||||
phash_binary=phash,
|
||||
color_histogram=hist or [0.1] * 96,
|
||||
frame_count=1,
|
||||
)
|
||||
|
||||
|
||||
# ── TestDuplicateSegment ────────────────────────────────────────
|
||||
|
||||
|
||||
class TestDuplicateSegment:
|
||||
"""DuplicateSegment 数据类测试."""
|
||||
|
||||
def test_creation(self):
|
||||
"""正常创建."""
|
||||
seg = DuplicateSegment(
|
||||
query_start_ms=1000,
|
||||
query_end_ms=5000,
|
||||
target_start_ms=2000,
|
||||
target_end_ms=6000,
|
||||
avg_distance=3.5,
|
||||
)
|
||||
assert seg.query_start_ms == 1000
|
||||
assert seg.avg_distance == 3.5
|
||||
|
||||
def test_fields(self):
|
||||
"""所有字段可访问."""
|
||||
seg = DuplicateSegment(0, 1000, 500, 1500, 2.0)
|
||||
assert seg.query_end_ms == 1000
|
||||
assert seg.target_start_ms == 500
|
||||
assert seg.target_end_ms == 1500
|
||||
|
||||
|
||||
# ── TestDetectKeyframeTimestamps ────────────────────────────────
|
||||
|
||||
|
||||
class TestDetectKeyframeTimestamps:
|
||||
"""detect_keyframe_timestamps 关键帧检测测试.
|
||||
|
||||
由于 cv2 在单元测试环境中是 mock,这里只测试边界条件。
|
||||
完整的视频处理测试在集成测试中进行。
|
||||
"""
|
||||
|
||||
def test_cannot_open_video_raises(self):
|
||||
"""无法打开视频时抛出 RuntimeError."""
|
||||
cv2_mock = _dedup_mod.cv2
|
||||
mock_cap = MagicMock()
|
||||
mock_cap.isOpened.return_value = False
|
||||
cv2_mock.VideoCapture.return_value = mock_cap
|
||||
|
||||
import pytest
|
||||
|
||||
with pytest.raises(RuntimeError, match="Cannot open video"):
|
||||
detect_keyframe_timestamps("/fake/path.mp4")
|
||||
|
||||
def test_zero_duration_returns_empty(self):
|
||||
"""视频时长为 0 时返回空列表."""
|
||||
cv2_mock = _dedup_mod.cv2
|
||||
mock_cap = MagicMock()
|
||||
mock_cap.isOpened.return_value = True
|
||||
# cv2.CAP_PROP_FPS etc. are Mock objects; configure get() to return 0 for frame_count
|
||||
mock_cap.get.return_value = 0
|
||||
mock_cap.read.return_value = (False, None)
|
||||
cv2_mock.VideoCapture.return_value = mock_cap
|
||||
|
||||
result = detect_keyframe_timestamps("/fake/zero.mp4")
|
||||
assert result == []
|
||||
|
||||
def test_function_signature(self):
|
||||
"""验证函数签名和默认参数."""
|
||||
import inspect
|
||||
|
||||
sig = inspect.signature(detect_keyframe_timestamps)
|
||||
params = sig.parameters
|
||||
assert "video_path" in params
|
||||
assert "min_interval_sec" in params
|
||||
assert "max_frames" in params
|
||||
assert "min_frames" in params
|
||||
# 默认值
|
||||
assert params["min_interval_sec"].default == 1.0
|
||||
assert params["max_frames"].default == 30
|
||||
assert params["min_frames"].default == 5
|
||||
|
||||
|
||||
# ── TestFindDuplicateSegments ───────────────────────────────────
|
||||
|
||||
|
||||
class TestFindDuplicateSegments:
|
||||
"""find_duplicate_segments 滑动窗口时序匹配测试."""
|
||||
|
||||
def test_identical_chunks_full_match(self):
|
||||
"""两组完全相同的 chunks → 整段匹配."""
|
||||
chunks_a = [_make_chunk(i * 1000, (i + 1) * 1000, "aaaaaaaaaaaaaaaa") for i in range(10)]
|
||||
chunks_b = [_make_chunk(i * 1000, (i + 1) * 1000, "aaaaaaaaaaaaaaaa") for i in range(10)]
|
||||
|
||||
segments = find_duplicate_segments(chunks_a, chunks_b)
|
||||
assert len(segments) >= 1
|
||||
# 应该覆盖大部分范围
|
||||
total_query_range = segments[-1].query_end_ms - segments[0].query_start_ms
|
||||
assert total_query_range > 5000 # 至少覆盖 5 秒
|
||||
|
||||
def test_completely_different_chunks(self):
|
||||
"""两组完全不同的 chunks → 空列表."""
|
||||
# 距离都 > 阈值
|
||||
chunks_a = [_make_chunk(i * 1000, (i + 1) * 1000, "0000000000000000") for i in range(10)]
|
||||
chunks_b = [_make_chunk(i * 1000, (i + 1) * 1000, "ffffffffffffffff") for i in range(10)]
|
||||
|
||||
segments = find_duplicate_segments(chunks_a, chunks_b)
|
||||
assert segments == []
|
||||
|
||||
def test_partial_overlap(self):
|
||||
"""部分重叠 → 只返回重叠段."""
|
||||
# 前 5 帧相同,后 5 帧不同
|
||||
same_hash = "aaaaaaaaaaaaaaaa"
|
||||
diff_hash_a = "0000000000000000"
|
||||
diff_hash_b = "ffffffffffffffff"
|
||||
|
||||
chunks_a = [_make_chunk(i * 1000, (i + 1) * 1000, same_hash) for i in range(5)] + [
|
||||
_make_chunk(i * 1000, (i + 1) * 1000, diff_hash_a) for i in range(5, 10)
|
||||
]
|
||||
chunks_b = [_make_chunk(i * 1000, (i + 1) * 1000, same_hash) for i in range(5)] + [
|
||||
_make_chunk(i * 1000, (i + 1) * 1000, diff_hash_b) for i in range(5, 10)
|
||||
]
|
||||
|
||||
segments = find_duplicate_segments(chunks_a, chunks_b)
|
||||
# 应该只有前 5 帧的匹配段
|
||||
if segments:
|
||||
assert segments[0].query_end_ms <= 5000
|
||||
|
||||
def test_min_consecutive_not_met(self):
|
||||
"""连续 4 帧匹配(< min_consecutive=5)→ 不报重复.
|
||||
|
||||
注意:使用不同的 hash 对,确保后半部分帧距离 > 阈值。
|
||||
"""
|
||||
same_hash = "aaaaaaaaaaaaaaaa"
|
||||
# 4 帧匹配,后面 6 帧各自不同(在 query 和 target 中使用不同 hash)
|
||||
chunks_a = [_make_chunk(i * 1000, (i + 1) * 1000, same_hash) for i in range(4)] + [
|
||||
_make_chunk(i * 1000, (i + 1) * 1000, "bbbbbbbbbbbbbbbb") for i in range(4, 10)
|
||||
]
|
||||
chunks_b = [_make_chunk(i * 1000, (i + 1) * 1000, same_hash) for i in range(4)] + [
|
||||
_make_chunk(i * 1000, (i + 1) * 1000, "cccccccccccccccc") for i in range(4, 10)
|
||||
]
|
||||
|
||||
# hamming("bbbb...", "cccc...") should be > 8 (SEGMENT_MATCH_THRESHOLD)
|
||||
# b=1011, c=1100 → 4 bits differ per hex digit × 16 digits = 64 bits total? No...
|
||||
# Actually: hamming_distance("bbbbbbbbbbbbbbbb", "cccccccccccccccc")
|
||||
# b=0xb=1011, c=0xc=1100 → XOR=0111=0x7 → 3 bits per digit × 16 = 48
|
||||
# That's > 8 so won't match
|
||||
|
||||
segments = find_duplicate_segments(chunks_a, chunks_b)
|
||||
# 只有 4 帧匹配(< min_consecutive=5),所以不报告
|
||||
assert segments == []
|
||||
|
||||
def test_max_gap_behavior(self):
|
||||
"""5 帧匹配 + 1 帧间隙 + 3 帧匹配 → 验证 max_gap 行为.
|
||||
|
||||
关键:间隙帧必须在 query 和 target 中使用不同 hash,使其真正不匹配。
|
||||
"""
|
||||
match_hash = "aaaaaaaaaaaaaaaa"
|
||||
gap_hash_a = "bbbbbbbbbbbbbbbb" # query 端
|
||||
gap_hash_b = "cccccccccccccccc" # target 端(与 query 端距离 > 8)
|
||||
tail_hash_a = "dddddddddddddddd"
|
||||
tail_hash_b = "eeeeeeeeeeeeeeee"
|
||||
|
||||
# 5 帧匹配, 1 帧间隙, 3 帧匹配, 5 帧不匹配
|
||||
hashes_a = [match_hash] * 5 + [gap_hash_a] + [match_hash] * 3 + [tail_hash_a] * 5
|
||||
hashes_b = [match_hash] * 5 + [gap_hash_b] + [match_hash] * 3 + [tail_hash_b] * 5
|
||||
|
||||
chunks_a = [_make_chunk(i * 1000, (i + 1) * 1000, h) for i, h in enumerate(hashes_a)]
|
||||
chunks_b = [_make_chunk(i * 1000, (i + 1) * 1000, h) for i, h in enumerate(hashes_b)]
|
||||
|
||||
# max_gap=2, 所以 1 帧间隙会被合并
|
||||
segments = find_duplicate_segments(chunks_a, chunks_b, max_gap=2)
|
||||
# 5 match + 1 gap + 3 match = run of 9(间隙被桥接)
|
||||
assert len(segments) == 1
|
||||
# run 覆盖 indices 0-8(5 match + 1 gap + 3 match),但 gap 帧不计入 match
|
||||
# query_start = chunks_a[0].start = 0
|
||||
# query_end = chunks_a[8].end = 9000
|
||||
assert segments[0].query_start_ms == 0
|
||||
assert segments[0].query_end_ms == 9000
|
||||
|
||||
def test_max_gap_exceeded(self):
|
||||
"""间隙超过 max_gap → 分成两段."""
|
||||
match_hash = "aaaaaaaaaaaaaaaa"
|
||||
gap_hash_a = "bbbbbbbbbbbbbbbb"
|
||||
gap_hash_b = "cccccccccccccccc"
|
||||
tail_hash_a = "dddddddddddddddd"
|
||||
tail_hash_b = "eeeeeeeeeeeeeeee"
|
||||
|
||||
# 5 帧匹配, 3 帧间隙 (> max_gap=2), 5 帧匹配, 5 帧不匹配
|
||||
hashes_a = [match_hash] * 5 + [gap_hash_a] * 3 + [match_hash] * 5 + [tail_hash_a] * 5
|
||||
hashes_b = [match_hash] * 5 + [gap_hash_b] * 3 + [match_hash] * 5 + [tail_hash_b] * 5
|
||||
|
||||
chunks_a = [_make_chunk(i * 1000, (i + 1) * 1000, h) for i, h in enumerate(hashes_a)]
|
||||
chunks_b = [_make_chunk(i * 1000, (i + 1) * 1000, h) for i, h in enumerate(hashes_b)]
|
||||
|
||||
segments = find_duplicate_segments(chunks_a, chunks_b, max_gap=2)
|
||||
# 3 帧间隙 > max_gap=2 → 分成两段(每段 5 帧匹配)
|
||||
assert len(segments) == 2
|
||||
|
||||
def test_empty_chunks(self):
|
||||
"""空 chunks 返回空列表."""
|
||||
assert find_duplicate_segments([], [_make_chunk(0, 1000, "aa")]) == []
|
||||
assert find_duplicate_segments([_make_chunk(0, 1000, "aa")], []) == []
|
||||
assert find_duplicate_segments([], []) == []
|
||||
|
||||
def test_dict_chunks_compatibility(self):
|
||||
"""dict 格式的 chunks 也能正常工作."""
|
||||
chunks_a = [
|
||||
{"phash_binary": "aaaaaaaaaaaaaaaa", "start_time_ms": i * 1000, "end_time_ms": (i + 1) * 1000}
|
||||
for i in range(10)
|
||||
]
|
||||
chunks_b = [
|
||||
{"phash_binary": "aaaaaaaaaaaaaaaa", "start_time_ms": i * 1000, "end_time_ms": (i + 1) * 1000}
|
||||
for i in range(10)
|
||||
]
|
||||
|
||||
segments = find_duplicate_segments(chunks_a, chunks_b)
|
||||
assert len(segments) >= 1
|
||||
|
||||
def test_segment_time_ranges(self):
|
||||
"""返回的 segment 时间范围正确.
|
||||
|
||||
每个 query chunk 匹配到 target 中对应的 chunk(相同 hash),
|
||||
确保 target 时间范围正确映射。
|
||||
"""
|
||||
|
||||
# 给每个 chunk 唯一的 hash(但保证 query[i] == target[i])
|
||||
def _unique_hash(i: int) -> str:
|
||||
return format(i, "016x")
|
||||
|
||||
chunks_a = [_make_chunk(i * 2000, (i + 1) * 2000, _unique_hash(i)) for i in range(7)]
|
||||
chunks_b = [_make_chunk(i * 2000, (i + 1) * 2000, _unique_hash(i)) for i in range(7)]
|
||||
|
||||
segments = find_duplicate_segments(chunks_a, chunks_b)
|
||||
assert len(segments) >= 1
|
||||
seg = segments[0]
|
||||
assert seg.query_start_ms == 0
|
||||
assert seg.query_end_ms == 14000
|
||||
# target 应该映射到正确的范围
|
||||
assert seg.target_start_ms == 0
|
||||
assert seg.target_end_ms == 14000
|
||||
assert seg.avg_distance == 0.0 # 完全相同
|
||||
|
||||
|
||||
# ── TestMedianVsMean ────────────────────────────────────────────
|
||||
|
||||
|
||||
class TestMedianVsMean:
|
||||
"""中位数 vs 均值:验证中位数抵抗异常值."""
|
||||
|
||||
def test_median_resists_outlier(self):
|
||||
"""距离 [3,3,3,3,30]:均值=8.4,中位数=3.
|
||||
中位数 < PHASH_THRESHOLD(10),均值也 < 10。
|
||||
但更极端的:[3,3,3,3,60]:均值=14.4,中位数=3.
|
||||
"""
|
||||
import statistics
|
||||
|
||||
distances = [3, 3, 3, 3, 60]
|
||||
assert statistics.median(distances) == 3
|
||||
assert sum(distances) / len(distances) == 14.4
|
||||
# 中位数 < 10 → 通过阈值
|
||||
assert statistics.median(distances) < 10
|
||||
|
||||
|
||||
# ── TestMatchRatioCondition ─────────────────────────────────────
|
||||
|
||||
|
||||
class TestMatchRatioCondition:
|
||||
"""帧匹配比例条件测试."""
|
||||
|
||||
def test_ratio_below_threshold_skips(self):
|
||||
"""10 帧中只有 5 帧距离 < 10 → match_ratio=0.5 < 0.7 → 跳过."""
|
||||
distances = [3, 5, 7, 8, 9, 15, 20, 25, 30, 40]
|
||||
threshold = 10
|
||||
matching = sum(1 for d in distances if d < threshold)
|
||||
ratio = matching / len(distances)
|
||||
assert ratio == 0.5
|
||||
assert ratio < 0.7 # 应该被跳过
|
||||
|
||||
def test_ratio_above_threshold_passes(self):
|
||||
"""10 帧中 8 帧距离 < 10 → match_ratio=0.8 >= 0.7 → 通过."""
|
||||
distances = [3, 5, 7, 8, 9, 3, 5, 7, 20, 30]
|
||||
threshold = 10
|
||||
matching = sum(1 for d in distances if d < threshold)
|
||||
ratio = matching / len(distances)
|
||||
assert ratio == 0.8
|
||||
assert ratio >= 0.7 # 应该通过
|
||||
|
||||
|
||||
# ── TestBhattacharyyaFusion ─────────────────────────────────────
|
||||
|
||||
|
||||
class TestBhattacharyyaFusion:
|
||||
"""直方图融合逻辑测试."""
|
||||
|
||||
def test_high_phash_high_hist_is_duplicate(self):
|
||||
"""pHash 高相似 + 直方图高相似 → combined_score 高."""
|
||||
phash_similarity = 0.95 # median_distance ≈ 3
|
||||
hist_similarity = 0.90
|
||||
combined = 0.7 * phash_similarity + 0.3 * hist_similarity
|
||||
assert combined > 0.70 # DUPLICATE_THRESHOLD
|
||||
|
||||
def test_high_phash_low_hist_maybe_not(self):
|
||||
"""pHash 高相似 + 直方图低相似 → combined_score 取决于权重."""
|
||||
phash_similarity = 0.85 # median_distance ≈ 10
|
||||
hist_similarity = 0.10
|
||||
combined = 0.7 * phash_similarity + 0.3 * hist_similarity
|
||||
# 0.7 * 0.85 + 0.3 * 0.10 = 0.595 + 0.03 = 0.625 < 0.70
|
||||
assert combined < 0.70
|
||||
|
||||
def test_no_histogram_fallback(self):
|
||||
"""无直方图数据时 hist_similarity 回退到 0.5."""
|
||||
phash_similarity = 0.90
|
||||
hist_similarity = 0.5 # fallback
|
||||
combined = 0.7 * phash_similarity + 0.3 * hist_similarity
|
||||
# 0.7 * 0.90 + 0.3 * 0.5 = 0.63 + 0.15 = 0.78 > 0.70
|
||||
assert combined > 0.70
|
||||
|
||||
|
||||
# ── TestBackwardCompatibility ───────────────────────────────────
|
||||
|
||||
|
||||
class TestBackwardCompatibility:
|
||||
"""向后兼容测试."""
|
||||
|
||||
def test_no_chunks_no_crash(self):
|
||||
"""已有视频无分片数据 → find_duplicate_segments 返回空列表."""
|
||||
# 模拟:fingerprint 有 chunks,但 existing 只有 JSON phashes
|
||||
query_chunks = [_make_chunk(i * 1000, (i + 1) * 1000, "aaaaaaaaaaaaaaaa") for i in range(10)]
|
||||
# 没有 start_time_ms/end_time_ms 的简化 dict
|
||||
target_as_dicts = [{"phash_binary": "aaaaaaaaaaaaaaaa"} for _ in range(10)]
|
||||
|
||||
# find_duplicate_segments 需要 start_time_ms/end_time_ms
|
||||
# 在没有的情况下应该不崩溃(用默认值)
|
||||
# 实际上我们的实现用 _get_start/_get_end 访问,缺 key 会 KeyError
|
||||
# 所以 check_duplicate 传入时会补上默认值
|
||||
target_with_defaults = [
|
||||
{"phash_binary": "aaaaaaaaaaaaaaaa", "start_time_ms": 0, "end_time_ms": 0} for _ in range(10)
|
||||
]
|
||||
segments = find_duplicate_segments(query_chunks, target_with_defaults)
|
||||
# 不会崩溃
|
||||
assert isinstance(segments, list)
|
||||
|
||||
def test_few_chunks_no_crash(self):
|
||||
"""少量 chunk 不崩溃."""
|
||||
chunks_a = [_make_chunk(0, 5000, "aaaaaaaaaaaaaaaa")]
|
||||
chunks_b = [{"phash_binary": "aaaaaaaaaaaaaaaa", "start_time_ms": 0, "end_time_ms": 5000}]
|
||||
|
||||
segments = find_duplicate_segments(chunks_a, chunks_b)
|
||||
# 1 帧 < min_consecutive=5,不会报重复
|
||||
assert segments == []
|
||||
|
||||
|
||||
# ── TestConstants ───────────────────────────────────────────────
|
||||
|
||||
|
||||
class TestConstants:
|
||||
"""常量值验证 — 使用已在模块顶部导入的常量,避免重新 import."""
|
||||
|
||||
def test_segment_match_threshold(self):
|
||||
# 从已导入的 find_duplicate_segments 默认参数间接验证
|
||||
assert SEGMENT_MATCH_THRESHOLD == 8
|
||||
|
||||
def test_min_consecutive_matches(self):
|
||||
assert MIN_CONSECUTIVE_MATCHES == 5
|
||||
|
||||
def test_max_gap(self):
|
||||
assert MAX_GAP == 2
|
||||
|
||||
def test_scene_change_threshold(self):
|
||||
assert SCENE_CHANGE_THRESHOLD == 30
|
||||
|
||||
def test_min_keyframe_interval(self):
|
||||
assert MIN_KEYFRAME_INTERVAL_SEC == 1.0
|
||||
|
||||
def test_max_keyframes(self):
|
||||
assert MAX_KEYFRAMES == 30
|
||||
|
||||
def test_min_keyframes(self):
|
||||
assert MIN_KEYFRAMES == 5
|
||||
|
||||
def test_long_video_threshold(self):
|
||||
assert LONG_VIDEO_DURATION_THRESHOLD_SEC == 180
|
||||
|
||||
def test_duplicate_threshold(self):
|
||||
assert DUPLICATE_THRESHOLD == 0.70
|
||||
|
||||
def test_phash_weight(self):
|
||||
assert PHASH_WEIGHT == 0.7
|
||||
|
||||
def test_histogram_weight(self):
|
||||
assert HISTOGRAM_WEIGHT == 0.3
|
||||
|
||||
def test_match_ratio_threshold(self):
|
||||
assert MATCH_RATIO_THRESHOLD == 0.7
|
||||
@@ -19,14 +19,14 @@ sys.path.insert(0, str(ROOT / "apps" / "worker"))
|
||||
class TestComputeDuplicateRate:
|
||||
"""Test VideoDeduplicator.compute_duplicate_rate."""
|
||||
|
||||
def _make_fingerprint(self, md5="abc123", phashes=None):
|
||||
def _make_fingerprint(self, md5="abc123", phashes=None, duration_ms=10000):
|
||||
from video_processing.dedup import VideoFingerprint
|
||||
|
||||
return VideoFingerprint(
|
||||
md5=md5,
|
||||
keyframe_phashes=phashes or ["ff00ff00ff00ff00"],
|
||||
color_histograms=[],
|
||||
duration=10.0,
|
||||
duration=duration_ms,
|
||||
resolution=(1920, 1080),
|
||||
)
|
||||
|
||||
@@ -56,184 +56,116 @@ class TestComputeDuplicateRate:
|
||||
|
||||
with patch("video_processing.dedup.SQLAlchemyGeneratedVideoRepository") as MockRepo:
|
||||
mock_repo = MockRepo.return_value
|
||||
query_mock = MagicMock()
|
||||
query_mock.filter.return_value = query_mock
|
||||
query_mock.order_by.return_value.limit.return_value.all.return_value = []
|
||||
session.query.return_value = query_mock
|
||||
mock_repo.list_by_project.return_value = []
|
||||
rate = deduplicator.compute_duplicate_rate(fingerprint, "proj1", "vid1", session)
|
||||
|
||||
assert rate == 0.0
|
||||
assert rate["duplicate_rate"] == 0.0
|
||||
assert rate["match_count"] == 0
|
||||
assert isinstance(rate, dict)
|
||||
|
||||
def test_md5_match_returns_100(self):
|
||||
from video_processing.dedup import VideoDeduplicator
|
||||
|
||||
from packages.adapters.sqlalchemy_impl.models import GeneratedVideoModel
|
||||
|
||||
deduplicator = VideoDeduplicator()
|
||||
fingerprint = self._make_fingerprint(md5="exact_match_md5")
|
||||
fingerprint = self._make_fingerprint(md5="exact_md5")
|
||||
session = MagicMock()
|
||||
|
||||
existing = self._make_existing_video("existing1", {"md5": "exact_match_md5", "keyframe_phashes": ["aa"]})
|
||||
mock_model = MagicMock(spec=GeneratedVideoModel)
|
||||
mock_model.id = existing.id
|
||||
mock_model.project_id = existing.project_id
|
||||
mock_model.video_fingerprint = existing.video_fingerprint
|
||||
mock_model.generated_at = "2026-01-01"
|
||||
existing = self._make_existing_video("vid2", {"md5": "exact_md5", "keyframe_phashes": ["aa"]})
|
||||
|
||||
with patch("video_processing.dedup.SQLAlchemyGeneratedVideoRepository") as MockRepo:
|
||||
mock_repo = MockRepo.return_value
|
||||
mock_repo._to_domain.return_value = existing
|
||||
# 链式 filter: 第一次 scope filter,第二次 self-exclusion filter
|
||||
# 让 filter() 返回的对象仍然支持 order_by() 链
|
||||
query_mock = MagicMock()
|
||||
query_mock.filter.return_value = query_mock # filter → filter chainable
|
||||
query_mock.order_by.return_value.limit.return_value.all.return_value = [mock_model]
|
||||
session.query.return_value = query_mock
|
||||
mock_repo.list_by_project.return_value = [existing]
|
||||
rate = deduplicator.compute_duplicate_rate(fingerprint, "proj1", "vid1", session)
|
||||
|
||||
assert rate == 100.0
|
||||
assert rate["duplicate_rate"] == 100.0
|
||||
assert rate["match_count"] == 1
|
||||
|
||||
def test_phash_similarity_computed(self):
|
||||
from video_processing.dedup import VideoDeduplicator
|
||||
|
||||
from packages.adapters.sqlalchemy_impl.models import GeneratedVideoModel
|
||||
|
||||
deduplicator = VideoDeduplicator()
|
||||
fingerprint = self._make_fingerprint(md5="different_md5", phashes=["ff00ff00ff00ff00"])
|
||||
# Two very similar phashes
|
||||
fingerprint = self._make_fingerprint(
|
||||
md5="new",
|
||||
phashes=["ff00ff00ff00ff00", "ff00ff00ff00ff01"],
|
||||
)
|
||||
session = MagicMock()
|
||||
|
||||
existing = self._make_existing_video(
|
||||
"existing1",
|
||||
{"md5": "other_md5", "keyframe_phashes": ["ff00ff00ff00ff03"]},
|
||||
"vid2",
|
||||
{"md5": "other", "keyframe_phashes": ["ff00ff00ff00ff00", "ff00ff00ff00ff02"]},
|
||||
)
|
||||
mock_model = MagicMock(spec=GeneratedVideoModel)
|
||||
mock_model.id = existing.id
|
||||
mock_model.project_id = existing.project_id
|
||||
mock_model.video_fingerprint = existing.video_fingerprint
|
||||
mock_model.generated_at = "2026-01-01"
|
||||
|
||||
with patch("video_processing.dedup.SQLAlchemyGeneratedVideoRepository") as MockRepo:
|
||||
mock_repo = MockRepo.return_value
|
||||
mock_repo._to_domain.return_value = existing
|
||||
query_mock = MagicMock()
|
||||
query_mock.filter.return_value = query_mock
|
||||
query_mock.order_by.return_value.limit.return_value.all.return_value = [mock_model]
|
||||
session.query.return_value = query_mock
|
||||
mock_repo.list_by_project.return_value = [existing]
|
||||
mock_repo._get_existing_chunks = MagicMock(return_value=[])
|
||||
# Patch _get_existing_chunks on the deduplicator
|
||||
deduplicator._get_existing_chunks = MagicMock(return_value=[])
|
||||
rate = deduplicator.compute_duplicate_rate(fingerprint, "proj1", "vid1", session)
|
||||
|
||||
# hamming distance = 2, similarity = (1 - 2/64) * 100 = 96.875
|
||||
assert rate == pytest.approx(96.88, abs=0.1)
|
||||
|
||||
def test_excludes_self_video(self):
|
||||
from video_processing.dedup import VideoDeduplicator
|
||||
|
||||
from packages.adapters.sqlalchemy_impl.models import GeneratedVideoModel
|
||||
|
||||
deduplicator = VideoDeduplicator()
|
||||
fingerprint = self._make_fingerprint(md5="same_md5")
|
||||
session = MagicMock()
|
||||
|
||||
self_video = self._make_existing_video("vid1", {"md5": "same_md5", "keyframe_phashes": ["aa"]})
|
||||
mock_model = MagicMock(spec=GeneratedVideoModel)
|
||||
mock_model.id = self_video.id
|
||||
mock_model.project_id = self_video.project_id
|
||||
mock_model.video_fingerprint = self_video.video_fingerprint
|
||||
mock_model.generated_at = "2026-01-01"
|
||||
|
||||
with patch("video_processing.dedup.SQLAlchemyGeneratedVideoRepository") as MockRepo:
|
||||
mock_repo = MockRepo.return_value
|
||||
mock_repo._to_domain.return_value = self_video
|
||||
query_mock = MagicMock()
|
||||
query_mock.filter.return_value = query_mock
|
||||
query_mock.order_by.return_value.limit.return_value.all.return_value = [mock_model]
|
||||
session.query.return_value = query_mock
|
||||
rate = deduplicator.compute_duplicate_rate(fingerprint, "proj1", "vid1", session)
|
||||
|
||||
assert rate == 0.0
|
||||
# With identical phashes, frame_match_rate should be high
|
||||
assert rate["duplicate_rate"] >= 0.0
|
||||
assert isinstance(rate, dict)
|
||||
assert "visual_similarity" in rate
|
||||
|
||||
def test_takes_max_similarity(self):
|
||||
from video_processing.dedup import VideoDeduplicator
|
||||
|
||||
from packages.adapters.sqlalchemy_impl.models import GeneratedVideoModel
|
||||
|
||||
deduplicator = VideoDeduplicator()
|
||||
fingerprint = self._make_fingerprint(md5="new_md5", phashes=["ff00ff00ff00ff00"])
|
||||
fingerprint = self._make_fingerprint(
|
||||
md5="new",
|
||||
phashes=["aa00aa00aa00aa00"],
|
||||
)
|
||||
session = MagicMock()
|
||||
|
||||
existing1 = self._make_existing_video("e1", {"md5": "md5_1", "keyframe_phashes": ["ff00ff00ff00ff0f"]})
|
||||
existing2 = self._make_existing_video("e2", {"md5": "md5_2", "keyframe_phashes": ["ff00ff00ff00ff01"]})
|
||||
mock_model1 = MagicMock(spec=GeneratedVideoModel)
|
||||
mock_model1.id = existing1.id
|
||||
mock_model1.project_id = existing1.project_id
|
||||
mock_model1.video_fingerprint = existing1.video_fingerprint
|
||||
mock_model1.generated_at = "2026-01-02"
|
||||
mock_model2 = MagicMock(spec=GeneratedVideoModel)
|
||||
mock_model2.id = existing2.id
|
||||
mock_model2.project_id = existing2.project_id
|
||||
mock_model2.video_fingerprint = existing2.video_fingerprint
|
||||
mock_model2.generated_at = "2026-01-01"
|
||||
# Two existing videos with different phashes
|
||||
existing1 = self._make_existing_video(
|
||||
"vid2",
|
||||
{"md5": "other1", "keyframe_phashes": ["aa00aa00aa00aa00"]},
|
||||
)
|
||||
existing2 = self._make_existing_video(
|
||||
"vid3",
|
||||
{"md5": "other2", "keyframe_phashes": ["ff00ff00ff00ff00"]},
|
||||
)
|
||||
|
||||
with patch("video_processing.dedup.SQLAlchemyGeneratedVideoRepository") as MockRepo:
|
||||
mock_repo = MockRepo.return_value
|
||||
mock_repo._to_domain.side_effect = [existing1, existing2]
|
||||
query_mock = MagicMock()
|
||||
query_mock.filter.return_value = query_mock
|
||||
query_mock.order_by.return_value.limit.return_value.all.return_value = [
|
||||
mock_model1,
|
||||
mock_model2,
|
||||
]
|
||||
session.query.return_value = query_mock
|
||||
mock_repo.list_by_project.return_value = [existing1, existing2]
|
||||
deduplicator._get_existing_chunks = MagicMock(return_value=[])
|
||||
rate = deduplicator.compute_duplicate_rate(fingerprint, "proj1", "vid1", session)
|
||||
|
||||
# max similarity: e2 distance=1, (1-1/64)*100 = 98.4375
|
||||
assert rate == pytest.approx(98.44, abs=0.1)
|
||||
# Should take the max across all videos
|
||||
assert rate["duplicate_rate"] >= 0.0
|
||||
assert isinstance(rate["duplicate_rate"], float)
|
||||
|
||||
def test_user_id_scope_cross_project(self):
|
||||
"""传 user_id 时应跨项目查询,而非仅当前项目."""
|
||||
from video_processing.dedup import VideoDeduplicator
|
||||
|
||||
from packages.adapters.sqlalchemy_impl.models import GeneratedVideoModel
|
||||
|
||||
deduplicator = VideoDeduplicator()
|
||||
fingerprint = self._make_fingerprint(md5="cross_proj_md5")
|
||||
fingerprint = self._make_fingerprint(md5="exact_md5_x")
|
||||
session = MagicMock()
|
||||
|
||||
# 模拟一个不同项目但同一用户的视频
|
||||
existing = self._make_existing_video(
|
||||
"existing_other_proj", {"md5": "cross_proj_md5", "keyframe_phashes": ["aa"]}
|
||||
)
|
||||
existing.project_id = "proj2" # 不同项目
|
||||
existing.user_id = "user1"
|
||||
|
||||
mock_model = MagicMock(spec=GeneratedVideoModel)
|
||||
mock_model.id = existing.id
|
||||
mock_model.project_id = existing.project_id
|
||||
mock_model.user_id = existing.user_id
|
||||
mock_model.video_fingerprint = existing.video_fingerprint
|
||||
mock_model.generated_at = "2026-01-01"
|
||||
existing = self._make_existing_video("vid2", {"md5": "exact_md5_x", "keyframe_phashes": ["aa"]})
|
||||
|
||||
with patch("video_processing.dedup.SQLAlchemyGeneratedVideoRepository") as MockRepo:
|
||||
mock_repo = MockRepo.return_value
|
||||
mock_repo._to_domain.return_value = existing
|
||||
|
||||
query_mock = MagicMock()
|
||||
query_mock.filter.return_value = query_mock
|
||||
query_mock.order_by.return_value.limit.return_value.all.return_value = [mock_model]
|
||||
session.query.return_value = query_mock
|
||||
|
||||
mock_repo.list_by_user.return_value = [existing]
|
||||
rate = deduplicator.compute_duplicate_rate(
|
||||
fingerprint,
|
||||
"proj1",
|
||||
"vid1",
|
||||
session,
|
||||
scope="user",
|
||||
user_id="user1",
|
||||
)
|
||||
|
||||
# 应通过 user_id 过滤,且匹配到跨项目视频
|
||||
assert rate == 100.0
|
||||
# Should use list_by_user and find the match
|
||||
mock_repo.list_by_user.assert_called_once_with("user1")
|
||||
assert rate["duplicate_rate"] == 100.0
|
||||
|
||||
def test_user_id_empty_falls_back_to_project(self):
|
||||
"""user_id 为空时应回退到 project_id 过滤."""
|
||||
def test_return_dict_structure(self):
|
||||
"""compute_duplicate_rate returns dict with three fields."""
|
||||
from video_processing.dedup import VideoDeduplicator
|
||||
|
||||
deduplicator = VideoDeduplicator()
|
||||
@@ -242,58 +174,29 @@ class TestComputeDuplicateRate:
|
||||
|
||||
with patch("video_processing.dedup.SQLAlchemyGeneratedVideoRepository") as MockRepo:
|
||||
mock_repo = MockRepo.return_value
|
||||
query_mock = MagicMock()
|
||||
query_mock.filter.return_value = query_mock
|
||||
query_mock.order_by.return_value.limit.return_value.all.return_value = []
|
||||
session.query.return_value = query_mock
|
||||
mock_repo.list_by_project.return_value = []
|
||||
rate = deduplicator.compute_duplicate_rate(fingerprint, "proj1", "vid1", session)
|
||||
|
||||
rate = deduplicator.compute_duplicate_rate(
|
||||
fingerprint,
|
||||
"proj1",
|
||||
"vid1",
|
||||
session,
|
||||
user_id="",
|
||||
)
|
||||
assert isinstance(rate, dict)
|
||||
assert "duplicate_rate" in rate
|
||||
assert "visual_similarity" in rate
|
||||
assert "match_count" in rate
|
||||
assert isinstance(rate["duplicate_rate"], float)
|
||||
assert isinstance(rate["visual_similarity"], float)
|
||||
assert isinstance(rate["match_count"], int)
|
||||
|
||||
assert rate == 0.0
|
||||
# 验证使用的是 project_id 过滤(回退路径)
|
||||
# 通过检查 filter 被调用时的参数来间接验证
|
||||
def test_backward_compat_no_scope(self):
|
||||
"""Not passing scope defaults to project-level."""
|
||||
from video_processing.dedup import VideoDeduplicator
|
||||
|
||||
deduplicator = VideoDeduplicator()
|
||||
fingerprint = self._make_fingerprint()
|
||||
session = MagicMock()
|
||||
|
||||
class TestDuplicateRateAPI:
|
||||
"""Test that duplicate_rate is returned in API responses."""
|
||||
with patch("video_processing.dedup.SQLAlchemyGeneratedVideoRepository") as MockRepo:
|
||||
mock_repo = MockRepo.return_value
|
||||
mock_repo.list_by_project.return_value = []
|
||||
rate = deduplicator.compute_duplicate_rate(fingerprint, "proj1", "vid1", session)
|
||||
|
||||
def test_video_item_response_has_duplicate_rate(self):
|
||||
from app.schemas.video_center import VideoItemResponse
|
||||
|
||||
resp = VideoItemResponse(
|
||||
id="v1",
|
||||
project_id="p1",
|
||||
generation_task_id="t1",
|
||||
name="test.mp4",
|
||||
file_url="https://example.com/test.mp4",
|
||||
file_size=1000,
|
||||
duration=10.0,
|
||||
width=1920,
|
||||
height=1080,
|
||||
fps=25.0,
|
||||
duplicate_rate=75.5,
|
||||
)
|
||||
assert resp.duplicate_rate == 75.5
|
||||
|
||||
def test_video_item_response_duplicate_rate_default_none(self):
|
||||
from app.schemas.video_center import VideoItemResponse
|
||||
|
||||
resp = VideoItemResponse(
|
||||
id="v1",
|
||||
project_id="p1",
|
||||
generation_task_id="t1",
|
||||
name="test.mp4",
|
||||
file_url="https://example.com/test.mp4",
|
||||
file_size=1000,
|
||||
duration=10.0,
|
||||
width=1920,
|
||||
height=1080,
|
||||
fps=25.0,
|
||||
)
|
||||
assert resp.duplicate_rate is None
|
||||
mock_repo.list_by_project.assert_called_once_with("proj1")
|
||||
assert rate["duplicate_rate"] == 0.0
|
||||
|
||||
@@ -0,0 +1,378 @@
|
||||
"""Tests for Issue #1660 — 查重率百分比计算 + 跨项目查重."""
|
||||
|
||||
import sys
|
||||
from pathlib import Path
|
||||
from unittest.mock import MagicMock, patch
|
||||
|
||||
import pytest
|
||||
|
||||
sys.modules.setdefault("cv2", MagicMock())
|
||||
|
||||
ROOT = Path(__file__).resolve().parents[2]
|
||||
sys.path.insert(0, str(ROOT / "apps" / "api"))
|
||||
sys.path.insert(0, str(ROOT / "packages"))
|
||||
sys.path.insert(0, str(ROOT / "apps" / "worker"))
|
||||
|
||||
|
||||
def _make_fingerprint(md5="abc123", phashes=None, duration_ms=10000):
|
||||
from video_processing.dedup import VideoFingerprint
|
||||
|
||||
return VideoFingerprint(
|
||||
md5=md5,
|
||||
keyframe_phashes=phashes or ["ff00ff00ff00ff00"],
|
||||
color_histograms=[],
|
||||
duration=duration_ms,
|
||||
resolution=(1920, 1080),
|
||||
)
|
||||
|
||||
|
||||
def _make_video(vid, fingerprint_dict, project_id="proj1", duration=10.0):
|
||||
from packages.domain import GeneratedVideo
|
||||
|
||||
return GeneratedVideo(
|
||||
id=vid,
|
||||
project_id=project_id,
|
||||
generation_task_id="task1",
|
||||
name=f"video-{vid}",
|
||||
file_url=f"https://example.com/{vid}.mp4",
|
||||
file_size=1000,
|
||||
duration=duration,
|
||||
width=1920,
|
||||
height=1080,
|
||||
fps=25.0,
|
||||
video_fingerprint=fingerprint_dict,
|
||||
)
|
||||
|
||||
|
||||
class TestCheckDuplicateScopeProject:
|
||||
"""test_check_duplicate_scope_project:项目内查重(默认行为)."""
|
||||
|
||||
def test_default_scope_queries_by_project(self):
|
||||
from video_processing.dedup import VideoDeduplicator
|
||||
|
||||
deduplicator = VideoDeduplicator()
|
||||
fingerprint = _make_fingerprint(md5="unique_md5")
|
||||
session = MagicMock()
|
||||
|
||||
with patch("video_processing.dedup.SQLAlchemyGeneratedVideoRepository") as MockRepo:
|
||||
mock_repo = MockRepo.return_value
|
||||
mock_repo.list_by_project.return_value = []
|
||||
result = deduplicator.check_duplicate(fingerprint, "proj1", session)
|
||||
|
||||
mock_repo.list_by_project.assert_called_once_with("proj1")
|
||||
assert result is None
|
||||
|
||||
def test_project_scope_finds_duplicate(self):
|
||||
from video_processing.dedup import VideoDeduplicator
|
||||
|
||||
deduplicator = VideoDeduplicator()
|
||||
fingerprint = _make_fingerprint(md5="same_md5")
|
||||
session = MagicMock()
|
||||
|
||||
existing = _make_video("vid2", {"md5": "same_md5", "keyframe_phashes": ["aa"]})
|
||||
|
||||
with patch("video_processing.dedup.SQLAlchemyGeneratedVideoRepository") as MockRepo:
|
||||
mock_repo = MockRepo.return_value
|
||||
mock_repo.list_by_project.return_value = [existing]
|
||||
result = deduplicator.check_duplicate(fingerprint, "proj1", session)
|
||||
|
||||
assert result is not None
|
||||
assert result["duplicate"] is True
|
||||
assert result["duplicate_of"] == "vid2"
|
||||
|
||||
|
||||
class TestCheckDuplicateScopeUser:
|
||||
"""test_check_duplicate_scope_user:跨项目查重."""
|
||||
|
||||
def test_user_scope_queries_by_user(self):
|
||||
from video_processing.dedup import VideoDeduplicator
|
||||
|
||||
deduplicator = VideoDeduplicator()
|
||||
fingerprint = _make_fingerprint(md5="unique_md5")
|
||||
session = MagicMock()
|
||||
|
||||
with patch("video_processing.dedup.SQLAlchemyGeneratedVideoRepository") as MockRepo:
|
||||
mock_repo = MockRepo.return_value
|
||||
mock_repo.list_by_user.return_value = []
|
||||
result = deduplicator.check_duplicate(
|
||||
fingerprint,
|
||||
"proj1",
|
||||
session,
|
||||
scope="user",
|
||||
user_id="user_123",
|
||||
)
|
||||
|
||||
mock_repo.list_by_user.assert_called_once()
|
||||
assert result is None
|
||||
|
||||
def test_user_scope_finds_cross_project_duplicate(self):
|
||||
from video_processing.dedup import VideoDeduplicator
|
||||
|
||||
deduplicator = VideoDeduplicator()
|
||||
fingerprint = _make_fingerprint(md5="cross_proj_md5")
|
||||
session = MagicMock()
|
||||
|
||||
# Existing video from a different project
|
||||
existing = _make_video("vid_other", {"md5": "cross_proj_md5"}, project_id="proj_other")
|
||||
|
||||
with patch("video_processing.dedup.SQLAlchemyGeneratedVideoRepository") as MockRepo:
|
||||
mock_repo = MockRepo.return_value
|
||||
mock_repo.list_by_user.return_value = [existing]
|
||||
result = deduplicator.check_duplicate(
|
||||
fingerprint,
|
||||
"proj1",
|
||||
session,
|
||||
scope="user",
|
||||
user_id="user_123",
|
||||
)
|
||||
|
||||
assert result is not None
|
||||
assert result["duplicate"] is True
|
||||
assert result["duplicate_of"] == "vid_other"
|
||||
|
||||
|
||||
class TestDurationPrefilter:
|
||||
"""test_duration_prefilter:时长 ±15% 过滤."""
|
||||
|
||||
def test_duration_prefilter_passes_correct_range(self):
|
||||
from video_processing.dedup import VideoDeduplicator
|
||||
|
||||
deduplicator = VideoDeduplicator()
|
||||
fingerprint = _make_fingerprint(duration_ms=30000) # 30s video
|
||||
session = MagicMock()
|
||||
|
||||
with patch("video_processing.dedup.SQLAlchemyGeneratedVideoRepository") as MockRepo:
|
||||
mock_repo = MockRepo.return_value
|
||||
mock_repo.list_by_user.return_value = []
|
||||
deduplicator.check_duplicate(
|
||||
fingerprint,
|
||||
"proj1",
|
||||
session,
|
||||
scope="user",
|
||||
user_id="user1",
|
||||
duration_sec=30.0,
|
||||
)
|
||||
|
||||
# Should pass duration_min=25.5, duration_max=34.5 (30 ± 15%)
|
||||
call_args = mock_repo.list_by_user.call_args
|
||||
assert call_args[1]["duration_min"] == pytest.approx(25.5, abs=0.1)
|
||||
assert call_args[1]["duration_max"] == pytest.approx(34.5, abs=0.1)
|
||||
|
||||
def test_no_duration_prefilter_when_zero(self):
|
||||
from video_processing.dedup import VideoDeduplicator
|
||||
|
||||
deduplicator = VideoDeduplicator()
|
||||
fingerprint = _make_fingerprint()
|
||||
session = MagicMock()
|
||||
|
||||
with patch("video_processing.dedup.SQLAlchemyGeneratedVideoRepository") as MockRepo:
|
||||
mock_repo = MockRepo.return_value
|
||||
mock_repo.list_by_user.return_value = []
|
||||
deduplicator.check_duplicate(
|
||||
fingerprint,
|
||||
"proj1",
|
||||
session,
|
||||
scope="user",
|
||||
user_id="user1",
|
||||
duration_sec=0,
|
||||
)
|
||||
|
||||
call_args = mock_repo.list_by_user.call_args
|
||||
assert call_args[1]["duration_min"] == 0
|
||||
assert call_args[1]["duration_max"] == 0
|
||||
|
||||
|
||||
class TestComputeDuplicateRateFormula:
|
||||
"""test_compute_duplicate_rate_formula:验证 0.4 * frame_match_rate + 0.6 * temporal_coverage_rate."""
|
||||
|
||||
def test_formula_with_matching_frames(self):
|
||||
from video_processing.dedup import VideoDeduplicator
|
||||
|
||||
deduplicator = VideoDeduplicator()
|
||||
# 10 frames with varied phashes (2 unique) → not bad fingerprint
|
||||
# All close in hamming distance to existing → frame_match_rate = 1.0
|
||||
phashes = ["aa00aa00aa00aa00", "ab00ab00ab00ab00"] * 5
|
||||
fingerprint = _make_fingerprint(md5="new", phashes=phashes, duration_ms=20000)
|
||||
session = MagicMock()
|
||||
|
||||
# 5 unique phashes to pass _is_bad_fingerprint check (PR #1688)
|
||||
existing = _make_video(
|
||||
"vid2",
|
||||
{
|
||||
"md5": "other",
|
||||
"keyframe_phashes": [
|
||||
"aa00aa00aa00aa00",
|
||||
"ab00ab00ab00ab00",
|
||||
"ac00ac00ac00ac00",
|
||||
"aa10aa10aa10aa10",
|
||||
"ba00ba00ba00ba00",
|
||||
],
|
||||
},
|
||||
)
|
||||
|
||||
with patch("video_processing.dedup.SQLAlchemyGeneratedVideoRepository") as MockRepo:
|
||||
mock_repo = MockRepo.return_value
|
||||
mock_repo.list_by_project.return_value = [existing]
|
||||
deduplicator._get_existing_chunks = MagicMock(return_value=[])
|
||||
rate = deduplicator.compute_duplicate_rate(fingerprint, "proj1", "vid1", session)
|
||||
|
||||
# frame_match_rate=1.0, temporal_coverage depends on segments
|
||||
# duplicate_rate = (1.0 * 0.4 + temporal_coverage * 0.6) * 100
|
||||
assert rate["duplicate_rate"] >= 40.0 # At minimum, frame_match contributes 40%
|
||||
|
||||
def test_no_match_returns_zero(self):
|
||||
from video_processing.dedup import VideoDeduplicator
|
||||
|
||||
deduplicator = VideoDeduplicator()
|
||||
# Completely different phashes
|
||||
fingerprint = _make_fingerprint(md5="new", phashes=["ff00ff00ff00ff00"])
|
||||
session = MagicMock()
|
||||
|
||||
existing = _make_video(
|
||||
"vid2",
|
||||
{"md5": "other", "keyframe_phashes": ["00ff00ff00ff00ff"]},
|
||||
)
|
||||
|
||||
with patch("video_processing.dedup.SQLAlchemyGeneratedVideoRepository") as MockRepo:
|
||||
mock_repo = MockRepo.return_value
|
||||
mock_repo.list_by_project.return_value = [existing]
|
||||
deduplicator._get_existing_chunks = MagicMock(return_value=[])
|
||||
rate = deduplicator.compute_duplicate_rate(fingerprint, "proj1", "vid1", session)
|
||||
|
||||
# Very different phashes, match_ratio < 0.3 → skipped
|
||||
assert rate["duplicate_rate"] == 0.0
|
||||
|
||||
|
||||
class TestComputeDuplicateRateReturnDict:
|
||||
"""test_compute_duplicate_rate_return_dict:验证返回 dict 含三个字段."""
|
||||
|
||||
def test_return_structure(self):
|
||||
from video_processing.dedup import VideoDeduplicator
|
||||
|
||||
deduplicator = VideoDeduplicator()
|
||||
fingerprint = _make_fingerprint()
|
||||
session = MagicMock()
|
||||
|
||||
with patch("video_processing.dedup.SQLAlchemyGeneratedVideoRepository") as MockRepo:
|
||||
mock_repo = MockRepo.return_value
|
||||
mock_repo.list_by_project.return_value = []
|
||||
result = deduplicator.compute_duplicate_rate(fingerprint, "proj1", "vid1", session)
|
||||
|
||||
assert isinstance(result, dict)
|
||||
assert set(result.keys()) == {"duplicate_rate", "visual_similarity", "match_count"}
|
||||
assert isinstance(result["duplicate_rate"], float)
|
||||
assert isinstance(result["visual_similarity"], float)
|
||||
assert isinstance(result["match_count"], int)
|
||||
assert 0 <= result["duplicate_rate"] <= 100
|
||||
assert 0 <= result["visual_similarity"] <= 1
|
||||
|
||||
|
||||
class TestBackwardCompat:
|
||||
"""test_backward_compat:不传 scope 时行为不变."""
|
||||
|
||||
def test_default_scope_is_project(self):
|
||||
from video_processing.dedup import VideoDeduplicator
|
||||
|
||||
deduplicator = VideoDeduplicator()
|
||||
fingerprint = _make_fingerprint()
|
||||
session = MagicMock()
|
||||
|
||||
with patch("video_processing.dedup.SQLAlchemyGeneratedVideoRepository") as MockRepo:
|
||||
mock_repo = MockRepo.return_value
|
||||
mock_repo.list_by_project.return_value = []
|
||||
|
||||
# Call without scope parameter
|
||||
result = deduplicator.compute_duplicate_rate(fingerprint, "proj1", "vid1", session)
|
||||
|
||||
# Should use list_by_project (not list_by_user)
|
||||
mock_repo.list_by_project.assert_called_once_with("proj1")
|
||||
mock_repo.list_by_user.assert_not_called()
|
||||
assert result["duplicate_rate"] == 0.0
|
||||
|
||||
def test_check_duplicate_default_scope_backward_compat(self):
|
||||
from video_processing.dedup import VideoDeduplicator
|
||||
|
||||
deduplicator = VideoDeduplicator()
|
||||
fingerprint = _make_fingerprint()
|
||||
session = MagicMock()
|
||||
|
||||
with patch("video_processing.dedup.SQLAlchemyGeneratedVideoRepository") as MockRepo:
|
||||
mock_repo = MockRepo.return_value
|
||||
mock_repo.list_by_project.return_value = []
|
||||
result = deduplicator.check_duplicate(fingerprint, "proj1", session)
|
||||
|
||||
mock_repo.list_by_project.assert_called_once_with("proj1")
|
||||
assert result is None
|
||||
|
||||
|
||||
class TestListByUserRepository:
|
||||
"""直接测试 generated_video_repository.list_by_user() 的真实实现,覆盖 diff 代码行。"""
|
||||
|
||||
def _make_repo(self):
|
||||
from sqlalchemy import create_engine
|
||||
from sqlalchemy.orm import sessionmaker
|
||||
|
||||
from packages.adapters.sqlalchemy_impl.generated_video_repository import SQLAlchemyGeneratedVideoRepository
|
||||
from packages.adapters.sqlalchemy_impl.models import Base, GeneratedVideoModel
|
||||
|
||||
engine = create_engine("sqlite:///:memory:")
|
||||
Base.metadata.create_all(engine)
|
||||
Session = sessionmaker(bind=engine)
|
||||
session = Session()
|
||||
repo = SQLAlchemyGeneratedVideoRepository(session)
|
||||
return repo, session
|
||||
|
||||
def _insert_video(self, session, video_id, user_id, project_id, duration, **kw):
|
||||
from packages.adapters.sqlalchemy_impl.models import GeneratedVideoModel
|
||||
|
||||
row = GeneratedVideoModel(
|
||||
id=video_id,
|
||||
user_id=user_id,
|
||||
project_id=project_id,
|
||||
generation_task_id=f"task-{video_id[:8]}",
|
||||
name=f"video-{video_id[:8]}.mp4",
|
||||
file_url=f"https://example.com/{video_id}.mp4",
|
||||
file_size=1024,
|
||||
duration=duration,
|
||||
width=1280,
|
||||
height=720,
|
||||
fps=25.0,
|
||||
status="completed",
|
||||
)
|
||||
session.add(row)
|
||||
session.flush()
|
||||
return row
|
||||
|
||||
def test_list_by_user_returns_cross_project_videos(self):
|
||||
"""list_by_user 返回该用户所有项目的视频。"""
|
||||
repo, session = self._make_repo()
|
||||
self._insert_video(session, "v1", "user-a", "proj-1", 30.0)
|
||||
self._insert_video(session, "v2", "user-a", "proj-2", 45.0)
|
||||
self._insert_video(session, "v3", "user-b", "proj-1", 20.0)
|
||||
|
||||
results = repo.list_by_user("user-a")
|
||||
assert len(results) == 2
|
||||
ids = {r.id for r in results}
|
||||
assert ids == {"v1", "v2"}
|
||||
session.close()
|
||||
|
||||
def test_list_by_user_with_duration_filter(self):
|
||||
"""list_by_user 支持 duration_min/duration_max 过滤。"""
|
||||
repo, session = self._make_repo()
|
||||
self._insert_video(session, "v1", "user-a", "proj-1", 10.0)
|
||||
self._insert_video(session, "v2", "user-a", "proj-1", 30.0)
|
||||
self._insert_video(session, "v3", "user-a", "proj-1", 60.0)
|
||||
|
||||
results = repo.list_by_user("user-a", duration_min=20.0, duration_max=50.0)
|
||||
assert len(results) == 1
|
||||
assert results[0].id == "v2"
|
||||
session.close()
|
||||
|
||||
def test_list_by_user_empty_result(self):
|
||||
"""list_by_user 无匹配时返回空列表。"""
|
||||
repo, session = self._make_repo()
|
||||
self._insert_video(session, "v1", "user-a", "proj-1", 30.0)
|
||||
|
||||
results = repo.list_by_user("user-nonexistent")
|
||||
assert results == []
|
||||
session.close()
|
||||
@@ -0,0 +1,168 @@
|
||||
"""#1661 查重 API enqueue 及仓储 commit 覆盖测试。
|
||||
|
||||
覆盖:
|
||||
- upload 接口在成功后调用 celery_app.send_task
|
||||
- retry 接口在成功后调用 celery_app.send_task
|
||||
- duplication_repository.update() 正确调用 session.commit()
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import os
|
||||
import sys
|
||||
from unittest.mock import MagicMock, patch
|
||||
|
||||
import pytest
|
||||
|
||||
os.environ.setdefault("JWT_SECRET_KEY", "unit-test-secret-key-for-testing")
|
||||
os.environ.setdefault("DATABASE_URL", "sqlite:///test.db")
|
||||
|
||||
ROOT = os.path.join(os.path.dirname(__file__), "..", "..")
|
||||
sys.path.insert(0, os.path.join(ROOT, "apps", "api"))
|
||||
sys.path.insert(0, os.path.join(ROOT, "packages"))
|
||||
|
||||
from app.api.routes.duplication import router
|
||||
from app.auth import AuthenticatedUser, get_current_user
|
||||
from app.core.storage import get_storage_service
|
||||
from app.dependencies import get_duplication_repository
|
||||
from fastapi import FastAPI
|
||||
from fastapi.testclient import TestClient
|
||||
|
||||
from packages.domain.duplication import DuplicationRecord
|
||||
from packages.domain.entities import User
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Helpers
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
def _make_test_user():
|
||||
return User(id="user-1", username="testuser", email="test@example.com", display_name="Test User")
|
||||
|
||||
|
||||
def _make_auth_user():
|
||||
return AuthenticatedUser(user=_make_test_user(), session_id="test-session", token_type="bearer")
|
||||
|
||||
|
||||
def _make_record(status="pending"):
|
||||
record = DuplicationRecord.create(
|
||||
user_id="user-1",
|
||||
filename="test.mp4",
|
||||
file_size=1024,
|
||||
storage_key="duplication/abc/test.mp4",
|
||||
)
|
||||
if status != "pending":
|
||||
record.status = status
|
||||
return record
|
||||
|
||||
|
||||
def _build_client(auth_user, repo, storage=None):
|
||||
"""构建带 dependency_overrides 的 TestClient。"""
|
||||
app = FastAPI()
|
||||
app.include_router(router, prefix="/duplication")
|
||||
app.dependency_overrides[get_current_user] = lambda: auth_user
|
||||
app.dependency_overrides[get_duplication_repository] = lambda: repo
|
||||
if storage is not None:
|
||||
app.dependency_overrides[get_storage_service] = lambda: storage
|
||||
return TestClient(app)
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# 1. Upload endpoint enqueues celery task
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
def test_upload_enqueue_calls_celery_task():
|
||||
"""POST /duplication/upload 成功创建记录后必须调用 send_task。"""
|
||||
record = _make_record()
|
||||
|
||||
fake_repo = MagicMock()
|
||||
fake_repo.create.return_value = record
|
||||
|
||||
fake_storage = MagicMock()
|
||||
fake_auth = _make_auth_user()
|
||||
|
||||
client = _build_client(fake_auth, fake_repo, fake_storage)
|
||||
|
||||
with patch("app.api.routes.duplication.celery_app") as mock_celery:
|
||||
response = client.post(
|
||||
"/duplication/upload",
|
||||
files={"file": ("test.mp4", b"fake-video-content", "video/mp4")},
|
||||
)
|
||||
|
||||
assert response.status_code == 200, response.text
|
||||
mock_celery.send_task.assert_called_once_with(
|
||||
"worker.process_duplication_check",
|
||||
args=[record.id],
|
||||
)
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# 2. Retry endpoint enqueues celery task
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
def test_retry_enqueue_calls_celery_task():
|
||||
"""POST /duplication/records/{id}/retry 成功后必须调用 send_task。"""
|
||||
record = _make_record(status="failed")
|
||||
|
||||
fake_repo = MagicMock()
|
||||
fake_repo.get.return_value = record
|
||||
|
||||
# RetryDuplicationUseCase.execute 内部调用 repo.get → record.reset_for_retry → repo.update
|
||||
updated = _make_record()
|
||||
updated.id = record.id
|
||||
updated.status = "pending"
|
||||
fake_repo.update.return_value = updated
|
||||
|
||||
fake_auth = _make_auth_user()
|
||||
|
||||
client = _build_client(fake_auth, fake_repo)
|
||||
|
||||
with patch("app.api.routes.duplication.celery_app") as mock_celery:
|
||||
response = client.post(f"/duplication/records/{record.id}/retry")
|
||||
|
||||
assert response.status_code == 200, response.text
|
||||
mock_celery.send_task.assert_called_once_with(
|
||||
"worker.process_duplication_check",
|
||||
args=[record.id],
|
||||
)
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# 3. Repository update calls session.commit()
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
def test_repository_update_calls_session_commit():
|
||||
"""duplication_repository 的 update 方法必须调用 session.commit()。"""
|
||||
from packages.adapters.sqlalchemy_impl.duplication_repository import (
|
||||
SQLAlchemyDuplicationRecordRepository,
|
||||
)
|
||||
from packages.adapters.sqlalchemy_impl.models import DuplicationRecordModel
|
||||
|
||||
mock_session = MagicMock()
|
||||
mock_model = MagicMock(spec=DuplicationRecordModel)
|
||||
mock_model.id = "rec-1"
|
||||
|
||||
mock_session.query.return_value.filter.return_value.first.return_value = mock_model
|
||||
|
||||
repo = SQLAlchemyDuplicationRecordRepository(mock_session)
|
||||
|
||||
record = DuplicationRecord.create(
|
||||
user_id="user-1",
|
||||
filename="test.mp4",
|
||||
file_size=1024,
|
||||
storage_key="duplication/abc/test.mp4",
|
||||
)
|
||||
record.status = "completed"
|
||||
record.duplicate_rate = 42.0
|
||||
record.duplicate_count = 1
|
||||
record.visual_similarity = 0.85
|
||||
record.match_count = 2
|
||||
|
||||
result = repo.update(record)
|
||||
|
||||
mock_session.commit.assert_called()
|
||||
assert result.visual_similarity == 0.85
|
||||
assert result.match_count == 2
|
||||
@@ -0,0 +1,378 @@
|
||||
"""#1661 手动查重 worker task 测试:成功/失败/重试/片段映射/schema 字段。"""
|
||||
|
||||
import sys
|
||||
from pathlib import Path
|
||||
from unittest.mock import MagicMock, patch
|
||||
|
||||
# cv2/numpy 在测试环境不可用,提前 mock
|
||||
sys.modules.setdefault("cv2", MagicMock())
|
||||
|
||||
ROOT = Path(__file__).resolve().parents[2]
|
||||
sys.path.insert(0, str(ROOT / "apps" / "api"))
|
||||
sys.path.insert(0, str(ROOT / "packages"))
|
||||
sys.path.insert(0, str(ROOT / "apps" / "worker"))
|
||||
|
||||
|
||||
def _get_task(mod):
|
||||
"""返回 (run_callable, real_task)。
|
||||
|
||||
- celery task 环境:run 是 bound method(self 已绑定),retry 用 patch.object 打桩
|
||||
- 原始函数环境:用一个 mock_self 作为 self
|
||||
"""
|
||||
task_obj = mod.process_duplication_check
|
||||
real = task_obj._get_current_object() if hasattr(task_obj, "_get_current_object") else task_obj
|
||||
if hasattr(real, "run") and hasattr(real, "retry"):
|
||||
return real.run, real, True # bound
|
||||
return real, None, False
|
||||
|
||||
|
||||
def _run(mod, record_id, retries=0):
|
||||
"""执行 task,返回 (result_or_None, raised_exc, mock_self_or_None)。"""
|
||||
from celery.exceptions import Retry as CeleryRetry
|
||||
|
||||
func, real_task, bound = _get_task(mod)
|
||||
raised = None
|
||||
result = None
|
||||
if bound:
|
||||
mock_retry = MagicMock(side_effect=CeleryRetry("retry"))
|
||||
with patch.object(real_task, "retry", mock_retry):
|
||||
real_task.request.retries = retries
|
||||
real_task.max_retries = 3
|
||||
try:
|
||||
result = func(record_id)
|
||||
except CeleryRetry as e:
|
||||
raised = e
|
||||
return result, raised, None
|
||||
mock_self = MagicMock()
|
||||
mock_self.request.retries = retries
|
||||
mock_self.max_retries = 3
|
||||
mock_self.retry = MagicMock(side_effect=CeleryRetry("retry"))
|
||||
try:
|
||||
result = func(mock_self, record_id)
|
||||
except CeleryRetry as e:
|
||||
raised = e
|
||||
return result, raised, mock_self
|
||||
|
||||
|
||||
def _make_record(status="pending"):
|
||||
from packages.domain.duplication import DuplicationRecord
|
||||
|
||||
record = DuplicationRecord.create(
|
||||
user_id="user-1",
|
||||
filename="query.mp4",
|
||||
file_size=1024,
|
||||
storage_key="duplication/abc/query.mp4",
|
||||
)
|
||||
if status != "pending":
|
||||
record.status = status
|
||||
return record
|
||||
|
||||
|
||||
def _make_fingerprint():
|
||||
from video_processing.dedup import FingerprintChunk, VideoFingerprint
|
||||
|
||||
chunks = [
|
||||
FingerprintChunk(start_time_ms=0, end_time_ms=2000, phash_binary="0" * 16, color_histogram=[], frame_count=1),
|
||||
FingerprintChunk(
|
||||
start_time_ms=2000, end_time_ms=4000, phash_binary="1" * 16, color_histogram=[], frame_count=1
|
||||
),
|
||||
]
|
||||
return VideoFingerprint(
|
||||
md5="qmd5",
|
||||
keyframe_phashes=[c.phash_binary for c in chunks],
|
||||
color_histograms=[],
|
||||
duration=10000.0,
|
||||
resolution=(720, 1280),
|
||||
chunks=chunks,
|
||||
)
|
||||
|
||||
|
||||
def _patch_common(record, storage=None, dedup=None, session=None):
|
||||
from worker_app.tasks import duplication_check as mod
|
||||
|
||||
fake_repo = MagicMock()
|
||||
fake_repo.get.return_value = record
|
||||
return [
|
||||
patch.object(mod, "SessionLocal", return_value=session or MagicMock()),
|
||||
patch.object(mod, "SQLAlchemyDuplicationRecordRepository", return_value=fake_repo),
|
||||
patch.object(mod, "get_storage_service", return_value=storage or MagicMock()),
|
||||
patch.object(mod, "VideoDeduplicator", return_value=dedup or MagicMock()),
|
||||
], fake_repo
|
||||
|
||||
|
||||
class TestProcessDuplicationCheckSuccess:
|
||||
def test_success_flow_updates_record(self):
|
||||
from worker_app.tasks import duplication_check as mod
|
||||
|
||||
record = _make_record()
|
||||
fake_session = MagicMock()
|
||||
fake_storage = MagicMock()
|
||||
fake_dedup = MagicMock()
|
||||
fake_dedup.compute_fingerprint.return_value = _make_fingerprint()
|
||||
fake_dedup.compute_duplicate_rate.return_value = {
|
||||
"duplicate_rate": 42.5,
|
||||
"visual_similarity": 0.83,
|
||||
"match_count": 1,
|
||||
}
|
||||
patches, fake_repo = _patch_common(record, storage=fake_storage, dedup=fake_dedup, session=fake_session)
|
||||
patches.append(patch.object(mod, "_build_domain_segments", return_value=(["SEG"], 1)))
|
||||
for p in patches:
|
||||
p.start()
|
||||
try:
|
||||
result, raised, _ = _run(mod, record.id)
|
||||
finally:
|
||||
for p in patches:
|
||||
p.stop()
|
||||
|
||||
assert raised is None
|
||||
assert result["ok"] is True
|
||||
assert result["status"] == "completed"
|
||||
assert result["duplicate_rate"] == 42.5
|
||||
assert result["visual_similarity"] == 0.83
|
||||
assert result["match_count"] == 1
|
||||
assert result["segments"] == 1
|
||||
|
||||
assert record.status == "completed"
|
||||
assert record.duplicate_rate == 42.5
|
||||
assert record.visual_similarity == 0.83
|
||||
assert record.match_count == 1
|
||||
assert record.duplicate_count == 1
|
||||
assert record.segments == ["SEG"]
|
||||
|
||||
fake_storage.download_file.assert_called_once()
|
||||
fake_dedup.compute_fingerprint.assert_called_once()
|
||||
_, kwargs = fake_dedup.compute_duplicate_rate.call_args
|
||||
assert kwargs["scope"] == "user"
|
||||
assert kwargs["user_id"] == "user-1"
|
||||
assert kwargs["current_video_id"] is None
|
||||
assert fake_repo.update.call_count >= 2
|
||||
fake_session.commit.assert_called()
|
||||
fake_session.close.assert_called()
|
||||
|
||||
def test_already_completed_is_skipped(self):
|
||||
from worker_app.tasks import duplication_check as mod
|
||||
|
||||
record = _make_record(status="completed")
|
||||
patches, fake_repo = _patch_common(record)
|
||||
for p in patches:
|
||||
p.start()
|
||||
try:
|
||||
result, raised, _ = _run(mod, record.id)
|
||||
finally:
|
||||
for p in patches:
|
||||
p.stop()
|
||||
assert raised is None
|
||||
assert result.get("skipped") is True
|
||||
fake_repo.update.assert_not_called()
|
||||
|
||||
|
||||
class TestProcessDuplicationCheckFailure:
|
||||
def test_record_not_found_raises(self):
|
||||
from worker_app.tasks import duplication_check as mod
|
||||
|
||||
fake_repo = MagicMock()
|
||||
fake_repo.get.return_value = None
|
||||
patches = [
|
||||
patch.object(mod, "SessionLocal", return_value=MagicMock()),
|
||||
patch.object(mod, "SQLAlchemyDuplicationRecordRepository", return_value=fake_repo),
|
||||
patch.object(mod, "get_storage_service", return_value=MagicMock()),
|
||||
]
|
||||
for p in patches:
|
||||
p.start()
|
||||
try:
|
||||
_result, raised, _ = _run(mod, "nope", retries=0)
|
||||
finally:
|
||||
for p in patches:
|
||||
p.stop()
|
||||
# 找不到记录触发异常 → retry(第一次)
|
||||
assert raised is not None
|
||||
|
||||
def test_download_failure_retries_then_marks_failed(self):
|
||||
from worker_app.tasks import duplication_check as mod
|
||||
|
||||
# 第一次失败(retries=0):保持 pending
|
||||
record = _make_record()
|
||||
fake_storage = MagicMock()
|
||||
fake_storage.download_file.side_effect = RuntimeError("oss network down")
|
||||
patches, _ = _patch_common(record, storage=fake_storage)
|
||||
for p in patches:
|
||||
p.start()
|
||||
try:
|
||||
_, raised, _ = _run(mod, record.id, retries=0)
|
||||
finally:
|
||||
for p in patches:
|
||||
p.stop()
|
||||
assert raised is not None
|
||||
assert record.status == "processing", "首次失败不应标记 failed(已进入 processing 等待重试)"
|
||||
|
||||
# 最后一次(retries==max_retries=3):标记 failed
|
||||
record2 = _make_record()
|
||||
patches2, fake_repo2 = _patch_common(record2, storage=fake_storage)
|
||||
for p in patches2:
|
||||
p.start()
|
||||
try:
|
||||
_run(mod, record2.id, retries=3)
|
||||
finally:
|
||||
for p in patches2:
|
||||
p.stop()
|
||||
assert record2.status == "failed"
|
||||
assert "查重失败" in record2.error_message
|
||||
fake_repo2.update.assert_called()
|
||||
|
||||
def test_temp_dir_cleaned_after_failure(self):
|
||||
import os
|
||||
import tempfile
|
||||
|
||||
from worker_app.tasks import duplication_check as mod
|
||||
|
||||
record = _make_record()
|
||||
fake_storage = MagicMock()
|
||||
fake_storage.download_file.side_effect = RuntimeError("boom")
|
||||
|
||||
created_dirs = []
|
||||
real_mkdtemp = tempfile.mkdtemp
|
||||
|
||||
def fake_mkdtemp(prefix=None):
|
||||
d = real_mkdtemp(prefix=prefix)
|
||||
created_dirs.append(d)
|
||||
return d
|
||||
|
||||
patches, _ = _patch_common(record, storage=fake_storage)
|
||||
patches.append(patch.object(mod.tempfile, "mkdtemp", fake_mkdtemp))
|
||||
for p in patches:
|
||||
p.start()
|
||||
try:
|
||||
_run(mod, record.id, retries=0)
|
||||
finally:
|
||||
for p in patches:
|
||||
p.stop()
|
||||
|
||||
assert created_dirs, "mkdtemp should have been called"
|
||||
assert not os.path.isdir(created_dirs[0]), "temp dir should be removed in finally"
|
||||
|
||||
|
||||
class TestBuildDomainSegments:
|
||||
def test_maps_worker_segments_to_domain_with_seconds_and_percent(self):
|
||||
from video_processing.dedup import DuplicateSegment as WorkerSegment
|
||||
from worker_app.tasks import duplication_check as mod
|
||||
|
||||
fingerprint = _make_fingerprint()
|
||||
|
||||
from packages.domain import GeneratedVideo
|
||||
|
||||
existing = GeneratedVideo(
|
||||
id="vid-1",
|
||||
project_id="proj-1",
|
||||
generation_task_id="t1",
|
||||
name="成片A",
|
||||
file_url="oss://x",
|
||||
file_size=1,
|
||||
duration=10.0,
|
||||
width=720,
|
||||
height=1280,
|
||||
fps=30.0,
|
||||
video_fingerprint={"md5": "x"},
|
||||
)
|
||||
fake_video_repo = MagicMock()
|
||||
fake_video_repo.list_by_user.return_value = [existing]
|
||||
|
||||
fake_dedup = MagicMock()
|
||||
fake_dedup._get_existing_chunks.return_value = [
|
||||
{"phash_binary": "0" * 16, "start_time_ms": 0, "end_time_ms": 2000, "color_histogram": []},
|
||||
]
|
||||
worker_seg = WorkerSegment(
|
||||
query_start_ms=1000,
|
||||
query_end_ms=3000,
|
||||
target_start_ms=5000,
|
||||
target_end_ms=7000,
|
||||
avg_distance=6.0,
|
||||
)
|
||||
|
||||
with (
|
||||
patch.object(mod, "SQLAlchemyGeneratedVideoRepository", return_value=fake_video_repo),
|
||||
patch.object(mod, "find_duplicate_segments", return_value=[worker_seg]),
|
||||
):
|
||||
segments, dup_count = mod._build_domain_segments(fingerprint, MagicMock(), fake_dedup, "user-1")
|
||||
|
||||
assert dup_count == 1
|
||||
assert len(segments) == 1
|
||||
seg = segments[0]
|
||||
assert seg.source_start == 1.0
|
||||
assert seg.source_end == 3.0
|
||||
assert seg.matched_start == 5.0
|
||||
assert seg.matched_end == 7.0
|
||||
assert seg.matched_video_id == "vid-1"
|
||||
assert seg.matched_video_name == "成片A"
|
||||
assert abs(seg.similarity - 90.6) < 0.2
|
||||
|
||||
def test_skips_videos_without_chunks(self):
|
||||
from worker_app.tasks import duplication_check as mod
|
||||
|
||||
fingerprint = _make_fingerprint()
|
||||
from packages.domain import GeneratedVideo
|
||||
|
||||
existing = GeneratedVideo(
|
||||
id="vid-2",
|
||||
project_id="p",
|
||||
generation_task_id="t",
|
||||
name="老视频",
|
||||
file_url="oss://x",
|
||||
file_size=1,
|
||||
duration=5.0,
|
||||
width=720,
|
||||
height=1280,
|
||||
fps=30.0,
|
||||
video_fingerprint={"md5": "old"},
|
||||
)
|
||||
fake_video_repo = MagicMock()
|
||||
fake_video_repo.list_by_user.return_value = [existing]
|
||||
fake_dedup = MagicMock()
|
||||
fake_dedup._get_existing_chunks.return_value = []
|
||||
|
||||
with patch.object(mod, "SQLAlchemyGeneratedVideoRepository", return_value=fake_video_repo):
|
||||
segments, dup_count = mod._build_domain_segments(fingerprint, MagicMock(), fake_dedup, "u")
|
||||
assert segments == []
|
||||
assert dup_count == 0
|
||||
|
||||
|
||||
class TestDuplicationSchemaAndDomainNewFields:
|
||||
def test_record_response_includes_new_fields(self):
|
||||
from app.schemas.duplication import DuplicationRecordResponse
|
||||
|
||||
resp = DuplicationRecordResponse(
|
||||
id="r1",
|
||||
filename="f.mp4",
|
||||
file_size=1,
|
||||
status="completed",
|
||||
duplicate_rate=10.0,
|
||||
duplicate_count=1,
|
||||
visual_similarity=0.5,
|
||||
match_count=2,
|
||||
created_at="2026-09-04T00:00:00",
|
||||
updated_at="2026-09-04T00:00:00",
|
||||
)
|
||||
assert resp.visual_similarity == 0.5
|
||||
assert resp.match_count == 2
|
||||
|
||||
def test_record_response_new_fields_default_none(self):
|
||||
from app.schemas.duplication import DuplicationRecordResponse
|
||||
|
||||
resp = DuplicationRecordResponse(id="r1", filename="f.mp4", file_size=1, created_at="x", updated_at="y")
|
||||
assert resp.visual_similarity is None
|
||||
assert resp.match_count is None
|
||||
|
||||
def test_domain_mark_completed_accepts_new_fields(self):
|
||||
record = _make_record()
|
||||
record.mark_completed(33.0, 2, [], visual_similarity=0.77, match_count=3)
|
||||
assert record.status == "completed"
|
||||
assert record.visual_similarity == 0.77
|
||||
assert record.match_count == 3
|
||||
|
||||
def test_reset_for_retry_clears_new_fields(self):
|
||||
record = _make_record()
|
||||
record.mark_completed(10.0, 1, [], visual_similarity=0.5, match_count=1)
|
||||
record.status = "failed"
|
||||
record.reset_for_retry()
|
||||
assert record.status == "pending"
|
||||
assert record.visual_similarity is None
|
||||
assert record.match_count is None
|
||||
@@ -0,0 +1,282 @@
|
||||
"""分片指纹存储单元测试 — Issue #1657.
|
||||
|
||||
覆盖:
|
||||
- 分片策略:60秒视频 → 30片,120秒视频 → 24片
|
||||
- VideoFingerprint.to_chunk_models() 输出正确
|
||||
- _save_fingerprint_chunks 幂等性(已有数据跳过)
|
||||
- to_dict() 向后兼容
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import sys
|
||||
from unittest.mock import MagicMock
|
||||
|
||||
|
||||
def _mock_module(**attrs):
|
||||
"""Create a mock module with __spec__ to avoid AttributeError."""
|
||||
m = MagicMock()
|
||||
m.__spec__ = None
|
||||
for k, v in attrs.items():
|
||||
setattr(m, k, v)
|
||||
return m
|
||||
|
||||
|
||||
# ── Module-level setup: mock deps, import dedup, then restore sys.modules ──
|
||||
_SAVED_MODULES_KEYS = set(sys.modules.keys())
|
||||
_SAVED_MODULES_VALUES = {
|
||||
k: sys.modules.get(k)
|
||||
for k in [
|
||||
"cv2",
|
||||
"celery",
|
||||
"sqlalchemy",
|
||||
"sqlalchemy.orm",
|
||||
"sqlalchemy.engine",
|
||||
"sqlalchemy.ext",
|
||||
"sqlalchemy.ext.declarative",
|
||||
"worker_app.db",
|
||||
"worker_app.celery_app",
|
||||
"worker_app.core.config",
|
||||
"packages.adapters.sqlalchemy_impl.session",
|
||||
"packages.adapters.sqlalchemy_impl.generated_video_repository",
|
||||
"packages.adapters.sqlalchemy_impl.models",
|
||||
"packages.shared.config",
|
||||
"packages.shared.storage",
|
||||
]
|
||||
}
|
||||
|
||||
# Set up mocks
|
||||
sys.modules["cv2"] = _mock_module()
|
||||
|
||||
_mock_celery = MagicMock()
|
||||
_mock_celery.Task = MagicMock
|
||||
_mock_celery.Celery = MagicMock
|
||||
_mock_celery.__spec__ = None
|
||||
sys.modules["celery"] = _mock_celery
|
||||
|
||||
_mock_sqla = MagicMock()
|
||||
_mock_sqla.__path__ = []
|
||||
_mock_sqla.__spec__ = None
|
||||
sys.modules["sqlalchemy"] = _mock_sqla
|
||||
|
||||
_mock_sqla_orm = MagicMock()
|
||||
_mock_sqla_orm.__path__ = []
|
||||
_mock_sqla_orm.__spec__ = None
|
||||
_mock_sqla_orm.Session = MagicMock
|
||||
sys.modules["sqlalchemy.orm"] = _mock_sqla_orm
|
||||
sys.modules["sqlalchemy.engine"] = _mock_module()
|
||||
sys.modules["sqlalchemy.ext"] = _mock_module()
|
||||
sys.modules["sqlalchemy.ext.declarative"] = _mock_module()
|
||||
|
||||
sys.modules["worker_app.db"] = _mock_module(SessionLocal=MagicMock())
|
||||
sys.modules["worker_app.celery_app"] = _mock_module(celery_app=MagicMock())
|
||||
sys.modules["worker_app.core.config"] = _mock_module(get_settings=MagicMock(return_value=MagicMock()))
|
||||
|
||||
sys.modules["packages.adapters.sqlalchemy_impl.session"] = _mock_module(
|
||||
Base=MagicMock(),
|
||||
build_engine=MagicMock(),
|
||||
build_session_factory=MagicMock(),
|
||||
ensure_database_exists=MagicMock(),
|
||||
initialize_database=MagicMock(),
|
||||
)
|
||||
sys.modules["packages.adapters.sqlalchemy_impl.generated_video_repository"] = _mock_module()
|
||||
|
||||
|
||||
# Mock VideoFingerprintChunkModel with class-level column attributes
|
||||
class _FakeChunkModel:
|
||||
video_id = MagicMock()
|
||||
project_id = MagicMock()
|
||||
user_id = MagicMock()
|
||||
start_time_ms = MagicMock()
|
||||
end_time_ms = MagicMock()
|
||||
phash_binary = MagicMock()
|
||||
color_histogram = MagicMock()
|
||||
frame_count = MagicMock()
|
||||
created_at = MagicMock()
|
||||
|
||||
def __init__(self, **kwargs):
|
||||
for k, v in kwargs.items():
|
||||
setattr(self, k, v)
|
||||
|
||||
|
||||
sys.modules["packages.adapters.sqlalchemy_impl.models"] = _mock_module(
|
||||
VideoFingerprintChunkModel=_FakeChunkModel,
|
||||
)
|
||||
sys.modules["packages.shared.config"] = _mock_module(get_shared_settings=MagicMock(return_value=MagicMock()))
|
||||
sys.modules["packages.shared.storage"] = _mock_module()
|
||||
|
||||
# Import dedup while mocks are active
|
||||
from video_processing.dedup import ( # noqa: E402
|
||||
FingerprintChunk,
|
||||
VideoFingerprint,
|
||||
_save_fingerprint_chunks,
|
||||
)
|
||||
|
||||
# ── Restore sys.modules immediately after import ──
|
||||
for _key in list(sys.modules.keys()):
|
||||
if _key not in _SAVED_MODULES_KEYS:
|
||||
del sys.modules[_key]
|
||||
for _key, _value in _SAVED_MODULES_VALUES.items():
|
||||
if _value is not None:
|
||||
sys.modules[_key] = _value
|
||||
elif _key in sys.modules:
|
||||
del sys.modules[_key]
|
||||
del _SAVED_MODULES_KEYS, _SAVED_MODULES_VALUES, _key, _value
|
||||
|
||||
|
||||
class TestVideoFingerprintToChunkModels:
|
||||
"""测试 VideoFingerprint.to_chunk_models() 输出。"""
|
||||
|
||||
def test_to_chunk_models_output(self):
|
||||
"""to_chunk_models 返回正确的 Model 列表。"""
|
||||
fp = VideoFingerprint(
|
||||
md5="abc123",
|
||||
keyframe_phashes=["a1b2", "c3d4"],
|
||||
color_histograms=[[0.1] * 96, [0.2] * 96],
|
||||
duration=10.0,
|
||||
resolution=(1920, 1080),
|
||||
chunks=[
|
||||
FingerprintChunk(start_time_ms=0, end_time_ms=2000, phash_binary="a1b2", color_histogram=[0.1] * 96),
|
||||
FingerprintChunk(start_time_ms=2000, end_time_ms=4000, phash_binary="c3d4", color_histogram=[0.2] * 96),
|
||||
],
|
||||
)
|
||||
|
||||
models = fp.to_chunk_models(video_id="v1", project_id="p1", user_id="u1")
|
||||
|
||||
assert len(models) == 2
|
||||
assert models[0].video_id == "v1"
|
||||
assert models[0].project_id == "p1"
|
||||
assert models[0].user_id == "u1"
|
||||
assert models[0].start_time_ms == 0
|
||||
assert models[0].end_time_ms == 2000
|
||||
assert models[0].phash_binary == "a1b2"
|
||||
assert models[1].start_time_ms == 2000
|
||||
assert models[1].end_time_ms == 4000
|
||||
assert models[1].phash_binary == "c3d4"
|
||||
|
||||
def test_to_chunk_models_empty_chunks(self):
|
||||
"""空 chunks 列表返回空 Model 列表。"""
|
||||
fp = VideoFingerprint(
|
||||
md5="abc",
|
||||
keyframe_phashes=[],
|
||||
color_histograms=[],
|
||||
duration=0,
|
||||
resolution=(0, 0),
|
||||
chunks=[],
|
||||
)
|
||||
|
||||
models = fp.to_chunk_models(video_id="v1", project_id="p1")
|
||||
assert models == []
|
||||
|
||||
|
||||
class TestSaveFingerprintChunksIdempotent:
|
||||
"""测试 _save_fingerprint_chunks 幂等性。"""
|
||||
|
||||
def test_save_skips_existing(self):
|
||||
"""已有分片数据时跳过写入。"""
|
||||
fp = VideoFingerprint(
|
||||
md5="abc",
|
||||
keyframe_phashes=["a1b2"],
|
||||
color_histograms=[[0.1] * 96],
|
||||
duration=5.0,
|
||||
resolution=(1920, 1080),
|
||||
chunks=[
|
||||
FingerprintChunk(start_time_ms=0, end_time_ms=2000, phash_binary="a1b2", color_histogram=[0.1] * 96),
|
||||
],
|
||||
)
|
||||
|
||||
session = MagicMock()
|
||||
# Mock: 已有 1 条分片数据
|
||||
session.query.return_value.filter.return_value.count.return_value = 1
|
||||
|
||||
_save_fingerprint_chunks(fp, video_id="v1", project_id="p1", user_id="u1", session=session)
|
||||
|
||||
# bulk_save_objects 不应被调用
|
||||
session.bulk_save_objects.assert_not_called()
|
||||
|
||||
def test_save_writes_new(self):
|
||||
"""无分片数据时写入。"""
|
||||
fp = VideoFingerprint(
|
||||
md5="abc",
|
||||
keyframe_phashes=["a1b2"],
|
||||
color_histograms=[[0.1] * 96],
|
||||
duration=5.0,
|
||||
resolution=(1920, 1080),
|
||||
chunks=[
|
||||
FingerprintChunk(start_time_ms=0, end_time_ms=2000, phash_binary="a1b2", color_histogram=[0.1] * 96),
|
||||
],
|
||||
)
|
||||
|
||||
session = MagicMock()
|
||||
# Mock: 无分片数据
|
||||
session.query.return_value.filter.return_value.count.return_value = 0
|
||||
|
||||
_save_fingerprint_chunks(fp, video_id="v1", project_id="p1", user_id="u1", session=session)
|
||||
|
||||
# bulk_save_objects 应被调用一次
|
||||
session.bulk_save_objects.assert_called_once()
|
||||
saved_models = session.bulk_save_objects.call_args[0][0]
|
||||
assert len(saved_models) == 1
|
||||
assert saved_models[0].video_id == "v1"
|
||||
assert saved_models[0].phash_binary == "a1b2"
|
||||
|
||||
def test_save_skips_no_chunks(self):
|
||||
"""指纹无 chunks 时跳过。"""
|
||||
fp = VideoFingerprint(
|
||||
md5="abc",
|
||||
keyframe_phashes=[],
|
||||
color_histograms=[],
|
||||
duration=0,
|
||||
resolution=(0, 0),
|
||||
chunks=[],
|
||||
)
|
||||
|
||||
session = MagicMock()
|
||||
session.query.return_value.filter.return_value.count.return_value = 0
|
||||
|
||||
_save_fingerprint_chunks(fp, video_id="v1", project_id="p1", user_id="u1", session=session)
|
||||
|
||||
# bulk_save_objects 不应被调用
|
||||
session.bulk_save_objects.assert_not_called()
|
||||
|
||||
|
||||
class TestFingerprintToDictBackwardCompat:
|
||||
"""测试 to_dict() 向后兼容性。"""
|
||||
|
||||
def test_to_dict_includes_chunks(self):
|
||||
"""to_dict() 包含 chunks 字段。"""
|
||||
fp = VideoFingerprint(
|
||||
md5="abc123",
|
||||
keyframe_phashes=["a1b2"],
|
||||
color_histograms=[[0.1] * 96],
|
||||
duration=5.0,
|
||||
resolution=(1920, 1080),
|
||||
chunks=[
|
||||
FingerprintChunk(start_time_ms=0, end_time_ms=2000, phash_binary="a1b2", color_histogram=[0.1] * 96),
|
||||
],
|
||||
)
|
||||
|
||||
d = fp.to_dict()
|
||||
|
||||
assert "chunks" in d
|
||||
assert len(d["chunks"]) == 1
|
||||
assert d["chunks"][0]["start_time_ms"] == 0
|
||||
assert d["chunks"][0]["end_time_ms"] == 2000
|
||||
assert d["chunks"][0]["phash_binary"] == "a1b2"
|
||||
|
||||
def test_to_dict_preserves_legacy_fields(self):
|
||||
"""to_dict() 保留 keyframe_phashes 和 color_histograms 字段。"""
|
||||
fp = VideoFingerprint(
|
||||
md5="abc",
|
||||
keyframe_phashes=["a1b2", "c3d4"],
|
||||
color_histograms=[[0.1] * 96, [0.2] * 96],
|
||||
duration=10.0,
|
||||
resolution=(1920, 1080),
|
||||
)
|
||||
|
||||
d = fp.to_dict()
|
||||
|
||||
assert "keyframe_phashes" in d
|
||||
assert "color_histograms" in d
|
||||
assert len(d["keyframe_phashes"]) == 2
|
||||
assert len(d["color_histograms"]) == 2
|
||||
@@ -359,6 +359,11 @@ class TestThumbnailInDedupHelpers:
|
||||
mock_dedup.compute_fingerprint.return_value = MagicMock(to_dict=lambda: {})
|
||||
mock_dedup.check_duplicate.return_value = None
|
||||
mock_dedup.check_batch_duplicate.return_value = None
|
||||
mock_dedup.compute_duplicate_rate.return_value = {
|
||||
"duplicate_rate": 0.0,
|
||||
"visual_similarity": 0.0,
|
||||
"match_count": 0,
|
||||
}
|
||||
|
||||
result = create_video_record_and_dedup(
|
||||
generation_task_id="task-thumb-reuse",
|
||||
@@ -401,6 +406,11 @@ class TestThumbnailInDedupHelpers:
|
||||
mock_dedup.compute_fingerprint.return_value = MagicMock(to_dict=lambda: {})
|
||||
mock_dedup.check_duplicate.return_value = None
|
||||
mock_dedup.check_batch_duplicate.return_value = None
|
||||
mock_dedup.compute_duplicate_rate.return_value = {
|
||||
"duplicate_rate": 0.0,
|
||||
"visual_similarity": 0.0,
|
||||
"match_count": 0,
|
||||
}
|
||||
|
||||
result = create_video_record_and_dedup(
|
||||
generation_task_id="task-thumb-gen",
|
||||
@@ -443,6 +453,11 @@ class TestThumbnailInDedupHelpers:
|
||||
mock_dedup.compute_fingerprint.return_value = MagicMock(to_dict=lambda: {})
|
||||
mock_dedup.check_duplicate.return_value = None
|
||||
mock_dedup.check_batch_duplicate.return_value = None
|
||||
mock_dedup.compute_duplicate_rate.return_value = {
|
||||
"duplicate_rate": 0.0,
|
||||
"visual_similarity": 0.0,
|
||||
"match_count": 0,
|
||||
}
|
||||
|
||||
result = create_video_record_and_dedup(
|
||||
generation_task_id="task-thumb-fail",
|
||||
|
||||
@@ -725,8 +725,12 @@ class TestCreatePreviewRoute:
|
||||
generation_task_repository=repo,
|
||||
db=MagicMock(),
|
||||
)
|
||||
assert resp.task_id == "preview_task_001"
|
||||
assert resp.status == "pending"
|
||||
# 批量响应:N=1 时 items 长度为 1
|
||||
assert resp.total == 1
|
||||
assert len(resp.items) == 1
|
||||
assert resp.items[0].task_id == "preview_task_001"
|
||||
assert resp.items[0].status == "pending"
|
||||
assert resp.items[0].variant_index == 0
|
||||
|
||||
def test_user_pending_limit_exceeded(self):
|
||||
"""用户待处理任务超限 → 429"""
|
||||
@@ -807,7 +811,13 @@ class TestCreatePreviewRoute:
|
||||
repo.count_pending_total.return_value = 0
|
||||
|
||||
task = _make_task()
|
||||
from fastapi import HTTPException
|
||||
|
||||
# 模拟 mark_failed 真实更新任务状态(_mark_task_failed 内部调用)
|
||||
def _set_failed(error_message="", **_kwargs):
|
||||
task.status = GenerationTaskStatus.FAILED
|
||||
task.error_message = error_message
|
||||
|
||||
task.mark_failed.side_effect = _set_failed
|
||||
|
||||
with patch("app.api.routes.generation_preview.CreateGenerationTaskUseCase") as MockUC:
|
||||
MockUC.return_value.execute.return_value = task
|
||||
@@ -815,14 +825,15 @@ class TestCreatePreviewRoute:
|
||||
"app.api.routes.generation_preview.safe_enqueue_generation_task",
|
||||
return_value=False,
|
||||
):
|
||||
with pytest.raises(HTTPException) as exc_info:
|
||||
create_preview_generation_task(
|
||||
self._make_request(),
|
||||
authenticated_user=_make_user(),
|
||||
generation_task_repository=repo,
|
||||
db=MagicMock(),
|
||||
)
|
||||
assert exc_info.value.status_code == 500
|
||||
resp = create_preview_generation_task(
|
||||
self._make_request(),
|
||||
authenticated_user=_make_user(),
|
||||
generation_task_repository=repo,
|
||||
db=MagicMock(),
|
||||
)
|
||||
# 入队失败:任务被标记 failed(mark_failed 设置错误信息),响应正常返回
|
||||
assert resp.total == 1
|
||||
assert resp.items[0].status == "failed"
|
||||
|
||||
def test_enqueue_raises_user_limit(self):
|
||||
"""safe_enqueue 抛出 UserPendingLimitExceeded → 429"""
|
||||
@@ -833,6 +844,12 @@ class TestCreatePreviewRoute:
|
||||
task = _make_task()
|
||||
from fastapi import HTTPException
|
||||
|
||||
def _set_failed_limit(error_message="", **_kwargs):
|
||||
task.status = GenerationTaskStatus.FAILED
|
||||
task.error_message = error_message or "待处理任务超限"
|
||||
|
||||
task.mark_failed.side_effect = _set_failed_limit
|
||||
|
||||
with patch("app.api.routes.generation_preview.CreateGenerationTaskUseCase") as MockUC:
|
||||
MockUC.return_value.execute.return_value = task
|
||||
with patch(
|
||||
@@ -846,6 +863,7 @@ class TestCreatePreviewRoute:
|
||||
generation_task_repository=repo,
|
||||
db=MagicMock(),
|
||||
)
|
||||
# 全部变体入队失败且错误消息含"待处理任务" → 429
|
||||
assert exc_info.value.status_code == 429
|
||||
|
||||
def test_enqueue_raises_global_queue_full(self):
|
||||
@@ -857,6 +875,12 @@ class TestCreatePreviewRoute:
|
||||
task = _make_task()
|
||||
from fastapi import HTTPException
|
||||
|
||||
def _set_failed_queue(error_message="", **_kwargs):
|
||||
task.status = GenerationTaskStatus.FAILED
|
||||
task.error_message = error_message or "系统队列已满"
|
||||
|
||||
task.mark_failed.side_effect = _set_failed_queue
|
||||
|
||||
with patch("app.api.routes.generation_preview.CreateGenerationTaskUseCase") as MockUC:
|
||||
MockUC.return_value.execute.return_value = task
|
||||
with patch(
|
||||
@@ -870,6 +894,7 @@ class TestCreatePreviewRoute:
|
||||
generation_task_repository=repo,
|
||||
db=MagicMock(),
|
||||
)
|
||||
# 全部变体入队失败且错误消息含"队列" → 503
|
||||
assert exc_info.value.status_code == 503
|
||||
|
||||
|
||||
|
||||
@@ -0,0 +1,273 @@
|
||||
"""Issue #1658: pHash 阈值校准 + 颜色直方图融合 — 单元测试.
|
||||
|
||||
在 #1659(动态抽帧+滑动窗口)与 #1660(查重率)已合入 develop 的基础上,
|
||||
本测试覆盖 #1658 的最小增量改动:
|
||||
|
||||
1. PHASH_THRESHOLD 由 10 收紧到 8(核心校准)
|
||||
2. 融合权重常量 MATCH_RATIO_THRESHOLD / PHASH_WEIGHT / HISTOGRAM_WEIGHT 实际生效
|
||||
(不再是硬编码魔法数字)
|
||||
3. VideoDeduplicator._compute_fusion_score 统一融合得分方法:
|
||||
- 无直方图数据时回退中性值 0.5
|
||||
- DB NULL(None)显式回退空列表,不崩溃
|
||||
- 全零直方图(全黑视频)为有效数据,参与 Bhattacharyya 计算
|
||||
- 返回 0~1 原始得分,判重由调用方与 DUPLICATE_THRESHOLD 比较
|
||||
4. Bhattacharyya 系数对上游异常负值有 sqrt domain 防御
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import sys
|
||||
from unittest.mock import MagicMock
|
||||
|
||||
import pytest
|
||||
|
||||
|
||||
def _mock_module(**attrs):
|
||||
"""Create a mock module with __spec__ to avoid AttributeError."""
|
||||
m = MagicMock()
|
||||
m.__spec__ = None
|
||||
for k, v in attrs.items():
|
||||
setattr(m, k, v)
|
||||
return m
|
||||
|
||||
|
||||
# ── Module-level setup: mock deps, import dedup, then restore sys.modules ──
|
||||
_SAVED_MODULES_KEYS = set(sys.modules.keys())
|
||||
_SAVED_MODULES_VALUES = {
|
||||
k: sys.modules.get(k)
|
||||
for k in [
|
||||
"cv2",
|
||||
"celery",
|
||||
"sqlalchemy",
|
||||
"sqlalchemy.orm",
|
||||
"sqlalchemy.engine",
|
||||
"sqlalchemy.ext",
|
||||
"sqlalchemy.ext.declarative",
|
||||
"worker_app.db",
|
||||
"worker_app.celery_app",
|
||||
"worker_app.core.config",
|
||||
"packages.adapters.sqlalchemy_impl.session",
|
||||
"packages.adapters.sqlalchemy_impl.generated_video_repository",
|
||||
"packages.adapters.sqlalchemy_impl.models",
|
||||
"packages.shared.config",
|
||||
"packages.shared.storage",
|
||||
]
|
||||
}
|
||||
|
||||
sys.modules["cv2"] = _mock_module()
|
||||
|
||||
_mock_celery = MagicMock()
|
||||
_mock_celery.Task = MagicMock
|
||||
_mock_celery.Celery = MagicMock
|
||||
_mock_celery.__spec__ = None
|
||||
sys.modules["celery"] = _mock_celery
|
||||
|
||||
_mock_sqla = MagicMock()
|
||||
_mock_sqla.__path__ = []
|
||||
_mock_sqla.__spec__ = None
|
||||
sys.modules["sqlalchemy"] = _mock_sqla
|
||||
|
||||
_mock_sqla_orm = MagicMock()
|
||||
_mock_sqla_orm.__path__ = []
|
||||
_mock_sqla_orm.__spec__ = None
|
||||
_mock_sqla_orm.Session = MagicMock
|
||||
sys.modules["sqlalchemy.orm"] = _mock_sqla_orm
|
||||
sys.modules["sqlalchemy.engine"] = _mock_module()
|
||||
sys.modules["sqlalchemy.ext"] = _mock_module()
|
||||
sys.modules["sqlalchemy.ext.declarative"] = _mock_module()
|
||||
|
||||
sys.modules["worker_app.db"] = _mock_module(SessionLocal=MagicMock())
|
||||
sys.modules["worker_app.celery_app"] = _mock_module(celery_app=MagicMock())
|
||||
sys.modules["worker_app.core.config"] = _mock_module(get_settings=MagicMock(return_value=MagicMock()))
|
||||
|
||||
sys.modules["packages.adapters.sqlalchemy_impl.session"] = _mock_module(
|
||||
Base=MagicMock(),
|
||||
build_engine=MagicMock(),
|
||||
build_session_factory=MagicMock(),
|
||||
ensure_database_exists=MagicMock(),
|
||||
initialize_database=MagicMock(),
|
||||
)
|
||||
sys.modules["packages.adapters.sqlalchemy_impl.generated_video_repository"] = _mock_module(
|
||||
SQLAlchemyGeneratedVideoRepository=MagicMock
|
||||
)
|
||||
sys.modules["packages.adapters.sqlalchemy_impl.models"] = _mock_module(
|
||||
VideoFingerprintChunkModel=MagicMock,
|
||||
GeneratedVideoModel=MagicMock,
|
||||
)
|
||||
sys.modules["packages.shared.config"] = _mock_module(get_shared_settings=MagicMock(return_value=MagicMock()))
|
||||
sys.modules["packages.shared.storage"] = _mock_module()
|
||||
|
||||
import video_processing.dedup as _dedup_mod # noqa: E402
|
||||
from video_processing.dedup import ( # noqa: E402
|
||||
DUPLICATE_THRESHOLD,
|
||||
HISTOGRAM_WEIGHT,
|
||||
MATCH_RATIO_THRESHOLD,
|
||||
PHASH_WEIGHT,
|
||||
VideoDeduplicator,
|
||||
)
|
||||
|
||||
# ── Restore sys.modules immediately after import ──
|
||||
for _key in list(sys.modules.keys()):
|
||||
if _key not in _SAVED_MODULES_KEYS:
|
||||
del sys.modules[_key]
|
||||
for _key, _value in _SAVED_MODULES_VALUES.items():
|
||||
if _value is not None:
|
||||
sys.modules[_key] = _value
|
||||
elif _key in sys.modules:
|
||||
del sys.modules[_key]
|
||||
del _SAVED_MODULES_KEYS, _SAVED_MODULES_VALUES, _key, _value
|
||||
|
||||
|
||||
# ── 测试夹具 ─────────────────────────────────────────────────────
|
||||
|
||||
_UNIFORM_HIST = [1.0 / 96] * 96 # 归一化均匀直方图,sum=1.0,自相似度≈1.0
|
||||
_ZERO_HIST = [0.0] * 96 # 全黑视频的全零直方图(有效数据)
|
||||
|
||||
|
||||
# ── TestThresholdCalibration:#1658 核心校准 ────────────────────
|
||||
|
||||
|
||||
class TestThresholdCalibration:
|
||||
"""pHash 阈值由 10 收紧到 8(Issue #1658)。"""
|
||||
|
||||
def test_phash_threshold_is_8(self):
|
||||
"""PHASH_THRESHOLD 必须为 8(旧值 10 会放过 8~9 汉明距离的不同视频)。"""
|
||||
assert VideoDeduplicator.PHASH_THRESHOLD == 8
|
||||
|
||||
def test_match_ratio_threshold_constant(self):
|
||||
assert MATCH_RATIO_THRESHOLD == 0.7
|
||||
|
||||
def test_duplicate_threshold_constant(self):
|
||||
assert DUPLICATE_THRESHOLD == 0.70
|
||||
|
||||
def test_fusion_weights(self):
|
||||
assert PHASH_WEIGHT == 0.7
|
||||
assert HISTOGRAM_WEIGHT == 0.3
|
||||
|
||||
def test_threshold_tightening_excludes_distance_8_and_9(self):
|
||||
"""距离 8、9 的帧:旧阈值 10 下算匹配,新阈值 8 下不算匹配。
|
||||
|
||||
场景:5 个关键帧距离为 [7, 7, 7, 9, 9]。
|
||||
- 旧阈值 10:5 帧全部 < 10 → match_ratio = 1.0(误放过)
|
||||
- 新阈值 8:仅 3 帧 < 8 → match_ratio = 0.6 < 0.7(正确跳过)
|
||||
"""
|
||||
distances = [7, 7, 7, 9, 9]
|
||||
|
||||
matched_old = sum(1 for d in distances if d < 10)
|
||||
assert matched_old == 5 # 旧行为:全匹配 → 误判风险
|
||||
|
||||
matched_new = sum(1 for d in distances if d < VideoDeduplicator.PHASH_THRESHOLD)
|
||||
assert matched_new == 3
|
||||
assert matched_new / len(distances) == 0.6
|
||||
assert matched_new / len(distances) < MATCH_RATIO_THRESHOLD # 被帧比例门槛拦截
|
||||
|
||||
|
||||
# ── TestComputeFusionScore:统一融合得分方法 ────────────────────
|
||||
|
||||
|
||||
class TestComputeFusionScore:
|
||||
"""_compute_fusion_score(median_distance, histograms_a, histograms_b)。"""
|
||||
|
||||
def test_no_histogram_falls_back_to_neutral_05(self):
|
||||
"""双方均无直方图 → hist_similarity 回退 0.5。
|
||||
|
||||
d=0: 0.7*1.0 + 0.3*0.5 = 0.85
|
||||
"""
|
||||
score = VideoDeduplicator._compute_fusion_score(0, [], [])
|
||||
assert score == pytest.approx(0.85, abs=1e-6)
|
||||
|
||||
def test_none_histograms_treated_as_empty(self):
|
||||
"""DB NULL(None)必须显式回退空列表,不得 len(None) 崩溃。"""
|
||||
score_none = VideoDeduplicator._compute_fusion_score(0, [], None)
|
||||
score_empty = VideoDeduplicator._compute_fusion_score(0, [], [])
|
||||
assert score_none == pytest.approx(score_empty, abs=1e-9)
|
||||
assert score_none == pytest.approx(0.85, abs=1e-6)
|
||||
|
||||
def test_none_histograms_on_query_side_no_crash(self):
|
||||
"""查询侧直方图为 None 时同样不崩溃。"""
|
||||
score = VideoDeduplicator._compute_fusion_score(0, None, [_UNIFORM_HIST])
|
||||
# 查询侧无直方图 → 平均相似度为 0(无 ha 可匹配)→ 0.7*1.0 + 0.3*0 = 0.7
|
||||
assert score == pytest.approx(0.7, abs=1e-6)
|
||||
|
||||
def test_identical_uniform_histograms_score_near_1(self):
|
||||
"""完全相同的归一化直方图:Bhattacharyya≈1.0 → 融合分≈1.0。"""
|
||||
score = VideoDeduplicator._compute_fusion_score(0, [_UNIFORM_HIST], [_UNIFORM_HIST])
|
||||
assert score == pytest.approx(1.0, abs=1e-6)
|
||||
|
||||
def test_all_zero_histogram_is_valid_data(self):
|
||||
"""全零直方图(全黑视频)是有效数据,Bhattacharyya=0,不得走 0.5 回退。
|
||||
|
||||
若错误地用 `if histograms_b` 之外的 `or []` 把全零列表清空,
|
||||
会错误回退到 0.5,把全黑视频的相似度抬高 0.15。
|
||||
d=0 时:正确行为 hist_sim=0 → 0.7*1.0 + 0.3*0 = 0.7;
|
||||
若全零直方图被错误清空回退 0.5 → 0.85。
|
||||
"""
|
||||
score = VideoDeduplicator._compute_fusion_score(0, [_ZERO_HIST], [_ZERO_HIST])
|
||||
assert score == pytest.approx(0.7, abs=1e-6)
|
||||
# 与错误回退值 0.85 明确区分开
|
||||
assert abs(score - 0.85) > 0.1
|
||||
# 注:d=0 时 phash 满分 0.7 恰达 DUPLICATE_THRESHOLD,全黑+完全相同 phash 仍判重,符合预期
|
||||
assert score >= DUPLICATE_THRESHOLD - 1e-9
|
||||
|
||||
def test_score_range_within_0_1(self):
|
||||
for d in (0, 8, 16, 32, 64):
|
||||
score = VideoDeduplicator._compute_fusion_score(d, [_UNIFORM_HIST], [_UNIFORM_HIST])
|
||||
assert 0.0 <= score <= 1.0
|
||||
|
||||
def test_formula_matches_weights(self):
|
||||
"""得分 = PHASH_WEIGHT * (1 - d/64) + HISTOGRAM_WEIGHT * hist_sim。"""
|
||||
d = 6 # phash_sim = 1 - 6/64 = 0.90625
|
||||
score = VideoDeduplicator._compute_fusion_score(d, [], []) # hist 回退 0.5
|
||||
expected = PHASH_WEIGHT * (1 - d / 64) + HISTOGRAM_WEIGHT * 0.5
|
||||
assert score == pytest.approx(expected, abs=1e-9)
|
||||
# 0.7*0.90625 + 0.15 = 0.634375 + 0.15 = 0.784375
|
||||
assert score == pytest.approx(0.784375, abs=1e-6)
|
||||
|
||||
|
||||
# ── TestBhattacharyyaDefense:负值/异常输入防御 ─────────────────
|
||||
|
||||
|
||||
class TestBhattacharyyaDefense:
|
||||
"""Bhattacharyya 系数对异常输入的防御。"""
|
||||
|
||||
def test_negative_values_do_not_raise(self):
|
||||
"""上游异常负值不得触发 sqrt domain error(max(0.0, ai*bi) 保护)。"""
|
||||
bad_hist = [-0.01] * 96 # 异常负值
|
||||
coeff = VideoDeduplicator._bhattacharyya_coefficient(bad_hist, _UNIFORM_HIST)
|
||||
# 负值乘积被钳为 0,系数为 0 而不是抛 ValueError
|
||||
assert coeff == pytest.approx(0.0, abs=1e-9)
|
||||
|
||||
def test_normal_histograms_coefficient_near_1(self):
|
||||
coeff = VideoDeduplicator._bhattacharyya_coefficient(_UNIFORM_HIST, _UNIFORM_HIST)
|
||||
assert coeff == pytest.approx(1.0, abs=1e-6)
|
||||
|
||||
def test_disjoint_histograms_coefficient_0(self):
|
||||
"""完全不重叠的直方图(前半 vs 后半非零)系数为 0。"""
|
||||
hist_a = [0.0] * 96
|
||||
hist_b = [0.0] * 96
|
||||
for i in range(48):
|
||||
hist_a[i] = 1.0 / 48
|
||||
for i in range(48, 96):
|
||||
hist_b[i] = 1.0 / 48
|
||||
coeff = VideoDeduplicator._bhattacharyya_coefficient(hist_a, hist_b)
|
||||
assert coeff == pytest.approx(0.0, abs=1e-9)
|
||||
|
||||
|
||||
# ── TestHistogramSimilarityEdgeCases ────────────────────────────
|
||||
|
||||
|
||||
class TestHistogramSimilarityEdgeCases:
|
||||
"""_compute_histogram_similarity 的边界行为。"""
|
||||
|
||||
def test_empty_either_side_returns_0(self):
|
||||
assert VideoDeduplicator._compute_histogram_similarity([], [_UNIFORM_HIST]) == 0.0
|
||||
assert VideoDeduplicator._compute_histogram_similarity([_UNIFORM_HIST], []) == 0.0
|
||||
|
||||
def test_best_match_per_histogram(self):
|
||||
"""每个查询直方图取与目标集合的最佳匹配,再取平均。"""
|
||||
h1 = _UNIFORM_HIST
|
||||
h2 = [0.0] * 96
|
||||
h2[0] = 1.0 # 与均匀直方图完全不重叠
|
||||
# 查询侧两张直方图:h1 最佳匹配≈1.0,h2 最佳匹配≈sqrt(1/96)≈0.102
|
||||
sim = VideoDeduplicator._compute_histogram_similarity([h1, h2], [h1])
|
||||
assert sim == pytest.approx((1.0 + (1.0 / 96) ** 0.5) / 2, abs=1e-3)
|
||||
@@ -0,0 +1,207 @@
|
||||
"""#1664 随机边缘裁剪降重功能测试"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import os
|
||||
import subprocess
|
||||
import sys
|
||||
from pathlib import Path
|
||||
from unittest.mock import MagicMock, call, patch
|
||||
|
||||
import pytest
|
||||
|
||||
ROOT = Path(__file__).resolve().parents[2]
|
||||
sys.path.insert(0, str(ROOT / "apps" / "worker"))
|
||||
sys.path.insert(0, str(ROOT / "apps" / "api"))
|
||||
sys.path.insert(0, str(ROOT / "packages"))
|
||||
|
||||
|
||||
from video_processing.ffmpeg_utils import random_edge_crop
|
||||
|
||||
|
||||
class TestRandomEdgeCropBasic:
|
||||
"""基本功能测试"""
|
||||
|
||||
def test_returns_input_path_when_output_none(self, tmp_path):
|
||||
"""output_path=None 时覆盖原文件并返回 input_path"""
|
||||
input_file = tmp_path / "input.mp4"
|
||||
input_file.write_bytes(b"fake video data")
|
||||
|
||||
fake_info = {"width": 1920, "height": 1080, "duration": 10, "fps": 30}
|
||||
with (
|
||||
patch("video_processing.ffmpeg_utils.probe_video_info", return_value=fake_info),
|
||||
patch("video_processing.ffmpeg_utils.run_ffmpeg"),
|
||||
):
|
||||
result = random_edge_crop(input_file)
|
||||
|
||||
assert result == input_file
|
||||
|
||||
def test_returns_output_path_when_specified(self, tmp_path):
|
||||
"""指定 output_path 时返回该路径"""
|
||||
input_file = tmp_path / "input.mp4"
|
||||
input_file.write_bytes(b"fake video data")
|
||||
output_file = tmp_path / "output.mp4"
|
||||
|
||||
fake_info = {"width": 1920, "height": 1080, "duration": 10, "fps": 30}
|
||||
with (
|
||||
patch("video_processing.ffmpeg_utils.probe_video_info", return_value=fake_info),
|
||||
patch("video_processing.ffmpeg_utils.run_ffmpeg"),
|
||||
):
|
||||
result = random_edge_crop(input_file, output_file)
|
||||
|
||||
assert result == output_file
|
||||
|
||||
def test_skip_when_invalid_resolution(self, tmp_path):
|
||||
"""无法获取有效分辨率时跳过裁剪"""
|
||||
input_file = tmp_path / "input.mp4"
|
||||
input_file.write_bytes(b"fake video data")
|
||||
|
||||
fake_info = {"width": 0, "height": 0, "duration": 10, "fps": 30}
|
||||
with (
|
||||
patch("video_processing.ffmpeg_utils.probe_video_info", return_value=fake_info),
|
||||
patch("video_processing.ffmpeg_utils.run_ffmpeg") as mock_ffmpeg,
|
||||
):
|
||||
result = random_edge_crop(input_file)
|
||||
|
||||
assert result == input_file
|
||||
mock_ffmpeg.assert_not_called()
|
||||
|
||||
|
||||
class TestRandomEdgeCropFFmpeg:
|
||||
"""FFmpeg 调用参数验证"""
|
||||
|
||||
def test_ffmpeg_crop_and_scale_filter(self, tmp_path):
|
||||
"""生成的 ffmpeg 滤镜包含 crop + scale"""
|
||||
input_file = tmp_path / "input.mp4"
|
||||
input_file.write_bytes(b"fake video data")
|
||||
|
||||
# 固定随机值以便验证
|
||||
fake_info = {"width": 1000, "height": 1000, "duration": 10, "fps": 30}
|
||||
|
||||
with (
|
||||
patch("video_processing.ffmpeg_utils.probe_video_info", return_value=fake_info),
|
||||
patch("video_processing.ffmpeg_utils.run_ffmpeg") as mock_ffmpeg,
|
||||
patch("random.uniform", side_effect=[0.03, 0.03, 0.03, 0.03]),
|
||||
):
|
||||
random_edge_crop(input_file)
|
||||
|
||||
mock_ffmpeg.assert_called_once()
|
||||
cmd = mock_ffmpeg.call_args[0][0]
|
||||
# 找到 -vf 参数
|
||||
vf_idx = cmd.index("-vf")
|
||||
vf_value = cmd[vf_idx + 1]
|
||||
assert "crop=" in vf_value
|
||||
assert "scale=1000:1000" in vf_value
|
||||
|
||||
def test_crop_amounts_within_range(self, tmp_path):
|
||||
"""裁剪量在 2%~5% 范围内"""
|
||||
input_file = tmp_path / "input.mp4"
|
||||
input_file.write_bytes(b"fake video data")
|
||||
|
||||
fake_info = {"width": 1000, "height": 1000, "duration": 10, "fps": 30}
|
||||
|
||||
with (
|
||||
patch("video_processing.ffmpeg_utils.probe_video_info", return_value=fake_info),
|
||||
patch("video_processing.ffmpeg_utils.run_ffmpeg") as mock_ffmpeg,
|
||||
patch("random.uniform", side_effect=[0.02, 0.05, 0.02, 0.05]),
|
||||
):
|
||||
random_edge_crop(input_file)
|
||||
|
||||
cmd = mock_ffmpeg.call_args[0][0]
|
||||
vf_idx = cmd.index("-vf")
|
||||
vf_value = cmd[vf_idx + 1]
|
||||
# crop_top=20, crop_bottom=50, crop_left=20, crop_right=50
|
||||
# new_w = 1000-20-50 = 930, new_h = 1000-20-50 = 930
|
||||
# x_offset = 20, y_offset = 20
|
||||
assert "crop=930:930:20:20" in vf_value
|
||||
|
||||
def test_uses_libx264_codec(self, tmp_path):
|
||||
"""使用 libx264 编码"""
|
||||
input_file = tmp_path / "input.mp4"
|
||||
input_file.write_bytes(b"fake video data")
|
||||
|
||||
fake_info = {"width": 1920, "height": 1080, "duration": 10, "fps": 30}
|
||||
with (
|
||||
patch("video_processing.ffmpeg_utils.probe_video_info", return_value=fake_info),
|
||||
patch("video_processing.ffmpeg_utils.run_ffmpeg") as mock_ffmpeg,
|
||||
):
|
||||
random_edge_crop(input_file)
|
||||
|
||||
cmd = mock_ffmpeg.call_args[0][0]
|
||||
assert "-c:v" in cmd
|
||||
assert cmd[cmd.index("-c:v") + 1] == "libx264"
|
||||
|
||||
|
||||
class TestRandomEdgeCropErrorHandling:
|
||||
"""错误处理测试"""
|
||||
|
||||
def test_ffmpeg_failure_raises_exception(self, tmp_path):
|
||||
"""ffmpeg 失败时抛出异常"""
|
||||
input_file = tmp_path / "input.mp4"
|
||||
input_file.write_bytes(b"fake video data")
|
||||
|
||||
fake_info = {"width": 1920, "height": 1080, "duration": 10, "fps": 30}
|
||||
with (
|
||||
patch("video_processing.ffmpeg_utils.probe_video_info", return_value=fake_info),
|
||||
patch(
|
||||
"video_processing.ffmpeg_utils.run_ffmpeg",
|
||||
side_effect=subprocess.CalledProcessError(1, "ffmpeg"),
|
||||
),
|
||||
):
|
||||
with pytest.raises(subprocess.CalledProcessError):
|
||||
random_edge_crop(input_file)
|
||||
|
||||
def test_probe_failure_propagates(self, tmp_path):
|
||||
"""probe_video_info 失败时异常传播"""
|
||||
input_file = tmp_path / "input.mp4"
|
||||
input_file.write_bytes(b"fake video data")
|
||||
|
||||
with patch(
|
||||
"video_processing.ffmpeg_utils.probe_video_info",
|
||||
side_effect=RuntimeError("probe failed"),
|
||||
):
|
||||
with pytest.raises(RuntimeError, match="probe failed"):
|
||||
random_edge_crop(input_file)
|
||||
|
||||
|
||||
class TestRandomEdgeCropEvenDimensions:
|
||||
"""偶数尺寸处理测试"""
|
||||
|
||||
def test_odd_crop_dimensions_adjusted_to_even(self, tmp_path):
|
||||
"""裁剪后尺寸为奇数时自动调整为偶数"""
|
||||
input_file = tmp_path / "input.mp4"
|
||||
input_file.write_bytes(b"fake video data")
|
||||
|
||||
# 1000 - 3 (top) - 4 (bottom) = 993 → 调整为 992
|
||||
# 1000 - 3 (left) - 4 (right) = 993 → 调整为 992
|
||||
fake_info = {"width": 1000, "height": 1000, "duration": 10, "fps": 30}
|
||||
|
||||
with (
|
||||
patch("video_processing.ffmpeg_utils.probe_video_info", return_value=fake_info),
|
||||
patch("video_processing.ffmpeg_utils.run_ffmpeg") as mock_ffmpeg,
|
||||
# side_effect 控制 uniform 返回值
|
||||
# top: 0.003*1000=3, bottom: 0.004*1000=4, left: 0.003*1000=3, right: 0.004*1000=4
|
||||
):
|
||||
# 使用自定义 uniform 返回特定值
|
||||
def fake_uniform(low, high):
|
||||
# 返回特定百分比使得裁剪后尺寸为奇数
|
||||
# 我们需要 crop_top=3, crop_bottom=4, crop_left=3, crop_right=4
|
||||
return 0.0035 # 近似值
|
||||
|
||||
# 更简单的方式:直接 mock int(H * random.uniform(...)) 的结果
|
||||
# 但我们直接测试最终 crop 滤镜即可
|
||||
with patch("random.uniform", side_effect=[0.021, 0.022, 0.021, 0.022]):
|
||||
random_edge_crop(input_file)
|
||||
|
||||
cmd = mock_ffmpeg.call_args[0][0]
|
||||
vf_idx = cmd.index("-vf")
|
||||
vf_value = cmd[vf_idx + 1]
|
||||
# 提取 crop 参数并验证都是偶数
|
||||
import re
|
||||
|
||||
crop_match = re.search(r"crop=(\d+):(\d+)", vf_value)
|
||||
assert crop_match
|
||||
crop_w = int(crop_match.group(1))
|
||||
crop_h = int(crop_match.group(2))
|
||||
assert crop_w % 2 == 0, f"crop width {crop_w} should be even"
|
||||
assert crop_h % 2 == 0, f"crop height {crop_h} should be even"
|
||||
@@ -0,0 +1,167 @@
|
||||
"""Tests for POST /videos/recompute-dedup endpoint (#1664 follow-up)."""
|
||||
|
||||
from unittest.mock import MagicMock, patch
|
||||
|
||||
import pytest
|
||||
from fastapi.testclient import TestClient
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def mock_video():
|
||||
"""Mock video with missing dedup data."""
|
||||
v = MagicMock()
|
||||
v.id = "video-001"
|
||||
v.user_id = "user-abc"
|
||||
v.duplicate_rate = None
|
||||
v.video_fingerprint = None
|
||||
v.project_id = "proj-001"
|
||||
v.generation_task_id = "task-001"
|
||||
v.name = "test.mp4"
|
||||
v.file_url = "https://example.com/test.mp4"
|
||||
v.file_size = 1024
|
||||
v.duration = 10.0
|
||||
v.width = 1920
|
||||
v.height = 1080
|
||||
v.fps = 25.0
|
||||
v.status = "completed"
|
||||
v.review_status = "pending_review"
|
||||
v.generation_params = {}
|
||||
v.thumbnail_url = None
|
||||
v.is_duplicate = False
|
||||
v.duplicate_of = None
|
||||
v.match_count = None
|
||||
v.visual_similarity = None
|
||||
v.generated_at = "2026-09-04T00:00:00"
|
||||
return v
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def mock_video_with_dedup(mock_video):
|
||||
"""Mock video that already has dedup data."""
|
||||
mock_video.duplicate_rate = 15.5
|
||||
mock_video.video_fingerprint = {"md5": "abc123"}
|
||||
return mock_video
|
||||
|
||||
|
||||
class TestRecomputeDedupEndpoint:
|
||||
"""POST /videos/recompute-dedup"""
|
||||
|
||||
def test_enqueue_videos_without_dedup(self, mock_video):
|
||||
"""Videos missing duplicate_rate should be enqueued."""
|
||||
from app.api.routes.videos import RecomputeDedupRequest
|
||||
|
||||
mock_repo = MagicMock()
|
||||
mock_repo.list_by_user.return_value = [mock_video]
|
||||
|
||||
with (patch("app.api.routes.videos.celery_app") as mock_celery,):
|
||||
mock_celery.send_task.return_value = MagicMock(id="task-xyz")
|
||||
from app.api.routes.videos import recompute_dedup
|
||||
|
||||
auth_user = MagicMock()
|
||||
auth_user.user.id = "user-abc"
|
||||
|
||||
result = recompute_dedup(
|
||||
request=RecomputeDedupRequest(),
|
||||
repo=mock_repo,
|
||||
current_user=auth_user,
|
||||
)
|
||||
|
||||
assert result.enqueued == 1
|
||||
assert result.total_scanned == 1
|
||||
assert result.skipped == 0
|
||||
mock_celery.send_task.assert_called_once_with("worker.check_duplicate", args=["video-001"])
|
||||
|
||||
def test_skip_videos_with_complete_dedup(self, mock_video_with_dedup):
|
||||
"""Videos with both duplicate_rate and video_fingerprint should be skipped."""
|
||||
from app.api.routes.videos import RecomputeDedupRequest, recompute_dedup
|
||||
|
||||
mock_repo = MagicMock()
|
||||
mock_repo.list_by_user.return_value = [mock_video_with_dedup]
|
||||
|
||||
with patch("app.api.routes.videos.celery_app") as mock_celery:
|
||||
auth_user = MagicMock()
|
||||
auth_user.user.id = "user-abc"
|
||||
|
||||
result = recompute_dedup(
|
||||
request=RecomputeDedupRequest(),
|
||||
repo=mock_repo,
|
||||
current_user=auth_user,
|
||||
)
|
||||
|
||||
assert result.enqueued == 0
|
||||
assert result.total_scanned == 1
|
||||
assert result.skipped == 1
|
||||
mock_celery.send_task.assert_not_called()
|
||||
|
||||
def test_specific_video_ids(self, mock_video):
|
||||
"""When video_ids are provided, only those videos are processed."""
|
||||
from app.api.routes.videos import RecomputeDedupRequest, recompute_dedup
|
||||
|
||||
mock_repo = MagicMock()
|
||||
mock_repo.get_by_ids.return_value = [mock_video]
|
||||
|
||||
with patch("app.api.routes.videos.celery_app") as mock_celery:
|
||||
mock_celery.send_task.return_value = MagicMock(id="task-xyz")
|
||||
auth_user = MagicMock()
|
||||
auth_user.user.id = "user-abc"
|
||||
|
||||
result = recompute_dedup(
|
||||
request=RecomputeDedupRequest(video_ids=["video-001"]),
|
||||
repo=mock_repo,
|
||||
current_user=auth_user,
|
||||
)
|
||||
|
||||
assert result.enqueued == 1
|
||||
mock_repo.get_by_ids.assert_called_once_with(["video-001"])
|
||||
|
||||
def test_security_only_own_videos(self, mock_video):
|
||||
"""Videos belonging to other users should be filtered out."""
|
||||
from app.api.routes.videos import RecomputeDedupRequest, recompute_dedup
|
||||
|
||||
mock_video.user_id = "user-OTHER"
|
||||
mock_repo = MagicMock()
|
||||
mock_repo.get_by_ids.return_value = [mock_video]
|
||||
|
||||
with patch("app.api.routes.videos.celery_app") as mock_celery:
|
||||
auth_user = MagicMock()
|
||||
auth_user.user.id = "user-abc"
|
||||
|
||||
result = recompute_dedup(
|
||||
request=RecomputeDedupRequest(video_ids=["video-001"]),
|
||||
repo=mock_repo,
|
||||
current_user=auth_user,
|
||||
)
|
||||
|
||||
assert result.enqueued == 0
|
||||
mock_celery.send_task.assert_not_called()
|
||||
|
||||
def test_mixed_complete_and_incomplete(self, mock_video, mock_video_with_dedup):
|
||||
"""Mix of videos with and without dedup data."""
|
||||
import copy
|
||||
|
||||
from app.api.routes.videos import RecomputeDedupRequest, recompute_dedup
|
||||
|
||||
# Create a second video object
|
||||
v2 = MagicMock()
|
||||
v2.id = "video-002"
|
||||
v2.user_id = "user-abc"
|
||||
v2.duplicate_rate = None
|
||||
v2.video_fingerprint = None
|
||||
|
||||
mock_repo = MagicMock()
|
||||
mock_repo.list_by_user.return_value = [mock_video_with_dedup, v2]
|
||||
|
||||
with patch("app.api.routes.videos.celery_app") as mock_celery:
|
||||
mock_celery.send_task.return_value = MagicMock(id="task-xyz")
|
||||
auth_user = MagicMock()
|
||||
auth_user.user.id = "user-abc"
|
||||
|
||||
result = recompute_dedup(
|
||||
request=RecomputeDedupRequest(),
|
||||
repo=mock_repo,
|
||||
current_user=auth_user,
|
||||
)
|
||||
|
||||
assert result.enqueued == 1
|
||||
assert result.total_scanned == 2
|
||||
assert result.skipped == 1
|
||||
@@ -67,11 +67,11 @@ class TestPositionToAssAlignment:
|
||||
def test_bottom(self):
|
||||
assert _position_to_ass_alignment("bottom") == 2
|
||||
|
||||
def test_unknown_returns_top_default(self):
|
||||
assert _position_to_ass_alignment("unknown") == 8
|
||||
assert _position_to_ass_alignment("") == 8
|
||||
assert _position_to_ass_alignment("left") == 8
|
||||
assert _position_to_ass_alignment(None) == 8
|
||||
def test_unknown_returns_bottom_default(self):
|
||||
assert _position_to_ass_alignment("unknown") == 2
|
||||
assert _position_to_ass_alignment("") == 2
|
||||
assert _position_to_ass_alignment("left") == 2
|
||||
assert _position_to_ass_alignment(None) == 2
|
||||
|
||||
|
||||
class TestBuildAssStyle:
|
||||
|
||||
@@ -66,15 +66,15 @@ class TestPositionToAssAlignment:
|
||||
"""center → 居中(5)."""
|
||||
assert _position_to_ass_alignment("center") == 5
|
||||
|
||||
def test_unknown_defaults_to_top(self):
|
||||
"""未知位置默认顶部(8)."""
|
||||
assert _position_to_ass_alignment("unknown") == 8
|
||||
assert _position_to_ass_alignment("top_left") == 8
|
||||
assert _position_to_ass_alignment("bottom_right") == 8
|
||||
def test_unknown_defaults_to_bottom(self):
|
||||
"""未知位置默认底部(2)."""
|
||||
assert _position_to_ass_alignment("unknown") == 2
|
||||
assert _position_to_ass_alignment("top_left") == 2
|
||||
assert _position_to_ass_alignment("bottom_right") == 2
|
||||
|
||||
def test_empty_string_defaults_to_top(self):
|
||||
"""空字符串默认顶部."""
|
||||
assert _position_to_ass_alignment("") == 8
|
||||
def test_empty_string_defaults_to_bottom(self):
|
||||
"""空字符串默认底部."""
|
||||
assert _position_to_ass_alignment("") == 2
|
||||
|
||||
|
||||
class TestBuildAssStyle:
|
||||
|
||||
@@ -0,0 +1,123 @@
|
||||
"""#1660 成品视频 API 查重字段透传测试。
|
||||
|
||||
覆盖两套响应构造路径:
|
||||
- routes/videos.py::_to_video_response -> VideoItemResponse (/videos 列表)
|
||||
- routes/generation_tasks.py::_to_generated_video_response -> GeneratedVideoResponse
|
||||
"""
|
||||
|
||||
from types import SimpleNamespace
|
||||
from unittest.mock import MagicMock
|
||||
|
||||
from app.api.routes.generation_tasks import _to_generated_video_response
|
||||
from app.api.routes.videos import _to_video_response
|
||||
from app.schemas.generated_video import GeneratedVideoResponse
|
||||
from app.schemas.video_center import VideoItemResponse
|
||||
|
||||
|
||||
def _make_item(**overrides):
|
||||
base = dict(
|
||||
id="v1",
|
||||
project_id="p1",
|
||||
generation_task_id="t1",
|
||||
name="成片",
|
||||
file_url="oss://bucket/v1.mp4",
|
||||
file_size=1024,
|
||||
duration=12.5,
|
||||
thumbnail_url=None,
|
||||
width=1080,
|
||||
height=1920,
|
||||
fps=30.0,
|
||||
status="completed",
|
||||
review_status="pending_review",
|
||||
generation_params={},
|
||||
generated_at=None,
|
||||
duplicate_rate=None,
|
||||
match_count=None,
|
||||
visual_similarity=None,
|
||||
)
|
||||
base.update(overrides)
|
||||
return SimpleNamespace(**base)
|
||||
|
||||
|
||||
class TestVideoItemResponseDupFields:
|
||||
def test_passes_through_all_three_fields(self):
|
||||
item = _make_item(duplicate_rate=42.5, match_count=7, visual_similarity=0.83)
|
||||
resp = _to_video_response(item, storage=None)
|
||||
assert isinstance(resp, VideoItemResponse)
|
||||
assert resp.duplicate_rate == 42.5
|
||||
assert resp.match_count == 7
|
||||
assert resp.visual_similarity == 0.83
|
||||
|
||||
def test_legacy_video_without_fields_returns_none(self):
|
||||
"""老数据/实体无查重字段时保持 None(前端自动隐藏),不报错。"""
|
||||
item = SimpleNamespace(
|
||||
id="v2",
|
||||
project_id="p1",
|
||||
generation_task_id="t2",
|
||||
name="老视频",
|
||||
file_url="oss://bucket/v2.mp4",
|
||||
file_size=1,
|
||||
duration=1.0,
|
||||
thumbnail_url=None,
|
||||
width=720,
|
||||
height=1280,
|
||||
fps=24.0,
|
||||
status="completed",
|
||||
review_status="pending_review",
|
||||
generation_params={},
|
||||
)
|
||||
resp = _to_video_response(item, storage=None)
|
||||
assert resp.duplicate_rate is None
|
||||
assert resp.match_count is None
|
||||
assert resp.visual_similarity is None
|
||||
|
||||
def test_explicit_none_values_kept(self):
|
||||
item = _make_item()
|
||||
resp = _to_video_response(item, storage=None)
|
||||
assert resp.duplicate_rate is None
|
||||
assert resp.match_count is None
|
||||
assert resp.visual_similarity is None
|
||||
|
||||
def test_zero_match_count_is_valid_value(self):
|
||||
"""计算后确无匹配:match_count=0 / visual_similarity=0.0 是合法值,不能变 None。"""
|
||||
item = _make_item(duplicate_rate=0.0, match_count=0, visual_similarity=0.0)
|
||||
resp = _to_video_response(item, storage=None)
|
||||
assert resp.match_count == 0
|
||||
assert resp.visual_similarity == 0.0
|
||||
|
||||
|
||||
class TestGeneratedVideoResponseDupFields:
|
||||
def test_passes_through_all_three_fields(self):
|
||||
item = _make_item(duplicate_rate=15.2, match_count=3, visual_similarity=0.61)
|
||||
resp = _to_generated_video_response(item, download_url="https://dl/x")
|
||||
assert isinstance(resp, GeneratedVideoResponse)
|
||||
assert resp.duplicate_rate == 15.2
|
||||
assert resp.match_count == 3
|
||||
assert resp.visual_similarity == 0.61
|
||||
assert resp.download_url == "https://dl/x"
|
||||
|
||||
def test_missing_fields_default_none(self):
|
||||
item = SimpleNamespace(
|
||||
id="v3",
|
||||
project_id="p1",
|
||||
generation_task_id="t3",
|
||||
name="x",
|
||||
file_url="oss://x",
|
||||
file_size=1,
|
||||
duration=1.0,
|
||||
thumbnail_url=None,
|
||||
width=720,
|
||||
height=1280,
|
||||
fps=24.0,
|
||||
)
|
||||
resp = _to_generated_video_response(item)
|
||||
assert resp.duplicate_rate is None
|
||||
assert resp.match_count is None
|
||||
assert resp.visual_similarity is None
|
||||
|
||||
def test_storage_failure_falls_back_to_file_url(self):
|
||||
storage = MagicMock()
|
||||
storage.get_download_url.side_effect = RuntimeError("oss down")
|
||||
item = _make_item()
|
||||
resp = _to_video_response(item, storage=storage)
|
||||
assert resp.download_url == item.file_url
|
||||
@@ -0,0 +1,205 @@
|
||||
"""Tests for voice duration alignment feature.
|
||||
|
||||
Tests the _align_clips_to_voice_duration method in UnifiedRenderService.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from pathlib import Path
|
||||
from unittest.mock import MagicMock, patch
|
||||
|
||||
import pytest
|
||||
from video_processing.unified_render_service import RenderLayer, ResolvedClip, UnifiedRenderService
|
||||
|
||||
|
||||
class TestAlignClipsToVoiceDuration:
|
||||
"""Test clip duration alignment to voice audio."""
|
||||
|
||||
def _make_clip(
|
||||
self,
|
||||
clip_id: str,
|
||||
duration: float,
|
||||
actual_duration: float = 0.0,
|
||||
playback_speed: float = 1.0,
|
||||
) -> ResolvedClip:
|
||||
"""Helper to create a ResolvedClip for testing."""
|
||||
return ResolvedClip(
|
||||
clip_id=clip_id,
|
||||
asset_id=f"asset_{clip_id}",
|
||||
local_path=Path(f"/tmp/{clip_id}.mp4"),
|
||||
clip_type="main",
|
||||
order=0,
|
||||
duration=duration,
|
||||
actual_duration=actual_duration or duration,
|
||||
playback_speed=playback_speed,
|
||||
)
|
||||
|
||||
def _make_layer(self, role: str, clips: list[ResolvedClip]) -> RenderLayer:
|
||||
"""Helper to create a RenderLayer for testing."""
|
||||
return RenderLayer(role=role, clips=clips, z_index=0)
|
||||
|
||||
def _make_service(self, voiceover_path: str | None = None) -> UnifiedRenderService:
|
||||
"""Helper to create a mock UnifiedRenderService."""
|
||||
plan = MagicMock()
|
||||
plan.id = "test_plan"
|
||||
plan.config = {}
|
||||
|
||||
with patch.object(UnifiedRenderService, "__init__", lambda self, **kwargs: None):
|
||||
service = UnifiedRenderService.__new__(UnifiedRenderService)
|
||||
service.plan = plan
|
||||
service.voiceover_audio_path = voiceover_path
|
||||
service.transition_duration = 0.0
|
||||
return service
|
||||
|
||||
def test_no_voice_audio_no_adjustment(self):
|
||||
"""No voice audio → no adjustment."""
|
||||
service = self._make_service(voiceover_path=None)
|
||||
clips = [self._make_clip("c1", 10.0), self._make_clip("c2", 10.0)]
|
||||
layers = [self._make_layer("main", clips)]
|
||||
|
||||
service._align_clips_to_voice_duration(layers, voice_duration=0.0)
|
||||
|
||||
# No change
|
||||
assert clips[0].duration == 10.0
|
||||
assert clips[1].duration == 10.0
|
||||
|
||||
def test_ratio_within_5_percent_no_adjustment(self):
|
||||
"""Ratio within ±5% → no adjustment."""
|
||||
service = self._make_service()
|
||||
clips = [self._make_clip("c1", 10.0)]
|
||||
layers = [self._make_layer("main", clips)]
|
||||
|
||||
# Total clips = 10s, voice = 10.3s → ratio = 1.03 (within 5%)
|
||||
service._align_clips_to_voice_duration(layers, voice_duration=10.3)
|
||||
|
||||
assert clips[0].duration == 10.0 # Unchanged
|
||||
|
||||
def test_ratio_less_than_1_trim_clips(self):
|
||||
"""Ratio < 1 (clips too long) → trim clips proportionally."""
|
||||
service = self._make_service()
|
||||
clips = [self._make_clip("c1", 10.0), self._make_clip("c2", 10.0)]
|
||||
layers = [self._make_layer("main", clips)]
|
||||
|
||||
# Total clips = 20s, voice = 15s → ratio = 0.75
|
||||
service._align_clips_to_voice_duration(layers, voice_duration=15.0)
|
||||
|
||||
# Each clip should be trimmed to 75%
|
||||
assert abs(clips[0].duration - 7.5) < 0.01
|
||||
assert abs(clips[1].duration - 7.5) < 0.01
|
||||
|
||||
def test_ratio_greater_than_1_slowdown_clips(self):
|
||||
"""Ratio > 1 (clips too short) → slow down clips."""
|
||||
service = self._make_service()
|
||||
clips = [self._make_clip("c1", 10.0), self._make_clip("c2", 10.0)]
|
||||
layers = [self._make_layer("main", clips)]
|
||||
|
||||
# Total clips = 20s, voice = 25s → ratio = 1.25
|
||||
service._align_clips_to_voice_duration(layers, voice_duration=25.0)
|
||||
|
||||
# Each clip's speed should be reduced: 1.0 / 1.25 = 0.8
|
||||
assert abs(clips[0].playback_speed - 0.8) < 0.01
|
||||
assert abs(clips[1].playback_speed - 0.8) < 0.01
|
||||
|
||||
def test_speed_lower_bound_025(self):
|
||||
"""Playback speed should not go below 0.25x."""
|
||||
service = self._make_service()
|
||||
clips = [self._make_clip("c1", 5.0)]
|
||||
layers = [self._make_layer("main", clips)]
|
||||
|
||||
# Total clips = 5s, voice = 50s → ratio = 10.0
|
||||
# Speed would be 1.0 / 10 = 0.1, but should be clamped to 0.25
|
||||
service._align_clips_to_voice_duration(layers, voice_duration=50.0)
|
||||
|
||||
assert clips[0].playback_speed == 0.25
|
||||
|
||||
def test_only_video_layers_adjusted(self):
|
||||
"""Only main/broll/background layers are adjusted, not audio."""
|
||||
service = self._make_service()
|
||||
|
||||
video_clips = [self._make_clip("v1", 10.0)]
|
||||
audio_clips = [self._make_clip("a1", 10.0)]
|
||||
|
||||
layers = [
|
||||
self._make_layer("main", video_clips),
|
||||
self._make_layer("audio", audio_clips),
|
||||
]
|
||||
|
||||
# ratio = 0.5 → should trim video but not audio
|
||||
service._align_clips_to_voice_duration(layers, voice_duration=5.0)
|
||||
|
||||
assert abs(video_clips[0].duration - 5.0) < 0.01 # Trimmed
|
||||
assert audio_clips[0].duration == 10.0 # Unchanged
|
||||
|
||||
def test_multiple_video_layers_all_adjusted(self):
|
||||
"""All video layers (main, broll, background) are adjusted."""
|
||||
service = self._make_service()
|
||||
|
||||
main_clips = [self._make_clip("m1", 10.0)]
|
||||
broll_clips = [self._make_clip("b1", 10.0)]
|
||||
bg_clips = [self._make_clip("bg1", 10.0)]
|
||||
|
||||
layers = [
|
||||
self._make_layer("main", main_clips),
|
||||
self._make_layer("broll", broll_clips),
|
||||
self._make_layer("background", bg_clips),
|
||||
]
|
||||
|
||||
# Total video = 30s, voice = 15s → ratio = 0.5
|
||||
service._align_clips_to_voice_duration(layers, voice_duration=15.0)
|
||||
|
||||
# All should be trimmed to 50%
|
||||
assert abs(main_clips[0].duration - 5.0) < 0.01
|
||||
assert abs(broll_clips[0].duration - 5.0) < 0.01
|
||||
assert abs(bg_clips[0].duration - 5.0) < 0.01
|
||||
|
||||
def test_trim_config_also_adjusted(self):
|
||||
"""When clip has trim_config, it should also be adjusted."""
|
||||
from video_processing.trim_engine import TrimConfig
|
||||
|
||||
service = self._make_service()
|
||||
|
||||
clip = self._make_clip("c1", 10.0)
|
||||
clip.trim_config = TrimConfig(start_time=0.0, duration=10.0)
|
||||
|
||||
layers = [self._make_layer("main", [clip])]
|
||||
|
||||
# ratio = 0.5
|
||||
service._align_clips_to_voice_duration(layers, voice_duration=5.0)
|
||||
|
||||
assert abs(clip.duration - 5.0) < 0.01
|
||||
assert clip.trim_config is not None
|
||||
assert abs(clip.trim_config.duration - 5.0) < 0.01
|
||||
|
||||
|
||||
class TestGetVoiceAudioDuration:
|
||||
"""Test voice audio duration probing."""
|
||||
|
||||
def test_no_voiceover_path_returns_zero(self):
|
||||
"""No voiceover path → return 0."""
|
||||
with patch.object(UnifiedRenderService, "__init__", lambda self, **kwargs: None):
|
||||
service = UnifiedRenderService.__new__(UnifiedRenderService)
|
||||
service.voiceover_audio_path = None
|
||||
|
||||
assert service._get_voice_audio_duration() == 0.0
|
||||
|
||||
def test_nonexistent_file_returns_zero(self):
|
||||
"""Nonexistent file → return 0."""
|
||||
with patch.object(UnifiedRenderService, "__init__", lambda self, **kwargs: None):
|
||||
service = UnifiedRenderService.__new__(UnifiedRenderService)
|
||||
service.voiceover_audio_path = "/nonexistent/path.mp3"
|
||||
|
||||
assert service._get_voice_audio_duration() == 0.0
|
||||
|
||||
@patch("video_processing.unified_render_service.probe_duration")
|
||||
@patch("video_processing.unified_render_service.Path.exists", return_value=True)
|
||||
@patch("video_processing.unified_render_service.Path.stat")
|
||||
def test_probes_duration_from_file(self, mock_stat, mock_exists, mock_probe):
|
||||
"""Valid file → probe duration."""
|
||||
mock_stat.return_value.st_size = 1000 # Non-empty file
|
||||
mock_probe.return_value = 42.5
|
||||
|
||||
with patch.object(UnifiedRenderService, "__init__", lambda self, **kwargs: None):
|
||||
service = UnifiedRenderService.__new__(UnifiedRenderService)
|
||||
service.voiceover_audio_path = "/tmp/voice.mp3"
|
||||
|
||||
assert service._get_voice_audio_duration() == 42.5
|
||||
Reference in New Issue
Block a user