From add93bb9de4b454f14785f5fb714463375a14b5e Mon Sep 17 00:00:00 2001 From: saas-backend-agent Date: Mon, 7 Sep 2026 19:46:43 +0800 Subject: [PATCH 1/2] =?UTF-8?q?feat(dedup):=20=E5=A2=9E=E5=8A=A0=E6=96=87?= =?UTF-8?q?=E6=A1=88+=E7=BB=93=E6=9E=84=E7=BB=B4=E5=BA=A6=E6=9F=A5?= =?UTF-8?q?=E9=87=8D=20(Issue=20#P2-=E5=90=8E=E7=AB=AF3)?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit - 新增 compute_text_similarity: 字符级 Jaccard 相似度比对配音文本 - 新增 compute_structure_similarity: 片段序列相似度(数量+类型+时长分布) - 多维度融合公式: duplicate_rate = visual*0.5 + text*0.25 + structure*0.25 - 文案/结构数据缺失时降级为纯视觉维度 - 21 个新测试覆盖文本+结构+权重常量 --- apps/worker/video_processing/dedup.py | 184 +++++++++++++++++++++-- tests/unit/test_dedup_enhanced.py | 206 ++++++++++++++++++++++++++ 2 files changed, 380 insertions(+), 10 deletions(-) create mode 100644 tests/unit/test_dedup_enhanced.py diff --git a/apps/worker/video_processing/dedup.py b/apps/worker/video_processing/dedup.py index 7e6ebdf06..9e62e568a 100755 --- a/apps/worker/video_processing/dedup.py +++ b/apps/worker/video_processing/dedup.py @@ -7,6 +7,7 @@ import hashlib import logging import math import os +import re import statistics import tempfile from dataclasses import dataclass, field @@ -56,8 +57,13 @@ MAX_GAP = 2 # 允许的最大间隙帧数 NEIGHBOR_WINDOW = 1 # 分片时序对齐:允许 ±1 邻接偏移(1s 密集采样下即 ±1s,缓解切点不一致) # ── 融合判定常量 ──────────────────────────────────────────────── -PHASH_WEIGHT = 0.7 # pHash 权重 -HISTOGRAM_WEIGHT = 0.3 # 直方图权重 +PHASH_WEIGHT = 0.7 # pHash 权重(视觉内部) +HISTOGRAM_WEIGHT = 0.3 # 直方图权重(视觉内部) + +# ── 多维度查重融合权重(Issue #P2-后端3) ──────────────────────── +VISUAL_WEIGHT = 0.5 # 视觉相似度权重(pHash+直方图) +TEXT_WEIGHT = 0.25 # 文案相似度权重(配音文本) +STRUCTURE_WEIGHT = 0.25 # 结构相似度权重(片段序列) MATCH_RATIO_THRESHOLD = 0.7 # 全片重复(is_duplicate)至少 70% 帧匹配 PARTIAL_COVERAGE_THRESHOLD = 0.5 # 局部复用覆盖率 >=50% 也判全片重复 DUPLICATE_THRESHOLD = 0.70 # 融合后相似度阈值 @@ -1045,6 +1051,111 @@ class VideoDeduplicator: logger.info("check_batch_duplicate no match (batch=%s): best_fusion=%.3f", batch_id, best_score) return None + +# ── 文案 & 结构维度查重(Issue #P2-后端3) ──────────────────────── + + +def _normalize_text(text: str) -> str: + """文本标准化:去空白、转小写、去标点。""" + if not text: + return "" + # 去空白字符 + text = re.sub(r"\s+", "", text) + # 转小写 + text = text.lower() + # 去标点(只保留中文、字母、数字) + text = re.sub(r"[^\w\u4e00-\u9fff]", "", text) + return text + + +def compute_text_similarity(text1: str, text2: str) -> float: + """计算两段文本的相似度(0~1)。 + + 使用字符级 Jaccard 相似度:交集 / 并集。 + 适合短文本(配音脚本)的相似度比对。 + + Args: + text1: 第一段文本 + text2: 第二段文本 + + Returns: + 0~1 之间的相似度 + """ + t1 = _normalize_text(text1) + t2 = _normalize_text(text2) + + if not t1 and not t2: + return 1.0 # 都为空,视为完全相同 + if not t1 or not t2: + return 0.0 # 一个为空,完全不同 + + # 字符级 Jaccard + set1 = set(t1) + set2 = set(t2) + intersection = set1 & set2 + union = set1 | set2 + + if not union: + return 0.0 + + return len(intersection) / len(union) + + +def compute_structure_similarity(clips1: list[dict], clips2: list[dict]) -> float: + """计算两个视频的结构相似度(0~1)。 + + 结构维度包括: + 1. 片段数差异(数量越接近越相似) + 2. 片段类型序列(相同位置的片段类型是否一致) + 3. 时长分布(各片段时长占比是否相似) + + Args: + clips1: 第一个视频的片段列表,每项包含 {clip_type, duration} + clips2: 第二个视频的片段列表 + + Returns: + 0~1 之间的相似度 + """ + if not clips1 and not clips2: + return 1.0 + if not clips1 or not clips2: + return 0.0 + + # 1. 片段数相似度(数量差异越大越低) + n1, n2 = len(clips1), len(clips2) + count_sim = min(n1, n2) / max(n1, n2) + + # 2. 类型序列相似度(逐位比较,相同位置类型是否一致) + min_len = min(n1, n2) + type_matches = sum(1 for i in range(min_len) if clips1[i].get("clip_type") == clips2[i].get("clip_type")) + type_sim = type_matches / min_len if min_len > 0 else 0.0 + + # 3. 时长分布相似度(归一化后比较分布) + total1 = sum(c.get("duration", 0) for c in clips1) + total2 = sum(c.get("duration", 0) for c in clips2) + + if total1 > 0 and total2 > 0: + # 归一化为占比 + dist1 = [c.get("duration", 0) / total1 for c in clips1] + dist2 = [c.get("duration", 0) / total2 for c in clips2] + + # 比较前 min_len 个片段的占比差异(L1 距离转相似度) + l1_dist = sum(abs(dist1[i] - dist2[i]) for i in range(min_len)) + # 加上多出的片段占比 + if n1 > n2: + l1_dist += sum(dist1[i] for i in range(n2, n1)) + elif n2 > n1: + l1_dist += sum(dist2[i] for i in range(n1, n2)) + + # L1 距离范围 [0, 2],转为相似度 [0, 1] + duration_sim = 1.0 - (l1_dist / 2.0) + else: + duration_sim = 0.0 + + # 三维度加权:数量 0.3 + 类型 0.4 + 时长 0.3 + return count_sim * 0.3 + type_sim * 0.4 + duration_sim * 0.3 + + def compute_duplicate_rate( self, fingerprint: VideoFingerprint, @@ -1057,12 +1168,11 @@ class VideoDeduplicator: ) -> dict: """计算当前视频与已有视频的查重率百分比。 - 新公式(双指标加权): - - frame_match_rate = 汉明距离 < PHASH_THRESHOLD 的帧数 / 总帧数 - - temporal_coverage_rate = 连续匹配片段总时长 / 视频总时长 - - duplicate_rate = (frame_match_rate * 0.4 + temporal_coverage_rate * 0.6) * 100 - - visual_similarity = 0.7 * phash_sim + 0.3 * hist_sim(归一化到 0~1) + 多维度融合公式(Issue #P2-后端3): + - visual_similarity = 0.7 * phash_sim + 0.3 * hist_sim(视觉维度) + - text_similarity = 文案 Jaccard 相似度(文案维度) + - structure_similarity = 片段序列相似度(结构维度) + - duplicate_rate = (visual*0.5 + text*0.25 + structure*0.25) * 100 对每个匹配视频都算,取最高 duplicate_rate。 @@ -1093,6 +1203,28 @@ class VideoDeduplicator: match_count = 0 evaluated = 0 + # Issue #P2-后端3: 加载当前视频的文案+结构数据 + from packages.adapters.sqlalchemy_impl.models import EditPlanClipModel, GeneratedVideoModel + + current_video_obj = session.query(GeneratedVideoModel).filter(GeneratedVideoModel.id == current_video_id).first() if current_video_id else None + current_plan_id = getattr(current_video_obj, "edit_plan_id", "") or "" + current_clips_data = [] + current_text_content = "" + + if current_plan_id: + current_clips = ( + session.query(EditPlanClipModel) + .filter(EditPlanClipModel.plan_id == current_plan_id) + .order_by(EditPlanClipModel.order) + .all() + ) + current_clips_data = [ + {"clip_type": c.clip_type, "duration": c.duration} + for c in current_clips + ] + # 拼接所有片段的文本内容 + current_text_content = " ".join(c.text_content for c in current_clips if c.text_content) + for existing in existing_videos: if current_video_id and existing.id == current_video_id: continue @@ -1160,8 +1292,40 @@ class VideoDeduplicator: # Issue #1702: 去掉 "frame_match_rate<0.3 整条跳过" 硬门槛—— # 局部片段复用帧比例天然低;coverage 为主指标,0 匹配自然得 0 分。 - # duplicate_rate = 0.4 * frame_match_rate + 0.6 * temporal_coverage - dup_rate = (min(ev["frame_match_rate"], 1.0) * 0.4 + ev["temporal_coverage"] * 0.6) * 100 + # 视觉维度:0.4 * frame_match_rate + 0.6 * temporal_coverage + visual_sim = min(ev["frame_match_rate"], 1.0) * 0.4 + ev["temporal_coverage"] * 0.6 + + # Issue #P2-后端3: 文案+结构维度 + existing_plan_id = getattr(existing, "edit_plan_id", "") or "" + existing_clips_data = [] + existing_text_content = "" + + if existing_plan_id: + existing_clips = ( + session.query(EditPlanClipModel) + .filter(EditPlanClipModel.plan_id == existing_plan_id) + .order_by(EditPlanClipModel.order) + .all() + ) + existing_clips_data = [ + {"clip_type": c.clip_type, "duration": c.duration} + for c in existing_clips + ] + existing_text_content = " ".join(c.text_content for c in existing_clips if c.text_content) + + # 计算文案相似度(有文案才算) + text_sim = compute_text_similarity(current_text_content, existing_text_content) if (current_text_content and existing_text_content) else 0.0 + + # 计算结构相似度(有片段才算) + structure_sim = compute_structure_similarity(current_clips_data, existing_clips_data) if (current_clips_data and existing_clips_data) else 0.0 + + # 多维度融合:visual*0.5 + text*0.25 + structure*0.25 + # 如果文案/结构数据缺失,只用视觉维度(visual 权重提升到 1.0) + if current_text_content and existing_text_content and current_clips_data and existing_clips_data: + dup_rate = (visual_sim * VISUAL_WEIGHT + text_sim * TEXT_WEIGHT + structure_sim * STRUCTURE_WEIGHT) * 100 + else: + # 降级:只有视觉维度 + dup_rate = visual_sim * 100 # 全片重复计数与 check_duplicate 判定口径一致 if ev["fusion"] >= DUPLICATE_THRESHOLD and ( diff --git a/tests/unit/test_dedup_enhanced.py b/tests/unit/test_dedup_enhanced.py new file mode 100644 index 000000000..bec63d84c --- /dev/null +++ b/tests/unit/test_dedup_enhanced.py @@ -0,0 +1,206 @@ +"""Tests for enhanced dedup: text + structure dimensions (Issue #P2-后端3).""" + +from __future__ import annotations + +import sys +from unittest.mock import MagicMock + +# --------------------------------------------------------------------------- +# Mock heavy deps before importing dedup module +# --------------------------------------------------------------------------- +_ORIGINAL_MODULES = dict(sys.modules) +_MOCKED_MODULE_NAMES: list[str] = [] + + +def _mock_if_absent(name: str, mock_obj=None): + """仅在模块不在 sys.modules 中时注入 mock,并记录以便清理。""" + if name not in sys.modules: + sys.modules[name] = mock_obj if mock_obj is not None else MagicMock() + _MOCKED_MODULE_NAMES.append(name) + + +# Mock heavy deps +_mock_if_absent("ffmpeg") +_mock_if_absent("ffmpeg.utils") +_mock_if_absent("worker_app.celery_app") +_mock_if_absent("worker_app.db") +_mock_if_absent("packages.adapters.sqlalchemy_impl.generated_video_repository") +_mock_if_absent("packages.adapters.sqlalchemy_impl.models") +_mock_if_absent("packages.shared.storage") + +# Mock cv2 and numpy if not available +try: + import cv2 as _cv2 + if not isinstance(_cv2, MagicMock): + _HAS_CV2 = True + else: + _HAS_CV2 = False +except ImportError: + _HAS_CV2 = False + _mock_if_absent("cv2") + _mock_if_absent("numpy") + +import pytest + +from apps.worker.video_processing.dedup import ( + STRUCTURE_WEIGHT, + TEXT_WEIGHT, + VISUAL_WEIGHT, + compute_structure_similarity, + compute_text_similarity, +) + + +class TestTextSimilarity: + """Tests for compute_text_similarity.""" + + def test_identical_texts(self): + """相同文本返回 1.0。""" + assert compute_text_similarity("你好世界", "你好世界") == 1.0 + + def test_empty_texts(self): + """都为空返回 1.0。""" + assert compute_text_similarity("", "") == 1.0 + + def test_one_empty(self): + """一个为空返回 0.0。""" + assert compute_text_similarity("你好", "") == 0.0 + assert compute_text_similarity("", "你好") == 0.0 + + def test_completely_different(self): + """完全不同文本返回低相似度。""" + sim = compute_text_similarity("你好世界", "abcdefgh") + assert sim < 0.3 + + def test_partial_overlap(self): + """部分重叠文本返回中等相似度。""" + sim = compute_text_similarity("今天天气真好", "今天天气不错") + assert 0.3 < sim < 0.9 + + def test_case_insensitive(self): + """英文大小写不敏感。""" + sim = compute_text_similarity("Hello World", "hello world") + assert sim == 1.0 + + def test_whitespace_ignored(self): + """空白字符被忽略。""" + sim = compute_text_similarity("你好 世界", "你好世界") + assert sim == 1.0 + + def test_punctuation_ignored(self): + """标点符号被忽略。""" + sim = compute_text_similarity("你好,世界!", "你好世界") + assert sim == 1.0 + + def test_long_texts(self): + """长文本也能计算。""" + t1 = "这是一段很长的配音文本,用于测试文案查重功能" + t2 = "这是一段较长的配音文字,用于测试文案去重功能" + sim = compute_text_similarity(t1, t2) + assert 0.0 <= sim <= 1.0 + + +class TestStructureSimilarity: + """Tests for compute_structure_similarity.""" + + def test_identical_structures(self): + """完全相同结构返回 1.0。""" + clips = [ + {"clip_type": "video", "duration": 5.0}, + {"clip_type": "title", "duration": 2.0}, + {"clip_type": "video", "duration": 8.0}, + ] + assert compute_structure_similarity(clips, clips) == 1.0 + + def test_empty_clips(self): + """都为空返回 1.0。""" + assert compute_structure_similarity([], []) == 1.0 + + def test_one_empty(self): + """一个为空返回 0.0。""" + clips = [{"clip_type": "video", "duration": 5.0}] + assert compute_structure_similarity(clips, []) == 0.0 + assert compute_structure_similarity([], clips) == 0.0 + + def test_different_count(self): + """片段数不同,相似度降低。""" + clips1 = [ + {"clip_type": "video", "duration": 5.0}, + {"clip_type": "title", "duration": 2.0}, + ] + clips2 = [ + {"clip_type": "video", "duration": 5.0}, + {"clip_type": "title", "duration": 2.0}, + {"clip_type": "video", "duration": 3.0}, + {"clip_type": "title", "duration": 1.0}, + ] + sim = compute_structure_similarity(clips1, clips2) + assert 0.0 < sim < 0.8 + + def test_different_types(self): + """片段类型不同,类型相似度低。""" + clips1 = [ + {"clip_type": "video", "duration": 5.0}, + {"clip_type": "video", "duration": 3.0}, + ] + clips2 = [ + {"clip_type": "title", "duration": 5.0}, + {"clip_type": "title", "duration": 3.0}, + ] + sim = compute_structure_similarity(clips1, clips2) + assert sim <= 0.6 # 类型全部不同,但数量和时长相同贡献 0.6 + + def test_different_duration_distribution(self): + """时长分布不同,时长相似度低。""" + clips1 = [ + {"clip_type": "video", "duration": 10.0}, # 占比 80% + {"clip_type": "title", "duration": 2.5}, # 占比 20% + ] + clips2 = [ + {"clip_type": "video", "duration": 2.0}, # 占比 20% + {"clip_type": "title", "duration": 8.0}, # 占比 80% + ] + sim = compute_structure_similarity(clips1, clips2) + assert 0.7 < sim < 0.9 # 类型相同但时长分布不同,sim=0.82 + + def test_similar_structure(self): + """相似结构返回较高相似度。""" + clips1 = [ + {"clip_type": "video", "duration": 5.0}, + {"clip_type": "title", "duration": 2.0}, + {"clip_type": "video", "duration": 8.0}, + ] + clips2 = [ + {"clip_type": "video", "duration": 5.5}, + {"clip_type": "title", "duration": 2.2}, + {"clip_type": "video", "duration": 7.5}, + ] + sim = compute_structure_similarity(clips1, clips2) + assert sim > 0.8 + + def test_single_clip(self): + """单片段也能计算。""" + clips1 = [{"clip_type": "video", "duration": 10.0}] + clips2 = [{"clip_type": "video", "duration": 12.0}] + sim = compute_structure_similarity(clips1, clips2) + assert sim > 0.5 # 类型相同,数量相同,只是时长不同 + + +class TestDimensionWeights: + """Tests for dimension weight constants.""" + + def test_weights_sum_to_one(self): + """多维度权重之和为 1.0。""" + assert abs(VISUAL_WEIGHT + TEXT_WEIGHT + STRUCTURE_WEIGHT - 1.0) < 1e-9 + + def test_visual_weight_is_half(self): + """视觉权重为 0.5。""" + assert VISUAL_WEIGHT == 0.5 + + def test_text_weight_is_quarter(self): + """文案权重为 0.25。""" + assert TEXT_WEIGHT == 0.25 + + def test_structure_weight_is_quarter(self): + """结构权重为 0.25。""" + assert STRUCTURE_WEIGHT == 0.25 -- 2.54.0 From 07aa03da244e2fefe54a678b8f6babcb56a6e2a7 Mon Sep 17 00:00:00 2001 From: CI Bot Date: Mon, 7 Sep 2026 11:49:16 +0000 Subject: [PATCH 2/2] style: auto-format with black + isort + prettier [skip ci-format-check] --- apps/worker/video_processing/dedup.py | 39 ++++++++++++++++----------- tests/unit/test_dedup_enhanced.py | 7 ++--- 2 files changed, 27 insertions(+), 19 deletions(-) diff --git a/apps/worker/video_processing/dedup.py b/apps/worker/video_processing/dedup.py index 9e62e568a..817f9034b 100755 --- a/apps/worker/video_processing/dedup.py +++ b/apps/worker/video_processing/dedup.py @@ -1155,7 +1155,6 @@ def compute_structure_similarity(clips1: list[dict], clips2: list[dict]) -> floa # 三维度加权:数量 0.3 + 类型 0.4 + 时长 0.3 return count_sim * 0.3 + type_sim * 0.4 + duration_sim * 0.3 - def compute_duplicate_rate( self, fingerprint: VideoFingerprint, @@ -1205,12 +1204,16 @@ def compute_structure_similarity(clips1: list[dict], clips2: list[dict]) -> floa # Issue #P2-后端3: 加载当前视频的文案+结构数据 from packages.adapters.sqlalchemy_impl.models import EditPlanClipModel, GeneratedVideoModel - - current_video_obj = session.query(GeneratedVideoModel).filter(GeneratedVideoModel.id == current_video_id).first() if current_video_id else None + + current_video_obj = ( + session.query(GeneratedVideoModel).filter(GeneratedVideoModel.id == current_video_id).first() + if current_video_id + else None + ) current_plan_id = getattr(current_video_obj, "edit_plan_id", "") or "" current_clips_data = [] current_text_content = "" - + if current_plan_id: current_clips = ( session.query(EditPlanClipModel) @@ -1218,10 +1221,7 @@ def compute_structure_similarity(clips1: list[dict], clips2: list[dict]) -> floa .order_by(EditPlanClipModel.order) .all() ) - current_clips_data = [ - {"clip_type": c.clip_type, "duration": c.duration} - for c in current_clips - ] + current_clips_data = [{"clip_type": c.clip_type, "duration": c.duration} for c in current_clips] # 拼接所有片段的文本内容 current_text_content = " ".join(c.text_content for c in current_clips if c.text_content) @@ -1299,7 +1299,7 @@ def compute_structure_similarity(clips1: list[dict], clips2: list[dict]) -> floa existing_plan_id = getattr(existing, "edit_plan_id", "") or "" existing_clips_data = [] existing_text_content = "" - + if existing_plan_id: existing_clips = ( session.query(EditPlanClipModel) @@ -1307,22 +1307,29 @@ def compute_structure_similarity(clips1: list[dict], clips2: list[dict]) -> floa .order_by(EditPlanClipModel.order) .all() ) - existing_clips_data = [ - {"clip_type": c.clip_type, "duration": c.duration} - for c in existing_clips - ] + existing_clips_data = [{"clip_type": c.clip_type, "duration": c.duration} for c in existing_clips] existing_text_content = " ".join(c.text_content for c in existing_clips if c.text_content) # 计算文案相似度(有文案才算) - text_sim = compute_text_similarity(current_text_content, existing_text_content) if (current_text_content and existing_text_content) else 0.0 + text_sim = ( + compute_text_similarity(current_text_content, existing_text_content) + if (current_text_content and existing_text_content) + else 0.0 + ) # 计算结构相似度(有片段才算) - structure_sim = compute_structure_similarity(current_clips_data, existing_clips_data) if (current_clips_data and existing_clips_data) else 0.0 + structure_sim = ( + compute_structure_similarity(current_clips_data, existing_clips_data) + if (current_clips_data and existing_clips_data) + else 0.0 + ) # 多维度融合:visual*0.5 + text*0.25 + structure*0.25 # 如果文案/结构数据缺失,只用视觉维度(visual 权重提升到 1.0) if current_text_content and existing_text_content and current_clips_data and existing_clips_data: - dup_rate = (visual_sim * VISUAL_WEIGHT + text_sim * TEXT_WEIGHT + structure_sim * STRUCTURE_WEIGHT) * 100 + dup_rate = ( + visual_sim * VISUAL_WEIGHT + text_sim * TEXT_WEIGHT + structure_sim * STRUCTURE_WEIGHT + ) * 100 else: # 降级:只有视觉维度 dup_rate = visual_sim * 100 diff --git a/tests/unit/test_dedup_enhanced.py b/tests/unit/test_dedup_enhanced.py index bec63d84c..e23c603d6 100644 --- a/tests/unit/test_dedup_enhanced.py +++ b/tests/unit/test_dedup_enhanced.py @@ -31,6 +31,7 @@ _mock_if_absent("packages.shared.storage") # Mock cv2 and numpy if not available try: import cv2 as _cv2 + if not isinstance(_cv2, MagicMock): _HAS_CV2 = True else: @@ -154,11 +155,11 @@ class TestStructureSimilarity: """时长分布不同,时长相似度低。""" clips1 = [ {"clip_type": "video", "duration": 10.0}, # 占比 80% - {"clip_type": "title", "duration": 2.5}, # 占比 20% + {"clip_type": "title", "duration": 2.5}, # 占比 20% ] clips2 = [ - {"clip_type": "video", "duration": 2.0}, # 占比 20% - {"clip_type": "title", "duration": 8.0}, # 占比 80% + {"clip_type": "video", "duration": 2.0}, # 占比 20% + {"clip_type": "title", "duration": 8.0}, # 占比 80% ] sim = compute_structure_similarity(clips1, clips2) assert 0.7 < sim < 0.9 # 类型相同但时长分布不同,sim=0.82 -- 2.54.0