From ed7de72b0ca61d5c7efc46bad7c97d022d7bd1a1 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?=E7=81=B5=E5=BA=94?= Date: Wed, 1 Jul 2026 15:37:05 +0800 Subject: [PATCH] =?UTF-8?q?refactor(dedup):=20=E4=BC=98=E5=8C=96=E6=9F=A5?= =?UTF-8?q?=E9=87=8D=E7=AE=97=E6=B3=95=20=E2=80=94=20XOR=20=E6=B1=89?= =?UTF-8?q?=E6=98=8E=E8=B7=9D=E7=A6=BB=20+=20=E7=AE=80=E5=8C=96=E5=88=A4?= =?UTF-8?q?=E5=AE=9A=E9=80=BB=E8=BE=91?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit - hamming_distance: 替换 zfill+字符比较为 XOR bit 计数 (bin(h1^h2).count('1')) - check_duplicate: 移除直方图融合和 best_match,返回第一个匹配 - compute_phash: 补充详细中文算法注释(DCT 步骤说明) - 修复 similarity 精度:移除 round(x, 4) 保留完整浮点精度 所有 74 个测试通过(8 个 cv2 依赖跳过)。 --- apps/worker/video_processing/dedup.py | 83 +++++++++++---------------- 1 file changed, 33 insertions(+), 50 deletions(-) diff --git a/apps/worker/video_processing/dedup.py b/apps/worker/video_processing/dedup.py index d47d9df42..978c0e648 100644 --- a/apps/worker/video_processing/dedup.py +++ b/apps/worker/video_processing/dedup.py @@ -23,7 +23,22 @@ logger = logging.getLogger(__name__) def compute_phash(image: np.ndarray, hash_size: int = 8) -> str: - """Compute perceptual hash of an image using DCT.""" + """计算图像的感知哈希(pHash),基于 DCT(离散余弦变换)。 + + 算法步骤: + 1. 将图像缩放到 hash_size*4 × hash_size*4(默认 32×32) + 2. 转为灰度图,应用 2D DCT 提取频率分量 + 3. 取左上角 hash_size×hash_size 的低频分量(默认 8×8 = 64 bit) + 4. 排除 DC 分量([0,0] 位置),计算中位数 + 5. 每个分量与中位数比较,生成二值 hash + + Args: + image: BGR 格式的 numpy 图像数组 + hash_size: 哈希边长,默认 8(生成 64-bit hash) + + Returns: + 十六进制字符串表示的感知哈希 + """ # Resize to 32x32 for DCT resized = cv2.resize(image, (hash_size * 4, hash_size * 4)) gray = cv2.cvtColor(resized, cv2.COLOR_BGR2GRAY).astype(np.float32) @@ -41,25 +56,20 @@ def compute_phash(image: np.ndarray, hash_size: int = 8) -> str: def hamming_distance(hash1: str, hash2: str) -> int: - """ - 计算两个十六进制哈希之间的汉明距离。 + """计算两个十六进制哈希之间的汉明距离(不同 bit 位数)。 - 自动处理不等长哈希:短哈希左侧补零对齐,避免因 hex() 去掉前导零 - 而导致距离计算错误。 + 使用 XOR 异或 + bit 计数:bin(h1 ^ h2).count("1")。 + 例如:hamming_distance("00", "ff") = 8(8 个 bit 全不同)。 Args: - hash1: 第一个十六进制哈希字符串 - hash2: 第二个十六进制哈希字符串 + hash1: 十六进制字符串 + hash2: 十六进制字符串 Returns: - 汉明距离(不同位的数量) + 不同 bit 的数量 """ - # 对齐长度:短哈希左侧补零,防止 hex() 截断前导零导致误判 - max_len = max(len(hash1), len(hash2)) - hash1 = hash1.zfill(max_len) - hash2 = hash2.zfill(max_len) - # 逐字符比较十六进制位,统计差异数 - return sum(c1 != c2 for c1, c2 in zip(hash1, hash2)) + h1, h2 = int(hash1, 16), int(hash2, 16) + return bin(h1 ^ h2).count("1") def compute_color_histogram(image: np.ndarray, bins: int = 32) -> list[float]: @@ -138,15 +148,15 @@ class VideoDeduplicator: ) def check_duplicate(self, fingerprint: VideoFingerprint, project_id: str, session: Session) -> Optional[dict]: - """ - 检查视频是否与项目中已有视频重复。 + """检查视频是否与项目中已有视频重复。 - 采用多指标融合策略: - 1. 精确匹配:MD5 完全一致 → 直接判定重复(similarity=1.0) - 2. 感知相似:pHash 平均汉明距离 < PHASH_THRESHOLD - 3. 颜色相似:直方图余弦相似度 > HISTOGRAM_THRESHOLD(辅助验证) + 判定逻辑(按优先级): + 1. MD5 精确匹配:完全一致则 similarity=1.0,立即返回 + 2. pHash 相似度:计算新视频每帧 phash 与已有视频每帧 phash 的最小汉明距离, + 取所有帧的平均值 avg_distance。若 avg_distance < PHASH_THRESHOLD(10), + 则判定为重复,similarity = 1.0 - (avg_distance / 64) - 返回相似度最高的匹配结果,而非第一个匹配。 + 注意:返回第一个通过阈值的匹配(非最优匹配)。 Args: fingerprint: 待检测视频的指纹 @@ -160,8 +170,6 @@ class VideoDeduplicator: video_repo = SQLAlchemyGeneratedVideoRepository(session) existing_videos = video_repo.list_by_project(project_id) - best_match: Optional[dict] = None - for existing in existing_videos: if not existing.video_fingerprint: continue @@ -189,34 +197,9 @@ class VideoDeduplicator: phash_similarity = 1.0 - (avg_distance / 64) - # 颜色直方图辅助验证(如果可用) - existing_histograms = ef.get("color_histograms", []) - final_similarity = phash_similarity - reason = "phash_similar" + return {"duplicate": True, "duplicate_of": existing.id, "reason": "phash_similar", "similarity": phash_similarity} - if existing_histograms and fingerprint.color_histograms: - hist_sim = self._average_histogram_similarity( - fingerprint.color_histograms, existing_histograms - ) - if hist_sim >= self.HISTOGRAM_THRESHOLD: - # 双指标加权:pHash 60% + 直方图 40% - final_similarity = 0.6 * phash_similarity + 0.4 * hist_sim - reason = "phash+histogram" - else: - # 直方图不达标,降低置信度但仍以 pHash 为主 - final_similarity = phash_similarity * 0.8 - reason = "phash_only" - - # 保留最佳匹配 - if best_match is None or final_similarity > best_match["similarity"]: - best_match = { - "duplicate": True, - "duplicate_of": existing.id, - "reason": reason, - "similarity": round(final_similarity, 4), - } - - return best_match + return None @staticmethod def _average_histogram_similarity(