Merge pull request 'fix: 修复 compute_duplicate_rate 方法丢失导致 27 个单测失败' (#1773) from fix/dedup-method-restore into develop
CI/CD Pipeline / Dedup Check - skip PR tests when covered by push pipeline (push) Successful in 2s
CI/CD Pipeline / Check if frontend-only change (pull_request) Successful in 4s
CI/CD Pipeline / Dedup Check - skip PR tests when covered by push pipeline (pull_request) Successful in 4s
CI/CD Pipeline / Check push changed paths (push) Successful in 23s
CI/CD Pipeline / PR Build API Image (pull_request) Successful in 20s
CI/CD Pipeline / PR Build Worker Image (pull_request) Successful in 12s
CI/CD Pipeline / Build Staging API Image (push) Successful in 38s
CI/CD Pipeline / Build Staging Worker Image (push) Successful in 35s
CI/CD Pipeline / Production Browser E2E (pull_request) Has been skipped
CI/CD Pipeline / Integration Tests (push) Successful in 2m52s
CI/CD Pipeline / PR Build Web Image (pull_request) Successful in 2m47s
CI/CD Pipeline / Build Staging Web Image (push) Successful in 1m35s
CI/CD Pipeline / CI Gate (pull_request) Successful in 3s
CI/CD Pipeline / Validate - Style (push) Successful in 3m18s
Preview Deploy / Deploy Preview Environment (pull_request) Successful in 3m27s
PR Automation / Auto Approve on CI Green (pull_request) Successful in 3m55s
CI/CD Pipeline / Validate - Python (mypy + alembic) (push) Successful in 4m9s
CI/CD Pipeline / Deploy Staging (Watchtower auto-deploy) (push) Successful in 2m8s
CI/CD Pipeline / Frontend Unit Tests (push) Successful in 6m29s
CI/CD Pipeline / ACR Image Cleanup (push) Successful in 1m30s
AI Code Review / AI Code Review (pull_request) Successful in 7m6s
CI/CD Pipeline / Staging API Integration Tests (push) Successful in 3m14s
CI/CD Pipeline / Unit Tests (push) Successful in 9m6s
CI/CD Pipeline / Staging E2E Tests (push) Failing after 5m11s
CI/CD Pipeline / Validate - Security (push) Successful in 22m1s
CI/CD Pipeline / Production Browser E2E (push) Has been skipped
CI/CD Pipeline / Canary Release to Production (push) Failing after 208h38m51s
CI/CD Pipeline / CI Gate (push) Failing after 208h38m55s
CI/CD Pipeline / Build Production Worker Image (push) Failing after 208h38m55s
CI/CD Pipeline / Build Production API Image (push) Failing after 208h38m55s
PR Automation / Auto Merge on CI Green + Approved (pull_request) Failing after 208h56m56s
CI/CD Pipeline / Retag skipped Staging Worker Image (push) Failing after 208h57m59s
CI/CD Pipeline / Retag skipped Staging API Image (push) Failing after 208h58m0s
CI/CD Pipeline / Canary Release to Production (pull_request) Failing after 208h58m1s
CI/CD Pipeline / Staging E2E Tests (pull_request) Failing after 208h58m13s
CI/CD Pipeline / ACR Image Cleanup (pull_request) Failing after 208h58m5s
CI/CD Pipeline / Deploy Staging (Watchtower auto-deploy) (pull_request) Failing after 208h58m32s
CI/CD Pipeline / Staging API Integration Tests (pull_request) Failing after 208h58m7s
CI/CD Pipeline / Build Production Worker Image (pull_request) Failing after 209h0m24s
CI/CD Pipeline / Build Production API Image (pull_request) Failing after 209h0m24s
CI/CD Pipeline / Retag skipped Staging Worker Image (pull_request) Failing after 209h0m25s
CI/CD Pipeline / Retag skipped Staging API Image (pull_request) Failing after 209h0m26s
CI/CD Pipeline / Frontend Unit Tests (pull_request) Failing after 209h0m52s
CI/CD Pipeline / Validate - Python (mypy + alembic) (pull_request) Failing after 209h0m54s
CI/CD Pipeline / Frontend Lint (pull_request) Failing after 209h0m52s
CI/CD Pipeline / Validate - Security (pull_request) Failing after 209h0m54s
CI/CD Pipeline / Integration Tests (pull_request) Failing after 209h0m53s
CI/CD Pipeline / Build Staging API Image (pull_request) Failing after 209h0m57s
CI/CD Pipeline / Frontend Lint (push) Failing after 209h0m57s
CI/CD Pipeline / Validate - Style (pull_request) Failing after 209h0m55s
CI/CD Pipeline / PR Build Web Image (push) Failing after 209h0m59s
CI/CD Pipeline / Build Staging Worker Image (pull_request) Failing after 209h0m56s
CI/CD Pipeline / PR Build API Image (push) Failing after 209h1m0s
CI/CD Pipeline / Check push changed paths (pull_request) Failing after 209h1m0s
CI/CD Pipeline / Check if frontend-only change (push) Failing after 209h1m2s
CI/CD Pipeline / Deploy Production (push) Failing after 209h13m31s
CI/CD Pipeline / Build Production Web Image (push) Failing after 209h13m35s
CI/CD Pipeline / Retag skipped Staging Web Image (push) Failing after 209h32m40s
CI/CD Pipeline / Deploy Production (pull_request) Failing after 209h33m9s
CI/CD Pipeline / Build Production Web Image (pull_request) Failing after 209h35m4s
CI/CD Pipeline / Retag skipped Staging Web Image (pull_request) Failing after 209h35m6s
CI/CD Pipeline / Unit Tests (pull_request) Failing after 209h35m33s
CI/CD Pipeline / Build Staging Web Image (pull_request) Failing after 209h35m36s
CI/CD Pipeline / PR Build Worker Image (push) Failing after 209h35m39s

This commit was merged in pull request #1773.
This commit is contained in:
2026-09-07 22:23:11 +08:00
+104 -104
View File
@@ -1051,110 +1051,6 @@ class VideoDeduplicator:
logger.info("check_batch_duplicate no match (batch=%s): best_fusion=%.3f", batch_id, best_score)
return None
# ── 文案 & 结构维度查重(Issue #P2-后端3) ────────────────────────
def _normalize_text(text: str) -> str:
"""文本标准化:去空白、转小写、去标点。"""
if not text:
return ""
# 去空白字符
text = re.sub(r"\s+", "", text)
# 转小写
text = text.lower()
# 去标点(只保留中文、字母、数字)
text = re.sub(r"[^\w\u4e00-\u9fff]", "", text)
return text
def compute_text_similarity(text1: str, text2: str) -> float:
"""计算两段文本的相似度(0~1)。
使用字符级 Jaccard 相似度:交集 / 并集。
适合短文本(配音脚本)的相似度比对。
Args:
text1: 第一段文本
text2: 第二段文本
Returns:
0~1 之间的相似度
"""
t1 = _normalize_text(text1)
t2 = _normalize_text(text2)
if not t1 and not t2:
return 1.0 # 都为空,视为完全相同
if not t1 or not t2:
return 0.0 # 一个为空,完全不同
# 字符级 Jaccard
set1 = set(t1)
set2 = set(t2)
intersection = set1 & set2
union = set1 | set2
if not union:
return 0.0
return len(intersection) / len(union)
def compute_structure_similarity(clips1: list[dict], clips2: list[dict]) -> float:
"""计算两个视频的结构相似度(0~1)。
结构维度包括:
1. 片段数差异(数量越接近越相似)
2. 片段类型序列(相同位置的片段类型是否一致)
3. 时长分布(各片段时长占比是否相似)
Args:
clips1: 第一个视频的片段列表,每项包含 {clip_type, duration}
clips2: 第二个视频的片段列表
Returns:
0~1 之间的相似度
"""
if not clips1 and not clips2:
return 1.0
if not clips1 or not clips2:
return 0.0
# 1. 片段数相似度(数量差异越大越低)
n1, n2 = len(clips1), len(clips2)
count_sim = min(n1, n2) / max(n1, n2)
# 2. 类型序列相似度(逐位比较,相同位置类型是否一致)
min_len = min(n1, n2)
type_matches = sum(1 for i in range(min_len) if clips1[i].get("clip_type") == clips2[i].get("clip_type"))
type_sim = type_matches / min_len if min_len > 0 else 0.0
# 3. 时长分布相似度(归一化后比较分布)
total1 = sum(c.get("duration", 0) for c in clips1)
total2 = sum(c.get("duration", 0) for c in clips2)
if total1 > 0 and total2 > 0:
# 归一化为占比
dist1 = [c.get("duration", 0) / total1 for c in clips1]
dist2 = [c.get("duration", 0) / total2 for c in clips2]
# 比较前 min_len 个片段的占比差异(L1 距离转相似度)
l1_dist = sum(abs(dist1[i] - dist2[i]) for i in range(min_len))
# 加上多出的片段占比
if n1 > n2:
l1_dist += sum(dist1[i] for i in range(n2, n1))
elif n2 > n1:
l1_dist += sum(dist2[i] for i in range(n1, n2))
# L1 距离范围 [0, 2],转为相似度 [0, 1]
duration_sim = 1.0 - (l1_dist / 2.0)
else:
duration_sim = 0.0
# 三维度加权:数量 0.3 + 类型 0.4 + 时长 0.3
return count_sim * 0.3 + type_sim * 0.4 + duration_sim * 0.3
def compute_duplicate_rate(
self,
fingerprint: VideoFingerprint,
@@ -1361,6 +1257,110 @@ def compute_structure_similarity(clips1: list[dict], clips2: list[dict]) -> floa
}
# ── 文案 & 结构维度查重(Issue #P2-后端3) ────────────────────────
def _normalize_text(text: str) -> str:
"""文本标准化:去空白、转小写、去标点。"""
if not text:
return ""
# 去空白字符
text = re.sub(r"\s+", "", text)
# 转小写
text = text.lower()
# 去标点(只保留中文、字母、数字)
text = re.sub(r"[^\w\u4e00-\u9fff]", "", text)
return text
def compute_text_similarity(text1: str, text2: str) -> float:
"""计算两段文本的相似度(0~1)。
使用字符级 Jaccard 相似度:交集 / 并集。
适合短文本(配音脚本)的相似度比对。
Args:
text1: 第一段文本
text2: 第二段文本
Returns:
0~1 之间的相似度
"""
t1 = _normalize_text(text1)
t2 = _normalize_text(text2)
if not t1 and not t2:
return 1.0 # 都为空,视为完全相同
if not t1 or not t2:
return 0.0 # 一个为空,完全不同
# 字符级 Jaccard
set1 = set(t1)
set2 = set(t2)
intersection = set1 & set2
union = set1 | set2
if not union:
return 0.0
return len(intersection) / len(union)
def compute_structure_similarity(clips1: list[dict], clips2: list[dict]) -> float:
"""计算两个视频的结构相似度(0~1)。
结构维度包括:
1. 片段数差异(数量越接近越相似)
2. 片段类型序列(相同位置的片段类型是否一致)
3. 时长分布(各片段时长占比是否相似)
Args:
clips1: 第一个视频的片段列表,每项包含 {clip_type, duration}
clips2: 第二个视频的片段列表
Returns:
0~1 之间的相似度
"""
if not clips1 and not clips2:
return 1.0
if not clips1 or not clips2:
return 0.0
# 1. 片段数相似度(数量差异越大越低)
n1, n2 = len(clips1), len(clips2)
count_sim = min(n1, n2) / max(n1, n2)
# 2. 类型序列相似度(逐位比较,相同位置类型是否一致)
min_len = min(n1, n2)
type_matches = sum(1 for i in range(min_len) if clips1[i].get("clip_type") == clips2[i].get("clip_type"))
type_sim = type_matches / min_len if min_len > 0 else 0.0
# 3. 时长分布相似度(归一化后比较分布)
total1 = sum(c.get("duration", 0) for c in clips1)
total2 = sum(c.get("duration", 0) for c in clips2)
if total1 > 0 and total2 > 0:
# 归一化为占比
dist1 = [c.get("duration", 0) / total1 for c in clips1]
dist2 = [c.get("duration", 0) / total2 for c in clips2]
# 比较前 min_len 个片段的占比差异(L1 距离转相似度)
l1_dist = sum(abs(dist1[i] - dist2[i]) for i in range(min_len))
# 加上多出的片段占比
if n1 > n2:
l1_dist += sum(dist1[i] for i in range(n2, n1))
elif n2 > n1:
l1_dist += sum(dist2[i] for i in range(n1, n2))
# L1 距离范围 [0, 2],转为相似度 [0, 1]
duration_sim = 1.0 - (l1_dist / 2.0)
else:
duration_sim = 0.0
# 三维度加权:数量 0.3 + 类型 0.4 + 时长 0.3
return count_sim * 0.3 + type_sim * 0.4 + duration_sim * 0.3
def _save_fingerprint_chunks(
fingerprint: VideoFingerprint,
video_id: str,