From 3c782f89d1cb6735b20d6db37757b642b2a27e2f Mon Sep 17 00:00:00 2001 From: xiaoxia Date: Mon, 5 Oct 2026 02:04:07 +0800 Subject: [PATCH] =?UTF-8?q?fix(#2187):=20outfit=E6=8F=90=E5=8F=96=E6=AD=A3?= =?UTF-8?q?=E5=88=99=E4=BC=98=E5=8C=96=EF=BC=8C=E4=BF=9D=E7=95=99=E9=A2=9C?= =?UTF-8?q?=E8=89=B2=E5=BD=A2=E5=AE=B9=E8=AF=8D=20(#2187)?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Co-authored-by: xiaoxia Co-committed-by: xiaoxia --- apps/worker/worker_app/tasks/viral_video.py | 18 ++++++++++++------ 1 file changed, 12 insertions(+), 6 deletions(-) diff --git a/apps/worker/worker_app/tasks/viral_video.py b/apps/worker/worker_app/tasks/viral_video.py index 58272d7eb..9e6c6f919 100644 --- a/apps/worker/worker_app/tasks/viral_video.py +++ b/apps/worker/worker_app/tasks/viral_video.py @@ -495,17 +495,23 @@ def _analyze_single_image( ] for _ckw in _cloth_kws: if _ckw in _ptxt: - # 提取含关键词的短语(关键词前后4字) _ci = _ptxt.find(_ckw) - _start = max(0, _ci - 6) - _end = min(len(_ptxt), _ci + len(_ckw) + 2) + # 向前找颜色/材质/款式形容词(白/黑/米/红/蓝/灰/棉/麻/长/短/厚/薄/长袖/短袖/翻领/圆领/V领/印花/条纹等) + _start = max(0, _ci - 8) + # 向后包含款式词(长袖/短袖/外套/套装/上衣等后续修饰) + _end = min(len(_ptxt), _ci + len(_ckw) + 4) _outfit_extract = _ptxt[_start:_end].strip(" ,,。.、") - # 清理掉品牌名/产品名词的干扰(只取服装描述部分) + # 仅清理明确的品牌/产品类前缀(不清理颜色/款式/尺寸形容词) _outfit_extract = re.sub( - r"^[\w一-龥]{0,2}(牌|品牌|的|款|女|男|新|装|大|小|长|短|厚|薄)", + r"^(\S{0,4}牌|\S{0,3}品牌|\S{0,3}款|产品|商品|的)", "", _outfit_extract + ).strip() + # 尾部清理:去掉残留的品牌字/型号字(如"标""ml""g""装"等单字杂字) + _outfit_extract = re.sub( + r"(标[0-9a-zA-Z]*|\d+\s*(?:ml|g|L|斤|件|个|瓶|盒|包|袋|装)|\s+\d+\s*)$", "", _outfit_extract, - ) + flags=re.IGNORECASE, + ).strip() if len(_outfit_extract) >= 2: _outfit = _outfit_extract break