Files
kefu/deploy/recognition-audit-20260916/fix_visual_identity.py
T
2026-09-21 10:34:06 +08:00

58 lines
3.2 KiB
Python

"""Apply narrow OCR fixes to both authorized directories, retaining backups."""
from pathlib import Path
import hashlib
import json
import shutil
root = Path(__file__).resolve().parent
summary = []
for project in (Path('C:/kefu/wechat_rpa'), Path('C:/wechat_rpa')):
file = project / 'session_name.py'
original = file.read_text(encoding='utf-8')
source = original.replace('import hashlib\n', 'import hashlib\nimport math\n', 1)
source = source.replace(
' if not text.strip() or score < 0.45 or len(points) < 4:\n',
' if (not text.strip() or not math.isfinite(score) or not 0.45 <= score <= 1.0\n'
' or len(points) < 4 or not all(math.isfinite(value) for point in points for value in point)):\n', 1)
source = source.replace(
' if not name or score < MIN_CONFIDENCE:\n',
' if not name or not math.isfinite(score) or not MIN_CONFIDENCE <= score <= 1.0:\n', 1)
start = source.index(' def canonical(self, name: str) -> str:\n')
end = source.index(' def remember(self, name: str) -> None:\n', start)
source = source[:start] + ''' def canonical(self, name: str) -> str:
"""只纠正已确认的固定界面词和外部联系人后缀,不模糊合并客户姓名。
编辑距离一不能证明同一个人:客户甲@微信、客户乙@微信以及编号只差
一位的会员,都可能是不同客户。即使 OCR 置信度很高,也必须保留各自
的档案键。受限别名只有在目标已经认识时才能使用。
"""
normalized = normalize_name(name)
if not normalized or normalized in self._known:
return normalized
corrections = {"亏业资讯": "行业资讯", "客户联糸": "客户联系"}
corrected = corrections.get(normalized, "")
# 这里只修正平台后缀,@ 前的客户姓名必须逐字相同。
if normalized.endswith("@徽信"):
corrected = normalized[:-3] + "@微信"
if corrected and corrected in self._known:
print(f" [身份] 已识别的界面词误读:{normalized!r} → {corrected!r}")
return corrected
return normalized
''' + source[end:]
assert source != original and 'best_distance = MAX_CORRECTION_DISTANCE' not in source
compile(source, str(file), 'exec')
backup = project / 'backups' / 'recognition-audit-20260916' / 'visual'
backup.mkdir(parents=True, exist_ok=True)
assert not (backup / file.name).exists(), 'Do not overwrite a prior backup'
shutil.copy2(file, backup / file.name)
file.write_text(source, encoding='utf-8', newline='\n')
target_test = project / 'test_visual_identity_audit.py'
assert not target_test.exists()
shutil.copy2(root / target_test.name, target_test)
summary.append({'project': str(project), 'file': str(file), 'backup': str(backup / file.name),
'before_sha256': hashlib.sha256(original.encode()).hexdigest(),
'after_sha256': hashlib.sha256(source.encode()).hexdigest()})
(root / 'visual-changes.json').write_text(json.dumps(summary, ensure_ascii=False, indent=2), encoding='utf-8')
print(json.dumps(summary, ensure_ascii=False))