58 lines
3.2 KiB
Python
58 lines
3.2 KiB
Python
"""Apply narrow OCR fixes to both authorized directories, retaining backups."""
|
|
from pathlib import Path
|
|
import hashlib
|
|
import json
|
|
import shutil
|
|
|
|
root = Path(__file__).resolve().parent
|
|
summary = []
|
|
for project in (Path('C:/kefu/wechat_rpa'), Path('C:/wechat_rpa')):
|
|
file = project / 'session_name.py'
|
|
original = file.read_text(encoding='utf-8')
|
|
source = original.replace('import hashlib\n', 'import hashlib\nimport math\n', 1)
|
|
source = source.replace(
|
|
' if not text.strip() or score < 0.45 or len(points) < 4:\n',
|
|
' if (not text.strip() or not math.isfinite(score) or not 0.45 <= score <= 1.0\n'
|
|
' or len(points) < 4 or not all(math.isfinite(value) for point in points for value in point)):\n', 1)
|
|
source = source.replace(
|
|
' if not name or score < MIN_CONFIDENCE:\n',
|
|
' if not name or not math.isfinite(score) or not MIN_CONFIDENCE <= score <= 1.0:\n', 1)
|
|
start = source.index(' def canonical(self, name: str) -> str:\n')
|
|
end = source.index(' def remember(self, name: str) -> None:\n', start)
|
|
source = source[:start] + ''' def canonical(self, name: str) -> str:
|
|
"""只纠正已确认的固定界面词和外部联系人后缀,不模糊合并客户姓名。
|
|
|
|
编辑距离一不能证明同一个人:客户甲@微信、客户乙@微信以及编号只差
|
|
一位的会员,都可能是不同客户。即使 OCR 置信度很高,也必须保留各自
|
|
的档案键。受限别名只有在目标已经认识时才能使用。
|
|
"""
|
|
normalized = normalize_name(name)
|
|
if not normalized or normalized in self._known:
|
|
return normalized
|
|
corrections = {"亏业资讯": "行业资讯", "客户联糸": "客户联系"}
|
|
corrected = corrections.get(normalized, "")
|
|
# 这里只修正平台后缀,@ 前的客户姓名必须逐字相同。
|
|
if normalized.endswith("@徽信"):
|
|
corrected = normalized[:-3] + "@微信"
|
|
if corrected and corrected in self._known:
|
|
print(f" [身份] 已识别的界面词误读:{normalized!r} → {corrected!r}")
|
|
return corrected
|
|
return normalized
|
|
|
|
''' + source[end:]
|
|
assert source != original and 'best_distance = MAX_CORRECTION_DISTANCE' not in source
|
|
compile(source, str(file), 'exec')
|
|
backup = project / 'backups' / 'recognition-audit-20260916' / 'visual'
|
|
backup.mkdir(parents=True, exist_ok=True)
|
|
assert not (backup / file.name).exists(), 'Do not overwrite a prior backup'
|
|
shutil.copy2(file, backup / file.name)
|
|
file.write_text(source, encoding='utf-8', newline='\n')
|
|
target_test = project / 'test_visual_identity_audit.py'
|
|
assert not target_test.exists()
|
|
shutil.copy2(root / target_test.name, target_test)
|
|
summary.append({'project': str(project), 'file': str(file), 'backup': str(backup / file.name),
|
|
'before_sha256': hashlib.sha256(original.encode()).hexdigest(),
|
|
'after_sha256': hashlib.sha256(source.encode()).hexdigest()})
|
|
(root / 'visual-changes.json').write_text(json.dumps(summary, ensure_ascii=False, indent=2), encoding='utf-8')
|
|
print(json.dumps(summary, ensure_ascii=False))
|