Files
kefu/deploy/review-voice-20260918/audit_voice_reference_matches.py
T
2026-09-21 10:34:06 +08:00

60 lines
3.1 KiB
Python

"""Only counts: reference matching and first 16-byte format header; no transcription."""
from collections import Counter
import json
from pathlib import Path
import re
import sqlite3
import sys
sys.dont_write_bytecode = True
sys.path.insert(0, r"C:\kefu\wechat_rpa")
from voice_messages import VoiceMessageReader, _references
cache = Path(r"C:\Users\isard\AppData\Local\ZhenYangTangRPA\wxwork_reply_cache")
source = Path(r"C:\Users\isard\Documents\WXWork")
reader = VoiceMessageReader(source, service_factory=lambda **kw: (_ for _ in ()).throw(AssertionError("No ASR permitted")))
report = {"scope": "metadata and 16-byte file magic only; no speech decode, no client calls", "accounts": []}
for folder in sorted(cache.iterdir()):
if not folder.is_dir() or not re.fullmatch(r"\d+", folder.name) or not (folder / "message.db").is_file():
continue
item = {"voice_messages": 0, "with_reference": 0, "one_exact_file": 0, "ambiguous": 0,
"missing": 0, "matched_suffixes": {}, "matched_formats": {}, "reference_shapes": {}}
formats, suffixes, shapes = Counter(), Counter(), Counter()
root = reader._root(folder.name)
index = reader._index(folder.name, root) if root and root.is_dir() else {}
item["default_voice_directory_exists"] = bool(root and root.is_dir())
db = sqlite3.connect((folder / "message.db").resolve().as_uri() + "?mode=ro&immutable=1", uri=True)
try:
db.execute("PRAGMA query_only=ON")
rows = db.execute("""SELECT m.content,v.voice_id FROM message_table m
LEFT JOIN msg_voice2text v ON v.message_id=m.server_id
WHERE m.content_type IN (4,16)""").fetchall()
for raw, voice_id in rows:
item["voice_messages"] += 1
refs = set(_references(raw) + _references(voice_id))
shapes.update("uuid" if re.fullmatch(r"[0-9a-f-]{36}", ref) else "filename" for ref in refs)
item["with_reference"] += bool(refs)
matches = set()
for ref in refs:
matches.update(index.get(ref, ()))
if len(matches) == 1:
item["one_exact_file"] += 1
path = Path(next(iter(matches)))
suffixes[path.suffix.lower() or "none"] += 1
with path.open("rb") as stream:
prefix = stream.read(16)
kind = ("silk" if prefix.startswith((b"#!SILK_V3", b"\x02#!SILK_V3")) else
"amr" if prefix.startswith((b"#!AMR\n", b"#!AMR-WB\n")) else
"wav" if prefix.startswith(b"RIFF") and prefix[8:12] == b"WAVE" else "unknown")
formats[kind] += 1
elif matches:
item["ambiguous"] += 1
else:
item["missing"] += 1
finally:
db.close()
item.update(matched_suffixes=dict(suffixes), matched_formats=dict(formats), reference_shapes=dict(shapes))
report["accounts"].append(item)
Path(__file__).with_name("voice-reference-counts.json").write_text(json.dumps(report, ensure_ascii=False, indent=2), encoding="utf-8")
print(json.dumps(report, ensure_ascii=False))