"""Only counts: reference matching and first 16-byte format header; no transcription.""" from collections import Counter import json from pathlib import Path import re import sqlite3 import sys sys.dont_write_bytecode = True sys.path.insert(0, r"C:\kefu\wechat_rpa") from voice_messages import VoiceMessageReader, _references cache = Path(r"C:\Users\isard\AppData\Local\ZhenYangTangRPA\wxwork_reply_cache") source = Path(r"C:\Users\isard\Documents\WXWork") reader = VoiceMessageReader(source, service_factory=lambda **kw: (_ for _ in ()).throw(AssertionError("No ASR permitted"))) report = {"scope": "metadata and 16-byte file magic only; no speech decode, no client calls", "accounts": []} for folder in sorted(cache.iterdir()): if not folder.is_dir() or not re.fullmatch(r"\d+", folder.name) or not (folder / "message.db").is_file(): continue item = {"voice_messages": 0, "with_reference": 0, "one_exact_file": 0, "ambiguous": 0, "missing": 0, "matched_suffixes": {}, "matched_formats": {}, "reference_shapes": {}} formats, suffixes, shapes = Counter(), Counter(), Counter() root = reader._root(folder.name) index = reader._index(folder.name, root) if root and root.is_dir() else {} item["default_voice_directory_exists"] = bool(root and root.is_dir()) db = sqlite3.connect((folder / "message.db").resolve().as_uri() + "?mode=ro&immutable=1", uri=True) try: db.execute("PRAGMA query_only=ON") rows = db.execute("""SELECT m.content,v.voice_id FROM message_table m LEFT JOIN msg_voice2text v ON v.message_id=m.server_id WHERE m.content_type IN (4,16)""").fetchall() for raw, voice_id in rows: item["voice_messages"] += 1 refs = set(_references(raw) + _references(voice_id)) shapes.update("uuid" if re.fullmatch(r"[0-9a-f-]{36}", ref) else "filename" for ref in refs) item["with_reference"] += bool(refs) matches = set() for ref in refs: matches.update(index.get(ref, ())) if len(matches) == 1: item["one_exact_file"] += 1 path = Path(next(iter(matches))) suffixes[path.suffix.lower() or "none"] += 1 with path.open("rb") as stream: prefix = stream.read(16) kind = ("silk" if prefix.startswith((b"#!SILK_V3", b"\x02#!SILK_V3")) else "amr" if prefix.startswith((b"#!AMR\n", b"#!AMR-WB\n")) else "wav" if prefix.startswith(b"RIFF") and prefix[8:12] == b"WAVE" else "unknown") formats[kind] += 1 elif matches: item["ambiguous"] += 1 else: item["missing"] += 1 finally: db.close() item.update(matched_suffixes=dict(suffixes), matched_formats=dict(formats), reference_shapes=dict(shapes)) report["accounts"].append(item) Path(__file__).with_name("voice-reference-counts.json").write_text(json.dumps(report, ensure_ascii=False, indent=2), encoding="utf-8") print(json.dumps(report, ensure_ascii=False))