60 lines
3.1 KiB
Python
60 lines
3.1 KiB
Python
"""Only counts: reference matching and first 16-byte format header; no transcription."""
|
|
from collections import Counter
|
|
import json
|
|
from pathlib import Path
|
|
import re
|
|
import sqlite3
|
|
import sys
|
|
|
|
sys.dont_write_bytecode = True
|
|
sys.path.insert(0, r"C:\kefu\wechat_rpa")
|
|
from voice_messages import VoiceMessageReader, _references
|
|
|
|
cache = Path(r"C:\Users\isard\AppData\Local\ZhenYangTangRPA\wxwork_reply_cache")
|
|
source = Path(r"C:\Users\isard\Documents\WXWork")
|
|
reader = VoiceMessageReader(source, service_factory=lambda **kw: (_ for _ in ()).throw(AssertionError("No ASR permitted")))
|
|
report = {"scope": "metadata and 16-byte file magic only; no speech decode, no client calls", "accounts": []}
|
|
for folder in sorted(cache.iterdir()):
|
|
if not folder.is_dir() or not re.fullmatch(r"\d+", folder.name) or not (folder / "message.db").is_file():
|
|
continue
|
|
item = {"voice_messages": 0, "with_reference": 0, "one_exact_file": 0, "ambiguous": 0,
|
|
"missing": 0, "matched_suffixes": {}, "matched_formats": {}, "reference_shapes": {}}
|
|
formats, suffixes, shapes = Counter(), Counter(), Counter()
|
|
root = reader._root(folder.name)
|
|
index = reader._index(folder.name, root) if root and root.is_dir() else {}
|
|
item["default_voice_directory_exists"] = bool(root and root.is_dir())
|
|
db = sqlite3.connect((folder / "message.db").resolve().as_uri() + "?mode=ro&immutable=1", uri=True)
|
|
try:
|
|
db.execute("PRAGMA query_only=ON")
|
|
rows = db.execute("""SELECT m.content,v.voice_id FROM message_table m
|
|
LEFT JOIN msg_voice2text v ON v.message_id=m.server_id
|
|
WHERE m.content_type IN (4,16)""").fetchall()
|
|
for raw, voice_id in rows:
|
|
item["voice_messages"] += 1
|
|
refs = set(_references(raw) + _references(voice_id))
|
|
shapes.update("uuid" if re.fullmatch(r"[0-9a-f-]{36}", ref) else "filename" for ref in refs)
|
|
item["with_reference"] += bool(refs)
|
|
matches = set()
|
|
for ref in refs:
|
|
matches.update(index.get(ref, ()))
|
|
if len(matches) == 1:
|
|
item["one_exact_file"] += 1
|
|
path = Path(next(iter(matches)))
|
|
suffixes[path.suffix.lower() or "none"] += 1
|
|
with path.open("rb") as stream:
|
|
prefix = stream.read(16)
|
|
kind = ("silk" if prefix.startswith((b"#!SILK_V3", b"\x02#!SILK_V3")) else
|
|
"amr" if prefix.startswith((b"#!AMR\n", b"#!AMR-WB\n")) else
|
|
"wav" if prefix.startswith(b"RIFF") and prefix[8:12] == b"WAVE" else "unknown")
|
|
formats[kind] += 1
|
|
elif matches:
|
|
item["ambiguous"] += 1
|
|
else:
|
|
item["missing"] += 1
|
|
finally:
|
|
db.close()
|
|
item.update(matched_suffixes=dict(suffixes), matched_formats=dict(formats), reference_shapes=dict(shapes))
|
|
report["accounts"].append(item)
|
|
Path(__file__).with_name("voice-reference-counts.json").write_text(json.dumps(report, ensure_ascii=False, indent=2), encoding="utf-8")
|
|
print(json.dumps(report, ensure_ascii=False))
|