Files
kefu/deploy/review-voice-20260918/audit_voice_schema_counts.py
T
2026-09-21 10:34:06 +08:00

78 lines
4.3 KiB
Python

"""Read only decrypted cache schemas and aggregate voice counts; never content."""
import json
import re
import sqlite3
from pathlib import Path
OUT = Path(__file__).resolve().parent
BASES = {"source": Path(r"C:\wechat_rpa"),
"installed": Path(r"C:\Users\isard\AppData\Local\ZhenYangTangRPA")}
RELATIVES = ("wxwork_reply_cache", "archive_auto_backup/decrypted", "wxwork_decrypted")
paths = []
for base_name, base in BASES.items():
for relative in RELATIVES:
root = base / relative
if not root.is_dir():
continue
for folder in sorted(root.iterdir()):
if not folder.is_dir() or not re.fullmatch(r"\d{1,24}", folder.name):
continue
path = folder / "message.db"
if path.is_file() and path.resolve().is_relative_to(root.resolve()):
paths.append((base_name, relative, folder.name, path))
labels = {account: "account-" + str(i + 1) for i, account in enumerate(sorted({r[2] for r in paths}))}
result = {"scope": "existing decrypted caches only; no active identity discovery",
"sqlite": "mode=ro&immutable=1; main-file snapshot only; WAL ignored",
"databases": []}
for base, relative, account, path in paths:
item = {"location": base, "cache": relative, "account": labels[account], "tables": {}}
result["databases"].append(item)
connection = None
try:
connection = sqlite3.connect(path.resolve().as_uri() + "?mode=ro&immutable=1", uri=True, timeout=3)
connection.execute("PRAGMA query_only=ON")
connection.execute("BEGIN")
names = {r[0] for r in connection.execute("SELECT name FROM sqlite_master WHERE type='table'")}
for table in ("message_table", "msg_voice2text"):
if table not in names:
continue
columns = [{"name": row[1], "type": row[2], "pk": row[5]}
for row in connection.execute('PRAGMA table_info("' + table + '")')]
item["tables"][table] = {"columns": columns,
"count": connection.execute('SELECT count(*) FROM "' + table + '"').fetchone()[0]}
if "message_table" not in item["tables"]:
continue
columns = {row["name"] for row in item["tables"]["message_table"]["columns"]}
if "content_type" not in columns:
continue
item["voice_types"] = {str(row[0]): row[1] for row in connection.execute(
"SELECT content_type,count(*) FROM message_table WHERE content_type IN (4,16) GROUP BY content_type")}
if {"sender_id", "conversation_id"} <= columns:
row = connection.execute("""SELECT count(*),
sum(CAST(sender_id AS TEXT) != ?),
sum(conversation_id LIKE 'M:%' OR conversation_id LIKE 'S:%')
FROM message_table WHERE content_type IN (4,16)""", (account,)).fetchone()
item["voice_counts"] = dict(zip(("total", "incoming", "personal"), (int(v or 0) for v in row)))
if "msg_voice2text" in item["tables"]:
trans_columns = {row["name"] for row in item["tables"]["msg_voice2text"]["columns"]}
if {"message_id", "text"} <= trans_columns:
item["nonempty_transcriptions"] = connection.execute(
"SELECT count(*) FROM msg_voice2text WHERE text IS NOT NULL AND trim(text) != ''").fetchone()[0]
if "server_id" in columns:
item["voice_with_cached_transcription"] = connection.execute("""
SELECT count(*) FROM message_table m WHERE m.content_type IN (4,16)
AND EXISTS (SELECT 1 FROM msg_voice2text v
WHERE CAST(v.message_id AS TEXT)=CAST(m.server_id AS TEXT)
AND v.text IS NOT NULL AND trim(v.text) != '')""").fetchone()[0]
except Exception as exc:
item["error_type"] = type(exc).__name__
finally:
if connection is not None:
connection.close()
OUT.mkdir(parents=True, exist_ok=True)
(OUT / "voice-schema-counts.json").write_text(json.dumps(result, ensure_ascii=False, indent=2), encoding="utf-8")
print(json.dumps({"databases": [{key: value for key, value in item.items() if key != "tables"} |
{"tables": {key: value["count"] for key, value in item["tables"].items()}}
for item in result["databases"]]}, ensure_ascii=False))