850 lines
34 KiB
Python
850 lines
34 KiB
Python
# -*- coding: utf-8 -*-
|
|
"""Read-only browser model for locally decrypted WeCom message databases.
|
|
|
|
The auto-reply engine and archive uploader already maintain decrypted SQLite
|
|
copies. This module only reads those copies and therefore never writes to the
|
|
live WeCom databases. A manual refresh may update the existing decrypted
|
|
cache through ``wxwork_db.decrypt_with_keys`` before the snapshot is built.
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
import sqlite3
|
|
import math
|
|
import threading
|
|
import base64
|
|
import mimetypes
|
|
import re
|
|
from array import array
|
|
from collections import OrderedDict
|
|
import time
|
|
from datetime import datetime
|
|
from pathlib import Path
|
|
from typing import Any, Iterable
|
|
|
|
from runtime_paths import application_data_dir
|
|
from wxwork_db import (
|
|
connect_sqlite,
|
|
decrypt_with_keys,
|
|
detect_wxwork_dir,
|
|
get_msg_type_name,
|
|
load_keys,
|
|
parse_content,
|
|
)
|
|
|
|
|
|
DATABASE_TABLES = {
|
|
"message.db": "message_table",
|
|
"session.db": "conversation_table",
|
|
"user.db": "user_table",
|
|
"company.db": "company_table",
|
|
}
|
|
|
|
|
|
# Cache only derived read-only data. A DB replacement or WAL commit invalidates
|
|
# its entry; no SQLite connection or file handle remains open between requests.
|
|
_CACHE_LOCK = threading.RLock()
|
|
_CACHES: dict[str, OrderedDict] = {}
|
|
_CACHE_LIMITS = {"index": 4, "names": 64, "files": 64, "counts": 64, "messages": 32}
|
|
_MEDIA_CACHE_LIMIT = 256
|
|
_MEDIA_REF = re.compile(r"(?<![0-9a-f])[0-9a-f]{8}-[0-9a-f]{4}-[0-9a-f]{4}-[0-9a-f]{4}-[0-9a-f]{12}(?![0-9a-f])|(?<![\w.-])[\w\u4e00-\u9fff()()【】\[\]-]{2,120}\.(?:png|jpe?g|gif|bmp|webp)(?!\w)", re.I)
|
|
_MEDIA_INDEX_CACHE: OrderedDict = OrderedDict()
|
|
_VOICE_READERS: dict[str, Any] = {}
|
|
|
|
|
|
def clear_browser_cache() -> None:
|
|
with _CACHE_LOCK:
|
|
_CACHES.clear()
|
|
_MEDIA_INDEX_CACHE.clear()
|
|
|
|
|
|
def _media_index(root: Path) -> dict[str, list[Path]]:
|
|
"""Index image cache files; references are matched by exact basename/stem."""
|
|
root = Path(root).resolve()
|
|
key = str(root).casefold()
|
|
try:
|
|
revision = tuple((str(p), p.stat().st_mtime_ns, p.stat().st_size)
|
|
for p in root.glob("**/*") if p.is_file())
|
|
except OSError:
|
|
revision = ()
|
|
with _CACHE_LOCK:
|
|
found = _MEDIA_INDEX_CACHE.get(key)
|
|
if found and found[0] == revision:
|
|
_MEDIA_INDEX_CACHE.move_to_end(key)
|
|
return found[1]
|
|
result: dict[str, list[Path]] = {}
|
|
if root.is_dir():
|
|
try:
|
|
files = [p for p in root.rglob("*") if p.is_file()][:20000]
|
|
except OSError:
|
|
files = []
|
|
for path in files:
|
|
if path.suffix.casefold() not in {".png", ".jpg", ".jpeg", ".gif", ".bmp", ".webp"}:
|
|
continue
|
|
tokens = {path.name.casefold(), path.stem.casefold()}
|
|
match = re.match(r"[0-9a-f]{8}-[0-9a-f]{4}-[0-9a-f]{4}-[0-9a-f]{4}-[0-9a-f]{12}", path.name, re.I)
|
|
if match:
|
|
tokens.add(match.group(0).casefold())
|
|
for token in tokens:
|
|
result.setdefault(token, []).append(path)
|
|
with _CACHE_LOCK:
|
|
_MEDIA_INDEX_CACHE[key] = (revision, result)
|
|
_MEDIA_INDEX_CACHE.move_to_end(key)
|
|
while len(_MEDIA_INDEX_CACHE) > _MEDIA_CACHE_LIMIT:
|
|
_MEDIA_INDEX_CACHE.popitem(last=False)
|
|
return result
|
|
|
|
|
|
def _media_data(raw: Any, media_root: Path | None, content_type: Any) -> dict[str, Any]:
|
|
"""Return a bounded data URI after an exact decoded image-cache match."""
|
|
try:
|
|
kind = int(content_type)
|
|
except (TypeError, ValueError):
|
|
kind = -1
|
|
if kind not in {3, 14, 123} or media_root is None:
|
|
return {}
|
|
if isinstance(raw, str):
|
|
raw = raw.encode("utf-8", errors="ignore")
|
|
if not isinstance(raw, (bytes, bytearray, memoryview)):
|
|
return {}
|
|
text = bytes(raw).decode("utf-8", errors="ignore")
|
|
refs = [match.group(0).casefold() for match in _MEDIA_REF.finditer(text)][:32]
|
|
if not refs:
|
|
return {}
|
|
index = _media_index(media_root)
|
|
matches = []
|
|
for ref in refs:
|
|
matches.extend(index.get(ref, ()))
|
|
unique = {str(path.resolve()): path for path in matches}
|
|
if len(unique) != 1:
|
|
return {}
|
|
path = next(iter(unique.values()))
|
|
try:
|
|
info = path.stat()
|
|
if info.st_size <= 0 or info.st_size > 2 * 1024 * 1024:
|
|
return {}
|
|
data = path.read_bytes()
|
|
except OSError:
|
|
return {}
|
|
if not data.startswith((b"\x89PNG\r\n\x1a\n", b"\xff\xd8\xff", b"GIF8", b"BM", b"RIFF")):
|
|
return {}
|
|
mime = mimetypes.guess_type(path.name)[0] or "image/png"
|
|
return {"src": "data:" + mime + ";base64," + base64.b64encode(data).decode("ascii"),
|
|
"name": path.name, "mime": mime}
|
|
|
|
|
|
def _database_revision(path: Path) -> tuple:
|
|
values = [str(path.resolve()).casefold()]
|
|
for candidate in (path, Path(str(path) + "-wal")):
|
|
try:
|
|
info = candidate.stat()
|
|
values.append((info.st_size, info.st_mtime_ns, info.st_ctime_ns, info.st_ino))
|
|
except OSError:
|
|
values.append(None)
|
|
return tuple(values)
|
|
|
|
|
|
def _cached(kind: str, key: Any, revision: tuple, build):
|
|
# Building under the lock coalesces simultaneous navigation requests. The
|
|
# GUI already loads in a worker; repeated searches cannot multiply scans.
|
|
with _CACHE_LOCK:
|
|
cache = _CACHES.setdefault(kind, OrderedDict())
|
|
found = cache.get(key)
|
|
if found is not None and found[0] == revision:
|
|
cache.move_to_end(key)
|
|
return found[1]
|
|
value = build()
|
|
cache[key] = (revision, value)
|
|
cache.move_to_end(key)
|
|
while len(cache) > _CACHE_LIMITS.get(kind, 16):
|
|
cache.popitem(last=False)
|
|
return value
|
|
|
|
|
|
def _message_index(message_path: Path) -> dict:
|
|
revision = _database_revision(message_path)
|
|
def build():
|
|
connection = connect_sqlite(str(message_path))
|
|
try:
|
|
columns = _columns(connection, "message_table")
|
|
result = {"columns": columns, "total_messages": 0, "conversations": {},
|
|
"ordered": [], "previews": {}, "revision": revision}
|
|
if not {"conversation_id", "send_time"}.issubset(columns):
|
|
return result
|
|
result["total_messages"] = _message_counts(message_path)[0]
|
|
# Sort/group inside SQLite instead of allocating one Python tuple
|
|
# per message. Only compact row IDs and one timestamp per chat are
|
|
# retained; message bodies are fetched for the visible page only.
|
|
cursor = connection.execute(
|
|
"SELECT conversation_id,GROUP_CONCAT(message_rowid),MAX(message_time) "
|
|
"FROM (SELECT rowid AS message_rowid,conversation_id,CAST(send_time AS REAL) AS message_time "
|
|
"FROM message_table WHERE conversation_id IS NOT NULL AND conversation_id<>'' "
|
|
"AND conversation_id NOT LIKE 'Y:%' "
|
|
"ORDER BY conversation_id,CAST(send_time AS REAL),rowid) GROUP BY conversation_id"
|
|
)
|
|
for conversation_id, encoded_rows, timestamp in cursor:
|
|
row_ids = array("q", map(int, str(encoded_rows).split(",")))
|
|
timestamp = float(timestamp or 0)
|
|
if not math.isfinite(timestamp):
|
|
timestamp = 0.0
|
|
result["conversations"][str(conversation_id)] = {
|
|
"rows": row_ids, "lastTimestamp": timestamp, "lastRowId": row_ids[-1],
|
|
}
|
|
result["ordered"] = sorted(result["conversations"], key=lambda key: (
|
|
result["conversations"][key]["lastTimestamp"],
|
|
result["conversations"][key]["lastRowId"], key), reverse=True)
|
|
return result
|
|
finally:
|
|
connection.close()
|
|
return _cached("index", revision[0], revision, build)
|
|
|
|
|
|
def _read_message_rows(message_path: Path, index: dict, row_ids: list[int]) -> dict[int, dict]:
|
|
if not row_ids:
|
|
return {}
|
|
columns = index["columns"]
|
|
wanted = [name for name in ("conversation_id", "sender_id", "send_time", "content_type", "content", "server_id", "client_id") if name in columns]
|
|
if not wanted:
|
|
return {}
|
|
connection = connect_sqlite(str(message_path))
|
|
try:
|
|
result = {}
|
|
for offset in range(0, len(row_ids), 500):
|
|
batch = row_ids[offset:offset + 500]
|
|
marks = ",".join("?" for _ in batch)
|
|
for row in connection.execute(f"SELECT rowid,{','.join(wanted)} FROM message_table WHERE rowid IN ({marks})", batch):
|
|
result[int(row[0])] = dict(zip(wanted, row[1:]))
|
|
return result
|
|
finally:
|
|
connection.close()
|
|
|
|
|
|
def _ensure_previews(message_path: Path, index: dict, conversation_ids: list[str]) -> None:
|
|
with _CACHE_LOCK:
|
|
missing = [key for key in conversation_ids if key not in index["previews"]]
|
|
for offset in range(0, len(missing), 500):
|
|
batch = missing[offset:offset + 500]
|
|
rows = _read_message_rows(message_path, index, [index["conversations"][key]["lastRowId"] for key in batch])
|
|
for key in batch:
|
|
item = rows.get(index["conversations"][key]["lastRowId"], {})
|
|
# A cache file can be atomically replaced between indexing and
|
|
# reading. Reused row IDs must never expose another chat.
|
|
if str(item.get("conversation_id") or "") != key:
|
|
index["previews"][key] = {"text": "", "sender": ""}
|
|
continue
|
|
index["previews"][key] = {
|
|
"text": _content_label(item.get("content"), item.get("content_type")).replace("\n", " ")[:2000],
|
|
"sender": str(item.get("sender_id") or ""),
|
|
}
|
|
|
|
|
|
def _load_names(databases: dict[str, Path], account: str):
|
|
revision = tuple((name, _database_revision(databases[name])) for name in ("user.db", "session.db") if name in databases)
|
|
key = (account, tuple((name, str(path.resolve()).casefold()) for name, path in databases.items() if name in {"user.db", "session.db"}))
|
|
return _cached("names", key, revision, lambda: _load_names_uncached(databases, account))
|
|
|
|
|
|
def _database_files(databases: dict[str, Path]):
|
|
revision = tuple((name, _database_revision(path)) for name, path in sorted(databases.items()))
|
|
key = tuple((name, str(path.resolve()).casefold()) for name, path in sorted(databases.items()))
|
|
files, size = _cached("files", key, revision, lambda: _database_files_uncached(databases))
|
|
return [dict(item) for item in files], size
|
|
|
|
|
|
def _message_counts(message_path: Path) -> tuple[int, int]:
|
|
revision = _database_revision(message_path)
|
|
def build():
|
|
connection = connect_sqlite(str(message_path))
|
|
try:
|
|
columns = _columns(connection, "message_table")
|
|
if "conversation_id" not in columns:
|
|
return _count(connection, "message_table"), 0
|
|
values = connection.execute(
|
|
"SELECT COUNT(*),COUNT(DISTINCT CASE WHEN conversation_id IS NOT NULL "
|
|
"AND conversation_id<>'' AND conversation_id NOT LIKE 'Y:%' "
|
|
"THEN conversation_id END) FROM message_table"
|
|
).fetchone()
|
|
return int(values[0]), int(values[1])
|
|
finally:
|
|
connection.close()
|
|
return _cached("counts", revision[0], revision, build)
|
|
|
|
|
|
def _format_time(value: Any) -> str:
|
|
try:
|
|
timestamp = float(value or 0)
|
|
except (TypeError, ValueError):
|
|
return ""
|
|
if timestamp > 10**12:
|
|
timestamp /= 1000
|
|
if timestamp <= 0:
|
|
return ""
|
|
try:
|
|
return datetime.fromtimestamp(timestamp).strftime("%Y-%m-%d %H:%M:%S")
|
|
except (OSError, OverflowError, ValueError):
|
|
return ""
|
|
|
|
|
|
def _size_label(size: int) -> str:
|
|
if size < 1024:
|
|
return f"{size} B"
|
|
if size < 1024 * 1024:
|
|
return f"{size / 1024:.1f} KB"
|
|
if size < 1024 * 1024 * 1024:
|
|
return f"{size / (1024 * 1024):.1f} MB"
|
|
return f"{size / (1024 * 1024 * 1024):.2f} GB"
|
|
|
|
|
|
def _columns(connection: sqlite3.Connection, table: str) -> set[str]:
|
|
try:
|
|
return {
|
|
str(row[1])
|
|
for row in connection.execute(f'PRAGMA table_info("{table}")').fetchall()
|
|
}
|
|
except sqlite3.Error:
|
|
return set()
|
|
|
|
|
|
def _count(connection: sqlite3.Connection, table: str) -> int:
|
|
try:
|
|
return int(connection.execute(f'SELECT COUNT(*) FROM "{table}"').fetchone()[0])
|
|
except (sqlite3.Error, TypeError, ValueError):
|
|
return 0
|
|
|
|
|
|
def default_cache_roots() -> list[Path]:
|
|
data_root = application_data_dir()
|
|
source_root = Path(__file__).resolve().parent
|
|
candidates = [
|
|
data_root / "archive_auto_backup" / "decrypted",
|
|
data_root / "wxwork_decrypted",
|
|
source_root / "archive_auto_backup" / "decrypted",
|
|
source_root / "wxwork_decrypted",
|
|
]
|
|
result: list[Path] = []
|
|
seen: set[str] = set()
|
|
for candidate in candidates:
|
|
identity = str(candidate.resolve()).casefold()
|
|
if identity in seen:
|
|
continue
|
|
seen.add(identity)
|
|
result.append(candidate)
|
|
return result
|
|
|
|
|
|
def _discover_databases(cache_roots: Iterable[Path]) -> dict[str, dict[str, Path]]:
|
|
"""Choose the newest readable copy of every database for each account."""
|
|
|
|
discovered: dict[str, dict[str, Path]] = {}
|
|
for root in cache_roots:
|
|
if not root.is_dir():
|
|
continue
|
|
try:
|
|
account_dirs = list(root.iterdir())
|
|
except OSError:
|
|
continue
|
|
for account_dir in account_dirs:
|
|
if not account_dir.is_dir():
|
|
continue
|
|
account = account_dir.name
|
|
for database_path in account_dir.glob("*.db"):
|
|
current = discovered.setdefault(account, {}).get(database_path.name)
|
|
try:
|
|
newer = current is None or database_path.stat().st_mtime > current.stat().st_mtime
|
|
except OSError:
|
|
continue
|
|
if newer:
|
|
discovered[account][database_path.name] = database_path
|
|
return discovered
|
|
|
|
|
|
def _refresh_cache(*, auto_acquire: bool = False) -> tuple[str, str]:
|
|
"""先复用密钥解密;自动初始化时,仅在无法读取后获取一次本机密钥。"""
|
|
source_root = detect_wxwork_dir()
|
|
if not source_root:
|
|
return "", "自动检测未找到企业微信数据库。请先在本机登录企业微信后刷新,或手动选择数据目录"
|
|
from wxwork_local_setup import acquire_local_keys, select_source_directory
|
|
|
|
try:
|
|
# 让独立 EXE 密钥工作进程使用同一个已识别的目录。
|
|
if auto_acquire:
|
|
select_source_directory(source_root)
|
|
expected = {p.parent.parent.name for p in Path(source_root).glob("*/Data/message.db")}
|
|
|
|
def decrypt() -> tuple[set[str], dict]:
|
|
try:
|
|
keys = load_keys()
|
|
except ValueError:
|
|
keys = {}
|
|
output = str(application_data_dir() / "wxwork_decrypted")
|
|
decrypted = decrypt_with_keys(source_root, output, keys, use_cache=True)
|
|
readable = set()
|
|
invalid = False
|
|
for path, name, account in decrypted:
|
|
if name != "message.db":
|
|
continue
|
|
try:
|
|
connection = connect_sqlite(path)
|
|
try:
|
|
connection.execute("SELECT conversation_id FROM message_table LIMIT 1").fetchone()
|
|
readable.add(account)
|
|
finally:
|
|
connection.close()
|
|
except sqlite3.Error:
|
|
invalid = True
|
|
# 缓存大小/时间正常也可能已经损坏;重建后再次检查。
|
|
if invalid:
|
|
decrypted = decrypt_with_keys(source_root, output, keys, use_cache=False)
|
|
readable.clear()
|
|
for path, name, account in decrypted:
|
|
if name != "message.db":
|
|
continue
|
|
try:
|
|
connection = connect_sqlite(path)
|
|
try:
|
|
connection.execute("SELECT conversation_id FROM message_table LIMIT 1").fetchone()
|
|
readable.add(account)
|
|
finally:
|
|
connection.close()
|
|
except sqlite3.Error:
|
|
continue
|
|
return readable, keys
|
|
|
|
readable, keys = decrypt()
|
|
if expected and expected <= readable:
|
|
return source_root, ""
|
|
if auto_acquire:
|
|
try:
|
|
acquire_local_keys()
|
|
except (OSError, ValueError) as exc:
|
|
return source_root, f"自动解密未完成:{exc}。可重新刷新自动检测,或使用下方手动设置"
|
|
readable, keys = decrypt()
|
|
if expected and expected <= readable:
|
|
return source_root, ""
|
|
return source_root, "自动获取密钥后仍有数据库无法解密,请确认企业微信已登录,或使用下方手动设置"
|
|
if not keys:
|
|
return source_root, "已找到企业微信数据库,但本机尚未配置密钥。请点击刷新自动获取,或导入当前电脑的密钥文件"
|
|
return source_root, "已找到企业微信数据库,但现有密钥无法解密全部数据库。请刷新重新获取本机密钥或导入匹配的密钥文件"
|
|
except Exception as exc:
|
|
return source_root, f"更新解密缓存失败:{exc}"
|
|
|
|
|
|
def _load_names_uncached(databases: dict[str, Path], account: str) -> tuple[dict[str, str], dict[str, str], str]:
|
|
users: dict[str, str] = {}
|
|
conversations: dict[str, str] = {}
|
|
account_name = account
|
|
|
|
user_path = databases.get("user.db")
|
|
if user_path:
|
|
try:
|
|
connection = connect_sqlite(str(user_path))
|
|
try:
|
|
columns = _columns(connection, "user_table")
|
|
wanted = [name for name in ("id", "name", "real_name", "account") if name in columns]
|
|
if "id" in wanted:
|
|
cursor = connection.execute(f"SELECT {','.join(wanted)} FROM user_table")
|
|
for values in cursor.fetchall():
|
|
item = dict(zip(wanted, values))
|
|
user_id = str(item.get("id") or "")
|
|
label = str(
|
|
item.get("name")
|
|
or item.get("real_name")
|
|
or item.get("account")
|
|
or user_id
|
|
)
|
|
if user_id:
|
|
users[user_id] = label
|
|
account_name = users.get(account) or account_name
|
|
finally:
|
|
connection.close()
|
|
except sqlite3.Error:
|
|
pass
|
|
|
|
session_path = databases.get("session.db")
|
|
if session_path:
|
|
try:
|
|
connection = connect_sqlite(str(session_path))
|
|
try:
|
|
columns = _columns(connection, "conversation_table")
|
|
wanted = [
|
|
name
|
|
for name in ("id", "name", "roomname_remark", "session_id")
|
|
if name in columns
|
|
]
|
|
if "id" in wanted:
|
|
cursor = connection.execute(
|
|
f"SELECT {','.join(wanted)} FROM conversation_table"
|
|
)
|
|
for values in cursor.fetchall():
|
|
item = dict(zip(wanted, values))
|
|
conversation_id = str(item.get("id") or "")
|
|
label = str(
|
|
item.get("roomname_remark")
|
|
or item.get("name")
|
|
or item.get("session_id")
|
|
or ""
|
|
)
|
|
if conversation_id and label:
|
|
conversations[conversation_id] = label
|
|
finally:
|
|
connection.close()
|
|
except sqlite3.Error:
|
|
pass
|
|
return users, conversations, account_name
|
|
|
|
|
|
def _conversation_name(
|
|
account: str,
|
|
conversation_id: str,
|
|
users: dict[str, str],
|
|
conversations: dict[str, str],
|
|
) -> str:
|
|
label = str(conversations.get(conversation_id) or "").strip()
|
|
if label and label != conversation_id:
|
|
return label
|
|
if conversation_id.startswith("M:"):
|
|
peer_id = conversation_id[2:]
|
|
return users.get(peer_id) or f"微信用户 {peer_id}"
|
|
if conversation_id.startswith("S:"):
|
|
peers = [item for item in conversation_id[2:].split("_") if item]
|
|
peer_id = next((item for item in peers if item != account), peers[0] if peers else "")
|
|
return users.get(peer_id) or f"企微用户 {peer_id}"
|
|
if conversation_id.startswith("R:"):
|
|
return label or f"群聊 {conversation_id[2:]}"
|
|
if conversation_id.startswith("Y:"):
|
|
return label or f"应用 {conversation_id[2:]}"
|
|
return label or conversation_id or "未知会话"
|
|
|
|
|
|
def _conversation_kind(conversation_id: str) -> str:
|
|
if conversation_id.startswith("M:"):
|
|
return "微信客户"
|
|
if conversation_id.startswith("S:"):
|
|
return "企业微信"
|
|
if conversation_id.startswith("R:"):
|
|
return "群聊"
|
|
if conversation_id.startswith("Y:"):
|
|
return "应用"
|
|
return "其他"
|
|
|
|
|
|
def _content_label(content: Any, content_type: Any) -> str:
|
|
from archive_content_parser import parse_message_content
|
|
parsed = parse_message_content(content, content_type)["text"].strip()
|
|
if parsed:
|
|
return parsed
|
|
return f"[{get_msg_type_name(content_type)}]"
|
|
|
|
|
|
def _database_files_uncached(databases: dict[str, Path]) -> tuple[list[dict[str, Any]], int]:
|
|
files: list[dict[str, Any]] = []
|
|
total_size = 0
|
|
preferred_order = {
|
|
name: index
|
|
for index, name in enumerate(
|
|
("message.db", "session.db", "user.db", "company.db", "message_lookup.db", "user_extend.db")
|
|
)
|
|
}
|
|
for name, path in sorted(
|
|
databases.items(), key=lambda item: (preferred_order.get(item[0], 99), item[0])
|
|
):
|
|
try:
|
|
stat = path.stat()
|
|
size = int(stat.st_size)
|
|
modified = _format_time(stat.st_mtime)
|
|
except OSError:
|
|
size, modified = 0, ""
|
|
total_size += size
|
|
row_count = 0
|
|
status = "可读取"
|
|
table = DATABASE_TABLES.get(name)
|
|
try:
|
|
connection = connect_sqlite(str(path))
|
|
try:
|
|
if table:
|
|
row_count = _message_counts(path)[0] if table == "message_table" else _count(connection, table)
|
|
connection.execute("SELECT COUNT(*) FROM sqlite_master").fetchone()
|
|
finally:
|
|
connection.close()
|
|
except sqlite3.Error:
|
|
status = "读取失败"
|
|
files.append(
|
|
{
|
|
"name": name,
|
|
"path": str(path),
|
|
"size": size,
|
|
"sizeLabel": _size_label(size),
|
|
"modified": modified,
|
|
"rows": row_count,
|
|
"status": status,
|
|
}
|
|
)
|
|
return files, total_size
|
|
|
|
|
|
def _conversation_rows(
|
|
message_path: Path,
|
|
account: str,
|
|
users: dict[str, str],
|
|
conversations: dict[str, str],
|
|
query: str,
|
|
limit: int,
|
|
offset: int = 0,
|
|
) -> tuple[list[dict[str, Any]], int, int]:
|
|
index = _message_index(message_path)
|
|
ordered = index["ordered"]
|
|
needle = str(query or "").strip().casefold()
|
|
offset = max(0, int(offset))
|
|
candidates = ordered if needle else ordered[offset:offset + limit]
|
|
_ensure_previews(message_path, index, candidates)
|
|
result = []
|
|
matching = 0
|
|
for conversation_id in candidates:
|
|
name = _conversation_name(account, conversation_id, users, conversations)
|
|
preview = index["previews"][conversation_id]
|
|
if needle and needle not in f"{name} {conversation_id} {preview['text']}".casefold():
|
|
continue
|
|
if needle:
|
|
matching += 1
|
|
if matching <= offset:
|
|
continue
|
|
group = index["conversations"][conversation_id]
|
|
result.append({
|
|
"id": conversation_id, "name": name, "kind": _conversation_kind(conversation_id),
|
|
"messageCount": len(group["rows"]), "lastTime": _format_time(group["lastTimestamp"]),
|
|
"lastTimestamp": group["lastTimestamp"], "preview": preview["text"][:160],
|
|
"lastDirection": "发出" if preview["sender"] == account else "收到",
|
|
})
|
|
if len(result) >= limit:
|
|
break
|
|
return result, len(ordered), index["total_messages"]
|
|
|
|
|
|
def _messages(
|
|
message_path: Path,
|
|
account: str,
|
|
conversation_id: str,
|
|
users: dict[str, str],
|
|
limit: int,
|
|
offset: int = 0,
|
|
media_root: Path | None = None,
|
|
) -> list[dict[str, Any]]:
|
|
if not conversation_id:
|
|
return []
|
|
index = _message_index(message_path)
|
|
rows = (index["conversations"].get(conversation_id) or {}).get("rows", [])
|
|
offset = max(0, int(offset))
|
|
limit = max(1, min(int(limit), 1000))
|
|
end = max(0, len(rows) - offset)
|
|
row_ids = list(rows[max(0, end - limit):end])
|
|
revision = index["revision"]
|
|
key = (revision[0], conversation_id, limit, offset)
|
|
records = _cached("messages", key, revision, lambda: _read_message_rows(message_path, index, row_ids))
|
|
result = []
|
|
connection = None
|
|
try:
|
|
from archive_content_parser import parse_message_content
|
|
from voice_messages import is_voice, cached_voice_fields
|
|
if any(is_voice(row) for row in records.values()):
|
|
connection = connect_sqlite(str(message_path))
|
|
for row_id in row_ids:
|
|
item = records.get(row_id)
|
|
if item is None or str(item.get("conversation_id") or "") != conversation_id:
|
|
continue
|
|
sender_id = str(item.get("sender_id") or "")
|
|
try:
|
|
timestamp = float(item.get("send_time") or 0)
|
|
if not math.isfinite(timestamp):
|
|
timestamp = 0.0
|
|
except (ValueError, TypeError, OverflowError):
|
|
timestamp = 0.0
|
|
content_type = item.get("content_type")
|
|
decoded = parse_message_content(item.get("content"), content_type)
|
|
fields = {}
|
|
server_id = str(item.get("server_id") or "")
|
|
if is_voice(item) and connection is not None:
|
|
fields = cached_voice_fields(connection, account, conversation_id, server_id, item.get("content"))
|
|
content = fields.get("content") or decoded.get("text") or _content_label(item.get("content"), content_type)
|
|
message = {
|
|
"id": server_id or str(item.get("client_id") or row_id),
|
|
"senderId": sender_id,
|
|
"sender": users.get(sender_id) or ("当前账号" if sender_id == account else sender_id),
|
|
"direction": "outbound" if sender_id == account else "inbound",
|
|
"time": _format_time(timestamp), "timestamp": timestamp,
|
|
"type": get_msg_type_name(content_type), "content": content,
|
|
"account": account, "conv_id": conversation_id, "server_id": server_id,
|
|
"content_type": content_type, "voice_refs": fields.get("voice_refs", []),
|
|
**fields,
|
|
}
|
|
result.append(message)
|
|
finally:
|
|
if connection is not None:
|
|
connection.close()
|
|
# Queue every visible voice in the history, including messages sent by us;
|
|
# the reply engine's default unanswered-only behavior remains unchanged.
|
|
if any(int(item.get("content_type") or 0) in {4, 16} for item in result):
|
|
source_key = str(media_root.parent.parent if media_root else "").casefold()
|
|
if source_key:
|
|
reader = _VOICE_READERS.get(source_key)
|
|
if reader is None:
|
|
from voice_messages import VoiceMessageReader
|
|
reader = _VOICE_READERS[source_key] = VoiceMessageReader(Path(source_key), background_index=True)
|
|
reader.prepare(result, include_history=True)
|
|
for message in result:
|
|
if message.get("voice_transcribed") and message.get("content"):
|
|
continue
|
|
image = _media_data(records.get(next((rid for rid, row in records.items() if str(row.get("server_id") or row.get("client_id") or rid) == message["id"]), 0), {}).get("content"), media_root, message.get("content_type"))
|
|
if image:
|
|
message["media"] = {"kind": "image", **image}
|
|
for message in result:
|
|
for key_name in ("account", "conv_id", "server_id", "content_type", "voice_refs"):
|
|
message.pop(key_name, None)
|
|
return result
|
|
|
|
|
|
def load_browser_snapshot(
|
|
*,
|
|
selected_account: str = "",
|
|
selected_conversation: str = "",
|
|
query: str = "",
|
|
refresh_cache: bool = False,
|
|
auto_initialize: bool = False,
|
|
cache_roots: Iterable[Path] | None = None,
|
|
conversation_limit: int = 200,
|
|
message_limit: int = 500,
|
|
conversation_offset: int = 0,
|
|
message_offset: int = 0,
|
|
) -> dict[str, Any]:
|
|
"""Build the JSON-ready state used by the desktop message-library page."""
|
|
|
|
source_root = ""
|
|
warning = ""
|
|
roots = [Path(item) for item in (cache_roots or default_cache_roots())]
|
|
discovered = _discover_databases(roots)
|
|
if refresh_cache or auto_initialize or not discovered:
|
|
source_root, warning = _refresh_cache(auto_acquire=True) if auto_initialize else _refresh_cache()
|
|
clear_browser_cache()
|
|
discovered = _discover_databases(roots)
|
|
if not source_root and cache_roots is None:
|
|
try:
|
|
source_root = str(detect_wxwork_dir() or "")
|
|
except Exception:
|
|
source_root = ""
|
|
|
|
state: dict[str, Any] = {
|
|
"loading": False,
|
|
"error": "",
|
|
"warning": warning,
|
|
"manualSetupRequired": bool(warning),
|
|
"initializationAttempted": bool(auto_initialize),
|
|
"sourceRoot": source_root,
|
|
"refreshedAt": _format_time(time.time()),
|
|
"query": str(query or ""),
|
|
"selectedAccount": "",
|
|
"selectedConversation": "",
|
|
"accountCount": len(discovered),
|
|
"databaseCount": 0,
|
|
"conversationCount": 0,
|
|
"messageCount": 0,
|
|
"accounts": [],
|
|
"conversations": [],
|
|
"messages": [],
|
|
"files": [],
|
|
"conversationLimit": max(1, min(int(conversation_limit), 1000)),
|
|
"messageLimit": max(1, min(int(message_limit), 1000)),
|
|
"messageTotal": 0,
|
|
"conversationTotal": 0,
|
|
"conversationHasMore": False,
|
|
"messageHasMore": False,
|
|
"messageLoadDirection": "newest",
|
|
}
|
|
if not discovered:
|
|
state["error"] = warning or "尚未找到可读取的企业微信数据库,请刷新自动检测,或使用手动设置"
|
|
state["manualSetupRequired"] = True
|
|
return state
|
|
|
|
account_models: list[dict[str, Any]] = []
|
|
account_metadata: dict[str, tuple[dict[str, str], dict[str, str], str]] = {}
|
|
for account, databases in discovered.items():
|
|
files, total_size = _database_files(databases)
|
|
users, conversations, account_name = _load_names(databases, account)
|
|
account_metadata[account] = (users, conversations, account_name)
|
|
message_count = 0
|
|
conversation_count = 0
|
|
message_path = databases.get("message.db")
|
|
if message_path:
|
|
try:
|
|
message_count, conversation_count = _message_counts(message_path)
|
|
except (sqlite3.Error, TypeError, ValueError):
|
|
pass
|
|
updated = max(
|
|
(item.get("modified") or "" for item in files),
|
|
default="",
|
|
)
|
|
account_models.append(
|
|
{
|
|
"id": account,
|
|
"name": account_name,
|
|
"databaseCount": len(files),
|
|
"messageCount": message_count,
|
|
"conversationCount": conversation_count,
|
|
"size": total_size,
|
|
"sizeLabel": _size_label(total_size),
|
|
"updated": updated,
|
|
"files": files,
|
|
}
|
|
)
|
|
account_models.sort(key=lambda item: (item["updated"], item["messageCount"]), reverse=True)
|
|
account_ids = {str(item["id"]) for item in account_models}
|
|
account = str(selected_account or "")
|
|
if account not in account_ids:
|
|
account = str(account_models[0]["id"])
|
|
state["selectedAccount"] = account
|
|
state["accounts"] = account_models
|
|
state["databaseCount"] = sum(int(item["databaseCount"]) for item in account_models)
|
|
state["messageCount"] = sum(int(item["messageCount"]) for item in account_models)
|
|
state["conversationCount"] = sum(int(item["conversationCount"]) for item in account_models)
|
|
|
|
current = next(item for item in account_models if str(item["id"]) == account)
|
|
state["files"] = current["files"]
|
|
users, conversation_names, _account_name = account_metadata[account]
|
|
message_path = discovered[account].get("message.db")
|
|
if not message_path:
|
|
state["error"] = "当前账号没有可读取的 message.db"
|
|
state["manualSetupRequired"] = True
|
|
return state
|
|
try:
|
|
conversations, total_conversations, total_messages = _conversation_rows(
|
|
message_path,
|
|
account,
|
|
users,
|
|
conversation_names,
|
|
str(query or ""),
|
|
max(1, min(int(conversation_limit), 1000)),
|
|
offset=max(0, int(conversation_offset)),
|
|
)
|
|
state["conversations"] = conversations
|
|
state["conversationTotal"] = total_conversations
|
|
state["messageTotal"] = len((_message_index(message_path).get("conversations", {}).get(str(selected_conversation or ""), {}).get("rows", []))) if selected_conversation else 0
|
|
state["conversationHasMore"] = len(conversations) < total_conversations
|
|
available = {str(item["id"]) for item in conversations}
|
|
conversation = str(selected_conversation or "")
|
|
if conversation not in available:
|
|
conversation = str(conversations[0]["id"]) if conversations else ""
|
|
state["selectedConversation"] = conversation
|
|
state["messages"] = _messages(
|
|
message_path,
|
|
account,
|
|
conversation,
|
|
users,
|
|
message_limit,
|
|
offset=max(0, int(message_offset)),
|
|
media_root=(Path(source_root) / account / "Cache") if source_root else None,
|
|
)
|
|
state["messageTotal"] = len((_message_index(message_path).get("conversations", {}).get(conversation, {}).get("rows", [])))
|
|
state["messageHasMore"] = len(state["messages"]) < state["messageTotal"]
|
|
except sqlite3.Error as exc:
|
|
state["error"] = f"读取消息数据库失败:{exc}"
|
|
state["manualSetupRequired"] = True
|
|
return state
|