Files
kefu/wechat_rpa/wxwork_message_browser.py
T
2026-09-21 10:34:06 +08:00

850 lines
34 KiB
Python

# -*- coding: utf-8 -*-
"""Read-only browser model for locally decrypted WeCom message databases.
The auto-reply engine and archive uploader already maintain decrypted SQLite
copies. This module only reads those copies and therefore never writes to the
live WeCom databases. A manual refresh may update the existing decrypted
cache through ``wxwork_db.decrypt_with_keys`` before the snapshot is built.
"""
from __future__ import annotations
import sqlite3
import math
import threading
import base64
import mimetypes
import re
from array import array
from collections import OrderedDict
import time
from datetime import datetime
from pathlib import Path
from typing import Any, Iterable
from runtime_paths import application_data_dir
from wxwork_db import (
connect_sqlite,
decrypt_with_keys,
detect_wxwork_dir,
get_msg_type_name,
load_keys,
parse_content,
)
DATABASE_TABLES = {
"message.db": "message_table",
"session.db": "conversation_table",
"user.db": "user_table",
"company.db": "company_table",
}
# Cache only derived read-only data. A DB replacement or WAL commit invalidates
# its entry; no SQLite connection or file handle remains open between requests.
_CACHE_LOCK = threading.RLock()
_CACHES: dict[str, OrderedDict] = {}
_CACHE_LIMITS = {"index": 4, "names": 64, "files": 64, "counts": 64, "messages": 32}
_MEDIA_CACHE_LIMIT = 256
_MEDIA_REF = re.compile(r"(?<![0-9a-f])[0-9a-f]{8}-[0-9a-f]{4}-[0-9a-f]{4}-[0-9a-f]{4}-[0-9a-f]{12}(?![0-9a-f])|(?<![\w.-])[\w\u4e00-\u9fff()()【】\[\]-]{2,120}\.(?:png|jpe?g|gif|bmp|webp)(?!\w)", re.I)
_MEDIA_INDEX_CACHE: OrderedDict = OrderedDict()
_VOICE_READERS: dict[str, Any] = {}
def clear_browser_cache() -> None:
with _CACHE_LOCK:
_CACHES.clear()
_MEDIA_INDEX_CACHE.clear()
def _media_index(root: Path) -> dict[str, list[Path]]:
"""Index image cache files; references are matched by exact basename/stem."""
root = Path(root).resolve()
key = str(root).casefold()
try:
revision = tuple((str(p), p.stat().st_mtime_ns, p.stat().st_size)
for p in root.glob("**/*") if p.is_file())
except OSError:
revision = ()
with _CACHE_LOCK:
found = _MEDIA_INDEX_CACHE.get(key)
if found and found[0] == revision:
_MEDIA_INDEX_CACHE.move_to_end(key)
return found[1]
result: dict[str, list[Path]] = {}
if root.is_dir():
try:
files = [p for p in root.rglob("*") if p.is_file()][:20000]
except OSError:
files = []
for path in files:
if path.suffix.casefold() not in {".png", ".jpg", ".jpeg", ".gif", ".bmp", ".webp"}:
continue
tokens = {path.name.casefold(), path.stem.casefold()}
match = re.match(r"[0-9a-f]{8}-[0-9a-f]{4}-[0-9a-f]{4}-[0-9a-f]{4}-[0-9a-f]{12}", path.name, re.I)
if match:
tokens.add(match.group(0).casefold())
for token in tokens:
result.setdefault(token, []).append(path)
with _CACHE_LOCK:
_MEDIA_INDEX_CACHE[key] = (revision, result)
_MEDIA_INDEX_CACHE.move_to_end(key)
while len(_MEDIA_INDEX_CACHE) > _MEDIA_CACHE_LIMIT:
_MEDIA_INDEX_CACHE.popitem(last=False)
return result
def _media_data(raw: Any, media_root: Path | None, content_type: Any) -> dict[str, Any]:
"""Return a bounded data URI after an exact decoded image-cache match."""
try:
kind = int(content_type)
except (TypeError, ValueError):
kind = -1
if kind not in {3, 14, 123} or media_root is None:
return {}
if isinstance(raw, str):
raw = raw.encode("utf-8", errors="ignore")
if not isinstance(raw, (bytes, bytearray, memoryview)):
return {}
text = bytes(raw).decode("utf-8", errors="ignore")
refs = [match.group(0).casefold() for match in _MEDIA_REF.finditer(text)][:32]
if not refs:
return {}
index = _media_index(media_root)
matches = []
for ref in refs:
matches.extend(index.get(ref, ()))
unique = {str(path.resolve()): path for path in matches}
if len(unique) != 1:
return {}
path = next(iter(unique.values()))
try:
info = path.stat()
if info.st_size <= 0 or info.st_size > 2 * 1024 * 1024:
return {}
data = path.read_bytes()
except OSError:
return {}
if not data.startswith((b"\x89PNG\r\n\x1a\n", b"\xff\xd8\xff", b"GIF8", b"BM", b"RIFF")):
return {}
mime = mimetypes.guess_type(path.name)[0] or "image/png"
return {"src": "data:" + mime + ";base64," + base64.b64encode(data).decode("ascii"),
"name": path.name, "mime": mime}
def _database_revision(path: Path) -> tuple:
values = [str(path.resolve()).casefold()]
for candidate in (path, Path(str(path) + "-wal")):
try:
info = candidate.stat()
values.append((info.st_size, info.st_mtime_ns, info.st_ctime_ns, info.st_ino))
except OSError:
values.append(None)
return tuple(values)
def _cached(kind: str, key: Any, revision: tuple, build):
# Building under the lock coalesces simultaneous navigation requests. The
# GUI already loads in a worker; repeated searches cannot multiply scans.
with _CACHE_LOCK:
cache = _CACHES.setdefault(kind, OrderedDict())
found = cache.get(key)
if found is not None and found[0] == revision:
cache.move_to_end(key)
return found[1]
value = build()
cache[key] = (revision, value)
cache.move_to_end(key)
while len(cache) > _CACHE_LIMITS.get(kind, 16):
cache.popitem(last=False)
return value
def _message_index(message_path: Path) -> dict:
revision = _database_revision(message_path)
def build():
connection = connect_sqlite(str(message_path))
try:
columns = _columns(connection, "message_table")
result = {"columns": columns, "total_messages": 0, "conversations": {},
"ordered": [], "previews": {}, "revision": revision}
if not {"conversation_id", "send_time"}.issubset(columns):
return result
result["total_messages"] = _message_counts(message_path)[0]
# Sort/group inside SQLite instead of allocating one Python tuple
# per message. Only compact row IDs and one timestamp per chat are
# retained; message bodies are fetched for the visible page only.
cursor = connection.execute(
"SELECT conversation_id,GROUP_CONCAT(message_rowid),MAX(message_time) "
"FROM (SELECT rowid AS message_rowid,conversation_id,CAST(send_time AS REAL) AS message_time "
"FROM message_table WHERE conversation_id IS NOT NULL AND conversation_id<>'' "
"AND conversation_id NOT LIKE 'Y:%' "
"ORDER BY conversation_id,CAST(send_time AS REAL),rowid) GROUP BY conversation_id"
)
for conversation_id, encoded_rows, timestamp in cursor:
row_ids = array("q", map(int, str(encoded_rows).split(",")))
timestamp = float(timestamp or 0)
if not math.isfinite(timestamp):
timestamp = 0.0
result["conversations"][str(conversation_id)] = {
"rows": row_ids, "lastTimestamp": timestamp, "lastRowId": row_ids[-1],
}
result["ordered"] = sorted(result["conversations"], key=lambda key: (
result["conversations"][key]["lastTimestamp"],
result["conversations"][key]["lastRowId"], key), reverse=True)
return result
finally:
connection.close()
return _cached("index", revision[0], revision, build)
def _read_message_rows(message_path: Path, index: dict, row_ids: list[int]) -> dict[int, dict]:
if not row_ids:
return {}
columns = index["columns"]
wanted = [name for name in ("conversation_id", "sender_id", "send_time", "content_type", "content", "server_id", "client_id") if name in columns]
if not wanted:
return {}
connection = connect_sqlite(str(message_path))
try:
result = {}
for offset in range(0, len(row_ids), 500):
batch = row_ids[offset:offset + 500]
marks = ",".join("?" for _ in batch)
for row in connection.execute(f"SELECT rowid,{','.join(wanted)} FROM message_table WHERE rowid IN ({marks})", batch):
result[int(row[0])] = dict(zip(wanted, row[1:]))
return result
finally:
connection.close()
def _ensure_previews(message_path: Path, index: dict, conversation_ids: list[str]) -> None:
with _CACHE_LOCK:
missing = [key for key in conversation_ids if key not in index["previews"]]
for offset in range(0, len(missing), 500):
batch = missing[offset:offset + 500]
rows = _read_message_rows(message_path, index, [index["conversations"][key]["lastRowId"] for key in batch])
for key in batch:
item = rows.get(index["conversations"][key]["lastRowId"], {})
# A cache file can be atomically replaced between indexing and
# reading. Reused row IDs must never expose another chat.
if str(item.get("conversation_id") or "") != key:
index["previews"][key] = {"text": "", "sender": ""}
continue
index["previews"][key] = {
"text": _content_label(item.get("content"), item.get("content_type")).replace("\n", " ")[:2000],
"sender": str(item.get("sender_id") or ""),
}
def _load_names(databases: dict[str, Path], account: str):
revision = tuple((name, _database_revision(databases[name])) for name in ("user.db", "session.db") if name in databases)
key = (account, tuple((name, str(path.resolve()).casefold()) for name, path in databases.items() if name in {"user.db", "session.db"}))
return _cached("names", key, revision, lambda: _load_names_uncached(databases, account))
def _database_files(databases: dict[str, Path]):
revision = tuple((name, _database_revision(path)) for name, path in sorted(databases.items()))
key = tuple((name, str(path.resolve()).casefold()) for name, path in sorted(databases.items()))
files, size = _cached("files", key, revision, lambda: _database_files_uncached(databases))
return [dict(item) for item in files], size
def _message_counts(message_path: Path) -> tuple[int, int]:
revision = _database_revision(message_path)
def build():
connection = connect_sqlite(str(message_path))
try:
columns = _columns(connection, "message_table")
if "conversation_id" not in columns:
return _count(connection, "message_table"), 0
values = connection.execute(
"SELECT COUNT(*),COUNT(DISTINCT CASE WHEN conversation_id IS NOT NULL "
"AND conversation_id<>'' AND conversation_id NOT LIKE 'Y:%' "
"THEN conversation_id END) FROM message_table"
).fetchone()
return int(values[0]), int(values[1])
finally:
connection.close()
return _cached("counts", revision[0], revision, build)
def _format_time(value: Any) -> str:
try:
timestamp = float(value or 0)
except (TypeError, ValueError):
return ""
if timestamp > 10**12:
timestamp /= 1000
if timestamp <= 0:
return ""
try:
return datetime.fromtimestamp(timestamp).strftime("%Y-%m-%d %H:%M:%S")
except (OSError, OverflowError, ValueError):
return ""
def _size_label(size: int) -> str:
if size < 1024:
return f"{size} B"
if size < 1024 * 1024:
return f"{size / 1024:.1f} KB"
if size < 1024 * 1024 * 1024:
return f"{size / (1024 * 1024):.1f} MB"
return f"{size / (1024 * 1024 * 1024):.2f} GB"
def _columns(connection: sqlite3.Connection, table: str) -> set[str]:
try:
return {
str(row[1])
for row in connection.execute(f'PRAGMA table_info("{table}")').fetchall()
}
except sqlite3.Error:
return set()
def _count(connection: sqlite3.Connection, table: str) -> int:
try:
return int(connection.execute(f'SELECT COUNT(*) FROM "{table}"').fetchone()[0])
except (sqlite3.Error, TypeError, ValueError):
return 0
def default_cache_roots() -> list[Path]:
data_root = application_data_dir()
source_root = Path(__file__).resolve().parent
candidates = [
data_root / "archive_auto_backup" / "decrypted",
data_root / "wxwork_decrypted",
source_root / "archive_auto_backup" / "decrypted",
source_root / "wxwork_decrypted",
]
result: list[Path] = []
seen: set[str] = set()
for candidate in candidates:
identity = str(candidate.resolve()).casefold()
if identity in seen:
continue
seen.add(identity)
result.append(candidate)
return result
def _discover_databases(cache_roots: Iterable[Path]) -> dict[str, dict[str, Path]]:
"""Choose the newest readable copy of every database for each account."""
discovered: dict[str, dict[str, Path]] = {}
for root in cache_roots:
if not root.is_dir():
continue
try:
account_dirs = list(root.iterdir())
except OSError:
continue
for account_dir in account_dirs:
if not account_dir.is_dir():
continue
account = account_dir.name
for database_path in account_dir.glob("*.db"):
current = discovered.setdefault(account, {}).get(database_path.name)
try:
newer = current is None or database_path.stat().st_mtime > current.stat().st_mtime
except OSError:
continue
if newer:
discovered[account][database_path.name] = database_path
return discovered
def _refresh_cache(*, auto_acquire: bool = False) -> tuple[str, str]:
"""先复用密钥解密;自动初始化时,仅在无法读取后获取一次本机密钥。"""
source_root = detect_wxwork_dir()
if not source_root:
return "", "自动检测未找到企业微信数据库。请先在本机登录企业微信后刷新,或手动选择数据目录"
from wxwork_local_setup import acquire_local_keys, select_source_directory
try:
# 让独立 EXE 密钥工作进程使用同一个已识别的目录。
if auto_acquire:
select_source_directory(source_root)
expected = {p.parent.parent.name for p in Path(source_root).glob("*/Data/message.db")}
def decrypt() -> tuple[set[str], dict]:
try:
keys = load_keys()
except ValueError:
keys = {}
output = str(application_data_dir() / "wxwork_decrypted")
decrypted = decrypt_with_keys(source_root, output, keys, use_cache=True)
readable = set()
invalid = False
for path, name, account in decrypted:
if name != "message.db":
continue
try:
connection = connect_sqlite(path)
try:
connection.execute("SELECT conversation_id FROM message_table LIMIT 1").fetchone()
readable.add(account)
finally:
connection.close()
except sqlite3.Error:
invalid = True
# 缓存大小/时间正常也可能已经损坏;重建后再次检查。
if invalid:
decrypted = decrypt_with_keys(source_root, output, keys, use_cache=False)
readable.clear()
for path, name, account in decrypted:
if name != "message.db":
continue
try:
connection = connect_sqlite(path)
try:
connection.execute("SELECT conversation_id FROM message_table LIMIT 1").fetchone()
readable.add(account)
finally:
connection.close()
except sqlite3.Error:
continue
return readable, keys
readable, keys = decrypt()
if expected and expected <= readable:
return source_root, ""
if auto_acquire:
try:
acquire_local_keys()
except (OSError, ValueError) as exc:
return source_root, f"自动解密未完成:{exc}。可重新刷新自动检测,或使用下方手动设置"
readable, keys = decrypt()
if expected and expected <= readable:
return source_root, ""
return source_root, "自动获取密钥后仍有数据库无法解密,请确认企业微信已登录,或使用下方手动设置"
if not keys:
return source_root, "已找到企业微信数据库,但本机尚未配置密钥。请点击刷新自动获取,或导入当前电脑的密钥文件"
return source_root, "已找到企业微信数据库,但现有密钥无法解密全部数据库。请刷新重新获取本机密钥或导入匹配的密钥文件"
except Exception as exc:
return source_root, f"更新解密缓存失败:{exc}"
def _load_names_uncached(databases: dict[str, Path], account: str) -> tuple[dict[str, str], dict[str, str], str]:
users: dict[str, str] = {}
conversations: dict[str, str] = {}
account_name = account
user_path = databases.get("user.db")
if user_path:
try:
connection = connect_sqlite(str(user_path))
try:
columns = _columns(connection, "user_table")
wanted = [name for name in ("id", "name", "real_name", "account") if name in columns]
if "id" in wanted:
cursor = connection.execute(f"SELECT {','.join(wanted)} FROM user_table")
for values in cursor.fetchall():
item = dict(zip(wanted, values))
user_id = str(item.get("id") or "")
label = str(
item.get("name")
or item.get("real_name")
or item.get("account")
or user_id
)
if user_id:
users[user_id] = label
account_name = users.get(account) or account_name
finally:
connection.close()
except sqlite3.Error:
pass
session_path = databases.get("session.db")
if session_path:
try:
connection = connect_sqlite(str(session_path))
try:
columns = _columns(connection, "conversation_table")
wanted = [
name
for name in ("id", "name", "roomname_remark", "session_id")
if name in columns
]
if "id" in wanted:
cursor = connection.execute(
f"SELECT {','.join(wanted)} FROM conversation_table"
)
for values in cursor.fetchall():
item = dict(zip(wanted, values))
conversation_id = str(item.get("id") or "")
label = str(
item.get("roomname_remark")
or item.get("name")
or item.get("session_id")
or ""
)
if conversation_id and label:
conversations[conversation_id] = label
finally:
connection.close()
except sqlite3.Error:
pass
return users, conversations, account_name
def _conversation_name(
account: str,
conversation_id: str,
users: dict[str, str],
conversations: dict[str, str],
) -> str:
label = str(conversations.get(conversation_id) or "").strip()
if label and label != conversation_id:
return label
if conversation_id.startswith("M:"):
peer_id = conversation_id[2:]
return users.get(peer_id) or f"微信用户 {peer_id}"
if conversation_id.startswith("S:"):
peers = [item for item in conversation_id[2:].split("_") if item]
peer_id = next((item for item in peers if item != account), peers[0] if peers else "")
return users.get(peer_id) or f"企微用户 {peer_id}"
if conversation_id.startswith("R:"):
return label or f"群聊 {conversation_id[2:]}"
if conversation_id.startswith("Y:"):
return label or f"应用 {conversation_id[2:]}"
return label or conversation_id or "未知会话"
def _conversation_kind(conversation_id: str) -> str:
if conversation_id.startswith("M:"):
return "微信客户"
if conversation_id.startswith("S:"):
return "企业微信"
if conversation_id.startswith("R:"):
return "群聊"
if conversation_id.startswith("Y:"):
return "应用"
return "其他"
def _content_label(content: Any, content_type: Any) -> str:
from archive_content_parser import parse_message_content
parsed = parse_message_content(content, content_type)["text"].strip()
if parsed:
return parsed
return f"[{get_msg_type_name(content_type)}]"
def _database_files_uncached(databases: dict[str, Path]) -> tuple[list[dict[str, Any]], int]:
files: list[dict[str, Any]] = []
total_size = 0
preferred_order = {
name: index
for index, name in enumerate(
("message.db", "session.db", "user.db", "company.db", "message_lookup.db", "user_extend.db")
)
}
for name, path in sorted(
databases.items(), key=lambda item: (preferred_order.get(item[0], 99), item[0])
):
try:
stat = path.stat()
size = int(stat.st_size)
modified = _format_time(stat.st_mtime)
except OSError:
size, modified = 0, ""
total_size += size
row_count = 0
status = "可读取"
table = DATABASE_TABLES.get(name)
try:
connection = connect_sqlite(str(path))
try:
if table:
row_count = _message_counts(path)[0] if table == "message_table" else _count(connection, table)
connection.execute("SELECT COUNT(*) FROM sqlite_master").fetchone()
finally:
connection.close()
except sqlite3.Error:
status = "读取失败"
files.append(
{
"name": name,
"path": str(path),
"size": size,
"sizeLabel": _size_label(size),
"modified": modified,
"rows": row_count,
"status": status,
}
)
return files, total_size
def _conversation_rows(
message_path: Path,
account: str,
users: dict[str, str],
conversations: dict[str, str],
query: str,
limit: int,
offset: int = 0,
) -> tuple[list[dict[str, Any]], int, int]:
index = _message_index(message_path)
ordered = index["ordered"]
needle = str(query or "").strip().casefold()
offset = max(0, int(offset))
candidates = ordered if needle else ordered[offset:offset + limit]
_ensure_previews(message_path, index, candidates)
result = []
matching = 0
for conversation_id in candidates:
name = _conversation_name(account, conversation_id, users, conversations)
preview = index["previews"][conversation_id]
if needle and needle not in f"{name} {conversation_id} {preview['text']}".casefold():
continue
if needle:
matching += 1
if matching <= offset:
continue
group = index["conversations"][conversation_id]
result.append({
"id": conversation_id, "name": name, "kind": _conversation_kind(conversation_id),
"messageCount": len(group["rows"]), "lastTime": _format_time(group["lastTimestamp"]),
"lastTimestamp": group["lastTimestamp"], "preview": preview["text"][:160],
"lastDirection": "发出" if preview["sender"] == account else "收到",
})
if len(result) >= limit:
break
return result, len(ordered), index["total_messages"]
def _messages(
message_path: Path,
account: str,
conversation_id: str,
users: dict[str, str],
limit: int,
offset: int = 0,
media_root: Path | None = None,
) -> list[dict[str, Any]]:
if not conversation_id:
return []
index = _message_index(message_path)
rows = (index["conversations"].get(conversation_id) or {}).get("rows", [])
offset = max(0, int(offset))
limit = max(1, min(int(limit), 1000))
end = max(0, len(rows) - offset)
row_ids = list(rows[max(0, end - limit):end])
revision = index["revision"]
key = (revision[0], conversation_id, limit, offset)
records = _cached("messages", key, revision, lambda: _read_message_rows(message_path, index, row_ids))
result = []
connection = None
try:
from archive_content_parser import parse_message_content
from voice_messages import is_voice, cached_voice_fields
if any(is_voice(row) for row in records.values()):
connection = connect_sqlite(str(message_path))
for row_id in row_ids:
item = records.get(row_id)
if item is None or str(item.get("conversation_id") or "") != conversation_id:
continue
sender_id = str(item.get("sender_id") or "")
try:
timestamp = float(item.get("send_time") or 0)
if not math.isfinite(timestamp):
timestamp = 0.0
except (ValueError, TypeError, OverflowError):
timestamp = 0.0
content_type = item.get("content_type")
decoded = parse_message_content(item.get("content"), content_type)
fields = {}
server_id = str(item.get("server_id") or "")
if is_voice(item) and connection is not None:
fields = cached_voice_fields(connection, account, conversation_id, server_id, item.get("content"))
content = fields.get("content") or decoded.get("text") or _content_label(item.get("content"), content_type)
message = {
"id": server_id or str(item.get("client_id") or row_id),
"senderId": sender_id,
"sender": users.get(sender_id) or ("当前账号" if sender_id == account else sender_id),
"direction": "outbound" if sender_id == account else "inbound",
"time": _format_time(timestamp), "timestamp": timestamp,
"type": get_msg_type_name(content_type), "content": content,
"account": account, "conv_id": conversation_id, "server_id": server_id,
"content_type": content_type, "voice_refs": fields.get("voice_refs", []),
**fields,
}
result.append(message)
finally:
if connection is not None:
connection.close()
# Queue every visible voice in the history, including messages sent by us;
# the reply engine's default unanswered-only behavior remains unchanged.
if any(int(item.get("content_type") or 0) in {4, 16} for item in result):
source_key = str(media_root.parent.parent if media_root else "").casefold()
if source_key:
reader = _VOICE_READERS.get(source_key)
if reader is None:
from voice_messages import VoiceMessageReader
reader = _VOICE_READERS[source_key] = VoiceMessageReader(Path(source_key), background_index=True)
reader.prepare(result, include_history=True)
for message in result:
if message.get("voice_transcribed") and message.get("content"):
continue
image = _media_data(records.get(next((rid for rid, row in records.items() if str(row.get("server_id") or row.get("client_id") or rid) == message["id"]), 0), {}).get("content"), media_root, message.get("content_type"))
if image:
message["media"] = {"kind": "image", **image}
for message in result:
for key_name in ("account", "conv_id", "server_id", "content_type", "voice_refs"):
message.pop(key_name, None)
return result
def load_browser_snapshot(
*,
selected_account: str = "",
selected_conversation: str = "",
query: str = "",
refresh_cache: bool = False,
auto_initialize: bool = False,
cache_roots: Iterable[Path] | None = None,
conversation_limit: int = 200,
message_limit: int = 500,
conversation_offset: int = 0,
message_offset: int = 0,
) -> dict[str, Any]:
"""Build the JSON-ready state used by the desktop message-library page."""
source_root = ""
warning = ""
roots = [Path(item) for item in (cache_roots or default_cache_roots())]
discovered = _discover_databases(roots)
if refresh_cache or auto_initialize or not discovered:
source_root, warning = _refresh_cache(auto_acquire=True) if auto_initialize else _refresh_cache()
clear_browser_cache()
discovered = _discover_databases(roots)
if not source_root and cache_roots is None:
try:
source_root = str(detect_wxwork_dir() or "")
except Exception:
source_root = ""
state: dict[str, Any] = {
"loading": False,
"error": "",
"warning": warning,
"manualSetupRequired": bool(warning),
"initializationAttempted": bool(auto_initialize),
"sourceRoot": source_root,
"refreshedAt": _format_time(time.time()),
"query": str(query or ""),
"selectedAccount": "",
"selectedConversation": "",
"accountCount": len(discovered),
"databaseCount": 0,
"conversationCount": 0,
"messageCount": 0,
"accounts": [],
"conversations": [],
"messages": [],
"files": [],
"conversationLimit": max(1, min(int(conversation_limit), 1000)),
"messageLimit": max(1, min(int(message_limit), 1000)),
"messageTotal": 0,
"conversationTotal": 0,
"conversationHasMore": False,
"messageHasMore": False,
"messageLoadDirection": "newest",
}
if not discovered:
state["error"] = warning or "尚未找到可读取的企业微信数据库,请刷新自动检测,或使用手动设置"
state["manualSetupRequired"] = True
return state
account_models: list[dict[str, Any]] = []
account_metadata: dict[str, tuple[dict[str, str], dict[str, str], str]] = {}
for account, databases in discovered.items():
files, total_size = _database_files(databases)
users, conversations, account_name = _load_names(databases, account)
account_metadata[account] = (users, conversations, account_name)
message_count = 0
conversation_count = 0
message_path = databases.get("message.db")
if message_path:
try:
message_count, conversation_count = _message_counts(message_path)
except (sqlite3.Error, TypeError, ValueError):
pass
updated = max(
(item.get("modified") or "" for item in files),
default="",
)
account_models.append(
{
"id": account,
"name": account_name,
"databaseCount": len(files),
"messageCount": message_count,
"conversationCount": conversation_count,
"size": total_size,
"sizeLabel": _size_label(total_size),
"updated": updated,
"files": files,
}
)
account_models.sort(key=lambda item: (item["updated"], item["messageCount"]), reverse=True)
account_ids = {str(item["id"]) for item in account_models}
account = str(selected_account or "")
if account not in account_ids:
account = str(account_models[0]["id"])
state["selectedAccount"] = account
state["accounts"] = account_models
state["databaseCount"] = sum(int(item["databaseCount"]) for item in account_models)
state["messageCount"] = sum(int(item["messageCount"]) for item in account_models)
state["conversationCount"] = sum(int(item["conversationCount"]) for item in account_models)
current = next(item for item in account_models if str(item["id"]) == account)
state["files"] = current["files"]
users, conversation_names, _account_name = account_metadata[account]
message_path = discovered[account].get("message.db")
if not message_path:
state["error"] = "当前账号没有可读取的 message.db"
state["manualSetupRequired"] = True
return state
try:
conversations, total_conversations, total_messages = _conversation_rows(
message_path,
account,
users,
conversation_names,
str(query or ""),
max(1, min(int(conversation_limit), 1000)),
offset=max(0, int(conversation_offset)),
)
state["conversations"] = conversations
state["conversationTotal"] = total_conversations
state["messageTotal"] = len((_message_index(message_path).get("conversations", {}).get(str(selected_conversation or ""), {}).get("rows", []))) if selected_conversation else 0
state["conversationHasMore"] = len(conversations) < total_conversations
available = {str(item["id"]) for item in conversations}
conversation = str(selected_conversation or "")
if conversation not in available:
conversation = str(conversations[0]["id"]) if conversations else ""
state["selectedConversation"] = conversation
state["messages"] = _messages(
message_path,
account,
conversation,
users,
message_limit,
offset=max(0, int(message_offset)),
media_root=(Path(source_root) / account / "Cache") if source_root else None,
)
state["messageTotal"] = len((_message_index(message_path).get("conversations", {}).get(conversation, {}).get("rows", [])))
state["messageHasMore"] = len(state["messages"]) < state["messageTotal"]
except sqlite3.Error as exc:
state["error"] = f"读取消息数据库失败:{exc}"
state["manualSetupRequired"] = True
return state