Files
kefu/deploy/loading-performance-20260917/browser_cache_helpers.txt
T
2026-09-21 10:34:06 +08:00

125 lines
5.5 KiB
Plaintext

# Cache only derived read-only data. A DB replacement or WAL commit invalidates
# its entry; no SQLite connection or file handle remains open between requests.
_CACHE_LOCK = threading.RLock()
_CACHES: dict[str, OrderedDict] = {}
_CACHE_LIMITS = {"index": 4, "names": 16, "files": 16, "messages": 32}
def clear_browser_cache() -> None:
with _CACHE_LOCK:
_CACHES.clear()
def _database_revision(path: Path) -> tuple:
values = [str(path.resolve()).casefold()]
for candidate in (path, Path(str(path) + "-wal")):
try:
info = candidate.stat()
values.append((info.st_size, info.st_mtime_ns, info.st_ctime_ns, info.st_ino))
except OSError:
values.append(None)
return tuple(values)
def _cached(kind: str, key: Any, revision: tuple, build):
# Building under the lock coalesces simultaneous navigation requests. The
# GUI already loads in a worker; repeated searches cannot multiply scans.
with _CACHE_LOCK:
cache = _CACHES.setdefault(kind, OrderedDict())
found = cache.get(key)
if found is not None and found[0] == revision:
cache.move_to_end(key)
return found[1]
value = build()
cache[key] = (revision, value)
cache.move_to_end(key)
while len(cache) > _CACHE_LIMITS.get(kind, 16):
cache.popitem(last=False)
return value
def _message_index(message_path: Path) -> dict:
revision = _database_revision(message_path)
def build():
connection = connect_sqlite(str(message_path))
try:
columns = _columns(connection, "message_table")
result = {"columns": columns, "total_messages": 0, "conversations": {},
"ordered": [], "previews": {}, "revision": revision}
if not {"conversation_id", "send_time"}.issubset(columns):
return result
groups: dict[str, list] = {}
cursor = connection.execute("SELECT rowid,conversation_id,send_time FROM message_table")
for row_id, conversation_id, timestamp in cursor:
result["total_messages"] += 1
conversation_id = str(conversation_id or "")
if not conversation_id or conversation_id.startswith("Y:"):
continue
try:
timestamp = float(timestamp or 0)
if not math.isfinite(timestamp):
timestamp = 0.0
except (ValueError, TypeError, OverflowError):
timestamp = 0.0
groups.setdefault(conversation_id, []).append((timestamp, row_id))
for conversation_id, rows in groups.items():
rows.sort()
result["conversations"][conversation_id] = {
"rows": array("q", (row_id for _timestamp, row_id in rows)),
"lastTimestamp": rows[-1][0], "lastRowId": rows[-1][1],
}
result["ordered"] = sorted(result["conversations"], key=lambda key: (
result["conversations"][key]["lastTimestamp"],
result["conversations"][key]["lastRowId"], key), reverse=True)
return result
finally:
connection.close()
return _cached("index", revision[0], revision, build)
def _read_message_rows(message_path: Path, index: dict, row_ids: list[int]) -> dict[int, dict]:
if not row_ids:
return {}
columns = index["columns"]
wanted = [name for name in ("sender_id", "send_time", "content_type", "content", "server_id", "client_id") if name in columns]
if not wanted:
return {}
connection = connect_sqlite(str(message_path))
try:
result = {}
for offset in range(0, len(row_ids), 500):
batch = row_ids[offset:offset + 500]
marks = ",".join("?" for _ in batch)
for row in connection.execute(f"SELECT rowid,{','.join(wanted)} FROM message_table WHERE rowid IN ({marks})", batch):
result[int(row[0])] = dict(zip(wanted, row[1:]))
return result
finally:
connection.close()
def _ensure_previews(message_path: Path, index: dict, conversation_ids: list[str]) -> None:
with _CACHE_LOCK:
missing = [key for key in conversation_ids if key not in index["previews"]]
rows = _read_message_rows(message_path, index, [index["conversations"][key]["lastRowId"] for key in missing])
for key in missing:
item = rows.get(index["conversations"][key]["lastRowId"], {})
index["previews"][key] = {
"text": _content_label(item.get("content"), item.get("content_type")).replace("\n", " ")[:2000],
"sender": str(item.get("sender_id") or ""),
}
def _load_names(databases: dict[str, Path], account: str):
revision = tuple((name, _database_revision(databases[name])) for name in ("user.db", "session.db") if name in databases)
key = (account, tuple((name, str(path.resolve()).casefold()) for name, path in databases.items() if name in {"user.db", "session.db"}))
return _cached("names", key, revision, lambda: _load_names_uncached(databases, account))
def _database_files(databases: dict[str, Path]):
revision = tuple((name, _database_revision(path)) for name, path in sorted(databases.items()))
key = tuple((name, str(path.resolve()).casefold()) for name, path in sorted(databases.items()))
files, size = _cached("files", key, revision, lambda: _database_files_uncached(databases))
return [dict(item) for item in files], size