Files
kefu/deploy/protocol-integration-20260916/payload/knowledge_store.py
T
2026-09-21 10:34:06 +08:00

510 lines
30 KiB
Python

"""Tenant-scoped knowledge drafts, publication, source lineage and durable jobs."""
from __future__ import annotations
import json
from datetime import date, datetime, time, timedelta, timezone
from archive_store import ArchiveStore, json_text, new_id, safe_scope, tenant_scope, utc_now
from knowledge_processing import PIPELINE_VERSION, fingerprint, redact, tokens
SCHEMA = """
CREATE INDEX IF NOT EXISTS idx_knowledge_archive_scan
ON archive_message(tenant_id,source_account_id,conversation_id,sent_at,id);
CREATE TABLE IF NOT EXISTS knowledge_discarded (
tenant_id TEXT NOT NULL, dedup_hash TEXT NOT NULL, created_at TEXT NOT NULL,
PRIMARY KEY(tenant_id,dedup_hash)
);
CREATE TABLE IF NOT EXISTS knowledge_settings (
tenant_id TEXT PRIMARY KEY, enabled INTEGER NOT NULL DEFAULT 0, updated_at TEXT NOT NULL
);
CREATE TABLE IF NOT EXISTS knowledge_job (
id TEXT PRIMARY KEY, tenant_id TEXT NOT NULL, status TEXT NOT NULL DEFAULT 'queued',
options_json TEXT NOT NULL, cutoff_at TEXT NOT NULL, cursor_json TEXT NOT NULL DEFAULT '{}',
state_json TEXT NOT NULL DEFAULT '{}', processed INTEGER NOT NULL DEFAULT 0,
created_items INTEGER NOT NULL DEFAULT 0, duplicates INTEGER NOT NULL DEFAULT 0,
skipped INTEGER NOT NULL DEFAULT 0, model_calls INTEGER NOT NULL DEFAULT 0,
input_chars INTEGER NOT NULL DEFAULT 0, output_chars INTEGER NOT NULL DEFAULT 0,
lease_owner TEXT NOT NULL DEFAULT '', lease_until REAL NOT NULL DEFAULT 0,
error TEXT NOT NULL DEFAULT '', created_by INTEGER NOT NULL, created_at TEXT NOT NULL,
updated_at TEXT NOT NULL
);
CREATE INDEX IF NOT EXISTS idx_knowledge_job_queue ON knowledge_job(status,lease_until,created_at);
CREATE TABLE IF NOT EXISTS knowledge_item (
id TEXT PRIMARY KEY, tenant_id TEXT NOT NULL, job_id TEXT, dedup_hash TEXT NOT NULL,
title TEXT NOT NULL, question TEXT NOT NULL, answer TEXT NOT NULL, conditions TEXT NOT NULL DEFAULT '',
category TEXT NOT NULL DEFAULT '', kind TEXT NOT NULL DEFAULT 'qa', flags_json TEXT NOT NULL DEFAULT '[]',
status TEXT NOT NULL DEFAULT 'draft', revision INTEGER NOT NULL DEFAULT 1,
valid_until TEXT NOT NULL DEFAULT '', vector_status TEXT NOT NULL DEFAULT 'pending',
vector_profile TEXT NOT NULL DEFAULT '', created_at TEXT NOT NULL, updated_at TEXT NOT NULL,
UNIQUE(tenant_id,dedup_hash)
);
CREATE INDEX IF NOT EXISTS idx_knowledge_item_list ON knowledge_item(tenant_id,status,updated_at DESC,id);
CREATE TABLE IF NOT EXISTS knowledge_source (
item_id TEXT NOT NULL REFERENCES knowledge_item(id), message_id TEXT NOT NULL,
version_no INTEGER NOT NULL, role TEXT NOT NULL, content TEXT NOT NULL, sent_at TEXT NOT NULL,
PRIMARY KEY(item_id,message_id,version_no)
);
CREATE INDEX IF NOT EXISTS idx_knowledge_source_message ON knowledge_source(message_id,item_id);
CREATE TABLE IF NOT EXISTS knowledge_revision (
id TEXT PRIMARY KEY, item_id TEXT NOT NULL REFERENCES knowledge_item(id), revision INTEGER NOT NULL,
action TEXT NOT NULL, snapshot_json TEXT NOT NULL, actor_id INTEGER NOT NULL, created_at TEXT NOT NULL
);
CREATE TABLE IF NOT EXISTS knowledge_retrieval_log (
id TEXT PRIMARY KEY, tenant_id TEXT NOT NULL, task_id TEXT NOT NULL DEFAULT '',
query TEXT NOT NULL, hits_json TEXT NOT NULL, mode TEXT NOT NULL, elapsed_ms INTEGER NOT NULL,
created_at TEXT NOT NULL
);
CREATE INDEX IF NOT EXISTS idx_knowledge_log_tenant ON knowledge_retrieval_log(tenant_id,created_at DESC);
CREATE TABLE IF NOT EXISTS knowledge_watch (
tenant_id TEXT NOT NULL, source_account_id TEXT NOT NULL, enabled INTEGER NOT NULL DEFAULT 0,
created_by INTEGER NOT NULL, PRIMARY KEY(tenant_id,source_account_id)
);
CREATE TABLE IF NOT EXISTS knowledge_dirty (
tenant_id TEXT NOT NULL, source_account_id TEXT NOT NULL, conversation_id TEXT NOT NULL,
changed_at TEXT NOT NULL, PRIMARY KEY(tenant_id,source_account_id,conversation_id)
);
CREATE TRIGGER IF NOT EXISTS knowledge_new_message AFTER INSERT ON archive_message
WHEN EXISTS(SELECT 1 FROM knowledge_watch WHERE tenant_id=NEW.tenant_id AND source_account_id=NEW.source_account_id AND enabled=1)
BEGIN
INSERT INTO knowledge_dirty VALUES (NEW.tenant_id,NEW.source_account_id,NEW.conversation_id,NEW.updated_at)
ON CONFLICT(tenant_id,source_account_id,conversation_id) DO UPDATE SET changed_at=excluded.changed_at;
END;
CREATE TRIGGER IF NOT EXISTS knowledge_updated_message AFTER UPDATE OF content,status ON archive_message
WHEN (OLD.content!=NEW.content OR OLD.status!=NEW.status) AND EXISTS(
SELECT 1 FROM knowledge_watch WHERE tenant_id=NEW.tenant_id AND source_account_id=NEW.source_account_id AND enabled=1)
BEGIN
INSERT INTO knowledge_dirty VALUES (NEW.tenant_id,NEW.source_account_id,NEW.conversation_id,NEW.updated_at)
ON CONFLICT(tenant_id,source_account_id,conversation_id) DO UPDATE SET changed_at=excluded.changed_at;
END;
CREATE VIRTUAL TABLE IF NOT EXISTS knowledge_fts USING fts5(item_id UNINDEXED, terms);
CREATE TRIGGER IF NOT EXISTS knowledge_source_changed AFTER UPDATE OF content,status ON archive_message
WHEN OLD.content != NEW.content OR OLD.status != NEW.status
BEGIN
UPDATE knowledge_item SET status='stale',updated_at=NEW.updated_at
WHERE id IN (SELECT item_id FROM knowledge_source WHERE message_id=NEW.id);
DELETE FROM knowledge_fts WHERE item_id IN
(SELECT item_id FROM knowledge_source WHERE message_id=NEW.id);
END;
CREATE TRIGGER IF NOT EXISTS knowledge_source_deleted BEFORE DELETE ON archive_message
BEGIN
UPDATE knowledge_item SET status='stale' WHERE id IN
(SELECT item_id FROM knowledge_source WHERE message_id=OLD.id);
DELETE FROM knowledge_fts WHERE item_id IN
(SELECT item_id FROM knowledge_source WHERE message_id=OLD.id);
END;
"""
def day_boundary(value: str, *, end=False) -> str:
if not value:
return ""
# UI dates are Asia/Shanghai business dates. Store / compare UTC half-open intervals.
day = date.fromisoformat(value)
if end:
day += timedelta(days=1)
return datetime.combine(day, time(), timezone(timedelta(hours=8))).astimezone(
timezone.utc).isoformat(timespec="milliseconds")
def job_scope(tenant, options, cutoff):
"""Use identical scope and cutoff predicates for preview, progress and worker."""
where = ["m.tenant_id=?", "m.source_account_id=?", "m.created_at<=?",
"c.conversation_type IN ('direct_wechat','direct_wecom')"]
args = [tenant, options["source_account_id"], cutoff]
for key, condition in (("date_start", "m.sent_at>=?"), ("date_end", "m.sent_at<?")):
if options.get(key):
where.append(condition)
args.append(options[key])
if options.get("conversation_id"):
where.append("m.conversation_id=?")
args.append(options["conversation_id"])
return where, args
def job_options(values):
options = dict(values)
options.setdefault("engine", "rules")
start = day_boundary(options.get("date_from", ""))
end = day_boundary(options.get("date_to", ""), end=True)
if start and end and start >= end:
raise ValueError("结束日期不能早于开始日期")
if options.get("engine", "rules") not in {"rules", "model"}:
raise ValueError("未知整理方式")
for key, default, maximum in (("max_messages", 10000, 1000000), ("max_model_calls", 100, 10000), ("model_timeout_seconds", 90, 180)):
value = options.get(key, default)
if not isinstance(value, int) or isinstance(value, bool) or not 1 <= value <= maximum:
raise ValueError("处理上限超出允许范围")
options[key] = value
if options['model_timeout_seconds'] < 30:
raise ValueError("模型等待时间应为 30 至 180 秒")
provider = options.get('model_provider_id', '')
if not isinstance(provider, str) or len(provider) > 128:
raise ValueError("模型标识无效")
options.update(date_start=start, date_end=end, pipeline_version=PIPELINE_VERSION)
return options
class KnowledgeStore:
def __init__(self, database):
self.database = database
def initialize(self):
ArchiveStore(self.database).initialize()
with self.database.connect() as db:
from knowledge_review import REVIEW_SCHEMA
db.executescript(SCHEMA + REVIEW_SCHEMA)
@staticmethod
def public(row):
result = dict(row)
for field in ("flags", "options", "cursor", "state"):
key = field + "_json"
if key in result:
result[field] = json.loads(result.pop(key))
result.pop("lease_owner", None)
result.pop("lease_until", None)
result["pending_items"] = len(result.get("state", {}).get("_pending", []))
result.pop("state", None)
result.pop("cursor", None)
if "options" in result:
total = result["options"].get("total_messages")
result["total_messages"] = total
result["completion_reason"] = result["options"].get("completion_reason", "")
result["progress_percent"] = (100 if result["status"] == "completed" else
min(99, int(result["processed"] * 100 / total)) if total else None)
return result
def settings(self, tenant):
tenant = safe_scope(tenant)
with self.database.connect() as db:
row = db.execute("SELECT * FROM knowledge_settings WHERE tenant_id=?", (tenant,)).fetchone()
return dict(row) if row else {"tenant_id": tenant, "enabled": False, "updated_at": ""}
def set_enabled(self, tenant, enabled, actor, ip):
tenant = safe_scope(tenant)
with self.database.connect() as db:
if enabled:
from knowledge_retriever import KnowledgeRetriever
available = db.execute("SELECT id FROM knowledge_item WHERE tenant_id=? AND status='published' "
"AND (valid_until='' OR valid_until>=?)", (tenant, KnowledgeRetriever._today()))
if not any(self._fresh(db, row["id"], tenant) for row in available):
raise ValueError("请先审核并发布至少一条有效知识,再开启客服使用知识")
db.execute("INSERT INTO knowledge_settings VALUES (?,?,?) ON CONFLICT(tenant_id) "
"DO UPDATE SET enabled=excluded.enabled,updated_at=excluded.updated_at",
(tenant, int(enabled), utc_now()))
self.database._audit(db, actor, "knowledge.settings", f"tenant={tenant} enabled={enabled}", ip)
return self.settings(tenant)
def watches(self, scope):
scope = tenant_scope(scope)
with self.database.connect() as db:
return [dict(r) for r in db.execute(f"SELECT * FROM knowledge_watch WHERE {scope.clause()}", scope.params)]
def set_watch(self, tenant, source_id, enabled, actor, ip):
with self.database.connect() as db:
row = db.execute("SELECT 1 FROM knowledge_watch WHERE tenant_id=? AND source_account_id=?",
(tenant, source_id)).fetchone()
if row is None:
raise KeyError("请先创建加工任务并勾选自动整理")
db.execute("UPDATE knowledge_watch SET enabled=?,created_by=? WHERE tenant_id=? AND source_account_id=?",
(int(enabled), actor, tenant, source_id))
self.database._audit(db, actor, "knowledge.watch", f"source={source_id} enabled={enabled}", ip)
return {"enabled": enabled}
def stats(self, scope):
scope = tenant_scope(scope)
with self.database.connect() as db:
from knowledge_retriever import KnowledgeRetriever
rows = db.execute(f"SELECT CASE WHEN status='published' AND valid_until!='' AND valid_until<? "
f"THEN 'expired' ELSE status END effective_status,COUNT(*) n FROM knowledge_item WHERE {scope.clause()} "
"GROUP BY effective_status", [KnowledgeRetriever._today(), *scope.params]).fetchall()
return {row["effective_status"]: row["n"] for row in rows}
def list_items(self, scope, status="", query="", page=1, page_size=30):
scope = tenant_scope(scope)
where, args = scope.clause(), scope.params
if status:
where += " AND status=?"
args.append(status)
if query:
where += " AND (title LIKE ? OR question LIKE ?)"
args.extend([f"%{query[:100]}%"] * 2)
with self.database.connect() as db:
total = db.execute(f"SELECT COUNT(*) FROM knowledge_item WHERE {where}", args).fetchone()[0]
rows = db.execute(f"SELECT * FROM knowledge_item WHERE {where} ORDER BY updated_at DESC,id "
"LIMIT ? OFFSET ?", [*args, page_size, (page - 1) * page_size]).fetchall()
return {"items": [self.public(r) for r in rows], "total": total}
@staticmethod
def _item(db, tenant, item_id):
row = db.execute("SELECT * FROM knowledge_item WHERE id=? AND tenant_id=?", (item_id, tenant)).fetchone()
if row is None:
raise KeyError("知识不存在或不在当前账号范围内")
return row
def detail(self, tenant, item_id, *, sources=True):
with self.database.connect() as db:
item = self.public(self._item(db, tenant, item_id))
item["sources"] = [dict(r) for r in db.execute(
"SELECT * FROM knowledge_source WHERE item_id=? ORDER BY sent_at,message_id", (item_id,))] if sources else []
item["history"] = [dict(r) for r in db.execute(
"SELECT revision,action,actor_id,created_at FROM knowledge_revision WHERE item_id=? "
"ORDER BY created_at DESC,id LIMIT 30", (item_id,))]
return item
def _record(self, db, item_id, actor, action, ip=""):
row = db.execute("SELECT * FROM knowledge_item WHERE id=?", (item_id,)).fetchone()
db.execute("INSERT INTO knowledge_revision VALUES (?,?,?,?,?,?,?)",
(new_id(), item_id, row["revision"], action, json_text(dict(row)), actor, utc_now()))
if actor:
self.database._audit(db, actor, "knowledge." + action,
f"item={item_id} revision={row['revision']}", ip)
@staticmethod
def _fresh(db, item_id, tenant):
return db.execute("""SELECT 1 FROM knowledge_source s LEFT JOIN archive_message m ON m.id=s.message_id
WHERE s.item_id=? AND (m.id IS NULL OR m.tenant_id!=? OR m.status!='normal' OR
s.version_no!=(SELECT MAX(version_no) FROM archive_message_version WHERE message_id=s.message_id))
LIMIT 1""", (item_id, tenant)).fetchone() is None
def save(self, tenant, item_id, values, actor, ip):
fields = {k: str(values.get(k) or "").strip() for k in
("title", "question", "answer", "conditions", "category", "kind", "valid_until")}
if not all(fields[k] for k in ("title", "question", "answer", "conditions")):
raise ValueError("标题、问题、答案和适用条件均不能为空")
if fields["kind"] not in {"qa", "procedure", "case"}:
raise ValueError("未知知识类型")
if fields["valid_until"]:
date.fromisoformat(fields["valid_until"])
# Defense in depth: the editor must not reintroduce obvious identifiers.
for key in ("title", "question", "answer", "conditions", "category"):
fields[key] = redact(fields[key])
with self.database.connect() as db:
db.execute("BEGIN IMMEDIATE")
row = self._item(db, tenant, item_id)
if row["revision"] != values["revision"]:
raise ValueError("知识已被其他人修改,请刷新后重试")
if not self._fresh(db, item_id, tenant):
raise ValueError("来源消息已变化,请重新加工原会话;旧草稿不能重新发布")
db.execute("UPDATE knowledge_item SET " + ",".join(k + "=?" for k in fields) +
",status='draft',revision=revision+1,vector_status='pending',updated_at=? WHERE id=?",
[*fields.values(), utc_now(), item_id])
db.execute("DELETE FROM knowledge_fts WHERE item_id=?", (item_id,))
self._record(db, item_id, actor, "edit", ip)
return self.detail(tenant, item_id)
def transition(self, tenant, item_id, revision, action, actor, ip, confirmed=False):
target = {"approve": "approved", "reject": "rejected", "disable": "disabled"}.get(action)
if target is None:
raise ValueError("未知操作")
with self.database.connect() as db:
db.execute("BEGIN IMMEDIATE")
row = self._item(db, tenant, item_id)
if row["revision"] != revision:
raise ValueError("知识版本已变化,请刷新")
if action == "disable" and row["status"] != "published":
raise ValueError("仅已发布知识可停用")
if action == "approve":
self._check_expiry(row)
if row["status"] != "draft" or not confirmed or not row["conditions"].strip():
raise ValueError("请先补充适用条件,并确认身份、事实、脱敏和通用适用性已审核")
if not self._fresh(db, item_id, tenant):
raise ValueError("来源已变化,请重新加工")
if action == "reject" and row["status"] not in {"draft", "approved"}:
raise ValueError("仅草稿或待发布知识可驳回")
db.execute("UPDATE knowledge_item SET status=?,updated_at=? WHERE id=?", (target, utc_now(), item_id))
db.execute("DELETE FROM knowledge_fts WHERE item_id=?", (item_id,))
self._record(db, item_id, actor, action, ip)
return self.detail(tenant, item_id)
def delete_rejected(self, tenant, item_id, revision, actor, ip):
with self.database.connect() as db:
db.execute("BEGIN IMMEDIATE")
row = self._item(db, tenant, item_id)
if row['revision'] != revision or row['status'] != 'rejected':
raise ValueError('仅当前版本的已驳回知识可以删除,请刷新确认状态')
db.execute('INSERT OR IGNORE INTO knowledge_discarded VALUES (?,?,?)',
(tenant, row['dedup_hash'], utc_now()))
db.execute('DELETE FROM knowledge_fts WHERE item_id=?', (item_id,))
db.execute('DELETE FROM knowledge_source WHERE item_id=?', (item_id,))
db.execute('DELETE FROM knowledge_revision WHERE item_id=?', (item_id,))
db.execute('DELETE FROM knowledge_item WHERE id=? AND tenant_id=?', (item_id, tenant))
self.database._audit(db, actor, 'knowledge.delete', f'item={item_id} revision={revision}', ip)
return {'id': item_id, 'status': 'deleted'}
@staticmethod
def _check_expiry(row):
from knowledge_retriever import KnowledgeRetriever
if row["valid_until"] and row["valid_until"] < KnowledgeRetriever._today():
raise ValueError("知识已过有效期,请更新内容和有效期后重新审核")
def publish(self, tenant, item_id, revision, actor, ip):
from knowledge_retriever import VectorIndex
with self.database.connect() as db:
row = dict(self._item(db, tenant, item_id))
if row["revision"] != revision or row["status"] != "approved":
raise ValueError("只能发布当前已审核版本")
self._check_expiry(row)
index = VectorIndex()
profile = index.publish(row) if index.configured else ""
with self.database.connect() as db:
db.execute("BEGIN IMMEDIATE")
current = self._item(db, tenant, item_id)
if current["revision"] != revision or current["status"] != "approved" or not self._fresh(db, item_id, tenant):
raise ValueError("发布期间知识或来源发生变化,请重新审核")
db.execute("DELETE FROM knowledge_fts WHERE item_id=?", (item_id,))
db.execute("INSERT INTO knowledge_fts(item_id,terms) VALUES (?,?)",
(item_id, " ".join(tokens(row["title"] + " " + row["question"] + " " + row["answer"]))))
db.execute("UPDATE knowledge_item SET status='published',vector_status=?,vector_profile=?,updated_at=? WHERE id=?",
("ready" if profile else "not_configured", profile, utc_now(), item_id))
self._record(db, item_id, actor, "publish", ip)
return self.detail(tenant, item_id)
def _scope_counts(self, db, tenant, options, cutoff, *, quality=True):
source = db.execute("SELECT id,display_name,external_account_id FROM archive_source_account "
"WHERE id=? AND tenant_id=?", (options["source_account_id"], tenant)).fetchone()
if source is None:
raise ValueError("请选择当前租户的客服来源账号")
where, args = job_scope(tenant, options, cutoff)
if not quality:
# Creating a job holds the writer lock. Count the snapshot without reading
# message bodies or resolving identities; quality analysis stays read-only.
total = db.execute("SELECT COUNT(*) FROM archive_message m "
"JOIN archive_conversation c ON c.id=m.conversation_id WHERE " + " AND ".join(where) +
" AND EXISTS(SELECT 1 FROM archive_message_version v WHERE v.message_id=m.id AND v.created_at<=?)",
[*args, cutoff]).fetchone()[0]
return {"available_messages": total, "total_messages": min(total, options["max_messages"]),
"source_name": source["display_name"] or source["external_account_id"]}
row = db.execute("""SELECT COUNT(*) available_messages,COUNT(DISTINCT m.conversation_id) conversations,
COALESCE(SUM(v.status='normal' AND lower(m.message_type) IN ('text','文本','文字','1')
AND trim(v.content)!='' AND m.sender_person_id IS NOT NULL AND m.direction IN ('inbound','outbound')
AND ((m.direction='outbound')=EXISTS(SELECT 1 FROM archive_person_identity pi
WHERE pi.person_id=m.sender_person_id AND pi.tenant_id=m.tenant_id
AND pi.external_id=a.external_account_id))),0) eligible_text_messages
FROM archive_message m JOIN archive_conversation c ON c.id=m.conversation_id
JOIN archive_source_account a ON a.id=m.source_account_id
JOIN archive_message_version v ON v.message_id=m.id AND v.version_no=(
SELECT MAX(v2.version_no) FROM archive_message_version v2
WHERE v2.message_id=m.id AND v2.created_at<=?) WHERE """ + " AND ".join(where),
[cutoff, *args]).fetchone()
result = dict(row)
result.update(total_messages=min(row["available_messages"], options["max_messages"]),
limited=row["available_messages"] > options["max_messages"], cutoff_at=cutoff,
source_name=source["display_name"] or source["external_account_id"])
return result
def preview_job(self, tenant, values):
tenant = safe_scope(tenant)
options = job_options(values)
with self.database.connect() as db:
return self._scope_counts(db, tenant, options, utc_now())
def create_job(self, tenant, options, actor, ip):
tenant = safe_scope(tenant)
options = job_options(options)
if not options.get("staff_confirmed"):
raise ValueError("请确认所选来源账号是客服账号,且这些聊天允许用于知识整理")
if options['engine'] == 'model':
from knowledge_models import resolve_model
resolve_model(self.database, options)
scope_key = fingerprint(json_text({k: options.get(k, "") for k in (
"source_account_id", "conversation_id", "date_start", "date_end", "engine", "max_messages", "max_model_calls", "model_provider_id", "model_timeout_seconds")}))
with self.database.connect() as db:
db.execute("BEGIN IMMEDIATE")
active = db.execute("SELECT options_json FROM knowledge_job WHERE tenant_id=? "
"AND status IN ('queued','running','paused')", (tenant,))
if any(json.loads(row[0]).get("scope_key") == scope_key for row in active):
raise ValueError("相同范围已有未完成任务,请在加工任务中继续或取消原任务")
job_id, now = new_id(), utc_now()
estimate = self._scope_counts(db, tenant, options, now, quality=False)
options.update(total_messages=estimate["total_messages"], available_messages=estimate["available_messages"],
source_name=estimate["source_name"], scope_key=scope_key)
db.execute("INSERT INTO knowledge_job(id,tenant_id,options_json,cutoff_at,created_by,created_at,updated_at) "
"VALUES (?,?,?,?,?,?,?)", (job_id, tenant, json_text(options), now, actor, now, now))
if options.get("auto_watch"):
db.execute("INSERT INTO knowledge_watch VALUES (?,?,1,?) ON CONFLICT(tenant_id,source_account_id) "
"DO UPDATE SET enabled=1,created_by=excluded.created_by", (tenant, options["source_account_id"], actor))
self.database._audit(db, actor, "knowledge.job.create", f"job={job_id} tenant={tenant}", ip)
return self.job(tenant, job_id)
def job(self, tenant, job_id):
with self.database.connect() as db:
row = db.execute("SELECT * FROM knowledge_job WHERE id=? AND tenant_id=?", (job_id, tenant)).fetchone()
if row is None:
raise KeyError("任务不存在")
return self.public(row)
def jobs(self, scope):
scope = tenant_scope(scope)
with self.database.connect() as db:
return [self.public(r) for r in db.execute(f"SELECT * FROM knowledge_job WHERE {scope.clause()} "
"ORDER BY created_at DESC LIMIT 100", scope.params)]
def configure_job(self, tenant, job_id, values, actor, ip):
options = dict(self.job(tenant, job_id)['options'])
options.update(values)
options = job_options(options)
if options['engine'] == 'model':
from knowledge_models import resolve_model
resolve_model(self.database, options)
with self.database.connect() as db:
db.execute('BEGIN IMMEDIATE')
row = db.execute('SELECT * FROM knowledge_job WHERE id=? AND tenant_id=?', (job_id, tenant)).fetchone()
if row is None:
raise KeyError('任务不存在')
if row['status'] not in {'paused', 'failed'}:
raise ValueError('请先暂停任务,再调整模型配置')
if options['engine'] == 'model' and options['max_model_calls'] <= row['model_calls']:
raise ValueError('模型调用总上限须大于已使用次数')
db.execute('UPDATE knowledge_job SET options_json=?,updated_at=? WHERE id=?',
(json_text(options), utc_now(), job_id))
self.database._audit(db, actor, 'knowledge.job.configure', f'job={job_id}', ip)
return self.job(tenant, job_id)
def job_action(self, tenant, job_id, action, actor, ip):
if action not in {"pause", "resume", "cancel"}:
raise ValueError("未知任务操作")
with self.database.connect() as db:
db.execute("BEGIN IMMEDIATE")
row = db.execute("SELECT * FROM knowledge_job WHERE id=? AND tenant_id=?", (job_id, tenant)).fetchone()
if row is None:
raise KeyError("任务不存在")
allowed = ({"queued", "running"} if action == "pause" else {"paused", "failed"}
if action == "resume" else {"queued", "running", "paused", "failed"})
if row["status"] not in allowed:
raise ValueError("当前任务状态不能执行该操作")
db.execute("UPDATE knowledge_job SET status=?,lease_owner='',lease_until=0,error='',updated_at=? WHERE id=?",
({"pause": "paused", "resume": "queued", "cancel": "cancelled"}[action], utc_now(), job_id))
if action == "cancel":
db.execute("UPDATE knowledge_job SET state_json='{}' WHERE id=?", (job_id,))
self.database._audit(db, actor, "knowledge.job." + action, f"job={job_id}", ip)
return self.job(tenant, job_id)
def add_draft(self, db, tenant, job_id, candidate):
# Keep original source/version identity in the key: changed sources get a fresh draft.
digest = fingerprint(candidate["question"] + "\n" + candidate["answer"])
if db.execute('SELECT 1 FROM knowledge_discarded WHERE tenant_id=? AND dedup_hash=?', (tenant, digest)).fetchone():
return False
existing = db.execute("SELECT id,status FROM knowledge_item WHERE tenant_id=? AND dedup_hash=?",
(tenant, digest)).fetchone()
if existing and existing["status"] != "stale":
return False
if existing:
digest = fingerprint(digest + json_text(candidate["sources"]))
item_id, now = new_id(), utc_now()
inserted = db.execute("""INSERT OR IGNORE INTO knowledge_item
(id,tenant_id,job_id,dedup_hash,title,question,answer,conditions,category,kind,flags_json,created_at,updated_at)
VALUES (?,?,?,?,?,?,?,?,?,?,?,?,?)""", (item_id, tenant, job_id, digest, candidate["title"],
candidate["question"], candidate["answer"], candidate["conditions"], candidate["category"],
candidate["kind"], json_text(candidate["flags"]), now, now))
if not inserted.rowcount:
return False
db.executemany("INSERT INTO knowledge_source VALUES (?,?,?,?,?,?)", [
(item_id, s["message_id"], s["version_no"], s["role"], s["content"], s["sent_at"])
for s in candidate["sources"]])
if not self._fresh(db, item_id, tenant):
db.execute("UPDATE knowledge_item SET status='stale' WHERE id=?", (item_id,))
self._record(db, item_id, 0, "extract")
return True