370 lines
16 KiB
Python
370 lines
16 KiB
Python
# -*- coding: utf-8 -*-
|
||
"""智能体(角色)与角色规则。
|
||
|
||
为什么要有这一层:原来"客服是谁、该怎么说话"只有 `ai_config.AI_SYSTEM_PROMPT_TEMPLATE`
|
||
里那一大段写死的提示词。想让客户经理和医生用不同口径,或者临时补一条"不许这么说",
|
||
都得改代码、重新打包、挨个客户端升级——而线上翻车的往往就是一句话的措辞。
|
||
|
||
这里把"角色"做成数据:后台配,随桌面端配置下发,客户端读。一个智能体 =
|
||
人设(persona)+ 一组规则(rules)。规则只有三种,多一种都是给运营添负担:
|
||
|
||
guide 给模型的补充指令。留空关键词 = 每轮都注入;填了关键词 = 客户这句
|
||
话命中才注入(提示词不会无限膨胀)。
|
||
forbid 回复里**不许**出现的词。命中就换成这条规则的兜底话术——模型已经
|
||
把话说错了,光靠提示词是拦不住的,得在出口再挡一道。
|
||
reply 客户这句话命中关键词时,直接用固定话术回答,不问模型。口径必须
|
||
一字不差的场景(面诊链接何时发、退款流程)用这个,比反复调教
|
||
提示词稳得多。
|
||
|
||
切换与协作由 `plan` 决定:`single` 只上一个角色;`collaborate` 同时在场,
|
||
按关键词认领本轮,其余角色的 forbid/guide 规则仍然全体生效——协作的价值在于
|
||
"各管一段但共守底线",如果不生效那就只是换了个提示词而已。
|
||
|
||
这个模块**不导入 ai_chat**(ai_chat 导入它),也不自己去抠聊天文本:客户这轮
|
||
说了什么由调用方提取好传进来。保持纯函数,测试里不用起任何环境。
|
||
"""
|
||
|
||
from __future__ import annotations
|
||
|
||
import re
|
||
from typing import Any
|
||
|
||
import ai_config
|
||
|
||
RULE_TYPES = ("guide", "forbid", "reply")
|
||
PLAN_MODES = ("single", "collaborate")
|
||
|
||
MAX_AGENTS = 20
|
||
MAX_RULES_PER_AGENT = 30
|
||
MAX_KEYWORDS_PER_RULE = 30
|
||
|
||
|
||
# ── 内置兜底:模型把客户消息说成"乱码/系统编码" ──────────────────────────────
|
||
# 这是线上真实翻过的车:客户发了个「13」(空腹血糖值),模型回"这串像系统编码,
|
||
# 解析后可能是13";客户补了一句文字,模型又回"这串内容像系统编码,我这边无法
|
||
# 确认具体意思"。对客户来说这是彻底的答非所问,而且连着两条,观感极差。
|
||
#
|
||
# 提示词里已经写死了不许这么说(见 ai_config),但提示词是"建议",不是"保证"。
|
||
# 这一条兜底关不掉,也不需要后台配——任何角色、任何配置下,把客户看得见的文字
|
||
# 说成乱码都是错的。
|
||
_ENCODING_EXCUSE_RE = re.compile(
|
||
r"(?:系统编码|乱码|编码错乱|一串编码|像(?:是)?编码|"
|
||
r"无法确认(?:具体)?意思|无法解析|解析(?:后|不了)|"
|
||
r"这串(?:内容|字符|数字|东西))"
|
||
)
|
||
# 纯数字(血糖值、时间点、年龄、数量)。客户发这个是在回答问题,不是发乱码。
|
||
_PURE_NUMBER_RE = re.compile(r"^\d{1,4}(?:[..]\d{1,2})?$")
|
||
|
||
|
||
def _text(value: Any) -> str:
|
||
return str(value or "").strip()
|
||
|
||
|
||
def _keywords(raw: Any) -> list[str]:
|
||
if not isinstance(raw, (list, tuple, set)):
|
||
return []
|
||
words = []
|
||
for item in raw:
|
||
word = _text(item)
|
||
if word and word not in words:
|
||
words.append(word)
|
||
return words[:MAX_KEYWORDS_PER_RULE]
|
||
|
||
|
||
def _hit(text: str, keywords: list[str]) -> str:
|
||
"""返回命中的第一个关键词,没命中返回空串。子串匹配、不分大小写。"""
|
||
lowered = str(text or "").lower()
|
||
if not lowered:
|
||
return ""
|
||
for word in keywords:
|
||
if word.lower() in lowered:
|
||
return word
|
||
return ""
|
||
|
||
|
||
def normalize_rules(raw: Any) -> list[dict[str, Any]]:
|
||
"""把后台下发的规则清洗成可安全消费的形状。坏数据丢掉,不抛异常。
|
||
|
||
下发的配置在后台已经校验过一遍(`admin_backend.validate_agents`)。这里再洗
|
||
一次是因为客户端读到的可能是上一个版本留在磁盘上的 `ai_settings.json`——
|
||
为了一条脏规则让整个自动回复崩掉,代价完全不成比例。
|
||
"""
|
||
if not isinstance(raw, (list, tuple)):
|
||
return []
|
||
rules = []
|
||
for item in raw[:MAX_RULES_PER_AGENT]:
|
||
if not isinstance(item, dict):
|
||
continue
|
||
kind = _text(item.get("type")).lower() or "guide"
|
||
if kind not in RULE_TYPES:
|
||
continue
|
||
rule = {
|
||
"id": _text(item.get("id")),
|
||
"label": _text(item.get("label")),
|
||
"type": kind,
|
||
"keywords": _keywords(item.get("keywords")),
|
||
"instruction": _text(item.get("instruction")),
|
||
"reply": _text(item.get("reply")),
|
||
"enabled": bool(item.get("enabled", True)),
|
||
}
|
||
# 关键词是 forbid / reply 的全部触发条件,没有就是一条永远不会生效的
|
||
# 规则;guide 不填关键词有明确含义(每轮都注入),不能一起丢掉。
|
||
if kind in ("forbid", "reply") and not rule["keywords"]:
|
||
continue
|
||
if kind == "guide" and not rule["instruction"]:
|
||
continue
|
||
if kind == "reply" and not rule["reply"]:
|
||
continue
|
||
rules.append(rule)
|
||
return rules
|
||
|
||
|
||
def normalize_agents(raw: Any = None) -> list[dict[str, Any]]:
|
||
"""读取并清洗智能体清单。传 None 时读 `ai_config.AI_AGENTS`。"""
|
||
if raw is None:
|
||
raw = getattr(ai_config, "AI_AGENTS", None) or []
|
||
if not isinstance(raw, (list, tuple)):
|
||
return []
|
||
agents = []
|
||
seen = set()
|
||
for item in raw[:MAX_AGENTS]:
|
||
if not isinstance(item, dict):
|
||
continue
|
||
agent_id = _text(item.get("id"))
|
||
name = _text(item.get("name"))
|
||
if not agent_id or not name or agent_id in seen:
|
||
continue
|
||
seen.add(agent_id)
|
||
try:
|
||
priority = int(item.get("priority", 100))
|
||
except (TypeError, ValueError):
|
||
priority = 100
|
||
agents.append({
|
||
"id": agent_id,
|
||
"name": name,
|
||
"role": _text(item.get("role")) or name,
|
||
"description": _text(item.get("description")),
|
||
"persona": _text(item.get("persona")),
|
||
"keywords": _keywords(item.get("keywords")),
|
||
"rules": normalize_rules(item.get("rules")),
|
||
"enabled": bool(item.get("enabled", True)),
|
||
"priority": priority,
|
||
})
|
||
return agents
|
||
|
||
|
||
def normalize_plan(raw: Any = None) -> dict[str, Any]:
|
||
"""读取并清洗启用方案。传 None 时读 `ai_config.AI_AGENT_PLAN`。"""
|
||
if raw is None:
|
||
raw = getattr(ai_config, "AI_AGENT_PLAN", None) or {}
|
||
if not isinstance(raw, dict):
|
||
raw = {}
|
||
mode = _text(raw.get("mode")).lower() or "single"
|
||
if mode not in PLAN_MODES:
|
||
mode = "single"
|
||
active = raw.get("active_ids")
|
||
if isinstance(active, str):
|
||
active = [part for part in active.split(",")]
|
||
active_ids = []
|
||
for item in active if isinstance(active, (list, tuple)) else []:
|
||
value = _text(item)
|
||
if value and value not in active_ids:
|
||
active_ids.append(value)
|
||
try:
|
||
version = int(raw.get("version") or 0)
|
||
except (TypeError, ValueError):
|
||
version = 0
|
||
return {
|
||
"version": version,
|
||
"mode": mode,
|
||
"primary_id": _text(raw.get("primary_id")),
|
||
"active_ids": active_ids,
|
||
}
|
||
|
||
|
||
def active_agents(agents: Any = None, plan: Any = None) -> list[dict[str, Any]]:
|
||
"""本次回复到底由哪些角色上场,按主答在前排好序。
|
||
|
||
没配任何智能体时返回空列表——这时全部行为和改造前一模一样,一个字的提示词
|
||
都不会变。这是这套东西能安全上线的前提。
|
||
"""
|
||
items = normalize_agents(agents)
|
||
plan = normalize_plan(plan)
|
||
enabled = [item for item in items if item["enabled"]]
|
||
if not enabled:
|
||
return []
|
||
by_id = {item["id"]: item for item in enabled}
|
||
|
||
if plan["mode"] == "single":
|
||
chosen = by_id.get(plan["primary_id"])
|
||
if chosen is None:
|
||
# 后台切走了一个已经被删掉/停用的角色。宁可用优先级最高的那个顶上,
|
||
# 也不能一个角色都不上——那等于配置一保存,人设整个消失。
|
||
chosen = sorted(enabled, key=lambda item: (item["priority"], item["name"]))[0]
|
||
return [chosen]
|
||
|
||
ordered = [by_id[item] for item in plan["active_ids"] if item in by_id]
|
||
if not ordered:
|
||
ordered = sorted(enabled, key=lambda item: (item["priority"], item["name"]))
|
||
primary = by_id.get(plan["primary_id"])
|
||
if primary is not None and primary in ordered:
|
||
ordered = [primary] + [item for item in ordered if item is not primary]
|
||
return ordered
|
||
|
||
|
||
def claim(customer_text: str, agents: list[dict[str, Any]]) -> dict[str, Any] | None:
|
||
"""本轮由谁主答:客户这句话命中谁的关键词就归谁,都没命中归第一个。"""
|
||
if not agents:
|
||
return None
|
||
for agent in agents:
|
||
if agent["keywords"] and _hit(customer_text, agent["keywords"]):
|
||
return agent
|
||
return agents[0]
|
||
|
||
|
||
def _rules_text(agent: dict[str, Any], customer_text: str) -> str:
|
||
"""一个角色在本轮真正要注入的规则文本。"""
|
||
lines = []
|
||
for rule in agent["rules"]:
|
||
if not rule["enabled"]:
|
||
continue
|
||
if rule["type"] == "guide":
|
||
# 填了关键词的 guide 只在命中时注入。全量注入会把提示词撑到几千字,
|
||
# 而模型对超长提示词的服从度是明显下降的——规则越多反而越不听话。
|
||
if rule["keywords"] and not _hit(customer_text, rule["keywords"]):
|
||
continue
|
||
label = f"({rule['label']})" if rule["label"] else ""
|
||
lines.append(f"- {label}{rule['instruction']}")
|
||
elif rule["type"] == "forbid":
|
||
words = "、".join(rule["keywords"])
|
||
lines.append(f"- 【禁止措辞】回复里绝对不能出现:{words}。")
|
||
elif rule["type"] == "reply" and _hit(customer_text, rule["keywords"]):
|
||
lines.append(f"- 【标准口径】这一轮必须按这个意思回答:{rule['reply']}")
|
||
return "\n".join(lines)
|
||
|
||
|
||
def prompt_sections(
|
||
customer_text: str,
|
||
agents: Any = None,
|
||
plan: Any = None,
|
||
) -> str:
|
||
"""追加在基础人设后面的角色段落。没配智能体时返回空串。"""
|
||
active = active_agents(agents, plan)
|
||
if not active:
|
||
return ""
|
||
owner = claim(customer_text, active)
|
||
blocks = []
|
||
|
||
if len(active) == 1:
|
||
agent = active[0]
|
||
header = f"\n【当前智能体|{agent['name']}({agent['role']})】"
|
||
body = [header]
|
||
if agent["description"]:
|
||
body.append(f"职责:{agent['description']}")
|
||
if agent["persona"]:
|
||
body.append(agent["persona"])
|
||
rules = _rules_text(agent, customer_text)
|
||
if rules:
|
||
body.append("这个角色的规则(优先级高于上面的通用表达习惯):\n" + rules)
|
||
return "\n".join(body) + "\n"
|
||
|
||
blocks.append(
|
||
"\n【多智能体协作|本轮只发一条消息】\n"
|
||
"下面几个角色同时在岗,但**客户看到的永远是同一个人**:不许自报角色名,"
|
||
"不许说「我帮您转给谁」,也不许把几个角色的话拼成一条。\n"
|
||
f"本轮主答:{owner['name']}({owner['role']})。由它决定这条回复说什么。\n"
|
||
"其余角色只做两件事:把自己的规则当成底线校验一遍;本轮确实涉及它负责的"
|
||
"事情时,最多补一句要点,补不进去就不补。"
|
||
)
|
||
for agent in active:
|
||
lines = [f"\n—— {agent['name']}({agent['role']})" + ("|本轮主答" if agent is owner else "")]
|
||
if agent["description"]:
|
||
lines.append(f"职责:{agent['description']}")
|
||
if agent["keywords"]:
|
||
lines.append("负责话题:" + "、".join(agent["keywords"]))
|
||
if agent["persona"]:
|
||
lines.append(agent["persona"])
|
||
rules = _rules_text(agent, customer_text)
|
||
if rules:
|
||
lines.append("规则:\n" + rules)
|
||
blocks.append("\n".join(lines))
|
||
return "\n".join(blocks) + "\n"
|
||
|
||
|
||
def canned_reply(customer_text: str, agents: Any = None, plan: Any = None) -> str:
|
||
"""客户这句话命中 `reply` 规则时的固定话术,没命中返回空串。
|
||
|
||
在场角色全体参与匹配,主答角色先匹配——口径类规则(链接什么时候发、退款怎么
|
||
走)由谁负责都一样,漏掉一条的代价比多匹配一次大得多。
|
||
"""
|
||
for agent in active_agents(agents, plan):
|
||
for rule in agent["rules"]:
|
||
if not rule["enabled"] or rule["type"] != "reply":
|
||
continue
|
||
if _hit(customer_text, rule["keywords"]):
|
||
return rule["reply"]
|
||
return ""
|
||
|
||
|
||
def builtin_repair(reply: str, customer_text: str) -> str:
|
||
"""把"你发的是乱码/系统编码"这类回复换成一句正常人话。
|
||
|
||
只在客户这轮**确实有可读文字**时才动手:真正收到未转写语音、看不清的图片时,
|
||
说自己没识别出来是对的,那条路不能被这里误伤。
|
||
"""
|
||
text = _text(reply)
|
||
visible = _text(customer_text)
|
||
if not text or not visible:
|
||
return text
|
||
if not _ENCODING_EXCUSE_RE.search(text):
|
||
return text
|
||
if _PURE_NUMBER_RE.fullmatch(visible):
|
||
return f"您发的{visible}我看到了,是说这个数值吧?您补一句是什么时候测的,我好判断。"
|
||
short = visible if len(visible) <= 20 else visible[:20] + "…"
|
||
return f"您说的是「{short}」是吧?我这边收到了,您再补一句具体想问什么就行。"
|
||
|
||
|
||
def enforce_reply(
|
||
reply: str,
|
||
customer_text: str,
|
||
agents: Any = None,
|
||
plan: Any = None,
|
||
) -> str:
|
||
"""出口校验:命中 forbid 规则就换成兜底话术,最后再过一遍内置兜底。
|
||
|
||
这一步必须在 `_humanize` 之后、真正发出去之前。放在模型那一侧靠提示词约束
|
||
是拦不住的——线上翻车的那两条回复,提示词里明明白白写着不许那么说。
|
||
"""
|
||
text = _text(reply)
|
||
if not text:
|
||
return text
|
||
for agent in active_agents(agents, plan):
|
||
for rule in agent["rules"]:
|
||
if not rule["enabled"] or rule["type"] != "forbid":
|
||
continue
|
||
word = _hit(text, rule["keywords"])
|
||
if not word:
|
||
continue
|
||
label = rule["label"] or word
|
||
print(f" [智能体] 回复命中「{agent['name']}·{label}」禁止措辞:{word}")
|
||
if rule["reply"]:
|
||
return rule["reply"]
|
||
# 没配兜底话术时交给内置兜底;它至少能保证不把客户的话说成乱码。
|
||
text = builtin_repair(text, customer_text)
|
||
if _hit(text, rule["keywords"]):
|
||
# 内置兜底没能改掉这个词。返回空串让调用方走"不自动回复"的路,
|
||
# 比把一条明确禁止的话发出去强。
|
||
return ""
|
||
return text
|
||
return builtin_repair(text, customer_text)
|
||
|
||
|
||
def describe(agents: Any = None, plan: Any = None) -> str:
|
||
"""一行状态描述,打日志用。"""
|
||
active = active_agents(agents, plan)
|
||
if not active:
|
||
return "未配置智能体(沿用内置人设)"
|
||
mode = normalize_plan(plan)["mode"]
|
||
names = "、".join(item["name"] for item in active)
|
||
return f"{'协作' if mode == 'collaborate' else '单角色'}:{names}"
|