Files
kefu/wechat_rpa/agent_rules.py
T
2026-09-21 10:34:06 +08:00

370 lines
16 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
# -*- coding: utf-8 -*-
"""智能体(角色)与角色规则。
为什么要有这一层:原来"客服是谁、该怎么说话"只有 `ai_config.AI_SYSTEM_PROMPT_TEMPLATE`
里那一大段写死的提示词。想让客户经理和医生用不同口径,或者临时补一条"不许这么说",
都得改代码、重新打包、挨个客户端升级——而线上翻车的往往就是一句话的措辞。
这里把"角色"做成数据:后台配,随桌面端配置下发,客户端读。一个智能体 =
人设(persona)+ 一组规则(rules)。规则只有三种,多一种都是给运营添负担:
guide 给模型的补充指令。留空关键词 = 每轮都注入;填了关键词 = 客户这句
话命中才注入(提示词不会无限膨胀)。
forbid 回复里**不许**出现的词。命中就换成这条规则的兜底话术——模型已经
把话说错了,光靠提示词是拦不住的,得在出口再挡一道。
reply 客户这句话命中关键词时,直接用固定话术回答,不问模型。口径必须
一字不差的场景(面诊链接何时发、退款流程)用这个,比反复调教
提示词稳得多。
切换与协作由 `plan` 决定:`single` 只上一个角色;`collaborate` 同时在场,
按关键词认领本轮,其余角色的 forbid/guide 规则仍然全体生效——协作的价值在于
"各管一段但共守底线",如果不生效那就只是换了个提示词而已。
这个模块**不导入 ai_chat**(ai_chat 导入它),也不自己去抠聊天文本:客户这轮
说了什么由调用方提取好传进来。保持纯函数,测试里不用起任何环境。
"""
from __future__ import annotations
import re
from typing import Any
import ai_config
RULE_TYPES = ("guide", "forbid", "reply")
PLAN_MODES = ("single", "collaborate")
MAX_AGENTS = 20
MAX_RULES_PER_AGENT = 30
MAX_KEYWORDS_PER_RULE = 30
# ── 内置兜底:模型把客户消息说成"乱码/系统编码" ──────────────────────────────
# 这是线上真实翻过的车:客户发了个「13」(空腹血糖值),模型回"这串像系统编码,
# 解析后可能是13";客户补了一句文字,模型又回"这串内容像系统编码,我这边无法
# 确认具体意思"。对客户来说这是彻底的答非所问,而且连着两条,观感极差。
#
# 提示词里已经写死了不许这么说(见 ai_config),但提示词是"建议",不是"保证"。
# 这一条兜底关不掉,也不需要后台配——任何角色、任何配置下,把客户看得见的文字
# 说成乱码都是错的。
_ENCODING_EXCUSE_RE = re.compile(
r"(?:系统编码|乱码|编码错乱|一串编码|像(?:是)?编码|"
r"无法确认(?:具体)?意思|无法解析|解析(?:后|不了)|"
r"这串(?:内容|字符|数字|东西))"
)
# 纯数字(血糖值、时间点、年龄、数量)。客户发这个是在回答问题,不是发乱码。
_PURE_NUMBER_RE = re.compile(r"^\d{1,4}(?:[..]\d{1,2})?$")
def _text(value: Any) -> str:
return str(value or "").strip()
def _keywords(raw: Any) -> list[str]:
if not isinstance(raw, (list, tuple, set)):
return []
words = []
for item in raw:
word = _text(item)
if word and word not in words:
words.append(word)
return words[:MAX_KEYWORDS_PER_RULE]
def _hit(text: str, keywords: list[str]) -> str:
"""返回命中的第一个关键词,没命中返回空串。子串匹配、不分大小写。"""
lowered = str(text or "").lower()
if not lowered:
return ""
for word in keywords:
if word.lower() in lowered:
return word
return ""
def normalize_rules(raw: Any) -> list[dict[str, Any]]:
"""把后台下发的规则清洗成可安全消费的形状。坏数据丢掉,不抛异常。
下发的配置在后台已经校验过一遍(`admin_backend.validate_agents`)。这里再洗
一次是因为客户端读到的可能是上一个版本留在磁盘上的 `ai_settings.json`——
为了一条脏规则让整个自动回复崩掉,代价完全不成比例。
"""
if not isinstance(raw, (list, tuple)):
return []
rules = []
for item in raw[:MAX_RULES_PER_AGENT]:
if not isinstance(item, dict):
continue
kind = _text(item.get("type")).lower() or "guide"
if kind not in RULE_TYPES:
continue
rule = {
"id": _text(item.get("id")),
"label": _text(item.get("label")),
"type": kind,
"keywords": _keywords(item.get("keywords")),
"instruction": _text(item.get("instruction")),
"reply": _text(item.get("reply")),
"enabled": bool(item.get("enabled", True)),
}
# 关键词是 forbid / reply 的全部触发条件,没有就是一条永远不会生效的
# 规则;guide 不填关键词有明确含义(每轮都注入),不能一起丢掉。
if kind in ("forbid", "reply") and not rule["keywords"]:
continue
if kind == "guide" and not rule["instruction"]:
continue
if kind == "reply" and not rule["reply"]:
continue
rules.append(rule)
return rules
def normalize_agents(raw: Any = None) -> list[dict[str, Any]]:
"""读取并清洗智能体清单。传 None 时读 `ai_config.AI_AGENTS`。"""
if raw is None:
raw = getattr(ai_config, "AI_AGENTS", None) or []
if not isinstance(raw, (list, tuple)):
return []
agents = []
seen = set()
for item in raw[:MAX_AGENTS]:
if not isinstance(item, dict):
continue
agent_id = _text(item.get("id"))
name = _text(item.get("name"))
if not agent_id or not name or agent_id in seen:
continue
seen.add(agent_id)
try:
priority = int(item.get("priority", 100))
except (TypeError, ValueError):
priority = 100
agents.append({
"id": agent_id,
"name": name,
"role": _text(item.get("role")) or name,
"description": _text(item.get("description")),
"persona": _text(item.get("persona")),
"keywords": _keywords(item.get("keywords")),
"rules": normalize_rules(item.get("rules")),
"enabled": bool(item.get("enabled", True)),
"priority": priority,
})
return agents
def normalize_plan(raw: Any = None) -> dict[str, Any]:
"""读取并清洗启用方案。传 None 时读 `ai_config.AI_AGENT_PLAN`。"""
if raw is None:
raw = getattr(ai_config, "AI_AGENT_PLAN", None) or {}
if not isinstance(raw, dict):
raw = {}
mode = _text(raw.get("mode")).lower() or "single"
if mode not in PLAN_MODES:
mode = "single"
active = raw.get("active_ids")
if isinstance(active, str):
active = [part for part in active.split(",")]
active_ids = []
for item in active if isinstance(active, (list, tuple)) else []:
value = _text(item)
if value and value not in active_ids:
active_ids.append(value)
try:
version = int(raw.get("version") or 0)
except (TypeError, ValueError):
version = 0
return {
"version": version,
"mode": mode,
"primary_id": _text(raw.get("primary_id")),
"active_ids": active_ids,
}
def active_agents(agents: Any = None, plan: Any = None) -> list[dict[str, Any]]:
"""本次回复到底由哪些角色上场,按主答在前排好序。
没配任何智能体时返回空列表——这时全部行为和改造前一模一样,一个字的提示词
都不会变。这是这套东西能安全上线的前提。
"""
items = normalize_agents(agents)
plan = normalize_plan(plan)
enabled = [item for item in items if item["enabled"]]
if not enabled:
return []
by_id = {item["id"]: item for item in enabled}
if plan["mode"] == "single":
chosen = by_id.get(plan["primary_id"])
if chosen is None:
# 后台切走了一个已经被删掉/停用的角色。宁可用优先级最高的那个顶上,
# 也不能一个角色都不上——那等于配置一保存,人设整个消失。
chosen = sorted(enabled, key=lambda item: (item["priority"], item["name"]))[0]
return [chosen]
ordered = [by_id[item] for item in plan["active_ids"] if item in by_id]
if not ordered:
ordered = sorted(enabled, key=lambda item: (item["priority"], item["name"]))
primary = by_id.get(plan["primary_id"])
if primary is not None and primary in ordered:
ordered = [primary] + [item for item in ordered if item is not primary]
return ordered
def claim(customer_text: str, agents: list[dict[str, Any]]) -> dict[str, Any] | None:
"""本轮由谁主答:客户这句话命中谁的关键词就归谁,都没命中归第一个。"""
if not agents:
return None
for agent in agents:
if agent["keywords"] and _hit(customer_text, agent["keywords"]):
return agent
return agents[0]
def _rules_text(agent: dict[str, Any], customer_text: str) -> str:
"""一个角色在本轮真正要注入的规则文本。"""
lines = []
for rule in agent["rules"]:
if not rule["enabled"]:
continue
if rule["type"] == "guide":
# 填了关键词的 guide 只在命中时注入。全量注入会把提示词撑到几千字,
# 而模型对超长提示词的服从度是明显下降的——规则越多反而越不听话。
if rule["keywords"] and not _hit(customer_text, rule["keywords"]):
continue
label = f"({rule['label']})" if rule["label"] else ""
lines.append(f"- {label}{rule['instruction']}")
elif rule["type"] == "forbid":
words = "、".join(rule["keywords"])
lines.append(f"- 【禁止措辞】回复里绝对不能出现:{words}。")
elif rule["type"] == "reply" and _hit(customer_text, rule["keywords"]):
lines.append(f"- 【标准口径】这一轮必须按这个意思回答:{rule['reply']}")
return "\n".join(lines)
def prompt_sections(
customer_text: str,
agents: Any = None,
plan: Any = None,
) -> str:
"""追加在基础人设后面的角色段落。没配智能体时返回空串。"""
active = active_agents(agents, plan)
if not active:
return ""
owner = claim(customer_text, active)
blocks = []
if len(active) == 1:
agent = active[0]
header = f"\n【当前智能体|{agent['name']}({agent['role']})】"
body = [header]
if agent["description"]:
body.append(f"职责:{agent['description']}")
if agent["persona"]:
body.append(agent["persona"])
rules = _rules_text(agent, customer_text)
if rules:
body.append("这个角色的规则(优先级高于上面的通用表达习惯):\n" + rules)
return "\n".join(body) + "\n"
blocks.append(
"\n【多智能体协作|本轮只发一条消息】\n"
"下面几个角色同时在岗,但**客户看到的永远是同一个人**:不许自报角色名,"
"不许说「我帮您转给谁」,也不许把几个角色的话拼成一条。\n"
f"本轮主答:{owner['name']}({owner['role']})。由它决定这条回复说什么。\n"
"其余角色只做两件事:把自己的规则当成底线校验一遍;本轮确实涉及它负责的"
"事情时,最多补一句要点,补不进去就不补。"
)
for agent in active:
lines = [f"\n—— {agent['name']}({agent['role']})" + ("|本轮主答" if agent is owner else "")]
if agent["description"]:
lines.append(f"职责:{agent['description']}")
if agent["keywords"]:
lines.append("负责话题:" + "、".join(agent["keywords"]))
if agent["persona"]:
lines.append(agent["persona"])
rules = _rules_text(agent, customer_text)
if rules:
lines.append("规则:\n" + rules)
blocks.append("\n".join(lines))
return "\n".join(blocks) + "\n"
def canned_reply(customer_text: str, agents: Any = None, plan: Any = None) -> str:
"""客户这句话命中 `reply` 规则时的固定话术,没命中返回空串。
在场角色全体参与匹配,主答角色先匹配——口径类规则(链接什么时候发、退款怎么
走)由谁负责都一样,漏掉一条的代价比多匹配一次大得多。
"""
for agent in active_agents(agents, plan):
for rule in agent["rules"]:
if not rule["enabled"] or rule["type"] != "reply":
continue
if _hit(customer_text, rule["keywords"]):
return rule["reply"]
return ""
def builtin_repair(reply: str, customer_text: str) -> str:
"""把"你发的是乱码/系统编码"这类回复换成一句正常人话。
只在客户这轮**确实有可读文字**时才动手:真正收到未转写语音、看不清的图片时,
说自己没识别出来是对的,那条路不能被这里误伤。
"""
text = _text(reply)
visible = _text(customer_text)
if not text or not visible:
return text
if not _ENCODING_EXCUSE_RE.search(text):
return text
if _PURE_NUMBER_RE.fullmatch(visible):
return f"您发的{visible}我看到了,是说这个数值吧?您补一句是什么时候测的,我好判断。"
short = visible if len(visible) <= 20 else visible[:20] + "…"
return f"您说的是「{short}」是吧?我这边收到了,您再补一句具体想问什么就行。"
def enforce_reply(
reply: str,
customer_text: str,
agents: Any = None,
plan: Any = None,
) -> str:
"""出口校验:命中 forbid 规则就换成兜底话术,最后再过一遍内置兜底。
这一步必须在 `_humanize` 之后、真正发出去之前。放在模型那一侧靠提示词约束
是拦不住的——线上翻车的那两条回复,提示词里明明白白写着不许那么说。
"""
text = _text(reply)
if not text:
return text
for agent in active_agents(agents, plan):
for rule in agent["rules"]:
if not rule["enabled"] or rule["type"] != "forbid":
continue
word = _hit(text, rule["keywords"])
if not word:
continue
label = rule["label"] or word
print(f" [智能体] 回复命中「{agent['name']}·{label}」禁止措辞:{word}")
if rule["reply"]:
return rule["reply"]
# 没配兜底话术时交给内置兜底;它至少能保证不把客户的话说成乱码。
text = builtin_repair(text, customer_text)
if _hit(text, rule["keywords"]):
# 内置兜底没能改掉这个词。返回空串让调用方走"不自动回复"的路,
# 比把一条明确禁止的话发出去强。
return ""
return text
return builtin_repair(text, customer_text)
def describe(agents: Any = None, plan: Any = None) -> str:
"""一行状态描述,打日志用。"""
active = active_agents(agents, plan)
if not active:
return "未配置智能体(沿用内置人设)"
mode = normalize_plan(plan)["mode"]
names = "、".join(item["name"] for item in active)
return f"{'协作' if mode == 'collaborate' else '单角色'}:{names}"