# -*- coding: utf-8 -*- """智能体(角色)与角色规则。 为什么要有这一层:原来"客服是谁、该怎么说话"只有 `ai_config.AI_SYSTEM_PROMPT_TEMPLATE` 里那一大段写死的提示词。想让客户经理和医生用不同口径,或者临时补一条"不许这么说", 都得改代码、重新打包、挨个客户端升级——而线上翻车的往往就是一句话的措辞。 这里把"角色"做成数据:后台配,随桌面端配置下发,客户端读。一个智能体 = 人设(persona)+ 一组规则(rules)。规则只有三种,多一种都是给运营添负担: guide 给模型的补充指令。留空关键词 = 每轮都注入;填了关键词 = 客户这句 话命中才注入(提示词不会无限膨胀)。 forbid 回复里**不许**出现的词。命中就换成这条规则的兜底话术——模型已经 把话说错了,光靠提示词是拦不住的,得在出口再挡一道。 reply 客户这句话命中关键词时,直接用固定话术回答,不问模型。口径必须 一字不差的场景(面诊链接何时发、退款流程)用这个,比反复调教 提示词稳得多。 切换与协作由 `plan` 决定:`single` 只上一个角色;`collaborate` 同时在场, 按关键词认领本轮,其余角色的 forbid/guide 规则仍然全体生效——协作的价值在于 "各管一段但共守底线",如果不生效那就只是换了个提示词而已。 这个模块**不导入 ai_chat**(ai_chat 导入它),也不自己去抠聊天文本:客户这轮 说了什么由调用方提取好传进来。保持纯函数,测试里不用起任何环境。 """ from __future__ import annotations import re from typing import Any import ai_config RULE_TYPES = ("guide", "forbid", "reply") PLAN_MODES = ("single", "collaborate") MAX_AGENTS = 20 MAX_RULES_PER_AGENT = 30 MAX_KEYWORDS_PER_RULE = 30 # ── 内置兜底:模型把客户消息说成"乱码/系统编码" ────────────────────────────── # 这是线上真实翻过的车:客户发了个「13」(空腹血糖值),模型回"这串像系统编码, # 解析后可能是13";客户补了一句文字,模型又回"这串内容像系统编码,我这边无法 # 确认具体意思"。对客户来说这是彻底的答非所问,而且连着两条,观感极差。 # # 提示词里已经写死了不许这么说(见 ai_config),但提示词是"建议",不是"保证"。 # 这一条兜底关不掉,也不需要后台配——任何角色、任何配置下,把客户看得见的文字 # 说成乱码都是错的。 _ENCODING_EXCUSE_RE = re.compile( r"(?:系统编码|乱码|编码错乱|一串编码|像(?:是)?编码|" r"无法确认(?:具体)?意思|无法解析|解析(?:后|不了)|" r"这串(?:内容|字符|数字|东西))" ) # 纯数字(血糖值、时间点、年龄、数量)。客户发这个是在回答问题,不是发乱码。 _PURE_NUMBER_RE = re.compile(r"^\d{1,4}(?:[..]\d{1,2})?$") def _text(value: Any) -> str: return str(value or "").strip() def _keywords(raw: Any) -> list[str]: if not isinstance(raw, (list, tuple, set)): return [] words = [] for item in raw: word = _text(item) if word and word not in words: words.append(word) return words[:MAX_KEYWORDS_PER_RULE] def _hit(text: str, keywords: list[str]) -> str: """返回命中的第一个关键词,没命中返回空串。子串匹配、不分大小写。""" lowered = str(text or "").lower() if not lowered: return "" for word in keywords: if word.lower() in lowered: return word return "" def normalize_rules(raw: Any) -> list[dict[str, Any]]: """把后台下发的规则清洗成可安全消费的形状。坏数据丢掉,不抛异常。 下发的配置在后台已经校验过一遍(`admin_backend.validate_agents`)。这里再洗 一次是因为客户端读到的可能是上一个版本留在磁盘上的 `ai_settings.json`—— 为了一条脏规则让整个自动回复崩掉,代价完全不成比例。 """ if not isinstance(raw, (list, tuple)): return [] rules = [] for item in raw[:MAX_RULES_PER_AGENT]: if not isinstance(item, dict): continue kind = _text(item.get("type")).lower() or "guide" if kind not in RULE_TYPES: continue rule = { "id": _text(item.get("id")), "label": _text(item.get("label")), "type": kind, "keywords": _keywords(item.get("keywords")), "instruction": _text(item.get("instruction")), "reply": _text(item.get("reply")), "enabled": bool(item.get("enabled", True)), } # 关键词是 forbid / reply 的全部触发条件,没有就是一条永远不会生效的 # 规则;guide 不填关键词有明确含义(每轮都注入),不能一起丢掉。 if kind in ("forbid", "reply") and not rule["keywords"]: continue if kind == "guide" and not rule["instruction"]: continue if kind == "reply" and not rule["reply"]: continue rules.append(rule) return rules def normalize_agents(raw: Any = None) -> list[dict[str, Any]]: """读取并清洗智能体清单。传 None 时读 `ai_config.AI_AGENTS`。""" if raw is None: raw = getattr(ai_config, "AI_AGENTS", None) or [] if not isinstance(raw, (list, tuple)): return [] agents = [] seen = set() for item in raw[:MAX_AGENTS]: if not isinstance(item, dict): continue agent_id = _text(item.get("id")) name = _text(item.get("name")) if not agent_id or not name or agent_id in seen: continue seen.add(agent_id) try: priority = int(item.get("priority", 100)) except (TypeError, ValueError): priority = 100 agents.append({ "id": agent_id, "name": name, "role": _text(item.get("role")) or name, "description": _text(item.get("description")), "persona": _text(item.get("persona")), "keywords": _keywords(item.get("keywords")), "rules": normalize_rules(item.get("rules")), "enabled": bool(item.get("enabled", True)), "priority": priority, }) return agents def normalize_plan(raw: Any = None) -> dict[str, Any]: """读取并清洗启用方案。传 None 时读 `ai_config.AI_AGENT_PLAN`。""" if raw is None: raw = getattr(ai_config, "AI_AGENT_PLAN", None) or {} if not isinstance(raw, dict): raw = {} mode = _text(raw.get("mode")).lower() or "single" if mode not in PLAN_MODES: mode = "single" active = raw.get("active_ids") if isinstance(active, str): active = [part for part in active.split(",")] active_ids = [] for item in active if isinstance(active, (list, tuple)) else []: value = _text(item) if value and value not in active_ids: active_ids.append(value) try: version = int(raw.get("version") or 0) except (TypeError, ValueError): version = 0 return { "version": version, "mode": mode, "primary_id": _text(raw.get("primary_id")), "active_ids": active_ids, } def active_agents(agents: Any = None, plan: Any = None) -> list[dict[str, Any]]: """本次回复到底由哪些角色上场,按主答在前排好序。 没配任何智能体时返回空列表——这时全部行为和改造前一模一样,一个字的提示词 都不会变。这是这套东西能安全上线的前提。 """ items = normalize_agents(agents) plan = normalize_plan(plan) enabled = [item for item in items if item["enabled"]] if not enabled: return [] by_id = {item["id"]: item for item in enabled} if plan["mode"] == "single": chosen = by_id.get(plan["primary_id"]) if chosen is None: # 后台切走了一个已经被删掉/停用的角色。宁可用优先级最高的那个顶上, # 也不能一个角色都不上——那等于配置一保存,人设整个消失。 chosen = sorted(enabled, key=lambda item: (item["priority"], item["name"]))[0] return [chosen] ordered = [by_id[item] for item in plan["active_ids"] if item in by_id] if not ordered: ordered = sorted(enabled, key=lambda item: (item["priority"], item["name"])) primary = by_id.get(plan["primary_id"]) if primary is not None and primary in ordered: ordered = [primary] + [item for item in ordered if item is not primary] return ordered def claim(customer_text: str, agents: list[dict[str, Any]]) -> dict[str, Any] | None: """本轮由谁主答:客户这句话命中谁的关键词就归谁,都没命中归第一个。""" if not agents: return None for agent in agents: if agent["keywords"] and _hit(customer_text, agent["keywords"]): return agent return agents[0] def _rules_text(agent: dict[str, Any], customer_text: str) -> str: """一个角色在本轮真正要注入的规则文本。""" lines = [] for rule in agent["rules"]: if not rule["enabled"]: continue if rule["type"] == "guide": # 填了关键词的 guide 只在命中时注入。全量注入会把提示词撑到几千字, # 而模型对超长提示词的服从度是明显下降的——规则越多反而越不听话。 if rule["keywords"] and not _hit(customer_text, rule["keywords"]): continue label = f"({rule['label']})" if rule["label"] else "" lines.append(f"- {label}{rule['instruction']}") elif rule["type"] == "forbid": words = "、".join(rule["keywords"]) lines.append(f"- 【禁止措辞】回复里绝对不能出现:{words}。") elif rule["type"] == "reply" and _hit(customer_text, rule["keywords"]): lines.append(f"- 【标准口径】这一轮必须按这个意思回答:{rule['reply']}") return "\n".join(lines) def prompt_sections( customer_text: str, agents: Any = None, plan: Any = None, ) -> str: """追加在基础人设后面的角色段落。没配智能体时返回空串。""" active = active_agents(agents, plan) if not active: return "" owner = claim(customer_text, active) blocks = [] if len(active) == 1: agent = active[0] header = f"\n【当前智能体|{agent['name']}({agent['role']})】" body = [header] if agent["description"]: body.append(f"职责:{agent['description']}") if agent["persona"]: body.append(agent["persona"]) rules = _rules_text(agent, customer_text) if rules: body.append("这个角色的规则(优先级高于上面的通用表达习惯):\n" + rules) return "\n".join(body) + "\n" blocks.append( "\n【多智能体协作|本轮只发一条消息】\n" "下面几个角色同时在岗,但**客户看到的永远是同一个人**:不许自报角色名," "不许说「我帮您转给谁」,也不许把几个角色的话拼成一条。\n" f"本轮主答:{owner['name']}({owner['role']})。由它决定这条回复说什么。\n" "其余角色只做两件事:把自己的规则当成底线校验一遍;本轮确实涉及它负责的" "事情时,最多补一句要点,补不进去就不补。" ) for agent in active: lines = [f"\n—— {agent['name']}({agent['role']})" + ("|本轮主答" if agent is owner else "")] if agent["description"]: lines.append(f"职责:{agent['description']}") if agent["keywords"]: lines.append("负责话题:" + "、".join(agent["keywords"])) if agent["persona"]: lines.append(agent["persona"]) rules = _rules_text(agent, customer_text) if rules: lines.append("规则:\n" + rules) blocks.append("\n".join(lines)) return "\n".join(blocks) + "\n" def canned_reply(customer_text: str, agents: Any = None, plan: Any = None) -> str: """客户这句话命中 `reply` 规则时的固定话术,没命中返回空串。 在场角色全体参与匹配,主答角色先匹配——口径类规则(链接什么时候发、退款怎么 走)由谁负责都一样,漏掉一条的代价比多匹配一次大得多。 """ for agent in active_agents(agents, plan): for rule in agent["rules"]: if not rule["enabled"] or rule["type"] != "reply": continue if _hit(customer_text, rule["keywords"]): return rule["reply"] return "" def builtin_repair(reply: str, customer_text: str) -> str: """把"你发的是乱码/系统编码"这类回复换成一句正常人话。 只在客户这轮**确实有可读文字**时才动手:真正收到未转写语音、看不清的图片时, 说自己没识别出来是对的,那条路不能被这里误伤。 """ text = _text(reply) visible = _text(customer_text) if not text or not visible: return text if not _ENCODING_EXCUSE_RE.search(text): return text if _PURE_NUMBER_RE.fullmatch(visible): return f"您发的{visible}我看到了,是说这个数值吧?您补一句是什么时候测的,我好判断。" short = visible if len(visible) <= 20 else visible[:20] + "…" return f"您说的是「{short}」是吧?我这边收到了,您再补一句具体想问什么就行。" def enforce_reply( reply: str, customer_text: str, agents: Any = None, plan: Any = None, ) -> str: """出口校验:命中 forbid 规则就换成兜底话术,最后再过一遍内置兜底。 这一步必须在 `_humanize` 之后、真正发出去之前。放在模型那一侧靠提示词约束 是拦不住的——线上翻车的那两条回复,提示词里明明白白写着不许那么说。 """ text = _text(reply) if not text: return text for agent in active_agents(agents, plan): for rule in agent["rules"]: if not rule["enabled"] or rule["type"] != "forbid": continue word = _hit(text, rule["keywords"]) if not word: continue label = rule["label"] or word print(f" [智能体] 回复命中「{agent['name']}·{label}」禁止措辞:{word}") if rule["reply"]: return rule["reply"] # 没配兜底话术时交给内置兜底;它至少能保证不把客户的话说成乱码。 text = builtin_repair(text, customer_text) if _hit(text, rule["keywords"]): # 内置兜底没能改掉这个词。返回空串让调用方走"不自动回复"的路, # 比把一条明确禁止的话发出去强。 return "" return text return builtin_repair(text, customer_text) def describe(agents: Any = None, plan: Any = None) -> str: """一行状态描述,打日志用。""" active = active_agents(agents, plan) if not active: return "未配置智能体(沿用内置人设)" mode = normalize_plan(plan)["mode"] names = "、".join(item["name"] for item in active) return f"{'协作' if mode == 'collaborate' else '单角色'}:{names}"