93 lines
3.9 KiB
Python
93 lines
3.9 KiB
Python
"""Validate model output without treating imperfect metadata as a job failure."""
|
|
from __future__ import annotations
|
|
|
|
import copy
|
|
import json
|
|
import re
|
|
|
|
from knowledge_processing import redact
|
|
|
|
|
|
class ModelOutputError(ValueError):
|
|
"""Safe local reason and usage only; never retains the provider response."""
|
|
|
|
def __init__(self, reason, input_chars=0, output_chars=0):
|
|
super().__init__(reason)
|
|
self.input_chars = input_chars
|
|
self.output_chars = output_chars
|
|
|
|
|
|
def normalize_output(candidate, text, input_chars):
|
|
output_chars = len(text) if isinstance(text, str) else 0
|
|
|
|
def invalid(reason):
|
|
raise ModelOutputError(reason, input_chars, output_chars)
|
|
|
|
if not isinstance(text, str) or not text.strip():
|
|
invalid('模型未返回知识正文')
|
|
raw = text.strip().lstrip('\ufeff')
|
|
fence = re.fullmatch(r'```(?:json)?\s*([\s\S]*?)\s*```', raw, flags=re.IGNORECASE)
|
|
if fence:
|
|
raw = fence.group(1).strip()
|
|
try:
|
|
data = json.loads(raw)
|
|
except (ValueError, RecursionError):
|
|
invalid('模型输出不是完整 JSON')
|
|
if not isinstance(data, dict):
|
|
invalid('模型输出不是 JSON 对象')
|
|
|
|
# Never stringify objects, invent an answer, or truncate clinical/business facts.
|
|
result = copy.deepcopy(candidate)
|
|
for field, label, maximum in (('question', '问题', 4000), ('answer', '答案', 8000)):
|
|
value = data.get(field)
|
|
if not isinstance(value, str) or not value.strip():
|
|
invalid(f'模型{label}字段缺失、为空或类型不正确')
|
|
value = redact(value)
|
|
if not value or len(value) > maximum:
|
|
invalid(f'模型{label}字段为空或超过 {maximum} 字符')
|
|
result[field] = value
|
|
|
|
flags = result.setdefault('flags', [])
|
|
title = data.get('title')
|
|
if not isinstance(title, str) or not title.strip():
|
|
result['title'] = result['question'][:80]
|
|
flags.append('模型未提供有效标题,已从问题生成标题,请核对')
|
|
else:
|
|
title = redact(title)
|
|
result['title'] = title[:160]
|
|
if len(title) > 160:
|
|
flags.append('模型标题过长,已缩短标题;问题与答案未截断')
|
|
|
|
category = data.get('category')
|
|
category = redact(category) if isinstance(category, str) else ''
|
|
result['category'] = category if 1 <= len(category) <= 100 else '待分类'
|
|
if result['category'] == '待分类':
|
|
flags.append('模型分类缺失或无效,请人工分类')
|
|
|
|
conditions = data.get('conditions')
|
|
conditions = redact(conditions) if isinstance(conditions, str) else ''
|
|
result['conditions'] = conditions if len(conditions) <= 2000 else ''
|
|
if not result['conditions']:
|
|
# Empty conditions deliberately block approval until an editor supplies them.
|
|
flags.append('模型未提供有效适用条件,请核对来源并补充后再审核')
|
|
|
|
kind = data.get('kind')
|
|
kind = kind.strip().lower() if isinstance(kind, str) else ''
|
|
aliases = {'问答': 'qa', '标准问答': 'qa', '流程': 'procedure', '业务流程': 'procedure',
|
|
'案例': 'case', '对话案例': 'case'}
|
|
kind = aliases.get(kind, kind)
|
|
result['kind'] = kind if kind in {'qa', 'procedure', 'case'} else 'qa'
|
|
if kind not in {'qa', 'procedure', 'case'}:
|
|
flags.append('模型知识类型无效,暂按问答保留,请核对')
|
|
flags.append('模型整理草稿,须逐项核对来源,不能直接作为正确答案')
|
|
return result, input_chars, output_chars
|
|
|
|
|
|
def original_draft(candidate, error):
|
|
"""Keep complete redacted evidence, requiring a human edit before approval."""
|
|
result = copy.deepcopy(candidate)
|
|
result['conditions'] = ''
|
|
result.setdefault('flags', []).append(
|
|
f'模型格式异常:{error};已保留原始脱敏问答,请人工整理并补充适用条件,不能直接发布')
|
|
return result
|