Files
zyt/server/app/common/service/followupaudio/FollowupAudioPolicy.php
T

385 lines
22 KiB
PHP
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
<?php
declare(strict_types=1);
namespace app\common\service\followupaudio;
use DateTimeImmutable;
use DateTimeZone;
use DomainException;
/** Pure conservative normalization. Model text is data, never a write instruction. */
final class FollowupAudioPolicy
{
public const PERIODS = ['凌晨', '早晨', '上午', '中午', '下午', '晚上', '睡前'];
public static function canonical(array $value): string
{
$sort = static function ($item) use (&$sort) {
if (!is_array($item)) { return $item; }
if (!array_is_list($item)) { ksort($item, SORT_STRING); }
foreach ($item as $key => $child) { $item[$key] = $sort($child); }
return $item;
};
return json_encode($sort($value), JSON_UNESCAPED_UNICODE | JSON_UNESCAPED_SLASHES | JSON_PRESERVE_ZERO_FRACTION | JSON_THROW_ON_ERROR);
}
public static function hash(array $value): string { return hash('sha256', self::canonical($value)); }
public static function strictDate($value): string
{
if (!is_string($value) || !preg_match('/^\d{4}-\d{2}-\d{2}$/D', $value)) {
throw new DomainException('FOLLOWUP_AUDIO_DATE_INVALID');
}
$date = DateTimeImmutable::createFromFormat('!Y-m-d', $value, new DateTimeZone('Asia/Shanghai'));
if (!$date || $date->format('Y-m-d') !== $value || $value < '1900-01-01') {
throw new DomainException('FOLLOWUP_AUDIO_DATE_INVALID');
}
return $value;
}
public static function strictRecordedAt(string $value): string
{
$date = DateTimeImmutable::createFromFormat('!Y-m-d H:i:s', $value, new DateTimeZone('Asia/Shanghai'));
if (!$date || $date->format('Y-m-d H:i:s') !== $value || $value < '1900-01-01 00:00:00' || $date->getTimestamp() > time() + 300) {
throw new DomainException('FOLLOWUP_AUDIO_RECORDED_AT_INVALID');
}
return $value;
}
public static function dayTimestamp(string $date): int
{
return (new DateTimeImmutable(self::strictDate($date) . ' 00:00:00', new DateTimeZone('Asia/Shanghai')))->getTimestamp();
}
public static function strictTime($value): ?string
{
if ($value === null || $value === '') { return null; }
if (!is_string($value) || !preg_match('/^(?:[01]\d|2[0-3]):[0-5]\d(?::[0-5]\d)?$/D', $value)) {
throw new DomainException('FOLLOWUP_AUDIO_TIME_INVALID');
}
return substr($value, 0, 5);
}
public static function normalizeExtraction(array $result, string $recordedAt): array
{
self::strictRecordedAt($recordedAt);
if (!is_string($result['transcript'] ?? null) || trim($result['transcript']) === '' || strlen($result['transcript']) > 2000000
|| !is_string($result['summary'] ?? null) || strlen($result['summary']) > 60000 || !is_array($result['items'] ?? null)
|| count($result['items']) > 500 || !is_array($result['uncertainties'] ?? [])) {
throw new DomainException('FOLLOWUP_AUDIO_EXTRACTION_INVALID');
}
$uncertainties = [];
foreach (($result['uncertainties'] ?? []) as $uncertainty) {
if (is_string($uncertainty) && count($uncertainties) < 200) { $uncertainties[] = mb_substr($uncertainty, 0, 1000); }
}
$items = [];
$seenEvents = [];
$segments = array_key_exists('transcript_segments', $result) ? self::segments($result['transcript_segments']) : null;
$segmentIndex = $segments === null ? [] : array_column($segments, null, 'id');
foreach ($result['items'] as $index => $raw) {
try {
if (!is_array($raw) || !is_string($raw['kind'] ?? null) || !is_array($raw['values'] ?? null)) {
throw new DomainException('FOLLOWUP_AUDIO_ITEM_INVALID');
}
$kind = $raw['kind'];
$values = FollowupAudioFields::validateValues($kind, $raw['values']);
$evidence = self::evidence($raw['evidence'] ?? []);
$quoted = $evidence !== [];
foreach ($evidence as $quote) {
if (!str_contains($result['transcript'], $quote['text'])) { $quoted = false; }
if ($segments !== null) {
$segment = $segmentIndex[$quote['segment_id'] ?? ''] ?? null;
if ($segment === null || !str_contains($segment['text'], $quote['text'])
|| ($quote['position_type'] ?? '') !== 'segment'
|| ($quote['start_ms'] ?? null) !== $segment['start_ms']
|| ($quote['end_ms'] ?? null) !== $segment['end_ms']) {
throw new DomainException('FOLLOWUP_AUDIO_EVIDENCE_SEGMENT_INVALID');
}
} elseif (isset($quote['segment_id'])) {
throw new DomainException('FOLLOWUP_AUDIO_EVIDENCE_SEGMENT_INVALID');
}
}
$dateText = self::shortText($raw['date_text'] ?? '');
$timeText = self::shortText($raw['time_text'] ?? '');
[$date, $dateText, $dateReview] = self::evidenceDate($raw['record_date'] ?? null, $dateText,
$recordedAt, $result['transcript'], $quoted ? $evidence : [], $segments !== null, $segmentIndex);
$time = self::strictTime($raw['record_time'] ?? null);
$period = self::period($raw['time_period'] ?? $timeText);
// Approved default clocks are estimates, never claims of an exact spoken measurement time.
$estimated = ($time === null && $kind !== 'diagnosis') || !empty($raw['time_estimated']);
if ($time === null && $kind !== 'diagnosis') { $time = self::estimatedTime($period); }
$needsReview = !empty($raw['needs_review']) || !$quoted || $dateReview
|| FollowupAudioFields::requiresClinicalReview($kind, $values)
|| self::riskyEvidenceContext($result['transcript'], $evidence)
|| ($kind !== 'diagnosis' && ($date === null || $estimated));
if ($date !== null && ($date > substr($recordedAt, 0, 10) || ($kind === 'diagnosis' && $date < substr($recordedAt, 0, 10)))) { $needsReview = true; }
$item = [
'id' => '', 'kind' => $kind, 'values' => $values, 'record_date' => $date, 'record_time' => $time,
'time_period' => $period, 'time_estimated' => $estimated, 'time_text' => $timeText, 'date_text' => $dateText,
'evidence' => $evidence, 'needs_review' => $needsReview, 'selected' => false, 'target_id' => null,
'current_values' => [], 'expected_hash' => '', 'conflict' => false, 'possible_duplicate' => false,
];
// Equal measurements/default times are NOT enough to collapse different utterances.
$identity = ['kind' => $kind, 'values' => $values, 'date' => $date, 'time' => $time, 'period' => $period, 'evidence' => $evidence];
$eventHash = self::hash($identity);
if ($evidence !== [] && isset($seenEvents[$eventHash])) { continue; }
$seenEvents[$eventHash] = true;
$item['id'] = 'i_' . substr(self::hash([$identity, 'index' => $index]), 0, 32);
$items[] = $item;
} catch (DomainException $exception) {
// Retain transcript for a human; do not silently turn unknown/invalid values into "normal".
$uncertainties[] = '第' . ($index + 1) . '项未进入可写字段:' . $exception->getMessage();
}
}
$normalized = ['summary' => trim($result['summary']), 'transcript' => $result['transcript'],
'uncertainties' => array_slice($uncertainties, 0, 200), 'items' => $items];
if ($segments !== null) { $normalized['transcript_segments'] = $segments; }
return $normalized;
}
/**
* Conservative review gate, NOT a semantic classifier. A substring proves neither
* speaker identity nor negation/current-vs-historical meaning. Look around every
* occurrence so a shortened quote cannot hide an adjacent question or family cue.
*/
private static function riskyEvidenceContext(string $transcript, array $evidence): bool
{
$pattern = '/(?:家属|家人|父亲|母亲|爸爸|妈妈|父母|儿子|女儿|丈夫|妻子|老伴|爱人|爷爷|奶奶|姥姥|姥爷|'
. '哥哥|姐姐|弟弟|妹妹|兄弟|姐妹|孩子|他们|她们|他(?:的|在|测|用)|她(?:的|在|测|用)|'
. '客服|工作人员|销售|推测|可能是|请问|问[::]|[??]|是否|有没有|是不是|多少|吗|呢|否认|没有|没|未|不是|不|'
. '更正|纠正|说错|口误|改成|其实|好像|可能|大概|大约|差不多|估计|似乎|左右|记不清|忘记|以前|曾经|过去|之前|上次|去年|前年|往年|当时|停药|停用|'
. '\b(?:family|father|mother|wife|husband|son|daughter|no|not|never|denied|maybe|uncertain|previously|stopped|correction)\b)/iu';
foreach ($evidence as $quote) {
$offset = 0;
$occurrences = 0;
$length = mb_strlen($quote['text'], 'UTF-8');
while (($position = mb_strpos($transcript, $quote['text'], $offset, 'UTF-8')) !== false) {
// Excessively repeated fragments cannot establish a unique reliable context.
if (++$occurrences > 20) { return true; }
$start = max(0, $position - 160);
$context = mb_substr($transcript, $start, $position - $start + $length + 160, 'UTF-8');
if (preg_match($pattern, $context)) { return true; }
$offset = $position + max(1, $length);
}
}
return false;
}
public static function period($value): ?string
{
if ($value === null || $value === '') { return null; }
if (!is_string($value)) { throw new DomainException('FOLLOWUP_AUDIO_PERIOD_INVALID'); }
$aliases = ['morning' => '上午', 'noon' => '中午', 'afternoon' => '下午', 'evening' => '晚上',
'night' => '晚上', 'bedtime' => '睡前', '早上' => '早晨', '清晨' => '早晨', '傍晚' => '晚上'];
$value = $aliases[$value] ?? $value;
foreach (self::PERIODS as $period) {
if (str_contains($value, $period)) { return $period; }
}
return null;
}
public static function estimatedTime(?string $period): ?string
{
return ['早晨' => '08:00', '上午' => '08:00', '中午' => '12:00', '下午' => '15:00',
'晚上' => '20:00', '睡前' => '22:00'][$period ?? ''] ?? null;
}
private static function resolveDate($explicit, string $text, string $recordedAt): ?string
{
// A model-supplied calendar date cannot resolve an explicitly ambiguous spoken range.
if (preg_match('/(?:或|至|到|最近|这几天|前几天|近几天|上周|上星期|上个月|不记得|记不清|大概|左右|between|around)/iu', $text)) { return null; }
$base = new DateTimeImmutable(substr($recordedAt, 0, 10), new DateTimeZone('Asia/Shanghai'));
$relative = null;
foreach (['前天' => '-2 days', '昨天' => '-1 day', '昨日' => '-1 day', '今天' => '+0 days', '今日' => '+0 days',
'yesterday' => '-1 day', 'today' => '+0 days'] as $word => $offset) {
if ($text === $word || str_starts_with($text, $word)) { $relative = $base->modify($offset)->format('Y-m-d'); break; }
}
if ($explicit !== null && $explicit !== '') {
$date = self::strictDate($explicit);
// A relative phrase and supplied date disagree: leave unresolved instead of trusting one silently.
return $relative !== null && $relative !== $date ? null : $date;
}
if ($relative !== null) { return $relative; }
if (preg_match('/^\d{4}-\d{2}-\d{2}$/D', $text)) { return self::strictDate($text); }
if (preg_match('/^(\d{4})年(\d{1,2})月(\d{1,2})日$/Du', $text, $match)) {
return self::strictDate(sprintf('%04d-%02d-%02d', $match[1], $match[2], $match[3]));
}
return null;
}
/** A model may omit or mislabel date_text; the cited words still constrain its date. */
private static function evidenceDate($explicit, string $text, string $recordedAt, string $transcript,
array $evidence, bool $requiresGrounding, array $segmentIndex): array
{
// Preserve invalid-date rejection and the established conflicting-date => null behavior.
$supplied = $explicit === null || $explicit === '' ? null : self::strictDate($explicit);
// A server-owned citation is a broad window, not an event/date binding. Preserve a
// model's abstention, including on repeated normalization of a cleared conflict.
// An explicit calendar without its spoken date label is equally unbound.
if ($requiresGrounding && ($supplied === null || $text === '')) { return [null, $text, true]; }
$dates = [];
$phrases = [];
$ambiguous = false;
$directlyGrounded = false;
$spokenRange = '';
foreach ($evidence as $quote) {
// A repeated fragment in another chunk is not evidence for this chunk's date.
$source = $segmentIndex[$quote['segment_id'] ?? '']['text'] ?? $transcript;
$contexts = [$quote['text']];
[$directDates, , $directAmbiguous] = self::dateCues($quote['text'], $recordedAt);
$directlyGrounded = $directlyGrounded || $directDates !== [];
if ($spokenRange === '' && ($directAmbiguous || count(array_unique($directDates)) > 1)) { $spokenRange = $quote['text']; }
if ($directDates === [] && !$directAmbiguous) {
// Recover an omitted date prefix only within the same sentence, never another event.
$contexts = [];
$offset = 0;
while (($position = mb_strpos($source, $quote['text'], $offset, 'UTF-8')) !== false) {
if (count($contexts) >= 20) { $ambiguous = true; break; }
$start = max(0, $position - 160);
$before = mb_substr($source, $start, $position - $start, 'UTF-8');
$after = mb_substr($source, $position + mb_strlen($quote['text'], 'UTF-8'), 160, 'UTF-8');
$prefix = preg_split('/[。!?!?;;\r\n]/u', $before);
$suffix = preg_split('/[。!?!?;;\r\n]/u', $after);
$contexts[] = end($prefix) . $quote['text'] . ($suffix[0] ?? '');
$offset = $position + max(1, mb_strlen($quote['text'], 'UTF-8'));
}
}
foreach ($contexts as $context) {
[$found, $words, $vague] = self::dateCues($context, $recordedAt);
$dates = array_merge($dates, $found);
$phrases = array_merge($phrases, $words);
$ambiguous = $ambiguous || $vague;
}
}
$dates = array_values(array_unique($dates));
[$claimedDates, , $claimedAmbiguous] = self::dateCues($text, $recordedAt);
$claimedDates = array_values(array_unique($claimedDates));
if ($text === '' && $phrases !== [] && ($directlyGrounded || $ambiguous)) {
$text = mb_substr($spokenRange !== '' ? $spokenRange : implode(';', array_unique($phrases)), 0, 255, 'UTF-8');
}
if ($ambiguous || $claimedAmbiguous || count($dates) > 1 || count($claimedDates) > 1) {
return [null, $text, true];
}
if ($dates !== []) {
$grounded = $dates[0];
if (($supplied !== null && $supplied !== $grounded)
|| ($claimedDates !== [] && $claimedDates[0] !== $grounded)) {
return [null, $text, true];
}
// Neighbor words can invalidate a conflicting date, but do not prove this event's date.
// Legacy transcripts may concatenate independent utterances without sentence boundaries.
if (!$directlyGrounded) {
return $requiresGrounding ? [null, $text, true] : [self::resolveDate($explicit, $text, $recordedAt), $text, false];
}
// An ambiguous model date label is not silently replaced by an exact evidence date.
if (self::resolveDate(null, $text, $recordedAt) === null && preg_match('/(?:或|至|到|大概|左右|around|between)/iu', $text)) {
return [null, $text, true];
}
return [$grounded, $text, false];
}
// A strict transcript pipeline cannot invent a calendar day when none was spoken.
if ($requiresGrounding) { return [null, $text, true]; }
return [self::resolveDate($explicit, $text, $recordedAt), $text, false];
}
/** Literal temporal cues only, not a general natural-language or speaker classifier. */
private static function dateCues(string $text, string $recordedAt): array
{
$base = new DateTimeImmutable(substr($recordedAt, 0, 10), new DateTimeZone('Asia/Shanghai'));
$offsets = ['大前天' => -3, '前天' => -2, '昨天' => -1, '昨日' => -1, '今天' => 0, '今日' => 0,
'明天' => 1, '明日' => 1, '后天' => 2, 'day before yesterday' => -2, 'yesterday' => -1, 'today' => 0, 'tomorrow' => 1];
$dates = [];
$phrases = [];
preg_match_all('/大前天|前天|昨天|昨日|今天|今日|明天|明日|后天|\b(?:day before yesterday|yesterday|today|tomorrow)\b/iu', $text, $matches);
foreach ($matches[0] as $word) {
$dates[] = $base->modify(sprintf('%+d days', $offsets[strtolower($word)]))->format('Y-m-d');
$phrases[] = $word;
}
preg_match_all('/(?<!\d)(\d{4})[-年](\d{1,2})[-月](\d{1,2})(?:日|号)?(?!\d)/u', $text, $matches, PREG_SET_ORDER);
foreach ($matches as $match) {
$dates[] = self::strictDate(sprintf('%04d-%02d-%02d', $match[1], $match[2], $match[3]));
$phrases[] = $match[0];
}
preg_match_all('/最近(?:几天|一周|一个月)?|这几天|前几天|近几天|前两天|前段时间|这段时间|上周|上星期|上个月|去年|前年|今年|往年|\d{4}年(?!\d{1,2}月)|不记得哪天|记不清哪天|\blast (?:week|month|year)\b|\brecently\b/iu', $text, $vague);
// Numeric uncertainty (e.g. 昨天读数大概120左右) does not erase an unambiguous date.
$ambiguous = $vague[0] !== [] || preg_match('/(?:大概|可能|不确定是|记不清是)(?:大前天|前天|昨天|今天)|(?:大前天|前天|昨天|今天)(?:左右|前后)/u', $text) === 1;
preg_match_all('/(?:星期|周|礼拜)([一二三四五六日天])/u', $text, $weekdays, PREG_SET_ORDER);
$weekdayNumbers = ['一' => 1, '二' => 2, '三' => 3, '四' => 4, '五' => 5, '六' => 6, '日' => 7, '天' => 7];
foreach ($weekdays as $weekday) {
$phrases[] = $weekday[0];
if ($dates === []) { $ambiguous = true; }
foreach ($dates as $date) {
if ((int) (new DateTimeImmutable($date, new DateTimeZone('Asia/Shanghai')))->format('N') !== $weekdayNumbers[$weekday[1]]) {
$ambiguous = true;
}
}
}
return [$dates, array_merge($phrases, $vague[0]), $ambiguous];
}
private static function segments($raw): array
{
if (!is_array($raw) || !array_is_list($raw) || $raw === [] || count($raw) > FollowupAudioTranscriptPrompt::MAX_SEGMENTS) {
throw new DomainException('FOLLOWUP_AUDIO_SEGMENTS_INVALID');
}
$seen = [];
foreach ($raw as $segment) {
if (!is_array($segment) || !is_string($segment['id'] ?? null)
|| !preg_match('/^[A-Za-z0-9_-]{1,128}$/D', $segment['id']) || isset($seen[$segment['id']])
|| !is_string($segment['text'] ?? null) || strlen($segment['text']) > 2000000
|| !is_int($segment['start_ms'] ?? null) || !is_int($segment['end_ms'] ?? null)
|| $segment['start_ms'] < 0 || $segment['end_ms'] <= $segment['start_ms'] || $segment['end_ms'] > 3600000) {
throw new DomainException('FOLLOWUP_AUDIO_SEGMENTS_INVALID');
}
$seen[$segment['id']] = true;
}
return $raw;
}
private static function shortText($value): string
{
if (!is_string($value) || mb_strlen($value) > 255) { throw new DomainException('FOLLOWUP_AUDIO_TEXT_INVALID'); }
return trim($value);
}
private static function evidence($raw): array
{
if (!is_array($raw) || count($raw) > 30) { throw new DomainException('FOLLOWUP_AUDIO_EVIDENCE_INVALID'); }
$evidence = [];
foreach ($raw as $entry) {
if (!is_array($entry) || !is_string($entry['text'] ?? null) || trim($entry['text']) === '' || mb_strlen($entry['text']) > 5000) {
throw new DomainException('FOLLOWUP_AUDIO_EVIDENCE_INVALID');
}
$quote = [];
if (array_key_exists('segment_id', $entry)) {
if (!is_string($entry['segment_id']) || !preg_match('/^[A-Za-z0-9_-]{1,128}$/D', $entry['segment_id'])
|| ($entry['position_type'] ?? '') !== 'segment' || !isset($entry['start_ms'], $entry['end_ms'])) {
throw new DomainException('FOLLOWUP_AUDIO_EVIDENCE_SEGMENT_INVALID');
}
$quote['segment_id'] = $entry['segment_id'];
}
$quote['text'] = trim($entry['text']);
foreach (['start_ms', 'end_ms'] as $key) {
if (isset($entry[$key])) {
if (!is_int($entry[$key]) || $entry[$key] < 0 || $entry[$key] > 3600000) {
throw new DomainException('FOLLOWUP_AUDIO_EVIDENCE_TIME_INVALID');
}
$quote[$key] = $entry[$key];
}
}
if (isset($quote['start_ms'], $quote['end_ms']) && $quote['end_ms'] < $quote['start_ms']) {
throw new DomainException('FOLLOWUP_AUDIO_EVIDENCE_TIME_INVALID');
}
if (array_key_exists('position_type', $entry)) {
if ($entry['position_type'] !== 'segment' || !isset($quote['segment_id'])) {
throw new DomainException('FOLLOWUP_AUDIO_EVIDENCE_SEGMENT_INVALID');
}
$quote['position_type'] = 'segment';
}
$evidence[] = $quote;
}
return $evidence;
}
}