feat: add durable ASR-to-patient extraction and guarded review
This commit is contained in:
@@ -75,6 +75,8 @@ final class FollowupAudioPolicy
|
||||
}
|
||||
$items = [];
|
||||
$seenEvents = [];
|
||||
$segments = array_key_exists('transcript_segments', $result) ? self::segments($result['transcript_segments']) : null;
|
||||
$segmentIndex = $segments === null ? [] : array_column($segments, null, 'id');
|
||||
foreach ($result['items'] as $index => $raw) {
|
||||
try {
|
||||
if (!is_array($raw) || !is_string($raw['kind'] ?? null) || !is_array($raw['values'] ?? null)) {
|
||||
@@ -86,16 +88,28 @@ final class FollowupAudioPolicy
|
||||
$quoted = $evidence !== [];
|
||||
foreach ($evidence as $quote) {
|
||||
if (!str_contains($result['transcript'], $quote['text'])) { $quoted = false; }
|
||||
if ($segments !== null) {
|
||||
$segment = $segmentIndex[$quote['segment_id'] ?? ''] ?? null;
|
||||
if ($segment === null || !str_contains($segment['text'], $quote['text'])
|
||||
|| ($quote['position_type'] ?? '') !== 'segment'
|
||||
|| ($quote['start_ms'] ?? null) !== $segment['start_ms']
|
||||
|| ($quote['end_ms'] ?? null) !== $segment['end_ms']) {
|
||||
throw new DomainException('FOLLOWUP_AUDIO_EVIDENCE_SEGMENT_INVALID');
|
||||
}
|
||||
} elseif (isset($quote['segment_id'])) {
|
||||
throw new DomainException('FOLLOWUP_AUDIO_EVIDENCE_SEGMENT_INVALID');
|
||||
}
|
||||
}
|
||||
$dateText = self::shortText($raw['date_text'] ?? '');
|
||||
$timeText = self::shortText($raw['time_text'] ?? '');
|
||||
$date = self::resolveDate($raw['record_date'] ?? null, $dateText, $recordedAt);
|
||||
[$date, $dateText, $dateReview] = self::evidenceDate($raw['record_date'] ?? null, $dateText,
|
||||
$recordedAt, $result['transcript'], $quoted ? $evidence : [], $segments !== null, $segmentIndex);
|
||||
$time = self::strictTime($raw['record_time'] ?? null);
|
||||
$period = self::period($raw['time_period'] ?? $timeText);
|
||||
// Approved default clocks are estimates, never claims of an exact spoken measurement time.
|
||||
$estimated = ($time === null && $kind !== 'diagnosis') || !empty($raw['time_estimated']);
|
||||
if ($time === null && $kind !== 'diagnosis') { $time = self::estimatedTime($period); }
|
||||
$needsReview = !empty($raw['needs_review']) || !$quoted
|
||||
$needsReview = !empty($raw['needs_review']) || !$quoted || $dateReview
|
||||
|| FollowupAudioFields::requiresClinicalReview($kind, $values)
|
||||
|| self::riskyEvidenceContext($result['transcript'], $evidence)
|
||||
|| ($kind !== 'diagnosis' && ($date === null || $estimated));
|
||||
@@ -118,8 +132,10 @@ final class FollowupAudioPolicy
|
||||
$uncertainties[] = '第' . ($index + 1) . '项未进入可写字段:' . $exception->getMessage();
|
||||
}
|
||||
}
|
||||
return ['summary' => trim($result['summary']), 'transcript' => $result['transcript'],
|
||||
$normalized = ['summary' => trim($result['summary']), 'transcript' => $result['transcript'],
|
||||
'uncertainties' => array_slice($uncertainties, 0, 200), 'items' => $items];
|
||||
if ($segments !== null) { $normalized['transcript_segments'] = $segments; }
|
||||
return $normalized;
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -131,8 +147,8 @@ final class FollowupAudioPolicy
|
||||
{
|
||||
$pattern = '/(?:家属|家人|父亲|母亲|爸爸|妈妈|父母|儿子|女儿|丈夫|妻子|老伴|爱人|爷爷|奶奶|姥姥|姥爷|'
|
||||
. '哥哥|姐姐|弟弟|妹妹|兄弟|姐妹|孩子|他们|她们|他(?:的|在|测|用)|她(?:的|在|测|用)|'
|
||||
. '客服|请问|问[::]|[??]|是否|有没有|是不是|多少|吗|呢|否认|没有|没|未|不是|不|'
|
||||
. '更正|纠正|说错|口误|改成|其实|好像|可能|大概|左右|记不清|忘记|以前|曾经|过去|之前|上次|停药|停用|'
|
||||
. '客服|工作人员|销售|推测|可能是|请问|问[::]|[??]|是否|有没有|是不是|多少|吗|呢|否认|没有|没|未|不是|不|'
|
||||
. '更正|纠正|说错|口误|改成|其实|好像|可能|大概|大约|差不多|估计|似乎|左右|记不清|忘记|以前|曾经|过去|之前|上次|去年|前年|往年|当时|停药|停用|'
|
||||
. '\b(?:family|father|mother|wife|husband|son|daughter|no|not|never|denied|maybe|uncertain|previously|stopped|correction)\b)/iu';
|
||||
foreach ($evidence as $quote) {
|
||||
$offset = 0;
|
||||
@@ -192,6 +208,135 @@ final class FollowupAudioPolicy
|
||||
return null;
|
||||
}
|
||||
|
||||
/** A model may omit or mislabel date_text; the cited words still constrain its date. */
|
||||
private static function evidenceDate($explicit, string $text, string $recordedAt, string $transcript,
|
||||
array $evidence, bool $requiresGrounding, array $segmentIndex): array
|
||||
{
|
||||
// Preserve invalid-date rejection and the established conflicting-date => null behavior.
|
||||
$supplied = $explicit === null || $explicit === '' ? null : self::strictDate($explicit);
|
||||
// A server-owned citation is a broad window, not an event/date binding. Preserve a
|
||||
// model's abstention, including on repeated normalization of a cleared conflict.
|
||||
// An explicit calendar without its spoken date label is equally unbound.
|
||||
if ($requiresGrounding && ($supplied === null || $text === '')) { return [null, $text, true]; }
|
||||
$dates = [];
|
||||
$phrases = [];
|
||||
$ambiguous = false;
|
||||
$directlyGrounded = false;
|
||||
$spokenRange = '';
|
||||
foreach ($evidence as $quote) {
|
||||
// A repeated fragment in another chunk is not evidence for this chunk's date.
|
||||
$source = $segmentIndex[$quote['segment_id'] ?? '']['text'] ?? $transcript;
|
||||
$contexts = [$quote['text']];
|
||||
[$directDates, , $directAmbiguous] = self::dateCues($quote['text'], $recordedAt);
|
||||
$directlyGrounded = $directlyGrounded || $directDates !== [];
|
||||
if ($spokenRange === '' && ($directAmbiguous || count(array_unique($directDates)) > 1)) { $spokenRange = $quote['text']; }
|
||||
if ($directDates === [] && !$directAmbiguous) {
|
||||
// Recover an omitted date prefix only within the same sentence, never another event.
|
||||
$contexts = [];
|
||||
$offset = 0;
|
||||
while (($position = mb_strpos($source, $quote['text'], $offset, 'UTF-8')) !== false) {
|
||||
if (count($contexts) >= 20) { $ambiguous = true; break; }
|
||||
$start = max(0, $position - 160);
|
||||
$before = mb_substr($source, $start, $position - $start, 'UTF-8');
|
||||
$after = mb_substr($source, $position + mb_strlen($quote['text'], 'UTF-8'), 160, 'UTF-8');
|
||||
$prefix = preg_split('/[。!?!?;;\r\n]/u', $before);
|
||||
$suffix = preg_split('/[。!?!?;;\r\n]/u', $after);
|
||||
$contexts[] = end($prefix) . $quote['text'] . ($suffix[0] ?? '');
|
||||
$offset = $position + max(1, mb_strlen($quote['text'], 'UTF-8'));
|
||||
}
|
||||
}
|
||||
foreach ($contexts as $context) {
|
||||
[$found, $words, $vague] = self::dateCues($context, $recordedAt);
|
||||
$dates = array_merge($dates, $found);
|
||||
$phrases = array_merge($phrases, $words);
|
||||
$ambiguous = $ambiguous || $vague;
|
||||
}
|
||||
}
|
||||
$dates = array_values(array_unique($dates));
|
||||
[$claimedDates, , $claimedAmbiguous] = self::dateCues($text, $recordedAt);
|
||||
$claimedDates = array_values(array_unique($claimedDates));
|
||||
if ($text === '' && $phrases !== [] && ($directlyGrounded || $ambiguous)) {
|
||||
$text = mb_substr($spokenRange !== '' ? $spokenRange : implode(';', array_unique($phrases)), 0, 255, 'UTF-8');
|
||||
}
|
||||
if ($ambiguous || $claimedAmbiguous || count($dates) > 1 || count($claimedDates) > 1) {
|
||||
return [null, $text, true];
|
||||
}
|
||||
if ($dates !== []) {
|
||||
$grounded = $dates[0];
|
||||
if (($supplied !== null && $supplied !== $grounded)
|
||||
|| ($claimedDates !== [] && $claimedDates[0] !== $grounded)) {
|
||||
return [null, $text, true];
|
||||
}
|
||||
// Neighbor words can invalidate a conflicting date, but do not prove this event's date.
|
||||
// Legacy transcripts may concatenate independent utterances without sentence boundaries.
|
||||
if (!$directlyGrounded) {
|
||||
return $requiresGrounding ? [null, $text, true] : [self::resolveDate($explicit, $text, $recordedAt), $text, false];
|
||||
}
|
||||
// An ambiguous model date label is not silently replaced by an exact evidence date.
|
||||
if (self::resolveDate(null, $text, $recordedAt) === null && preg_match('/(?:或|至|到|大概|左右|around|between)/iu', $text)) {
|
||||
return [null, $text, true];
|
||||
}
|
||||
return [$grounded, $text, false];
|
||||
}
|
||||
// A strict transcript pipeline cannot invent a calendar day when none was spoken.
|
||||
if ($requiresGrounding) { return [null, $text, true]; }
|
||||
return [self::resolveDate($explicit, $text, $recordedAt), $text, false];
|
||||
}
|
||||
|
||||
/** Literal temporal cues only, not a general natural-language or speaker classifier. */
|
||||
private static function dateCues(string $text, string $recordedAt): array
|
||||
{
|
||||
$base = new DateTimeImmutable(substr($recordedAt, 0, 10), new DateTimeZone('Asia/Shanghai'));
|
||||
$offsets = ['大前天' => -3, '前天' => -2, '昨天' => -1, '昨日' => -1, '今天' => 0, '今日' => 0,
|
||||
'明天' => 1, '明日' => 1, '后天' => 2, 'day before yesterday' => -2, 'yesterday' => -1, 'today' => 0, 'tomorrow' => 1];
|
||||
$dates = [];
|
||||
$phrases = [];
|
||||
preg_match_all('/大前天|前天|昨天|昨日|今天|今日|明天|明日|后天|\b(?:day before yesterday|yesterday|today|tomorrow)\b/iu', $text, $matches);
|
||||
foreach ($matches[0] as $word) {
|
||||
$dates[] = $base->modify(sprintf('%+d days', $offsets[strtolower($word)]))->format('Y-m-d');
|
||||
$phrases[] = $word;
|
||||
}
|
||||
preg_match_all('/(?<!\d)(\d{4})[-年](\d{1,2})[-月](\d{1,2})(?:日|号)?(?!\d)/u', $text, $matches, PREG_SET_ORDER);
|
||||
foreach ($matches as $match) {
|
||||
$dates[] = self::strictDate(sprintf('%04d-%02d-%02d', $match[1], $match[2], $match[3]));
|
||||
$phrases[] = $match[0];
|
||||
}
|
||||
preg_match_all('/最近(?:几天|一周|一个月)?|这几天|前几天|近几天|前两天|前段时间|这段时间|上周|上星期|上个月|去年|前年|今年|往年|\d{4}年(?!\d{1,2}月)|不记得哪天|记不清哪天|\blast (?:week|month|year)\b|\brecently\b/iu', $text, $vague);
|
||||
// Numeric uncertainty (e.g. 昨天读数大概120左右) does not erase an unambiguous date.
|
||||
$ambiguous = $vague[0] !== [] || preg_match('/(?:大概|可能|不确定是|记不清是)(?:大前天|前天|昨天|今天)|(?:大前天|前天|昨天|今天)(?:左右|前后)/u', $text) === 1;
|
||||
preg_match_all('/(?:星期|周|礼拜)([一二三四五六日天])/u', $text, $weekdays, PREG_SET_ORDER);
|
||||
$weekdayNumbers = ['一' => 1, '二' => 2, '三' => 3, '四' => 4, '五' => 5, '六' => 6, '日' => 7, '天' => 7];
|
||||
foreach ($weekdays as $weekday) {
|
||||
$phrases[] = $weekday[0];
|
||||
if ($dates === []) { $ambiguous = true; }
|
||||
foreach ($dates as $date) {
|
||||
if ((int) (new DateTimeImmutable($date, new DateTimeZone('Asia/Shanghai')))->format('N') !== $weekdayNumbers[$weekday[1]]) {
|
||||
$ambiguous = true;
|
||||
}
|
||||
}
|
||||
}
|
||||
return [$dates, array_merge($phrases, $vague[0]), $ambiguous];
|
||||
}
|
||||
|
||||
private static function segments($raw): array
|
||||
{
|
||||
if (!is_array($raw) || !array_is_list($raw) || $raw === [] || count($raw) > FollowupAudioTranscriptPrompt::MAX_SEGMENTS) {
|
||||
throw new DomainException('FOLLOWUP_AUDIO_SEGMENTS_INVALID');
|
||||
}
|
||||
$seen = [];
|
||||
foreach ($raw as $segment) {
|
||||
if (!is_array($segment) || !is_string($segment['id'] ?? null)
|
||||
|| !preg_match('/^[A-Za-z0-9_-]{1,128}$/D', $segment['id']) || isset($seen[$segment['id']])
|
||||
|| !is_string($segment['text'] ?? null) || strlen($segment['text']) > 2000000
|
||||
|| !is_int($segment['start_ms'] ?? null) || !is_int($segment['end_ms'] ?? null)
|
||||
|| $segment['start_ms'] < 0 || $segment['end_ms'] <= $segment['start_ms'] || $segment['end_ms'] > 3600000) {
|
||||
throw new DomainException('FOLLOWUP_AUDIO_SEGMENTS_INVALID');
|
||||
}
|
||||
$seen[$segment['id']] = true;
|
||||
}
|
||||
return $raw;
|
||||
}
|
||||
|
||||
private static function shortText($value): string
|
||||
{
|
||||
if (!is_string($value) || mb_strlen($value) > 255) { throw new DomainException('FOLLOWUP_AUDIO_TEXT_INVALID'); }
|
||||
@@ -206,7 +351,15 @@ final class FollowupAudioPolicy
|
||||
if (!is_array($entry) || !is_string($entry['text'] ?? null) || trim($entry['text']) === '' || mb_strlen($entry['text']) > 5000) {
|
||||
throw new DomainException('FOLLOWUP_AUDIO_EVIDENCE_INVALID');
|
||||
}
|
||||
$quote = ['text' => trim($entry['text'])];
|
||||
$quote = [];
|
||||
if (array_key_exists('segment_id', $entry)) {
|
||||
if (!is_string($entry['segment_id']) || !preg_match('/^[A-Za-z0-9_-]{1,128}$/D', $entry['segment_id'])
|
||||
|| ($entry['position_type'] ?? '') !== 'segment' || !isset($entry['start_ms'], $entry['end_ms'])) {
|
||||
throw new DomainException('FOLLOWUP_AUDIO_EVIDENCE_SEGMENT_INVALID');
|
||||
}
|
||||
$quote['segment_id'] = $entry['segment_id'];
|
||||
}
|
||||
$quote['text'] = trim($entry['text']);
|
||||
foreach (['start_ms', 'end_ms'] as $key) {
|
||||
if (isset($entry[$key])) {
|
||||
if (!is_int($entry[$key]) || $entry[$key] < 0 || $entry[$key] > 3600000) {
|
||||
@@ -218,6 +371,12 @@ final class FollowupAudioPolicy
|
||||
if (isset($quote['start_ms'], $quote['end_ms']) && $quote['end_ms'] < $quote['start_ms']) {
|
||||
throw new DomainException('FOLLOWUP_AUDIO_EVIDENCE_TIME_INVALID');
|
||||
}
|
||||
if (array_key_exists('position_type', $entry)) {
|
||||
if ($entry['position_type'] !== 'segment' || !isset($quote['segment_id'])) {
|
||||
throw new DomainException('FOLLOWUP_AUDIO_EVIDENCE_SEGMENT_INVALID');
|
||||
}
|
||||
$quote['position_type'] = 'segment';
|
||||
}
|
||||
$evidence[] = $quote;
|
||||
}
|
||||
return $evidence;
|
||||
|
||||
Reference in New Issue
Block a user