's_a', 'start_ms' => 0, 'end_ms' => 120000, 'text' => $text]]; $citations = Prompt::citations($segments); $seen = []; $covered = []; $exact = true; foreach ($citations as $index => $citation) { $start = $index * 200; $expected = mb_substr($text, $start, 240); $exact = $exact && array_keys($citation) === ['id', 'segment_id', 'text'] && $citation['segment_id'] === 's_a' && $citation['text'] === $expected && mb_strlen($citation['text']) <= 240 && str_contains($text, $citation['text']) && preg_match('/^c_[a-f0-9]{32}$/D', $citation['id']) && !isset($seen[$citation['id']]); $seen[$citation['id']] = true; for ($j = $start; $j < $start + mb_strlen($citation['text']); $j++) { $covered[$j] = true; } } $check('exact_bounded_unique_windows_' . $length, $exact); $check('complete_source_coverage_' . $length, count($covered) === $length); $check('deterministic_' . $length, $citations === Prompt::citations($segments)); } $text = " 开头空格\n" . str_repeat('合成来源文字不可改写。', 70) . "\n末尾空格 "; $segments = [ ['id' => 'one', 'text' => $text, 'start_ms' => 0, 'end_ms' => 60000], ['id' => 'two', 'text' => $text, 'start_ms' => 60000, 'end_ms' => 120000], ]; $citations = Prompt::citations($segments); $check('identical_text_different_segment_unique_ids', count(array_unique(array_column($citations, 'id'))) === count($citations)); $check('whitespace_not_trimmed', str_starts_with($citations[0]['text'], " 开头空格\n") && str_ends_with(end($citations)['text'], "\n末尾空格 ")); $changed = $segments; $changed[0]['text'] .= '变更'; $check('source_change_invalidates_old_id', Prompt::citations($changed)[0]['id'] !== $citations[0]['id']); $changed = $segments; $changed[0]['end_ms'] = 61000; $check('boundary_change_invalidates_old_id', Prompt::citations($changed)[0]['id'] !== $citations[0]['id']); $check('silence_has_no_invented_citation', Prompt::citations([array_replace($segments[0], ['text' => ''])]) === []); $withSilence = $segments; array_unshift($withSilence, ['id' => 'silence', 'text' => '', 'start_ms' => 0, 'end_ms' => 100]); $check('silence_does_not_change_speech_ids', Prompt::citations($withSilence) === $citations); $boundary = []; for ($i = 0; $i < 1000; $i++) { $boundary[] = ['id' => 's_' . $i, 'start_ms' => $i * 1000, 'end_ms' => ($i + 1) * 1000, 'text' => '合成']; } $check('exact_maximum_segments_allowed', count(Prompt::citations($boundary)) === 1000); $normalized = Policy::normalizeExtraction(['summary' => '', 'items' => [], 'transcript' => implode("\n", array_column($boundary, 'text')), 'transcript_segments' => $boundary], '2026-01-01 09:00:00'); $check('policy_exact_maximum_segments_allowed', count($normalized['transcript_segments']) === 1000); $boundary[] = ['id' => 's_1000', 'start_ms' => 1000000, 'end_ms' => 1001000, 'text' => '合成']; $rejected = false; try { Prompt::citations($boundary); } catch (DomainException $error) { $rejected = true; } $check('maximum_plus_one_segments_rejected', $rejected); $rejected = false; try { Policy::normalizeExtraction(['summary' => '', 'items' => [], 'transcript' => implode("\n", array_column($boundary, 'text')), 'transcript_segments' => $boundary], '2026-01-01 09:00:00'); } catch (DomainException $error) { $rejected = true; } $check('policy_maximum_plus_one_segments_rejected', $rejected); foreach ([[], [$segments[0], $segments[0]], [array_replace($segments[0], ['start_ms' => -1])], [array_replace($segments[0], ['text' => "\xFF"])]] as $index => $invalid) { $rejected = false; try { Prompt::citations($invalid); } catch (DomainException $error) { $rejected = true; } $check('invalid_segments_rejected_' . $index, $rejected); } } $failed = array_keys(array_filter($checks, static fn (bool $ok): bool => !$ok)); echo json_encode(['suite' => 'followup-audio-citations-v2', 'synthetic_only' => true, 'checks' => $checks, 'passed' => count($checks) - count($failed), 'total' => count($checks), 'failures' => $failed], JSON_UNESCAPED_UNICODE | JSON_PRETTY_PRINT | JSON_THROW_ON_ERROR) . PHP_EOL; exit($failed === [] ? 0 : 1);