', 'GB18030', 'UTF-8'); foreach ($terms as $term) { $termBytes[$term] = mb_convert_encoding($term, 'GB18030', 'UTF-8'); } $paths = glob($root . '/*.txt') ?: []; sort($paths, SORT_STRING); if (count($paths) !== 701) { throw new RuntimeException('Expected 701 source text files, found ' . count($paths)); } $byTerm = array_fill_keys($terms, []); $chapterTerms = []; $invalidLineCount = 0; foreach ($paths as $path) { $filename = basename($path); $title = preg_replace('/^\d+[\-.]/u', '', preg_replace('/\.txt$/u', '', $filename) ?? $filename) ?? $filename; $chapter = ''; $chapterTerm = ''; $chapterBodyLines = 0; $bookHits = []; $file = new SplFileObject($path, 'r'); $lineNumber = 0; while (!$file->eof()) { $raw = $file->fgets(); if ($raw === false) { break; } $lineNumber++; if (!mb_check_encoding($raw, 'GB18030')) { // A few upstream files have malformed bytes. Preserve their original // line numbers and omit only those lines from the search index. $invalidLineCount++; continue; } if (str_contains($raw, $chapterMarker)) { $chapter = trim(mb_convert_encoding(str_replace($chapterMarker, '', $raw), 'UTF-8', 'GB18030')); $chapter = mb_substr($chapter, 0, 80, 'UTF-8'); $chapterTerm = preg_replace('/第[一二三四五六七八九十百千万零〇0-9].*$/u', '', $chapter) ?? $chapter; $chapterTerm = preg_replace('/(?:脉证|病证|诸证|篇|论)$/u', '', $chapterTerm) ?? $chapterTerm; $chapterTerm = trim($chapterTerm, " \t\r\n ::"); if (mb_strlen($chapterTerm, 'UTF-8') < 2 || mb_strlen($chapterTerm, 'UTF-8') > 6 || preg_match('/^\p{Han}+$/u', $chapterTerm) !== 1 || in_array($chapterTerm, ['序言', '总论', '凡例', '目录', '医论', '医案', '方论', '药论', '本草', '脉法', '用药', '病机', '证治', '杂病'], true)) { $chapterTerm = ''; } $chapterBodyLines = 0; continue; } $line = trim(mb_convert_encoding($raw, 'UTF-8', 'GB18030')); if ($line === '' || str_starts_with($line, '书名:') || str_starts_with($line, '作者:') || str_starts_with($line, '<目录>')) { continue; } $chapterBodyLines++; if ($chapterTerm !== '' && $chapterBodyLines === 1) { $chapterTerms[$chapterTerm][] = [ 'title' => $title, 'chapter' => $chapter, 'path' => $filename, 'line' => $lineNumber, 'excerpt' => mb_substr(preg_replace('/^内容:/u', '', $line) ?? $line, 0, 120, 'UTF-8'), 'score' => 12, ]; } $matched = []; foreach ($termBytes as $term => $encoded) { $bodyMatch = str_contains($raw, $encoded); $chapterMatch = $chapterBodyLines <= 3 && mb_strpos($chapter, $term, 0, 'UTF-8') !== false; if ($bodyMatch || $chapterMatch) { $matched[$term] = [$bodyMatch, $chapterMatch]; } } if ($matched === []) { continue; } $line = preg_replace('/^内容:/u', '', $line) ?? $line; $line = preg_replace('/\s+/u', ' ', $line) ?? $line; foreach ($matched as $term => [$bodyMatch, $chapterMatch]) { $pos = mb_strpos($line, $term, 0, 'UTF-8'); $start = max(0, (int) $pos - 55); $excerpt = mb_substr($line, $start, 180, 'UTF-8'); $score = ($bodyMatch ? 5 : 0) + ($chapterMatch ? 10 : 0) + (mb_strpos($title, $term, 0, 'UTF-8') !== false ? 3 : 0); if ($score > (int) ($bookHits[$term]['score'] ?? -1)) { $bookHits[$term] = [ 'title' => $title, 'chapter' => $chapter, 'path' => $filename, 'line' => $lineNumber, 'excerpt' => $excerpt, 'score' => $score, ]; } } } foreach ($bookHits as $term => $hit) { $byTerm[$term][] = $hit; } } foreach ($byTerm as &$hits) { usort($hits, static fn (array $a, array $b): int => ($b['score'] <=> $a['score']) ?: strcmp($a['path'], $b['path'])); $hits = array_slice($hits, 0, 30); } unset($hits); foreach ($chapterTerms as &$hits) { usort($hits, static fn (array $a, array $b): int => strcmp($a['path'], $b['path'])); $hits = array_slice($hits, 0, 3); } unset($hits); $shardDirectory = TcmAncientBooksReference::shardDirectory(); if (!is_dir($shardDirectory) && !mkdir($shardDirectory, 0755, true)) { throw new RuntimeException('Could not create chapter index directory'); } $shards = []; foreach ($chapterTerms as $term => $hits) { $prefix = substr(hash('sha256', (string) $term), 0, 2); $shards[$prefix][$term] = $hits; } foreach ($shards as $prefix => $entries) { $shard = json_encode([ 'commit' => TcmAncientBooksReference::COMMIT, 'index_version' => TcmAncientBooksReference::INDEX_VERSION, 'terms' => $entries, ], JSON_UNESCAPED_UNICODE | JSON_UNESCAPED_SLASHES | JSON_THROW_ON_ERROR); $shardPath = $shardDirectory . '/' . $prefix . '.json'; $shardTemporary = $shardPath . '.tmp.' . getmypid(); if (file_put_contents($shardTemporary, $shard, LOCK_EX) === false || !rename($shardTemporary, $shardPath)) { throw new RuntimeException('Could not write chapter index shard ' . $prefix); } } $index = [ 'repository' => 'local:tcm-ancient-books', 'commit' => TcmAncientBooksReference::COMMIT, 'index_version' => TcmAncientBooksReference::INDEX_VERSION, 'encoding' => 'GB18030', 'book_count' => count($paths), 'skipped_invalid_lines' => $invalidLineCount, 'terms' => $byTerm, 'chapter_term_list' => array_keys($chapterTerms), ]; $encoded = json_encode($index, JSON_UNESCAPED_UNICODE | JSON_UNESCAPED_SLASHES | JSON_THROW_ON_ERROR); $temporary = $indexPath . '.tmp.' . getmypid(); if (file_put_contents($temporary, $encoded, LOCK_EX) === false || !rename($temporary, $indexPath)) { throw new RuntimeException('Could not write index'); } echo 'Indexed ' . count($paths) . ' books from source commit ' . TcmAncientBooksReference::COMMIT . ', ' . strlen($encoded) . ' index bytes, skipped ' . $invalidLineCount . " malformed lines.\n"; } catch (Throwable $error) { fwrite(STDERR, $error->getMessage() . "\n"); exit(1); }