/** * 书库全书搜索:章节 markdown → 检索纯文本 + 关键词匹配/片段生成。 * 纯函数、无 DOM 依赖,供阅读页搜索面板(client)与单测共用。 */ import { publicMarkdownForSeo } from "./hideBlocks.ts"; import { parseEntrySource } from "./entryBlocks.ts"; /** 传给客户端的章节索引:no 为书内章序号(1 基,与 ?c=/ ?sec= 一致) */ export interface BookChapterIndex { no: number; title: string; /** markdown 经剥离后的可见纯文本 */ plain: string; } export interface BookSnippet { /** 命中前缀文(约 SNIPPET_CONTEXT 字) */ before: string; /** 命中原文(保留原始大小写) */ match: string; /** 命中后缀文 */ after: string; /** 该片段是本章内第几个命中(0 基,与 DOM 高亮顺序对应) */ ordinal: number; } export interface BookChapterHit { no: number; title: string; /** 本章命中总数 */ count: number; /** 前若干个命中的上下文片段 */ snippets: BookSnippet[]; } export const SNIPPET_CONTEXT = 28; export const SNIPPET_LIMIT = 3; const TAGS_RE = /^@tags\s+(.+)$/i; const SUMMARY_RE = /^@summary(?:\s+(.*))?$/i; const SOURCE_RE = /^@source\s+(.+)$/i; const FIELD_RE = /^([^:@\n]{1,40}?)\s*::\s*(.*)$/; /** [entries] / [entries:off] / [timeline] 等块壳标记行(hide 已由 SEO 逻辑处理) */ const DIRECTIVE_SHELL_RE = /^\[\/?(?:entries(?::off)?|timeline)\]$/i; /** * 章节 markdown → 检索用纯文本。 * 目标:页面可见的文字基本都能被搜到,且不命中标记符号。 * - 锁定的 [hide] 块整体剔除,未锁块保留正文(复用 SEO 公开文本逻辑) * - 条目卡指令保留「值」:@tags 留标签、字段::值 留「字段 值」、@summary/@source 留文本 * - 块壳标记、代码围栏、图片/链接语法、HTML 标签、强调符号一律剥离 */ export function markdownToSearchText(md: string): string { if (!md) return ""; // 统一换行;剔除锁块(未锁 hide 仅留正文,标记行已不在) const lines = publicMarkdownForSeo(md).replace(/\r\n?/g, "\n").split("\n"); const kept: string[] = []; let inCode = false; for (const raw of lines) { const t = raw.trim(); if (t.startsWith("```")) { inCode = !inCode; continue; // 围栏本身去掉,代码内容保留可搜 } if (inCode) { kept.push(raw); continue; } if (!t) continue; if (DIRECTIVE_SHELL_RE.test(t)) continue; const tags = t.match(TAGS_RE); if (tags) { kept.push(tags[1].replace(/[||]/g, " ")); continue; } const summary = t.match(SUMMARY_RE); if (summary) { if (summary[1]) kept.push(summary[1]); continue; } const source = t.match(SOURCE_RE); if (source) { // 仅保留来源文本,丢弃尾部 URL kept.push(parseEntrySource(source[1]).text); continue; } if (/^@plain$/i.test(t)) continue; const field = t.match(FIELD_RE); if (field) { kept.push(`${field[1]} ${field[2]}`); continue; } kept.push(raw); } return kept .join("\n") .replace(//g, " ") // HTML 注释(如 ) .replace(/!\[([^\]]*)\]\([^)]*\)/g, "$1") // 图片 → alt .replace(/\[([^\]]+)\]\([^)]*\)/g, "$1") // 链接 → 锚文本 .replace(/<\/?[a-z][^>]*>/gi, " ") // 内联 HTML 标签 .replace(/^#{1,6}\s+/gm, "") // 标题井号 .replace(/^\s*>+\s?/gm, "") // 引用前缀 .replace(/^\s*[-+*]\s+/gm, "") // 无序列表点 .replace(/^\s*\d+[.)]\s+/gm, "") // 有序列表序号 .replace(/[`*_~]+/g, "") // 行内代码/强调/删除线 .replace(/\|/g, " ") // 表格竖线 .replace(/\s+/g, " ") .trim(); } /** 全书关键词匹配:大小写不敏感子串,命中按章聚合,仅前 SNIPPET_LIMIT 个命中生成片段 */ export function searchBook(chapters: BookChapterIndex[], query: string): BookChapterHit[] { const q = query.trim().toLowerCase(); if (!q) return []; const hits: BookChapterHit[] = []; for (const ch of chapters) { const hay = ch.plain.toLowerCase(); const snippets: BookSnippet[] = []; let count = 0; let from = 0; for (;;) { const idx = hay.indexOf(q, from); if (idx < 0) break; if (snippets.length < SNIPPET_LIMIT) { snippets.push({ ordinal: count, before: ch.plain.slice(Math.max(0, idx - SNIPPET_CONTEXT), idx), match: ch.plain.slice(idx, idx + q.length), after: ch.plain.slice(idx + q.length, idx + q.length + SNIPPET_CONTEXT), }); } count++; from = idx + q.length; } if (count > 0) hits.push({ no: ch.no, title: ch.title, count, snippets }); } return hits; } /** 全书命中总数 */ export function countBookHits(chapters: BookChapterIndex[], query: string): number { return searchBook(chapters, query).reduce((acc, h) => acc + h.count, 0); }