/** * epub 导入解析(浏览器端):epub 本质是 zip + XHTML。 * 流程:JSZip 解压 → META-INF/container.xml 找 OPF → 按 spine 顺序读 XHTML * → 清理(去 script/style、剔除内部相对路径图片)→ turndown 转 Markdown。 * 产物直接作为伪 .md 文件走现有批量导入接口,后端零改动。 * 两个依赖均在本模块内动态 import,避免打进主包。 */ export interface EpubChapter { title: string; markdown: string; } /** 以 OPF 所在目录为基准解析 zip 内相对路径(处理 ./ ../ 与 URL 编码) */ function resolveZipPath(base: string, href: string): string { const parts = `${base}${href}`.split("/"); const out: string[] = []; for (const p of parts) { if (!p || p === ".") continue; if (p === "..") out.pop(); else out.push(p); } return out.join("/"); } /** 清理标题里不适合做文件名的字符 */ function sanitizeTitle(t: string): string { return t.replace(/[\\/:*?"<>|\n\r\t]+/g, " ").trim().slice(0, 60); } export async function parseEpub(file: File): Promise { const { default: JSZip } = await import("jszip"); const { default: TurndownService } = await import("turndown"); const zip = await JSZip.loadAsync(await file.arrayBuffer()).catch(() => { throw new Error("无法读取 epub:不是有效的 zip 文件"); }); // container.xml → OPF 路径 const containerEntry = zip.file("META-INF/container.xml"); if (!containerEntry) throw new Error("无法读取 epub:缺少 META-INF/container.xml"); const containerDoc = new DOMParser().parseFromString(await containerEntry.async("text"), "application/xml"); const rootfilePath = containerDoc.querySelector("rootfile")?.getAttribute("full-path"); if (!rootfilePath) throw new Error("无法读取 epub:container.xml 未声明 OPF 文件"); // OPF → manifest(id → href/mediaType)+ spine(阅读顺序) const opfEntry = zip.file(rootfilePath) ?? zip.file(decodeURIComponent(rootfilePath)); if (!opfEntry) throw new Error(`无法读取 epub:缺少 OPF 文件 ${rootfilePath}`); const opfDoc = new DOMParser().parseFromString(await opfEntry.async("text"), "application/xml"); const opfDir = rootfilePath.includes("/") ? rootfilePath.slice(0, rootfilePath.lastIndexOf("/") + 1) : ""; const manifest = new Map(); opfDoc.querySelectorAll("manifest > item").forEach((item) => { const id = item.getAttribute("id"); const href = item.getAttribute("href"); if (id && href) manifest.set(id, { href, mediaType: item.getAttribute("media-type") ?? "" }); }); const spineIds = Array.from(opfDoc.querySelectorAll("spine > itemref")) .map((el) => el.getAttribute("idref")) .filter((id): id is string => !!id); const td = new TurndownService({ headingStyle: "atx", codeBlockStyle: "fenced" }); const chapters: EpubChapter[] = []; for (const id of spineIds) { const item = manifest.get(id); // spine 里也会有图片/样式等资源项,只处理 HTML 文档 if (!item || !/xhtml|html/i.test(item.mediaType)) continue; const path = resolveZipPath(opfDir, item.href); const entry = zip.file(path) ?? zip.file(decodeURIComponent(path)); if (!entry) continue; const doc = new DOMParser().parseFromString(await entry.async("text"), "text/html"); doc.querySelectorAll("script, style").forEach((el) => el.remove()); // 内部相对路径图片导入后无法解析,剔除;外链 http(s) 图片保留 doc.querySelectorAll("img").forEach((img) => { if (!/^https?:\/\//i.test(img.getAttribute("src") ?? "")) img.remove(); }); const body = doc.body; if (!body) continue; const markdown = td .turndown(body.innerHTML) .replace(/\n{3,}/g, "\n\n") .trim(); if (!markdown) continue; // 封面页等纯图片页 const heading = Array.from(body.querySelectorAll("h1, h2, h3, h4, h5, h6")).find((h) => h.textContent?.trim()); const title = sanitizeTitle(heading?.textContent ?? doc.title ?? "") || `第 ${chapters.length + 1} 章`; chapters.push({ title, markdown }); } if (chapters.length === 0) throw new Error("未能从 epub 中解析出任何章节"); return chapters; }