diff --git a/docs/CHANGELOG_web.md b/docs/CHANGELOG_web.md index 726d590..f64fe15 100644 --- a/docs/CHANGELOG_web.md +++ b/docs/CHANGELOG_web.md @@ -8,6 +8,13 @@ The format loosely follows Keep a Changelog and can be adapted to the team's hab ## [Unreleased] +### Fixed / 修复 + +- .txt/.md files encoded in GBK/GB2312/GB18030 or UTF-16 (typical web-novel exports) no longer render as mojibake: the reader strictly tries UTF-8 first and falls back to GB18030, recognizing UTF-16 by BOM. +- GBK/GB2312/GB18030 或 UTF-16 编码的 .txt/.md(网文导出常见)不再乱码:阅读器先按 UTF-8 严格解码,失败自动回退 GB18030,UTF-16 靠 BOM 识别。 +- Reading a large .txt no longer freezes the browser: the text is split into chapters (「第X章」/Chapter X/序章… markers, pseudo-sections for unmarked files) and rendered one chapter at a time with a chapter selector and prev/next navigation; 「第X卷/部」 are volume groupings (dropdown optgroups), not chapter breaks, and books with only volumes split by volume; reading progress restores to the saved chapter. +- 阅读大 txt 不再卡死浏览器:正文按章节切分(识别「第X章」/Chapter X/序章等标题,无标记的按段落切成小节),每次只渲染一章,并提供章节目录选择与上一章/下一章导航;「第X卷/部」按卷分组(下拉框分组标签)而非章节边界,只有卷没有章的书按卷切分;阅读进度会恢复到上次的章节。 + ### Changed / 变更 - Creating a library on the admin Libraries page no longer asks for a server-side absolute path; just a name (the directory lands at `BOOKS_DIR/` automatically). diff --git a/frontend/src/lib/chapters.ts b/frontend/src/lib/chapters.ts new file mode 100644 index 0000000..146a539 --- /dev/null +++ b/frontend/src/lib/chapters.ts @@ -0,0 +1,65 @@ +// txt 章节切分:优先识别「第X章」类标题行;「第X卷/部」是层级不是章节边界, +// 只做分组标签;整本没有章标记时才退而用卷作边界;再没有就按段落边界切 +// 伪章节,保证任何单块都能在浏览器里流畅渲染。纯逻辑,无 DOM,vitest 覆盖。 +export interface TxtChapter { + title: string; + start: number; // 正文中的起始字符偏移(含标题行本身) + vol?: string; // 所属卷标题(仅用于导航分组) +} + +const NUM = "[0-9〇零一二三四五六七八九十百千万两]+"; +const CH_RE = new RegExp( + `^[ \\t\\u3000]*(?:第\\s*${NUM}\\s*[章节回篇话]|Chapter\\s+${NUM}|序[章言]|楔子|引子|尾声|终章|后记|番外)[^\\n]{0,60}[ \\t]*$`, + "gim", +); +const VOL_RE = new RegExp(`^[ \\t\\u3000]*第\\s*${NUM}\\s*[卷部][^\\n]{0,60}[ \\t]*$`, "gm"); + +const MIN_GAP = 200; // 距上一保留标题不足此数视为目录行/正文提及,丢弃 +const CHUNK = 20_000; // 伪章节目标字数 + +function markers(text: string, re: RegExp): TxtChapter[] { + const out: TxtChapter[] = []; + for (const m of text.matchAll(re)) { + const i = m.index ?? 0; + if (out.length && i - out[out.length - 1].start < MIN_GAP) continue; + out.push({ title: m[0].trim().slice(0, 40), start: i }); + } + return out; +} + +export function splitChapters(text: string): TxtChapter[] { + const vols = markers(text, VOL_RE); + let chs = markers(text, CH_RE); + let volMarks = vols; + if (chs.length < 2 && vols.length > 1) { + chs = vols; // 只有卷的书:卷即章节 + volMarks = []; + } + if (chs.length > 1) { + if (chs[0].start > MIN_GAP) chs.unshift({ title: "开篇", start: 0 }); + for (const c of chs) { + let v: TxtChapter | undefined; + for (const m of volMarks) if (m.start < c.start) v = m; + else break; + if (!v) continue; + c.vol = v.title; + if (c.start - v.start < MIN_GAP) c.start = v.start; // 卷题行并入本章开头,不留在上一章尾部 + } + return chs; + } + const out: TxtChapter[] = []; + for (let s = 0, i = 1; s < text.length; i++) { + let e = Math.min(s + CHUNK, text.length); + if (e < text.length) { + const brk = text.lastIndexOf("\n", e); + if (brk > s) e = brk + 1; + } + out.push({ title: `第 ${i} 节`, start: s }); + s = e; + } + return out; +} + +export function chapterText(text: string, chapters: TxtChapter[], i: number): string { + return text.slice(chapters[i].start, chapters[i + 1]?.start ?? text.length); +} diff --git a/frontend/src/lib/encoding.ts b/frontend/src/lib/encoding.ts new file mode 100644 index 0000000..38849d4 --- /dev/null +++ b/frontend/src/lib/encoding.ts @@ -0,0 +1,12 @@ +// 网络 txt 常是 GBK/GB2312 而服务端一律标 charset=utf-8:先严格试 UTF-8, +// 失败回退 GB18030(GBK 超集);UTF-16 靠 BOM 识别。纯逻辑,vitest 覆盖。 +export function decodeText(buf: ArrayBuffer): string { + const b = new Uint8Array(buf); + if (b[0] === 0xff && b[1] === 0xfe) return new TextDecoder("utf-16le").decode(b); + if (b[0] === 0xfe && b[1] === 0xff) return new TextDecoder("utf-16be").decode(b); + try { + return new TextDecoder("utf-8", { fatal: true }).decode(b); + } catch { + return new TextDecoder("gb18030").decode(b); + } +} diff --git a/frontend/src/readers/TextReader.tsx b/frontend/src/readers/TextReader.tsx index aa90433..85506ba 100644 --- a/frontend/src/readers/TextReader.tsx +++ b/frontend/src/readers/TextReader.tsx @@ -2,34 +2,57 @@ import DOMPurify from "dompurify"; import { marked } from "marked"; import { useEffect, useMemo, useRef, useState } from "react"; import { apiRaw } from "../api/client"; -import { btn } from "../components/ui"; +import { btn, input } from "../components/ui"; +import { chapterText, splitChapters } from "../lib/chapters"; +import { decodeText } from "../lib/encoding"; import { useProgressSaver, type ReaderProps } from "../lib/useProgress"; -export default function TextReader({ book, initialLocator }: ReaderProps) { - const boxRef = useRef(null); - const saver = useProgressSaver(book.id); - const restored = useRef(false); +function useFileText(url: string) { const [text, setText] = useState(null); const [err, setErr] = useState(""); const [nonce, setNonce] = useState(0); - useEffect(() => { let dead = false; setText(null); setErr(""); - apiRaw(book.file_url ?? "") - .then((r) => r.text()) - .then((t) => !dead && setText(t)) + apiRaw(url) + .then((r) => r.arrayBuffer()) + .then((b) => !dead && setText(decodeText(b))) .catch((e) => !dead && setErr(e instanceof Error ? e.message : "读取失败")); return () => { dead = true; }; - }, [book.file_url, nonce]); + }, [url, nonce]); + return { text, err, retry: () => setNonce((n) => n + 1) }; +} - const html = useMemo(() => { - if (text == null || book.format !== "md") return null; - return DOMPurify.sanitize(marked.parse(text, { async: false })); - }, [text, book.format]); +function LoadState({ text, onRetry }: { text: string; onRetry?: () => void }) { + return ( +
+
+ {onRetry &&

{text}

} + {!onRetry &&

{text}

} + {onRetry && ( + + )} +
+
+ ); +} + +export default function TextReader(props: ReaderProps) { + return props.book.format === "md" ? : ; +} + +function MdView({ book, initialLocator }: ReaderProps) { + const boxRef = useRef(null); + const saver = useProgressSaver(book.id); + const restored = useRef(false); + const { text, err, retry } = useFileText(book.file_url ?? ""); + + const html = useMemo(() => (text == null ? null : DOMPurify.sanitize(marked.parse(text, { async: false }))), [text]); useEffect(() => { if (restored.current || text == null || !boxRef.current) return; @@ -49,25 +72,124 @@ export default function TextReader({ book, initialLocator }: ReaderProps) { saver.report({ scrollFraction: frac }, frac); } - if (err) - return ( -
-
-

{err}

- -
-
- ); - if (text == null) return
正文加载中…
; - + if (err) return ; + if (text == null) return ; return (
- {html !== null ? ( -
- ) : ( -
{text}
+
+
+ ); +} + +// 整本塞一个
 会冻死浏览器:按章渲染,一章一个滚动块。
+function TxtView({ book, initialLocator }: ReaderProps) {
+  const boxRef = useRef(null);
+  const saver = useProgressSaver(book.id);
+  const inited = useRef(false);
+  const restoreFrac = useRef(null);
+  const targetCi = useRef(0);
+  const [ci, setCi] = useState(0);
+  const { text, err, retry } = useFileText(book.file_url ?? "");
+
+  const chapters = useMemo(() => (text == null ? null : splitChapters(text)), [text]);
+
+  useEffect(() => {
+    if (inited.current || !chapters) return;
+    inited.current = true;
+    const f = Number(initialLocator?.scrollFraction);
+    const hasF = Number.isFinite(f) && f > 0 && f <= 1;
+    const rawCh = Number(initialLocator?.ch);
+    let idx = 0;
+    if (Number.isFinite(rawCh) && rawCh >= 0 && rawCh < chapters.length) {
+      idx = Math.floor(rawCh);
+      if (hasF) restoreFrac.current = f;
+    } else if (hasF) {
+      const p = f * chapters.length; // 旧版全局分数 → 章节近似定位
+      idx = Math.min(chapters.length - 1, Math.floor(p));
+      restoreFrac.current = p - idx;
+    }
+    if (idx > 0 || restoreFrac.current != null) {
+      targetCi.current = idx;
+      setCi(idx);
+    }
+  }, [chapters, initialLocator]);
+
+  useEffect(() => {
+    const el = boxRef.current;
+    if (!el || !chapters || ci !== targetCi.current) return;
+    if (restoreFrac.current != null) {
+      el.scrollTop = restoreFrac.current * (el.scrollHeight - el.clientHeight);
+      restoreFrac.current = null;
+    } else el.scrollTop = 0;
+  }, [ci, chapters]);
+
+  function goChapter(i: number) {
+    if (!chapters) return;
+    restoreFrac.current = null;
+    setCi(i);
+    saver.report({ ch: i, scrollFraction: 0 }, i / chapters.length);
+  }
+
+  function onScroll() {
+    const el = boxRef.current;
+    if (!el || !chapters) return;
+    const max = el.scrollHeight - el.clientHeight;
+    const frac = max > 0 ? Math.min(1, Math.max(0, el.scrollTop / max)) : 0;
+    saver.report({ ch: ci, scrollFraction: frac }, (ci + frac) / chapters.length);
+  }
+
+  if (err) return ;
+  if (text == null || !chapters) return ;
+
+  return (
+    
+ {chapters.length > 1 && ( +
+ + + {ci + 1}/{chapters.length} + +
+ )} +
+
+          {chapterText(text, chapters, ci)}
+        
+
+ {chapters.length > 1 && ( +
+ + +
)}
); diff --git a/frontend/test/chapters.test.ts b/frontend/test/chapters.test.ts new file mode 100644 index 0000000..d4ef59b --- /dev/null +++ b/frontend/test/chapters.test.ts @@ -0,0 +1,58 @@ +import { expect, it } from "vitest"; +import { chapterText, splitChapters } from "../src/lib/chapters"; + +const pad = (n: number) => "字".repeat(n); + +it("识别中文/英文章节标题,首章前的引子归入「开篇」", () => { + const t = `${pad(300)}\n简介正文\n\n第一章 初入江湖\n${pad(500)}\n\n第二章 风起云涌\n${pad(500)}\n\nChapter 3 the end\n${pad(100)}`; + const chs = splitChapters(t); + expect(chs.map((c) => c.title)).toEqual(["开篇", "第一章 初入江湖", "第二章 风起云涌", "Chapter 3 the end"]); + expect(chs[1].start).toBe(t.indexOf("第一章")); + expect(chapterText(t, chs, 1)).toMatch(/^第一章 初入江湖/); + expect(chapterText(t, chs, chs.length - 1).trim().length).toBeLessThanOrEqual(200 + 15); +}); + +it("密集目录行(间隔 < MIN_GAP)被折叠,正文里的真章节仍可导航", () => { + const toc = ["第一章 aaa", "第二章 bbb", "第三章 ccc"].map((s) => s + "\n").join(""); + const t = toc + "\n" + pad(500) + "\n第三章 ccc 正式开讲\n" + pad(500) + "\n第四章 ddd\n" + pad(500); + const chs = splitChapters(t); + expect(chs.map((c) => c.title)).toEqual(["第一章 aaa", "第三章 ccc 正式开讲", "第四章 ddd"]); +}); + +it("无章节标记的长文按段落边界切伪章节,覆盖全文不重叠", () => { + const para = pad(2500) + "\n\n"; + const t = para.repeat(40); // 10 万字 + const chs = splitChapters(t); + expect(chs.length).toBeGreaterThanOrEqual(5); + expect(chs[0].title).toBe("第 1 节"); + expect(chs[0].start).toBe(0); + for (let i = 0; i < chs.length; i++) expect(chs[i].start).toBeLessThan(chs[i + 1]?.start ?? t.length); + if (chs.length > 1) expect(t[chs[1].start - 1]).toBe("\n"); // 落点在换行后,不切断一行 + const total = chs.reduce((n, c, i) => n + (chs[i + 1]?.start ?? t.length) - c.start, 0); + expect(total).toBe(t.length); +}); + +it("短文无标记 → 单节,chapterText 即全文", () => { + const t = "hello\nworld"; + const chs = splitChapters(t); + expect(chs).toEqual([{ title: "第 1 节", start: 0 }]); + expect(chapterText(t, chs, 0)).toBe(t); +}); + +it("卷不是章节边界:只按章切,卷题作分组标签并并入所属章开头", () => { + const t = + "第一卷 天云世界\n第一章 铜镜\n" + pad(500) + "\n第二章 修炼\n" + pad(500) + + "\n\n\n第二卷 仙界\n第三章 飞升\n" + pad(300); + const chs = splitChapters(t); + expect(chs.map((c) => c.title)).toEqual(["第一章 铜镜", "第二章 修炼", "第三章 飞升"]); + expect(chs.map((c) => c.vol)).toEqual(["第一卷 天云世界", "第一卷 天云世界", "第二卷 仙界"]); + expect(chs[2].start).toBe(t.indexOf("第二卷 仙界")); // 卷题行从本章开头渲染,而非上一章结尾 + expect(chapterText(t, chs, 2)).toMatch(/^第二卷 仙界/); +}); + +it("只有卷没有章的书:卷即章节", () => { + const t = "第一卷 起\n" + pad(600) + "\n第二卷 承\n" + pad(600) + "\n第三卷 转\n" + pad(600); + const chs = splitChapters(t); + expect(chs.map((c) => c.title)).toEqual(["第一卷 起", "第二卷 承", "第三卷 转"]); + expect(chs.every((c) => c.vol === undefined)).toBe(true); +}); diff --git a/frontend/test/encoding.test.ts b/frontend/test/encoding.test.ts new file mode 100644 index 0000000..0853210 --- /dev/null +++ b/frontend/test/encoding.test.ts @@ -0,0 +1,23 @@ +import { expect, it } from "vitest"; +import { decodeText } from "../src/lib/encoding"; + +const u8 = (...rows: number[][]) => new Uint8Array(rows.flat()); +const ab = (v: Uint8Array) => v.slice().buffer; + +it("GBK 编码的中文回退 GB18030 正确解码", () => { + expect(decodeText(ab(u8([0xd6, 0xd0], [0xce, 0xc4])))).toBe("中文"); +}); + +it("合法 UTF-8 不被 GBK 路径污染;UTF-8 BOM 被剥掉", () => { + expect(decodeText(ab(u8([0xe4, 0xb8, 0xad], [0xe6, 0x96, 0x87])))).toBe("中文"); + expect(decodeText(ab(u8([0xef, 0xbb, 0xbf], [0xe4, 0xb8, 0xad])))).toBe("中"); +}); + +it("UTF-16 靠 BOM 识别", () => { + expect(decodeText(ab(u8([0xff, 0xfe], [0x2d, 0x4e])))).toBe("中"); + expect(decodeText(ab(u8([0xfe, 0xff], [0x4e, 0x2d])))).toBe("中"); +}); + +it("纯 ASCII 原样通过", () => { + expect(decodeText(ab(u8([...new TextEncoder().encode("hello\nworld")])))).toBe("hello\nworld"); +});