fix(reader): split large txt into chapters (章 as boundaries, 卷 as optgroup labels, pseudo-sections when unmarked) rendered one at a time so the browser stops freezing; decode via strict-UTF-8-then-GB18030 with UTF-16 BOM sniffing to kill mojibake; progress locator {ch,scrollFraction} with old global-fraction fallback

This commit is contained in:
2026-09-07 22:21:29 +08:00
parent 991de0ef26
commit f47f2a6199
6 changed files with 318 additions and 31 deletions
+7
View File
@@ -8,6 +8,13 @@ The format loosely follows Keep a Changelog and can be adapted to the team's hab
## [Unreleased] ## [Unreleased]
### Fixed / 修复
- .txt/.md files encoded in GBK/GB2312/GB18030 or UTF-16 (typical web-novel exports) no longer render as mojibake: the reader strictly tries UTF-8 first and falls back to GB18030, recognizing UTF-16 by BOM.
- GBK/GB2312/GB18030 或 UTF-16 编码的 .txt/.md(网文导出常见)不再乱码:阅读器先按 UTF-8 严格解码,失败自动回退 GB18030,UTF-16 靠 BOM 识别。
- Reading a large .txt no longer freezes the browser: the text is split into chapters (「第X章」/Chapter X/序章… markers, pseudo-sections for unmarked files) and rendered one chapter at a time with a chapter selector and prev/next navigation; 「第X卷/部」 are volume groupings (dropdown optgroups), not chapter breaks, and books with only volumes split by volume; reading progress restores to the saved chapter.
- 阅读大 txt 不再卡死浏览器:正文按章节切分(识别「第X章」/Chapter X/序章等标题,无标记的按段落切成小节),每次只渲染一章,并提供章节目录选择与上一章/下一章导航;「第X卷/部」按卷分组(下拉框分组标签)而非章节边界,只有卷没有章的书按卷切分;阅读进度会恢复到上次的章节。
### Changed / 变更 ### Changed / 变更
- Creating a library on the admin Libraries page no longer asks for a server-side absolute path; just a name (the directory lands at `BOOKS_DIR/<name>` automatically). - Creating a library on the admin Libraries page no longer asks for a server-side absolute path; just a name (the directory lands at `BOOKS_DIR/<name>` automatically).
+65
View File
@@ -0,0 +1,65 @@
// txt 章节切分:优先识别「第X章」类标题行;「第X卷/部」是层级不是章节边界,
// 只做分组标签;整本没有章标记时才退而用卷作边界;再没有就按段落边界切
// 伪章节,保证任何单块都能在浏览器里流畅渲染。纯逻辑,无 DOM,vitest 覆盖。
export interface TxtChapter {
title: string;
start: number; // 正文中的起始字符偏移(含标题行本身)
vol?: string; // 所属卷标题(仅用于导航分组)
}
const NUM = "[0-9〇零一二三四五六七八九十百千万两]+";
const CH_RE = new RegExp(
`^[ \\t\\u3000]*(?:第\\s*${NUM}\\s*[章节回篇话]|Chapter\\s+${NUM}|序[章言]|楔子|引子|尾声|终章|后记|番外)[^\\n]{0,60}[ \\t]*$`,
"gim",
);
const VOL_RE = new RegExp(`^[ \\t\\u3000]*第\\s*${NUM}\\s*[卷部][^\\n]{0,60}[ \\t]*$`, "gm");
const MIN_GAP = 200; // 距上一保留标题不足此数视为目录行/正文提及,丢弃
const CHUNK = 20_000; // 伪章节目标字数
function markers(text: string, re: RegExp): TxtChapter[] {
const out: TxtChapter[] = [];
for (const m of text.matchAll(re)) {
const i = m.index ?? 0;
if (out.length && i - out[out.length - 1].start < MIN_GAP) continue;
out.push({ title: m[0].trim().slice(0, 40), start: i });
}
return out;
}
export function splitChapters(text: string): TxtChapter[] {
const vols = markers(text, VOL_RE);
let chs = markers(text, CH_RE);
let volMarks = vols;
if (chs.length < 2 && vols.length > 1) {
chs = vols; // 只有卷的书:卷即章节
volMarks = [];
}
if (chs.length > 1) {
if (chs[0].start > MIN_GAP) chs.unshift({ title: "开篇", start: 0 });
for (const c of chs) {
let v: TxtChapter | undefined;
for (const m of volMarks) if (m.start < c.start) v = m;
else break;
if (!v) continue;
c.vol = v.title;
if (c.start - v.start < MIN_GAP) c.start = v.start; // 卷题行并入本章开头,不留在上一章尾部
}
return chs;
}
const out: TxtChapter[] = [];
for (let s = 0, i = 1; s < text.length; i++) {
let e = Math.min(s + CHUNK, text.length);
if (e < text.length) {
const brk = text.lastIndexOf("\n", e);
if (brk > s) e = brk + 1;
}
out.push({ title: `第 ${i} 节`, start: s });
s = e;
}
return out;
}
export function chapterText(text: string, chapters: TxtChapter[], i: number): string {
return text.slice(chapters[i].start, chapters[i + 1]?.start ?? text.length);
}
+12
View File
@@ -0,0 +1,12 @@
// 网络 txt 常是 GBK/GB2312 而服务端一律标 charset=utf-8:先严格试 UTF-8,
// 失败回退 GB18030(GBK 超集);UTF-16 靠 BOM 识别。纯逻辑,vitest 覆盖。
export function decodeText(buf: ArrayBuffer): string {
const b = new Uint8Array(buf);
if (b[0] === 0xff && b[1] === 0xfe) return new TextDecoder("utf-16le").decode(b);
if (b[0] === 0xfe && b[1] === 0xff) return new TextDecoder("utf-16be").decode(b);
try {
return new TextDecoder("utf-8", { fatal: true }).decode(b);
} catch {
return new TextDecoder("gb18030").decode(b);
}
}
+152 -30
View File
@@ -2,34 +2,57 @@ import DOMPurify from "dompurify";
import { marked } from "marked"; import { marked } from "marked";
import { useEffect, useMemo, useRef, useState } from "react"; import { useEffect, useMemo, useRef, useState } from "react";
import { apiRaw } from "../api/client"; import { apiRaw } from "../api/client";
import { btn } from "../components/ui"; import { btn, input } from "../components/ui";
import { chapterText, splitChapters } from "../lib/chapters";
import { decodeText } from "../lib/encoding";
import { useProgressSaver, type ReaderProps } from "../lib/useProgress"; import { useProgressSaver, type ReaderProps } from "../lib/useProgress";
export default function TextReader({ book, initialLocator }: ReaderProps) { function useFileText(url: string) {
const boxRef = useRef<HTMLDivElement>(null);
const saver = useProgressSaver(book.id);
const restored = useRef(false);
const [text, setText] = useState<string | null>(null); const [text, setText] = useState<string | null>(null);
const [err, setErr] = useState(""); const [err, setErr] = useState("");
const [nonce, setNonce] = useState(0); const [nonce, setNonce] = useState(0);
useEffect(() => { useEffect(() => {
let dead = false; let dead = false;
setText(null); setText(null);
setErr(""); setErr("");
apiRaw(book.file_url ?? "") apiRaw(url)
.then((r) => r.text()) .then((r) => r.arrayBuffer())
.then((t) => !dead && setText(t)) .then((b) => !dead && setText(decodeText(b)))
.catch((e) => !dead && setErr(e instanceof Error ? e.message : "读取失败")); .catch((e) => !dead && setErr(e instanceof Error ? e.message : "读取失败"));
return () => { return () => {
dead = true; dead = true;
}; };
}, [book.file_url, nonce]); }, [url, nonce]);
return { text, err, retry: () => setNonce((n) => n + 1) };
}
const html = useMemo(() => { function LoadState({ text, onRetry }: { text: string; onRetry?: () => void }) {
if (text == null || book.format !== "md") return null; return (
return DOMPurify.sanitize(marked.parse(text, { async: false })); <div className="grid h-full place-items-center p-8 text-center">
}, [text, book.format]); <div>
{onRetry && <p className="mb-3 text-red-400">{text}</p>}
{!onRetry && <p className="text-zinc-500">{text}</p>}
{onRetry && (
<button className={btn} onClick={onRetry}>
重试
</button>
)}
</div>
</div>
);
}
export default function TextReader(props: ReaderProps) {
return props.book.format === "md" ? <MdView {...props} /> : <TxtView {...props} />;
}
function MdView({ book, initialLocator }: ReaderProps) {
const boxRef = useRef<HTMLDivElement>(null);
const saver = useProgressSaver(book.id);
const restored = useRef(false);
const { text, err, retry } = useFileText(book.file_url ?? "");
const html = useMemo(() => (text == null ? null : DOMPurify.sanitize(marked.parse(text, { async: false }))), [text]);
useEffect(() => { useEffect(() => {
if (restored.current || text == null || !boxRef.current) return; if (restored.current || text == null || !boxRef.current) return;
@@ -49,25 +72,124 @@ export default function TextReader({ book, initialLocator }: ReaderProps) {
saver.report({ scrollFraction: frac }, frac); saver.report({ scrollFraction: frac }, frac);
} }
if (err) if (err) return <LoadState text={err} onRetry={retry} />;
return ( if (text == null) return <LoadState text="正文加载中…" />;
<div className="grid h-full place-items-center p-8 text-center">
<div>
<p className="mb-3 text-red-400">{err}</p>
<button className={btn} onClick={() => setNonce((n) => n + 1)}>
重试
</button>
</div>
</div>
);
if (text == null) return <div className="grid h-full place-items-center text-zinc-500">正文加载中…</div>;
return ( return (
<div ref={boxRef} onScroll={onScroll} className="h-full overflow-y-auto p-6"> <div ref={boxRef} onScroll={onScroll} className="h-full overflow-y-auto p-6">
{html !== null ? ( <article className="md-body mx-auto max-w-2xl text-[15px] leading-7" dangerouslySetInnerHTML={{ __html: html ?? "" }} />
<article className="md-body mx-auto max-w-2xl text-[15px] leading-7" dangerouslySetInnerHTML={{ __html: html }} /> </div>
);
}
// 整本塞一个 <pre> 会冻死浏览器:按章渲染,一章一个滚动块。
function TxtView({ book, initialLocator }: ReaderProps) {
const boxRef = useRef<HTMLDivElement>(null);
const saver = useProgressSaver(book.id);
const inited = useRef(false);
const restoreFrac = useRef<number | null>(null);
const targetCi = useRef(0);
const [ci, setCi] = useState(0);
const { text, err, retry } = useFileText(book.file_url ?? "");
const chapters = useMemo(() => (text == null ? null : splitChapters(text)), [text]);
useEffect(() => {
if (inited.current || !chapters) return;
inited.current = true;
const f = Number(initialLocator?.scrollFraction);
const hasF = Number.isFinite(f) && f > 0 && f <= 1;
const rawCh = Number(initialLocator?.ch);
let idx = 0;
if (Number.isFinite(rawCh) && rawCh >= 0 && rawCh < chapters.length) {
idx = Math.floor(rawCh);
if (hasF) restoreFrac.current = f;
} else if (hasF) {
const p = f * chapters.length; // 旧版全局分数 → 章节近似定位
idx = Math.min(chapters.length - 1, Math.floor(p));
restoreFrac.current = p - idx;
}
if (idx > 0 || restoreFrac.current != null) {
targetCi.current = idx;
setCi(idx);
}
}, [chapters, initialLocator]);
useEffect(() => {
const el = boxRef.current;
if (!el || !chapters || ci !== targetCi.current) return;
if (restoreFrac.current != null) {
el.scrollTop = restoreFrac.current * (el.scrollHeight - el.clientHeight);
restoreFrac.current = null;
} else el.scrollTop = 0;
}, [ci, chapters]);
function goChapter(i: number) {
if (!chapters) return;
restoreFrac.current = null;
setCi(i);
saver.report({ ch: i, scrollFraction: 0 }, i / chapters.length);
}
function onScroll() {
const el = boxRef.current;
if (!el || !chapters) return;
const max = el.scrollHeight - el.clientHeight;
const frac = max > 0 ? Math.min(1, Math.max(0, el.scrollTop / max)) : 0;
saver.report({ ch: ci, scrollFraction: frac }, (ci + frac) / chapters.length);
}
if (err) return <LoadState text={err} onRetry={retry} />;
if (text == null || !chapters) return <LoadState text="正文加载中…" />;
return (
<div className="flex h-full flex-col">
{chapters.length > 1 && (
<div className="flex shrink-0 items-center gap-2 border-b border-zinc-800 px-4 py-2 text-sm">
<select className={input + " max-w-64"} value={ci} onChange={(e) => goChapter(Number(e.target.value))}>
{(() => {
const groups: { vol: string | undefined; items: number[] }[] = [];
chapters.forEach((c, i) => {
const last = groups[groups.length - 1];
if (!last || last.vol !== c.vol) groups.push({ vol: c.vol, items: [i] });
else last.items.push(i);
});
const opt = (i: number) => (
<option key={i} value={i}>
{i + 1}. {chapters[i].title}
</option>
);
return groups.length === 1 && !groups[0].vol
? chapters.map((_, i) => opt(i))
: groups.map((g, gi) =>
g.vol ? (
<optgroup key={gi} label={g.vol}>
{g.items.map(opt)}
</optgroup>
) : ( ) : (
<pre className="mx-auto max-w-2xl font-sans text-[15px] leading-7 break-words whitespace-pre-wrap">{text}</pre> g.items.map(opt)
),
);
})()}
</select>
<span className="ml-auto text-zinc-500">
{ci + 1}/{chapters.length}
</span>
</div>
)}
<div ref={boxRef} onScroll={onScroll} className="min-h-0 flex-1 overflow-y-auto p-6">
<pre className="mx-auto max-w-2xl font-sans text-[15px] leading-7 break-words whitespace-pre-wrap">
{chapterText(text, chapters, ci)}
</pre>
</div>
{chapters.length > 1 && (
<div className="flex shrink-0 items-center justify-center gap-3 border-t border-zinc-800 px-4 py-2 text-sm">
<button className={btn} disabled={ci === 0} onClick={() => goChapter(ci - 1)}>
上一章
</button>
<button className={btn} disabled={ci === chapters.length - 1} onClick={() => goChapter(ci + 1)}>
下一章
</button>
</div>
)} )}
</div> </div>
); );
+58
View File
@@ -0,0 +1,58 @@
import { expect, it } from "vitest";
import { chapterText, splitChapters } from "../src/lib/chapters";
const pad = (n: number) => "字".repeat(n);
it("识别中文/英文章节标题,首章前的引子归入「开篇」", () => {
const t = `${pad(300)}\n简介正文\n\n第一章 初入江湖\n${pad(500)}\n\n第二章 风起云涌\n${pad(500)}\n\nChapter 3 the end\n${pad(100)}`;
const chs = splitChapters(t);
expect(chs.map((c) => c.title)).toEqual(["开篇", "第一章 初入江湖", "第二章 风起云涌", "Chapter 3 the end"]);
expect(chs[1].start).toBe(t.indexOf("第一章"));
expect(chapterText(t, chs, 1)).toMatch(/^第一章 初入江湖/);
expect(chapterText(t, chs, chs.length - 1).trim().length).toBeLessThanOrEqual(200 + 15);
});
it("密集目录行(间隔 < MIN_GAP)被折叠,正文里的真章节仍可导航", () => {
const toc = ["第一章 aaa", "第二章 bbb", "第三章 ccc"].map((s) => s + "\n").join("");
const t = toc + "\n" + pad(500) + "\n第三章 ccc 正式开讲\n" + pad(500) + "\n第四章 ddd\n" + pad(500);
const chs = splitChapters(t);
expect(chs.map((c) => c.title)).toEqual(["第一章 aaa", "第三章 ccc 正式开讲", "第四章 ddd"]);
});
it("无章节标记的长文按段落边界切伪章节,覆盖全文不重叠", () => {
const para = pad(2500) + "\n\n";
const t = para.repeat(40); // 10 万字
const chs = splitChapters(t);
expect(chs.length).toBeGreaterThanOrEqual(5);
expect(chs[0].title).toBe("第 1 节");
expect(chs[0].start).toBe(0);
for (let i = 0; i < chs.length; i++) expect(chs[i].start).toBeLessThan(chs[i + 1]?.start ?? t.length);
if (chs.length > 1) expect(t[chs[1].start - 1]).toBe("\n"); // 落点在换行后,不切断一行
const total = chs.reduce((n, c, i) => n + (chs[i + 1]?.start ?? t.length) - c.start, 0);
expect(total).toBe(t.length);
});
it("短文无标记 → 单节,chapterText 即全文", () => {
const t = "hello\nworld";
const chs = splitChapters(t);
expect(chs).toEqual([{ title: "第 1 节", start: 0 }]);
expect(chapterText(t, chs, 0)).toBe(t);
});
it("卷不是章节边界:只按章切,卷题作分组标签并并入所属章开头", () => {
const t =
"第一卷 天云世界\n第一章 铜镜\n" + pad(500) + "\n第二章 修炼\n" + pad(500) +
"\n\n\n第二卷 仙界\n第三章 飞升\n" + pad(300);
const chs = splitChapters(t);
expect(chs.map((c) => c.title)).toEqual(["第一章 铜镜", "第二章 修炼", "第三章 飞升"]);
expect(chs.map((c) => c.vol)).toEqual(["第一卷 天云世界", "第一卷 天云世界", "第二卷 仙界"]);
expect(chs[2].start).toBe(t.indexOf("第二卷 仙界")); // 卷题行从本章开头渲染,而非上一章结尾
expect(chapterText(t, chs, 2)).toMatch(/^第二卷 仙界/);
});
it("只有卷没有章的书:卷即章节", () => {
const t = "第一卷 起\n" + pad(600) + "\n第二卷 承\n" + pad(600) + "\n第三卷 转\n" + pad(600);
const chs = splitChapters(t);
expect(chs.map((c) => c.title)).toEqual(["第一卷 起", "第二卷 承", "第三卷 转"]);
expect(chs.every((c) => c.vol === undefined)).toBe(true);
});
+23
View File
@@ -0,0 +1,23 @@
import { expect, it } from "vitest";
import { decodeText } from "../src/lib/encoding";
const u8 = (...rows: number[][]) => new Uint8Array(rows.flat());
const ab = (v: Uint8Array) => v.slice().buffer;
it("GBK 编码的中文回退 GB18030 正确解码", () => {
expect(decodeText(ab(u8([0xd6, 0xd0], [0xce, 0xc4])))).toBe("中文");
});
it("合法 UTF-8 不被 GBK 路径污染;UTF-8 BOM 被剥掉", () => {
expect(decodeText(ab(u8([0xe4, 0xb8, 0xad], [0xe6, 0x96, 0x87])))).toBe("中文");
expect(decodeText(ab(u8([0xef, 0xbb, 0xbf], [0xe4, 0xb8, 0xad])))).toBe("中");
});
it("UTF-16 靠 BOM 识别", () => {
expect(decodeText(ab(u8([0xff, 0xfe], [0x2d, 0x4e])))).toBe("中");
expect(decodeText(ab(u8([0xfe, 0xff], [0x4e, 0x2d])))).toBe("中");
});
it("纯 ASCII 原样通过", () => {
expect(decodeText(ab(u8([...new TextEncoder().encode("hello\nworld")])))).toBe("hello\nworld");
});