fix(reader): split large txt into chapters (章 as boundaries, 卷 as optgroup labels, pseudo-sections when unmarked) rendered one at a time so the browser stops freezing; decode via strict-UTF-8-then-GB18030 with UTF-16 BOM sniffing to kill mojibake; progress locator {ch,scrollFraction} with old global-fraction fallback

This commit is contained in:
2026-09-07 22:21:29 +08:00
parent 991de0ef26
commit f47f2a6199
6 changed files with 318 additions and 31 deletions
+58
View File
@@ -0,0 +1,58 @@
import { expect, it } from "vitest";
import { chapterText, splitChapters } from "../src/lib/chapters";
const pad = (n: number) => "字".repeat(n);
it("识别中文/英文章节标题,首章前的引子归入「开篇」", () => {
const t = `${pad(300)}\n简介正文\n\n第一章 初入江湖\n${pad(500)}\n\n第二章 风起云涌\n${pad(500)}\n\nChapter 3 the end\n${pad(100)}`;
const chs = splitChapters(t);
expect(chs.map((c) => c.title)).toEqual(["开篇", "第一章 初入江湖", "第二章 风起云涌", "Chapter 3 the end"]);
expect(chs[1].start).toBe(t.indexOf("第一章"));
expect(chapterText(t, chs, 1)).toMatch(/^第一章 初入江湖/);
expect(chapterText(t, chs, chs.length - 1).trim().length).toBeLessThanOrEqual(200 + 15);
});
it("密集目录行(间隔 < MIN_GAP)被折叠,正文里的真章节仍可导航", () => {
const toc = ["第一章 aaa", "第二章 bbb", "第三章 ccc"].map((s) => s + "\n").join("");
const t = toc + "\n" + pad(500) + "\n第三章 ccc 正式开讲\n" + pad(500) + "\n第四章 ddd\n" + pad(500);
const chs = splitChapters(t);
expect(chs.map((c) => c.title)).toEqual(["第一章 aaa", "第三章 ccc 正式开讲", "第四章 ddd"]);
});
it("无章节标记的长文按段落边界切伪章节,覆盖全文不重叠", () => {
const para = pad(2500) + "\n\n";
const t = para.repeat(40); // 10 万字
const chs = splitChapters(t);
expect(chs.length).toBeGreaterThanOrEqual(5);
expect(chs[0].title).toBe("第 1 节");
expect(chs[0].start).toBe(0);
for (let i = 0; i < chs.length; i++) expect(chs[i].start).toBeLessThan(chs[i + 1]?.start ?? t.length);
if (chs.length > 1) expect(t[chs[1].start - 1]).toBe("\n"); // 落点在换行后,不切断一行
const total = chs.reduce((n, c, i) => n + (chs[i + 1]?.start ?? t.length) - c.start, 0);
expect(total).toBe(t.length);
});
it("短文无标记 → 单节,chapterText 即全文", () => {
const t = "hello\nworld";
const chs = splitChapters(t);
expect(chs).toEqual([{ title: "第 1 节", start: 0 }]);
expect(chapterText(t, chs, 0)).toBe(t);
});
it("卷不是章节边界:只按章切,卷题作分组标签并并入所属章开头", () => {
const t =
"第一卷 天云世界\n第一章 铜镜\n" + pad(500) + "\n第二章 修炼\n" + pad(500) +
"\n\n\n第二卷 仙界\n第三章 飞升\n" + pad(300);
const chs = splitChapters(t);
expect(chs.map((c) => c.title)).toEqual(["第一章 铜镜", "第二章 修炼", "第三章 飞升"]);
expect(chs.map((c) => c.vol)).toEqual(["第一卷 天云世界", "第一卷 天云世界", "第二卷 仙界"]);
expect(chs[2].start).toBe(t.indexOf("第二卷 仙界")); // 卷题行从本章开头渲染,而非上一章结尾
expect(chapterText(t, chs, 2)).toMatch(/^第二卷 仙界/);
});
it("只有卷没有章的书:卷即章节", () => {
const t = "第一卷 起\n" + pad(600) + "\n第二卷 承\n" + pad(600) + "\n第三卷 转\n" + pad(600);
const chs = splitChapters(t);
expect(chs.map((c) => c.title)).toEqual(["第一卷 起", "第二卷 承", "第三卷 转"]);
expect(chs.every((c) => c.vol === undefined)).toBe(true);
});
+23
View File
@@ -0,0 +1,23 @@
import { expect, it } from "vitest";
import { decodeText } from "../src/lib/encoding";
const u8 = (...rows: number[][]) => new Uint8Array(rows.flat());
const ab = (v: Uint8Array) => v.slice().buffer;
it("GBK 编码的中文回退 GB18030 正确解码", () => {
expect(decodeText(ab(u8([0xd6, 0xd0], [0xce, 0xc4])))).toBe("中文");
});
it("合法 UTF-8 不被 GBK 路径污染;UTF-8 BOM 被剥掉", () => {
expect(decodeText(ab(u8([0xe4, 0xb8, 0xad], [0xe6, 0x96, 0x87])))).toBe("中文");
expect(decodeText(ab(u8([0xef, 0xbb, 0xbf], [0xe4, 0xb8, 0xad])))).toBe("中");
});
it("UTF-16 靠 BOM 识别", () => {
expect(decodeText(ab(u8([0xff, 0xfe], [0x2d, 0x4e])))).toBe("中");
expect(decodeText(ab(u8([0xfe, 0xff], [0x4e, 0x2d])))).toBe("中");
});
it("纯 ASCII 原样通过", () => {
expect(decodeText(ab(u8([...new TextEncoder().encode("hello\nworld")])))).toBe("hello\nworld");
});