fix(reader): split large txt into chapters (章 as boundaries, 卷 as optgroup labels, pseudo-sections when unmarked) rendered one at a time so the browser stops freezing; decode via strict-UTF-8-then-GB18030 with UTF-16 BOM sniffing to kill mojibake; progress locator {ch,scrollFraction} with old global-fraction fallback
This commit is contained in:
@@ -0,0 +1,58 @@
|
||||
import { expect, it } from "vitest";
|
||||
import { chapterText, splitChapters } from "../src/lib/chapters";
|
||||
|
||||
const pad = (n: number) => "字".repeat(n);
|
||||
|
||||
it("识别中文/英文章节标题,首章前的引子归入「开篇」", () => {
|
||||
const t = `${pad(300)}\n简介正文\n\n第一章 初入江湖\n${pad(500)}\n\n第二章 风起云涌\n${pad(500)}\n\nChapter 3 the end\n${pad(100)}`;
|
||||
const chs = splitChapters(t);
|
||||
expect(chs.map((c) => c.title)).toEqual(["开篇", "第一章 初入江湖", "第二章 风起云涌", "Chapter 3 the end"]);
|
||||
expect(chs[1].start).toBe(t.indexOf("第一章"));
|
||||
expect(chapterText(t, chs, 1)).toMatch(/^第一章 初入江湖/);
|
||||
expect(chapterText(t, chs, chs.length - 1).trim().length).toBeLessThanOrEqual(200 + 15);
|
||||
});
|
||||
|
||||
it("密集目录行(间隔 < MIN_GAP)被折叠,正文里的真章节仍可导航", () => {
|
||||
const toc = ["第一章 aaa", "第二章 bbb", "第三章 ccc"].map((s) => s + "\n").join("");
|
||||
const t = toc + "\n" + pad(500) + "\n第三章 ccc 正式开讲\n" + pad(500) + "\n第四章 ddd\n" + pad(500);
|
||||
const chs = splitChapters(t);
|
||||
expect(chs.map((c) => c.title)).toEqual(["第一章 aaa", "第三章 ccc 正式开讲", "第四章 ddd"]);
|
||||
});
|
||||
|
||||
it("无章节标记的长文按段落边界切伪章节,覆盖全文不重叠", () => {
|
||||
const para = pad(2500) + "\n\n";
|
||||
const t = para.repeat(40); // 10 万字
|
||||
const chs = splitChapters(t);
|
||||
expect(chs.length).toBeGreaterThanOrEqual(5);
|
||||
expect(chs[0].title).toBe("第 1 节");
|
||||
expect(chs[0].start).toBe(0);
|
||||
for (let i = 0; i < chs.length; i++) expect(chs[i].start).toBeLessThan(chs[i + 1]?.start ?? t.length);
|
||||
if (chs.length > 1) expect(t[chs[1].start - 1]).toBe("\n"); // 落点在换行后,不切断一行
|
||||
const total = chs.reduce((n, c, i) => n + (chs[i + 1]?.start ?? t.length) - c.start, 0);
|
||||
expect(total).toBe(t.length);
|
||||
});
|
||||
|
||||
it("短文无标记 → 单节,chapterText 即全文", () => {
|
||||
const t = "hello\nworld";
|
||||
const chs = splitChapters(t);
|
||||
expect(chs).toEqual([{ title: "第 1 节", start: 0 }]);
|
||||
expect(chapterText(t, chs, 0)).toBe(t);
|
||||
});
|
||||
|
||||
it("卷不是章节边界:只按章切,卷题作分组标签并并入所属章开头", () => {
|
||||
const t =
|
||||
"第一卷 天云世界\n第一章 铜镜\n" + pad(500) + "\n第二章 修炼\n" + pad(500) +
|
||||
"\n\n\n第二卷 仙界\n第三章 飞升\n" + pad(300);
|
||||
const chs = splitChapters(t);
|
||||
expect(chs.map((c) => c.title)).toEqual(["第一章 铜镜", "第二章 修炼", "第三章 飞升"]);
|
||||
expect(chs.map((c) => c.vol)).toEqual(["第一卷 天云世界", "第一卷 天云世界", "第二卷 仙界"]);
|
||||
expect(chs[2].start).toBe(t.indexOf("第二卷 仙界")); // 卷题行从本章开头渲染,而非上一章结尾
|
||||
expect(chapterText(t, chs, 2)).toMatch(/^第二卷 仙界/);
|
||||
});
|
||||
|
||||
it("只有卷没有章的书:卷即章节", () => {
|
||||
const t = "第一卷 起\n" + pad(600) + "\n第二卷 承\n" + pad(600) + "\n第三卷 转\n" + pad(600);
|
||||
const chs = splitChapters(t);
|
||||
expect(chs.map((c) => c.title)).toEqual(["第一卷 起", "第二卷 承", "第三卷 转"]);
|
||||
expect(chs.every((c) => c.vol === undefined)).toBe(true);
|
||||
});
|
||||
@@ -0,0 +1,23 @@
|
||||
import { expect, it } from "vitest";
|
||||
import { decodeText } from "../src/lib/encoding";
|
||||
|
||||
const u8 = (...rows: number[][]) => new Uint8Array(rows.flat());
|
||||
const ab = (v: Uint8Array) => v.slice().buffer;
|
||||
|
||||
it("GBK 编码的中文回退 GB18030 正确解码", () => {
|
||||
expect(decodeText(ab(u8([0xd6, 0xd0], [0xce, 0xc4])))).toBe("中文");
|
||||
});
|
||||
|
||||
it("合法 UTF-8 不被 GBK 路径污染;UTF-8 BOM 被剥掉", () => {
|
||||
expect(decodeText(ab(u8([0xe4, 0xb8, 0xad], [0xe6, 0x96, 0x87])))).toBe("中文");
|
||||
expect(decodeText(ab(u8([0xef, 0xbb, 0xbf], [0xe4, 0xb8, 0xad])))).toBe("中");
|
||||
});
|
||||
|
||||
it("UTF-16 靠 BOM 识别", () => {
|
||||
expect(decodeText(ab(u8([0xff, 0xfe], [0x2d, 0x4e])))).toBe("中");
|
||||
expect(decodeText(ab(u8([0xfe, 0xff], [0x4e, 0x2d])))).toBe("中");
|
||||
});
|
||||
|
||||
it("纯 ASCII 原样通过", () => {
|
||||
expect(decodeText(ab(u8([...new TextEncoder().encode("hello\nworld")])))).toBe("hello\nworld");
|
||||
});
|
||||
Reference in New Issue
Block a user