fix(reader): split large txt into chapters (章 as boundaries, 卷 as optgroup labels, pseudo-sections when unmarked) rendered one at a time so the browser stops freezing; decode via strict-UTF-8-then-GB18030 with UTF-16 BOM sniffing to kill mojibake; progress locator {ch,scrollFraction} with old global-fraction fallback
This commit is contained in:
@@ -0,0 +1,65 @@
|
||||
// txt 章节切分:优先识别「第X章」类标题行;「第X卷/部」是层级不是章节边界,
|
||||
// 只做分组标签;整本没有章标记时才退而用卷作边界;再没有就按段落边界切
|
||||
// 伪章节,保证任何单块都能在浏览器里流畅渲染。纯逻辑,无 DOM,vitest 覆盖。
|
||||
export interface TxtChapter {
|
||||
title: string;
|
||||
start: number; // 正文中的起始字符偏移(含标题行本身)
|
||||
vol?: string; // 所属卷标题(仅用于导航分组)
|
||||
}
|
||||
|
||||
const NUM = "[0-9〇零一二三四五六七八九十百千万两]+";
|
||||
const CH_RE = new RegExp(
|
||||
`^[ \\t\\u3000]*(?:第\\s*${NUM}\\s*[章节回篇话]|Chapter\\s+${NUM}|序[章言]|楔子|引子|尾声|终章|后记|番外)[^\\n]{0,60}[ \\t]*$`,
|
||||
"gim",
|
||||
);
|
||||
const VOL_RE = new RegExp(`^[ \\t\\u3000]*第\\s*${NUM}\\s*[卷部][^\\n]{0,60}[ \\t]*$`, "gm");
|
||||
|
||||
const MIN_GAP = 200; // 距上一保留标题不足此数视为目录行/正文提及,丢弃
|
||||
const CHUNK = 20_000; // 伪章节目标字数
|
||||
|
||||
function markers(text: string, re: RegExp): TxtChapter[] {
|
||||
const out: TxtChapter[] = [];
|
||||
for (const m of text.matchAll(re)) {
|
||||
const i = m.index ?? 0;
|
||||
if (out.length && i - out[out.length - 1].start < MIN_GAP) continue;
|
||||
out.push({ title: m[0].trim().slice(0, 40), start: i });
|
||||
}
|
||||
return out;
|
||||
}
|
||||
|
||||
export function splitChapters(text: string): TxtChapter[] {
|
||||
const vols = markers(text, VOL_RE);
|
||||
let chs = markers(text, CH_RE);
|
||||
let volMarks = vols;
|
||||
if (chs.length < 2 && vols.length > 1) {
|
||||
chs = vols; // 只有卷的书:卷即章节
|
||||
volMarks = [];
|
||||
}
|
||||
if (chs.length > 1) {
|
||||
if (chs[0].start > MIN_GAP) chs.unshift({ title: "开篇", start: 0 });
|
||||
for (const c of chs) {
|
||||
let v: TxtChapter | undefined;
|
||||
for (const m of volMarks) if (m.start < c.start) v = m;
|
||||
else break;
|
||||
if (!v) continue;
|
||||
c.vol = v.title;
|
||||
if (c.start - v.start < MIN_GAP) c.start = v.start; // 卷题行并入本章开头,不留在上一章尾部
|
||||
}
|
||||
return chs;
|
||||
}
|
||||
const out: TxtChapter[] = [];
|
||||
for (let s = 0, i = 1; s < text.length; i++) {
|
||||
let e = Math.min(s + CHUNK, text.length);
|
||||
if (e < text.length) {
|
||||
const brk = text.lastIndexOf("\n", e);
|
||||
if (brk > s) e = brk + 1;
|
||||
}
|
||||
out.push({ title: `第 ${i} 节`, start: s });
|
||||
s = e;
|
||||
}
|
||||
return out;
|
||||
}
|
||||
|
||||
export function chapterText(text: string, chapters: TxtChapter[], i: number): string {
|
||||
return text.slice(chapters[i].start, chapters[i + 1]?.start ?? text.length);
|
||||
}
|
||||
@@ -0,0 +1,12 @@
|
||||
// 网络 txt 常是 GBK/GB2312 而服务端一律标 charset=utf-8:先严格试 UTF-8,
|
||||
// 失败回退 GB18030(GBK 超集);UTF-16 靠 BOM 识别。纯逻辑,vitest 覆盖。
|
||||
export function decodeText(buf: ArrayBuffer): string {
|
||||
const b = new Uint8Array(buf);
|
||||
if (b[0] === 0xff && b[1] === 0xfe) return new TextDecoder("utf-16le").decode(b);
|
||||
if (b[0] === 0xfe && b[1] === 0xff) return new TextDecoder("utf-16be").decode(b);
|
||||
try {
|
||||
return new TextDecoder("utf-8", { fatal: true }).decode(b);
|
||||
} catch {
|
||||
return new TextDecoder("gb18030").decode(b);
|
||||
}
|
||||
}
|
||||
Reference in New Issue
Block a user