feat(cbz): adapt reader to real archive structure — PageIndex skips macOS junk (__MACOSX/ and ._ AppleDouble were polluting pages with 163-byte blacks), pages endpoint derives chapters[{title,start}] from folder layout, imageless zip now lands state=error instead of an empty reader, CbzReader gains 目录 sidebar + prev/next chapter (TextReader pattern, rd-* classes); pagesidx cache key bumped for one-time invalidation; e2e on 311MB 調教開關:第二季.zip: 15048→7524 real pages, 52 chapters, all JPEGs

This commit is contained in:
2026-09-08 00:22:13 +08:00
parent 7d650107d2
commit b8da9fd92e
9 changed files with 216 additions and 3 deletions
+44 -2
View File
@@ -4,6 +4,7 @@ import (
"fmt"
"net/http"
"os"
"path"
"path/filepath"
"strconv"
"strings"
@@ -118,7 +119,7 @@ func (h *H) openBook(c *gin.Context, b store.Book, root string) (*os.File, int64
func (h *H) pageIndex(c *gin.Context, b store.Book, root string) ([]string, error) {
hash := bookfile.Hash(b.FileSize, b.ModTS)
key := fmt.Sprintf("pagesidx:%d:%s", b.ID, hash)
key := fmt.Sprintf("pagesidx2:%d:%s", b.ID, hash) // 前缀换代 = 索引逻辑变更时一次性作废旧缓存
if v, ok := h.rdb.Get(c, key); ok && v != "" {
return strings.Split(v, "\n"), nil
}
@@ -158,7 +159,48 @@ func (h *H) PagesCount(c *gin.Context) {
err(c, http.StatusUnprocessableEntity, "broken", e.Error())
return
}
c.JSON(http.StatusOK, gin.H{"count": len(idx)})
c.JSON(http.StatusOK, gin.H{"count": len(idx), "chapters": chaptersOf(idx)})
}
type cbzChapter struct {
Title string `json:"title"`
Start int `json:"start"`
}
// chaptersOf 页索引已自然序排列,按父目录分组:每个新目录开一章,标题取目录名;扁平包或只有一组返回 nil
func chaptersOf(idx []string) []cbzChapter {
type grp struct {
dir string
start int
}
var grps []grp
last := "\x00"
for i, n := range idx {
d := path.Dir(n)
if d == last {
continue
}
last = d
if d == "." { // 根目录散页不开章
continue
}
grps = append(grps, grp{d, i})
}
if len(grps) < 2 {
return nil
}
titles := make(map[string]int)
out := make([]cbzChapter, len(grps))
for i, g := range grps {
out[i] = cbzChapter{Title: path.Base(g.dir), Start: g.start}
titles[out[i].Title]++
}
for i, g := range grps { // 同名目录(不同父级)撞车 → 用全路径消歧
if titles[out[i].Title] > 1 {
out[i].Title = g.dir
}
}
return out
}
func (h *H) Page(c *gin.Context) {
@@ -1,6 +1,8 @@
package handlers_test
import (
"archive/zip"
"bytes"
"context"
"encoding/json"
"net/http"
@@ -103,6 +105,76 @@ func TestPages(t *testing.T) {
}
}
func zipTo(t *testing.T, path string, kv map[string]string) {
t.Helper()
os.MkdirAll(filepath.Dir(path), 0o755)
buf := &bytes.Buffer{}
zw := zip.NewWriter(buf)
for n, v := range kv {
w, err := zw.Create(n)
if err != nil {
t.Fatal(err)
}
w.Write([]byte(v))
}
zw.Close()
os.WriteFile(path, buf.Bytes(), 0o644)
}
func TestPagesChaptersAndNoImageZip(t *testing.T) {
st, sc, h, booksDir := setupAPI(t)
tok := adminToken(t, h)
lib, root := newLibrary(t, st, h, tok, booksDir, "ch")
zipTo(t, filepath.Join(root, "show.cbz"), map[string]string{
"第2季/第2話/0001.jpg": "IMG2",
"第2季/第1話/0001.jpg": "IMG1a",
"第2季/第1話/._0001.jpg": "junk",
"__MACOSX/第2季/._0001.jpg": "junk",
})
zipTo(t, filepath.Join(root, "videos.zip"), map[string]string{"ep/01.mkv": "x"})
scanNow(t, sc, lib)
bookID := func(q string) string {
w := do(h, "GET", "/api/books?q="+q, tok, nil)
var bs []map[string]any
json.Unmarshal(w.Body.Bytes(), &bs)
if len(bs) != 1 {
t.Fatalf("q=%s books: %s", q, w.Body)
}
return itoa(bs[0]["id"])
}
cbzID := bookID("show")
w := do(h, "GET", "/api/books/"+cbzID+"/pages", tok, nil)
var pg struct {
Count int `json:"count"`
Chapters []struct {
Title string `json:"title"`
Start int `json:"start"`
} `json:"chapters"`
}
json.Unmarshal(w.Body.Bytes(), &pg)
if pg.Count != 2 || len(pg.Chapters) != 2 {
t.Fatalf("pages want 2/2ch got %s", w.Body)
}
if pg.Chapters[0].Title != "第1話" || pg.Chapters[0].Start != 0 || pg.Chapters[1].Start != 1 {
t.Fatalf("chapters: %s", w.Body)
}
ww := do(h, "GET", "/api/books/"+cbzID+"/pages/0", tok, nil)
if ww.Code != 200 || ww.Body.String() != "IMG1a" {
t.Fatalf("page0 must be real image (junk filtered): %d %s", ww.Code, ww.Body)
}
if ww := do(h, "GET", "/api/books/"+cbzID+"/pages/2", tok, nil); ww.Code != 404 {
t.Fatalf("junk-indexed page must be gone: %d", ww.Code)
}
// 无图 zip → state=error,而不是空白 reader
w = do(h, "GET", "/api/books/"+bookID("videos"), tok, nil)
var bk map[string]any
json.Unmarshal(w.Body.Bytes(), &bk)
if bk["state"] != "error" || !strings.Contains(bk["error"].(string), "no images") {
t.Fatalf("empty-image zip want error, got %s", w.Body)
}
}
func TestBrokenCBZ(t *testing.T) {
st, sc, h, booksDir := setupAPI(t)
atok := adminToken(t, h)
+11
View File
@@ -26,6 +26,14 @@ func isImage(name string) bool {
return false
}
// junkEntry: macOS 打包混入的资源叉垃圾(__MACOSX/ 目录与 ._* AppleDouble),不是页
func junkEntry(name string) bool {
if strings.HasPrefix(strings.ToLower(name), "__macosx/") {
return true
}
return strings.HasPrefix(path.Base(name), "._")
}
func PageIndex(f io.ReaderAt, size int64) ([]string, error) {
zr, err := zip.NewReader(f, size)
if err != nil {
@@ -36,6 +44,9 @@ func PageIndex(f io.ReaderAt, size int64) ([]string, error) {
if unsafeEntry(zf.Name) {
return nil, fmt.Errorf("%w: %s", ErrUnsafeZip, zf.Name)
}
if junkEntry(zf.Name) {
continue
}
if isImage(zf.Name) {
names = append(names, zf.Name)
}
+17
View File
@@ -38,6 +38,23 @@ func TestPageIndexSortAndFilter(t *testing.T) {
}
}
func TestPageIndexSkipsAppleDouble(t *testing.T) {
r := zipOf(t, "第2季/第2話/0001.jpg", "第2季/第1話/._0001.jpg", "__MACOSX/第2季/._0001.jpg", "第2季/第1話/0001.jpg", "._top.jpg")
idx, err := PageIndex(r, int64(r.Len()))
if err != nil {
t.Fatal(err)
}
want := []string{"第2季/第1話/0001.jpg", "第2季/第2話/0001.jpg"}
if len(idx) != len(want) {
t.Fatalf("got %v want %v", idx, want)
}
for i := range want {
if idx[i] != want[i] {
t.Fatalf("got %v want %v", idx, want)
}
}
}
func TestPageIndexRejectsSlip(t *testing.T) {
for _, bad := range []string{"../evil.jpg", "/etc/passwd.jpg", "a\\..\\b.jpg", "pag\ne.jpg", "pag\re.jpg"} {
r := zipOf(t, bad)
+7
View File
@@ -2,6 +2,7 @@ package scanner
import (
"context"
"errors"
"fmt"
"io"
"io/fs"
@@ -146,6 +147,9 @@ func (s *Scanner) add(ctx context.Context, libID int64, root, rel string, ds dis
idx, err := s.zipIndex(root, rel)
pageCount = len(idx)
idxErr = err
if idxErr == nil && pageCount == 0 { // 视频/文档 zip 不是漫画,空白 reader 没有意义
idxErr = errors.New("no images in archive")
}
}
id, err := s.st.InsertBook(ctx, libID, rel, titleOf(rel), format, ds.size, ds.modTS, pageCount)
if err != nil {
@@ -167,6 +171,9 @@ func (s *Scanner) update(ctx context.Context, libID, bookID int64, root, rel str
idx, err := s.zipIndex(root, rel)
pageCount = len(idx)
idxErr = err
if idxErr == nil && pageCount == 0 { // 视频/文档 zip 不是漫画,空白 reader 没有意义
idxErr = errors.New("no images in archive")
}
}
if err := s.st.UpdateBookFile(ctx, bookID, ds.size, ds.modTS, pageCount); err != nil {
log.Printf("scan: update %s: %v", rel, err)
+6
View File
@@ -10,11 +10,17 @@ The format loosely follows Keep a Changelog and can be adapted to the team's hab
### Added / 新增
- API: CBZ page indexing now skips macOS packaging junk (`__MACOSX/…` and `._*` AppleDouble files), which used to land in the page list as ~163-byte black "pages"; `GET /api/books/:id/pages` additionally returns `chapters:[{title,start}]` derived from the archive's folder structure (e.g. 第1話…), so per-folder comics expose their real organization.
- API:CBZ 页索引现会跳过 macOS 打包垃圾(`__MACOSX/…` 与 `._*` 资源叉文件),此前它们以 ~163 字节黑页混入页列表;`GET /api/books/:id/pages` 新增 `chapters:[{title,start}]`,按压缩包内目录结构(如 第1話…)给出真实章节。
- API: resumable chunked upload protocol for large files — `POST /api/libraries/:id/upload/init` (fingerprint-derived deterministic `uploadId`, rejects totals over `UPLOAD_MAX_MB` with `413 too_large`), `PUT /api/uploads/:uid/parts/:index` (parts ≤ 32MB), `GET /api/uploads/:uid` (received parts, for resume), `POST /api/uploads/:uid/complete` (assemble + atomic land, same path contract as single-POST upload). Sessions persist under `BOOKS_DIR/.uploads/` with 24h opportunistic sweep. `UPLOAD_MAX_MB` is now wired through both compose stacks/`.env`; `.env.example` sets 2048 and drops `NGINX_CLIENT_MAX_BODY_SIZE` to 32m (nginx only ever sees one chunk).
- API: 新增大文件可续传分片上传协议——`POST /api/libraries/:id/upload/init`(按指纹派生确定性 `uploadId`,总量超 `UPLOAD_MAX_MB` 返回 `413 too_large`)、`PUT /api/uploads/:uid/parts/:index`(单片 ≤32MB)、`GET /api/uploads/:uid`(查询已传分片以续传)、`POST /api/uploads/:uid/complete`(拼接后原子落盘,返回与单发上传一致的 `path`)。会话存于 `BOOKS_DIR/.uploads/`,超 24h 顺手清理。`UPLOAD_MAX_MB` 已接入两份 compose/`.env`;`.env.example` 调至 2048 并将 `NGINX_CLIENT_MAX_BODY_SIZE` 降为 32m(nginx 只见单个分片)。
### Changed / 变更
- Scanner: an image-list-less archive (`.zip`/`.cbz` with no page images — video packs, document dumps) is now recorded as `state=error` ("no images in archive") instead of registering as an empty CBZ with a blank reader.
- 扫描器:不含任何图片条目的 `.zip`/`.cbz`(视频包、文档包)现记录为 `state=error`("no images in archive"),不再注册成空 CBZ 留下一个白板阅读器。
- API: upload over `UPLOAD_MAX_MB` now returns `413 too_large` with the limit in the message; previously the size abort was misreported as `400 bad_request "multipart field 'file' required"`.
- API:超过 `UPLOAD_MAX_MB` 的上传现在返回 `413 too_large` 并在消息中带上限额;此前体积超限被误报为 `400 bad_request "multipart field 'file' required"`。
- API: `POST /api/libraries` now takes only `{name}`; `root_path` is generated server-side as `BOOKS_DIR/<sanitized name>` (no client-supplied paths, validated at creation).
+3
View File
@@ -10,6 +10,9 @@ The format loosely follows Keep a Changelog and can be adapted to the team's hab
### Added / 新增
- Folder-structured comics gain a 目录 sidebar in the CBZ reader: each archive folder (第N話…) becomes a chapter that jumps to its first page, with 上一章/下一章 buttons, a current-chapter counter, and the active row highlighted and scrolled into view — flat archives show no extra UI.
- 按目录组织的漫画在 CBZ 阅读器新增「目录」侧栏:压缩包内每个文件夹(第N話…)即一章,点击直达该话首页;工具条配上一章/下一章与当前章计数,目录内高亮当前章并自动居中;平铺无目录的包不显示多余控件。
- Library uploads over 16MB now auto-switch to the resumable chunked protocol (8MB parts): failed parts retry in place up to 3 times and already-sent parts are skipped, so a flaky upload continues instead of starting over; small files keep the original single request.
- 书库上传超过 16MB 自动切换为可续传分片协议(每片 8MB):失败的分片原地重试至多 3 次,已传片自动跳过,中断后继续而不是从头再来;小文件仍走原单次上传。
+2 -1
View File
@@ -172,7 +172,8 @@ export const api = {
},
getBook: (id: number) => apiFetch<Book>(`/books/${id}`),
deleteBook: (id: number) => apiFetch<void>(`/books/${id}`, { method: "DELETE" }),
pageCount: (pagesUrl: string) => apiFetch<{ count: number }>(pagesUrl),
pageCount: (pagesUrl: string) =>
apiFetch<{ count: number; chapters?: { title: string; start: number }[] | null }>(pagesUrl),
putProgress: (id: number, locator: Record<string, unknown>, percent: number, keepalive = false) =>
apiFetch<void>(`/books/${id}/progress`, { method: "PUT", body: { locator, percent }, keepalive }),
+54
View File
@@ -84,6 +84,9 @@ export default function CbzReader({ book, initialLocator, chrome }: ReaderProps)
retry: 0,
});
const count = countQ.data?.count ?? 0;
const chapters = countQ.data?.chapters ?? null; // 包内目录结构(第N話/上中下卷)
const [toc, setToc] = useState(false);
const activeRow = useRef<HTMLButtonElement>(null);
const saver = useProgressSaver(book.id);
const boxRef = useRef<HTMLDivElement>(null);
const [width, setWidth] = useState(480);
@@ -168,6 +171,10 @@ export default function CbzReader({ book, initialLocator, chrome }: ReaderProps)
}
const cur = ph.pageAt(top + vh / 2);
const ci = chapters ? chapters.reduce((acc, c, i) => (c.start <= cur ? i : acc), 0) : 0;
useEffect(() => {
if (toc) activeRow.current?.scrollIntoView({ block: "center" });
}, [toc]);
function jump(i: number, smooth = true) {
const el = boxRef.current;
@@ -284,6 +291,26 @@ export default function CbzReader({ book, initialLocator, chrome }: ReaderProps)
>
{MODE_LABEL[mode]}
</button>
{chapters && (
<>
<button className="rd-btn" aria-expanded={toc} aria-controls="cbz-toc" onClick={() => setToc((v) => !v)}>
目录
</button>
<span className="text-xs tabular-nums opacity-60">
{ci + 1}/{chapters.length}
</span>
<button className="rd-btn" disabled={ci === 0} onClick={() => jump(chapters[ci - 1].start)}>
← 上一章
</button>
<button
className="rd-btn"
disabled={ci === chapters.length - 1}
onClick={() => jump(chapters[ci + 1].start)}
>
下一章 →
</button>
</>
)}
</div>
</div>
) : (
@@ -294,6 +321,33 @@ export default function CbzReader({ book, initialLocator, chrome }: ReaderProps)
<span className="opacity-60">{Math.round(((cur + 1) / Math.max(1, count)) * 100)}%</span>
</div>
)}
{toc && chapters && (
<>
<div className="fixed inset-0 z-10" aria-hidden="true" onClick={() => setToc(false)} />
<nav
id="cbz-toc"
aria-label="章节目录"
className="rd-divider absolute inset-y-0 left-0 z-20 w-64 overflow-y-auto border-r bg-[var(--rd-bg)] p-2 shadow-2xl"
>
<ul className="space-y-0.5">
{chapters.map((c, i) => (
<li key={i}>
<button
ref={i === ci ? activeRow : undefined}
className={"rd-row" + (i === ci ? " rd-btn-on" : "")}
onClick={() => {
jump(c.start);
setToc(false);
}}
>
{i + 1}. {c.title}
</button>
</li>
))}
</ul>
</nav>
</>
)}
</div>
);
}