diff --git a/userscript/AGENTS.md b/userscript/AGENTS.md index 6c9e9c3..2a0b198 100644 --- a/userscript/AGENTS.md +++ b/userscript/AGENTS.md @@ -44,6 +44,25 @@ Guidance for OpenCode (and Claude Code) working under `userscript/`. See root `A Encodings (incl. triple-encoded punctuation like `%25252D`) identical on /manga/ and /title/ pages, so decode-once seriesIds match — verified 2026-07-28. +- **comix.to**: series `/title/-`, chapter + `/title/-/-chapter-`. Only the leading `` is + identity — the slug re-renders when a series is renamed (`comixSeriesId`). + An SPA that **never rewrites `og:title`**: the server-rendered head keeps + whatever document loaded first, so on a cold load `og:title` is the homepage's + "Comix — Read Comics online for free" and after an in-page hop it is the + *previous* series' name. `document.title` is the one thing client routing does + update, so titles come from there, with the chapter page's `" · Ch."` tail + stripped. Covers likewise: `og:image` is absent, so the cover is the `img` + whose `alt` matches the cleaned title — verified live 2026-08-08. +- **kagane.to**: series `/series/`, reader + `/series//reader/`. Reader URLs carry no chapter number, so + the number comes out of `og:title`. Two shapes exist: `" - Chapter + [ - Episode ]"` and, for volume-numbered series, `" - Volume + Chapter "` with no episode name — both must yield a bare series title, or + the volume tail lands in the bookmark's title. Its covers are challenge- and + CORP-protected, so the web UI proxies them; the userscript still stores the + raw `og:image`. Behind a Cloudflare JS challenge, so the backend polls it + through the headless browser. - **novelfull.com** (novel script): series `/.html`, chapter `//chapter-[-].html`. No `og:*` tags at all — title from `h3.title` (series) or `a.truyen-title` (chapter), cover from diff --git a/userscript/manga-bookmark.user.js b/userscript/manga-bookmark.user.js index f2689a4..c7e0a26 100644 --- a/userscript/manga-bookmark.user.js +++ b/userscript/manga-bookmark.user.js @@ -242,6 +242,12 @@ matches: (loc) => /(^|\.)comix\.to$/.test(loc.hostname), detect(loc) { const path = loc.pathname; + // comix client-routes without ever rewriting og:title — the head keeps + // whatever the first server-rendered document carried, so a bookmark + // taken after a client route got the homepage's title, then the + // previous series'. document.title is the one thing its router does + // update. Verified live 2026-08-08; do not "restore" meta("og:title"). + const pageTitle = cleanTitle(document.title); // /title/-/-chapter-. Several uploads (different // groups or languages) share one chapter number; the number is the // progress identity, the upload id is not. @@ -252,8 +258,8 @@ type: "chapter", site: this.site, seriesId: comixSeriesId(m[1]), - title: cleanTitle(meta("og:title")), - cover: coverFromPage(), + title: pageTitle, + cover: coverFromPage(pageTitle), seriesUrl: loc.origin + "/title/" + m[1], chapterLabel: "Chapter " + m[2], chapterNum: isNaN(num) ? null : num, @@ -267,8 +273,8 @@ type: "series", site: this.site, seriesId: comixSeriesId(m[1]), - title: cleanTitle(meta("og:title")), - cover: coverFromPage(), + title: pageTitle, + cover: coverFromPage(pageTitle), seriesUrl: loc.origin + "/title/" + m[1], chapterLabel: null, chapterNum: null, @@ -277,7 +283,7 @@ } return { type: "other" }; - // comix chapter og:title is " · Ch.<n>"; series is clean. + // comix chapter document.title is "<Title> · Ch.<n>"; series is clean. function cleanTitle(t) { if (!t) return ""; return t.replace(/\s*·\s*Ch\.[\d.]+\s*$/i, "").trim(); @@ -287,8 +293,7 @@ // the DOM for a cover. Matching on alt rather than a class keeps it off // the site's styling: the cover is the image whose alt is the title. // Do not "simplify" this into meta("og:image") — that returns null. - function coverFromPage() { - const title = cleanTitle(meta("og:title")); + function coverFromPage(title) { if (!title || !document.querySelectorAll) return ""; for (const img of document.querySelectorAll("img[alt]")) { if (img.getAttribute("alt") === title) return img.getAttribute("src") || ""; @@ -313,6 +318,14 @@ }, }; + // Kagane builds the reader og:title suffix out of the book's metadata, so + // every combination occurs: the volume part appears only when the book has a + // volume_no, the episode part only when it has a non-empty title. All four + // shapes captured live 2026-08-08 — "SP Baby - Volume 1 Chapter 1" is the one + // the old trailing-space regex missed, which left both the number and the + // series title wrong. + const KAGANE_CHAPTER_SUFFIX = /\s-\s(?:Volume\s[\d.]+\s)?Chapter\s([\d.]+)(?:\s-\s.*)?$/i; + const kagane = { site: "kagane", matches: (loc) => /(^|\.)kagane\.to$/.test(loc.hostname), @@ -353,9 +366,8 @@ } return { type: "other" }; - // Reader og:title is "<Title> - Chapter <n> - <episode name>". function chapterNumFromTitle(t) { - const m = t && t.match(/\s-\sChapter\s([\d.]+)\s/); + const m = t && t.match(KAGANE_CHAPTER_SUFFIX); if (!m) return null; const num = parseFloat(m[1]); return isNaN(num) ? null : num; @@ -363,7 +375,7 @@ function cleanTitle(t) { if (!t) return ""; - return t.replace(/\s-\sChapter\s[\d.]+\s-\s.*$/i, "").trim(); + return t.replace(KAGANE_CHAPTER_SUFFIX, "").trim(); } }, // Reader hrefs are uuids with no number in them, so no maximum can be taken @@ -1450,8 +1462,15 @@ } let lastUrl = location.href; - function onNavigate() { + let lastPageSig = ""; + + function setPage() { state.page = detect(); + lastPageSig = JSON.stringify(state.page); + } + + function onNavigate() { + setPage(); render(); maybeAutoUpdate(); maybeCaptureLatestOnSeriesPage(); @@ -1484,7 +1503,14 @@ wrap("pushState"); wrap("replaceState"); window.addEventListener("popstate", fire); - setInterval(fire, 1500); // catch routes that bypass history + // Also catches routes that bypass history — and comix, which fills + // document.title a beat after the route changes, so the 300ms snapshot + // above can still hold the previous page's title. Re-detect whenever what + // we would read has changed, not only when the URL has. + setInterval(() => { + fire(); + if (JSON.stringify(detect()) !== lastPageSig) onNavigate(); + }, 1500); } // ============================================================ @@ -1563,7 +1589,7 @@ function init() { buildUI(); - state.page = detect(); + setPage(); render(); installNavWatcher(); installLongPress(); diff --git a/userscript/test/logic.test.js b/userscript/test/logic.test.js index 73230d0..fdeff51 100644 --- a/userscript/test/logic.test.js +++ b/userscript/test/logic.test.js @@ -31,6 +31,9 @@ globalThis.location = { // og: meta tags the adapters read through meta(). Reassigned per test. let metaTags = {}; +// document.title. comix's SPA rewrites this on client routing but never +// og:title, so the comix adapter reads it instead. Reassigned per test. +let docTitle = ""; // img[alt] elements comix's coverFromPage() scans. Reassigned per test; each // entry is {alt, src}. let pageImages = []; @@ -48,6 +51,9 @@ globalThis.document = { })); }, addEventListener() {}, + get title() { + return docTitle; + }, body: undefined, }; @@ -193,8 +199,15 @@ test("comixSeriesId leaves a bare id untouched", () => { assert.equal(comixSeriesId("n8we"), "n8we"); }); +// comix is an SPA that rewrites document.title on client routing but leaves the +// server-rendered og:title untouched, so every test here pins og:title to a +// STALE value — the homepage title on first hop, the previous series after +// that. Captured live 2026-08-08. +const COMIX_STALE_HOME = "Comix - Read Comics online for free"; + test("comix detects a series page", () => { - metaTags = { "og:title": "Dungeons and Crayons" }; + metaTags = { "og:title": COMIX_STALE_HOME }; + docTitle = "Dungeons and Crayons"; const p = comix.detect(loc("https://comix.to/title/n8we-dungeons-and-crayons")); assert.equal(p.type, "series"); assert.equal(p.site, "comix"); @@ -204,8 +217,16 @@ test("comix detects a series page", () => { assert.equal(p.chapterNum, null); }); +test("comix ignores a previous series' stale og:title", () => { + metaTags = { "og:title": "Full-Time Awakening" }; + docTitle = "Dungeons and Crayons"; + const p = comix.detect(loc("https://comix.to/title/n8we-dungeons-and-crayons")); + assert.equal(p.title, "Dungeons and Crayons"); +}); + test("comix detects a chapter page and strips the Ch. suffix from the title", () => { - metaTags = { "og:title": "Dungeons and Crayons · Ch.80" }; + metaTags = { "og:title": COMIX_STALE_HOME }; + docTitle = "Dungeons and Crayons · Ch.80"; const p = comix.detect( loc("https://comix.to/title/n8we-dungeons-and-crayons/11139891-chapter-80") ); @@ -218,7 +239,8 @@ test("comix detects a chapter page and strips the Ch. suffix from the title", () }); test("comix parses decimal chapter numbers", () => { - metaTags = { "og:title": "Dungeons and Crayons · Ch.80.5" }; + metaTags = {}; + docTitle = "Dungeons and Crayons · Ch.80.5"; const p = comix.detect( loc("https://comix.to/title/n8we-dungeons-and-crayons/11139891-chapter-80.5") ); @@ -226,19 +248,19 @@ test("comix parses decimal chapter numbers", () => { }); test("comix.detect reads the cover from an img whose alt matches the cleaned title", () => { - metaTags = { "og:title": "Dungeons and Crayons · Ch.80" }; + metaTags = { "og:title": COMIX_STALE_HOME }; + docTitle = "Dungeons and Crayons"; pageImages = [ { alt: "Some Other Series", src: "https://cdn.example/other.jpg" }, { alt: "Dungeons and Crayons", src: "https://cdn.example/cover.jpg" }, ]; - const p = comix.detect( - loc("https://comix.to/title/n8we-dungeons-and-crayons/11139891-chapter-80") - ); + const p = comix.detect(loc("https://comix.to/title/n8we-dungeons-and-crayons")); assert.equal(p.cover, "https://cdn.example/cover.jpg"); }); test("comix.detect leaves cover empty when no img alt matches the title", () => { - metaTags = { "og:title": "Dungeons and Crayons" }; + metaTags = {}; + docTitle = "Dungeons and Crayons"; pageImages = [{ alt: "Some Other Series", src: "https://cdn.example/other.jpg" }]; const p = comix.detect(loc("https://comix.to/title/n8we-dungeons-and-crayons")); assert.equal(p.cover, ""); @@ -315,6 +337,33 @@ test("kagane reads the chapter number out of og:title", () => { assert.equal(p.seriesUrl, "https://kagane.to/series/" + KAGANE_SERIES); }); +// Volume-numbered series render the suffix as "- Volume <v> Chapter <n>" with +// no episode name, because the book carries volume_no and an empty title. +// Captured live 2026-08-08 from SP Baby. +test("kagane reads through a Volume-numbered chapter suffix", () => { + metaTags = { + "og:title": "SP Baby - Volume 1 Chapter 1", + "og:image": "https://kagane.to/api/v2/image/abc/compressed", + }; + const p = kagane.detect( + loc("https://kagane.to/series/" + KAGANE_SERIES + "/reader/" + KAGANE_BOOK) + ); + assert.equal(p.title, "SP Baby"); + assert.equal(p.chapterNum, 1); + assert.equal(p.chapterLabel, "Chapter 1"); +}); + +// A book with neither a volume nor an episode name ends the title right after +// the number, which the old trailing-\s regex could not match. +test("kagane reads a chapter suffix with no episode name", () => { + metaTags = { "og:title": "Some Series - Chapter 7.5" }; + const p = kagane.detect( + loc("https://kagane.to/series/" + KAGANE_SERIES + "/reader/" + KAGANE_BOOK) + ); + assert.equal(p.title, "Some Series"); + assert.equal(p.chapterNum, 7.5); +}); + test("kagane yields a null chapterNum when og:title has no chapter", () => { metaTags = { "og:title": "Infinite Decryption: The Strongest Level 0" }; const p = kagane.detect(