Fix comix titles and covers, kagane volume chapters, and kagane cover rendering #37

Merged
sulthan merged 6 commits from fix/comix-kagane-titles-and-covers into main 2026-08-08 23:27:33 +07:00
3 changed files with 115 additions and 21 deletions
Showing only changes of commit e72765df85 - Show all commits
+19
View File
@@ -44,6 +44,25 @@ Guidance for OpenCode (and Claude Code) working under `userscript/`. See root `A
Encodings (incl. triple-encoded punctuation like `%25252D`) identical Encodings (incl. triple-encoded punctuation like `%25252D`) identical
on /manga/ and /title/ pages, so decode-once seriesIds match — verified on /manga/ and /title/ pages, so decode-once seriesIds match — verified
2026-07-28. 2026-07-28.
- **comix.to**: series `/title/<id>-<slug>`, chapter
`/title/<id>-<slug>/<uploadId>-chapter-<n>`. Only the leading `<id>` is
identity — the slug re-renders when a series is renamed (`comixSeriesId`).
An SPA that **never rewrites `og:title`**: the server-rendered head keeps
whatever document loaded first, so on a cold load `og:title` is the homepage's
"Comix — Read Comics online for free" and after an in-page hop it is the
*previous* series' name. `document.title` is the one thing client routing does
update, so titles come from there, with the chapter page's `" · Ch.<n>"` tail
stripped. Covers likewise: `og:image` is absent, so the cover is the `img`
whose `alt` matches the cleaned title — verified live 2026-08-08.
- **kagane.to**: series `/series/<uuid>`, reader
`/series/<uuid>/reader/<bookUuid>`. Reader URLs carry no chapter number, so
the number comes out of `og:title`. Two shapes exist: `"<Series> - Chapter
<n>[ - Episode <n>]"` and, for volume-numbered series, `"<Series> - Volume <v>
Chapter <n>"` with no episode name — both must yield a bare series title, or
the volume tail lands in the bookmark's title. Its covers are challenge- and
CORP-protected, so the web UI proxies them; the userscript still stores the
raw `og:image`. Behind a Cloudflare JS challenge, so the backend polls it
through the headless browser.
- **novelfull.com** (novel script): series `/<slug>.html`, chapter - **novelfull.com** (novel script): series `/<slug>.html`, chapter
`/<slug>/chapter-<n>[-<title-slug>].html`. No `og:*` tags at all — title from `/<slug>/chapter-<n>[-<title-slug>].html`. No `og:*` tags at all — title from
`h3.title` (series) or `a.truyen-title` (chapter), cover from `h3.title` (series) or `a.truyen-title` (chapter), cover from
+39 -13
View File
@@ -242,6 +242,12 @@
matches: (loc) => /(^|\.)comix\.to$/.test(loc.hostname), matches: (loc) => /(^|\.)comix\.to$/.test(loc.hostname),
detect(loc) { detect(loc) {
const path = loc.pathname; const path = loc.pathname;
// comix client-routes without ever rewriting og:title — the head keeps
// whatever the first server-rendered document carried, so a bookmark
// taken after a client route got the homepage's title, then the
// previous series'. document.title is the one thing its router does
// update. Verified live 2026-08-08; do not "restore" meta("og:title").
const pageTitle = cleanTitle(document.title);
// /title/<id>-<slug>/<uploadId>-chapter-<n>. Several uploads (different // /title/<id>-<slug>/<uploadId>-chapter-<n>. Several uploads (different
// groups or languages) share one chapter number; the number is the // groups or languages) share one chapter number; the number is the
// progress identity, the upload id is not. // progress identity, the upload id is not.
@@ -252,8 +258,8 @@
type: "chapter", type: "chapter",
site: this.site, site: this.site,
seriesId: comixSeriesId(m[1]), seriesId: comixSeriesId(m[1]),
title: cleanTitle(meta("og:title")), title: pageTitle,
cover: coverFromPage(), cover: coverFromPage(pageTitle),
seriesUrl: loc.origin + "/title/" + m[1], seriesUrl: loc.origin + "/title/" + m[1],
chapterLabel: "Chapter " + m[2], chapterLabel: "Chapter " + m[2],
chapterNum: isNaN(num) ? null : num, chapterNum: isNaN(num) ? null : num,
@@ -267,8 +273,8 @@
type: "series", type: "series",
site: this.site, site: this.site,
seriesId: comixSeriesId(m[1]), seriesId: comixSeriesId(m[1]),
title: cleanTitle(meta("og:title")), title: pageTitle,
cover: coverFromPage(), cover: coverFromPage(pageTitle),
seriesUrl: loc.origin + "/title/" + m[1], seriesUrl: loc.origin + "/title/" + m[1],
chapterLabel: null, chapterLabel: null,
chapterNum: null, chapterNum: null,
@@ -277,7 +283,7 @@
} }
return { type: "other" }; return { type: "other" };
// comix chapter og:title is "<Title> · Ch.<n>"; series is clean. // comix chapter document.title is "<Title> · Ch.<n>"; series is clean.
function cleanTitle(t) { function cleanTitle(t) {
if (!t) return ""; if (!t) return "";
return t.replace(/\s*·\s*Ch\.[\d.]+\s*$/i, "").trim(); return t.replace(/\s*·\s*Ch\.[\d.]+\s*$/i, "").trim();
@@ -287,8 +293,7 @@
// the DOM for a cover. Matching on alt rather than a class keeps it off // the DOM for a cover. Matching on alt rather than a class keeps it off
// the site's styling: the cover is the image whose alt is the title. // the site's styling: the cover is the image whose alt is the title.
// Do not "simplify" this into meta("og:image") — that returns null. // Do not "simplify" this into meta("og:image") — that returns null.
function coverFromPage() { function coverFromPage(title) {
const title = cleanTitle(meta("og:title"));
if (!title || !document.querySelectorAll) return ""; if (!title || !document.querySelectorAll) return "";
for (const img of document.querySelectorAll("img[alt]")) { for (const img of document.querySelectorAll("img[alt]")) {
if (img.getAttribute("alt") === title) return img.getAttribute("src") || ""; if (img.getAttribute("alt") === title) return img.getAttribute("src") || "";
@@ -313,6 +318,14 @@
}, },
}; };
// Kagane builds the reader og:title suffix out of the book's metadata, so
// every combination occurs: the volume part appears only when the book has a
// volume_no, the episode part only when it has a non-empty title. All four
// shapes captured live 2026-08-08 — "SP Baby - Volume 1 Chapter 1" is the one
// the old trailing-space regex missed, which left both the number and the
// series title wrong.
const KAGANE_CHAPTER_SUFFIX = /\s-\s(?:Volume\s[\d.]+\s)?Chapter\s([\d.]+)(?:\s-\s.*)?$/i;
const kagane = { const kagane = {
site: "kagane", site: "kagane",
matches: (loc) => /(^|\.)kagane\.to$/.test(loc.hostname), matches: (loc) => /(^|\.)kagane\.to$/.test(loc.hostname),
@@ -353,9 +366,8 @@
} }
return { type: "other" }; return { type: "other" };
// Reader og:title is "<Title> - Chapter <n> - <episode name>".
function chapterNumFromTitle(t) { function chapterNumFromTitle(t) {
const m = t && t.match(/\s-\sChapter\s([\d.]+)\s/); const m = t && t.match(KAGANE_CHAPTER_SUFFIX);
if (!m) return null; if (!m) return null;
const num = parseFloat(m[1]); const num = parseFloat(m[1]);
return isNaN(num) ? null : num; return isNaN(num) ? null : num;
@@ -363,7 +375,7 @@
function cleanTitle(t) { function cleanTitle(t) {
if (!t) return ""; if (!t) return "";
return t.replace(/\s-\sChapter\s[\d.]+\s-\s.*$/i, "").trim(); return t.replace(KAGANE_CHAPTER_SUFFIX, "").trim();
} }
}, },
// Reader hrefs are uuids with no number in them, so no maximum can be taken // Reader hrefs are uuids with no number in them, so no maximum can be taken
@@ -1450,8 +1462,15 @@
} }
let lastUrl = location.href; let lastUrl = location.href;
function onNavigate() { let lastPageSig = "";
function setPage() {
state.page = detect(); state.page = detect();
lastPageSig = JSON.stringify(state.page);
}
function onNavigate() {
setPage();
render(); render();
maybeAutoUpdate(); maybeAutoUpdate();
maybeCaptureLatestOnSeriesPage(); maybeCaptureLatestOnSeriesPage();
@@ -1484,7 +1503,14 @@
wrap("pushState"); wrap("pushState");
wrap("replaceState"); wrap("replaceState");
window.addEventListener("popstate", fire); window.addEventListener("popstate", fire);
setInterval(fire, 1500); // catch routes that bypass history // Also catches routes that bypass history — and comix, which fills
// document.title a beat after the route changes, so the 300ms snapshot
// above can still hold the previous page's title. Re-detect whenever what
// we would read has changed, not only when the URL has.
setInterval(() => {
fire();
if (JSON.stringify(detect()) !== lastPageSig) onNavigate();
}, 1500);
} }
// ============================================================ // ============================================================
@@ -1563,7 +1589,7 @@
function init() { function init() {
buildUI(); buildUI();
state.page = detect(); setPage();
render(); render();
installNavWatcher(); installNavWatcher();
installLongPress(); installLongPress();
+57 -8
View File
@@ -31,6 +31,9 @@ globalThis.location = {
// og: meta tags the adapters read through meta(). Reassigned per test. // og: meta tags the adapters read through meta(). Reassigned per test.
let metaTags = {}; let metaTags = {};
// document.title. comix's SPA rewrites this on client routing but never
// og:title, so the comix adapter reads it instead. Reassigned per test.
let docTitle = "";
// img[alt] elements comix's coverFromPage() scans. Reassigned per test; each // img[alt] elements comix's coverFromPage() scans. Reassigned per test; each
// entry is {alt, src}. // entry is {alt, src}.
let pageImages = []; let pageImages = [];
@@ -48,6 +51,9 @@ globalThis.document = {
})); }));
}, },
addEventListener() {}, addEventListener() {},
get title() {
return docTitle;
},
body: undefined, body: undefined,
}; };
@@ -193,8 +199,15 @@ test("comixSeriesId leaves a bare id untouched", () => {
assert.equal(comixSeriesId("n8we"), "n8we"); assert.equal(comixSeriesId("n8we"), "n8we");
}); });
// comix is an SPA that rewrites document.title on client routing but leaves the
// server-rendered og:title untouched, so every test here pins og:title to a
// STALE value — the homepage title on first hop, the previous series after
// that. Captured live 2026-08-08.
const COMIX_STALE_HOME = "Comix - Read Comics online for free";
test("comix detects a series page", () => { test("comix detects a series page", () => {
metaTags = { "og:title": "Dungeons and Crayons" }; metaTags = { "og:title": COMIX_STALE_HOME };
docTitle = "Dungeons and Crayons";
const p = comix.detect(loc("https://comix.to/title/n8we-dungeons-and-crayons")); const p = comix.detect(loc("https://comix.to/title/n8we-dungeons-and-crayons"));
assert.equal(p.type, "series"); assert.equal(p.type, "series");
assert.equal(p.site, "comix"); assert.equal(p.site, "comix");
@@ -204,8 +217,16 @@ test("comix detects a series page", () => {
assert.equal(p.chapterNum, null); assert.equal(p.chapterNum, null);
}); });
test("comix ignores a previous series' stale og:title", () => {
metaTags = { "og:title": "Full-Time Awakening" };
docTitle = "Dungeons and Crayons";
const p = comix.detect(loc("https://comix.to/title/n8we-dungeons-and-crayons"));
assert.equal(p.title, "Dungeons and Crayons");
});
test("comix detects a chapter page and strips the Ch. suffix from the title", () => { test("comix detects a chapter page and strips the Ch. suffix from the title", () => {
metaTags = { "og:title": "Dungeons and Crayons · Ch.80" }; metaTags = { "og:title": COMIX_STALE_HOME };
docTitle = "Dungeons and Crayons · Ch.80";
const p = comix.detect( const p = comix.detect(
loc("https://comix.to/title/n8we-dungeons-and-crayons/11139891-chapter-80") loc("https://comix.to/title/n8we-dungeons-and-crayons/11139891-chapter-80")
); );
@@ -218,7 +239,8 @@ test("comix detects a chapter page and strips the Ch. suffix from the title", ()
}); });
test("comix parses decimal chapter numbers", () => { test("comix parses decimal chapter numbers", () => {
metaTags = { "og:title": "Dungeons and Crayons · Ch.80.5" }; metaTags = {};
docTitle = "Dungeons and Crayons · Ch.80.5";
const p = comix.detect( const p = comix.detect(
loc("https://comix.to/title/n8we-dungeons-and-crayons/11139891-chapter-80.5") loc("https://comix.to/title/n8we-dungeons-and-crayons/11139891-chapter-80.5")
); );
@@ -226,19 +248,19 @@ test("comix parses decimal chapter numbers", () => {
}); });
test("comix.detect reads the cover from an img whose alt matches the cleaned title", () => { test("comix.detect reads the cover from an img whose alt matches the cleaned title", () => {
metaTags = { "og:title": "Dungeons and Crayons · Ch.80" }; metaTags = { "og:title": COMIX_STALE_HOME };
docTitle = "Dungeons and Crayons";
pageImages = [ pageImages = [
{ alt: "Some Other Series", src: "https://cdn.example/other.jpg" }, { alt: "Some Other Series", src: "https://cdn.example/other.jpg" },
{ alt: "Dungeons and Crayons", src: "https://cdn.example/cover.jpg" }, { alt: "Dungeons and Crayons", src: "https://cdn.example/cover.jpg" },
]; ];
const p = comix.detect( const p = comix.detect(loc("https://comix.to/title/n8we-dungeons-and-crayons"));
loc("https://comix.to/title/n8we-dungeons-and-crayons/11139891-chapter-80")
);
assert.equal(p.cover, "https://cdn.example/cover.jpg"); assert.equal(p.cover, "https://cdn.example/cover.jpg");
}); });
test("comix.detect leaves cover empty when no img alt matches the title", () => { test("comix.detect leaves cover empty when no img alt matches the title", () => {
metaTags = { "og:title": "Dungeons and Crayons" }; metaTags = {};
docTitle = "Dungeons and Crayons";
pageImages = [{ alt: "Some Other Series", src: "https://cdn.example/other.jpg" }]; pageImages = [{ alt: "Some Other Series", src: "https://cdn.example/other.jpg" }];
const p = comix.detect(loc("https://comix.to/title/n8we-dungeons-and-crayons")); const p = comix.detect(loc("https://comix.to/title/n8we-dungeons-and-crayons"));
assert.equal(p.cover, ""); assert.equal(p.cover, "");
@@ -315,6 +337,33 @@ test("kagane reads the chapter number out of og:title", () => {
assert.equal(p.seriesUrl, "https://kagane.to/series/" + KAGANE_SERIES); assert.equal(p.seriesUrl, "https://kagane.to/series/" + KAGANE_SERIES);
}); });
// Volume-numbered series render the suffix as "- Volume <v> Chapter <n>" with
// no episode name, because the book carries volume_no and an empty title.
// Captured live 2026-08-08 from SP Baby.
test("kagane reads through a Volume-numbered chapter suffix", () => {
metaTags = {
"og:title": "SP Baby - Volume 1 Chapter 1",
"og:image": "https://kagane.to/api/v2/image/abc/compressed",
};
const p = kagane.detect(
loc("https://kagane.to/series/" + KAGANE_SERIES + "/reader/" + KAGANE_BOOK)
);
assert.equal(p.title, "SP Baby");
assert.equal(p.chapterNum, 1);
assert.equal(p.chapterLabel, "Chapter 1");
});
// A book with neither a volume nor an episode name ends the title right after
// the number, which the old trailing-\s regex could not match.
test("kagane reads a chapter suffix with no episode name", () => {
metaTags = { "og:title": "Some Series - Chapter 7.5" };
const p = kagane.detect(
loc("https://kagane.to/series/" + KAGANE_SERIES + "/reader/" + KAGANE_BOOK)
);
assert.equal(p.title, "Some Series");
assert.equal(p.chapterNum, 7.5);
});
test("kagane yields a null chapterNum when og:title has no chapter", () => { test("kagane yields a null chapterNum when og:title has no chapter", () => {
metaTags = { "og:title": "Infinite Decryption: The Strongest Level 0" }; metaTags = { "og:title": "Infinite Decryption: The Strongest Level 0" };
const p = kagane.detect( const p = kagane.detect(