diff --git a/userscript/AGENTS.md b/userscript/AGENTS.md index 9966c62..98fe033 100644 --- a/userscript/AGENTS.md +++ b/userscript/AGENTS.md @@ -70,9 +70,13 @@ Guidance for OpenCode (and Claude Code) working under `userscript/`. See root `A Behind a Cloudflare JS challenge no TLS fingerprint clears, so the backend polls it through the headless browser. - **lightnovelworld.net** (novel script): series `/novel//`, chapter - `/-chapter-/` — flat, at the site root. `h1.entry-title` is the clean - title on a series page and ` Chapter <n>` on a chapter page. Its series - page lists every chapter with an + `/<slug>-chapter-<n>/` — flat, at the site root. The chapter path's slug is a + Chapter Slug, not an identity: the Series address is read off the page's + `a[aria-label='All Chapter']` (fallback: the BreadcrumbList's second crumb), + and a Series may publish under several Chapter Slugs. A chapter page with no + pointer resolves to `other`, so no Bookmark is offered. `h1.entry-title` is + the clean title on a series page and `<Title> Chapter <n>` on a chapter page. + Its series page lists every chapter with an absolute href, so the backend polls it with the plain TLS client. ### Second script: `novel-bookmark.user.js` diff --git a/userscript/novel-bookmark.user.js b/userscript/novel-bookmark.user.js index cca85f8..07480f5 100644 --- a/userscript/novel-bookmark.user.js +++ b/userscript/novel-bookmark.user.js @@ -132,6 +132,23 @@ }, }; + // A chapter page's pointer is its own link back to its Series. The href is + // page markup, so validate before trusting: the host must be this Site's (a + // leading subdomain is allowed, as in `matches`) and the path the + // /novel/<slug>/ Series shape. Anything else is not a pointer. + function seriesIdFromLnwPointer(href) { + if (!href) return null; + let u; + try { + u = new URL(href); + } catch (e) { + return null; + } + if (!/(^|\.)lightnovelworld\.net$/.test(u.hostname)) return null; + const m = u.pathname.match(/^\/novel\/([^/]+)\/?$/); + return m ? m[1] : null; + } + const lightnovelworld = { site: "lightnovelworld", matches: (loc) => /(^|\.)lightnovelworld\.net$/.test(loc.hostname), @@ -142,16 +159,38 @@ // words still resolves to the right series. let m = path.match(/^\/(.+)-chapter-([0-9]+(?:\.[0-9]+)?)\/?$/); if (m) { + // The address is not the identity on this Site: the slug in the path + // is a Chapter Slug, which can differ from the Series slug and is + // never computable from it. The Series address is read from the + // page's own pointer — a silent fallback to derivation is the defect + // this replaced, not a safety net. + const chapterSlug = m[1]; + const pointer = document.querySelector("a[aria-label='All Chapter']"); + let seriesId = pointer ? seriesIdFromLnwPointer(pointer.getAttribute("href")) : null; + if (!seriesId) { + // Fallback: the microdata breadcrumb's second crumb is the Series. + // Scoped to the BreadcrumbList because itemprop="item" is not + // unique to it (the header nav uses microdata too). + const crumb = document.querySelector( + '[itemtype="http://schema.org/BreadcrumbList"] a[itemprop="item"][href*="/novel/"]' + ); + seriesId = crumb ? seriesIdFromLnwPointer(crumb.getAttribute("href")) : null; + } + // Neither pointer present, nor either pointing at a /novel/<slug>/ + // address on this host: not a page the script understands, so no + // Bookmark under an invented identity. + if (!seriesId) return { type: "other" }; const num = parseFloat(m[2]); const h1 = document.querySelector("h1.entry-title"); const heading = h1 ? h1.textContent || "" : ""; return { type: "chapter", site: this.site, - seriesId: m[1], + seriesId: seriesId, + chapterSlug: chapterSlug, // The heading is "<Series> Chapter <n>"; drop the suffix. title: heading.replace(/\s*Chapter\s+[0-9.]+\s*$/i, "").trim(), - seriesUrl: "https://lightnovelworld.net/novel/" + m[1] + "/", + seriesUrl: "https://lightnovelworld.net/novel/" + seriesId + "/", chapterLabel: "Chapter " + m[2], chapterNum: isNaN(num) ? null : num, chapterUrl: loc.href, @@ -165,6 +204,7 @@ type: "series", site: this.site, seriesId: m[1], + chapterSlug: null, title: h1 ? (h1.textContent || "").trim() : "", seriesUrl: "https://lightnovelworld.net/novel/" + m[1] + "/", chapterLabel: null, diff --git a/userscript/test/novel-logic.test.js b/userscript/test/novel-logic.test.js index 926e8c7..86d5173 100644 --- a/userscript/test/novel-logic.test.js +++ b/userscript/test/novel-logic.test.js @@ -20,8 +20,13 @@ globalThis.location = { href: "about:blank", hostname: "", pathname: "/", origin let metaTags = {}; let elements = {}; +// Attribute selectors (the lightnovelworld Series pointer) answer with an +// element exposing getAttribute, like the meta branch below. +let attrEls = {}; globalThis.document = { querySelector(sel) { + const attr = attrEls[sel]; + if (attr != null) return attr; const m = sel.match(/^meta\[property="([^"]+)"\]$/); if (m) { const v = metaTags[m[1]]; @@ -52,6 +57,7 @@ function loc(href) { function reset() { metaTags = {}; elements = {}; + attrEls = {}; } // ============================================================ @@ -115,11 +121,18 @@ test("lightnovelworld.detect reads a series page", () => { assert.equal(p.site, "lightnovelworld"); assert.equal(p.seriesId, "a-will-eternal"); assert.equal(p.title, "A Will Eternal"); + // A series page's address *is* its identity; there is no Chapter Slug. + assert.equal(p.chapterSlug, null); }); test("lightnovelworld.detect strips the chapter suffix off the heading", () => { reset(); elements = { "h1.entry-title": "A Will Eternal Chapter 1298" }; + attrEls = { + "a[aria-label='All Chapter']": { + getAttribute: () => "https://lightnovelworld.net/novel/a-will-eternal/", + }, + }; const url = "https://lightnovelworld.net/a-will-eternal-chapter-1298/"; const p = lightnovelworld.detect(loc(url)); assert.equal(p.type, "chapter"); @@ -130,6 +143,126 @@ test("lightnovelworld.detect strips the chapter suffix off the heading", () => { assert.equal(p.seriesUrl, "https://lightnovelworld.net/novel/a-will-eternal/"); }); +test("lightnovelworld.detect reads the Series identity from the page pointer on a chapter page", () => { + reset(); + elements = { "h1.entry-title": "A Will Eternal Chapter 1298" }; + attrEls = { + // Verbatim from the real page (research note §3): the All Chapter anchor + // carries the absolute Series address. + "a[aria-label='All Chapter']": { + getAttribute: () => "https://lightnovelworld.net/novel/a-will-eternal/", + }, + }; + const url = "https://lightnovelworld.net/a-will-eternal-chapter-1298/"; + const p = lightnovelworld.detect(loc(url)); + assert.equal(p.type, "chapter"); + assert.equal(p.seriesId, "a-will-eternal"); + assert.equal(p.seriesUrl, "https://lightnovelworld.net/novel/a-will-eternal/"); + assert.equal(p.chapterSlug, "a-will-eternal"); + assert.equal(p.chapterNum, 1298); +}); + +test("lightnovelworld.detect falls back to the breadcrumb when the pointer is absent", () => { + reset(); + elements = { "h1.entry-title": "My Longevity Simulation Chapter 1" }; + attrEls = { + // Verbatim from the real page (research note §3): position 2 of the + // microdata BreadcrumbList is the Series. + '[itemtype="http://schema.org/BreadcrumbList"] a[itemprop="item"][href*="/novel/"]': { + getAttribute: () => "https://lightnovelworld.net/novel/immortality-simulator/", + }, + }; + const url = "https://lightnovelworld.net/my-longevity-simulation-chapter-1/"; + const p = lightnovelworld.detect(loc(url)); + assert.equal(p.type, "chapter"); + assert.equal(p.seriesId, "immortality-simulator"); + assert.equal(p.seriesUrl, "https://lightnovelworld.net/novel/immortality-simulator/"); + assert.equal(p.chapterSlug, "my-longevity-simulation"); +}); + +test("lightnovelworld.detect resolves to other when the page carries no pointer", () => { + reset(); + elements = { "h1.entry-title": "My Longevity Simulation Chapter 1" }; + const p = lightnovelworld.detect( + loc("https://lightnovelworld.net/my-longevity-simulation-chapter-1/") + ); + assert.equal(p.type, "other"); +}); + +test("lightnovelworld pins the divergent novel: the pointer's slug wins over the address's", () => { + // Regression pin for the derivation defect (research note §2/§3): the + // chapter address is built from "my-longevity-simulation" but the Series + // is published as "immortality-simulator". If the adapter ever derives the + // identity from the address again, this test goes red. + reset(); + elements = { "h1.entry-title": "My Longevity Simulation Chapter 1" }; + attrEls = { + "a[aria-label='All Chapter']": { + getAttribute: () => "https://lightnovelworld.net/novel/immortality-simulator/", + }, + }; + const url = "https://lightnovelworld.net/my-longevity-simulation-chapter-1/"; + const p = lightnovelworld.detect(loc(url)); + assert.equal(p.type, "chapter"); + assert.equal(p.seriesId, "immortality-simulator"); + assert.equal(p.seriesUrl, "https://lightnovelworld.net/novel/immortality-simulator/"); + assert.equal(p.chapterSlug, "my-longevity-simulation"); + assert.notEqual(p.chapterSlug, p.seriesId); +}); + +test("lightnovelworld.detect resolves a novel whose heading ends in a chapter number", () => { + // The split novel from research §4.4, chapter 200 (published under the + // current slug). Its heading ends "…Not Them All Chapter 200", which would + // false-match a selector that looks for the text "All Chapter". + reset(); + elements = { + "h1.entry-title": "All Jobs and Classes I Just Wanted One Skill Not Them All Chapter 200", + }; + attrEls = { + "a[aria-label='All Chapter']": { + getAttribute: () => + "https://lightnovelworld.net/novel/all-jobs-and-classes-i-just-wanted-one-skill-not-them-all/", + }, + }; + const url = + "https://lightnovelworld.net/all-jobs-and-classes-i-just-wanted-one-skill-not-them-all-chapter-200/"; + const p = lightnovelworld.detect(loc(url)); + assert.equal(p.type, "chapter"); + assert.equal(p.seriesId, "all-jobs-and-classes-i-just-wanted-one-skill-not-them-all"); + assert.equal(p.title, "All Jobs and Classes I Just Wanted One Skill Not Them All"); + assert.equal(p.chapterNum, 200); + assert.equal(p.chapterSlug, "all-jobs-and-classes-i-just-wanted-one-skill-not-them-all"); +}); + +test("lightnovelworld.detect resolves both Chapter Slugs of the split novel to one Series", () => { + // Research note §4.4: this Series serves chapters 1-99 under one Chapter + // Slug and 100-423 under another; both chapter addresses are live and both + // point back at the same Series. An old deep link must not get a different + // identity than a current one. + reset(); + elements = { "h1.entry-title": "All Jobs and Classes I Just Wanted One Skill Not Chapter 1" }; + attrEls = { + "a[aria-label='All Chapter']": { + getAttribute: () => + "https://lightnovelworld.net/novel/all-jobs-and-classes-i-just-wanted-one-skill-not-them-all/", + }, + }; + const oldSlug = lightnovelworld.detect( + loc("https://lightnovelworld.net/all-jobs-and-classes-i-just-wanted-one-skill-not-chapter-1/") + ); + const currentSlug = lightnovelworld.detect( + loc( + "https://lightnovelworld.net/all-jobs-and-classes-i-just-wanted-one-skill-not-them-all-chapter-404/" + ) + ); + assert.equal(oldSlug.type, "chapter"); + assert.equal(oldSlug.seriesId, "all-jobs-and-classes-i-just-wanted-one-skill-not-them-all"); + assert.equal(oldSlug.chapterSlug, "all-jobs-and-classes-i-just-wanted-one-skill-not"); + assert.equal(currentSlug.type, "chapter"); + assert.equal(currentSlug.seriesId, "all-jobs-and-classes-i-just-wanted-one-skill-not-them-all"); + assert.equal(currentSlug.chapterSlug, "all-jobs-and-classes-i-just-wanted-one-skill-not-them-all"); +}); + test("lightnovelworld.detect returns other for non-series paths", () => { reset(); assert.equal(lightnovelworld.detect(loc("https://lightnovelworld.net/")).type, "other");