From a4da2e2f00be77262998fce678899cf507ab9060 Mon Sep 17 00:00:00 2001 From: Sulthan Zaki Date: Tue, 11 Aug 2026 11:48:46 +0700 Subject: [PATCH 1/2] Read lnw Series identity from the chapter page pointer (#89) The lightnovelworld chapter branch no longer derives the Series address by string-manipulating the chapter path: on ~7% of novels the Chapter Slug diverges from the Series slug and the derived address 404s on every Poll. The identity now comes from the page's own a[aria-label='All Chapter'] pointer (breadcrumb's second crumb as fallback), validated as a lightnovelworld.net /novel// address; no pointer resolves to type: other. The Chapter Slug rides on the detected page object only (chapterSlug; null on a series page) and is written to no store, for the stale-row repair in #90. Harness: document.querySelector now answers attribute selectors with an element exposing getAttribute (attrEls table, cleared by reset()). AGENTS.md records that the Series address is discovered, not derived. --- userscript/AGENTS.md | 10 ++- userscript/novel-bookmark.user.js | 44 ++++++++- userscript/test/novel-logic.test.js | 133 ++++++++++++++++++++++++++++ 3 files changed, 182 insertions(+), 5 deletions(-) diff --git a/userscript/AGENTS.md b/userscript/AGENTS.md index 9966c62..98fe033 100644 --- a/userscript/AGENTS.md +++ b/userscript/AGENTS.md @@ -70,9 +70,13 @@ Guidance for OpenCode (and Claude Code) working under `userscript/`. See root `A Behind a Cloudflare JS challenge no TLS fingerprint clears, so the backend polls it through the headless browser. - **lightnovelworld.net** (novel script): series `/novel//`, chapter - `/-chapter-/` — flat, at the site root. `h1.entry-title` is the clean - title on a series page and ` Chapter <n>` on a chapter page. Its series - page lists every chapter with an + `/<slug>-chapter-<n>/` — flat, at the site root. The chapter path's slug is a + Chapter Slug, not an identity: the Series address is read off the page's + `a[aria-label='All Chapter']` (fallback: the BreadcrumbList's second crumb), + and a Series may publish under several Chapter Slugs. A chapter page with no + pointer resolves to `other`, so no Bookmark is offered. `h1.entry-title` is + the clean title on a series page and `<Title> Chapter <n>` on a chapter page. + Its series page lists every chapter with an absolute href, so the backend polls it with the plain TLS client. ### Second script: `novel-bookmark.user.js` diff --git a/userscript/novel-bookmark.user.js b/userscript/novel-bookmark.user.js index cca85f8..07480f5 100644 --- a/userscript/novel-bookmark.user.js +++ b/userscript/novel-bookmark.user.js @@ -132,6 +132,23 @@ }, }; + // A chapter page's pointer is its own link back to its Series. The href is + // page markup, so validate before trusting: the host must be this Site's (a + // leading subdomain is allowed, as in `matches`) and the path the + // /novel/<slug>/ Series shape. Anything else is not a pointer. + function seriesIdFromLnwPointer(href) { + if (!href) return null; + let u; + try { + u = new URL(href); + } catch (e) { + return null; + } + if (!/(^|\.)lightnovelworld\.net$/.test(u.hostname)) return null; + const m = u.pathname.match(/^\/novel\/([^/]+)\/?$/); + return m ? m[1] : null; + } + const lightnovelworld = { site: "lightnovelworld", matches: (loc) => /(^|\.)lightnovelworld\.net$/.test(loc.hostname), @@ -142,16 +159,38 @@ // words still resolves to the right series. let m = path.match(/^\/(.+)-chapter-([0-9]+(?:\.[0-9]+)?)\/?$/); if (m) { + // The address is not the identity on this Site: the slug in the path + // is a Chapter Slug, which can differ from the Series slug and is + // never computable from it. The Series address is read from the + // page's own pointer — a silent fallback to derivation is the defect + // this replaced, not a safety net. + const chapterSlug = m[1]; + const pointer = document.querySelector("a[aria-label='All Chapter']"); + let seriesId = pointer ? seriesIdFromLnwPointer(pointer.getAttribute("href")) : null; + if (!seriesId) { + // Fallback: the microdata breadcrumb's second crumb is the Series. + // Scoped to the BreadcrumbList because itemprop="item" is not + // unique to it (the header nav uses microdata too). + const crumb = document.querySelector( + '[itemtype="http://schema.org/BreadcrumbList"] a[itemprop="item"][href*="/novel/"]' + ); + seriesId = crumb ? seriesIdFromLnwPointer(crumb.getAttribute("href")) : null; + } + // Neither pointer present, nor either pointing at a /novel/<slug>/ + // address on this host: not a page the script understands, so no + // Bookmark under an invented identity. + if (!seriesId) return { type: "other" }; const num = parseFloat(m[2]); const h1 = document.querySelector("h1.entry-title"); const heading = h1 ? h1.textContent || "" : ""; return { type: "chapter", site: this.site, - seriesId: m[1], + seriesId: seriesId, + chapterSlug: chapterSlug, // The heading is "<Series> Chapter <n>"; drop the suffix. title: heading.replace(/\s*Chapter\s+[0-9.]+\s*$/i, "").trim(), - seriesUrl: "https://lightnovelworld.net/novel/" + m[1] + "/", + seriesUrl: "https://lightnovelworld.net/novel/" + seriesId + "/", chapterLabel: "Chapter " + m[2], chapterNum: isNaN(num) ? null : num, chapterUrl: loc.href, @@ -165,6 +204,7 @@ type: "series", site: this.site, seriesId: m[1], + chapterSlug: null, title: h1 ? (h1.textContent || "").trim() : "", seriesUrl: "https://lightnovelworld.net/novel/" + m[1] + "/", chapterLabel: null, diff --git a/userscript/test/novel-logic.test.js b/userscript/test/novel-logic.test.js index 926e8c7..86d5173 100644 --- a/userscript/test/novel-logic.test.js +++ b/userscript/test/novel-logic.test.js @@ -20,8 +20,13 @@ globalThis.location = { href: "about:blank", hostname: "", pathname: "/", origin let metaTags = {}; let elements = {}; +// Attribute selectors (the lightnovelworld Series pointer) answer with an +// element exposing getAttribute, like the meta branch below. +let attrEls = {}; globalThis.document = { querySelector(sel) { + const attr = attrEls[sel]; + if (attr != null) return attr; const m = sel.match(/^meta\[property="([^"]+)"\]$/); if (m) { const v = metaTags[m[1]]; @@ -52,6 +57,7 @@ function loc(href) { function reset() { metaTags = {}; elements = {}; + attrEls = {}; } // ============================================================ @@ -115,11 +121,18 @@ test("lightnovelworld.detect reads a series page", () => { assert.equal(p.site, "lightnovelworld"); assert.equal(p.seriesId, "a-will-eternal"); assert.equal(p.title, "A Will Eternal"); + // A series page's address *is* its identity; there is no Chapter Slug. + assert.equal(p.chapterSlug, null); }); test("lightnovelworld.detect strips the chapter suffix off the heading", () => { reset(); elements = { "h1.entry-title": "A Will Eternal Chapter 1298" }; + attrEls = { + "a[aria-label='All Chapter']": { + getAttribute: () => "https://lightnovelworld.net/novel/a-will-eternal/", + }, + }; const url = "https://lightnovelworld.net/a-will-eternal-chapter-1298/"; const p = lightnovelworld.detect(loc(url)); assert.equal(p.type, "chapter"); @@ -130,6 +143,126 @@ test("lightnovelworld.detect strips the chapter suffix off the heading", () => { assert.equal(p.seriesUrl, "https://lightnovelworld.net/novel/a-will-eternal/"); }); +test("lightnovelworld.detect reads the Series identity from the page pointer on a chapter page", () => { + reset(); + elements = { "h1.entry-title": "A Will Eternal Chapter 1298" }; + attrEls = { + // Verbatim from the real page (research note §3): the All Chapter anchor + // carries the absolute Series address. + "a[aria-label='All Chapter']": { + getAttribute: () => "https://lightnovelworld.net/novel/a-will-eternal/", + }, + }; + const url = "https://lightnovelworld.net/a-will-eternal-chapter-1298/"; + const p = lightnovelworld.detect(loc(url)); + assert.equal(p.type, "chapter"); + assert.equal(p.seriesId, "a-will-eternal"); + assert.equal(p.seriesUrl, "https://lightnovelworld.net/novel/a-will-eternal/"); + assert.equal(p.chapterSlug, "a-will-eternal"); + assert.equal(p.chapterNum, 1298); +}); + +test("lightnovelworld.detect falls back to the breadcrumb when the pointer is absent", () => { + reset(); + elements = { "h1.entry-title": "My Longevity Simulation Chapter 1" }; + attrEls = { + // Verbatim from the real page (research note §3): position 2 of the + // microdata BreadcrumbList is the Series. + '[itemtype="http://schema.org/BreadcrumbList"] a[itemprop="item"][href*="/novel/"]': { + getAttribute: () => "https://lightnovelworld.net/novel/immortality-simulator/", + }, + }; + const url = "https://lightnovelworld.net/my-longevity-simulation-chapter-1/"; + const p = lightnovelworld.detect(loc(url)); + assert.equal(p.type, "chapter"); + assert.equal(p.seriesId, "immortality-simulator"); + assert.equal(p.seriesUrl, "https://lightnovelworld.net/novel/immortality-simulator/"); + assert.equal(p.chapterSlug, "my-longevity-simulation"); +}); + +test("lightnovelworld.detect resolves to other when the page carries no pointer", () => { + reset(); + elements = { "h1.entry-title": "My Longevity Simulation Chapter 1" }; + const p = lightnovelworld.detect( + loc("https://lightnovelworld.net/my-longevity-simulation-chapter-1/") + ); + assert.equal(p.type, "other"); +}); + +test("lightnovelworld pins the divergent novel: the pointer's slug wins over the address's", () => { + // Regression pin for the derivation defect (research note §2/§3): the + // chapter address is built from "my-longevity-simulation" but the Series + // is published as "immortality-simulator". If the adapter ever derives the + // identity from the address again, this test goes red. + reset(); + elements = { "h1.entry-title": "My Longevity Simulation Chapter 1" }; + attrEls = { + "a[aria-label='All Chapter']": { + getAttribute: () => "https://lightnovelworld.net/novel/immortality-simulator/", + }, + }; + const url = "https://lightnovelworld.net/my-longevity-simulation-chapter-1/"; + const p = lightnovelworld.detect(loc(url)); + assert.equal(p.type, "chapter"); + assert.equal(p.seriesId, "immortality-simulator"); + assert.equal(p.seriesUrl, "https://lightnovelworld.net/novel/immortality-simulator/"); + assert.equal(p.chapterSlug, "my-longevity-simulation"); + assert.notEqual(p.chapterSlug, p.seriesId); +}); + +test("lightnovelworld.detect resolves a novel whose heading ends in a chapter number", () => { + // The split novel from research §4.4, chapter 200 (published under the + // current slug). Its heading ends "…Not Them All Chapter 200", which would + // false-match a selector that looks for the text "All Chapter". + reset(); + elements = { + "h1.entry-title": "All Jobs and Classes I Just Wanted One Skill Not Them All Chapter 200", + }; + attrEls = { + "a[aria-label='All Chapter']": { + getAttribute: () => + "https://lightnovelworld.net/novel/all-jobs-and-classes-i-just-wanted-one-skill-not-them-all/", + }, + }; + const url = + "https://lightnovelworld.net/all-jobs-and-classes-i-just-wanted-one-skill-not-them-all-chapter-200/"; + const p = lightnovelworld.detect(loc(url)); + assert.equal(p.type, "chapter"); + assert.equal(p.seriesId, "all-jobs-and-classes-i-just-wanted-one-skill-not-them-all"); + assert.equal(p.title, "All Jobs and Classes I Just Wanted One Skill Not Them All"); + assert.equal(p.chapterNum, 200); + assert.equal(p.chapterSlug, "all-jobs-and-classes-i-just-wanted-one-skill-not-them-all"); +}); + +test("lightnovelworld.detect resolves both Chapter Slugs of the split novel to one Series", () => { + // Research note §4.4: this Series serves chapters 1-99 under one Chapter + // Slug and 100-423 under another; both chapter addresses are live and both + // point back at the same Series. An old deep link must not get a different + // identity than a current one. + reset(); + elements = { "h1.entry-title": "All Jobs and Classes I Just Wanted One Skill Not Chapter 1" }; + attrEls = { + "a[aria-label='All Chapter']": { + getAttribute: () => + "https://lightnovelworld.net/novel/all-jobs-and-classes-i-just-wanted-one-skill-not-them-all/", + }, + }; + const oldSlug = lightnovelworld.detect( + loc("https://lightnovelworld.net/all-jobs-and-classes-i-just-wanted-one-skill-not-chapter-1/") + ); + const currentSlug = lightnovelworld.detect( + loc( + "https://lightnovelworld.net/all-jobs-and-classes-i-just-wanted-one-skill-not-them-all-chapter-404/" + ) + ); + assert.equal(oldSlug.type, "chapter"); + assert.equal(oldSlug.seriesId, "all-jobs-and-classes-i-just-wanted-one-skill-not-them-all"); + assert.equal(oldSlug.chapterSlug, "all-jobs-and-classes-i-just-wanted-one-skill-not"); + assert.equal(currentSlug.type, "chapter"); + assert.equal(currentSlug.seriesId, "all-jobs-and-classes-i-just-wanted-one-skill-not-them-all"); + assert.equal(currentSlug.chapterSlug, "all-jobs-and-classes-i-just-wanted-one-skill-not-them-all"); +}); + test("lightnovelworld.detect returns other for non-series paths", () => { reset(); assert.equal(lightnovelworld.detect(loc("https://lightnovelworld.net/")).type, "other"); From 8beaa4fd007fe23eab866f795ab9ee01ed616153 Mon Sep 17 00:00:00 2001 From: Sulthan Zaki <sultankiki05@gmail.com> Date: Tue, 11 Aug 2026 11:52:31 +0700 Subject: [PATCH 2/2] Address review findings on the lnw pointer read (#89) - Share one lnw host regex between matches and the pointer validator so the two cannot drift apart. - Resolve the pointer href against the page address before validating, so a relative pointer still yields its Series instead of silently resolving to type other. - Pin pointer/breadcrumb agreement on the divergent page and pin that a pointer off-host or not /novel/<slug>/ resolves to other. --- userscript/novel-bookmark.user.js | 22 ++++++----- userscript/test/novel-logic.test.js | 58 +++++++++++++++++++++++++---- 2 files changed, 64 insertions(+), 16 deletions(-) diff --git a/userscript/novel-bookmark.user.js b/userscript/novel-bookmark.user.js index 07480f5..3d4884d 100644 --- a/userscript/novel-bookmark.user.js +++ b/userscript/novel-bookmark.user.js @@ -132,26 +132,30 @@ }, }; + // The host shape this adapter owns. One definition so the pointer validator + // and `matches` cannot drift apart (a leading subdomain is allowed). + const lnwHostRe = /(^|\.)lightnovelworld\.net$/; + // A chapter page's pointer is its own link back to its Series. The href is - // page markup, so validate before trusting: the host must be this Site's (a - // leading subdomain is allowed, as in `matches`) and the path the - // /novel/<slug>/ Series shape. Anything else is not a pointer. - function seriesIdFromLnwPointer(href) { + // page markup, so validate before trusting: resolve it against the page + // address, require the host above and the /novel/<slug>/ Series path. + // Anything else is not a pointer. + function seriesIdFromLnwPointer(href, base) { if (!href) return null; let u; try { - u = new URL(href); + u = new URL(href, base); } catch (e) { return null; } - if (!/(^|\.)lightnovelworld\.net$/.test(u.hostname)) return null; + if (!lnwHostRe.test(u.hostname)) return null; const m = u.pathname.match(/^\/novel\/([^/]+)\/?$/); return m ? m[1] : null; } const lightnovelworld = { site: "lightnovelworld", - matches: (loc) => /(^|\.)lightnovelworld\.net$/.test(loc.hostname), + matches: (loc) => lnwHostRe.test(loc.hostname), detect(loc) { const path = loc.pathname; // /<slug>-chapter-<n>/ — flat, at the site root, not under /novel/. The @@ -166,7 +170,7 @@ // this replaced, not a safety net. const chapterSlug = m[1]; const pointer = document.querySelector("a[aria-label='All Chapter']"); - let seriesId = pointer ? seriesIdFromLnwPointer(pointer.getAttribute("href")) : null; + let seriesId = pointer ? seriesIdFromLnwPointer(pointer.getAttribute("href"), loc.href) : null; if (!seriesId) { // Fallback: the microdata breadcrumb's second crumb is the Series. // Scoped to the BreadcrumbList because itemprop="item" is not @@ -174,7 +178,7 @@ const crumb = document.querySelector( '[itemtype="http://schema.org/BreadcrumbList"] a[itemprop="item"][href*="/novel/"]' ); - seriesId = crumb ? seriesIdFromLnwPointer(crumb.getAttribute("href")) : null; + seriesId = crumb ? seriesIdFromLnwPointer(crumb.getAttribute("href"), loc.href) : null; } // Neither pointer present, nor either pointing at a /novel/<slug>/ // address on this host: not a page the script understands, so no diff --git a/userscript/test/novel-logic.test.js b/userscript/test/novel-logic.test.js index 86d5173..9b8ef5b 100644 --- a/userscript/test/novel-logic.test.js +++ b/userscript/test/novel-logic.test.js @@ -165,19 +165,63 @@ test("lightnovelworld.detect reads the Series identity from the page pointer on test("lightnovelworld.detect falls back to the breadcrumb when the pointer is absent", () => { reset(); elements = { "h1.entry-title": "My Longevity Simulation Chapter 1" }; + const url = "https://lightnovelworld.net/my-longevity-simulation-chapter-1/"; + // Same divergent page, both pointers from the real markup (research note + // §3): breadcrumb position 2 must say what the All Chapter anchor says. + attrEls = { + "a[aria-label='All Chapter']": { + getAttribute: () => "https://lightnovelworld.net/novel/immortality-simulator/", + }, + }; + const viaPointer = lightnovelworld.detect(loc(url)); attrEls = { - // Verbatim from the real page (research note §3): position 2 of the - // microdata BreadcrumbList is the Series. '[itemtype="http://schema.org/BreadcrumbList"] a[itemprop="item"][href*="/novel/"]': { getAttribute: () => "https://lightnovelworld.net/novel/immortality-simulator/", }, }; - const url = "https://lightnovelworld.net/my-longevity-simulation-chapter-1/"; - const p = lightnovelworld.detect(loc(url)); + const viaBreadcrumb = lightnovelworld.detect(loc(url)); + assert.equal(viaBreadcrumb.type, "chapter"); + assert.equal(viaBreadcrumb.seriesId, viaPointer.seriesId); + assert.equal(viaBreadcrumb.seriesUrl, viaPointer.seriesUrl); + assert.equal(viaBreadcrumb.seriesId, "immortality-simulator"); + assert.equal(viaBreadcrumb.chapterSlug, "my-longevity-simulation"); +}); + +test("lightnovelworld.detect resolves a relative pointer against the page address", () => { + // Defensive: every measured page ships an absolute pointer href, but a + // theme change could go relative — the pointer is still this page's own + // link back to its Series. + reset(); + elements = { "h1.entry-title": "A Will Eternal Chapter 1298" }; + attrEls = { + "a[aria-label='All Chapter']": { getAttribute: () => "/novel/a-will-eternal/" }, + }; + const p = lightnovelworld.detect(loc("https://lightnovelworld.net/a-will-eternal-chapter-1298/")); assert.equal(p.type, "chapter"); - assert.equal(p.seriesId, "immortality-simulator"); - assert.equal(p.seriesUrl, "https://lightnovelworld.net/novel/immortality-simulator/"); - assert.equal(p.chapterSlug, "my-longevity-simulation"); + assert.equal(p.seriesId, "a-will-eternal"); + assert.equal(p.seriesUrl, "https://lightnovelworld.net/novel/a-will-eternal/"); +}); + +test("lightnovelworld.detect rejects a pointer that is not a /novel/ address on its own host", () => { + // Counterfactual pointers, exercising the criterion that a pointer "present + // but not parseable as /novel/<slug>/ on lightnovelworld.net" must not + // become a seriesUrl the backend is later asked to Poll: off-host, and + // on-host but the wrong path shape. + reset(); + elements = { "h1.entry-title": "My Longevity Simulation Chapter 1" }; + const url = "https://lightnovelworld.net/my-longevity-simulation-chapter-1/"; + attrEls = { + "a[aria-label='All Chapter']": { + getAttribute: () => "https://evil.example/novel/immortality-simulator/", + }, + }; + assert.equal(lightnovelworld.detect(loc(url)).type, "other"); + attrEls = { + "a[aria-label='All Chapter']": { + getAttribute: () => "https://lightnovelworld.net/fiction/immortality-simulator/", + }, + }; + assert.equal(lightnovelworld.detect(loc(url)).type, "other"); }); test("lightnovelworld.detect resolves to other when the page carries no pointer", () => {