Read lnw Series identity from the chapter page pointer (#89)

The lightnovelworld chapter branch no longer derives the Series address by
string-manipulating the chapter path: on ~7% of novels the Chapter Slug
diverges from the Series slug and the derived address 404s on every Poll.
The identity now comes from the page's own a[aria-label='All Chapter']
pointer (breadcrumb's second crumb as fallback), validated as a
lightnovelworld.net /novel/<slug>/ address; no pointer resolves to
type: other. The Chapter Slug rides on the detected page object only
(chapterSlug; null on a series page) and is written to no store, for the
stale-row repair in #90.

Harness: document.querySelector now answers attribute selectors with an
element exposing getAttribute (attrEls table, cleared by reset()).
AGENTS.md records that the Series address is discovered, not derived.
This commit is contained in:
2026-08-11 11:48:46 +07:00
parent c400c91a80
commit a4da2e2f00
3 changed files with 182 additions and 5 deletions
+7 -3
View File
@@ -70,9 +70,13 @@ Guidance for OpenCode (and Claude Code) working under `userscript/`. See root `A
Behind a Cloudflare JS challenge no TLS fingerprint Behind a Cloudflare JS challenge no TLS fingerprint
clears, so the backend polls it through the headless browser. clears, so the backend polls it through the headless browser.
- **lightnovelworld.net** (novel script): series `/novel/<slug>/`, chapter - **lightnovelworld.net** (novel script): series `/novel/<slug>/`, chapter
`/<slug>-chapter-<n>/` — flat, at the site root. `h1.entry-title` is the clean `/<slug>-chapter-<n>/` — flat, at the site root. The chapter path's slug is a
title on a series page and `<Title> Chapter <n>` on a chapter page. Its series Chapter Slug, not an identity: the Series address is read off the page's
page lists every chapter with an `a[aria-label='All Chapter']` (fallback: the BreadcrumbList's second crumb),
and a Series may publish under several Chapter Slugs. A chapter page with no
pointer resolves to `other`, so no Bookmark is offered. `h1.entry-title` is
the clean title on a series page and `<Title> Chapter <n>` on a chapter page.
Its series page lists every chapter with an
absolute href, so the backend polls it with the plain TLS client. absolute href, so the backend polls it with the plain TLS client.
### Second script: `novel-bookmark.user.js` ### Second script: `novel-bookmark.user.js`
+42 -2
View File
@@ -132,6 +132,23 @@
}, },
}; };
// A chapter page's pointer is its own link back to its Series. The href is
// page markup, so validate before trusting: the host must be this Site's (a
// leading subdomain is allowed, as in `matches`) and the path the
// /novel/<slug>/ Series shape. Anything else is not a pointer.
function seriesIdFromLnwPointer(href) {
if (!href) return null;
let u;
try {
u = new URL(href);
} catch (e) {
return null;
}
if (!/(^|\.)lightnovelworld\.net$/.test(u.hostname)) return null;
const m = u.pathname.match(/^\/novel\/([^/]+)\/?$/);
return m ? m[1] : null;
}
const lightnovelworld = { const lightnovelworld = {
site: "lightnovelworld", site: "lightnovelworld",
matches: (loc) => /(^|\.)lightnovelworld\.net$/.test(loc.hostname), matches: (loc) => /(^|\.)lightnovelworld\.net$/.test(loc.hostname),
@@ -142,16 +159,38 @@
// words still resolves to the right series. // words still resolves to the right series.
let m = path.match(/^\/(.+)-chapter-([0-9]+(?:\.[0-9]+)?)\/?$/); let m = path.match(/^\/(.+)-chapter-([0-9]+(?:\.[0-9]+)?)\/?$/);
if (m) { if (m) {
// The address is not the identity on this Site: the slug in the path
// is a Chapter Slug, which can differ from the Series slug and is
// never computable from it. The Series address is read from the
// page's own pointer — a silent fallback to derivation is the defect
// this replaced, not a safety net.
const chapterSlug = m[1];
const pointer = document.querySelector("a[aria-label='All Chapter']");
let seriesId = pointer ? seriesIdFromLnwPointer(pointer.getAttribute("href")) : null;
if (!seriesId) {
// Fallback: the microdata breadcrumb's second crumb is the Series.
// Scoped to the BreadcrumbList because itemprop="item" is not
// unique to it (the header nav uses microdata too).
const crumb = document.querySelector(
'[itemtype="http://schema.org/BreadcrumbList"] a[itemprop="item"][href*="/novel/"]'
);
seriesId = crumb ? seriesIdFromLnwPointer(crumb.getAttribute("href")) : null;
}
// Neither pointer present, nor either pointing at a /novel/<slug>/
// address on this host: not a page the script understands, so no
// Bookmark under an invented identity.
if (!seriesId) return { type: "other" };
const num = parseFloat(m[2]); const num = parseFloat(m[2]);
const h1 = document.querySelector("h1.entry-title"); const h1 = document.querySelector("h1.entry-title");
const heading = h1 ? h1.textContent || "" : ""; const heading = h1 ? h1.textContent || "" : "";
return { return {
type: "chapter", type: "chapter",
site: this.site, site: this.site,
seriesId: m[1], seriesId: seriesId,
chapterSlug: chapterSlug,
// The heading is "<Series> Chapter <n>"; drop the suffix. // The heading is "<Series> Chapter <n>"; drop the suffix.
title: heading.replace(/\s*Chapter\s+[0-9.]+\s*$/i, "").trim(), title: heading.replace(/\s*Chapter\s+[0-9.]+\s*$/i, "").trim(),
seriesUrl: "https://lightnovelworld.net/novel/" + m[1] + "/", seriesUrl: "https://lightnovelworld.net/novel/" + seriesId + "/",
chapterLabel: "Chapter " + m[2], chapterLabel: "Chapter " + m[2],
chapterNum: isNaN(num) ? null : num, chapterNum: isNaN(num) ? null : num,
chapterUrl: loc.href, chapterUrl: loc.href,
@@ -165,6 +204,7 @@
type: "series", type: "series",
site: this.site, site: this.site,
seriesId: m[1], seriesId: m[1],
chapterSlug: null,
title: h1 ? (h1.textContent || "").trim() : "", title: h1 ? (h1.textContent || "").trim() : "",
seriesUrl: "https://lightnovelworld.net/novel/" + m[1] + "/", seriesUrl: "https://lightnovelworld.net/novel/" + m[1] + "/",
chapterLabel: null, chapterLabel: null,
+133
View File
@@ -20,8 +20,13 @@ globalThis.location = { href: "about:blank", hostname: "", pathname: "/", origin
let metaTags = {}; let metaTags = {};
let elements = {}; let elements = {};
// Attribute selectors (the lightnovelworld Series pointer) answer with an
// element exposing getAttribute, like the meta branch below.
let attrEls = {};
globalThis.document = { globalThis.document = {
querySelector(sel) { querySelector(sel) {
const attr = attrEls[sel];
if (attr != null) return attr;
const m = sel.match(/^meta\[property="([^"]+)"\]$/); const m = sel.match(/^meta\[property="([^"]+)"\]$/);
if (m) { if (m) {
const v = metaTags[m[1]]; const v = metaTags[m[1]];
@@ -52,6 +57,7 @@ function loc(href) {
function reset() { function reset() {
metaTags = {}; metaTags = {};
elements = {}; elements = {};
attrEls = {};
} }
// ============================================================ // ============================================================
@@ -115,11 +121,18 @@ test("lightnovelworld.detect reads a series page", () => {
assert.equal(p.site, "lightnovelworld"); assert.equal(p.site, "lightnovelworld");
assert.equal(p.seriesId, "a-will-eternal"); assert.equal(p.seriesId, "a-will-eternal");
assert.equal(p.title, "A Will Eternal"); assert.equal(p.title, "A Will Eternal");
// A series page's address *is* its identity; there is no Chapter Slug.
assert.equal(p.chapterSlug, null);
}); });
test("lightnovelworld.detect strips the chapter suffix off the heading", () => { test("lightnovelworld.detect strips the chapter suffix off the heading", () => {
reset(); reset();
elements = { "h1.entry-title": "A Will Eternal Chapter 1298" }; elements = { "h1.entry-title": "A Will Eternal Chapter 1298" };
attrEls = {
"a[aria-label='All Chapter']": {
getAttribute: () => "https://lightnovelworld.net/novel/a-will-eternal/",
},
};
const url = "https://lightnovelworld.net/a-will-eternal-chapter-1298/"; const url = "https://lightnovelworld.net/a-will-eternal-chapter-1298/";
const p = lightnovelworld.detect(loc(url)); const p = lightnovelworld.detect(loc(url));
assert.equal(p.type, "chapter"); assert.equal(p.type, "chapter");
@@ -130,6 +143,126 @@ test("lightnovelworld.detect strips the chapter suffix off the heading", () => {
assert.equal(p.seriesUrl, "https://lightnovelworld.net/novel/a-will-eternal/"); assert.equal(p.seriesUrl, "https://lightnovelworld.net/novel/a-will-eternal/");
}); });
test("lightnovelworld.detect reads the Series identity from the page pointer on a chapter page", () => {
reset();
elements = { "h1.entry-title": "A Will Eternal Chapter 1298" };
attrEls = {
// Verbatim from the real page (research note §3): the All Chapter anchor
// carries the absolute Series address.
"a[aria-label='All Chapter']": {
getAttribute: () => "https://lightnovelworld.net/novel/a-will-eternal/",
},
};
const url = "https://lightnovelworld.net/a-will-eternal-chapter-1298/";
const p = lightnovelworld.detect(loc(url));
assert.equal(p.type, "chapter");
assert.equal(p.seriesId, "a-will-eternal");
assert.equal(p.seriesUrl, "https://lightnovelworld.net/novel/a-will-eternal/");
assert.equal(p.chapterSlug, "a-will-eternal");
assert.equal(p.chapterNum, 1298);
});
test("lightnovelworld.detect falls back to the breadcrumb when the pointer is absent", () => {
reset();
elements = { "h1.entry-title": "My Longevity Simulation Chapter 1" };
attrEls = {
// Verbatim from the real page (research note §3): position 2 of the
// microdata BreadcrumbList is the Series.
'[itemtype="http://schema.org/BreadcrumbList"] a[itemprop="item"][href*="/novel/"]': {
getAttribute: () => "https://lightnovelworld.net/novel/immortality-simulator/",
},
};
const url = "https://lightnovelworld.net/my-longevity-simulation-chapter-1/";
const p = lightnovelworld.detect(loc(url));
assert.equal(p.type, "chapter");
assert.equal(p.seriesId, "immortality-simulator");
assert.equal(p.seriesUrl, "https://lightnovelworld.net/novel/immortality-simulator/");
assert.equal(p.chapterSlug, "my-longevity-simulation");
});
test("lightnovelworld.detect resolves to other when the page carries no pointer", () => {
reset();
elements = { "h1.entry-title": "My Longevity Simulation Chapter 1" };
const p = lightnovelworld.detect(
loc("https://lightnovelworld.net/my-longevity-simulation-chapter-1/")
);
assert.equal(p.type, "other");
});
test("lightnovelworld pins the divergent novel: the pointer's slug wins over the address's", () => {
// Regression pin for the derivation defect (research note §2/§3): the
// chapter address is built from "my-longevity-simulation" but the Series
// is published as "immortality-simulator". If the adapter ever derives the
// identity from the address again, this test goes red.
reset();
elements = { "h1.entry-title": "My Longevity Simulation Chapter 1" };
attrEls = {
"a[aria-label='All Chapter']": {
getAttribute: () => "https://lightnovelworld.net/novel/immortality-simulator/",
},
};
const url = "https://lightnovelworld.net/my-longevity-simulation-chapter-1/";
const p = lightnovelworld.detect(loc(url));
assert.equal(p.type, "chapter");
assert.equal(p.seriesId, "immortality-simulator");
assert.equal(p.seriesUrl, "https://lightnovelworld.net/novel/immortality-simulator/");
assert.equal(p.chapterSlug, "my-longevity-simulation");
assert.notEqual(p.chapterSlug, p.seriesId);
});
test("lightnovelworld.detect resolves a novel whose heading ends in a chapter number", () => {
// The split novel from research §4.4, chapter 200 (published under the
// current slug). Its heading ends "…Not Them All Chapter 200", which would
// false-match a selector that looks for the text "All Chapter".
reset();
elements = {
"h1.entry-title": "All Jobs and Classes I Just Wanted One Skill Not Them All Chapter 200",
};
attrEls = {
"a[aria-label='All Chapter']": {
getAttribute: () =>
"https://lightnovelworld.net/novel/all-jobs-and-classes-i-just-wanted-one-skill-not-them-all/",
},
};
const url =
"https://lightnovelworld.net/all-jobs-and-classes-i-just-wanted-one-skill-not-them-all-chapter-200/";
const p = lightnovelworld.detect(loc(url));
assert.equal(p.type, "chapter");
assert.equal(p.seriesId, "all-jobs-and-classes-i-just-wanted-one-skill-not-them-all");
assert.equal(p.title, "All Jobs and Classes I Just Wanted One Skill Not Them All");
assert.equal(p.chapterNum, 200);
assert.equal(p.chapterSlug, "all-jobs-and-classes-i-just-wanted-one-skill-not-them-all");
});
test("lightnovelworld.detect resolves both Chapter Slugs of the split novel to one Series", () => {
// Research note §4.4: this Series serves chapters 1-99 under one Chapter
// Slug and 100-423 under another; both chapter addresses are live and both
// point back at the same Series. An old deep link must not get a different
// identity than a current one.
reset();
elements = { "h1.entry-title": "All Jobs and Classes I Just Wanted One Skill Not Chapter 1" };
attrEls = {
"a[aria-label='All Chapter']": {
getAttribute: () =>
"https://lightnovelworld.net/novel/all-jobs-and-classes-i-just-wanted-one-skill-not-them-all/",
},
};
const oldSlug = lightnovelworld.detect(
loc("https://lightnovelworld.net/all-jobs-and-classes-i-just-wanted-one-skill-not-chapter-1/")
);
const currentSlug = lightnovelworld.detect(
loc(
"https://lightnovelworld.net/all-jobs-and-classes-i-just-wanted-one-skill-not-them-all-chapter-404/"
)
);
assert.equal(oldSlug.type, "chapter");
assert.equal(oldSlug.seriesId, "all-jobs-and-classes-i-just-wanted-one-skill-not-them-all");
assert.equal(oldSlug.chapterSlug, "all-jobs-and-classes-i-just-wanted-one-skill-not");
assert.equal(currentSlug.type, "chapter");
assert.equal(currentSlug.seriesId, "all-jobs-and-classes-i-just-wanted-one-skill-not-them-all");
assert.equal(currentSlug.chapterSlug, "all-jobs-and-classes-i-just-wanted-one-skill-not-them-all");
});
test("lightnovelworld.detect returns other for non-series paths", () => { test("lightnovelworld.detect returns other for non-series paths", () => {
reset(); reset();
assert.equal(lightnovelworld.detect(loc("https://lightnovelworld.net/")).type, "other"); assert.equal(lightnovelworld.detect(loc("https://lightnovelworld.net/")).type, "other");