Read lnw Series identity from the chapter page pointer (#89)

The lightnovelworld chapter branch no longer derives the Series address by
string-manipulating the chapter path: on ~7% of novels the Chapter Slug
diverges from the Series slug and the derived address 404s on every Poll.
The identity now comes from the page's own a[aria-label='All Chapter']
pointer (breadcrumb's second crumb as fallback), validated as a
lightnovelworld.net /novel/<slug>/ address; no pointer resolves to
type: other. The Chapter Slug rides on the detected page object only
(chapterSlug; null on a series page) and is written to no store, for the
stale-row repair in #90.

Harness: document.querySelector now answers attribute selectors with an
element exposing getAttribute (attrEls table, cleared by reset()).
AGENTS.md records that the Series address is discovered, not derived.
This commit is contained in:
2026-08-11 11:48:46 +07:00
parent c400c91a80
commit a4da2e2f00
3 changed files with 182 additions and 5 deletions
+42 -2
View File
@@ -132,6 +132,23 @@
},
};
// A chapter page's pointer is its own link back to its Series. The href is
// page markup, so validate before trusting: the host must be this Site's (a
// leading subdomain is allowed, as in `matches`) and the path the
// /novel/<slug>/ Series shape. Anything else is not a pointer.
function seriesIdFromLnwPointer(href) {
if (!href) return null;
let u;
try {
u = new URL(href);
} catch (e) {
return null;
}
if (!/(^|\.)lightnovelworld\.net$/.test(u.hostname)) return null;
const m = u.pathname.match(/^\/novel\/([^/]+)\/?$/);
return m ? m[1] : null;
}
const lightnovelworld = {
site: "lightnovelworld",
matches: (loc) => /(^|\.)lightnovelworld\.net$/.test(loc.hostname),
@@ -142,16 +159,38 @@
// words still resolves to the right series.
let m = path.match(/^\/(.+)-chapter-([0-9]+(?:\.[0-9]+)?)\/?$/);
if (m) {
// The address is not the identity on this Site: the slug in the path
// is a Chapter Slug, which can differ from the Series slug and is
// never computable from it. The Series address is read from the
// page's own pointer — a silent fallback to derivation is the defect
// this replaced, not a safety net.
const chapterSlug = m[1];
const pointer = document.querySelector("a[aria-label='All Chapter']");
let seriesId = pointer ? seriesIdFromLnwPointer(pointer.getAttribute("href")) : null;
if (!seriesId) {
// Fallback: the microdata breadcrumb's second crumb is the Series.
// Scoped to the BreadcrumbList because itemprop="item" is not
// unique to it (the header nav uses microdata too).
const crumb = document.querySelector(
'[itemtype="http://schema.org/BreadcrumbList"] a[itemprop="item"][href*="/novel/"]'
);
seriesId = crumb ? seriesIdFromLnwPointer(crumb.getAttribute("href")) : null;
}
// Neither pointer present, nor either pointing at a /novel/<slug>/
// address on this host: not a page the script understands, so no
// Bookmark under an invented identity.
if (!seriesId) return { type: "other" };
const num = parseFloat(m[2]);
const h1 = document.querySelector("h1.entry-title");
const heading = h1 ? h1.textContent || "" : "";
return {
type: "chapter",
site: this.site,
seriesId: m[1],
seriesId: seriesId,
chapterSlug: chapterSlug,
// The heading is "<Series> Chapter <n>"; drop the suffix.
title: heading.replace(/\s*Chapter\s+[0-9.]+\s*$/i, "").trim(),
seriesUrl: "https://lightnovelworld.net/novel/" + m[1] + "/",
seriesUrl: "https://lightnovelworld.net/novel/" + seriesId + "/",
chapterLabel: "Chapter " + m[2],
chapterNum: isNaN(num) ? null : num,
chapterUrl: loc.href,
@@ -165,6 +204,7 @@
type: "series",
site: this.site,
seriesId: m[1],
chapterSlug: null,
title: h1 ? (h1.textContent || "").trim() : "",
seriesUrl: "https://lightnovelworld.net/novel/" + m[1] + "/",
chapterLabel: null,