diff --git a/backend/internal/latest/browser.go b/backend/internal/latest/browser.go index 0d68292..1227551 100644 --- a/backend/internal/latest/browser.go +++ b/backend/internal/latest/browser.go @@ -86,46 +86,58 @@ func (f *BrowserFetcher) Close() { f.cancel() } -// Get navigates to seriesURL, lets any challenge resolve, then reads either the -// site's JSON API (kagane) from inside the page so the request carries the -// clearance cookie, or the served HTML itself (novelfull) — see -// novelfullSeriesURL for the latter case. The returned body is whatever the -// site's chapter list lives in, which is what latestChapterFrom's per-site -// switch expects. +// Get navigates to seriesURL, lets any challenge resolve, then reads the +// payload the Site's registry entry describes — kagane's chapter-list API from +// inside the page so the request carries the clearance cookie, novelfull's +// served HTML. The returned body is whatever the Site's chapter list lives in, +// which is what the entry's LatestChapter parse expects. func (f *BrowserFetcher) Get(ctx context.Context, seriesURL string) (string, int, error) { - apiURL, isKagane := kaganeAPIURL(seriesURL) - if !isKagane && !novelfullSeriesURL(seriesURL) { - return "", 0, fmt.Errorf("not a fetchable browser series url: %q", seriesURL) - } - var body string - // kagane's chapter list is only in its JSON API, which must be called from - // inside the page so the request carries the clearance cookie. novelfull - // renders its chapters into the HTML, so the cleared DOM is the answer. - // chromedp.OuterHTML returns a QueryAction and chromedp.Evaluate an - // EvaluateAction, so the variable has to be the interface both implement. - var read chromedp.Action = chromedp.OuterHTML("html", &body, chromedp.ByQuery) - if isKagane { - read = chromedp.Evaluate( - `fetch(`+jsString(apiURL)+`).then(r => r.ok ? r.text() : "")`, - &body, - awaitPromise, - ) - } - - // novelfull's payload is the DOM itself, and the interstitial has a DOM - // too, so "we have an answer" has to exclude it explicitly. kagane's - // in-page fetch just fails while challenged, which is already the signal. - done := func() bool { return body != "" && (isKagane || !isInterstitial(body)) } - if err := f.run(ctx, seriesURL, read, done); err != nil { - // Challenge never cleared, or the API refused. Indistinguishable from - // here and handled identically by the caller. - if errors.Is(err, errChallengeHeld) { - return "", 403, nil + for _, s := range sites { + if s.Browser == nil { + continue } - return "", 0, fmt.Errorf("browser fetch %q: %w", seriesURL, err) + read, ok := s.Browser.Read(seriesURL, &body) + if !ok { + continue + } + if err := f.run(ctx, seriesURL, read, + func() bool { return s.Browser.Done(body) }); err != nil { + // Challenge never cleared, or the payload was refused. + // Indistinguishable from here and handled identically by the caller. + if errors.Is(err, errChallengeHeld) { + return "", 403, nil + } + return "", 0, fmt.Errorf("browser fetch %q: %w", seriesURL, err) + } + return body, 200, nil } - return body, 200, nil + return "", 0, fmt.Errorf("not a fetchable browser series url: %q", seriesURL) +} + +// kaganeRead builds the in-tab fetch of kagane's chapter-list API: the +// request must be made from inside the page so it carries the clearance +// cookie, and the API is the only place the list exists. Refusing any other +// address is the per-Site half of the SSRF gate, kept deliberately behind +// fetchableSeriesURL (see browserRead.Read). +func kaganeRead(seriesURL string, out *string) (chromedp.Action, bool) { + apiURL, ok := kaganeAPIURL(seriesURL) + if !ok { + return nil, false + } + return chromedp.Evaluate( + `fetch(`+jsString(apiURL)+`).then(r => r.ok ? r.text() : "")`, + out, awaitPromise), true +} + +// novelfullRead reads the cleared DOM. novelfull renders its chapter list +// into the served HTML, so there is no API to call from inside the page — the +// challenge-cleared DOM is the payload. +func novelfullRead(seriesURL string, out *string) (chromedp.Action, bool) { + if !novelfullSeriesURL(seriesURL) { + return nil, false + } + return chromedp.OuterHTML("html", out, chromedp.ByQuery), true } // Image retrieves one cover's bytes through the browser sidecar, and its diff --git a/backend/internal/latest/poller.go b/backend/internal/latest/poller.go index 7578f87..8adab5c 100644 --- a/backend/internal/latest/poller.go +++ b/backend/internal/latest/poller.go @@ -4,7 +4,6 @@ import ( "context" "log" "net/url" - "slices" "time" "bookmarkmanager/backend/internal/store" @@ -57,8 +56,6 @@ type Poller struct { Batch int } -var browserBackedSites = []string{"kagane", "novelfull"} - // fillBlankCover gives a Series its Cover when it has none. The blank state is // what "no Cover yet" means on the wire (ADR-0007): permanently-blank rows // created before acquisition existed, and rows whose creation-time fetch @@ -120,27 +117,23 @@ func (p *Poller) storeCover(ctx context.Context, sr store.Series, sourceURL stri } // fetcherFor returns the fetcher a site's page needs, or nil when the site -// cannot be fetched at all right now. kagane and novelfull pages sit behind a -// Cloudflare JavaScript challenge that no TLS fingerprint clears (kagane -// verified 2026-08-03, novelfull verified 2026-08-05, both against the same -// Chrome_133 profile TLSFetcher uses), so both prefer the browser; novelfull -// alone falls back to the plain-TLS fetcher when no browser is configured, -// because its challenge is a live time-varying fact (AGENTS.md) and its cover -// bytes never need the browser. kagane never falls back: a plain fetch of a -// kagane page or cover would only ever retrieve a challenge page. One routing -// rule for the poll and the acquirer, so the two cannot drift apart. +// cannot be fetched at all right now. A Site whose registry entry carries a +// Browser read — kagane and novelfull, both behind a Cloudflare JavaScript +// challenge no TLS fingerprint clears — prefers the browser; when it is +// absent, the entry's Fallback decides whether plain TLS may take over. One +// routing rule for the poll and the acquirer, so the two cannot drift apart. func fetcherFor(site string, browser, tls Fetcher) Fetcher { - switch { - case site == "kagane": - return browser - case slices.Contains(browserBackedSites, site): // novelfull - if browser != nil { - return browser - } - return tls - default: + s := sites[site] + if s.Browser == nil { return tls } + if browser != nil { + return browser + } + if s.Browser.Fallback { + return tls + } + return nil } // Run polls until ctx is cancelled. @@ -170,7 +163,7 @@ func (p *Poller) runOnce(ctx context.Context) { now := p.Now() cutoff := now.Add(-p.Cooldown).UnixMilli() browserCutoff := now.Add(-p.BrowserCooldown).UnixMilli() - due, err := p.Store.DueForLatestCheck(cutoff, browserCutoff, browserBackedSites, p.Batch) + due, err := p.Store.DueForLatestCheck(cutoff, browserCutoff, browserBackedSites(), p.Batch) if err != nil { log.Printf("latest poll: due query: %v", err) return @@ -283,45 +276,23 @@ func (p *Poller) checkOne(ctx context.Context, sr store.Series) { log.Printf("latest poll %q: latest is now %s", sr.Key(), latest.Label) } -// fetchableSeriesURL reports whether site is a site latestChapterFrom knows how -// to parse and seriesURL is safe to hand to a fetcher: an https URL with a -// non-empty host. series_url comes from client-supplied PUT bodies, so this is -// a defence against the poller being used to probe arbitrary hosts from the -// server's own network position, not just a check against wasted requests. -// -// Three sites are held to a stricter rule, each for a different reason: -// -// - kagane and novelfull are fetched by a headless browser, which executes -// JavaScript and carries cookies, and is therefore a far stronger SSRF -// primitive than an HTTP GET. Their hosts must match exactly, not merely -// be non-empty. -// - lightnovelworld's parser regex hardcodes its host, so a URL anywhere -// else could never yield a match — reject it here rather than burn the -// request. +// fetchableSeriesURL reports whether site is a Site the registry knows and +// seriesURL is safe to hand to a fetcher: an https URL whose host matches the +// Site's pinned hostname exactly. series_url comes from client-supplied PUT +// bodies, so this is a defence against the poller being used to probe +// arbitrary hosts from the server's own network position, not just a check +// against wasted requests. The pin guards different things per Site — a +// browser Site guards a control that executes JavaScript and carries cookies, +// a parser Site guards a wasted request — but the rule is one rule, from the +// registry. func fetchableSeriesURL(site, seriesURL string) bool { - switch site { - case "asura", "demonic", "comix", "kagane", "novelfull", "lightnovelworld": - default: + s, known := sites[site] + if !known { return false } u, err := url.Parse(seriesURL) if err != nil { return false } - if u.Scheme != "https" || u.Host == "" { - return false - } - switch site { - case "kagane": - return u.Hostname() == "kagane.to" - case "novelfull": - // Fetched by a real browser, same as kagane, so the host is pinned - // rather than merely non-empty. - return u.Hostname() == "novelfull.com" - case "lightnovelworld": - // Its parser regex hardcodes this host, so a URL anywhere else could - // never yield a match — reject it here rather than burn the request. - return u.Hostname() == "lightnovelworld.net" - } - return true + return u.Scheme == "https" && u.Hostname() == s.Host } diff --git a/backend/internal/latest/poller_test.go b/backend/internal/latest/poller_test.go index 59b3cf8..8b38fa0 100644 --- a/backend/internal/latest/poller_test.go +++ b/backend/internal/latest/poller_test.go @@ -623,6 +623,13 @@ func TestFetchableSeriesURL(t *testing.T) { {"demonic https", "demonic", "https://demonicscans.org/manga/X", true}, {"comix https", "comix", "https://comix.to/title/n8we-dungeons-and-crayons", true}, {"kagane on its own host", "kagane", "https://kagane.to/series/019f84bc-9ba0-7ed9-86f5-8b905ec7c28b", true}, + // The pins added in #94 cover the three plain-TLS Sites too: a + // client-supplied series_url must not aim a fetcher at a lookalike + // host, even when the fetcher is only an HTTP GET. + {"asura on a foreign host", "asura", "https://asurascans.com.evil.example/comics/x", false}, + {"asura on the dead old domain", "asura", "https://asuracomic.net/comics/x", false}, + {"demonic on a lookalike host", "demonic", "https://demonicscans.org.evil.example/manga/X", false}, + {"comix on a foreign host", "comix", "https://evil.example/title/x", false}, // The browser fetcher runs JavaScript and carries cookies, so a // client-supplied series_url must not be able to aim it anywhere else. {"kagane on a foreign host", "kagane", "https://evil.example/series/x", false}, diff --git a/backend/internal/latest/sites.go b/backend/internal/latest/sites.go index 0c354ec..702233f 100644 --- a/backend/internal/latest/sites.go +++ b/backend/internal/latest/sites.go @@ -8,6 +8,8 @@ import ( "regexp" "strconv" "strings" + + "github.com/chromedp/chromedp" ) // latestChapter is the newest chapter a series page advertises. @@ -16,6 +18,40 @@ type latestChapter struct { Label string } +// site answers the fixed questions every series-page read asks of its Site +// (ADR-0009): the host its addresses must carry, how to find the Latest +// Chapter and the Cover address in a body, and — for a Site behind a +// JavaScript challenge — how to read its payload from a cleared tab. One +// entry describes everything about one Site, and nowhere else gets to compare +// the site string. +type site struct { + // Host is the exact hostname a series_url for this Site must carry. + Host string + // LatestChapter finds the newest chapter in a fetched body. + LatestChapter func(seriesURL, body string) (latestChapter, bool) + // Cover finds the Cover address in a fetched body. + Cover func(seriesURL, body string) (string, bool) + // Browser reads this Site's payload from a cleared browser tab; nil + // means the page is fetched over plain TLS. + Browser *browserRead +} + +type browserRead struct { + // Read builds the tab read for seriesURL, refusing (false) an address + // this Site will not open in a browser — the per-Site half of the SSRF + // gate, kept deliberately behind fetchableSeriesURL: a headless browser + // executes JavaScript and carries cookies, and series_url is + // client-supplied. + Read func(seriesURL string, out *string) (chromedp.Action, bool) + // Done reports whether the payload arrived. + Done func(body string) bool + // Fallback allows the plain-TLS fetcher when no browser is configured. + // False skips the Site instead. kagane is false — a plain fetch would + // only ever retrieve a challenge page — and novelfull is true, because + // its challenge is a live time-varying fact (AGENTS.md). + Fallback bool +} + // asuraSlugRe pulls the series slug out of a stored series_url. // Shape verified live 2026-07-26: https://asurascans.com/comics/, where // the slug carries a trailing build-hash suffix (e.g. "-f886a8af") that @@ -66,7 +102,7 @@ var novelfullSlugRe = regexp.MustCompile(`^/([^/?#]+)\.html$`) // under several Chapter Slugs (measured 2026-08-11: a sampled novel serves // 1-99 under one slug and 100-423 under another), so no stored-slug pattern can // cover a Series' whole list. An unscoped match is safe because -// latestChapterFrom truncates the body at the comment thread before scanning +// lnwLatestChapter truncates the body at the comment thread before scanning // (lnwCommentMarker); without that, a visitor's comment could set the Latest // Chapter on the shared Series row. var lnwChapterRe = regexp.MustCompile(`lightnovelworld\.net/[a-z0-9-]+-chapter-([0-9.]+)/`) @@ -80,86 +116,16 @@ var lnwChapterRe = regexp.MustCompile(`lightnovelworld\.net/[a-z0-9-]+-chapter-( // body is skipped, never scanned whole. const lnwCommentMarker = "wpd-threads" -// latestChapterFrom returns the highest chapter number body advertises for this -// series. ok is false when the body yields nothing usable — an unknown site, an -// empty body, a Cloudflare challenge page, and a site redesign all land here, -// and the caller treats all four identically. -// -// Ported from the userscript's latestChapterFromAnchors (asura L123-133, -// demonic L183-193), including its reason for taking a maximum rather than a -// first or last: neither site lists chapters in a dependable order. +// maxChapter returns the highest chapter number the regex finds in body. A +// maximum rather than a first or last, ported from the userscript's +// latestChapterFromAnchors (asura L123-133, demonic L183-193): neither site +// lists chapters in a dependable order. // // The userscript's asura rule additionally requires the anchor text to match // /Chapter\s+[\d.]+/i. That check exists only to skip the "First Chapter" -// shortcut, which points at chapter/1 and therefore can never win a maximum, so -// it is redundant here. For asura, scoping the pattern to this series' own slug -// replaces it with a stronger guarantee: a chapter link belonging to some other -// series cannot contribute even if the page starts carrying them. demonic has no -// such guarantee — demonicChapterRe matches any chaptered.php?manga= anchor -// with no per-series scoping, because the stored series_id for demonic is a -// slug, not the numeric id the URL carries, so it cannot easily be scoped. -// lightnovelworld is unscoped and body-truncated instead — see lnwChapterRe. -func latestChapterFrom(site, seriesURL, body string) (latestChapter, bool) { - var re *regexp.Regexp - switch site { - case "asura": - m := asuraSlugRe.FindStringSubmatch(seriesURL) - if m == nil { - return latestChapter{}, false - } - // Stored URLs predating a redeploy may carry a stale build hash; - // chapter hrefs in the fetched body carry the current one. Strip to - // the stable ID and make the hash optional in the pattern, so scoping - // survives rotations. - slug := asuraBuildHash.ReplaceAllString(m[1], "") - // Compiled per call rather than cached: this runs once per fetch, which - // is at most a few times a minute, and the slug varies per series. - re = regexp.MustCompile(`/comics/` + regexp.QuoteMeta(slug) + `(?:-[0-9a-f]{8})?/chapter/([0-9.]+)`) - case "demonic": - re = demonicChapterRe - case "comix": - id, ok := comixSeriesID(seriesURL) - if !ok { - return latestChapter{}, false - } - // comix ships an SPA: the served HTML carries a JSON state blob instead - // of chapter anchors, and latestChapterUrl is the only place the newest - // chapter appears. Scoping to this series' id prefix keeps a - // "recommended" strip's entries from winning the maximum. - re = regexp.MustCompile(`"latestChapterUrl":"/title/` + regexp.QuoteMeta(id) + `-[^"]*-chapter-([0-9.]+)"`) - case "kagane": - re = kaganeChapterRe - case "novelfull": - u, err := url.Parse(seriesURL) - if err != nil { - return latestChapter{}, false - } - m := novelfullSlugRe.FindStringSubmatch(u.Path) - if m == nil { - return latestChapter{}, false - } - // Scoped to this series' slug for the same reason asura is: page 1 - // carries a "latest chapters" widget and a "you may also like" strip, - // and neither may contribute to the maximum. - re = regexp.MustCompile(`/` + regexp.QuoteMeta(m[1]) + `/chapter-([0-9.]+)`) - case "lightnovelworld": - // The comment thread below the chapter list is the one region of the - // page any visitor can write to, so the scan never reads past it (see - // lnwChapterRe). A body without the marker is skipped, never scanned - // whole — a redesign must degrade into staleness, not into a wrong - // shared value; the logged body length tells a markup change from a - // body the size cap cut short. - i := strings.Index(body, lnwCommentMarker) - if i < 0 { - log.Printf("latest poll %q: no %s marker in %d bytes", seriesURL, lnwCommentMarker, len(body)) - return latestChapter{}, false - } - body = body[:i] - re = lnwChapterRe - default: - return latestChapter{}, false - } - +// shortcut, which points at chapter/1 and therefore can never win a maximum, +// so it is redundant once a maximum is taken. +func maxChapter(re *regexp.Regexp, body string) (latestChapter, bool) { var best latestChapter found := false for _, m := range re.FindAllStringSubmatch(body, -1) { @@ -178,6 +144,91 @@ func latestChapterFrom(site, seriesURL, body string) (latestChapter, bool) { return best, found } +// asuraLatestChapter scopes chapter links to this series' own slug, which +// replaces the userscript's anchor-text check with a stronger guarantee: a +// chapter link belonging to some other series cannot contribute even if the +// page starts carrying them. +func asuraLatestChapter(seriesURL, body string) (latestChapter, bool) { + m := asuraSlugRe.FindStringSubmatch(seriesURL) + if m == nil { + return latestChapter{}, false + } + // Stored URLs predating a redeploy may carry a stale build hash; chapter + // hrefs in the fetched body carry the current one. Strip to the stable ID + // and make the hash optional in the pattern, so scoping survives + // rotations. + slug := asuraBuildHash.ReplaceAllString(m[1], "") + // Compiled per call rather than cached: this runs once per fetch, which is + // at most a few times a minute, and the slug varies per series. + re := regexp.MustCompile(`/comics/` + regexp.QuoteMeta(slug) + `(?:-[0-9a-f]{8})?/chapter/([0-9.]+)`) + return maxChapter(re, body) +} + +// demonicLatestChapter is not scoped: demonicChapterRe matches any +// chaptered.php?manga= anchor, because the stored series_id is a slug, +// not the numeric id the URL carries, so it cannot be scoped. +func demonicLatestChapter(_, body string) (latestChapter, bool) { + return maxChapter(demonicChapterRe, body) +} + +// comixLatestChapter reads comix's SPA: the served HTML carries a JSON state +// blob instead of chapter anchors, and latestChapterUrl is the only place the +// newest chapter appears. Scoping to this series' id prefix keeps a +// "recommended" strip's entries from winning the maximum. +func comixLatestChapter(seriesURL, body string) (latestChapter, bool) { + id, ok := comixSeriesID(seriesURL) + if !ok { + return latestChapter{}, false + } + re := regexp.MustCompile(`"latestChapterUrl":"/title/` + regexp.QuoteMeta(id) + `-[^"]*-chapter-([0-9.]+)"`) + return maxChapter(re, body) +} + +func kaganeLatestChapter(_, body string) (latestChapter, bool) { + return maxChapter(kaganeChapterRe, body) +} + +// novelfullLatestChapter is scoped to this series' slug for the same reason +// asura is: page 1 carries a "latest chapters" widget and a "you may also +// like" strip, and neither may contribute to the maximum. +func novelfullLatestChapter(seriesURL, body string) (latestChapter, bool) { + u, err := url.Parse(seriesURL) + if err != nil { + return latestChapter{}, false + } + m := novelfullSlugRe.FindStringSubmatch(u.Path) + if m == nil { + return latestChapter{}, false + } + re := regexp.MustCompile(`/` + regexp.QuoteMeta(m[1]) + `/chapter-([0-9.]+)`) + return maxChapter(re, body) +} + +// lnwLatestChapter truncates the body at the comment thread before scanning: +// it is the one region of the page any visitor can write to (see lnwChapterRe). +// A body without the marker is skipped, never scanned whole — a redesign must +// degrade into staleness, not into a wrong shared value; the logged body length +// tells a markup change from a body the size cap cut short. +func lnwLatestChapter(seriesURL, body string) (latestChapter, bool) { + i := strings.Index(body, lnwCommentMarker) + if i < 0 { + log.Printf("latest poll %q: no %s marker in %d bytes", seriesURL, lnwCommentMarker, len(body)) + return latestChapter{}, false + } + return maxChapter(lnwChapterRe, body[:i]) +} + +// latestChapterFrom returns the highest chapter number body advertises for this +// series, via the Site's registry entry. ok is false when the body yields +// nothing usable — an unknown site, an empty body, a Cloudflare challenge page, +// and a site redesign all land here, and the caller treats all four identically. +func latestChapterFrom(site, seriesURL, body string) (latestChapter, bool) { + if fn := sites[site].LatestChapter; fn != nil { + return fn(seriesURL, body) + } + return latestChapter{}, false +} + var metaTagRe = regexp.MustCompile(`(?is)]*>`) var doubleQuotedMetaAttrRe = regexp.MustCompile(`(?is)([a-z][a-z0-9:_-]*)\s*=\s*"([^"]*)"`) var singleQuotedMetaAttrRe = regexp.MustCompile(`(?is)([a-z][a-z0-9:_-]*)\s*=\s*'([^']*)'`) @@ -257,24 +308,40 @@ func comixCoverURL(seriesURL, body string) string { return publishedCoverURL(detail.Poster.Medium) } -// coverFrom reports false for unknown sites, challenge bodies, and pages with -// no usable cover. Metadata extraction keeps scanning after an empty match so -// a later published cover is not hidden by an empty tag. -func coverFrom(site, seriesURL, body string) (string, bool) { - var cover string - switch site { - case "asura", "demonic", "lightnovelworld": - cover = metaContent(body, "property", "og:image") - case "novelfull": - cover = metaContent(body, "name", "image") - case "comix": - cover = comixCoverURL(seriesURL, body) - case "kagane": - cover = kaganeCoverURL(body) - } +// ogImageCover reads the og:image metadata shared by asura, demonic and +// lightnovelworld. +func ogImageCover(_, body string) (string, bool) { + cover := metaContent(body, "property", "og:image") return cover, cover != "" } +func novelfullCoverEntry(_, body string) (string, bool) { + cover := metaContent(body, "name", "image") + return cover, cover != "" +} + +func comixCoverEntry(seriesURL, body string) (string, bool) { + cover := comixCoverURL(seriesURL, body) + return cover, cover != "" +} + +func kaganeCoverEntry(_, body string) (string, bool) { + cover := kaganeCoverURL(body) + return cover, cover != "" +} + +// coverFrom reports false for unknown sites, challenge bodies, and pages with +// no usable cover, via the Site's registry entry. +func coverFrom(site, seriesURL, body string) (string, bool) { + if fn := sites[site].Cover; fn != nil { + return fn(seriesURL, body) + } + return "", false +} + +// metaContent returns the content of the first whose attrName is +// attrValue. It keeps scanning after an empty match so a later published cover +// is not hidden by an empty tag. func metaContent(body, attrName, attrValue string) string { for _, tag := range metaTagRe.FindAllString(body, -1) { attrs := make(map[string]string) @@ -297,3 +364,68 @@ func publishedCoverURL(value string) string { value = strings.TrimSpace(html.UnescapeString(value)) return strings.ReplaceAll(value, " ", "%20") } + +// sites is the registry: one entry per Site, keyed by the stored site string. +// Adding a Site means adding an entry here and nowhere else — the dispatch +// functions above and the poller's route list are lookups into this map. An +// unknown site string resolves to the zero entry, which fails the existing +// not-fetchable and no-fetcher paths unchanged. +var sites = map[string]site{ + "asura": { + Host: "asurascans.com", + LatestChapter: asuraLatestChapter, + Cover: ogImageCover, + }, + "demonic": { + Host: "demonicscans.org", + LatestChapter: demonicLatestChapter, + Cover: ogImageCover, + }, + "comix": { + Host: "comix.to", + LatestChapter: comixLatestChapter, + Cover: comixCoverEntry, + }, + "kagane": { + Host: "kagane.to", + LatestChapter: kaganeLatestChapter, + Cover: kaganeCoverEntry, + Browser: &browserRead{ + Read: kaganeRead, + Done: func(body string) bool { return body != "" }, + // Never falls back: a plain fetch of a kagane page or cover would + // only ever retrieve a challenge page (verified 2026-08-03). + Fallback: false, + }, + }, + "novelfull": { + Host: "novelfull.com", + LatestChapter: novelfullLatestChapter, + Cover: novelfullCoverEntry, + Browser: &browserRead{ + Read: novelfullRead, + // The interstitial has a DOM too, so "the payload arrived" has to + // exclude it explicitly. + Done: func(body string) bool { return body != "" && !isInterstitial(body) }, + Fallback: true, + }, + }, + "lightnovelworld": { + Host: "lightnovelworld.net", + LatestChapter: lnwLatestChapter, + Cover: ogImageCover, + }, +} + +// browserBackedSites is derived from the registry: the Sites whose pages are +// read through the browser sidecar, which are also the ones granted the longer +// cooldown. +func browserBackedSites() []string { + out := make([]string, 0, len(sites)) + for name, s := range sites { + if s.Browser != nil { + out = append(out, name) + } + } + return out +}