feat: one registry entry per Site, derived browser route and gate (#94)

Collapse the six per-site comparison points into a sites map in sites.go:
Latest Chapter parse, Cover parse, browser-backed list, fetcher route,
host pins, and the browser payload read all become lookups into it. All
six hosts are now pinned in fetchableSeriesURL; asura/demonic/comix were
previously accepted on any https host.
This commit is contained in:
2026-08-12 00:21:28 +07:00
parent 4a92e956cf
commit 2d134fb05c
4 changed files with 307 additions and 185 deletions
+48 -36
View File
@@ -86,46 +86,58 @@ func (f *BrowserFetcher) Close() {
f.cancel() f.cancel()
} }
// Get navigates to seriesURL, lets any challenge resolve, then reads either the // Get navigates to seriesURL, lets any challenge resolve, then reads the
// site's JSON API (kagane) from inside the page so the request carries the // payload the Site's registry entry describes — kagane's chapter-list API from
// clearance cookie, or the served HTML itself (novelfull) — see // inside the page so the request carries the clearance cookie, novelfull's
// novelfullSeriesURL for the latter case. The returned body is whatever the // served HTML. The returned body is whatever the Site's chapter list lives in,
// site's chapter list lives in, which is what latestChapterFrom's per-site // which is what the entry's LatestChapter parse expects.
// switch expects.
func (f *BrowserFetcher) Get(ctx context.Context, seriesURL string) (string, int, error) { func (f *BrowserFetcher) Get(ctx context.Context, seriesURL string) (string, int, error) {
apiURL, isKagane := kaganeAPIURL(seriesURL)
if !isKagane && !novelfullSeriesURL(seriesURL) {
return "", 0, fmt.Errorf("not a fetchable browser series url: %q", seriesURL)
}
var body string var body string
// kagane's chapter list is only in its JSON API, which must be called from for _, s := range sites {
// inside the page so the request carries the clearance cookie. novelfull if s.Browser == nil {
// renders its chapters into the HTML, so the cleared DOM is the answer. continue
// chromedp.OuterHTML returns a QueryAction and chromedp.Evaluate an
// EvaluateAction, so the variable has to be the interface both implement.
var read chromedp.Action = chromedp.OuterHTML("html", &body, chromedp.ByQuery)
if isKagane {
read = chromedp.Evaluate(
`fetch(`+jsString(apiURL)+`).then(r => r.ok ? r.text() : "")`,
&body,
awaitPromise,
)
}
// novelfull's payload is the DOM itself, and the interstitial has a DOM
// too, so "we have an answer" has to exclude it explicitly. kagane's
// in-page fetch just fails while challenged, which is already the signal.
done := func() bool { return body != "" && (isKagane || !isInterstitial(body)) }
if err := f.run(ctx, seriesURL, read, done); err != nil {
// Challenge never cleared, or the API refused. Indistinguishable from
// here and handled identically by the caller.
if errors.Is(err, errChallengeHeld) {
return "", 403, nil
} }
return "", 0, fmt.Errorf("browser fetch %q: %w", seriesURL, err) read, ok := s.Browser.Read(seriesURL, &body)
if !ok {
continue
}
if err := f.run(ctx, seriesURL, read,
func() bool { return s.Browser.Done(body) }); err != nil {
// Challenge never cleared, or the payload was refused.
// Indistinguishable from here and handled identically by the caller.
if errors.Is(err, errChallengeHeld) {
return "", 403, nil
}
return "", 0, fmt.Errorf("browser fetch %q: %w", seriesURL, err)
}
return body, 200, nil
} }
return body, 200, nil return "", 0, fmt.Errorf("not a fetchable browser series url: %q", seriesURL)
}
// kaganeRead builds the in-tab fetch of kagane's chapter-list API: the
// request must be made from inside the page so it carries the clearance
// cookie, and the API is the only place the list exists. Refusing any other
// address is the per-Site half of the SSRF gate, kept deliberately behind
// fetchableSeriesURL (see browserRead.Read).
func kaganeRead(seriesURL string, out *string) (chromedp.Action, bool) {
apiURL, ok := kaganeAPIURL(seriesURL)
if !ok {
return nil, false
}
return chromedp.Evaluate(
`fetch(`+jsString(apiURL)+`).then(r => r.ok ? r.text() : "")`,
out, awaitPromise), true
}
// novelfullRead reads the cleared DOM. novelfull renders its chapter list
// into the served HTML, so there is no API to call from inside the page — the
// challenge-cleared DOM is the payload.
func novelfullRead(seriesURL string, out *string) (chromedp.Action, bool) {
if !novelfullSeriesURL(seriesURL) {
return nil, false
}
return chromedp.OuterHTML("html", out, chromedp.ByQuery), true
} }
// Image retrieves one cover's bytes through the browser sidecar, and its // Image retrieves one cover's bytes through the browser sidecar, and its
+27 -56
View File
@@ -4,7 +4,6 @@ import (
"context" "context"
"log" "log"
"net/url" "net/url"
"slices"
"time" "time"
"bookmarkmanager/backend/internal/store" "bookmarkmanager/backend/internal/store"
@@ -57,8 +56,6 @@ type Poller struct {
Batch int Batch int
} }
var browserBackedSites = []string{"kagane", "novelfull"}
// fillBlankCover gives a Series its Cover when it has none. The blank state is // fillBlankCover gives a Series its Cover when it has none. The blank state is
// what "no Cover yet" means on the wire (ADR-0007): permanently-blank rows // what "no Cover yet" means on the wire (ADR-0007): permanently-blank rows
// created before acquisition existed, and rows whose creation-time fetch // created before acquisition existed, and rows whose creation-time fetch
@@ -120,27 +117,23 @@ func (p *Poller) storeCover(ctx context.Context, sr store.Series, sourceURL stri
} }
// fetcherFor returns the fetcher a site's page needs, or nil when the site // fetcherFor returns the fetcher a site's page needs, or nil when the site
// cannot be fetched at all right now. kagane and novelfull pages sit behind a // cannot be fetched at all right now. A Site whose registry entry carries a
// Cloudflare JavaScript challenge that no TLS fingerprint clears (kagane // Browser read — kagane and novelfull, both behind a Cloudflare JavaScript
// verified 2026-08-03, novelfull verified 2026-08-05, both against the same // challenge no TLS fingerprint clears — prefers the browser; when it is
// Chrome_133 profile TLSFetcher uses), so both prefer the browser; novelfull // absent, the entry's Fallback decides whether plain TLS may take over. One
// alone falls back to the plain-TLS fetcher when no browser is configured, // routing rule for the poll and the acquirer, so the two cannot drift apart.
// because its challenge is a live time-varying fact (AGENTS.md) and its cover
// bytes never need the browser. kagane never falls back: a plain fetch of a
// kagane page or cover would only ever retrieve a challenge page. One routing
// rule for the poll and the acquirer, so the two cannot drift apart.
func fetcherFor(site string, browser, tls Fetcher) Fetcher { func fetcherFor(site string, browser, tls Fetcher) Fetcher {
switch { s := sites[site]
case site == "kagane": if s.Browser == nil {
return browser
case slices.Contains(browserBackedSites, site): // novelfull
if browser != nil {
return browser
}
return tls
default:
return tls return tls
} }
if browser != nil {
return browser
}
if s.Browser.Fallback {
return tls
}
return nil
} }
// Run polls until ctx is cancelled. // Run polls until ctx is cancelled.
@@ -170,7 +163,7 @@ func (p *Poller) runOnce(ctx context.Context) {
now := p.Now() now := p.Now()
cutoff := now.Add(-p.Cooldown).UnixMilli() cutoff := now.Add(-p.Cooldown).UnixMilli()
browserCutoff := now.Add(-p.BrowserCooldown).UnixMilli() browserCutoff := now.Add(-p.BrowserCooldown).UnixMilli()
due, err := p.Store.DueForLatestCheck(cutoff, browserCutoff, browserBackedSites, p.Batch) due, err := p.Store.DueForLatestCheck(cutoff, browserCutoff, browserBackedSites(), p.Batch)
if err != nil { if err != nil {
log.Printf("latest poll: due query: %v", err) log.Printf("latest poll: due query: %v", err)
return return
@@ -283,45 +276,23 @@ func (p *Poller) checkOne(ctx context.Context, sr store.Series) {
log.Printf("latest poll %q: latest is now %s", sr.Key(), latest.Label) log.Printf("latest poll %q: latest is now %s", sr.Key(), latest.Label)
} }
// fetchableSeriesURL reports whether site is a site latestChapterFrom knows how // fetchableSeriesURL reports whether site is a Site the registry knows and
// to parse and seriesURL is safe to hand to a fetcher: an https URL with a // seriesURL is safe to hand to a fetcher: an https URL whose host matches the
// non-empty host. series_url comes from client-supplied PUT bodies, so this is // Site's pinned hostname exactly. series_url comes from client-supplied PUT
// a defence against the poller being used to probe arbitrary hosts from the // bodies, so this is a defence against the poller being used to probe
// server's own network position, not just a check against wasted requests. // arbitrary hosts from the server's own network position, not just a check
// // against wasted requests. The pin guards different things per Site — a
// Three sites are held to a stricter rule, each for a different reason: // browser Site guards a control that executes JavaScript and carries cookies,
// // a parser Site guards a wasted request — but the rule is one rule, from the
// - kagane and novelfull are fetched by a headless browser, which executes // registry.
// JavaScript and carries cookies, and is therefore a far stronger SSRF
// primitive than an HTTP GET. Their hosts must match exactly, not merely
// be non-empty.
// - lightnovelworld's parser regex hardcodes its host, so a URL anywhere
// else could never yield a match — reject it here rather than burn the
// request.
func fetchableSeriesURL(site, seriesURL string) bool { func fetchableSeriesURL(site, seriesURL string) bool {
switch site { s, known := sites[site]
case "asura", "demonic", "comix", "kagane", "novelfull", "lightnovelworld": if !known {
default:
return false return false
} }
u, err := url.Parse(seriesURL) u, err := url.Parse(seriesURL)
if err != nil { if err != nil {
return false return false
} }
if u.Scheme != "https" || u.Host == "" { return u.Scheme == "https" && u.Hostname() == s.Host
return false
}
switch site {
case "kagane":
return u.Hostname() == "kagane.to"
case "novelfull":
// Fetched by a real browser, same as kagane, so the host is pinned
// rather than merely non-empty.
return u.Hostname() == "novelfull.com"
case "lightnovelworld":
// Its parser regex hardcodes this host, so a URL anywhere else could
// never yield a match — reject it here rather than burn the request.
return u.Hostname() == "lightnovelworld.net"
}
return true
} }
+7
View File
@@ -623,6 +623,13 @@ func TestFetchableSeriesURL(t *testing.T) {
{"demonic https", "demonic", "https://demonicscans.org/manga/X", true}, {"demonic https", "demonic", "https://demonicscans.org/manga/X", true},
{"comix https", "comix", "https://comix.to/title/n8we-dungeons-and-crayons", true}, {"comix https", "comix", "https://comix.to/title/n8we-dungeons-and-crayons", true},
{"kagane on its own host", "kagane", "https://kagane.to/series/019f84bc-9ba0-7ed9-86f5-8b905ec7c28b", true}, {"kagane on its own host", "kagane", "https://kagane.to/series/019f84bc-9ba0-7ed9-86f5-8b905ec7c28b", true},
// The pins added in #94 cover the three plain-TLS Sites too: a
// client-supplied series_url must not aim a fetcher at a lookalike
// host, even when the fetcher is only an HTTP GET.
{"asura on a foreign host", "asura", "https://asurascans.com.evil.example/comics/x", false},
{"asura on the dead old domain", "asura", "https://asuracomic.net/comics/x", false},
{"demonic on a lookalike host", "demonic", "https://demonicscans.org.evil.example/manga/X", false},
{"comix on a foreign host", "comix", "https://evil.example/title/x", false},
// The browser fetcher runs JavaScript and carries cookies, so a // The browser fetcher runs JavaScript and carries cookies, so a
// client-supplied series_url must not be able to aim it anywhere else. // client-supplied series_url must not be able to aim it anywhere else.
{"kagane on a foreign host", "kagane", "https://evil.example/series/x", false}, {"kagane on a foreign host", "kagane", "https://evil.example/series/x", false},
+225 -93
View File
@@ -8,6 +8,8 @@ import (
"regexp" "regexp"
"strconv" "strconv"
"strings" "strings"
"github.com/chromedp/chromedp"
) )
// latestChapter is the newest chapter a series page advertises. // latestChapter is the newest chapter a series page advertises.
@@ -16,6 +18,40 @@ type latestChapter struct {
Label string Label string
} }
// site answers the fixed questions every series-page read asks of its Site
// (ADR-0009): the host its addresses must carry, how to find the Latest
// Chapter and the Cover address in a body, and — for a Site behind a
// JavaScript challenge — how to read its payload from a cleared tab. One
// entry describes everything about one Site, and nowhere else gets to compare
// the site string.
type site struct {
// Host is the exact hostname a series_url for this Site must carry.
Host string
// LatestChapter finds the newest chapter in a fetched body.
LatestChapter func(seriesURL, body string) (latestChapter, bool)
// Cover finds the Cover address in a fetched body.
Cover func(seriesURL, body string) (string, bool)
// Browser reads this Site's payload from a cleared browser tab; nil
// means the page is fetched over plain TLS.
Browser *browserRead
}
type browserRead struct {
// Read builds the tab read for seriesURL, refusing (false) an address
// this Site will not open in a browser — the per-Site half of the SSRF
// gate, kept deliberately behind fetchableSeriesURL: a headless browser
// executes JavaScript and carries cookies, and series_url is
// client-supplied.
Read func(seriesURL string, out *string) (chromedp.Action, bool)
// Done reports whether the payload arrived.
Done func(body string) bool
// Fallback allows the plain-TLS fetcher when no browser is configured.
// False skips the Site instead. kagane is false — a plain fetch would
// only ever retrieve a challenge page — and novelfull is true, because
// its challenge is a live time-varying fact (AGENTS.md).
Fallback bool
}
// asuraSlugRe pulls the series slug out of a stored series_url. // asuraSlugRe pulls the series slug out of a stored series_url.
// Shape verified live 2026-07-26: https://asurascans.com/comics/<slug>, where // Shape verified live 2026-07-26: https://asurascans.com/comics/<slug>, where
// the slug carries a trailing build-hash suffix (e.g. "-f886a8af") that // the slug carries a trailing build-hash suffix (e.g. "-f886a8af") that
@@ -66,7 +102,7 @@ var novelfullSlugRe = regexp.MustCompile(`^/([^/?#]+)\.html$`)
// under several Chapter Slugs (measured 2026-08-11: a sampled novel serves // under several Chapter Slugs (measured 2026-08-11: a sampled novel serves
// 1-99 under one slug and 100-423 under another), so no stored-slug pattern can // 1-99 under one slug and 100-423 under another), so no stored-slug pattern can
// cover a Series' whole list. An unscoped match is safe because // cover a Series' whole list. An unscoped match is safe because
// latestChapterFrom truncates the body at the comment thread before scanning // lnwLatestChapter truncates the body at the comment thread before scanning
// (lnwCommentMarker); without that, a visitor's comment could set the Latest // (lnwCommentMarker); without that, a visitor's comment could set the Latest
// Chapter on the shared Series row. // Chapter on the shared Series row.
var lnwChapterRe = regexp.MustCompile(`lightnovelworld\.net/[a-z0-9-]+-chapter-([0-9.]+)/`) var lnwChapterRe = regexp.MustCompile(`lightnovelworld\.net/[a-z0-9-]+-chapter-([0-9.]+)/`)
@@ -80,86 +116,16 @@ var lnwChapterRe = regexp.MustCompile(`lightnovelworld\.net/[a-z0-9-]+-chapter-(
// body is skipped, never scanned whole. // body is skipped, never scanned whole.
const lnwCommentMarker = "wpd-threads" const lnwCommentMarker = "wpd-threads"
// latestChapterFrom returns the highest chapter number body advertises for this // maxChapter returns the highest chapter number the regex finds in body. A
// series. ok is false when the body yields nothing usable — an unknown site, an // maximum rather than a first or last, ported from the userscript's
// empty body, a Cloudflare challenge page, and a site redesign all land here, // latestChapterFromAnchors (asura L123-133, demonic L183-193): neither site
// and the caller treats all four identically. // lists chapters in a dependable order.
//
// Ported from the userscript's latestChapterFromAnchors (asura L123-133,
// demonic L183-193), including its reason for taking a maximum rather than a
// first or last: neither site lists chapters in a dependable order.
// //
// The userscript's asura rule additionally requires the anchor text to match // The userscript's asura rule additionally requires the anchor text to match
// /Chapter\s+[\d.]+/i. That check exists only to skip the "First Chapter" // /Chapter\s+[\d.]+/i. That check exists only to skip the "First Chapter"
// shortcut, which points at chapter/1 and therefore can never win a maximum, so // shortcut, which points at chapter/1 and therefore can never win a maximum,
// it is redundant here. For asura, scoping the pattern to this series' own slug // so it is redundant once a maximum is taken.
// replaces it with a stronger guarantee: a chapter link belonging to some other func maxChapter(re *regexp.Regexp, body string) (latestChapter, bool) {
// series cannot contribute even if the page starts carrying them. demonic has no
// such guarantee — demonicChapterRe matches any chaptered.php?manga=<id> anchor
// with no per-series scoping, because the stored series_id for demonic is a
// slug, not the numeric id the URL carries, so it cannot easily be scoped.
// lightnovelworld is unscoped and body-truncated instead — see lnwChapterRe.
func latestChapterFrom(site, seriesURL, body string) (latestChapter, bool) {
var re *regexp.Regexp
switch site {
case "asura":
m := asuraSlugRe.FindStringSubmatch(seriesURL)
if m == nil {
return latestChapter{}, false
}
// Stored URLs predating a redeploy may carry a stale build hash;
// chapter hrefs in the fetched body carry the current one. Strip to
// the stable ID and make the hash optional in the pattern, so scoping
// survives rotations.
slug := asuraBuildHash.ReplaceAllString(m[1], "")
// Compiled per call rather than cached: this runs once per fetch, which
// is at most a few times a minute, and the slug varies per series.
re = regexp.MustCompile(`/comics/` + regexp.QuoteMeta(slug) + `(?:-[0-9a-f]{8})?/chapter/([0-9.]+)`)
case "demonic":
re = demonicChapterRe
case "comix":
id, ok := comixSeriesID(seriesURL)
if !ok {
return latestChapter{}, false
}
// comix ships an SPA: the served HTML carries a JSON state blob instead
// of chapter anchors, and latestChapterUrl is the only place the newest
// chapter appears. Scoping to this series' id prefix keeps a
// "recommended" strip's entries from winning the maximum.
re = regexp.MustCompile(`"latestChapterUrl":"/title/` + regexp.QuoteMeta(id) + `-[^"]*-chapter-([0-9.]+)"`)
case "kagane":
re = kaganeChapterRe
case "novelfull":
u, err := url.Parse(seriesURL)
if err != nil {
return latestChapter{}, false
}
m := novelfullSlugRe.FindStringSubmatch(u.Path)
if m == nil {
return latestChapter{}, false
}
// Scoped to this series' slug for the same reason asura is: page 1
// carries a "latest chapters" widget and a "you may also like" strip,
// and neither may contribute to the maximum.
re = regexp.MustCompile(`/` + regexp.QuoteMeta(m[1]) + `/chapter-([0-9.]+)`)
case "lightnovelworld":
// The comment thread below the chapter list is the one region of the
// page any visitor can write to, so the scan never reads past it (see
// lnwChapterRe). A body without the marker is skipped, never scanned
// whole — a redesign must degrade into staleness, not into a wrong
// shared value; the logged body length tells a markup change from a
// body the size cap cut short.
i := strings.Index(body, lnwCommentMarker)
if i < 0 {
log.Printf("latest poll %q: no %s marker in %d bytes", seriesURL, lnwCommentMarker, len(body))
return latestChapter{}, false
}
body = body[:i]
re = lnwChapterRe
default:
return latestChapter{}, false
}
var best latestChapter var best latestChapter
found := false found := false
for _, m := range re.FindAllStringSubmatch(body, -1) { for _, m := range re.FindAllStringSubmatch(body, -1) {
@@ -178,6 +144,91 @@ func latestChapterFrom(site, seriesURL, body string) (latestChapter, bool) {
return best, found return best, found
} }
// asuraLatestChapter scopes chapter links to this series' own slug, which
// replaces the userscript's anchor-text check with a stronger guarantee: a
// chapter link belonging to some other series cannot contribute even if the
// page starts carrying them.
func asuraLatestChapter(seriesURL, body string) (latestChapter, bool) {
m := asuraSlugRe.FindStringSubmatch(seriesURL)
if m == nil {
return latestChapter{}, false
}
// Stored URLs predating a redeploy may carry a stale build hash; chapter
// hrefs in the fetched body carry the current one. Strip to the stable ID
// and make the hash optional in the pattern, so scoping survives
// rotations.
slug := asuraBuildHash.ReplaceAllString(m[1], "")
// Compiled per call rather than cached: this runs once per fetch, which is
// at most a few times a minute, and the slug varies per series.
re := regexp.MustCompile(`/comics/` + regexp.QuoteMeta(slug) + `(?:-[0-9a-f]{8})?/chapter/([0-9.]+)`)
return maxChapter(re, body)
}
// demonicLatestChapter is not scoped: demonicChapterRe matches any
// chaptered.php?manga=<id> anchor, because the stored series_id is a slug,
// not the numeric id the URL carries, so it cannot be scoped.
func demonicLatestChapter(_, body string) (latestChapter, bool) {
return maxChapter(demonicChapterRe, body)
}
// comixLatestChapter reads comix's SPA: the served HTML carries a JSON state
// blob instead of chapter anchors, and latestChapterUrl is the only place the
// newest chapter appears. Scoping to this series' id prefix keeps a
// "recommended" strip's entries from winning the maximum.
func comixLatestChapter(seriesURL, body string) (latestChapter, bool) {
id, ok := comixSeriesID(seriesURL)
if !ok {
return latestChapter{}, false
}
re := regexp.MustCompile(`"latestChapterUrl":"/title/` + regexp.QuoteMeta(id) + `-[^"]*-chapter-([0-9.]+)"`)
return maxChapter(re, body)
}
func kaganeLatestChapter(_, body string) (latestChapter, bool) {
return maxChapter(kaganeChapterRe, body)
}
// novelfullLatestChapter is scoped to this series' slug for the same reason
// asura is: page 1 carries a "latest chapters" widget and a "you may also
// like" strip, and neither may contribute to the maximum.
func novelfullLatestChapter(seriesURL, body string) (latestChapter, bool) {
u, err := url.Parse(seriesURL)
if err != nil {
return latestChapter{}, false
}
m := novelfullSlugRe.FindStringSubmatch(u.Path)
if m == nil {
return latestChapter{}, false
}
re := regexp.MustCompile(`/` + regexp.QuoteMeta(m[1]) + `/chapter-([0-9.]+)`)
return maxChapter(re, body)
}
// lnwLatestChapter truncates the body at the comment thread before scanning:
// it is the one region of the page any visitor can write to (see lnwChapterRe).
// A body without the marker is skipped, never scanned whole — a redesign must
// degrade into staleness, not into a wrong shared value; the logged body length
// tells a markup change from a body the size cap cut short.
func lnwLatestChapter(seriesURL, body string) (latestChapter, bool) {
i := strings.Index(body, lnwCommentMarker)
if i < 0 {
log.Printf("latest poll %q: no %s marker in %d bytes", seriesURL, lnwCommentMarker, len(body))
return latestChapter{}, false
}
return maxChapter(lnwChapterRe, body[:i])
}
// latestChapterFrom returns the highest chapter number body advertises for this
// series, via the Site's registry entry. ok is false when the body yields
// nothing usable — an unknown site, an empty body, a Cloudflare challenge page,
// and a site redesign all land here, and the caller treats all four identically.
func latestChapterFrom(site, seriesURL, body string) (latestChapter, bool) {
if fn := sites[site].LatestChapter; fn != nil {
return fn(seriesURL, body)
}
return latestChapter{}, false
}
var metaTagRe = regexp.MustCompile(`(?is)<meta\b[^>]*>`) var metaTagRe = regexp.MustCompile(`(?is)<meta\b[^>]*>`)
var doubleQuotedMetaAttrRe = regexp.MustCompile(`(?is)([a-z][a-z0-9:_-]*)\s*=\s*"([^"]*)"`) var doubleQuotedMetaAttrRe = regexp.MustCompile(`(?is)([a-z][a-z0-9:_-]*)\s*=\s*"([^"]*)"`)
var singleQuotedMetaAttrRe = regexp.MustCompile(`(?is)([a-z][a-z0-9:_-]*)\s*=\s*'([^']*)'`) var singleQuotedMetaAttrRe = regexp.MustCompile(`(?is)([a-z][a-z0-9:_-]*)\s*=\s*'([^']*)'`)
@@ -257,24 +308,40 @@ func comixCoverURL(seriesURL, body string) string {
return publishedCoverURL(detail.Poster.Medium) return publishedCoverURL(detail.Poster.Medium)
} }
// coverFrom reports false for unknown sites, challenge bodies, and pages with // ogImageCover reads the og:image metadata shared by asura, demonic and
// no usable cover. Metadata extraction keeps scanning after an empty match so // lightnovelworld.
// a later published cover is not hidden by an empty tag. func ogImageCover(_, body string) (string, bool) {
func coverFrom(site, seriesURL, body string) (string, bool) { cover := metaContent(body, "property", "og:image")
var cover string
switch site {
case "asura", "demonic", "lightnovelworld":
cover = metaContent(body, "property", "og:image")
case "novelfull":
cover = metaContent(body, "name", "image")
case "comix":
cover = comixCoverURL(seriesURL, body)
case "kagane":
cover = kaganeCoverURL(body)
}
return cover, cover != "" return cover, cover != ""
} }
func novelfullCoverEntry(_, body string) (string, bool) {
cover := metaContent(body, "name", "image")
return cover, cover != ""
}
func comixCoverEntry(seriesURL, body string) (string, bool) {
cover := comixCoverURL(seriesURL, body)
return cover, cover != ""
}
func kaganeCoverEntry(_, body string) (string, bool) {
cover := kaganeCoverURL(body)
return cover, cover != ""
}
// coverFrom reports false for unknown sites, challenge bodies, and pages with
// no usable cover, via the Site's registry entry.
func coverFrom(site, seriesURL, body string) (string, bool) {
if fn := sites[site].Cover; fn != nil {
return fn(seriesURL, body)
}
return "", false
}
// metaContent returns the content of the first <meta> whose attrName is
// attrValue. It keeps scanning after an empty match so a later published cover
// is not hidden by an empty tag.
func metaContent(body, attrName, attrValue string) string { func metaContent(body, attrName, attrValue string) string {
for _, tag := range metaTagRe.FindAllString(body, -1) { for _, tag := range metaTagRe.FindAllString(body, -1) {
attrs := make(map[string]string) attrs := make(map[string]string)
@@ -297,3 +364,68 @@ func publishedCoverURL(value string) string {
value = strings.TrimSpace(html.UnescapeString(value)) value = strings.TrimSpace(html.UnescapeString(value))
return strings.ReplaceAll(value, " ", "%20") return strings.ReplaceAll(value, " ", "%20")
} }
// sites is the registry: one entry per Site, keyed by the stored site string.
// Adding a Site means adding an entry here and nowhere else — the dispatch
// functions above and the poller's route list are lookups into this map. An
// unknown site string resolves to the zero entry, which fails the existing
// not-fetchable and no-fetcher paths unchanged.
var sites = map[string]site{
"asura": {
Host: "asurascans.com",
LatestChapter: asuraLatestChapter,
Cover: ogImageCover,
},
"demonic": {
Host: "demonicscans.org",
LatestChapter: demonicLatestChapter,
Cover: ogImageCover,
},
"comix": {
Host: "comix.to",
LatestChapter: comixLatestChapter,
Cover: comixCoverEntry,
},
"kagane": {
Host: "kagane.to",
LatestChapter: kaganeLatestChapter,
Cover: kaganeCoverEntry,
Browser: &browserRead{
Read: kaganeRead,
Done: func(body string) bool { return body != "" },
// Never falls back: a plain fetch of a kagane page or cover would
// only ever retrieve a challenge page (verified 2026-08-03).
Fallback: false,
},
},
"novelfull": {
Host: "novelfull.com",
LatestChapter: novelfullLatestChapter,
Cover: novelfullCoverEntry,
Browser: &browserRead{
Read: novelfullRead,
// The interstitial has a DOM too, so "the payload arrived" has to
// exclude it explicitly.
Done: func(body string) bool { return body != "" && !isInterstitial(body) },
Fallback: true,
},
},
"lightnovelworld": {
Host: "lightnovelworld.net",
LatestChapter: lnwLatestChapter,
Cover: ogImageCover,
},
}
// browserBackedSites is derived from the registry: the Sites whose pages are
// read through the browser sidecar, which are also the ones granted the longer
// cooldown.
func browserBackedSites() []string {
out := make([]string, 0, len(sites))
for name, s := range sites {
if s.Browser != nil {
out = append(out, name)
}
}
return out
}