b38dfe54e8
Spec review flagged the live-verification acceptance criterion as unproven in the diff: the brief's adapter rule wants the probe recorded, not just the date. Each fixture comment now names how the live body was fetched on 2026-08-22 — curl probe for asura, demonic and lightnovelworld; cleared Chrome tab (CDP sidecar) for comix, kagane and novelfull, including the challenge/403 fallback story for novelfull. Also renamed lnwCompletedRe to lnwStatusRe for symmetry with the other site-marker vars.
640 lines
25 KiB
Go
640 lines
25 KiB
Go
package latest
|
|
|
|
import (
|
|
"encoding/json"
|
|
"html"
|
|
"log"
|
|
"net/url"
|
|
"regexp"
|
|
"sort"
|
|
"strconv"
|
|
"strings"
|
|
"time"
|
|
|
|
"github.com/chromedp/chromedp"
|
|
)
|
|
|
|
// latestChapter is the newest chapter a series page advertises.
|
|
type latestChapter struct {
|
|
Num float64
|
|
Label string
|
|
}
|
|
|
|
// site answers the fixed questions every series-page read asks of its Site
|
|
// (ADR-0009): the host its addresses must carry, how to find the Latest
|
|
// Chapter and the Cover address in a body, whether the Site calls the work
|
|
// completed, and — for a Site behind a JavaScript challenge — how to read its
|
|
// payload from a cleared tab. One entry describes everything about one Site,
|
|
// and nowhere else gets to compare the site string.
|
|
type site struct {
|
|
// Host is the exact hostname a series_url for this Site must carry.
|
|
Host string
|
|
// LatestChapter finds the newest chapter in a fetched body.
|
|
LatestChapter func(seriesURL, body string) (latestChapter, bool)
|
|
// Cover finds the Cover address in a fetched body.
|
|
Cover func(seriesURL, body string) (string, bool)
|
|
// Completed reports whether this body carries the Site's own completed
|
|
// value. False for a body that carries any other value, and false for a
|
|
// failed extraction — never an error and never a third state.
|
|
Completed func(seriesURL, body string) bool
|
|
// Rest is how long a Series of this Site rests between Polls.
|
|
Rest time.Duration
|
|
// Gap is the Lane's strictest pace: at least one second must pass between
|
|
// two consecutive Series-page Polls of this Site (issue #100).
|
|
Gap time.Duration
|
|
// Browser reads this Site's payload from a cleared browser tab; nil
|
|
// means the page is fetched over plain TLS.
|
|
Browser *browserRead
|
|
}
|
|
|
|
type browserRead struct {
|
|
// Read builds the tab read for seriesURL, refusing (false) an address
|
|
// this Site will not open in a browser — the per-Site half of the SSRF
|
|
// gate, kept deliberately behind FetchableSeriesURL: a headless browser
|
|
// executes JavaScript and carries cookies, and series_url is
|
|
// client-supplied.
|
|
Read func(seriesURL string, out *string) (chromedp.Action, bool)
|
|
// Done reports whether the payload arrived.
|
|
Done func(body string) bool
|
|
// Fallback allows the plain-TLS fetcher when no browser is configured.
|
|
// False skips the Site instead. kagane and comix are false — a plain fetch
|
|
// would only ever retrieve a challenge page — and novelfull is true,
|
|
// because its challenge is a live time-varying fact (AGENTS.md).
|
|
Fallback bool
|
|
}
|
|
|
|
// asuraSlugRe pulls the series slug out of a stored series_url.
|
|
// Shape verified live 2026-07-26: https://asurascans.com/comics/<slug>, where
|
|
// the slug carries a trailing build-hash suffix (e.g. "-f886a8af") that
|
|
// rotates on every site redeploy — callers must strip it (asuraBuildHash)
|
|
// before using the slug to scope anything.
|
|
var asuraSlugRe = regexp.MustCompile(`/comics/([^/?#]+)`)
|
|
|
|
// asuraBuildHash matches the trailing "-xxxxxxxx" site-wide build ID Asura
|
|
// appends to every series slug. It rotates on each site redeploy, so it is
|
|
// never part of a stable series_id. Must stay in sync with stripBuildHash in
|
|
// userscript/manga-bookmark.user.js.
|
|
var asuraBuildHash = regexp.MustCompile(`-[0-9a-f]{8}$`)
|
|
|
|
// demonicChapterRe matches the pre-redirect anchors demonic series pages link
|
|
// through. Both the raw "&" and the HTML-escaped "&" forms occur.
|
|
var demonicChapterRe = regexp.MustCompile(`chaptered\.php\?manga=\d+&(?:amp;)?chapter=([0-9.]+)`)
|
|
|
|
// comixSlugRe pulls the "<id>-<slug>" segment out of a stored series_url.
|
|
// Only the id prefix is stable; the slug tail follows the title.
|
|
var comixSlugRe = regexp.MustCompile(`/title/([^/?#]+)`)
|
|
|
|
func comixSeriesID(seriesURL string) (string, bool) {
|
|
m := comixSlugRe.FindStringSubmatch(seriesURL)
|
|
if m == nil {
|
|
return "", false
|
|
}
|
|
id := m[1]
|
|
if i := strings.Index(id, "-"); i != -1 {
|
|
id = id[:i]
|
|
}
|
|
return id, true
|
|
}
|
|
|
|
// kaganeChapterRe matches the chapter numbers in a kagane API response. This
|
|
// branch is fed by the browser fetcher, so the body is JSON rather than HTML —
|
|
// there are no anchors to scan.
|
|
var kaganeChapterRe = regexp.MustCompile(`"chapter_no":"([0-9.]+)"`)
|
|
|
|
// novelfullSlugRe pulls the series slug out of a stored series_url. novelfull
|
|
// series pages are "/<slug>.html"; their chapter anchors are
|
|
// "/<slug>/chapter-<n>[-<title-slug>].html". Verified live 2026-08-05.
|
|
var novelfullSlugRe = regexp.MustCompile(`^/([^/?#]+)\.html$`)
|
|
|
|
// lnwChapterRe matches any chapter-shaped address on lightnovelworld. Unlike
|
|
// asura, novelfull and comix — which scope to their stored series slug so a
|
|
// foreign chapter link cannot contribute — this Site's chapter addresses carry
|
|
// the Chapter Slug, which is not the Series identity: one Series may publish
|
|
// under several Chapter Slugs (measured 2026-08-11: a sampled novel serves
|
|
// 1-99 under one slug and 100-423 under another), so no stored-slug pattern can
|
|
// cover a Series' whole list. An unscoped match is safe because
|
|
// lnwLatestChapter truncates the body at the comment thread before scanning
|
|
// (lnwCommentMarker); without that, a visitor's comment could set the Latest
|
|
// Chapter on the shared Series row.
|
|
var lnwChapterRe = regexp.MustCompile(`lightnovelworld\.net/[a-z0-9-]+-chapter-([0-9.]+)/`)
|
|
|
|
// lnwCommentMarker is the boundary of lightnovelworld's server-rendered
|
|
// wpdiscuz comment thread. It occurs exactly once per page and follows every
|
|
// chapter anchor (measured 2026-08-11,
|
|
// docs/research/lightnovelworld-chapter-vs-series-slug.md §6), so cutting the
|
|
// body at its first occurrence keeps the whole chapter list while excluding a
|
|
// region any visitor can write to. Absent means the page shape changed: the
|
|
// body is skipped, never scanned whole.
|
|
const lnwCommentMarker = "wpd-threads"
|
|
|
|
// maxChapter returns the highest chapter number the regex finds in body. A
|
|
// maximum rather than a first or last, ported from the userscript's
|
|
// latestChapterFromAnchors (asura L123-133, demonic L183-193): neither site
|
|
// lists chapters in a dependable order.
|
|
//
|
|
// The userscript's asura rule additionally requires the anchor text to match
|
|
// /Chapter\s+[\d.]+/i. That check exists only to skip the "First Chapter"
|
|
// shortcut, which points at chapter/1 and therefore can never win a maximum,
|
|
// so it is redundant once a maximum is taken.
|
|
func maxChapter(re *regexp.Regexp, body string) (latestChapter, bool) {
|
|
var best latestChapter
|
|
found := false
|
|
for _, m := range re.FindAllStringSubmatch(body, -1) {
|
|
// [0-9.]+ can swallow a trailing separator, e.g. "chapter/12." in a
|
|
// sentence; ParseFloat would reject the whole match.
|
|
raw := strings.Trim(m[1], ".")
|
|
num, err := strconv.ParseFloat(raw, 64)
|
|
if err != nil {
|
|
continue
|
|
}
|
|
if !found || num > best.Num {
|
|
best = latestChapter{Num: num, Label: "Chapter " + raw}
|
|
found = true
|
|
}
|
|
}
|
|
return best, found
|
|
}
|
|
|
|
// asuraLatestChapter scopes chapter links to this series' own slug, which
|
|
// replaces the userscript's anchor-text check with a stronger guarantee: a
|
|
// chapter link belonging to some other series cannot contribute even if the
|
|
// page starts carrying them.
|
|
func asuraLatestChapter(seriesURL, body string) (latestChapter, bool) {
|
|
m := asuraSlugRe.FindStringSubmatch(seriesURL)
|
|
if m == nil {
|
|
return latestChapter{}, false
|
|
}
|
|
// Stored URLs predating a redeploy may carry a stale build hash; chapter
|
|
// hrefs in the fetched body carry the current one. Strip to the stable ID
|
|
// and make the hash optional in the pattern, so scoping survives
|
|
// rotations.
|
|
slug := asuraBuildHash.ReplaceAllString(m[1], "")
|
|
// Compiled per call rather than cached: this runs once per fetch, which is
|
|
// at most a few times a minute, and the slug varies per series.
|
|
re := regexp.MustCompile(`/comics/` + regexp.QuoteMeta(slug) + `(?:-[0-9a-f]{8})?/chapter/([0-9.]+)`)
|
|
return maxChapter(re, body)
|
|
}
|
|
|
|
// demonicLatestChapter is not scoped: demonicChapterRe matches any
|
|
// chaptered.php?manga=<id> anchor, because the stored series_id is a slug,
|
|
// not the numeric id the URL carries, so it cannot be scoped.
|
|
func demonicLatestChapter(_, body string) (latestChapter, bool) {
|
|
return maxChapter(demonicChapterRe, body)
|
|
}
|
|
|
|
// comixLatestChapter reads comix's SPA: the served HTML carries a JSON state
|
|
// blob instead of chapter anchors, and latestChapterUrl is the only place the
|
|
// newest chapter appears. Scoping to this series' id prefix keeps a
|
|
// "recommended" strip's entries from winning the maximum.
|
|
func comixLatestChapter(seriesURL, body string) (latestChapter, bool) {
|
|
id, ok := comixSeriesID(seriesURL)
|
|
if !ok {
|
|
return latestChapter{}, false
|
|
}
|
|
re := regexp.MustCompile(`"latestChapterUrl":"/title/` + regexp.QuoteMeta(id) + `-[^"]*-chapter-([0-9.]+)"`)
|
|
return maxChapter(re, body)
|
|
}
|
|
|
|
// kaganeLatestChapter scans the kagane series API JSON that the browser read
|
|
// fetched from inside the page; the match rides on the property name,
|
|
// regardless of the surrounding JSON shape.
|
|
func kaganeLatestChapter(_, body string) (latestChapter, bool) {
|
|
return maxChapter(kaganeChapterRe, body)
|
|
}
|
|
|
|
// novelfullLatestChapter is scoped to this series' slug for the same reason
|
|
// asura is: page 1 carries a "latest chapters" widget and a "you may also
|
|
// like" strip, and neither may contribute to the maximum.
|
|
func novelfullLatestChapter(seriesURL, body string) (latestChapter, bool) {
|
|
u, err := url.Parse(seriesURL)
|
|
if err != nil {
|
|
return latestChapter{}, false
|
|
}
|
|
m := novelfullSlugRe.FindStringSubmatch(u.Path)
|
|
if m == nil {
|
|
return latestChapter{}, false
|
|
}
|
|
re := regexp.MustCompile(`/` + regexp.QuoteMeta(m[1]) + `/chapter-([0-9.]+)`)
|
|
return maxChapter(re, body)
|
|
}
|
|
|
|
// lnwLatestChapter truncates the body at the comment thread before scanning:
|
|
// it is the one region of the page any visitor can write to (see lnwChapterRe).
|
|
// A body without the marker is skipped, never scanned whole — a redesign must
|
|
// degrade into staleness, not into a wrong shared value; the logged body length
|
|
// tells a markup change from a body the size cap cut short.
|
|
func lnwLatestChapter(seriesURL, body string) (latestChapter, bool) {
|
|
i := strings.Index(body, lnwCommentMarker)
|
|
if i < 0 {
|
|
log.Printf("latest poll %q: no %s marker in %d bytes", seriesURL, lnwCommentMarker, len(body))
|
|
return latestChapter{}, false
|
|
}
|
|
return maxChapter(lnwChapterRe, body[:i])
|
|
}
|
|
|
|
// latestChapterFrom returns the highest chapter number body advertises for this
|
|
// series, via the Site's registry entry. ok is false when the body yields
|
|
// nothing usable — an unknown site, an empty body, a Cloudflare challenge page,
|
|
// and a site redesign all land here, and the caller treats all four identically.
|
|
func latestChapterFrom(site, seriesURL, body string) (latestChapter, bool) {
|
|
if fn := sites[site].LatestChapter; fn != nil {
|
|
return fn(seriesURL, body)
|
|
}
|
|
return latestChapter{}, false
|
|
}
|
|
|
|
var metaTagRe = regexp.MustCompile(`(?is)<meta\b[^>]*>`)
|
|
var doubleQuotedMetaAttrRe = regexp.MustCompile(`(?is)([a-z][a-z0-9:_-]*)\s*=\s*"([^"]*)"`)
|
|
var singleQuotedMetaAttrRe = regexp.MustCompile(`(?is)([a-z][a-z0-9:_-]*)\s*=\s*'([^']*)'`)
|
|
|
|
// comix's server-rendered page embeds query data in this JSON script; parsing
|
|
// the target detail entry avoids matching posters from recommended results.
|
|
var comixInitialDataRe = regexp.MustCompile(`(?is)<script\b[^>]*\bid\s*=\s*["']initial-data["'][^>]*>(.*?)</script>`)
|
|
|
|
// kaganeImageURLRe matches the canonical compressed image route kagane's API
|
|
// publishes — the only cover URL form the extractor emits and the browser
|
|
// fetcher accepts. The URL is matched in full (scheme, host, id shape) rather
|
|
// than trusted: the value a fetcher is pointed at may have been client-
|
|
// supplied, and a headless browser is a strong SSRF primitive.
|
|
var kaganeImageURLRe = regexp.MustCompile(`^https://kagane\.to/api/v2/image/([0-9a-f-]{36})/compressed$`)
|
|
|
|
// comixImageURLRe matches comix's cover host and path shape. Pinned in full
|
|
// (scheme, host, path characters, image extension) for the same reason
|
|
// kaganeImageURLRe is: the address reaches a headless browser, and it can
|
|
// originate in a client-supplied PUT body. No dot is allowed inside the path,
|
|
// so no traversal or second extension can hide in it. Shape from a live page,
|
|
// 2026-08-10: /039d/i/1/34/6a6742bf15736@280.jpg.
|
|
var comixImageURLRe = regexp.MustCompile(`^https://static\.comix\.to/[A-Za-z0-9@/_-]+\.(?:jpg|jpeg|png|webp)$`)
|
|
|
|
// browserOnlyCoverURL reports whether the browser sidecar is the only fetcher
|
|
// for cover bytes at imageURL. kagane's image route answers a plain fetch with
|
|
// a challenge and `cross-origin-resource-policy: same-origin`, and
|
|
// static.comix.to answers one with the same Cloudflare challenge its pages
|
|
// serve (measured 2026-08-12, issue #98), so a TLS fetch would only ever
|
|
// retrieve a challenge page and must not be attempted (ADR-0007). This is the
|
|
// byte-fetch router's per-Site knowledge; it lives in the extraction module,
|
|
// which owns those URL shapes.
|
|
func browserOnlyCoverURL(imageURL string) bool {
|
|
return kaganeImageURLRe.MatchString(imageURL) ||
|
|
comixImageURLRe.MatchString(imageURL)
|
|
}
|
|
|
|
// kagane's browser-fetched series response publishes cover image IDs under
|
|
// series_covers. The API's canonical compressed image route is the only URL
|
|
// form accepted by the store and browser fetcher; no rendition is guessed.
|
|
func kaganeCoverURL(body string) string {
|
|
var response struct {
|
|
SeriesCovers []struct {
|
|
ImageID string `json:"image_id"`
|
|
} `json:"series_covers"`
|
|
}
|
|
if err := json.Unmarshal([]byte(body), &response); err != nil {
|
|
return ""
|
|
}
|
|
for _, cover := range response.SeriesCovers {
|
|
// Validate the assembled URL against the same regex the browser
|
|
// fetcher enforces, so the extractor can never emit an address the
|
|
// fetch would refuse.
|
|
imageURL := "https://kagane.to/api/v2/image/" + cover.ImageID + "/compressed"
|
|
if kaganeImageURLRe.MatchString(imageURL) {
|
|
return imageURL
|
|
}
|
|
}
|
|
return ""
|
|
}
|
|
|
|
// comixDetailQuery returns the ["manga","detail","<id>"] query entry of
|
|
// comix's initial-data JSON, or nil. The cover and completed reads share the
|
|
// scoped lookup so a "recommended" strip entry can never contribute either
|
|
// answer.
|
|
func comixDetailQuery(seriesURL, body string) json.RawMessage {
|
|
id, ok := comixSeriesID(seriesURL)
|
|
if !ok {
|
|
return nil
|
|
}
|
|
data := comixInitialDataRe.FindStringSubmatch(body)
|
|
if data == nil {
|
|
return nil
|
|
}
|
|
var state struct {
|
|
Queries map[string]json.RawMessage `json:"queries"`
|
|
}
|
|
if err := json.Unmarshal([]byte(data[1]), &state); err != nil {
|
|
return nil
|
|
}
|
|
return state.Queries[`["manga","detail","`+id+`"]`]
|
|
}
|
|
|
|
func comixCoverURL(seriesURL, body string) string {
|
|
detail := comixDetailQuery(seriesURL, body)
|
|
if len(detail) == 0 {
|
|
return ""
|
|
}
|
|
var entry struct {
|
|
Poster struct {
|
|
Medium string `json:"medium"`
|
|
} `json:"poster"`
|
|
}
|
|
if err := json.Unmarshal(detail, &entry); err != nil {
|
|
return ""
|
|
}
|
|
return publishedCoverURL(entry.Poster.Medium)
|
|
}
|
|
|
|
// ogImageCover reads the og:image metadata shared by asura, demonic and
|
|
// lightnovelworld.
|
|
func ogImageCover(_, body string) (string, bool) {
|
|
cover := metaContent(body, "property", "og:image")
|
|
return cover, cover != ""
|
|
}
|
|
|
|
func novelfullCoverEntry(_, body string) (string, bool) {
|
|
cover := metaContent(body, "name", "image")
|
|
return cover, cover != ""
|
|
}
|
|
|
|
func comixCoverEntry(seriesURL, body string) (string, bool) {
|
|
cover := comixCoverURL(seriesURL, body)
|
|
return cover, cover != ""
|
|
}
|
|
|
|
func kaganeCoverEntry(_, body string) (string, bool) {
|
|
cover := kaganeCoverURL(body)
|
|
return cover, cover != ""
|
|
}
|
|
|
|
// coverFrom reports false for unknown sites, challenge bodies, and pages with
|
|
// no usable cover, via the Site's registry entry.
|
|
func coverFrom(site, seriesURL, body string) (string, bool) {
|
|
if fn := sites[site].Cover; fn != nil {
|
|
return fn(seriesURL, body)
|
|
}
|
|
return "", false
|
|
}
|
|
|
|
// asuraStatusRe matches the status value inside the escaped astro-island
|
|
// props blob, in both the " form the served document carries and the "
|
|
// form a decoded copy would. "completed" is the only true value: "dropped"
|
|
// is scanlation editorial (the work itself continues elsewhere) and hiatus is
|
|
// its own value.
|
|
var asuraStatusRe = regexp.MustCompile(`(?:"|")status(?:"|"):\[0,(?:"|")completed(?:"|")\]`)
|
|
|
|
func asuraCompleted(_, body string) bool {
|
|
return asuraStatusRe.MatchString(body)
|
|
}
|
|
|
|
// demonicStatusRe matches the info block's status pair: a Status label <li>
|
|
// immediately followed by the value <li>. The site's whole status vocabulary
|
|
// is {Ongoing, Completed} (its advanced-search status filter), so the literal
|
|
// Completed value is the entire signal.
|
|
var demonicStatusRe = regexp.MustCompile(`<li[^>]*>\s*Status\s*</li>\s*<li[^>]*>\s*Completed\s*</li>`)
|
|
|
|
func demonicCompleted(_, body string) bool {
|
|
return demonicStatusRe.MatchString(body)
|
|
}
|
|
|
|
// comixCompleted reads "status" from the scoped detail entry only;
|
|
// "finished" is the completed value, and on_hiatus and discontinued are
|
|
// distinct values.
|
|
func comixCompleted(seriesURL, body string) bool {
|
|
detail := comixDetailQuery(seriesURL, body)
|
|
if len(detail) == 0 {
|
|
return false
|
|
}
|
|
var entry struct {
|
|
Status string `json:"status"`
|
|
}
|
|
if err := json.Unmarshal(detail, &entry); err != nil {
|
|
return false
|
|
}
|
|
return entry.Status == "finished"
|
|
}
|
|
|
|
// kaganeCompleted reads publication_status only: upload_status is the
|
|
// release's state, and the two provably diverge ('Cause Calypso Can,
|
|
// 2026-08-19: publication Ongoing, upload Hiatus), so a Completed upload
|
|
// must never read as a Completed work.
|
|
func kaganeCompleted(_, body string) bool {
|
|
var series struct {
|
|
PublicationStatus string `json:"publication_status"`
|
|
}
|
|
if err := json.Unmarshal([]byte(body), &series); err != nil {
|
|
return false
|
|
}
|
|
return series.PublicationStatus == "Completed"
|
|
}
|
|
|
|
// novelfullStatusRe matches the info panel's status link. The page's whole
|
|
// status vocabulary is {Ongoing, Completed} (the "OnGoing" spelling aliases
|
|
// "Ongoing" on the taxonomy), so the Completed href is the signal.
|
|
var novelfullStatusRe = regexp.MustCompile(`href="/status/Completed"`)
|
|
|
|
func novelfullCompleted(_, body string) bool {
|
|
return novelfullStatusRe.MatchString(body)
|
|
}
|
|
|
|
// lnwCompleted matches creativeWorkStatus in the head's JSON-LD block. The
|
|
// whole body is scanned and the comment marker is not required, unlike
|
|
// lnwLatestChapter: the status block sits ahead of the visitor-writable
|
|
// thread, and hiatus maps to a distinct PotentialActionStatus value.
|
|
var lnwStatusRe = regexp.MustCompile(`"creativeWorkStatus"\s*:\s*"https://schema\.org/CompletedActionStatus"`)
|
|
|
|
func lnwCompleted(_, body string) bool {
|
|
return lnwStatusRe.MatchString(body)
|
|
}
|
|
|
|
// siteCompletedFrom reports whether the Site calls this work completed, via
|
|
// the Site's registry entry. False for an unknown site, a challenge body and
|
|
// a redesign alike: an absent hint, never a claim.
|
|
func siteCompletedFrom(site, seriesURL, body string) bool {
|
|
if fn := sites[site].Completed; fn != nil {
|
|
return fn(seriesURL, body)
|
|
}
|
|
return false
|
|
}
|
|
|
|
// metaContent returns the content of the first <meta> whose attrName is
|
|
// attrValue. It keeps scanning after an empty match so a later published cover
|
|
// is not hidden by an empty tag.
|
|
func metaContent(body, attrName, attrValue string) string {
|
|
for _, tag := range metaTagRe.FindAllString(body, -1) {
|
|
attrs := make(map[string]string)
|
|
for _, m := range doubleQuotedMetaAttrRe.FindAllStringSubmatch(tag, -1) {
|
|
attrs[strings.ToLower(m[1])] = m[2]
|
|
}
|
|
for _, m := range singleQuotedMetaAttrRe.FindAllStringSubmatch(tag, -1) {
|
|
attrs[strings.ToLower(m[1])] = m[2]
|
|
}
|
|
if strings.EqualFold(attrs[strings.ToLower(attrName)], attrValue) {
|
|
if cover := publishedCoverURL(attrs["content"]); cover != "" {
|
|
return cover
|
|
}
|
|
}
|
|
}
|
|
return ""
|
|
}
|
|
|
|
func publishedCoverURL(value string) string {
|
|
value = strings.TrimSpace(html.UnescapeString(value))
|
|
return strings.ReplaceAll(value, " ", "%20")
|
|
}
|
|
|
|
// Poll Lane constants (issue #100). The per-Site structure is deliberately
|
|
// uniform at first — every Site rests an hour and gaps ten seconds — but it
|
|
// exists so a single Site can be slowed if it turns hostile, and the numbers
|
|
// stay in the registry so the structure has a place to differ.
|
|
const (
|
|
// defaultRest is how long every Series rests between Polls.
|
|
defaultRest = time.Hour
|
|
// defaultGap is the strictest pace of every Lane unless the eligible
|
|
// Series count forces it tighter.
|
|
defaultGap = 10 * time.Second
|
|
// minGap floors the effective gap. One request per second is already an
|
|
// order of magnitude past the strictest rate rule a free-plan Site can
|
|
// express (docs/research/cloudflare-bot-scoring-and-poll-cadence.md);
|
|
// below it the Lane is outrunning its own plan and says so loudly.
|
|
minGap = time.Second
|
|
// RefuseBackoff is how long a Lane waits after its Site refused twice in
|
|
// one run before attempting it again. Exported so the web layer can derive
|
|
// browser reachability from the pass log over the same window (issue #145).
|
|
RefuseBackoff = 15 * time.Minute
|
|
// browserWakeCount and browserWakeAge gate a browser Lane's run: five or
|
|
// more due Series, or any one of them waiting this long, or Chrome stays
|
|
// asleep (ADR-0005 on-demand browser).
|
|
browserWakeCount = 5
|
|
browserWakeAge = 15 * time.Minute
|
|
// sightingCeilingRests caps Sighting deferral (issue #103): however many
|
|
// Sightings arrive, a Series unpolled for this many of its Site's rests is
|
|
// Polled. It is what makes a client report safe to trust — a wrong Latest
|
|
// Chapter dies within the ceiling deterministically, rather than in
|
|
// expectation the way a randomised audit would have it. Six, so a Series a
|
|
// Reader visits constantly still gets one authoritative check per working
|
|
// day-part.
|
|
sightingCeilingRests = 6
|
|
)
|
|
|
|
// effectiveGap is a Site's pace: the registry gap, or one rest divided by the
|
|
// eligible Series count when that is smaller, never below one second. The
|
|
// denominator follows defaultRest rather than a literal hour so a Site whose
|
|
// rest is ever changed keeps its per-Series pace in step. The second return is
|
|
// true when the one-second floor engaged (and the Lane logs a warning naming
|
|
// the Site, every round it does).
|
|
func effectiveGap(s site, eligible int) (time.Duration, bool) {
|
|
gap := s.Gap
|
|
if eligible > 0 {
|
|
if perSeries := defaultRest / time.Duration(eligible); perSeries < gap {
|
|
gap = perSeries
|
|
}
|
|
}
|
|
if gap < minGap {
|
|
return minGap, true
|
|
}
|
|
return gap, false
|
|
}
|
|
|
|
// sites is the registry: one entry per Site, keyed by the stored site string.
|
|
// Adding a Site means adding an entry here and nowhere else — the dispatch
|
|
// functions above and the poller's route list are lookups into this map. An
|
|
// unknown site string resolves to the zero entry, which fails the existing
|
|
// not-fetchable and no-fetcher paths unchanged.
|
|
var sites = map[string]site{
|
|
"asura": {
|
|
Host: "asurascans.com",
|
|
LatestChapter: asuraLatestChapter,
|
|
Cover: ogImageCover,
|
|
Completed: asuraCompleted,
|
|
Rest: defaultRest,
|
|
Gap: defaultGap,
|
|
},
|
|
"demonic": {
|
|
Host: "demonicscans.org",
|
|
LatestChapter: demonicLatestChapter,
|
|
Cover: ogImageCover,
|
|
Completed: demonicCompleted,
|
|
Rest: defaultRest,
|
|
Gap: defaultGap,
|
|
},
|
|
"comix": {
|
|
Host: "comix.to",
|
|
LatestChapter: comixLatestChapter,
|
|
Cover: comixCoverEntry,
|
|
Completed: comixCompleted,
|
|
Rest: defaultRest,
|
|
Gap: defaultGap,
|
|
Browser: &browserRead{
|
|
Read: comixRead,
|
|
// The interstitial is served in place of the page, so "arrived"
|
|
// has to exclude it explicitly, as novelfull's does.
|
|
Done: func(body string) bool { return body != "" && !isInterstitial(body) },
|
|
// Never falls back: a plain fetch of a comix page or cover
|
|
// retrieves only a challenge page (measured 2026-08-12).
|
|
Fallback: false,
|
|
},
|
|
},
|
|
"kagane": {
|
|
Host: "kagane.to",
|
|
LatestChapter: kaganeLatestChapter,
|
|
Cover: kaganeCoverEntry,
|
|
Completed: kaganeCompleted,
|
|
Rest: defaultRest,
|
|
Gap: defaultGap,
|
|
Browser: &browserRead{
|
|
Read: kaganeRead,
|
|
Done: func(body string) bool { return body != "" },
|
|
// Never falls back: a plain fetch of a kagane page or cover would
|
|
// only ever retrieve a challenge page (verified 2026-08-03).
|
|
Fallback: false,
|
|
},
|
|
},
|
|
"novelfull": {
|
|
Host: "novelfull.com",
|
|
LatestChapter: novelfullLatestChapter,
|
|
Cover: novelfullCoverEntry,
|
|
Completed: novelfullCompleted,
|
|
Rest: defaultRest,
|
|
Gap: defaultGap,
|
|
Browser: &browserRead{
|
|
Read: novelfullRead,
|
|
// The interstitial has a DOM too, so "the payload arrived" has to
|
|
// exclude it explicitly.
|
|
Done: func(body string) bool { return body != "" && !isInterstitial(body) },
|
|
Fallback: true,
|
|
},
|
|
},
|
|
"lightnovelworld": {
|
|
Host: "lightnovelworld.net",
|
|
LatestChapter: lnwLatestChapter,
|
|
Cover: ogImageCover,
|
|
Completed: lnwCompleted,
|
|
Rest: defaultRest,
|
|
Gap: defaultGap,
|
|
},
|
|
}
|
|
|
|
// SiteNames returns every registry Site, sorted. The admin Series list's Site
|
|
// select needs the full registry, not just the Sites that have rows, and
|
|
// laneNames() is the poller's copy of the same list — both read this.
|
|
func SiteNames() []string {
|
|
names := make([]string, 0, len(sites))
|
|
for name := range sites {
|
|
names = append(names, name)
|
|
}
|
|
sort.Strings(names)
|
|
return names
|
|
}
|
|
|
|
// browserBackedSites is derived from the registry: the Sites whose pages are
|
|
// read through the browser sidecar. Sorted so callers that range it (the
|
|
// browser fetcher's dispatch) see a stable order instead of map-iteration
|
|
// noise.
|
|
func browserBackedSites() []string {
|
|
out := make([]string, 0, len(sites))
|
|
for name, s := range sites {
|
|
if s.Browser != nil {
|
|
out = append(out, name)
|
|
}
|
|
}
|
|
sort.Strings(out)
|
|
return out
|
|
}
|