package latest import ( "encoding/json" "html" "log" "net/url" "regexp" "sort" "strconv" "strings" "time" "github.com/chromedp/chromedp" ) // latestChapter is the newest chapter a series page advertises. type latestChapter struct { Num float64 Label string } // site answers the fixed questions every series-page read asks of its Site // (ADR-0009): the host its addresses must carry, how to find the Latest // Chapter and the Cover address in a body, and — for a Site behind a // JavaScript challenge — how to read its payload from a cleared tab. One // entry describes everything about one Site, and nowhere else gets to compare // the site string. type site struct { // Host is the exact hostname a series_url for this Site must carry. Host string // LatestChapter finds the newest chapter in a fetched body. LatestChapter func(seriesURL, body string) (latestChapter, bool) // Cover finds the Cover address in a fetched body. Cover func(seriesURL, body string) (string, bool) // Rest is how long a Series of this Site rests between Polls. Rest time.Duration // Gap is the Lane's strictest pace: at least one second must pass between // two consecutive Series-page Polls of this Site (issue #100). Gap time.Duration // Browser reads this Site's payload from a cleared browser tab; nil // means the page is fetched over plain TLS. Browser *browserRead } type browserRead struct { // Read builds the tab read for seriesURL, refusing (false) an address // this Site will not open in a browser — the per-Site half of the SSRF // gate, kept deliberately behind fetchableSeriesURL: a headless browser // executes JavaScript and carries cookies, and series_url is // client-supplied. Read func(seriesURL string, out *string) (chromedp.Action, bool) // Done reports whether the payload arrived. Done func(body string) bool // Fallback allows the plain-TLS fetcher when no browser is configured. // False skips the Site instead. kagane and comix are false — a plain fetch // would only ever retrieve a challenge page — and novelfull is true, // because its challenge is a live time-varying fact (AGENTS.md). Fallback bool } // asuraSlugRe pulls the series slug out of a stored series_url. // Shape verified live 2026-07-26: https://asurascans.com/comics/, where // the slug carries a trailing build-hash suffix (e.g. "-f886a8af") that // rotates on every site redeploy — callers must strip it (asuraBuildHash) // before using the slug to scope anything. var asuraSlugRe = regexp.MustCompile(`/comics/([^/?#]+)`) // asuraBuildHash matches the trailing "-xxxxxxxx" site-wide build ID Asura // appends to every series slug. It rotates on each site redeploy, so it is // never part of a stable series_id. Must stay in sync with stripBuildHash in // userscript/manga-bookmark.user.js. var asuraBuildHash = regexp.MustCompile(`-[0-9a-f]{8}$`) // demonicChapterRe matches the pre-redirect anchors demonic series pages link // through. Both the raw "&" and the HTML-escaped "&" forms occur. var demonicChapterRe = regexp.MustCompile(`chaptered\.php\?manga=\d+&(?:amp;)?chapter=([0-9.]+)`) // comixSlugRe pulls the "-" segment out of a stored series_url. // Only the id prefix is stable; the slug tail follows the title. var comixSlugRe = regexp.MustCompile(`/title/([^/?#]+)`) func comixSeriesID(seriesURL string) (string, bool) { m := comixSlugRe.FindStringSubmatch(seriesURL) if m == nil { return "", false } id := m[1] if i := strings.Index(id, "-"); i != -1 { id = id[:i] } return id, true } // kaganeChapterRe matches the chapter numbers in a kagane API response. This // branch is fed by the browser fetcher, so the body is JSON rather than HTML — // there are no anchors to scan. var kaganeChapterRe = regexp.MustCompile(`"chapter_no":"([0-9.]+)"`) // novelfullSlugRe pulls the series slug out of a stored series_url. novelfull // series pages are "/.html"; their chapter anchors are // "//chapter-[-].html". Verified live 2026-08-05. var novelfullSlugRe = regexp.MustCompile(`^/([^/?#]+)\.html$`) // lnwChapterRe matches any chapter-shaped address on lightnovelworld. Unlike // asura, novelfull and comix — which scope to their stored series slug so a // foreign chapter link cannot contribute — this Site's chapter addresses carry // the Chapter Slug, which is not the Series identity: one Series may publish // under several Chapter Slugs (measured 2026-08-11: a sampled novel serves // 1-99 under one slug and 100-423 under another), so no stored-slug pattern can // cover a Series' whole list. An unscoped match is safe because // lnwLatestChapter truncates the body at the comment thread before scanning // (lnwCommentMarker); without that, a visitor's comment could set the Latest // Chapter on the shared Series row. var lnwChapterRe = regexp.MustCompile(`lightnovelworld\.net/[a-z0-9-]+-chapter-([0-9.]+)/`) // lnwCommentMarker is the boundary of lightnovelworld's server-rendered // wpdiscuz comment thread. It occurs exactly once per page and follows every // chapter anchor (measured 2026-08-11, // docs/research/lightnovelworld-chapter-vs-series-slug.md §6), so cutting the // body at its first occurrence keeps the whole chapter list while excluding a // region any visitor can write to. Absent means the page shape changed: the // body is skipped, never scanned whole. const lnwCommentMarker = "wpd-threads" // maxChapter returns the highest chapter number the regex finds in body. A // maximum rather than a first or last, ported from the userscript's // latestChapterFromAnchors (asura L123-133, demonic L183-193): neither site // lists chapters in a dependable order. // // The userscript's asura rule additionally requires the anchor text to match // /Chapter\s+[\d.]+/i. That check exists only to skip the "First Chapter" // shortcut, which points at chapter/1 and therefore can never win a maximum, // so it is redundant once a maximum is taken. func maxChapter(re *regexp.Regexp, body string) (latestChapter, bool) { var best latestChapter found := false for _, m := range re.FindAllStringSubmatch(body, -1) { // [0-9.]+ can swallow a trailing separator, e.g. "chapter/12." in a // sentence; ParseFloat would reject the whole match. raw := strings.Trim(m[1], ".") num, err := strconv.ParseFloat(raw, 64) if err != nil { continue } if !found || num > best.Num { best = latestChapter{Num: num, Label: "Chapter " + raw} found = true } } return best, found } // asuraLatestChapter scopes chapter links to this series' own slug, which // replaces the userscript's anchor-text check with a stronger guarantee: a // chapter link belonging to some other series cannot contribute even if the // page starts carrying them. func asuraLatestChapter(seriesURL, body string) (latestChapter, bool) { m := asuraSlugRe.FindStringSubmatch(seriesURL) if m == nil { return latestChapter{}, false } // Stored URLs predating a redeploy may carry a stale build hash; chapter // hrefs in the fetched body carry the current one. Strip to the stable ID // and make the hash optional in the pattern, so scoping survives // rotations. slug := asuraBuildHash.ReplaceAllString(m[1], "") // Compiled per call rather than cached: this runs once per fetch, which is // at most a few times a minute, and the slug varies per series. re := regexp.MustCompile(`/comics/` + regexp.QuoteMeta(slug) + `(?:-[0-9a-f]{8})?/chapter/([0-9.]+)`) return maxChapter(re, body) } // demonicLatestChapter is not scoped: demonicChapterRe matches any // chaptered.php?manga= anchor, because the stored series_id is a slug, // not the numeric id the URL carries, so it cannot be scoped. func demonicLatestChapter(_, body string) (latestChapter, bool) { return maxChapter(demonicChapterRe, body) } // comixLatestChapter reads comix's SPA: the served HTML carries a JSON state // blob instead of chapter anchors, and latestChapterUrl is the only place the // newest chapter appears. Scoping to this series' id prefix keeps a // "recommended" strip's entries from winning the maximum. func comixLatestChapter(seriesURL, body string) (latestChapter, bool) { id, ok := comixSeriesID(seriesURL) if !ok { return latestChapter{}, false } re := regexp.MustCompile(`"latestChapterUrl":"/title/` + regexp.QuoteMeta(id) + `-[^"]*-chapter-([0-9.]+)"`) return maxChapter(re, body) } // kaganeLatestChapter scans the kagane series API JSON that the browser read // fetched from inside the page; the match rides on the property name, // regardless of the surrounding JSON shape. func kaganeLatestChapter(_, body string) (latestChapter, bool) { return maxChapter(kaganeChapterRe, body) } // novelfullLatestChapter is scoped to this series' slug for the same reason // asura is: page 1 carries a "latest chapters" widget and a "you may also // like" strip, and neither may contribute to the maximum. func novelfullLatestChapter(seriesURL, body string) (latestChapter, bool) { u, err := url.Parse(seriesURL) if err != nil { return latestChapter{}, false } m := novelfullSlugRe.FindStringSubmatch(u.Path) if m == nil { return latestChapter{}, false } re := regexp.MustCompile(`/` + regexp.QuoteMeta(m[1]) + `/chapter-([0-9.]+)`) return maxChapter(re, body) } // lnwLatestChapter truncates the body at the comment thread before scanning: // it is the one region of the page any visitor can write to (see lnwChapterRe). // A body without the marker is skipped, never scanned whole — a redesign must // degrade into staleness, not into a wrong shared value; the logged body length // tells a markup change from a body the size cap cut short. func lnwLatestChapter(seriesURL, body string) (latestChapter, bool) { i := strings.Index(body, lnwCommentMarker) if i < 0 { log.Printf("latest poll %q: no %s marker in %d bytes", seriesURL, lnwCommentMarker, len(body)) return latestChapter{}, false } return maxChapter(lnwChapterRe, body[:i]) } // latestChapterFrom returns the highest chapter number body advertises for this // series, via the Site's registry entry. ok is false when the body yields // nothing usable — an unknown site, an empty body, a Cloudflare challenge page, // and a site redesign all land here, and the caller treats all four identically. func latestChapterFrom(site, seriesURL, body string) (latestChapter, bool) { if fn := sites[site].LatestChapter; fn != nil { return fn(seriesURL, body) } return latestChapter{}, false } var metaTagRe = regexp.MustCompile(`(?is)]*>`) var doubleQuotedMetaAttrRe = regexp.MustCompile(`(?is)([a-z][a-z0-9:_-]*)\s*=\s*"([^"]*)"`) var singleQuotedMetaAttrRe = regexp.MustCompile(`(?is)([a-z][a-z0-9:_-]*)\s*=\s*'([^']*)'`) // comix's server-rendered page embeds query data in this JSON script; parsing // the target detail entry avoids matching posters from recommended results. var comixInitialDataRe = regexp.MustCompile(`(?is)]*\bid\s*=\s*["']initial-data["'][^>]*>(.*?)`) // kaganeImageURLRe matches the canonical compressed image route kagane's API // publishes — the only cover URL form the extractor emits and the browser // fetcher accepts. The URL is matched in full (scheme, host, id shape) rather // than trusted: the value a fetcher is pointed at may have been client- // supplied, and a headless browser is a strong SSRF primitive. var kaganeImageURLRe = regexp.MustCompile(`^https://kagane\.to/api/v2/image/([0-9a-f-]{36})/compressed$`) // comixImageURLRe matches comix's cover host and path shape. Pinned in full // (scheme, host, path characters, image extension) for the same reason // kaganeImageURLRe is: the address reaches a headless browser, and it can // originate in a client-supplied PUT body. No dot is allowed inside the path, // so no traversal or second extension can hide in it. Shape from a live page, // 2026-08-10: /039d/i/1/34/6a6742bf15736@280.jpg. var comixImageURLRe = regexp.MustCompile(`^https://static\.comix\.to/[A-Za-z0-9@/_-]+\.(?:jpg|jpeg|png|webp)$`) // browserOnlyCoverURL reports whether the browser sidecar is the only fetcher // for cover bytes at imageURL. kagane's image route answers a plain fetch with // a challenge and `cross-origin-resource-policy: same-origin`, and // static.comix.to answers one with the same Cloudflare challenge its pages // serve (measured 2026-08-12, issue #98), so a TLS fetch would only ever // retrieve a challenge page and must not be attempted (ADR-0007). This is the // byte-fetch router's per-Site knowledge; it lives in the extraction module, // which owns those URL shapes. func browserOnlyCoverURL(imageURL string) bool { return kaganeImageURLRe.MatchString(imageURL) || comixImageURLRe.MatchString(imageURL) } // kagane's browser-fetched series response publishes cover image IDs under // series_covers. The API's canonical compressed image route is the only URL // form accepted by the store and browser fetcher; no rendition is guessed. func kaganeCoverURL(body string) string { var response struct { SeriesCovers []struct { ImageID string `json:"image_id"` } `json:"series_covers"` } if err := json.Unmarshal([]byte(body), &response); err != nil { return "" } for _, cover := range response.SeriesCovers { // Validate the assembled URL against the same regex the browser // fetcher enforces, so the extractor can never emit an address the // fetch would refuse. imageURL := "https://kagane.to/api/v2/image/" + cover.ImageID + "/compressed" if kaganeImageURLRe.MatchString(imageURL) { return imageURL } } return "" } func comixCoverURL(seriesURL, body string) string { id, ok := comixSeriesID(seriesURL) if !ok { return "" } data := comixInitialDataRe.FindStringSubmatch(body) if data == nil { return "" } var state struct { Queries map[string]json.RawMessage `json:"queries"` } if err := json.Unmarshal([]byte(data[1]), &state); err != nil { return "" } raw := state.Queries[`["manga","detail","`+id+`"]`] if len(raw) == 0 { return "" } var detail struct { Poster struct { Medium string `json:"medium"` } `json:"poster"` } if err := json.Unmarshal(raw, &detail); err != nil { return "" } return publishedCoverURL(detail.Poster.Medium) } // ogImageCover reads the og:image metadata shared by asura, demonic and // lightnovelworld. func ogImageCover(_, body string) (string, bool) { cover := metaContent(body, "property", "og:image") return cover, cover != "" } func novelfullCoverEntry(_, body string) (string, bool) { cover := metaContent(body, "name", "image") return cover, cover != "" } func comixCoverEntry(seriesURL, body string) (string, bool) { cover := comixCoverURL(seriesURL, body) return cover, cover != "" } func kaganeCoverEntry(_, body string) (string, bool) { cover := kaganeCoverURL(body) return cover, cover != "" } // coverFrom reports false for unknown sites, challenge bodies, and pages with // no usable cover, via the Site's registry entry. func coverFrom(site, seriesURL, body string) (string, bool) { if fn := sites[site].Cover; fn != nil { return fn(seriesURL, body) } return "", false } // metaContent returns the content of the first whose attrName is // attrValue. It keeps scanning after an empty match so a later published cover // is not hidden by an empty tag. func metaContent(body, attrName, attrValue string) string { for _, tag := range metaTagRe.FindAllString(body, -1) { attrs := make(map[string]string) for _, m := range doubleQuotedMetaAttrRe.FindAllStringSubmatch(tag, -1) { attrs[strings.ToLower(m[1])] = m[2] } for _, m := range singleQuotedMetaAttrRe.FindAllStringSubmatch(tag, -1) { attrs[strings.ToLower(m[1])] = m[2] } if strings.EqualFold(attrs[strings.ToLower(attrName)], attrValue) { if cover := publishedCoverURL(attrs["content"]); cover != "" { return cover } } } return "" } func publishedCoverURL(value string) string { value = strings.TrimSpace(html.UnescapeString(value)) return strings.ReplaceAll(value, " ", "%20") } // Poll Lane constants (issue #100). The per-Site structure is deliberately // uniform at first — every Site rests an hour and gaps ten seconds — but it // exists so a single Site can be slowed if it turns hostile, and the numbers // stay in the registry so the structure has a place to differ. const ( // defaultRest is how long every Series rests between Polls. defaultRest = time.Hour // defaultGap is the strictest pace of every Lane unless the eligible // Series count forces it tighter. defaultGap = 10 * time.Second // minGap floors the effective gap. One request per second is already an // order of magnitude past the strictest rate rule a free-plan Site can // express (docs/research/cloudflare-bot-scoring-and-poll-cadence.md); // below it the Lane is outrunning its own plan and says so loudly. minGap = time.Second // RefuseBackoff is how long a Lane waits after its Site refused twice in // one run before attempting it again. Exported so the web layer can derive // browser reachability from the pass log over the same window (issue #145). RefuseBackoff = 15 * time.Minute // browserWakeCount and browserWakeAge gate a browser Lane's run: five or // more due Series, or any one of them waiting this long, or Chrome stays // asleep (ADR-0005 on-demand browser). browserWakeCount = 5 browserWakeAge = 15 * time.Minute // sightingCeilingRests caps Sighting deferral (issue #103): however many // Sightings arrive, a Series unpolled for this many of its Site's rests is // Polled. It is what makes a client report safe to trust — a wrong Latest // Chapter dies within the ceiling deterministically, rather than in // expectation the way a randomised audit would have it. Six, so a Series a // Reader visits constantly still gets one authoritative check per working // day-part. sightingCeilingRests = 6 ) // effectiveGap is a Site's pace: the registry gap, or one rest divided by the // eligible Series count when that is smaller, never below one second. The // denominator follows defaultRest rather than a literal hour so a Site whose // rest is ever changed keeps its per-Series pace in step. The second return is // true when the one-second floor engaged (and the Lane logs a warning naming // the Site, every round it does). func effectiveGap(s site, eligible int) (time.Duration, bool) { gap := s.Gap if eligible > 0 { if perSeries := defaultRest / time.Duration(eligible); perSeries < gap { gap = perSeries } } if gap < minGap { return minGap, true } return gap, false } // sites is the registry: one entry per Site, keyed by the stored site string. // Adding a Site means adding an entry here and nowhere else — the dispatch // functions above and the poller's route list are lookups into this map. An // unknown site string resolves to the zero entry, which fails the existing // not-fetchable and no-fetcher paths unchanged. var sites = map[string]site{ "asura": { Host: "asurascans.com", LatestChapter: asuraLatestChapter, Cover: ogImageCover, Rest: defaultRest, Gap: defaultGap, }, "demonic": { Host: "demonicscans.org", LatestChapter: demonicLatestChapter, Cover: ogImageCover, Rest: defaultRest, Gap: defaultGap, }, "comix": { Host: "comix.to", LatestChapter: comixLatestChapter, Cover: comixCoverEntry, Rest: defaultRest, Gap: defaultGap, Browser: &browserRead{ Read: comixRead, // The interstitial is served in place of the page, so "arrived" // has to exclude it explicitly, as novelfull's does. Done: func(body string) bool { return body != "" && !isInterstitial(body) }, // Never falls back: a plain fetch of a comix page or cover // retrieves only a challenge page (measured 2026-08-12). Fallback: false, }, }, "kagane": { Host: "kagane.to", LatestChapter: kaganeLatestChapter, Cover: kaganeCoverEntry, Rest: defaultRest, Gap: defaultGap, Browser: &browserRead{ Read: kaganeRead, Done: func(body string) bool { return body != "" }, // Never falls back: a plain fetch of a kagane page or cover would // only ever retrieve a challenge page (verified 2026-08-03). Fallback: false, }, }, "novelfull": { Host: "novelfull.com", LatestChapter: novelfullLatestChapter, Cover: novelfullCoverEntry, Rest: defaultRest, Gap: defaultGap, Browser: &browserRead{ Read: novelfullRead, // The interstitial has a DOM too, so "the payload arrived" has to // exclude it explicitly. Done: func(body string) bool { return body != "" && !isInterstitial(body) }, Fallback: true, }, }, "lightnovelworld": { Host: "lightnovelworld.net", LatestChapter: lnwLatestChapter, Cover: ogImageCover, Rest: defaultRest, Gap: defaultGap, }, } // browserBackedSites is derived from the registry: the Sites whose pages are // read through the browser sidecar. Sorted so callers that range it (the // browser fetcher's dispatch) see a stable order instead of map-iteration // noise. func browserBackedSites() []string { out := make([]string, 0, len(sites)) for name, s := range sites { if s.Browser != nil { out = append(out, name) } } sort.Strings(out) return out }