Files
mangaBookmark/backend/internal/latest/poller.go
T

240 lines
8.7 KiB
Go

package latest
import (
"context"
"log"
"net/url"
"slices"
"time"
"bookmarkmanager/backend/internal/store"
)
// Fetcher retrieves a series page. It exists as an interface so tests can inject
// a fake: nothing in the test suite may touch the network or the TLS client.
type Fetcher interface {
Get(ctx context.Context, url string) (body string, status int, err error)
}
// Poller re-checks each bookmarked series' newest published chapter on a
// schedule, independent of the userscript's own in-browser checks. The two run
// in parallel and report the same observable fact, so whichever writes last wins
// and neither needs to know about the other.
//
// Two clocks, deliberately independent:
//
// - Interval is how often this goroutine wakes up and looks.
// - Cooldowns are how long a series rests since its own last check. Browser-
// backed sites use the longer BrowserCooldown.
//
// Cooldowns are enforced by the WHERE clause in DueForLatestCheck rather than
// by any timer. Shortening Interval therefore cannot shorten anyone's cooldown;
// it only makes the poller wake up and find nothing due more often.
type Poller struct {
Store *store.Store
Fetch Fetcher
// BrowserFetch handles sites behind a JavaScript challenge that Fetch
// cannot clear. Nil disables those sites entirely rather than falling back
// to Fetch, which would only ever retrieve a challenge page.
BrowserFetch Fetcher
Now func() time.Time // injected so tests can freeze it
Cooldown time.Duration
BrowserCooldown time.Duration
Interval time.Duration
Stagger time.Duration
Batch int
}
var browserBackedSites = []string{"kagane", "novelfull"}
// fetcherFor returns the fetcher a site needs, or nil when the site cannot be
// fetched at all right now. kagane and novelfull both sit behind a Cloudflare
// JavaScript challenge that no TLS fingerprint clears — kagane verified
// 2026-08-03, novelfull verified 2026-08-05, both against the same Chrome_133
// profile TLSFetcher uses — so they are browser-only or nothing.
func (p *Poller) fetcherFor(site string) Fetcher {
if slices.Contains(browserBackedSites, site) {
return p.BrowserFetch
}
return p.Fetch
}
// Run polls until ctx is cancelled.
//
// runOnce is called synchronously, so a batch that overruns the tick delays the
// next one instead of stacking a second batch on top of it. That is the intended
// failure mode for a misconfigured batch x stagger: a slower cadence, never
// concurrent fetch storms.
func (p *Poller) Run(ctx context.Context) {
log.Printf("latest-chapter poller: interval=%s cooldown=%s browser-cooldown=%s batch=%d stagger=%s",
p.Interval, p.Cooldown, p.BrowserCooldown, p.Batch, p.Stagger)
t := time.NewTicker(p.Interval)
defer t.Stop()
for {
select {
case <-ctx.Done():
log.Println("latest-chapter poller: stopped")
return
case <-t.C:
p.runOnce(ctx)
}
}
}
// runOnce processes one batch of due series.
func (p *Poller) runOnce(ctx context.Context) {
now := p.Now()
cutoff := now.Add(-p.Cooldown).UnixMilli()
browserCutoff := now.Add(-p.BrowserCooldown).UnixMilli()
due, err := p.Store.DueForLatestCheck(cutoff, browserCutoff, browserBackedSites, p.Batch)
if err != nil {
log.Printf("latest poll: due query: %v", err)
return
}
checked := 0
for i, sr := range due {
if ctx.Err() != nil {
break
}
// Staggered rather than fired together: a burst of simultaneous requests
// from one server IP is the traffic shape most likely to move that IP's
// bot score. This is the server-side analogue of the userscript's "one
// series per navigation ... indistinguishable from browsing" (L455-456).
stopped := false
if i > 0 && p.Stagger > 0 {
select {
case <-ctx.Done():
stopped = true
case <-time.After(p.Stagger):
}
}
if stopped {
break
}
p.checkOne(ctx, sr)
checked++
}
// due vs checked is how you tell which constraint is binding: ticks that
// report due=0 mean the cooldown is the limit, ticks that report due==batch
// every time mean throughput is.
log.Printf("latest poll: due=%d checked=%d", len(due), checked)
}
// checkOne re-checks one series. Every failure path here is "log and move on":
// the poller is a best-effort enhancement, and no single bad series may stall a
// batch or take down the process.
func (p *Poller) checkOne(ctx context.Context, sr store.Series) {
defer func() {
if r := recover(); r != nil {
log.Printf("latest poll %q: recovered from panic: %v", sr.Key(), r)
}
}()
// Stamped before the fetch, not after, so an error, a timeout, or a shutdown
// mid-request still consumes the cooldown. Otherwise a renamed or deleted
// series would be retried on every single tick forever. The userscript
// stamps in the same order and for the same reason (L471-473).
if err := p.Store.MarkLatestChecked(sr.Site, sr.SeriesID, p.Now().UnixMilli()); err != nil {
log.Printf("latest poll %q: mark checked: %v", sr.Key(), err)
return
}
// series_url is client-supplied (PUT /bookmarks/{key} accepts any string),
// so this is not just an optimisation against burning a request on an
// unknown site: without it, the server would issue a GET from its own
// network position to whatever URL a token-holder writes, including
// link-local/internal addresses or non-https schemes. The cooldown above
// is already consumed, so a row that never passes this check is retried at
// cooldown pace rather than hot-looping.
if !fetchableSeriesURL(sr.Site, sr.SeriesURL) {
log.Printf("latest poll %q: not fetchable: site=%q url=%q", sr.Key(), sr.Site, sr.SeriesURL)
return
}
f := p.fetcherFor(sr.Site)
if f == nil {
log.Printf("latest poll %q: no fetcher for site %q", sr.Key(), sr.Site)
return
}
body, status, err := f.Get(ctx, sr.SeriesURL)
if err != nil {
log.Printf("latest poll %q: fetch %s: %v", sr.Key(), sr.SeriesURL, err)
return
}
if status != 200 {
log.Printf("latest poll %q: fetch %s: status %d", sr.Key(), sr.SeriesURL, status)
return
}
latest, ok := latestChapterFrom(sr.Site, sr.SeriesURL, body)
if !ok {
// Most likely a challenge page or a layout change. Either way the row is
// already stamped, so this waits out a cooldown instead of hot-looping.
log.Printf("latest poll %q: no chapter links in %d bytes", sr.Key(), len(body))
return
}
// Equality, not >, mirroring the userscript (L427): a site that retracts a
// chapter should correct the stored number downward. The comparison is
// against the due-query snapshot; a concurrent write in between only costs
// one redundant UPDATE of the same absolute value, never a wrong one.
if sr.LatestChapterNum != nil && *sr.LatestChapterNum == latest.Num {
return
}
// Series-level write: the row is shared, so one update refreshes every
// bookmark joining to it, and the bookmark's updated_at is never touched —
// a newly published chapter is not reading progress and must not reorder
// the list.
if err := p.Store.SetLatestChapter(sr.Site, sr.SeriesID, latest.Label, latest.Num); err != nil {
log.Printf("latest poll %q: set latest chapter: %v", sr.Key(), err)
return
}
log.Printf("latest poll %q: latest is now %s", sr.Key(), latest.Label)
}
// fetchableSeriesURL reports whether site is a site latestChapterFrom knows how
// to parse and seriesURL is safe to hand to a fetcher: an https URL with a
// non-empty host. series_url comes from client-supplied PUT bodies, so this is
// a defence against the poller being used to probe arbitrary hosts from the
// server's own network position, not just a check against wasted requests.
//
// Three sites are held to a stricter rule, each for a different reason:
//
// - kagane and novelfull are fetched by a headless browser, which executes
// JavaScript and carries cookies, and is therefore a far stronger SSRF
// primitive than an HTTP GET. Their hosts must match exactly, not merely
// be non-empty.
// - lightnovelworld's parser regex hardcodes its host, so a URL anywhere
// else could never yield a match — reject it here rather than burn the
// request.
func fetchableSeriesURL(site, seriesURL string) bool {
switch site {
case "asura", "demonic", "comix", "kagane", "novelfull", "lightnovelworld":
default:
return false
}
u, err := url.Parse(seriesURL)
if err != nil {
return false
}
if u.Scheme != "https" || u.Host == "" {
return false
}
switch site {
case "kagane":
return u.Hostname() == "kagane.to"
case "novelfull":
// Fetched by a real browser, same as kagane, so the host is pinned
// rather than merely non-empty.
return u.Hostname() == "novelfull.com"
case "lightnovelworld":
// Its parser regex hardcodes this host, so a URL anywhere else could
// never yield a match — reject it here rather than burn the request.
return u.Hostname() == "lightnovelworld.net"
}
return true
}