Browser-backed Sites join the Cover pipeline (#62)

This commit is contained in:
2026-08-10 10:49:51 +07:00
parent b9220b3dfc
commit 40ce68b7ab
7 changed files with 315 additions and 31 deletions
+43 -8
View File
@@ -32,11 +32,23 @@ const acquireTimeout = 45 * time.Second
// left blank until the poll's own cover pass (#61) fills it.
type Acquirer struct {
Store *store.Store
// Fetch retrieves the series page. Nil disables acquisition entirely.
// Fetch retrieves the series page over plain TLS. Nil with a nil
// BrowserFetch disables acquisition entirely.
Fetch Fetcher
// BrowserFetch retrieves kagane and novelfull pages through the browser
// sidecar, which is the only thing that clears their Cloudflare
// challenge. Nil leaves those Sites unacquired; kagane never falls back
// to Fetch (a plain request only retrieves a challenge page), while
// novelfull does, because its challenge is a live time-varying fact and
// its cover bytes never need the browser.
BrowserFetch Fetcher
// Covers retrieves the cover bytes. Nil leaves the Cover blank and the
// chapter half working.
Covers CoverBytesFetcher
// BrowserCoverFetch retrieves kagane cover bytes through the browser
// sidecar. Nil leaves kagane Covers blank; nothing falls back to a plain
// fetch, which would only ever retrieve a challenge page.
BrowserCoverFetch BrowserCoverFetcher
// Ctx cancels in-flight acquisitions at shutdown. A hook signature has
// nowhere to pass one, so it lives here; nil means context.Background.
Ctx context.Context
@@ -85,10 +97,7 @@ func (a *Acquirer) Acquire(sr store.Series) {
func (a *Acquirer) Wait() { a.inflight.Wait() }
func (a *Acquirer) acquire(ctx context.Context, sr store.Series) {
// Browser-backed Sites are deliberately not acquired here: their pages
// only yield a Cloudflare challenge to the TLS client, so the request
// would be spent for nothing.
if a.Fetch == nil || slices.Contains(browserBackedSites, sr.Site) {
if a.Fetch == nil && a.BrowserFetch == nil {
return
}
// series_url arrives in a client-supplied PUT body, so the same gate the
@@ -99,7 +108,12 @@ func (a *Acquirer) acquire(ctx context.Context, sr store.Series) {
return
}
body, status, err := a.Fetch.Get(ctx, sr.SeriesURL)
f := a.fetcherFor(sr.Site)
if f == nil {
log.Printf("acquire %q: no fetcher for site %q", sr.Key(), sr.Site)
return
}
body, status, err := f.Get(ctx, sr.SeriesURL)
if err != nil {
log.Printf("acquire %q: fetch %s: %v", sr.Key(), sr.SeriesURL, err)
return
@@ -122,10 +136,10 @@ func (a *Acquirer) acquire(ctx context.Context, sr store.Series) {
}
cover, ok := coverFrom(sr.Site, sr.SeriesURL, body)
if !ok || a.Covers == nil {
if !ok {
return
}
bytes, contentType, err := a.Covers.Fetch(ctx, cover)
bytes, contentType, err := fetchCoverBytes(ctx, sr.Site, cover, a.BrowserCoverFetch, a.Covers)
if err != nil {
log.Printf("acquire %q: fetch cover %s: %v", sr.Key(), cover, err)
return
@@ -134,3 +148,24 @@ func (a *Acquirer) acquire(ctx context.Context, sr store.Series) {
log.Printf("acquire %q: persist cover: %v", sr.Key(), err)
}
}
// fetcherFor returns the fetcher a site's page needs, or nil when the site
// cannot be fetched at all right now. kagane and novelfull pages sit behind a
// Cloudflare JavaScript challenge, so they prefer the browser; novelfull alone
// falls back to the plain-TLS fetcher when no browser is configured, because
// its challenge is a live time-varying fact (AGENTS.md) and its cover bytes
// never need the browser. kagane never falls back: a plain fetch of a kagane
// page or cover would only ever retrieve a challenge page.
func (a *Acquirer) fetcherFor(site string) Fetcher {
switch {
case site == "kagane":
return a.BrowserFetch
case slices.Contains(browserBackedSites, site): // novelfull
if a.BrowserFetch != nil {
return a.BrowserFetch
}
return a.Fetch
default:
return a.Fetch
}
}