Browser-backed Sites join the Cover pipeline (#62)

This commit is contained in:
2026-08-10 10:49:51 +07:00
parent b9220b3dfc
commit 40ce68b7ab
7 changed files with 315 additions and 31 deletions
+43 -8
View File
@@ -32,11 +32,23 @@ const acquireTimeout = 45 * time.Second
// left blank until the poll's own cover pass (#61) fills it.
type Acquirer struct {
Store *store.Store
// Fetch retrieves the series page. Nil disables acquisition entirely.
// Fetch retrieves the series page over plain TLS. Nil with a nil
// BrowserFetch disables acquisition entirely.
Fetch Fetcher
// BrowserFetch retrieves kagane and novelfull pages through the browser
// sidecar, which is the only thing that clears their Cloudflare
// challenge. Nil leaves those Sites unacquired; kagane never falls back
// to Fetch (a plain request only retrieves a challenge page), while
// novelfull does, because its challenge is a live time-varying fact and
// its cover bytes never need the browser.
BrowserFetch Fetcher
// Covers retrieves the cover bytes. Nil leaves the Cover blank and the
// chapter half working.
Covers CoverBytesFetcher
// BrowserCoverFetch retrieves kagane cover bytes through the browser
// sidecar. Nil leaves kagane Covers blank; nothing falls back to a plain
// fetch, which would only ever retrieve a challenge page.
BrowserCoverFetch BrowserCoverFetcher
// Ctx cancels in-flight acquisitions at shutdown. A hook signature has
// nowhere to pass one, so it lives here; nil means context.Background.
Ctx context.Context
@@ -85,10 +97,7 @@ func (a *Acquirer) Acquire(sr store.Series) {
func (a *Acquirer) Wait() { a.inflight.Wait() }
func (a *Acquirer) acquire(ctx context.Context, sr store.Series) {
// Browser-backed Sites are deliberately not acquired here: their pages
// only yield a Cloudflare challenge to the TLS client, so the request
// would be spent for nothing.
if a.Fetch == nil || slices.Contains(browserBackedSites, sr.Site) {
if a.Fetch == nil && a.BrowserFetch == nil {
return
}
// series_url arrives in a client-supplied PUT body, so the same gate the
@@ -99,7 +108,12 @@ func (a *Acquirer) acquire(ctx context.Context, sr store.Series) {
return
}
body, status, err := a.Fetch.Get(ctx, sr.SeriesURL)
f := a.fetcherFor(sr.Site)
if f == nil {
log.Printf("acquire %q: no fetcher for site %q", sr.Key(), sr.Site)
return
}
body, status, err := f.Get(ctx, sr.SeriesURL)
if err != nil {
log.Printf("acquire %q: fetch %s: %v", sr.Key(), sr.SeriesURL, err)
return
@@ -122,10 +136,10 @@ func (a *Acquirer) acquire(ctx context.Context, sr store.Series) {
}
cover, ok := coverFrom(sr.Site, sr.SeriesURL, body)
if !ok || a.Covers == nil {
if !ok {
return
}
bytes, contentType, err := a.Covers.Fetch(ctx, cover)
bytes, contentType, err := fetchCoverBytes(ctx, sr.Site, cover, a.BrowserCoverFetch, a.Covers)
if err != nil {
log.Printf("acquire %q: fetch cover %s: %v", sr.Key(), cover, err)
return
@@ -134,3 +148,24 @@ func (a *Acquirer) acquire(ctx context.Context, sr store.Series) {
log.Printf("acquire %q: persist cover: %v", sr.Key(), err)
}
}
// fetcherFor returns the fetcher a site's page needs, or nil when the site
// cannot be fetched at all right now. kagane and novelfull pages sit behind a
// Cloudflare JavaScript challenge, so they prefer the browser; novelfull alone
// falls back to the plain-TLS fetcher when no browser is configured, because
// its challenge is a live time-varying fact (AGENTS.md) and its cover bytes
// never need the browser. kagane never falls back: a plain fetch of a kagane
// page or cover would only ever retrieve a challenge page.
func (a *Acquirer) fetcherFor(site string) Fetcher {
switch {
case site == "kagane":
return a.BrowserFetch
case slices.Contains(browserBackedSites, site): // novelfull
if a.BrowserFetch != nil {
return a.BrowserFetch
}
return a.Fetch
default:
return a.Fetch
}
}
+167
View File
@@ -20,6 +20,29 @@ const (
acquireCoverURL = "https://cdn.asurascans.com/asura-images/covers/chronicles-of-the-demon-faction.d4dcb8.webp"
)
const (
kaganeKey = "kagane:019f84bc-9ba0-7ed9-86f5-8b905ec7c28b"
kaganeSeriesID = "019f84bc-9ba0-7ed9-86f5-8b905ec7c28b"
kaganeSeriesURL = "https://kagane.to/series/019f84bc-9ba0-7ed9-86f5-8b905ec7c28b"
kaganeImageID = "019fe11a-84c3-7fc3-a84b-88787374b617"
kaganeCoverSrc = "https://kagane.to/api/v2/image/" + kaganeImageID + "/compressed"
)
// kagane's browser-fetched body is one JSON object carrying both the chapter
// list (series_books) and the cover image ids (series_covers), so the single
// acquisition fetch yields both facts.
const kaganeSeriesAndCoverFixture = `{"series_id":"019f84bc-9ba0-7ed9-86f5-8b905ec7c28b",` +
`"series_books":[{"book_id":"b","title":"Episode 41","chapter_no":"41","sort_no":41}],` +
`"series_covers":[{"cover_id":"019fe11a-84d1-714b-9cf4-2827f277f3c0","language":"en",` +
`"image_id":"019fe11a-84c3-7fc3-a84b-88787374b617"}]}`
const (
novelfullKey = "novelfull:reverend-insanity"
novelfullSeriesID = "reverend-insanity"
novelfullSeriesURI = "https://novelfull.com/reverend-insanity.html"
novelfullCoverURL = "https://novelfull.com/uploads/webp/novel/reverend-insanity-82661d911a.webp"
)
// newAcquirer wires an acquirer onto the store's creation hook, which is how
// main wires it: the write path is what starts an acquisition.
func newAcquirer(s *store.Store, page *fakeFetcher, covers *fakeBytesCoverFetcher) *Acquirer {
@@ -50,6 +73,30 @@ func readBookmark(t *testing.T, s *store.Store, key string) store.Bookmark {
return b
}
func bookmarkNewKaganeSeries(t *testing.T, s *store.Store) store.Bookmark {
t.Helper()
stored, err := s.Upsert(s.OwnerID(), store.Bookmark{
Key: kaganeKey, Site: "kagane", SeriesID: kaganeSeriesID,
Title: "Infinite Decryption", SeriesURL: kaganeSeriesURL, UpdatedAt: 1000,
})
if err != nil {
t.Fatalf("Upsert: %v", err)
}
return stored
}
func bookmarkNewNovelfullSeries(t *testing.T, s *store.Store) store.Bookmark {
t.Helper()
stored, err := s.Upsert(s.OwnerID(), store.Bookmark{
Key: novelfullKey, Site: "novelfull", SeriesID: novelfullSeriesID,
Title: "Reverend Insanity", SeriesURL: novelfullSeriesURI, UpdatedAt: 1000,
})
if err != nil {
t.Fatalf("Upsert: %v", err)
}
return stored
}
// The reported bug: a Reader bookmarks a Series nobody holds and expects the
// Cover, not a broken image. Both facts come from the one series-page fetch.
func TestAcquireFillsChapterAndCoverFromOneFetch(t *testing.T) {
@@ -237,3 +284,123 @@ func TestAcquireDoesNotBlockTheWrite(t *testing.T) {
close(release)
acq.Wait()
}
// The second symptom of #47: a kagane Series bookmarked from a chapter page
// gets its Cover at creation, with the bytes fetched through the browser
// sidecar — the only path that clears the challenge — into the
// content-addressed store.
func TestAcquireKaganeCoverThroughBrowser(t *testing.T) {
s, _ := newTestStore(t)
tlsPage := &fakeFetcher{body: "", status: 403}
browserPage := &fakeFetcher{body: kaganeSeriesAndCoverFixture, status: 200}
covers := &fakeCoverFetcher{body: []byte("cover-bytes"), contentType: "image/webp"}
acq := &Acquirer{
Store: s, Fetch: tlsPage, BrowserFetch: browserPage,
BrowserCoverFetch: covers, Covers: &fakeBytesCoverFetcher{},
}
s.OnSeriesCreated = acq.Acquire
bookmarkNewKaganeSeries(t, s)
acq.Wait()
if got := tlsPage.callCount(); got != 0 {
t.Fatalf("plain-TLS page fetches = %d, want 0 — kagane pages are browser-only", got)
}
if got := browserPage.callCount(); got != 1 {
t.Fatalf("browser page fetches = %d, want 1", got)
}
if got := covers.callCount(); got != 1 {
t.Fatalf("browser cover fetches = %d, want 1", got)
}
if got := covers.calls[0]; got != kaganeImageID {
t.Fatalf("browser cover fetched image id %q, want %q", got, kaganeImageID)
}
got := readBookmark(t, s, kaganeKey)
if want := testCoverBaseURL + "/covers/" + store.CoverAddress(kaganeCoverSrc); got.Cover != want {
t.Fatalf("Cover = %q, want the content-addressed URL %q", got.Cover, want)
}
body, contentType, ok, err := s.CoverByAddress(store.CoverAddress(kaganeCoverSrc))
if err != nil || !ok {
t.Fatalf("CoverByAddress = %v, %v", ok, err)
}
if string(body) != "cover-bytes" || contentType != "image/webp" {
t.Fatalf("stored cover = (%q, %q), want the browser-fetched bytes", body, contentType)
}
}
// novelfull needs the browser only for its HTML: the cover URL comes out of
// the browser-fetched page, but the bytes go over plain TLS through the
// ordinary gated fetcher, never through the browser (issue #62).
func TestAcquireNovelfullCoverOverPlainTLS(t *testing.T) {
s, _ := newTestStore(t)
browserPage := &fakeFetcher{body: novelfullSeriesFixture + novelfullCoverFixture, status: 200}
covers := &fakeBytesCoverFetcher{body: []byte("cover-bytes"), contentType: "image/webp"}
acq := &Acquirer{
Store: s, Fetch: &fakeFetcher{body: "", status: 403},
BrowserFetch: browserPage, Covers: covers,
}
s.OnSeriesCreated = acq.Acquire
bookmarkNewNovelfullSeries(t, s)
acq.Wait()
if got := browserPage.callCount(); got != 1 {
t.Fatalf("browser page fetches = %d, want 1", got)
}
if got := covers.callCount(); got != 1 {
t.Fatalf("cover fetches = %d, want 1 — novelfull bytes never touch the browser", got)
}
if got := covers.calls[0]; got != novelfullCoverURL {
t.Fatalf("cover fetched from %q, want %q", got, novelfullCoverURL)
}
got := readBookmark(t, s, novelfullKey)
if want := testCoverBaseURL + "/covers/" + store.CoverAddress(novelfullCoverURL); got.Cover != want {
t.Fatalf("Cover = %q, want %q", got.Cover, want)
}
}
// With no browser sidecar configured, kagane is simply not acquired: no
// request is spent on a page that could only ever answer with a challenge,
// and nothing falls back to a plain fetch.
func TestAcquireKaganeSkippedWithoutBrowser(t *testing.T) {
s, _ := newTestStore(t)
tlsPage := &fakeFetcher{body: kaganeSeriesAndCoverFixture, status: 200}
acq := &Acquirer{
Store: s, Fetch: tlsPage,
Covers: &fakeBytesCoverFetcher{body: []byte("x"), contentType: "image/webp"},
}
s.OnSeriesCreated = acq.Acquire
bookmarkNewKaganeSeries(t, s)
acq.Wait()
if got := tlsPage.callCount(); got != 0 {
t.Fatalf("plain-TLS fetches for kagane = %d, want 0", got)
}
if got := readBookmark(t, s, kaganeKey); got.Cover != "" {
t.Fatalf("Cover = %q, want empty without a browser", got.Cover)
}
}
// novelfull's no-browser degradation differs from kagane's: only its HTML
// needs the sidecar, so when the page body is available — the challenge is a
// live time-varying fact that sometimes answers a plain request — the Cover
// still lands, bytes over plain TLS.
func TestAcquireNovelfullCoverWithoutBrowser(t *testing.T) {
s, _ := newTestStore(t)
tlsPage := &fakeFetcher{body: novelfullSeriesFixture + novelfullCoverFixture, status: 200}
covers := &fakeBytesCoverFetcher{body: []byte("cover-bytes"), contentType: "image/webp"}
acq := &Acquirer{Store: s, Fetch: tlsPage, Covers: covers}
s.OnSeriesCreated = acq.Acquire
bookmarkNewNovelfullSeries(t, s)
acq.Wait()
if got := covers.callCount(); got != 1 {
t.Fatalf("cover fetches = %d, want 1", got)
}
got := readBookmark(t, s, novelfullKey)
if want := testCoverBaseURL + "/covers/" + store.CoverAddress(novelfullCoverURL); got.Cover != want {
t.Fatalf("Cover = %q, want %q", got.Cover, want)
}
}
+23
View File
@@ -22,6 +22,29 @@ type CoverBytesFetcher interface {
Fetch(ctx context.Context, sourceURL string) (body []byte, contentType string, err error)
}
// fetchCoverBytes routes a cover's byte retrieval by Site: only kagane needs
// the browser for image bytes — its covers answer a plain fetch with a
// challenge and `cross-origin-resource-policy: same-origin` — while every
// other Site's CDN answers plain TLS. Missing fetchers degrade to an error the
// caller logs, never a fallback onto a path that cannot succeed. One routing
// rule for the poll and the acquirer, so the two cannot drift apart.
func fetchCoverBytes(ctx context.Context, site, cover string, browser BrowserCoverFetcher, tls CoverBytesFetcher) ([]byte, string, error) {
if site == "kagane" {
if browser == nil {
return nil, "", errors.New("no cover fetcher")
}
imageID, ok := store.KaganeImageID(cover)
if !ok {
return nil, "", errors.New("invalid kagane cover URL")
}
return browser.Image(ctx, imageID)
}
if tls == nil {
return nil, "", errors.New("no cover fetcher")
}
return tls.Fetch(ctx, cover)
}
// CoverResolver resolves a host before any connection is attempted. Tests
// inject it to exercise hostile DNS results without touching the live network.
type CoverResolver func(context.Context, string) ([]netip.Addr, error)
+1 -22
View File
@@ -2,7 +2,6 @@ package latest
import (
"context"
"errors"
"log"
"net/url"
"slices"
@@ -107,7 +106,7 @@ func (p *Poller) prefetchCover(ctx context.Context, sr store.Series) {
// failure is logged against the Series and swallowed so the chapter poll
// cannot see it.
func (p *Poller) storeCover(ctx context.Context, sr store.Series, sourceURL string) {
bytes, contentType, err := p.fetchCoverBytes(ctx, sr, sourceURL)
bytes, contentType, err := fetchCoverBytes(ctx, sr.Site, sourceURL, p.CoverFetch, p.CoverBytesFetch)
if err != nil {
log.Printf("latest poll %q: fetch cover %s: %v", sr.Key(), sourceURL, err)
return
@@ -117,26 +116,6 @@ func (p *Poller) storeCover(ctx context.Context, sr store.Series, sourceURL stri
}
}
// fetchCoverBytes routes by Site: only kagane needs the browser for image
// bytes; every other Site's CDN answers plain TLS. Missing fetchers degrade to
// a blank Cover rather than falling back onto a path that cannot succeed.
func (p *Poller) fetchCoverBytes(ctx context.Context, sr store.Series, cover string) ([]byte, string, error) {
if sr.Site == "kagane" {
if p.CoverFetch == nil {
return nil, "", errors.New("no cover fetcher")
}
imageID, ok := store.KaganeImageID(cover)
if !ok {
return nil, "", errors.New("invalid kagane cover URL")
}
return p.CoverFetch.Image(ctx, imageID)
}
if p.CoverBytesFetch == nil {
return nil, "", errors.New("no cover fetcher")
}
return p.CoverBytesFetch.Fetch(ctx, cover)
}
// fetcherFor returns the fetcher a site needs, or nil when the site cannot be
// fetched at all right now. kagane and novelfull both sit behind a Cloudflare
// JavaScript challenge that no TLS fingerprint clears — kagane verified
@@ -6,6 +6,8 @@ import (
"os"
"testing"
"time"
"bookmarkmanager/backend/internal/store"
)
// TestSmokeKaganeImage is the live proof that the cover proxy's fetch actually
@@ -92,3 +94,62 @@ func TestSmokeKaganeGet(t *testing.T) {
t.Fatalf("status = %d, want 200 — the sidecar is not clearing the challenge", status)
}
}
// TestSmokeAcquireKaganeCover proves the #62 acquisition path end to end
// against the real browser: a kagane Series bookmarked at creation gets its
// Cover, bytes fetched through the sidecar into the content-addressed store.
// Same SMOKE_BROWSER_WS_URL gate as the tests above; a red run means the
// challenge is not clearing from this IP (a live fact to re-check), not
// necessarily a defect in the pipeline.
func TestSmokeAcquireKaganeCover(t *testing.T) {
ws := os.Getenv("SMOKE_BROWSER_WS_URL")
if ws == "" {
t.Skip("SMOKE_BROWSER_WS_URL unset")
}
const (
seriesID = "019fe11a-8670-7cf3-8343-0b02057d3787"
coverURL = "https://kagane.to/api/v2/image/019fe11a-84c3-7fc3-a84b-88787374b617/compressed"
)
s, _ := newTestStore(t)
bf, err := NewBrowserFetcher(ws)
if err != nil {
t.Fatalf("NewBrowserFetcher: %v", err)
}
defer bf.Close()
tlsF, err := NewTLSFetcher()
if err != nil {
t.Fatalf("NewTLSFetcher: %v", err)
}
acq := &Acquirer{
Store: s, Fetch: tlsF, BrowserFetch: bf,
BrowserCoverFetch: bf, Covers: NewCoverFetcher(),
}
s.OnSeriesCreated = acq.Acquire
if _, err := s.Upsert(s.OwnerID(), store.Bookmark{
Key: "kagane:" + seriesID, Site: "kagane", SeriesID: seriesID,
Title: "smoke", SeriesURL: "https://kagane.to/series/" + seriesID, UpdatedAt: 1000,
}); err != nil {
t.Fatalf("Upsert: %v", err)
}
acq.Wait()
got, found, err := s.Get(s.OwnerID(), "kagane:"+seriesID)
if err != nil || !found {
t.Fatalf("Get: %v found=%v", err, found)
}
if want := testCoverBaseURL + "/covers/" + store.CoverAddress(coverURL); got.Cover != want {
t.Fatalf("Cover = %q, want %q — the acquire path did not store the browser-fetched bytes", got.Cover, want)
}
body, contentType, ok, err := s.CoverByAddress(store.CoverAddress(coverURL))
if err != nil || !ok {
t.Fatalf("CoverByAddress: %v found=%v", err, ok)
}
if len(body) < 1000 {
t.Fatalf("stored cover is %d bytes, want a real image", len(body))
}
if contentType != "image/webp" {
t.Fatalf("content type = %q, want image/webp", contentType)
}
t.Logf("stored %d bytes of %s", len(body), contentType)
}