feat(latest): poll comix through the browser sidecar (#98)

comix.to began answering plain-TLS fetches with a Cloudflare JavaScript
challenge on 2026-08-12, so every poll got a 403 interstitial and its cover
host static.comix.to is gated the same way. comix joins kagane and novelfull
as a browser Site: one registry entry, no plain-TLS fallback, and cover bytes
routed through the browser's image path behind a fully pinned URL pattern.

The read is an in-tab fetch of the Series URL, not a DOM render: comix is an
SPA, so rendering costs ~65 requests for the same server-rendered HTML one
fetch returns (24.5 KB, ~480 ms). Parsers and stored Series identity are
untouched.

Verified live against the real browser unit: page 24793 bytes in one fetch,
chapter 53, cover accepted by the pin and 26862 image bytes retrieved by
direct navigation (comix's Series page sets cross-origin-embedder-policy:
require-corp, so an in-page fetch of the cover host cannot work).
This commit is contained in:
2026-08-16 12:05:11 +07:00
parent 3f53c79cf4
commit 86160c164a
11 changed files with 336 additions and 55 deletions
+52 -17
View File
@@ -24,12 +24,16 @@ const challengeTimeout = 45 * time.Second
var kaganeSeriesRe = regexp.MustCompile(`^/series/([0-9a-f-]{36})/?$`)
// comixSeriesPathRe matches the one path shape comixRead will open: a Series
// page, "/title/<id>-<slug>". Verified live 2026-08-12.
var comixSeriesPathRe = regexp.MustCompile(`^/title/[^/?#]+/?$`)
// BrowserFetcher retrieves pages through a remote headless Chrome over the
// DevTools Protocol.
//
// It exists for one reason: kagane.to and novelfull.com sit behind a
// Cloudflare JavaScript challenge. Verified 2026-08-03 (kagane) and 2026-08-05
// (novelfull) from the deployment host, plain HTTP and bogdanfinn/tls-client
// It exists for one reason: kagane.to, novelfull.com and comix.to sit behind a
// Cloudflare JavaScript challenge. Verified 2026-08-03 (kagane), 2026-08-05
// (novelfull) and 2026-08-12 (comix), plain HTTP and bogdanfinn/tls-client
// with a Chrome_133 profile both get 403 with cf-mitigated: challenge on every
// path, including the API, robots.txt and images. Clearing it requires
// executing the challenge script, which only a real browser does.
@@ -141,29 +145,47 @@ func novelfullRead(seriesURL string, out *string) (chromedp.Action, bool) {
return chromedp.OuterHTML("html", out, chromedp.ByQuery), true
}
// comixRead fetches the Series page from inside the cleared tab. comix is an
// SPA: rendering the page costs ~65 requests, while one same-origin fetch of
// the same address returns the server-rendered HTML — 24.5 KB, ~480 ms,
// carrying both parser anchors (measured 2026-08-12, issue #98). So this is
// kaganeRead's shape, not novelfullRead's, even though the payload is HTML.
// Refusing any other address is the per-Site half of the SSRF gate.
func comixRead(seriesURL string, out *string) (chromedp.Action, bool) {
pageURL, ok := comixSeriesPageURL(seriesURL)
if !ok {
return nil, false
}
return chromedp.Evaluate(
`fetch(`+jsString(pageURL)+`).then(r => r.ok ? r.text() : "")`,
out, awaitPromise), true
}
// Image retrieves one cover's bytes through the browser sidecar, and its
// content type.
//
// It exists because kagane serves covers behind the same challenge as its
// pages *and* with `cross-origin-resource-policy: same-origin`, so the bytes
// are only reachable from inside a browser that already holds the clearance
// cookie (verified 2026-08-08). Acquisition through the sidecar is the only
// route.
// It exists because kagane and comix serve covers behind the same challenge as
// their pages — kagane additionally with
// `cross-origin-resource-policy: same-origin` — so the bytes are only
// reachable from inside a browser that already holds the clearance cookie
// (verified 2026-08-08 for kagane, 2026-08-12 for comix). Acquisition through
// the sidecar is the only route.
//
// The image URL is navigated to rather than fetched from some other kagane
// page: the challenge only runs on a top-level navigation, and once it clears
// The image URL is navigated to rather than fetched from another page of the
// Site: the challenge only runs on a top-level navigation, and once it clears
// the document *is* the image, so a same-origin fetch of location.href reads
// it straight back out of the cache.
// it straight back out of the cache. For comix the navigation is also the only
// route that works at all — its Series page sets
// `cross-origin-embedder-policy: require-corp`, which fails a page-context
// fetch of the cover host.
//
// The challenge is not solved by the first read: WaitReady("body") is satisfied
// by the interstitial too. run holds the tab open until the in-page fetch
// succeeds, which is what gives the challenge script the seconds it needs.
func (f *BrowserFetcher) Image(ctx context.Context, imageURL string) ([]byte, string, error) {
m := kaganeImageURLRe.FindStringSubmatch(imageURL)
if m == nil {
if !browserOnlyCoverURL(imageURL) {
return nil, "", fmt.Errorf("not a browser-fetchable cover url: %q", imageURL)
}
imageID := m[1]
var dataURL string
err := f.run(ctx, imageURL,
chromedp.Evaluate(`fetch(location.href).then(r => r.ok
@@ -175,16 +197,16 @@ func (f *BrowserFetcher) Image(ctx context.Context, imageURL string) ([]byte, st
: "")`, &dataURL, awaitPromise),
func() bool { return dataURL != "" })
if err != nil {
return nil, "", fmt.Errorf("browser image %s: %w", imageID, err)
return nil, "", fmt.Errorf("browser image %s: %w", imageURL, err)
}
// "data:image/webp;base64,<payload>".
head, payload, ok := strings.Cut(dataURL, ";base64,")
if !ok {
return nil, "", fmt.Errorf("browser image %s: not a data url", imageID)
return nil, "", fmt.Errorf("browser image %s: not a data url", imageURL)
}
raw, err := base64.StdEncoding.DecodeString(payload)
if err != nil {
return nil, "", fmt.Errorf("browser image %s: %w", imageID, err)
return nil, "", fmt.Errorf("browser image %s: %w", imageURL, err)
}
return raw, strings.TrimPrefix(head, "data:"), nil
}
@@ -323,6 +345,19 @@ func novelfullSeriesURL(seriesURL string) bool {
strings.HasSuffix(u.Path, ".html")
}
// comixSeriesPageURL returns the address comixRead fetches inside the tab: the
// Series page itself, rebuilt from the pinned host and path so nothing else
// travels. Host-pinned here for the same reason kagane's is — series_url is
// client-supplied and a headless browser is a strong SSRF primitive.
func comixSeriesPageURL(seriesURL string) (string, bool) {
u, err := url.Parse(seriesURL)
if err != nil || u.Scheme != "https" || u.Hostname() != "comix.to" ||
!comixSeriesPathRe.MatchString(u.Path) {
return "", false
}
return "https://comix.to" + u.Path, true
}
// awaitPromise makes Evaluate resolve the promise rather than returning a
// serialised Promise object.
func awaitPromise(p *runtime.EvaluateParams) *runtime.EvaluateParams {
+56
View File
@@ -61,6 +61,62 @@ func TestNovelfullSeriesURL(t *testing.T) {
})
}
}
func TestComixSeriesPageURL(t *testing.T) {
const series = "https://comix.to/title/n8we-dungeons-and-crayons"
cases := []struct {
name string
url string
want string
}{
{"series page", series, series},
{"trailing slash kept", series + "/", series + "/"},
// Query and fragment are dropped: only the pinned path travels.
{"query dropped", series + "?tab=chapters", series},
{"foreign host", "https://evil.example/title/x", ""},
{"lookalike host", "https://comix.to.evil.example/title/x", ""},
{"not https", "http://comix.to/title/x", ""},
{"not a series path", "https://comix.to/search", ""},
{"chapter page", series + "/11139891-chapter-80", ""},
{"garbage", "://nope", ""},
}
for _, tc := range cases {
t.Run(tc.name, func(t *testing.T) {
got, ok := comixSeriesPageURL(tc.url)
if ok != (tc.want != "") || got != tc.want {
t.Fatalf("comixSeriesPageURL(%q) = %q, %v; want %q", tc.url, got, ok, tc.want)
}
})
}
}
// The browser is an SSRF primitive and a cover address can originate in a
// client-supplied PUT body, so this gate decides what it may navigate to.
func TestBrowserOnlyCoverURL(t *testing.T) {
cases := []struct {
url string
want bool
}{
{"https://static.comix.to/039d/i/1/34/6a6742bf15736@280.jpg", true},
{"https://kagane.to/api/v2/image/019fe11a-84c3-7fc3-a84b-88787374b617/compressed", true},
// Every other Site's CDN answers plain TLS.
{"https://gg.asuracomic.net/covers/x.webp", false},
{"http://static.comix.to/039d/x.jpg", false},
{"https://static.comix.to.evil.example/039d/x.jpg", false},
{"https://evil.example/static.comix.to/x.jpg", false},
{"https://static.comix.to/039d/x.jpg?next=http://169.254.169.254/", false},
{"https://static.comix.to/039d/x.svg", false},
{"https://static.comix.to/../etc/passwd.jpg", false},
{"https://static.comix.to/", false},
}
for _, tc := range cases {
t.Run(tc.url, func(t *testing.T) {
if got := browserOnlyCoverURL(tc.url); got != tc.want {
t.Fatalf("browserOnlyCoverURL(%q) = %v, want %v", tc.url, got, tc.want)
}
})
}
}
func TestClassifyBrowserInterruption(t *testing.T) {
if err := classifyBrowserError(context.Background(), true, context.Canceled); !errors.Is(err, errBrowserInterrupted) {
t.Fatalf("classifyBrowserError(context.Canceled) = %v, want browser interruption", err)
+2 -1
View File
@@ -25,7 +25,8 @@ type CoverBytesFetcher interface {
// fetchCoverBytes routes a cover's byte retrieval by URL shape, not by Site
// name: the browser fetcher's module claims the addresses only it can fetch
// (kagane's image route answers a plain fetch with a challenge and
// `cross-origin-resource-policy: same-origin`), and everything else goes over
// `cross-origin-resource-policy: same-origin`, static.comix.to answers one with
// the same challenge its pages serve), and everything else goes over
// plain TLS. Missing fetchers degrade to an error the caller logs, never a
// fallback onto a path that cannot succeed. One routing rule for the poll and
// the acquirer, so the two cannot drift apart.
+5 -5
View File
@@ -17,8 +17,8 @@ type Fetcher interface {
}
// BrowserCoverFetcher retrieves one cover's bytes through the browser-backed
// path — the only route that clears the challenge kagane's image URLs answer
// a plain fetch with. Satisfied by BrowserFetcher.
// path — the only route that clears the challenge kagane's and comix's image
// URLs answer a plain fetch with. Satisfied by BrowserFetcher.
type BrowserCoverFetcher interface {
Image(ctx context.Context, imageURL string) (body []byte, contentType string, err error)
}
@@ -119,9 +119,9 @@ func (p *Poller) storeCover(ctx context.Context, sr store.Series, sourceURL stri
// fetcherFor returns the fetcher a site's page needs, or nil when the site
// cannot be fetched at all right now. A Site whose registry entry carries a
// Browser read — kagane and novelfull, both behind a Cloudflare JavaScript
// challenge no TLS fingerprint clears — prefers the browser; when it is
// absent, the entry's Fallback decides whether plain TLS may take over. One
// Browser read — kagane, comix and novelfull, all behind a Cloudflare
// JavaScript challenge no TLS fingerprint clears — prefers the browser; when it
// is absent, the entry's Fallback decides whether plain TLS may take over. One
// routing rule for the poll and the acquirer, so the two cannot drift apart.
func fetcherFor(site string, browser, tls Fetcher) Fetcher {
s, known := sites[site]
+80
View File
@@ -762,6 +762,86 @@ func TestKaganeUsesBrowserFetcher(t *testing.T) {
}
}
// comix joined kagane behind the challenge on 2026-08-12 (#98): its page goes
// to the browser, its Cover bytes go through the browser's image route because
// static.comix.to is gated the same way, and the TLS fetcher is never asked
// for either.
func TestComixUsesBrowserFetcher(t *testing.T) {
s, dbURL := newTestStore(t)
const (
key = "comix:n8we-dungeons-and-crayons"
seriesID = "n8we-dungeons-and-crayons"
seriesURL = "https://comix.to/title/n8we-dungeons-and-crayons"
coverURL = "https://static.comix.to/039d/i/1/34/6a6742bf15736@280.jpg"
)
if _, err := s.Upsert(s.OwnerID(), store.Bookmark{
Key: key, Site: "comix", SeriesID: seriesID, SeriesURL: seriesURL,
UpdatedAt: 1000,
}); err != nil {
t.Fatalf("seed: %v", err)
}
seedCoverSource(t, dbURL, "comix", seriesID, coverURL)
tlsF := &fakeFetcher{body: "", status: 200}
browserF := &fakeFetcher{body: comixSeriesFixture, status: 200}
covers := &fakeCoverFetcher{body: []byte("cover-bytes"), contentType: "image/jpeg"}
tlsCovers := &fakeBytesCoverFetcher{body: []byte("tls-bytes"), contentType: "image/jpeg"}
p := &Poller{
Store: s, Fetch: tlsF, BrowserFetch: browserF,
CoverFetch: covers, CoverBytesFetch: tlsCovers,
Now: func() time.Time { return time.UnixMilli(5_000_000) },
Cooldown: time.Hour, BrowserCooldown: time.Hour,
Interval: time.Hour, Batch: 10,
}
p.runOnce(context.Background())
if len(tlsF.calls) != 0 {
t.Errorf("TLS fetcher was called for comix: %v", tlsF.calls)
}
if len(browserF.calls) != 1 {
t.Fatalf("browser fetcher calls = %v, want 1", browserF.calls)
}
if got := tlsCovers.callCount(); got != 0 {
t.Errorf("TLS cover fetches = %d, want 0: static.comix.to answers a challenge", got)
}
if got := covers.callCount(); got != 1 {
t.Fatalf("browser cover fetches = %d, want 1", got)
}
got, found, err := s.Get(s.OwnerID(), key)
if err != nil || !found {
t.Fatalf("Get: %v found=%v", err, found)
}
if got.LatestChapterNum == nil || *got.LatestChapterNum != 80 {
t.Errorf("LatestChapterNum = %v, want 80", got.LatestChapterNum)
}
}
// Without a browser, comix is skipped outright rather than handed to plain
// TLS: a plain fetch retrieves only a challenge page (measured 2026-08-12).
func TestComixSkippedWhenNoBrowserFetcher(t *testing.T) {
s, _ := newTestStore(t)
if _, err := s.Upsert(s.OwnerID(), store.Bookmark{
Key: "comix:n8we-dungeons-and-crayons", Site: "comix",
SeriesID: "n8we-dungeons-and-crayons",
SeriesURL: "https://comix.to/title/n8we-dungeons-and-crayons",
UpdatedAt: 1000,
}); err != nil {
t.Fatalf("seed: %v", err)
}
f := &fakeFetcher{body: comixSeriesFixture, status: 200}
p := &Poller{
Store: s, Fetch: f,
Now: func() time.Time { return time.UnixMilli(5_000_000) },
Cooldown: time.Hour, BrowserCooldown: time.Hour,
Interval: time.Hour, Batch: 10,
}
p.runOnce(context.Background())
if len(f.calls) != 0 {
t.Errorf("TLS fetcher was called for comix: %v", f.calls)
}
}
func TestRunOncePrefetchesKaganeCover(t *testing.T) {
s, dbURL := newTestStore(t)
const (
+25 -5
View File
@@ -248,14 +248,25 @@ var comixInitialDataRe = regexp.MustCompile(`(?is)<script\b[^>]*\bid\s*=\s*["']i
// supplied, and a headless browser is a strong SSRF primitive.
var kaganeImageURLRe = regexp.MustCompile(`^https://kagane\.to/api/v2/image/([0-9a-f-]{36})/compressed$`)
// comixImageURLRe matches comix's cover host and path shape. Pinned in full
// (scheme, host, path characters, image extension) for the same reason
// kaganeImageURLRe is: the address reaches a headless browser, and it can
// originate in a client-supplied PUT body. No dot is allowed inside the path,
// so no traversal or second extension can hide in it. Shape from a live page,
// 2026-08-10: /039d/i/1/34/6a6742bf15736@280.jpg.
var comixImageURLRe = regexp.MustCompile(`^https://static\.comix\.to/[A-Za-z0-9@/_-]+\.(?:jpg|jpeg|png|webp)$`)
// browserOnlyCoverURL reports whether the browser sidecar is the only fetcher
// for cover bytes at imageURL. kagane's image route answers a plain fetch with
// a challenge and `cross-origin-resource-policy: same-origin`, so a TLS fetch
// would only ever retrieve a challenge page and must not be attempted
// (ADR-0007). This is the byte-fetch router's per-Site knowledge; it lives in
// the extraction module, which owns kagane's URL shapes.
// a challenge and `cross-origin-resource-policy: same-origin`, and
// static.comix.to answers one with the same Cloudflare challenge its pages
// serve (measured 2026-08-12, issue #98), so a TLS fetch would only ever
// retrieve a challenge page and must not be attempted (ADR-0007). This is the
// byte-fetch router's per-Site knowledge; it lives in the extraction module,
// which owns those URL shapes.
func browserOnlyCoverURL(imageURL string) bool {
return kaganeImageURLRe.MatchString(imageURL)
return kaganeImageURLRe.MatchString(imageURL) ||
comixImageURLRe.MatchString(imageURL)
}
// kagane's browser-fetched series response publishes cover image IDs under
@@ -389,6 +400,15 @@ var sites = map[string]site{
Host: "comix.to",
LatestChapter: comixLatestChapter,
Cover: comixCoverEntry,
Browser: &browserRead{
Read: comixRead,
// The interstitial is served in place of the page, so "arrived"
// has to exclude it explicitly, as novelfull's does.
Done: func(body string) bool { return body != "" && !isInterstitial(body) },
// Never falls back: a plain fetch of a comix page or cover
// retrieves only a challenge page (measured 2026-08-12).
Fallback: false,
},
},
"kagane": {
Host: "kagane.to",
+4
View File
@@ -41,6 +41,10 @@ const challengeFixture = `<!DOCTYPE html><html><head><title>Just a moment...</ti
// https://comix.to/title/n8we-dungeons-and-crayons fetched 2026-08-03. comix is
// an SPA: the page ships a JSON state blob rather than a list of chapter
// anchors, and latestChapterUrl is where the newest chapter actually lives.
//
// Still the right fixture after comix moved behind the challenge (#98): the
// browser read is an in-tab fetch of the Series URL, so the body a poll parses
// is this same server-rendered HTML, not a rendered DOM.
const comixSeriesFixture = `
{"firstChapterUrl":"/title/n8we-dungeons-and-crayons/5038739-chapter-1","latestChapterUrl":"/title/n8we-dungeons-and-crayons/11139891-chapter-80"},
{""manga","recommended","n8we",1]":{"items":[{"latestChapterUrl":"/title/qqwrm-full-time-awakening/99999999-chapter-999"}]}
@@ -0,0 +1,72 @@
package latest
import (
"context"
"os"
"testing"
"time"
"bookmarkmanager/backend/internal/store"
)
// TestSmokeComix answers "is comix's challenge clearing from this browser right
// now" — a live, time-varying fact, so a red run is something to re-check
// before it is a defect. Needs the real browser unit with outbound network:
//
// cd chrome && BROWSER_BIND_ADDR=127.0.0.1 docker compose up -d --build
// SMOKE_BROWSER_WS_URL=ws://127.0.0.1:9222 go test -run TestSmokeComix ./internal/latest
//
// It walks the whole read: the in-tab page fetch, both parses, and the Cover
// bytes by direct navigation to static.comix.to. The Cover address comes out of
// the page rather than being pinned in the test, because a stored one rots.
func TestSmokeComix(t *testing.T) {
ws := os.Getenv("SMOKE_BROWSER_WS_URL")
if ws == "" {
t.Skip("SMOKE_BROWSER_WS_URL unset")
}
const seriesURL = "https://comix.to/title/m12d-classmate"
f, err := NewBrowserFetcher(ws)
if err != nil {
t.Fatalf("NewBrowserFetcher: %v", err)
}
defer f.Close()
ctx, cancel := context.WithTimeout(context.Background(), 120*time.Second)
defer cancel()
body, status, err := f.Get(ctx, seriesURL)
if err != nil {
t.Fatalf("Get: %v", err)
}
t.Logf("status=%d bytes=%d", status, len(body))
if status != 200 {
t.Fatalf("status = %d, want 200 — the sidecar is not clearing the challenge", status)
}
chapter, ok := latestChapterFrom("comix", seriesURL, body)
if !ok {
t.Fatalf("no latest chapter in %d bytes — page shape changed", len(body))
}
t.Logf("latest chapter: %v %q", chapter.Num, chapter.Label)
cover, ok := coverFrom("comix", seriesURL, body)
if !ok {
t.Fatalf("no cover address in %d bytes — page shape changed", len(body))
}
t.Logf("cover: %s", cover)
if !browserOnlyCoverURL(cover) {
t.Fatalf("cover %q is not claimed by the browser gate: the pin and the live URL shape disagree", cover)
}
bytes, contentType, err := f.Image(ctx, cover)
if err != nil {
t.Fatalf("Image: %v", err)
}
if len(bytes) < 1000 {
t.Fatalf("cover is %d bytes, want a real image", len(bytes))
}
t.Logf("fetched %d bytes of %s", len(bytes), contentType)
if _, ok := store.CoverContentType(contentType); !ok {
t.Fatalf("content type %q is not storable", contentType)
}
}