feat(latest): poll comix through the browser sidecar (#98)
comix.to began answering plain-TLS fetches with a Cloudflare JavaScript challenge on 2026-08-12, so every poll got a 403 interstitial and its cover host static.comix.to is gated the same way. comix joins kagane and novelfull as a browser Site: one registry entry, no plain-TLS fallback, and cover bytes routed through the browser's image path behind a fully pinned URL pattern. The read is an in-tab fetch of the Series URL, not a DOM render: comix is an SPA, so rendering costs ~65 requests for the same server-rendered HTML one fetch returns (24.5 KB, ~480 ms). Parsers and stored Series identity are untouched. Verified live against the real browser unit: page 24793 bytes in one fetch, chapter 53, cover accepted by the pin and 26862 image bytes retrieved by direct navigation (comix's Series page sets cross-origin-embedder-policy: require-corp, so an in-page fetch of the cover host cannot work).
This commit is contained in:
@@ -24,12 +24,16 @@ const challengeTimeout = 45 * time.Second
|
||||
|
||||
var kaganeSeriesRe = regexp.MustCompile(`^/series/([0-9a-f-]{36})/?$`)
|
||||
|
||||
// comixSeriesPathRe matches the one path shape comixRead will open: a Series
|
||||
// page, "/title/<id>-<slug>". Verified live 2026-08-12.
|
||||
var comixSeriesPathRe = regexp.MustCompile(`^/title/[^/?#]+/?$`)
|
||||
|
||||
// BrowserFetcher retrieves pages through a remote headless Chrome over the
|
||||
// DevTools Protocol.
|
||||
//
|
||||
// It exists for one reason: kagane.to and novelfull.com sit behind a
|
||||
// Cloudflare JavaScript challenge. Verified 2026-08-03 (kagane) and 2026-08-05
|
||||
// (novelfull) from the deployment host, plain HTTP and bogdanfinn/tls-client
|
||||
// It exists for one reason: kagane.to, novelfull.com and comix.to sit behind a
|
||||
// Cloudflare JavaScript challenge. Verified 2026-08-03 (kagane), 2026-08-05
|
||||
// (novelfull) and 2026-08-12 (comix), plain HTTP and bogdanfinn/tls-client
|
||||
// with a Chrome_133 profile both get 403 with cf-mitigated: challenge on every
|
||||
// path, including the API, robots.txt and images. Clearing it requires
|
||||
// executing the challenge script, which only a real browser does.
|
||||
@@ -141,29 +145,47 @@ func novelfullRead(seriesURL string, out *string) (chromedp.Action, bool) {
|
||||
return chromedp.OuterHTML("html", out, chromedp.ByQuery), true
|
||||
}
|
||||
|
||||
// comixRead fetches the Series page from inside the cleared tab. comix is an
|
||||
// SPA: rendering the page costs ~65 requests, while one same-origin fetch of
|
||||
// the same address returns the server-rendered HTML — 24.5 KB, ~480 ms,
|
||||
// carrying both parser anchors (measured 2026-08-12, issue #98). So this is
|
||||
// kaganeRead's shape, not novelfullRead's, even though the payload is HTML.
|
||||
// Refusing any other address is the per-Site half of the SSRF gate.
|
||||
func comixRead(seriesURL string, out *string) (chromedp.Action, bool) {
|
||||
pageURL, ok := comixSeriesPageURL(seriesURL)
|
||||
if !ok {
|
||||
return nil, false
|
||||
}
|
||||
return chromedp.Evaluate(
|
||||
`fetch(`+jsString(pageURL)+`).then(r => r.ok ? r.text() : "")`,
|
||||
out, awaitPromise), true
|
||||
}
|
||||
|
||||
// Image retrieves one cover's bytes through the browser sidecar, and its
|
||||
// content type.
|
||||
//
|
||||
// It exists because kagane serves covers behind the same challenge as its
|
||||
// pages *and* with `cross-origin-resource-policy: same-origin`, so the bytes
|
||||
// are only reachable from inside a browser that already holds the clearance
|
||||
// cookie (verified 2026-08-08). Acquisition through the sidecar is the only
|
||||
// route.
|
||||
// It exists because kagane and comix serve covers behind the same challenge as
|
||||
// their pages — kagane additionally with
|
||||
// `cross-origin-resource-policy: same-origin` — so the bytes are only
|
||||
// reachable from inside a browser that already holds the clearance cookie
|
||||
// (verified 2026-08-08 for kagane, 2026-08-12 for comix). Acquisition through
|
||||
// the sidecar is the only route.
|
||||
//
|
||||
// The image URL is navigated to rather than fetched from some other kagane
|
||||
// page: the challenge only runs on a top-level navigation, and once it clears
|
||||
// The image URL is navigated to rather than fetched from another page of the
|
||||
// Site: the challenge only runs on a top-level navigation, and once it clears
|
||||
// the document *is* the image, so a same-origin fetch of location.href reads
|
||||
// it straight back out of the cache.
|
||||
// it straight back out of the cache. For comix the navigation is also the only
|
||||
// route that works at all — its Series page sets
|
||||
// `cross-origin-embedder-policy: require-corp`, which fails a page-context
|
||||
// fetch of the cover host.
|
||||
//
|
||||
// The challenge is not solved by the first read: WaitReady("body") is satisfied
|
||||
// by the interstitial too. run holds the tab open until the in-page fetch
|
||||
// succeeds, which is what gives the challenge script the seconds it needs.
|
||||
func (f *BrowserFetcher) Image(ctx context.Context, imageURL string) ([]byte, string, error) {
|
||||
m := kaganeImageURLRe.FindStringSubmatch(imageURL)
|
||||
if m == nil {
|
||||
if !browserOnlyCoverURL(imageURL) {
|
||||
return nil, "", fmt.Errorf("not a browser-fetchable cover url: %q", imageURL)
|
||||
}
|
||||
imageID := m[1]
|
||||
var dataURL string
|
||||
err := f.run(ctx, imageURL,
|
||||
chromedp.Evaluate(`fetch(location.href).then(r => r.ok
|
||||
@@ -175,16 +197,16 @@ func (f *BrowserFetcher) Image(ctx context.Context, imageURL string) ([]byte, st
|
||||
: "")`, &dataURL, awaitPromise),
|
||||
func() bool { return dataURL != "" })
|
||||
if err != nil {
|
||||
return nil, "", fmt.Errorf("browser image %s: %w", imageID, err)
|
||||
return nil, "", fmt.Errorf("browser image %s: %w", imageURL, err)
|
||||
}
|
||||
// "data:image/webp;base64,<payload>".
|
||||
head, payload, ok := strings.Cut(dataURL, ";base64,")
|
||||
if !ok {
|
||||
return nil, "", fmt.Errorf("browser image %s: not a data url", imageID)
|
||||
return nil, "", fmt.Errorf("browser image %s: not a data url", imageURL)
|
||||
}
|
||||
raw, err := base64.StdEncoding.DecodeString(payload)
|
||||
if err != nil {
|
||||
return nil, "", fmt.Errorf("browser image %s: %w", imageID, err)
|
||||
return nil, "", fmt.Errorf("browser image %s: %w", imageURL, err)
|
||||
}
|
||||
return raw, strings.TrimPrefix(head, "data:"), nil
|
||||
}
|
||||
@@ -323,6 +345,19 @@ func novelfullSeriesURL(seriesURL string) bool {
|
||||
strings.HasSuffix(u.Path, ".html")
|
||||
}
|
||||
|
||||
// comixSeriesPageURL returns the address comixRead fetches inside the tab: the
|
||||
// Series page itself, rebuilt from the pinned host and path so nothing else
|
||||
// travels. Host-pinned here for the same reason kagane's is — series_url is
|
||||
// client-supplied and a headless browser is a strong SSRF primitive.
|
||||
func comixSeriesPageURL(seriesURL string) (string, bool) {
|
||||
u, err := url.Parse(seriesURL)
|
||||
if err != nil || u.Scheme != "https" || u.Hostname() != "comix.to" ||
|
||||
!comixSeriesPathRe.MatchString(u.Path) {
|
||||
return "", false
|
||||
}
|
||||
return "https://comix.to" + u.Path, true
|
||||
}
|
||||
|
||||
// awaitPromise makes Evaluate resolve the promise rather than returning a
|
||||
// serialised Promise object.
|
||||
func awaitPromise(p *runtime.EvaluateParams) *runtime.EvaluateParams {
|
||||
|
||||
Reference in New Issue
Block a user