05e93a4869
BrowserFetcher.run navigated, waited for "body", read once, and closed the tab - about half a second end to end. The Cloudflare interstitial has a body too, so WaitReady was satisfied by the challenge page itself, and the read that followed was of the interstitial rather than the site. That made the challenge unclearable rather than merely slow. An interstitial needs several seconds of a live page to solve itself and write clearance into the browser's shared cookie jar; tearing the tab down first means the clearance that would have unblocked every later fetch is never obtained, so each call is challenged exactly like the one before it. run now holds one tab and re-reads until the caller's predicate reports an answer, bounded by challengeTimeout and by the caller's own deadline. Each caller supplies the predicate that fits its payload: kagane's in-page fetch simply returns nothing while challenged, whereas novelfull's payload is the DOM, and the interstitial has a DOM as well, so that one excludes the challenge markup explicitly. Exhausting the budget is now reported as errChallengeHeld and mapped back to the 403 the poller already expects, keeping a challenged site distinct from a broken transport. Image loses its own retry loop, which run now subsumes. Measured against a real kagane cover from a cold browser profile: no image at all before, 4.9s to a 56710-byte image/webp after. The live proof is TestSmokeKagane* in smoke_image_test.go, which skips unless SMOKE_BROWSER_WS_URL names a sidecar, so `go test ./...` stays hermetic.
286 lines
12 KiB
Go
286 lines
12 KiB
Go
package latest
|
|
|
|
import (
|
|
"context"
|
|
"encoding/base64"
|
|
"encoding/json"
|
|
"errors"
|
|
"fmt"
|
|
"net/url"
|
|
"regexp"
|
|
"strings"
|
|
"sync"
|
|
"time"
|
|
|
|
"github.com/chromedp/cdproto/runtime"
|
|
"github.com/chromedp/chromedp"
|
|
)
|
|
|
|
// challengeTimeout bounds one navigate-and-solve. A Cloudflare managed
|
|
// challenge clears in a few seconds when it clears at all; anything longer is a
|
|
// challenge that is not going to pass, and the caller's cooldown was already
|
|
// stamped before this ran.
|
|
const challengeTimeout = 45 * time.Second
|
|
|
|
var kaganeSeriesRe = regexp.MustCompile(`^/series/([0-9a-f-]{36})/?$`)
|
|
|
|
// kaganeImageIDRe pins the only path segment Image interpolates into an
|
|
// outbound URL. The id arrives from a stored cover URL, which a client
|
|
// supplied, so it is matched rather than trusted: a headless browser is a
|
|
// strong SSRF primitive.
|
|
var kaganeImageIDRe = regexp.MustCompile(`^[0-9a-f-]{36}$`)
|
|
|
|
// BrowserFetcher retrieves pages through a remote headless Chrome over the
|
|
// DevTools Protocol.
|
|
//
|
|
// It exists for one reason: kagane.to and novelfull.com sit behind a
|
|
// Cloudflare JavaScript challenge. Verified 2026-08-03 (kagane) and 2026-08-05
|
|
// (novelfull) from the deployment host, plain HTTP and bogdanfinn/tls-client
|
|
// with a Chrome_133 profile both get 403 with cf-mitigated: challenge on every
|
|
// path, including the API, robots.txt and images. Clearing it requires
|
|
// executing the challenge script, which only a real browser does.
|
|
//
|
|
// The request is made *inside* the page rather than by extracting cf_clearance
|
|
// and replaying it through TLSFetcher. That cookie is bound to IP, User-Agent
|
|
// and often the TLS fingerprint, so replaying it means keeping three things in
|
|
// sync that break silently and separately. The browser's own cookie jar
|
|
// persists across polls, so the challenge is solved once every few hours.
|
|
//
|
|
// The two sites differ in how the chapter list is read: kagane serves it from
|
|
// a JSON API that must be called from inside the page (so the request carries
|
|
// the clearance cookie), while novelfull renders it into the HTML so the
|
|
// cleared DOM is the payload.
|
|
type BrowserFetcher struct {
|
|
allocCtx context.Context
|
|
cancel context.CancelFunc
|
|
// One page at a time: caps the sidecar's memory and keeps series from
|
|
// sharing page state.
|
|
mu sync.Mutex
|
|
}
|
|
|
|
var _ Fetcher = (*BrowserFetcher)(nil)
|
|
|
|
// NewBrowserFetcher connects to a headless-shell over CDP. wsURL must name the
|
|
// sidecar by IP, e.g. ws://172.28.0.10:9222 — not by Docker DNS name. Chrome's
|
|
// DevTools HTTP handler 500s any /json/version request whose Host header
|
|
// isn't an IP or "localhost" (confirmed 2026-08-03 against
|
|
// chromedp/headless-shell:stable), so the compose network pins the sidecar's
|
|
// address for this to resolve at all.
|
|
//
|
|
// Do not add chromedp.NoModifyURL here: that option skips the /json/version
|
|
// discovery request entirely and dials wsURL as if it were already the full
|
|
// debugger endpoint, but Chrome only accepts connections at
|
|
// /devtools/browser/<uuid>, a path chosen fresh at every Chrome start — dialing
|
|
// the bare host:port 404s. The default (discovery) path works precisely
|
|
// because Chrome's /json/version response echoes back the Host header of the
|
|
// discovery request in webSocketDebuggerUrl, so as long as wsURL is a
|
|
// container-reachable IP, the URL chromedp gets back already points at it.
|
|
func NewBrowserFetcher(wsURL string) (*BrowserFetcher, error) {
|
|
if wsURL == "" {
|
|
return nil, fmt.Errorf("empty browser websocket url")
|
|
}
|
|
ctx, cancel := chromedp.NewRemoteAllocator(context.Background(), wsURL)
|
|
return &BrowserFetcher{allocCtx: ctx, cancel: cancel}, nil
|
|
}
|
|
|
|
func (f *BrowserFetcher) Close() {
|
|
f.cancel()
|
|
}
|
|
|
|
// Get navigates to seriesURL, lets any challenge resolve, then reads either the
|
|
// site's JSON API (kagane) from inside the page so the request carries the
|
|
// clearance cookie, or the served HTML itself (novelfull) — see
|
|
// novelfullSeriesURL for the latter case. The returned body is whatever the
|
|
// site's chapter list lives in, which is what latestChapterFrom's per-site
|
|
// switch expects.
|
|
func (f *BrowserFetcher) Get(ctx context.Context, seriesURL string) (string, int, error) {
|
|
apiURL, isKagane := kaganeAPIURL(seriesURL)
|
|
if !isKagane && !novelfullSeriesURL(seriesURL) {
|
|
return "", 0, fmt.Errorf("not a fetchable browser series url: %q", seriesURL)
|
|
}
|
|
|
|
var body string
|
|
// kagane's chapter list is only in its JSON API, which must be called from
|
|
// inside the page so the request carries the clearance cookie. novelfull
|
|
// renders its chapters into the HTML, so the cleared DOM is the answer.
|
|
// chromedp.OuterHTML returns a QueryAction and chromedp.Evaluate an
|
|
// EvaluateAction, so the variable has to be the interface both implement.
|
|
var read chromedp.Action = chromedp.OuterHTML("html", &body, chromedp.ByQuery)
|
|
if isKagane {
|
|
read = chromedp.Evaluate(
|
|
`fetch(`+jsString(apiURL)+`).then(r => r.ok ? r.text() : "")`,
|
|
&body,
|
|
awaitPromise,
|
|
)
|
|
}
|
|
|
|
// novelfull's payload is the DOM itself, and the interstitial has a DOM
|
|
// too, so "we have an answer" has to exclude it explicitly. kagane's
|
|
// in-page fetch just fails while challenged, which is already the signal.
|
|
done := func() bool { return body != "" && (isKagane || !isInterstitial(body)) }
|
|
if err := f.run(ctx, seriesURL, read, done); err != nil {
|
|
// Challenge never cleared, or the API refused. Indistinguishable from
|
|
// here and handled identically by the caller.
|
|
if errors.Is(err, errChallengeHeld) {
|
|
return "", 403, nil
|
|
}
|
|
return "", 0, fmt.Errorf("browser fetch %q: %w", seriesURL, err)
|
|
}
|
|
return body, 200, nil
|
|
}
|
|
|
|
// Image retrieves one kagane cover as raw bytes and its content type.
|
|
//
|
|
// It exists because kagane serves covers behind the same challenge as its
|
|
// pages *and* with `cross-origin-resource-policy: same-origin`, so an <img> on
|
|
// the web UI's origin cannot load one even from a browser that already holds
|
|
// the clearance cookie (verified 2026-08-08). Proxying is the only route.
|
|
//
|
|
// The image URL is navigated to rather than fetched from some other kagane
|
|
// page: the challenge only runs on a top-level navigation, and once it clears
|
|
// the document *is* the image, so a same-origin fetch of location.href reads
|
|
// it straight back out of the cache.
|
|
//
|
|
// The challenge is not solved by the first read: WaitReady("body") is satisfied
|
|
// by the interstitial too. run holds the tab open until the in-page fetch
|
|
// succeeds, which is what gives the challenge script the seconds it needs.
|
|
func (f *BrowserFetcher) Image(ctx context.Context, imageID string) ([]byte, string, error) {
|
|
if !kaganeImageIDRe.MatchString(imageID) {
|
|
return nil, "", fmt.Errorf("not a kagane image id: %q", imageID)
|
|
}
|
|
var dataURL string
|
|
err := f.run(ctx, "https://kagane.to/api/v2/image/"+imageID+"/compressed",
|
|
chromedp.Evaluate(`fetch(location.href).then(r => r.ok
|
|
? r.blob().then(b => new Promise(res => {
|
|
const fr = new FileReader();
|
|
fr.onload = () => res(fr.result);
|
|
fr.readAsDataURL(b);
|
|
}))
|
|
: "")`, &dataURL, awaitPromise),
|
|
func() bool { return dataURL != "" })
|
|
if err != nil {
|
|
return nil, "", fmt.Errorf("browser image %s: %w", imageID, err)
|
|
}
|
|
// "data:image/webp;base64,<payload>".
|
|
head, payload, ok := strings.Cut(dataURL, ";base64,")
|
|
if !ok {
|
|
return nil, "", fmt.Errorf("browser image %s: not a data url", imageID)
|
|
}
|
|
raw, err := base64.StdEncoding.DecodeString(payload)
|
|
if err != nil {
|
|
return nil, "", fmt.Errorf("browser image %s: %w", imageID, err)
|
|
}
|
|
return raw, strings.TrimPrefix(head, "data:"), nil
|
|
}
|
|
|
|
// errChallengeHeld reports that the budget ran out with the interstitial still
|
|
// up. Distinct from a transport failure: it means "this site said no", which
|
|
// the poller answers with a 403 and its ordinary cooldown.
|
|
var errChallengeHeld = errors.New("challenge held")
|
|
|
|
// challengePollInterval paces re-reads while a challenge solves itself.
|
|
const challengePollInterval = 2 * time.Second
|
|
|
|
// isInterstitial reports whether html is Cloudflare's challenge page rather
|
|
// than the site's own. Matched on the challenge runtime's script path, which is
|
|
// stable across the interstitial's wording and locale — the visible "Just a
|
|
// moment..." title is neither.
|
|
func isInterstitial(html string) bool {
|
|
return strings.Contains(html, "/cdn-cgi/challenge-platform/")
|
|
}
|
|
|
|
// run navigates to target and re-reads until done reports an answer, bounded by
|
|
// challengeTimeout and by the caller's own deadline, in a tab that is closed on
|
|
// return so one wedged page cannot poison later calls.
|
|
//
|
|
// Holding the tab open across re-reads is the whole point. A Cloudflare
|
|
// interstitial needs several seconds of a live page to solve itself and write
|
|
// clearance into the browser's shared cookie jar; reading once and closing the
|
|
// tab — which is what this did before 2026-08-08 — never gives it that window,
|
|
// so every fetch lands on the interstitial and the clearance that would have
|
|
// unblocked all the later ones is never obtained.
|
|
func (f *BrowserFetcher) run(ctx context.Context, target string, read chromedp.Action, done func() bool) error {
|
|
f.mu.Lock()
|
|
defer f.mu.Unlock()
|
|
|
|
ctx, cancel := context.WithTimeout(ctx, challengeTimeout)
|
|
defer cancel()
|
|
tabCtx, cancelTab := chromedp.NewContext(f.allocCtx)
|
|
defer cancelTab()
|
|
// Bind the caller's deadline to the tab.
|
|
tabCtx, cancelDeadline := context.WithCancel(tabCtx)
|
|
defer cancelDeadline()
|
|
go func() {
|
|
<-ctx.Done()
|
|
cancelDeadline()
|
|
}()
|
|
|
|
if err := chromedp.Run(tabCtx,
|
|
chromedp.Navigate(target),
|
|
chromedp.WaitReady("body", chromedp.ByQuery),
|
|
); err != nil {
|
|
return err
|
|
}
|
|
|
|
var lastErr error
|
|
for {
|
|
// The challenge reloads the page when it passes, which tears down the
|
|
// execution context mid-read. That is a retry, not a failure.
|
|
if err := chromedp.Run(tabCtx, read); err != nil {
|
|
lastErr = err
|
|
} else if done() {
|
|
return nil
|
|
}
|
|
select {
|
|
case <-ctx.Done():
|
|
if lastErr != nil {
|
|
return fmt.Errorf("%w (last read: %v)", errChallengeHeld, lastErr)
|
|
}
|
|
return errChallengeHeld
|
|
case <-time.After(challengePollInterval):
|
|
}
|
|
}
|
|
}
|
|
|
|
// kaganeAPIURL maps a stored series_url to the JSON endpoint carrying its
|
|
// chapter list. Returning false for anything else is a second line of defence
|
|
// behind fetchableSeriesURL: a headless browser is a strong SSRF primitive and
|
|
// series_url is client-supplied, so the host is pinned here too.
|
|
func kaganeAPIURL(seriesURL string) (string, bool) {
|
|
u, err := url.Parse(seriesURL)
|
|
if err != nil || u.Scheme != "https" || u.Hostname() != "kagane.to" {
|
|
return "", false
|
|
}
|
|
m := kaganeSeriesRe.FindStringSubmatch(u.Path)
|
|
if m == nil {
|
|
return "", false
|
|
}
|
|
return "https://kagane.to/api/v2/series/" + m[1], true
|
|
}
|
|
|
|
// novelfullSeriesURL reports whether seriesURL is a novelfull series page this
|
|
// fetcher will open. novelfull's chapter list is in the served HTML, so unlike
|
|
// kagane there is no API to call from inside the page — the challenge-cleared
|
|
// DOM is the payload. The host is pinned here for the same reason kagane's is:
|
|
// series_url is client-supplied and a headless browser is a strong SSRF
|
|
// primitive.
|
|
func novelfullSeriesURL(seriesURL string) bool {
|
|
u, err := url.Parse(seriesURL)
|
|
return err == nil && u.Scheme == "https" && u.Hostname() == "novelfull.com" &&
|
|
strings.HasSuffix(u.Path, ".html")
|
|
}
|
|
|
|
// awaitPromise makes Evaluate resolve the promise rather than returning a
|
|
// serialised Promise object.
|
|
func awaitPromise(p *runtime.EvaluateParams) *runtime.EvaluateParams {
|
|
return p.WithAwaitPromise(true)
|
|
}
|
|
|
|
// jsString renders s as a JavaScript string literal for embedding in an
|
|
// Evaluate expression. The URL is host-pinned by kaganeAPIURL before it gets
|
|
// here, but quoting it properly is what keeps that guarantee intact.
|
|
func jsString(s string) string {
|
|
b, _ := json.Marshal(s)
|
|
return string(b)
|
|
}
|