package latest import ( "context" "encoding/base64" "encoding/json" "errors" "fmt" "net/url" "regexp" "strings" "sync" "time" "github.com/chromedp/cdproto/runtime" "github.com/chromedp/chromedp" ) // challengeTimeout bounds one navigate-and-solve. A Cloudflare managed // challenge clears in a few seconds when it clears at all; anything longer is a // challenge that is not going to pass, and the caller's rest was already // stamped before this ran. const challengeTimeout = 45 * time.Second var kaganeSeriesRe = regexp.MustCompile(`^/series/([0-9a-f-]{36})/?$`) // comixSeriesPathRe matches the one path shape comixRead will open: a Series // page, "/title/-". Verified live 2026-08-12. var comixSeriesPathRe = regexp.MustCompile(`^/title/[^/?#]+/?$`) // BrowserFetcher retrieves pages through a remote headless Chrome over the // DevTools Protocol. // // It exists for one reason: kagane.to, novelfull.com and comix.to sit behind a // Cloudflare JavaScript challenge. Verified 2026-08-03 (kagane), 2026-08-05 // (novelfull) and 2026-08-12 (comix), plain HTTP and bogdanfinn/tls-client // with a Chrome_133 profile both get 403 with cf-mitigated: challenge on every // path, including the API, robots.txt and images. Clearing it requires // executing the challenge script, which only a real browser does. // // The request is made *inside* the page rather than by extracting cf_clearance // and replaying it through TLSFetcher. That cookie is bound to IP, User-Agent // and often the TLS fingerprint, so replaying it means keeping three things in // sync that break silently and separately. The browser's own cookie jar // persists across polls, so the challenge is solved once every few hours. // // The three sites differ in what a cleared tab is asked for: kagane fetches a // JSON API from inside the page (the list exists nowhere else), comix fetches // its own Series URL from inside the page (the served HTML carries the facts, // and rendering the SPA costs ~65 requests instead of one), and novelfull // renders its list into the HTML so the cleared DOM is the payload. type BrowserFetcher struct { allocCtx context.Context cancel context.CancelFunc // One page at a time: caps the browser's memory — it runs under a hard // cgroup cap on a shared machine — and keeps series from sharing page state. mu sync.Mutex } var _ Fetcher = (*BrowserFetcher)(nil) // NewBrowserFetcher connects to a Chrome over CDP. The browser is not a // sidecar: it runs on a separate machine and is reached over the tailnet // (ADR-0006), so wsURL is that machine's tailnet address, e.g. // ws://100.64.0.5:9222. // // It must be an IP, never a hostname — not MagicDNS, not a Docker service // name. Chrome's DevTools HTTP handler 500s any /json/version request whose // Host header isn't an IP or "localhost" (confirmed 2026-08-03), so a name // fails at discovery and surfaces as a dead site rather than a bad URL. // // Do not add chromedp.NoModifyURL here: that option skips the /json/version // discovery request entirely and dials wsURL as if it were already the full // debugger endpoint, but Chrome only accepts connections at // /devtools/browser/, a path chosen fresh at every Chrome start — dialing // the bare host:port 404s. The default (discovery) path works precisely // because Chrome's /json/version response echoes back the Host header of the // discovery request in webSocketDebuggerUrl, so as long as wsURL is an IP this // process can reach, the URL chromedp gets back already points at it. That is // also why a Chrome restarted behind a stable endpoint needs no reconnect // here: the fresh UUID arrives with the next discovery. func NewBrowserFetcher(wsURL string) (*BrowserFetcher, error) { if wsURL == "" { return nil, fmt.Errorf("empty browser websocket url") } ctx, cancel := chromedp.NewRemoteAllocator(context.Background(), wsURL) return &BrowserFetcher{allocCtx: ctx, cancel: cancel}, nil } func (f *BrowserFetcher) Close() { f.cancel() } // Get navigates to seriesURL, lets any challenge resolve, then reads the // payload the Site's registry entry describes (the shapes are listed on // BrowserFetcher). The returned body is whatever the Site's chapter list lives // in, which is what the entry's LatestChapter parse expects. func (f *BrowserFetcher) Get(ctx context.Context, seriesURL string) (string, int, error) { var body string // Sorted order (browserBackedSites sorts) makes dispatch deterministic: // entries' Read funcs are expected to refuse any address owned by another // Site, and the loop must not depend on that staying true. for _, name := range browserBackedSites() { s := sites[name] read, ok := s.Browser.Read(seriesURL, &body) if !ok { continue } if err := f.run(ctx, seriesURL, read, func() bool { return s.Browser.Done(body) }); err != nil { // Challenge never cleared, or the payload was refused. // Indistinguishable from here and handled identically by the caller. if errors.Is(err, errChallengeHeld) { return "", 403, nil } return "", 0, fmt.Errorf("browser fetch %q: %w", seriesURL, err) } return body, 200, nil } return "", 0, fmt.Errorf("not a fetchable browser series url: %q", seriesURL) } // kaganeRead builds the in-tab fetch of kagane's chapter-list API: the // request must be made from inside the page so it carries the clearance // cookie, and the API is the only place the list exists. Refusing any other // address is the per-Site half of the SSRF gate, kept deliberately behind // FetchableSeriesURL (see browserRead.Read). func kaganeRead(seriesURL string, out *string) (chromedp.Action, bool) { apiURL, ok := kaganeAPIURL(seriesURL) if !ok { return nil, false } return chromedp.Evaluate( `fetch(`+jsString(apiURL)+`).then(r => r.ok ? r.text() : "")`, out, awaitPromise), true } // novelfullRead reads the cleared DOM. novelfull renders its chapter list // into the served HTML, so there is no API to call from inside the page — the // challenge-cleared DOM is the payload. func novelfullRead(seriesURL string, out *string) (chromedp.Action, bool) { if !novelfullSeriesURL(seriesURL) { return nil, false } return chromedp.OuterHTML("html", out, chromedp.ByQuery), true } // comixRead fetches the Series page from inside the cleared tab. comix is an // SPA: rendering the page costs ~65 requests, while one same-origin fetch of // the same address returns the server-rendered HTML — 24.5 KB, ~480 ms, // carrying both parser anchors (measured 2026-08-12, issue #98). So this is // kaganeRead's shape, not novelfullRead's, even though the payload is HTML. // Refusing any other address is the per-Site half of the SSRF gate. func comixRead(seriesURL string, out *string) (chromedp.Action, bool) { pageURL, ok := comixSeriesPageURL(seriesURL) if !ok { return nil, false } return chromedp.Evaluate( `fetch(`+jsString(pageURL)+`).then(r => r.ok ? r.text() : "")`, out, awaitPromise), true } // Image retrieves one cover's bytes through the browser sidecar, and its // content type. // // It exists because kagane and comix serve covers behind the same challenge as // their pages — kagane additionally with // `cross-origin-resource-policy: same-origin` — so the bytes are only // reachable from inside a browser that already holds the clearance cookie // (verified 2026-08-08 for kagane, 2026-08-12 for comix). Acquisition through // the sidecar is the only route. // // The image URL is navigated to rather than fetched from another page of the // Site: the challenge only runs on a top-level navigation, and once it clears // the document *is* the image, so a same-origin fetch of location.href reads // it straight back out of the cache. For comix the navigation is also the only // route that works at all — its Series page sets // `cross-origin-embedder-policy: require-corp`, which fails a page-context // fetch of the cover host. // // The challenge is not solved by the first read: WaitReady("body") is satisfied // by the interstitial too. run holds the tab open until the in-page fetch // succeeds, which is what gives the challenge script the seconds it needs. func (f *BrowserFetcher) Image(ctx context.Context, imageURL string) ([]byte, string, error) { if !browserOnlyCoverURL(imageURL) { return nil, "", fmt.Errorf("not a browser-fetchable cover url: %q", imageURL) } var dataURL string err := f.run(ctx, imageURL, chromedp.Evaluate(`fetch(location.href).then(r => r.ok ? r.blob().then(b => new Promise(res => { const fr = new FileReader(); fr.onload = () => res(fr.result); fr.readAsDataURL(b); })) : "")`, &dataURL, awaitPromise), func() bool { return dataURL != "" }) if err != nil { return nil, "", fmt.Errorf("browser image %s: %w", imageURL, err) } // "data:image/webp;base64,". head, payload, ok := strings.Cut(dataURL, ";base64,") if !ok { return nil, "", fmt.Errorf("browser image %s: not a data url", imageURL) } raw, err := base64.StdEncoding.DecodeString(payload) if err != nil { return nil, "", fmt.Errorf("browser image %s: %w", imageURL, err) } return raw, strings.TrimPrefix(head, "data:"), nil } // errChallengeHeld reports that the budget ran out with the interstitial still // up. Distinct from a transport failure: it means "this site said no", which // the poller answers with a refusal backoff for that Site's Lane (issue #100). var errChallengeHeld = errors.New("challenge held") // errBrowserInterrupted distinguishes a remote Chrome restart from the // caller's own deadline. chromedp reports both as context.Canceled. var errBrowserInterrupted = errors.New("browser interrupted") func classifyBrowserError(ctx context.Context, browserLost bool, err error) error { if err == nil || ctx.Err() != nil { return err } if !browserLost { return err } if !errors.Is(err, context.Canceled) { return err } return fmt.Errorf("%w: %w", errBrowserInterrupted, err) } func browserConnectionLost(ctx context.Context) bool { c := chromedp.FromContext(ctx) if c == nil || c.Browser == nil { return true } select { case <-c.Browser.LostConnection: return true default: return false } } // challengePollInterval paces re-reads while a challenge solves itself. const challengePollInterval = 2 * time.Second // isInterstitial reports whether html is Cloudflare's challenge page rather // than the site's own. Matched on the challenge orchestration path // (/cdn-cgi/challenge-platform/h//orchestrate/...), which is stable // across the interstitial's wording and locale — the visible "Just a // moment..." title is neither. // // The bare "/cdn-cgi/challenge-platform/" prefix is NOT enough: Cloudflare // injects /cdn-cgi/challenge-platform/scripts/jsd/main.js into ordinary 200 // pages when JS detections are on, so matching the prefix declared every real // demonic page a refusal and parked that Lane in 15m backoff (observed // 2026-08-16, demonic turned detections on). func isInterstitial(html string) bool { return strings.Contains(html, "/cdn-cgi/challenge-platform/h/") } // run navigates to target and re-reads until done reports an answer, bounded by // challengeTimeout and by the caller's own deadline, in a tab that is closed on // return so one wedged page cannot poison later calls. // // Holding the tab open across re-reads is the whole point. A Cloudflare // interstitial needs several seconds of a live page to solve itself and write // clearance into the browser's shared cookie jar; reading once and closing the // tab — which is what this did before 2026-08-08 — never gives it that window, // so every fetch lands on the interstitial and the clearance that would have // unblocked all the later ones is never obtained. func (f *BrowserFetcher) run(ctx context.Context, target string, read chromedp.Action, done func() bool) error { f.mu.Lock() defer f.mu.Unlock() callerCtx := ctx ctx, cancel := context.WithTimeout(ctx, challengeTimeout) defer cancel() tabCtx, cancelTab := chromedp.NewContext(f.allocCtx) defer cancelTab() // Bind the caller's deadline to the tab. tabCtx, cancelDeadline := context.WithCancel(tabCtx) defer cancelDeadline() go func() { <-ctx.Done() cancelDeadline() }() if err := chromedp.Run(tabCtx, chromedp.Navigate(target), chromedp.WaitReady("body", chromedp.ByQuery), ); err != nil { return classifyBrowserError(callerCtx, browserConnectionLost(tabCtx), err) } var lastErr error for { // The challenge reloads the page when it passes, which tears down the // execution context mid-read. That is a retry, not a failure. if err := chromedp.Run(tabCtx, read); err != nil { err = classifyBrowserError(callerCtx, browserConnectionLost(tabCtx), err) if errors.Is(err, errBrowserInterrupted) { return err } lastErr = err } else if done() { return nil } select { case <-ctx.Done(): if err := callerCtx.Err(); err != nil { return err } if lastErr != nil { return fmt.Errorf("%w (last read: %v)", errChallengeHeld, lastErr) } return errChallengeHeld case <-time.After(challengePollInterval): } } } // kaganeAPIURL maps a stored series_url to the JSON endpoint carrying its // chapter list. Returning false for anything else is a second line of defence // behind FetchableSeriesURL: a headless browser is a strong SSRF primitive and // series_url is client-supplied, so the host is pinned here too. func kaganeAPIURL(seriesURL string) (string, bool) { u, err := url.Parse(seriesURL) if err != nil || u.Scheme != "https" || u.Hostname() != "kagane.to" { return "", false } m := kaganeSeriesRe.FindStringSubmatch(u.Path) if m == nil { return "", false } return "https://kagane.to/api/v2/series/" + m[1], true } // novelfullSeriesURL reports whether seriesURL is a novelfull series page this // fetcher will open. novelfull's chapter list is in the served HTML, so unlike // kagane there is no API to call from inside the page — the challenge-cleared // DOM is the payload. The host is pinned here for the same reason kagane's is: // series_url is client-supplied and a headless browser is a strong SSRF // primitive. func novelfullSeriesURL(seriesURL string) bool { u, err := url.Parse(seriesURL) return err == nil && u.Scheme == "https" && u.Hostname() == "novelfull.com" && strings.HasSuffix(u.Path, ".html") } // comixSeriesPageURL returns the address comixRead fetches inside the tab: the // Series page itself, rebuilt from the pinned host and path so nothing else // travels. Host-pinned here for the same reason kagane's is — series_url is // client-supplied and a headless browser is a strong SSRF primitive. func comixSeriesPageURL(seriesURL string) (string, bool) { u, err := url.Parse(seriesURL) if err != nil || u.Scheme != "https" || u.Hostname() != "comix.to" || !comixSeriesPathRe.MatchString(u.Path) { return "", false } return "https://comix.to" + u.Path, true } // awaitPromise makes Evaluate resolve the promise rather than returning a // serialised Promise object. func awaitPromise(p *runtime.EvaluateParams) *runtime.EvaluateParams { return p.WithAwaitPromise(true) } // jsString renders s as a JavaScript string literal for embedding in an // Evaluate expression. The URL is host-pinned by kaganeAPIURL before it gets // here, but quoting it properly is what keeps that guarantee intact. func jsString(s string) string { b, _ := json.Marshal(s) return string(b) }