package latest import ( "context" "encoding/base64" "encoding/json" "errors" "fmt" "net/url" "regexp" "strings" "sync" "time" "github.com/chromedp/cdproto/runtime" "github.com/chromedp/chromedp" ) // challengeTimeout bounds one navigate-and-solve. A Cloudflare managed // challenge clears in a few seconds when it clears at all; anything longer is a // challenge that is not going to pass, and the caller's cooldown was already // stamped before this ran. const challengeTimeout = 45 * time.Second var kaganeSeriesRe = regexp.MustCompile(`^/series/([0-9a-f-]{36})/?$`) // kaganeImageIDRe pins the only path segment Image interpolates into an // outbound URL. The id arrives from a stored cover URL, which a client // supplied, so it is matched rather than trusted: a headless browser is a // strong SSRF primitive. var kaganeImageIDRe = regexp.MustCompile(`^[0-9a-f-]{36}$`) // BrowserFetcher retrieves pages through a remote headless Chrome over the // DevTools Protocol. // // It exists for one reason: kagane.to and novelfull.com sit behind a // Cloudflare JavaScript challenge. Verified 2026-08-03 (kagane) and 2026-08-05 // (novelfull) from the deployment host, plain HTTP and bogdanfinn/tls-client // with a Chrome_133 profile both get 403 with cf-mitigated: challenge on every // path, including the API, robots.txt and images. Clearing it requires // executing the challenge script, which only a real browser does. // // The request is made *inside* the page rather than by extracting cf_clearance // and replaying it through TLSFetcher. That cookie is bound to IP, User-Agent // and often the TLS fingerprint, so replaying it means keeping three things in // sync that break silently and separately. The browser's own cookie jar // persists across polls, so the challenge is solved once every few hours. // // The two sites differ in how the chapter list is read: kagane serves it from // a JSON API that must be called from inside the page (so the request carries // the clearance cookie), while novelfull renders it into the HTML so the // cleared DOM is the payload. type BrowserFetcher struct { allocCtx context.Context cancel context.CancelFunc // One page at a time: caps the sidecar's memory and keeps series from // sharing page state. mu sync.Mutex } var _ Fetcher = (*BrowserFetcher)(nil) // NewBrowserFetcher connects to a headless-shell over CDP. wsURL must name the // sidecar by IP, e.g. ws://172.28.0.10:9222 — not by Docker DNS name. Chrome's // DevTools HTTP handler 500s any /json/version request whose Host header // isn't an IP or "localhost" (confirmed 2026-08-03 against // chromedp/headless-shell:stable), so the compose network pins the sidecar's // address for this to resolve at all. // // Do not add chromedp.NoModifyURL here: that option skips the /json/version // discovery request entirely and dials wsURL as if it were already the full // debugger endpoint, but Chrome only accepts connections at // /devtools/browser/, a path chosen fresh at every Chrome start — dialing // the bare host:port 404s. The default (discovery) path works precisely // because Chrome's /json/version response echoes back the Host header of the // discovery request in webSocketDebuggerUrl, so as long as wsURL is a // container-reachable IP, the URL chromedp gets back already points at it. func NewBrowserFetcher(wsURL string) (*BrowserFetcher, error) { if wsURL == "" { return nil, fmt.Errorf("empty browser websocket url") } ctx, cancel := chromedp.NewRemoteAllocator(context.Background(), wsURL) return &BrowserFetcher{allocCtx: ctx, cancel: cancel}, nil } func (f *BrowserFetcher) Close() { f.cancel() } // Get navigates to seriesURL, lets any challenge resolve, then reads either the // site's JSON API (kagane) from inside the page so the request carries the // clearance cookie, or the served HTML itself (novelfull) — see // novelfullSeriesURL for the latter case. The returned body is whatever the // site's chapter list lives in, which is what latestChapterFrom's per-site // switch expects. func (f *BrowserFetcher) Get(ctx context.Context, seriesURL string) (string, int, error) { apiURL, isKagane := kaganeAPIURL(seriesURL) if !isKagane && !novelfullSeriesURL(seriesURL) { return "", 0, fmt.Errorf("not a fetchable browser series url: %q", seriesURL) } var body string // kagane's chapter list is only in its JSON API, which must be called from // inside the page so the request carries the clearance cookie. novelfull // renders its chapters into the HTML, so the cleared DOM is the answer. // chromedp.OuterHTML returns a QueryAction and chromedp.Evaluate an // EvaluateAction, so the variable has to be the interface both implement. var read chromedp.Action = chromedp.OuterHTML("html", &body, chromedp.ByQuery) if isKagane { read = chromedp.Evaluate( `fetch(`+jsString(apiURL)+`).then(r => r.ok ? r.text() : "")`, &body, awaitPromise, ) } // novelfull's payload is the DOM itself, and the interstitial has a DOM // too, so "we have an answer" has to exclude it explicitly. kagane's // in-page fetch just fails while challenged, which is already the signal. done := func() bool { return body != "" && (isKagane || !isInterstitial(body)) } if err := f.run(ctx, seriesURL, read, done); err != nil { // Challenge never cleared, or the API refused. Indistinguishable from // here and handled identically by the caller. if errors.Is(err, errChallengeHeld) { return "", 403, nil } return "", 0, fmt.Errorf("browser fetch %q: %w", seriesURL, err) } return body, 200, nil } // Image retrieves one kagane cover as raw bytes and its content type. // // It exists because kagane serves covers behind the same challenge as its // pages *and* with `cross-origin-resource-policy: same-origin`, so an on // the web UI's origin cannot load one even from a browser that already holds // the clearance cookie (verified 2026-08-08). Proxying is the only route. // // The image URL is navigated to rather than fetched from some other kagane // page: the challenge only runs on a top-level navigation, and once it clears // the document *is* the image, so a same-origin fetch of location.href reads // it straight back out of the cache. // // The challenge is not solved by the first read: WaitReady("body") is satisfied // by the interstitial too. run holds the tab open until the in-page fetch // succeeds, which is what gives the challenge script the seconds it needs. func (f *BrowserFetcher) Image(ctx context.Context, imageID string) ([]byte, string, error) { if !kaganeImageIDRe.MatchString(imageID) { return nil, "", fmt.Errorf("not a kagane image id: %q", imageID) } var dataURL string err := f.run(ctx, "https://kagane.to/api/v2/image/"+imageID+"/compressed", chromedp.Evaluate(`fetch(location.href).then(r => r.ok ? r.blob().then(b => new Promise(res => { const fr = new FileReader(); fr.onload = () => res(fr.result); fr.readAsDataURL(b); })) : "")`, &dataURL, awaitPromise), func() bool { return dataURL != "" }) if err != nil { return nil, "", fmt.Errorf("browser image %s: %w", imageID, err) } // "data:image/webp;base64,". head, payload, ok := strings.Cut(dataURL, ";base64,") if !ok { return nil, "", fmt.Errorf("browser image %s: not a data url", imageID) } raw, err := base64.StdEncoding.DecodeString(payload) if err != nil { return nil, "", fmt.Errorf("browser image %s: %w", imageID, err) } return raw, strings.TrimPrefix(head, "data:"), nil } // errChallengeHeld reports that the budget ran out with the interstitial still // up. Distinct from a transport failure: it means "this site said no", which // the poller answers with a 403 and its ordinary cooldown. var errChallengeHeld = errors.New("challenge held") // errBrowserInterrupted distinguishes a remote Chrome restart from the // caller's own deadline. chromedp reports both as context.Canceled. var errBrowserInterrupted = errors.New("browser interrupted") func classifyBrowserError(ctx context.Context, browserLost bool, err error) error { if err == nil || ctx.Err() != nil { return err } if !browserLost { return err } if !errors.Is(err, context.Canceled) { return err } return fmt.Errorf("%w: %w", errBrowserInterrupted, err) } func browserConnectionLost(ctx context.Context) bool { c := chromedp.FromContext(ctx) if c == nil || c.Browser == nil { return true } select { case <-c.Browser.LostConnection: return true default: return false } } // challengePollInterval paces re-reads while a challenge solves itself. const challengePollInterval = 2 * time.Second // isInterstitial reports whether html is Cloudflare's challenge page rather // than the site's own. Matched on the challenge runtime's script path, which is // stable across the interstitial's wording and locale — the visible "Just a // moment..." title is neither. func isInterstitial(html string) bool { return strings.Contains(html, "/cdn-cgi/challenge-platform/") } // run navigates to target and re-reads until done reports an answer, bounded by // challengeTimeout and by the caller's own deadline, in a tab that is closed on // return so one wedged page cannot poison later calls. // // Holding the tab open across re-reads is the whole point. A Cloudflare // interstitial needs several seconds of a live page to solve itself and write // clearance into the browser's shared cookie jar; reading once and closing the // tab — which is what this did before 2026-08-08 — never gives it that window, // so every fetch lands on the interstitial and the clearance that would have // unblocked all the later ones is never obtained. func (f *BrowserFetcher) run(ctx context.Context, target string, read chromedp.Action, done func() bool) error { f.mu.Lock() defer f.mu.Unlock() callerCtx := ctx ctx, cancel := context.WithTimeout(ctx, challengeTimeout) defer cancel() tabCtx, cancelTab := chromedp.NewContext(f.allocCtx) defer cancelTab() // Bind the caller's deadline to the tab. tabCtx, cancelDeadline := context.WithCancel(tabCtx) defer cancelDeadline() go func() { <-ctx.Done() cancelDeadline() }() if err := chromedp.Run(tabCtx, chromedp.Navigate(target), chromedp.WaitReady("body", chromedp.ByQuery), ); err != nil { return classifyBrowserError(callerCtx, browserConnectionLost(tabCtx), err) } var lastErr error for { // The challenge reloads the page when it passes, which tears down the // execution context mid-read. That is a retry, not a failure. if err := chromedp.Run(tabCtx, read); err != nil { err = classifyBrowserError(callerCtx, browserConnectionLost(tabCtx), err) if errors.Is(err, errBrowserInterrupted) { return err } lastErr = err } else if done() { return nil } select { case <-ctx.Done(): if err := callerCtx.Err(); err != nil { return err } if lastErr != nil { return fmt.Errorf("%w (last read: %v)", errChallengeHeld, lastErr) } return errChallengeHeld case <-time.After(challengePollInterval): } } } // kaganeAPIURL maps a stored series_url to the JSON endpoint carrying its // chapter list. Returning false for anything else is a second line of defence // behind fetchableSeriesURL: a headless browser is a strong SSRF primitive and // series_url is client-supplied, so the host is pinned here too. func kaganeAPIURL(seriesURL string) (string, bool) { u, err := url.Parse(seriesURL) if err != nil || u.Scheme != "https" || u.Hostname() != "kagane.to" { return "", false } m := kaganeSeriesRe.FindStringSubmatch(u.Path) if m == nil { return "", false } return "https://kagane.to/api/v2/series/" + m[1], true } // novelfullSeriesURL reports whether seriesURL is a novelfull series page this // fetcher will open. novelfull's chapter list is in the served HTML, so unlike // kagane there is no API to call from inside the page — the challenge-cleared // DOM is the payload. The host is pinned here for the same reason kagane's is: // series_url is client-supplied and a headless browser is a strong SSRF // primitive. func novelfullSeriesURL(seriesURL string) bool { u, err := url.Parse(seriesURL) return err == nil && u.Scheme == "https" && u.Hostname() == "novelfull.com" && strings.HasSuffix(u.Path, ".html") } // awaitPromise makes Evaluate resolve the promise rather than returning a // serialised Promise object. func awaitPromise(p *runtime.EvaluateParams) *runtime.EvaluateParams { return p.WithAwaitPromise(true) } // jsString renders s as a JavaScript string literal for embedding in an // Evaluate expression. The URL is host-pinned by kaganeAPIURL before it gets // here, but quoting it properly is what keeps that guarantee intact. func jsString(s string) string { b, _ := json.Marshal(s) return string(b) }