package latest import ( "context" "encoding/json" "fmt" "net/url" "regexp" "strings" "sync" "time" "github.com/chromedp/cdproto/runtime" "github.com/chromedp/chromedp" ) // challengeTimeout bounds one navigate-and-solve. A Cloudflare managed // challenge clears in a few seconds when it clears at all; anything longer is a // challenge that is not going to pass, and the caller's cooldown was already // stamped before this ran. const challengeTimeout = 45 * time.Second var kaganeSeriesRe = regexp.MustCompile(`^/series/([0-9a-f-]{36})/?$`) // BrowserFetcher retrieves pages through a remote headless Chrome over the // DevTools Protocol. // // It exists for one reason: kagane.to and novelfull.com sit behind a // Cloudflare JavaScript challenge. Verified 2026-08-03 (kagane) and 2026-08-05 // (novelfull) from the deployment host, plain HTTP and bogdanfinn/tls-client // with a Chrome_133 profile both get 403 with cf-mitigated: challenge on every // path, including the API, robots.txt and images. Clearing it requires // executing the challenge script, which only a real browser does. // // The request is made *inside* the page rather than by extracting cf_clearance // and replaying it through TLSFetcher. That cookie is bound to IP, User-Agent // and often the TLS fingerprint, so replaying it means keeping three things in // sync that break silently and separately. The browser's own cookie jar // persists across polls, so the challenge is solved once every few hours. // // The two sites differ in how the chapter list is read: kagane serves it from // a JSON API that must be called from inside the page (so the request carries // the clearance cookie), while novelfull renders it into the HTML so the // cleared DOM is the payload. type BrowserFetcher struct { allocCtx context.Context cancel context.CancelFunc // One page at a time: caps the sidecar's memory and keeps series from // sharing page state. mu sync.Mutex } var _ Fetcher = (*BrowserFetcher)(nil) // NewBrowserFetcher connects to a headless-shell over CDP. wsURL must name the // sidecar by IP, e.g. ws://172.28.0.10:9222 — not by Docker DNS name. Chrome's // DevTools HTTP handler 500s any /json/version request whose Host header // isn't an IP or "localhost" (confirmed 2026-08-03 against // chromedp/headless-shell:stable), so the compose network pins the sidecar's // address for this to resolve at all. // // Do not add chromedp.NoModifyURL here: that option skips the /json/version // discovery request entirely and dials wsURL as if it were already the full // debugger endpoint, but Chrome only accepts connections at // /devtools/browser/, a path chosen fresh at every Chrome start — dialing // the bare host:port 404s. The default (discovery) path works precisely // because Chrome's /json/version response echoes back the Host header of the // discovery request in webSocketDebuggerUrl, so as long as wsURL is a // container-reachable IP, the URL chromedp gets back already points at it. func NewBrowserFetcher(wsURL string) (*BrowserFetcher, error) { if wsURL == "" { return nil, fmt.Errorf("empty browser websocket url") } ctx, cancel := chromedp.NewRemoteAllocator(context.Background(), wsURL) return &BrowserFetcher{allocCtx: ctx, cancel: cancel}, nil } func (f *BrowserFetcher) Close() { f.cancel() } // Get navigates to seriesURL, lets any challenge resolve, then reads either the // site's JSON API (kagane) from inside the page so the request carries the // clearance cookie, or the served HTML itself (novelfull) — see // novelfullSeriesURL for the latter case. The returned body is whatever the // site's chapter list lives in, which is what latestChapterFrom's per-site // switch expects. func (f *BrowserFetcher) Get(ctx context.Context, seriesURL string) (string, int, error) { apiURL, isKagane := kaganeAPIURL(seriesURL) if !isKagane && !novelfullSeriesURL(seriesURL) { return "", 0, fmt.Errorf("not a fetchable browser series url: %q", seriesURL) } f.mu.Lock() defer f.mu.Unlock() ctx, cancel := context.WithTimeout(ctx, challengeTimeout) defer cancel() // A fresh tab per fetch, closed on return, so one wedged page cannot // poison later polls. tabCtx, cancelTab := chromedp.NewContext(f.allocCtx) defer cancelTab() // Bind the caller's deadline to the tab. tabCtx, cancelDeadline := context.WithCancel(tabCtx) defer cancelDeadline() go func() { <-ctx.Done() cancelDeadline() }() var body string // kagane's chapter list is only in its JSON API, which must be called from // inside the page so the request carries the clearance cookie. novelfull // renders its chapters into the HTML, so the cleared DOM is the answer. // chromedp.OuterHTML returns a QueryAction and chromedp.Evaluate an // EvaluateAction, so the variable has to be the interface both implement. var read chromedp.Action = chromedp.OuterHTML("html", &body, chromedp.ByQuery) if isKagane { read = chromedp.Evaluate( `fetch(`+jsString(apiURL)+`).then(r => r.ok ? r.text() : "")`, &body, awaitPromise, ) } err := chromedp.Run(tabCtx, chromedp.Navigate(seriesURL), // The challenge reloads the page itself when it passes; waiting for the // site's own root element is what tells us we are through it. chromedp.WaitReady("body", chromedp.ByQuery), read, ) if err != nil { return "", 0, fmt.Errorf("browser fetch %q: %w", seriesURL, err) } if body == "" { // Challenge still up, or the API refused. Indistinguishable from here // and handled identically by the caller. return "", 403, nil } return body, 200, nil } // kaganeAPIURL maps a stored series_url to the JSON endpoint carrying its // chapter list. Returning false for anything else is a second line of defence // behind fetchableSeriesURL: a headless browser is a strong SSRF primitive and // series_url is client-supplied, so the host is pinned here too. func kaganeAPIURL(seriesURL string) (string, bool) { u, err := url.Parse(seriesURL) if err != nil || u.Scheme != "https" || u.Hostname() != "kagane.to" { return "", false } m := kaganeSeriesRe.FindStringSubmatch(u.Path) if m == nil { return "", false } return "https://kagane.to/api/v2/series/" + m[1], true } // novelfullSeriesURL reports whether seriesURL is a novelfull series page this // fetcher will open. novelfull's chapter list is in the served HTML, so unlike // kagane there is no API to call from inside the page — the challenge-cleared // DOM is the payload. The host is pinned here for the same reason kagane's is: // series_url is client-supplied and a headless browser is a strong SSRF // primitive. func novelfullSeriesURL(seriesURL string) bool { u, err := url.Parse(seriesURL) return err == nil && u.Scheme == "https" && u.Hostname() == "novelfull.com" && strings.HasSuffix(u.Path, ".html") } // awaitPromise makes Evaluate resolve the promise rather than returning a // serialised Promise object. func awaitPromise(p *runtime.EvaluateParams) *runtime.EvaluateParams { return p.WithAwaitPromise(true) } // jsString renders s as a JavaScript string literal for embedding in an // Evaluate expression. The URL is host-pinned by kaganeAPIURL before it gets // here, but quoting it properly is what keeps that guarantee intact. func jsString(s string) string { b, _ := json.Marshal(s) return string(b) }