84cfd1b2c1
Closes #44. Chrome now starts on first CDP connection, tracks concurrent helpers, reaps after 300 seconds idle, preserves the named profile, and classifies reap interruptions. Shutdown stops Chrome's process group so cookie batches flush. ADR-0005 records the measured constraints and decisions. Verification: docker build, live CDP wake, graceful stop cleanup, sh -n, and go test ./... (7 packages, 3 no tests). Reviewed-on: #50 Co-authored-by: Sulthan Zaki <sultankiki05@gmail.com> Co-committed-by: Sulthan Zaki <sultankiki05@gmail.com>
323 lines
12 KiB
Go
323 lines
12 KiB
Go
package latest
|
|
|
|
import (
|
|
"context"
|
|
"encoding/base64"
|
|
"encoding/json"
|
|
"errors"
|
|
"fmt"
|
|
"net/url"
|
|
"regexp"
|
|
"strings"
|
|
"sync"
|
|
"time"
|
|
|
|
"github.com/chromedp/cdproto/runtime"
|
|
"github.com/chromedp/chromedp"
|
|
)
|
|
|
|
// challengeTimeout bounds one navigate-and-solve. A Cloudflare managed
|
|
// challenge clears in a few seconds when it clears at all; anything longer is a
|
|
// challenge that is not going to pass, and the caller's cooldown was already
|
|
// stamped before this ran.
|
|
const challengeTimeout = 45 * time.Second
|
|
|
|
var kaganeSeriesRe = regexp.MustCompile(`^/series/([0-9a-f-]{36})/?$`)
|
|
|
|
// kaganeImageIDRe pins the only path segment Image interpolates into an
|
|
// outbound URL. The id arrives from a stored cover URL, which a client
|
|
// supplied, so it is matched rather than trusted: a headless browser is a
|
|
// strong SSRF primitive.
|
|
var kaganeImageIDRe = regexp.MustCompile(`^[0-9a-f-]{36}$`)
|
|
|
|
// BrowserFetcher retrieves pages through a remote headless Chrome over the
|
|
// DevTools Protocol.
|
|
//
|
|
// It exists for one reason: kagane.to and novelfull.com sit behind a
|
|
// Cloudflare JavaScript challenge. Verified 2026-08-03 (kagane) and 2026-08-05
|
|
// (novelfull) from the deployment host, plain HTTP and bogdanfinn/tls-client
|
|
// with a Chrome_133 profile both get 403 with cf-mitigated: challenge on every
|
|
// path, including the API, robots.txt and images. Clearing it requires
|
|
// executing the challenge script, which only a real browser does.
|
|
//
|
|
// The request is made *inside* the page rather than by extracting cf_clearance
|
|
// and replaying it through TLSFetcher. That cookie is bound to IP, User-Agent
|
|
// and often the TLS fingerprint, so replaying it means keeping three things in
|
|
// sync that break silently and separately. The browser's own cookie jar
|
|
// persists across polls, so the challenge is solved once every few hours.
|
|
//
|
|
// The two sites differ in how the chapter list is read: kagane serves it from
|
|
// a JSON API that must be called from inside the page (so the request carries
|
|
// the clearance cookie), while novelfull renders it into the HTML so the
|
|
// cleared DOM is the payload.
|
|
type BrowserFetcher struct {
|
|
allocCtx context.Context
|
|
cancel context.CancelFunc
|
|
// One page at a time: caps the sidecar's memory and keeps series from
|
|
// sharing page state.
|
|
mu sync.Mutex
|
|
}
|
|
|
|
var _ Fetcher = (*BrowserFetcher)(nil)
|
|
|
|
// NewBrowserFetcher connects to a headless-shell over CDP. wsURL must name the
|
|
// sidecar by IP, e.g. ws://172.28.0.10:9222 — not by Docker DNS name. Chrome's
|
|
// DevTools HTTP handler 500s any /json/version request whose Host header
|
|
// isn't an IP or "localhost" (confirmed 2026-08-03 against
|
|
// chromedp/headless-shell:stable), so the compose network pins the sidecar's
|
|
// address for this to resolve at all.
|
|
//
|
|
// Do not add chromedp.NoModifyURL here: that option skips the /json/version
|
|
// discovery request entirely and dials wsURL as if it were already the full
|
|
// debugger endpoint, but Chrome only accepts connections at
|
|
// /devtools/browser/<uuid>, a path chosen fresh at every Chrome start — dialing
|
|
// the bare host:port 404s. The default (discovery) path works precisely
|
|
// because Chrome's /json/version response echoes back the Host header of the
|
|
// discovery request in webSocketDebuggerUrl, so as long as wsURL is a
|
|
// container-reachable IP, the URL chromedp gets back already points at it.
|
|
func NewBrowserFetcher(wsURL string) (*BrowserFetcher, error) {
|
|
if wsURL == "" {
|
|
return nil, fmt.Errorf("empty browser websocket url")
|
|
}
|
|
ctx, cancel := chromedp.NewRemoteAllocator(context.Background(), wsURL)
|
|
return &BrowserFetcher{allocCtx: ctx, cancel: cancel}, nil
|
|
}
|
|
|
|
func (f *BrowserFetcher) Close() {
|
|
f.cancel()
|
|
}
|
|
|
|
// Get navigates to seriesURL, lets any challenge resolve, then reads either the
|
|
// site's JSON API (kagane) from inside the page so the request carries the
|
|
// clearance cookie, or the served HTML itself (novelfull) — see
|
|
// novelfullSeriesURL for the latter case. The returned body is whatever the
|
|
// site's chapter list lives in, which is what latestChapterFrom's per-site
|
|
// switch expects.
|
|
func (f *BrowserFetcher) Get(ctx context.Context, seriesURL string) (string, int, error) {
|
|
apiURL, isKagane := kaganeAPIURL(seriesURL)
|
|
if !isKagane && !novelfullSeriesURL(seriesURL) {
|
|
return "", 0, fmt.Errorf("not a fetchable browser series url: %q", seriesURL)
|
|
}
|
|
|
|
var body string
|
|
// kagane's chapter list is only in its JSON API, which must be called from
|
|
// inside the page so the request carries the clearance cookie. novelfull
|
|
// renders its chapters into the HTML, so the cleared DOM is the answer.
|
|
// chromedp.OuterHTML returns a QueryAction and chromedp.Evaluate an
|
|
// EvaluateAction, so the variable has to be the interface both implement.
|
|
var read chromedp.Action = chromedp.OuterHTML("html", &body, chromedp.ByQuery)
|
|
if isKagane {
|
|
read = chromedp.Evaluate(
|
|
`fetch(`+jsString(apiURL)+`).then(r => r.ok ? r.text() : "")`,
|
|
&body,
|
|
awaitPromise,
|
|
)
|
|
}
|
|
|
|
// novelfull's payload is the DOM itself, and the interstitial has a DOM
|
|
// too, so "we have an answer" has to exclude it explicitly. kagane's
|
|
// in-page fetch just fails while challenged, which is already the signal.
|
|
done := func() bool { return body != "" && (isKagane || !isInterstitial(body)) }
|
|
if err := f.run(ctx, seriesURL, read, done); err != nil {
|
|
// Challenge never cleared, or the API refused. Indistinguishable from
|
|
// here and handled identically by the caller.
|
|
if errors.Is(err, errChallengeHeld) {
|
|
return "", 403, nil
|
|
}
|
|
return "", 0, fmt.Errorf("browser fetch %q: %w", seriesURL, err)
|
|
}
|
|
return body, 200, nil
|
|
}
|
|
|
|
// Image retrieves one kagane cover as raw bytes and its content type.
|
|
//
|
|
// It exists because kagane serves covers behind the same challenge as its
|
|
// pages *and* with `cross-origin-resource-policy: same-origin`, so an <img> on
|
|
// the web UI's origin cannot load one even from a browser that already holds
|
|
// the clearance cookie (verified 2026-08-08). Proxying is the only route.
|
|
//
|
|
// The image URL is navigated to rather than fetched from some other kagane
|
|
// page: the challenge only runs on a top-level navigation, and once it clears
|
|
// the document *is* the image, so a same-origin fetch of location.href reads
|
|
// it straight back out of the cache.
|
|
//
|
|
// The challenge is not solved by the first read: WaitReady("body") is satisfied
|
|
// by the interstitial too. run holds the tab open until the in-page fetch
|
|
// succeeds, which is what gives the challenge script the seconds it needs.
|
|
func (f *BrowserFetcher) Image(ctx context.Context, imageID string) ([]byte, string, error) {
|
|
if !kaganeImageIDRe.MatchString(imageID) {
|
|
return nil, "", fmt.Errorf("not a kagane image id: %q", imageID)
|
|
}
|
|
var dataURL string
|
|
err := f.run(ctx, "https://kagane.to/api/v2/image/"+imageID+"/compressed",
|
|
chromedp.Evaluate(`fetch(location.href).then(r => r.ok
|
|
? r.blob().then(b => new Promise(res => {
|
|
const fr = new FileReader();
|
|
fr.onload = () => res(fr.result);
|
|
fr.readAsDataURL(b);
|
|
}))
|
|
: "")`, &dataURL, awaitPromise),
|
|
func() bool { return dataURL != "" })
|
|
if err != nil {
|
|
return nil, "", fmt.Errorf("browser image %s: %w", imageID, err)
|
|
}
|
|
// "data:image/webp;base64,<payload>".
|
|
head, payload, ok := strings.Cut(dataURL, ";base64,")
|
|
if !ok {
|
|
return nil, "", fmt.Errorf("browser image %s: not a data url", imageID)
|
|
}
|
|
raw, err := base64.StdEncoding.DecodeString(payload)
|
|
if err != nil {
|
|
return nil, "", fmt.Errorf("browser image %s: %w", imageID, err)
|
|
}
|
|
return raw, strings.TrimPrefix(head, "data:"), nil
|
|
}
|
|
|
|
// errChallengeHeld reports that the budget ran out with the interstitial still
|
|
// up. Distinct from a transport failure: it means "this site said no", which
|
|
// the poller answers with a 403 and its ordinary cooldown.
|
|
var errChallengeHeld = errors.New("challenge held")
|
|
|
|
// errBrowserInterrupted distinguishes a remote Chrome restart from the
|
|
// caller's own deadline. chromedp reports both as context.Canceled.
|
|
var errBrowserInterrupted = errors.New("browser interrupted")
|
|
|
|
func classifyBrowserError(ctx context.Context, browserLost bool, err error) error {
|
|
if err == nil || ctx.Err() != nil {
|
|
return err
|
|
}
|
|
if !browserLost {
|
|
return err
|
|
}
|
|
if !errors.Is(err, context.Canceled) {
|
|
return err
|
|
}
|
|
return fmt.Errorf("%w: %w", errBrowserInterrupted, err)
|
|
}
|
|
|
|
func browserConnectionLost(ctx context.Context) bool {
|
|
c := chromedp.FromContext(ctx)
|
|
if c == nil || c.Browser == nil {
|
|
return true
|
|
}
|
|
select {
|
|
case <-c.Browser.LostConnection:
|
|
return true
|
|
default:
|
|
return false
|
|
}
|
|
}
|
|
|
|
// challengePollInterval paces re-reads while a challenge solves itself.
|
|
const challengePollInterval = 2 * time.Second
|
|
|
|
// isInterstitial reports whether html is Cloudflare's challenge page rather
|
|
// than the site's own. Matched on the challenge runtime's script path, which is
|
|
// stable across the interstitial's wording and locale — the visible "Just a
|
|
// moment..." title is neither.
|
|
func isInterstitial(html string) bool {
|
|
return strings.Contains(html, "/cdn-cgi/challenge-platform/")
|
|
}
|
|
|
|
// run navigates to target and re-reads until done reports an answer, bounded by
|
|
// challengeTimeout and by the caller's own deadline, in a tab that is closed on
|
|
// return so one wedged page cannot poison later calls.
|
|
//
|
|
// Holding the tab open across re-reads is the whole point. A Cloudflare
|
|
// interstitial needs several seconds of a live page to solve itself and write
|
|
// clearance into the browser's shared cookie jar; reading once and closing the
|
|
// tab — which is what this did before 2026-08-08 — never gives it that window,
|
|
// so every fetch lands on the interstitial and the clearance that would have
|
|
// unblocked all the later ones is never obtained.
|
|
func (f *BrowserFetcher) run(ctx context.Context, target string, read chromedp.Action, done func() bool) error {
|
|
f.mu.Lock()
|
|
defer f.mu.Unlock()
|
|
|
|
callerCtx := ctx
|
|
ctx, cancel := context.WithTimeout(ctx, challengeTimeout)
|
|
defer cancel()
|
|
tabCtx, cancelTab := chromedp.NewContext(f.allocCtx)
|
|
defer cancelTab()
|
|
// Bind the caller's deadline to the tab.
|
|
tabCtx, cancelDeadline := context.WithCancel(tabCtx)
|
|
defer cancelDeadline()
|
|
go func() {
|
|
<-ctx.Done()
|
|
cancelDeadline()
|
|
}()
|
|
|
|
if err := chromedp.Run(tabCtx,
|
|
chromedp.Navigate(target),
|
|
chromedp.WaitReady("body", chromedp.ByQuery),
|
|
); err != nil {
|
|
return classifyBrowserError(callerCtx, browserConnectionLost(tabCtx), err)
|
|
}
|
|
var lastErr error
|
|
for {
|
|
// The challenge reloads the page when it passes, which tears down the
|
|
// execution context mid-read. That is a retry, not a failure.
|
|
if err := chromedp.Run(tabCtx, read); err != nil {
|
|
err = classifyBrowserError(callerCtx, browserConnectionLost(tabCtx), err)
|
|
if errors.Is(err, errBrowserInterrupted) {
|
|
return err
|
|
}
|
|
lastErr = err
|
|
} else if done() {
|
|
return nil
|
|
}
|
|
select {
|
|
case <-ctx.Done():
|
|
if err := callerCtx.Err(); err != nil {
|
|
return err
|
|
}
|
|
if lastErr != nil {
|
|
return fmt.Errorf("%w (last read: %v)", errChallengeHeld, lastErr)
|
|
}
|
|
return errChallengeHeld
|
|
case <-time.After(challengePollInterval):
|
|
}
|
|
}
|
|
}
|
|
|
|
// kaganeAPIURL maps a stored series_url to the JSON endpoint carrying its
|
|
// chapter list. Returning false for anything else is a second line of defence
|
|
// behind fetchableSeriesURL: a headless browser is a strong SSRF primitive and
|
|
// series_url is client-supplied, so the host is pinned here too.
|
|
func kaganeAPIURL(seriesURL string) (string, bool) {
|
|
u, err := url.Parse(seriesURL)
|
|
if err != nil || u.Scheme != "https" || u.Hostname() != "kagane.to" {
|
|
return "", false
|
|
}
|
|
m := kaganeSeriesRe.FindStringSubmatch(u.Path)
|
|
if m == nil {
|
|
return "", false
|
|
}
|
|
return "https://kagane.to/api/v2/series/" + m[1], true
|
|
}
|
|
|
|
// novelfullSeriesURL reports whether seriesURL is a novelfull series page this
|
|
// fetcher will open. novelfull's chapter list is in the served HTML, so unlike
|
|
// kagane there is no API to call from inside the page — the challenge-cleared
|
|
// DOM is the payload. The host is pinned here for the same reason kagane's is:
|
|
// series_url is client-supplied and a headless browser is a strong SSRF
|
|
// primitive.
|
|
func novelfullSeriesURL(seriesURL string) bool {
|
|
u, err := url.Parse(seriesURL)
|
|
return err == nil && u.Scheme == "https" && u.Hostname() == "novelfull.com" &&
|
|
strings.HasSuffix(u.Path, ".html")
|
|
}
|
|
|
|
// awaitPromise makes Evaluate resolve the promise rather than returning a
|
|
// serialised Promise object.
|
|
func awaitPromise(p *runtime.EvaluateParams) *runtime.EvaluateParams {
|
|
return p.WithAwaitPromise(true)
|
|
}
|
|
|
|
// jsString renders s as a JavaScript string literal for embedding in an
|
|
// Evaluate expression. The URL is host-pinned by kaganeAPIURL before it gets
|
|
// here, but quoting it properly is what keeps that guarantee intact.
|
|
func jsString(s string) string {
|
|
b, _ := json.Marshal(s)
|
|
return string(b)
|
|
}
|