d998f8725c
Restore two review findings against spec claims 'the Poll keeps its own Cover policy' and 'pinning is the only behavioural change': prefetchCover now also runs on ticks where the page fetch itself fails (the heal is independent of the page read, and its source may answer while the origin does not), and the no-chapter log surfaces the fetched body length again via a BodyLen fact on seriesRead. Browser dispatch iterates the sorted browser site list so its outcome cannot depend on map order.
339 lines
13 KiB
Go
339 lines
13 KiB
Go
package latest
|
|
|
|
import (
|
|
"context"
|
|
"encoding/base64"
|
|
"encoding/json"
|
|
"errors"
|
|
"fmt"
|
|
"net/url"
|
|
"regexp"
|
|
"strings"
|
|
"sync"
|
|
"time"
|
|
|
|
"github.com/chromedp/cdproto/runtime"
|
|
"github.com/chromedp/chromedp"
|
|
)
|
|
|
|
// challengeTimeout bounds one navigate-and-solve. A Cloudflare managed
|
|
// challenge clears in a few seconds when it clears at all; anything longer is a
|
|
// challenge that is not going to pass, and the caller's cooldown was already
|
|
// stamped before this ran.
|
|
const challengeTimeout = 45 * time.Second
|
|
|
|
var kaganeSeriesRe = regexp.MustCompile(`^/series/([0-9a-f-]{36})/?$`)
|
|
|
|
// BrowserFetcher retrieves pages through a remote headless Chrome over the
|
|
// DevTools Protocol.
|
|
//
|
|
// It exists for one reason: kagane.to and novelfull.com sit behind a
|
|
// Cloudflare JavaScript challenge. Verified 2026-08-03 (kagane) and 2026-08-05
|
|
// (novelfull) from the deployment host, plain HTTP and bogdanfinn/tls-client
|
|
// with a Chrome_133 profile both get 403 with cf-mitigated: challenge on every
|
|
// path, including the API, robots.txt and images. Clearing it requires
|
|
// executing the challenge script, which only a real browser does.
|
|
//
|
|
// The request is made *inside* the page rather than by extracting cf_clearance
|
|
// and replaying it through TLSFetcher. That cookie is bound to IP, User-Agent
|
|
// and often the TLS fingerprint, so replaying it means keeping three things in
|
|
// sync that break silently and separately. The browser's own cookie jar
|
|
// persists across polls, so the challenge is solved once every few hours.
|
|
//
|
|
// The two sites differ in how the chapter list is read: kagane serves it from
|
|
// a JSON API that must be called from inside the page (so the request carries
|
|
// the clearance cookie), while novelfull renders it into the HTML so the
|
|
// cleared DOM is the payload.
|
|
type BrowserFetcher struct {
|
|
allocCtx context.Context
|
|
cancel context.CancelFunc
|
|
// One page at a time: caps the browser's memory — it runs under a hard
|
|
// cgroup cap on a shared machine — and keeps series from sharing page state.
|
|
mu sync.Mutex
|
|
}
|
|
|
|
var _ Fetcher = (*BrowserFetcher)(nil)
|
|
|
|
// NewBrowserFetcher connects to a Chrome over CDP. The browser is not a
|
|
// sidecar: it runs on a separate machine and is reached over the tailnet
|
|
// (ADR-0006), so wsURL is that machine's tailnet address, e.g.
|
|
// ws://100.64.0.5:9222.
|
|
//
|
|
// It must be an IP, never a hostname — not MagicDNS, not a Docker service
|
|
// name. Chrome's DevTools HTTP handler 500s any /json/version request whose
|
|
// Host header isn't an IP or "localhost" (confirmed 2026-08-03), so a name
|
|
// fails at discovery and surfaces as a dead site rather than a bad URL.
|
|
//
|
|
// Do not add chromedp.NoModifyURL here: that option skips the /json/version
|
|
// discovery request entirely and dials wsURL as if it were already the full
|
|
// debugger endpoint, but Chrome only accepts connections at
|
|
// /devtools/browser/<uuid>, a path chosen fresh at every Chrome start — dialing
|
|
// the bare host:port 404s. The default (discovery) path works precisely
|
|
// because Chrome's /json/version response echoes back the Host header of the
|
|
// discovery request in webSocketDebuggerUrl, so as long as wsURL is an IP this
|
|
// process can reach, the URL chromedp gets back already points at it. That is
|
|
// also why a Chrome restarted behind a stable endpoint needs no reconnect
|
|
// here: the fresh UUID arrives with the next discovery.
|
|
func NewBrowserFetcher(wsURL string) (*BrowserFetcher, error) {
|
|
if wsURL == "" {
|
|
return nil, fmt.Errorf("empty browser websocket url")
|
|
}
|
|
ctx, cancel := chromedp.NewRemoteAllocator(context.Background(), wsURL)
|
|
return &BrowserFetcher{allocCtx: ctx, cancel: cancel}, nil
|
|
}
|
|
|
|
func (f *BrowserFetcher) Close() {
|
|
f.cancel()
|
|
}
|
|
|
|
// Get navigates to seriesURL, lets any challenge resolve, then reads the
|
|
// payload the Site's registry entry describes — kagane's chapter-list API from
|
|
// inside the page so the request carries the clearance cookie, novelfull's
|
|
// served HTML. The returned body is whatever the Site's chapter list lives in,
|
|
// which is what the entry's LatestChapter parse expects.
|
|
func (f *BrowserFetcher) Get(ctx context.Context, seriesURL string) (string, int, error) {
|
|
var body string
|
|
// Sorted order (browserBackedSites sorts) makes dispatch deterministic:
|
|
// entries' Read funcs are expected to refuse any address owned by another
|
|
// Site, and the loop must not depend on that staying true.
|
|
for _, name := range browserBackedSites() {
|
|
s := sites[name]
|
|
read, ok := s.Browser.Read(seriesURL, &body)
|
|
if !ok {
|
|
continue
|
|
}
|
|
if err := f.run(ctx, seriesURL, read,
|
|
func() bool { return s.Browser.Done(body) }); err != nil {
|
|
// Challenge never cleared, or the payload was refused.
|
|
// Indistinguishable from here and handled identically by the caller.
|
|
if errors.Is(err, errChallengeHeld) {
|
|
return "", 403, nil
|
|
}
|
|
return "", 0, fmt.Errorf("browser fetch %q: %w", seriesURL, err)
|
|
}
|
|
return body, 200, nil
|
|
}
|
|
return "", 0, fmt.Errorf("not a fetchable browser series url: %q", seriesURL)
|
|
}
|
|
|
|
// kaganeRead builds the in-tab fetch of kagane's chapter-list API: the
|
|
// request must be made from inside the page so it carries the clearance
|
|
// cookie, and the API is the only place the list exists. Refusing any other
|
|
// address is the per-Site half of the SSRF gate, kept deliberately behind
|
|
// fetchableSeriesURL (see browserRead.Read).
|
|
func kaganeRead(seriesURL string, out *string) (chromedp.Action, bool) {
|
|
apiURL, ok := kaganeAPIURL(seriesURL)
|
|
if !ok {
|
|
return nil, false
|
|
}
|
|
return chromedp.Evaluate(
|
|
`fetch(`+jsString(apiURL)+`).then(r => r.ok ? r.text() : "")`,
|
|
out, awaitPromise), true
|
|
}
|
|
|
|
// novelfullRead reads the cleared DOM. novelfull renders its chapter list
|
|
// into the served HTML, so there is no API to call from inside the page — the
|
|
// challenge-cleared DOM is the payload.
|
|
func novelfullRead(seriesURL string, out *string) (chromedp.Action, bool) {
|
|
if !novelfullSeriesURL(seriesURL) {
|
|
return nil, false
|
|
}
|
|
return chromedp.OuterHTML("html", out, chromedp.ByQuery), true
|
|
}
|
|
|
|
// Image retrieves one cover's bytes through the browser sidecar, and its
|
|
// content type.
|
|
//
|
|
// It exists because kagane serves covers behind the same challenge as its
|
|
// pages *and* with `cross-origin-resource-policy: same-origin`, so the bytes
|
|
// are only reachable from inside a browser that already holds the clearance
|
|
// cookie (verified 2026-08-08). Acquisition through the sidecar is the only
|
|
// route.
|
|
//
|
|
// The image URL is navigated to rather than fetched from some other kagane
|
|
// page: the challenge only runs on a top-level navigation, and once it clears
|
|
// the document *is* the image, so a same-origin fetch of location.href reads
|
|
// it straight back out of the cache.
|
|
//
|
|
// The challenge is not solved by the first read: WaitReady("body") is satisfied
|
|
// by the interstitial too. run holds the tab open until the in-page fetch
|
|
// succeeds, which is what gives the challenge script the seconds it needs.
|
|
func (f *BrowserFetcher) Image(ctx context.Context, imageURL string) ([]byte, string, error) {
|
|
m := kaganeImageURLRe.FindStringSubmatch(imageURL)
|
|
if m == nil {
|
|
return nil, "", fmt.Errorf("not a browser-fetchable cover url: %q", imageURL)
|
|
}
|
|
imageID := m[1]
|
|
var dataURL string
|
|
err := f.run(ctx, imageURL,
|
|
chromedp.Evaluate(`fetch(location.href).then(r => r.ok
|
|
? r.blob().then(b => new Promise(res => {
|
|
const fr = new FileReader();
|
|
fr.onload = () => res(fr.result);
|
|
fr.readAsDataURL(b);
|
|
}))
|
|
: "")`, &dataURL, awaitPromise),
|
|
func() bool { return dataURL != "" })
|
|
if err != nil {
|
|
return nil, "", fmt.Errorf("browser image %s: %w", imageID, err)
|
|
}
|
|
// "data:image/webp;base64,<payload>".
|
|
head, payload, ok := strings.Cut(dataURL, ";base64,")
|
|
if !ok {
|
|
return nil, "", fmt.Errorf("browser image %s: not a data url", imageID)
|
|
}
|
|
raw, err := base64.StdEncoding.DecodeString(payload)
|
|
if err != nil {
|
|
return nil, "", fmt.Errorf("browser image %s: %w", imageID, err)
|
|
}
|
|
return raw, strings.TrimPrefix(head, "data:"), nil
|
|
}
|
|
|
|
// errChallengeHeld reports that the budget ran out with the interstitial still
|
|
// up. Distinct from a transport failure: it means "this site said no", which
|
|
// the poller answers with a 403 and its ordinary cooldown.
|
|
var errChallengeHeld = errors.New("challenge held")
|
|
|
|
// errBrowserInterrupted distinguishes a remote Chrome restart from the
|
|
// caller's own deadline. chromedp reports both as context.Canceled.
|
|
var errBrowserInterrupted = errors.New("browser interrupted")
|
|
|
|
func classifyBrowserError(ctx context.Context, browserLost bool, err error) error {
|
|
if err == nil || ctx.Err() != nil {
|
|
return err
|
|
}
|
|
if !browserLost {
|
|
return err
|
|
}
|
|
if !errors.Is(err, context.Canceled) {
|
|
return err
|
|
}
|
|
return fmt.Errorf("%w: %w", errBrowserInterrupted, err)
|
|
}
|
|
|
|
func browserConnectionLost(ctx context.Context) bool {
|
|
c := chromedp.FromContext(ctx)
|
|
if c == nil || c.Browser == nil {
|
|
return true
|
|
}
|
|
select {
|
|
case <-c.Browser.LostConnection:
|
|
return true
|
|
default:
|
|
return false
|
|
}
|
|
}
|
|
|
|
// challengePollInterval paces re-reads while a challenge solves itself.
|
|
const challengePollInterval = 2 * time.Second
|
|
|
|
// isInterstitial reports whether html is Cloudflare's challenge page rather
|
|
// than the site's own. Matched on the challenge runtime's script path, which is
|
|
// stable across the interstitial's wording and locale — the visible "Just a
|
|
// moment..." title is neither.
|
|
func isInterstitial(html string) bool {
|
|
return strings.Contains(html, "/cdn-cgi/challenge-platform/")
|
|
}
|
|
|
|
// run navigates to target and re-reads until done reports an answer, bounded by
|
|
// challengeTimeout and by the caller's own deadline, in a tab that is closed on
|
|
// return so one wedged page cannot poison later calls.
|
|
//
|
|
// Holding the tab open across re-reads is the whole point. A Cloudflare
|
|
// interstitial needs several seconds of a live page to solve itself and write
|
|
// clearance into the browser's shared cookie jar; reading once and closing the
|
|
// tab — which is what this did before 2026-08-08 — never gives it that window,
|
|
// so every fetch lands on the interstitial and the clearance that would have
|
|
// unblocked all the later ones is never obtained.
|
|
func (f *BrowserFetcher) run(ctx context.Context, target string, read chromedp.Action, done func() bool) error {
|
|
f.mu.Lock()
|
|
defer f.mu.Unlock()
|
|
|
|
callerCtx := ctx
|
|
ctx, cancel := context.WithTimeout(ctx, challengeTimeout)
|
|
defer cancel()
|
|
tabCtx, cancelTab := chromedp.NewContext(f.allocCtx)
|
|
defer cancelTab()
|
|
// Bind the caller's deadline to the tab.
|
|
tabCtx, cancelDeadline := context.WithCancel(tabCtx)
|
|
defer cancelDeadline()
|
|
go func() {
|
|
<-ctx.Done()
|
|
cancelDeadline()
|
|
}()
|
|
|
|
if err := chromedp.Run(tabCtx,
|
|
chromedp.Navigate(target),
|
|
chromedp.WaitReady("body", chromedp.ByQuery),
|
|
); err != nil {
|
|
return classifyBrowserError(callerCtx, browserConnectionLost(tabCtx), err)
|
|
}
|
|
var lastErr error
|
|
for {
|
|
// The challenge reloads the page when it passes, which tears down the
|
|
// execution context mid-read. That is a retry, not a failure.
|
|
if err := chromedp.Run(tabCtx, read); err != nil {
|
|
err = classifyBrowserError(callerCtx, browserConnectionLost(tabCtx), err)
|
|
if errors.Is(err, errBrowserInterrupted) {
|
|
return err
|
|
}
|
|
lastErr = err
|
|
} else if done() {
|
|
return nil
|
|
}
|
|
select {
|
|
case <-ctx.Done():
|
|
if err := callerCtx.Err(); err != nil {
|
|
return err
|
|
}
|
|
if lastErr != nil {
|
|
return fmt.Errorf("%w (last read: %v)", errChallengeHeld, lastErr)
|
|
}
|
|
return errChallengeHeld
|
|
case <-time.After(challengePollInterval):
|
|
}
|
|
}
|
|
}
|
|
|
|
// kaganeAPIURL maps a stored series_url to the JSON endpoint carrying its
|
|
// chapter list. Returning false for anything else is a second line of defence
|
|
// behind fetchableSeriesURL: a headless browser is a strong SSRF primitive and
|
|
// series_url is client-supplied, so the host is pinned here too.
|
|
func kaganeAPIURL(seriesURL string) (string, bool) {
|
|
u, err := url.Parse(seriesURL)
|
|
if err != nil || u.Scheme != "https" || u.Hostname() != "kagane.to" {
|
|
return "", false
|
|
}
|
|
m := kaganeSeriesRe.FindStringSubmatch(u.Path)
|
|
if m == nil {
|
|
return "", false
|
|
}
|
|
return "https://kagane.to/api/v2/series/" + m[1], true
|
|
}
|
|
|
|
// novelfullSeriesURL reports whether seriesURL is a novelfull series page this
|
|
// fetcher will open. novelfull's chapter list is in the served HTML, so unlike
|
|
// kagane there is no API to call from inside the page — the challenge-cleared
|
|
// DOM is the payload. The host is pinned here for the same reason kagane's is:
|
|
// series_url is client-supplied and a headless browser is a strong SSRF
|
|
// primitive.
|
|
func novelfullSeriesURL(seriesURL string) bool {
|
|
u, err := url.Parse(seriesURL)
|
|
return err == nil && u.Scheme == "https" && u.Hostname() == "novelfull.com" &&
|
|
strings.HasSuffix(u.Path, ".html")
|
|
}
|
|
|
|
// awaitPromise makes Evaluate resolve the promise rather than returning a
|
|
// serialised Promise object.
|
|
func awaitPromise(p *runtime.EvaluateParams) *runtime.EvaluateParams {
|
|
return p.WithAwaitPromise(true)
|
|
}
|
|
|
|
// jsString renders s as a JavaScript string literal for embedding in an
|
|
// Evaluate expression. The URL is host-pinned by kaganeAPIURL before it gets
|
|
// here, but quoting it properly is what keeps that guarantee intact.
|
|
func jsString(s string) string {
|
|
b, _ := json.Marshal(s)
|
|
return string(b)
|
|
}
|