Files
2026-08-16 21:17:29 +07:00

381 lines
15 KiB
Go

package latest
import (
"context"
"encoding/base64"
"encoding/json"
"errors"
"fmt"
"net/url"
"regexp"
"strings"
"sync"
"time"
"github.com/chromedp/cdproto/runtime"
"github.com/chromedp/chromedp"
)
// challengeTimeout bounds one navigate-and-solve. A Cloudflare managed
// challenge clears in a few seconds when it clears at all; anything longer is a
// challenge that is not going to pass, and the caller's rest was already
// stamped before this ran.
const challengeTimeout = 45 * time.Second
var kaganeSeriesRe = regexp.MustCompile(`^/series/([0-9a-f-]{36})/?$`)
// comixSeriesPathRe matches the one path shape comixRead will open: a Series
// page, "/title/<id>-<slug>". Verified live 2026-08-12.
var comixSeriesPathRe = regexp.MustCompile(`^/title/[^/?#]+/?$`)
// BrowserFetcher retrieves pages through a remote headless Chrome over the
// DevTools Protocol.
//
// It exists for one reason: kagane.to, novelfull.com and comix.to sit behind a
// Cloudflare JavaScript challenge. Verified 2026-08-03 (kagane), 2026-08-05
// (novelfull) and 2026-08-12 (comix), plain HTTP and bogdanfinn/tls-client
// with a Chrome_133 profile both get 403 with cf-mitigated: challenge on every
// path, including the API, robots.txt and images. Clearing it requires
// executing the challenge script, which only a real browser does.
//
// The request is made *inside* the page rather than by extracting cf_clearance
// and replaying it through TLSFetcher. That cookie is bound to IP, User-Agent
// and often the TLS fingerprint, so replaying it means keeping three things in
// sync that break silently and separately. The browser's own cookie jar
// persists across polls, so the challenge is solved once every few hours.
//
// The three sites differ in what a cleared tab is asked for: kagane fetches a
// JSON API from inside the page (the list exists nowhere else), comix fetches
// its own Series URL from inside the page (the served HTML carries the facts,
// and rendering the SPA costs ~65 requests instead of one), and novelfull
// renders its list into the HTML so the cleared DOM is the payload.
type BrowserFetcher struct {
allocCtx context.Context
cancel context.CancelFunc
// One page at a time: caps the browser's memory — it runs under a hard
// cgroup cap on a shared machine — and keeps series from sharing page state.
mu sync.Mutex
}
var _ Fetcher = (*BrowserFetcher)(nil)
// NewBrowserFetcher connects to a Chrome over CDP. The browser is not a
// sidecar: it runs on a separate machine and is reached over the tailnet
// (ADR-0006), so wsURL is that machine's tailnet address, e.g.
// ws://100.64.0.5:9222.
//
// It must be an IP, never a hostname — not MagicDNS, not a Docker service
// name. Chrome's DevTools HTTP handler 500s any /json/version request whose
// Host header isn't an IP or "localhost" (confirmed 2026-08-03), so a name
// fails at discovery and surfaces as a dead site rather than a bad URL.
//
// Do not add chromedp.NoModifyURL here: that option skips the /json/version
// discovery request entirely and dials wsURL as if it were already the full
// debugger endpoint, but Chrome only accepts connections at
// /devtools/browser/<uuid>, a path chosen fresh at every Chrome start — dialing
// the bare host:port 404s. The default (discovery) path works precisely
// because Chrome's /json/version response echoes back the Host header of the
// discovery request in webSocketDebuggerUrl, so as long as wsURL is an IP this
// process can reach, the URL chromedp gets back already points at it. That is
// also why a Chrome restarted behind a stable endpoint needs no reconnect
// here: the fresh UUID arrives with the next discovery.
func NewBrowserFetcher(wsURL string) (*BrowserFetcher, error) {
if wsURL == "" {
return nil, fmt.Errorf("empty browser websocket url")
}
ctx, cancel := chromedp.NewRemoteAllocator(context.Background(), wsURL)
return &BrowserFetcher{allocCtx: ctx, cancel: cancel}, nil
}
func (f *BrowserFetcher) Close() {
f.cancel()
}
// Get navigates to seriesURL, lets any challenge resolve, then reads the
// payload the Site's registry entry describes (the shapes are listed on
// BrowserFetcher). The returned body is whatever the Site's chapter list lives
// in, which is what the entry's LatestChapter parse expects.
func (f *BrowserFetcher) Get(ctx context.Context, seriesURL string) (string, int, error) {
var body string
// Sorted order (browserBackedSites sorts) makes dispatch deterministic:
// entries' Read funcs are expected to refuse any address owned by another
// Site, and the loop must not depend on that staying true.
for _, name := range browserBackedSites() {
s := sites[name]
read, ok := s.Browser.Read(seriesURL, &body)
if !ok {
continue
}
if err := f.run(ctx, seriesURL, read,
func() bool { return s.Browser.Done(body) }); err != nil {
// Challenge never cleared, or the payload was refused.
// Indistinguishable from here and handled identically by the caller.
if errors.Is(err, errChallengeHeld) {
return "", 403, nil
}
return "", 0, fmt.Errorf("browser fetch %q: %w", seriesURL, err)
}
return body, 200, nil
}
return "", 0, fmt.Errorf("not a fetchable browser series url: %q", seriesURL)
}
// kaganeRead builds the in-tab fetch of kagane's chapter-list API: the
// request must be made from inside the page so it carries the clearance
// cookie, and the API is the only place the list exists. Refusing any other
// address is the per-Site half of the SSRF gate, kept deliberately behind
// fetchableSeriesURL (see browserRead.Read).
func kaganeRead(seriesURL string, out *string) (chromedp.Action, bool) {
apiURL, ok := kaganeAPIURL(seriesURL)
if !ok {
return nil, false
}
return chromedp.Evaluate(
`fetch(`+jsString(apiURL)+`).then(r => r.ok ? r.text() : "")`,
out, awaitPromise), true
}
// novelfullRead reads the cleared DOM. novelfull renders its chapter list
// into the served HTML, so there is no API to call from inside the page — the
// challenge-cleared DOM is the payload.
func novelfullRead(seriesURL string, out *string) (chromedp.Action, bool) {
if !novelfullSeriesURL(seriesURL) {
return nil, false
}
return chromedp.OuterHTML("html", out, chromedp.ByQuery), true
}
// comixRead fetches the Series page from inside the cleared tab. comix is an
// SPA: rendering the page costs ~65 requests, while one same-origin fetch of
// the same address returns the server-rendered HTML — 24.5 KB, ~480 ms,
// carrying both parser anchors (measured 2026-08-12, issue #98). So this is
// kaganeRead's shape, not novelfullRead's, even though the payload is HTML.
// Refusing any other address is the per-Site half of the SSRF gate.
func comixRead(seriesURL string, out *string) (chromedp.Action, bool) {
pageURL, ok := comixSeriesPageURL(seriesURL)
if !ok {
return nil, false
}
return chromedp.Evaluate(
`fetch(`+jsString(pageURL)+`).then(r => r.ok ? r.text() : "")`,
out, awaitPromise), true
}
// Image retrieves one cover's bytes through the browser sidecar, and its
// content type.
//
// It exists because kagane and comix serve covers behind the same challenge as
// their pages — kagane additionally with
// `cross-origin-resource-policy: same-origin` — so the bytes are only
// reachable from inside a browser that already holds the clearance cookie
// (verified 2026-08-08 for kagane, 2026-08-12 for comix). Acquisition through
// the sidecar is the only route.
//
// The image URL is navigated to rather than fetched from another page of the
// Site: the challenge only runs on a top-level navigation, and once it clears
// the document *is* the image, so a same-origin fetch of location.href reads
// it straight back out of the cache. For comix the navigation is also the only
// route that works at all — its Series page sets
// `cross-origin-embedder-policy: require-corp`, which fails a page-context
// fetch of the cover host.
//
// The challenge is not solved by the first read: WaitReady("body") is satisfied
// by the interstitial too. run holds the tab open until the in-page fetch
// succeeds, which is what gives the challenge script the seconds it needs.
func (f *BrowserFetcher) Image(ctx context.Context, imageURL string) ([]byte, string, error) {
if !browserOnlyCoverURL(imageURL) {
return nil, "", fmt.Errorf("not a browser-fetchable cover url: %q", imageURL)
}
var dataURL string
err := f.run(ctx, imageURL,
chromedp.Evaluate(`fetch(location.href).then(r => r.ok
? r.blob().then(b => new Promise(res => {
const fr = new FileReader();
fr.onload = () => res(fr.result);
fr.readAsDataURL(b);
}))
: "")`, &dataURL, awaitPromise),
func() bool { return dataURL != "" })
if err != nil {
return nil, "", fmt.Errorf("browser image %s: %w", imageURL, err)
}
// "data:image/webp;base64,<payload>".
head, payload, ok := strings.Cut(dataURL, ";base64,")
if !ok {
return nil, "", fmt.Errorf("browser image %s: not a data url", imageURL)
}
raw, err := base64.StdEncoding.DecodeString(payload)
if err != nil {
return nil, "", fmt.Errorf("browser image %s: %w", imageURL, err)
}
return raw, strings.TrimPrefix(head, "data:"), nil
}
// errChallengeHeld reports that the budget ran out with the interstitial still
// up. Distinct from a transport failure: it means "this site said no", which
// the poller answers with a refusal backoff for that Site's Lane (issue #100).
var errChallengeHeld = errors.New("challenge held")
// errBrowserInterrupted distinguishes a remote Chrome restart from the
// caller's own deadline. chromedp reports both as context.Canceled.
var errBrowserInterrupted = errors.New("browser interrupted")
func classifyBrowserError(ctx context.Context, browserLost bool, err error) error {
if err == nil || ctx.Err() != nil {
return err
}
if !browserLost {
return err
}
if !errors.Is(err, context.Canceled) {
return err
}
return fmt.Errorf("%w: %w", errBrowserInterrupted, err)
}
func browserConnectionLost(ctx context.Context) bool {
c := chromedp.FromContext(ctx)
if c == nil || c.Browser == nil {
return true
}
select {
case <-c.Browser.LostConnection:
return true
default:
return false
}
}
// challengePollInterval paces re-reads while a challenge solves itself.
const challengePollInterval = 2 * time.Second
// isInterstitial reports whether html is Cloudflare's challenge page rather
// than the site's own. Matched on the challenge orchestration path
// (/cdn-cgi/challenge-platform/h/<b|g|x>/orchestrate/...), which is stable
// across the interstitial's wording and locale — the visible "Just a
// moment..." title is neither.
//
// The bare "/cdn-cgi/challenge-platform/" prefix is NOT enough: Cloudflare
// injects /cdn-cgi/challenge-platform/scripts/jsd/main.js into ordinary 200
// pages when JS detections are on, so matching the prefix declared every real
// demonic page a refusal and parked that Lane in 15m backoff (observed
// 2026-08-16, demonic turned detections on).
func isInterstitial(html string) bool {
return strings.Contains(html, "/cdn-cgi/challenge-platform/h/")
}
// run navigates to target and re-reads until done reports an answer, bounded by
// challengeTimeout and by the caller's own deadline, in a tab that is closed on
// return so one wedged page cannot poison later calls.
//
// Holding the tab open across re-reads is the whole point. A Cloudflare
// interstitial needs several seconds of a live page to solve itself and write
// clearance into the browser's shared cookie jar; reading once and closing the
// tab — which is what this did before 2026-08-08 — never gives it that window,
// so every fetch lands on the interstitial and the clearance that would have
// unblocked all the later ones is never obtained.
func (f *BrowserFetcher) run(ctx context.Context, target string, read chromedp.Action, done func() bool) error {
f.mu.Lock()
defer f.mu.Unlock()
callerCtx := ctx
ctx, cancel := context.WithTimeout(ctx, challengeTimeout)
defer cancel()
tabCtx, cancelTab := chromedp.NewContext(f.allocCtx)
defer cancelTab()
// Bind the caller's deadline to the tab.
tabCtx, cancelDeadline := context.WithCancel(tabCtx)
defer cancelDeadline()
go func() {
<-ctx.Done()
cancelDeadline()
}()
if err := chromedp.Run(tabCtx,
chromedp.Navigate(target),
chromedp.WaitReady("body", chromedp.ByQuery),
); err != nil {
return classifyBrowserError(callerCtx, browserConnectionLost(tabCtx), err)
}
var lastErr error
for {
// The challenge reloads the page when it passes, which tears down the
// execution context mid-read. That is a retry, not a failure.
if err := chromedp.Run(tabCtx, read); err != nil {
err = classifyBrowserError(callerCtx, browserConnectionLost(tabCtx), err)
if errors.Is(err, errBrowserInterrupted) {
return err
}
lastErr = err
} else if done() {
return nil
}
select {
case <-ctx.Done():
if err := callerCtx.Err(); err != nil {
return err
}
if lastErr != nil {
return fmt.Errorf("%w (last read: %v)", errChallengeHeld, lastErr)
}
return errChallengeHeld
case <-time.After(challengePollInterval):
}
}
}
// kaganeAPIURL maps a stored series_url to the JSON endpoint carrying its
// chapter list. Returning false for anything else is a second line of defence
// behind fetchableSeriesURL: a headless browser is a strong SSRF primitive and
// series_url is client-supplied, so the host is pinned here too.
func kaganeAPIURL(seriesURL string) (string, bool) {
u, err := url.Parse(seriesURL)
if err != nil || u.Scheme != "https" || u.Hostname() != "kagane.to" {
return "", false
}
m := kaganeSeriesRe.FindStringSubmatch(u.Path)
if m == nil {
return "", false
}
return "https://kagane.to/api/v2/series/" + m[1], true
}
// novelfullSeriesURL reports whether seriesURL is a novelfull series page this
// fetcher will open. novelfull's chapter list is in the served HTML, so unlike
// kagane there is no API to call from inside the page — the challenge-cleared
// DOM is the payload. The host is pinned here for the same reason kagane's is:
// series_url is client-supplied and a headless browser is a strong SSRF
// primitive.
func novelfullSeriesURL(seriesURL string) bool {
u, err := url.Parse(seriesURL)
return err == nil && u.Scheme == "https" && u.Hostname() == "novelfull.com" &&
strings.HasSuffix(u.Path, ".html")
}
// comixSeriesPageURL returns the address comixRead fetches inside the tab: the
// Series page itself, rebuilt from the pinned host and path so nothing else
// travels. Host-pinned here for the same reason kagane's is — series_url is
// client-supplied and a headless browser is a strong SSRF primitive.
func comixSeriesPageURL(seriesURL string) (string, bool) {
u, err := url.Parse(seriesURL)
if err != nil || u.Scheme != "https" || u.Hostname() != "comix.to" ||
!comixSeriesPathRe.MatchString(u.Path) {
return "", false
}
return "https://comix.to" + u.Path, true
}
// awaitPromise makes Evaluate resolve the promise rather than returning a
// serialised Promise object.
func awaitPromise(p *runtime.EvaluateParams) *runtime.EvaluateParams {
return p.WithAwaitPromise(true)
}
// jsString renders s as a JavaScript string literal for embedding in an
// Evaluate expression. The URL is host-pinned by kaganeAPIURL before it gets
// here, but quoting it properly is what keeps that guarantee intact.
func jsString(s string) string {
b, _ := json.Marshal(s)
return string(b)
}