Files
mangaBookmark/backend/internal/latest/browser.go
T
sulthan 0a79e5f3d7 Address review findings on the cover-path deletion (#63)
- Move the kagane cover URL shape into the extraction module (sites.go):
  browserOnlyCoverURL + kaganeImageURLRe now own the claim; the byte-fetch
  router and BrowserFetcher.Image reference it. One shape gate for producer
  and fetcher (the id regex is folded into the full-URL match), so no Site
  name appears in a cover path outside the extraction module and the
  producer cannot emit an address the fetch would refuse.
- Restore the serving-boundary guarantee: GET /covers/{addr} re-checks the
  stored media type via store.CoverContentType and 404s a poisoned row;
  TestPublicCoverNeverEchoesNonImage now seeds one directly behind the
  write gate and pins the refusal where bytes leave.
- Restore the SSRF rationale (client-supplied stored URL, headless browser
  as a strong primitive) on the URL regex.
2026-08-10 11:36:54 +07:00

326 lines
13 KiB
Go

package latest
import (
"context"
"encoding/base64"
"encoding/json"
"errors"
"fmt"
"net/url"
"regexp"
"strings"
"sync"
"time"
"github.com/chromedp/cdproto/runtime"
"github.com/chromedp/chromedp"
)
// challengeTimeout bounds one navigate-and-solve. A Cloudflare managed
// challenge clears in a few seconds when it clears at all; anything longer is a
// challenge that is not going to pass, and the caller's cooldown was already
// stamped before this ran.
const challengeTimeout = 45 * time.Second
var kaganeSeriesRe = regexp.MustCompile(`^/series/([0-9a-f-]{36})/?$`)
// BrowserFetcher retrieves pages through a remote headless Chrome over the
// DevTools Protocol.
//
// It exists for one reason: kagane.to and novelfull.com sit behind a
// Cloudflare JavaScript challenge. Verified 2026-08-03 (kagane) and 2026-08-05
// (novelfull) from the deployment host, plain HTTP and bogdanfinn/tls-client
// with a Chrome_133 profile both get 403 with cf-mitigated: challenge on every
// path, including the API, robots.txt and images. Clearing it requires
// executing the challenge script, which only a real browser does.
//
// The request is made *inside* the page rather than by extracting cf_clearance
// and replaying it through TLSFetcher. That cookie is bound to IP, User-Agent
// and often the TLS fingerprint, so replaying it means keeping three things in
// sync that break silently and separately. The browser's own cookie jar
// persists across polls, so the challenge is solved once every few hours.
//
// The two sites differ in how the chapter list is read: kagane serves it from
// a JSON API that must be called from inside the page (so the request carries
// the clearance cookie), while novelfull renders it into the HTML so the
// cleared DOM is the payload.
type BrowserFetcher struct {
allocCtx context.Context
cancel context.CancelFunc
// One page at a time: caps the browser's memory — it runs under a hard
// cgroup cap on a shared machine — and keeps series from sharing page state.
mu sync.Mutex
}
var _ Fetcher = (*BrowserFetcher)(nil)
// NewBrowserFetcher connects to a Chrome over CDP. The browser is not a
// sidecar: it runs on a separate machine and is reached over the tailnet
// (ADR-0006), so wsURL is that machine's tailnet address, e.g.
// ws://100.64.0.5:9222.
//
// It must be an IP, never a hostname — not MagicDNS, not a Docker service
// name. Chrome's DevTools HTTP handler 500s any /json/version request whose
// Host header isn't an IP or "localhost" (confirmed 2026-08-03), so a name
// fails at discovery and surfaces as a dead site rather than a bad URL.
//
// Do not add chromedp.NoModifyURL here: that option skips the /json/version
// discovery request entirely and dials wsURL as if it were already the full
// debugger endpoint, but Chrome only accepts connections at
// /devtools/browser/<uuid>, a path chosen fresh at every Chrome start — dialing
// the bare host:port 404s. The default (discovery) path works precisely
// because Chrome's /json/version response echoes back the Host header of the
// discovery request in webSocketDebuggerUrl, so as long as wsURL is an IP this
// process can reach, the URL chromedp gets back already points at it. That is
// also why a Chrome restarted behind a stable endpoint needs no reconnect
// here: the fresh UUID arrives with the next discovery.
func NewBrowserFetcher(wsURL string) (*BrowserFetcher, error) {
if wsURL == "" {
return nil, fmt.Errorf("empty browser websocket url")
}
ctx, cancel := chromedp.NewRemoteAllocator(context.Background(), wsURL)
return &BrowserFetcher{allocCtx: ctx, cancel: cancel}, nil
}
func (f *BrowserFetcher) Close() {
f.cancel()
}
// Get navigates to seriesURL, lets any challenge resolve, then reads either the
// site's JSON API (kagane) from inside the page so the request carries the
// clearance cookie, or the served HTML itself (novelfull) — see
// novelfullSeriesURL for the latter case. The returned body is whatever the
// site's chapter list lives in, which is what latestChapterFrom's per-site
// switch expects.
func (f *BrowserFetcher) Get(ctx context.Context, seriesURL string) (string, int, error) {
apiURL, isKagane := kaganeAPIURL(seriesURL)
if !isKagane && !novelfullSeriesURL(seriesURL) {
return "", 0, fmt.Errorf("not a fetchable browser series url: %q", seriesURL)
}
var body string
// kagane's chapter list is only in its JSON API, which must be called from
// inside the page so the request carries the clearance cookie. novelfull
// renders its chapters into the HTML, so the cleared DOM is the answer.
// chromedp.OuterHTML returns a QueryAction and chromedp.Evaluate an
// EvaluateAction, so the variable has to be the interface both implement.
var read chromedp.Action = chromedp.OuterHTML("html", &body, chromedp.ByQuery)
if isKagane {
read = chromedp.Evaluate(
`fetch(`+jsString(apiURL)+`).then(r => r.ok ? r.text() : "")`,
&body,
awaitPromise,
)
}
// novelfull's payload is the DOM itself, and the interstitial has a DOM
// too, so "we have an answer" has to exclude it explicitly. kagane's
// in-page fetch just fails while challenged, which is already the signal.
done := func() bool { return body != "" && (isKagane || !isInterstitial(body)) }
if err := f.run(ctx, seriesURL, read, done); err != nil {
// Challenge never cleared, or the API refused. Indistinguishable from
// here and handled identically by the caller.
if errors.Is(err, errChallengeHeld) {
return "", 403, nil
}
return "", 0, fmt.Errorf("browser fetch %q: %w", seriesURL, err)
}
return body, 200, nil
}
// Image retrieves one cover's bytes through the browser sidecar, and its
// content type.
//
// It exists because kagane serves covers behind the same challenge as its
// pages *and* with `cross-origin-resource-policy: same-origin`, so the bytes
// are only reachable from inside a browser that already holds the clearance
// cookie (verified 2026-08-08). Acquisition through the sidecar is the only
// route.
//
// The image URL is navigated to rather than fetched from some other kagane
// page: the challenge only runs on a top-level navigation, and once it clears
// the document *is* the image, so a same-origin fetch of location.href reads
// it straight back out of the cache.
//
// The challenge is not solved by the first read: WaitReady("body") is satisfied
// by the interstitial too. run holds the tab open until the in-page fetch
// succeeds, which is what gives the challenge script the seconds it needs.
func (f *BrowserFetcher) Image(ctx context.Context, imageURL string) ([]byte, string, error) {
m := kaganeImageURLRe.FindStringSubmatch(imageURL)
if m == nil {
return nil, "", fmt.Errorf("not a browser-fetchable cover url: %q", imageURL)
}
imageID := m[1]
var dataURL string
err := f.run(ctx, imageURL,
chromedp.Evaluate(`fetch(location.href).then(r => r.ok
? r.blob().then(b => new Promise(res => {
const fr = new FileReader();
fr.onload = () => res(fr.result);
fr.readAsDataURL(b);
}))
: "")`, &dataURL, awaitPromise),
func() bool { return dataURL != "" })
if err != nil {
return nil, "", fmt.Errorf("browser image %s: %w", imageID, err)
}
// "data:image/webp;base64,<payload>".
head, payload, ok := strings.Cut(dataURL, ";base64,")
if !ok {
return nil, "", fmt.Errorf("browser image %s: not a data url", imageID)
}
raw, err := base64.StdEncoding.DecodeString(payload)
if err != nil {
return nil, "", fmt.Errorf("browser image %s: %w", imageID, err)
}
return raw, strings.TrimPrefix(head, "data:"), nil
}
// errChallengeHeld reports that the budget ran out with the interstitial still
// up. Distinct from a transport failure: it means "this site said no", which
// the poller answers with a 403 and its ordinary cooldown.
var errChallengeHeld = errors.New("challenge held")
// errBrowserInterrupted distinguishes a remote Chrome restart from the
// caller's own deadline. chromedp reports both as context.Canceled.
var errBrowserInterrupted = errors.New("browser interrupted")
func classifyBrowserError(ctx context.Context, browserLost bool, err error) error {
if err == nil || ctx.Err() != nil {
return err
}
if !browserLost {
return err
}
if !errors.Is(err, context.Canceled) {
return err
}
return fmt.Errorf("%w: %w", errBrowserInterrupted, err)
}
func browserConnectionLost(ctx context.Context) bool {
c := chromedp.FromContext(ctx)
if c == nil || c.Browser == nil {
return true
}
select {
case <-c.Browser.LostConnection:
return true
default:
return false
}
}
// challengePollInterval paces re-reads while a challenge solves itself.
const challengePollInterval = 2 * time.Second
// isInterstitial reports whether html is Cloudflare's challenge page rather
// than the site's own. Matched on the challenge runtime's script path, which is
// stable across the interstitial's wording and locale — the visible "Just a
// moment..." title is neither.
func isInterstitial(html string) bool {
return strings.Contains(html, "/cdn-cgi/challenge-platform/")
}
// run navigates to target and re-reads until done reports an answer, bounded by
// challengeTimeout and by the caller's own deadline, in a tab that is closed on
// return so one wedged page cannot poison later calls.
//
// Holding the tab open across re-reads is the whole point. A Cloudflare
// interstitial needs several seconds of a live page to solve itself and write
// clearance into the browser's shared cookie jar; reading once and closing the
// tab — which is what this did before 2026-08-08 — never gives it that window,
// so every fetch lands on the interstitial and the clearance that would have
// unblocked all the later ones is never obtained.
func (f *BrowserFetcher) run(ctx context.Context, target string, read chromedp.Action, done func() bool) error {
f.mu.Lock()
defer f.mu.Unlock()
callerCtx := ctx
ctx, cancel := context.WithTimeout(ctx, challengeTimeout)
defer cancel()
tabCtx, cancelTab := chromedp.NewContext(f.allocCtx)
defer cancelTab()
// Bind the caller's deadline to the tab.
tabCtx, cancelDeadline := context.WithCancel(tabCtx)
defer cancelDeadline()
go func() {
<-ctx.Done()
cancelDeadline()
}()
if err := chromedp.Run(tabCtx,
chromedp.Navigate(target),
chromedp.WaitReady("body", chromedp.ByQuery),
); err != nil {
return classifyBrowserError(callerCtx, browserConnectionLost(tabCtx), err)
}
var lastErr error
for {
// The challenge reloads the page when it passes, which tears down the
// execution context mid-read. That is a retry, not a failure.
if err := chromedp.Run(tabCtx, read); err != nil {
err = classifyBrowserError(callerCtx, browserConnectionLost(tabCtx), err)
if errors.Is(err, errBrowserInterrupted) {
return err
}
lastErr = err
} else if done() {
return nil
}
select {
case <-ctx.Done():
if err := callerCtx.Err(); err != nil {
return err
}
if lastErr != nil {
return fmt.Errorf("%w (last read: %v)", errChallengeHeld, lastErr)
}
return errChallengeHeld
case <-time.After(challengePollInterval):
}
}
}
// kaganeAPIURL maps a stored series_url to the JSON endpoint carrying its
// chapter list. Returning false for anything else is a second line of defence
// behind fetchableSeriesURL: a headless browser is a strong SSRF primitive and
// series_url is client-supplied, so the host is pinned here too.
func kaganeAPIURL(seriesURL string) (string, bool) {
u, err := url.Parse(seriesURL)
if err != nil || u.Scheme != "https" || u.Hostname() != "kagane.to" {
return "", false
}
m := kaganeSeriesRe.FindStringSubmatch(u.Path)
if m == nil {
return "", false
}
return "https://kagane.to/api/v2/series/" + m[1], true
}
// novelfullSeriesURL reports whether seriesURL is a novelfull series page this
// fetcher will open. novelfull's chapter list is in the served HTML, so unlike
// kagane there is no API to call from inside the page — the challenge-cleared
// DOM is the payload. The host is pinned here for the same reason kagane's is:
// series_url is client-supplied and a headless browser is a strong SSRF
// primitive.
func novelfullSeriesURL(seriesURL string) bool {
u, err := url.Parse(seriesURL)
return err == nil && u.Scheme == "https" && u.Hostname() == "novelfull.com" &&
strings.HasSuffix(u.Path, ".html")
}
// awaitPromise makes Evaluate resolve the promise rather than returning a
// serialised Promise object.
func awaitPromise(p *runtime.EvaluateParams) *runtime.EvaluateParams {
return p.WithAwaitPromise(true)
}
// jsString renders s as a JavaScript string literal for embedding in an
// Evaluate expression. The URL is host-pinned by kaganeAPIURL before it gets
// here, but quoting it properly is what keeps that guarantee intact.
func jsString(s string) string {
b, _ := json.Marshal(s)
return string(b)
}