feat(latest): poll novelfull via headless browser, lightnovelworld via TLS

This commit is contained in:
2026-08-06 03:10:50 +07:00
parent c81e50c7f1
commit 92f1fbf6ec
4 changed files with 141 additions and 29 deletions
+48 -18
View File
@@ -6,6 +6,7 @@ import (
"fmt"
"net/url"
"regexp"
"strings"
"sync"
"time"
@@ -24,18 +25,23 @@ var kaganeSeriesRe = regexp.MustCompile(`^/series/([0-9a-f-]{36})/?$`)
// BrowserFetcher retrieves pages through a remote headless Chrome over the
// DevTools Protocol.
//
// It exists for one reason: kagane.to sits behind a Cloudflare JavaScript
// challenge. Verified 2026-08-03 from the deployment host, plain HTTP and
// bogdanfinn/tls-client with a Chrome_133 profile both get 403 with
// cf-mitigated: challenge on every path, including the API, robots.txt and
// images. Clearing it requires executing the challenge script, which only a
// real browser does.
// It exists for one reason: kagane.to and novelfull.com sit behind a
// Cloudflare JavaScript challenge. Verified 2026-08-03 (kagane) and 2026-08-05
// (novelfull) from the deployment host, plain HTTP and bogdanfinn/tls-client
// with a Chrome_133 profile both get 403 with cf-mitigated: challenge on every
// path, including the API, robots.txt and images. Clearing it requires
// executing the challenge script, which only a real browser does.
//
// The request is made *inside* the page rather than by extracting cf_clearance
// and replaying it through TLSFetcher. That cookie is bound to IP, User-Agent
// and often the TLS fingerprint, so replaying it means keeping three things in
// sync that break silently and separately. The browser's own cookie jar
// persists across polls, so the challenge is solved once every few hours.
//
// The two sites differ in how the chapter list is read: kagane serves it from
// a JSON API that must be called from inside the page (so the request carries
// the clearance cookie), while novelfull renders it into the HTML so the
// cleared DOM is the payload.
type BrowserFetcher struct {
allocCtx context.Context
cancel context.CancelFunc
@@ -73,14 +79,16 @@ func (f *BrowserFetcher) Close() {
f.cancel()
}
// Get navigates to seriesURL, lets any challenge resolve, then reads the site's
// JSON API from inside the page so the request carries the clearance cookie.
// The returned body is API JSON, which is what latestChapterFrom's kagane case
// expects — it is not HTML.
// Get navigates to seriesURL, lets any challenge resolve, then reads either the
// site's JSON API (kagane) from inside the page so the request carries the
// clearance cookie, or the served HTML itself (novelfull) — see
// novelfullSeriesURL for the latter case. The returned body is whatever the
// site's chapter list lives in, which is what latestChapterFrom's per-site
// switch expects.
func (f *BrowserFetcher) Get(ctx context.Context, seriesURL string) (string, int, error) {
apiURL, ok := kaganeAPIURL(seriesURL)
if !ok {
return "", 0, fmt.Errorf("not a fetchable kagane series url: %q", seriesURL)
apiURL, isKagane := kaganeAPIURL(seriesURL)
if !isKagane && !novelfullSeriesURL(seriesURL) {
return "", 0, fmt.Errorf("not a fetchable browser series url: %q", seriesURL)
}
f.mu.Lock()
@@ -101,16 +109,26 @@ func (f *BrowserFetcher) Get(ctx context.Context, seriesURL string) (string, int
}()
var body string
// kagane's chapter list is only in its JSON API, which must be called from
// inside the page so the request carries the clearance cookie. novelfull
// renders its chapters into the HTML, so the cleared DOM is the answer.
// chromedp.OuterHTML returns a QueryAction and chromedp.Evaluate an
// EvaluateAction, so the variable has to be the interface both implement.
var read chromedp.Action = chromedp.OuterHTML("html", &body, chromedp.ByQuery)
if isKagane {
read = chromedp.Evaluate(
`fetch(`+jsString(apiURL)+`).then(r => r.ok ? r.text() : "")`,
&body,
awaitPromise,
)
}
err := chromedp.Run(tabCtx,
chromedp.Navigate(seriesURL),
// The challenge reloads the page itself when it passes; waiting for the
// site's own root element is what tells us we are through it.
chromedp.WaitReady("body", chromedp.ByQuery),
chromedp.Evaluate(
`fetch(`+jsString(apiURL)+`).then(r => r.ok ? r.text() : "")`,
&body,
awaitPromise,
),
read,
)
if err != nil {
return "", 0, fmt.Errorf("browser fetch %q: %w", seriesURL, err)
@@ -139,6 +157,18 @@ func kaganeAPIURL(seriesURL string) (string, bool) {
return "https://kagane.to/api/v2/series/" + m[1], true
}
// novelfullSeriesURL reports whether seriesURL is a novelfull series page this
// fetcher will open. novelfull's chapter list is in the served HTML, so unlike
// kagane there is no API to call from inside the page — the challenge-cleared
// DOM is the payload. The host is pinned here for the same reason kagane's is:
// series_url is client-supplied and a headless browser is a strong SSRF
// primitive.
func novelfullSeriesURL(seriesURL string) bool {
u, err := url.Parse(seriesURL)
return err == nil && u.Scheme == "https" && u.Hostname() == "novelfull.com" &&
strings.HasSuffix(u.Path, ".html")
}
// awaitPromise makes Evaluate resolve the promise rather than returning a
// serialised Promise object.
func awaitPromise(p *runtime.EvaluateParams) *runtime.EvaluateParams {