feat(latest): poll novelfull via headless browser, lightnovelworld via TLS

This commit is contained in:
2026-08-06 03:10:50 +07:00
parent c81e50c7f1
commit 92f1fbf6ec
4 changed files with 141 additions and 29 deletions
+48 -18
View File
@@ -6,6 +6,7 @@ import (
"fmt" "fmt"
"net/url" "net/url"
"regexp" "regexp"
"strings"
"sync" "sync"
"time" "time"
@@ -24,18 +25,23 @@ var kaganeSeriesRe = regexp.MustCompile(`^/series/([0-9a-f-]{36})/?$`)
// BrowserFetcher retrieves pages through a remote headless Chrome over the // BrowserFetcher retrieves pages through a remote headless Chrome over the
// DevTools Protocol. // DevTools Protocol.
// //
// It exists for one reason: kagane.to sits behind a Cloudflare JavaScript // It exists for one reason: kagane.to and novelfull.com sit behind a
// challenge. Verified 2026-08-03 from the deployment host, plain HTTP and // Cloudflare JavaScript challenge. Verified 2026-08-03 (kagane) and 2026-08-05
// bogdanfinn/tls-client with a Chrome_133 profile both get 403 with // (novelfull) from the deployment host, plain HTTP and bogdanfinn/tls-client
// cf-mitigated: challenge on every path, including the API, robots.txt and // with a Chrome_133 profile both get 403 with cf-mitigated: challenge on every
// images. Clearing it requires executing the challenge script, which only a // path, including the API, robots.txt and images. Clearing it requires
// real browser does. // executing the challenge script, which only a real browser does.
// //
// The request is made *inside* the page rather than by extracting cf_clearance // The request is made *inside* the page rather than by extracting cf_clearance
// and replaying it through TLSFetcher. That cookie is bound to IP, User-Agent // and replaying it through TLSFetcher. That cookie is bound to IP, User-Agent
// and often the TLS fingerprint, so replaying it means keeping three things in // and often the TLS fingerprint, so replaying it means keeping three things in
// sync that break silently and separately. The browser's own cookie jar // sync that break silently and separately. The browser's own cookie jar
// persists across polls, so the challenge is solved once every few hours. // persists across polls, so the challenge is solved once every few hours.
//
// The two sites differ in how the chapter list is read: kagane serves it from
// a JSON API that must be called from inside the page (so the request carries
// the clearance cookie), while novelfull renders it into the HTML so the
// cleared DOM is the payload.
type BrowserFetcher struct { type BrowserFetcher struct {
allocCtx context.Context allocCtx context.Context
cancel context.CancelFunc cancel context.CancelFunc
@@ -73,14 +79,16 @@ func (f *BrowserFetcher) Close() {
f.cancel() f.cancel()
} }
// Get navigates to seriesURL, lets any challenge resolve, then reads the site's // Get navigates to seriesURL, lets any challenge resolve, then reads either the
// JSON API from inside the page so the request carries the clearance cookie. // site's JSON API (kagane) from inside the page so the request carries the
// The returned body is API JSON, which is what latestChapterFrom's kagane case // clearance cookie, or the served HTML itself (novelfull) — see
// expects — it is not HTML. // novelfullSeriesURL for the latter case. The returned body is whatever the
// site's chapter list lives in, which is what latestChapterFrom's per-site
// switch expects.
func (f *BrowserFetcher) Get(ctx context.Context, seriesURL string) (string, int, error) { func (f *BrowserFetcher) Get(ctx context.Context, seriesURL string) (string, int, error) {
apiURL, ok := kaganeAPIURL(seriesURL) apiURL, isKagane := kaganeAPIURL(seriesURL)
if !ok { if !isKagane && !novelfullSeriesURL(seriesURL) {
return "", 0, fmt.Errorf("not a fetchable kagane series url: %q", seriesURL) return "", 0, fmt.Errorf("not a fetchable browser series url: %q", seriesURL)
} }
f.mu.Lock() f.mu.Lock()
@@ -101,16 +109,26 @@ func (f *BrowserFetcher) Get(ctx context.Context, seriesURL string) (string, int
}() }()
var body string var body string
// kagane's chapter list is only in its JSON API, which must be called from
// inside the page so the request carries the clearance cookie. novelfull
// renders its chapters into the HTML, so the cleared DOM is the answer.
// chromedp.OuterHTML returns a QueryAction and chromedp.Evaluate an
// EvaluateAction, so the variable has to be the interface both implement.
var read chromedp.Action = chromedp.OuterHTML("html", &body, chromedp.ByQuery)
if isKagane {
read = chromedp.Evaluate(
`fetch(`+jsString(apiURL)+`).then(r => r.ok ? r.text() : "")`,
&body,
awaitPromise,
)
}
err := chromedp.Run(tabCtx, err := chromedp.Run(tabCtx,
chromedp.Navigate(seriesURL), chromedp.Navigate(seriesURL),
// The challenge reloads the page itself when it passes; waiting for the // The challenge reloads the page itself when it passes; waiting for the
// site's own root element is what tells us we are through it. // site's own root element is what tells us we are through it.
chromedp.WaitReady("body", chromedp.ByQuery), chromedp.WaitReady("body", chromedp.ByQuery),
chromedp.Evaluate( read,
`fetch(`+jsString(apiURL)+`).then(r => r.ok ? r.text() : "")`,
&body,
awaitPromise,
),
) )
if err != nil { if err != nil {
return "", 0, fmt.Errorf("browser fetch %q: %w", seriesURL, err) return "", 0, fmt.Errorf("browser fetch %q: %w", seriesURL, err)
@@ -139,6 +157,18 @@ func kaganeAPIURL(seriesURL string) (string, bool) {
return "https://kagane.to/api/v2/series/" + m[1], true return "https://kagane.to/api/v2/series/" + m[1], true
} }
// novelfullSeriesURL reports whether seriesURL is a novelfull series page this
// fetcher will open. novelfull's chapter list is in the served HTML, so unlike
// kagane there is no API to call from inside the page — the challenge-cleared
// DOM is the payload. The host is pinned here for the same reason kagane's is:
// series_url is client-supplied and a headless browser is a strong SSRF
// primitive.
func novelfullSeriesURL(seriesURL string) bool {
u, err := url.Parse(seriesURL)
return err == nil && u.Scheme == "https" && u.Hostname() == "novelfull.com" &&
strings.HasSuffix(u.Path, ".html")
}
// awaitPromise makes Evaluate resolve the promise rather than returning a // awaitPromise makes Evaluate resolve the promise rather than returning a
// serialised Promise object. // serialised Promise object.
func awaitPromise(p *runtime.EvaluateParams) *runtime.EvaluateParams { func awaitPromise(p *runtime.EvaluateParams) *runtime.EvaluateParams {
+21
View File
@@ -36,3 +36,24 @@ func TestKaganeAPIURL(t *testing.T) {
}) })
} }
} }
func TestNovelfullSeriesURL(t *testing.T) {
cases := []struct {
name string
url string
want bool
}{
{"series page", "https://novelfull.com/reverend-insanity.html", true},
{"foreign host", "https://evil.example/reverend-insanity.html", false},
{"not https", "http://novelfull.com/reverend-insanity.html", false},
{"not a series page", "https://novelfull.com/genre/Fantasy", false},
{"garbage", "://nope", false},
}
for _, tc := range cases {
t.Run(tc.name, func(t *testing.T) {
if got := novelfullSeriesURL(tc.url); got != tc.want {
t.Fatalf("novelfullSeriesURL(%q) = %v, want %v", tc.url, got, tc.want)
}
})
}
}
+26 -11
View File
@@ -44,12 +44,13 @@ type Poller struct {
} }
// fetcherFor returns the fetcher a site needs, or nil when the site cannot be // fetcherFor returns the fetcher a site needs, or nil when the site cannot be
// fetched at all right now. kagane sits behind a Cloudflare JavaScript // fetched at all right now. kagane and novelfull both sit behind a Cloudflare
// challenge that no TLS fingerprint clears — verified 2026-08-03 from the // JavaScript challenge that no TLS fingerprint clears — kagane verified
// deployment host with the same Chrome profile TLSFetcher uses — so it is // 2026-08-03, novelfull verified 2026-08-05, both against the same Chrome_133
// browser-only or nothing. // profile TLSFetcher uses — so they are browser-only or nothing.
func (p *Poller) fetcherFor(site string) Fetcher { func (p *Poller) fetcherFor(site string) Fetcher {
if site == "kagane" { switch site {
case "kagane", "novelfull":
return p.BrowserFetch return p.BrowserFetch
} }
return p.Fetch return p.Fetch
@@ -212,13 +213,18 @@ func (p *Poller) checkOne(ctx context.Context, b store.Bookmark) {
// a defence against the poller being used to probe arbitrary hosts from the // a defence against the poller being used to probe arbitrary hosts from the
// server's own network position, not just a check against wasted requests. // server's own network position, not just a check against wasted requests.
// //
// kagane is held to a stricter rule: it is fetched by a headless browser, which // Three sites are held to a stricter rule, each for a different reason:
// executes JavaScript and carries cookies, and is therefore a far stronger SSRF //
// primitive than an HTTP GET. Its host must match exactly, not merely be // - kagane and novelfull are fetched by a headless browser, which executes
// non-empty. // JavaScript and carries cookies, and is therefore a far stronger SSRF
// primitive than an HTTP GET. Their hosts must match exactly, not merely
// be non-empty.
// - lightnovelworld's parser regex hardcodes its host, so a URL anywhere
// else could never yield a match — reject it here rather than burn the
// request.
func fetchableSeriesURL(site, seriesURL string) bool { func fetchableSeriesURL(site, seriesURL string) bool {
switch site { switch site {
case "asura", "demonic", "comix", "kagane": case "asura", "demonic", "comix", "kagane", "novelfull", "lightnovelworld":
default: default:
return false return false
} }
@@ -229,8 +235,17 @@ func fetchableSeriesURL(site, seriesURL string) bool {
if u.Scheme != "https" || u.Host == "" { if u.Scheme != "https" || u.Host == "" {
return false return false
} }
if site == "kagane" { switch site {
case "kagane":
return u.Hostname() == "kagane.to" return u.Hostname() == "kagane.to"
case "novelfull":
// Fetched by a real browser, same as kagane, so the host is pinned
// rather than merely non-empty.
return u.Hostname() == "novelfull.com"
case "lightnovelworld":
// Its parser regex hardcodes this host, so a URL anywhere else could
// never yield a match — reject it here rather than burn the request.
return u.Hostname() == "lightnovelworld.net"
} }
return true return true
} }
+46
View File
@@ -453,3 +453,49 @@ func TestKaganeUsesBrowserFetcher(t *testing.T) {
t.Errorf("LatestChapterNum = %v, want 41", got.LatestChapterNum) t.Errorf("LatestChapterNum = %v, want 41", got.LatestChapterNum)
} }
} }
func TestFetcherForRoutesNovelSites(t *testing.T) {
tls := &fakeFetcher{}
browser := &fakeFetcher{}
p := &Poller{Fetch: tls, BrowserFetch: browser}
cases := []struct {
site string
want Fetcher
}{
{"asura", tls},
{"lightnovelworld", tls},
{"kagane", browser},
{"novelfull", browser},
}
for _, tc := range cases {
t.Run(tc.site, func(t *testing.T) {
if got := p.fetcherFor(tc.site); got != tc.want {
t.Fatalf("fetcherFor(%q) = %v, want %v", tc.site, got, tc.want)
}
})
}
}
func TestFetchableSeriesURLPinsNovelHosts(t *testing.T) {
cases := []struct {
name string
site string
url string
want bool
}{
{"novelfull on its own host", "novelfull", "https://novelfull.com/reverend-insanity.html", true},
{"novelfull on a foreign host", "novelfull", "https://evil.example/x.html", false},
{"novelfull over http", "novelfull", "http://novelfull.com/x.html", false},
{"lightnovelworld on its own host", "lightnovelworld", "https://lightnovelworld.net/novel/a-will-eternal/", true},
{"lightnovelworld on a foreign host", "lightnovelworld", "https://evil.example/novel/x/", false},
{"unknown site", "webnovel", "https://webnovel.com/x", false},
}
for _, tc := range cases {
t.Run(tc.name, func(t *testing.T) {
if got := fetchableSeriesURL(tc.site, tc.url); got != tc.want {
t.Fatalf("fetchableSeriesURL(%q, %q) = %v, want %v", tc.site, tc.url, got, tc.want)
}
})
}
}