From 92f1fbf6ec670400275a9b3a6bc6906d719f3980 Mon Sep 17 00:00:00 2001 From: Sulthan Zaki Date: Thu, 6 Aug 2026 03:10:50 +0700 Subject: [PATCH] feat(latest): poll novelfull via headless browser, lightnovelworld via TLS --- backend/internal/latest/browser.go | 66 ++++++++++++++++++------- backend/internal/latest/browser_test.go | 21 ++++++++ backend/internal/latest/poller.go | 37 +++++++++----- backend/internal/latest/poller_test.go | 46 +++++++++++++++++ 4 files changed, 141 insertions(+), 29 deletions(-) diff --git a/backend/internal/latest/browser.go b/backend/internal/latest/browser.go index 10bcf21..fb856af 100644 --- a/backend/internal/latest/browser.go +++ b/backend/internal/latest/browser.go @@ -6,6 +6,7 @@ import ( "fmt" "net/url" "regexp" + "strings" "sync" "time" @@ -24,18 +25,23 @@ var kaganeSeriesRe = regexp.MustCompile(`^/series/([0-9a-f-]{36})/?$`) // BrowserFetcher retrieves pages through a remote headless Chrome over the // DevTools Protocol. // -// It exists for one reason: kagane.to sits behind a Cloudflare JavaScript -// challenge. Verified 2026-08-03 from the deployment host, plain HTTP and -// bogdanfinn/tls-client with a Chrome_133 profile both get 403 with -// cf-mitigated: challenge on every path, including the API, robots.txt and -// images. Clearing it requires executing the challenge script, which only a -// real browser does. +// It exists for one reason: kagane.to and novelfull.com sit behind a +// Cloudflare JavaScript challenge. Verified 2026-08-03 (kagane) and 2026-08-05 +// (novelfull) from the deployment host, plain HTTP and bogdanfinn/tls-client +// with a Chrome_133 profile both get 403 with cf-mitigated: challenge on every +// path, including the API, robots.txt and images. Clearing it requires +// executing the challenge script, which only a real browser does. // // The request is made *inside* the page rather than by extracting cf_clearance // and replaying it through TLSFetcher. That cookie is bound to IP, User-Agent // and often the TLS fingerprint, so replaying it means keeping three things in // sync that break silently and separately. The browser's own cookie jar // persists across polls, so the challenge is solved once every few hours. +// +// The two sites differ in how the chapter list is read: kagane serves it from +// a JSON API that must be called from inside the page (so the request carries +// the clearance cookie), while novelfull renders it into the HTML so the +// cleared DOM is the payload. type BrowserFetcher struct { allocCtx context.Context cancel context.CancelFunc @@ -73,14 +79,16 @@ func (f *BrowserFetcher) Close() { f.cancel() } -// Get navigates to seriesURL, lets any challenge resolve, then reads the site's -// JSON API from inside the page so the request carries the clearance cookie. -// The returned body is API JSON, which is what latestChapterFrom's kagane case -// expects — it is not HTML. +// Get navigates to seriesURL, lets any challenge resolve, then reads either the +// site's JSON API (kagane) from inside the page so the request carries the +// clearance cookie, or the served HTML itself (novelfull) — see +// novelfullSeriesURL for the latter case. The returned body is whatever the +// site's chapter list lives in, which is what latestChapterFrom's per-site +// switch expects. func (f *BrowserFetcher) Get(ctx context.Context, seriesURL string) (string, int, error) { - apiURL, ok := kaganeAPIURL(seriesURL) - if !ok { - return "", 0, fmt.Errorf("not a fetchable kagane series url: %q", seriesURL) + apiURL, isKagane := kaganeAPIURL(seriesURL) + if !isKagane && !novelfullSeriesURL(seriesURL) { + return "", 0, fmt.Errorf("not a fetchable browser series url: %q", seriesURL) } f.mu.Lock() @@ -101,16 +109,26 @@ func (f *BrowserFetcher) Get(ctx context.Context, seriesURL string) (string, int }() var body string + // kagane's chapter list is only in its JSON API, which must be called from + // inside the page so the request carries the clearance cookie. novelfull + // renders its chapters into the HTML, so the cleared DOM is the answer. + // chromedp.OuterHTML returns a QueryAction and chromedp.Evaluate an + // EvaluateAction, so the variable has to be the interface both implement. + var read chromedp.Action = chromedp.OuterHTML("html", &body, chromedp.ByQuery) + if isKagane { + read = chromedp.Evaluate( + `fetch(`+jsString(apiURL)+`).then(r => r.ok ? r.text() : "")`, + &body, + awaitPromise, + ) + } + err := chromedp.Run(tabCtx, chromedp.Navigate(seriesURL), // The challenge reloads the page itself when it passes; waiting for the // site's own root element is what tells us we are through it. chromedp.WaitReady("body", chromedp.ByQuery), - chromedp.Evaluate( - `fetch(`+jsString(apiURL)+`).then(r => r.ok ? r.text() : "")`, - &body, - awaitPromise, - ), + read, ) if err != nil { return "", 0, fmt.Errorf("browser fetch %q: %w", seriesURL, err) @@ -139,6 +157,18 @@ func kaganeAPIURL(seriesURL string) (string, bool) { return "https://kagane.to/api/v2/series/" + m[1], true } +// novelfullSeriesURL reports whether seriesURL is a novelfull series page this +// fetcher will open. novelfull's chapter list is in the served HTML, so unlike +// kagane there is no API to call from inside the page — the challenge-cleared +// DOM is the payload. The host is pinned here for the same reason kagane's is: +// series_url is client-supplied and a headless browser is a strong SSRF +// primitive. +func novelfullSeriesURL(seriesURL string) bool { + u, err := url.Parse(seriesURL) + return err == nil && u.Scheme == "https" && u.Hostname() == "novelfull.com" && + strings.HasSuffix(u.Path, ".html") +} + // awaitPromise makes Evaluate resolve the promise rather than returning a // serialised Promise object. func awaitPromise(p *runtime.EvaluateParams) *runtime.EvaluateParams { diff --git a/backend/internal/latest/browser_test.go b/backend/internal/latest/browser_test.go index fa91a3a..c94c1f4 100644 --- a/backend/internal/latest/browser_test.go +++ b/backend/internal/latest/browser_test.go @@ -36,3 +36,24 @@ func TestKaganeAPIURL(t *testing.T) { }) } } + +func TestNovelfullSeriesURL(t *testing.T) { + cases := []struct { + name string + url string + want bool + }{ + {"series page", "https://novelfull.com/reverend-insanity.html", true}, + {"foreign host", "https://evil.example/reverend-insanity.html", false}, + {"not https", "http://novelfull.com/reverend-insanity.html", false}, + {"not a series page", "https://novelfull.com/genre/Fantasy", false}, + {"garbage", "://nope", false}, + } + for _, tc := range cases { + t.Run(tc.name, func(t *testing.T) { + if got := novelfullSeriesURL(tc.url); got != tc.want { + t.Fatalf("novelfullSeriesURL(%q) = %v, want %v", tc.url, got, tc.want) + } + }) + } +} diff --git a/backend/internal/latest/poller.go b/backend/internal/latest/poller.go index 2693c43..57e764f 100644 --- a/backend/internal/latest/poller.go +++ b/backend/internal/latest/poller.go @@ -44,12 +44,13 @@ type Poller struct { } // fetcherFor returns the fetcher a site needs, or nil when the site cannot be -// fetched at all right now. kagane sits behind a Cloudflare JavaScript -// challenge that no TLS fingerprint clears — verified 2026-08-03 from the -// deployment host with the same Chrome profile TLSFetcher uses — so it is -// browser-only or nothing. +// fetched at all right now. kagane and novelfull both sit behind a Cloudflare +// JavaScript challenge that no TLS fingerprint clears — kagane verified +// 2026-08-03, novelfull verified 2026-08-05, both against the same Chrome_133 +// profile TLSFetcher uses — so they are browser-only or nothing. func (p *Poller) fetcherFor(site string) Fetcher { - if site == "kagane" { + switch site { + case "kagane", "novelfull": return p.BrowserFetch } return p.Fetch @@ -212,13 +213,18 @@ func (p *Poller) checkOne(ctx context.Context, b store.Bookmark) { // a defence against the poller being used to probe arbitrary hosts from the // server's own network position, not just a check against wasted requests. // -// kagane is held to a stricter rule: it is fetched by a headless browser, which -// executes JavaScript and carries cookies, and is therefore a far stronger SSRF -// primitive than an HTTP GET. Its host must match exactly, not merely be -// non-empty. +// Three sites are held to a stricter rule, each for a different reason: +// +// - kagane and novelfull are fetched by a headless browser, which executes +// JavaScript and carries cookies, and is therefore a far stronger SSRF +// primitive than an HTTP GET. Their hosts must match exactly, not merely +// be non-empty. +// - lightnovelworld's parser regex hardcodes its host, so a URL anywhere +// else could never yield a match — reject it here rather than burn the +// request. func fetchableSeriesURL(site, seriesURL string) bool { switch site { - case "asura", "demonic", "comix", "kagane": + case "asura", "demonic", "comix", "kagane", "novelfull", "lightnovelworld": default: return false } @@ -229,8 +235,17 @@ func fetchableSeriesURL(site, seriesURL string) bool { if u.Scheme != "https" || u.Host == "" { return false } - if site == "kagane" { + switch site { + case "kagane": return u.Hostname() == "kagane.to" + case "novelfull": + // Fetched by a real browser, same as kagane, so the host is pinned + // rather than merely non-empty. + return u.Hostname() == "novelfull.com" + case "lightnovelworld": + // Its parser regex hardcodes this host, so a URL anywhere else could + // never yield a match — reject it here rather than burn the request. + return u.Hostname() == "lightnovelworld.net" } return true } diff --git a/backend/internal/latest/poller_test.go b/backend/internal/latest/poller_test.go index 30754cb..fdb9985 100644 --- a/backend/internal/latest/poller_test.go +++ b/backend/internal/latest/poller_test.go @@ -453,3 +453,49 @@ func TestKaganeUsesBrowserFetcher(t *testing.T) { t.Errorf("LatestChapterNum = %v, want 41", got.LatestChapterNum) } } + +func TestFetcherForRoutesNovelSites(t *testing.T) { + tls := &fakeFetcher{} + browser := &fakeFetcher{} + p := &Poller{Fetch: tls, BrowserFetch: browser} + + cases := []struct { + site string + want Fetcher + }{ + {"asura", tls}, + {"lightnovelworld", tls}, + {"kagane", browser}, + {"novelfull", browser}, + } + for _, tc := range cases { + t.Run(tc.site, func(t *testing.T) { + if got := p.fetcherFor(tc.site); got != tc.want { + t.Fatalf("fetcherFor(%q) = %v, want %v", tc.site, got, tc.want) + } + }) + } +} + +func TestFetchableSeriesURLPinsNovelHosts(t *testing.T) { + cases := []struct { + name string + site string + url string + want bool + }{ + {"novelfull on its own host", "novelfull", "https://novelfull.com/reverend-insanity.html", true}, + {"novelfull on a foreign host", "novelfull", "https://evil.example/x.html", false}, + {"novelfull over http", "novelfull", "http://novelfull.com/x.html", false}, + {"lightnovelworld on its own host", "lightnovelworld", "https://lightnovelworld.net/novel/a-will-eternal/", true}, + {"lightnovelworld on a foreign host", "lightnovelworld", "https://evil.example/novel/x/", false}, + {"unknown site", "webnovel", "https://webnovel.com/x", false}, + } + for _, tc := range cases { + t.Run(tc.name, func(t *testing.T) { + if got := fetchableSeriesURL(tc.site, tc.url); got != tc.want { + t.Fatalf("fetchableSeriesURL(%q, %q) = %v, want %v", tc.site, tc.url, got, tc.want) + } + }) + } +}