rebrand: MangaBM → BookmarkManager, add novel library support #15
@@ -6,6 +6,7 @@ import (
|
||||
"fmt"
|
||||
"net/url"
|
||||
"regexp"
|
||||
"strings"
|
||||
"sync"
|
||||
"time"
|
||||
|
||||
@@ -24,18 +25,23 @@ var kaganeSeriesRe = regexp.MustCompile(`^/series/([0-9a-f-]{36})/?$`)
|
||||
// BrowserFetcher retrieves pages through a remote headless Chrome over the
|
||||
// DevTools Protocol.
|
||||
//
|
||||
// It exists for one reason: kagane.to sits behind a Cloudflare JavaScript
|
||||
// challenge. Verified 2026-08-03 from the deployment host, plain HTTP and
|
||||
// bogdanfinn/tls-client with a Chrome_133 profile both get 403 with
|
||||
// cf-mitigated: challenge on every path, including the API, robots.txt and
|
||||
// images. Clearing it requires executing the challenge script, which only a
|
||||
// real browser does.
|
||||
// It exists for one reason: kagane.to and novelfull.com sit behind a
|
||||
// Cloudflare JavaScript challenge. Verified 2026-08-03 (kagane) and 2026-08-05
|
||||
// (novelfull) from the deployment host, plain HTTP and bogdanfinn/tls-client
|
||||
// with a Chrome_133 profile both get 403 with cf-mitigated: challenge on every
|
||||
// path, including the API, robots.txt and images. Clearing it requires
|
||||
// executing the challenge script, which only a real browser does.
|
||||
//
|
||||
// The request is made *inside* the page rather than by extracting cf_clearance
|
||||
// and replaying it through TLSFetcher. That cookie is bound to IP, User-Agent
|
||||
// and often the TLS fingerprint, so replaying it means keeping three things in
|
||||
// sync that break silently and separately. The browser's own cookie jar
|
||||
// persists across polls, so the challenge is solved once every few hours.
|
||||
//
|
||||
// The two sites differ in how the chapter list is read: kagane serves it from
|
||||
// a JSON API that must be called from inside the page (so the request carries
|
||||
// the clearance cookie), while novelfull renders it into the HTML so the
|
||||
// cleared DOM is the payload.
|
||||
type BrowserFetcher struct {
|
||||
allocCtx context.Context
|
||||
cancel context.CancelFunc
|
||||
@@ -73,14 +79,16 @@ func (f *BrowserFetcher) Close() {
|
||||
f.cancel()
|
||||
}
|
||||
|
||||
// Get navigates to seriesURL, lets any challenge resolve, then reads the site's
|
||||
// JSON API from inside the page so the request carries the clearance cookie.
|
||||
// The returned body is API JSON, which is what latestChapterFrom's kagane case
|
||||
// expects — it is not HTML.
|
||||
// Get navigates to seriesURL, lets any challenge resolve, then reads either the
|
||||
// site's JSON API (kagane) from inside the page so the request carries the
|
||||
// clearance cookie, or the served HTML itself (novelfull) — see
|
||||
// novelfullSeriesURL for the latter case. The returned body is whatever the
|
||||
// site's chapter list lives in, which is what latestChapterFrom's per-site
|
||||
// switch expects.
|
||||
func (f *BrowserFetcher) Get(ctx context.Context, seriesURL string) (string, int, error) {
|
||||
apiURL, ok := kaganeAPIURL(seriesURL)
|
||||
if !ok {
|
||||
return "", 0, fmt.Errorf("not a fetchable kagane series url: %q", seriesURL)
|
||||
apiURL, isKagane := kaganeAPIURL(seriesURL)
|
||||
if !isKagane && !novelfullSeriesURL(seriesURL) {
|
||||
return "", 0, fmt.Errorf("not a fetchable browser series url: %q", seriesURL)
|
||||
}
|
||||
|
||||
f.mu.Lock()
|
||||
@@ -101,16 +109,26 @@ func (f *BrowserFetcher) Get(ctx context.Context, seriesURL string) (string, int
|
||||
}()
|
||||
|
||||
var body string
|
||||
// kagane's chapter list is only in its JSON API, which must be called from
|
||||
// inside the page so the request carries the clearance cookie. novelfull
|
||||
// renders its chapters into the HTML, so the cleared DOM is the answer.
|
||||
// chromedp.OuterHTML returns a QueryAction and chromedp.Evaluate an
|
||||
// EvaluateAction, so the variable has to be the interface both implement.
|
||||
var read chromedp.Action = chromedp.OuterHTML("html", &body, chromedp.ByQuery)
|
||||
if isKagane {
|
||||
read = chromedp.Evaluate(
|
||||
`fetch(`+jsString(apiURL)+`).then(r => r.ok ? r.text() : "")`,
|
||||
&body,
|
||||
awaitPromise,
|
||||
)
|
||||
}
|
||||
|
||||
err := chromedp.Run(tabCtx,
|
||||
chromedp.Navigate(seriesURL),
|
||||
// The challenge reloads the page itself when it passes; waiting for the
|
||||
// site's own root element is what tells us we are through it.
|
||||
chromedp.WaitReady("body", chromedp.ByQuery),
|
||||
chromedp.Evaluate(
|
||||
`fetch(`+jsString(apiURL)+`).then(r => r.ok ? r.text() : "")`,
|
||||
&body,
|
||||
awaitPromise,
|
||||
),
|
||||
read,
|
||||
)
|
||||
if err != nil {
|
||||
return "", 0, fmt.Errorf("browser fetch %q: %w", seriesURL, err)
|
||||
@@ -139,6 +157,18 @@ func kaganeAPIURL(seriesURL string) (string, bool) {
|
||||
return "https://kagane.to/api/v2/series/" + m[1], true
|
||||
}
|
||||
|
||||
// novelfullSeriesURL reports whether seriesURL is a novelfull series page this
|
||||
// fetcher will open. novelfull's chapter list is in the served HTML, so unlike
|
||||
// kagane there is no API to call from inside the page — the challenge-cleared
|
||||
// DOM is the payload. The host is pinned here for the same reason kagane's is:
|
||||
// series_url is client-supplied and a headless browser is a strong SSRF
|
||||
// primitive.
|
||||
func novelfullSeriesURL(seriesURL string) bool {
|
||||
u, err := url.Parse(seriesURL)
|
||||
return err == nil && u.Scheme == "https" && u.Hostname() == "novelfull.com" &&
|
||||
strings.HasSuffix(u.Path, ".html")
|
||||
}
|
||||
|
||||
// awaitPromise makes Evaluate resolve the promise rather than returning a
|
||||
// serialised Promise object.
|
||||
func awaitPromise(p *runtime.EvaluateParams) *runtime.EvaluateParams {
|
||||
|
||||
@@ -36,3 +36,24 @@ func TestKaganeAPIURL(t *testing.T) {
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
func TestNovelfullSeriesURL(t *testing.T) {
|
||||
cases := []struct {
|
||||
name string
|
||||
url string
|
||||
want bool
|
||||
}{
|
||||
{"series page", "https://novelfull.com/reverend-insanity.html", true},
|
||||
{"foreign host", "https://evil.example/reverend-insanity.html", false},
|
||||
{"not https", "http://novelfull.com/reverend-insanity.html", false},
|
||||
{"not a series page", "https://novelfull.com/genre/Fantasy", false},
|
||||
{"garbage", "://nope", false},
|
||||
}
|
||||
for _, tc := range cases {
|
||||
t.Run(tc.name, func(t *testing.T) {
|
||||
if got := novelfullSeriesURL(tc.url); got != tc.want {
|
||||
t.Fatalf("novelfullSeriesURL(%q) = %v, want %v", tc.url, got, tc.want)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
@@ -44,12 +44,13 @@ type Poller struct {
|
||||
}
|
||||
|
||||
// fetcherFor returns the fetcher a site needs, or nil when the site cannot be
|
||||
// fetched at all right now. kagane sits behind a Cloudflare JavaScript
|
||||
// challenge that no TLS fingerprint clears — verified 2026-08-03 from the
|
||||
// deployment host with the same Chrome profile TLSFetcher uses — so it is
|
||||
// browser-only or nothing.
|
||||
// fetched at all right now. kagane and novelfull both sit behind a Cloudflare
|
||||
// JavaScript challenge that no TLS fingerprint clears — kagane verified
|
||||
// 2026-08-03, novelfull verified 2026-08-05, both against the same Chrome_133
|
||||
// profile TLSFetcher uses — so they are browser-only or nothing.
|
||||
func (p *Poller) fetcherFor(site string) Fetcher {
|
||||
if site == "kagane" {
|
||||
switch site {
|
||||
case "kagane", "novelfull":
|
||||
return p.BrowserFetch
|
||||
}
|
||||
return p.Fetch
|
||||
@@ -212,13 +213,18 @@ func (p *Poller) checkOne(ctx context.Context, b store.Bookmark) {
|
||||
// a defence against the poller being used to probe arbitrary hosts from the
|
||||
// server's own network position, not just a check against wasted requests.
|
||||
//
|
||||
// kagane is held to a stricter rule: it is fetched by a headless browser, which
|
||||
// executes JavaScript and carries cookies, and is therefore a far stronger SSRF
|
||||
// primitive than an HTTP GET. Its host must match exactly, not merely be
|
||||
// non-empty.
|
||||
// Three sites are held to a stricter rule, each for a different reason:
|
||||
//
|
||||
// - kagane and novelfull are fetched by a headless browser, which executes
|
||||
// JavaScript and carries cookies, and is therefore a far stronger SSRF
|
||||
// primitive than an HTTP GET. Their hosts must match exactly, not merely
|
||||
// be non-empty.
|
||||
// - lightnovelworld's parser regex hardcodes its host, so a URL anywhere
|
||||
// else could never yield a match — reject it here rather than burn the
|
||||
// request.
|
||||
func fetchableSeriesURL(site, seriesURL string) bool {
|
||||
switch site {
|
||||
case "asura", "demonic", "comix", "kagane":
|
||||
case "asura", "demonic", "comix", "kagane", "novelfull", "lightnovelworld":
|
||||
default:
|
||||
return false
|
||||
}
|
||||
@@ -229,8 +235,17 @@ func fetchableSeriesURL(site, seriesURL string) bool {
|
||||
if u.Scheme != "https" || u.Host == "" {
|
||||
return false
|
||||
}
|
||||
if site == "kagane" {
|
||||
switch site {
|
||||
case "kagane":
|
||||
return u.Hostname() == "kagane.to"
|
||||
case "novelfull":
|
||||
// Fetched by a real browser, same as kagane, so the host is pinned
|
||||
// rather than merely non-empty.
|
||||
return u.Hostname() == "novelfull.com"
|
||||
case "lightnovelworld":
|
||||
// Its parser regex hardcodes this host, so a URL anywhere else could
|
||||
// never yield a match — reject it here rather than burn the request.
|
||||
return u.Hostname() == "lightnovelworld.net"
|
||||
}
|
||||
return true
|
||||
}
|
||||
|
||||
@@ -453,3 +453,49 @@ func TestKaganeUsesBrowserFetcher(t *testing.T) {
|
||||
t.Errorf("LatestChapterNum = %v, want 41", got.LatestChapterNum)
|
||||
}
|
||||
}
|
||||
|
||||
func TestFetcherForRoutesNovelSites(t *testing.T) {
|
||||
tls := &fakeFetcher{}
|
||||
browser := &fakeFetcher{}
|
||||
p := &Poller{Fetch: tls, BrowserFetch: browser}
|
||||
|
||||
cases := []struct {
|
||||
site string
|
||||
want Fetcher
|
||||
}{
|
||||
{"asura", tls},
|
||||
{"lightnovelworld", tls},
|
||||
{"kagane", browser},
|
||||
{"novelfull", browser},
|
||||
}
|
||||
for _, tc := range cases {
|
||||
t.Run(tc.site, func(t *testing.T) {
|
||||
if got := p.fetcherFor(tc.site); got != tc.want {
|
||||
t.Fatalf("fetcherFor(%q) = %v, want %v", tc.site, got, tc.want)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
func TestFetchableSeriesURLPinsNovelHosts(t *testing.T) {
|
||||
cases := []struct {
|
||||
name string
|
||||
site string
|
||||
url string
|
||||
want bool
|
||||
}{
|
||||
{"novelfull on its own host", "novelfull", "https://novelfull.com/reverend-insanity.html", true},
|
||||
{"novelfull on a foreign host", "novelfull", "https://evil.example/x.html", false},
|
||||
{"novelfull over http", "novelfull", "http://novelfull.com/x.html", false},
|
||||
{"lightnovelworld on its own host", "lightnovelworld", "https://lightnovelworld.net/novel/a-will-eternal/", true},
|
||||
{"lightnovelworld on a foreign host", "lightnovelworld", "https://evil.example/novel/x/", false},
|
||||
{"unknown site", "webnovel", "https://webnovel.com/x", false},
|
||||
}
|
||||
for _, tc := range cases {
|
||||
t.Run(tc.name, func(t *testing.T) {
|
||||
if got := fetchableSeriesURL(tc.site, tc.url); got != tc.want {
|
||||
t.Fatalf("fetchableSeriesURL(%q, %q) = %v, want %v", tc.site, tc.url, got, tc.want)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user