Split backend into internal packages by responsibility

All Go files lived flat in backend/ as one package main. Move store,
latest-chapter polling, sessions, HTTP middleware, the JSON API, the
userscript handler, and the web UI (with its templates/static assets)
into backend/internal/{store,latest,session,httpmw,api,userscript,web},
each with an exported API. main.go becomes the composition root wiring
them into newRouter; root-level tests cover the assembled router while
package-local tests cover unit behavior. Update Dockerfile/.dockerignore
for the new internal/ tree and CLAUDE.md to describe the layout.

Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
This commit is contained in:
2026-08-02 19:41:38 +07:00
parent e250762ea6
commit eeb601cbe2
36 changed files with 971 additions and 843 deletions
+76
View File
@@ -0,0 +1,76 @@
package latest
import (
"context"
"fmt"
"io"
fhttp "github.com/bogdanfinn/fhttp"
tls_client "github.com/bogdanfinn/tls-client"
"github.com/bogdanfinn/tls-client/profiles"
)
// maxBodyBytes caps what a single series page can cost in memory. Real pages
// measured 100-400 KB on 2026-07-26, so this is roughly 10x headroom and mostly
// guards against a proxy handing back something enormous.
const maxBodyBytes = 4 << 20
// chromeUA matches the client profile below. A Chrome fingerprint paired with a
// non-Chrome user agent is itself a signal.
const chromeUA = "Mozilla/5.0 (Linux; Android 10; K) AppleWebKit/537.36 " +
"(KHTML, like Gecko) Chrome/133.0.0.0 Mobile Safari/537.36"
// TLSFetcher fetches series pages with a Chrome TLS fingerprint.
//
// Plain net/http was verified working against both sites on 2026-07-26, so this
// is not fixing an observed block — it is deliberate defence-in-depth against a
// future fingerprint-based one, chosen up front rather than reacted to later.
// The library is pure Go, so CGO_ENABLED=0, the static binary, and the
// distroless image are all unaffected.
type TLSFetcher struct {
client tls_client.HttpClient
}
var _ Fetcher = (*TLSFetcher)(nil)
func NewTLSFetcher() (*TLSFetcher, error) {
c, err := tls_client.NewHttpClient(tls_client.NewNoopLogger(),
tls_client.WithTimeoutSeconds(30),
tls_client.WithClientProfile(profiles.Chrome_133),
)
if err != nil {
return nil, fmt.Errorf("new tls client: %w", err)
}
return &TLSFetcher{client: c}, nil
}
// Get fetches url and returns the body and status. Redirects are followed: the
// demonic chapter anchors are a redirect form, and asura has moved domains
// before.
func (f *TLSFetcher) Get(ctx context.Context, url string) (string, int, error) {
req, err := fhttp.NewRequest(fhttp.MethodGet, url, nil)
if err != nil {
return "", 0, fmt.Errorf("build request %q: %w", url, err)
}
req = req.WithContext(ctx)
// Header order is part of what is being fingerprinted, so it is stated
// explicitly instead of left to Go's map iteration order.
req.Header = fhttp.Header{
"user-agent": {chromeUA},
"accept": {"text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8"},
"accept-language": {"en-US,en;q=0.9"},
fhttp.HeaderOrderKey: {"user-agent", "accept", "accept-language"},
}
resp, err := f.client.Do(req)
if err != nil {
return "", 0, fmt.Errorf("get %q: %w", url, err)
}
defer resp.Body.Close()
body, err := io.ReadAll(io.LimitReader(resp.Body, maxBodyBytes))
if err != nil {
return "", resp.StatusCode, fmt.Errorf("read %q: %w", url, err)
}
return string(body), resp.StatusCode, nil
}
+203
View File
@@ -0,0 +1,203 @@
package latest
import (
"context"
"log"
"net/url"
"time"
"mangabm/backend/internal/store"
)
// Fetcher retrieves a series page. It exists as an interface so tests can inject
// a fake: nothing in the test suite may touch the network or the TLS client.
type Fetcher interface {
Get(ctx context.Context, url string) (body string, status int, err error)
}
// Poller re-checks each bookmarked series' newest published chapter on a
// schedule, independent of the userscript's own in-browser checks. The two run
// in parallel and report the same observable fact, so whichever writes last wins
// and neither needs to know about the other.
//
// Two clocks, deliberately independent:
//
// - Interval is how often this goroutine wakes up and looks.
// - Cooldown is how long one bookmark rests since its own last check.
//
// Only the cooldown is per bookmark, and it is enforced by the WHERE clause in
// DueForLatestCheck rather than by any timer. Shortening Interval therefore
// cannot shorten anyone's cooldown; it only makes the poller wake up and find
// nothing due more often.
type Poller struct {
Store *store.Store
Fetch Fetcher
Now func() time.Time // injected so tests can freeze it
Cooldown time.Duration
Interval time.Duration
Stagger time.Duration
Batch int
}
// Run polls until ctx is cancelled.
//
// runOnce is called synchronously, so a batch that overruns the tick delays the
// next one instead of stacking a second batch on top of it. That is the intended
// failure mode for a misconfigured batch x stagger: a slower cadence, never
// concurrent fetch storms.
func (p *Poller) Run(ctx context.Context) {
log.Printf("latest-chapter poller: interval=%s cooldown=%s batch=%d stagger=%s",
p.Interval, p.Cooldown, p.Batch, p.Stagger)
t := time.NewTicker(p.Interval)
defer t.Stop()
for {
select {
case <-ctx.Done():
log.Println("latest-chapter poller: stopped")
return
case <-t.C:
p.runOnce(ctx)
}
}
}
// runOnce processes one batch of due bookmarks.
func (p *Poller) runOnce(ctx context.Context) {
cutoff := p.Now().Add(-p.Cooldown).UnixMilli()
due, err := p.Store.DueForLatestCheck(cutoff, p.Batch)
if err != nil {
log.Printf("latest poll: due query: %v", err)
return
}
checked := 0
for i, b := range due {
if ctx.Err() != nil {
break
}
// Staggered rather than fired together: a burst of simultaneous requests
// from one server IP is the traffic shape most likely to move that IP's
// bot score. This is the server-side analogue of the userscript's "one
// series per navigation ... indistinguishable from browsing" (L455-456).
stopped := false
if i > 0 && p.Stagger > 0 {
select {
case <-ctx.Done():
stopped = true
case <-time.After(p.Stagger):
}
}
if stopped {
break
}
p.checkOne(ctx, b)
checked++
}
// due vs checked is how you tell which constraint is binding: ticks that
// report due=0 mean the cooldown is the limit, ticks that report due==batch
// every time mean throughput is.
log.Printf("latest poll: due=%d checked=%d", len(due), checked)
}
// checkOne re-checks one series. Every failure path here is "log and move on":
// the poller is a best-effort enhancement, and no single bad series may stall a
// batch or take down the process.
func (p *Poller) checkOne(ctx context.Context, b store.Bookmark) {
defer func() {
if r := recover(); r != nil {
log.Printf("latest poll %q: recovered from panic: %v", b.Key, r)
}
}()
// Stamped before the fetch, not after, so an error, a timeout, or a shutdown
// mid-request still consumes the cooldown. Otherwise a renamed or deleted
// series would be retried on every single tick forever. The userscript
// stamps in the same order and for the same reason (L471-473).
if err := p.Store.MarkLatestChecked(b.Key, p.Now().UnixMilli()); err != nil {
log.Printf("latest poll %q: mark checked: %v", b.Key, err)
return
}
// series_url is client-supplied (PUT /bookmarks/{key} accepts any string),
// so this is not just an optimisation against burning a request on an
// unknown site: without it, the server would issue a GET from its own
// network position to whatever URL a token-holder writes, including
// link-local/internal addresses or non-https schemes. The cooldown above
// is already consumed, so a row that never passes this check is retried at
// cooldown pace rather than hot-looping.
if !fetchableSeriesURL(b.Site, b.SeriesURL) {
log.Printf("latest poll %q: not fetchable: site=%q url=%q", b.Key, b.Site, b.SeriesURL)
return
}
body, status, err := p.Fetch.Get(ctx, b.SeriesURL)
if err != nil {
log.Printf("latest poll %q: fetch %s: %v", b.Key, b.SeriesURL, err)
return
}
if status != 200 {
log.Printf("latest poll %q: fetch %s: status %d", b.Key, b.SeriesURL, status)
return
}
latest, ok := latestChapterFrom(b.Site, b.SeriesURL, body)
if !ok {
// Most likely a challenge page or a layout change. Either way the row is
// already stamped, so this waits out a cooldown instead of hot-looping.
log.Printf("latest poll %q: no chapter links in %d bytes", b.Key, len(body))
return
}
// Re-read: the row may have been updated or deleted while the fetch was in
// flight, and writing b back wholesale would undo that.
//
// ponytail: non-transactional read-modify-write, wrap Get+Upsert in a tx if
// this ever runs for more than one user. A client PUT that commits between
// these two statements is lost to the stale re-read — reverting read
// progress or a status change, and moving updated_at because the stored
// value now differs. Accepted for a single-user deployment: the window is
// milliseconds and the loser is one poll cycle.
cur, found, err := p.Store.Get(b.Key)
if err != nil {
log.Printf("latest poll %q: reread: %v", b.Key, err)
return
}
if !found {
return
}
// Equality, not >, mirroring the userscript (L427): a site that retracts a
// chapter should correct the stored number downward.
if cur.LatestChapterNum != nil && *cur.LatestChapterNum == latest.Num {
return
}
num := latest.Num
cur.LatestChapter = latest.Label
cur.LatestChapterNum = &num
// A candidate only. last_chapter_num is untouched, so the CASE in Upsert
// keeps the stored updated_at and the bookmark list does not reorder.
cur.UpdatedAt = p.Now().UnixMilli()
if _, err := p.Store.Upsert(cur); err != nil {
log.Printf("latest poll %q: upsert: %v", b.Key, err)
return
}
log.Printf("latest poll %q: latest is now %s", b.Key, latest.Label)
}
// fetchableSeriesURL reports whether site is a site latestChapterFrom knows how
// to parse and seriesURL is safe to hand to the fetcher: an https URL with a
// non-empty host. series_url comes from client-supplied PUT bodies, so this is
// a defence against the poller being used to probe arbitrary hosts from the
// server's own network position, not just a check against wasted requests.
func fetchableSeriesURL(site, seriesURL string) bool {
switch site {
case "asura", "demonic":
default:
return false
}
u, err := url.Parse(seriesURL)
if err != nil {
return false
}
return u.Scheme == "https" && u.Host != ""
}
+360
View File
@@ -0,0 +1,360 @@
package latest
import (
"context"
"errors"
"path/filepath"
"sync"
"testing"
"time"
"mangabm/backend/internal/store"
)
// newTestStore opens a fresh SQLite store in a temp dir.
func newTestStore(t *testing.T) *store.Store {
t.Helper()
s, err := store.Open(filepath.Join(t.TempDir(), "test.db"))
if err != nil {
t.Fatalf("Open: %v", err)
}
t.Cleanup(func() { s.Close() })
return s
}
// seedForCheck inserts a bookmark and forces its latest_checked_at.
func seedForCheck(t *testing.T, s *store.Store, key, seriesURL string, checkedAt int64) {
t.Helper()
if _, err := s.Upsert(store.Bookmark{
Key: key,
Site: "asura",
SeriesID: key,
SeriesURL: seriesURL,
UpdatedAt: 1000,
}); err != nil {
t.Fatalf("seed %q: %v", key, err)
}
if err := s.MarkLatestChecked(key, checkedAt); err != nil {
t.Fatalf("seed mark %q: %v", key, err)
}
}
func readLatestCheckedAt(t *testing.T, s *store.Store, key string) int64 {
t.Helper()
ts, err := s.LatestCheckedAt(key)
if err != nil {
t.Fatalf("LatestCheckedAt %q: %v", key, err)
}
return ts
}
// fakeFetcher stands in for the network. Every poller test uses it, so nothing
// in this file can reach tls-client or a real site.
type fakeFetcher struct {
mu sync.Mutex
calls []string
body string
status int
err error
// perURL overrides body/status/err for specific URLs.
perURL map[string]fakeResponse
}
type fakeResponse struct {
body string
status int
err error
}
func (f *fakeFetcher) Get(ctx context.Context, url string) (string, int, error) {
f.mu.Lock()
f.calls = append(f.calls, url)
f.mu.Unlock()
if r, ok := f.perURL[url]; ok {
return r.body, r.status, r.err
}
return f.body, f.status, f.err
}
func (f *fakeFetcher) callCount() int {
f.mu.Lock()
defer f.mu.Unlock()
return len(f.calls)
}
// newTestPoller wires a poller with a frozen clock and no stagger, so tests run
// instantly and deterministically.
func newTestPoller(t *testing.T, s *store.Store, f Fetcher, at time.Time) *Poller {
t.Helper()
return &Poller{
Store: s,
Fetch: f,
Now: func() time.Time { return at },
Cooldown: time.Hour,
Interval: 10 * time.Minute,
Stagger: 0,
Batch: 14,
}
}
func TestRunOnceRecordsLatestChapter(t *testing.T) {
s := newTestStore(t)
const url = "https://asurascans.com/comics/chronicles-of-the-demon-faction-f886a8af"
seedForCheck(t, s, "asura:chronicles-of-the-demon-faction-f886a8af", url, 0)
now := time.UnixMilli(5_000_000)
f := &fakeFetcher{body: asuraSeriesFixture, status: 200}
newTestPoller(t, s, f, now).runOnce(context.Background())
b, ok, err := s.Get("asura:chronicles-of-the-demon-faction-f886a8af")
if err != nil || !ok {
t.Fatalf("Get: %v ok=%v", err, ok)
}
if b.LatestChapterNum == nil || *b.LatestChapterNum != 181 {
t.Fatalf("LatestChapterNum = %v, want 181", b.LatestChapterNum)
}
if b.LatestChapter != "Chapter 181" {
t.Fatalf("LatestChapter = %q, want %q", b.LatestChapter, "Chapter 181")
}
if got := readLatestCheckedAt(t, s, "asura:chronicles-of-the-demon-faction-f886a8af"); got != now.UnixMilli() {
t.Fatalf("latest_checked_at = %d, want %d", got, now.UnixMilli())
}
}
// The whole point of the updated_at CASE in Upsert: a newly published chapter is
// not reading progress and must not move the series up the list.
func TestRunOnceDoesNotReorderList(t *testing.T) {
s := newTestStore(t)
const url = "https://asurascans.com/comics/chronicles-of-the-demon-faction-f886a8af"
const key = "asura:chronicles-of-the-demon-faction-f886a8af"
// "other" is the most recently read, so it must stay at the top of List().
if _, err := s.Upsert(store.Bookmark{
Key: "asura:other", Site: "asura", SeriesID: "other",
SeriesURL: "https://asurascans.com/comics/other", UpdatedAt: 9_000_000,
}); err != nil {
t.Fatalf("seed other: %v", err)
}
seedForCheck(t, s, key, url, 0)
before, _, err := s.Get(key)
if err != nil {
t.Fatalf("Get before: %v", err)
}
f := &fakeFetcher{body: asuraSeriesFixture, status: 200}
newTestPoller(t, s, f, time.UnixMilli(9_999_999)).runOnce(context.Background())
after, _, err := s.Get(key)
if err != nil {
t.Fatalf("Get after: %v", err)
}
if after.UpdatedAt != before.UpdatedAt {
t.Fatalf("updated_at moved from %d to %d on a latest-chapter bump",
before.UpdatedAt, after.UpdatedAt)
}
list, err := s.List()
if err != nil {
t.Fatalf("List: %v", err)
}
if list[0].Key != "asura:other" {
t.Fatalf("list reordered: head is %q, want asura:other", list[0].Key)
}
}
// A failed fetch must still consume the cooldown, or a renamed series gets
// retried on every tick forever.
func TestRunOnceMarksCheckedOnFailure(t *testing.T) {
tests := []struct {
name string
resp fakeResponse
}{
{"network error", fakeResponse{err: errors.New("dial tcp: refused")}},
{"non-200", fakeResponse{body: "nope", status: 503}},
{"challenge page", fakeResponse{body: challengeFixture, status: 200}},
{"empty body", fakeResponse{body: "", status: 200}},
}
for _, tt := range tests {
t.Run(tt.name, func(t *testing.T) {
s := newTestStore(t)
const url = "https://asurascans.com/comics/x"
seedForCheck(t, s, "asura:x", url, 0)
now := time.UnixMilli(7_000_000)
f := &fakeFetcher{perURL: map[string]fakeResponse{url: tt.resp}}
newTestPoller(t, s, f, now).runOnce(context.Background())
if got := readLatestCheckedAt(t, s, "asura:x"); got != now.UnixMilli() {
t.Fatalf("latest_checked_at = %d, want %d", got, now.UnixMilli())
}
b, _, err := s.Get("asura:x")
if err != nil {
t.Fatalf("Get: %v", err)
}
if b.LatestChapterNum != nil {
t.Fatalf("LatestChapterNum = %v, want nil on a failed check", *b.LatestChapterNum)
}
})
}
}
func TestRunOnceRespectsBatchLimit(t *testing.T) {
s := newTestStore(t)
for i := 0; i < 20; i++ {
key := "asura:s" + string(rune('a'+i))
seedForCheck(t, s, key, "https://asurascans.com/comics/"+key, 0)
}
f := &fakeFetcher{body: "", status: 200}
p := newTestPoller(t, s, f, time.UnixMilli(5_000_000))
p.Batch = 5
p.runOnce(context.Background())
if got := f.callCount(); got != 5 {
t.Fatalf("fetched %d series, want 5 (batch limit)", got)
}
}
// One unreachable series must not abandon the rest of the batch.
func TestRunOnceOneBadSeriesDoesNotStallBatch(t *testing.T) {
s := newTestStore(t)
keys := []string{"asura:a", "asura:b", "asura:c", "asura:d", "asura:e"}
for _, k := range keys {
seedForCheck(t, s, k, "https://asurascans.com/comics/"+k, 0)
}
now := time.UnixMilli(6_000_000)
f := &fakeFetcher{
body: "", status: 200,
perURL: map[string]fakeResponse{
"https://asurascans.com/comics/asura:b": {err: errors.New("boom")},
},
}
newTestPoller(t, s, f, now).runOnce(context.Background())
if got := f.callCount(); got != 5 {
t.Fatalf("fetched %d series, want all 5 attempted", got)
}
for _, k := range keys {
if got := readLatestCheckedAt(t, s, k); got != now.UnixMilli() {
t.Fatalf("%s latest_checked_at = %d, want %d", k, got, now.UnixMilli())
}
}
}
// The cooldown is enforced by the due query, so a second immediate pass must do
// nothing at all — this is what makes the tick interval independent of it.
func TestRunOnceHonoursCooldownAcrossPasses(t *testing.T) {
s := newTestStore(t)
const url = "https://asurascans.com/comics/x"
seedForCheck(t, s, "asura:x", url, 0)
now := time.UnixMilli(8_000_000)
f := &fakeFetcher{body: asuraSeriesFixture, status: 200}
p := newTestPoller(t, s, f, now)
p.runOnce(context.Background())
if got := f.callCount(); got != 1 {
t.Fatalf("first pass fetched %d, want 1", got)
}
// Same instant, and again 59 minutes later: both inside the 1h cooldown.
p.runOnce(context.Background())
p.Now = func() time.Time { return now.Add(59 * time.Minute) }
p.runOnce(context.Background())
if got := f.callCount(); got != 1 {
t.Fatalf("fetched %d times inside the cooldown, want 1", got)
}
// Past the cooldown, it is due again.
p.Now = func() time.Time { return now.Add(61 * time.Minute) }
p.runOnce(context.Background())
if got := f.callCount(); got != 2 {
t.Fatalf("fetched %d times after the cooldown, want 2", got)
}
}
// A site that retracts a chapter should correct the stored number downward,
// mirroring the userscript's equality check (L427) rather than a >.
func TestRunOnceCorrectsDownward(t *testing.T) {
s := newTestStore(t)
const url = "https://demonicscans.org/manga/Catastrophic-Necromancer"
const key = "demonic:Catastrophic-Necromancer"
high := 400.0
if _, err := s.Upsert(store.Bookmark{
Key: key, Site: "demonic", SeriesID: "Catastrophic-Necromancer",
SeriesURL: url, LatestChapter: "Chapter 400", LatestChapterNum: &high,
UpdatedAt: 1000,
}); err != nil {
t.Fatalf("seed: %v", err)
}
f := &fakeFetcher{body: demonicSeriesFixture, status: 200}
newTestPoller(t, s, f, time.UnixMilli(5_000_000)).runOnce(context.Background())
b, _, err := s.Get(key)
if err != nil {
t.Fatalf("Get: %v", err)
}
if b.LatestChapterNum == nil || *b.LatestChapterNum != 296 {
t.Fatalf("LatestChapterNum = %v, want 296", b.LatestChapterNum)
}
}
// series_url is client-supplied via PUT /bookmarks/{key}, so checkOne must
// reject anything that is not a known site with an https URL before spending a
// request on it — the cooldown still gets consumed either way.
func TestCheckOneValidatesSeriesURLBeforeFetching(t *testing.T) {
tests := []struct {
name string
site string
seriesURL string
wantCalls int
}{
{"unknown site", "mangadex", "https://mangadex.org/title/x", 0},
{"http scheme", "asura", "http://asurascans.com/comics/x", 0},
{"unparseable url", "asura", "http://[::1", 0},
{"valid https asura", "asura", "https://asurascans.com/comics/x", 1},
}
for _, tt := range tests {
t.Run(tt.name, func(t *testing.T) {
s := newTestStore(t)
key := tt.site + ":x"
if _, err := s.Upsert(store.Bookmark{
Key: key, Site: tt.site, SeriesID: "x", SeriesURL: tt.seriesURL,
UpdatedAt: 1000,
}); err != nil {
t.Fatalf("seed: %v", err)
}
now := time.UnixMilli(4_000_000)
f := &fakeFetcher{body: asuraSeriesFixture, status: 200}
newTestPoller(t, s, f, now).checkOne(context.Background(), store.Bookmark{
Key: key, Site: tt.site, SeriesURL: tt.seriesURL,
})
if got := f.callCount(); got != tt.wantCalls {
t.Fatalf("fetch calls = %d, want %d", got, tt.wantCalls)
}
if got := readLatestCheckedAt(t, s, key); got != now.UnixMilli() {
t.Fatalf("latest_checked_at = %d, want %d (cooldown must be consumed regardless)", got, now.UnixMilli())
}
})
}
}
// A cancelled context must abandon the batch rather than run it to completion.
func TestRunOnceStopsOnCancelledContext(t *testing.T) {
s := newTestStore(t)
for _, k := range []string{"asura:a", "asura:b", "asura:c"} {
seedForCheck(t, s, k, "https://asurascans.com/comics/"+k, 0)
}
ctx, cancel := context.WithCancel(context.Background())
cancel()
f := &fakeFetcher{body: "", status: 200}
newTestPoller(t, s, f, time.UnixMilli(5_000_000)).runOnce(ctx)
if got := f.callCount(); got != 0 {
t.Fatalf("fetched %d series with a cancelled context, want 0", got)
}
}
+84
View File
@@ -0,0 +1,84 @@
package latest
import (
"regexp"
"strconv"
"strings"
"mangabm/backend/internal/store"
)
// latestChapter is the newest chapter a series page advertises.
type latestChapter struct {
Num float64
Label string
}
// asuraSlugRe pulls the series slug out of a stored series_url.
// Shape verified live 2026-07-26: https://asurascans.com/comics/<slug>, where
// the slug carries a trailing build-hash suffix (e.g. "-f886a8af") that
// rotates on every site redeploy — callers must strip it (asuraBuildHash)
// before using the slug to scope anything.
var asuraSlugRe = regexp.MustCompile(`/comics/([^/?#]+)`)
// demonicChapterRe matches the pre-redirect anchors demonic series pages link
// through. Both the raw "&" and the HTML-escaped "&amp;" forms occur.
var demonicChapterRe = regexp.MustCompile(`chaptered\.php\?manga=\d+&(?:amp;)?chapter=([0-9.]+)`)
// latestChapterFrom returns the highest chapter number body advertises for this
// series. ok is false when the body yields nothing usable — an unknown site, an
// empty body, a Cloudflare challenge page, and a site redesign all land here,
// and the caller treats all four identically.
//
// Ported from the userscript's latestChapterFromAnchors (asura L123-133,
// demonic L183-193), including its reason for taking a maximum rather than a
// first or last: neither site lists chapters in a dependable order.
//
// The userscript's asura rule additionally requires the anchor text to match
// /Chapter\s+[\d.]+/i. That check exists only to skip the "First Chapter"
// shortcut, which points at chapter/1 and therefore can never win a maximum, so
// it is redundant here. For asura, scoping the pattern to this series' own slug
// replaces it with a stronger guarantee: a chapter link belonging to some other
// series cannot contribute even if the page starts carrying them. demonic has no
// such guarantee — demonicChapterRe matches any chaptered.php?manga=<id> anchor
// with no per-series scoping, because the stored series_id for demonic is a
// slug, not the numeric id the URL carries, so it cannot easily be scoped.
func latestChapterFrom(site, seriesURL, body string) (latestChapter, bool) {
var re *regexp.Regexp
switch site {
case "asura":
m := asuraSlugRe.FindStringSubmatch(seriesURL)
if m == nil {
return latestChapter{}, false
}
// Stored URLs predating a redeploy may carry a stale build hash;
// chapter hrefs in the fetched body carry the current one. Strip to
// the stable ID (same rule as migrateAsuraKeys) and make the hash
// optional in the pattern, so scoping survives rotations.
slug := store.AsuraBuildHash.ReplaceAllString(m[1], "")
// Compiled per call rather than cached: this runs once per fetch, which
// is at most a few times a minute, and the slug varies per series.
re = regexp.MustCompile(`/comics/` + regexp.QuoteMeta(slug) + `(?:-[0-9a-f]{8})?/chapter/([0-9.]+)`)
case "demonic":
re = demonicChapterRe
default:
return latestChapter{}, false
}
var best latestChapter
found := false
for _, m := range re.FindAllStringSubmatch(body, -1) {
// [0-9.]+ can swallow a trailing separator, e.g. "chapter/12." in a
// sentence; ParseFloat would reject the whole match.
raw := strings.Trim(m[1], ".")
num, err := strconv.ParseFloat(raw, 64)
if err != nil {
continue
}
if !found || num > best.Num {
best = latestChapter{Num: num, Label: "Chapter " + raw}
found = true
}
}
return best, found
}
+122
View File
@@ -0,0 +1,122 @@
package latest
import "testing"
// Trimmed from https://asurascans.com/comics/chronicles-of-the-demon-faction-f886a8af
// fetched 2026-07-26. The first anchor is the "First Chapter" shortcut: it is a
// real chapter link with no "Chapter N" text, and it must not be mistaken for
// the latest just because it parses.
const asuraSeriesFixture = `
<a href="/comics/chronicles-of-the-demon-faction-f886a8af/chapter/1" class="py-3 rounded-md bg-[#E8E8E8]"><svg class="w-4 h-4"></svg>First Chapter</a>
<a href="/comics/chronicles-of-the-demon-faction-f886a8af/chapter/179" data-astro-prefetch="hover" class="group flex"><span class="font-medium">Chapter 179</span></a>
<a href="/comics/chronicles-of-the-demon-faction-f886a8af/chapter/181" data-astro-prefetch="hover" class="group flex"><span class="font-medium">Chapter 181</span></a>
<a href="/comics/chronicles-of-the-demon-faction-f886a8af/chapter/180" data-astro-prefetch="hover" class="group flex"><span class="font-medium">Chapter 180</span></a>
`
// A chapter link belonging to a different series, of the kind a "you might also
// like" strip would introduce. Slug scoping must exclude it.
const asuraCrossSeriesFixture = asuraSeriesFixture + `
<a href="/comics/some-other-series-aabbccdd/chapter/999" class="group flex"><span>Chapter 999</span></a>
`
// Trimmed from https://demonicscans.org/manga/Catastrophic-Necromancer fetched
// 2026-07-26. Note the raw "&", the doubled space after <a, and the decimal
// chapters, all as they appear live.
const demonicSeriesFixture = `
<a href="/chaptered.php?manga=11799&chapter=0.5" class="chplinks" title="Catastrophic Necromancer 0.5">Chapter 0.5</a>
<a href="/chaptered.php?manga=11799&chapter=294" class="chplinks" title="Catastrophic Necromancer 294">Chapter 294</a>
<a href="/chaptered.php?manga=11799&amp;chapter=296" class="chplinks" title="Catastrophic Necromancer 296">Chapter 296</a>
<a href="/chaptered.php?manga=11799&chapter=295" class="chplinks" title="Catastrophic Necromancer 295">Chapter 295</a>
`
// What Cloudflare serves instead of the page when an IP's bot score flips.
const challengeFixture = `<!DOCTYPE html><html><head><title>Just a moment...</title>
<script src="/cdn-cgi/challenge-platform/h/b/orchestrate/chl_page/v1"></script></head>
<body><div id="challenge-running">Checking your browser</div></body></html>`
func TestLatestChapterFrom(t *testing.T) {
const asuraURL = "https://asurascans.com/comics/chronicles-of-the-demon-faction-f886a8af"
const demonicURL = "https://demonicscans.org/manga/Catastrophic-Necromancer"
tests := []struct {
name string
site string
seriesURL string
body string
wantOK bool
wantNum float64
wantLabel string
}{
{
name: "asura takes the max, not the last listed",
site: "asura", seriesURL: asuraURL, body: asuraSeriesFixture,
wantOK: true, wantNum: 181, wantLabel: "Chapter 181",
},
{
name: "asura ignores another series' chapter links",
site: "asura", seriesURL: asuraURL, body: asuraCrossSeriesFixture,
wantOK: true, wantNum: 181, wantLabel: "Chapter 181",
},
{
name: "asura scoping survives a build-hash rotation",
site: "asura",
seriesURL: "https://asurascans.com/comics/chronicles-of-the-demon-faction-059befe1",
body: asuraCrossSeriesFixture,
wantOK: true, wantNum: 181, wantLabel: "Chapter 181",
},
{
name: "asura with an unparseable series url",
site: "asura", seriesURL: "https://asurascans.com/", body: asuraSeriesFixture,
wantOK: false,
},
{
name: "demonic takes the max across raw and escaped ampersands",
site: "demonic", seriesURL: demonicURL, body: demonicSeriesFixture,
wantOK: true, wantNum: 296, wantLabel: "Chapter 296",
},
{
name: "demonic keeps decimal chapters parseable",
site: "demonic", seriesURL: demonicURL,
body: `<a href="/chaptered.php?manga=11799&chapter=0.5">Chapter 0.5</a>`,
wantOK: true, wantNum: 0.5, wantLabel: "Chapter 0.5",
},
{
name: "empty body",
site: "asura", seriesURL: asuraURL, body: "",
wantOK: false,
},
{
name: "cloudflare challenge page",
site: "asura", seriesURL: asuraURL, body: challengeFixture,
wantOK: false,
},
{
name: "demonic markup handed to the asura rule",
site: "asura", seriesURL: asuraURL, body: demonicSeriesFixture,
wantOK: false,
},
{
name: "unknown site",
site: "mangadex", seriesURL: "https://example.com/x", body: asuraSeriesFixture,
wantOK: false,
},
}
for _, tt := range tests {
t.Run(tt.name, func(t *testing.T) {
got, ok := latestChapterFrom(tt.site, tt.seriesURL, tt.body)
if ok != tt.wantOK {
t.Fatalf("ok = %v, want %v (got %+v)", ok, tt.wantOK, got)
}
if !tt.wantOK {
return
}
if got.Num != tt.wantNum {
t.Errorf("Num = %v, want %v", got.Num, tt.wantNum)
}
if got.Label != tt.wantLabel {
t.Errorf("Label = %q, want %q", got.Label, tt.wantLabel)
}
})
}
}