Split backend into internal packages by responsibility

All Go files lived flat in backend/ as one package main. Move store,
latest-chapter polling, sessions, HTTP middleware, the JSON API, the
userscript handler, and the web UI (with its templates/static assets)
into backend/internal/{store,latest,session,httpmw,api,userscript,web},
each with an exported API. main.go becomes the composition root wiring
them into newRouter; root-level tests cover the assembled router while
package-local tests cover unit behavior. Update Dockerfile/.dockerignore
for the new internal/ tree and CLAUDE.md to describe the layout.

Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
This commit is contained in:
2026-08-02 19:41:38 +07:00
parent e250762ea6
commit eeb601cbe2
36 changed files with 971 additions and 843 deletions
+203
View File
@@ -0,0 +1,203 @@
package latest
import (
"context"
"log"
"net/url"
"time"
"mangabm/backend/internal/store"
)
// Fetcher retrieves a series page. It exists as an interface so tests can inject
// a fake: nothing in the test suite may touch the network or the TLS client.
type Fetcher interface {
Get(ctx context.Context, url string) (body string, status int, err error)
}
// Poller re-checks each bookmarked series' newest published chapter on a
// schedule, independent of the userscript's own in-browser checks. The two run
// in parallel and report the same observable fact, so whichever writes last wins
// and neither needs to know about the other.
//
// Two clocks, deliberately independent:
//
// - Interval is how often this goroutine wakes up and looks.
// - Cooldown is how long one bookmark rests since its own last check.
//
// Only the cooldown is per bookmark, and it is enforced by the WHERE clause in
// DueForLatestCheck rather than by any timer. Shortening Interval therefore
// cannot shorten anyone's cooldown; it only makes the poller wake up and find
// nothing due more often.
type Poller struct {
Store *store.Store
Fetch Fetcher
Now func() time.Time // injected so tests can freeze it
Cooldown time.Duration
Interval time.Duration
Stagger time.Duration
Batch int
}
// Run polls until ctx is cancelled.
//
// runOnce is called synchronously, so a batch that overruns the tick delays the
// next one instead of stacking a second batch on top of it. That is the intended
// failure mode for a misconfigured batch x stagger: a slower cadence, never
// concurrent fetch storms.
func (p *Poller) Run(ctx context.Context) {
log.Printf("latest-chapter poller: interval=%s cooldown=%s batch=%d stagger=%s",
p.Interval, p.Cooldown, p.Batch, p.Stagger)
t := time.NewTicker(p.Interval)
defer t.Stop()
for {
select {
case <-ctx.Done():
log.Println("latest-chapter poller: stopped")
return
case <-t.C:
p.runOnce(ctx)
}
}
}
// runOnce processes one batch of due bookmarks.
func (p *Poller) runOnce(ctx context.Context) {
cutoff := p.Now().Add(-p.Cooldown).UnixMilli()
due, err := p.Store.DueForLatestCheck(cutoff, p.Batch)
if err != nil {
log.Printf("latest poll: due query: %v", err)
return
}
checked := 0
for i, b := range due {
if ctx.Err() != nil {
break
}
// Staggered rather than fired together: a burst of simultaneous requests
// from one server IP is the traffic shape most likely to move that IP's
// bot score. This is the server-side analogue of the userscript's "one
// series per navigation ... indistinguishable from browsing" (L455-456).
stopped := false
if i > 0 && p.Stagger > 0 {
select {
case <-ctx.Done():
stopped = true
case <-time.After(p.Stagger):
}
}
if stopped {
break
}
p.checkOne(ctx, b)
checked++
}
// due vs checked is how you tell which constraint is binding: ticks that
// report due=0 mean the cooldown is the limit, ticks that report due==batch
// every time mean throughput is.
log.Printf("latest poll: due=%d checked=%d", len(due), checked)
}
// checkOne re-checks one series. Every failure path here is "log and move on":
// the poller is a best-effort enhancement, and no single bad series may stall a
// batch or take down the process.
func (p *Poller) checkOne(ctx context.Context, b store.Bookmark) {
defer func() {
if r := recover(); r != nil {
log.Printf("latest poll %q: recovered from panic: %v", b.Key, r)
}
}()
// Stamped before the fetch, not after, so an error, a timeout, or a shutdown
// mid-request still consumes the cooldown. Otherwise a renamed or deleted
// series would be retried on every single tick forever. The userscript
// stamps in the same order and for the same reason (L471-473).
if err := p.Store.MarkLatestChecked(b.Key, p.Now().UnixMilli()); err != nil {
log.Printf("latest poll %q: mark checked: %v", b.Key, err)
return
}
// series_url is client-supplied (PUT /bookmarks/{key} accepts any string),
// so this is not just an optimisation against burning a request on an
// unknown site: without it, the server would issue a GET from its own
// network position to whatever URL a token-holder writes, including
// link-local/internal addresses or non-https schemes. The cooldown above
// is already consumed, so a row that never passes this check is retried at
// cooldown pace rather than hot-looping.
if !fetchableSeriesURL(b.Site, b.SeriesURL) {
log.Printf("latest poll %q: not fetchable: site=%q url=%q", b.Key, b.Site, b.SeriesURL)
return
}
body, status, err := p.Fetch.Get(ctx, b.SeriesURL)
if err != nil {
log.Printf("latest poll %q: fetch %s: %v", b.Key, b.SeriesURL, err)
return
}
if status != 200 {
log.Printf("latest poll %q: fetch %s: status %d", b.Key, b.SeriesURL, status)
return
}
latest, ok := latestChapterFrom(b.Site, b.SeriesURL, body)
if !ok {
// Most likely a challenge page or a layout change. Either way the row is
// already stamped, so this waits out a cooldown instead of hot-looping.
log.Printf("latest poll %q: no chapter links in %d bytes", b.Key, len(body))
return
}
// Re-read: the row may have been updated or deleted while the fetch was in
// flight, and writing b back wholesale would undo that.
//
// ponytail: non-transactional read-modify-write, wrap Get+Upsert in a tx if
// this ever runs for more than one user. A client PUT that commits between
// these two statements is lost to the stale re-read — reverting read
// progress or a status change, and moving updated_at because the stored
// value now differs. Accepted for a single-user deployment: the window is
// milliseconds and the loser is one poll cycle.
cur, found, err := p.Store.Get(b.Key)
if err != nil {
log.Printf("latest poll %q: reread: %v", b.Key, err)
return
}
if !found {
return
}
// Equality, not >, mirroring the userscript (L427): a site that retracts a
// chapter should correct the stored number downward.
if cur.LatestChapterNum != nil && *cur.LatestChapterNum == latest.Num {
return
}
num := latest.Num
cur.LatestChapter = latest.Label
cur.LatestChapterNum = &num
// A candidate only. last_chapter_num is untouched, so the CASE in Upsert
// keeps the stored updated_at and the bookmark list does not reorder.
cur.UpdatedAt = p.Now().UnixMilli()
if _, err := p.Store.Upsert(cur); err != nil {
log.Printf("latest poll %q: upsert: %v", b.Key, err)
return
}
log.Printf("latest poll %q: latest is now %s", b.Key, latest.Label)
}
// fetchableSeriesURL reports whether site is a site latestChapterFrom knows how
// to parse and seriesURL is safe to hand to the fetcher: an https URL with a
// non-empty host. series_url comes from client-supplied PUT bodies, so this is
// a defence against the poller being used to probe arbitrary hosts from the
// server's own network position, not just a check against wasted requests.
func fetchableSeriesURL(site, seriesURL string) bool {
switch site {
case "asura", "demonic":
default:
return false
}
u, err := url.Parse(seriesURL)
if err != nil {
return false
}
return u.Scheme == "https" && u.Host != ""
}