feat(backend)!: split Series from Bookmark, keeping the wire format flat

Series becomes a shared row keyed (site, series_id) owning title, cover,
canonical URL, kind, latest chapter and last-checked time (ADR-0003).
A bookmark keeps only progress, favourite, lifecycle bucket, updated_at.

Store.Upsert decomposes one flat body across both tables in one
transaction: client title/series_url/cover apply only when the series row
is new, then only the poll may change them (security boundary — the row
is shared and the values are scraped page content). Reads join series
back in, so GET/PUT emit and accept exactly the flat field set they did
before (ADR-0004), asserted by TestFlatWireFieldSet.

The poller walks Series instead of Bookmarks: one fetch per shared
series per due cycle, due queue ordered reader_count DESC then
latest_checked_at ASC, orphaned series never due and never deleted, row
still stamped before the fetch. Batch/stagger/interval unchanged.

Migration 0002 backfills series from existing bookmarks; verified by
TestMigration0002BackfillsExistingBookmarks.
This commit is contained in:
2026-08-08 07:11:43 +07:00
parent 08749df050
commit a48aba67b8
8 changed files with 756 additions and 158 deletions
+29 -46
View File
@@ -23,9 +23,9 @@ type Fetcher interface {
// Two clocks, deliberately independent:
//
// - Interval is how often this goroutine wakes up and looks.
// - Cooldown is how long one bookmark rests since its own last check.
// - Cooldown is how long one series rests since its own last check.
//
// Only the cooldown is per bookmark, and it is enforced by the WHERE clause in
// Only the cooldown is per series, and it is enforced by the WHERE clause in
// DueForLatestCheck rather than by any timer. Shortening Interval therefore
// cannot shorten anyone's cooldown; it only makes the poller wake up and find
// nothing due more often.
@@ -78,7 +78,7 @@ func (p *Poller) Run(ctx context.Context) {
}
}
// runOnce processes one batch of due bookmarks.
// runOnce processes one batch of due series.
func (p *Poller) runOnce(ctx context.Context) {
cutoff := p.Now().Add(-p.Cooldown).UnixMilli()
due, err := p.Store.DueForLatestCheck(cutoff, p.Batch)
@@ -88,7 +88,7 @@ func (p *Poller) runOnce(ctx context.Context) {
}
checked := 0
for i, b := range due {
for i, sr := range due {
if ctx.Err() != nil {
break
}
@@ -107,7 +107,7 @@ func (p *Poller) runOnce(ctx context.Context) {
if stopped {
break
}
p.checkOne(ctx, b)
p.checkOne(ctx, sr)
checked++
}
// due vs checked is how you tell which constraint is binding: ticks that
@@ -119,10 +119,10 @@ func (p *Poller) runOnce(ctx context.Context) {
// checkOne re-checks one series. Every failure path here is "log and move on":
// the poller is a best-effort enhancement, and no single bad series may stall a
// batch or take down the process.
func (p *Poller) checkOne(ctx context.Context, b store.Bookmark) {
func (p *Poller) checkOne(ctx context.Context, sr store.Series) {
defer func() {
if r := recover(); r != nil {
log.Printf("latest poll %q: recovered from panic: %v", b.Key, r)
log.Printf("latest poll %q: recovered from panic: %v", sr.Key(), r)
}
}()
@@ -130,8 +130,8 @@ func (p *Poller) checkOne(ctx context.Context, b store.Bookmark) {
// mid-request still consumes the cooldown. Otherwise a renamed or deleted
// series would be retried on every single tick forever. The userscript
// stamps in the same order and for the same reason (L471-473).
if err := p.Store.MarkLatestChecked(b.Key, p.Now().UnixMilli()); err != nil {
log.Printf("latest poll %q: mark checked: %v", b.Key, err)
if err := p.Store.MarkLatestChecked(sr.Site, sr.SeriesID, p.Now().UnixMilli()); err != nil {
log.Printf("latest poll %q: mark checked: %v", sr.Key(), err)
return
}
@@ -142,69 +142,52 @@ func (p *Poller) checkOne(ctx context.Context, b store.Bookmark) {
// link-local/internal addresses or non-https schemes. The cooldown above
// is already consumed, so a row that never passes this check is retried at
// cooldown pace rather than hot-looping.
if !fetchableSeriesURL(b.Site, b.SeriesURL) {
log.Printf("latest poll %q: not fetchable: site=%q url=%q", b.Key, b.Site, b.SeriesURL)
if !fetchableSeriesURL(sr.Site, sr.SeriesURL) {
log.Printf("latest poll %q: not fetchable: site=%q url=%q", sr.Key(), sr.Site, sr.SeriesURL)
return
}
f := p.fetcherFor(b.Site)
f := p.fetcherFor(sr.Site)
if f == nil {
log.Printf("latest poll %q: no fetcher for site %q", b.Key, b.Site)
log.Printf("latest poll %q: no fetcher for site %q", sr.Key(), sr.Site)
return
}
body, status, err := f.Get(ctx, b.SeriesURL)
body, status, err := f.Get(ctx, sr.SeriesURL)
if err != nil {
log.Printf("latest poll %q: fetch %s: %v", b.Key, b.SeriesURL, err)
log.Printf("latest poll %q: fetch %s: %v", sr.Key(), sr.SeriesURL, err)
return
}
if status != 200 {
log.Printf("latest poll %q: fetch %s: status %d", b.Key, b.SeriesURL, status)
log.Printf("latest poll %q: fetch %s: status %d", sr.Key(), sr.SeriesURL, status)
return
}
latest, ok := latestChapterFrom(b.Site, b.SeriesURL, body)
latest, ok := latestChapterFrom(sr.Site, sr.SeriesURL, body)
if !ok {
// Most likely a challenge page or a layout change. Either way the row is
// already stamped, so this waits out a cooldown instead of hot-looping.
log.Printf("latest poll %q: no chapter links in %d bytes", b.Key, len(body))
log.Printf("latest poll %q: no chapter links in %d bytes", sr.Key(), len(body))
return
}
// Re-read: the row may have been updated or deleted while the fetch was in
// flight, and writing b back wholesale would undo that.
//
// ponytail: non-transactional read-modify-write, wrap Get+Upsert in a tx if
// this ever runs for more than one user. A client PUT that commits between
// these two statements is lost to the stale re-read — reverting read
// progress or a status change, and moving updated_at because the stored
// value now differs. Accepted for a single-user deployment: the window is
// milliseconds and the loser is one poll cycle.
cur, found, err := p.Store.Get(b.Key)
if err != nil {
log.Printf("latest poll %q: reread: %v", b.Key, err)
return
}
if !found {
return
}
// Equality, not >, mirroring the userscript (L427): a site that retracts a
// chapter should correct the stored number downward.
if cur.LatestChapterNum != nil && *cur.LatestChapterNum == latest.Num {
// chapter should correct the stored number downward. The comparison is
// against the due-query snapshot; a concurrent write in between only costs
// one redundant UPDATE of the same absolute value, never a wrong one.
if sr.LatestChapterNum != nil && *sr.LatestChapterNum == latest.Num {
return
}
num := latest.Num
cur.LatestChapter = latest.Label
cur.LatestChapterNum = &num
// A candidate only. last_chapter_num is untouched, so the CASE in Upsert
// keeps the stored updated_at and the bookmark list does not reorder.
cur.UpdatedAt = p.Now().UnixMilli()
if _, err := p.Store.Upsert(cur); err != nil {
log.Printf("latest poll %q: upsert: %v", b.Key, err)
// Series-level write: the row is shared, so one update refreshes every
// bookmark joining to it, and the bookmark's updated_at is never touched —
// a newly published chapter is not reading progress and must not reorder
// the list.
if err := p.Store.SetLatestChapter(sr.Site, sr.SeriesID, latest.Label, latest.Num); err != nil {
log.Printf("latest poll %q: set latest chapter: %v", sr.Key(), err)
return
}
log.Printf("latest poll %q: latest is now %s", b.Key, latest.Label)
log.Printf("latest poll %q: latest is now %s", sr.Key(), latest.Label)
}
// fetchableSeriesURL reports whether site is a site latestChapterFrom knows how