Files
mangaBookmark/backend/latest_sites.go
T
sulthan e966182e6f fix: address final-review findings on the latest-chapter poller
Wires the five poll env vars into docker-compose (the documented kill
switch was inert), skips fetching unknown sites and non-https URLs
before spending a request, and makes runOnce's summary log fire on
empty and cancelled ticks. Records the accepted non-atomic Get+Upsert
window and the one-interval startup delay in the docs.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
2026-07-26 18:18:11 +07:00

76 lines
2.9 KiB
Go

package main
import (
"regexp"
"strconv"
"strings"
)
// latestChapter is the newest chapter a series page advertises.
type latestChapter struct {
Num float64
Label string
}
// asuraSlugRe pulls the series slug out of a stored series_url.
// Shape verified live 2026-07-26: https://asurascans.com/comics/<slug>, where
// the slug carries a trailing hash-like suffix (e.g. "-f886a8af").
var asuraSlugRe = regexp.MustCompile(`/comics/([^/?#]+)`)
// demonicChapterRe matches the pre-redirect anchors demonic series pages link
// through. Both the raw "&" and the HTML-escaped "&amp;" forms occur.
var demonicChapterRe = regexp.MustCompile(`chaptered\.php\?manga=\d+&(?:amp;)?chapter=([0-9.]+)`)
// latestChapterFrom returns the highest chapter number body advertises for this
// series. ok is false when the body yields nothing usable — an unknown site, an
// empty body, a Cloudflare challenge page, and a site redesign all land here,
// and the caller treats all four identically.
//
// Ported from the userscript's latestChapterFromAnchors (asura L123-133,
// demonic L183-193), including its reason for taking a maximum rather than a
// first or last: neither site lists chapters in a dependable order.
//
// The userscript's asura rule additionally requires the anchor text to match
// /Chapter\s+[\d.]+/i. That check exists only to skip the "First Chapter"
// shortcut, which points at chapter/1 and therefore can never win a maximum, so
// it is redundant here. For asura, scoping the pattern to this series' own slug
// replaces it with a stronger guarantee: a chapter link belonging to some other
// series cannot contribute even if the page starts carrying them. demonic has no
// such guarantee — demonicChapterRe matches any chaptered.php?manga=<id> anchor
// with no per-series scoping, because the stored series_id for demonic is a
// slug, not the numeric id the URL carries, so it cannot easily be scoped.
func latestChapterFrom(site, seriesURL, body string) (latestChapter, bool) {
var re *regexp.Regexp
switch site {
case "asura":
m := asuraSlugRe.FindStringSubmatch(seriesURL)
if m == nil {
return latestChapter{}, false
}
// Compiled per call rather than cached: this runs once per fetch, which
// is at most a few times a minute, and the slug varies per series.
re = regexp.MustCompile(`/comics/` + regexp.QuoteMeta(m[1]) + `/chapter/([0-9.]+)`)
case "demonic":
re = demonicChapterRe
default:
return latestChapter{}, false
}
var best latestChapter
found := false
for _, m := range re.FindAllStringSubmatch(body, -1) {
// [0-9.]+ can swallow a trailing separator, e.g. "chapter/12." in a
// sentence; ParseFloat would reject the whole match.
raw := strings.Trim(m[1], ".")
num, err := strconv.ParseFloat(raw, 64)
if err != nil {
continue
}
if !found || num > best.Num {
best = latestChapter{Num: num, Label: "Chapter " + raw}
found = true
}
}
return best, found
}