feat: per-site latest-chapter extraction from series HTML
Ports latestChapterFromAnchors from the userscript for asura and demonic. Asura's pattern is scoped to the series' own slug, which subsumes the userscript's anchor-text check and also excludes chapter links belonging to other series. Fixtures are trimmed from real pages.
This commit is contained in:
@@ -0,0 +1,72 @@
|
||||
package main
|
||||
|
||||
import (
|
||||
"regexp"
|
||||
"strconv"
|
||||
"strings"
|
||||
)
|
||||
|
||||
// latestChapter is the newest chapter a series page advertises.
|
||||
type latestChapter struct {
|
||||
Num float64
|
||||
Label string
|
||||
}
|
||||
|
||||
// asuraSlugRe pulls the series slug out of a stored series_url.
|
||||
// Shape verified live 2026-07-26: https://asurascans.com/comics/<slug>, where
|
||||
// the slug carries a trailing hash-like suffix (e.g. "-f886a8af").
|
||||
var asuraSlugRe = regexp.MustCompile(`/comics/([^/?#]+)`)
|
||||
|
||||
// demonicChapterRe matches the pre-redirect anchors demonic series pages link
|
||||
// through. Both the raw "&" and the HTML-escaped "&" forms occur.
|
||||
var demonicChapterRe = regexp.MustCompile(`chaptered\.php\?manga=\d+&(?:amp;)?chapter=([0-9.]+)`)
|
||||
|
||||
// latestChapterFrom returns the highest chapter number body advertises for this
|
||||
// series. ok is false when the body yields nothing usable — an unknown site, an
|
||||
// empty body, a Cloudflare challenge page, and a site redesign all land here,
|
||||
// and the caller treats all four identically.
|
||||
//
|
||||
// Ported from the userscript's latestChapterFromAnchors (asura L123-133,
|
||||
// demonic L183-193), including its reason for taking a maximum rather than a
|
||||
// first or last: neither site lists chapters in a dependable order.
|
||||
//
|
||||
// The userscript's asura rule additionally requires the anchor text to match
|
||||
// /Chapter\s+[\d.]+/i. That check exists only to skip the "First Chapter"
|
||||
// shortcut, which points at chapter/1 and therefore can never win a maximum, so
|
||||
// it is redundant here. Scoping the pattern to this series' own slug replaces it
|
||||
// with a stronger guarantee: a chapter link belonging to some other series
|
||||
// cannot contribute even if the page starts carrying them.
|
||||
func latestChapterFrom(site, seriesURL, body string) (latestChapter, bool) {
|
||||
var re *regexp.Regexp
|
||||
switch site {
|
||||
case "asura":
|
||||
m := asuraSlugRe.FindStringSubmatch(seriesURL)
|
||||
if m == nil {
|
||||
return latestChapter{}, false
|
||||
}
|
||||
// Compiled per call rather than cached: this runs once per fetch, which
|
||||
// is at most a few times a minute, and the slug varies per series.
|
||||
re = regexp.MustCompile(`/comics/` + regexp.QuoteMeta(m[1]) + `/chapter/([0-9.]+)`)
|
||||
case "demonic":
|
||||
re = demonicChapterRe
|
||||
default:
|
||||
return latestChapter{}, false
|
||||
}
|
||||
|
||||
var best latestChapter
|
||||
found := false
|
||||
for _, m := range re.FindAllStringSubmatch(body, -1) {
|
||||
// [0-9.]+ can swallow a trailing separator, e.g. "chapter/12." in a
|
||||
// sentence; ParseFloat would reject the whole match.
|
||||
raw := strings.Trim(m[1], ".")
|
||||
num, err := strconv.ParseFloat(raw, 64)
|
||||
if err != nil {
|
||||
continue
|
||||
}
|
||||
if !found || num > best.Num {
|
||||
best = latestChapter{Num: num, Label: "Chapter " + raw}
|
||||
found = true
|
||||
}
|
||||
}
|
||||
return best, found
|
||||
}
|
||||
Reference in New Issue
Block a user