51b0094430
Cross-reference lnwChapterRe from the function doc instead of restating its rationale, drop the redundant '; skipping' from the fail-closed log to match the package's bare verb:detail form, and reword the truncation comment so the logged body length is not described as a code-made distinction.
300 lines
12 KiB
Go
300 lines
12 KiB
Go
package latest
|
|
|
|
import (
|
|
"encoding/json"
|
|
"html"
|
|
"log"
|
|
"net/url"
|
|
"regexp"
|
|
"strconv"
|
|
"strings"
|
|
)
|
|
|
|
// latestChapter is the newest chapter a series page advertises.
|
|
type latestChapter struct {
|
|
Num float64
|
|
Label string
|
|
}
|
|
|
|
// asuraSlugRe pulls the series slug out of a stored series_url.
|
|
// Shape verified live 2026-07-26: https://asurascans.com/comics/<slug>, where
|
|
// the slug carries a trailing build-hash suffix (e.g. "-f886a8af") that
|
|
// rotates on every site redeploy — callers must strip it (asuraBuildHash)
|
|
// before using the slug to scope anything.
|
|
var asuraSlugRe = regexp.MustCompile(`/comics/([^/?#]+)`)
|
|
|
|
// asuraBuildHash matches the trailing "-xxxxxxxx" site-wide build ID Asura
|
|
// appends to every series slug. It rotates on each site redeploy, so it is
|
|
// never part of a stable series_id. Must stay in sync with stripBuildHash in
|
|
// userscript/manga-bookmark.user.js.
|
|
var asuraBuildHash = regexp.MustCompile(`-[0-9a-f]{8}$`)
|
|
|
|
// demonicChapterRe matches the pre-redirect anchors demonic series pages link
|
|
// through. Both the raw "&" and the HTML-escaped "&" forms occur.
|
|
var demonicChapterRe = regexp.MustCompile(`chaptered\.php\?manga=\d+&(?:amp;)?chapter=([0-9.]+)`)
|
|
|
|
// comixSlugRe pulls the "<id>-<slug>" segment out of a stored series_url.
|
|
// Only the id prefix is stable; the slug tail follows the title.
|
|
var comixSlugRe = regexp.MustCompile(`/title/([^/?#]+)`)
|
|
|
|
func comixSeriesID(seriesURL string) (string, bool) {
|
|
m := comixSlugRe.FindStringSubmatch(seriesURL)
|
|
if m == nil {
|
|
return "", false
|
|
}
|
|
id := m[1]
|
|
if i := strings.Index(id, "-"); i != -1 {
|
|
id = id[:i]
|
|
}
|
|
return id, true
|
|
}
|
|
|
|
// kaganeChapterRe matches the chapter numbers in a kagane API response. This
|
|
// branch is fed by the browser fetcher, so the body is JSON rather than HTML —
|
|
// there are no anchors to scan.
|
|
var kaganeChapterRe = regexp.MustCompile(`"chapter_no":"([0-9.]+)"`)
|
|
|
|
// novelfullSlugRe pulls the series slug out of a stored series_url. novelfull
|
|
// series pages are "/<slug>.html"; their chapter anchors are
|
|
// "/<slug>/chapter-<n>[-<title-slug>].html". Verified live 2026-08-05.
|
|
var novelfullSlugRe = regexp.MustCompile(`^/([^/?#]+)\.html$`)
|
|
|
|
// lnwChapterRe matches any chapter-shaped address on lightnovelworld. Unlike
|
|
// asura, novelfull and comix — which scope to their stored series slug so a
|
|
// foreign chapter link cannot contribute — this Site's chapter addresses carry
|
|
// the Chapter Slug, which is not the Series identity: one Series may publish
|
|
// under several Chapter Slugs (measured 2026-08-11: a sampled novel serves
|
|
// 1-99 under one slug and 100-423 under another), so no stored-slug pattern can
|
|
// cover a Series' whole list. An unscoped match is safe because
|
|
// latestChapterFrom truncates the body at the comment thread before scanning
|
|
// (lnwCommentMarker); without that, a visitor's comment could set the Latest
|
|
// Chapter on the shared Series row.
|
|
var lnwChapterRe = regexp.MustCompile(`lightnovelworld\.net/[a-z0-9-]+-chapter-([0-9.]+)/`)
|
|
|
|
// lnwCommentMarker is the boundary of lightnovelworld's server-rendered
|
|
// wpdiscuz comment thread. It occurs exactly once per page and follows every
|
|
// chapter anchor (measured 2026-08-11,
|
|
// docs/research/lightnovelworld-chapter-vs-series-slug.md §6), so cutting the
|
|
// body at its first occurrence keeps the whole chapter list while excluding a
|
|
// region any visitor can write to. Absent means the page shape changed: the
|
|
// body is skipped, never scanned whole.
|
|
const lnwCommentMarker = "wpd-threads"
|
|
|
|
// latestChapterFrom returns the highest chapter number body advertises for this
|
|
// series. ok is false when the body yields nothing usable — an unknown site, an
|
|
// empty body, a Cloudflare challenge page, and a site redesign all land here,
|
|
// and the caller treats all four identically.
|
|
//
|
|
// Ported from the userscript's latestChapterFromAnchors (asura L123-133,
|
|
// demonic L183-193), including its reason for taking a maximum rather than a
|
|
// first or last: neither site lists chapters in a dependable order.
|
|
//
|
|
// The userscript's asura rule additionally requires the anchor text to match
|
|
// /Chapter\s+[\d.]+/i. That check exists only to skip the "First Chapter"
|
|
// shortcut, which points at chapter/1 and therefore can never win a maximum, so
|
|
// it is redundant here. For asura, scoping the pattern to this series' own slug
|
|
// replaces it with a stronger guarantee: a chapter link belonging to some other
|
|
// series cannot contribute even if the page starts carrying them. demonic has no
|
|
// such guarantee — demonicChapterRe matches any chaptered.php?manga=<id> anchor
|
|
// with no per-series scoping, because the stored series_id for demonic is a
|
|
// slug, not the numeric id the URL carries, so it cannot easily be scoped.
|
|
// lightnovelworld is unscoped and body-truncated instead — see lnwChapterRe.
|
|
func latestChapterFrom(site, seriesURL, body string) (latestChapter, bool) {
|
|
var re *regexp.Regexp
|
|
switch site {
|
|
case "asura":
|
|
m := asuraSlugRe.FindStringSubmatch(seriesURL)
|
|
if m == nil {
|
|
return latestChapter{}, false
|
|
}
|
|
// Stored URLs predating a redeploy may carry a stale build hash;
|
|
// chapter hrefs in the fetched body carry the current one. Strip to
|
|
// the stable ID and make the hash optional in the pattern, so scoping
|
|
// survives rotations.
|
|
slug := asuraBuildHash.ReplaceAllString(m[1], "")
|
|
// Compiled per call rather than cached: this runs once per fetch, which
|
|
// is at most a few times a minute, and the slug varies per series.
|
|
re = regexp.MustCompile(`/comics/` + regexp.QuoteMeta(slug) + `(?:-[0-9a-f]{8})?/chapter/([0-9.]+)`)
|
|
case "demonic":
|
|
re = demonicChapterRe
|
|
case "comix":
|
|
id, ok := comixSeriesID(seriesURL)
|
|
if !ok {
|
|
return latestChapter{}, false
|
|
}
|
|
// comix ships an SPA: the served HTML carries a JSON state blob instead
|
|
// of chapter anchors, and latestChapterUrl is the only place the newest
|
|
// chapter appears. Scoping to this series' id prefix keeps a
|
|
// "recommended" strip's entries from winning the maximum.
|
|
re = regexp.MustCompile(`"latestChapterUrl":"/title/` + regexp.QuoteMeta(id) + `-[^"]*-chapter-([0-9.]+)"`)
|
|
case "kagane":
|
|
re = kaganeChapterRe
|
|
case "novelfull":
|
|
u, err := url.Parse(seriesURL)
|
|
if err != nil {
|
|
return latestChapter{}, false
|
|
}
|
|
m := novelfullSlugRe.FindStringSubmatch(u.Path)
|
|
if m == nil {
|
|
return latestChapter{}, false
|
|
}
|
|
// Scoped to this series' slug for the same reason asura is: page 1
|
|
// carries a "latest chapters" widget and a "you may also like" strip,
|
|
// and neither may contribute to the maximum.
|
|
re = regexp.MustCompile(`/` + regexp.QuoteMeta(m[1]) + `/chapter-([0-9.]+)`)
|
|
case "lightnovelworld":
|
|
// The comment thread below the chapter list is the one region of the
|
|
// page any visitor can write to, so the scan never reads past it (see
|
|
// lnwChapterRe). A body without the marker is skipped, never scanned
|
|
// whole — a redesign must degrade into staleness, not into a wrong
|
|
// shared value; the logged body length tells a markup change from a
|
|
// body the size cap cut short.
|
|
i := strings.Index(body, lnwCommentMarker)
|
|
if i < 0 {
|
|
log.Printf("latest poll %q: no %s marker in %d bytes", seriesURL, lnwCommentMarker, len(body))
|
|
return latestChapter{}, false
|
|
}
|
|
body = body[:i]
|
|
re = lnwChapterRe
|
|
default:
|
|
return latestChapter{}, false
|
|
}
|
|
|
|
var best latestChapter
|
|
found := false
|
|
for _, m := range re.FindAllStringSubmatch(body, -1) {
|
|
// [0-9.]+ can swallow a trailing separator, e.g. "chapter/12." in a
|
|
// sentence; ParseFloat would reject the whole match.
|
|
raw := strings.Trim(m[1], ".")
|
|
num, err := strconv.ParseFloat(raw, 64)
|
|
if err != nil {
|
|
continue
|
|
}
|
|
if !found || num > best.Num {
|
|
best = latestChapter{Num: num, Label: "Chapter " + raw}
|
|
found = true
|
|
}
|
|
}
|
|
return best, found
|
|
}
|
|
|
|
var metaTagRe = regexp.MustCompile(`(?is)<meta\b[^>]*>`)
|
|
var doubleQuotedMetaAttrRe = regexp.MustCompile(`(?is)([a-z][a-z0-9:_-]*)\s*=\s*"([^"]*)"`)
|
|
var singleQuotedMetaAttrRe = regexp.MustCompile(`(?is)([a-z][a-z0-9:_-]*)\s*=\s*'([^']*)'`)
|
|
|
|
// comix's server-rendered page embeds query data in this JSON script; parsing
|
|
// the target detail entry avoids matching posters from recommended results.
|
|
var comixInitialDataRe = regexp.MustCompile(`(?is)<script\b[^>]*\bid\s*=\s*["']initial-data["'][^>]*>(.*?)</script>`)
|
|
|
|
// kaganeImageURLRe matches the canonical compressed image route kagane's API
|
|
// publishes — the only cover URL form the extractor emits and the browser
|
|
// fetcher accepts. The URL is matched in full (scheme, host, id shape) rather
|
|
// than trusted: the value a fetcher is pointed at may have been client-
|
|
// supplied, and a headless browser is a strong SSRF primitive.
|
|
var kaganeImageURLRe = regexp.MustCompile(`^https://kagane\.to/api/v2/image/([0-9a-f-]{36})/compressed$`)
|
|
|
|
// browserOnlyCoverURL reports whether the browser sidecar is the only fetcher
|
|
// for cover bytes at imageURL. kagane's image route answers a plain fetch with
|
|
// a challenge and `cross-origin-resource-policy: same-origin`, so a TLS fetch
|
|
// would only ever retrieve a challenge page and must not be attempted
|
|
// (ADR-0007). This is the byte-fetch router's per-Site knowledge; it lives in
|
|
// the extraction module, which owns kagane's URL shapes.
|
|
func browserOnlyCoverURL(imageURL string) bool {
|
|
return kaganeImageURLRe.MatchString(imageURL)
|
|
}
|
|
|
|
// kagane's browser-fetched series response publishes cover image IDs under
|
|
// series_covers. The API's canonical compressed image route is the only URL
|
|
// form accepted by the store and browser fetcher; no rendition is guessed.
|
|
func kaganeCoverURL(body string) string {
|
|
var response struct {
|
|
SeriesCovers []struct {
|
|
ImageID string `json:"image_id"`
|
|
} `json:"series_covers"`
|
|
}
|
|
if err := json.Unmarshal([]byte(body), &response); err != nil {
|
|
return ""
|
|
}
|
|
for _, cover := range response.SeriesCovers {
|
|
// Validate the assembled URL against the same regex the browser
|
|
// fetcher enforces, so the extractor can never emit an address the
|
|
// fetch would refuse.
|
|
imageURL := "https://kagane.to/api/v2/image/" + cover.ImageID + "/compressed"
|
|
if kaganeImageURLRe.MatchString(imageURL) {
|
|
return imageURL
|
|
}
|
|
}
|
|
return ""
|
|
}
|
|
|
|
func comixCoverURL(seriesURL, body string) string {
|
|
id, ok := comixSeriesID(seriesURL)
|
|
if !ok {
|
|
return ""
|
|
}
|
|
data := comixInitialDataRe.FindStringSubmatch(body)
|
|
if data == nil {
|
|
return ""
|
|
}
|
|
var state struct {
|
|
Queries map[string]json.RawMessage `json:"queries"`
|
|
}
|
|
if err := json.Unmarshal([]byte(data[1]), &state); err != nil {
|
|
return ""
|
|
}
|
|
raw := state.Queries[`["manga","detail","`+id+`"]`]
|
|
if len(raw) == 0 {
|
|
return ""
|
|
}
|
|
var detail struct {
|
|
Poster struct {
|
|
Medium string `json:"medium"`
|
|
} `json:"poster"`
|
|
}
|
|
if err := json.Unmarshal(raw, &detail); err != nil {
|
|
return ""
|
|
}
|
|
return publishedCoverURL(detail.Poster.Medium)
|
|
}
|
|
|
|
// coverFrom reports false for unknown sites, challenge bodies, and pages with
|
|
// no usable cover. Metadata extraction keeps scanning after an empty match so
|
|
// a later published cover is not hidden by an empty tag.
|
|
func coverFrom(site, seriesURL, body string) (string, bool) {
|
|
var cover string
|
|
switch site {
|
|
case "asura", "demonic", "lightnovelworld":
|
|
cover = metaContent(body, "property", "og:image")
|
|
case "novelfull":
|
|
cover = metaContent(body, "name", "image")
|
|
case "comix":
|
|
cover = comixCoverURL(seriesURL, body)
|
|
case "kagane":
|
|
cover = kaganeCoverURL(body)
|
|
}
|
|
return cover, cover != ""
|
|
}
|
|
|
|
func metaContent(body, attrName, attrValue string) string {
|
|
for _, tag := range metaTagRe.FindAllString(body, -1) {
|
|
attrs := make(map[string]string)
|
|
for _, m := range doubleQuotedMetaAttrRe.FindAllStringSubmatch(tag, -1) {
|
|
attrs[strings.ToLower(m[1])] = m[2]
|
|
}
|
|
for _, m := range singleQuotedMetaAttrRe.FindAllStringSubmatch(tag, -1) {
|
|
attrs[strings.ToLower(m[1])] = m[2]
|
|
}
|
|
if strings.EqualFold(attrs[strings.ToLower(attrName)], attrValue) {
|
|
if cover := publishedCoverURL(attrs["content"]); cover != "" {
|
|
return cover
|
|
}
|
|
}
|
|
}
|
|
return ""
|
|
}
|
|
|
|
func publishedCoverURL(value string) string {
|
|
value = strings.TrimSpace(html.UnescapeString(value))
|
|
return strings.ReplaceAll(value, " ", "%20")
|
|
}
|