0a79e5f3d7
- Move the kagane cover URL shape into the extraction module (sites.go):
browserOnlyCoverURL + kaganeImageURLRe now own the claim; the byte-fetch
router and BrowserFetcher.Image reference it. One shape gate for producer
and fetcher (the id regex is folded into the full-URL match), so no Site
name appears in a cover path outside the extraction module and the
producer cannot emit an address the fetch would refuse.
- Restore the serving-boundary guarantee: GET /covers/{addr} re-checks the
stored media type via store.CoverContentType and 404s a poisoned row;
TestPublicCoverNeverEchoesNonImage now seeds one directly behind the
write gate and pins the refusal where bytes leave.
- Restore the SSRF rationale (client-supplied stored URL, headless browser
as a strong primitive) on the URL regex.
279 lines
10 KiB
Go
279 lines
10 KiB
Go
package latest
|
|
|
|
import (
|
|
"encoding/json"
|
|
"html"
|
|
"net/url"
|
|
"regexp"
|
|
"strconv"
|
|
"strings"
|
|
)
|
|
|
|
// latestChapter is the newest chapter a series page advertises.
|
|
type latestChapter struct {
|
|
Num float64
|
|
Label string
|
|
}
|
|
|
|
// asuraSlugRe pulls the series slug out of a stored series_url.
|
|
// Shape verified live 2026-07-26: https://asurascans.com/comics/<slug>, where
|
|
// the slug carries a trailing build-hash suffix (e.g. "-f886a8af") that
|
|
// rotates on every site redeploy — callers must strip it (asuraBuildHash)
|
|
// before using the slug to scope anything.
|
|
var asuraSlugRe = regexp.MustCompile(`/comics/([^/?#]+)`)
|
|
|
|
// asuraBuildHash matches the trailing "-xxxxxxxx" site-wide build ID Asura
|
|
// appends to every series slug. It rotates on each site redeploy, so it is
|
|
// never part of a stable series_id. Must stay in sync with stripBuildHash in
|
|
// userscript/manga-bookmark.user.js.
|
|
var asuraBuildHash = regexp.MustCompile(`-[0-9a-f]{8}$`)
|
|
|
|
// demonicChapterRe matches the pre-redirect anchors demonic series pages link
|
|
// through. Both the raw "&" and the HTML-escaped "&" forms occur.
|
|
var demonicChapterRe = regexp.MustCompile(`chaptered\.php\?manga=\d+&(?:amp;)?chapter=([0-9.]+)`)
|
|
|
|
// comixSlugRe pulls the "<id>-<slug>" segment out of a stored series_url.
|
|
// Only the id prefix is stable; the slug tail follows the title.
|
|
var comixSlugRe = regexp.MustCompile(`/title/([^/?#]+)`)
|
|
|
|
func comixSeriesID(seriesURL string) (string, bool) {
|
|
m := comixSlugRe.FindStringSubmatch(seriesURL)
|
|
if m == nil {
|
|
return "", false
|
|
}
|
|
id := m[1]
|
|
if i := strings.Index(id, "-"); i != -1 {
|
|
id = id[:i]
|
|
}
|
|
return id, true
|
|
}
|
|
|
|
// kaganeChapterRe matches the chapter numbers in a kagane API response. This
|
|
// branch is fed by the browser fetcher, so the body is JSON rather than HTML —
|
|
// there are no anchors to scan.
|
|
var kaganeChapterRe = regexp.MustCompile(`"chapter_no":"([0-9.]+)"`)
|
|
|
|
// novelfullSlugRe pulls the series slug out of a stored series_url. novelfull
|
|
// series pages are "/<slug>.html"; their chapter anchors are
|
|
// "/<slug>/chapter-<n>[-<title-slug>].html". Verified live 2026-08-05.
|
|
var novelfullSlugRe = regexp.MustCompile(`^/([^/?#]+)\.html$`)
|
|
|
|
// lnwSlugRe does the same for lightnovelworld, whose series pages live under
|
|
// /novel/<slug>/ while its chapter URLs are flat at the site root:
|
|
// "/<slug>-chapter-<n>/", absolute in the page's own anchors. Verified live
|
|
// 2026-08-05.
|
|
var lnwSlugRe = regexp.MustCompile(`^/novel/([^/?#]+)/?$`)
|
|
|
|
// latestChapterFrom returns the highest chapter number body advertises for this
|
|
// series. ok is false when the body yields nothing usable — an unknown site, an
|
|
// empty body, a Cloudflare challenge page, and a site redesign all land here,
|
|
// and the caller treats all four identically.
|
|
//
|
|
// Ported from the userscript's latestChapterFromAnchors (asura L123-133,
|
|
// demonic L183-193), including its reason for taking a maximum rather than a
|
|
// first or last: neither site lists chapters in a dependable order.
|
|
//
|
|
// The userscript's asura rule additionally requires the anchor text to match
|
|
// /Chapter\s+[\d.]+/i. That check exists only to skip the "First Chapter"
|
|
// shortcut, which points at chapter/1 and therefore can never win a maximum, so
|
|
// it is redundant here. For asura, scoping the pattern to this series' own slug
|
|
// replaces it with a stronger guarantee: a chapter link belonging to some other
|
|
// series cannot contribute even if the page starts carrying them. demonic has no
|
|
// such guarantee — demonicChapterRe matches any chaptered.php?manga=<id> anchor
|
|
// with no per-series scoping, because the stored series_id for demonic is a
|
|
// slug, not the numeric id the URL carries, so it cannot easily be scoped.
|
|
func latestChapterFrom(site, seriesURL, body string) (latestChapter, bool) {
|
|
var re *regexp.Regexp
|
|
switch site {
|
|
case "asura":
|
|
m := asuraSlugRe.FindStringSubmatch(seriesURL)
|
|
if m == nil {
|
|
return latestChapter{}, false
|
|
}
|
|
// Stored URLs predating a redeploy may carry a stale build hash;
|
|
// chapter hrefs in the fetched body carry the current one. Strip to
|
|
// the stable ID and make the hash optional in the pattern, so scoping
|
|
// survives rotations.
|
|
slug := asuraBuildHash.ReplaceAllString(m[1], "")
|
|
// Compiled per call rather than cached: this runs once per fetch, which
|
|
// is at most a few times a minute, and the slug varies per series.
|
|
re = regexp.MustCompile(`/comics/` + regexp.QuoteMeta(slug) + `(?:-[0-9a-f]{8})?/chapter/([0-9.]+)`)
|
|
case "demonic":
|
|
re = demonicChapterRe
|
|
case "comix":
|
|
id, ok := comixSeriesID(seriesURL)
|
|
if !ok {
|
|
return latestChapter{}, false
|
|
}
|
|
// comix ships an SPA: the served HTML carries a JSON state blob instead
|
|
// of chapter anchors, and latestChapterUrl is the only place the newest
|
|
// chapter appears. Scoping to this series' id prefix keeps a
|
|
// "recommended" strip's entries from winning the maximum.
|
|
re = regexp.MustCompile(`"latestChapterUrl":"/title/` + regexp.QuoteMeta(id) + `-[^"]*-chapter-([0-9.]+)"`)
|
|
case "kagane":
|
|
re = kaganeChapterRe
|
|
case "novelfull":
|
|
u, err := url.Parse(seriesURL)
|
|
if err != nil {
|
|
return latestChapter{}, false
|
|
}
|
|
m := novelfullSlugRe.FindStringSubmatch(u.Path)
|
|
if m == nil {
|
|
return latestChapter{}, false
|
|
}
|
|
// Scoped to this series' slug for the same reason asura is: page 1
|
|
// carries a "latest chapters" widget and a "you may also like" strip,
|
|
// and neither may contribute to the maximum.
|
|
re = regexp.MustCompile(`/` + regexp.QuoteMeta(m[1]) + `/chapter-([0-9.]+)`)
|
|
case "lightnovelworld":
|
|
u, err := url.Parse(seriesURL)
|
|
if err != nil {
|
|
return latestChapter{}, false
|
|
}
|
|
m := lnwSlugRe.FindStringSubmatch(u.Path)
|
|
if m == nil {
|
|
return latestChapter{}, false
|
|
}
|
|
re = regexp.MustCompile(`lightnovelworld\.net/` + regexp.QuoteMeta(m[1]) + `-chapter-([0-9.]+)/`)
|
|
default:
|
|
return latestChapter{}, false
|
|
}
|
|
|
|
var best latestChapter
|
|
found := false
|
|
for _, m := range re.FindAllStringSubmatch(body, -1) {
|
|
// [0-9.]+ can swallow a trailing separator, e.g. "chapter/12." in a
|
|
// sentence; ParseFloat would reject the whole match.
|
|
raw := strings.Trim(m[1], ".")
|
|
num, err := strconv.ParseFloat(raw, 64)
|
|
if err != nil {
|
|
continue
|
|
}
|
|
if !found || num > best.Num {
|
|
best = latestChapter{Num: num, Label: "Chapter " + raw}
|
|
found = true
|
|
}
|
|
}
|
|
return best, found
|
|
}
|
|
|
|
var metaTagRe = regexp.MustCompile(`(?is)<meta\b[^>]*>`)
|
|
var doubleQuotedMetaAttrRe = regexp.MustCompile(`(?is)([a-z][a-z0-9:_-]*)\s*=\s*"([^"]*)"`)
|
|
var singleQuotedMetaAttrRe = regexp.MustCompile(`(?is)([a-z][a-z0-9:_-]*)\s*=\s*'([^']*)'`)
|
|
|
|
// comix's server-rendered page embeds query data in this JSON script; parsing
|
|
// the target detail entry avoids matching posters from recommended results.
|
|
var comixInitialDataRe = regexp.MustCompile(`(?is)<script\b[^>]*\bid\s*=\s*["']initial-data["'][^>]*>(.*?)</script>`)
|
|
|
|
// kaganeImageURLRe matches the canonical compressed image route kagane's API
|
|
// publishes — the only cover URL form the extractor emits and the browser
|
|
// fetcher accepts. The URL is matched in full (scheme, host, id shape) rather
|
|
// than trusted: the value a fetcher is pointed at may have been client-
|
|
// supplied, and a headless browser is a strong SSRF primitive.
|
|
var kaganeImageURLRe = regexp.MustCompile(`^https://kagane\.to/api/v2/image/([0-9a-f-]{36})/compressed$`)
|
|
|
|
// browserOnlyCoverURL reports whether the browser sidecar is the only fetcher
|
|
// for cover bytes at imageURL. kagane's image route answers a plain fetch with
|
|
// a challenge and `cross-origin-resource-policy: same-origin`, so a TLS fetch
|
|
// would only ever retrieve a challenge page and must not be attempted
|
|
// (ADR-0007). This is the byte-fetch router's per-Site knowledge; it lives in
|
|
// the extraction module, which owns kagane's URL shapes.
|
|
func browserOnlyCoverURL(imageURL string) bool {
|
|
return kaganeImageURLRe.MatchString(imageURL)
|
|
}
|
|
|
|
// kagane's browser-fetched series response publishes cover image IDs under
|
|
// series_covers. The API's canonical compressed image route is the only URL
|
|
// form accepted by the store and browser fetcher; no rendition is guessed.
|
|
func kaganeCoverURL(body string) string {
|
|
var response struct {
|
|
SeriesCovers []struct {
|
|
ImageID string `json:"image_id"`
|
|
} `json:"series_covers"`
|
|
}
|
|
if err := json.Unmarshal([]byte(body), &response); err != nil {
|
|
return ""
|
|
}
|
|
for _, cover := range response.SeriesCovers {
|
|
// Validate the assembled URL against the same regex the browser
|
|
// fetcher enforces, so the extractor can never emit an address the
|
|
// fetch would refuse.
|
|
imageURL := "https://kagane.to/api/v2/image/" + cover.ImageID + "/compressed"
|
|
if kaganeImageURLRe.MatchString(imageURL) {
|
|
return imageURL
|
|
}
|
|
}
|
|
return ""
|
|
}
|
|
|
|
func comixCoverURL(seriesURL, body string) string {
|
|
id, ok := comixSeriesID(seriesURL)
|
|
if !ok {
|
|
return ""
|
|
}
|
|
data := comixInitialDataRe.FindStringSubmatch(body)
|
|
if data == nil {
|
|
return ""
|
|
}
|
|
var state struct {
|
|
Queries map[string]json.RawMessage `json:"queries"`
|
|
}
|
|
if err := json.Unmarshal([]byte(data[1]), &state); err != nil {
|
|
return ""
|
|
}
|
|
raw := state.Queries[`["manga","detail","`+id+`"]`]
|
|
if len(raw) == 0 {
|
|
return ""
|
|
}
|
|
var detail struct {
|
|
Poster struct {
|
|
Medium string `json:"medium"`
|
|
} `json:"poster"`
|
|
}
|
|
if err := json.Unmarshal(raw, &detail); err != nil {
|
|
return ""
|
|
}
|
|
return publishedCoverURL(detail.Poster.Medium)
|
|
}
|
|
|
|
// coverFrom reports false for unknown sites, challenge bodies, and pages with
|
|
// no usable cover. Metadata extraction keeps scanning after an empty match so
|
|
// a later published cover is not hidden by an empty tag.
|
|
func coverFrom(site, seriesURL, body string) (string, bool) {
|
|
var cover string
|
|
switch site {
|
|
case "asura", "demonic", "lightnovelworld":
|
|
cover = metaContent(body, "property", "og:image")
|
|
case "novelfull":
|
|
cover = metaContent(body, "name", "image")
|
|
case "comix":
|
|
cover = comixCoverURL(seriesURL, body)
|
|
case "kagane":
|
|
cover = kaganeCoverURL(body)
|
|
}
|
|
return cover, cover != ""
|
|
}
|
|
|
|
func metaContent(body, attrName, attrValue string) string {
|
|
for _, tag := range metaTagRe.FindAllString(body, -1) {
|
|
attrs := make(map[string]string)
|
|
for _, m := range doubleQuotedMetaAttrRe.FindAllStringSubmatch(tag, -1) {
|
|
attrs[strings.ToLower(m[1])] = m[2]
|
|
}
|
|
for _, m := range singleQuotedMetaAttrRe.FindAllStringSubmatch(tag, -1) {
|
|
attrs[strings.ToLower(m[1])] = m[2]
|
|
}
|
|
if strings.EqualFold(attrs[strings.ToLower(attrName)], attrValue) {
|
|
if cover := publishedCoverURL(attrs["content"]); cover != "" {
|
|
return cover
|
|
}
|
|
}
|
|
}
|
|
return ""
|
|
}
|
|
|
|
func publishedCoverURL(value string) string {
|
|
value = strings.TrimSpace(html.UnescapeString(value))
|
|
return strings.ReplaceAll(value, " ", "%20")
|
|
}
|