package latest import ( "encoding/json" "html" "net/url" "regexp" "strconv" "strings" ) // latestChapter is the newest chapter a series page advertises. type latestChapter struct { Num float64 Label string } // asuraSlugRe pulls the series slug out of a stored series_url. // Shape verified live 2026-07-26: https://asurascans.com/comics/, where // the slug carries a trailing build-hash suffix (e.g. "-f886a8af") that // rotates on every site redeploy — callers must strip it (asuraBuildHash) // before using the slug to scope anything. var asuraSlugRe = regexp.MustCompile(`/comics/([^/?#]+)`) // asuraBuildHash matches the trailing "-xxxxxxxx" site-wide build ID Asura // appends to every series slug. It rotates on each site redeploy, so it is // never part of a stable series_id. Must stay in sync with stripBuildHash in // userscript/manga-bookmark.user.js. var asuraBuildHash = regexp.MustCompile(`-[0-9a-f]{8}$`) // demonicChapterRe matches the pre-redirect anchors demonic series pages link // through. Both the raw "&" and the HTML-escaped "&" forms occur. var demonicChapterRe = regexp.MustCompile(`chaptered\.php\?manga=\d+&(?:amp;)?chapter=([0-9.]+)`) // comixSlugRe pulls the "-" segment out of a stored series_url. // Only the id prefix is stable; the slug tail follows the title. var comixSlugRe = regexp.MustCompile(`/title/([^/?#]+)`) func comixSeriesID(seriesURL string) (string, bool) { m := comixSlugRe.FindStringSubmatch(seriesURL) if m == nil { return "", false } id := m[1] if i := strings.Index(id, "-"); i != -1 { id = id[:i] } return id, true } // kaganeChapterRe matches the chapter numbers in a kagane API response. This // branch is fed by the browser fetcher, so the body is JSON rather than HTML — // there are no anchors to scan. var kaganeChapterRe = regexp.MustCompile(`"chapter_no":"([0-9.]+)"`) // novelfullSlugRe pulls the series slug out of a stored series_url. novelfull // series pages are "/.html"; their chapter anchors are // "//chapter-[-].html". Verified live 2026-08-05. var novelfullSlugRe = regexp.MustCompile(`^/([^/?#]+)\.html$`) // lnwSlugRe does the same for lightnovelworld, whose series pages live under // /novel// while its chapter URLs are flat at the site root: // "/-chapter-/", absolute in the page's own anchors. Verified live // 2026-08-05. var lnwSlugRe = regexp.MustCompile(`^/novel/([^/?#]+)/?$`) // latestChapterFrom returns the highest chapter number body advertises for this // series. ok is false when the body yields nothing usable — an unknown site, an // empty body, a Cloudflare challenge page, and a site redesign all land here, // and the caller treats all four identically. // // Ported from the userscript's latestChapterFromAnchors (asura L123-133, // demonic L183-193), including its reason for taking a maximum rather than a // first or last: neither site lists chapters in a dependable order. // // The userscript's asura rule additionally requires the anchor text to match // /Chapter\s+[\d.]+/i. That check exists only to skip the "First Chapter" // shortcut, which points at chapter/1 and therefore can never win a maximum, so // it is redundant here. For asura, scoping the pattern to this series' own slug // replaces it with a stronger guarantee: a chapter link belonging to some other // series cannot contribute even if the page starts carrying them. demonic has no // such guarantee — demonicChapterRe matches any chaptered.php?manga= anchor // with no per-series scoping, because the stored series_id for demonic is a // slug, not the numeric id the URL carries, so it cannot easily be scoped. func latestChapterFrom(site, seriesURL, body string) (latestChapter, bool) { var re *regexp.Regexp switch site { case "asura": m := asuraSlugRe.FindStringSubmatch(seriesURL) if m == nil { return latestChapter{}, false } // Stored URLs predating a redeploy may carry a stale build hash; // chapter hrefs in the fetched body carry the current one. Strip to // the stable ID and make the hash optional in the pattern, so scoping // survives rotations. slug := asuraBuildHash.ReplaceAllString(m[1], "") // Compiled per call rather than cached: this runs once per fetch, which // is at most a few times a minute, and the slug varies per series. re = regexp.MustCompile(`/comics/` + regexp.QuoteMeta(slug) + `(?:-[0-9a-f]{8})?/chapter/([0-9.]+)`) case "demonic": re = demonicChapterRe case "comix": id, ok := comixSeriesID(seriesURL) if !ok { return latestChapter{}, false } // comix ships an SPA: the served HTML carries a JSON state blob instead // of chapter anchors, and latestChapterUrl is the only place the newest // chapter appears. Scoping to this series' id prefix keeps a // "recommended" strip's entries from winning the maximum. re = regexp.MustCompile(`"latestChapterUrl":"/title/` + regexp.QuoteMeta(id) + `-[^"]*-chapter-([0-9.]+)"`) case "kagane": re = kaganeChapterRe case "novelfull": u, err := url.Parse(seriesURL) if err != nil { return latestChapter{}, false } m := novelfullSlugRe.FindStringSubmatch(u.Path) if m == nil { return latestChapter{}, false } // Scoped to this series' slug for the same reason asura is: page 1 // carries a "latest chapters" widget and a "you may also like" strip, // and neither may contribute to the maximum. re = regexp.MustCompile(`/` + regexp.QuoteMeta(m[1]) + `/chapter-([0-9.]+)`) case "lightnovelworld": u, err := url.Parse(seriesURL) if err != nil { return latestChapter{}, false } m := lnwSlugRe.FindStringSubmatch(u.Path) if m == nil { return latestChapter{}, false } re = regexp.MustCompile(`lightnovelworld\.net/` + regexp.QuoteMeta(m[1]) + `-chapter-([0-9.]+)/`) default: return latestChapter{}, false } var best latestChapter found := false for _, m := range re.FindAllStringSubmatch(body, -1) { // [0-9.]+ can swallow a trailing separator, e.g. "chapter/12." in a // sentence; ParseFloat would reject the whole match. raw := strings.Trim(m[1], ".") num, err := strconv.ParseFloat(raw, 64) if err != nil { continue } if !found || num > best.Num { best = latestChapter{Num: num, Label: "Chapter " + raw} found = true } } return best, found } var metaTagRe = regexp.MustCompile(`(?is)]*>`) var doubleQuotedMetaAttrRe = regexp.MustCompile(`(?is)([a-z][a-z0-9:_-]*)\s*=\s*"([^"]*)"`) var singleQuotedMetaAttrRe = regexp.MustCompile(`(?is)([a-z][a-z0-9:_-]*)\s*=\s*'([^']*)'`) // comix's server-rendered page embeds query data in this JSON script; parsing // the target detail entry avoids matching posters from recommended results. var comixInitialDataRe = regexp.MustCompile(`(?is)]*\bid\s*=\s*["']initial-data["'][^>]*>(.*?)`) // kaganeImageURLRe matches the canonical compressed image route kagane's API // publishes — the only cover URL form the extractor emits and the browser // fetcher accepts. The URL is matched in full (scheme, host, id shape) rather // than trusted: the value a fetcher is pointed at may have been client- // supplied, and a headless browser is a strong SSRF primitive. var kaganeImageURLRe = regexp.MustCompile(`^https://kagane\.to/api/v2/image/([0-9a-f-]{36})/compressed$`) // browserOnlyCoverURL reports whether the browser sidecar is the only fetcher // for cover bytes at imageURL. kagane's image route answers a plain fetch with // a challenge and `cross-origin-resource-policy: same-origin`, so a TLS fetch // would only ever retrieve a challenge page and must not be attempted // (ADR-0007). This is the byte-fetch router's per-Site knowledge; it lives in // the extraction module, which owns kagane's URL shapes. func browserOnlyCoverURL(imageURL string) bool { return kaganeImageURLRe.MatchString(imageURL) } // kagane's browser-fetched series response publishes cover image IDs under // series_covers. The API's canonical compressed image route is the only URL // form accepted by the store and browser fetcher; no rendition is guessed. func kaganeCoverURL(body string) string { var response struct { SeriesCovers []struct { ImageID string `json:"image_id"` } `json:"series_covers"` } if err := json.Unmarshal([]byte(body), &response); err != nil { return "" } for _, cover := range response.SeriesCovers { // Validate the assembled URL against the same regex the browser // fetcher enforces, so the extractor can never emit an address the // fetch would refuse. imageURL := "https://kagane.to/api/v2/image/" + cover.ImageID + "/compressed" if kaganeImageURLRe.MatchString(imageURL) { return imageURL } } return "" } func comixCoverURL(seriesURL, body string) string { id, ok := comixSeriesID(seriesURL) if !ok { return "" } data := comixInitialDataRe.FindStringSubmatch(body) if data == nil { return "" } var state struct { Queries map[string]json.RawMessage `json:"queries"` } if err := json.Unmarshal([]byte(data[1]), &state); err != nil { return "" } raw := state.Queries[`["manga","detail","`+id+`"]`] if len(raw) == 0 { return "" } var detail struct { Poster struct { Medium string `json:"medium"` } `json:"poster"` } if err := json.Unmarshal(raw, &detail); err != nil { return "" } return publishedCoverURL(detail.Poster.Medium) } // coverFrom reports false for unknown sites, challenge bodies, and pages with // no usable cover. Metadata extraction keeps scanning after an empty match so // a later published cover is not hidden by an empty tag. func coverFrom(site, seriesURL, body string) (string, bool) { var cover string switch site { case "asura", "demonic", "lightnovelworld": cover = metaContent(body, "property", "og:image") case "novelfull": cover = metaContent(body, "name", "image") case "comix": cover = comixCoverURL(seriesURL, body) case "kagane": cover = kaganeCoverURL(body) } return cover, cover != "" } func metaContent(body, attrName, attrValue string) string { for _, tag := range metaTagRe.FindAllString(body, -1) { attrs := make(map[string]string) for _, m := range doubleQuotedMetaAttrRe.FindAllStringSubmatch(tag, -1) { attrs[strings.ToLower(m[1])] = m[2] } for _, m := range singleQuotedMetaAttrRe.FindAllStringSubmatch(tag, -1) { attrs[strings.ToLower(m[1])] = m[2] } if strings.EqualFold(attrs[strings.ToLower(attrName)], attrValue) { if cover := publishedCoverURL(attrs["content"]); cover != "" { return cover } } } return "" } func publishedCoverURL(value string) string { value = strings.TrimSpace(html.UnescapeString(value)) return strings.ReplaceAll(value, " ", "%20") }