package latest import ( "regexp" "strconv" "strings" "bookmarkmanager/backend/internal/store" ) // latestChapter is the newest chapter a series page advertises. type latestChapter struct { Num float64 Label string } // asuraSlugRe pulls the series slug out of a stored series_url. // Shape verified live 2026-07-26: https://asurascans.com/comics/, where // the slug carries a trailing build-hash suffix (e.g. "-f886a8af") that // rotates on every site redeploy — callers must strip it (asuraBuildHash) // before using the slug to scope anything. var asuraSlugRe = regexp.MustCompile(`/comics/([^/?#]+)`) // demonicChapterRe matches the pre-redirect anchors demonic series pages link // through. Both the raw "&" and the HTML-escaped "&" forms occur. var demonicChapterRe = regexp.MustCompile(`chaptered\.php\?manga=\d+&(?:amp;)?chapter=([0-9.]+)`) // comixSlugRe pulls the "-" segment out of a stored series_url. // Only the id prefix is stable; the slug tail follows the title. var comixSlugRe = regexp.MustCompile(`/title/([^/?#]+)`) // kaganeChapterRe matches the chapter numbers in a kagane API response. This // branch is fed by the browser fetcher, so the body is JSON rather than HTML — // there are no anchors to scan. var kaganeChapterRe = regexp.MustCompile(`"chapter_no":"([0-9.]+)"`) // latestChapterFrom returns the highest chapter number body advertises for this // series. ok is false when the body yields nothing usable — an unknown site, an // empty body, a Cloudflare challenge page, and a site redesign all land here, // and the caller treats all four identically. // // Ported from the userscript's latestChapterFromAnchors (asura L123-133, // demonic L183-193), including its reason for taking a maximum rather than a // first or last: neither site lists chapters in a dependable order. // // The userscript's asura rule additionally requires the anchor text to match // /Chapter\s+[\d.]+/i. That check exists only to skip the "First Chapter" // shortcut, which points at chapter/1 and therefore can never win a maximum, so // it is redundant here. For asura, scoping the pattern to this series' own slug // replaces it with a stronger guarantee: a chapter link belonging to some other // series cannot contribute even if the page starts carrying them. demonic has no // such guarantee — demonicChapterRe matches any chaptered.php?manga= anchor // with no per-series scoping, because the stored series_id for demonic is a // slug, not the numeric id the URL carries, so it cannot easily be scoped. func latestChapterFrom(site, seriesURL, body string) (latestChapter, bool) { var re *regexp.Regexp switch site { case "asura": m := asuraSlugRe.FindStringSubmatch(seriesURL) if m == nil { return latestChapter{}, false } // Stored URLs predating a redeploy may carry a stale build hash; // chapter hrefs in the fetched body carry the current one. Strip to // the stable ID (same rule as migrateAsuraKeys) and make the hash // optional in the pattern, so scoping survives rotations. slug := store.AsuraBuildHash.ReplaceAllString(m[1], "") // Compiled per call rather than cached: this runs once per fetch, which // is at most a few times a minute, and the slug varies per series. re = regexp.MustCompile(`/comics/` + regexp.QuoteMeta(slug) + `(?:-[0-9a-f]{8})?/chapter/([0-9.]+)`) case "demonic": re = demonicChapterRe case "comix": m := comixSlugRe.FindStringSubmatch(seriesURL) if m == nil { return latestChapter{}, false } // comix ships an SPA: the served HTML carries a JSON state blob instead // of chapter anchors, and latestChapterUrl is the only place the newest // chapter appears. Scoping to this series' id prefix keeps a // "recommended" strip's entries from winning the maximum. id := m[1] if i := strings.Index(id, "-"); i != -1 { id = id[:i] } re = regexp.MustCompile(`"latestChapterUrl":"/title/` + regexp.QuoteMeta(id) + `-[^"]*-chapter-([0-9.]+)"`) case "kagane": re = kaganeChapterRe default: return latestChapter{}, false } var best latestChapter found := false for _, m := range re.FindAllStringSubmatch(body, -1) { // [0-9.]+ can swallow a trailing separator, e.g. "chapter/12." in a // sentence; ParseFloat would reject the whole match. raw := strings.Trim(m[1], ".") num, err := strconv.ParseFloat(raw, 64) if err != nil { continue } if !found || num > best.Num { best = latestChapter{Num: num, Label: "Chapter " + raw} found = true } } return best, found }