Files
mangaBookmark/backend/internal/latest/sites.go
T
sulthan 7c7d597019 Delete the kagane-specific cover path (#63) (#73)
Closes #63

Deletes the second way to reach a Cover. Since #62, every Site's cover bytes land in the content-addressed store at creation or on the poll, and the one public route serves them all — nothing needs the kagane proxy anymore.

## What went

- **Template-level rewrite:** `Bookmark.CoverURL()` and both templates' use of it. Cards and chrome now render `.Cover` — the wire value — and nothing else. `Bookmark.CoverSource` was dead once `CoverURL` went, so it and its `bookmarkColumns` entry are gone too.
- **Kagane-only cover route and its identifier validation:** `GET /img/kagane/{id}`, `web.CoverFetcher`, `coverIDRe`, and the whole `internal/web/cover.go`.
- **The proxy's persistence:** `store.KaganeImageID`, `GetKaganeCover`, `PutKaganeCover`, `kaganeCoverSourceURL`, `kaganeCoverRe`.
- **The kagane-shaped branch in the byte-fetch routing:** `fetchCoverBytes` no longer takes a `site` argument and no longer names a Site. The URL shape kagane's API publishes is claimed by the browser module itself — `kaganeImageURLRe` + `browserCoverURL` live in `latest/browser.go` with the rest of the per-Site knowledge — and `BrowserFetcher.Image` is now URL-driven (it validates the URL it will navigate to, same SSRF discipline as before). The no-plain-TLS-fallback rule for a claimed URL is preserved: a claimed address with no browser is an error, never a challenge-page fetch.

## What stayed (deliberately)

- `BrowserFetcher.Image` and the browser-backed acquisition path: kagane genuinely serves cover bytes behind the challenge + `cross-origin-resource-policy: same-origin`, so the sidecar remains the only fetcher for them — it just routes by URL claim now instead of by Site name.
- `fetcherFor`'s per-Site page routing (kagane/novelfull page fetches) — that is the page path, not a cover path.

## Acceptance criteria

- [x] Template-level kagane cover rewrite gone
- [x] Kagane-only cover route and its identifier validation gone
- [x] Tests removed/rewritten against the general route, guarantees kept: unstored + traversal-shaped addresses serve nothing (`TestPublicCoverRejectsUnknownAddress`), non-image content types never echoed (`TestPublicCoverNeverEchoesNonImage` — new; the store-side gate was already pinned by `TestCoverStoreAcceptsAnySourceURL`). Store reopen-persistence and filesystem content-addressing tests rewritten against `PutCover`/`GetCover`, no guarantee lost.
- [x] No Site name in a cover code path outside the acquisition module (`grep kagane backend`: store/web/templates/api are clean; remaining hits are `latest/browser.go` + `latest/sites.go`, tests, docs)
- [x] Web UI and panel render Covers for all six Sites (templates render the wire address; panel renders `b.cover` — untouched, it never had a kagane path)
- [x] `go test ./...` green

## Verification

- `go vet ./...` clean
- `go test ./...` — all packages pass (root 16.9s, latest 12.7s, store 12.7s, web 0.004s)
- `CGO_ENABLED=0 go build` produces the static binary
- Cover-path tests run verbosely: `TestPublicCoverServesStoredBytesUnauthenticated`, `TestPublicCoverRejectsUnknownAddress` (unknown/malformed/traversal/empty), `TestPublicCoverNeverEchoesNonImage`, `TestListRendersAcquiredCover`, `TestAcquireKaganeCoverThroughBrowser`, `TestRunOncePrefetchesKaganeCover`, `TestRunOnceRoutesNonKaganeCoverToPublicFetcher` all pass; the three `SMOKE_*` tests skip without the browser sidecar, as designed

Live browser verification of the "web UI and panel render Covers for all six Sites" criterion is being run separately with Playwright against real Site pages and a locally mocked backend.

Reviewed-on: #73
Co-authored-by: Sulthan Zaki <sultankiki05@gmail.com>
Co-committed-by: Sulthan Zaki <sultankiki05@gmail.com>
2026-08-10 18:02:47 +07:00

279 lines
10 KiB
Go

package latest
import (
"encoding/json"
"html"
"net/url"
"regexp"
"strconv"
"strings"
)
// latestChapter is the newest chapter a series page advertises.
type latestChapter struct {
Num float64
Label string
}
// asuraSlugRe pulls the series slug out of a stored series_url.
// Shape verified live 2026-07-26: https://asurascans.com/comics/<slug>, where
// the slug carries a trailing build-hash suffix (e.g. "-f886a8af") that
// rotates on every site redeploy — callers must strip it (asuraBuildHash)
// before using the slug to scope anything.
var asuraSlugRe = regexp.MustCompile(`/comics/([^/?#]+)`)
// asuraBuildHash matches the trailing "-xxxxxxxx" site-wide build ID Asura
// appends to every series slug. It rotates on each site redeploy, so it is
// never part of a stable series_id. Must stay in sync with stripBuildHash in
// userscript/manga-bookmark.user.js.
var asuraBuildHash = regexp.MustCompile(`-[0-9a-f]{8}$`)
// demonicChapterRe matches the pre-redirect anchors demonic series pages link
// through. Both the raw "&" and the HTML-escaped "&amp;" forms occur.
var demonicChapterRe = regexp.MustCompile(`chaptered\.php\?manga=\d+&(?:amp;)?chapter=([0-9.]+)`)
// comixSlugRe pulls the "<id>-<slug>" segment out of a stored series_url.
// Only the id prefix is stable; the slug tail follows the title.
var comixSlugRe = regexp.MustCompile(`/title/([^/?#]+)`)
func comixSeriesID(seriesURL string) (string, bool) {
m := comixSlugRe.FindStringSubmatch(seriesURL)
if m == nil {
return "", false
}
id := m[1]
if i := strings.Index(id, "-"); i != -1 {
id = id[:i]
}
return id, true
}
// kaganeChapterRe matches the chapter numbers in a kagane API response. This
// branch is fed by the browser fetcher, so the body is JSON rather than HTML —
// there are no anchors to scan.
var kaganeChapterRe = regexp.MustCompile(`"chapter_no":"([0-9.]+)"`)
// novelfullSlugRe pulls the series slug out of a stored series_url. novelfull
// series pages are "/<slug>.html"; their chapter anchors are
// "/<slug>/chapter-<n>[-<title-slug>].html". Verified live 2026-08-05.
var novelfullSlugRe = regexp.MustCompile(`^/([^/?#]+)\.html$`)
// lnwSlugRe does the same for lightnovelworld, whose series pages live under
// /novel/<slug>/ while its chapter URLs are flat at the site root:
// "/<slug>-chapter-<n>/", absolute in the page's own anchors. Verified live
// 2026-08-05.
var lnwSlugRe = regexp.MustCompile(`^/novel/([^/?#]+)/?$`)
// latestChapterFrom returns the highest chapter number body advertises for this
// series. ok is false when the body yields nothing usable — an unknown site, an
// empty body, a Cloudflare challenge page, and a site redesign all land here,
// and the caller treats all four identically.
//
// Ported from the userscript's latestChapterFromAnchors (asura L123-133,
// demonic L183-193), including its reason for taking a maximum rather than a
// first or last: neither site lists chapters in a dependable order.
//
// The userscript's asura rule additionally requires the anchor text to match
// /Chapter\s+[\d.]+/i. That check exists only to skip the "First Chapter"
// shortcut, which points at chapter/1 and therefore can never win a maximum, so
// it is redundant here. For asura, scoping the pattern to this series' own slug
// replaces it with a stronger guarantee: a chapter link belonging to some other
// series cannot contribute even if the page starts carrying them. demonic has no
// such guarantee — demonicChapterRe matches any chaptered.php?manga=<id> anchor
// with no per-series scoping, because the stored series_id for demonic is a
// slug, not the numeric id the URL carries, so it cannot easily be scoped.
func latestChapterFrom(site, seriesURL, body string) (latestChapter, bool) {
var re *regexp.Regexp
switch site {
case "asura":
m := asuraSlugRe.FindStringSubmatch(seriesURL)
if m == nil {
return latestChapter{}, false
}
// Stored URLs predating a redeploy may carry a stale build hash;
// chapter hrefs in the fetched body carry the current one. Strip to
// the stable ID and make the hash optional in the pattern, so scoping
// survives rotations.
slug := asuraBuildHash.ReplaceAllString(m[1], "")
// Compiled per call rather than cached: this runs once per fetch, which
// is at most a few times a minute, and the slug varies per series.
re = regexp.MustCompile(`/comics/` + regexp.QuoteMeta(slug) + `(?:-[0-9a-f]{8})?/chapter/([0-9.]+)`)
case "demonic":
re = demonicChapterRe
case "comix":
id, ok := comixSeriesID(seriesURL)
if !ok {
return latestChapter{}, false
}
// comix ships an SPA: the served HTML carries a JSON state blob instead
// of chapter anchors, and latestChapterUrl is the only place the newest
// chapter appears. Scoping to this series' id prefix keeps a
// "recommended" strip's entries from winning the maximum.
re = regexp.MustCompile(`"latestChapterUrl":"/title/` + regexp.QuoteMeta(id) + `-[^"]*-chapter-([0-9.]+)"`)
case "kagane":
re = kaganeChapterRe
case "novelfull":
u, err := url.Parse(seriesURL)
if err != nil {
return latestChapter{}, false
}
m := novelfullSlugRe.FindStringSubmatch(u.Path)
if m == nil {
return latestChapter{}, false
}
// Scoped to this series' slug for the same reason asura is: page 1
// carries a "latest chapters" widget and a "you may also like" strip,
// and neither may contribute to the maximum.
re = regexp.MustCompile(`/` + regexp.QuoteMeta(m[1]) + `/chapter-([0-9.]+)`)
case "lightnovelworld":
u, err := url.Parse(seriesURL)
if err != nil {
return latestChapter{}, false
}
m := lnwSlugRe.FindStringSubmatch(u.Path)
if m == nil {
return latestChapter{}, false
}
re = regexp.MustCompile(`lightnovelworld\.net/` + regexp.QuoteMeta(m[1]) + `-chapter-([0-9.]+)/`)
default:
return latestChapter{}, false
}
var best latestChapter
found := false
for _, m := range re.FindAllStringSubmatch(body, -1) {
// [0-9.]+ can swallow a trailing separator, e.g. "chapter/12." in a
// sentence; ParseFloat would reject the whole match.
raw := strings.Trim(m[1], ".")
num, err := strconv.ParseFloat(raw, 64)
if err != nil {
continue
}
if !found || num > best.Num {
best = latestChapter{Num: num, Label: "Chapter " + raw}
found = true
}
}
return best, found
}
var metaTagRe = regexp.MustCompile(`(?is)<meta\b[^>]*>`)
var doubleQuotedMetaAttrRe = regexp.MustCompile(`(?is)([a-z][a-z0-9:_-]*)\s*=\s*"([^"]*)"`)
var singleQuotedMetaAttrRe = regexp.MustCompile(`(?is)([a-z][a-z0-9:_-]*)\s*=\s*'([^']*)'`)
// comix's server-rendered page embeds query data in this JSON script; parsing
// the target detail entry avoids matching posters from recommended results.
var comixInitialDataRe = regexp.MustCompile(`(?is)<script\b[^>]*\bid\s*=\s*["']initial-data["'][^>]*>(.*?)</script>`)
// kaganeImageURLRe matches the canonical compressed image route kagane's API
// publishes — the only cover URL form the extractor emits and the browser
// fetcher accepts. The URL is matched in full (scheme, host, id shape) rather
// than trusted: the value a fetcher is pointed at may have been client-
// supplied, and a headless browser is a strong SSRF primitive.
var kaganeImageURLRe = regexp.MustCompile(`^https://kagane\.to/api/v2/image/([0-9a-f-]{36})/compressed$`)
// browserOnlyCoverURL reports whether the browser sidecar is the only fetcher
// for cover bytes at imageURL. kagane's image route answers a plain fetch with
// a challenge and `cross-origin-resource-policy: same-origin`, so a TLS fetch
// would only ever retrieve a challenge page and must not be attempted
// (ADR-0007). This is the byte-fetch router's per-Site knowledge; it lives in
// the extraction module, which owns kagane's URL shapes.
func browserOnlyCoverURL(imageURL string) bool {
return kaganeImageURLRe.MatchString(imageURL)
}
// kagane's browser-fetched series response publishes cover image IDs under
// series_covers. The API's canonical compressed image route is the only URL
// form accepted by the store and browser fetcher; no rendition is guessed.
func kaganeCoverURL(body string) string {
var response struct {
SeriesCovers []struct {
ImageID string `json:"image_id"`
} `json:"series_covers"`
}
if err := json.Unmarshal([]byte(body), &response); err != nil {
return ""
}
for _, cover := range response.SeriesCovers {
// Validate the assembled URL against the same regex the browser
// fetcher enforces, so the extractor can never emit an address the
// fetch would refuse.
imageURL := "https://kagane.to/api/v2/image/" + cover.ImageID + "/compressed"
if kaganeImageURLRe.MatchString(imageURL) {
return imageURL
}
}
return ""
}
func comixCoverURL(seriesURL, body string) string {
id, ok := comixSeriesID(seriesURL)
if !ok {
return ""
}
data := comixInitialDataRe.FindStringSubmatch(body)
if data == nil {
return ""
}
var state struct {
Queries map[string]json.RawMessage `json:"queries"`
}
if err := json.Unmarshal([]byte(data[1]), &state); err != nil {
return ""
}
raw := state.Queries[`["manga","detail","`+id+`"]`]
if len(raw) == 0 {
return ""
}
var detail struct {
Poster struct {
Medium string `json:"medium"`
} `json:"poster"`
}
if err := json.Unmarshal(raw, &detail); err != nil {
return ""
}
return publishedCoverURL(detail.Poster.Medium)
}
// coverFrom reports false for unknown sites, challenge bodies, and pages with
// no usable cover. Metadata extraction keeps scanning after an empty match so
// a later published cover is not hidden by an empty tag.
func coverFrom(site, seriesURL, body string) (string, bool) {
var cover string
switch site {
case "asura", "demonic", "lightnovelworld":
cover = metaContent(body, "property", "og:image")
case "novelfull":
cover = metaContent(body, "name", "image")
case "comix":
cover = comixCoverURL(seriesURL, body)
case "kagane":
cover = kaganeCoverURL(body)
}
return cover, cover != ""
}
func metaContent(body, attrName, attrValue string) string {
for _, tag := range metaTagRe.FindAllString(body, -1) {
attrs := make(map[string]string)
for _, m := range doubleQuotedMetaAttrRe.FindAllStringSubmatch(tag, -1) {
attrs[strings.ToLower(m[1])] = m[2]
}
for _, m := range singleQuotedMetaAttrRe.FindAllStringSubmatch(tag, -1) {
attrs[strings.ToLower(m[1])] = m[2]
}
if strings.EqualFold(attrs[strings.ToLower(attrName)], attrValue) {
if cover := publishedCoverURL(attrs["content"]); cover != "" {
return cover
}
}
}
return ""
}
func publishedCoverURL(value string) string {
value = strings.TrimSpace(html.UnescapeString(value))
return strings.ReplaceAll(value, " ", "%20")
}