From d1801f4f2a2687915fbcb7d848625f36a59f8774 Mon Sep 17 00:00:00 2001 From: Sulthan Zaki Date: Mon, 10 Aug 2026 00:53:54 +0700 Subject: [PATCH] feat(latest): extract per-site covers (#58) --- backend/internal/latest/sites.go | 110 +++++++++++++++++++++++-- backend/internal/latest/sites_test.go | 113 ++++++++++++++++++++++++++ 2 files changed, 217 insertions(+), 6 deletions(-) diff --git a/backend/internal/latest/sites.go b/backend/internal/latest/sites.go index 52ebfa1..35e7293 100644 --- a/backend/internal/latest/sites.go +++ b/backend/internal/latest/sites.go @@ -1,6 +1,8 @@ package latest import ( + "encoding/json" + "html" "net/url" "regexp" "strconv" @@ -34,6 +36,18 @@ var demonicChapterRe = regexp.MustCompile(`chaptered\.php\?manga=\d+&(?:amp;)?ch // Only the id prefix is stable; the slug tail follows the title. var comixSlugRe = regexp.MustCompile(`/title/([^/?#]+)`) +func comixSeriesID(seriesURL string) (string, bool) { + m := comixSlugRe.FindStringSubmatch(seriesURL) + if m == nil { + return "", false + } + id := m[1] + if i := strings.Index(id, "-"); i != -1 { + id = id[:i] + } + return id, true +} + // kaganeChapterRe matches the chapter numbers in a kagane API response. This // branch is fed by the browser fetcher, so the body is JSON rather than HTML — // there are no anchors to scan. @@ -87,18 +101,14 @@ func latestChapterFrom(site, seriesURL, body string) (latestChapter, bool) { case "demonic": re = demonicChapterRe case "comix": - m := comixSlugRe.FindStringSubmatch(seriesURL) - if m == nil { + id, ok := comixSeriesID(seriesURL) + if !ok { return latestChapter{}, false } // comix ships an SPA: the served HTML carries a JSON state blob instead // of chapter anchors, and latestChapterUrl is the only place the newest // chapter appears. Scoping to this series' id prefix keeps a // "recommended" strip's entries from winning the maximum. - id := m[1] - if i := strings.Index(id, "-"); i != -1 { - id = id[:i] - } re = regexp.MustCompile(`"latestChapterUrl":"/title/` + regexp.QuoteMeta(id) + `-[^"]*-chapter-([0-9.]+)"`) case "kagane": re = kaganeChapterRe @@ -146,3 +156,91 @@ func latestChapterFrom(site, seriesURL, body string) (latestChapter, bool) { } return best, found } + +var metaTagRe = regexp.MustCompile(`(?is)]*>`) +var doubleQuotedMetaAttrRe = regexp.MustCompile(`(?is)([a-z][a-z0-9:_-]*)\s*=\s*"([^"]*)"`) +var singleQuotedMetaAttrRe = regexp.MustCompile(`(?is)([a-z][a-z0-9:_-]*)\s*=\s*'([^']*)'`) + +// comix's server-rendered page embeds query data in this JSON script; parsing +// the target detail entry avoids matching posters from recommended results. +var comixInitialDataRe = regexp.MustCompile(`(?is)]*\bid\s*=\s*["']initial-data["'][^>]*>(.*?)`) + +// kagane's browser-fetched series response carries this published cover URL. +// Decode the top-level field, not merely an image-shaped URL elsewhere in JSON. +func kaganeCoverURL(body string) string { + var response struct { + Cover string `json:"cover"` + } + if err := json.Unmarshal([]byte(body), &response); err != nil { + return "" + } + return publishedCoverURL(response.Cover) +} + +func comixCoverURL(seriesURL, body string) string { + id, ok := comixSeriesID(seriesURL) + if !ok { + return "" + } + data := comixInitialDataRe.FindStringSubmatch(body) + if data == nil { + return "" + } + var state struct { + Queries map[string]json.RawMessage `json:"queries"` + } + if err := json.Unmarshal([]byte(data[1]), &state); err != nil { + return "" + } + raw := state.Queries[`["manga","detail","`+id+`"]`] + if len(raw) == 0 { + return "" + } + var detail struct { + Poster struct { + Medium string `json:"medium"` + } `json:"poster"` + } + if err := json.Unmarshal(raw, &detail); err != nil { + return "" + } + return publishedCoverURL(detail.Poster.Medium) +} + +func coverFrom(site, seriesURL, body string) (string, bool) { + var cover string + switch site { + case "asura", "demonic", "lightnovelworld": + cover = metaContent(body, "property", "og:image") + case "novelfull": + cover = metaContent(body, "name", "image") + case "comix": + cover = comixCoverURL(seriesURL, body) + case "kagane": + cover = kaganeCoverURL(body) + } + return cover, cover != "" +} + +func metaContent(body, attrName, attrValue string) string { + for _, tag := range metaTagRe.FindAllString(body, -1) { + attrs := make(map[string]string) + for _, m := range doubleQuotedMetaAttrRe.FindAllStringSubmatch(tag, -1) { + attrs[strings.ToLower(m[1])] = m[2] + } + for _, m := range singleQuotedMetaAttrRe.FindAllStringSubmatch(tag, -1) { + attrs[strings.ToLower(m[1])] = m[2] + } + if strings.EqualFold(attrs[strings.ToLower(attrName)], attrValue) { + if cover := publishedCoverURL(attrs["content"]); cover != "" { + return cover + } + } + } + return "" +} + +func publishedCoverURL(value string) string { + value = strings.TrimSpace(html.UnescapeString(value)) + return strings.ReplaceAll(value, " ", "%20") +} diff --git a/backend/internal/latest/sites_test.go b/backend/internal/latest/sites_test.go index 49423ad..a82eac9 100644 --- a/backend/internal/latest/sites_test.go +++ b/backend/internal/latest/sites_test.go @@ -79,6 +79,119 @@ const lnwSeriesFixture = ` Chapter 9999 ` +// Trimmed from https://asurascans.com/comics/chronicles-of-the-demon-faction-f886a8af +// (redirected to ...-00dcbf97) on 2026-08-10. +const asuraCoverFixture = `` + +// Trimmed from https://demonicscans.org/manga/Catastrophic-Necromancer on 2026-08-10. +// The source publishes the raw space in this URL. +const demonicCoverFixture = `` + +// Trimmed from https://comix.to/title/n8we-dungeons-and-crayons on 2026-08-10. +// The state includes a recommended poster before the target detail object and +// nested IDs inside that object; no og:image is present. +const comixCoverFixture = `` + +// Trimmed from GET https://kagane.to/api/v2/series/019f84bc-9ba0-7ed9-86f5-8b905ec7c28b on 2026-08-09. +const kaganeCoverFixture = `{"books":[{"cover":"https://kagane.to/api/v2/image/00000000-0000-0000-0000-000000000000/compressed"}],"cover":"https://kagane.to/api/v2/image/019f84bc-9ba0-7ed9-86f5-8b905ec7c28b/compressed"}` + +// Trimmed from https://novelfull.com/reverend-insanity.html on 2026-08-10. +const novelfullCoverFixture = `` + +// Trimmed from https://lightnovelworld.net/novel/a-will-eternal/ on 2026-08-10. +const lnwCoverFixture = `` + +func TestCoverFrom(t *testing.T) { + const comixURL = "https://comix.to/title/n8we-dungeons-and-crayons" + tests := []struct { + name string + site string + seriesURL string + body string + wantOK bool + wantCover string + }{ + { + name: "asura uses published metadata URL", + site: "asura", body: asuraCoverFixture, wantOK: true, + wantCover: "https://cdn.asurascans.com/asura-images/covers/chronicles-of-the-demon-faction.d4dcb8.webp", + }, + { + name: "demonic escapes raw spaces", + site: "demonic", body: demonicCoverFixture, wantOK: true, + wantCover: "https://readermc.org/images/thumbnails/Catastrophic%20Necromancer.webp", + }, + { + name: "comix takes target medium poster", + site: "comix", seriesURL: comixURL, body: comixCoverFixture, wantOK: true, + wantCover: "https://static.comix.to/039d/i/1/34/6a6742bf15736@280.jpg", + }, + { + name: "kagane reads API cover field", + site: "kagane", body: kaganeCoverFixture, wantOK: true, + wantCover: "https://kagane.to/api/v2/image/019f84bc-9ba0-7ed9-86f5-8b905ec7c28b/compressed", + }, + { + name: "novelfull reads image metadata", + site: "novelfull", body: novelfullCoverFixture, wantOK: true, + wantCover: "https://novelfull.com/uploads/webp/novel/reverend-insanity-82661d911a.webp", + }, + { + name: "lightnovelworld reads og image", + site: "lightnovelworld", body: lnwCoverFixture, wantOK: true, + wantCover: "https://lightnovelworld.net/wp-content/uploads/2026/03/a-will-eternal-1.webp", + }, + { + name: "later metadata cover survives empty match", + site: "asura", + body: `` + asuraCoverFixture, + wantOK: true, + wantCover: "https://cdn.asurascans.com/asura-images/covers/chronicles-of-the-demon-faction.d4dcb8.webp", + }, + { + name: "page without cover is empty", + site: "asura", body: ``, + }, + { + name: "unknown site is empty", + site: "unknown", body: asuraCoverFixture, + }, + } + + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + got, ok := coverFrom(tt.site, tt.seriesURL, tt.body) + if ok != tt.wantOK { + t.Fatalf("ok = %v, want %v (got %q)", ok, tt.wantOK, got) + } + if got != tt.wantCover { + t.Errorf("cover = %q, want %q", got, tt.wantCover) + } + }) + } +} + +func TestCoverFromChallenge(t *testing.T) { + tests := []struct { + site string + seriesURL string + }{ + {"asura", "https://asurascans.com/comics/chronicles-of-the-demon-faction-f886a8af"}, + {"demonic", "https://demonicscans.org/manga/Catastrophic-Necromancer"}, + {"comix", "https://comix.to/title/n8we-dungeons-and-crayons"}, + {"kagane", "https://kagane.to/series/019f84bc-9ba0-7ed9-86f5-8b905ec7c28b"}, + {"novelfull", "https://novelfull.com/reverend-insanity.html"}, + {"lightnovelworld", "https://lightnovelworld.net/novel/a-will-eternal/"}, + } + for _, tt := range tests { + t.Run(tt.site, func(t *testing.T) { + if got, ok := coverFrom(tt.site, tt.seriesURL, challengeFixture); ok || got != "" { + t.Fatalf("cover = %q, ok = %v, want empty", got, ok) + } + }) + } +} + func TestLatestChapterFrom(t *testing.T) { const asuraURL = "https://asurascans.com/comics/chronicles-of-the-demon-faction-f886a8af" const demonicURL = "https://demonicscans.org/manga/Catastrophic-Necromancer"