feat(latest): extract per-site covers (#58)

This commit is contained in:
2026-08-10 00:53:54 +07:00
parent e8d1cba6c5
commit ff84eec3a0
2 changed files with 178 additions and 0 deletions
+77
View File
@@ -1,6 +1,7 @@
package latest package latest
import ( import (
"html"
"net/url" "net/url"
"regexp" "regexp"
"strconv" "strconv"
@@ -146,3 +147,79 @@ func latestChapterFrom(site, seriesURL, body string) (latestChapter, bool) {
} }
return best, found return best, found
} }
var metaTagRe = regexp.MustCompile(`(?is)<meta\b[^>]*>`)
var doubleQuotedMetaAttrRe = regexp.MustCompile(`(?is)([a-z][a-z0-9:_-]*)\s*=\s*"([^"]*)"`)
var singleQuotedMetaAttrRe = regexp.MustCompile(`(?is)([a-z][a-z0-9:_-]*)\s*=\s*'([^']*)'`)
// comix stores its cover in the target series object in the page state rather
// than an HTML metadata tag. The medium rendition is deliberately selected
// instead of deriving a URL for a thumbnail or retaining the larger rendition.
var comixHIDRe = regexp.MustCompile(`"hid"\s*:\s*"[^"]+"`)
var comixPosterMediumRe = regexp.MustCompile(`(?s)"poster"\s*:\s*\{.*?"medium"\s*:\s*"([^"]+)"`)
// kagane's browser-fetched series response carries this published cover URL.
// Match the cover field, not merely an image-shaped URL elsewhere in JSON.
var kaganeCoverFieldRe = regexp.MustCompile(`"cover"\s*:\s*"([^"]+)"`)
func comixCoverURL(seriesURL, body string) string {
m := comixSlugRe.FindStringSubmatch(seriesURL)
if m == nil {
return ""
}
id := m[1]
if i := strings.Index(id, "-"); i != -1 {
id = id[:i]
}
hidRe := regexp.MustCompile(`"hid"\s*:\s*"` + regexp.QuoteMeta(id) + `"`)
hid := hidRe.FindStringIndex(body)
if hid == nil {
return ""
}
end := len(body)
if next := comixHIDRe.FindStringIndex(body[hid[1]:]); next != nil {
end = hid[1] + next[0]
}
if m := comixPosterMediumRe.FindStringSubmatch(body[hid[0]:end]); m != nil {
return publishedCoverURL(m[1])
}
return ""
}
func coverFrom(site, seriesURL, body string) (string, bool) {
var cover string
switch site {
case "asura", "demonic", "lightnovelworld":
cover = metaContent(body, "property", "og:image")
case "novelfull":
cover = metaContent(body, "name", "image")
case "comix":
cover = comixCoverURL(seriesURL, body)
case "kagane":
if m := kaganeCoverFieldRe.FindStringSubmatch(body); m != nil {
cover = publishedCoverURL(m[1])
}
}
return cover, cover != ""
}
func metaContent(body, attrName, attrValue string) string {
for _, tag := range metaTagRe.FindAllString(body, -1) {
attrs := make(map[string]string)
for _, m := range doubleQuotedMetaAttrRe.FindAllStringSubmatch(tag, -1) {
attrs[strings.ToLower(m[1])] = m[2]
}
for _, m := range singleQuotedMetaAttrRe.FindAllStringSubmatch(tag, -1) {
attrs[strings.ToLower(m[1])] = m[2]
}
if strings.EqualFold(attrs[strings.ToLower(attrName)], attrValue) {
return publishedCoverURL(attrs["content"])
}
}
return ""
}
func publishedCoverURL(value string) string {
value = strings.TrimSpace(html.UnescapeString(value))
return strings.ReplaceAll(value, " ", "%20")
}
+101
View File
@@ -79,6 +79,107 @@ const lnwSeriesFixture = `
<a href="https://lightnovelworld.net/overgeared-chapter-9999/">Chapter 9999</a> <a href="https://lightnovelworld.net/overgeared-chapter-9999/">Chapter 9999</a>
` `
// Trimmed from https://asurascans.com/comics/chronicles-of-the-demon-faction-00dcbf97 on 2026-08-10.
const asuraCoverFixture = `<meta property="og:image" content="https://cdn.asurascans.com/asura-images/covers/chronicles-of-the-demon-faction.d4dcb8.webp">`
// Trimmed from https://demonicscans.org/manga/Catastrophic-Necromancer on 2026-08-10.
// The source publishes the raw space in this URL.
const demonicCoverFixture = `<meta property="og:image" content="https://readermc.org/images/thumbnails/Catastrophic Necromancer.webp">`
// Trimmed from https://comix.to/title/n8we-dungeons-and-crayons on 2026-08-10.
// comix publishes medium and large renditions in its state blob, with no og:image.
const comixCoverFixture = `{"hid":"recommended","poster":{"medium":"https://static.comix.to/recommended@280.jpg","large":"https://static.comix.to/recommended.jpg"},"hid":"n8we","poster":{"medium":"https://static.comix.to/039d/i/1/34/6a6742bf15736@280.jpg","large":"https://static.comix.to/039d/i/1/34/6a6742bf15736.jpg"}}`
// Trimmed from GET https://kagane.to/api/v2/series/019f84bc-9ba0-7ed9-86f5-8b905ec7c28b on 2026-08-09.
const kaganeCoverFixture = `{"image":"https://kagane.to/api/v2/image/00000000-0000-0000-0000-000000000000/compressed","cover":"https://kagane.to/api/v2/image/019f84bc-9ba0-7ed9-86f5-8b905ec7c28b/compressed"}`
// Trimmed from https://novelfull.com/reverend-insanity.html on 2026-08-10.
const novelfullCoverFixture = `<meta name="image" content="https://novelfull.com/uploads/webp/novel/reverend-insanity-82661d911a.webp">`
// Trimmed from https://lightnovelworld.net/novel/a-will-eternal/ on 2026-08-10.
const lnwCoverFixture = `<meta content='https://lightnovelworld.net/wp-content/uploads/2026/03/a-will-eternal-1.webp' property='og:image'>`
func TestCoverFrom(t *testing.T) {
const comixURL = "https://comix.to/title/n8we-dungeons-and-crayons"
tests := []struct {
name string
site string
seriesURL string
body string
wantOK bool
wantCover string
}{
{
name: "asura uses published metadata URL",
site: "asura", body: asuraCoverFixture, wantOK: true,
wantCover: "https://cdn.asurascans.com/asura-images/covers/chronicles-of-the-demon-faction.d4dcb8.webp",
},
{
name: "demonic escapes raw spaces",
site: "demonic", body: demonicCoverFixture, wantOK: true,
wantCover: "https://readermc.org/images/thumbnails/Catastrophic%20Necromancer.webp",
},
{
name: "comix takes target medium poster",
site: "comix", seriesURL: comixURL, body: comixCoverFixture, wantOK: true,
wantCover: "https://static.comix.to/039d/i/1/34/6a6742bf15736@280.jpg",
},
{
name: "kagane reads API cover field",
site: "kagane", body: kaganeCoverFixture, wantOK: true,
wantCover: "https://kagane.to/api/v2/image/019f84bc-9ba0-7ed9-86f5-8b905ec7c28b/compressed",
},
{
name: "novelfull reads image metadata",
site: "novelfull", body: novelfullCoverFixture, wantOK: true,
wantCover: "https://novelfull.com/uploads/webp/novel/reverend-insanity-82661d911a.webp",
},
{
name: "lightnovelworld reads og image",
site: "lightnovelworld", body: lnwCoverFixture, wantOK: true,
wantCover: "https://lightnovelworld.net/wp-content/uploads/2026/03/a-will-eternal-1.webp",
},
{
name: "page without cover is empty",
site: "asura", body: `<meta property="og:title" content="No Cover">`,
},
{
name: "unknown site is empty",
site: "unknown", body: asuraCoverFixture,
},
}
for _, tt := range tests {
t.Run(tt.name, func(t *testing.T) {
got, ok := coverFrom(tt.site, tt.seriesURL, tt.body)
if ok != tt.wantOK {
t.Fatalf("ok = %v, want %v (got %q)", ok, tt.wantOK, got)
}
if got != tt.wantCover {
t.Errorf("cover = %q, want %q", got, tt.wantCover)
}
})
}
}
func TestCoverFromChallenge(t *testing.T) {
seriesURLs := map[string]string{
"asura": "https://asurascans.com/comics/chronicles-of-the-demon-faction-f886a8af",
"demonic": "https://demonicscans.org/manga/Catastrophic-Necromancer",
"comix": "https://comix.to/title/n8we-dungeons-and-crayons",
"kagane": "https://kagane.to/series/019f84bc-9ba0-7ed9-86f5-8b905ec7c28b",
"novelfull": "https://novelfull.com/reverend-insanity.html",
"lightnovelworld": "https://lightnovelworld.net/novel/a-will-eternal/",
}
for site, seriesURL := range seriesURLs {
t.Run(site, func(t *testing.T) {
if got, ok := coverFrom(site, seriesURL, challengeFixture); ok || got != "" {
t.Fatalf("cover = %q, ok = %v, want empty", got, ok)
}
})
}
}
func TestLatestChapterFrom(t *testing.T) { func TestLatestChapterFrom(t *testing.T) {
const asuraURL = "https://asurascans.com/comics/chronicles-of-the-demon-faction-f886a8af" const asuraURL = "https://asurascans.com/comics/chronicles-of-the-demon-faction-f886a8af"
const demonicURL = "https://demonicscans.org/manga/Catastrophic-Necromancer" const demonicURL = "https://demonicscans.org/manga/Catastrophic-Necromancer"