Closes #58 ## Summary - Add pure per-Site cover extraction beside latest-chapter parsing for all six Sites. - Read Asura, Demonic, LightNovelWorld, and NovelFull metadata; read the Comix target detail state; read Kagane's browser-fetched `series_covers[].image_id` JSON. - Preserve published cover URLs, percent-encode Demonic raw spaces, select Comix's smaller published `medium`, and avoid thumbnail rendition URL synthesis. - Add live-source fixtures plus no-cover and Cloudflare challenge coverage for every Site. ## Correctness - Scope Comix extraction to the requested series detail key, avoiding recommended posters. - Parse Kagane's current live API shape and emit its canonical compressed image route from the published image ID; unrelated JSON fields are ignored. - Validate Kagane image IDs against the existing UUID-shaped route constraint. - Keep extraction pure; storage, polling, and wire integration remain outside issue #58. ## Acceptance criteria - [x] Cover extraction exists for all six Sites in the existing latest parser module. - [x] Each Site has a live-source fixture with source URL and date. - [x] Comix reads the state blob, not metadata. - [x] Demonic raw spaces are percent-encoded. - [x] Comix returns the smaller published rendition. - [x] No-cover pages return empty. - [x] Cloudflare challenge pages return empty. - [x] No thumbnail URL is synthesized by editing a published URL. - [x] `go test ./...` passes. ## Verification - `go test ./...` - `go vet ./...` - `git diff --check` Parent issues #47 and #55 remain open as requested. Reviewed-on: #67 Co-authored-by: Sulthan Zaki <sultankiki05@gmail.com> Co-committed-by: Sulthan Zaki <sultankiki05@gmail.com>
This commit was merged in pull request #67.
This commit is contained in:
@@ -1,6 +1,8 @@
|
||||
package latest
|
||||
|
||||
import (
|
||||
"encoding/json"
|
||||
"html"
|
||||
"net/url"
|
||||
"regexp"
|
||||
"strconv"
|
||||
@@ -34,6 +36,18 @@ var demonicChapterRe = regexp.MustCompile(`chaptered\.php\?manga=\d+&(?:amp;)?ch
|
||||
// Only the id prefix is stable; the slug tail follows the title.
|
||||
var comixSlugRe = regexp.MustCompile(`/title/([^/?#]+)`)
|
||||
|
||||
func comixSeriesID(seriesURL string) (string, bool) {
|
||||
m := comixSlugRe.FindStringSubmatch(seriesURL)
|
||||
if m == nil {
|
||||
return "", false
|
||||
}
|
||||
id := m[1]
|
||||
if i := strings.Index(id, "-"); i != -1 {
|
||||
id = id[:i]
|
||||
}
|
||||
return id, true
|
||||
}
|
||||
|
||||
// kaganeChapterRe matches the chapter numbers in a kagane API response. This
|
||||
// branch is fed by the browser fetcher, so the body is JSON rather than HTML —
|
||||
// there are no anchors to scan.
|
||||
@@ -87,18 +101,14 @@ func latestChapterFrom(site, seriesURL, body string) (latestChapter, bool) {
|
||||
case "demonic":
|
||||
re = demonicChapterRe
|
||||
case "comix":
|
||||
m := comixSlugRe.FindStringSubmatch(seriesURL)
|
||||
if m == nil {
|
||||
id, ok := comixSeriesID(seriesURL)
|
||||
if !ok {
|
||||
return latestChapter{}, false
|
||||
}
|
||||
// comix ships an SPA: the served HTML carries a JSON state blob instead
|
||||
// of chapter anchors, and latestChapterUrl is the only place the newest
|
||||
// chapter appears. Scoping to this series' id prefix keeps a
|
||||
// "recommended" strip's entries from winning the maximum.
|
||||
id := m[1]
|
||||
if i := strings.Index(id, "-"); i != -1 {
|
||||
id = id[:i]
|
||||
}
|
||||
re = regexp.MustCompile(`"latestChapterUrl":"/title/` + regexp.QuoteMeta(id) + `-[^"]*-chapter-([0-9.]+)"`)
|
||||
case "kagane":
|
||||
re = kaganeChapterRe
|
||||
@@ -146,3 +156,102 @@ func latestChapterFrom(site, seriesURL, body string) (latestChapter, bool) {
|
||||
}
|
||||
return best, found
|
||||
}
|
||||
|
||||
var metaTagRe = regexp.MustCompile(`(?is)<meta\b[^>]*>`)
|
||||
var doubleQuotedMetaAttrRe = regexp.MustCompile(`(?is)([a-z][a-z0-9:_-]*)\s*=\s*"([^"]*)"`)
|
||||
var singleQuotedMetaAttrRe = regexp.MustCompile(`(?is)([a-z][a-z0-9:_-]*)\s*=\s*'([^']*)'`)
|
||||
|
||||
// comix's server-rendered page embeds query data in this JSON script; parsing
|
||||
// the target detail entry avoids matching posters from recommended results.
|
||||
var comixInitialDataRe = regexp.MustCompile(`(?is)<script\b[^>]*\bid\s*=\s*["']initial-data["'][^>]*>(.*?)</script>`)
|
||||
|
||||
// kagane's browser-fetched series response publishes cover image IDs under
|
||||
// series_covers. The API's canonical compressed image route is the only URL
|
||||
// form accepted by the store and browser fetcher; no rendition is guessed.
|
||||
func kaganeCoverURL(body string) string {
|
||||
var response struct {
|
||||
SeriesCovers []struct {
|
||||
ImageID string `json:"image_id"`
|
||||
} `json:"series_covers"`
|
||||
}
|
||||
if err := json.Unmarshal([]byte(body), &response); err != nil {
|
||||
return ""
|
||||
}
|
||||
for _, cover := range response.SeriesCovers {
|
||||
if kaganeImageIDRe.MatchString(cover.ImageID) {
|
||||
return "https://kagane.to/api/v2/image/" + cover.ImageID + "/compressed"
|
||||
}
|
||||
}
|
||||
return ""
|
||||
}
|
||||
|
||||
func comixCoverURL(seriesURL, body string) string {
|
||||
id, ok := comixSeriesID(seriesURL)
|
||||
if !ok {
|
||||
return ""
|
||||
}
|
||||
data := comixInitialDataRe.FindStringSubmatch(body)
|
||||
if data == nil {
|
||||
return ""
|
||||
}
|
||||
var state struct {
|
||||
Queries map[string]json.RawMessage `json:"queries"`
|
||||
}
|
||||
if err := json.Unmarshal([]byte(data[1]), &state); err != nil {
|
||||
return ""
|
||||
}
|
||||
raw := state.Queries[`["manga","detail","`+id+`"]`]
|
||||
if len(raw) == 0 {
|
||||
return ""
|
||||
}
|
||||
var detail struct {
|
||||
Poster struct {
|
||||
Medium string `json:"medium"`
|
||||
} `json:"poster"`
|
||||
}
|
||||
if err := json.Unmarshal(raw, &detail); err != nil {
|
||||
return ""
|
||||
}
|
||||
return publishedCoverURL(detail.Poster.Medium)
|
||||
}
|
||||
|
||||
// coverFrom reports false for unknown sites, challenge bodies, and pages with
|
||||
// no usable cover. Metadata extraction keeps scanning after an empty match so
|
||||
// a later published cover is not hidden by an empty tag.
|
||||
func coverFrom(site, seriesURL, body string) (string, bool) {
|
||||
var cover string
|
||||
switch site {
|
||||
case "asura", "demonic", "lightnovelworld":
|
||||
cover = metaContent(body, "property", "og:image")
|
||||
case "novelfull":
|
||||
cover = metaContent(body, "name", "image")
|
||||
case "comix":
|
||||
cover = comixCoverURL(seriesURL, body)
|
||||
case "kagane":
|
||||
cover = kaganeCoverURL(body)
|
||||
}
|
||||
return cover, cover != ""
|
||||
}
|
||||
|
||||
func metaContent(body, attrName, attrValue string) string {
|
||||
for _, tag := range metaTagRe.FindAllString(body, -1) {
|
||||
attrs := make(map[string]string)
|
||||
for _, m := range doubleQuotedMetaAttrRe.FindAllStringSubmatch(tag, -1) {
|
||||
attrs[strings.ToLower(m[1])] = m[2]
|
||||
}
|
||||
for _, m := range singleQuotedMetaAttrRe.FindAllStringSubmatch(tag, -1) {
|
||||
attrs[strings.ToLower(m[1])] = m[2]
|
||||
}
|
||||
if strings.EqualFold(attrs[strings.ToLower(attrName)], attrValue) {
|
||||
if cover := publishedCoverURL(attrs["content"]); cover != "" {
|
||||
return cover
|
||||
}
|
||||
}
|
||||
}
|
||||
return ""
|
||||
}
|
||||
|
||||
func publishedCoverURL(value string) string {
|
||||
value = strings.TrimSpace(html.UnescapeString(value))
|
||||
return strings.ReplaceAll(value, " ", "%20")
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user