From c38ed247df188f21a639d76aa91516213761bf70 Mon Sep 17 00:00:00 2001 From: Sulthan Zaki Date: Sun, 26 Jul 2026 15:34:54 +0700 Subject: [PATCH] feat: per-site latest-chapter extraction from series HTML Ports latestChapterFromAnchors from the userscript for asura and demonic. Asura's pattern is scoped to the series' own slug, which subsumes the userscript's anchor-text check and also excludes chapter links belonging to other series. Fixtures are trimmed from real pages. --- backend/latest_sites.go | 72 ++++++++++++++++++++++ backend/latest_sites_test.go | 115 +++++++++++++++++++++++++++++++++++ 2 files changed, 187 insertions(+) create mode 100644 backend/latest_sites.go create mode 100644 backend/latest_sites_test.go diff --git a/backend/latest_sites.go b/backend/latest_sites.go new file mode 100644 index 0000000..b19aa15 --- /dev/null +++ b/backend/latest_sites.go @@ -0,0 +1,72 @@ +package main + +import ( + "regexp" + "strconv" + "strings" +) + +// latestChapter is the newest chapter a series page advertises. +type latestChapter struct { + Num float64 + Label string +} + +// asuraSlugRe pulls the series slug out of a stored series_url. +// Shape verified live 2026-07-26: https://asurascans.com/comics/, where +// the slug carries a trailing hash-like suffix (e.g. "-f886a8af"). +var asuraSlugRe = regexp.MustCompile(`/comics/([^/?#]+)`) + +// demonicChapterRe matches the pre-redirect anchors demonic series pages link +// through. Both the raw "&" and the HTML-escaped "&" forms occur. +var demonicChapterRe = regexp.MustCompile(`chaptered\.php\?manga=\d+&(?:amp;)?chapter=([0-9.]+)`) + +// latestChapterFrom returns the highest chapter number body advertises for this +// series. ok is false when the body yields nothing usable — an unknown site, an +// empty body, a Cloudflare challenge page, and a site redesign all land here, +// and the caller treats all four identically. +// +// Ported from the userscript's latestChapterFromAnchors (asura L123-133, +// demonic L183-193), including its reason for taking a maximum rather than a +// first or last: neither site lists chapters in a dependable order. +// +// The userscript's asura rule additionally requires the anchor text to match +// /Chapter\s+[\d.]+/i. That check exists only to skip the "First Chapter" +// shortcut, which points at chapter/1 and therefore can never win a maximum, so +// it is redundant here. Scoping the pattern to this series' own slug replaces it +// with a stronger guarantee: a chapter link belonging to some other series +// cannot contribute even if the page starts carrying them. +func latestChapterFrom(site, seriesURL, body string) (latestChapter, bool) { + var re *regexp.Regexp + switch site { + case "asura": + m := asuraSlugRe.FindStringSubmatch(seriesURL) + if m == nil { + return latestChapter{}, false + } + // Compiled per call rather than cached: this runs once per fetch, which + // is at most a few times a minute, and the slug varies per series. + re = regexp.MustCompile(`/comics/` + regexp.QuoteMeta(m[1]) + `/chapter/([0-9.]+)`) + case "demonic": + re = demonicChapterRe + default: + return latestChapter{}, false + } + + var best latestChapter + found := false + for _, m := range re.FindAllStringSubmatch(body, -1) { + // [0-9.]+ can swallow a trailing separator, e.g. "chapter/12." in a + // sentence; ParseFloat would reject the whole match. + raw := strings.Trim(m[1], ".") + num, err := strconv.ParseFloat(raw, 64) + if err != nil { + continue + } + if !found || num > best.Num { + best = latestChapter{Num: num, Label: "Chapter " + raw} + found = true + } + } + return best, found +} diff --git a/backend/latest_sites_test.go b/backend/latest_sites_test.go new file mode 100644 index 0000000..a65b346 --- /dev/null +++ b/backend/latest_sites_test.go @@ -0,0 +1,115 @@ +package main + +import "testing" + +// Trimmed from https://asurascans.com/comics/chronicles-of-the-demon-faction-f886a8af +// fetched 2026-07-26. The first anchor is the "First Chapter" shortcut: it is a +// real chapter link with no "Chapter N" text, and it must not be mistaken for +// the latest just because it parses. +const asuraSeriesFixture = ` +First Chapter +Chapter 179 +Chapter 181 +Chapter 180 +` + +// A chapter link belonging to a different series, of the kind a "you might also +// like" strip would introduce. Slug scoping must exclude it. +const asuraCrossSeriesFixture = asuraSeriesFixture + ` +Chapter 999 +` + +// Trimmed from https://demonicscans.org/manga/Catastrophic-Necromancer fetched +// 2026-07-26. Note the raw "&", the doubled space after Chapter 0.5 +Chapter 294 +Chapter 296 +Chapter 295 +` + +// What Cloudflare serves instead of the page when an IP's bot score flips. +const challengeFixture = `Just a moment... + +
Checking your browser
` + +func TestLatestChapterFrom(t *testing.T) { + const asuraURL = "https://asurascans.com/comics/chronicles-of-the-demon-faction-f886a8af" + const demonicURL = "https://demonicscans.org/manga/Catastrophic-Necromancer" + + tests := []struct { + name string + site string + seriesURL string + body string + wantOK bool + wantNum float64 + wantLabel string + }{ + { + name: "asura takes the max, not the last listed", + site: "asura", seriesURL: asuraURL, body: asuraSeriesFixture, + wantOK: true, wantNum: 181, wantLabel: "Chapter 181", + }, + { + name: "asura ignores another series' chapter links", + site: "asura", seriesURL: asuraURL, body: asuraCrossSeriesFixture, + wantOK: true, wantNum: 181, wantLabel: "Chapter 181", + }, + { + name: "asura with an unparseable series url", + site: "asura", seriesURL: "https://asurascans.com/", body: asuraSeriesFixture, + wantOK: false, + }, + { + name: "demonic takes the max across raw and escaped ampersands", + site: "demonic", seriesURL: demonicURL, body: demonicSeriesFixture, + wantOK: true, wantNum: 296, wantLabel: "Chapter 296", + }, + { + name: "demonic keeps decimal chapters parseable", + site: "demonic", seriesURL: demonicURL, + body: `Chapter 0.5`, + wantOK: true, wantNum: 0.5, wantLabel: "Chapter 0.5", + }, + { + name: "empty body", + site: "asura", seriesURL: asuraURL, body: "", + wantOK: false, + }, + { + name: "cloudflare challenge page", + site: "asura", seriesURL: asuraURL, body: challengeFixture, + wantOK: false, + }, + { + name: "demonic markup handed to the asura rule", + site: "asura", seriesURL: asuraURL, body: demonicSeriesFixture, + wantOK: false, + }, + { + name: "unknown site", + site: "mangadex", seriesURL: "https://example.com/x", body: asuraSeriesFixture, + wantOK: false, + }, + } + + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + got, ok := latestChapterFrom(tt.site, tt.seriesURL, tt.body) + if ok != tt.wantOK { + t.Fatalf("ok = %v, want %v (got %+v)", ok, tt.wantOK, got) + } + if !tt.wantOK { + return + } + if got.Num != tt.wantNum { + t.Errorf("Num = %v, want %v", got.Num, tt.wantNum) + } + if got.Label != tt.wantLabel { + t.Errorf("Label = %q, want %q", got.Label, tt.wantLabel) + } + }) + } +}