diff --git a/backend/internal/latest/browser.go b/backend/internal/latest/browser.go index b23c244..621ee28 100644 --- a/backend/internal/latest/browser.go +++ b/backend/internal/latest/browser.go @@ -178,6 +178,36 @@ func (f *BrowserFetcher) Image(ctx context.Context, imageID string) ([]byte, str // the poller answers with a 403 and its ordinary cooldown. var errChallengeHeld = errors.New("challenge held") +// errBrowserInterrupted distinguishes a remote Chrome restart from the +// caller's own deadline. chromedp reports both as context.Canceled. +var errBrowserInterrupted = errors.New("browser interrupted") + +func classifyBrowserError(ctx context.Context, browserLost bool, err error) error { + if err == nil || ctx.Err() != nil { + return err + } + if !browserLost { + return err + } + if !errors.Is(err, context.Canceled) { + return err + } + return fmt.Errorf("%w: %w", errBrowserInterrupted, err) +} + +func browserConnectionLost(ctx context.Context) bool { + c := chromedp.FromContext(ctx) + if c == nil || c.Browser == nil { + return true + } + select { + case <-c.Browser.LostConnection: + return true + default: + return false + } +} + // challengePollInterval paces re-reads while a challenge solves itself. const challengePollInterval = 2 * time.Second @@ -203,6 +233,7 @@ func (f *BrowserFetcher) run(ctx context.Context, target string, read chromedp.A f.mu.Lock() defer f.mu.Unlock() + callerCtx := ctx ctx, cancel := context.WithTimeout(ctx, challengeTimeout) defer cancel() tabCtx, cancelTab := chromedp.NewContext(f.allocCtx) @@ -219,20 +250,26 @@ func (f *BrowserFetcher) run(ctx context.Context, target string, read chromedp.A chromedp.Navigate(target), chromedp.WaitReady("body", chromedp.ByQuery), ); err != nil { - return err + return classifyBrowserError(callerCtx, browserConnectionLost(tabCtx), err) } - var lastErr error for { // The challenge reloads the page when it passes, which tears down the // execution context mid-read. That is a retry, not a failure. if err := chromedp.Run(tabCtx, read); err != nil { + err = classifyBrowserError(callerCtx, browserConnectionLost(tabCtx), err) + if errors.Is(err, errBrowserInterrupted) { + return err + } lastErr = err } else if done() { return nil } select { case <-ctx.Done(): + if err := callerCtx.Err(); err != nil { + return err + } if lastErr != nil { return fmt.Errorf("%w (last read: %v)", errChallengeHeld, lastErr) } diff --git a/backend/internal/latest/browser_test.go b/backend/internal/latest/browser_test.go index c94c1f4..41dcf6b 100644 --- a/backend/internal/latest/browser_test.go +++ b/backend/internal/latest/browser_test.go @@ -1,6 +1,10 @@ package latest -import "testing" +import ( + "context" + "errors" + "testing" +) func TestKaganeAPIURL(t *testing.T) { const uuid = "019f84bc-9ba0-7ed9-86f5-8b905ec7c28b" @@ -57,3 +61,17 @@ func TestNovelfullSeriesURL(t *testing.T) { }) } } +func TestClassifyBrowserInterruption(t *testing.T) { + if err := classifyBrowserError(context.Background(), true, context.Canceled); !errors.Is(err, errBrowserInterrupted) { + t.Fatalf("classifyBrowserError(context.Canceled) = %v, want browser interruption", err) + } + if err := classifyBrowserError(context.Background(), false, context.Canceled); errors.Is(err, errBrowserInterrupted) { + t.Fatalf("ordinary cancellation misclassified as browser interruption: %v", err) + } + + caller, cancel := context.WithCancel(context.Background()) + cancel() + if err := classifyBrowserError(caller, true, context.Canceled); errors.Is(err, errBrowserInterrupted) { + t.Fatalf("caller cancellation misclassified as browser interruption: %v", err) + } +} diff --git a/chrome/Dockerfile b/chrome/Dockerfile index 714d700..7c52c7c 100644 --- a/chrome/Dockerfile +++ b/chrome/Dockerfile @@ -18,7 +18,7 @@ FROM debian:trixie-slim # stale, and a stale browser is exactly what Cloudflare turns away — the 124 in # alpine-chrome is the worked example. Rebuild is the upgrade path. RUN apt-get update \ - && apt-get install -y --no-install-recommends ca-certificates wget gnupg \ + && apt-get install -y --no-install-recommends ca-certificates wget gnupg util-linux \ && wget -qO- https://dl.google.com/linux/linux_signing_key.pub \ | gpg --dearmor -o /usr/share/keyrings/google-chrome.gpg \ && echo "deb [arch=amd64 signed-by=/usr/share/keyrings/google-chrome.gpg] https://dl.google.com/linux/chrome/deb/ stable main" \ @@ -27,9 +27,15 @@ RUN apt-get update \ && apt-get install -y --no-install-recommends google-chrome-stable socat \ && rm -rf /var/lib/apt/lists/* +# Keep the profile path present so Docker initializes the named volume with +# the unprivileged user's ownership. + +RUN useradd --create-home --shell /usr/sbin/nologin chrome \ + && mkdir -p /home/chrome/profile /home/chrome/state \ + && chown -R chrome:chrome /home/chrome + # Unprivileged: Chrome refuses to run as root, and the CDP endpoint is a shell # on whatever user owns it. -RUN useradd --create-home --shell /usr/sbin/nologin chrome USER chrome WORKDIR /home/chrome diff --git a/chrome/entrypoint.sh b/chrome/entrypoint.sh index fefde7d..fb3a9d6 100755 --- a/chrome/entrypoint.sh +++ b/chrome/entrypoint.sh @@ -16,46 +16,197 @@ set -eu [ -n "${TZ:-}" ] || TZ=$(cat /etc/timezone 2>/dev/null || echo UTC) export TZ -# Chrome's own UA advertises "HeadlessChrome" under --headless=new, and that one -# token is the difference between kagane.to's challenge clearing in ~4s and -# never clearing at all (measured 2026-08-08, same host, same Chrome, only the -# UA changed). Overriding it does not touch the Sec-CH-UA client hints, which -# report the real version, so the version is read back out of the binary rather -# than hardcoded: a hardcoded one would drift out of step with the hints on the -# next Chrome update and become a fresh tell. +state=/home/chrome/state +profile=/home/chrome/profile +lock_file=$state/lock +pid_file=$state/chrome.pid +connections_dir=$state/connections +last_use_file=$state/last-use +idle_seconds=300 + +mkdir -p "$state" "$profile" "$connections_dir" +exec 9>>"$lock_file" + +# Chrome's own UA advertises "HeadlessChrome" under --headless=new, and that +# one token is the difference between kagane.to's challenge clearing in ~4s and +# never clearing at all. Read the installed major version so client hints and +# the UA stay aligned after an image rebuild. major=$(google-chrome-stable --version | sed -E 's/[^0-9]*([0-9]+)\..*/\1/') ua="Mozilla/5.0 (X11; Linux x86_64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/${major}.0.0.0 Safari/537.36" -# Chrome binds its DevTools port to loopback and silently ignores -# --remote-debugging-address (verified 2026-08-08: Chrome 151 with -# --remote-debugging-address=0.0.0.0 still listened on 127.0.0.1 only), so the -# caller — another container — cannot reach it directly. socat fronting the -# loopback port is how chromedp/headless-shell solved the same problem and is -# why this image is a drop-in for it. -# -# Nothing publishes 9222; reachability is the `browser` network in -# docker-compose.yml, and an exposed CDP endpoint is remote code execution. -socat TCP-LISTEN:9222,fork,reuseaddr TCP:127.0.0.1:9223 & +lock() { + flock 9 +} -# Chrome stays in the foreground so that its death takes the container down and -# compose's restart policy applies; a backgrounded browser behind a live socat -# would leave the sidecar looking healthy while answering nothing. -# -# No --enable-automation: it sets navigator.webdriver, the first thing a bot -# check reads. -# -# --no-sandbox because Chrome's zygote wants user namespaces, which Docker's -# default profile does not hand out; the alternative is --cap-add=SYS_ADMIN, -# which gives the container strictly more than it takes away. Containment here -# is the unprivileged user, the isolated network, and the fact that this -# browser only ever navigates to kagane.to and novelfull.com. -exec google-chrome-stable \ - --headless=new \ - --no-sandbox \ - --remote-debugging-port=9223 \ - --user-agent="$ua" \ - --user-data-dir=/home/chrome/profile \ - --no-first-run \ - --no-default-browser-check \ - --disable-gpu \ - about:blank +unlock() { + flock -u 9 +} + +browser_alive() { + [ -s "$pid_file" ] || return 1 + pid=$(cat "$pid_file") + [ -n "$pid" ] && kill -0 "$pid" 2>/dev/null +} + +has_connections() { + for marker in "$connections_dir"/*; do + [ -e "$marker" ] || continue + pid=${marker##*/} + if kill -0 "$pid" 2>/dev/null; then + return 0 + fi + # A SIGKILLed helper cannot run its cleanup trap. Reconcile its marker + # here so one dead client cannot pin Chrome forever. + rm -f "$marker" + done + return 1 +} + +start_browser() { + # No --enable-automation: it sets navigator.webdriver, the first thing a + # bot check reads. setsid gives Chrome a process group so the reaper can + # terminate its renderer children with the browser. + # --no-sandbox avoids granting SYS_ADMIN solely for Docker's unavailable + # user namespaces; containment is the unprivileged user and private network. + setsid google-chrome-stable \ + --headless=new \ + --no-sandbox \ + --remote-debugging-port=9223 \ + --user-agent="$ua" \ + --user-data-dir="$profile" \ + --no-first-run \ + --no-default-browser-check \ + --disable-gpu \ + about:blank >/dev/null 2>&1 & + printf '%s\n' "$!" >"$pid_file" +} + +stop_browser() { + pid=$(cat "$pid_file") + kill -TERM -- "-$pid" 2>/dev/null || kill -TERM "$pid" 2>/dev/null || true + i=0 + while kill -0 "$pid" 2>/dev/null && [ "$i" -lt 100 ]; do + i=$((i + 1)) + sleep 0.1 + done + if kill -0 "$pid" 2>/dev/null; then + kill -KILL -- "-$pid" 2>/dev/null || kill -KILL "$pid" 2>/dev/null || true + fi + rm -f "$pid_file" +} + +wait_for_browser() { + i=0 + while [ "$i" -lt 300 ]; do + if wget -qO /dev/null http://127.0.0.1:9223/json/version; then + return 0 + fi + browser_alive || return 1 + i=$((i + 1)) + sleep 0.1 + done + return 1 +} + +finish_connection() { + lock + rm -f "$connection_marker" + date +%s >"$last_use_file" + unlock +} + +connection_signal() { + trap - INT TERM HUP + finish_connection + exit 143 +} + +connection() { + connection_marker=$connections_dir/$$ + lock + : >"$connection_marker" + if ! browser_alive; then + rm -f "$pid_file" + start_browser + fi + date +%s >"$last_use_file" + unlock + + trap connection_signal INT TERM HUP + if wait_for_browser; then + if socat STDIO TCP:127.0.0.1:9223; then + result=0 + else + result=$? + fi + else + result=1 + fi + finish_connection + return "$result" +} + +reaper() { + while :; do + sleep 10 + lock + if ! has_connections && browser_alive; then + now=$(date +%s) + last=$(cat "$last_use_file" 2>/dev/null || printf '%s' "$now") + if [ $((now - last)) -ge "$idle_seconds" ]; then + stop_browser + fi + fi + unlock + done +} + +if [ "${1:-}" = connection ]; then + connection + exit $? +fi + +# The files are process state, not the Chrome profile. The profile is a named +# volume in Compose, so clearance survives both a reap and a container rebuild. +for marker in "$connections_dir"/*; do + [ -e "$marker" ] || continue + rm -f "$marker" +done +rm -f "$pid_file" "$last_use_file" + +# Chrome binds DevTools to loopback and silently ignores +# --remote-debugging-address. socat remains the network front-end, but each +# accepted connection now starts a browser on demand and is tracked by a +# per-helper marker. A connection held by Go's transport delays reap by its +# idle timeout; the 300-second threshold starts once the last connection closes. +reaper & +reaper_pid=$! +socat TCP-LISTEN:9222,reuseaddr,fork EXEC:'/entrypoint.sh connection',nofork & +front_pid=$! + +stop_browser_gracefully() { + lock + if browser_alive; then + # Chrome is a separate session, so stop its process group explicitly; + # this gives its cookie batch time to flush before the container exits. + stop_browser + fi + unlock +} + +shutdown() { + trap - INT TERM HUP + stop_browser_gracefully + kill "$front_pid" "$reaper_pid" 2>/dev/null || true + exit 143 +} +trap shutdown INT TERM HUP + +if wait "$front_pid"; then + status=0 +else + status=$? +fi +stop_browser_gracefully +kill "$reaper_pid" 2>/dev/null || true +exit "$status" diff --git a/docker-compose.yml b/docker-compose.yml index e01dece..71b64a8 100644 --- a/docker-compose.yml +++ b/docker-compose.yml @@ -122,6 +122,8 @@ services: # The zone *name*, which is what Chrome's ICU needs — see chrome/entrypoint.sh. # Absent on a non-Debian host, which the entrypoint handles by falling back to UTC. - /etc/timezone:/etc/timezone:ro + # Cloudflare clearance must survive Chrome reaping and image recreation. + - chrome-profile:/home/chrome/profile # Chrome allocates shared memory per tab and dies on Docker's 64MB default. shm_size: '1gb' # Reaps zombie renderer processes, which otherwise accumulate for the @@ -143,6 +145,7 @@ volumes: # declared here: undeclared means `docker compose down -v` cannot take it # with the rest, so the old database survives the cutover until someone # removes it by hand. + chrome-profile: networks: # Not `internal: true`: headless Chrome still needs outbound access to reach diff --git a/docs/adr/0005-on-demand-browser.md b/docs/adr/0005-on-demand-browser.md new file mode 100644 index 0000000..c3ddba6 --- /dev/null +++ b/docs/adr/0005-on-demand-browser.md @@ -0,0 +1,48 @@ +# ADR-0005: On-demand browser sidecar + +Date: 2026-08-09 +Status: accepted + +## Decision + +Keep the `headless-shell` service and its CDP port alive, but start Google Chrome +only when the first CDP connection arrives. The entrypoint supervises a `socat` +front-end, serializes browser start/reap state with `flock`, and tracks each +connection with a marker named for its helper PID. A reaper stops Chrome after +300 seconds with no live markers. Marker reconciliation covers a helper killed +before its cleanup trap runs. + +Chrome runs in its own process group so reap sends the termination signal to +Chrome and its renderer children. The explicit `/home/chrome/profile` user-data +directory remains: Chrome remaps remote debugging to loopback on modern builds, +and Chrome ignores remote-debugging flags on a default profile. `socat` therefore +continues to front Chrome's loopback CDP port. + +The profile is a named Compose volume. Clearance cookies survive both a reap and +`docker compose up --build`; the browser still starts with a fresh debugger UUID, +so chromedp must keep endpoint discovery enabled and must not use +`chromedp.NoModifyURL`. + +The socat front-end and explicit profile are retained because Chromium remaps a +non-loopback debugging address to loopback since M113, while Chrome ignores the +remote-debugging flags on a default profile since Chrome 136. Flag tuning is +deliberately not adopted: its roughly 30% idle-footprint saving is irrelevant +to a browser that exists for seconds per wake and risks an untested fingerprint. + +## Constraints + +The 300-second floor is deliberate. Chromium batches cookie persistence on a +roughly 31-second timer, and Go's default HTTP transport can keep the discovery +connection parked for about 90 seconds after use. Reaping only with zero live +connections holds Chrome through both windows and through the poller's staggered +batch plus cover prefetch. + +The anti-bot properties remain unchanged: a plausible non-UTC timezone, a +Chrome-version-derived User-Agent without `HeadlessChrome`, and no automation +flag. A remote browser restart can surface as `context.Canceled`, the same error +as a caller deadline, so the backend wraps cancellation observed with a closed +CDP connection as `browser interrupted`; the focused test asserts that +classification without killing a real browser. +The same process-group stop runs during supervisor shutdown, not only during +idle reap, so Chrome can flush its cookie batch before a container rebuild or +graceful stop.