Files
mangaBookmark/chrome/entrypoint.sh
T
sulthan 6489f97686 Clear the stale Chrome singleton lock at browser boot
A Chrome killed uncleanly leaves SingletonLock in the chrome-profile
volume, naming the hostname and pid that took it. Both change when the
container is rebuilt, so the next Chrome reads it as "another computer
holds this profile" and exits at startup — permanently, since the volume
outlives every recreate.

The only symptom was a bare connection reset at 9222: socat accepts,
start_browser fires, Chrome dies, wait_for_browser bails on browser_alive
and the helper closes the socket. Chrome's stderr went to /dev/null and
the give-up path logged nothing, so neither docker logs nor the client
could tell a dead browser from a dead network — which is most of why this
cost an afternoon. Both are now on the container's stderr.

Clearing the lock at boot is safe because container_name pins the volume
to one container: nothing can hold the profile when that line runs. A
lock left mid-lifetime is self-healing (same hostname, dead pid), so boot
is the whole exposure.
2026-08-10 19:10:43 +07:00

226 lines
6.3 KiB
Bash
Executable File

#!/bin/sh
set -eu
# A UTC clock is itself the bot signal — Cloudflare treats it as the datacenter
# default — and kagane's challenge then never clears. Measured 2026-08-08 with
# an identical container on one Indonesian egress IP: UTC never cleared in 60s
# (twice), while Asia/Jakarta and America/New_York both cleared in 4s. Any real
# zone will do; the zone does not have to match the IP's country, it just must
# not be UTC. It does have to be right the way Chrome reads it.
#
# TZ must carry the zone *name*. Chrome resolves the zone through ICU, which
# takes the name from /etc/localtime's symlink target and ignores the file's
# contents; bind-mounting the host's /etc/localtime therefore lands on the
# image's own symlink to Etc/UTC and leaves glibc reporting +07 while Chrome
# still reports UTC. /etc/timezone, mounted by docker-compose.yml, is the name.
[ -n "${TZ:-}" ] || TZ=$(cat /etc/timezone 2>/dev/null || echo UTC)
export TZ
state=/home/chrome/state
profile=/home/chrome/profile
lock_file=$state/lock
pid_file=$state/chrome.pid
connections_dir=$state/connections
last_use_file=$state/last-use
idle_seconds=300
mkdir -p "$state" "$profile" "$connections_dir"
exec 9>>"$lock_file"
# Chrome's own UA advertises "HeadlessChrome" under --headless=new, and that
# one token is the difference between kagane.to's challenge clearing in ~4s and
# never clearing at all. Read the installed major version so client hints and
# the UA stay aligned after an image rebuild.
major=$(google-chrome-stable --version | sed -E 's/[^0-9]*([0-9]+)\..*/\1/')
ua="Mozilla/5.0 (X11; Linux x86_64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/${major}.0.0.0 Safari/537.36"
lock() {
flock 9
}
unlock() {
flock -u 9
}
browser_alive() {
[ -s "$pid_file" ] || return 1
pid=$(cat "$pid_file")
[ -n "$pid" ] && kill -0 "$pid" 2>/dev/null
}
has_connections() {
for marker in "$connections_dir"/*; do
[ -e "$marker" ] || continue
pid=${marker##*/}
if kill -0 "$pid" 2>/dev/null; then
return 0
fi
# A SIGKILLed helper cannot run its cleanup trap. Reconcile its marker
# here so one dead client cannot pin Chrome forever.
rm -f "$marker"
done
return 1
}
start_browser() {
# No --enable-automation: it sets navigator.webdriver, the first thing a
# bot check reads. setsid gives Chrome a process group so the reaper can
# terminate its renderer children with the browser.
# --no-sandbox avoids granting SYS_ADMIN solely for Docker's unavailable
# user namespaces; containment is the unprivileged user and private network.
setsid google-chrome-stable \
--headless=new \
--no-sandbox \
--remote-debugging-port=9223 \
--user-agent="$ua" \
--user-data-dir="$profile" \
--no-first-run \
--no-default-browser-check \
--disable-gpu \
about:blank >/dev/null &
printf '%s\n' "$!" >"$pid_file"
}
stop_browser() {
pid=$(cat "$pid_file")
kill -TERM -- "-$pid" 2>/dev/null || kill -TERM "$pid" 2>/dev/null || true
i=0
while kill -0 "$pid" 2>/dev/null && [ "$i" -lt 100 ]; do
i=$((i + 1))
sleep 0.1
done
if kill -0 "$pid" 2>/dev/null; then
kill -KILL -- "-$pid" 2>/dev/null || kill -KILL "$pid" 2>/dev/null || true
fi
rm -f "$pid_file"
}
wait_for_browser() {
i=0
while [ "$i" -lt 300 ]; do
if wget -qO /dev/null http://127.0.0.1:9223/json/version; then
return 0
fi
browser_alive || return 1
i=$((i + 1))
sleep 0.1
done
return 1
}
finish_connection() {
lock
rm -f "$connection_marker"
date +%s >"$last_use_file"
unlock
}
connection_signal() {
trap - INT TERM HUP
finish_connection
exit 143
}
connection() {
connection_marker=$connections_dir/$$
lock
: >"$connection_marker"
if ! browser_alive; then
rm -f "$pid_file"
start_browser
fi
date +%s >"$last_use_file"
unlock
trap connection_signal INT TERM HUP
if wait_for_browser; then
if socat STDIO TCP:127.0.0.1:9223; then
result=0
else
result=$?
fi
else
# The client only ever sees a bare connection reset here, so this is
# the sole record that the browser, not the network, was the problem.
echo "browser did not come up; dropping connection" >&2
result=1
fi
finish_connection
return "$result"
}
reaper() {
while :; do
sleep 10
lock
if ! has_connections && browser_alive; then
now=$(date +%s)
last=$(cat "$last_use_file" 2>/dev/null || printf '%s' "$now")
if [ $((now - last)) -ge "$idle_seconds" ]; then
stop_browser
fi
fi
unlock
done
}
if [ "${1:-}" = connection ]; then
connection
exit $?
fi
# The files are process state, not the Chrome profile. The profile is a named
# volume in Compose, so clearance survives both a reap and a container rebuild.
for marker in "$connections_dir"/*; do
[ -e "$marker" ] || continue
rm -f "$marker"
done
rm -f "$pid_file" "$last_use_file"
# Chrome's singleton lock names the hostname and pid that took it, and a
# container rebuild changes both — so a Chrome killed uncleanly (OOM, docker
# kill) leaves a lock the next container reads as "another computer holds this
# profile" and refuses to start behind, permanently, with the only symptom a
# bare connection reset at 9222. Clearing it here is safe precisely because
# container_name pins this volume to one container: nothing can be holding the
# profile at the moment this line runs. The lock is process state; the
# clearance cookies it sits beside are not, and are left alone.
rm -f "$profile"/Singleton*
# Chrome binds DevTools to loopback and silently ignores
# --remote-debugging-address. socat remains the network front-end, but each
# accepted connection now starts a browser on demand and is tracked by a
# per-helper marker. A connection held by Go's transport delays reap by its
# idle timeout; the 300-second threshold starts once the last connection closes.
reaper &
reaper_pid=$!
socat TCP-LISTEN:9222,reuseaddr,fork EXEC:'/entrypoint.sh connection',nofork &
front_pid=$!
stop_browser_gracefully() {
lock
if browser_alive; then
# Chrome is a separate session, so stop its process group explicitly;
# this gives its cookie batch time to flush before the container exits.
stop_browser
fi
unlock
}
shutdown() {
trap - INT TERM HUP
stop_browser_gracefully
kill "$front_pid" "$reaper_pid" 2>/dev/null || true
exit 143
}
trap shutdown INT TERM HUP
if wait "$front_pid"; then
status=0
else
status=$?
fi
stop_browser_gracefully
kill "$reaper_pid" 2>/dev/null || true
exit "$status"