6489f97686
A Chrome killed uncleanly leaves SingletonLock in the chrome-profile volume, naming the hostname and pid that took it. Both change when the container is rebuilt, so the next Chrome reads it as "another computer holds this profile" and exits at startup — permanently, since the volume outlives every recreate. The only symptom was a bare connection reset at 9222: socat accepts, start_browser fires, Chrome dies, wait_for_browser bails on browser_alive and the helper closes the socket. Chrome's stderr went to /dev/null and the give-up path logged nothing, so neither docker logs nor the client could tell a dead browser from a dead network — which is most of why this cost an afternoon. Both are now on the container's stderr. Clearing the lock at boot is safe because container_name pins the volume to one container: nothing can hold the profile when that line runs. A lock left mid-lifetime is self-healing (same hostname, dead pid), so boot is the whole exposure.
226 lines
6.3 KiB
Bash
Executable File
226 lines
6.3 KiB
Bash
Executable File
#!/bin/sh
|
|
set -eu
|
|
|
|
# A UTC clock is itself the bot signal — Cloudflare treats it as the datacenter
|
|
# default — and kagane's challenge then never clears. Measured 2026-08-08 with
|
|
# an identical container on one Indonesian egress IP: UTC never cleared in 60s
|
|
# (twice), while Asia/Jakarta and America/New_York both cleared in 4s. Any real
|
|
# zone will do; the zone does not have to match the IP's country, it just must
|
|
# not be UTC. It does have to be right the way Chrome reads it.
|
|
#
|
|
# TZ must carry the zone *name*. Chrome resolves the zone through ICU, which
|
|
# takes the name from /etc/localtime's symlink target and ignores the file's
|
|
# contents; bind-mounting the host's /etc/localtime therefore lands on the
|
|
# image's own symlink to Etc/UTC and leaves glibc reporting +07 while Chrome
|
|
# still reports UTC. /etc/timezone, mounted by docker-compose.yml, is the name.
|
|
[ -n "${TZ:-}" ] || TZ=$(cat /etc/timezone 2>/dev/null || echo UTC)
|
|
export TZ
|
|
|
|
state=/home/chrome/state
|
|
profile=/home/chrome/profile
|
|
lock_file=$state/lock
|
|
pid_file=$state/chrome.pid
|
|
connections_dir=$state/connections
|
|
last_use_file=$state/last-use
|
|
idle_seconds=300
|
|
|
|
mkdir -p "$state" "$profile" "$connections_dir"
|
|
exec 9>>"$lock_file"
|
|
|
|
# Chrome's own UA advertises "HeadlessChrome" under --headless=new, and that
|
|
# one token is the difference between kagane.to's challenge clearing in ~4s and
|
|
# never clearing at all. Read the installed major version so client hints and
|
|
# the UA stay aligned after an image rebuild.
|
|
major=$(google-chrome-stable --version | sed -E 's/[^0-9]*([0-9]+)\..*/\1/')
|
|
ua="Mozilla/5.0 (X11; Linux x86_64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/${major}.0.0.0 Safari/537.36"
|
|
|
|
lock() {
|
|
flock 9
|
|
}
|
|
|
|
unlock() {
|
|
flock -u 9
|
|
}
|
|
|
|
browser_alive() {
|
|
[ -s "$pid_file" ] || return 1
|
|
pid=$(cat "$pid_file")
|
|
[ -n "$pid" ] && kill -0 "$pid" 2>/dev/null
|
|
}
|
|
|
|
has_connections() {
|
|
for marker in "$connections_dir"/*; do
|
|
[ -e "$marker" ] || continue
|
|
pid=${marker##*/}
|
|
if kill -0 "$pid" 2>/dev/null; then
|
|
return 0
|
|
fi
|
|
# A SIGKILLed helper cannot run its cleanup trap. Reconcile its marker
|
|
# here so one dead client cannot pin Chrome forever.
|
|
rm -f "$marker"
|
|
done
|
|
return 1
|
|
}
|
|
|
|
start_browser() {
|
|
# No --enable-automation: it sets navigator.webdriver, the first thing a
|
|
# bot check reads. setsid gives Chrome a process group so the reaper can
|
|
# terminate its renderer children with the browser.
|
|
# --no-sandbox avoids granting SYS_ADMIN solely for Docker's unavailable
|
|
# user namespaces; containment is the unprivileged user and private network.
|
|
setsid google-chrome-stable \
|
|
--headless=new \
|
|
--no-sandbox \
|
|
--remote-debugging-port=9223 \
|
|
--user-agent="$ua" \
|
|
--user-data-dir="$profile" \
|
|
--no-first-run \
|
|
--no-default-browser-check \
|
|
--disable-gpu \
|
|
about:blank >/dev/null &
|
|
printf '%s\n' "$!" >"$pid_file"
|
|
}
|
|
|
|
stop_browser() {
|
|
pid=$(cat "$pid_file")
|
|
kill -TERM -- "-$pid" 2>/dev/null || kill -TERM "$pid" 2>/dev/null || true
|
|
i=0
|
|
while kill -0 "$pid" 2>/dev/null && [ "$i" -lt 100 ]; do
|
|
i=$((i + 1))
|
|
sleep 0.1
|
|
done
|
|
if kill -0 "$pid" 2>/dev/null; then
|
|
kill -KILL -- "-$pid" 2>/dev/null || kill -KILL "$pid" 2>/dev/null || true
|
|
fi
|
|
rm -f "$pid_file"
|
|
}
|
|
|
|
wait_for_browser() {
|
|
i=0
|
|
while [ "$i" -lt 300 ]; do
|
|
if wget -qO /dev/null http://127.0.0.1:9223/json/version; then
|
|
return 0
|
|
fi
|
|
browser_alive || return 1
|
|
i=$((i + 1))
|
|
sleep 0.1
|
|
done
|
|
return 1
|
|
}
|
|
|
|
finish_connection() {
|
|
lock
|
|
rm -f "$connection_marker"
|
|
date +%s >"$last_use_file"
|
|
unlock
|
|
}
|
|
|
|
connection_signal() {
|
|
trap - INT TERM HUP
|
|
finish_connection
|
|
exit 143
|
|
}
|
|
|
|
connection() {
|
|
connection_marker=$connections_dir/$$
|
|
lock
|
|
: >"$connection_marker"
|
|
if ! browser_alive; then
|
|
rm -f "$pid_file"
|
|
start_browser
|
|
fi
|
|
date +%s >"$last_use_file"
|
|
unlock
|
|
|
|
trap connection_signal INT TERM HUP
|
|
if wait_for_browser; then
|
|
if socat STDIO TCP:127.0.0.1:9223; then
|
|
result=0
|
|
else
|
|
result=$?
|
|
fi
|
|
else
|
|
# The client only ever sees a bare connection reset here, so this is
|
|
# the sole record that the browser, not the network, was the problem.
|
|
echo "browser did not come up; dropping connection" >&2
|
|
result=1
|
|
fi
|
|
finish_connection
|
|
return "$result"
|
|
}
|
|
|
|
reaper() {
|
|
while :; do
|
|
sleep 10
|
|
lock
|
|
if ! has_connections && browser_alive; then
|
|
now=$(date +%s)
|
|
last=$(cat "$last_use_file" 2>/dev/null || printf '%s' "$now")
|
|
if [ $((now - last)) -ge "$idle_seconds" ]; then
|
|
stop_browser
|
|
fi
|
|
fi
|
|
unlock
|
|
done
|
|
}
|
|
|
|
if [ "${1:-}" = connection ]; then
|
|
connection
|
|
exit $?
|
|
fi
|
|
|
|
# The files are process state, not the Chrome profile. The profile is a named
|
|
# volume in Compose, so clearance survives both a reap and a container rebuild.
|
|
for marker in "$connections_dir"/*; do
|
|
[ -e "$marker" ] || continue
|
|
rm -f "$marker"
|
|
done
|
|
rm -f "$pid_file" "$last_use_file"
|
|
|
|
# Chrome's singleton lock names the hostname and pid that took it, and a
|
|
# container rebuild changes both — so a Chrome killed uncleanly (OOM, docker
|
|
# kill) leaves a lock the next container reads as "another computer holds this
|
|
# profile" and refuses to start behind, permanently, with the only symptom a
|
|
# bare connection reset at 9222. Clearing it here is safe precisely because
|
|
# container_name pins this volume to one container: nothing can be holding the
|
|
# profile at the moment this line runs. The lock is process state; the
|
|
# clearance cookies it sits beside are not, and are left alone.
|
|
rm -f "$profile"/Singleton*
|
|
|
|
# Chrome binds DevTools to loopback and silently ignores
|
|
# --remote-debugging-address. socat remains the network front-end, but each
|
|
# accepted connection now starts a browser on demand and is tracked by a
|
|
# per-helper marker. A connection held by Go's transport delays reap by its
|
|
# idle timeout; the 300-second threshold starts once the last connection closes.
|
|
reaper &
|
|
reaper_pid=$!
|
|
socat TCP-LISTEN:9222,reuseaddr,fork EXEC:'/entrypoint.sh connection',nofork &
|
|
front_pid=$!
|
|
|
|
stop_browser_gracefully() {
|
|
lock
|
|
if browser_alive; then
|
|
# Chrome is a separate session, so stop its process group explicitly;
|
|
# this gives its cookie batch time to flush before the container exits.
|
|
stop_browser
|
|
fi
|
|
unlock
|
|
}
|
|
|
|
shutdown() {
|
|
trap - INT TERM HUP
|
|
stop_browser_gracefully
|
|
kill "$front_pid" "$reaper_pid" 2>/dev/null || true
|
|
exit 143
|
|
}
|
|
trap shutdown INT TERM HUP
|
|
|
|
if wait "$front_pid"; then
|
|
status=0
|
|
else
|
|
status=$?
|
|
fi
|
|
stop_browser_gracefully
|
|
kill "$reaper_pid" 2>/dev/null || true
|
|
exit "$status"
|