Delegate orphan cleanup to the backend instead of killing containers across stores
This commit is contained in:
@@ -8,7 +8,7 @@
|
|||||||
#
|
#
|
||||||
# Fixes:
|
# Fixes:
|
||||||
# 1. Stale podman ps/stats processes (>10 = pileup)
|
# 1. Stale podman ps/stats processes (>10 = pileup)
|
||||||
# 2. Orphaned conmon/crun processes holding ports
|
# Orphaned container cleanup is owned by the backend store-scoped reaper.
|
||||||
# 3. System tor conflicting with container tor
|
# 3. System tor conflicting with container tor
|
||||||
# 4. Tor hidden service directory permissions (group/other must have no
|
# 4. Tor hidden service directory permissions (group/other must have no
|
||||||
# access; Tor's own setgid 2700 is fine, restart is backed off)
|
# access; Tor's own setgid 2700 is fine, restart is backed off)
|
||||||
@@ -113,39 +113,13 @@ fix_stale_podman() {
|
|||||||
return 1
|
return 1
|
||||||
}
|
}
|
||||||
|
|
||||||
# ── Fix 2: Orphaned conmon holding ports ─────────────────────
|
# Orphan cleanup deliberately has one owner: the backend's ghost_reaper,
|
||||||
fix_orphaned_conmon() {
|
# which verifies the process UID, exact Podman storage root, successful complete
|
||||||
local fixed=false
|
# inventory and container identity before signalling anything. The former doctor
|
||||||
# Find conmon processes whose containers no longer exist
|
# loop scanned every conmon on the host and treated any failed default-store
|
||||||
local pids
|
# inspect as an orphan. It killed healthy containers in other stores (including
|
||||||
pids=$(pgrep -f "conmon.*--exit-command" 2>/dev/null || true)
|
# operator/test workloads), and runtime errors could make it kill managed apps.
|
||||||
if [ -z "$pids" ]; then
|
# Do not reintroduce independent process killing here.
|
||||||
return 1
|
|
||||||
fi
|
|
||||||
|
|
||||||
# Doctor runs as root but containers are rootless under archipelago user.
|
|
||||||
# Must check container existence using the rootless podman database.
|
|
||||||
local PODMANCMD="sudo -u archipelago XDG_RUNTIME_DIR=/run/user/1000 podman"
|
|
||||||
|
|
||||||
for pid in $pids; do
|
|
||||||
# Extract container ID from conmon args
|
|
||||||
local cid
|
|
||||||
cid=$(tr '\0' ' ' < /proc/"$pid"/cmdline 2>/dev/null | grep -oP '(?<=-c )[a-f0-9]{64}' || true)
|
|
||||||
if [ -z "$cid" ]; then
|
|
||||||
continue
|
|
||||||
fi
|
|
||||||
# Check if container still exists in rootless podman
|
|
||||||
if ! $PODMANCMD inspect "$cid" &>/dev/null; then
|
|
||||||
local port_info
|
|
||||||
port_info=$(ss -tlnp 2>/dev/null | grep "pid=$pid" | grep -oP ':\K\d+' | head -3 | tr '\n' ',' | sed 's/,$//')
|
|
||||||
log "Killing orphaned conmon pid=$pid (ports: ${port_info:-none})"
|
|
||||||
kill "$pid" 2>/dev/null || kill -9 "$pid" 2>/dev/null || true
|
|
||||||
fixed=true
|
|
||||||
fi
|
|
||||||
done
|
|
||||||
|
|
||||||
$fixed && return 0 || return 1
|
|
||||||
}
|
|
||||||
|
|
||||||
# ── Fix 3: Ensure system Tor is running (preferred over container) ──
|
# ── Fix 3: Ensure system Tor is running (preferred over container) ──
|
||||||
fix_system_tor_conflict() {
|
fix_system_tor_conflict() {
|
||||||
@@ -651,7 +625,6 @@ fi
|
|||||||
log "Starting container health check"
|
log "Starting container health check"
|
||||||
|
|
||||||
run_fix "stale-podman" fix_stale_podman
|
run_fix "stale-podman" fix_stale_podman
|
||||||
run_fix "orphaned-conmon" fix_orphaned_conmon
|
|
||||||
run_fix "system-tor" fix_system_tor_conflict
|
run_fix "system-tor" fix_system_tor_conflict
|
||||||
run_fix "tor-permissions" fix_tor_permissions
|
run_fix "tor-permissions" fix_tor_permissions
|
||||||
run_fix "searxng" fix_searxng
|
run_fix "searxng" fix_searxng
|
||||||
|
|||||||
Reference in New Issue
Block a user