fix(monitor): remove unfiltered docker container prune from health-monitor.sh

health-monitor.sh still carried the docker prune mechanism behind the
2026-07-30 ollama/SOLARIA incident (see kb/incidents/2026-07-30-ollama-
solaria-vanish.md, root cause R1). Verified 2026-08-06 it is not wired
into cron/systemd on any node — node_agent.py's R1-filtered prune is
the only cleanup path actually running in the fleet.

Remove the cleanup section entirely rather than backporting the R1
filter here too: node-agent is the sole owner of Docker cleanup, and a
second copy of the filter logic would just be a future drift risk.
Health checks (disk/RAM/CPU/container status) and the VPS control-
plane filesystem rotation are untouched.

Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01VMn76yx2CNuHKFVMrYcKWA
This commit is contained in:
oskar 2026-08-27 14:45:30 +02:00
parent 89f75c34b0
commit 9a86843ddd

View file

@ -1,17 +1,17 @@
#!/usr/bin/env bash #!/usr/bin/env bash
# health-monitor.sh - Homelab node health monitor and safe disk cleanup # health-monitor.sh - Homelab node health monitor
# #
# Designed to run standalone on the host (cron or direct) or to be called by # Designed to run standalone on the host (cron or direct) or to be called by
# the node-agent Python daemon. All cleanup decisions follow the conservative # the node-agent Python daemon.
# policy agreed in the design review:
# #
# lte_node (chelsty-infra, chelsty-ha) : NO cleanup at all # Docker cleanup (image/container/build-cache prune) does NOT live here.
# sd_card (piha, saturn) : dangling images + stopped containers, # node_agent.py (R1) is the sole owner of that cleanup, with a filtered
# rate-limited to once per 24 h # container prune that respects restart policy and compose ownership —
# ai_node (solaria) : dangling images + stopped containers # see kb/incidents/2026-07-30-ollama-solaria-vanish.md. This script used to
# + build cache (NEVER -a) # carry its own unfiltered `docker container prune -f` (never wired into
# standard (vps) : dangling images + stopped containers # cron/systemd on any node, verified 2026-08-06); it was removed rather than
# + build cache # backporting the R1 filter here too, to avoid two independent copies of
# cleanup logic drifting apart.
# #
# VPS additionally rotates control-plane filesystem artefacts: # VPS additionally rotates control-plane filesystem artefacts:
# actions/completed + failed > 7 days # actions/completed + failed > 7 days
@ -43,15 +43,6 @@ DISK_CRIT_PCT=85
MEM_WARN_PCT=85 MEM_WARN_PCT=85
MEM_CRIT_PCT=95 MEM_CRIT_PCT=95
# Rate-limit file for SD-card nodes (max one Docker cleanup per 24 h)
CLEANUP_LOCK="${STATE_DIR}/last-docker-cleanup"
CLEANUP_INTERVAL=86400 # seconds
# Node classifications
LTE_NODES="chelsty-infra chelsty-ha"
SD_CARD_NODES="piha saturn"
AI_NODES="solaria"
# --------------------------------------------------------------------------- # ---------------------------------------------------------------------------
# Helpers # Helpers
# --------------------------------------------------------------------------- # ---------------------------------------------------------------------------
@ -60,20 +51,6 @@ log() { echo "$(date -u +%H:%M:%S) [INFO] $*"; }
warn() { echo "$(date -u +%H:%M:%S) [WARN] $*" >&2; } warn() { echo "$(date -u +%H:%M:%S) [WARN] $*" >&2; }
err() { echo "$(date -u +%H:%M:%S) [ERROR] $*" >&2; } err() { echo "$(date -u +%H:%M:%S) [ERROR] $*" >&2; }
contains() {
local word="$1"; shift
for w in "$@"; do [[ "$w" == "$word" ]] && return 0; done
return 1
}
get_node_type() {
# shellcheck disable=SC2086
if contains "$NODE_NAME" $LTE_NODES; then echo "lte_node"; return; fi
if contains "$NODE_NAME" $SD_CARD_NODES; then echo "sd_card"; return; fi
if contains "$NODE_NAME" $AI_NODES; then echo "ai_node"; return; fi
echo "standard"
}
# --------------------------------------------------------------------------- # ---------------------------------------------------------------------------
# Event emission # Event emission
# --------------------------------------------------------------------------- # ---------------------------------------------------------------------------
@ -195,69 +172,6 @@ check_containers() {
--format "{{.Names}}" 2>/dev/null || true) --format "{{.Names}}" 2>/dev/null || true)
} }
# ---------------------------------------------------------------------------
# Safe Docker cleanup (per policy)
# ---------------------------------------------------------------------------
_sd_card_rate_ok() {
if [[ -f "${CLEANUP_LOCK}" ]]; then
local last_ts elapsed
last_ts=$(cat "${CLEANUP_LOCK}" 2>/dev/null || echo 0)
elapsed=$(( TIMESTAMP - last_ts ))
if [[ "${elapsed}" -lt "${CLEANUP_INTERVAL}" ]]; then
log "Docker cleanup skipped: last run ${elapsed}s ago (limit ${CLEANUP_INTERVAL}s)"
return 1
fi
fi
return 0
}
_mark_cleanup_done() {
echo "${TIMESTAMP}" > "${CLEANUP_LOCK}"
}
run_safe_cleanup() {
command -v docker &>/dev/null || return
local node_type
node_type=$(get_node_type)
case "${node_type}" in
lte_node)
# NO cleanup on LTE nodes. Any docker operation risks triggering
# a pull over a metered/intermittent connection.
log "Skipping Docker cleanup: LTE node (${NODE_NAME})"
;;
sd_card)
# Dangling images + stopped containers only.
# Rate-limited to once per 24 hours to protect SD card write endurance.
_sd_card_rate_ok || return
log "Running rate-limited Docker cleanup (SD card node)"
docker image prune -f >/dev/null 2>&1 || true
docker container prune -f >/dev/null 2>&1 || true
_mark_cleanup_done
;;
ai_node)
# Dangling images + stopped containers + build cache.
# NEVER docker image prune -a (would remove Ollama runtime images,
# requiring a multi-hour re-pull of model weights).
log "Running AI-node Docker cleanup (dangling images + containers + build cache)"
docker image prune -f >/dev/null 2>&1 || true
docker container prune -f >/dev/null 2>&1 || true
docker builder prune -f >/dev/null 2>&1 || true
;;
standard)
# VPS and other standard nodes: full safe cleanup.
log "Running standard Docker cleanup"
docker image prune -f >/dev/null 2>&1 || true
docker container prune -f >/dev/null 2>&1 || true
docker builder prune -f >/dev/null 2>&1 || true
;;
esac
}
# --------------------------------------------------------------------------- # ---------------------------------------------------------------------------
# VPS-specific: control-plane filesystem rotation # VPS-specific: control-plane filesystem rotation
# --------------------------------------------------------------------------- # ---------------------------------------------------------------------------
@ -315,15 +229,13 @@ except Exception:
mkdir -p "${EVENTS_DIR}/${NODE_NAME}" "${STATE_DIR}" mkdir -p "${EVENTS_DIR}/${NODE_NAME}" "${STATE_DIR}"
log "Health check starting on ${NODE_NAME} (type=$(get_node_type))" log "Health check starting on ${NODE_NAME}"
disk_pct=$(check_disk || echo 0) disk_pct=$(check_disk || echo 0)
mem_pct=$(check_memory || echo 0) mem_pct=$(check_memory || echo 0)
cpu_pct=$(check_cpu || echo 0) cpu_pct=$(check_cpu || echo 0)
check_containers check_containers
run_safe_cleanup
# VPS: also rotate control-plane filesystem artefacts # VPS: also rotate control-plane filesystem artefacts
if [[ "${NODE_NAME}" == "vps" ]]; then if [[ "${NODE_NAME}" == "vps" ]]; then
cleanup_control_plane_fs cleanup_control_plane_fs