homelab-codex-ws/services/control-plane/src/executor.py
oskar da151fc8d3 fix(control-plane): redeploy wykonywalny — dispatch do host-side deploy-runnera
Executor odpalal scripts/deploy/deploy-node.sh <node> <service> wewnatrz
swojego kontenera: skrypt ignoruje oba argumenty i wymaga repo w
${HOME}/homelab-codex-ws (w kontenerze HOME=/home/homelab) -> exit 1 w 18.
linii. Za tym brak git, brak klienta docker w obrazie, a gdyby przeszedl —
deploy calego zestawu uslug hosta executora zamiast wezla z akcji. Kazdy
redeploy padal (recon D14/D15; 18 pending / 0 completed).

Redeploy idzie teraz ta sama sciezka pull co container_restart — VPS nigdy
nie inicjuje polaczenia do wezla:
  executor -> actions/deploy/<node>/<id>.json
  -> deploy-runner (systemd na hoscie) rsync-pull, walidacja, deploy
  -> action_result event -> executor rozlicza completed/failed

- scripts/deploy/deploy-service.sh: deploy jednej uslugi, wspoldzielony z
  deploy-node.sh, wiec inwokacja compose (a przez to nazwa projektu) jest
  identyczna jak przy deployu recznym
- jobs/deploy-runner/: host-level, nie kontener — compose rozwiazuje
  wzgledne bindy i nazwe projektu tak jak przy deployu czlowieka;
  niezalezny od node-agenta, wiec potrafi zredeployowac takze jego
- walidacja: tylko typ redeploy, node musi sie zgadzac, usluga musi byc w
  hosts/<node>/services.yaml, zadna tresc z payloadu nie trafia do shella
- --force-recreate bez --build i bez --remove-orphans: redeploy to
  rekoncyliacja, nie wysylka kodu
- executor: REDEPLOY_TIMEOUT_SECS=900, /repo zjechany do :ro (nieuzywany)

248 testow zielonych; deploy-node.sh przecwiczony na atrapie dockera —
argv compose bez zmian. Instalacja unitow na wezlach i E2E: backlog.

Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
2026-08-03 18:26:30 +02:00

419 lines
18 KiB
Python

import os
import re
import json
import time
import logging
import subprocess
from pathlib import Path
def _atomic_write_json(path: Path, data) -> None:
"""Write JSON atomically: write to a sibling .tmp, fsync, then os.replace."""
tmp = path.with_suffix(".tmp")
with open(tmp, "w") as f:
json.dump(data, f, indent=2)
f.flush()
os.fsync(f.fileno())
os.replace(tmp, path)
# Constants and Paths
RUNTIME_PATH = os.getenv("RUNTIME_PATH", "/opt/homelab")
ACTIONS_DIR = Path(RUNTIME_PATH) / "actions"
EVENTS_DIR = Path(RUNTIME_PATH) / "events"
DISPATCH_DIR = ACTIONS_DIR / "dispatch"
# Separate inbox for redeploy actions, consumed by the host-side deploy runner
# (jobs/deploy-runner/) rather than by node-agent. It must NOT share
# DISPATCH_DIR: node-agent's process_dispatched_actions() deletes and
# failure-reports every file it finds in its inbox, so a redeploy landing there
# would be killed before the runner ever saw it.
DEPLOY_DISPATCH_DIR = ACTIONS_DIR / "deploy"
# The executor no longer reads the repo at all (the old redeploy path ran a
# script out of it). Kept only so an operator can still see which checkout the
# container is wired to; nothing in this module resolves paths against it.
REPO_ROOT = Path(os.getenv("REPO_ROOT", "/repo"))
# SSH configuration
# SSH_USER can be overridden per-deployment environment.
# Still used by _execute_disk_cleanup (out of scope for this change — see
# docs/backlog.md "shadow_mode -> remediacja"). container_restart no longer
# uses SSH: see _dispatch_container_restart / _reconcile_running_actions.
SSH_USER = os.getenv("SSH_USER", "oskar")
SSH_OPTIONS = [
"-o", "StrictHostKeyChecking=no",
"-o", "ConnectTimeout=10",
"-o", "BatchMode=yes",
]
# How long a container_restart action may sit in "running" waiting for the
# target node-agent to report a result before the executor gives up and marks
# it failed. Env-overridable so a slow/flaky LTE node can be tuned without a
# code change.
ACTION_TIMEOUT_SECS = int(os.getenv("ACTION_TIMEOUT_SECS", "300"))
# Redeploy gets its own, longer budget: the deploy runner polls once a minute
# and a `docker compose up` may pull images before it can report anything.
REDEPLOY_TIMEOUT_SECS = int(os.getenv("REDEPLOY_TIMEOUT_SECS", "900"))
# Action types the executor hands to an on-node agent instead of resolving
# synchronously. They stay in running/ until an action_result event arrives (or
# the type's timeout expires) — see _reconcile_running_actions.
DISPATCHED_ACTION_TYPES = {"container_restart", "redeploy"}
# Matches evt-<node>-<unixts>-<type>-<svc>.json, same convention as
# node_agent.py / observer.py / operator_ui.py.
_EVENT_TS_RE = re.compile(r"-(\d{9,11})-")
# Logging setup
logging.basicConfig(level=logging.INFO, format='%(asctime)s - %(levelname)s - %(message)s')
logger = logging.getLogger("executor")
class Executor:
def __init__(self):
self._ensure_dirs()
def _ensure_dirs(self):
for s in ["approved", "running", "completed", "failed", "rejected"]:
(ACTIONS_DIR / s).mkdir(parents=True, exist_ok=True)
DISPATCH_DIR.mkdir(parents=True, exist_ok=True)
DEPLOY_DISPATCH_DIR.mkdir(parents=True, exist_ok=True)
def process_actions(self):
# Update heartbeat
heartbeat_file = ACTIONS_DIR.parent / "state" / "executor.heartbeat"
try:
heartbeat_file.touch()
except Exception as e:
logger.error(f"Failed to touch heartbeat file: {e}")
# Resolve container_restart actions dispatched to a node-agent on a
# previous cycle before dispatching new work, so a result that landed
# this cycle is reflected immediately rather than waiting a cycle.
self._reconcile_running_actions()
approved_dir = ACTIONS_DIR / "approved"
action_files = sorted(approved_dir.glob("*.json"))
for action_file in action_files:
self._execute_action(action_file)
def _execute_action(self, action_file):
action_id = action_file.stem
logger.info(f"Executing action: {action_id}")
# Move to running
running_path = ACTIONS_DIR / "running" / f"{action_id}.json"
try:
with open(action_file, "r") as f:
data = json.load(f)
data["status"] = "running"
data["started_at"] = time.time()
_atomic_write_json(running_path, data)
action_file.unlink()
except Exception as e:
logger.error(f"Failed to move {action_id} to running: {e}")
return
# Dispatch by action type
success = False
error_msg = ""
try:
action_type = data.get("type")
node = data.get("node")
service = data.get("service")
if action_type == "redeploy":
# Single-service redeploy, dispatched to the host-side deploy
# runner on the target node (jobs/deploy-runner/). Same pull
# architecture as container_restart — the executor never runs a
# deploy itself. It used to try: it ran deploy-node.sh inside
# this container, a script that ignores its arguments and needs
# a repo at ${HOME}/homelab-codex-ws (plus git and the docker
# CLI) that does not exist here, so every redeploy failed at the
# first line (recon D14/D15).
if not node or not service:
success = False
error_msg = (
f"redeploy requires both node and service "
f"(node={node!r}, service={service!r})"
)
else:
self._dispatch_redeploy(action_id, node, service)
# Stays in "running" until the runner reports an
# action_result — same as container_restart below.
return
elif action_type == "container_restart":
# No SSH from the executor (see CLAUDE.md / docs/backlog.md): the
# action is handed to the node-agent running on the target node,
# which restarts the container locally via its own docker socket.
# container_name is set by the supervisor; falls back to service name.
container_name = data.get("container_name") or service
self._dispatch_container_restart(action_id, node, service, container_name)
# Stays in "running" — _reconcile_running_actions() resolves it
# on a later cycle once the node-agent reports a result (or the
# action times out). Do not fall through to the completed/failed
# move below.
return
elif action_type == "disk_cleanup":
# Operator-approved aggressive Docker cleanup (image prune -a +
# volume prune). Commands come from the action payload so the
# supervisor controls exactly what runs; the executor adds a
# safety check to reject anything touching protected paths.
payload = data.get("payload", {})
success, error_msg = self._execute_disk_cleanup(node, payload)
elif action_type == "alert_only":
# Operator acknowledged the alert; no automated execution needed.
success = True
else:
success = False
error_msg = f"Unknown action type: {action_type}"
except Exception as e:
success = False
error_msg = str(e)
self._finalize_action(action_id, running_path, data, success, error_msg)
def _finalize_action(self, action_id, running_path, data, success, error_msg):
"""Move a running action to completed/ or failed/, recording the result."""
target_status = "completed" if success else "failed"
target_path = ACTIONS_DIR / target_status / f"{action_id}.json"
try:
data["status"] = target_status
data["finished_at"] = time.time()
if not success:
data["error"] = error_msg
_atomic_write_json(target_path, data)
running_path.unlink()
logger.info(f"Action {action_id} {target_status}")
except Exception as e:
logger.error(f"Failed to move {action_id} to {target_status}: {e}")
# ------------------------------------------------------------------
# container_restart: dispatch to node-agent, no SSH from the executor
# ------------------------------------------------------------------
#
# The executor has no SSH client and no key to the fleet (deliberate —
# see docs/backlog.md "PROJEKT: remediacja bez SSH"). Instead, it writes a
# small dispatch file that the node-agent running ON the target node picks
# up and executes locally through its own docker socket. This works
# identically whether the target is a remote node (piha/solaria/... reach
# the file via their existing rsync-pull, reusing the SSH key node-agent
# already has for event shipping) or the local VPS node (its node-agent
# reads the same on-disk path directly — no network hop at all).
def _dispatch_container_restart(self, action_id, node, service, container_name):
"""Write a dispatch file for the node-agent on `node` to pick up.
Does not resolve the action itself — it stays in running/ until
_reconcile_running_actions() sees an action_result event or the
action times out.
"""
if not node:
logger.error(f"Action {action_id}: container_restart with no node set")
return
inbox = DISPATCH_DIR / node
inbox.mkdir(parents=True, exist_ok=True)
payload = {
"action_id": action_id,
"type": "container_restart",
"node": node,
"service": service,
"container_name": container_name,
"dispatched_at": time.time(),
}
try:
_atomic_write_json(inbox / f"{action_id}.json", payload)
logger.info(
f"Dispatched container_restart {action_id} "
f"(container={container_name}) to node-agent on {node}"
)
except Exception as e:
logger.error(f"Failed to dispatch {action_id} to {node}: {e}")
def _dispatch_redeploy(self, action_id, node, service):
"""Write a redeploy action for the deploy runner on `node` to pick up.
Deliberately a different inbox from container_restart: node-agent
consumes actions/dispatch/<node>/ and rejects (and deletes) anything
that is not container_restart, so redeploys get their own
actions/deploy/<node>/ tree. The runner on the node pulls it over the
same rsync/ssh channel node-agent already uses, deploys locally with
scripts/deploy/deploy-service.sh, and reports back as an action_result
event — the executor still never initiates a connection to a node.
Does not resolve the action itself; _reconcile_running_actions() does.
"""
inbox = DEPLOY_DISPATCH_DIR / node
inbox.mkdir(parents=True, exist_ok=True)
payload = {
"action_id": action_id,
"type": "redeploy",
"node": node,
"service": service,
"dispatched_at": time.time(),
}
try:
_atomic_write_json(inbox / f"{action_id}.json", payload)
logger.info(
f"Dispatched redeploy {action_id} (service={service}) "
f"to deploy runner on {node}"
)
except Exception as e:
logger.error(f"Failed to dispatch {action_id} to {node}: {e}")
def _reconcile_running_actions(self):
"""Resolve actions previously dispatched to an on-node agent.
Covers both dispatched types: container_restart (executed by node-agent)
and redeploy (executed by the host-side deploy runner). Both report the
outcome the same way — an action_result event carrying the action_id —
so the resolution logic is identical; only the timeout differs.
Synchronous types (disk_cleanup/alert_only) resolve inside
_execute_action and never linger in running/, so they are not touched.
"""
running_dir = ACTIONS_DIR / "running"
if not running_dir.exists():
return
for action_file in sorted(running_dir.glob("*.json")):
try:
with open(action_file, "r") as f:
data = json.load(f)
except Exception as e:
logger.error(f"Failed to read running action {action_file.name}: {e}")
continue
action_type = data.get("type")
if action_type not in DISPATCHED_ACTION_TYPES:
continue
action_id = data.get("action_id") or action_file.stem
node = data.get("node")
started_at = data.get("started_at") or 0
executor_name = "deploy runner" if action_type == "redeploy" else "node-agent"
timeout_secs = (
REDEPLOY_TIMEOUT_SECS if action_type == "redeploy" else ACTION_TIMEOUT_SECS
)
result = self._find_action_result(node, action_id, started_at)
if result is not None:
success = bool(result.get("success"))
error_msg = "" if success else (result.get("error") or f"{executor_name} reported failure")
self._finalize_action(action_id, action_file, data, success, error_msg)
continue
if started_at and (time.time() - started_at) > timeout_secs:
error_msg = (
f"Timed out after {timeout_secs}s waiting for {executor_name} "
f"on '{node}' to report a result for {action_type} "
f"(action_id={action_id})"
)
logger.error(f"Action {action_id} timed out: {error_msg}")
self._finalize_action(action_id, action_file, data, False, error_msg)
def _find_action_result(self, node, action_id, started_at):
"""Look for an action_result event from `node` reporting on `action_id`.
Only considers events at or after `started_at` so a stale action_result
left over from a PREVIOUS run of the same (deterministic) action_id can
never be mistaken for the current run's outcome. Returns the event's
payload dict, or None if no matching result has arrived yet.
"""
if not node:
return None
node_events_dir = EVENTS_DIR / node
if not node_events_dir.exists():
return None
# Event timestamps are always whole seconds (int(time.time()) in
# node_agent.emit_event); started_at is a float. Floor started_at to
# whole seconds before comparing so a result emitted in the SAME
# wall-clock second as the dispatch (started_at's fractional part
# ahead of the truncated event second) is not mistaken for "stale".
started_at_floor = int(started_at)
for event_file in node_events_dir.glob("evt-*-action_result-*.json"):
m = _EVENT_TS_RE.search(event_file.stem)
if m and int(m.group(1)) < started_at_floor:
continue
try:
with open(event_file, "r") as f:
event = json.load(f)
except Exception:
continue
payload = event.get("payload", {})
if payload.get("action_id") == action_id:
event_ts = event.get("timestamp") or 0
if event_ts and event_ts < started_at_floor:
continue
return payload
return None
def _execute_disk_cleanup(self, node: str, payload: dict):
"""
SSH to the target node and run the operator-approved disk cleanup
commands from the action payload.
Safety invariants enforced here regardless of payload content:
- No command may reference /opt/homelab/data/, /opt/homelab/config/,
or /opt/homelab/state/ (application data and configuration).
- No command may contain rm -rf / or similar destructive patterns.
If any command fails the safety check the entire action is rejected
(not run at all) and the rejection reason is recorded.
Returns (success: bool, error_msg: str).
"""
commands = payload.get("commands", [
"docker image prune -a -f",
"docker volume prune -f",
])
# Safety gate: reject commands that touch protected paths
FORBIDDEN = [
"/opt/homelab/data",
"/opt/homelab/config",
"/opt/homelab/state",
"rm -rf /",
]
for cmd in commands:
for forbidden in FORBIDDEN:
if forbidden in cmd:
msg = f"Rejected: command contains forbidden pattern '{forbidden}': {cmd}"
logger.error(msg)
return False, msg
full_command = " && ".join(commands)
cmd = [
"ssh",
*SSH_OPTIONS,
f"{SSH_USER}@{node}",
full_command,
]
logger.info(f"Disk cleanup on {node}: {full_command}")
result = subprocess.run(cmd, capture_output=True, text=True)
if result.returncode == 0:
logger.info(f"Disk cleanup on {node} succeeded")
return True, ""
error_msg = (result.stderr or result.stdout).strip()
logger.error(f"Disk cleanup on {node} failed: {error_msg}")
return False, error_msg
def loop(self, interval=10):
logger.info("Starting executor loop")
while True:
self.process_actions()
time.sleep(interval)
if __name__ == "__main__":
executor = Executor()
executor.loop()