homelab-codex-ws/services/control-plane/tests/test_executor_redeploy.py

190 lines
7.3 KiB
Python
Raw Normal View History

fix(control-plane): redeploy wykonywalny — dispatch do host-side deploy-runnera Executor odpalal scripts/deploy/deploy-node.sh <node> <service> wewnatrz swojego kontenera: skrypt ignoruje oba argumenty i wymaga repo w ${HOME}/homelab-codex-ws (w kontenerze HOME=/home/homelab) -> exit 1 w 18. linii. Za tym brak git, brak klienta docker w obrazie, a gdyby przeszedl — deploy calego zestawu uslug hosta executora zamiast wezla z akcji. Kazdy redeploy padal (recon D14/D15; 18 pending / 0 completed). Redeploy idzie teraz ta sama sciezka pull co container_restart — VPS nigdy nie inicjuje polaczenia do wezla: executor -> actions/deploy/<node>/<id>.json -> deploy-runner (systemd na hoscie) rsync-pull, walidacja, deploy -> action_result event -> executor rozlicza completed/failed - scripts/deploy/deploy-service.sh: deploy jednej uslugi, wspoldzielony z deploy-node.sh, wiec inwokacja compose (a przez to nazwa projektu) jest identyczna jak przy deployu recznym - jobs/deploy-runner/: host-level, nie kontener — compose rozwiazuje wzgledne bindy i nazwe projektu tak jak przy deployu czlowieka; niezalezny od node-agenta, wiec potrafi zredeployowac takze jego - walidacja: tylko typ redeploy, node musi sie zgadzac, usluga musi byc w hosts/<node>/services.yaml, zadna tresc z payloadu nie trafia do shella - --force-recreate bez --build i bez --remove-orphans: redeploy to rekoncyliacja, nie wysylka kodu - executor: REDEPLOY_TIMEOUT_SECS=900, /repo zjechany do :ro (nieuzywany) 248 testow zielonych; deploy-node.sh przecwiczony na atrapie dockera — argv compose bez zmian. Instalacja unitow na wezlach i E2E: backlog. Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
2026-08-03 18:26:30 +02:00
"""Tests for the Executor redeploy path (recon D14/D15 — etap 2).
Before this, `redeploy` ran scripts/deploy/deploy-node.sh inside the executor
container: a script that ignores its arguments, expects a repo at
${HOME}/homelab-codex-ws (absent there, along with git and the docker CLI), and
would have deployed the EXECUTOR's own host's full service set rather than the
action's target. Every redeploy failed, which is why healthcheck_failed had to
be rerouted to container_restart and the action queue never drained.
Now redeploy follows the same pull architecture as container_restart: the
executor writes a dispatch file, the host-side deploy runner on the target node
(jobs/deploy-runner/) executes it and reports back an action_result event.
"""
from __future__ import annotations
import json
import sys
import time
from pathlib import Path
import pytest
sys.path.insert(0, str(Path(__file__).parent.parent / "src"))
import executor as executor_module
from executor import Executor
def _setup_executor(tmp_path: Path, monkeypatch) -> Executor:
actions = tmp_path / "actions"
events = tmp_path / "events"
for d in (actions, events):
d.mkdir(parents=True, exist_ok=True)
monkeypatch.setattr(executor_module, "ACTIONS_DIR", actions)
monkeypatch.setattr(executor_module, "EVENTS_DIR", events)
monkeypatch.setattr(executor_module, "DISPATCH_DIR", actions / "dispatch")
monkeypatch.setattr(executor_module, "DEPLOY_DISPATCH_DIR", actions / "deploy")
return Executor()
def _write_approved(tmp_path, action_id, node="piha", service="vikunja"):
action = {
"action_id": action_id,
"type": "redeploy",
"node": node,
"service": service,
"status": "approved",
"timestamp": time.time(),
}
path = tmp_path / "actions" / "approved" / f"{action_id}.json"
path.write_text(json.dumps(action))
return path
def _write_action_result_event(tmp_path, node, action_id, success, error="", ts=None):
ts = int(ts if ts is not None else time.time())
node_dir = tmp_path / "events" / node
node_dir.mkdir(parents=True, exist_ok=True)
event_id = f"evt-{node}-{ts}-action_result-{action_id}"
(node_dir / f"{event_id}.json").write_text(json.dumps({
"id": event_id,
"timestamp": ts,
"type": "action_result",
"node": node,
"payload": {"action_id": action_id, "success": success, "error": error,
"node": node, "source": "deploy-runner"},
}))
def _exists(tmp_path, state, action_id):
return (tmp_path / "actions" / state / f"{action_id}.json").exists()
def _read(tmp_path, state, action_id):
return json.loads((tmp_path / "actions" / state / f"{action_id}.json").read_text())
# ---------------------------------------------------------------------------
# Dispatch
# ---------------------------------------------------------------------------
def test_redeploy_is_dispatched_not_executed_locally(tmp_path, monkeypatch):
ex = _setup_executor(tmp_path, monkeypatch)
def fail(*args, **kwargs):
raise AssertionError("executor must never run a deploy itself")
monkeypatch.setattr(executor_module.subprocess, "run", fail)
ex._execute_action(_write_approved(tmp_path, "redeploy-piha-vikunja"))
dispatch_file = tmp_path / "actions" / "deploy" / "piha" / "redeploy-piha-vikunja.json"
assert dispatch_file.exists()
payload = json.loads(dispatch_file.read_text())
assert payload["type"] == "redeploy"
assert payload["node"] == "piha"
assert payload["service"] == "vikunja"
assert "dispatched_at" in payload
def test_redeploy_inbox_is_separate_from_node_agent_dispatch(tmp_path, monkeypatch):
"""node-agent deletes and failure-reports anything in its own inbox that is
not container_restart, so a redeploy must never be written there."""
ex = _setup_executor(tmp_path, monkeypatch)
ex._execute_action(_write_approved(tmp_path, "redeploy-piha-vikunja"))
node_agent_inbox = tmp_path / "actions" / "dispatch"
assert not list(node_agent_inbox.glob("**/*.json"))
def test_redeploy_stays_running_until_reported(tmp_path, monkeypatch):
ex = _setup_executor(tmp_path, monkeypatch)
ex._execute_action(_write_approved(tmp_path, "rd-1"))
assert _exists(tmp_path, "running", "rd-1")
assert not _exists(tmp_path, "completed", "rd-1")
assert not _exists(tmp_path, "failed", "rd-1")
@pytest.mark.parametrize("node,service", [("", "vikunja"), ("piha", ""), ("", "")])
def test_redeploy_without_node_or_service_fails_immediately(tmp_path, monkeypatch, node, service):
ex = _setup_executor(tmp_path, monkeypatch)
ex._execute_action(_write_approved(tmp_path, "rd-bad", node=node, service=service))
assert _exists(tmp_path, "failed", "rd-bad")
assert "requires both node and service" in _read(tmp_path, "failed", "rd-bad")["error"]
assert not list((tmp_path / "actions" / "deploy").glob("**/*.json"))
# ---------------------------------------------------------------------------
# Reconcile
# ---------------------------------------------------------------------------
def test_redeploy_completes_on_success_result(tmp_path, monkeypatch):
ex = _setup_executor(tmp_path, monkeypatch)
ex._execute_action(_write_approved(tmp_path, "rd-2"))
_write_action_result_event(tmp_path, "piha", "rd-2", success=True)
ex._reconcile_running_actions()
assert _exists(tmp_path, "completed", "rd-2")
assert not _exists(tmp_path, "running", "rd-2")
def test_redeploy_fails_on_failure_result(tmp_path, monkeypatch):
ex = _setup_executor(tmp_path, monkeypatch)
ex._execute_action(_write_approved(tmp_path, "rd-3"))
_write_action_result_event(tmp_path, "piha", "rd-3", success=False,
error="deploy-service.sh exited 1")
ex._reconcile_running_actions()
assert _read(tmp_path, "failed", "rd-3")["error"] == "deploy-service.sh exited 1"
def test_redeploy_uses_its_own_longer_timeout(tmp_path, monkeypatch):
"""A redeploy must not be timed out on the container_restart budget: the
runner polls once a minute and may pull images before reporting."""
monkeypatch.setattr(executor_module, "ACTION_TIMEOUT_SECS", 300)
monkeypatch.setattr(executor_module, "REDEPLOY_TIMEOUT_SECS", 900)
ex = _setup_executor(tmp_path, monkeypatch)
ex._execute_action(_write_approved(tmp_path, "rd-4"))
running_path = tmp_path / "actions" / "running" / "rd-4.json"
data = json.loads(running_path.read_text())
data["started_at"] = time.time() - 600 # past 300s, inside 900s
running_path.write_text(json.dumps(data))
ex._reconcile_running_actions()
assert _exists(tmp_path, "running", "rd-4")
data["started_at"] = time.time() - 1000 # past 900s
running_path.write_text(json.dumps(data))
ex._reconcile_running_actions()
assert _exists(tmp_path, "failed", "rd-4")
error = _read(tmp_path, "failed", "rd-4")["error"]
assert "Timed out after 900s" in error
assert "deploy runner" in error
def test_stale_result_from_previous_run_does_not_resolve(tmp_path, monkeypatch):
ex = _setup_executor(tmp_path, monkeypatch)
_write_action_result_event(tmp_path, "piha", "rd-5", success=True,
ts=int(time.time()) - 86_400)
ex._execute_action(_write_approved(tmp_path, "rd-5"))
ex._reconcile_running_actions()
assert _exists(tmp_path, "running", "rd-5")