services: ollama-piha: image: ollama/ollama:latest container_name: ollama-piha restart: unless-stopped ports: # Loopback: healthcheck.sh curls localhost directly on the node. LAN IP: kb-query@PIHA # reaches this over the host's LAN interface -- kb-query runs in its own Docker network # (separate compose project), same reasoning as kb-query's KB_DSN reaching kb-postgres. # Requires .env (from env.example) next to this file at deploy. - "127.0.0.1:11434:11434" - "${LAN_BIND_IP}:11434:11434" environment: # Module 5 phase 4 plan §2 decision 2 / §5 requirement: the model is loaded only for the # duration of a request and released immediately after -- RPi5 has no GPU and limited RAM # (hosts/piha/capabilities.yaml: arm64, 4 cores, no acceleration), so this is a short burst # spike (idle Ollama binary ~100 MB) rather than a permanent ~1.5-2 GB resident cost. This # is a fallback-only path (kb-query only reaches this when SOLARIA is down/times out), not # the default hot path, so paying a cold-load per request here is the correct trade-off. - OLLAMA_KEEP_ALIVE=0 volumes: - /opt/homelab/data/ollama-piha:/root/.ollama # No GPU reservation -- PIHA is arm64 with no acceleration (hosts/piha/capabilities.yaml), # unlike services/ollama@SOLARIA. CPU-only inference here is expected to be slower; that is # exactly what the plan §5 live calibration step measures before this is trusted as a # default fallback (see README.md "Calibration status").