From 01db57ab822666730d99eb9903b8f1afc92f6e53 Mon Sep 17 00:00:00 2001 From: oskar Date: Tue, 4 Aug 2026 15:12:24 +0200 Subject: [PATCH] fix(kb): przepiecie wszystkich odwolan wewnetrznych po migracji MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit 126 plikow (md, yaml, sh, py) odwolywalo sie do sciezek sprzed migracji. 15 markdown-linkow [..](..) -> policzona sciezka WZGLEDNA wobec pliku odsylajacego (wczesniej czesc z nich byla repo-root-relative i nie rozwiazywala sie z katalogu, w ktorym lezala) 200 odwolan tekstowych (backticki, proza, yaml, importy w kodzie) -> nowa sciezka repo-root-relative, zgodnie z konwencja repo 5 linkow rodzenstwa (gole nazwy plikow, np. "](DEPLOY.md)") — dzialaly tylko w starym katalogu; przeliczone recznie Objete m.in.: CLAUDE.md (scripts/onboard/README.md -> kb/runbooks/ node-onboarding-tool.md, docs/backlog.md -> kb/phases/backlog.md), README.md, .claude/skills/, 20 session logow, kod jobow. Ostatnie 5 odwolan pochodzi z tresci wciagnietej rebasem z origin/master (session log 2026-07-31, override node-agenta na SOLARII, dwie pozycje backlogu) — wskazywaly na docs/incidents/, docs/kb/modules/ i services/narty27/README.md sprzed migracji. Dodany wzajemny link miedzy kb/services/control-plane.md (stub kodu) a kb/subsystems/control-plane.md (opis, deprecated) — dwa dokumenty o tym samym systemie, latwe do pomylenia. Weryfikacja na 790 plikach: 0 odwolan do starych sciezek, 0 martwych linkow markdown. Lint OKF: 190/190 plikow ZGODNE. Co-Authored-By: Claude Opus 5 (1M context) --- .claude/skills/node-onboarding/SKILL.md | 2 +- CLAUDE.md | 4 +-- README.md | 26 +++++++++---------- docs/sessions/2026-06-08-lustro-onboarding.md | 2 +- ...26-06-09-flota-recovery-lustro-register.md | 2 +- .../2026-06-11-lustro-ssh-shipping.md | 8 +++--- docs/sessions/2026-06-17-kb-foundations.md | 4 +-- docs/sessions/2026-06-24-kb-gmail-importer.md | 4 +-- docs/sessions/2026-06-24.md | 2 +- docs/sessions/2026-06-30-fleet-inventory.md | 2 +- .../2026-07-02-modul0-cert-migracja.md | 2 +- docs/sessions/2026-07-02.md | 4 +-- docs/sessions/2026-07-06.md | 2 +- .../sessions/2026-07-12-deploy2-ocr-worker.md | 2 +- docs/sessions/2026-07-15.md | 4 +-- docs/sessions/2026-07-16.md | 2 +- ...026-07-23-control-plane-remediation-e2e.md | 4 +-- docs/sessions/2026-07-23-kb-f4-ingress.md | 8 +++--- docs/sessions/2026-07-27-kb-f4-fallback.md | 4 +-- docs/sessions/2026-07-28.md | 4 +-- .../2026-07-31-kb-f4-final-narty27.md | 4 +-- .../node_exporter/docker-compose.override.yml | 2 +- hosts/piha/services.yaml | 4 +-- .../node-agent/docker-compose.override.yml | 2 +- hosts/vps/services.yaml | 4 +-- jobs/deploy-runner/deploy-runner.sh | 4 +-- jobs/documents-ingest/eval/queries.yaml | 6 ++--- jobs/documents-ingest/eval/retrieval_eval.py | 6 ++--- .../src/documents_ingest/chunk_embed.py | 6 ++--- .../src/documents_ingest/cyclic_ingest.py | 2 +- .../src/documents_ingest/extractor.py | 2 +- .../src/documents_ingest/paperless_adapter.py | 2 +- .../src/documents_ingest/retrieval.py | 2 +- .../src/documents_ingest/summarize.py | 2 +- .../documents-ingest/systemd/kb-ingest-run.sh | 2 +- .../tests/test_chunk_embed.py | 4 +-- .../src/gmail_header_backfill/backfill.py | 2 +- .../src/mail_body_ingest/ingest.py | 2 +- kb/audits/czujniki-2026-07-30.md | 6 ++--- kb/audits/monitoring-coverage-2026-07-14.md | 2 +- kb/audits/piha-slim-2026-07-02.md | 2 +- kb/audits/prometheus-cutover-2026-07-06.md | 10 +++---- kb/decisions/ai-cluster-legacy.md | 2 +- kb/decisions/architektura-2026-07-28.md | 6 ++--- kb/decisions/backlog-aktywne.md | 12 ++++----- .../backlog-m1-solaria-prune-mitigation.md | 2 +- .../backlog-rozjazdy-repo-rzeczywistosc.md | 6 ++--- kb/decisions/backlog-zamkniete.md | 6 ++--- kb/decisions/deploy-runner-uzasadnienie.md | 2 +- kb/decisions/ha-configs-as-code.md | 10 +++---- kb/decisions/kb-dokumenty-otwarte.md | 6 ++--- kb/decisions/paperless-split-ocr.md | 2 +- kb/decisions/tech-debt-legacy.md | 2 +- .../2026-07-12-paperless-worker-config.md | 2 +- ...26-07-16-ollama-solaria-brak-sterownika.md | 6 ++--- kb/incidents/2026-07-22-ha-dwie-instancje.md | 2 +- .../2026-07-22-ha-ken-cutover-legacy.md | 2 +- kb/phases/ha-configs-as-code.md | 2 +- kb/phases/kb-m0-piha-slim.md | 4 +-- kb/phases/kb-m5-documents-ingest-fazy.md | 10 +++---- kb/phases/kb-m5-faza2.md | 2 +- kb/phases/kb-m5-faza3.md | 4 +-- kb/phases/kb-m5-faza4-fallback-dedup.md | 10 +++---- kb/phases/kb-m5-faza4.md | 8 +++--- kb/phases/monitoring-floty-prometheus.md | 2 +- kb/phases/prometheus-cutover-etap2.md | 2 +- kb/phases/subsystem-a-naprawa.md | 2 +- kb/runbooks/ha-diag-agent-runbook.md | 2 +- kb/runbooks/kb-query-deploy.md | 2 +- kb/runbooks/ollama-solaria-cutover.md | 4 +-- kb/runbooks/paperless-cutover.md | 2 +- kb/runbooks/paperless-worker-deploy.md | 2 +- kb/services/control-plane.md | 3 ++- kb/services/ha-mcp.md | 6 ++--- kb/services/home-assistant-ken-legacy.md | 2 +- kb/services/job-documents-ingest.md | 4 +-- kb/services/job-gmail-header-backfill.md | 2 +- kb/services/job-mail-body-ingest.md | 2 +- kb/services/kb-postgres.md | 2 +- kb/services/kb-query.md | 2 +- kb/services/mosquitto.md | 4 +-- kb/services/nextcloud.md | 2 +- kb/services/ollama-piha.md | 2 +- kb/services/paperless-worker.md | 2 +- kb/services/paperless.md | 2 +- kb/subsystems/agent-operating-procedures.md | 4 +-- kb/subsystems/control-plane.md | 1 + kb/subsystems/fleet-inventory-verify.md | 2 +- kb/subsystems/recon-multiagent.md | 2 +- packages/kb-mail/src/kb_mail/chunking.py | 2 +- packages/kb-mail/tests/test_migration.py | 2 +- .../kb-retrieval/src/kb_retrieval/embed.py | 4 +-- .../src/kb_retrieval/retrieval.py | 8 +++--- packages/kb-retrieval/tests/test_retrieval.py | 2 +- scripts/ha/deploy.sh | 2 +- scripts/ha/import.sh | 2 +- scripts/ha/lib/deploy_api.py | 2 +- scripts/ha/lib/ha_api.py | 2 +- scripts/ha/lib/ha_write_api.py | 2 +- scripts/ha/lib/ha_ws.py | 4 +-- scripts/ha/lib/normalize.py | 2 +- scripts/ha/lib/split.py | 2 +- scripts/kb/check_okf.py | 3 +-- scripts/npm/npm_api.py | 2 +- scripts/observer/observer.py | 2 +- services/control-plane/src/executor.py | 6 ++--- services/control-plane/src/supervisor.py | 4 +-- .../control-plane/tests/test_dormant_nodes.py | 2 +- .../tests/test_executor_dispatch.py | 2 +- services/fleet-prometheus/rules/kb-ingest.yml | 2 +- services/ha-mcp/run.sh | 2 +- services/ha-mcp/src/ha_mcp/server.py | 2 +- services/kb-query/app/embed_router.py | 2 +- services/kb-query/app/links.py | 2 +- services/kb-query/app/main.py | 4 +-- services/kb-query/app/search.py | 2 +- services/kb-query/app/startup.py | 2 +- services/nextcloud/docker-compose.yml | 2 +- services/nextcloud/service.yaml | 2 +- services/node-agent/src/node_agent.py | 8 +++--- .../node-agent/tests/test_action_dispatch.py | 2 +- services/node_exporter/service.yaml | 4 +-- services/ollama/docker-compose.yml | 2 +- services/paperless/docker-compose.yml | 4 +-- services/stability-agent/service.yaml | 2 +- 125 files changed, 221 insertions(+), 220 deletions(-) diff --git a/.claude/skills/node-onboarding/SKILL.md b/.claude/skills/node-onboarding/SKILL.md index 703c2f4..0123931 100644 --- a/.claude/skills/node-onboarding/SKILL.md +++ b/.claude/skills/node-onboarding/SKILL.md @@ -93,7 +93,7 @@ services: preflight fills `arch`, `ram_mb`, `docker_present`, `mm_runtime` — do NOT guess these. -Full schema: `scripts/onboard/README.md`. +Full schema: `kb/runbooks/node-onboarding-tool.md`. --- diff --git a/CLAUDE.md b/CLAUDE.md index 1029ec9..7a090b8 100644 --- a/CLAUDE.md +++ b/CLAUDE.md @@ -40,7 +40,7 @@ Pipeline stages: **prepare → validate → deploy → verify → diagnose (on f ## Node Onboarding New nodes are onboarded via `scripts/onboard/` — an idempotent bash tool driven by -`hosts//node.yaml` manifests (no Ansible). See `scripts/onboard/README.md` for +`hosts//node.yaml` manifests (no Ansible). See `kb/runbooks/node-onboarding-tool.md` for the full schema, step status table, and gotchas. Key fields in `node.yaml`: `ssh_user`, `first_contact` (LAN IP — not `.local`), @@ -93,7 +93,7 @@ Agent → /opt/homelab/actions/pending/.json → Executor dispatches to the target node → completed / failed ``` -The executor never connects to a node (deliberate — see docs/backlog.md +The executor never connects to a node (deliberate — see kb/phases/backlog.md "Remediacja floty bez SSH"). It writes a dispatch file that the node collects: | Action type | Inbox | Executed on the node by | diff --git a/README.md b/README.md index 752596a..6c42679 100644 --- a/README.md +++ b/README.md @@ -31,29 +31,29 @@ Action approval flow: `pending/` → operator approves → `approved/` → execu ## Repository Structure -- `docs/`: [Infrastructure Standards](docs/standards.md) and [Deployment Conventions](docs/deployment.md). -- `docs/architecture/PLAN-subsystem-a-2026-07-28.md`: [Current Maintenance Plan (Control Plane)](docs/architecture/PLAN-subsystem-a-2026-07-28.md). +- `docs/`: [Infrastructure Standards](kb/subsystems/standards.md) and [Deployment Conventions](kb/subsystems/deployment.md). +- `kb/phases/subsystem-a-naprawa.md`: [Current Maintenance Plan (Control Plane)](kb/phases/subsystem-a-naprawa.md). - `hosts/`: Host-specific configurations and service assignments. - `services/`: Reusable Docker Compose service definitions. - `scripts/`: Deployment and management scripts. ## Getting Started -1. **Standardization**: Follow the [Infrastructure Standards](docs/standards.md). -2. **Deployment**: See [Deployment Conventions](docs/deployment.md) for how to roll out changes. +1. **Standardization**: Follow the [Infrastructure Standards](kb/subsystems/standards.md). +2. **Deployment**: See [Deployment Conventions](kb/subsystems/deployment.md) for how to roll out changes. 3. **SATURN**: Remember that SATURN is the only node where commits should be made. ## Documentation Index -- [Current Maintenance Plan (Control Plane)](docs/architecture/PLAN-subsystem-a-2026-07-28.md) -- [Infrastructure Standards](docs/standards.md) -- [Agent Operating Procedures](docs/agents.md) (For AI/Non-Human Agents) -- [Deployment Conventions](docs/deployment.md) -- [Hardware](docs/hardware.md) -- [Networking](docs/networking.md) -- [Services](docs/services.md) -- [Node Capabilities](docs/capabilities.md) -- [Action Model](services/agent-system/action-model.md) +- [Current Maintenance Plan (Control Plane)](kb/phases/subsystem-a-naprawa.md) +- [Infrastructure Standards](kb/subsystems/standards.md) +- [Agent Operating Procedures](kb/subsystems/agent-operating-procedures.md) (For AI/Non-Human Agents) +- [Deployment Conventions](kb/subsystems/deployment.md) +- [Hardware](kb/nodes/legacy-hardware.md) +- [Networking](kb/subsystems/networking.md) +- [Services](kb/subsystems/legacy-services-list.md) +- [Node Capabilities](kb/subsystems/capability-model.md) +- [Action Model](kb/subsystems/action-approval-model.md) --- *Note: This repository documents the state of the homelab. Runtime state lives outside the repository in `/opt/homelab`.* diff --git a/docs/sessions/2026-06-08-lustro-onboarding.md b/docs/sessions/2026-06-08-lustro-onboarding.md index 61896f3..2f63de2 100644 --- a/docs/sessions/2026-06-08-lustro-onboarding.md +++ b/docs/sessions/2026-06-08-lustro-onboarding.md @@ -90,7 +90,7 @@ przez Tailscale działa bezhasłowo. Verify czysty (arch=aarch64). ## Learnings -(odzwierciedlone też w `scripts/onboard/README.md`) +(odzwierciedlone też w `kb/runbooks/node-onboarding-tool.md`) - mDNS `.local` zawodny do automatyzacji → `first_contact` przez IP lub tailscale, nie `.local` - istniejący node z userem uid=1000: użyj go zamiast tworzyć `oskar` (kolizja uid) diff --git a/docs/sessions/2026-06-09-flota-recovery-lustro-register.md b/docs/sessions/2026-06-09-flota-recovery-lustro-register.md index 90e8015..ecaf4ae 100644 --- a/docs/sessions/2026-06-09-flota-recovery-lustro-register.md +++ b/docs/sessions/2026-06-09-flota-recovery-lustro-register.md @@ -130,4 +130,4 @@ Docelowo: osobny worktree per task. ## Tech-debt złapany w sesji -→ wpisany do `docs/backlog.md` +→ wpisany do `kb/phases/backlog.md` diff --git a/docs/sessions/2026-06-11-lustro-ssh-shipping.md b/docs/sessions/2026-06-11-lustro-ssh-shipping.md index e6e8d2e..93bb072 100644 --- a/docs/sessions/2026-06-11-lustro-ssh-shipping.md +++ b/docs/sessions/2026-06-11-lustro-ssh-shipping.md @@ -79,13 +79,13 @@ solaria / piha / chelsty to wciąż **stare root kontenery** node-agenta (piha Created 2026-05-27, uid 0). Ich mount `/root/.ssh` działa tylko dlatego, że kontenery są sprzed `user: "1000:1000"`. Pierwszy `--force-recreate` / reboot hosta / update obrazu przełączy je na uid 1000 i shipping padnie jak na lustrze. -**NIE RECREATE bez fixu.** Szczegóły i fix: `docs/backlog.md`. +**NIE RECREATE bez fixu.** Szczegóły i fix: `kb/phases/backlog.md`. --- ## Tech-debt złapany w sesji -→ wpisany do `docs/backlog.md` (flota-bomba, ha-diag-agent blocked, +→ wpisany do `kb/phases/backlog.md` (flota-bomba, ha-diag-agent blocked, poison-quarantine review, `--omit-dir-times`, stale komentarz node_agent.py, shipping success na `logger.debug`, event-bloat lustro na VPS). @@ -96,8 +96,8 @@ fa59625 docs(ha-diag-agent): replace curl verify commands with docker exec d7e0d31 fix(ha-diag-agent): remove host port mapping for 8087 ### Files changed - services/ha-diag-agent/DEPLOY.md | 4 ++-- - services/ha-diag-agent/README.md | 4 ++-- + kb/runbooks/ha-diag-agent-deploy.md | 4 ++-- + kb/services/ha-diag-agent.md | 4 ++-- services/ha-diag-agent/docker-compose.yml | 3 --- services/ha-diag-agent/service.yaml | 3 --- 4 files changed, 4 insertions(+), 10 deletions(-)) diff --git a/docs/sessions/2026-06-17-kb-foundations.md b/docs/sessions/2026-06-17-kb-foundations.md index a3ae8fd..943e71a 100644 --- a/docs/sessions/2026-06-17-kb-foundations.md +++ b/docs/sessions/2026-06-17-kb-foundations.md @@ -120,14 +120,14 @@ services/kb-postgres/service.yaml services/kb-postgres/env.example services/kb-postgres/healthcheck.sh services/kb-postgres/init/001_envelope.sql -services/kb-postgres/README.md +kb/services/kb-postgres.md hosts/solaria/runtime/kb-postgres/docker-compose.override.yml hosts/solaria/services.yaml inventory/topology.yaml packages/kb-mail/pyproject.toml packages/kb-mail/src/kb_mail/{__init__,envelope,db,archive}.py packages/kb-mail/tests/{conftest,test_envelope,test_archive,test_db,test_migration}.py -docs/kb/kb-00-overview.md (etap 1 done, konwencja packages/) +kb/subsystems/kb-overview.md (etap 1 done, konwencja packages/) CLAUDE.md (sekcja Shared Python Libraries) docs/sessions/2026-06-17-kb-foundations.md ``` diff --git a/docs/sessions/2026-06-24-kb-gmail-importer.md b/docs/sessions/2026-06-24-kb-gmail-importer.md index 540c205..d985cb5 100644 --- a/docs/sessions/2026-06-24-kb-gmail-importer.md +++ b/docs/sessions/2026-06-24-kb-gmail-importer.md @@ -171,7 +171,7 @@ jobs/gmail-bulk-import/pyproject.toml jobs/gmail-bulk-import/tests/test_importer.py hosts/piha/capabilities.yaml .gitignore -docs/kb/kb-00-overview.md (etap 2 gotowy, konwencja jobs/) -docs/kb/kb-01-email-design.md (§8 krok 2 = kod gotowy) +kb/subsystems/kb-overview.md (etap 2 gotowy, konwencja jobs/) +kb/subsystems/kb-mail-pillar.md (§8 krok 2 = kod gotowy) docs/sessions/2026-06-24-kb-gmail-importer.md ``` diff --git a/docs/sessions/2026-06-24.md b/docs/sessions/2026-06-24.md index 6af6d22..7f6f627 100644 --- a/docs/sessions/2026-06-24.md +++ b/docs/sessions/2026-06-24.md @@ -108,7 +108,7 @@ docker compose \ --- -## Nowe tech-debty (dodane do `docs/backlog.md`) +## Nowe tech-debty (dodane do `kb/phases/backlog.md`) 1. **Rozjazd stanu Docker Compose na VPS** — serwisy `node-agent`, `control-plane` i inne stworzone innym `project-name` niż `deploy-node.sh` oczekuje; Recreate pada na stale diff --git a/docs/sessions/2026-06-30-fleet-inventory.md b/docs/sessions/2026-06-30-fleet-inventory.md index d6ffed7..fae392e 100644 --- a/docs/sessions/2026-06-30-fleet-inventory.md +++ b/docs/sessions/2026-06-30-fleet-inventory.md @@ -19,7 +19,7 @@ architekturą dokumentów KB. Start od weryfikacji stanu faktycznego wszystkich - CC (Sonnet 4.6) w worktree `fleet-inventory` zebrał stan faktyczny (docker ps + free/df/nproc) z 4 dostępnych nodów: PIHA, VPS, SOLARIA, SATURN. LUSTRO+CHELSTY offline (timeout :22) -> oznaczone UNREACHABLE. -- Wynik: `docs/infra/inventory-2026-06-30.md` — 23 zpriorytetyzowane rozjazdy. +- Wynik: `kb/subsystems/fleet-inventory.md` — 23 zpriorytetyzowane rozjazdy. - Kluczowe ustalenia: - **forgejo** biega na PIHA (always-on), ale `service.yaml owner_node=saturn` — rozjazd - **mosquitto** biega na VPS, `service.yaml owner=piha`, na PIHA go nie ma diff --git a/docs/sessions/2026-07-02-modul0-cert-migracja.md b/docs/sessions/2026-07-02-modul0-cert-migracja.md index 6f0a9ec..6be4c67 100644 --- a/docs/sessions/2026-07-02-modul0-cert-migracja.md +++ b/docs/sessions/2026-07-02-modul0-cert-migracja.md @@ -10,7 +10,7 @@ links: [] # Sesja 2026-07-02 — Modul 0 (odchudzenie PIHA) + migracja Forgejo/Vikunja na kapala.org ## Modul 0 kb-02 — WYKONANY (faza 1 audyt + faza 2 egzekucja) -- Audyt CC (read-only): docs/infra/piha-slim-audit-2026-07-02.md +- Audyt CC (read-only): kb/audits/piha-slim-2026-07-02.md - Review Oskara skorygowal audyt: llm-gateway = WLASNY kod (FastAPI-router LLM, /opt/llm-gateway, proxy do Ollama@SOLARIA) — NIE martwy; immich MUSI byc 24/7 na PIHA (SOLARIA sesyjna) — rekomendacja przeniesienia wykreslona. diff --git a/docs/sessions/2026-07-02.md b/docs/sessions/2026-07-02.md index bfe17f3..5980274 100644 --- a/docs/sessions/2026-07-02.md +++ b/docs/sessions/2026-07-02.md @@ -21,7 +21,7 @@ Sesja tylko-recon + minimalne fixy; bez deployów nowych feature'ów. ### Recon-weryfikacja inwentaryzacji floty (commit `57a6dff`, read-only) -Wynik: `docs/infra/inventory-verify-2026-07-02.md`. +Wynik: `kb/subsystems/fleet-inventory-verify.md`. **Bilans 23 rozjazdów z audytu 2026-06-30**: - **20 wciąż aktualnych** — nic się samo nie naprawiło. @@ -100,7 +100,7 @@ Po jednej linii per plik; `owner_node` nie występował nigdzie indziej w repo. nie rezolwuje z SOLARII; brak formalnego override mem_limit fleet-prometheus w `hosts/vps/runtime/` (siedzi w bazowym compose — kosmetyka). -Wpisy dodane do `docs/backlog.md` w tej sesji. +Wpisy dodane do `kb/phases/backlog.md` w tej sesji. --- diff --git a/docs/sessions/2026-07-06.md b/docs/sessions/2026-07-06.md index 1d8e9c6..e63e41d 100644 --- a/docs/sessions/2026-07-06.md +++ b/docs/sessions/2026-07-06.md @@ -22,7 +22,7 @@ Prometheus → brain-watchdog → Telegram, którego brakowało od 2026-06-30. ### Recon cutoveru — wmergowany (commit `d94bb38`) -Wynik: `docs/infra/prometheus-cutover-recon-2026-07-06.md` (517 linii, read-only, +Wynik: `kb/audits/prometheus-cutover-2026-07-06.md` (517 linii, read-only, zero zmian w kodzie). Kluczowe ustalenia: - **Cutover to podmiana klasyfikacji liveności w JEDNYM miejscu** — diff --git a/docs/sessions/2026-07-12-deploy2-ocr-worker.md b/docs/sessions/2026-07-12-deploy2-ocr-worker.md index 20619e9..4c034a3 100644 --- a/docs/sessions/2026-07-12-deploy2-ocr-worker.md +++ b/docs/sessions/2026-07-12-deploy2-ocr-worker.md @@ -46,7 +46,7 @@ maila = ta sama encja, DOWOD zasady kb-00 #7), (3) interfejs pytan (RAG) — pie realnej uzytecznosci. Dopiero POTEM dopelniac importy (reszta Takeout, zdjecia, transakcje). ## TODO nastepne -- MODUL 5 (koperta + ingest + embeddingi + cross-source) — docs/kb/modules/05-documents-ingest.md +- MODUL 5 (koperta + ingest + embeddingi + cross-source) — kb/phases/kb-m5-documents-ingest.md - Import probki zalacznikow z maili (kilkaset, nie 70k) — do zbudowania RAG - Interfejs pytan / RAG — warstwa uzytkowa - Deploy 3 (Nextcloud), Deploy 4 (Gokapi) — configi gotowe, czekaja diff --git a/docs/sessions/2026-07-15.md b/docs/sessions/2026-07-15.md index 92d696a..f1f83bf 100644 --- a/docs/sessions/2026-07-15.md +++ b/docs/sessions/2026-07-15.md @@ -27,12 +27,12 @@ parsowany z nazwy `evt---...` (fallback mtime, **nigdy 0** dla istniejącego pliku — 0 = leksykalne "starszy niż checkpoint" = dokładnie ten poison), migracja starych path-checkpointów przy starcie. Zdeployowany na VPS (observer `StartedAt` 07-14). Zweryfikowany dziś jako kompletny i zdeployowany. -Szczegóły: `docs/backlog.md` (sekcja "Bug: checkpoint observera po ścieżce +Szczegóły: `kb/phases/backlog.md` (sekcja "Bug: checkpoint observera po ścieżce leksykalnej"). ### 2. docs(infra) analiza Etapu 2 shadow-run (Fable, `8fec62d`) -`docs/infra/prometheus-shadow-etap2-analiza-2026-07-15.md` — 165 mismatchy +`kb/phases/prometheus-cutover-etap2.md` — 165 mismatchy `SHADOW_LIVENESS_MISMATCH` solaria/lustro w dobie 2026-07-14 WYJAŚNIONE: dwa nocne wyłączenia węzłów (lustro 21:30 UTC — regularny power-off, solaria 21:34 UTC). Wzorzec `event=fresh prom=down` to **nie** "żywy węzeł niewidziany diff --git a/docs/sessions/2026-07-16.md b/docs/sessions/2026-07-16.md index 59cafa7..79832e1 100644 --- a/docs/sessions/2026-07-16.md +++ b/docs/sessions/2026-07-16.md @@ -31,7 +31,7 @@ links: [] ### 1. RECON lustro shipping (Fable, commit `542bba4`) -`docs/infra/lustro-shipping-recon-2026-07-16.md` — 1507 mismatchy `lustro event=dead prom=up` w trwałym logu WYJAŚNIONE: (a) wczorajszy kontrolowany test (node-agent stał 3h20m, nie 15 min jak zakładano) + (b) poranny boot-race 56s. **Werdykt: shipping lustro działa, ZERO recurring problemu.** Prometheus 0 pomyłek w 48h — wzmacnia rekomendację GO dla Etapu 3 cutoveru. +`kb/audits/lustro-shipping-2026-07-16.md` — 1507 mismatchy `lustro event=dead prom=up` w trwałym logu WYJAŚNIONE: (a) wczorajszy kontrolowany test (node-agent stał 3h20m, nie 15 min jak zakładano) + (b) poranny boot-race 56s. **Werdykt: shipping lustro działa, ZERO recurring problemu.** Prometheus 0 pomyłek w 48h — wzmacnia rekomendację GO dla Etapu 3 cutoveru. Znaleziska poboczne: lustro biega na obrazie sprzed 5 tyg (deploy-node bez `--build` — patrz fix #2 niżej); fake-hwclock boot-race (RPi bez RTC — pierwszy event po boocie ma stary stempel, dropnięty przez timestamp checkpoint). diff --git a/docs/sessions/2026-07-23-control-plane-remediation-e2e.md b/docs/sessions/2026-07-23-control-plane-remediation-e2e.md index b3783b3..17fe103 100644 --- a/docs/sessions/2026-07-23-control-plane-remediation-e2e.md +++ b/docs/sessions/2026-07-23-control-plane-remediation-e2e.md @@ -70,7 +70,7 @@ approval → executor → node-agent → docker restart → completed. test E2E padł: agent rzucał `[Errno 13] Permission denied: /opt/homelab/actions/dispatch` co cykl. Root cause to ZNANY, POWRACAJĄCY (już 4. raz — patrz sekcja "Tech-debt: globalny porządek uid/gid/uprawnień - we flocie" w `docs/backlog.md`) motyw + we flocie" w `kb/phases/backlog.md`) motyw uid/gid na PIHA: oskar ma uid 1004, kontener agenta biega jako uid 1000 (= user `pi` na hoście). `/opt/homelab/actions` było `oskar:oskar drwxr-xr-x` (utworzone w maju), podczas gdy DZIAŁAJĄCY wzorzec to @@ -111,7 +111,7 @@ approval → executor → node-agent → docker restart → completed. Pierwszy w historii systemu pełny cykl remediacji end-to-end potwierdzony w produkcji (PIHA), bez SSH z control-plane do węzłów. Publiczna dziura autoryzacji na `operator_ui.py:18180` zamknięta (bind ograniczony do -Tailscale). Otwarte follow-upy — patrz `docs/backlog.md` (retry-w-nieskończoność +Tailscale). Otwarte follow-upy — patrz `kb/phases/backlog.md` (retry-w-nieskończoność zepsutego JSON, uprawnienia `actions/` na innych węzłach, brak twardego checka `.env`/`TAILSCALE_BIND_IP` w `deploy-local.sh`, brak autoryzacji w `operator_ui.py`, zapchana approval queue przez `alert_only`, brak diff --git a/docs/sessions/2026-07-23-kb-f4-ingress.md b/docs/sessions/2026-07-23-kb-f4-ingress.md index 95c1b85..3cfbbd9 100644 --- a/docs/sessions/2026-07-23-kb-f4-ingress.md +++ b/docs/sessions/2026-07-23-kb-f4-ingress.md @@ -9,7 +9,7 @@ links: [] # Sesja 2026-07-23 — KB faza 4: ingress kb.kapala.org (krok 5/§8) -**Zakres**: wyłącznie ingress (`docs/kb/modules/05-faza4-plan.md` §8, krok 5) — +**Zakres**: wyłącznie ingress (`kb/phases/kb-m5-faza4.md` §8, krok 5) — frontend i `/search` już LIVE na PIHA (port 8230) od sesji 2026-07-22. Zero zmian w kodzie kb-query w tej sesji. @@ -70,8 +70,8 @@ w tej samej sesji, osobnym przebiegiem po zgłoszeniu przez operatora: Potwierdzone w repo (zgodnie z `05-faza4-plan.md` §1.4): **brak wzorca forward-auth/reverse-proxy-level auth** — NPM community edition go nie ma (sprawdzone: brak `oauth2-proxy`/`authelia`/`forward_auth` w kodzie repo poza -wzmiankami "przyszła opcja" w `docs/kb/kb-02-documents-design.md` i -`hosts/vps/README.md`). Wszystkie 3 precedensy (paperless/nextcloud/vikunja) +wzmiankami "przyszła opcja" w `kb/subsystems/kb-documents-pillar.md` i +`kb/nodes/vps.md`). Wszystkie 3 precedensy (paperless/nextcloud/vikunja) robią OIDC **wewnątrz aplikacji**. kb-query nie ma dziś żadnego logowania. Zgodnie z instrukcją zadania: **nie budowano** nowego komponentu auth. @@ -122,7 +122,7 @@ username collision, `docs/sessions/2026-07-10-paperless-deploy.md`). ## Pliki repo zmienione -- `services/kb-query/README.md` — sekcja "Ingress" (co żyje, co nie, dlaczego +- `kb/services/kb-query.md` — sekcja "Ingress" (co żyje, co nie, dlaczego auth odłożone) zastępuje starą notatkę "not wired up yet". - `docs/sessions/2026-07-23-kb-f4-ingress.md` — ten dokument. diff --git a/docs/sessions/2026-07-27-kb-f4-fallback.md b/docs/sessions/2026-07-27-kb-f4-fallback.md index 34fc311..f72a058 100644 --- a/docs/sessions/2026-07-27-kb-f4-fallback.md +++ b/docs/sessions/2026-07-27-kb-f4-fallback.md @@ -9,7 +9,7 @@ links: [] # Sesja 2026-07-27 — KB faza 4: fallback embed SOLARIA→PIHA (krok 3, ostatni element rdzenia) -> **Dopisek redakcyjny (2026-07-30, dedup — `docs/kb/modules/05-fallback-dedup-raport.md`):** +> **Dopisek redakcyjny (2026-07-30, dedup — `kb/phases/kb-m5-faza4-fallback-dedup.md`):** > implementacja kodu z tej sesji (`app/fallback.py`, branch `task/kb-f4-fallback`, 3d4ee38) > została **porzucona** — do mastera weszła równoległa, szersza implementacja tego samego > kroku planu (e7625cd, `app/embed_router.py`, 2026-07-29) i to ona biega na PIHA. Ten log @@ -23,7 +23,7 @@ links: [] > Z delty brancha uratowano ponadto: `retrieval_eval.py --transport http` (plan §2 D6/§9) > i luki testowe T1/T2 przeniesione do `test_embed_router.py`. -**Zakres**: `docs/kb/modules/05-faza4-plan.md` §2 decyzja 2 / §5 — aktywny fallback +**Zakres**: `kb/phases/kb-m5-faza4.md` §2 decyzja 2 / §5 — aktywny fallback embedu, ostatni brakujący element rdzenia fazy 4 (frontend i ingress LIVE od 2026-07-22/23, `docs/sessions/2026-07-23-kb-f4-ingress.md`). Zero zmian w schemacie DB, zero zmian w `kb_retrieval`'s retrieval logice — wyłącznie warstwa embed + health. diff --git a/docs/sessions/2026-07-28.md b/docs/sessions/2026-07-28.md index 675fbf5..f11f665 100644 --- a/docs/sessions/2026-07-28.md +++ b/docs/sessions/2026-07-28.md @@ -19,8 +19,8 @@ e8aa3e3 docs(architecture): recon multiagent 2026-07-27 ### Files changed ``` - docs/architecture/PLAN-subsystem-a-2026-07-28.md | 60 +++ - docs/architecture/RECON-multiagent-2026-07-27.md | 551 +++++++++++++++++++++++ + kb/phases/subsystem-a-naprawa.md | 60 +++ + kb/subsystems/recon-multiagent.md | 551 +++++++++++++++++++++++ 2 files changed, 611 insertions(+) ``` diff --git a/docs/sessions/2026-07-31-kb-f4-final-narty27.md b/docs/sessions/2026-07-31-kb-f4-final-narty27.md index 96ac9e6..6c09e1b 100644 --- a/docs/sessions/2026-07-31-kb-f4-final-narty27.md +++ b/docs/sessions/2026-07-31-kb-f4-final-narty27.md @@ -22,7 +22,7 @@ pilota fazy 5 (narty27), wykonanego równolegle. - session log 27.07 - testy luk T1–T3 - komentarz kalibracji progów dist. -- Pełny rozbiór obu implementacji: `docs/kb/modules/05-fallback-dedup-raport.md`. +- Pełny rozbiór obu implementacji: `kb/phases/kb-m5-faza4-fallback-dedup.md`. ### Deploy na PIHA (z mastera) - `ollama-piha`: named volume `ollama_piha_models`, model bge-m3, `KEEP_ALIVE=0`. @@ -76,7 +76,7 @@ infry**. Infra: `services/narty27`. stubie nie przechodzi wyłącznie z tego powodu. 2. **`hosts/solaria/runtime/ollama/docker-compose.override.yml` — brak w repo** (rozjazd repo↔runtime na SOLARII). -3. **R1–R3 node-agent** (incydent `docs/incidents/2026-07-30-ollama-solaria-vanish.md`) +3. **R1–R3 node-agent** (incydent `kb/incidents/2026-07-30-ollama-solaria-vanish.md`) — **niezrobione**: R1 `_prune_stopped_containers` nie może kasować kontenerów zarządzanych, R2 rate-limit dla `ai_node`/`standard`, R3 logowanie usuniętych zasobów. Przyczyna nadal aktywna → blokuje/warunkuje fazę mailową diff --git a/hosts/piha/runtime/node_exporter/docker-compose.override.yml b/hosts/piha/runtime/node_exporter/docker-compose.override.yml index b261a10..8291c71 100644 --- a/hosts/piha/runtime/node_exporter/docker-compose.override.yml +++ b/hosts/piha/runtime/node_exporter/docker-compose.override.yml @@ -1,6 +1,6 @@ # PIHA-specific override for node_exporter. # -# WHY: KB module 5 phase 3 step 5 (docs/kb/modules/05-faza3-plan.md §7.2) needs the +# WHY: KB module 5 phase 3 step 5 (kb/phases/kb-m5-faza3.md §7.2) needs the # textfile collector so kb-ingest's systemd timer can publish # kb_ingest_last_success_timestamp / kb_ingest_last_exit_code / kb_ingest_embed_backlog # etc. for fleet-prometheus to scrape and alert on. diff --git a/hosts/piha/services.yaml b/hosts/piha/services.yaml index 81f6b53..74f6c11 100644 --- a/hosts/piha/services.yaml +++ b/hosts/piha/services.yaml @@ -84,7 +84,7 @@ services: external: [] runtime: # textfile collector reads /opt/homelab/state/node-exporter (module 5 phase 3 step 5, - # docs/kb/modules/05-faza3-plan.md §7.2 — kb-ingest.prom) via the existing /:/host:ro + # kb/phases/kb-m5-faza3.md §7.2 — kb-ingest.prom) via the existing /:/host:ro # mount, see hosts/piha/runtime/node_exporter/docker-compose.override.yml. data_path: /opt/homelab/state/node-exporter @@ -161,7 +161,7 @@ services: runtime: # No config and no secrets. Content is PERSONAL and deliberately outside # the repo — it lives only in the Docker named volume - # narty27_narty27_content, refreshed from SOLARIA (services/narty27/README.md). + # narty27_narty27_content, refreshed from SOLARIA (kb/runbooks/narty27-deploy.md). # No backup job, no /opt/homelab/data bind. config_path: services/narty27 diff --git a/hosts/solaria/runtime/node-agent/docker-compose.override.yml b/hosts/solaria/runtime/node-agent/docker-compose.override.yml index c603333..df9b609 100644 --- a/hosts/solaria/runtime/node-agent/docker-compose.override.yml +++ b/hosts/solaria/runtime/node-agent/docker-compose.override.yml @@ -1,7 +1,7 @@ # MITYGACJA TYMCZASOWA (M1) — założona 2026-08-04. # NODE_TYPE=lte_node wyłącza run_safe_cleanup() (niefiltrowany # `docker container prune`) na czas backfillu embed (faza mailowa KB). -# Incydent: docs/incidents/2026-07-30-ollama-solaria-vanish.md (§7, M1). +# Incydent: kb/incidents/2026-07-30-ollama-solaria-vanish.md (§7, M1). # Bez tego każdy zatrzymany kontener na SOLARII znika w ≤60 s — również taki # z `restart: unless-stopped`, zatrzymany świadomie przez operatora. # Warunek zdjęcia: R1 (filtrowanie prune po restart policy / labelu compose) diff --git a/hosts/vps/services.yaml b/hosts/vps/services.yaml index ea9f4d5..3dd3f09 100644 --- a/hosts/vps/services.yaml +++ b/hosts/vps/services.yaml @@ -167,5 +167,5 @@ services: # redis, mosquitto): legacy stack, runs UNMANAGED on vps and is scheduled # for retirement — its codex/* bus has been idle since 2026-06-09. # Deliberately NO entry here: legacy is not pulled into desired state. - # See docs/architecture/ai-cluster-LEGACY.md and - # docs/architecture/RECON-multiagent-2026-07-27.md (C9). + # See kb/decisions/ai-cluster-legacy.md and + # kb/subsystems/recon-multiagent.md (C9). diff --git a/jobs/deploy-runner/deploy-runner.sh b/jobs/deploy-runner/deploy-runner.sh index de7e0aa..0f94be4 100755 --- a/jobs/deploy-runner/deploy-runner.sh +++ b/jobs/deploy-runner/deploy-runner.sh @@ -1,7 +1,7 @@ #!/usr/bin/env bash # jobs/deploy-runner/deploy-runner.sh — host-side executor for `redeploy` actions. # -# WHY THIS EXISTS (docs/architecture/RECON-multiagent-2026-07-27.md D14/D15): +# WHY THIS EXISTS (kb/subsystems/recon-multiagent.md D14/D15): # the control-plane executor could never carry out a redeploy — it ran # deploy-node.sh inside its own container, a script that ignores its arguments # and expects a repo at ${HOME}/homelab-codex-ws that does not exist there (nor @@ -9,7 +9,7 @@ # dead-ended, which is the single biggest reason the self-healing loop had # 18 pending / 0 completed actions. # -# The fix keeps the architecture decision from docs/backlog.md ("Remediacja +# The fix keeps the architecture decision from kb/phases/backlog.md ("Remediacja # floty bez SSH"): the VPS never initiates a connection to a node. The executor # only WRITES an action file; this runner, on the target node, pulls it, runs # the deploy locally, and reports the outcome back through the existing event diff --git a/jobs/documents-ingest/eval/queries.yaml b/jobs/documents-ingest/eval/queries.yaml index be670fb..223d391 100644 --- a/jobs/documents-ingest/eval/queries.yaml +++ b/jobs/documents-ingest/eval/queries.yaml @@ -1,6 +1,6 @@ # Eval-set for the retrieval quality gate (module 5, phase 3, plan §6.2, -# docs/kb/modules/05-faza3-plan.md). Transcribed 1:1 from the pilot baseline, -# docs/kb/eval/retrieval-pilot-2026-07-16.md (read-only source -- this file is the versioned +# kb/phases/kb-m5-faza3.md). Transcribed 1:1 from the pilot baseline, +# kb/phases/kb-m5-eval-retrieval-pilot.md (read-only source -- this file is the versioned # copy the plan asked for, so the set stops living only in a session transcript). # # kind: @@ -80,7 +80,7 @@ queries: wpis, patrz historia sesji). Próg tej kontroli obniżony do 0.50 (z 0.55) właśnie z powodu tej znanej kolizji, żeby bramka nie płonęła co uruchomienie na nie-problemie. -# Faza mailowa (docs/kb/modules/05-faza-mailowa-plan.md, §8, Krok 5) -- bramka jakościowa dla +# Faza mailowa (kb/phases/kb-m5-faza-mailowa.md, §8, Krok 5) -- bramka jakościowa dla # treści mailowej wprowadzonej w Etapie A (ostatnie 12 miesięcy, plan §7 Krok 4). Wypełniona # przez operatora 2026-07-23 (5 zapytań "wiem że to mam w mailach z ostatniego roku"; M5 # odrzucone po weryfikacji, patrz N2 powyżej i historia sesji). expected_envelope celowo null diff --git a/jobs/documents-ingest/eval/retrieval_eval.py b/jobs/documents-ingest/eval/retrieval_eval.py index f0aa063..e370413 100644 --- a/jobs/documents-ingest/eval/retrieval_eval.py +++ b/jobs/documents-ingest/eval/retrieval_eval.py @@ -1,5 +1,5 @@ -"""Retrieval quality gate -- module 5, phase 3, plan §6.2 (docs/kb/modules/05-faza3-plan.md), -extended in faza mailowa Krok 5 (docs/kb/modules/05-faza-mailowa-plan.md, §8) to add the +"""Retrieval quality gate -- module 5, phase 3, plan §6.2 (kb/phases/kb-m5-faza3.md), +extended in faza mailowa Krok 5 (kb/phases/kb-m5-faza-mailowa.md, §8) to add the `hybrid` track once mail content exists in `document_chunk` (faza mailowa Krok 2/4). Read-only integration script (NOT collected by pytest -- it hits the live kb-postgres DB and @@ -343,7 +343,7 @@ def print_report( rows: list[dict], n_values: list[int], gate_result: dict, mail_rows: Optional[list[dict]] = None ) -> None: print("=" * 100) - print("RETRIEVAL QUALITY GATE -- plan §6.2 (docs/kb/modules/05-faza3-plan.md), " + print("RETRIEVAL QUALITY GATE -- plan §6.2 (kb/phases/kb-m5-faza3.md), " "+ hybrid/mail extension (05-faza-mailowa-plan.md §8)") print("=" * 100) header = f"{'id':<3} {'kind':<28} {'expected':<16} {'flat d1':>8} {'flat@3':>7}" diff --git a/jobs/documents-ingest/src/documents_ingest/chunk_embed.py b/jobs/documents-ingest/src/documents_ingest/chunk_embed.py index b6e27d8..79c8f2b 100644 --- a/jobs/documents-ingest/src/documents_ingest/chunk_embed.py +++ b/jobs/documents-ingest/src/documents_ingest/chunk_embed.py @@ -1,8 +1,8 @@ -"""Chunk + embed job — module 5 phase 2, plan step 6 (docs/kb/modules/05-faza2-plan.md, +"""Chunk + embed job — module 5 phase 2, plan step 6 (kb/phases/kb-m5-faza2.md, §6 step 6, decision 3). `chunk_text`/`hard_split`/`split_paragraphs`/`TARGET_CHARS`/`OVERLAP_CHARS` moved to -`kb_mail.chunking` in module 5 faza mailowa, Krok 0 (docs/kb/modules/05-faza-mailowa-plan.md, §3) +`kb_mail.chunking` in module 5 faza mailowa, Krok 0 (kb/phases/kb-m5-faza-mailowa.md, §3) so `jobs/mail-body-ingest` shares the exact same chunker instead of a copy-pasted drift; re-exported here unchanged so nothing importing them from this module breaks. @@ -23,7 +23,7 @@ Install (from repo root): pip install -e jobs/documents-ingest/ `embed_chunk`/`_vector_literal`/`DEFAULT_MODEL`/`DEFAULT_OLLAMA_URL` moved to -`kb_retrieval.embed` in module 5 phase 4 (docs/kb/modules/05-faza4-plan.md, §3, decision 1) so +`kb_retrieval.embed` in module 5 phase 4 (kb/phases/kb-m5-faza4.md, §3, decision 1) so `kb-query` (Docker service) can share the same client without pulling in this job's `anthropic` dependency; re-exported here unchanged so nothing importing them from this module breaks. diff --git a/jobs/documents-ingest/src/documents_ingest/cyclic_ingest.py b/jobs/documents-ingest/src/documents_ingest/cyclic_ingest.py index d0f086f..c4afaa6 100644 --- a/jobs/documents-ingest/src/documents_ingest/cyclic_ingest.py +++ b/jobs/documents-ingest/src/documents_ingest/cyclic_ingest.py @@ -1,4 +1,4 @@ -"""Cyclic ingest wrapper -- module 5 phase 3, plan step 5 (docs/kb/modules/05-faza3-plan.md, +"""Cyclic ingest wrapper -- module 5 phase 3, plan step 5 (kb/phases/kb-m5-faza3.md, §7). Orchestrates one run of the recurring ingest pipeline for `kb-ingest.timer` on PIHA: paperless_adapter.run() -- new source='paperless' envelopes diff --git a/jobs/documents-ingest/src/documents_ingest/extractor.py b/jobs/documents-ingest/src/documents_ingest/extractor.py index c1abbbf..853b605 100644 --- a/jobs/documents-ingest/src/documents_ingest/extractor.py +++ b/jobs/documents-ingest/src/documents_ingest/extractor.py @@ -1,6 +1,6 @@ """PDF attachment extractor — kb-postgres mail archive -> Paperless consume/. -Module 5 ("documents-ingest"), Phase 1 (docs/kb/modules/05-documents-ingest.md, +Module 5 ("documents-ingest"), Phase 1 (kb/phases/kb-m5-documents-ingest.md, section "Domkniecie dlugu z maili"): pull a sample of PDF attachments out of the Gmail .eml archive and drop them into Paperless' consume/ dir so Paperless does the OCR + correspondent-detection. This is NOT the Paperless/Nextcloud envelope diff --git a/jobs/documents-ingest/src/documents_ingest/paperless_adapter.py b/jobs/documents-ingest/src/documents_ingest/paperless_adapter.py index a13c823..561640d 100644 --- a/jobs/documents-ingest/src/documents_ingest/paperless_adapter.py +++ b/jobs/documents-ingest/src/documents_ingest/paperless_adapter.py @@ -1,4 +1,4 @@ -"""Paperless -> envelope adapter — module 5 phase 2 (docs/kb/modules/05-faza2-plan.md, +"""Paperless -> envelope adapter — module 5 phase 2 (kb/phases/kb-m5-faza2.md, §4.2-4.3, §6 step 5). Reads documents from the Paperless REST API (read-only — GET only, this job never diff --git a/jobs/documents-ingest/src/documents_ingest/retrieval.py b/jobs/documents-ingest/src/documents_ingest/retrieval.py index 8934076..6c45291 100644 --- a/jobs/documents-ingest/src/documents_ingest/retrieval.py +++ b/jobs/documents-ingest/src/documents_ingest/retrieval.py @@ -1,5 +1,5 @@ """Re-export shim -- the retrieval module moved to `packages/kb-retrieval/` in module 5 phase -4 (docs/kb/modules/05-faza4-plan.md, §3, decision 1) so both this job and `kb-query` (Docker +4 (kb/phases/kb-m5-faza4.md, §3, decision 1) so both this job and `kb-query` (Docker service) share one tested module. Kept here unchanged so nothing importing `documents_ingest.retrieval` breaks; new code should import `kb_retrieval.retrieval` directly. """ diff --git a/jobs/documents-ingest/src/documents_ingest/summarize.py b/jobs/documents-ingest/src/documents_ingest/summarize.py index c039ace..cadfb0b 100644 --- a/jobs/documents-ingest/src/documents_ingest/summarize.py +++ b/jobs/documents-ingest/src/documents_ingest/summarize.py @@ -1,4 +1,4 @@ -"""Summarize + tag job — module 5 phase 3, plan step 3 (docs/kb/modules/05-faza3-plan.md, +"""Summarize + tag job — module 5 phase 3, plan step 3 (kb/phases/kb-m5-faza3.md, §5 step 3, §2 decision 3: two-track A/B pilot). Pipeline: `document_chunk.text WHERE excluded_reason IS NULL ORDER BY chunk_index` per diff --git a/jobs/documents-ingest/systemd/kb-ingest-run.sh b/jobs/documents-ingest/systemd/kb-ingest-run.sh index c2f0bb3..ee38f1e 100755 --- a/jobs/documents-ingest/systemd/kb-ingest-run.sh +++ b/jobs/documents-ingest/systemd/kb-ingest-run.sh @@ -1,6 +1,6 @@ #!/usr/bin/env bash # kb-ingest-run.sh — thin launcher for kb-ingest.service (module 5 phase 3 step 5, -# docs/kb/modules/05-faza3-plan.md §7.1). +# kb/phases/kb-m5-faza3.md §7.1). # # All sequencing/tolerance/exit-code logic lives in documents-ingest-cyclic (Python, # jobs/documents-ingest/src/documents_ingest/cyclic_ingest.py) — this script only computes diff --git a/jobs/documents-ingest/tests/test_chunk_embed.py b/jobs/documents-ingest/tests/test_chunk_embed.py index 75a90bc..1a9a804 100644 --- a/jobs/documents-ingest/tests/test_chunk_embed.py +++ b/jobs/documents-ingest/tests/test_chunk_embed.py @@ -49,8 +49,8 @@ class TestVectorLiteral: class TestIsOcrJunk: - """Pilot cases from docs/kb/modules/05-faza3-plan.md §3.1 and the 2026-07-16 calibration - session (docs/kb/eval/retrieval-pilot-2026-07-16.md).""" + """Pilot cases from kb/phases/kb-m5-faza3.md §3.1 and the 2026-07-16 calibration + session (kb/phases/kb-m5-eval-retrieval-pilot.md).""" def test_clean_text_is_not_junk(self): text = ( diff --git a/jobs/gmail-header-backfill/src/gmail_header_backfill/backfill.py b/jobs/gmail-header-backfill/src/gmail_header_backfill/backfill.py index 1ea6b19..7cb46f9 100644 --- a/jobs/gmail-header-backfill/src/gmail_header_backfill/backfill.py +++ b/jobs/gmail-header-backfill/src/gmail_header_backfill/backfill.py @@ -1,4 +1,4 @@ -"""Gmail header backfill — one-shot job, module 5 phase 2 (docs/kb/modules/05-faza2-plan.md, §5). +"""Gmail header backfill — one-shot job, module 5 phase 2 (kb/phases/kb-m5-faza2.md, §5). Backfills `{"type": "headers", ...}` (§4.1) onto the 225 030 existing `source='gmail'` envelope rows in kb-postgres, which today carry only an attachment manifest. This is an diff --git a/jobs/mail-body-ingest/src/mail_body_ingest/ingest.py b/jobs/mail-body-ingest/src/mail_body_ingest/ingest.py index 594a566..5dcb67e 100644 --- a/jobs/mail-body-ingest/src/mail_body_ingest/ingest.py +++ b/jobs/mail-body-ingest/src/mail_body_ingest/ingest.py @@ -1,5 +1,5 @@ """Mail body ingest job — module 5, faza mailowa, plan Krok 2 -(docs/kb/modules/05-faza-mailowa-plan.md, §5). Second full pass over the gmail .eml archive +(kb/phases/kb-m5-faza-mailowa.md, §5). Second full pass over the gmail .eml archive (the first was gmail-bulk-import's manifest-only import): extracts inline body text that `_parse_attachments` deliberately skipped, chunks it, embeds it (batched), and appends `entities[type=threading]` — the only new writes are `document_chunk` INSERTs and an additive diff --git a/kb/audits/czujniki-2026-07-30.md b/kb/audits/czujniki-2026-07-30.md index 9f99edb..155118b 100644 --- a/kb/audits/czujniki-2026-07-30.md +++ b/kb/audits/czujniki-2026-07-30.md @@ -10,7 +10,7 @@ links: [] # Recon — node-agent vs stability-agent (2026-07-30) -Read-only recon. Ground truth: `docs/architecture/RECON-multiagent-2026-07-27.md` +Read-only recon. Ground truth: `kb/subsystems/recon-multiagent.md` (A1, A2, B7, D15). Source read at master @ `473bf8e`; this branch is cut from master @ `0650eb8`. The two intervening merges (`0650eb8` ha-mcp, `cb8a19d` kb-query tests) touch none of the audited paths — `services/node-agent/`, @@ -66,7 +66,7 @@ control plane, stability-agent feeds the agent-system UI, and stability-agent's half of the event store is a write-only archive nothing has ever read. A prior recon reached the same conclusion about the event path on 2026-07-06 -(`docs/infra/prometheus-cutover-recon-2026-07-06.md:88-91`); it has not been acted on. +(`kb/audits/prometheus-cutover-2026-07-06.md:88-91`); it has not been acted on. --- @@ -307,7 +307,7 @@ prefix that `container_service_name()` was written to strip. - Remove entries from `hosts/solaria/services.yaml` and `hosts/vps/services.yaml`. - `docker compose down` on vps, piha, solaria (chelsty-infra when reachable). - Purge or archive `/opt/homelab/events/2026-*/` on all four nodes (~5 MB on piha, mostly May). -- Update CLAUDE.md (agent-system architecture §1, event-path claim at line 100), `docs/chelsty-stability-agent.md`, recon A1/A2/B7. +- Update CLAUDE.md (agent-system architecture §1, event-path claim at line 100), `kb/services/chelsty-stability-agent.md`, recon A1/A2/B7. - **Not covered by the cleanup:** the Redis publisher must be rehomed first or the UI loss is permanent. ### (b) Merge stability-agent's unique checks into node-agent diff --git a/kb/audits/monitoring-coverage-2026-07-14.md b/kb/audits/monitoring-coverage-2026-07-14.md index 3a3572c..d2767ac 100644 --- a/kb/audits/monitoring-coverage-2026-07-14.md +++ b/kb/audits/monitoring-coverage-2026-07-14.md @@ -131,7 +131,7 @@ compose-service kontenera. | owntracks-prometheus-exporter-prometheus-owntracks-exporter-1 | linusgroh/prometheus-owntracks-exporter | Up 2w | 0.0.0.0:8780→80 | | own-tracks-frontend-owntracks-frontend-1 | owntracks/frontend | Up 2w | 0.0.0.0:8084→80 | -Zmiany vs audyt 2026-06-30 (`docs/infra/inventory-2026-06-30.md`): **przybyły** paperless, +Zmiany vs audyt 2026-06-30 (`kb/subsystems/fleet-inventory.md`): **przybyły** paperless, paperless-db, paperless-broker (Deploy 1, 2026-07-10); **zniknęły** diskover i elasticsearch (w audycie 06-30 były w 33 shadow; dziś nie biegają). 06-30: 40 kontenerów → dziś: 42. diff --git a/kb/audits/piha-slim-2026-07-02.md b/kb/audits/piha-slim-2026-07-02.md index 63fafa7..5b3465e 100644 --- a/kb/audits/piha-slim-2026-07-02.md +++ b/kb/audits/piha-slim-2026-07-02.md @@ -10,7 +10,7 @@ links: [] # Audyt odchudzania PIHA — 2026-07-02 -> Faza 1 (READ-ONLY) modulu 0 filaru dokumentow (`docs/kb/modules/00-piha-slim.md`). +> Faza 1 (READ-ONLY) modulu 0 filaru dokumentow (`kb/phases/kb-m0-piha-slim.md`). > Zadna akcja nie zostala wykonana — wylacznie `docker stats/inspect/logs`, `ss`, `curl` (odczyt). > Stan w momencie audytu: **RAM 7.9Gi total, 5.0Gi used, 2.9Gi available; swap 4Gi total, 2.0Gi uzyty.** > 41 kontenerow Up (inwentaryzacja 2026-06-30 liczyla 40; wszystkie nadal biega). diff --git a/kb/audits/prometheus-cutover-2026-07-06.md b/kb/audits/prometheus-cutover-2026-07-06.md index 18bb307..c7e62f3 100644 --- a/kb/audits/prometheus-cutover-2026-07-06.md +++ b/kb/audits/prometheus-cutover-2026-07-06.md @@ -300,7 +300,7 @@ renderuje `health`/`status`/`last_seen`), `/unhealthy` (`:302-334`), `/summary`, chelsty/chelsty-infra (exporter DOWN po LTE, 2026-06-26). - Reguła `NodeDown` (`rules/liveness.yml:21-28`): `up{node=~"vps|piha"} == 0`, `for: 5m`, severity critical. solaria/lustro świadomie wykluczone (`:12-16` — - planned power-off; docelowo anomaly detection, `docs/backlog.md:390-406`). + planned power-off; docelowo anomaly detection, `kb/phases/backlog.md:390-406`). Bez Alertmanagera by design (`:3-7`) — delivery = brain-watchdog poll `/api/v1/alerts`. - node_exporter w repo tylko dla VPS (`services/node_exporter/`, network_mode: host, `hosts/vps/services.yaml:36-43`). Exportery na piha/solaria/lustro **nie mają @@ -348,7 +348,7 @@ chelsty-infra, chelsty-ha, lustro. | lustro | TAK (`:50-52`) | NIE | TAK (bez stability-agenta) | jw. | | **chelsty-infra** | **NIE** (`:57-60`, exporter DOWN po LTE) | NIE | **TAK** (remote TTL 900/3600, `liveness.py:43-50`) | **tylko stary tor — cutover totalny zostawiłby go bez liveności** | | chelsty-ha | NIE | NIE | NIE (`hosts/chelsty-ha/services.yaml:6-12`, `monitor: false`) | już dziś bez liveności (pośrednio przez MQTT chelsty-infra) — cutover nic nie zmienia | -| saturn | NIE (`:55`, laptop) | NIE | NIE (brak `hosts/saturn/services.yaml`, `docs/backlog.md:423`) | już dziś bez liveności — cutover nic nie zmienia | +| saturn | NIE (`:55`, laptop) | NIE | NIE (brak `hosts/saturn/services.yaml`, `kb/phases/backlog.md:423`) | już dziś bez liveności — cutover nic nie zmienia | **Chelsty offline ~34 dni — jak traktuje go stara rura:** eventy buforują się lokalnie (rsync fail = non-fatal, `node_agent.py:560-566`), `last_seen` na VPS zamrożone sprzed @@ -361,7 +361,7 @@ totalnym chelsty-infra nie miałby żadnej liveności i żadnego przejścia offl Dodatkowo docs sygnalizują konflikt IP w komentarzach `prometheus.yml:57` vs `hosts/chelsty-infra/host.yaml:12` — do wyjaśnienia przy ewentualnym dodawaniu scrape. *(rzeczywisty bieżący stan chelsty — do weryfikacji na żywo; ostatni zapis: -UNREACHABLE, `docs/infra/inventory-verify-2026-07-02.md:17,151`)* +UNREACHABLE, `kb/subsystems/fleet-inventory-verify.md:17,151`)* **Wniosek twardy:** cutover NIE może być globalny. Docelowa architektura to **hybryda per-node**: `up{}` dla scrape'owanych (vps, piha, solaria, lustro), @@ -460,7 +460,7 @@ z `last_seen` — rzadszy heartbeat przy niezmienionych TTL-ach = fałszywe degr - chelsty-infra: zbadać exporter-over-LTE (`prometheus.yml:57-60` + konflikt IP z `hosts/chelsty-infra/host.yaml:12`); do tego czasu zostaje na torze eventowym. - NodeDown dla solaria/lustro: świadomie odroczone do anomaly detection - (`docs/backlog.md:390-406`) — nie wciągać do cutoveru. + (`kb/phases/backlog.md:390-406`) — nie wciągać do cutoveru. - Watchdog na sam Prometheus (D.2 pkt 5) — mały task przy etapie 3. - saturn / chelsty-ha: świadomie poza monitoringiem — status quo. @@ -468,7 +468,7 @@ z `last_seen` — rzadszy heartbeat przy niezmienionych TTL-ach = fałszywe degr **Seria.** Minimalnie trzy taski implementacyjne + weryfikacje między nimi: (1) etap 1 shadow-read; (2) etap 3 flaga per-node (po tygodniu etapu 2); -(3) watchdog-na-Prometheusa + aktualizacja `docs/observer-runtime.md`. +(3) watchdog-na-Prometheusa + aktualizacja `kb/subsystems/observer.md`. Etap 0 to czynność operatorska (runtime, nie repo). Etap 5 to niezależny backlog. --- diff --git a/kb/decisions/ai-cluster-legacy.md b/kb/decisions/ai-cluster-legacy.md index 46604be..71aa922 100644 --- a/kb/decisions/ai-cluster-legacy.md +++ b/kb/decisions/ai-cluster-legacy.md @@ -14,7 +14,7 @@ links: [] Stack **ai-cluster** działający na vps (`ai-cluster-openclaw-1`, `codex-worker`, `planner-worker`, `service-ops-worker`, `redis`, `mosquitto`) jest **wygaszany, nie migrowany**. Podstawa (recon -[RECON-multiagent-2026-07-27.md](RECON-multiagent-2026-07-27.md), C9): +[RECON-multiagent-2026-07-27.md](../subsystems/recon-multiagent.md), C9): bus `codex/*` jest martwy od **2026-06-09** — zero nowych połączeń przez ~7 tygodni, workery trzymają tylko puste długożyjące połączenia. diff --git a/kb/decisions/architektura-2026-07-28.md b/kb/decisions/architektura-2026-07-28.md index 5a6cefe..cd4ff61 100644 --- a/kb/decisions/architektura-2026-07-28.md +++ b/kb/decisions/architektura-2026-07-28.md @@ -10,8 +10,8 @@ links: [] # Architektura — decyzje obowiązujące Stan decyzji na 2026-07-28. Podstawa dowodowa: -[RECON-multiagent-2026-07-27.md](RECON-multiagent-2026-07-27.md); plan wykonawczy: -[PLAN-subsystem-a-2026-07-28.md](PLAN-subsystem-a-2026-07-28.md). Zmiana którejkolwiek +[RECON-multiagent-2026-07-27.md](../subsystems/recon-multiagent.md); plan wykonawczy: +[PLAN-subsystem-a-2026-07-28.md](../phases/subsystem-a-naprawa.md). Zmiana którejkolwiek decyzji wymaga aktualizacji tego pliku z nową datą. ## Dwa subsystemy (2026-07-28) @@ -42,7 +42,7 @@ Stack ai-cluster na vps (openclaw, codex-worker, planner-worker, service-ops-wor redis, mosquitto) jest **wygaszany, nie migrowany**. Bus `codex/*` martwy od 2026-06-09 (zero nowych połączeń). Branch `task/ai-cluster-solaria` zostaje **niezmergowany** — pełni rolę dokumentacji. Kontenery na vps zostaną zatrzymane w -osobnej, nadzorowanej sesji. Szczegóły: [ai-cluster-LEGACY.md](ai-cluster-LEGACY.md). +osobnej, nadzorowanej sesji. Szczegóły: [ai-cluster-LEGACY.md](ai-cluster-legacy.md). ## Approvale zostają HITL (2026-07-28) diff --git a/kb/decisions/backlog-aktywne.md b/kb/decisions/backlog-aktywne.md index d471ebd..e6be503 100644 --- a/kb/decisions/backlog-aktywne.md +++ b/kb/decisions/backlog-aktywne.md @@ -54,7 +54,7 @@ z repo daje kontener identyczny z obecnym **zanim** ktokolwiek zrobi redeploy. znalezione przy porządkowaniu backlogu po tej sesji **Problem**: kontrakt serwisu deklaruje `private` — `services/narty27/service.yaml:5` (`exposure: private # LAN/Tailscale only; no npm vhost, no public ingress`) i to samo -w `services/narty27/README.md` — podczas gdy `narty27.kapala.org` jest **publiczne** +w `kb/runbooks/narty27-deploy.md` — podczas gdy `narty27.kapala.org` jest **publiczne** (NPM na VPS + cert Let's Encrypt). Pole `exposure` steruje traktowaniem ekspozycji przez agentów (patrz „Discovery Entry Points for Agents" w CLAUDE.md — `service.yaml` jest kontraktem operacyjnym, z którego agent czyta, jak zarządzać serwisem), więc @@ -110,7 +110,7 @@ na próg wyłączenia) w `1784804667795`, tak samo jak istniejący trigger na ### HA ken: guard TRV kalibracji przed sezonem grzewczym (~2026-09) **Data**: 2026-07-23 -**Źródło**: `services/home-assistant/docs/audyt-automatyzacji-2026-07-23.md` sekcja 1.4 +**Źródło**: `kb/audits/ha-automatyzacje-2026-07-23.md` sekcja 1.4 (pkt 2 checklisty operatora) **Problem**: kalibracje TRV sypialnia (`1764751049013`) i Tymek (`1765817937658`) liczą `room_temp` jako średnią z czujnika zhimi, który jest `unavailable` od 2026-07-17 → @@ -129,7 +129,7 @@ urządzenie zniknęło z sieci). ### HA ken: przycisk graceful shutdown klimy salonowej na kartę dashboardu **Data**: 2026-07-23 -**Źródło**: `services/home-assistant/docs/audyt-automatyzacji-2026-07-23.md` sekcja 3.1, +**Źródło**: `kb/audits/ha-automatyzacje-2026-07-23.md` sekcja 3.1, fix-pack 1 (`task/ha-fix-pack-1`, DESIGN.md „Decyzje operatora po audycie 2026-07-23") **Kontekst**: po fix-packu 1 „Klima salon: wyłącz…" (`1784804668795`) respektuje `klima_salon_auto = on` dla gałęzi sunset/balkon; suszenie parownika przy ręcznym @@ -145,7 +145,7 @@ wymagało znajomości wewnętrznej logiki automatyzacji. ### HA ken: diagnoza wspólnej awarii sprzętowej 2026-07-17 (czujniki ruchu, pilot 4button, xiaomi_miot) **Data**: 2026-07-23 -**Źródło**: `services/home-assistant/docs/audyt-automatyzacji-2026-07-23.md` sekcje 1.2, +**Źródło**: `kb/audits/ha-automatyzacje-2026-07-23.md` sekcje 1.2, 1.6, 4.5 (pkt 1, 3, 15 checklisty operatora) **Problem**: restart HA / update Supervisora 2026-07-17 15:07 zbiega się z `unavailable` na: klaster czujników ruchu/obecności (mdwejscie, mdsypialnia, mdheli — @@ -166,7 +166,7 @@ automatyzacji (alerty on-leave, nocne gaszenie, `Poranek start`) — patrz audyt ### HA ken: projekt „architektura night_mode" (konsolidacja sleep/night mode) **Data**: 2026-07-23 -**Źródło**: `services/home-assistant/docs/audyt-automatyzacji-2026-07-23.md` sekcje 2.2, +**Źródło**: `kb/audits/ha-automatyzacje-2026-07-23.md` sekcje 2.2, 2.4, 6 (pkt 7, 9 checklisty operatora — świadomie NIE załatane w fix-packu 1, patrz DESIGN.md „Decyzje operatora po audycie 2026-07-23") **Problem**: cztery flagi trybu (`sleep_mode`, `night_mode`, `passive_mode`, @@ -284,7 +284,7 @@ przyciskami approve/reject per pending action. **Data**: 2026-07-22 **Źródło**: sesja 2026-07-22/23 (`docs/sessions/2026-07-23-control-plane-remediation-e2e.md`) **Problem**: kontener `homeassistant5` (legacy HA "ken", patrz cutover -`docs/backlog.md` sekcja "Cutover HA ken") zaobserwowany w stanie `Exited (0)` — exit +`kb/phases/backlog.md` sekcja "Cutover HA ken") zaobserwowany w stanie `Exited (0)` — exit code 0 = czyste zatrzymanie, więc Docker `restart: unless-stopped` świadomie go nie podnosi (to nie crash). Do zweryfikowania, czy to zamierzone wygaszenie z cutoveru czy coś zatrzymało kontener niezamierzenie. diff --git a/kb/decisions/backlog-m1-solaria-prune-mitigation.md b/kb/decisions/backlog-m1-solaria-prune-mitigation.md index 9749fd9..2fd30dd 100644 --- a/kb/decisions/backlog-m1-solaria-prune-mitigation.md +++ b/kb/decisions/backlog-m1-solaria-prune-mitigation.md @@ -12,7 +12,7 @@ links: ## M1 aktywna na SOLARII — cleanup node-agenta wyłączony do czasu R1 (2026-08-04) **Data**: 2026-08-04 -**Źródło**: `docs/incidents/2026-07-30-ollama-solaria-vanish.md` (§7, M1). +**Źródło**: `kb/incidents/2026-07-30-ollama-solaria-vanish.md` (§7, M1). **Stan**: w `hosts/solaria/runtime/node-agent/docker-compose.override.yml` ustawiono `NODE_TYPE=lte_node` (było `ai_node`) na czas backfillu embed (faza mailowa KB). `lte_node` powoduje wczesny return w `run_safe_cleanup()`, więc na SOLARII nie działa diff --git a/kb/decisions/backlog-rozjazdy-repo-rzeczywistosc.md b/kb/decisions/backlog-rozjazdy-repo-rzeczywistosc.md index d29ba71..6751b07 100644 --- a/kb/decisions/backlog-rozjazdy-repo-rzeczywistosc.md +++ b/kb/decisions/backlog-rozjazdy-repo-rzeczywistosc.md @@ -9,8 +9,8 @@ links: --- ## Rozjazdy repo<->rzeczywistosc (z inwentaryzacji 2026-06-30) -**Zrodlo**: `docs/infra/inventory-2026-06-30.md` (23 rozjazdy, pelna tabela tam). -**Weryfikacja 2026-07-02**: `docs/infra/inventory-verify-2026-07-02.md` — bilans: +**Zrodlo**: `kb/subsystems/fleet-inventory.md` (23 rozjazdy, pelna tabela tam). +**Weryfikacja 2026-07-02**: `kb/subsystems/fleet-inventory-verify.md` — bilans: 20 wciaz aktualnych, 2 zmienione, 1 wyjasniony (storage SOLARIA = partycja Windows dual-boot, NIE rozjazd — zdjety z listy). Ponizej te wymagajace akcji, pogrupowane wg ryzyka. Naprawa = osobny task/kilka. @@ -50,7 +50,7 @@ Ponizej te wymagajace akcji, pogrupowane wg ryzyka. Naprawa = osobny task/kilka. - **stability-agent / node_exporter** owner_node single, biegaja wielomiejscowo -> per-host ### Followupy z weryfikacji + rozbrajania min (2026-07-02) -**Zrodlo**: `docs/infra/inventory-verify-2026-07-02.md` + sesja 2026-07-02. +**Zrodlo**: `kb/subsystems/fleet-inventory-verify.md` + sesja 2026-07-02. Zgloszone przy fixie owner_node (`886bc85`), swiadomie NIE ruszone — osobne decyzje. - **forgejo** brak wpisu w `hosts/piha/services.yaml`; **mosquitto** brak diff --git a/kb/decisions/backlog-zamkniete.md b/kb/decisions/backlog-zamkniete.md index c8652e3..ae6d50e 100644 --- a/kb/decisions/backlog-zamkniete.md +++ b/kb/decisions/backlog-zamkniete.md @@ -116,7 +116,7 @@ konsumentów tego pola — szeroki `except` maskował dokładnie tę klasę regr > „Ghosty B WRÓCIŁY" w Aktywnych (`docs/sessions/2026-07-06.md`). **Data**: 2026-06-24 (wykryte), 2026-07-02 (zamknięte) -**Źródło**: sesje 2026-06-24/25/26; recon `docs/infra/inventory-verify-2026-07-02.md` +**Źródło**: sesje 2026-06-24/25/26; recon `kb/subsystems/fleet-inventory-verify.md` **Było**: martwe kontenery ze starych project-name'ów (`8547b46c0317_control-plane-supervisor` itp.) raportowane przez observera jako `error` → `System Status ERROR` w panelu mimo zdrowego mózgu. @@ -130,7 +130,7 @@ Rozważenie czyszczenia obcych project-name przy deployu — już nieaktualne ### brain-watchdog: poll Prometheus — POTWIERDZONY (recon 2026-07-02) **Data**: 2026-06-30 (pending), 2026-07-02 (zamknięte) -**Źródło**: recon `docs/infra/inventory-verify-2026-07-02.md` +**Źródło**: recon `kb/subsystems/fleet-inventory-verify.md` **Było**: log startowy nie wypisuje `PROMETHEUS_URL` → brak pewności, że polling aktywny; diagnoza opierała się na `.env` i braku błędów. **Zamknięte**: recon potwierdził — obraz zbudowany po `62d6fc0`, `PROMETHEUS_URL` @@ -144,7 +144,7 @@ bez czekania na realną awarię. ### Miny #1/#2/#3 z weryfikacji inwentaryzacji — ROZBROJONE (2026-07-02) -**Źródło**: `docs/infra/inventory-verify-2026-07-02.md`, sesja +**Źródło**: `kb/subsystems/fleet-inventory-verify.md`, sesja `docs/sessions/2026-07-02.md` (tam szczegóły i lekcje). - **#1 PIHA checkout**: gałąź wciąż `task/kb-gmail-import` po resecie z 2026-06-30 (reset --hard przesuwa gałąź, nie przełącza) → `checkout master && pull`, diff --git a/kb/decisions/deploy-runner-uzasadnienie.md b/kb/decisions/deploy-runner-uzasadnienie.md index 438d19f..f84e82e 100644 --- a/kb/decisions/deploy-runner-uzasadnienie.md +++ b/kb/decisions/deploy-runner-uzasadnienie.md @@ -20,5 +20,5 @@ exited at line 18 with `Error: Repository not found`. Behind that failure sat three more: no `git`, no `docker` CLI in the image, and — had it ever got that far — it would have deployed the **executor host's** entire service set, not the action's target node/service. Every `healthcheck_failed → redeploy` dead-ended -(recon `docs/architecture/RECON-multiagent-2026-07-27.md`, D14/D15). +(recon `kb/subsystems/recon-multiagent.md`, D14/D15). diff --git a/kb/decisions/ha-configs-as-code.md b/kb/decisions/ha-configs-as-code.md index 9804462..241d97d 100644 --- a/kb/decisions/ha-configs-as-code.md +++ b/kb/decisions/ha-configs-as-code.md @@ -13,7 +13,7 @@ links: Status: **phase 1 (partial)**. `scripts/ha/deploy.sh` implements the write path for the `api` adapter's automations/scripts/scenes scope (see "Deploy path" and "Sync model" below); dashboards/helpers and the `docker-exec` -adapter have no write path yet. See `docs/backlog.md` for the tracking +adapter have no write path yet. See `kb/phases/backlog.md` for the tracking entry. ## Phasing @@ -59,7 +59,7 @@ behind a common interface (`import.sh`/eventual `deploy.sh `): |---|---|---| | `ken` (RPi4, HAOS, LAN `192.168.31.7:8123`) | **api** | Canonical home instance since the 2026-07-22 cutover (see Incident log). HAOS has no SSH access, so there is no `docker exec`/filesystem path — only the HA REST/websocket API is reachable. Full `/config` import is deferred until an alternative access path exists; for now the api adapter's import scope is limited to what the API exposes: automations, scripts, scenes, dashboards. | | `ken-legacy` (piha, container `homeassistant5`) | **docker-exec over SSH** (archive-only) | Pre-migration container instance, superseded by `ken` at 31.7 (see Incident log) — same container filesystem access as the old `ken` entry (`ssh oskar@piha "docker exec homeassistant5 ..."`). Import only, for historical reference; never a deploy target. | -| `chelsty-ha` | **api** | Reachable over Tailscale at `100.70.180.90:8123` (confirmed working path — `services/ha-diag-agent/DEPLOY.md` already curls this for health checks). Config-as-code deploy will reuse the same reachability, calling the HA REST/websocket API rather than shelling into the container. | +| `chelsty-ha` | **api** | Reachable over Tailscale at `100.70.180.90:8123` (confirmed working path — `kb/runbooks/ha-diag-agent-deploy.md` already curls this for health checks). Config-as-code deploy will reuse the same reachability, calling the HA REST/websocket API rather than shelling into the container. | **Open**: a `file` adapter (direct bind-mount / SSH `rsync` to the config directory, bypassing `docker exec`) is worth revisiting once SSH access to @@ -124,7 +124,7 @@ copied into the repo. - A dedicated `deploy_agent` HA user account (admin rights, **local-only** — never exposed through the public API/ingress) is created per instance, mirroring the existing `diag_agent` account pattern documented in - `services/ha-diag-agent/DEPLOY.md`. Reusing `diag_agent` is explicitly + `kb/runbooks/ha-diag-agent-deploy.md`. Reusing `diag_agent` is explicitly rejected — deploy tooling and the diagnostic agent must be revocable independently. - Long-lived access tokens for `deploy_agent` live at @@ -150,7 +150,7 @@ Zobacz `docs/audyt-automatyzacji-2026-07-23.md` (sekcja "Do decyzji operatora", 17 punktów) — poniżej wyłącznie decyzje, które doprowadziły do zmian w fix-pack 1 (`task/ha-fix-pack-1`) albo świadomie do braku zmian. Reszta checklisty (baterie/re-pairing czujników, kalibracje TRV, xiaomi_miot, -konsolidacja aliasów, higiena 4.x) zostaje otwarta w `docs/backlog.md`. +konsolidacja aliasów, higiena 4.x) zostaje otwarta w `kb/phases/backlog.md`. - **Pkt 6 (klima salon: sunset ubija też ręczne chłodzenie?)** — decyzja: NIE. „Klima salon: wyłącz…" (`1784804668795`) ma teraz respektować @@ -166,7 +166,7 @@ konsolidacja aliasów, higiena 4.x) zostaje otwarta w `docs/backlog.md`. - **Pkt 7 (enforcer sleep mode gasi światła cyklicznie całą noc) i pkt 9 (konsolidacja czterech nocnych wyłączników)** — bez zmian w tym fix-packu. Oba wchłania przyszły projekt „architektura night_mode" (patrz - `docs/backlog.md`) — punktowa łatka tu tylko dodałaby kolejny wariant do + `kb/phases/backlog.md`) — punktowa łatka tu tylko dodałaby kolejny wariant do już przegęszczonego zestawu nakładających się automatyzacji (audyt 2.2). - **Pkt 11 (OwnTracks: przywrócić czy skasować) i pkt 12 (Leave auto on: batch 02 — włączyć z powrotem?)** — świadomie bez zmian; obie wymagają diff --git a/kb/decisions/kb-dokumenty-otwarte.md b/kb/decisions/kb-dokumenty-otwarte.md index 8018060..ca76810 100644 --- a/kb/decisions/kb-dokumenty-otwarte.md +++ b/kb/decisions/kb-dokumenty-otwarte.md @@ -63,7 +63,7 @@ zdecydować, czy podbić concurrency, czy zostawić zapas na ollama/AI. rsync/borg → SOLARIA (2 TB, ta sama LAN). Retencja: 7 dziennych + 4 tygodniowe + 6 miesięcznych. Offsite (np. restic → chmura) zostaje jako future-note, poza zakresem tego etapu. Cron/skrypt deployowy powstaje przy - deployu modułu 2, nie teraz. Szczegóły: `services/paperless/README.md`. + deployu modułu 2, nie teraz. Szczegóły: `kb/services/paperless.md`. - **4. Redis brokera: `requirepass`.** Broker (6380) dostaje hasło — `PAPERLESS_REDIS_PASSWORD` w `.env` po obu stronach (paperless@PIHA, @@ -76,7 +76,7 @@ zdecydować, czy podbić concurrency, czy zostawić zapas na ollama/AI. + SOLARIA po NFS) świadomie zaakceptowane — indeks jest odtwarzalny (`document_index reindex`), oryginałom nic nie grozi. Bez zmian w configu; fallback-worker na PIHA zostaje. Szczegóły: - `services/paperless-worker/README.md`. + `kb/services/paperless-worker.md`. - **6. Domeny: `kapala.org` (mesh, prywatne).** `paper.kapala.org` (Paperless), `cloud.kapala.org` (Nextcloud) — potwierdzone, `*.okit.pl` @@ -99,7 +99,7 @@ zdecydować, czy podbić concurrency, czy zostawić zapas na ollama/AI. maintainerów paperless-ngx (nieoficjalnie wspierany): ten sam obraz, `command: celery --app paperless worker`, wspólny Redis+Postgres+storage, identyczne ścieżki kontenerowe i numeryczny UID po obu stronach. Pełny - wynik badania + ryzyka: `services/paperless-worker/README.md`. + wynik badania + ryzyka: `kb/services/paperless-worker.md`. - Storage dokumentów na PIHA; NFS export → SOLARIA po LAN (192.168.31.5 → 192.168.31.70), nie Tailscale. - AOF w Redis brokera (kolejka przeżywa restart — zero utraty zadań). diff --git a/kb/decisions/paperless-split-ocr.md b/kb/decisions/paperless-split-ocr.md index f59a893..e24a64e 100644 --- a/kb/decisions/paperless-split-ocr.md +++ b/kb/decisions/paperless-split-ocr.md @@ -29,5 +29,5 @@ widzieć **ten sam Redis, tego samego Postgresa i te same pliki**. Stąd: restart brokera nie gubi kolejki). Szczegóły NFS (export na PIHA, mount na SOLARIA, ryzyko indeksu Whoosh) — -`services/paperless-worker/README.md`. +`kb/services/paperless-worker.md`. diff --git a/kb/decisions/tech-debt-legacy.md b/kb/decisions/tech-debt-legacy.md index 3b779f7..156725b 100644 --- a/kb/decisions/tech-debt-legacy.md +++ b/kb/decisions/tech-debt-legacy.md @@ -5,7 +5,7 @@ visibility: private status: deprecated updated: 2026-06-24 links: [] -superseded_by: "kb/phases/backlog.md (docs/backlog.md przejal ewidencje dlugu)" +superseded_by: "kb/phases/backlog.md (kb/phases/backlog.md przejal ewidencje dlugu)" --- # Tech Debt diff --git a/kb/incidents/2026-07-12-paperless-worker-config.md b/kb/incidents/2026-07-12-paperless-worker-config.md index 0c54273..456c1b0 100644 --- a/kb/incidents/2026-07-12-paperless-worker-config.md +++ b/kb/incidents/2026-07-12-paperless-worker-config.md @@ -30,7 +30,7 @@ zdeployowany i brał zadania z kolejki, ale miał dwa bugi w compose: mount na `/tmp/paperless` po obu stronach. Oba fixy + uzasadnienie: `services/paperless/docker-compose.yml`, -`services/paperless-worker/docker-compose.yml`, `services/paperless-worker/README.md`. +`services/paperless-worker/docker-compose.yml`, `kb/services/paperless-worker.md`. Zweryfikowane end-to-end na żywo (branch `task/paperless-worker-fix`, jeszcze niezmergowany do master w momencie pisania tego wpisu): 3 dokumenty testowe wrzucone do `consume/` na PIHA, jeden odebrany i dokończony przez worker@SOLARIA diff --git a/kb/incidents/2026-07-16-ollama-solaria-brak-sterownika.md b/kb/incidents/2026-07-16-ollama-solaria-brak-sterownika.md index cd47a9f..88652e4 100644 --- a/kb/incidents/2026-07-16-ollama-solaria-brak-sterownika.md +++ b/kb/incidents/2026-07-16-ollama-solaria-brak-sterownika.md @@ -10,7 +10,7 @@ links: ## Ollama SOLARIA: brak sterownika NVIDII — ZAMKNIĘTE (2026-07-16) -**Kontekst.** Cutover 2026-07-15 (`docs/infra/ollama-solaria-cutover-2026-07-15.md`) +**Kontekst.** Cutover 2026-07-15 (`kb/runbooks/ollama-solaria-cutover.md`) odkrył, że SOLARIA nie miała zainstalowanego żadnego sterownika NVIDII — `nvidia-smi` nie istniał na hoście. `hosts/solaria/services.yaml` opisywał ollama jako "GPU-backed" od dawna, ale to było aspiracyjne — Ollama zawsze @@ -24,7 +24,7 @@ jako blokujący fazę mailową embeddingów (moduł 5). hoście. `nvidia-container-toolkit` był już obecny (doinstalowany jako prerequisite przy cutoverze 07-15). GPU reservation przywrócona w compose. Pomiar throughput GPU vs CPU baseline (0.79s/chunk) — patrz -`jobs/documents-ingest/README.md`, sekcja timing. +`kb/phases/kb-m5-documents-ingest-fazy.md`, sekcja timing. **Status:** ZAMKNIĘTE. @@ -35,6 +35,6 @@ Pomiar throughput GPU vs CPU baseline (0.79s/chunk) — patrz - **`UNIQUE(envelope_id, chunk_index)` bez `model`** w `document_chunk` (`services/kb-postgres/init/002_chunks.sql`) — re-embedding innym modelem cicho no-opuje się przez istniejący constraint. Schema change do zrobienia - przy fazie 3 (patrz `jobs/documents-ingest/README.md`, sekcja "Idempotency" + przy fazie 3 (patrz `kb/phases/kb-m5-documents-ingest-fazy.md`, sekcja "Idempotency" kroku 6 embed). diff --git a/kb/incidents/2026-07-22-ha-dwie-instancje.md b/kb/incidents/2026-07-22-ha-dwie-instancje.md index e2983c4..528761b 100644 --- a/kb/incidents/2026-07-22-ha-dwie-instancje.md +++ b/kb/incidents/2026-07-22-ha-dwie-instancje.md @@ -37,7 +37,7 @@ followed the move. **Decision**: 192.168.31.7 (HAOS/RPi4) is canonical `ken`. The piha container is renamed `ken-legacy` in `instances.yaml`, `status: archived`. Plan: archival import for historical reference → `docker stop` (not `rm`) -→ one week of observation → decide on `docker rm`. See `docs/backlog.md` +→ one week of observation → decide on `docker rm`. See `kb/phases/backlog.md` for the ha-diag-agent re-pointing and wind-down follow-ups this incident generated. diff --git a/kb/incidents/2026-07-22-ha-ken-cutover-legacy.md b/kb/incidents/2026-07-22-ha-ken-cutover-legacy.md index 474b56c..c1d9abb 100644 --- a/kb/incidents/2026-07-22-ha-ken-cutover-legacy.md +++ b/kb/incidents/2026-07-22-ha-ken-cutover-legacy.md @@ -13,7 +13,7 @@ links: **Data**: 2026-07-22 **Źródło**: recon — dwie instancje HA równolegle sterowały domem (kontener `homeassistant5` na piha + RPi4 HAOS 192.168.31.7), patrz -`services/home-assistant/DESIGN.md` sekcja "Incident log". `instances.yaml` +`kb/decisions/ha-configs-as-code.md` sekcja "Incident log". `instances.yaml` naprawiony w tej samej sesji: `ken` = 192.168.31.7 (api), `ken-legacy` = dawny kontener piha (docker-exec, archived). diff --git a/kb/phases/ha-configs-as-code.md b/kb/phases/ha-configs-as-code.md index 8fa07b7..644072f 100644 --- a/kb/phases/ha-configs-as-code.md +++ b/kb/phases/ha-configs-as-code.md @@ -17,7 +17,7 @@ Szkielet struktury dla `services/home-assistant/` — configs-as-code dla instancji HA (`ken` na PIHA, `chelsty-ha`). Na razie tylko struktura + read-only import (`scripts/ha/import.sh`), bez deployu. Fazowanie, wybór adaptera per instancja, model sync i otwarte pytania — -`services/home-assistant/DESIGN.md`. +`kb/decisions/ha-configs-as-code.md`. --- diff --git a/kb/phases/kb-m0-piha-slim.md b/kb/phases/kb-m0-piha-slim.md index 72fa71e..f7b29c2 100644 --- a/kb/phases/kb-m0-piha-slim.md +++ b/kb/phases/kb-m0-piha-slim.md @@ -15,7 +15,7 @@ links: [] ## STATUS: prerekwizyt RAM SPELNIONY (2026-07-02) Faza 1 (audyt read-only) + faza 2 (egzekucja po review Oskara) wykonane — -szczegoly: `docs/infra/piha-slim-audit-2026-07-02.md` (sekcja "Korekta po review +szczegoly: `kb/audits/piha-slim-2026-07-02.md` (sekcja "Korekta po review + egzekucja"). - **Kryterium >= 1.5Gi available: SPELNIONE.** Przed egzekucja: 2.8Gi available @@ -37,7 +37,7 @@ Zwolnic RAM na PIHA (dzis: 3.1Gi available, swap 2G uzyty) tak, by lekki Paperle serwis wszedl z zapasem, nie na styku swap. ## Wymogi -- Audyt 33 shadow-kontenerow (lista w `docs/infra/inventory-2026-06-30.md`). +- Audyt 33 shadow-kontenerow (lista w `kb/subsystems/fleet-inventory.md`). - Zidentyfikowac kandydatow do usuniecia/przeniesienia/wylaczenia: - **elasticsearch 1Gi** — kto tego uzywa? (wikijs? diskover?) — jesli martwy, ubic - **diskover** — jednorazowy indekser? czy chodzi ciagle bez potrzeby? diff --git a/kb/phases/kb-m5-documents-ingest-fazy.md b/kb/phases/kb-m5-documents-ingest-fazy.md index b8b0f91..ee366d9 100644 --- a/kb/phases/kb-m5-documents-ingest-fazy.md +++ b/kb/phases/kb-m5-documents-ingest-fazy.md @@ -13,7 +13,7 @@ links: ## Phase 2 — `documents-ingest-paperless` (Paperless -> envelope adapter) -Module 5, phase 2 (`docs/kb/modules/05-faza2-plan.md`, §4.2-4.3, §6 step 5). +Module 5, phase 2 (`kb/phases/kb-m5-faza2.md`, §4.2-4.3, §6 step 5). Reads documents from the **Paperless REST API** (read-only — GET only, never writes to Paperless) and inserts them as `source='paperless'` rows into the `envelope` table on kb-postgres, reusing `kb_mail.envelope.Envelope` / @@ -133,7 +133,7 @@ this commit. ## Phase 2 step 6 — `documents-ingest-embed` (chunk + embed) -Module 5, phase 2, plan step 6 (`docs/kb/modules/05-faza2-plan.md`, §6 step 6, +Module 5, phase 2, plan step 6 (`kb/phases/kb-m5-faza2.md`, §6 step 6, §2 decision 3). Reads `entities[type=content].text` off every `source='paperless'` envelope, chunks it, calls Ollama (`POST /api/embeddings`, model `bge-m3`) for each chunk, and inserts the result into `document_chunk` @@ -311,7 +311,7 @@ kb-postgres@PIHA: ## Phase 3 step 4 — retrieval cascade (`documents_ingest.retrieval`) + quality gate -Module 5, phase 3, plan step 4 (`docs/kb/modules/05-faza3-plan.md`, §6). Two retrieval +Module 5, phase 3, plan step 4 (`kb/phases/kb-m5-faza3.md`, §6). Two retrieval paths, both `query_text -> chunk hits (dist, source)` — the intended clean API surface for phase 4's kb-query, not just this eval: @@ -328,7 +328,7 @@ phase 4's kb-query, not just this eval: ### Quality gate `eval/queries.yaml` — 7 queries transcribed 1:1 from the phase-2 pilot baseline -(`docs/kb/eval/retrieval-pilot-2026-07-16.md`, left untouched — this is its versioned working +(`kb/phases/kb-m5-eval-retrieval-pilot.md`, left untouched — this is its versioned working copy) with expected envelope / kind (`hit`, `grey_zone`, `negative_control`, `negative_control_borderline`) per query. @@ -373,7 +373,7 @@ short-circuit (stage 2 never queried), and both query entry points embedding exa ## Phase 3 step 5 — cyclic ingest (`documents-ingest-cyclic`) + systemd timer -Module 5, phase 3, plan step 5 (`docs/kb/modules/05-faza3-plan.md`, §7). Orchestrates one +Module 5, phase 3, plan step 5 (`kb/phases/kb-m5-faza3.md`, §7). Orchestrates one run of the recurring ingest pipeline: `paperless_adapter.run()` (new `source='paperless'` envelopes) → `chunk_embed.run()` (new `document_chunk` rows) → `summarize.run_summarize(backend='anthropic')` (new `document_summary` rows, diff --git a/kb/phases/kb-m5-faza2.md b/kb/phases/kb-m5-faza2.md index 996734d..58528bd 100644 --- a/kb/phases/kb-m5-faza2.md +++ b/kb/phases/kb-m5-faza2.md @@ -545,7 +545,7 @@ zeby dalo sie uruchomic partiami i zweryfikowac progres bez czekania na cale 225 rzedu dziesiatek-set chunkow/s. Caly pilot (2–3k chunkow) → **rzedu minut**, nie wymaga specjalnego batchowania/partii. - **Skala docelowa (70k zalacznikow z maili)**: modul 5 faza-1 to swiadomie **probka, nie - bulk** (`jobs/documents-ingest/README.md` — decyzja architektoniczna). Realny wolumen + bulk** (`kb/phases/kb-m5-documents-ingest-fazy.md` — decyzja architektoniczna). Realny wolumen ktory trafi do embeddingu zalezy od (a) throughput OCR-workera na SOLARII (modul 3) — **to jest waskie gardlo skalowania, nie embedding** — oraz (b) filtra selektywnosci (decyzja #6). Sam embedding bge-m3 nie bedzie bottleneckiem nawet przy tysiacach diff --git a/kb/phases/kb-m5-faza3.md b/kb/phases/kb-m5-faza3.md index 35176f6..2fb0f92 100644 --- a/kb/phases/kb-m5-faza3.md +++ b/kb/phases/kb-m5-faza3.md @@ -505,7 +505,7 @@ aktywnych chunków, N większe niż liczba kopert, no-summaries short-circuit) pakietu przechodzi. **Eval-set utrwalony**: `jobs/documents-ingest/eval/queries.yaml` (7 zapytań z pilota -07-16, 1:1 z `docs/kb/eval/retrieval-pilot-2026-07-16.md`, ten plik pozostał nietknięty — +07-16, 1:1 z `kb/phases/kb-m5-eval-retrieval-pilot.md`, ten plik pozostał nietknięty — `queries.yaml` to jego wersjonowana kopia robocza). Skrypt bramki (read-only, integracyjny, **nie wchodzi do pytest**): `jobs/documents-ingest/eval/retrieval_eval.py`. @@ -700,7 +700,7 @@ streszczeń). Kandydaci na pierwsze strony (encje z pilota): PZU/WARTA (polisy), ## 9. Poza zakresem fazy 3 Granice planu — wszystko poniżej jest świadomie odłożone, z istniejącym miejscem w -roadmapie (`docs/kb/kb-00-overview.md` „Stan etapów/Backlog", sesja +roadmapie (`kb/subsystems/kb-overview.md` „Stan etapów/Backlog", sesja `docs/sessions/2026-07-16.md`, backlog operatora): | Temat | Gdzie zakotwiczone | Kiedy | diff --git a/kb/phases/kb-m5-faza4-fallback-dedup.md b/kb/phases/kb-m5-faza4-fallback-dedup.md index 5762ffa..5b0abae 100644 --- a/kb/phases/kb-m5-faza4-fallback-dedup.md +++ b/kb/phases/kb-m5-faza4-fallback-dedup.md @@ -71,10 +71,10 @@ zegarem, master trzyma to samo wewnątrz `EmbedRouter` (też injektowalny zegar) | `services/kb-query/{README,env.example,service.yaml,docker-compose.yml}` | ~±58 | duplikat | **porzuć** | inny (porzucony) schemat env `OLLAMA_PIHA_URL`; master bogatszy (testy A/B/C, rename `OLLAMA_URL`→`EMBED_PRIMARY_URL`) | | `packages/kb-retrieval/src/kb_retrieval/embed.py` (`timeout_s` w `embed_chunk`) | +16 | duplikat funkcji | **porzuć** | master osiąga twardy timeout przez `asyncio.wait_for` bez zmiany współdzielonego pakietu — mniejsza powierzchnia zmian, ten sam efekt | | `jobs/documents-ingest/eval/retrieval_eval.py` (`--transport {direct,http}` + `--base-url`) | +109 | **unikalna wartość** | **cherry-pick** | plan §2 decyzja 6 / §9 (bramka HTTP-equivalence) — **na masterze w ogóle nie istnieje**; e7625cd nie tknął tego pliku, patch aplikuje się czysto; kod woła tylko `GET /search` i czyta `envelope_id`/`dist`/`source` — w pełni zgodny z odpowiedzią mastera | -| `jobs/documents-ingest/README.md` | +6 | **unikalna wartość** | **cherry-pick** | dokumentacja powyższego, idzie w parze | +| `kb/phases/kb-m5-documents-ingest-fazy.md` | +6 | **unikalna wartość** | **cherry-pick** | dokumentacja powyższego, idzie w parze | | `docs/sessions/2026-07-27-kb-f4-fallback.md` | +189 | **unikalna wartość** | **adaptuj** | jedyny zapis: (1) znalezisko osieroconego natywnego `ollama.service` na PIHA + jego wyłączenie 2026-07-27 i backlog odinstalowania, (2) kalibracja live ollama-piha (GO: peak ~983 MiB, ~4.2–5.3 s/embed), (3) metodologia i wyniki bramki §9 (HTTP-equivalence 0 rozbieżności; sol-down Δ~3e-4), (4) rsync-deploy → dirty working tree na PIHA. Wciągnąć z dopiskiem redakcyjnym, że zmergowana implementacja to **inny kod** (e7625cd) i wyniki bramki wymagają powtórki | | `services/ollama-piha/*` (5 plików) | +155 | duplikat | **porzuć** | wersja mastera lepsza: named volume `ollama_piha_models` (uzasadnienie uid-pattern PIHA), healthcheck sprawdza obecność `bge-m3`, bind tylko 127.0.0.1+LAN | -| `hosts/piha/runtime/ollama-piha/docker-compose.override.yml` | +13 | duplikat + **1 unikalny fakt** | **adaptuj (mikro)** | ten sam `mem_limit: 2560m`; ale komentarz brancha zawiera potwierdzony pomiar (peak ~983 MiB), a master wciąż mówi „Confirm/trim after live calibration" — dopisać wynik kalibracji do komentarza override'u i/lub sekcji „Calibration" w `services/ollama-piha/README.md` | +| `hosts/piha/runtime/ollama-piha/docker-compose.override.yml` | +13 | duplikat + **1 unikalny fakt** | **adaptuj (mikro)** | ten sam `mem_limit: 2560m`; ale komentarz brancha zawiera potwierdzony pomiar (peak ~983 MiB), a master wciąż mówi „Confirm/trim after live calibration" — dopisać wynik kalibracji do komentarza override'u i/lub sekcji „Calibration" w `kb/services/ollama-piha.md` | | `hosts/piha/services.yaml` | ±26 | duplikat | **porzuć** | master ma własny wpis `ollama-piha` + soft-dependency kb-query; drobna różnica (`offline_required: true` na branchu vs `false` na masterze) — master źródłem prawdy | --- @@ -107,7 +107,7 @@ skonfigurowanego → `EmbedBackendError`, `fallback_status` (up/unconfigured), s ## 4. Rekomendacja zbiorcza (lista do zatwierdzenia) 1. **S1 — cherry-pick**: `retrieval_eval.py --transport http --base-url` + akapit w - `jobs/documents-ingest/README.md` (plan §2 D6/§9; aplikuje się czysto, zero zależności + `kb/phases/kb-m5-documents-ingest-fazy.md` (plan §2 D6/§9; aplikuje się czysto, zero zależności od porzuconego kodu brancha). 2. **S2 — adaptuj**: `docs/sessions/2026-07-27-kb-f4-fallback.md` → `docs/sessions/` z dopiskiem redakcyjnym na górze (implementacja z tej sesji porzucona na rzecz @@ -116,7 +116,7 @@ skonfigurowanego → `EmbedBackendError`, `fallback_status` (up/unconfigured), s 3. **S3 — adaptuj**: luki testowe T1 + T2 (T3 opcjonalnie) do `test_embed_router.py`. 4. **S4 — adaptuj (mikro)**: wynik kalibracji 2026-07-27 (peak ~983 MiB, ~4.2–5.3 s, werdykt GO) do komentarza `hosts/piha/runtime/ollama-piha/docker-compose.override.yml` - i sekcji Calibration w `services/ollama-piha/README.md` — pomiar dotyczył kontenera + i sekcji Calibration w `kb/services/ollama-piha.md` — pomiar dotyczył kontenera ollama-piha (ta sama konfiguracja: obraz, `OLLAMA_KEEP_ALIVE=0`, `mem_limit 2560m`), więc **przenosi się** na wersję mastera; różni się tylko storage (bind vs named volume), co nie wpływa na RAM/latencję. @@ -130,7 +130,7 @@ skonfigurowanego → `EmbedBackendError`, `fallback_status` (up/unconfigured), s (3d4ee38) i worktree `~/homelab-codex-ws-kb-f4-fallback` po zakończeniu salvage. - **(b) Powtórka testu sol-down na żywym masterze**: kalibracja i bramka z 2026-07-27 dotyczyły **innego kodu** (`fallback.py`, env `OLLAMA_PIHA_URL`) — na wdrożonym - e7625cd trzeba przejść testy A/B/C z `services/kb-query/README.md` oraz bramkę + e7625cd trzeba przejść testy A/B/C z `kb/services/kb-query.md` oraz bramkę `retrieval_eval.py --transport http` (po S1): HTTP-equivalence przy SOLARIA-up (identyczne `dist`) i sol-down (Δ≤epsilon, kolejność top-k identyczna; baseline z 27.07: Δ~3e-4). Symulacja wg README: `EMBED_PRIMARY_URL=http://192.0.2.1:11434` diff --git a/kb/phases/kb-m5-faza4.md b/kb/phases/kb-m5-faza4.md index 0e11693..0fd7a22 100644 --- a/kb/phases/kb-m5-faza4.md +++ b/kb/phases/kb-m5-faza4.md @@ -68,7 +68,7 @@ links: [] ### 1.3 PIHA — budżet RAM (istotne dla decyzji D2 / lokalnego fallbacku) -- Audyt `docs/infra/piha-slim-audit-2026-07-02.md`: po "bezpiecznym usuń" +- Audyt `kb/audits/piha-slim-2026-07-02.md`: po "bezpiecznym usuń" (elasticsearch+diskover+stary llm-gateway) available rosło z **2.9Gi do ~3.8Gi** (na 8 GB total, +4 GB swap). To jest **jedyna zweryfikowana liczba w repo i ma 3 tygodnie** — od tego dnia na PIHA doszły (GitOps, z `mem_limit`): paperless+db+broker @@ -97,7 +97,7 @@ domeny `*.kapala.org`: (cert #51 *.okit.pl, cert osobny dla *.kapala.org — sesje ją tylko konfigurują ręcznie/SQL-em, nigdy nie deployują z repo). kb-query dogania się do tego samego wzorca: nowy vhost w `npm@PIHA`, TLS z **już istniejącego** wildcard `*.kapala.org` - (pokrywa `paper.`/`cloud.`/`vikunja.kapala.org` — `docs/kb/modules/DECYZJE-do-podjecia.md` + (pokrywa `paper.`/`cloud.`/`vikunja.kapala.org` — `kb/decisions/kb-dokumenty-otwarte.md` #6) — **żaden nowy certyfikat nie jest potrzebny**. - **DNS — dwie warstwy, obie trzeba dotknąć** (lekcja `okit-cloudflare-migracja.md` §"WAZNE: split-horizon DNS"): (a) Cloudflare rekord A → Tailscale IP PIHA @@ -261,7 +261,7 @@ przed backfillem" z fazy 3 §3.1 (zmierz, obejrzyj, dopiero wtedy zaufaj progowi ### Decyzja 3 — Linki do źródeł: paperless vs gmail **Paperless**: URL do dokumentu — **do zweryfikowania na żywym Paperless przed -implementacją** (repo nie ma zapisanego przykładu, `docs/kb/modules/02-paperless-service.md` +implementacją** (repo nie ma zapisanego przykładu, `kb/phases/kb-m2-paperless.md` dokumentuje tylko subdomenę, nie ścieżkę). Kandydat wg konwencji paperless-ngx UI: `https://paper.kapala.org/documents//details` (Angular routing) — `envelope_id` `paperless:` już niesie surowy `` do wstawienia. Krok implementacji: jeden @@ -471,7 +471,7 @@ Jedna strona (Jinja2 template + vanilla JS + CSS, serwowane z tego samego FastAP - Pole zapytania + submit (Enter albo przycisk). - Wyniki grupowane po `envelope_id` (dokument), w obrębie dokumentu chunki posortowane po `dist`. -- Kolorowanie progów (progi z fazy 3, `docs/kb/modules/05-faza3-plan.md` §1.2, +- Kolorowanie progów (progi z fazy 3, `kb/phases/kb-m5-faza3.md` §1.2, zweryfikowane bramką): `dist < 0.45` zielony, `0.45–0.55` żółty, `> 0.55` — **nie renderować wyniku**, tylko komunikat "brak odpowiedzi w KB" (żółta/czerwona strefa nadal renderuje wynik z ostrzeżeniem wizualnym; czerwona = brak sensownego diff --git a/kb/phases/monitoring-floty-prometheus.md b/kb/phases/monitoring-floty-prometheus.md index 9db151f..36cec2b 100644 --- a/kb/phases/monitoring-floty-prometheus.md +++ b/kb/phases/monitoring-floty-prometheus.md @@ -40,7 +40,7 @@ historia incydentów, out-of-band watchdog. mózg NIETKNIĘTY, debounce per-alert (klucz alertname:node w state.json). 12 testów pass. PENDING zamknięty 2026-07-02: poll POTWIERDZONY reconem (obraz zbudowany po `62d6fc0`, `PROMETHEUS_URL` w `.env` i w env kontenera, zero poll failed) — patrz - `docs/infra/inventory-verify-2026-07-02.md`. + `kb/subsystems/fleet-inventory-verify.md`. ✅ END-TO-END UDOWODNIONE 2026-07-06 (Etap 0 cutoveru): tor Prometheus → watchdog → Telegram potwierdzony w produkcji testem `AlertTestEtap0` (firing → log polla → Telegram → rollback) — patrz `docs/sessions/2026-07-06.md`. diff --git a/kb/phases/prometheus-cutover-etap2.md b/kb/phases/prometheus-cutover-etap2.md index 1bfc4dd..82d83f6 100644 --- a/kb/phases/prometheus-cutover-etap2.md +++ b/kb/phases/prometheus-cutover-etap2.md @@ -164,7 +164,7 @@ Skutek uboczny do zaakceptowania świadomie: syntetyczne `node_stale`/ wyłączeniu **wcześniej, ale nie liczniej** — te eventy już dziś powstają co noc (21:32/21:39 dla lustro, każdorazowo dla solaria). Cutover nie zwiększa wolumenu alertów. Docelowe wyciszenie planowych okien off to wątek anomaly-detection -z backlogu (`docs/backlog.md:390-406`) — **niezależny od cutoveru i nieblokujący**; +z backlogu (`kb/phases/backlog.md:390-406`) — **niezależny od cutoveru i nieblokujący**; `NodeDown` dla solaria/lustro słusznie pozostaje wyłączony. ### Rekomendowany mapping (potwierdzenie rekomendacji z recon F/Etap 2) diff --git a/kb/phases/subsystem-a-naprawa.md b/kb/phases/subsystem-a-naprawa.md index 986a1a0..7742cbe 100644 --- a/kb/phases/subsystem-a-naprawa.md +++ b/kb/phases/subsystem-a-naprawa.md @@ -9,7 +9,7 @@ links: [] # Plan naprawy subsystemu A (control-plane) — 2026-07-28 -Kontekst: docs/architecture/RECON-multiagent-2026-07-27.md. Decyzje bazowe: +Kontekst: kb/subsystems/recon-multiagent.md. Decyzje bazowe: - Flota dzieli się na dwa subsystemy. A = utrzymaniowy (control-plane, node-agenty, self-healing) — ten plan. B = zleceniowy (dyspozytor + Telegram + KB + HA + homelab-ops) — prowadzony w osobnym projekcie, poza tym planem. diff --git a/kb/runbooks/ha-diag-agent-runbook.md b/kb/runbooks/ha-diag-agent-runbook.md index 607d703..2007c23 100644 --- a/kb/runbooks/ha-diag-agent-runbook.md +++ b/kb/runbooks/ha-diag-agent-runbook.md @@ -12,7 +12,7 @@ links: ## First-time deployment -See **[DEPLOY.md](DEPLOY.md)** for the full procedure: HA token creation, +See **[DEPLOY.md](ha-diag-agent-deploy.md)** for the full procedure: HA token creation, per-host `.env` config, deploy commands, verification steps, 48h shadow-mode observation, and rollback. diff --git a/kb/runbooks/kb-query-deploy.md b/kb/runbooks/kb-query-deploy.md index e4fd5f4..33d6da0 100644 --- a/kb/runbooks/kb-query-deploy.md +++ b/kb/runbooks/kb-query-deploy.md @@ -13,7 +13,7 @@ links: ## Deploy (PIHA) 0. Prerequisite: `ollama-piha` deployed and `bge-m3` pulled — see - `services/ollama-piha/README.md` (the pull is a **manual** deploy step). + `kb/services/ollama-piha.md` (the pull is a **manual** deploy step). 1. `git pull` on PIHA (`~/homelab-codex-ws`). 2. `cp services/kb-query/env.example services/kb-query/.env` and fill in the real `KB_DSN` password (the template already sets `EMBED_FALLBACK_URL`). diff --git a/kb/runbooks/ollama-solaria-cutover.md b/kb/runbooks/ollama-solaria-cutover.md index bbce353..011d069 100644 --- a/kb/runbooks/ollama-solaria-cutover.md +++ b/kb/runbooks/ollama-solaria-cutover.md @@ -97,7 +97,7 @@ bind-mount the existing model directory.** docker exec ollama ollama ps ``` 3. Embeddings endpoint + vector dimension (deferred check from - `docs/kb/modules/05-faza2-plan.md` §6 step 2): + `kb/phases/kb-m5-faza2.md` §6 step 2): ```bash curl -s http://localhost:11434/api/embeddings -d '{"model":"bge-m3","prompt":"test"}' \ | python3 -c "import json,sys; v=json.load(sys.stdin)['embedding']; print(len(v))" @@ -145,7 +145,7 @@ SOLARIA: - Given the missing driver, the cutover proceeded **in CPU-only mode**: the `deploy.resources` GPU reservation was commented out in `services/ollama/docker-compose.yml` (commit `f57a01a`), and the driver fix - was filed as a backlog item (see `docs/backlog.md`) blocking the module 5 + was filed as a backlog item (see `kb/phases/backlog.md`) blocking the module 5 mail-embedding phase. - **2026-07-16: driver fixed.** Installed `nvidia-driver-595-open` from the distro repository — not the old `ppa:graphics-drivers/ppa` (jammy), which diff --git a/kb/runbooks/paperless-cutover.md b/kb/runbooks/paperless-cutover.md index 2352d26..3b4963d 100644 --- a/kb/runbooks/paperless-cutover.md +++ b/kb/runbooks/paperless-cutover.md @@ -23,7 +23,7 @@ links: wildcard `*.kapala.org` już pokrywa tę subdomenę, nowy cert niepotrzebny. Plus vhost w npm@PIHA (HTTPS → `192.168.31.5:8210`, Advanced puste). 6. `docker compose up -d` + `./healthcheck.sh` + testowy login OIDC. -7. Export NFS dla workera (patrz `services/paperless-worker/README.md`) — +7. Export NFS dla workera (patrz `kb/services/paperless-worker.md`) — dopiero przy module 3. 8. Wpis w `hosts/piha/services.yaml` (dopiero przy deployu — wcześniej supervisor widziałby drift dla nieistniejącego serwisu). diff --git a/kb/runbooks/paperless-worker-deploy.md b/kb/runbooks/paperless-worker-deploy.md index 7c517cb..30ecdb7 100644 --- a/kb/runbooks/paperless-worker-deploy.md +++ b/kb/runbooks/paperless-worker-deploy.md @@ -64,4 +64,4 @@ Weryfikacja po deployu: `./healthcheck.sh` robi test zapisu na mount. `inventory/topology.yaml` (obecnie SOLARIA ma tam tylko `node-agent`) — bez tego supervisor/observer nie widzą tego serwisu w desired-state, więc drift między `hosts/solaria/services.yaml` a rzeczywistością nie jest - wykrywany. Patrz `docs/backlog.md`. + wykrywany. Patrz `kb/phases/backlog.md`. diff --git a/kb/services/control-plane.md b/kb/services/control-plane.md index c0c479c..c8742c0 100644 --- a/kb/services/control-plane.md +++ b/kb/services/control-plane.md @@ -5,7 +5,8 @@ visibility: private status: active updated: 2026-08-03 stub: true -links: [] +links: + - ../subsystems/control-plane.md --- # control-plane diff --git a/kb/services/ha-mcp.md b/kb/services/ha-mcp.md index 8f0dab0..71d7a71 100644 --- a/kb/services/ha-mcp.md +++ b/kb/services/ha-mcp.md @@ -10,7 +10,7 @@ links: # ha-mcp — read-only MCP server for Home Assistant -**Status: phase 2a** of `services/home-assistant/DESIGN.md` — own minimal MCP +**Status: phase 2a** of `kb/decisions/ha-configs-as-code.md` — own minimal MCP server (the alternative, adopting `hass-mcp`, was the other option in that document's Open questions; operator decision 2026-07-30: build our own). @@ -38,7 +38,7 @@ in this server that changes anything in Home Assistant: **The write path back into Home Assistant is unchanged and lives elsewhere:** edit `services/home-assistant/config//` in the repo, then `scripts/ha/deploy.sh ` (drift-abort → `check_config` → write → -verify). See `services/home-assistant/DESIGN.md`, "Sync model" and +verify). See `kb/decisions/ha-configs-as-code.md`, "Sync model" and "Validation gate". Nothing in this server bypasses that, and nothing in this server should ever learn to. @@ -140,7 +140,7 @@ down — you then get `live_error` instead of `live`. ## Why `unavailable` is in every result -The 2026-07-23 audit (`services/home-assistant/docs/audyt-automatyzacji-2026-07-23.md`, +The 2026-07-23 audit (`kb/audits/ha-automatyzacje-2026-07-23.md`, 1.2–1.4) traced ~15 silently broken automations to dead sensors: a condition on an `unavailable` entity is simply never true, and HA reports no error. So every entity view here carries an explicit `unavailable: true` plus diff --git a/kb/services/home-assistant-ken-legacy.md b/kb/services/home-assistant-ken-legacy.md index 280325d..5fbc03b 100644 --- a/kb/services/home-assistant-ken-legacy.md +++ b/kb/services/home-assistant-ken-legacy.md @@ -15,7 +15,7 @@ Assistant instance: the `homeassistant5` container on piha, wound down 2026-07. It is kept for historical reference only — e.g. recovering the logic of an old automation — not as a live or deployable instance. -See `services/home-assistant/DESIGN.md`, "Incident log" (2026-07-22) for +See `kb/decisions/ha-configs-as-code.md`, "Incident log" (2026-07-22) for why this instance exists separately from the canonical `ken` (now the Home Assistant OS instance on the RPi4 at 192.168.31.7): this container kept running after the real migration and was firing automations in diff --git a/kb/services/job-documents-ingest.md b/kb/services/job-documents-ingest.md index f2f85b3..da1c5dd 100644 --- a/kb/services/job-documents-ingest.md +++ b/kb/services/job-documents-ingest.md @@ -11,7 +11,7 @@ links: # documents-ingest -One-shot job CLI, Phase 1 of module 5 (`docs/kb/modules/05-documents-ingest.md`, +One-shot job CLI, Phase 1 of module 5 (`kb/phases/kb-m5-documents-ingest.md`, "Domkniecie dlugu z maili"). Extracts a **sample** of PDF attachments from the Gmail `.eml` archive (already indexed in the `envelope` table of kb-postgres) and drops them into Paperless' `consume/` directory so Paperless does the OCR @@ -98,7 +98,7 @@ skipped. `__.pdf`. Files are written with a best-effort `chown` to uid:gid `1000:1000` (the -Paperless container's `USERMAP_UID/GID`, see `services/paperless/README.md`) +Paperless container's `USERMAP_UID/GID`, see `kb/services/paperless.md`) so Paperless can read them. If the chown fails (e.g. the job isn't running as root/uid 1000), a warning is logged but the run continues — the write itself already succeeded; fix ownership/perms on `consume/` separately if needed. diff --git a/kb/services/job-gmail-header-backfill.md b/kb/services/job-gmail-header-backfill.md index 9870d82..5ee4e12 100644 --- a/kb/services/job-gmail-header-backfill.md +++ b/kb/services/job-gmail-header-backfill.md @@ -10,7 +10,7 @@ links: # gmail-header-backfill -One-shot job, module 5 phase 2 backfill (`docs/kb/modules/05-faza2-plan.md`, §5). +One-shot job, module 5 phase 2 backfill (`kb/phases/kb-m5-faza2.md`, §5). Backfills `{"type": "headers", ...}` onto the 225 030 existing `source='gmail'` envelope rows in kb-postgres, which today carry only an attachment manifest (`{"type": "attachment", ...}`) — no `from`/`to`/`cc`/`delivered_to`/`subject` diff --git a/kb/services/job-mail-body-ingest.md b/kb/services/job-mail-body-ingest.md index 757499e..ee4a9c1 100644 --- a/kb/services/job-mail-body-ingest.md +++ b/kb/services/job-mail-body-ingest.md @@ -10,7 +10,7 @@ links: # mail-body-ingest -Module 5, faza mailowa (`docs/kb/modules/05-faza-mailowa-plan.md`, §5, Krok 2). Second full +Module 5, faza mailowa (`kb/phases/kb-m5-faza-mailowa.md`, §5, Krok 2). Second full pass over the gmail `.eml` archive — `gmail-bulk-import` deliberately skipped inline `text/plain`/`text/html` parts (`_parse_attachments` does `continue` on them); this job reads exactly the content that gap left out, chunks it, embeds it, and inserts it into diff --git a/kb/services/kb-postgres.md b/kb/services/kb-postgres.md index e414799..13846ae 100644 --- a/kb/services/kb-postgres.md +++ b/kb/services/kb-postgres.md @@ -98,7 +98,7 @@ Indexes: ## Schema contract -The `envelope` table is the frozen cross-source envelope (see `docs/kb/kb-00-overview.md` §Zasady przekrojowe). Adding columns is OK; removing or renaming existing ones is NOT. +The `envelope` table is the frozen cross-source envelope (see `kb/subsystems/kb-overview.md` §Zasady przekrojowe). Adding columns is OK; removing or renaming existing ones is NOT. Future migrations go in `init/` as `002_*.sql`, `003_*.sql`, …. Postgres runs `initdb` scripts only on a fresh volume — for existing instances apply migrations with `psql` directly. diff --git a/kb/services/kb-query.md b/kb/services/kb-query.md index 0df1fee..0b086ea 100644 --- a/kb/services/kb-query.md +++ b/kb/services/kb-query.md @@ -167,4 +167,4 @@ existed. - Calibration verdict for the fallback (plan §5 steps 4–5: live PIHA measurements → keep/tune/degrade decision) — the mechanism is built and default-on when `EMBED_FALLBACK_URL` is set; the measurements are the - operator's post-deploy step (see `services/ollama-piha/README.md`). + operator's post-deploy step (see `kb/services/ollama-piha.md`). diff --git a/kb/services/mosquitto.md b/kb/services/mosquitto.md index d809584..f229fb1 100644 --- a/kb/services/mosquitto.md +++ b/kb/services/mosquitto.md @@ -11,9 +11,9 @@ superseded_by: "hosts/chelsty-infra/runtime/mosquitto/ (chelsty) + host systemd # Mosquitto MQTT Broker — NOT DEPLOYED / LEGACY > **Status (2026-07-28, truth cleanup):** this manifest matches **nothing that -> actually runs** (recon `docs/architecture/RECON-multiagent-2026-07-27.md`, C8). +> actually runs** (recon `kb/subsystems/recon-multiagent.md`, C8). > The broker on **vps** is part of the legacy ai-cluster stack (being -> decommissioned, see `docs/architecture/ai-cluster-LEGACY.md`); the broker on +> decommissioned, see `kb/decisions/ai-cluster-legacy.md`); the broker on > **piha** is a host systemd mosquitto (OS package) with no repo definition; > **chelsty** has its own config under `hosts/chelsty-infra/runtime/mosquitto/`. > Do not deploy from this directory. Kept until the MQTT topology decision diff --git a/kb/services/nextcloud.md b/kb/services/nextcloud.md index edb38b5..54ac896 100644 --- a/kb/services/nextcloud.md +++ b/kb/services/nextcloud.md @@ -11,7 +11,7 @@ links: # Nextcloud (drive / WebDAV) -Drugi adapter dokumentów filaru KB #2 (moduł 4, `docs/kb/modules/04-nextcloud.md`): +Drugi adapter dokumentów filaru KB #2 (moduł 4, `kb/phases/kb-m4-nextcloud.md`): zamiennik Google Drive — dowolne pliki + sync telefon/desktop, źródło dla ingestu KB (moduł 5) przez WebDAV. diff --git a/kb/services/ollama-piha.md b/kb/services/ollama-piha.md index 6f8b73e..f8da905 100644 --- a/kb/services/ollama-piha.md +++ b/kb/services/ollama-piha.md @@ -46,5 +46,5 @@ SOLARIA with a ~30 s cache and only sends embeds here while SOLARIA is down. kb-query verifies at first use that this backend actually serves `bge-m3` (`/api/tags`) and refuses to embed against a mismatched model. Configuration: `EMBED_FALLBACK_URL=http://192.168.31.5:11434` in `services/kb-query/.env`. -See `services/kb-query/README.md` for the fallback verification plan (tests +See `kb/services/kb-query.md` for the fallback verification plan (tests A/B/C). diff --git a/kb/services/paperless-worker.md b/kb/services/paperless-worker.md index f0db938..ffbfa72 100644 --- a/kb/services/paperless-worker.md +++ b/kb/services/paperless-worker.md @@ -10,7 +10,7 @@ links: # Paperless OCR worker (SOLARIA) -Ciężki OCR filaru KB #2 (moduł 3, `docs/kb/modules/03-paperless-ocr-worker.md`). +Ciężki OCR filaru KB #2 (moduł 3, `kb/phases/kb-m3-ocr-worker.md`). Ten sam obraz co `services/paperless/`, ale z nadpisanym poleceniem — działa **wyłącznie** jako celery worker podpięty do brokera/bazy paperless@PIHA, ze storage widzianym przez NFS z PIHA. Realizuje wzorzec kb-02 diff --git a/kb/services/paperless.md b/kb/services/paperless.md index 7a467a3..d647145 100644 --- a/kb/services/paperless.md +++ b/kb/services/paperless.md @@ -11,7 +11,7 @@ links: # Paperless-ngx (serwis, PIHA) -Serwis dokumentów filaru KB #2 (moduł 2, `docs/kb/modules/02-paperless-service.md`): +Serwis dokumentów filaru KB #2 (moduł 2, `kb/phases/kb-m2-paperless.md`): UI + API + Postgres + Redis. Always-on na PIHA. Ciężki OCR wykonuje osobny worker na SOLARIA (`services/paperless-worker/`, moduł 3) — tu zostaje tylko wolny fallback (1 worker × 1 wątek). diff --git a/kb/subsystems/agent-operating-procedures.md b/kb/subsystems/agent-operating-procedures.md index 4bedc6a..5281669 100644 --- a/kb/subsystems/agent-operating-procedures.md +++ b/kb/subsystems/agent-operating-procedures.md @@ -15,9 +15,9 @@ This document defines the operating procedures, constraints, and interaction pro 1. **Read-Only by Default**: Agents should assume read-only access to the `/opt/homelab` runtime unless explicitly executing an approved action. 2. **Git as Authority**: The repository on **SATURN** is the source of truth. Agents must not modify the runtime state on nodes directly without corresponding (or pending) Git state, unless it's an emergency mitigation. -3. **Human-in-the-Loop (HIL)**: All destructive or structural changes (restarts, deployments, config changes) must follow the [Action Approval Model](../services/agent-system/action-model.md). +3. **Human-in-the-Loop (HIL)**: All destructive or structural changes (restarts, deployments, config changes) must follow the [Action Approval Model](action-approval-model.md). 4. **Idempotency**: All scripts and actions proposed or executed by agents MUST be idempotent. -5. **Context-Awareness**: Agents MUST read the `README.md` and `docs/agents.md` at the start of every session to align with current infrastructure standards. +5. **Context-Awareness**: Agents MUST read the `README.md` and `kb/subsystems/agent-operating-procedures.md` at the start of every session to align with current infrastructure standards. ## 2. Agent Roles diff --git a/kb/subsystems/control-plane.md b/kb/subsystems/control-plane.md index e04a780..4161cfb 100644 --- a/kb/subsystems/control-plane.md +++ b/kb/subsystems/control-plane.md @@ -5,6 +5,7 @@ visibility: private status: deprecated updated: 2026-05-27 links: + - ../services/control-plane.md - ../runbooks/control-plane-deploy-recovery.md superseded_by: "przepisany tor redeploy, commity da151fc/79bfe8c 2026-08-03" --- diff --git a/kb/subsystems/fleet-inventory-verify.md b/kb/subsystems/fleet-inventory-verify.md index 409b4b2..fbc1bb2 100644 --- a/kb/subsystems/fleet-inventory-verify.md +++ b/kb/subsystems/fleet-inventory-verify.md @@ -11,7 +11,7 @@ links: [] Zebrano: 2026-07-02 ~15:20 CEST (read-only recon, zero zmian na nodach). Metoda: ssh + `docker ps -a / inspect / logs`, `free/df/nproc/lscpu/lsblk`, `git branch/log` (odczyt), -`curl` do fleet-prometheus API. Porównanie z `docs/infra/inventory-2026-06-30.md` (23 rozjazdy) +`curl` do fleet-prometheus API. Porównanie z `kb/subsystems/fleet-inventory.md` (23 rozjazdy) oraz z repo na `master` (HEAD `22adfb1`). Dostępność nodów podczas weryfikacji: diff --git a/kb/subsystems/recon-multiagent.md b/kb/subsystems/recon-multiagent.md index 50ce545..517c066 100644 --- a/kb/subsystems/recon-multiagent.md +++ b/kb/subsystems/recon-multiagent.md @@ -615,7 +615,7 @@ No runtime state was touched. **Legacy** -- `docs/architecture/ai-cluster-LEGACY.md`: ai-cluster is retired in place, +- `kb/decisions/ai-cluster-legacy.md`: ai-cluster is retired in place, not migrated (bus idle since 2026-06-09, C9); branch `task/ai-cluster-solaria` stays unmerged as documentation; surviving patterns listed; runtime retirement runbook (stop stack on vps, observe `free -m`, remove containers) diff --git a/packages/kb-mail/src/kb_mail/chunking.py b/packages/kb-mail/src/kb_mail/chunking.py index fc910b4..f525330 100644 --- a/packages/kb-mail/src/kb_mail/chunking.py +++ b/packages/kb-mail/src/kb_mail/chunking.py @@ -1,4 +1,4 @@ -"""Shared chunker — module 5, phase mailowa, plan step 0 (docs/kb/modules/05-faza-mailowa-plan.md, +"""Shared chunker — module 5, phase mailowa, plan step 0 (kb/phases/kb-m5-faza-mailowa.md, §3). Moved 1:1 out of `jobs/documents_ingest/chunk_embed.py` so both the paperless job and `jobs/mail-body-ingest` chunk against the exact same tested logic instead of two implementations drifting apart (the phase-4 lesson for `retrieval.py` -> `packages/kb-retrieval`, applied again). diff --git a/packages/kb-mail/tests/test_migration.py b/packages/kb-mail/tests/test_migration.py index f58a5d3..d4bac31 100644 --- a/packages/kb-mail/tests/test_migration.py +++ b/packages/kb-mail/tests/test_migration.py @@ -71,7 +71,7 @@ def test_migration_002_references_envelope_with_cascade(): def test_migration_002_embedding_dimension_matches_bge_m3(): sql = (INIT_DIR / "002_chunks.sql").read_text() - # bge-m3 dense embedding dimension is 1024 (docs/kb/modules/05-faza2-plan.md §1.4) + # bge-m3 dense embedding dimension is 1024 (kb/phases/kb-m5-faza2.md §1.4) assert "VECTOR(1024)" in sql.upper() diff --git a/packages/kb-retrieval/src/kb_retrieval/embed.py b/packages/kb-retrieval/src/kb_retrieval/embed.py index a2aad28..0ae5e06 100644 --- a/packages/kb-retrieval/src/kb_retrieval/embed.py +++ b/packages/kb-retrieval/src/kb_retrieval/embed.py @@ -1,4 +1,4 @@ -"""Ollama embedding client -- module 5, phase 4, plan step 0 (docs/kb/modules/05-faza4-plan.md, +"""Ollama embedding client -- module 5, phase 4, plan step 0 (kb/phases/kb-m5-faza4.md, §3, decision 1). Moved 1:1 out of `jobs/documents_ingest/chunk_embed.py` (`embed_chunk`, `_vector_literal`, `DEFAULT_MODEL`, `DEFAULT_OLLAMA_URL`) so both a venv-based host job (`documents-ingest`) and a long-lived Docker service (`kb-query`) can depend on the same tested @@ -8,7 +8,7 @@ client without the service image pulling in `jobs/`'s `anthropic` dependency and health-check + circuit breaker (plan §2 decision 2) builds on, kept in this module so the HTTP probing logic lives in exactly one place. -`embed_batch` -- module 5, faza mailowa, plan Krok 1 (docs/kb/modules/05-faza-mailowa-plan.md, +`embed_batch` -- module 5, faza mailowa, plan Krok 1 (kb/phases/kb-m5-faza-mailowa.md, §4, decision 5): `/api/embed` with `input` as a list, batch 64 measured live at ~8-18 ms/chunk vs ~150-200 ms/chunk sequential through `embed_chunk`'s `/api/embeddings` (plan §1.4). Used by `jobs/mail-body-ingest` only -- `embed_chunk` stays the single-text path for kb-query (one query diff --git a/packages/kb-retrieval/src/kb_retrieval/retrieval.py b/packages/kb-retrieval/src/kb_retrieval/retrieval.py index 443686f..8b8f816 100644 --- a/packages/kb-retrieval/src/kb_retrieval/retrieval.py +++ b/packages/kb-retrieval/src/kb_retrieval/retrieval.py @@ -1,12 +1,12 @@ -"""Retrieval module -- module 5, phase 3, plan step 4 (docs/kb/modules/05-faza3-plan.md, §6). -Moved 1:1 into `packages/kb-retrieval` in phase 4 (docs/kb/modules/05-faza4-plan.md, §3, +"""Retrieval module -- module 5, phase 3, plan step 4 (kb/phases/kb-m5-faza3.md, §6). +Moved 1:1 into `packages/kb-retrieval` in phase 4 (kb/phases/kb-m5-faza4.md, §3, decision 1) so both `documents-ingest` (venv job) and `kb-query` (Docker service) share one tested module instead of the service image needing to pull in all of `jobs/documents-ingest`. Three retrieval paths over the same corpus, sharing one query embedding (bge-m3, via Ollama): - `flat_retrieve`: baseline -- ranks every active `document_chunk` row directly. This - formalizes the pilot's ad hoc `/tmp/kbq.sh` query (docs/kb/eval/retrieval-pilot-2026-07-16.md) + formalizes the pilot's ad hoc `/tmp/kbq.sh` query (kb/phases/kb-m5-eval-retrieval-pilot.md) into a tested, versioned module instead of a script living only in a session transcript. - `cascade_retrieve`: pre-filters to the top-N `document_summary` matches for one configured `model` (plan §2 decision 3, resolved 2026-07-17 as D3: `claude-haiku-4-5` is the compilation @@ -15,7 +15,7 @@ Three retrieval paths over the same corpus, sharing one query embedding (bge-m3, -- it is an architecture test for the mail-scale corpus (225k envelopes, plan §1.1) where a flat chunk scan stops being cheap. `eval/retrieval_eval.py` runs the quality gate (plan §6.2) that decides whether it becomes the default path. -- `hybrid_retrieve` -- module 5, faza mailowa, plan Krok 3 (docs/kb/modules/05-faza-mailowa-plan.md, +- `hybrid_retrieve` -- module 5, faza mailowa, plan Krok 3 (kb/phases/kb-m5-faza-mailowa.md, §6, decision 6): mail (gmail) envelopes never get a `document_summary` (streszczenie maila would usually be longer than the mail itself -- decision 6's rejected-summaries reasoning), so they are invisible to the cascade's stage 1 pre-filter. Hybrid runs the cascade for sources that diff --git a/packages/kb-retrieval/tests/test_retrieval.py b/packages/kb-retrieval/tests/test_retrieval.py index 904c2d8..410bca3 100644 --- a/packages/kb-retrieval/tests/test_retrieval.py +++ b/packages/kb-retrieval/tests/test_retrieval.py @@ -98,7 +98,7 @@ class _FakeSession: class TestDefaults: def test_plan_start_values(self): # plan §6.1: "Start: N=10, k=5" -- a regression guard against silently drifting off - # the value the quality gate (docs/kb/modules/05-faza3-plan.md §6.2) was run against. + # the value the quality gate (kb/phases/kb-m5-faza3.md §6.2) was run against. assert DEFAULT_N == 10 assert DEFAULT_K == 5 diff --git a/scripts/ha/deploy.sh b/scripts/ha/deploy.sh index 215a910..793c71c 100755 --- a/scripts/ha/deploy.sh +++ b/scripts/ha/deploy.sh @@ -1,6 +1,6 @@ #!/usr/bin/env bash # Write path: repo -> HA instance, for the "api" adapter only (see -# services/home-assistant/DESIGN.md, "Deploy path: adapter per instance", +# kb/decisions/ha-configs-as-code.md, "Deploy path: adapter per instance", # "Sync model", "Validation gate"). # # Scope: automations/scripts/scenes only. Dashboards and helpers are diff --git a/scripts/ha/import.sh b/scripts/ha/import.sh index 50a5bab..b8713fd 100755 --- a/scripts/ha/import.sh +++ b/scripts/ha/import.sh @@ -1,7 +1,7 @@ #!/usr/bin/env bash # Read-only import of one Home Assistant instance's /config tree (plus a # curated .storage/* export and an /api/states fixture snapshot) into this -# repo, normalized and split per services/home-assistant/DESIGN.md. +# repo, normalized and split per kb/decisions/ha-configs-as-code.md. # # This script NEVER writes to the HA instance. It is idempotent: re-running # against an unchanged instance produces no diff under diff --git a/scripts/ha/lib/deploy_api.py b/scripts/ha/lib/deploy_api.py index fe421fa..2b47270 100755 --- a/scripts/ha/lib/deploy_api.py +++ b/scripts/ha/lib/deploy_api.py @@ -1,6 +1,6 @@ #!/usr/bin/env python3 """Deploy path (repo -> instance) for the HA "api" adapter — see -services/home-assistant/DESIGN.md, "Sync model" and "Validation gate". +kb/decisions/ha-configs-as-code.md, "Sync model" and "Validation gate". Scope: automations/scripts/scenes only (the domains reachable via `/api/config//config/`). Dashboards and helpers are WS-only — diff --git a/scripts/ha/lib/ha_api.py b/scripts/ha/lib/ha_api.py index 4ea7019..c7fac99 100755 --- a/scripts/ha/lib/ha_api.py +++ b/scripts/ha/lib/ha_api.py @@ -5,7 +5,7 @@ adapter per instance"). Every entrypoint here takes a *token_path*, never a token value, and reads the token from disk itself. The bearer token must never appear as a process argv (visible to any local user via `ps`) or in a log line — see DESIGN.md, -"Tokens", and the api-adapter requirements in docs/backlog.md. Callers +"Tokens", and the api-adapter requirements in kb/phases/backlog.md. Callers needing the token value in-process (e.g. the WebSocket client in ha_ws.py) should call read_token() directly rather than shelling out. diff --git a/scripts/ha/lib/ha_write_api.py b/scripts/ha/lib/ha_write_api.py index 24d1ef6..06325e5 100755 --- a/scripts/ha/lib/ha_write_api.py +++ b/scripts/ha/lib/ha_write_api.py @@ -1,6 +1,6 @@ #!/usr/bin/env python3 """Write-capable REST client for the HA "api" adapter's deploy path (see -services/home-assistant/DESIGN.md, "Sync model", "Validation gate"). +kb/decisions/ha-configs-as-code.md, "Sync model", "Validation gate"). Deliberately kept out of ha_api.py, whose module docstring guarantees that module is read-only by construction — the import path relies on that diff --git a/scripts/ha/lib/ha_ws.py b/scripts/ha/lib/ha_ws.py index 82a5ace..1c3b73c 100755 --- a/scripts/ha/lib/ha_ws.py +++ b/scripts/ha/lib/ha_ws.py @@ -9,7 +9,7 @@ commands this module ever sends are `auth` and whatever `*_list`/ has no method that issues a create/update/delete command. Depends on the `websocket-client` PyPI package (import name `websocket`), -which is not in the stdlib. See services/home-assistant/README.md, +which is not in the stdlib. See kb/runbooks/home-assistant-deploy.md, "WebSocket dependency", for the install command. Importing this module raises ImportError with an actionable message if it's missing, instead of the default traceback, so the api adapter can surface one clear line @@ -25,7 +25,7 @@ except ImportError as exc: # pragma: no cover - exercised via import.sh error p "the 'websocket-client' package is required for the api adapter's " "dashboard/registry/helper export (HA's WebSocket API) — install it " "with 'sudo apt install python3-websocket' or 'pip install --user " - "websocket-client' (see services/home-assistant/README.md, " + "websocket-client' (see kb/runbooks/home-assistant-deploy.md, " "'WebSocket dependency')" ) from exc diff --git a/scripts/ha/lib/normalize.py b/scripts/ha/lib/normalize.py index 711e07d..25656e9 100755 --- a/scripts/ha/lib/normalize.py +++ b/scripts/ha/lib/normalize.py @@ -2,7 +2,7 @@ """Canonical YAML normalization used by every HA import path. One function, one style, used everywhere data lands in the repo (see -services/home-assistant/DESIGN.md, "Canonical format"): block collections, +kb/decisions/ha-configs-as-code.md, "Canonical format"): block collections, indent width 2, unlimited line width, sorted keys, UTF-8. HA configs use custom `!`-prefixed tags (`!include`, `!secret`, `!env_var`, diff --git a/scripts/ha/lib/split.py b/scripts/ha/lib/split.py index 405cae2..478754e 100755 --- a/scripts/ha/lib/split.py +++ b/scripts/ha/lib/split.py @@ -1,7 +1,7 @@ #!/usr/bin/env python3 """Split HA list/dict config files into one-object-per-file layouts. -See services/home-assistant/DESIGN.md ("Split + normalization"): +See kb/decisions/ha-configs-as-code.md ("Split + normalization"): automations.yaml -> automations/.yaml scenes.yaml -> scenes/.yaml scripts.yaml -> scripts/.yaml diff --git a/scripts/kb/check_okf.py b/scripts/kb/check_okf.py index 335cbb8..c0c6d6c 100755 --- a/scripts/kb/check_okf.py +++ b/scripts/kb/check_okf.py @@ -20,8 +20,7 @@ reguły tego repo: 11. `stub` — o ile obecne — musi być boolem. Zakres domyślny: kb/ oraz docs/sessions/. Reszta repo (CLAUDE.md, README.md, -.claude/skills/, docs/backlog.md itd.) leży poza bazą wiedzy i nie podlega -walidacji. +.claude/skills/ itd.) leży poza bazą wiedzy i nie podlega walidacji. Tylko biblioteka standardowa: minimalny parser YAML wystarczający dla frontmatterów w tym repo (klucze skalarne, listy inline, listy blokowe). diff --git a/scripts/npm/npm_api.py b/scripts/npm/npm_api.py index 0660a17..60f00da 100644 --- a/scripts/npm/npm_api.py +++ b/scripts/npm/npm_api.py @@ -30,7 +30,7 @@ INSTANCES = { "vps": { "url_env": "NPM_VPS_URL", # Tailscale mesh address of the admin panel — the public IP - # (135.181.153.108:81) must never be used here, see docs/backlog.md. + # (135.181.153.108:81) must never be used here, see kb/phases/backlog.md. "url_default": "http://100.95.58.48:81", "user_env": "NPM_VPS_USER", "pass_env": "NPM_VPS_PASS", diff --git a/scripts/observer/observer.py b/scripts/observer/observer.py index 27784c1..d2d8656 100644 --- a/scripts/observer/observer.py +++ b/scripts/observer/observer.py @@ -239,7 +239,7 @@ class Observer: # offline (e.g. chelsty site, hardware down since ~2026-06-01). The # observer keeps their last-known world state but never reclassifies # liveness for them, so no node_offline/node_stale/node_online events - # are emitted while a node is dormant. See docs/architecture/ARCHITEKTURA.md. + # are emitted while a node is dormant. See kb/decisions/architektura-2026-07-28.md. self.dormant_nodes = { name for name, info in self.inventory["nodes"].items() if info.get("status") == "dormant" diff --git a/services/control-plane/src/executor.py b/services/control-plane/src/executor.py index 3618d9d..c9feee1 100644 --- a/services/control-plane/src/executor.py +++ b/services/control-plane/src/executor.py @@ -35,7 +35,7 @@ REPO_ROOT = Path(os.getenv("REPO_ROOT", "/repo")) # SSH configuration # SSH_USER can be overridden per-deployment environment. # Still used by _execute_disk_cleanup (out of scope for this change — see -# docs/backlog.md "shadow_mode -> remediacja"). container_restart no longer +# kb/phases/backlog.md "shadow_mode -> remediacja"). container_restart no longer # uses SSH: see _dispatch_container_restart / _reconcile_running_actions. SSH_USER = os.getenv("SSH_USER", "oskar") SSH_OPTIONS = [ @@ -144,7 +144,7 @@ class Executor: return elif action_type == "container_restart": - # No SSH from the executor (see CLAUDE.md / docs/backlog.md): the + # No SSH from the executor (see CLAUDE.md / kb/phases/backlog.md): the # action is handed to the node-agent running on the target node, # which restarts the container locally via its own docker socket. # container_name is set by the supervisor; falls back to service name. @@ -198,7 +198,7 @@ class Executor: # ------------------------------------------------------------------ # # The executor has no SSH client and no key to the fleet (deliberate — - # see docs/backlog.md "PROJEKT: remediacja bez SSH"). Instead, it writes a + # see kb/phases/backlog.md "PROJEKT: remediacja bez SSH"). Instead, it writes a # small dispatch file that the node-agent running ON the target node picks # up and executes locally through its own docker socket. This works # identically whether the target is a remote node (piha/solaria/... reach diff --git a/services/control-plane/src/supervisor.py b/services/control-plane/src/supervisor.py index 4ca5303..48407e4 100644 --- a/services/control-plane/src/supervisor.py +++ b/services/control-plane/src/supervisor.py @@ -47,7 +47,7 @@ except Exception: # redeploy, which now actually executes. # mqtt_unreachable was removed 2026-07-28: the observer never creates incidents # with that trigger_type, so the branch was dead code (recon -# docs/architecture/RECON-multiagent-2026-07-27.md, D15). stability-agent still +# kb/subsystems/recon-multiagent.md, D15). stability-agent still # emits the mqtt_unreachable *event*; it just never becomes an incident. CONTAINER_RESTART_TRIGGERS = {"containers_not_running", "healthcheck_failed"} @@ -147,7 +147,7 @@ class Supervisor: # generates NO actions of any kind for them: their services are excluded # from desired state (which also auto-cancels their stale pending # actions), and node/HA events from them are not routed to alerts. - # See docs/architecture/ARCHITEKTURA.md. + # See kb/decisions/architektura-2026-07-28.md. self.dormant_nodes: set = set() # In-memory set of already-routed HA event IDs; prevents re-processing # on each reconcile cycle. Grows to at most ~hundreds of entries/day. diff --git a/services/control-plane/tests/test_dormant_nodes.py b/services/control-plane/tests/test_dormant_nodes.py index 0e85234..d71c759 100644 --- a/services/control-plane/tests/test_dormant_nodes.py +++ b/services/control-plane/tests/test_dormant_nodes.py @@ -1,6 +1,6 @@ """Dormant-node handling (topology `status: dormant`, etap 0 truth cleanup). -Contract (docs/architecture/ARCHITEKTURA.md): a dormant node keeps its +Contract (kb/decisions/architektura-2026-07-28.md): a dormant node keeps its last-known world state, the observer emits NO liveness events for it, and the supervisor generates NO actions of any kind for it. """ diff --git a/services/control-plane/tests/test_executor_dispatch.py b/services/control-plane/tests/test_executor_dispatch.py index 9b3011d..f073ae3 100644 --- a/services/control-plane/tests/test_executor_dispatch.py +++ b/services/control-plane/tests/test_executor_dispatch.py @@ -1,6 +1,6 @@ """Tests for Executor container_restart dispatch — the no-SSH remediation path. -Covers docs/backlog.md "PROJEKT: remediacja bez SSH": the executor never +Covers kb/phases/backlog.md "PROJEKT: remediacja bez SSH": the executor never shells out to SSH for container_restart. Instead it drops a dispatch file for the target node's node-agent to pick up, and later resolves the action from either a matching action_result event or a timeout — never leaving it running diff --git a/services/fleet-prometheus/rules/kb-ingest.yml b/services/fleet-prometheus/rules/kb-ingest.yml index 5ef6325..6a5aa6a 100644 --- a/services/fleet-prometheus/rules/kb-ingest.yml +++ b/services/fleet-prometheus/rules/kb-ingest.yml @@ -1,5 +1,5 @@ # fleet-prometheus kb-ingest rules — module 5 phase 3 step 5 -# (docs/kb/modules/05-faza3-plan.md §7.2). +# (kb/phases/kb-m5-faza3.md §7.2). # # Same delivery convention as liveness.yml: no Alertmanager, these rules only make alerts # FIRING (visible at GET /api/v1/alerts on this Prometheus instance); brain-watchdog@PIHA diff --git a/services/ha-mcp/run.sh b/services/ha-mcp/run.sh index b38e265..4355e93 100755 --- a/services/ha-mcp/run.sh +++ b/services/ha-mcp/run.sh @@ -17,7 +17,7 @@ else fi if ! "$PY" -c "import mcp" 2>/dev/null; then - echo "ha-mcp: the 'mcp' SDK is not importable by $PY — see services/ha-mcp/README.md, 'Install'" >&2 + echo "ha-mcp: the 'mcp' SDK is not importable by $PY — see kb/services/ha-mcp.md, 'Install'" >&2 exit 1 fi diff --git a/services/ha-mcp/src/ha_mcp/server.py b/services/ha-mcp/src/ha_mcp/server.py index b15f787..c15b4ee 100644 --- a/services/ha-mcp/src/ha_mcp/server.py +++ b/services/ha-mcp/src/ha_mcp/server.py @@ -34,7 +34,7 @@ mcp = MCPServer( "canonical home instance); pass `instance` to target another one. " "These tools cannot change anything in Home Assistant — to change an " "automation, edit services/home-assistant/config// in the repo " - "and deploy with scripts/ha/deploy.sh (see services/home-assistant/DESIGN.md). " + "and deploy with scripts/ha/deploy.sh (see kb/decisions/ha-configs-as-code.md). " "Entities reported with unavailable=true are dead: their automations " "silently do not fire." ), diff --git a/services/kb-query/app/embed_router.py b/services/kb-query/app/embed_router.py index bbf7cf8..75ea64e 100644 --- a/services/kb-query/app/embed_router.py +++ b/services/kb-query/app/embed_router.py @@ -1,4 +1,4 @@ -"""Embed-backend router -- module 5 phase 4, Krok 2 (docs/kb/modules/05-faza4-plan.md §2 +"""Embed-backend router -- module 5 phase 4, Krok 2 (kb/phases/kb-m5-faza4.md §2 Decyzja 2, §5): the active SOLARIA->PIHA fallback state machine `/search` embeds through. One instance per process (`app.state.embed_router`), holding the single global `sol_status` cache the plan describes: diff --git a/services/kb-query/app/links.py b/services/kb-query/app/links.py index 4d27642..ab36574 100644 --- a/services/kb-query/app/links.py +++ b/services/kb-query/app/links.py @@ -1,4 +1,4 @@ -"""Per-source result shaping for `/search` -- module 5 phase 4 (docs/kb/modules/05-faza4-plan.md, +"""Per-source result shaping for `/search` -- module 5 phase 4 (kb/phases/kb-m5-faza4.md, §4 response shape, §6/decision 3 links). `kb_retrieval.retrieval` only ever touches `document_chunk`/`document_summary`; the join to `envelope` (for `source` and link/metadata fields) is new HTTP-layer code that belongs to kb-query, not the shared retrieval package. diff --git a/services/kb-query/app/main.py b/services/kb-query/app/main.py index 812ec94..9258d9f 100644 --- a/services/kb-query/app/main.py +++ b/services/kb-query/app/main.py @@ -1,4 +1,4 @@ -"""kb-query -- module 5 phase 4 (docs/kb/modules/05-faza4-plan.md §4): first user-facing HTTP +"""kb-query -- module 5 phase 4 (kb/phases/kb-m5-faza4.md §4): first user-facing HTTP entry point to the KB. Wraps `kb_retrieval` retrieval (module 5 phase 3, gated PASS -- docs/sessions/2026-07-21.md) in FastAPI. This is a search API, not chat: no answer synthesis, no LLM call over the results (that is phase 5, out of scope here). @@ -15,7 +15,7 @@ silent cross-space distance computation. container (plan §2 decision 4): a Jinja2 shell + a static vanilla-JS file, no node build step. `/` and `/static/*` need no DB/Ollama, so they stay reachable even while `/search` is 503ing. -`mode=hybrid` (faza mailowa, plan Krok 3, docs/kb/modules/05-faza-mailowa-plan.md §6) is +`mode=hybrid` (faza mailowa, plan Krok 3, kb/phases/kb-m5-faza-mailowa.md §6) is available explicitly starting here, but the default stays `cascade` until the quality gate (plan §8) PASSes on the full mail corpus -- flipping the default is a separate, later change. """ diff --git a/services/kb-query/app/search.py b/services/kb-query/app/search.py index 3ef8748..5995468 100644 --- a/services/kb-query/app/search.py +++ b/services/kb-query/app/search.py @@ -1,4 +1,4 @@ -"""`/search` core -- module 5 phase 4 (docs/kb/modules/05-faza4-plan.md §4). Kept decoupled +"""`/search` core -- module 5 phase 4 (kb/phases/kb-m5-faza4.md §4). Kept decoupled from FastAPI so it can be unit-tested with fake `conn`/`session`/`router` objects, the same style as `kb_retrieval`'s own tests, instead of needing a live DB/Ollama behind a TestClient. diff --git a/services/kb-query/app/startup.py b/services/kb-query/app/startup.py index 0189ccc..abecd34 100644 --- a/services/kb-query/app/startup.py +++ b/services/kb-query/app/startup.py @@ -1,4 +1,4 @@ -"""Startup invariant -- module 5 phase 4 (docs/kb/modules/05-faza4-plan.md, §2 decision 2, +"""Startup invariant -- module 5 phase 4 (kb/phases/kb-m5-faza4.md, §2 decision 2, "twardy inwariant"): the configured `EMBED_MODEL` must already be the model behind the active embeddings in both `document_chunk` and `document_summary`, or kb-query refuses to start (crash-loop, visible via container restarts in monitoring -- deliberately loud, never a silent diff --git a/services/nextcloud/docker-compose.yml b/services/nextcloud/docker-compose.yml index 2417d7f..6320d2e 100644 --- a/services/nextcloud/docker-compose.yml +++ b/services/nextcloud/docker-compose.yml @@ -8,7 +8,7 @@ # Host decided: PIHA (not SOLARIA) — Oskar uses Nextcloud actively (phone # sync, family), needs always-on availability; SOLARIA's session-only uptime # was not acceptable for this workload. See -# docs/kb/modules/DECYZJE-do-podjecia.md #1 and README.md. Compose stays +# kb/decisions/kb-dokumenty-otwarte.md #1 and README.md. Compose stays # portable: paths follow the /opt/homelab/data convention, bind IP and proxy # come from .env. # diff --git a/services/nextcloud/service.yaml b/services/nextcloud/service.yaml index 0bc88e5..5995330 100644 --- a/services/nextcloud/service.yaml +++ b/services/nextcloud/service.yaml @@ -5,7 +5,7 @@ service: # dostepnosc SOLARII nie jest akceptowalna dla tego workloadu (w # przeciwienstwie do KB-zapytan, ktore i tak czytaja wlasna kopie z # archiwum na PIHA niezaleznie od hosta Nextclouda). Patrz - # docs/kb/modules/DECYZJE-do-podjecia.md #1 i README.md. + # kb/decisions/kb-dokumenty-otwarte.md #1 i README.md. owner_node: piha role: file-sync-drive # Drive replacement + WebDAV source for KB ingest (archiwum = KOPIA) exposure: private # LAN/Tailscale only via npm@PIHA (cloud.kapala.org); NO public ingress diff --git a/services/node-agent/src/node_agent.py b/services/node-agent/src/node_agent.py index ed37bbc..2e34dba 100644 --- a/services/node-agent/src/node_agent.py +++ b/services/node-agent/src/node_agent.py @@ -90,7 +90,7 @@ VPS_EVENTS_PATH = os.getenv("VPS_EVENTS_PATH", "/opt/homelab/events") # --------------------------------------------------------------------------- # Remediation dispatch (pull-based — reuses the same VPS_EVENTS_HOST / SSH key # as event shipping above, just in the opposite direction). See -# docs/backlog.md "PROJEKT: remediacja bez SSH": the control-plane executor on +# kb/phases/backlog.md "PROJEKT: remediacja bez SSH": the control-plane executor on # VPS has no SSH client and no key to the fleet, so it never reaches out to a # node directly. Instead it drops a small JSON file under # /opt/homelab/actions/dispatch// on VPS; the node-agent for that node @@ -103,7 +103,7 @@ VPS_DISPATCH_PATH = os.getenv("VPS_DISPATCH_PATH", "/opt/homelab/actions/dispatc # Action types the agent is willing to execute on its own. Deliberately just # one to start: redeploy stays VPS/manual, disk_cleanup is out of scope for -# this change (see docs/backlog.md). Anything not in this set is refused with +# this change (see kb/phases/backlog.md). Anything not in this set is refused with # a clear action_result error rather than silently ignored. ALLOWED_DISPATCH_ACTION_TYPES = {"container_restart"} @@ -859,7 +859,7 @@ class NodeAgent: # file transferred fine. Files still keep their mtime via -t; # this only skips the (harmless, doomed) directory mtime set. # TODO tech-debt: fix ownership of /opt/homelab/events/ on - # VPS so this workaround isn't needed (see docs/backlog.md, + # VPS so this workaround isn't needed (see kb/phases/backlog.md, # "Tech-debt: globalny porządek uid/gid/uprawnień"). "--omit-dir-times", # --no-perms/--no-owner/--no-group: same root cause as --omit-dir-times above. @@ -869,7 +869,7 @@ class NodeAgent: # destination dir. File CONTENT still transfers correctly (that is all the # observer needs); only attribute-setting on the remote dir is skipped. # TODO tech-debt: fix ownership of /opt/homelab/events/ on VPS so none - # of these workarounds are needed (see docs/backlog.md, uid/gid section). + # of these workarounds are needed (see kb/phases/backlog.md, uid/gid section). "--no-perms", "--no-owner", "--no-group", diff --git a/services/node-agent/tests/test_action_dispatch.py b/services/node-agent/tests/test_action_dispatch.py index f7b033c..68249c1 100644 --- a/services/node-agent/tests/test_action_dispatch.py +++ b/services/node-agent/tests/test_action_dispatch.py @@ -1,7 +1,7 @@ """Tests for NodeAgent remediation dispatch: pull_dispatched_actions / process_dispatched_actions / _execute_dispatched_action. -Covers the security gates required by docs/backlog.md "PROJEKT: remediacja +Covers the security gates required by kb/phases/backlog.md "PROJEKT: remediacja bez SSH": the agent executes only actions addressed to its own node, refuses anything outside the container_restart whitelist, never restarts its own container, and treats a repeated dispatch of the same action_id as a no-op. diff --git a/services/node_exporter/service.yaml b/services/node_exporter/service.yaml index f0e1f1d..8556f82 100644 --- a/services/node_exporter/service.yaml +++ b/services/node_exporter/service.yaml @@ -1,8 +1,8 @@ name: node_exporter # Deployed per-host: vps and piha (each with its own hosts//services.yaml entry + # hosts//runtime/node_exporter/docker-compose.override.yml). Was owner_node: vps only -# until KB module 5 phase 3 step 5 (docs/kb/modules/05-faza3-plan.md §7.2) needed the -# textfile collector on PIHA — closes the docs/backlog.md "stability-agent / node_exporter +# until KB module 5 phase 3 step 5 (kb/phases/kb-m5-faza3.md §7.2) needed the +# textfile collector on PIHA — closes the kb/phases/backlog.md "stability-agent / node_exporter # owner_node single, biegaja wielomiejscowo -> per-host" item for node_exporter's half. owner_node: per-host role: metrics-exporter diff --git a/services/ollama/docker-compose.yml b/services/ollama/docker-compose.yml index 17258a4..8b30e48 100644 --- a/services/ollama/docker-compose.yml +++ b/services/ollama/docker-compose.yml @@ -15,7 +15,7 @@ services: - /opt/homelab/data/ollama:/root/.ollama # GPU przywrócone 2026-07-16: sterownik nvidia-driver-595-open (repo distro) # zainstalowany, CUDA 13.2, nvidia-container-toolkit już obecny. Patrz - # docs/infra/ollama-solaria-cutover-2026-07-15.md dla historii cutoveru. + # kb/runbooks/ollama-solaria-cutover.md dla historii cutoveru. deploy: resources: reservations: diff --git a/services/paperless/docker-compose.yml b/services/paperless/docker-compose.yml index 31ac7e3..bd73836 100644 --- a/services/paperless/docker-compose.yml +++ b/services/paperless/docker-compose.yml @@ -1,6 +1,6 @@ # Paperless-ngx service stack (PIHA) — UI + API + Postgres + Redis broker. # -# KB module 2 (docs/kb/modules/02-paperless-service.md). Heavy OCR runs on a +# KB module 2 (kb/phases/kb-m2-paperless.md). Heavy OCR runs on a # SEPARATE worker on SOLARIA (services/paperless-worker/, module 3) that shares # this stack's Redis broker, Postgres and document storage (NFS export from # PIHA). The built-in celery worker here stays at concurrency 1 as the slow @@ -54,7 +54,7 @@ services: - PAPERLESS_WEBSERVER_WORKERS=1 # Files on disk (media/, data/, consume/) are owned by this numeric UID. # MUST equal USERMAP_UID/GID of paperless-worker@SOLARIA — the NFS export - # carries numeric IDs, not names. See services/paperless-worker/README.md. + # carries numeric IDs, not names. See kb/services/paperless-worker.md. - USERMAP_UID=1000 - USERMAP_GID=1000 # OIDC via Forgejo (django-allauth openid_connect). Provider JSON with diff --git a/services/stability-agent/service.yaml b/services/stability-agent/service.yaml index 8e2dc66..6aaa5fb 100644 --- a/services/stability-agent/service.yaml +++ b/services/stability-agent/service.yaml @@ -2,7 +2,7 @@ service: name: stability-agent # Deployed per-host: runs on vps, piha and solaria (and chelsty-infra once the # site is revived). Was owner_node: chelsty — not a real node name; fixed during - # the 2026-07 truth cleanup (recon B7/F20.5), closing the docs/backlog.md + # the 2026-07 truth cleanup (recon B7/F20.5), closing the kb/phases/backlog.md # "stability-agent / node_exporter owner_node single" item for stability-agent. owner_node: per-host exposure: private