"""kb-query -- module 5 phase 4 (kb/phases/kb-m5-faza4.md §4): first user-facing HTTP entry point to the KB. Wraps `kb_retrieval` retrieval (module 5 phase 3, gated PASS -- docs/sessions/2026-07-21.md) in FastAPI. This is a search API, not chat: no answer synthesis, no LLM call over the results (that is phase 5, out of scope here). Embed path (Krok 2, plan §2 decision 2 / §5 -- the active fallback): queries embed through `app.embed_router.EmbedRouter` -- SOLARIA's GPU Ollama (EMBED_PRIMARY_URL) when its cached ~30 s health-check says up, the local CPU `ollama-piha` (EMBED_FALLBACK_URL) when SOLARIA sleeps. A mid-query primary failure fails over the SAME request instead of surfacing an error. Only when BOTH backends are unavailable does `/search` return 503; a backend serving the wrong model (vs the bge-m3 space `document_chunk` is indexed in) is a loud 500, never a silent cross-space distance computation. `GET /` (Krok 4, plan §7) serves the search UI from this same FastAPI process -- one image, one container (plan §2 decision 4): a Jinja2 shell + a static vanilla-JS file, no node build step. `/` and `/static/*` need no DB/Ollama, so they stay reachable even while `/search` is 503ing. `mode=hybrid` (faza mailowa, plan Krok 3, kb/phases/kb-m5-faza-mailowa.md §6) is the DEFAULT since 2026-08-06 (DoD (d) of that phase): the quality gate (§8) re-ran on the full mail corpus (187k embedded mail chunks in HNSW) and PASSed -- no paperless hit degraded, mail hit@3 5/5 -- so mail bodies are now visible without asking for them. Cost: one extra SQL query per search (the summaryless-source chunk scan). `mode=cascade` (documents only, the old default) and `mode=flat` (debug, no summary pre-filter) stay available explicitly and behave exactly as before. """ from __future__ import annotations import logging import os import pathlib from contextlib import asynccontextmanager import aiohttp from fastapi import FastAPI, HTTPException, Query, Request from fastapi.responses import HTMLResponse from fastapi.staticfiles import StaticFiles from fastapi.templating import Jinja2Templates from app.db import create_pool from app.embed_router import EmbedBackendError, EmbedRouter, ModelMismatchError from app.search import run_search from app.startup import validate_embed_model # uvicorn only configures its own loggers; without this the router's # backend=solaria/piha lines (kb-query.embed) would never reach docker logs. logging.basicConfig(level=logging.INFO, format="%(asctime)s %(levelname)s %(name)s %(message)s") BASE_DIR = pathlib.Path(__file__).resolve().parent KB_DSN = os.environ.get("KB_DSN") # EMBED_PRIMARY_URL replaced OLLAMA_URL in Krok 2 (aktywny fallback) -- one naming scheme # for both legs. Same default upstream as before: Ollama @ SOLARIA over Tailscale. EMBED_PRIMARY_URL = os.environ.get("EMBED_PRIMARY_URL", "http://solaria:11434") # Unset/empty -> no fallback leg: /search 503s when SOLARIA is down (pre-Krok-2 behaviour). EMBED_FALLBACK_URL = os.environ.get("EMBED_FALLBACK_URL") or None EMBED_PRIMARY_NAME = os.environ.get("EMBED_PRIMARY_NAME", "solaria") EMBED_FALLBACK_NAME = os.environ.get("EMBED_FALLBACK_NAME", "piha") EMBED_MODEL = os.environ.get("EMBED_MODEL", "bge-m3") SUMMARY_MODEL = os.environ.get("SUMMARY_MODEL", "claude-haiku-4-5") # State-machine parameters (plan §2 D2 table; timeouts rationale in app/embed_router.py). EMBED_HEALTH_TTL_S = float(os.environ.get("EMBED_HEALTH_TTL_S", "30")) EMBED_HEALTH_TIMEOUT_S = float(os.environ.get("EMBED_HEALTH_TIMEOUT_S", "1.5")) EMBED_PRIMARY_TIMEOUT_S = float(os.environ.get("EMBED_PRIMARY_TIMEOUT_S", "3")) @asynccontextmanager async def lifespan(app: FastAPI): if not KB_DSN: raise RuntimeError("KB_DSN is required (see env.example)") pool = await create_pool(KB_DSN) async with pool.acquire() as conn: # Hard invariant (plan §2 decision 2): refuse to start rather than silently serve # queries against a mismatched embedding space. The per-backend half of the same # invariant (does the Ollama actually serve EMBED_MODEL?) lives in EmbedRouter and # runs lazily at each backend's first use -- a sleeping SOLARIA must not block boot. await validate_embed_model(conn, EMBED_MODEL) app.state.pool = pool app.state.http = aiohttp.ClientSession() app.state.embed_router = EmbedRouter( EMBED_PRIMARY_URL, EMBED_FALLBACK_URL, embed_model=EMBED_MODEL, primary_name=EMBED_PRIMARY_NAME, fallback_name=EMBED_FALLBACK_NAME, health_ttl_s=EMBED_HEALTH_TTL_S, health_timeout_s=EMBED_HEALTH_TIMEOUT_S, primary_embed_timeout_s=EMBED_PRIMARY_TIMEOUT_S, ) try: yield finally: await app.state.http.close() await pool.close() app = FastAPI(lifespan=lifespan) app.mount("/static", StaticFiles(directory=BASE_DIR / "static"), name="static") templates = Jinja2Templates(directory=BASE_DIR / "templates") @app.get("/", response_class=HTMLResponse) async def index(request: Request): return templates.TemplateResponse(request, "index.html") @app.get("/healthz") async def healthz() -> dict: # sol_status goes through the router's 30 s cache -- /healthz and /search routing # deliberately share one world view (and monitoring no longer re-probes a sleeping # SOLARIA on every scrape). fallback_status is a live cheap /api/tags probe. router = app.state.embed_router return { "status": "ok", "sol_status": await router.primary_status(app.state.http), "fallback_status": await router.fallback_status(app.state.http), } @app.get("/search") async def search( q: str = Query(..., min_length=1), mode: str = Query("hybrid", pattern="^(cascade|flat|hybrid)$"), ) -> dict: try: async with app.state.pool.acquire() as conn: return await run_search( conn, app.state.http, app.state.embed_router, q, mode, SUMMARY_MODEL ) except ModelMismatchError as exc: # Config/ops error (backend up but wrong model set) -- 500, deliberately loud. raise HTTPException(status_code=500, detail=str(exc)) from exc except (EmbedBackendError, aiohttp.ClientError) as exc: raise HTTPException(status_code=503, detail=f"embed backend unavailable: {exc}") from exc