Self-hosted knowledge management for SMEs: a split-screen Markdown editor whose sections an LLM refines while you write, and RAG question answering over the documents that result. FastAPI + Postgres/pgvector on the back, SvelteKit on the front, everything OpenAI-compatible and self-hostable. Squashed into a single commit; the development history stays local. Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01CA43ZJda8Rbp2hKXNy8f6b
287 lines
9.9 KiB
Python
287 lines
9.9 KiB
Python
"""LLM endpoint configuration: bootstrapped from `.env` once, then owned by
|
|
the database, applied without a restart, and the api_key never leaves the
|
|
server (rule 12)."""
|
|
|
|
import logging
|
|
|
|
import pytest
|
|
from httpx import AsyncClient
|
|
from sqlalchemy import select
|
|
from sqlalchemy.ext.asyncio import AsyncSession
|
|
|
|
from app.llm import client as llm_client
|
|
from app.llm import overrides
|
|
from app.llm.overrides import bootstrap_llm_settings, env_defaults, load_config
|
|
from app.log import JsonFormatter
|
|
from app.models import LLMSetting, User
|
|
|
|
pytestmark = pytest.mark.usefixtures("fake_llm")
|
|
|
|
SECRET = "sk-super-secret-key-9876"
|
|
|
|
|
|
@pytest.fixture(autouse=True)
|
|
def _clean_config():
|
|
overrides.clear()
|
|
llm_client.rebuild_clients()
|
|
yield
|
|
overrides.clear()
|
|
llm_client.rebuild_clients()
|
|
|
|
|
|
async def _login_admin(client: AsyncClient) -> None:
|
|
response = await client.post(
|
|
"/api/auth/login", json={"email": "florian@test.dev", "password": "secret123"}
|
|
)
|
|
assert response.status_code == 200
|
|
|
|
|
|
async def test_bootstrap_copies_the_environment_once(db: AsyncSession) -> None:
|
|
written = await bootstrap_llm_settings(db)
|
|
assert written > 0
|
|
|
|
rows = {row.role: row for row in (await db.execute(select(LLMSetting))).scalars()}
|
|
for role in ("chat", "utility", "embedding"):
|
|
defaults = env_defaults(role)
|
|
assert rows[role].base_url == defaults.base_url
|
|
assert rows[role].base_url_from_env is True
|
|
assert rows[role].model_from_env is True
|
|
|
|
# A second start changes nothing — the rows are the admin's now.
|
|
assert await bootstrap_llm_settings(db) == 0
|
|
|
|
|
|
async def test_bootstrap_fills_a_field_an_upgrade_left_deferring_to_env(
|
|
db: AsyncSession,
|
|
) -> None:
|
|
"""The upgrade path: before this milestone a NULL column meant "inherit
|
|
from .env", so the migration flags those fields `*_from_env` and leaves
|
|
them empty. Startup has to fill them in — otherwise a row whose
|
|
overrides had been cleared comes out as an empty configuration and
|
|
takes the endpoint down."""
|
|
db.add(
|
|
LLMSetting(
|
|
role="chat",
|
|
base_url=None,
|
|
model="hand-picked",
|
|
api_key=None,
|
|
base_url_from_env=True,
|
|
model_from_env=False,
|
|
api_key_from_env=True,
|
|
)
|
|
)
|
|
await db.commit()
|
|
|
|
await bootstrap_llm_settings(db)
|
|
|
|
row = (
|
|
await db.execute(select(LLMSetting).where(LLMSetting.role == "chat"))
|
|
).scalar_one()
|
|
assert row.base_url == env_defaults("chat").base_url
|
|
# The admin's own value is never overwritten.
|
|
assert row.model == "hand-picked"
|
|
assert row.model_from_env is False
|
|
|
|
|
|
async def test_startup_logging_survives_the_reserved_name_trap(
|
|
db: AsyncSession, caplog: pytest.LogCaptureFixture
|
|
) -> None:
|
|
"""`extra={"created": ...}` raises KeyError inside logging, because
|
|
LogRecord already owns that attribute — and the process dies on startup.
|
|
|
|
This slipped through once: the log line only builds when the logger is
|
|
enabled for INFO, and the suite otherwise runs above that level, so
|
|
every existing test passed while the app refused to boot.
|
|
"""
|
|
with caplog.at_level(logging.INFO, logger="pablan.llm"):
|
|
await bootstrap_llm_settings(db)
|
|
await load_config(db)
|
|
|
|
rendered = "\n".join(JsonFormatter().format(record) for record in caplog.records)
|
|
assert "llm_bootstrap" in rendered
|
|
|
|
|
|
async def test_the_database_wins_over_a_later_env_change(
|
|
client: AsyncClient, db: AsyncSession, seeded_admin: User
|
|
) -> None:
|
|
"""The point of bootstrap-then-DB: once a value is stored, the process
|
|
uses it even though `.env` still says something else."""
|
|
await _login_admin(client)
|
|
env_url = env_defaults("chat").base_url
|
|
|
|
saved = await client.put(
|
|
"/api/admin/llm/settings/chat", json={"base_url": "http://stored.invalid/v1"}
|
|
)
|
|
assert saved.status_code == 200
|
|
assert saved.json()["base_url"] == "http://stored.invalid/v1"
|
|
assert saved.json()["base_url_from_env"] is False
|
|
assert llm_client.role_config("chat")[0] == "http://stored.invalid/v1"
|
|
|
|
# Re-running bootstrap (i.e. a restart) must not undo it.
|
|
await bootstrap_llm_settings(db)
|
|
await load_config(db)
|
|
assert llm_client.role_config("chat")[0] == "http://stored.invalid/v1"
|
|
assert env_url != "http://stored.invalid/v1"
|
|
|
|
|
|
async def test_resetting_a_field_restores_the_env_value(
|
|
client: AsyncClient, db: AsyncSession, seeded_admin: User
|
|
) -> None:
|
|
await _login_admin(client)
|
|
env_model = env_defaults("chat").model
|
|
|
|
changed = await client.put(
|
|
"/api/admin/llm/settings/chat", json={"model": "gemma-9000"}
|
|
)
|
|
assert changed.json()["model"] == "gemma-9000"
|
|
assert changed.json()["model_from_env"] is False
|
|
|
|
reset = await client.put("/api/admin/llm/settings/chat", json={"reset_model": True})
|
|
assert reset.json()["model_from_env"] is True
|
|
assert reset.json()["model"] == (env_model or "")
|
|
assert llm_client.role_config("chat")[2] == env_model
|
|
|
|
|
|
async def test_provenance_is_tracked_per_field(
|
|
client: AsyncClient, db: AsyncSession, seeded_admin: User
|
|
) -> None:
|
|
"""Changing the model must not relabel the URL as hand-edited."""
|
|
await _login_admin(client)
|
|
body = (
|
|
await client.put("/api/admin/llm/settings/chat", json={"model": "gemma-9000"})
|
|
).json()
|
|
assert body["model_from_env"] is False
|
|
assert body["base_url_from_env"] is True
|
|
assert body["api_key_from_env"] is True
|
|
|
|
|
|
async def test_saving_rebuilds_the_client_so_no_restart_is_needed(
|
|
client: AsyncClient, db: AsyncSession, seeded_admin: User
|
|
) -> None:
|
|
await _login_admin(client)
|
|
before = llm_client._client_for("chat")
|
|
|
|
await client.put(
|
|
"/api/admin/llm/settings/chat",
|
|
json={"base_url": "http://elsewhere.invalid/v1"},
|
|
)
|
|
|
|
after = llm_client._client_for("chat")
|
|
assert after is not before, "cached client kept the old base_url"
|
|
assert str(after.base_url).startswith("http://elsewhere.invalid")
|
|
|
|
|
|
async def test_the_api_key_is_write_only(
|
|
client: AsyncClient, db: AsyncSession, seeded_admin: User
|
|
) -> None:
|
|
await _login_admin(client)
|
|
saved = await client.put("/api/admin/llm/settings/chat", json={"api_key": SECRET})
|
|
assert saved.status_code == 200
|
|
|
|
# It is stored...
|
|
row = (
|
|
await db.execute(select(LLMSetting).where(LLMSetting.role == "chat"))
|
|
).scalar_one()
|
|
assert row.api_key == SECRET
|
|
# ...and used...
|
|
assert llm_client.role_config("chat")[1] == SECRET
|
|
|
|
# ...but no response body ever contains it.
|
|
assert SECRET not in saved.text
|
|
assert saved.json()["api_key_set"] is True
|
|
assert saved.json()["api_key_from_env"] is False
|
|
|
|
listing = await client.get("/api/admin/llm/settings")
|
|
assert SECRET not in listing.text
|
|
assert "api_key" not in listing.json()[0]
|
|
|
|
|
|
async def test_the_api_key_never_reaches_a_log_line(
|
|
client: AsyncClient,
|
|
db: AsyncSession,
|
|
seeded_admin: User,
|
|
caplog: pytest.LogCaptureFixture,
|
|
) -> None:
|
|
await _login_admin(client)
|
|
with caplog.at_level(logging.DEBUG):
|
|
await client.put("/api/admin/llm/settings/chat", json={"api_key": SECRET})
|
|
await client.post(
|
|
"/api/admin/llm/test",
|
|
json={"role": "chat", "api_key": SECRET, "base_url": "http://x.invalid/v1"},
|
|
)
|
|
await client.post(
|
|
"/api/admin/llm/models/chat",
|
|
json={"api_key": SECRET, "base_url": "http://x.invalid/v1"},
|
|
)
|
|
|
|
formatter = JsonFormatter()
|
|
rendered = "\n".join(formatter.format(record) for record in caplog.records)
|
|
assert SECRET not in rendered
|
|
|
|
|
|
async def test_testing_a_candidate_does_not_persist_it(
|
|
client: AsyncClient, db: AsyncSession, seeded_admin: User
|
|
) -> None:
|
|
"""The test button must not change the running configuration."""
|
|
await _login_admin(client)
|
|
before = llm_client.role_config("chat")
|
|
|
|
response = await client.post(
|
|
"/api/admin/llm/test",
|
|
json={"role": "chat", "base_url": "http://candidate.invalid/v1"},
|
|
)
|
|
assert response.status_code == 200
|
|
assert [role["role"] for role in response.json()["roles"]] == ["chat"]
|
|
|
|
assert llm_client.role_config("chat") == before
|
|
assert (await db.execute(select(LLMSetting))).scalars().all() == []
|
|
|
|
|
|
async def test_available_models_come_from_the_endpoint(
|
|
client: AsyncClient, db: AsyncSession, seeded_admin: User
|
|
) -> None:
|
|
await _login_admin(client)
|
|
response = await client.post("/api/admin/llm/models/chat", json={})
|
|
assert response.status_code == 200
|
|
body = response.json()
|
|
assert body["supported"] is True
|
|
assert body["models"] == ["bge-m3", "gemma-3-27b"]
|
|
|
|
|
|
async def test_an_endpoint_without_the_route_degrades_quietly(
|
|
client: AsyncClient, db: AsyncSession, seeded_admin: User, fake_llm
|
|
) -> None:
|
|
"""Plenty of OpenAI-compatible servers do not implement /v1/models. That
|
|
is a missing convenience, not an error worth showing."""
|
|
fake_llm.served_models = None
|
|
await _login_admin(client)
|
|
|
|
body = (await client.post("/api/admin/llm/models/chat", json={})).json()
|
|
assert body["supported"] is False
|
|
assert body["models"] == []
|
|
assert body["error"] is None
|
|
|
|
|
|
async def test_listing_models_does_not_persist_the_candidate(
|
|
client: AsyncClient, db: AsyncSession, seeded_admin: User
|
|
) -> None:
|
|
await _login_admin(client)
|
|
await client.post(
|
|
"/api/admin/llm/models/chat",
|
|
json={"base_url": "http://candidate.invalid/v1", "api_key": SECRET},
|
|
)
|
|
assert (await db.execute(select(LLMSetting))).scalars().all() == []
|
|
|
|
|
|
async def test_settings_require_an_admin(
|
|
client: AsyncClient, seeded_user: User
|
|
) -> None:
|
|
await client.post(
|
|
"/api/auth/login", json={"email": "pablo@test.dev", "password": "secret123"}
|
|
)
|
|
assert (await client.get("/api/admin/llm/settings")).status_code == 403
|
|
assert (
|
|
await client.put("/api/admin/llm/settings/chat", json={"model": "x"})
|
|
).status_code == 403
|
|
assert (await client.post("/api/admin/llm/models/chat", json={})).status_code == 403
|