Self-hosted knowledge management for SMEs: a split-screen Markdown editor whose sections an LLM refines while you write, and RAG question answering over the documents that result. FastAPI + Postgres/pgvector on the back, SvelteKit on the front, everything OpenAI-compatible and self-hostable. Squashed into a single commit; the development history stays local. Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01CA43ZJda8Rbp2hKXNy8f6b
266 lines
8.3 KiB
Python
266 lines
8.3 KiB
Python
"""Model endpoints: test them, configure them, ask what they serve.
|
|
|
|
Configuration is bootstrapped from `.env` at first start and lives in the DB
|
|
afterwards, so this is where an admin changes an endpoint without a restart.
|
|
Every call runs server-side: the API key must never reach the browser, and the
|
|
browser must never reach the model endpoint.
|
|
"""
|
|
|
|
import asyncio
|
|
import time
|
|
from typing import Annotated
|
|
|
|
from fastapi import Depends
|
|
from pydantic import BaseModel, Field
|
|
from sqlalchemy import select
|
|
from sqlalchemy.ext.asyncio import AsyncSession
|
|
|
|
from app.api.admin.routing import admin_router
|
|
from app.db import get_db
|
|
from app.llm.client import (
|
|
Role,
|
|
chat_stream,
|
|
embed,
|
|
list_models,
|
|
probe,
|
|
rebuild_clients,
|
|
role_config,
|
|
)
|
|
from app.llm.errors import LLMError
|
|
from app.llm.overrides import env_defaults, load_config
|
|
from app.models import LLMSetting
|
|
|
|
router = admin_router()
|
|
|
|
ROLES: tuple[Role, ...] = ("chat", "utility", "embedding")
|
|
|
|
|
|
class LLMSettingUpdate(BaseModel):
|
|
# None means "leave as is"; the reset_* flags restore the value the
|
|
# environment currently states (same sentinel pattern as
|
|
# UserUpdate.clear_department).
|
|
base_url: str | None = Field(default=None, max_length=500)
|
|
model: str | None = Field(default=None, max_length=200)
|
|
api_key: str | None = Field(default=None, max_length=500)
|
|
reset_base_url: bool | None = None
|
|
reset_model: bool | None = None
|
|
reset_api_key: bool | None = None
|
|
|
|
|
|
class LLMRoleStatus(BaseModel):
|
|
role: Role
|
|
ok: bool
|
|
base_url: str
|
|
model: str
|
|
latency_ms: int | None
|
|
# Why it failed: `code` is phrased by the frontend, `error` is the
|
|
# sanitized technical detail (exception class + role) for the admin.
|
|
code: str | None
|
|
error: str | None
|
|
|
|
|
|
class LLMTestResponse(BaseModel):
|
|
roles: list[LLMRoleStatus]
|
|
|
|
|
|
class LLMTestRequest(BaseModel):
|
|
"""Optional candidate config: test an endpoint BEFORE saving it."""
|
|
|
|
role: Role | None = None
|
|
base_url: str | None = None
|
|
model: str | None = None
|
|
api_key: str | None = None
|
|
|
|
|
|
async def _ping_role(
|
|
role: Role, candidate: LLMSettingUpdate | None = None
|
|
) -> LLMRoleStatus:
|
|
"""Ping a role — either its effective config, or a candidate an admin is
|
|
about to save."""
|
|
base_url, _, model = role_config(role)
|
|
if candidate is not None:
|
|
base_url = candidate.base_url or base_url
|
|
model = candidate.model or model
|
|
|
|
started = time.monotonic()
|
|
try:
|
|
if candidate is None:
|
|
if role == "embedding":
|
|
await embed(["ping"])
|
|
else:
|
|
# Drain the (max_tokens=1) stream so the call records as
|
|
# "ok", not "aborted".
|
|
async for _ in chat_stream(
|
|
[{"role": "user", "content": "ping"}], role=role, max_tokens=1
|
|
):
|
|
pass
|
|
else:
|
|
await probe(
|
|
role,
|
|
base_url=candidate.base_url,
|
|
api_key=candidate.api_key,
|
|
model=candidate.model,
|
|
)
|
|
return LLMRoleStatus(
|
|
role=role,
|
|
ok=True,
|
|
base_url=base_url,
|
|
model=model,
|
|
latency_ms=round((time.monotonic() - started) * 1000),
|
|
code=None,
|
|
error=None,
|
|
)
|
|
except LLMError as exc:
|
|
return LLMRoleStatus(
|
|
role=role,
|
|
ok=False,
|
|
base_url=base_url,
|
|
model=model,
|
|
latency_ms=None,
|
|
code=exc.code,
|
|
error=str(exc),
|
|
)
|
|
|
|
|
|
@router.post("/llm/test")
|
|
async def llm_test(body: LLMTestRequest | None = None) -> LLMTestResponse:
|
|
"""First-line support tool: pings all three roles, or one candidate
|
|
configuration without persisting anything."""
|
|
if body is not None and body.role is not None:
|
|
candidate = LLMSettingUpdate(
|
|
base_url=body.base_url, model=body.model, api_key=body.api_key
|
|
)
|
|
return LLMTestResponse(roles=[await _ping_role(body.role, candidate)])
|
|
results = await asyncio.gather(*(_ping_role(role) for role in ROLES))
|
|
return LLMTestResponse(roles=list(results))
|
|
|
|
|
|
class LLMSettingOut(BaseModel):
|
|
"""Stored config for one role.
|
|
|
|
The api_key is NEVER returned — only whether one is set, and where each
|
|
field's value came from. `*_from_env` drives the per-field
|
|
"taken from .env" / "changed here" label and the reset action; it is
|
|
provenance, not a fallback.
|
|
"""
|
|
|
|
role: Role
|
|
base_url: str
|
|
model: str
|
|
base_url_from_env: bool
|
|
model_from_env: bool
|
|
api_key_set: bool
|
|
api_key_from_env: bool
|
|
|
|
|
|
async def _row_for(db: AsyncSession, role: Role) -> LLMSetting:
|
|
row = (
|
|
await db.execute(select(LLMSetting).where(LLMSetting.role == role))
|
|
).scalar_one_or_none()
|
|
if row is None:
|
|
# Only reachable if bootstrap never ran (a test, or a role added
|
|
# after install). Seed it from the environment, same as bootstrap.
|
|
defaults = env_defaults(role)
|
|
row = LLMSetting(
|
|
role=role,
|
|
base_url=defaults.base_url or None,
|
|
model=defaults.model or None,
|
|
api_key=defaults.api_key or None,
|
|
)
|
|
db.add(row)
|
|
await db.flush()
|
|
return row
|
|
|
|
|
|
def _setting_out(row: LLMSetting) -> LLMSettingOut:
|
|
return LLMSettingOut(
|
|
role=row.role, # type: ignore[arg-type]
|
|
base_url=row.base_url or "",
|
|
model=row.model or "",
|
|
base_url_from_env=row.base_url_from_env,
|
|
model_from_env=row.model_from_env,
|
|
api_key_set=bool(row.api_key),
|
|
api_key_from_env=row.api_key_from_env,
|
|
)
|
|
|
|
|
|
@router.get("/llm/settings")
|
|
async def llm_settings(
|
|
db: Annotated[AsyncSession, Depends(get_db)],
|
|
) -> list[LLMSettingOut]:
|
|
return [_setting_out(await _row_for(db, role)) for role in ROLES]
|
|
|
|
|
|
@router.put("/llm/settings/{role}")
|
|
async def update_llm_setting(
|
|
role: Role,
|
|
body: LLMSettingUpdate,
|
|
db: Annotated[AsyncSession, Depends(get_db)],
|
|
) -> LLMSettingOut:
|
|
"""Change a role's stored config and apply it without a restart.
|
|
|
|
Writing a field marks it as changed here; resetting it writes back what
|
|
`.env` currently says and marks it as coming from the environment
|
|
again.
|
|
"""
|
|
row = await _row_for(db, role)
|
|
defaults = env_defaults(role)
|
|
|
|
if body.reset_base_url:
|
|
row.base_url, row.base_url_from_env = defaults.base_url or None, True
|
|
elif body.base_url is not None:
|
|
row.base_url, row.base_url_from_env = body.base_url or None, False
|
|
if body.reset_model:
|
|
row.model, row.model_from_env = defaults.model or None, True
|
|
elif body.model is not None:
|
|
row.model, row.model_from_env = body.model or None, False
|
|
if body.reset_api_key:
|
|
row.api_key, row.api_key_from_env = defaults.api_key or None, True
|
|
elif body.api_key:
|
|
row.api_key, row.api_key_from_env = body.api_key, False
|
|
|
|
await db.commit()
|
|
await load_config(db)
|
|
# The cached OpenAI clients hold the old base_url and key.
|
|
rebuild_clients()
|
|
return _setting_out(row)
|
|
|
|
|
|
class LLMModelsRequest(BaseModel):
|
|
"""Optional candidate endpoint, so an admin can list the models of a
|
|
URL they have typed but not saved."""
|
|
|
|
base_url: str | None = Field(default=None, max_length=500)
|
|
api_key: str | None = Field(default=None, max_length=500)
|
|
|
|
|
|
class LLMModelsResponse(BaseModel):
|
|
models: list[str]
|
|
# False when the endpoint does not implement GET /v1/models — the UI
|
|
# keeps its free-text field instead of showing an error.
|
|
supported: bool
|
|
error: str | None = None
|
|
|
|
|
|
@router.post("/llm/models/{role}")
|
|
async def llm_models(
|
|
role: Role, body: LLMModelsRequest | None = None
|
|
) -> LLMModelsResponse:
|
|
"""List what the endpoint serves, so the model field can be a dropdown."""
|
|
try:
|
|
models = await list_models(
|
|
role,
|
|
base_url=(body.base_url if body else None) or None,
|
|
api_key=(body.api_key if body else None) or None,
|
|
)
|
|
except LLMError as exc:
|
|
# A 404 means "this server has no /v1/models", which is common
|
|
# enough to be a normal outcome rather than a failure to report.
|
|
supported = exc.status_code != 404
|
|
return LLMModelsResponse(
|
|
models=[],
|
|
supported=supported,
|
|
error=exc.cause_type if supported else None,
|
|
)
|
|
return LLMModelsResponse(models=models, supported=True)
|