Pablan, as it stands

Self-hosted knowledge management for SMEs: a split-screen Markdown editor
whose sections an LLM refines while you write, and RAG question answering
over the documents that result. FastAPI + Postgres/pgvector on the back,
SvelteKit on the front, everything OpenAI-compatible and self-hostable.

Squashed into a single commit; the development history stays local.

Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01CA43ZJda8Rbp2hKXNy8f6b
This commit is contained in:
ProfessorNova
2026-09-04 09:21:37 +02:00
co-authored by Claude Opus 5
parent 68d3a43191
commit 784b76baf7
346 changed files with 43430 additions and 0 deletions
+320
View File
@@ -0,0 +1,320 @@
"""The shared similarity mechanic: one permission-filtered vector search,
two calibrated thresholds.
Uses deterministic fake embeddings (fake_embed): identical text → distance
0, unrelated text → near-orthogonal. That is enough to prove the SQL
plumbing, the permission boundary and the threshold behaviour; whether the
thresholds are set at useful VALUES is measured against the real model in
tests/evals/test_duplicate_eval.py.
"""
import uuid
import pytest
from sqlalchemy import select
from sqlalchemy.ext.asyncio import AsyncSession
from app.auth.passwords import hash_password
from app.metrics import metrics
from app.models import (
Chunk,
Department,
DocPermission,
Document,
DocumentStatus,
DocumentVisibility,
User,
UserRole,
)
from app.rag.indexing import embedding_text, reindex_document
from app.rag.similarity import (
CAPTURE_CONTEXT_MAX_DISTANCE,
DUPLICATE_MAX_DISTANCE,
similar_chunks,
similar_documents,
)
pytestmark = pytest.mark.usefixtures("fake_embed")
class Setup:
pablo: User # Engineering
max: User # Sales
pub: Document
dept_sales: Document
restricted_sales: Document
granted_eng: Document
draft: Document
async def _user(db: AsyncSession, email: str, department_id) -> User:
user = User(
email=email,
name=email.split("@")[0],
role=UserRole.member,
password_hash=hash_password("secret123"),
department_id=department_id,
)
db.add(user)
await db.flush()
return user
async def _doc(
db: AsyncSession,
*,
title: str,
content: str,
author: User,
department_id=None,
visibility: DocumentVisibility,
status: DocumentStatus = DocumentStatus.published,
) -> Document:
document = Document(
title=title,
status=status,
visibility=visibility,
content_md=content,
author_id=author.id,
department_id=department_id,
)
db.add(document)
await db.flush()
return document
@pytest.fixture
async def setup(db: AsyncSession) -> Setup:
s = Setup()
engineering = Department(name="Engineering")
sales = Department(name="Sales")
db.add_all([engineering, sales])
await db.flush()
s.pablo = await _user(db, "pablo@test.dev", engineering.id)
s.max = await _user(db, "max@test.dev", sales.id)
s.pub = await _doc(
db,
title="Wartung der Kaffeemaschine",
content="## Pflege\n\nDie Kaffeemaschine wird jeden Freitag entkalkt.",
author=s.pablo,
department_id=engineering.id,
visibility=DocumentVisibility.public,
)
s.dept_sales = await _doc(
db,
title="Angebotsfristen",
content="## Fristen\n\nAngebote gelten dreißig Tage.",
author=s.max,
department_id=sales.id,
visibility=DocumentVisibility.department,
)
s.restricted_sales = await _doc(
db,
title="Geheime Preisliste",
content="## Preise\n\nDer Rabattdeckel liegt bei zwölf Prozent.",
author=s.max,
department_id=sales.id,
visibility=DocumentVisibility.restricted,
)
s.granted_eng = await _doc(
db,
title="Ersatzteillager",
content="## Zugang\n\nDie Zugangskarte liegt im Tresorfach drei.",
author=s.max,
department_id=sales.id,
visibility=DocumentVisibility.restricted,
)
db.add(DocPermission(document_id=s.granted_eng.id, department_id=engineering.id))
s.draft = await _doc(
db,
title="Pausenregelung",
content="## Entwurf\n\nNeue Pausenregelung ab Oktober.",
author=s.pablo,
department_id=engineering.id,
visibility=DocumentVisibility.public,
status=DocumentStatus.draft,
)
for document in (
s.pub,
s.dept_sales,
s.restricted_sales,
s.granted_eng,
s.draft,
):
await reindex_document(db, document)
await db.commit()
return s
async def _chunk_text(db: AsyncSession, document: Document) -> str:
"""A chunk exactly as it was embedded — heading path and all, so a search
for it lands at distance 0 (see indexing.embedding_text)."""
row = (
await db.execute(
select(Chunk.content, Chunk.meta).where(Chunk.document_id == document.id)
)
).first()
return embedding_text(row.meta["heading_path"], row.content)
def _doc_ids(results) -> set[uuid.UUID]:
return {result.document_id for result in results}
async def test_finds_the_matching_chunk_with_a_usable_distance(
db: AsyncSession, setup: Setup
) -> None:
text = await _chunk_text(db, setup.pub)
results = await similar_chunks(
db, text, user=setup.pablo, max_distance=DUPLICATE_MAX_DISTANCE
)
assert results
top = results[0]
assert top.document_id == setup.pub.id
assert top.title == "Wartung der Kaffeemaschine"
# Unlike the hybrid path, a distance is always present — that is the
# whole reason this search exists.
assert top.distance < 0.001
async def test_unrelated_text_is_filtered_by_the_threshold(
db: AsyncSession, setup: Setup
) -> None:
loose = await similar_chunks(
db,
"Völlig anderes Thema ohne Bezug zu irgendetwas",
user=setup.pablo,
max_distance=CAPTURE_CONTEXT_MAX_DISTANCE,
)
assert loose == []
async def test_the_tight_threshold_rejects_what_the_loose_one_accepts(
db: AsyncSession, setup: Setup
) -> None:
"""One mechanic, two thresholds: the same call with a smaller limit is
strictly more selective."""
text = await _chunk_text(db, setup.pub)
near_miss = text + " Zusätzlich wird der Wasserfilter getauscht."
loose = await similar_chunks(db, near_miss, user=setup.pablo, max_distance=1.0)
assert loose, "the near miss should be retrievable at all"
distance = loose[0].distance
accepted = await similar_chunks(
db, near_miss, user=setup.pablo, max_distance=distance
)
rejected = await similar_chunks(
db, near_miss, user=setup.pablo, max_distance=distance / 2
)
assert accepted and not rejected
async def test_restricted_document_never_surfaces(
db: AsyncSession, setup: Setup
) -> None:
"""The permission filter is the same CTE search() uses: Pablo has no
grant for the Sales price list, so no threshold can reveal it."""
text = await _chunk_text(db, setup.restricted_sales)
results = await similar_chunks(db, text, user=setup.pablo, max_distance=1.0)
assert setup.restricted_sales.id not in _doc_ids(results)
# Max authored it, so he still finds it — the filter is about the user,
# not about the document being hidden from everyone.
mine = await similar_chunks(db, text, user=setup.max, max_distance=1.0)
assert setup.restricted_sales.id in _doc_ids(mine)
async def test_excluding_a_document_drops_its_own_chunks(
db: AsyncSession, setup: Setup
) -> None:
"""A document must never ground a suggestion on itself: excluding its id
removes its own chunks even when the query is its exact text."""
text = await _chunk_text(db, setup.pub)
included = await similar_chunks(db, text, user=setup.pablo, max_distance=1.0)
assert setup.pub.id in _doc_ids(included)
excluded = await similar_chunks(
db,
text,
user=setup.pablo,
max_distance=1.0,
exclude_document_id=setup.pub.id,
)
assert setup.pub.id not in _doc_ids(excluded)
async def test_department_grant_is_honoured(db: AsyncSession, setup: Setup) -> None:
text = await _chunk_text(db, setup.granted_eng)
results = await similar_chunks(db, text, user=setup.pablo, max_distance=1.0)
assert setup.granted_eng.id in _doc_ids(results)
async def test_unpublished_documents_are_never_similar(
db: AsyncSession, setup: Setup
) -> None:
"""Why duplicate detection cannot match the draft it just created: an
unpublished document is outside the searchable set by construction."""
text = await _chunk_text(db, setup.draft)
results = await similar_chunks(db, text, user=setup.pablo, max_distance=1.0)
assert setup.draft.id not in _doc_ids(results)
async def test_documents_are_grouped_by_their_closest_chunk(
db: AsyncSession, setup: Setup
) -> None:
document = await _doc(
db,
title="Mehrteilige Anleitung",
content=(
"## Erster Abschnitt\n\nHier steht der erste Teil der Anleitung.\n\n"
"## Zweiter Abschnitt\n\nHier steht der zweite Teil der Anleitung."
),
author=setup.pablo,
department_id=setup.pablo.department_id,
visibility=DocumentVisibility.public,
)
await reindex_document(db, document)
await db.commit()
chunks = (
await db.execute(
select(Chunk.content, Chunk.meta).where(Chunk.document_id == document.id)
)
).all()
assert len(chunks) > 1, "fixture needs a multi-chunk document"
query = embedding_text(chunks[1].meta["heading_path"], chunks[1].content)
results = await similar_documents(db, query, user=setup.pablo, max_distance=1.0)
mine = [r for r in results if r.document_id == document.id]
assert len(mine) == 1, "a document must appear once, not once per chunk"
assert mine[0].distance < 0.001, "grouping must keep the CLOSEST chunk's distance"
async def test_similar_documents_respects_top_k(db: AsyncSession, setup: Setup) -> None:
results = await similar_documents(
db, "Kaffeemaschine", user=setup.pablo, top_k=1, max_distance=1.0
)
assert len(results) <= 1
async def test_similarity_records_metrics(db: AsyncSession, setup: Setup) -> None:
await similar_chunks(db, "Kaffeemaschine", user=setup.pablo, max_distance=1.0)
snapshot = metrics.snapshot()
assert snapshot["counters"]["similarity_searches_total"][0]["value"] >= 1
assert "similarity_seconds" in snapshot["histograms"]
async def test_similarity_logs_no_content(
db: AsyncSession, setup: Setup, caplog: pytest.LogCaptureFixture
) -> None:
"""Rule 12: the searched text is user content and never reaches a log."""
secret = "GEHEIM-SUCHTEXT-42 Kaffeemaschine entkalken"
with caplog.at_level("INFO", logger="pablan.rag"):
await similar_chunks(db, secret, user=setup.pablo, max_distance=1.0)
rendered = "\n".join(
record.getMessage() + str(record.__dict__) for record in caplog.records
)
assert "GEHEIM-SUCHTEXT-42" not in rendered