"""Dev seed data: departments, users and the fixture corpus as documents with a plausible past. Never runs in production. Seeded documents are not inserted as finished rows — they are given the history they would have if someone had written them in the app: an empty draft, one edit per section as the author works down the page, the publish, and the questions colleagues asked afterwards. Without that, every history view, diff and "recently changed" list in the dev stack is empty or lies. """ import asyncio import re import sys from dataclasses import dataclass from datetime import UTC, datetime, timedelta from typing import TYPE_CHECKING from sqlalchemy import select from sqlalchemy.ext.asyncio import AsyncSession from app.auth.passwords import hash_password from app.config import get_settings from app.db import async_session_factory, engine from app.ingestion.handlers import INDEX_DOCUMENT from app.ingestion.queue import enqueue from app.models import ( Department, DocPermission, Document, DocumentEvent, DocumentEventAction, DocumentStatus, ReviewRequest, User, UserRole, ) if TYPE_CHECKING: # the corpus is a dev-only import, see _seed_corpus from tests.fixtures.loader import CorpusDoc DEV_PASSWORD = "pablan-dev" DEPARTMENTS = ["Engineering", "Sales", "Administration"] # The dev team, one per department, so the permission scenarios the e2e # suite relies on stay intact: an admin in Administration, a member in # Engineering, a member in Sales. USERS = [ ("florian@pablan.dev", "Florian", UserRole.admin, "Administration"), ("pablo@pablan.dev", "Pablo", UserRole.member, "Engineering"), ("max@pablan.dev", "Max", UserRole.member, "Sales"), ] # Which seeded user authors a department's corpus documents. AUTHORS_BY_DEPARTMENT = { "Engineering": "pablo@pablan.dev", "Sales": "max@pablan.dev", "Administration": "florian@pablan.dev", } # Deliberately authorless: it doubles as the demo for the "department" access # reason (a member sees it without owning it) and, being the knowledge of # someone who has since left, fits the offboarding theme. AUTHORLESS_SLUG = "wartungsplan-cnc-f350" # Documents whose life stops before the publish: unpublished work in progress, # one per author, so whoever logs in finds their own drafts waiting on the # home page. DRAFT_SLUGS = { "netzwerk-produktions-it", # Engineering "messevorbereitung", # Sales "it-onboarding-arbeitsplatz", # Administration } # Published once, then retired — so the archive is not an empty concept in the # dev stack. ARCHIVED_SLUGS = {"edi-rechnungen"} # How far back the corpus starts and how far apart the documents were written. # Deterministic rather than random: the "recently changed" order stays stable # across re-seeds, and the oldest documents are the ones that look oldest. CORPUS_STARTS_DAYS_AGO = 120 DAYS_BETWEEN_DOCUMENTS = 6 @dataclass(frozen=True) class SeededReview: """A "please check this" the author sent a colleague. Open ones are the interesting case: the document is published and readable and still carries an unanswered question, which is exactly what every surface — list, detail page, chat sources — has to mark. """ reviewer: str question: str answered: bool = False # What the reviewer corrected, as (current text, the text it replaced). # Applied in reverse to every version before the answer, so the history # holds a real diff and the answer is visibly a fix, not a rubber stamp. correction: tuple[str, str] | None = None REVIEWS = { # Published, public, unanswered: the case a reader most needs to see. "urlaubsantrag-prozess": SeededReview( reviewer="pablo@pablan.dev", question=( "Stimmt das für die Fertigung noch so, dass pro Schicht maximal " "zwei Personen gleichzeitig Urlaub haben dürfen?" ), ), # On a draft: being asked is what lets a colleague see it at all. "messevorbereitung": SeededReview( reviewer="florian@pablan.dev", question="Passt der Budgetrahmen so, bevor ich das veröffentliche?", ), # Answered — and the reviewer fixed the number before answering. "rabattrichtlinie": SeededReview( reviewer="florian@pablan.dev", question="Gilt für Ersatzteile weiterhin die 3-%-Grenze?", answered=True, correction=( "- Bis 5 %: eigenverantwortlich durch den Vertriebsmitarbeiter", "- Bis 3 %: eigenverantwortlich durch den Vertriebsmitarbeiter", ), ), } def _seed_order(doc: "CorpusDoc") -> tuple[int, str]: """The order the corpus was "written" in, oldest first. Not alphabetical: the knowledge of someone who has left is the oldest thing in the base, and work still in draft has to be the most recent. """ if doc.slug == AUTHORLESS_SLUG: return (0, doc.slug) if doc.slug in DRAFT_SLUGS: return (2, doc.slug) return (1, doc.slug) def _writing_steps(content_md: str) -> list[str]: """The document as it grew: the empty draft it starts as, then one state per section — the shape the writing editor produces, where a section is refined and saved before the next one is started.""" sections = re.split(r"(?m)^(?=## )", content_md) return [""] + ["".join(sections[: index + 1]) for index in range(len(sections))] async def _get_or_create_department(db: AsyncSession, name: str) -> Department: existing = ( await db.execute(select(Department).where(Department.name == name)) ).scalar_one_or_none() if existing is not None: return existing department = Department(name=name) db.add(department) await db.flush() return department async def seed() -> None: async with async_session_factory() as db: departments = { name: await _get_or_create_department(db, name) for name in DEPARTMENTS } created = 0 for email, name, role, department_name in USERS: existing = ( await db.execute(select(User).where(User.email == email)) ).scalar_one_or_none() if existing is not None: continue db.add( User( email=email, name=name, role=role, password_hash=hash_password(DEV_PASSWORD), department_id=departments[department_name].id, ) ) created += 1 documents_created = await _seed_corpus(db, departments) await db.commit() await engine.dispose() print(f"Seeded {len(DEPARTMENTS)} departments, {created} new users.") print(f"Dev logins (password: {DEV_PASSWORD!r}):") for email, _, role, department in USERS: print(f" {email} ({role}, {department})") print( f"Seeded {documents_created} new corpus documents with their history " f"({len(DRAFT_SLUGS)} drafts, {len(REVIEWS)} review requests); " "index jobs enqueued (processed once the backend runs)." ) async def _seed_corpus(db: AsyncSession, departments: dict[str, Department]) -> int: # Dev-only import: the corpus lives with the test fixtures on purpose — # seeds and tests draw from the same product asset. from tests.fixtures.loader import load_corpus users_by_email = {u.email: u for u in (await db.execute(select(User))).scalars()} created = 0 for index, doc in enumerate(sorted(load_corpus(), key=_seed_order)): existing = ( await db.execute( select(Document.id).where(Document.meta["slug"].astext == doc.slug) ) ).first() if existing is not None: continue document = await _seed_document( db, doc, department=departments[doc.department], users_by_email=users_by_email, started_at=datetime.now(UTC) - timedelta(days=CORPUS_STARTS_DAYS_AGO - index * DAYS_BETWEEN_DOCUMENTS), ) for grant in doc.grants: db.add( DocPermission( document_id=document.id, department_id=departments[grant].id, ) ) if document.status == DocumentStatus.published: await enqueue(db, INDEX_DOCUMENT, {"document_id": str(document.id)}) created += 1 return created async def _seed_document( db: AsyncSession, doc: "CorpusDoc", *, department: Department, users_by_email: dict[str, User], started_at: datetime, ) -> Document: """One corpus document plus the trail of everything that happened to it.""" author: User | None = users_by_email[AUTHORS_BY_DEPARTMENT[doc.department]] if doc.slug == AUTHORLESS_SLUG: author = None slug = doc.slug content = doc.content_md review = REVIEWS.get(slug) is_draft = slug in DRAFT_SLUGS document = Document( title=doc.title, status=DocumentStatus.draft if is_draft else DocumentStatus.published, visibility=doc.visibility, content_md=content, meta={"slug": slug}, author_id=author.id if author else None, department_id=department.id, reviews=[], ) db.add(document) await db.flush() at = started_at def happened( action: DocumentEventAction, actor: User | None, snapshot: str | None = None, *, after: timedelta = timedelta(), ) -> datetime: nonlocal at at += after db.add( DocumentEvent( document_id=document.id, actor_id=actor.id if actor else None, action=action, content_md=snapshot, title=document.title if snapshot is not None else None, visibility=document.visibility, meta=document.meta if snapshot is not None else None, created_at=at, updated_at=at, ) ) return at # The text as it stood before the reviewer's correction — everything up to # their answer holds the old wording. written = content if review and review.correction: current, previous = review.correction if current not in content: raise ValueError(f"correction text not found in {slug}: {current!r}") written = content.replace(current, previous) steps = _writing_steps(written) happened(DocumentEventAction.created, author, steps[0]) for step in steps[1:]: happened(DocumentEventAction.edited, author, step, after=timedelta(minutes=40)) if not is_draft: happened(DocumentEventAction.published, author, after=timedelta(days=1)) if review: reviewer = users_by_email[review.reviewer] asked_at = happened( DocumentEventAction.review_requested, author, after=timedelta(days=3) ) resolved_at = None if review.answered: if review.correction: happened( DocumentEventAction.edited, reviewer, content, after=timedelta(days=1), ) resolved_at = happened( DocumentEventAction.review_resolved, reviewer, after=timedelta(minutes=20), ) db.add( ReviewRequest( document_id=document.id, requester_id=author.id if author else None, reviewer_id=reviewer.id, question=review.question, resolved_at=resolved_at, resolved_by_id=reviewer.id if resolved_at else None, created_at=asked_at, updated_at=resolved_at or asked_at, ) ) if slug in ARCHIVED_SLUGS: document.status = DocumentStatus.archived happened(DocumentEventAction.archived, author, after=timedelta(days=30)) document.created_at = started_at document.updated_at = at return document def main() -> None: if get_settings().env == "production": sys.exit( "Refusing to seed: PABLAN_ENV=production. " "Seed data contains known dev credentials." ) asyncio.run(seed()) if __name__ == "__main__": main()