Add rulebook RAG pipeline and LLM-driven game setup wizard

RAG: switch Postgres to pgvector, chunk and embed the three D&D
rulebooks locally via sentence-transformers, and retrieve relevant
excerpts per DM turn (query = latest player message) to ground the
system prompt. Retrieval runs off the event loop and is capped by a
relevance threshold and a max character budget so it can't blow up
context size or cost.

Game setup wizard: creating a game now opens a short chat where the
DM asks about genre, length, and the player's experience level, then
proposes a name and description via a tool call. The player can edit
both before creating the game. Stateless endpoint — the frontend
carries the conversation, no DB needed since the game doesn't exist
yet.

Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
This commit is contained in:
Thorsten
2026-08-31 20:03:05 +02:00
parent f37dc9fa76
commit 419f5e3a89
26 changed files with 82471 additions and 51 deletions
View File
+33
View File
@@ -0,0 +1,33 @@
import re
# Rough word-count targets standing in for the 300-800 token / 10-20% overlap guideline —
# English prose runs ~0.75 words per token, so ~380 words ≈ 500 tokens.
WORDS_PER_CHUNK = 380
OVERLAP_WORDS = 60
_WHITESPACE_RE = re.compile(r"[ \t]+")
_BLANK_LINES_RE = re.compile(r"\n{3,}")
def _normalize(text: str) -> str:
text = _WHITESPACE_RE.sub(" ", text)
text = _BLANK_LINES_RE.sub("\n\n", text)
return text.strip()
def chunk_text(
text: str, words_per_chunk: int = WORDS_PER_CHUNK, overlap_words: int = OVERLAP_WORDS
) -> list[str]:
words = _normalize(text).split()
if not words:
return []
step = words_per_chunk - overlap_words
chunks = []
start = 0
while start < len(words):
chunks.append(" ".join(words[start : start + words_per_chunk]))
if start + words_per_chunk >= len(words):
break
start += step
return chunks
+22
View File
@@ -0,0 +1,22 @@
from functools import lru_cache
from sentence_transformers import SentenceTransformer
# Multilingual so German player messages retrieve relevant chunks from the
# English-language rulebooks. 384-dim output, matching RulebookChunk.embedding.
MODEL_NAME = "paraphrase-multilingual-MiniLM-L12-v2"
@lru_cache
def get_embedding_model() -> SentenceTransformer:
return SentenceTransformer(MODEL_NAME)
def embed_texts(texts: list[str]) -> list[list[float]]:
model = get_embedding_model()
embeddings = model.encode(texts, normalize_embeddings=True, show_progress_bar=False)
return embeddings.tolist()
def embed_query(text: str) -> list[float]:
return embed_texts([text])[0]
+51
View File
@@ -0,0 +1,51 @@
import asyncio
from sqlalchemy import select
from sqlalchemy.ext.asyncio import AsyncSession
from app.models.rulebook_chunk import RulebookChunk
from app.rag.embeddings import embed_query
DEFAULT_TOP_K = 5
# Cosine distance cutoff (0 = identical, 2 = opposite) — drop weak matches rather than
# stuffing the prompt with irrelevant rulebook text when nothing actually fits the query.
# Calibrated against test queries: genuinely relevant chunks scored 0.30-0.45.
MAX_DISTANCE = 0.6
# Hard cap on injected rulebook text so one turn's context/cost doesn't balloon even when
# several chunks all clear the relevance threshold.
MAX_BLOCK_CHARS = 4000
async def retrieve_relevant_chunks(
session: AsyncSession, query: str, top_k: int = DEFAULT_TOP_K
) -> list[tuple[RulebookChunk, float]]:
# embed_query is a synchronous, CPU-bound sentence-transformers call — run it off the
# event loop so it doesn't stall other concurrent requests/WebSocket connections.
query_embedding = await asyncio.to_thread(embed_query, query)
distance = RulebookChunk.embedding.cosine_distance(query_embedding)
rows = (
await session.execute(
select(RulebookChunk, distance.label("distance")).order_by(distance).limit(top_k)
)
).all()
return [(row[0], row[1]) for row in rows]
async def build_rag_block(session: AsyncSession, query: str, top_k: int = DEFAULT_TOP_K) -> str:
results = await retrieve_relevant_chunks(session, query, top_k=top_k)
relevant = [chunk for chunk, distance in results if distance <= MAX_DISTANCE]
if not relevant:
return ""
parts = [
"Relevante Auszüge aus den Regelwerken — richte dich danach und nenne bei Bedarf die Quelle:"
]
used_chars = 0
for chunk in relevant:
entry = f"\n[{chunk.source_document}]\n{chunk.content}"
if used_chars + len(entry) > MAX_BLOCK_CHARS and used_chars > 0:
break
parts.append(entry)
used_chars += len(entry)
return "\n".join(parts)