Add rulebook RAG pipeline and LLM-driven game setup wizard

RAG: switch Postgres to pgvector, chunk and embed the three D&D
rulebooks locally via sentence-transformers, and retrieve relevant
excerpts per DM turn (query = latest player message) to ground the
system prompt. Retrieval runs off the event loop and is capped by a
relevance threshold and a max character budget so it can't blow up
context size or cost.

Game setup wizard: creating a game now opens a short chat where the
DM asks about genre, length, and the player's experience level, then
proposes a name and description via a tool call. The player can edit
both before creating the game. Stateless endpoint — the frontend
carries the conversation, no DB needed since the game doesn't exist
yet.

Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
This commit is contained in:
Thorsten
2026-08-31 20:03:05 +02:00
parent f37dc9fa76
commit 419f5e3a89
26 changed files with 82471 additions and 51 deletions
+3
View File
@@ -7,6 +7,9 @@ RUN apt-get update && apt-get install -y --no-install-recommends \
&& rm -rf /var/lib/apt/lists/*
COPY requirements.txt .
# CPU-only torch first — sentence-transformers would otherwise pull the much larger CUDA wheels,
# which are useless in this container and roughly double the image size.
RUN pip install --no-cache-dir torch==2.5.1 --index-url https://download.pytorch.org/whl/cpu
RUN pip install --no-cache-dir -r requirements.txt
COPY . .
@@ -0,0 +1,43 @@
"""add rulebook_chunks (pgvector RAG)
Revision ID: 7c3f9a2e1d44
Revises: 48846ed4fe0e
Create Date: 2026-08-31 19:00:00.000000
"""
from typing import Sequence, Union
from alembic import op
import sqlalchemy as sa
from sqlalchemy.dialects import postgresql
from pgvector.sqlalchemy import Vector
# revision identifiers, used by Alembic.
revision: str = '7c3f9a2e1d44'
down_revision: Union[str, None] = '48846ed4fe0e'
branch_labels: Union[str, Sequence[str], None] = None
depends_on: Union[str, Sequence[str], None] = None
def upgrade() -> None:
op.execute("CREATE EXTENSION IF NOT EXISTS vector")
op.create_table(
'rulebook_chunks',
sa.Column('id', sa.UUID(), nullable=False),
sa.Column('source_document', sa.String(length=200), nullable=False),
sa.Column('chunk_index', sa.Integer(), nullable=False),
sa.Column('content', sa.Text(), nullable=False),
sa.Column('embedding', Vector(384), nullable=False),
sa.Column('doc_metadata', postgresql.JSONB(astext_type=sa.Text()), nullable=False),
sa.PrimaryKeyConstraint('id'),
)
op.create_index(
op.f('ix_rulebook_chunks_source_document'), 'rulebook_chunks', ['source_document'], unique=False
)
def downgrade() -> None:
op.drop_index(op.f('ix_rulebook_chunks_source_document'), table_name='rulebook_chunks')
op.drop_table('rulebook_chunks')
op.execute("DROP EXTENSION IF EXISTS vector")
+11
View File
@@ -8,11 +8,13 @@ from sqlalchemy.ext.asyncio import AsyncSession
from app.auth.users import current_active_user
from app.db import get_async_session
from app.llm.game_setup import run_game_setup_turn
from app.models.character import Character
from app.models.game import Game, GameParticipant
from app.models.message import Message
from app.models.user import User
from app.schemas.game import GameCreate, GameJoin, GameRead
from app.schemas.game_setup import GameSetupChatRequest, GameSetupChatResponse
from app.schemas.message import MessageRead
from app.ws_tickets import issue_ticket
@@ -108,6 +110,15 @@ async def create_game(
return serialized[0]
@router.post("/setup-chat", response_model=GameSetupChatResponse)
async def game_setup_chat(
payload: GameSetupChatRequest,
user: User = Depends(current_active_user),
) -> GameSetupChatResponse:
result = await run_game_setup_turn(payload.messages)
return GameSetupChatResponse(**result)
@router.get("/{game_id}", response_model=GameRead)
async def get_game(
game_id: uuid.UUID,
+1 -1
View File
@@ -120,7 +120,7 @@ async def game_websocket(websocket: WebSocket, game_id: uuid.UUID, ticket: str)
)
try:
dm_message = await run_dm_turn(session, game_id)
dm_message = await run_dm_turn(session, game_id, latest_player_message=content)
except Exception: # noqa: BLE001
logger.exception("DM turn failed for game %s", game_id)
await websocket.send_json({"type": "error", "detail": "dm_turn_failed"})
+5
View File
@@ -13,6 +13,11 @@ def get_dm_system_prompt() -> str:
return (PROMPTS_DIR / "dm_system_prompt.txt").read_text(encoding="utf-8")
@lru_cache
def get_game_setup_prompt() -> str:
return (PROMPTS_DIR / "game_setup_prompt.txt").read_text(encoding="utf-8")
@lru_cache
def get_llm_client() -> AsyncOpenAI:
return AsyncOpenAI(api_key=settings.xai_api_key, base_url=settings.xai_base_url)
+55
View File
@@ -0,0 +1,55 @@
import json
from app.config import settings
from app.llm.client import get_game_setup_prompt, get_llm_client
from app.llm.tools import game_setup as game_setup_tool
MAX_ROUNDS = 3
MAX_TOKENS = 1024
TOOLS = [{"type": "function", "function": game_setup_tool.TOOL_SCHEMA}]
async def run_game_setup_turn(client_messages: list[dict]) -> dict:
"""Stateless setup-wizard turn: the caller owns conversation history (no DB, no game yet).
Returns {"messages": <history to resend next turn>, "assistant_text": str | None,
"proposal": {"name": str, "description": str} | None}.
"""
client = get_llm_client()
messages = [{"role": "system", "content": get_game_setup_prompt()}] + client_messages
assistant_text: str | None = None
proposal: dict | None = None
for _ in range(MAX_ROUNDS):
response = await client.chat.completions.create(
model=settings.dm_model,
max_tokens=MAX_TOKENS,
messages=messages,
tools=TOOLS,
)
message = response.choices[0].message
if message.content:
assistant_text = message.content
if not message.tool_calls:
messages.append({"role": "assistant", "content": message.content or ""})
break
messages.append(message.model_dump(exclude_unset=True))
for tool_call in message.tool_calls:
args = json.loads(tool_call.function.arguments)
if tool_call.function.name == "propose_game_setup":
proposal = {"name": args.get("name", ""), "description": args.get("description", "")}
messages.append(
{
"role": "tool",
"tool_call_id": tool_call.id,
"content": json.dumps({"status": "ok"}),
}
)
if proposal:
break
return {"messages": messages[1:], "assistant_text": assistant_text, "proposal": proposal}
+14 -4
View File
@@ -4,13 +4,14 @@ import uuid
from sqlalchemy.ext.asyncio import AsyncSession
logger = logging.getLogger("app.llm.orchestrator")
from app.config import settings
from app.llm.client import get_dm_system_prompt, get_llm_client
from app.llm.context import build_context
from app.llm.tools import character_sheet, dice
from app.models.message import Message
from app.rag.retrieval import build_rag_block
logger = logging.getLogger("app.llm.orchestrator")
MAX_TOOL_ROUNDS = 5
MAX_TOKENS = 4096
@@ -38,9 +39,18 @@ async def _execute_tool_call(
return {"error": str(exc)}
async def run_dm_turn(session: AsyncSession, game_id: uuid.UUID) -> Message:
async def run_dm_turn(
session: AsyncSession, game_id: uuid.UUID, latest_player_message: str | None = None
) -> Message:
client = get_llm_client()
messages = [{"role": "system", "content": get_dm_system_prompt()}]
system_prompt = get_dm_system_prompt()
if latest_player_message:
rag_block = await build_rag_block(session, latest_player_message)
if rag_block:
system_prompt = f"{system_prompt}\n\n{rag_block}"
messages = [{"role": "system", "content": system_prompt}]
messages.extend(await build_context(session, game_id))
final_text = ""
@@ -0,0 +1,21 @@
Rolle
Du hilfst einem Spieler dabei, ein neues D&D-Spiel einzurichten, bevor es losgeht. Du bist derselbe erfahrene Dungeon Master, aber in dieser kurzen Setup-Phase geht es nicht ums eigentliche Abenteuer, sondern nur darum, die Eckdaten zu klären.
Ablauf
Stelle dem Spieler nacheinander (eine Frage nach der anderen, nicht alle auf einmal) folgende drei Fragen, bis du zu allen dreien eine Antwort hast:
1. Welches Genre bzw. welche Welt wünscht er sich (z. B. High Fantasy, düster, humorvoll, Horror, Piraten, eine eigene Idee)?
2. Wie lang soll das Spiel werden — ein einzelnes One-Shot-Abenteuer oder eine längere Kampagne über mehrere Sitzungen?
3. Wie ist sein Erfahrungsstand mit D&D — komplette(r) Anfänger(in), gelegentlich gespielt, oder erfahrene(r) Regelkenner(in)?
Sobald du alle drei Antworten hast, rufe das Werkzeug propose_game_setup auf mit:
- einem passenden, einprägsamen Namen für das Spiel
- einer kurzen Beschreibung (Genre und Welt, optional ein Hinweis auf Umfang und Erfahrungsstand)
Der Spieler sieht deinen Vorschlag danach in einem Formular und kann Name und Beschreibung noch selbst anpassen, bevor er das Spiel erstellt.
Ton
Bleib locker, freundlich und kurz angebunden — das ist nur die Einrichtung, kein Rollenspiel. Sprich den Spieler per "Du" an.
Sprache
Antworte in der Sprache, in der der Spieler schreibt.
+22
View File
@@ -0,0 +1,22 @@
TOOL_SCHEMA = {
"name": "propose_game_setup",
"description": (
"Call this once you know the desired genre/world, the intended length (one-shot vs. "
"campaign), and the player's experience level. Propose a fitting name and a short "
"description for the new game — the player can still edit both before creating it."
),
"parameters": {
"type": "object",
"properties": {
"name": {
"type": "string",
"description": "A fitting, evocative name for the game.",
},
"description": {
"type": "string",
"description": "Short description covering genre and world/setting.",
},
},
"required": ["name", "description"],
},
}
+2 -1
View File
@@ -1,6 +1,7 @@
from app.models.character import Character
from app.models.game import Game, GameParticipant
from app.models.message import Message
from app.models.rulebook_chunk import RulebookChunk
from app.models.user import User
__all__ = ["User", "Game", "GameParticipant", "Character", "Message"]
__all__ = ["User", "Game", "GameParticipant", "Character", "Message", "RulebookChunk"]
+23
View File
@@ -0,0 +1,23 @@
import uuid
from pgvector.sqlalchemy import Vector
from sqlalchemy import Integer, String, Text
from sqlalchemy.dialects.postgresql import JSONB, UUID
from sqlalchemy.orm import Mapped, mapped_column
from app.db import Base
EMBEDDING_DIM = 384
class RulebookChunk(Base):
__tablename__ = "rulebook_chunks"
id: Mapped[uuid.UUID] = mapped_column(
UUID(as_uuid=True), primary_key=True, default=uuid.uuid4
)
source_document: Mapped[str] = mapped_column(String(length=200), nullable=False, index=True)
chunk_index: Mapped[int] = mapped_column(Integer, nullable=False)
content: Mapped[str] = mapped_column(Text, nullable=False)
embedding: Mapped[list[float]] = mapped_column(Vector(EMBEDDING_DIM), nullable=False)
doc_metadata: Mapped[dict] = mapped_column(JSONB, nullable=False, default=dict)
View File
+33
View File
@@ -0,0 +1,33 @@
import re
# Rough word-count targets standing in for the 300-800 token / 10-20% overlap guideline —
# English prose runs ~0.75 words per token, so ~380 words ≈ 500 tokens.
WORDS_PER_CHUNK = 380
OVERLAP_WORDS = 60
_WHITESPACE_RE = re.compile(r"[ \t]+")
_BLANK_LINES_RE = re.compile(r"\n{3,}")
def _normalize(text: str) -> str:
text = _WHITESPACE_RE.sub(" ", text)
text = _BLANK_LINES_RE.sub("\n\n", text)
return text.strip()
def chunk_text(
text: str, words_per_chunk: int = WORDS_PER_CHUNK, overlap_words: int = OVERLAP_WORDS
) -> list[str]:
words = _normalize(text).split()
if not words:
return []
step = words_per_chunk - overlap_words
chunks = []
start = 0
while start < len(words):
chunks.append(" ".join(words[start : start + words_per_chunk]))
if start + words_per_chunk >= len(words):
break
start += step
return chunks
+22
View File
@@ -0,0 +1,22 @@
from functools import lru_cache
from sentence_transformers import SentenceTransformer
# Multilingual so German player messages retrieve relevant chunks from the
# English-language rulebooks. 384-dim output, matching RulebookChunk.embedding.
MODEL_NAME = "paraphrase-multilingual-MiniLM-L12-v2"
@lru_cache
def get_embedding_model() -> SentenceTransformer:
return SentenceTransformer(MODEL_NAME)
def embed_texts(texts: list[str]) -> list[list[float]]:
model = get_embedding_model()
embeddings = model.encode(texts, normalize_embeddings=True, show_progress_bar=False)
return embeddings.tolist()
def embed_query(text: str) -> list[float]:
return embed_texts([text])[0]
+51
View File
@@ -0,0 +1,51 @@
import asyncio
from sqlalchemy import select
from sqlalchemy.ext.asyncio import AsyncSession
from app.models.rulebook_chunk import RulebookChunk
from app.rag.embeddings import embed_query
DEFAULT_TOP_K = 5
# Cosine distance cutoff (0 = identical, 2 = opposite) — drop weak matches rather than
# stuffing the prompt with irrelevant rulebook text when nothing actually fits the query.
# Calibrated against test queries: genuinely relevant chunks scored 0.30-0.45.
MAX_DISTANCE = 0.6
# Hard cap on injected rulebook text so one turn's context/cost doesn't balloon even when
# several chunks all clear the relevance threshold.
MAX_BLOCK_CHARS = 4000
async def retrieve_relevant_chunks(
session: AsyncSession, query: str, top_k: int = DEFAULT_TOP_K
) -> list[tuple[RulebookChunk, float]]:
# embed_query is a synchronous, CPU-bound sentence-transformers call — run it off the
# event loop so it doesn't stall other concurrent requests/WebSocket connections.
query_embedding = await asyncio.to_thread(embed_query, query)
distance = RulebookChunk.embedding.cosine_distance(query_embedding)
rows = (
await session.execute(
select(RulebookChunk, distance.label("distance")).order_by(distance).limit(top_k)
)
).all()
return [(row[0], row[1]) for row in rows]
async def build_rag_block(session: AsyncSession, query: str, top_k: int = DEFAULT_TOP_K) -> str:
results = await retrieve_relevant_chunks(session, query, top_k=top_k)
relevant = [chunk for chunk, distance in results if distance <= MAX_DISTANCE]
if not relevant:
return ""
parts = [
"Relevante Auszüge aus den Regelwerken — richte dich danach und nenne bei Bedarf die Quelle:"
]
used_chars = 0
for chunk in relevant:
entry = f"\n[{chunk.source_document}]\n{chunk.content}"
if used_chars + len(entry) > MAX_BLOCK_CHARS and used_chars > 0:
break
parts.append(entry)
used_chars += len(entry)
return "\n".join(parts)
+16
View File
@@ -0,0 +1,16 @@
from pydantic import BaseModel
class GameSetupChatRequest(BaseModel):
messages: list[dict] = []
class GameSetupProposal(BaseModel):
name: str
description: str
class GameSetupChatResponse(BaseModel):
messages: list[dict]
assistant_text: str | None
proposal: GameSetupProposal | None
+2
View File
@@ -9,3 +9,5 @@ openai>=1.50.0,<2.0.0
aiosmtplib==3.0.2
python-multipart==0.0.20
websockets==14.1
pgvector==0.3.6
sentence-transformers==3.3.1
File diff suppressed because it is too large Load Diff
File diff suppressed because it is too large Load Diff
File diff suppressed because it is too large Load Diff
+75
View File
@@ -0,0 +1,75 @@
"""Chunk and embed the D&D rulebooks into the rulebook_chunks table.
Usage (inside the backend container):
python scripts/ingest_rulebooks.py
Safe to re-run: each document's previous chunks are deleted before re-ingesting it.
"""
import asyncio
import sys
from pathlib import Path
sys.path.insert(0, str(Path(__file__).resolve().parent.parent))
from sqlalchemy import delete # noqa: E402
from app.db import async_session_maker # noqa: E402
from app.models.rulebook_chunk import RulebookChunk # noqa: E402
from app.rag.chunking import chunk_text # noqa: E402
from app.rag.embeddings import embed_texts # noqa: E402
RULEBOOKS_DIR = Path(__file__).resolve().parent.parent / "rulebooks"
DOCUMENTS = {
"DnDPlayersHandbook.txt": "Player's Handbook",
"DnDMastersGuide.txt": "Dungeon Master's Guide",
"DnDMonsterHandbook.txt": "Monster Manual",
}
EMBED_BATCH_SIZE = 32
async def ingest_document(session, filename: str, display_name: str) -> int:
path = RULEBOOKS_DIR / filename
if not path.exists():
print(f" Skipping {display_name} — {path} not found")
return 0
text = path.read_text(encoding="utf-8", errors="ignore")
chunks = chunk_text(text)
if not chunks:
print(f" {display_name}: empty, nothing to ingest")
return 0
print(f" {display_name}: {len(chunks)} chunks")
await session.execute(delete(RulebookChunk).where(RulebookChunk.source_document == display_name))
for batch_start in range(0, len(chunks), EMBED_BATCH_SIZE):
batch = chunks[batch_start : batch_start + EMBED_BATCH_SIZE]
embeddings = embed_texts(batch)
for i, (content, embedding) in enumerate(zip(batch, embeddings)):
session.add(
RulebookChunk(
source_document=display_name,
chunk_index=batch_start + i,
content=content,
embedding=embedding,
)
)
await session.commit()
print(f" {min(batch_start + EMBED_BATCH_SIZE, len(chunks))}/{len(chunks)}")
return len(chunks)
async def main() -> None:
total = 0
async with async_session_maker() as session:
for filename, display_name in DOCUMENTS.items():
total += await ingest_document(session, filename, display_name)
print(f"Done — {total} chunks ingested.")
if __name__ == "__main__":
asyncio.run(main())