Files
DungeonsDragons/backend/scripts/ingest_rulebooks.py
T
Thorsten 419f5e3a89 Add rulebook RAG pipeline and LLM-driven game setup wizard
RAG: switch Postgres to pgvector, chunk and embed the three D&D
rulebooks locally via sentence-transformers, and retrieve relevant
excerpts per DM turn (query = latest player message) to ground the
system prompt. Retrieval runs off the event loop and is capped by a
relevance threshold and a max character budget so it can't blow up
context size or cost.

Game setup wizard: creating a game now opens a short chat where the
DM asks about genre, length, and the player's experience level, then
proposes a name and description via a tool call. The player can edit
both before creating the game. Stateless endpoint — the frontend
carries the conversation, no DB needed since the game doesn't exist
yet.

Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
2026-08-31 20:03:05 +02:00

76 lines
2.4 KiB
Python

"""Chunk and embed the D&D rulebooks into the rulebook_chunks table.
Usage (inside the backend container):
python scripts/ingest_rulebooks.py
Safe to re-run: each document's previous chunks are deleted before re-ingesting it.
"""
import asyncio
import sys
from pathlib import Path
sys.path.insert(0, str(Path(__file__).resolve().parent.parent))
from sqlalchemy import delete # noqa: E402
from app.db import async_session_maker # noqa: E402
from app.models.rulebook_chunk import RulebookChunk # noqa: E402
from app.rag.chunking import chunk_text # noqa: E402
from app.rag.embeddings import embed_texts # noqa: E402
RULEBOOKS_DIR = Path(__file__).resolve().parent.parent / "rulebooks"
DOCUMENTS = {
"DnDPlayersHandbook.txt": "Player's Handbook",
"DnDMastersGuide.txt": "Dungeon Master's Guide",
"DnDMonsterHandbook.txt": "Monster Manual",
}
EMBED_BATCH_SIZE = 32
async def ingest_document(session, filename: str, display_name: str) -> int:
path = RULEBOOKS_DIR / filename
if not path.exists():
print(f" Skipping {display_name} — {path} not found")
return 0
text = path.read_text(encoding="utf-8", errors="ignore")
chunks = chunk_text(text)
if not chunks:
print(f" {display_name}: empty, nothing to ingest")
return 0
print(f" {display_name}: {len(chunks)} chunks")
await session.execute(delete(RulebookChunk).where(RulebookChunk.source_document == display_name))
for batch_start in range(0, len(chunks), EMBED_BATCH_SIZE):
batch = chunks[batch_start : batch_start + EMBED_BATCH_SIZE]
embeddings = embed_texts(batch)
for i, (content, embedding) in enumerate(zip(batch, embeddings)):
session.add(
RulebookChunk(
source_document=display_name,
chunk_index=batch_start + i,
content=content,
embedding=embedding,
)
)
await session.commit()
print(f" {min(batch_start + EMBED_BATCH_SIZE, len(chunks))}/{len(chunks)}")
return len(chunks)
async def main() -> None:
total = 0
async with async_session_maker() as session:
for filename, display_name in DOCUMENTS.items():
total += await ingest_document(session, filename, display_name)
print(f"Done — {total} chunks ingested.")
if __name__ == "__main__":
asyncio.run(main())