Add rulebook RAG pipeline and LLM-driven game setup wizard
RAG: switch Postgres to pgvector, chunk and embed the three D&D rulebooks locally via sentence-transformers, and retrieve relevant excerpts per DM turn (query = latest player message) to ground the system prompt. Retrieval runs off the event loop and is capped by a relevance threshold and a max character budget so it can't blow up context size or cost. Game setup wizard: creating a game now opens a short chat where the DM asks about genre, length, and the player's experience level, then proposes a name and description via a tool call. The player can edit both before creating the game. Stateless endpoint — the frontend carries the conversation, no DB needed since the game doesn't exist yet. Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
This commit is contained in:
@@ -0,0 +1,75 @@
|
||||
"""Chunk and embed the D&D rulebooks into the rulebook_chunks table.
|
||||
|
||||
Usage (inside the backend container):
|
||||
python scripts/ingest_rulebooks.py
|
||||
|
||||
Safe to re-run: each document's previous chunks are deleted before re-ingesting it.
|
||||
"""
|
||||
|
||||
import asyncio
|
||||
import sys
|
||||
from pathlib import Path
|
||||
|
||||
sys.path.insert(0, str(Path(__file__).resolve().parent.parent))
|
||||
|
||||
from sqlalchemy import delete # noqa: E402
|
||||
|
||||
from app.db import async_session_maker # noqa: E402
|
||||
from app.models.rulebook_chunk import RulebookChunk # noqa: E402
|
||||
from app.rag.chunking import chunk_text # noqa: E402
|
||||
from app.rag.embeddings import embed_texts # noqa: E402
|
||||
|
||||
RULEBOOKS_DIR = Path(__file__).resolve().parent.parent / "rulebooks"
|
||||
|
||||
DOCUMENTS = {
|
||||
"DnDPlayersHandbook.txt": "Player's Handbook",
|
||||
"DnDMastersGuide.txt": "Dungeon Master's Guide",
|
||||
"DnDMonsterHandbook.txt": "Monster Manual",
|
||||
}
|
||||
|
||||
EMBED_BATCH_SIZE = 32
|
||||
|
||||
|
||||
async def ingest_document(session, filename: str, display_name: str) -> int:
|
||||
path = RULEBOOKS_DIR / filename
|
||||
if not path.exists():
|
||||
print(f" Skipping {display_name} — {path} not found")
|
||||
return 0
|
||||
|
||||
text = path.read_text(encoding="utf-8", errors="ignore")
|
||||
chunks = chunk_text(text)
|
||||
if not chunks:
|
||||
print(f" {display_name}: empty, nothing to ingest")
|
||||
return 0
|
||||
|
||||
print(f" {display_name}: {len(chunks)} chunks")
|
||||
await session.execute(delete(RulebookChunk).where(RulebookChunk.source_document == display_name))
|
||||
|
||||
for batch_start in range(0, len(chunks), EMBED_BATCH_SIZE):
|
||||
batch = chunks[batch_start : batch_start + EMBED_BATCH_SIZE]
|
||||
embeddings = embed_texts(batch)
|
||||
for i, (content, embedding) in enumerate(zip(batch, embeddings)):
|
||||
session.add(
|
||||
RulebookChunk(
|
||||
source_document=display_name,
|
||||
chunk_index=batch_start + i,
|
||||
content=content,
|
||||
embedding=embedding,
|
||||
)
|
||||
)
|
||||
await session.commit()
|
||||
print(f" {min(batch_start + EMBED_BATCH_SIZE, len(chunks))}/{len(chunks)}")
|
||||
|
||||
return len(chunks)
|
||||
|
||||
|
||||
async def main() -> None:
|
||||
total = 0
|
||||
async with async_session_maker() as session:
|
||||
for filename, display_name in DOCUMENTS.items():
|
||||
total += await ingest_document(session, filename, display_name)
|
||||
print(f"Done — {total} chunks ingested.")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
asyncio.run(main())
|
||||
Reference in New Issue
Block a user