419f5e3a89
RAG: switch Postgres to pgvector, chunk and embed the three D&D rulebooks locally via sentence-transformers, and retrieve relevant excerpts per DM turn (query = latest player message) to ground the system prompt. Retrieval runs off the event loop and is capped by a relevance threshold and a max character budget so it can't blow up context size or cost. Game setup wizard: creating a game now opens a short chat where the DM asks about genre, length, and the player's experience level, then proposes a name and description via a tool call. The player can edit both before creating the game. Stateless endpoint — the frontend carries the conversation, no DB needed since the game doesn't exist yet. Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
76 lines
2.4 KiB
Python
76 lines
2.4 KiB
Python
"""Chunk and embed the D&D rulebooks into the rulebook_chunks table.
|
|
|
|
Usage (inside the backend container):
|
|
python scripts/ingest_rulebooks.py
|
|
|
|
Safe to re-run: each document's previous chunks are deleted before re-ingesting it.
|
|
"""
|
|
|
|
import asyncio
|
|
import sys
|
|
from pathlib import Path
|
|
|
|
sys.path.insert(0, str(Path(__file__).resolve().parent.parent))
|
|
|
|
from sqlalchemy import delete # noqa: E402
|
|
|
|
from app.db import async_session_maker # noqa: E402
|
|
from app.models.rulebook_chunk import RulebookChunk # noqa: E402
|
|
from app.rag.chunking import chunk_text # noqa: E402
|
|
from app.rag.embeddings import embed_texts # noqa: E402
|
|
|
|
RULEBOOKS_DIR = Path(__file__).resolve().parent.parent / "rulebooks"
|
|
|
|
DOCUMENTS = {
|
|
"DnDPlayersHandbook.txt": "Player's Handbook",
|
|
"DnDMastersGuide.txt": "Dungeon Master's Guide",
|
|
"DnDMonsterHandbook.txt": "Monster Manual",
|
|
}
|
|
|
|
EMBED_BATCH_SIZE = 32
|
|
|
|
|
|
async def ingest_document(session, filename: str, display_name: str) -> int:
|
|
path = RULEBOOKS_DIR / filename
|
|
if not path.exists():
|
|
print(f" Skipping {display_name} — {path} not found")
|
|
return 0
|
|
|
|
text = path.read_text(encoding="utf-8", errors="ignore")
|
|
chunks = chunk_text(text)
|
|
if not chunks:
|
|
print(f" {display_name}: empty, nothing to ingest")
|
|
return 0
|
|
|
|
print(f" {display_name}: {len(chunks)} chunks")
|
|
await session.execute(delete(RulebookChunk).where(RulebookChunk.source_document == display_name))
|
|
|
|
for batch_start in range(0, len(chunks), EMBED_BATCH_SIZE):
|
|
batch = chunks[batch_start : batch_start + EMBED_BATCH_SIZE]
|
|
embeddings = embed_texts(batch)
|
|
for i, (content, embedding) in enumerate(zip(batch, embeddings)):
|
|
session.add(
|
|
RulebookChunk(
|
|
source_document=display_name,
|
|
chunk_index=batch_start + i,
|
|
content=content,
|
|
embedding=embedding,
|
|
)
|
|
)
|
|
await session.commit()
|
|
print(f" {min(batch_start + EMBED_BATCH_SIZE, len(chunks))}/{len(chunks)}")
|
|
|
|
return len(chunks)
|
|
|
|
|
|
async def main() -> None:
|
|
total = 0
|
|
async with async_session_maker() as session:
|
|
for filename, display_name in DOCUMENTS.items():
|
|
total += await ingest_document(session, filename, display_name)
|
|
print(f"Done — {total} chunks ingested.")
|
|
|
|
|
|
if __name__ == "__main__":
|
|
asyncio.run(main())
|