import re # Rough word-count targets standing in for the 300-800 token / 10-20% overlap guideline — # English prose runs ~0.75 words per token, so ~380 words ≈ 500 tokens. WORDS_PER_CHUNK = 380 OVERLAP_WORDS = 60 _WHITESPACE_RE = re.compile(r"[ \t]+") _BLANK_LINES_RE = re.compile(r"\n{3,}") def _normalize(text: str) -> str: text = _WHITESPACE_RE.sub(" ", text) text = _BLANK_LINES_RE.sub("\n\n", text) return text.strip() def chunk_text( text: str, words_per_chunk: int = WORDS_PER_CHUNK, overlap_words: int = OVERLAP_WORDS ) -> list[str]: words = _normalize(text).split() if not words: return [] step = words_per_chunk - overlap_words chunks = [] start = 0 while start < len(words): chunks.append(" ".join(words[start : start + words_per_chunk])) if start + words_per_chunk >= len(words): break start += step return chunks