"""Contextual chunker — structure-preserving splitting for technical corpora. The chunker treats programming books, system configuration, and policy documents as what they are: mixed prose and code. Invariants: * Fenced code blocks (``` / ~~~) are atomic — a block is never split mid-listing, never merged with prose, and keeps every newline and indentation character verbatim. Only oversized blocks (beyond ~2x the budget) are line-windowed, and every window carries the language tag. * Prose chunks are paragraph-aligned: paragraphs are never split mid-way (an oversized paragraph is line-windowed with its lines preserved), and intra-chunk newlines are kept — lists, commands, and tables hold their line structure in the stored payload. * ATX headings require whitespace after the hashes (``^#{1,6}\\s``), so shebangs (``#!``), machine comments (``#x``), and fenced code never masquerade as section headers. * Each chunk carries a contextual prefix — ``[Source: … | Section: …]`` plus ``| code:lang`` for code — which is what gets embedded; the plain text is stored verbatim. """ from __future__ import annotations import re from dataclasses import dataclass # ATX heading: 1-6 hashes, mandatory whitespace, then content — rejects # "#!/shebang", "#no-space-comment", and bare "#######" runs. HEADING_RE = re.compile(r"^#{1,6}\s+\S") # Fenced block opening: ``` or ~~~ with an optional language tag. FENCE_OPEN_RE = re.compile(r"^(`{3,}|~{3,})\s*(\S*)\s*$") # Indented block (pandoc-style code): 4+ spaces or a tab, not a list. INDENT_RE = re.compile(r"^(?: {4,}|\t)\S") # Code blocks larger than chunk_size * this factor are line-windowed. OVERSIZED_FACTOR = 2 @dataclass(slots=True) class Chunk: header: str text: str contextual_text: str kind: str = "prose" # "prose" | "code" lang: str | None = None # language tag for code chunks source: str | None = None # input-tree path of the origin file def _words(text: str) -> int: return len(text.split()) class ContextualChunker: """Splits markdown-ish text into header-aware, code-preserving contextual chunks.""" def __init__(self, chunk_size: int = 500, overlap: int = 50): self.chunk_size = max(1, int(chunk_size)) self.overlap = max(0, int(overlap)) # ── block parsing ── def _blocks(self, text: str): """Yield (kind, body, lang) blocks: fenced code carries its whole verbatim body; everything else is blank-line-delimited.""" lines = text.split("\n") i = 0 while i < len(lines): fence = FENCE_OPEN_RE.match(lines[i]) if fence: marker, lang = fence.group(1), fence.group(2) j = i + 1 while j < len(lines) and not lines[j].startswith(marker[:3]): j += 1 yield ("code", "\n".join(lines[i + 1:j]), lang or None) i = j + 1 continue if not lines[i].strip(): i += 1 continue j = i while j < len(lines) and lines[j].strip(): j += 1 yield ("para", "\n".join(lines[i:j]), None) i = j # ── chunk assembly ── def chunk(self, title: str, text: str, source_path: str | None = None) -> list[Chunk]: chunks: list[Chunk] = [] header = "Introduction" buffer: list[str] = [] # prose paragraphs, verbatim def _flush() -> None: nonlocal buffer if not buffer: return chunks.append(self._prose_chunk(title, header, "\n\n".join(buffer))) # Paragraph-level overlap: carry the tail paragraph into # the next chunk when it fits the overlap budget. buffer = ([buffer[-1]] if _words(buffer[-1]) <= self.overlap else []) for kind, body, lang in self._blocks(text): if kind == "code": _flush() chunks.extend(self._code_chunks(title, header, body, lang)) continue first_line = body.split("\n", 1)[0] if "\n" not in body and HEADING_RE.match(first_line): _flush() # section boundary never straddles chunks header = first_line.lstrip("#").strip() or header continue if len(body.split("\n")) > 1 and all( INDENT_RE.match(ln) or not ln.strip() for ln in body.split("\n")): # Indented block (pandoc-style code): treat as code. _flush() chunks.extend(self._code_chunks( title, header, re.sub(r"^ {0,4}", "", body, flags=re.M), None)) continue if buffer and _words("\n\n".join(buffer)) + _words(body) > self.chunk_size: _flush() buffer.append(body) if _words("\n\n".join(buffer)) >= self.chunk_size: _flush() _flush() if source_path: for chunk in chunks: chunk.source = source_path return chunks # ── chunk constructors ── def _prose_chunk(self, title: str, header: str, body: str) -> Chunk: return Chunk( header=header, text=body, contextual_text=f"[Source: {title} | Section: {header}]\n{body}", kind="prose", ) def _code_chunk(self, title: str, header: str, body: str, lang: str | None) -> Chunk: tag = f" | code:{lang}" if lang else " | code" return Chunk( header=header, text=body, contextual_text=f"[Source: {title} | Section: {header}{tag}]\n{body}", kind="code", lang=lang, ) def _code_chunks(self, title: str, header: str, body: str, lang: str | None) -> list[Chunk]: """One atomic code chunk — or line windows when the block is oversized; every window keeps the language tag.""" if _words(body) <= self.chunk_size * OVERSIZED_FACTOR: return [self._code_chunk(title, header, body, lang)] lines = body.split("\n") windows: list[list[str]] = [] current: list[str] = [] for line in lines: current.append(line) if _words("\n".join(current)) >= self.chunk_size: windows.append(current) # Two tail lines carry into the next window as overlap. current = current[-2:] if current and (not windows or current != windows[-1]): windows.append(current) return [self._code_chunk(title, header, "\n".join(w), lang) for w in windows]