Thicket/thicket/chunker.py

185 lines
6.8 KiB
Python

"""Contextual chunker — structure-preserving splitting for technical
corpora.
The chunker treats programming books, system configuration, and policy
documents as what they are: mixed prose and code.
Invariants:
* Fenced code blocks (``` / ~~~) are atomic — a block is never split
mid-listing, never merged with prose, and keeps every newline and
indentation character verbatim. Only oversized blocks (beyond
~2x the budget) are line-windowed, and every window carries the
language tag.
* Prose chunks are paragraph-aligned: paragraphs are never split
mid-way (an oversized paragraph is line-windowed with its lines
preserved), and intra-chunk newlines are kept — lists, commands,
and tables hold their line structure in the stored payload.
* ATX headings require whitespace after the hashes (``^#{1,6}\\s``),
so shebangs (``#!``), machine comments (``#x``), and fenced code
never masquerade as section headers.
* Each chunk carries a contextual prefix — ``[Source: … | Section: …]``
plus ``| code:lang`` for code — which is what gets embedded; the
plain text is stored verbatim.
"""
from __future__ import annotations
import re
from dataclasses import dataclass
# ATX heading: 1-6 hashes, mandatory whitespace, then content — rejects
# "#!/shebang", "#no-space-comment", and bare "#######" runs.
HEADING_RE = re.compile(r"^#{1,6}\s+\S")
# Fenced block opening: ``` or ~~~ with an optional language tag.
FENCE_OPEN_RE = re.compile(r"^(`{3,}|~{3,})\s*(\S*)\s*$")
# Indented block (pandoc-style code): 4+ spaces or a tab, not a list.
INDENT_RE = re.compile(r"^(?: {4,}|\t)\S")
# Code blocks larger than chunk_size * this factor are line-windowed.
OVERSIZED_FACTOR = 2
@dataclass(slots=True)
class Chunk:
header: str
text: str
contextual_text: str
kind: str = "prose" # "prose" | "code"
lang: str | None = None # language tag for code chunks
source: str | None = None # input-tree path of the origin file
def _words(text: str) -> int:
return len(text.split())
class ContextualChunker:
"""Splits markdown-ish text into header-aware, code-preserving
contextual chunks."""
def __init__(self, chunk_size: int = 500, overlap: int = 50):
self.chunk_size = max(1, int(chunk_size))
self.overlap = max(0, int(overlap))
# ── block parsing ──
def _blocks(self, text: str):
"""Yield (kind, body, lang) blocks: fenced code carries its
whole verbatim body; everything else is blank-line-delimited."""
lines = text.split("\n")
i = 0
while i < len(lines):
fence = FENCE_OPEN_RE.match(lines[i])
if fence:
marker, lang = fence.group(1), fence.group(2)
j = i + 1
while j < len(lines) and not lines[j].startswith(marker[:3]):
j += 1
yield ("code", "\n".join(lines[i + 1:j]), lang or None)
i = j + 1
continue
if not lines[i].strip():
i += 1
continue
j = i
while j < len(lines) and lines[j].strip():
j += 1
yield ("para", "\n".join(lines[i:j]), None)
i = j
# ── chunk assembly ──
def chunk(self, title: str, text: str,
source_path: str | None = None) -> list[Chunk]:
chunks: list[Chunk] = []
header = "Introduction"
buffer: list[str] = [] # prose paragraphs, verbatim
def _flush() -> None:
nonlocal buffer
if not buffer:
return
chunks.append(self._prose_chunk(title, header,
"\n\n".join(buffer)))
# Paragraph-level overlap: carry the tail paragraph into
# the next chunk when it fits the overlap budget.
buffer = ([buffer[-1]] if _words(buffer[-1]) <= self.overlap
else [])
for kind, body, lang in self._blocks(text):
if kind == "code":
_flush()
chunks.extend(self._code_chunks(title, header, body, lang))
continue
first_line = body.split("\n", 1)[0]
if "\n" not in body and HEADING_RE.match(first_line):
_flush() # section boundary never straddles chunks
header = first_line.lstrip("#").strip() or header
continue
if len(body.split("\n")) > 1 and all(
INDENT_RE.match(ln) or not ln.strip()
for ln in body.split("\n")):
# Indented block (pandoc-style code): treat as code.
_flush()
chunks.extend(self._code_chunks(
title, header,
re.sub(r"^ {0,4}", "", body, flags=re.M), None))
continue
if buffer and _words("\n\n".join(buffer)) + _words(body) > self.chunk_size:
_flush()
buffer.append(body)
if _words("\n\n".join(buffer)) >= self.chunk_size:
_flush()
_flush()
if source_path:
for chunk in chunks:
chunk.source = source_path
return chunks
# ── chunk constructors ──
def _prose_chunk(self, title: str, header: str, body: str) -> Chunk:
return Chunk(
header=header,
text=body,
contextual_text=f"[Source: {title} | Section: {header}]\n{body}",
kind="prose",
)
def _code_chunk(self, title: str, header: str, body: str,
lang: str | None) -> Chunk:
tag = f" | code:{lang}" if lang else " | code"
return Chunk(
header=header,
text=body,
contextual_text=f"[Source: {title} | Section: {header}{tag}]\n{body}",
kind="code",
lang=lang,
)
def _code_chunks(self, title: str, header: str, body: str,
lang: str | None) -> list[Chunk]:
"""One atomic code chunk — or line windows when the block is
oversized; every window keeps the language tag."""
if _words(body) <= self.chunk_size * OVERSIZED_FACTOR:
return [self._code_chunk(title, header, body, lang)]
lines = body.split("\n")
windows: list[list[str]] = []
current: list[str] = []
for line in lines:
current.append(line)
if _words("\n".join(current)) >= self.chunk_size:
windows.append(current)
# Two tail lines carry into the next window as overlap.
current = current[-2:]
if current and (not windows or current != windows[-1]):
windows.append(current)
return [self._code_chunk(title, header, "\n".join(w), lang)
for w in windows]