185 lines
6.8 KiB
Python
185 lines
6.8 KiB
Python
"""Contextual chunker — structure-preserving splitting for technical
|
|
corpora.
|
|
|
|
The chunker treats programming books, system configuration, and policy
|
|
documents as what they are: mixed prose and code.
|
|
|
|
Invariants:
|
|
|
|
* Fenced code blocks (``` / ~~~) are atomic — a block is never split
|
|
mid-listing, never merged with prose, and keeps every newline and
|
|
indentation character verbatim. Only oversized blocks (beyond
|
|
~2x the budget) are line-windowed, and every window carries the
|
|
language tag.
|
|
* Prose chunks are paragraph-aligned: paragraphs are never split
|
|
mid-way (an oversized paragraph is line-windowed with its lines
|
|
preserved), and intra-chunk newlines are kept — lists, commands,
|
|
and tables hold their line structure in the stored payload.
|
|
* ATX headings require whitespace after the hashes (``^#{1,6}\\s``),
|
|
so shebangs (``#!``), machine comments (``#x``), and fenced code
|
|
never masquerade as section headers.
|
|
* Each chunk carries a contextual prefix — ``[Source: … | Section: …]``
|
|
plus ``| code:lang`` for code — which is what gets embedded; the
|
|
plain text is stored verbatim.
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
import re
|
|
from dataclasses import dataclass
|
|
|
|
# ATX heading: 1-6 hashes, mandatory whitespace, then content — rejects
|
|
# "#!/shebang", "#no-space-comment", and bare "#######" runs.
|
|
HEADING_RE = re.compile(r"^#{1,6}\s+\S")
|
|
# Fenced block opening: ``` or ~~~ with an optional language tag.
|
|
FENCE_OPEN_RE = re.compile(r"^(`{3,}|~{3,})\s*(\S*)\s*$")
|
|
# Indented block (pandoc-style code): 4+ spaces or a tab, not a list.
|
|
INDENT_RE = re.compile(r"^(?: {4,}|\t)\S")
|
|
|
|
# Code blocks larger than chunk_size * this factor are line-windowed.
|
|
OVERSIZED_FACTOR = 2
|
|
|
|
|
|
@dataclass(slots=True)
|
|
class Chunk:
|
|
header: str
|
|
text: str
|
|
contextual_text: str
|
|
kind: str = "prose" # "prose" | "code"
|
|
lang: str | None = None # language tag for code chunks
|
|
source: str | None = None # input-tree path of the origin file
|
|
|
|
|
|
def _words(text: str) -> int:
|
|
return len(text.split())
|
|
|
|
|
|
class ContextualChunker:
|
|
"""Splits markdown-ish text into header-aware, code-preserving
|
|
contextual chunks."""
|
|
|
|
def __init__(self, chunk_size: int = 500, overlap: int = 50):
|
|
self.chunk_size = max(1, int(chunk_size))
|
|
self.overlap = max(0, int(overlap))
|
|
|
|
# ── block parsing ──
|
|
|
|
def _blocks(self, text: str):
|
|
"""Yield (kind, body, lang) blocks: fenced code carries its
|
|
whole verbatim body; everything else is blank-line-delimited."""
|
|
lines = text.split("\n")
|
|
i = 0
|
|
while i < len(lines):
|
|
fence = FENCE_OPEN_RE.match(lines[i])
|
|
if fence:
|
|
marker, lang = fence.group(1), fence.group(2)
|
|
j = i + 1
|
|
while j < len(lines) and not lines[j].startswith(marker[:3]):
|
|
j += 1
|
|
yield ("code", "\n".join(lines[i + 1:j]), lang or None)
|
|
i = j + 1
|
|
continue
|
|
if not lines[i].strip():
|
|
i += 1
|
|
continue
|
|
j = i
|
|
while j < len(lines) and lines[j].strip():
|
|
j += 1
|
|
yield ("para", "\n".join(lines[i:j]), None)
|
|
i = j
|
|
|
|
# ── chunk assembly ──
|
|
|
|
def chunk(self, title: str, text: str,
|
|
source_path: str | None = None) -> list[Chunk]:
|
|
chunks: list[Chunk] = []
|
|
header = "Introduction"
|
|
buffer: list[str] = [] # prose paragraphs, verbatim
|
|
|
|
def _flush() -> None:
|
|
nonlocal buffer
|
|
if not buffer:
|
|
return
|
|
chunks.append(self._prose_chunk(title, header,
|
|
"\n\n".join(buffer)))
|
|
# Paragraph-level overlap: carry the tail paragraph into
|
|
# the next chunk when it fits the overlap budget.
|
|
buffer = ([buffer[-1]] if _words(buffer[-1]) <= self.overlap
|
|
else [])
|
|
|
|
for kind, body, lang in self._blocks(text):
|
|
if kind == "code":
|
|
_flush()
|
|
chunks.extend(self._code_chunks(title, header, body, lang))
|
|
continue
|
|
|
|
first_line = body.split("\n", 1)[0]
|
|
if "\n" not in body and HEADING_RE.match(first_line):
|
|
_flush() # section boundary never straddles chunks
|
|
header = first_line.lstrip("#").strip() or header
|
|
continue
|
|
|
|
if len(body.split("\n")) > 1 and all(
|
|
INDENT_RE.match(ln) or not ln.strip()
|
|
for ln in body.split("\n")):
|
|
# Indented block (pandoc-style code): treat as code.
|
|
_flush()
|
|
chunks.extend(self._code_chunks(
|
|
title, header,
|
|
re.sub(r"^ {0,4}", "", body, flags=re.M), None))
|
|
continue
|
|
|
|
if buffer and _words("\n\n".join(buffer)) + _words(body) > self.chunk_size:
|
|
_flush()
|
|
buffer.append(body)
|
|
if _words("\n\n".join(buffer)) >= self.chunk_size:
|
|
_flush()
|
|
|
|
_flush()
|
|
if source_path:
|
|
for chunk in chunks:
|
|
chunk.source = source_path
|
|
return chunks
|
|
|
|
# ── chunk constructors ──
|
|
|
|
def _prose_chunk(self, title: str, header: str, body: str) -> Chunk:
|
|
return Chunk(
|
|
header=header,
|
|
text=body,
|
|
contextual_text=f"[Source: {title} | Section: {header}]\n{body}",
|
|
kind="prose",
|
|
)
|
|
|
|
def _code_chunk(self, title: str, header: str, body: str,
|
|
lang: str | None) -> Chunk:
|
|
tag = f" | code:{lang}" if lang else " | code"
|
|
return Chunk(
|
|
header=header,
|
|
text=body,
|
|
contextual_text=f"[Source: {title} | Section: {header}{tag}]\n{body}",
|
|
kind="code",
|
|
lang=lang,
|
|
)
|
|
|
|
def _code_chunks(self, title: str, header: str, body: str,
|
|
lang: str | None) -> list[Chunk]:
|
|
"""One atomic code chunk — or line windows when the block is
|
|
oversized; every window keeps the language tag."""
|
|
if _words(body) <= self.chunk_size * OVERSIZED_FACTOR:
|
|
return [self._code_chunk(title, header, body, lang)]
|
|
|
|
lines = body.split("\n")
|
|
windows: list[list[str]] = []
|
|
current: list[str] = []
|
|
for line in lines:
|
|
current.append(line)
|
|
if _words("\n".join(current)) >= self.chunk_size:
|
|
windows.append(current)
|
|
# Two tail lines carry into the next window as overlap.
|
|
current = current[-2:]
|
|
if current and (not windows or current != windows[-1]):
|
|
windows.append(current)
|
|
return [self._code_chunk(title, header, "\n".join(w), lang)
|
|
for w in windows]
|