Thicket/thicket/extractors.py

150 lines
5.4 KiB
Python

"""Document extraction — PDF / EPUB / Markdown / plain text.
Extracts ``(title, text)`` from every supported format. Structural
headers are preserved in the extracted text (PDF pages become
``## Page N`` headings, EPUB chapters are joined with rules) so the
chunker can attribute each chunk to its section.
All third-party parsers are imported lazily inside the extract methods:
the GUI must launch and probe on a bare system where EbookLib or pypdf
are not installed, and fail with an actionable message only when the
matching file type is actually encountered.
"""
from __future__ import annotations
import re
from pathlib import Path
# Source files ingested verbatim as language-tagged code blocks.
CODE_LANGUAGES: dict[str, str] = {
".py": "python", ".sh": "bash", ".bash": "bash", ".zsh": "zsh",
".yaml": "yaml", ".yml": "yaml", ".toml": "toml", ".ini": "ini",
".conf": "ini", ".cfg": "ini", ".json": "json", ".sql": "sql",
".rs": "rust", ".go": "go", ".c": "c", ".h": "c", ".cpp": "cpp",
".js": "javascript", ".ts": "typescript", ".tf": "hcl", ".nix": "nix",
}
SUPPORTED_EXTENSIONS = frozenset(
{".pdf", ".epub", ".md", ".markdown", ".txt"}
| set(CODE_LANGUAGES)
)
class ExtractionError(Exception):
"""Raised when a document cannot be parsed (missing dep, bad file)."""
def scan_files(
input_dir: Path,
extensions: frozenset[str] | set[str] = SUPPORTED_EXTENSIONS,
) -> list[Path]:
"""Recursively list supported documents under *input_dir*, sorted
for deterministic queue order."""
return sorted(
(p for p in input_dir.rglob("*")
if p.is_file() and p.suffix.lower() in extensions),
key=lambda p: str(p).lower(),
)
def _title_from_stem(filepath: Path) -> str:
return filepath.stem.replace("-", " ").replace("_", " ").title()
class DocumentExtractor:
"""Extracts raw text and structural headers from supported formats."""
@staticmethod
def extract(filepath: Path) -> tuple[str, str]:
"""Dispatch on extension via the runner table. Raises
ExtractionError for unsupported types or missing parser
dependencies."""
runner = _EXTENSION_RUNNERS.get(filepath.suffix.lower())
if runner is None:
raise ExtractionError(f"unsupported file type: {filepath.suffix}")
return runner(filepath)
@staticmethod
def extract_md_txt(filepath: Path) -> tuple[str, str]:
text = filepath.read_text(encoding="utf-8", errors="ignore")
return _title_from_stem(filepath), text
@staticmethod
def extract_code(filepath: Path) -> tuple[str, str]:
"""Source/config file: verbatim content wrapped in a
language-tagged fence so the chunker treats it as one code
document and the vault note renders it as a listing."""
lang = CODE_LANGUAGES.get(filepath.suffix.lower(), "")
body = filepath.read_text(encoding="utf-8", errors="ignore")
return _title_from_stem(filepath), f"```{lang}\n{body}\n```"
@staticmethod
def extract_pdf(filepath: Path) -> tuple[str, str]:
try:
from pypdf import PdfReader
except ImportError as e:
raise ExtractionError(
"pypdf not installed — run: pip install 'thicket[ingest]'"
) from e
reader = PdfReader(str(filepath))
title = _title_from_stem(filepath)
# Prefer the embedded metadata title when present.
try:
if reader.metadata and reader.metadata.title:
title = str(reader.metadata.title).strip() or title
except Exception:
pass # malformed metadata must not kill the extract
pages_text: list[str] = []
for i, page in enumerate(reader.pages):
try:
txt = page.extract_text() or ""
except Exception:
txt = ""
if txt.strip():
pages_text.append(f"## Page {i + 1}\n\n{txt}")
return title, "\n\n".join(pages_text)
@staticmethod
def extract_epub(filepath: Path) -> tuple[str, str]:
try:
import bs4
from ebooklib import epub, ITEM_DOCUMENT
except ImportError as e:
raise ExtractionError(
"ebooklib / beautifulsoup4 not installed — run: "
"pip install 'thicket[ingest]'"
) from e
book = epub.read_epub(str(filepath))
title = _title_from_stem(filepath)
meta_titles = book.get_metadata("DC", "title")
if meta_titles:
title = str(meta_titles[0][0]).strip() or title
chapters: list[str] = []
for item in book.get_items_of_type(ITEM_DOCUMENT):
soup = bs4.BeautifulSoup(item.get_content(), "html.parser")
text = soup.get_text(separator="\n")
clean_text = re.sub(r"\n+", "\n", text).strip()
if clean_text:
chapters.append(clean_text)
return title, "\n\n---\n\n".join(chapters)
# Extension dispatch table — the single source of routing for extract().
# Adding a format means adding one entry here and its extractor method.
_EXTENSION_RUNNERS: dict[str, object] = {
".md": DocumentExtractor.extract_md_txt,
".markdown": DocumentExtractor.extract_md_txt,
".txt": DocumentExtractor.extract_md_txt,
".pdf": DocumentExtractor.extract_pdf,
".epub": DocumentExtractor.extract_epub,
**{ext: DocumentExtractor.extract_code for ext in CODE_LANGUAGES},
}