150 lines
5.4 KiB
Python
150 lines
5.4 KiB
Python
"""Document extraction — PDF / EPUB / Markdown / plain text.
|
|
|
|
Extracts ``(title, text)`` from every supported format. Structural
|
|
headers are preserved in the extracted text (PDF pages become
|
|
``## Page N`` headings, EPUB chapters are joined with rules) so the
|
|
chunker can attribute each chunk to its section.
|
|
|
|
All third-party parsers are imported lazily inside the extract methods:
|
|
the GUI must launch and probe on a bare system where EbookLib or pypdf
|
|
are not installed, and fail with an actionable message only when the
|
|
matching file type is actually encountered.
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
import re
|
|
from pathlib import Path
|
|
|
|
# Source files ingested verbatim as language-tagged code blocks.
|
|
CODE_LANGUAGES: dict[str, str] = {
|
|
".py": "python", ".sh": "bash", ".bash": "bash", ".zsh": "zsh",
|
|
".yaml": "yaml", ".yml": "yaml", ".toml": "toml", ".ini": "ini",
|
|
".conf": "ini", ".cfg": "ini", ".json": "json", ".sql": "sql",
|
|
".rs": "rust", ".go": "go", ".c": "c", ".h": "c", ".cpp": "cpp",
|
|
".js": "javascript", ".ts": "typescript", ".tf": "hcl", ".nix": "nix",
|
|
}
|
|
|
|
SUPPORTED_EXTENSIONS = frozenset(
|
|
{".pdf", ".epub", ".md", ".markdown", ".txt"}
|
|
| set(CODE_LANGUAGES)
|
|
)
|
|
|
|
|
|
class ExtractionError(Exception):
|
|
"""Raised when a document cannot be parsed (missing dep, bad file)."""
|
|
|
|
|
|
def scan_files(
|
|
input_dir: Path,
|
|
extensions: frozenset[str] | set[str] = SUPPORTED_EXTENSIONS,
|
|
) -> list[Path]:
|
|
"""Recursively list supported documents under *input_dir*, sorted
|
|
for deterministic queue order."""
|
|
return sorted(
|
|
(p for p in input_dir.rglob("*")
|
|
if p.is_file() and p.suffix.lower() in extensions),
|
|
key=lambda p: str(p).lower(),
|
|
)
|
|
|
|
|
|
def _title_from_stem(filepath: Path) -> str:
|
|
return filepath.stem.replace("-", " ").replace("_", " ").title()
|
|
|
|
|
|
class DocumentExtractor:
|
|
"""Extracts raw text and structural headers from supported formats."""
|
|
|
|
@staticmethod
|
|
def extract(filepath: Path) -> tuple[str, str]:
|
|
"""Dispatch on extension via the runner table. Raises
|
|
ExtractionError for unsupported types or missing parser
|
|
dependencies."""
|
|
runner = _EXTENSION_RUNNERS.get(filepath.suffix.lower())
|
|
if runner is None:
|
|
raise ExtractionError(f"unsupported file type: {filepath.suffix}")
|
|
return runner(filepath)
|
|
|
|
@staticmethod
|
|
def extract_md_txt(filepath: Path) -> tuple[str, str]:
|
|
text = filepath.read_text(encoding="utf-8", errors="ignore")
|
|
return _title_from_stem(filepath), text
|
|
|
|
@staticmethod
|
|
def extract_code(filepath: Path) -> tuple[str, str]:
|
|
"""Source/config file: verbatim content wrapped in a
|
|
language-tagged fence so the chunker treats it as one code
|
|
document and the vault note renders it as a listing."""
|
|
lang = CODE_LANGUAGES.get(filepath.suffix.lower(), "")
|
|
body = filepath.read_text(encoding="utf-8", errors="ignore")
|
|
return _title_from_stem(filepath), f"```{lang}\n{body}\n```"
|
|
|
|
@staticmethod
|
|
def extract_pdf(filepath: Path) -> tuple[str, str]:
|
|
try:
|
|
from pypdf import PdfReader
|
|
except ImportError as e:
|
|
raise ExtractionError(
|
|
"pypdf not installed — run: pip install 'thicket[ingest]'"
|
|
) from e
|
|
|
|
reader = PdfReader(str(filepath))
|
|
title = _title_from_stem(filepath)
|
|
|
|
# Prefer the embedded metadata title when present.
|
|
try:
|
|
if reader.metadata and reader.metadata.title:
|
|
title = str(reader.metadata.title).strip() or title
|
|
except Exception:
|
|
pass # malformed metadata must not kill the extract
|
|
|
|
pages_text: list[str] = []
|
|
for i, page in enumerate(reader.pages):
|
|
try:
|
|
txt = page.extract_text() or ""
|
|
except Exception:
|
|
txt = ""
|
|
if txt.strip():
|
|
pages_text.append(f"## Page {i + 1}\n\n{txt}")
|
|
|
|
return title, "\n\n".join(pages_text)
|
|
|
|
@staticmethod
|
|
def extract_epub(filepath: Path) -> tuple[str, str]:
|
|
try:
|
|
import bs4
|
|
from ebooklib import epub, ITEM_DOCUMENT
|
|
except ImportError as e:
|
|
raise ExtractionError(
|
|
"ebooklib / beautifulsoup4 not installed — run: "
|
|
"pip install 'thicket[ingest]'"
|
|
) from e
|
|
|
|
book = epub.read_epub(str(filepath))
|
|
title = _title_from_stem(filepath)
|
|
|
|
meta_titles = book.get_metadata("DC", "title")
|
|
if meta_titles:
|
|
title = str(meta_titles[0][0]).strip() or title
|
|
|
|
chapters: list[str] = []
|
|
for item in book.get_items_of_type(ITEM_DOCUMENT):
|
|
soup = bs4.BeautifulSoup(item.get_content(), "html.parser")
|
|
text = soup.get_text(separator="\n")
|
|
clean_text = re.sub(r"\n+", "\n", text).strip()
|
|
if clean_text:
|
|
chapters.append(clean_text)
|
|
|
|
return title, "\n\n---\n\n".join(chapters)
|
|
|
|
# Extension dispatch table — the single source of routing for extract().
|
|
# Adding a format means adding one entry here and its extractor method.
|
|
_EXTENSION_RUNNERS: dict[str, object] = {
|
|
".md": DocumentExtractor.extract_md_txt,
|
|
".markdown": DocumentExtractor.extract_md_txt,
|
|
".txt": DocumentExtractor.extract_md_txt,
|
|
".pdf": DocumentExtractor.extract_pdf,
|
|
".epub": DocumentExtractor.extract_epub,
|
|
**{ext: DocumentExtractor.extract_code for ext in CODE_LANGUAGES},
|
|
}
|