"""Document extraction — PDF / EPUB / Markdown / plain text. Extracts ``(title, text)`` from every supported format. Structural headers are preserved in the extracted text (PDF pages become ``## Page N`` headings, EPUB chapters are joined with rules) so the chunker can attribute each chunk to its section. All third-party parsers are imported lazily inside the extract methods: the GUI must launch and probe on a bare system where EbookLib or pypdf are not installed, and fail with an actionable message only when the matching file type is actually encountered. """ from __future__ import annotations import re from pathlib import Path # Source files ingested verbatim as language-tagged code blocks. CODE_LANGUAGES: dict[str, str] = { ".py": "python", ".sh": "bash", ".bash": "bash", ".zsh": "zsh", ".yaml": "yaml", ".yml": "yaml", ".toml": "toml", ".ini": "ini", ".conf": "ini", ".cfg": "ini", ".json": "json", ".sql": "sql", ".rs": "rust", ".go": "go", ".c": "c", ".h": "c", ".cpp": "cpp", ".js": "javascript", ".ts": "typescript", ".tf": "hcl", ".nix": "nix", } SUPPORTED_EXTENSIONS = frozenset( {".pdf", ".epub", ".md", ".markdown", ".txt"} | set(CODE_LANGUAGES) ) class ExtractionError(Exception): """Raised when a document cannot be parsed (missing dep, bad file).""" def scan_files( input_dir: Path, extensions: frozenset[str] | set[str] = SUPPORTED_EXTENSIONS, ) -> list[Path]: """Recursively list supported documents under *input_dir*, sorted for deterministic queue order.""" return sorted( (p for p in input_dir.rglob("*") if p.is_file() and p.suffix.lower() in extensions), key=lambda p: str(p).lower(), ) def _title_from_stem(filepath: Path) -> str: return filepath.stem.replace("-", " ").replace("_", " ").title() class DocumentExtractor: """Extracts raw text and structural headers from supported formats.""" @staticmethod def extract(filepath: Path) -> tuple[str, str]: """Dispatch on extension via the runner table. Raises ExtractionError for unsupported types or missing parser dependencies.""" runner = _EXTENSION_RUNNERS.get(filepath.suffix.lower()) if runner is None: raise ExtractionError(f"unsupported file type: {filepath.suffix}") return runner(filepath) @staticmethod def extract_md_txt(filepath: Path) -> tuple[str, str]: text = filepath.read_text(encoding="utf-8", errors="ignore") return _title_from_stem(filepath), text @staticmethod def extract_code(filepath: Path) -> tuple[str, str]: """Source/config file: verbatim content wrapped in a language-tagged fence so the chunker treats it as one code document and the vault note renders it as a listing.""" lang = CODE_LANGUAGES.get(filepath.suffix.lower(), "") body = filepath.read_text(encoding="utf-8", errors="ignore") return _title_from_stem(filepath), f"```{lang}\n{body}\n```" @staticmethod def extract_pdf(filepath: Path) -> tuple[str, str]: try: from pypdf import PdfReader except ImportError as e: raise ExtractionError( "pypdf not installed — run: pip install 'thicket[ingest]'" ) from e reader = PdfReader(str(filepath)) title = _title_from_stem(filepath) # Prefer the embedded metadata title when present. try: if reader.metadata and reader.metadata.title: title = str(reader.metadata.title).strip() or title except Exception: pass # malformed metadata must not kill the extract pages_text: list[str] = [] for i, page in enumerate(reader.pages): try: txt = page.extract_text() or "" except Exception: txt = "" if txt.strip(): pages_text.append(f"## Page {i + 1}\n\n{txt}") return title, "\n\n".join(pages_text) @staticmethod def extract_epub(filepath: Path) -> tuple[str, str]: try: import bs4 from ebooklib import epub, ITEM_DOCUMENT except ImportError as e: raise ExtractionError( "ebooklib / beautifulsoup4 not installed — run: " "pip install 'thicket[ingest]'" ) from e book = epub.read_epub(str(filepath)) title = _title_from_stem(filepath) meta_titles = book.get_metadata("DC", "title") if meta_titles: title = str(meta_titles[0][0]).strip() or title chapters: list[str] = [] for item in book.get_items_of_type(ITEM_DOCUMENT): soup = bs4.BeautifulSoup(item.get_content(), "html.parser") text = soup.get_text(separator="\n") clean_text = re.sub(r"\n+", "\n", text).strip() if clean_text: chapters.append(clean_text) return title, "\n\n---\n\n".join(chapters) # Extension dispatch table — the single source of routing for extract(). # Adding a format means adding one entry here and its extractor method. _EXTENSION_RUNNERS: dict[str, object] = { ".md": DocumentExtractor.extract_md_txt, ".markdown": DocumentExtractor.extract_md_txt, ".txt": DocumentExtractor.extract_md_txt, ".pdf": DocumentExtractor.extract_pdf, ".epub": DocumentExtractor.extract_epub, **{ext: DocumentExtractor.extract_code for ext in CODE_LANGUAGES}, }