"""DocumentExtractor — dispatch, md/txt extraction, scanning.""" from __future__ import annotations import pytest from thicket.extractors import ( ExtractionError, DocumentExtractor, scan_files, ) def test_md_extraction_title_from_stem(sample_docs): title, text = DocumentExtractor.extract(sample_docs / "deep-learning-notes.md") assert title == "Deep Learning Notes" assert "# Deep Learning" in text assert "## Optimizers" in text def test_txt_extraction(sample_docs): title, text = DocumentExtractor.extract(sample_docs / "reading-list.txt") assert title == "Reading List" assert "Pragmatic Programmer" in text def test_unsupported_extension_raises(tmp_path): bogus = tmp_path / "photo.jpg" bogus.write_bytes(b"\xff\xd8fake") with pytest.raises(ExtractionError): DocumentExtractor.extract(bogus) def test_scan_files_sorted_and_filtered(sample_docs, tmp_path): (sample_docs / "nested").mkdir() (sample_docs / "nested" / "zz-last.md").write_text("x", encoding="utf-8") (sample_docs / "image.png").write_bytes(b"\x89PNG") # ignored files = scan_files(sample_docs) names = [f.name for f in files] assert "image.png" not in names # Sorted by full path (case-insensitive), so the nested file sits # between the top-level entries alphabetically. assert names == ["deep-learning-notes.md", "zz-last.md", "reading-list.txt"] def test_scan_files_custom_extensions(sample_docs): files = scan_files(sample_docs, {".txt"}) assert [f.name for f in files] == ["reading-list.txt"]