Thicket/tests/test_extractors.py

48 lines
1.5 KiB
Python

"""DocumentExtractor — dispatch, md/txt extraction, scanning."""
from __future__ import annotations
import pytest
from thicket.extractors import (
ExtractionError, DocumentExtractor, scan_files,
)
def test_md_extraction_title_from_stem(sample_docs):
title, text = DocumentExtractor.extract(sample_docs / "deep-learning-notes.md")
assert title == "Deep Learning Notes"
assert "# Deep Learning" in text
assert "## Optimizers" in text
def test_txt_extraction(sample_docs):
title, text = DocumentExtractor.extract(sample_docs / "reading-list.txt")
assert title == "Reading List"
assert "Pragmatic Programmer" in text
def test_unsupported_extension_raises(tmp_path):
bogus = tmp_path / "photo.jpg"
bogus.write_bytes(b"\xff\xd8fake")
with pytest.raises(ExtractionError):
DocumentExtractor.extract(bogus)
def test_scan_files_sorted_and_filtered(sample_docs, tmp_path):
(sample_docs / "nested").mkdir()
(sample_docs / "nested" / "zz-last.md").write_text("x", encoding="utf-8")
(sample_docs / "image.png").write_bytes(b"\x89PNG") # ignored
files = scan_files(sample_docs)
names = [f.name for f in files]
assert "image.png" not in names
# Sorted by full path (case-insensitive), so the nested file sits
# between the top-level entries alphabetically.
assert names == ["deep-learning-notes.md", "zz-last.md", "reading-list.txt"]
def test_scan_files_custom_extensions(sample_docs):
files = scan_files(sample_docs, {".txt"})
assert [f.name for f in files] == ["reading-list.txt"]