48 lines
1.5 KiB
Python
48 lines
1.5 KiB
Python
"""DocumentExtractor — dispatch, md/txt extraction, scanning."""
|
|
|
|
from __future__ import annotations
|
|
|
|
import pytest
|
|
|
|
from thicket.extractors import (
|
|
ExtractionError, DocumentExtractor, scan_files,
|
|
)
|
|
|
|
|
|
def test_md_extraction_title_from_stem(sample_docs):
|
|
title, text = DocumentExtractor.extract(sample_docs / "deep-learning-notes.md")
|
|
assert title == "Deep Learning Notes"
|
|
assert "# Deep Learning" in text
|
|
assert "## Optimizers" in text
|
|
|
|
|
|
def test_txt_extraction(sample_docs):
|
|
title, text = DocumentExtractor.extract(sample_docs / "reading-list.txt")
|
|
assert title == "Reading List"
|
|
assert "Pragmatic Programmer" in text
|
|
|
|
|
|
def test_unsupported_extension_raises(tmp_path):
|
|
bogus = tmp_path / "photo.jpg"
|
|
bogus.write_bytes(b"\xff\xd8fake")
|
|
with pytest.raises(ExtractionError):
|
|
DocumentExtractor.extract(bogus)
|
|
|
|
|
|
def test_scan_files_sorted_and_filtered(sample_docs, tmp_path):
|
|
(sample_docs / "nested").mkdir()
|
|
(sample_docs / "nested" / "zz-last.md").write_text("x", encoding="utf-8")
|
|
(sample_docs / "image.png").write_bytes(b"\x89PNG") # ignored
|
|
|
|
files = scan_files(sample_docs)
|
|
names = [f.name for f in files]
|
|
assert "image.png" not in names
|
|
# Sorted by full path (case-insensitive), so the nested file sits
|
|
# between the top-level entries alphabetically.
|
|
assert names == ["deep-learning-notes.md", "zz-last.md", "reading-list.txt"]
|
|
|
|
|
|
def test_scan_files_custom_extensions(sample_docs):
|
|
files = scan_files(sample_docs, {".txt"})
|
|
assert [f.name for f in files] == ["reading-list.txt"]
|