182 lines
6.1 KiB
Python
182 lines
6.1 KiB
Python
"""Vector store targets — registry, doc_key contract, and per-target
|
|
roundtrips against a deterministic stub engine (no model download).
|
|
|
|
The per-target roundtrips are skipped automatically when a target's
|
|
library is not installed; the registry tests always run.
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
import pytest
|
|
|
|
from thicket.chunker import Chunk
|
|
from thicket.vector_stores import (
|
|
TARGETS, VectorStoreError, create_store, doc_key,
|
|
)
|
|
|
|
CHUNKS = [
|
|
Chunk(header="Alpha", text="alpha content one",
|
|
contextual_text="[Source: T | Section: Alpha]\nalpha content one"),
|
|
Chunk(header="Beta", text="beta content two",
|
|
contextual_text="[Source: T | Section: Beta]\nbeta content two"),
|
|
]
|
|
|
|
|
|
class StubEngine:
|
|
"""Deterministic 4-dim embeddings: direction keyed by first word."""
|
|
model_name = "stub-model"
|
|
dim = 4
|
|
|
|
_BASIS = {
|
|
"alpha": [1.0, 0.0, 0.0, 0.0],
|
|
"beta": [0.0, 1.0, 0.0, 0.0],
|
|
}
|
|
|
|
def embed(self, texts: list[str]) -> list[list[float]]:
|
|
return [self._vector(t) for t in texts]
|
|
|
|
def embed_query(self, text: str) -> list[float]:
|
|
return self._vector(text)
|
|
|
|
@classmethod
|
|
def _vector(cls, text: str) -> list[float]:
|
|
for word, vec in cls._BASIS.items():
|
|
if word in text.lower():
|
|
return list(vec)
|
|
return [0.0, 0.0, 1.0, 0.0]
|
|
|
|
|
|
def _roundtrip(store) -> None:
|
|
store.set_embedder(StubEngine())
|
|
store.ensure_collection()
|
|
|
|
written = store.replace_document("Doc One", "Ingested_Brain/doc-one.md", CHUNKS)
|
|
assert written == 2
|
|
|
|
hits = store.search(StubEngine().embed_query("alpha query"), limit=2)
|
|
assert hits, "expected at least one hit"
|
|
top = hits[0]
|
|
assert top["payload"]["document_title"] == "Doc One"
|
|
assert top["payload"]["section_header"] == "Alpha"
|
|
assert "doc_key" not in top["payload"] # internal key never surfaces
|
|
|
|
# Exact replacement: shrink to one chunk, count must follow exactly.
|
|
store.replace_document("Doc One", "Ingested_Brain/doc-one.md", CHUNKS[:1])
|
|
hits_after = store.search(StubEngine().embed_query("alpha query"), limit=10)
|
|
alpha_hits = [h for h in hits_after
|
|
if h["payload"]["section_header"] == "Alpha"]
|
|
beta_hits = [h for h in hits_after
|
|
if h["payload"]["section_header"] == "Beta"]
|
|
assert len(alpha_hits) == 1 and not beta_hits, \
|
|
"re-ingest must replace exactly, leaving no stale chunks"
|
|
|
|
|
|
# ── registry (always runs) ──
|
|
|
|
def test_registry_covers_ten_targets():
|
|
assert sorted(TARGETS) == [
|
|
"chroma", "duckdb", "faiss", "lancedb", "mariadb", "milvus",
|
|
"pgvector", "qdrant", "sqlitevec", "weaviate",
|
|
]
|
|
|
|
|
|
def test_service_requirements_are_exact():
|
|
services = {key: spec.service for key, spec in TARGETS.items()}
|
|
assert services == {
|
|
"qdrant": "qdrant", "pgvector": "postgres",
|
|
"weaviate": "weaviate", "mariadb": "mariadb",
|
|
"chroma": None, "lancedb": None, "faiss": None, "milvus": None,
|
|
"duckdb": None, "sqlitevec": None,
|
|
}
|
|
|
|
|
|
def test_identifier_mappers_never_emit_raw_names():
|
|
from thicket.vector_stores import _mariadb_ident, _weaviate_name
|
|
assert _mariadb_ident("second-brain; DROP TABLE x") == "second_brain__DROP_TABLE_x"
|
|
assert _weaviate_name("second_brain") == "Second_brain"
|
|
assert _mariadb_ident("") == "thicket"
|
|
|
|
|
|
def test_doc_key_is_stable_and_injective():
|
|
a = doc_key("Title", "path/one.md")
|
|
assert a == doc_key("Title", "path/one.md")
|
|
assert a != doc_key("Title", "path/two.md")
|
|
assert a != doc_key("Other", "path/one.md")
|
|
|
|
|
|
def test_unknown_target_names_every_choice():
|
|
with pytest.raises(VectorStoreError, match="known:"):
|
|
create_store("vespa", collection="x", dim=4)
|
|
|
|
|
|
def test_missing_module_error_names_the_fix():
|
|
from thicket.vector_stores import BaseVectorStore
|
|
base = BaseVectorStore("c", 4)
|
|
with pytest.raises(VectorStoreError, match="pip install"):
|
|
base._require_module("definitely_not_a_module_xyz")
|
|
|
|
|
|
# ── per-target roundtrips (skip when library absent) ──
|
|
|
|
def test_qdrant_roundtrip_needs_service():
|
|
assert TARGETS["qdrant"].service == "qdrant"
|
|
|
|
|
|
def test_chroma_roundtrip(tmp_path):
|
|
pytest.importorskip("chromadb")
|
|
from thicket.vector_stores import ChromaStore
|
|
_roundtrip(ChromaStore(data_dir=tmp_path / "data", collection="test_col", dim=4))
|
|
|
|
|
|
def test_lancedb_roundtrip(tmp_path):
|
|
pytest.importorskip("lancedb")
|
|
from thicket.vector_stores import LanceStore
|
|
_roundtrip(LanceStore(data_dir=tmp_path, collection="t", dim=4))
|
|
|
|
|
|
def test_faiss_roundtrip(tmp_path):
|
|
pytest.importorskip("faiss")
|
|
from thicket.vector_stores import FaissStore
|
|
_roundtrip(FaissStore(data_dir=tmp_path, collection="t", dim=4))
|
|
|
|
|
|
def test_milvus_roundtrip(tmp_path):
|
|
pytest.importorskip("pymilvus")
|
|
from thicket.vector_stores import MilvusStore
|
|
_roundtrip(MilvusStore(data_dir=tmp_path / "data", collection="test_col", dim=4))
|
|
|
|
|
|
def test_duckdb_roundtrip(tmp_path):
|
|
pytest.importorskip("duckdb")
|
|
from thicket.vector_stores import DuckStore
|
|
_roundtrip(DuckStore(data_dir=tmp_path / "data", collection="test_col", dim=4))
|
|
|
|
|
|
def test_sqlitevec_roundtrip(tmp_path):
|
|
pytest.importorskip("sqlite_vec")
|
|
from thicket.vector_stores import SqliteVecStore
|
|
_roundtrip(SqliteVecStore(data_dir=tmp_path / "data", collection="test_col", dim=4))
|
|
|
|
|
|
def test_weaviate_name_mapping():
|
|
from thicket.vector_stores import _weaviate_name
|
|
assert _weaviate_name("second_brain") == "Second_brain"
|
|
assert _weaviate_name("my-papers 2") == "My_papers_2"
|
|
|
|
|
|
def test_payload_carries_code_provenance():
|
|
from thicket.chunker import Chunk
|
|
from thicket.vector_stores import _payload
|
|
|
|
chunk = Chunk("H", "def x(): pass", "[S|H]\ndef x(): pass",
|
|
kind="code", lang="python", source="pkg/mod.py")
|
|
payload = _payload("Doc", "notes/doc.md", chunk, 0)
|
|
assert payload["chunk_kind"] == "code"
|
|
assert payload["lang"] == "python"
|
|
assert payload["source_path"] == "pkg/mod.py"
|
|
assert payload["doc_key"]
|
|
|
|
plain = Chunk("H", "prose", "[S|H]\nprose")
|
|
payload = _payload("Doc", "notes/doc.md", plain, 0)
|
|
assert "lang" not in payload and "source_path" not in payload
|