"""Tests for arborist.embed — the supported embedding surface. These pin the contract a downstream Python app (e.g. neopig) depends on: open a store, ingest Documents, get content-level dedup, search them. """ from __future__ import annotations from arborist.embed import ( Document, Edge, Source, open_store, ingest_documents, search, ) def _doc(uri, content, title=None, source_type="embedded"): return Document(uri=uri, content=content, source_type=source_type, title=title) def test_open_store_creates_schema(tmp_path): conn = open_store(tmp_path / "embed.db") try: # A fresh store has the documents table (schema migrated on open). names = {r[0] for r in conn.execute("SELECT name FROM sqlite_master WHERE type='table'")} assert "documents" in names assert "audit_events" in names finally: conn.close() def test_ingest_documents_lands_and_is_searchable(tmp_path): conn = open_store(tmp_path / "embed.db") try: stats = ingest_documents( conn, [ _doc("https://x.com/a", "the quick brown fox jumps", title="A"), _doc("https://x.com/b", "lorem ipsum dolor sit amet", title="B"), ], ) assert stats.inserted == 2 n = conn.execute("SELECT COUNT(*) FROM documents").fetchone()[0] assert n == 2 hits = search(conn, "brown fox", limit=10) assert any("x.com/a" in h.document_uri for h in hits), hits finally: conn.close() def test_ingest_is_idempotent_on_content(tmp_path): """Same content → same document_root → no-op re-insert (dedup).""" conn = open_store(tmp_path / "embed.db") try: ingest_documents(conn, [_doc("https://x.com/a", "identical body text here")]) first = conn.execute("SELECT COUNT(*) FROM documents").fetchone()[0] # Re-ingest the same content (same uri) — must not duplicate. stats = ingest_documents(conn, [_doc("https://x.com/a", "identical body text here")]) second = conn.execute("SELECT COUNT(*) FROM documents").fetchone()[0] assert first == second == 1 assert stats.inserted == 0 finally: conn.close() def test_dedup_is_content_not_uri(tmp_path): """Two different URIs with identical content share one document_root — the property an embedder leans on for cross-source dedup.""" conn = open_store(tmp_path / "embed.db") try: ingest_documents( conn, [ _doc("https://x.com/a", "byte identical content"), _doc("https://y.com/mirror", "byte identical content"), ], ) roots = [r[0] for r in conn.execute("SELECT DISTINCT document_root FROM documents")] assert len(roots) == 1, "identical content must collapse to one root" finally: conn.close() def test_edges_survive_ingest(tmp_path): conn = open_store(tmp_path / "embed.db") try: page = Document( uri="https://x.com/page", content="a page that links to an image", source_type="neopig_html", title="Page", edges=[Edge(edge_type="embeds_media", dst_uri="https://x.com/img.jpg")], ) ingest_documents(conn, [page]) n_edges = conn.execute("SELECT COUNT(*) FROM edges").fetchone()[0] assert n_edges >= 1 finally: conn.close() def test_iterable_source_is_a_source_subclass(): """The embed wrapper must satisfy the Source contract so it flows through the same ingest pipeline as any first-class corpus.""" from arborist.embed import _IterableSource src = _IterableSource([], source_type="embedded") assert isinstance(src, Source) assert src.source_type == "embedded"