crawl central-db + query auto-include + read-seam provenance
- make crawl-ingest writes to one central crawl db (CRAWL_DB, default ~/.arborist/crawl/web.db) instead of per-domain shards in the peer-shared main dir: keeps locally-crawled content out of peer sharing by default and a growing domain set under SQLite's 10-attach cap (Makefile, docs/crawler.md). - arborist query auto-includes the local crawl db (query() gains extra_shards; CLI --include-shard / --no-crawl-db, default-on when web.db exists). Fix latent --db single-file query AttributeError (cli.py). Persist used / used_pointer_ids + retrieval_purity into merkle_proof so read-only consumers can see which chunks fed the answer (qa/query.py). - arborist.read: read-only seam for dashboards / verifiers; on a multi-source context root surface the real primary source instead of the opaque corpus://multi-source sentinel (read.py). Backs the arborist-viz Merkle Command Center (#000069). - tests for extra_shards, the CLI crawl-db resolver, and the read seam.
This commit is contained in:
parent
6eada6dc89
commit
7f7eeefeb9
10 changed files with 3460 additions and 20 deletions
|
|
@ -383,3 +383,40 @@ def test_render_label_format_collapsed_surfaces_tail():
|
|||
out = _render_query_human(r, "winners of all major sports?")
|
||||
assert "format collapsed" in out
|
||||
assert "UNGROUNDED" in out
|
||||
|
||||
|
||||
# --- auto-include of the local crawl db -----------------------------
|
||||
|
||||
|
||||
def test_resolve_extra_shards_auto_includes_crawl_db(tmp_path, monkeypatch):
|
||||
"""`arborist query` auto-includes the local crawl db (when present),
|
||||
honors --no-crawl-db, and appends --include-shard extras de-duped."""
|
||||
import argparse
|
||||
|
||||
from arborist.cli import _default_crawl_db, _resolve_extra_shards
|
||||
|
||||
crawl_db = tmp_path / "web.db"
|
||||
crawl_db.write_bytes(b"") # resolver only checks presence
|
||||
monkeypatch.setenv("ARBORIST_CRAWL_DB", str(crawl_db))
|
||||
assert _default_crawl_db() == crawl_db
|
||||
|
||||
# present → auto-included
|
||||
ns = argparse.Namespace(no_crawl_db=False, include_shard=None)
|
||||
assert _resolve_extra_shards(ns) == [crawl_db]
|
||||
|
||||
# --no-crawl-db opts out
|
||||
ns_off = argparse.Namespace(no_crawl_db=True, include_shard=None)
|
||||
assert _resolve_extra_shards(ns_off) == []
|
||||
|
||||
# explicit --include-shard appends extras; crawl db deduped
|
||||
extra = tmp_path / "extra.db"
|
||||
extra.write_bytes(b"")
|
||||
ns_inc = argparse.Namespace(
|
||||
no_crawl_db=False, include_shard=[str(extra), str(crawl_db)]
|
||||
)
|
||||
assert _resolve_extra_shards(ns_inc) == [crawl_db, extra]
|
||||
|
||||
# missing crawl db → silently dropped
|
||||
monkeypatch.setenv("ARBORIST_CRAWL_DB", str(tmp_path / "nope.db"))
|
||||
ns_missing = argparse.Namespace(no_crawl_db=False, include_shard=None)
|
||||
assert _resolve_extra_shards(ns_missing) == []
|
||||
|
|
|
|||
|
|
@ -1424,3 +1424,57 @@ def test_integration_hyphenated_query_retrieves_joined_title(tmp_path):
|
|||
assert any("bipolar-disorder" in u for u in src_uris), (
|
||||
f"Ticket #000007 regression: bipolar-disorder not in {src_uris}"
|
||||
)
|
||||
|
||||
|
||||
def test_query_extra_shards_pulls_in_separate_db(tmp_path):
|
||||
"""`extra_shards` searches a separate db alongside the main corpus —
|
||||
the seam the CLI uses to auto-include the local crawl db without
|
||||
folding it into the peer-shared main shards."""
|
||||
main_db = tmp_path / "corpus.db"
|
||||
crawl_db = tmp_path / "web.db"
|
||||
conn = connect(main_db)
|
||||
try:
|
||||
ingest_source(conn, FakeSource(DOCS))
|
||||
finally:
|
||||
conn.close()
|
||||
crawl_doc = _doc(
|
||||
"test://virtback",
|
||||
"Virtback is a python libvirt backup utility for kvm xen virtualbox. " * 12
|
||||
+ "Virtback restores domains from backups. " * 8,
|
||||
)
|
||||
conn = connect(crawl_db)
|
||||
try:
|
||||
ingest_source(conn, FakeSource([crawl_doc]))
|
||||
finally:
|
||||
conn.close()
|
||||
|
||||
client = StubClient(
|
||||
answer='Per the source: "Virtback is a python libvirt backup utility for kvm xen virtualbox"'
|
||||
)
|
||||
|
||||
# Without extra_shards: the crawl-only topic is absent from the main corpus.
|
||||
r_without = query(
|
||||
question="What is virtback?",
|
||||
qa_db=tmp_path / "qa1.db",
|
||||
chat_client=client,
|
||||
model_id="m",
|
||||
single_db=main_db,
|
||||
top_k=3,
|
||||
)
|
||||
assert all(
|
||||
"virtback" not in s["document_uri"] for s in r_without.get("sources") or []
|
||||
)
|
||||
|
||||
# With extra_shards: the separate crawl db is searched alongside it.
|
||||
r_with = query(
|
||||
question="What is virtback?",
|
||||
qa_db=tmp_path / "qa2.db",
|
||||
chat_client=client,
|
||||
model_id="m",
|
||||
single_db=main_db,
|
||||
extra_shards=[crawl_db],
|
||||
top_k=3,
|
||||
)
|
||||
assert any(
|
||||
"virtback" in s["document_uri"] for s in r_with.get("sources") or []
|
||||
), f"extra_shards crawl db not searched: {[s['document_uri'] for s in r_with.get('sources') or []]}"
|
||||
|
|
|
|||
395
tests/test_read.py
Normal file
395
tests/test_read.py
Normal file
|
|
@ -0,0 +1,395 @@
|
|||
"""Tests for arborist.read — the supported read-only seam.
|
||||
|
||||
These pin the contract that downstream read-only consumers
|
||||
(arborist-viz / Merkle Command Center, third-party verifiers, archival
|
||||
mirrors) depend on: open shards, query roots/leaves/proofs/audit/qa,
|
||||
resolve any hex hash.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import sqlite3
|
||||
|
||||
import pytest
|
||||
|
||||
from arborist.embed import Document, ingest_documents, open_store
|
||||
from arborist.read import (
|
||||
open_shards,
|
||||
Root,
|
||||
Leaf,
|
||||
Proof,
|
||||
AuditEvent,
|
||||
QaRecord,
|
||||
ResolveResult,
|
||||
ShardCounts,
|
||||
)
|
||||
|
||||
|
||||
# ---------- fixtures ----------------------------------------------------
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def shard_path(tmp_path):
|
||||
p = str(tmp_path / "shard.db")
|
||||
conn = open_store(p)
|
||||
ingest_documents(
|
||||
conn,
|
||||
[
|
||||
Document(
|
||||
uri="https://test.local/doc1",
|
||||
title="Doc 1",
|
||||
source_type="test",
|
||||
content="One sentence. Two sentence. Three sentence. Four sentence. Five.",
|
||||
),
|
||||
Document(
|
||||
uri="https://test.local/doc2",
|
||||
title="Doc 2",
|
||||
source_type="test",
|
||||
content="Alpha beta gamma delta epsilon.",
|
||||
),
|
||||
],
|
||||
source_type="test",
|
||||
chunker_name="sent-v1",
|
||||
)
|
||||
conn.close()
|
||||
return p
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def shards(shard_path):
|
||||
s = open_shards([shard_path])
|
||||
yield s
|
||||
s.close()
|
||||
|
||||
|
||||
def _root_of(shards) -> str:
|
||||
return shards.roots(limit=10)[0].document_root
|
||||
|
||||
|
||||
# ---------- handle ------------------------------------------------------
|
||||
|
||||
|
||||
def test_open_skips_broken_shards(tmp_path):
|
||||
s = open_shards([str(tmp_path / "missing.db")])
|
||||
# Broken / missing shard is skipped, handle is still usable.
|
||||
assert s.paths == [str(tmp_path / "missing.db")] or s.paths == []
|
||||
|
||||
|
||||
def test_counts(shards):
|
||||
c = shards.counts()
|
||||
assert isinstance(c, ShardCounts)
|
||||
assert c.documents == 2
|
||||
assert c.shard_count == 1
|
||||
assert c.audit_events >= 2 # at least one ingest event per doc
|
||||
|
||||
|
||||
# ---------- roots / leaves ---------------------------------------------
|
||||
|
||||
|
||||
def test_roots_listing(shards):
|
||||
rs = shards.roots(limit=10)
|
||||
assert len(rs) == 2
|
||||
titles = {r.title for r in rs}
|
||||
assert {"Doc 1", "Doc 2"} == titles
|
||||
for r in rs:
|
||||
assert isinstance(r, Root)
|
||||
assert r.shard_path # source-of-record citation
|
||||
|
||||
|
||||
def test_root_lookup(shards):
|
||||
root_hash = _root_of(shards)
|
||||
r = shards.root(root_hash)
|
||||
assert r is not None
|
||||
assert r.document_root == root_hash
|
||||
assert r.leaf_count >= 1
|
||||
|
||||
|
||||
def test_root_not_found(shards):
|
||||
assert shards.root("0" * 64) is None
|
||||
|
||||
|
||||
def test_leaves_privacy_default_deny(shards):
|
||||
"""Default: bytes-under-the-hash NOT decompressed (§14)."""
|
||||
root_hash = _root_of(shards)
|
||||
leaves = shards.leaves(root_hash)
|
||||
assert leaves, "shard must have at least one chunk"
|
||||
for L in leaves:
|
||||
assert isinstance(L, Leaf)
|
||||
assert L.content is None
|
||||
assert L.prose is None
|
||||
|
||||
|
||||
def test_leaves_reveal_returns_content(shards):
|
||||
root_hash = _root_of(shards)
|
||||
leaves = shards.leaves(root_hash, reveal_private=True)
|
||||
assert any(L.content for L in leaves), "reveal must surface chunk content"
|
||||
|
||||
|
||||
def test_leaves_project_off(shards):
|
||||
root_hash = _root_of(shards)
|
||||
leaves = shards.leaves(root_hash, reveal_private=True, project=False)
|
||||
for L in leaves:
|
||||
assert L.prose is None
|
||||
assert L.base_version is None
|
||||
|
||||
|
||||
# ---------- proof / tree ------------------------------------------------
|
||||
|
||||
|
||||
def test_proof_passes(shards):
|
||||
root_hash = _root_of(shards)
|
||||
p = shards.proof(root_hash, 0)
|
||||
assert isinstance(p, Proof)
|
||||
assert p.passed is True
|
||||
assert p.computed_root == p.expected_root == root_hash
|
||||
|
||||
|
||||
def test_proof_out_of_range(shards):
|
||||
root_hash = _root_of(shards)
|
||||
with pytest.raises(IndexError):
|
||||
shards.proof(root_hash, 9999)
|
||||
|
||||
|
||||
def test_proof_unknown_root(shards):
|
||||
assert shards.proof("0" * 64, 0) is None
|
||||
|
||||
|
||||
def test_tree_layers(shards):
|
||||
root_hash = _root_of(shards)
|
||||
layers = shards.tree_layers(root_hash)
|
||||
assert layers is not None
|
||||
assert len(layers[-1]) == 1
|
||||
assert layers[-1][0] == root_hash
|
||||
|
||||
|
||||
# ---------- audit -------------------------------------------------------
|
||||
|
||||
|
||||
def test_audit_recent(shards):
|
||||
evs = shards.audit_recent(limit=10)
|
||||
assert evs, "ingest writes at least one audit event per document"
|
||||
for ev in evs:
|
||||
assert isinstance(ev, AuditEvent)
|
||||
assert ev.shard_path
|
||||
|
||||
|
||||
def test_audit_event_lookup(shards):
|
||||
ev = shards.audit_recent(limit=1)[0]
|
||||
assert shards.audit_event(ev.event_hash) == ev
|
||||
|
||||
|
||||
def test_audit_by_root(shards):
|
||||
root_hash = _root_of(shards)
|
||||
evs = shards.audit_by_root(root_hash)
|
||||
assert all(ev.subject_root == root_hash for ev in evs)
|
||||
|
||||
|
||||
def test_audit_chain_walks_prev(shards):
|
||||
head = shards.audit_recent(limit=1)[0]
|
||||
chain = shards.audit_chain(head.event_hash, limit=10)
|
||||
assert chain[0].event_hash == head.event_hash
|
||||
|
||||
|
||||
def test_audit_since_yields_then_empties(shards):
|
||||
# First call (empty cursor) returns every event.
|
||||
events, cur = shards.audit_since(None)
|
||||
assert events
|
||||
# Second call with the cursor returns nothing new.
|
||||
again, _ = shards.audit_since(cur)
|
||||
assert again == []
|
||||
|
||||
|
||||
def test_audit_cursor_is_head(shards):
|
||||
cur = shards.audit_cursor()
|
||||
assert all(seq >= 1 for seq in cur.values())
|
||||
|
||||
|
||||
# ---------- hash resolver ----------------------------------------------
|
||||
|
||||
|
||||
def test_resolve_document_root(shards):
|
||||
root_hash = _root_of(shards)
|
||||
r = shards.resolve(root_hash)
|
||||
assert r.kind == "document_root"
|
||||
assert r.hash == root_hash
|
||||
assert r.shard_path
|
||||
|
||||
|
||||
def test_resolve_leaf_hash(shards):
|
||||
root_hash = _root_of(shards)
|
||||
leaf = shards.leaves(root_hash)[0]
|
||||
r = shards.resolve(leaf.leaf_hash)
|
||||
assert r.kind == "leaf_hash"
|
||||
assert r.extra["document_root"] == root_hash
|
||||
assert r.extra["leaf_index"] == leaf.idx
|
||||
|
||||
|
||||
def test_resolve_audit_event(shards):
|
||||
ev = shards.audit_recent(limit=1)[0]
|
||||
r = shards.resolve(ev.event_hash)
|
||||
assert r.kind == "audit_event"
|
||||
|
||||
|
||||
def test_resolve_unknown(shards):
|
||||
assert shards.resolve("0" * 64).kind == "unknown"
|
||||
|
||||
|
||||
# ---------- multi-shard fan-out ----------------------------------------
|
||||
|
||||
|
||||
def test_multi_shard_resolve_and_count(tmp_path):
|
||||
"""Fan-out: roots / counts / resolve all aggregate across shards."""
|
||||
paths = []
|
||||
for i, content in enumerate(["First doc one.", "Second doc two.", "Third doc three."]):
|
||||
p = str(tmp_path / f"s{i}.db")
|
||||
c = open_store(p)
|
||||
ingest_documents(
|
||||
c,
|
||||
[Document(uri=f"test://{i}", content=content, source_type="test")],
|
||||
source_type="test",
|
||||
)
|
||||
c.close()
|
||||
paths.append(p)
|
||||
|
||||
s = open_shards(paths)
|
||||
try:
|
||||
assert s.counts().documents == 3
|
||||
assert s.counts().shard_count == 3
|
||||
roots = s.roots(limit=10)
|
||||
assert len(roots) == 3
|
||||
# Each root's shard_path matches one of the three configured paths.
|
||||
assert {r.shard_path for r in roots} == set(paths)
|
||||
# Resolving each root identifies its owning shard.
|
||||
for r in roots:
|
||||
res = s.resolve(r.document_root)
|
||||
assert res.kind == "document_root"
|
||||
assert res.shard_path == r.shard_path
|
||||
finally:
|
||||
s.close()
|
||||
|
||||
|
||||
# ---------- providence_cache (Q&A) -------------------------------------
|
||||
|
||||
|
||||
def _seed_qa(path: str, *, cache_key: str, source_root: str, question: str, answer: str):
|
||||
"""Manually insert a providence_cache row — arborist's qa.runner
|
||||
would normally do this; we synthesize a row so the test doesn't need
|
||||
a live LLM."""
|
||||
c = sqlite3.connect(path)
|
||||
c.execute(
|
||||
"INSERT OR REPLACE INTO providence_cache ("
|
||||
" cache_key, source_root, document_uri, question_hash, question_text, "
|
||||
" answer_text, merkle_proof, model_profile_hash, conversation_hash, "
|
||||
" governance_policy_hash, schema_version, canonicalization_version, "
|
||||
" chunking_version, falsification_state, created_at, hit_count"
|
||||
") VALUES (?,?,?,?,?,?,?,?,?,?,?,?,?,?,?,?)",
|
||||
(
|
||||
cache_key, source_root, "test://doc", "q" * 64, question, answer,
|
||||
"{}", "m" * 64, "c" * 64, "g" * 64,
|
||||
"v9.8.0", "norm-v1", "tok-512-v1", "live", 1, 1,
|
||||
),
|
||||
)
|
||||
c.commit()
|
||||
c.close()
|
||||
|
||||
|
||||
def test_qa_lookup_and_search(shard_path):
|
||||
_seed_qa(
|
||||
shard_path,
|
||||
cache_key="a" * 64,
|
||||
source_root="b" * 64,
|
||||
question="who wrote the song?",
|
||||
answer="Joey Tempest.",
|
||||
)
|
||||
s = open_shards([shard_path])
|
||||
try:
|
||||
rec = s.qa("a" * 64)
|
||||
assert rec is not None
|
||||
assert rec.question_text == "who wrote the song?"
|
||||
assert rec.answer_text == "Joey Tempest."
|
||||
|
||||
hits = s.qa_search("wrote")
|
||||
assert any(r.cache_key == "a" * 64 for r in hits)
|
||||
|
||||
by_root = s.qa_by_root("b" * 64)
|
||||
assert by_root[0].cache_key == "a" * 64
|
||||
|
||||
# cache_key resolves
|
||||
r = s.resolve("a" * 64)
|
||||
assert r.kind == "qa_cache_key"
|
||||
# source_root resolves as a synthetic context_root (no
|
||||
# corresponding row in the documents table).
|
||||
r = s.resolve("b" * 64)
|
||||
assert r.kind == "context_root"
|
||||
assert r.extra.get("cache_key") == "a" * 64
|
||||
|
||||
# And the dedicated context() lookup returns the synthesized root.
|
||||
ctx = s.context("b" * 64)
|
||||
assert ctx is not None
|
||||
assert ctx.context_root == "b" * 64
|
||||
assert ctx.cache_key == "a" * 64
|
||||
assert ctx.question_text == "who wrote the song?"
|
||||
finally:
|
||||
s.close()
|
||||
|
||||
|
||||
def test_summarize_sources_picks_primary_and_domains():
|
||||
"""The provenance helper ranks the primary answer source first and
|
||||
de-dupes contributing domains (primary first)."""
|
||||
from arborist.read import _summarize_sources
|
||||
|
||||
sources = [
|
||||
{"document_uri": "https://en.wikipedia.org/wiki/Virt",
|
||||
"source_role": "background_source"},
|
||||
{"document_uri": "https://russell.ballestrini.net/virt-back-restoring-from-backups/",
|
||||
"source_role": "primary_answer_source"},
|
||||
{"document_uri": "https://russell.ballestrini.net/virt-backs-domfetcher/",
|
||||
"source_role": "primary_answer_source"},
|
||||
]
|
||||
primary, domains = _summarize_sources(sources)
|
||||
assert primary == "https://russell.ballestrini.net/virt-back-restoring-from-backups/"
|
||||
assert domains == ["russell.ballestrini.net", "en.wikipedia.org"]
|
||||
assert _summarize_sources([]) == (None, [])
|
||||
|
||||
|
||||
def test_context_headlines_real_source_not_sentinel(shard_path):
|
||||
"""A multi-source context root stores the opaque
|
||||
``corpus://multi-source`` sentinel as document_uri; the read seam
|
||||
must headline the real primary source so a consumer (dashboard /
|
||||
verifier) shows where the knowledge came from."""
|
||||
import json
|
||||
|
||||
c = sqlite3.connect(shard_path)
|
||||
c.execute(
|
||||
"INSERT OR REPLACE INTO providence_cache ("
|
||||
" cache_key, source_root, document_uri, question_hash, question_text, "
|
||||
" answer_text, merkle_proof, model_profile_hash, conversation_hash, "
|
||||
" governance_policy_hash, schema_version, canonicalization_version, "
|
||||
" chunking_version, falsification_state, created_at, hit_count"
|
||||
") VALUES (?,?,?,?,?,?,?,?,?,?,?,?,?,?,?,?)",
|
||||
(
|
||||
"d" * 64, "e" * 64, "corpus://multi-source", "q" * 64,
|
||||
"what is virt-back?", "answer",
|
||||
json.dumps({"sources": [
|
||||
{"document_uri": "https://en.wikipedia.org/wiki/Virt",
|
||||
"source_role": "background_source", "document_root": "1" * 64},
|
||||
{"document_uri": "https://russell.ballestrini.net/virt-back/",
|
||||
"source_role": "primary_answer_source", "document_root": "2" * 64},
|
||||
]}),
|
||||
"m" * 64, "c" * 64, "g" * 64,
|
||||
"v9.8.0", "norm-v1", "tok-512-v1", "live", 1, 1,
|
||||
),
|
||||
)
|
||||
c.commit()
|
||||
c.close()
|
||||
|
||||
s = open_shards([shard_path])
|
||||
try:
|
||||
ctx = s.context("e" * 64)
|
||||
assert ctx is not None
|
||||
# Headline is the real primary source, not the opaque sentinel.
|
||||
assert ctx.document_uri == "https://russell.ballestrini.net/virt-back/"
|
||||
assert ctx.primary_source_uri == "https://russell.ballestrini.net/virt-back/"
|
||||
assert ctx.source_domains == ["russell.ballestrini.net", "en.wikipedia.org"]
|
||||
finally:
|
||||
s.close()
|
||||
Loading…
Add table
Add a link
Reference in a new issue