feat: optional arborist provenance sink (opt-in, default off)

Adds neopig/arborist_sink.py: when enabled, every crawled page neopig
stores is also committed to an arborist content-addressed store — a
Merkle document_root + append-only audit chain — alongside neopig's
existing md5/FileVault storage. Verifiable dedup + FTS5 + tamper-evident
provenance; replaces nothing.

Strictly opt-in: a hard no-op unless NEOPIG_ARBORIST_ENABLED is set AND
arborist is importable. neopig behaves byte-identically without it. This
keeps neopig public-domain by default — arborist is AGPL, pulled only
when an operator opts in. neopig never touches arborist's tables; all
writes go through arborist.embed (no-raw-SQL rule preserved). Sync
SQLite writes run off the event loop via to_thread, lock-serialized,
and failures are swallowed so the mirror can never break a crawl.

Wired into NeoPig.__init__ (self.arborist_sink) and the store_page hook
in the crawl path. Phase-0 scope: page text only; media-manifest edges
are a follow-on. Tests cover disabled-by-default, flag-without-arborist,
the page->Document mapping, and enabled end-to-end + content-idempotence.
This commit is contained in:
russell@unturf.com 2026-05-22 13:06:59 -04:00
parent 341bcb325a
commit ced92de67c
No known key found for this signature in database
3 changed files with 263 additions and 0 deletions

View file

@ -66,6 +66,7 @@ from neopig.database import Database, SCORE_SCREENSHOT, SCORE_OG_IMAGE, SCORE_TH
from neopig.screenshot import ScreenshotCapture, ScreenshotConfig
from neopig.domain_vault import VaultManager, DomainHtmlVault, DomainMediaVault, DomainLinkpeekVault, extract_media_urls
from neopig.repo import detect_vcs, clone_repo_async, pull_repo_async, get_repo_path, walk_files, get_commit_hash, get_file_language, is_binary_file
from neopig.arborist_sink import ArboristSink
from tqdm import tqdm
logger = logging.getLogger(__name__)
@ -355,6 +356,9 @@ class NeoPig:
self.screenshot = ScreenshotCapture(screenshot_config or ScreenshotConfig())
self.screenshot_config = screenshot_config or ScreenshotConfig()
self.vault_path = vault_path
# Optional, opt-in arborist provenance mirror (default OFF; no-op
# unless NEOPIG_ARBORIST_ENABLED is set and arborist is installed).
self.arborist_sink = ArboristSink()
# Track stats
self.stats = {
@ -818,6 +822,16 @@ class NeoPig:
crawl_job_id=crawl_job_id,
)
# Opt-in: also commit the page text to arborist for verifiable
# dedup + audit-chained provenance. No-op when the sink is off.
await self.arborist_sink.ingest_page(
uri=uri,
title=title,
content=content,
path=path,
crawl_job_id=crawl_job_id,
)
async def _archive_media_to_vault(
self,
url: str,

149
neopig/arborist_sink.py Normal file
View file

@ -0,0 +1,149 @@
# This is free software for the public good of a permacomputer hosted at
# permacomputer.com, an always-on computer by the people, for the people.
# One which is durable, easy to repair, & distributed like tap water
# for machine learning intelligence.
#
# The permacomputer is community-owned infrastructure optimized around
# four values:
#
# TRUTH First principles, math & science, open source code freely distributed
# FREEDOM Voluntary partnerships, freedom from tyranny & corporate control
# HARMONY Minimal waste, self-renewing systems with diverse thriving connections
# LOVE Be yourself without hurting others, cooperation through natural law
#
# This software contributes to that vision by archiving the web, preserving digital knowledge before it disappears.
# Code is seeds to sprout on any abandoned technology.
"""Optional arborist provenance sink.
When enabled, every crawled page neopig stores is ALSO committed to an
arborist content-addressed store: a Merkle ``document_root`` plus an
append-only audit chain, sitting beside neopig's md5/FileVault storage.
It adds verifiable dedup + an FTS5 index + tamper-evident provenance; it
replaces nothing.
Default OFF, and a hard no-op when the flag is unset OR arborist is not
installed neopig behaves identically without it. arborist is AGPL and
neopig is public domain, so the dependency stays strictly opt-in: an
operator who enables it pulls AGPL code by choice.
Enable::
pip install -e ~/git/arborist # core only (httpx/zstandard/crypto)
export NEOPIG_ARBORIST_ENABLED=1
export NEOPIG_ARBORIST_DB=data/arborist.db # optional, default shown
neopig never touches arborist's tables — every write goes through
``arborist.embed``, so neopig's no-raw-SQL rule is preserved (there is no
SQL to write).
"""
from __future__ import annotations
import asyncio
import hashlib
import logging
import os
from typing import List, Optional
logger = logging.getLogger(__name__)
try:
from arborist.embed import Document, Edge, ingest_documents, open_store
_ARBORIST_AVAILABLE = True
except ImportError:
_ARBORIST_AVAILABLE = False
def _truthy(value: Optional[str]) -> bool:
return (value or "").strip().lower() in {"1", "true", "yes", "on"}
class ArboristSink:
"""Opt-in mirror of crawled pages into an arborist store.
Every public method is a no-op unless the sink is enabled which
requires both the ``NEOPIG_ARBORIST_ENABLED`` flag AND arborist being
importable. arborist's sync SQLite writes run in a worker thread so
they never block neopig's event loop, serialized by a lock so
concurrent pages don't contend on the one writer.
"""
def __init__(self, db_path: Optional[str] = None, enabled: Optional[bool] = None):
flag = _truthy(os.environ.get("NEOPIG_ARBORIST_ENABLED")) if enabled is None else bool(enabled)
self.available = _ARBORIST_AVAILABLE
self.enabled = bool(flag and self.available)
self.db_path = db_path or os.environ.get("NEOPIG_ARBORIST_DB", "data/arborist.db")
self._lock = asyncio.Lock()
if flag and not self.available:
logger.warning(
"NEOPIG_ARBORIST_ENABLED is set but arborist is not installed; "
"provenance sink disabled. `pip install -e ~/git/arborist` to enable."
)
elif self.enabled:
logger.info("arborist provenance sink enabled -> %s", self.db_path)
def _page_to_document(
self,
uri: str,
title: Optional[str],
content: str,
*,
path: str = "",
description: str = "",
keywords: Optional[list] = None,
crawl_job_id: Optional[int] = None,
media_uris: Optional[List[str]] = None,
):
"""Map a crawled page onto an arborist Document.
Provenance neopig already tracks (its md5 uri-hash, crawl job,
meta) rides along in ``extra``; page->media links become edges.
"""
edges = [Edge(edge_type="embeds_media", dst_uri=m) for m in (media_uris or []) if m]
return Document(
uri=uri,
content=content or "",
source_type="neopig_html",
title=title or None,
edges=edges,
extra={
"path": path,
"description": description,
"keywords": keywords or [],
"crawl_job_id": crawl_job_id,
"md5_uri_hash": hashlib.md5(uri.encode()).hexdigest(),
},
)
async def ingest_page(self, uri: str, title: Optional[str], content: str, **kwargs) -> bool:
"""Commit one crawled page to arborist.
No-op returning ``False`` when disabled. Runs the sync arborist
write off the event loop. Failures are swallowed and logged a
provenance mirror must NEVER break a crawl.
"""
if not self.enabled:
return False
document = self._page_to_document(uri, title, content, **kwargs)
async with self._lock:
try:
await asyncio.to_thread(self._ingest_sync, [document])
return True
except Exception as exc: # never let the mirror break a crawl
logger.warning("arborist ingest failed for %s: %s", uri, exc)
return False
def _ingest_sync(self, documents) -> None:
"""Open a fresh connection in this worker thread, ingest, close.
Per-page open/close keeps each SQLite connection bound to one
thread (safe) and leans on arborist's idempotent ingest. Batching
is a later optimization; correctness first.
"""
conn = open_store(self.db_path)
try:
ingest_documents(conn, documents)
finally:
conn.close()

View file

@ -0,0 +1,100 @@
# This is free software for the public good of a permacomputer hosted at
# permacomputer.com, an always-on computer by the people, for the people.
# One which is durable, easy to repair, & distributed like tap water
# for machine learning intelligence.
#
# The permacomputer is community-owned infrastructure optimized around
# four values:
#
# TRUTH First principles, math & science, open source code freely distributed
# FREEDOM Voluntary partnerships, freedom from tyranny & corporate control
# HARMONY Minimal waste, self-renewing systems with diverse thriving connections
# LOVE Be yourself without hurting others, cooperation through natural law
#
# This software contributes to that vision by archiving the web, preserving digital knowledge before it disappears.
# Code is seeds to sprout on any abandoned technology.
"""Tests for the optional arborist provenance sink.
Disabled-path tests run anywhere. The enabled-path test skips when
arborist is not installed (it is an opt-in dependency).
"""
import pytest
from neopig.arborist_sink import ArboristSink
def test_disabled_by_default(monkeypatch):
"""No flag -> sink off, regardless of whether arborist is installed."""
monkeypatch.delenv("NEOPIG_ARBORIST_ENABLED", raising=False)
sink = ArboristSink()
assert sink.enabled is False
@pytest.mark.asyncio
async def test_ingest_page_is_noop_when_disabled(tmp_path, monkeypatch):
monkeypatch.delenv("NEOPIG_ARBORIST_ENABLED", raising=False)
db = tmp_path / "arborist.db"
sink = ArboristSink(db_path=str(db), enabled=False)
result = await sink.ingest_page("https://x.com/a", "Title", "body text")
assert result is False
assert not db.exists(), "disabled sink must not create a store"
def test_flag_without_arborist_stays_disabled(monkeypatch):
"""Flag on but arborist missing -> still disabled (no crash)."""
monkeypatch.setattr("neopig.arborist_sink._ARBORIST_AVAILABLE", False)
sink = ArboristSink(enabled=True)
assert sink.available is False
assert sink.enabled is False
def test_page_to_document_mapping():
"""Mapping is pure and available even when the sink is disabled,
as long as arborist's types import. Skip if arborist absent."""
pytest.importorskip("arborist")
sink = ArboristSink(enabled=False)
doc = sink._page_to_document(
"https://x.com/post",
"A Post",
"the page prose",
path="/post",
crawl_job_id=7,
media_uris=["https://x.com/img.jpg"],
)
assert doc.uri == "https://x.com/post"
assert doc.source_type == "neopig_html"
assert doc.title == "A Post"
assert doc.extra["crawl_job_id"] == 7
assert doc.extra["md5_uri_hash"] # provenance carried along
assert any(e.dst_uri == "https://x.com/img.jpg" for e in doc.edges)
@pytest.mark.asyncio
async def test_enabled_path_ingests_into_arborist(tmp_path):
"""End-to-end: flag on + arborist installed -> page lands in the store
and is content-deduped. Skips when arborist is not installed."""
pytest.importorskip("arborist")
from arborist.embed import open_store
db = tmp_path / "arborist.db"
sink = ArboristSink(db_path=str(db), enabled=True)
assert sink.enabled is True
ok = await sink.ingest_page(
"https://x.com/post", "A Post", "the quick brown fox", path="/post", crawl_job_id=1
)
assert ok is True
assert db.exists()
conn = open_store(str(db))
try:
n = conn.execute("SELECT COUNT(*) FROM documents").fetchone()[0]
assert n == 1
# Re-ingest identical content -> idempotent, still one doc.
await sink.ingest_page("https://x.com/post", "A Post", "the quick brown fox", path="/post")
n2 = conn.execute("SELECT COUNT(*) FROM documents").fetchone()[0]
assert n2 == 1
finally:
conn.close()