#!/usr/bin/env python3 r"""arborist quickstart task runner — the no-`make` path (Windows-first). The Makefile is the canonical Unix entry point and remains so. This runner mirrors **only the quickstart subset** of it so the README quickstart runs on native Windows (cmd / PowerShell) with no `make`, no `bzip2`, no `curl`, no `bash` — just Python 3.10+ and sqlite3 (sqlite3 ships with CPython). It also works on POSIX (`python3 tasks.py `), but on POSIX the Makefile stays the documented path. Ticket: docs/tickets/ticket-000055-windows-quickstart-no-make.md Keep the supported-target set in sync with the README quickstart and the Makefile; tests/test_tasks_runner.py pins it. Usage (make-style KEY=VALUE args, identical to the Makefile): py -3 tasks.py bootstrap py -3 tasks.py fetch-cur py -3 tasks.py ingest-cur-attached SHARDS=4 py -3 tasks.py distill-shards-parallel py -3 tasks.py distill-shards-tfidf-parallel py -3 tasks.py query Q="What is anarcho-capitalism?" py -3 tasks.py inspect KEY= JSON=1 py -3 tasks.py help On Windows `make.bat` forwards to this file, so `make ` (cmd) or `.\make.bat ` (PowerShell) works exactly like `make` does on Unix. """ from __future__ import annotations import os import shlex import shutil import subprocess import sys import urllib.error import urllib.request from collections.abc import Callable from pathlib import Path # -------------------------------------------------------------------------- # Config — mirrors the Makefile's `?=` defaults. Every value is overridable # the same two ways the Makefile allows: an environment variable, or a # make-style KEY=VALUE token on the command line (the CLI token wins). # -------------------------------------------------------------------------- REPO = Path(__file__).resolve().parent IS_WINDOWS = os.name == "nt" DEFAULTS = { "VENV": ".venv", "DATA_DIR": "data", "WP_BASE_URL": "https://dumps.wikimedia.org/archive/2003/2003-05-16/en", "WP_CUR_NAME": "20030516_cur_tablesql.bz2", "DB": str(Path.home() / ".arborist" / "arborist.db"), "SHARDS_DIR": str(Path.home() / ".arborist" / "shards"), "SHARDS": "4", "INGEST_LIMIT": "500", "VERIFY_N": "10", "SEARCH_Q": "computer", "QUERY_TOP_K": "8", "ANSWER_MODE": "claim_lattice", "KG_SECONDS": "3600", "CRAWL_DEPTH": "2", "CRAWL_MAX": "0", "RECRAWL_LIMIT": "100", } # CLI KEY=VALUE tokens collected by main(); read through cfg(). _CLI_VARS: dict[str, str] = {} def cfg(key: str, default: str | None = None) -> str: """Resolve a config value: CLI token > environment > DEFAULTS > default.""" if key in _CLI_VARS: return _CLI_VARS[key] if key in os.environ: return os.environ[key] if key in DEFAULTS: return DEFAULTS[key] return "" if default is None else default def truthy(key: str) -> bool: """Make semantics: a variable is 'set' iff it has a non-empty value.""" return cfg(key).strip() != "" # -------------------------------------------------------------------------- # Platform-aware venv layout + Python launcher detection. # -------------------------------------------------------------------------- def venv_dir() -> Path: return REPO / cfg("VENV") def venv_bin_dir() -> Path: # Windows venvs put executables in Scripts\; POSIX in bin/. return venv_dir() / ("Scripts" if IS_WINDOWS else "bin") def venv_exe(name: str) -> Path: suffix = ".exe" if IS_WINDOWS else "" return venv_bin_dir() / f"{name}{suffix}" def venv_python() -> Path: return venv_exe("python") def arborist_exe() -> Path: return venv_exe("arborist") def base_python_cmd() -> list[str]: """The interpreter used to *create* the venv. Override with ARBORIST_PYTHON (e.g. 'py -3.12' or 'C:\\py\\python.exe'). Windows ships the `py` launcher rather than `python3`, so prefer it. """ override = os.environ.get("ARBORIST_PYTHON", "").strip() if override: return shlex.split(override, posix=not IS_WINDOWS) if IS_WINDOWS: if shutil.which("py"): return ["py", "-3"] if shutil.which("python"): return ["python"] return ["python3"] return ["python3"] if shutil.which("python3") else ["python"] # -------------------------------------------------------------------------- # Process helpers. # -------------------------------------------------------------------------- def _env() -> dict[str, str]: env = dict(os.environ) env["PYTHONUNBUFFERED"] = "1" # CLAUDE.md: never buffer long-running procs return env def run(argv: list[str], *, cwd: Path | None = None) -> None: """Run a command, inheriting stdio. Abort the runner on failure.""" printable = " ".join(shlex.quote(str(a)) for a in argv) print(f">> {printable}", flush=True) rc = subprocess.run(argv, cwd=str(cwd or REPO), env=_env()).returncode if rc != 0: sys.exit(rc) def run_parallel(cmds: list[list[str]]) -> None: """Fan out N commands, wait for all (the bash `for … & done; wait`).""" if not cmds: print(" (nothing to do — no shards found)", flush=True) return procs = [] for argv in cmds: printable = " ".join(shlex.quote(str(a)) for a in argv) print(f">> {printable}", flush=True) procs.append(subprocess.Popen(argv, cwd=str(REPO), env=_env())) failed = 0 for p in procs: if p.wait() != 0: failed += 1 if failed: sys.exit(f"{failed} of {len(procs)} parallel job(s) failed") def arborist(*args: str) -> list[str]: return [str(arborist_exe()), *args] def shard_dbs(shards_dir: Path) -> list[Path]: return sorted(shards_dir.glob("*.db")) if shards_dir.is_dir() else [] # -------------------------------------------------------------------------- # Bootstrap (idempotent, mirrors the Makefile's pyproject-mtime rule). # -------------------------------------------------------------------------- def _venv_marker() -> Path: return venv_bin_dir() / ("activate" if not IS_WINDOWS else "activate.bat") def ensure_bootstrap() -> None: """Create the venv + editable install. No-op when already up to date.""" marker = _venv_marker() pyproject = REPO / "pyproject.toml" up_to_date = ( arborist_exe().exists() and marker.exists() and marker.stat().st_mtime >= pyproject.stat().st_mtime ) if up_to_date: return if not venv_python().exists(): run([*base_python_cmd(), "-m", "venv", cfg("VENV")]) pip = [str(venv_python()), "-m", "pip"] run([*pip, "install", "--upgrade", "pip", "wheel"]) # Matches Makefile:52 — `.[dev]` (NOT `[dev,html]`; the README claim of # [dev,html] was stale, corrected with this ticket — artifact wins). run([*pip, "install", "-e", ".[dev]"]) marker.touch() def ensure_crawler() -> None: ensure_bootstrap() run([str(venv_python()), "-m", "pip", "install", "-e", ".[crawler]"]) # -------------------------------------------------------------------------- # Targets. # -------------------------------------------------------------------------- def t_bootstrap() -> None: """create venv and install the editable package ([dev] extras)""" ensure_bootstrap() print("bootstrap OK ·", arborist_exe(), flush=True) def _wp_cur_path() -> Path: return REPO / cfg("DATA_DIR") / cfg("WP_CUR_NAME") def t_fetch_cur() -> None: """download the 2003-05-16 cur table dump (~82 MB) — idempotent""" dest = _wp_cur_path() if dest.exists() and dest.stat().st_size > 0: print(f"fetch-cur: {dest} already present — skipping", flush=True) return dest.parent.mkdir(parents=True, exist_ok=True) url = f"{cfg('WP_BASE_URL')}/{cfg('WP_CUR_NAME')}" tmp = dest.with_suffix(dest.suffix + ".part") print(f">> fetching {url}", flush=True) last_err: Exception | None = None for attempt in range(1, 4): # curl --retry 3 try: req = urllib.request.Request(url, headers={"User-Agent": "arborist-tasks/1"}) with urllib.request.urlopen(req) as resp, open(tmp, "wb") as fh: total = int(resp.headers.get("Content-Length") or 0) got = 0 while chunk := resp.read(1 << 20): fh.write(chunk) got += len(chunk) if total: pct = got * 100 // total print(f"\r {got >> 20} / {total >> 20} MB ({pct}%)", end="", flush=True) print(flush=True) tmp.replace(dest) print(f"fetch-cur: saved {dest}", flush=True) return except (urllib.error.URLError, OSError) as exc: # noqa: PERF203 last_err = exc print(f" attempt {attempt}/3 failed: {exc}", flush=True) tmp.unlink(missing_ok=True) sys.exit(f"fetch-cur: download failed after 3 attempts: {last_err}") def t_ingest_cur_attached() -> None: """sharded ingest of the cur snapshot, one process per shard (Phase 2)""" ensure_bootstrap() t_fetch_cur() shards_dir = Path(cfg("SHARDS_DIR")) shards_dir.mkdir(parents=True, exist_ok=True) n = int(cfg("SHARDS")) cur = str(_wp_cur_path()) cmds = [ arborist("ingest", "--source", "wikipedia_cur", "--path", cur, "--shards-dir", str(shards_dir), "--shard", f"{i}/{n}") for i in range(n) ] run_parallel(cmds) def _distill_parallel(process: str) -> None: ensure_bootstrap() shards = shard_dbs(Path(cfg("SHARDS_DIR"))) cmds = [ arborist("--db", str(s), "distill", "--process", process, "--kind", "surface") for s in shards ] run_parallel(cmds) def t_distill_shards_parallel() -> None: """surface → core, first-sentence-v1, one process per shard""" _distill_parallel("first-sentence-v1") def t_distill_shards_tfidf_parallel() -> None: """core → TF-IDF keyword sets for retrieval, one process per shard""" _distill_parallel("tfidf-keywords-v1") def _question() -> str: q = cfg("Q") or " ".join(_POSITIONALS) if not q.strip(): sys.exit('usage: query Q="your question" [JSON=1 ANSWER_MODE=... ' 'K="extra keywords" BURN=1 REPAIR=1 WITNESS=1 ' 'BROAD=1 REJECT_BROAD=1 ALLOW_BROAD=1 XLANG=1 XLANG_MT=1]') return q def _query_argv(*, dry: bool) -> list[str]: a = ["--shards-dir", cfg("SHARDS_DIR"), "query", "--top-k", cfg("QUERY_TOP_K")] if dry: a.append("--dry-run") if truthy("JSON"): a.append("--json") if truthy("BURN"): a.append("--burn") if not dry and truthy("REPAIR"): a.append("--repair") if not dry and truthy("REPROMPTS"): a += ["--repair-reprompts", cfg("REPROMPTS")] if cfg("ANSWER_MODE").strip(): a += ["--answer-mode", cfg("ANSWER_MODE")] if not dry and truthy("K"): a += ["--retrieval-keywords", cfg("K")] if truthy("BROAD"): a.append("--apply-quantifier-caps") if truthy("REJECT_BROAD"): a.append("--reject-broad") if truthy("ALLOW_BROAD"): a.append("--allow-broad") if not dry and truthy("WITNESS"): a.append("--witness") if truthy("XLANG"): # ticket #000001 §7 cross-language guard (default OFF) a.append("--crosslang-guard") if truthy("XLANG_MT"): # ticket #000056 Operation Sandwich (default OFF) a.append("--crosslang-translate") a.append(_question()) return a def t_query() -> None: """ask the corpus a question [JSON=1 ANSWER_MODE=... K=... BURN=1 …]""" ensure_bootstrap() run(arborist(*_query_argv(dry=False))) def t_query_dry() -> None: """like query but skip the LLM call (dry-run)""" ensure_bootstrap() run(arborist(*_query_argv(dry=True))) def t_inspect() -> None: """sidecar: classify unverified spans for a cache_key — KEY=hex [JSON=1]""" ensure_bootstrap() if not cfg("KEY"): sys.exit("usage: inspect KEY= [JSON=1]") a = ["--shards-dir", cfg("SHARDS_DIR"), "inspect", "--cache-key", cfg("KEY")] if truthy("JSON"): a.append("--json") run(arborist(*a)) def t_falsify() -> None: """mark a cached answer wrong — KEY=hex REASON='why'""" ensure_bootstrap() if not cfg("KEY"): sys.exit("usage: falsify KEY= REASON='why'") run(arborist("--shards-dir", cfg("SHARDS_DIR"), "providence", "--falsify", cfg("KEY"), "--reason", cfg("REASON"))) def t_burn() -> None: """delete a childless leaf — KEY=hex (providence) | KIND=document|core ROOT=hex; REASON='why' [FORCE=1]""" ensure_bootstrap() kind = cfg("KIND") or "providence" base = ["--shards-dir", cfg("SHARDS_DIR"), "burn", "--kind", kind] if kind == "providence": if not cfg("KEY"): sys.exit("usage: burn KEY= REASON='why' [FORCE=1]") base += ["--cache-key", cfg("KEY")] elif kind in ("document", "core"): if not cfg("ROOT"): sys.exit(f"usage: burn KIND={kind} ROOT= REASON='why' [FORCE=1]") base += ["--root", cfg("ROOT")] else: sys.exit(f"unknown KIND: {kind} (expected: providence|document|core)") base += ["--reason", cfg("REASON")] if truthy("FORCE"): base.append("--force") run(arborist(*base)) def t_burn_kindergarten() -> None: """bust providence rows younger than SECONDS [SECONDS=3600 FORCE=1 DRY_RUN=1 REASON='why']""" ensure_bootstrap() # Makefile var is KG_SECONDS; its help advertises SECONDS — accept both, # SECONDS (the documented token) wins. seconds = cfg("SECONDS") or cfg("KG_SECONDS") a = ["--shards-dir", cfg("SHARDS_DIR"), "burn-kindergarten", "--kindergarten-seconds", seconds] if truthy("REASON"): a += ["--reason", cfg("REASON")] if truthy("FORCE"): a.append("--force") if truthy("DRY_RUN"): a.append("--dry-run") run(arborist(*a)) def t_bootstrap_crawler() -> None: """install the [crawler] extras into the venv""" ensure_crawler() print("bootstrap-crawler OK", flush=True) def _crawl_shard_path() -> Path: custom = cfg("CRAWL_SHARD") if custom: return Path(custom) url = cfg("URL") # Mirror the Makefile sed: strip scheme, take up to first '/', '.'→'_'. rest = url.split("://", 1)[-1] host = rest.split("/", 1)[0].replace(".", "_") return Path(cfg("SHARDS_DIR")) / f"crawl_{host}.db" def t_crawl_ingest() -> None: """BFS-crawl URL=https://x.com [DEPTH=2 MAX=0 FAST=1] into a per-host shard""" if not cfg("URL"): sys.exit("usage: crawl-ingest URL=https://example.com " "[DEPTH=2 MAX=0 FAST=1 CRAWL_SHARD=path]") ensure_crawler() Path(cfg("SHARDS_DIR")).mkdir(parents=True, exist_ok=True) shard = _crawl_shard_path() print(f" shard: {shard}", flush=True) a = ["--db", str(shard), "crawl", "--seed-url", cfg("URL"), "--depth", cfg("DEPTH", cfg("CRAWL_DEPTH")), "--max-pages", cfg("MAX", cfg("CRAWL_MAX"))] if truthy("FAST"): a.append("--fast") a.append("--ingest") run(arborist(*a)) def t_recrawl_check() -> None: """conditional-HEAD freshness probe [DOMAIN=x.com LIMIT=100 CRAWL_SHARD=path]""" ensure_crawler() limit = cfg("LIMIT", cfg("RECRAWL_LIMIT")) domain = cfg("DOMAIN") def one(db: str) -> list[str]: a = ["--db", db, "crawler", "recrawl-check"] if domain: a += ["--domain", domain] a += ["--limit", limit] return arborist(*a) if cfg("CRAWL_SHARD"): run(one(cfg("CRAWL_SHARD"))) return for db in shard_dbs(Path(cfg("SHARDS_DIR"))): if db.name in ("qa.db", "snapshots.db"): continue print(f" shard: {db}", flush=True) run(one(str(db))) def t_stats() -> None: """counts: documents, chunks, edges, audit chain (single DB)""" ensure_bootstrap() run(arborist("--db", cfg("DB"), "stats")) def t_stats_shards() -> None: """cross-shard stats via UNION views over SHARDS_DIR""" ensure_bootstrap() run(arborist("--shards-dir", cfg("SHARDS_DIR"), "stats")) def t_verify() -> None: """round-trip Merkle proofs for VERIFY_N random documents""" ensure_bootstrap() run(arborist("--db", cfg("DB"), "verify", "-n", cfg("VERIFY_N"))) def t_search() -> None: """keyword search [Q=... | SEARCH_Q=computer]""" ensure_bootstrap() run(arborist("--db", cfg("DB"), "search", cfg("Q") or cfg("SEARCH_Q"))) def t_test() -> None: """run the pytest suite (excludes opt-in crawler tests)""" ensure_bootstrap() run([str(venv_exe("pytest")), "-q", "--ignore=tests/crawler", "-n", "auto"]) def t_clean() -> None: """remove venv + caches (keeps fetched data and db)""" for path in (venv_dir(), REPO / ".pytest_cache", REPO / "arborist.egg-info"): shutil.rmtree(path, ignore_errors=True) for pyc in REPO.rglob("__pycache__"): shutil.rmtree(pyc, ignore_errors=True) print("clean OK", flush=True) def t_clean_db() -> None: """drop the arborist db (keeps fetched data and venv)""" db = Path(cfg("DB")) for suffix in ("", "-journal", "-wal", "-shm"): Path(str(db) + suffix).unlink(missing_ok=True) print(f"clean-db OK · {db}", flush=True) def t_clean_data() -> None: """remove fetched dumps (DATA_DIR)""" shutil.rmtree(REPO / cfg("DATA_DIR"), ignore_errors=True) print("clean-data OK", flush=True) # Ordered so `help` reads top-to-bottom like the quickstart. TARGETS: dict[str, Callable[[], None]] = { "bootstrap": t_bootstrap, "fetch-cur": t_fetch_cur, "ingest-cur-attached": t_ingest_cur_attached, "distill-shards-parallel": t_distill_shards_parallel, "distill-shards-tfidf-parallel": t_distill_shards_tfidf_parallel, "query": t_query, "query-dry": t_query_dry, "inspect": t_inspect, "falsify": t_falsify, "burn": t_burn, "burn-kindergarten": t_burn_kindergarten, "bootstrap-crawler": t_bootstrap_crawler, "crawl-ingest": t_crawl_ingest, "recrawl-check": t_recrawl_check, "stats": t_stats, "stats-shards": t_stats_shards, "verify": t_verify, "search": t_search, "test": t_test, "clean": t_clean, "clean-db": t_clean_db, "clean-data": t_clean_data, } _POSITIONALS: list[str] = [] def _unquote(s: str) -> str: """Strip one layer of matching surrounding quotes. POSIX shells strip quotes before exec; cmd/PowerShell forward them literally through `%*`. Stripping here makes the single documented `KEY="value with spaces"` form behave identically on every shell — including the cmd-friendly whole-token form `"KEY=value with spaces"`. """ if len(s) >= 2 and s[0] == s[-1] and s[0] in ("'", '"'): return s[1:-1] return s def t_help() -> None: """show this help""" print("arborist quickstart runner — Windows/no-make path " "(Makefile is canonical on Unix)\n") print("usage: py -3 tasks.py [KEY=VALUE ...]\n") for name, fn in TARGETS.items(): print(f" {name:<30} {(fn.__doc__ or '').strip()}") print(f" {'help':<30} {t_help.__doc__.strip()}") def main(argv: list[str]) -> None: args = argv[1:] if not args or args[0] in ("help", "-h", "--help"): t_help() return target = args[0] for raw in args[1:]: tok = _unquote(raw) # cmd-friendly "KEY=value with spaces" if "=" in tok and not tok.startswith("="): key, val = tok.split("=", 1) _CLI_VARS[key.strip().upper()] = _unquote(val) # KEY="value" else: _POSITIONALS.append(tok) fn = TARGETS.get(target) if fn is None: print(f"unknown target: {target}\n", file=sys.stderr) t_help() sys.exit(2) fn() if __name__ == "__main__": main(sys.argv)