modified: .gitlab-ci.yml modified: bench/qa_questions.txt modified: bench/qa_sweep.py modified: bench/run.sh modified: docs/TICKETS.md modified: docs/_source/README.md modified: docs/_source/_ext/makefile_targets.py modified: docs/_source/api/cli.rst modified: docs/_source/api/distill.rst modified: docs/_source/api/mesh.rst modified: docs/_source/api/qa.rst modified: docs/_source/api/retrieval.rst modified: docs/_source/api/storage.rst modified: docs/_source/api/substrate.rst modified: docs/_source/concepts.rst modified: docs/_source/conf.py modified: docs/_source/cookbook.rst modified: docs/_source/index.rst modified: docs/_source/license.rst modified: docs/_source/quickstart.rst modified: docs/bench-maxing.md modified: docs/benchmarks.md modified: docs/cti-architecture.md modified: docs/diagrams/aborist-modules.dot modified: docs/diagrams/aborist-modules.svg modified: docs/diagrams/mesh-data-flow.dot modified: docs/diagrams/mesh-epoch-lifecycle.dot modified: docs/diagrams/mesh-epoch-lifecycle.svg modified: docs/diagrams/mesh-group-decisions.dot modified: docs/diagrams/mesh-group-decisions.svg modified: docs/diagrams/mesh-identity-stack.dot modified: docs/diagrams/mesh-secret-envelope.dot modified: docs/mesh.md modified: docs/qa-modes-bench.md modified: docs/seven-point-program.md modified: docs/tickets/ticket-000001-retrieval-keywords-audit-gap.md modified: docs/tickets/ticket-000002-reference-frame-polarity-contract.md modified: docs/tickets/ticket-000003-anchor-class-warrant.md modified: docs/tickets/ticket-000005-label-ladder-migration.md modified: docs/tickets/ticket-000006-bench-emergent-findings.md modified: docs/tickets/ticket-000007-query-layer-hyphen-fold.md modified: docs/tickets/ticket-000008-broad-quantifier-preflight-guard.md modified: docs/tickets/ticket-000009-quantifier-preflight-dag-binding.md modified: docs/tickets/ticket-000010-metacognition-preflight-guard.md modified: docs/tickets/ticket-000011-soft-preflight-hint-sidecar.md modified: scripts/backfill_concepts.py modified: scripts/bench_emergent.py modified: tests/crawler/test_async_web_fetcher.py modified: tests/crawler/test_bridge.py modified: tests/crawler/test_web_fetch.py modified: tests/test_bench_qa_sweep.py modified: tests/test_burn.py modified: tests/test_burn_doc.py modified: tests/test_claim_lattice.py modified: tests/test_cli_render.py modified: tests/test_compress.py modified: tests/test_concepts.py modified: tests/test_dag.py modified: tests/test_directives.py modified: tests/test_distill.py modified: tests/test_distill_recursive.py modified: tests/test_evict.py modified: tests/test_frame.py modified: tests/test_grok_source.py modified: tests/test_html_source.py modified: tests/test_ingest.py modified: tests/test_inspect.py modified: tests/test_journal.py modified: tests/test_keys.py modified: tests/test_llm_context_base.py modified: tests/test_merkle.py modified: tests/test_mesh.py modified: tests/test_mesh_aead.py modified: tests/test_mesh_chain.py modified: tests/test_mesh_cli.py modified: tests/test_mesh_cli_pull.py modified: tests/test_mesh_wire.py modified: tests/test_mesh_wire_e2e.py modified: tests/test_metacognition.py modified: tests/test_migration_audit_mode.py modified: tests/test_providence_source.py modified: tests/test_qa.py modified: tests/test_qa_quality_live.py modified: tests/test_quantifier_caps.py modified: tests/test_quantifier_classifier.py modified: tests/test_quantifier_phase4.py modified: tests/test_quantifier_reminder.py modified: tests/test_query.py modified: tests/test_reclassify.py modified: tests/test_repair.py modified: tests/test_resume.py modified: tests/test_snapshot.py modified: tests/test_soft_preflight.py modified: tests/test_tfidf.py modified: tests/test_vcs_source.py modified: tests/test_verify.py modified: tests/test_verify_json.py modified: tests/test_versioned_ingest.py modified: tests/test_warrant.py modified: tests/test_wikipedia_old.py modified: tests/test_wikipedia_xml.py modified: tests/test_wikitext.py
96 lines
3.5 KiB
Python
96 lines
3.5 KiB
Python
"""WikipediaSqlDump on the 'old' (revision history) table."""
|
|
|
|
from __future__ import annotations
|
|
|
|
import bz2
|
|
from pathlib import Path
|
|
|
|
import pytest
|
|
|
|
from arborist.sources import WikipediaOldDump, WikipediaSqlDump
|
|
|
|
|
|
# Minimal fabricated 'old' table dump matching the 2003-05-16 schema.
|
|
# Columns: old_id, old_namespace, old_title, old_text, old_comment,
|
|
# old_user, old_user_text, old_timestamp, old_minor_edit, old_flags,
|
|
# inverse_timestamp
|
|
_SAMPLE_SQL = """\
|
|
-- MySQL dump
|
|
DROP TABLE IF EXISTS old;
|
|
CREATE TABLE old (
|
|
old_id int(8) unsigned NOT NULL auto_increment
|
|
);
|
|
INSERT INTO old VALUES (1, 0, 'Anarchism', 'first revision text [[liberty]] [[freedom]]', 'init', 0, 'alice', '20020101000000', 0, '', '79979898'),(2, 0, 'Anarchism', 'second revision text with [[autonomy]]', 'edit', 0, 'bob', '20020201000000', 0, '', '79979897'),(3, 1, 'Talk:Anarchism', 'talk page (skipped: namespace 1)', 'tk', 0, 'carol', '20020301000000', 0, '', '79979896');
|
|
"""
|
|
|
|
|
|
def test_old_dump_yields_revisions(tmp_path):
|
|
p = tmp_path / "fake_old.sql.bz2"
|
|
with bz2.open(p, "wt", encoding="utf-8") as f:
|
|
f.write(_SAMPLE_SQL)
|
|
|
|
src = WikipediaOldDump(path=p)
|
|
docs = list(src.iter_documents())
|
|
# Talk:Anarchism (namespace 1) is filtered; main namespace yields 2 revisions.
|
|
assert len(docs) == 2
|
|
|
|
# Revision 1
|
|
assert docs[0].title == "Anarchism"
|
|
assert docs[0].source_type == "wikipedia_old"
|
|
assert "first revision" in docs[0].content
|
|
assert docs[0].extra["old_timestamp"] == "20020101000000"
|
|
assert docs[0].extra["old_id"] == "1"
|
|
assert any(e.dst_uri.endswith("liberty") for e in docs[0].edges)
|
|
|
|
# Revision 2 — same title, different content, different timestamp.
|
|
assert docs[1].title == "Anarchism"
|
|
assert "second revision" in docs[1].content
|
|
assert docs[1].extra["old_timestamp"] == "20020201000000"
|
|
assert docs[1].extra["old_id"] == "2"
|
|
|
|
|
|
def test_cur_and_old_share_uri_namespace(tmp_path):
|
|
"""Both tables produce the same URI for the same article title."""
|
|
cur_sql = (
|
|
"INSERT INTO cur VALUES "
|
|
"(7, 0, 'Anarchism', 'cur revision text', '', 0, 'sys', '20030516000000', "
|
|
"'', 0, 0, 0, 0, 0, '', '');\n"
|
|
)
|
|
old_sql = (
|
|
"INSERT INTO old VALUES "
|
|
"(11, 0, 'Anarchism', 'old revision text', '', 0, 'sys', "
|
|
"'20020101000000', 0, '', '');\n"
|
|
)
|
|
cur_path = tmp_path / "c.sql.bz2"
|
|
old_path = tmp_path / "o.sql.bz2"
|
|
with bz2.open(cur_path, "wt", encoding="utf-8") as f:
|
|
f.write(cur_sql)
|
|
with bz2.open(old_path, "wt", encoding="utf-8") as f:
|
|
f.write(old_sql)
|
|
|
|
cur_doc = next(WikipediaSqlDump(cur_path, table="cur").iter_documents())
|
|
old_doc = next(WikipediaSqlDump(old_path, table="old").iter_documents())
|
|
assert cur_doc.uri == old_doc.uri
|
|
assert cur_doc.source_type == "wikipedia_cur"
|
|
assert old_doc.source_type == "wikipedia_old"
|
|
# But content differs -> different document_root.
|
|
assert cur_doc.content != old_doc.content
|
|
|
|
|
|
def test_real_dump_smoke():
|
|
"""If the real concatenated old dump exists locally, parse the first few rows."""
|
|
real = Path("data/20030516_old_tablesql.bz2")
|
|
if not real.exists():
|
|
pytest.skip("real old dump not fetched; run 'make fetch-old'")
|
|
src = WikipediaOldDump(path=real)
|
|
docs: list = []
|
|
for d in src.iter_documents():
|
|
docs.append(d)
|
|
if len(docs) >= 3:
|
|
break
|
|
assert len(docs) == 3
|
|
for d in docs:
|
|
assert d.title
|
|
assert d.content
|
|
assert d.extra.get("old_id")
|
|
assert d.extra.get("old_timestamp")
|