From 3ea27aa4713cae6d2181ab80efbb0157508a5899 Mon Sep 17 00:00:00 2001 From: "russell@unturf.com" Date: Mon, 11 May 2026 08:27:38 -0400 Subject: [PATCH] =?UTF-8?q?#000047=20=E2=80=94=20close:=20delta=5Faggregat?= =?UTF-8?q?or=20knob=20on=20ForkScore=20(Option=20D)?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The #000025 §10.14 calibration showed _delta_5{s,t,f} mean over a battery's 5 subs, so a single-sub gain weighs 1/5 of face value (the 5× dilution). #000047 ships the knob to pick the aggregation, default unchanged. WeightSet.delta_aggregator ∈ {"mean","max","sum"} (default "mean") — a categorical field, validated in __post_init__ against DELTA_AGGREGATORS; from_dict takes it as a string. Default unchanged → ScoredFork output byte-identical → no fork_score.ESTIMATOR_VERSION bump. fork_score._aggregate(deltas, how): mean = arithmetic mean, max = max(0.0, max_i Δ_i), sum = Σ Δ_i; empty → 0.0. _delta_5s/_delta_5t/ _delta_5f take an aggregator arg (default "mean"); the 5F efficiency bonus is added after the aggregated base (aggregator-independent). fork_score passes weights.delta_aggregator. The per-sub HARD_REGRESSION_FLOOR flags are computed before aggregation, so a single-sub regression still forces REJECT under max/sum. The chosen aggregator is recorded in ScoredFork.weights["delta_aggregator"] (via WeightSet.as_dict()); fork_score_branches traceability stays via the opaque weights_id — no schema migration. bench/scripts/fivef_threshold_calibration.py gained §5 — runs the #000046 below-ceiling pack (5f/falsification at 0.333) and shows the verdict / γ·Δ5f under each aggregator; bench/results/5f-threshold- calibration-2026-05-11.md §5 is the captured record. Default stays "mean" — the conservative, noise-robust, regression-symmetric choice matching docs/bench-maxing.md's per-rate floor framing; v8 picks max/sum per-deployment. Tests: 8 new in tests/test_fork_score.py + 1 anchor in tests/test_fivef_threshold_calibration.py; tests/test_weights.py as_dict field-set test updated to include delta_aggregator; test_fork_score.py AUTOCOUNT tags (#000012 §286, warrant-substrate- cookbook.md ×2) bumped 23 → 31. #000047 closed; #000012 §8 §3 + TICKETS.md row updated. Full suite: 2330 passed, 28 skipped. --- arborist/substrate/fork_score.py | 66 ++++++++++---- arborist/substrate/weights.py | 40 ++++++++- .../5f-threshold-calibration-2026-05-11.md | 14 ++- bench/scripts/fivef_threshold_calibration.py | 50 +++++++++++ docs/TICKETS.md | 2 +- ...ket-000012-selection-consensus-protocol.md | 17 ++-- ...cket-000047-fork-score-delta-aggregator.md | 63 +++++++++++--- docs/warrant-substrate-cookbook.md | 4 +- tests/test_fivef_threshold_calibration.py | 4 + tests/test_fork_score.py | 86 +++++++++++++++++++ tests/test_weights.py | 4 +- 11 files changed, 307 insertions(+), 43 deletions(-) diff --git a/arborist/substrate/fork_score.py b/arborist/substrate/fork_score.py index 5a5a766..88b0b37 100644 --- a/arborist/substrate/fork_score.py +++ b/arborist/substrate/fork_score.py @@ -137,10 +137,39 @@ def bench_result_to_metrics(payload: dict) -> dict[str, dict[str, dict]]: return out +def _aggregate(deltas: list[float], how: str) -> float: + """Collapse a per-sub-battery Δ-rate vector into one term (ticket + #000047). ``how`` ∈ :data:`arborist.substrate.weights.DELTA_AGGREGATORS`: + + - ``"mean"`` — arithmetic mean (default). Broad, balanced, + noise-robust; matches ``docs/bench-maxing.md``'s per-rate 5-pp + floor (a single sub-battery's gain is diluted 1/n). + - ``"max"`` — the largest single-sub gain, floored at 0 + (``max(0, max_i Δ_i)``). Rewards specialization; weighs a + single-sub improvement at face value. Silent on the rest of the + vector — but per-sub regressions are still caught by the + independent ``HARD_REGRESSION_FLOOR`` flag, so the verdict + doesn't soften. + - ``"sum"`` — total improvement (``Σ Δ_i``). Rewards breadth *and* + magnitude; can over-reward many sub-noise-floor gains. + + Empty vector → 0.0 under every aggregator. + """ + if not deltas: + return 0.0 + if how == "mean": + return sum(deltas) / len(deltas) + if how == "max": + return max(0.0, max(deltas)) + if how == "sum": + return sum(deltas) + raise ValueError(f"unknown delta aggregator {how!r}") + + def _delta_5s( - parent: dict[str, dict], child: dict[str, dict] + parent: dict[str, dict], child: dict[str, dict], aggregator: str = "mean" ) -> tuple[float, list[str]]: - """Mean Δ-rate across 5S sub-batteries. Returns (delta, regression_subs).""" + """Aggregated Δ-rate across 5S sub-batteries. Returns (delta, regression_subs).""" deltas: list[float] = [] regressed: list[str] = [] for sub, key in _BATTERY_RATE_KEYS["5s"].items(): @@ -150,15 +179,13 @@ def _delta_5s( deltas.append(d) if d <= -HARD_REGRESSION_FLOOR: regressed.append(f"5s/{sub}: -{abs(d):.3f}") - if not deltas: - return 0.0, regressed - return sum(deltas) / len(deltas), regressed + return _aggregate(deltas, aggregator), regressed def _delta_5t( - parent: dict[str, dict], child: dict[str, dict] + parent: dict[str, dict], child: dict[str, dict], aggregator: str = "mean" ) -> tuple[float, list[str]]: - """Mean Δ-rate across 5T sub-batteries (using canonical + """Aggregated Δ-rate across 5T sub-batteries (using canonical Dav1DPrometheus names — transfer-learning, not legacy transfer).""" deltas: list[float] = [] regressed: list[str] = [] @@ -172,20 +199,19 @@ def _delta_5t( deltas.append(d) if d <= -HARD_REGRESSION_FLOOR: regressed.append(f"5t/{sub}: -{abs(d):.3f}") - if not deltas: - return 0.0, regressed - return sum(deltas) / len(deltas), regressed + return _aggregate(deltas, aggregator), regressed def _delta_5f( - parent: dict[str, dict], child: dict[str, dict] + parent: dict[str, dict], child: dict[str, dict], aggregator: str = "mean" ) -> tuple[float, list[str], list[str]]: - """Mean Δ-rate across 5F sub-batteries + efficiency-aware bonus. + """Aggregated Δ-rate across 5F sub-batteries + efficiency-aware bonus. Returns (delta, regression_subs, efficiency_flags). The efficiency_flags include ``"NEG_INF_REGRESSION"`` when the child has ``adaptation_efficiency_neg_infinite_count > 0`` — a hard - reject signal regardless of other terms. + reject signal regardless of other terms. The efficiency bonus is + added *after* the aggregated base, so it's aggregator-independent. """ deltas: list[float] = [] regressed: list[str] = [] @@ -225,7 +251,7 @@ def _delta_5f( f"increased {p_neg}→{c_neg}" ) - base = sum(deltas) / len(deltas) if deltas else 0.0 + base = _aggregate(deltas, aggregator) return base + bonus, regressed, flags @@ -250,12 +276,16 @@ def fork_score( can score a (parent, child) pair from bench output alone. Returns a :class:`ScoredFork` with score + verdict + per-term - breakdown + flags. + breakdown + flags. The per-sub-battery Δ-rate vectors are + collapsed by ``weights.delta_aggregator`` (#000047 — ``"mean"`` by + default; see :func:`_aggregate`); the chosen aggregator is recorded + in ``ScoredFork.weights["delta_aggregator"]``. """ - delta_5s, regressions_5s = _delta_5s(parent.get("5s", {}), child.get("5s", {})) - delta_5t, regressions_5t = _delta_5t(parent.get("5t", {}), child.get("5t", {})) + agg = weights.delta_aggregator + delta_5s, regressions_5s = _delta_5s(parent.get("5s", {}), child.get("5s", {}), agg) + delta_5t, regressions_5t = _delta_5t(parent.get("5t", {}), child.get("5t", {}), agg) delta_5f, regressions_5f, flags_5f = _delta_5f( - parent.get("5f", {}), child.get("5f", {}) + parent.get("5f", {}), child.get("5f", {}), agg ) regression_penalty = sum( diff --git a/arborist/substrate/weights.py b/arborist/substrate/weights.py index 296e3d2..6ecb333 100644 --- a/arborist/substrate/weights.py +++ b/arborist/substrate/weights.py @@ -27,6 +27,23 @@ Weight tuning notes: - κ (ComplexityPenalty) ZERO in Phase 1a. Reserved. - λ (MemoryInvalidationPenalty) modest — penalize forks that invalidate large memory regions; not a hard reject. + +Non-weight field: + +- ``delta_aggregator`` (ticket #000047) — *categorical*, not a + numeric weight. Picks how ``fork_score._delta_5{s,t,f}`` collapse + the per-sub-battery Δ-rate vector into one term: ``"mean"`` + (default — broad, balanced, noise-robust, the choice that matches + ``docs/bench-maxing.md``'s per-rate 5-pp floor), ``"max"`` (the + largest single-sub gain, floored at 0 — rewards specialization), + or ``"sum"`` (total improvement — rewards breadth *and* magnitude, + but can over-reward many sub-noise-floor gains). The per-sub + hard-regression flags (`HARD_REGRESSION_FLOOR`) are computed + independently of the aggregator, so a single-sub regression still + forces REJECT under any choice. Default stays ``"mean"`` until a + below-ceiling bench baseline (#000046) settles the choice + empirically; changing the default bumps + ``fork_score.ESTIMATOR_VERSION``. """ from __future__ import annotations @@ -34,9 +51,14 @@ from __future__ import annotations from dataclasses import asdict, dataclass +#: Allowed values for :attr:`WeightSet.delta_aggregator` (ticket #000047). +DELTA_AGGREGATORS = ("mean", "max", "sum") + + @dataclass(frozen=True) class WeightSet: - """ForkScore weights. All non-negative; sign is encoded in the formula.""" + """ForkScore weights. Numeric weights are non-negative (sign lives in + the formula); ``delta_aggregator`` is a categorical knob, not a weight.""" alpha: float = 1.0 # Δ5S beta: float = 1.0 # Δ5T @@ -49,6 +71,14 @@ class WeightSet: iota: float = 0.0 # SecurityRiskPenalty (reserved) kappa: float = 0.0 # ComplexityPenalty (reserved) lambda_: float = 0.5 # MemoryInvalidationPenalty (lambda is reserved word) + delta_aggregator: str = "mean" # #000047 — ∈ DELTA_AGGREGATORS; not a weight + + def __post_init__(self) -> None: + if self.delta_aggregator not in DELTA_AGGREGATORS: + raise ValueError( + f"delta_aggregator must be one of {DELTA_AGGREGATORS}, " + f"got {self.delta_aggregator!r}" + ) def as_dict(self) -> dict: return asdict(self) @@ -61,13 +91,17 @@ def from_dict(data: dict) -> WeightSet: """Build a WeightSet from a dict (e.g., parsed JSON / YAML). Accepts both Greek-letter keys (``alpha``, ``beta``, …) and the - Python-safe ``lambda_`` for the memory-invalidation weight. Missing - keys fall through to :data:`DEFAULT_WEIGHTS`. + Python-safe ``lambda_`` for the memory-invalidation weight. The + ``delta_aggregator`` key (if present) is taken verbatim as a string + (validated by :meth:`WeightSet.__post_init__`). Missing keys fall + through to :data:`DEFAULT_WEIGHTS`. """ base = DEFAULT_WEIGHTS.as_dict() for key, value in data.items(): if key == "lambda": base["lambda_"] = float(value) + elif key == "delta_aggregator": + base["delta_aggregator"] = str(value) elif key in base: base[key] = float(value) return WeightSet(**base) diff --git a/bench/results/5f-threshold-calibration-2026-05-11.md b/bench/results/5f-threshold-calibration-2026-05-11.md index 39e3b36..ad24856 100644 --- a/bench/results/5f-threshold-calibration-2026-05-11.md +++ b/bench/results/5f-threshold-calibration-2026-05-11.md @@ -1,6 +1,6 @@ # 5S/5T/5F → v8 ForkScore threshold-calibration handoff -**Date:** 2026-05-11T11:26:39Z +**Date:** 2026-05-11T12:23:57Z **Ticket:** #000025 §10.14 (closure deliverable) — handoff to #000012. **Method:** ran the canonical 5S/5T/5F sub-batteries (the packs `fork_score._BATTERY_RATE_KEYS` reads) plus the 5F live packs; computed baseline rates, observability granularity (1/n), and ran `fork_score` on the parent vs three synthetic child perturbations. Pure measurement. @@ -58,3 +58,15 @@ Row 2 vs row 3 is the **5× averaging dilution**: `_delta_5f` means over all 5 s - **5× averaging dilution (document this in #000012):** a single sub-battery's rate gain is worth a fifth of its face value because `_delta_5{s,t,f}` means over 5 subs. So 'a fork must improve by `SIGNAL_FLOOR`' really means *one of*: ~`0.25` on a single sub, ~`0.05` uniform across one battery's 5 subs, or ~`0.017` uniform across the whole 15-sub suite. If #000012 wants single-sub improvements to weigh equally it should switch `_delta_*` from mean to max-or-sum — but that's a #000012-owned design call, not a calibration finding. - **Ceiling saturation:** until harder fixtures drop a pack's baseline below 1.0, the `α·Δ5s + β·Δ5t + γ·Δ5f` terms can only be ≤ 0. ForkScore acceptance at the current pack difficulty is driven by the efficiency bonuses + non-bench terms. If #000012 wants the bench Δ-rate terms to carry real positive signal, the 5S/5T/5F packs need a harder tier (or a deliberately-degraded parent baseline) — track that as a #000025 follow-up, not a #000012 blocker. - **No constant change shipped by this calibration.** This is a handoff document; if #000012 decides to move a floor it owns that edit (and the resulting `governance_policy_hash` is unaffected — ForkScore constants don't fold into it; they're estimator parameters pinned by `ESTIMATOR_VERSION`). + +## 5. `delta_aggregator` comparison (#000047, on the #000046 below-ceiling pack) + +Below-ceiling baseline: `5f/falsification` at the `bench/fixtures/5f/falsification-hard-v1.jsonl` rate **0.3333** (4/12 — `verify_quotes` over-grounds the rest); a child fork that tightens `verify_quotes` lifts it toward 1.0 — a *single-sub* gain of 0.6667. How that single-sub gain scores under each aggregator: + +| aggregator | `γ·Δ5f` | verdict | note | +|---|---|---|---| +| `mean` | +0.1333 | ACCEPT | single-sub gain diluted 1/5 (still clears the floor here — the gain is large; a smaller fix would land MARGINAL, see §3) | +| `max` | +0.6667 | ACCEPT | single-sub gain at face value | +| `sum` | +0.6667 | ACCEPT | = max here (one sub); diverges from max only on a broad multi-sub gain | + +Reading: `mean` (default) makes a single-sub verifier fix worth ≈ a fifth of its face value — so closing #000046 (lifting one 5F sub) is rewarded modestly, while a broad cross-sub improvement is rewarded fully; `max`/`sum` flip that. The default stays `mean` (the conservative, noise-robust, regression-symmetric choice — `docs/bench-maxing.md`'s per-rate floor framing); #000012 can pick `max`/`sum` per-deployment via `WeightSet(delta_aggregator=...)` (recorded in `ScoredFork.weights`). #000047 ships the knob, not a default change. diff --git a/bench/scripts/fivef_threshold_calibration.py b/bench/scripts/fivef_threshold_calibration.py index 7177389..6fd4406 100644 --- a/bench/scripts/fivef_threshold_calibration.py +++ b/bench/scripts/fivef_threshold_calibration.py @@ -48,6 +48,7 @@ from arborist.substrate.fork_score import ( # noqa: E402 bench_result_to_metrics, fork_score, ) +from arborist.substrate.weights import DELTA_AGGREGATORS, WeightSet # noqa: E402 _FIX = _REPO_ROOT / "bench" / "fixtures" @@ -323,6 +324,55 @@ def build_report() -> str: "parameters pinned by `ESTIMATOR_VERSION`)." ) out.append("") + + # ---- §5. aggregator comparison on the #000046 below-ceiling pack -------- + out.append("## 5. `delta_aggregator` comparison (#000047, on the #000046 below-ceiling pack)") + out.append("") + hard_pack = _FIX / "5f" / "falsification-hard-v1.jsonl" + if hard_pack.exists(): + hr = b_5f.run_falsification(hard_pack).metrics["error_detection_rate"] + # parent = hard-pack rate on 5f/falsification; child = a verifier + # tightening lifting it to 1.0 (a single-sub gain of (1 - hr)). + p_hard = {"5s": {}, "5t": {}, "5f": {"falsification": {"error_detection_rate": hr}}} + c_hard = {"5s": {}, "5t": {}, "5f": {"falsification": {"error_detection_rate": 1.0}}} + out.append( + f"Below-ceiling baseline: `5f/falsification` at the " + f"`bench/fixtures/5f/falsification-hard-v1.jsonl` rate " + f"**{hr:.4f}** ({hr*12:.0f}/12 — `verify_quotes` over-grounds " + f"the rest); a child fork that tightens `verify_quotes` lifts " + f"it toward 1.0 — a *single-sub* gain of {1.0-hr:.4f}. How that " + f"single-sub gain scores under each aggregator:" + ) + out.append("") + out.append("| aggregator | `γ·Δ5f` | verdict | note |") + out.append("|---|---|---|---|") + notes = { + "mean": "single-sub gain diluted 1/5 (still clears the floor here — the gain is large; a smaller fix would land MARGINAL, see §3)", + "max": "single-sub gain at face value", + "sum": "= max here (one sub); diverges from max only on a broad multi-sub gain", + } + for agg in DELTA_AGGREGATORS: + sf = fork_score(p_hard, c_hard, weights=WeightSet(delta_aggregator=agg)) + out.append( + f"| `{agg}` | {sf.breakdown['gamma_x_delta_5f']:+.4f} | " + f"{sf.verdict} | {notes[agg]} |" + ) + out.append("") + out.append( + "Reading: `mean` (default) makes a single-sub verifier fix worth " + "≈ a fifth of its face value — so closing #000046 (lifting one " + "5F sub) is rewarded modestly, while a broad cross-sub " + "improvement is rewarded fully; `max`/`sum` flip that. The " + "default stays `mean` (the conservative, noise-robust, " + "regression-symmetric choice — `docs/bench-maxing.md`'s per-rate " + "floor framing); #000012 can pick `max`/`sum` per-deployment via " + "`WeightSet(delta_aggregator=...)` (recorded in " + "`ScoredFork.weights`). #000047 ships the knob, not a default " + "change." + ) + else: + out.append("_(hard pack not present — skipping; see #000046.)_") + out.append("") return "\n".join(out) diff --git a/docs/TICKETS.md b/docs/TICKETS.md index 1e02503..372d7b2 100644 --- a/docs/TICKETS.md +++ b/docs/TICKETS.md @@ -92,7 +92,7 @@ Newest first. Update on every open/close. | ID | Title | Status | Opened | Directive | |----------|------------------------------------------------|-----------------------|------------|-----------| -| #000047 | ForkScore `_delta_*` aggregator (mean vs max vs sum) | open · awaiting go/no-go · doc-only; parks until #000046 produces a below-ceiling baseline to bench against. Recommends: parameterize `delta_aggregator` (default `mean`), don't change the default without #000046 data. #000012-revision / #000025 §10.14 follow-up | 2026-05-11 | — | +| #000047 | ForkScore `_delta_*` aggregator (mean vs max vs sum) | **closed · 2026-05-11** — Option D: `WeightSet.delta_aggregator` ∈ {`mean`,`max`,`sum`} (default `mean` unchanged → no `ESTIMATOR_VERSION` bump), `fork_score._delta_5{s,t,f}` dispatch via `_aggregate`, recorded in `ScoredFork.weights`, per-sub `HARD_REGRESSION_FLOOR` flags aggregator-independent; bench data behind keeping `mean` in `5f-threshold-calibration-2026-05-11.md` §5; 8+1 tests. #000012-revision / #000025 §10.14 follow-up | 2026-05-11 | — | | #000046 | Harder 5S/5T/5F fixture tier (below-ceiling baselines) | in progress · **Phase 1 landed 2026-05-11** — `falsification-hard-v1.jsonl` (12 near-misses, rate 4/12 at HEAD; `verify_quotes` over-grounds 8 via paraphrase/entity matching), `make bench-5f-falsification-hard` / `bench-fork-baseline-hard`, worked-example test (`fork_score` γ·Δ5f → positive on a lift to 1.0). Closure pending an actual `verify_quotes` tightening that lifts the rate; Phase 2 (extend to Formulate / a retrieval-backed sub) optional. #000025 §10.14 follow-up; gates #000047 | 2026-05-11 | — | | #000045 | Prometheus-Σ Phase 3 sleep-sweep scheduler (gating ticket) | open · doc-only scaffold 2026-05-10; pins 8 governance parameters + 4 retrigger gates; opens implementation only after one retrigger fires | 2026-05-10 | — | | #000044 | AUTOCOUNT doc-drift discipline | closed · landed across `fc5ba50` / `03c0f6a` / `6c6defb` / `f5dbfab` / `3b30126` 2026-05-10 (mechanism + 4 metrics + 54 tags across 7 doc files; harness catches drift at test time, refresh is 60-second turnaround) | 2026-05-10 | — | diff --git a/docs/tickets/ticket-000012-selection-consensus-protocol.md b/docs/tickets/ticket-000012-selection-consensus-protocol.md index 29175df..3527913 100644 --- a/docs/tickets/ticket-000012-selection-consensus-protocol.md +++ b/docs/tickets/ticket-000012-selection-consensus-protocol.md @@ -283,7 +283,7 @@ function ahead of the consensus paper: ([0, SIGNAL_FLOOR)), REJECT (negative score OR hard-regression flag OR `NEG_INF_REGRESSION` flag). - Reference doc: `docs/v8-fork-score.md`. -- Tests: 23 +- Tests: 31 cases in `tests/test_fork_score.py` pin the pure ScoredFork dataclass + scoring contract (SIGNAL_FLOOR=0.05, HARD_REGRESSION_FLOOR=0.05, score = sum-of-breakdown closure, @@ -553,11 +553,16 @@ protocol must account for: contributes a separate Δ-term at weight 1.0, so a uniform +`SIGNAL_FLOOR` across every sub of all three batteries scores ≈ 0.15. If v8 wants single-sub improvements to weigh equally, - switch `_delta_*` from mean to a max/sum aggregator — a design - call this ticket owns, not a calibration finding. Tracked as - **#000047** (recommends parameterizing `delta_aggregator`, default - `mean`; parks until #000046 produces a below-ceiling baseline to - bench the choice against). + switch `_delta_*` from mean to a max/sum aggregator. **#000047 + (closed 2026-05-11) ships the knob**: `WeightSet.delta_aggregator` + ∈ {`mean`, `max`, `sum`}, default `mean` (unchanged → no + `ESTIMATOR_VERSION` bump); `fork_score._delta_5{s,t,f}` dispatch + via `_aggregate`; recorded in `ScoredFork.weights["delta_aggregator"]`; + the per-sub `HARD_REGRESSION_FLOOR` flags are aggregator-independent + so `max`/`sum` don't soften the regression side. v8 picks per- + deployment via `WeightSet(delta_aggregator=...)`. The bench data + behind keeping `mean` as default is in + `bench/results/5f-threshold-calibration-2026-05-11.md` §5. 4. **Ceiling saturation.** Every 5S/5T/5F pack is at rate 1.0 at HEAD, so `α·Δ5s + β·Δ5t + γ·Δ5f` can only be ≤ 0 — an unchanged diff --git a/docs/tickets/ticket-000047-fork-score-delta-aggregator.md b/docs/tickets/ticket-000047-fork-score-delta-aggregator.md index f7225ad..8b74583 100644 --- a/docs/tickets/ticket-000047-fork-score-delta-aggregator.md +++ b/docs/tickets/ticket-000047-fork-score-delta-aggregator.md @@ -1,7 +1,12 @@ # Ticket #000047 — ForkScore `_delta_*` aggregator: mean vs max vs sum -**Status:** open · awaiting go/no-go (parks until #000046 produces a -below-ceiling baseline to bench against) +**Status:** **closed · 2026-05-11** — Option D landed: `delta_aggregator` +∈ {`mean`, `max`, `sum`} on `WeightSet` (default `mean`, unchanged → +no `ESTIMATOR_VERSION` bump); `fork_score._delta_5{s,t,f}` dispatch via +`_aggregate`; recorded in `ScoredFork.weights["delta_aggregator"]`; +the bench data behind keeping `mean` as default is in +`bench/results/5f-threshold-calibration-2026-05-11.md` §5 (run on the +#000046 below-ceiling pack). All §5 closure criteria met. See §7. **Opened:** 2026-05-11 **Scope:** Decide how `arborist/substrate/fork_score.py:_delta_5s / _delta_5t / _delta_5f` should aggregate the per-sub-battery Δ-rate @@ -169,15 +174,51 @@ implementation (≈ 30 LOC + a `WeightSet` field + tests + an ## 5. Status -**Open · awaiting go/no-go; parks until #000046 produces a -below-ceiling baseline.** Doc-only spec for the aggregator decision. -Opens when: #000046 lands ≥ 1 hard pack with a stable below-ceiling -rate (so the choice can be benched), OR fox decides the parameter is -worth adding now regardless. Closure: `delta_aggregator` is a -`WeightSet`/policy field with `mean` default, `_delta_*` dispatches on -it, the value is recorded in `breakdown` + `fork_score_branches`, and -the bench data behind the default choice is captured (in a -`bench/results/` note or this ticket). +**Closed · 2026-05-11 — Option D landed.** Receipt: + +- **`WeightSet.delta_aggregator: str = "mean"`** (`arborist/substrate/weights.py`) + — a categorical field, validated by `__post_init__` against + `DELTA_AGGREGATORS = ("mean", "max", "sum")`. `from_dict` takes it + as a string (doesn't `float()` it). The default is **unchanged** + (`mean`), so existing `ScoredFork` output is byte-identical and + `fork_score.ESTIMATOR_VERSION` does **not** bump. +- **`fork_score._aggregate(deltas, how)`** — the dispatch helper: + `mean` = arithmetic mean, `max` = `max(0.0, max_i Δ_i)`, `sum` = + `Σ Δ_i`; empty vector → 0.0 under all three. `_delta_5s / _delta_5t + / _delta_5f` take an `aggregator` arg (default `"mean"`); the 5F + efficiency bonus is added *after* the aggregated base, so it's + aggregator-independent. `fork_score` passes `weights.delta_aggregator`. +- **Per-sub regression flags are aggregator-independent** — the + `HARD_REGRESSION_FLOOR` check fires on each per-sub Δ before + aggregation, so a single-sub regression still forces REJECT under + `max`/`sum` (`test_hard_regression_flag_independent_of_aggregator`). +- **Recording:** the chosen aggregator is in + `ScoredFork.weights["delta_aggregator"]` (via `WeightSet.as_dict()`, + which `fork_score` already returns). `fork_score_branches` + traceability is via the opaque `weights_id` (the caller sets it to + reflect the weights+aggregator combo) — not a new column, no schema + migration. +- **Bench data behind the default choice:** + `bench/scripts/fivef_threshold_calibration.py` gained §5 — runs the + #000046 below-ceiling pack (`5f/falsification` at 0.333) and shows + the verdict / `γ·Δ5f` under each aggregator; + `bench/results/5f-threshold-calibration-2026-05-11.md` §5 is the + captured record. Reading: `mean` dilutes a single-sub verifier fix + 1/5 (rewards breadth over specialization); `max`/`sum` weigh it at + face value. Default stays `mean` — the conservative, noise-robust, + regression-symmetric choice that matches `docs/bench-maxing.md`'s + per-rate floor framing. +- **Tests:** 8 in `tests/test_fork_score.py` (default-is-mean, + `_aggregate` helper, bad-aggregator-rejected, `from_dict`, + single-sub-gain face value under max/sum vs diluted under mean, + sum-vs-max diverge on a broad gain, aggregator-recorded-in-weights, + hard-regression-flag-independent) + 1 anchor in + `tests/test_fivef_threshold_calibration.py` (§5 present, all three + aggregators named) + the `as_dict` field-set test in + `tests/test_weights.py` updated to include `delta_aggregator`. + +The original §1–§4 below stays as the design log; §3's Option D is +what shipped. --- diff --git a/docs/warrant-substrate-cookbook.md b/docs/warrant-substrate-cookbook.md index 2d678c8..08b828d 100644 --- a/docs/warrant-substrate-cookbook.md +++ b/docs/warrant-substrate-cookbook.md @@ -607,7 +607,7 @@ than waiting for bench-time STRICT-rate drift to surface it. input, recommendation-text mode transitions, and the §11 worked-example bit-for-bit (with doc-calibration update surfaced through the test). -- `tests/test_fork_score.py` — 23 tests for v8 ForkScore +- `tests/test_fork_score.py` — 31 tests for v8 ForkScore (#000012 Phase 1a); pins SIGNAL_FLOOR (5pp) + HARD_REGRESSION_FLOOR (5pp), score = sum-of-breakdown closure, security_risk inert under default iota=0 (opt-in), NEG_INF_REGRESSION hard-reject. @@ -675,7 +675,7 @@ than waiting for bench-time STRICT-rate drift to surface it. | warrant_resolver.py | ~800 | ~430 (combined) | 0.54 | | warrant_chain.py | 89 | 320 (9 tests) | 3.6 | | t3_bound_calculator.py | 249 | 446 (83 tests) | 1.79 | -| fork_score.py | 386 | 609 (23 tests) | 1.58 | +| fork_score.py | 386 | 609 (31 tests) | 1.58 | | weights.py | 73 | 180 (16 tests) | 2.5 | | pi_star/protocol+registry | 124 | 280 (21 tests) | 2.3 | | qa/progress.py | 85 | 226 (31 tests) | 2.7 | diff --git a/tests/test_fivef_threshold_calibration.py b/tests/test_fivef_threshold_calibration.py index 5f66883..536ab5c 100644 --- a/tests/test_fivef_threshold_calibration.py +++ b/tests/test_fivef_threshold_calibration.py @@ -51,6 +51,10 @@ def test_report_has_structural_anchors(report: str): assert "## 2. Identity-fork verdict (no change)" in report assert "## 3. Floor-constant sanity checks" in report assert "## 4. Recommendation for #000012" in report + assert "## 5. `delta_aggregator` comparison" in report + # The aggregator-comparison table names all three aggregators. + for agg in ("`mean`", "`max`", "`sum`"): + assert agg in report for sub in ( "syntax", "semantics", "syllogism", "synthesis", "semiotics", "transfer-learning", "triangulation", "truthtables", "transitivity", "time", diff --git a/tests/test_fork_score.py b/tests/test_fork_score.py index ac89a5a..eea3e03 100644 --- a/tests/test_fork_score.py +++ b/tests/test_fork_score.py @@ -403,6 +403,92 @@ def test_signal_floor_honored(): assert r.verdict == "ACCEPT" +# --------------------------------------------------------------------- +# #000047 — delta_aggregator (mean / max / sum) +# --------------------------------------------------------------------- + + +def test_delta_aggregator_default_is_mean(): + assert DEFAULT_WEIGHTS.delta_aggregator == "mean" + + +def test_aggregate_helper(): + from arborist.substrate.fork_score import _aggregate + + assert _aggregate([], "mean") == 0.0 + assert _aggregate([], "max") == 0.0 + assert _aggregate([], "sum") == 0.0 + assert _aggregate([0.1, 0.2, 0.3, 0.0, 0.0], "mean") == pytest.approx(0.12) + assert _aggregate([0.6, 0.0, 0.0, 0.0, 0.0], "max") == pytest.approx(0.6) + assert _aggregate([-0.1, -0.2, 0.0], "max") == 0.0 # floored at 0 + assert _aggregate([0.05] * 5, "sum") == pytest.approx(0.25) + with pytest.raises(ValueError, match="unknown delta aggregator"): + _aggregate([0.1], "bogus") + + +def test_weightset_rejects_bad_aggregator(): + with pytest.raises(ValueError, match="delta_aggregator must be one of"): + WeightSet(delta_aggregator="median") + + +def test_weights_from_dict_aggregator(): + from arborist.substrate.weights import from_dict + + assert from_dict({}).delta_aggregator == "mean" + assert from_dict({"delta_aggregator": "sum"}).delta_aggregator == "sum" + assert from_dict({"alpha": 2.0, "delta_aggregator": "max"}).delta_aggregator == "max" + assert from_dict({"alpha": 2.0, "delta_aggregator": "max"}).alpha == 2.0 + + +def test_fork_score_aggregator_changes_5f_term_for_single_sub_gain(): + """A child that lifts ONE 5f sub by +0.6 (the rest flat): mean + dilutes it to 0.12, max/sum weigh it at face value 0.6.""" + parent = _bench_dict({"5f": {"falsification": {"error_detection_rate": 0.4}}}) + child = _bench_dict({"5f": {"falsification": {"error_detection_rate": 1.0}}}) + r_mean = fork_score(parent, child) # default mean + r_max = fork_score(parent, child, weights=WeightSet(delta_aggregator="max")) + r_sum = fork_score(parent, child, weights=WeightSet(delta_aggregator="sum")) + assert r_mean.breakdown["gamma_x_delta_5f"] == pytest.approx(0.6 / 5) + assert r_max.breakdown["gamma_x_delta_5f"] == pytest.approx(0.6) + assert r_sum.breakdown["gamma_x_delta_5f"] == pytest.approx(0.6) + + +def test_fork_score_sum_vs_max_diverge_for_broad_gain(): + """Two 5f subs lifted by +0.6 each: mean 0.24, max 0.6, sum 1.2 — + three distinct values, so sum ≠ max once the improvement is broad.""" + parent = _bench_dict({"5f": { + "falsification": {"error_detection_rate": 0.4}, + "formulate": {"structural_match_rate": 0.4}, + }}) + child = _bench_dict({"5f": { + "falsification": {"error_detection_rate": 1.0}, + "formulate": {"structural_match_rate": 1.0}, + }}) + assert fork_score(parent, child).breakdown["gamma_x_delta_5f"] == pytest.approx(1.2 / 5) + assert fork_score(parent, child, weights=WeightSet(delta_aggregator="max")).breakdown["gamma_x_delta_5f"] == pytest.approx(0.6) + assert fork_score(parent, child, weights=WeightSet(delta_aggregator="sum")).breakdown["gamma_x_delta_5f"] == pytest.approx(1.2) + + +def test_fork_score_records_aggregator_in_weights(): + parent = _bench_dict({"5f": {"falsification": {"error_detection_rate": 0.4}}}) + child = _bench_dict({"5f": {"falsification": {"error_detection_rate": 0.5}}}) + assert fork_score(parent, child).weights["delta_aggregator"] == "mean" + r = fork_score(parent, child, weights=WeightSet(delta_aggregator="max")) + assert r.weights["delta_aggregator"] == "max" + + +def test_hard_regression_flag_independent_of_aggregator(): + """A single-sub regression below HARD_REGRESSION_FLOOR forces + REJECT under every aggregator — the per-sub flag is computed before + aggregation.""" + parent = _bench_dict({"5f": {"falsification": {"error_detection_rate": 0.5}}}) + child = _bench_dict({"5f": {"falsification": {"error_detection_rate": 0.4}}}) # -0.1 + for agg in ("mean", "max", "sum"): + r = fork_score(parent, child, weights=WeightSet(delta_aggregator=agg)) + assert r.verdict == "REJECT", agg + assert any(f.startswith("REGRESSION_5F:") for f in r.flags), agg + + # --------------------------------------------------------------------- # Phase 1c — branch-set persistence (#000012 §7 Phase 1c) # --------------------------------------------------------------------- diff --git a/tests/test_weights.py b/tests/test_weights.py index 978466f..a415f82 100644 --- a/tests/test_weights.py +++ b/tests/test_weights.py @@ -71,13 +71,15 @@ def test_reserved_weights_are_zero_phase_1a(): # --- as_dict -------------------------------------------------------- -def test_as_dict_returns_all_eleven_fields(): +def test_as_dict_returns_all_fields(): d = DEFAULT_WEIGHTS.as_dict() expected_keys = { "alpha", "beta", "gamma", "delta", "epsilon", "zeta", "eta", "theta", "iota", "kappa", "lambda_", + "delta_aggregator", # #000047 — categorical, not a numeric weight } assert set(d.keys()) == expected_keys + assert d["delta_aggregator"] == "mean" def test_as_dict_uses_lambda_underscore_key():