feat(adr-0249): P5 arithmetic-chain lift domain on real GSM8K holdout

Wires the turn-program executor into the generalized-lift instrument and
measures it against the REAL GSM8K dev holdout (evals/gsm8k_math/dev, never the
templated cases), using each problem's ground_truth_graph. Baseline is a
symbolic fold of the SAME compiled turn program: both paths consume identical
problems, the corridor relaxes each step, the baseline folds it numerically.

Honest result on the sealed 50-case dev holdout: 26 problems are Tier-1 affine
single-accumulator ingestible and solved wrong=0; the corridor matches the
symbolic fold on every one (PARITY, delta=0) — the field matches arithmetic, it
does not beat it, so this is real-holdout coverage, NOT a manufactured lift. The
24 refused (all not_single_accumulator / multi-entity) are the recorded Tier-2
frontier, never silently dropped.

Corrects the now-stale scope-limitation note ('no reader-to-Hamiltonian compiler
exists…') — the compiler exists as of this arc; the note now records real-data
Tier-1 coverage and the multi-entity frontier. wrong_zero_guard_held extended to
bind both exact-regime domains (deductive flagship + arithmetic).

10/10 pins (5 new + 5 existing instrument tests still green).

[Verification]: uv run python -m pytest tests/test_adr_0249_arithmetic_lift.py tests/test_generalized_lift_instrument.py -q
This commit is contained in:
Shay 2026-07-18 13:07:40 -07:00
parent 805752db72
commit 324db57373
2 changed files with 192 additions and 6 deletions

View file

@ -32,8 +32,10 @@ composition frontier, and this instrument is the harness waiting for it.
from __future__ import annotations from __future__ import annotations
import json
from dataclasses import dataclass from dataclasses import dataclass
from itertools import product from itertools import product
from pathlib import Path
from typing import Any, Sequence from typing import Any, Sequence
import numpy as np import numpy as np
@ -56,8 +58,15 @@ from core.physics.linguistic_readback import (
from core.physics.sensorium_wave_feed import ModalityPacket from core.physics.sensorium_wave_feed import ModalityPacket
from core.physics.wave_manifold import WaveManifold from core.physics.wave_manifold import WaveManifold
from generate.logic_equivalence import Verdict, check_equivalence from generate.logic_equivalence import Verdict, check_equivalence
from generate.math_problem_graph import MathGraphError, graph_from_dict
from vocab.manifold import VocabManifold from vocab.manifold import VocabManifold
from evals.turn_program import (
TurnProgramError,
compile_turn_program,
execute_turn_program,
)
__all__ = [ __all__ = [
"DomainOutcome", "DomainOutcome",
"LiftInstrumentReport", "LiftInstrumentReport",
@ -65,6 +74,7 @@ __all__ = [
"run_propositional_domain", "run_propositional_domain",
"run_recognition_domain", "run_recognition_domain",
"run_multimodal_domain", "run_multimodal_domain",
"run_arithmetic_chain_domain",
] ]
# Hot-band energy axes (existing E3/E4 precedent; caller-supplied, never invented). # Hot-band energy axes (existing E3/E4 precedent; caller-supplied, never invented).
@ -450,22 +460,130 @@ def run_multimodal_domain() -> DomainOutcome:
# --- Composed instrument --------------------------------------------------------------- # --- Composed instrument ---------------------------------------------------------------
# --- Domain D: multi-step arithmetic on the real GSM8K dev holdout (ADR-0249) ----------
# The sealed GSM8K dev holdout — real problems, never the templated cases.
_DEV_HOLDOUT = Path(__file__).parent / "gsm8k_math" / "dev" / "cases.jsonl"
def _load_dev_cases() -> tuple[dict[str, Any], ...]:
if not _DEV_HOLDOUT.exists():
return ()
with _DEV_HOLDOUT.open() as handle:
return tuple(json.loads(line) for line in handle if line.strip())
def _symbolic_fold(seed: float, steps) -> float:
"""The baseline: solve the SAME compiled turn program by plain arithmetic.
Corridor and baseline consume the identical `TurnProgram`; the corridor
relaxes each step, the baseline folds it numerically. The honest question
is whether field relaxation matches arithmetic on the same problem.
"""
answer = float(seed)
for step in steps:
answer = step.scale * answer + step.offset
return answer
def run_arithmetic_chain_domain() -> DomainOutcome:
"""Corridor turn-program executor vs symbolic fold on real GSM8K problems.
Both paths consume the graph's `ground_truth_graph` compiled to a turn
program; non-Tier-1 graphs (multi-entity, transfer, rate, ) are refused by
BOTH and recorded as the composition frontier never silently dropped. The
expected honest verdict is PARITY with wrong=0: the field does not beat
arithmetic at arithmetic; the result is real-holdout coverage, not a lift.
"""
corridor_correct = corridor_wrong = corridor_refused = 0
baseline_correct = baseline_wrong = baseline_refused = 0
rows: list[dict[str, Any]] = []
cases = _load_dev_cases()
for case in cases:
case_id = case.get("id", "")
gold_raw = case.get("expected_answer")
graph_dict = case.get("ground_truth_graph")
try:
graph = graph_from_dict(graph_dict) if graph_dict else None
except (MathGraphError, KeyError, TypeError, ValueError):
graph = None
try:
program = compile_turn_program(graph) if graph is not None else None
except TurnProgramError as refusal:
corridor_refused += 1
baseline_refused += 1
rows.append({"case_id": case_id, "ingested": False, "refusal": refusal.reason})
continue
if program is None:
corridor_refused += 1
baseline_refused += 1
rows.append({"case_id": case_id, "ingested": False, "refusal": "no_graph"})
continue
gold = None if gold_raw is None else float(gold_raw)
corridor_answer = execute_turn_program(program).answer
baseline_answer = _symbolic_fold(program.seed, program.steps)
corridor_ok = gold is not None and abs(corridor_answer - gold) < 1e-4
baseline_ok = gold is not None and abs(baseline_answer - gold) < 1e-4
corridor_correct += int(corridor_ok)
corridor_wrong += int(not corridor_ok)
baseline_correct += int(baseline_ok)
baseline_wrong += int(not baseline_ok)
rows.append(
{
"case_id": case_id,
"ingested": True,
"gold": gold,
"corridor_answer": corridor_answer,
"baseline_answer": baseline_answer,
"corridor_ok": corridor_ok,
}
)
ingested = corridor_correct + corridor_wrong
return DomainOutcome(
domain_id="arithmetic-chain",
n_cases=len(cases),
corridor_correct=corridor_correct,
corridor_wrong=corridor_wrong,
corridor_refused=corridor_refused,
baseline_correct=baseline_correct,
baseline_wrong=baseline_wrong,
baseline_refused=baseline_refused,
notes=(
f"Real GSM8K dev holdout: {ingested}/{len(cases)} problems are Tier-1 "
f"affine single-accumulator ingestible and solved wrong=0; the rest are "
f"refused (multi-entity/transfer/rate/…) — the recorded Tier-2 frontier.",
"Baseline is a symbolic fold of the SAME compiled turn program; PARITY "
"with wrong=0 is the honest outcome — the field matches arithmetic, it "
"does not beat it. The result is real-holdout coverage, not a lift.",
),
cases=tuple(rows),
)
def run_generalized_lift_instrument() -> LiftInstrumentReport: def run_generalized_lift_instrument() -> LiftInstrumentReport:
outcomes = ( outcomes = (
run_propositional_domain(), run_propositional_domain(),
run_recognition_domain(), run_recognition_domain(),
run_multimodal_domain(), run_multimodal_domain(),
run_arithmetic_chain_domain(),
) )
propositional = outcomes[0] propositional = outcomes[0]
arithmetic = outcomes[3]
return LiftInstrumentReport( return LiftInstrumentReport(
outcomes=outcomes, outcomes=outcomes,
wrong_zero_guard_held=(propositional.corridor_wrong == 0), # Both exact-regime domains (deductive flagship + arithmetic) preserve wrong=0.
wrong_zero_guard_held=(propositional.corridor_wrong == 0 and arithmetic.corridor_wrong == 0),
honest_null=all(o.delta_correct <= 0 for o in outcomes), honest_null=all(o.delta_correct <= 0 for o in outcomes),
scope_limitations=( scope_limitations=(
"GSM8K / natural-language arithmetic is NOT ingestible by corridor " "The reader→Hamiltonian compiler now exists for the affine "
"v1: no reader-to-Hamiltonian compiler exists beyond the <=5-atom " "single-accumulator arithmetic subset (ADR-0249): on the real GSM8K "
"propositional and quadratic-well domains. Recorded, not silently " "dev holdout it solves the Tier-1 subset wrong=0 (see the "
"dropped; that compiler is the composition frontier this " "arithmetic-chain domain). Non-affine / multi-entity GSM8K problems "
"instrument is waiting to measure.", "(transfer, rate, comparison, fraction, partition) remain the "
"recorded Tier-2 frontier — refused, never silently dropped.",
), ),
) )

View file

@ -0,0 +1,68 @@
"""ADR-0249 P5 — arithmetic-chain lift domain on the real GSM8K holdout.
Wires the turn-program executor into the generalized-lift instrument and
measures it against the *real* GSM8K dev holdout (never the templated cases),
with a symbolic fold of the same compiled program as the honest baseline. The
expected, honest outcome is PARITY with wrong=0 on the Tier-1 affine subset,
with the multi-entity remainder recorded not a manufactured lift.
"""
from __future__ import annotations
import json
import pytest
from evals.generalized_lift_instrument import (
run_arithmetic_chain_domain,
run_generalized_lift_instrument,
)
@pytest.fixture(scope="module")
def domain():
return run_arithmetic_chain_domain()
def test_domain_solves_real_gsm8k_wrong_zero(domain) -> None:
assert domain.domain_id == "arithmetic-chain"
assert domain.corridor_wrong == 0 # wrong=0 on real GSM8K, not just templates
assert domain.corridor_correct > 0 # the corridor ingests and solves real problems
assert domain.corridor_refused > 0 # and honestly refuses what it can't ingest
def test_every_ingested_case_is_correct(domain) -> None:
# The load-bearing claim: on real problems the corridor CAN ingest, it is
# exactly right — refusals, never wrong answers.
for row in domain.cases:
if row["ingested"]:
assert row["corridor_ok"] is True
def test_parity_with_symbolic_fold_honest_null(domain) -> None:
# The field matches arithmetic on the same compiled program; it does not
# beat it. PARITY is the honest verdict here, never an inflated lift.
assert domain.corridor_correct == domain.baseline_correct
assert domain.verdict == "PARITY"
assert domain.delta_correct == 0
def test_coverage_is_recorded_not_dropped(domain) -> None:
ingested = domain.corridor_correct + domain.corridor_wrong
assert ingested + domain.corridor_refused == domain.n_cases
assert domain.n_cases > 0
# A coverage tracker against the sealed dev holdout (50 cases, 26 Tier-1).
assert domain.corridor_correct == 26
assert domain.corridor_refused == 24
assert any("Tier-1" in note for note in domain.notes)
def test_domain_joins_full_report_and_scope_corrected() -> None:
report = run_generalized_lift_instrument()
ids = [o.domain_id for o in report.outcomes]
assert "arithmetic-chain" in ids
assert report.wrong_zero_guard_held # deductive + arithmetic both wrong=0
# The stale "no compiler exists" scope note must be gone; the compiler exists now.
joined = " ".join(report.scope_limitations)
assert "now exists" in joined
assert "no reader-to-Hamiltonian compiler exists" not in joined
json.dumps(report.as_dict()) # JSON-safe artifact