diff --git a/core/cli_test.py b/core/cli_test.py index bc88aba3..c01fe3be 100644 --- a/core/cli_test.py +++ b/core/cli_test.py @@ -246,6 +246,21 @@ TEST_SUITES: dict[str, tuple[str, ...]] = { "tests/test_vocab_trigger_instrument.py", "tests/test_grammar_roundtrip.py", "tests/test_lexicon_single_source.py", + # Phase 4 — which realizer serves, and what the other one's score + # means. Registered here deliberately: this file exists because a lane + # reported 117/117 for a function nothing calls, and a pin that no + # curated suite runs is the same defect one level up. Pure in-process + # lane replay, ~0.4s. + "tests/test_phase4_realizer_resolution.py", + # The grammar arc's agreement oracle and the tail-preservation + # invariant. It lived ONLY in the `cognition` suite, which is not on + # the AGENTS.md pre-push gate (smoke + deductive) — so every pin + # Phases 3 and 4 added to it, including the invariant that covers all + # twelve inflection branches, ran in no gate at all. Exactly the + # silent-red shape called out for test_adr_index.py above, and the + # reason smoke stayed at 621 across two PRs that added 13 tests. + # ~0.4s; it belongs with the other grammar pins. + "tests/test_realizer_quantifier_agreement.py", ), "full": ("tests/",), } diff --git a/docs/plans/grammar-unification-2026-07-26.md b/docs/plans/grammar-unification-2026-07-26.md index 95d2619d..001892e1 100644 --- a/docs/plans/grammar-unification-2026-07-26.md +++ b/docs/plans/grammar-unification-2026-07-26.md @@ -766,7 +766,14 @@ not guessed). Mutation now: | baseline | 13/13 | | phrase-head plural agreement | **4/13** | | `-es` stem rule | **12/13** | -| closed `ves` set | **11/13** | +| closed `ves` set | **12/13** | + +> **Correction (Phase 4).** The `ves` row was first recorded here as 11/13. Re-run +> against merged `main` by reverting the hunk in source — not by monkeypatch — it +> is **12/13**, and the single sensitive case is `gram_C14_p06` (`proofs`). The +> pin is still load-bearing; the recorded number was wrong. Corrected rather than +> left standing, because a mutation matrix nobody can reproduce is worth less +> than no matrix at all. The 4 survivors of the first mutation are the mass-noun and unchanged controls, which is the correct behaviour — `all evidence is grounded in truth` must *not* @@ -792,6 +799,55 @@ is taken is recorded with its reasoning. **Why this is gated:** (a) changes what users see and would move surface hashes. It is a serving change and belongs to Shay, not to this plan. +**RESULT — (b), and the framing of the choice was wrong.** + +The plan posed this as "which realizer is better". Reading the source says it is +not a quality question. `render_semantic`'s signature is +`(intent, subject, predicate, obj, secondary, language, root)` — there is **no +`negated`, `quantifier`, `tense` or `aspect` parameter**, and `realize_semantic` +never reads them off the step. So the serving writer does not merely inflect +worse; it cannot express content the `ArticulationStep` is carrying: + +``` +negated=False -> 'Knowledge is defined as opinion.' +negated=True -> 'Knowledge is defined as opinion.' <-- byte-identical +``` + +**It serves the affirmative of a negated proposition.** That is not a fluency +defect. It is the ADR-0261 §5.1 family — v1b served WRONG by dropping premises +it could not express — and it is pinned here as a defect, not fixed, because +fixing it changes served output. + +Scored on the identical contract across all seven corpora that a lane runs +through `realize_target`: + +| bucket | n | `realize_target` | `realize_semantic` | +|---|---|---|---| +| feature-bearing, single node | 214 | 207 | **49** | +| no features, multi-node | 100 | 100 | **3** | +| no features, single node — **CONTROL** | 33 | 33 | **33** | +| total | 347 | **340** | **85** | + +The control is what makes the rest mean anything. Every corpus hardcodes +`IntentTag.UNKNOWN`, so "the serving writer scores badly" could have been an +artifact of never giving it a real intent. On the 33 cases carrying nothing it +cannot express, the two realizers are **identical** — so the gap on the other +314 is the dropped features (214) and clause joining (100), not the intent. + +**What was delivered.** `grammatical_coverage/runner.py` now reports +`serving_accuracy` beside `accuracy`, computed by re-scoring the same cases +through `realize_semantic`; `english_fluency_ood` delegates to that `run_lane` +and gains it for free. The claim is restated at both places it was made — the +lane docstring and `grammar_roundtrip/contract.md` — as a claim about +**eval-only code**. Zero served bytes change, so this needed no serving +authorization. + +**What was not delivered, deliberately.** (a) is still open and still Shay's. +The difference now is that it can be decided against a measured cost: promoting +`realize_target` buys 340/347 over 85/347 and the ability to say "not", and +costs a move in every surface hash plus whatever the Shadow Coherence Gate +ruling in `core/cognition/surface_resolution.py` was protecting. + ### Phase 5 — Raise round-trip, then earn diversity *Depends on:* Phases 1–3 plus 2B, and on the Phase 1 baseline being non-zero-able. @@ -860,6 +916,58 @@ the ADR rather than at fluid prose. --- +### RESULT — measured after Phase 4. **Neither fork. The pre-commitment above was wrong.** + +`g_read_rate` did not stay at zero, and it did not rise materially either. It +went to **1 of 293** (`0.003413`) — the first non-zero in the arc — and the one +case tells you more than the rate does. + +The unblocked case is `gram_C14_p01`. What was blocking it: the writer emitted +`all molecules are defined as compound`, a predicate nominal that does not agree +with its subject. The reader **accepts** `...as compounds` and **refuses** +`...as compound`. So the blocker was a one-line *writer* defect — not §1.8. + +The other 292 decompose, and not one of them is a model mismatch either: + +| refusal reason | n | whose problem | +|---|---|---| +| `no_template_match` | 289 | reader has no SUBJ-VERB-OBJ template at all | +| `unknown_morphology` | 2 | prepositional objects (`reserved_word_in_np`) | +| `unsupported_negation` | 1 | reader has no negated-categorical template | +| read | 1 | — | + +Every one is the reader declining a **construction it has no template for**, not +a projection disagreeing about a graph it successfully parsed. And where a +construction *is* in both inventories the round trip closes exactly — +`s_surface_match_rate == s_renderable_rate` — while the reader independently +reads `all dogs are mammals`, `all wolves are canines`, and +`all molecules are defined as compounds`. + +**So the barrier is the overlap of the two construction inventories, which is +currently one construction wide.** That is Phase 5's item 1 — grow the +inventory until `read_rate` clears a ratcheted floor — and it is tractable work, +not an ADR-scale model decision. + +**Why the pre-commitment was wrong, kept here rather than quietly edited out:** +§1.6 read "uniform `no_template_match` across all 280" as evidence *for* the +type mismatch. It is not evidence for it. `no_template_match` is the reader +saying it has no template — a coverage fact — and §1.8's type mismatch was +inferred from it rather than measured. The corpus was 289/293 bare transitives, +a construction the reader has never claimed to read, so the measurement was +mostly reporting the corpus's composition. The one construction the corpus *did* +share with the reader (Phase 3's C14) was blocked by a writer bug, and until +that was fixed there was no case in the whole corpus capable of testing the +graph-model question at all. + +**Consequence for Phase 5.** Its scope is now decidable. The inventory question +is the arc's live frontier, and it should be sized by measuring the reader's +construction set against the writer's rather than by growing corpora blindly. +The graph-model ADR is **not** cancelled — §1.8's type mismatch is still real — +but nothing measured here obliges it, and it should not be opened until a case +exists that fails for that reason and no other. + +--- + ## 7. Verification protocol Per `AGENTS.md`, unchanged from prior arcs: diff --git a/evals/grammar_roundtrip/contract.md b/evals/grammar_roundtrip/contract.md index 26f5bfd9..4b4e6ff9 100644 --- a/evals/grammar_roundtrip/contract.md +++ b/evals/grammar_roundtrip/contract.md @@ -36,6 +36,14 @@ Round-trip is falsifiable with no judge, no embedding, and no gold aesthetic. | direction | pipeline | what a failure means | |---|---|---| | **G-round-trip** | graph → `realize_target` → surface → `comprehend` → graph | CORE cannot read its own writing | + +> **`realize_target` is eval-only** (Phase 4). `core/cognition/pipeline.py` calls +> `realize_semantic`, never `realize_target`, so `g_read_rate` is a measurement of +> the grammar *library* and not of served English. Run through the serving writer +> instead, the G-direction reads back **nothing** (0.0 vs 0.003413) — see +> `tests/test_phase4_realizer_resolution.py`. Any claim made from this metric must +> say which writer it is about. + | **S-round-trip** | surface → `comprehend` → graph → categorical renderer → surface | CORE cannot reproduce what it just understood | Both are reported because they fail for different reasons and have different diff --git a/evals/grammatical_coverage/runner.py b/evals/grammatical_coverage/runner.py index a5ce005c..10d3b141 100644 --- a/evals/grammatical_coverage/runner.py +++ b/evals/grammatical_coverage/runner.py @@ -5,6 +5,23 @@ English surfaces from PropositionGraph inputs. Each case specifies a construction family (e.g. negation, conjunction, embedded clause) and acceptance criteria (exact surfaces, constraint checks). +WHICH REALIZER (Phase 4) +------------------------ +``accuracy`` scores ``realize_target``, which **no serving path calls** — +``core/cognition/pipeline.py`` calls ``realize_semantic``. So ``accuracy`` is a +measurement of the grammar *library*, not of English CORE has ever spoken, and +citing it as "CORE's fluency" is a category error. It was cited that way for +the whole arc: 117/117 + 39/39 for a function that does not speak. + +``serving_accuracy`` scores ``realize_semantic`` on the identical cases and the +identical rubric. Across all seven corpora the two read **340/347** and +**85/347**. The gap is not a matter of polish — ``render_semantic`` has no +``negated``, ``quantifier``, ``tense`` or ``aspect`` parameter, so it serves a +negated proposition as its affirmative. See +``tests/test_phase4_realizer_resolution.py``, which pins that defect and the +control proving the gap is the dropped content rather than the corpora's +hardcoded ``IntentTag.UNKNOWN``. + Conforms to the framework interface: ``run_lane(cases, config=None) -> report``. """ from __future__ import annotations @@ -62,11 +79,16 @@ def _check_word_order(order: list[str], surface_words: list[str]) -> bool: return True -def _realize_from_graph(case: dict[str, Any]) -> str: +def _realize_from_graph(case: dict[str, Any], realize: Any = None) -> str: """Realize a surface from a proposition graph case. This calls the actual realizer infrastructure. The graph format in the eval cases maps to the realizer's PropositionGraph -> surface path. + + ``realize`` defaults to ``realize_target``, which is the realizer this lane + has always scored -- and which **nothing in the serving path calls** + (Phase 4). Passing ``realize_semantic`` scores the writer that actually + ships, against this identical contract. """ from generate.graph_planner import ( ArticulationStep, @@ -133,17 +155,17 @@ def _realize_from_graph(case: dict[str, Any]) -> str: )) target = ArticulationTarget(steps=tuple(steps), source_intent=IntentTag.UNKNOWN) - plan = realize_target(target, graph) + plan = (realize or realize_target)(target, graph) surface = plan.surface.rstrip(".") return surface -def _score_case(case: dict[str, Any]) -> CaseResult: +def _score_case(case: dict[str, Any], realize: Any = None) -> CaseResult: construction = case["construction"] construction_name = case["construction_name"] try: - surface = _realize_from_graph(case) + surface = _realize_from_graph(case, realize) except Exception as exc: return CaseResult( case_id=case["id"], @@ -235,11 +257,30 @@ def run_lane( for k, v in sorted(by_construction.items()) } + # ----------------------------------------------------------------- # + # Phase 4: score the SERVING writer on the identical contract. + # + # `accuracy` above is `realize_target`'s, and `core/cognition/pipeline.py` + # never calls `realize_target` -- it calls `realize_semantic`. For the whole + # arc this lane reported ~1.00 for a function that does not speak. Both + # numbers are reported now so the headline can never again be read as + # "CORE's fluency" when it is a measurement of a grammar library. + # + # This is the honest half of Phase 4 option (b). It changes no served byte; + # it only stops the lane from being silent about the gap. + # ----------------------------------------------------------------- # + from generate.realizer import realize_semantic + + serving_passed = sum(1 for case in cases if _score_case(case, realize_semantic).passed) + metrics = { "total": total, "passed": passed, "accuracy": round(passed / total, 4) if total else 0.0, "by_construction": construction_scores, + # What ships, on the same cases and the same rubric. + "serving_passed": serving_passed, + "serving_accuracy": round(serving_passed / total, 4) if total else 0.0, } return LaneReport(metrics=metrics, case_details=case_details) diff --git a/tests/test_phase4_realizer_resolution.py b/tests/test_phase4_realizer_resolution.py new file mode 100644 index 00000000..1dd5a460 --- /dev/null +++ b/tests/test_phase4_realizer_resolution.py @@ -0,0 +1,268 @@ +"""Phase 4 — which realizer serves, and what the other one's score means. + +The arc opened with a fact that had gone unremarked for its whole life: +``english_fluency_ood`` reported 117/117 + 39/39 for ``realize_target``, and +``core/cognition/pipeline.py`` **never calls** ``realize_target``. It calls +``realize_semantic``. 149 green cases were scoring a function that does not +speak, and the lane said nothing about the one that does. + +Phase 4 resolves that by option (b) of the plan: the lanes now report the +serving writer's score alongside, and the claim "realizer fluency is +mechanistic" is restated as a claim about **eval-only code**. Nothing here +changes a served byte — promoting ``realize_target`` to the serving path is +option (a), it moves surface hashes, and it is Shay's call, not this file's. + +What these tests are for +------------------------ +Every number below was measured, and each one is pinned so that it cannot +quietly stop being true. The load-bearing ones are the *controls*: without +them the headline gap could be dismissed as an artifact of the corpora, and +without the negation pin the most serious finding here would live only in a +commit message. +""" + +from __future__ import annotations + +import json +from pathlib import Path + +import pytest + +from evals.grammatical_coverage.runner import run_lane +from generate.graph_planner import ( + ArticulationStep, + ArticulationTarget, + GraphNode, + PropositionGraph, + RhetoricalMove, +) +from generate.intent import IntentTag +from generate.realizer import realize_semantic, realize_target + + +_EVALS = Path(__file__).resolve().parents[1] / "evals" + +#: Every corpus scored through ``realize_target`` by a lane. +_CORPORA = ( + "grammatical_coverage/public/v1", + "grammatical_coverage/public/v2", + "grammatical_coverage/dev", + "grammatical_coverage/holdouts/v1", + "english_fluency_ood/public/v1", + "english_fluency_ood/holdouts/v1", + "english_fluency_ood/dev", +) + +#: Content an ``ArticulationStep`` carries and ``render_semantic`` has no +#: parameter for. +_UNEXPRESSIBLE = ("quantifier", "negated", "tense", "aspect") + + +def _load(name: str) -> list[dict]: + path = _EVALS / name / "cases.jsonl" + return [json.loads(line) for line in path.read_text().splitlines() if line.strip()] + + +def _all_cases() -> list[dict]: + return [case for name in _CORPORA for case in _load(name)] + + +def _carries_unexpressible(case: dict) -> bool: + return any( + node.get(feature) not in (None, False, "") + for node in case.get("proposition_graph", {}).get("nodes", []) + for feature in _UNEXPRESSIBLE + ) + + +def _is_multi_node(case: dict) -> bool: + return len(case.get("proposition_graph", {}).get("nodes", [])) > 1 + + +def _one_step(**step_kwargs) -> tuple[ArticulationTarget, PropositionGraph]: + node = GraphNode( + node_id="n1", subject="knowledge", predicate="is_grounded_in", + obj="opinion", source_intent=IntentTag.DEFINITION, + ) + step = ArticulationStep( + node_id="n1", subject="knowledge", predicate="is_grounded_in", + move=RhetoricalMove.ASSERT, **step_kwargs, + ) + return ( + ArticulationTarget(steps=(step,), source_intent=IntentTag.DEFINITION), + PropositionGraph(nodes=(node,), edges=()), + ) + + +# --------------------------------------------------------------------------- # +# The headline: the lane now reports what ships +# --------------------------------------------------------------------------- # + + +def test_the_lane_reports_the_serving_writer_too() -> None: + """Before Phase 4 a lane could report 1.00 while the writer that actually + speaks scored 0.23 on the identical cases, and nothing surfaced it.""" + metrics = run_lane(_load("english_fluency_ood/public/v1")).metrics + assert metrics["passed"] == 117 + assert metrics["serving_passed"] == 27 + assert "serving_accuracy" in metrics + + +def test_the_measured_gap_across_every_scored_corpus() -> None: + """340/347 for the realizer the lanes score; 85/347 for the one that ships.""" + cases = _all_cases() + assert len(cases) == 347 + metrics = run_lane(cases).metrics + assert metrics["passed"] == 340 + assert metrics["serving_passed"] == 85 + + +# --------------------------------------------------------------------------- # +# CONTROL — the gap is the missing features, not the corpora's IntentTag +# --------------------------------------------------------------------------- # + + +def test_the_two_realizers_agree_exactly_where_nothing_is_dropped() -> None: + """THE CONTROL, and the most important test in this file. + + Every corpus here hardcodes ``IntentTag.UNKNOWN``, so "the serving realizer + scores badly" could be an artifact of never giving it a real intent. It is + not. On the 33 cases that carry no unexpressible feature and have a single + node, the two realizers score **identically** — which is only possible if + the gap on the other 314 is the dropped content and the clause joining. + + If this ever goes red, the gap measured above has stopped meaning what the + Phase 4 ruling says it means, and the ruling needs revisiting. + """ + control = [c for c in _all_cases() + if not _carries_unexpressible(c) and not _is_multi_node(c)] + assert len(control) == 33, "the control bucket changed size" + metrics = run_lane(control).metrics + assert metrics["passed"] == 33 + assert metrics["serving_passed"] == 33 + + +@pytest.mark.parametrize( + ("bucket", "expected_n", "expected_eval", "expected_serving"), + [ + ("features", 214, 207, 49), + ("multi_node", 100, 100, 3), + ], +) +def test_the_gap_decomposes_into_dropped_features_and_clause_joining( + bucket: str, expected_n: int, expected_eval: int, expected_serving: int +) -> None: + """Two separate causes, measured separately: content the serving writer has + no parameter for, and clauses it can only join with a full stop.""" + if bucket == "features": + cases = [c for c in _all_cases() + if _carries_unexpressible(c) and not _is_multi_node(c)] + else: + cases = [c for c in _all_cases() + if not _carries_unexpressible(c) and _is_multi_node(c)] + assert len(cases) == expected_n + metrics = run_lane(cases).metrics + assert metrics["passed"] == expected_eval + assert metrics["serving_passed"] == expected_serving + + +# --------------------------------------------------------------------------- # +# DEFECT PIN — the serving writer cannot express negation +# --------------------------------------------------------------------------- # + + +def test_the_serving_realizer_emits_the_same_surface_negated_or_not() -> None: + """DEFECT PIN, not a goal. Revise when fixed; never relax. + + ``render_semantic``'s signature is ``(intent, subject, predicate, obj, + secondary, language, root)``. There is no ``negated`` parameter, and + ``realize_semantic`` never reads ``step.negated``. So a negated + proposition is served as its **affirmative**: + + negated=False -> 'Knowledge is defined as opinion.' + negated=True -> 'Knowledge is defined as opinion.' + + This is not a fluency defect. It is the same family as ADR-0261 §5.1 + refuse-don't-drop: v1b served WRONG by dropping premises it could not + express. Fixing it changes served output, so it is authorization-gated; + pinning it here is what stops it from being forgotten. + """ + affirmative = realize_semantic(*_one_step(negated=False)).surface + negated = realize_semantic(*_one_step(negated=True)).surface + assert affirmative == negated, "the defect this pin records has changed shape" + + # The eval-only realizer distinguishes them, which is how we know the + # information reaches the realizer boundary intact. + assert realize_target(*_one_step(negated=True)).surface != ( + realize_target(*_one_step(negated=False)).surface + ) + assert "not" in realize_target(*_one_step(negated=True)).surface + + +@pytest.mark.parametrize("feature", _UNEXPRESSIBLE) +def test_render_semantic_has_no_parameter_for_the_content_it_drops(feature: str) -> None: + """Derived from the signature, so it cannot rot into a stale comment.""" + import inspect + + from generate.semantic_templates import render_semantic + + assert feature not in inspect.signature(render_semantic).parameters + + +def test_how_much_of_the_corpus_carries_content_the_serving_writer_drops() -> None: + """214 of 347. The scale of the gap, independent of any rubric.""" + cases = _all_cases() + carrying = [c for c in cases if _carries_unexpressible(c)] + assert len(carrying) == 214 + assert len(cases) == 347 + + +# --------------------------------------------------------------------------- # +# §6 evidence — the writer is not what round-trip is blocked on +# --------------------------------------------------------------------------- # + + +def test_the_serving_writer_round_trips_nothing_at_all() -> None: + """§6 of the plan forks on ``read_rate`` after unification, and this is the + measurement that says which fork. + + Two numbers, and they say different things: + + * **292 of 293** cases refuse with ``no_template_match`` under *either* + writer. For those the writer is irrelevant — the reader has no + SUBJ-VERB-OBJ template at all, so nothing the writer does can help. + * The **one** case that reads is writer-sensitive: it reads through + ``realize_target`` (0.003413) and refuses through ``realize_semantic`` + (0.0), because predicate-nominal object agreement lives in + ``render_step`` and the serving writer has no equivalent. + + So the serving writer round-trips **nothing**, and the eval writer + round-trips one case — and the gap between 1 and 293 is reader construction + coverage, not a graph-model mismatch. That is why §6's pre-committed + reading (§1.8 ⇒ next step is an ADR) is not what the evidence supports; see + ``test_grammar_roundtrip.py::test_the_remaining_blockers_are_reader_construction_coverage`` + for the full refusal census. + """ + import evals.grammar_roundtrip.runner as rt + + baseline = rt.run_lane().metrics["g_read_rate"] + original = rt.realize_target + rt.realize_target = realize_semantic + try: + swapped = rt.run_lane().metrics["g_read_rate"] + finally: + rt.realize_target = original + + assert baseline == 0.003413, "the eval writer reads back exactly one case" + assert swapped == 0.0, "the serving writer reads back none" + # Sentinel: prove the swap is actually reaching the lane, so that an + # equality between two identical runs can never be mistaken for a result. + def _boom(*_args, **_kwargs): # pragma: no cover - must raise + raise RuntimeError("sentinel") + + rt.realize_target = _boom + try: + with pytest.raises(RuntimeError, match="sentinel"): + rt.run_lane() + finally: + rt.realize_target = original