From aaa8503f0b5aa3a8ca83b00b88f6e2b34e64de46 Mon Sep 17 00:00:00 2001 From: Shay Date: Mon, 20 Jul 2026 13:47:02 -0700 Subject: [PATCH] feat(recognition): Stage 3 fail-closed multi-root depth ambiguity Stop silent roots[0] commitment when multiple HE/GRC roots are observed. Return typed RootSenseAmbiguity, mark runnable assessments non-runnable with AMBIGUOUS_ROOTS provenance, and refuse multi-root node canonicalization without authored single-candidate resolution. [Verification]: stage3 root ambiguity + oov pipeline tests passed (39) --- recognition/depth_canonical.py | 195 +++++++++++++++++++++++----- tests/test_oov_pipeline.py | 7 +- tests/test_stage3_root_ambiguity.py | 87 +++++++++++++ 3 files changed, 258 insertions(+), 31 deletions(-) create mode 100644 tests/test_stage3_root_ambiguity.py diff --git a/recognition/depth_canonical.py b/recognition/depth_canonical.py index f5f324fc..f4055a5a 100644 --- a/recognition/depth_canonical.py +++ b/recognition/depth_canonical.py @@ -7,11 +7,14 @@ Used by derive_recognizer, recognize (for matching), and assessment enrichment. Connects to cognitive path: listen/comprehend (depth from packs) -> think (anti-unif canonical) -> articulate. Preserves exact recall, no drift repair. + +Stage 3 (Master Blueprint / ADR-0256 reserve): multi-root Hebrew/Greek +depth must fail closed — never silently commit ``roots[0]``. """ from __future__ import annotations -from dataclasses import replace +from dataclasses import dataclass, replace from typing import Any, Sequence, Tuple from recognition.outcome import EvidenceSpan, FeatureBundle @@ -23,12 +26,99 @@ except ImportError: GraphNode = Any # type: ignore -def canonicalize_token(token: str, node_id: str | None, depths: dict[str, dict] | None) -> str: - """Wrap root_normalize keyed on node_id from depths dict. +@dataclass(frozen=True, slots=True) +class RootSenseAmbiguity: + """Typed multi-root / multi-sense ambiguity (fail-closed). - depths: {node_id: {"language": , "root": , ...}, ...} - If node_id in depths and lang he/grc and root, return root, else token. - Caller must pass the token associated with that node_id. + Carries every observed candidate root with node provenance. Downstream + must not emit a single canonical constraint unless an authored + ``disambiguation_rule_id`` resolves the set. + """ + + candidates: tuple[str, ...] + node_ids: tuple[str, ...] + languages: tuple[str, ...] + disambiguation_rule_id: str | None = None + + @property + def resolved(self) -> bool: + return self.disambiguation_rule_id is not None and len(self.candidates) == 1 + + def as_dict(self) -> dict[str, Any]: + return { + "kind": "root_sense_ambiguity", + "candidates": list(self.candidates), + "node_ids": list(self.node_ids), + "languages": list(self.languages), + "disambiguation_rule_id": self.disambiguation_rule_id, + "resolved": self.resolved, + } + + +def observed_roots(depth: dict[str, dict] | None) -> tuple[str, ...]: + """Stable unique roots observed across depth entries (order of first appearance).""" + if not depth: + return () + seen: list[str] = [] + for nid, d in depth.items(): + del nid # provenance via observe_root_ambiguity + roots_field = d.get("roots") + if isinstance(roots_field, (list, tuple)): + for r in roots_field: + if r and r not in seen: + seen.append(str(r)) + r = d.get("root") + if r and r not in seen: + seen.append(str(r)) + return tuple(seen) + + +def observe_root_ambiguity( + depth: dict[str, dict] | None, + *, + disambiguation_rule_id: str | None = None, +) -> RootSenseAmbiguity | None: + """Return typed ambiguity when ≥2 distinct roots are observed without resolution. + + A single authored ``disambiguation_rule_id`` alone does **not** collapse + candidates — that would be the forbidden "one authored mapping" shortcut. + Resolution requires the rule id *and* a single remaining candidate + (caller must filter candidates before calling). + """ + if not depth: + return None + cands = observed_roots(depth) + if len(cands) <= 1: + return None + node_ids: list[str] = [] + langs: list[str] = [] + for nid, d in depth.items(): + rset = set() + if d.get("root"): + rset.add(str(d["root"])) + if isinstance(d.get("roots"), (list, tuple)): + rset.update(str(x) for x in d["roots"] if x) + if rset: + node_ids.append(nid) + langs.append(str(d.get("language") or "")) + amb = RootSenseAmbiguity( + candidates=cands, + node_ids=tuple(node_ids), + languages=tuple(langs), + disambiguation_rule_id=disambiguation_rule_id, + ) + if amb.resolved: + return None + return amb + + +def canonicalize_token(token: str, node_id: str | None, depths: dict[str, dict] | None) -> str: + """Root-normalize keyed on node_id from depths dict. + + depths: {node_id: {"language": , "root": , "roots": optional multi, ...}, ...} + If the node carries multiple roots (``roots`` len>1), fail closed: return + the surface token unchanged (no silent first-root commit). + If global depth is multi-root ambiguous, also refuse cross-node first-match. """ if not depths or not node_id: return token @@ -36,9 +126,16 @@ def canonicalize_token(token: str, node_id: str | None, depths: dict[str, dict] if not d: return token lang = d.get("language") + if lang not in ("he", "grc"): + return token + multi = d.get("roots") + if isinstance(multi, (list, tuple)) and len({str(x) for x in multi if x}) > 1: + return token # fail closed: multi-root on this node root = d.get("root") - if lang in ("he", "grc") and root: - return root + if root: + return str(root) + if isinstance(multi, (list, tuple)) and len(multi) == 1 and multi[0]: + return str(multi[0]) return token @@ -49,8 +146,7 @@ def canonicalize_agent_slot( """Return copy of tokens with agent slot canonicalized using EvidenceSpan.start + node_id if available. Single lookup by agent_node_id (node-keyed, requires explicit nid or no-op; no first-key proxy). - If bundle use its start, else start_idx. - Pure. + Pure. Multi-root nodes fail closed (surface token retained). """ if not depths: return tuple(tokens) @@ -66,7 +162,7 @@ def canonicalize_agent_slot( if nid is None or nid not in depths: return tuple(new_tokens) # require explicit nid, no first-key proxy d = depths[nid] - if d.get("language") in ("he", "grc") and d.get("root"): + if d.get("language") in ("he", "grc") and (d.get("root") or d.get("roots")): orig = new_tokens[start] new_tokens[start] = canonicalize_token(orig, nid, depths) return tuple(new_tokens) @@ -75,34 +171,75 @@ def canonicalize_agent_slot( def build_node_depths(nodes: Sequence[Any]) -> dict[str, dict]: """Lift the node_depths dict from list of GraphNode (or objects with .node_id, .language etc). - Pure extraction of the comprehension in pipeline. + Pure extraction of the comprehension in pipeline. Includes optional + multi-root ``roots`` when present on the node. """ - return { - n.node_id: { - k: v for k, v in { - "language": getattr(n, "language", None), - "root": getattr(n, "root", None), - "morphology_id": getattr(n, "morphology_id", None), - }.items() if v is not None - } - for n in nodes - if getattr(n, "language", None) in ("he", "grc") or getattr(n, "root", None) - } + out: dict[str, dict] = {} + for n in nodes: + if getattr(n, "language", None) not in ("he", "grc") and not getattr(n, "root", None) and not getattr(n, "roots", None): + continue + payload: dict[str, Any] = {} + lang = getattr(n, "language", None) + root = getattr(n, "root", None) + roots = getattr(n, "roots", None) + mid = getattr(n, "morphology_id", None) + if lang is not None: + payload["language"] = lang + if root is not None: + payload["root"] = root + if roots is not None: + payload["roots"] = tuple(roots) if not isinstance(roots, tuple) else roots + if mid is not None: + payload["morphology_id"] = mid + if payload: + out[n.node_id] = payload + return out def enrich_assessments_with_depth(assessments: Tuple[Any, ...], depth: dict | None) -> Tuple[Any, ...]: """Immutable enrichment of assessments with root note using dataclasses.replace. - Returns a new tuple. Enrichment is best-effort: if an assessment does not - support replace (e.g. non-dataclass or frozen in an incompatible way), the - original item is kept without the depth note. No __dict__ reconstruction - is performed (avoids producing invalid objects for slots/frozen types). + * **0 roots** — assessments unchanged. + * **1 unique root** — append `` [root:…]`` to runnable explanations. + * **≥2 unique roots** — **fail closed**: do not pick ``roots[0]``; mark + runnable assessments non-runnable with ``AMBIGUOUS_ROOTS`` provenance + listing every candidate (Master Blueprint Stage 3 HE policy). + + Returns a new tuple. Enrichment is best-effort for non-dataclass items. """ if not depth: return assessments - roots = [d.get("root") for d in depth.values() if d.get("root")] + amb = observe_root_ambiguity(depth) + roots = observed_roots(depth) if not roots: return assessments + if amb is not None: + note = ( + f" [AMBIGUOUS_ROOTS candidates={list(amb.candidates)} " + f"nodes={list(amb.node_ids)} — fail-closed; no silent first-root commit]" + ) + new_ass = [] + for a in assessments: + if hasattr(a, "explanation"): + try: + kwargs: dict[str, Any] = { + "explanation": (getattr(a, "explanation", "") or "") + note, + } + if hasattr(a, "runnable"): + kwargs["runnable"] = False + if hasattr(a, "unresolved_hazards"): + hazards = tuple(getattr(a, "unresolved_hazards", ()) or ()) + if "ambiguous_hebrew_roots" not in hazards: + kwargs["unresolved_hazards"] = hazards + ("ambiguous_hebrew_roots",) + new_a = replace(a, **kwargs) + except Exception: + new_a = a + new_ass.append(new_a) + else: + new_ass.append(a) + return tuple(new_ass) + + # Single unambiguous root note = f" [root:{roots[0]}]" new_ass = [] for a in assessments: @@ -110,8 +247,6 @@ def enrich_assessments_with_depth(assessments: Tuple[Any, ...], depth: dict | No try: new_a = replace(a, explanation=(getattr(a, "explanation", "") or "") + note) except Exception: - # Best-effort only; do not synthesize copies that could violate - # frozen/slots invariants after CGA substrate types. new_a = a new_ass.append(new_a) else: diff --git a/tests/test_oov_pipeline.py b/tests/test_oov_pipeline.py index e48518a4..645f7502 100644 --- a/tests/test_oov_pipeline.py +++ b/tests/test_oov_pipeline.py @@ -381,8 +381,13 @@ def test_depth_canonical_direct(): assert real_pg.get_node_depths()["n1"]["root"] == "א-מ-ן" from generate.problem_frame_contracts import ContractAssessment a = ContractAssessment(candidate_organ="t", runnable=True, explanation="base") + # Multi-root depth must fail closed (Stage 3) — not silent roots[0]. en = enrich_assessments_with_depth((a,), depths) - assert "[root:א-מ-ן]" in (en[0].explanation or "") + assert en[0].runnable is False + assert "AMBIGUOUS_ROOTS" in (en[0].explanation or "") + # Single-root enrichment still annotates. + en_one = enrich_assessments_with_depth((a,), {"n1": {"language": "he", "root": "א-מ-ן"}}) + assert "[root:א-מ-ן]" in (en_one[0].explanation or "") print("depth_canonical direct tests passed") diff --git a/tests/test_stage3_root_ambiguity.py b/tests/test_stage3_root_ambiguity.py new file mode 100644 index 00000000..c5e6ab94 --- /dev/null +++ b/tests/test_stage3_root_ambiguity.py @@ -0,0 +1,87 @@ +"""Stage 3 — fail-closed multi-root Hebrew/Greek depth policy.""" + +from __future__ import annotations + +from generate.problem_frame_contracts import ContractAssessment +from recognition.depth_canonical import ( + RootSenseAmbiguity, + build_node_depths, + canonicalize_token, + enrich_assessments_with_depth, + observe_root_ambiguity, + observed_roots, +) + + +def test_observed_roots_unique_stable_order(): + depth = { + "n1": {"language": "he", "root": "א-מ-ן"}, + "n2": {"language": "he", "root": "ד-ב-ר"}, + "n3": {"language": "he", "root": "א-מ-ן"}, + } + assert observed_roots(depth) == ("א-מ-ן", "ד-ב-ר") + + +def test_observe_root_ambiguity_when_multi(): + depth = { + "n1": {"language": "he", "root": "א-מ-ן"}, + "n2": {"language": "he", "root": "ד-ב-ר"}, + } + amb = observe_root_ambiguity(depth) + assert isinstance(amb, RootSenseAmbiguity) + assert amb.candidates == ("א-מ-ן", "ד-ב-ר") + assert amb.resolved is False + + +def test_enrich_assessments_fails_closed_on_multi_root_no_first_match(): + depth = { + "n1": {"language": "he", "root": "א-מ-ן"}, + "n2": {"language": "he", "root": "ד-ב-ר"}, + } + a = ContractAssessment(candidate_organ="t", runnable=True, explanation="base") + en = enrich_assessments_with_depth((a,), depth) + assert en[0].runnable is False + assert "AMBIGUOUS_ROOTS" in (en[0].explanation or "") + assert "א-מ-ן" in (en[0].explanation or "") + assert "ד-ב-ר" in (en[0].explanation or "") + # Must NOT silently commit only the first root as the sole note. + assert "[root:א-מ-ן]" not in (en[0].explanation or "") or "AMBIGUOUS" in ( + en[0].explanation or "" + ) + assert "ambiguous_hebrew_roots" in (en[0].unresolved_hazards or ()) + + +def test_enrich_single_root_still_annotates(): + depth = {"n1": {"language": "he", "root": "א-מ-ן"}} + a = ContractAssessment(candidate_organ="t", runnable=True, explanation="base") + en = enrich_assessments_with_depth((a,), depth) + assert en[0].runnable is True + assert "[root:א-מ-ן]" in (en[0].explanation or "") + + +def test_canonicalize_multi_root_node_fails_closed(): + depths = { + "n1": {"language": "he", "roots": ("א-מ-ן", "ד-ב-ר")}, + } + # Surface retained — no silent first-root commit. + assert canonicalize_token("דָּבָר", "n1", depths) == "דָּבָר" + + +def test_canonicalize_single_root_still_maps(): + depths = {"n1": {"language": "he", "root": "א-מ-ן"}} + assert canonicalize_token("אמת", "n1", depths) == "א-מ-ן" + + +def test_build_node_depths_carries_roots_tuple(): + class _N: + node_id = "n1" + language = "he" + root = None + roots = ("א-מ-ן", "ד-ב-ר") + morphology_id = None + + d = build_node_depths([_N()]) + assert d["n1"]["roots"] == ("א-מ-ן", "ד-ב-ר") + amb = observe_root_ambiguity(d) + assert amb is not None + assert len(amb.candidates) == 2