feat(recognition): Stage 3 fail-closed multi-root depth ambiguity

Stop silent roots[0] commitment when multiple HE/GRC roots are observed.
Return typed RootSenseAmbiguity, mark runnable assessments non-runnable
with AMBIGUOUS_ROOTS provenance, and refuse multi-root node canonicalization
without authored single-candidate resolution.

[Verification]: stage3 root ambiguity + oov pipeline tests passed (39)
This commit is contained in:
Shay 2026-07-20 13:47:02 -07:00
parent b1dc57ff22
commit aaa8503f0b
3 changed files with 258 additions and 31 deletions

View file

@ -7,11 +7,14 @@ Used by derive_recognizer, recognize (for matching), and assessment enrichment.
Connects to cognitive path: listen/comprehend (depth from packs) -> think (anti-unif canonical) -> articulate.
Preserves exact recall, no drift repair.
Stage 3 (Master Blueprint / ADR-0256 reserve): multi-root Hebrew/Greek
depth must fail closed never silently commit ``roots[0]``.
"""
from __future__ import annotations
from dataclasses import replace
from dataclasses import dataclass, replace
from typing import Any, Sequence, Tuple
from recognition.outcome import EvidenceSpan, FeatureBundle
@ -23,12 +26,99 @@ except ImportError:
GraphNode = Any # type: ignore
def canonicalize_token(token: str, node_id: str | None, depths: dict[str, dict] | None) -> str:
"""Wrap root_normalize keyed on node_id from depths dict.
@dataclass(frozen=True, slots=True)
class RootSenseAmbiguity:
"""Typed multi-root / multi-sense ambiguity (fail-closed).
depths: {node_id: {"language": , "root": , ...}, ...}
If node_id in depths and lang he/grc and root, return root, else token.
Caller must pass the token associated with that node_id.
Carries every observed candidate root with node provenance. Downstream
must not emit a single canonical constraint unless an authored
``disambiguation_rule_id`` resolves the set.
"""
candidates: tuple[str, ...]
node_ids: tuple[str, ...]
languages: tuple[str, ...]
disambiguation_rule_id: str | None = None
@property
def resolved(self) -> bool:
return self.disambiguation_rule_id is not None and len(self.candidates) == 1
def as_dict(self) -> dict[str, Any]:
return {
"kind": "root_sense_ambiguity",
"candidates": list(self.candidates),
"node_ids": list(self.node_ids),
"languages": list(self.languages),
"disambiguation_rule_id": self.disambiguation_rule_id,
"resolved": self.resolved,
}
def observed_roots(depth: dict[str, dict] | None) -> tuple[str, ...]:
"""Stable unique roots observed across depth entries (order of first appearance)."""
if not depth:
return ()
seen: list[str] = []
for nid, d in depth.items():
del nid # provenance via observe_root_ambiguity
roots_field = d.get("roots")
if isinstance(roots_field, (list, tuple)):
for r in roots_field:
if r and r not in seen:
seen.append(str(r))
r = d.get("root")
if r and r not in seen:
seen.append(str(r))
return tuple(seen)
def observe_root_ambiguity(
depth: dict[str, dict] | None,
*,
disambiguation_rule_id: str | None = None,
) -> RootSenseAmbiguity | None:
"""Return typed ambiguity when ≥2 distinct roots are observed without resolution.
A single authored ``disambiguation_rule_id`` alone does **not** collapse
candidates that would be the forbidden "one authored mapping" shortcut.
Resolution requires the rule id *and* a single remaining candidate
(caller must filter candidates before calling).
"""
if not depth:
return None
cands = observed_roots(depth)
if len(cands) <= 1:
return None
node_ids: list[str] = []
langs: list[str] = []
for nid, d in depth.items():
rset = set()
if d.get("root"):
rset.add(str(d["root"]))
if isinstance(d.get("roots"), (list, tuple)):
rset.update(str(x) for x in d["roots"] if x)
if rset:
node_ids.append(nid)
langs.append(str(d.get("language") or ""))
amb = RootSenseAmbiguity(
candidates=cands,
node_ids=tuple(node_ids),
languages=tuple(langs),
disambiguation_rule_id=disambiguation_rule_id,
)
if amb.resolved:
return None
return amb
def canonicalize_token(token: str, node_id: str | None, depths: dict[str, dict] | None) -> str:
"""Root-normalize keyed on node_id from depths dict.
depths: {node_id: {"language": , "root": , "roots": optional multi, ...}, ...}
If the node carries multiple roots (``roots`` len>1), fail closed: return
the surface token unchanged (no silent first-root commit).
If global depth is multi-root ambiguous, also refuse cross-node first-match.
"""
if not depths or not node_id:
return token
@ -36,9 +126,16 @@ def canonicalize_token(token: str, node_id: str | None, depths: dict[str, dict]
if not d:
return token
lang = d.get("language")
if lang not in ("he", "grc"):
return token
multi = d.get("roots")
if isinstance(multi, (list, tuple)) and len({str(x) for x in multi if x}) > 1:
return token # fail closed: multi-root on this node
root = d.get("root")
if lang in ("he", "grc") and root:
return root
if root:
return str(root)
if isinstance(multi, (list, tuple)) and len(multi) == 1 and multi[0]:
return str(multi[0])
return token
@ -49,8 +146,7 @@ def canonicalize_agent_slot(
"""Return copy of tokens with agent slot canonicalized using EvidenceSpan.start + node_id if available.
Single lookup by agent_node_id (node-keyed, requires explicit nid or no-op; no first-key proxy).
If bundle use its start, else start_idx.
Pure.
Pure. Multi-root nodes fail closed (surface token retained).
"""
if not depths:
return tuple(tokens)
@ -66,7 +162,7 @@ def canonicalize_agent_slot(
if nid is None or nid not in depths:
return tuple(new_tokens) # require explicit nid, no first-key proxy
d = depths[nid]
if d.get("language") in ("he", "grc") and d.get("root"):
if d.get("language") in ("he", "grc") and (d.get("root") or d.get("roots")):
orig = new_tokens[start]
new_tokens[start] = canonicalize_token(orig, nid, depths)
return tuple(new_tokens)
@ -75,34 +171,75 @@ def canonicalize_agent_slot(
def build_node_depths(nodes: Sequence[Any]) -> dict[str, dict]:
"""Lift the node_depths dict from list of GraphNode (or objects with .node_id, .language etc).
Pure extraction of the comprehension in pipeline.
Pure extraction of the comprehension in pipeline. Includes optional
multi-root ``roots`` when present on the node.
"""
return {
n.node_id: {
k: v for k, v in {
"language": getattr(n, "language", None),
"root": getattr(n, "root", None),
"morphology_id": getattr(n, "morphology_id", None),
}.items() if v is not None
}
for n in nodes
if getattr(n, "language", None) in ("he", "grc") or getattr(n, "root", None)
}
out: dict[str, dict] = {}
for n in nodes:
if getattr(n, "language", None) not in ("he", "grc") and not getattr(n, "root", None) and not getattr(n, "roots", None):
continue
payload: dict[str, Any] = {}
lang = getattr(n, "language", None)
root = getattr(n, "root", None)
roots = getattr(n, "roots", None)
mid = getattr(n, "morphology_id", None)
if lang is not None:
payload["language"] = lang
if root is not None:
payload["root"] = root
if roots is not None:
payload["roots"] = tuple(roots) if not isinstance(roots, tuple) else roots
if mid is not None:
payload["morphology_id"] = mid
if payload:
out[n.node_id] = payload
return out
def enrich_assessments_with_depth(assessments: Tuple[Any, ...], depth: dict | None) -> Tuple[Any, ...]:
"""Immutable enrichment of assessments with root note using dataclasses.replace.
Returns a new tuple. Enrichment is best-effort: if an assessment does not
support replace (e.g. non-dataclass or frozen in an incompatible way), the
original item is kept without the depth note. No __dict__ reconstruction
is performed (avoids producing invalid objects for slots/frozen types).
* **0 roots** assessments unchanged.
* **1 unique root** append `` [root:]`` to runnable explanations.
* **2 unique roots** **fail closed**: do not pick ``roots[0]``; mark
runnable assessments non-runnable with ``AMBIGUOUS_ROOTS`` provenance
listing every candidate (Master Blueprint Stage 3 HE policy).
Returns a new tuple. Enrichment is best-effort for non-dataclass items.
"""
if not depth:
return assessments
roots = [d.get("root") for d in depth.values() if d.get("root")]
amb = observe_root_ambiguity(depth)
roots = observed_roots(depth)
if not roots:
return assessments
if amb is not None:
note = (
f" [AMBIGUOUS_ROOTS candidates={list(amb.candidates)} "
f"nodes={list(amb.node_ids)} — fail-closed; no silent first-root commit]"
)
new_ass = []
for a in assessments:
if hasattr(a, "explanation"):
try:
kwargs: dict[str, Any] = {
"explanation": (getattr(a, "explanation", "") or "") + note,
}
if hasattr(a, "runnable"):
kwargs["runnable"] = False
if hasattr(a, "unresolved_hazards"):
hazards = tuple(getattr(a, "unresolved_hazards", ()) or ())
if "ambiguous_hebrew_roots" not in hazards:
kwargs["unresolved_hazards"] = hazards + ("ambiguous_hebrew_roots",)
new_a = replace(a, **kwargs)
except Exception:
new_a = a
new_ass.append(new_a)
else:
new_ass.append(a)
return tuple(new_ass)
# Single unambiguous root
note = f" [root:{roots[0]}]"
new_ass = []
for a in assessments:
@ -110,8 +247,6 @@ def enrich_assessments_with_depth(assessments: Tuple[Any, ...], depth: dict | No
try:
new_a = replace(a, explanation=(getattr(a, "explanation", "") or "") + note)
except Exception:
# Best-effort only; do not synthesize copies that could violate
# frozen/slots invariants after CGA substrate types.
new_a = a
new_ass.append(new_a)
else:

View file

@ -381,8 +381,13 @@ def test_depth_canonical_direct():
assert real_pg.get_node_depths()["n1"]["root"] == "א-מ-ן"
from generate.problem_frame_contracts import ContractAssessment
a = ContractAssessment(candidate_organ="t", runnable=True, explanation="base")
# Multi-root depth must fail closed (Stage 3) — not silent roots[0].
en = enrich_assessments_with_depth((a,), depths)
assert "[root:א-מ-ן]" in (en[0].explanation or "")
assert en[0].runnable is False
assert "AMBIGUOUS_ROOTS" in (en[0].explanation or "")
# Single-root enrichment still annotates.
en_one = enrich_assessments_with_depth((a,), {"n1": {"language": "he", "root": "א-מ-ן"}})
assert "[root:א-מ-ן]" in (en_one[0].explanation or "")
print("depth_canonical direct tests passed")

View file

@ -0,0 +1,87 @@
"""Stage 3 — fail-closed multi-root Hebrew/Greek depth policy."""
from __future__ import annotations
from generate.problem_frame_contracts import ContractAssessment
from recognition.depth_canonical import (
RootSenseAmbiguity,
build_node_depths,
canonicalize_token,
enrich_assessments_with_depth,
observe_root_ambiguity,
observed_roots,
)
def test_observed_roots_unique_stable_order():
depth = {
"n1": {"language": "he", "root": "א-מ-ן"},
"n2": {"language": "he", "root": "ד-ב-ר"},
"n3": {"language": "he", "root": "א-מ-ן"},
}
assert observed_roots(depth) == ("א-מ-ן", "ד-ב-ר")
def test_observe_root_ambiguity_when_multi():
depth = {
"n1": {"language": "he", "root": "א-מ-ן"},
"n2": {"language": "he", "root": "ד-ב-ר"},
}
amb = observe_root_ambiguity(depth)
assert isinstance(amb, RootSenseAmbiguity)
assert amb.candidates == ("א-מ-ן", "ד-ב-ר")
assert amb.resolved is False
def test_enrich_assessments_fails_closed_on_multi_root_no_first_match():
depth = {
"n1": {"language": "he", "root": "א-מ-ן"},
"n2": {"language": "he", "root": "ד-ב-ר"},
}
a = ContractAssessment(candidate_organ="t", runnable=True, explanation="base")
en = enrich_assessments_with_depth((a,), depth)
assert en[0].runnable is False
assert "AMBIGUOUS_ROOTS" in (en[0].explanation or "")
assert "א-מ-ן" in (en[0].explanation or "")
assert "ד-ב-ר" in (en[0].explanation or "")
# Must NOT silently commit only the first root as the sole note.
assert "[root:א-מ-ן]" not in (en[0].explanation or "") or "AMBIGUOUS" in (
en[0].explanation or ""
)
assert "ambiguous_hebrew_roots" in (en[0].unresolved_hazards or ())
def test_enrich_single_root_still_annotates():
depth = {"n1": {"language": "he", "root": "א-מ-ן"}}
a = ContractAssessment(candidate_organ="t", runnable=True, explanation="base")
en = enrich_assessments_with_depth((a,), depth)
assert en[0].runnable is True
assert "[root:א-מ-ן]" in (en[0].explanation or "")
def test_canonicalize_multi_root_node_fails_closed():
depths = {
"n1": {"language": "he", "roots": ("א-מ-ן", "ד-ב-ר")},
}
# Surface retained — no silent first-root commit.
assert canonicalize_token("דָּבָר", "n1", depths) == "דָּבָר"
def test_canonicalize_single_root_still_maps():
depths = {"n1": {"language": "he", "root": "א-מ-ן"}}
assert canonicalize_token("אמת", "n1", depths) == "א-מ-ן"
def test_build_node_depths_carries_roots_tuple():
class _N:
node_id = "n1"
language = "he"
root = None
roots = ("א-מ-ן", "ד-ב-ר")
morphology_id = None
d = build_node_depths([_N()])
assert d["n1"]["roots"] == ("א-מ-ן", "ד-ב-ר")
amb = observe_root_ambiguity(d)
assert amb is not None
assert len(amb.candidates) == 2