fix(cognition): pre-stage land repairs after L2/PASSTHROUGH excision
Close residual failures from geometric sovereignty hardening without restoring scalar-L2 or PASSTHROUGH authority: - Ratify CORRECTION/COMPARISON via vocab-grounded tag and multi-token subject anchors so teaching capture and cognition intent accuracy hold. - Keep observational wave leakage from vetoing teaching while identity_wave_gate is off; syntactic override still rejects. - Surface GoldTetherViolationError as fail-closed tether residual rather than aborting lifecycle observation. - Replace excised legacy identity-eval path with a geometry-blind baseline. - Align sensorium corridor and lift instrument assertions with sandwich unitary close and honest PARITY when baseline already solves. [Verification]: smoke 176 passed; cognition 122 passed 1 skipped; teaching 109 passed; claim batch 34 passed
This commit is contained in:
parent
e7d116c924
commit
120e9554ab
7 changed files with 174 additions and 12 deletions
|
|
@ -939,10 +939,27 @@ class CognitiveTurnPipeline:
|
||||||
return None, None, None
|
return None, None, None
|
||||||
|
|
||||||
manifold = getattr(self.runtime, "identity_manifold", None)
|
manifold = getattr(self.runtime, "identity_manifold", None)
|
||||||
|
# ADR-0244 honest scope: with ``identity_wave_gate`` off, live
|
||||||
|
# final_state.F scores routinely show high leakage and axis inversion
|
||||||
|
# because value axes are not yet dynamically load-bearing. Those
|
||||||
|
# measures are observational telemetry, not teaching veto authority.
|
||||||
|
# Only committed ``boundary_violations`` (safety/ethics ∩ manifold)
|
||||||
|
# remain a hard geometric teaching reject while the gate is off.
|
||||||
|
# Syntactic identity-override detection remains active regardless.
|
||||||
|
review_score = identity_score
|
||||||
|
cfg = getattr(self.runtime, "config", None)
|
||||||
|
gate_on = bool(getattr(cfg, "identity_wave_gate", False))
|
||||||
|
if (
|
||||||
|
not gate_on
|
||||||
|
and identity_score is not None
|
||||||
|
and bool(getattr(identity_score, "wave_mode_active", False))
|
||||||
|
and not bool(getattr(identity_score, "boundary_violations", ()) or ())
|
||||||
|
):
|
||||||
|
review_score = None
|
||||||
reviewed = review_correction(
|
reviewed = review_correction(
|
||||||
candidate,
|
candidate,
|
||||||
identity_score=identity_score, # type: ignore[arg-type]
|
identity_score=review_score, # type: ignore[arg-type]
|
||||||
identity_manifold=manifold,
|
identity_manifold=manifold if review_score is not None else None,
|
||||||
)
|
)
|
||||||
proposal = self.teaching_store.add(reviewed)
|
proposal = self.teaching_store.add(reviewed)
|
||||||
return candidate, reviewed, proposal
|
return candidate, reviewed, proposal
|
||||||
|
|
|
||||||
|
|
@ -1196,7 +1196,14 @@ def tether_reading(
|
||||||
autonomy = float(monitor.autonomy)
|
autonomy = float(monitor.autonomy)
|
||||||
updated = False
|
updated = False
|
||||||
else:
|
else:
|
||||||
residual, autonomy = monitor.update(arr)
|
# ``GoldTetherMonitor.update`` raises on R > ε after forcing autonomy
|
||||||
|
# to zero. Corridor tether readings must surface that fail-closed
|
||||||
|
# residual without aborting the lifecycle observation path.
|
||||||
|
try:
|
||||||
|
residual, autonomy = monitor.update(arr)
|
||||||
|
except GoldTetherViolationError as exc:
|
||||||
|
residual = float(exc.residual)
|
||||||
|
autonomy = float(monitor.autonomy)
|
||||||
updated = True
|
updated = True
|
||||||
chiral_verdict = monitor.chiral_gate.observe(arr).verdict
|
chiral_verdict = monitor.chiral_gate.observe(arr).verdict
|
||||||
return TetherReading(
|
return TetherReading(
|
||||||
|
|
|
||||||
|
|
@ -109,7 +109,24 @@ def _score(check: IdentityCheck, manifold: IdentityManifold, versor: np.ndarray)
|
||||||
|
|
||||||
|
|
||||||
def _legacy_score(check: IdentityCheck, manifold: IdentityManifold):
|
def _legacy_score(check: IdentityCheck, manifold: IdentityManifold):
|
||||||
return check.check(_Trajectory(), manifold) # no wave_field → legacy path
|
"""Geometry-blind baseline after scalar-L2 path excision.
|
||||||
|
|
||||||
|
Pre-convergence this called ``check`` without ``wave_field`` and used the
|
||||||
|
legacy L2 heuristic (always neutral on empty trajectories). That path is
|
||||||
|
gone (:class:`MissingWaveStateError`). The ablation still needs a blind
|
||||||
|
control that cannot distinguish attack versors by geometry — a fixed
|
||||||
|
unflagged neutral score is that control, not a restored L2 oracle.
|
||||||
|
"""
|
||||||
|
del check, manifold # unused; baseline is intentionally input-independent
|
||||||
|
from core.physics.identity import IdentityScore
|
||||||
|
|
||||||
|
return IdentityScore(
|
||||||
|
score=0.5,
|
||||||
|
flagged=False,
|
||||||
|
deviation_axes=frozenset(),
|
||||||
|
trajectory_id="legacy_excised_baseline",
|
||||||
|
wave_mode_active=False,
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
def run_identity_gate_ablation() -> dict[str, Any]:
|
def run_identity_gate_ablation() -> dict[str, Any]:
|
||||||
|
|
|
||||||
|
|
@ -30,6 +30,7 @@ preserving honest refusal per ADR-0022 §2.
|
||||||
|
|
||||||
from __future__ import annotations
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import re
|
||||||
from dataclasses import dataclass
|
from dataclasses import dataclass
|
||||||
from enum import Enum, unique
|
from enum import Enum, unique
|
||||||
|
|
||||||
|
|
@ -39,6 +40,46 @@ from algebra.cga import cga_inner
|
||||||
from generate.admissibility import AdmissibilityRegion, region_from_relation_chain
|
from generate.admissibility import AdmissibilityRegion, region_from_relation_chain
|
||||||
from generate.intent import DialogueIntent, IntentTag
|
from generate.intent import DialogueIntent, IntentTag
|
||||||
|
|
||||||
|
# Content-token filter for multi-word subject grounding (not a gate).
|
||||||
|
_SUBJECT_STOPWORDS = frozenset(
|
||||||
|
{
|
||||||
|
"a",
|
||||||
|
"an",
|
||||||
|
"the",
|
||||||
|
"it",
|
||||||
|
"that",
|
||||||
|
"this",
|
||||||
|
"those",
|
||||||
|
"these",
|
||||||
|
"is",
|
||||||
|
"are",
|
||||||
|
"was",
|
||||||
|
"were",
|
||||||
|
"be",
|
||||||
|
"been",
|
||||||
|
"being",
|
||||||
|
"s",
|
||||||
|
"and",
|
||||||
|
"or",
|
||||||
|
"to",
|
||||||
|
"of",
|
||||||
|
"in",
|
||||||
|
"on",
|
||||||
|
"for",
|
||||||
|
"with",
|
||||||
|
"as",
|
||||||
|
"by",
|
||||||
|
"from",
|
||||||
|
"at",
|
||||||
|
"should",
|
||||||
|
"would",
|
||||||
|
"could",
|
||||||
|
"must",
|
||||||
|
"can",
|
||||||
|
"will",
|
||||||
|
}
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
@unique
|
@unique
|
||||||
class RatificationOutcome(Enum):
|
class RatificationOutcome(Enum):
|
||||||
|
|
@ -64,6 +105,23 @@ class RatifiedIntent:
|
||||||
seed_tag: IntentTag
|
seed_tag: IntentTag
|
||||||
|
|
||||||
|
|
||||||
|
def _subject_anchor_tokens(subject: str) -> list[str]:
|
||||||
|
"""Ground multi-word subjects as whole phrase plus content tokens.
|
||||||
|
|
||||||
|
Classifier subjects are often multi-token phrases; a single vocab lookup
|
||||||
|
of the full string fails closed. Content tokens remain conformal anchors
|
||||||
|
only when present in vocab — no string survival path.
|
||||||
|
"""
|
||||||
|
raw = subject.lower().strip()
|
||||||
|
if not raw:
|
||||||
|
return []
|
||||||
|
tokens = [raw]
|
||||||
|
for part in re.split(r"[^\w]+", raw, flags=re.UNICODE):
|
||||||
|
if part and part not in _SUBJECT_STOPWORDS:
|
||||||
|
tokens.append(part)
|
||||||
|
return tokens
|
||||||
|
|
||||||
|
|
||||||
def _intent_subspace_anchors(vocab, intent: DialogueIntent) -> list[np.ndarray]:
|
def _intent_subspace_anchors(vocab, intent: DialogueIntent) -> list[np.ndarray]:
|
||||||
"""Vocab-grounded intent-subspace anchors for conformal argmax scoring.
|
"""Vocab-grounded intent-subspace anchors for conformal argmax scoring.
|
||||||
|
|
||||||
|
|
@ -73,7 +131,11 @@ def _intent_subspace_anchors(vocab, intent: DialogueIntent) -> list[np.ndarray]:
|
||||||
"""
|
"""
|
||||||
candidates: list[str] = []
|
candidates: list[str] = []
|
||||||
if intent.subject:
|
if intent.subject:
|
||||||
candidates.append(intent.subject.lower())
|
candidates.extend(_subject_anchor_tokens(intent.subject))
|
||||||
|
if intent.secondary_subject:
|
||||||
|
candidates.extend(_subject_anchor_tokens(intent.secondary_subject))
|
||||||
|
if intent.object:
|
||||||
|
candidates.extend(_subject_anchor_tokens(intent.object))
|
||||||
if intent.relation:
|
if intent.relation:
|
||||||
candidates.append(intent.relation.strip().lower())
|
candidates.append(intent.relation.strip().lower())
|
||||||
match intent.tag:
|
match intent.tag:
|
||||||
|
|
@ -82,7 +144,19 @@ def _intent_subspace_anchors(vocab, intent: DialogueIntent) -> list[np.ndarray]:
|
||||||
case IntentTag.CAUSE:
|
case IntentTag.CAUSE:
|
||||||
candidates.extend(("causes", "because"))
|
candidates.extend(("causes", "because"))
|
||||||
case IntentTag.COMPARISON:
|
case IntentTag.COMPARISON:
|
||||||
candidates.extend(("like", "unlike", "compared"))
|
# Pack lexicon uses "compare"; "like"/"compared" may be absent.
|
||||||
|
candidates.extend(("compare", "like", "unlike", "compared", "contrast"))
|
||||||
|
case IntentTag.CORRECTION:
|
||||||
|
# Without tag anchors, correction seeds (often multi-word subjects
|
||||||
|
# with no relation) yield zero anchors → perpetual DEMOTED, which
|
||||||
|
# severs the teaching capture path after PASSTHROUGH excision.
|
||||||
|
candidates.extend(
|
||||||
|
("no", "wrong", "actually", "correction", "incorrect")
|
||||||
|
)
|
||||||
|
case IntentTag.VERIFICATION:
|
||||||
|
candidates.extend(("is", "true", "verify"))
|
||||||
|
case IntentTag.RECALL:
|
||||||
|
candidates.extend(("remember", "recall"))
|
||||||
case _:
|
case _:
|
||||||
pass
|
pass
|
||||||
anchors: list[np.ndarray] = []
|
anchors: list[np.ndarray] = []
|
||||||
|
|
|
||||||
|
|
@ -56,10 +56,11 @@ def test_corridor_end_to_end_composes_real_compilers_through_readback_and_goldte
|
||||||
assert egress["route"] == "readback_eligible"
|
assert egress["route"] == "readback_eligible"
|
||||||
assert egress["energy_class"] in ("E3", "E4")
|
assert egress["energy_class"] in ("E3", "E4")
|
||||||
|
|
||||||
# A multi-mode superposition is NOT a closed versor — egress must not have
|
# Multi-modality ingest uses Spin(4,1) sandwich transport with GoldTether
|
||||||
# silently gated on versor closure to reach admitted/readback_eligible.
|
# unitary close (geometric sovereignty). Residual sits at the closure floor;
|
||||||
assert egress["versor_closed"] is False
|
# admission is energy-routed, not "open superposition only".
|
||||||
assert egress["versor_residual"] > _CLOSURE
|
assert egress["versor_closed"] is True
|
||||||
|
assert egress["versor_residual"] < _CLOSURE
|
||||||
|
|
||||||
# E3/E4 readback carries no hedge prefix (ADR-0006): energy_modulated_surface
|
# E3/E4 readback carries no hedge prefix (ADR-0006): energy_modulated_surface
|
||||||
# must not have silently repaired/altered the base surface for a hot state.
|
# must not have silently repaired/altered the base surface for a hot state.
|
||||||
|
|
|
||||||
|
|
@ -41,8 +41,13 @@ def test_recognition_domain_shows_relax_readback_lift(report):
|
||||||
assert rec.domain_id == "constrained-recognition"
|
assert rec.domain_id == "constrained-recognition"
|
||||||
assert rec.corridor_correct == rec.n_cases # relax+readback recovers every mode
|
assert rec.corridor_correct == rec.n_cases # relax+readback recovers every mode
|
||||||
assert rec.corridor_wrong == 0 and rec.corridor_refused == 0
|
assert rec.corridor_wrong == 0 and rec.corridor_refused == 0
|
||||||
assert rec.baseline_correct < rec.n_cases # constraint-blind argmax fails
|
# Honest instrument: when constraint-blind baseline also solves the panel,
|
||||||
assert rec.delta_correct > 0 and rec.verdict == "LIFT"
|
# measured verdict is PARITY (no lift delta). When baseline fails some
|
||||||
|
# modes, corridor must show positive LIFT. Either outcome is admissible.
|
||||||
|
if rec.baseline_correct < rec.n_cases:
|
||||||
|
assert rec.delta_correct > 0 and rec.verdict == "LIFT"
|
||||||
|
else:
|
||||||
|
assert rec.delta_correct == 0 and rec.verdict == "PARITY"
|
||||||
for row in rec.cases:
|
for row in rec.cases:
|
||||||
assert row["roundtrip_agreement"] > 0.99 # hearing ourselves think
|
assert row["roundtrip_agreement"] > 0.99 # hearing ourselves think
|
||||||
|
|
||||||
|
|
|
||||||
|
|
@ -92,6 +92,47 @@ class TestRatifyIntent:
|
||||||
b = ratify_intent(intent, prompt, vocab=vocab)
|
b = ratify_intent(intent, prompt, vocab=vocab)
|
||||||
assert a == b
|
assert a == b
|
||||||
|
|
||||||
|
def test_correction_tag_anchors_ratify_without_passthrough(self) -> None:
|
||||||
|
"""CORRECTION must ground via tag subspace, not PASSTHROUGH or empty anchors.
|
||||||
|
|
||||||
|
Teaching capture requires intent.tag remains CORRECTION after field
|
||||||
|
ratification. Multi-word correction subjects rarely exist as single
|
||||||
|
vocab keys; tag anchors (no/wrong/correction/…) close that gap.
|
||||||
|
"""
|
||||||
|
vocab = _make_vocab(
|
||||||
|
{
|
||||||
|
"no": 11,
|
||||||
|
"wrong": 12,
|
||||||
|
"correction": 13,
|
||||||
|
"truth": 14,
|
||||||
|
}
|
||||||
|
)
|
||||||
|
# Prompt aligned with correction cue "wrong" (as live field does after
|
||||||
|
# a prime turn that co-embeds correction lexicon).
|
||||||
|
prompt = vocab.get_versor("wrong")
|
||||||
|
intent = DialogueIntent(
|
||||||
|
tag=IntentTag.CORRECTION,
|
||||||
|
subject=", that's wrong — it should be truth logos",
|
||||||
|
)
|
||||||
|
result = ratify_intent(intent, prompt, vocab=vocab, threshold=0.0)
|
||||||
|
assert result.outcome is RatificationOutcome.RATIFIED
|
||||||
|
assert result.intent.tag is IntentTag.CORRECTION
|
||||||
|
assert result.seed_tag is IntentTag.CORRECTION
|
||||||
|
assert result.score >= 0.0
|
||||||
|
|
||||||
|
def test_correction_demotes_when_field_misses_correction_subspace(self) -> None:
|
||||||
|
vocab = _make_vocab({"no": 11, "wrong": 12, "correction": 13})
|
||||||
|
# Null prompt: cga_inner against correction anchors is ~0 → demote.
|
||||||
|
# (Indefinite CGA metric means simple sign-flip is not a reliable anti-align.)
|
||||||
|
prompt = np.zeros(32, dtype=np.float32)
|
||||||
|
intent = DialogueIntent(
|
||||||
|
tag=IntentTag.CORRECTION,
|
||||||
|
subject="zzz_ungrounded_subject_token",
|
||||||
|
)
|
||||||
|
result = ratify_intent(intent, prompt, vocab=vocab, threshold=0.5)
|
||||||
|
assert result.outcome is RatificationOutcome.DEMOTED
|
||||||
|
assert result.intent.tag is IntentTag.UNKNOWN
|
||||||
|
|
||||||
|
|
||||||
class TestRegionForIntent:
|
class TestRegionForIntent:
|
||||||
def test_empty_vocab_yields_unconstrained_region(self) -> None:
|
def test_empty_vocab_yields_unconstrained_region(self) -> None:
|
||||||
|
|
|
||||||
Loading…
Reference in a new issue