From 9f85832baaf8d0cdad6856a9bd2def8a9c54239e Mon Sep 17 00:00:00 2001 From: Shay Date: Mon, 20 Jul 2026 14:58:21 -0700 Subject: [PATCH] =?UTF-8?q?fix:=20full-suite=20gates=20=E2=80=94=20suite?= =?UTF-8?q?=20SoT=20+=20multi-root=20depth=20pin?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit - Re-export core.cli._TEST_SUITES from cli_test.TEST_SUITES so argparse and tests cannot drift from the runtime packs/smoke/algebra pins (Stage dual-pack draft boundary was only on cli_test). - Align relational ablation multi-root metadata test with Stage 3 fail-closed policy: ≥2 unique roots do not emit [root:] notes. [Verification]: Smoke suite passed locally (~133s, 180 passed); postfix targeted 5/5; full pre-fix was 4 failed / 13281 passed (2 env dirt/flake classified separately). --- core/cli.py | 156 +-------------------- tests/test_cli_test_suites.py | 7 +- tests/test_relational_operator_ablation.py | 10 +- 3 files changed, 21 insertions(+), 152 deletions(-) diff --git a/core/cli.py b/core/cli.py index ee03d8c3..140991a0 100644 --- a/core/cli.py +++ b/core/cli.py @@ -25,156 +25,12 @@ _CORE_RS_MANIFEST = _CORE_RS_DIR / "Cargo.toml" DESCRIPTION = "CORE versor engine command suite." EPILOG = 'Examples:\n core chat\n core pulse "What is truth?"\n core pulse --no-glove --json "Compare knowledge and wisdom"\n core bench\n core bench --suite all\n core bench --suite all --json --report bench_all.json\n core bench --suite determinism --runs 50\n core bench --suite speedup --json\n core trace "word beginning truth"\n core trace --output-language grc --frame-pack grc --json "logos"\n core rust status\n core rust build\n core oov covenant\n core pack list\n core pack verify en_minimal_v1\n core teaching audit\n core teaching audit --json\n core teaching gaps --top 10\n core teaching queue --threshold 3\n core teaching hitl-queue list\n core teaching hitl-queue list --state all --json\n core teaching hitl-queue show \n core teaching propose \n core teaching propose-from-exemplars teaching/admissibility_exemplars/rate_with_currency_v1.jsonl\n core teaching propose-from-exemplars --all\n core teaching proposals --state pending\n core teaching review --accept --review-date 2026-05-18\n core teaching supersede cause_light_reveals_truth --subject light --intent cause --connective grounds --object truth --review-date 2026-05-18\n core teaching supersessions\n core teaching supersessions --json\n core test --suite fast -q\n core test --suite pulse -q\n core test --suite proof -q\n core test --suite cognition -q\n core test -- tests/test_alignment_graph.py -q\n core demo audit-tour\n core demo register-tour\n core demo anchor-lens-tour\n core demo orthogonality-tour\n core demo pack-measurements\n core demo long-context-comparison\n core demo anti-regression\n core demo learning-loop\n core demo learning-arc\n core demo articulation\n core demo conversation\n core demo conversation --no-stream\n core demo all\n core demo adr-0024-chain\n core eval --list\n core eval cognition\n core eval cognition --json --save\n core eval cognition --split dev --version v1\n core eval cognition --split holdout\n core eval contemplation_quality\n core eval contemplation_quality --json --save\n core eval math-contemplation\n core eval math-contemplation --audit evals/gsm8k_math/train_sample/v1/audit_brief_11.json\n core eval math-contemplation --output teaching/math_proposals/proposals.jsonl\n core workbench api\n core workbench api --port 9000\n core workbench api --host 0.0.0.0 --allow-nonlocal-bind' -_TEST_SUITES: dict[str, tuple[str, ...]] = { - "fast": ( - "tests/test_cli_test_suites.py", - "tests/test_runtime_config.py", - "tests/test_core_semantic_seed_pack.py", - "tests/test_intent_proposition_graph.py", - "tests/test_articulation_realizer_v2.py", - "tests/test_reviewed_teaching_loop.py", - "tests/test_cognitive_eval_harness.py", - ), - "smoke": ( - "tests/test_chat_runtime.py", - "tests/test_achat.py", - "tests/test_runtime_config.py", - "tests/test_cognitive_turn_pipeline.py", - "tests/test_architectural_invariants.py", - # ADR-0043 — identity falsifiability: ratified identity packs must - # produce distinct, directionally-correct articulations, with a - # pack-invariant grounding/refusal floor and zero fabrication. Lives - # only under ``full`` historically, so a divergence regression cleared - # the PR gate and surfaced only post-merge. Promoted into smoke so - # the falsifiability claim blocks-on-regression rather than - # detect-after-merge. - "tests/test_pack_measurements_phase2.py", - ), - "runtime": ( - "tests/test_chat_runtime.py", - "tests/test_achat.py", - "tests/test_runtime_config.py", - "tests/test_session_coherence.py", - ), - "cognition": ( - "tests/test_intent_proposition_graph.py", - "tests/test_cognitive_turn_pipeline.py", - "tests/test_articulation_realizer_v2.py", - "tests/test_semantic_realizer_integration.py", - "tests/test_cognitive_eval_harness.py", - "tests/test_deterministic_hash.py", - "tests/test_morphology_irregular.py", - "tests/test_realizer_quantifier_agreement.py", - "tests/test_benchmarks_profiler.py", - "tests/test_compose_relations.py", - "tests/test_replay_vs_llm_benchmark.py", - ), - "teaching": ( - "tests/test_reviewed_teaching_loop.py", - "tests/test_pipeline_teaching_integration.py", - "tests/test_epistemic_invariants.py", - "tests/test_adr_0172_w2_decomposer.py", - "tests/test_adr_0172_w5_inference_proposal.py", - "tests/test_math_frame_ratification.py", - "tests/test_math_composition_ratification.py", - "tests/test_teaching_coverage_cli.py", - ), - "packs": ( - "tests/test_core_semantic_seed_pack.py", - "tests/test_adr_0127_pack_ratification.py", - "tests/test_frame_registry_load.py", - "tests/test_composition_registry_load.py", - "tests/test_composition_consult_in_injector.py", - "tests/test_consumption_case_0050_hazard_pin.py", - "tests/test_consumption_empty_registry_no_op.py", - "tests/test_consumption_partition.py", - "tests/test_matcher_extension_currency_per_unit.py", - "tests/test_matcher_extension_case_0050_hazard_pin.py", - "tests/test_matcher_extension_end_to_end_admission.py", - "tests/test_me2_cross_sentence_subject.py", - "tests/test_me2_case_0019_admits.py", - "tests/test_me3_additive_composition.py", - "tests/test_me4_subtractive_composition.py", - "tests/test_me5_all_categories_integration.py", - "tests/test_rat1_end_to_end_admission.py", - "tests/test_wave_a_multiplicative_aggregation_injector.py", - ), - "algebra": ( - "tests/test_versor_closure.py", - "tests/test_holonomy.py", - "tests/test_holonomy_resonance.py", - "tests/test_energy.py", - "tests/test_motor.py", - "tests/test_null_cone.py", - "tests/test_vault_recall.py", - "tests/test_vault_recall_vectorised.py", - "tests/test_vault_recall_rust_parity.py", - "tests/test_cga_inner_rust_parity.py", - "tests/test_geometric_product_rust_parity.py", - "tests/test_versor_condition_rust_parity.py", - "tests/test_versor_apply_rust_parity.py", - ), - "sensorium": ( - "tests/test_sensorium_compiler_delta.py", - "tests/test_audio_compiler.py", - "tests/test_audio_crdt_merge.py", - "tests/test_audio_eval_gates.py", - "tests/test_audio_pack_manifest.py", - "tests/test_audio_sensorium_mount.py", - "tests/test_vision_compiler.py", - "tests/test_event_vision_compiler.py", - "tests/test_vision_crdt_merge.py", - "tests/test_vision_eval_gates.py", - "tests/test_vision_sensorium_mount.py", - "tests/test_sensorimotor_contract.py", - "tests/test_sensorimotor_pack_manifest.py", - "tests/test_observation_frame_contract.py", - "tests/test_observation_frame_harness.py", - "tests/test_environment_falsification.py", - "tests/test_environment_falsification_eval_cli.py", - "tests/test_witness_log_importer.py", - "tests/test_tabletop_lab_protocol.py", - "tests/test_sensorium_eval_cli.py", - "tests/test_efferent_gate.py", - ), - "pulse": ( - "tests/test_pulse_integration.py", - "tests/test_graph_diffusion.py", - ), - "formation": ("tests/formation",), - "proof": ("tests/test_proof_properties.py",), - # ADR-0024 chain suites (Phases 2-6). Each phase has its own - # contract tests so investors / reviewers can run them - # independently; ``adr-0024`` runs the full chain end-to-end. - "refusal": ("tests/test_refusal_contract.py",), - "margin": ("tests/test_margin_admissibility.py",), - "rotor": ("tests/test_rotor_admissibility.py",), - "inner-loop": ( - "tests/test_inner_loop_admissibility.py", - "tests/test_inner_loop_phase2.py", - "tests/test_inner_loop_phase3.py", - "tests/test_inner_loop_phase4.py", - ), - "phase5": ("tests/test_phase5_corpus.py",), - "phase6": ("tests/test_phase6_demo.py",), - "adr-0024": ( - "tests/test_refusal_contract.py", - "tests/test_margin_admissibility.py", - "tests/test_rotor_admissibility.py", - "tests/test_inner_loop_admissibility.py", - "tests/test_inner_loop_phase2.py", - "tests/test_inner_loop_phase3.py", - "tests/test_inner_loop_phase4.py", - "tests/test_phase5_corpus.py", - "tests/test_phase6_demo.py", - ), - # ADR-0126 P6 — measurement harness for the GSM8K candidate-graph - # parser exit criterion. ``wrong == 0`` is a hard gate (Obligation - # #4: refuse rather than confabulate). - "math": ("tests/test_adr_0126_train_sample_runner.py",), - "deductive": ("tests/test_deductive_logic_entail.py",), - "full": ("tests/",), -} +# Canonical suite map lives in core.cli_test (runtime + --list-suites). +# Re-export here for argparse choices and tests that import core.cli. +# Do not maintain a second copy — Stage dual-pack / algebra pins drifted +# when cli.py lagged cli_test.TEST_SUITES (full-suite failure 2026-07-20). +from core.cli_test import TEST_SUITES as _TEST_SUITES # noqa: E402 + def _run(*args: str, check: bool = False, cwd: Path | None = None) -> int: diff --git a/tests/test_cli_test_suites.py b/tests/test_cli_test_suites.py index 71147c7e..b04f2e59 100644 --- a/tests/test_cli_test_suites.py +++ b/tests/test_cli_test_suites.py @@ -157,11 +157,16 @@ def test_core_test_suite_accepts_pytest_flags_without_separator(monkeypatch) -> # (growing) packs file list: the contract under test is that a curated # suite expands to its files followed by the forwarded "-q", with no "--" # separator needed. + # Assert against the runtime suite map (core.cli_test.TEST_SUITES), which + # cli.cmd_test actually expands. core.cli._TEST_SUITES is a re-export of + # the same object — do not pin a second stale list. + from core.cli_test import TEST_SUITES + assert calls[0] == ( cli.sys.executable, "-m", "pytest", - *cli._TEST_SUITES["packs"], + *TEST_SUITES["packs"], "-q", ) diff --git a/tests/test_relational_operator_ablation.py b/tests/test_relational_operator_ablation.py index 393f6e96..2dc7bf5e 100644 --- a/tests/test_relational_operator_ablation.py +++ b/tests/test_relational_operator_ablation.py @@ -269,6 +269,13 @@ def test_malformed_depth_labels_cannot_elevate_runnable() -> None: def test_valid_depth_metadata_on_gold_does_not_change_answer() -> None: + """Multi-root labels must not change the operator answer (Stage 3 fail-closed). + + With ≥2 unique roots, enrichment appends AMBIGUOUS_ROOTS provenance and does + **not** commit a silent ``[root:…]`` first-root note. Single-root decoration + with ``[root:…]`` is covered by + ``test_metadata_only_decorates_but_does_not_change_answer``. + """ depths = { "p0": {"language": "he", "root": "א-מ-נ"}, "p1": {"language": "grc", "root": "λογ"}, @@ -276,7 +283,8 @@ def test_valid_depth_metadata_on_gold_does_not_change_answer() -> None: op = run_operator(GOLD_0001) meta = run_metadata_only(GOLD_0001, depth_labels=depths) assert answers_match(op, meta) - assert meta.explanation_has_root_note is True + # Multi-root: fail-closed — no silent first-root [root:] commit. + assert meta.explanation_has_root_note is False def test_distractor_quantity_case_refuses() -> None: