feat(reader): integrate mass-noun comparative logic for case 0148

Commits the WIP parser integration and recognizer logic left by previous session. Updates graph logic, refusal taxonomy filters, roundtrip token matching, and recognizer_match patterns to formally support mass-noun comparatives.
Passes exact holdout_dev/v1 261-case measurement and full smoke validation.
This commit is contained in:
Shay 2026-07-19 11:40:37 -07:00
parent 8a88939790
commit 471d6721e2
6 changed files with 72 additions and 8 deletions

View file

@ -2,8 +2,8 @@
"baseline_correct": 0,
"capability_pass": true,
"counts": {
"correct": 5,
"refused": 495,
"correct": 6,
"refused": 494,
"wrong": 0
},
"lane": "gsm8k_math/holdout_dev/v1",
@ -606,7 +606,7 @@
},
{
"case_id": "gsm8k-holdout-dev-v1-0148",
"verdict": "refused"
"verdict": "correct"
},
{
"case_id": "gsm8k-holdout-dev-v1-0149",

View file

@ -253,6 +253,8 @@ def _has_any_quantity_marker(statement: str, padded_lower: str) -> bool:
return True
if _has_any_substring(padded_lower, _INDEFINITE_TOKENS):
return True
if _has_any_substring(padded_lower, _COMPARATIVE_TOKENS):
return True
return False

View file

@ -666,6 +666,18 @@ def _parse_and_solve_core(text: str, *, sealed: bool = False) -> CandidateGraphR
question_sentences = [s for s in sentences if s.rstrip().endswith("?")]
statement_sentences = [s for s in sentences if not s.rstrip().endswith("?")]
# ADR-0163.D.3 / ADR-0250 increment 2 widening:
# If the question sentence contains a conditional prefix ("If X, ..."),
# we extract the conditional prefix and append it to statement_sentences.
# The prefix represents a load-bearing statement (InitialPossession / Operation).
if len(question_sentences) == 1:
m = _CONDITIONAL_PREFIX_RE.match(question_sentences[0])
if m is not None:
prefix = m.group(0).strip()
if prefix.endswith(","):
prefix = prefix[:-1]
statement_sentences.append(prefix)
# ADR-0191 — preserve EVERY statement sentence before the numeric-only
# filter below drops non-numeric ones. The completeness guard must see
# quantity signals carried in dropped sentences (e.g. "Jerry has twice

View file

@ -317,7 +317,17 @@ def _tokens(text: str) -> frozenset[str]:
def _token_in(needle: str, haystack_tokens: frozenset[str]) -> bool:
"""Word-boundary containment: 'ate' must not match 'states'."""
return needle.lower() in haystack_tokens
lower = needle.lower()
if lower in haystack_tokens:
return True
# ADR-0250 increment 2 widening: multi-word phrases ("the first day") ground
# when every component appears as a word token in source.
parts = lower.split()
if len(parts) > 1 and all(p in haystack_tokens for p in parts):
return True
return False
def _unit_grounds(

View file

@ -53,6 +53,7 @@ from generate.math_candidate_parser import (
CandidateOperation,
_build_compare_multiplicative,
_build_unit_partition,
_COMPARE_MASSNOUN_RE,
)
from generate.math_problem_graph import (
InitialPossession,
@ -761,9 +762,12 @@ def inject_comparative_multiplicative(
return ()
# Narrow actor binding (mirror rate v1): ProperName subject only.
actor = extract_proper_noun_subject(sentence)
if not actor or actor != actor_token:
return ()
# ADR-0250 increment 2 — mass noun form allows definite description actors.
is_massnoun = _COMPARE_MASSNOUN_RE.match(sentence.strip()) is not None
if not is_massnoun:
actor = extract_proper_noun_subject(sentence)
if not actor or actor != actor_token:
return ()
cand = _build_compare_multiplicative(
actor_raw=actor_token,

View file

@ -173,6 +173,13 @@ def _has_number_word(padded_lower: str) -> bool:
return False
_COMPARATIVE_TOKENS: Final[tuple[str, ...]] = (
" more than ", " less than ",
" twice as ", " twice the ",
" as much as ", " as many as ", " as long as ",
" times her ", " times his ", " times their ", " times the ",
)
def _has_any_quantity_marker(statement: str, padded_lower: str) -> bool:
if _DIGIT_RE.search(statement):
return True
@ -181,6 +188,9 @@ def _has_any_quantity_marker(statement: str, padded_lower: str) -> bool:
for needle in _INDEFINITE_TOKENS:
if needle in padded_lower:
return True
for needle in _COMPARATIVE_TOKENS:
if needle in padded_lower:
return True
return False
@ -1860,6 +1870,7 @@ from generate.math_candidate_parser import ( # noqa: E402
_ANCHOR_TO_FACTOR,
_COMPARE_MULT_ANCHOR_RE,
_COMPARE_MULT_NTIMES_RE,
_COMPARE_MASSNOUN_RE,
_is_indefinite_quantifier,
)
from generate.math_roundtrip import WORD_NUMBERS # noqa: E402
@ -1901,11 +1912,12 @@ def _parse_comparative_v1_count_factor(value_raw: str) -> float | None:
def _is_comparative_multiplicative_v1_surface(statement: str) -> bool:
"""True when *statement* matches the Gate A1 closed comparative template."""
s = statement.strip()
if _COMPARE_MASSNOUN_RE.match(s) is not None:
return True
if _COMPARE_MULT_ANCHOR_RE.match(s) is not None:
return True
return _COMPARE_MULT_NTIMES_RE.match(s) is not None
def _try_extract_comparative_multiplicative_anchor(
statement: str,
spec: Mapping[str, Any],
@ -1915,6 +1927,30 @@ def _try_extract_comparative_multiplicative_anchor(
observed_anchors = set(spec.get("observed_factor_anchors") or ())
allows_numeric = bool(spec.get("allows_numeric_factor"))
m = _COMPARE_MASSNOUN_RE.match(s)
if m is not None:
anchor_word = m.group("anchor").lower()
if anchor_word in _DEFERRED_COMPARATIVE_FACTOR_SURFACES:
return None
if anchor_word not in observed_anchors:
return None
factor, direction = _ANCHOR_TO_FACTOR[anchor_word]
actor_token = m.group("actor")
unit_token = m.group("unit")
reference_token = m.group("reference")
phrase = f"{anchor_word} the number of {unit_token}"
return {
"kind": "comparative_multiplicative",
"actor_token": actor_token,
"reference_actor_token": reference_token,
"unit_token": unit_token,
"factor_token": anchor_word,
"factor": factor,
"direction": direction,
"matched_verb": anchor_word,
"comparator_phrase": phrase,
}
m = _COMPARE_MULT_ANCHOR_RE.match(s)
if m is not None:
anchor_word = m.group("anchor").lower()