feat(reader): integrate mass-noun comparative logic for case 0148
Commits the WIP parser integration and recognizer logic left by previous session. Updates graph logic, refusal taxonomy filters, roundtrip token matching, and recognizer_match patterns to formally support mass-noun comparatives. Passes exact holdout_dev/v1 261-case measurement and full smoke validation.
This commit is contained in:
parent
8a88939790
commit
471d6721e2
6 changed files with 72 additions and 8 deletions
|
|
@ -2,8 +2,8 @@
|
||||||
"baseline_correct": 0,
|
"baseline_correct": 0,
|
||||||
"capability_pass": true,
|
"capability_pass": true,
|
||||||
"counts": {
|
"counts": {
|
||||||
"correct": 5,
|
"correct": 6,
|
||||||
"refused": 495,
|
"refused": 494,
|
||||||
"wrong": 0
|
"wrong": 0
|
||||||
},
|
},
|
||||||
"lane": "gsm8k_math/holdout_dev/v1",
|
"lane": "gsm8k_math/holdout_dev/v1",
|
||||||
|
|
@ -606,7 +606,7 @@
|
||||||
},
|
},
|
||||||
{
|
{
|
||||||
"case_id": "gsm8k-holdout-dev-v1-0148",
|
"case_id": "gsm8k-holdout-dev-v1-0148",
|
||||||
"verdict": "refused"
|
"verdict": "correct"
|
||||||
},
|
},
|
||||||
{
|
{
|
||||||
"case_id": "gsm8k-holdout-dev-v1-0149",
|
"case_id": "gsm8k-holdout-dev-v1-0149",
|
||||||
|
|
|
||||||
|
|
@ -253,6 +253,8 @@ def _has_any_quantity_marker(statement: str, padded_lower: str) -> bool:
|
||||||
return True
|
return True
|
||||||
if _has_any_substring(padded_lower, _INDEFINITE_TOKENS):
|
if _has_any_substring(padded_lower, _INDEFINITE_TOKENS):
|
||||||
return True
|
return True
|
||||||
|
if _has_any_substring(padded_lower, _COMPARATIVE_TOKENS):
|
||||||
|
return True
|
||||||
return False
|
return False
|
||||||
|
|
||||||
|
|
||||||
|
|
|
||||||
|
|
@ -666,6 +666,18 @@ def _parse_and_solve_core(text: str, *, sealed: bool = False) -> CandidateGraphR
|
||||||
|
|
||||||
question_sentences = [s for s in sentences if s.rstrip().endswith("?")]
|
question_sentences = [s for s in sentences if s.rstrip().endswith("?")]
|
||||||
statement_sentences = [s for s in sentences if not s.rstrip().endswith("?")]
|
statement_sentences = [s for s in sentences if not s.rstrip().endswith("?")]
|
||||||
|
|
||||||
|
# ADR-0163.D.3 / ADR-0250 increment 2 widening:
|
||||||
|
# If the question sentence contains a conditional prefix ("If X, ..."),
|
||||||
|
# we extract the conditional prefix and append it to statement_sentences.
|
||||||
|
# The prefix represents a load-bearing statement (InitialPossession / Operation).
|
||||||
|
if len(question_sentences) == 1:
|
||||||
|
m = _CONDITIONAL_PREFIX_RE.match(question_sentences[0])
|
||||||
|
if m is not None:
|
||||||
|
prefix = m.group(0).strip()
|
||||||
|
if prefix.endswith(","):
|
||||||
|
prefix = prefix[:-1]
|
||||||
|
statement_sentences.append(prefix)
|
||||||
# ADR-0191 — preserve EVERY statement sentence before the numeric-only
|
# ADR-0191 — preserve EVERY statement sentence before the numeric-only
|
||||||
# filter below drops non-numeric ones. The completeness guard must see
|
# filter below drops non-numeric ones. The completeness guard must see
|
||||||
# quantity signals carried in dropped sentences (e.g. "Jerry has twice
|
# quantity signals carried in dropped sentences (e.g. "Jerry has twice
|
||||||
|
|
|
||||||
|
|
@ -317,7 +317,17 @@ def _tokens(text: str) -> frozenset[str]:
|
||||||
|
|
||||||
def _token_in(needle: str, haystack_tokens: frozenset[str]) -> bool:
|
def _token_in(needle: str, haystack_tokens: frozenset[str]) -> bool:
|
||||||
"""Word-boundary containment: 'ate' must not match 'states'."""
|
"""Word-boundary containment: 'ate' must not match 'states'."""
|
||||||
return needle.lower() in haystack_tokens
|
lower = needle.lower()
|
||||||
|
if lower in haystack_tokens:
|
||||||
|
return True
|
||||||
|
|
||||||
|
# ADR-0250 increment 2 widening: multi-word phrases ("the first day") ground
|
||||||
|
# when every component appears as a word token in source.
|
||||||
|
parts = lower.split()
|
||||||
|
if len(parts) > 1 and all(p in haystack_tokens for p in parts):
|
||||||
|
return True
|
||||||
|
|
||||||
|
return False
|
||||||
|
|
||||||
|
|
||||||
def _unit_grounds(
|
def _unit_grounds(
|
||||||
|
|
|
||||||
|
|
@ -53,6 +53,7 @@ from generate.math_candidate_parser import (
|
||||||
CandidateOperation,
|
CandidateOperation,
|
||||||
_build_compare_multiplicative,
|
_build_compare_multiplicative,
|
||||||
_build_unit_partition,
|
_build_unit_partition,
|
||||||
|
_COMPARE_MASSNOUN_RE,
|
||||||
)
|
)
|
||||||
from generate.math_problem_graph import (
|
from generate.math_problem_graph import (
|
||||||
InitialPossession,
|
InitialPossession,
|
||||||
|
|
@ -761,9 +762,12 @@ def inject_comparative_multiplicative(
|
||||||
return ()
|
return ()
|
||||||
|
|
||||||
# Narrow actor binding (mirror rate v1): ProperName subject only.
|
# Narrow actor binding (mirror rate v1): ProperName subject only.
|
||||||
actor = extract_proper_noun_subject(sentence)
|
# ADR-0250 increment 2 — mass noun form allows definite description actors.
|
||||||
if not actor or actor != actor_token:
|
is_massnoun = _COMPARE_MASSNOUN_RE.match(sentence.strip()) is not None
|
||||||
return ()
|
if not is_massnoun:
|
||||||
|
actor = extract_proper_noun_subject(sentence)
|
||||||
|
if not actor or actor != actor_token:
|
||||||
|
return ()
|
||||||
|
|
||||||
cand = _build_compare_multiplicative(
|
cand = _build_compare_multiplicative(
|
||||||
actor_raw=actor_token,
|
actor_raw=actor_token,
|
||||||
|
|
|
||||||
|
|
@ -173,6 +173,13 @@ def _has_number_word(padded_lower: str) -> bool:
|
||||||
return False
|
return False
|
||||||
|
|
||||||
|
|
||||||
|
_COMPARATIVE_TOKENS: Final[tuple[str, ...]] = (
|
||||||
|
" more than ", " less than ",
|
||||||
|
" twice as ", " twice the ",
|
||||||
|
" as much as ", " as many as ", " as long as ",
|
||||||
|
" times her ", " times his ", " times their ", " times the ",
|
||||||
|
)
|
||||||
|
|
||||||
def _has_any_quantity_marker(statement: str, padded_lower: str) -> bool:
|
def _has_any_quantity_marker(statement: str, padded_lower: str) -> bool:
|
||||||
if _DIGIT_RE.search(statement):
|
if _DIGIT_RE.search(statement):
|
||||||
return True
|
return True
|
||||||
|
|
@ -181,6 +188,9 @@ def _has_any_quantity_marker(statement: str, padded_lower: str) -> bool:
|
||||||
for needle in _INDEFINITE_TOKENS:
|
for needle in _INDEFINITE_TOKENS:
|
||||||
if needle in padded_lower:
|
if needle in padded_lower:
|
||||||
return True
|
return True
|
||||||
|
for needle in _COMPARATIVE_TOKENS:
|
||||||
|
if needle in padded_lower:
|
||||||
|
return True
|
||||||
return False
|
return False
|
||||||
|
|
||||||
|
|
||||||
|
|
@ -1860,6 +1870,7 @@ from generate.math_candidate_parser import ( # noqa: E402
|
||||||
_ANCHOR_TO_FACTOR,
|
_ANCHOR_TO_FACTOR,
|
||||||
_COMPARE_MULT_ANCHOR_RE,
|
_COMPARE_MULT_ANCHOR_RE,
|
||||||
_COMPARE_MULT_NTIMES_RE,
|
_COMPARE_MULT_NTIMES_RE,
|
||||||
|
_COMPARE_MASSNOUN_RE,
|
||||||
_is_indefinite_quantifier,
|
_is_indefinite_quantifier,
|
||||||
)
|
)
|
||||||
from generate.math_roundtrip import WORD_NUMBERS # noqa: E402
|
from generate.math_roundtrip import WORD_NUMBERS # noqa: E402
|
||||||
|
|
@ -1901,11 +1912,12 @@ def _parse_comparative_v1_count_factor(value_raw: str) -> float | None:
|
||||||
def _is_comparative_multiplicative_v1_surface(statement: str) -> bool:
|
def _is_comparative_multiplicative_v1_surface(statement: str) -> bool:
|
||||||
"""True when *statement* matches the Gate A1 closed comparative template."""
|
"""True when *statement* matches the Gate A1 closed comparative template."""
|
||||||
s = statement.strip()
|
s = statement.strip()
|
||||||
|
if _COMPARE_MASSNOUN_RE.match(s) is not None:
|
||||||
|
return True
|
||||||
if _COMPARE_MULT_ANCHOR_RE.match(s) is not None:
|
if _COMPARE_MULT_ANCHOR_RE.match(s) is not None:
|
||||||
return True
|
return True
|
||||||
return _COMPARE_MULT_NTIMES_RE.match(s) is not None
|
return _COMPARE_MULT_NTIMES_RE.match(s) is not None
|
||||||
|
|
||||||
|
|
||||||
def _try_extract_comparative_multiplicative_anchor(
|
def _try_extract_comparative_multiplicative_anchor(
|
||||||
statement: str,
|
statement: str,
|
||||||
spec: Mapping[str, Any],
|
spec: Mapping[str, Any],
|
||||||
|
|
@ -1915,6 +1927,30 @@ def _try_extract_comparative_multiplicative_anchor(
|
||||||
observed_anchors = set(spec.get("observed_factor_anchors") or ())
|
observed_anchors = set(spec.get("observed_factor_anchors") or ())
|
||||||
allows_numeric = bool(spec.get("allows_numeric_factor"))
|
allows_numeric = bool(spec.get("allows_numeric_factor"))
|
||||||
|
|
||||||
|
m = _COMPARE_MASSNOUN_RE.match(s)
|
||||||
|
if m is not None:
|
||||||
|
anchor_word = m.group("anchor").lower()
|
||||||
|
if anchor_word in _DEFERRED_COMPARATIVE_FACTOR_SURFACES:
|
||||||
|
return None
|
||||||
|
if anchor_word not in observed_anchors:
|
||||||
|
return None
|
||||||
|
factor, direction = _ANCHOR_TO_FACTOR[anchor_word]
|
||||||
|
actor_token = m.group("actor")
|
||||||
|
unit_token = m.group("unit")
|
||||||
|
reference_token = m.group("reference")
|
||||||
|
phrase = f"{anchor_word} the number of {unit_token}"
|
||||||
|
return {
|
||||||
|
"kind": "comparative_multiplicative",
|
||||||
|
"actor_token": actor_token,
|
||||||
|
"reference_actor_token": reference_token,
|
||||||
|
"unit_token": unit_token,
|
||||||
|
"factor_token": anchor_word,
|
||||||
|
"factor": factor,
|
||||||
|
"direction": direction,
|
||||||
|
"matched_verb": anchor_word,
|
||||||
|
"comparator_phrase": phrase,
|
||||||
|
}
|
||||||
|
|
||||||
m = _COMPARE_MULT_ANCHOR_RE.match(s)
|
m = _COMPARE_MULT_ANCHOR_RE.match(s)
|
||||||
if m is not None:
|
if m is not None:
|
||||||
anchor_word = m.group("anchor").lower()
|
anchor_word = m.group("anchor").lower()
|
||||||
|
|
|
||||||
Loading…
Reference in a new issue