"""Tests that prove the verifier is trustworthy (it is the metric AND the later RL reward). Run: pytest tests/test_verifier.py -q """ import sys from pathlib import Path sys.path.insert(0, str(Path(__file__).resolve().parents[1])) from verifier.step_verifier import ( # noqa: E402 extract_final_answer, verify_assertion, verify_chain, ) # --------------------------------------------------------------------------- # # 1. correct + verifiable chain # --------------------------------------------------------------------------- # def test_correct_and_verifiable(): text = ( "Tom packs 3 bags with 8 apples each. " "So the total is <<3*8 = 24>> apples. " "The answer is \\boxed{24}." ) r = verify_chain(text, gold_answer=24) assert r["verifiable"] is True assert r["correct"] is True assert r["verified_and_correct"] is True assert r["n_assertions"] == 1 and r["n_verified"] == 1 assert r["composes_to_final"] is True # --------------------------------------------------------------------------- # # 2. one arithmetic error -> not verifiable, the bad step flagged # --------------------------------------------------------------------------- # def test_arithmetic_error_flagged(): text = "We compute <<2 + 2 = 5>> and conclude \\boxed{5}." r = verify_chain(text, gold_answer=5) assert r["verifiable"] is False assert r["n_assertions"] == 1 and r["n_verified"] == 0 assert any("2 + 2" in f.get("expr", "") for f in r["failures"]) # --------------------------------------------------------------------------- # # 3. THE fluency trap: fluent prose, right answer, NO assertions -> not verifiable # (proves the verifier rewards mechanical grounding, not plausible text) # --------------------------------------------------------------------------- # def test_fluency_trap(): text = ( "We carefully add up all the quantities involved, and after thinking it " "through it is clear the total works out to twenty-four, so the answer is " "\\boxed{24}." ) r = verify_chain(text, gold_answer=24) assert r["has_loadbearing_assertions"] is False assert r["verifiable"] is False # no load-bearing assertions assert r["correct"] is True # final answer still matches gold assert r["verified_and_correct"] is False # the two axes are independent # --------------------------------------------------------------------------- # # 4. steps verify but final answer does not compose from them -> not verifiable # --------------------------------------------------------------------------- # def test_non_composing_final(): text = ( "First <<10 * 2 = 20>>, then <<20 + 5 = 25>>. " "Therefore the answer is \\boxed{30}." ) r = verify_chain(text, gold_answer=30) assert r["all_assertions_verified"] is True assert r["n_verified"] == 2 assert r["composes_to_final"] is False assert r["verifiable"] is False assert any("compose" in f.get("reason", "") for f in r["failures"]) # --------------------------------------------------------------------------- # # 5a. verifiable BUT WRONG (internally consistent + composes, final != gold) # --------------------------------------------------------------------------- # def test_verifiable_but_wrong(): text = "Clearly <<2 + 2 = 4>>, so \\boxed{4}." r = verify_chain(text, gold_answer=5) assert r["verifiable"] is True assert r["correct"] is False assert r["verified_and_correct"] is False # --------------------------------------------------------------------------- # # 5b. correct BUT UNVERIFIABLE — same as the fluency trap axis, asserted distinctly # --------------------------------------------------------------------------- # def test_correct_but_unverifiable(): text = "After some mental arithmetic the answer is \\boxed{42}." r = verify_chain(text, gold_answer=42) assert r["correct"] is True assert r["verifiable"] is False # --------------------------------------------------------------------------- # # 6. variable binding threads forward # --------------------------------------------------------------------------- # def test_variable_binding(): text = ( "Let total = <<3 * 8 = 24>>. " "Adding the bonus: <>. " "So \\boxed{30}." ) r = verify_chain(text, gold_answer=30) assert r["n_assertions"] == 2 and r["n_verified"] == 2 assert r["verifiable"] is True and r["correct"] is True # --------------------------------------------------------------------------- # # 7. fail-closed on an unbound symbol (v1 out-of-scope algebra) # --------------------------------------------------------------------------- # def test_unbound_symbol_fails_closed(): text = "We have <> hence \\boxed{3}." r = verify_chain(text, gold_answer=3) assert r["n_verified"] == 0 assert r["verifiable"] is False assert any("unbound" in f.get("reason", "") for f in r["failures"]) # --------------------------------------------------------------------------- # # 8. tolerance policy: exact rationals + 1e-6 float # --------------------------------------------------------------------------- # def test_exact_rational(): # decimals parse as exact rationals, so this is an EXACT match assert verify_assertion("0.1 + 0.2", "0.3", {}).ok is True def test_float_tolerance(): r = verify_assertion("2**0.5", "1.4142135", {}) assert r.ok is True # within 1e-6 relative def test_float_outside_tolerance(): r = verify_assertion("2**0.5", "1.41", {}) assert r.ok is False def test_thousands_separator(): assert verify_assertion("1,000 + 234", "1,234", {}).ok is True # --------------------------------------------------------------------------- # # 9. final-answer extraction priority # --------------------------------------------------------------------------- # def test_final_answer_extraction(): assert extract_final_answer("blah \\boxed{7} blah") == "7" assert extract_final_answer("steps...\n#### 42") == "42" assert extract_final_answer("so the answer is 13.") == "13" assert extract_final_answer("no answer here") is None def test_unparseable_fails_closed(): r = verify_assertion("3 +* 4", "7", {}) assert r.ok is False