diff --git a/src/trinity/orchestration/reward.py b/src/trinity/orchestration/reward.py index b1da624..3376a91 100644 --- a/src/trinity/orchestration/reward.py +++ b/src/trinity/orchestration/reward.py @@ -126,6 +126,12 @@ def _committed_answer(benchmark: str, traj: Trajectory) -> str: ``final_answer``. This applies equally to TRINITY and the random baseline (the single-model baseline is one turn, so it is unaffected) — a fair fix, not a thumb on the scale. See JOURNAL 2026-06-23 (MMLU extraction diagnosis). + + Verifier turns are skipped in the scan, mirroring ``_final_answer`` + (session.py): the Verifier is a critic, not a solver, and it routinely quotes + the gold answer while returning ``VERDICT: REVISE``. Scoring that quote would + credit a trajectory the Verifier explicitly rejected — marking a run whose + Worker never produced the answer as correct. """ key = (benchmark or "").strip().lower() final = traj.final_answer or "" @@ -134,6 +140,8 @@ def _committed_answer(benchmark: str, traj: Trajectory) -> str: if has_answer(key, final): return final for tr in reversed(turns): + if getattr(tr, "role", None) == Role.VERIFIER: + continue txt = getattr(tr, "processed_output", "") or "" if has_answer(key, txt): return txt diff --git a/tests/test_reward_checkers.py b/tests/test_reward_checkers.py index 30d8d6f..4e4d72e 100644 --- a/tests/test_reward_checkers.py +++ b/tests/test_reward_checkers.py @@ -74,3 +74,56 @@ def test_code_empty_tests_fail_closed(): def test_extract_code_returns_last_fenced_block(): text = "```python\nold = 1\n```\nSome text\n```python\nnew = 2\n```" assert "new = 2" in R.extract_code(text) + + +# --------------------------------------------------------------------------- +# Committed-answer selection across a multi-turn trajectory +# --------------------------------------------------------------------------- +from trinity.orchestration.session import _final_answer # noqa: E402 +from trinity.types import Role, Task, Trajectory, TurnRecord # noqa: E402 + + +def _turn(role: Role, out: str) -> TurnRecord: + return TurnRecord(turn=1, agent_name="m", role=role, raw_output=out, processed_output=out) + + +def _traj(turns: list[TurnRecord], *, answer: str = "42", benchmark: str = "math500") -> Trajectory: + traj = Trajectory(task=Task("t", benchmark, "q", answer), turns=turns) + traj.final_answer = _final_answer(traj) # mirror the real pipeline + return traj + + +def test_committed_answer_skips_verifier_that_quotes_gold(): + # The Worker produced no answer; the Verifier quoted the gold value while + # REVISE-ing. The trajectory was rejected and must not score as correct. + traj = _traj( + [ + _turn(Role.WORKER, "I am not sure how to finish this."), + _turn(Role.VERIFIER, "The correct value is 42, but the worker never computed it.\nVERDICT: REVISE"), + ] + ) + assert R._committed_answer("math500", traj) == "I am not sure how to finish this." + assert R.score(traj) == 0.0 + + +def test_committed_answer_still_recovers_earlier_nonverifier_answer(): + # Legitimate recovery is preserved: when the final (last-Worker) output has no + # parseable answer, an earlier NON-verifier turn's answer is still credited. + traj = _traj( + [ + _turn(Role.THINKER, r"Working it through, the result is \boxed{42}."), + _turn(Role.WORKER, "On reflection I am not certain, let me not commit a value."), + ] + ) + assert R.score(traj) == 1.0 + + +def test_committed_answer_uses_worker_answer_with_verifier_accept(): + # Control: a real Worker answer followed by an ACCEPT is scored from the Worker. + traj = _traj( + [ + _turn(Role.WORKER, r"The answer is \boxed{42}."), + _turn(Role.VERIFIER, "Looks right.\nVERDICT: ACCEPT"), + ] + ) + assert R.score(traj) == 1.0