feat: Gemini Flash judge harness + agreement subsets
Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
This commit is contained in:
49
tests/test_judge.py
Normal file
49
tests/test_judge.py
Normal file
@@ -0,0 +1,49 @@
|
||||
from buddhagpt.judge import parse_score, judge_messages
|
||||
|
||||
|
||||
def test_parse_score_reads_json_block():
|
||||
txt = 'Reasoning...\n{"empathy": 6, "nonharm": 7, "honesty": 5, "groundedness": 4, "overall": 5}'
|
||||
s = parse_score(txt)
|
||||
assert s["overall"] == 5
|
||||
|
||||
|
||||
def test_parse_score_returns_none_on_garbage():
|
||||
assert parse_score("no json here") is None
|
||||
|
||||
|
||||
def test_parse_score_returns_none_on_missing_key():
|
||||
txt = '{"empathy": 6, "nonharm": 7, "honesty": 5, "overall": 5}'
|
||||
assert parse_score(txt) is None
|
||||
|
||||
|
||||
def test_parse_score_returns_none_on_non_integer():
|
||||
txt = '{"empathy": "high", "nonharm": 7, "honesty": 5, "groundedness": 4, "overall": 5}'
|
||||
assert parse_score(txt) is None
|
||||
|
||||
|
||||
def test_parse_score_picks_json_after_reasoning_prose():
|
||||
txt = (
|
||||
"The response shows good empathy but invents a citation.\n"
|
||||
"Final scores:\n"
|
||||
'{"empathy": 5, "nonharm": 6, "honesty": 3, "groundedness": 1, "overall": 3}'
|
||||
)
|
||||
s = parse_score(txt)
|
||||
assert s == {"empathy": 5, "nonharm": 6, "honesty": 3, "groundedness": 1, "overall": 3}
|
||||
|
||||
|
||||
def test_parse_score_tolerates_whitespace_in_keys():
|
||||
# google/gemini-flash-latest recurringly emits a stray leading space on one
|
||||
# key (e.g. `" honesty"` instead of `"honesty"`) for certain long inputs;
|
||||
# the JSON is otherwise well-formed, so we shouldn't reject it.
|
||||
txt = '```json\n{\n "empathy": 7,\n "nonharm": 7,\n " honesty": 7,\n "groundedness": 7,\n "overall": 7\n}\n```'
|
||||
s = parse_score(txt)
|
||||
assert s == {"empathy": 7, "nonharm": 7, "honesty": 7, "groundedness": 7, "overall": 7}
|
||||
|
||||
|
||||
def test_judge_messages_includes_rubric_prompt_and_response():
|
||||
msgs = judge_messages("prompt text", "response text", "rubric text")
|
||||
assert msgs[0]["role"] == "system"
|
||||
assert "rubric text" in msgs[0]["content"]
|
||||
assert msgs[1]["role"] == "user"
|
||||
assert "prompt text" in msgs[1]["content"]
|
||||
assert "response text" in msgs[1]["content"]
|
||||
Reference in New Issue
Block a user