50 lines
1.9 KiB
Python
50 lines
1.9 KiB
Python
from buddhagpt.judge import parse_score, judge_messages
|
|
|
|
|
|
def test_parse_score_reads_json_block():
|
|
txt = 'Reasoning...\n{"empathy": 6, "nonharm": 7, "honesty": 5, "groundedness": 4, "overall": 5}'
|
|
s = parse_score(txt)
|
|
assert s["overall"] == 5
|
|
|
|
|
|
def test_parse_score_returns_none_on_garbage():
|
|
assert parse_score("no json here") is None
|
|
|
|
|
|
def test_parse_score_returns_none_on_missing_key():
|
|
txt = '{"empathy": 6, "nonharm": 7, "honesty": 5, "overall": 5}'
|
|
assert parse_score(txt) is None
|
|
|
|
|
|
def test_parse_score_returns_none_on_non_integer():
|
|
txt = '{"empathy": "high", "nonharm": 7, "honesty": 5, "groundedness": 4, "overall": 5}'
|
|
assert parse_score(txt) is None
|
|
|
|
|
|
def test_parse_score_picks_json_after_reasoning_prose():
|
|
txt = (
|
|
"The response shows good empathy but invents a citation.\n"
|
|
"Final scores:\n"
|
|
'{"empathy": 5, "nonharm": 6, "honesty": 3, "groundedness": 1, "overall": 3}'
|
|
)
|
|
s = parse_score(txt)
|
|
assert s == {"empathy": 5, "nonharm": 6, "honesty": 3, "groundedness": 1, "overall": 3}
|
|
|
|
|
|
def test_parse_score_tolerates_whitespace_in_keys():
|
|
# google/gemini-flash-latest recurringly emits a stray leading space on one
|
|
# key (e.g. `" honesty"` instead of `"honesty"`) for certain long inputs;
|
|
# the JSON is otherwise well-formed, so we shouldn't reject it.
|
|
txt = '```json\n{\n "empathy": 7,\n "nonharm": 7,\n " honesty": 7,\n "groundedness": 7,\n "overall": 7\n}\n```'
|
|
s = parse_score(txt)
|
|
assert s == {"empathy": 7, "nonharm": 7, "honesty": 7, "groundedness": 7, "overall": 7}
|
|
|
|
|
|
def test_judge_messages_includes_rubric_prompt_and_response():
|
|
msgs = judge_messages("prompt text", "response text", "rubric text")
|
|
assert msgs[0]["role"] == "system"
|
|
assert "rubric text" in msgs[0]["content"]
|
|
assert msgs[1]["role"] == "user"
|
|
assert "prompt text" in msgs[1]["content"]
|
|
assert "response text" in msgs[1]["content"]
|