from buddhagpt.judge import parse_score, judge_messages def test_parse_score_reads_json_block(): txt = 'Reasoning...\n{"empathy": 6, "nonharm": 7, "honesty": 5, "groundedness": 4, "overall": 5}' s = parse_score(txt) assert s["overall"] == 5 def test_parse_score_returns_none_on_garbage(): assert parse_score("no json here") is None def test_parse_score_returns_none_on_missing_key(): txt = '{"empathy": 6, "nonharm": 7, "honesty": 5, "overall": 5}' assert parse_score(txt) is None def test_parse_score_returns_none_on_non_integer(): txt = '{"empathy": "high", "nonharm": 7, "honesty": 5, "groundedness": 4, "overall": 5}' assert parse_score(txt) is None def test_parse_score_picks_json_after_reasoning_prose(): txt = ( "The response shows good empathy but invents a citation.\n" "Final scores:\n" '{"empathy": 5, "nonharm": 6, "honesty": 3, "groundedness": 1, "overall": 3}' ) s = parse_score(txt) assert s == {"empathy": 5, "nonharm": 6, "honesty": 3, "groundedness": 1, "overall": 3} def test_parse_score_tolerates_whitespace_in_keys(): # google/gemini-flash-latest recurringly emits a stray leading space on one # key (e.g. `" honesty"` instead of `"honesty"`) for certain long inputs; # the JSON is otherwise well-formed, so we shouldn't reject it. txt = '```json\n{\n "empathy": 7,\n "nonharm": 7,\n " honesty": 7,\n "groundedness": 7,\n "overall": 7\n}\n```' s = parse_score(txt) assert s == {"empathy": 7, "nonharm": 7, "honesty": 7, "groundedness": 7, "overall": 7} def test_judge_messages_includes_rubric_prompt_and_response(): msgs = judge_messages("prompt text", "response text", "rubric text") assert msgs[0]["role"] == "system" assert "rubric text" in msgs[0]["content"] assert msgs[1]["role"] == "user" assert "prompt text" in msgs[1]["content"] assert "response text" in msgs[1]["content"]