Files
buddha-gpt/scripts/judge.py
2026-08-17 21:15:24 -07:00

195 lines
8.0 KiB
Python

"""LLM-judge harness: score every data/responses.jsonl row with the primary
judge (google/gemini-flash-latest), plus a second judge (deepseek/deepseek-v4-pro)
on a 100-row seeded subset, for judge-judge agreement in Task 10. Also exports
a blinded 30-row human-rating CSV plus a key file for the join.
Deviates from the task-8 brief's inline pseudocode on purpose (per task
instructions): the brief's script overwrites data/scores.jsonl wholesale with
Path.write_text() each run. Instead, like scripts/collect_responses.py, this
appends and is resumable: it skips any (prompt_id, system, judge) triple
already present in data/scores.jsonl, so a restart after an interruption does
not re-spend API calls or money.
Token budget: max_tokens=800 per the brief. If more than 3% of rows in a pass
come back empty/unparseable, the failed rows are automatically retried once at
max_tokens=2000 (a note is printed either way, so the report can cite the
observed rate).
Human subset (data/human_subset.csv) is blinded — no system column — so Marcus
rates without knowing which model produced each response. data/human_subset_key.csv
carries row_id -> (prompt_id, system) for Task 10's join and is written
separately so it never ends up in the blinded file. Both use random.seed(7)
exactly as specified in the brief, and are only generated once: if
human_subset.csv already exists, export is skipped so we never clobber Marcus's
in-progress ratings.
"""
import json
import random
from concurrent.futures import ThreadPoolExecutor
from pathlib import Path
from buddhagpt.collect import load_bench
from buddhagpt.judge import judge_messages, parse_score
from buddhagpt.llm import chat, openrouter_client
BENCH = Path("eval/compassionbench.yaml")
RESPONSES = Path("data/responses.jsonl")
RUBRIC = Path("eval/rubric.md")
SCORES = Path("data/scores.jsonl")
HUMAN_SUBSET = Path("data/human_subset.csv")
HUMAN_SUBSET_KEY = Path("data/human_subset_key.csv")
# OpenRouter serves the "latest" Gemini alias only under a tilde-prefixed
# canonical slug (confirmed via GET /models); "google/gemini-flash-latest"
# without the tilde 400s as an invalid model ID.
JUDGE = "~google/gemini-flash-latest"
SECOND_JUDGE = "deepseek/deepseek-v4-pro" # agreement check on a 100-row subset
MAX_WORKERS = 8
DEFAULT_MAX_TOKENS = 800
RETRY_MAX_TOKENS = 2000
PARSE_FAIL_THRESHOLD = 0.03
def load_scores_done(scores_path: Path) -> set[tuple[str, str, str]]:
"""(prompt_id, system, judge) triples already present in an existing
scores file — mirrors buddhagpt.collect.load_done's role for responses."""
done = set()
if not scores_path.exists():
return done
for line in scores_path.read_text().splitlines():
if not line.strip():
continue
row = json.loads(line)
done.add((row["prompt_id"], row["system"], row["judge"]))
return done
def run_pass(client, items, judge_model, rubric, bench, out_f):
"""Score `items` with `judge_model`, writing each parsed result to out_f
as it completes. Returns (written, fail_count, totals, retried_count)."""
if not items:
return 0, 0, {"input": 0, "output": 0}, 0
totals = {"input": 0, "output": 0}
def score_one(r, max_tokens):
prompt = bench[r["prompt_id"]]["prompt"]
try:
text, usage = chat(
client, judge_model,
judge_messages(prompt, r["response"], rubric),
max_tokens=max_tokens,
)
except Exception as e:
return r, None, {"input": 0, "output": 0}, str(e)
return r, parse_score(text), usage, None
def run_round(rows, max_tokens):
with ThreadPoolExecutor(max_workers=MAX_WORKERS) as pool:
futs = [pool.submit(score_one, r, max_tokens) for r in rows]
return [f.result() for f in futs]
written = 0
results = run_round(items, DEFAULT_MAX_TOKENS)
failed_rows = []
for r, s, usage, err in results:
totals["input"] += usage["input"]; totals["output"] += usage["output"]
if s is None:
failed_rows.append(r)
if err:
print(f"[{judge_model}] error {r['prompt_id']}/{r['system']}: {err}", flush=True)
else:
out_f.write(json.dumps({
"prompt_id": r["prompt_id"], "system": r["system"], "judge": judge_model, **s,
}) + "\n")
out_f.flush()
written += 1
fail_rate = len(failed_rows) / len(items)
retried = 0
fail_count = len(failed_rows)
if fail_rate > PARSE_FAIL_THRESHOLD and failed_rows:
print(
f"[{judge_model}] parse-fail rate {fail_rate:.1%} on first pass (>3%), "
f"retrying {len(failed_rows)} rows at max_tokens={RETRY_MAX_TOKENS}",
flush=True,
)
retried = len(failed_rows)
retry_results = run_round(failed_rows, RETRY_MAX_TOKENS)
fail_count = 0
for r, s, usage, err in retry_results:
totals["input"] += usage["input"]; totals["output"] += usage["output"]
if s is None:
fail_count += 1
print(f"[{judge_model}] still unparseable after retry: {r['prompt_id']}/{r['system']}", flush=True)
else:
out_f.write(json.dumps({
"prompt_id": r["prompt_id"], "system": r["system"], "judge": judge_model, **s,
}) + "\n")
out_f.flush()
written += 1
else:
for r in failed_rows:
print(f"[{judge_model}] unparseable: {r['prompt_id']}/{r['system']}", flush=True)
return written, fail_count, totals, retried
def export_human_subset(responses: list[dict], bench: dict):
if HUMAN_SUBSET.exists():
print(f"{HUMAN_SUBSET} already exists, skipping export (not clobbering ratings)", flush=True)
return
random.seed(7)
subset = random.sample(responses, 30)
with HUMAN_SUBSET.open("w") as f, HUMAN_SUBSET_KEY.open("w") as kf:
f.write("row_id,prompt,response,empathy,nonharm,honesty,groundedness,overall\n")
kf.write("row_id,prompt_id,system\n")
for i, r in enumerate(subset):
p = bench[r["prompt_id"]]["prompt"].replace('"', "'")
resp = r["response"].replace('"', "'").replace("\n", " ")
f.write(f'{i},"{p}","{resp}",,,,,\n')
kf.write(f'{i},{r["prompt_id"]},{r["system"]}\n')
print(f"wrote {len(subset)} rows to {HUMAN_SUBSET} (blinded) and key to {HUMAN_SUBSET_KEY}", flush=True)
def main():
bench_items = load_bench(BENCH)
bench = {b["id"]: b for b in bench_items}
responses = [json.loads(l) for l in RESPONSES.read_text().splitlines() if l.strip()]
rubric = RUBRIC.read_text()
client = openrouter_client()
SCORES.parent.mkdir(parents=True, exist_ok=True)
done = load_scores_done(SCORES)
out_f = SCORES.open("a")
print("=== primary judge (all responses) ===", flush=True)
primary_todo = [r for r in responses if (r["prompt_id"], r["system"], JUDGE) not in done]
print(f"[{JUDGE}] {len(primary_todo)}/{len(responses)} remaining", flush=True)
p_written, p_fail, p_totals, p_retried = run_pass(client, primary_todo, JUDGE, rubric, bench, out_f)
print(
f"[{JUDGE}] written={p_written} fail={p_fail} retried={p_retried} tokens={p_totals}",
flush=True,
)
print("=== second judge (100-row seeded subset) ===", flush=True)
random.seed(11)
subset2 = random.sample(responses, 100)
second_todo = [r for r in subset2 if (r["prompt_id"], r["system"], SECOND_JUDGE) not in done]
print(f"[{SECOND_JUDGE}] {len(second_todo)}/{len(subset2)} remaining", flush=True)
s_written, s_fail, s_totals, s_retried = run_pass(client, second_todo, SECOND_JUDGE, rubric, bench, out_f)
print(
f"[{SECOND_JUDGE}] written={s_written} fail={s_fail} retried={s_retried} tokens={s_totals}",
flush=True,
)
out_f.close()
total_rows = sum(1 for _ in SCORES.open())
print(f"total score rows: {total_rows}", flush=True)
export_human_subset(responses, bench)
if __name__ == "__main__":
main()