feat: Gemini Flash judge harness + agreement subsets
Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
This commit is contained in:
194
scripts/judge.py
Normal file
194
scripts/judge.py
Normal file
@@ -0,0 +1,194 @@
|
||||
"""LLM-judge harness: score every data/responses.jsonl row with the primary
|
||||
judge (google/gemini-flash-latest), plus a second judge (deepseek/deepseek-v4-pro)
|
||||
on a 100-row seeded subset, for judge-judge agreement in Task 10. Also exports
|
||||
a blinded 30-row human-rating CSV plus a key file for the join.
|
||||
|
||||
Deviates from the task-8 brief's inline pseudocode on purpose (per task
|
||||
instructions): the brief's script overwrites data/scores.jsonl wholesale with
|
||||
Path.write_text() each run. Instead, like scripts/collect_responses.py, this
|
||||
appends and is resumable: it skips any (prompt_id, system, judge) triple
|
||||
already present in data/scores.jsonl, so a restart after an interruption does
|
||||
not re-spend API calls or money.
|
||||
|
||||
Token budget: max_tokens=800 per the brief. If more than 3% of rows in a pass
|
||||
come back empty/unparseable, the failed rows are automatically retried once at
|
||||
max_tokens=2000 (a note is printed either way, so the report can cite the
|
||||
observed rate).
|
||||
|
||||
Human subset (data/human_subset.csv) is blinded — no system column — so Marcus
|
||||
rates without knowing which model produced each response. data/human_subset_key.csv
|
||||
carries row_id -> (prompt_id, system) for Task 10's join and is written
|
||||
separately so it never ends up in the blinded file. Both use random.seed(7)
|
||||
exactly as specified in the brief, and are only generated once: if
|
||||
human_subset.csv already exists, export is skipped so we never clobber Marcus's
|
||||
in-progress ratings.
|
||||
"""
|
||||
import json
|
||||
import random
|
||||
from concurrent.futures import ThreadPoolExecutor
|
||||
from pathlib import Path
|
||||
|
||||
from buddhagpt.collect import load_bench
|
||||
from buddhagpt.judge import judge_messages, parse_score
|
||||
from buddhagpt.llm import chat, openrouter_client
|
||||
|
||||
BENCH = Path("eval/compassionbench.yaml")
|
||||
RESPONSES = Path("data/responses.jsonl")
|
||||
RUBRIC = Path("eval/rubric.md")
|
||||
SCORES = Path("data/scores.jsonl")
|
||||
HUMAN_SUBSET = Path("data/human_subset.csv")
|
||||
HUMAN_SUBSET_KEY = Path("data/human_subset_key.csv")
|
||||
|
||||
# OpenRouter serves the "latest" Gemini alias only under a tilde-prefixed
|
||||
# canonical slug (confirmed via GET /models); "google/gemini-flash-latest"
|
||||
# without the tilde 400s as an invalid model ID.
|
||||
JUDGE = "~google/gemini-flash-latest"
|
||||
SECOND_JUDGE = "deepseek/deepseek-v4-pro" # agreement check on a 100-row subset
|
||||
MAX_WORKERS = 8
|
||||
DEFAULT_MAX_TOKENS = 800
|
||||
RETRY_MAX_TOKENS = 2000
|
||||
PARSE_FAIL_THRESHOLD = 0.03
|
||||
|
||||
|
||||
def load_scores_done(scores_path: Path) -> set[tuple[str, str, str]]:
|
||||
"""(prompt_id, system, judge) triples already present in an existing
|
||||
scores file — mirrors buddhagpt.collect.load_done's role for responses."""
|
||||
done = set()
|
||||
if not scores_path.exists():
|
||||
return done
|
||||
for line in scores_path.read_text().splitlines():
|
||||
if not line.strip():
|
||||
continue
|
||||
row = json.loads(line)
|
||||
done.add((row["prompt_id"], row["system"], row["judge"]))
|
||||
return done
|
||||
|
||||
|
||||
def run_pass(client, items, judge_model, rubric, bench, out_f):
|
||||
"""Score `items` with `judge_model`, writing each parsed result to out_f
|
||||
as it completes. Returns (written, fail_count, totals, retried_count)."""
|
||||
if not items:
|
||||
return 0, 0, {"input": 0, "output": 0}, 0
|
||||
|
||||
totals = {"input": 0, "output": 0}
|
||||
|
||||
def score_one(r, max_tokens):
|
||||
prompt = bench[r["prompt_id"]]["prompt"]
|
||||
try:
|
||||
text, usage = chat(
|
||||
client, judge_model,
|
||||
judge_messages(prompt, r["response"], rubric),
|
||||
max_tokens=max_tokens,
|
||||
)
|
||||
except Exception as e:
|
||||
return r, None, {"input": 0, "output": 0}, str(e)
|
||||
return r, parse_score(text), usage, None
|
||||
|
||||
def run_round(rows, max_tokens):
|
||||
with ThreadPoolExecutor(max_workers=MAX_WORKERS) as pool:
|
||||
futs = [pool.submit(score_one, r, max_tokens) for r in rows]
|
||||
return [f.result() for f in futs]
|
||||
|
||||
written = 0
|
||||
results = run_round(items, DEFAULT_MAX_TOKENS)
|
||||
failed_rows = []
|
||||
for r, s, usage, err in results:
|
||||
totals["input"] += usage["input"]; totals["output"] += usage["output"]
|
||||
if s is None:
|
||||
failed_rows.append(r)
|
||||
if err:
|
||||
print(f"[{judge_model}] error {r['prompt_id']}/{r['system']}: {err}", flush=True)
|
||||
else:
|
||||
out_f.write(json.dumps({
|
||||
"prompt_id": r["prompt_id"], "system": r["system"], "judge": judge_model, **s,
|
||||
}) + "\n")
|
||||
out_f.flush()
|
||||
written += 1
|
||||
|
||||
fail_rate = len(failed_rows) / len(items)
|
||||
retried = 0
|
||||
fail_count = len(failed_rows)
|
||||
if fail_rate > PARSE_FAIL_THRESHOLD and failed_rows:
|
||||
print(
|
||||
f"[{judge_model}] parse-fail rate {fail_rate:.1%} on first pass (>3%), "
|
||||
f"retrying {len(failed_rows)} rows at max_tokens={RETRY_MAX_TOKENS}",
|
||||
flush=True,
|
||||
)
|
||||
retried = len(failed_rows)
|
||||
retry_results = run_round(failed_rows, RETRY_MAX_TOKENS)
|
||||
fail_count = 0
|
||||
for r, s, usage, err in retry_results:
|
||||
totals["input"] += usage["input"]; totals["output"] += usage["output"]
|
||||
if s is None:
|
||||
fail_count += 1
|
||||
print(f"[{judge_model}] still unparseable after retry: {r['prompt_id']}/{r['system']}", flush=True)
|
||||
else:
|
||||
out_f.write(json.dumps({
|
||||
"prompt_id": r["prompt_id"], "system": r["system"], "judge": judge_model, **s,
|
||||
}) + "\n")
|
||||
out_f.flush()
|
||||
written += 1
|
||||
else:
|
||||
for r in failed_rows:
|
||||
print(f"[{judge_model}] unparseable: {r['prompt_id']}/{r['system']}", flush=True)
|
||||
|
||||
return written, fail_count, totals, retried
|
||||
|
||||
|
||||
def export_human_subset(responses: list[dict], bench: dict):
|
||||
if HUMAN_SUBSET.exists():
|
||||
print(f"{HUMAN_SUBSET} already exists, skipping export (not clobbering ratings)", flush=True)
|
||||
return
|
||||
random.seed(7)
|
||||
subset = random.sample(responses, 30)
|
||||
with HUMAN_SUBSET.open("w") as f, HUMAN_SUBSET_KEY.open("w") as kf:
|
||||
f.write("row_id,prompt,response,empathy,nonharm,honesty,groundedness,overall\n")
|
||||
kf.write("row_id,prompt_id,system\n")
|
||||
for i, r in enumerate(subset):
|
||||
p = bench[r["prompt_id"]]["prompt"].replace('"', "'")
|
||||
resp = r["response"].replace('"', "'").replace("\n", " ")
|
||||
f.write(f'{i},"{p}","{resp}",,,,,\n')
|
||||
kf.write(f'{i},{r["prompt_id"]},{r["system"]}\n')
|
||||
print(f"wrote {len(subset)} rows to {HUMAN_SUBSET} (blinded) and key to {HUMAN_SUBSET_KEY}", flush=True)
|
||||
|
||||
|
||||
def main():
|
||||
bench_items = load_bench(BENCH)
|
||||
bench = {b["id"]: b for b in bench_items}
|
||||
responses = [json.loads(l) for l in RESPONSES.read_text().splitlines() if l.strip()]
|
||||
rubric = RUBRIC.read_text()
|
||||
client = openrouter_client()
|
||||
|
||||
SCORES.parent.mkdir(parents=True, exist_ok=True)
|
||||
done = load_scores_done(SCORES)
|
||||
out_f = SCORES.open("a")
|
||||
|
||||
print("=== primary judge (all responses) ===", flush=True)
|
||||
primary_todo = [r for r in responses if (r["prompt_id"], r["system"], JUDGE) not in done]
|
||||
print(f"[{JUDGE}] {len(primary_todo)}/{len(responses)} remaining", flush=True)
|
||||
p_written, p_fail, p_totals, p_retried = run_pass(client, primary_todo, JUDGE, rubric, bench, out_f)
|
||||
print(
|
||||
f"[{JUDGE}] written={p_written} fail={p_fail} retried={p_retried} tokens={p_totals}",
|
||||
flush=True,
|
||||
)
|
||||
|
||||
print("=== second judge (100-row seeded subset) ===", flush=True)
|
||||
random.seed(11)
|
||||
subset2 = random.sample(responses, 100)
|
||||
second_todo = [r for r in subset2 if (r["prompt_id"], r["system"], SECOND_JUDGE) not in done]
|
||||
print(f"[{SECOND_JUDGE}] {len(second_todo)}/{len(subset2)} remaining", flush=True)
|
||||
s_written, s_fail, s_totals, s_retried = run_pass(client, second_todo, SECOND_JUDGE, rubric, bench, out_f)
|
||||
print(
|
||||
f"[{SECOND_JUDGE}] written={s_written} fail={s_fail} retried={s_retried} tokens={s_totals}",
|
||||
flush=True,
|
||||
)
|
||||
|
||||
out_f.close()
|
||||
total_rows = sum(1 for _ in SCORES.open())
|
||||
print(f"total score rows: {total_rows}", flush=True)
|
||||
|
||||
export_human_subset(responses, bench)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
Reference in New Issue
Block a user