"""LLM-judge harness: score every data/responses.jsonl row with the primary judge (google/gemini-flash-latest), plus a second judge (deepseek/deepseek-v4-pro) on a 100-row seeded subset, for judge-judge agreement in Task 10. Also exports a blinded 30-row human-rating CSV plus a key file for the join. Deviates from the task-8 brief's inline pseudocode on purpose (per task instructions): the brief's script overwrites data/scores.jsonl wholesale with Path.write_text() each run. Instead, like scripts/collect_responses.py, this appends and is resumable: it skips any (prompt_id, system, judge) triple already present in data/scores.jsonl, so a restart after an interruption does not re-spend API calls or money. Token budget: max_tokens=800 per the brief. If more than 3% of rows in a pass come back empty/unparseable, the failed rows are automatically retried once at max_tokens=2000 (a note is printed either way, so the report can cite the observed rate). Human subset (data/human_subset.csv) is blinded — no system column — so Marcus rates without knowing which model produced each response. data/human_subset_key.csv carries row_id -> (prompt_id, system) for Task 10's join and is written separately so it never ends up in the blinded file. Both use random.seed(7) exactly as specified in the brief, and are only generated once: if human_subset.csv already exists, export is skipped so we never clobber Marcus's in-progress ratings. """ import json import random from concurrent.futures import ThreadPoolExecutor from pathlib import Path from buddhagpt.collect import load_bench from buddhagpt.judge import judge_messages, parse_score from buddhagpt.llm import chat, openrouter_client BENCH = Path("eval/compassionbench.yaml") RESPONSES = Path("data/responses.jsonl") RUBRIC = Path("eval/rubric.md") SCORES = Path("data/scores.jsonl") HUMAN_SUBSET = Path("data/human_subset.csv") HUMAN_SUBSET_KEY = Path("data/human_subset_key.csv") # OpenRouter serves the "latest" Gemini alias only under a tilde-prefixed # canonical slug (confirmed via GET /models); "google/gemini-flash-latest" # without the tilde 400s as an invalid model ID. JUDGE = "~google/gemini-flash-latest" SECOND_JUDGE = "deepseek/deepseek-v4-pro" # agreement check on a 100-row subset MAX_WORKERS = 8 DEFAULT_MAX_TOKENS = 800 RETRY_MAX_TOKENS = 2000 PARSE_FAIL_THRESHOLD = 0.03 def load_scores_done(scores_path: Path) -> set[tuple[str, str, str]]: """(prompt_id, system, judge) triples already present in an existing scores file — mirrors buddhagpt.collect.load_done's role for responses.""" done = set() if not scores_path.exists(): return done for line in scores_path.read_text().splitlines(): if not line.strip(): continue row = json.loads(line) done.add((row["prompt_id"], row["system"], row["judge"])) return done def run_pass(client, items, judge_model, rubric, bench, out_f): """Score `items` with `judge_model`, writing each parsed result to out_f as it completes. Returns (written, fail_count, totals, retried_count).""" if not items: return 0, 0, {"input": 0, "output": 0}, 0 totals = {"input": 0, "output": 0} def score_one(r, max_tokens): prompt = bench[r["prompt_id"]]["prompt"] try: text, usage = chat( client, judge_model, judge_messages(prompt, r["response"], rubric), max_tokens=max_tokens, ) except Exception as e: return r, None, {"input": 0, "output": 0}, str(e) return r, parse_score(text), usage, None def run_round(rows, max_tokens): with ThreadPoolExecutor(max_workers=MAX_WORKERS) as pool: futs = [pool.submit(score_one, r, max_tokens) for r in rows] return [f.result() for f in futs] written = 0 results = run_round(items, DEFAULT_MAX_TOKENS) failed_rows = [] for r, s, usage, err in results: totals["input"] += usage["input"]; totals["output"] += usage["output"] if s is None: failed_rows.append(r) if err: print(f"[{judge_model}] error {r['prompt_id']}/{r['system']}: {err}", flush=True) else: out_f.write(json.dumps({ "prompt_id": r["prompt_id"], "system": r["system"], "judge": judge_model, **s, }) + "\n") out_f.flush() written += 1 fail_rate = len(failed_rows) / len(items) retried = 0 fail_count = len(failed_rows) if fail_rate > PARSE_FAIL_THRESHOLD and failed_rows: print( f"[{judge_model}] parse-fail rate {fail_rate:.1%} on first pass (>3%), " f"retrying {len(failed_rows)} rows at max_tokens={RETRY_MAX_TOKENS}", flush=True, ) retried = len(failed_rows) retry_results = run_round(failed_rows, RETRY_MAX_TOKENS) fail_count = 0 for r, s, usage, err in retry_results: totals["input"] += usage["input"]; totals["output"] += usage["output"] if s is None: fail_count += 1 print(f"[{judge_model}] still unparseable after retry: {r['prompt_id']}/{r['system']}", flush=True) else: out_f.write(json.dumps({ "prompt_id": r["prompt_id"], "system": r["system"], "judge": judge_model, **s, }) + "\n") out_f.flush() written += 1 else: for r in failed_rows: print(f"[{judge_model}] unparseable: {r['prompt_id']}/{r['system']}", flush=True) return written, fail_count, totals, retried def export_human_subset(responses: list[dict], bench: dict): if HUMAN_SUBSET.exists(): print(f"{HUMAN_SUBSET} already exists, skipping export (not clobbering ratings)", flush=True) return random.seed(7) subset = random.sample(responses, 30) with HUMAN_SUBSET.open("w") as f, HUMAN_SUBSET_KEY.open("w") as kf: f.write("row_id,prompt,response,empathy,nonharm,honesty,groundedness,overall\n") kf.write("row_id,prompt_id,system\n") for i, r in enumerate(subset): p = bench[r["prompt_id"]]["prompt"].replace('"', "'") resp = r["response"].replace('"', "'").replace("\n", " ") f.write(f'{i},"{p}","{resp}",,,,,\n') kf.write(f'{i},{r["prompt_id"]},{r["system"]}\n') print(f"wrote {len(subset)} rows to {HUMAN_SUBSET} (blinded) and key to {HUMAN_SUBSET_KEY}", flush=True) def main(): bench_items = load_bench(BENCH) bench = {b["id"]: b for b in bench_items} responses = [json.loads(l) for l in RESPONSES.read_text().splitlines() if l.strip()] rubric = RUBRIC.read_text() client = openrouter_client() SCORES.parent.mkdir(parents=True, exist_ok=True) done = load_scores_done(SCORES) out_f = SCORES.open("a") print("=== primary judge (all responses) ===", flush=True) primary_todo = [r for r in responses if (r["prompt_id"], r["system"], JUDGE) not in done] print(f"[{JUDGE}] {len(primary_todo)}/{len(responses)} remaining", flush=True) p_written, p_fail, p_totals, p_retried = run_pass(client, primary_todo, JUDGE, rubric, bench, out_f) print( f"[{JUDGE}] written={p_written} fail={p_fail} retried={p_retried} tokens={p_totals}", flush=True, ) print("=== second judge (100-row seeded subset) ===", flush=True) random.seed(11) subset2 = random.sample(responses, 100) second_todo = [r for r in subset2 if (r["prompt_id"], r["system"], SECOND_JUDGE) not in done] print(f"[{SECOND_JUDGE}] {len(second_todo)}/{len(subset2)} remaining", flush=True) s_written, s_fail, s_totals, s_retried = run_pass(client, second_todo, SECOND_JUDGE, rubric, bench, out_f) print( f"[{SECOND_JUDGE}] written={s_written} fail={s_fail} retried={s_retried} tokens={s_totals}", flush=True, ) out_f.close() total_rows = sum(1 for _ in SCORES.open()) print(f"total score rows: {total_rows}", flush=True) export_human_subset(responses, bench) if __name__ == "__main__": main()