From e92782639a6955ca1bd9be839e196b6ce890495b Mon Sep 17 00:00:00 2001 From: marcuspaico Date: Mon, 17 Aug 2026 20:54:34 -0700 Subject: [PATCH] fix: align collect_frontier token budget with driver (8k) kimi-k3 spends heavily on hidden reasoning tokens that draw from the same budget as the visible answer. 3000-token budget resulted in empty responses on hard prompts. Increased to 8000 to accommodate reasoning overhead while ensuring room for substantive visible answers. Co-Authored-By: Claude Fable 5 --- src/buddhagpt/collect.py | 10 +++++----- 1 file changed, 5 insertions(+), 5 deletions(-) diff --git a/src/buddhagpt/collect.py b/src/buddhagpt/collect.py index abc0085..1c0d8dd 100644 --- a/src/buddhagpt/collect.py +++ b/src/buddhagpt/collect.py @@ -64,11 +64,11 @@ def collect_frontier(items: list[dict], out: Path, model: str = "moonshotai/kimi for it in items: if (it["id"], system) in done: continue - # max_tokens=3000, not 1000: kimi-k3 emits hidden reasoning tokens that - # draw from the same budget as the visible answer even with - # reasoning.enabled=False; 1000 was observed to exhaust on harder - # prompts and return empty content (finish_reason="length"). - text, _ = chat(client, model, [{"role": "user", "content": it["prompt"]}], max_tokens=3000) + # max_tokens=8000: kimi-k3 spends heavily on hidden reasoning tokens; 8000 + # leaves room for the visible answer — 3000 produced empty responses on hard + # prompts (finish_reason="length"). Hidden reasoning tokens draw from the + # same budget as the visible answer even with reasoning.enabled=False. + text, _ = chat(client, model, [{"role": "user", "content": it["prompt"]}], max_tokens=8000) if not text or not text.strip(): continue # leave undone for a future rerun rather than write empty content f.write(json.dumps({"prompt_id": it["id"], "system": system, "response": text}) + "\n")