fix: complete template coverage in training data + disclose model alias

This commit is contained in:
marcuspaico
2026-08-15 15:50:27 -07:00
parent cc86a97d10
commit d55d9a648c
4 changed files with 160 additions and 68 deletions

View File

@@ -26,6 +26,14 @@ SYSTEM = (
'{"question": "Example question three?", "answer": "Example answer three, 120-250 words..."}'
)
def interleaved_variant(call_index: int) -> int:
"""Map a running call index to a template variant, cycling through all templates.
Used so any generation run (however many calls it makes) covers every template in
TEMPLATES rather than exhausting one variant per full pass over the corpus.
"""
return call_index % len(TEMPLATES)
def build_messages(sutta: dict, variant: int) -> list[dict]:
tmpl = TEMPLATES[variant % len(TEMPLATES)]
return [