feat: bilara corpus ingestion to suttas.jsonl

This commit is contained in:
marcuspaico
2026-08-14 17:06:26 -07:00
parent 76d1858ffe
commit f456877b58
4 changed files with 52 additions and 0 deletions

31
src/buddhagpt/corpus.py Normal file
View File

@@ -0,0 +1,31 @@
import json, re
from pathlib import Path
from typing import Iterator
def _seg_sort_key(k: str):
# "mn21:1.2" -> numeric-aware ordering of the segment path
# Handle mixed types (integers and strings) by wrapping in tuples
parts = []
for p in re.split(r"[:.]", k):
if p.isdigit():
parts.append((0, int(p))) # 0 for int, sorts before strings
else:
parts.append((1, p)) # 1 for string
return parts
def segments_to_text(segments: dict[str, str]) -> str:
ordered = [segments[k] for k in sorted(segments, key=_seg_sort_key)]
return " ".join(s.strip() for s in ordered if s.strip())
def load_bilara_suttas(root: Path) -> Iterator[dict]:
base = root / "translation/en/sujato/sutta"
for f in sorted(base.rglob("*_translation-en-sujato.json")):
segments = json.loads(f.read_text())
uid = f.name.split("_")[0]
title = next(iter(segments.values()), uid)
yield {
"uid": uid,
"title": title.strip(),
"text": segments_to_text(segments),
"collection": f.parent.name,
}