feat: bilara corpus ingestion to suttas.jsonl
This commit is contained in:
17
tests/test_corpus.py
Normal file
17
tests/test_corpus.py
Normal file
@@ -0,0 +1,17 @@
|
||||
import json
|
||||
from pathlib import Path
|
||||
from buddhagpt.corpus import segments_to_text, load_bilara_suttas
|
||||
|
||||
def test_segments_to_text_joins_in_key_order():
|
||||
segs = {"mn21:1.2": "second.", "mn21:1.1": "First."}
|
||||
assert segments_to_text(segs) == "First. second."
|
||||
|
||||
def test_load_bilara_suttas(tmp_path):
|
||||
d = tmp_path / "translation/en/sujato/sutta/mn"
|
||||
d.mkdir(parents=True)
|
||||
(d / "mn21_translation-en-sujato.json").write_text(
|
||||
json.dumps({"mn21:0.1": "Middle Discourses 21", "mn21:1.1": "So I have heard."})
|
||||
)
|
||||
suttas = list(load_bilara_suttas(tmp_path))
|
||||
assert suttas[0]["uid"] == "mn21"
|
||||
assert "So I have heard." in suttas[0]["text"]
|
||||
Reference in New Issue
Block a user