diff options
Diffstat (limited to 'scripts/gen-deutero.py')
| -rw-r--r-- | scripts/gen-deutero.py | 62 |
1 files changed, 62 insertions, 0 deletions
diff --git a/scripts/gen-deutero.py b/scripts/gen-deutero.py new file mode 100644 index 0000000..ace721e --- /dev/null +++ b/scripts/gen-deutero.py @@ -0,0 +1,62 @@ +#!/usr/bin/env python3 +# gen-deutero.py -- backfill the deuterocanonical Daniel and Esther additions +# that the base drb (Douay-Rheims) corpus omits: Daniel 13 (Susanna) & 14 (Bel +# and the Dragon), and the Greek additions to Esther (chapters 11-16). The base +# corpus used Hebrew-canon chapter counts (Daniel 1-12, Esther 1-10) even though +# the Douay-Rheims itself carries these chapters; the EF Lenten lectionary reads +# them (e.g. Susanna on Saturday of the 3rd week of Lent), so they are required. +# +# Source: get.bible v2 "douayrheims" (public domain). Latin (vul) already has +# them; the Polish Wujek source (biblia.info.pl) uses the truncated 12-chapter +# Daniel, so wuj cannot be filled from the existing pipeline (Latin fallback +# covers Polish). +# +# Usage (from repo root): +# python3 scripts/gen-deutero.py # print the rows (inspect) +# python3 scripts/gen-deutero.py --apply # append to drb.tsv if not present +# +# Row format matches the corpus: Book\tAbbrev\tBookNum\tChapter\tVerse\tText +import json, sys, urllib.request + +DRB = "internal/bible/corpora/drb.tsv" +# (canonical book name, abbrev, book-number, [chapters]) -- must match drb.tsv. +TARGETS = [ + ("Daniel", "Dan", 27, range(13, 15)), # 13 Susanna, 14 Bel & the Dragon + ("Esther", "Est", 17, range(11, 17)), # 11-16 Greek additions +] + +def get(url): + req = urllib.request.Request(url, headers={"User-Agent": "curl/8.0"}) + with urllib.request.urlopen(req, timeout=25) as r: + return json.load(r) + +def rows(): + out = [] + for name, abbr, nr, chapters in TARGETS: + for ch in chapters: + d = get(f"https://api.getbible.net/v2/douayrheims/{nr}/{ch}.json") + for v in d.get("verses", []): + text = " ".join(v["text"].split()) # collapse whitespace, no tabs/newlines + out.append(f"{name}\t{abbr}\t{nr}\t{ch}\t{v['verse']}\t{text}") + return out + +def main(): + apply = "--apply" in sys.argv[1:] + new = rows() + if not apply: + for r in new: + print(r) + print(f"# {len(new)} rows (not written; pass --apply to append)", file=sys.stderr) + return + existing = open(DRB, encoding="utf-8").read() + if "\nDaniel\tDan\t27\t13\t" in existing: + print("drb.tsv already has Daniel 13 -- refusing to duplicate", file=sys.stderr) + sys.exit(1) + with open(DRB, "a", encoding="utf-8") as f: + if not existing.endswith("\n"): + f.write("\n") + f.write("\n".join(new) + "\n") + print(f"appended {len(new)} rows to {DRB}", file=sys.stderr) + +if __name__ == "__main__": + main() |
