aboutsummaryrefslogtreecommitdiff
path: root/scripts/gen-deutero.py
blob: ace721ea6ea461be2a1b27db93c04371e04a406e (plain) (blame)
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
#!/usr/bin/env python3
# gen-deutero.py -- backfill the deuterocanonical Daniel and Esther additions
# that the base drb (Douay-Rheims) corpus omits: Daniel 13 (Susanna) & 14 (Bel
# and the Dragon), and the Greek additions to Esther (chapters 11-16). The base
# corpus used Hebrew-canon chapter counts (Daniel 1-12, Esther 1-10) even though
# the Douay-Rheims itself carries these chapters; the EF Lenten lectionary reads
# them (e.g. Susanna on Saturday of the 3rd week of Lent), so they are required.
#
# Source: get.bible v2 "douayrheims" (public domain). Latin (vul) already has
# them; the Polish Wujek source (biblia.info.pl) uses the truncated 12-chapter
# Daniel, so wuj cannot be filled from the existing pipeline (Latin fallback
# covers Polish).
#
# Usage (from repo root):
#   python3 scripts/gen-deutero.py            # print the rows (inspect)
#   python3 scripts/gen-deutero.py --apply    # append to drb.tsv if not present
#
# Row format matches the corpus: Book\tAbbrev\tBookNum\tChapter\tVerse\tText
import json, sys, urllib.request

DRB = "internal/bible/corpora/drb.tsv"
# (canonical book name, abbrev, book-number, [chapters]) -- must match drb.tsv.
TARGETS = [
    ("Daniel", "Dan", 27, range(13, 15)),   # 13 Susanna, 14 Bel & the Dragon
    ("Esther", "Est", 17, range(11, 17)),   # 11-16 Greek additions
]

def get(url):
    req = urllib.request.Request(url, headers={"User-Agent": "curl/8.0"})
    with urllib.request.urlopen(req, timeout=25) as r:
        return json.load(r)

def rows():
    out = []
    for name, abbr, nr, chapters in TARGETS:
        for ch in chapters:
            d = get(f"https://api.getbible.net/v2/douayrheims/{nr}/{ch}.json")
            for v in d.get("verses", []):
                text = " ".join(v["text"].split())  # collapse whitespace, no tabs/newlines
                out.append(f"{name}\t{abbr}\t{nr}\t{ch}\t{v['verse']}\t{text}")
    return out

def main():
    apply = "--apply" in sys.argv[1:]
    new = rows()
    if not apply:
        for r in new:
            print(r)
        print(f"# {len(new)} rows (not written; pass --apply to append)", file=sys.stderr)
        return
    existing = open(DRB, encoding="utf-8").read()
    if "\nDaniel\tDan\t27\t13\t" in existing:
        print("drb.tsv already has Daniel 13 -- refusing to duplicate", file=sys.stderr)
        sys.exit(1)
    with open(DRB, "a", encoding="utf-8") as f:
        if not existing.endswith("\n"):
            f.write("\n")
        f.write("\n".join(new) + "\n")
    print(f"appended {len(new)} rows to {DRB}", file=sys.stderr)

if __name__ == "__main__":
    main()