summaryrefslogtreecommitdiff
path: root/tools/check_citations.py
blob: 750f0a4b96249a4d442b3563ff793a76cf80c663 (plain) (blame)
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
#!/usr/bin/env python3
"""check_citations.py -- verify every "LT.txt:<n>" citation in lang/la.ini
actually resolves to the Latin text it claims, in docs/research/LT.txt.

Run via `make check-citations`. Exits 2 with a report if any citation is
wrong; exits 0 (silently, bar a summary line) if every citation checked out;
exits 0 with a loud "SKIPPED" line if docs/research/LT.txt is not present
(it is gitignored -- see below).

WHAT THIS CHECKS, PRECISELY (a heuristic, not a proof)
-------------------------------------------------------
lang/la.ini's own comments cite a Missal heading in one of two shapes:

  1. A LEADING comment block, then a group of entries it covers, e.g.
     "; Ash Wednesday and the three days after it -- LT.txt:8686-8689."
     followed by four `key = value` lines.
  2. A TRAILING comment immediately under the ONE entry it explains, e.g.
     "advent = Tempus Adventus" then "; LT.txt:8609." on the next line,
     with no blank line -- [season]'s own style.

For each individual cited line number (after expanding "A-B" ranges and
comma lists), this script builds a POOL of candidate Latin phrases: the
entry/entries the citation is attached to (the single preceding entry for
the trailing shape, the group of following entries for the leading shape),
PLUS every double-quoted Latin phrase appearing anywhere in that comment
block (comments routinely quote an ALTERNATIVE heading being discussed, not
only the chosen entry's own value -- see e.g. the [season] block's
time-after-epiphany caveat). Each pool item keeps its OWN distinctive-word
set (>=4 letters, not on the small stopword list below, j/i and
ae/oe/diacritics normalised) -- items are not flattened into one bag. A
citation PASSES if a window of LT.txt[n-2 .. n+2] (+-2 lines, since a
heading can wrap) contains ALL of at least one single pool item's words --
not merely ANY word from ANY item. That distinction matters: a flattened
any-word-overlap check let a citation bundling two claims onto one line
number ("D.NI NOSTRI JESU CHRISTI REGIS / I classis", cited at LT.txt:12459)
pass on the strength of the first half alone (found two lines away, at the
edge of tolerance) even though the second half ("I classis") was three
lines away and never actually checked -- one of the two real citation bugs
this script exists to catch. Requiring one item's FULL word-set closes
that gap.

This is deliberately a LOOSE, word-overlap check, not an exact-phrase
match: la.ini spells abbreviations out in full (Sanctissimi, not Ss.mi) and
normalises j->i, and requiring a byte-exact substring would either force
every citation's prose to repeat the raw OCR text verbatim (defeating the
point of writing readable comments) or produce false failures having
nothing to do with a wrong line number. The trade-off is disclosed, not
hidden: this catches a citation pointing at UNRELATED content (the two real
bugs this script exists because of: LT.txt:8631 cited for "Tempus
Adventus" is actually "Tempus Nativitatis"; LT.txt:12459 cited for
"D.NI NOSTRI JESU CHRISTI REGIS / I classis" is actually just "Dominica
ultima Octobris") -- it does not, and cannot, prove a citation is the BEST
possible line, only that it is not obviously wrong.

Only citations OUTSIDE a "PATTERN" block are checked: a PATTERN entry makes
no claim that its own line is a direct heading, so a "LT.txt:N" mentioned
in its comment (e.g. citing the GRAMMAR another day's heading attests, not
this day's own heading) is not a provenance claim for THIS entry and would
otherwise produce a meaningless failure.
"""
import re
import sys
import unicodedata
from pathlib import Path

ROOT = Path(__file__).resolve().parent.parent
LA_INI = ROOT / "lang" / "la.ini"
LT_TXT = ROOT / "docs" / "research" / "LT.txt"

STOPWORDS = {
    "in", "de", "et", "ad", "post", "ante", "cum", "per", "seu", "infra",
    "vel", "si", "haec", "hoc", "hic", "qui", "quae", "quod", "quia",
    "tempus", "dominica", "dominicam", "dominicae", "feria", "feriae",
    "sabbato", "sabbatum", "die", "diebus", "eodem", "anno", "eius",
    "sancti", "sancta", "sanctae", "sancto", "sanctorum", "sanctus",
    "domini", "dominus", "octava", "octavam", "octavas",
    "missae", "missa", "proprium", "gregorianus", "cantus", "pdf",
    "forma", "longior", "brevior", "vide", "etiam", "dom", "prosper",
    "sacro", "actio", "electronica", "formam", "novissimae", "variationes",
    "copyright", "archivum", "liturgicum", "missale", "romanum", "index",
    "www", "http", "https", "htm", "html", "com", "romanum", "text",
}


def normalize_word(w: str) -> str:
    w = w.lower()
    w = unicodedata.normalize("NFKD", w)
    w = "".join(c for c in w if not unicodedata.combining(c))
    w = w.replace("æ", "ae").replace("œ", "oe")
    w = re.sub(r"[^a-z]", "", w)
    w = w.replace("j", "i")
    return w


def distinctive_words(text: str) -> set:
    out = set()
    for tok in re.split(r"\s+", text):
        w = normalize_word(tok)
        if len(w) >= 4 and w not in STOPWORDS:
            out.add(w)
    return out


def expand_citation_spec(spec: str):
    """'8618,8620,8622' -> [8618,8620,8622]; '8691-8717' -> [8691..8717]."""
    nums = []
    for tok in spec.split(","):
        tok = tok.strip()
        m = re.fullmatch(r"(\d{2,6})-(\d{2,6})", tok)
        if m:
            a, b = int(m.group(1)), int(m.group(2))
            if a <= b and (b - a) <= 200:
                nums.extend(range(a, b + 1))
            continue
        m = re.fullmatch(r"(\d{2,6})", tok)
        if m:
            nums.append(int(m.group(1)))
    return nums


CITATION_RE = re.compile(r"LT\.txt:\s*((?:\d{2,6}(?:-\d{2,6})?)(?:\s*,\s*\d{2,6}(?:-\d{2,6})?)*)")
QUOTE_RE = re.compile(r'"([^"]{3,})"')


def parse_blocks(la_ini_text: str):
    """Split la.ini into blocks on blank lines and [section] headers. Each
    block is a list of (kind, content) where kind is 'comment' or 'entry',
    content is the stripped comment text or (key, value)."""
    blocks = []
    cur = []
    for raw in la_ini_text.split("\n"):
        line = raw.rstrip("\n")
        stripped = line.strip()
        if stripped == "" or stripped.startswith("["):
            if cur:
                blocks.append(cur)
            cur = []
            continue
        if stripped.startswith(";"):
            cur.append(("comment", stripped[1:].strip()))
        elif "=" in stripped:
            k, _, v = stripped.partition("=")
            cur.append(("entry", (k.strip(), v.strip())))
        # anything else (shouldn't occur) is ignored
    if cur:
        blocks.append(cur)
    return blocks


def check(la_ini_text: str, lt_lines: list):
    findings = []
    checked = 0
    for block in parse_blocks(la_ini_text):
        block_comment = "\n".join(c for k, c in block if k == "comment")
        if "PATTERN" in block_comment:
            continue
        entries = [c for k, c in block if k == "entry"]
        if not entries:
            continue
        quotes = QUOTE_RE.findall(block_comment)
        entries_before = []
        for k, c in block:
            if k == "comment":
                for m in CITATION_RE.finditer(c):
                    nums = expand_citation_spec(m.group(1))
                    if entries_before:
                        pool_entries = [entries_before[-1]]
                    else:
                        # leading citation: pool = every entry in the block
                        # (entries after this comment, i.e. all of them,
                        # since none has been seen yet)
                        pool_entries = entries
                    # Pool items are kept SEPARATE (not flattened into one
                    # bag of words): a citation passes only if the window
                    # fully covers -- ALL the distinctive words of -- at
                    # least one single pool item (one quoted phrase, or one
                    # entry's own value). A flattened "any word from any
                    # pool item" bag is too permissive: it let a citation
                    # bundling two claims onto one line number ("D.NI
                    # NOSTRI JESU CHRISTI REGIS / I classis") pass on the
                    # strength of the FIRST half alone, even though the
                    # second half ("I classis") was not actually nearby --
                    # exactly the shape of one of the two real citation
                    # bugs this script was written to catch. Verified by
                    # replay against the pre-fix file (see the task report).
                    pool_items = [distinctive_words(q) for q in quotes]
                    for _, v in pool_entries:
                        pool_items.append(distinctive_words(v))
                    pool_items = [p for p in pool_items if p]
                    for n in nums:
                        checked += 1
                        lo, hi = max(1, n - 2), min(len(lt_lines), n + 2)
                        window_text = " ".join(lt_lines[lo - 1 : hi])
                        window_words = distinctive_words(window_text)
                        if not any(item <= window_words for item in pool_items):
                            entry_desc = ", ".join(f"{k}={v}" for k, v in pool_entries)
                            findings.append(
                                {
                                    "line": n,
                                    "entries": entry_desc,
                                    "actual": lt_lines[n - 1].strip() if 1 <= n <= len(lt_lines) else "(out of range)",
                                    "window": window_text.strip()[:160],
                                }
                            )
            else:
                entries_before.append(c)
    return checked, findings


def main():
    if not LT_TXT.exists():
        print(
            "SKIPPED: docs/research/LT.txt is absent (docs/ is gitignored -- "
            "present locally only). Citations are NOT verified this run."
        )
        return 0
    la_ini_text = LA_INI.read_text(encoding="utf-8")
    lt_lines = LT_TXT.read_text(encoding="utf-8", errors="replace").split("\n")
    checked, findings = check(la_ini_text, lt_lines)
    if findings:
        print(f"check-citations: {len(findings)} of {checked} citations look wrong:\n")
        for f in findings:
            print(f"  LT.txt:{f['line']} cited for [{f['entries']}]")
            print(f"    actual line {f['line']}: {f['actual']!r}")
            print(f"    window (+-2): {f['window']!r}\n")
        return 2
    print(f"check-citations: {checked} LT.txt citations checked, 0 look wrong.")
    return 0


if __name__ == "__main__":
    sys.exit(main())