aboutsummaryrefslogtreecommitdiff
path: root/internal/liturgy/parse.go
diff options
context:
space:
mode:
Diffstat (limited to 'internal/liturgy/parse.go')
-rw-r--r--internal/liturgy/parse.go156
1 files changed, 156 insertions, 0 deletions
diff --git a/internal/liturgy/parse.go b/internal/liturgy/parse.go
new file mode 100644
index 0000000..64e6928
--- /dev/null
+++ b/internal/liturgy/parse.go
@@ -0,0 +1,156 @@
+package liturgy
+
+import (
+ "fmt"
+ "html"
+ "regexp"
+ "strings"
+)
+
+// lectionaryTabs lists the two reading sets carried on the page: the
+// lectionary in force since Advent 2015 ("nowy") and the one it replaced
+// ("stary"). They are not interchangeable -- the acclamation can cite a
+// different book entirely, and the older set contains malformed citations.
+// Pick by id, never by position in the markup, or a reordering silently
+// switches edition.
+var lectionaryTabs = []string{"tabnowy0all", "tabstary0all"}
+
+var (
+ brRe = regexp.MustCompile(`(?i)<br\s*/?>`)
+ tagRe = regexp.MustCompile(`<[^>]+>`)
+ wsRe = regexp.MustCompile(`[ \t]+`)
+ h2Re = regexp.MustCompile(`(?s)<h2>(.*?)</h2>`)
+ h4Re = regexp.MustCompile(`(?s)<h4>(.*?)</h4>`)
+ pRe = regexp.MustCompile(`(?s)<p>(.*?)</p>`)
+ citationRe = regexp.MustCompile(`\((.+)\)\s*$`)
+)
+
+// panePattern matches the opening tag of the tab-pane div carrying the given
+// tab id, e.g. `<div class="tab-pane fade " id="tabnowy0all">`.
+func panePattern(tab string) *regexp.Regexp {
+ return regexp.MustCompile(`<div class="tab-pane[^"]*"\s+id="` + regexp.QuoteMeta(tab) + `">`)
+}
+
+// htmlToLines turns an HTML fragment into a list of non-empty text lines.
+// <br> marks a verse line break (used in psalms/acclamations); other tags
+// are dropped, entities decoded, and intra-line whitespace collapsed.
+func htmlToLines(fragment string) []string {
+ fragment = brRe.ReplaceAllString(fragment, "\n")
+ fragment = tagRe.ReplaceAllString(fragment, "")
+ text := html.UnescapeString(fragment)
+ var lines []string
+ for _, ln := range strings.Split(text, "\n") {
+ ln = strings.TrimSpace(wsRe.ReplaceAllString(ln, " "))
+ if ln != "" {
+ lines = append(lines, ln)
+ }
+ }
+ return lines
+}
+
+// Parse extracts every reading section from the page's preferred lectionary
+// tab (falling back to the superseded one), erroring loudly if neither tab is
+// present on the page, or the tab is found but carries no sections -- a
+// layout change should never be mistaken for a quiet day with no readings.
+func Parse(pageHTML string) ([]Section, error) {
+ var loc []int
+ for _, tab := range lectionaryTabs {
+ if m := panePattern(tab).FindStringIndex(pageHTML); m != nil {
+ loc = m
+ break
+ }
+ }
+ if loc == nil {
+ if strings.Contains(pageHTML, "Przykro nam") {
+ return nil, fmt.Errorf("no reading published for this date yet")
+ }
+ return nil, fmt.Errorf(
+ "no reading tab (%s) found on page -- the site layout may have changed",
+ strings.Join(lectionaryTabs, "/"),
+ )
+ }
+
+ rest := pageHTML[loc[1]:]
+ block := rest
+ if nxt := strings.Index(rest, `<div class="tab-pane`); nxt != -1 {
+ block = rest[:nxt]
+ }
+
+ heads := h2Re.FindAllStringSubmatchIndex(block, -1)
+ var sections []Section
+ czytanieCount := 0
+ for i, h := range heads {
+ bodyEnd := len(block)
+ if i+1 < len(heads) {
+ bodyEnd = heads[i+1][0]
+ }
+ body := block[h[1]:bodyEnd]
+
+ heading := strings.Join(htmlToLines(block[h[2]:h[3]]), " ")
+
+ subtitle := ""
+ if sub := h4Re.FindStringSubmatch(body); sub != nil {
+ subtitle = strings.Join(htmlToLines(sub[1]), " ")
+ }
+
+ var paragraphs [][]string
+ for _, p := range pRe.FindAllStringSubmatch(body, -1) {
+ if lines := htmlToLines(p[1]); len(lines) > 0 {
+ paragraphs = append(paragraphs, lines)
+ }
+ }
+
+ // A heading is expected to always carry a parenthetical citation;
+ // if one is somehow missing, leave Citation empty rather than
+ // failing the whole parse over one section.
+ citation, _ := ExtractCitation(heading)
+
+ sections = append(sections, Section{
+ Heading: heading,
+ Subtitle: subtitle,
+ Citation: citation,
+ PartID: partID(heading, &czytanieCount),
+ Paragraphs: paragraphs,
+ })
+ }
+
+ // A layout change can leave the tab findable but empty; say so rather
+ // than returning nothing and looking like a quiet day.
+ if len(sections) == 0 {
+ return nil, fmt.Errorf("reading tab found but no sections in it -- the site layout may have changed")
+ }
+ return sections, nil
+}
+
+// partID assigns the stable liturgical-part identifier for a section heading.
+// A second "1. czytanie" heading on the same day (a split feast offering two
+// alternative first readings) becomes "drugie_czytanie" instead of colliding
+// with the first ("pierwsze_czytanie"). Anything unrecognised is "".
+func partID(heading string, czytanieCount *int) string {
+ switch {
+ case strings.HasPrefix(heading, "1. czytanie"):
+ *czytanieCount++
+ if *czytanieCount == 1 {
+ return "pierwsze_czytanie"
+ }
+ return "drugie_czytanie"
+ case strings.HasPrefix(heading, "Psalm"):
+ return "psalm"
+ case strings.HasPrefix(heading, "Aklamacja"):
+ return "aklamacja"
+ case strings.HasPrefix(heading, "Ewangelia"):
+ return "ewangelia"
+ default:
+ return ""
+ }
+}
+
+// ExtractCitation pulls the citation from a section heading:
+// "Ewangelia (Mt 7, 1-5)" -> "Mt 7, 1-5".
+func ExtractCitation(heading string) (string, error) {
+ m := citationRe.FindStringSubmatch(heading)
+ if m == nil {
+ return "", fmt.Errorf("no reference found in heading: %q", heading)
+ }
+ return strings.TrimSpace(m[1]), nil
+}