aboutsummaryrefslogtreecommitdiff
path: root/internal/liturgy/fetch.go
blob: 1b1458ff82f9be7132f3c4f9b5b024bf51184dba (plain) (blame)
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
package liturgy

import (
	"encoding/json"
	"fmt"
	"io"
	"net/http"
	"os"
	"path/filepath"
	"regexp"
	"time"
)

// baseURL is the niedziela.pl URL template ("%s" is the date, YYYY-MM-DD).
// It is a package var so tests can point it at an httptest server.
var baseURL = "https://niezbednik.niedziela.pl/liturgia/%s/Ewangelia"

// userAgent is sent on every fetch; the site serves a different (broken)
// page to non-browser clients without it.
const userAgent = "Mozilla/5.0 (X11; Linux x86_64) AppleWebKit/537.36 " +
	"(KHTML, like Gecko) Chrome/124.0 Safari/537.36"

// publishedRe matches the tab-pane id niedziela.pl gives a fully published
// day's page (e.g. "tabnowy0all"). Its absence -- together with a "Przykro
// nam" placeholder -- marks a date that has not been published yet; see
// ewangelia.py's fetch() for the original behaviour this mirrors.
var publishedRe = regexp.MustCompile(`id="\w*0all"`)

// Options controls how Load resolves a day's readings.
type Options struct {
	// Date is the day to load, formatted YYYY-MM-DD.
	Date string
	// Refresh bypasses both cache layers and re-fetches from the network.
	Refresh bool
	// Offline restricts Load to previously cached/harvested data, never
	// hitting the network. Its behavior is added in a later task; for now
	// it is a no-op on the modern-source path.
	Offline bool
}

// cacheDir is where the HTML/JSON cache layers live:
// ${XDG_CACHE_HOME:-~/.cache}/lectio/
func cacheDir() string {
	base := os.Getenv("XDG_CACHE_HOME")
	if base == "" {
		home, err := os.UserHomeDir()
		if err != nil {
			home = "."
		}
		base = filepath.Join(home, ".cache")
	}
	return filepath.Join(base, "lectio")
}

// Load returns the day's reading sections, from the modern (niedziela.pl)
// source, preferring cache over network:
//
//  1. Unless Refresh, the parsed JSON cache ({date}.json).
//  2. Unless Refresh, the raw HTML cache ({date}.html) -- parsed, and the
//     result written to the JSON cache for next time.
//  3. Otherwise, fetch the page over the network, parse it, and -- only if
//     the page is fully published -- write both cache layers.
func Load(opts Options) ([]Section, error) {
	dir := cacheDir()
	jsonPath := filepath.Join(dir, opts.Date+".json")
	htmlPath := filepath.Join(dir, opts.Date+".html")

	if !opts.Refresh {
		if secs, err := loadJSONCache(jsonPath); err == nil {
			return secs, nil
		}
		if page, err := os.ReadFile(htmlPath); err == nil {
			secs, err := Parse(string(page))
			if err != nil {
				return nil, err
			}
			writeJSONCache(jsonPath, secs)
			return secs, nil
		}
	}

	page, err := fetch(opts.Date)
	if err != nil {
		return nil, err
	}

	// Parse itself reports an unpublished date as "no reading published for
	// this date yet" (via the "Przykro nam" placeholder), so an unpublished
	// page is neither cached nor returned as sections here.
	secs, err := Parse(page)
	if err != nil {
		return nil, err
	}

	// Cache only fully-published pages, so an as-yet-unpublished future date
	// keeps being retried instead of caching a "no reading" placeholder.
	if publishedRe.MatchString(page) {
		if err := os.MkdirAll(dir, 0o755); err == nil {
			_ = os.WriteFile(htmlPath, []byte(page), 0o644)
			writeJSONCache(jsonPath, secs)
		}
	}

	return secs, nil
}

// loadJSONCache reads and unmarshals the parsed-sections cache file.
func loadJSONCache(path string) ([]Section, error) {
	data, err := os.ReadFile(path)
	if err != nil {
		return nil, err
	}
	var secs []Section
	if err := json.Unmarshal(data, &secs); err != nil {
		return nil, err
	}
	return secs, nil
}

// writeJSONCache best-effort writes the parsed sections cache; a failure to
// cache should never fail the load itself.
func writeJSONCache(path string, secs []Section) {
	data, err := json.Marshal(secs)
	if err != nil {
		return
	}
	_ = os.WriteFile(path, data, 0o644)
}

// fetch GETs the day's page from baseURL with the browser User-Agent.
// Whether the page is actually published is left to the caller (Parse
// detects an unpublished date; Load re-checks the tab id to decide whether
// the result is cache-worthy).
func fetch(dateStr string) (string, error) {
	url := fmt.Sprintf(baseURL, dateStr)
	req, err := http.NewRequest(http.MethodGet, url, nil)
	if err != nil {
		return "", err
	}
	req.Header.Set("User-Agent", userAgent)

	client := &http.Client{Timeout: 20 * time.Second}
	resp, err := client.Do(req)
	if err != nil {
		return "", fmt.Errorf("failed to fetch %s: %w", url, err)
	}
	defer resp.Body.Close()

	body, err := io.ReadAll(resp.Body)
	if err != nil {
		return "", fmt.Errorf("failed to read response from %s: %w", url, err)
	}
	return string(body), nil
}