aboutsummaryrefslogtreecommitdiff
path: root/internal/liturgy/fetch.go
diff options
context:
space:
mode:
Diffstat (limited to 'internal/liturgy/fetch.go')
-rw-r--r--internal/liturgy/fetch.go154
1 files changed, 154 insertions, 0 deletions
diff --git a/internal/liturgy/fetch.go b/internal/liturgy/fetch.go
new file mode 100644
index 0000000..1b1458f
--- /dev/null
+++ b/internal/liturgy/fetch.go
@@ -0,0 +1,154 @@
+package liturgy
+
+import (
+ "encoding/json"
+ "fmt"
+ "io"
+ "net/http"
+ "os"
+ "path/filepath"
+ "regexp"
+ "time"
+)
+
+// baseURL is the niedziela.pl URL template ("%s" is the date, YYYY-MM-DD).
+// It is a package var so tests can point it at an httptest server.
+var baseURL = "https://niezbednik.niedziela.pl/liturgia/%s/Ewangelia"
+
+// userAgent is sent on every fetch; the site serves a different (broken)
+// page to non-browser clients without it.
+const userAgent = "Mozilla/5.0 (X11; Linux x86_64) AppleWebKit/537.36 " +
+ "(KHTML, like Gecko) Chrome/124.0 Safari/537.36"
+
+// publishedRe matches the tab-pane id niedziela.pl gives a fully published
+// day's page (e.g. "tabnowy0all"). Its absence -- together with a "Przykro
+// nam" placeholder -- marks a date that has not been published yet; see
+// ewangelia.py's fetch() for the original behaviour this mirrors.
+var publishedRe = regexp.MustCompile(`id="\w*0all"`)
+
+// Options controls how Load resolves a day's readings.
+type Options struct {
+ // Date is the day to load, formatted YYYY-MM-DD.
+ Date string
+ // Refresh bypasses both cache layers and re-fetches from the network.
+ Refresh bool
+ // Offline restricts Load to previously cached/harvested data, never
+ // hitting the network. Its behavior is added in a later task; for now
+ // it is a no-op on the modern-source path.
+ Offline bool
+}
+
+// cacheDir is where the HTML/JSON cache layers live:
+// ${XDG_CACHE_HOME:-~/.cache}/lectio/
+func cacheDir() string {
+ base := os.Getenv("XDG_CACHE_HOME")
+ if base == "" {
+ home, err := os.UserHomeDir()
+ if err != nil {
+ home = "."
+ }
+ base = filepath.Join(home, ".cache")
+ }
+ return filepath.Join(base, "lectio")
+}
+
+// Load returns the day's reading sections, from the modern (niedziela.pl)
+// source, preferring cache over network:
+//
+// 1. Unless Refresh, the parsed JSON cache ({date}.json).
+// 2. Unless Refresh, the raw HTML cache ({date}.html) -- parsed, and the
+// result written to the JSON cache for next time.
+// 3. Otherwise, fetch the page over the network, parse it, and -- only if
+// the page is fully published -- write both cache layers.
+func Load(opts Options) ([]Section, error) {
+ dir := cacheDir()
+ jsonPath := filepath.Join(dir, opts.Date+".json")
+ htmlPath := filepath.Join(dir, opts.Date+".html")
+
+ if !opts.Refresh {
+ if secs, err := loadJSONCache(jsonPath); err == nil {
+ return secs, nil
+ }
+ if page, err := os.ReadFile(htmlPath); err == nil {
+ secs, err := Parse(string(page))
+ if err != nil {
+ return nil, err
+ }
+ writeJSONCache(jsonPath, secs)
+ return secs, nil
+ }
+ }
+
+ page, err := fetch(opts.Date)
+ if err != nil {
+ return nil, err
+ }
+
+ // Parse itself reports an unpublished date as "no reading published for
+ // this date yet" (via the "Przykro nam" placeholder), so an unpublished
+ // page is neither cached nor returned as sections here.
+ secs, err := Parse(page)
+ if err != nil {
+ return nil, err
+ }
+
+ // Cache only fully-published pages, so an as-yet-unpublished future date
+ // keeps being retried instead of caching a "no reading" placeholder.
+ if publishedRe.MatchString(page) {
+ if err := os.MkdirAll(dir, 0o755); err == nil {
+ _ = os.WriteFile(htmlPath, []byte(page), 0o644)
+ writeJSONCache(jsonPath, secs)
+ }
+ }
+
+ return secs, nil
+}
+
+// loadJSONCache reads and unmarshals the parsed-sections cache file.
+func loadJSONCache(path string) ([]Section, error) {
+ data, err := os.ReadFile(path)
+ if err != nil {
+ return nil, err
+ }
+ var secs []Section
+ if err := json.Unmarshal(data, &secs); err != nil {
+ return nil, err
+ }
+ return secs, nil
+}
+
+// writeJSONCache best-effort writes the parsed sections cache; a failure to
+// cache should never fail the load itself.
+func writeJSONCache(path string, secs []Section) {
+ data, err := json.Marshal(secs)
+ if err != nil {
+ return
+ }
+ _ = os.WriteFile(path, data, 0o644)
+}
+
+// fetch GETs the day's page from baseURL with the browser User-Agent.
+// Whether the page is actually published is left to the caller (Parse
+// detects an unpublished date; Load re-checks the tab id to decide whether
+// the result is cache-worthy).
+func fetch(dateStr string) (string, error) {
+ url := fmt.Sprintf(baseURL, dateStr)
+ req, err := http.NewRequest(http.MethodGet, url, nil)
+ if err != nil {
+ return "", err
+ }
+ req.Header.Set("User-Agent", userAgent)
+
+ client := &http.Client{Timeout: 20 * time.Second}
+ resp, err := client.Do(req)
+ if err != nil {
+ return "", fmt.Errorf("failed to fetch %s: %w", url, err)
+ }
+ defer resp.Body.Close()
+
+ body, err := io.ReadAll(resp.Body)
+ if err != nil {
+ return "", fmt.Errorf("failed to read response from %s: %w", url, err)
+ }
+ return string(body), nil
+}