aboutsummaryrefslogtreecommitdiff
path: root/internal/liturgy/fetch.go
diff options
context:
space:
mode:
authorLukasz Kasprzak <lukas@labunix.xyz>2026-07-28 12:52:13 +0200
committerLukasz Kasprzak <lukas@labunix.xyz>2026-07-28 12:52:13 +0200
commit44472bbe31475a3c672ae003d928d7caffb4b50a (patch)
treee74bae03c61b62479ef4f68b046e3345618933c8 /internal/liturgy/fetch.go
parent626f2275438a37e5cd2efb5964281b9bef5ce8cc (diff)
parent7b220084cf3951c8cde0582efdfcf628afc64336 (diff)
downloadlectio-44472bbe31475a3c672ae003d928d7caffb4b50a.tar.gz
lectio-44472bbe31475a3c672ae003d928d7caffb4b50a.zip
Merge branch 'of-offline-migration': offline daily view, remove scrapers
Diffstat (limited to 'internal/liturgy/fetch.go')
-rw-r--r--internal/liturgy/fetch.go259
1 files changed, 0 insertions, 259 deletions
diff --git a/internal/liturgy/fetch.go b/internal/liturgy/fetch.go
deleted file mode 100644
index e66b77f..0000000
--- a/internal/liturgy/fetch.go
+++ /dev/null
@@ -1,259 +0,0 @@
-package liturgy
-
-import (
- "encoding/json"
- "fmt"
- "io"
- "net/http"
- "os"
- "path/filepath"
- "regexp"
- "time"
-)
-
-// baseURL is the niedziela.pl URL template ("%s" is the date, YYYY-MM-DD).
-// It is a package var so tests can point it at an httptest server.
-var baseURL = "https://niezbednik.niedziela.pl/liturgia/%s/Ewangelia"
-
-// SetBaseURL overrides the fetch URL template used by Load. It exists so
-// tests in other packages (e.g. internal/readings) can point Load at an
-// httptest server; production code must never call it.
-func SetBaseURL(url string) {
- baseURL = url
-}
-
-// userAgent is sent on every fetch; the site serves a different (broken)
-// page to non-browser clients without it.
-const userAgent = "Mozilla/5.0 (X11; Linux x86_64) AppleWebKit/537.36 " +
- "(KHTML, like Gecko) Chrome/124.0 Safari/537.36"
-
-// publishedRe matches the tab-pane id niedziela.pl gives a fully published
-// day's page (e.g. "tabnowy0all"). Its absence -- together with a "Przykro
-// nam" placeholder -- marks a date that has not been published yet; see
-// ewangelia.py's fetch() for the original behaviour this mirrors.
-var publishedRe = regexp.MustCompile(`id="\w*0all"`)
-
-// dateRe is the same YYYY-MM-DD shape internal/cli's dateRe validates
-// against. Load checks opts.Date against it before building any filesystem
-// path (jsonPath/htmlPath below are built by string concatenation, so an
-// unvalidated Date is a path-traversal vector) -- defense-in-depth so every
-// caller (web, cli, tui) is protected even if a future caller forgets to
-// validate its own input first.
-var dateRe = regexp.MustCompile(`^\d{4}-\d{2}-\d{2}$`)
-
-// cacheFileRe matches the date-prefixed cache file names Load/Harvest/tradlit
-// write into CacheDir(): "<YYYY-MM-DD>.html", "<YYYY-MM-DD>.json", and
-// "<YYYY-MM-DD>.trad.<lang>.json" (the traditional-lectionary cache; see
-// internal/tradlit). CleanCache uses it to tell cache entries apart from
-// anything else that might be sitting in the directory, and to recover the
-// date (group 1) for the age check regardless of which cache file it is.
-var cacheFileRe = regexp.MustCompile(`^(\d{4}-\d{2}-\d{2})\.[a-z0-9.]+$`)
-
-// Options controls how Load resolves a day's readings.
-type Options struct {
- // Date is the day to load, formatted YYYY-MM-DD.
- Date string
- // Refresh bypasses both cache layers and re-fetches from the network.
- Refresh bool
- // Offline restricts Load to previously cached/harvested data (see
- // LoadOffline), never hitting the network.
- Offline bool
-}
-
-// CacheDir is where the HTML/JSON cache layers live:
-// ${XDG_CACHE_HOME:-~/.cache}/lectio/
-// Exported so internal/tradlit shares the same cache root for the
-// traditional lectionary's propers.
-func CacheDir() string {
- base := os.Getenv("XDG_CACHE_HOME")
- if base == "" {
- home, err := os.UserHomeDir()
- if err != nil {
- home = "."
- }
- base = filepath.Join(home, ".cache")
- }
- return filepath.Join(base, "lectio")
-}
-
-// Load returns the day's reading sections and DayInfo, from the modern
-// (niedziela.pl) source, preferring cache over network:
-//
-// 0. If Offline, skip straight to LoadOffline (the harvested sigla TSV) --
-// that store carries citations only, so DayInfo comes back zero.
-// 1. Unless Refresh, the parsed JSON cache ({date}.json), which round-trips
-// DayInfo alongside the sections (see cachedDay/loadCache).
-// 2. Unless Refresh, the raw HTML cache ({date}.html) -- parsed (both
-// sections and DayInfo re-derived from the same HTML via Parse/
-// ParseDayInfo), and the result written to the JSON cache for next time.
-// 3. Otherwise, fetch the page over the network, parse it, and -- only if
-// the page is fully published -- write both cache layers. A fetch error
-// here (e.g. no network) falls back to LoadOffline for this date if it
-// has been harvested, and only surfaces the original fetch error if
-// that fallback also fails.
-func Load(opts Options) ([]Section, DayInfo, error) {
- if !dateRe.MatchString(opts.Date) {
- return nil, DayInfo{}, fmt.Errorf("invalid date %q: want YYYY-MM-DD", opts.Date)
- }
-
- if opts.Offline {
- secs, err := LoadOffline(opts.Date)
- return secs, DayInfo{}, err
- }
-
- dir := CacheDir()
- jsonPath := filepath.Join(dir, opts.Date+".json")
- htmlPath := filepath.Join(dir, opts.Date+".html")
-
- if !opts.Refresh {
- if secs, info, err := loadCache(jsonPath); err == nil {
- return secs, info, nil
- }
- if page, err := os.ReadFile(htmlPath); err == nil {
- pageStr := string(page)
- secs, err := Parse(pageStr)
- if err != nil {
- return nil, DayInfo{}, err
- }
- info := ParseDayInfo(pageStr)
- writeCache(jsonPath, secs, info)
- return secs, info, nil
- }
- }
-
- page, err := fetch(opts.Date)
- if err != nil {
- // Network is unreachable: fall back to a prior harvest of this date
- // if there is one, rather than failing outright. The sigla store
- // carries no DayInfo, so this path always reports it zero.
- if secs, offErr := LoadOffline(opts.Date); offErr == nil {
- return secs, DayInfo{}, nil
- }
- return nil, DayInfo{}, err
- }
-
- // Parse itself reports an unpublished date as "no reading published for
- // this date yet" (via the "Przykro nam" placeholder), so an unpublished
- // page is neither cached nor returned as sections here.
- secs, err := Parse(page)
- if err != nil {
- return nil, DayInfo{}, err
- }
- info := ParseDayInfo(page)
-
- // Cache only fully-published pages, so an as-yet-unpublished future date
- // keeps being retried instead of caching a "no reading" placeholder.
- if publishedRe.MatchString(page) {
- if err := os.MkdirAll(dir, 0o755); err == nil {
- _ = os.WriteFile(htmlPath, []byte(page), 0o644)
- writeCache(jsonPath, secs, info)
- }
- }
-
- return secs, info, nil
-}
-
-// CleanCache removes cached readings whose date is before `before` from
-// CacheDir(). It matches only date-prefixed cache files (see cacheFileRe:
-// "<YYYY-MM-DD>.html", ".json", or the traditional lectionary's
-// ".trad.<lang>.json"); anything else in the directory (e.g. a stray
-// notes.txt, or the sigla store, which lives elsewhere entirely) is left
-// alone. A missing cache dir is not an error -- it just means there is
-// nothing to clean yet.
-func CleanCache(before time.Time) (removed int, freed int64, err error) {
- dir := CacheDir()
- entries, err := os.ReadDir(dir)
- if err != nil {
- if os.IsNotExist(err) {
- return 0, 0, nil
- }
- return 0, 0, err
- }
-
- for _, entry := range entries {
- if entry.IsDir() {
- continue
- }
- m := cacheFileRe.FindStringSubmatch(entry.Name())
- if m == nil {
- continue
- }
- date, perr := time.Parse("2006-01-02", m[1])
- if perr != nil || !date.Before(before) {
- continue
- }
-
- path := filepath.Join(dir, entry.Name())
- info, serr := os.Stat(path)
- if serr != nil {
- return removed, freed, serr
- }
- if rerr := os.Remove(path); rerr != nil {
- return removed, freed, rerr
- }
- removed++
- freed += info.Size()
- }
- return removed, freed, nil
-}
-
-// cachedDay is the on-disk shape of the parsed-sections JSON cache
-// ({date}.json): the sections plus the day's liturgical identity, so a
-// cache hit round-trips DayInfo without re-parsing the HTML cache. (An
-// older cache file written before DayInfo existed was a bare JSON array,
-// not an object -- json.Unmarshal into cachedDay then fails, which
-// loadCache treats as an ordinary cache miss, falling through to the HTML
-// cache or the network like any other stale/missing cache entry.)
-type cachedDay struct {
- Sections []Section
- DayInfo DayInfo
-}
-
-// loadCache reads and unmarshals the parsed-sections + DayInfo cache file.
-func loadCache(path string) ([]Section, DayInfo, error) {
- data, err := os.ReadFile(path)
- if err != nil {
- return nil, DayInfo{}, err
- }
- var cd cachedDay
- if err := json.Unmarshal(data, &cd); err != nil {
- return nil, DayInfo{}, err
- }
- return cd.Sections, cd.DayInfo, nil
-}
-
-// writeCache best-effort writes the parsed sections + DayInfo cache; a
-// failure to cache should never fail the load itself.
-func writeCache(path string, secs []Section, info DayInfo) {
- data, err := json.Marshal(cachedDay{Sections: secs, DayInfo: info})
- if err != nil {
- return
- }
- _ = os.WriteFile(path, data, 0o644)
-}
-
-// fetch GETs the day's page from baseURL with the browser User-Agent.
-// Whether the page is actually published is left to the caller (Parse
-// detects an unpublished date; Load re-checks the tab id to decide whether
-// the result is cache-worthy).
-func fetch(dateStr string) (string, error) {
- url := fmt.Sprintf(baseURL, dateStr)
- req, err := http.NewRequest(http.MethodGet, url, nil)
- if err != nil {
- return "", err
- }
- req.Header.Set("User-Agent", userAgent)
-
- client := &http.Client{Timeout: 20 * time.Second}
- resp, err := client.Do(req)
- if err != nil {
- return "", fmt.Errorf("failed to fetch %s: %w", url, err)
- }
- defer resp.Body.Close()
-
- body, err := io.ReadAll(resp.Body)
- if err != nil {
- return "", fmt.Errorf("failed to read response from %s: %w", url, err)
- }
- return string(body), nil
-}