package liturgy import ( "encoding/json" "fmt" "io" "net/http" "os" "path/filepath" "regexp" "time" ) // baseURL is the niedziela.pl URL template ("%s" is the date, YYYY-MM-DD). // It is a package var so tests can point it at an httptest server. var baseURL = "https://niezbednik.niedziela.pl/liturgia/%s/Ewangelia" // SetBaseURL overrides the fetch URL template used by Load. It exists so // tests in other packages (e.g. internal/readings) can point Load at an // httptest server; production code must never call it. func SetBaseURL(url string) { baseURL = url } // userAgent is sent on every fetch; the site serves a different (broken) // page to non-browser clients without it. const userAgent = "Mozilla/5.0 (X11; Linux x86_64) AppleWebKit/537.36 " + "(KHTML, like Gecko) Chrome/124.0 Safari/537.36" // publishedRe matches the tab-pane id niedziela.pl gives a fully published // day's page (e.g. "tabnowy0all"). Its absence -- together with a "Przykro // nam" placeholder -- marks a date that has not been published yet; see // ewangelia.py's fetch() for the original behaviour this mirrors. var publishedRe = regexp.MustCompile(`id="\w*0all"`) // dateRe is the same YYYY-MM-DD shape internal/cli's dateRe validates // against. Load checks opts.Date against it before building any filesystem // path (jsonPath/htmlPath below are built by string concatenation, so an // unvalidated Date is a path-traversal vector) -- defense-in-depth so every // caller (web, cli, tui) is protected even if a future caller forgets to // validate its own input first. var dateRe = regexp.MustCompile(`^\d{4}-\d{2}-\d{2}$`) // Options controls how Load resolves a day's readings. type Options struct { // Date is the day to load, formatted YYYY-MM-DD. Date string // Refresh bypasses both cache layers and re-fetches from the network. Refresh bool // Offline restricts Load to previously cached/harvested data (see // LoadOffline), never hitting the network. Offline bool } // cacheDir is where the HTML/JSON cache layers live: // ${XDG_CACHE_HOME:-~/.cache}/lectio/ func cacheDir() string { base := os.Getenv("XDG_CACHE_HOME") if base == "" { home, err := os.UserHomeDir() if err != nil { home = "." } base = filepath.Join(home, ".cache") } return filepath.Join(base, "lectio") } // Load returns the day's reading sections, from the modern (niedziela.pl) // source, preferring cache over network: // // 0. If Offline, skip straight to LoadOffline (the harvested sigla TSV). // 1. Unless Refresh, the parsed JSON cache ({date}.json). // 2. Unless Refresh, the raw HTML cache ({date}.html) -- parsed, and the // result written to the JSON cache for next time. // 3. Otherwise, fetch the page over the network, parse it, and -- only if // the page is fully published -- write both cache layers. A fetch error // here (e.g. no network) falls back to LoadOffline for this date if it // has been harvested, and only surfaces the original fetch error if // that fallback also fails. func Load(opts Options) ([]Section, error) { if !dateRe.MatchString(opts.Date) { return nil, fmt.Errorf("invalid date %q: want YYYY-MM-DD", opts.Date) } if opts.Offline { return LoadOffline(opts.Date) } dir := cacheDir() jsonPath := filepath.Join(dir, opts.Date+".json") htmlPath := filepath.Join(dir, opts.Date+".html") if !opts.Refresh { if secs, err := loadJSONCache(jsonPath); err == nil { return secs, nil } if page, err := os.ReadFile(htmlPath); err == nil { secs, err := Parse(string(page)) if err != nil { return nil, err } writeJSONCache(jsonPath, secs) return secs, nil } } page, err := fetch(opts.Date) if err != nil { // Network is unreachable: fall back to a prior harvest of this date // if there is one, rather than failing outright. if secs, offErr := LoadOffline(opts.Date); offErr == nil { return secs, nil } return nil, err } // Parse itself reports an unpublished date as "no reading published for // this date yet" (via the "Przykro nam" placeholder), so an unpublished // page is neither cached nor returned as sections here. secs, err := Parse(page) if err != nil { return nil, err } // Cache only fully-published pages, so an as-yet-unpublished future date // keeps being retried instead of caching a "no reading" placeholder. if publishedRe.MatchString(page) { if err := os.MkdirAll(dir, 0o755); err == nil { _ = os.WriteFile(htmlPath, []byte(page), 0o644) writeJSONCache(jsonPath, secs) } } return secs, nil } // loadJSONCache reads and unmarshals the parsed-sections cache file. func loadJSONCache(path string) ([]Section, error) { data, err := os.ReadFile(path) if err != nil { return nil, err } var secs []Section if err := json.Unmarshal(data, &secs); err != nil { return nil, err } return secs, nil } // writeJSONCache best-effort writes the parsed sections cache; a failure to // cache should never fail the load itself. func writeJSONCache(path string, secs []Section) { data, err := json.Marshal(secs) if err != nil { return } _ = os.WriteFile(path, data, 0o644) } // fetch GETs the day's page from baseURL with the browser User-Agent. // Whether the page is actually published is left to the caller (Parse // detects an unpublished date; Load re-checks the tab id to decide whether // the result is cache-worthy). func fetch(dateStr string) (string, error) { url := fmt.Sprintf(baseURL, dateStr) req, err := http.NewRequest(http.MethodGet, url, nil) if err != nil { return "", err } req.Header.Set("User-Agent", userAgent) client := &http.Client{Timeout: 20 * time.Second} resp, err := client.Do(req) if err != nil { return "", fmt.Errorf("failed to fetch %s: %w", url, err) } defer resp.Body.Close() body, err := io.ReadAll(resp.Body) if err != nil { return "", fmt.Errorf("failed to read response from %s: %w", url, err) } return string(body), nil }