changed method of fetching crontab schedule
This commit is contained in:
+14
-4
@@ -31,8 +31,17 @@ DATA_FILE=/path/to/events.json ARCHIVE_DATA_FILE=/path/to/archive.json uvicorn m
|
|||||||
### Scrape schedule reporting
|
### Scrape schedule reporting
|
||||||
|
|
||||||
`/health` reports when each scraper output file was last written (its mtime)
|
`/health` reports when each scraper output file was last written (its mtime)
|
||||||
and, if you tell it the cron schedule, when it's next due. This doesn't read
|
and, if it can determine the cron schedule, when it's next due. The schedule
|
||||||
your crontab — set the same expression(s) you put there as env vars:
|
is sourced in order:
|
||||||
|
|
||||||
|
1. `SCRAPE_CRON` / `ARCHIVE_SCRAPE_CRON` env vars, if set — always wins, and
|
||||||
|
the only option that works when the API doesn't run as the same user/host
|
||||||
|
as the cron job.
|
||||||
|
2. Otherwise, the API's own OS user's crontab (`crontab -l`), looked up for a
|
||||||
|
line invoking `scraper.py` (with vs. without `--archive` picks archive vs.
|
||||||
|
plain). Only works when the API process runs as the same user whose
|
||||||
|
personal crontab holds the scrape job — not a system crontab/cron.d entry,
|
||||||
|
not a job scheduled under a different user or host.
|
||||||
|
|
||||||
```bash
|
```bash
|
||||||
SCRAPE_CRON="0 * * * *" ARCHIVE_SCRAPE_CRON="0 4 * * *" uvicorn main:app
|
SCRAPE_CRON="0 * * * *" ARCHIVE_SCRAPE_CRON="0 4 * * *" uvicorn main:app
|
||||||
@@ -40,8 +49,9 @@ SCRAPE_CRON="0 * * * *" ARCHIVE_SCRAPE_CRON="0 4 * * *" uvicorn main:app
|
|||||||
|
|
||||||
`SCRAPE_CRON` covers the plain scrape (`fundi-scraped-output.json`);
|
`SCRAPE_CRON` covers the plain scrape (`fundi-scraped-output.json`);
|
||||||
`ARCHIVE_SCRAPE_CRON` covers `--archive` (`fundi-archive-output.json`) and
|
`ARCHIVE_SCRAPE_CRON` covers `--archive` (`fundi-archive-output.json`) and
|
||||||
falls back to `SCRAPE_CRON` if unset — set it separately only if the archive
|
falls back to `SCRAPE_CRON` (env or crontab-discovered) if unset — set it
|
||||||
scrape runs on its own cron line. Leave both unset to just get the
|
separately only if the archive scrape runs on its own cron line. If neither
|
||||||
|
an env var nor a matching crontab line is found, you just get the
|
||||||
last-scrape info with `next_scrape_at` / `seconds_until_next_scrape` as
|
last-scrape info with `next_scrape_at` / `seconds_until_next_scrape` as
|
||||||
`null`.
|
`null`.
|
||||||
|
|
||||||
|
|||||||
Binary file not shown.
+9
-3
@@ -17,7 +17,7 @@ from fastapi.responses import PlainTextResponse
|
|||||||
from calendar_feed import build_ics
|
from calendar_feed import build_ics
|
||||||
from datasource import EventStore, iter_artist_names, parse_event_date
|
from datasource import EventStore, iter_artist_names, parse_event_date
|
||||||
from models import Event, EventList
|
from models import Event, EventList
|
||||||
from scrape_schedule import get_schedule_info
|
from scrape_schedule import discover_cron_expr, get_schedule_info
|
||||||
|
|
||||||
app = FastAPI(
|
app = FastAPI(
|
||||||
title="Fundi Scraper API",
|
title="Fundi Scraper API",
|
||||||
@@ -59,8 +59,14 @@ def _matches(
|
|||||||
|
|
||||||
@app.get("/health")
|
@app.get("/health")
|
||||||
def health():
|
def health():
|
||||||
scrape_cron = os.environ.get("SCRAPE_CRON")
|
scrape_cron = os.environ.get("SCRAPE_CRON") or discover_cron_expr(
|
||||||
archive_scrape_cron = os.environ.get("ARCHIVE_SCRAPE_CRON") or scrape_cron
|
"scraper.py", archive=False
|
||||||
|
)
|
||||||
|
archive_scrape_cron = (
|
||||||
|
os.environ.get("ARCHIVE_SCRAPE_CRON")
|
||||||
|
or discover_cron_expr("scraper.py", archive=True)
|
||||||
|
or scrape_cron
|
||||||
|
)
|
||||||
return {
|
return {
|
||||||
"status": "ok",
|
"status": "ok",
|
||||||
"events_loaded": len(store.all()),
|
"events_loaded": len(store.all()),
|
||||||
|
|||||||
+64
-5
@@ -1,15 +1,22 @@
|
|||||||
"""Reports when a scrape last ran and when cron will next run it.
|
"""Reports when a scrape last ran and when cron will next run it.
|
||||||
|
|
||||||
"Last ran" is read from the mtime of the scraper's output file - true
|
"Last ran" is read from the mtime of the scraper's output file - true
|
||||||
regardless of how/when the scrape was actually invoked. "Next run" requires
|
regardless of how/when the scrape was actually invoked. "Next run" needs the
|
||||||
telling this module the cron schedule via an env var (there's no reliable,
|
cron schedule, sourced in order of preference:
|
||||||
portable way to introspect another process's crontab from here), given as
|
|
||||||
a standard 5-field cron expression, e.g. "0 * * * *".
|
1. An explicit env var (SCRAPE_CRON / ARCHIVE_SCRAPE_CRON) - always wins,
|
||||||
|
works anywhere.
|
||||||
|
2. The current OS user's crontab (``discover_cron_expr``) - only works when
|
||||||
|
this process runs as the same user whose personal crontab (``crontab -e``)
|
||||||
|
contains the scrape job, and the ``crontab`` binary is available. Not
|
||||||
|
portable beyond that (different user, no cron, system crontab/cron.d
|
||||||
|
instead of a personal one, containers split across hosts, ...).
|
||||||
"""
|
"""
|
||||||
|
|
||||||
|
import subprocess
|
||||||
from datetime import datetime
|
from datetime import datetime
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
from typing import Optional
|
from typing import List, Optional, Tuple
|
||||||
|
|
||||||
from croniter import croniter
|
from croniter import croniter
|
||||||
|
|
||||||
@@ -18,6 +25,58 @@ def _now_local() -> datetime:
|
|||||||
return datetime.now().astimezone()
|
return datetime.now().astimezone()
|
||||||
|
|
||||||
|
|
||||||
|
def _read_crontab_lines() -> List[str]:
|
||||||
|
"""Return the invoking OS user's crontab lines, or [] if unavailable."""
|
||||||
|
try:
|
||||||
|
result = subprocess.run(
|
||||||
|
["crontab", "-l"],
|
||||||
|
capture_output=True,
|
||||||
|
text=True,
|
||||||
|
timeout=5,
|
||||||
|
)
|
||||||
|
except (OSError, subprocess.SubprocessError):
|
||||||
|
return []
|
||||||
|
if result.returncode != 0:
|
||||||
|
return []
|
||||||
|
return result.stdout.splitlines()
|
||||||
|
|
||||||
|
|
||||||
|
def _parse_crontab_line(line: str) -> Optional[Tuple[str, str]]:
|
||||||
|
"""Split one crontab line into (schedule, command); None if not a job."""
|
||||||
|
line = line.strip()
|
||||||
|
if not line or line.startswith("#"):
|
||||||
|
return None
|
||||||
|
if line.startswith("@"):
|
||||||
|
parts = line.split(maxsplit=1)
|
||||||
|
if len(parts) != 2:
|
||||||
|
return None
|
||||||
|
return parts[0], parts[1]
|
||||||
|
parts = line.split(maxsplit=5)
|
||||||
|
if len(parts) != 6:
|
||||||
|
return None
|
||||||
|
return " ".join(parts[:5]), parts[5]
|
||||||
|
|
||||||
|
|
||||||
|
def discover_cron_expr(command_hint: str, *, archive: bool) -> Optional[str]:
|
||||||
|
"""Find a matching scrape job in the current user's crontab.
|
||||||
|
|
||||||
|
``command_hint`` is a substring identifying the scrape command (e.g.
|
||||||
|
"scraper.py"); ``archive`` picks the "--archive" line vs. the plain one.
|
||||||
|
Returns None if the crontab can't be read or no line matches.
|
||||||
|
"""
|
||||||
|
for line in _read_crontab_lines():
|
||||||
|
parsed = _parse_crontab_line(line)
|
||||||
|
if parsed is None:
|
||||||
|
continue
|
||||||
|
schedule, command = parsed
|
||||||
|
if command_hint not in command:
|
||||||
|
continue
|
||||||
|
if ("--archive" in command) != archive:
|
||||||
|
continue
|
||||||
|
return schedule
|
||||||
|
return None
|
||||||
|
|
||||||
|
|
||||||
def _file_mtime_local(path: Path) -> Optional[datetime]:
|
def _file_mtime_local(path: Path) -> Optional[datetime]:
|
||||||
try:
|
try:
|
||||||
return datetime.fromtimestamp(path.stat().st_mtime).astimezone()
|
return datetime.fromtimestamp(path.stat().st_mtime).astimezone()
|
||||||
|
|||||||
Reference in New Issue
Block a user