Files
fundi-scraper-api/api/scrape_schedule.py
T
2026-09-09 13:50:22 +02:00

123 lines
4.2 KiB
Python

"""Reports when a scrape last ran and when cron will next run it.
"Last ran" is read from the mtime of the scraper's output file - true
regardless of how/when the scrape was actually invoked. "Next run" needs the
cron schedule, sourced in order of preference:
1. An explicit env var (SCRAPE_CRON / ARCHIVE_SCRAPE_CRON) - always wins,
works anywhere.
2. The current OS user's crontab (``discover_cron_expr``) - only works when
this process runs as the same user whose personal crontab (``crontab -e``)
contains the scrape job, and the ``crontab`` binary is available. Not
portable beyond that (different user, no cron, system crontab/cron.d
instead of a personal one, containers split across hosts, ...).
"""
import subprocess
from datetime import datetime
from pathlib import Path
from typing import List, Optional, Tuple
from croniter import croniter
def _now_local() -> datetime:
return datetime.now().astimezone()
def _read_crontab_lines() -> List[str]:
"""Return the invoking OS user's crontab lines, or [] if unavailable."""
try:
result = subprocess.run(
["crontab", "-l"],
capture_output=True,
text=True,
timeout=5,
)
except (OSError, subprocess.SubprocessError):
return []
if result.returncode != 0:
return []
return result.stdout.splitlines()
def _parse_crontab_line(line: str) -> Optional[Tuple[str, str]]:
"""Split one crontab line into (schedule, command); None if not a job."""
line = line.strip()
if not line or line.startswith("#"):
return None
if line.startswith("@"):
parts = line.split(maxsplit=1)
if len(parts) != 2:
return None
return parts[0], parts[1]
parts = line.split(maxsplit=5)
if len(parts) != 6:
return None
return " ".join(parts[:5]), parts[5]
def discover_cron_expr(command_hint: str, *, archive: bool) -> Optional[str]:
"""Find a matching scrape job in the current user's crontab.
``command_hint`` is a substring identifying the scrape command (e.g.
"scraper.py"); ``archive`` picks the "--archive" line vs. the plain one.
Each scrape typically has two crontab lines - an "@reboot" one and a
timed one; "@reboot" isn't a recurring schedule (croniter can't compute
a next run from it), so it's skipped in favor of the timed line.
Returns None if the crontab can't be read or no timed line matches.
"""
for line in _read_crontab_lines():
parsed = _parse_crontab_line(line)
if parsed is None:
continue
schedule, command = parsed
if schedule.lower() == "@reboot":
continue
if command_hint not in command:
continue
if ("--archive" in command) != archive:
continue
return schedule
return None
def _file_mtime_local(path: Path) -> Optional[datetime]:
try:
return datetime.fromtimestamp(path.stat().st_mtime).astimezone()
except FileNotFoundError:
return None
def _next_cron_run(cron_expr: str, after: datetime) -> Optional[datetime]:
try:
naive_after = after.replace(tzinfo=None)
return croniter(cron_expr, naive_after).get_next(datetime).astimezone()
except (ValueError, KeyError):
return None
def get_schedule_info(path: Path, cron_expr: Optional[str]) -> dict:
"""Build the last/next-scrape block for one scraper output file.
``cron_expr`` is that scrape's cron expression (e.g. "0 * * * *");
None/invalid -> next-run fields are null.
"""
now = _now_local()
cron_expr = (cron_expr or "").strip() or None
last_scrape_at = _file_mtime_local(path)
next_scrape_at = _next_cron_run(cron_expr, now) if cron_expr else None
return {
"cron_schedule": cron_expr,
"last_scrape_at": last_scrape_at.isoformat() if last_scrape_at else None,
"seconds_since_last_scrape": (
round((now - last_scrape_at).total_seconds()) if last_scrape_at else None
),
"next_scrape_at": next_scrape_at.isoformat() if next_scrape_at else None,
"seconds_until_next_scrape": (
round((next_scrape_at - now).total_seconds()) if next_scrape_at else None
),
}