osint-dashboard/news/scraper/run_news_scraper.py
Sirius DevOps 8643153954 feat: continuous news scrape + 15-min analyst, expand urls.txt
Scraper loops with NEWS_SCRAPE_INTERVAL_S (default 10s after each
crawl). Summarizer runs every NEWS_SUMMARIZE_INTERVAL_S (default 900)
over the last 15 minutes of articles. Feed list replaced from the
k8s scrapy-urls configmap (334 sources).
2026-08-28 20:53:32 -04:00

53 lines
1.6 KiB
Python

#!/usr/bin/env python3
"""Scheduler loop for the news scraper — crawl continuously.
As soon as one Scrapy pass finishes, wait NEWS_SCRAPE_INTERVAL_S seconds
and start the next. Two crawls never overlap (the loop is serial).
Env (all optional, 12-factor):
NEWS_SCRAPE_INTERVAL_S seconds between crawls (default 10)
NEWS_SCRAPE_RUN_ON_START "1" to crawl immediately on boot (default 1)
"""
from __future__ import annotations
import datetime
import logging
import os
import subprocess
import sys
import time
logging.basicConfig(level=logging.INFO, format="%(asctime)s %(levelname)s %(name)s: %(message)s")
logger = logging.getLogger("news.scraper")
INTERVAL_S = max(0, int(os.getenv("NEWS_SCRAPE_INTERVAL_S", "10")))
RUN_ON_START = os.getenv("NEWS_SCRAPE_RUN_ON_START", "1").lower() in ("1", "true", "yes")
CRAWL_CMD = ["scrapy", "crawl", "articles"]
def run_crawl() -> None:
logger.info("scrape starting at %s", datetime.datetime.now().isoformat(timespec="seconds"))
try:
proc = subprocess.run(CRAWL_CMD, cwd="/app")
logger.info("scrape finished rc=%s", proc.returncode)
except Exception: # noqa: BLE001 — keep the loop alive across failures
logger.exception("scrape failed")
def main() -> None:
logger.info(
"news scraper loop starting (interval_s=%s, run_on_start=%s)",
INTERVAL_S, RUN_ON_START,
)
if RUN_ON_START:
run_crawl()
while True:
logger.info("next scrape in %ss", INTERVAL_S)
time.sleep(INTERVAL_S)
run_crawl()
if __name__ == "__main__":
sys.exit(main())