Compare commits
No commits in common. "master" and "fix/news-ticker-hud" have entirely different histories.
master
...
fix/news-t
109 changed files with 1565 additions and 14896 deletions
72
.env.example
72
.env.example
|
|
@ -31,6 +31,20 @@ NOMINATIM_URL=https://nominatim.openstreetmap.org
|
|||
NOMINATIM_MIN_INTERVAL=1.1
|
||||
SNAPSHOT_TTL_SECONDS=300
|
||||
|
||||
# ── masscan active camera discovery (host-level systemd service, NOT compose) ─
|
||||
# Continuous rolling sweep for open RTSP port 554 across a range. Runs on the
|
||||
# Pi host via deploy/osint-masscan.service (needs root + raw sockets). Results
|
||||
# land in the same `cameras` table as the scraper (discovery_source=masscan).
|
||||
# NOTE: 200 pps is the residential-safe default. 1k/10k pps saturated a home
|
||||
# uplink. A full 0.0.0.0/0 sweep at 200 pps takes ~8 months (rolling).
|
||||
MASSCAN_RANGE=0.0.0.0/0
|
||||
MASSCAN_PORTS=554
|
||||
MASSCAN_RATE=200
|
||||
MASSCAN_RETRIES=1
|
||||
MASSCAN_WAIT=0
|
||||
MASSCAN_EXCLUDEFILE=/etc/osint-dashboard/masscan-excludes.txt
|
||||
MASSCAN_FLUSH_EVERY=250
|
||||
|
||||
# ── NASA FIRMS (active fire / hotspot ingest) ──────────────────────────────
|
||||
# MAP_KEY is FREE — get one at https://firms.modaps.eosdis.nasa.gov/api/map_key_info/
|
||||
# (1-minute signup, no payment). Leave blank to keep fire ingest idle.
|
||||
|
|
@ -45,65 +59,35 @@ FIRMS_INTERVAL=900
|
|||
# Set to 0 to disable the fire loop entirely.
|
||||
INGEST_FIRES=1
|
||||
|
||||
# ── VesselAPI (commercial REST AIS — Hormuz, 5×/day, 150 calls/mo cap) ─────
|
||||
# Independent of AISStream (open/shared live US-coast WebSocket). Both stay
|
||||
# on when their keys are set; missing one never disables the other.
|
||||
# Prefer pasting VESSELAPI_API_KEY on the dashboard Keys page.
|
||||
# The poller idles when the key is unset. Never called from map pans
|
||||
# (GET /api/vessels serves the shared last-known cache only).
|
||||
VESSELAPI_API_KEY=
|
||||
# Bounding box(es) as minlat,minlon,maxlat,maxlon (lat/lon order). Semicolon-
|
||||
# separated for multiple boxes. Default = Strait of Hormuz (span 3.6 ≤ 4° cap).
|
||||
VESSELAPI_BBOX=25.5,55.4,27.3,57.2
|
||||
# Poll cadence in seconds (17280 = 4.8h → 5 polls/day = 150/mo).
|
||||
VESSELAPI_INTERVAL=17280
|
||||
# Local hard cap on successful 2xx calls per UTC day (persisted in Postgres).
|
||||
VESSELAPI_MAX_CALLS_PER_DAY=5
|
||||
# 1 = run the poller inside the dashboard process (default); ingester off.
|
||||
VESSELAPI_IN_APP=1
|
||||
VESSELAPI_IN_INGEST=0
|
||||
|
||||
# ── API keys (managed from the dashboard UI) ──────────────────────────────
|
||||
# Keys such as NOUS_API_KEY and TELEGRAM_TOKEN are stored in the Postgres
|
||||
# Keys such as GEMINI_API_KEY and TELEGRAM_TOKEN are stored in the Postgres
|
||||
# `api_keys` table and managed from the dashboard's "Keys" tab
|
||||
# (GET/POST/DELETE /api/keys/{name}) — see app/keystore.py. The FIRMS ingestor
|
||||
# currently reads FIRMS_MAP_KEY from .env (above); wiring the Keys-UI store as
|
||||
# its lookup/fallback is a planned follow-up.
|
||||
|
||||
# ── News pipeline (scraper + summarizer, profile `ingest`) ────────────────
|
||||
# Scraper crawls urls.txt continuously (default 10s between crawls).
|
||||
# Summarizer runs Nous map-reduce every 15 min (NEWS_SUMMARIZE_INTERVAL_S=900).
|
||||
# NOUS_API_KEY is also (preferably) set in the Keys UI; env is an override.
|
||||
# Unset in both env and api_keys = summarizer logs and idles (never crashes).
|
||||
NOUS_API_KEY=
|
||||
NOUS_BASE_URL=https://inference-api.nousresearch.com/v1
|
||||
# Hourly: the scraper crawls 257 RSS sources at minute :00 and the summarizer
|
||||
# runs the Gemini map-reduce at minute :05, both writing to the shared osint-db
|
||||
# (tables `articles` + `article_summaries`, created by alembic 003_news).
|
||||
# Consume via GET /api/news and GET /api/news/summaries.
|
||||
# GEMINI_API_KEY is REQUIRED for summarization; unset = summarizer idles.
|
||||
GEMINI_API_KEY=
|
||||
# Optional LLM knobs
|
||||
# SUMMARY_MODEL is an optional override. Leave unset so Settings
|
||||
# (app_settings.SUMMARY_MODEL) can reach the summarizer. Code default
|
||||
# Hermes-4.3-36B remains after a Postgres miss. Env wins when set.
|
||||
# SUMMARY_MODEL=
|
||||
SUMMARY_MODEL=gemini-2.0-flash
|
||||
NEWS_BATCH_SIZE=50
|
||||
SUMMARY_WINDOW_MINUTES=15
|
||||
SUMMARY_WINDOW_HOURS=1
|
||||
# Futures/markets coupling from the upstream pipeline is OFF by default
|
||||
# (irrelevant to OSINT). Set INCLUDE_FUTURES=1 + install yfinance to enable.
|
||||
INCLUDE_FUTURES=0
|
||||
NEWS_SCRAPE_INTERVAL_S=10
|
||||
NEWS_SUMMARIZE_INTERVAL_S=900
|
||||
# Wall-clock scheduling (k8s CronJob replacement): scrape minute, summarize minute
|
||||
NEWS_SCRAPE_MINUTE=0
|
||||
NEWS_SUMMARIZE_MINUTE=5
|
||||
# Run once immediately on container start (seeds data fast), then align to the
|
||||
# scheduled minute.
|
||||
NEWS_SCRAPE_RUN_ON_START=1
|
||||
NEWS_SUMMARIZE_RUN_ON_START=1
|
||||
# "1" ignores the interval idempotency skip (double-pins on recreate).
|
||||
NEWS_SUMMARIZE_FORCE=0
|
||||
NEWS_LOG_LEVEL=INFO
|
||||
# Reserved for the (out-of-scope) Telegram delivery bot.
|
||||
TELEGRAM_TOKEN=
|
||||
TELEGRAM_CHAT_ID=
|
||||
|
||||
# =============================================================================
|
||||
# Forgejo container registry (CI publishes here; NOT ghcr.io)
|
||||
# =============================================================================
|
||||
# forgejo.siriusdevops.com/sirius/osint-dashboard[:tag]
|
||||
# forgejo.siriusdevops.com/sirius/osint-dashboard-pg[:tag]
|
||||
# forgejo.siriusdevops.com/sirius/osint-news-scraper[:tag]
|
||||
# forgejo.siriusdevops.com/sirius/osint-news-summarizer[:tag]
|
||||
FORGEJO_REGISTRY=forgejo.siriusdevops.com
|
||||
FORGEJO_OWNER=sirius
|
||||
|
|
|
|||
|
|
@ -1,249 +1,27 @@
|
|||
# Build changed OSINT images, publish to the Forgejo container registry, then
|
||||
# redeploy on the Pi runner (docker.sock mounted).
|
||||
#
|
||||
# Unchanged images are skipped. Dockerfile.pg / osint-dashboard-pg is NOT
|
||||
# rebuilt or pulled on a normal merge — Postgres stays up. Rebuild it only
|
||||
# when Dockerfile.pg changes, or via workflow_dispatch rebuild_pg.
|
||||
#
|
||||
# Public pull host: forgejo.siriusdevops.com (NOT ghcr.io)
|
||||
# CI push host: 127.0.0.1:3000 — Cloudflare 413s layers ≳100MB on the public
|
||||
# hostname, even from the Pi (hairpins out through the tunnel).
|
||||
# Images (public names):
|
||||
# forgejo.siriusdevops.com/sirius/osint-dashboard
|
||||
# forgejo.siriusdevops.com/sirius/osint-dashboard-pg
|
||||
# forgejo.siriusdevops.com/sirius/osint-news-scraper
|
||||
# forgejo.siriusdevops.com/sirius/osint-news-summarizer
|
||||
#
|
||||
# Optional repo variable FORGEJO_REGISTRY overrides the *public* pull host.
|
||||
# Deploy still uses the local docker socket on the runner host (rpi).
|
||||
|
||||
name: build-and-deploy
|
||||
|
||||
on:
|
||||
push:
|
||||
branches: [main, master]
|
||||
workflow_dispatch:
|
||||
inputs:
|
||||
rebuild_pg:
|
||||
description: Rebuild Timescale+PostGIS (Dockerfile.pg)
|
||||
type: boolean
|
||||
default: false
|
||||
rebuild_all:
|
||||
description: Rebuild every app image (ignore path filter)
|
||||
type: boolean
|
||||
default: false
|
||||
|
||||
env:
|
||||
PUBLIC_REGISTRY: ${{ vars.FORGEJO_REGISTRY || 'forgejo.siriusdevops.com' }}
|
||||
PUSH_REGISTRY: 127.0.0.1:3000
|
||||
OWNER: sirius
|
||||
# Keep compose project/volumes stable on the Pi
|
||||
COMPOSE_PROJECT_NAME: osint-dashboard
|
||||
|
||||
jobs:
|
||||
build-push-deploy:
|
||||
build:
|
||||
runs-on: docker
|
||||
permissions:
|
||||
contents: read
|
||||
packages: write
|
||||
steps:
|
||||
- name: Checkout
|
||||
uses: https://code.forgejo.org/actions/checkout@v4
|
||||
with:
|
||||
fetch-depth: 50
|
||||
|
||||
- name: Plan image builds
|
||||
id: plan
|
||||
run: |
|
||||
set -euo pipefail
|
||||
APP=0
|
||||
SCRAPER=0
|
||||
SUM=0
|
||||
PG=0
|
||||
COMPOSE=0
|
||||
|
||||
mark() {
|
||||
case "$1" in
|
||||
Dockerfile.pg)
|
||||
PG=1 ;;
|
||||
Dockerfile|app/*|alembic/*|alembic.ini)
|
||||
APP=1 ;;
|
||||
news/scraper/*)
|
||||
SCRAPER=1 ;;
|
||||
news/summerizer/*)
|
||||
SUM=1 ;;
|
||||
docker-compose.yml|scripts/compose-reup.sh)
|
||||
COMPOSE=1 ;;
|
||||
esac
|
||||
}
|
||||
|
||||
if [ "${{ github.event_name }}" = "workflow_dispatch" ]; then
|
||||
APP=1; SCRAPER=1; SUM=1
|
||||
if [ "${{ github.event.inputs.rebuild_all }}" = "true" ]; then
|
||||
APP=1; SCRAPER=1; SUM=1; PG=1
|
||||
fi
|
||||
if [ "${{ github.event.inputs.rebuild_pg }}" = "true" ]; then
|
||||
PG=1
|
||||
fi
|
||||
else
|
||||
BEFORE="${{ github.event.before }}"
|
||||
SHA="${GITHUB_SHA}"
|
||||
ZEROS="0000000000000000000000000000000000000000"
|
||||
if [ -z "$BEFORE" ] || [ "$BEFORE" = "$ZEROS" ]; then
|
||||
echo "No previous SHA — build app images, skip pg"
|
||||
APP=1; SCRAPER=1; SUM=1
|
||||
elif ! git cat-file -e "${BEFORE}^{commit}" 2>/dev/null; then
|
||||
echo "Previous SHA $BEFORE not in history — build app images, skip pg"
|
||||
APP=1; SCRAPER=1; SUM=1
|
||||
else
|
||||
while IFS= read -r f; do
|
||||
[ -z "$f" ] && continue
|
||||
mark "$f"
|
||||
done < <(git diff --name-only "$BEFORE" "$SHA")
|
||||
fi
|
||||
fi
|
||||
|
||||
{
|
||||
echo "app=$APP"
|
||||
echo "scraper=$SCRAPER"
|
||||
echo "summarizer=$SUM"
|
||||
echo "pg=$PG"
|
||||
echo "compose=$COMPOSE"
|
||||
} >> "$GITHUB_OUTPUT"
|
||||
echo "plan app=$APP scraper=$SCRAPER summarizer=$SUM pg=$PG compose=$COMPOSE"
|
||||
|
||||
- name: Image refs
|
||||
id: img
|
||||
run: |
|
||||
set -euo pipefail
|
||||
PUSH="${PUSH_REGISTRY}"
|
||||
PUB="${PUBLIC_REGISTRY}"
|
||||
OWN="${OWNER}"
|
||||
SHA="${GITHUB_SHA::12}"
|
||||
{
|
||||
echo "reg=$PUSH"
|
||||
echo "pub=$PUB"
|
||||
echo "sha=$SHA"
|
||||
echo "app=$PUSH/$OWN/osint-dashboard"
|
||||
echo "pg=$PUSH/$OWN/osint-dashboard-pg"
|
||||
echo "scraper=$PUSH/$OWN/osint-news-scraper"
|
||||
echo "summarizer=$PUSH/$OWN/osint-news-summarizer"
|
||||
} >> "$GITHUB_OUTPUT"
|
||||
echo "Push registry: $PUSH"
|
||||
echo "Public pull: $PUB"
|
||||
echo "SHA tag: $SHA"
|
||||
|
||||
- name: Login to Forgejo registry
|
||||
if: steps.plan.outputs.app == '1' || steps.plan.outputs.scraper == '1' || steps.plan.outputs.summarizer == '1' || steps.plan.outputs.pg == '1'
|
||||
run: |
|
||||
set -euo pipefail
|
||||
# GITHUB_TOKEN login "succeeds" but blob uploads 401 (Forgejo packages
|
||||
# reject the Actions token). Use a user PAT with write:package.
|
||||
if [ -z "${{ secrets.FORGEJO_TOKEN }}" ]; then
|
||||
echo "Missing repo secret FORGEJO_TOKEN (user PAT, not GITHUB_TOKEN)"
|
||||
exit 1
|
||||
fi
|
||||
echo "${{ secrets.FORGEJO_TOKEN }}" | docker login "${{ steps.img.outputs.reg }}" \
|
||||
-u sirius --password-stdin
|
||||
|
||||
- name: Build application image (api / ingester / cameras)
|
||||
if: steps.plan.outputs.app == '1'
|
||||
run: |
|
||||
set -ex
|
||||
APP="${{ steps.img.outputs.app }}"
|
||||
SHA="${{ steps.img.outputs.sha }}"
|
||||
docker build -f Dockerfile -t "${APP}:latest" -t "${APP}:${SHA}" \
|
||||
-t "localhost/osint-dashboard:latest" .
|
||||
docker push "${APP}:latest"
|
||||
docker push "${APP}:${SHA}"
|
||||
|
||||
- name: Build news-scraper image
|
||||
if: steps.plan.outputs.scraper == '1'
|
||||
run: |
|
||||
set -ex
|
||||
IMG="${{ steps.img.outputs.scraper }}"
|
||||
SHA="${{ steps.img.outputs.sha }}"
|
||||
docker build -f news/scraper/Dockerfile -t "${IMG}:latest" -t "${IMG}:${SHA}" \
|
||||
-t "localhost/osint-news-scraper:latest" news/scraper
|
||||
docker push "${IMG}:latest"
|
||||
docker push "${IMG}:${SHA}"
|
||||
|
||||
- name: Build news-summarizer image
|
||||
if: steps.plan.outputs.summarizer == '1'
|
||||
run: |
|
||||
set -ex
|
||||
IMG="${{ steps.img.outputs.summarizer }}"
|
||||
SHA="${{ steps.img.outputs.sha }}"
|
||||
docker build -f news/summerizer/Dockerfile -t "${IMG}:latest" -t "${IMG}:${SHA}" \
|
||||
-t "localhost/osint-news-summarizer:latest" news/summerizer
|
||||
docker push "${IMG}:latest"
|
||||
docker push "${IMG}:${SHA}"
|
||||
|
||||
- name: Build / refresh Timescale+PostGIS image
|
||||
if: steps.plan.outputs.pg == '1'
|
||||
run: |
|
||||
set -ex
|
||||
PG="${{ steps.img.outputs.pg }}"
|
||||
SHA="${{ steps.img.outputs.sha }}"
|
||||
if docker build -f Dockerfile.pg -t "${PG}:latest" -t "${PG}:${SHA}" \
|
||||
-t "localhost/osint-dashboard-pg:latest" .; then
|
||||
docker push "${PG}:latest"
|
||||
docker push "${PG}:${SHA}"
|
||||
elif docker image inspect "localhost/osint-dashboard-pg:latest" >/dev/null 2>&1; then
|
||||
echo "WARN: Dockerfile.pg build failed; keeping existing local pg image"
|
||||
else
|
||||
echo "ERROR: cannot build or find osint-dashboard-pg image"
|
||||
exit 1
|
||||
fi
|
||||
|
||||
- name: Deploy on runner host (compose)
|
||||
- name: Build and deploy on the Pi (local docker)
|
||||
run: |
|
||||
set -ex
|
||||
# The forgejo-runner runs on the Pi with /var/run/docker.sock and
|
||||
# /opt/siriusdevops mounted, so we build + deploy LOCALLY — no SSH/scp.
|
||||
# Deploy straight from the checked-out workspace.
|
||||
cd "${GITHUB_WORKSPACE}"
|
||||
APP="${{ steps.plan.outputs.app }}"
|
||||
SCRAPER="${{ steps.plan.outputs.scraper }}"
|
||||
SUM="${{ steps.plan.outputs.summarizer }}"
|
||||
PG="${{ steps.plan.outputs.pg }}"
|
||||
COMPOSE="${{ steps.plan.outputs.compose }}"
|
||||
|
||||
SVCS=()
|
||||
[ "$APP" = "1" ] && SVCS+=(app ingester camera-service)
|
||||
[ "$SCRAPER" = "1" ] && SVCS+=(news-scraper)
|
||||
[ "$SUM" = "1" ] && SVCS+=(news-summarizer)
|
||||
if [ "$COMPOSE" = "1" ]; then
|
||||
# compose/script change: bounce workers so env/command updates apply.
|
||||
# Still do not bounce Postgres.
|
||||
for s in app ingester camera-service news-scraper news-summarizer; do
|
||||
case " ${SVCS[*]} " in
|
||||
*" $s "*) ;;
|
||||
*) SVCS+=("$s") ;;
|
||||
esac
|
||||
done
|
||||
fi
|
||||
|
||||
chmod +x scripts/compose-reup.sh
|
||||
if [ "$PG" = "1" ]; then
|
||||
FORCE_RECREATE_DB=1 COMPOSE_PROJECT_NAME=osint-dashboard COMPOSE_PROFILES=ingest \
|
||||
scripts/compose-reup.sh "${SVCS[@]}" db
|
||||
elif [ "${#SVCS[@]}" -gt 0 ]; then
|
||||
COMPOSE_PROJECT_NAME=osint-dashboard COMPOSE_PROFILES=ingest \
|
||||
scripts/compose-reup.sh "${SVCS[@]}"
|
||||
else
|
||||
echo "No image or compose changes — leave running containers alone"
|
||||
docker compose --profile ingest ps
|
||||
fi
|
||||
# Project name is pinned by `name:` in docker-compose.yml
|
||||
# (osint-dashboard), matching the live volume
|
||||
# osint-dashboard_osint-pgdata. Do NOT override COMPOSE_PROJECT_NAME
|
||||
# here — a different name would create a fresh empty pgdata volume.
|
||||
docker compose build --no-cache
|
||||
docker compose up -d --force-recreate
|
||||
docker image prune -f
|
||||
echo "osint-dashboard deploy done; db image left in place unless pg=1"
|
||||
echo 'osint-dashboard deployed (local)'
|
||||
|
||||
- name: Summary
|
||||
if: always()
|
||||
run: |
|
||||
{
|
||||
echo "## Image plan"
|
||||
echo "- app: \`${{ steps.plan.outputs.app }}\`"
|
||||
echo "- news-scraper: \`${{ steps.plan.outputs.scraper }}\`"
|
||||
echo "- news-summarizer: \`${{ steps.plan.outputs.summarizer }}\`"
|
||||
echo "- pg (Timescale): \`${{ steps.plan.outputs.pg }}\`"
|
||||
echo
|
||||
echo "Postgres is rebuilt/pulled only when \`Dockerfile.pg\` changes (or workflow_dispatch rebuild_pg)."
|
||||
} >> "$GITHUB_STEP_SUMMARY"
|
||||
# deployed locally on the Pi via runner (docker socket + cli-plugins mounted)
|
||||
|
|
|
|||
|
|
@ -1,64 +0,0 @@
|
|||
"""news_items table + article_summaries.model
|
||||
|
||||
Revision ID: 005_news_items
|
||||
Revises: 004_camera_enum
|
||||
Create Date: 2026-08-28
|
||||
"""
|
||||
|
||||
from alembic import op
|
||||
|
||||
# revision identifiers, used by Alembic.
|
||||
revision = "005_news_items"
|
||||
down_revision = "004_camera_enum"
|
||||
branch_labels = None
|
||||
depends_on = None
|
||||
|
||||
|
||||
def upgrade() -> None:
|
||||
# Idempotent DDL: ingest (summarizer ensure_tables) may create the same
|
||||
# shapes first depending on container startup order. IF NOT EXISTS makes
|
||||
# both orders safe — whichever runs first wins, the other no-ops.
|
||||
# One statement per op.execute: asyncpg rejects multi-command prepared
|
||||
# statements (same style as 003_news).
|
||||
op.execute(
|
||||
"""
|
||||
ALTER TABLE article_summaries
|
||||
ADD COLUMN IF NOT EXISTS model TEXT
|
||||
"""
|
||||
)
|
||||
op.execute(
|
||||
"""
|
||||
CREATE TABLE IF NOT EXISTS news_items (
|
||||
id SERIAL PRIMARY KEY,
|
||||
summary_id INTEGER REFERENCES article_summaries(id) ON DELETE CASCADE,
|
||||
kind TEXT NOT NULL,
|
||||
headline TEXT NOT NULL,
|
||||
importance TEXT NOT NULL,
|
||||
location_name TEXT,
|
||||
lat DOUBLE PRECISION,
|
||||
lon DOUBLE PRECISION,
|
||||
location_confidence TEXT,
|
||||
category TEXT,
|
||||
url TEXT,
|
||||
created_at TIMESTAMPTZ NOT NULL DEFAULT NOW()
|
||||
)
|
||||
"""
|
||||
)
|
||||
op.execute(
|
||||
"""
|
||||
CREATE INDEX IF NOT EXISTS ix_news_items_kind_created
|
||||
ON news_items (kind, created_at DESC)
|
||||
"""
|
||||
)
|
||||
op.execute(
|
||||
"""
|
||||
CREATE INDEX IF NOT EXISTS ix_news_items_map_bbox
|
||||
ON news_items (lon, lat)
|
||||
WHERE kind = 'map' AND lat IS NOT NULL AND lon IS NOT NULL
|
||||
"""
|
||||
)
|
||||
|
||||
|
||||
def downgrade() -> None:
|
||||
op.execute("DROP TABLE IF EXISTS news_items")
|
||||
op.execute("ALTER TABLE article_summaries DROP COLUMN IF EXISTS model")
|
||||
|
|
@ -1,196 +0,0 @@
|
|||
"""phase 2: geofences, 1-min track CAGGs, fire/aircraft hits
|
||||
|
||||
Revision ID: 005_phase2
|
||||
Revises: 004_camera_enum
|
||||
Create Date: 2026-08-28
|
||||
"""
|
||||
|
||||
from alembic import op
|
||||
|
||||
revision = "005_phase2"
|
||||
down_revision = "004_camera_enum"
|
||||
branch_labels = None
|
||||
depends_on = None
|
||||
|
||||
|
||||
def upgrade() -> None:
|
||||
op.execute("CREATE EXTENSION IF NOT EXISTS postgis")
|
||||
op.execute("CREATE EXTENSION IF NOT EXISTS timescaledb")
|
||||
|
||||
op.execute("""
|
||||
CREATE TABLE IF NOT EXISTS geofences (
|
||||
id UUID PRIMARY KEY,
|
||||
name TEXT NOT NULL,
|
||||
geojson JSONB NOT NULL,
|
||||
geom geometry(Polygon, 4326),
|
||||
active INTEGER NOT NULL DEFAULT 1,
|
||||
created_at TIMESTAMPTZ NOT NULL DEFAULT now(),
|
||||
updated_at TIMESTAMPTZ NOT NULL DEFAULT now()
|
||||
)
|
||||
""")
|
||||
op.execute("""
|
||||
CREATE INDEX IF NOT EXISTS ix_geofences_geom
|
||||
ON geofences USING gist (geom)
|
||||
""")
|
||||
|
||||
op.execute("""
|
||||
CREATE TABLE IF NOT EXISTS geofence_alerts (
|
||||
id UUID PRIMARY KEY,
|
||||
geofence_id UUID NOT NULL,
|
||||
source_kind TEXT NOT NULL,
|
||||
entity_id TEXT NOT NULL,
|
||||
lat DOUBLE PRECISION,
|
||||
lon DOUBLE PRECISION,
|
||||
payload JSONB,
|
||||
created_at TIMESTAMPTZ NOT NULL DEFAULT now()
|
||||
)
|
||||
""")
|
||||
op.execute("""
|
||||
CREATE INDEX IF NOT EXISTS ix_geofence_alerts_created
|
||||
ON geofence_alerts (created_at DESC)
|
||||
""")
|
||||
|
||||
op.execute("""
|
||||
CREATE TABLE IF NOT EXISTS vessel_positions (
|
||||
mmsi TEXT NOT NULL,
|
||||
ts TIMESTAMPTZ NOT NULL,
|
||||
lat DOUBLE PRECISION NOT NULL,
|
||||
lon DOUBLE PRECISION NOT NULL,
|
||||
heading DOUBLE PRECISION,
|
||||
speed DOUBLE PRECISION,
|
||||
label TEXT,
|
||||
extra JSONB,
|
||||
PRIMARY KEY (mmsi, ts)
|
||||
)
|
||||
""")
|
||||
op.execute("""
|
||||
SELECT create_hypertable(
|
||||
'vessel_positions', 'ts', if_not_exists => TRUE
|
||||
)
|
||||
""")
|
||||
op.execute("""
|
||||
CREATE INDEX IF NOT EXISTS ix_vessel_positions_bbox
|
||||
ON vessel_positions (lon, lat)
|
||||
""")
|
||||
|
||||
op.execute("""
|
||||
CREATE TABLE IF NOT EXISTS aircraft_positions (
|
||||
hex TEXT NOT NULL,
|
||||
ts TIMESTAMPTZ NOT NULL,
|
||||
lat DOUBLE PRECISION NOT NULL,
|
||||
lon DOUBLE PRECISION NOT NULL,
|
||||
heading DOUBLE PRECISION,
|
||||
speed DOUBLE PRECISION,
|
||||
label TEXT,
|
||||
extra JSONB,
|
||||
PRIMARY KEY (hex, ts)
|
||||
)
|
||||
""")
|
||||
op.execute("""
|
||||
SELECT create_hypertable(
|
||||
'aircraft_positions', 'ts', if_not_exists => TRUE
|
||||
)
|
||||
""")
|
||||
op.execute("""
|
||||
CREATE INDEX IF NOT EXISTS ix_aircraft_positions_bbox
|
||||
ON aircraft_positions (lon, lat)
|
||||
""")
|
||||
|
||||
op.execute("""
|
||||
CREATE TABLE IF NOT EXISTS fire_aircraft_hits (
|
||||
id UUID NOT NULL,
|
||||
fire_id TEXT NOT NULL,
|
||||
fire_lat DOUBLE PRECISION NOT NULL,
|
||||
fire_lon DOUBLE PRECISION NOT NULL,
|
||||
aircraft_hex TEXT NOT NULL,
|
||||
aircraft_type TEXT,
|
||||
aircraft_lat DOUBLE PRECISION NOT NULL,
|
||||
aircraft_lon DOUBLE PRECISION NOT NULL,
|
||||
distance_mi DOUBLE PRECISION NOT NULL,
|
||||
seen_at TIMESTAMPTZ NOT NULL,
|
||||
PRIMARY KEY (fire_id, aircraft_hex, seen_at)
|
||||
)
|
||||
""")
|
||||
op.execute("""
|
||||
CREATE INDEX IF NOT EXISTS ix_fire_aircraft_hits_seen
|
||||
ON fire_aircraft_hits (seen_at DESC)
|
||||
""")
|
||||
|
||||
# 1-minute continuous aggregates (Timescale). last() keeps the newest
|
||||
# sample in each bucket — the DVR slider reads these, not the raw table.
|
||||
op.execute("""
|
||||
DO $$
|
||||
BEGIN
|
||||
IF NOT EXISTS (
|
||||
SELECT 1 FROM timescaledb_information.continuous_aggregates
|
||||
WHERE view_name = 'vessel_tracks_1min'
|
||||
) THEN
|
||||
EXECUTE $v$
|
||||
CREATE MATERIALIZED VIEW vessel_tracks_1min
|
||||
WITH (timescaledb.continuous) AS
|
||||
SELECT time_bucket('1 minute', ts) AS bucket,
|
||||
mmsi,
|
||||
last(lat, ts) AS lat,
|
||||
last(lon, ts) AS lon,
|
||||
last(heading, ts) AS heading,
|
||||
last(speed, ts) AS speed,
|
||||
last(label, ts) AS label
|
||||
FROM vessel_positions
|
||||
GROUP BY bucket, mmsi
|
||||
WITH NO DATA
|
||||
$v$;
|
||||
END IF;
|
||||
IF NOT EXISTS (
|
||||
SELECT 1 FROM timescaledb_information.continuous_aggregates
|
||||
WHERE view_name = 'aircraft_tracks_1min'
|
||||
) THEN
|
||||
EXECUTE $a$
|
||||
CREATE MATERIALIZED VIEW aircraft_tracks_1min
|
||||
WITH (timescaledb.continuous) AS
|
||||
SELECT time_bucket('1 minute', ts) AS bucket,
|
||||
hex,
|
||||
last(lat, ts) AS lat,
|
||||
last(lon, ts) AS lon,
|
||||
last(heading, ts) AS heading,
|
||||
last(speed, ts) AS speed,
|
||||
last(label, ts) AS label
|
||||
FROM aircraft_positions
|
||||
GROUP BY bucket, hex
|
||||
WITH NO DATA
|
||||
$a$;
|
||||
END IF;
|
||||
END
|
||||
$$;
|
||||
""")
|
||||
op.execute("""
|
||||
DO $$
|
||||
BEGIN
|
||||
PERFORM add_continuous_aggregate_policy(
|
||||
'vessel_tracks_1min',
|
||||
start_offset => INTERVAL '3 hours',
|
||||
end_offset => INTERVAL '1 minute',
|
||||
schedule_interval => INTERVAL '1 minute',
|
||||
if_not_exists => TRUE
|
||||
);
|
||||
PERFORM add_continuous_aggregate_policy(
|
||||
'aircraft_tracks_1min',
|
||||
start_offset => INTERVAL '3 hours',
|
||||
end_offset => INTERVAL '1 minute',
|
||||
schedule_interval => INTERVAL '1 minute',
|
||||
if_not_exists => TRUE
|
||||
);
|
||||
EXCEPTION WHEN OTHERS THEN
|
||||
NULL;
|
||||
END
|
||||
$$;
|
||||
""")
|
||||
|
||||
|
||||
def downgrade() -> None:
|
||||
op.execute("DROP MATERIALIZED VIEW IF EXISTS aircraft_tracks_1min CASCADE")
|
||||
op.execute("DROP MATERIALIZED VIEW IF EXISTS vessel_tracks_1min CASCADE")
|
||||
op.execute("DROP TABLE IF EXISTS fire_aircraft_hits")
|
||||
op.execute("DROP TABLE IF EXISTS aircraft_positions")
|
||||
op.execute("DROP TABLE IF EXISTS vessel_positions")
|
||||
op.execute("DROP TABLE IF EXISTS geofence_alerts")
|
||||
op.execute("DROP TABLE IF EXISTS geofences")
|
||||
|
|
@ -1,24 +0,0 @@
|
|||
"""merge 005_phase2 and 005_news_items
|
||||
|
||||
Revision ID: 006_merge_heads
|
||||
Revises: 005_phase2, 005_news_items
|
||||
Create Date: 2026-08-28
|
||||
|
||||
Two PRs both parented 004_camera_enum (phase2 geofences + news_items).
|
||||
`alembic upgrade head` then fails and crash-loops the dashboard.
|
||||
"""
|
||||
|
||||
from alembic import op # noqa: F401
|
||||
|
||||
revision = "006_merge_heads"
|
||||
down_revision = ("005_phase2", "005_news_items")
|
||||
branch_labels = None
|
||||
depends_on = None
|
||||
|
||||
|
||||
def upgrade() -> None:
|
||||
pass
|
||||
|
||||
|
||||
def downgrade() -> None:
|
||||
pass
|
||||
|
|
@ -1,115 +0,0 @@
|
|||
"""event_dedup + Timescale compression/retention
|
||||
|
||||
Revision ID: 007_event_dedup
|
||||
Revises: 006_merge_heads
|
||||
Create Date: 2026-08-28
|
||||
"""
|
||||
|
||||
from alembic import op
|
||||
|
||||
revision = "007_event_dedup"
|
||||
down_revision = "006_merge_heads"
|
||||
branch_labels = None
|
||||
depends_on = None
|
||||
|
||||
|
||||
def upgrade() -> None:
|
||||
op.execute("""
|
||||
CREATE TABLE IF NOT EXISTS event_dedup (
|
||||
url TEXT PRIMARY KEY,
|
||||
created_at TIMESTAMPTZ NOT NULL DEFAULT now()
|
||||
)
|
||||
""")
|
||||
|
||||
# Keep the earliest row per URL, drop the 10× USGS/camera dupes.
|
||||
op.execute("""
|
||||
DELETE FROM events a
|
||||
USING events b
|
||||
WHERE a.url IS NOT NULL AND a.url <> ''
|
||||
AND a.url = b.url
|
||||
AND (a.ingested_at, a.id) > (b.ingested_at, b.id)
|
||||
""")
|
||||
op.execute("""
|
||||
INSERT INTO event_dedup (url)
|
||||
SELECT DISTINCT url FROM events
|
||||
WHERE url IS NOT NULL AND url <> ''
|
||||
ON CONFLICT (url) DO NOTHING
|
||||
""")
|
||||
|
||||
# Compression + retention. Policies no-op if Timescale rejects (fresh PG).
|
||||
op.execute("""
|
||||
DO $$
|
||||
BEGIN
|
||||
PERFORM add_compression_policy('events', INTERVAL '7 days', if_not_exists => TRUE);
|
||||
EXCEPTION WHEN OTHERS THEN
|
||||
BEGIN
|
||||
ALTER TABLE events SET (
|
||||
timescaledb.compress,
|
||||
timescaledb.compress_segmentby = 'source_type',
|
||||
timescaledb.compress_orderby = 'ingested_at DESC'
|
||||
);
|
||||
PERFORM add_compression_policy('events', INTERVAL '7 days', if_not_exists => TRUE);
|
||||
EXCEPTION WHEN OTHERS THEN
|
||||
NULL;
|
||||
END;
|
||||
END
|
||||
$$;
|
||||
""")
|
||||
op.execute("""
|
||||
DO $$
|
||||
BEGIN
|
||||
PERFORM add_retention_policy('events', INTERVAL '180 days', if_not_exists => TRUE);
|
||||
EXCEPTION WHEN OTHERS THEN
|
||||
NULL;
|
||||
END
|
||||
$$;
|
||||
""")
|
||||
op.execute("""
|
||||
DO $$
|
||||
BEGIN
|
||||
ALTER TABLE fires SET (
|
||||
timescaledb.compress,
|
||||
timescaledb.compress_segmentby = 'satellite',
|
||||
timescaledb.compress_orderby = 'acq_time DESC'
|
||||
);
|
||||
PERFORM add_compression_policy('fires', INTERVAL '7 days', if_not_exists => TRUE);
|
||||
PERFORM add_retention_policy('fires', INTERVAL '90 days', if_not_exists => TRUE);
|
||||
EXCEPTION WHEN OTHERS THEN
|
||||
NULL;
|
||||
END
|
||||
$$;
|
||||
""")
|
||||
op.execute("""
|
||||
DO $$
|
||||
BEGIN
|
||||
ALTER TABLE aircraft_positions SET (
|
||||
timescaledb.compress,
|
||||
timescaledb.compress_segmentby = 'hex',
|
||||
timescaledb.compress_orderby = 'ts DESC'
|
||||
);
|
||||
PERFORM add_compression_policy('aircraft_positions', INTERVAL '1 day', if_not_exists => TRUE);
|
||||
PERFORM add_retention_policy('aircraft_positions', INTERVAL '14 days', if_not_exists => TRUE);
|
||||
EXCEPTION WHEN OTHERS THEN
|
||||
NULL;
|
||||
END
|
||||
$$;
|
||||
""")
|
||||
op.execute("""
|
||||
DO $$
|
||||
BEGIN
|
||||
ALTER TABLE vessel_positions SET (
|
||||
timescaledb.compress,
|
||||
timescaledb.compress_segmentby = 'mmsi',
|
||||
timescaledb.compress_orderby = 'ts DESC'
|
||||
);
|
||||
PERFORM add_compression_policy('vessel_positions', INTERVAL '1 day', if_not_exists => TRUE);
|
||||
PERFORM add_retention_policy('vessel_positions', INTERVAL '14 days', if_not_exists => TRUE);
|
||||
EXCEPTION WHEN OTHERS THEN
|
||||
NULL;
|
||||
END
|
||||
$$;
|
||||
""")
|
||||
|
||||
|
||||
def downgrade() -> None:
|
||||
op.execute("DROP TABLE IF EXISTS event_dedup")
|
||||
|
|
@ -1,26 +0,0 @@
|
|||
"""article_summaries.kind — interval vs daily_recap
|
||||
|
||||
Revision ID: 008_summary_kind
|
||||
Revises: 007_event_dedup
|
||||
Create Date: 2026-08-28
|
||||
"""
|
||||
|
||||
from alembic import op
|
||||
|
||||
revision = "008_summary_kind"
|
||||
down_revision = "007_event_dedup"
|
||||
branch_labels = None
|
||||
depends_on = None
|
||||
|
||||
|
||||
def upgrade() -> None:
|
||||
op.execute(
|
||||
"""
|
||||
ALTER TABLE article_summaries
|
||||
ADD COLUMN IF NOT EXISTS kind TEXT
|
||||
"""
|
||||
)
|
||||
|
||||
|
||||
def downgrade() -> None:
|
||||
op.execute("ALTER TABLE article_summaries DROP COLUMN IF EXISTS kind")
|
||||
|
|
@ -1,41 +0,0 @@
|
|||
"""vessels — daily VesselAPI snapshots for DVR as-of
|
||||
|
||||
Revision ID: 009_vessels
|
||||
Revises: 008_summary_kind
|
||||
Create Date: 2026-08-29
|
||||
"""
|
||||
|
||||
from alembic import op
|
||||
|
||||
revision = "009_vessels"
|
||||
down_revision = "008_summary_kind"
|
||||
branch_labels = None
|
||||
depends_on = None
|
||||
|
||||
|
||||
def upgrade() -> None:
|
||||
op.execute(
|
||||
"""
|
||||
CREATE TABLE IF NOT EXISTS vessels (
|
||||
mmsi TEXT NOT NULL,
|
||||
poll_at TIMESTAMPTZ NOT NULL,
|
||||
lat DOUBLE PRECISION NOT NULL,
|
||||
lon DOUBLE PRECISION NOT NULL,
|
||||
heading DOUBLE PRECISION,
|
||||
speed DOUBLE PRECISION,
|
||||
label TEXT,
|
||||
extra JSONB,
|
||||
PRIMARY KEY (mmsi, poll_at)
|
||||
)
|
||||
"""
|
||||
)
|
||||
op.execute(
|
||||
"CREATE INDEX IF NOT EXISTS ix_vessels_poll_at ON vessels (poll_at DESC)"
|
||||
)
|
||||
op.execute(
|
||||
"CREATE INDEX IF NOT EXISTS ix_vessels_bbox ON vessels (lon, lat)"
|
||||
)
|
||||
|
||||
|
||||
def downgrade() -> None:
|
||||
op.execute("DROP TABLE IF EXISTS vessels")
|
||||
|
|
@ -1,29 +0,0 @@
|
|||
"""GIST bbox indexes for events/fires map-pan queries.
|
||||
|
||||
Revision ID: 010_bbox_gist
|
||||
Revises: 009_vessels
|
||||
Create Date: 2026-09-01
|
||||
"""
|
||||
|
||||
from alembic import op
|
||||
|
||||
revision = "010_bbox_gist"
|
||||
down_revision = "009_vessels"
|
||||
branch_labels = None
|
||||
depends_on = None
|
||||
|
||||
|
||||
def upgrade() -> None:
|
||||
op.execute(
|
||||
"CREATE INDEX IF NOT EXISTS ix_events_geom_gist ON events "
|
||||
"USING gist (ST_SetSRID(ST_MakePoint(location_lon, location_lat), 4326))"
|
||||
)
|
||||
op.execute(
|
||||
"CREATE INDEX IF NOT EXISTS ix_fires_geom_gist ON fires "
|
||||
"USING gist (ST_SetSRID(ST_MakePoint(longitude, latitude), 4326))"
|
||||
)
|
||||
|
||||
|
||||
def downgrade() -> None:
|
||||
op.execute("DROP INDEX IF EXISTS ix_fires_geom_gist")
|
||||
op.execute("DROP INDEX IF EXISTS ix_events_geom_gist")
|
||||
|
|
@ -1,24 +0,0 @@
|
|||
"""geofence_alerts (geofence_id, created_at DESC) for fence-scoped hit log
|
||||
|
||||
Revision ID: 011_geofence_alerts_fence
|
||||
Revises: 010_bbox_gist
|
||||
Create Date: 2026-09-01
|
||||
"""
|
||||
|
||||
from alembic import op
|
||||
|
||||
revision = "011_geofence_alerts_fence"
|
||||
down_revision = "010_bbox_gist"
|
||||
branch_labels = None
|
||||
depends_on = None
|
||||
|
||||
|
||||
def upgrade() -> None:
|
||||
op.execute(
|
||||
"CREATE INDEX IF NOT EXISTS ix_geofence_alerts_fence_created "
|
||||
"ON geofence_alerts (geofence_id, created_at DESC)"
|
||||
)
|
||||
|
||||
|
||||
def downgrade() -> None:
|
||||
op.execute("DROP INDEX IF EXISTS ix_geofence_alerts_fence_created")
|
||||
|
|
@ -14,7 +14,6 @@ import json
|
|||
import logging
|
||||
import os
|
||||
import random
|
||||
import time
|
||||
|
||||
from config import AISSTREAM_API_KEY, AISSTREAM_BBOX
|
||||
from keystore import get_api_key
|
||||
|
|
@ -30,36 +29,6 @@ FILTER_TYPES = [
|
|||
"ShipStaticData",
|
||||
]
|
||||
|
||||
# ── Viewport-following ─────────────────────────────────────────────────────
|
||||
# The frontend POSTs its current viewport box to /api/vessels/subscribe; the
|
||||
# worker retunes the AISStream subscription to it (throttled to 1/s, the
|
||||
# service's subscription-update cap). ``None`` keeps the env AISSTREAM_BBOX
|
||||
# default. Last writer wins; the key never reaches the browser.
|
||||
_desired_boxes: list[list[list[float]]] | None = None
|
||||
_bbox_guard = asyncio.Lock()
|
||||
|
||||
|
||||
async def request_viewport_bbox(
|
||||
minlon: float, minlat: float, maxlon: float, maxlat: float
|
||||
) -> None:
|
||||
"""Retune the live subscription to a viewport box (lon/lat input order)."""
|
||||
global _desired_boxes
|
||||
box = [[minlat, minlon], [maxlat, maxlon]] # AISStream wants [lat, lon] corners
|
||||
async with _bbox_guard:
|
||||
_desired_boxes = [box]
|
||||
|
||||
|
||||
async def reset_viewport_bbox() -> None:
|
||||
"""Fall back to the env AISSTREAM_BBOX default."""
|
||||
global _desired_boxes
|
||||
async with _bbox_guard:
|
||||
_desired_boxes = None
|
||||
|
||||
|
||||
async def _take_desired_boxes() -> list[list[list[float]]] | None:
|
||||
async with _bbox_guard:
|
||||
return _desired_boxes
|
||||
|
||||
|
||||
def _parse_boxes(raw: str) -> list[list[list[float]]]:
|
||||
"""Env format: minlat,minlon,maxlat,maxlon[; ...]. AIS wants [[lat,lon],[lat,lon]]."""
|
||||
|
|
@ -96,7 +65,7 @@ async def run_ais_worker() -> None:
|
|||
)
|
||||
await asyncio.sleep(60)
|
||||
continue
|
||||
boxes = await _take_desired_boxes() or _parse_boxes(AISSTREAM_BBOX)
|
||||
boxes = _parse_boxes(AISSTREAM_BBOX)
|
||||
try:
|
||||
async with websockets.connect(
|
||||
WS_URL,
|
||||
|
|
@ -113,28 +82,7 @@ async def run_ais_worker() -> None:
|
|||
await ws.send(json.dumps(sub))
|
||||
logger.info("AISStream subscribed (%d bbox(es))", len(boxes))
|
||||
backoff = 2.0
|
||||
last_submit = time.monotonic()
|
||||
while True:
|
||||
# Follow the client viewport: coalesce to the latest request
|
||||
# and honor AISStream's 1 subscription-update/s cap.
|
||||
desired = await _take_desired_boxes()
|
||||
if (
|
||||
desired is not None
|
||||
and desired != boxes
|
||||
and time.monotonic() - last_submit >= 1.0
|
||||
):
|
||||
boxes = desired
|
||||
sub["BoundingBoxes"] = boxes
|
||||
await ws.send(json.dumps(sub))
|
||||
last_submit = time.monotonic()
|
||||
logger.info(
|
||||
"AISStream re-subscribed to viewport (%d bbox)",
|
||||
len(boxes),
|
||||
)
|
||||
try:
|
||||
raw = await asyncio.wait_for(ws.recv(), timeout=0.25)
|
||||
except asyncio.TimeoutError:
|
||||
continue
|
||||
async for raw in ws:
|
||||
if isinstance(raw, bytes):
|
||||
raw = raw.decode("utf-8", errors="replace")
|
||||
try:
|
||||
|
|
|
|||
|
|
@ -1,72 +0,0 @@
|
|||
"""Background ffmpeg — never block a FastAPI request on a frame grab.
|
||||
|
||||
ffmpeg frame grabs are scheduled with asyncio.create_task and shared per URL.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import asyncio
|
||||
import shutil
|
||||
from cachetools import TTLCache
|
||||
|
||||
_ffmpeg_cache: TTLCache = TTLCache(maxsize=100, ttl=300)
|
||||
_ffmpeg_tasks: dict[str, asyncio.Task] = {}
|
||||
_FFMPEG = shutil.which("ffmpeg")
|
||||
|
||||
|
||||
def cached_ffmpeg_jpeg(url: str) -> bytes | None:
|
||||
return _ffmpeg_cache.get(url)
|
||||
|
||||
|
||||
def schedule_ffmpeg_snapshot(url: str, timeout: float = 8.0) -> asyncio.Task:
|
||||
"""Start (or reuse) an ffmpeg JPEG grab. Caller may await the task."""
|
||||
existing = _ffmpeg_tasks.get(url)
|
||||
if existing is not None and not existing.done():
|
||||
return existing
|
||||
task = asyncio.create_task(_ffmpeg_grab_and_cache(url, timeout))
|
||||
_ffmpeg_tasks[url] = task
|
||||
return task
|
||||
|
||||
|
||||
async def _ffmpeg_grab(url: str, timeout: float = 8.0) -> bytes | None:
|
||||
"""Grab one JPEG frame. Isolated so tests can stub it."""
|
||||
if not _FFMPEG or not url.lower().startswith("rtsp://"):
|
||||
return None
|
||||
cmd = [
|
||||
_FFMPEG, "-hide_banner", "-loglevel", "error", "-nostdin",
|
||||
"-rtsp_transport", "tcp",
|
||||
"-timeout", "4000000",
|
||||
"-i", url,
|
||||
"-frames:v", "1",
|
||||
"-f", "image2pipe", "-vcodec", "mjpeg",
|
||||
"pipe:1",
|
||||
]
|
||||
try:
|
||||
proc = await asyncio.create_subprocess_exec(
|
||||
*cmd,
|
||||
stdout=asyncio.subprocess.PIPE,
|
||||
stderr=asyncio.subprocess.DEVNULL,
|
||||
)
|
||||
except FileNotFoundError:
|
||||
return None
|
||||
try:
|
||||
stdout, _ = await asyncio.wait_for(proc.communicate(), timeout=timeout)
|
||||
except asyncio.TimeoutError:
|
||||
proc.kill()
|
||||
try:
|
||||
await proc.wait()
|
||||
except Exception: # noqa: BLE001
|
||||
pass
|
||||
return None
|
||||
if proc.returncode not in (0, None) or not stdout or len(stdout) < 64:
|
||||
return None
|
||||
if stdout[:2] != b"\xff\xd8":
|
||||
return None
|
||||
return stdout
|
||||
|
||||
|
||||
async def _ffmpeg_grab_and_cache(url: str, timeout: float) -> bytes | None:
|
||||
data = await _ffmpeg_grab(url, timeout)
|
||||
if data:
|
||||
_ffmpeg_cache[url] = data
|
||||
return data
|
||||
|
|
@ -16,8 +16,6 @@ CALTRANS_CCTV_URLS = tuple(
|
|||
f"https://cwwp2.dot.ca.gov/data/d{n}/cctv/cctvStatusD{n:02d}.json"
|
||||
for n in range(1, 13)
|
||||
)
|
||||
# MDOT MiDrive official DOT CCTV list (fields carry rendered HTML).
|
||||
MDOT_CAMERA_URL = "https://mdotjboss.state.mi.us/MiDrive/camera/list"
|
||||
_DEFAULT_SOURCE_URL = ",".join((
|
||||
# Publicly published open-camera list (markdown bullets of stream URLs).
|
||||
"https://raw.githubusercontent.com/fury999io/public-ip-cams/main/README.md",
|
||||
|
|
@ -27,10 +25,6 @@ _DEFAULT_SOURCE_URL = ",".join((
|
|||
"https://raw.githubusercontent.com/willytop8/Live-Environment-Streams/main/streams.geojson",
|
||||
# Official Caltrans CWWP2 JPEG + HLS CCTV (districts 1–12).
|
||||
*CALTRANS_CCTV_URLS,
|
||||
# Oregon DOT TripCheck public CCTV JPEG inventory (Esri JSON).
|
||||
"https://www.tripcheck.com/Scripts/map/data/cctvinventory.js",
|
||||
# Official MDOT MiDrive CCTV (JPEG stills, Michigan).
|
||||
MDOT_CAMERA_URL,
|
||||
))
|
||||
CAMERA_SOURCE_URLS = [
|
||||
u.strip()
|
||||
|
|
@ -61,15 +55,3 @@ SNAPSHOT_TIMEOUT = float(os.getenv("SNAPSHOT_TIMEOUT", "8.0"))
|
|||
|
||||
# NATS subject cameras are published on (consumed by the shared ingester).
|
||||
CAMERA_NATS_SUBJECT = os.getenv("CAMERA_NATS_SUBJECT", "events.camera")
|
||||
|
||||
|
||||
# ── UDOT IBI 511 traffic cameras ──────────────────────────────────────────
|
||||
# DataTables endpoint (POST form-encoded; server caps at 100 rows/page no
|
||||
# matter what `length` is sent). No API key. Snapshot stills live at a stable
|
||||
# /map/Cctv/{id} URL — same URL always serves the latest frame, so we store
|
||||
# the URL and never scrape every frame ourselves.
|
||||
UDOT_IBI_URL = "https://prod-ut.ibi511.com/List/GetData/Cameras"
|
||||
UDOT_IBI_BASE = "https://prod-ut.ibi511.com"
|
||||
UDOT_IBI_PAGE_SIZE = 100
|
||||
# Safety cap on pages per cycle so a runaway recordsTotal cannot fan out.
|
||||
UDOT_IBI_MAX_PAGES = int(os.getenv("UDOT_IBI_MAX_PAGES", "40"))
|
||||
|
|
|
|||
|
|
@ -1,7 +1,7 @@
|
|||
"""Resolve a browser-renderable preview for a camera.
|
||||
|
||||
HTTP/MJPEG cameras already expose a snapshot_url the existing proxy can
|
||||
stream. Some scraper sources store `rtsp://` URLs with no snapshot_url, so
|
||||
stream. masscan finds are stored as `rtsp://IP/` with no snapshot_url, so
|
||||
the map popup used to skip the <img> entirely and the leftover source link
|
||||
handed the browser an rtsp:// URL (which opens VLC).
|
||||
|
||||
|
|
@ -42,6 +42,16 @@ _HTTP_PATHS = (
|
|||
"/tmpfs/auto.jpg",
|
||||
)
|
||||
|
||||
# Browser-playable MJPEG paths the /stream proxy can pass through.
|
||||
_MJPEG_PATHS = (
|
||||
"/mjpg/video.mjpg",
|
||||
"/video.mjpg",
|
||||
"/cgi-bin/mjpg/video.cgi",
|
||||
"/axis-cgi/mjpg/video.cgi",
|
||||
"/nphMotionJpeg",
|
||||
"/mjpeg.cgi",
|
||||
)
|
||||
|
||||
_FFMPEG = shutil.which("ffmpeg")
|
||||
|
||||
|
||||
|
|
@ -77,18 +87,88 @@ async def _http_get_image(url: str, timeout: float = 2.5) -> bytes | None:
|
|||
return None
|
||||
|
||||
|
||||
async def ffmpeg_snapshot(url: str, timeout: float = 8.0) -> bytes | None:
|
||||
"""Grab a single JPEG frame from an RTSP URL. None if ffmpeg missing/fails.
|
||||
async def _http_feed_url(url: str, timeout: float = 2.5) -> str | None:
|
||||
"""Return url if it looks like an unauthenticated image/MJPEG feed."""
|
||||
try:
|
||||
async with httpx.AsyncClient(
|
||||
timeout=timeout, follow_redirects=True,
|
||||
headers={"User-Agent": USER_AGENT},
|
||||
) as c:
|
||||
async with c.stream("GET", url) as r:
|
||||
if r.status_code != 200:
|
||||
return None
|
||||
ctype = (r.headers.get("content-type") or "").lower()
|
||||
if "html" in ctype or ctype.startswith("text/"):
|
||||
return None
|
||||
if any(x in ctype for x in ("image/", "multipart", "mjpeg", "octet-stream")):
|
||||
# Read a little to reject empty/error bodies.
|
||||
chunk = b""
|
||||
async for b in r.aiter_bytes():
|
||||
chunk += b
|
||||
if len(chunk) >= 64:
|
||||
break
|
||||
if len(chunk) < 64:
|
||||
return None
|
||||
if b"html" in chunk[:64].lower():
|
||||
return None
|
||||
return url
|
||||
except Exception: # noqa: BLE001
|
||||
return None
|
||||
return None
|
||||
|
||||
The subprocess is scheduled via asyncio.create_task (shared per URL) so
|
||||
concurrent popup clicks do not stack ffmpeg processes on the request path.
|
||||
|
||||
async def probe_public_feed(host: str) -> str | None:
|
||||
"""Unauthenticated HTTP still or MJPEG URL for this host, or None.
|
||||
|
||||
Used at masscan ingest time so dead RTSP-only hosts never hit the map.
|
||||
No credentials, no RTSP path-walking (too slow / rarely public).
|
||||
"""
|
||||
from bg_jobs import cached_ffmpeg_jpeg, schedule_ffmpeg_snapshot
|
||||
urls = [f"http://{host}{p}" for p in _HTTP_PATHS]
|
||||
urls.append(f"http://{host}:8080/shot.jpg")
|
||||
urls.extend(f"http://{host}{p}" for p in _MJPEG_PATHS)
|
||||
results = await asyncio.gather(
|
||||
*(_http_feed_url(u) for u in urls),
|
||||
return_exceptions=True,
|
||||
)
|
||||
for url, hit in zip(urls, results):
|
||||
if isinstance(hit, str) and hit:
|
||||
return hit
|
||||
return None
|
||||
|
||||
hit = cached_ffmpeg_jpeg(url)
|
||||
if hit:
|
||||
return hit
|
||||
return await schedule_ffmpeg_snapshot(url, timeout)
|
||||
|
||||
async def ffmpeg_snapshot(url: str, timeout: float = 8.0) -> bytes | None:
|
||||
"""Grab a single JPEG frame from an RTSP URL. None if ffmpeg missing/fails."""
|
||||
if not _FFMPEG or not url.lower().startswith("rtsp://"):
|
||||
return None
|
||||
cmd = [
|
||||
_FFMPEG, "-hide_banner", "-loglevel", "error", "-nostdin",
|
||||
"-rtsp_transport", "tcp",
|
||||
"-timeout", "4000000", # 4s socket timeout, microseconds
|
||||
"-i", url,
|
||||
"-frames:v", "1",
|
||||
"-f", "image2pipe", "-vcodec", "mjpeg",
|
||||
"pipe:1",
|
||||
]
|
||||
try:
|
||||
proc = await asyncio.create_subprocess_exec(
|
||||
*cmd,
|
||||
stdout=asyncio.subprocess.PIPE,
|
||||
stderr=asyncio.subprocess.DEVNULL,
|
||||
)
|
||||
except FileNotFoundError:
|
||||
return None
|
||||
try:
|
||||
stdout, _ = await asyncio.wait_for(proc.communicate(), timeout=timeout)
|
||||
except asyncio.TimeoutError:
|
||||
proc.kill()
|
||||
try:
|
||||
await proc.wait()
|
||||
except Exception: # noqa: BLE001
|
||||
pass
|
||||
return None
|
||||
if proc.returncode not in (0, None) or not _looks_like_jpeg(stdout or b""):
|
||||
return None
|
||||
return stdout
|
||||
|
||||
|
||||
async def ffmpeg_mjpeg_stream(url: str):
|
||||
|
|
|
|||
|
|
@ -25,7 +25,6 @@ import hashlib
|
|||
import ipaddress
|
||||
import json
|
||||
import logging
|
||||
import math
|
||||
import re
|
||||
import time
|
||||
from datetime import datetime, timezone
|
||||
|
|
@ -38,7 +37,6 @@ from camera_config import (
|
|||
CAMERA_SOURCE_URLS, CAMERA_REQUEST_DELAY, CAMERA_MAX_PER_SOURCE,
|
||||
NOMINATIM_URL, NOMINATIM_MIN_INTERVAL, USER_AGENT,
|
||||
SNAPSHOT_CACHE_DIR, SNAPSHOT_TTL_SECONDS, SNAPSHOT_TIMEOUT,
|
||||
UDOT_IBI_URL, UDOT_IBI_BASE, UDOT_IBI_PAGE_SIZE, UDOT_IBI_MAX_PAGES,
|
||||
)
|
||||
from camera_models import cameras
|
||||
from database import async_session
|
||||
|
|
@ -116,15 +114,6 @@ class RateLimitedClient:
|
|||
self._last[host] = time.monotonic()
|
||||
return await self.client.get(url, **kw)
|
||||
|
||||
async def post(self, url: str, **kw) -> httpx.Response:
|
||||
host = urlparse(url).netloc
|
||||
now = time.monotonic()
|
||||
wait = self._last.get(host, 0.0) + self._delay - now
|
||||
if wait > 0:
|
||||
await asyncio.sleep(wait)
|
||||
self._last[host] = time.monotonic()
|
||||
return await self.client.post(url, **kw)
|
||||
|
||||
async def aclose(self):
|
||||
await self.client.aclose()
|
||||
|
||||
|
|
@ -390,201 +379,6 @@ def parse_caltrans_json(text: str, source_name: str) -> list[dict]:
|
|||
return out
|
||||
|
||||
|
||||
# ── UDOT IBI 511 ──────────────────────────────────────────────────────────
|
||||
# Utah bbox (lat 36.9–42.1, lon -114.2–-108.9). WKT is `POINT (lng lat)`.
|
||||
_UDOT_IBI_MIN_LAT, _UDOT_IBI_MAX_LAT = 36.9, 42.1
|
||||
_UDOT_IBI_MIN_LON, _UDOT_IBI_MAX_LON = -114.2, -108.9
|
||||
_UDOT_WKT_POINT_RE = re.compile(
|
||||
r"POINT\s*\(\s*(-?\d+(?:\.\d+)?)\s+(-?\d+(?:\.\d+)?)\s*\)", re.I,
|
||||
)
|
||||
|
||||
|
||||
def parse_udot_ibi_page(text: str, source_name: str = "udot") -> list[dict]:
|
||||
"""Parse one UDOT IBI 511 DataTables camera page (`{"data": [...]}`).
|
||||
|
||||
Skips rows whose first image is `blocked` or `disabled`, and drops any
|
||||
point outside the Utah bbox. The `/map/Cctv/{id}` URL is a stable identity
|
||||
(always serves the latest frame), so it is stored as both source_url and
|
||||
snapshot_url — we never scrape frames ourselves.
|
||||
"""
|
||||
try:
|
||||
payload = json.loads(text)
|
||||
except (json.JSONDecodeError, ValueError):
|
||||
return []
|
||||
rows = payload.get("data") if isinstance(payload, dict) else None
|
||||
if not isinstance(rows, list):
|
||||
return []
|
||||
out: list[dict] = []
|
||||
for row in rows:
|
||||
if not isinstance(row, dict):
|
||||
continue
|
||||
cam_id = row.get("id")
|
||||
images = row.get("images") or []
|
||||
if cam_id is None or not images:
|
||||
continue
|
||||
img = images[0] or {}
|
||||
if img.get("blocked") or img.get("disabled"):
|
||||
continue
|
||||
lon = lat = None
|
||||
try:
|
||||
wkt = (row.get("latLng") or {}).get("geography") or {}
|
||||
wkt = wkt.get("wellKnownText") or ""
|
||||
m = _UDOT_WKT_POINT_RE.match(str(wkt).strip())
|
||||
if m:
|
||||
lon, lat = float(m.group(1)), float(m.group(2))
|
||||
except (AttributeError, TypeError, ValueError):
|
||||
lon = lat = None
|
||||
if lat is None or lon is None:
|
||||
continue
|
||||
if not (_UDOT_IBI_MIN_LAT <= lat <= _UDOT_IBI_MAX_LAT
|
||||
and _UDOT_IBI_MIN_LON <= lon <= _UDOT_IBI_MAX_LON):
|
||||
continue
|
||||
snap = f"{UDOT_IBI_BASE}/map/Cctv/{cam_id}"
|
||||
roadway, direction, location = (
|
||||
row.get("roadway"), row.get("direction"), row.get("location"),
|
||||
)
|
||||
name = ", ".join(
|
||||
str(b) for b in (roadway, direction, location)
|
||||
if b and str(b).strip() and str(b).strip().lower() != "unknown"
|
||||
) or None
|
||||
out.append({
|
||||
"source_url": snap,
|
||||
"snapshot_url": snap,
|
||||
"discovery_source": source_name,
|
||||
"location_lat": lat,
|
||||
"location_lon": lon,
|
||||
"location_name": name,
|
||||
"vendor": "UDOT",
|
||||
"device_type": "http",
|
||||
"raw": {
|
||||
"udot_id": cam_id,
|
||||
"agency": row.get("source"),
|
||||
"source_id": row.get("sourceId"),
|
||||
"roadway": roadway,
|
||||
"direction": direction,
|
||||
},
|
||||
})
|
||||
return out
|
||||
|
||||
|
||||
# Oregon DOT TripCheck inventory bounding box (approx state extent).
|
||||
ODOT_BBOX = (41.9, 46.3, -124.6, -116.4) # lat_min, lat_max, lon_min, lon_max
|
||||
|
||||
|
||||
def parse_odot_json(text: str, source_name: str) -> list[dict]:
|
||||
"""Parse Oregon DOT TripCheck cctvinventory Esri-style JSON.
|
||||
|
||||
Store the JPEG still as snapshot_url (map thumbs); never RTSP. Keep only
|
||||
rows with finite coordinates inside Oregon and a usable filename.
|
||||
"""
|
||||
try:
|
||||
payload = json.loads(text)
|
||||
except (json.JSONDecodeError, ValueError):
|
||||
return []
|
||||
lat_min, lat_max, lon_min, lon_max = ODOT_BBOX
|
||||
out: list[dict] = []
|
||||
for feat in payload.get("features") or []:
|
||||
attrs = (feat or {}).get("attributes") or {}
|
||||
filename = (attrs.get("filename") or "").strip()
|
||||
if not filename:
|
||||
continue
|
||||
try:
|
||||
lat = float(attrs.get("latitude"))
|
||||
lon = float(attrs.get("longitude"))
|
||||
except (TypeError, ValueError):
|
||||
continue
|
||||
if not (math.isfinite(lat) and math.isfinite(lon)):
|
||||
continue
|
||||
if not (lat_min <= lat <= lat_max and lon_min <= lon <= lon_max):
|
||||
continue
|
||||
jpeg = f"https://tripcheck.com/RoadCams/cams/{filename}"
|
||||
title = (attrs.get("title") or "").strip()
|
||||
out.append({
|
||||
"source_url": jpeg,
|
||||
"snapshot_url": jpeg,
|
||||
"discovery_source": "odot",
|
||||
"location_lat": lat,
|
||||
"location_lon": lon,
|
||||
"location_name": title or None,
|
||||
"vendor": "ODOT",
|
||||
"device_type": "http",
|
||||
})
|
||||
return out
|
||||
|
||||
|
||||
# MDOT MiDrive field extractors (fields carry rendered HTML).
|
||||
_MDOT_LAT_RE = re.compile(r"lat=(-?\d+(?:\.\d+)?)", re.I)
|
||||
_MDOT_LON_RE = re.compile(r"lon=(-?\d+(?:\.\d+)?)", re.I)
|
||||
_MDOT_ID_RE = re.compile(r"[?&]id=(\d+)", re.I)
|
||||
_MDOT_IMG_RE = re.compile(r'<img[^>]+src=["\']([^"\']+)["\']', re.I)
|
||||
|
||||
# Michigan bbox (docs/osiris-ideas.md §3.2): lat 41.6–48.3, lon -90.5–-82.1.
|
||||
MDOT_LAT_RANGE = (41.6, 48.3)
|
||||
MDOT_LON_RANGE = (-90.5, -82.1)
|
||||
|
||||
|
||||
def parse_mdot_json(text: str, source_name: str) -> list[dict]:
|
||||
"""Parse MDOT MiDrive `camera/list` JSON (fields carry rendered HTML).
|
||||
|
||||
Coordinates and the stable id live in the `county` field's map link
|
||||
(`/MiDrive/map?...lat=&lon=&id=`); the `image` field carries an `<img>`
|
||||
whose src is the JPEG still. Out-of-bbox and coord-less rows are dropped.
|
||||
"""
|
||||
try:
|
||||
payload = json.loads(text)
|
||||
except (json.JSONDecodeError, ValueError):
|
||||
return []
|
||||
if not isinstance(payload, list):
|
||||
return []
|
||||
out: list[dict] = []
|
||||
for row in payload:
|
||||
if not isinstance(row, dict):
|
||||
continue
|
||||
county_html = row.get("county") or ""
|
||||
m_lat = _MDOT_LAT_RE.search(county_html)
|
||||
m_lon = _MDOT_LON_RE.search(county_html)
|
||||
m_id = _MDOT_ID_RE.search(county_html)
|
||||
if not (m_lat and m_lon and m_id):
|
||||
continue # missing coordinates / stable id → drop
|
||||
try:
|
||||
lat = float(m_lat.group(1))
|
||||
lon = float(m_lon.group(1))
|
||||
except ValueError:
|
||||
continue
|
||||
if not (MDOT_LAT_RANGE[0] <= lat <= MDOT_LAT_RANGE[1]
|
||||
and MDOT_LON_RANGE[0] <= lon <= MDOT_LON_RANGE[1]):
|
||||
continue # out of Michigan bbox → drop
|
||||
img_m = _MDOT_IMG_RE.search(row.get("image") or "")
|
||||
if not img_m:
|
||||
continue
|
||||
snap = img_m.group(1).strip()
|
||||
low = snap.lower()
|
||||
if not (low.startswith("http://") or low.startswith("https://")):
|
||||
continue
|
||||
if low.startswith("rtsp"):
|
||||
continue
|
||||
cam_id = m_id.group(1)
|
||||
route = (row.get("route") or "").strip()
|
||||
loc = (row.get("location") or "").strip().lstrip("@").strip()
|
||||
county_name = county_html.split("<a", 1)[0].strip()
|
||||
bits = [
|
||||
f"{route} @ {loc}" if (route and loc) else (route or loc or None),
|
||||
county_name or None,
|
||||
]
|
||||
name = ", ".join(b for b in bits if b) or None
|
||||
out.append({
|
||||
"source_url": f"https://mdotjboss.state.mi.us/MiDrive/camera/{cam_id}",
|
||||
"snapshot_url": snap,
|
||||
"discovery_source": "mdot",
|
||||
"location_lat": lat,
|
||||
"location_lon": lon,
|
||||
"location_name": name,
|
||||
"vendor": "MDOT",
|
||||
"device_type": "http",
|
||||
})
|
||||
return out
|
||||
|
||||
|
||||
def parse_live_streams_geojson(text: str, source_name: str) -> list[dict]:
|
||||
"""Parse willytop8/Live-Environment-Streams GeoJSON.
|
||||
|
||||
|
|
@ -696,10 +490,6 @@ async def scrape_source(client: RateLimitedClient, geo: Geocoder,
|
|||
body = resp.text
|
||||
if "cwwp2.dot.ca.gov" in src_url or "cctvStatus" in src_url:
|
||||
cams = parse_caltrans_json(body, name)
|
||||
elif "cctvinventory" in src_url or "tripcheck.com" in src_url:
|
||||
cams = parse_odot_json(body, name)
|
||||
elif "mdotjboss.state.mi.us" in src_url or "/MiDrive/camera/list" in src_url:
|
||||
cams = parse_mdot_json(body, name)
|
||||
elif ("getCameraDataByLoc" in src_url
|
||||
or ("json" in ctype and '"locs"' in body[:4000] and '"cams"' in body[:8000])):
|
||||
cams = parse_alertwest_json(body, name)
|
||||
|
|
@ -759,54 +549,6 @@ async def scrape_source(client: RateLimitedClient, geo: Geocoder,
|
|||
return out
|
||||
|
||||
|
||||
# ── UDOT IBI 511 paginated fetcher ────────────────────────────────────────
|
||||
|
||||
async def scrape_udot_ibi(client: RateLimitedClient) -> list[dict]:
|
||||
"""Page through the UDOT IBI 511 DataTables endpoint and normalize.
|
||||
|
||||
POSTs `start`/`length` form fields (server caps at 100 rows/page), walking
|
||||
pages until `recordsTotal` is exhausted or UDOT_IBI_MAX_PAGES is hit.
|
||||
"""
|
||||
out: list[dict] = []
|
||||
seen: set[str] = set()
|
||||
start = 0
|
||||
for _ in range(UDOT_IBI_MAX_PAGES):
|
||||
try:
|
||||
resp = await client.post(
|
||||
UDOT_IBI_URL,
|
||||
data={
|
||||
"start": str(start),
|
||||
"length": str(UDOT_IBI_PAGE_SIZE),
|
||||
"lang": "en-US",
|
||||
},
|
||||
headers={"X-Requested-With": "XMLHttpRequest"},
|
||||
)
|
||||
resp.raise_for_status()
|
||||
body = resp.text
|
||||
except Exception: # noqa: BLE001
|
||||
logger.exception("failed to fetch UDOT IBI page start=%d", start)
|
||||
break
|
||||
try:
|
||||
payload = json.loads(body)
|
||||
except ValueError:
|
||||
logger.warning("UDOT IBI non-JSON response at start=%d", start)
|
||||
break
|
||||
total = int(payload.get("recordsTotal") or 0)
|
||||
rows = payload.get("data") or []
|
||||
if not isinstance(rows, list) or not rows:
|
||||
break
|
||||
for cam in parse_udot_ibi_page(body, "udot"):
|
||||
if cam["source_url"] in seen:
|
||||
continue
|
||||
seen.add(cam["source_url"])
|
||||
out.append(cam)
|
||||
if start + len(rows) >= total:
|
||||
break
|
||||
start += len(rows)
|
||||
logger.info("UDOT IBI yielded %d cameras", len(out))
|
||||
return out
|
||||
|
||||
|
||||
# ── Persistence ────────────────────────────────────────────────────────────
|
||||
|
||||
async def upsert_cameras(cams: list[dict]) -> int:
|
||||
|
|
@ -858,7 +600,6 @@ async def run_cycle() -> int:
|
|||
try:
|
||||
results = await asyncio.gather(
|
||||
*(scrape_source(client, geo, s) for s in CAMERA_SOURCE_URLS),
|
||||
scrape_udot_ibi(client),
|
||||
return_exceptions=True,
|
||||
)
|
||||
all_cams: list[dict] = []
|
||||
|
|
|
|||
|
|
@ -1,62 +0,0 @@
|
|||
"""Static chokepoint preset catalog — one-tap fly-to targets for the map.
|
||||
|
||||
Pure data, no upstream calls and no VesselAPI quota spend. ``vesselapi`` is
|
||||
``True`` only for Hormuz (the single box the VesselAPI poller already covers);
|
||||
every other strait is AISStream-only until a human later spends quota. Never
|
||||
call VesselAPI from here.
|
||||
|
||||
Bounding boxes are ``minlat,minlon,maxlat,maxlon`` (VesselAPI order) and each
|
||||
stays within the ``|dLat|+|dLon| <= 4`` span rule enforced by
|
||||
``vesselapi.validate_bbox_span``.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
# id → preset. ``center`` is ``[lat, lon]`` for Leaflet ``setView``.
|
||||
_CHOKEPOINTS: tuple[dict, ...] = (
|
||||
{
|
||||
"id": "hormuz",
|
||||
"title": "Strait of Hormuz",
|
||||
"bbox": "25.5,55.4,27.3,57.2",
|
||||
"center": [26.4, 56.5],
|
||||
"zoom": 9,
|
||||
"vesselapi": True,
|
||||
},
|
||||
{
|
||||
"id": "bab_el_mandeb",
|
||||
"title": "Bab el-Mandeb",
|
||||
"bbox": "12.0,42.8,13.5,44.3",
|
||||
"center": [12.7, 43.4],
|
||||
"zoom": 9,
|
||||
"vesselapi": False,
|
||||
},
|
||||
{
|
||||
"id": "suez",
|
||||
"title": "Suez / N. Red Sea",
|
||||
"bbox": "29.5,32.0,31.0,33.5",
|
||||
"center": [30.0,32.5],
|
||||
"zoom": 9,
|
||||
"vesselapi": False,
|
||||
},
|
||||
{
|
||||
"id": "malacca",
|
||||
"title": "Malacca / Singapore",
|
||||
"bbox": "1.0,103.0,2.5,104.5",
|
||||
"center": [1.3, 103.8],
|
||||
"zoom": 9,
|
||||
"vesselapi": False,
|
||||
},
|
||||
{
|
||||
"id": "taiwan",
|
||||
"title": "Taiwan Strait",
|
||||
"bbox": "23.5,119.0,25.0,120.5",
|
||||
"center": [24.2, 119.8],
|
||||
"zoom": 9,
|
||||
"vesselapi": False,
|
||||
},
|
||||
)
|
||||
|
||||
|
||||
def chokepoints() -> list[dict]:
|
||||
"""Return a fresh copy of the catalog (callers must not mutate the source)."""
|
||||
return [dict(p) for p in _CHOKEPOINTS]
|
||||
|
|
@ -71,16 +71,6 @@ FIRMS_DATASETS = [d.strip() for d in _FIRMS_DATASETS_RAW.split(",") if d.strip()
|
|||
OSINT_USER_AGENT = os.getenv(
|
||||
"OSINT_USER_AGENT", "osint-dashboard/1.0 (self-hosted; lancewalters94@gmail.com)"
|
||||
)
|
||||
# Nominatim reverse (GET /api/place). Camera scraper has its own copy in camera_config.
|
||||
NOMINATIM_URL = os.getenv("NOMINATIM_URL", "https://nominatim.openstreetmap.org")
|
||||
NOMINATIM_MIN_INTERVAL = float(os.getenv("NOMINATIM_MIN_INTERVAL", "1.0"))
|
||||
|
||||
# Self-hosted TiTiler (warps Sentinel-1 signed COGs into XYZ tiles on the Pi).
|
||||
# TITILER_PUBLIC_BASE is the same-origin path prefix the browser hits through
|
||||
# the osint.rpi.local nginx vhost (`location /titiler/` → 127.0.0.1:8001).
|
||||
# TITILER_INTERNAL_URL is the compose-DNS address, used only for health checks.
|
||||
TITILER_PUBLIC_BASE = os.getenv("TITILER_PUBLIC_BASE", "/titiler").rstrip("/")
|
||||
TITILER_INTERNAL_URL = os.getenv("TITILER_INTERNAL_URL", "http://titiler:8000")
|
||||
|
||||
# AISStream (server-side WebSocket only). Idle when unset.
|
||||
AISSTREAM_API_KEY = os.getenv("AISSTREAM_API_KEY", "")
|
||||
|
|
@ -91,19 +81,3 @@ AISSTREAM_BBOX = os.getenv("AISSTREAM_BBOX", "24,-125,50,-66")
|
|||
# without the ingest profile). Set 0 if the ingester owns the only connection.
|
||||
AISSTREAM_IN_APP = os.getenv("AISSTREAM_IN_APP", "1").lower() in ("1", "true", "yes")
|
||||
AISSTREAM_IN_INGEST = os.getenv("AISSTREAM_IN_INGEST", "0").lower() in ("1", "true", "yes")
|
||||
|
||||
# VesselAPI (quota-capped REST AIS poller — free tier 150 calls/mo).
|
||||
# AISStream keeps US coasts; VesselAPI fills the Middle East blind spot. The
|
||||
# poller idles when VESSELAPI_API_KEY is unset (never from GET /api/vessels).
|
||||
VESSELAPI_API_KEY = os.getenv("VESSELAPI_API_KEY", "")
|
||||
# Bounding box(es) as minlat,minlon,maxlat,maxlon — note lat/lon order (same as
|
||||
# AISSTREAM_BBOX). Semicolon-separated for multiple boxes. Default: Strait of
|
||||
# Hormuz (|dLat|+|dLon| = 3.6 ≤ 4° span cap). VesselAPI 400s any box over 4°.
|
||||
VESSELAPI_BBOX = os.getenv("VESSELAPI_BBOX", "25.5,55.4,27.3,57.2")
|
||||
# Poll cadence in seconds. 17280 = 4.8h → 5 polls/day (150/mo free tier).
|
||||
VESSELAPI_INTERVAL = int(os.getenv("VESSELAPI_INTERVAL", "17280"))
|
||||
# Local hard cap on successful 2xx calls per UTC day (persisted in Postgres).
|
||||
VESSELAPI_MAX_CALLS_PER_DAY = int(os.getenv("VESSELAPI_MAX_CALLS_PER_DAY", "5"))
|
||||
# Run the VesselAPI poller inside the dashboard process (default on, like AIS).
|
||||
VESSELAPI_IN_APP = os.getenv("VESSELAPI_IN_APP", "1").lower() in ("1", "true", "yes")
|
||||
VESSELAPI_IN_INGEST = os.getenv("VESSELAPI_IN_INGEST", "0").lower() in ("1", "true", "yes")
|
||||
|
|
|
|||
168
app/conflicts.py
168
app/conflicts.py
|
|
@ -1,168 +0,0 @@
|
|||
"""Curated OSINT conflict-zone catalog + point-in-bbox event counting.
|
||||
|
||||
A static, human-curated list of active conflict theatres (war / high /
|
||||
elevated). Purely descriptive — this is a catalog, not a live feed and not a
|
||||
scrape of LiveUAMap or any other source. Severity and descriptions are
|
||||
editorial judgement kept short and factual.
|
||||
|
||||
Each zone carries an internal ``bbox`` (``min_lat, min_lon, max_lat, max_lon``)
|
||||
used only to count pre-existing geocoded news/GDELT/``/api/news/map`` rows that
|
||||
fall inside it. The bbox is not part of the API response; callers get the
|
||||
``eventCount`` roll-up instead.
|
||||
|
||||
Never call an upstream API from here — event counts come from rows already in
|
||||
the local database (``events`` with geocoords + ``news_items`` map pins).
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from datetime import datetime
|
||||
|
||||
# id → zone. ``lat``/``lon`` is the fly-to anchor; ``bbox`` is the internal
|
||||
# count window in ``min_lat, min_lon, max_lat, max_lon`` order.
|
||||
_ZONES: tuple[dict, ...] = (
|
||||
{
|
||||
"id": "ukraine",
|
||||
"label": "Ukraine",
|
||||
"severity": "war",
|
||||
"lat": 48.5,
|
||||
"lon": 31.0,
|
||||
"description": "Full-scale Russian invasion since 2022; active front lines in the east and south.",
|
||||
"bbox": (44.3, 22.1, 52.4, 40.2),
|
||||
},
|
||||
{
|
||||
"id": "gaza",
|
||||
"label": "Gaza",
|
||||
"severity": "war",
|
||||
"lat": 31.4,
|
||||
"lon": 34.4,
|
||||
"description": "Israel–Hamas war; sustained fighting and a severe humanitarian crisis in the Gaza Strip.",
|
||||
"bbox": (31.0, 34.1, 31.8, 34.7),
|
||||
},
|
||||
{
|
||||
"id": "sudan",
|
||||
"label": "Sudan",
|
||||
"severity": "war",
|
||||
"lat": 15.5,
|
||||
"lon": 30.0,
|
||||
"description": "Civil war between the SAF and RSF since 2023, with mass displacement across the country.",
|
||||
"bbox": (8.7, 21.8, 22.0, 38.6),
|
||||
},
|
||||
{
|
||||
"id": "myanmar",
|
||||
"label": "Myanmar",
|
||||
"severity": "war",
|
||||
"lat": 21.5,
|
||||
"lon": 96.0,
|
||||
"description": "Post-2021 coup conflict pitting the junta against resistance and ethnic armed groups.",
|
||||
"bbox": (9.5, 92.2, 28.5, 101.2),
|
||||
},
|
||||
{
|
||||
"id": "drc",
|
||||
"label": "DR Congo",
|
||||
"severity": "war",
|
||||
"lat": -1.5,
|
||||
"lon": 28.0,
|
||||
"description": "Eastern DRC conflict involving M23 and other armed groups; heavy displacement around Goma.",
|
||||
"bbox": (-5.0, 26.0, 3.0, 31.0),
|
||||
},
|
||||
{
|
||||
"id": "yemen",
|
||||
"label": "Yemen",
|
||||
"severity": "war",
|
||||
"lat": 15.5,
|
||||
"lon": 47.5,
|
||||
"description": "Protracted Houthi–government/coalition war with one of the world's worst humanitarian emergencies.",
|
||||
"bbox": (12.6, 42.5, 19.0, 54.0),
|
||||
},
|
||||
{
|
||||
"id": "syria",
|
||||
"label": "Syria",
|
||||
"severity": "war",
|
||||
"lat": 34.5,
|
||||
"lon": 38.5,
|
||||
"description": "Multi-sided civil war; government, opposition, and external actors continue to engage.",
|
||||
"bbox": (32.3, 35.7, 37.3, 42.4),
|
||||
},
|
||||
{
|
||||
"id": "lebanon",
|
||||
"label": "Lebanon",
|
||||
"severity": "high",
|
||||
"lat": 33.9,
|
||||
"lon": 35.9,
|
||||
"description": "Israel–Hezbollah hostilities with periodic escalation along the southern border.",
|
||||
"bbox": (33.0, 35.0, 34.7, 36.6),
|
||||
},
|
||||
{
|
||||
"id": "sahel",
|
||||
"label": "Sahel",
|
||||
"severity": "high",
|
||||
"lat": 14.5,
|
||||
"lon": 0.0,
|
||||
"description": "Jihadist insurgencies across Mali, Burkina Faso, and Niger destabilising the central Sahel.",
|
||||
"bbox": (10.0, -10.0, 20.0, 12.0),
|
||||
},
|
||||
{
|
||||
"id": "somalia",
|
||||
"label": "Somalia",
|
||||
"severity": "high",
|
||||
"lat": 6.0,
|
||||
"lon": 45.0,
|
||||
"description": "Al-Shabaab insurgency against the federal government and security forces.",
|
||||
"bbox": (-2.0, 41.0, 12.0, 51.5),
|
||||
},
|
||||
{
|
||||
"id": "red_sea",
|
||||
"label": "Red Sea",
|
||||
"severity": "high",
|
||||
"lat": 18.0,
|
||||
"lon": 40.0,
|
||||
"description": "Houthi attacks on commercial shipping transiting the Red Sea corridor.",
|
||||
"bbox": (12.0, 34.0, 22.0, 44.0),
|
||||
},
|
||||
{
|
||||
"id": "taiwan_strait",
|
||||
"label": "Taiwan Strait",
|
||||
"severity": "elevated",
|
||||
"lat": 24.5,
|
||||
"lon": 119.5,
|
||||
"description": "Heightened military standoff between China and Taiwan, including deterrence patrols.",
|
||||
"bbox": (21.9, 117.0, 26.5, 122.0),
|
||||
},
|
||||
{
|
||||
"id": "korean_dmz",
|
||||
"label": "Korean DMZ",
|
||||
"severity": "elevated",
|
||||
"lat": 38.3,
|
||||
"lon": 127.0,
|
||||
"description": "Heavily fortified inter-Korean border with periodic tensions and military drills.",
|
||||
"bbox": (37.5, 126.0, 39.0, 128.5),
|
||||
},
|
||||
)
|
||||
|
||||
SEVERITIES: frozenset[str] = frozenset({"war", "high", "elevated"})
|
||||
|
||||
|
||||
def conflict_zones() -> list[dict]:
|
||||
"""Return a fresh shallow copy of the catalog (callers must not mutate)."""
|
||||
return [dict(z) for z in _ZONES]
|
||||
|
||||
|
||||
def zone_event_stats(
|
||||
points: list[tuple[float, float, datetime | None]],
|
||||
bbox: tuple[float, float, float, float],
|
||||
) -> tuple[int, datetime | None]:
|
||||
"""Count points inside ``bbox`` and return (count, latest timestamp).
|
||||
|
||||
``points`` is an iterable of ``(lat, lon, ts)``; ``ts`` may be ``None``.
|
||||
``bbox`` is ``(min_lat, min_lon, max_lat, max_lon)``.
|
||||
"""
|
||||
min_lat, min_lon, max_lat, max_lon = bbox
|
||||
count = 0
|
||||
latest: datetime | None = None
|
||||
for lat, lon, ts in points:
|
||||
if min_lat <= lat <= max_lat and min_lon <= lon <= max_lon:
|
||||
count += 1
|
||||
if ts is not None and (latest is None or ts > latest):
|
||||
latest = ts
|
||||
return count, latest
|
||||
|
|
@ -4,9 +4,7 @@ set -e
|
|||
|
||||
cd /app
|
||||
echo "[entrypoint] running migrations..."
|
||||
# `heads` (plural): parallel PRs can fork the chain (005_phase2 + 005_news_items).
|
||||
# `upgrade head` then exits 255 and crash-loops the container.
|
||||
alembic upgrade heads
|
||||
alembic upgrade head
|
||||
|
||||
echo "[entrypoint] starting uvicorn..."
|
||||
exec uvicorn app.main:app --host 0.0.0.0 --port 8000
|
||||
|
|
|
|||
|
|
@ -1,186 +0,0 @@
|
|||
"""Flag firefighting ADS-B aircraft within 20 miles of an active fire."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from datetime import datetime, timedelta, timezone
|
||||
from uuid import uuid4
|
||||
|
||||
from sqlalchemy import text
|
||||
|
||||
from database import async_session
|
||||
from live_layers import _haversine_km
|
||||
|
||||
|
||||
RADIUS_MILES = 20.0
|
||||
RADIUS_KM = RADIUS_MILES * 1.609344
|
||||
_recent: dict[tuple[str, str], datetime] = {}
|
||||
_COOLDOWN = timedelta(minutes=5)
|
||||
|
||||
# ICAO type designators commonly used on wildfire air tankers, scoopers,
|
||||
# helitack, and air-attack platforms. Uppercase; compared case-insensitively.
|
||||
FIREFIGHTER_ICAO = frozenset({
|
||||
"AT802", "AT8T", "AT8P", "AT8B",
|
||||
"C130", "C30J", "C130J",
|
||||
"DC10", "MD10", "MD11", "MD87",
|
||||
"B737", "B38M",
|
||||
"CL2T", "CL215", "CL415", "CL5T",
|
||||
"S64", "SK64",
|
||||
"UH1", "UH1Y", "UH60", "H60", "S70",
|
||||
"B412", "B212", "B205",
|
||||
"AS50", "AS350", "A119", "A109",
|
||||
"B350", "BE20",
|
||||
"OV10",
|
||||
"RJ85", "RJ1H", "B461", "B462", "B463",
|
||||
"PC12",
|
||||
"TBM7", "TBM8", "TBM9",
|
||||
"C208",
|
||||
"DH8D", "Q400",
|
||||
})
|
||||
|
||||
|
||||
def _icao(marker: dict) -> str:
|
||||
extra = marker.get("extra") or {}
|
||||
return str(extra.get("type") or extra.get("t") or "").strip().upper()
|
||||
|
||||
|
||||
def is_firefighter(marker: dict) -> bool:
|
||||
return _icao(marker) in FIREFIGHTER_ICAO
|
||||
|
||||
|
||||
def _haversine_mi(lat1: float, lon1: float, lat2: float, lon2: float) -> float:
|
||||
return _haversine_km(lat1, lon1, lat2, lon2) / 1.609344
|
||||
|
||||
|
||||
def correlate_aircraft_to_fires(
|
||||
fires: list[dict],
|
||||
aircraft: list[dict],
|
||||
radius_mi: float = RADIUS_MILES,
|
||||
) -> list[dict]:
|
||||
hits: list[dict] = []
|
||||
for fire in fires:
|
||||
flat, flon = fire.get("lat"), fire.get("lon")
|
||||
if flat is None or flon is None:
|
||||
continue
|
||||
fid = str(fire.get("id") or fire.get("label") or "fire")
|
||||
for ac in aircraft:
|
||||
if not is_firefighter(ac):
|
||||
continue
|
||||
alat, alon = ac.get("lat"), ac.get("lon")
|
||||
if alat is None or alon is None:
|
||||
continue
|
||||
dist = _haversine_mi(float(flat), float(flon), float(alat), float(alon))
|
||||
if dist > radius_mi:
|
||||
continue
|
||||
hits.append({
|
||||
"fire_id": fid,
|
||||
"fire_lat": float(flat),
|
||||
"fire_lon": float(flon),
|
||||
"aircraft_hex": str(ac.get("id")),
|
||||
"aircraft_type": _icao(ac),
|
||||
"aircraft_lat": float(alat),
|
||||
"aircraft_lon": float(alon),
|
||||
"distance_mi": round(dist, 2),
|
||||
"label": ac.get("label") or ac.get("id"),
|
||||
})
|
||||
return hits
|
||||
|
||||
|
||||
async def persist_hits(hits: list[dict], seen_at: datetime | None = None) -> int:
|
||||
if not hits:
|
||||
return 0
|
||||
ts = seen_at or datetime.now(timezone.utc)
|
||||
n = 0
|
||||
async with async_session() as session:
|
||||
for h in hits:
|
||||
try:
|
||||
await session.execute(
|
||||
text(
|
||||
"""
|
||||
INSERT INTO fire_aircraft_hits
|
||||
(id, fire_id, fire_lat, fire_lon,
|
||||
aircraft_hex, aircraft_type, aircraft_lat, aircraft_lon,
|
||||
distance_mi, seen_at)
|
||||
VALUES (
|
||||
CAST(:id AS uuid), :fire_id, :fire_lat, :fire_lon,
|
||||
:aircraft_hex, :aircraft_type, :aircraft_lat, :aircraft_lon,
|
||||
:distance_mi, :seen_at
|
||||
)
|
||||
ON CONFLICT (fire_id, aircraft_hex, seen_at) DO NOTHING
|
||||
"""
|
||||
),
|
||||
{
|
||||
"id": str(uuid4()),
|
||||
"fire_id": h["fire_id"],
|
||||
"fire_lat": h["fire_lat"],
|
||||
"fire_lon": h["fire_lon"],
|
||||
"aircraft_hex": h["aircraft_hex"],
|
||||
"aircraft_type": h["aircraft_type"],
|
||||
"aircraft_lat": h["aircraft_lat"],
|
||||
"aircraft_lon": h["aircraft_lon"],
|
||||
"distance_mi": h["distance_mi"],
|
||||
"seen_at": ts.replace(microsecond=0),
|
||||
},
|
||||
)
|
||||
n += 1
|
||||
except Exception:
|
||||
continue
|
||||
try:
|
||||
await session.commit()
|
||||
except Exception:
|
||||
return 0
|
||||
return n
|
||||
|
||||
|
||||
async def recent_hits(limit: int = 200) -> list[dict]:
|
||||
try:
|
||||
async with async_session() as session:
|
||||
rows = (await session.execute(
|
||||
text(
|
||||
"""
|
||||
SELECT fire_id, fire_lat, fire_lon,
|
||||
aircraft_hex, aircraft_type, aircraft_lat, aircraft_lon,
|
||||
distance_mi, seen_at
|
||||
FROM fire_aircraft_hits
|
||||
ORDER BY seen_at DESC
|
||||
LIMIT :limit
|
||||
"""
|
||||
),
|
||||
{"limit": limit},
|
||||
)).mappings().all()
|
||||
out = []
|
||||
for r in rows:
|
||||
item = dict(r)
|
||||
if item.get("seen_at") is not None:
|
||||
item["seen_at"] = item["seen_at"].isoformat()
|
||||
out.append(item)
|
||||
return out
|
||||
except Exception:
|
||||
return []
|
||||
|
||||
|
||||
async def correlate_and_notify(fires: list[dict], aircraft: list[dict]) -> list[dict]:
|
||||
hits = correlate_aircraft_to_fires(fires, aircraft)
|
||||
if not hits:
|
||||
return []
|
||||
now = datetime.now(timezone.utc)
|
||||
fresh = []
|
||||
for h in hits:
|
||||
key = (h["fire_id"], h["aircraft_hex"])
|
||||
prev = _recent.get(key)
|
||||
if prev is not None and now - prev < _COOLDOWN:
|
||||
continue
|
||||
_recent[key] = now
|
||||
fresh.append(h)
|
||||
if not fresh:
|
||||
return []
|
||||
await persist_hits(fresh, seen_at=now)
|
||||
from ws_manager import manager
|
||||
for h in fresh:
|
||||
body = {**h, "seen_at": now.isoformat()}
|
||||
await manager.publish_point(
|
||||
"fire_aircraft", body, lat=h["aircraft_lat"], lon=h["aircraft_lon"],
|
||||
)
|
||||
await manager.publish_point(
|
||||
"fire_aircraft", body, lat=h["fire_lat"], lon=h["fire_lon"],
|
||||
)
|
||||
return fresh
|
||||
|
|
@ -22,7 +22,6 @@ UTC date (YYYY-MM-DD).
|
|||
from __future__ import annotations
|
||||
|
||||
import csv
|
||||
import hashlib
|
||||
import io
|
||||
import json
|
||||
import logging
|
||||
|
|
@ -41,15 +40,9 @@ from config import (
|
|||
NATS_URL,
|
||||
)
|
||||
from keystore import get_api_key
|
||||
from upstream_cache import firms_cache
|
||||
|
||||
logger = logging.getLogger("osint.firms")
|
||||
|
||||
# In-process poll state: skip byte-identical CSVs, persist only new hotspots.
|
||||
# Survives the 15-minute loop; one full ON CONFLICT after process start.
|
||||
_csv_digest: dict[tuple, bytes] = {}
|
||||
_seen_ids: dict[tuple, set[int]] = {}
|
||||
|
||||
# ── FIRMS API ─────────────────────────────────────────────────────────────
|
||||
|
||||
FIRMS_AREA_CSV = (
|
||||
|
|
@ -92,34 +85,33 @@ def normalize_acq_time(acq_date: object, acq_time: object) -> datetime | None:
|
|||
return None
|
||||
|
||||
|
||||
def _hotspot_id(lat: float, lon: float, acq_iso: str, satellite: str) -> int:
|
||||
return hash((round(lat, 5), round(lon, 5), acq_iso, satellite))
|
||||
def parse_firms_csv(text: str) -> list[dict]:
|
||||
"""Parse a FIRMS area CSV payload into normalized fire messages.
|
||||
|
||||
|
||||
def parse_firms_csv_delta(
|
||||
text: str, skip_ids: set[int] | None = None,
|
||||
) -> tuple[list[dict], set[int]]:
|
||||
"""Parse FIRMS CSV; optionally drop hotspots already seen this process.
|
||||
|
||||
Returns (new_or_all_points, ids_for_every_valid_row). Streaming — does not
|
||||
materialize the raw CSV as a list of lists.
|
||||
Returns one dict per hotspot with the fields stored in the ``fires`` table
|
||||
(acq_time already combined into a UTC ISO timestamp). Rows that don't look
|
||||
like valid VIIRS detections are skipped rather than failing the whole poll.
|
||||
"""
|
||||
reader = csv.reader(io.StringIO(text))
|
||||
header = None
|
||||
for row in reader:
|
||||
rows = list(csv.reader(io.StringIO(text)))
|
||||
if not rows:
|
||||
return []
|
||||
|
||||
# Locate the real header row. FIRMS normally returns the CSV header first,
|
||||
# but occasionally prepends a legend/info line, so scan until we see the
|
||||
# canonical header.
|
||||
header_idx = 0
|
||||
for i, row in enumerate(rows):
|
||||
if row and row[0].strip().lower() == "latitude" and len(row) >= 4:
|
||||
header = [c.strip().lower() for c in row]
|
||||
header_idx = i
|
||||
break
|
||||
if not header or "latitude" not in header or "longitude" not in header:
|
||||
logger.warning(
|
||||
"FIRMS payload does not look like a hotspot CSV (first row: %r)",
|
||||
(header or [])[:6],
|
||||
)
|
||||
return [], set()
|
||||
header = [c.strip().lower() for c in rows[header_idx]]
|
||||
# Guard against a header that isn't actually the FIRMS one.
|
||||
if "latitude" not in header or "longitude" not in header:
|
||||
logger.warning("FIRMS payload does not look like a hotspot CSV (first row: %r)", header[:6])
|
||||
return []
|
||||
|
||||
points: list[dict] = []
|
||||
ids: set[int] = set()
|
||||
for row in reader:
|
||||
for row in rows[header_idx + 1:]:
|
||||
if len(row) < len(header):
|
||||
continue
|
||||
rec = dict(zip(header, row))
|
||||
|
|
@ -130,19 +122,13 @@ def parse_firms_csv_delta(
|
|||
acq_time = normalize_acq_time(rec.get("acq_date"), rec.get("acq_time"))
|
||||
if acq_time is None:
|
||||
continue
|
||||
sat = str(rec.get("satellite") or "").strip()
|
||||
acq_iso = acq_time.isoformat()
|
||||
hid = _hotspot_id(lat, lon, acq_iso, sat)
|
||||
ids.add(hid)
|
||||
if skip_ids is not None and hid in skip_ids:
|
||||
continue
|
||||
points.append({
|
||||
"latitude": lat,
|
||||
"longitude": lon,
|
||||
"brightness": _to_float(rec.get("bright_ti4")),
|
||||
"confidence": str(rec.get("confidence") or "").strip(),
|
||||
"acq_time": acq_iso,
|
||||
"satellite": sat,
|
||||
"acq_time": acq_time.isoformat(),
|
||||
"satellite": str(rec.get("satellite") or "").strip(),
|
||||
"instrument": str(rec.get("instrument") or "").strip(),
|
||||
"bright_ti5": _to_float(rec.get("bright_ti5")),
|
||||
"frp": _to_float(rec.get("frp")),
|
||||
|
|
@ -151,17 +137,6 @@ def parse_firms_csv_delta(
|
|||
"track": _to_float(rec.get("track")),
|
||||
"version": str(rec.get("version") or "").strip(),
|
||||
})
|
||||
return points, ids
|
||||
|
||||
|
||||
def parse_firms_csv(text: str) -> list[dict]:
|
||||
"""Parse a FIRMS area CSV payload into normalized fire messages.
|
||||
|
||||
Returns one dict per hotspot with the fields stored in the ``fires`` table
|
||||
(acq_time already combined into a UTC ISO timestamp). Rows that don't look
|
||||
like valid VIIRS detections are skipped rather than failing the whole poll.
|
||||
"""
|
||||
points, _ids = parse_firms_csv_delta(text)
|
||||
return points
|
||||
|
||||
|
||||
|
|
@ -186,16 +161,6 @@ async def publish_fire_batch(points: list[dict]) -> int:
|
|||
return len(points)
|
||||
|
||||
|
||||
async def persist_hotspots(points: list[dict]) -> int:
|
||||
"""Write a FIRMS poll to Postgres in one ON CONFLICT batch.
|
||||
|
||||
NATS-per-row was 93k commits + geofence/correlation per hotspot.
|
||||
"""
|
||||
from ingestor import ingest_fire_rows
|
||||
|
||||
return await ingest_fire_rows(points)
|
||||
|
||||
|
||||
async def ingest_fires(bbox: str | None = None) -> int:
|
||||
"""Fetch the FIRMS hotspot CSV for an area and publish it to NATS.
|
||||
|
||||
|
|
@ -220,16 +185,12 @@ async def ingest_fires(bbox: str | None = None) -> int:
|
|||
total_published = 0
|
||||
async with httpx.AsyncClient(timeout=FIRMS_TIMEOUT) as client:
|
||||
for dataset in datasets:
|
||||
cache_key = (dataset, area, FIRMS_DAYS)
|
||||
text = firms_cache.get(cache_key)
|
||||
if text is None:
|
||||
url = FIRMS_AREA_CSV.format(
|
||||
key=map_key, dataset=dataset, bbox=area, days=FIRMS_DAYS
|
||||
)
|
||||
resp = await client.get(url)
|
||||
resp.raise_for_status()
|
||||
text = resp.text
|
||||
firms_cache[cache_key] = text
|
||||
url = FIRMS_AREA_CSV.format(
|
||||
key=map_key, dataset=dataset, bbox=area, days=FIRMS_DAYS
|
||||
)
|
||||
resp = await client.get(url)
|
||||
resp.raise_for_status()
|
||||
text = resp.text
|
||||
# FIRMS returns HTTP 200 with a plain-text error for some failure modes
|
||||
# (bad key, invalid bbox); surface the first line for debuggability.
|
||||
if "latitude" not in text.lower()[:4096]:
|
||||
|
|
@ -239,18 +200,11 @@ async def ingest_fires(bbox: str | None = None) -> int:
|
|||
dataset, first_line,
|
||||
)
|
||||
continue
|
||||
poll_key = (dataset, area, FIRMS_DAYS)
|
||||
digest = hashlib.sha256(text.encode("utf-8", "surrogatepass")).digest()
|
||||
if _csv_digest.get(poll_key) == digest:
|
||||
logger.info("FIRMS %s CSV unchanged, skip parse/insert", dataset)
|
||||
continue
|
||||
points, ids = parse_firms_csv_delta(text, skip_ids=_seen_ids.get(poll_key))
|
||||
published = await persist_hotspots(points) if points else 0
|
||||
_csv_digest[poll_key] = digest
|
||||
_seen_ids[poll_key] = ids
|
||||
points = parse_firms_csv(text)
|
||||
published = await publish_fire_batch(points)
|
||||
total_published += published
|
||||
logger.info(
|
||||
"FIRMS: fetched %d hotspot(s) for bbox=%s (%s), published %d",
|
||||
len(ids), area, dataset, published,
|
||||
len(points), area, dataset, published,
|
||||
)
|
||||
return total_published
|
||||
|
|
|
|||
472
app/geofence.py
472
app/geofence.py
|
|
@ -1,472 +0,0 @@
|
|||
"""Geofences: GeoJSON polygons, ST_Intersects on ingest, WS alerts.
|
||||
|
||||
``/api/alerts`` is the dashboard entity/keyword table — geofence hits live
|
||||
in ``geofence_alerts`` and fan out as WS type ``geofence_alert``.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
from datetime import datetime, timedelta, timezone
|
||||
from typing import Any
|
||||
from uuid import uuid4
|
||||
|
||||
from sqlalchemy import text
|
||||
|
||||
from database import async_session
|
||||
|
||||
|
||||
def _rings_from_geojson(geojson: dict) -> list[list[list[float]]]:
|
||||
if not isinstance(geojson, dict):
|
||||
raise ValueError("geojson must be an object")
|
||||
gj = geojson
|
||||
if gj.get("type") == "Feature":
|
||||
gj = gj.get("geometry") or {}
|
||||
if gj.get("type") == "FeatureCollection":
|
||||
raise ValueError("FeatureCollection is not a single polygon")
|
||||
if gj.get("type") != "Polygon":
|
||||
raise ValueError("geojson must be a Polygon")
|
||||
coords = gj.get("coordinates")
|
||||
if not isinstance(coords, list) or not coords:
|
||||
raise ValueError("polygon has no rings")
|
||||
rings: list[list[list[float]]] = []
|
||||
for ring in coords:
|
||||
if not isinstance(ring, list) or len(ring) < 4:
|
||||
raise ValueError("polygon ring needs ≥4 positions (closed)")
|
||||
pts = []
|
||||
for pt in ring:
|
||||
if not isinstance(pt, (list, tuple)) or len(pt) < 2:
|
||||
raise ValueError("position must be [lon, lat]")
|
||||
pts.append([float(pt[0]), float(pt[1])])
|
||||
rings.append(pts)
|
||||
return rings
|
||||
|
||||
|
||||
def validate_polygon_geojson(geojson: dict) -> dict:
|
||||
"""Return a canonical Polygon GeoJSON or raise ValueError."""
|
||||
rings = _rings_from_geojson(geojson)
|
||||
return {"type": "Polygon", "coordinates": rings}
|
||||
|
||||
|
||||
def _ring_contains(lon: float, lat: float, ring: list[list[float]]) -> bool:
|
||||
"""Ray-cast even-odd rule. Ring is [lon, lat] positions."""
|
||||
inside = False
|
||||
n = len(ring)
|
||||
if n < 4:
|
||||
return False
|
||||
j = n - 1
|
||||
for i in range(n):
|
||||
xi, yi = ring[i][0], ring[i][1]
|
||||
xj, yj = ring[j][0], ring[j][1]
|
||||
intersects = ((yi > lat) != (yj > lat)) and (
|
||||
lon < (xj - xi) * (lat - yi) / ((yj - yi) or 1e-16) + xi
|
||||
)
|
||||
if intersects:
|
||||
inside = not inside
|
||||
j = i
|
||||
return inside
|
||||
|
||||
|
||||
def point_in_geojson(lon: float, lat: float, geojson: dict) -> bool:
|
||||
"""True if (lon, lat) is inside the outer ring and outside holes."""
|
||||
try:
|
||||
rings = _rings_from_geojson(geojson)
|
||||
except (ValueError, TypeError, KeyError):
|
||||
return False
|
||||
if not _ring_contains(lon, lat, rings[0]):
|
||||
return False
|
||||
for hole in rings[1:]:
|
||||
if _ring_contains(lon, lat, hole):
|
||||
return False
|
||||
return True
|
||||
|
||||
|
||||
def matching_geofences(lon: float, lat: float, fences: list[dict]) -> list[dict]:
|
||||
hits = []
|
||||
for fence in fences:
|
||||
if not fence.get("active", True):
|
||||
continue
|
||||
gj = fence.get("geojson") or {}
|
||||
if point_in_geojson(lon, lat, gj):
|
||||
hits.append(fence)
|
||||
return hits
|
||||
|
||||
|
||||
# In-process copy of active fences so ingest does not round-trip Postgres
|
||||
# on every AIS frame. CRUD endpoints refresh this list.
|
||||
_cache: list[dict] = []
|
||||
_recent_hits: dict[tuple[str, str], datetime] = {}
|
||||
_HIT_COOLDOWN = timedelta(minutes=5)
|
||||
|
||||
|
||||
async def refresh_cache() -> list[dict]:
|
||||
global _cache
|
||||
async with async_session() as session:
|
||||
rows = (await session.execute(text(
|
||||
"SELECT id::text, name, geojson, active FROM geofences"
|
||||
))).mappings().all()
|
||||
_cache = [
|
||||
{
|
||||
"id": r["id"],
|
||||
"name": r["name"],
|
||||
"geojson": r["geojson"] if isinstance(r["geojson"], dict)
|
||||
else json.loads(r["geojson"] or "{}"),
|
||||
"active": bool(r["active"]),
|
||||
}
|
||||
for r in rows
|
||||
]
|
||||
return _cache
|
||||
|
||||
|
||||
def cached_fences() -> list[dict]:
|
||||
return list(_cache)
|
||||
|
||||
|
||||
async def list_geofences() -> list[dict]:
|
||||
if not _cache:
|
||||
try:
|
||||
await refresh_cache()
|
||||
except Exception:
|
||||
return []
|
||||
return cached_fences()
|
||||
|
||||
|
||||
async def create_geofence(name: str, geojson: dict, active: bool = True) -> dict:
|
||||
polygon = validate_polygon_geojson(geojson)
|
||||
gid = str(uuid4())
|
||||
gj = json.dumps(polygon)
|
||||
async with async_session() as session:
|
||||
await session.execute(
|
||||
text(
|
||||
"""
|
||||
INSERT INTO geofences (id, name, geojson, geom, active)
|
||||
VALUES (
|
||||
:id, :name, CAST(:geojson AS jsonb),
|
||||
ST_SetSRID(ST_GeomFromGeoJSON(:geojson), 4326),
|
||||
:active
|
||||
)
|
||||
"""
|
||||
),
|
||||
{"id": gid, "name": name, "geojson": gj, "active": 1 if active else 0},
|
||||
)
|
||||
await session.commit()
|
||||
row = {"id": gid, "name": name, "geojson": polygon, "active": active}
|
||||
_cache.append(row)
|
||||
return row
|
||||
|
||||
|
||||
async def update_geofence(gid: str, *, name: str | None = None,
|
||||
geojson: dict | None = None,
|
||||
active: bool | None = None) -> dict | None:
|
||||
current = next((f for f in _cache if f["id"] == gid), None)
|
||||
if current is None:
|
||||
await refresh_cache()
|
||||
current = next((f for f in _cache if f["id"] == gid), None)
|
||||
if current is None:
|
||||
return None
|
||||
if name is not None:
|
||||
current["name"] = name
|
||||
if geojson is not None:
|
||||
current["geojson"] = validate_polygon_geojson(geojson)
|
||||
if active is not None:
|
||||
current["active"] = active
|
||||
gj = json.dumps(current["geojson"])
|
||||
async with async_session() as session:
|
||||
await session.execute(
|
||||
text(
|
||||
"""
|
||||
UPDATE geofences SET
|
||||
name = :name,
|
||||
geojson = CAST(:geojson AS jsonb),
|
||||
geom = ST_SetSRID(ST_GeomFromGeoJSON(:geojson), 4326),
|
||||
active = :active,
|
||||
updated_at = now()
|
||||
WHERE id = CAST(:id AS uuid)
|
||||
"""
|
||||
),
|
||||
{
|
||||
"id": gid,
|
||||
"name": current["name"],
|
||||
"geojson": gj,
|
||||
"active": 1 if current["active"] else 0,
|
||||
},
|
||||
)
|
||||
await session.commit()
|
||||
return current
|
||||
|
||||
|
||||
async def delete_geofence(gid: str) -> bool:
|
||||
async with async_session() as session:
|
||||
result = await session.execute(
|
||||
text("DELETE FROM geofences WHERE id = CAST(:id AS uuid)"),
|
||||
{"id": gid},
|
||||
)
|
||||
await session.commit()
|
||||
_cache[:] = [f for f in _cache if f["id"] != gid]
|
||||
return bool(result.rowcount)
|
||||
|
||||
|
||||
async def st_intersects(lon: float, lat: float) -> list[dict]:
|
||||
"""PostGIS ST_Intersects against active geofences.
|
||||
|
||||
Falls back to the in-memory GeoJSON test if the DB is unreachable so
|
||||
ingest never dies because a fence check failed.
|
||||
"""
|
||||
try:
|
||||
async with async_session() as session:
|
||||
rows = (await session.execute(
|
||||
text(
|
||||
"""
|
||||
SELECT id::text, name, geojson, active
|
||||
FROM geofences
|
||||
WHERE active = 1
|
||||
AND ST_Intersects(
|
||||
geom,
|
||||
ST_SetSRID(ST_MakePoint(:lon, :lat), 4326)
|
||||
)
|
||||
"""
|
||||
),
|
||||
{"lon": lon, "lat": lat},
|
||||
)).mappings().all()
|
||||
return [
|
||||
{
|
||||
"id": r["id"],
|
||||
"name": r["name"],
|
||||
"geojson": r["geojson"] if isinstance(r["geojson"], dict)
|
||||
else json.loads(r["geojson"] or "{}"),
|
||||
"active": True,
|
||||
}
|
||||
for r in rows
|
||||
]
|
||||
except Exception:
|
||||
return matching_geofences(lon, lat, cached_fences())
|
||||
|
||||
|
||||
async def record_and_notify(
|
||||
*,
|
||||
source_kind: str,
|
||||
entity_id: str,
|
||||
lat: float,
|
||||
lon: float,
|
||||
payload: dict[str, Any] | None = None,
|
||||
) -> int:
|
||||
"""Insert a geofence_alerts row per hit and WS-push to viewport clients.
|
||||
|
||||
PostGIS ST_Intersects is the source of truth. The in-process GeoJSON
|
||||
cache is not a reject filter — the FIRMS ingester never fills it.
|
||||
"""
|
||||
hits = await st_intersects(lon, lat)
|
||||
if not hits:
|
||||
return 0
|
||||
from ws_manager import manager
|
||||
|
||||
sent = 0
|
||||
now = datetime.now(timezone.utc)
|
||||
fresh = []
|
||||
for fence in hits:
|
||||
key = (str(fence["id"]), str(entity_id))
|
||||
prev = _recent_hits.get(key)
|
||||
if prev is not None and now - prev < _HIT_COOLDOWN:
|
||||
continue
|
||||
_recent_hits[key] = now
|
||||
fresh.append(fence)
|
||||
if not fresh:
|
||||
return 0
|
||||
hits = fresh
|
||||
async with async_session() as session:
|
||||
for fence in hits:
|
||||
aid = str(uuid4())
|
||||
body = {
|
||||
"id": aid,
|
||||
"geofence_id": fence["id"],
|
||||
"geofence_name": fence.get("name"),
|
||||
"source_kind": source_kind,
|
||||
"entity_id": str(entity_id),
|
||||
"lat": lat,
|
||||
"lon": lon,
|
||||
"payload": payload or {},
|
||||
"created_at": now.isoformat(),
|
||||
}
|
||||
try:
|
||||
await session.execute(
|
||||
text(
|
||||
"""
|
||||
INSERT INTO geofence_alerts
|
||||
(id, geofence_id, source_kind, entity_id, lat, lon, payload)
|
||||
VALUES (
|
||||
CAST(:id AS uuid), CAST(:geofence_id AS uuid),
|
||||
:source_kind, :entity_id, :lat, :lon, CAST(:payload AS jsonb)
|
||||
)
|
||||
"""
|
||||
),
|
||||
{
|
||||
"id": aid,
|
||||
"geofence_id": fence["id"],
|
||||
"source_kind": source_kind,
|
||||
"entity_id": str(entity_id),
|
||||
"lat": lat,
|
||||
"lon": lon,
|
||||
"payload": json.dumps(payload or {}),
|
||||
},
|
||||
)
|
||||
except Exception:
|
||||
pass
|
||||
sent += await manager.publish_point(
|
||||
"geofence_alert", body, lat=lat, lon=lon,
|
||||
)
|
||||
try:
|
||||
await session.commit()
|
||||
except Exception:
|
||||
pass
|
||||
return sent
|
||||
|
||||
|
||||
async def list_alerts(
|
||||
*,
|
||||
geofence_id: str | None = None,
|
||||
since: datetime | None = None,
|
||||
until: datetime | None = None,
|
||||
source_kind: str | None = None,
|
||||
limit: int = 100,
|
||||
) -> list[dict]:
|
||||
"""Filterable hit log. Empty list if the DB is down — never raises."""
|
||||
where = ["TRUE"]
|
||||
params: dict[str, Any] = {"limit": int(limit)}
|
||||
if geofence_id:
|
||||
where.append("geofence_id = CAST(:geofence_id AS uuid)")
|
||||
params["geofence_id"] = geofence_id
|
||||
if since is not None:
|
||||
where.append("created_at >= :since")
|
||||
params["since"] = since
|
||||
if until is not None:
|
||||
where.append("created_at <= :until")
|
||||
params["until"] = until
|
||||
if source_kind:
|
||||
where.append("source_kind = :source_kind")
|
||||
params["source_kind"] = source_kind
|
||||
sql = f"""
|
||||
SELECT id::text, geofence_id::text, source_kind, entity_id,
|
||||
lat, lon, payload, created_at
|
||||
FROM geofence_alerts
|
||||
WHERE {' AND '.join(where)}
|
||||
ORDER BY created_at DESC
|
||||
LIMIT :limit
|
||||
"""
|
||||
try:
|
||||
async with async_session() as session:
|
||||
rows = (await session.execute(text(sql), params)).mappings().all()
|
||||
out = []
|
||||
for r in rows:
|
||||
item = dict(r)
|
||||
if item.get("created_at") is not None:
|
||||
item["created_at"] = item["created_at"].isoformat()
|
||||
out.append(item)
|
||||
return out
|
||||
except Exception:
|
||||
return []
|
||||
|
||||
|
||||
async def get_geofence(gid: str) -> dict | None:
|
||||
current = next((f for f in _cache if f["id"] == gid), None)
|
||||
if current is not None:
|
||||
return current
|
||||
try:
|
||||
await refresh_cache()
|
||||
except Exception:
|
||||
return None
|
||||
return next((f for f in _cache if f["id"] == gid), None)
|
||||
|
||||
|
||||
def _marker_from_track(row) -> dict:
|
||||
from live_layers import to_marker
|
||||
|
||||
extra = {"bucket": row["bucket"].isoformat() if row.get("bucket") else None, "dvr": True}
|
||||
return to_marker(
|
||||
row["id"], row["lat"], row["lon"],
|
||||
heading=row.get("heading"), speed=row.get("speed"),
|
||||
label=row.get("label") or row["id"],
|
||||
extra=extra,
|
||||
)
|
||||
|
||||
|
||||
async def _cagg_inside(gid: str, kind: str, bucket: datetime, limit: int = 2000) -> list[dict]:
|
||||
table = "aircraft_tracks_1min" if kind == "aircraft" else "vessel_tracks_1min"
|
||||
id_col = "hex" if kind == "aircraft" else "mmsi"
|
||||
sql = f"""
|
||||
SELECT {id_col} AS id, lat, lon, heading, speed, label, bucket
|
||||
FROM {table}
|
||||
WHERE bucket = :bucket
|
||||
AND ST_Intersects(
|
||||
(SELECT geom FROM geofences WHERE id = CAST(:gid AS uuid)),
|
||||
ST_SetSRID(ST_MakePoint(lon, lat), 4326)
|
||||
)
|
||||
LIMIT :limit
|
||||
"""
|
||||
try:
|
||||
async with async_session() as session:
|
||||
rows = (await session.execute(
|
||||
text(sql), {"bucket": bucket, "gid": gid, "limit": limit},
|
||||
)).mappings().all()
|
||||
return [
|
||||
_marker_from_track(r)
|
||||
for r in rows
|
||||
if r["lat"] is not None and r["lon"] is not None
|
||||
]
|
||||
except Exception:
|
||||
return []
|
||||
|
||||
|
||||
async def _fires_inside(gid: str, ts: datetime, limit: int = 2000) -> list[dict]:
|
||||
from tracks import minute_bucket
|
||||
|
||||
bucket = minute_bucket(ts)
|
||||
t1 = bucket + timedelta(minutes=1)
|
||||
sql = """
|
||||
SELECT latitude, longitude, brightness, confidence, acq_time, satellite,
|
||||
instrument, bright_ti5, frp, daynight
|
||||
FROM fires
|
||||
WHERE acq_time >= :t0 AND acq_time < :t1
|
||||
AND ST_Intersects(
|
||||
(SELECT geom FROM geofences WHERE id = CAST(:gid AS uuid)),
|
||||
ST_SetSRID(ST_MakePoint(longitude, latitude), 4326)
|
||||
)
|
||||
LIMIT :limit
|
||||
"""
|
||||
try:
|
||||
async with async_session() as session:
|
||||
rows = (await session.execute(
|
||||
text(sql),
|
||||
{"t0": bucket, "t1": t1, "gid": gid, "limit": limit},
|
||||
)).mappings().all()
|
||||
out = []
|
||||
for r in rows:
|
||||
item = dict(r)
|
||||
if item.get("acq_time") is not None:
|
||||
item["acq_time"] = item["acq_time"].isoformat()
|
||||
out.append(item)
|
||||
return out
|
||||
except Exception:
|
||||
return []
|
||||
|
||||
|
||||
async def snapshot_at(gid: str, ts: datetime) -> dict | None:
|
||||
"""Positions inside the fence at time T. None if the fence is missing.
|
||||
|
||||
Does not persist or notify. Empty lists if track/fire queries fail.
|
||||
"""
|
||||
fence = await get_geofence(gid)
|
||||
if fence is None:
|
||||
return None
|
||||
from tracks import minute_bucket
|
||||
|
||||
bucket = minute_bucket(ts)
|
||||
aircraft = await _cagg_inside(gid, "aircraft", bucket)
|
||||
vessels = await _cagg_inside(gid, "vessel", bucket)
|
||||
fires = await _fires_inside(gid, ts)
|
||||
return {
|
||||
"geofence_id": gid,
|
||||
"timestamp": ts.isoformat(),
|
||||
"aircraft": aircraft,
|
||||
"vessels": vessels,
|
||||
"fires": fires,
|
||||
}
|
||||
|
|
@ -14,17 +14,10 @@ from sqlalchemy.dialects.postgresql import insert as pg_insert
|
|||
from database import async_session
|
||||
from models import events as events_table
|
||||
from models import fires as fires_table
|
||||
from models import event_dedup as event_dedup_table
|
||||
from config import NATS_URL
|
||||
from sources import event_dedup_key
|
||||
|
||||
logger = logging.getLogger("osint.ingestor")
|
||||
|
||||
# asyncpg rejects statements with >32767 bind params. A FIRMS poll is ~90k
|
||||
# rows × 14 columns. Chunk inserts; still one transaction / one commit.
|
||||
FIRE_ROW_BIND_PARAMS = 14
|
||||
FIRE_INSERT_CHUNK = 2000
|
||||
|
||||
# NATS connection settings
|
||||
NATS_URLS = NATS_URL
|
||||
NATS_STREAM = "events"
|
||||
|
|
@ -102,62 +95,6 @@ async def ingest_fire_row(msg: dict) -> bool:
|
|||
"ingested fire %.5f,%.5f %s satellite=%s",
|
||||
row["latitude"], row["longitude"], row["acq_time"], row["satellite"],
|
||||
)
|
||||
lat, lon = row["latitude"], row["longitude"]
|
||||
from geofence import record_and_notify
|
||||
await record_and_notify(
|
||||
source_kind="firms",
|
||||
entity_id=f"{lat},{lon},{row['satellite']}",
|
||||
lat=lat, lon=lon, payload={"satellite": row["satellite"]},
|
||||
)
|
||||
from live_layers import aircraft_last_known
|
||||
from fire_aircraft import correlate_and_notify
|
||||
from tracks import recent_markers
|
||||
acs = list(aircraft_last_known.values()) or await recent_markers("aircraft")
|
||||
if acs:
|
||||
fire = {
|
||||
"id": f"firms:{lat:.4f},{lon:.4f}",
|
||||
"lat": lat, "lon": lon, "label": "FIRMS",
|
||||
}
|
||||
await correlate_and_notify([fire], acs)
|
||||
return inserted
|
||||
|
||||
|
||||
async def ingest_fire_rows(msgs: list[dict]) -> int:
|
||||
"""Bulk-insert FIRMS hotspots: one INSERT, one ON CONFLICT, one commit."""
|
||||
rows = []
|
||||
for msg in msgs:
|
||||
row = _fire_row_from_msg(msg)
|
||||
if row is not None:
|
||||
rows.append(row)
|
||||
if not rows:
|
||||
return 0
|
||||
inserted = 0
|
||||
async with async_session() as session:
|
||||
for i in range(0, len(rows), FIRE_INSERT_CHUNK):
|
||||
chunk = rows[i:i + FIRE_INSERT_CHUNK]
|
||||
stmt = (
|
||||
pg_insert(fires_table)
|
||||
.values(chunk)
|
||||
.on_conflict_do_nothing(constraint="pk_fires_natural_key")
|
||||
)
|
||||
result = await session.execute(stmt)
|
||||
inserted += int(result.rowcount or 0)
|
||||
await session.commit()
|
||||
if inserted:
|
||||
logger.info("bulk ingested %d/%d FIRMS hotspots", inserted, len(rows))
|
||||
from live_layers import aircraft_last_known
|
||||
from fire_aircraft import correlate_and_notify
|
||||
from tracks import recent_markers
|
||||
acs = list(aircraft_last_known.values()) or await recent_markers("aircraft")
|
||||
if acs:
|
||||
fires = [
|
||||
{
|
||||
"id": f"firms:{r['latitude']:.4f},{r['longitude']:.4f}",
|
||||
"lat": r["latitude"], "lon": r["longitude"], "label": "FIRMS",
|
||||
}
|
||||
for r in rows[:500]
|
||||
]
|
||||
await correlate_and_notify(fires, acs)
|
||||
return inserted
|
||||
|
||||
|
||||
|
|
@ -195,26 +132,10 @@ async def ingest_event(msg: dict):
|
|||
}
|
||||
|
||||
# Parse timestamp if string
|
||||
ts = event_row["source_timestamp"]
|
||||
if isinstance(ts, str):
|
||||
ts = datetime.fromisoformat(ts.replace("Z", "+00:00"))
|
||||
if isinstance(ts, datetime) and ts.tzinfo is None:
|
||||
ts = ts.replace(tzinfo=timezone.utc)
|
||||
event_row["source_timestamp"] = ts
|
||||
if isinstance(event_row["source_timestamp"], str):
|
||||
event_row["source_timestamp"] = datetime.fromisoformat(event_row["source_timestamp"])
|
||||
|
||||
key = event_dedup_key(event_row)
|
||||
async with async_session() as session:
|
||||
if key:
|
||||
dedup = (
|
||||
pg_insert(event_dedup_table)
|
||||
.values(url=key)
|
||||
.on_conflict_do_nothing(index_elements=["url"])
|
||||
)
|
||||
claimed = await session.execute(dedup)
|
||||
if not claimed.rowcount:
|
||||
await session.commit()
|
||||
logger.debug("skip duplicate event url=%s", key)
|
||||
return None
|
||||
result = await session.execute(events_table.insert().values(**event_row))
|
||||
await session.commit()
|
||||
event_id = result.inserted_primary_key[0] # type: ignore[union-attr]
|
||||
|
|
|
|||
|
|
@ -57,10 +57,10 @@ KEY_REGISTRY: dict[str, dict] = {
|
|||
"pattern": r"^[0-9a-fA-F]{32}$",
|
||||
"example": "32-char hex string (e.g. 5f3c…9a02)",
|
||||
},
|
||||
"NOUS_API_KEY": {
|
||||
"description": "Nous Portal API key — 15-min news summarizer (inference-api.nousresearch.com).",
|
||||
"pattern": r"^.{16,}$",
|
||||
"example": "key from https://portal.nousresearch.com (API keys page)",
|
||||
"GEMINI_API_KEY": {
|
||||
"description": "Google Gemini API key — LLM event analysis / summarization.",
|
||||
"pattern": r"^AIza[0-9A-Za-z_-]{35}$",
|
||||
"example": "AIza… (Google API key, 39 chars)",
|
||||
},
|
||||
"TELEGRAM_TOKEN": {
|
||||
"description": "Telegram bot token — push alert notifications to a channel.",
|
||||
|
|
@ -68,15 +68,10 @@ KEY_REGISTRY: dict[str, dict] = {
|
|||
"example": "123456789:AA… (bot token from @BotFather)",
|
||||
},
|
||||
"AISSTREAM_API_KEY": {
|
||||
"description": "AISStream (open/shared) — live US-coast AIS. Server-side WebSocket only.",
|
||||
"description": "AISStream WebSocket key — live vessel positions (server-side only).",
|
||||
"pattern": r"^.{8,}$",
|
||||
"example": "key from https://aisstream.io/account (GitHub login)",
|
||||
},
|
||||
"VESSELAPI_API_KEY": {
|
||||
"description": "VesselAPI (commercial) — Strait of Hormuz AIS, 5×/day cache. Paste the Bearer token from dashboard.vesselapi.com. Not a US-coast feed.",
|
||||
"pattern": r"^.{8,}$",
|
||||
"example": "Bearer token from https://dashboard.vesselapi.com/",
|
||||
},
|
||||
"OPENSKY_CLIENT_ID": {
|
||||
"description": "OpenSky OAuth client id — optional ADS-B fallback (unused until enabled).",
|
||||
"example": "client id from opensky-network.org account",
|
||||
|
|
@ -219,7 +214,7 @@ async def get_api_key(name: str) -> str | None:
|
|||
"""Read a stored key value — used by ingest services, never by the API.
|
||||
|
||||
Returns the raw value (or None when unset) so producers can pass it to
|
||||
external APIs (FIRMS, Nous, Telegram, …). Reads live from Postgres, so a
|
||||
external APIs (FIRMS, Gemini, Telegram, …). Reads live from Postgres, so a
|
||||
key set via the dashboard is picked up on the next poll — no restart.
|
||||
"""
|
||||
await ensure_api_keys_table()
|
||||
|
|
|
|||
File diff suppressed because it is too large
Load diff
853
app/main.py
853
app/main.py
File diff suppressed because it is too large
Load diff
65
app/masscan_config.py
Normal file
65
app/masscan_config.py
Normal file
|
|
@ -0,0 +1,65 @@
|
|||
"""Active camera-discovery configuration (masscan-based, env-driven).
|
||||
|
||||
All knobs read from the environment with safe defaults. The scanner targets
|
||||
open TCP port 554 (RTSP — the typical IP-camera port) across a configured
|
||||
range and feeds results into the same `cameras` table as the passive scraper
|
||||
(discovery_source='masscan'), deduped by URL hash.
|
||||
|
||||
ETHICS / SCOPE (mirrors camera_scraper.py):
|
||||
* Detection only — a SYN port scan for OPEN hosts. No credential guessing,
|
||||
no login attempts, no banner grabbing, and no access to camera feeds.
|
||||
* Private / reserved ranges are excluded via MASSCAN_EXCLUDEFILE so the
|
||||
scanner never probes RFC1918, loopback, link-local, multicast, or the
|
||||
bogons. Fail closed if the excludefile is missing.
|
||||
|
||||
TIMING REALITY: at the residential-safe default of 200 pps a full IPv4
|
||||
sweep (0.0.0.0/0, ~4.29B addresses) takes ~8 months. This is therefore a
|
||||
CONTINUOUS ROLLING SWEEP, not a "finish in a day" job: masscan streams
|
||||
open hosts to stdout and the runner ingests them incrementally, then
|
||||
restarts the sweep when a pass completes. New cameras are detected as they
|
||||
appear on each pass. 1k/10k pps saturated a home uplink — do not raise the
|
||||
rate unless you are on a VPS / unmetered link.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import os
|
||||
|
||||
# Path to the masscan binary (installed on the Pi host).
|
||||
MASSCAN_BIN = os.getenv("MASSCAN_BIN", "masscan")
|
||||
|
||||
# CIDR(s) to sweep. Default = the whole public IPv4 space.
|
||||
MASSCAN_RANGE = os.getenv("MASSCAN_RANGE", "0.0.0.0/0")
|
||||
|
||||
# Port(s) to probe. Default 554 = RTSP, the typical IP-camera port.
|
||||
MASSCAN_PORTS = os.getenv("MASSCAN_PORTS", "554")
|
||||
|
||||
# Packets/sec. 200 is the residential-safe default — 1k/10k pps saturated
|
||||
# a home uplink. Raise only on a VPS / unmetered link.
|
||||
MASSCAN_RATE = int(os.getenv("MASSCAN_RATE", "200"))
|
||||
|
||||
# Retransmission count. 1 maximizes unique-host coverage at low rate; the
|
||||
# default (10) spends most of the budget re-probing the same hosts.
|
||||
MASSCAN_RETRIES = int(os.getenv("MASSCAN_RETRIES", "1"))
|
||||
|
||||
# Seconds to keep listening for straggler responses after the last probe.
|
||||
# 0 avoids a 10s tail per pass; tiny loss of the very last hosts is fine
|
||||
# since the sweep repeats.
|
||||
MASSCAN_WAIT = int(os.getenv("MASSCAN_WAIT", "0"))
|
||||
|
||||
# Excludefile path on the Pi host. Must contain RFC1918/loopback/link-local/
|
||||
# multicast/bogons so the scanner never probes private ranges. Fail closed if
|
||||
# the file is absent (the runner refuses to start rather than scan wide).
|
||||
MASSCAN_EXCLUDEFILE = os.getenv(
|
||||
"MASSCAN_EXCLUDEFILE", "/etc/osint-dashboard/masscan-excludes.txt"
|
||||
)
|
||||
|
||||
# Ingest batch size — flush this many newly-seen hosts to the DB per round.
|
||||
MASSCAN_FLUSH_EVERY = int(os.getenv("MASSCAN_FLUSH_EVERY", "250"))
|
||||
|
||||
# NATS subject newly-found cameras are published on (same feed as the
|
||||
# passive scraper so the shared ingester persists them).
|
||||
MASSCAN_NATS_SUBJECT = os.getenv("MASSCAN_NATS_SUBJECT", "events.camera")
|
||||
|
||||
# discovery_source tag written into the cameras table.
|
||||
MASSCAN_DISCOVERY_SOURCE = os.getenv("MASSCAN_DISCOVERY_SOURCE", "masscan")
|
||||
226
app/masscan_scanner.py
Normal file
226
app/masscan_scanner.py
Normal file
|
|
@ -0,0 +1,226 @@
|
|||
"""masscan result parsing + ingestion for the OSINT dashboard.
|
||||
|
||||
Turns a stream of masscan JSON-lines (open port 554 hosts) into rows in the
|
||||
`cameras` table with discovery_source='masscan', deduped by URL hash against
|
||||
whatever the passive scraper already found. Newly discovered hosts are also
|
||||
published to NATS (`events.camera`) so the shared ingester pipeline persists
|
||||
them exactly like scraper finds.
|
||||
|
||||
Scope: detection of OPEN hosts only. No credentials, no banners, no feed
|
||||
access. Private/reserved ranges never enter masscan (see excludefile).
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import asyncio
|
||||
import json
|
||||
import logging
|
||||
from datetime import datetime, timezone
|
||||
|
||||
from camera_models import cameras
|
||||
from camera_scraper import url_hash, geolocate_ips
|
||||
from database import async_session
|
||||
|
||||
from masscan_config import (
|
||||
MASSCAN_NATS_SUBJECT, MASSCAN_DISCOVERY_SOURCE,
|
||||
)
|
||||
|
||||
logger = logging.getLogger("osint.masscan_scanner")
|
||||
|
||||
|
||||
# ── URL building ──────────────────────────────────────────────────────────
|
||||
|
||||
def build_rtsp_url(ip: str) -> str:
|
||||
"""Canonical URL for an open-RTSP host. Used as the dedupe key."""
|
||||
return f"rtsp://{ip}/"
|
||||
|
||||
|
||||
# ── masscan JSON parsing ──────────────────────────────────────────────────
|
||||
# masscan --output-format=json --output-file=- emits line-delimited JSON on a
|
||||
# pipe (a bare object per open host), not the array form used for seekable
|
||||
# files. We parse per-line and tolerate an accidental leading '['.
|
||||
|
||||
def parse_masscan_line(line: str) -> list[dict]:
|
||||
"""Parse one masscan stdout line into a list of host records.
|
||||
|
||||
A line may contain one JSON object or, defensively, be wrapped in an
|
||||
array. Returns [] on anything unparseable (harmless — the sweep repeats).
|
||||
"""
|
||||
s = line.strip()
|
||||
if not s:
|
||||
return []
|
||||
s = s.lstrip("[").rstrip("]").strip()
|
||||
if not s:
|
||||
return []
|
||||
# Multiple records may share a line separated by '},{'.
|
||||
if s.endswith(","):
|
||||
s = s[:-1].rstrip()
|
||||
out: list[dict] = []
|
||||
for cand in _split_records(s):
|
||||
try:
|
||||
obj = json.loads(cand)
|
||||
except (json.JSONDecodeError, ValueError):
|
||||
continue
|
||||
if isinstance(obj, dict) and obj.get("ip"):
|
||||
out.append(obj)
|
||||
return out
|
||||
|
||||
|
||||
def _split_records(s: str) -> list[str]:
|
||||
"""Split a buffer into individual JSON object strings, honoring nesting."""
|
||||
records, depth, start = [], 0, 0
|
||||
for i, ch in enumerate(s):
|
||||
if ch == "{":
|
||||
if depth == 0:
|
||||
start = i
|
||||
depth += 1
|
||||
elif ch == "}":
|
||||
depth -= 1
|
||||
if depth == 0:
|
||||
records.append(s[start:i + 1])
|
||||
return records
|
||||
|
||||
|
||||
def extract_open_ips(records: list[dict], port: int) -> list[str]:
|
||||
"""Return the list of IPs from records that have `port` open."""
|
||||
ips: list[str] = []
|
||||
for rec in records:
|
||||
for p in rec.get("ports", []):
|
||||
if p.get("port") == port and p.get("status") == "open":
|
||||
ips.append(rec["ip"])
|
||||
break
|
||||
return ips
|
||||
|
||||
|
||||
# ── Persistence ───────────────────────────────────────────────────────────
|
||||
|
||||
async def ingest_open_hosts(ips: list[str]) -> tuple[int, list[str]]:
|
||||
"""Insert-or-refresh camera rows for open RTSP hosts that have a public feed.
|
||||
|
||||
A host only lands in the table (and therefore on the map) if an
|
||||
unauthenticated HTTP still or MJPEG URL responds. Port-554-only hosts
|
||||
are skipped. Returns (newly_inserted, hosts_with_working_feed).
|
||||
"""
|
||||
if not ips:
|
||||
return 0, []
|
||||
from camera_preview import probe_public_feed
|
||||
|
||||
unique = list(dict.fromkeys(ips))
|
||||
sem = asyncio.Semaphore(20)
|
||||
|
||||
async def _probe(ip: str) -> tuple[str, str | None]:
|
||||
async with sem:
|
||||
return ip, await probe_public_feed(ip)
|
||||
|
||||
probed = await asyncio.gather(*(_probe(ip) for ip in unique))
|
||||
live = [(ip, feed) for ip, feed in probed if feed]
|
||||
if not live:
|
||||
logger.info("masscan ingest: 0 working feeds of %d open-554 hosts",
|
||||
len(unique))
|
||||
return 0, []
|
||||
|
||||
now = datetime.now(timezone.utc)
|
||||
coords = await geolocate_ips([ip for ip, _ in live])
|
||||
new = 0
|
||||
async with async_session() as session:
|
||||
for ip, feed in live:
|
||||
url = build_rtsp_url(ip)
|
||||
h = url_hash(url)
|
||||
lat, lon = coords.get(ip, (None, None))
|
||||
existing = (await session.execute(
|
||||
cameras.select().where(cameras.c.url_hash == h)
|
||||
)).one_or_none()
|
||||
if existing is None:
|
||||
await session.execute(cameras.insert().values(
|
||||
url_hash=h,
|
||||
source_url=url,
|
||||
snapshot_url=feed,
|
||||
discovery_source=MASSCAN_DISCOVERY_SOURCE,
|
||||
location_lat=lat,
|
||||
location_lon=lon,
|
||||
location_name=f"{ip} (IP-geo)" if lat is not None else None,
|
||||
vendor=None,
|
||||
device_type="rtsp",
|
||||
first_seen=now,
|
||||
last_seen=now,
|
||||
raw={"discovered_via": "masscan", "port": 554,
|
||||
"public_feed": feed},
|
||||
))
|
||||
new += 1
|
||||
else:
|
||||
await session.execute(cameras.update().where(
|
||||
cameras.c.url_hash == h
|
||||
).values(
|
||||
last_seen=now,
|
||||
snapshot_url=feed,
|
||||
location_lat=lat,
|
||||
location_lon=lon,
|
||||
location_name=f"{ip} (IP-geo)" if lat is not None else None,
|
||||
))
|
||||
await session.commit()
|
||||
logger.info("masscan ingest: %d new working feeds (%d probed, %d open-554)",
|
||||
new, len(live), len(unique))
|
||||
return new, [ip for ip, _ in live]
|
||||
|
||||
|
||||
# ── NATS publish ──────────────────────────────────────────────────────────
|
||||
|
||||
async def publish_new_hosts(ips: list[str]) -> int:
|
||||
"""Publish newly-found open hosts to NATS for the shared ingester.
|
||||
|
||||
Returns the number of messages published (0 if NATS is down).
|
||||
"""
|
||||
import json as _json
|
||||
import nats
|
||||
from config import NATS_URL
|
||||
|
||||
if not ips:
|
||||
return 0
|
||||
try:
|
||||
nc = await nats.connect(NATS_URL)
|
||||
except Exception: # noqa: BLE001
|
||||
logger.warning("NATS unavailable — skipping publish pass")
|
||||
return 0
|
||||
published = 0
|
||||
try:
|
||||
js = nc.jetstream()
|
||||
for ip in dict.fromkeys(ips):
|
||||
url = build_rtsp_url(ip)
|
||||
msg = {
|
||||
"source_type": "camera",
|
||||
"title": f"Open RTSP camera ({ip})",
|
||||
"url": url,
|
||||
"location_lat": None,
|
||||
"location_lon": None,
|
||||
"location_name": None,
|
||||
"tags": ["osint", "camera", MASSCAN_DISCOVERY_SOURCE],
|
||||
"raw": {
|
||||
"url_hash": url_hash(url),
|
||||
"source_url": url,
|
||||
"snapshot_url": None,
|
||||
"vendor": None,
|
||||
"device_type": "rtsp",
|
||||
"discovered_via": "masscan",
|
||||
"port": 554,
|
||||
},
|
||||
"source_timestamp": datetime.now(timezone.utc).isoformat(),
|
||||
}
|
||||
await js.publish(MASSCAN_NATS_SUBJECT, _json.dumps(msg).encode())
|
||||
published += 1
|
||||
finally:
|
||||
await nc.close()
|
||||
logger.info("published %d masscan finds to %s", published, MASSCAN_NATS_SUBJECT)
|
||||
return published
|
||||
|
||||
|
||||
# ── Batch drain helper used by the runner ─────────────────────────────────
|
||||
|
||||
async def flush(seen: set[str], new_accum: int) -> tuple[int, int]:
|
||||
"""Ingest + publish the accumulated host set; return (new, published)."""
|
||||
if not seen:
|
||||
return 0, 0
|
||||
ips = list(seen)
|
||||
new, live = await ingest_open_hosts(ips)
|
||||
published = await publish_new_hosts(live)
|
||||
seen.clear()
|
||||
return new, published
|
||||
|
|
@ -68,15 +68,6 @@ Index("ix_events_search_vector", events.c.search_vector, postgresql_using="gin")
|
|||
# Spatial index on location
|
||||
Index("ix_events_location", events.c.location_lat, events.c.location_lon)
|
||||
|
||||
# Timescale unique indexes must include the partition column, so URL
|
||||
# idempotency lives on a regular table — not the events hypertable.
|
||||
event_dedup = Table(
|
||||
"event_dedup",
|
||||
metadata,
|
||||
Column("url", Text, primary_key=True),
|
||||
Column("created_at", DateTime(timezone=True), server_default=func.now(), nullable=False),
|
||||
)
|
||||
|
||||
|
||||
# ── Entities (people, organizations, locations of interest) ──────────────
|
||||
|
||||
|
|
@ -194,8 +185,8 @@ Index("ix_fires_bbox", fires.c.longitude, fires.c.latitude)
|
|||
|
||||
|
||||
# ── News pipeline (scraper + summarizer) ──────────────────────────────────
|
||||
# Written by the vendored news-scraper (Scrapy) / news-summarizer services;
|
||||
# schema must match the idempotent alembic migrations 003_news + 005_news_items.
|
||||
# Written by the vendored news-scraper (Scrapy) / news-summarizer (Gemini)
|
||||
# services; schema must match the idempotent alembic migration 003_news.
|
||||
|
||||
articles = Table(
|
||||
"articles",
|
||||
|
|
@ -218,27 +209,6 @@ article_summaries = Table(
|
|||
Column("summary_text", Text, nullable=False),
|
||||
Column("batch_timestamp", DateTime(timezone=True),
|
||||
server_default=func.now(), nullable=False),
|
||||
Column("model", Text), # LLM id used for this batch; nullable for old rows
|
||||
Column("kind", Text), # interval | daily_recap; nullable for old rows
|
||||
)
|
||||
|
||||
Index("ix_article_summaries_batch_timestamp", article_summaries.c.batch_timestamp)
|
||||
|
||||
|
||||
news_items = Table(
|
||||
"news_items",
|
||||
metadata,
|
||||
Column("id", Integer, primary_key=True, autoincrement=True),
|
||||
Column("summary_id", Integer),
|
||||
Column("kind", Text, nullable=False),
|
||||
Column("headline", Text, nullable=False),
|
||||
Column("importance", Text, nullable=False),
|
||||
Column("location_name", Text),
|
||||
Column("lat", Float),
|
||||
Column("lon", Float),
|
||||
Column("location_confidence", Text),
|
||||
Column("category", Text),
|
||||
Column("url", Text),
|
||||
Column("created_at", DateTime(timezone=True), server_default=func.now(), nullable=False),
|
||||
)
|
||||
Index("ix_news_items_kind_created", news_items.c.kind, news_items.c.created_at)
|
||||
|
|
|
|||
99
app/place.py
99
app/place.py
|
|
@ -1,99 +0,0 @@
|
|||
"""Nominatim reverse-geocode proxy for the map place dossier.
|
||||
|
||||
Browser clients cannot set an identifying User-Agent, and Nominatim typically
|
||||
blocks CORS — so the HUD calls GET /api/place instead of talking to OSM
|
||||
directly. Cache 60s / 500 keys; never exceed 1 req/s upstream.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import asyncio
|
||||
import time
|
||||
|
||||
import httpx
|
||||
from cachetools import TTLCache
|
||||
|
||||
from config import NOMINATIM_MIN_INTERVAL, NOMINATIM_URL, OSINT_USER_AGENT
|
||||
|
||||
_NOMINATIM = NOMINATIM_URL.rstrip("/")
|
||||
|
||||
place_cache: TTLCache = TTLCache(maxsize=500, ttl=60)
|
||||
|
||||
_lock = asyncio.Lock()
|
||||
_last_req = 0.0
|
||||
|
||||
_ADDR_KEEP = (
|
||||
"house_number", "road", "neighbourhood", "suburb", "city", "town",
|
||||
"village", "hamlet", "county", "state", "postcode", "country", "country_code",
|
||||
)
|
||||
|
||||
|
||||
def cache_key(lat: float, lon: float) -> str:
|
||||
return f"{lat:.4f},{lon:.4f}"
|
||||
|
||||
|
||||
def slim_place(lat: float, lon: float, data: dict | None) -> dict:
|
||||
data = data or {}
|
||||
raw_addr = data.get("address")
|
||||
addr_in: dict = raw_addr if isinstance(raw_addr, dict) else {}
|
||||
address = {k: addr_in[k] for k in _ADDR_KEEP if addr_in.get(k)}
|
||||
err = data.get("error")
|
||||
display = None if err else (data.get("display_name") or None)
|
||||
name = None if err else (data.get("name") or address.get("city")
|
||||
or address.get("town") or address.get("village") or None)
|
||||
return {
|
||||
"lat": lat,
|
||||
"lon": lon,
|
||||
"display_name": display,
|
||||
"name": name,
|
||||
"address": address,
|
||||
"osm_type": None if err else data.get("osm_type"),
|
||||
"osm_id": None if err else data.get("osm_id"),
|
||||
"attribution": "© OpenStreetMap contributors",
|
||||
}
|
||||
|
||||
|
||||
async def reverse_geocode(lat: float, lon: float) -> dict:
|
||||
"""Reverse-geocode a point. Cache hits skip Nominatim entirely."""
|
||||
if not (-90.0 <= lat <= 90.0 and -180.0 <= lon <= 180.0):
|
||||
raise ValueError("lat/lon out of range")
|
||||
key = cache_key(lat, lon)
|
||||
qlat, qlon = (float(p) for p in key.split(","))
|
||||
async with _lock:
|
||||
hit = place_cache.get(key)
|
||||
if hit is not None:
|
||||
return hit
|
||||
global _last_req
|
||||
wait = _last_req + NOMINATIM_MIN_INTERVAL - time.monotonic()
|
||||
if wait > 0:
|
||||
await asyncio.sleep(wait)
|
||||
body = await _fetch_nominatim(qlat, qlon)
|
||||
_last_req = time.monotonic()
|
||||
place_cache[key] = body
|
||||
return body
|
||||
|
||||
|
||||
async def _fetch_nominatim(lat: float, lon: float) -> dict:
|
||||
headers = {
|
||||
"User-Agent": OSINT_USER_AGENT,
|
||||
"Accept": "application/json",
|
||||
}
|
||||
url = f"{_NOMINATIM}/reverse"
|
||||
params = {
|
||||
"lat": f"{lat:.6f}",
|
||||
"lon": f"{lon:.6f}",
|
||||
"format": "jsonv2",
|
||||
"addressdetails": "1",
|
||||
"zoom": "18",
|
||||
}
|
||||
async with _http_client(timeout=10.0, follow_redirects=True) as client:
|
||||
r = await client.get(url, params=params, headers=headers)
|
||||
r.raise_for_status()
|
||||
data = r.json()
|
||||
if not isinstance(data, dict):
|
||||
data = {}
|
||||
return slim_place(lat, lon, data)
|
||||
|
||||
|
||||
def _http_client(**kwargs):
|
||||
return httpx.AsyncClient(**kwargs)
|
||||
|
|
@ -11,6 +11,3 @@ feedparser>=6.0
|
|||
python-dateutil>=2.9
|
||||
structlog>=24.4
|
||||
websockets>=14
|
||||
cachetools>=5.5
|
||||
h3>=4.0
|
||||
sgp4>=2.23
|
||||
|
|
|
|||
|
|
@ -24,8 +24,8 @@ import sys
|
|||
|
||||
sys.path.insert(0, sys_path)
|
||||
|
||||
from config import NATS_URL, FIRMS_INTERVAL, FIRMS_DATASET, AISSTREAM_IN_INGEST, VESSELAPI_IN_INGEST # noqa: E402
|
||||
from sources import ingest_rss_feed, ingest_gdelt, ingest_earthquakes, ingest_eonet, ingest_cisa_kev # noqa: E402
|
||||
from config import NATS_URL, FIRMS_INTERVAL, FIRMS_DATASET, AISSTREAM_IN_INGEST # noqa: E402
|
||||
from sources import ingest_rss_feed, ingest_gdelt, ingest_earthquakes # noqa: E402
|
||||
from fire_sources import ingest_fires # noqa: E402
|
||||
from ingestor import ingest_event, start_nats_consumer # noqa: E402
|
||||
|
||||
|
|
@ -37,8 +37,6 @@ INTERVAL = int(os.getenv("INGEST_INTERVAL", "300"))
|
|||
GDELT_QUERY = os.getenv("GDELT_QUERY", "")
|
||||
ENABLE_QUAKES = os.getenv("INGEST_EARTHQUAKES", "1").lower() in ("1", "true", "yes")
|
||||
ENABLE_FIRES = os.getenv("INGEST_FIRES", "1").lower() in ("1", "true", "yes")
|
||||
ENABLE_EONET = os.getenv("INGEST_EONET", "1").lower() in ("1", "true", "yes")
|
||||
ENABLE_KEV = os.getenv("INGEST_KEV", "1").lower() in ("1", "true", "yes")
|
||||
|
||||
NATS_STREAM = "events"
|
||||
|
||||
|
|
@ -64,18 +62,6 @@ async def producer_loop() -> None:
|
|||
logger.info("USGS -> %d events", q)
|
||||
except Exception: # noqa: BLE001
|
||||
logger.exception("USGS fetch failed")
|
||||
if ENABLE_EONET:
|
||||
try:
|
||||
n = await ingest_eonet()
|
||||
logger.info("EONET -> %d events", n)
|
||||
except Exception: # noqa: BLE001
|
||||
logger.exception("EONET fetch failed")
|
||||
if ENABLE_KEV:
|
||||
try:
|
||||
k = await ingest_cisa_kev()
|
||||
logger.info("CISA KEV -> %d events", k)
|
||||
except Exception: # noqa: BLE001
|
||||
logger.exception("CISA KEV fetch failed")
|
||||
except Exception: # noqa: BLE001
|
||||
logger.exception("producer cycle error")
|
||||
await asyncio.sleep(INTERVAL)
|
||||
|
|
@ -121,11 +107,6 @@ async def main() -> None:
|
|||
"ingester starting (rss=%d feeds, gdelt_q=%r, quakes=%s, fires=%s, interval=%ss)",
|
||||
len(RSS_URLS), GDELT_QUERY, ENABLE_QUAKES, ENABLE_FIRES, INTERVAL,
|
||||
)
|
||||
try:
|
||||
from geofence import refresh_cache
|
||||
await refresh_cache()
|
||||
except Exception:
|
||||
logger.exception("geofence cache refresh failed (ST_Intersects still runs on ingest)")
|
||||
tasks: list[asyncio.Task] = []
|
||||
if ENABLE_FIRES:
|
||||
# Fire ingest only starts once FIRMS_MAP_KEY is set (ingest_fires logs
|
||||
|
|
@ -134,9 +115,6 @@ async def main() -> None:
|
|||
if AISSTREAM_IN_INGEST:
|
||||
from ais_stream import run_ais_worker # noqa: E402
|
||||
tasks.append(asyncio.create_task(run_ais_worker()))
|
||||
if VESSELAPI_IN_INGEST:
|
||||
from vesselapi import run_vesselapi_worker # noqa: E402
|
||||
tasks.append(asyncio.create_task(run_vesselapi_worker()))
|
||||
await asyncio.gather(producer_loop(), consumer_loop(), *tasks)
|
||||
|
||||
|
||||
|
|
|
|||
149
app/run_masscan_service.py
Normal file
149
app/run_masscan_service.py
Normal file
|
|
@ -0,0 +1,149 @@
|
|||
"""Continuous masscan rolling-sweep service for the OSINT dashboard.
|
||||
|
||||
Runs masscan against the configured range for open port 554 (RTSP), streams
|
||||
the JSON-lines output, and ingests open hosts into the `cameras` table (new
|
||||
finds only) plus publishes them to NATS — exactly like the passive scraper.
|
||||
|
||||
Because a full IPv4 sweep at a conservative rate takes days, this runs
|
||||
masscan CONTINUOUSLY: each pass streams results in as they're found, and when
|
||||
a pass completes the sweep restarts from the top. New cameras are picked up
|
||||
on every pass.
|
||||
|
||||
Ethics: detection-only (open-port SYN scan). Private/reserved ranges are
|
||||
excluded and the service REFUSES to start if the excludefile is missing, so
|
||||
we never probe private space by accident.
|
||||
|
||||
Run once (for a manual/test pass): python app/run_masscan_service.py --once
|
||||
Run forever (systemd): python app/run_masscan_service.py
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import asyncio
|
||||
import logging
|
||||
import os
|
||||
import sys
|
||||
from pathlib import Path
|
||||
|
||||
sys_path = str(Path(__file__).parent)
|
||||
sys.path.insert(0, sys_path)
|
||||
|
||||
import masscan_config as cfg # noqa: E402
|
||||
from database import init_extensions # noqa: E402
|
||||
from masscan_scanner import ( # noqa: E402
|
||||
parse_masscan_line, extract_open_ips, flush,
|
||||
)
|
||||
|
||||
logging.basicConfig(level=logging.INFO,
|
||||
format="%(asctime)s %(levelname)s %(name)s: %(message)s")
|
||||
logger = logging.getLogger("osint.masscan_service")
|
||||
|
||||
ONCE = "--once" in sys.argv[1:]
|
||||
|
||||
|
||||
def _verify_excludefile() -> None:
|
||||
"""Fail closed: refuse to sweep the wide range without an excludefile."""
|
||||
if not cfg.MASSCAN_EXCLUDEFILE:
|
||||
raise SystemExit("MASSCAN_EXCLUDEFILE is empty — refusing to run")
|
||||
if not Path(cfg.MASSCAN_EXCLUDEFILE).is_file():
|
||||
raise SystemExit(
|
||||
f"excludefile {cfg.MASSCAN_EXCLUDEFILE!r} missing — refusing to "
|
||||
f"run (would risk probing private ranges). Install the excludefile "
|
||||
f"first (see deploy/masscan-excludes.txt)."
|
||||
)
|
||||
|
||||
|
||||
def build_command() -> list[str]:
|
||||
cmd = [
|
||||
cfg.MASSCAN_BIN,
|
||||
cfg.MASSCAN_RANGE,
|
||||
f"-p{cfg.MASSCAN_PORTS}",
|
||||
f"--rate={cfg.MASSCAN_RATE}",
|
||||
f"--retries={cfg.MASSCAN_RETRIES}",
|
||||
f"--wait={cfg.MASSCAN_WAIT}",
|
||||
"--output-format=json",
|
||||
"--output-file=-",
|
||||
]
|
||||
if cfg.MASSCAN_EXCLUDEFILE:
|
||||
cmd.append(f"--excludefile={cfg.MASSCAN_EXCLUDEFILE}")
|
||||
return cmd
|
||||
|
||||
|
||||
async def _drain_stderr(stream: asyncio.StreamReader) -> None:
|
||||
"""Consume masscan's progress chatter so its stderr pipe never fills."""
|
||||
while True:
|
||||
line = await stream.readline()
|
||||
if not line:
|
||||
break
|
||||
text = line.decode(errors="ignore").strip()
|
||||
if text and not text.startswith("rate:"):
|
||||
logger.debug("masscan: %s", text)
|
||||
|
||||
|
||||
async def run_pass() -> tuple[int, int]:
|
||||
"""Run one full sweep pass, ingesting incrementally.
|
||||
|
||||
Returns (new_hosts, total_hosts_seen) for the whole pass.
|
||||
"""
|
||||
cmd = build_command()
|
||||
logger.info("starting masscan pass: %s", " ".join(cmd))
|
||||
proc = await asyncio.create_subprocess_exec(
|
||||
*cmd,
|
||||
stdout=asyncio.subprocess.PIPE,
|
||||
stderr=asyncio.subprocess.PIPE,
|
||||
)
|
||||
if proc.stderr is not None:
|
||||
asyncio.ensure_future(_drain_stderr(proc.stderr))
|
||||
|
||||
seen: set[str] = set()
|
||||
total_seen = 0
|
||||
total_new = 0
|
||||
try:
|
||||
while True:
|
||||
raw = await proc.stdout.readline()
|
||||
if not raw:
|
||||
break
|
||||
records = parse_masscan_line(raw.decode(errors="ignore"))
|
||||
for ip in extract_open_ips(records, 554):
|
||||
if ip in seen:
|
||||
continue
|
||||
seen.add(ip)
|
||||
if len(seen) >= cfg.MASSCAN_FLUSH_EVERY:
|
||||
new, _published = await flush(seen, total_new)
|
||||
total_new += new
|
||||
total_seen += new
|
||||
# Drain the final partial batch.
|
||||
if seen:
|
||||
new, _published = await flush(seen, total_new)
|
||||
total_new += new
|
||||
rc = await proc.wait()
|
||||
except asyncio.CancelledError:
|
||||
proc.kill()
|
||||
raise
|
||||
logger.info("masscan pass finished (rc=%s): %d new hosts ingested",
|
||||
rc, total_new)
|
||||
return total_new, total_seen
|
||||
|
||||
|
||||
async def main() -> None:
|
||||
_verify_excludefile()
|
||||
await init_extensions()
|
||||
logger.info(
|
||||
"masscan service starting: range=%s ports=%s rate=%s pps (full sweep "
|
||||
"~%.0fh at this rate)",
|
||||
cfg.MASSCAN_RANGE, cfg.MASSCAN_PORTS, cfg.MASSCAN_RATE,
|
||||
4.29e9 / cfg.MASSCAN_RATE / 3600,
|
||||
)
|
||||
while True:
|
||||
try:
|
||||
await run_pass()
|
||||
except Exception: # noqa: BLE001
|
||||
logger.exception("masscan pass error")
|
||||
if ONCE:
|
||||
return
|
||||
# Small gap between passes so the restart is visible in logs.
|
||||
await asyncio.sleep(5)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
asyncio.run(main())
|
||||
|
|
@ -1,289 +0,0 @@
|
|||
"""CelesTrak satellites last-known overlay.
|
||||
|
||||
Fetches GP **JSON** (OMM mean elements — not TLE) per group at most once per
|
||||
2 hours, caches the element blob, and propagates positions with a real SGP4
|
||||
library on every request. Positions move every second; the *element set* is
|
||||
what we cache, not the derived lat/lon.
|
||||
|
||||
Catalog numbers >= 100000 only fit OMM/JSON, never a 5-column TLE field, so
|
||||
elements are initialized through :func:`sgp4.omm.initialize` (which consumes
|
||||
the CelesTrak GP JSON fields verbatim) rather than round-tripping to TLE.
|
||||
|
||||
CelesTrak usage policy is non-negotiable: fetch the GP JSON blob at most once
|
||||
per 2 hours per group, never fan out every GROUP, never also fetch
|
||||
``GROUP=active`` plus subsets, and identify with ``OSINT_USER_AGENT``.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
import math
|
||||
from datetime import datetime, timezone
|
||||
from urllib.parse import quote
|
||||
|
||||
logger = logging.getLogger("osint.satellites")
|
||||
|
||||
CELESTRAK_GP = "https://celestrak.org/NORAD/elements/gp.php"
|
||||
SATNOGS_TLE = "https://db.satnogs.org/api/tle/"
|
||||
DEFAULT_GROUPS = ("stations", "weather")
|
||||
ALLOWED_GROUPS = ("stations", "weather", "gps-ops", "starlink")
|
||||
# CelesTrak policy: do not hit gp.php more than once per 2 hours per group.
|
||||
SATELLITE_TTL = 2 * 3600.0
|
||||
SOURCE_CELESTRAK = "celestrak"
|
||||
SOURCE_SATNOGS = "satnogs"
|
||||
DEFAULT_LIMIT = 2000
|
||||
|
||||
# WGS-84 ellipsoid for TEME -> geodetic.
|
||||
_WGS84_A = 6378.137
|
||||
_WGS84_F = 1.0 / 298.257223563
|
||||
|
||||
# Last-good element blob per group, kept past TTL so a 403 / "has not updated
|
||||
# since ..." still serves the previous set instead of failing the overlay.
|
||||
_last_good: dict[str, list[dict]] = {}
|
||||
|
||||
|
||||
def parse_groups(raw: str | None) -> list[str]:
|
||||
"""Validate + normalize a comma-separated group list. Raises ValueError.
|
||||
|
||||
Starlink is allowed only when explicitly requested (never in the default);
|
||||
it is a large supplemental feed, not part of the stations/weather default.
|
||||
"""
|
||||
groups = [g.strip().lower() for g in (raw or "").split(",") if g.strip()]
|
||||
if not groups:
|
||||
raise ValueError("groups must be a non-empty comma-separated list")
|
||||
bad = [g for g in groups if g not in ALLOWED_GROUPS]
|
||||
if bad:
|
||||
raise ValueError(f"unknown group(s): {', '.join(bad)}")
|
||||
# Dedup, preserve order.
|
||||
seen: set[str] = set()
|
||||
out: list[str] = []
|
||||
for g in groups:
|
||||
if g not in seen:
|
||||
seen.add(g)
|
||||
out.append(g)
|
||||
return out
|
||||
|
||||
|
||||
def _teme_to_geodetic(
|
||||
r: tuple[float, float, float],
|
||||
jd: float,
|
||||
fr: float,
|
||||
) -> tuple[float, float, float]:
|
||||
"""SGP4 TEME position (km) -> geodetic (lat_deg, lon_deg, alt_km).
|
||||
|
||||
Rotate TEME into an Earth-fixed frame via GMST, then iterate the WGS-84
|
||||
geodetic conversion. Good to well under a km for a ground-track overlay.
|
||||
"""
|
||||
# GMST (radians) from UT1 ~= UTC here (sub-second error is negligible).
|
||||
d = (jd + fr) - 2451545.0
|
||||
t = d / 36525.0
|
||||
gmst_s = (
|
||||
67310.54841
|
||||
+ (876600.0 * 3600.0 + 8640184.812866) * t
|
||||
+ 0.093104 * t * t
|
||||
- 6.2e-6 * t * t * t
|
||||
)
|
||||
theta = math.radians((gmst_s % 86400.0) / 240.0)
|
||||
|
||||
x, y, z = r
|
||||
xe = x * math.cos(theta) + y * math.sin(theta)
|
||||
ye = -x * math.sin(theta) + y * math.cos(theta)
|
||||
ze = z
|
||||
|
||||
e2 = _WGS84_F * (2.0 - _WGS84_F)
|
||||
p = math.sqrt(xe * xe + ye * ye)
|
||||
lon = math.atan2(ye, xe)
|
||||
lat = math.atan2(ze, p * (1.0 - e2))
|
||||
alt = 0.0
|
||||
for _ in range(10):
|
||||
n = _WGS84_A / math.sqrt(1.0 - e2 * math.sin(lat) ** 2)
|
||||
alt = p / math.cos(lat) - n
|
||||
lat = math.atan2(ze, p * (1.0 - e2 * n / (n + alt)))
|
||||
n = _WGS84_A / math.sqrt(1.0 - e2 * math.sin(lat) ** 2)
|
||||
alt = p / math.cos(lat) - n
|
||||
return math.degrees(lat), math.degrees(lon), alt
|
||||
|
||||
|
||||
def propagate_gp(
|
||||
elements: list[dict],
|
||||
group: str,
|
||||
now: datetime,
|
||||
) -> list[dict]:
|
||||
"""Propagate CelesTrak GP JSON elements to geodetic positions at ``now``.
|
||||
|
||||
Pure and deterministic given ``now``. Returns ``[{id, name, lat, lon,
|
||||
alt_km, group}]``; malformed elements and propagation errors are skipped.
|
||||
"""
|
||||
from sgp4.api import Satrec, jday
|
||||
import sgp4.omm as omm
|
||||
|
||||
jd, fr = jday(
|
||||
now.year, now.month, now.day,
|
||||
now.hour, now.minute, now.second + now.microsecond / 1e6,
|
||||
)
|
||||
out: list[dict] = []
|
||||
for rec in elements:
|
||||
if not isinstance(rec, dict):
|
||||
continue
|
||||
sat = Satrec()
|
||||
try:
|
||||
omm.initialize(sat, rec)
|
||||
except (KeyError, ValueError, TypeError):
|
||||
continue
|
||||
err, r, _v = sat.sgp4(jd, fr)
|
||||
if err != 0:
|
||||
continue
|
||||
lat, lon, alt = _teme_to_geodetic(r, jd, fr)
|
||||
norad = rec.get("NORAD_CAT_ID")
|
||||
out.append({
|
||||
"id": str(norad) if norad is not None else "",
|
||||
"name": rec.get("OBJECT_NAME") or str(norad or ""),
|
||||
"lat": round(lat, 5),
|
||||
"lon": round(lon, 5),
|
||||
"alt_km": round(alt, 2),
|
||||
"group": group,
|
||||
})
|
||||
return out
|
||||
|
||||
|
||||
def _max_epoch(elements: list[dict]) -> str | None:
|
||||
"""Most recent EPOCH across an element set (ISO-8601 lexical max)."""
|
||||
epochs = [
|
||||
str(e["EPOCH"]) for e in elements
|
||||
if isinstance(e, dict) and e.get("EPOCH")
|
||||
]
|
||||
return max(epochs) if epochs else None
|
||||
|
||||
|
||||
def propagate_satnogs_tle(
|
||||
payload: list[dict],
|
||||
group: str,
|
||||
now: datetime,
|
||||
) -> tuple[list[dict], str | None]:
|
||||
"""Fallback parser for SatNOGS TLE JSON (``[{tle0,tle1,tle2,updated}]``).
|
||||
|
||||
Returns ``(satellites, epoch)`` where epoch is the max ``updated`` time.
|
||||
Only used when the CelesTrak cache is completely empty.
|
||||
"""
|
||||
from sgp4.api import Satrec, jday
|
||||
|
||||
jd, fr = jday(
|
||||
now.year, now.month, now.day,
|
||||
now.hour, now.minute, now.second + now.microsecond / 1e6,
|
||||
)
|
||||
out: list[dict] = []
|
||||
epochs: list[str] = []
|
||||
for rec in payload or []:
|
||||
if not isinstance(rec, dict):
|
||||
continue
|
||||
line1 = rec.get("tle1")
|
||||
line2 = rec.get("tle2")
|
||||
if not line1 or not line2:
|
||||
continue
|
||||
try:
|
||||
sat = Satrec.twoline2rv(line1, line2)
|
||||
except (ValueError, TypeError):
|
||||
continue
|
||||
e, r, _v = sat.sgp4(jd, fr)
|
||||
if e != 0:
|
||||
continue
|
||||
lat, lon, alt = _teme_to_geodetic(r, jd, fr)
|
||||
satnum = getattr(sat, "satnum_str", None) or rec.get("norad_cat_id")
|
||||
name = (rec.get("tle0") or "").strip().lstrip("0").strip() or str(satnum or "")
|
||||
out.append({
|
||||
"id": str(satnum).strip() or "",
|
||||
"name": name,
|
||||
"lat": round(lat, 5),
|
||||
"lon": round(lon, 5),
|
||||
"alt_km": round(alt, 2),
|
||||
"group": group,
|
||||
})
|
||||
if rec.get("updated"):
|
||||
epochs.append(str(rec["updated"]))
|
||||
return out, (max(epochs) if epochs else None)
|
||||
|
||||
|
||||
async def _group_elements(group: str) -> tuple[list[dict], str | None]:
|
||||
"""CelesTrak GP blob for one group, TTL-cached with a last-good fallback.
|
||||
|
||||
Returns ``(elements, epoch)``. On a fetch failure (403 / "has not updated
|
||||
since ...") falls back to the previous successful blob for that group.
|
||||
"""
|
||||
from live_layers import _get_json, _ttl_get
|
||||
|
||||
url = f"{CELESTRAK_GP}?GROUP={quote(group)}&FORMAT=JSON"
|
||||
|
||||
async def _load() -> list[dict]:
|
||||
data = await _get_json(url)
|
||||
if not isinstance(data, list):
|
||||
raise ValueError(f"unexpected CelesTrak payload for {group}")
|
||||
if data:
|
||||
_last_good[group] = data
|
||||
return data
|
||||
|
||||
key = f"celestrak:gp:{group}"
|
||||
try:
|
||||
elements = await _ttl_get(key, SATELLITE_TTL, _load)
|
||||
except Exception as exc: # noqa: BLE001
|
||||
logger.warning("celestrak_fetch_failed group=%s: %s", group, exc)
|
||||
elements = _last_good.get(group, [])
|
||||
if not elements:
|
||||
return [], None
|
||||
return elements, _max_epoch(elements)
|
||||
|
||||
|
||||
async def fetch_satellites(
|
||||
groups: list[str],
|
||||
bbox: str | None = None,
|
||||
limit: int = DEFAULT_LIMIT,
|
||||
) -> dict:
|
||||
"""Assemble the ``/api/satellites`` payload for the requested groups."""
|
||||
from live_layers import _get_json, _ttl_get, filter_points_bbox, parse_bbox
|
||||
|
||||
now = datetime.now(timezone.utc)
|
||||
satellites: list[dict] = []
|
||||
epoch: str | None = None
|
||||
source = SOURCE_CELESTRAK
|
||||
|
||||
for group in groups:
|
||||
elements, group_epoch = await _group_elements(group)
|
||||
if not elements:
|
||||
continue
|
||||
if group_epoch and (epoch is None or group_epoch > epoch):
|
||||
epoch = group_epoch
|
||||
satellites.extend(propagate_gp(elements, group, now))
|
||||
|
||||
if not satellites:
|
||||
# Fallback only when the CelesTrak cache is entirely empty — never
|
||||
# poll both providers every cycle.
|
||||
async def _load_satnogs() -> list[dict]:
|
||||
data = await _get_json(SATNOGS_TLE, params={"format": "json"})
|
||||
return data if isinstance(data, list) else []
|
||||
|
||||
try:
|
||||
satnogs = await _ttl_get("satnogs:tle", SATELLITE_TTL, _load_satnogs)
|
||||
except Exception as exc: # noqa: BLE001
|
||||
logger.warning("satnogs_fetch_failed: %s", exc)
|
||||
satnogs = []
|
||||
if satnogs:
|
||||
source = SOURCE_SATNOGS
|
||||
for group in groups:
|
||||
rows, sn_epoch = propagate_satnogs_tle(satnogs, group, now)
|
||||
if sn_epoch and (epoch is None or sn_epoch > epoch):
|
||||
epoch = sn_epoch
|
||||
satellites.extend(rows)
|
||||
|
||||
if bbox:
|
||||
minlon, minlat, maxlon, maxlat = parse_bbox(bbox)
|
||||
satellites = filter_points_bbox(
|
||||
satellites, minlon, minlat, maxlon, maxlat, limit,
|
||||
)
|
||||
else:
|
||||
satellites = satellites[:limit]
|
||||
|
||||
return {
|
||||
"satellites": satellites,
|
||||
"source": source,
|
||||
"tle_epoch": epoch,
|
||||
"timestamp": now.isoformat(),
|
||||
}
|
||||
122
app/schemas.py
122
app/schemas.py
|
|
@ -4,10 +4,10 @@ from __future__ import annotations
|
|||
|
||||
from datetime import datetime
|
||||
from enum import Enum
|
||||
from typing import Literal, Optional
|
||||
from typing import Optional
|
||||
from uuid import UUID
|
||||
|
||||
from pydantic import BaseModel, ConfigDict, Field, field_validator
|
||||
from pydantic import BaseModel, Field
|
||||
|
||||
|
||||
# ─── Enums ───────────────────────────────────────────────────────────────
|
||||
|
|
@ -63,17 +63,6 @@ class FeedSourceCreate(BaseModel):
|
|||
config: Optional[dict] = None
|
||||
|
||||
|
||||
class FeedSourceUpdate(BaseModel):
|
||||
"""PATCH /api/sources/{id} — only these keys may be set."""
|
||||
|
||||
model_config = ConfigDict(extra="forbid")
|
||||
|
||||
name: Optional[str] = None
|
||||
url: Optional[str] = None
|
||||
config: Optional[dict] = None
|
||||
enabled: Optional[bool] = None
|
||||
|
||||
|
||||
class FeedSourceOut(BaseModel):
|
||||
id: UUID
|
||||
name: str
|
||||
|
|
@ -284,70 +273,6 @@ class NewsSummaryOut(BaseModel):
|
|||
id: int
|
||||
summary_text: str
|
||||
batch_timestamp: datetime
|
||||
model: Optional[str] = None
|
||||
kind: Optional[str] = None
|
||||
|
||||
|
||||
class NewsTickerItemOut(BaseModel):
|
||||
"""One flagged ticker row as exposed by GET /api/news/ticker."""
|
||||
|
||||
id: int
|
||||
headline: str
|
||||
importance: str
|
||||
location_name: Optional[str] = None
|
||||
url: Optional[str] = None
|
||||
created_at: datetime
|
||||
|
||||
|
||||
class NewsMapItemOut(BaseModel):
|
||||
"""One flagged map pin as exposed by GET /api/news/map."""
|
||||
|
||||
id: int
|
||||
headline: str
|
||||
importance: str
|
||||
location_name: Optional[str] = None
|
||||
lat: float
|
||||
lon: float
|
||||
location_confidence: Optional[str] = None
|
||||
category: Optional[str] = None
|
||||
url: Optional[str] = None
|
||||
created_at: datetime
|
||||
|
||||
|
||||
class NewsModelId(BaseModel):
|
||||
"""One model id as exposed by GET /api/news/models."""
|
||||
|
||||
id: str
|
||||
|
||||
|
||||
class NewsModelsOut(BaseModel):
|
||||
"""Catalog for the summarizer model selector."""
|
||||
|
||||
source: str
|
||||
models: list[NewsModelId]
|
||||
|
||||
|
||||
class SettingsIn(BaseModel):
|
||||
"""Body for PUT /api/settings. ``nous_base_url`` is not writable."""
|
||||
|
||||
summary_model: str = Field(..., min_length=1, max_length=128)
|
||||
|
||||
@field_validator("summary_model")
|
||||
@classmethod
|
||||
def summary_model_not_blank(cls, v: str) -> str:
|
||||
stripped = v.strip()
|
||||
if not stripped:
|
||||
raise ValueError("summary_model must be 1–128 chars, not whitespace-only")
|
||||
if len(stripped) > 128:
|
||||
raise ValueError("summary_model must be 1–128 chars, not whitespace-only")
|
||||
return stripped
|
||||
|
||||
|
||||
class SettingsOut(BaseModel):
|
||||
"""Current summarizer settings. ``nous_base_url`` is read-only."""
|
||||
|
||||
summary_model: str
|
||||
nous_base_url: str
|
||||
|
||||
|
||||
# ─── Aggregations ────────────────────────────────────────────────────────
|
||||
|
|
@ -375,46 +300,3 @@ class DashboardSummary(BaseModel):
|
|||
sentiment: SentimentSummary
|
||||
top_entities: list[EntityOut]
|
||||
|
||||
|
||||
class VesselBboxUpdate(BaseModel):
|
||||
"""Retune the server-side AISStream subscription to a client viewport box.
|
||||
|
||||
``bbox`` is "minlon,minlat,maxlon,maxlat" (Leaflet order). ``None``/empty
|
||||
resets to the env AISSTREAM_BBOX default.
|
||||
"""
|
||||
|
||||
bbox: str | None = None
|
||||
|
||||
|
||||
class GeofenceCreate(BaseModel):
|
||||
name: str
|
||||
geojson: dict
|
||||
active: bool = True
|
||||
|
||||
|
||||
class GeofenceUpdate(BaseModel):
|
||||
name: Optional[str] = None
|
||||
geojson: Optional[dict] = None
|
||||
active: Optional[bool] = None
|
||||
|
||||
|
||||
class ConflictZoneOut(BaseModel):
|
||||
"""One curated conflict theatre as exposed by GET /api/conflicts."""
|
||||
|
||||
id: str
|
||||
label: str
|
||||
severity: Literal["war", "high", "elevated"]
|
||||
lat: float
|
||||
lon: float
|
||||
description: str
|
||||
eventCount: int
|
||||
lastUpdated: Optional[datetime] = None
|
||||
|
||||
|
||||
class ConflictsOut(BaseModel):
|
||||
"""Response envelope for GET /api/conflicts."""
|
||||
|
||||
zones: list[ConflictZoneOut]
|
||||
timestamp: datetime
|
||||
|
||||
|
||||
|
|
|
|||
|
|
@ -1,219 +0,0 @@
|
|||
"""OSINT Dashboard — non-secret app settings (keyv-style Postgres table).
|
||||
|
||||
Model choice lives here so the summarizer container can read it from Postgres.
|
||||
Only whitelisted names are stored — this is not a generic dump.
|
||||
|
||||
Storage: the table is created lazily with ``CREATE TABLE IF NOT EXISTS`` on
|
||||
first use in each process (same bootstrap pattern as ``api_keys``).
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import asyncio
|
||||
import os
|
||||
import time
|
||||
from datetime import datetime, timezone
|
||||
|
||||
import httpx
|
||||
from sqlalchemy import Column, DateTime, String, Table, Text, func, select, text
|
||||
|
||||
import keystore
|
||||
from database import async_session, engine, metadata
|
||||
|
||||
DEFAULT_NOUS_BASE_URL = "https://inference-api.nousresearch.com/v1"
|
||||
DEFAULT_SUMMARY_MODEL = "Hermes-4.3-36B"
|
||||
SETTING_SUMMARY_MODEL = "SUMMARY_MODEL"
|
||||
ALLOWED_SETTINGS = frozenset({SETTING_SUMMARY_MODEL})
|
||||
MODELS_CACHE_TTL_S = 600.0
|
||||
MODELS_TIMEOUT_S = 8.0
|
||||
DEFAULT_MODELS_USER_AGENT = "osint-dashboard-news-summarizer"
|
||||
|
||||
FALLBACK_MODELS = [
|
||||
"Hermes-4.3-36B",
|
||||
"Hermes-4-70B",
|
||||
"google/gemini-2.5-flash",
|
||||
"anthropic/claude-haiku-4.5",
|
||||
"openai/gpt-4.1-mini",
|
||||
"x-ai/grok-4",
|
||||
]
|
||||
|
||||
app_settings = Table(
|
||||
"app_settings",
|
||||
metadata,
|
||||
Column("name", String(128), primary_key=True),
|
||||
Column("value", Text, nullable=False),
|
||||
Column("updated_at", DateTime(timezone=True), server_default=func.now(), nullable=False),
|
||||
)
|
||||
|
||||
_CREATE_TABLE_SQL = text(
|
||||
"""
|
||||
CREATE TABLE IF NOT EXISTS app_settings (
|
||||
name VARCHAR(128) PRIMARY KEY,
|
||||
value TEXT NOT NULL,
|
||||
updated_at TIMESTAMPTZ NOT NULL DEFAULT now()
|
||||
)
|
||||
"""
|
||||
)
|
||||
|
||||
_ensure_lock = asyncio.Lock()
|
||||
_ensured = False
|
||||
_models_cache: tuple[float, dict] | None = None
|
||||
|
||||
|
||||
class SettingsError(ValueError):
|
||||
"""Raised when a setting name or value fails validation."""
|
||||
|
||||
|
||||
async def ensure_app_settings_table() -> None:
|
||||
"""Create the app_settings table if it doesn't exist (idempotent, per process)."""
|
||||
global _ensured
|
||||
if _ensured:
|
||||
return
|
||||
async with _ensure_lock:
|
||||
if _ensured:
|
||||
return
|
||||
async with engine.begin() as conn:
|
||||
await conn.execute(_CREATE_TABLE_SQL)
|
||||
_ensured = True
|
||||
|
||||
|
||||
def nous_base_url() -> str:
|
||||
"""Read-only Nous inference base URL (env, never writable from the UI)."""
|
||||
raw = (os.getenv("NOUS_BASE_URL") or "").strip().rstrip("/")
|
||||
return raw or DEFAULT_NOUS_BASE_URL
|
||||
|
||||
|
||||
def _validate_summary_model(value: str) -> str:
|
||||
stripped = (value or "").strip()
|
||||
if not stripped or len(stripped) > 128:
|
||||
raise SettingsError("summary_model must be 1–128 chars, not whitespace-only")
|
||||
return stripped
|
||||
|
||||
|
||||
async def get_summary_model() -> str:
|
||||
"""Stored SUMMARY_MODEL, else env, else Hermes-4.3-36B."""
|
||||
await ensure_app_settings_table()
|
||||
async with async_session() as session:
|
||||
row = (
|
||||
await session.execute(
|
||||
select(app_settings).where(app_settings.c.name == SETTING_SUMMARY_MODEL)
|
||||
)
|
||||
).mappings().one_or_none()
|
||||
if row and row["value"]:
|
||||
return row["value"]
|
||||
return os.getenv("SUMMARY_MODEL", DEFAULT_SUMMARY_MODEL)
|
||||
|
||||
|
||||
async def set_summary_model(value: str) -> dict:
|
||||
"""Upsert SUMMARY_MODEL and return the public settings payload."""
|
||||
value = _validate_summary_model(value)
|
||||
now = datetime.now(timezone.utc)
|
||||
|
||||
await ensure_app_settings_table()
|
||||
async with async_session() as session:
|
||||
existing = (
|
||||
await session.execute(
|
||||
select(app_settings).where(app_settings.c.name == SETTING_SUMMARY_MODEL)
|
||||
)
|
||||
).mappings().one_or_none()
|
||||
if existing:
|
||||
await session.execute(
|
||||
app_settings.update()
|
||||
.where(app_settings.c.name == SETTING_SUMMARY_MODEL)
|
||||
.values(value=value, updated_at=now)
|
||||
)
|
||||
else:
|
||||
await session.execute(
|
||||
app_settings.insert().values(
|
||||
name=SETTING_SUMMARY_MODEL, value=value, updated_at=now
|
||||
)
|
||||
)
|
||||
await session.commit()
|
||||
return await get_app_settings()
|
||||
|
||||
|
||||
async def get_app_settings() -> dict:
|
||||
return {
|
||||
"summary_model": await get_summary_model(),
|
||||
"nous_base_url": nous_base_url(),
|
||||
}
|
||||
|
||||
|
||||
def _fallback_payload() -> dict:
|
||||
return {
|
||||
"source": "fallback",
|
||||
"models": [{"id": mid} for mid in FALLBACK_MODELS],
|
||||
}
|
||||
|
||||
|
||||
async def _nous_api_key() -> str | None:
|
||||
"""Keystore first, then env. Any lookup failure is treated as missing."""
|
||||
try:
|
||||
stored = await keystore.get_api_key("NOUS_API_KEY")
|
||||
except Exception:
|
||||
stored = None
|
||||
if stored and str(stored).strip():
|
||||
return str(stored).strip()
|
||||
env = (os.getenv("NOUS_API_KEY") or "").strip()
|
||||
return env or None
|
||||
|
||||
|
||||
def _models_user_agent() -> str:
|
||||
return os.getenv("OSINT_USER_AGENT") or DEFAULT_MODELS_USER_AGENT
|
||||
|
||||
|
||||
def _parse_models_payload(body: object) -> list[dict[str, str]]:
|
||||
if isinstance(body, dict):
|
||||
raw = body.get("data", body.get("models", []))
|
||||
elif isinstance(body, list):
|
||||
raw = body
|
||||
else:
|
||||
raw = []
|
||||
out: list[dict[str, str]] = []
|
||||
for item in raw or []:
|
||||
if isinstance(item, str) and item.strip():
|
||||
out.append({"id": item.strip()})
|
||||
elif isinstance(item, dict):
|
||||
mid = item.get("id") or item.get("name")
|
||||
if mid:
|
||||
out.append({"id": str(mid)})
|
||||
return out
|
||||
|
||||
|
||||
async def _http_get(url: str, *, headers: dict[str, str], timeout: float) -> httpx.Response:
|
||||
async with httpx.AsyncClient(timeout=timeout, headers=headers) as client:
|
||||
return await client.get(url)
|
||||
|
||||
|
||||
async def list_models() -> dict:
|
||||
"""Live ``GET {base}/models`` when a key is present; otherwise curated fallback.
|
||||
|
||||
Never raises to the caller for missing key or upstream failure.
|
||||
"""
|
||||
global _models_cache
|
||||
key = await _nous_api_key()
|
||||
if not key:
|
||||
return _fallback_payload()
|
||||
|
||||
now = time.monotonic()
|
||||
hit = _models_cache
|
||||
if hit and now - hit[0] < MODELS_CACHE_TTL_S:
|
||||
return hit[1]
|
||||
|
||||
url = f"{nous_base_url()}/models"
|
||||
headers = {
|
||||
"Authorization": f"Bearer {key}",
|
||||
"User-Agent": _models_user_agent(),
|
||||
"Accept": "application/json",
|
||||
}
|
||||
try:
|
||||
resp = await _http_get(url, headers=headers, timeout=MODELS_TIMEOUT_S)
|
||||
resp.raise_for_status()
|
||||
models = _parse_models_payload(resp.json())
|
||||
if not models:
|
||||
return _fallback_payload()
|
||||
payload = {"source": "live", "models": models}
|
||||
_models_cache = (now, payload)
|
||||
return payload
|
||||
except Exception:
|
||||
return _fallback_payload()
|
||||
334
app/sources.py
334
app/sources.py
|
|
@ -11,8 +11,7 @@ import httpx
|
|||
import feedparser
|
||||
import nats
|
||||
|
||||
from config import NATS_URL, OSINT_USER_AGENT
|
||||
from upstream_cache import rss_cache
|
||||
from config import NATS_URL
|
||||
|
||||
logger = logging.getLogger("osint.sources")
|
||||
|
||||
|
|
@ -26,80 +25,27 @@ def _parse_rfc822(date_str: object) -> str | None:
|
|||
except (ValueError, TypeError):
|
||||
return None
|
||||
|
||||
|
||||
# NATS connection
|
||||
NATS_URLS = NATS_URL
|
||||
_nc = None
|
||||
|
||||
|
||||
async def _jetstream():
|
||||
"""Reuse one NATS connection across publishes (no connect/close per event)."""
|
||||
global _nc
|
||||
if _nc is None or _nc.is_closed:
|
||||
_nc = await nats.connect(NATS_URLS)
|
||||
return _nc.jetstream()
|
||||
|
||||
|
||||
async def publish_event(subject: str, event: dict):
|
||||
"""Publish an event to NATS JetStream."""
|
||||
js = await _jetstream()
|
||||
nc = await nats.connect(NATS_URLS)
|
||||
js = nc.jetstream()
|
||||
await js.publish(subject, json.dumps(event).encode())
|
||||
await nc.close()
|
||||
logger.debug("Published event to %s", subject)
|
||||
|
||||
|
||||
def event_dedup_key(msg: dict) -> str | None:
|
||||
"""Natural key for generic events. URL when present; else None (always insert)."""
|
||||
url = msg.get("url")
|
||||
if not isinstance(url, str):
|
||||
return None
|
||||
url = url.strip()
|
||||
return url or None
|
||||
|
||||
|
||||
async def existing_event_urls(urls: list[str]) -> set[str]:
|
||||
"""URLs already claimed in event_dedup. Empty input -> empty set."""
|
||||
if not urls:
|
||||
return set()
|
||||
from sqlalchemy import select
|
||||
|
||||
from database import async_session
|
||||
from models import event_dedup as event_dedup_table
|
||||
|
||||
async with async_session() as session:
|
||||
result = await session.execute(
|
||||
select(event_dedup_table.c.url).where(event_dedup_table.c.url.in_(urls))
|
||||
)
|
||||
return {row[0] for row in result}
|
||||
|
||||
|
||||
async def _publish_unknown(subject: str, events: list[dict]) -> int:
|
||||
"""Publish only events whose URL is not already in event_dedup."""
|
||||
keys = [event_dedup_key(e) for e in events]
|
||||
known = await existing_event_urls([k for k in keys if k])
|
||||
published = 0
|
||||
for event, key in zip(events, keys):
|
||||
if key and key in known:
|
||||
continue
|
||||
await publish_event(subject, event)
|
||||
published += 1
|
||||
return published
|
||||
|
||||
|
||||
def _ua_headers() -> dict[str, str]:
|
||||
return {"User-Agent": OSINT_USER_AGENT}
|
||||
|
||||
|
||||
# ─── RSS Feed Ingestor ──────────────────────────────────────────────────
|
||||
|
||||
async def ingest_rss_feed(feed_url: str, source_id: str | None = None):
|
||||
async def ingest_rss_feed(feed_url: str):
|
||||
"""Fetch and parse an RSS feed, publish items to NATS."""
|
||||
text = rss_cache.get(feed_url)
|
||||
if text is None:
|
||||
async with httpx.AsyncClient(timeout=30, headers=_ua_headers()) as client:
|
||||
resp = await client.get(feed_url)
|
||||
resp.raise_for_status()
|
||||
text = resp.text
|
||||
rss_cache[feed_url] = text
|
||||
feed = feedparser.parse(text)
|
||||
async with httpx.AsyncClient(timeout=30) as client:
|
||||
resp = await client.get(feed_url)
|
||||
resp.raise_for_status()
|
||||
feed = feedparser.parse(resp.text)
|
||||
|
||||
count = 0
|
||||
for entry in feed.entries[:100]: # max 100 per run
|
||||
|
|
@ -125,80 +71,51 @@ async def ingest_rss_feed(feed_url: str, source_id: str | None = None):
|
|||
return count
|
||||
|
||||
|
||||
# ─── GDELT 2.0 DOC API ──────────────────────────────────────────────────
|
||||
# ─── GDELT 2.0 Ingestor ─────────────────────────────────────────────────
|
||||
|
||||
GDELT_API = "https://api.gdeltproject.org/api/v2/doc/doc"
|
||||
GDELT_DEFAULT_QUERY = '(unrest OR protest OR outage OR cyber OR "power outage")'
|
||||
|
||||
|
||||
def gdelt_params(query: str = "", max_articles: int = 50) -> dict[str, str]:
|
||||
"""DOC 2.0 query string (not the retired gdeltv2 ``search`` param)."""
|
||||
q = (query or "").strip() or GDELT_DEFAULT_QUERY
|
||||
return {
|
||||
"query": q,
|
||||
"mode": "ArtList",
|
||||
"format": "json",
|
||||
"maxrecords": str(int(max_articles)),
|
||||
"timespan": "1d",
|
||||
}
|
||||
|
||||
|
||||
def _parse_gdelt_seendate(value: object) -> str:
|
||||
if isinstance(value, str) and len(value) >= 15:
|
||||
try:
|
||||
return datetime.strptime(value[:15], "%Y%m%dT%H%M%S").replace(
|
||||
tzinfo=timezone.utc
|
||||
).isoformat()
|
||||
except ValueError:
|
||||
pass
|
||||
return datetime.now(timezone.utc).isoformat()
|
||||
|
||||
|
||||
def parse_gdelt_articles(data: dict) -> list[dict]:
|
||||
events = []
|
||||
for article in data.get("articles") or []:
|
||||
if not isinstance(article, dict):
|
||||
continue
|
||||
url = article.get("url")
|
||||
if not url:
|
||||
continue
|
||||
events.append({
|
||||
"source_type": "gdel-t2",
|
||||
"title": article.get("title"),
|
||||
"body": article.get("domain") or article.get("language"),
|
||||
"url": url,
|
||||
"location_name": article.get("sourcecountry"),
|
||||
"source_timestamp": _parse_gdelt_seendate(article.get("seendate")),
|
||||
"tags": [t for t in (article.get("language"), article.get("sourcecountry")) if t],
|
||||
"raw": article,
|
||||
})
|
||||
return events
|
||||
GDELT_API = "https://api.gdeltproject.org/gdeltv2"
|
||||
|
||||
|
||||
async def ingest_gdelt(query: str = "", max_articles: int = 50):
|
||||
"""Fetch articles from the GDELT DOC 2.0 API."""
|
||||
params = gdelt_params(query=query, max_articles=max_articles)
|
||||
data: dict = {"articles": []}
|
||||
async with httpx.AsyncClient(
|
||||
timeout=60, headers=_ua_headers(), follow_redirects=True,
|
||||
) as client:
|
||||
try:
|
||||
resp = await client.get(GDELT_API, params=params)
|
||||
resp.raise_for_status()
|
||||
data = resp.json()
|
||||
except (httpx.TransportError, httpx.HTTPStatusError) as exc:
|
||||
# gdeltproject.org certs have expired in the wild; HTTP fallback.
|
||||
logger.warning("GDELT HTTPS failed (%s); retrying HTTP", exc)
|
||||
http_url = GDELT_API.replace("https://", "http://", 1)
|
||||
resp = await client.get(http_url, params=params)
|
||||
resp.raise_for_status()
|
||||
data = resp.json()
|
||||
"""Fetch articles from GDELT 2.0 API."""
|
||||
params = {
|
||||
"mode": "artlist",
|
||||
"format": "json",
|
||||
"maxrecords": max_articles,
|
||||
"mode": "artlist",
|
||||
}
|
||||
if query:
|
||||
params["search"] = query
|
||||
|
||||
events = parse_gdelt_articles(data if isinstance(data, dict) else {})
|
||||
for event in events:
|
||||
async with httpx.AsyncClient(timeout=60) as client:
|
||||
resp = await client.get(GDELT_API, params=params)
|
||||
resp.raise_for_status()
|
||||
data = resp.json()
|
||||
|
||||
count = 0
|
||||
for article in data.get("articles", []):
|
||||
event = {
|
||||
"source_type": "gdel-t2",
|
||||
"title": article.get("title"),
|
||||
"body": article.get("articleBody"),
|
||||
"url": article.get("url"),
|
||||
"sentiment_score": _parse_gdelt_tone(article.get("Tone", "0")),
|
||||
"location_lat": article.get("Latitude"),
|
||||
"location_lon": article.get("Longitude"),
|
||||
"location_name": article.get("Location"),
|
||||
"source_timestamp": article.get("FirstCreated"),
|
||||
"entities": [
|
||||
{"name": e.get("Topic"), "type": "topic"}
|
||||
for e in article.get("Mentions", [])
|
||||
if e.get("Topic")
|
||||
],
|
||||
"raw": article,
|
||||
}
|
||||
await publish_event("events.gdelt", event)
|
||||
logger.info("Ingested %d articles from GDELT", len(events))
|
||||
return len(events)
|
||||
count += 1
|
||||
|
||||
logger.info("Ingested %d articles from GDELT", count)
|
||||
return count
|
||||
|
||||
|
||||
def _parse_gdelt_tone(tone: str) -> float | None:
|
||||
|
|
@ -215,45 +132,34 @@ def _parse_gdelt_tone(tone: str) -> float | None:
|
|||
USGS_API = "https://earthquake.usgs.gov/earthquakes/feed/v1.0/summary/all_hour.geojson"
|
||||
|
||||
|
||||
def parse_usgs_feature(feature: dict) -> dict:
|
||||
"""Map one USGS GeoJSON feature, keeping the stable event id."""
|
||||
props = feature.get("properties") or {}
|
||||
geometry = (feature.get("geometry") or {}).get("coordinates") or []
|
||||
usgs_id = feature.get("id")
|
||||
url = props.get("url") or (
|
||||
f"https://earthquake.usgs.gov/earthquakes/eventpage/{usgs_id}" if usgs_id else None
|
||||
)
|
||||
raw = dict(props)
|
||||
raw["usgs_id"] = usgs_id
|
||||
return {
|
||||
"source_type": "earthquake",
|
||||
"title": props.get("title"),
|
||||
"body": props.get("description"),
|
||||
"url": url,
|
||||
"location_lat": geometry[1] if len(geometry) > 1 else None,
|
||||
"location_lon": geometry[0] if len(geometry) > 0 else None,
|
||||
"location_name": props.get("place"),
|
||||
"sentiment_label": "neutral",
|
||||
"tags": [f"magnitude:{props.get('mag')}"] if props.get("mag") else [],
|
||||
"source_timestamp": (
|
||||
datetime.utcfromtimestamp(props.get("time", 0) / 1000)
|
||||
.replace(tzinfo=timezone.utc)
|
||||
.isoformat()
|
||||
),
|
||||
"raw": raw,
|
||||
}
|
||||
|
||||
|
||||
async def ingest_earthquakes():
|
||||
"""Fetch recent earthquakes from USGS."""
|
||||
async with httpx.AsyncClient(timeout=30, headers=_ua_headers()) as client:
|
||||
async with httpx.AsyncClient(timeout=30) as client:
|
||||
resp = await client.get(USGS_API)
|
||||
resp.raise_for_status()
|
||||
data = resp.json()
|
||||
|
||||
count = 0
|
||||
for feature in data.get("features", []):
|
||||
event = parse_usgs_feature(feature)
|
||||
props = feature.get("properties", {})
|
||||
geometry = feature.get("geometry", {}).get("coordinates", [])
|
||||
event = {
|
||||
"source_type": "earthquake",
|
||||
"title": props.get("title"),
|
||||
"body": props.get("description"),
|
||||
"url": props.get("url"),
|
||||
"location_lat": geometry[1] if len(geometry) > 1 else None,
|
||||
"location_lon": geometry[0] if len(geometry) > 0 else None,
|
||||
"location_name": props.get("place"),
|
||||
"sentiment_label": "neutral",
|
||||
"tags": [f"magnitude:{props.get('mag')}"] if props.get("mag") else [],
|
||||
"source_timestamp": (
|
||||
datetime.utcfromtimestamp(props.get("time", 0) / 1000)
|
||||
.replace(tzinfo=timezone.utc)
|
||||
.isoformat()
|
||||
),
|
||||
"raw": props,
|
||||
}
|
||||
await publish_event("events.earthquake", event)
|
||||
count += 1
|
||||
|
||||
|
|
@ -261,108 +167,6 @@ async def ingest_earthquakes():
|
|||
return count
|
||||
|
||||
|
||||
# ─── NASA EONET v3 ──────────────────────────────────────────────────────
|
||||
|
||||
EONET_API = "https://eonet.gsfc.nasa.gov/api/v3/events"
|
||||
|
||||
|
||||
def parse_eonet_events(payload: dict) -> list[dict]:
|
||||
events = []
|
||||
for item in payload.get("events") or []:
|
||||
if not isinstance(item, dict):
|
||||
continue
|
||||
eid = item.get("id")
|
||||
geoms = item.get("geometry") or []
|
||||
point = None
|
||||
for g in geoms:
|
||||
if isinstance(g, dict) and g.get("type") == "Point":
|
||||
point = g
|
||||
if point is None:
|
||||
continue
|
||||
coords = point.get("coordinates") or []
|
||||
if len(coords) < 2:
|
||||
continue
|
||||
lon, lat = float(coords[0]), float(coords[1])
|
||||
cats = item.get("categories") or []
|
||||
tags = []
|
||||
for c in cats:
|
||||
if isinstance(c, dict) and c.get("id"):
|
||||
tags.append(str(c["id"]))
|
||||
url = item.get("link") or (f"https://eonet.gsfc.nasa.gov/api/v3/events/{eid}" if eid else None)
|
||||
ts = point.get("date") or datetime.now(timezone.utc).isoformat()
|
||||
events.append({
|
||||
"source_type": "disaster",
|
||||
"title": item.get("title"),
|
||||
"body": ", ".join(tags) if tags else None,
|
||||
"url": url,
|
||||
"location_lat": lat,
|
||||
"location_lon": lon,
|
||||
"location_name": item.get("title"),
|
||||
"tags": tags,
|
||||
"source_timestamp": ts,
|
||||
"raw": {**item, "eonet_id": eid},
|
||||
})
|
||||
return events
|
||||
|
||||
|
||||
async def ingest_eonet():
|
||||
"""Volcanoes, storms, floods, drought — gaps USGS/FIRMS don't cover."""
|
||||
async with httpx.AsyncClient(timeout=30, headers=_ua_headers()) as client:
|
||||
resp = await client.get(EONET_API, params={"status": "open", "limit": 100})
|
||||
resp.raise_for_status()
|
||||
data = resp.json()
|
||||
events = parse_eonet_events(data if isinstance(data, dict) else {})
|
||||
published = await _publish_unknown("events.disaster", events)
|
||||
logger.info("Ingested %d EONET events (%d already known)", published, len(events) - published)
|
||||
return published
|
||||
|
||||
|
||||
# ─── CISA KEV ───────────────────────────────────────────────────────────
|
||||
|
||||
CISA_KEV_API = (
|
||||
"https://www.cisa.gov/sites/default/files/feeds/known_exploited_vulnerabilities.json"
|
||||
)
|
||||
|
||||
|
||||
def parse_cisa_kev(payload: dict) -> list[dict]:
|
||||
events = []
|
||||
for row in payload.get("vulnerabilities") or []:
|
||||
if not isinstance(row, dict):
|
||||
continue
|
||||
cve = row.get("cveID")
|
||||
if not cve:
|
||||
continue
|
||||
title = row.get("vulnerabilityName") or cve
|
||||
vendor = row.get("vendorProject") or ""
|
||||
product = row.get("product") or ""
|
||||
events.append({
|
||||
"source_type": "disaster",
|
||||
"title": f"{cve}: {title}",
|
||||
"body": row.get("shortDescription") or f"{vendor} {product}".strip(),
|
||||
"url": f"https://nvd.nist.gov/vuln/detail/{cve}",
|
||||
"location_lat": None,
|
||||
"location_lon": None,
|
||||
"location_name": None,
|
||||
"tags": ["cisa-kev", cve, "ransomware" if row.get("knownRansomwareCampaignUse") == "Known" else None],
|
||||
"source_timestamp": row.get("dateAdded") or datetime.now(timezone.utc).isoformat(),
|
||||
"raw": {**row, "cveID": cve},
|
||||
})
|
||||
events[-1]["tags"] = [t for t in events[-1]["tags"] if t]
|
||||
return events
|
||||
|
||||
|
||||
async def ingest_cisa_kev():
|
||||
"""Exploited-in-the-wild CVEs. No fake map coords — ticker/events only."""
|
||||
async with httpx.AsyncClient(timeout=30, headers=_ua_headers(), follow_redirects=True) as client:
|
||||
resp = await client.get(CISA_KEV_API)
|
||||
resp.raise_for_status()
|
||||
data = resp.json()
|
||||
events = parse_cisa_kev(data if isinstance(data, dict) else {})
|
||||
published = await _publish_unknown("events.disaster", events)
|
||||
logger.info("Ingested %d CISA KEV rows (%d already known)", published, len(events) - published)
|
||||
return published
|
||||
|
||||
|
||||
# ─── Social Signals (Twitter/X-like placeholder) ────────────────────────
|
||||
|
||||
async def ingest_social_signals(query: str = "", max_items: int = 50):
|
||||
|
|
|
|||
File diff suppressed because it is too large
Load diff
244
app/tracks.py
244
app/tracks.py
|
|
@ -1,244 +0,0 @@
|
|||
"""Timescale 1-minute track rollups for DVR playback.
|
||||
|
||||
Live overlays stay in memory. Historical `?timestamp=` reads the 1-minute
|
||||
continuous aggregates (or an in-process downsample when the DB is down).
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
from datetime import datetime, timedelta, timezone
|
||||
from typing import Any, Literal
|
||||
|
||||
from sqlalchemy import text
|
||||
|
||||
from database import async_session
|
||||
from live_layers import parse_bbox, to_marker
|
||||
|
||||
|
||||
TRACK_BUCKET = "1 minute"
|
||||
Kind = Literal["vessel", "aircraft"]
|
||||
|
||||
_RAW_TABLE = {
|
||||
"vessel": "vessel_positions",
|
||||
"aircraft": "aircraft_positions",
|
||||
}
|
||||
_CAGG = {
|
||||
"vessel": "vessel_tracks_1min",
|
||||
"aircraft": "aircraft_tracks_1min",
|
||||
}
|
||||
_ID_COL = {
|
||||
"vessel": "mmsi",
|
||||
"aircraft": "hex",
|
||||
}
|
||||
|
||||
# Last persist time per entity so AIS/ADS-B does not write every frame.
|
||||
_last_write: dict[tuple[str, str], datetime] = {}
|
||||
_MIN_WRITE_GAP = timedelta(seconds=20)
|
||||
|
||||
|
||||
def minute_bucket(ts: datetime) -> datetime:
|
||||
if ts.tzinfo is None:
|
||||
ts = ts.replace(tzinfo=timezone.utc)
|
||||
return ts.replace(second=0, microsecond=0)
|
||||
|
||||
|
||||
def downsample_tracks(rows: list[dict]) -> list[dict]:
|
||||
"""Last sample per id per 1-minute bucket (mirrors the CAGG)."""
|
||||
last: dict[tuple[str, datetime], dict] = {}
|
||||
for row in rows:
|
||||
rid = str(row.get("id") or "")
|
||||
ts = row.get("ts")
|
||||
if not rid or not isinstance(ts, datetime):
|
||||
continue
|
||||
bucket = minute_bucket(ts)
|
||||
key = (rid, bucket)
|
||||
prev = last.get(key)
|
||||
if prev is None or ts >= prev["ts"]:
|
||||
last[key] = {**row, "id": rid, "bucket": bucket, "ts": ts}
|
||||
out = []
|
||||
for (_id, bucket), row in last.items():
|
||||
out.append({
|
||||
"id": row["id"],
|
||||
"bucket": bucket,
|
||||
"lat": row.get("lat"),
|
||||
"lon": row.get("lon"),
|
||||
"heading": row.get("heading"),
|
||||
"speed": row.get("speed"),
|
||||
"label": row.get("label"),
|
||||
})
|
||||
return out
|
||||
|
||||
|
||||
def positions_at_timestamp(rows: list[dict], ts: datetime) -> list[dict]:
|
||||
"""Positions whose 1-minute bucket equals floor(ts)."""
|
||||
want = minute_bucket(ts)
|
||||
picked = [r for r in downsample_tracks(rows) if r["bucket"] == want]
|
||||
return [
|
||||
to_marker(
|
||||
r["id"], r.get("lat"), r.get("lon"),
|
||||
heading=r.get("heading"), speed=r.get("speed"),
|
||||
label=r.get("label") or r["id"],
|
||||
)
|
||||
for r in picked
|
||||
if r.get("lat") is not None and r.get("lon") is not None
|
||||
]
|
||||
|
||||
|
||||
def parse_timestamp(value: str | datetime | None) -> datetime | None:
|
||||
if value is None or value == "":
|
||||
return None
|
||||
if isinstance(value, datetime):
|
||||
ts = value
|
||||
else:
|
||||
raw = str(value).strip().replace("Z", "+00:00")
|
||||
ts = datetime.fromisoformat(raw)
|
||||
if ts.tzinfo is None:
|
||||
ts = ts.replace(tzinfo=timezone.utc)
|
||||
return ts
|
||||
|
||||
|
||||
async def record_position(kind: Kind, marker: dict, ts: datetime | None = None) -> bool:
|
||||
"""Insert one sample into the raw hypertable (rate-limited)."""
|
||||
vid = str(marker.get("id") or "")
|
||||
lat, lon = marker.get("lat"), marker.get("lon")
|
||||
if not vid or lat is None or lon is None:
|
||||
return False
|
||||
now = ts or datetime.now(timezone.utc)
|
||||
key = (kind, vid)
|
||||
prev = _last_write.get(key)
|
||||
if prev is not None and now - prev < _MIN_WRITE_GAP:
|
||||
return False
|
||||
_last_write[key] = now
|
||||
table = _RAW_TABLE[kind]
|
||||
id_col = _ID_COL[kind]
|
||||
extra = marker.get("extra") or {}
|
||||
try:
|
||||
async with async_session() as session:
|
||||
await session.execute(
|
||||
text(
|
||||
f"""
|
||||
INSERT INTO {table} ({id_col}, ts, lat, lon, heading, speed, label, extra)
|
||||
VALUES (:id, :ts, :lat, :lon, :heading, :speed, :label, CAST(:extra AS jsonb))
|
||||
ON CONFLICT ({id_col}, ts) DO NOTHING
|
||||
"""
|
||||
),
|
||||
{
|
||||
"id": vid,
|
||||
"ts": now,
|
||||
"lat": float(lat),
|
||||
"lon": float(lon),
|
||||
"heading": marker.get("heading"),
|
||||
"speed": marker.get("speed"),
|
||||
"label": marker.get("label") or vid,
|
||||
"extra": json.dumps(extra),
|
||||
},
|
||||
)
|
||||
await session.commit()
|
||||
return True
|
||||
except Exception:
|
||||
return False
|
||||
|
||||
|
||||
async def fetch_positions_at(
|
||||
kind: Kind,
|
||||
ts: datetime,
|
||||
bbox: str | None = None,
|
||||
limit: int = 2000,
|
||||
) -> list[dict]:
|
||||
"""Read the 1-minute CAGG for the bucket containing ``ts``."""
|
||||
bucket = minute_bucket(ts)
|
||||
table = _CAGG[kind]
|
||||
id_col = _ID_COL[kind]
|
||||
where = "bucket = :bucket"
|
||||
params: dict[str, Any] = {"bucket": bucket, "limit": limit}
|
||||
if bbox:
|
||||
minlon, minlat, maxlon, maxlat = parse_bbox(bbox)
|
||||
where += " AND lon BETWEEN :minlon AND :maxlon AND lat BETWEEN :minlat AND :maxlat"
|
||||
params.update(minlon=minlon, minlat=minlat, maxlon=maxlon, maxlat=maxlat)
|
||||
sql = f"""
|
||||
SELECT {id_col} AS id, lat, lon, heading, speed, label, bucket
|
||||
FROM {table}
|
||||
WHERE {where}
|
||||
LIMIT :limit
|
||||
"""
|
||||
try:
|
||||
async with async_session() as session:
|
||||
rows = (await session.execute(text(sql), params)).mappings().all()
|
||||
points = [
|
||||
to_marker(
|
||||
r["id"], r["lat"], r["lon"],
|
||||
heading=r["heading"], speed=r["speed"],
|
||||
label=r["label"] or r["id"],
|
||||
extra={"bucket": r["bucket"].isoformat() if r["bucket"] else None, "dvr": True},
|
||||
)
|
||||
for r in rows
|
||||
if r["lat"] is not None and r["lon"] is not None
|
||||
]
|
||||
return points
|
||||
except Exception:
|
||||
return []
|
||||
|
||||
|
||||
async def track_range() -> dict:
|
||||
"""Earliest/latest buckets across both CAGGs — slider bounds."""
|
||||
try:
|
||||
async with async_session() as session:
|
||||
row = (await session.execute(text(
|
||||
"""
|
||||
SELECT min(t) AS tmin, max(t) AS tmax FROM (
|
||||
SELECT min(bucket) AS t FROM vessel_tracks_1min
|
||||
UNION ALL SELECT max(bucket) FROM vessel_tracks_1min
|
||||
UNION ALL SELECT min(bucket) FROM aircraft_tracks_1min
|
||||
UNION ALL SELECT max(bucket) FROM aircraft_tracks_1min
|
||||
UNION ALL SELECT min(poll_at) FROM vessels
|
||||
UNION ALL SELECT max(poll_at) FROM vessels
|
||||
) s
|
||||
"""
|
||||
))).mappings().first()
|
||||
if not row or row["tmin"] is None:
|
||||
now = datetime.now(timezone.utc).replace(second=0, microsecond=0)
|
||||
return {"min": (now - timedelta(hours=6)).isoformat(), "max": now.isoformat()}
|
||||
return {
|
||||
"min": row["tmin"].isoformat(),
|
||||
"max": row["tmax"].isoformat(),
|
||||
}
|
||||
except Exception:
|
||||
now = datetime.now(timezone.utc).replace(second=0, microsecond=0)
|
||||
return {"min": (now - timedelta(hours=6)).isoformat(), "max": now.isoformat()}
|
||||
|
||||
|
||||
async def recent_markers(kind: Kind, limit: int = 2000) -> list[dict]:
|
||||
"""Latest raw sample per id — used when in-process last-known is empty."""
|
||||
table = _RAW_TABLE[kind]
|
||||
id_col = _ID_COL[kind]
|
||||
sql = f"""
|
||||
SELECT DISTINCT ON ({id_col})
|
||||
{id_col} AS id, lat, lon, heading, speed, label, extra
|
||||
FROM {table}
|
||||
WHERE ts > now() - interval '15 minutes'
|
||||
ORDER BY {id_col}, ts DESC
|
||||
LIMIT :limit
|
||||
"""
|
||||
try:
|
||||
async with async_session() as session:
|
||||
rows = (await session.execute(text(sql), {"limit": limit})).mappings().all()
|
||||
out = []
|
||||
for r in rows:
|
||||
extra = r.get("extra") or {}
|
||||
if isinstance(extra, str):
|
||||
try:
|
||||
extra = json.loads(extra)
|
||||
except (TypeError, ValueError):
|
||||
extra = {}
|
||||
m = to_marker(
|
||||
r["id"], r["lat"], r["lon"],
|
||||
heading=r["heading"], speed=r["speed"],
|
||||
label=r["label"] or r["id"],
|
||||
extra=extra if isinstance(extra, dict) else {},
|
||||
)
|
||||
if m.get("lat") is not None and m.get("lon") is not None:
|
||||
out.append(m)
|
||||
return out
|
||||
except Exception:
|
||||
return []
|
||||
|
|
@ -1,11 +0,0 @@
|
|||
"""In-process TTL caches for chatty upstreams (FIRMS, RSS). No Redis."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from cachetools import TTLCache
|
||||
|
||||
# FIRMS NRT updates every ~5–10 min; 5 min / 100 keys is enough for bbox×dataset.
|
||||
firms_cache: TTLCache = TTLCache(maxsize=100, ttl=300)
|
||||
|
||||
# News RSS: 1 minute is enough to absorb dashboard double-clicks / retries.
|
||||
rss_cache: TTLCache = TTLCache(maxsize=100, ttl=60)
|
||||
671
app/vesselapi.py
671
app/vesselapi.py
|
|
@ -1,671 +0,0 @@
|
|||
"""VesselAPI REST poller — quota-capped AIS for the Middle East (free tier 150 calls/mo).
|
||||
|
||||
VesselAPI and AISStream are two independent, first-class vessel providers —
|
||||
not a primary/fallback pair. AISStream (WebSocket) owns live US-coast AIS;
|
||||
VesselAPI (REST) covers the Strait of Hormuz (default box) where AISStream
|
||||
has no coverage. Missing one key never disables the other. This worker polls
|
||||
the REST ``GET /v1/location/vessels/bounding-box`` endpoint at most
|
||||
``VESSELAPI_MAX_CALLS_PER_DAY`` (default 5) *successful 2xx* calls per UTC day
|
||||
and upserts the results into the shared ``vessel_last_known`` store.
|
||||
|
||||
Idle (no crash) when VESSELAPI_API_KEY is unset. Never called from the GET
|
||||
/api/vessels path — map pans must not hit upstream. One request per poll,
|
||||
``pagination.limit=50``, never follow ``nextToken``, never send ``filter.sat``.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import asyncio
|
||||
import calendar
|
||||
import json
|
||||
import logging
|
||||
import os
|
||||
from datetime import date, datetime, timezone
|
||||
|
||||
import httpx
|
||||
from sqlalchemy import Column, Date, DateTime, Integer, Table, func, select, text
|
||||
|
||||
from config import (
|
||||
OSINT_USER_AGENT,
|
||||
VESSELAPI_API_KEY,
|
||||
VESSELAPI_BBOX,
|
||||
VESSELAPI_INTERVAL,
|
||||
VESSELAPI_MAX_CALLS_PER_DAY,
|
||||
)
|
||||
from database import async_session, engine, metadata
|
||||
from live_layers import parse_bbox, to_marker, upsert_vessel, vessel_last_known, vessel_lock
|
||||
|
||||
logger = logging.getLogger("osint.vesselapi")
|
||||
|
||||
BASE_URL = "https://api.vesselapi.com/v1"
|
||||
ENDPOINT = f"{BASE_URL}/location/vessels/bounding-box"
|
||||
MAX_SPAN_DEG = 4.0 # |dLat| + |dLon| — VesselAPI 400s above this.
|
||||
PAGE_LIMIT = 50 # pagination.limit; never follow nextToken on the free tier.
|
||||
|
||||
_client: httpx.AsyncClient | None = None
|
||||
_client_lock = asyncio.Lock()
|
||||
|
||||
|
||||
# ── Box parsing / span validation ─────────────────────────────────────────
|
||||
|
||||
class BboxError(ValueError):
|
||||
"""A configured VesselAPI box violates the 4° span rule or is malformed."""
|
||||
|
||||
|
||||
def validate_bbox_span(
|
||||
minlat: float, minlon: float, maxlat: float, maxlon: float,
|
||||
) -> None:
|
||||
"""Reject boxes VesselAPI would 400 on (span > 4°, bad order, bad range)."""
|
||||
if not (-90 <= minlat <= 90 and -90 <= maxlat <= 90
|
||||
and -180 <= minlon <= 180 and -180 <= maxlon <= 180):
|
||||
raise BboxError("coordinates out of range")
|
||||
if minlat >= maxlat or minlon >= maxlon:
|
||||
raise BboxError("bbox must have min < max on both axes")
|
||||
dlat = abs(maxlat - minlat)
|
||||
dlon = abs(maxlon - minlon)
|
||||
if dlat + dlon > MAX_SPAN_DEG:
|
||||
raise BboxError(
|
||||
f"span |dLat|+|dLon| = {dlat + dlon:.2f}° exceeds {MAX_SPAN_DEG}° cap"
|
||||
)
|
||||
|
||||
|
||||
def parse_boxes(raw: str) -> list[tuple[float, float, float, float]]:
|
||||
"""Env format: ``minlat,minlon,maxlat,maxlon[; ...]`` (lat/lon order)."""
|
||||
out: list[tuple[float, float, float, float]] = []
|
||||
for chunk in (raw or "").split(";"):
|
||||
parts = [p.strip() for p in chunk.split(",") if p.strip()]
|
||||
if len(parts) != 4:
|
||||
continue
|
||||
try:
|
||||
minlat = float(parts[0])
|
||||
minlon = float(parts[1])
|
||||
maxlat = float(parts[2])
|
||||
maxlon = float(parts[3])
|
||||
except ValueError:
|
||||
continue
|
||||
out.append((minlat, minlon, maxlat, maxlon))
|
||||
return out
|
||||
|
||||
|
||||
def parse_boxes_validated(raw: str) -> list[tuple[float, float, float, float]]:
|
||||
"""Parse boxes, log + skip any that violate the span/order/range rules."""
|
||||
valid: list[tuple[float, float, float, float]] = []
|
||||
for box in parse_boxes(raw):
|
||||
try:
|
||||
validate_bbox_span(*box)
|
||||
valid.append(box)
|
||||
except BboxError as exc:
|
||||
logger.warning("VesselAPI bbox %r skipped: %s", box, exc)
|
||||
return valid
|
||||
|
||||
|
||||
# ── Position → marker transform ───────────────────────────────────────────
|
||||
|
||||
def _f(value: object) -> float | None:
|
||||
if value is None or value == "":
|
||||
return None
|
||||
try:
|
||||
return float(value)
|
||||
except (TypeError, ValueError):
|
||||
return None
|
||||
|
||||
|
||||
def _s(value: object) -> str | None:
|
||||
if value is None:
|
||||
return None
|
||||
text = str(value).strip()
|
||||
return text or None
|
||||
|
||||
|
||||
def transform_vesselapi_position(obj: dict | None) -> dict | None:
|
||||
"""Map one VesselAPI position object to the shared marker contract.
|
||||
|
||||
Returns None for glitch rows, missing MMSI, or missing coordinates.
|
||||
"""
|
||||
if not obj or not isinstance(obj, dict):
|
||||
return None
|
||||
if obj.get("suspected_glitch") is True:
|
||||
return None
|
||||
mmsi = obj.get("mmsi")
|
||||
if mmsi is None:
|
||||
return None
|
||||
lat = _f(obj.get("latitude"))
|
||||
lon = _f(obj.get("longitude"))
|
||||
if lat is None or lon is None:
|
||||
return None
|
||||
mmsi_s = str(mmsi)
|
||||
name = _s(obj.get("vessel_name") or obj.get("name"))
|
||||
heading = _f(obj.get("heading"))
|
||||
if heading is None:
|
||||
heading = _f(obj.get("cog"))
|
||||
sog = _f(obj.get("sog"))
|
||||
extra: dict = {
|
||||
"src": "vesselapi",
|
||||
"mmsi": mmsi_s,
|
||||
"cog": obj.get("cog"),
|
||||
"sog": obj.get("sog"),
|
||||
"navstat": obj.get("nav_status"),
|
||||
}
|
||||
imo = obj.get("imo")
|
||||
if imo:
|
||||
extra["imo"] = imo
|
||||
dest = _s(obj.get("dest") or obj.get("destination"))
|
||||
if dest:
|
||||
extra["dest"] = dest
|
||||
ts = obj.get("timestamp") or obj.get("processed_timestamp")
|
||||
if ts:
|
||||
extra["timestamp"] = ts
|
||||
return to_marker(
|
||||
mmsi_s, lat, lon,
|
||||
heading=heading,
|
||||
speed=sog,
|
||||
label=name or mmsi_s,
|
||||
extra=extra,
|
||||
)
|
||||
|
||||
|
||||
def transform_vesselapi_payload(payload: dict | None) -> list[dict]:
|
||||
"""Flatten a bounding-box response ``{vessels: [...]}`` to markers."""
|
||||
if not payload or not isinstance(payload, dict):
|
||||
return []
|
||||
rows = payload.get("vessels") or []
|
||||
out = []
|
||||
for row in rows:
|
||||
marker = transform_vesselapi_position(row)
|
||||
if marker:
|
||||
out.append(marker)
|
||||
return out
|
||||
|
||||
|
||||
def utc_day_start(now: datetime) -> datetime:
|
||||
"""Floor ``now`` to 00:00:00 UTC."""
|
||||
if now.tzinfo is None:
|
||||
now = now.replace(tzinfo=timezone.utc)
|
||||
now = now.astimezone(timezone.utc)
|
||||
return now.replace(hour=0, minute=0, second=0, microsecond=0)
|
||||
|
||||
|
||||
def pick_poll_at(poll_times: list[datetime], as_of: datetime) -> datetime | None:
|
||||
"""Latest poll timestamp at or before ``as_of`` (DVR as-of)."""
|
||||
if as_of.tzinfo is None:
|
||||
as_of = as_of.replace(tzinfo=timezone.utc)
|
||||
else:
|
||||
as_of = as_of.astimezone(timezone.utc)
|
||||
eligible: list[datetime] = []
|
||||
for raw in poll_times:
|
||||
ts = raw if raw.tzinfo else raw.replace(tzinfo=timezone.utc)
|
||||
ts = ts.astimezone(timezone.utc)
|
||||
if ts <= as_of:
|
||||
eligible.append(ts)
|
||||
return max(eligible) if eligible else None
|
||||
|
||||
|
||||
def snapshot_as_of(rows: list[dict], as_of: datetime) -> list[dict]:
|
||||
"""Keep only rows from the latest poll_at ≤ ``as_of``."""
|
||||
chosen = pick_poll_at(
|
||||
[r["poll_at"] for r in rows if r.get("poll_at") is not None],
|
||||
as_of,
|
||||
)
|
||||
if chosen is None:
|
||||
return []
|
||||
out = []
|
||||
for row in rows:
|
||||
ts = row.get("poll_at")
|
||||
if ts is None:
|
||||
continue
|
||||
if ts.tzinfo is None:
|
||||
ts = ts.replace(tzinfo=timezone.utc)
|
||||
if ts.astimezone(timezone.utc) == chosen:
|
||||
out.append(row)
|
||||
return out
|
||||
|
||||
|
||||
# ── Durable daily quota (Postgres, survives restarts) ─────────────────────
|
||||
# Mirrors keystore.api_keys: lazy CREATE TABLE IF NOT EXISTS, no alembic fork.
|
||||
|
||||
vesselapi_quota = Table(
|
||||
"vesselapi_quota",
|
||||
metadata,
|
||||
Column("day", Date, primary_key=True),
|
||||
Column("calls", Integer, nullable=False, server_default="0"),
|
||||
Column("remaining", Integer, nullable=True),
|
||||
Column("updated_at", DateTime(timezone=True), server_default=func.now(), nullable=False),
|
||||
)
|
||||
|
||||
_CREATE_QUOTA_SQL = text(
|
||||
"""
|
||||
CREATE TABLE IF NOT EXISTS vesselapi_quota (
|
||||
day DATE PRIMARY KEY,
|
||||
calls INTEGER NOT NULL DEFAULT 0,
|
||||
remaining INTEGER,
|
||||
updated_at TIMESTAMPTZ NOT NULL DEFAULT now()
|
||||
)
|
||||
"""
|
||||
)
|
||||
|
||||
_quota_lock = asyncio.Lock()
|
||||
_quota_ensured = False
|
||||
|
||||
|
||||
async def ensure_quota_table() -> None:
|
||||
global _quota_ensured
|
||||
if _quota_ensured:
|
||||
return
|
||||
async with _quota_lock:
|
||||
if _quota_ensured:
|
||||
return
|
||||
async with engine.begin() as conn:
|
||||
await conn.execute(_CREATE_QUOTA_SQL)
|
||||
_quota_ensured = True
|
||||
|
||||
|
||||
class PgQuotaStore:
|
||||
"""Postgres-backed daily call counter. Injected for tests."""
|
||||
|
||||
async def calls_today(self, day: date) -> int:
|
||||
await ensure_quota_table()
|
||||
async with async_session() as session:
|
||||
row = (await session.execute(
|
||||
select(vesselapi_quota.c.calls).where(vesselapi_quota.c.day == day)
|
||||
)).scalar()
|
||||
return int(row) if row else 0
|
||||
|
||||
async def remaining_today(self, day: date) -> int | None:
|
||||
await ensure_quota_table()
|
||||
async with async_session() as session:
|
||||
row = (await session.execute(
|
||||
select(vesselapi_quota.c.remaining).where(vesselapi_quota.c.day == day)
|
||||
)).scalar()
|
||||
return int(row) if row is not None else None
|
||||
|
||||
async def bump(self, day: date, remaining: int | None) -> int:
|
||||
await ensure_quota_table()
|
||||
now = datetime.now(timezone.utc)
|
||||
async with async_session() as session:
|
||||
existing = (await session.execute(
|
||||
select(vesselapi_quota.c.calls).where(vesselapi_quota.c.day == day)
|
||||
)).scalar()
|
||||
if existing is None:
|
||||
await session.execute(
|
||||
vesselapi_quota.insert().values(
|
||||
day=day, calls=1, remaining=remaining, updated_at=now,
|
||||
)
|
||||
)
|
||||
else:
|
||||
await session.execute(
|
||||
vesselapi_quota.update()
|
||||
.where(vesselapi_quota.c.day == day)
|
||||
.values(
|
||||
calls=vesselapi_quota.c.calls + 1,
|
||||
remaining=remaining,
|
||||
updated_at=now,
|
||||
)
|
||||
)
|
||||
await session.commit()
|
||||
return (int(existing) if existing else 0) + 1
|
||||
|
||||
|
||||
# ── Daily VesselAPI snapshots (DVR as-of + survive restarts) ──────────────
|
||||
# Cleared at the UTC day boundary so the table holds today's 5 polls only.
|
||||
|
||||
_CREATE_VESSELS_SQL = text(
|
||||
"""
|
||||
CREATE TABLE IF NOT EXISTS vessels (
|
||||
mmsi TEXT NOT NULL,
|
||||
poll_at TIMESTAMPTZ NOT NULL,
|
||||
lat DOUBLE PRECISION NOT NULL,
|
||||
lon DOUBLE PRECISION NOT NULL,
|
||||
heading DOUBLE PRECISION,
|
||||
speed DOUBLE PRECISION,
|
||||
label TEXT,
|
||||
extra JSONB,
|
||||
PRIMARY KEY (mmsi, poll_at)
|
||||
)
|
||||
"""
|
||||
)
|
||||
_CREATE_VESSELS_POLL_IDX = text(
|
||||
"CREATE INDEX IF NOT EXISTS ix_vessels_poll_at ON vessels (poll_at DESC)"
|
||||
)
|
||||
_CREATE_VESSELS_BBOX_IDX = text(
|
||||
"CREATE INDEX IF NOT EXISTS ix_vessels_bbox ON vessels (lon, lat)"
|
||||
)
|
||||
|
||||
_vessels_lock = asyncio.Lock()
|
||||
_vessels_ensured = False
|
||||
|
||||
|
||||
async def ensure_vessels_table() -> None:
|
||||
global _vessels_ensured
|
||||
if _vessels_ensured:
|
||||
return
|
||||
async with _vessels_lock:
|
||||
if _vessels_ensured:
|
||||
return
|
||||
async with engine.begin() as conn:
|
||||
await conn.execute(_CREATE_VESSELS_SQL)
|
||||
await conn.execute(_CREATE_VESSELS_POLL_IDX)
|
||||
await conn.execute(_CREATE_VESSELS_BBOX_IDX)
|
||||
_vessels_ensured = True
|
||||
|
||||
|
||||
def _marker_from_vessel_row(r) -> dict:
|
||||
extra = r.get("extra") or {}
|
||||
if isinstance(extra, str):
|
||||
try:
|
||||
extra = json.loads(extra)
|
||||
except (TypeError, ValueError):
|
||||
extra = {}
|
||||
if not isinstance(extra, dict):
|
||||
extra = {}
|
||||
extra.setdefault("src", "vesselapi")
|
||||
poll_at = r.get("poll_at")
|
||||
if poll_at is not None and hasattr(poll_at, "isoformat"):
|
||||
extra["poll_at"] = poll_at.isoformat()
|
||||
marker = to_marker(
|
||||
str(r["id"]), r["lat"], r["lon"],
|
||||
heading=r.get("heading"), speed=r.get("speed"),
|
||||
label=r.get("label") or str(r["id"]),
|
||||
extra=extra,
|
||||
)
|
||||
marker["seen_at"] = extra.get("poll_at") or datetime.now(timezone.utc).isoformat()
|
||||
return marker
|
||||
|
||||
|
||||
async def persist_vessel_snapshot(markers: list[dict], poll_at: datetime) -> None:
|
||||
"""Write one VesselAPI poll into ``vessels`` (today's snapshots)."""
|
||||
await ensure_vessels_table()
|
||||
if not markers:
|
||||
return
|
||||
async with async_session() as session:
|
||||
for m in markers:
|
||||
vid = str(m.get("id") or "")
|
||||
lat, lon = m.get("lat"), m.get("lon")
|
||||
if not vid or lat is None or lon is None:
|
||||
continue
|
||||
extra = dict(m.get("extra") or {})
|
||||
extra.setdefault("src", "vesselapi")
|
||||
await session.execute(
|
||||
text(
|
||||
"""
|
||||
INSERT INTO vessels
|
||||
(mmsi, poll_at, lat, lon, heading, speed, label, extra)
|
||||
VALUES
|
||||
(:mmsi, :poll_at, :lat, :lon, :heading, :speed, :label,
|
||||
CAST(:extra AS jsonb))
|
||||
ON CONFLICT (mmsi, poll_at) DO UPDATE SET
|
||||
lat = EXCLUDED.lat,
|
||||
lon = EXCLUDED.lon,
|
||||
heading = EXCLUDED.heading,
|
||||
speed = EXCLUDED.speed,
|
||||
label = EXCLUDED.label,
|
||||
extra = EXCLUDED.extra
|
||||
"""
|
||||
),
|
||||
{
|
||||
"mmsi": vid,
|
||||
"poll_at": poll_at,
|
||||
"lat": float(lat),
|
||||
"lon": float(lon),
|
||||
"heading": m.get("heading"),
|
||||
"speed": m.get("speed"),
|
||||
"label": m.get("label") or vid,
|
||||
"extra": json.dumps(extra),
|
||||
},
|
||||
)
|
||||
await session.commit()
|
||||
|
||||
|
||||
async def purge_old_vessels(before: datetime | None = None) -> None:
|
||||
"""Drop snapshots from before the current UTC day (or ``before``)."""
|
||||
await ensure_vessels_table()
|
||||
cutoff = before or utc_day_start(datetime.now(timezone.utc))
|
||||
async with async_session() as session:
|
||||
await session.execute(
|
||||
text("DELETE FROM vessels WHERE poll_at < :cutoff"),
|
||||
{"cutoff": cutoff},
|
||||
)
|
||||
await session.commit()
|
||||
|
||||
|
||||
async def fetch_vessels_as_of(
|
||||
ts: datetime,
|
||||
bbox: str | None = None,
|
||||
limit: int = 2000,
|
||||
) -> list[dict]:
|
||||
"""Latest VesselAPI poll at or before ``ts`` (DVR as-of, not exact minute)."""
|
||||
try:
|
||||
await ensure_vessels_table()
|
||||
async with async_session() as session:
|
||||
poll = (await session.execute(
|
||||
text("SELECT max(poll_at) FROM vessels WHERE poll_at <= :ts"),
|
||||
{"ts": ts},
|
||||
)).scalar()
|
||||
if poll is None:
|
||||
return []
|
||||
sql = """
|
||||
SELECT mmsi AS id, lat, lon, heading, speed, label, extra, poll_at
|
||||
FROM vessels
|
||||
WHERE poll_at = :poll
|
||||
"""
|
||||
params: dict = {"poll": poll, "limit": limit}
|
||||
if bbox:
|
||||
minlon, minlat, maxlon, maxlat = parse_bbox(bbox)
|
||||
sql += (
|
||||
" AND lon BETWEEN :minlon AND :maxlon"
|
||||
" AND lat BETWEEN :minlat AND :maxlat"
|
||||
)
|
||||
params.update(
|
||||
minlon=minlon, minlat=minlat, maxlon=maxlon, maxlat=maxlat,
|
||||
)
|
||||
sql += " LIMIT :limit"
|
||||
rows = (await session.execute(text(sql), params)).mappings().all()
|
||||
return [_marker_from_vessel_row(r) for r in rows]
|
||||
except Exception:
|
||||
logger.exception("VesselAPI snapshot fetch failed")
|
||||
return []
|
||||
|
||||
|
||||
async def hydrate_last_known() -> int:
|
||||
"""Seed in-memory last-known from today's latest poll (app boot)."""
|
||||
try:
|
||||
rows = await fetch_vessels_as_of(datetime.now(timezone.utc))
|
||||
except Exception:
|
||||
logger.exception("VesselAPI hydrate failed")
|
||||
return 0
|
||||
if not rows:
|
||||
return 0
|
||||
async with vessel_lock:
|
||||
for m in rows:
|
||||
vid = str(m.get("id") or "")
|
||||
if vid:
|
||||
vessel_last_known[vid] = m
|
||||
return len(rows)
|
||||
|
||||
|
||||
# ── Budget / scheduling (pure, unit-testable) ─────────────────────────────
|
||||
|
||||
def days_left_in_month(now: datetime) -> int:
|
||||
"""UTC days remaining in the current month, inclusive of today."""
|
||||
_, last = calendar.monthrange(now.year, now.month)
|
||||
return last - now.day + 1
|
||||
|
||||
|
||||
def budget_allows(
|
||||
calls_today: int,
|
||||
remaining: int | None,
|
||||
days_left: int,
|
||||
max_per_day: int,
|
||||
) -> bool:
|
||||
"""True if another poll is permitted today.
|
||||
|
||||
Local hard cap: fewer than ``max_per_day`` successful calls today.
|
||||
Monthly floor: if ``X-RateLimit-Remaining`` is known, keep at least
|
||||
``max_per_day * days_left`` in reserve for the rest of the month.
|
||||
"""
|
||||
if calls_today >= max_per_day:
|
||||
return False
|
||||
if remaining is not None and remaining <= max_per_day * days_left:
|
||||
return False
|
||||
return True
|
||||
|
||||
|
||||
def choose_box(
|
||||
boxes: list[tuple[float, float, float, float]],
|
||||
calls_today: int,
|
||||
max_per_day: int,
|
||||
) -> int:
|
||||
"""Index into ``boxes`` for the next poll.
|
||||
|
||||
Prefer refreshing the first (primary) box rather than spraying one call
|
||||
across every region — round-robin only when the remaining daily budget is
|
||||
enough to cover all boxes.
|
||||
"""
|
||||
if len(boxes) <= 1:
|
||||
return 0
|
||||
budget_left = max_per_day - calls_today
|
||||
if budget_left >= len(boxes):
|
||||
return calls_today % len(boxes)
|
||||
return 0
|
||||
|
||||
|
||||
# ── HTTP / poll ───────────────────────────────────────────────────────────
|
||||
|
||||
async def _resolve_key() -> str:
|
||||
from keystore import get_api_key
|
||||
return (
|
||||
os.getenv("VESSELAPI_API_KEY")
|
||||
or VESSELAPI_API_KEY
|
||||
or (await get_api_key("VESSELAPI_API_KEY"))
|
||||
or ""
|
||||
).strip()
|
||||
|
||||
|
||||
async def _get_client() -> httpx.AsyncClient:
|
||||
global _client
|
||||
if _client is None:
|
||||
async with _client_lock:
|
||||
if _client is None:
|
||||
_client = httpx.AsyncClient(
|
||||
timeout=httpx.Timeout(15.0, connect=5.0),
|
||||
follow_redirects=True,
|
||||
headers={"User-Agent": OSINT_USER_AGENT, "Accept": "application/json"},
|
||||
limits=httpx.Limits(max_connections=1, max_keepalive_connections=1),
|
||||
)
|
||||
return _client
|
||||
|
||||
|
||||
async def close_client() -> None:
|
||||
global _client
|
||||
if _client is not None:
|
||||
await _client.aclose()
|
||||
_client = None
|
||||
|
||||
|
||||
def _int_header(value: str | None) -> int | None:
|
||||
if value is None:
|
||||
return None
|
||||
try:
|
||||
return int(value)
|
||||
except (TypeError, ValueError):
|
||||
return None
|
||||
|
||||
|
||||
async def poll_once(store, boxes: list[tuple[float, float, float, float]], key: str) -> bool:
|
||||
"""One quota-checked poll. Returns True if a successful 2xx was made.
|
||||
|
||||
Only successful 2xx responses count against the monthly quota; 4xx/5xx/429
|
||||
are skipped without retry-storming (Retry-After respected by simply
|
||||
sleeping the interval).
|
||||
"""
|
||||
now = datetime.now(timezone.utc)
|
||||
today = now.date()
|
||||
calls = await store.calls_today(today)
|
||||
remaining = await store.remaining_today(today)
|
||||
days_left = days_left_in_month(now)
|
||||
if not budget_allows(calls, remaining, days_left, VESSELAPI_MAX_CALLS_PER_DAY):
|
||||
logger.info(
|
||||
"VesselAPI quota reached (calls_today=%d, remaining=%s, days_left=%d) — skip poll",
|
||||
calls, remaining, days_left,
|
||||
)
|
||||
return False
|
||||
|
||||
idx = choose_box(boxes, calls, VESSELAPI_MAX_CALLS_PER_DAY)
|
||||
minlat, minlon, maxlat, maxlon = boxes[idx]
|
||||
client = await _get_client()
|
||||
params = {
|
||||
"filter.latBottom": str(minlat),
|
||||
"filter.latTop": str(maxlat),
|
||||
"filter.lonLeft": str(minlon),
|
||||
"filter.lonRight": str(maxlon),
|
||||
"pagination.limit": str(PAGE_LIMIT),
|
||||
}
|
||||
headers = {"Authorization": f"Bearer {key}"}
|
||||
try:
|
||||
resp = await client.get(ENDPOINT, params=params, headers=headers)
|
||||
except httpx.HTTPError as exc:
|
||||
logger.warning("VesselAPI request failed: %s", exc)
|
||||
return False
|
||||
|
||||
if resp.status_code == 429:
|
||||
logger.warning(
|
||||
"VesselAPI rate-limited (Retry-After=%s) — skip poll",
|
||||
resp.headers.get("Retry-After"),
|
||||
)
|
||||
return False
|
||||
if resp.status_code >= 400:
|
||||
logger.warning("VesselAPI HTTP %d — not counted against quota", resp.status_code)
|
||||
return False
|
||||
|
||||
# 2xx success — counts against the monthly quota.
|
||||
remaining = _int_header(resp.headers.get("X-RateLimit-Remaining"))
|
||||
calls = await store.bump(today, remaining)
|
||||
try:
|
||||
data = resp.json()
|
||||
except ValueError:
|
||||
logger.warning("VesselAPI 2xx with non-JSON body — counted but ignored")
|
||||
return True
|
||||
markers = transform_vesselapi_payload(data)
|
||||
for m in markers:
|
||||
await upsert_vessel(m)
|
||||
try:
|
||||
await persist_vessel_snapshot(markers, now)
|
||||
await purge_old_vessels(utc_day_start(now))
|
||||
except Exception: # noqa: BLE001 — live overlay must not die on persist
|
||||
logger.exception("VesselAPI snapshot persist failed")
|
||||
logger.info(
|
||||
"VesselAPI poll OK: %d vessels (remaining=%s, calls_today=%d)",
|
||||
len(markers), remaining, calls,
|
||||
)
|
||||
return True
|
||||
|
||||
|
||||
# ── Worker loop ───────────────────────────────────────────────────────────
|
||||
|
||||
async def run_vesselapi_worker(store: PgQuotaStore | None = None) -> None:
|
||||
"""Long-lived poll loop. Idle when the key is unset; never crashes the app."""
|
||||
if store is None:
|
||||
store = PgQuotaStore()
|
||||
boxes = parse_boxes_validated(VESSELAPI_BBOX)
|
||||
if not boxes:
|
||||
logger.warning(
|
||||
"VESSELAPI_BBOX has no valid boxes (span ≤ %.1f°) — poller idle", MAX_SPAN_DEG,
|
||||
)
|
||||
while True:
|
||||
try:
|
||||
if not boxes:
|
||||
await asyncio.sleep(VESSELAPI_INTERVAL)
|
||||
continue
|
||||
key = await _resolve_key()
|
||||
if not key:
|
||||
logger.warning(
|
||||
"VESSELAPI_API_KEY not set — VesselAPI poller idle. "
|
||||
"Create a free key at https://dashboard.vesselapi.com/"
|
||||
)
|
||||
await asyncio.sleep(VESSELAPI_INTERVAL)
|
||||
continue
|
||||
await poll_once(store, boxes, key)
|
||||
except asyncio.CancelledError:
|
||||
raise
|
||||
except Exception: # noqa: BLE001 — keep the loop alive across transient failures
|
||||
logger.exception("VesselAPI poll error")
|
||||
await asyncio.sleep(VESSELAPI_INTERVAL)
|
||||
|
|
@ -1,115 +0,0 @@
|
|||
"""In-memory WebSocket pub/sub with viewport filtering.
|
||||
|
||||
Zero extra deps. Ingest workers publish AIS/ADS-B points; only clients whose
|
||||
current map bbox contains the point receive the payload. No Redis/Kafka.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import asyncio
|
||||
from typing import Any
|
||||
from uuid import UUID
|
||||
|
||||
BBox = tuple[float, float, float, float] # minlon, minlat, maxlon, maxlat
|
||||
|
||||
|
||||
def _uuid_str(value: object) -> str | None:
|
||||
try:
|
||||
return str(UUID(str(value)))
|
||||
except (ValueError, TypeError, AttributeError):
|
||||
return None
|
||||
|
||||
|
||||
def point_in_bbox(lon: float, lat: float, bbox: BBox | None) -> bool:
|
||||
"""True if (lon, lat) sits inside an axis-aligned viewport."""
|
||||
if bbox is None:
|
||||
return False
|
||||
minlon, minlat, maxlon, maxlat = bbox
|
||||
return minlon <= lon <= maxlon and minlat <= lat <= maxlat
|
||||
|
||||
|
||||
class ConnectionManager:
|
||||
"""Maps Tailscale/browser clients → viewport bbox + per-client queue."""
|
||||
|
||||
def __init__(self) -> None:
|
||||
self._queues: dict[str, asyncio.Queue] = {}
|
||||
self._viewports: dict[str, BBox] = {}
|
||||
self._watched: dict[str, set[str]] = {}
|
||||
|
||||
def register(self, client_id: str, maxsize: int = 256) -> asyncio.Queue:
|
||||
q: asyncio.Queue = asyncio.Queue(maxsize=maxsize)
|
||||
self._queues[client_id] = q
|
||||
return q
|
||||
|
||||
def unregister(self, client_id: str) -> None:
|
||||
self._queues.pop(client_id, None)
|
||||
self._viewports.pop(client_id, None)
|
||||
self._watched.pop(client_id, None)
|
||||
|
||||
def set_watched_geofences(self, client_id: str, ids: list[str]) -> None:
|
||||
"""Watch these fence UUIDs so geofence_alert delivers off-viewport.
|
||||
|
||||
Invalid UUIDs are ignored. Empty list = watch none (viewport-only).
|
||||
"""
|
||||
if client_id not in self._queues:
|
||||
return
|
||||
watched: set[str] = set()
|
||||
for raw in ids:
|
||||
uid = _uuid_str(raw)
|
||||
if uid is not None:
|
||||
watched.add(uid)
|
||||
self._watched[client_id] = watched
|
||||
|
||||
def set_viewport(self, client_id: str, bbox: BBox) -> None:
|
||||
if client_id in self._queues:
|
||||
self._viewports[client_id] = bbox
|
||||
|
||||
def viewport_of(self, client_id: str) -> BBox | None:
|
||||
return self._viewports.get(client_id)
|
||||
|
||||
def viewports(self) -> list[BBox]:
|
||||
return list(self._viewports.values())
|
||||
|
||||
def has_clients(self) -> bool:
|
||||
return bool(self._queues)
|
||||
|
||||
async def publish_point(
|
||||
self,
|
||||
kind: str,
|
||||
payload: dict[str, Any],
|
||||
*,
|
||||
lat: float,
|
||||
lon: float,
|
||||
) -> int:
|
||||
"""Enqueue `{type, payload}` for clients whose viewport contains the point.
|
||||
|
||||
kind=geofence_alert also delivers when payload.geofence_id is in the
|
||||
client's watch set (even if the point is off-viewport). Other kinds
|
||||
stay viewport-only. Drops the oldest queued message if a client's
|
||||
buffer is full. Returns the number of clients that got a copy.
|
||||
"""
|
||||
msg = {"type": kind, "payload": payload}
|
||||
sent = 0
|
||||
gid = _uuid_str(payload.get("geofence_id")) if kind == "geofence_alert" else None
|
||||
for client_id, queue in list(self._queues.items()):
|
||||
in_view = point_in_bbox(lon, lat, self._viewports.get(client_id))
|
||||
if kind == "geofence_alert":
|
||||
watching = gid is not None and gid in self._watched.get(client_id, set())
|
||||
if not in_view and not watching:
|
||||
continue
|
||||
elif not in_view:
|
||||
continue
|
||||
if queue.full():
|
||||
try:
|
||||
queue.get_nowait()
|
||||
except asyncio.QueueEmpty:
|
||||
pass
|
||||
try:
|
||||
queue.put_nowait(msg)
|
||||
except asyncio.QueueFull:
|
||||
continue
|
||||
sent += 1
|
||||
return sent
|
||||
|
||||
|
||||
manager = ConnectionManager()
|
||||
31
deploy/README.md
Normal file
31
deploy/README.md
Normal file
|
|
@ -0,0 +1,31 @@
|
|||
# systemd unit template — copy to /etc/systemd/system/osint-masscan.service
|
||||
#
|
||||
# The masscan service is a CONTINUOUS rolling sweep (a full IPv4 pass at a
|
||||
# conservative rate takes ~5 days), so it runs as a long-lived service, NOT a
|
||||
# daily timer. The [Install] WantedBy means it starts at boot and Restart=always
|
||||
# keeps it up. Install steps (run once on the Pi, as root):
|
||||
#
|
||||
# apt install -y masscan # or: apt-get install masscan
|
||||
# mkdir -p /etc/osint-dashboard /opt/siriusdevops
|
||||
# cp deploy/masscan-excludes.txt /etc/osint-dashboard/masscan-excludes.txt
|
||||
#
|
||||
# # Optional tuning (override env in this file; the DB_* values in the unit
|
||||
# # already point at the host-published Postgres on 127.0.0.1:5432):
|
||||
# cat > /etc/osint-dashboard/masscan.env <<'EOF'
|
||||
# MASSCAN_RANGE=0.0.0.0/0
|
||||
# MASSCAN_PORTS=554
|
||||
# MASSCAN_RATE=1000
|
||||
# EOF
|
||||
#
|
||||
# # Venv for the scanner (host-level, not the compose image):
|
||||
# cd /opt/siriusdevops/osint-dashboard
|
||||
# python3 -m venv .venv-masscan
|
||||
# .venv-masscan/bin/pip install -r app/requirements.txt
|
||||
#
|
||||
# install -m 644 deploy/osint-masscan.service /etc/systemd/system/
|
||||
# systemctl daemon-reload
|
||||
# systemctl enable --now osint-masscan
|
||||
#
|
||||
# Watch: journalctl -u osint-masscan -f
|
||||
# DB: writes into the same Postgres the compose stack uses (127.0.0.1:5432)
|
||||
# so findings appear on the dashboard camera map automatically.
|
||||
33
deploy/masscan-excludes.txt
Normal file
33
deploy/masscan-excludes.txt
Normal file
|
|
@ -0,0 +1,33 @@
|
|||
# masscan excludefile — never probe these ranges.
|
||||
# RFC1918 private + loopback + link-local + multicast + documentation/bogons.
|
||||
# The service refuses to start if this file is missing (fail closed).
|
||||
|
||||
# Loopback
|
||||
127.0.0.0/8
|
||||
|
||||
# RFC1918 private
|
||||
10.0.0.0/8
|
||||
172.16.0.0/12
|
||||
192.168.0.0/16
|
||||
|
||||
# Link-local
|
||||
169.254.0.0/16
|
||||
|
||||
# CGNAT (RFC 6598)
|
||||
100.64.0.0/10
|
||||
|
||||
# Multicast + reserved
|
||||
224.0.0.0/4
|
||||
240.0.0.0/4
|
||||
|
||||
# Documentation / benchmark / example ranges (never real hosts)
|
||||
0.0.0.0/8
|
||||
192.0.2.0/24
|
||||
198.51.100.0/24
|
||||
203.0.113.0/24
|
||||
192.0.0.0/24
|
||||
198.18.0.0/15
|
||||
255.255.255.255/32
|
||||
|
||||
# Carrier NAT / TEST-NET leftovers
|
||||
233.252.0.0/24
|
||||
29
deploy/osint-masscan.service
Normal file
29
deploy/osint-masscan.service
Normal file
|
|
@ -0,0 +1,29 @@
|
|||
[Unit]
|
||||
Description=OSINT dashboard — masscan rolling sweep (open RTSP port 554)
|
||||
Documentation=https://forgejo.siriusdevops.com/sirius/osint-dashboard
|
||||
After=network-online.target
|
||||
Wants=network-online.target
|
||||
|
||||
[Service]
|
||||
Type=simple
|
||||
# masscan needs raw sockets (CAP_NET_RAW) — run as root on the Pi host.
|
||||
User=root
|
||||
WorkingDirectory=/opt/siriusdevops/osint-dashboard
|
||||
EnvironmentFile=-/etc/osint-dashboard/masscan.env
|
||||
# Point at the compose-published Postgres on the HOST (127.0.0.1:5432), not the
|
||||
# docker service name 'postgres' which doesn't resolve outside the compose net.
|
||||
Environment=DB_HOST=127.0.0.1
|
||||
Environment=DB_PORT=5432
|
||||
Environment=DB_USER=osint
|
||||
Environment=DB_PASSWORD=osint
|
||||
Environment=DB_NAME=osint_data
|
||||
Environment=MASSCAN_EXCLUDEFILE=/etc/osint-dashboard/masscan-excludes.txt
|
||||
ExecStart=/opt/siriusdevops/osint-dashboard/.venv-masscan/bin/python app/run_masscan_service.py
|
||||
Restart=always
|
||||
RestartSec=10
|
||||
# Log the sweep to journald (read with: journalctl -u osint-masscan -f)
|
||||
StandardOutput=journal
|
||||
StandardError=journal
|
||||
|
||||
[Install]
|
||||
WantedBy=multi-user.target
|
||||
|
|
@ -1,21 +0,0 @@
|
|||
# osint.rpi.local — Sentinel-1 SAR tile proxy (/titiler/)
|
||||
#
|
||||
# GitOps: this file is the source of truth. On the Pi:
|
||||
# sudo cp deploy/osint-titiler.nginx.conf /etc/nginx/snippets/osint-titiler.conf
|
||||
# then `include snippets/osint-titiler.conf;` inside the osint.rpi.local server
|
||||
# block (before `location /`), `nginx -t && systemctl reload nginx`.
|
||||
#
|
||||
# The browser hits /titiler/cog/tiles/... (same-origin). We strip the /titiler
|
||||
# prefix so self-hosted TiTiler (127.0.0.1:8001) sees /cog/tiles/... and proxy
|
||||
# its response straight back. Tiles are heavy PNGs — disable buffering so a
|
||||
# slow client doesn't hold a worker open.
|
||||
|
||||
location /titiler/ {
|
||||
proxy_pass http://127.0.0.1:8001/;
|
||||
proxy_set_header Host $host;
|
||||
proxy_set_header X-Real-IP $remote_addr;
|
||||
proxy_set_header X-Forwarded-For $proxy_add_x_forwarded_for;
|
||||
proxy_set_header X-Forwarded-Proto $scheme;
|
||||
proxy_buffering off;
|
||||
proxy_read_timeout 300s;
|
||||
}
|
||||
|
|
@ -1,23 +0,0 @@
|
|||
# osint.rpi.local — WebSocket upgrade for /ws/live
|
||||
#
|
||||
# GitOps: this file is the source of truth. On the Pi:
|
||||
# sudo cp deploy/osint-ws.nginx.conf /etc/nginx/snippets/osint-ws.conf
|
||||
# then `include snippets/osint-ws.conf;` inside the osint.rpi.local server
|
||||
# block (before `location /`), `nginx -t && systemctl reload nginx`.
|
||||
#
|
||||
# Without these headers nginx proxies GET /ws/live as HTTP/1.0 → FastAPI 404
|
||||
# and the HUD reconnects every few seconds.
|
||||
|
||||
location /ws/ {
|
||||
proxy_pass http://127.0.0.1:8000;
|
||||
proxy_http_version 1.1;
|
||||
proxy_set_header Upgrade $http_upgrade;
|
||||
proxy_set_header Connection "upgrade";
|
||||
proxy_set_header Host $host;
|
||||
proxy_set_header X-Real-IP $remote_addr;
|
||||
proxy_set_header X-Forwarded-For $proxy_add_x_forwarded_for;
|
||||
proxy_set_header X-Forwarded-Proto $scheme;
|
||||
proxy_read_timeout 3600s;
|
||||
proxy_send_timeout 3600s;
|
||||
proxy_buffering off;
|
||||
}
|
||||
|
|
@ -20,10 +20,8 @@ services:
|
|||
# networks (and ISP abuse-mitigation blackholes) block, so rebuilding it
|
||||
# on every CI deploy made the pipeline flaky. Rebuild manually when the
|
||||
# base image or extensions need bumping:
|
||||
# docker build -f Dockerfile.pg -t localhost/osint-dashboard-pg:latest .
|
||||
# FORCE_RECREATE_DB=1 scripts/compose-reup.sh db
|
||||
# docker compose build db && docker compose up -d db
|
||||
image: localhost/osint-dashboard-pg:latest
|
||||
pull_policy: never
|
||||
container_name: osint-db
|
||||
restart: unless-stopped
|
||||
environment:
|
||||
|
|
@ -32,16 +30,7 @@ services:
|
|||
POSTGRES_DB: ${DB_NAME:-osint_data}
|
||||
# Ensure TimescaleDB is preloaded (conf.d drop-in may be ignored by the
|
||||
# official image's runtime-generated postgresql.conf, so pass it explicitly).
|
||||
# shared_buffers capped at 2GB for Pi 5 8GB / 4 cores.
|
||||
command:
|
||||
[
|
||||
"-c", "shared_preload_libraries=timescaledb",
|
||||
"-c", "shared_buffers=2GB",
|
||||
]
|
||||
deploy:
|
||||
resources:
|
||||
limits:
|
||||
memory: 3G
|
||||
command: ["-c", "shared_preload_libraries=timescaledb"]
|
||||
ports:
|
||||
- "127.0.0.1:5432:5432"
|
||||
volumes:
|
||||
|
|
@ -54,7 +43,6 @@ services:
|
|||
|
||||
nats:
|
||||
image: nats:2.10
|
||||
pull_policy: missing
|
||||
platform: linux/arm64
|
||||
container_name: osint-nats
|
||||
restart: unless-stopped
|
||||
|
|
@ -70,7 +58,6 @@ services:
|
|||
dockerfile: Dockerfile
|
||||
platforms: ["linux/arm64"]
|
||||
image: localhost/osint-dashboard:latest
|
||||
pull_policy: never
|
||||
container_name: osint-ingester
|
||||
restart: unless-stopped
|
||||
profiles: ["ingest"]
|
||||
|
|
@ -97,15 +84,10 @@ services:
|
|||
FIRMS_DATASETS: ${FIRMS_DATASETS:-VIIRS_NOAA20_NRT,VIIRS_NOAA21_NRT}
|
||||
FIRMS_BBOX: ${FIRMS_BBOX:--180,-60,180,75}
|
||||
FIRMS_INTERVAL: ${FIRMS_INTERVAL:-900}
|
||||
OSINT_USER_AGENT: ${OSINT_USER_AGENT:-osint-dashboard/1.0 (self-hosted; lancewalters94@gmail.com)}
|
||||
OSINT_USER_AGENT: ${OSINT_USER_AGENT:-osint-dashboard/1.0 (self-hosted)}
|
||||
AISSTREAM_API_KEY: ${AISSTREAM_API_KEY:-}
|
||||
AISSTREAM_BBOX: ${AISSTREAM_BBOX:-24,-125,50,-66}
|
||||
AISSTREAM_IN_INGEST: ${AISSTREAM_IN_INGEST:-0}
|
||||
VESSELAPI_API_KEY: ${VESSELAPI_API_KEY:-}
|
||||
VESSELAPI_BBOX: ${VESSELAPI_BBOX:-25.5,55.4,27.3,57.2}
|
||||
VESSELAPI_INTERVAL: ${VESSELAPI_INTERVAL:-17280}
|
||||
VESSELAPI_MAX_CALLS_PER_DAY: ${VESSELAPI_MAX_CALLS_PER_DAY:-5}
|
||||
VESSELAPI_IN_INGEST: ${VESSELAPI_IN_INGEST:-0}
|
||||
command: ["python", "app/run_ingester.py"]
|
||||
entrypoint: ["python", "app/run_ingester.py"]
|
||||
|
||||
|
|
@ -115,7 +97,6 @@ services:
|
|||
dockerfile: Dockerfile
|
||||
platforms: ["linux/arm64"]
|
||||
image: localhost/osint-dashboard:latest
|
||||
pull_policy: never
|
||||
container_name: osint-dashboard
|
||||
restart: unless-stopped
|
||||
depends_on:
|
||||
|
|
@ -137,61 +118,24 @@ services:
|
|||
FIRMS_DATASET: ${FIRMS_DATASET:-VIIRS_NOAA20_NRT}
|
||||
FIRMS_DATASETS: ${FIRMS_DATASETS:-VIIRS_NOAA20_NRT,VIIRS_NOAA21_NRT}
|
||||
FIRMS_BBOX: ${FIRMS_BBOX:--180,-60,180,75}
|
||||
OSINT_USER_AGENT: ${OSINT_USER_AGENT:-osint-dashboard/1.0 (self-hosted; lancewalters94@gmail.com)}
|
||||
NOMINATIM_URL: ${NOMINATIM_URL:-https://nominatim.openstreetmap.org}
|
||||
NOMINATIM_MIN_INTERVAL: ${NOMINATIM_MIN_INTERVAL:-1.0}
|
||||
OSINT_USER_AGENT: ${OSINT_USER_AGENT:-osint-dashboard/1.0 (self-hosted)}
|
||||
AISSTREAM_API_KEY: ${AISSTREAM_API_KEY:-}
|
||||
AISSTREAM_BBOX: ${AISSTREAM_BBOX:-24,-125,50,-66}
|
||||
AISSTREAM_IN_APP: ${AISSTREAM_IN_APP:-1}
|
||||
VESSELAPI_API_KEY: ${VESSELAPI_API_KEY:-}
|
||||
VESSELAPI_BBOX: ${VESSELAPI_BBOX:-25.5,55.4,27.3,57.2}
|
||||
VESSELAPI_INTERVAL: ${VESSELAPI_INTERVAL:-17280}
|
||||
VESSELAPI_MAX_CALLS_PER_DAY: ${VESSELAPI_MAX_CALLS_PER_DAY:-5}
|
||||
VESSELAPI_IN_APP: ${VESSELAPI_IN_APP:-1}
|
||||
# ── Self-hosted TiTiler (Sentinel-1 SAR tiles) ──
|
||||
TITILER_PUBLIC_BASE: ${TITILER_PUBLIC_BASE:-/titiler}
|
||||
TITILER_INTERNAL_URL: ${TITILER_INTERNAL_URL:-http://titiler:8000}
|
||||
ports:
|
||||
- "127.0.0.1:8000:8000"
|
||||
deploy:
|
||||
resources:
|
||||
limits:
|
||||
memory: 2G
|
||||
healthcheck:
|
||||
test: ["CMD-SHELL", "python -c \"import urllib.request,sys; sys.exit(0 if urllib.request.urlopen('http://127.0.0.1:8000/api/health').status==200 else 1)\""]
|
||||
interval: 30s
|
||||
timeout: 5s
|
||||
retries: 5
|
||||
|
||||
# ── Self-hosted TiTiler (Sentinel-1 SAR COG → XYZ tiles) ────────────────
|
||||
# Warps the signed Planetary Computer COG into WebMercator XYZ tiles so the
|
||||
# browser never loads a multi-GB GeoTIFF. The FastAPI app signs the COG URL
|
||||
# and returns a /titiler/... template; nginx routes /titiler/ here.
|
||||
# Listens on 8000 INSIDE the container (the app already owns host 8000);
|
||||
# published on host loopback 127.0.0.1:8001 only.
|
||||
titiler:
|
||||
image: ghcr.io/developmentseed/titiler:latest@sha256:1809958d063543e3ec858259536002b2de78e9f8f09a22a8d9591bdc2b550b14
|
||||
pull_policy: missing
|
||||
container_name: osint-titiler
|
||||
platform: linux/arm64
|
||||
restart: unless-stopped
|
||||
environment:
|
||||
- PORT=8000
|
||||
- WORKERS_PER_CORE=1
|
||||
ports:
|
||||
- "127.0.0.1:8001:8000"
|
||||
deploy:
|
||||
resources:
|
||||
limits:
|
||||
memory: 1G
|
||||
|
||||
camera-service:
|
||||
build:
|
||||
context: .
|
||||
dockerfile: Dockerfile
|
||||
platforms: ["linux/arm64"]
|
||||
image: localhost/osint-dashboard:latest
|
||||
pull_policy: never
|
||||
container_name: osint-camera-scraper
|
||||
restart: unless-stopped
|
||||
profiles: ["ingest"]
|
||||
|
|
@ -219,7 +163,7 @@ services:
|
|||
volumes:
|
||||
- camera-snapshots:/data/snapshots
|
||||
|
||||
# ── News pipeline: continuous scraper + 15-min summarizer ───────────────
|
||||
# ── News pipeline: hourly scraper (:00) + summarizer (:05) ───────────────
|
||||
# Both services point at the EXISTING osint-db (tables articles +
|
||||
# article_summaries, created by idempotent alembic migration 003_news).
|
||||
# Scheduling replaces the upstream k8s CronJobs with in-compose wall-clock
|
||||
|
|
@ -230,7 +174,6 @@ services:
|
|||
dockerfile: Dockerfile
|
||||
platforms: ["linux/arm64"]
|
||||
image: localhost/osint-news-scraper:latest
|
||||
pull_policy: never
|
||||
container_name: osint-news-scraper
|
||||
restart: unless-stopped
|
||||
profiles: ["ingest"]
|
||||
|
|
@ -244,7 +187,7 @@ services:
|
|||
DB_PORT: ${DB_PORT:-5432}
|
||||
DB_NAME: ${DB_NAME:-osint_data}
|
||||
LOG_LEVEL: ${NEWS_LOG_LEVEL:-INFO}
|
||||
NEWS_SCRAPE_INTERVAL_S: ${NEWS_SCRAPE_INTERVAL_S:-10}
|
||||
NEWS_SCRAPE_MINUTE: ${NEWS_SCRAPE_MINUTE:-0}
|
||||
NEWS_SCRAPE_RUN_ON_START: ${NEWS_SCRAPE_RUN_ON_START:-1}
|
||||
# Override the image ENTRYPOINT ["scrapy"] with the scheduler loop.
|
||||
entrypoint: []
|
||||
|
|
@ -256,7 +199,6 @@ services:
|
|||
dockerfile: Dockerfile
|
||||
platforms: ["linux/arm64"]
|
||||
image: localhost/osint-news-summarizer:latest
|
||||
pull_policy: never
|
||||
container_name: osint-news-summarizer
|
||||
restart: unless-stopped
|
||||
profiles: ["ingest"]
|
||||
|
|
@ -269,19 +211,14 @@ services:
|
|||
DB_HOST: db
|
||||
DB_PORT: ${DB_PORT:-5432}
|
||||
DB_NAME: ${DB_NAME:-osint_data}
|
||||
NOUS_API_KEY: ${NOUS_API_KEY:-}
|
||||
NOUS_BASE_URL: ${NOUS_BASE_URL:-https://inference-api.nousresearch.com/v1}
|
||||
SUMMARY_MODEL: ${SUMMARY_MODEL:-}
|
||||
OSINT_USER_AGENT: ${OSINT_USER_AGENT:-osint-dashboard-news-summarizer}
|
||||
# Required to do real work; unset → the loop logs and idles.
|
||||
GEMINI_API_KEY: ${GEMINI_API_KEY:-}
|
||||
SUMMARY_MODEL: ${SUMMARY_MODEL:-gemini-2.0-flash}
|
||||
BATCH_SIZE: ${NEWS_BATCH_SIZE:-50}
|
||||
SUMMARY_WINDOW_MINUTES: ${SUMMARY_WINDOW_MINUTES:-15}
|
||||
SUMMARY_WINDOW_HOURS: ${SUMMARY_WINDOW_HOURS:-1}
|
||||
INCLUDE_FUTURES: ${INCLUDE_FUTURES:-0}
|
||||
NEWS_SUMMARIZE_INTERVAL_S: ${NEWS_SUMMARIZE_INTERVAL_S:-900}
|
||||
NEWS_SUMMARIZE_MINUTE: ${NEWS_SUMMARIZE_MINUTE:-5}
|
||||
NEWS_SUMMARIZE_RUN_ON_START: ${NEWS_SUMMARIZE_RUN_ON_START:-1}
|
||||
NEWS_SUMMARIZE_FORCE: ${NEWS_SUMMARIZE_FORCE:-0}
|
||||
TZ: ${TZ:-America/New_York}
|
||||
NEWS_RECAP_HOUR: ${NEWS_RECAP_HOUR:-23}
|
||||
NEWS_RECAP_MINUTE: ${NEWS_RECAP_MINUTE:-0}
|
||||
command: ["python", "run_news_summarizer.py"]
|
||||
|
||||
volumes:
|
||||
|
|
|
|||
|
|
@ -2,7 +2,7 @@
|
|||
|
||||
Builder brief for backend + frontend. Researched 2026-08-27. Every endpoint below was either live-probed from this machine or taken from the provider’s current docs. Prefer **free, no-key, CORS-open** sources first. Keys are called out explicitly.
|
||||
|
||||
This is **not** a camera-discovery change. Existing camera rules still apply: never emit `rtsp://` hrefs; camera pins go through `/api/cameras/{id}/snapshot`; HTTP directory cams use `/stream` MJPEG.
|
||||
This is **not** a camera-discovery / masscan change. Existing camera rules still apply: never emit `rtsp://` hrefs; masscan pins go through `/api/cameras/{id}/snapshot`; HTTP directory cams use `/stream` MJPEG.
|
||||
|
||||
---
|
||||
|
||||
|
|
@ -13,7 +13,7 @@ This is **not** a camera-discovery change. Existing camera rules still apply: ne
|
|||
| NASA FIRMS VIIRS hotspots | Ingested (`app/fire_sources.py` → NATS `events.fire` → `fires` hypertable → `GET /api/fires`) | Needs free `FIRMS_MAP_KEY`. See `docs/firms.md`. |
|
||||
| NASA GIBS basemaps | Frontend tiles via `app/gibs_map.py` | No key. CORS `*`. |
|
||||
| GIBS VIIRS thermal tiles | Documented, not wired as overlay | Same GIBS stack; no key. |
|
||||
| Cameras | Scraper → `cameras` table | Defaults already include ALERTWest JPEGs + Live-Environment-Streams HLS/YouTube GeoJSON. |
|
||||
| Cameras | Scraper + masscan → `cameras` table | Defaults already include ALERTWest JPEGs + Live-Environment-Streams HLS/YouTube GeoJSON. |
|
||||
| News / RSS / GDELT / USGS quakes | Ingest | Out of scope for this brief. |
|
||||
|
||||
**Action for existing fire ingest:** NASA will stop Suomi NPP product delivery on **2026-11-01**. Switch `FIRMS_DATASET` from `VIIRS_SNPP_NRT` to `VIIRS_NOAA20_NRT` and/or `VIIRS_NOAA21_NRT` before then.[20]
|
||||
|
|
@ -261,7 +261,7 @@ Use later if you want commuter rail / subway vehicle positions (LA Metro, MTA, e
|
|||
|
||||
## 6. Open video / camera feeds (official public only)
|
||||
|
||||
Do **not** add Insecam-style random IP cams as a new source. The scraper already has a public list; this section is **agency-published** JPEG/HLS.
|
||||
Do **not** add Insecam-style random IP cams as a new source. The scraper already has a public list + masscan; this section is **agency-published** JPEG/HLS.
|
||||
|
||||
### 6.1 Already wired
|
||||
|
||||
|
|
@ -304,7 +304,7 @@ Do not call the YouTube Data API unless you want search. Embedding existing stre
|
|||
|
||||
### 6.5 Skip
|
||||
|
||||
- Insecam / random “public IP cam” aggregators — ToS / privacy.
|
||||
- Insecam / random “public IP cam” aggregators — ToS / privacy / already covered by masscan ethics.
|
||||
- TrafficLand, EarthCam commercial APIs.
|
||||
- SkylineWebcams — scraping, not an API.
|
||||
|
||||
|
|
@ -523,7 +523,7 @@ Attribution bar (required): OpenSky / ADSB.lol ODbL / Amtraker / RainViewer / IE
|
|||
|
||||
## 12. Legal / ethics (non-negotiable)
|
||||
|
||||
- RTSP policy unchanged (never emit `rtsp://` hrefs).
|
||||
- Masscan / RTSP policy unchanged.
|
||||
- AISStream: server-side only; do not put the key in JS.[5]
|
||||
- OpenSky: non-commercial unless licensed; cite if you publish.[2]
|
||||
- ADSB.lol: ODbL share-alike on derived databases.[4]
|
||||
|
|
|
|||
261
docs/news.md
261
docs/news.md
|
|
@ -1,95 +1,67 @@
|
|||
# News pipeline — scraper + Nous Portal summarizer
|
||||
# News pipeline — scraper + summarizer
|
||||
|
||||
The OSINT dashboard ingests a large curated feed list (`news/scraper/urls.txt`)
|
||||
continuously and produces an English LLM brief plus flagged ticker/map rows
|
||||
every 15 minutes. Both services were vendored from the upstream
|
||||
`~/Projects/newsPipeline` project and re-integrated here against the EXISTING
|
||||
osint-db — **no second Postgres**. The LLM is **Nous Portal**
|
||||
(`inference-api.nousresearch.com`) — not Gemini.
|
||||
The OSINT dashboard ingests ~257 global news RSS sources hourly and produces
|
||||
LLM master summaries. Both services were vendored from the upstream
|
||||
`~/Projects/newsPipeline` project and re-integrated here to replace the old
|
||||
k8s CronJob choreography with in-compose scheduling against the EXISTING
|
||||
osint-db — **no second Postgres**.
|
||||
|
||||
## Architecture
|
||||
|
||||
```
|
||||
urls.txt (RSS + homepages)
|
||||
257 RSS feeds (news/scraper/urls.txt)
|
||||
│
|
||||
▼
|
||||
news-scraper (Scrapy, continuous) ──► articles table (osint-db)
|
||||
news-scraper (Scrapy, hourly :00) ──► articles table (osint-db)
|
||||
│ │
|
||||
│ ▼
|
||||
news-summarizer (Nous Portal, every 15m + 23:00 recap) ──► article_summaries + news_items
|
||||
news-summarizer (Gemini map-reduce, hourly :05) ──► article_summaries table
|
||||
│
|
||||
▼
|
||||
GET /api/news · /api/news/summaries · /api/news/ticker · /api/news/map
|
||||
GET /api/news/models · GET/PUT /api/settings
|
||||
GET /api/news · GET /api/news/summaries
|
||||
```
|
||||
|
||||
| Component | Image | Container | Scheduling |
|
||||
|---|---|---|---|
|
||||
| Scraper | `localhost/osint-news-scraper` | `osint-news-scraper` | loop, `NEWS_SCRAPE_INTERVAL_S` (default 10s after each crawl) |
|
||||
| Summarizer | `localhost/osint-news-summarizer` | `osint-news-summarizer` | loop, `NEWS_SUMMARIZE_INTERVAL_S` (default 900s) |
|
||||
| Scraper | `localhost/osint-news-scraper` | `osint-news-scraper` | wall-clock loop, minute `NEWS_SCRAPE_MINUTE` (default :00) |
|
||||
| Summarizer | `localhost/osint-news-summarizer` | `osint-news-summarizer` | wall-clock loop, minute `NEWS_SUMMARIZE_MINUTE` (default :05) |
|
||||
|
||||
Both services live under the `ingest` compose profile (same as the ingester
|
||||
and camera-scraper): `docker compose --profile ingest up -d`.
|
||||
|
||||
The summarizer is a batch sidecar, **not** a live overlay. Do **not** reuse
|
||||
`GET /api/alerts` (dashboard entity/keyword alerts). Do **not** stuff news
|
||||
into `overlay_catalog()` — `/api/map/layers` `overlays` stays live upstream
|
||||
feeds (`GET /api/news` exact key set is unchanged on purpose).
|
||||
|
||||
## Data flow
|
||||
|
||||
1. **Scraper** — `news/scraper/run_news_scraper.py` runs
|
||||
`scrapy crawl articles` back-to-back (default 10s pause). The spider reads
|
||||
URLs from `urls.txt` (homepages autodiscover RSS; feed URLs are parsed
|
||||
directly), follows each `<item>` link, extracts the main article body, and
|
||||
the `PostgresPipeline` writes to `articles` with URL-based dedup
|
||||
`scrapy crawl articles` (spider `news/scraper/newsScraper/spiders/news_spider.py`)
|
||||
at the top of each hour. The spider reads the RSS feed URLs from `urls.txt`,
|
||||
follows each `<item>` link, extracts the main article body, and the
|
||||
`PostgresPipeline` writes to `articles` with URL-based dedup
|
||||
(`ON CONFLICT (url) DO NOTHING`).
|
||||
2. **Summarizer** — `news/summerizer/run_news_summarizer.py` runs
|
||||
`summarizer.py` every `NEWS_SUMMARIZE_INTERVAL_S` (default 900) over the
|
||||
last `SUMMARY_WINDOW_MINUTES` (default 15), and again at 23:00
|
||||
`America/New_York` (`TZ`) over the last 24 hours as a daily recap
|
||||
(`kind=daily_recap`). Both map-reduce through Nous Portal (`SUMMARY_MODEL`
|
||||
/ Settings, default `Hermes-4.3-36B`), write the English brief to
|
||||
`article_summaries` (column `model` is the LLM id; `kind` is
|
||||
`interval` or `daily_recap`), and flagged ticker/map rows to `news_items`.
|
||||
`summarizer.py` at :05 past each hour. It reads articles from the last
|
||||
`SUMMARY_WINDOW_HOURS`, map-reduces them through Gemini
|
||||
(`SUMMARY_MODEL`, default `gemini-2.0-flash`), and inserts one master
|
||||
summary into `article_summaries`.
|
||||
|
||||
Loops are serial (two crawls/summaries never overlap). Interval idempotency:
|
||||
if `article_summaries` already has a row in the last interval, the summarizer
|
||||
**skips** (prevents double-pins on `RUN_ON_START` recreate). Set
|
||||
`NEWS_SUMMARIZE_FORCE=1` to ignore that skip.
|
||||
Scheduling is done with small in-compose wall-clock loops (not host cron): each
|
||||
loop runs once on boot (`*_RUN_ON_START=1`, seeds data fast) then sleeps until
|
||||
the next scheduled minute. The loop is serial, so a run that overruns its slot
|
||||
simply shifts to the next boundary — two crawls/summaries never overlap.
|
||||
|
||||
The `articles` and `article_summaries` tables are created by the idempotent
|
||||
alembic migration `003_news` (also created by the scraper's own
|
||||
`CREATE TABLE IF NOT EXISTS`). `news_items` is alembic `005_news_items`.
|
||||
Container startup order doesn't matter.
|
||||
|
||||
## Keys and Settings
|
||||
|
||||
- **`NOUS_API_KEY`** — paste in the dashboard **Keys** UI (`api_keys` /
|
||||
`keystore.KEY_REGISTRY`). Env / `.env` is an **override** (env wins, same
|
||||
as FIRMS). Never returned by any API; never emitted into `index.html`;
|
||||
never proxied from the browser.
|
||||
- **Idle without a key** — if env is unset **and** the keystore row is empty,
|
||||
the summarizer logs and idles (never crashes). News intel APIs return `[]`.
|
||||
- **Model** — non-secret. Settings UI model selector `PUT /api/settings`
|
||||
`{ "summary_model": "…" }` stores `SUMMARY_MODEL` in `app_settings` (1–128
|
||||
chars). `GET /api/settings` echoes `{summary_model, nous_base_url}`.
|
||||
`nous_base_url` is read-only. Default `Hermes-4.3-36B`. Live catalog is
|
||||
best-effort `GET /api/news/models`.
|
||||
`CREATE TABLE IF NOT EXISTS`, so container startup order doesn't matter).
|
||||
|
||||
## Endpoints
|
||||
|
||||
### GET /api/news — recent articles
|
||||
|
||||
Key set **unchanged** (no `lat`/`lon` on articles; geo lives on `/api/news/map`).
|
||||
|
||||
| Query param | Meaning | Default |
|
||||
|---|---|---|
|
||||
| `domain` | filter by source domain (e.g. `www.reuters.com`) | none |
|
||||
| `since` | only articles captured at/after this UTC instant (ISO-8601) | none |
|
||||
| `limit` | max rows | `50` (max `500`) |
|
||||
| `offset` | pagination offset | `0` |
|
||||
| `include_content` | include full article body | `false` |
|
||||
|
||||
```json
|
||||
[
|
||||
|
|
@ -97,14 +69,14 @@ Key set **unchanged** (no `lat`/`lon` on articles; geo lives on `/api/news/map`)
|
|||
"id": 1,
|
||||
"title": "…",
|
||||
"url": "https://…",
|
||||
"content": null,
|
||||
"content": "full extracted article text…",
|
||||
"domain": "www.reuters.com",
|
||||
"timestamp": "2026-08-24T18:10:00Z"
|
||||
}
|
||||
]
|
||||
```
|
||||
|
||||
### GET /api/news/summaries — master LLM briefs
|
||||
### GET /api/news/summaries — master LLM summaries
|
||||
|
||||
| Query param | Meaning | Default |
|
||||
|---|---|---|
|
||||
|
|
@ -116,185 +88,52 @@ Key set **unchanged** (no `lat`/`lon` on articles; geo lives on `/api/news/map`)
|
|||
[
|
||||
{
|
||||
"id": 1,
|
||||
"summary_text": "English markdown brief…",
|
||||
"batch_timestamp": "2026-08-24T18:10:00Z",
|
||||
"model": "Hermes-4.3-36B",
|
||||
"kind": "daily_recap"
|
||||
"summary_text": "master LLM summary (markdown)…",
|
||||
"batch_timestamp": "2026-08-24T18:10:00Z"
|
||||
}
|
||||
]
|
||||
```
|
||||
|
||||
`model` and `kind` are additive (`interval` | `daily_recap` | `null` for old rows).
|
||||
`?kind=daily_recap` pins the nightly 24h recap. Empty DB → `[]` (no crash).
|
||||
Malformed `kind` → `422`.
|
||||
|
||||
### GET /api/news/ticker — HUD headlines
|
||||
|
||||
Critical/high `news_items` with `kind=ticker` first. If none are flagged,
|
||||
medium/low ticker rows fill the tape so the dock is not blank. Do **not**
|
||||
reuse `GET /api/alerts`. Bottom HUD `#nt-track` scrolls these rows, not a
|
||||
dump of the whole brief.
|
||||
|
||||
| Query param | Meaning | Default |
|
||||
|---|---|---|
|
||||
| `since` | only items created at/after this UTC instant | none |
|
||||
| `limit` | max rows | `20` (max `50`) |
|
||||
|
||||
```json
|
||||
[
|
||||
{
|
||||
"id": 1,
|
||||
"headline": "…",
|
||||
"importance": "critical",
|
||||
"location_name": "Kyiv",
|
||||
"url": "https://…",
|
||||
"created_at": "2026-08-24T18:10:00Z"
|
||||
}
|
||||
]
|
||||
```
|
||||
|
||||
### GET /api/news/map — geolocated critical/high pins
|
||||
|
||||
Only rows with valid `lat`/`lon`. Optional bbox. **No zoom skip** — world
|
||||
view is the point. Layer-panel toggle uses this dedicated path (same as
|
||||
event blips), not `overlay_catalog`.
|
||||
|
||||
| Query param | Meaning | Default |
|
||||
|---|---|---|
|
||||
| `bbox` | `minlon,minlat,maxlon,maxlat` | all flagged pins |
|
||||
| `since` | only items created at/after this UTC instant | last 24 hours |
|
||||
| `limit` | max rows | `200` (max `500`) |
|
||||
|
||||
Malformed bbox → `422`.
|
||||
|
||||
```json
|
||||
[
|
||||
{
|
||||
"id": 1,
|
||||
"headline": "…",
|
||||
"importance": "high",
|
||||
"location_name": "Kyiv",
|
||||
"lat": 50.45,
|
||||
"lon": 30.52,
|
||||
"location_confidence": "city",
|
||||
"category": "military/conflict",
|
||||
"url": "https://…",
|
||||
"created_at": "2026-08-24T18:10:00Z"
|
||||
}
|
||||
]
|
||||
```
|
||||
|
||||
Pins are LLM-estimated and clamped (`lat∈[-90,90]`, `lon∈[-180,180]`). No
|
||||
Nominatim. No writes into `events`.
|
||||
|
||||
### GET /api/news/models — Settings dropdown catalog
|
||||
|
||||
Never 502s. `{ "source": "live"|"fallback", "models": [{"id": "…"}] }`.
|
||||
|
||||
### GET /api/settings · PUT /api/settings
|
||||
|
||||
```json
|
||||
{ "summary_model": "Hermes-4.3-36B", "nous_base_url": "https://inference-api.nousresearch.com/v1" }
|
||||
```
|
||||
|
||||
PUT body is `{ "summary_model": "<1–128 char id>" }`. `nous_base_url` is
|
||||
ignored even if sent.
|
||||
|
||||
## Reduce JSON contract
|
||||
|
||||
Reduce phase (`response_format: json_object`, English only) must be a single
|
||||
object. Parser (`intel.parse_reduce_json`) strips `<think>…</think>` and
|
||||
markdown json fences, then brace-slices:
|
||||
|
||||
```json
|
||||
{
|
||||
"summary_en": "English markdown brief or the no-qualifying-events sentence",
|
||||
"ticker": [
|
||||
{"headline": "", "importance": "critical", "url": "", "location_name": ""}
|
||||
],
|
||||
"map_items": [
|
||||
{
|
||||
"headline": "",
|
||||
"importance": "critical",
|
||||
"location_name": "",
|
||||
"lat": 0,
|
||||
"lon": 0,
|
||||
"location_confidence": "city",
|
||||
"category": "military/conflict",
|
||||
"url": ""
|
||||
}
|
||||
]
|
||||
}
|
||||
```
|
||||
|
||||
Persist ticker for critical/high first; if none, persist medium/low so the
|
||||
tape is not empty. Map rows stay critical/high with valid coords; Unknown /
|
||||
invented places are dropped. Caps: 12 ticker (≤140 chars, no markdown), 20
|
||||
map. `summary_en` lands in `article_summaries.summary_text`.
|
||||
|
||||
## Configuration (all via env / `.env`)
|
||||
|
||||
| Var | Default | Notes |
|
||||
|---|---|---|
|
||||
| `NOUS_API_KEY` | *(blank)* | **Required for summaries.** Prefer Keys UI; env overrides. Unset in **both** env and `api_keys` = summarizer logs and idles (never crashes); APIs return `[]`. |
|
||||
| `NOUS_BASE_URL` | `https://inference-api.nousresearch.com/v1` | Read-only in Settings. |
|
||||
| `SUMMARY_MODEL` | `Hermes-4.3-36B` | Compose default. Operator-facing choice is Settings → `app_settings.SUMMARY_MODEL`. |
|
||||
| `NEWS_BATCH_SIZE` | `50` | Articles per map-phase batch (compose maps to container `BATCH_SIZE`). |
|
||||
| `SUMMARY_WINDOW_MINUTES` | `15` | How far back the summarizer looks for new articles. |
|
||||
| `NEWS_SCRAPE_INTERVAL_S` | `10` | Pause after each crawl before the next (scraper is otherwise continuous). |
|
||||
| `NEWS_SUMMARIZE_INTERVAL_S` | `900` | Seconds between analyst runs (default 15 min). |
|
||||
| `TZ` | `America/New_York` | Timezone for the 23:00 daily recap. |
|
||||
| `NEWS_RECAP_HOUR` | `23` | Local hour of the daily 24h recap. |
|
||||
| `NEWS_RECAP_MINUTE` | `0` | Local minute of the daily recap. |
|
||||
| `GEMINI_API_KEY` | *(blank)* | **Required for summaries.** Unset = summarizer logs and idles (never crashes). |
|
||||
| `SUMMARY_MODEL` | `gemini-2.0-flash` | Gemini model id. |
|
||||
| `NEWS_BATCH_SIZE` | `50` | Articles per map-phase batch. |
|
||||
| `SUMMARY_WINDOW_HOURS` | `1` | How far back the summarizer looks for new articles. |
|
||||
| `INCLUDE_FUTURES` | `0` | Legacy futures-prices coupling (upstream pipeline). OFF for OSINT; set `1` + install `yfinance` to enable. |
|
||||
| `NEWS_SCRAPE_MINUTE` | `0` | Wall-clock minute the scraper fires. |
|
||||
| `NEWS_SUMMARIZE_MINUTE` | `5` | Wall-clock minute the summarizer fires. |
|
||||
| `NEWS_SCRAPE_RUN_ON_START` | `1` | Run one scrape immediately on container start. |
|
||||
| `NEWS_SUMMARIZE_RUN_ON_START` | `1` | Run one summarize immediately on container start. |
|
||||
| `NEWS_SUMMARIZE_FORCE` | `0` | `1` ignores the interval/recap idempotency skip (double-pins on recreate). |
|
||||
| `INCLUDE_FUTURES` | `0` | Legacy. Ignored — prompts never inject futures/market tape. |
|
||||
| `NEWS_LOG_LEVEL` | `INFO` | Scrapy log level. |
|
||||
| `OSINT_USER_AGENT` | `osint-dashboard-news-summarizer` | Sent on every outbound Nous call. |
|
||||
| `TELEGRAM_TOKEN` / `TELEGRAM_CHAT_ID` | *(blank)* | Reserved for the (out-of-scope) Telegram delivery bot. |
|
||||
|
||||
DB_* for both services is mapped to the shared osint-db credentials
|
||||
(`DB_HOST=db`, same `DB_USER/DB_PASSWORD/DB_NAME` as the rest of the stack).
|
||||
|
||||
Nous chat: `POST {NOUS_BASE_URL}/chat/completions` via `news/summerizer/nous_client.py`
|
||||
(`httpx`, no `openai` SDK). Auth is a Bearer token from `NOUS_API_KEY`.
|
||||
No Hermes-4 reasoning system prompt. Reduce uses `json_mode=True`.
|
||||
|
||||
## Prompts
|
||||
|
||||
Both prompts are env-overridable. Defaults recap the articles actually
|
||||
provided, ranked by breaking important news, and ignore futures / commodity
|
||||
tape. ticker/map may be empty; `summary_en` must still be a real brief.
|
||||
`RECAP_PROMPT` (23:00, 24h window) is the daily recap; `SUMMARY_PROMPT` is the
|
||||
15-min analyst. `INCLUDE_FUTURES` is ignored.
|
||||
Both prompts are env-overridable — the default `MAP_PROMPT` is OSINT-neutral
|
||||
(facts, locations, entities, category, OSINT signal per article) and the default
|
||||
`SUMMARY_PROMPT` produces a concise executive summary of the most impactful
|
||||
items (with a "no qualifying events" escape hatch). Upstream's futures/markets
|
||||
prompt language is gated behind `INCLUDE_FUTURES=1`.
|
||||
|
||||
## Tests
|
||||
|
||||
```bash
|
||||
PYTHONPATH=news/summerizer pytest news/summerizer/tests -v
|
||||
# intel + nous_client tests PASS (no network)
|
||||
`tests/test_api_news.py` — DB-backed API contract tests (auto-skip without a
|
||||
reachable test database, same as the FIRMS tests):
|
||||
|
||||
PYTHONPATH=app pytest tests/test_api_news.py \
|
||||
tests/test_api_settings.py tests/test_api_live_layers.py -v
|
||||
# DB-marked tests skip without Postgres; live_layers must still PASS
|
||||
# /api/map/layers overlays key set UNCHANGED
|
||||
```bash
|
||||
DB_HOST=... DB_PORT=... DB_USER=osint DB_PASSWORD=... DB_NAME=osint_data \
|
||||
pytest tests/test_api_news.py -v
|
||||
```
|
||||
|
||||
## Live verification
|
||||
|
||||
After deploy / compose rebuild of `news-summarizer` on the Pi:
|
||||
|
||||
1. Keys UI: save `NOUS_API_KEY` → status `****last4`.
|
||||
2. Settings: pick a model → Save → `GET /api/settings` echoes it.
|
||||
3. `docker compose --profile ingest logs -f news-summarizer` — next run (or
|
||||
`NEWS_SUMMARIZE_RUN_ON_START=1` recreate) logs `Processing N articles with <model>`.
|
||||
4. `curl -s localhost:8000/api/news/summaries?limit=1` — English `summary_text`, `model` set.
|
||||
5. `curl -s localhost:8000/api/news/ticker` — flagged headlines only.
|
||||
6. `curl -s localhost:8000/api/news/map` — only rows with lat/lon.
|
||||
7. HUD: NEWS ticker scrolls flagged items; map overlay pins popup with location.
|
||||
8. Unset key + empty keystore → summarizer logs idle, APIs return `[]`, no crash.
|
||||
|
||||
**Operator action after merge:** paste a Nous Portal API key in API Keys; pick
|
||||
a model in Settings if the default `Hermes-4.3-36B` is not wanted; rebuild
|
||||
`osint-news-summarizer` on the Pi (`pi-app-deploy` / compose).
|
||||
End-to-end (real crawl → DB → API) is verified after deploy on the Pi: check
|
||||
`docker compose --profile ingest logs -f news-scraper news-summarizer`, then
|
||||
`curl -s localhost:8000/api/news | head`. Summaries additionally require
|
||||
`GEMINI_API_KEY` to be set in `.env` on the Pi.
|
||||
|
|
|
|||
|
|
@ -1,347 +0,0 @@
|
|||
# Free satellite feeds for the OSINT map
|
||||
|
||||
Builder inventory (research profile). Probed **2026-08-29** from this machine. Do **not** treat search snippets as live — every row below had a `curl`/GET (tile, GetCapabilities, STAC, or GetMap). 404 tile rows are omitted unless Capabilities/DescribeDomains still prove the layer exists (sparse fire overlays 404 on empty tiles).
|
||||
|
||||
**Pi rules:** browser `L.tileLayer` when CORS `*`; do not proxy multi-GB COGs through the Pi; STAC+SAS like existing Sentinel-1 is “backend same as S-1”; no Redis; home uplink is small.
|
||||
|
||||
GIBS Web Mercator REST template (no key):[2]
|
||||
|
||||
```
|
||||
https://gibs.earthdata.nasa.gov/wmts/epsg3857/best/{layer}/default/{time}/{TileMatrixSet}/{z}/{y}/{x}.{jpg|png}
|
||||
```
|
||||
|
||||
Omit `{time}` for static layers. Sub-daily GOES/Himawari accept `YYYY-MM-DD` **or** `YYYY-MM-DDTHH:MI:SSZ` (GIBS snaps to nearest).[2] Attribution: NASA asks clients to acknowledge GIBS/ESDIS.[1]
|
||||
|
||||
Live GetCapabilities `epsg3857/best` on 2026-08-29: **1315** `Layer` entries, **all** with a `GoogleMapsCompatible_LevelN` matrix, `access-control-allow-origin: *`.[4] GIBS documents **1000+** visualizations; many LANCE layers appear within **3.5 hours** of observation.[3]
|
||||
|
||||
Worldview is the interactive catalog of the same tiles.[5] GIBS developer portal: Earthdata GIBS API page (HTTP 403 from this host at probe time; docs site above is the working copy).[27]
|
||||
|
||||
---
|
||||
|
||||
## Already in the product (do not rediscover)
|
||||
|
||||
| id | status |
|
||||
|---|---|
|
||||
| `BlueMarble_ShadedRelief_Bathymetry` | GIBS basemap (`app/gibs_map.py`) |
|
||||
| `VIIRS_SNPP_CorrectedReflectance_TrueColor` | GIBS basemap |
|
||||
| `MODIS_Terra_CorrectedReflectance_TrueColor` | GIBS basemap |
|
||||
| `MODIS_Aqua_CorrectedReflectance_TrueColor` | GIBS basemap |
|
||||
| `VIIRS_SNPP_DayNightBand_ENCC` | GIBS night lights |
|
||||
| FIRMS VIIRS hotspot CSV | ingest + `FIRMS_MAP_KEY` |
|
||||
| `VIIRS_SNPP_Thermal_Anomalies_375m_All` | overlay in `app/live_layers.py` (`gibs_thermal`). **Caps now say TMS `GoogleMapsCompatible_Level8`**, not Level9 — the wired URL uses Level9 (will 400). |
|
||||
| Sentinel-1 GRD | Planetary Computer STAC + SAS + TiTiler `GET /api/map/sentinel1` |
|
||||
| IEM NEXRAD / RainViewer | weather radar, not satellite |
|
||||
|
||||
Repo docs already flag **Suomi NPP product stop 2026-11-01** — swap SNPP true color / DNB / thermal / FIRMS `VIIRS_SNPP_NRT` to NOAA-20/21 before then.
|
||||
|
||||
---
|
||||
|
||||
## Ranked “add tomorrow” (sections 1–2)
|
||||
|
||||
Most new OSINT signal per **zero dollars**, browser tiles only:
|
||||
|
||||
1. **VIIRS NOAA-20 + NOAA-21 true color** — SNPP replacement, same dropdown pattern.
|
||||
2. **VIIRS false-color SWIR** (`BandsM11-I2-I1`, `BandsM3-I3-M11`, MODIS 7-2-1) — burn scars, flood, bare soil.
|
||||
3. **GIBS GOES-East/West GeoColor + Band13 IR** — 10-minute weather-sat, Hormuz + CONUS.
|
||||
4. **GIBS Himawari AHI vis + IR** — same for IO/WestPac.
|
||||
5. **HLS S30/L30** — 30 m Landsat/Sentinel-2 look without TiTiler.
|
||||
6. **OPERA RTC Sentinel-1 + DIST-ALERT + DSWx** — SAR / disturbance / flood as GIBS tiles (not COGs).
|
||||
7. **NOAA-20/21 DNB** — night lights after SNPP.
|
||||
8. **IEM GOES XYZ** — “latest” tiles, no time in the URL, already CORS `*` like NEXRAD.[6]
|
||||
9. **EUMETView WMS** — Meteosat/MTG for Europe–Africa–IO, CORS `*`.[18]
|
||||
10. **GFW GLAD-S2 / integrated deforestation alerts** — raster tiles, CORS `*` when `Origin` is sent.[15]
|
||||
11. **MUR SST + VIIRS/PACE/OLCI chlorophyll** — ocean.
|
||||
12. **MODIS NDVI 8-day + IMERG rain** — veg / flood context.
|
||||
13. **SRTM / ASTER GDEM color index** — satellite-derived DEM, static.
|
||||
14. **NOAA-20/21 thermal anomalies** — FIRMS-shaped overlay after SNPP; empty tiles 404.
|
||||
|
||||
---
|
||||
|
||||
## 1. Drop-in GIBS WMTS
|
||||
|
||||
All rows: **key? no**. **CORS `*`**. **Pi fit: browser `L.tileLayer`**. Same time-domain helper as `gibs_map.py` (`…/1.0.0/{id}/default/{tms}/all/all.xml`).
|
||||
|
||||
Format of URL column: layer id + TMS + ext. Date used in probes: `2026-08-27` unless noted.
|
||||
|
||||
### 1.1 Optical (true / false / SWIR)
|
||||
|
||||
| id | what you see | tile pattern | cadence | max zoom | license | already have? | probe |
|
||||
|---|---|---|---|---|---|---|---|
|
||||
| `VIIRS_NOAA20_CorrectedReflectance_TrueColor` | Daily true color, JPSS-1 | `…/{id}/default/{time}/GoogleMapsCompatible_Level9/{z}/{y}/{x}.jpg` | daily | 9 (~250 m) | NASA GIBS ack[1] | **no** (SNPP only) | 200 `*` jpeg |
|
||||
| `VIIRS_NOAA21_CorrectedReflectance_TrueColor` | Daily true color, JPSS-2 | same Level9 jpg | daily | 9 | NASA GIBS[1] | no | 200 |
|
||||
| `VIIRS_SNPP_CorrectedReflectance_BandsM11-I2-I1` | False color SWIR (burns, flood) | Level9 jpg | daily | 9 | NASA GIBS[1] | no | 200 |
|
||||
| `VIIRS_SNPP_CorrectedReflectance_BandsM3-I3-M11` | False color (snow/ice/desert) | Level9 jpg | daily | 9 | NASA GIBS[1] | no | 200 |
|
||||
| `VIIRS_NOAA20_CorrectedReflectance_BandsM11-I2-I1` | NOAA-20 SWIR false color | Level9 jpg | daily | 9 | NASA GIBS[1] | no | 200 |
|
||||
| `VIIRS_NOAA21_CorrectedReflectance_BandsM11-I2-I1` | NOAA-21 SWIR false color | Level9 jpg | daily | 9 | NASA GIBS[1] | no | 200 |
|
||||
| `MODIS_Terra_CorrectedReflectance_Bands721` | Classic 7-2-1 burn/SWIR | Level9 jpg | daily | 9 | NASA GIBS[1] | no | 200 |
|
||||
| `MODIS_Terra_CorrectedReflectance_Bands367` | 3-6-7 false color | Level9 jpg | daily | 9 | NASA GIBS[1] | no | 200 |
|
||||
| `MODIS_Aqua_CorrectedReflectance_Bands721` | Aqua 7-2-1 | Level9 jpg | daily | 9 | NASA GIBS[1] | no | 200 |
|
||||
| `HLS_S30_Nadir_BRDF_Adjusted_Reflectance` | Harmonized Sentinel-2 30 m RGB | Level12 png | ~2–3 d when a granule exists | 12 (~30 m) | NASA GIBS[1] | no | 200 at z=5 NC; 404 on empty granules is normal. Domain from 2015–present |
|
||||
| `HLS_L30_Nadir_BRDF_Adjusted_Reflectance` | Harmonized Landsat 30 m | Level12 png | 8–16 d | 12 | NASA GIBS[1] | no | in caps; tile 404 on empty scene |
|
||||
| `Landsat_WELD_CorrectedReflectance_TrueColor_Global_Monthly` | Landsat WELD monthly mosaic | Level12 jpg | monthly, **not NRT** | 12 | NASA GIBS[1] | no | 200 |
|
||||
| `Landsat_WELD_CorrectedReflectance_TrueColor_Global_Annual` | WELD annual | Level12 jpg | yearly | 12 | NASA GIBS[1] | no | 200 |
|
||||
|
||||
### 1.2 Weather satellites (imagery, not NEXRAD)
|
||||
|
||||
Sub-daily. Probe with `2026-08-27` **and** `2026-08-27T18:00:00Z` both 200 (nearestValue).[2]
|
||||
|
||||
| id | what you see | TMS / ext | cadence | max zoom | already have? | probe |
|
||||
|---|---|---|---|---|---|---|
|
||||
| `GOES-East_ABI_GeoColor` | GeoColor full disk (Americas, Atlantic, Hormuz west edge) | Level7 png | ~10 min | 7 | no | 200 |
|
||||
| `GOES-West_ABI_GeoColor` | GeoColor Pacific / CONUS west | Level7 png | ~10 min | 7 | no | 200 (also 200 over CA with ISO time) |
|
||||
| `GOES-East_ABI_Band2_Red_Visible_1km` | ABI vis | Level7 png | ~10 min | 7 | no | 200 |
|
||||
| `GOES-West_ABI_Band2_Red_Visible_1km` | ABI vis | Level7 png | ~10 min | 7 | no | 200 |
|
||||
| `GOES-East_ABI_Band13_Clean_Infrared` | Clean IR window | Level6 png | ~10 min | 6 | no | 200 |
|
||||
| `GOES-West_ABI_Band13_Clean_Infrared` | Clean IR | Level6 png | ~10 min | 6 | no | 200 |
|
||||
| `GOES-East_ABI_FireTemp` | Fire temperature RGB | Level7 png | ~10 min | 7 | no | 200 |
|
||||
| `GOES-West_ABI_FireTemp` | Fire temperature RGB | Level7 png | ~10 min | 7 | no | 404 on NC tile (wrong disk); use west longitudes |
|
||||
| `GOES-East_ABI_Dust` | Dust RGB | Level7 png | ~10 min | 7 | no | 200 |
|
||||
| `GOES-East_ABI_Air_Mass` | Air mass RGB | Level6 png | ~10 min | 6 | no | 200 |
|
||||
| `GOES-West_ABI_Air_Mass` | Air mass RGB | Level6 png | ~10 min | 6 | no | 200 |
|
||||
| `Himawari_AHI_Band3_Red_Visible_1km` | Himawari vis (IO / WestPac / Aus) | Level7 png | ~10 min | 7 | no | 200 (ISO time over Japan) |
|
||||
| `Himawari_AHI_Band13_Clean_Infrared` | Himawari IR | Level6 png | ~10 min | 6 | no | 200 |
|
||||
| `Himawari_AHI_Air_Mass` | Himawari air mass | Level6 png | ~10 min | 6 | no | 200 |
|
||||
|
||||
**Meteosat is not in GIBS.** Use section 2 EUMETView.
|
||||
|
||||
### 1.3 SAR / flood / disturbance (GIBS tiles — skip TiTiler)
|
||||
|
||||
| id | what you see | TMS | cadence | max zoom | already have? | probe |
|
||||
|---|---|---|---|---|---|---|
|
||||
| `OPERA_L2_Radiometric_Terrain_Corrected_SAR_Sentinel-1` | S-1 RTC browse (better than GRD for terrain) | Level12 png | scene-based from 2025-01 | 12 | **no** (you have GRD COGs, not RTC tiles) | 200 at z=5; domain 2025-01-10/… |
|
||||
| `OPERA_L3_DIST-ALERT-HLS_Color_Index` | Vegetation disturbance / clearing alert | Level12 png | ~2–3 d | 12 | no | 200 |
|
||||
| `OPERA_L3_DIST-ANN-HLS_Color_Index` | Annual DIST | Level12 png | yearly | 12 | no | in caps |
|
||||
| `OPERA_L3_Dynamic_Surface_Water_Extent-HLS` | Surface water / flood (HLS, 30 m) | Level12 png | ~2–3 d | 12 | no | 200 at z=5 |
|
||||
| `OPERA_L3_Dynamic_Surface_Water_Extent-Sentinel-1` | Surface water from S-1 (clouds irrelevant) | Level12 png | S-1 revisit | 12 | no | 200 at z=5 |
|
||||
| `NISAR_L2_Geocoded_Polarimetric_Covariance` | NISAR early browse | Level13 png | when downlinked | 13 | no | 200 (layer exists; coverage still sparse) |
|
||||
| `SMAP_L4_Analyzed_Surface_Soil_Moisture` | Soil moisture | Level6 png | daily | 6 | no | 200 |
|
||||
| `SMAP_L3_Active_Sigma0_VV` | SMAP radar σ0 | Level6 png | 2–3 d | 6 | no | in caps (SMAP radar died 2015 — historical) |
|
||||
|
||||
No ICEYE / Capella / Umbra / ALOS PALSAR **daily** layers in this GIBS 3857 dump.[4] ALOS shows up as mosaics on Planetary Computer (section 3).
|
||||
|
||||
### 1.4 Thermal / fire / volcano
|
||||
|
||||
Sparse PNG overlays: **empty tiles 404**. Capabilities + DescribeDomains still 200. Frontend must tolerate 404 (Leaflet does).
|
||||
|
||||
| id | what you see | TMS | cadence | max zoom | already have? | probe |
|
||||
|---|---|---|---|---|---|---|
|
||||
| `VIIRS_SNPP_Thermal_Anomalies_375m_All` | 375 m hotspots | **Level8** png (not 9) | daily | 8 | **yes**, but wired as Level9 | Domain 200; many tiles 404 |
|
||||
| `VIIRS_NOAA20_Thermal_Anomalies_375m_All` | NOAA-20 hotspots | Level8 png | daily | 8 | no | Domain 200 (`2020-01-01/…` through at least 2025-09); tiles 404 if no fire in tile |
|
||||
| `VIIRS_NOAA21_Thermal_Anomalies_375m_All` | NOAA-21 hotspots | Level8 png | daily | 8 | no | same |
|
||||
| `VIIRS_*_Thermal_Anomalies_375m_{Day,Night}` | day/night split | Level8 png | daily | 8 | no | in caps |
|
||||
| `MODIS_{Terra,Aqua,Combined}_Thermal_Anomalies_All` | 1 km MODIS fire | Level7 png | daily | 7 | no | in caps |
|
||||
| `GOES-East_ABI_FireTemp` | geostationary fire RGB | Level7 png | ~10 min | 7 | no | 200 |
|
||||
|
||||
Also keep FIRMS CSV — points beat raster for click/query.
|
||||
|
||||
### 1.5 Night lights (beyond current DNB ENCC)
|
||||
|
||||
| id | what you see | TMS / ext | cadence | max zoom | already have? | probe |
|
||||
|---|---|---|---|---|---|---|
|
||||
| `VIIRS_NOAA20_DayNightBand` | NOAA-20 DNB | Level7 png | daily | 7 | no | 200 |
|
||||
| `VIIRS_NOAA21_DayNightBand` | NOAA-21 DNB | Level7 png | daily | 7 | no | 200 |
|
||||
| `VIIRS_NOAA20_DayNightBand_At_Sensor_Radiance` | radiance, not ENCC | Level8 png | daily | 8 | no | 200 |
|
||||
| `VIIRS_SNPP_DayNightBand_At_Sensor_Radiance` | SNPP radiance | Level8 png | daily | 8 | no | 200 |
|
||||
| `VIIRS_NOAA20_DayNightBand_AtSensor_M15` | DNB+M15 composite jpg | Level8 jpg | daily | 8 | no | 200 |
|
||||
| `VIIRS_Night_Lights` | Black-marble-style annual-ish | Level8 png | time-dim | 8 | no | 200 on 2026-08-27 mosaic date |
|
||||
| `VIIRS_CityLights_2012` | Static 2012 city lights | Level8 jpg | **static** (`has_time=false`) | 8 | no | 200 |
|
||||
|
||||
`VIIRS_Black_Marble` and `VIIRS_NOAA20_DayNightBand_ENCC` are in caps; ENCC-NOAA20 returned HTTP 400 on the Level8 template we tried — do not ship until DescribeDomains + a known-good date are wired. SNPP ENCC stays as the current layer.
|
||||
|
||||
### 1.6 Ocean
|
||||
|
||||
| id | what you see | TMS | cadence | max zoom | probe |
|
||||
|---|---|---|---|---|---|
|
||||
| `GHRSST_L4_MUR_Sea_Surface_Temperature` | 1 km MUR SST | Level7 png | daily | 7 | 200 |
|
||||
| `GHRSST_L4_MUR_Sea_Surface_Temperature_Anomalies` | SST anomaly | Level7 png | daily | 7 | in caps |
|
||||
| `MODIS_Aqua_L3_SST_MidIR_4km_Night_Daily` | MODIS SST | Level6 png | daily | 6 | 200 |
|
||||
| `MODIS_Aqua_L2_Chlorophyll_A` | Aqua chl-a | Level7 png | daily | 7 | 200 |
|
||||
| `VIIRS_SNPP_L2_Chlorophyll_A` | VIIRS chl-a | Level7 png | daily | 7 | 200 |
|
||||
| `VIIRS_NOAA20_Chlorophyll_a` | NOAA-20 chl-a | Level7 png | daily | 7 | 200 |
|
||||
| `OCI_PACE_Chlorophyll_a` | PACE OCI chl-a | Level7 png | daily | 7 | 200 |
|
||||
| `S3A_OLCI_Chlorophyll_a` | Sentinel-3A OLCI chl-a | Level7 png | daily | 7 | 200 |
|
||||
| `S3B_OLCI_Chlorophyll_a` | Sentinel-3B OLCI | Level7 png | daily | 7 | in caps |
|
||||
| `MODIS_Terra_Sea_Ice` | sea ice | Level7 png | daily | 7 | 200 |
|
||||
| `GHRSST_L4_MUR_Sea_Ice_Concentration` | MUR ice | Level7 png | daily | 7 | in caps |
|
||||
|
||||
No dedicated “SAR oil slick” GIBS layer in the 3857 dump. Closest: OPERA RTC / DSWx-S1 + existing S-1 GRD TiTiler.
|
||||
|
||||
### 1.7 Vegetation / burn / flood / precip / atm
|
||||
|
||||
| id | what you see | TMS | cadence | max zoom | probe |
|
||||
|---|---|---|---|---|---|
|
||||
| `MODIS_Terra_NDVI_8Day` | NDVI | Level9 png | 8-day | 9 | 200 |
|
||||
| `MODIS_Terra_L3_NDVI_16Day` | NDVI 16-day | Level9 png | 16-day | 9 | 200 |
|
||||
| `IMERG_Precipitation_Rate` | GPM IMERG rain | Level6 png | sub-daily | 6 | 200 |
|
||||
| `MODIS_Terra_Aerosol` | AOD | Level6 png | daily | 6 | 200 |
|
||||
| `MODIS_Terra_Land_Surface_Temp_Day` | LST | Level7 png | daily | 7 | 200 |
|
||||
| `VIIRS_SNPP_Land_Surface_Temp_Day` | VIIRS LST | Level7 png | daily | 7 | 200 |
|
||||
| `AIRS_L3_Carbon_Monoxide_500hPa_Volume_Mixing_Ratio_Daily_Night` | CO (fires, industry) | Level6 png | daily | 6 | 200 |
|
||||
| `OMI_NO2` / `OMI_Aerosol_Index` | NO2 / smoke index | Level6 png | daily | 6 | OMI AI 200; several OMPS 200 |
|
||||
| `MODIS_Water_Mask` | static water mask | Level9 png | static | 9 | 200 |
|
||||
|
||||
MODIS burned-area monthly (`MCD64` / `MODIS_Combined_L3_Burned_Area_Monthly`) is in caps; our dated tile 400’d — wire only after a DescribeDomains date hits 200.
|
||||
|
||||
### 1.8 DEM (satellite-derived, tileable)
|
||||
|
||||
| id | what you see | TMS / ext | cadence | max zoom | probe |
|
||||
|---|---|---|---|---|---|
|
||||
| `SRTM_Color_Index` | SRTM elevation color | Level12 png | static | 12 | 200 |
|
||||
| `ASTER_GDEM_Color_Index` | ASTER GDEM color | Level12 png | static | 12 | 200 |
|
||||
| `ASTER_GDEM_Color_Shaded_Relief` | ASTER hillshade | Level12 jpg | static | 12 | 200 |
|
||||
| `ASTER_GDEM_Greyscale_Shaded_Relief` | grey hillshade | Level12 jpg | static | 12 | in caps |
|
||||
| `GEDI_ISS_L3_Canopy_Height_Mean_RH100_201904-202303` | GEDI canopy height | Level7 png | static epoch | 7 | related GEDI biomass 200 |
|
||||
|
||||
Blue Marble shaded relief is **already** the basemap — these are extra.
|
||||
|
||||
---
|
||||
|
||||
## 2. Other XYZ / WMTS / WMS (no key)
|
||||
|
||||
Ranked after GIBS for signal/$; still free.
|
||||
|
||||
| id | what you see | provider | URL pattern | key? | CORS | cadence | max zoom / res | license / attribution | Pi fit | already have? | probe 2026-08-29 |
|
||||
|---|---|---|---|---|---|---|---|---|---|---|---|
|
||||
| `iem_goes_east_conus_ch02` | GOES-East CONUS ABI ch02 vis, **latest** | Iowa State IEM | `https://mesonet.agron.iastate.edu/cache/tile.py/1.0.0/goes_east_conus_ch02/{z}/{x}/{y}.png` | no | `*` | ~5 min cache header | TMS; vis ~1 km | Cite IEM / NOAA GOES[6][7] | **browser** (same stack as NEXRAD) | no | 200 image/png `*` |
|
||||
| `iem_goes_east_conus_ch13` | GOES-East CONUS IR ch13 | IEM | `…/goes_east_conus_ch13/{z}/{x}/{y}.png` | no | `*` | ~5 min | IR ~2 km | IEM[6] | browser | no | 200 |
|
||||
| `iem_goes_east_fulldisk_ch02` | GOES-East full disk vis | IEM | `…/goes_east_fulldisk_ch02/{z}/{x}/{y}.png` | no | `*` | ~5 min | full disk | IEM[6] | browser | no | 200 |
|
||||
| `iem_goes_west_conus_ch02` | GOES-West CONUS vis | IEM | `…/goes_west_conus_ch02/{z}/{x}/{y}.png` | no | `*` | ~5 min | | IEM[6] | browser | no | 200 |
|
||||
| `iem_goes_vis_1km` | Legacy name → GOES-East vis | IEM | `…/goes-vis-1km/{z}/{x}/{y}.png` | no | `*` | ~5 min | | IEM[6] | browser | no | 200 |
|
||||
| IEM GOES template | Any bird/sector/channel | IEM | `goes_{east\|west}_{fulldisk\|conus\|mesoscale-1\|mesoscale-2\|alaska\|puertorico}_ch{01–16}`[6] | no | `*` | NRT | 16 ABI bands | IEM[6] | browser | no | template documented; ch02/ch13 probed |
|
||||
| `eumet_msg_natural` | Meteosat natural color | EUMETSAT EUMETView GeoServer | WMS `https://view.eumetsat.int/geoserver/ows` layer `msg_fes:rgb_natural` EPSG:3857 GetMap | no | `*` | NRT | SEVIRI ~3 km | EUMETSAT viz; cite EUMETSAT[18] | **browser `L.tileLayer.wms`** (not XYZ). Caps 200, 165 layer names | no | GetMap 200 image/png `*` |
|
||||
| `eumet_msg_ir108` | Meteosat IR 10.8 | EUMETView | WMS `msg_fes:ir108` | no | `*` | NRT | | EUMETSAT | browser WMS | no | name in caps |
|
||||
| `eumet_msg_fire` | Meteosat fire | EUMETView | WMS `msg_fes:fire` | no | `*` | NRT | | EUMETSAT | browser WMS | no | name in caps |
|
||||
| `eumet_mtg_ir105` | MTG-I IR | EUMETView | WMS `mtg_fd:ir105_hrfi` | no | `*` | NRT | FCI | EUMETSAT | browser WMS | no | name in caps |
|
||||
| `eumet_s3_olci_rgb` | S3 OLCI RGB mosaic | EUMETView | WMS `copernicus:daily_sentinel3ab_olci_l1_rgb_fulres` | no | `*` | daily | OLCI | Copernicus/EUMETSAT | browser WMS | no | name in caps |
|
||||
| `eumet_s3_chl` | S3 chl-a | EUMETView | WMS `copernicus:daily_sentinel3ab_olci_l2_chl_fullres` | no | `*` | daily | | Copernicus | browser WMS | no | name in caps |
|
||||
| `gfw_glad_s2` | GLAD Sentinel-2 deforestation alerts | GFW tile cache | `https://tiles.globalforestwatch.org/umd_glad_sentinel2_alerts/latest/default/{z}/{x}/{y}.png` | no | `*` **if `Origin` header** (null without it) | ~daily | raster z 0–22 documented[15] | WRI/UMD; cite GFW | **browser** (Leaflet sends Origin) | no | 200 image/png; with Origin → CORS `*` |
|
||||
| `gfw_integrated` | Integrated deforestation alerts | GFW | `https://tiles.globalforestwatch.org/gfw_integrated_alerts/latest/default/{z}/{x}/{y}.png` | no | `*` + Origin | ~daily | | GFW | browser | no | 200 |
|
||||
| `gfw_tcl` | UMD tree-cover loss | GFW | `https://tiles.globalforestwatch.org/umd_tree_cover_loss/latest/tcd_30/{z}/{x}/{y}.png` | no | (same host) | annual | | GFW/UMD | browser | no | 200 |
|
||||
| `star_goes19_fd_geocolor` | GOES-19 full-disk GeoColor **JPEG** (not XYZ) | NOAA NESDIS STAR CDN | `https://cdn.star.nesdis.noaa.gov/GOES19/ABI/FD/GEOCOLOR/latest.jpg` also `…/CONUS/GEOCOLOR/latest.jpg` | no | `*` | minutes | full-disk / CONUS image | NOAA | **not a map layer** — optional lightbox. Do not tile-proxy | no | 200 jpeg `*`[21] |
|
||||
| `star_goes18_fd_geocolor` | GOES-18 FD GeoColor JPEG | STAR | `https://cdn.star.nesdis.noaa.gov/GOES18/ABI/FD/GEOCOLOR/latest.jpg` | no | `*` | minutes | | NOAA | lightbox only | no | 200 |
|
||||
|
||||
IEM JSON `…/GOES/conus/channel02/GOES-16_C02.json` is **stale** (`generated_at` 2025-04-07) but the **tile names still 200**. Prefer GIBS GeoColor when you need a time slider; prefer IEM when you want “whatever is latest” with zero time plumbing.[6][7]
|
||||
|
||||
### 2.x Works but **not** browser-direct (no CORS)
|
||||
|
||||
| id | what you see | URL | CORS | Pi fit | probe |
|
||||
|---|---|---|---|---|---|
|
||||
| RAMMB/CIRA SLIDER GeoColor tiles | GOES-19 / Himawari / JPSS loops, ~10 min | Times: `https://rammb-slider.cira.colostate.edu/data/json/goes-19/full_disk/geocolor/latest_times.json` (`timestamps_int`). Tile: `https://rammb-slider.cira.colostate.edu/data/imagery/{YYYY}/{MM}/{DD}/goes-19---full_disk/geocolor/{ts}/{zz}/{yyy}_{xxx}.png` e.g. `…/2026/08/28/goes-19---full_disk/geocolor/20260828225021/00/000_000.png`. Himawari times JSON also 200. | **none** | **Do not proxy tiles through the Pi.** Bookmark / deep-link SLIDER instead.[19][26] | times 200; tile 200 png; CORS null |
|
||||
| NICT Himawari-8 Real-time Web | 10-min full disk PNG grid | `https://himawari8.nict.go.jp/img/D531106/latest.json` then `https://himawari8.nict.go.jp/img/D531106/2d/550/{YYYY}/{MM}/{DD}/{HHMMSS}_{x}_{y}.png` | **none** | same — no Pi proxy | latest.json 200; tile 200 png; CORS null[22] |
|
||||
| OpenAerialMap | Per-scene TMS of open UAV/sat | `https://api.openaerialmap.org/meta` → `properties.tms` | CORS **only** `https://map.openaerialmap.org` | not usable from the dashboard origin without a proxy; opportunistic, not a global basemap[23][25] | meta 200 |
|
||||
| USGS LandsatLook STAC | Landsat C2 STAC | `https://landsatlook.usgs.gov/stac-server` | CORS locked to `https://landsatlook.usgs.gov/stac-server` | backend-only if ever; prefer Earth Search / PC / GIBS HLS[24] | collections + search 200 |
|
||||
| NOAA CoastWatch ERDDAP WMS (`jplMURSST41`) | MUR SST WMS | `https://coastwatch.pfeg.noaa.gov/erddap/wms/jplMURSST41/request` | mixed | **flaky**: GetCapabilities 200 earlier, **503** on later GetMap/GetCapabilities. Prefer GIBS MUR | 503 on 2nd pass |
|
||||
| RainViewer `satellite.infrared` | would be IR sat frames | `https://api.rainviewer.com/public/weather-maps.json` | `*` | **empty list** (`"infrared": []`) at probe time — do not ship. Radar path already in product[20] | JSON 200, satellite IR empty |
|
||||
|
||||
---
|
||||
|
||||
## 3. STAC / COG (TiTiler, same pattern as Sentinel-1)
|
||||
|
||||
Do **not** stream COGs through the Pi for a basemap. Viewport bbox + short datetime window + SAS/public HTTPS + existing TiTiler. Prefer GIBS HLS / OPERA tiles (section 1) when a browse PNG is enough.
|
||||
|
||||
| id | what you see | provider | STAC | key? | CORS | cadence | res | license | Pi fit | already have? | probe |
|
||||
|---|---|---|---|---|---|---|---|---|---|---|---|
|
||||
| `sentinel-2-l2a` (Earth Search) | S2 L2A COGs, public HTTPS | Element 84 / AWS Open Data | `https://earth-search.aws.element84.com/v1` collections: `sentinel-2-l2a`, `sentinel-2-c1-l2a`, `sentinel-2-l1c`, `sentinel-2-pre-c1-l2a`, `sentinel-1-grd`, `landsat-c2-l2`, `naip`, `cop-dem-glo-30`, `cop-dem-glo-90`[8][9] | no | STAC `*` | S2 ~5 d | 10 m | Copernicus open; AWS public bucket HTTPS (not requester-pays for these COGs)[9] | **backend same as S-1**: search → TCI/visual COG → TiTiler. Live item `S2A_40RCP_20260827_0_L2A` href `https://sentinel-cogs.s3.us-west-2.amazonaws.com/…/TCI.tif` | no | collections + search 200 `*` |
|
||||
| `sentinel-2-l2a` (Planetary Computer) | same S2 on Azure | Microsoft PC | `https://planetarycomputer.microsoft.com/api/stac/v1/collections/sentinel-2-l2a` | SAS token (unsigned search works) | STAC `*` | ~5 d | 10 m | Copernicus; Azure blob needs SAS like current S-1 | backend same as S-1 | no | collection + search 200 `*` (136 collections listed)[10][11] |
|
||||
| `sentinel-1-rtc` | S-1 IW RTC γ0 COGs | PC / Catalyst | `/collections/sentinel-1-rtc` | **PC account required to retrieve SAS** for RTC blobs[12] | STAC `*` | IW land | ~10 m pixels | **CC BY 4.0**[12] | backend same as S-1 **plus** PC login for SAS. Prefer GIBS OPERA RTC tiles if browse is enough | no | collection 200; search item `S1D_IW_GRDH_…_rtc` assets `vv,vh,tilejson,rendered_preview` |
|
||||
| `sentinel-1-grd` (PC) | GRD | PC | `/collections/sentinel-1-grd` | SAS | `*` | 6–12 d | | Copernicus | **already have** | yes | 200 |
|
||||
| `sentinel-1-grd` (Earth Search) | GRD on AWS | E84 | `/collections/sentinel-1-grd` | requester-pays **s3://** URLs per E84 README[9] | `*` | | | Copernicus | worse than PC for the Pi (AWS creds) | no | collection 200 |
|
||||
| `landsat-c2-l2` | Landsat 8/9 SR | E84 + PC | both catalogs | no / SAS | `*` | 8–16 d | 30 m | USGS public | backend TiTiler; or just use GIBS HLS | no | both 200 |
|
||||
| `hls2-s30` / `hls2-l30` | HLS v2 COGs | PC | `/collections/hls2-s30`, `hls2-l30` | SAS | `*` | 2–3 d | 30 m | NASA | prefer GIBS HLS tiles | no | collections 200 |
|
||||
| `goes-cmi` | GOES Cloud & Moisture Imagery COGs | PC | `/collections/goes-cmi` | SAS | `*` | 5–10 min | ABI | NOAA | **overkill vs GIBS/IEM tiles** | no | collection 200 |
|
||||
| `modis-14A1-061` / `modis-64A1-061` | MODIS fire / burned area | PC | `/collections/modis-14A1-061`, `modis-64A1-061` | SAS | `*` | daily / monthly | 1 km / 500 m | NASA | prefer GIBS fire tiles + FIRMS | no | 200 |
|
||||
| `alos-palsar-mosaic` / `alos-fnf-mosaic` | ALOS PALSAR yearly mosaic / forest-nonforest | PC | those collection ids | SAS | `*` | **annual** | 25 m | JAXA (check collection) | backend mosaic, not live SAR | no | 200 |
|
||||
| `nasadem` / `cop-dem-glo-30` | DEM COGs | PC + E84 | `nasadem`, `cop-dem-glo-30` | public / SAS | `*` | static | 30 m | NASA / Copernicus | prefer GIBS SRTM/ASTER tiles | no | 200 |
|
||||
| `naip` | USDA NAIP aerial (CONUS) | E84 + PC | `naip` | no | `*` | leaf-on, not NRT | ~0.6 m | USDA | CONUS only; huge. Optional TiTiler | no | 200 |
|
||||
| `io-lulc-annual-v02` | 10 m land cover | PC | `io-lulc-annual-v02` | SAS | `*` | annual | 10 m | various | overlay, not sat photo | no | 200 |
|
||||
| CDSE `sentinel-2-l2a` / `sentinel-1-grd` | Copernicus Dataspace STAC | ESA CDSE | `https://stac.dataspace.copernicus.eu/v1/collections/sentinel-2-l2a` (lowercase ids work; `SENTINEL-2` 404) | **free account** for many assets | **CORS none** | same as ESA | | Copernicus | backend only; Earth Search/PC easier on a Pi | no | collection 200, CORS null. List endpoint is paginated (first page was CLMS burned-area COGs)[14] |
|
||||
|
||||
PC catalog also has Sentinel-3 OLCI/SLSTR NetCDF, Sentinel-5P, GOES-GLM — NetCDF is a bad TiTiler citizen; use GIBS/EUMETView for those.
|
||||
|
||||
---
|
||||
|
||||
## 4. Free-account / license-gated (no card this week)
|
||||
|
||||
| id | note | why not a dropdown tomorrow |
|
||||
|---|---|---|
|
||||
| Microsoft PC SAS for RTC (and some blobs) | “A Planetary Computer account is required to retrieve SAS tokens to read the RTC data.”[12] | Search is open; **read** needs an account. GRD path you already have may not need this. |
|
||||
| Copernicus Data Space (`stac.dataspace.copernicus.eu`) | STAC search 200 without cookie; **no CORS**; downloads often need a free CDSE login | Use Earth Search/PC unless you want official ESA provenance |
|
||||
| JAXA P-Tree / Himawari Monitor | Himawari standard data, account | NICT/GIBS already cover browse |
|
||||
| EUMETSAT Data Store | full MTG/MSG granules | EUMETView WMS is the browse path |
|
||||
| FIRMS map key | already in product | add `VIIRS_NOAA20_NRT` / `VIIRS_NOAA21_NRT` before SNPP sunset |
|
||||
| USGS ERS / EarthExplorer | Landsat/ASTER download login | GIBS HLS + Earth Search cover browse |
|
||||
| Planet Tropical Forest Observatory | paid successor after NICFI | see skip |
|
||||
|
||||
---
|
||||
|
||||
## 5. Skip / costs money / dead
|
||||
|
||||
| id | why |
|
||||
|---|---|
|
||||
| **NICFI / Planet tropical mosaics (free)** | Free NICFI phase **ended 1 Apr 2025**. Removed from GFW and Collect Earth Online. Successor is Planet **Tropical Forest Observatory (subscription)** or a future NICFI re-compete.[16][17] PC collections `planet-nicfi-analytic` / `planet-nicfi-visual` still exist but assets are **RFP winners only** + proprietary PLA.[13] |
|
||||
| Sentinel Hub (paid tiers) | billed processing units |
|
||||
| Google Earth Engine | billing project |
|
||||
| Maxar / Planet commercial | $ |
|
||||
| ICEYE commercial | no free global tile/STAC found this pass |
|
||||
| Umbra / Capella / Maxar **open data** STAC | catalogs 200 (`maxar-opendata`, `umbra-open-data-catalog`) but **disaster events only**, not a standing layer |
|
||||
| Esri World Imagery / Clarity | tiles 200 CORS `*` — **ToS not a free basemap we should wrap** |
|
||||
| Mapbox / Google satellite | key + ToS |
|
||||
| GEE Dynamic World / NICFI in EE | EE billing |
|
||||
| `nowcoast.noaa.gov` | HTTP **403** |
|
||||
| FIRMS WMS (`firms.modaps.eosdis.nasa.gov/wms/…`) | HTTP **404** — use CSV + GIBS |
|
||||
| RainViewer satellite IR | payload empty[20] |
|
||||
| CoastWatch ERDDAP | 503 at probe; GIBS MUR replaces SST |
|
||||
| Proxying RAMMB or NICT tiles | no CORS; would soak the home uplink |
|
||||
|
||||
---
|
||||
|
||||
## Implementation notes for builders
|
||||
|
||||
1. **GIBS dropdown:** reuse `MAP_LAYERS` in `app/gibs_map.py`. New rows are `{id, title, tms, format, has_time, max_zoom}`. Time-domain fetch already exists.
|
||||
2. **SNPP sunset:** NOAA-20/21 true color, DNB, thermal, FIRMS datasets first. SNPP true color can stay as fallback until 2026-11-01.
|
||||
3. **Fix thermal TMS:** caps say `GoogleMapsCompatible_Level8` for `VIIRS_*_Thermal_Anomalies_375m_*`. Level9 GetTile is HTTP 400 XML.
|
||||
4. **GOES time:** either GIBS `{time}` ISO + existing date slider, or IEM “latest” XYZ with no time (simpler, CONUS/FD only).
|
||||
5. **HLS / OPERA:** empty granules 404 — same as “today’s MODIS isn’t ingested yet”. Clamp latest date via DescribeDomains like current daily mosaics.
|
||||
6. **Do not add TiTiler S2 as a global basemap.** 10 m COGs will thrash the Pi. GIBS HLS Level12 is the browse path; Earth Search TCI is a “inspect this viewport” action like S-1.
|
||||
7. **EUMETView:** `L.tileLayer.wms` against `https://view.eumetsat.int/geoserver/ows`, layers `msg_fes:rgb_natural` / `msg_fes:ir108`. Caps CORS `*`.
|
||||
8. **GFW:** send browser Origin (Leaflet does). No key.
|
||||
9. **Attribution strings:** NASA GIBS acknowledgment[1]; IEM; EUMETSAT; GFW/UMD; NOAA STAR.
|
||||
|
||||
### Probe stats (this run)
|
||||
|
||||
- GIBS WMTS caps: 5 796 177 bytes, CORS `*`, 1315 layers.[4]
|
||||
- Curated GIBS GetTile: **78/90 HTTP 200** first batch; extra GOES/DNB/HLS/OPERA/ocean 200 as tabulated.
|
||||
- Earth Search collections (complete list): `sentinel-2-pre-c1-l2a`, `cop-dem-glo-30`, `naip`, `cop-dem-glo-90`, `landsat-c2-l2`, `sentinel-2-l2a`, `sentinel-2-l1c`, `sentinel-2-c1-l2a`, `sentinel-1-grd`.[8]
|
||||
- Planetary Computer: 136 collections; S-1 RTC CC-BY-4.0.[11][12]
|
||||
|
||||
Raw probe JSON lives next to this file in the kanban workspace (`gibs_probes.json`, `wave2_probes.json`, `wave3_probes.json`, `gibs_all_layers.json`).
|
||||
|
||||
## Sources
|
||||
|
||||
[1] https://nasa-gibs.github.io/gibs-api-docs
|
||||
[2] https://nasa-gibs.github.io/gibs-api-docs/access-basics
|
||||
[3] https://nasa-gibs.github.io/gibs-api-docs/available-visualizations
|
||||
[4] https://gibs.earthdata.nasa.gov/wmts/epsg3857/best/1.0.0/WMTSCapabilities.xml
|
||||
[5] https://worldview.earthdata.nasa.gov
|
||||
[6] https://mesonet.agron.iastate.edu/ogc
|
||||
[7] https://mesonet.agron.iastate.edu/GIS/goes.phtml
|
||||
[8] https://earth-search.aws.element84.com/v1/collections
|
||||
[9] https://github.com/Element84/earth-search
|
||||
[10] https://planetarycomputer.microsoft.com/catalog
|
||||
[11] https://planetarycomputer.microsoft.com/api/stac/v1/collections
|
||||
[12] https://planetarycomputer.microsoft.com/dataset/sentinel-1-rtc
|
||||
[13] https://planetarycomputer.microsoft.com/dataset/planet-nicfi-analytic
|
||||
[14] https://stac.dataspace.copernicus.eu/v1/collections
|
||||
[15] https://tiles.globalforestwatch.org
|
||||
[16] https://www.collect.earth/planet-imagery-via-nicfi-is-no-longer-available-on-ceo
|
||||
[17] https://www.globalforestwatch.org/blog/data-and-tools/planet-imagery-changes-gfw
|
||||
[18] https://view.eumetsat.int/geoserver/ows?service=WMS&request=GetCapabilities
|
||||
[19] https://rammb-slider.cira.colostate.edu
|
||||
[20] https://api.rainviewer.com/public/weather-maps.json
|
||||
[21] https://cdn.star.nesdis.noaa.gov/GOES19/ABI/FD/GEOCOLOR/latest.jpg
|
||||
[22] https://himawari8.nict.go.jp
|
||||
[23] https://api.openaerialmap.org/meta?limit=1
|
||||
[24] https://landsatlook.usgs.gov/stac-server/collections
|
||||
[25] https://openaerialmap.org
|
||||
[26] https://bellingcat.gitbook.io/toolkit/more/all-tools/rammb-slider
|
||||
[27] https://www.earthdata.nasa.gov/engage/open-data-services-software/earthdata-developer-portal/gibs-api
|
||||
|
|
@ -1,32 +0,0 @@
|
|||
"""Feed helpers shared by the news spider (no Scrapy import)."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import datetime
|
||||
from email.utils import parsedate_to_datetime
|
||||
from urllib.parse import urlparse
|
||||
|
||||
AUDIO_EXT = (".mp3", ".m4a", ".ogg", ".wav", ".aac", ".flac", ".opus")
|
||||
|
||||
|
||||
def is_audio_url(url: str) -> bool:
|
||||
path = urlparse(url or "").path.lower()
|
||||
return any(path.endswith(ext) for ext in AUDIO_EXT)
|
||||
|
||||
|
||||
def article_timestamp(pub_date: str | None) -> datetime.datetime:
|
||||
"""Prefer the feed's pubDate/published; fall back to now (UTC)."""
|
||||
if pub_date:
|
||||
try:
|
||||
return parsedate_to_datetime(pub_date).astimezone(datetime.timezone.utc)
|
||||
except (TypeError, ValueError, IndexError):
|
||||
pass
|
||||
try:
|
||||
raw = pub_date.replace("Z", "+00:00")
|
||||
ts = datetime.datetime.fromisoformat(raw)
|
||||
if ts.tzinfo is None:
|
||||
ts = ts.replace(tzinfo=datetime.timezone.utc)
|
||||
return ts
|
||||
except ValueError:
|
||||
pass
|
||||
return datetime.datetime.now(datetime.timezone.utc)
|
||||
|
|
@ -3,7 +3,7 @@
|
|||
# Don't forget to add your pipeline to the ITEM_PIPELINES setting
|
||||
# See: https://docs.scrapy.org/en/latest/topics/item-pipeline.html
|
||||
|
||||
import logging
|
||||
import logging
|
||||
import psycopg2
|
||||
import os
|
||||
from scrapy.exceptions import DropItem
|
||||
|
|
@ -53,10 +53,8 @@ class PostgresPipeline:
|
|||
self.connection.commit()
|
||||
|
||||
def process_item(self, item, spider):
|
||||
url = item['url']
|
||||
if url in self.seen_urls:
|
||||
raise DropItem(f"Duplicate URL (in-memory): {url}")
|
||||
self.seen_urls.add(url)
|
||||
if item ['url'] in self.seen_urls:
|
||||
raise DropItem()
|
||||
try:
|
||||
self.cur.execute("""
|
||||
INSERT INTO articles (title, url, content, domain, timestamp)
|
||||
|
|
@ -70,16 +68,15 @@ class PostgresPipeline:
|
|||
item['timestamp']
|
||||
))
|
||||
if self.cur.rowcount == 0:
|
||||
raise DropItem(f"Duplicate URL (database): {url}")
|
||||
e = DropItem("Duplicate URL (database conflict)")
|
||||
e.log_level = logging.DEBUG
|
||||
raise e
|
||||
self.connection.commit()
|
||||
return item
|
||||
except DropItem:
|
||||
self.connection.rollback()
|
||||
raise
|
||||
return item
|
||||
except Exception as e:
|
||||
spider.logger.error(f"Error saving to Postgres: {e}")
|
||||
self.connection.rollback()
|
||||
raise
|
||||
raise
|
||||
|
||||
def close_spider(self, spider):
|
||||
self.cur.close()
|
||||
|
|
@ -91,3 +88,5 @@ from itemadapter import ItemAdapter
|
|||
class NewsscraperPipeline:
|
||||
def process_item(self, item, spider):
|
||||
return item
|
||||
|
||||
|
||||
|
|
|
|||
|
|
@ -4,13 +4,11 @@ from urllib.parse import urljoin, urlparse
|
|||
import datetime
|
||||
import re
|
||||
|
||||
from newsScraper.feed_util import article_timestamp, is_audio_url
|
||||
|
||||
|
||||
class NewsRSSSpider(Spider):
|
||||
"""Crawl the curated news sources in urls.txt and extract articles.
|
||||
|
||||
urls.txt contains curated news HOMEPAGES and RSS/Atom feeds, so this spider
|
||||
urls.txt contains 257 news HOMEPAGES (not feed URLs), so this spider
|
||||
implements feed autodiscovery: it fetches each start URL, finds the
|
||||
RSS/Atom feed link (`<link rel="alternate" type="application/rss+xml">`
|
||||
or a visible /rss|/feed link), follows it, and then follows each feed
|
||||
|
|
@ -89,8 +87,6 @@ class NewsRSSSpider(Spider):
|
|||
or node.xpath('updated/text()').get()
|
||||
)
|
||||
if link:
|
||||
if is_audio_url(link):
|
||||
continue
|
||||
yield scrapy.Request(
|
||||
link,
|
||||
callback=self.parse_article,
|
||||
|
|
@ -116,5 +112,5 @@ class NewsRSSSpider(Spider):
|
|||
'url': response.url,
|
||||
'text': pure_text,
|
||||
'domain': urlparse(response.url).netloc,
|
||||
'timestamp': article_timestamp(response.meta.get('date')).isoformat()
|
||||
'timestamp': datetime.datetime.now().isoformat()
|
||||
}
|
||||
|
|
|
|||
|
|
@ -1,12 +1,18 @@
|
|||
#!/usr/bin/env python3
|
||||
"""Scheduler loop for the news scraper — crawl continuously.
|
||||
"""Scheduler loop for the news scraper — hourly scrape at minute :00.
|
||||
|
||||
As soon as one Scrapy pass finishes, wait NEWS_SCRAPE_INTERVAL_S seconds
|
||||
and start the next. Two crawls never overlap (the loop is serial).
|
||||
Replaces the k8s CronJob (`0 * * * *`) with an in-compose loop so the whole
|
||||
news pipeline lives inside docker-compose. Each iteration:
|
||||
|
||||
1. (optionally, on first boot) runs the Scrapy crawl once to seed data fast
|
||||
2. sleeps until the next :NEWS_SCRAPE_MINUTE wall-clock boundary
|
||||
|
||||
Because the loop is serial, a crawl that overruns its hour simply delays the
|
||||
next run to the following boundary — two crawls never overlap.
|
||||
|
||||
Env (all optional, 12-factor):
|
||||
NEWS_SCRAPE_INTERVAL_S seconds between crawls (default 10)
|
||||
NEWS_SCRAPE_RUN_ON_START "1" to crawl immediately on boot (default 1)
|
||||
NEWS_SCRAPE_MINUTE minute of the hour to fire (default 0)
|
||||
NEWS_SCRAPE_RUN_ON_START "1" to crawl once immediately on boot (default 1)
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
|
@ -21,12 +27,19 @@ import time
|
|||
logging.basicConfig(level=logging.INFO, format="%(asctime)s %(levelname)s %(name)s: %(message)s")
|
||||
logger = logging.getLogger("news.scraper")
|
||||
|
||||
INTERVAL_S = max(0, int(os.getenv("NEWS_SCRAPE_INTERVAL_S", "10")))
|
||||
MINUTE = int(os.getenv("NEWS_SCRAPE_MINUTE", "0"))
|
||||
RUN_ON_START = os.getenv("NEWS_SCRAPE_RUN_ON_START", "1").lower() in ("1", "true", "yes")
|
||||
|
||||
CRAWL_CMD = ["scrapy", "crawl", "articles"]
|
||||
|
||||
|
||||
def seconds_until_next(minute: int) -> float:
|
||||
"""Seconds until the next occurrence of ``minute`` past the hour (local time)."""
|
||||
now = datetime.datetime.now()
|
||||
nxt = now.replace(minute=minute, second=0, microsecond=0) + datetime.timedelta(hours=1)
|
||||
return (nxt - now).total_seconds()
|
||||
|
||||
|
||||
def run_crawl() -> None:
|
||||
logger.info("scrape starting at %s", datetime.datetime.now().isoformat(timespec="seconds"))
|
||||
try:
|
||||
|
|
@ -38,14 +51,15 @@ def run_crawl() -> None:
|
|||
|
||||
def main() -> None:
|
||||
logger.info(
|
||||
"news scraper loop starting (interval_s=%s, run_on_start=%s)",
|
||||
INTERVAL_S, RUN_ON_START,
|
||||
"news scraper loop starting (minute=%s, run_on_start=%s)",
|
||||
MINUTE, RUN_ON_START,
|
||||
)
|
||||
if RUN_ON_START:
|
||||
run_crawl()
|
||||
while True:
|
||||
logger.info("next scrape in %ss", INTERVAL_S)
|
||||
time.sleep(INTERVAL_S)
|
||||
delay = seconds_until_next(MINUTE)
|
||||
logger.info("next scrape at :%02d (in %.0fs)", MINUTE, delay)
|
||||
time.sleep(delay)
|
||||
run_crawl()
|
||||
|
||||
|
||||
|
|
|
|||
|
|
@ -1,336 +1,257 @@
|
|||
https://www.investing.com/rss/news.rss
|
||||
https://www.ftchinese.com/rss
|
||||
https://www.alwatan.com
|
||||
https://albiladpress.com
|
||||
https://www.aletihad.ae/
|
||||
https://www.albayan.ae
|
||||
https://www.aljazeera.com/xml/rss/all.xml
|
||||
https://english.alarabiya.net/.mrss/en.xml
|
||||
https://english.aawsat.com/home/rss
|
||||
https://www.newarab.com/rss
|
||||
https://www.skynewsarabia.com/rss/feeds/rss-1.xml
|
||||
https://www.thenationalnews.com/arc/outboundfeeds/rss/
|
||||
https://www.arabnews.com/rss.xml
|
||||
https://gulfnews.com/rss
|
||||
https://www.kuwaittimes.com/feed/
|
||||
https://www.omanobserver.om/feed/
|
||||
https://www.khaleejtimes.com/rss/news
|
||||
http://www.akhbar-alkhaleej.com/rss/all
|
||||
https://today.lorientleyour.com/rss
|
||||
https://www.annahar.com/english/rss
|
||||
https://english.almayadeen.net/rss
|
||||
https://english.ahram.org.eg/rss/0/Home.aspx
|
||||
https://www.dailynewsegypt.com/feed/
|
||||
http://www.jordantimes.com/rss
|
||||
https://www.alraimedia.com/rss
|
||||
https://alghad.com/feed/
|
||||
https://nypost.com/feed/
|
||||
https://gothamist.com/feed/
|
||||
https://www.cityandstateny.com/rss
|
||||
https://feeds.nytimes.com/nyt/rss/HomePage
|
||||
https://www.thecity.nyc/rss/index.xml
|
||||
https://brooklyneagle.com/feed/
|
||||
https://www.reutersagency.com/feed/
|
||||
https://newsatme.com/api/v1/rss/ap/world
|
||||
https://feeds.bbci.co.uk/news/world/rss.xml
|
||||
https://rss.dw.com/rdf/rss-en-all
|
||||
https://www.france24.com/en/rss
|
||||
https://www3.nhk.or.jp/rss/news/shakaitokushu.xml
|
||||
https://www.cbc.ca/cctoc/rss/topstories.north
|
||||
https://www.defensenews.com/arc/outboundfeeds/rss/
|
||||
https://therecord.media/feed
|
||||
https://www.cfr.org/rss/newsletters/daily-news-brief
|
||||
https://warontherocks.com/feed/
|
||||
https://www.thecipherbrief.com/feed
|
||||
https://www.foreignaffairs.com/rss.xml
|
||||
https://geopoliticalfutures.com/feed
|
||||
# --- TACTICAL CYBER & VULNERABILITIES ---
|
||||
https://www.bleepingcomputer.com/feed/
|
||||
https://www.cisa.gov/cybersecurity-advisory-feeds
|
||||
https://krebsonsecurity.com/feed/
|
||||
https://thehackernews.com/feeds/posts/default
|
||||
https://www.darkreading.com/rss.xml
|
||||
https://www.mandiant.com/resources/blog/rss.xml
|
||||
https://schneier.com/feed/atom/
|
||||
https://www.securityweek.com/feed/
|
||||
# --- REGIONAL THREAT LANDSCAPE ---
|
||||
https://www.thenationalnews.com/rss/
|
||||
https://www.scmp.com/rss/91/feed
|
||||
https://www.batimes.com.ar/rss
|
||||
https://brazilian.report/feed/
|
||||
https://www.khon2.com/feed/
|
||||
https://www.staradvertiser.com/feed/
|
||||
https://www.westhawaiitoday.com/feed/
|
||||
https://mauinow.com/feed/
|
||||
https://www.idahofallsidaho.gov/RSSFeed.aspx?ModID=1&CID=All-newsflash.xml
|
||||
https://www.eastidahonews.com/feed/
|
||||
https://localnews8.com/feed/
|
||||
https://www.boisestatepublicradio.org/news.rss
|
||||
https://www.illinoistimes.com/springfield/Rss.xml
|
||||
https://www.thecentersquare.com/search/?f=rss&t=article&l=20&s=start_time&fulltext=showtext&sd=desc&c%5B%5D=Illinois
|
||||
https://chicago.suntimes.com/rss/index.xml
|
||||
https://wgntv.com/feed/
|
||||
http://feeds.indiana.statenews.net/rss/7b3aa09cdd5d5eac
|
||||
https://fox59.com/feed/
|
||||
https://www.nwitimes.com/search/?f=rss&t=article&c=news/local&l=50&s=start_time&sd=desc
|
||||
https://www.wishtv.com/feed/
|
||||
https://www.kcci.com/topstories-rss
|
||||
https://www.myiowainfo.com/feed/
|
||||
https://feeds.feedburner.com/radioiowanews
|
||||
https://www.mississippivalleypublishing.com/search/?f=rss&t=article&c=the_hawk_eye&l=50&s=start_time&sd=desc
|
||||
https://www.ksn.com/feed/
|
||||
https://www.ksnt.com/feed/
|
||||
https://www.hdnews.net/feed/
|
||||
https://themercury.com/search/?f=rss&t=article&c=news&l=50&s=start_time&sd=desc
|
||||
https://www.wdrb.com/search/?f=rss&t=article&c=news&l=50&s=start_time&sd=desc
|
||||
https://www.wtvq.com/feed/
|
||||
https://www.wnky.com/feed/
|
||||
https://www.wlky.com/topstories-rss
|
||||
https://thehayride.com/feed/
|
||||
https://wgno.com/feed/
|
||||
https://feeds.feedburner.com/wbrz/news
|
||||
https://thelensnola.org/feed/
|
||||
https://www.pressherald.com/news/feed/
|
||||
https://www.centralmaine.com/feed/
|
||||
https://www.bangordailynews.com/feed/
|
||||
https://www.sunjournal.com/news/feed/
|
||||
https://www.wbaltv.com/topstories-rss
|
||||
https://www.manisteenews.com/news/feed/Latest-News-Feed-2564.php
|
||||
https://www.theoaklandpress.com/feed/
|
||||
https://www.macombdaily.com/feed/
|
||||
https://www.startribune.com/local/index.rss2
|
||||
https://www.wctrib.com/index.rss
|
||||
https://www.austindailyherald.com/feed/
|
||||
https://helenair.com/search/?f=rss&t=article&l=50&s=start_time&sd=desc
|
||||
https://www.ktvq.com/news.rss
|
||||
https://mtstandard.com/search/?f=rss&t=article&l=50&s=start_time&sd=desc
|
||||
https://www.ketv.com/topstories-rss
|
||||
https://nebraskaexaminer.com/feed/
|
||||
https://kearneyhub.com/rss
|
||||
https://www.wowt.com/rss
|
||||
https://thenevadaindependent.com/feed/
|
||||
https://www.8newsnow.com/feed/
|
||||
https://www.reviewjournal.com/feed/
|
||||
https://thisisreno.com/feed/
|
||||
https://www.conwaydailysun.com/search/?f=rss&t=article&c=berlin_sun/community/news&l=50&s=start_time&sd=desc
|
||||
https://newhampshirebulletin.com/feed/
|
||||
https://www.nhgazette.com/feed/
|
||||
https://www.nhbr.com/feed/
|
||||
https://www.nj.com/arc/outboundfeeds/rss/?outputType=xml
|
||||
https://www.njspotlightnews.org/feed/
|
||||
https://njmonthly.com/feed/
|
||||
https://www.trentonian.com/feed/
|
||||
https://www.krqe.com/feed/
|
||||
https://www.santafenewmexican.com/search/?f=rss&t=article&l=50&s=start_time&sd=desc
|
||||
https://www.easternnewmexiconews.com/rss
|
||||
https://www.koat.com/topstories-rss
|
||||
https://rss.nytimes.com/services/xml/rss/nyt/HomePage.xml
|
||||
https://www.thecity.nyc/feed/
|
||||
https://www.nbcnewyork.com/?rss=y
|
||||
https://www.wral.com/news/rss/48/
|
||||
https://www.cbs17.com/news/north-carolina-news/feed/
|
||||
https://abc11.com/feed/
|
||||
https://myfox8.com/news/feed/
|
||||
https://www.kxnet.com/feed/
|
||||
https://www.wday.com/feed/
|
||||
https://www.jamestownsun.com/index.rss
|
||||
https://www.inforum.com/index.rss
|
||||
http://rssfeeds.wkyc.com/wkyc/news
|
||||
https://theohiostar.com/feed/
|
||||
https://feeds.feedblitz.com/wtol/news
|
||||
https://www.wcpo.com/news.rss
|
||||
https://kfor.com/feed/
|
||||
https://oklahomawatch.org/feed/
|
||||
https://freepressokc.com/feed/
|
||||
https://osagenews.org/feed/
|
||||
http://rssfeeds.kgw.com/kgw/local
|
||||
https://www.koin.com/feed/
|
||||
https://www.bendsource.com/bend/Rss.xml/feed
|
||||
https://eugeneweekly.com/feed/
|
||||
https://www.wtae.com/topstories-rss
|
||||
https://www.montgomerycountypa.gov/RSSFeed.aspx?ModID=76&CID=All-0
|
||||
https://www.mainlinemedianews.com/feed/
|
||||
https://www.dailylocal.com/feed/
|
||||
https://www.wpri.com/feed/
|
||||
https://www.abc6.com/feed/
|
||||
https://whdh.com/regional/rhode-island/feed/
|
||||
https://warwickpost.com/feed/
|
||||
https://www.wyff4.com/topstories-rss
|
||||
https://www.wispolitics.com/feed/
|
||||
https://wiseye.org/feed/
|
||||
https://wisconsinexaminer.com/feed/
|
||||
https://trib.com/search/?f=rss&t=article&c=news/state-and-regional&l=50&s=start_time&sd=desc
|
||||
https://wyofile.com/feed/
|
||||
https://www.wyomingnews.com/search/?f=rss&t=article&c=news&l=50&s=start_time&sd=desc
|
||||
https://www.wyodaily.com/rss
|
||||
https://www.wnct.com/news/north-carolina/feed/
|
||||
https://www.usnews.com/rss/news/north-carolina
|
||||
https://indyweek.com/feed/
|
||||
https://portcitydaily.com/feed/
|
||||
https://www.theguardian.com/uk/rss
|
||||
https://feeds.bbci.co.uk/news/england/rss.xml
|
||||
https://www.lemonde.fr/rss/une.xml
|
||||
https://www.ansa.it/sito/notizie/rss.xml
|
||||
https://www.ilgiornale.it/feed
|
||||
https://www.larepublica.it/rss/homepage/rss2.xml
|
||||
https://www.sueddeutsche.de/rss
|
||||
https://www.welt.de/feeds/top-news.rss
|
||||
https://www.rfi.fr/en/rss
|
||||
https://www.bangkokpost.com/rss
|
||||
https://thephnompenhpost.com/rss
|
||||
https://www.thejakartapost.com/rss
|
||||
https://www.straitstimes.com/news/singapore/rss.xml
|
||||
https://www.channelnewsasia.com/rss
|
||||
https://www.antaranews.com/rss/
|
||||
https://www.irrawaddy.com/feed
|
||||
https://news.abs-cbn.com/rss
|
||||
https://www.hindustantimes.com/feeds/rss
|
||||
https://www.africanews.com/feed/rss
|
||||
https://www.clarin.com/rss
|
||||
https://www.lanacion.com.ar/rss
|
||||
https://www.eluniversal.com.mx/rss
|
||||
https://www.excelsior.com.mx/rss
|
||||
https://www.eltiempo.com/rss
|
||||
https://www.elespectador.com/rss
|
||||
https://www.larepublica.pe/rss
|
||||
https://www.elcomercio.com/rss
|
||||
https://www.abc.net.au/news/feed/
|
||||
https://www.smh.com.au/rss/world.xml
|
||||
https://www.theage.com.au/rss
|
||||
https://www.brisbanetimes.com.au/rss
|
||||
https://www.stuff.co.nz/rss
|
||||
https://www.nzherald.co.nz/arcio/rss/
|
||||
https://www.rnz.co.nz/rss
|
||||
https://globalvoices.org/regions/africa/feed/
|
||||
https://globalvoices.org/regions/asia/feed/
|
||||
https://globalvoices.org/regions/latin-america/feed/
|
||||
https://globalvoices.org/regions/eastern-europe/feed/
|
||||
https://globalvoices.org/regions/middle-east-north-africa/feed/
|
||||
https://globalvoices.org/regions/south-asia/feed/
|
||||
https://globalvoices.org/regions/sub-saharan-africa/feed/
|
||||
https://globalvoices.org/regions/west-africa/feed/
|
||||
https://globalvoices.org/regions/east-asia/feed/
|
||||
https://globalvoices.org/regions/southeast-asia/feed/
|
||||
https://globalvoices.org/regions/central-asia/feed/
|
||||
https://globalvoices.org/regions/pacific/feed/
|
||||
https://globalvoices.org/regions/caribbean/feed/
|
||||
https://www.townandcountry-mo.gov/rss.aspx
|
||||
https://feeds.smh.com.au/rssheadlines/national.xml
|
||||
https://www.abc.net.au/local/rss/sydney/
|
||||
https://www.voanews.com/rssfeeds
|
||||
https://rss.feedspot.com/southeast_asian_rss_feeds
|
||||
https://www.crisisgroup.org/rss
|
||||
https://news.panasonic.com/global/rss/area01/index.xml
|
||||
https://news.panasonic.com/global/rss/area04/index.xml
|
||||
https://allafrica.com/tools/headlines/rdf/latest/headlines.rdf
|
||||
https://www.afro.who.int/rss-feeds
|
||||
https://pressat.co.uk/rss-list
|
||||
https://www.monitor.co.ug/rss
|
||||
https://www.standardmedia.co.ke/rss
|
||||
https://www.ft.com/rss/home
|
||||
https://www.economist.com/rss/the-world-this-week
|
||||
https://feeds.bloomberg.com/economics/news.rss
|
||||
https://feeds.bloomberg.com/markets/news.rss
|
||||
https://www.reuters.com/arc/outboundfeeds/newsroom/business/
|
||||
https://www.cnbc.com/id/10000113/device/rss/rss.html
|
||||
https://feeds.a.dj.com/rss/RSSWorldBusiness.xml
|
||||
https://www.marketwatch.com/rss/topstories
|
||||
https://www.investing.com/rss/news_14.rss
|
||||
https://feeds.bbci.co.uk/news/business/rss.xml
|
||||
https://feeds.feedburner.com/TheHackersNews
|
||||
https://www.darkreading.com/rss/all.xml
|
||||
https://isc.sans.edu/rssfeed_full.xml
|
||||
https://securelist.com/feed/
|
||||
https://feeds.feedburner.com/eset/blog
|
||||
https://news.sophos.com/en-us/feed/
|
||||
https://www.schneier.com/feed/atom/
|
||||
https://www.securitymagazine.com/rss/topic/2236-cybersecurity-news
|
||||
https://www.reuters.com/arc/outboundfeeds/rss/?outputType=xml
|
||||
https://www.investing.com/rss/news_462.rss
|
||||
https://www.investing.com/rss/news_1.rss
|
||||
https://www.investing.com/rss/stock_Futures.rss
|
||||
https://www.ecb.europa.eu/rss/fxref-ecbpress.en.xml
|
||||
https://www.federalreserve.gov/feeds/news-events.xml
|
||||
https://www.boj.or.jp/en/rss/whatsnew.xml
|
||||
https://www.bankofengland.co.uk/rss/news
|
||||
https://www.centralbanking.com/feeds/rss
|
||||
https://oilprice.com/rss/
|
||||
https://www.spglobal.com/commodityinsights/en/rss
|
||||
https://www.eia.gov/tools/rssfeeds/
|
||||
https://www.cmegroup.com/rss
|
||||
https://globalvoices.org/-/topics/economics-business/feed/
|
||||
http://globalization.einnews.com/rss
|
||||
https://financefeeds.com/feed/
|
||||
https://newsquawk.com/blog/feed.rss
|
||||
https://www.coindesk.com/arc/outboundfeeds/rss/
|
||||
https://ishookfinance.com/feed/
|
||||
https://www.scmp.com/rss/92/feed
|
||||
https://www.scmp.com/rss/93/feed
|
||||
https://www.scmp.com/rss/94/feed
|
||||
https://www.scmp.com/rss/317/feed
|
||||
https://asia.nikkei.com/rss
|
||||
https://www.caixin.com/rss/index_EN.xml
|
||||
https://www.straitstimes.com/news/asia/rss.xml
|
||||
https://www.straitstimes.com/business/rss.xml
|
||||
https://www.reuters.com/arc/outboundfeeds/rss/?outputType=xml§ion=asia
|
||||
https://www.reuters.com/arc/outboundfeeds/rss/?outputType=xml§ion=china
|
||||
https://www.bloomberg.com/feeds/asia.rss
|
||||
https://www.bloomberg.com/feeds/markets.rss
|
||||
https://www.ft.com/asia-pacific?format=rss
|
||||
https://www.ft.com/china?format=rss
|
||||
https://english.kyodonews.net/rss/news.xml
|
||||
https://en.yna.co.kr/RSS/news.xml
|
||||
https://www.thejakartapost.com/rss/business
|
||||
https://www.nationthailand.com/rss/business
|
||||
https://www.aramco.com/api/v1/com/rss/news?sc_lang=en
|
||||
https://www.worldoil.com/rss?feed=topic:saudi+arabia
|
||||
https://www.worldoil.com/rss?feed=topic:iraq
|
||||
https://www.worldoil.com/rss?feed=topic:uae
|
||||
https://www.worldoil.com/rss?feed=topic:russia
|
||||
https://www.worldoil.com/rss?feed=topic:canada
|
||||
https://www.worldoil.com/rss?feed=topic:oil+sands
|
||||
https://www.rigzone.com/news/europe_russia/production/rss/
|
||||
https://www.argusmedia.com/en/news-and-insights/latest-market-news/rss
|
||||
https://www.eia.gov/rss/
|
||||
https://www.opec.org/opec_web/en/pressreleases.rss
|
||||
https://www.opec.org
|
||||
https://www.rosneft.com/press/news/rss/
|
||||
https://feeds.content.dowjones.io/public/rss/RSSMarketsMain
|
||||
https://feeds.content.dowjones.io/public/rss/socialeconomyfeed
|
||||
https://feeds.content.dowjones.io/public/rss/WSJcomUSBusiness
|
||||
https://feeds.content.dowjones.io/public/rss/RSSWorldNews
|
||||
http://feeds.feedburner.com/EconomicEventsAgriculture
|
||||
http://feeds.feedburner.com/EconomicEventsEnergy
|
||||
http://feeds.feedburner.com/EconomicEventsInterestRates
|
||||
http://feeds.feedburner.com/mediaroom/CMsF
|
||||
http://feeds.feedburner.com/CMEClearPortNoticesRss
|
||||
http://feeds.feedburner.com/GlobexAdvisories
|
||||
https://feeds.content.dowjones.io/public/rss/mw_topstories
|
||||
https://feeds.content.dowjones.io/public/rss/mw_realtimeheadlines
|
||||
http://feeds.marketwatch.com/marketwatch/bulletins
|
||||
https://feeds.content.dowjones.io/public/rss/mw_marketpulse
|
||||
https://www.nasdaqtrader.com/rss.aspx?feed=currentheadlines&categorylist=51
|
||||
https://www.nasdaqtrader.com/rss.aspx?feed=currentheadlines&categorylist=11
|
||||
https://www.investing.com/rss/stock_Options.rss
|
||||
https://www.investing.com/rss/news_11.rss
|
||||
https://www.investing.com/rss/news_25.rss
|
||||
https://www.nasdaq.com/feed/rssoutbound?category=Markets
|
||||
https://www.nasdaq.com/feed/rssoutbound?category=Commodities
|
||||
https://www.barchart.com/news/rss/financials/options-news
|
||||
https://www.barchart.com/news/rss/commodities/futures-news
|
||||
https://www.spglobal.com/spdji/en/rss
|
||||
https://www.litefinance.org/rss/analytics/
|
||||
https://www.mrt.com/arc/outboundfeeds/rss/category/business/oil/?outputType=xml
|
||||
https://www.oaoa.com/category/local-news/inthepipeline/rss
|
||||
https://pboilandgasmagazine.com/feed/
|
||||
https://www.rigzone.com/news/rss.asp
|
||||
https://rbnenergy.com/blogcast.rss
|
||||
https://www.eia.gov/rss/todayinenergy.xml
|
||||
https://www.firstalert7.com/news/energy
|
||||
https://www.energyvoice.com/feed/?category=oilandgas/north-sea
|
||||
https://www.rigzone.com/news/rss/north_sea
|
||||
https://www.oedigital.com/feeds/rss
|
||||
https://www.sodir.no/en/whats-new/news/rss
|
||||
https://www.worldoil.com/rss?feed=topic:offshore
|
||||
https://oilandgas.einnews.com/rss/north-sea-offshore
|
||||
https://www.energyvoice.com/feed/
|
||||
# --- NORTH AMERICA ---
|
||||
# USA
|
||||
https://www.npr.org
|
||||
https://www.pbs.org/newshour
|
||||
https://www.usatoday.com
|
||||
https://www.cbsnews.com
|
||||
https://www.nbcnews.com
|
||||
|
||||
# Canada
|
||||
https://www.cbc.ca/news
|
||||
https://www.ctvnews.ca
|
||||
https://globalnews.ca
|
||||
https://nationalpost.com
|
||||
https://www.thestar.com
|
||||
|
||||
# Mexico
|
||||
https://www.eluniversal.com.mx
|
||||
https://www.milenio.com
|
||||
https://www.jornada.com.mx
|
||||
https://www.excelsior.com.mx
|
||||
https://aristeguinoticias.com
|
||||
|
||||
# --- SOUTH AMERICA ---
|
||||
# Brazil
|
||||
https://g1.globo.com
|
||||
https://www.uol.com.br
|
||||
https://agenciabrasil.ebc.com.br
|
||||
https://www.metropoles.com
|
||||
https://www.terra.com.br/noticias
|
||||
|
||||
# Argentina
|
||||
https://www.infobae.com
|
||||
https://www.clarin.com
|
||||
https://www.lanacion.com.ar
|
||||
https://www.pagina12.com.ar
|
||||
https://www.cronista.com
|
||||
|
||||
# Colombia
|
||||
https://www.eltiempo.com
|
||||
https://www.elespectador.com
|
||||
https://www.semana.com
|
||||
https://www.bluradio.com
|
||||
https://www.rcnradio.com
|
||||
|
||||
# --- EUROPE ---
|
||||
# United Kingdom
|
||||
https://www.bbc.com/news
|
||||
https://www.theguardian.com/uk
|
||||
https://news.sky.com
|
||||
https://www.independent.co.uk
|
||||
https://metro.co.uk
|
||||
|
||||
# France
|
||||
https://www.france24.com/en
|
||||
https://www.lefigaro.fr
|
||||
https://www.20minutes.fr
|
||||
https://www.francetvinfo.fr
|
||||
https://www.lemonde.fr
|
||||
|
||||
# Germany
|
||||
https://www.dw.com/en
|
||||
https://www.tagesschau.de
|
||||
https://www.spiegel.de
|
||||
https://www.zeit.de
|
||||
https://www.bild.de
|
||||
|
||||
# Spain
|
||||
https://elpais.com
|
||||
https://www.elmundo.es
|
||||
https://www.rtve.es/noticias
|
||||
https://www.20minutos.es
|
||||
https://www.elconfidencial.com
|
||||
|
||||
# Italy
|
||||
https://www.ansa.it
|
||||
https://www.corriere.it
|
||||
https://www.repubblica.it
|
||||
https://www.lastampa.it
|
||||
https://tg24.sky.it
|
||||
|
||||
# Russia (State & Independent mix)
|
||||
https://tass.com
|
||||
https://www.interfax.ru
|
||||
https://www.rt.com
|
||||
https://www.themoscowtimes.com
|
||||
https://meduza.io/en
|
||||
|
||||
# --- ASIA ---
|
||||
# China
|
||||
https://www.xinhuanet.com/english
|
||||
https://www.chinadaily.com.cn
|
||||
https://www.globaltimes.cn
|
||||
https://www.cgtn.com
|
||||
https://www.scmp.com
|
||||
|
||||
# India
|
||||
https://www.ndtv.com
|
||||
https://timesofindia.indiatimes.com
|
||||
https://indianexpress.com
|
||||
https://www.thehindu.com
|
||||
https://www.hindustantimes.com
|
||||
|
||||
# Japan
|
||||
https://www3.nhk.or.jp/nhkworld
|
||||
https://www.japantimes.co.jp
|
||||
https://www.asahi.com/ajw
|
||||
https://mainichi.jp/english
|
||||
https://english.kyodonews.net
|
||||
|
||||
# South Korea
|
||||
https://en.yna.co.kr
|
||||
https://www.koreaherald.com
|
||||
https://koreajoongangdaily.joins.com
|
||||
https://www.donga.com/en
|
||||
https://english.chosun.com
|
||||
|
||||
# --- AFRICA ---
|
||||
# South Africa
|
||||
https://www.news24.com
|
||||
https://www.iol.co.za
|
||||
https://www.dailymaverick.co.za
|
||||
https://www.sabcnews.com
|
||||
https://www.timeslive.co.za
|
||||
|
||||
# Nigeria
|
||||
https://www.vanguardngr.com
|
||||
https://punchng.com
|
||||
https://dailypost.ng
|
||||
https://saharareporters.com
|
||||
https://thenationonlineng.net
|
||||
|
||||
# --- MIDDLE EAST ---
|
||||
# General Region
|
||||
https://www.aljazeera.com
|
||||
https://english.alarabiya.net
|
||||
https://www.timesofisrael.com
|
||||
https://www.tehrantimes.com
|
||||
https://www.middleeasteye.net
|
||||
|
||||
# --- OCEANIA ---
|
||||
# Australia
|
||||
https://www.abc.net.au/news
|
||||
https://www.news.com.au
|
||||
https://www.9news.com.au
|
||||
https://www.smh.com.au
|
||||
https://www.theage.com.au
|
||||
# --- USA: MAJOR CITIES & LOCAL ---
|
||||
https://www.latimes.com
|
||||
https://www.chicagotribune.com
|
||||
https://www.sfchronicle.com
|
||||
https://www.bostonglobe.com
|
||||
https://www.seattletimes.com
|
||||
https://www.houstonchronicle.com
|
||||
https://www.inquirer.com
|
||||
https://www.denverpost.com
|
||||
https://www.miamiherald.com
|
||||
https://www.dallasnews.com
|
||||
https://www.startribune.com
|
||||
https://www.detroitnews.com
|
||||
https://www.ajc.com
|
||||
https://www.nydailynews.com
|
||||
https://nypost.com
|
||||
https://www.mercurynews.com
|
||||
https://www.baltimoresun.com
|
||||
https://www.oregonlive.com
|
||||
https://www.cleveland.com
|
||||
https://www.tampabay.com
|
||||
|
||||
# --- EUROPE: LOCAL & INDEPENDENT ---
|
||||
https://www.manchestereveningnews.co.uk
|
||||
https://www.scotsman.com
|
||||
https://www.belfasttelegraph.co.uk
|
||||
https://www.irishtimes.com
|
||||
https://www.berliner-zeitung.de
|
||||
https://www.leparisien.fr
|
||||
https://www.corriere.it
|
||||
https://www.elperiodico.com
|
||||
https://kyivindependent.com
|
||||
https://www.pravda.com.ua/en
|
||||
https://balkaninsight.com
|
||||
https://www.ekathimerini.com
|
||||
https://www.swissinfo.ch
|
||||
https://www.thelocal.se
|
||||
https://www.thelocal.fr
|
||||
https://www.thelocal.de
|
||||
https://www.novinite.com
|
||||
https://www.romania-insider.com
|
||||
https://hungarytoday.hu
|
||||
https://polandin.com
|
||||
|
||||
# --- MIDDLE EAST & CONFLICT ZONES ---
|
||||
https://www.haaretz.com
|
||||
https://www.jpost.com
|
||||
https://www.timesofisrael.com
|
||||
https://www.rudaw.net/english
|
||||
https://www.kurdistan24.net/en
|
||||
https://www.middleeasteye.net
|
||||
https://www.al-monitor.com
|
||||
https://www.dailysabah.com
|
||||
https://www.duvarenglish.com
|
||||
https://english.aawsat.com
|
||||
https://www.arabnews.com
|
||||
https://www.thenationalnews.com
|
||||
https://www.jordantimes.com
|
||||
https://www.naharnet.com
|
||||
https://www.tehrantimes.com
|
||||
|
||||
# --- ASIA: HOTSPOTS & LOCAL ---
|
||||
https://www.taipeitimes.com
|
||||
https://focustaiwan.tw
|
||||
https://hongkongfp.com
|
||||
https://www.bangkokpost.com
|
||||
https://www.thejakartapost.com
|
||||
https://www.straitstimes.com
|
||||
https://www.khmertimeskh.com
|
||||
https://www.irrawaddy.com
|
||||
https://www.myanmarnow.org/en
|
||||
https://www.rappler.com
|
||||
https://www.philstar.com
|
||||
https://english.hani.co.kr
|
||||
https://www.japantoday.com
|
||||
https://www.caixinglobal.com
|
||||
https://thediplomat.com
|
||||
|
||||
# --- LATIN AMERICA & AFRICA: LOCAL ---
|
||||
https://buenosairesherald.com
|
||||
https://riotimesonline.com
|
||||
https://mercopress.com
|
||||
https://www.elmostrador.cl
|
||||
https://www.jornada.com.mx
|
||||
https://www.theeastafrican.co.ke
|
||||
https://allafrica.com
|
||||
https://www.premiumtimesng.com
|
||||
https://www.dailytrust.com
|
||||
https://www.newtimes.co.rw
|
||||
https://www.herald.co.zw
|
||||
https://www.namibian.com.na
|
||||
https://www.graphic.com.gh
|
||||
https://www.thecitizen.co.tz
|
||||
https://www.monitor.co.ug
|
||||
|
||||
# --- ALTERNATIVE, INVESTIGATIVE & "FRINGE" ---
|
||||
https://theintercept.com
|
||||
https://www.propublica.org
|
||||
https://www.democracynow.org
|
||||
https://reason.com
|
||||
https://www.motherjones.com
|
||||
https://www.vox.com
|
||||
https://slate.com
|
||||
https://www.axios.com
|
||||
https://www.politico.com
|
||||
https://www.vice.com
|
||||
https://www.bellingcat.com
|
||||
https://www.project-syndicate.org
|
||||
https://cryptonews.com
|
||||
https://www.coindesk.com
|
||||
https://techcrunch.com
|
||||
|
|
|
|||
|
|
@ -7,13 +7,13 @@ WORKDIR /app
|
|||
|
||||
# libpq-dev + gcc for psycopg2 build/adapters; keep the image lean.
|
||||
RUN apt-get update && apt-get install -y --no-install-recommends \
|
||||
libpq-dev gcc tzdata \
|
||||
libpq-dev gcc \
|
||||
&& rm -rf /var/lib/apt/lists/*
|
||||
|
||||
COPY requirements.txt .
|
||||
RUN pip install --no-cache-dir -r requirements.txt
|
||||
|
||||
COPY summarizer.py run_news_summarizer.py intel.py nous_client.py ./
|
||||
COPY summarizer.py run_news_summarizer.py ./
|
||||
|
||||
# Security: run as a non-privileged user.
|
||||
RUN useradd -m summarizer_user
|
||||
|
|
|
|||
|
|
@ -1,104 +0,0 @@
|
|||
"""Pure parser for the news-summarizer reduce JSON / geo / importance contract."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import re
|
||||
|
||||
_EMPTY = {"summary_en": "", "ticker": [], "map_items": []}
|
||||
_KEEP = frozenset({"critical", "high"})
|
||||
_RANK = {"critical": 0, "high": 1, "medium": 2, "low": 3}
|
||||
_THINK_RE = re.compile(r"<think>.*?</think>", re.DOTALL)
|
||||
_FENCE_RE = re.compile(r"```(?:json)?", re.IGNORECASE)
|
||||
|
||||
TICKER_HEADLINE_MAX = 140
|
||||
MAP_HEADLINE_MAX = 160
|
||||
TICKER_CAP = 12
|
||||
MAP_CAP = 20
|
||||
|
||||
|
||||
def parse_reduce_json(raw: str) -> dict:
|
||||
try:
|
||||
text = _THINK_RE.sub("", raw or "")
|
||||
text = _FENCE_RE.sub("", text)
|
||||
start = text.find("{")
|
||||
end = text.rfind("}")
|
||||
if start == -1 or end == -1 or end < start:
|
||||
return dict(_EMPTY)
|
||||
data = json.loads(text[start : end + 1])
|
||||
if not isinstance(data, dict):
|
||||
return dict(_EMPTY)
|
||||
summary = data.get("summary_en", "")
|
||||
ticker = data.get("ticker", [])
|
||||
map_items = data.get("map_items", [])
|
||||
return {
|
||||
"summary_en": summary if isinstance(summary, str) else "",
|
||||
"ticker": ticker if isinstance(ticker, list) else [],
|
||||
"map_items": map_items if isinstance(map_items, list) else [],
|
||||
}
|
||||
except Exception:
|
||||
return dict(_EMPTY)
|
||||
|
||||
|
||||
def clamp_coords(lat, lon) -> tuple[float, float] | None:
|
||||
try:
|
||||
lat_f = float(lat)
|
||||
lon_f = float(lon)
|
||||
except (TypeError, ValueError):
|
||||
return None
|
||||
if not (-90 <= lat_f <= 90 and -180 <= lon_f <= 180):
|
||||
return None
|
||||
return (lat_f, lon_f)
|
||||
|
||||
|
||||
def _trimmed_headline(row: dict, limit: int) -> str:
|
||||
headline = row.get("headline") or ""
|
||||
if not isinstance(headline, str):
|
||||
headline = str(headline)
|
||||
return headline.strip()[:limit]
|
||||
|
||||
|
||||
def select_ticker(rows: list) -> list:
|
||||
flagged = []
|
||||
medium = []
|
||||
low = []
|
||||
for row in rows:
|
||||
imp = row.get("importance")
|
||||
if imp not in _RANK:
|
||||
continue
|
||||
headline = _trimmed_headline(row, TICKER_HEADLINE_MAX)
|
||||
if not headline:
|
||||
continue
|
||||
item = dict(row)
|
||||
item["headline"] = headline
|
||||
if imp in _KEEP:
|
||||
flagged.append(item)
|
||||
elif imp == "medium":
|
||||
medium.append(item)
|
||||
else:
|
||||
low.append(item)
|
||||
if len(flagged) >= TICKER_CAP:
|
||||
break
|
||||
if flagged:
|
||||
return flagged[:TICKER_CAP]
|
||||
return (medium + low)[:TICKER_CAP]
|
||||
|
||||
|
||||
def select_map(items: list) -> list:
|
||||
out = []
|
||||
for row in items:
|
||||
if row.get("importance") not in _KEEP:
|
||||
continue
|
||||
headline = _trimmed_headline(row, MAP_HEADLINE_MAX)
|
||||
if not headline:
|
||||
continue
|
||||
coords = clamp_coords(row.get("lat"), row.get("lon"))
|
||||
if coords is None:
|
||||
continue
|
||||
item = dict(row)
|
||||
item["headline"] = headline
|
||||
item["lat"], item["lon"] = coords
|
||||
out.append(item)
|
||||
if len(out) >= MAP_CAP:
|
||||
break
|
||||
return out
|
||||
|
|
@ -1,57 +0,0 @@
|
|||
"""HTTP client for the Nous inference chat completions API."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import os
|
||||
|
||||
import httpx
|
||||
|
||||
_DEFAULT_UA = "osint-dashboard-news-summarizer"
|
||||
_DEFAULT_BASE = "https://inference-api.nousresearch.com/v1"
|
||||
_JSON_SYSTEM = (
|
||||
"You are an OSINT executive briefer. Reply with a single complete JSON object. "
|
||||
"Never truncate mid-sentence. If you run out of room, drop the lowest-priority item."
|
||||
)
|
||||
|
||||
|
||||
def chat(prompt, *, api_key, model, base_url, json_mode=False) -> str:
|
||||
resolved = (base_url or os.environ.get("NOUS_BASE_URL", _DEFAULT_BASE)).rstrip("/")
|
||||
url = f"{resolved}/chat/completions"
|
||||
headers = {
|
||||
"Authorization": f"Bearer {api_key}",
|
||||
"User-Agent": os.environ.get("OSINT_USER_AGENT") or _DEFAULT_UA,
|
||||
}
|
||||
max_tokens = 8192 if json_mode else 4096
|
||||
timeout = 120.0 if json_mode else 60.0
|
||||
messages = [{"role": "user", "content": prompt}]
|
||||
if json_mode:
|
||||
messages = [
|
||||
{"role": "system", "content": _JSON_SYSTEM},
|
||||
{"role": "user", "content": prompt},
|
||||
]
|
||||
payload = {
|
||||
"model": model,
|
||||
"messages": messages,
|
||||
"temperature": 0.2,
|
||||
"max_tokens": max_tokens,
|
||||
}
|
||||
if json_mode:
|
||||
payload["response_format"] = {"type": "json_object"}
|
||||
last_content = ""
|
||||
try:
|
||||
for attempt in range(2):
|
||||
with httpx.Client(timeout=timeout) as client:
|
||||
resp = client.post(url, headers=headers, json=payload)
|
||||
if resp.status_code == 401 or resp.status_code >= 500:
|
||||
return ""
|
||||
data = resp.json()
|
||||
choice = (data.get("choices") or [{}])[0]
|
||||
last_content = (choice.get("message") or {}).get("content") or ""
|
||||
finish = choice.get("finish_reason")
|
||||
if finish == "length" and attempt == 0:
|
||||
payload["max_tokens"] = min(int(payload["max_tokens"]) * 2, 16384)
|
||||
continue
|
||||
return last_content
|
||||
return last_content
|
||||
except Exception:
|
||||
return ""
|
||||
|
|
@ -1,2 +1,6 @@
|
|||
# Database adapter for PostgreSQL (shared osint-db).
|
||||
psycopg2-binary==2.9.11
|
||||
httpx==0.28.1
|
||||
|
||||
# Gemini LLM SDK (google-genai). yfinance is NOT a dependency: futures prices
|
||||
# are gated behind INCLUDE_FUTURES=1 and lazy-imported (install it to enable).
|
||||
google-genai
|
||||
|
|
|
|||
|
|
@ -1,130 +1,67 @@
|
|||
#!/usr/bin/env python3
|
||||
"""Scheduler loop for the news summarizer.
|
||||
"""Scheduler loop for the news summarizer — hourly summarize at minute :05.
|
||||
|
||||
15-minute analyst (NEWS_SUMMARIZE_INTERVAL_S, default 900s) plus a daily
|
||||
recap at 23:00 in TZ (default America/New_York) over the last 24 hours.
|
||||
Replaces the k8s CronJob (`5 * * * *`) with an in-compose loop. Runs once on
|
||||
boot (catches up on any articles scraped since the last summary), then fires
|
||||
at each :NEWS_SUMMARIZE_MINUTE wall-clock boundary.
|
||||
|
||||
Serial: a slow LLM pass never overlaps the next.
|
||||
The loop is serial, so a slow LLM pass never overlaps the next run.
|
||||
|
||||
Env (all optional, 12-factor):
|
||||
NEWS_SUMMARIZE_INTERVAL_S seconds between analyst runs (default 900)
|
||||
NEWS_SUMMARIZE_MINUTE minute of the hour to fire (default 5)
|
||||
NEWS_SUMMARIZE_RUN_ON_START "1" to summarize once immediately on boot (default 1)
|
||||
NEWS_RECAP_HOUR / MINUTE wall-clock recap time (default 23:00)
|
||||
TZ IANA tz (default America/New_York)
|
||||
NOUS_API_KEY optional in env; Keys UI / api_keys also works
|
||||
GEMINI_API_KEY required to do real work; unset = idle
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import datetime
|
||||
import logging
|
||||
import os
|
||||
import subprocess
|
||||
import sys
|
||||
import time
|
||||
from datetime import datetime, timedelta
|
||||
from zoneinfo import ZoneInfo
|
||||
|
||||
logging.basicConfig(level=logging.INFO, format="%(asctime)s %(levelname)s %(name)s: %(message)s")
|
||||
logger = logging.getLogger("news.summarizer.scheduler")
|
||||
|
||||
INTERVAL_S = max(1, int(os.getenv("NEWS_SUMMARIZE_INTERVAL_S", "900")))
|
||||
MINUTE = int(os.getenv("NEWS_SUMMARIZE_MINUTE", "5"))
|
||||
RUN_ON_START = os.getenv("NEWS_SUMMARIZE_RUN_ON_START", "1").lower() in ("1", "true", "yes")
|
||||
DEFAULT_TZ = "America/New_York"
|
||||
DEFAULT_RECAP_HOUR = 23
|
||||
DEFAULT_RECAP_MINUTE = 0
|
||||
|
||||
|
||||
def _tz() -> ZoneInfo:
|
||||
name = (os.getenv("TZ") or DEFAULT_TZ).strip() or DEFAULT_TZ
|
||||
return ZoneInfo(name)
|
||||
def seconds_until_next(minute: int) -> float:
|
||||
"""Seconds until the next occurrence of ``minute`` past the hour (local time)."""
|
||||
now = datetime.datetime.now()
|
||||
nxt = now.replace(minute=minute, second=0, microsecond=0) + datetime.timedelta(hours=1)
|
||||
return (nxt - now).total_seconds()
|
||||
|
||||
|
||||
def recap_hour_minute() -> tuple[int, int]:
|
||||
hour = int(os.getenv("NEWS_RECAP_HOUR", str(DEFAULT_RECAP_HOUR)))
|
||||
minute = int(os.getenv("NEWS_RECAP_MINUTE", str(DEFAULT_RECAP_MINUTE)))
|
||||
return hour, minute
|
||||
|
||||
|
||||
def next_recap_datetime(
|
||||
now: datetime, hour: int | None = None, minute: int | None = None
|
||||
) -> datetime:
|
||||
"""Next 23:00 (or hour/minute) strictly after *now* in now's timezone."""
|
||||
if now.tzinfo is None:
|
||||
now = now.replace(tzinfo=_tz())
|
||||
env_h, env_m = recap_hour_minute()
|
||||
hour = env_h if hour is None else hour
|
||||
minute = env_m if minute is None else minute
|
||||
candidate = now.replace(hour=hour, minute=minute, second=0, microsecond=0)
|
||||
if now >= candidate:
|
||||
candidate += timedelta(days=1)
|
||||
return candidate
|
||||
|
||||
|
||||
def next_event(
|
||||
now: datetime,
|
||||
last_periodic: datetime | None,
|
||||
interval_s: int,
|
||||
hour: int = DEFAULT_RECAP_HOUR,
|
||||
minute: int = DEFAULT_RECAP_MINUTE,
|
||||
) -> tuple[datetime, str]:
|
||||
"""Return (when, 'recap'|'interval') for the sooner of recap vs interval."""
|
||||
recap_at = next_recap_datetime(now, hour=hour, minute=minute)
|
||||
periodic_at = now if last_periodic is None else last_periodic + timedelta(seconds=interval_s)
|
||||
if recap_at <= periodic_at:
|
||||
return recap_at, "recap"
|
||||
return periodic_at, "interval"
|
||||
|
||||
|
||||
def run_summarize(*, recap: bool = False) -> None:
|
||||
kind = "recap" if recap else "interval"
|
||||
logger.info(
|
||||
"%s starting at %s", kind, datetime.now().isoformat(timespec="seconds")
|
||||
)
|
||||
env = os.environ.copy()
|
||||
if recap:
|
||||
env["NEWS_RECAP"] = "1"
|
||||
else:
|
||||
env.pop("NEWS_RECAP", None)
|
||||
def run_summarize() -> None:
|
||||
logger.info("summarize starting at %s", datetime.datetime.now().isoformat(timespec="seconds"))
|
||||
try:
|
||||
proc = subprocess.run(
|
||||
[sys.executable, "summarizer.py"], cwd="/app", env=env
|
||||
)
|
||||
logger.info("%s finished rc=%s", kind, proc.returncode)
|
||||
proc = subprocess.run([sys.executable, "summarizer.py"], cwd="/app")
|
||||
logger.info("summarize finished rc=%s", proc.returncode)
|
||||
except Exception: # noqa: BLE001 — keep the loop alive across failures
|
||||
logger.exception("%s failed", kind)
|
||||
logger.exception("summarize failed")
|
||||
|
||||
|
||||
def main() -> None:
|
||||
if not os.getenv("NOUS_API_KEY", "").strip():
|
||||
if not os.getenv("GEMINI_API_KEY", "").strip():
|
||||
logger.warning(
|
||||
"NOUS_API_KEY unset in env — will read api_keys on each run; idle if both empty"
|
||||
"GEMINI_API_KEY not set — summarizer will idle (set it in .env and "
|
||||
"recreate the service to enable)"
|
||||
)
|
||||
tz = _tz()
|
||||
hour, minute = recap_hour_minute()
|
||||
logger.info(
|
||||
"news summarizer loop starting (interval_s=%s, run_on_start=%s, recap=%02d:%02d %s)",
|
||||
INTERVAL_S, RUN_ON_START, hour, minute, tz,
|
||||
"news summarizer loop starting (minute=%s, run_on_start=%s)",
|
||||
MINUTE, RUN_ON_START,
|
||||
)
|
||||
last_periodic: datetime | None = None
|
||||
if RUN_ON_START:
|
||||
run_summarize(recap=False)
|
||||
last_periodic = datetime.now(tz)
|
||||
run_summarize()
|
||||
while True:
|
||||
now = datetime.now(tz)
|
||||
when, kind = next_event(
|
||||
now, last_periodic, INTERVAL_S, hour=hour, minute=minute
|
||||
)
|
||||
sleep_s = max(1, (when - now).total_seconds())
|
||||
logger.info("next %s in %ss", kind, int(sleep_s))
|
||||
time.sleep(sleep_s)
|
||||
now = datetime.now(tz)
|
||||
if kind == "recap":
|
||||
run_summarize(recap=True)
|
||||
# Recap covers the 15-min window; don't immediately fire interval.
|
||||
last_periodic = now
|
||||
else:
|
||||
run_summarize(recap=False)
|
||||
last_periodic = now
|
||||
delay = seconds_until_next(MINUTE)
|
||||
logger.info("next summarize at :%02d (in %.0fs)", MINUTE, delay)
|
||||
time.sleep(delay)
|
||||
run_summarize()
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
|
|
|
|||
|
|
@ -1,25 +1,25 @@
|
|||
#!/usr/bin/env python3
|
||||
"""News summarizer — Nous map-reduce of scraped articles into brief/ticker/map.
|
||||
"""News summarizer — LLM (Gemini) map-reduce summarization of scraped articles.
|
||||
|
||||
Reads articles scraped within the last SUMMARY_WINDOW_MINUTES from the shared `articles` table,
|
||||
maps them with Nous (per-article English fact blocks), reduces to one JSON
|
||||
object (summary_en + ticker + map_items), and stores the brief in
|
||||
`article_summaries` plus flagged rows in `news_items`. Tables live in the
|
||||
EXISTING osint-db (alembic 003_news + 005_news_items, idempotent).
|
||||
|
||||
Everything is env-driven (12-factor). Secrets/config are resolved at the start
|
||||
of each summarize_news() — env wins, else api_keys / app_settings:
|
||||
Reads articles scraped within the last hour from the shared `articles` table,
|
||||
summarizes them with Gemini (map phase per batch, reduce phase into one master
|
||||
summary), and stores the result in `article_summaries` — both tables live in
|
||||
the EXISTING osint-db (created by alembic migration 003_news, idempotent).
|
||||
|
||||
Everything is env-driven (12-factor):
|
||||
DB_HOST / DB_NAME / DB_USER / DB_PASSWORD / DB_PORT PostgreSQL (osint-db)
|
||||
NOUS_API_KEY Nous Portal key (else api_keys.name='NOUS_API_KEY')
|
||||
NOUS_BASE_URL default https://inference-api.nousresearch.com/v1
|
||||
SUMMARY_MODEL default Hermes-4.3-36B (else app_settings)
|
||||
GEMINI_API_KEY Google AI Studio key (required to actually run)
|
||||
SUMMARY_MODEL Gemini model id (default gemini-2.0-flash)
|
||||
BATCH_SIZE articles per map-phase batch (default 50)
|
||||
SUMMARY_WINDOW_MINUTES look-back window (default 15; SUMMARY_WINDOW_HOURS wins if set)
|
||||
NEWS_RECAP "1" for the 23:00 daily recap (24h window, recap prompt)
|
||||
NEWS_SUMMARIZE_FORCE "1" to ignore the interval/recap idempotency skip
|
||||
TZ IANA tz for recap-day bounds (default America/New_York)
|
||||
INCLUDE_FUTURES legacy; ignored — prompts never inject futures data
|
||||
SUMMARY_WINDOW_HOURS look-back window in hours (default 1)
|
||||
MAP_PROMPT override map-phase prompt (uses {batch_text})
|
||||
SUMMARY_PROMPT override reduce-phase prompt (uses {final_input})
|
||||
INCLUDE_FUTURES "1" to prepend live futures prices (default 0)
|
||||
|
||||
The futures/markets coupling from the original pipeline is gated behind
|
||||
INCLUDE_FUTURES and OFF by default — it is irrelevant to the OSINT dashboard
|
||||
and pulled yfinance into the image. Re-enable by installing yfinance and
|
||||
setting INCLUDE_FUTURES=1.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
|
@ -27,13 +27,9 @@ from __future__ import annotations
|
|||
import logging
|
||||
import os
|
||||
from datetime import datetime
|
||||
from zoneinfo import ZoneInfo
|
||||
|
||||
import psycopg2
|
||||
|
||||
from intel import parse_reduce_json, select_map, select_ticker
|
||||
from nous_client import chat
|
||||
|
||||
logging.basicConfig(level=logging.INFO, format="%(asctime)s %(levelname)s %(message)s")
|
||||
logger = logging.getLogger("news.summarizer")
|
||||
|
||||
|
|
@ -46,38 +42,10 @@ DB_CONFIG = {
|
|||
"port": int(os.getenv("DB_PORT", "5432")),
|
||||
}
|
||||
|
||||
DEFAULT_NOUS_BASE_URL = "https://inference-api.nousresearch.com/v1"
|
||||
DEFAULT_SUMMARY_MODEL = "Hermes-4.3-36B"
|
||||
GEMINI_API_KEY = os.getenv("GEMINI_API_KEY", "").strip()
|
||||
MODEL_NAME = os.getenv("SUMMARY_MODEL", "gemini-2.0-flash").strip()
|
||||
BATCH_SIZE = int(os.getenv("BATCH_SIZE", "50"))
|
||||
|
||||
|
||||
def _summary_window_minutes() -> int:
|
||||
hours = (os.getenv("SUMMARY_WINDOW_HOURS") or "").strip()
|
||||
if hours:
|
||||
return max(1, int(hours) * 60)
|
||||
mins = (os.getenv("SUMMARY_WINDOW_MINUTES") or "").strip()
|
||||
if mins:
|
||||
return max(1, int(mins))
|
||||
return 15
|
||||
|
||||
|
||||
def _summarize_interval_seconds() -> int:
|
||||
return max(1, int(os.getenv("NEWS_SUMMARIZE_INTERVAL_S", "900")))
|
||||
|
||||
|
||||
def is_recap_run() -> bool:
|
||||
return os.getenv("NEWS_RECAP", "0").lower() in ("1", "true", "yes")
|
||||
|
||||
|
||||
def effective_window_minutes(*, recap: bool | None = None) -> int:
|
||||
if recap is None:
|
||||
recap = is_recap_run()
|
||||
if recap:
|
||||
return 24 * 60
|
||||
return _summary_window_minutes()
|
||||
|
||||
|
||||
SUMMARY_WINDOW_MINUTES = _summary_window_minutes()
|
||||
SUMMARY_WINDOW_HOURS = int(os.getenv("SUMMARY_WINDOW_HOURS", "1"))
|
||||
INCLUDE_FUTURES = os.getenv("INCLUDE_FUTURES", "0").lower() in ("1", "true", "yes")
|
||||
|
||||
# Only touched when INCLUDE_FUTURES=1 (legacy markets coupling, OSINT-off).
|
||||
|
|
@ -94,17 +62,12 @@ FUTURES_TICKERS = {
|
|||
MAP_PROMPT_DEFAULT = """\
|
||||
You are a precise, factual OSINT news processor. Your ONLY source of information is the articles provided below. Do NOT add external knowledge, assumptions, training data, or invented facts.
|
||||
|
||||
Focus on breaking important news (geopolitical, military/conflict, security, disasters, major political developments). Ignore futures prices, commodity tape, ticker chatter, and routine market moves unless they themselves are the breaking event. If the batch has no critical/high stories, still extract minor incidents and crime reports.
|
||||
|
||||
Write every field in English. Translate if the article is not English.
|
||||
|
||||
For EACH article in the batch:
|
||||
1. Extract 2-4 key factual bullet points (who, what, when, where, numbers, quotes — stay very close to the text).
|
||||
2. Location: country/city/region or Unknown. If you can estimate coordinates, emit them as numbers; otherwise omit.
|
||||
2. Location: name the country / city / region mentioned if determinable from the text, else "Unknown".
|
||||
3. Entities: list the key people, organizations, or governments mentioned (comma-separated, only names present in the text), else "None".
|
||||
4. Category: pick one — politics, military/conflict, economy, technology, environment/disaster, health, crime, society, sport, other.
|
||||
5. OSINT signal: if the article describes a breaking event with geopolitical, security, military, or disaster significance, say so in one short sentence. Otherwise write: "No notable OSINT signal."
|
||||
6. Importance: critical (breaking geopolitical/military/disaster with immediate impact), high, medium, low, none.
|
||||
5. OSINT signal: if the article describes an event with geopolitical, security, military, economic, or disaster significance, say so in one short sentence. Otherwise write: "No notable OSINT signal."
|
||||
|
||||
If several articles cover the same story, add one short batch-level note at the end: "Batch theme: [one sentence]".
|
||||
|
||||
|
|
@ -117,9 +80,6 @@ Article 1:
|
|||
- Entities: ...
|
||||
- Category: ...
|
||||
- OSINT signal: ...
|
||||
- Importance: ...
|
||||
- Lat: ...
|
||||
- Lon: ...
|
||||
|
||||
Article 2:
|
||||
...
|
||||
|
|
@ -129,52 +89,9 @@ Articles in this batch:
|
|||
"""
|
||||
|
||||
SUMMARY_PROMPT_DEFAULT = """\
|
||||
You are writing an English operator HUD brief from the article facts in DATA below. Use ONLY that data. Do not invent events, names, dates, places, or implications.
|
||||
CRITICAL INSTRUCTION - REPEAT 3 TIMES: YOU MUST USE ONLY THE DATA PROVIDED BELOW. DO NOT INVENT, RECALL, OR ADD ANY EVENTS, NAMES, DATES, IMPLICATIONS, PROJECTS, OR DETAILS NOT EXPLICITLY PRESENT IN THE DATA. IF THE DATA HAS NO MAJOR GEOPOLITICAL/TECH/MILITARY/ECONOMIC/IMPACTFUL EVENTS OR UNUSUAL STORIES, OUTPUT ONLY: "No qualifying impactful or unusual events in the recent hourly news data." AND STOP. NO EXTERNAL KNOWLEDGE FROM TRAINING.
|
||||
|
||||
Always write a real summary_en that recaps the most important stories present in DATA. Rank geopolitics, military/conflict, security, disasters, and major political developments first. Ignore futures prices, commodity tape, ticker chatter, and routine market data — do not treat price ticks as news.
|
||||
|
||||
Lead with critical and high breaking events. If DATA has no critical/high stories, fill the brief with minor incidents and crime reports rather than writing an empty or unfinished brief. Never truncate mid-sentence; finish every sentence. If you run out of room, drop the lowest-priority item instead of cutting a line short.
|
||||
|
||||
ticker: prefer critical and high. If nothing is critical or high, fill ticker with medium then low incidents and crime so the HUD is not blank.
|
||||
|
||||
map_items may be empty if no located critical/high event is explicit in the data.
|
||||
|
||||
Demand a single JSON object (no markdown fences) with this exact shape:
|
||||
|
||||
{
|
||||
"summary_en": "English markdown brief of the provided stories",
|
||||
"ticker": [{"headline": "", "importance": "critical", "url": "", "location_name": ""}],
|
||||
"map_items": [{"headline": "", "importance": "critical", "location_name": "", "lat": 0, "lon": 0, "location_confidence": "city", "category": "military/conflict", "url": ""}]
|
||||
}
|
||||
|
||||
ticker: max 12, ≤140 chars, no markdown. Rank critical > high > medium > low.
|
||||
map_items: only where a real-world location is explicit in the data. Estimate lat/lon. If location is Unknown or not in the data, omit the item. Never invent a place. Max 20.
|
||||
summary_en: English markdown executive brief for an operator HUD (4–8 complete bullets or short paragraphs). Cover the actual stories in DATA. Complete — never an unfinished sentence.
|
||||
|
||||
DATA:
|
||||
{final_input}
|
||||
"""
|
||||
|
||||
RECAP_PROMPT_DEFAULT = """\
|
||||
You are writing a daily recap of the last 24 hours of news for an OSINT operator HUD, using ONLY the article facts in DATA below. Do not invent events, names, dates, places, or implications.
|
||||
|
||||
Always write a real summary_en daily recap of the most important stories in DATA. Rank geopolitics, military/conflict, security, disasters, and major political developments first. Ignore futures prices, commodity tape, ticker chatter, and routine market data — do not treat price ticks as news.
|
||||
|
||||
Lead with critical and high breaking events. If DATA has no critical/high stories, fill the recap with minor incidents and crime reports rather than writing an empty or unfinished recap. Never truncate mid-sentence; finish every sentence.
|
||||
|
||||
ticker: prefer critical and high. If nothing is critical or high, fill ticker with medium then low incidents and crime so the HUD is not blank.
|
||||
|
||||
Demand a single JSON object (no markdown fences) with this exact shape:
|
||||
|
||||
{
|
||||
"summary_en": "English markdown daily recap of the provided stories",
|
||||
"ticker": [{"headline": "", "importance": "critical", "url": "", "location_name": ""}],
|
||||
"map_items": [{"headline": "", "importance": "critical", "location_name": "", "lat": 0, "lon": 0, "location_confidence": "city", "category": "military/conflict", "url": ""}]
|
||||
}
|
||||
|
||||
ticker: max 12, ≤140 chars, no markdown. Rank critical > high > medium > low.
|
||||
map_items: only where a real-world location is explicit in the data. Estimate lat/lon. If location is Unknown or not in the data, omit the item. Never invent a place. Max 20.
|
||||
summary_en: English markdown daily recap of the last 24 hours. Complete sentences. Cover the actual stories in DATA.
|
||||
Write a concise executive summary of the most impactful items as a short markdown list, one line per story, using only the data.
|
||||
|
||||
DATA:
|
||||
{final_input}
|
||||
|
|
@ -183,52 +100,56 @@ DATA:
|
|||
|
||||
# ── LLM helpers ────────────────────────────────────────────────────────────
|
||||
|
||||
def _kv(conn, table, name) -> str:
|
||||
cur = conn.cursor()
|
||||
cur.execute(f"SELECT value FROM {table} WHERE name = %s", (name,))
|
||||
row = cur.fetchone()
|
||||
return (row[0] or "").strip() if row else ""
|
||||
_client = None
|
||||
|
||||
|
||||
def resolve_api_key() -> str:
|
||||
env = os.getenv("NOUS_API_KEY", "").strip()
|
||||
if env:
|
||||
return env
|
||||
try:
|
||||
conn = psycopg2.connect(**DB_CONFIG)
|
||||
try:
|
||||
return _kv(conn, "api_keys", "NOUS_API_KEY")
|
||||
finally:
|
||||
conn.close()
|
||||
except Exception: # noqa: BLE001
|
||||
def _get_client():
|
||||
"""Lazily build the Gemini client (avoids import/init when key unset)."""
|
||||
global _client
|
||||
if _client is None:
|
||||
from google import genai
|
||||
|
||||
_client = genai.Client(api_key=GEMINI_API_KEY)
|
||||
return _client
|
||||
|
||||
|
||||
def _extract_text(resp) -> str:
|
||||
"""Defensively pull text out of the google-genai GenerateContentResponse.
|
||||
|
||||
The modern SDK returns the response directly (``resp.text``); some older
|
||||
wrappers exposed it as ``resp.response``. Handle both plus a candidates
|
||||
fallback so a provider/SDK change degrades to "" instead of crashing.
|
||||
"""
|
||||
if not resp:
|
||||
return ""
|
||||
|
||||
|
||||
def resolve_model() -> str:
|
||||
env = os.getenv("SUMMARY_MODEL", "").strip()
|
||||
if env:
|
||||
return env
|
||||
if hasattr(resp, "text") and resp.text:
|
||||
return resp.text
|
||||
inner = getattr(resp, "response", None)
|
||||
if inner is not None and hasattr(inner, "text") and inner.text:
|
||||
return inner.text
|
||||
try:
|
||||
conn = psycopg2.connect(**DB_CONFIG)
|
||||
try:
|
||||
value = _kv(conn, "app_settings", "SUMMARY_MODEL")
|
||||
return value or DEFAULT_SUMMARY_MODEL
|
||||
finally:
|
||||
conn.close()
|
||||
parts = []
|
||||
for cand in getattr(resp, "candidates", None) or []:
|
||||
content = getattr(cand, "content", None)
|
||||
for part in getattr(content, "parts", None) or []:
|
||||
if getattr(part, "text", None):
|
||||
parts.append(part.text)
|
||||
return "\n".join(parts)
|
||||
except Exception: # noqa: BLE001
|
||||
return DEFAULT_SUMMARY_MODEL
|
||||
return str(resp)
|
||||
|
||||
|
||||
def resolve_base_url() -> str:
|
||||
return os.getenv("NOUS_BASE_URL", DEFAULT_NOUS_BASE_URL).strip() or DEFAULT_NOUS_BASE_URL
|
||||
|
||||
|
||||
def call_llm(prompt: str, *, api_key: str, model: str, base_url: str, json_mode: bool = False) -> str:
|
||||
"""Send a prompt to Nous chat completions and return the text (\"\" on failure)."""
|
||||
if not api_key:
|
||||
logger.warning("NOUS_API_KEY not set — skipping LLM call")
|
||||
def call_llm(prompt: str) -> str:
|
||||
"""Send a prompt to Gemini and return the text ("" on any failure)."""
|
||||
if not GEMINI_API_KEY:
|
||||
logger.warning("GEMINI_API_KEY not set — skipping LLM call")
|
||||
return ""
|
||||
try:
|
||||
resp = _get_client().models.generate_content(model=MODEL_NAME, contents=prompt)
|
||||
return _extract_text(resp)
|
||||
except Exception as exc: # noqa: BLE001
|
||||
logger.error("Gemini API error: %s", exc)
|
||||
return ""
|
||||
return chat(prompt, api_key=api_key, model=model, base_url=base_url, json_mode=json_mode)
|
||||
|
||||
|
||||
# ── Futures (legacy, gated) ────────────────────────────────────────────────
|
||||
|
|
@ -285,11 +206,10 @@ def build_futures_context() -> str:
|
|||
def ensure_tables() -> None:
|
||||
"""Idempotently create the news tables if missing.
|
||||
|
||||
Normally created by alembic 003_news + 005_news_items when the app
|
||||
container starts, but this summarizer may boot before the app has run
|
||||
migrations (compose only guarantees `db` is up, not that alembic has
|
||||
run). Mirrors the scraper pipeline's own CREATE TABLE IF NOT EXISTS so
|
||||
either start order is safe.
|
||||
Normally created by alembic 003_news when the app container starts, but
|
||||
this summarizer may boot before the app has run migrations (compose only
|
||||
guarantees `db` is up, not that alembic has run). Mirrors the scraper
|
||||
pipeline's own CREATE TABLE IF NOT EXISTS so either start order is safe.
|
||||
"""
|
||||
ddl = """
|
||||
CREATE TABLE IF NOT EXISTS articles (
|
||||
|
|
@ -305,27 +225,6 @@ def ensure_tables() -> None:
|
|||
summary_text TEXT NOT NULL,
|
||||
batch_timestamp TIMESTAMPTZ NOT NULL DEFAULT NOW()
|
||||
);
|
||||
ALTER TABLE article_summaries ADD COLUMN IF NOT EXISTS model TEXT;
|
||||
ALTER TABLE article_summaries ADD COLUMN IF NOT EXISTS kind TEXT;
|
||||
CREATE TABLE IF NOT EXISTS news_items (
|
||||
id SERIAL PRIMARY KEY,
|
||||
summary_id INTEGER REFERENCES article_summaries(id) ON DELETE CASCADE,
|
||||
kind TEXT NOT NULL,
|
||||
headline TEXT NOT NULL,
|
||||
importance TEXT NOT NULL,
|
||||
location_name TEXT,
|
||||
lat DOUBLE PRECISION,
|
||||
lon DOUBLE PRECISION,
|
||||
location_confidence TEXT,
|
||||
category TEXT,
|
||||
url TEXT,
|
||||
created_at TIMESTAMPTZ NOT NULL DEFAULT NOW()
|
||||
);
|
||||
CREATE INDEX IF NOT EXISTS ix_news_items_kind_created
|
||||
ON news_items (kind, created_at DESC);
|
||||
CREATE INDEX IF NOT EXISTS ix_news_items_map_bbox
|
||||
ON news_items (lon, lat)
|
||||
WHERE kind = 'map' AND lat IS NOT NULL AND lon IS NOT NULL;
|
||||
"""
|
||||
try:
|
||||
conn = psycopg2.connect(**DB_CONFIG)
|
||||
|
|
@ -338,20 +237,19 @@ def ensure_tables() -> None:
|
|||
logger.error("Error ensuring news tables: %s", exc)
|
||||
|
||||
|
||||
def get_recent_news(window_minutes: int | None = None) -> list[dict]:
|
||||
"""Fetch articles from the look-back window (content > 100 chars)."""
|
||||
mins = window_minutes if window_minutes is not None else effective_window_minutes()
|
||||
def get_recent_news() -> list[dict]:
|
||||
"""Fetch articles from the last SUMMARY_WINDOW_HOURS (content > 100 chars)."""
|
||||
query = """
|
||||
SELECT title, content, url, domain
|
||||
FROM articles
|
||||
WHERE timestamp > NOW() - make_interval(mins => %s)
|
||||
WHERE timestamp > NOW() - make_interval(hours => %s)
|
||||
AND content IS NOT NULL AND length(content) > 100
|
||||
ORDER BY timestamp DESC;
|
||||
"""
|
||||
try:
|
||||
conn = psycopg2.connect(**DB_CONFIG)
|
||||
cur = conn.cursor()
|
||||
cur.execute(query, (mins,))
|
||||
cur.execute(query, (SUMMARY_WINDOW_HOURS,))
|
||||
rows = cur.fetchall()
|
||||
cur.close()
|
||||
conn.close()
|
||||
|
|
@ -364,117 +262,24 @@ def get_recent_news(window_minutes: int | None = None) -> list[dict]:
|
|||
return []
|
||||
|
||||
|
||||
def _already_summarized_this_interval() -> bool:
|
||||
"""True when article_summaries already has a row in the last interval."""
|
||||
if os.getenv("NEWS_SUMMARIZE_FORCE", "") == "1":
|
||||
return False
|
||||
query = (
|
||||
"SELECT 1 FROM article_summaries "
|
||||
"WHERE batch_timestamp >= NOW() - make_interval(secs => %s)"
|
||||
)
|
||||
try:
|
||||
conn = psycopg2.connect(**DB_CONFIG)
|
||||
cur = conn.cursor()
|
||||
cur.execute(query, (_summarize_interval_seconds(),))
|
||||
row = cur.fetchone()
|
||||
cur.close()
|
||||
conn.close()
|
||||
return row is not None
|
||||
except Exception as exc: # noqa: BLE001
|
||||
logger.error("Error checking interval idempotency: %s", exc)
|
||||
return False
|
||||
|
||||
|
||||
def _already_recapped_today() -> bool:
|
||||
"""True when a daily_recap row already exists for the local calendar day."""
|
||||
if os.getenv("NEWS_SUMMARIZE_FORCE", "") == "1":
|
||||
return False
|
||||
tz_name = (os.getenv("TZ") or "America/New_York").strip() or "America/New_York"
|
||||
start = datetime.now(ZoneInfo(tz_name)).replace(
|
||||
hour=0, minute=0, second=0, microsecond=0
|
||||
)
|
||||
query = (
|
||||
"SELECT 1 FROM article_summaries "
|
||||
"WHERE kind = 'daily_recap' AND batch_timestamp >= %s"
|
||||
)
|
||||
try:
|
||||
conn = psycopg2.connect(**DB_CONFIG)
|
||||
cur = conn.cursor()
|
||||
cur.execute(query, (start,))
|
||||
row = cur.fetchone()
|
||||
cur.close()
|
||||
conn.close()
|
||||
return row is not None
|
||||
except Exception as exc: # noqa: BLE001
|
||||
logger.error("Error checking recap idempotency: %s", exc)
|
||||
return False
|
||||
|
||||
|
||||
def save_batch(
|
||||
summary_en: str, model: str, ticker: list, map_items: list, *, kind: str = "interval"
|
||||
) -> None:
|
||||
"""Insert the master brief plus flagged ticker/map rows."""
|
||||
ticker_rows = select_ticker(ticker or [])
|
||||
map_rows = select_map(map_items or [])
|
||||
text = (summary_en or "").strip()
|
||||
if len(text) < 10 and not ticker_rows and not map_rows:
|
||||
def save_summary_to_db(summary_text: str) -> None:
|
||||
"""Insert one master summary row (table created by alembic 003_news)."""
|
||||
if not summary_text or len(summary_text.strip()) < 10:
|
||||
logger.info("Summary too short or empty. Skipping save.")
|
||||
return
|
||||
insert_item = """
|
||||
INSERT INTO news_items (
|
||||
summary_id, kind, headline, importance, location_name,
|
||||
lat, lon, location_confidence, category, url
|
||||
) VALUES (%s, %s, %s, %s, %s, %s, %s, %s, %s, %s)
|
||||
"""
|
||||
try:
|
||||
conn = psycopg2.connect(**DB_CONFIG)
|
||||
cur = conn.cursor()
|
||||
cur.execute(
|
||||
"INSERT INTO article_summaries (summary_text, model, kind) VALUES (%s, %s, %s) RETURNING id",
|
||||
(text, model, kind),
|
||||
"INSERT INTO article_summaries (summary_text) VALUES (%s)",
|
||||
(summary_text.strip(),),
|
||||
)
|
||||
summary_id = cur.fetchone()[0]
|
||||
for row in ticker_rows:
|
||||
cur.execute(
|
||||
insert_item,
|
||||
(
|
||||
summary_id,
|
||||
"ticker",
|
||||
row.get("headline"),
|
||||
row.get("importance"),
|
||||
row.get("location_name"),
|
||||
None,
|
||||
None,
|
||||
None,
|
||||
None,
|
||||
row.get("url"),
|
||||
),
|
||||
)
|
||||
for row in map_rows:
|
||||
cur.execute(
|
||||
insert_item,
|
||||
(
|
||||
summary_id,
|
||||
"map",
|
||||
row.get("headline"),
|
||||
row.get("importance"),
|
||||
row.get("location_name"),
|
||||
row.get("lat"),
|
||||
row.get("lon"),
|
||||
row.get("location_confidence"),
|
||||
row.get("category"),
|
||||
row.get("url"),
|
||||
),
|
||||
)
|
||||
conn.commit()
|
||||
logger.info(
|
||||
"Master summary saved id=%s model=%s kind=%s ticker=%d map=%d",
|
||||
summary_id, model, kind, len(ticker_rows), len(map_rows),
|
||||
)
|
||||
logger.info("Master summary saved to database successfully.")
|
||||
cur.close()
|
||||
conn.close()
|
||||
except Exception as exc: # noqa: BLE001
|
||||
logger.error("Error saving batch to DB: %s", exc)
|
||||
logger.error("Error saving summary to DB: %s", exc)
|
||||
|
||||
|
||||
# ── Orchestration ──────────────────────────────────────────────────────────
|
||||
|
|
@ -485,73 +290,40 @@ def build_map_prompt(batch: list[dict]) -> str:
|
|||
for a in batch
|
||||
)
|
||||
template = os.getenv("MAP_PROMPT", MAP_PROMPT_DEFAULT)
|
||||
prefix = build_futures_context() + "\n" if INCLUDE_FUTURES else ""
|
||||
try:
|
||||
return template.format(batch_text=batch_text)
|
||||
return prefix + template.format(batch_text=batch_text)
|
||||
except KeyError:
|
||||
return template
|
||||
return prefix + template
|
||||
|
||||
|
||||
def build_master_prompt(final_input: str, recap: bool = False) -> str:
|
||||
if recap:
|
||||
template = os.getenv("RECAP_PROMPT", RECAP_PROMPT_DEFAULT)
|
||||
else:
|
||||
template = os.getenv("SUMMARY_PROMPT", SUMMARY_PROMPT_DEFAULT)
|
||||
if "{final_input}" in template:
|
||||
return template.replace("{final_input}", final_input)
|
||||
return template
|
||||
def build_master_prompt(final_input: str) -> str:
|
||||
template = os.getenv("SUMMARY_PROMPT", SUMMARY_PROMPT_DEFAULT)
|
||||
prefix = build_futures_context() + "\n" if INCLUDE_FUTURES else ""
|
||||
try:
|
||||
return prefix + template.format(final_input=final_input)
|
||||
except KeyError:
|
||||
return prefix + template
|
||||
|
||||
|
||||
def summarize_news() -> None:
|
||||
"""Map-reduce summarize recent articles and store brief + ticker + map."""
|
||||
recap = is_recap_run()
|
||||
window = effective_window_minutes(recap=recap)
|
||||
"""Map-reduce summarize recent articles and store the master summary."""
|
||||
ensure_tables()
|
||||
if recap:
|
||||
if _already_recapped_today():
|
||||
logger.info(
|
||||
"Skipping recap: article_summaries already has daily_recap today "
|
||||
"(set NEWS_SUMMARIZE_FORCE=1 to override)"
|
||||
)
|
||||
return
|
||||
elif _already_summarized_this_interval():
|
||||
logger.info(
|
||||
"Skipping summarize: article_summaries already has a row in the last %ss "
|
||||
"(set NEWS_SUMMARIZE_FORCE=1 to override)",
|
||||
_summarize_interval_seconds(),
|
||||
)
|
||||
return
|
||||
|
||||
api_key = resolve_api_key()
|
||||
model = resolve_model()
|
||||
base_url = resolve_base_url()
|
||||
if not api_key:
|
||||
logger.warning("NOUS_API_KEY unset in env and api_keys — idle this run")
|
||||
return
|
||||
|
||||
articles = get_recent_news(window)
|
||||
articles = get_recent_news()
|
||||
if not articles:
|
||||
logger.info("No new articles found in the last %s min.", window)
|
||||
logger.info("No new articles found in the last %sh.", SUMMARY_WINDOW_HOURS)
|
||||
return
|
||||
|
||||
logger.info(
|
||||
"Processing %d articles with %s (batch_size=%d, recap=%s, window_min=%s)...",
|
||||
len(articles), model, BATCH_SIZE, recap, window,
|
||||
"Processing %d articles with %s (batch_size=%d, futures=%s)...",
|
||||
len(articles), MODEL_NAME, BATCH_SIZE, INCLUDE_FUTURES,
|
||||
)
|
||||
|
||||
partial_summaries: list[str] = []
|
||||
for i in range(0, len(articles), BATCH_SIZE):
|
||||
batch = articles[i : i + BATCH_SIZE]
|
||||
logger.info(
|
||||
"map batch %d/%d (%d articles)",
|
||||
i // BATCH_SIZE + 1, -(-len(articles) // BATCH_SIZE), len(batch),
|
||||
)
|
||||
summary = call_llm(
|
||||
build_map_prompt(batch),
|
||||
api_key=api_key,
|
||||
model=model,
|
||||
base_url=base_url,
|
||||
json_mode=False,
|
||||
)
|
||||
logger.info("map batch %d/%d (%d articles)", i // BATCH_SIZE + 1, -(-len(articles) // BATCH_SIZE), len(batch))
|
||||
summary = call_llm(build_map_prompt(batch))
|
||||
if summary:
|
||||
partial_summaries.append(summary)
|
||||
|
||||
|
|
@ -561,24 +333,9 @@ def summarize_news() -> None:
|
|||
return
|
||||
|
||||
logger.info("reduce phase over %d partial summaries", len(partial_summaries))
|
||||
master_raw = call_llm(
|
||||
build_master_prompt(final_input, recap=recap),
|
||||
api_key=api_key,
|
||||
model=model,
|
||||
base_url=base_url,
|
||||
json_mode=True,
|
||||
)
|
||||
if not master_raw:
|
||||
logger.warning("Reduce phase returned empty — nothing to persist.")
|
||||
return
|
||||
parsed = parse_reduce_json(master_raw)
|
||||
save_batch(
|
||||
parsed["summary_en"],
|
||||
model,
|
||||
parsed["ticker"],
|
||||
parsed["map_items"],
|
||||
kind="daily_recap" if recap else "interval",
|
||||
)
|
||||
master_summary = call_llm(build_master_prompt(final_input))
|
||||
if master_summary:
|
||||
save_summary_to_db(master_summary)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
|
|
|
|||
|
|
@ -1,11 +0,0 @@
|
|||
"""Keep summarizer unit tests importable without Postgres drivers."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import sys
|
||||
from types import ModuleType
|
||||
|
||||
if "psycopg2" not in sys.modules:
|
||||
fake = ModuleType("psycopg2")
|
||||
fake.connect = lambda **kwargs: None # type: ignore[attr-defined]
|
||||
sys.modules["psycopg2"] = fake
|
||||
|
|
@ -1,61 +0,0 @@
|
|||
from intel import parse_reduce_json, clamp_coords, select_ticker, select_map
|
||||
|
||||
FENCED = """```json
|
||||
{"summary_en": "Brief.", "ticker": [
|
||||
{"headline": "Blast in Kyiv", "importance": "critical", "url": "https://ex", "location_name": "Kyiv"}
|
||||
], "map_items": [
|
||||
{"headline": "Blast in Kyiv", "importance": "critical", "location_name": "Kyiv, Ukraine",
|
||||
"lat": 50.45, "lon": 30.52, "location_confidence": "city", "category": "military/conflict", "url": "https://ex"}
|
||||
]}
|
||||
```"""
|
||||
|
||||
def test_parse_strips_fence_and_think_tags():
|
||||
raw = "<think>nope</think>\n" + FENCED
|
||||
out = parse_reduce_json(raw)
|
||||
assert out["summary_en"] == "Brief."
|
||||
assert len(out["ticker"]) == 1
|
||||
|
||||
def test_parse_empty_and_garbage_returns_empty_struct():
|
||||
assert parse_reduce_json("")["summary_en"] == ""
|
||||
assert parse_reduce_json("not json")["ticker"] == []
|
||||
|
||||
def test_clamp_coords_drops_out_of_range_and_unknown():
|
||||
assert clamp_coords(50.45, 30.52) == (50.45, 30.52)
|
||||
assert clamp_coords(95.0, 10.0) is None
|
||||
assert clamp_coords(None, 10.0) is None
|
||||
assert clamp_coords("50.45", "30.52") == (50.45, 30.52)
|
||||
|
||||
def test_select_ticker_keeps_critical_high_caps_12():
|
||||
rows = [{"headline": f"h{i}", "importance": "critical"} for i in range(15)]
|
||||
rows.append({"headline": "skip", "importance": "low"})
|
||||
out = select_ticker(rows)
|
||||
assert len(out) == 12
|
||||
assert all(r["importance"] in ("critical", "high") for r in out)
|
||||
|
||||
|
||||
def test_select_ticker_falls_back_to_medium_low_when_nothing_flagged():
|
||||
rows = [
|
||||
{"headline": "shop theft", "importance": "low"},
|
||||
{"headline": "highway crash", "importance": "medium"},
|
||||
{"headline": "none", "importance": "none"},
|
||||
]
|
||||
out = select_ticker(rows)
|
||||
assert [r["headline"] for r in out] == ["highway crash", "shop theft"]
|
||||
|
||||
def test_select_map_requires_valid_coords_and_flag():
|
||||
items = [
|
||||
{"headline": "A", "importance": "critical", "lat": 50.45, "lon": 30.52, "location_name": "Kyiv"},
|
||||
{"headline": "B", "importance": "critical", "lat": None, "lon": None, "location_name": "Unknown"},
|
||||
{"headline": "C", "importance": "low", "lat": 1.0, "lon": 2.0, "location_name": "x"},
|
||||
]
|
||||
out = select_map(items)
|
||||
assert [r["headline"] for r in out] == ["A"]
|
||||
|
||||
def test_select_map_caps_20():
|
||||
items = [
|
||||
{"headline": f"h{i}", "importance": "critical", "lat": 1.0, "lon": 2.0}
|
||||
for i in range(25)
|
||||
]
|
||||
out = select_map(items)
|
||||
assert len(out) == 20
|
||||
assert all(r["importance"] in ("critical", "high") for r in out)
|
||||
|
|
@ -1,124 +0,0 @@
|
|||
from unittest.mock import MagicMock
|
||||
|
||||
import httpx
|
||||
from nous_client import chat
|
||||
|
||||
DEFAULT_UA = "osint-dashboard-news-summarizer"
|
||||
BASE = "https://inference-api.nousresearch.com/v1"
|
||||
|
||||
|
||||
def _ok_response(content="hello"):
|
||||
resp = MagicMock()
|
||||
resp.status_code = 200
|
||||
resp.json.return_value = {"choices": [{"message": {"content": content}}]}
|
||||
return resp
|
||||
|
||||
|
||||
def _install_fake(monkeypatch, post_impl):
|
||||
captured = {}
|
||||
|
||||
class FakeClient:
|
||||
def __init__(self, timeout=None, **kwargs):
|
||||
captured["timeout"] = timeout
|
||||
|
||||
def __enter__(self):
|
||||
return self
|
||||
|
||||
def __exit__(self, *exc):
|
||||
return False
|
||||
|
||||
def post(self, url, *, headers=None, json=None, **kwargs):
|
||||
captured["url"] = url
|
||||
captured["headers"] = headers
|
||||
captured["json"] = json
|
||||
return post_impl(url, headers, json)
|
||||
|
||||
monkeypatch.setattr(httpx, "Client", FakeClient)
|
||||
return captured
|
||||
|
||||
|
||||
def test_posts_chat_completions_with_auth_body_and_returns_content(monkeypatch):
|
||||
captured = _install_fake(monkeypatch, lambda *a: _ok_response("the-content"))
|
||||
out = chat(
|
||||
"summarize this",
|
||||
api_key="secret-key",
|
||||
model="hermes-3",
|
||||
base_url=BASE,
|
||||
)
|
||||
assert out == "the-content"
|
||||
assert captured["url"] == f"{BASE}/chat/completions"
|
||||
assert captured["headers"]["Authorization"] == "Bearer secret-key"
|
||||
assert captured["headers"]["User-Agent"] == DEFAULT_UA
|
||||
assert captured["json"]["model"] == "hermes-3"
|
||||
assert captured["json"]["messages"] == [{"role": "user", "content": "summarize this"}]
|
||||
assert captured["json"]["temperature"] == 0.2
|
||||
assert captured["json"]["max_tokens"] == 4096
|
||||
assert "response_format" not in captured["json"]
|
||||
|
||||
|
||||
def test_user_agent_equals_osint_user_agent_env(monkeypatch):
|
||||
monkeypatch.setenv("OSINT_USER_AGENT", "custom-ua/2.0")
|
||||
captured = _install_fake(monkeypatch, lambda *a: _ok_response("ok"))
|
||||
chat("p", api_key="k", model="m", base_url=BASE)
|
||||
assert captured["headers"]["User-Agent"] == "custom-ua/2.0"
|
||||
|
||||
|
||||
def test_json_mode_sets_response_format(monkeypatch):
|
||||
captured = _install_fake(monkeypatch, lambda *a: _ok_response("{}"))
|
||||
chat("p", api_key="k", model="m", base_url=BASE, json_mode=True)
|
||||
assert captured["json"]["response_format"] == {"type": "json_object"}
|
||||
assert captured["json"]["max_tokens"] >= 8192
|
||||
roles = [m["role"] for m in captured["json"]["messages"]]
|
||||
assert "system" in roles
|
||||
assert "user" in roles
|
||||
|
||||
|
||||
def test_retries_once_when_finish_reason_is_length(monkeypatch):
|
||||
calls = {"n": 0}
|
||||
|
||||
def post_impl(*a):
|
||||
calls["n"] += 1
|
||||
if calls["n"] == 1:
|
||||
resp = MagicMock()
|
||||
resp.status_code = 200
|
||||
resp.json.return_value = {
|
||||
"choices": [{
|
||||
"message": {"content": "{\"summary_en\": \"cut off"},
|
||||
"finish_reason": "length",
|
||||
}]
|
||||
}
|
||||
return resp
|
||||
return _ok_response('{"summary_en": "complete brief."}')
|
||||
|
||||
_install_fake(monkeypatch, post_impl)
|
||||
out = chat("p", api_key="k", model="m", base_url=BASE, json_mode=True)
|
||||
assert calls["n"] == 2
|
||||
assert "complete brief" in out
|
||||
|
||||
|
||||
def test_401_returns_empty_string(monkeypatch):
|
||||
def post_impl(*a):
|
||||
resp = MagicMock()
|
||||
resp.status_code = 401
|
||||
return resp
|
||||
|
||||
_install_fake(monkeypatch, post_impl)
|
||||
assert chat("p", api_key="bad", model="m", base_url=BASE) == ""
|
||||
|
||||
|
||||
def test_5xx_returns_empty_string(monkeypatch):
|
||||
def post_impl(*a):
|
||||
resp = MagicMock()
|
||||
resp.status_code = 503
|
||||
return resp
|
||||
|
||||
_install_fake(monkeypatch, post_impl)
|
||||
assert chat("p", api_key="k", model="m", base_url=BASE) == ""
|
||||
|
||||
|
||||
def test_timeout_returns_empty_string(monkeypatch):
|
||||
def post_impl(*a):
|
||||
raise httpx.TimeoutException("timed out")
|
||||
|
||||
_install_fake(monkeypatch, post_impl)
|
||||
assert chat("p", api_key="k", model="m", base_url=BASE) == ""
|
||||
|
|
@ -1,60 +0,0 @@
|
|||
"""Default prompts: breaking news, ignore futures/market tape."""
|
||||
|
||||
from summarizer import MAP_PROMPT_DEFAULT, RECAP_PROMPT_DEFAULT, SUMMARY_PROMPT_DEFAULT
|
||||
|
||||
|
||||
def _assert_breaking_not_futures(prompt: str) -> None:
|
||||
p = prompt.lower()
|
||||
assert "breaking" in p
|
||||
assert "futures" in p
|
||||
assert "ignore" in p or "do not" in p or "not" in p
|
||||
assert "es=f" not in p
|
||||
assert "yfinance" not in p
|
||||
|
||||
|
||||
def test_map_prompt_focuses_on_breaking_news_not_futures():
|
||||
_assert_breaking_not_futures(MAP_PROMPT_DEFAULT)
|
||||
|
||||
|
||||
def test_summary_prompt_focuses_on_breaking_news_not_futures():
|
||||
_assert_breaking_not_futures(SUMMARY_PROMPT_DEFAULT)
|
||||
p = SUMMARY_PROMPT_DEFAULT.lower()
|
||||
assert "commodity" in p or "market" in p
|
||||
|
||||
|
||||
def test_summary_prompt_covers_critical_then_incidents():
|
||||
p = SUMMARY_PROMPT_DEFAULT.lower()
|
||||
assert "critical" in p
|
||||
assert "crime" in p
|
||||
assert "incident" in p
|
||||
assert "complete" in p or "truncat" in p or "unfinished" in p or "mid-sentence" in p
|
||||
|
||||
|
||||
def test_summary_prompt_does_not_bail_out_with_canned_empty_brief():
|
||||
p = SUMMARY_PROMPT_DEFAULT
|
||||
assert "AND STOP" not in p
|
||||
assert "REPEAT 3 TIMES" not in p
|
||||
assert "No qualifying" not in p
|
||||
assert "no-qualifying" not in p.lower()
|
||||
low = p.lower()
|
||||
assert "always" in low
|
||||
assert "recap" in low or "summar" in low
|
||||
|
||||
|
||||
def test_recap_prompt_does_not_bail_out_with_canned_empty_brief():
|
||||
p = RECAP_PROMPT_DEFAULT
|
||||
assert "AND STOP" not in p
|
||||
assert "No qualifying" not in p
|
||||
assert "no-qualifying" not in p.lower()
|
||||
low = p.lower()
|
||||
assert "always" in low
|
||||
assert "daily" in low
|
||||
assert "24" in low
|
||||
|
||||
p = RECAP_PROMPT_DEFAULT.lower()
|
||||
_assert_breaking_not_futures(RECAP_PROMPT_DEFAULT)
|
||||
assert "daily" in p
|
||||
assert "24" in p
|
||||
assert "summary_en" in p
|
||||
assert "ticker" in p
|
||||
assert "map_items" in p
|
||||
|
|
@ -1,42 +0,0 @@
|
|||
"""Wall-clock scheduling: 15-min analyst + 23:00 America/New_York recap."""
|
||||
|
||||
from datetime import datetime, timedelta
|
||||
from zoneinfo import ZoneInfo
|
||||
|
||||
from run_news_summarizer import next_event, next_recap_datetime
|
||||
|
||||
TZ = ZoneInfo("America/New_York")
|
||||
|
||||
|
||||
def test_next_recap_is_11pm_same_day_before_2300():
|
||||
now = datetime(2026, 8, 28, 15, 4, tzinfo=TZ)
|
||||
got = next_recap_datetime(now)
|
||||
assert got == datetime(2026, 8, 28, 23, 0, tzinfo=TZ)
|
||||
|
||||
|
||||
def test_next_recap_is_11pm_next_day_at_or_after_2300():
|
||||
now = datetime(2026, 8, 28, 23, 0, tzinfo=TZ)
|
||||
got = next_recap_datetime(now)
|
||||
assert got == datetime(2026, 8, 29, 23, 0, tzinfo=TZ)
|
||||
|
||||
|
||||
def test_next_recap_honors_custom_hour():
|
||||
now = datetime(2026, 8, 28, 10, 0, tzinfo=TZ)
|
||||
got = next_recap_datetime(now, hour=22, minute=30)
|
||||
assert got == datetime(2026, 8, 28, 22, 30, tzinfo=TZ)
|
||||
|
||||
|
||||
def test_next_event_picks_recap_when_sooner_than_interval():
|
||||
now = datetime(2026, 8, 28, 22, 50, tzinfo=TZ)
|
||||
last_periodic = now - timedelta(seconds=100)
|
||||
when, kind = next_event(now, last_periodic=last_periodic, interval_s=900)
|
||||
assert kind == "recap"
|
||||
assert when == datetime(2026, 8, 28, 23, 0, tzinfo=TZ)
|
||||
|
||||
|
||||
def test_next_event_picks_interval_when_recap_is_hours_away():
|
||||
now = datetime(2026, 8, 28, 10, 0, tzinfo=TZ)
|
||||
last_periodic = now
|
||||
when, kind = next_event(now, last_periodic=last_periodic, interval_s=900)
|
||||
assert kind == "interval"
|
||||
assert when == now + timedelta(seconds=900)
|
||||
|
|
@ -1,45 +0,0 @@
|
|||
"""Summarizer window, recap flag, and no futures injection into prompts."""
|
||||
|
||||
from summarizer import (
|
||||
build_map_prompt,
|
||||
build_master_prompt,
|
||||
effective_window_minutes,
|
||||
is_recap_run,
|
||||
)
|
||||
|
||||
|
||||
def test_interval_window_defaults_to_15_minutes(monkeypatch):
|
||||
monkeypatch.delenv("NEWS_RECAP", raising=False)
|
||||
monkeypatch.delenv("SUMMARY_WINDOW_HOURS", raising=False)
|
||||
monkeypatch.setenv("SUMMARY_WINDOW_MINUTES", "15")
|
||||
assert is_recap_run() is False
|
||||
assert effective_window_minutes() == 15
|
||||
|
||||
|
||||
def test_recap_window_is_24_hours(monkeypatch):
|
||||
monkeypatch.setenv("NEWS_RECAP", "1")
|
||||
monkeypatch.setenv("SUMMARY_WINDOW_MINUTES", "15")
|
||||
assert is_recap_run() is True
|
||||
assert effective_window_minutes() == 24 * 60
|
||||
|
||||
|
||||
def test_build_map_prompt_does_not_inject_futures():
|
||||
prompt = build_map_prompt(
|
||||
[{"title": "Blast", "domain": "ex.com", "url": "https://ex.com/1", "content": "x" * 120}]
|
||||
)
|
||||
assert "FUTURES PRICES" not in prompt
|
||||
assert "ES=F" not in prompt
|
||||
assert "Blast" in prompt
|
||||
|
||||
|
||||
def test_build_master_prompt_interval_uses_summary_not_recap():
|
||||
prompt = build_master_prompt("partial facts", recap=False)
|
||||
assert "partial facts" in prompt
|
||||
assert "daily recap" not in prompt.lower()
|
||||
|
||||
|
||||
def test_build_master_prompt_recap_uses_daily_template():
|
||||
prompt = build_master_prompt("partial facts", recap=True)
|
||||
assert "partial facts" in prompt
|
||||
assert "daily recap" in prompt.lower()
|
||||
assert "24" in prompt
|
||||
|
|
@ -1,3 +0,0 @@
|
|||
[pytest]
|
||||
testpaths = tests
|
||||
python_files = test_*.py
|
||||
|
|
@ -1,74 +0,0 @@
|
|||
#!/usr/bin/env bash
|
||||
# Recreate selected OSINT compose services WITHOUT bouncing Postgres.
|
||||
#
|
||||
# The old path was `compose down` + up, which stopped osint-db on every merge
|
||||
# even when Dockerfile.pg did not change. Name-pinned leftovers are still
|
||||
# removed, but only for the services we are actually replacing.
|
||||
#
|
||||
# Usage: scripts/compose-reup.sh [compose-service ...]
|
||||
# (default: app ingester camera-service news-scraper news-summarizer)
|
||||
# Env: COMPOSE_PROJECT_NAME (default osint-dashboard)
|
||||
# COMPOSE_PROFILES (default ingest)
|
||||
# FORCE_RECREATE_DB=1 also recreate db
|
||||
set -euo pipefail
|
||||
|
||||
ROOT="$(cd "$(dirname "$0")/.." && pwd)"
|
||||
cd "$ROOT"
|
||||
|
||||
export COMPOSE_PROJECT_NAME="${COMPOSE_PROJECT_NAME:-osint-dashboard}"
|
||||
PROFILE="${COMPOSE_PROFILES:-ingest}"
|
||||
|
||||
DEFAULT_SVCS=(app ingester camera-service news-scraper news-summarizer)
|
||||
if [ "$#" -gt 0 ]; then
|
||||
SVCS=("$@")
|
||||
else
|
||||
SVCS=("${DEFAULT_SVCS[@]}")
|
||||
fi
|
||||
|
||||
if [ "${FORCE_RECREATE_DB:-0}" = "1" ]; then
|
||||
SVCS+=(db)
|
||||
fi
|
||||
|
||||
# Never recreate db unless it was requested.
|
||||
FILTERED=()
|
||||
for svc in "${SVCS[@]}"; do
|
||||
if [ "$svc" = "db" ] && [ "${FORCE_RECREATE_DB:-0}" != "1" ]; then
|
||||
echo "compose-reup: skipping db (set FORCE_RECREATE_DB=1 to bounce Postgres)"
|
||||
continue
|
||||
fi
|
||||
FILTERED+=("$svc")
|
||||
done
|
||||
SVCS=("${FILTERED[@]}")
|
||||
|
||||
declare -A CONTAINER_NAME=(
|
||||
[app]=osint-dashboard
|
||||
[ingester]=osint-ingester
|
||||
[camera-service]=osint-camera-scraper
|
||||
[news-scraper]=osint-news-scraper
|
||||
[news-summarizer]=osint-news-summarizer
|
||||
[db]=osint-db
|
||||
[nats]=osint-nats
|
||||
[titiler]=osint-titiler
|
||||
)
|
||||
|
||||
echo "compose-reup: project=${COMPOSE_PROJECT_NAME} profile=${PROFILE} dir=${ROOT}"
|
||||
echo "compose-reup: recreate=${SVCS[*]:-none}"
|
||||
|
||||
# Keep data-plane containers running (db / nats / titiler).
|
||||
docker compose --profile "${PROFILE}" up -d --no-build --no-recreate db nats titiler || true
|
||||
|
||||
if [ "${#SVCS[@]}" -eq 0 ]; then
|
||||
docker compose --profile "${PROFILE}" ps
|
||||
exit 0
|
||||
fi
|
||||
|
||||
for svc in "${SVCS[@]}"; do
|
||||
c="${CONTAINER_NAME[$svc]:-}"
|
||||
if [ -n "$c" ] && docker inspect "$c" >/dev/null 2>&1; then
|
||||
echo "compose-reup: replacing ${c}"
|
||||
docker rm -f "$c" >/dev/null
|
||||
fi
|
||||
done
|
||||
|
||||
docker compose --profile "${PROFILE}" up -d --no-build --no-deps "${SVCS[@]}"
|
||||
docker compose --profile "${PROFILE}" ps
|
||||
|
|
@ -1,10 +0,0 @@
|
|||
#!/usr/bin/env bash
|
||||
# Pull OSINT images from Forgejo registry and retag for docker-compose (localhost/*).
|
||||
set -euo pipefail
|
||||
REG="${FORGEJO_REGISTRY:-forgejo.siriusdevops.com}"
|
||||
OWN="${FORGEJO_OWNER:-sirius}"
|
||||
for name in osint-dashboard osint-dashboard-pg osint-news-scraper osint-news-summarizer; do
|
||||
docker pull "${REG}/${OWN}/${name}:latest"
|
||||
docker tag "${REG}/${OWN}/${name}:latest" "localhost/${name}:latest"
|
||||
echo "ok ${name}"
|
||||
done
|
||||
|
|
@ -1,57 +0,0 @@
|
|||
"""Aircraft popup enrichment + emergency/MIL layer contract (static HTML)."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from pathlib import Path
|
||||
|
||||
ROOT = Path(__file__).resolve().parent.parent
|
||||
HTML = (ROOT / "app/static/index.html").read_text()
|
||||
|
||||
|
||||
def _fn(name: str, nxt: str) -> str:
|
||||
return HTML.split(f"function {name}", 1)[1].split(f"function {nxt}", 1)[0]
|
||||
|
||||
|
||||
def test_popup_has_required_adsb_fields_and_photo():
|
||||
js = _fn("pointPopup", "loadPlanePhoto")
|
||||
for field in ("callsign", "hex", "registration", "type", "alt", "gs", "squawk"):
|
||||
assert f"add('{field}'" in js
|
||||
assert "class=\"ps-photo\"" in js or "class='ps-photo'" in js
|
||||
assert "wikipedia" not in js.lower()
|
||||
assert "ceo" not in js.lower()
|
||||
|
||||
|
||||
def test_emergency_badge_and_squawk_codes():
|
||||
assert "role-badge emergency" in HTML
|
||||
assert "hdg-emerg" in HTML
|
||||
assert "EMERG_SQUAWK" in HTML
|
||||
assert "['7700', '7600', '7500']" in HTML
|
||||
emerg = HTML.split("function acIsEmergency", 1)[1].split("function acVisible", 1)[0]
|
||||
assert "EMERG_SQUAWK.has(sq)" in emerg
|
||||
color = HTML.split("function acColor", 1)[1].split("function connectLiveWs", 1)[0]
|
||||
assert "acIsEmergency(p)" in color
|
||||
assert "#ff5d5d" in color
|
||||
|
||||
|
||||
def test_mil_toggle_hidden_until_role_flag_and_never_hits_adsb_lol():
|
||||
assert 'id="lp-ac-mil-row"' in HTML
|
||||
assert 'id="lp-ac-mil-on"' in HTML
|
||||
row = HTML.split('id="lp-ac-mil-row"', 1)[1].split(">", 1)[0]
|
||||
assert "hidden" in row
|
||||
on = HTML.split('id="lp-ac-mil-on"', 1)[1].split(">", 1)[0]
|
||||
assert "checked" not in on
|
||||
load = HTML.split("async function loadAircraft", 1)[1].split("async function toggleTrains", 1)[0]
|
||||
assert "/api/aircraft?bbox=" in load
|
||||
assert "api.adsb.lol" not in load
|
||||
assert "noteMilSupport" in load
|
||||
assert "acMilOn" in load
|
||||
note = HTML.split("function noteMilSupport", 1)[1].split("function acColor", 1)[0]
|
||||
assert "extra.role" in note
|
||||
assert "lp-ac-mil-row" in note
|
||||
assert "hidden = false" in note
|
||||
|
||||
|
||||
def test_planespotters_lazy_photo_still_wired():
|
||||
assert "function loadPlanePhoto" in HTML
|
||||
assert "/api/aircraft/photo?" in HTML
|
||||
assert "map.on('popupopen', (e) => { loadPlanePhoto(e.popup); });" in HTML
|
||||
|
|
@ -16,12 +16,6 @@ async def _get(path: str) -> httpx.Response:
|
|||
return await client.get(path)
|
||||
|
||||
|
||||
async def _post(path: str, payload: dict | None) -> httpx.Response:
|
||||
transport = httpx.ASGITransport(app=app)
|
||||
async with httpx.AsyncClient(transport=transport, base_url=BASE) as client:
|
||||
return await client.post(path, json=payload)
|
||||
|
||||
|
||||
def test_map_layers_includes_overlays():
|
||||
body = asyncio.run(_get("/api/map/layers")).json()
|
||||
assert "layers" in body
|
||||
|
|
@ -43,67 +37,3 @@ def test_vessels_empty_without_ais_key():
|
|||
assert resp.status_code == 200
|
||||
assert resp.json() == []
|
||||
assert "max-age" in (resp.headers.get("cache-control") or "").lower()
|
||||
|
||||
|
||||
def _desired_boxes():
|
||||
import asyncio as _a
|
||||
from ais_stream import _take_desired_boxes
|
||||
return _a.run(_take_desired_boxes())
|
||||
|
||||
|
||||
def test_vessels_subscribe_sets_viewport_box():
|
||||
assert _desired_boxes() is None
|
||||
resp = asyncio.run(_post("/api/vessels/subscribe", {"bbox": "-70,40,-60,45"}))
|
||||
assert resp.status_code == 200
|
||||
body = resp.json()
|
||||
assert body["ok"] is True and body["bbox"] == "-70,40,-60,45"
|
||||
# AISStream corner order: [[lat, lon], [lat, lon]] (southwest, northeast).
|
||||
assert _desired_boxes() == [[[40.0, -70.0], [45.0, -60.0]]]
|
||||
|
||||
|
||||
def test_vessels_subscribe_empty_resets():
|
||||
assert asyncio.run(_post("/api/vessels/subscribe", {"bbox": "-70,40,-60,45"})).status_code == 200
|
||||
resp = asyncio.run(_post("/api/vessels/subscribe", {"bbox": ""}))
|
||||
assert resp.status_code == 200
|
||||
assert resp.json()["bbox"] is None
|
||||
assert _desired_boxes() is None
|
||||
|
||||
|
||||
def test_vessels_subscribe_rejects_bad_bbox():
|
||||
for bad in ("1,2,3", "a,b,c,d", "20,30,10,40", "0,0,0,200"):
|
||||
resp = asyncio.run(_post("/api/vessels/subscribe", {"bbox": bad}))
|
||||
assert resp.status_code == 422, bad
|
||||
# Explicit null bbox is the "reset to env default" path (still 200).
|
||||
assert asyncio.run(_post("/api/vessels/subscribe", {"bbox": None})).status_code == 200
|
||||
|
||||
|
||||
def test_aircraft_photo_requires_hex_or_reg():
|
||||
assert asyncio.run(_get("/api/aircraft/photo")).status_code == 422
|
||||
|
||||
|
||||
def test_aircraft_photo_rejects_bad_hex():
|
||||
# hex must be exactly 6 hex chars
|
||||
resp = asyncio.run(_get("/api/aircraft/photo?hex=xyz1234"))
|
||||
assert resp.status_code == 422
|
||||
|
||||
|
||||
def test_aircraft_photo_returns_photo(monkeypatch):
|
||||
async def fake(hex_code=None, reg=None):
|
||||
return {"id": "1", "src": "https://t.plnspttrs.net/a_280.jpg",
|
||||
"link": "https://www.planespotters.net/photo/1/x", "photographer": "A"}
|
||||
|
||||
monkeypatch.setattr("main.fetch_planespotters_photo", fake)
|
||||
resp = asyncio.run(_get("/api/aircraft/photo?hex=e8027e"))
|
||||
assert resp.status_code == 200
|
||||
body = resp.json()
|
||||
assert body["src"].startswith("https://t.plnspttrs.net/")
|
||||
assert "max-age" in (resp.headers.get("cache-control") or "").lower()
|
||||
|
||||
|
||||
def test_aircraft_photo_404_when_no_photo(monkeypatch):
|
||||
async def fake(hex_code=None, reg=None):
|
||||
return None
|
||||
|
||||
monkeypatch.setattr("main.fetch_planespotters_photo", fake)
|
||||
resp = asyncio.run(_get("/api/aircraft/photo?reg=D-ABCD"))
|
||||
assert resp.status_code == 404
|
||||
|
|
|
|||
|
|
@ -1,9 +1,9 @@
|
|||
"""Integration tests for the news pipeline API (GET /api/news + summaries + intel).
|
||||
"""Integration tests for the news pipeline API (GET /api/news + summaries).
|
||||
|
||||
DB-backed: marked `requires_db` and auto-skip when the test database is
|
||||
unreachable (see tests/conftest.py). Seeding writes directly to the shared
|
||||
`articles` / `article_summaries` / `news_items` tables, exactly as the scraper
|
||||
+ summarizer services would.
|
||||
`articles` / `article_summaries` tables, exactly as the scraper + summarizer
|
||||
services would.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
|
@ -37,9 +37,7 @@ def _truncate() -> None:
|
|||
async def run():
|
||||
conn = await asyncpg.connect(**_conn_kwargs())
|
||||
try:
|
||||
await conn.execute(
|
||||
"TRUNCATE articles, article_summaries, news_items CASCADE"
|
||||
)
|
||||
await conn.execute("TRUNCATE articles, article_summaries")
|
||||
finally:
|
||||
await conn.close()
|
||||
|
||||
|
|
@ -68,45 +66,14 @@ def _seed_article(title: str, url: str, domain: str, ts: str, content: str = "bo
|
|||
asyncio.run(run())
|
||||
|
||||
|
||||
def _seed_summary(text: str, ts: str, model: str | None = None, kind: str | None = None) -> int:
|
||||
async def run() -> int:
|
||||
conn = await asyncpg.connect(**_conn_kwargs())
|
||||
try:
|
||||
row = await conn.fetchrow(
|
||||
"INSERT INTO article_summaries (summary_text, batch_timestamp, model, kind) "
|
||||
"VALUES ($1, $2, $3, $4) RETURNING id",
|
||||
text, datetime.fromisoformat(ts), model, kind,
|
||||
)
|
||||
return int(row["id"])
|
||||
finally:
|
||||
await conn.close()
|
||||
|
||||
return asyncio.run(run())
|
||||
|
||||
|
||||
def _seed_news_item(
|
||||
summary_id: int,
|
||||
kind: str,
|
||||
headline: str,
|
||||
importance: str,
|
||||
*,
|
||||
location_name: str | None = None,
|
||||
lat: float | None = None,
|
||||
lon: float | None = None,
|
||||
location_confidence: str | None = None,
|
||||
category: str | None = None,
|
||||
url: str | None = None,
|
||||
) -> None:
|
||||
def _seed_summary(text: str, ts: str) -> None:
|
||||
async def run():
|
||||
conn = await asyncpg.connect(**_conn_kwargs())
|
||||
try:
|
||||
await conn.execute(
|
||||
"INSERT INTO news_items "
|
||||
"(summary_id, kind, headline, importance, location_name, "
|
||||
" lat, lon, location_confidence, category, url) "
|
||||
"VALUES ($1, $2, $3, $4, $5, $6, $7, $8, $9, $10)",
|
||||
summary_id, kind, headline, importance, location_name,
|
||||
lat, lon, location_confidence, category, url,
|
||||
"INSERT INTO article_summaries (summary_text, batch_timestamp) "
|
||||
"VALUES ($1, $2)",
|
||||
text, datetime.fromisoformat(ts),
|
||||
)
|
||||
finally:
|
||||
await conn.close()
|
||||
|
|
@ -164,101 +131,12 @@ def test_api_news_summaries_contract(clean_news):
|
|||
assert isinstance(body, list)
|
||||
assert len(body) == 1
|
||||
s = body[0]
|
||||
assert set(s.keys()) == {"id", "summary_text", "batch_timestamp", "model", "kind"}
|
||||
assert set(s.keys()) == {"id", "summary_text", "batch_timestamp"}
|
||||
assert s["summary_text"] == "master summary markdown…"
|
||||
assert s["batch_timestamp"].startswith("2026-08-24T18:05")
|
||||
assert s["kind"] is None
|
||||
|
||||
|
||||
@requires_db
|
||||
def test_api_news_summaries_kind_filter(clean_news):
|
||||
_seed_summary("interval brief", "2026-08-28T22:05:00+00:00", kind="interval")
|
||||
_seed_summary("daily recap", "2026-08-28T03:00:00+00:00", kind="daily_recap")
|
||||
recap = _get("/api/news/summaries?kind=daily_recap").json()
|
||||
assert len(recap) == 1
|
||||
assert recap[0]["summary_text"] == "daily recap"
|
||||
assert recap[0]["kind"] == "daily_recap"
|
||||
assert _get("/api/news/summaries?kind=nope").status_code == 422
|
||||
|
||||
|
||||
@requires_db
|
||||
def test_api_news_empty(clean_news):
|
||||
assert _get("/api/news").json() == []
|
||||
assert _get("/api/news/summaries").json() == []
|
||||
assert _get("/api/news/ticker").json() == []
|
||||
assert _get("/api/news/map").json() == []
|
||||
|
||||
|
||||
TICKER_KEYS = {"id", "headline", "importance", "location_name", "url", "created_at"}
|
||||
MAP_KEYS = {
|
||||
"id", "headline", "importance", "location_name", "lat", "lon",
|
||||
"location_confidence", "category", "url", "created_at",
|
||||
}
|
||||
|
||||
|
||||
def _seed_flagged_items() -> None:
|
||||
sid = _seed_summary("batch brief", "2026-08-27T18:05:00+00:00", "Hermes-4.3-36B")
|
||||
_seed_news_item(
|
||||
sid, "ticker", "Critical ticker", "critical",
|
||||
location_name="Kyiv", url="https://example.com/ticker",
|
||||
)
|
||||
_seed_news_item(
|
||||
sid, "ticker", "Low ticker", "low",
|
||||
location_name="Somewhere", url="https://example.com/low",
|
||||
)
|
||||
_seed_news_item(
|
||||
sid, "map", "Critical map", "critical",
|
||||
location_name="Taipei", lat=25.03, lon=121.56,
|
||||
location_confidence="high", category="conflict",
|
||||
url="https://example.com/map",
|
||||
)
|
||||
|
||||
|
||||
@requires_db
|
||||
def test_api_news_ticker_returns_only_flagged(clean_news):
|
||||
_seed_flagged_items()
|
||||
resp = _get("/api/news/ticker")
|
||||
assert resp.status_code == 200
|
||||
body = resp.json()
|
||||
assert isinstance(body, list)
|
||||
assert len(body) == 1
|
||||
item = body[0]
|
||||
assert set(item.keys()) == TICKER_KEYS
|
||||
assert item["headline"] == "Critical ticker"
|
||||
assert item["importance"] == "critical"
|
||||
assert item["location_name"] == "Kyiv"
|
||||
assert item["url"] == "https://example.com/ticker"
|
||||
|
||||
|
||||
@requires_db
|
||||
def test_api_news_ticker_falls_back_to_lesser_when_nothing_flagged(clean_news):
|
||||
sid = _seed_summary("quiet brief", "2026-08-27T18:05:00+00:00", "Hermes-4.3-36B")
|
||||
_seed_news_item(
|
||||
sid, "ticker", "Shop theft downtown", "low",
|
||||
location_name="Raleigh", url="https://example.com/theft",
|
||||
)
|
||||
resp = _get("/api/news/ticker")
|
||||
assert resp.status_code == 200
|
||||
body = resp.json()
|
||||
assert len(body) == 1
|
||||
assert body[0]["headline"] == "Shop theft downtown"
|
||||
assert body[0]["importance"] == "low"
|
||||
|
||||
|
||||
@requires_db
|
||||
def test_api_news_map_returns_only_flagged_with_coords(clean_news):
|
||||
_seed_flagged_items()
|
||||
resp = _get("/api/news/map")
|
||||
assert resp.status_code == 200
|
||||
body = resp.json()
|
||||
assert isinstance(body, list)
|
||||
assert len(body) == 1
|
||||
item = body[0]
|
||||
assert set(item.keys()) == MAP_KEYS
|
||||
assert item["headline"] == "Critical map"
|
||||
assert item["importance"] == "critical"
|
||||
assert item["lat"] == 25.03
|
||||
assert item["lon"] == 121.56
|
||||
assert item["location_confidence"] == "high"
|
||||
assert item["category"] == "conflict"
|
||||
assert _get("/api/news/map?bbox=1,2,3").status_code == 422
|
||||
|
|
|
|||
|
|
@ -1,127 +0,0 @@
|
|||
"""GET /api/place — Nominatim reverse proxy (60s cache, 500 keys, 1 req/s)."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import asyncio
|
||||
|
||||
import httpx
|
||||
import pytest
|
||||
|
||||
from main import app
|
||||
from place import cache_key, place_cache, slim_place
|
||||
|
||||
BASE = "http://test"
|
||||
|
||||
SAMPLE = {
|
||||
"display_name": "Raleigh, Wake County, North Carolina, United States",
|
||||
"name": "Raleigh",
|
||||
"osm_type": "relation",
|
||||
"osm_id": 123,
|
||||
"address": {
|
||||
"city": "Raleigh",
|
||||
"state": "North Carolina",
|
||||
"country": "United States",
|
||||
"country_code": "us",
|
||||
"tourism": "ignore-me",
|
||||
},
|
||||
}
|
||||
|
||||
|
||||
class _FakeResp:
|
||||
def __init__(self, payload, status=200):
|
||||
self._payload = payload
|
||||
self.status_code = status
|
||||
|
||||
def raise_for_status(self):
|
||||
if self.status_code >= 400:
|
||||
req = httpx.Request("GET", "https://nominatim.openstreetmap.org/reverse")
|
||||
raise httpx.HTTPStatusError(
|
||||
"upstream", request=req,
|
||||
response=httpx.Response(self.status_code, request=req),
|
||||
)
|
||||
|
||||
def json(self):
|
||||
return self._payload
|
||||
|
||||
|
||||
class _FakeNominatim:
|
||||
calls: list[dict] = []
|
||||
|
||||
def __init__(self, *args, **kwargs):
|
||||
pass
|
||||
|
||||
async def __aenter__(self):
|
||||
return self
|
||||
|
||||
async def __aexit__(self, *args):
|
||||
return False
|
||||
|
||||
async def get(self, url, params=None, headers=None):
|
||||
_FakeNominatim.calls.append({"url": url, "params": params, "headers": headers})
|
||||
return _FakeResp(SAMPLE)
|
||||
|
||||
|
||||
def _nominatim_client(**kwargs):
|
||||
return _FakeNominatim()
|
||||
|
||||
|
||||
async def _get(path: str) -> httpx.Response:
|
||||
transport = httpx.ASGITransport(app=app)
|
||||
async with httpx.AsyncClient(transport=transport, base_url=BASE) as client:
|
||||
return await client.get(path)
|
||||
|
||||
|
||||
@pytest.fixture(autouse=True)
|
||||
def _reset_place(monkeypatch):
|
||||
place_cache.clear()
|
||||
_FakeNominatim.calls = []
|
||||
monkeypatch.setattr("place._http_client", _nominatim_client)
|
||||
monkeypatch.setattr("place.NOMINATIM_MIN_INTERVAL", 0.0)
|
||||
monkeypatch.setattr("place._last_req", 0.0)
|
||||
yield
|
||||
place_cache.clear()
|
||||
|
||||
|
||||
def test_slim_place_keeps_address_subset():
|
||||
body = slim_place(35.78, -78.64, SAMPLE)
|
||||
assert body["display_name"].startswith("Raleigh")
|
||||
assert body["name"] == "Raleigh"
|
||||
assert body["address"]["city"] == "Raleigh"
|
||||
assert "tourism" not in body["address"]
|
||||
assert body["attribution"].startswith("© OpenStreetMap")
|
||||
|
||||
|
||||
def test_cache_key_quantizes_to_4_decimals():
|
||||
assert cache_key(35.77961, -78.63821) == cache_key(35.77964, -78.63819)
|
||||
|
||||
|
||||
def test_place_requires_lat_lon():
|
||||
resp = asyncio.run(_get("/api/place"))
|
||||
assert resp.status_code == 422
|
||||
|
||||
|
||||
def test_place_rejects_out_of_range():
|
||||
assert asyncio.run(_get("/api/place?lat=99&lon=0")).status_code == 422
|
||||
assert asyncio.run(_get("/api/place?lat=0&lon=200")).status_code == 422
|
||||
|
||||
|
||||
def test_place_reverse_and_cache():
|
||||
r1 = asyncio.run(_get("/api/place?lat=35.7796&lon=-78.6382"))
|
||||
assert r1.status_code == 200
|
||||
body = r1.json()
|
||||
assert body["display_name"].startswith("Raleigh")
|
||||
assert body["lat"] == pytest.approx(35.7796, abs=0.001)
|
||||
assert "max-age=60" in (r1.headers.get("cache-control") or "").lower()
|
||||
assert len(_FakeNominatim.calls) == 1
|
||||
ua = _FakeNominatim.calls[0]["headers"]["User-Agent"]
|
||||
assert "osint-dashboard" in ua.lower() or "@" in ua
|
||||
r2 = asyncio.run(_get("/api/place?lat=35.77961&lon=-78.63821"))
|
||||
assert r2.status_code == 200
|
||||
assert len(_FakeNominatim.calls) == 1 # cache hit, same 4-decimal key
|
||||
|
||||
|
||||
def test_place_cache_cap_500():
|
||||
from cachetools import TTLCache
|
||||
assert isinstance(place_cache, TTLCache)
|
||||
assert place_cache.maxsize == 500
|
||||
assert place_cache.ttl == 60
|
||||
|
|
@ -1,172 +0,0 @@
|
|||
"""API tests for GET/PUT /api/settings and GET /api/news/models."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import asyncio
|
||||
import os
|
||||
|
||||
import asyncpg
|
||||
import httpx
|
||||
import pytest
|
||||
|
||||
from conftest import requires_db
|
||||
|
||||
from main import app
|
||||
|
||||
BASE = "http://test"
|
||||
|
||||
|
||||
def _conn_kwargs() -> dict:
|
||||
return {
|
||||
"host": os.environ["DB_HOST"],
|
||||
"port": int(os.environ["DB_PORT"]),
|
||||
"user": os.environ["DB_USER"],
|
||||
"password": os.environ["DB_PASSWORD"],
|
||||
"database": os.environ["DB_NAME"],
|
||||
}
|
||||
|
||||
|
||||
def _truncate_settings() -> None:
|
||||
async def run():
|
||||
conn = await asyncpg.connect(**_conn_kwargs())
|
||||
try:
|
||||
await conn.execute("DROP TABLE IF EXISTS app_settings")
|
||||
finally:
|
||||
await conn.close()
|
||||
|
||||
asyncio.run(run())
|
||||
|
||||
|
||||
@pytest.fixture()
|
||||
def clean_settings():
|
||||
import settings_store
|
||||
settings_store._ensured = False
|
||||
_truncate_settings()
|
||||
yield
|
||||
settings_store._ensured = False
|
||||
_truncate_settings()
|
||||
|
||||
|
||||
def _request(method: str, path: str, json: dict | None = None) -> httpx.Response:
|
||||
async def _run() -> httpx.Response:
|
||||
transport = httpx.ASGITransport(app=app)
|
||||
async with httpx.AsyncClient(transport=transport, base_url=BASE) as client:
|
||||
return await client.request(method, path, json=json)
|
||||
|
||||
return asyncio.run(_run())
|
||||
|
||||
|
||||
def _get(path: str) -> httpx.Response:
|
||||
return _request("GET", path)
|
||||
|
||||
|
||||
def _put(path: str, json: dict) -> httpx.Response:
|
||||
return _request("PUT", path, json=json)
|
||||
|
||||
|
||||
def test_get_news_models_without_key_returns_fallback(monkeypatch):
|
||||
monkeypatch.delenv("NOUS_API_KEY", raising=False)
|
||||
monkeypatch.setattr("keystore.get_api_key", _missing_key)
|
||||
|
||||
resp = _get("/api/news/models")
|
||||
assert resp.status_code == 200
|
||||
body = resp.json()
|
||||
assert body["source"] == "fallback"
|
||||
ids = [m["id"] for m in body["models"]]
|
||||
assert "Hermes-4.3-36B" in ids
|
||||
from settings_store import FALLBACK_MODELS
|
||||
assert ids == FALLBACK_MODELS
|
||||
|
||||
|
||||
async def _missing_key(name: str):
|
||||
return None
|
||||
|
||||
|
||||
async def _present_key(name: str):
|
||||
return "test-nous-api-key-1234"
|
||||
|
||||
|
||||
def test_get_news_models_live_from_upstream(monkeypatch):
|
||||
monkeypatch.setattr("keystore.get_api_key", _present_key)
|
||||
monkeypatch.setenv("NOUS_BASE_URL", "https://inference-api.nousresearch.com/v1")
|
||||
monkeypatch.delenv("OSINT_USER_AGENT", raising=False)
|
||||
|
||||
captured: dict = {}
|
||||
|
||||
class FakeResponse:
|
||||
status_code = 200
|
||||
|
||||
def raise_for_status(self):
|
||||
return None
|
||||
|
||||
def json(self):
|
||||
return {"data": [{"id": "live-model-a"}, {"id": "Hermes-4.3-36B"}]}
|
||||
|
||||
async def fake_http_get(url, *, headers, timeout):
|
||||
captured["url"] = url
|
||||
captured["headers"] = headers
|
||||
captured["timeout"] = timeout
|
||||
return FakeResponse()
|
||||
|
||||
import settings_store
|
||||
monkeypatch.setattr(settings_store, "_http_get", fake_http_get)
|
||||
settings_store._models_cache = None
|
||||
|
||||
resp = _get("/api/news/models")
|
||||
assert resp.status_code == 200
|
||||
body = resp.json()
|
||||
assert body["source"] == "live"
|
||||
ids = [m["id"] for m in body["models"]]
|
||||
assert ids == ["live-model-a", "Hermes-4.3-36B"]
|
||||
assert captured["url"] == "https://inference-api.nousresearch.com/v1/models"
|
||||
assert captured["headers"]["User-Agent"] == "osint-dashboard-news-summarizer"
|
||||
assert captured["headers"]["Authorization"] == "Bearer test-nous-api-key-1234"
|
||||
timeout = captured["timeout"]
|
||||
assert timeout == 8 or getattr(timeout, "read", timeout) == 8 or float(timeout) == 8.0
|
||||
|
||||
|
||||
def test_get_news_models_upstream_failure_returns_fallback(monkeypatch):
|
||||
monkeypatch.setattr("keystore.get_api_key", _present_key)
|
||||
|
||||
async def boom_http_get(url, *, headers, timeout):
|
||||
raise httpx.ConnectError("upstream down")
|
||||
|
||||
import settings_store
|
||||
monkeypatch.setattr(settings_store, "_http_get", boom_http_get)
|
||||
settings_store._models_cache = None
|
||||
|
||||
resp = _get("/api/news/models")
|
||||
assert resp.status_code == 200
|
||||
body = resp.json()
|
||||
assert body["source"] == "fallback"
|
||||
ids = [m["id"] for m in body["models"]]
|
||||
assert "Hermes-4.3-36B" in ids
|
||||
|
||||
|
||||
def test_put_settings_empty_returns_422():
|
||||
resp = _put("/api/settings", {"summary_model": ""})
|
||||
assert resp.status_code == 422
|
||||
|
||||
|
||||
def test_put_settings_whitespace_only_returns_422():
|
||||
resp = _put("/api/settings", {"summary_model": " "})
|
||||
assert resp.status_code == 422
|
||||
|
||||
|
||||
@requires_db
|
||||
def test_put_settings_round_trip(clean_settings, monkeypatch):
|
||||
monkeypatch.delenv("SUMMARY_MODEL", raising=False)
|
||||
monkeypatch.delenv("NOUS_BASE_URL", raising=False)
|
||||
|
||||
put_resp = _put("/api/settings", {"summary_model": "google/gemini-2.5-flash"})
|
||||
assert put_resp.status_code == 200
|
||||
put_body = put_resp.json()
|
||||
assert put_body["summary_model"] == "google/gemini-2.5-flash"
|
||||
assert put_body["nous_base_url"] == "https://inference-api.nousresearch.com/v1"
|
||||
|
||||
get_resp = _get("/api/settings")
|
||||
assert get_resp.status_code == 200
|
||||
get_body = get_resp.json()
|
||||
assert get_body["summary_model"] == "google/gemini-2.5-flash"
|
||||
assert get_body["nous_base_url"] == "https://inference-api.nousresearch.com/v1"
|
||||
assert set(get_body.keys()) == {"summary_model", "nous_base_url"}
|
||||
|
|
@ -1,87 +0,0 @@
|
|||
"""GET /api/stats HUD counter contract (counts only, small, never 500)."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import asyncio
|
||||
import re
|
||||
from datetime import timezone
|
||||
|
||||
import httpx
|
||||
|
||||
from main import app, _stats_counts
|
||||
|
||||
BASE = "http://test"
|
||||
|
||||
EXPECTED_KEYS = ("aircraft", "vessels", "trains", "cameras",
|
||||
"fires", "quakes", "alerts", "timestamp")
|
||||
|
||||
|
||||
async def _get(path: str) -> httpx.Response:
|
||||
transport = httpx.ASGITransport(app=app)
|
||||
async with httpx.AsyncClient(transport=transport, base_url=BASE) as client:
|
||||
return await client.get(path)
|
||||
|
||||
|
||||
def test_stats_200_all_keys_present():
|
||||
resp = asyncio.run(_get("/api/stats"))
|
||||
assert resp.status_code == 200
|
||||
body = resp.json()
|
||||
for key in EXPECTED_KEYS:
|
||||
assert key in body, f"missing key {key}"
|
||||
assert "max-age" in (resp.headers.get("cache-control") or "").lower()
|
||||
|
||||
|
||||
def test_stats_counters_are_ints():
|
||||
body = asyncio.run(_get("/api/stats")).json()
|
||||
for key in EXPECTED_KEYS:
|
||||
if key == "timestamp":
|
||||
continue
|
||||
assert isinstance(body[key], int), f"{key} is not an int: {body[key]!r}"
|
||||
|
||||
|
||||
def test_stats_timestamp_is_iso8601_z():
|
||||
body = asyncio.run(_get("/api/stats")).json()
|
||||
ts = body["timestamp"]
|
||||
# ISO8601 with a trailing Z (we normalize +00:00 -> Z).
|
||||
assert isinstance(ts, str) and ts.endswith("Z")
|
||||
assert re.match(r"^\d{4}-\d{2}-\d{2}T\d{2}:\d{2}:\d{2}", ts)
|
||||
|
||||
|
||||
def test_stats_payload_is_tiny():
|
||||
resp = asyncio.run(_get("/api/stats"))
|
||||
assert len(resp.content) < 2048, "stats payload must be counts-only, not GeoJSON"
|
||||
|
||||
|
||||
def test_stats_counts_reflect_last_known(monkeypatch):
|
||||
"""aircraft/vessels/trains/alerts come from in-memory last-known state."""
|
||||
import live_layers
|
||||
|
||||
monkeypatch.setattr(live_layers, "aircraft_last_known", {str(i): {} for i in range(7)})
|
||||
monkeypatch.setattr(live_layers, "vessel_last_known", {str(i): {} for i in range(3)})
|
||||
monkeypatch.setattr(live_layers, "train_count", 11)
|
||||
monkeypatch.setattr(live_layers, "nws_alert_count", 5)
|
||||
|
||||
# _stats_counts imports the dicts/counters inside the function from live_layers,
|
||||
# so monkeypatching the module attributes is what it observes.
|
||||
from main import _stats_counts as fn
|
||||
|
||||
body = asyncio.run(fn())
|
||||
assert body["aircraft"] == 7
|
||||
assert body["vessels"] == 3
|
||||
assert body["trains"] == 11
|
||||
assert body["alerts"] == 5
|
||||
|
||||
|
||||
def test_stats_db_failure_degrades_to_zero(monkeypatch):
|
||||
"""A down DB yields zeros for the SQL-backed counters, never a 500."""
|
||||
# Make the session factory raise synchronously so the try/except in
|
||||
# _stats_counts degrades the SQL counters to zero (no dangling coroutine).
|
||||
def _raise(*args, **kwargs):
|
||||
raise RuntimeError("db down")
|
||||
|
||||
monkeypatch.setattr("main.async_session", _raise)
|
||||
body = asyncio.run(_stats_counts())
|
||||
assert body["cameras"] == 0
|
||||
assert body["fires"] == 0
|
||||
assert body["quakes"] == 0
|
||||
assert isinstance(body["timestamp"], str)
|
||||
|
|
@ -1,55 +0,0 @@
|
|||
"""ffmpeg snapshots stay off the request path (asyncio.create_task)."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import asyncio
|
||||
|
||||
import bg_jobs
|
||||
|
||||
|
||||
def test_bg_jobs_has_no_pps_cap():
|
||||
assert not any(name.endswith("_PPS_CAP") for name in dir(bg_jobs))
|
||||
|
||||
|
||||
def test_camera_preview_has_no_public_feed_probe():
|
||||
import camera_preview
|
||||
|
||||
assert not hasattr(camera_preview, "probe_public_feed")
|
||||
assert not hasattr(camera_preview, "_http_feed_url")
|
||||
|
||||
|
||||
def test_ingest_routes_exclude_active_discovery():
|
||||
from main import app
|
||||
|
||||
ingest = [
|
||||
getattr(r, "path", "")
|
||||
for r in app.routes
|
||||
if getattr(r, "path", "").startswith("/api/ingest/")
|
||||
]
|
||||
assert "/api/ingest/fires" in ingest
|
||||
assert all("scan" not in path for path in ingest)
|
||||
|
||||
|
||||
def test_schedule_ffmpeg_snapshot_is_a_task_not_inline(monkeypatch):
|
||||
calls = {"n": 0}
|
||||
|
||||
async def fake_grab(url, timeout=8.0):
|
||||
calls["n"] += 1
|
||||
await asyncio.sleep(5)
|
||||
return b"\xff\xd8fakejpeg"
|
||||
|
||||
monkeypatch.setattr(bg_jobs, "_ffmpeg_grab", fake_grab)
|
||||
bg_jobs._ffmpeg_tasks.clear()
|
||||
bg_jobs._ffmpeg_cache.clear()
|
||||
|
||||
async def run():
|
||||
task = bg_jobs.schedule_ffmpeg_snapshot("rtsp://10.0.0.1/")
|
||||
assert isinstance(task, asyncio.Task)
|
||||
assert not task.done()
|
||||
task.cancel()
|
||||
try:
|
||||
await task
|
||||
except (asyncio.CancelledError, Exception):
|
||||
pass
|
||||
|
||||
asyncio.run(run())
|
||||
|
|
@ -1,21 +0,0 @@
|
|||
"""Timeline bucket_hours + static Cache-Control."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from pathlib import Path
|
||||
|
||||
ROOT = Path(__file__).resolve().parent.parent
|
||||
|
||||
|
||||
def test_timeline_uses_bucket_hours():
|
||||
src = (ROOT / "app/main.py").read_text()
|
||||
fn = src.split("async def get_timeline")[1].split("async def sentiment_by_source")[0]
|
||||
assert "bucket_hours" in fn
|
||||
assert "date_trunc('hour'" not in fn or "bucket" in fn.lower()
|
||||
# Must not ignore the query param.
|
||||
assert ":bucket" in fn or "bucket_hours" in fn.split("text(")[1][:800]
|
||||
|
||||
|
||||
def test_static_vendor_cache_control():
|
||||
src = (ROOT / "app/main.py").read_text()
|
||||
assert "max-age=31536000" in src or "immutable" in src.lower()
|
||||
|
|
@ -1,15 +0,0 @@
|
|||
"""Camera list is slim (no URLs). Popup must fetch GET /api/cameras/{id}."""
|
||||
|
||||
from pathlib import Path
|
||||
|
||||
HTML = Path(__file__).resolve().parents[1] / "app/static/index.html"
|
||||
|
||||
|
||||
def test_popupopen_fetches_camera_detail_row():
|
||||
html = HTML.read_text()
|
||||
start = html.index("map.on('popupopen'")
|
||||
end = html.index("map.on('popupclose'")
|
||||
block = html[start:end]
|
||||
assert "fetch(" in block
|
||||
assert "/api/cameras/" in block
|
||||
assert "camPopupHtml(" in block
|
||||
|
|
@ -1,122 +0,0 @@
|
|||
"""Tests for the chokepoint preset catalog + vessels ``src=`` filter.
|
||||
|
||||
- Span: every catalog box passes VesselAPI's ``|dLat|+|dLon| <= 4`` validator.
|
||||
- Catalog: ``GET /api/map/chokepoints`` returns 200 with the documented shape.
|
||||
- Vessels filter: ``GET /api/vessels?src=`` narrows the union store by provider.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import asyncio
|
||||
|
||||
import httpx
|
||||
|
||||
from chokepoints import chokepoints
|
||||
from live_layers import fetch_vessels, vessel_last_known
|
||||
from main import app
|
||||
from vesselapi import validate_bbox_span
|
||||
|
||||
BASE = "http://test"
|
||||
|
||||
|
||||
async def _get(path: str) -> httpx.Response:
|
||||
transport = httpx.ASGITransport(app=app)
|
||||
async with httpx.AsyncClient(transport=transport, base_url=BASE) as client:
|
||||
return await client.get(path)
|
||||
|
||||
|
||||
# ── Span validation (VesselAPI rule) ──────────────────────────────────────
|
||||
|
||||
def test_all_catalog_boxes_within_span() -> None:
|
||||
for preset in chokepoints():
|
||||
minlat, minlon, maxlat, maxlon = (float(p) for p in preset["bbox"].split(","))
|
||||
dlat = abs(maxlat - minlat)
|
||||
dlon = abs(maxlon - minlon)
|
||||
assert dlat + dlon <= 4.0, preset["id"]
|
||||
validate_bbox_span(minlat, minlon, maxlat, maxlon) # no raise
|
||||
|
||||
|
||||
# ── Catalog API contract ──────────────────────────────────────────────────
|
||||
|
||||
def test_chokepoints_catalog_shape() -> None:
|
||||
resp = asyncio.run(_get("/api/map/chokepoints"))
|
||||
assert resp.status_code == 200
|
||||
body = resp.json()
|
||||
assert set(body) == {"chokepoints"}
|
||||
rows = body["chokepoints"]
|
||||
assert [r["id"] for r in rows] == [
|
||||
"hormuz", "bab_el_mandeb", "suez", "malacca", "taiwan",
|
||||
]
|
||||
for r in rows:
|
||||
assert set(r) == {"id", "title", "bbox", "center", "zoom", "vesselapi"}
|
||||
assert isinstance(r["center"], list) and len(r["center"]) == 2
|
||||
assert r["zoom"] == 9
|
||||
assert isinstance(r["vesselapi"], bool)
|
||||
# bbox is minlat,minlon,maxlat,maxlon
|
||||
minlat, minlon, maxlat, maxlon = (float(p) for p in r["bbox"].split(","))
|
||||
assert minlat < maxlat and minlon < maxlon
|
||||
|
||||
|
||||
def test_only_hormuz_is_vesselapi() -> None:
|
||||
rows = chokepoints()
|
||||
by_id = {r["id"]: r for r in rows}
|
||||
assert by_id["hormuz"]["vesselapi"] is True
|
||||
for cid in ("bab_el_mandeb", "suez", "malacca", "taiwan"):
|
||||
assert by_id[cid]["vesselapi"] is False
|
||||
|
||||
|
||||
# ── Vessels src= filter (mocked store) ────────────────────────────────────
|
||||
|
||||
def _seed_store() -> None:
|
||||
vessel_last_known.clear()
|
||||
vessel_last_known["422050100"] = {
|
||||
"id": "422050100", "lat": 26.5, "lon": 56.3, "label": "HORMUZ STAR",
|
||||
"extra": {"src": "vesselapi", "mmsi": "422050100"},
|
||||
}
|
||||
vessel_last_known["366001230"] = {
|
||||
"id": "366001230", "lat": 35.0, "lon": -79.0, "label": "CONUS SHIP",
|
||||
"extra": {"src": "aisstream", "mmsi": "366001230"},
|
||||
}
|
||||
vessel_last_known["366001231"] = {
|
||||
"id": "366001231", "lat": 36.0, "lon": -78.0, "label": "CONUS SHIP 2",
|
||||
"extra": {"src": "aisstream", "mmsi": "366001231"},
|
||||
}
|
||||
|
||||
|
||||
def test_fetch_vessels_src_filters() -> None:
|
||||
_seed_store()
|
||||
assert {v["id"] for v in asyncio.run(fetch_vessels(None, src="vesselapi"))} == {"422050100"}
|
||||
assert {v["id"] for v in asyncio.run(fetch_vessels(None, src="aisstream"))} == {
|
||||
"366001230", "366001231",
|
||||
}
|
||||
assert len(asyncio.run(fetch_vessels(None, src="all"))) == 3
|
||||
assert len(asyncio.run(fetch_vessels(None))) == 3 # default all
|
||||
|
||||
|
||||
def test_vessels_src_query_param(monkeypatch) -> None:
|
||||
_seed_store()
|
||||
|
||||
async def _fake_fetch(bbox, limit, src=None):
|
||||
rows = [
|
||||
{"id": k, **{kk: v[kk] for kk in ("lat", "lon", "label", "extra")}}
|
||||
for k, v in vessel_last_known.items()
|
||||
]
|
||||
if src and src != "all":
|
||||
rows = [r for r in rows if (r.get("extra") or {}).get("src") == src]
|
||||
return rows
|
||||
|
||||
monkeypatch.setattr("main.fetch_vessels", _fake_fetch)
|
||||
|
||||
body = asyncio.run(_get("/api/vessels?src=vesselapi")).json()
|
||||
assert [r["id"] for r in body] == ["422050100"]
|
||||
|
||||
body = asyncio.run(_get("/api/vessels?src=aisstream")).json()
|
||||
assert {r["id"] for r in body} == {"366001230", "366001231"}
|
||||
|
||||
body = asyncio.run(_get("/api/vessels?src=all")).json()
|
||||
assert len(body) == 3
|
||||
|
||||
|
||||
def test_vessels_src_rejects_bad_value() -> None:
|
||||
resp = asyncio.run(_get("/api/vessels?src=marine-traffic"))
|
||||
assert resp.status_code == 422
|
||||
|
|
@ -1,132 +0,0 @@
|
|||
"""GET /api/conflicts — curated conflict-zone catalog + event-count roll-up.
|
||||
|
||||
No outbound HTTP: event counts come from geocoded rows already (or not) in the
|
||||
DB, and the API tests monkeypatch ``main._fetch_geocoded_points`` so no database
|
||||
is required for the contract checks.
|
||||
"""
|
||||
|
||||
from datetime import datetime, timezone
|
||||
|
||||
import httpx
|
||||
|
||||
from conflicts import SEVERITIES, conflict_zones, zone_event_stats
|
||||
from live_layers import overlay_catalog
|
||||
from main import app
|
||||
|
||||
BASE = "http://test"
|
||||
|
||||
|
||||
def _get(path: str, monkeypatch=None, points=None) -> httpx.Response:
|
||||
import asyncio
|
||||
|
||||
async def run() -> httpx.Response:
|
||||
if monkeypatch is not None:
|
||||
async def fake():
|
||||
return points or []
|
||||
|
||||
monkeypatch.setattr("main._fetch_geocoded_points", fake)
|
||||
transport = httpx.ASGITransport(app=app)
|
||||
async with httpx.AsyncClient(transport=transport, base_url=BASE) as client:
|
||||
return await client.get(path)
|
||||
|
||||
return asyncio.run(run())
|
||||
|
||||
|
||||
# ── Catalog shape ──────────────────────────────────────────────────────
|
||||
|
||||
|
||||
def test_catalog_length():
|
||||
zones = conflict_zones()
|
||||
assert len(zones) == 13
|
||||
|
||||
|
||||
def test_catalog_severity_enum():
|
||||
zones = conflict_zones()
|
||||
sevs = {z["severity"] for z in zones}
|
||||
assert sevs.issubset(SEVERITIES)
|
||||
# All three tiers are represented.
|
||||
assert sevs == SEVERITIES
|
||||
|
||||
|
||||
def test_catalog_fields_factual_and_complete():
|
||||
zones = conflict_zones()
|
||||
ids = [z["id"] for z in zones]
|
||||
assert len(set(ids)) == len(ids) # unique ids
|
||||
for z in zones:
|
||||
assert z["label"]
|
||||
assert z["description"].strip()
|
||||
assert -90.0 <= z["lat"] <= 90.0
|
||||
assert -180.0 <= z["lon"] <= 180.0
|
||||
# internal-only bbox is well-formed: (min_lat, min_lon, max_lat, max_lon)
|
||||
min_lat, min_lon, max_lat, max_lon = z["bbox"]
|
||||
assert min_lat <= max_lat and min_lon <= max_lon
|
||||
assert min_lat <= z["lat"] <= max_lat and min_lon <= z["lon"] <= max_lon
|
||||
|
||||
|
||||
def test_overlay_catalog_has_conflicts():
|
||||
entry = overlay_catalog()["conflicts"]
|
||||
assert entry["kind"] == "points"
|
||||
assert entry["endpoint"] == "/api/conflicts"
|
||||
|
||||
|
||||
# ── Pure counting ──────────────────────────────────────────────────────
|
||||
|
||||
TS1 = datetime(2026, 8, 30, 12, 0, tzinfo=timezone.utc)
|
||||
TS2 = datetime(2026, 8, 30, 13, 0, tzinfo=timezone.utc)
|
||||
|
||||
|
||||
def test_zone_event_stats_counts_and_picks_latest():
|
||||
bbox = (40.0, 20.0, 52.0, 40.0) # roughly Ukraine
|
||||
points = [
|
||||
(50.45, 30.52, TS1), # inside
|
||||
(48.0, 25.0, TS2), # inside, later
|
||||
(0.0, -60.0, TS1), # outside
|
||||
(15.0, 45.0, TS2), # outside (lat ok, lon out)
|
||||
]
|
||||
count, latest = zone_event_stats(points, bbox)
|
||||
assert count == 2
|
||||
assert latest == TS2
|
||||
|
||||
|
||||
def test_zone_event_stats_empty_bbox():
|
||||
count, latest = zone_event_stats([], (0.0, 0.0, 1.0, 1.0))
|
||||
assert count == 0
|
||||
assert latest is None
|
||||
|
||||
|
||||
# ── API contract (mocked map items, no DB) ─────────────────────────────
|
||||
|
||||
|
||||
def test_conflicts_returns_catalog_with_mocked_counts(monkeypatch):
|
||||
points = [
|
||||
(50.45, 30.52, TS1), # Ukraine
|
||||
(25.03, 121.56, TS2), # Taiwan Strait
|
||||
(0.0, -60.0, TS1), # nowhere
|
||||
]
|
||||
resp = _get("/api/conflicts", monkeypatch=monkeypatch, points=points)
|
||||
assert resp.status_code == 200
|
||||
body = resp.json()
|
||||
assert "zones" in body and "timestamp" in body
|
||||
by_id = {z["id"]: z for z in body["zones"]}
|
||||
assert len(body["zones"]) == 13
|
||||
|
||||
zone = by_id["ukraine"]
|
||||
assert zone["eventCount"] == 1
|
||||
assert zone["lastUpdated"] == TS1.isoformat().replace("+00:00", "Z")
|
||||
assert zone["severity"] == "war"
|
||||
|
||||
assert by_id["taiwan_strait"]["eventCount"] == 1
|
||||
assert by_id["gaza"]["eventCount"] == 0
|
||||
# exact per-zone key contract the frontend consumes
|
||||
assert set(zone.keys()) == {
|
||||
"id", "label", "severity", "lat", "lon",
|
||||
"description", "eventCount", "lastUpdated",
|
||||
}
|
||||
|
||||
|
||||
def test_conflicts_empty_db_yields_zero_counts(monkeypatch):
|
||||
resp = _get("/api/conflicts", monkeypatch=monkeypatch, points=[])
|
||||
assert resp.status_code == 200
|
||||
body = resp.json()
|
||||
assert all(z["eventCount"] == 0 for z in body["zones"])
|
||||
assert all(z["lastUpdated"] is None for z in body["zones"])
|
||||
|
|
@ -1,66 +0,0 @@
|
|||
"""Conflicts Leaflet overlay: default-off toggle, catalog fetch, no jitter."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from pathlib import Path
|
||||
|
||||
ROOT = Path(__file__).resolve().parent.parent
|
||||
HTML = (ROOT / "app/static/index.html").read_text()
|
||||
|
||||
|
||||
def _fn(name: str, until: str | None = None) -> str:
|
||||
chunk = HTML.split(f"function {name}", 1)[1]
|
||||
if until:
|
||||
chunk = chunk.split(until, 1)[0]
|
||||
return chunk
|
||||
|
||||
|
||||
def test_conflicts_toggle_default_off():
|
||||
assert 'id="lp-conflicts-on"' in HTML
|
||||
assert 'id="conflicts-layer"' in HTML
|
||||
assert "> Conflicts<" in HTML or "> Conflicts</" in HTML
|
||||
on = HTML.split('id="lp-conflicts-on"', 1)[1].split(">", 1)[0]
|
||||
assert "checked" not in on
|
||||
|
||||
|
||||
def test_conflicts_fetches_catalog_not_liveuamap():
|
||||
js = _fn("loadConflicts", "/* ═══════════════ INITIAL LOAD")
|
||||
assert "/api/conflicts" in js
|
||||
assert "liveuamap.com" not in HTML.lower()
|
||||
assert "Math.random" not in js
|
||||
assert "jitter" not in js.lower()
|
||||
|
||||
|
||||
def test_conflicts_not_refetched_on_moveend():
|
||||
refresh = HTML.split("function refreshLiveOverlays", 1)[1].split(
|
||||
"function addExtraAttrib", 1
|
||||
)[0]
|
||||
assert "loadConflicts" not in refresh
|
||||
assert "probeConflicts" not in refresh
|
||||
init = HTML.split("function initMap", 1)[1].split("function readMapPrefs", 1)[0]
|
||||
assert "probeConflicts()" in init
|
||||
assert "loadConflicts(true)" not in init
|
||||
assert "paintConflicts()" not in init
|
||||
|
||||
|
||||
def test_conflicts_hides_toggle_on_404():
|
||||
js = _fn("loadConflicts", "/* ═══════════════ INITIAL LOAD")
|
||||
assert "r.status === 404" in js
|
||||
assert "hideConflictsToggle()" in js
|
||||
hide = _fn("hideConflictsToggle", "function paintConflicts")
|
||||
assert "row.hidden = true" in hide
|
||||
assert "lp-conflicts-on" in hide
|
||||
|
||||
|
||||
def test_conflicts_popup_and_severity_colors():
|
||||
paint = _fn("paintConflicts", "async function probeConflicts")
|
||||
assert "z.label" in paint
|
||||
assert "z.description" in paint
|
||||
assert "eventCount" in paint
|
||||
assert "L.circleMarker" in paint
|
||||
assert "z.lat == null || z.lon == null" in paint
|
||||
assert "Number.isFinite(lat)" in paint
|
||||
color = _fn("conflictSeverityColor", "function hideConflictsToggle")
|
||||
assert "war" in color and "#ff2a6d" in color
|
||||
assert "high" in color and "#fb923c" in color
|
||||
assert "elevated" in color and "#facc15" in color
|
||||
|
|
@ -1,133 +0,0 @@
|
|||
"""Generic event ingest: idempotency, USGS ids, GDELT DOC URL."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import asyncio
|
||||
from datetime import datetime, timezone
|
||||
|
||||
|
||||
def test_event_dedup_key_prefers_url():
|
||||
from sources import event_dedup_key
|
||||
|
||||
assert event_dedup_key({"url": "https://earthquake.usgs.gov/earthquakes/eventpage/ci1"}) == (
|
||||
"https://earthquake.usgs.gov/earthquakes/eventpage/ci1"
|
||||
)
|
||||
assert event_dedup_key({"url": " "}) is None
|
||||
assert event_dedup_key({}) is None
|
||||
|
||||
|
||||
def test_usgs_feature_keeps_id_and_url():
|
||||
from sources import parse_usgs_feature
|
||||
|
||||
feature = {
|
||||
"id": "ci39818991",
|
||||
"properties": {
|
||||
"title": "M 2.1 - 5 km W of",
|
||||
"url": "https://earthquake.usgs.gov/earthquakes/eventpage/ci39818991",
|
||||
"place": "5 km W of",
|
||||
"mag": 2.1,
|
||||
"time": 1_700_000_000_000,
|
||||
},
|
||||
"geometry": {"coordinates": [-118.5, 34.1, 10.0]},
|
||||
}
|
||||
event = parse_usgs_feature(feature)
|
||||
assert event["url"] == "https://earthquake.usgs.gov/earthquakes/eventpage/ci39818991"
|
||||
assert event["raw"]["usgs_id"] == "ci39818991"
|
||||
assert event["location_lat"] == 34.1
|
||||
assert event["location_lon"] == -118.5
|
||||
assert event["source_type"] == "earthquake"
|
||||
|
||||
|
||||
def test_gdelt_uses_doc_api_and_query_param():
|
||||
from sources import GDELT_API, gdelt_params
|
||||
|
||||
assert GDELT_API == "https://api.gdeltproject.org/api/v2/doc/doc"
|
||||
params = gdelt_params(query="unrest", max_articles=50)
|
||||
assert params["query"] == "unrest"
|
||||
assert "search" not in params
|
||||
assert params["mode"] == "ArtList"
|
||||
assert params["format"] == "json"
|
||||
assert int(params["maxrecords"]) == 50
|
||||
|
||||
|
||||
def test_gdelt_default_query_when_empty():
|
||||
from sources import gdelt_params
|
||||
|
||||
params = gdelt_params(query="", max_articles=25)
|
||||
assert params["query"]
|
||||
assert "unrest" in params["query"].lower() or "cyber" in params["query"].lower()
|
||||
|
||||
|
||||
def test_parse_gdelt_articles_maps_doc_payload():
|
||||
from sources import parse_gdelt_articles
|
||||
|
||||
payload = {
|
||||
"articles": [
|
||||
{
|
||||
"url": "https://example.com/a",
|
||||
"title": "Outage",
|
||||
"seendate": "20240101T120000Z",
|
||||
"domain": "example.com",
|
||||
"language": "English",
|
||||
"sourcecountry": "US",
|
||||
}
|
||||
]
|
||||
}
|
||||
events = parse_gdelt_articles(payload)
|
||||
assert len(events) == 1
|
||||
assert events[0]["source_type"] == "gdel-t2"
|
||||
assert events[0]["url"] == "https://example.com/a"
|
||||
assert events[0]["title"] == "Outage"
|
||||
|
||||
|
||||
def test_ingest_event_skips_duplicate_url(monkeypatch):
|
||||
"""Second insert with the same url must not hit events_table.insert."""
|
||||
from ingestor import ingest_event
|
||||
|
||||
calls = {"insert": 0, "dedup": 0}
|
||||
|
||||
class _Result:
|
||||
rowcount = 1
|
||||
inserted_primary_key = ["evt-1"]
|
||||
|
||||
class _Session:
|
||||
async def execute(self, stmt):
|
||||
sql = str(stmt).lower()
|
||||
if "event_dedup" in sql or "on conflict" in sql:
|
||||
calls["dedup"] += 1
|
||||
self_result = _Result()
|
||||
if calls["dedup"] > 1:
|
||||
self_result.rowcount = 0
|
||||
return self_result
|
||||
calls["insert"] += 1
|
||||
return _Result()
|
||||
|
||||
async def commit(self):
|
||||
return None
|
||||
|
||||
async def __aenter__(self):
|
||||
return self
|
||||
|
||||
async def __aexit__(self, *a):
|
||||
return False
|
||||
|
||||
import ingestor
|
||||
|
||||
monkeypatch.setattr(ingestor, "async_session", lambda: _Session())
|
||||
|
||||
msg = {
|
||||
"source_type": "earthquake",
|
||||
"title": "M 2.1",
|
||||
"url": "https://earthquake.usgs.gov/earthquakes/eventpage/ci1",
|
||||
"source_timestamp": datetime(2026, 1, 1, tzinfo=timezone.utc).isoformat(),
|
||||
}
|
||||
|
||||
async def run():
|
||||
first = await ingest_event(msg)
|
||||
second = await ingest_event(msg)
|
||||
return first, second
|
||||
|
||||
first, second = asyncio.run(run())
|
||||
assert first is not None
|
||||
assert second is None
|
||||
assert calls["insert"] == 1
|
||||
|
|
@ -1,66 +0,0 @@
|
|||
"""WFIGS/FIRMS × firefighting ADS-B correlation within 20 miles."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from fire_aircraft import (
|
||||
FIREFIGHTER_ICAO,
|
||||
RADIUS_MILES,
|
||||
correlate_aircraft_to_fires,
|
||||
is_firefighter,
|
||||
)
|
||||
|
||||
|
||||
def _ac(hex_id, lat, lon, icao, **extra):
|
||||
return {
|
||||
"id": hex_id,
|
||||
"lat": lat,
|
||||
"lon": lon,
|
||||
"label": hex_id,
|
||||
"heading": 0,
|
||||
"speed": 120,
|
||||
"extra": {"type": icao, "hex": hex_id, **extra},
|
||||
}
|
||||
|
||||
|
||||
def _fire(name, lat, lon, **extra):
|
||||
return {
|
||||
"id": name,
|
||||
"lat": lat,
|
||||
"lon": lon,
|
||||
"label": name,
|
||||
"extra": extra,
|
||||
}
|
||||
|
||||
|
||||
def test_air_tractor_is_firefighter_airliner_is_not():
|
||||
assert is_firefighter(_ac("aaa", 0, 0, "AT802")) is True
|
||||
assert is_firefighter(_ac("bbb", 0, 0, "C130")) is True
|
||||
assert is_firefighter(_ac("ccc", 0, 0, "B738")) is False
|
||||
assert "AT802" in FIREFIGHTER_ICAO
|
||||
|
||||
|
||||
def test_flags_tanker_within_20_miles_of_fire():
|
||||
# ~10 miles north of a Piedmont fire
|
||||
fire = _fire("Jones Gap", 35.00, -82.00)
|
||||
tanker = _ac("acf001", 35.145, -82.00, "AT802")
|
||||
airliner = _ac("a0b738", 35.145, -82.00, "B738")
|
||||
far = _ac("acfar", 35.50, -82.00, "C130") # ~34 miles
|
||||
hits = correlate_aircraft_to_fires([fire], [tanker, airliner, far])
|
||||
assert RADIUS_MILES == 20.0
|
||||
assert len(hits) == 1
|
||||
h = hits[0]
|
||||
assert h["aircraft_hex"] == "acf001"
|
||||
assert h["fire_id"] == "Jones Gap"
|
||||
assert h["aircraft_type"] == "AT802"
|
||||
assert 0 < h["distance_mi"] <= 20.0
|
||||
|
||||
|
||||
def test_persist_shape_has_reload_fields():
|
||||
fire = _fire("Jones Gap", 35.00, -82.00, src="wfigs")
|
||||
tanker = _ac("acf001", 35.10, -82.00, "S64")
|
||||
hits = correlate_aircraft_to_fires([fire], [tanker])
|
||||
assert set(hits[0]).issuperset({
|
||||
"fire_id", "fire_lat", "fire_lon",
|
||||
"aircraft_hex", "aircraft_type", "aircraft_lat", "aircraft_lon",
|
||||
"distance_mi",
|
||||
})
|
||||
|
|
@ -85,183 +85,3 @@ def test_ingest_fire_row_drops_malformed(clean_fires):
|
|||
assert await ingest_fire_row(make_fire_msg(acq_time="garbage")) is False
|
||||
|
||||
asyncio.run(run())
|
||||
|
||||
|
||||
class _InsertSession:
|
||||
"""async_session stand-in: fire insert succeeds (rowcount=1)."""
|
||||
|
||||
def __init__(self):
|
||||
self.rowcount = 1
|
||||
|
||||
async def execute(self, *a, **k):
|
||||
return self
|
||||
|
||||
async def commit(self):
|
||||
return None
|
||||
|
||||
async def __aenter__(self):
|
||||
return self
|
||||
|
||||
async def __aexit__(self, *a):
|
||||
return False
|
||||
|
||||
|
||||
def test_ingest_fire_row_geofence_when_cache_empty(monkeypatch):
|
||||
"""Ingester process has empty geofence cache; still notify on FIRMS insert."""
|
||||
import geofence
|
||||
import ingestor
|
||||
import live_layers
|
||||
|
||||
geofence._cache.clear()
|
||||
live_layers.aircraft_last_known.clear()
|
||||
notified = []
|
||||
|
||||
async def fake_record(**kw):
|
||||
notified.append(kw)
|
||||
return 1
|
||||
|
||||
async def no_markers(*a, **k):
|
||||
return []
|
||||
|
||||
monkeypatch.setattr(ingestor, "async_session", _InsertSession)
|
||||
monkeypatch.setattr(geofence, "record_and_notify", fake_record)
|
||||
monkeypatch.setattr("tracks.recent_markers", no_markers)
|
||||
|
||||
assert asyncio.run(ingest_fire_row(make_fire_msg())) is True
|
||||
assert len(notified) == 1
|
||||
assert notified[0]["source_kind"] == "firms"
|
||||
assert notified[0]["lat"] == 39.45678
|
||||
assert notified[0]["lon"] == -121.12345
|
||||
|
||||
|
||||
def test_ingest_fire_row_correlates_from_hypertable_when_last_known_empty(monkeypatch):
|
||||
"""FIRMS ingester has no ADS-B last-known; still correlate from aircraft_positions."""
|
||||
import geofence
|
||||
import ingestor
|
||||
import live_layers
|
||||
|
||||
geofence._cache.clear()
|
||||
live_layers.aircraft_last_known.clear()
|
||||
correlated = []
|
||||
|
||||
async def fake_record(**kw):
|
||||
return 0
|
||||
|
||||
async def fake_recent(kind, limit=2000):
|
||||
assert kind == "aircraft"
|
||||
return [{
|
||||
"id": "acf001",
|
||||
"lat": 39.45,
|
||||
"lon": -121.12,
|
||||
"extra": {"type": "AT802"},
|
||||
}]
|
||||
|
||||
async def fake_corr(fires, aircraft):
|
||||
correlated.append((fires, aircraft))
|
||||
return aircraft
|
||||
|
||||
monkeypatch.setattr(ingestor, "async_session", _InsertSession)
|
||||
monkeypatch.setattr(geofence, "record_and_notify", fake_record)
|
||||
monkeypatch.setattr("tracks.recent_markers", fake_recent)
|
||||
monkeypatch.setattr("fire_aircraft.correlate_and_notify", fake_corr)
|
||||
|
||||
assert asyncio.run(ingest_fire_row(make_fire_msg())) is True
|
||||
assert len(correlated) == 1
|
||||
assert correlated[0][1][0]["id"] == "acf001"
|
||||
assert correlated[0][0][0]["lat"] == 39.45678
|
||||
|
||||
|
||||
def test_ingest_fire_rows_one_execute_one_commit(monkeypatch):
|
||||
"""93k FIRMS points must not be 93k commits."""
|
||||
from ingestor import ingest_fire_rows
|
||||
|
||||
class _Session:
|
||||
def __init__(self):
|
||||
self.executes = 0
|
||||
self.commits = 0
|
||||
self.rowcount = 3
|
||||
|
||||
async def execute(self, *a, **k):
|
||||
self.executes += 1
|
||||
return self
|
||||
|
||||
async def commit(self):
|
||||
self.commits += 1
|
||||
|
||||
async def __aenter__(self):
|
||||
return self
|
||||
|
||||
async def __aexit__(self, *a):
|
||||
return False
|
||||
|
||||
session = _Session()
|
||||
import ingestor
|
||||
monkeypatch.setattr(ingestor, "async_session", lambda: session)
|
||||
|
||||
msgs = [
|
||||
make_fire_msg(latitude=39.1 + i * 0.01, longitude=-121.1)
|
||||
for i in range(3)
|
||||
]
|
||||
|
||||
async def no_corr(*a, **k):
|
||||
return []
|
||||
|
||||
monkeypatch.setattr("fire_aircraft.correlate_and_notify", no_corr)
|
||||
monkeypatch.setattr("geofence.record_and_notify", no_corr)
|
||||
|
||||
inserted = asyncio.run(ingest_fire_rows(msgs))
|
||||
assert inserted == 3
|
||||
assert session.executes == 1
|
||||
assert session.commits == 1
|
||||
|
||||
|
||||
def test_fire_insert_chunk_stays_under_asyncpg_bind_limit():
|
||||
"""asyncpg caps bind params at 32767 — a 93k-row INSERT dies."""
|
||||
from ingestor import FIRE_INSERT_CHUNK, FIRE_ROW_BIND_PARAMS
|
||||
|
||||
assert FIRE_INSERT_CHUNK * FIRE_ROW_BIND_PARAMS < 32767
|
||||
assert FIRE_INSERT_CHUNK >= 500
|
||||
|
||||
|
||||
def test_ingest_fire_rows_chunks_when_over_limit(monkeypatch):
|
||||
from ingestor import ingest_fire_rows
|
||||
import ingestor
|
||||
|
||||
class _Session:
|
||||
def __init__(self):
|
||||
self.executes = 0
|
||||
self.commits = 0
|
||||
self.rowcount = 0
|
||||
|
||||
async def execute(self, *a, **k):
|
||||
self.executes += 1
|
||||
self.rowcount = 2 if self.executes < 3 else 1
|
||||
return self
|
||||
|
||||
async def commit(self):
|
||||
self.commits += 1
|
||||
|
||||
async def __aenter__(self):
|
||||
return self
|
||||
|
||||
async def __aexit__(self, *a):
|
||||
return False
|
||||
|
||||
session = _Session()
|
||||
monkeypatch.setattr(ingestor, "async_session", lambda: session)
|
||||
monkeypatch.setattr(ingestor, "FIRE_INSERT_CHUNK", 2)
|
||||
|
||||
async def no_corr(*a, **k):
|
||||
return []
|
||||
|
||||
monkeypatch.setattr("fire_aircraft.correlate_and_notify", no_corr)
|
||||
monkeypatch.setattr("geofence.record_and_notify", no_corr)
|
||||
|
||||
msgs = [
|
||||
make_fire_msg(latitude=39.1 + i * 0.01, longitude=-121.1)
|
||||
for i in range(5)
|
||||
]
|
||||
inserted = asyncio.run(ingest_fire_rows(msgs))
|
||||
assert inserted == 5
|
||||
assert session.executes == 3
|
||||
assert session.commits == 1
|
||||
|
|
|
|||
|
|
@ -83,8 +83,6 @@ def test_ingest_fires_idles_without_map_key(monkeypatch):
|
|||
|
||||
def test_ingest_fires_uses_keystore_key(monkeypatch):
|
||||
# Key saved via the dashboard Keys page (Postgres) is picked up.
|
||||
from upstream_cache import firms_cache
|
||||
firms_cache.clear()
|
||||
monkeypatch.setenv("FIRMS_MAP_KEY", "")
|
||||
monkeypatch.setattr(
|
||||
"fire_sources.get_api_key",
|
||||
|
|
@ -123,7 +121,7 @@ def test_ingest_fires_uses_keystore_key(monkeypatch):
|
|||
published.extend(points)
|
||||
return len(points)
|
||||
|
||||
monkeypatch.setattr("fire_sources.persist_hotspots", fake_publish)
|
||||
monkeypatch.setattr("fire_sources.publish_fire_batch", fake_publish)
|
||||
|
||||
assert asyncio.run(ingest_fires()) == 10 # NOAA-20 + NOAA-21 dual-write
|
||||
assert "a" * 32 in captured["url"]
|
||||
|
|
@ -131,100 +129,6 @@ def test_ingest_fires_uses_keystore_key(monkeypatch):
|
|||
assert any("VIIRS_NOAA21_NRT" in u for u in captured["urls"])
|
||||
|
||||
|
||||
def _reset_firms_poll_state():
|
||||
from upstream_cache import firms_cache
|
||||
import fire_sources
|
||||
|
||||
firms_cache.clear()
|
||||
if hasattr(fire_sources, "_csv_digest"):
|
||||
fire_sources._csv_digest.clear()
|
||||
if hasattr(fire_sources, "_seen_ids"):
|
||||
fire_sources._seen_ids.clear()
|
||||
|
||||
|
||||
def _fake_firms_http(monkeypatch, bodies_by_call: list[str] | None = None, body: str = SAMPLE_CSV):
|
||||
hits = {"n": 0}
|
||||
|
||||
class FakeResp:
|
||||
def __init__(self, text):
|
||||
self.text = text
|
||||
|
||||
def raise_for_status(self):
|
||||
pass
|
||||
|
||||
class FakeClient:
|
||||
def __init__(self, **kw):
|
||||
pass
|
||||
|
||||
async def __aenter__(self):
|
||||
return self
|
||||
|
||||
async def __aexit__(self, *exc):
|
||||
return False
|
||||
|
||||
async def get(self, url):
|
||||
idx = hits["n"]
|
||||
hits["n"] += 1
|
||||
if bodies_by_call is not None:
|
||||
text = bodies_by_call[min(idx, len(bodies_by_call) - 1)]
|
||||
else:
|
||||
text = body
|
||||
return FakeResp(text)
|
||||
|
||||
monkeypatch.setenv("FIRMS_MAP_KEY", "k" * 32)
|
||||
monkeypatch.setattr("fire_sources.FIRMS_DATASETS", ["VIIRS_NOAA20_NRT"])
|
||||
monkeypatch.setattr("fire_sources.httpx.AsyncClient", FakeClient)
|
||||
return hits
|
||||
|
||||
|
||||
def test_ingest_fires_skips_unchanged_csv(monkeypatch):
|
||||
"""Same FIRMS CSV must not be re-parsed into a 100k-row ON CONFLICT insert."""
|
||||
_reset_firms_poll_state()
|
||||
hits = _fake_firms_http(monkeypatch)
|
||||
persisted = []
|
||||
|
||||
async def fake_persist(points):
|
||||
persisted.append(len(points))
|
||||
return len(points)
|
||||
|
||||
monkeypatch.setattr("fire_sources.persist_hotspots", fake_persist)
|
||||
|
||||
assert asyncio.run(ingest_fires()) == 5
|
||||
assert persisted == [5]
|
||||
firms_cache_hits = hits["n"]
|
||||
persisted.clear()
|
||||
assert asyncio.run(ingest_fires()) == 0
|
||||
assert persisted == []
|
||||
# TTL cache may skip HTTP; either way we must not persist again.
|
||||
assert hits["n"] >= firms_cache_hits
|
||||
|
||||
|
||||
def test_ingest_fires_persists_only_new_hotspots(monkeypatch):
|
||||
"""When the CSV grows, persist the delta — not the whole 2-day dump."""
|
||||
_reset_firms_poll_state()
|
||||
extra = (
|
||||
SAMPLE_CSV
|
||||
+ "16.00000,-12.00000,340.00,0.40,0.40,2025-06-06,1500,N20,VIIRS,h,2.0NRT,310.00,8.00,D\n"
|
||||
)
|
||||
hits = _fake_firms_http(monkeypatch, bodies_by_call=[SAMPLE_CSV, extra])
|
||||
persisted = []
|
||||
|
||||
async def fake_persist(points):
|
||||
persisted.append([p["latitude"] for p in points])
|
||||
return len(points)
|
||||
|
||||
monkeypatch.setattr("fire_sources.persist_hotspots", fake_persist)
|
||||
|
||||
from upstream_cache import firms_cache
|
||||
|
||||
assert asyncio.run(ingest_fires()) == 5
|
||||
firms_cache.clear() # force the next poll to see the grown CSV
|
||||
persisted.clear()
|
||||
assert asyncio.run(ingest_fires()) == 1
|
||||
assert persisted == [[16.0]]
|
||||
assert hits["n"] == 2
|
||||
|
||||
|
||||
def _async_return(value):
|
||||
async def inner():
|
||||
return value
|
||||
|
|
|
|||
|
|
@ -1,86 +0,0 @@
|
|||
"""HUD load-time: no market 404 poll, deferred overlays, WS backoff, nginx snippet."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from pathlib import Path
|
||||
|
||||
ROOT = Path(__file__).resolve().parent.parent
|
||||
HTML = (ROOT / "app/static/index.html").read_text()
|
||||
|
||||
|
||||
def test_summarizer_dockerfile_copies_intel_modules():
|
||||
df = (ROOT / "news/summerizer/Dockerfile").read_text()
|
||||
assert "intel.py" in df
|
||||
assert "nous_client.py" in df
|
||||
|
||||
|
||||
def test_news_panel_pins_daily_recap():
|
||||
assert "kind=daily_recap" in HTML
|
||||
assert "DAILY RECAP" in HTML
|
||||
|
||||
|
||||
def test_nginx_ws_snippet_has_upgrade_headers():
|
||||
conf = (ROOT / "deploy/osint-ws.nginx.conf").read_text()
|
||||
assert "proxy_http_version 1.1" in conf
|
||||
assert "Upgrade" in conf
|
||||
assert "Connection" in conf
|
||||
assert "/ws/" in conf
|
||||
|
||||
|
||||
def test_market_ticker_does_not_poll_unwired_endpoint():
|
||||
assert "setInterval(probeMarket" not in HTML
|
||||
assert "initMarketTicker()" not in HTML or "probeMarket();" not in HTML.split("function initMarketTicker")[1][:400]
|
||||
|
||||
|
||||
def test_startup_defers_nonessential_overlays():
|
||||
init = HTML.split("function initMap")[1].split("function readMapPrefs")[0]
|
||||
# Must not fire all four DB loads + live overlays in the same tick.
|
||||
assert "setTimeout" in init or "requestAnimationFrame" in init
|
||||
|
||||
|
||||
def test_ws_reconnect_uses_backoff():
|
||||
assert "setTimeout(connectLiveWs, 4000)" not in HTML
|
||||
ws = HTML.split("function connectLiveWs")[1][:1200]
|
||||
assert "backoff" in ws.lower() or "wsRetry" in ws or "wsDelay" in ws
|
||||
|
||||
|
||||
def test_check_health_treats_degraded_status():
|
||||
fn = HTML.split("async function checkHealth")[1].split("/* ═══════════════ NAV")[0]
|
||||
assert "degraded" in fn.lower() or "d.status" in fn
|
||||
|
||||
|
||||
def test_chokepoint_presets_in_toolbar():
|
||||
assert 'id="chokepoint-btns"' in HTML
|
||||
assert 'id="chokepoint-select"' in HTML
|
||||
assert "loadChokepoints()" in HTML
|
||||
assert "/api/map/chokepoints" in HTML
|
||||
assert "function applyChokepoint" in HTML
|
||||
for name in ("Hormuz", "Bab el-Mandeb", "Suez", "Malacca", "Taiwan"):
|
||||
assert name in HTML
|
||||
|
||||
|
||||
def test_chokepoint_skips_aisstream_subscribe_outside_conus():
|
||||
load = HTML.split("async function loadVessels")[1].split("async function toggleStorms")[0]
|
||||
assert "intersectsConus()" in load
|
||||
assert "api/vessels/subscribe" in load
|
||||
assert "src=${encodeURIComponent(vesselSrcPref)}" in load or "&src=" in load
|
||||
apply = HTML.split("function applyChokepoint")[1].split("function currentBBox")[0]
|
||||
assert "vesselapi" in apply
|
||||
assert "lp-vessels-on" in apply
|
||||
assert "lp-sentinel-on" in apply
|
||||
assert "map.setView" in apply
|
||||
assert "minlat,minlon,maxlat,maxlon" in HTML.split("function chokepointLeafletBounds")[1][:400]
|
||||
|
||||
|
||||
def test_news_ticker_polls_more_often_than_summarizer_cycle():
|
||||
assert "NEWS_REFRESH_MS" in HTML
|
||||
# Summarizer is 15 min; ticker should refresh on a shorter cadence so
|
||||
# lesser-news fills show up without waiting for the next brief.
|
||||
line = [ln for ln in HTML.splitlines() if "NEWS_REFRESH_MS" in ln][0]
|
||||
assert "900000" not in line
|
||||
|
||||
|
||||
def test_phone_chokepoints_use_select_not_buttons():
|
||||
mobile = HTML.split("@media (max-width: 820px)")[1].split("@media (prefers-reduced-motion")[0]
|
||||
assert "#chokepoint-select { display: block; }" in mobile
|
||||
assert ".chokepoint-btns { display: none; }" in mobile or "#chokepoint-label, .chokepoint-btns { display: none; }" in mobile
|
||||
|
|
@ -1,363 +0,0 @@
|
|||
"""Geofence hit detection and WS alert routing (no Redis)."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import asyncio
|
||||
|
||||
import geofence
|
||||
from geofence import (
|
||||
matching_geofences,
|
||||
point_in_geojson,
|
||||
validate_polygon_geojson,
|
||||
)
|
||||
from ws_manager import ConnectionManager
|
||||
|
||||
|
||||
NC_BOX = {
|
||||
"type": "Polygon",
|
||||
"coordinates": [[
|
||||
[-80.0, 35.0],
|
||||
[-78.0, 35.0],
|
||||
[-78.0, 36.0],
|
||||
[-80.0, 36.0],
|
||||
[-80.0, 35.0],
|
||||
]],
|
||||
}
|
||||
|
||||
|
||||
def test_point_inside_polygon_hits():
|
||||
assert point_in_geojson(-79.0, 35.5, NC_BOX) is True
|
||||
|
||||
|
||||
def test_point_outside_polygon_misses():
|
||||
assert point_in_geojson(-122.4, 37.7, NC_BOX) is False
|
||||
|
||||
|
||||
def test_validate_rejects_non_polygon():
|
||||
try:
|
||||
validate_polygon_geojson({"type": "Point", "coordinates": [-79.0, 35.5]})
|
||||
assert False, "expected ValueError"
|
||||
except ValueError:
|
||||
pass
|
||||
|
||||
|
||||
def test_matching_geofences_only_active_hits():
|
||||
fences = [
|
||||
{"id": "a", "name": "NC", "active": True, "geojson": NC_BOX},
|
||||
{"id": "b", "name": "off", "active": False, "geojson": NC_BOX},
|
||||
]
|
||||
hits = matching_geofences(-79.0, 35.5, fences)
|
||||
assert [h["id"] for h in hits] == ["a"]
|
||||
assert matching_geofences(-122.4, 37.7, fences) == []
|
||||
|
||||
|
||||
FENCE_ID = "11111111-1111-1111-1111-111111111111"
|
||||
NC_VIEW = (-80.0, 35.0, -78.0, 36.0)
|
||||
SF_VIEW = (-123.0, 37.0, -121.0, 38.0)
|
||||
|
||||
|
||||
def _alert_payload(gid=FENCE_ID):
|
||||
return {
|
||||
"geofence_id": gid,
|
||||
"geofence_name": "NC",
|
||||
"source_kind": "ais",
|
||||
"entity_id": "366123456",
|
||||
"lat": 35.5,
|
||||
"lon": -79.0,
|
||||
}
|
||||
|
||||
|
||||
def test_geofence_alert_fans_out_only_to_viewport_clients():
|
||||
mgr = ConnectionManager()
|
||||
q_nc = mgr.register("nc")
|
||||
q_sf = mgr.register("sf")
|
||||
mgr.set_viewport("nc", NC_VIEW)
|
||||
mgr.set_viewport("sf", SF_VIEW)
|
||||
|
||||
async def run():
|
||||
n = await mgr.publish_point(
|
||||
"geofence_alert", _alert_payload(), lat=35.5, lon=-79.0,
|
||||
)
|
||||
assert n == 1
|
||||
msg = q_nc.get_nowait()
|
||||
assert msg["type"] == "geofence_alert"
|
||||
assert msg["payload"]["entity_id"] == "366123456"
|
||||
assert q_sf.empty()
|
||||
|
||||
asyncio.run(run())
|
||||
|
||||
|
||||
def test_off_viewport_watch_receives_geofence_alert():
|
||||
mgr = ConnectionManager()
|
||||
q_sf = mgr.register("sf")
|
||||
mgr.set_viewport("sf", SF_VIEW)
|
||||
mgr.set_watched_geofences("sf", [FENCE_ID])
|
||||
|
||||
async def run():
|
||||
n = await mgr.publish_point(
|
||||
"geofence_alert", _alert_payload(), lat=35.5, lon=-79.0,
|
||||
)
|
||||
assert n == 1
|
||||
msg = q_sf.get_nowait()
|
||||
assert msg["type"] == "geofence_alert"
|
||||
assert msg["payload"]["geofence_id"] == FENCE_ID
|
||||
|
||||
asyncio.run(run())
|
||||
|
||||
|
||||
def test_off_viewport_without_watch_does_not_receive_geofence_alert():
|
||||
mgr = ConnectionManager()
|
||||
q_sf = mgr.register("sf")
|
||||
mgr.set_viewport("sf", SF_VIEW)
|
||||
|
||||
async def run():
|
||||
n = await mgr.publish_point(
|
||||
"geofence_alert", _alert_payload(), lat=35.5, lon=-79.0,
|
||||
)
|
||||
assert n == 0
|
||||
assert q_sf.empty()
|
||||
|
||||
asyncio.run(run())
|
||||
|
||||
|
||||
def test_on_viewport_receives_geofence_alert_without_watch():
|
||||
mgr = ConnectionManager()
|
||||
q_nc = mgr.register("nc")
|
||||
mgr.set_viewport("nc", NC_VIEW)
|
||||
|
||||
async def run():
|
||||
n = await mgr.publish_point(
|
||||
"geofence_alert", _alert_payload(), lat=35.5, lon=-79.0,
|
||||
)
|
||||
assert n == 1
|
||||
assert q_nc.get_nowait()["type"] == "geofence_alert"
|
||||
|
||||
asyncio.run(run())
|
||||
|
||||
|
||||
def test_ais_stays_viewport_only_even_when_watching():
|
||||
mgr = ConnectionManager()
|
||||
q_sf = mgr.register("sf")
|
||||
mgr.set_viewport("sf", SF_VIEW)
|
||||
mgr.set_watched_geofences("sf", [FENCE_ID])
|
||||
|
||||
async def run():
|
||||
n = await mgr.publish_point("ais", {"id": "366123456"}, lat=35.5, lon=-79.0)
|
||||
assert n == 0
|
||||
assert q_sf.empty()
|
||||
|
||||
asyncio.run(run())
|
||||
|
||||
|
||||
def test_invalid_watch_uuids_ignored_empty_list_clears():
|
||||
mgr = ConnectionManager()
|
||||
q = mgr.register("sf")
|
||||
mgr.set_viewport("sf", SF_VIEW)
|
||||
mgr.set_watched_geofences("sf", ["not-a-uuid", FENCE_ID, "also-bad"])
|
||||
|
||||
async def run():
|
||||
n = await mgr.publish_point(
|
||||
"geofence_alert", _alert_payload(), lat=35.5, lon=-79.0,
|
||||
)
|
||||
assert n == 1
|
||||
q.get_nowait()
|
||||
mgr.set_watched_geofences("sf", [])
|
||||
n2 = await mgr.publish_point(
|
||||
"geofence_alert", _alert_payload(), lat=35.5, lon=-79.0,
|
||||
)
|
||||
assert n2 == 0
|
||||
assert q.empty()
|
||||
|
||||
asyncio.run(run())
|
||||
|
||||
|
||||
def test_unregister_clears_watched_geofences():
|
||||
mgr = ConnectionManager()
|
||||
q = mgr.register("sf")
|
||||
mgr.set_viewport("sf", SF_VIEW)
|
||||
mgr.set_watched_geofences("sf", [FENCE_ID])
|
||||
mgr.unregister("sf")
|
||||
q2 = mgr.register("sf")
|
||||
mgr.set_viewport("sf", SF_VIEW)
|
||||
|
||||
async def run():
|
||||
n = await mgr.publish_point(
|
||||
"geofence_alert", _alert_payload(), lat=35.5, lon=-79.0,
|
||||
)
|
||||
assert n == 0
|
||||
assert q2.empty()
|
||||
|
||||
asyncio.run(run())
|
||||
|
||||
|
||||
def test_record_and_notify_queries_postgis_when_cache_empty(monkeypatch):
|
||||
"""FIRMS ingest in the ingester has an empty in-process cache — still ST_Intersects."""
|
||||
import geofence
|
||||
|
||||
geofence._cache.clear()
|
||||
geofence._recent_hits.clear()
|
||||
st_called = []
|
||||
|
||||
async def fake_st(lon, lat):
|
||||
st_called.append((lon, lat))
|
||||
return [{
|
||||
"id": "11111111-1111-1111-1111-111111111111",
|
||||
"name": "NC",
|
||||
"geojson": NC_BOX,
|
||||
"active": True,
|
||||
}]
|
||||
|
||||
monkeypatch.setattr(geofence, "st_intersects", fake_st)
|
||||
|
||||
executed: list = []
|
||||
|
||||
class FakeSession:
|
||||
async def execute(self, stmt, params=None):
|
||||
executed.append(params or {})
|
||||
return None
|
||||
|
||||
async def commit(self):
|
||||
executed.append("commit")
|
||||
|
||||
async def __aenter__(self):
|
||||
return self
|
||||
|
||||
async def __aexit__(self, *a):
|
||||
return False
|
||||
|
||||
monkeypatch.setattr(geofence, "async_session", FakeSession)
|
||||
|
||||
async def run():
|
||||
from ws_manager import manager
|
||||
q = manager.register("nc")
|
||||
manager.set_viewport("nc", (-80.0, 35.0, -78.0, 36.0))
|
||||
n = await geofence.record_and_notify(
|
||||
source_kind="firms", entity_id="35.5,-79.0,N",
|
||||
lat=35.5, lon=-79.0, payload={"satellite": "N"},
|
||||
)
|
||||
msg = None if q.empty() else q.get_nowait()
|
||||
manager.unregister("nc")
|
||||
return n, msg
|
||||
|
||||
n, msg = asyncio.run(run())
|
||||
assert st_called == [(-79.0, 35.5)]
|
||||
assert n == 1
|
||||
assert msg["type"] == "geofence_alert"
|
||||
assert msg["payload"]["source_kind"] == "firms"
|
||||
inserts = [p for p in executed if isinstance(p, dict)]
|
||||
assert inserts and inserts[0]["source_kind"] == "firms"
|
||||
assert "commit" in executed
|
||||
|
||||
|
||||
def test_list_alerts_sql_filters(monkeypatch):
|
||||
captured: dict = {}
|
||||
|
||||
class FakeResult:
|
||||
def mappings(self):
|
||||
return self
|
||||
|
||||
def all(self):
|
||||
return []
|
||||
|
||||
class FakeSession:
|
||||
async def execute(self, stmt, params=None):
|
||||
captured["sql"] = str(stmt)
|
||||
captured["params"] = params
|
||||
return FakeResult()
|
||||
|
||||
async def __aenter__(self):
|
||||
return self
|
||||
|
||||
async def __aexit__(self, *a):
|
||||
return False
|
||||
|
||||
monkeypatch.setattr(geofence, "async_session", FakeSession)
|
||||
from datetime import datetime, timezone
|
||||
|
||||
since = datetime(2026, 8, 28, tzinfo=timezone.utc)
|
||||
until = datetime(2026, 8, 29, tzinfo=timezone.utc)
|
||||
|
||||
async def run():
|
||||
return await geofence.list_alerts(
|
||||
geofence_id=FENCE_ID, since=since, until=until,
|
||||
source_kind="firms", limit=5,
|
||||
)
|
||||
|
||||
assert asyncio.run(run()) == []
|
||||
sql = captured["sql"].lower()
|
||||
assert "geofence_id" in sql
|
||||
assert "created_at >=" in sql
|
||||
assert "created_at <=" in sql
|
||||
assert "source_kind" in sql
|
||||
assert captured["params"]["geofence_id"] == FENCE_ID
|
||||
assert captured["params"]["source_kind"] == "firms"
|
||||
assert captured["params"]["limit"] == 5
|
||||
|
||||
|
||||
def test_alembic_fence_created_index_exists():
|
||||
from pathlib import Path
|
||||
text = Path(__file__).resolve().parent.parent.joinpath(
|
||||
"alembic/versions/011_geofence_alerts_fence.py",
|
||||
).read_text()
|
||||
assert "ix_geofence_alerts_fence_created" in text
|
||||
assert "010_bbox_gist" in text
|
||||
|
||||
|
||||
def test_snapshot_at_404_when_fence_missing(monkeypatch):
|
||||
geofence._cache.clear()
|
||||
|
||||
async def boom():
|
||||
raise RuntimeError("db down")
|
||||
|
||||
monkeypatch.setattr(geofence, "refresh_cache", boom)
|
||||
|
||||
async def run():
|
||||
from datetime import datetime, timezone
|
||||
return await geofence.snapshot_at(
|
||||
FENCE_ID, datetime(2026, 8, 28, 12, 4, tzinfo=timezone.utc),
|
||||
)
|
||||
|
||||
assert asyncio.run(run()) is None
|
||||
|
||||
|
||||
def test_snapshot_queries_st_intersects(monkeypatch):
|
||||
geofence._cache[:] = [{
|
||||
"id": FENCE_ID, "name": "NC", "geojson": NC_BOX, "active": True,
|
||||
}]
|
||||
sqls: list[str] = []
|
||||
|
||||
class FakeResult:
|
||||
def mappings(self):
|
||||
return self
|
||||
|
||||
def all(self):
|
||||
return []
|
||||
|
||||
class FakeSession:
|
||||
async def execute(self, stmt, params=None):
|
||||
sqls.append(str(stmt))
|
||||
return FakeResult()
|
||||
|
||||
async def __aenter__(self):
|
||||
return self
|
||||
|
||||
async def __aexit__(self, *a):
|
||||
return False
|
||||
|
||||
monkeypatch.setattr(geofence, "async_session", FakeSession)
|
||||
|
||||
async def run():
|
||||
from datetime import datetime, timezone
|
||||
return await geofence.snapshot_at(
|
||||
FENCE_ID, datetime(2026, 8, 28, 12, 4, 30, tzinfo=timezone.utc),
|
||||
)
|
||||
|
||||
body = asyncio.run(run())
|
||||
assert body["aircraft"] == []
|
||||
assert body["vessels"] == []
|
||||
assert body["fires"] == []
|
||||
blob = "\n".join(sqls).lower()
|
||||
assert "st_intersects" in blob
|
||||
assert "aircraft_tracks_1min" in blob
|
||||
assert "vessel_tracks_1min" in blob
|
||||
assert "from fires" in blob
|
||||
|
|
@ -1,56 +0,0 @@
|
|||
"""Geofence layer panel: draw, watch, inbox, delete (HTML contract)."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from pathlib import Path
|
||||
|
||||
ROOT = Path(__file__).resolve().parent.parent
|
||||
HTML = (ROOT / "app/static/index.html").read_text()
|
||||
|
||||
|
||||
def test_geofence_panel_has_list_and_delete_hook():
|
||||
assert 'id="gf-draw"' in HTML
|
||||
assert 'id="gf-list"' in HTML
|
||||
assert "function deleteGeofence" in HTML
|
||||
assert "method: 'DELETE'" in HTML or 'method: "DELETE"' in HTML
|
||||
assert "/api/geofences/" in HTML
|
||||
|
||||
|
||||
def test_load_geofences_renders_delete_controls():
|
||||
js = HTML.split("async function loadGeofences", 1)[1].split(
|
||||
"async function loadFireAircraftHits", 1
|
||||
)[0]
|
||||
assert "gf-list" in js
|
||||
assert "deleteGeofence" in js
|
||||
assert "onEachFeature" in js
|
||||
assert "bindPopup" in js
|
||||
|
||||
|
||||
def test_finish_cancel_draw_controls():
|
||||
assert 'id="gf-finish"' in HTML
|
||||
assert 'id="gf-cancel"' in HTML
|
||||
assert "function cancelGeofenceDraw" in HTML
|
||||
assert "function onGfClose" in HTML
|
||||
|
||||
|
||||
def test_watch_geofences_ws_payload():
|
||||
assert "watch_geofences" in HTML
|
||||
assert "function sendWatchGeofences" in HTML
|
||||
|
||||
|
||||
def test_geofence_alert_inbox():
|
||||
assert 'id="gf-inbox"' in HTML
|
||||
assert "/api/geofence-alerts" in HTML
|
||||
assert "function loadGfInbox" in HTML
|
||||
assert "function pushGfInbox" in HTML
|
||||
|
||||
|
||||
def test_delete_geofence_still_present():
|
||||
assert "function deleteGeofence" in HTML
|
||||
assert "method: 'DELETE'" in HTML or 'method: "DELETE"' in HTML
|
||||
|
||||
|
||||
def test_fence_dvr_at_endpoint():
|
||||
assert "/at?timestamp=" in HTML or "/at?timestamp=${" in HTML
|
||||
assert "function dvrScrubFence" in HTML
|
||||
assert "gfSelectedId" in HTML
|
||||
|
|
@ -1,150 +0,0 @@
|
|||
"""GPSJAM GPS-interference overlay: level mapping, CSV→GeoJSON, API contract."""
|
||||
|
||||
import asyncio
|
||||
|
||||
import httpx
|
||||
|
||||
from live_layers import gpsjam_csv_to_geojson, gpsjam_level, overlay_catalog, _cache
|
||||
from main import app
|
||||
|
||||
BASE = "http://test"
|
||||
|
||||
# A valid H3 resolution-4 cell id (the payload hex column carries these).
|
||||
HEX_A = "8400c57ffffffff"
|
||||
|
||||
CSV = (
|
||||
"hex,count_good_aircraft,count_bad_aircraft\n"
|
||||
f"{HEX_A},0,20\n" # 100*(20-1)/20 = 95 -> high
|
||||
f"{HEX_A},8,2\n" # 100*(2-1)/10 = 10 -> medium
|
||||
f"{HEX_A},98,2\n" # 100*(2-1)/100 = 1 -> low
|
||||
f"{HEX_A},100,0\n" # bad == 0 -> dropped
|
||||
)
|
||||
|
||||
|
||||
async def _get(path: str) -> httpx.Response:
|
||||
transport = httpx.ASGITransport(app=app)
|
||||
async with httpx.AsyncClient(transport=transport, base_url=BASE) as client:
|
||||
return await client.get(path)
|
||||
|
||||
|
||||
def test_gpsjam_level_thresholds():
|
||||
assert gpsjam_level(0.0) == "low"
|
||||
assert gpsjam_level(2.0) == "low"
|
||||
assert gpsjam_level(2.1) == "medium"
|
||||
assert gpsjam_level(10.0) == "medium"
|
||||
assert gpsjam_level(10.1) == "high"
|
||||
assert gpsjam_level(95.0) == "high"
|
||||
|
||||
|
||||
def test_gpsjam_csv_to_geojson_levels_and_drop_zero_bad():
|
||||
fc = gpsjam_csv_to_geojson(CSV)
|
||||
assert fc["type"] == "FeatureCollection"
|
||||
assert len(fc["features"]) == 3 # bad==0 row dropped
|
||||
levels = [f["properties"]["level"] for f in fc["features"]]
|
||||
assert levels == ["high", "medium", "low"]
|
||||
for f in fc["features"]:
|
||||
geom = f["geometry"]
|
||||
assert geom["type"] == "Polygon"
|
||||
ring = geom["coordinates"][0]
|
||||
assert len(ring) == 7 # 6 verts + closing point
|
||||
assert ring[0] == ring[-1]
|
||||
assert f["properties"]["hex"] == HEX_A
|
||||
assert set(f["properties"]).issuperset({"level", "percent_bad", "good", "bad", "hex"})
|
||||
|
||||
|
||||
def test_gpsjam_csv_skips_malformed_rows():
|
||||
bad_csv = "hex,count_good_aircraft,count_bad_aircraft\n" \
|
||||
",1,5\n" \
|
||||
f"{HEX_A},x,5\n" \
|
||||
f"{HEX_A},1,notanint\n" \
|
||||
"not_a_cell,1,5\n"
|
||||
fc = gpsjam_csv_to_geojson(bad_csv)
|
||||
assert fc["features"] == []
|
||||
|
||||
|
||||
def test_overlay_catalog_has_gpsjam_stub():
|
||||
entry = overlay_catalog()["gpsjam"]
|
||||
assert entry["kind"] == "geojson"
|
||||
assert entry["endpoint"] == "/api/map/gpsjam"
|
||||
assert "GPSJAM" in entry["attribution"]
|
||||
|
||||
|
||||
def test_map_gpsjam_returns_featurecollection(monkeypatch):
|
||||
async def fake_fetch(date):
|
||||
return {"type": "FeatureCollection", "features": [{"type": "Feature"}]}
|
||||
|
||||
monkeypatch.setattr("main.fetch_gpsjam", fake_fetch)
|
||||
resp = asyncio.run(_get("/api/map/gpsjam?date=2026-08-28"))
|
||||
assert resp.status_code == 200
|
||||
assert resp.json()["type"] == "FeatureCollection"
|
||||
assert "max-age" in (resp.headers.get("cache-control") or "").lower()
|
||||
|
||||
|
||||
def test_map_gpsjam_rejects_bad_date():
|
||||
resp = asyncio.run(_get("/api/map/gpsjam?date=08-28-2026"))
|
||||
assert resp.status_code == 422
|
||||
|
||||
|
||||
def test_map_gpsjam_unavailable_on_404(monkeypatch):
|
||||
import httpx as _httpx
|
||||
|
||||
async def fake_fetch(date):
|
||||
exc = _httpx.HTTPStatusError(
|
||||
"404", request=_httpx.Request("GET", "http://x"), response=_httpx.Response(404)
|
||||
)
|
||||
raise exc
|
||||
|
||||
monkeypatch.setattr("main.fetch_gpsjam", fake_fetch)
|
||||
resp = asyncio.run(_get("/api/map/gpsjam?date=2026-08-28"))
|
||||
assert resp.status_code == 200
|
||||
body = resp.json()
|
||||
assert body["error"] == "unavailable"
|
||||
assert body["href"] == "https://gpsjam.org/"
|
||||
|
||||
|
||||
def test_map_gpsjam_unavailable_on_empty_features(monkeypatch):
|
||||
async def fake_fetch(date):
|
||||
return {"type": "FeatureCollection", "features": []}
|
||||
|
||||
monkeypatch.setattr("main.fetch_gpsjam", fake_fetch)
|
||||
resp = asyncio.run(_get("/api/map/gpsjam?date=2026-08-28"))
|
||||
assert resp.status_code == 200
|
||||
assert resp.json()["error"] == "unavailable"
|
||||
|
||||
|
||||
def test_fetch_gpsjam_hits_http_once_within_ttl(monkeypatch):
|
||||
_cache.clear()
|
||||
hits = {"n": 0}
|
||||
|
||||
class FakeResp:
|
||||
text = CSV
|
||||
|
||||
def raise_for_status(self):
|
||||
pass
|
||||
|
||||
class FakeClient:
|
||||
def __init__(self, **kw):
|
||||
pass
|
||||
|
||||
async def __aenter__(self):
|
||||
return self
|
||||
|
||||
async def __aexit__(self, *exc):
|
||||
return False
|
||||
|
||||
async def get(self, url):
|
||||
hits["n"] += 1
|
||||
assert url == "https://gpsjam.org/data/2026-08-28-h3_4.csv"
|
||||
return FakeResp()
|
||||
|
||||
monkeypatch.setattr("live_layers.httpx.AsyncClient", FakeClient)
|
||||
monkeypatch.setattr("live_layers._http", None)
|
||||
|
||||
from live_layers import fetch_gpsjam
|
||||
|
||||
fc1 = asyncio.run(fetch_gpsjam("2026-08-28"))
|
||||
fc2 = asyncio.run(fetch_gpsjam("2026-08-28"))
|
||||
assert len(fc1["features"]) == 3
|
||||
assert fc2 == fc1
|
||||
assert hits["n"] == 1
|
||||
_cache.clear()
|
||||
|
|
@ -1,33 +0,0 @@
|
|||
"""Liveness stays up; readiness/freshness is explicit."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import asyncio
|
||||
|
||||
import httpx
|
||||
|
||||
from main import app
|
||||
|
||||
BASE = "http://test"
|
||||
|
||||
|
||||
async def _get(path: str) -> httpx.Response:
|
||||
transport = httpx.ASGITransport(app=app)
|
||||
async with httpx.AsyncClient(transport=transport, base_url=BASE) as client:
|
||||
return await client.get(path)
|
||||
|
||||
|
||||
def test_health_includes_checks_even_when_db_ok(monkeypatch):
|
||||
"""HUD must be able to show degraded without docker killing the container."""
|
||||
body = asyncio.run(_get("/api/health")).json()
|
||||
assert "status" in body
|
||||
assert "checks" in body
|
||||
assert "db" in body["checks"]
|
||||
|
||||
|
||||
def test_ready_endpoint_exists():
|
||||
resp = asyncio.run(_get("/api/ready"))
|
||||
assert resp.status_code in (200, 503)
|
||||
body = resp.json()
|
||||
assert "checks" in body
|
||||
assert "status" in body
|
||||
|
|
@ -1,113 +0,0 @@
|
|||
"""Quiet HUD chrome: VIIRS default, collapsed rail, no Orbitron/MKT dashes."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from pathlib import Path
|
||||
|
||||
ROOT = Path(__file__).resolve().parent.parent
|
||||
HTML = (ROOT / "app/static/index.html").read_text()
|
||||
|
||||
|
||||
def _attr(html: str, elem_id: str) -> str:
|
||||
chunk = html.split(f'id="{elem_id}"', 1)[1].split(">", 1)[0]
|
||||
return chunk
|
||||
|
||||
|
||||
def test_initmap_prefers_viirs_true_color():
|
||||
init = HTML.split("async function initMap", 1)[1].split("function readMapPrefs", 1)[0]
|
||||
assert "VIIRS_SNPP_CorrectedReflectance_TrueColor" in init
|
||||
assert init.index("VIIRS_SNPP_CorrectedReflectance_TrueColor") < init.index(
|
||||
"MODIS_Terra_CorrectedReflectance_TrueColor"
|
||||
)
|
||||
assert init.index("MODIS_Terra_CorrectedReflectance_TrueColor") < init.index(
|
||||
"BlueMarble_ShadedRelief_Bathymetry"
|
||||
)
|
||||
|
||||
|
||||
def test_orbitron_gone():
|
||||
assert "Orbitron" not in HTML
|
||||
assert "IBM Plex Sans" in HTML
|
||||
assert "IBM Plex Mono" in HTML
|
||||
|
||||
|
||||
def test_lp_note_stripped_from_layer_list():
|
||||
assert 'class="lp-note"' not in HTML
|
||||
body = HTML.split('class="lp-body"', 1)[1].split("lp-legend", 1)[0]
|
||||
assert "lp-note" not in body
|
||||
|
||||
|
||||
def test_default_overlays_basemap_and_firms_only():
|
||||
fires = _attr(HTML, "lp-fires-on")
|
||||
assert "checked" in fires
|
||||
for eid in (
|
||||
"lp-cams-on",
|
||||
"lp-blips-on",
|
||||
"lp-news-on",
|
||||
"lp-radar-on",
|
||||
"lp-alerts-on",
|
||||
"lp-perim-on",
|
||||
"lp-ac-on",
|
||||
"lp-trains-on",
|
||||
"lp-storms-on",
|
||||
):
|
||||
assert "checked" not in _attr(HTML, eid), eid
|
||||
|
||||
|
||||
def test_geofence_markup_before_cameras():
|
||||
assert 'id="gf-draw"' in HTML
|
||||
assert HTML.index('id="gf-draw"') < HTML.index('id="lp-cams-on"')
|
||||
assert HTML.index('id="lp-base-on"') < HTML.index('id="gf-draw"')
|
||||
|
||||
|
||||
def test_parent_geofence_hud_survives():
|
||||
assert "watch_geofences" in HTML
|
||||
assert "function deleteGeofence" in HTML
|
||||
assert 'id="gf-finish"' in HTML
|
||||
assert 'id="gf-cancel"' in HTML
|
||||
assert 'id="gf-inbox"' in HTML
|
||||
|
||||
|
||||
def test_layer_rail_collapsed_on_load():
|
||||
head = HTML.split('class="lp-head"', 1)[1].split("</div>", 1)[0]
|
||||
assert 'aria-expanded="false"' in head
|
||||
assert 'id="layer-panel" class="collapsed"' in HTML
|
||||
|
||||
|
||||
def test_market_ticker_hidden_no_poll():
|
||||
mkt = HTML.split('class="ticker market"', 1)[1].split(">", 1)[0]
|
||||
assert "hidden" in mkt
|
||||
assert "setInterval(probeMarket" not in HTML
|
||||
assert "setInterval(loadMarket" not in HTML
|
||||
init = HTML.split("function initMarketTicker", 1)[1].split("function ", 1)[0]
|
||||
assert "/api/market" in init or "404-poll" in init
|
||||
assert "setInterval" not in init
|
||||
|
||||
|
||||
def test_news_ticker_fills_news_only_dock():
|
||||
css = HTML.split("</style>", 1)[0]
|
||||
compact = css.replace(" ", "").replace("\n", "")
|
||||
assert ".dock.news-only{height:32px;}" in compact
|
||||
assert ".dock.news-only.ticker{height:100%;}" in compact
|
||||
assert ".ticker{display:flex;align-items:stretch;height:50%;" in compact
|
||||
|
||||
|
||||
def test_news_pins_are_circle_markers():
|
||||
js = HTML.split("async function loadNewsPins", 1)[1].split("function refreshLiveOverlays", 1)[0]
|
||||
assert "L.circleMarker" in js
|
||||
assert "fillOpacity: 0.7" in js or "fillOpacity:0.7" in js
|
||||
assert "rotate(45deg)" not in js
|
||||
assert "L.divIcon" not in js
|
||||
|
||||
|
||||
def test_chokepoint_buttons_not_in_toolbar_flow():
|
||||
assert 'id="chokepoint-select"' in HTML
|
||||
css = HTML.split("</style>", 1)[0]
|
||||
assert ".chokepoint-btns { display: none; }" in css or ".chokepoint-btns{display:none" in css.replace(
|
||||
" ", ""
|
||||
)
|
||||
|
||||
|
||||
def test_brand_is_osint_slash():
|
||||
assert "GLOBAL SITUATIONAL AWARENESS TERMINAL" not in HTML
|
||||
assert "OSINT" in HTML
|
||||
assert 'class="accent">//</span>' in HTML
|
||||
|
|
@ -1,90 +0,0 @@
|
|||
"""HUD: layer-rail stats, shortcuts, terminator, zoom-gated cams, SWPC chip."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from pathlib import Path
|
||||
|
||||
ROOT = Path(__file__).resolve().parent.parent
|
||||
HTML = (ROOT / "app/static/index.html").read_text()
|
||||
|
||||
|
||||
def _fn(name: str, nxt: str | None = None) -> str:
|
||||
start = HTML.index(f"function {name}")
|
||||
if nxt:
|
||||
return HTML[start : HTML.index(f"function {nxt}", start + 1)]
|
||||
return HTML[start : start + 4000]
|
||||
|
||||
|
||||
def test_stats_poll_uses_api_then_falls_back():
|
||||
assert "/api/stats" in HTML
|
||||
assert "30000" in HTML.split("pollLayerStats")[1][:2500] or "STATS_POLL_MS" in HTML
|
||||
poll = HTML.split("async function pollLayerStats")[1].split("async function ")[0]
|
||||
assert "404" in poll
|
||||
assert "catch" in poll
|
||||
ids = HTML.split("STATS_COUNT_IDS")[1].split("};")[0]
|
||||
assert "aircraft" in ids and "cameras" in ids and "fires" in ids and "vessels" in ids
|
||||
# Overlay loaders still write array lengths when stats is down.
|
||||
assert "setLayerCount('lp-fires-count'" in HTML or 'setLayerCount("lp-fires-count"' in HTML
|
||||
assert "setLayerCount('lp-cams-count'" in HTML or 'setLayerCount("lp-cams-count"' in HTML
|
||||
assert "setLayerCount('lp-ac-count'" in HTML or 'setLayerCount("lp-ac-count"' in HTML
|
||||
assert "setLayerCount('lp-vessels-count'" in HTML or 'setLayerCount("lp-vessels-count"' in HTML
|
||||
|
||||
|
||||
def test_keyboard_shortcuts_do_not_steal_osiris_fs():
|
||||
keys = HTML.split("function initHudKeys")[1].split("function ")[0]
|
||||
assert "Escape" in keys
|
||||
assert "cheat-sheet" in keys or "toggleCheatSheet" in keys
|
||||
assert "mapResetView" in keys
|
||||
assert "toggleLayerPanel" in keys or "closeLayerPanel" in keys
|
||||
# Do not bind Osiris's conflicting F/S (flights vs fullscreen / search).
|
||||
assert "e.key === 'f'" not in keys.lower()
|
||||
assert "e.key === 's'" not in keys.lower()
|
||||
assert "case 'f'" not in keys.lower()
|
||||
assert "case 's'" not in keys.lower()
|
||||
assert 'id="cheat-sheet"' in HTML
|
||||
assert "?" in keys or "Shift" in keys
|
||||
|
||||
|
||||
def test_terminator_toggle_defaults_off():
|
||||
assert 'id="lp-terminator-on"' in HTML
|
||||
row = HTML.split('id="lp-terminator-on"')[0][-120:] + HTML.split('id="lp-terminator-on"')[1][:80]
|
||||
assert "checked" not in row.split(">")[0]
|
||||
assert "function toggleTerminator" in HTML
|
||||
assert "subsolarPoint" in HTML or "terminator" in HTML.lower()
|
||||
|
||||
|
||||
def test_camera_thumbs_gated_at_zoom_12():
|
||||
assert "CAM_THUMB_MIN_ZOOM" in HTML
|
||||
assert "CAM_THUMB_MIN_ZOOM = 12" in HTML
|
||||
thumb = _fn("camThumb", "camPopupHtml")
|
||||
assert "camThumbsAllowed" in thumb or "CAM_THUMB_MIN_ZOOM" in thumb
|
||||
assert "zoom in for preview" in HTML or "zoom for preview" in HTML
|
||||
assert "preview unavailable" in HTML
|
||||
# RTSP still proxy through snapshot; never emit rtsp hrefs.
|
||||
src = _fn("camSourceLink", "youtubeId")
|
||||
assert "rtsp://" in src
|
||||
assert "href=" not in src.split("rtsp://")[1].split("return")[0] or "Never emit" in src
|
||||
assert 'href="${esc(url)}"' in src or "href=\"${esc(url)}\"" in src
|
||||
assert src.index("rtsp://") < src.index("href=")
|
||||
|
||||
|
||||
def test_swpc_chip_browser_direct_correct_urls():
|
||||
assert 'id="swpc-chip"' in HTML
|
||||
assert "services.swpc.noaa.gov/json/planetary_k_index_1m.json" in HTML
|
||||
assert "services.swpc.noaa.gov/json/goes/primary/xray-flares-latest.json" in HTML
|
||||
assert "services.swpc.noaa.gov/products/alerts.json" in HTML
|
||||
assert "services.swpc.noaa.gov/json/alerts.json" not in HTML
|
||||
sw = HTML.split("async function pollSwpc")[1].split("async function ")[0]
|
||||
assert "hidden" in sw
|
||||
assert "kp_index" in sw
|
||||
assert "90000" in HTML or "SWPC_POLL_MS" in HTML
|
||||
|
||||
|
||||
def test_new_chrome_does_not_cover_mobile_layers_zoom():
|
||||
mobile = HTML.split("@media (max-width: 820px)")[1].split("@media (prefers-reduced-motion")[0]
|
||||
assert "#layer-panel" in mobile
|
||||
assert ".leaflet-top.leaflet-right .leaflet-control-zoom" in mobile
|
||||
assert 'id="cheat-sheet"' in HTML
|
||||
cheat = HTML.split(".cheat-sheet")[1][:500]
|
||||
assert "z-index" in cheat
|
||||
assert "calc(100% - 96px)" in cheat or "96px" in cheat
|
||||
|
|
@ -1,145 +0,0 @@
|
|||
"""GET /api/infrastructure — Overpass nuclear markers."""
|
||||
|
||||
import asyncio
|
||||
|
||||
import httpx
|
||||
|
||||
from live_layers import (
|
||||
normalize_infra_element,
|
||||
overlay_catalog,
|
||||
overpass_nuclear_to_markers,
|
||||
_cache,
|
||||
)
|
||||
from main import app
|
||||
|
||||
BASE = "http://test"
|
||||
|
||||
OVERPASS = {
|
||||
"version": 0.6,
|
||||
"generator": "Overpass API",
|
||||
"elements": [
|
||||
{
|
||||
"type": "node",
|
||||
"id": 12345,
|
||||
"lat": 44.0,
|
||||
"lon": -1.5,
|
||||
"tags": {"name": "Test NPP", "operator": "EDF", "plant:source": "nuclear"},
|
||||
},
|
||||
{
|
||||
"type": "way",
|
||||
"id": 67890,
|
||||
"center": {"lat": 43.5, "lon": -1.25},
|
||||
"tags": {"name": "Test Plant Way", "plant:source": "nuclear"},
|
||||
},
|
||||
{
|
||||
"type": "relation",
|
||||
"id": 999,
|
||||
"center": {"lat": 43.0, "lon": -1.0},
|
||||
"tags": {},
|
||||
},
|
||||
],
|
||||
}
|
||||
|
||||
|
||||
async def _get(path: str) -> httpx.Response:
|
||||
transport = httpx.ASGITransport(app=app)
|
||||
async with httpx.AsyncClient(transport=transport, base_url=BASE) as client:
|
||||
return await client.get(path)
|
||||
|
||||
|
||||
def test_normalize_node_to_marker():
|
||||
m = normalize_infra_element(OVERPASS["elements"][0], "nuclear")
|
||||
assert m["id"] == "node/12345"
|
||||
assert m["name"] == "Test NPP"
|
||||
assert m["lat"] == 44.0
|
||||
assert m["lon"] == -1.5
|
||||
assert m["type"] == "nuclear"
|
||||
assert m["extra"]["operator"] == "EDF"
|
||||
assert "name" not in m["extra"]
|
||||
|
||||
|
||||
def test_way_center_and_unnamed_fallback():
|
||||
way = normalize_infra_element(OVERPASS["elements"][1], "nuclear")
|
||||
assert way["lat"] == 43.5
|
||||
assert way["lon"] == -1.25
|
||||
rel = normalize_infra_element(OVERPASS["elements"][2], "nuclear")
|
||||
assert rel["name"] == "relation/999"
|
||||
|
||||
|
||||
def test_overpass_json_to_markers():
|
||||
markers = overpass_nuclear_to_markers(OVERPASS)
|
||||
assert len(markers) == 3
|
||||
assert markers[0]["id"] == "node/12345"
|
||||
|
||||
|
||||
def test_missing_bbox_400():
|
||||
resp = asyncio.run(_get("/api/infrastructure?types=nuclear"))
|
||||
assert resp.status_code == 400
|
||||
|
||||
|
||||
def test_unknown_type_422():
|
||||
resp = asyncio.run(_get("/api/infrastructure?types=military&bbox=-2,43,-1,44"))
|
||||
assert resp.status_code == 422
|
||||
|
||||
|
||||
def test_map_infrastructure_returns_markers(monkeypatch):
|
||||
async def fake_fetch(types, bbox):
|
||||
return [
|
||||
{"id": "node/1", "name": "X", "lat": 1.0, "lon": 2.0,
|
||||
"type": "nuclear", "extra": {}}
|
||||
]
|
||||
|
||||
monkeypatch.setattr("main.fetch_infrastructure", fake_fetch)
|
||||
resp = asyncio.run(_get("/api/infrastructure?types=nuclear&bbox=-2,43,-1,44"))
|
||||
assert resp.status_code == 200
|
||||
body = resp.json()
|
||||
assert body[0]["name"] == "X"
|
||||
assert body[0]["type"] == "nuclear"
|
||||
assert "max-age" in (resp.headers.get("cache-control") or "").lower()
|
||||
|
||||
|
||||
def test_overlay_catalog_has_infra_nuclear():
|
||||
entry = overlay_catalog()["infra_nuclear"]
|
||||
assert entry["kind"] == "points"
|
||||
assert "nuclear" in entry["endpoint"]
|
||||
|
||||
|
||||
def test_fetch_infrastructure_cache_hit_no_refetch(monkeypatch):
|
||||
_cache.clear()
|
||||
hits = {"n": 0}
|
||||
|
||||
class FakeResp:
|
||||
def raise_for_status(self):
|
||||
pass
|
||||
|
||||
def json(self):
|
||||
return OVERPASS
|
||||
|
||||
class FakeClient:
|
||||
def __init__(self, **kw):
|
||||
pass
|
||||
|
||||
async def __aenter__(self):
|
||||
return self
|
||||
|
||||
async def __aexit__(self, *exc):
|
||||
return False
|
||||
|
||||
async def post(self, url, data=None, timeout=None):
|
||||
hits["n"] += 1
|
||||
assert "overpass-api.de" in url
|
||||
assert "plant:source" in data["data"]
|
||||
assert "nuclear" in data["data"]
|
||||
return FakeResp()
|
||||
|
||||
monkeypatch.setattr("live_layers.httpx.AsyncClient", FakeClient)
|
||||
monkeypatch.setattr("live_layers._http", None)
|
||||
|
||||
from live_layers import fetch_infrastructure
|
||||
|
||||
m1 = asyncio.run(fetch_infrastructure("nuclear", "-2,43,-1,44"))
|
||||
m2 = asyncio.run(fetch_infrastructure("nuclear", "-2,43,-1,44"))
|
||||
assert len(m1) == 3
|
||||
assert m2 == m1
|
||||
assert hits["n"] == 1
|
||||
_cache.clear()
|
||||
|
|
@ -1,67 +0,0 @@
|
|||
"""SSRF guard on ingest triggers + PATCH /api/sources allowlist."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import asyncio
|
||||
|
||||
import httpx
|
||||
import pytest
|
||||
from pydantic import ValidationError
|
||||
|
||||
from main import app
|
||||
|
||||
BASE = "http://test"
|
||||
LINK_LOCAL_META = "http://169.254.169.254/latest/meta-data/"
|
||||
LOOPBACK = "http://127.0.0.1/secret"
|
||||
|
||||
|
||||
async def _req(method: str, path: str, **kw) -> httpx.Response:
|
||||
transport = httpx.ASGITransport(app=app)
|
||||
async with httpx.AsyncClient(transport=transport, base_url=BASE) as client:
|
||||
return await client.request(method, path, **kw)
|
||||
|
||||
|
||||
def test_rss_ingest_rejects_link_local_metadata_url(monkeypatch):
|
||||
called = {"n": 0}
|
||||
|
||||
async def _boom(*_a, **_k):
|
||||
called["n"] += 1
|
||||
raise AssertionError("ingest_rss_feed must not run for a private URL")
|
||||
|
||||
monkeypatch.setattr("main.ingest_rss_feed", _boom)
|
||||
resp = asyncio.run(_req("POST", "/api/ingest/rss", params={"feed_url": LINK_LOCAL_META}))
|
||||
assert resp.status_code == 400
|
||||
assert called["n"] == 0
|
||||
|
||||
|
||||
def test_gdelt_ingest_rejects_private_query_url(monkeypatch):
|
||||
called = {"n": 0}
|
||||
|
||||
async def _boom(*_a, **_k):
|
||||
called["n"] += 1
|
||||
raise AssertionError("ingest_gdelt must not run for a private URL query")
|
||||
|
||||
monkeypatch.setattr("main.ingest_gdelt", _boom)
|
||||
resp = asyncio.run(_req("POST", "/api/ingest/gdelt", params={"query": LOOPBACK}))
|
||||
assert resp.status_code == 400
|
||||
assert called["n"] == 0
|
||||
|
||||
|
||||
def test_update_source_rejects_unknown_fields():
|
||||
sid = "00000000-0000-0000-0000-000000000001"
|
||||
resp = asyncio.run(_req("PATCH", f"/api/sources/{sid}", json={"enabled": True, "source_type": "rss"}))
|
||||
assert resp.status_code == 422
|
||||
|
||||
|
||||
def test_feed_source_update_allowlist_only():
|
||||
from schemas import FeedSourceUpdate
|
||||
|
||||
payload = FeedSourceUpdate(name="n", url="https://example.com/rss", config={"k": 1}, enabled=False)
|
||||
assert payload.model_dump(exclude_unset=True) == {
|
||||
"name": "n",
|
||||
"url": "https://example.com/rss",
|
||||
"config": {"k": 1},
|
||||
"enabled": False,
|
||||
}
|
||||
with pytest.raises(ValidationError):
|
||||
FeedSourceUpdate.model_validate({"enabled": True, "id": "00000000-0000-0000-0000-000000000001"})
|
||||
|
|
@ -1,7 +1,5 @@
|
|||
"""Unit tests for live map-layer mappers (aircraft, trains, AIS, WFIGS, Caltrans)."""
|
||||
|
||||
import json
|
||||
|
||||
from live_layers import (
|
||||
MARKER_FIELDS,
|
||||
bbox_center_radius_nm,
|
||||
|
|
@ -9,10 +7,7 @@ from live_layers import (
|
|||
filter_points_bbox,
|
||||
parse_bbox,
|
||||
quantize_bbox,
|
||||
pick_sentinel_feature,
|
||||
rainviewer_tile_url,
|
||||
sign_cog_url,
|
||||
sentinel1_tile_url,
|
||||
slim_alert_properties,
|
||||
to_marker,
|
||||
transform_adsb_lol,
|
||||
|
|
@ -20,14 +15,12 @@ from live_layers import (
|
|||
transform_amtraker,
|
||||
transform_nhc_storms,
|
||||
transform_wfigs_incidents,
|
||||
SENTINEL1_ATTRIBUTION,
|
||||
TITILER_COG_TILES,
|
||||
_cache,
|
||||
_ttl_get,
|
||||
_wfigs_params,
|
||||
)
|
||||
|
||||
from camera_scraper import parse_caltrans_json, parse_udot_ibi_page, parse_odot_json, parse_mdot_json
|
||||
from camera_scraper import parse_caltrans_json
|
||||
|
||||
|
||||
def test_parse_bbox_and_radius_clamps_to_150_nm():
|
||||
|
|
@ -259,186 +252,6 @@ def test_parse_caltrans_skips_oos_and_maps_jpeg_hls():
|
|||
assert "rtsp://" not in cam["snapshot_url"].lower()
|
||||
|
||||
|
||||
# ── UDOT IBI 511 parser ──────────────────────────────────────────────────
|
||||
|
||||
def _udot_row(cam_id, lng, lat, **img_overrides):
|
||||
img = {
|
||||
"id": cam_id, "cameraSiteId": cam_id,
|
||||
"imageUrl": f"/map/Cctv/{cam_id}", "disabled": False, "blocked": False,
|
||||
}
|
||||
img.update(img_overrides)
|
||||
return {
|
||||
"id": cam_id, "sourceId": "102771", "source": "ADX",
|
||||
"roadway": "Unknown", "direction": "Unknown",
|
||||
"location": "Freedom Blvd / 200 W @ 1100 N, PVO",
|
||||
"latLng": {"geography": {
|
||||
"coordinateSystemId": 4326,
|
||||
"wellKnownText": f"POINT ({lng} {lat})"}},
|
||||
"images": [img],
|
||||
}
|
||||
|
||||
|
||||
def _udot_page(rows):
|
||||
import json
|
||||
return json.dumps({"draw": 0, "recordsTotal": len(rows),
|
||||
"recordsFiltered": len(rows), "data": rows})
|
||||
|
||||
|
||||
def test_parse_udot_wkt_maps_lng_lat():
|
||||
cams = parse_udot_ibi_page(_udot_page([_udot_row(112731, -111.66204, 40.24863)]))
|
||||
assert len(cams) == 1
|
||||
cam = cams[0]
|
||||
# WKT is `POINT (lng lat)` — order must not be swapped.
|
||||
assert cam["location_lat"] == 40.24863
|
||||
assert cam["location_lon"] == -111.66204
|
||||
assert cam["discovery_source"] == "udot"
|
||||
assert cam["vendor"] == "UDOT"
|
||||
assert cam["source_url"] == "https://prod-ut.ibi511.com/map/Cctv/112731"
|
||||
assert cam["snapshot_url"] == cam["source_url"]
|
||||
assert "rtsp://" not in cam["source_url"].lower()
|
||||
assert cam["raw"]["udot_id"] == 112731
|
||||
|
||||
|
||||
def test_parse_udot_skips_blocked_and_disabled():
|
||||
rows = [
|
||||
_udot_row(1, -111.0, 40.0),
|
||||
_udot_row(2, -111.1, 40.1, blocked=True),
|
||||
_udot_row(3, -111.2, 40.2, disabled=True),
|
||||
]
|
||||
rows.append(_udot_row(4, -111.3, 40.3))
|
||||
rows[3]["images"] = [] # no images → drop
|
||||
cams = parse_udot_ibi_page(_udot_page(rows))
|
||||
assert [c["raw"]["udot_id"] for c in cams] == [1]
|
||||
|
||||
|
||||
def test_parse_udot_drops_out_of_bbox():
|
||||
rows = [
|
||||
_udot_row(1, -111.0, 40.0), # inside Utah
|
||||
_udot_row(2, -100.0, 40.0), # east of -108.9
|
||||
_udot_row(3, -120.0, 40.0), # west of -114.2
|
||||
_udot_row(4, -111.0, 44.0), # north of 42.1
|
||||
_udot_row(5, -111.0, 30.0), # south of 36.9
|
||||
]
|
||||
cams = parse_udot_ibi_page(_udot_page(rows))
|
||||
assert [c["raw"]["udot_id"] for c in cams] == [1]
|
||||
|
||||
|
||||
def test_parse_udot_bad_payload_returns_empty():
|
||||
import json
|
||||
assert parse_udot_ibi_page("not json") == []
|
||||
assert parse_udot_ibi_page(json.dumps({"data": None})) == []
|
||||
assert parse_udot_ibi_page(json.dumps({"data": "nope"})) == []
|
||||
|
||||
|
||||
def test_parse_udot_missing_wkt_skipped():
|
||||
row = _udot_row(1, -111.0, 40.0)
|
||||
row["latLng"] = {}
|
||||
assert parse_udot_ibi_page(_udot_page([row])) == []
|
||||
|
||||
|
||||
def test_parse_odot_tripcheck_keeps_valid_skips_missing_and_oob():
|
||||
payload = """
|
||||
{"features":[
|
||||
{"attributes":{
|
||||
"cameraId":277,"filename":"AstoriaUS101_pid392.jpg",
|
||||
"latitude":46.18785,"longitude":-123.85347,
|
||||
"route":"US101 ","title":"US101 at Astoria"
|
||||
}},
|
||||
{"attributes":{
|
||||
"cameraId":200,"filename":"","latitude":45.0,"longitude":-122.0,
|
||||
"route":"I-5","title":"missing filename"
|
||||
}},
|
||||
{"attributes":{
|
||||
"cameraId":300,"filename":"nocal_pid1.jpg",
|
||||
"latitude":40.0,"longitude":-122.0,
|
||||
"route":"US97","title":"out of bbox"
|
||||
}},
|
||||
{"attributes":{
|
||||
"cameraId":400,"filename":"badcoord_pid2.jpg",
|
||||
"latitude":null,"longitude":-122.0,
|
||||
"route":"OR22","title":"null coord"
|
||||
}}
|
||||
]}
|
||||
"""
|
||||
cams = parse_odot_json(payload, "www.tripcheck.com")
|
||||
assert len(cams) == 1
|
||||
cam = cams[0]
|
||||
assert cam["discovery_source"] == "odot"
|
||||
assert cam["snapshot_url"] == (
|
||||
"https://tripcheck.com/RoadCams/cams/AstoriaUS101_pid392.jpg")
|
||||
assert cam["source_url"] == cam["snapshot_url"]
|
||||
assert cam["location_lat"] == 46.18785
|
||||
assert cam["location_lon"] == -123.85347
|
||||
assert "US101 at Astoria" in cam["location_name"]
|
||||
assert cam["vendor"] == "ODOT"
|
||||
assert cam["device_type"] == "http"
|
||||
assert "rtsp://" not in cam["snapshot_url"].lower()
|
||||
|
||||
|
||||
def test_parse_odot_tripcheck_handles_malformed():
|
||||
assert parse_odot_json("not json", "www.tripcheck.com") == []
|
||||
assert parse_odot_json('{"features":null}', "www.tripcheck.com") == []
|
||||
|
||||
|
||||
def test_parse_mdot_extracts_html_fields_and_bbox_filters():
|
||||
rows = [
|
||||
# In-bbox, full fields.
|
||||
{
|
||||
"route": "11 Mile",
|
||||
"county": 'Wayne County <a href="/MiDrive/map?cameras=true&lat=42.491304&lon=-83.04479&zoom=15&id=1129"target="_blank">Go to</a>',
|
||||
"location": " @ Mound NB",
|
||||
"direction": "Traffic closest to camera is traveling north.",
|
||||
"image": '<img alt="x" class="cameraImageForActivePane" id="1129Img" src="https://micamerasimages.net/thumbs/semtoc_cam_253.flv.jpg?item=1" height="170" width="250" onerror="cameraImageBroken(this)">',
|
||||
},
|
||||
# Out of bbox (lat 50) → drop.
|
||||
{
|
||||
"route": "Far",
|
||||
"county": 'Nowhere <a href="/MiDrive/map?lat=50.0&lon=-83.0&zoom=15&id=9999">Go to</a>',
|
||||
"location": "",
|
||||
"image": '<img src="https://micamerasimages.net/thumbs/x.jpg">',
|
||||
},
|
||||
# Missing coordinates → drop.
|
||||
{
|
||||
"route": "NoCoords",
|
||||
"county": 'Somewhere <a href="/MiDrive/map?zoom=15&id=8888">Go to</a>',
|
||||
"location": "",
|
||||
"image": '<img src="https://micamerasimages.net/thumbs/y.jpg">',
|
||||
},
|
||||
# Missing image → drop.
|
||||
{
|
||||
"route": "NoImage",
|
||||
"county": 'Kent <a href="/MiDrive/map?lat=42.8841&lon=-85.6646&zoom=15&id=2113">Go to</a>',
|
||||
"location": " @ Division",
|
||||
"image": "",
|
||||
},
|
||||
# RTSP image src → drop.
|
||||
{
|
||||
"route": "Rtsp",
|
||||
"county": 'Wayne <a href="/MiDrive/map?lat=42.4&lon=-83.1&zoom=15&id=1234">Go to</a>',
|
||||
"location": "",
|
||||
"image": '<img src="rtsp://10.0.0.1/stream">',
|
||||
},
|
||||
]
|
||||
cams = parse_mdot_json(json.dumps(rows), "mdotjboss.state.mi.us")
|
||||
assert len(cams) == 1
|
||||
cam = cams[0]
|
||||
assert cam["discovery_source"] == "mdot"
|
||||
assert cam["location_lat"] == 42.491304
|
||||
assert cam["location_lon"] == -83.04479
|
||||
assert cam["snapshot_url"] == "https://micamerasimages.net/thumbs/semtoc_cam_253.flv.jpg?item=1"
|
||||
assert cam["source_url"] == "https://mdotjboss.state.mi.us/MiDrive/camera/1129"
|
||||
assert cam["device_type"] == "http"
|
||||
assert cam["vendor"] == "MDOT"
|
||||
assert "11 Mile @ Mound NB" in cam["location_name"]
|
||||
assert "Wayne County" in cam["location_name"]
|
||||
|
||||
|
||||
def test_parse_mdot_handles_malformed_payload():
|
||||
assert parse_mdot_json("not json", "mdot") == []
|
||||
assert parse_mdot_json('{"not": "a list"}', "mdot") == []
|
||||
assert parse_mdot_json("[]", "mdot") == []
|
||||
|
||||
|
||||
def test_quantize_bbox_stable_under_jitter():
|
||||
a = quantize_bbox(*parse_bbox("-78.7912,35.7711,-78.6101,35.9102"))
|
||||
b = quantize_bbox(*parse_bbox("-78.7900,35.7700,-78.6110,35.9090"))
|
||||
|
|
@ -549,440 +362,3 @@ def test_wfigs_params_requests_simplified_geometry():
|
|||
# Envelope is the quantized cell, not the raw pan box.
|
||||
geom = params["geometry"]
|
||||
assert geom != "-84.5,33.8,-75.4,36.6"
|
||||
|
||||
|
||||
def test_transform_adsb_lol_flags_military_from_dbflags():
|
||||
payload = {
|
||||
"ac": [
|
||||
{
|
||||
"hex": "ae01ab",
|
||||
"flight": "RCH123 ",
|
||||
"r": "04-1234",
|
||||
"t": "C17",
|
||||
"lat": 35.1,
|
||||
"lon": -77.9,
|
||||
"alt_baro": 24000,
|
||||
"gs": 410,
|
||||
"track": 90,
|
||||
"squawk": "5101",
|
||||
"emergency": "none",
|
||||
"category": "A5",
|
||||
"dbFlags": 1,
|
||||
"baro_rate": 64,
|
||||
"alt_geom": 24500,
|
||||
"desc": "Boeing C-17A Globemaster III",
|
||||
"ownOp": "USAF",
|
||||
},
|
||||
{
|
||||
"hex": "a1b2c3",
|
||||
"flight": "AAL123",
|
||||
"r": "N123AA",
|
||||
"t": "B738",
|
||||
"lat": 35.88,
|
||||
"lon": -78.79,
|
||||
"alt_baro": 32000,
|
||||
"gs": 430,
|
||||
"track": 87,
|
||||
"squawk": "1200",
|
||||
"emergency": "none",
|
||||
"category": "A3",
|
||||
},
|
||||
]
|
||||
}
|
||||
rows = {r["id"]: r for r in transform_adsb_lol(payload)}
|
||||
mil = rows["ae01ab"]["extra"]
|
||||
civ = rows["a1b2c3"]["extra"]
|
||||
assert mil["role"] == "military"
|
||||
assert mil["role_src"] == "dbFlags"
|
||||
assert mil["emitter"] == "heavy"
|
||||
assert mil["desc"] == "Boeing C-17A Globemaster III"
|
||||
assert mil["ownOp"] == "USAF"
|
||||
assert mil["vs"] == 64
|
||||
assert mil["alt_geom"] == 24500
|
||||
assert civ["role"] == "civilian"
|
||||
assert civ["emitter"] == "large"
|
||||
|
||||
|
||||
def test_transform_adsb_lol_military_from_icao_type_and_hex():
|
||||
payload = {
|
||||
"ac": [
|
||||
{"hex": "3b76aa", "flight": "FAF123", "t": "F16", "lat": 1, "lon": 2, "category": "A1"},
|
||||
{"hex": "ae1234", "flight": "BOXER1", "t": "C172", "lat": 1, "lon": 2, "category": "A1"},
|
||||
]
|
||||
}
|
||||
rows = {r["id"]: r for r in transform_adsb_lol(payload)}
|
||||
assert rows["3b76aa"]["extra"]["role"] == "military"
|
||||
assert rows["3b76aa"]["extra"]["role_src"] == "type"
|
||||
assert rows["ae1234"]["extra"]["role"] == "military"
|
||||
assert rows["ae1234"]["extra"]["role_src"] == "hex"
|
||||
|
||||
|
||||
def test_transform_ais_static_classifies_military_and_cargo():
|
||||
mil = transform_ais_frame({
|
||||
"MessageType": "ShipStaticData",
|
||||
"MetaData": {"MMSI": 338123456, "ShipName": "USNS BOB", "Latitude": 32.7, "Longitude": -117.2},
|
||||
"Message": {"ShipStaticData": {
|
||||
"Type": 35, "CallSign": "NBXX", "ImoNumber": 0,
|
||||
"Destination": "SAN DIEGO", "MaximumStaticDraught": 8.2,
|
||||
"Dimension": {"A": 80, "B": 20, "C": 8, "D": 8},
|
||||
"Eta": {"Month": 8, "Day": 29, "Hour": 14, "Minute": 0},
|
||||
}},
|
||||
})
|
||||
cargo = transform_ais_frame({
|
||||
"MessageType": "ShipStaticData",
|
||||
"MetaData": {"MMSI": 477123456, "ShipName": "EVER GIVEN", "Latitude": 36.9, "Longitude": -76.3},
|
||||
"Message": {"ShipStaticData": {
|
||||
"Type": 70, "CallSign": "VRXX", "ImoNumber": 9811000,
|
||||
"Destination": "NORFOLK", "MaximumStaticDraught": 14.5,
|
||||
"Dimension": {"A": 200, "B": 150, "C": 20, "D": 20},
|
||||
}},
|
||||
})
|
||||
assert mil is not None and cargo is not None
|
||||
assert mil["extra"]["role"] == "military"
|
||||
assert mil["extra"]["kind"] == "military"
|
||||
assert mil["extra"]["callsign"] == "NBXX"
|
||||
assert mil["extra"]["length"] == 100
|
||||
assert mil["extra"]["beam"] == 16
|
||||
assert mil["extra"]["dest"] == "SAN DIEGO"
|
||||
assert mil["extra"]["country"] == "United States"
|
||||
assert cargo["extra"]["role"] == "civilian"
|
||||
assert cargo["extra"]["kind"] == "cargo"
|
||||
assert cargo["extra"]["imo"] == 9811000
|
||||
|
||||
|
||||
def test_transform_ais_position_decodes_navstat():
|
||||
row = transform_ais_frame({
|
||||
"MessageType": "PositionReport",
|
||||
"MetaData": {"MMSI": 366912810, "ShipName": "EVER GIVEN", "latitude": 36.9, "longitude": -76.3},
|
||||
"Message": {"PositionReport": {"Sog": 0.1, "Cog": 88.0, "TrueHeading": 90, "NavigationalStatus": 5}},
|
||||
})
|
||||
assert row is not None
|
||||
assert row["extra"]["nav"] == "moored"
|
||||
assert row["extra"]["navstat"] == 5
|
||||
|
||||
|
||||
def test_nws_alerts_does_not_send_bbox_param(monkeypatch):
|
||||
"""api.weather.gov/alerts/active 400s on bbox — clip locally instead."""
|
||||
import asyncio
|
||||
|
||||
from live_layers import fetch_weather_alerts, _cache
|
||||
|
||||
seen = []
|
||||
|
||||
async def fake_get(url, params=None):
|
||||
seen.append((url, dict(params or {})))
|
||||
if "weather.gov" in url:
|
||||
return {
|
||||
"type": "FeatureCollection",
|
||||
"features": [{
|
||||
"type": "Feature",
|
||||
"properties": {"event": "Tornado Warning", "severity": "Extreme"},
|
||||
"geometry": {"type": "Point", "coordinates": [-78.7, 35.8]},
|
||||
}],
|
||||
}
|
||||
return {"type": "FeatureCollection", "features": []}
|
||||
|
||||
monkeypatch.setattr("live_layers._get_json", fake_get)
|
||||
_cache.clear()
|
||||
fc = asyncio.run(fetch_weather_alerts(None, "-79.0,35.5,-78.0,36.0"))
|
||||
nws_calls = [p for u, p in seen if "weather.gov" in u]
|
||||
assert nws_calls, "NWS should still be fetched"
|
||||
assert "bbox" not in nws_calls[0]
|
||||
assert fc.get("nws_ok") is True
|
||||
assert len(fc["features"]) == 1
|
||||
|
||||
|
||||
def test_nws_alerts_failure_is_flagged(monkeypatch):
|
||||
import asyncio
|
||||
|
||||
from live_layers import fetch_weather_alerts, _cache
|
||||
|
||||
async def fake_get(url, params=None):
|
||||
if "weather.gov" in url:
|
||||
raise RuntimeError("400 Bad Request")
|
||||
return {"type": "FeatureCollection", "features": []}
|
||||
|
||||
monkeypatch.setattr("live_layers._get_json", fake_get)
|
||||
_cache.clear()
|
||||
fc = asyncio.run(fetch_weather_alerts(None, None))
|
||||
assert fc.get("nws_ok") is False
|
||||
|
||||
|
||||
def test_fetch_aircraft_get_path_does_not_persist(monkeypatch):
|
||||
"""GET /api/aircraft must serve last-known without track/geofence writes."""
|
||||
import asyncio
|
||||
|
||||
from live_layers import (
|
||||
aircraft_last_known, fetch_aircraft, persist_aircraft_snapshot, _cache,
|
||||
)
|
||||
|
||||
aircraft_last_known.clear()
|
||||
aircraft_last_known["abc"] = {
|
||||
"id": "abc", "lat": 35.8, "lon": -78.7, "heading": 90, "speed": 400,
|
||||
"label": "ABC", "extra": {},
|
||||
}
|
||||
writes = {"n": 0}
|
||||
|
||||
async def boom(*a, **k):
|
||||
writes["n"] += 1
|
||||
raise AssertionError("GET path must not persist")
|
||||
|
||||
monkeypatch.setattr("tracks.record_position", boom)
|
||||
monkeypatch.setattr("geofence.record_and_notify", boom)
|
||||
_cache.clear()
|
||||
rows = asyncio.run(fetch_aircraft("-79,35,-78,36", persist=False))
|
||||
assert writes["n"] == 0
|
||||
assert any(r["id"] == "abc" for r in rows)
|
||||
|
||||
|
||||
def test_persist_aircraft_snapshot_writes_tracks(monkeypatch):
|
||||
import asyncio
|
||||
|
||||
from live_layers import persist_aircraft_snapshot
|
||||
|
||||
recorded = []
|
||||
|
||||
async def fake_record(kind, marker):
|
||||
recorded.append((kind, marker["id"]))
|
||||
return True
|
||||
|
||||
async def fake_gf(**kw):
|
||||
return 0
|
||||
|
||||
monkeypatch.setattr("tracks.record_position", fake_record)
|
||||
monkeypatch.setattr("geofence.record_and_notify", fake_gf)
|
||||
monkeypatch.setattr("ws_manager.manager.has_clients", lambda: False)
|
||||
|
||||
rows = [{
|
||||
"id": "abc", "lat": 35.8, "lon": -78.7, "heading": 90, "speed": 400,
|
||||
"label": "ABC", "extra": {},
|
||||
}]
|
||||
asyncio.run(persist_aircraft_snapshot(rows))
|
||||
assert recorded == [("aircraft", "abc")]
|
||||
|
||||
|
||||
# ── Planespotters.net photo lookup ────────────────────────────────────────
|
||||
|
||||
def test_normalize_planespotter_photo_prefers_large_thumbnail():
|
||||
from live_layers import _normalize_planespotter_photo
|
||||
|
||||
out = _normalize_planespotter_photo({
|
||||
"id": "1053982",
|
||||
"thumbnail": {"src": "https://t.plnspttrs.net/x_t.jpg", "size": {"width": 200, "height": 141}},
|
||||
"thumbnail_large": {"src": "https://t.plnspttrs.net/x_280.jpg", "size": {"width": 395, "height": 280}},
|
||||
"link": "https://www.planespotters.net/photo/1053982/foo",
|
||||
"photographer": "Günther Feniuk",
|
||||
})
|
||||
assert out["id"] == "1053982"
|
||||
assert out["src"] == "https://t.plnspttrs.net/x_280.jpg"
|
||||
assert out["width"] == 395
|
||||
assert out["height"] == 280
|
||||
assert out["photographer"] == "Günther Feniuk"
|
||||
assert "planespotters.net" in out["link"]
|
||||
|
||||
|
||||
def test_normalize_planespotter_photo_empty_or_malformed_returns_none():
|
||||
from live_layers import _normalize_planespotter_photo
|
||||
|
||||
assert _normalize_planespotter_photo({}) is None
|
||||
assert _normalize_planespotter_photo({"thumbnail": {}}) is None
|
||||
assert _normalize_planespotter_photo(None) is None
|
||||
assert _normalize_planespotter_photo("not-a-dict") is None
|
||||
|
||||
|
||||
def test_fetch_planespotters_photo_hex_builds_url_and_normalizes(monkeypatch):
|
||||
import asyncio
|
||||
|
||||
from live_layers import fetch_planespotters_photo, _cache
|
||||
|
||||
seen = []
|
||||
|
||||
async def fake_get(url, params=None, headers=None):
|
||||
seen.append((url, (headers or {}).get("User-Agent", "")))
|
||||
return {"photos": [{
|
||||
"id": "1", "thumbnail": {"src": "https://t.plnspttrs.net/a_t.jpg"},
|
||||
"thumbnail_large": {"src": "https://t.plnspttrs.net/a_280.jpg"},
|
||||
"link": "https://www.planespotters.net/photo/1/x", "photographer": "A",
|
||||
}]}
|
||||
|
||||
monkeypatch.setattr("live_layers._get_json", fake_get)
|
||||
_cache.clear()
|
||||
out = asyncio.run(fetch_planespotters_photo(hex_code="e8027e"))
|
||||
assert out["src"] == "https://t.plnspttrs.net/a_280.jpg"
|
||||
assert seen[0][0] == "https://api.planespotters.net/pub/photos/hex/e8027e"
|
||||
assert "@" in seen[0][1] or "http" in seen[0][1]
|
||||
|
||||
|
||||
def test_fetch_planespotters_photo_reg_fallback_and_no_result(monkeypatch):
|
||||
import asyncio
|
||||
|
||||
from live_layers import fetch_planespotters_photo, _cache
|
||||
|
||||
seen = []
|
||||
|
||||
async def fake_get(url, params=None, headers=None):
|
||||
seen.append(url)
|
||||
return {"photos": []}
|
||||
|
||||
monkeypatch.setattr("live_layers._get_json", fake_get)
|
||||
_cache.clear()
|
||||
assert asyncio.run(fetch_planespotters_photo(reg="D-ABCD")) is None
|
||||
assert seen == ["https://api.planespotters.net/pub/photos/reg/D-ABCD"]
|
||||
# no hex and no reg → no upstream call at all
|
||||
assert asyncio.run(fetch_planespotters_photo()) is None
|
||||
|
||||
|
||||
def test_planespotters_headers_add_contact_when_ua_is_generic(monkeypatch):
|
||||
import live_layers
|
||||
|
||||
monkeypatch.setattr(live_layers, "OSINT_USER_AGENT", "osint-dashboard/1.0 (self-hosted)")
|
||||
ua = live_layers._planespotters_headers()["User-Agent"]
|
||||
assert "osint-dashboard" in ua
|
||||
assert "@" in ua
|
||||
|
||||
|
||||
# ── Sentinel-1 SAR (Planetary Computer STAC → signed COG template) ────────
|
||||
|
||||
def test_sign_cog_url_appends_token():
|
||||
# PC returns the token pre-encoded as a query string; append verbatim.
|
||||
assert sign_cog_url("https://blob.example/x.tif", "st=s&se=e&sig=x%3D") == \
|
||||
"https://blob.example/x.tif?st=s&se=e&sig=x%3D"
|
||||
# Existing query string → append with &
|
||||
assert sign_cog_url("https://blob.example/x.tif?foo=1", "st=s&sig=x") == \
|
||||
"https://blob.example/x.tif?foo=1&st=s&sig=x"
|
||||
|
||||
|
||||
def test_sentinel1_tile_url_contains_titiler_rescale_and_cfastie():
|
||||
signed = "https://blob.example/x.tif?token=secret"
|
||||
url = sentinel1_tile_url(signed)
|
||||
assert url.startswith(TITILER_COG_TILES + "?")
|
||||
assert "WebMercatorQuad/{z}/{x}/{y}?" in url
|
||||
assert "url=https%3A%2F%2Fblob.example%2Fx.tif%3Ftoken%3Dsecret" in url
|
||||
assert "rescale=0%2C500" in url
|
||||
assert "colormap_name=cfastie" in url
|
||||
|
||||
|
||||
def test_sentinel1_tile_url_is_same_origin_relative():
|
||||
# Self-hosted TiTiler: the browser must hit the Pi's nginx vhost, not
|
||||
# titiler.xyz or a raw host:port. The template is a root-relative path.
|
||||
url = sentinel1_tile_url("https://blob.example/x.tif")
|
||||
assert url.startswith("/titiler/cog/tiles/WebMercatorQuad/")
|
||||
assert "://" not in url
|
||||
assert "titiler.xyz" not in url
|
||||
|
||||
|
||||
def _stac_feature(assets: dict) -> dict:
|
||||
return {
|
||||
"type": "Feature",
|
||||
"id": "S1A_IW_GRDH_1SDV_20240820T000000",
|
||||
"properties": {"datetime": "2024-08-20T00:00:00Z"},
|
||||
"assets": assets,
|
||||
}
|
||||
|
||||
|
||||
def test_fetch_sentinel1_vv_signed_tile_url(monkeypatch):
|
||||
import asyncio
|
||||
from live_layers import fetch_sentinel1, _cache
|
||||
|
||||
calls = []
|
||||
|
||||
async def fake_post(url, json=None, headers=None):
|
||||
calls.append(("post", url, json))
|
||||
return {"features": [_stac_feature({
|
||||
"vv": {"href": "https://blob.example/grd-vv.tif"},
|
||||
})]}
|
||||
|
||||
async def fake_get(url, params=None, headers=None):
|
||||
calls.append(("get", url))
|
||||
return {"token": "sig=abc123"}
|
||||
|
||||
monkeypatch.setattr("live_layers._post_json", fake_post)
|
||||
monkeypatch.setattr("live_layers._get_json", fake_get)
|
||||
_cache.clear()
|
||||
|
||||
out = asyncio.run(fetch_sentinel1("-80,35,-79,36"))
|
||||
assert out["id"] == "sentinel-1-sar"
|
||||
assert out["kind"] == "raster"
|
||||
assert out["polarization"] == "vv"
|
||||
assert out["opacity"] == 0.8
|
||||
assert out["itemId"].startswith("S1A")
|
||||
assert out["attribution"] == SENTINEL1_ATTRIBUTION
|
||||
assert "WebMercatorQuad/{z}/{x}/{y}?" in out["tileUrl"]
|
||||
assert "rescale=0%2C500" in out["tileUrl"]
|
||||
assert "colormap_name=cfastie" in out["tileUrl"]
|
||||
# SAS token "sig=abc123" is appended top-level, then the whole COG URL is
|
||||
# percent-encoded again as a query param (=> sig%3Dabc123).
|
||||
assert "sig%3Dabc123" in out["tileUrl"]
|
||||
# STAC search payload shape
|
||||
post_url, post_json = calls[0][1], calls[0][2]
|
||||
assert post_url.endswith("/api/stac/v1/search")
|
||||
assert post_json["collections"] == ["sentinel-1-grd"]
|
||||
assert post_json["limit"] >= 1
|
||||
assert post_json["sortby"][0]["direction"] == "desc"
|
||||
assert "bbox" in out
|
||||
|
||||
|
||||
def test_fetch_sentinel1_uses_hh_when_vv_missing(monkeypatch):
|
||||
import asyncio
|
||||
from live_layers import fetch_sentinel1, _cache
|
||||
|
||||
async def fake_post(url, json=None, headers=None):
|
||||
return {"features": [_stac_feature({
|
||||
"hh": {"href": "https://blob.example/grd-hh.tif"},
|
||||
})]}
|
||||
|
||||
async def fake_get(url, params=None, headers=None):
|
||||
return {"token": "tok"}
|
||||
|
||||
monkeypatch.setattr("live_layers._post_json", fake_post)
|
||||
monkeypatch.setattr("live_layers._get_json", fake_get)
|
||||
_cache.clear()
|
||||
|
||||
out = asyncio.run(fetch_sentinel1("-80,35,-79,36"))
|
||||
assert out["polarization"] == "hh"
|
||||
assert "url=https%3A%2F%2Fblob.example%2Fgrd-hh.tif" in out["tileUrl"]
|
||||
|
||||
|
||||
def test_fetch_sentinel1_none_on_empty_features(monkeypatch):
|
||||
import asyncio
|
||||
from live_layers import fetch_sentinel1, _cache
|
||||
|
||||
async def fake_post(url, json=None, headers=None):
|
||||
return {"features": []}
|
||||
|
||||
monkeypatch.setattr("live_layers._post_json", fake_post)
|
||||
_cache.clear()
|
||||
|
||||
assert asyncio.run(fetch_sentinel1("-80,35,-79,36")) is None
|
||||
|
||||
|
||||
def test_fetch_sentinel1_none_when_no_vv_or_hh(monkeypatch):
|
||||
import asyncio
|
||||
from live_layers import fetch_sentinel1, _cache
|
||||
|
||||
async def fake_post(url, json=None, headers=None):
|
||||
return {"features": [_stac_feature({"thumbnail": {"href": "https://x"}})]}
|
||||
|
||||
monkeypatch.setattr("live_layers._post_json", fake_post)
|
||||
_cache.clear()
|
||||
|
||||
assert asyncio.run(fetch_sentinel1("-80,35,-79,36")) is None
|
||||
|
||||
|
||||
def test_pick_sentinel_feature_prefers_scene_covering_center():
|
||||
features = [
|
||||
{"id": "far", "bbox": [10.0, 10.0, 12.0, 12.0]},
|
||||
{"id": "cover", "bbox": [-80.5, 34.5, -78.5, 36.5]},
|
||||
{"id": "also-far", "bbox": [-10.0, 0.0, -8.0, 2.0]},
|
||||
]
|
||||
picked = pick_sentinel_feature(features, -79.5, 35.5)
|
||||
assert picked["id"] == "cover"
|
||||
|
||||
|
||||
def test_pick_sentinel_feature_falls_back_to_first_when_none_cover():
|
||||
features = [
|
||||
{"id": "a", "bbox": [10.0, 10.0, 12.0, 12.0]},
|
||||
{"id": "b", "bbox": [20.0, 20.0, 22.0, 22.0]},
|
||||
]
|
||||
assert pick_sentinel_feature(features, -79.5, 35.5)["id"] == "a"
|
||||
assert pick_sentinel_feature([], -79.5, 35.5) is None
|
||||
|
|
|
|||
|
|
@ -1,128 +0,0 @@
|
|||
"""NASA EONET + CISA KEV parsers (no network)."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
|
||||
def test_parse_eonet_keeps_stable_ids_and_points():
|
||||
from sources import parse_eonet_events
|
||||
|
||||
payload = {
|
||||
"events": [
|
||||
{
|
||||
"id": "EONET_6363",
|
||||
"title": "Etna Volcano",
|
||||
"categories": [{"id": "volcanoes", "title": "Volcanoes"}],
|
||||
"geometry": [
|
||||
{"date": "2024-01-01T00:00:00Z", "type": "Point", "coordinates": [15.0, 37.7]},
|
||||
],
|
||||
"link": "https://eonet.gsfc.nasa.gov/api/v3/events/EONET_6363",
|
||||
},
|
||||
{
|
||||
"id": "EONET_skip",
|
||||
"title": "No geometry",
|
||||
"categories": [],
|
||||
"geometry": [],
|
||||
"link": "https://eonet.gsfc.nasa.gov/api/v3/events/EONET_skip",
|
||||
},
|
||||
]
|
||||
}
|
||||
events = parse_eonet_events(payload)
|
||||
assert len(events) == 1
|
||||
ev = events[0]
|
||||
assert ev["url"] == "https://eonet.gsfc.nasa.gov/api/v3/events/EONET_6363"
|
||||
assert ev["source_type"] == "disaster"
|
||||
assert ev["location_lat"] == 37.7
|
||||
assert ev["location_lon"] == 15.0
|
||||
assert ev["raw"]["eonet_id"] == "EONET_6363"
|
||||
assert "volcanoes" in ev["tags"]
|
||||
|
||||
|
||||
def test_parse_cisa_kev_emits_cve_url_no_coords():
|
||||
from sources import parse_cisa_kev
|
||||
|
||||
payload = {
|
||||
"vulnerabilities": [
|
||||
{
|
||||
"cveID": "CVE-2024-1234",
|
||||
"vendorProject": "Acme",
|
||||
"product": "Widget",
|
||||
"vulnerabilityName": "RCE",
|
||||
"dateAdded": "2024-06-01",
|
||||
"shortDescription": "Remote code execution",
|
||||
"requiredAction": "Apply updates",
|
||||
"dueDate": "2024-06-22",
|
||||
"knownRansomwareCampaignUse": "Known",
|
||||
}
|
||||
]
|
||||
}
|
||||
events = parse_cisa_kev(payload)
|
||||
assert len(events) == 1
|
||||
ev = events[0]
|
||||
assert ev["url"] == "https://nvd.nist.gov/vuln/detail/CVE-2024-1234"
|
||||
assert ev["location_lat"] is None
|
||||
assert ev["location_lon"] is None
|
||||
assert "cisa-kev" in ev["tags"]
|
||||
assert "CVE-2024-1234" in ev["tags"]
|
||||
assert ev["raw"]["cveID"] == "CVE-2024-1234"
|
||||
|
||||
|
||||
def test_ingest_cisa_kev_does_not_republish_known_nist_urls(monkeypatch):
|
||||
"""Producer must not push the whole KEV catalog to NATS every cycle."""
|
||||
import asyncio
|
||||
|
||||
from sources import ingest_cisa_kev
|
||||
|
||||
payload = {
|
||||
"vulnerabilities": [
|
||||
{
|
||||
"cveID": "CVE-2024-1111",
|
||||
"vulnerabilityName": "old",
|
||||
"dateAdded": "2024-01-01",
|
||||
"shortDescription": "already in db",
|
||||
},
|
||||
{
|
||||
"cveID": "CVE-2024-2222",
|
||||
"vulnerabilityName": "new",
|
||||
"dateAdded": "2024-06-01",
|
||||
"shortDescription": "not in db yet",
|
||||
},
|
||||
]
|
||||
}
|
||||
|
||||
class FakeResp:
|
||||
def raise_for_status(self):
|
||||
pass
|
||||
|
||||
def json(self):
|
||||
return payload
|
||||
|
||||
class FakeClient:
|
||||
def __init__(self, **kw):
|
||||
pass
|
||||
|
||||
async def __aenter__(self):
|
||||
return self
|
||||
|
||||
async def __aexit__(self, *exc):
|
||||
return False
|
||||
|
||||
async def get(self, url):
|
||||
return FakeResp()
|
||||
|
||||
published: list[str] = []
|
||||
|
||||
async def fake_publish(subject, event):
|
||||
published.append(event["url"])
|
||||
|
||||
known = {"https://nvd.nist.gov/vuln/detail/CVE-2024-1111"}
|
||||
|
||||
async def fake_existing(urls):
|
||||
return {u for u in urls if u in known}
|
||||
|
||||
monkeypatch.setattr("sources.httpx.AsyncClient", FakeClient)
|
||||
monkeypatch.setattr("sources.publish_event", fake_publish)
|
||||
monkeypatch.setattr("sources.existing_event_urls", fake_existing, raising=False)
|
||||
|
||||
n = asyncio.run(ingest_cisa_kev())
|
||||
assert n == 1
|
||||
assert published == ["https://nvd.nist.gov/vuln/detail/CVE-2024-2222"]
|
||||
|
|
@ -1,42 +0,0 @@
|
|||
"""News spider/pipeline: skip audio, use pubDate, don't log dupes as errors."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import sys
|
||||
from pathlib import Path
|
||||
|
||||
ROOT = Path(__file__).resolve().parent.parent
|
||||
_SCRAPER = ROOT / "news/scraper"
|
||||
if str(_SCRAPER) not in sys.path:
|
||||
sys.path.insert(0, str(_SCRAPER))
|
||||
|
||||
|
||||
def test_is_audio_enclosure():
|
||||
from newsScraper.feed_util import is_audio_url
|
||||
|
||||
assert is_audio_url("https://cdn.example/podcast.mp3") is True
|
||||
assert is_audio_url("https://cdn.example/show.m4a?x=1") is True
|
||||
assert is_audio_url("https://www.example.com/world/story") is False
|
||||
|
||||
|
||||
def test_article_timestamp_prefers_pubdate():
|
||||
from newsScraper.feed_util import article_timestamp
|
||||
|
||||
ts = article_timestamp("Tue, 01 Apr 2025 12:00:00 GMT")
|
||||
assert ts.tzinfo is not None
|
||||
assert ts.year == 2025
|
||||
assert ts.month == 4
|
||||
assert ts.day == 1
|
||||
|
||||
|
||||
def test_pipeline_does_not_wrap_dropitem_as_error():
|
||||
src = (ROOT / "news/scraper/newsScraper/pipelines.py").read_text()
|
||||
assert "except DropItem" in src
|
||||
assert "seen_urls.add" in src or "self.seen_urls.add" in src
|
||||
|
||||
|
||||
def test_spider_skips_audio_before_request():
|
||||
src = (ROOT / "news/scraper/newsScraper/spiders/news_spider.py").read_text()
|
||||
assert "is_audio_url" in src
|
||||
assert "article_timestamp" in src
|
||||
assert "datetime.datetime.now()" not in src
|
||||
|
|
@ -1,65 +0,0 @@
|
|||
"""Phase 1 guardrails: compose limits, 500ms debounce, no extra brokers."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from pathlib import Path
|
||||
|
||||
ROOT = Path(__file__).resolve().parent.parent
|
||||
|
||||
|
||||
def test_compose_memory_limits_and_shared_buffers():
|
||||
text = (ROOT / "docker-compose.yml").read_text()
|
||||
assert "shared_buffers=2GB" in text
|
||||
assert "shared_preload_libraries=timescaledb" in text
|
||||
assert "memory: 3G" in text or "memory: 3GB" in text
|
||||
assert "memory: 2G" in text or "memory: 2GB" in text
|
||||
|
||||
|
||||
def test_map_moveend_debounced_500ms():
|
||||
html = (ROOT / "app" / "static" / "index.html").read_text()
|
||||
assert "map.on('moveend'" in html
|
||||
# Existing 300ms debounce must be 500ms so pans don't spam bbox POSTs/WS.
|
||||
assert "}, 500);" in html
|
||||
assert "}, 300);" not in html.split("map.on('moveend'")[1][:800]
|
||||
|
||||
|
||||
def test_no_redis_kafka_celery():
|
||||
req = (ROOT / "app" / "requirements.txt").read_text().lower()
|
||||
compose = (ROOT / "docker-compose.yml").read_text().lower()
|
||||
for blob in (req, compose):
|
||||
assert "redis" not in blob
|
||||
assert "kafka" not in blob
|
||||
assert "celery" not in blob
|
||||
assert "cachetools" in req
|
||||
|
||||
|
||||
def test_titiler_image_pinned_by_digest():
|
||||
text = (ROOT / "docker-compose.yml").read_text()
|
||||
assert (
|
||||
"ghcr.io/developmentseed/titiler:latest@sha256:"
|
||||
"1809958d063543e3ec858259536002b2de78e9f8f09a22a8d9591bdc2b550b14"
|
||||
in text
|
||||
)
|
||||
# Unpinned :latest would drift on every pull.
|
||||
for line in text.splitlines():
|
||||
if "titiler" in line.lower() and "image:" in line:
|
||||
assert "@sha256:" in line
|
||||
|
||||
|
||||
def test_uvicorn_single_worker_guard():
|
||||
text = (ROOT / "app" / "main.py").read_text()
|
||||
main_block = text.split('if __name__ == "__main__":', 1)[1]
|
||||
assert "workers=1" in main_block
|
||||
|
||||
|
||||
def test_bbox_gist_migration_keeps_btree_and_adds_gist():
|
||||
text = (ROOT / "alembic" / "versions" / "010_bbox_gist.py").read_text()
|
||||
assert "down_revision" in text and "009_vessels" in text
|
||||
assert "ix_events_geom_gist" in text
|
||||
assert "ix_fires_geom_gist" in text
|
||||
assert "ST_MakePoint(location_lon, location_lat)" in text
|
||||
assert "ST_MakePoint(longitude, latitude)" in text
|
||||
assert "USING gist" in text
|
||||
models = (ROOT / "app" / "models.py").read_text()
|
||||
assert 'Index("ix_events_location"' in models
|
||||
assert 'Index("ix_fires_bbox"' in models
|
||||
Some files were not shown because too many files have changed in this diff Show more
Loading…
Add table
Reference in a new issue