Compare commits

..

No commits in common. "master" and "fix/news-ticker-hud" have entirely different histories.

109 changed files with 1565 additions and 14896 deletions

View file

@ -31,6 +31,20 @@ NOMINATIM_URL=https://nominatim.openstreetmap.org
NOMINATIM_MIN_INTERVAL=1.1 NOMINATIM_MIN_INTERVAL=1.1
SNAPSHOT_TTL_SECONDS=300 SNAPSHOT_TTL_SECONDS=300
# ── masscan active camera discovery (host-level systemd service, NOT compose) ─
# Continuous rolling sweep for open RTSP port 554 across a range. Runs on the
# Pi host via deploy/osint-masscan.service (needs root + raw sockets). Results
# land in the same `cameras` table as the scraper (discovery_source=masscan).
# NOTE: 200 pps is the residential-safe default. 1k/10k pps saturated a home
# uplink. A full 0.0.0.0/0 sweep at 200 pps takes ~8 months (rolling).
MASSCAN_RANGE=0.0.0.0/0
MASSCAN_PORTS=554
MASSCAN_RATE=200
MASSCAN_RETRIES=1
MASSCAN_WAIT=0
MASSCAN_EXCLUDEFILE=/etc/osint-dashboard/masscan-excludes.txt
MASSCAN_FLUSH_EVERY=250
# ── NASA FIRMS (active fire / hotspot ingest) ────────────────────────────── # ── NASA FIRMS (active fire / hotspot ingest) ──────────────────────────────
# MAP_KEY is FREE — get one at https://firms.modaps.eosdis.nasa.gov/api/map_key_info/ # MAP_KEY is FREE — get one at https://firms.modaps.eosdis.nasa.gov/api/map_key_info/
# (1-minute signup, no payment). Leave blank to keep fire ingest idle. # (1-minute signup, no payment). Leave blank to keep fire ingest idle.
@ -45,65 +59,35 @@ FIRMS_INTERVAL=900
# Set to 0 to disable the fire loop entirely. # Set to 0 to disable the fire loop entirely.
INGEST_FIRES=1 INGEST_FIRES=1
# ── VesselAPI (commercial REST AIS — Hormuz, 5×/day, 150 calls/mo cap) ─────
# Independent of AISStream (open/shared live US-coast WebSocket). Both stay
# on when their keys are set; missing one never disables the other.
# Prefer pasting VESSELAPI_API_KEY on the dashboard Keys page.
# The poller idles when the key is unset. Never called from map pans
# (GET /api/vessels serves the shared last-known cache only).
VESSELAPI_API_KEY=
# Bounding box(es) as minlat,minlon,maxlat,maxlon (lat/lon order). Semicolon-
# separated for multiple boxes. Default = Strait of Hormuz (span 3.6 ≤ 4° cap).
VESSELAPI_BBOX=25.5,55.4,27.3,57.2
# Poll cadence in seconds (17280 = 4.8h → 5 polls/day = 150/mo).
VESSELAPI_INTERVAL=17280
# Local hard cap on successful 2xx calls per UTC day (persisted in Postgres).
VESSELAPI_MAX_CALLS_PER_DAY=5
# 1 = run the poller inside the dashboard process (default); ingester off.
VESSELAPI_IN_APP=1
VESSELAPI_IN_INGEST=0
# ── API keys (managed from the dashboard UI) ────────────────────────────── # ── API keys (managed from the dashboard UI) ──────────────────────────────
# Keys such as NOUS_API_KEY and TELEGRAM_TOKEN are stored in the Postgres # Keys such as GEMINI_API_KEY and TELEGRAM_TOKEN are stored in the Postgres
# `api_keys` table and managed from the dashboard's "Keys" tab # `api_keys` table and managed from the dashboard's "Keys" tab
# (GET/POST/DELETE /api/keys/{name}) — see app/keystore.py. The FIRMS ingestor # (GET/POST/DELETE /api/keys/{name}) — see app/keystore.py. The FIRMS ingestor
# currently reads FIRMS_MAP_KEY from .env (above); wiring the Keys-UI store as # currently reads FIRMS_MAP_KEY from .env (above); wiring the Keys-UI store as
# its lookup/fallback is a planned follow-up. # its lookup/fallback is a planned follow-up.
# ── News pipeline (scraper + summarizer, profile `ingest`) ──────────────── # ── News pipeline (scraper + summarizer, profile `ingest`) ────────────────
# Scraper crawls urls.txt continuously (default 10s between crawls). # Hourly: the scraper crawls 257 RSS sources at minute :00 and the summarizer
# Summarizer runs Nous map-reduce every 15 min (NEWS_SUMMARIZE_INTERVAL_S=900). # runs the Gemini map-reduce at minute :05, both writing to the shared osint-db
# NOUS_API_KEY is also (preferably) set in the Keys UI; env is an override. # (tables `articles` + `article_summaries`, created by alembic 003_news).
# Unset in both env and api_keys = summarizer logs and idles (never crashes). # Consume via GET /api/news and GET /api/news/summaries.
NOUS_API_KEY= # GEMINI_API_KEY is REQUIRED for summarization; unset = summarizer idles.
NOUS_BASE_URL=https://inference-api.nousresearch.com/v1 GEMINI_API_KEY=
# Optional LLM knobs # Optional LLM knobs
# SUMMARY_MODEL is an optional override. Leave unset so Settings SUMMARY_MODEL=gemini-2.0-flash
# (app_settings.SUMMARY_MODEL) can reach the summarizer. Code default
# Hermes-4.3-36B remains after a Postgres miss. Env wins when set.
# SUMMARY_MODEL=
NEWS_BATCH_SIZE=50 NEWS_BATCH_SIZE=50
SUMMARY_WINDOW_MINUTES=15 SUMMARY_WINDOW_HOURS=1
# Futures/markets coupling from the upstream pipeline is OFF by default # Futures/markets coupling from the upstream pipeline is OFF by default
# (irrelevant to OSINT). Set INCLUDE_FUTURES=1 + install yfinance to enable. # (irrelevant to OSINT). Set INCLUDE_FUTURES=1 + install yfinance to enable.
INCLUDE_FUTURES=0 INCLUDE_FUTURES=0
NEWS_SCRAPE_INTERVAL_S=10 # Wall-clock scheduling (k8s CronJob replacement): scrape minute, summarize minute
NEWS_SUMMARIZE_INTERVAL_S=900 NEWS_SCRAPE_MINUTE=0
NEWS_SUMMARIZE_MINUTE=5
# Run once immediately on container start (seeds data fast), then align to the
# scheduled minute.
NEWS_SCRAPE_RUN_ON_START=1 NEWS_SCRAPE_RUN_ON_START=1
NEWS_SUMMARIZE_RUN_ON_START=1 NEWS_SUMMARIZE_RUN_ON_START=1
# "1" ignores the interval idempotency skip (double-pins on recreate).
NEWS_SUMMARIZE_FORCE=0
NEWS_LOG_LEVEL=INFO NEWS_LOG_LEVEL=INFO
# Reserved for the (out-of-scope) Telegram delivery bot. # Reserved for the (out-of-scope) Telegram delivery bot.
TELEGRAM_TOKEN= TELEGRAM_TOKEN=
TELEGRAM_CHAT_ID= TELEGRAM_CHAT_ID=
# =============================================================================
# Forgejo container registry (CI publishes here; NOT ghcr.io)
# =============================================================================
# forgejo.siriusdevops.com/sirius/osint-dashboard[:tag]
# forgejo.siriusdevops.com/sirius/osint-dashboard-pg[:tag]
# forgejo.siriusdevops.com/sirius/osint-news-scraper[:tag]
# forgejo.siriusdevops.com/sirius/osint-news-summarizer[:tag]
FORGEJO_REGISTRY=forgejo.siriusdevops.com
FORGEJO_OWNER=sirius

View file

@ -1,249 +1,27 @@
# Build changed OSINT images, publish to the Forgejo container registry, then
# redeploy on the Pi runner (docker.sock mounted).
#
# Unchanged images are skipped. Dockerfile.pg / osint-dashboard-pg is NOT
# rebuilt or pulled on a normal merge — Postgres stays up. Rebuild it only
# when Dockerfile.pg changes, or via workflow_dispatch rebuild_pg.
#
# Public pull host: forgejo.siriusdevops.com (NOT ghcr.io)
# CI push host: 127.0.0.1:3000 — Cloudflare 413s layers ≳100MB on the public
# hostname, even from the Pi (hairpins out through the tunnel).
# Images (public names):
# forgejo.siriusdevops.com/sirius/osint-dashboard
# forgejo.siriusdevops.com/sirius/osint-dashboard-pg
# forgejo.siriusdevops.com/sirius/osint-news-scraper
# forgejo.siriusdevops.com/sirius/osint-news-summarizer
#
# Optional repo variable FORGEJO_REGISTRY overrides the *public* pull host.
# Deploy still uses the local docker socket on the runner host (rpi).
name: build-and-deploy name: build-and-deploy
on: on:
push: push:
branches: [main, master] branches: [main, master]
workflow_dispatch:
inputs:
rebuild_pg:
description: Rebuild Timescale+PostGIS (Dockerfile.pg)
type: boolean
default: false
rebuild_all:
description: Rebuild every app image (ignore path filter)
type: boolean
default: false
env:
PUBLIC_REGISTRY: ${{ vars.FORGEJO_REGISTRY || 'forgejo.siriusdevops.com' }}
PUSH_REGISTRY: 127.0.0.1:3000
OWNER: sirius
# Keep compose project/volumes stable on the Pi
COMPOSE_PROJECT_NAME: osint-dashboard
jobs: jobs:
build-push-deploy: build:
runs-on: docker runs-on: docker
permissions:
contents: read
packages: write
steps: steps:
- name: Checkout - name: Checkout
uses: https://code.forgejo.org/actions/checkout@v4 uses: https://code.forgejo.org/actions/checkout@v4
with: - name: Build and deploy on the Pi (local docker)
fetch-depth: 50
- name: Plan image builds
id: plan
run: |
set -euo pipefail
APP=0
SCRAPER=0
SUM=0
PG=0
COMPOSE=0
mark() {
case "$1" in
Dockerfile.pg)
PG=1 ;;
Dockerfile|app/*|alembic/*|alembic.ini)
APP=1 ;;
news/scraper/*)
SCRAPER=1 ;;
news/summerizer/*)
SUM=1 ;;
docker-compose.yml|scripts/compose-reup.sh)
COMPOSE=1 ;;
esac
}
if [ "${{ github.event_name }}" = "workflow_dispatch" ]; then
APP=1; SCRAPER=1; SUM=1
if [ "${{ github.event.inputs.rebuild_all }}" = "true" ]; then
APP=1; SCRAPER=1; SUM=1; PG=1
fi
if [ "${{ github.event.inputs.rebuild_pg }}" = "true" ]; then
PG=1
fi
else
BEFORE="${{ github.event.before }}"
SHA="${GITHUB_SHA}"
ZEROS="0000000000000000000000000000000000000000"
if [ -z "$BEFORE" ] || [ "$BEFORE" = "$ZEROS" ]; then
echo "No previous SHA — build app images, skip pg"
APP=1; SCRAPER=1; SUM=1
elif ! git cat-file -e "${BEFORE}^{commit}" 2>/dev/null; then
echo "Previous SHA $BEFORE not in history — build app images, skip pg"
APP=1; SCRAPER=1; SUM=1
else
while IFS= read -r f; do
[ -z "$f" ] && continue
mark "$f"
done < <(git diff --name-only "$BEFORE" "$SHA")
fi
fi
{
echo "app=$APP"
echo "scraper=$SCRAPER"
echo "summarizer=$SUM"
echo "pg=$PG"
echo "compose=$COMPOSE"
} >> "$GITHUB_OUTPUT"
echo "plan app=$APP scraper=$SCRAPER summarizer=$SUM pg=$PG compose=$COMPOSE"
- name: Image refs
id: img
run: |
set -euo pipefail
PUSH="${PUSH_REGISTRY}"
PUB="${PUBLIC_REGISTRY}"
OWN="${OWNER}"
SHA="${GITHUB_SHA::12}"
{
echo "reg=$PUSH"
echo "pub=$PUB"
echo "sha=$SHA"
echo "app=$PUSH/$OWN/osint-dashboard"
echo "pg=$PUSH/$OWN/osint-dashboard-pg"
echo "scraper=$PUSH/$OWN/osint-news-scraper"
echo "summarizer=$PUSH/$OWN/osint-news-summarizer"
} >> "$GITHUB_OUTPUT"
echo "Push registry: $PUSH"
echo "Public pull: $PUB"
echo "SHA tag: $SHA"
- name: Login to Forgejo registry
if: steps.plan.outputs.app == '1' || steps.plan.outputs.scraper == '1' || steps.plan.outputs.summarizer == '1' || steps.plan.outputs.pg == '1'
run: |
set -euo pipefail
# GITHUB_TOKEN login "succeeds" but blob uploads 401 (Forgejo packages
# reject the Actions token). Use a user PAT with write:package.
if [ -z "${{ secrets.FORGEJO_TOKEN }}" ]; then
echo "Missing repo secret FORGEJO_TOKEN (user PAT, not GITHUB_TOKEN)"
exit 1
fi
echo "${{ secrets.FORGEJO_TOKEN }}" | docker login "${{ steps.img.outputs.reg }}" \
-u sirius --password-stdin
- name: Build application image (api / ingester / cameras)
if: steps.plan.outputs.app == '1'
run: |
set -ex
APP="${{ steps.img.outputs.app }}"
SHA="${{ steps.img.outputs.sha }}"
docker build -f Dockerfile -t "${APP}:latest" -t "${APP}:${SHA}" \
-t "localhost/osint-dashboard:latest" .
docker push "${APP}:latest"
docker push "${APP}:${SHA}"
- name: Build news-scraper image
if: steps.plan.outputs.scraper == '1'
run: |
set -ex
IMG="${{ steps.img.outputs.scraper }}"
SHA="${{ steps.img.outputs.sha }}"
docker build -f news/scraper/Dockerfile -t "${IMG}:latest" -t "${IMG}:${SHA}" \
-t "localhost/osint-news-scraper:latest" news/scraper
docker push "${IMG}:latest"
docker push "${IMG}:${SHA}"
- name: Build news-summarizer image
if: steps.plan.outputs.summarizer == '1'
run: |
set -ex
IMG="${{ steps.img.outputs.summarizer }}"
SHA="${{ steps.img.outputs.sha }}"
docker build -f news/summerizer/Dockerfile -t "${IMG}:latest" -t "${IMG}:${SHA}" \
-t "localhost/osint-news-summarizer:latest" news/summerizer
docker push "${IMG}:latest"
docker push "${IMG}:${SHA}"
- name: Build / refresh Timescale+PostGIS image
if: steps.plan.outputs.pg == '1'
run: |
set -ex
PG="${{ steps.img.outputs.pg }}"
SHA="${{ steps.img.outputs.sha }}"
if docker build -f Dockerfile.pg -t "${PG}:latest" -t "${PG}:${SHA}" \
-t "localhost/osint-dashboard-pg:latest" .; then
docker push "${PG}:latest"
docker push "${PG}:${SHA}"
elif docker image inspect "localhost/osint-dashboard-pg:latest" >/dev/null 2>&1; then
echo "WARN: Dockerfile.pg build failed; keeping existing local pg image"
else
echo "ERROR: cannot build or find osint-dashboard-pg image"
exit 1
fi
- name: Deploy on runner host (compose)
run: | run: |
set -ex set -ex
# The forgejo-runner runs on the Pi with /var/run/docker.sock and
# /opt/siriusdevops mounted, so we build + deploy LOCALLY — no SSH/scp.
# Deploy straight from the checked-out workspace.
cd "${GITHUB_WORKSPACE}" cd "${GITHUB_WORKSPACE}"
APP="${{ steps.plan.outputs.app }}" # Project name is pinned by `name:` in docker-compose.yml
SCRAPER="${{ steps.plan.outputs.scraper }}" # (osint-dashboard), matching the live volume
SUM="${{ steps.plan.outputs.summarizer }}" # osint-dashboard_osint-pgdata. Do NOT override COMPOSE_PROJECT_NAME
PG="${{ steps.plan.outputs.pg }}" # here — a different name would create a fresh empty pgdata volume.
COMPOSE="${{ steps.plan.outputs.compose }}" docker compose build --no-cache
docker compose up -d --force-recreate
SVCS=()
[ "$APP" = "1" ] && SVCS+=(app ingester camera-service)
[ "$SCRAPER" = "1" ] && SVCS+=(news-scraper)
[ "$SUM" = "1" ] && SVCS+=(news-summarizer)
if [ "$COMPOSE" = "1" ]; then
# compose/script change: bounce workers so env/command updates apply.
# Still do not bounce Postgres.
for s in app ingester camera-service news-scraper news-summarizer; do
case " ${SVCS[*]} " in
*" $s "*) ;;
*) SVCS+=("$s") ;;
esac
done
fi
chmod +x scripts/compose-reup.sh
if [ "$PG" = "1" ]; then
FORCE_RECREATE_DB=1 COMPOSE_PROJECT_NAME=osint-dashboard COMPOSE_PROFILES=ingest \
scripts/compose-reup.sh "${SVCS[@]}" db
elif [ "${#SVCS[@]}" -gt 0 ]; then
COMPOSE_PROJECT_NAME=osint-dashboard COMPOSE_PROFILES=ingest \
scripts/compose-reup.sh "${SVCS[@]}"
else
echo "No image or compose changes — leave running containers alone"
docker compose --profile ingest ps
fi
docker image prune -f docker image prune -f
echo "osint-dashboard deploy done; db image left in place unless pg=1" echo 'osint-dashboard deployed (local)'
- name: Summary # deployed locally on the Pi via runner (docker socket + cli-plugins mounted)
if: always()
run: |
{
echo "## Image plan"
echo "- app: \`${{ steps.plan.outputs.app }}\`"
echo "- news-scraper: \`${{ steps.plan.outputs.scraper }}\`"
echo "- news-summarizer: \`${{ steps.plan.outputs.summarizer }}\`"
echo "- pg (Timescale): \`${{ steps.plan.outputs.pg }}\`"
echo
echo "Postgres is rebuilt/pulled only when \`Dockerfile.pg\` changes (or workflow_dispatch rebuild_pg)."
} >> "$GITHUB_STEP_SUMMARY"

View file

@ -1,64 +0,0 @@
"""news_items table + article_summaries.model
Revision ID: 005_news_items
Revises: 004_camera_enum
Create Date: 2026-08-28
"""
from alembic import op
# revision identifiers, used by Alembic.
revision = "005_news_items"
down_revision = "004_camera_enum"
branch_labels = None
depends_on = None
def upgrade() -> None:
# Idempotent DDL: ingest (summarizer ensure_tables) may create the same
# shapes first depending on container startup order. IF NOT EXISTS makes
# both orders safe — whichever runs first wins, the other no-ops.
# One statement per op.execute: asyncpg rejects multi-command prepared
# statements (same style as 003_news).
op.execute(
"""
ALTER TABLE article_summaries
ADD COLUMN IF NOT EXISTS model TEXT
"""
)
op.execute(
"""
CREATE TABLE IF NOT EXISTS news_items (
id SERIAL PRIMARY KEY,
summary_id INTEGER REFERENCES article_summaries(id) ON DELETE CASCADE,
kind TEXT NOT NULL,
headline TEXT NOT NULL,
importance TEXT NOT NULL,
location_name TEXT,
lat DOUBLE PRECISION,
lon DOUBLE PRECISION,
location_confidence TEXT,
category TEXT,
url TEXT,
created_at TIMESTAMPTZ NOT NULL DEFAULT NOW()
)
"""
)
op.execute(
"""
CREATE INDEX IF NOT EXISTS ix_news_items_kind_created
ON news_items (kind, created_at DESC)
"""
)
op.execute(
"""
CREATE INDEX IF NOT EXISTS ix_news_items_map_bbox
ON news_items (lon, lat)
WHERE kind = 'map' AND lat IS NOT NULL AND lon IS NOT NULL
"""
)
def downgrade() -> None:
op.execute("DROP TABLE IF EXISTS news_items")
op.execute("ALTER TABLE article_summaries DROP COLUMN IF EXISTS model")

View file

@ -1,196 +0,0 @@
"""phase 2: geofences, 1-min track CAGGs, fire/aircraft hits
Revision ID: 005_phase2
Revises: 004_camera_enum
Create Date: 2026-08-28
"""
from alembic import op
revision = "005_phase2"
down_revision = "004_camera_enum"
branch_labels = None
depends_on = None
def upgrade() -> None:
op.execute("CREATE EXTENSION IF NOT EXISTS postgis")
op.execute("CREATE EXTENSION IF NOT EXISTS timescaledb")
op.execute("""
CREATE TABLE IF NOT EXISTS geofences (
id UUID PRIMARY KEY,
name TEXT NOT NULL,
geojson JSONB NOT NULL,
geom geometry(Polygon, 4326),
active INTEGER NOT NULL DEFAULT 1,
created_at TIMESTAMPTZ NOT NULL DEFAULT now(),
updated_at TIMESTAMPTZ NOT NULL DEFAULT now()
)
""")
op.execute("""
CREATE INDEX IF NOT EXISTS ix_geofences_geom
ON geofences USING gist (geom)
""")
op.execute("""
CREATE TABLE IF NOT EXISTS geofence_alerts (
id UUID PRIMARY KEY,
geofence_id UUID NOT NULL,
source_kind TEXT NOT NULL,
entity_id TEXT NOT NULL,
lat DOUBLE PRECISION,
lon DOUBLE PRECISION,
payload JSONB,
created_at TIMESTAMPTZ NOT NULL DEFAULT now()
)
""")
op.execute("""
CREATE INDEX IF NOT EXISTS ix_geofence_alerts_created
ON geofence_alerts (created_at DESC)
""")
op.execute("""
CREATE TABLE IF NOT EXISTS vessel_positions (
mmsi TEXT NOT NULL,
ts TIMESTAMPTZ NOT NULL,
lat DOUBLE PRECISION NOT NULL,
lon DOUBLE PRECISION NOT NULL,
heading DOUBLE PRECISION,
speed DOUBLE PRECISION,
label TEXT,
extra JSONB,
PRIMARY KEY (mmsi, ts)
)
""")
op.execute("""
SELECT create_hypertable(
'vessel_positions', 'ts', if_not_exists => TRUE
)
""")
op.execute("""
CREATE INDEX IF NOT EXISTS ix_vessel_positions_bbox
ON vessel_positions (lon, lat)
""")
op.execute("""
CREATE TABLE IF NOT EXISTS aircraft_positions (
hex TEXT NOT NULL,
ts TIMESTAMPTZ NOT NULL,
lat DOUBLE PRECISION NOT NULL,
lon DOUBLE PRECISION NOT NULL,
heading DOUBLE PRECISION,
speed DOUBLE PRECISION,
label TEXT,
extra JSONB,
PRIMARY KEY (hex, ts)
)
""")
op.execute("""
SELECT create_hypertable(
'aircraft_positions', 'ts', if_not_exists => TRUE
)
""")
op.execute("""
CREATE INDEX IF NOT EXISTS ix_aircraft_positions_bbox
ON aircraft_positions (lon, lat)
""")
op.execute("""
CREATE TABLE IF NOT EXISTS fire_aircraft_hits (
id UUID NOT NULL,
fire_id TEXT NOT NULL,
fire_lat DOUBLE PRECISION NOT NULL,
fire_lon DOUBLE PRECISION NOT NULL,
aircraft_hex TEXT NOT NULL,
aircraft_type TEXT,
aircraft_lat DOUBLE PRECISION NOT NULL,
aircraft_lon DOUBLE PRECISION NOT NULL,
distance_mi DOUBLE PRECISION NOT NULL,
seen_at TIMESTAMPTZ NOT NULL,
PRIMARY KEY (fire_id, aircraft_hex, seen_at)
)
""")
op.execute("""
CREATE INDEX IF NOT EXISTS ix_fire_aircraft_hits_seen
ON fire_aircraft_hits (seen_at DESC)
""")
# 1-minute continuous aggregates (Timescale). last() keeps the newest
# sample in each bucket — the DVR slider reads these, not the raw table.
op.execute("""
DO $$
BEGIN
IF NOT EXISTS (
SELECT 1 FROM timescaledb_information.continuous_aggregates
WHERE view_name = 'vessel_tracks_1min'
) THEN
EXECUTE $v$
CREATE MATERIALIZED VIEW vessel_tracks_1min
WITH (timescaledb.continuous) AS
SELECT time_bucket('1 minute', ts) AS bucket,
mmsi,
last(lat, ts) AS lat,
last(lon, ts) AS lon,
last(heading, ts) AS heading,
last(speed, ts) AS speed,
last(label, ts) AS label
FROM vessel_positions
GROUP BY bucket, mmsi
WITH NO DATA
$v$;
END IF;
IF NOT EXISTS (
SELECT 1 FROM timescaledb_information.continuous_aggregates
WHERE view_name = 'aircraft_tracks_1min'
) THEN
EXECUTE $a$
CREATE MATERIALIZED VIEW aircraft_tracks_1min
WITH (timescaledb.continuous) AS
SELECT time_bucket('1 minute', ts) AS bucket,
hex,
last(lat, ts) AS lat,
last(lon, ts) AS lon,
last(heading, ts) AS heading,
last(speed, ts) AS speed,
last(label, ts) AS label
FROM aircraft_positions
GROUP BY bucket, hex
WITH NO DATA
$a$;
END IF;
END
$$;
""")
op.execute("""
DO $$
BEGIN
PERFORM add_continuous_aggregate_policy(
'vessel_tracks_1min',
start_offset => INTERVAL '3 hours',
end_offset => INTERVAL '1 minute',
schedule_interval => INTERVAL '1 minute',
if_not_exists => TRUE
);
PERFORM add_continuous_aggregate_policy(
'aircraft_tracks_1min',
start_offset => INTERVAL '3 hours',
end_offset => INTERVAL '1 minute',
schedule_interval => INTERVAL '1 minute',
if_not_exists => TRUE
);
EXCEPTION WHEN OTHERS THEN
NULL;
END
$$;
""")
def downgrade() -> None:
op.execute("DROP MATERIALIZED VIEW IF EXISTS aircraft_tracks_1min CASCADE")
op.execute("DROP MATERIALIZED VIEW IF EXISTS vessel_tracks_1min CASCADE")
op.execute("DROP TABLE IF EXISTS fire_aircraft_hits")
op.execute("DROP TABLE IF EXISTS aircraft_positions")
op.execute("DROP TABLE IF EXISTS vessel_positions")
op.execute("DROP TABLE IF EXISTS geofence_alerts")
op.execute("DROP TABLE IF EXISTS geofences")

View file

@ -1,24 +0,0 @@
"""merge 005_phase2 and 005_news_items
Revision ID: 006_merge_heads
Revises: 005_phase2, 005_news_items
Create Date: 2026-08-28
Two PRs both parented 004_camera_enum (phase2 geofences + news_items).
`alembic upgrade head` then fails and crash-loops the dashboard.
"""
from alembic import op # noqa: F401
revision = "006_merge_heads"
down_revision = ("005_phase2", "005_news_items")
branch_labels = None
depends_on = None
def upgrade() -> None:
pass
def downgrade() -> None:
pass

View file

@ -1,115 +0,0 @@
"""event_dedup + Timescale compression/retention
Revision ID: 007_event_dedup
Revises: 006_merge_heads
Create Date: 2026-08-28
"""
from alembic import op
revision = "007_event_dedup"
down_revision = "006_merge_heads"
branch_labels = None
depends_on = None
def upgrade() -> None:
op.execute("""
CREATE TABLE IF NOT EXISTS event_dedup (
url TEXT PRIMARY KEY,
created_at TIMESTAMPTZ NOT NULL DEFAULT now()
)
""")
# Keep the earliest row per URL, drop the 10× USGS/camera dupes.
op.execute("""
DELETE FROM events a
USING events b
WHERE a.url IS NOT NULL AND a.url <> ''
AND a.url = b.url
AND (a.ingested_at, a.id) > (b.ingested_at, b.id)
""")
op.execute("""
INSERT INTO event_dedup (url)
SELECT DISTINCT url FROM events
WHERE url IS NOT NULL AND url <> ''
ON CONFLICT (url) DO NOTHING
""")
# Compression + retention. Policies no-op if Timescale rejects (fresh PG).
op.execute("""
DO $$
BEGIN
PERFORM add_compression_policy('events', INTERVAL '7 days', if_not_exists => TRUE);
EXCEPTION WHEN OTHERS THEN
BEGIN
ALTER TABLE events SET (
timescaledb.compress,
timescaledb.compress_segmentby = 'source_type',
timescaledb.compress_orderby = 'ingested_at DESC'
);
PERFORM add_compression_policy('events', INTERVAL '7 days', if_not_exists => TRUE);
EXCEPTION WHEN OTHERS THEN
NULL;
END;
END
$$;
""")
op.execute("""
DO $$
BEGIN
PERFORM add_retention_policy('events', INTERVAL '180 days', if_not_exists => TRUE);
EXCEPTION WHEN OTHERS THEN
NULL;
END
$$;
""")
op.execute("""
DO $$
BEGIN
ALTER TABLE fires SET (
timescaledb.compress,
timescaledb.compress_segmentby = 'satellite',
timescaledb.compress_orderby = 'acq_time DESC'
);
PERFORM add_compression_policy('fires', INTERVAL '7 days', if_not_exists => TRUE);
PERFORM add_retention_policy('fires', INTERVAL '90 days', if_not_exists => TRUE);
EXCEPTION WHEN OTHERS THEN
NULL;
END
$$;
""")
op.execute("""
DO $$
BEGIN
ALTER TABLE aircraft_positions SET (
timescaledb.compress,
timescaledb.compress_segmentby = 'hex',
timescaledb.compress_orderby = 'ts DESC'
);
PERFORM add_compression_policy('aircraft_positions', INTERVAL '1 day', if_not_exists => TRUE);
PERFORM add_retention_policy('aircraft_positions', INTERVAL '14 days', if_not_exists => TRUE);
EXCEPTION WHEN OTHERS THEN
NULL;
END
$$;
""")
op.execute("""
DO $$
BEGIN
ALTER TABLE vessel_positions SET (
timescaledb.compress,
timescaledb.compress_segmentby = 'mmsi',
timescaledb.compress_orderby = 'ts DESC'
);
PERFORM add_compression_policy('vessel_positions', INTERVAL '1 day', if_not_exists => TRUE);
PERFORM add_retention_policy('vessel_positions', INTERVAL '14 days', if_not_exists => TRUE);
EXCEPTION WHEN OTHERS THEN
NULL;
END
$$;
""")
def downgrade() -> None:
op.execute("DROP TABLE IF EXISTS event_dedup")

View file

@ -1,26 +0,0 @@
"""article_summaries.kind — interval vs daily_recap
Revision ID: 008_summary_kind
Revises: 007_event_dedup
Create Date: 2026-08-28
"""
from alembic import op
revision = "008_summary_kind"
down_revision = "007_event_dedup"
branch_labels = None
depends_on = None
def upgrade() -> None:
op.execute(
"""
ALTER TABLE article_summaries
ADD COLUMN IF NOT EXISTS kind TEXT
"""
)
def downgrade() -> None:
op.execute("ALTER TABLE article_summaries DROP COLUMN IF EXISTS kind")

View file

@ -1,41 +0,0 @@
"""vessels — daily VesselAPI snapshots for DVR as-of
Revision ID: 009_vessels
Revises: 008_summary_kind
Create Date: 2026-08-29
"""
from alembic import op
revision = "009_vessels"
down_revision = "008_summary_kind"
branch_labels = None
depends_on = None
def upgrade() -> None:
op.execute(
"""
CREATE TABLE IF NOT EXISTS vessels (
mmsi TEXT NOT NULL,
poll_at TIMESTAMPTZ NOT NULL,
lat DOUBLE PRECISION NOT NULL,
lon DOUBLE PRECISION NOT NULL,
heading DOUBLE PRECISION,
speed DOUBLE PRECISION,
label TEXT,
extra JSONB,
PRIMARY KEY (mmsi, poll_at)
)
"""
)
op.execute(
"CREATE INDEX IF NOT EXISTS ix_vessels_poll_at ON vessels (poll_at DESC)"
)
op.execute(
"CREATE INDEX IF NOT EXISTS ix_vessels_bbox ON vessels (lon, lat)"
)
def downgrade() -> None:
op.execute("DROP TABLE IF EXISTS vessels")

View file

@ -1,29 +0,0 @@
"""GIST bbox indexes for events/fires map-pan queries.
Revision ID: 010_bbox_gist
Revises: 009_vessels
Create Date: 2026-09-01
"""
from alembic import op
revision = "010_bbox_gist"
down_revision = "009_vessels"
branch_labels = None
depends_on = None
def upgrade() -> None:
op.execute(
"CREATE INDEX IF NOT EXISTS ix_events_geom_gist ON events "
"USING gist (ST_SetSRID(ST_MakePoint(location_lon, location_lat), 4326))"
)
op.execute(
"CREATE INDEX IF NOT EXISTS ix_fires_geom_gist ON fires "
"USING gist (ST_SetSRID(ST_MakePoint(longitude, latitude), 4326))"
)
def downgrade() -> None:
op.execute("DROP INDEX IF EXISTS ix_fires_geom_gist")
op.execute("DROP INDEX IF EXISTS ix_events_geom_gist")

View file

@ -1,24 +0,0 @@
"""geofence_alerts (geofence_id, created_at DESC) for fence-scoped hit log
Revision ID: 011_geofence_alerts_fence
Revises: 010_bbox_gist
Create Date: 2026-09-01
"""
from alembic import op
revision = "011_geofence_alerts_fence"
down_revision = "010_bbox_gist"
branch_labels = None
depends_on = None
def upgrade() -> None:
op.execute(
"CREATE INDEX IF NOT EXISTS ix_geofence_alerts_fence_created "
"ON geofence_alerts (geofence_id, created_at DESC)"
)
def downgrade() -> None:
op.execute("DROP INDEX IF EXISTS ix_geofence_alerts_fence_created")

View file

@ -14,7 +14,6 @@ import json
import logging import logging
import os import os
import random import random
import time
from config import AISSTREAM_API_KEY, AISSTREAM_BBOX from config import AISSTREAM_API_KEY, AISSTREAM_BBOX
from keystore import get_api_key from keystore import get_api_key
@ -30,36 +29,6 @@ FILTER_TYPES = [
"ShipStaticData", "ShipStaticData",
] ]
# ── Viewport-following ─────────────────────────────────────────────────────
# The frontend POSTs its current viewport box to /api/vessels/subscribe; the
# worker retunes the AISStream subscription to it (throttled to 1/s, the
# service's subscription-update cap). ``None`` keeps the env AISSTREAM_BBOX
# default. Last writer wins; the key never reaches the browser.
_desired_boxes: list[list[list[float]]] | None = None
_bbox_guard = asyncio.Lock()
async def request_viewport_bbox(
minlon: float, minlat: float, maxlon: float, maxlat: float
) -> None:
"""Retune the live subscription to a viewport box (lon/lat input order)."""
global _desired_boxes
box = [[minlat, minlon], [maxlat, maxlon]] # AISStream wants [lat, lon] corners
async with _bbox_guard:
_desired_boxes = [box]
async def reset_viewport_bbox() -> None:
"""Fall back to the env AISSTREAM_BBOX default."""
global _desired_boxes
async with _bbox_guard:
_desired_boxes = None
async def _take_desired_boxes() -> list[list[list[float]]] | None:
async with _bbox_guard:
return _desired_boxes
def _parse_boxes(raw: str) -> list[list[list[float]]]: def _parse_boxes(raw: str) -> list[list[list[float]]]:
"""Env format: minlat,minlon,maxlat,maxlon[; ...]. AIS wants [[lat,lon],[lat,lon]].""" """Env format: minlat,minlon,maxlat,maxlon[; ...]. AIS wants [[lat,lon],[lat,lon]]."""
@ -96,7 +65,7 @@ async def run_ais_worker() -> None:
) )
await asyncio.sleep(60) await asyncio.sleep(60)
continue continue
boxes = await _take_desired_boxes() or _parse_boxes(AISSTREAM_BBOX) boxes = _parse_boxes(AISSTREAM_BBOX)
try: try:
async with websockets.connect( async with websockets.connect(
WS_URL, WS_URL,
@ -113,28 +82,7 @@ async def run_ais_worker() -> None:
await ws.send(json.dumps(sub)) await ws.send(json.dumps(sub))
logger.info("AISStream subscribed (%d bbox(es))", len(boxes)) logger.info("AISStream subscribed (%d bbox(es))", len(boxes))
backoff = 2.0 backoff = 2.0
last_submit = time.monotonic() async for raw in ws:
while True:
# Follow the client viewport: coalesce to the latest request
# and honor AISStream's 1 subscription-update/s cap.
desired = await _take_desired_boxes()
if (
desired is not None
and desired != boxes
and time.monotonic() - last_submit >= 1.0
):
boxes = desired
sub["BoundingBoxes"] = boxes
await ws.send(json.dumps(sub))
last_submit = time.monotonic()
logger.info(
"AISStream re-subscribed to viewport (%d bbox)",
len(boxes),
)
try:
raw = await asyncio.wait_for(ws.recv(), timeout=0.25)
except asyncio.TimeoutError:
continue
if isinstance(raw, bytes): if isinstance(raw, bytes):
raw = raw.decode("utf-8", errors="replace") raw = raw.decode("utf-8", errors="replace")
try: try:

View file

@ -1,72 +0,0 @@
"""Background ffmpeg — never block a FastAPI request on a frame grab.
ffmpeg frame grabs are scheduled with asyncio.create_task and shared per URL.
"""
from __future__ import annotations
import asyncio
import shutil
from cachetools import TTLCache
_ffmpeg_cache: TTLCache = TTLCache(maxsize=100, ttl=300)
_ffmpeg_tasks: dict[str, asyncio.Task] = {}
_FFMPEG = shutil.which("ffmpeg")
def cached_ffmpeg_jpeg(url: str) -> bytes | None:
return _ffmpeg_cache.get(url)
def schedule_ffmpeg_snapshot(url: str, timeout: float = 8.0) -> asyncio.Task:
"""Start (or reuse) an ffmpeg JPEG grab. Caller may await the task."""
existing = _ffmpeg_tasks.get(url)
if existing is not None and not existing.done():
return existing
task = asyncio.create_task(_ffmpeg_grab_and_cache(url, timeout))
_ffmpeg_tasks[url] = task
return task
async def _ffmpeg_grab(url: str, timeout: float = 8.0) -> bytes | None:
"""Grab one JPEG frame. Isolated so tests can stub it."""
if not _FFMPEG or not url.lower().startswith("rtsp://"):
return None
cmd = [
_FFMPEG, "-hide_banner", "-loglevel", "error", "-nostdin",
"-rtsp_transport", "tcp",
"-timeout", "4000000",
"-i", url,
"-frames:v", "1",
"-f", "image2pipe", "-vcodec", "mjpeg",
"pipe:1",
]
try:
proc = await asyncio.create_subprocess_exec(
*cmd,
stdout=asyncio.subprocess.PIPE,
stderr=asyncio.subprocess.DEVNULL,
)
except FileNotFoundError:
return None
try:
stdout, _ = await asyncio.wait_for(proc.communicate(), timeout=timeout)
except asyncio.TimeoutError:
proc.kill()
try:
await proc.wait()
except Exception: # noqa: BLE001
pass
return None
if proc.returncode not in (0, None) or not stdout or len(stdout) < 64:
return None
if stdout[:2] != b"\xff\xd8":
return None
return stdout
async def _ffmpeg_grab_and_cache(url: str, timeout: float) -> bytes | None:
data = await _ffmpeg_grab(url, timeout)
if data:
_ffmpeg_cache[url] = data
return data

View file

@ -16,8 +16,6 @@ CALTRANS_CCTV_URLS = tuple(
f"https://cwwp2.dot.ca.gov/data/d{n}/cctv/cctvStatusD{n:02d}.json" f"https://cwwp2.dot.ca.gov/data/d{n}/cctv/cctvStatusD{n:02d}.json"
for n in range(1, 13) for n in range(1, 13)
) )
# MDOT MiDrive official DOT CCTV list (fields carry rendered HTML).
MDOT_CAMERA_URL = "https://mdotjboss.state.mi.us/MiDrive/camera/list"
_DEFAULT_SOURCE_URL = ",".join(( _DEFAULT_SOURCE_URL = ",".join((
# Publicly published open-camera list (markdown bullets of stream URLs). # Publicly published open-camera list (markdown bullets of stream URLs).
"https://raw.githubusercontent.com/fury999io/public-ip-cams/main/README.md", "https://raw.githubusercontent.com/fury999io/public-ip-cams/main/README.md",
@ -27,10 +25,6 @@ _DEFAULT_SOURCE_URL = ",".join((
"https://raw.githubusercontent.com/willytop8/Live-Environment-Streams/main/streams.geojson", "https://raw.githubusercontent.com/willytop8/Live-Environment-Streams/main/streams.geojson",
# Official Caltrans CWWP2 JPEG + HLS CCTV (districts 112). # Official Caltrans CWWP2 JPEG + HLS CCTV (districts 112).
*CALTRANS_CCTV_URLS, *CALTRANS_CCTV_URLS,
# Oregon DOT TripCheck public CCTV JPEG inventory (Esri JSON).
"https://www.tripcheck.com/Scripts/map/data/cctvinventory.js",
# Official MDOT MiDrive CCTV (JPEG stills, Michigan).
MDOT_CAMERA_URL,
)) ))
CAMERA_SOURCE_URLS = [ CAMERA_SOURCE_URLS = [
u.strip() u.strip()
@ -61,15 +55,3 @@ SNAPSHOT_TIMEOUT = float(os.getenv("SNAPSHOT_TIMEOUT", "8.0"))
# NATS subject cameras are published on (consumed by the shared ingester). # NATS subject cameras are published on (consumed by the shared ingester).
CAMERA_NATS_SUBJECT = os.getenv("CAMERA_NATS_SUBJECT", "events.camera") CAMERA_NATS_SUBJECT = os.getenv("CAMERA_NATS_SUBJECT", "events.camera")
# ── UDOT IBI 511 traffic cameras ──────────────────────────────────────────
# DataTables endpoint (POST form-encoded; server caps at 100 rows/page no
# matter what `length` is sent). No API key. Snapshot stills live at a stable
# /map/Cctv/{id} URL — same URL always serves the latest frame, so we store
# the URL and never scrape every frame ourselves.
UDOT_IBI_URL = "https://prod-ut.ibi511.com/List/GetData/Cameras"
UDOT_IBI_BASE = "https://prod-ut.ibi511.com"
UDOT_IBI_PAGE_SIZE = 100
# Safety cap on pages per cycle so a runaway recordsTotal cannot fan out.
UDOT_IBI_MAX_PAGES = int(os.getenv("UDOT_IBI_MAX_PAGES", "40"))

View file

@ -1,7 +1,7 @@
"""Resolve a browser-renderable preview for a camera. """Resolve a browser-renderable preview for a camera.
HTTP/MJPEG cameras already expose a snapshot_url the existing proxy can HTTP/MJPEG cameras already expose a snapshot_url the existing proxy can
stream. Some scraper sources store `rtsp://` URLs with no snapshot_url, so stream. masscan finds are stored as `rtsp://IP/` with no snapshot_url, so
the map popup used to skip the <img> entirely and the leftover source link the map popup used to skip the <img> entirely and the leftover source link
handed the browser an rtsp:// URL (which opens VLC). handed the browser an rtsp:// URL (which opens VLC).
@ -42,6 +42,16 @@ _HTTP_PATHS = (
"/tmpfs/auto.jpg", "/tmpfs/auto.jpg",
) )
# Browser-playable MJPEG paths the /stream proxy can pass through.
_MJPEG_PATHS = (
"/mjpg/video.mjpg",
"/video.mjpg",
"/cgi-bin/mjpg/video.cgi",
"/axis-cgi/mjpg/video.cgi",
"/nphMotionJpeg",
"/mjpeg.cgi",
)
_FFMPEG = shutil.which("ffmpeg") _FFMPEG = shutil.which("ffmpeg")
@ -77,18 +87,88 @@ async def _http_get_image(url: str, timeout: float = 2.5) -> bytes | None:
return None return None
async def ffmpeg_snapshot(url: str, timeout: float = 8.0) -> bytes | None: async def _http_feed_url(url: str, timeout: float = 2.5) -> str | None:
"""Grab a single JPEG frame from an RTSP URL. None if ffmpeg missing/fails. """Return url if it looks like an unauthenticated image/MJPEG feed."""
try:
async with httpx.AsyncClient(
timeout=timeout, follow_redirects=True,
headers={"User-Agent": USER_AGENT},
) as c:
async with c.stream("GET", url) as r:
if r.status_code != 200:
return None
ctype = (r.headers.get("content-type") or "").lower()
if "html" in ctype or ctype.startswith("text/"):
return None
if any(x in ctype for x in ("image/", "multipart", "mjpeg", "octet-stream")):
# Read a little to reject empty/error bodies.
chunk = b""
async for b in r.aiter_bytes():
chunk += b
if len(chunk) >= 64:
break
if len(chunk) < 64:
return None
if b"html" in chunk[:64].lower():
return None
return url
except Exception: # noqa: BLE001
return None
return None
The subprocess is scheduled via asyncio.create_task (shared per URL) so
concurrent popup clicks do not stack ffmpeg processes on the request path. async def probe_public_feed(host: str) -> str | None:
"""Unauthenticated HTTP still or MJPEG URL for this host, or None.
Used at masscan ingest time so dead RTSP-only hosts never hit the map.
No credentials, no RTSP path-walking (too slow / rarely public).
""" """
from bg_jobs import cached_ffmpeg_jpeg, schedule_ffmpeg_snapshot urls = [f"http://{host}{p}" for p in _HTTP_PATHS]
urls.append(f"http://{host}:8080/shot.jpg")
urls.extend(f"http://{host}{p}" for p in _MJPEG_PATHS)
results = await asyncio.gather(
*(_http_feed_url(u) for u in urls),
return_exceptions=True,
)
for url, hit in zip(urls, results):
if isinstance(hit, str) and hit:
return hit
return None
hit = cached_ffmpeg_jpeg(url)
if hit: async def ffmpeg_snapshot(url: str, timeout: float = 8.0) -> bytes | None:
return hit """Grab a single JPEG frame from an RTSP URL. None if ffmpeg missing/fails."""
return await schedule_ffmpeg_snapshot(url, timeout) if not _FFMPEG or not url.lower().startswith("rtsp://"):
return None
cmd = [
_FFMPEG, "-hide_banner", "-loglevel", "error", "-nostdin",
"-rtsp_transport", "tcp",
"-timeout", "4000000", # 4s socket timeout, microseconds
"-i", url,
"-frames:v", "1",
"-f", "image2pipe", "-vcodec", "mjpeg",
"pipe:1",
]
try:
proc = await asyncio.create_subprocess_exec(
*cmd,
stdout=asyncio.subprocess.PIPE,
stderr=asyncio.subprocess.DEVNULL,
)
except FileNotFoundError:
return None
try:
stdout, _ = await asyncio.wait_for(proc.communicate(), timeout=timeout)
except asyncio.TimeoutError:
proc.kill()
try:
await proc.wait()
except Exception: # noqa: BLE001
pass
return None
if proc.returncode not in (0, None) or not _looks_like_jpeg(stdout or b""):
return None
return stdout
async def ffmpeg_mjpeg_stream(url: str): async def ffmpeg_mjpeg_stream(url: str):

View file

@ -25,7 +25,6 @@ import hashlib
import ipaddress import ipaddress
import json import json
import logging import logging
import math
import re import re
import time import time
from datetime import datetime, timezone from datetime import datetime, timezone
@ -38,7 +37,6 @@ from camera_config import (
CAMERA_SOURCE_URLS, CAMERA_REQUEST_DELAY, CAMERA_MAX_PER_SOURCE, CAMERA_SOURCE_URLS, CAMERA_REQUEST_DELAY, CAMERA_MAX_PER_SOURCE,
NOMINATIM_URL, NOMINATIM_MIN_INTERVAL, USER_AGENT, NOMINATIM_URL, NOMINATIM_MIN_INTERVAL, USER_AGENT,
SNAPSHOT_CACHE_DIR, SNAPSHOT_TTL_SECONDS, SNAPSHOT_TIMEOUT, SNAPSHOT_CACHE_DIR, SNAPSHOT_TTL_SECONDS, SNAPSHOT_TIMEOUT,
UDOT_IBI_URL, UDOT_IBI_BASE, UDOT_IBI_PAGE_SIZE, UDOT_IBI_MAX_PAGES,
) )
from camera_models import cameras from camera_models import cameras
from database import async_session from database import async_session
@ -116,15 +114,6 @@ class RateLimitedClient:
self._last[host] = time.monotonic() self._last[host] = time.monotonic()
return await self.client.get(url, **kw) return await self.client.get(url, **kw)
async def post(self, url: str, **kw) -> httpx.Response:
host = urlparse(url).netloc
now = time.monotonic()
wait = self._last.get(host, 0.0) + self._delay - now
if wait > 0:
await asyncio.sleep(wait)
self._last[host] = time.monotonic()
return await self.client.post(url, **kw)
async def aclose(self): async def aclose(self):
await self.client.aclose() await self.client.aclose()
@ -390,201 +379,6 @@ def parse_caltrans_json(text: str, source_name: str) -> list[dict]:
return out return out
# ── UDOT IBI 511 ──────────────────────────────────────────────────────────
# Utah bbox (lat 36.942.1, lon -114.2-108.9). WKT is `POINT (lng lat)`.
_UDOT_IBI_MIN_LAT, _UDOT_IBI_MAX_LAT = 36.9, 42.1
_UDOT_IBI_MIN_LON, _UDOT_IBI_MAX_LON = -114.2, -108.9
_UDOT_WKT_POINT_RE = re.compile(
r"POINT\s*\(\s*(-?\d+(?:\.\d+)?)\s+(-?\d+(?:\.\d+)?)\s*\)", re.I,
)
def parse_udot_ibi_page(text: str, source_name: str = "udot") -> list[dict]:
"""Parse one UDOT IBI 511 DataTables camera page (`{"data": [...]}`).
Skips rows whose first image is `blocked` or `disabled`, and drops any
point outside the Utah bbox. The `/map/Cctv/{id}` URL is a stable identity
(always serves the latest frame), so it is stored as both source_url and
snapshot_url we never scrape frames ourselves.
"""
try:
payload = json.loads(text)
except (json.JSONDecodeError, ValueError):
return []
rows = payload.get("data") if isinstance(payload, dict) else None
if not isinstance(rows, list):
return []
out: list[dict] = []
for row in rows:
if not isinstance(row, dict):
continue
cam_id = row.get("id")
images = row.get("images") or []
if cam_id is None or not images:
continue
img = images[0] or {}
if img.get("blocked") or img.get("disabled"):
continue
lon = lat = None
try:
wkt = (row.get("latLng") or {}).get("geography") or {}
wkt = wkt.get("wellKnownText") or ""
m = _UDOT_WKT_POINT_RE.match(str(wkt).strip())
if m:
lon, lat = float(m.group(1)), float(m.group(2))
except (AttributeError, TypeError, ValueError):
lon = lat = None
if lat is None or lon is None:
continue
if not (_UDOT_IBI_MIN_LAT <= lat <= _UDOT_IBI_MAX_LAT
and _UDOT_IBI_MIN_LON <= lon <= _UDOT_IBI_MAX_LON):
continue
snap = f"{UDOT_IBI_BASE}/map/Cctv/{cam_id}"
roadway, direction, location = (
row.get("roadway"), row.get("direction"), row.get("location"),
)
name = ", ".join(
str(b) for b in (roadway, direction, location)
if b and str(b).strip() and str(b).strip().lower() != "unknown"
) or None
out.append({
"source_url": snap,
"snapshot_url": snap,
"discovery_source": source_name,
"location_lat": lat,
"location_lon": lon,
"location_name": name,
"vendor": "UDOT",
"device_type": "http",
"raw": {
"udot_id": cam_id,
"agency": row.get("source"),
"source_id": row.get("sourceId"),
"roadway": roadway,
"direction": direction,
},
})
return out
# Oregon DOT TripCheck inventory bounding box (approx state extent).
ODOT_BBOX = (41.9, 46.3, -124.6, -116.4) # lat_min, lat_max, lon_min, lon_max
def parse_odot_json(text: str, source_name: str) -> list[dict]:
"""Parse Oregon DOT TripCheck cctvinventory Esri-style JSON.
Store the JPEG still as snapshot_url (map thumbs); never RTSP. Keep only
rows with finite coordinates inside Oregon and a usable filename.
"""
try:
payload = json.loads(text)
except (json.JSONDecodeError, ValueError):
return []
lat_min, lat_max, lon_min, lon_max = ODOT_BBOX
out: list[dict] = []
for feat in payload.get("features") or []:
attrs = (feat or {}).get("attributes") or {}
filename = (attrs.get("filename") or "").strip()
if not filename:
continue
try:
lat = float(attrs.get("latitude"))
lon = float(attrs.get("longitude"))
except (TypeError, ValueError):
continue
if not (math.isfinite(lat) and math.isfinite(lon)):
continue
if not (lat_min <= lat <= lat_max and lon_min <= lon <= lon_max):
continue
jpeg = f"https://tripcheck.com/RoadCams/cams/{filename}"
title = (attrs.get("title") or "").strip()
out.append({
"source_url": jpeg,
"snapshot_url": jpeg,
"discovery_source": "odot",
"location_lat": lat,
"location_lon": lon,
"location_name": title or None,
"vendor": "ODOT",
"device_type": "http",
})
return out
# MDOT MiDrive field extractors (fields carry rendered HTML).
_MDOT_LAT_RE = re.compile(r"lat=(-?\d+(?:\.\d+)?)", re.I)
_MDOT_LON_RE = re.compile(r"lon=(-?\d+(?:\.\d+)?)", re.I)
_MDOT_ID_RE = re.compile(r"[?&]id=(\d+)", re.I)
_MDOT_IMG_RE = re.compile(r'<img[^>]+src=["\']([^"\']+)["\']', re.I)
# Michigan bbox (docs/osiris-ideas.md §3.2): lat 41.648.3, lon -90.5-82.1.
MDOT_LAT_RANGE = (41.6, 48.3)
MDOT_LON_RANGE = (-90.5, -82.1)
def parse_mdot_json(text: str, source_name: str) -> list[dict]:
"""Parse MDOT MiDrive `camera/list` JSON (fields carry rendered HTML).
Coordinates and the stable id live in the `county` field's map link
(`/MiDrive/map?...lat=&lon=&id=`); the `image` field carries an `<img>`
whose src is the JPEG still. Out-of-bbox and coord-less rows are dropped.
"""
try:
payload = json.loads(text)
except (json.JSONDecodeError, ValueError):
return []
if not isinstance(payload, list):
return []
out: list[dict] = []
for row in payload:
if not isinstance(row, dict):
continue
county_html = row.get("county") or ""
m_lat = _MDOT_LAT_RE.search(county_html)
m_lon = _MDOT_LON_RE.search(county_html)
m_id = _MDOT_ID_RE.search(county_html)
if not (m_lat and m_lon and m_id):
continue # missing coordinates / stable id → drop
try:
lat = float(m_lat.group(1))
lon = float(m_lon.group(1))
except ValueError:
continue
if not (MDOT_LAT_RANGE[0] <= lat <= MDOT_LAT_RANGE[1]
and MDOT_LON_RANGE[0] <= lon <= MDOT_LON_RANGE[1]):
continue # out of Michigan bbox → drop
img_m = _MDOT_IMG_RE.search(row.get("image") or "")
if not img_m:
continue
snap = img_m.group(1).strip()
low = snap.lower()
if not (low.startswith("http://") or low.startswith("https://")):
continue
if low.startswith("rtsp"):
continue
cam_id = m_id.group(1)
route = (row.get("route") or "").strip()
loc = (row.get("location") or "").strip().lstrip("@").strip()
county_name = county_html.split("<a", 1)[0].strip()
bits = [
f"{route} @ {loc}" if (route and loc) else (route or loc or None),
county_name or None,
]
name = ", ".join(b for b in bits if b) or None
out.append({
"source_url": f"https://mdotjboss.state.mi.us/MiDrive/camera/{cam_id}",
"snapshot_url": snap,
"discovery_source": "mdot",
"location_lat": lat,
"location_lon": lon,
"location_name": name,
"vendor": "MDOT",
"device_type": "http",
})
return out
def parse_live_streams_geojson(text: str, source_name: str) -> list[dict]: def parse_live_streams_geojson(text: str, source_name: str) -> list[dict]:
"""Parse willytop8/Live-Environment-Streams GeoJSON. """Parse willytop8/Live-Environment-Streams GeoJSON.
@ -696,10 +490,6 @@ async def scrape_source(client: RateLimitedClient, geo: Geocoder,
body = resp.text body = resp.text
if "cwwp2.dot.ca.gov" in src_url or "cctvStatus" in src_url: if "cwwp2.dot.ca.gov" in src_url or "cctvStatus" in src_url:
cams = parse_caltrans_json(body, name) cams = parse_caltrans_json(body, name)
elif "cctvinventory" in src_url or "tripcheck.com" in src_url:
cams = parse_odot_json(body, name)
elif "mdotjboss.state.mi.us" in src_url or "/MiDrive/camera/list" in src_url:
cams = parse_mdot_json(body, name)
elif ("getCameraDataByLoc" in src_url elif ("getCameraDataByLoc" in src_url
or ("json" in ctype and '"locs"' in body[:4000] and '"cams"' in body[:8000])): or ("json" in ctype and '"locs"' in body[:4000] and '"cams"' in body[:8000])):
cams = parse_alertwest_json(body, name) cams = parse_alertwest_json(body, name)
@ -759,54 +549,6 @@ async def scrape_source(client: RateLimitedClient, geo: Geocoder,
return out return out
# ── UDOT IBI 511 paginated fetcher ────────────────────────────────────────
async def scrape_udot_ibi(client: RateLimitedClient) -> list[dict]:
"""Page through the UDOT IBI 511 DataTables endpoint and normalize.
POSTs `start`/`length` form fields (server caps at 100 rows/page), walking
pages until `recordsTotal` is exhausted or UDOT_IBI_MAX_PAGES is hit.
"""
out: list[dict] = []
seen: set[str] = set()
start = 0
for _ in range(UDOT_IBI_MAX_PAGES):
try:
resp = await client.post(
UDOT_IBI_URL,
data={
"start": str(start),
"length": str(UDOT_IBI_PAGE_SIZE),
"lang": "en-US",
},
headers={"X-Requested-With": "XMLHttpRequest"},
)
resp.raise_for_status()
body = resp.text
except Exception: # noqa: BLE001
logger.exception("failed to fetch UDOT IBI page start=%d", start)
break
try:
payload = json.loads(body)
except ValueError:
logger.warning("UDOT IBI non-JSON response at start=%d", start)
break
total = int(payload.get("recordsTotal") or 0)
rows = payload.get("data") or []
if not isinstance(rows, list) or not rows:
break
for cam in parse_udot_ibi_page(body, "udot"):
if cam["source_url"] in seen:
continue
seen.add(cam["source_url"])
out.append(cam)
if start + len(rows) >= total:
break
start += len(rows)
logger.info("UDOT IBI yielded %d cameras", len(out))
return out
# ── Persistence ──────────────────────────────────────────────────────────── # ── Persistence ────────────────────────────────────────────────────────────
async def upsert_cameras(cams: list[dict]) -> int: async def upsert_cameras(cams: list[dict]) -> int:
@ -858,7 +600,6 @@ async def run_cycle() -> int:
try: try:
results = await asyncio.gather( results = await asyncio.gather(
*(scrape_source(client, geo, s) for s in CAMERA_SOURCE_URLS), *(scrape_source(client, geo, s) for s in CAMERA_SOURCE_URLS),
scrape_udot_ibi(client),
return_exceptions=True, return_exceptions=True,
) )
all_cams: list[dict] = [] all_cams: list[dict] = []

View file

@ -1,62 +0,0 @@
"""Static chokepoint preset catalog — one-tap fly-to targets for the map.
Pure data, no upstream calls and no VesselAPI quota spend. ``vesselapi`` is
``True`` only for Hormuz (the single box the VesselAPI poller already covers);
every other strait is AISStream-only until a human later spends quota. Never
call VesselAPI from here.
Bounding boxes are ``minlat,minlon,maxlat,maxlon`` (VesselAPI order) and each
stays within the ``|dLat|+|dLon| <= 4`` span rule enforced by
``vesselapi.validate_bbox_span``.
"""
from __future__ import annotations
# id → preset. ``center`` is ``[lat, lon]`` for Leaflet ``setView``.
_CHOKEPOINTS: tuple[dict, ...] = (
{
"id": "hormuz",
"title": "Strait of Hormuz",
"bbox": "25.5,55.4,27.3,57.2",
"center": [26.4, 56.5],
"zoom": 9,
"vesselapi": True,
},
{
"id": "bab_el_mandeb",
"title": "Bab el-Mandeb",
"bbox": "12.0,42.8,13.5,44.3",
"center": [12.7, 43.4],
"zoom": 9,
"vesselapi": False,
},
{
"id": "suez",
"title": "Suez / N. Red Sea",
"bbox": "29.5,32.0,31.0,33.5",
"center": [30.0,32.5],
"zoom": 9,
"vesselapi": False,
},
{
"id": "malacca",
"title": "Malacca / Singapore",
"bbox": "1.0,103.0,2.5,104.5",
"center": [1.3, 103.8],
"zoom": 9,
"vesselapi": False,
},
{
"id": "taiwan",
"title": "Taiwan Strait",
"bbox": "23.5,119.0,25.0,120.5",
"center": [24.2, 119.8],
"zoom": 9,
"vesselapi": False,
},
)
def chokepoints() -> list[dict]:
"""Return a fresh copy of the catalog (callers must not mutate the source)."""
return [dict(p) for p in _CHOKEPOINTS]

View file

@ -71,16 +71,6 @@ FIRMS_DATASETS = [d.strip() for d in _FIRMS_DATASETS_RAW.split(",") if d.strip()
OSINT_USER_AGENT = os.getenv( OSINT_USER_AGENT = os.getenv(
"OSINT_USER_AGENT", "osint-dashboard/1.0 (self-hosted; lancewalters94@gmail.com)" "OSINT_USER_AGENT", "osint-dashboard/1.0 (self-hosted; lancewalters94@gmail.com)"
) )
# Nominatim reverse (GET /api/place). Camera scraper has its own copy in camera_config.
NOMINATIM_URL = os.getenv("NOMINATIM_URL", "https://nominatim.openstreetmap.org")
NOMINATIM_MIN_INTERVAL = float(os.getenv("NOMINATIM_MIN_INTERVAL", "1.0"))
# Self-hosted TiTiler (warps Sentinel-1 signed COGs into XYZ tiles on the Pi).
# TITILER_PUBLIC_BASE is the same-origin path prefix the browser hits through
# the osint.rpi.local nginx vhost (`location /titiler/` → 127.0.0.1:8001).
# TITILER_INTERNAL_URL is the compose-DNS address, used only for health checks.
TITILER_PUBLIC_BASE = os.getenv("TITILER_PUBLIC_BASE", "/titiler").rstrip("/")
TITILER_INTERNAL_URL = os.getenv("TITILER_INTERNAL_URL", "http://titiler:8000")
# AISStream (server-side WebSocket only). Idle when unset. # AISStream (server-side WebSocket only). Idle when unset.
AISSTREAM_API_KEY = os.getenv("AISSTREAM_API_KEY", "") AISSTREAM_API_KEY = os.getenv("AISSTREAM_API_KEY", "")
@ -91,19 +81,3 @@ AISSTREAM_BBOX = os.getenv("AISSTREAM_BBOX", "24,-125,50,-66")
# without the ingest profile). Set 0 if the ingester owns the only connection. # without the ingest profile). Set 0 if the ingester owns the only connection.
AISSTREAM_IN_APP = os.getenv("AISSTREAM_IN_APP", "1").lower() in ("1", "true", "yes") AISSTREAM_IN_APP = os.getenv("AISSTREAM_IN_APP", "1").lower() in ("1", "true", "yes")
AISSTREAM_IN_INGEST = os.getenv("AISSTREAM_IN_INGEST", "0").lower() in ("1", "true", "yes") AISSTREAM_IN_INGEST = os.getenv("AISSTREAM_IN_INGEST", "0").lower() in ("1", "true", "yes")
# VesselAPI (quota-capped REST AIS poller — free tier 150 calls/mo).
# AISStream keeps US coasts; VesselAPI fills the Middle East blind spot. The
# poller idles when VESSELAPI_API_KEY is unset (never from GET /api/vessels).
VESSELAPI_API_KEY = os.getenv("VESSELAPI_API_KEY", "")
# Bounding box(es) as minlat,minlon,maxlat,maxlon — note lat/lon order (same as
# AISSTREAM_BBOX). Semicolon-separated for multiple boxes. Default: Strait of
# Hormuz (|dLat|+|dLon| = 3.6 ≤ 4° span cap). VesselAPI 400s any box over 4°.
VESSELAPI_BBOX = os.getenv("VESSELAPI_BBOX", "25.5,55.4,27.3,57.2")
# Poll cadence in seconds. 17280 = 4.8h → 5 polls/day (150/mo free tier).
VESSELAPI_INTERVAL = int(os.getenv("VESSELAPI_INTERVAL", "17280"))
# Local hard cap on successful 2xx calls per UTC day (persisted in Postgres).
VESSELAPI_MAX_CALLS_PER_DAY = int(os.getenv("VESSELAPI_MAX_CALLS_PER_DAY", "5"))
# Run the VesselAPI poller inside the dashboard process (default on, like AIS).
VESSELAPI_IN_APP = os.getenv("VESSELAPI_IN_APP", "1").lower() in ("1", "true", "yes")
VESSELAPI_IN_INGEST = os.getenv("VESSELAPI_IN_INGEST", "0").lower() in ("1", "true", "yes")

View file

@ -1,168 +0,0 @@
"""Curated OSINT conflict-zone catalog + point-in-bbox event counting.
A static, human-curated list of active conflict theatres (war / high /
elevated). Purely descriptive this is a catalog, not a live feed and not a
scrape of LiveUAMap or any other source. Severity and descriptions are
editorial judgement kept short and factual.
Each zone carries an internal ``bbox`` (``min_lat, min_lon, max_lat, max_lon``)
used only to count pre-existing geocoded news/GDELT/``/api/news/map`` rows that
fall inside it. The bbox is not part of the API response; callers get the
``eventCount`` roll-up instead.
Never call an upstream API from here event counts come from rows already in
the local database (``events`` with geocoords + ``news_items`` map pins).
"""
from __future__ import annotations
from datetime import datetime
# id → zone. ``lat``/``lon`` is the fly-to anchor; ``bbox`` is the internal
# count window in ``min_lat, min_lon, max_lat, max_lon`` order.
_ZONES: tuple[dict, ...] = (
{
"id": "ukraine",
"label": "Ukraine",
"severity": "war",
"lat": 48.5,
"lon": 31.0,
"description": "Full-scale Russian invasion since 2022; active front lines in the east and south.",
"bbox": (44.3, 22.1, 52.4, 40.2),
},
{
"id": "gaza",
"label": "Gaza",
"severity": "war",
"lat": 31.4,
"lon": 34.4,
"description": "IsraelHamas war; sustained fighting and a severe humanitarian crisis in the Gaza Strip.",
"bbox": (31.0, 34.1, 31.8, 34.7),
},
{
"id": "sudan",
"label": "Sudan",
"severity": "war",
"lat": 15.5,
"lon": 30.0,
"description": "Civil war between the SAF and RSF since 2023, with mass displacement across the country.",
"bbox": (8.7, 21.8, 22.0, 38.6),
},
{
"id": "myanmar",
"label": "Myanmar",
"severity": "war",
"lat": 21.5,
"lon": 96.0,
"description": "Post-2021 coup conflict pitting the junta against resistance and ethnic armed groups.",
"bbox": (9.5, 92.2, 28.5, 101.2),
},
{
"id": "drc",
"label": "DR Congo",
"severity": "war",
"lat": -1.5,
"lon": 28.0,
"description": "Eastern DRC conflict involving M23 and other armed groups; heavy displacement around Goma.",
"bbox": (-5.0, 26.0, 3.0, 31.0),
},
{
"id": "yemen",
"label": "Yemen",
"severity": "war",
"lat": 15.5,
"lon": 47.5,
"description": "Protracted Houthigovernment/coalition war with one of the world's worst humanitarian emergencies.",
"bbox": (12.6, 42.5, 19.0, 54.0),
},
{
"id": "syria",
"label": "Syria",
"severity": "war",
"lat": 34.5,
"lon": 38.5,
"description": "Multi-sided civil war; government, opposition, and external actors continue to engage.",
"bbox": (32.3, 35.7, 37.3, 42.4),
},
{
"id": "lebanon",
"label": "Lebanon",
"severity": "high",
"lat": 33.9,
"lon": 35.9,
"description": "IsraelHezbollah hostilities with periodic escalation along the southern border.",
"bbox": (33.0, 35.0, 34.7, 36.6),
},
{
"id": "sahel",
"label": "Sahel",
"severity": "high",
"lat": 14.5,
"lon": 0.0,
"description": "Jihadist insurgencies across Mali, Burkina Faso, and Niger destabilising the central Sahel.",
"bbox": (10.0, -10.0, 20.0, 12.0),
},
{
"id": "somalia",
"label": "Somalia",
"severity": "high",
"lat": 6.0,
"lon": 45.0,
"description": "Al-Shabaab insurgency against the federal government and security forces.",
"bbox": (-2.0, 41.0, 12.0, 51.5),
},
{
"id": "red_sea",
"label": "Red Sea",
"severity": "high",
"lat": 18.0,
"lon": 40.0,
"description": "Houthi attacks on commercial shipping transiting the Red Sea corridor.",
"bbox": (12.0, 34.0, 22.0, 44.0),
},
{
"id": "taiwan_strait",
"label": "Taiwan Strait",
"severity": "elevated",
"lat": 24.5,
"lon": 119.5,
"description": "Heightened military standoff between China and Taiwan, including deterrence patrols.",
"bbox": (21.9, 117.0, 26.5, 122.0),
},
{
"id": "korean_dmz",
"label": "Korean DMZ",
"severity": "elevated",
"lat": 38.3,
"lon": 127.0,
"description": "Heavily fortified inter-Korean border with periodic tensions and military drills.",
"bbox": (37.5, 126.0, 39.0, 128.5),
},
)
SEVERITIES: frozenset[str] = frozenset({"war", "high", "elevated"})
def conflict_zones() -> list[dict]:
"""Return a fresh shallow copy of the catalog (callers must not mutate)."""
return [dict(z) for z in _ZONES]
def zone_event_stats(
points: list[tuple[float, float, datetime | None]],
bbox: tuple[float, float, float, float],
) -> tuple[int, datetime | None]:
"""Count points inside ``bbox`` and return (count, latest timestamp).
``points`` is an iterable of ``(lat, lon, ts)``; ``ts`` may be ``None``.
``bbox`` is ``(min_lat, min_lon, max_lat, max_lon)``.
"""
min_lat, min_lon, max_lat, max_lon = bbox
count = 0
latest: datetime | None = None
for lat, lon, ts in points:
if min_lat <= lat <= max_lat and min_lon <= lon <= max_lon:
count += 1
if ts is not None and (latest is None or ts > latest):
latest = ts
return count, latest

View file

@ -4,9 +4,7 @@ set -e
cd /app cd /app
echo "[entrypoint] running migrations..." echo "[entrypoint] running migrations..."
# `heads` (plural): parallel PRs can fork the chain (005_phase2 + 005_news_items). alembic upgrade head
# `upgrade head` then exits 255 and crash-loops the container.
alembic upgrade heads
echo "[entrypoint] starting uvicorn..." echo "[entrypoint] starting uvicorn..."
exec uvicorn app.main:app --host 0.0.0.0 --port 8000 exec uvicorn app.main:app --host 0.0.0.0 --port 8000

View file

@ -1,186 +0,0 @@
"""Flag firefighting ADS-B aircraft within 20 miles of an active fire."""
from __future__ import annotations
from datetime import datetime, timedelta, timezone
from uuid import uuid4
from sqlalchemy import text
from database import async_session
from live_layers import _haversine_km
RADIUS_MILES = 20.0
RADIUS_KM = RADIUS_MILES * 1.609344
_recent: dict[tuple[str, str], datetime] = {}
_COOLDOWN = timedelta(minutes=5)
# ICAO type designators commonly used on wildfire air tankers, scoopers,
# helitack, and air-attack platforms. Uppercase; compared case-insensitively.
FIREFIGHTER_ICAO = frozenset({
"AT802", "AT8T", "AT8P", "AT8B",
"C130", "C30J", "C130J",
"DC10", "MD10", "MD11", "MD87",
"B737", "B38M",
"CL2T", "CL215", "CL415", "CL5T",
"S64", "SK64",
"UH1", "UH1Y", "UH60", "H60", "S70",
"B412", "B212", "B205",
"AS50", "AS350", "A119", "A109",
"B350", "BE20",
"OV10",
"RJ85", "RJ1H", "B461", "B462", "B463",
"PC12",
"TBM7", "TBM8", "TBM9",
"C208",
"DH8D", "Q400",
})
def _icao(marker: dict) -> str:
extra = marker.get("extra") or {}
return str(extra.get("type") or extra.get("t") or "").strip().upper()
def is_firefighter(marker: dict) -> bool:
return _icao(marker) in FIREFIGHTER_ICAO
def _haversine_mi(lat1: float, lon1: float, lat2: float, lon2: float) -> float:
return _haversine_km(lat1, lon1, lat2, lon2) / 1.609344
def correlate_aircraft_to_fires(
fires: list[dict],
aircraft: list[dict],
radius_mi: float = RADIUS_MILES,
) -> list[dict]:
hits: list[dict] = []
for fire in fires:
flat, flon = fire.get("lat"), fire.get("lon")
if flat is None or flon is None:
continue
fid = str(fire.get("id") or fire.get("label") or "fire")
for ac in aircraft:
if not is_firefighter(ac):
continue
alat, alon = ac.get("lat"), ac.get("lon")
if alat is None or alon is None:
continue
dist = _haversine_mi(float(flat), float(flon), float(alat), float(alon))
if dist > radius_mi:
continue
hits.append({
"fire_id": fid,
"fire_lat": float(flat),
"fire_lon": float(flon),
"aircraft_hex": str(ac.get("id")),
"aircraft_type": _icao(ac),
"aircraft_lat": float(alat),
"aircraft_lon": float(alon),
"distance_mi": round(dist, 2),
"label": ac.get("label") or ac.get("id"),
})
return hits
async def persist_hits(hits: list[dict], seen_at: datetime | None = None) -> int:
if not hits:
return 0
ts = seen_at or datetime.now(timezone.utc)
n = 0
async with async_session() as session:
for h in hits:
try:
await session.execute(
text(
"""
INSERT INTO fire_aircraft_hits
(id, fire_id, fire_lat, fire_lon,
aircraft_hex, aircraft_type, aircraft_lat, aircraft_lon,
distance_mi, seen_at)
VALUES (
CAST(:id AS uuid), :fire_id, :fire_lat, :fire_lon,
:aircraft_hex, :aircraft_type, :aircraft_lat, :aircraft_lon,
:distance_mi, :seen_at
)
ON CONFLICT (fire_id, aircraft_hex, seen_at) DO NOTHING
"""
),
{
"id": str(uuid4()),
"fire_id": h["fire_id"],
"fire_lat": h["fire_lat"],
"fire_lon": h["fire_lon"],
"aircraft_hex": h["aircraft_hex"],
"aircraft_type": h["aircraft_type"],
"aircraft_lat": h["aircraft_lat"],
"aircraft_lon": h["aircraft_lon"],
"distance_mi": h["distance_mi"],
"seen_at": ts.replace(microsecond=0),
},
)
n += 1
except Exception:
continue
try:
await session.commit()
except Exception:
return 0
return n
async def recent_hits(limit: int = 200) -> list[dict]:
try:
async with async_session() as session:
rows = (await session.execute(
text(
"""
SELECT fire_id, fire_lat, fire_lon,
aircraft_hex, aircraft_type, aircraft_lat, aircraft_lon,
distance_mi, seen_at
FROM fire_aircraft_hits
ORDER BY seen_at DESC
LIMIT :limit
"""
),
{"limit": limit},
)).mappings().all()
out = []
for r in rows:
item = dict(r)
if item.get("seen_at") is not None:
item["seen_at"] = item["seen_at"].isoformat()
out.append(item)
return out
except Exception:
return []
async def correlate_and_notify(fires: list[dict], aircraft: list[dict]) -> list[dict]:
hits = correlate_aircraft_to_fires(fires, aircraft)
if not hits:
return []
now = datetime.now(timezone.utc)
fresh = []
for h in hits:
key = (h["fire_id"], h["aircraft_hex"])
prev = _recent.get(key)
if prev is not None and now - prev < _COOLDOWN:
continue
_recent[key] = now
fresh.append(h)
if not fresh:
return []
await persist_hits(fresh, seen_at=now)
from ws_manager import manager
for h in fresh:
body = {**h, "seen_at": now.isoformat()}
await manager.publish_point(
"fire_aircraft", body, lat=h["aircraft_lat"], lon=h["aircraft_lon"],
)
await manager.publish_point(
"fire_aircraft", body, lat=h["fire_lat"], lon=h["fire_lon"],
)
return fresh

View file

@ -22,7 +22,6 @@ UTC date (YYYY-MM-DD).
from __future__ import annotations from __future__ import annotations
import csv import csv
import hashlib
import io import io
import json import json
import logging import logging
@ -41,15 +40,9 @@ from config import (
NATS_URL, NATS_URL,
) )
from keystore import get_api_key from keystore import get_api_key
from upstream_cache import firms_cache
logger = logging.getLogger("osint.firms") logger = logging.getLogger("osint.firms")
# In-process poll state: skip byte-identical CSVs, persist only new hotspots.
# Survives the 15-minute loop; one full ON CONFLICT after process start.
_csv_digest: dict[tuple, bytes] = {}
_seen_ids: dict[tuple, set[int]] = {}
# ── FIRMS API ───────────────────────────────────────────────────────────── # ── FIRMS API ─────────────────────────────────────────────────────────────
FIRMS_AREA_CSV = ( FIRMS_AREA_CSV = (
@ -92,34 +85,33 @@ def normalize_acq_time(acq_date: object, acq_time: object) -> datetime | None:
return None return None
def _hotspot_id(lat: float, lon: float, acq_iso: str, satellite: str) -> int: def parse_firms_csv(text: str) -> list[dict]:
return hash((round(lat, 5), round(lon, 5), acq_iso, satellite)) """Parse a FIRMS area CSV payload into normalized fire messages.
Returns one dict per hotspot with the fields stored in the ``fires`` table
def parse_firms_csv_delta( (acq_time already combined into a UTC ISO timestamp). Rows that don't look
text: str, skip_ids: set[int] | None = None, like valid VIIRS detections are skipped rather than failing the whole poll.
) -> tuple[list[dict], set[int]]:
"""Parse FIRMS CSV; optionally drop hotspots already seen this process.
Returns (new_or_all_points, ids_for_every_valid_row). Streaming does not
materialize the raw CSV as a list of lists.
""" """
reader = csv.reader(io.StringIO(text)) rows = list(csv.reader(io.StringIO(text)))
header = None if not rows:
for row in reader: return []
# Locate the real header row. FIRMS normally returns the CSV header first,
# but occasionally prepends a legend/info line, so scan until we see the
# canonical header.
header_idx = 0
for i, row in enumerate(rows):
if row and row[0].strip().lower() == "latitude" and len(row) >= 4: if row and row[0].strip().lower() == "latitude" and len(row) >= 4:
header = [c.strip().lower() for c in row] header_idx = i
break break
if not header or "latitude" not in header or "longitude" not in header: header = [c.strip().lower() for c in rows[header_idx]]
logger.warning( # Guard against a header that isn't actually the FIRMS one.
"FIRMS payload does not look like a hotspot CSV (first row: %r)", if "latitude" not in header or "longitude" not in header:
(header or [])[:6], logger.warning("FIRMS payload does not look like a hotspot CSV (first row: %r)", header[:6])
) return []
return [], set()
points: list[dict] = [] points: list[dict] = []
ids: set[int] = set() for row in rows[header_idx + 1:]:
for row in reader:
if len(row) < len(header): if len(row) < len(header):
continue continue
rec = dict(zip(header, row)) rec = dict(zip(header, row))
@ -130,19 +122,13 @@ def parse_firms_csv_delta(
acq_time = normalize_acq_time(rec.get("acq_date"), rec.get("acq_time")) acq_time = normalize_acq_time(rec.get("acq_date"), rec.get("acq_time"))
if acq_time is None: if acq_time is None:
continue continue
sat = str(rec.get("satellite") or "").strip()
acq_iso = acq_time.isoformat()
hid = _hotspot_id(lat, lon, acq_iso, sat)
ids.add(hid)
if skip_ids is not None and hid in skip_ids:
continue
points.append({ points.append({
"latitude": lat, "latitude": lat,
"longitude": lon, "longitude": lon,
"brightness": _to_float(rec.get("bright_ti4")), "brightness": _to_float(rec.get("bright_ti4")),
"confidence": str(rec.get("confidence") or "").strip(), "confidence": str(rec.get("confidence") or "").strip(),
"acq_time": acq_iso, "acq_time": acq_time.isoformat(),
"satellite": sat, "satellite": str(rec.get("satellite") or "").strip(),
"instrument": str(rec.get("instrument") or "").strip(), "instrument": str(rec.get("instrument") or "").strip(),
"bright_ti5": _to_float(rec.get("bright_ti5")), "bright_ti5": _to_float(rec.get("bright_ti5")),
"frp": _to_float(rec.get("frp")), "frp": _to_float(rec.get("frp")),
@ -151,17 +137,6 @@ def parse_firms_csv_delta(
"track": _to_float(rec.get("track")), "track": _to_float(rec.get("track")),
"version": str(rec.get("version") or "").strip(), "version": str(rec.get("version") or "").strip(),
}) })
return points, ids
def parse_firms_csv(text: str) -> list[dict]:
"""Parse a FIRMS area CSV payload into normalized fire messages.
Returns one dict per hotspot with the fields stored in the ``fires`` table
(acq_time already combined into a UTC ISO timestamp). Rows that don't look
like valid VIIRS detections are skipped rather than failing the whole poll.
"""
points, _ids = parse_firms_csv_delta(text)
return points return points
@ -186,16 +161,6 @@ async def publish_fire_batch(points: list[dict]) -> int:
return len(points) return len(points)
async def persist_hotspots(points: list[dict]) -> int:
"""Write a FIRMS poll to Postgres in one ON CONFLICT batch.
NATS-per-row was 93k commits + geofence/correlation per hotspot.
"""
from ingestor import ingest_fire_rows
return await ingest_fire_rows(points)
async def ingest_fires(bbox: str | None = None) -> int: async def ingest_fires(bbox: str | None = None) -> int:
"""Fetch the FIRMS hotspot CSV for an area and publish it to NATS. """Fetch the FIRMS hotspot CSV for an area and publish it to NATS.
@ -220,16 +185,12 @@ async def ingest_fires(bbox: str | None = None) -> int:
total_published = 0 total_published = 0
async with httpx.AsyncClient(timeout=FIRMS_TIMEOUT) as client: async with httpx.AsyncClient(timeout=FIRMS_TIMEOUT) as client:
for dataset in datasets: for dataset in datasets:
cache_key = (dataset, area, FIRMS_DAYS) url = FIRMS_AREA_CSV.format(
text = firms_cache.get(cache_key) key=map_key, dataset=dataset, bbox=area, days=FIRMS_DAYS
if text is None: )
url = FIRMS_AREA_CSV.format( resp = await client.get(url)
key=map_key, dataset=dataset, bbox=area, days=FIRMS_DAYS resp.raise_for_status()
) text = resp.text
resp = await client.get(url)
resp.raise_for_status()
text = resp.text
firms_cache[cache_key] = text
# FIRMS returns HTTP 200 with a plain-text error for some failure modes # FIRMS returns HTTP 200 with a plain-text error for some failure modes
# (bad key, invalid bbox); surface the first line for debuggability. # (bad key, invalid bbox); surface the first line for debuggability.
if "latitude" not in text.lower()[:4096]: if "latitude" not in text.lower()[:4096]:
@ -239,18 +200,11 @@ async def ingest_fires(bbox: str | None = None) -> int:
dataset, first_line, dataset, first_line,
) )
continue continue
poll_key = (dataset, area, FIRMS_DAYS) points = parse_firms_csv(text)
digest = hashlib.sha256(text.encode("utf-8", "surrogatepass")).digest() published = await publish_fire_batch(points)
if _csv_digest.get(poll_key) == digest:
logger.info("FIRMS %s CSV unchanged, skip parse/insert", dataset)
continue
points, ids = parse_firms_csv_delta(text, skip_ids=_seen_ids.get(poll_key))
published = await persist_hotspots(points) if points else 0
_csv_digest[poll_key] = digest
_seen_ids[poll_key] = ids
total_published += published total_published += published
logger.info( logger.info(
"FIRMS: fetched %d hotspot(s) for bbox=%s (%s), published %d", "FIRMS: fetched %d hotspot(s) for bbox=%s (%s), published %d",
len(ids), area, dataset, published, len(points), area, dataset, published,
) )
return total_published return total_published

View file

@ -1,472 +0,0 @@
"""Geofences: GeoJSON polygons, ST_Intersects on ingest, WS alerts.
``/api/alerts`` is the dashboard entity/keyword table geofence hits live
in ``geofence_alerts`` and fan out as WS type ``geofence_alert``.
"""
from __future__ import annotations
import json
from datetime import datetime, timedelta, timezone
from typing import Any
from uuid import uuid4
from sqlalchemy import text
from database import async_session
def _rings_from_geojson(geojson: dict) -> list[list[list[float]]]:
if not isinstance(geojson, dict):
raise ValueError("geojson must be an object")
gj = geojson
if gj.get("type") == "Feature":
gj = gj.get("geometry") or {}
if gj.get("type") == "FeatureCollection":
raise ValueError("FeatureCollection is not a single polygon")
if gj.get("type") != "Polygon":
raise ValueError("geojson must be a Polygon")
coords = gj.get("coordinates")
if not isinstance(coords, list) or not coords:
raise ValueError("polygon has no rings")
rings: list[list[list[float]]] = []
for ring in coords:
if not isinstance(ring, list) or len(ring) < 4:
raise ValueError("polygon ring needs ≥4 positions (closed)")
pts = []
for pt in ring:
if not isinstance(pt, (list, tuple)) or len(pt) < 2:
raise ValueError("position must be [lon, lat]")
pts.append([float(pt[0]), float(pt[1])])
rings.append(pts)
return rings
def validate_polygon_geojson(geojson: dict) -> dict:
"""Return a canonical Polygon GeoJSON or raise ValueError."""
rings = _rings_from_geojson(geojson)
return {"type": "Polygon", "coordinates": rings}
def _ring_contains(lon: float, lat: float, ring: list[list[float]]) -> bool:
"""Ray-cast even-odd rule. Ring is [lon, lat] positions."""
inside = False
n = len(ring)
if n < 4:
return False
j = n - 1
for i in range(n):
xi, yi = ring[i][0], ring[i][1]
xj, yj = ring[j][0], ring[j][1]
intersects = ((yi > lat) != (yj > lat)) and (
lon < (xj - xi) * (lat - yi) / ((yj - yi) or 1e-16) + xi
)
if intersects:
inside = not inside
j = i
return inside
def point_in_geojson(lon: float, lat: float, geojson: dict) -> bool:
"""True if (lon, lat) is inside the outer ring and outside holes."""
try:
rings = _rings_from_geojson(geojson)
except (ValueError, TypeError, KeyError):
return False
if not _ring_contains(lon, lat, rings[0]):
return False
for hole in rings[1:]:
if _ring_contains(lon, lat, hole):
return False
return True
def matching_geofences(lon: float, lat: float, fences: list[dict]) -> list[dict]:
hits = []
for fence in fences:
if not fence.get("active", True):
continue
gj = fence.get("geojson") or {}
if point_in_geojson(lon, lat, gj):
hits.append(fence)
return hits
# In-process copy of active fences so ingest does not round-trip Postgres
# on every AIS frame. CRUD endpoints refresh this list.
_cache: list[dict] = []
_recent_hits: dict[tuple[str, str], datetime] = {}
_HIT_COOLDOWN = timedelta(minutes=5)
async def refresh_cache() -> list[dict]:
global _cache
async with async_session() as session:
rows = (await session.execute(text(
"SELECT id::text, name, geojson, active FROM geofences"
))).mappings().all()
_cache = [
{
"id": r["id"],
"name": r["name"],
"geojson": r["geojson"] if isinstance(r["geojson"], dict)
else json.loads(r["geojson"] or "{}"),
"active": bool(r["active"]),
}
for r in rows
]
return _cache
def cached_fences() -> list[dict]:
return list(_cache)
async def list_geofences() -> list[dict]:
if not _cache:
try:
await refresh_cache()
except Exception:
return []
return cached_fences()
async def create_geofence(name: str, geojson: dict, active: bool = True) -> dict:
polygon = validate_polygon_geojson(geojson)
gid = str(uuid4())
gj = json.dumps(polygon)
async with async_session() as session:
await session.execute(
text(
"""
INSERT INTO geofences (id, name, geojson, geom, active)
VALUES (
:id, :name, CAST(:geojson AS jsonb),
ST_SetSRID(ST_GeomFromGeoJSON(:geojson), 4326),
:active
)
"""
),
{"id": gid, "name": name, "geojson": gj, "active": 1 if active else 0},
)
await session.commit()
row = {"id": gid, "name": name, "geojson": polygon, "active": active}
_cache.append(row)
return row
async def update_geofence(gid: str, *, name: str | None = None,
geojson: dict | None = None,
active: bool | None = None) -> dict | None:
current = next((f for f in _cache if f["id"] == gid), None)
if current is None:
await refresh_cache()
current = next((f for f in _cache if f["id"] == gid), None)
if current is None:
return None
if name is not None:
current["name"] = name
if geojson is not None:
current["geojson"] = validate_polygon_geojson(geojson)
if active is not None:
current["active"] = active
gj = json.dumps(current["geojson"])
async with async_session() as session:
await session.execute(
text(
"""
UPDATE geofences SET
name = :name,
geojson = CAST(:geojson AS jsonb),
geom = ST_SetSRID(ST_GeomFromGeoJSON(:geojson), 4326),
active = :active,
updated_at = now()
WHERE id = CAST(:id AS uuid)
"""
),
{
"id": gid,
"name": current["name"],
"geojson": gj,
"active": 1 if current["active"] else 0,
},
)
await session.commit()
return current
async def delete_geofence(gid: str) -> bool:
async with async_session() as session:
result = await session.execute(
text("DELETE FROM geofences WHERE id = CAST(:id AS uuid)"),
{"id": gid},
)
await session.commit()
_cache[:] = [f for f in _cache if f["id"] != gid]
return bool(result.rowcount)
async def st_intersects(lon: float, lat: float) -> list[dict]:
"""PostGIS ST_Intersects against active geofences.
Falls back to the in-memory GeoJSON test if the DB is unreachable so
ingest never dies because a fence check failed.
"""
try:
async with async_session() as session:
rows = (await session.execute(
text(
"""
SELECT id::text, name, geojson, active
FROM geofences
WHERE active = 1
AND ST_Intersects(
geom,
ST_SetSRID(ST_MakePoint(:lon, :lat), 4326)
)
"""
),
{"lon": lon, "lat": lat},
)).mappings().all()
return [
{
"id": r["id"],
"name": r["name"],
"geojson": r["geojson"] if isinstance(r["geojson"], dict)
else json.loads(r["geojson"] or "{}"),
"active": True,
}
for r in rows
]
except Exception:
return matching_geofences(lon, lat, cached_fences())
async def record_and_notify(
*,
source_kind: str,
entity_id: str,
lat: float,
lon: float,
payload: dict[str, Any] | None = None,
) -> int:
"""Insert a geofence_alerts row per hit and WS-push to viewport clients.
PostGIS ST_Intersects is the source of truth. The in-process GeoJSON
cache is not a reject filter the FIRMS ingester never fills it.
"""
hits = await st_intersects(lon, lat)
if not hits:
return 0
from ws_manager import manager
sent = 0
now = datetime.now(timezone.utc)
fresh = []
for fence in hits:
key = (str(fence["id"]), str(entity_id))
prev = _recent_hits.get(key)
if prev is not None and now - prev < _HIT_COOLDOWN:
continue
_recent_hits[key] = now
fresh.append(fence)
if not fresh:
return 0
hits = fresh
async with async_session() as session:
for fence in hits:
aid = str(uuid4())
body = {
"id": aid,
"geofence_id": fence["id"],
"geofence_name": fence.get("name"),
"source_kind": source_kind,
"entity_id": str(entity_id),
"lat": lat,
"lon": lon,
"payload": payload or {},
"created_at": now.isoformat(),
}
try:
await session.execute(
text(
"""
INSERT INTO geofence_alerts
(id, geofence_id, source_kind, entity_id, lat, lon, payload)
VALUES (
CAST(:id AS uuid), CAST(:geofence_id AS uuid),
:source_kind, :entity_id, :lat, :lon, CAST(:payload AS jsonb)
)
"""
),
{
"id": aid,
"geofence_id": fence["id"],
"source_kind": source_kind,
"entity_id": str(entity_id),
"lat": lat,
"lon": lon,
"payload": json.dumps(payload or {}),
},
)
except Exception:
pass
sent += await manager.publish_point(
"geofence_alert", body, lat=lat, lon=lon,
)
try:
await session.commit()
except Exception:
pass
return sent
async def list_alerts(
*,
geofence_id: str | None = None,
since: datetime | None = None,
until: datetime | None = None,
source_kind: str | None = None,
limit: int = 100,
) -> list[dict]:
"""Filterable hit log. Empty list if the DB is down — never raises."""
where = ["TRUE"]
params: dict[str, Any] = {"limit": int(limit)}
if geofence_id:
where.append("geofence_id = CAST(:geofence_id AS uuid)")
params["geofence_id"] = geofence_id
if since is not None:
where.append("created_at >= :since")
params["since"] = since
if until is not None:
where.append("created_at <= :until")
params["until"] = until
if source_kind:
where.append("source_kind = :source_kind")
params["source_kind"] = source_kind
sql = f"""
SELECT id::text, geofence_id::text, source_kind, entity_id,
lat, lon, payload, created_at
FROM geofence_alerts
WHERE {' AND '.join(where)}
ORDER BY created_at DESC
LIMIT :limit
"""
try:
async with async_session() as session:
rows = (await session.execute(text(sql), params)).mappings().all()
out = []
for r in rows:
item = dict(r)
if item.get("created_at") is not None:
item["created_at"] = item["created_at"].isoformat()
out.append(item)
return out
except Exception:
return []
async def get_geofence(gid: str) -> dict | None:
current = next((f for f in _cache if f["id"] == gid), None)
if current is not None:
return current
try:
await refresh_cache()
except Exception:
return None
return next((f for f in _cache if f["id"] == gid), None)
def _marker_from_track(row) -> dict:
from live_layers import to_marker
extra = {"bucket": row["bucket"].isoformat() if row.get("bucket") else None, "dvr": True}
return to_marker(
row["id"], row["lat"], row["lon"],
heading=row.get("heading"), speed=row.get("speed"),
label=row.get("label") or row["id"],
extra=extra,
)
async def _cagg_inside(gid: str, kind: str, bucket: datetime, limit: int = 2000) -> list[dict]:
table = "aircraft_tracks_1min" if kind == "aircraft" else "vessel_tracks_1min"
id_col = "hex" if kind == "aircraft" else "mmsi"
sql = f"""
SELECT {id_col} AS id, lat, lon, heading, speed, label, bucket
FROM {table}
WHERE bucket = :bucket
AND ST_Intersects(
(SELECT geom FROM geofences WHERE id = CAST(:gid AS uuid)),
ST_SetSRID(ST_MakePoint(lon, lat), 4326)
)
LIMIT :limit
"""
try:
async with async_session() as session:
rows = (await session.execute(
text(sql), {"bucket": bucket, "gid": gid, "limit": limit},
)).mappings().all()
return [
_marker_from_track(r)
for r in rows
if r["lat"] is not None and r["lon"] is not None
]
except Exception:
return []
async def _fires_inside(gid: str, ts: datetime, limit: int = 2000) -> list[dict]:
from tracks import minute_bucket
bucket = minute_bucket(ts)
t1 = bucket + timedelta(minutes=1)
sql = """
SELECT latitude, longitude, brightness, confidence, acq_time, satellite,
instrument, bright_ti5, frp, daynight
FROM fires
WHERE acq_time >= :t0 AND acq_time < :t1
AND ST_Intersects(
(SELECT geom FROM geofences WHERE id = CAST(:gid AS uuid)),
ST_SetSRID(ST_MakePoint(longitude, latitude), 4326)
)
LIMIT :limit
"""
try:
async with async_session() as session:
rows = (await session.execute(
text(sql),
{"t0": bucket, "t1": t1, "gid": gid, "limit": limit},
)).mappings().all()
out = []
for r in rows:
item = dict(r)
if item.get("acq_time") is not None:
item["acq_time"] = item["acq_time"].isoformat()
out.append(item)
return out
except Exception:
return []
async def snapshot_at(gid: str, ts: datetime) -> dict | None:
"""Positions inside the fence at time T. None if the fence is missing.
Does not persist or notify. Empty lists if track/fire queries fail.
"""
fence = await get_geofence(gid)
if fence is None:
return None
from tracks import minute_bucket
bucket = minute_bucket(ts)
aircraft = await _cagg_inside(gid, "aircraft", bucket)
vessels = await _cagg_inside(gid, "vessel", bucket)
fires = await _fires_inside(gid, ts)
return {
"geofence_id": gid,
"timestamp": ts.isoformat(),
"aircraft": aircraft,
"vessels": vessels,
"fires": fires,
}

View file

@ -14,17 +14,10 @@ from sqlalchemy.dialects.postgresql import insert as pg_insert
from database import async_session from database import async_session
from models import events as events_table from models import events as events_table
from models import fires as fires_table from models import fires as fires_table
from models import event_dedup as event_dedup_table
from config import NATS_URL from config import NATS_URL
from sources import event_dedup_key
logger = logging.getLogger("osint.ingestor") logger = logging.getLogger("osint.ingestor")
# asyncpg rejects statements with >32767 bind params. A FIRMS poll is ~90k
# rows × 14 columns. Chunk inserts; still one transaction / one commit.
FIRE_ROW_BIND_PARAMS = 14
FIRE_INSERT_CHUNK = 2000
# NATS connection settings # NATS connection settings
NATS_URLS = NATS_URL NATS_URLS = NATS_URL
NATS_STREAM = "events" NATS_STREAM = "events"
@ -102,62 +95,6 @@ async def ingest_fire_row(msg: dict) -> bool:
"ingested fire %.5f,%.5f %s satellite=%s", "ingested fire %.5f,%.5f %s satellite=%s",
row["latitude"], row["longitude"], row["acq_time"], row["satellite"], row["latitude"], row["longitude"], row["acq_time"], row["satellite"],
) )
lat, lon = row["latitude"], row["longitude"]
from geofence import record_and_notify
await record_and_notify(
source_kind="firms",
entity_id=f"{lat},{lon},{row['satellite']}",
lat=lat, lon=lon, payload={"satellite": row["satellite"]},
)
from live_layers import aircraft_last_known
from fire_aircraft import correlate_and_notify
from tracks import recent_markers
acs = list(aircraft_last_known.values()) or await recent_markers("aircraft")
if acs:
fire = {
"id": f"firms:{lat:.4f},{lon:.4f}",
"lat": lat, "lon": lon, "label": "FIRMS",
}
await correlate_and_notify([fire], acs)
return inserted
async def ingest_fire_rows(msgs: list[dict]) -> int:
"""Bulk-insert FIRMS hotspots: one INSERT, one ON CONFLICT, one commit."""
rows = []
for msg in msgs:
row = _fire_row_from_msg(msg)
if row is not None:
rows.append(row)
if not rows:
return 0
inserted = 0
async with async_session() as session:
for i in range(0, len(rows), FIRE_INSERT_CHUNK):
chunk = rows[i:i + FIRE_INSERT_CHUNK]
stmt = (
pg_insert(fires_table)
.values(chunk)
.on_conflict_do_nothing(constraint="pk_fires_natural_key")
)
result = await session.execute(stmt)
inserted += int(result.rowcount or 0)
await session.commit()
if inserted:
logger.info("bulk ingested %d/%d FIRMS hotspots", inserted, len(rows))
from live_layers import aircraft_last_known
from fire_aircraft import correlate_and_notify
from tracks import recent_markers
acs = list(aircraft_last_known.values()) or await recent_markers("aircraft")
if acs:
fires = [
{
"id": f"firms:{r['latitude']:.4f},{r['longitude']:.4f}",
"lat": r["latitude"], "lon": r["longitude"], "label": "FIRMS",
}
for r in rows[:500]
]
await correlate_and_notify(fires, acs)
return inserted return inserted
@ -195,26 +132,10 @@ async def ingest_event(msg: dict):
} }
# Parse timestamp if string # Parse timestamp if string
ts = event_row["source_timestamp"] if isinstance(event_row["source_timestamp"], str):
if isinstance(ts, str): event_row["source_timestamp"] = datetime.fromisoformat(event_row["source_timestamp"])
ts = datetime.fromisoformat(ts.replace("Z", "+00:00"))
if isinstance(ts, datetime) and ts.tzinfo is None:
ts = ts.replace(tzinfo=timezone.utc)
event_row["source_timestamp"] = ts
key = event_dedup_key(event_row)
async with async_session() as session: async with async_session() as session:
if key:
dedup = (
pg_insert(event_dedup_table)
.values(url=key)
.on_conflict_do_nothing(index_elements=["url"])
)
claimed = await session.execute(dedup)
if not claimed.rowcount:
await session.commit()
logger.debug("skip duplicate event url=%s", key)
return None
result = await session.execute(events_table.insert().values(**event_row)) result = await session.execute(events_table.insert().values(**event_row))
await session.commit() await session.commit()
event_id = result.inserted_primary_key[0] # type: ignore[union-attr] event_id = result.inserted_primary_key[0] # type: ignore[union-attr]

View file

@ -57,10 +57,10 @@ KEY_REGISTRY: dict[str, dict] = {
"pattern": r"^[0-9a-fA-F]{32}$", "pattern": r"^[0-9a-fA-F]{32}$",
"example": "32-char hex string (e.g. 5f3c…9a02)", "example": "32-char hex string (e.g. 5f3c…9a02)",
}, },
"NOUS_API_KEY": { "GEMINI_API_KEY": {
"description": "Nous Portal API key — 15-min news summarizer (inference-api.nousresearch.com).", "description": "Google Gemini API key — LLM event analysis / summarization.",
"pattern": r"^.{16,}$", "pattern": r"^AIza[0-9A-Za-z_-]{35}$",
"example": "key from https://portal.nousresearch.com (API keys page)", "example": "AIza… (Google API key, 39 chars)",
}, },
"TELEGRAM_TOKEN": { "TELEGRAM_TOKEN": {
"description": "Telegram bot token — push alert notifications to a channel.", "description": "Telegram bot token — push alert notifications to a channel.",
@ -68,15 +68,10 @@ KEY_REGISTRY: dict[str, dict] = {
"example": "123456789:AA… (bot token from @BotFather)", "example": "123456789:AA… (bot token from @BotFather)",
}, },
"AISSTREAM_API_KEY": { "AISSTREAM_API_KEY": {
"description": "AISStream (open/shared) — live US-coast AIS. Server-side WebSocket only.", "description": "AISStream WebSocket key — live vessel positions (server-side only).",
"pattern": r"^.{8,}$", "pattern": r"^.{8,}$",
"example": "key from https://aisstream.io/account (GitHub login)", "example": "key from https://aisstream.io/account (GitHub login)",
}, },
"VESSELAPI_API_KEY": {
"description": "VesselAPI (commercial) — Strait of Hormuz AIS, 5×/day cache. Paste the Bearer token from dashboard.vesselapi.com. Not a US-coast feed.",
"pattern": r"^.{8,}$",
"example": "Bearer token from https://dashboard.vesselapi.com/",
},
"OPENSKY_CLIENT_ID": { "OPENSKY_CLIENT_ID": {
"description": "OpenSky OAuth client id — optional ADS-B fallback (unused until enabled).", "description": "OpenSky OAuth client id — optional ADS-B fallback (unused until enabled).",
"example": "client id from opensky-network.org account", "example": "client id from opensky-network.org account",
@ -219,7 +214,7 @@ async def get_api_key(name: str) -> str | None:
"""Read a stored key value — used by ingest services, never by the API. """Read a stored key value — used by ingest services, never by the API.
Returns the raw value (or None when unset) so producers can pass it to Returns the raw value (or None when unset) so producers can pass it to
external APIs (FIRMS, Nous, Telegram, ). Reads live from Postgres, so a external APIs (FIRMS, Gemini, Telegram, ). Reads live from Postgres, so a
key set via the dashboard is picked up on the next poll no restart. key set via the dashboard is picked up on the next poll no restart.
""" """
await ensure_api_keys_table() await ensure_api_keys_table()

File diff suppressed because it is too large Load diff

File diff suppressed because it is too large Load diff

65
app/masscan_config.py Normal file
View file

@ -0,0 +1,65 @@
"""Active camera-discovery configuration (masscan-based, env-driven).
All knobs read from the environment with safe defaults. The scanner targets
open TCP port 554 (RTSP the typical IP-camera port) across a configured
range and feeds results into the same `cameras` table as the passive scraper
(discovery_source='masscan'), deduped by URL hash.
ETHICS / SCOPE (mirrors camera_scraper.py):
* Detection only a SYN port scan for OPEN hosts. No credential guessing,
no login attempts, no banner grabbing, and no access to camera feeds.
* Private / reserved ranges are excluded via MASSCAN_EXCLUDEFILE so the
scanner never probes RFC1918, loopback, link-local, multicast, or the
bogons. Fail closed if the excludefile is missing.
TIMING REALITY: at the residential-safe default of 200 pps a full IPv4
sweep (0.0.0.0/0, ~4.29B addresses) takes ~8 months. This is therefore a
CONTINUOUS ROLLING SWEEP, not a "finish in a day" job: masscan streams
open hosts to stdout and the runner ingests them incrementally, then
restarts the sweep when a pass completes. New cameras are detected as they
appear on each pass. 1k/10k pps saturated a home uplink do not raise the
rate unless you are on a VPS / unmetered link.
"""
from __future__ import annotations
import os
# Path to the masscan binary (installed on the Pi host).
MASSCAN_BIN = os.getenv("MASSCAN_BIN", "masscan")
# CIDR(s) to sweep. Default = the whole public IPv4 space.
MASSCAN_RANGE = os.getenv("MASSCAN_RANGE", "0.0.0.0/0")
# Port(s) to probe. Default 554 = RTSP, the typical IP-camera port.
MASSCAN_PORTS = os.getenv("MASSCAN_PORTS", "554")
# Packets/sec. 200 is the residential-safe default — 1k/10k pps saturated
# a home uplink. Raise only on a VPS / unmetered link.
MASSCAN_RATE = int(os.getenv("MASSCAN_RATE", "200"))
# Retransmission count. 1 maximizes unique-host coverage at low rate; the
# default (10) spends most of the budget re-probing the same hosts.
MASSCAN_RETRIES = int(os.getenv("MASSCAN_RETRIES", "1"))
# Seconds to keep listening for straggler responses after the last probe.
# 0 avoids a 10s tail per pass; tiny loss of the very last hosts is fine
# since the sweep repeats.
MASSCAN_WAIT = int(os.getenv("MASSCAN_WAIT", "0"))
# Excludefile path on the Pi host. Must contain RFC1918/loopback/link-local/
# multicast/bogons so the scanner never probes private ranges. Fail closed if
# the file is absent (the runner refuses to start rather than scan wide).
MASSCAN_EXCLUDEFILE = os.getenv(
"MASSCAN_EXCLUDEFILE", "/etc/osint-dashboard/masscan-excludes.txt"
)
# Ingest batch size — flush this many newly-seen hosts to the DB per round.
MASSCAN_FLUSH_EVERY = int(os.getenv("MASSCAN_FLUSH_EVERY", "250"))
# NATS subject newly-found cameras are published on (same feed as the
# passive scraper so the shared ingester persists them).
MASSCAN_NATS_SUBJECT = os.getenv("MASSCAN_NATS_SUBJECT", "events.camera")
# discovery_source tag written into the cameras table.
MASSCAN_DISCOVERY_SOURCE = os.getenv("MASSCAN_DISCOVERY_SOURCE", "masscan")

226
app/masscan_scanner.py Normal file
View file

@ -0,0 +1,226 @@
"""masscan result parsing + ingestion for the OSINT dashboard.
Turns a stream of masscan JSON-lines (open port 554 hosts) into rows in the
`cameras` table with discovery_source='masscan', deduped by URL hash against
whatever the passive scraper already found. Newly discovered hosts are also
published to NATS (`events.camera`) so the shared ingester pipeline persists
them exactly like scraper finds.
Scope: detection of OPEN hosts only. No credentials, no banners, no feed
access. Private/reserved ranges never enter masscan (see excludefile).
"""
from __future__ import annotations
import asyncio
import json
import logging
from datetime import datetime, timezone
from camera_models import cameras
from camera_scraper import url_hash, geolocate_ips
from database import async_session
from masscan_config import (
MASSCAN_NATS_SUBJECT, MASSCAN_DISCOVERY_SOURCE,
)
logger = logging.getLogger("osint.masscan_scanner")
# ── URL building ──────────────────────────────────────────────────────────
def build_rtsp_url(ip: str) -> str:
"""Canonical URL for an open-RTSP host. Used as the dedupe key."""
return f"rtsp://{ip}/"
# ── masscan JSON parsing ──────────────────────────────────────────────────
# masscan --output-format=json --output-file=- emits line-delimited JSON on a
# pipe (a bare object per open host), not the array form used for seekable
# files. We parse per-line and tolerate an accidental leading '['.
def parse_masscan_line(line: str) -> list[dict]:
"""Parse one masscan stdout line into a list of host records.
A line may contain one JSON object or, defensively, be wrapped in an
array. Returns [] on anything unparseable (harmless the sweep repeats).
"""
s = line.strip()
if not s:
return []
s = s.lstrip("[").rstrip("]").strip()
if not s:
return []
# Multiple records may share a line separated by '},{'.
if s.endswith(","):
s = s[:-1].rstrip()
out: list[dict] = []
for cand in _split_records(s):
try:
obj = json.loads(cand)
except (json.JSONDecodeError, ValueError):
continue
if isinstance(obj, dict) and obj.get("ip"):
out.append(obj)
return out
def _split_records(s: str) -> list[str]:
"""Split a buffer into individual JSON object strings, honoring nesting."""
records, depth, start = [], 0, 0
for i, ch in enumerate(s):
if ch == "{":
if depth == 0:
start = i
depth += 1
elif ch == "}":
depth -= 1
if depth == 0:
records.append(s[start:i + 1])
return records
def extract_open_ips(records: list[dict], port: int) -> list[str]:
"""Return the list of IPs from records that have `port` open."""
ips: list[str] = []
for rec in records:
for p in rec.get("ports", []):
if p.get("port") == port and p.get("status") == "open":
ips.append(rec["ip"])
break
return ips
# ── Persistence ───────────────────────────────────────────────────────────
async def ingest_open_hosts(ips: list[str]) -> tuple[int, list[str]]:
"""Insert-or-refresh camera rows for open RTSP hosts that have a public feed.
A host only lands in the table (and therefore on the map) if an
unauthenticated HTTP still or MJPEG URL responds. Port-554-only hosts
are skipped. Returns (newly_inserted, hosts_with_working_feed).
"""
if not ips:
return 0, []
from camera_preview import probe_public_feed
unique = list(dict.fromkeys(ips))
sem = asyncio.Semaphore(20)
async def _probe(ip: str) -> tuple[str, str | None]:
async with sem:
return ip, await probe_public_feed(ip)
probed = await asyncio.gather(*(_probe(ip) for ip in unique))
live = [(ip, feed) for ip, feed in probed if feed]
if not live:
logger.info("masscan ingest: 0 working feeds of %d open-554 hosts",
len(unique))
return 0, []
now = datetime.now(timezone.utc)
coords = await geolocate_ips([ip for ip, _ in live])
new = 0
async with async_session() as session:
for ip, feed in live:
url = build_rtsp_url(ip)
h = url_hash(url)
lat, lon = coords.get(ip, (None, None))
existing = (await session.execute(
cameras.select().where(cameras.c.url_hash == h)
)).one_or_none()
if existing is None:
await session.execute(cameras.insert().values(
url_hash=h,
source_url=url,
snapshot_url=feed,
discovery_source=MASSCAN_DISCOVERY_SOURCE,
location_lat=lat,
location_lon=lon,
location_name=f"{ip} (IP-geo)" if lat is not None else None,
vendor=None,
device_type="rtsp",
first_seen=now,
last_seen=now,
raw={"discovered_via": "masscan", "port": 554,
"public_feed": feed},
))
new += 1
else:
await session.execute(cameras.update().where(
cameras.c.url_hash == h
).values(
last_seen=now,
snapshot_url=feed,
location_lat=lat,
location_lon=lon,
location_name=f"{ip} (IP-geo)" if lat is not None else None,
))
await session.commit()
logger.info("masscan ingest: %d new working feeds (%d probed, %d open-554)",
new, len(live), len(unique))
return new, [ip for ip, _ in live]
# ── NATS publish ──────────────────────────────────────────────────────────
async def publish_new_hosts(ips: list[str]) -> int:
"""Publish newly-found open hosts to NATS for the shared ingester.
Returns the number of messages published (0 if NATS is down).
"""
import json as _json
import nats
from config import NATS_URL
if not ips:
return 0
try:
nc = await nats.connect(NATS_URL)
except Exception: # noqa: BLE001
logger.warning("NATS unavailable — skipping publish pass")
return 0
published = 0
try:
js = nc.jetstream()
for ip in dict.fromkeys(ips):
url = build_rtsp_url(ip)
msg = {
"source_type": "camera",
"title": f"Open RTSP camera ({ip})",
"url": url,
"location_lat": None,
"location_lon": None,
"location_name": None,
"tags": ["osint", "camera", MASSCAN_DISCOVERY_SOURCE],
"raw": {
"url_hash": url_hash(url),
"source_url": url,
"snapshot_url": None,
"vendor": None,
"device_type": "rtsp",
"discovered_via": "masscan",
"port": 554,
},
"source_timestamp": datetime.now(timezone.utc).isoformat(),
}
await js.publish(MASSCAN_NATS_SUBJECT, _json.dumps(msg).encode())
published += 1
finally:
await nc.close()
logger.info("published %d masscan finds to %s", published, MASSCAN_NATS_SUBJECT)
return published
# ── Batch drain helper used by the runner ─────────────────────────────────
async def flush(seen: set[str], new_accum: int) -> tuple[int, int]:
"""Ingest + publish the accumulated host set; return (new, published)."""
if not seen:
return 0, 0
ips = list(seen)
new, live = await ingest_open_hosts(ips)
published = await publish_new_hosts(live)
seen.clear()
return new, published

View file

@ -68,15 +68,6 @@ Index("ix_events_search_vector", events.c.search_vector, postgresql_using="gin")
# Spatial index on location # Spatial index on location
Index("ix_events_location", events.c.location_lat, events.c.location_lon) Index("ix_events_location", events.c.location_lat, events.c.location_lon)
# Timescale unique indexes must include the partition column, so URL
# idempotency lives on a regular table — not the events hypertable.
event_dedup = Table(
"event_dedup",
metadata,
Column("url", Text, primary_key=True),
Column("created_at", DateTime(timezone=True), server_default=func.now(), nullable=False),
)
# ── Entities (people, organizations, locations of interest) ────────────── # ── Entities (people, organizations, locations of interest) ──────────────
@ -194,8 +185,8 @@ Index("ix_fires_bbox", fires.c.longitude, fires.c.latitude)
# ── News pipeline (scraper + summarizer) ────────────────────────────────── # ── News pipeline (scraper + summarizer) ──────────────────────────────────
# Written by the vendored news-scraper (Scrapy) / news-summarizer services; # Written by the vendored news-scraper (Scrapy) / news-summarizer (Gemini)
# schema must match the idempotent alembic migrations 003_news + 005_news_items. # services; schema must match the idempotent alembic migration 003_news.
articles = Table( articles = Table(
"articles", "articles",
@ -218,27 +209,6 @@ article_summaries = Table(
Column("summary_text", Text, nullable=False), Column("summary_text", Text, nullable=False),
Column("batch_timestamp", DateTime(timezone=True), Column("batch_timestamp", DateTime(timezone=True),
server_default=func.now(), nullable=False), server_default=func.now(), nullable=False),
Column("model", Text), # LLM id used for this batch; nullable for old rows
Column("kind", Text), # interval | daily_recap; nullable for old rows
) )
Index("ix_article_summaries_batch_timestamp", article_summaries.c.batch_timestamp) Index("ix_article_summaries_batch_timestamp", article_summaries.c.batch_timestamp)
news_items = Table(
"news_items",
metadata,
Column("id", Integer, primary_key=True, autoincrement=True),
Column("summary_id", Integer),
Column("kind", Text, nullable=False),
Column("headline", Text, nullable=False),
Column("importance", Text, nullable=False),
Column("location_name", Text),
Column("lat", Float),
Column("lon", Float),
Column("location_confidence", Text),
Column("category", Text),
Column("url", Text),
Column("created_at", DateTime(timezone=True), server_default=func.now(), nullable=False),
)
Index("ix_news_items_kind_created", news_items.c.kind, news_items.c.created_at)

View file

@ -1,99 +0,0 @@
"""Nominatim reverse-geocode proxy for the map place dossier.
Browser clients cannot set an identifying User-Agent, and Nominatim typically
blocks CORS so the HUD calls GET /api/place instead of talking to OSM
directly. Cache 60s / 500 keys; never exceed 1 req/s upstream.
"""
from __future__ import annotations
import asyncio
import time
import httpx
from cachetools import TTLCache
from config import NOMINATIM_MIN_INTERVAL, NOMINATIM_URL, OSINT_USER_AGENT
_NOMINATIM = NOMINATIM_URL.rstrip("/")
place_cache: TTLCache = TTLCache(maxsize=500, ttl=60)
_lock = asyncio.Lock()
_last_req = 0.0
_ADDR_KEEP = (
"house_number", "road", "neighbourhood", "suburb", "city", "town",
"village", "hamlet", "county", "state", "postcode", "country", "country_code",
)
def cache_key(lat: float, lon: float) -> str:
return f"{lat:.4f},{lon:.4f}"
def slim_place(lat: float, lon: float, data: dict | None) -> dict:
data = data or {}
raw_addr = data.get("address")
addr_in: dict = raw_addr if isinstance(raw_addr, dict) else {}
address = {k: addr_in[k] for k in _ADDR_KEEP if addr_in.get(k)}
err = data.get("error")
display = None if err else (data.get("display_name") or None)
name = None if err else (data.get("name") or address.get("city")
or address.get("town") or address.get("village") or None)
return {
"lat": lat,
"lon": lon,
"display_name": display,
"name": name,
"address": address,
"osm_type": None if err else data.get("osm_type"),
"osm_id": None if err else data.get("osm_id"),
"attribution": "© OpenStreetMap contributors",
}
async def reverse_geocode(lat: float, lon: float) -> dict:
"""Reverse-geocode a point. Cache hits skip Nominatim entirely."""
if not (-90.0 <= lat <= 90.0 and -180.0 <= lon <= 180.0):
raise ValueError("lat/lon out of range")
key = cache_key(lat, lon)
qlat, qlon = (float(p) for p in key.split(","))
async with _lock:
hit = place_cache.get(key)
if hit is not None:
return hit
global _last_req
wait = _last_req + NOMINATIM_MIN_INTERVAL - time.monotonic()
if wait > 0:
await asyncio.sleep(wait)
body = await _fetch_nominatim(qlat, qlon)
_last_req = time.monotonic()
place_cache[key] = body
return body
async def _fetch_nominatim(lat: float, lon: float) -> dict:
headers = {
"User-Agent": OSINT_USER_AGENT,
"Accept": "application/json",
}
url = f"{_NOMINATIM}/reverse"
params = {
"lat": f"{lat:.6f}",
"lon": f"{lon:.6f}",
"format": "jsonv2",
"addressdetails": "1",
"zoom": "18",
}
async with _http_client(timeout=10.0, follow_redirects=True) as client:
r = await client.get(url, params=params, headers=headers)
r.raise_for_status()
data = r.json()
if not isinstance(data, dict):
data = {}
return slim_place(lat, lon, data)
def _http_client(**kwargs):
return httpx.AsyncClient(**kwargs)

View file

@ -11,6 +11,3 @@ feedparser>=6.0
python-dateutil>=2.9 python-dateutil>=2.9
structlog>=24.4 structlog>=24.4
websockets>=14 websockets>=14
cachetools>=5.5
h3>=4.0
sgp4>=2.23

View file

@ -24,8 +24,8 @@ import sys
sys.path.insert(0, sys_path) sys.path.insert(0, sys_path)
from config import NATS_URL, FIRMS_INTERVAL, FIRMS_DATASET, AISSTREAM_IN_INGEST, VESSELAPI_IN_INGEST # noqa: E402 from config import NATS_URL, FIRMS_INTERVAL, FIRMS_DATASET, AISSTREAM_IN_INGEST # noqa: E402
from sources import ingest_rss_feed, ingest_gdelt, ingest_earthquakes, ingest_eonet, ingest_cisa_kev # noqa: E402 from sources import ingest_rss_feed, ingest_gdelt, ingest_earthquakes # noqa: E402
from fire_sources import ingest_fires # noqa: E402 from fire_sources import ingest_fires # noqa: E402
from ingestor import ingest_event, start_nats_consumer # noqa: E402 from ingestor import ingest_event, start_nats_consumer # noqa: E402
@ -37,8 +37,6 @@ INTERVAL = int(os.getenv("INGEST_INTERVAL", "300"))
GDELT_QUERY = os.getenv("GDELT_QUERY", "") GDELT_QUERY = os.getenv("GDELT_QUERY", "")
ENABLE_QUAKES = os.getenv("INGEST_EARTHQUAKES", "1").lower() in ("1", "true", "yes") ENABLE_QUAKES = os.getenv("INGEST_EARTHQUAKES", "1").lower() in ("1", "true", "yes")
ENABLE_FIRES = os.getenv("INGEST_FIRES", "1").lower() in ("1", "true", "yes") ENABLE_FIRES = os.getenv("INGEST_FIRES", "1").lower() in ("1", "true", "yes")
ENABLE_EONET = os.getenv("INGEST_EONET", "1").lower() in ("1", "true", "yes")
ENABLE_KEV = os.getenv("INGEST_KEV", "1").lower() in ("1", "true", "yes")
NATS_STREAM = "events" NATS_STREAM = "events"
@ -64,18 +62,6 @@ async def producer_loop() -> None:
logger.info("USGS -> %d events", q) logger.info("USGS -> %d events", q)
except Exception: # noqa: BLE001 except Exception: # noqa: BLE001
logger.exception("USGS fetch failed") logger.exception("USGS fetch failed")
if ENABLE_EONET:
try:
n = await ingest_eonet()
logger.info("EONET -> %d events", n)
except Exception: # noqa: BLE001
logger.exception("EONET fetch failed")
if ENABLE_KEV:
try:
k = await ingest_cisa_kev()
logger.info("CISA KEV -> %d events", k)
except Exception: # noqa: BLE001
logger.exception("CISA KEV fetch failed")
except Exception: # noqa: BLE001 except Exception: # noqa: BLE001
logger.exception("producer cycle error") logger.exception("producer cycle error")
await asyncio.sleep(INTERVAL) await asyncio.sleep(INTERVAL)
@ -121,11 +107,6 @@ async def main() -> None:
"ingester starting (rss=%d feeds, gdelt_q=%r, quakes=%s, fires=%s, interval=%ss)", "ingester starting (rss=%d feeds, gdelt_q=%r, quakes=%s, fires=%s, interval=%ss)",
len(RSS_URLS), GDELT_QUERY, ENABLE_QUAKES, ENABLE_FIRES, INTERVAL, len(RSS_URLS), GDELT_QUERY, ENABLE_QUAKES, ENABLE_FIRES, INTERVAL,
) )
try:
from geofence import refresh_cache
await refresh_cache()
except Exception:
logger.exception("geofence cache refresh failed (ST_Intersects still runs on ingest)")
tasks: list[asyncio.Task] = [] tasks: list[asyncio.Task] = []
if ENABLE_FIRES: if ENABLE_FIRES:
# Fire ingest only starts once FIRMS_MAP_KEY is set (ingest_fires logs # Fire ingest only starts once FIRMS_MAP_KEY is set (ingest_fires logs
@ -134,9 +115,6 @@ async def main() -> None:
if AISSTREAM_IN_INGEST: if AISSTREAM_IN_INGEST:
from ais_stream import run_ais_worker # noqa: E402 from ais_stream import run_ais_worker # noqa: E402
tasks.append(asyncio.create_task(run_ais_worker())) tasks.append(asyncio.create_task(run_ais_worker()))
if VESSELAPI_IN_INGEST:
from vesselapi import run_vesselapi_worker # noqa: E402
tasks.append(asyncio.create_task(run_vesselapi_worker()))
await asyncio.gather(producer_loop(), consumer_loop(), *tasks) await asyncio.gather(producer_loop(), consumer_loop(), *tasks)

149
app/run_masscan_service.py Normal file
View file

@ -0,0 +1,149 @@
"""Continuous masscan rolling-sweep service for the OSINT dashboard.
Runs masscan against the configured range for open port 554 (RTSP), streams
the JSON-lines output, and ingests open hosts into the `cameras` table (new
finds only) plus publishes them to NATS exactly like the passive scraper.
Because a full IPv4 sweep at a conservative rate takes days, this runs
masscan CONTINUOUSLY: each pass streams results in as they're found, and when
a pass completes the sweep restarts from the top. New cameras are picked up
on every pass.
Ethics: detection-only (open-port SYN scan). Private/reserved ranges are
excluded and the service REFUSES to start if the excludefile is missing, so
we never probe private space by accident.
Run once (for a manual/test pass): python app/run_masscan_service.py --once
Run forever (systemd): python app/run_masscan_service.py
"""
from __future__ import annotations
import asyncio
import logging
import os
import sys
from pathlib import Path
sys_path = str(Path(__file__).parent)
sys.path.insert(0, sys_path)
import masscan_config as cfg # noqa: E402
from database import init_extensions # noqa: E402
from masscan_scanner import ( # noqa: E402
parse_masscan_line, extract_open_ips, flush,
)
logging.basicConfig(level=logging.INFO,
format="%(asctime)s %(levelname)s %(name)s: %(message)s")
logger = logging.getLogger("osint.masscan_service")
ONCE = "--once" in sys.argv[1:]
def _verify_excludefile() -> None:
"""Fail closed: refuse to sweep the wide range without an excludefile."""
if not cfg.MASSCAN_EXCLUDEFILE:
raise SystemExit("MASSCAN_EXCLUDEFILE is empty — refusing to run")
if not Path(cfg.MASSCAN_EXCLUDEFILE).is_file():
raise SystemExit(
f"excludefile {cfg.MASSCAN_EXCLUDEFILE!r} missing — refusing to "
f"run (would risk probing private ranges). Install the excludefile "
f"first (see deploy/masscan-excludes.txt)."
)
def build_command() -> list[str]:
cmd = [
cfg.MASSCAN_BIN,
cfg.MASSCAN_RANGE,
f"-p{cfg.MASSCAN_PORTS}",
f"--rate={cfg.MASSCAN_RATE}",
f"--retries={cfg.MASSCAN_RETRIES}",
f"--wait={cfg.MASSCAN_WAIT}",
"--output-format=json",
"--output-file=-",
]
if cfg.MASSCAN_EXCLUDEFILE:
cmd.append(f"--excludefile={cfg.MASSCAN_EXCLUDEFILE}")
return cmd
async def _drain_stderr(stream: asyncio.StreamReader) -> None:
"""Consume masscan's progress chatter so its stderr pipe never fills."""
while True:
line = await stream.readline()
if not line:
break
text = line.decode(errors="ignore").strip()
if text and not text.startswith("rate:"):
logger.debug("masscan: %s", text)
async def run_pass() -> tuple[int, int]:
"""Run one full sweep pass, ingesting incrementally.
Returns (new_hosts, total_hosts_seen) for the whole pass.
"""
cmd = build_command()
logger.info("starting masscan pass: %s", " ".join(cmd))
proc = await asyncio.create_subprocess_exec(
*cmd,
stdout=asyncio.subprocess.PIPE,
stderr=asyncio.subprocess.PIPE,
)
if proc.stderr is not None:
asyncio.ensure_future(_drain_stderr(proc.stderr))
seen: set[str] = set()
total_seen = 0
total_new = 0
try:
while True:
raw = await proc.stdout.readline()
if not raw:
break
records = parse_masscan_line(raw.decode(errors="ignore"))
for ip in extract_open_ips(records, 554):
if ip in seen:
continue
seen.add(ip)
if len(seen) >= cfg.MASSCAN_FLUSH_EVERY:
new, _published = await flush(seen, total_new)
total_new += new
total_seen += new
# Drain the final partial batch.
if seen:
new, _published = await flush(seen, total_new)
total_new += new
rc = await proc.wait()
except asyncio.CancelledError:
proc.kill()
raise
logger.info("masscan pass finished (rc=%s): %d new hosts ingested",
rc, total_new)
return total_new, total_seen
async def main() -> None:
_verify_excludefile()
await init_extensions()
logger.info(
"masscan service starting: range=%s ports=%s rate=%s pps (full sweep "
"~%.0fh at this rate)",
cfg.MASSCAN_RANGE, cfg.MASSCAN_PORTS, cfg.MASSCAN_RATE,
4.29e9 / cfg.MASSCAN_RATE / 3600,
)
while True:
try:
await run_pass()
except Exception: # noqa: BLE001
logger.exception("masscan pass error")
if ONCE:
return
# Small gap between passes so the restart is visible in logs.
await asyncio.sleep(5)
if __name__ == "__main__":
asyncio.run(main())

View file

@ -1,289 +0,0 @@
"""CelesTrak satellites last-known overlay.
Fetches GP **JSON** (OMM mean elements not TLE) per group at most once per
2 hours, caches the element blob, and propagates positions with a real SGP4
library on every request. Positions move every second; the *element set* is
what we cache, not the derived lat/lon.
Catalog numbers >= 100000 only fit OMM/JSON, never a 5-column TLE field, so
elements are initialized through :func:`sgp4.omm.initialize` (which consumes
the CelesTrak GP JSON fields verbatim) rather than round-tripping to TLE.
CelesTrak usage policy is non-negotiable: fetch the GP JSON blob at most once
per 2 hours per group, never fan out every GROUP, never also fetch
``GROUP=active`` plus subsets, and identify with ``OSINT_USER_AGENT``.
"""
from __future__ import annotations
import logging
import math
from datetime import datetime, timezone
from urllib.parse import quote
logger = logging.getLogger("osint.satellites")
CELESTRAK_GP = "https://celestrak.org/NORAD/elements/gp.php"
SATNOGS_TLE = "https://db.satnogs.org/api/tle/"
DEFAULT_GROUPS = ("stations", "weather")
ALLOWED_GROUPS = ("stations", "weather", "gps-ops", "starlink")
# CelesTrak policy: do not hit gp.php more than once per 2 hours per group.
SATELLITE_TTL = 2 * 3600.0
SOURCE_CELESTRAK = "celestrak"
SOURCE_SATNOGS = "satnogs"
DEFAULT_LIMIT = 2000
# WGS-84 ellipsoid for TEME -> geodetic.
_WGS84_A = 6378.137
_WGS84_F = 1.0 / 298.257223563
# Last-good element blob per group, kept past TTL so a 403 / "has not updated
# since ..." still serves the previous set instead of failing the overlay.
_last_good: dict[str, list[dict]] = {}
def parse_groups(raw: str | None) -> list[str]:
"""Validate + normalize a comma-separated group list. Raises ValueError.
Starlink is allowed only when explicitly requested (never in the default);
it is a large supplemental feed, not part of the stations/weather default.
"""
groups = [g.strip().lower() for g in (raw or "").split(",") if g.strip()]
if not groups:
raise ValueError("groups must be a non-empty comma-separated list")
bad = [g for g in groups if g not in ALLOWED_GROUPS]
if bad:
raise ValueError(f"unknown group(s): {', '.join(bad)}")
# Dedup, preserve order.
seen: set[str] = set()
out: list[str] = []
for g in groups:
if g not in seen:
seen.add(g)
out.append(g)
return out
def _teme_to_geodetic(
r: tuple[float, float, float],
jd: float,
fr: float,
) -> tuple[float, float, float]:
"""SGP4 TEME position (km) -> geodetic (lat_deg, lon_deg, alt_km).
Rotate TEME into an Earth-fixed frame via GMST, then iterate the WGS-84
geodetic conversion. Good to well under a km for a ground-track overlay.
"""
# GMST (radians) from UT1 ~= UTC here (sub-second error is negligible).
d = (jd + fr) - 2451545.0
t = d / 36525.0
gmst_s = (
67310.54841
+ (876600.0 * 3600.0 + 8640184.812866) * t
+ 0.093104 * t * t
- 6.2e-6 * t * t * t
)
theta = math.radians((gmst_s % 86400.0) / 240.0)
x, y, z = r
xe = x * math.cos(theta) + y * math.sin(theta)
ye = -x * math.sin(theta) + y * math.cos(theta)
ze = z
e2 = _WGS84_F * (2.0 - _WGS84_F)
p = math.sqrt(xe * xe + ye * ye)
lon = math.atan2(ye, xe)
lat = math.atan2(ze, p * (1.0 - e2))
alt = 0.0
for _ in range(10):
n = _WGS84_A / math.sqrt(1.0 - e2 * math.sin(lat) ** 2)
alt = p / math.cos(lat) - n
lat = math.atan2(ze, p * (1.0 - e2 * n / (n + alt)))
n = _WGS84_A / math.sqrt(1.0 - e2 * math.sin(lat) ** 2)
alt = p / math.cos(lat) - n
return math.degrees(lat), math.degrees(lon), alt
def propagate_gp(
elements: list[dict],
group: str,
now: datetime,
) -> list[dict]:
"""Propagate CelesTrak GP JSON elements to geodetic positions at ``now``.
Pure and deterministic given ``now``. Returns ``[{id, name, lat, lon,
alt_km, group}]``; malformed elements and propagation errors are skipped.
"""
from sgp4.api import Satrec, jday
import sgp4.omm as omm
jd, fr = jday(
now.year, now.month, now.day,
now.hour, now.minute, now.second + now.microsecond / 1e6,
)
out: list[dict] = []
for rec in elements:
if not isinstance(rec, dict):
continue
sat = Satrec()
try:
omm.initialize(sat, rec)
except (KeyError, ValueError, TypeError):
continue
err, r, _v = sat.sgp4(jd, fr)
if err != 0:
continue
lat, lon, alt = _teme_to_geodetic(r, jd, fr)
norad = rec.get("NORAD_CAT_ID")
out.append({
"id": str(norad) if norad is not None else "",
"name": rec.get("OBJECT_NAME") or str(norad or ""),
"lat": round(lat, 5),
"lon": round(lon, 5),
"alt_km": round(alt, 2),
"group": group,
})
return out
def _max_epoch(elements: list[dict]) -> str | None:
"""Most recent EPOCH across an element set (ISO-8601 lexical max)."""
epochs = [
str(e["EPOCH"]) for e in elements
if isinstance(e, dict) and e.get("EPOCH")
]
return max(epochs) if epochs else None
def propagate_satnogs_tle(
payload: list[dict],
group: str,
now: datetime,
) -> tuple[list[dict], str | None]:
"""Fallback parser for SatNOGS TLE JSON (``[{tle0,tle1,tle2,updated}]``).
Returns ``(satellites, epoch)`` where epoch is the max ``updated`` time.
Only used when the CelesTrak cache is completely empty.
"""
from sgp4.api import Satrec, jday
jd, fr = jday(
now.year, now.month, now.day,
now.hour, now.minute, now.second + now.microsecond / 1e6,
)
out: list[dict] = []
epochs: list[str] = []
for rec in payload or []:
if not isinstance(rec, dict):
continue
line1 = rec.get("tle1")
line2 = rec.get("tle2")
if not line1 or not line2:
continue
try:
sat = Satrec.twoline2rv(line1, line2)
except (ValueError, TypeError):
continue
e, r, _v = sat.sgp4(jd, fr)
if e != 0:
continue
lat, lon, alt = _teme_to_geodetic(r, jd, fr)
satnum = getattr(sat, "satnum_str", None) or rec.get("norad_cat_id")
name = (rec.get("tle0") or "").strip().lstrip("0").strip() or str(satnum or "")
out.append({
"id": str(satnum).strip() or "",
"name": name,
"lat": round(lat, 5),
"lon": round(lon, 5),
"alt_km": round(alt, 2),
"group": group,
})
if rec.get("updated"):
epochs.append(str(rec["updated"]))
return out, (max(epochs) if epochs else None)
async def _group_elements(group: str) -> tuple[list[dict], str | None]:
"""CelesTrak GP blob for one group, TTL-cached with a last-good fallback.
Returns ``(elements, epoch)``. On a fetch failure (403 / "has not updated
since ...") falls back to the previous successful blob for that group.
"""
from live_layers import _get_json, _ttl_get
url = f"{CELESTRAK_GP}?GROUP={quote(group)}&FORMAT=JSON"
async def _load() -> list[dict]:
data = await _get_json(url)
if not isinstance(data, list):
raise ValueError(f"unexpected CelesTrak payload for {group}")
if data:
_last_good[group] = data
return data
key = f"celestrak:gp:{group}"
try:
elements = await _ttl_get(key, SATELLITE_TTL, _load)
except Exception as exc: # noqa: BLE001
logger.warning("celestrak_fetch_failed group=%s: %s", group, exc)
elements = _last_good.get(group, [])
if not elements:
return [], None
return elements, _max_epoch(elements)
async def fetch_satellites(
groups: list[str],
bbox: str | None = None,
limit: int = DEFAULT_LIMIT,
) -> dict:
"""Assemble the ``/api/satellites`` payload for the requested groups."""
from live_layers import _get_json, _ttl_get, filter_points_bbox, parse_bbox
now = datetime.now(timezone.utc)
satellites: list[dict] = []
epoch: str | None = None
source = SOURCE_CELESTRAK
for group in groups:
elements, group_epoch = await _group_elements(group)
if not elements:
continue
if group_epoch and (epoch is None or group_epoch > epoch):
epoch = group_epoch
satellites.extend(propagate_gp(elements, group, now))
if not satellites:
# Fallback only when the CelesTrak cache is entirely empty — never
# poll both providers every cycle.
async def _load_satnogs() -> list[dict]:
data = await _get_json(SATNOGS_TLE, params={"format": "json"})
return data if isinstance(data, list) else []
try:
satnogs = await _ttl_get("satnogs:tle", SATELLITE_TTL, _load_satnogs)
except Exception as exc: # noqa: BLE001
logger.warning("satnogs_fetch_failed: %s", exc)
satnogs = []
if satnogs:
source = SOURCE_SATNOGS
for group in groups:
rows, sn_epoch = propagate_satnogs_tle(satnogs, group, now)
if sn_epoch and (epoch is None or sn_epoch > epoch):
epoch = sn_epoch
satellites.extend(rows)
if bbox:
minlon, minlat, maxlon, maxlat = parse_bbox(bbox)
satellites = filter_points_bbox(
satellites, minlon, minlat, maxlon, maxlat, limit,
)
else:
satellites = satellites[:limit]
return {
"satellites": satellites,
"source": source,
"tle_epoch": epoch,
"timestamp": now.isoformat(),
}

View file

@ -4,10 +4,10 @@ from __future__ import annotations
from datetime import datetime from datetime import datetime
from enum import Enum from enum import Enum
from typing import Literal, Optional from typing import Optional
from uuid import UUID from uuid import UUID
from pydantic import BaseModel, ConfigDict, Field, field_validator from pydantic import BaseModel, Field
# ─── Enums ─────────────────────────────────────────────────────────────── # ─── Enums ───────────────────────────────────────────────────────────────
@ -63,17 +63,6 @@ class FeedSourceCreate(BaseModel):
config: Optional[dict] = None config: Optional[dict] = None
class FeedSourceUpdate(BaseModel):
"""PATCH /api/sources/{id} — only these keys may be set."""
model_config = ConfigDict(extra="forbid")
name: Optional[str] = None
url: Optional[str] = None
config: Optional[dict] = None
enabled: Optional[bool] = None
class FeedSourceOut(BaseModel): class FeedSourceOut(BaseModel):
id: UUID id: UUID
name: str name: str
@ -284,70 +273,6 @@ class NewsSummaryOut(BaseModel):
id: int id: int
summary_text: str summary_text: str
batch_timestamp: datetime batch_timestamp: datetime
model: Optional[str] = None
kind: Optional[str] = None
class NewsTickerItemOut(BaseModel):
"""One flagged ticker row as exposed by GET /api/news/ticker."""
id: int
headline: str
importance: str
location_name: Optional[str] = None
url: Optional[str] = None
created_at: datetime
class NewsMapItemOut(BaseModel):
"""One flagged map pin as exposed by GET /api/news/map."""
id: int
headline: str
importance: str
location_name: Optional[str] = None
lat: float
lon: float
location_confidence: Optional[str] = None
category: Optional[str] = None
url: Optional[str] = None
created_at: datetime
class NewsModelId(BaseModel):
"""One model id as exposed by GET /api/news/models."""
id: str
class NewsModelsOut(BaseModel):
"""Catalog for the summarizer model selector."""
source: str
models: list[NewsModelId]
class SettingsIn(BaseModel):
"""Body for PUT /api/settings. ``nous_base_url`` is not writable."""
summary_model: str = Field(..., min_length=1, max_length=128)
@field_validator("summary_model")
@classmethod
def summary_model_not_blank(cls, v: str) -> str:
stripped = v.strip()
if not stripped:
raise ValueError("summary_model must be 1128 chars, not whitespace-only")
if len(stripped) > 128:
raise ValueError("summary_model must be 1128 chars, not whitespace-only")
return stripped
class SettingsOut(BaseModel):
"""Current summarizer settings. ``nous_base_url`` is read-only."""
summary_model: str
nous_base_url: str
# ─── Aggregations ──────────────────────────────────────────────────────── # ─── Aggregations ────────────────────────────────────────────────────────
@ -375,46 +300,3 @@ class DashboardSummary(BaseModel):
sentiment: SentimentSummary sentiment: SentimentSummary
top_entities: list[EntityOut] top_entities: list[EntityOut]
class VesselBboxUpdate(BaseModel):
"""Retune the server-side AISStream subscription to a client viewport box.
``bbox`` is "minlon,minlat,maxlon,maxlat" (Leaflet order). ``None``/empty
resets to the env AISSTREAM_BBOX default.
"""
bbox: str | None = None
class GeofenceCreate(BaseModel):
name: str
geojson: dict
active: bool = True
class GeofenceUpdate(BaseModel):
name: Optional[str] = None
geojson: Optional[dict] = None
active: Optional[bool] = None
class ConflictZoneOut(BaseModel):
"""One curated conflict theatre as exposed by GET /api/conflicts."""
id: str
label: str
severity: Literal["war", "high", "elevated"]
lat: float
lon: float
description: str
eventCount: int
lastUpdated: Optional[datetime] = None
class ConflictsOut(BaseModel):
"""Response envelope for GET /api/conflicts."""
zones: list[ConflictZoneOut]
timestamp: datetime

View file

@ -1,219 +0,0 @@
"""OSINT Dashboard — non-secret app settings (keyv-style Postgres table).
Model choice lives here so the summarizer container can read it from Postgres.
Only whitelisted names are stored this is not a generic dump.
Storage: the table is created lazily with ``CREATE TABLE IF NOT EXISTS`` on
first use in each process (same bootstrap pattern as ``api_keys``).
"""
from __future__ import annotations
import asyncio
import os
import time
from datetime import datetime, timezone
import httpx
from sqlalchemy import Column, DateTime, String, Table, Text, func, select, text
import keystore
from database import async_session, engine, metadata
DEFAULT_NOUS_BASE_URL = "https://inference-api.nousresearch.com/v1"
DEFAULT_SUMMARY_MODEL = "Hermes-4.3-36B"
SETTING_SUMMARY_MODEL = "SUMMARY_MODEL"
ALLOWED_SETTINGS = frozenset({SETTING_SUMMARY_MODEL})
MODELS_CACHE_TTL_S = 600.0
MODELS_TIMEOUT_S = 8.0
DEFAULT_MODELS_USER_AGENT = "osint-dashboard-news-summarizer"
FALLBACK_MODELS = [
"Hermes-4.3-36B",
"Hermes-4-70B",
"google/gemini-2.5-flash",
"anthropic/claude-haiku-4.5",
"openai/gpt-4.1-mini",
"x-ai/grok-4",
]
app_settings = Table(
"app_settings",
metadata,
Column("name", String(128), primary_key=True),
Column("value", Text, nullable=False),
Column("updated_at", DateTime(timezone=True), server_default=func.now(), nullable=False),
)
_CREATE_TABLE_SQL = text(
"""
CREATE TABLE IF NOT EXISTS app_settings (
name VARCHAR(128) PRIMARY KEY,
value TEXT NOT NULL,
updated_at TIMESTAMPTZ NOT NULL DEFAULT now()
)
"""
)
_ensure_lock = asyncio.Lock()
_ensured = False
_models_cache: tuple[float, dict] | None = None
class SettingsError(ValueError):
"""Raised when a setting name or value fails validation."""
async def ensure_app_settings_table() -> None:
"""Create the app_settings table if it doesn't exist (idempotent, per process)."""
global _ensured
if _ensured:
return
async with _ensure_lock:
if _ensured:
return
async with engine.begin() as conn:
await conn.execute(_CREATE_TABLE_SQL)
_ensured = True
def nous_base_url() -> str:
"""Read-only Nous inference base URL (env, never writable from the UI)."""
raw = (os.getenv("NOUS_BASE_URL") or "").strip().rstrip("/")
return raw or DEFAULT_NOUS_BASE_URL
def _validate_summary_model(value: str) -> str:
stripped = (value or "").strip()
if not stripped or len(stripped) > 128:
raise SettingsError("summary_model must be 1128 chars, not whitespace-only")
return stripped
async def get_summary_model() -> str:
"""Stored SUMMARY_MODEL, else env, else Hermes-4.3-36B."""
await ensure_app_settings_table()
async with async_session() as session:
row = (
await session.execute(
select(app_settings).where(app_settings.c.name == SETTING_SUMMARY_MODEL)
)
).mappings().one_or_none()
if row and row["value"]:
return row["value"]
return os.getenv("SUMMARY_MODEL", DEFAULT_SUMMARY_MODEL)
async def set_summary_model(value: str) -> dict:
"""Upsert SUMMARY_MODEL and return the public settings payload."""
value = _validate_summary_model(value)
now = datetime.now(timezone.utc)
await ensure_app_settings_table()
async with async_session() as session:
existing = (
await session.execute(
select(app_settings).where(app_settings.c.name == SETTING_SUMMARY_MODEL)
)
).mappings().one_or_none()
if existing:
await session.execute(
app_settings.update()
.where(app_settings.c.name == SETTING_SUMMARY_MODEL)
.values(value=value, updated_at=now)
)
else:
await session.execute(
app_settings.insert().values(
name=SETTING_SUMMARY_MODEL, value=value, updated_at=now
)
)
await session.commit()
return await get_app_settings()
async def get_app_settings() -> dict:
return {
"summary_model": await get_summary_model(),
"nous_base_url": nous_base_url(),
}
def _fallback_payload() -> dict:
return {
"source": "fallback",
"models": [{"id": mid} for mid in FALLBACK_MODELS],
}
async def _nous_api_key() -> str | None:
"""Keystore first, then env. Any lookup failure is treated as missing."""
try:
stored = await keystore.get_api_key("NOUS_API_KEY")
except Exception:
stored = None
if stored and str(stored).strip():
return str(stored).strip()
env = (os.getenv("NOUS_API_KEY") or "").strip()
return env or None
def _models_user_agent() -> str:
return os.getenv("OSINT_USER_AGENT") or DEFAULT_MODELS_USER_AGENT
def _parse_models_payload(body: object) -> list[dict[str, str]]:
if isinstance(body, dict):
raw = body.get("data", body.get("models", []))
elif isinstance(body, list):
raw = body
else:
raw = []
out: list[dict[str, str]] = []
for item in raw or []:
if isinstance(item, str) and item.strip():
out.append({"id": item.strip()})
elif isinstance(item, dict):
mid = item.get("id") or item.get("name")
if mid:
out.append({"id": str(mid)})
return out
async def _http_get(url: str, *, headers: dict[str, str], timeout: float) -> httpx.Response:
async with httpx.AsyncClient(timeout=timeout, headers=headers) as client:
return await client.get(url)
async def list_models() -> dict:
"""Live ``GET {base}/models`` when a key is present; otherwise curated fallback.
Never raises to the caller for missing key or upstream failure.
"""
global _models_cache
key = await _nous_api_key()
if not key:
return _fallback_payload()
now = time.monotonic()
hit = _models_cache
if hit and now - hit[0] < MODELS_CACHE_TTL_S:
return hit[1]
url = f"{nous_base_url()}/models"
headers = {
"Authorization": f"Bearer {key}",
"User-Agent": _models_user_agent(),
"Accept": "application/json",
}
try:
resp = await _http_get(url, headers=headers, timeout=MODELS_TIMEOUT_S)
resp.raise_for_status()
models = _parse_models_payload(resp.json())
if not models:
return _fallback_payload()
payload = {"source": "live", "models": models}
_models_cache = (now, payload)
return payload
except Exception:
return _fallback_payload()

View file

@ -11,8 +11,7 @@ import httpx
import feedparser import feedparser
import nats import nats
from config import NATS_URL, OSINT_USER_AGENT from config import NATS_URL
from upstream_cache import rss_cache
logger = logging.getLogger("osint.sources") logger = logging.getLogger("osint.sources")
@ -26,80 +25,27 @@ def _parse_rfc822(date_str: object) -> str | None:
except (ValueError, TypeError): except (ValueError, TypeError):
return None return None
# NATS connection
NATS_URLS = NATS_URL NATS_URLS = NATS_URL
_nc = None
async def _jetstream():
"""Reuse one NATS connection across publishes (no connect/close per event)."""
global _nc
if _nc is None or _nc.is_closed:
_nc = await nats.connect(NATS_URLS)
return _nc.jetstream()
async def publish_event(subject: str, event: dict): async def publish_event(subject: str, event: dict):
"""Publish an event to NATS JetStream.""" """Publish an event to NATS JetStream."""
js = await _jetstream() nc = await nats.connect(NATS_URLS)
js = nc.jetstream()
await js.publish(subject, json.dumps(event).encode()) await js.publish(subject, json.dumps(event).encode())
await nc.close()
logger.debug("Published event to %s", subject) logger.debug("Published event to %s", subject)
def event_dedup_key(msg: dict) -> str | None:
"""Natural key for generic events. URL when present; else None (always insert)."""
url = msg.get("url")
if not isinstance(url, str):
return None
url = url.strip()
return url or None
async def existing_event_urls(urls: list[str]) -> set[str]:
"""URLs already claimed in event_dedup. Empty input -> empty set."""
if not urls:
return set()
from sqlalchemy import select
from database import async_session
from models import event_dedup as event_dedup_table
async with async_session() as session:
result = await session.execute(
select(event_dedup_table.c.url).where(event_dedup_table.c.url.in_(urls))
)
return {row[0] for row in result}
async def _publish_unknown(subject: str, events: list[dict]) -> int:
"""Publish only events whose URL is not already in event_dedup."""
keys = [event_dedup_key(e) for e in events]
known = await existing_event_urls([k for k in keys if k])
published = 0
for event, key in zip(events, keys):
if key and key in known:
continue
await publish_event(subject, event)
published += 1
return published
def _ua_headers() -> dict[str, str]:
return {"User-Agent": OSINT_USER_AGENT}
# ─── RSS Feed Ingestor ────────────────────────────────────────────────── # ─── RSS Feed Ingestor ──────────────────────────────────────────────────
async def ingest_rss_feed(feed_url: str, source_id: str | None = None): async def ingest_rss_feed(feed_url: str):
"""Fetch and parse an RSS feed, publish items to NATS.""" """Fetch and parse an RSS feed, publish items to NATS."""
text = rss_cache.get(feed_url) async with httpx.AsyncClient(timeout=30) as client:
if text is None: resp = await client.get(feed_url)
async with httpx.AsyncClient(timeout=30, headers=_ua_headers()) as client: resp.raise_for_status()
resp = await client.get(feed_url) feed = feedparser.parse(resp.text)
resp.raise_for_status()
text = resp.text
rss_cache[feed_url] = text
feed = feedparser.parse(text)
count = 0 count = 0
for entry in feed.entries[:100]: # max 100 per run for entry in feed.entries[:100]: # max 100 per run
@ -125,80 +71,51 @@ async def ingest_rss_feed(feed_url: str, source_id: str | None = None):
return count return count
# ─── GDELT 2.0 DOC API ────────────────────────────────────────────────── # ─── GDELT 2.0 Ingestor ─────────────────────────────────────────────────
GDELT_API = "https://api.gdeltproject.org/api/v2/doc/doc" GDELT_API = "https://api.gdeltproject.org/gdeltv2"
GDELT_DEFAULT_QUERY = '(unrest OR protest OR outage OR cyber OR "power outage")'
def gdelt_params(query: str = "", max_articles: int = 50) -> dict[str, str]:
"""DOC 2.0 query string (not the retired gdeltv2 ``search`` param)."""
q = (query or "").strip() or GDELT_DEFAULT_QUERY
return {
"query": q,
"mode": "ArtList",
"format": "json",
"maxrecords": str(int(max_articles)),
"timespan": "1d",
}
def _parse_gdelt_seendate(value: object) -> str:
if isinstance(value, str) and len(value) >= 15:
try:
return datetime.strptime(value[:15], "%Y%m%dT%H%M%S").replace(
tzinfo=timezone.utc
).isoformat()
except ValueError:
pass
return datetime.now(timezone.utc).isoformat()
def parse_gdelt_articles(data: dict) -> list[dict]:
events = []
for article in data.get("articles") or []:
if not isinstance(article, dict):
continue
url = article.get("url")
if not url:
continue
events.append({
"source_type": "gdel-t2",
"title": article.get("title"),
"body": article.get("domain") or article.get("language"),
"url": url,
"location_name": article.get("sourcecountry"),
"source_timestamp": _parse_gdelt_seendate(article.get("seendate")),
"tags": [t for t in (article.get("language"), article.get("sourcecountry")) if t],
"raw": article,
})
return events
async def ingest_gdelt(query: str = "", max_articles: int = 50): async def ingest_gdelt(query: str = "", max_articles: int = 50):
"""Fetch articles from the GDELT DOC 2.0 API.""" """Fetch articles from GDELT 2.0 API."""
params = gdelt_params(query=query, max_articles=max_articles) params = {
data: dict = {"articles": []} "mode": "artlist",
async with httpx.AsyncClient( "format": "json",
timeout=60, headers=_ua_headers(), follow_redirects=True, "maxrecords": max_articles,
) as client: "mode": "artlist",
try: }
resp = await client.get(GDELT_API, params=params) if query:
resp.raise_for_status() params["search"] = query
data = resp.json()
except (httpx.TransportError, httpx.HTTPStatusError) as exc:
# gdeltproject.org certs have expired in the wild; HTTP fallback.
logger.warning("GDELT HTTPS failed (%s); retrying HTTP", exc)
http_url = GDELT_API.replace("https://", "http://", 1)
resp = await client.get(http_url, params=params)
resp.raise_for_status()
data = resp.json()
events = parse_gdelt_articles(data if isinstance(data, dict) else {}) async with httpx.AsyncClient(timeout=60) as client:
for event in events: resp = await client.get(GDELT_API, params=params)
resp.raise_for_status()
data = resp.json()
count = 0
for article in data.get("articles", []):
event = {
"source_type": "gdel-t2",
"title": article.get("title"),
"body": article.get("articleBody"),
"url": article.get("url"),
"sentiment_score": _parse_gdelt_tone(article.get("Tone", "0")),
"location_lat": article.get("Latitude"),
"location_lon": article.get("Longitude"),
"location_name": article.get("Location"),
"source_timestamp": article.get("FirstCreated"),
"entities": [
{"name": e.get("Topic"), "type": "topic"}
for e in article.get("Mentions", [])
if e.get("Topic")
],
"raw": article,
}
await publish_event("events.gdelt", event) await publish_event("events.gdelt", event)
logger.info("Ingested %d articles from GDELT", len(events)) count += 1
return len(events)
logger.info("Ingested %d articles from GDELT", count)
return count
def _parse_gdelt_tone(tone: str) -> float | None: def _parse_gdelt_tone(tone: str) -> float | None:
@ -215,45 +132,34 @@ def _parse_gdelt_tone(tone: str) -> float | None:
USGS_API = "https://earthquake.usgs.gov/earthquakes/feed/v1.0/summary/all_hour.geojson" USGS_API = "https://earthquake.usgs.gov/earthquakes/feed/v1.0/summary/all_hour.geojson"
def parse_usgs_feature(feature: dict) -> dict:
"""Map one USGS GeoJSON feature, keeping the stable event id."""
props = feature.get("properties") or {}
geometry = (feature.get("geometry") or {}).get("coordinates") or []
usgs_id = feature.get("id")
url = props.get("url") or (
f"https://earthquake.usgs.gov/earthquakes/eventpage/{usgs_id}" if usgs_id else None
)
raw = dict(props)
raw["usgs_id"] = usgs_id
return {
"source_type": "earthquake",
"title": props.get("title"),
"body": props.get("description"),
"url": url,
"location_lat": geometry[1] if len(geometry) > 1 else None,
"location_lon": geometry[0] if len(geometry) > 0 else None,
"location_name": props.get("place"),
"sentiment_label": "neutral",
"tags": [f"magnitude:{props.get('mag')}"] if props.get("mag") else [],
"source_timestamp": (
datetime.utcfromtimestamp(props.get("time", 0) / 1000)
.replace(tzinfo=timezone.utc)
.isoformat()
),
"raw": raw,
}
async def ingest_earthquakes(): async def ingest_earthquakes():
"""Fetch recent earthquakes from USGS.""" """Fetch recent earthquakes from USGS."""
async with httpx.AsyncClient(timeout=30, headers=_ua_headers()) as client: async with httpx.AsyncClient(timeout=30) as client:
resp = await client.get(USGS_API) resp = await client.get(USGS_API)
resp.raise_for_status() resp.raise_for_status()
data = resp.json() data = resp.json()
count = 0 count = 0
for feature in data.get("features", []): for feature in data.get("features", []):
event = parse_usgs_feature(feature) props = feature.get("properties", {})
geometry = feature.get("geometry", {}).get("coordinates", [])
event = {
"source_type": "earthquake",
"title": props.get("title"),
"body": props.get("description"),
"url": props.get("url"),
"location_lat": geometry[1] if len(geometry) > 1 else None,
"location_lon": geometry[0] if len(geometry) > 0 else None,
"location_name": props.get("place"),
"sentiment_label": "neutral",
"tags": [f"magnitude:{props.get('mag')}"] if props.get("mag") else [],
"source_timestamp": (
datetime.utcfromtimestamp(props.get("time", 0) / 1000)
.replace(tzinfo=timezone.utc)
.isoformat()
),
"raw": props,
}
await publish_event("events.earthquake", event) await publish_event("events.earthquake", event)
count += 1 count += 1
@ -261,108 +167,6 @@ async def ingest_earthquakes():
return count return count
# ─── NASA EONET v3 ──────────────────────────────────────────────────────
EONET_API = "https://eonet.gsfc.nasa.gov/api/v3/events"
def parse_eonet_events(payload: dict) -> list[dict]:
events = []
for item in payload.get("events") or []:
if not isinstance(item, dict):
continue
eid = item.get("id")
geoms = item.get("geometry") or []
point = None
for g in geoms:
if isinstance(g, dict) and g.get("type") == "Point":
point = g
if point is None:
continue
coords = point.get("coordinates") or []
if len(coords) < 2:
continue
lon, lat = float(coords[0]), float(coords[1])
cats = item.get("categories") or []
tags = []
for c in cats:
if isinstance(c, dict) and c.get("id"):
tags.append(str(c["id"]))
url = item.get("link") or (f"https://eonet.gsfc.nasa.gov/api/v3/events/{eid}" if eid else None)
ts = point.get("date") or datetime.now(timezone.utc).isoformat()
events.append({
"source_type": "disaster",
"title": item.get("title"),
"body": ", ".join(tags) if tags else None,
"url": url,
"location_lat": lat,
"location_lon": lon,
"location_name": item.get("title"),
"tags": tags,
"source_timestamp": ts,
"raw": {**item, "eonet_id": eid},
})
return events
async def ingest_eonet():
"""Volcanoes, storms, floods, drought — gaps USGS/FIRMS don't cover."""
async with httpx.AsyncClient(timeout=30, headers=_ua_headers()) as client:
resp = await client.get(EONET_API, params={"status": "open", "limit": 100})
resp.raise_for_status()
data = resp.json()
events = parse_eonet_events(data if isinstance(data, dict) else {})
published = await _publish_unknown("events.disaster", events)
logger.info("Ingested %d EONET events (%d already known)", published, len(events) - published)
return published
# ─── CISA KEV ───────────────────────────────────────────────────────────
CISA_KEV_API = (
"https://www.cisa.gov/sites/default/files/feeds/known_exploited_vulnerabilities.json"
)
def parse_cisa_kev(payload: dict) -> list[dict]:
events = []
for row in payload.get("vulnerabilities") or []:
if not isinstance(row, dict):
continue
cve = row.get("cveID")
if not cve:
continue
title = row.get("vulnerabilityName") or cve
vendor = row.get("vendorProject") or ""
product = row.get("product") or ""
events.append({
"source_type": "disaster",
"title": f"{cve}: {title}",
"body": row.get("shortDescription") or f"{vendor} {product}".strip(),
"url": f"https://nvd.nist.gov/vuln/detail/{cve}",
"location_lat": None,
"location_lon": None,
"location_name": None,
"tags": ["cisa-kev", cve, "ransomware" if row.get("knownRansomwareCampaignUse") == "Known" else None],
"source_timestamp": row.get("dateAdded") or datetime.now(timezone.utc).isoformat(),
"raw": {**row, "cveID": cve},
})
events[-1]["tags"] = [t for t in events[-1]["tags"] if t]
return events
async def ingest_cisa_kev():
"""Exploited-in-the-wild CVEs. No fake map coords — ticker/events only."""
async with httpx.AsyncClient(timeout=30, headers=_ua_headers(), follow_redirects=True) as client:
resp = await client.get(CISA_KEV_API)
resp.raise_for_status()
data = resp.json()
events = parse_cisa_kev(data if isinstance(data, dict) else {})
published = await _publish_unknown("events.disaster", events)
logger.info("Ingested %d CISA KEV rows (%d already known)", published, len(events) - published)
return published
# ─── Social Signals (Twitter/X-like placeholder) ──────────────────────── # ─── Social Signals (Twitter/X-like placeholder) ────────────────────────
async def ingest_social_signals(query: str = "", max_items: int = 50): async def ingest_social_signals(query: str = "", max_items: int = 50):

File diff suppressed because it is too large Load diff

View file

@ -1,244 +0,0 @@
"""Timescale 1-minute track rollups for DVR playback.
Live overlays stay in memory. Historical `?timestamp=` reads the 1-minute
continuous aggregates (or an in-process downsample when the DB is down).
"""
from __future__ import annotations
import json
from datetime import datetime, timedelta, timezone
from typing import Any, Literal
from sqlalchemy import text
from database import async_session
from live_layers import parse_bbox, to_marker
TRACK_BUCKET = "1 minute"
Kind = Literal["vessel", "aircraft"]
_RAW_TABLE = {
"vessel": "vessel_positions",
"aircraft": "aircraft_positions",
}
_CAGG = {
"vessel": "vessel_tracks_1min",
"aircraft": "aircraft_tracks_1min",
}
_ID_COL = {
"vessel": "mmsi",
"aircraft": "hex",
}
# Last persist time per entity so AIS/ADS-B does not write every frame.
_last_write: dict[tuple[str, str], datetime] = {}
_MIN_WRITE_GAP = timedelta(seconds=20)
def minute_bucket(ts: datetime) -> datetime:
if ts.tzinfo is None:
ts = ts.replace(tzinfo=timezone.utc)
return ts.replace(second=0, microsecond=0)
def downsample_tracks(rows: list[dict]) -> list[dict]:
"""Last sample per id per 1-minute bucket (mirrors the CAGG)."""
last: dict[tuple[str, datetime], dict] = {}
for row in rows:
rid = str(row.get("id") or "")
ts = row.get("ts")
if not rid or not isinstance(ts, datetime):
continue
bucket = minute_bucket(ts)
key = (rid, bucket)
prev = last.get(key)
if prev is None or ts >= prev["ts"]:
last[key] = {**row, "id": rid, "bucket": bucket, "ts": ts}
out = []
for (_id, bucket), row in last.items():
out.append({
"id": row["id"],
"bucket": bucket,
"lat": row.get("lat"),
"lon": row.get("lon"),
"heading": row.get("heading"),
"speed": row.get("speed"),
"label": row.get("label"),
})
return out
def positions_at_timestamp(rows: list[dict], ts: datetime) -> list[dict]:
"""Positions whose 1-minute bucket equals floor(ts)."""
want = minute_bucket(ts)
picked = [r for r in downsample_tracks(rows) if r["bucket"] == want]
return [
to_marker(
r["id"], r.get("lat"), r.get("lon"),
heading=r.get("heading"), speed=r.get("speed"),
label=r.get("label") or r["id"],
)
for r in picked
if r.get("lat") is not None and r.get("lon") is not None
]
def parse_timestamp(value: str | datetime | None) -> datetime | None:
if value is None or value == "":
return None
if isinstance(value, datetime):
ts = value
else:
raw = str(value).strip().replace("Z", "+00:00")
ts = datetime.fromisoformat(raw)
if ts.tzinfo is None:
ts = ts.replace(tzinfo=timezone.utc)
return ts
async def record_position(kind: Kind, marker: dict, ts: datetime | None = None) -> bool:
"""Insert one sample into the raw hypertable (rate-limited)."""
vid = str(marker.get("id") or "")
lat, lon = marker.get("lat"), marker.get("lon")
if not vid or lat is None or lon is None:
return False
now = ts or datetime.now(timezone.utc)
key = (kind, vid)
prev = _last_write.get(key)
if prev is not None and now - prev < _MIN_WRITE_GAP:
return False
_last_write[key] = now
table = _RAW_TABLE[kind]
id_col = _ID_COL[kind]
extra = marker.get("extra") or {}
try:
async with async_session() as session:
await session.execute(
text(
f"""
INSERT INTO {table} ({id_col}, ts, lat, lon, heading, speed, label, extra)
VALUES (:id, :ts, :lat, :lon, :heading, :speed, :label, CAST(:extra AS jsonb))
ON CONFLICT ({id_col}, ts) DO NOTHING
"""
),
{
"id": vid,
"ts": now,
"lat": float(lat),
"lon": float(lon),
"heading": marker.get("heading"),
"speed": marker.get("speed"),
"label": marker.get("label") or vid,
"extra": json.dumps(extra),
},
)
await session.commit()
return True
except Exception:
return False
async def fetch_positions_at(
kind: Kind,
ts: datetime,
bbox: str | None = None,
limit: int = 2000,
) -> list[dict]:
"""Read the 1-minute CAGG for the bucket containing ``ts``."""
bucket = minute_bucket(ts)
table = _CAGG[kind]
id_col = _ID_COL[kind]
where = "bucket = :bucket"
params: dict[str, Any] = {"bucket": bucket, "limit": limit}
if bbox:
minlon, minlat, maxlon, maxlat = parse_bbox(bbox)
where += " AND lon BETWEEN :minlon AND :maxlon AND lat BETWEEN :minlat AND :maxlat"
params.update(minlon=minlon, minlat=minlat, maxlon=maxlon, maxlat=maxlat)
sql = f"""
SELECT {id_col} AS id, lat, lon, heading, speed, label, bucket
FROM {table}
WHERE {where}
LIMIT :limit
"""
try:
async with async_session() as session:
rows = (await session.execute(text(sql), params)).mappings().all()
points = [
to_marker(
r["id"], r["lat"], r["lon"],
heading=r["heading"], speed=r["speed"],
label=r["label"] or r["id"],
extra={"bucket": r["bucket"].isoformat() if r["bucket"] else None, "dvr": True},
)
for r in rows
if r["lat"] is not None and r["lon"] is not None
]
return points
except Exception:
return []
async def track_range() -> dict:
"""Earliest/latest buckets across both CAGGs — slider bounds."""
try:
async with async_session() as session:
row = (await session.execute(text(
"""
SELECT min(t) AS tmin, max(t) AS tmax FROM (
SELECT min(bucket) AS t FROM vessel_tracks_1min
UNION ALL SELECT max(bucket) FROM vessel_tracks_1min
UNION ALL SELECT min(bucket) FROM aircraft_tracks_1min
UNION ALL SELECT max(bucket) FROM aircraft_tracks_1min
UNION ALL SELECT min(poll_at) FROM vessels
UNION ALL SELECT max(poll_at) FROM vessels
) s
"""
))).mappings().first()
if not row or row["tmin"] is None:
now = datetime.now(timezone.utc).replace(second=0, microsecond=0)
return {"min": (now - timedelta(hours=6)).isoformat(), "max": now.isoformat()}
return {
"min": row["tmin"].isoformat(),
"max": row["tmax"].isoformat(),
}
except Exception:
now = datetime.now(timezone.utc).replace(second=0, microsecond=0)
return {"min": (now - timedelta(hours=6)).isoformat(), "max": now.isoformat()}
async def recent_markers(kind: Kind, limit: int = 2000) -> list[dict]:
"""Latest raw sample per id — used when in-process last-known is empty."""
table = _RAW_TABLE[kind]
id_col = _ID_COL[kind]
sql = f"""
SELECT DISTINCT ON ({id_col})
{id_col} AS id, lat, lon, heading, speed, label, extra
FROM {table}
WHERE ts > now() - interval '15 minutes'
ORDER BY {id_col}, ts DESC
LIMIT :limit
"""
try:
async with async_session() as session:
rows = (await session.execute(text(sql), {"limit": limit})).mappings().all()
out = []
for r in rows:
extra = r.get("extra") or {}
if isinstance(extra, str):
try:
extra = json.loads(extra)
except (TypeError, ValueError):
extra = {}
m = to_marker(
r["id"], r["lat"], r["lon"],
heading=r["heading"], speed=r["speed"],
label=r["label"] or r["id"],
extra=extra if isinstance(extra, dict) else {},
)
if m.get("lat") is not None and m.get("lon") is not None:
out.append(m)
return out
except Exception:
return []

View file

@ -1,11 +0,0 @@
"""In-process TTL caches for chatty upstreams (FIRMS, RSS). No Redis."""
from __future__ import annotations
from cachetools import TTLCache
# FIRMS NRT updates every ~510 min; 5 min / 100 keys is enough for bbox×dataset.
firms_cache: TTLCache = TTLCache(maxsize=100, ttl=300)
# News RSS: 1 minute is enough to absorb dashboard double-clicks / retries.
rss_cache: TTLCache = TTLCache(maxsize=100, ttl=60)

View file

@ -1,671 +0,0 @@
"""VesselAPI REST poller — quota-capped AIS for the Middle East (free tier 150 calls/mo).
VesselAPI and AISStream are two independent, first-class vessel providers
not a primary/fallback pair. AISStream (WebSocket) owns live US-coast AIS;
VesselAPI (REST) covers the Strait of Hormuz (default box) where AISStream
has no coverage. Missing one key never disables the other. This worker polls
the REST ``GET /v1/location/vessels/bounding-box`` endpoint at most
``VESSELAPI_MAX_CALLS_PER_DAY`` (default 5) *successful 2xx* calls per UTC day
and upserts the results into the shared ``vessel_last_known`` store.
Idle (no crash) when VESSELAPI_API_KEY is unset. Never called from the GET
/api/vessels path map pans must not hit upstream. One request per poll,
``pagination.limit=50``, never follow ``nextToken``, never send ``filter.sat``.
"""
from __future__ import annotations
import asyncio
import calendar
import json
import logging
import os
from datetime import date, datetime, timezone
import httpx
from sqlalchemy import Column, Date, DateTime, Integer, Table, func, select, text
from config import (
OSINT_USER_AGENT,
VESSELAPI_API_KEY,
VESSELAPI_BBOX,
VESSELAPI_INTERVAL,
VESSELAPI_MAX_CALLS_PER_DAY,
)
from database import async_session, engine, metadata
from live_layers import parse_bbox, to_marker, upsert_vessel, vessel_last_known, vessel_lock
logger = logging.getLogger("osint.vesselapi")
BASE_URL = "https://api.vesselapi.com/v1"
ENDPOINT = f"{BASE_URL}/location/vessels/bounding-box"
MAX_SPAN_DEG = 4.0 # |dLat| + |dLon| — VesselAPI 400s above this.
PAGE_LIMIT = 50 # pagination.limit; never follow nextToken on the free tier.
_client: httpx.AsyncClient | None = None
_client_lock = asyncio.Lock()
# ── Box parsing / span validation ─────────────────────────────────────────
class BboxError(ValueError):
"""A configured VesselAPI box violates the 4° span rule or is malformed."""
def validate_bbox_span(
minlat: float, minlon: float, maxlat: float, maxlon: float,
) -> None:
"""Reject boxes VesselAPI would 400 on (span > 4°, bad order, bad range)."""
if not (-90 <= minlat <= 90 and -90 <= maxlat <= 90
and -180 <= minlon <= 180 and -180 <= maxlon <= 180):
raise BboxError("coordinates out of range")
if minlat >= maxlat or minlon >= maxlon:
raise BboxError("bbox must have min < max on both axes")
dlat = abs(maxlat - minlat)
dlon = abs(maxlon - minlon)
if dlat + dlon > MAX_SPAN_DEG:
raise BboxError(
f"span |dLat|+|dLon| = {dlat + dlon:.2f}° exceeds {MAX_SPAN_DEG}° cap"
)
def parse_boxes(raw: str) -> list[tuple[float, float, float, float]]:
"""Env format: ``minlat,minlon,maxlat,maxlon[; ...]`` (lat/lon order)."""
out: list[tuple[float, float, float, float]] = []
for chunk in (raw or "").split(";"):
parts = [p.strip() for p in chunk.split(",") if p.strip()]
if len(parts) != 4:
continue
try:
minlat = float(parts[0])
minlon = float(parts[1])
maxlat = float(parts[2])
maxlon = float(parts[3])
except ValueError:
continue
out.append((minlat, minlon, maxlat, maxlon))
return out
def parse_boxes_validated(raw: str) -> list[tuple[float, float, float, float]]:
"""Parse boxes, log + skip any that violate the span/order/range rules."""
valid: list[tuple[float, float, float, float]] = []
for box in parse_boxes(raw):
try:
validate_bbox_span(*box)
valid.append(box)
except BboxError as exc:
logger.warning("VesselAPI bbox %r skipped: %s", box, exc)
return valid
# ── Position → marker transform ───────────────────────────────────────────
def _f(value: object) -> float | None:
if value is None or value == "":
return None
try:
return float(value)
except (TypeError, ValueError):
return None
def _s(value: object) -> str | None:
if value is None:
return None
text = str(value).strip()
return text or None
def transform_vesselapi_position(obj: dict | None) -> dict | None:
"""Map one VesselAPI position object to the shared marker contract.
Returns None for glitch rows, missing MMSI, or missing coordinates.
"""
if not obj or not isinstance(obj, dict):
return None
if obj.get("suspected_glitch") is True:
return None
mmsi = obj.get("mmsi")
if mmsi is None:
return None
lat = _f(obj.get("latitude"))
lon = _f(obj.get("longitude"))
if lat is None or lon is None:
return None
mmsi_s = str(mmsi)
name = _s(obj.get("vessel_name") or obj.get("name"))
heading = _f(obj.get("heading"))
if heading is None:
heading = _f(obj.get("cog"))
sog = _f(obj.get("sog"))
extra: dict = {
"src": "vesselapi",
"mmsi": mmsi_s,
"cog": obj.get("cog"),
"sog": obj.get("sog"),
"navstat": obj.get("nav_status"),
}
imo = obj.get("imo")
if imo:
extra["imo"] = imo
dest = _s(obj.get("dest") or obj.get("destination"))
if dest:
extra["dest"] = dest
ts = obj.get("timestamp") or obj.get("processed_timestamp")
if ts:
extra["timestamp"] = ts
return to_marker(
mmsi_s, lat, lon,
heading=heading,
speed=sog,
label=name or mmsi_s,
extra=extra,
)
def transform_vesselapi_payload(payload: dict | None) -> list[dict]:
"""Flatten a bounding-box response ``{vessels: [...]}`` to markers."""
if not payload or not isinstance(payload, dict):
return []
rows = payload.get("vessels") or []
out = []
for row in rows:
marker = transform_vesselapi_position(row)
if marker:
out.append(marker)
return out
def utc_day_start(now: datetime) -> datetime:
"""Floor ``now`` to 00:00:00 UTC."""
if now.tzinfo is None:
now = now.replace(tzinfo=timezone.utc)
now = now.astimezone(timezone.utc)
return now.replace(hour=0, minute=0, second=0, microsecond=0)
def pick_poll_at(poll_times: list[datetime], as_of: datetime) -> datetime | None:
"""Latest poll timestamp at or before ``as_of`` (DVR as-of)."""
if as_of.tzinfo is None:
as_of = as_of.replace(tzinfo=timezone.utc)
else:
as_of = as_of.astimezone(timezone.utc)
eligible: list[datetime] = []
for raw in poll_times:
ts = raw if raw.tzinfo else raw.replace(tzinfo=timezone.utc)
ts = ts.astimezone(timezone.utc)
if ts <= as_of:
eligible.append(ts)
return max(eligible) if eligible else None
def snapshot_as_of(rows: list[dict], as_of: datetime) -> list[dict]:
"""Keep only rows from the latest poll_at ≤ ``as_of``."""
chosen = pick_poll_at(
[r["poll_at"] for r in rows if r.get("poll_at") is not None],
as_of,
)
if chosen is None:
return []
out = []
for row in rows:
ts = row.get("poll_at")
if ts is None:
continue
if ts.tzinfo is None:
ts = ts.replace(tzinfo=timezone.utc)
if ts.astimezone(timezone.utc) == chosen:
out.append(row)
return out
# ── Durable daily quota (Postgres, survives restarts) ─────────────────────
# Mirrors keystore.api_keys: lazy CREATE TABLE IF NOT EXISTS, no alembic fork.
vesselapi_quota = Table(
"vesselapi_quota",
metadata,
Column("day", Date, primary_key=True),
Column("calls", Integer, nullable=False, server_default="0"),
Column("remaining", Integer, nullable=True),
Column("updated_at", DateTime(timezone=True), server_default=func.now(), nullable=False),
)
_CREATE_QUOTA_SQL = text(
"""
CREATE TABLE IF NOT EXISTS vesselapi_quota (
day DATE PRIMARY KEY,
calls INTEGER NOT NULL DEFAULT 0,
remaining INTEGER,
updated_at TIMESTAMPTZ NOT NULL DEFAULT now()
)
"""
)
_quota_lock = asyncio.Lock()
_quota_ensured = False
async def ensure_quota_table() -> None:
global _quota_ensured
if _quota_ensured:
return
async with _quota_lock:
if _quota_ensured:
return
async with engine.begin() as conn:
await conn.execute(_CREATE_QUOTA_SQL)
_quota_ensured = True
class PgQuotaStore:
"""Postgres-backed daily call counter. Injected for tests."""
async def calls_today(self, day: date) -> int:
await ensure_quota_table()
async with async_session() as session:
row = (await session.execute(
select(vesselapi_quota.c.calls).where(vesselapi_quota.c.day == day)
)).scalar()
return int(row) if row else 0
async def remaining_today(self, day: date) -> int | None:
await ensure_quota_table()
async with async_session() as session:
row = (await session.execute(
select(vesselapi_quota.c.remaining).where(vesselapi_quota.c.day == day)
)).scalar()
return int(row) if row is not None else None
async def bump(self, day: date, remaining: int | None) -> int:
await ensure_quota_table()
now = datetime.now(timezone.utc)
async with async_session() as session:
existing = (await session.execute(
select(vesselapi_quota.c.calls).where(vesselapi_quota.c.day == day)
)).scalar()
if existing is None:
await session.execute(
vesselapi_quota.insert().values(
day=day, calls=1, remaining=remaining, updated_at=now,
)
)
else:
await session.execute(
vesselapi_quota.update()
.where(vesselapi_quota.c.day == day)
.values(
calls=vesselapi_quota.c.calls + 1,
remaining=remaining,
updated_at=now,
)
)
await session.commit()
return (int(existing) if existing else 0) + 1
# ── Daily VesselAPI snapshots (DVR as-of + survive restarts) ──────────────
# Cleared at the UTC day boundary so the table holds today's 5 polls only.
_CREATE_VESSELS_SQL = text(
"""
CREATE TABLE IF NOT EXISTS vessels (
mmsi TEXT NOT NULL,
poll_at TIMESTAMPTZ NOT NULL,
lat DOUBLE PRECISION NOT NULL,
lon DOUBLE PRECISION NOT NULL,
heading DOUBLE PRECISION,
speed DOUBLE PRECISION,
label TEXT,
extra JSONB,
PRIMARY KEY (mmsi, poll_at)
)
"""
)
_CREATE_VESSELS_POLL_IDX = text(
"CREATE INDEX IF NOT EXISTS ix_vessels_poll_at ON vessels (poll_at DESC)"
)
_CREATE_VESSELS_BBOX_IDX = text(
"CREATE INDEX IF NOT EXISTS ix_vessels_bbox ON vessels (lon, lat)"
)
_vessels_lock = asyncio.Lock()
_vessels_ensured = False
async def ensure_vessels_table() -> None:
global _vessels_ensured
if _vessels_ensured:
return
async with _vessels_lock:
if _vessels_ensured:
return
async with engine.begin() as conn:
await conn.execute(_CREATE_VESSELS_SQL)
await conn.execute(_CREATE_VESSELS_POLL_IDX)
await conn.execute(_CREATE_VESSELS_BBOX_IDX)
_vessels_ensured = True
def _marker_from_vessel_row(r) -> dict:
extra = r.get("extra") or {}
if isinstance(extra, str):
try:
extra = json.loads(extra)
except (TypeError, ValueError):
extra = {}
if not isinstance(extra, dict):
extra = {}
extra.setdefault("src", "vesselapi")
poll_at = r.get("poll_at")
if poll_at is not None and hasattr(poll_at, "isoformat"):
extra["poll_at"] = poll_at.isoformat()
marker = to_marker(
str(r["id"]), r["lat"], r["lon"],
heading=r.get("heading"), speed=r.get("speed"),
label=r.get("label") or str(r["id"]),
extra=extra,
)
marker["seen_at"] = extra.get("poll_at") or datetime.now(timezone.utc).isoformat()
return marker
async def persist_vessel_snapshot(markers: list[dict], poll_at: datetime) -> None:
"""Write one VesselAPI poll into ``vessels`` (today's snapshots)."""
await ensure_vessels_table()
if not markers:
return
async with async_session() as session:
for m in markers:
vid = str(m.get("id") or "")
lat, lon = m.get("lat"), m.get("lon")
if not vid or lat is None or lon is None:
continue
extra = dict(m.get("extra") or {})
extra.setdefault("src", "vesselapi")
await session.execute(
text(
"""
INSERT INTO vessels
(mmsi, poll_at, lat, lon, heading, speed, label, extra)
VALUES
(:mmsi, :poll_at, :lat, :lon, :heading, :speed, :label,
CAST(:extra AS jsonb))
ON CONFLICT (mmsi, poll_at) DO UPDATE SET
lat = EXCLUDED.lat,
lon = EXCLUDED.lon,
heading = EXCLUDED.heading,
speed = EXCLUDED.speed,
label = EXCLUDED.label,
extra = EXCLUDED.extra
"""
),
{
"mmsi": vid,
"poll_at": poll_at,
"lat": float(lat),
"lon": float(lon),
"heading": m.get("heading"),
"speed": m.get("speed"),
"label": m.get("label") or vid,
"extra": json.dumps(extra),
},
)
await session.commit()
async def purge_old_vessels(before: datetime | None = None) -> None:
"""Drop snapshots from before the current UTC day (or ``before``)."""
await ensure_vessels_table()
cutoff = before or utc_day_start(datetime.now(timezone.utc))
async with async_session() as session:
await session.execute(
text("DELETE FROM vessels WHERE poll_at < :cutoff"),
{"cutoff": cutoff},
)
await session.commit()
async def fetch_vessels_as_of(
ts: datetime,
bbox: str | None = None,
limit: int = 2000,
) -> list[dict]:
"""Latest VesselAPI poll at or before ``ts`` (DVR as-of, not exact minute)."""
try:
await ensure_vessels_table()
async with async_session() as session:
poll = (await session.execute(
text("SELECT max(poll_at) FROM vessels WHERE poll_at <= :ts"),
{"ts": ts},
)).scalar()
if poll is None:
return []
sql = """
SELECT mmsi AS id, lat, lon, heading, speed, label, extra, poll_at
FROM vessels
WHERE poll_at = :poll
"""
params: dict = {"poll": poll, "limit": limit}
if bbox:
minlon, minlat, maxlon, maxlat = parse_bbox(bbox)
sql += (
" AND lon BETWEEN :minlon AND :maxlon"
" AND lat BETWEEN :minlat AND :maxlat"
)
params.update(
minlon=minlon, minlat=minlat, maxlon=maxlon, maxlat=maxlat,
)
sql += " LIMIT :limit"
rows = (await session.execute(text(sql), params)).mappings().all()
return [_marker_from_vessel_row(r) for r in rows]
except Exception:
logger.exception("VesselAPI snapshot fetch failed")
return []
async def hydrate_last_known() -> int:
"""Seed in-memory last-known from today's latest poll (app boot)."""
try:
rows = await fetch_vessels_as_of(datetime.now(timezone.utc))
except Exception:
logger.exception("VesselAPI hydrate failed")
return 0
if not rows:
return 0
async with vessel_lock:
for m in rows:
vid = str(m.get("id") or "")
if vid:
vessel_last_known[vid] = m
return len(rows)
# ── Budget / scheduling (pure, unit-testable) ─────────────────────────────
def days_left_in_month(now: datetime) -> int:
"""UTC days remaining in the current month, inclusive of today."""
_, last = calendar.monthrange(now.year, now.month)
return last - now.day + 1
def budget_allows(
calls_today: int,
remaining: int | None,
days_left: int,
max_per_day: int,
) -> bool:
"""True if another poll is permitted today.
Local hard cap: fewer than ``max_per_day`` successful calls today.
Monthly floor: if ``X-RateLimit-Remaining`` is known, keep at least
``max_per_day * days_left`` in reserve for the rest of the month.
"""
if calls_today >= max_per_day:
return False
if remaining is not None and remaining <= max_per_day * days_left:
return False
return True
def choose_box(
boxes: list[tuple[float, float, float, float]],
calls_today: int,
max_per_day: int,
) -> int:
"""Index into ``boxes`` for the next poll.
Prefer refreshing the first (primary) box rather than spraying one call
across every region round-robin only when the remaining daily budget is
enough to cover all boxes.
"""
if len(boxes) <= 1:
return 0
budget_left = max_per_day - calls_today
if budget_left >= len(boxes):
return calls_today % len(boxes)
return 0
# ── HTTP / poll ───────────────────────────────────────────────────────────
async def _resolve_key() -> str:
from keystore import get_api_key
return (
os.getenv("VESSELAPI_API_KEY")
or VESSELAPI_API_KEY
or (await get_api_key("VESSELAPI_API_KEY"))
or ""
).strip()
async def _get_client() -> httpx.AsyncClient:
global _client
if _client is None:
async with _client_lock:
if _client is None:
_client = httpx.AsyncClient(
timeout=httpx.Timeout(15.0, connect=5.0),
follow_redirects=True,
headers={"User-Agent": OSINT_USER_AGENT, "Accept": "application/json"},
limits=httpx.Limits(max_connections=1, max_keepalive_connections=1),
)
return _client
async def close_client() -> None:
global _client
if _client is not None:
await _client.aclose()
_client = None
def _int_header(value: str | None) -> int | None:
if value is None:
return None
try:
return int(value)
except (TypeError, ValueError):
return None
async def poll_once(store, boxes: list[tuple[float, float, float, float]], key: str) -> bool:
"""One quota-checked poll. Returns True if a successful 2xx was made.
Only successful 2xx responses count against the monthly quota; 4xx/5xx/429
are skipped without retry-storming (Retry-After respected by simply
sleeping the interval).
"""
now = datetime.now(timezone.utc)
today = now.date()
calls = await store.calls_today(today)
remaining = await store.remaining_today(today)
days_left = days_left_in_month(now)
if not budget_allows(calls, remaining, days_left, VESSELAPI_MAX_CALLS_PER_DAY):
logger.info(
"VesselAPI quota reached (calls_today=%d, remaining=%s, days_left=%d) — skip poll",
calls, remaining, days_left,
)
return False
idx = choose_box(boxes, calls, VESSELAPI_MAX_CALLS_PER_DAY)
minlat, minlon, maxlat, maxlon = boxes[idx]
client = await _get_client()
params = {
"filter.latBottom": str(minlat),
"filter.latTop": str(maxlat),
"filter.lonLeft": str(minlon),
"filter.lonRight": str(maxlon),
"pagination.limit": str(PAGE_LIMIT),
}
headers = {"Authorization": f"Bearer {key}"}
try:
resp = await client.get(ENDPOINT, params=params, headers=headers)
except httpx.HTTPError as exc:
logger.warning("VesselAPI request failed: %s", exc)
return False
if resp.status_code == 429:
logger.warning(
"VesselAPI rate-limited (Retry-After=%s) — skip poll",
resp.headers.get("Retry-After"),
)
return False
if resp.status_code >= 400:
logger.warning("VesselAPI HTTP %d — not counted against quota", resp.status_code)
return False
# 2xx success — counts against the monthly quota.
remaining = _int_header(resp.headers.get("X-RateLimit-Remaining"))
calls = await store.bump(today, remaining)
try:
data = resp.json()
except ValueError:
logger.warning("VesselAPI 2xx with non-JSON body — counted but ignored")
return True
markers = transform_vesselapi_payload(data)
for m in markers:
await upsert_vessel(m)
try:
await persist_vessel_snapshot(markers, now)
await purge_old_vessels(utc_day_start(now))
except Exception: # noqa: BLE001 — live overlay must not die on persist
logger.exception("VesselAPI snapshot persist failed")
logger.info(
"VesselAPI poll OK: %d vessels (remaining=%s, calls_today=%d)",
len(markers), remaining, calls,
)
return True
# ── Worker loop ───────────────────────────────────────────────────────────
async def run_vesselapi_worker(store: PgQuotaStore | None = None) -> None:
"""Long-lived poll loop. Idle when the key is unset; never crashes the app."""
if store is None:
store = PgQuotaStore()
boxes = parse_boxes_validated(VESSELAPI_BBOX)
if not boxes:
logger.warning(
"VESSELAPI_BBOX has no valid boxes (span ≤ %.1f°) — poller idle", MAX_SPAN_DEG,
)
while True:
try:
if not boxes:
await asyncio.sleep(VESSELAPI_INTERVAL)
continue
key = await _resolve_key()
if not key:
logger.warning(
"VESSELAPI_API_KEY not set — VesselAPI poller idle. "
"Create a free key at https://dashboard.vesselapi.com/"
)
await asyncio.sleep(VESSELAPI_INTERVAL)
continue
await poll_once(store, boxes, key)
except asyncio.CancelledError:
raise
except Exception: # noqa: BLE001 — keep the loop alive across transient failures
logger.exception("VesselAPI poll error")
await asyncio.sleep(VESSELAPI_INTERVAL)

View file

@ -1,115 +0,0 @@
"""In-memory WebSocket pub/sub with viewport filtering.
Zero extra deps. Ingest workers publish AIS/ADS-B points; only clients whose
current map bbox contains the point receive the payload. No Redis/Kafka.
"""
from __future__ import annotations
import asyncio
from typing import Any
from uuid import UUID
BBox = tuple[float, float, float, float] # minlon, minlat, maxlon, maxlat
def _uuid_str(value: object) -> str | None:
try:
return str(UUID(str(value)))
except (ValueError, TypeError, AttributeError):
return None
def point_in_bbox(lon: float, lat: float, bbox: BBox | None) -> bool:
"""True if (lon, lat) sits inside an axis-aligned viewport."""
if bbox is None:
return False
minlon, minlat, maxlon, maxlat = bbox
return minlon <= lon <= maxlon and minlat <= lat <= maxlat
class ConnectionManager:
"""Maps Tailscale/browser clients → viewport bbox + per-client queue."""
def __init__(self) -> None:
self._queues: dict[str, asyncio.Queue] = {}
self._viewports: dict[str, BBox] = {}
self._watched: dict[str, set[str]] = {}
def register(self, client_id: str, maxsize: int = 256) -> asyncio.Queue:
q: asyncio.Queue = asyncio.Queue(maxsize=maxsize)
self._queues[client_id] = q
return q
def unregister(self, client_id: str) -> None:
self._queues.pop(client_id, None)
self._viewports.pop(client_id, None)
self._watched.pop(client_id, None)
def set_watched_geofences(self, client_id: str, ids: list[str]) -> None:
"""Watch these fence UUIDs so geofence_alert delivers off-viewport.
Invalid UUIDs are ignored. Empty list = watch none (viewport-only).
"""
if client_id not in self._queues:
return
watched: set[str] = set()
for raw in ids:
uid = _uuid_str(raw)
if uid is not None:
watched.add(uid)
self._watched[client_id] = watched
def set_viewport(self, client_id: str, bbox: BBox) -> None:
if client_id in self._queues:
self._viewports[client_id] = bbox
def viewport_of(self, client_id: str) -> BBox | None:
return self._viewports.get(client_id)
def viewports(self) -> list[BBox]:
return list(self._viewports.values())
def has_clients(self) -> bool:
return bool(self._queues)
async def publish_point(
self,
kind: str,
payload: dict[str, Any],
*,
lat: float,
lon: float,
) -> int:
"""Enqueue `{type, payload}` for clients whose viewport contains the point.
kind=geofence_alert also delivers when payload.geofence_id is in the
client's watch set (even if the point is off-viewport). Other kinds
stay viewport-only. Drops the oldest queued message if a client's
buffer is full. Returns the number of clients that got a copy.
"""
msg = {"type": kind, "payload": payload}
sent = 0
gid = _uuid_str(payload.get("geofence_id")) if kind == "geofence_alert" else None
for client_id, queue in list(self._queues.items()):
in_view = point_in_bbox(lon, lat, self._viewports.get(client_id))
if kind == "geofence_alert":
watching = gid is not None and gid in self._watched.get(client_id, set())
if not in_view and not watching:
continue
elif not in_view:
continue
if queue.full():
try:
queue.get_nowait()
except asyncio.QueueEmpty:
pass
try:
queue.put_nowait(msg)
except asyncio.QueueFull:
continue
sent += 1
return sent
manager = ConnectionManager()

31
deploy/README.md Normal file
View file

@ -0,0 +1,31 @@
# systemd unit template — copy to /etc/systemd/system/osint-masscan.service
#
# The masscan service is a CONTINUOUS rolling sweep (a full IPv4 pass at a
# conservative rate takes ~5 days), so it runs as a long-lived service, NOT a
# daily timer. The [Install] WantedBy means it starts at boot and Restart=always
# keeps it up. Install steps (run once on the Pi, as root):
#
# apt install -y masscan # or: apt-get install masscan
# mkdir -p /etc/osint-dashboard /opt/siriusdevops
# cp deploy/masscan-excludes.txt /etc/osint-dashboard/masscan-excludes.txt
#
# # Optional tuning (override env in this file; the DB_* values in the unit
# # already point at the host-published Postgres on 127.0.0.1:5432):
# cat > /etc/osint-dashboard/masscan.env <<'EOF'
# MASSCAN_RANGE=0.0.0.0/0
# MASSCAN_PORTS=554
# MASSCAN_RATE=1000
# EOF
#
# # Venv for the scanner (host-level, not the compose image):
# cd /opt/siriusdevops/osint-dashboard
# python3 -m venv .venv-masscan
# .venv-masscan/bin/pip install -r app/requirements.txt
#
# install -m 644 deploy/osint-masscan.service /etc/systemd/system/
# systemctl daemon-reload
# systemctl enable --now osint-masscan
#
# Watch: journalctl -u osint-masscan -f
# DB: writes into the same Postgres the compose stack uses (127.0.0.1:5432)
# so findings appear on the dashboard camera map automatically.

View file

@ -0,0 +1,33 @@
# masscan excludefile — never probe these ranges.
# RFC1918 private + loopback + link-local + multicast + documentation/bogons.
# The service refuses to start if this file is missing (fail closed).
# Loopback
127.0.0.0/8
# RFC1918 private
10.0.0.0/8
172.16.0.0/12
192.168.0.0/16
# Link-local
169.254.0.0/16
# CGNAT (RFC 6598)
100.64.0.0/10
# Multicast + reserved
224.0.0.0/4
240.0.0.0/4
# Documentation / benchmark / example ranges (never real hosts)
0.0.0.0/8
192.0.2.0/24
198.51.100.0/24
203.0.113.0/24
192.0.0.0/24
198.18.0.0/15
255.255.255.255/32
# Carrier NAT / TEST-NET leftovers
233.252.0.0/24

View file

@ -0,0 +1,29 @@
[Unit]
Description=OSINT dashboard — masscan rolling sweep (open RTSP port 554)
Documentation=https://forgejo.siriusdevops.com/sirius/osint-dashboard
After=network-online.target
Wants=network-online.target
[Service]
Type=simple
# masscan needs raw sockets (CAP_NET_RAW) — run as root on the Pi host.
User=root
WorkingDirectory=/opt/siriusdevops/osint-dashboard
EnvironmentFile=-/etc/osint-dashboard/masscan.env
# Point at the compose-published Postgres on the HOST (127.0.0.1:5432), not the
# docker service name 'postgres' which doesn't resolve outside the compose net.
Environment=DB_HOST=127.0.0.1
Environment=DB_PORT=5432
Environment=DB_USER=osint
Environment=DB_PASSWORD=osint
Environment=DB_NAME=osint_data
Environment=MASSCAN_EXCLUDEFILE=/etc/osint-dashboard/masscan-excludes.txt
ExecStart=/opt/siriusdevops/osint-dashboard/.venv-masscan/bin/python app/run_masscan_service.py
Restart=always
RestartSec=10
# Log the sweep to journald (read with: journalctl -u osint-masscan -f)
StandardOutput=journal
StandardError=journal
[Install]
WantedBy=multi-user.target

View file

@ -1,21 +0,0 @@
# osint.rpi.local — Sentinel-1 SAR tile proxy (/titiler/)
#
# GitOps: this file is the source of truth. On the Pi:
# sudo cp deploy/osint-titiler.nginx.conf /etc/nginx/snippets/osint-titiler.conf
# then `include snippets/osint-titiler.conf;` inside the osint.rpi.local server
# block (before `location /`), `nginx -t && systemctl reload nginx`.
#
# The browser hits /titiler/cog/tiles/... (same-origin). We strip the /titiler
# prefix so self-hosted TiTiler (127.0.0.1:8001) sees /cog/tiles/... and proxy
# its response straight back. Tiles are heavy PNGs — disable buffering so a
# slow client doesn't hold a worker open.
location /titiler/ {
proxy_pass http://127.0.0.1:8001/;
proxy_set_header Host $host;
proxy_set_header X-Real-IP $remote_addr;
proxy_set_header X-Forwarded-For $proxy_add_x_forwarded_for;
proxy_set_header X-Forwarded-Proto $scheme;
proxy_buffering off;
proxy_read_timeout 300s;
}

View file

@ -1,23 +0,0 @@
# osint.rpi.local — WebSocket upgrade for /ws/live
#
# GitOps: this file is the source of truth. On the Pi:
# sudo cp deploy/osint-ws.nginx.conf /etc/nginx/snippets/osint-ws.conf
# then `include snippets/osint-ws.conf;` inside the osint.rpi.local server
# block (before `location /`), `nginx -t && systemctl reload nginx`.
#
# Without these headers nginx proxies GET /ws/live as HTTP/1.0 → FastAPI 404
# and the HUD reconnects every few seconds.
location /ws/ {
proxy_pass http://127.0.0.1:8000;
proxy_http_version 1.1;
proxy_set_header Upgrade $http_upgrade;
proxy_set_header Connection "upgrade";
proxy_set_header Host $host;
proxy_set_header X-Real-IP $remote_addr;
proxy_set_header X-Forwarded-For $proxy_add_x_forwarded_for;
proxy_set_header X-Forwarded-Proto $scheme;
proxy_read_timeout 3600s;
proxy_send_timeout 3600s;
proxy_buffering off;
}

View file

@ -20,10 +20,8 @@ services:
# networks (and ISP abuse-mitigation blackholes) block, so rebuilding it # networks (and ISP abuse-mitigation blackholes) block, so rebuilding it
# on every CI deploy made the pipeline flaky. Rebuild manually when the # on every CI deploy made the pipeline flaky. Rebuild manually when the
# base image or extensions need bumping: # base image or extensions need bumping:
# docker build -f Dockerfile.pg -t localhost/osint-dashboard-pg:latest . # docker compose build db && docker compose up -d db
# FORCE_RECREATE_DB=1 scripts/compose-reup.sh db
image: localhost/osint-dashboard-pg:latest image: localhost/osint-dashboard-pg:latest
pull_policy: never
container_name: osint-db container_name: osint-db
restart: unless-stopped restart: unless-stopped
environment: environment:
@ -32,16 +30,7 @@ services:
POSTGRES_DB: ${DB_NAME:-osint_data} POSTGRES_DB: ${DB_NAME:-osint_data}
# Ensure TimescaleDB is preloaded (conf.d drop-in may be ignored by the # Ensure TimescaleDB is preloaded (conf.d drop-in may be ignored by the
# official image's runtime-generated postgresql.conf, so pass it explicitly). # official image's runtime-generated postgresql.conf, so pass it explicitly).
# shared_buffers capped at 2GB for Pi 5 8GB / 4 cores. command: ["-c", "shared_preload_libraries=timescaledb"]
command:
[
"-c", "shared_preload_libraries=timescaledb",
"-c", "shared_buffers=2GB",
]
deploy:
resources:
limits:
memory: 3G
ports: ports:
- "127.0.0.1:5432:5432" - "127.0.0.1:5432:5432"
volumes: volumes:
@ -54,7 +43,6 @@ services:
nats: nats:
image: nats:2.10 image: nats:2.10
pull_policy: missing
platform: linux/arm64 platform: linux/arm64
container_name: osint-nats container_name: osint-nats
restart: unless-stopped restart: unless-stopped
@ -70,7 +58,6 @@ services:
dockerfile: Dockerfile dockerfile: Dockerfile
platforms: ["linux/arm64"] platforms: ["linux/arm64"]
image: localhost/osint-dashboard:latest image: localhost/osint-dashboard:latest
pull_policy: never
container_name: osint-ingester container_name: osint-ingester
restart: unless-stopped restart: unless-stopped
profiles: ["ingest"] profiles: ["ingest"]
@ -97,15 +84,10 @@ services:
FIRMS_DATASETS: ${FIRMS_DATASETS:-VIIRS_NOAA20_NRT,VIIRS_NOAA21_NRT} FIRMS_DATASETS: ${FIRMS_DATASETS:-VIIRS_NOAA20_NRT,VIIRS_NOAA21_NRT}
FIRMS_BBOX: ${FIRMS_BBOX:--180,-60,180,75} FIRMS_BBOX: ${FIRMS_BBOX:--180,-60,180,75}
FIRMS_INTERVAL: ${FIRMS_INTERVAL:-900} FIRMS_INTERVAL: ${FIRMS_INTERVAL:-900}
OSINT_USER_AGENT: ${OSINT_USER_AGENT:-osint-dashboard/1.0 (self-hosted; lancewalters94@gmail.com)} OSINT_USER_AGENT: ${OSINT_USER_AGENT:-osint-dashboard/1.0 (self-hosted)}
AISSTREAM_API_KEY: ${AISSTREAM_API_KEY:-} AISSTREAM_API_KEY: ${AISSTREAM_API_KEY:-}
AISSTREAM_BBOX: ${AISSTREAM_BBOX:-24,-125,50,-66} AISSTREAM_BBOX: ${AISSTREAM_BBOX:-24,-125,50,-66}
AISSTREAM_IN_INGEST: ${AISSTREAM_IN_INGEST:-0} AISSTREAM_IN_INGEST: ${AISSTREAM_IN_INGEST:-0}
VESSELAPI_API_KEY: ${VESSELAPI_API_KEY:-}
VESSELAPI_BBOX: ${VESSELAPI_BBOX:-25.5,55.4,27.3,57.2}
VESSELAPI_INTERVAL: ${VESSELAPI_INTERVAL:-17280}
VESSELAPI_MAX_CALLS_PER_DAY: ${VESSELAPI_MAX_CALLS_PER_DAY:-5}
VESSELAPI_IN_INGEST: ${VESSELAPI_IN_INGEST:-0}
command: ["python", "app/run_ingester.py"] command: ["python", "app/run_ingester.py"]
entrypoint: ["python", "app/run_ingester.py"] entrypoint: ["python", "app/run_ingester.py"]
@ -115,7 +97,6 @@ services:
dockerfile: Dockerfile dockerfile: Dockerfile
platforms: ["linux/arm64"] platforms: ["linux/arm64"]
image: localhost/osint-dashboard:latest image: localhost/osint-dashboard:latest
pull_policy: never
container_name: osint-dashboard container_name: osint-dashboard
restart: unless-stopped restart: unless-stopped
depends_on: depends_on:
@ -137,61 +118,24 @@ services:
FIRMS_DATASET: ${FIRMS_DATASET:-VIIRS_NOAA20_NRT} FIRMS_DATASET: ${FIRMS_DATASET:-VIIRS_NOAA20_NRT}
FIRMS_DATASETS: ${FIRMS_DATASETS:-VIIRS_NOAA20_NRT,VIIRS_NOAA21_NRT} FIRMS_DATASETS: ${FIRMS_DATASETS:-VIIRS_NOAA20_NRT,VIIRS_NOAA21_NRT}
FIRMS_BBOX: ${FIRMS_BBOX:--180,-60,180,75} FIRMS_BBOX: ${FIRMS_BBOX:--180,-60,180,75}
OSINT_USER_AGENT: ${OSINT_USER_AGENT:-osint-dashboard/1.0 (self-hosted; lancewalters94@gmail.com)} OSINT_USER_AGENT: ${OSINT_USER_AGENT:-osint-dashboard/1.0 (self-hosted)}
NOMINATIM_URL: ${NOMINATIM_URL:-https://nominatim.openstreetmap.org}
NOMINATIM_MIN_INTERVAL: ${NOMINATIM_MIN_INTERVAL:-1.0}
AISSTREAM_API_KEY: ${AISSTREAM_API_KEY:-} AISSTREAM_API_KEY: ${AISSTREAM_API_KEY:-}
AISSTREAM_BBOX: ${AISSTREAM_BBOX:-24,-125,50,-66} AISSTREAM_BBOX: ${AISSTREAM_BBOX:-24,-125,50,-66}
AISSTREAM_IN_APP: ${AISSTREAM_IN_APP:-1} AISSTREAM_IN_APP: ${AISSTREAM_IN_APP:-1}
VESSELAPI_API_KEY: ${VESSELAPI_API_KEY:-}
VESSELAPI_BBOX: ${VESSELAPI_BBOX:-25.5,55.4,27.3,57.2}
VESSELAPI_INTERVAL: ${VESSELAPI_INTERVAL:-17280}
VESSELAPI_MAX_CALLS_PER_DAY: ${VESSELAPI_MAX_CALLS_PER_DAY:-5}
VESSELAPI_IN_APP: ${VESSELAPI_IN_APP:-1}
# ── Self-hosted TiTiler (Sentinel-1 SAR tiles) ──
TITILER_PUBLIC_BASE: ${TITILER_PUBLIC_BASE:-/titiler}
TITILER_INTERNAL_URL: ${TITILER_INTERNAL_URL:-http://titiler:8000}
ports: ports:
- "127.0.0.1:8000:8000" - "127.0.0.1:8000:8000"
deploy:
resources:
limits:
memory: 2G
healthcheck: healthcheck:
test: ["CMD-SHELL", "python -c \"import urllib.request,sys; sys.exit(0 if urllib.request.urlopen('http://127.0.0.1:8000/api/health').status==200 else 1)\""] test: ["CMD-SHELL", "python -c \"import urllib.request,sys; sys.exit(0 if urllib.request.urlopen('http://127.0.0.1:8000/api/health').status==200 else 1)\""]
interval: 30s interval: 30s
timeout: 5s timeout: 5s
retries: 5 retries: 5
# ── Self-hosted TiTiler (Sentinel-1 SAR COG → XYZ tiles) ────────────────
# Warps the signed Planetary Computer COG into WebMercator XYZ tiles so the
# browser never loads a multi-GB GeoTIFF. The FastAPI app signs the COG URL
# and returns a /titiler/... template; nginx routes /titiler/ here.
# Listens on 8000 INSIDE the container (the app already owns host 8000);
# published on host loopback 127.0.0.1:8001 only.
titiler:
image: ghcr.io/developmentseed/titiler:latest@sha256:1809958d063543e3ec858259536002b2de78e9f8f09a22a8d9591bdc2b550b14
pull_policy: missing
container_name: osint-titiler
platform: linux/arm64
restart: unless-stopped
environment:
- PORT=8000
- WORKERS_PER_CORE=1
ports:
- "127.0.0.1:8001:8000"
deploy:
resources:
limits:
memory: 1G
camera-service: camera-service:
build: build:
context: . context: .
dockerfile: Dockerfile dockerfile: Dockerfile
platforms: ["linux/arm64"] platforms: ["linux/arm64"]
image: localhost/osint-dashboard:latest image: localhost/osint-dashboard:latest
pull_policy: never
container_name: osint-camera-scraper container_name: osint-camera-scraper
restart: unless-stopped restart: unless-stopped
profiles: ["ingest"] profiles: ["ingest"]
@ -219,7 +163,7 @@ services:
volumes: volumes:
- camera-snapshots:/data/snapshots - camera-snapshots:/data/snapshots
# ── News pipeline: continuous scraper + 15-min summarizer ─────────────── # ── News pipeline: hourly scraper (:00) + summarizer (:05) ───────────────
# Both services point at the EXISTING osint-db (tables articles + # Both services point at the EXISTING osint-db (tables articles +
# article_summaries, created by idempotent alembic migration 003_news). # article_summaries, created by idempotent alembic migration 003_news).
# Scheduling replaces the upstream k8s CronJobs with in-compose wall-clock # Scheduling replaces the upstream k8s CronJobs with in-compose wall-clock
@ -230,7 +174,6 @@ services:
dockerfile: Dockerfile dockerfile: Dockerfile
platforms: ["linux/arm64"] platforms: ["linux/arm64"]
image: localhost/osint-news-scraper:latest image: localhost/osint-news-scraper:latest
pull_policy: never
container_name: osint-news-scraper container_name: osint-news-scraper
restart: unless-stopped restart: unless-stopped
profiles: ["ingest"] profiles: ["ingest"]
@ -244,7 +187,7 @@ services:
DB_PORT: ${DB_PORT:-5432} DB_PORT: ${DB_PORT:-5432}
DB_NAME: ${DB_NAME:-osint_data} DB_NAME: ${DB_NAME:-osint_data}
LOG_LEVEL: ${NEWS_LOG_LEVEL:-INFO} LOG_LEVEL: ${NEWS_LOG_LEVEL:-INFO}
NEWS_SCRAPE_INTERVAL_S: ${NEWS_SCRAPE_INTERVAL_S:-10} NEWS_SCRAPE_MINUTE: ${NEWS_SCRAPE_MINUTE:-0}
NEWS_SCRAPE_RUN_ON_START: ${NEWS_SCRAPE_RUN_ON_START:-1} NEWS_SCRAPE_RUN_ON_START: ${NEWS_SCRAPE_RUN_ON_START:-1}
# Override the image ENTRYPOINT ["scrapy"] with the scheduler loop. # Override the image ENTRYPOINT ["scrapy"] with the scheduler loop.
entrypoint: [] entrypoint: []
@ -256,7 +199,6 @@ services:
dockerfile: Dockerfile dockerfile: Dockerfile
platforms: ["linux/arm64"] platforms: ["linux/arm64"]
image: localhost/osint-news-summarizer:latest image: localhost/osint-news-summarizer:latest
pull_policy: never
container_name: osint-news-summarizer container_name: osint-news-summarizer
restart: unless-stopped restart: unless-stopped
profiles: ["ingest"] profiles: ["ingest"]
@ -269,19 +211,14 @@ services:
DB_HOST: db DB_HOST: db
DB_PORT: ${DB_PORT:-5432} DB_PORT: ${DB_PORT:-5432}
DB_NAME: ${DB_NAME:-osint_data} DB_NAME: ${DB_NAME:-osint_data}
NOUS_API_KEY: ${NOUS_API_KEY:-} # Required to do real work; unset → the loop logs and idles.
NOUS_BASE_URL: ${NOUS_BASE_URL:-https://inference-api.nousresearch.com/v1} GEMINI_API_KEY: ${GEMINI_API_KEY:-}
SUMMARY_MODEL: ${SUMMARY_MODEL:-} SUMMARY_MODEL: ${SUMMARY_MODEL:-gemini-2.0-flash}
OSINT_USER_AGENT: ${OSINT_USER_AGENT:-osint-dashboard-news-summarizer}
BATCH_SIZE: ${NEWS_BATCH_SIZE:-50} BATCH_SIZE: ${NEWS_BATCH_SIZE:-50}
SUMMARY_WINDOW_MINUTES: ${SUMMARY_WINDOW_MINUTES:-15} SUMMARY_WINDOW_HOURS: ${SUMMARY_WINDOW_HOURS:-1}
INCLUDE_FUTURES: ${INCLUDE_FUTURES:-0} INCLUDE_FUTURES: ${INCLUDE_FUTURES:-0}
NEWS_SUMMARIZE_INTERVAL_S: ${NEWS_SUMMARIZE_INTERVAL_S:-900} NEWS_SUMMARIZE_MINUTE: ${NEWS_SUMMARIZE_MINUTE:-5}
NEWS_SUMMARIZE_RUN_ON_START: ${NEWS_SUMMARIZE_RUN_ON_START:-1} NEWS_SUMMARIZE_RUN_ON_START: ${NEWS_SUMMARIZE_RUN_ON_START:-1}
NEWS_SUMMARIZE_FORCE: ${NEWS_SUMMARIZE_FORCE:-0}
TZ: ${TZ:-America/New_York}
NEWS_RECAP_HOUR: ${NEWS_RECAP_HOUR:-23}
NEWS_RECAP_MINUTE: ${NEWS_RECAP_MINUTE:-0}
command: ["python", "run_news_summarizer.py"] command: ["python", "run_news_summarizer.py"]
volumes: volumes:

View file

@ -2,7 +2,7 @@
Builder brief for backend + frontend. Researched 2026-08-27. Every endpoint below was either live-probed from this machine or taken from the providers current docs. Prefer **free, no-key, CORS-open** sources first. Keys are called out explicitly. Builder brief for backend + frontend. Researched 2026-08-27. Every endpoint below was either live-probed from this machine or taken from the providers current docs. Prefer **free, no-key, CORS-open** sources first. Keys are called out explicitly.
This is **not** a camera-discovery change. Existing camera rules still apply: never emit `rtsp://` hrefs; camera pins go through `/api/cameras/{id}/snapshot`; HTTP directory cams use `/stream` MJPEG. This is **not** a camera-discovery / masscan change. Existing camera rules still apply: never emit `rtsp://` hrefs; masscan pins go through `/api/cameras/{id}/snapshot`; HTTP directory cams use `/stream` MJPEG.
--- ---
@ -13,7 +13,7 @@ This is **not** a camera-discovery change. Existing camera rules still apply: ne
| NASA FIRMS VIIRS hotspots | Ingested (`app/fire_sources.py` → NATS `events.fire``fires` hypertable → `GET /api/fires`) | Needs free `FIRMS_MAP_KEY`. See `docs/firms.md`. | | NASA FIRMS VIIRS hotspots | Ingested (`app/fire_sources.py` → NATS `events.fire``fires` hypertable → `GET /api/fires`) | Needs free `FIRMS_MAP_KEY`. See `docs/firms.md`. |
| NASA GIBS basemaps | Frontend tiles via `app/gibs_map.py` | No key. CORS `*`. | | NASA GIBS basemaps | Frontend tiles via `app/gibs_map.py` | No key. CORS `*`. |
| GIBS VIIRS thermal tiles | Documented, not wired as overlay | Same GIBS stack; no key. | | GIBS VIIRS thermal tiles | Documented, not wired as overlay | Same GIBS stack; no key. |
| Cameras | Scraper → `cameras` table | Defaults already include ALERTWest JPEGs + Live-Environment-Streams HLS/YouTube GeoJSON. | | Cameras | Scraper + masscan `cameras` table | Defaults already include ALERTWest JPEGs + Live-Environment-Streams HLS/YouTube GeoJSON. |
| News / RSS / GDELT / USGS quakes | Ingest | Out of scope for this brief. | | News / RSS / GDELT / USGS quakes | Ingest | Out of scope for this brief. |
**Action for existing fire ingest:** NASA will stop Suomi NPP product delivery on **2026-11-01**. Switch `FIRMS_DATASET` from `VIIRS_SNPP_NRT` to `VIIRS_NOAA20_NRT` and/or `VIIRS_NOAA21_NRT` before then.[20] **Action for existing fire ingest:** NASA will stop Suomi NPP product delivery on **2026-11-01**. Switch `FIRMS_DATASET` from `VIIRS_SNPP_NRT` to `VIIRS_NOAA20_NRT` and/or `VIIRS_NOAA21_NRT` before then.[20]
@ -261,7 +261,7 @@ Use later if you want commuter rail / subway vehicle positions (LA Metro, MTA, e
## 6. Open video / camera feeds (official public only) ## 6. Open video / camera feeds (official public only)
Do **not** add Insecam-style random IP cams as a new source. The scraper already has a public list; this section is **agency-published** JPEG/HLS. Do **not** add Insecam-style random IP cams as a new source. The scraper already has a public list + masscan; this section is **agency-published** JPEG/HLS.
### 6.1 Already wired ### 6.1 Already wired
@ -304,7 +304,7 @@ Do not call the YouTube Data API unless you want search. Embedding existing stre
### 6.5 Skip ### 6.5 Skip
- Insecam / random “public IP cam” aggregators — ToS / privacy. - Insecam / random “public IP cam” aggregators — ToS / privacy / already covered by masscan ethics.
- TrafficLand, EarthCam commercial APIs. - TrafficLand, EarthCam commercial APIs.
- SkylineWebcams — scraping, not an API. - SkylineWebcams — scraping, not an API.
@ -523,7 +523,7 @@ Attribution bar (required): OpenSky / ADSB.lol ODbL / Amtraker / RainViewer / IE
## 12. Legal / ethics (non-negotiable) ## 12. Legal / ethics (non-negotiable)
- RTSP policy unchanged (never emit `rtsp://` hrefs). - Masscan / RTSP policy unchanged.
- AISStream: server-side only; do not put the key in JS.[5] - AISStream: server-side only; do not put the key in JS.[5]
- OpenSky: non-commercial unless licensed; cite if you publish.[2] - OpenSky: non-commercial unless licensed; cite if you publish.[2]
- ADSB.lol: ODbL share-alike on derived databases.[4] - ADSB.lol: ODbL share-alike on derived databases.[4]

View file

@ -1,95 +1,67 @@
# News pipeline — scraper + Nous Portal summarizer # News pipeline — scraper + summarizer
The OSINT dashboard ingests a large curated feed list (`news/scraper/urls.txt`) The OSINT dashboard ingests ~257 global news RSS sources hourly and produces
continuously and produces an English LLM brief plus flagged ticker/map rows LLM master summaries. Both services were vendored from the upstream
every 15 minutes. Both services were vendored from the upstream `~/Projects/newsPipeline` project and re-integrated here to replace the old
`~/Projects/newsPipeline` project and re-integrated here against the EXISTING k8s CronJob choreography with in-compose scheduling against the EXISTING
osint-db — **no second Postgres**. The LLM is **Nous Portal** osint-db — **no second Postgres**.
(`inference-api.nousresearch.com`) — not Gemini.
## Architecture ## Architecture
``` ```
urls.txt (RSS + homepages) 257 RSS feeds (news/scraper/urls.txt)
news-scraper (Scrapy, continuous) ──► articles table (osint-db) news-scraper (Scrapy, hourly :00) ──► articles table (osint-db)
│ │ │ │
│ ▼ │ ▼
news-summarizer (Nous Portal, every 15m + 23:00 recap) ──► article_summaries + news_items news-summarizer (Gemini map-reduce, hourly :05) ──► article_summaries table
GET /api/news · /api/news/summaries · /api/news/ticker · /api/news/map GET /api/news · GET /api/news/summaries
GET /api/news/models · GET/PUT /api/settings
``` ```
| Component | Image | Container | Scheduling | | Component | Image | Container | Scheduling |
|---|---|---|---| |---|---|---|---|
| Scraper | `localhost/osint-news-scraper` | `osint-news-scraper` | loop, `NEWS_SCRAPE_INTERVAL_S` (default 10s after each crawl) | | Scraper | `localhost/osint-news-scraper` | `osint-news-scraper` | wall-clock loop, minute `NEWS_SCRAPE_MINUTE` (default :00) |
| Summarizer | `localhost/osint-news-summarizer` | `osint-news-summarizer` | loop, `NEWS_SUMMARIZE_INTERVAL_S` (default 900s) | | Summarizer | `localhost/osint-news-summarizer` | `osint-news-summarizer` | wall-clock loop, minute `NEWS_SUMMARIZE_MINUTE` (default :05) |
Both services live under the `ingest` compose profile (same as the ingester Both services live under the `ingest` compose profile (same as the ingester
and camera-scraper): `docker compose --profile ingest up -d`. and camera-scraper): `docker compose --profile ingest up -d`.
The summarizer is a batch sidecar, **not** a live overlay. Do **not** reuse
`GET /api/alerts` (dashboard entity/keyword alerts). Do **not** stuff news
into `overlay_catalog()``/api/map/layers` `overlays` stays live upstream
feeds (`GET /api/news` exact key set is unchanged on purpose).
## Data flow ## Data flow
1. **Scraper**`news/scraper/run_news_scraper.py` runs 1. **Scraper**`news/scraper/run_news_scraper.py` runs
`scrapy crawl articles` back-to-back (default 10s pause). The spider reads `scrapy crawl articles` (spider `news/scraper/newsScraper/spiders/news_spider.py`)
URLs from `urls.txt` (homepages autodiscover RSS; feed URLs are parsed at the top of each hour. The spider reads the RSS feed URLs from `urls.txt`,
directly), follows each `<item>` link, extracts the main article body, and follows each `<item>` link, extracts the main article body, and the
the `PostgresPipeline` writes to `articles` with URL-based dedup `PostgresPipeline` writes to `articles` with URL-based dedup
(`ON CONFLICT (url) DO NOTHING`). (`ON CONFLICT (url) DO NOTHING`).
2. **Summarizer**`news/summerizer/run_news_summarizer.py` runs 2. **Summarizer**`news/summerizer/run_news_summarizer.py` runs
`summarizer.py` every `NEWS_SUMMARIZE_INTERVAL_S` (default 900) over the `summarizer.py` at :05 past each hour. It reads articles from the last
last `SUMMARY_WINDOW_MINUTES` (default 15), and again at 23:00 `SUMMARY_WINDOW_HOURS`, map-reduces them through Gemini
`America/New_York` (`TZ`) over the last 24 hours as a daily recap (`SUMMARY_MODEL`, default `gemini-2.0-flash`), and inserts one master
(`kind=daily_recap`). Both map-reduce through Nous Portal (`SUMMARY_MODEL` summary into `article_summaries`.
/ Settings, default `Hermes-4.3-36B`), write the English brief to
`article_summaries` (column `model` is the LLM id; `kind` is
`interval` or `daily_recap`), and flagged ticker/map rows to `news_items`.
Loops are serial (two crawls/summaries never overlap). Interval idempotency: Scheduling is done with small in-compose wall-clock loops (not host cron): each
if `article_summaries` already has a row in the last interval, the summarizer loop runs once on boot (`*_RUN_ON_START=1`, seeds data fast) then sleeps until
**skips** (prevents double-pins on `RUN_ON_START` recreate). Set the next scheduled minute. The loop is serial, so a run that overruns its slot
`NEWS_SUMMARIZE_FORCE=1` to ignore that skip. simply shifts to the next boundary — two crawls/summaries never overlap.
The `articles` and `article_summaries` tables are created by the idempotent The `articles` and `article_summaries` tables are created by the idempotent
alembic migration `003_news` (also created by the scraper's own alembic migration `003_news` (also created by the scraper's own
`CREATE TABLE IF NOT EXISTS`). `news_items` is alembic `005_news_items`. `CREATE TABLE IF NOT EXISTS`, so container startup order doesn't matter).
Container startup order doesn't matter.
## Keys and Settings
- **`NOUS_API_KEY`** — paste in the dashboard **Keys** UI (`api_keys` /
`keystore.KEY_REGISTRY`). Env / `.env` is an **override** (env wins, same
as FIRMS). Never returned by any API; never emitted into `index.html`;
never proxied from the browser.
- **Idle without a key** — if env is unset **and** the keystore row is empty,
the summarizer logs and idles (never crashes). News intel APIs return `[]`.
- **Model** — non-secret. Settings UI model selector `PUT /api/settings`
`{ "summary_model": "…" }` stores `SUMMARY_MODEL` in `app_settings` (1128
chars). `GET /api/settings` echoes `{summary_model, nous_base_url}`.
`nous_base_url` is read-only. Default `Hermes-4.3-36B`. Live catalog is
best-effort `GET /api/news/models`.
## Endpoints ## Endpoints
### GET /api/news — recent articles ### GET /api/news — recent articles
Key set **unchanged** (no `lat`/`lon` on articles; geo lives on `/api/news/map`).
| Query param | Meaning | Default | | Query param | Meaning | Default |
|---|---|---| |---|---|---|
| `domain` | filter by source domain (e.g. `www.reuters.com`) | none | | `domain` | filter by source domain (e.g. `www.reuters.com`) | none |
| `since` | only articles captured at/after this UTC instant (ISO-8601) | none | | `since` | only articles captured at/after this UTC instant (ISO-8601) | none |
| `limit` | max rows | `50` (max `500`) | | `limit` | max rows | `50` (max `500`) |
| `offset` | pagination offset | `0` | | `offset` | pagination offset | `0` |
| `include_content` | include full article body | `false` |
```json ```json
[ [
@ -97,14 +69,14 @@ Key set **unchanged** (no `lat`/`lon` on articles; geo lives on `/api/news/map`)
"id": 1, "id": 1,
"title": "…", "title": "…",
"url": "https://…", "url": "https://…",
"content": null, "content": "full extracted article text…",
"domain": "www.reuters.com", "domain": "www.reuters.com",
"timestamp": "2026-08-24T18:10:00Z" "timestamp": "2026-08-24T18:10:00Z"
} }
] ]
``` ```
### GET /api/news/summaries — master LLM briefs ### GET /api/news/summaries — master LLM summaries
| Query param | Meaning | Default | | Query param | Meaning | Default |
|---|---|---| |---|---|---|
@ -116,185 +88,52 @@ Key set **unchanged** (no `lat`/`lon` on articles; geo lives on `/api/news/map`)
[ [
{ {
"id": 1, "id": 1,
"summary_text": "English markdown brief…", "summary_text": "master LLM summary (markdown)…",
"batch_timestamp": "2026-08-24T18:10:00Z", "batch_timestamp": "2026-08-24T18:10:00Z"
"model": "Hermes-4.3-36B",
"kind": "daily_recap"
} }
] ]
``` ```
`model` and `kind` are additive (`interval` | `daily_recap` | `null` for old rows).
`?kind=daily_recap` pins the nightly 24h recap. Empty DB → `[]` (no crash).
Malformed `kind``422`.
### GET /api/news/ticker — HUD headlines
Critical/high `news_items` with `kind=ticker` first. If none are flagged,
medium/low ticker rows fill the tape so the dock is not blank. Do **not**
reuse `GET /api/alerts`. Bottom HUD `#nt-track` scrolls these rows, not a
dump of the whole brief.
| Query param | Meaning | Default |
|---|---|---|
| `since` | only items created at/after this UTC instant | none |
| `limit` | max rows | `20` (max `50`) |
```json
[
{
"id": 1,
"headline": "…",
"importance": "critical",
"location_name": "Kyiv",
"url": "https://…",
"created_at": "2026-08-24T18:10:00Z"
}
]
```
### GET /api/news/map — geolocated critical/high pins
Only rows with valid `lat`/`lon`. Optional bbox. **No zoom skip** — world
view is the point. Layer-panel toggle uses this dedicated path (same as
event blips), not `overlay_catalog`.
| Query param | Meaning | Default |
|---|---|---|
| `bbox` | `minlon,minlat,maxlon,maxlat` | all flagged pins |
| `since` | only items created at/after this UTC instant | last 24 hours |
| `limit` | max rows | `200` (max `500`) |
Malformed bbox → `422`.
```json
[
{
"id": 1,
"headline": "…",
"importance": "high",
"location_name": "Kyiv",
"lat": 50.45,
"lon": 30.52,
"location_confidence": "city",
"category": "military/conflict",
"url": "https://…",
"created_at": "2026-08-24T18:10:00Z"
}
]
```
Pins are LLM-estimated and clamped (`lat∈[-90,90]`, `lon∈[-180,180]`). No
Nominatim. No writes into `events`.
### GET /api/news/models — Settings dropdown catalog
Never 502s. `{ "source": "live"|"fallback", "models": [{"id": "…"}] }`.
### GET /api/settings · PUT /api/settings
```json
{ "summary_model": "Hermes-4.3-36B", "nous_base_url": "https://inference-api.nousresearch.com/v1" }
```
PUT body is `{ "summary_model": "<1128 char id>" }`. `nous_base_url` is
ignored even if sent.
## Reduce JSON contract
Reduce phase (`response_format: json_object`, English only) must be a single
object. Parser (`intel.parse_reduce_json`) strips `<think>…</think>` and
markdown json fences, then brace-slices:
```json
{
"summary_en": "English markdown brief or the no-qualifying-events sentence",
"ticker": [
{"headline": "", "importance": "critical", "url": "", "location_name": ""}
],
"map_items": [
{
"headline": "",
"importance": "critical",
"location_name": "",
"lat": 0,
"lon": 0,
"location_confidence": "city",
"category": "military/conflict",
"url": ""
}
]
}
```
Persist ticker for critical/high first; if none, persist medium/low so the
tape is not empty. Map rows stay critical/high with valid coords; Unknown /
invented places are dropped. Caps: 12 ticker (≤140 chars, no markdown), 20
map. `summary_en` lands in `article_summaries.summary_text`.
## Configuration (all via env / `.env`) ## Configuration (all via env / `.env`)
| Var | Default | Notes | | Var | Default | Notes |
|---|---|---| |---|---|---|
| `NOUS_API_KEY` | *(blank)* | **Required for summaries.** Prefer Keys UI; env overrides. Unset in **both** env and `api_keys` = summarizer logs and idles (never crashes); APIs return `[]`. | | `GEMINI_API_KEY` | *(blank)* | **Required for summaries.** Unset = summarizer logs and idles (never crashes). |
| `NOUS_BASE_URL` | `https://inference-api.nousresearch.com/v1` | Read-only in Settings. | | `SUMMARY_MODEL` | `gemini-2.0-flash` | Gemini model id. |
| `SUMMARY_MODEL` | `Hermes-4.3-36B` | Compose default. Operator-facing choice is Settings → `app_settings.SUMMARY_MODEL`. | | `NEWS_BATCH_SIZE` | `50` | Articles per map-phase batch. |
| `NEWS_BATCH_SIZE` | `50` | Articles per map-phase batch (compose maps to container `BATCH_SIZE`). | | `SUMMARY_WINDOW_HOURS` | `1` | How far back the summarizer looks for new articles. |
| `SUMMARY_WINDOW_MINUTES` | `15` | How far back the summarizer looks for new articles. | | `INCLUDE_FUTURES` | `0` | Legacy futures-prices coupling (upstream pipeline). OFF for OSINT; set `1` + install `yfinance` to enable. |
| `NEWS_SCRAPE_INTERVAL_S` | `10` | Pause after each crawl before the next (scraper is otherwise continuous). | | `NEWS_SCRAPE_MINUTE` | `0` | Wall-clock minute the scraper fires. |
| `NEWS_SUMMARIZE_INTERVAL_S` | `900` | Seconds between analyst runs (default 15 min). | | `NEWS_SUMMARIZE_MINUTE` | `5` | Wall-clock minute the summarizer fires. |
| `TZ` | `America/New_York` | Timezone for the 23:00 daily recap. |
| `NEWS_RECAP_HOUR` | `23` | Local hour of the daily 24h recap. |
| `NEWS_RECAP_MINUTE` | `0` | Local minute of the daily recap. |
| `NEWS_SCRAPE_RUN_ON_START` | `1` | Run one scrape immediately on container start. | | `NEWS_SCRAPE_RUN_ON_START` | `1` | Run one scrape immediately on container start. |
| `NEWS_SUMMARIZE_RUN_ON_START` | `1` | Run one summarize immediately on container start. | | `NEWS_SUMMARIZE_RUN_ON_START` | `1` | Run one summarize immediately on container start. |
| `NEWS_SUMMARIZE_FORCE` | `0` | `1` ignores the interval/recap idempotency skip (double-pins on recreate). |
| `INCLUDE_FUTURES` | `0` | Legacy. Ignored — prompts never inject futures/market tape. |
| `NEWS_LOG_LEVEL` | `INFO` | Scrapy log level. | | `NEWS_LOG_LEVEL` | `INFO` | Scrapy log level. |
| `OSINT_USER_AGENT` | `osint-dashboard-news-summarizer` | Sent on every outbound Nous call. |
| `TELEGRAM_TOKEN` / `TELEGRAM_CHAT_ID` | *(blank)* | Reserved for the (out-of-scope) Telegram delivery bot. | | `TELEGRAM_TOKEN` / `TELEGRAM_CHAT_ID` | *(blank)* | Reserved for the (out-of-scope) Telegram delivery bot. |
DB_* for both services is mapped to the shared osint-db credentials DB_* for both services is mapped to the shared osint-db credentials
(`DB_HOST=db`, same `DB_USER/DB_PASSWORD/DB_NAME` as the rest of the stack). (`DB_HOST=db`, same `DB_USER/DB_PASSWORD/DB_NAME` as the rest of the stack).
Nous chat: `POST {NOUS_BASE_URL}/chat/completions` via `news/summerizer/nous_client.py`
(`httpx`, no `openai` SDK). Auth is a Bearer token from `NOUS_API_KEY`.
No Hermes-4 reasoning system prompt. Reduce uses `json_mode=True`.
## Prompts ## Prompts
Both prompts are env-overridable. Defaults recap the articles actually Both prompts are env-overridable — the default `MAP_PROMPT` is OSINT-neutral
provided, ranked by breaking important news, and ignore futures / commodity (facts, locations, entities, category, OSINT signal per article) and the default
tape. ticker/map may be empty; `summary_en` must still be a real brief. `SUMMARY_PROMPT` produces a concise executive summary of the most impactful
`RECAP_PROMPT` (23:00, 24h window) is the daily recap; `SUMMARY_PROMPT` is the items (with a "no qualifying events" escape hatch). Upstream's futures/markets
15-min analyst. `INCLUDE_FUTURES` is ignored. prompt language is gated behind `INCLUDE_FUTURES=1`.
## Tests ## Tests
```bash `tests/test_api_news.py` — DB-backed API contract tests (auto-skip without a
PYTHONPATH=news/summerizer pytest news/summerizer/tests -v reachable test database, same as the FIRMS tests):
# intel + nous_client tests PASS (no network)
PYTHONPATH=app pytest tests/test_api_news.py \ ```bash
tests/test_api_settings.py tests/test_api_live_layers.py -v DB_HOST=... DB_PORT=... DB_USER=osint DB_PASSWORD=... DB_NAME=osint_data \
# DB-marked tests skip without Postgres; live_layers must still PASS pytest tests/test_api_news.py -v
# /api/map/layers overlays key set UNCHANGED
``` ```
## Live verification ## Live verification
After deploy / compose rebuild of `news-summarizer` on the Pi: End-to-end (real crawl → DB → API) is verified after deploy on the Pi: check
`docker compose --profile ingest logs -f news-scraper news-summarizer`, then
1. Keys UI: save `NOUS_API_KEY` → status `****last4`. `curl -s localhost:8000/api/news | head`. Summaries additionally require
2. Settings: pick a model → Save → `GET /api/settings` echoes it. `GEMINI_API_KEY` to be set in `.env` on the Pi.
3. `docker compose --profile ingest logs -f news-summarizer` — next run (or
`NEWS_SUMMARIZE_RUN_ON_START=1` recreate) logs `Processing N articles with <model>`.
4. `curl -s localhost:8000/api/news/summaries?limit=1` — English `summary_text`, `model` set.
5. `curl -s localhost:8000/api/news/ticker` — flagged headlines only.
6. `curl -s localhost:8000/api/news/map` — only rows with lat/lon.
7. HUD: NEWS ticker scrolls flagged items; map overlay pins popup with location.
8. Unset key + empty keystore → summarizer logs idle, APIs return `[]`, no crash.
**Operator action after merge:** paste a Nous Portal API key in API Keys; pick
a model in Settings if the default `Hermes-4.3-36B` is not wanted; rebuild
`osint-news-summarizer` on the Pi (`pi-app-deploy` / compose).

View file

@ -1,347 +0,0 @@
# Free satellite feeds for the OSINT map
Builder inventory (research profile). Probed **2026-08-29** from this machine. Do **not** treat search snippets as live — every row below had a `curl`/GET (tile, GetCapabilities, STAC, or GetMap). 404 tile rows are omitted unless Capabilities/DescribeDomains still prove the layer exists (sparse fire overlays 404 on empty tiles).
**Pi rules:** browser `L.tileLayer` when CORS `*`; do not proxy multi-GB COGs through the Pi; STAC+SAS like existing Sentinel-1 is “backend same as S-1”; no Redis; home uplink is small.
GIBS Web Mercator REST template (no key):[2]
```
https://gibs.earthdata.nasa.gov/wmts/epsg3857/best/{layer}/default/{time}/{TileMatrixSet}/{z}/{y}/{x}.{jpg|png}
```
Omit `{time}` for static layers. Sub-daily GOES/Himawari accept `YYYY-MM-DD` **or** `YYYY-MM-DDTHH:MI:SSZ` (GIBS snaps to nearest).[2] Attribution: NASA asks clients to acknowledge GIBS/ESDIS.[1]
Live GetCapabilities `epsg3857/best` on 2026-08-29: **1315** `Layer` entries, **all** with a `GoogleMapsCompatible_LevelN` matrix, `access-control-allow-origin: *`.[4] GIBS documents **1000+** visualizations; many LANCE layers appear within **3.5 hours** of observation.[3]
Worldview is the interactive catalog of the same tiles.[5] GIBS developer portal: Earthdata GIBS API page (HTTP 403 from this host at probe time; docs site above is the working copy).[27]
---
## Already in the product (do not rediscover)
| id | status |
|---|---|
| `BlueMarble_ShadedRelief_Bathymetry` | GIBS basemap (`app/gibs_map.py`) |
| `VIIRS_SNPP_CorrectedReflectance_TrueColor` | GIBS basemap |
| `MODIS_Terra_CorrectedReflectance_TrueColor` | GIBS basemap |
| `MODIS_Aqua_CorrectedReflectance_TrueColor` | GIBS basemap |
| `VIIRS_SNPP_DayNightBand_ENCC` | GIBS night lights |
| FIRMS VIIRS hotspot CSV | ingest + `FIRMS_MAP_KEY` |
| `VIIRS_SNPP_Thermal_Anomalies_375m_All` | overlay in `app/live_layers.py` (`gibs_thermal`). **Caps now say TMS `GoogleMapsCompatible_Level8`**, not Level9 — the wired URL uses Level9 (will 400). |
| Sentinel-1 GRD | Planetary Computer STAC + SAS + TiTiler `GET /api/map/sentinel1` |
| IEM NEXRAD / RainViewer | weather radar, not satellite |
Repo docs already flag **Suomi NPP product stop 2026-11-01** — swap SNPP true color / DNB / thermal / FIRMS `VIIRS_SNPP_NRT` to NOAA-20/21 before then.
---
## Ranked “add tomorrow” (sections 12)
Most new OSINT signal per **zero dollars**, browser tiles only:
1. **VIIRS NOAA-20 + NOAA-21 true color** — SNPP replacement, same dropdown pattern.
2. **VIIRS false-color SWIR** (`BandsM11-I2-I1`, `BandsM3-I3-M11`, MODIS 7-2-1) — burn scars, flood, bare soil.
3. **GIBS GOES-East/West GeoColor + Band13 IR** — 10-minute weather-sat, Hormuz + CONUS.
4. **GIBS Himawari AHI vis + IR** — same for IO/WestPac.
5. **HLS S30/L30** — 30 m Landsat/Sentinel-2 look without TiTiler.
6. **OPERA RTC Sentinel-1 + DIST-ALERT + DSWx** — SAR / disturbance / flood as GIBS tiles (not COGs).
7. **NOAA-20/21 DNB** — night lights after SNPP.
8. **IEM GOES XYZ** — “latest” tiles, no time in the URL, already CORS `*` like NEXRAD.[6]
9. **EUMETView WMS** — Meteosat/MTG for EuropeAfricaIO, CORS `*`.[18]
10. **GFW GLAD-S2 / integrated deforestation alerts** — raster tiles, CORS `*` when `Origin` is sent.[15]
11. **MUR SST + VIIRS/PACE/OLCI chlorophyll** — ocean.
12. **MODIS NDVI 8-day + IMERG rain** — veg / flood context.
13. **SRTM / ASTER GDEM color index** — satellite-derived DEM, static.
14. **NOAA-20/21 thermal anomalies** — FIRMS-shaped overlay after SNPP; empty tiles 404.
---
## 1. Drop-in GIBS WMTS
All rows: **key? no**. **CORS `*`**. **Pi fit: browser `L.tileLayer`**. Same time-domain helper as `gibs_map.py` (`…/1.0.0/{id}/default/{tms}/all/all.xml`).
Format of URL column: layer id + TMS + ext. Date used in probes: `2026-08-27` unless noted.
### 1.1 Optical (true / false / SWIR)
| id | what you see | tile pattern | cadence | max zoom | license | already have? | probe |
|---|---|---|---|---|---|---|---|
| `VIIRS_NOAA20_CorrectedReflectance_TrueColor` | Daily true color, JPSS-1 | `…/{id}/default/{time}/GoogleMapsCompatible_Level9/{z}/{y}/{x}.jpg` | daily | 9 (~250 m) | NASA GIBS ack[1] | **no** (SNPP only) | 200 `*` jpeg |
| `VIIRS_NOAA21_CorrectedReflectance_TrueColor` | Daily true color, JPSS-2 | same Level9 jpg | daily | 9 | NASA GIBS[1] | no | 200 |
| `VIIRS_SNPP_CorrectedReflectance_BandsM11-I2-I1` | False color SWIR (burns, flood) | Level9 jpg | daily | 9 | NASA GIBS[1] | no | 200 |
| `VIIRS_SNPP_CorrectedReflectance_BandsM3-I3-M11` | False color (snow/ice/desert) | Level9 jpg | daily | 9 | NASA GIBS[1] | no | 200 |
| `VIIRS_NOAA20_CorrectedReflectance_BandsM11-I2-I1` | NOAA-20 SWIR false color | Level9 jpg | daily | 9 | NASA GIBS[1] | no | 200 |
| `VIIRS_NOAA21_CorrectedReflectance_BandsM11-I2-I1` | NOAA-21 SWIR false color | Level9 jpg | daily | 9 | NASA GIBS[1] | no | 200 |
| `MODIS_Terra_CorrectedReflectance_Bands721` | Classic 7-2-1 burn/SWIR | Level9 jpg | daily | 9 | NASA GIBS[1] | no | 200 |
| `MODIS_Terra_CorrectedReflectance_Bands367` | 3-6-7 false color | Level9 jpg | daily | 9 | NASA GIBS[1] | no | 200 |
| `MODIS_Aqua_CorrectedReflectance_Bands721` | Aqua 7-2-1 | Level9 jpg | daily | 9 | NASA GIBS[1] | no | 200 |
| `HLS_S30_Nadir_BRDF_Adjusted_Reflectance` | Harmonized Sentinel-2 30 m RGB | Level12 png | ~23 d when a granule exists | 12 (~30 m) | NASA GIBS[1] | no | 200 at z=5 NC; 404 on empty granules is normal. Domain from 2015present |
| `HLS_L30_Nadir_BRDF_Adjusted_Reflectance` | Harmonized Landsat 30 m | Level12 png | 816 d | 12 | NASA GIBS[1] | no | in caps; tile 404 on empty scene |
| `Landsat_WELD_CorrectedReflectance_TrueColor_Global_Monthly` | Landsat WELD monthly mosaic | Level12 jpg | monthly, **not NRT** | 12 | NASA GIBS[1] | no | 200 |
| `Landsat_WELD_CorrectedReflectance_TrueColor_Global_Annual` | WELD annual | Level12 jpg | yearly | 12 | NASA GIBS[1] | no | 200 |
### 1.2 Weather satellites (imagery, not NEXRAD)
Sub-daily. Probe with `2026-08-27` **and** `2026-08-27T18:00:00Z` both 200 (nearestValue).[2]
| id | what you see | TMS / ext | cadence | max zoom | already have? | probe |
|---|---|---|---|---|---|---|
| `GOES-East_ABI_GeoColor` | GeoColor full disk (Americas, Atlantic, Hormuz west edge) | Level7 png | ~10 min | 7 | no | 200 |
| `GOES-West_ABI_GeoColor` | GeoColor Pacific / CONUS west | Level7 png | ~10 min | 7 | no | 200 (also 200 over CA with ISO time) |
| `GOES-East_ABI_Band2_Red_Visible_1km` | ABI vis | Level7 png | ~10 min | 7 | no | 200 |
| `GOES-West_ABI_Band2_Red_Visible_1km` | ABI vis | Level7 png | ~10 min | 7 | no | 200 |
| `GOES-East_ABI_Band13_Clean_Infrared` | Clean IR window | Level6 png | ~10 min | 6 | no | 200 |
| `GOES-West_ABI_Band13_Clean_Infrared` | Clean IR | Level6 png | ~10 min | 6 | no | 200 |
| `GOES-East_ABI_FireTemp` | Fire temperature RGB | Level7 png | ~10 min | 7 | no | 200 |
| `GOES-West_ABI_FireTemp` | Fire temperature RGB | Level7 png | ~10 min | 7 | no | 404 on NC tile (wrong disk); use west longitudes |
| `GOES-East_ABI_Dust` | Dust RGB | Level7 png | ~10 min | 7 | no | 200 |
| `GOES-East_ABI_Air_Mass` | Air mass RGB | Level6 png | ~10 min | 6 | no | 200 |
| `GOES-West_ABI_Air_Mass` | Air mass RGB | Level6 png | ~10 min | 6 | no | 200 |
| `Himawari_AHI_Band3_Red_Visible_1km` | Himawari vis (IO / WestPac / Aus) | Level7 png | ~10 min | 7 | no | 200 (ISO time over Japan) |
| `Himawari_AHI_Band13_Clean_Infrared` | Himawari IR | Level6 png | ~10 min | 6 | no | 200 |
| `Himawari_AHI_Air_Mass` | Himawari air mass | Level6 png | ~10 min | 6 | no | 200 |
**Meteosat is not in GIBS.** Use section 2 EUMETView.
### 1.3 SAR / flood / disturbance (GIBS tiles — skip TiTiler)
| id | what you see | TMS | cadence | max zoom | already have? | probe |
|---|---|---|---|---|---|---|
| `OPERA_L2_Radiometric_Terrain_Corrected_SAR_Sentinel-1` | S-1 RTC browse (better than GRD for terrain) | Level12 png | scene-based from 2025-01 | 12 | **no** (you have GRD COGs, not RTC tiles) | 200 at z=5; domain 2025-01-10/… |
| `OPERA_L3_DIST-ALERT-HLS_Color_Index` | Vegetation disturbance / clearing alert | Level12 png | ~23 d | 12 | no | 200 |
| `OPERA_L3_DIST-ANN-HLS_Color_Index` | Annual DIST | Level12 png | yearly | 12 | no | in caps |
| `OPERA_L3_Dynamic_Surface_Water_Extent-HLS` | Surface water / flood (HLS, 30 m) | Level12 png | ~23 d | 12 | no | 200 at z=5 |
| `OPERA_L3_Dynamic_Surface_Water_Extent-Sentinel-1` | Surface water from S-1 (clouds irrelevant) | Level12 png | S-1 revisit | 12 | no | 200 at z=5 |
| `NISAR_L2_Geocoded_Polarimetric_Covariance` | NISAR early browse | Level13 png | when downlinked | 13 | no | 200 (layer exists; coverage still sparse) |
| `SMAP_L4_Analyzed_Surface_Soil_Moisture` | Soil moisture | Level6 png | daily | 6 | no | 200 |
| `SMAP_L3_Active_Sigma0_VV` | SMAP radar σ0 | Level6 png | 23 d | 6 | no | in caps (SMAP radar died 2015 — historical) |
No ICEYE / Capella / Umbra / ALOS PALSAR **daily** layers in this GIBS 3857 dump.[4] ALOS shows up as mosaics on Planetary Computer (section 3).
### 1.4 Thermal / fire / volcano
Sparse PNG overlays: **empty tiles 404**. Capabilities + DescribeDomains still 200. Frontend must tolerate 404 (Leaflet does).
| id | what you see | TMS | cadence | max zoom | already have? | probe |
|---|---|---|---|---|---|---|
| `VIIRS_SNPP_Thermal_Anomalies_375m_All` | 375 m hotspots | **Level8** png (not 9) | daily | 8 | **yes**, but wired as Level9 | Domain 200; many tiles 404 |
| `VIIRS_NOAA20_Thermal_Anomalies_375m_All` | NOAA-20 hotspots | Level8 png | daily | 8 | no | Domain 200 (`2020-01-01/…` through at least 2025-09); tiles 404 if no fire in tile |
| `VIIRS_NOAA21_Thermal_Anomalies_375m_All` | NOAA-21 hotspots | Level8 png | daily | 8 | no | same |
| `VIIRS_*_Thermal_Anomalies_375m_{Day,Night}` | day/night split | Level8 png | daily | 8 | no | in caps |
| `MODIS_{Terra,Aqua,Combined}_Thermal_Anomalies_All` | 1 km MODIS fire | Level7 png | daily | 7 | no | in caps |
| `GOES-East_ABI_FireTemp` | geostationary fire RGB | Level7 png | ~10 min | 7 | no | 200 |
Also keep FIRMS CSV — points beat raster for click/query.
### 1.5 Night lights (beyond current DNB ENCC)
| id | what you see | TMS / ext | cadence | max zoom | already have? | probe |
|---|---|---|---|---|---|---|
| `VIIRS_NOAA20_DayNightBand` | NOAA-20 DNB | Level7 png | daily | 7 | no | 200 |
| `VIIRS_NOAA21_DayNightBand` | NOAA-21 DNB | Level7 png | daily | 7 | no | 200 |
| `VIIRS_NOAA20_DayNightBand_At_Sensor_Radiance` | radiance, not ENCC | Level8 png | daily | 8 | no | 200 |
| `VIIRS_SNPP_DayNightBand_At_Sensor_Radiance` | SNPP radiance | Level8 png | daily | 8 | no | 200 |
| `VIIRS_NOAA20_DayNightBand_AtSensor_M15` | DNB+M15 composite jpg | Level8 jpg | daily | 8 | no | 200 |
| `VIIRS_Night_Lights` | Black-marble-style annual-ish | Level8 png | time-dim | 8 | no | 200 on 2026-08-27 mosaic date |
| `VIIRS_CityLights_2012` | Static 2012 city lights | Level8 jpg | **static** (`has_time=false`) | 8 | no | 200 |
`VIIRS_Black_Marble` and `VIIRS_NOAA20_DayNightBand_ENCC` are in caps; ENCC-NOAA20 returned HTTP 400 on the Level8 template we tried — do not ship until DescribeDomains + a known-good date are wired. SNPP ENCC stays as the current layer.
### 1.6 Ocean
| id | what you see | TMS | cadence | max zoom | probe |
|---|---|---|---|---|---|
| `GHRSST_L4_MUR_Sea_Surface_Temperature` | 1 km MUR SST | Level7 png | daily | 7 | 200 |
| `GHRSST_L4_MUR_Sea_Surface_Temperature_Anomalies` | SST anomaly | Level7 png | daily | 7 | in caps |
| `MODIS_Aqua_L3_SST_MidIR_4km_Night_Daily` | MODIS SST | Level6 png | daily | 6 | 200 |
| `MODIS_Aqua_L2_Chlorophyll_A` | Aqua chl-a | Level7 png | daily | 7 | 200 |
| `VIIRS_SNPP_L2_Chlorophyll_A` | VIIRS chl-a | Level7 png | daily | 7 | 200 |
| `VIIRS_NOAA20_Chlorophyll_a` | NOAA-20 chl-a | Level7 png | daily | 7 | 200 |
| `OCI_PACE_Chlorophyll_a` | PACE OCI chl-a | Level7 png | daily | 7 | 200 |
| `S3A_OLCI_Chlorophyll_a` | Sentinel-3A OLCI chl-a | Level7 png | daily | 7 | 200 |
| `S3B_OLCI_Chlorophyll_a` | Sentinel-3B OLCI | Level7 png | daily | 7 | in caps |
| `MODIS_Terra_Sea_Ice` | sea ice | Level7 png | daily | 7 | 200 |
| `GHRSST_L4_MUR_Sea_Ice_Concentration` | MUR ice | Level7 png | daily | 7 | in caps |
No dedicated “SAR oil slick” GIBS layer in the 3857 dump. Closest: OPERA RTC / DSWx-S1 + existing S-1 GRD TiTiler.
### 1.7 Vegetation / burn / flood / precip / atm
| id | what you see | TMS | cadence | max zoom | probe |
|---|---|---|---|---|---|
| `MODIS_Terra_NDVI_8Day` | NDVI | Level9 png | 8-day | 9 | 200 |
| `MODIS_Terra_L3_NDVI_16Day` | NDVI 16-day | Level9 png | 16-day | 9 | 200 |
| `IMERG_Precipitation_Rate` | GPM IMERG rain | Level6 png | sub-daily | 6 | 200 |
| `MODIS_Terra_Aerosol` | AOD | Level6 png | daily | 6 | 200 |
| `MODIS_Terra_Land_Surface_Temp_Day` | LST | Level7 png | daily | 7 | 200 |
| `VIIRS_SNPP_Land_Surface_Temp_Day` | VIIRS LST | Level7 png | daily | 7 | 200 |
| `AIRS_L3_Carbon_Monoxide_500hPa_Volume_Mixing_Ratio_Daily_Night` | CO (fires, industry) | Level6 png | daily | 6 | 200 |
| `OMI_NO2` / `OMI_Aerosol_Index` | NO2 / smoke index | Level6 png | daily | 6 | OMI AI 200; several OMPS 200 |
| `MODIS_Water_Mask` | static water mask | Level9 png | static | 9 | 200 |
MODIS burned-area monthly (`MCD64` / `MODIS_Combined_L3_Burned_Area_Monthly`) is in caps; our dated tile 400d — wire only after a DescribeDomains date hits 200.
### 1.8 DEM (satellite-derived, tileable)
| id | what you see | TMS / ext | cadence | max zoom | probe |
|---|---|---|---|---|---|
| `SRTM_Color_Index` | SRTM elevation color | Level12 png | static | 12 | 200 |
| `ASTER_GDEM_Color_Index` | ASTER GDEM color | Level12 png | static | 12 | 200 |
| `ASTER_GDEM_Color_Shaded_Relief` | ASTER hillshade | Level12 jpg | static | 12 | 200 |
| `ASTER_GDEM_Greyscale_Shaded_Relief` | grey hillshade | Level12 jpg | static | 12 | in caps |
| `GEDI_ISS_L3_Canopy_Height_Mean_RH100_201904-202303` | GEDI canopy height | Level7 png | static epoch | 7 | related GEDI biomass 200 |
Blue Marble shaded relief is **already** the basemap — these are extra.
---
## 2. Other XYZ / WMTS / WMS (no key)
Ranked after GIBS for signal/$; still free.
| id | what you see | provider | URL pattern | key? | CORS | cadence | max zoom / res | license / attribution | Pi fit | already have? | probe 2026-08-29 |
|---|---|---|---|---|---|---|---|---|---|---|---|
| `iem_goes_east_conus_ch02` | GOES-East CONUS ABI ch02 vis, **latest** | Iowa State IEM | `https://mesonet.agron.iastate.edu/cache/tile.py/1.0.0/goes_east_conus_ch02/{z}/{x}/{y}.png` | no | `*` | ~5 min cache header | TMS; vis ~1 km | Cite IEM / NOAA GOES[6][7] | **browser** (same stack as NEXRAD) | no | 200 image/png `*` |
| `iem_goes_east_conus_ch13` | GOES-East CONUS IR ch13 | IEM | `…/goes_east_conus_ch13/{z}/{x}/{y}.png` | no | `*` | ~5 min | IR ~2 km | IEM[6] | browser | no | 200 |
| `iem_goes_east_fulldisk_ch02` | GOES-East full disk vis | IEM | `…/goes_east_fulldisk_ch02/{z}/{x}/{y}.png` | no | `*` | ~5 min | full disk | IEM[6] | browser | no | 200 |
| `iem_goes_west_conus_ch02` | GOES-West CONUS vis | IEM | `…/goes_west_conus_ch02/{z}/{x}/{y}.png` | no | `*` | ~5 min | | IEM[6] | browser | no | 200 |
| `iem_goes_vis_1km` | Legacy name → GOES-East vis | IEM | `…/goes-vis-1km/{z}/{x}/{y}.png` | no | `*` | ~5 min | | IEM[6] | browser | no | 200 |
| IEM GOES template | Any bird/sector/channel | IEM | `goes_{east\|west}_{fulldisk\|conus\|mesoscale-1\|mesoscale-2\|alaska\|puertorico}_ch{0116}`[6] | no | `*` | NRT | 16 ABI bands | IEM[6] | browser | no | template documented; ch02/ch13 probed |
| `eumet_msg_natural` | Meteosat natural color | EUMETSAT EUMETView GeoServer | WMS `https://view.eumetsat.int/geoserver/ows` layer `msg_fes:rgb_natural` EPSG:3857 GetMap | no | `*` | NRT | SEVIRI ~3 km | EUMETSAT viz; cite EUMETSAT[18] | **browser `L.tileLayer.wms`** (not XYZ). Caps 200, 165 layer names | no | GetMap 200 image/png `*` |
| `eumet_msg_ir108` | Meteosat IR 10.8 | EUMETView | WMS `msg_fes:ir108` | no | `*` | NRT | | EUMETSAT | browser WMS | no | name in caps |
| `eumet_msg_fire` | Meteosat fire | EUMETView | WMS `msg_fes:fire` | no | `*` | NRT | | EUMETSAT | browser WMS | no | name in caps |
| `eumet_mtg_ir105` | MTG-I IR | EUMETView | WMS `mtg_fd:ir105_hrfi` | no | `*` | NRT | FCI | EUMETSAT | browser WMS | no | name in caps |
| `eumet_s3_olci_rgb` | S3 OLCI RGB mosaic | EUMETView | WMS `copernicus:daily_sentinel3ab_olci_l1_rgb_fulres` | no | `*` | daily | OLCI | Copernicus/EUMETSAT | browser WMS | no | name in caps |
| `eumet_s3_chl` | S3 chl-a | EUMETView | WMS `copernicus:daily_sentinel3ab_olci_l2_chl_fullres` | no | `*` | daily | | Copernicus | browser WMS | no | name in caps |
| `gfw_glad_s2` | GLAD Sentinel-2 deforestation alerts | GFW tile cache | `https://tiles.globalforestwatch.org/umd_glad_sentinel2_alerts/latest/default/{z}/{x}/{y}.png` | no | `*` **if `Origin` header** (null without it) | ~daily | raster z 022 documented[15] | WRI/UMD; cite GFW | **browser** (Leaflet sends Origin) | no | 200 image/png; with Origin → CORS `*` |
| `gfw_integrated` | Integrated deforestation alerts | GFW | `https://tiles.globalforestwatch.org/gfw_integrated_alerts/latest/default/{z}/{x}/{y}.png` | no | `*` + Origin | ~daily | | GFW | browser | no | 200 |
| `gfw_tcl` | UMD tree-cover loss | GFW | `https://tiles.globalforestwatch.org/umd_tree_cover_loss/latest/tcd_30/{z}/{x}/{y}.png` | no | (same host) | annual | | GFW/UMD | browser | no | 200 |
| `star_goes19_fd_geocolor` | GOES-19 full-disk GeoColor **JPEG** (not XYZ) | NOAA NESDIS STAR CDN | `https://cdn.star.nesdis.noaa.gov/GOES19/ABI/FD/GEOCOLOR/latest.jpg` also `…/CONUS/GEOCOLOR/latest.jpg` | no | `*` | minutes | full-disk / CONUS image | NOAA | **not a map layer** — optional lightbox. Do not tile-proxy | no | 200 jpeg `*`[21] |
| `star_goes18_fd_geocolor` | GOES-18 FD GeoColor JPEG | STAR | `https://cdn.star.nesdis.noaa.gov/GOES18/ABI/FD/GEOCOLOR/latest.jpg` | no | `*` | minutes | | NOAA | lightbox only | no | 200 |
IEM JSON `…/GOES/conus/channel02/GOES-16_C02.json` is **stale** (`generated_at` 2025-04-07) but the **tile names still 200**. Prefer GIBS GeoColor when you need a time slider; prefer IEM when you want “whatever is latest” with zero time plumbing.[6][7]
### 2.x Works but **not** browser-direct (no CORS)
| id | what you see | URL | CORS | Pi fit | probe |
|---|---|---|---|---|---|
| RAMMB/CIRA SLIDER GeoColor tiles | GOES-19 / Himawari / JPSS loops, ~10 min | Times: `https://rammb-slider.cira.colostate.edu/data/json/goes-19/full_disk/geocolor/latest_times.json` (`timestamps_int`). Tile: `https://rammb-slider.cira.colostate.edu/data/imagery/{YYYY}/{MM}/{DD}/goes-19---full_disk/geocolor/{ts}/{zz}/{yyy}_{xxx}.png` e.g. `…/2026/08/28/goes-19---full_disk/geocolor/20260828225021/00/000_000.png`. Himawari times JSON also 200. | **none** | **Do not proxy tiles through the Pi.** Bookmark / deep-link SLIDER instead.[19][26] | times 200; tile 200 png; CORS null |
| NICT Himawari-8 Real-time Web | 10-min full disk PNG grid | `https://himawari8.nict.go.jp/img/D531106/latest.json` then `https://himawari8.nict.go.jp/img/D531106/2d/550/{YYYY}/{MM}/{DD}/{HHMMSS}_{x}_{y}.png` | **none** | same — no Pi proxy | latest.json 200; tile 200 png; CORS null[22] |
| OpenAerialMap | Per-scene TMS of open UAV/sat | `https://api.openaerialmap.org/meta``properties.tms` | CORS **only** `https://map.openaerialmap.org` | not usable from the dashboard origin without a proxy; opportunistic, not a global basemap[23][25] | meta 200 |
| USGS LandsatLook STAC | Landsat C2 STAC | `https://landsatlook.usgs.gov/stac-server` | CORS locked to `https://landsatlook.usgs.gov/stac-server` | backend-only if ever; prefer Earth Search / PC / GIBS HLS[24] | collections + search 200 |
| NOAA CoastWatch ERDDAP WMS (`jplMURSST41`) | MUR SST WMS | `https://coastwatch.pfeg.noaa.gov/erddap/wms/jplMURSST41/request` | mixed | **flaky**: GetCapabilities 200 earlier, **503** on later GetMap/GetCapabilities. Prefer GIBS MUR | 503 on 2nd pass |
| RainViewer `satellite.infrared` | would be IR sat frames | `https://api.rainviewer.com/public/weather-maps.json` | `*` | **empty list** (`"infrared": []`) at probe time — do not ship. Radar path already in product[20] | JSON 200, satellite IR empty |
---
## 3. STAC / COG (TiTiler, same pattern as Sentinel-1)
Do **not** stream COGs through the Pi for a basemap. Viewport bbox + short datetime window + SAS/public HTTPS + existing TiTiler. Prefer GIBS HLS / OPERA tiles (section 1) when a browse PNG is enough.
| id | what you see | provider | STAC | key? | CORS | cadence | res | license | Pi fit | already have? | probe |
|---|---|---|---|---|---|---|---|---|---|---|---|
| `sentinel-2-l2a` (Earth Search) | S2 L2A COGs, public HTTPS | Element 84 / AWS Open Data | `https://earth-search.aws.element84.com/v1` collections: `sentinel-2-l2a`, `sentinel-2-c1-l2a`, `sentinel-2-l1c`, `sentinel-2-pre-c1-l2a`, `sentinel-1-grd`, `landsat-c2-l2`, `naip`, `cop-dem-glo-30`, `cop-dem-glo-90`[8][9] | no | STAC `*` | S2 ~5 d | 10 m | Copernicus open; AWS public bucket HTTPS (not requester-pays for these COGs)[9] | **backend same as S-1**: search → TCI/visual COG → TiTiler. Live item `S2A_40RCP_20260827_0_L2A` href `https://sentinel-cogs.s3.us-west-2.amazonaws.com/…/TCI.tif` | no | collections + search 200 `*` |
| `sentinel-2-l2a` (Planetary Computer) | same S2 on Azure | Microsoft PC | `https://planetarycomputer.microsoft.com/api/stac/v1/collections/sentinel-2-l2a` | SAS token (unsigned search works) | STAC `*` | ~5 d | 10 m | Copernicus; Azure blob needs SAS like current S-1 | backend same as S-1 | no | collection + search 200 `*` (136 collections listed)[10][11] |
| `sentinel-1-rtc` | S-1 IW RTC γ0 COGs | PC / Catalyst | `/collections/sentinel-1-rtc` | **PC account required to retrieve SAS** for RTC blobs[12] | STAC `*` | IW land | ~10 m pixels | **CC BY 4.0**[12] | backend same as S-1 **plus** PC login for SAS. Prefer GIBS OPERA RTC tiles if browse is enough | no | collection 200; search item `S1D_IW_GRDH_…_rtc` assets `vv,vh,tilejson,rendered_preview` |
| `sentinel-1-grd` (PC) | GRD | PC | `/collections/sentinel-1-grd` | SAS | `*` | 612 d | | Copernicus | **already have** | yes | 200 |
| `sentinel-1-grd` (Earth Search) | GRD on AWS | E84 | `/collections/sentinel-1-grd` | requester-pays **s3://** URLs per E84 README[9] | `*` | | | Copernicus | worse than PC for the Pi (AWS creds) | no | collection 200 |
| `landsat-c2-l2` | Landsat 8/9 SR | E84 + PC | both catalogs | no / SAS | `*` | 816 d | 30 m | USGS public | backend TiTiler; or just use GIBS HLS | no | both 200 |
| `hls2-s30` / `hls2-l30` | HLS v2 COGs | PC | `/collections/hls2-s30`, `hls2-l30` | SAS | `*` | 23 d | 30 m | NASA | prefer GIBS HLS tiles | no | collections 200 |
| `goes-cmi` | GOES Cloud & Moisture Imagery COGs | PC | `/collections/goes-cmi` | SAS | `*` | 510 min | ABI | NOAA | **overkill vs GIBS/IEM tiles** | no | collection 200 |
| `modis-14A1-061` / `modis-64A1-061` | MODIS fire / burned area | PC | `/collections/modis-14A1-061`, `modis-64A1-061` | SAS | `*` | daily / monthly | 1 km / 500 m | NASA | prefer GIBS fire tiles + FIRMS | no | 200 |
| `alos-palsar-mosaic` / `alos-fnf-mosaic` | ALOS PALSAR yearly mosaic / forest-nonforest | PC | those collection ids | SAS | `*` | **annual** | 25 m | JAXA (check collection) | backend mosaic, not live SAR | no | 200 |
| `nasadem` / `cop-dem-glo-30` | DEM COGs | PC + E84 | `nasadem`, `cop-dem-glo-30` | public / SAS | `*` | static | 30 m | NASA / Copernicus | prefer GIBS SRTM/ASTER tiles | no | 200 |
| `naip` | USDA NAIP aerial (CONUS) | E84 + PC | `naip` | no | `*` | leaf-on, not NRT | ~0.6 m | USDA | CONUS only; huge. Optional TiTiler | no | 200 |
| `io-lulc-annual-v02` | 10 m land cover | PC | `io-lulc-annual-v02` | SAS | `*` | annual | 10 m | various | overlay, not sat photo | no | 200 |
| CDSE `sentinel-2-l2a` / `sentinel-1-grd` | Copernicus Dataspace STAC | ESA CDSE | `https://stac.dataspace.copernicus.eu/v1/collections/sentinel-2-l2a` (lowercase ids work; `SENTINEL-2` 404) | **free account** for many assets | **CORS none** | same as ESA | | Copernicus | backend only; Earth Search/PC easier on a Pi | no | collection 200, CORS null. List endpoint is paginated (first page was CLMS burned-area COGs)[14] |
PC catalog also has Sentinel-3 OLCI/SLSTR NetCDF, Sentinel-5P, GOES-GLM — NetCDF is a bad TiTiler citizen; use GIBS/EUMETView for those.
---
## 4. Free-account / license-gated (no card this week)
| id | note | why not a dropdown tomorrow |
|---|---|---|
| Microsoft PC SAS for RTC (and some blobs) | “A Planetary Computer account is required to retrieve SAS tokens to read the RTC data.”[12] | Search is open; **read** needs an account. GRD path you already have may not need this. |
| Copernicus Data Space (`stac.dataspace.copernicus.eu`) | STAC search 200 without cookie; **no CORS**; downloads often need a free CDSE login | Use Earth Search/PC unless you want official ESA provenance |
| JAXA P-Tree / Himawari Monitor | Himawari standard data, account | NICT/GIBS already cover browse |
| EUMETSAT Data Store | full MTG/MSG granules | EUMETView WMS is the browse path |
| FIRMS map key | already in product | add `VIIRS_NOAA20_NRT` / `VIIRS_NOAA21_NRT` before SNPP sunset |
| USGS ERS / EarthExplorer | Landsat/ASTER download login | GIBS HLS + Earth Search cover browse |
| Planet Tropical Forest Observatory | paid successor after NICFI | see skip |
---
## 5. Skip / costs money / dead
| id | why |
|---|---|
| **NICFI / Planet tropical mosaics (free)** | Free NICFI phase **ended 1 Apr 2025**. Removed from GFW and Collect Earth Online. Successor is Planet **Tropical Forest Observatory (subscription)** or a future NICFI re-compete.[16][17] PC collections `planet-nicfi-analytic` / `planet-nicfi-visual` still exist but assets are **RFP winners only** + proprietary PLA.[13] |
| Sentinel Hub (paid tiers) | billed processing units |
| Google Earth Engine | billing project |
| Maxar / Planet commercial | $ |
| ICEYE commercial | no free global tile/STAC found this pass |
| Umbra / Capella / Maxar **open data** STAC | catalogs 200 (`maxar-opendata`, `umbra-open-data-catalog`) but **disaster events only**, not a standing layer |
| Esri World Imagery / Clarity | tiles 200 CORS `*`**ToS not a free basemap we should wrap** |
| Mapbox / Google satellite | key + ToS |
| GEE Dynamic World / NICFI in EE | EE billing |
| `nowcoast.noaa.gov` | HTTP **403** |
| FIRMS WMS (`firms.modaps.eosdis.nasa.gov/wms/…`) | HTTP **404** — use CSV + GIBS |
| RainViewer satellite IR | payload empty[20] |
| CoastWatch ERDDAP | 503 at probe; GIBS MUR replaces SST |
| Proxying RAMMB or NICT tiles | no CORS; would soak the home uplink |
---
## Implementation notes for builders
1. **GIBS dropdown:** reuse `MAP_LAYERS` in `app/gibs_map.py`. New rows are `{id, title, tms, format, has_time, max_zoom}`. Time-domain fetch already exists.
2. **SNPP sunset:** NOAA-20/21 true color, DNB, thermal, FIRMS datasets first. SNPP true color can stay as fallback until 2026-11-01.
3. **Fix thermal TMS:** caps say `GoogleMapsCompatible_Level8` for `VIIRS_*_Thermal_Anomalies_375m_*`. Level9 GetTile is HTTP 400 XML.
4. **GOES time:** either GIBS `{time}` ISO + existing date slider, or IEM “latest” XYZ with no time (simpler, CONUS/FD only).
5. **HLS / OPERA:** empty granules 404 — same as “todays MODIS isnt ingested yet”. Clamp latest date via DescribeDomains like current daily mosaics.
6. **Do not add TiTiler S2 as a global basemap.** 10 m COGs will thrash the Pi. GIBS HLS Level12 is the browse path; Earth Search TCI is a “inspect this viewport” action like S-1.
7. **EUMETView:** `L.tileLayer.wms` against `https://view.eumetsat.int/geoserver/ows`, layers `msg_fes:rgb_natural` / `msg_fes:ir108`. Caps CORS `*`.
8. **GFW:** send browser Origin (Leaflet does). No key.
9. **Attribution strings:** NASA GIBS acknowledgment[1]; IEM; EUMETSAT; GFW/UMD; NOAA STAR.
### Probe stats (this run)
- GIBS WMTS caps: 5796177 bytes, CORS `*`, 1315 layers.[4]
- Curated GIBS GetTile: **78/90 HTTP 200** first batch; extra GOES/DNB/HLS/OPERA/ocean 200 as tabulated.
- Earth Search collections (complete list): `sentinel-2-pre-c1-l2a`, `cop-dem-glo-30`, `naip`, `cop-dem-glo-90`, `landsat-c2-l2`, `sentinel-2-l2a`, `sentinel-2-l1c`, `sentinel-2-c1-l2a`, `sentinel-1-grd`.[8]
- Planetary Computer: 136 collections; S-1 RTC CC-BY-4.0.[11][12]
Raw probe JSON lives next to this file in the kanban workspace (`gibs_probes.json`, `wave2_probes.json`, `wave3_probes.json`, `gibs_all_layers.json`).
## Sources
[1] https://nasa-gibs.github.io/gibs-api-docs
[2] https://nasa-gibs.github.io/gibs-api-docs/access-basics
[3] https://nasa-gibs.github.io/gibs-api-docs/available-visualizations
[4] https://gibs.earthdata.nasa.gov/wmts/epsg3857/best/1.0.0/WMTSCapabilities.xml
[5] https://worldview.earthdata.nasa.gov
[6] https://mesonet.agron.iastate.edu/ogc
[7] https://mesonet.agron.iastate.edu/GIS/goes.phtml
[8] https://earth-search.aws.element84.com/v1/collections
[9] https://github.com/Element84/earth-search
[10] https://planetarycomputer.microsoft.com/catalog
[11] https://planetarycomputer.microsoft.com/api/stac/v1/collections
[12] https://planetarycomputer.microsoft.com/dataset/sentinel-1-rtc
[13] https://planetarycomputer.microsoft.com/dataset/planet-nicfi-analytic
[14] https://stac.dataspace.copernicus.eu/v1/collections
[15] https://tiles.globalforestwatch.org
[16] https://www.collect.earth/planet-imagery-via-nicfi-is-no-longer-available-on-ceo
[17] https://www.globalforestwatch.org/blog/data-and-tools/planet-imagery-changes-gfw
[18] https://view.eumetsat.int/geoserver/ows?service=WMS&request=GetCapabilities
[19] https://rammb-slider.cira.colostate.edu
[20] https://api.rainviewer.com/public/weather-maps.json
[21] https://cdn.star.nesdis.noaa.gov/GOES19/ABI/FD/GEOCOLOR/latest.jpg
[22] https://himawari8.nict.go.jp
[23] https://api.openaerialmap.org/meta?limit=1
[24] https://landsatlook.usgs.gov/stac-server/collections
[25] https://openaerialmap.org
[26] https://bellingcat.gitbook.io/toolkit/more/all-tools/rammb-slider
[27] https://www.earthdata.nasa.gov/engage/open-data-services-software/earthdata-developer-portal/gibs-api

View file

@ -1,32 +0,0 @@
"""Feed helpers shared by the news spider (no Scrapy import)."""
from __future__ import annotations
import datetime
from email.utils import parsedate_to_datetime
from urllib.parse import urlparse
AUDIO_EXT = (".mp3", ".m4a", ".ogg", ".wav", ".aac", ".flac", ".opus")
def is_audio_url(url: str) -> bool:
path = urlparse(url or "").path.lower()
return any(path.endswith(ext) for ext in AUDIO_EXT)
def article_timestamp(pub_date: str | None) -> datetime.datetime:
"""Prefer the feed's pubDate/published; fall back to now (UTC)."""
if pub_date:
try:
return parsedate_to_datetime(pub_date).astimezone(datetime.timezone.utc)
except (TypeError, ValueError, IndexError):
pass
try:
raw = pub_date.replace("Z", "+00:00")
ts = datetime.datetime.fromisoformat(raw)
if ts.tzinfo is None:
ts = ts.replace(tzinfo=datetime.timezone.utc)
return ts
except ValueError:
pass
return datetime.datetime.now(datetime.timezone.utc)

View file

@ -3,7 +3,7 @@
# Don't forget to add your pipeline to the ITEM_PIPELINES setting # Don't forget to add your pipeline to the ITEM_PIPELINES setting
# See: https://docs.scrapy.org/en/latest/topics/item-pipeline.html # See: https://docs.scrapy.org/en/latest/topics/item-pipeline.html
import logging import logging
import psycopg2 import psycopg2
import os import os
from scrapy.exceptions import DropItem from scrapy.exceptions import DropItem
@ -53,10 +53,8 @@ class PostgresPipeline:
self.connection.commit() self.connection.commit()
def process_item(self, item, spider): def process_item(self, item, spider):
url = item['url'] if item ['url'] in self.seen_urls:
if url in self.seen_urls: raise DropItem()
raise DropItem(f"Duplicate URL (in-memory): {url}")
self.seen_urls.add(url)
try: try:
self.cur.execute(""" self.cur.execute("""
INSERT INTO articles (title, url, content, domain, timestamp) INSERT INTO articles (title, url, content, domain, timestamp)
@ -70,16 +68,15 @@ class PostgresPipeline:
item['timestamp'] item['timestamp']
)) ))
if self.cur.rowcount == 0: if self.cur.rowcount == 0:
raise DropItem(f"Duplicate URL (database): {url}") e = DropItem("Duplicate URL (database conflict)")
e.log_level = logging.DEBUG
raise e
self.connection.commit() self.connection.commit()
return item return item
except DropItem:
self.connection.rollback()
raise
except Exception as e: except Exception as e:
spider.logger.error(f"Error saving to Postgres: {e}") spider.logger.error(f"Error saving to Postgres: {e}")
self.connection.rollback() self.connection.rollback()
raise raise
def close_spider(self, spider): def close_spider(self, spider):
self.cur.close() self.cur.close()
@ -91,3 +88,5 @@ from itemadapter import ItemAdapter
class NewsscraperPipeline: class NewsscraperPipeline:
def process_item(self, item, spider): def process_item(self, item, spider):
return item return item

View file

@ -4,13 +4,11 @@ from urllib.parse import urljoin, urlparse
import datetime import datetime
import re import re
from newsScraper.feed_util import article_timestamp, is_audio_url
class NewsRSSSpider(Spider): class NewsRSSSpider(Spider):
"""Crawl the curated news sources in urls.txt and extract articles. """Crawl the curated news sources in urls.txt and extract articles.
urls.txt contains curated news HOMEPAGES and RSS/Atom feeds, so this spider urls.txt contains 257 news HOMEPAGES (not feed URLs), so this spider
implements feed autodiscovery: it fetches each start URL, finds the implements feed autodiscovery: it fetches each start URL, finds the
RSS/Atom feed link (`<link rel="alternate" type="application/rss+xml">` RSS/Atom feed link (`<link rel="alternate" type="application/rss+xml">`
or a visible /rss|/feed link), follows it, and then follows each feed or a visible /rss|/feed link), follows it, and then follows each feed
@ -89,8 +87,6 @@ class NewsRSSSpider(Spider):
or node.xpath('updated/text()').get() or node.xpath('updated/text()').get()
) )
if link: if link:
if is_audio_url(link):
continue
yield scrapy.Request( yield scrapy.Request(
link, link,
callback=self.parse_article, callback=self.parse_article,
@ -116,5 +112,5 @@ class NewsRSSSpider(Spider):
'url': response.url, 'url': response.url,
'text': pure_text, 'text': pure_text,
'domain': urlparse(response.url).netloc, 'domain': urlparse(response.url).netloc,
'timestamp': article_timestamp(response.meta.get('date')).isoformat() 'timestamp': datetime.datetime.now().isoformat()
} }

View file

@ -1,12 +1,18 @@
#!/usr/bin/env python3 #!/usr/bin/env python3
"""Scheduler loop for the news scraper — crawl continuously. """Scheduler loop for the news scraper — hourly scrape at minute :00.
As soon as one Scrapy pass finishes, wait NEWS_SCRAPE_INTERVAL_S seconds Replaces the k8s CronJob (`0 * * * *`) with an in-compose loop so the whole
and start the next. Two crawls never overlap (the loop is serial). news pipeline lives inside docker-compose. Each iteration:
1. (optionally, on first boot) runs the Scrapy crawl once to seed data fast
2. sleeps until the next :NEWS_SCRAPE_MINUTE wall-clock boundary
Because the loop is serial, a crawl that overruns its hour simply delays the
next run to the following boundary two crawls never overlap.
Env (all optional, 12-factor): Env (all optional, 12-factor):
NEWS_SCRAPE_INTERVAL_S seconds between crawls (default 10) NEWS_SCRAPE_MINUTE minute of the hour to fire (default 0)
NEWS_SCRAPE_RUN_ON_START "1" to crawl immediately on boot (default 1) NEWS_SCRAPE_RUN_ON_START "1" to crawl once immediately on boot (default 1)
""" """
from __future__ import annotations from __future__ import annotations
@ -21,12 +27,19 @@ import time
logging.basicConfig(level=logging.INFO, format="%(asctime)s %(levelname)s %(name)s: %(message)s") logging.basicConfig(level=logging.INFO, format="%(asctime)s %(levelname)s %(name)s: %(message)s")
logger = logging.getLogger("news.scraper") logger = logging.getLogger("news.scraper")
INTERVAL_S = max(0, int(os.getenv("NEWS_SCRAPE_INTERVAL_S", "10"))) MINUTE = int(os.getenv("NEWS_SCRAPE_MINUTE", "0"))
RUN_ON_START = os.getenv("NEWS_SCRAPE_RUN_ON_START", "1").lower() in ("1", "true", "yes") RUN_ON_START = os.getenv("NEWS_SCRAPE_RUN_ON_START", "1").lower() in ("1", "true", "yes")
CRAWL_CMD = ["scrapy", "crawl", "articles"] CRAWL_CMD = ["scrapy", "crawl", "articles"]
def seconds_until_next(minute: int) -> float:
"""Seconds until the next occurrence of ``minute`` past the hour (local time)."""
now = datetime.datetime.now()
nxt = now.replace(minute=minute, second=0, microsecond=0) + datetime.timedelta(hours=1)
return (nxt - now).total_seconds()
def run_crawl() -> None: def run_crawl() -> None:
logger.info("scrape starting at %s", datetime.datetime.now().isoformat(timespec="seconds")) logger.info("scrape starting at %s", datetime.datetime.now().isoformat(timespec="seconds"))
try: try:
@ -38,14 +51,15 @@ def run_crawl() -> None:
def main() -> None: def main() -> None:
logger.info( logger.info(
"news scraper loop starting (interval_s=%s, run_on_start=%s)", "news scraper loop starting (minute=%s, run_on_start=%s)",
INTERVAL_S, RUN_ON_START, MINUTE, RUN_ON_START,
) )
if RUN_ON_START: if RUN_ON_START:
run_crawl() run_crawl()
while True: while True:
logger.info("next scrape in %ss", INTERVAL_S) delay = seconds_until_next(MINUTE)
time.sleep(INTERVAL_S) logger.info("next scrape at :%02d (in %.0fs)", MINUTE, delay)
time.sleep(delay)
run_crawl() run_crawl()

View file

@ -1,336 +1,257 @@
https://www.investing.com/rss/news.rss # --- NORTH AMERICA ---
https://www.ftchinese.com/rss # USA
https://www.alwatan.com https://www.npr.org
https://albiladpress.com https://www.pbs.org/newshour
https://www.aletihad.ae/ https://www.usatoday.com
https://www.albayan.ae https://www.cbsnews.com
https://www.aljazeera.com/xml/rss/all.xml https://www.nbcnews.com
https://english.alarabiya.net/.mrss/en.xml
https://english.aawsat.com/home/rss # Canada
https://www.newarab.com/rss https://www.cbc.ca/news
https://www.skynewsarabia.com/rss/feeds/rss-1.xml https://www.ctvnews.ca
https://www.thenationalnews.com/arc/outboundfeeds/rss/ https://globalnews.ca
https://www.arabnews.com/rss.xml https://nationalpost.com
https://gulfnews.com/rss https://www.thestar.com
https://www.kuwaittimes.com/feed/
https://www.omanobserver.om/feed/ # Mexico
https://www.khaleejtimes.com/rss/news https://www.eluniversal.com.mx
http://www.akhbar-alkhaleej.com/rss/all https://www.milenio.com
https://today.lorientleyour.com/rss https://www.jornada.com.mx
https://www.annahar.com/english/rss https://www.excelsior.com.mx
https://english.almayadeen.net/rss https://aristeguinoticias.com
https://english.ahram.org.eg/rss/0/Home.aspx
https://www.dailynewsegypt.com/feed/ # --- SOUTH AMERICA ---
http://www.jordantimes.com/rss # Brazil
https://www.alraimedia.com/rss https://g1.globo.com
https://alghad.com/feed/ https://www.uol.com.br
https://nypost.com/feed/ https://agenciabrasil.ebc.com.br
https://gothamist.com/feed/ https://www.metropoles.com
https://www.cityandstateny.com/rss https://www.terra.com.br/noticias
https://feeds.nytimes.com/nyt/rss/HomePage
https://www.thecity.nyc/rss/index.xml # Argentina
https://brooklyneagle.com/feed/ https://www.infobae.com
https://www.reutersagency.com/feed/ https://www.clarin.com
https://newsatme.com/api/v1/rss/ap/world https://www.lanacion.com.ar
https://feeds.bbci.co.uk/news/world/rss.xml https://www.pagina12.com.ar
https://rss.dw.com/rdf/rss-en-all https://www.cronista.com
https://www.france24.com/en/rss
https://www3.nhk.or.jp/rss/news/shakaitokushu.xml # Colombia
https://www.cbc.ca/cctoc/rss/topstories.north https://www.eltiempo.com
https://www.defensenews.com/arc/outboundfeeds/rss/ https://www.elespectador.com
https://therecord.media/feed https://www.semana.com
https://www.cfr.org/rss/newsletters/daily-news-brief https://www.bluradio.com
https://warontherocks.com/feed/ https://www.rcnradio.com
https://www.thecipherbrief.com/feed
https://www.foreignaffairs.com/rss.xml # --- EUROPE ---
https://geopoliticalfutures.com/feed # United Kingdom
# --- TACTICAL CYBER & VULNERABILITIES --- https://www.bbc.com/news
https://www.bleepingcomputer.com/feed/ https://www.theguardian.com/uk
https://www.cisa.gov/cybersecurity-advisory-feeds https://news.sky.com
https://krebsonsecurity.com/feed/ https://www.independent.co.uk
https://thehackernews.com/feeds/posts/default https://metro.co.uk
https://www.darkreading.com/rss.xml
https://www.mandiant.com/resources/blog/rss.xml # France
https://schneier.com/feed/atom/ https://www.france24.com/en
https://www.securityweek.com/feed/ https://www.lefigaro.fr
# --- REGIONAL THREAT LANDSCAPE --- https://www.20minutes.fr
https://www.thenationalnews.com/rss/ https://www.francetvinfo.fr
https://www.scmp.com/rss/91/feed https://www.lemonde.fr
https://www.batimes.com.ar/rss
https://brazilian.report/feed/ # Germany
https://www.khon2.com/feed/ https://www.dw.com/en
https://www.staradvertiser.com/feed/ https://www.tagesschau.de
https://www.westhawaiitoday.com/feed/ https://www.spiegel.de
https://mauinow.com/feed/ https://www.zeit.de
https://www.idahofallsidaho.gov/RSSFeed.aspx?ModID=1&CID=All-newsflash.xml https://www.bild.de
https://www.eastidahonews.com/feed/
https://localnews8.com/feed/ # Spain
https://www.boisestatepublicradio.org/news.rss https://elpais.com
https://www.illinoistimes.com/springfield/Rss.xml https://www.elmundo.es
https://www.thecentersquare.com/search/?f=rss&t=article&l=20&s=start_time&fulltext=showtext&sd=desc&c%5B%5D=Illinois https://www.rtve.es/noticias
https://chicago.suntimes.com/rss/index.xml https://www.20minutos.es
https://wgntv.com/feed/ https://www.elconfidencial.com
http://feeds.indiana.statenews.net/rss/7b3aa09cdd5d5eac
https://fox59.com/feed/ # Italy
https://www.nwitimes.com/search/?f=rss&t=article&c=news/local&l=50&s=start_time&sd=desc https://www.ansa.it
https://www.wishtv.com/feed/ https://www.corriere.it
https://www.kcci.com/topstories-rss https://www.repubblica.it
https://www.myiowainfo.com/feed/ https://www.lastampa.it
https://feeds.feedburner.com/radioiowanews https://tg24.sky.it
https://www.mississippivalleypublishing.com/search/?f=rss&t=article&c=the_hawk_eye&l=50&s=start_time&sd=desc
https://www.ksn.com/feed/ # Russia (State & Independent mix)
https://www.ksnt.com/feed/ https://tass.com
https://www.hdnews.net/feed/ https://www.interfax.ru
https://themercury.com/search/?f=rss&t=article&c=news&l=50&s=start_time&sd=desc https://www.rt.com
https://www.wdrb.com/search/?f=rss&t=article&c=news&l=50&s=start_time&sd=desc https://www.themoscowtimes.com
https://www.wtvq.com/feed/ https://meduza.io/en
https://www.wnky.com/feed/
https://www.wlky.com/topstories-rss # --- ASIA ---
https://thehayride.com/feed/ # China
https://wgno.com/feed/ https://www.xinhuanet.com/english
https://feeds.feedburner.com/wbrz/news https://www.chinadaily.com.cn
https://thelensnola.org/feed/ https://www.globaltimes.cn
https://www.pressherald.com/news/feed/ https://www.cgtn.com
https://www.centralmaine.com/feed/ https://www.scmp.com
https://www.bangordailynews.com/feed/
https://www.sunjournal.com/news/feed/ # India
https://www.wbaltv.com/topstories-rss https://www.ndtv.com
https://www.manisteenews.com/news/feed/Latest-News-Feed-2564.php https://timesofindia.indiatimes.com
https://www.theoaklandpress.com/feed/ https://indianexpress.com
https://www.macombdaily.com/feed/ https://www.thehindu.com
https://www.startribune.com/local/index.rss2 https://www.hindustantimes.com
https://www.wctrib.com/index.rss
https://www.austindailyherald.com/feed/ # Japan
https://helenair.com/search/?f=rss&t=article&l=50&s=start_time&sd=desc https://www3.nhk.or.jp/nhkworld
https://www.ktvq.com/news.rss https://www.japantimes.co.jp
https://mtstandard.com/search/?f=rss&t=article&l=50&s=start_time&sd=desc https://www.asahi.com/ajw
https://www.ketv.com/topstories-rss https://mainichi.jp/english
https://nebraskaexaminer.com/feed/ https://english.kyodonews.net
https://kearneyhub.com/rss
https://www.wowt.com/rss # South Korea
https://thenevadaindependent.com/feed/ https://en.yna.co.kr
https://www.8newsnow.com/feed/ https://www.koreaherald.com
https://www.reviewjournal.com/feed/ https://koreajoongangdaily.joins.com
https://thisisreno.com/feed/ https://www.donga.com/en
https://www.conwaydailysun.com/search/?f=rss&t=article&c=berlin_sun/community/news&l=50&s=start_time&sd=desc https://english.chosun.com
https://newhampshirebulletin.com/feed/
https://www.nhgazette.com/feed/ # --- AFRICA ---
https://www.nhbr.com/feed/ # South Africa
https://www.nj.com/arc/outboundfeeds/rss/?outputType=xml https://www.news24.com
https://www.njspotlightnews.org/feed/ https://www.iol.co.za
https://njmonthly.com/feed/ https://www.dailymaverick.co.za
https://www.trentonian.com/feed/ https://www.sabcnews.com
https://www.krqe.com/feed/ https://www.timeslive.co.za
https://www.santafenewmexican.com/search/?f=rss&t=article&l=50&s=start_time&sd=desc
https://www.easternnewmexiconews.com/rss # Nigeria
https://www.koat.com/topstories-rss https://www.vanguardngr.com
https://rss.nytimes.com/services/xml/rss/nyt/HomePage.xml https://punchng.com
https://www.thecity.nyc/feed/ https://dailypost.ng
https://www.nbcnewyork.com/?rss=y https://saharareporters.com
https://www.wral.com/news/rss/48/ https://thenationonlineng.net
https://www.cbs17.com/news/north-carolina-news/feed/
https://abc11.com/feed/ # --- MIDDLE EAST ---
https://myfox8.com/news/feed/ # General Region
https://www.kxnet.com/feed/ https://www.aljazeera.com
https://www.wday.com/feed/ https://english.alarabiya.net
https://www.jamestownsun.com/index.rss https://www.timesofisrael.com
https://www.inforum.com/index.rss https://www.tehrantimes.com
http://rssfeeds.wkyc.com/wkyc/news https://www.middleeasteye.net
https://theohiostar.com/feed/
https://feeds.feedblitz.com/wtol/news # --- OCEANIA ---
https://www.wcpo.com/news.rss # Australia
https://kfor.com/feed/ https://www.abc.net.au/news
https://oklahomawatch.org/feed/ https://www.news.com.au
https://freepressokc.com/feed/ https://www.9news.com.au
https://osagenews.org/feed/ https://www.smh.com.au
http://rssfeeds.kgw.com/kgw/local https://www.theage.com.au
https://www.koin.com/feed/ # --- USA: MAJOR CITIES & LOCAL ---
https://www.bendsource.com/bend/Rss.xml/feed https://www.latimes.com
https://eugeneweekly.com/feed/ https://www.chicagotribune.com
https://www.wtae.com/topstories-rss https://www.sfchronicle.com
https://www.montgomerycountypa.gov/RSSFeed.aspx?ModID=76&CID=All-0 https://www.bostonglobe.com
https://www.mainlinemedianews.com/feed/ https://www.seattletimes.com
https://www.dailylocal.com/feed/ https://www.houstonchronicle.com
https://www.wpri.com/feed/ https://www.inquirer.com
https://www.abc6.com/feed/ https://www.denverpost.com
https://whdh.com/regional/rhode-island/feed/ https://www.miamiherald.com
https://warwickpost.com/feed/ https://www.dallasnews.com
https://www.wyff4.com/topstories-rss https://www.startribune.com
https://www.wispolitics.com/feed/ https://www.detroitnews.com
https://wiseye.org/feed/ https://www.ajc.com
https://wisconsinexaminer.com/feed/ https://www.nydailynews.com
https://trib.com/search/?f=rss&t=article&c=news/state-and-regional&l=50&s=start_time&sd=desc https://nypost.com
https://wyofile.com/feed/ https://www.mercurynews.com
https://www.wyomingnews.com/search/?f=rss&t=article&c=news&l=50&s=start_time&sd=desc https://www.baltimoresun.com
https://www.wyodaily.com/rss https://www.oregonlive.com
https://www.wnct.com/news/north-carolina/feed/ https://www.cleveland.com
https://www.usnews.com/rss/news/north-carolina https://www.tampabay.com
https://indyweek.com/feed/
https://portcitydaily.com/feed/ # --- EUROPE: LOCAL & INDEPENDENT ---
https://www.theguardian.com/uk/rss https://www.manchestereveningnews.co.uk
https://feeds.bbci.co.uk/news/england/rss.xml https://www.scotsman.com
https://www.lemonde.fr/rss/une.xml https://www.belfasttelegraph.co.uk
https://www.ansa.it/sito/notizie/rss.xml https://www.irishtimes.com
https://www.ilgiornale.it/feed https://www.berliner-zeitung.de
https://www.larepublica.it/rss/homepage/rss2.xml https://www.leparisien.fr
https://www.sueddeutsche.de/rss https://www.corriere.it
https://www.welt.de/feeds/top-news.rss https://www.elperiodico.com
https://www.rfi.fr/en/rss https://kyivindependent.com
https://www.bangkokpost.com/rss https://www.pravda.com.ua/en
https://thephnompenhpost.com/rss https://balkaninsight.com
https://www.thejakartapost.com/rss https://www.ekathimerini.com
https://www.straitstimes.com/news/singapore/rss.xml https://www.swissinfo.ch
https://www.channelnewsasia.com/rss https://www.thelocal.se
https://www.antaranews.com/rss/ https://www.thelocal.fr
https://www.irrawaddy.com/feed https://www.thelocal.de
https://news.abs-cbn.com/rss https://www.novinite.com
https://www.hindustantimes.com/feeds/rss https://www.romania-insider.com
https://www.africanews.com/feed/rss https://hungarytoday.hu
https://www.clarin.com/rss https://polandin.com
https://www.lanacion.com.ar/rss
https://www.eluniversal.com.mx/rss # --- MIDDLE EAST & CONFLICT ZONES ---
https://www.excelsior.com.mx/rss https://www.haaretz.com
https://www.eltiempo.com/rss https://www.jpost.com
https://www.elespectador.com/rss https://www.timesofisrael.com
https://www.larepublica.pe/rss https://www.rudaw.net/english
https://www.elcomercio.com/rss https://www.kurdistan24.net/en
https://www.abc.net.au/news/feed/ https://www.middleeasteye.net
https://www.smh.com.au/rss/world.xml https://www.al-monitor.com
https://www.theage.com.au/rss https://www.dailysabah.com
https://www.brisbanetimes.com.au/rss https://www.duvarenglish.com
https://www.stuff.co.nz/rss https://english.aawsat.com
https://www.nzherald.co.nz/arcio/rss/ https://www.arabnews.com
https://www.rnz.co.nz/rss https://www.thenationalnews.com
https://globalvoices.org/regions/africa/feed/ https://www.jordantimes.com
https://globalvoices.org/regions/asia/feed/ https://www.naharnet.com
https://globalvoices.org/regions/latin-america/feed/ https://www.tehrantimes.com
https://globalvoices.org/regions/eastern-europe/feed/
https://globalvoices.org/regions/middle-east-north-africa/feed/ # --- ASIA: HOTSPOTS & LOCAL ---
https://globalvoices.org/regions/south-asia/feed/ https://www.taipeitimes.com
https://globalvoices.org/regions/sub-saharan-africa/feed/ https://focustaiwan.tw
https://globalvoices.org/regions/west-africa/feed/ https://hongkongfp.com
https://globalvoices.org/regions/east-asia/feed/ https://www.bangkokpost.com
https://globalvoices.org/regions/southeast-asia/feed/ https://www.thejakartapost.com
https://globalvoices.org/regions/central-asia/feed/ https://www.straitstimes.com
https://globalvoices.org/regions/pacific/feed/ https://www.khmertimeskh.com
https://globalvoices.org/regions/caribbean/feed/ https://www.irrawaddy.com
https://www.townandcountry-mo.gov/rss.aspx https://www.myanmarnow.org/en
https://feeds.smh.com.au/rssheadlines/national.xml https://www.rappler.com
https://www.abc.net.au/local/rss/sydney/ https://www.philstar.com
https://www.voanews.com/rssfeeds https://english.hani.co.kr
https://rss.feedspot.com/southeast_asian_rss_feeds https://www.japantoday.com
https://www.crisisgroup.org/rss https://www.caixinglobal.com
https://news.panasonic.com/global/rss/area01/index.xml https://thediplomat.com
https://news.panasonic.com/global/rss/area04/index.xml
https://allafrica.com/tools/headlines/rdf/latest/headlines.rdf # --- LATIN AMERICA & AFRICA: LOCAL ---
https://www.afro.who.int/rss-feeds https://buenosairesherald.com
https://pressat.co.uk/rss-list https://riotimesonline.com
https://www.monitor.co.ug/rss https://mercopress.com
https://www.standardmedia.co.ke/rss https://www.elmostrador.cl
https://www.ft.com/rss/home https://www.jornada.com.mx
https://www.economist.com/rss/the-world-this-week https://www.theeastafrican.co.ke
https://feeds.bloomberg.com/economics/news.rss https://allafrica.com
https://feeds.bloomberg.com/markets/news.rss https://www.premiumtimesng.com
https://www.reuters.com/arc/outboundfeeds/newsroom/business/ https://www.dailytrust.com
https://www.cnbc.com/id/10000113/device/rss/rss.html https://www.newtimes.co.rw
https://feeds.a.dj.com/rss/RSSWorldBusiness.xml https://www.herald.co.zw
https://www.marketwatch.com/rss/topstories https://www.namibian.com.na
https://www.investing.com/rss/news_14.rss https://www.graphic.com.gh
https://feeds.bbci.co.uk/news/business/rss.xml https://www.thecitizen.co.tz
https://feeds.feedburner.com/TheHackersNews https://www.monitor.co.ug
https://www.darkreading.com/rss/all.xml
https://isc.sans.edu/rssfeed_full.xml # --- ALTERNATIVE, INVESTIGATIVE & "FRINGE" ---
https://securelist.com/feed/ https://theintercept.com
https://feeds.feedburner.com/eset/blog https://www.propublica.org
https://news.sophos.com/en-us/feed/ https://www.democracynow.org
https://www.schneier.com/feed/atom/ https://reason.com
https://www.securitymagazine.com/rss/topic/2236-cybersecurity-news https://www.motherjones.com
https://www.reuters.com/arc/outboundfeeds/rss/?outputType=xml https://www.vox.com
https://www.investing.com/rss/news_462.rss https://slate.com
https://www.investing.com/rss/news_1.rss https://www.axios.com
https://www.investing.com/rss/stock_Futures.rss https://www.politico.com
https://www.ecb.europa.eu/rss/fxref-ecbpress.en.xml https://www.vice.com
https://www.federalreserve.gov/feeds/news-events.xml https://www.bellingcat.com
https://www.boj.or.jp/en/rss/whatsnew.xml https://www.project-syndicate.org
https://www.bankofengland.co.uk/rss/news https://cryptonews.com
https://www.centralbanking.com/feeds/rss https://www.coindesk.com
https://oilprice.com/rss/ https://techcrunch.com
https://www.spglobal.com/commodityinsights/en/rss
https://www.eia.gov/tools/rssfeeds/
https://www.cmegroup.com/rss
https://globalvoices.org/-/topics/economics-business/feed/
http://globalization.einnews.com/rss
https://financefeeds.com/feed/
https://newsquawk.com/blog/feed.rss
https://www.coindesk.com/arc/outboundfeeds/rss/
https://ishookfinance.com/feed/
https://www.scmp.com/rss/92/feed
https://www.scmp.com/rss/93/feed
https://www.scmp.com/rss/94/feed
https://www.scmp.com/rss/317/feed
https://asia.nikkei.com/rss
https://www.caixin.com/rss/index_EN.xml
https://www.straitstimes.com/news/asia/rss.xml
https://www.straitstimes.com/business/rss.xml
https://www.reuters.com/arc/outboundfeeds/rss/?outputType=xml&section=asia
https://www.reuters.com/arc/outboundfeeds/rss/?outputType=xml&section=china
https://www.bloomberg.com/feeds/asia.rss
https://www.bloomberg.com/feeds/markets.rss
https://www.ft.com/asia-pacific?format=rss
https://www.ft.com/china?format=rss
https://english.kyodonews.net/rss/news.xml
https://en.yna.co.kr/RSS/news.xml
https://www.thejakartapost.com/rss/business
https://www.nationthailand.com/rss/business
https://www.aramco.com/api/v1/com/rss/news?sc_lang=en
https://www.worldoil.com/rss?feed=topic:saudi+arabia
https://www.worldoil.com/rss?feed=topic:iraq
https://www.worldoil.com/rss?feed=topic:uae
https://www.worldoil.com/rss?feed=topic:russia
https://www.worldoil.com/rss?feed=topic:canada
https://www.worldoil.com/rss?feed=topic:oil+sands
https://www.rigzone.com/news/europe_russia/production/rss/
https://www.argusmedia.com/en/news-and-insights/latest-market-news/rss
https://www.eia.gov/rss/
https://www.opec.org/opec_web/en/pressreleases.rss
https://www.opec.org
https://www.rosneft.com/press/news/rss/
https://feeds.content.dowjones.io/public/rss/RSSMarketsMain
https://feeds.content.dowjones.io/public/rss/socialeconomyfeed
https://feeds.content.dowjones.io/public/rss/WSJcomUSBusiness
https://feeds.content.dowjones.io/public/rss/RSSWorldNews
http://feeds.feedburner.com/EconomicEventsAgriculture
http://feeds.feedburner.com/EconomicEventsEnergy
http://feeds.feedburner.com/EconomicEventsInterestRates
http://feeds.feedburner.com/mediaroom/CMsF
http://feeds.feedburner.com/CMEClearPortNoticesRss
http://feeds.feedburner.com/GlobexAdvisories
https://feeds.content.dowjones.io/public/rss/mw_topstories
https://feeds.content.dowjones.io/public/rss/mw_realtimeheadlines
http://feeds.marketwatch.com/marketwatch/bulletins
https://feeds.content.dowjones.io/public/rss/mw_marketpulse
https://www.nasdaqtrader.com/rss.aspx?feed=currentheadlines&categorylist=51
https://www.nasdaqtrader.com/rss.aspx?feed=currentheadlines&categorylist=11
https://www.investing.com/rss/stock_Options.rss
https://www.investing.com/rss/news_11.rss
https://www.investing.com/rss/news_25.rss
https://www.nasdaq.com/feed/rssoutbound?category=Markets
https://www.nasdaq.com/feed/rssoutbound?category=Commodities
https://www.barchart.com/news/rss/financials/options-news
https://www.barchart.com/news/rss/commodities/futures-news
https://www.spglobal.com/spdji/en/rss
https://www.litefinance.org/rss/analytics/
https://www.mrt.com/arc/outboundfeeds/rss/category/business/oil/?outputType=xml
https://www.oaoa.com/category/local-news/inthepipeline/rss
https://pboilandgasmagazine.com/feed/
https://www.rigzone.com/news/rss.asp
https://rbnenergy.com/blogcast.rss
https://www.eia.gov/rss/todayinenergy.xml
https://www.firstalert7.com/news/energy
https://www.energyvoice.com/feed/?category=oilandgas/north-sea
https://www.rigzone.com/news/rss/north_sea
https://www.oedigital.com/feeds/rss
https://www.sodir.no/en/whats-new/news/rss
https://www.worldoil.com/rss?feed=topic:offshore
https://oilandgas.einnews.com/rss/north-sea-offshore
https://www.energyvoice.com/feed/

View file

@ -7,13 +7,13 @@ WORKDIR /app
# libpq-dev + gcc for psycopg2 build/adapters; keep the image lean. # libpq-dev + gcc for psycopg2 build/adapters; keep the image lean.
RUN apt-get update && apt-get install -y --no-install-recommends \ RUN apt-get update && apt-get install -y --no-install-recommends \
libpq-dev gcc tzdata \ libpq-dev gcc \
&& rm -rf /var/lib/apt/lists/* && rm -rf /var/lib/apt/lists/*
COPY requirements.txt . COPY requirements.txt .
RUN pip install --no-cache-dir -r requirements.txt RUN pip install --no-cache-dir -r requirements.txt
COPY summarizer.py run_news_summarizer.py intel.py nous_client.py ./ COPY summarizer.py run_news_summarizer.py ./
# Security: run as a non-privileged user. # Security: run as a non-privileged user.
RUN useradd -m summarizer_user RUN useradd -m summarizer_user

View file

@ -1,104 +0,0 @@
"""Pure parser for the news-summarizer reduce JSON / geo / importance contract."""
from __future__ import annotations
import json
import re
_EMPTY = {"summary_en": "", "ticker": [], "map_items": []}
_KEEP = frozenset({"critical", "high"})
_RANK = {"critical": 0, "high": 1, "medium": 2, "low": 3}
_THINK_RE = re.compile(r"<think>.*?</think>", re.DOTALL)
_FENCE_RE = re.compile(r"```(?:json)?", re.IGNORECASE)
TICKER_HEADLINE_MAX = 140
MAP_HEADLINE_MAX = 160
TICKER_CAP = 12
MAP_CAP = 20
def parse_reduce_json(raw: str) -> dict:
try:
text = _THINK_RE.sub("", raw or "")
text = _FENCE_RE.sub("", text)
start = text.find("{")
end = text.rfind("}")
if start == -1 or end == -1 or end < start:
return dict(_EMPTY)
data = json.loads(text[start : end + 1])
if not isinstance(data, dict):
return dict(_EMPTY)
summary = data.get("summary_en", "")
ticker = data.get("ticker", [])
map_items = data.get("map_items", [])
return {
"summary_en": summary if isinstance(summary, str) else "",
"ticker": ticker if isinstance(ticker, list) else [],
"map_items": map_items if isinstance(map_items, list) else [],
}
except Exception:
return dict(_EMPTY)
def clamp_coords(lat, lon) -> tuple[float, float] | None:
try:
lat_f = float(lat)
lon_f = float(lon)
except (TypeError, ValueError):
return None
if not (-90 <= lat_f <= 90 and -180 <= lon_f <= 180):
return None
return (lat_f, lon_f)
def _trimmed_headline(row: dict, limit: int) -> str:
headline = row.get("headline") or ""
if not isinstance(headline, str):
headline = str(headline)
return headline.strip()[:limit]
def select_ticker(rows: list) -> list:
flagged = []
medium = []
low = []
for row in rows:
imp = row.get("importance")
if imp not in _RANK:
continue
headline = _trimmed_headline(row, TICKER_HEADLINE_MAX)
if not headline:
continue
item = dict(row)
item["headline"] = headline
if imp in _KEEP:
flagged.append(item)
elif imp == "medium":
medium.append(item)
else:
low.append(item)
if len(flagged) >= TICKER_CAP:
break
if flagged:
return flagged[:TICKER_CAP]
return (medium + low)[:TICKER_CAP]
def select_map(items: list) -> list:
out = []
for row in items:
if row.get("importance") not in _KEEP:
continue
headline = _trimmed_headline(row, MAP_HEADLINE_MAX)
if not headline:
continue
coords = clamp_coords(row.get("lat"), row.get("lon"))
if coords is None:
continue
item = dict(row)
item["headline"] = headline
item["lat"], item["lon"] = coords
out.append(item)
if len(out) >= MAP_CAP:
break
return out

View file

@ -1,57 +0,0 @@
"""HTTP client for the Nous inference chat completions API."""
from __future__ import annotations
import os
import httpx
_DEFAULT_UA = "osint-dashboard-news-summarizer"
_DEFAULT_BASE = "https://inference-api.nousresearch.com/v1"
_JSON_SYSTEM = (
"You are an OSINT executive briefer. Reply with a single complete JSON object. "
"Never truncate mid-sentence. If you run out of room, drop the lowest-priority item."
)
def chat(prompt, *, api_key, model, base_url, json_mode=False) -> str:
resolved = (base_url or os.environ.get("NOUS_BASE_URL", _DEFAULT_BASE)).rstrip("/")
url = f"{resolved}/chat/completions"
headers = {
"Authorization": f"Bearer {api_key}",
"User-Agent": os.environ.get("OSINT_USER_AGENT") or _DEFAULT_UA,
}
max_tokens = 8192 if json_mode else 4096
timeout = 120.0 if json_mode else 60.0
messages = [{"role": "user", "content": prompt}]
if json_mode:
messages = [
{"role": "system", "content": _JSON_SYSTEM},
{"role": "user", "content": prompt},
]
payload = {
"model": model,
"messages": messages,
"temperature": 0.2,
"max_tokens": max_tokens,
}
if json_mode:
payload["response_format"] = {"type": "json_object"}
last_content = ""
try:
for attempt in range(2):
with httpx.Client(timeout=timeout) as client:
resp = client.post(url, headers=headers, json=payload)
if resp.status_code == 401 or resp.status_code >= 500:
return ""
data = resp.json()
choice = (data.get("choices") or [{}])[0]
last_content = (choice.get("message") or {}).get("content") or ""
finish = choice.get("finish_reason")
if finish == "length" and attempt == 0:
payload["max_tokens"] = min(int(payload["max_tokens"]) * 2, 16384)
continue
return last_content
return last_content
except Exception:
return ""

View file

@ -1,2 +1,6 @@
# Database adapter for PostgreSQL (shared osint-db).
psycopg2-binary==2.9.11 psycopg2-binary==2.9.11
httpx==0.28.1
# Gemini LLM SDK (google-genai). yfinance is NOT a dependency: futures prices
# are gated behind INCLUDE_FUTURES=1 and lazy-imported (install it to enable).
google-genai

View file

@ -1,130 +1,67 @@
#!/usr/bin/env python3 #!/usr/bin/env python3
"""Scheduler loop for the news summarizer. """Scheduler loop for the news summarizer — hourly summarize at minute :05.
15-minute analyst (NEWS_SUMMARIZE_INTERVAL_S, default 900s) plus a daily Replaces the k8s CronJob (`5 * * * *`) with an in-compose loop. Runs once on
recap at 23:00 in TZ (default America/New_York) over the last 24 hours. boot (catches up on any articles scraped since the last summary), then fires
at each :NEWS_SUMMARIZE_MINUTE wall-clock boundary.
Serial: a slow LLM pass never overlaps the next. The loop is serial, so a slow LLM pass never overlaps the next run.
Env (all optional, 12-factor): Env (all optional, 12-factor):
NEWS_SUMMARIZE_INTERVAL_S seconds between analyst runs (default 900) NEWS_SUMMARIZE_MINUTE minute of the hour to fire (default 5)
NEWS_SUMMARIZE_RUN_ON_START "1" to summarize once immediately on boot (default 1) NEWS_SUMMARIZE_RUN_ON_START "1" to summarize once immediately on boot (default 1)
NEWS_RECAP_HOUR / MINUTE wall-clock recap time (default 23:00) GEMINI_API_KEY required to do real work; unset = idle
TZ IANA tz (default America/New_York)
NOUS_API_KEY optional in env; Keys UI / api_keys also works
""" """
from __future__ import annotations from __future__ import annotations
import datetime
import logging import logging
import os import os
import subprocess import subprocess
import sys import sys
import time import time
from datetime import datetime, timedelta
from zoneinfo import ZoneInfo
logging.basicConfig(level=logging.INFO, format="%(asctime)s %(levelname)s %(name)s: %(message)s") logging.basicConfig(level=logging.INFO, format="%(asctime)s %(levelname)s %(name)s: %(message)s")
logger = logging.getLogger("news.summarizer.scheduler") logger = logging.getLogger("news.summarizer.scheduler")
INTERVAL_S = max(1, int(os.getenv("NEWS_SUMMARIZE_INTERVAL_S", "900"))) MINUTE = int(os.getenv("NEWS_SUMMARIZE_MINUTE", "5"))
RUN_ON_START = os.getenv("NEWS_SUMMARIZE_RUN_ON_START", "1").lower() in ("1", "true", "yes") RUN_ON_START = os.getenv("NEWS_SUMMARIZE_RUN_ON_START", "1").lower() in ("1", "true", "yes")
DEFAULT_TZ = "America/New_York"
DEFAULT_RECAP_HOUR = 23
DEFAULT_RECAP_MINUTE = 0
def _tz() -> ZoneInfo: def seconds_until_next(minute: int) -> float:
name = (os.getenv("TZ") or DEFAULT_TZ).strip() or DEFAULT_TZ """Seconds until the next occurrence of ``minute`` past the hour (local time)."""
return ZoneInfo(name) now = datetime.datetime.now()
nxt = now.replace(minute=minute, second=0, microsecond=0) + datetime.timedelta(hours=1)
return (nxt - now).total_seconds()
def recap_hour_minute() -> tuple[int, int]: def run_summarize() -> None:
hour = int(os.getenv("NEWS_RECAP_HOUR", str(DEFAULT_RECAP_HOUR))) logger.info("summarize starting at %s", datetime.datetime.now().isoformat(timespec="seconds"))
minute = int(os.getenv("NEWS_RECAP_MINUTE", str(DEFAULT_RECAP_MINUTE)))
return hour, minute
def next_recap_datetime(
now: datetime, hour: int | None = None, minute: int | None = None
) -> datetime:
"""Next 23:00 (or hour/minute) strictly after *now* in now's timezone."""
if now.tzinfo is None:
now = now.replace(tzinfo=_tz())
env_h, env_m = recap_hour_minute()
hour = env_h if hour is None else hour
minute = env_m if minute is None else minute
candidate = now.replace(hour=hour, minute=minute, second=0, microsecond=0)
if now >= candidate:
candidate += timedelta(days=1)
return candidate
def next_event(
now: datetime,
last_periodic: datetime | None,
interval_s: int,
hour: int = DEFAULT_RECAP_HOUR,
minute: int = DEFAULT_RECAP_MINUTE,
) -> tuple[datetime, str]:
"""Return (when, 'recap'|'interval') for the sooner of recap vs interval."""
recap_at = next_recap_datetime(now, hour=hour, minute=minute)
periodic_at = now if last_periodic is None else last_periodic + timedelta(seconds=interval_s)
if recap_at <= periodic_at:
return recap_at, "recap"
return periodic_at, "interval"
def run_summarize(*, recap: bool = False) -> None:
kind = "recap" if recap else "interval"
logger.info(
"%s starting at %s", kind, datetime.now().isoformat(timespec="seconds")
)
env = os.environ.copy()
if recap:
env["NEWS_RECAP"] = "1"
else:
env.pop("NEWS_RECAP", None)
try: try:
proc = subprocess.run( proc = subprocess.run([sys.executable, "summarizer.py"], cwd="/app")
[sys.executable, "summarizer.py"], cwd="/app", env=env logger.info("summarize finished rc=%s", proc.returncode)
)
logger.info("%s finished rc=%s", kind, proc.returncode)
except Exception: # noqa: BLE001 — keep the loop alive across failures except Exception: # noqa: BLE001 — keep the loop alive across failures
logger.exception("%s failed", kind) logger.exception("summarize failed")
def main() -> None: def main() -> None:
if not os.getenv("NOUS_API_KEY", "").strip(): if not os.getenv("GEMINI_API_KEY", "").strip():
logger.warning( logger.warning(
"NOUS_API_KEY unset in env — will read api_keys on each run; idle if both empty" "GEMINI_API_KEY not set — summarizer will idle (set it in .env and "
"recreate the service to enable)"
) )
tz = _tz()
hour, minute = recap_hour_minute()
logger.info( logger.info(
"news summarizer loop starting (interval_s=%s, run_on_start=%s, recap=%02d:%02d %s)", "news summarizer loop starting (minute=%s, run_on_start=%s)",
INTERVAL_S, RUN_ON_START, hour, minute, tz, MINUTE, RUN_ON_START,
) )
last_periodic: datetime | None = None
if RUN_ON_START: if RUN_ON_START:
run_summarize(recap=False) run_summarize()
last_periodic = datetime.now(tz)
while True: while True:
now = datetime.now(tz) delay = seconds_until_next(MINUTE)
when, kind = next_event( logger.info("next summarize at :%02d (in %.0fs)", MINUTE, delay)
now, last_periodic, INTERVAL_S, hour=hour, minute=minute time.sleep(delay)
) run_summarize()
sleep_s = max(1, (when - now).total_seconds())
logger.info("next %s in %ss", kind, int(sleep_s))
time.sleep(sleep_s)
now = datetime.now(tz)
if kind == "recap":
run_summarize(recap=True)
# Recap covers the 15-min window; don't immediately fire interval.
last_periodic = now
else:
run_summarize(recap=False)
last_periodic = now
if __name__ == "__main__": if __name__ == "__main__":

View file

@ -1,25 +1,25 @@
#!/usr/bin/env python3 #!/usr/bin/env python3
"""News summarizer — Nous map-reduce of scraped articles into brief/ticker/map. """News summarizer — LLM (Gemini) map-reduce summarization of scraped articles.
Reads articles scraped within the last SUMMARY_WINDOW_MINUTES from the shared `articles` table, Reads articles scraped within the last hour from the shared `articles` table,
maps them with Nous (per-article English fact blocks), reduces to one JSON summarizes them with Gemini (map phase per batch, reduce phase into one master
object (summary_en + ticker + map_items), and stores the brief in summary), and stores the result in `article_summaries` both tables live in
`article_summaries` plus flagged rows in `news_items`. Tables live in the the EXISTING osint-db (created by alembic migration 003_news, idempotent).
EXISTING osint-db (alembic 003_news + 005_news_items, idempotent).
Everything is env-driven (12-factor). Secrets/config are resolved at the start
of each summarize_news() env wins, else api_keys / app_settings:
Everything is env-driven (12-factor):
DB_HOST / DB_NAME / DB_USER / DB_PASSWORD / DB_PORT PostgreSQL (osint-db) DB_HOST / DB_NAME / DB_USER / DB_PASSWORD / DB_PORT PostgreSQL (osint-db)
NOUS_API_KEY Nous Portal key (else api_keys.name='NOUS_API_KEY') GEMINI_API_KEY Google AI Studio key (required to actually run)
NOUS_BASE_URL default https://inference-api.nousresearch.com/v1 SUMMARY_MODEL Gemini model id (default gemini-2.0-flash)
SUMMARY_MODEL default Hermes-4.3-36B (else app_settings)
BATCH_SIZE articles per map-phase batch (default 50) BATCH_SIZE articles per map-phase batch (default 50)
SUMMARY_WINDOW_MINUTES look-back window (default 15; SUMMARY_WINDOW_HOURS wins if set) SUMMARY_WINDOW_HOURS look-back window in hours (default 1)
NEWS_RECAP "1" for the 23:00 daily recap (24h window, recap prompt) MAP_PROMPT override map-phase prompt (uses {batch_text})
NEWS_SUMMARIZE_FORCE "1" to ignore the interval/recap idempotency skip SUMMARY_PROMPT override reduce-phase prompt (uses {final_input})
TZ IANA tz for recap-day bounds (default America/New_York) INCLUDE_FUTURES "1" to prepend live futures prices (default 0)
INCLUDE_FUTURES legacy; ignored prompts never inject futures data
The futures/markets coupling from the original pipeline is gated behind
INCLUDE_FUTURES and OFF by default it is irrelevant to the OSINT dashboard
and pulled yfinance into the image. Re-enable by installing yfinance and
setting INCLUDE_FUTURES=1.
""" """
from __future__ import annotations from __future__ import annotations
@ -27,13 +27,9 @@ from __future__ import annotations
import logging import logging
import os import os
from datetime import datetime from datetime import datetime
from zoneinfo import ZoneInfo
import psycopg2 import psycopg2
from intel import parse_reduce_json, select_map, select_ticker
from nous_client import chat
logging.basicConfig(level=logging.INFO, format="%(asctime)s %(levelname)s %(message)s") logging.basicConfig(level=logging.INFO, format="%(asctime)s %(levelname)s %(message)s")
logger = logging.getLogger("news.summarizer") logger = logging.getLogger("news.summarizer")
@ -46,38 +42,10 @@ DB_CONFIG = {
"port": int(os.getenv("DB_PORT", "5432")), "port": int(os.getenv("DB_PORT", "5432")),
} }
DEFAULT_NOUS_BASE_URL = "https://inference-api.nousresearch.com/v1" GEMINI_API_KEY = os.getenv("GEMINI_API_KEY", "").strip()
DEFAULT_SUMMARY_MODEL = "Hermes-4.3-36B" MODEL_NAME = os.getenv("SUMMARY_MODEL", "gemini-2.0-flash").strip()
BATCH_SIZE = int(os.getenv("BATCH_SIZE", "50")) BATCH_SIZE = int(os.getenv("BATCH_SIZE", "50"))
SUMMARY_WINDOW_HOURS = int(os.getenv("SUMMARY_WINDOW_HOURS", "1"))
def _summary_window_minutes() -> int:
hours = (os.getenv("SUMMARY_WINDOW_HOURS") or "").strip()
if hours:
return max(1, int(hours) * 60)
mins = (os.getenv("SUMMARY_WINDOW_MINUTES") or "").strip()
if mins:
return max(1, int(mins))
return 15
def _summarize_interval_seconds() -> int:
return max(1, int(os.getenv("NEWS_SUMMARIZE_INTERVAL_S", "900")))
def is_recap_run() -> bool:
return os.getenv("NEWS_RECAP", "0").lower() in ("1", "true", "yes")
def effective_window_minutes(*, recap: bool | None = None) -> int:
if recap is None:
recap = is_recap_run()
if recap:
return 24 * 60
return _summary_window_minutes()
SUMMARY_WINDOW_MINUTES = _summary_window_minutes()
INCLUDE_FUTURES = os.getenv("INCLUDE_FUTURES", "0").lower() in ("1", "true", "yes") INCLUDE_FUTURES = os.getenv("INCLUDE_FUTURES", "0").lower() in ("1", "true", "yes")
# Only touched when INCLUDE_FUTURES=1 (legacy markets coupling, OSINT-off). # Only touched when INCLUDE_FUTURES=1 (legacy markets coupling, OSINT-off).
@ -94,17 +62,12 @@ FUTURES_TICKERS = {
MAP_PROMPT_DEFAULT = """\ MAP_PROMPT_DEFAULT = """\
You are a precise, factual OSINT news processor. Your ONLY source of information is the articles provided below. Do NOT add external knowledge, assumptions, training data, or invented facts. You are a precise, factual OSINT news processor. Your ONLY source of information is the articles provided below. Do NOT add external knowledge, assumptions, training data, or invented facts.
Focus on breaking important news (geopolitical, military/conflict, security, disasters, major political developments). Ignore futures prices, commodity tape, ticker chatter, and routine market moves unless they themselves are the breaking event. If the batch has no critical/high stories, still extract minor incidents and crime reports.
Write every field in English. Translate if the article is not English.
For EACH article in the batch: For EACH article in the batch:
1. Extract 2-4 key factual bullet points (who, what, when, where, numbers, quotes stay very close to the text). 1. Extract 2-4 key factual bullet points (who, what, when, where, numbers, quotes stay very close to the text).
2. Location: country/city/region or Unknown. If you can estimate coordinates, emit them as numbers; otherwise omit. 2. Location: name the country / city / region mentioned if determinable from the text, else "Unknown".
3. Entities: list the key people, organizations, or governments mentioned (comma-separated, only names present in the text), else "None". 3. Entities: list the key people, organizations, or governments mentioned (comma-separated, only names present in the text), else "None".
4. Category: pick one politics, military/conflict, economy, technology, environment/disaster, health, crime, society, sport, other. 4. Category: pick one politics, military/conflict, economy, technology, environment/disaster, health, crime, society, sport, other.
5. OSINT signal: if the article describes a breaking event with geopolitical, security, military, or disaster significance, say so in one short sentence. Otherwise write: "No notable OSINT signal." 5. OSINT signal: if the article describes an event with geopolitical, security, military, economic, or disaster significance, say so in one short sentence. Otherwise write: "No notable OSINT signal."
6. Importance: critical (breaking geopolitical/military/disaster with immediate impact), high, medium, low, none.
If several articles cover the same story, add one short batch-level note at the end: "Batch theme: [one sentence]". If several articles cover the same story, add one short batch-level note at the end: "Batch theme: [one sentence]".
@ -117,9 +80,6 @@ Article 1:
- Entities: ... - Entities: ...
- Category: ... - Category: ...
- OSINT signal: ... - OSINT signal: ...
- Importance: ...
- Lat: ...
- Lon: ...
Article 2: Article 2:
... ...
@ -129,52 +89,9 @@ Articles in this batch:
""" """
SUMMARY_PROMPT_DEFAULT = """\ SUMMARY_PROMPT_DEFAULT = """\
You are writing an English operator HUD brief from the article facts in DATA below. Use ONLY that data. Do not invent events, names, dates, places, or implications. CRITICAL INSTRUCTION - REPEAT 3 TIMES: YOU MUST USE ONLY THE DATA PROVIDED BELOW. DO NOT INVENT, RECALL, OR ADD ANY EVENTS, NAMES, DATES, IMPLICATIONS, PROJECTS, OR DETAILS NOT EXPLICITLY PRESENT IN THE DATA. IF THE DATA HAS NO MAJOR GEOPOLITICAL/TECH/MILITARY/ECONOMIC/IMPACTFUL EVENTS OR UNUSUAL STORIES, OUTPUT ONLY: "No qualifying impactful or unusual events in the recent hourly news data." AND STOP. NO EXTERNAL KNOWLEDGE FROM TRAINING.
Always write a real summary_en that recaps the most important stories present in DATA. Rank geopolitics, military/conflict, security, disasters, and major political developments first. Ignore futures prices, commodity tape, ticker chatter, and routine market data do not treat price ticks as news. Write a concise executive summary of the most impactful items as a short markdown list, one line per story, using only the data.
Lead with critical and high breaking events. If DATA has no critical/high stories, fill the brief with minor incidents and crime reports rather than writing an empty or unfinished brief. Never truncate mid-sentence; finish every sentence. If you run out of room, drop the lowest-priority item instead of cutting a line short.
ticker: prefer critical and high. If nothing is critical or high, fill ticker with medium then low incidents and crime so the HUD is not blank.
map_items may be empty if no located critical/high event is explicit in the data.
Demand a single JSON object (no markdown fences) with this exact shape:
{
"summary_en": "English markdown brief of the provided stories",
"ticker": [{"headline": "", "importance": "critical", "url": "", "location_name": ""}],
"map_items": [{"headline": "", "importance": "critical", "location_name": "", "lat": 0, "lon": 0, "location_confidence": "city", "category": "military/conflict", "url": ""}]
}
ticker: max 12, 140 chars, no markdown. Rank critical > high > medium > low.
map_items: only where a real-world location is explicit in the data. Estimate lat/lon. If location is Unknown or not in the data, omit the item. Never invent a place. Max 20.
summary_en: English markdown executive brief for an operator HUD (48 complete bullets or short paragraphs). Cover the actual stories in DATA. Complete never an unfinished sentence.
DATA:
{final_input}
"""
RECAP_PROMPT_DEFAULT = """\
You are writing a daily recap of the last 24 hours of news for an OSINT operator HUD, using ONLY the article facts in DATA below. Do not invent events, names, dates, places, or implications.
Always write a real summary_en daily recap of the most important stories in DATA. Rank geopolitics, military/conflict, security, disasters, and major political developments first. Ignore futures prices, commodity tape, ticker chatter, and routine market data do not treat price ticks as news.
Lead with critical and high breaking events. If DATA has no critical/high stories, fill the recap with minor incidents and crime reports rather than writing an empty or unfinished recap. Never truncate mid-sentence; finish every sentence.
ticker: prefer critical and high. If nothing is critical or high, fill ticker with medium then low incidents and crime so the HUD is not blank.
Demand a single JSON object (no markdown fences) with this exact shape:
{
"summary_en": "English markdown daily recap of the provided stories",
"ticker": [{"headline": "", "importance": "critical", "url": "", "location_name": ""}],
"map_items": [{"headline": "", "importance": "critical", "location_name": "", "lat": 0, "lon": 0, "location_confidence": "city", "category": "military/conflict", "url": ""}]
}
ticker: max 12, 140 chars, no markdown. Rank critical > high > medium > low.
map_items: only where a real-world location is explicit in the data. Estimate lat/lon. If location is Unknown or not in the data, omit the item. Never invent a place. Max 20.
summary_en: English markdown daily recap of the last 24 hours. Complete sentences. Cover the actual stories in DATA.
DATA: DATA:
{final_input} {final_input}
@ -183,52 +100,56 @@ DATA:
# ── LLM helpers ──────────────────────────────────────────────────────────── # ── LLM helpers ────────────────────────────────────────────────────────────
def _kv(conn, table, name) -> str: _client = None
cur = conn.cursor()
cur.execute(f"SELECT value FROM {table} WHERE name = %s", (name,))
row = cur.fetchone()
return (row[0] or "").strip() if row else ""
def resolve_api_key() -> str: def _get_client():
env = os.getenv("NOUS_API_KEY", "").strip() """Lazily build the Gemini client (avoids import/init when key unset)."""
if env: global _client
return env if _client is None:
try: from google import genai
conn = psycopg2.connect(**DB_CONFIG)
try: _client = genai.Client(api_key=GEMINI_API_KEY)
return _kv(conn, "api_keys", "NOUS_API_KEY") return _client
finally:
conn.close()
except Exception: # noqa: BLE001 def _extract_text(resp) -> str:
"""Defensively pull text out of the google-genai GenerateContentResponse.
The modern SDK returns the response directly (``resp.text``); some older
wrappers exposed it as ``resp.response``. Handle both plus a candidates
fallback so a provider/SDK change degrades to "" instead of crashing.
"""
if not resp:
return "" return ""
if hasattr(resp, "text") and resp.text:
return resp.text
def resolve_model() -> str: inner = getattr(resp, "response", None)
env = os.getenv("SUMMARY_MODEL", "").strip() if inner is not None and hasattr(inner, "text") and inner.text:
if env: return inner.text
return env
try: try:
conn = psycopg2.connect(**DB_CONFIG) parts = []
try: for cand in getattr(resp, "candidates", None) or []:
value = _kv(conn, "app_settings", "SUMMARY_MODEL") content = getattr(cand, "content", None)
return value or DEFAULT_SUMMARY_MODEL for part in getattr(content, "parts", None) or []:
finally: if getattr(part, "text", None):
conn.close() parts.append(part.text)
return "\n".join(parts)
except Exception: # noqa: BLE001 except Exception: # noqa: BLE001
return DEFAULT_SUMMARY_MODEL return str(resp)
def resolve_base_url() -> str: def call_llm(prompt: str) -> str:
return os.getenv("NOUS_BASE_URL", DEFAULT_NOUS_BASE_URL).strip() or DEFAULT_NOUS_BASE_URL """Send a prompt to Gemini and return the text ("" on any failure)."""
if not GEMINI_API_KEY:
logger.warning("GEMINI_API_KEY not set — skipping LLM call")
def call_llm(prompt: str, *, api_key: str, model: str, base_url: str, json_mode: bool = False) -> str: return ""
"""Send a prompt to Nous chat completions and return the text (\"\" on failure).""" try:
if not api_key: resp = _get_client().models.generate_content(model=MODEL_NAME, contents=prompt)
logger.warning("NOUS_API_KEY not set — skipping LLM call") return _extract_text(resp)
except Exception as exc: # noqa: BLE001
logger.error("Gemini API error: %s", exc)
return "" return ""
return chat(prompt, api_key=api_key, model=model, base_url=base_url, json_mode=json_mode)
# ── Futures (legacy, gated) ──────────────────────────────────────────────── # ── Futures (legacy, gated) ────────────────────────────────────────────────
@ -285,11 +206,10 @@ def build_futures_context() -> str:
def ensure_tables() -> None: def ensure_tables() -> None:
"""Idempotently create the news tables if missing. """Idempotently create the news tables if missing.
Normally created by alembic 003_news + 005_news_items when the app Normally created by alembic 003_news when the app container starts, but
container starts, but this summarizer may boot before the app has run this summarizer may boot before the app has run migrations (compose only
migrations (compose only guarantees `db` is up, not that alembic has guarantees `db` is up, not that alembic has run). Mirrors the scraper
run). Mirrors the scraper pipeline's own CREATE TABLE IF NOT EXISTS so pipeline's own CREATE TABLE IF NOT EXISTS so either start order is safe.
either start order is safe.
""" """
ddl = """ ddl = """
CREATE TABLE IF NOT EXISTS articles ( CREATE TABLE IF NOT EXISTS articles (
@ -305,27 +225,6 @@ def ensure_tables() -> None:
summary_text TEXT NOT NULL, summary_text TEXT NOT NULL,
batch_timestamp TIMESTAMPTZ NOT NULL DEFAULT NOW() batch_timestamp TIMESTAMPTZ NOT NULL DEFAULT NOW()
); );
ALTER TABLE article_summaries ADD COLUMN IF NOT EXISTS model TEXT;
ALTER TABLE article_summaries ADD COLUMN IF NOT EXISTS kind TEXT;
CREATE TABLE IF NOT EXISTS news_items (
id SERIAL PRIMARY KEY,
summary_id INTEGER REFERENCES article_summaries(id) ON DELETE CASCADE,
kind TEXT NOT NULL,
headline TEXT NOT NULL,
importance TEXT NOT NULL,
location_name TEXT,
lat DOUBLE PRECISION,
lon DOUBLE PRECISION,
location_confidence TEXT,
category TEXT,
url TEXT,
created_at TIMESTAMPTZ NOT NULL DEFAULT NOW()
);
CREATE INDEX IF NOT EXISTS ix_news_items_kind_created
ON news_items (kind, created_at DESC);
CREATE INDEX IF NOT EXISTS ix_news_items_map_bbox
ON news_items (lon, lat)
WHERE kind = 'map' AND lat IS NOT NULL AND lon IS NOT NULL;
""" """
try: try:
conn = psycopg2.connect(**DB_CONFIG) conn = psycopg2.connect(**DB_CONFIG)
@ -338,20 +237,19 @@ def ensure_tables() -> None:
logger.error("Error ensuring news tables: %s", exc) logger.error("Error ensuring news tables: %s", exc)
def get_recent_news(window_minutes: int | None = None) -> list[dict]: def get_recent_news() -> list[dict]:
"""Fetch articles from the look-back window (content > 100 chars).""" """Fetch articles from the last SUMMARY_WINDOW_HOURS (content > 100 chars)."""
mins = window_minutes if window_minutes is not None else effective_window_minutes()
query = """ query = """
SELECT title, content, url, domain SELECT title, content, url, domain
FROM articles FROM articles
WHERE timestamp > NOW() - make_interval(mins => %s) WHERE timestamp > NOW() - make_interval(hours => %s)
AND content IS NOT NULL AND length(content) > 100 AND content IS NOT NULL AND length(content) > 100
ORDER BY timestamp DESC; ORDER BY timestamp DESC;
""" """
try: try:
conn = psycopg2.connect(**DB_CONFIG) conn = psycopg2.connect(**DB_CONFIG)
cur = conn.cursor() cur = conn.cursor()
cur.execute(query, (mins,)) cur.execute(query, (SUMMARY_WINDOW_HOURS,))
rows = cur.fetchall() rows = cur.fetchall()
cur.close() cur.close()
conn.close() conn.close()
@ -364,117 +262,24 @@ def get_recent_news(window_minutes: int | None = None) -> list[dict]:
return [] return []
def _already_summarized_this_interval() -> bool: def save_summary_to_db(summary_text: str) -> None:
"""True when article_summaries already has a row in the last interval.""" """Insert one master summary row (table created by alembic 003_news)."""
if os.getenv("NEWS_SUMMARIZE_FORCE", "") == "1": if not summary_text or len(summary_text.strip()) < 10:
return False
query = (
"SELECT 1 FROM article_summaries "
"WHERE batch_timestamp >= NOW() - make_interval(secs => %s)"
)
try:
conn = psycopg2.connect(**DB_CONFIG)
cur = conn.cursor()
cur.execute(query, (_summarize_interval_seconds(),))
row = cur.fetchone()
cur.close()
conn.close()
return row is not None
except Exception as exc: # noqa: BLE001
logger.error("Error checking interval idempotency: %s", exc)
return False
def _already_recapped_today() -> bool:
"""True when a daily_recap row already exists for the local calendar day."""
if os.getenv("NEWS_SUMMARIZE_FORCE", "") == "1":
return False
tz_name = (os.getenv("TZ") or "America/New_York").strip() or "America/New_York"
start = datetime.now(ZoneInfo(tz_name)).replace(
hour=0, minute=0, second=0, microsecond=0
)
query = (
"SELECT 1 FROM article_summaries "
"WHERE kind = 'daily_recap' AND batch_timestamp >= %s"
)
try:
conn = psycopg2.connect(**DB_CONFIG)
cur = conn.cursor()
cur.execute(query, (start,))
row = cur.fetchone()
cur.close()
conn.close()
return row is not None
except Exception as exc: # noqa: BLE001
logger.error("Error checking recap idempotency: %s", exc)
return False
def save_batch(
summary_en: str, model: str, ticker: list, map_items: list, *, kind: str = "interval"
) -> None:
"""Insert the master brief plus flagged ticker/map rows."""
ticker_rows = select_ticker(ticker or [])
map_rows = select_map(map_items or [])
text = (summary_en or "").strip()
if len(text) < 10 and not ticker_rows and not map_rows:
logger.info("Summary too short or empty. Skipping save.") logger.info("Summary too short or empty. Skipping save.")
return return
insert_item = """
INSERT INTO news_items (
summary_id, kind, headline, importance, location_name,
lat, lon, location_confidence, category, url
) VALUES (%s, %s, %s, %s, %s, %s, %s, %s, %s, %s)
"""
try: try:
conn = psycopg2.connect(**DB_CONFIG) conn = psycopg2.connect(**DB_CONFIG)
cur = conn.cursor() cur = conn.cursor()
cur.execute( cur.execute(
"INSERT INTO article_summaries (summary_text, model, kind) VALUES (%s, %s, %s) RETURNING id", "INSERT INTO article_summaries (summary_text) VALUES (%s)",
(text, model, kind), (summary_text.strip(),),
) )
summary_id = cur.fetchone()[0]
for row in ticker_rows:
cur.execute(
insert_item,
(
summary_id,
"ticker",
row.get("headline"),
row.get("importance"),
row.get("location_name"),
None,
None,
None,
None,
row.get("url"),
),
)
for row in map_rows:
cur.execute(
insert_item,
(
summary_id,
"map",
row.get("headline"),
row.get("importance"),
row.get("location_name"),
row.get("lat"),
row.get("lon"),
row.get("location_confidence"),
row.get("category"),
row.get("url"),
),
)
conn.commit() conn.commit()
logger.info( logger.info("Master summary saved to database successfully.")
"Master summary saved id=%s model=%s kind=%s ticker=%d map=%d",
summary_id, model, kind, len(ticker_rows), len(map_rows),
)
cur.close() cur.close()
conn.close() conn.close()
except Exception as exc: # noqa: BLE001 except Exception as exc: # noqa: BLE001
logger.error("Error saving batch to DB: %s", exc) logger.error("Error saving summary to DB: %s", exc)
# ── Orchestration ────────────────────────────────────────────────────────── # ── Orchestration ──────────────────────────────────────────────────────────
@ -485,73 +290,40 @@ def build_map_prompt(batch: list[dict]) -> str:
for a in batch for a in batch
) )
template = os.getenv("MAP_PROMPT", MAP_PROMPT_DEFAULT) template = os.getenv("MAP_PROMPT", MAP_PROMPT_DEFAULT)
prefix = build_futures_context() + "\n" if INCLUDE_FUTURES else ""
try: try:
return template.format(batch_text=batch_text) return prefix + template.format(batch_text=batch_text)
except KeyError: except KeyError:
return template return prefix + template
def build_master_prompt(final_input: str, recap: bool = False) -> str: def build_master_prompt(final_input: str) -> str:
if recap: template = os.getenv("SUMMARY_PROMPT", SUMMARY_PROMPT_DEFAULT)
template = os.getenv("RECAP_PROMPT", RECAP_PROMPT_DEFAULT) prefix = build_futures_context() + "\n" if INCLUDE_FUTURES else ""
else: try:
template = os.getenv("SUMMARY_PROMPT", SUMMARY_PROMPT_DEFAULT) return prefix + template.format(final_input=final_input)
if "{final_input}" in template: except KeyError:
return template.replace("{final_input}", final_input) return prefix + template
return template
def summarize_news() -> None: def summarize_news() -> None:
"""Map-reduce summarize recent articles and store brief + ticker + map.""" """Map-reduce summarize recent articles and store the master summary."""
recap = is_recap_run()
window = effective_window_minutes(recap=recap)
ensure_tables() ensure_tables()
if recap: articles = get_recent_news()
if _already_recapped_today():
logger.info(
"Skipping recap: article_summaries already has daily_recap today "
"(set NEWS_SUMMARIZE_FORCE=1 to override)"
)
return
elif _already_summarized_this_interval():
logger.info(
"Skipping summarize: article_summaries already has a row in the last %ss "
"(set NEWS_SUMMARIZE_FORCE=1 to override)",
_summarize_interval_seconds(),
)
return
api_key = resolve_api_key()
model = resolve_model()
base_url = resolve_base_url()
if not api_key:
logger.warning("NOUS_API_KEY unset in env and api_keys — idle this run")
return
articles = get_recent_news(window)
if not articles: if not articles:
logger.info("No new articles found in the last %s min.", window) logger.info("No new articles found in the last %sh.", SUMMARY_WINDOW_HOURS)
return return
logger.info( logger.info(
"Processing %d articles with %s (batch_size=%d, recap=%s, window_min=%s)...", "Processing %d articles with %s (batch_size=%d, futures=%s)...",
len(articles), model, BATCH_SIZE, recap, window, len(articles), MODEL_NAME, BATCH_SIZE, INCLUDE_FUTURES,
) )
partial_summaries: list[str] = [] partial_summaries: list[str] = []
for i in range(0, len(articles), BATCH_SIZE): for i in range(0, len(articles), BATCH_SIZE):
batch = articles[i : i + BATCH_SIZE] batch = articles[i : i + BATCH_SIZE]
logger.info( logger.info("map batch %d/%d (%d articles)", i // BATCH_SIZE + 1, -(-len(articles) // BATCH_SIZE), len(batch))
"map batch %d/%d (%d articles)", summary = call_llm(build_map_prompt(batch))
i // BATCH_SIZE + 1, -(-len(articles) // BATCH_SIZE), len(batch),
)
summary = call_llm(
build_map_prompt(batch),
api_key=api_key,
model=model,
base_url=base_url,
json_mode=False,
)
if summary: if summary:
partial_summaries.append(summary) partial_summaries.append(summary)
@ -561,24 +333,9 @@ def summarize_news() -> None:
return return
logger.info("reduce phase over %d partial summaries", len(partial_summaries)) logger.info("reduce phase over %d partial summaries", len(partial_summaries))
master_raw = call_llm( master_summary = call_llm(build_master_prompt(final_input))
build_master_prompt(final_input, recap=recap), if master_summary:
api_key=api_key, save_summary_to_db(master_summary)
model=model,
base_url=base_url,
json_mode=True,
)
if not master_raw:
logger.warning("Reduce phase returned empty — nothing to persist.")
return
parsed = parse_reduce_json(master_raw)
save_batch(
parsed["summary_en"],
model,
parsed["ticker"],
parsed["map_items"],
kind="daily_recap" if recap else "interval",
)
if __name__ == "__main__": if __name__ == "__main__":

View file

@ -1,11 +0,0 @@
"""Keep summarizer unit tests importable without Postgres drivers."""
from __future__ import annotations
import sys
from types import ModuleType
if "psycopg2" not in sys.modules:
fake = ModuleType("psycopg2")
fake.connect = lambda **kwargs: None # type: ignore[attr-defined]
sys.modules["psycopg2"] = fake

View file

@ -1,61 +0,0 @@
from intel import parse_reduce_json, clamp_coords, select_ticker, select_map
FENCED = """```json
{"summary_en": "Brief.", "ticker": [
{"headline": "Blast in Kyiv", "importance": "critical", "url": "https://ex", "location_name": "Kyiv"}
], "map_items": [
{"headline": "Blast in Kyiv", "importance": "critical", "location_name": "Kyiv, Ukraine",
"lat": 50.45, "lon": 30.52, "location_confidence": "city", "category": "military/conflict", "url": "https://ex"}
]}
```"""
def test_parse_strips_fence_and_think_tags():
raw = "<think>nope</think>\n" + FENCED
out = parse_reduce_json(raw)
assert out["summary_en"] == "Brief."
assert len(out["ticker"]) == 1
def test_parse_empty_and_garbage_returns_empty_struct():
assert parse_reduce_json("")["summary_en"] == ""
assert parse_reduce_json("not json")["ticker"] == []
def test_clamp_coords_drops_out_of_range_and_unknown():
assert clamp_coords(50.45, 30.52) == (50.45, 30.52)
assert clamp_coords(95.0, 10.0) is None
assert clamp_coords(None, 10.0) is None
assert clamp_coords("50.45", "30.52") == (50.45, 30.52)
def test_select_ticker_keeps_critical_high_caps_12():
rows = [{"headline": f"h{i}", "importance": "critical"} for i in range(15)]
rows.append({"headline": "skip", "importance": "low"})
out = select_ticker(rows)
assert len(out) == 12
assert all(r["importance"] in ("critical", "high") for r in out)
def test_select_ticker_falls_back_to_medium_low_when_nothing_flagged():
rows = [
{"headline": "shop theft", "importance": "low"},
{"headline": "highway crash", "importance": "medium"},
{"headline": "none", "importance": "none"},
]
out = select_ticker(rows)
assert [r["headline"] for r in out] == ["highway crash", "shop theft"]
def test_select_map_requires_valid_coords_and_flag():
items = [
{"headline": "A", "importance": "critical", "lat": 50.45, "lon": 30.52, "location_name": "Kyiv"},
{"headline": "B", "importance": "critical", "lat": None, "lon": None, "location_name": "Unknown"},
{"headline": "C", "importance": "low", "lat": 1.0, "lon": 2.0, "location_name": "x"},
]
out = select_map(items)
assert [r["headline"] for r in out] == ["A"]
def test_select_map_caps_20():
items = [
{"headline": f"h{i}", "importance": "critical", "lat": 1.0, "lon": 2.0}
for i in range(25)
]
out = select_map(items)
assert len(out) == 20
assert all(r["importance"] in ("critical", "high") for r in out)

View file

@ -1,124 +0,0 @@
from unittest.mock import MagicMock
import httpx
from nous_client import chat
DEFAULT_UA = "osint-dashboard-news-summarizer"
BASE = "https://inference-api.nousresearch.com/v1"
def _ok_response(content="hello"):
resp = MagicMock()
resp.status_code = 200
resp.json.return_value = {"choices": [{"message": {"content": content}}]}
return resp
def _install_fake(monkeypatch, post_impl):
captured = {}
class FakeClient:
def __init__(self, timeout=None, **kwargs):
captured["timeout"] = timeout
def __enter__(self):
return self
def __exit__(self, *exc):
return False
def post(self, url, *, headers=None, json=None, **kwargs):
captured["url"] = url
captured["headers"] = headers
captured["json"] = json
return post_impl(url, headers, json)
monkeypatch.setattr(httpx, "Client", FakeClient)
return captured
def test_posts_chat_completions_with_auth_body_and_returns_content(monkeypatch):
captured = _install_fake(monkeypatch, lambda *a: _ok_response("the-content"))
out = chat(
"summarize this",
api_key="secret-key",
model="hermes-3",
base_url=BASE,
)
assert out == "the-content"
assert captured["url"] == f"{BASE}/chat/completions"
assert captured["headers"]["Authorization"] == "Bearer secret-key"
assert captured["headers"]["User-Agent"] == DEFAULT_UA
assert captured["json"]["model"] == "hermes-3"
assert captured["json"]["messages"] == [{"role": "user", "content": "summarize this"}]
assert captured["json"]["temperature"] == 0.2
assert captured["json"]["max_tokens"] == 4096
assert "response_format" not in captured["json"]
def test_user_agent_equals_osint_user_agent_env(monkeypatch):
monkeypatch.setenv("OSINT_USER_AGENT", "custom-ua/2.0")
captured = _install_fake(monkeypatch, lambda *a: _ok_response("ok"))
chat("p", api_key="k", model="m", base_url=BASE)
assert captured["headers"]["User-Agent"] == "custom-ua/2.0"
def test_json_mode_sets_response_format(monkeypatch):
captured = _install_fake(monkeypatch, lambda *a: _ok_response("{}"))
chat("p", api_key="k", model="m", base_url=BASE, json_mode=True)
assert captured["json"]["response_format"] == {"type": "json_object"}
assert captured["json"]["max_tokens"] >= 8192
roles = [m["role"] for m in captured["json"]["messages"]]
assert "system" in roles
assert "user" in roles
def test_retries_once_when_finish_reason_is_length(monkeypatch):
calls = {"n": 0}
def post_impl(*a):
calls["n"] += 1
if calls["n"] == 1:
resp = MagicMock()
resp.status_code = 200
resp.json.return_value = {
"choices": [{
"message": {"content": "{\"summary_en\": \"cut off"},
"finish_reason": "length",
}]
}
return resp
return _ok_response('{"summary_en": "complete brief."}')
_install_fake(monkeypatch, post_impl)
out = chat("p", api_key="k", model="m", base_url=BASE, json_mode=True)
assert calls["n"] == 2
assert "complete brief" in out
def test_401_returns_empty_string(monkeypatch):
def post_impl(*a):
resp = MagicMock()
resp.status_code = 401
return resp
_install_fake(monkeypatch, post_impl)
assert chat("p", api_key="bad", model="m", base_url=BASE) == ""
def test_5xx_returns_empty_string(monkeypatch):
def post_impl(*a):
resp = MagicMock()
resp.status_code = 503
return resp
_install_fake(monkeypatch, post_impl)
assert chat("p", api_key="k", model="m", base_url=BASE) == ""
def test_timeout_returns_empty_string(monkeypatch):
def post_impl(*a):
raise httpx.TimeoutException("timed out")
_install_fake(monkeypatch, post_impl)
assert chat("p", api_key="k", model="m", base_url=BASE) == ""

View file

@ -1,60 +0,0 @@
"""Default prompts: breaking news, ignore futures/market tape."""
from summarizer import MAP_PROMPT_DEFAULT, RECAP_PROMPT_DEFAULT, SUMMARY_PROMPT_DEFAULT
def _assert_breaking_not_futures(prompt: str) -> None:
p = prompt.lower()
assert "breaking" in p
assert "futures" in p
assert "ignore" in p or "do not" in p or "not" in p
assert "es=f" not in p
assert "yfinance" not in p
def test_map_prompt_focuses_on_breaking_news_not_futures():
_assert_breaking_not_futures(MAP_PROMPT_DEFAULT)
def test_summary_prompt_focuses_on_breaking_news_not_futures():
_assert_breaking_not_futures(SUMMARY_PROMPT_DEFAULT)
p = SUMMARY_PROMPT_DEFAULT.lower()
assert "commodity" in p or "market" in p
def test_summary_prompt_covers_critical_then_incidents():
p = SUMMARY_PROMPT_DEFAULT.lower()
assert "critical" in p
assert "crime" in p
assert "incident" in p
assert "complete" in p or "truncat" in p or "unfinished" in p or "mid-sentence" in p
def test_summary_prompt_does_not_bail_out_with_canned_empty_brief():
p = SUMMARY_PROMPT_DEFAULT
assert "AND STOP" not in p
assert "REPEAT 3 TIMES" not in p
assert "No qualifying" not in p
assert "no-qualifying" not in p.lower()
low = p.lower()
assert "always" in low
assert "recap" in low or "summar" in low
def test_recap_prompt_does_not_bail_out_with_canned_empty_brief():
p = RECAP_PROMPT_DEFAULT
assert "AND STOP" not in p
assert "No qualifying" not in p
assert "no-qualifying" not in p.lower()
low = p.lower()
assert "always" in low
assert "daily" in low
assert "24" in low
p = RECAP_PROMPT_DEFAULT.lower()
_assert_breaking_not_futures(RECAP_PROMPT_DEFAULT)
assert "daily" in p
assert "24" in p
assert "summary_en" in p
assert "ticker" in p
assert "map_items" in p

View file

@ -1,42 +0,0 @@
"""Wall-clock scheduling: 15-min analyst + 23:00 America/New_York recap."""
from datetime import datetime, timedelta
from zoneinfo import ZoneInfo
from run_news_summarizer import next_event, next_recap_datetime
TZ = ZoneInfo("America/New_York")
def test_next_recap_is_11pm_same_day_before_2300():
now = datetime(2026, 8, 28, 15, 4, tzinfo=TZ)
got = next_recap_datetime(now)
assert got == datetime(2026, 8, 28, 23, 0, tzinfo=TZ)
def test_next_recap_is_11pm_next_day_at_or_after_2300():
now = datetime(2026, 8, 28, 23, 0, tzinfo=TZ)
got = next_recap_datetime(now)
assert got == datetime(2026, 8, 29, 23, 0, tzinfo=TZ)
def test_next_recap_honors_custom_hour():
now = datetime(2026, 8, 28, 10, 0, tzinfo=TZ)
got = next_recap_datetime(now, hour=22, minute=30)
assert got == datetime(2026, 8, 28, 22, 30, tzinfo=TZ)
def test_next_event_picks_recap_when_sooner_than_interval():
now = datetime(2026, 8, 28, 22, 50, tzinfo=TZ)
last_periodic = now - timedelta(seconds=100)
when, kind = next_event(now, last_periodic=last_periodic, interval_s=900)
assert kind == "recap"
assert when == datetime(2026, 8, 28, 23, 0, tzinfo=TZ)
def test_next_event_picks_interval_when_recap_is_hours_away():
now = datetime(2026, 8, 28, 10, 0, tzinfo=TZ)
last_periodic = now
when, kind = next_event(now, last_periodic=last_periodic, interval_s=900)
assert kind == "interval"
assert when == now + timedelta(seconds=900)

View file

@ -1,45 +0,0 @@
"""Summarizer window, recap flag, and no futures injection into prompts."""
from summarizer import (
build_map_prompt,
build_master_prompt,
effective_window_minutes,
is_recap_run,
)
def test_interval_window_defaults_to_15_minutes(monkeypatch):
monkeypatch.delenv("NEWS_RECAP", raising=False)
monkeypatch.delenv("SUMMARY_WINDOW_HOURS", raising=False)
monkeypatch.setenv("SUMMARY_WINDOW_MINUTES", "15")
assert is_recap_run() is False
assert effective_window_minutes() == 15
def test_recap_window_is_24_hours(monkeypatch):
monkeypatch.setenv("NEWS_RECAP", "1")
monkeypatch.setenv("SUMMARY_WINDOW_MINUTES", "15")
assert is_recap_run() is True
assert effective_window_minutes() == 24 * 60
def test_build_map_prompt_does_not_inject_futures():
prompt = build_map_prompt(
[{"title": "Blast", "domain": "ex.com", "url": "https://ex.com/1", "content": "x" * 120}]
)
assert "FUTURES PRICES" not in prompt
assert "ES=F" not in prompt
assert "Blast" in prompt
def test_build_master_prompt_interval_uses_summary_not_recap():
prompt = build_master_prompt("partial facts", recap=False)
assert "partial facts" in prompt
assert "daily recap" not in prompt.lower()
def test_build_master_prompt_recap_uses_daily_template():
prompt = build_master_prompt("partial facts", recap=True)
assert "partial facts" in prompt
assert "daily recap" in prompt.lower()
assert "24" in prompt

View file

@ -1,3 +0,0 @@
[pytest]
testpaths = tests
python_files = test_*.py

View file

@ -1,74 +0,0 @@
#!/usr/bin/env bash
# Recreate selected OSINT compose services WITHOUT bouncing Postgres.
#
# The old path was `compose down` + up, which stopped osint-db on every merge
# even when Dockerfile.pg did not change. Name-pinned leftovers are still
# removed, but only for the services we are actually replacing.
#
# Usage: scripts/compose-reup.sh [compose-service ...]
# (default: app ingester camera-service news-scraper news-summarizer)
# Env: COMPOSE_PROJECT_NAME (default osint-dashboard)
# COMPOSE_PROFILES (default ingest)
# FORCE_RECREATE_DB=1 also recreate db
set -euo pipefail
ROOT="$(cd "$(dirname "$0")/.." && pwd)"
cd "$ROOT"
export COMPOSE_PROJECT_NAME="${COMPOSE_PROJECT_NAME:-osint-dashboard}"
PROFILE="${COMPOSE_PROFILES:-ingest}"
DEFAULT_SVCS=(app ingester camera-service news-scraper news-summarizer)
if [ "$#" -gt 0 ]; then
SVCS=("$@")
else
SVCS=("${DEFAULT_SVCS[@]}")
fi
if [ "${FORCE_RECREATE_DB:-0}" = "1" ]; then
SVCS+=(db)
fi
# Never recreate db unless it was requested.
FILTERED=()
for svc in "${SVCS[@]}"; do
if [ "$svc" = "db" ] && [ "${FORCE_RECREATE_DB:-0}" != "1" ]; then
echo "compose-reup: skipping db (set FORCE_RECREATE_DB=1 to bounce Postgres)"
continue
fi
FILTERED+=("$svc")
done
SVCS=("${FILTERED[@]}")
declare -A CONTAINER_NAME=(
[app]=osint-dashboard
[ingester]=osint-ingester
[camera-service]=osint-camera-scraper
[news-scraper]=osint-news-scraper
[news-summarizer]=osint-news-summarizer
[db]=osint-db
[nats]=osint-nats
[titiler]=osint-titiler
)
echo "compose-reup: project=${COMPOSE_PROJECT_NAME} profile=${PROFILE} dir=${ROOT}"
echo "compose-reup: recreate=${SVCS[*]:-none}"
# Keep data-plane containers running (db / nats / titiler).
docker compose --profile "${PROFILE}" up -d --no-build --no-recreate db nats titiler || true
if [ "${#SVCS[@]}" -eq 0 ]; then
docker compose --profile "${PROFILE}" ps
exit 0
fi
for svc in "${SVCS[@]}"; do
c="${CONTAINER_NAME[$svc]:-}"
if [ -n "$c" ] && docker inspect "$c" >/dev/null 2>&1; then
echo "compose-reup: replacing ${c}"
docker rm -f "$c" >/dev/null
fi
done
docker compose --profile "${PROFILE}" up -d --no-build --no-deps "${SVCS[@]}"
docker compose --profile "${PROFILE}" ps

View file

@ -1,10 +0,0 @@
#!/usr/bin/env bash
# Pull OSINT images from Forgejo registry and retag for docker-compose (localhost/*).
set -euo pipefail
REG="${FORGEJO_REGISTRY:-forgejo.siriusdevops.com}"
OWN="${FORGEJO_OWNER:-sirius}"
for name in osint-dashboard osint-dashboard-pg osint-news-scraper osint-news-summarizer; do
docker pull "${REG}/${OWN}/${name}:latest"
docker tag "${REG}/${OWN}/${name}:latest" "localhost/${name}:latest"
echo "ok ${name}"
done

View file

@ -1,57 +0,0 @@
"""Aircraft popup enrichment + emergency/MIL layer contract (static HTML)."""
from __future__ import annotations
from pathlib import Path
ROOT = Path(__file__).resolve().parent.parent
HTML = (ROOT / "app/static/index.html").read_text()
def _fn(name: str, nxt: str) -> str:
return HTML.split(f"function {name}", 1)[1].split(f"function {nxt}", 1)[0]
def test_popup_has_required_adsb_fields_and_photo():
js = _fn("pointPopup", "loadPlanePhoto")
for field in ("callsign", "hex", "registration", "type", "alt", "gs", "squawk"):
assert f"add('{field}'" in js
assert "class=\"ps-photo\"" in js or "class='ps-photo'" in js
assert "wikipedia" not in js.lower()
assert "ceo" not in js.lower()
def test_emergency_badge_and_squawk_codes():
assert "role-badge emergency" in HTML
assert "hdg-emerg" in HTML
assert "EMERG_SQUAWK" in HTML
assert "['7700', '7600', '7500']" in HTML
emerg = HTML.split("function acIsEmergency", 1)[1].split("function acVisible", 1)[0]
assert "EMERG_SQUAWK.has(sq)" in emerg
color = HTML.split("function acColor", 1)[1].split("function connectLiveWs", 1)[0]
assert "acIsEmergency(p)" in color
assert "#ff5d5d" in color
def test_mil_toggle_hidden_until_role_flag_and_never_hits_adsb_lol():
assert 'id="lp-ac-mil-row"' in HTML
assert 'id="lp-ac-mil-on"' in HTML
row = HTML.split('id="lp-ac-mil-row"', 1)[1].split(">", 1)[0]
assert "hidden" in row
on = HTML.split('id="lp-ac-mil-on"', 1)[1].split(">", 1)[0]
assert "checked" not in on
load = HTML.split("async function loadAircraft", 1)[1].split("async function toggleTrains", 1)[0]
assert "/api/aircraft?bbox=" in load
assert "api.adsb.lol" not in load
assert "noteMilSupport" in load
assert "acMilOn" in load
note = HTML.split("function noteMilSupport", 1)[1].split("function acColor", 1)[0]
assert "extra.role" in note
assert "lp-ac-mil-row" in note
assert "hidden = false" in note
def test_planespotters_lazy_photo_still_wired():
assert "function loadPlanePhoto" in HTML
assert "/api/aircraft/photo?" in HTML
assert "map.on('popupopen', (e) => { loadPlanePhoto(e.popup); });" in HTML

View file

@ -16,12 +16,6 @@ async def _get(path: str) -> httpx.Response:
return await client.get(path) return await client.get(path)
async def _post(path: str, payload: dict | None) -> httpx.Response:
transport = httpx.ASGITransport(app=app)
async with httpx.AsyncClient(transport=transport, base_url=BASE) as client:
return await client.post(path, json=payload)
def test_map_layers_includes_overlays(): def test_map_layers_includes_overlays():
body = asyncio.run(_get("/api/map/layers")).json() body = asyncio.run(_get("/api/map/layers")).json()
assert "layers" in body assert "layers" in body
@ -43,67 +37,3 @@ def test_vessels_empty_without_ais_key():
assert resp.status_code == 200 assert resp.status_code == 200
assert resp.json() == [] assert resp.json() == []
assert "max-age" in (resp.headers.get("cache-control") or "").lower() assert "max-age" in (resp.headers.get("cache-control") or "").lower()
def _desired_boxes():
import asyncio as _a
from ais_stream import _take_desired_boxes
return _a.run(_take_desired_boxes())
def test_vessels_subscribe_sets_viewport_box():
assert _desired_boxes() is None
resp = asyncio.run(_post("/api/vessels/subscribe", {"bbox": "-70,40,-60,45"}))
assert resp.status_code == 200
body = resp.json()
assert body["ok"] is True and body["bbox"] == "-70,40,-60,45"
# AISStream corner order: [[lat, lon], [lat, lon]] (southwest, northeast).
assert _desired_boxes() == [[[40.0, -70.0], [45.0, -60.0]]]
def test_vessels_subscribe_empty_resets():
assert asyncio.run(_post("/api/vessels/subscribe", {"bbox": "-70,40,-60,45"})).status_code == 200
resp = asyncio.run(_post("/api/vessels/subscribe", {"bbox": ""}))
assert resp.status_code == 200
assert resp.json()["bbox"] is None
assert _desired_boxes() is None
def test_vessels_subscribe_rejects_bad_bbox():
for bad in ("1,2,3", "a,b,c,d", "20,30,10,40", "0,0,0,200"):
resp = asyncio.run(_post("/api/vessels/subscribe", {"bbox": bad}))
assert resp.status_code == 422, bad
# Explicit null bbox is the "reset to env default" path (still 200).
assert asyncio.run(_post("/api/vessels/subscribe", {"bbox": None})).status_code == 200
def test_aircraft_photo_requires_hex_or_reg():
assert asyncio.run(_get("/api/aircraft/photo")).status_code == 422
def test_aircraft_photo_rejects_bad_hex():
# hex must be exactly 6 hex chars
resp = asyncio.run(_get("/api/aircraft/photo?hex=xyz1234"))
assert resp.status_code == 422
def test_aircraft_photo_returns_photo(monkeypatch):
async def fake(hex_code=None, reg=None):
return {"id": "1", "src": "https://t.plnspttrs.net/a_280.jpg",
"link": "https://www.planespotters.net/photo/1/x", "photographer": "A"}
monkeypatch.setattr("main.fetch_planespotters_photo", fake)
resp = asyncio.run(_get("/api/aircraft/photo?hex=e8027e"))
assert resp.status_code == 200
body = resp.json()
assert body["src"].startswith("https://t.plnspttrs.net/")
assert "max-age" in (resp.headers.get("cache-control") or "").lower()
def test_aircraft_photo_404_when_no_photo(monkeypatch):
async def fake(hex_code=None, reg=None):
return None
monkeypatch.setattr("main.fetch_planespotters_photo", fake)
resp = asyncio.run(_get("/api/aircraft/photo?reg=D-ABCD"))
assert resp.status_code == 404

View file

@ -1,9 +1,9 @@
"""Integration tests for the news pipeline API (GET /api/news + summaries + intel). """Integration tests for the news pipeline API (GET /api/news + summaries).
DB-backed: marked `requires_db` and auto-skip when the test database is DB-backed: marked `requires_db` and auto-skip when the test database is
unreachable (see tests/conftest.py). Seeding writes directly to the shared unreachable (see tests/conftest.py). Seeding writes directly to the shared
`articles` / `article_summaries` / `news_items` tables, exactly as the scraper `articles` / `article_summaries` tables, exactly as the scraper + summarizer
+ summarizer services would. services would.
""" """
from __future__ import annotations from __future__ import annotations
@ -37,9 +37,7 @@ def _truncate() -> None:
async def run(): async def run():
conn = await asyncpg.connect(**_conn_kwargs()) conn = await asyncpg.connect(**_conn_kwargs())
try: try:
await conn.execute( await conn.execute("TRUNCATE articles, article_summaries")
"TRUNCATE articles, article_summaries, news_items CASCADE"
)
finally: finally:
await conn.close() await conn.close()
@ -68,45 +66,14 @@ def _seed_article(title: str, url: str, domain: str, ts: str, content: str = "bo
asyncio.run(run()) asyncio.run(run())
def _seed_summary(text: str, ts: str, model: str | None = None, kind: str | None = None) -> int: def _seed_summary(text: str, ts: str) -> None:
async def run() -> int:
conn = await asyncpg.connect(**_conn_kwargs())
try:
row = await conn.fetchrow(
"INSERT INTO article_summaries (summary_text, batch_timestamp, model, kind) "
"VALUES ($1, $2, $3, $4) RETURNING id",
text, datetime.fromisoformat(ts), model, kind,
)
return int(row["id"])
finally:
await conn.close()
return asyncio.run(run())
def _seed_news_item(
summary_id: int,
kind: str,
headline: str,
importance: str,
*,
location_name: str | None = None,
lat: float | None = None,
lon: float | None = None,
location_confidence: str | None = None,
category: str | None = None,
url: str | None = None,
) -> None:
async def run(): async def run():
conn = await asyncpg.connect(**_conn_kwargs()) conn = await asyncpg.connect(**_conn_kwargs())
try: try:
await conn.execute( await conn.execute(
"INSERT INTO news_items " "INSERT INTO article_summaries (summary_text, batch_timestamp) "
"(summary_id, kind, headline, importance, location_name, " "VALUES ($1, $2)",
" lat, lon, location_confidence, category, url) " text, datetime.fromisoformat(ts),
"VALUES ($1, $2, $3, $4, $5, $6, $7, $8, $9, $10)",
summary_id, kind, headline, importance, location_name,
lat, lon, location_confidence, category, url,
) )
finally: finally:
await conn.close() await conn.close()
@ -164,101 +131,12 @@ def test_api_news_summaries_contract(clean_news):
assert isinstance(body, list) assert isinstance(body, list)
assert len(body) == 1 assert len(body) == 1
s = body[0] s = body[0]
assert set(s.keys()) == {"id", "summary_text", "batch_timestamp", "model", "kind"} assert set(s.keys()) == {"id", "summary_text", "batch_timestamp"}
assert s["summary_text"] == "master summary markdown…" assert s["summary_text"] == "master summary markdown…"
assert s["batch_timestamp"].startswith("2026-08-24T18:05") assert s["batch_timestamp"].startswith("2026-08-24T18:05")
assert s["kind"] is None
@requires_db
def test_api_news_summaries_kind_filter(clean_news):
_seed_summary("interval brief", "2026-08-28T22:05:00+00:00", kind="interval")
_seed_summary("daily recap", "2026-08-28T03:00:00+00:00", kind="daily_recap")
recap = _get("/api/news/summaries?kind=daily_recap").json()
assert len(recap) == 1
assert recap[0]["summary_text"] == "daily recap"
assert recap[0]["kind"] == "daily_recap"
assert _get("/api/news/summaries?kind=nope").status_code == 422
@requires_db @requires_db
def test_api_news_empty(clean_news): def test_api_news_empty(clean_news):
assert _get("/api/news").json() == [] assert _get("/api/news").json() == []
assert _get("/api/news/summaries").json() == [] assert _get("/api/news/summaries").json() == []
assert _get("/api/news/ticker").json() == []
assert _get("/api/news/map").json() == []
TICKER_KEYS = {"id", "headline", "importance", "location_name", "url", "created_at"}
MAP_KEYS = {
"id", "headline", "importance", "location_name", "lat", "lon",
"location_confidence", "category", "url", "created_at",
}
def _seed_flagged_items() -> None:
sid = _seed_summary("batch brief", "2026-08-27T18:05:00+00:00", "Hermes-4.3-36B")
_seed_news_item(
sid, "ticker", "Critical ticker", "critical",
location_name="Kyiv", url="https://example.com/ticker",
)
_seed_news_item(
sid, "ticker", "Low ticker", "low",
location_name="Somewhere", url="https://example.com/low",
)
_seed_news_item(
sid, "map", "Critical map", "critical",
location_name="Taipei", lat=25.03, lon=121.56,
location_confidence="high", category="conflict",
url="https://example.com/map",
)
@requires_db
def test_api_news_ticker_returns_only_flagged(clean_news):
_seed_flagged_items()
resp = _get("/api/news/ticker")
assert resp.status_code == 200
body = resp.json()
assert isinstance(body, list)
assert len(body) == 1
item = body[0]
assert set(item.keys()) == TICKER_KEYS
assert item["headline"] == "Critical ticker"
assert item["importance"] == "critical"
assert item["location_name"] == "Kyiv"
assert item["url"] == "https://example.com/ticker"
@requires_db
def test_api_news_ticker_falls_back_to_lesser_when_nothing_flagged(clean_news):
sid = _seed_summary("quiet brief", "2026-08-27T18:05:00+00:00", "Hermes-4.3-36B")
_seed_news_item(
sid, "ticker", "Shop theft downtown", "low",
location_name="Raleigh", url="https://example.com/theft",
)
resp = _get("/api/news/ticker")
assert resp.status_code == 200
body = resp.json()
assert len(body) == 1
assert body[0]["headline"] == "Shop theft downtown"
assert body[0]["importance"] == "low"
@requires_db
def test_api_news_map_returns_only_flagged_with_coords(clean_news):
_seed_flagged_items()
resp = _get("/api/news/map")
assert resp.status_code == 200
body = resp.json()
assert isinstance(body, list)
assert len(body) == 1
item = body[0]
assert set(item.keys()) == MAP_KEYS
assert item["headline"] == "Critical map"
assert item["importance"] == "critical"
assert item["lat"] == 25.03
assert item["lon"] == 121.56
assert item["location_confidence"] == "high"
assert item["category"] == "conflict"
assert _get("/api/news/map?bbox=1,2,3").status_code == 422

View file

@ -1,127 +0,0 @@
"""GET /api/place — Nominatim reverse proxy (60s cache, 500 keys, 1 req/s)."""
from __future__ import annotations
import asyncio
import httpx
import pytest
from main import app
from place import cache_key, place_cache, slim_place
BASE = "http://test"
SAMPLE = {
"display_name": "Raleigh, Wake County, North Carolina, United States",
"name": "Raleigh",
"osm_type": "relation",
"osm_id": 123,
"address": {
"city": "Raleigh",
"state": "North Carolina",
"country": "United States",
"country_code": "us",
"tourism": "ignore-me",
},
}
class _FakeResp:
def __init__(self, payload, status=200):
self._payload = payload
self.status_code = status
def raise_for_status(self):
if self.status_code >= 400:
req = httpx.Request("GET", "https://nominatim.openstreetmap.org/reverse")
raise httpx.HTTPStatusError(
"upstream", request=req,
response=httpx.Response(self.status_code, request=req),
)
def json(self):
return self._payload
class _FakeNominatim:
calls: list[dict] = []
def __init__(self, *args, **kwargs):
pass
async def __aenter__(self):
return self
async def __aexit__(self, *args):
return False
async def get(self, url, params=None, headers=None):
_FakeNominatim.calls.append({"url": url, "params": params, "headers": headers})
return _FakeResp(SAMPLE)
def _nominatim_client(**kwargs):
return _FakeNominatim()
async def _get(path: str) -> httpx.Response:
transport = httpx.ASGITransport(app=app)
async with httpx.AsyncClient(transport=transport, base_url=BASE) as client:
return await client.get(path)
@pytest.fixture(autouse=True)
def _reset_place(monkeypatch):
place_cache.clear()
_FakeNominatim.calls = []
monkeypatch.setattr("place._http_client", _nominatim_client)
monkeypatch.setattr("place.NOMINATIM_MIN_INTERVAL", 0.0)
monkeypatch.setattr("place._last_req", 0.0)
yield
place_cache.clear()
def test_slim_place_keeps_address_subset():
body = slim_place(35.78, -78.64, SAMPLE)
assert body["display_name"].startswith("Raleigh")
assert body["name"] == "Raleigh"
assert body["address"]["city"] == "Raleigh"
assert "tourism" not in body["address"]
assert body["attribution"].startswith("© OpenStreetMap")
def test_cache_key_quantizes_to_4_decimals():
assert cache_key(35.77961, -78.63821) == cache_key(35.77964, -78.63819)
def test_place_requires_lat_lon():
resp = asyncio.run(_get("/api/place"))
assert resp.status_code == 422
def test_place_rejects_out_of_range():
assert asyncio.run(_get("/api/place?lat=99&lon=0")).status_code == 422
assert asyncio.run(_get("/api/place?lat=0&lon=200")).status_code == 422
def test_place_reverse_and_cache():
r1 = asyncio.run(_get("/api/place?lat=35.7796&lon=-78.6382"))
assert r1.status_code == 200
body = r1.json()
assert body["display_name"].startswith("Raleigh")
assert body["lat"] == pytest.approx(35.7796, abs=0.001)
assert "max-age=60" in (r1.headers.get("cache-control") or "").lower()
assert len(_FakeNominatim.calls) == 1
ua = _FakeNominatim.calls[0]["headers"]["User-Agent"]
assert "osint-dashboard" in ua.lower() or "@" in ua
r2 = asyncio.run(_get("/api/place?lat=35.77961&lon=-78.63821"))
assert r2.status_code == 200
assert len(_FakeNominatim.calls) == 1 # cache hit, same 4-decimal key
def test_place_cache_cap_500():
from cachetools import TTLCache
assert isinstance(place_cache, TTLCache)
assert place_cache.maxsize == 500
assert place_cache.ttl == 60

View file

@ -1,172 +0,0 @@
"""API tests for GET/PUT /api/settings and GET /api/news/models."""
from __future__ import annotations
import asyncio
import os
import asyncpg
import httpx
import pytest
from conftest import requires_db
from main import app
BASE = "http://test"
def _conn_kwargs() -> dict:
return {
"host": os.environ["DB_HOST"],
"port": int(os.environ["DB_PORT"]),
"user": os.environ["DB_USER"],
"password": os.environ["DB_PASSWORD"],
"database": os.environ["DB_NAME"],
}
def _truncate_settings() -> None:
async def run():
conn = await asyncpg.connect(**_conn_kwargs())
try:
await conn.execute("DROP TABLE IF EXISTS app_settings")
finally:
await conn.close()
asyncio.run(run())
@pytest.fixture()
def clean_settings():
import settings_store
settings_store._ensured = False
_truncate_settings()
yield
settings_store._ensured = False
_truncate_settings()
def _request(method: str, path: str, json: dict | None = None) -> httpx.Response:
async def _run() -> httpx.Response:
transport = httpx.ASGITransport(app=app)
async with httpx.AsyncClient(transport=transport, base_url=BASE) as client:
return await client.request(method, path, json=json)
return asyncio.run(_run())
def _get(path: str) -> httpx.Response:
return _request("GET", path)
def _put(path: str, json: dict) -> httpx.Response:
return _request("PUT", path, json=json)
def test_get_news_models_without_key_returns_fallback(monkeypatch):
monkeypatch.delenv("NOUS_API_KEY", raising=False)
monkeypatch.setattr("keystore.get_api_key", _missing_key)
resp = _get("/api/news/models")
assert resp.status_code == 200
body = resp.json()
assert body["source"] == "fallback"
ids = [m["id"] for m in body["models"]]
assert "Hermes-4.3-36B" in ids
from settings_store import FALLBACK_MODELS
assert ids == FALLBACK_MODELS
async def _missing_key(name: str):
return None
async def _present_key(name: str):
return "test-nous-api-key-1234"
def test_get_news_models_live_from_upstream(monkeypatch):
monkeypatch.setattr("keystore.get_api_key", _present_key)
monkeypatch.setenv("NOUS_BASE_URL", "https://inference-api.nousresearch.com/v1")
monkeypatch.delenv("OSINT_USER_AGENT", raising=False)
captured: dict = {}
class FakeResponse:
status_code = 200
def raise_for_status(self):
return None
def json(self):
return {"data": [{"id": "live-model-a"}, {"id": "Hermes-4.3-36B"}]}
async def fake_http_get(url, *, headers, timeout):
captured["url"] = url
captured["headers"] = headers
captured["timeout"] = timeout
return FakeResponse()
import settings_store
monkeypatch.setattr(settings_store, "_http_get", fake_http_get)
settings_store._models_cache = None
resp = _get("/api/news/models")
assert resp.status_code == 200
body = resp.json()
assert body["source"] == "live"
ids = [m["id"] for m in body["models"]]
assert ids == ["live-model-a", "Hermes-4.3-36B"]
assert captured["url"] == "https://inference-api.nousresearch.com/v1/models"
assert captured["headers"]["User-Agent"] == "osint-dashboard-news-summarizer"
assert captured["headers"]["Authorization"] == "Bearer test-nous-api-key-1234"
timeout = captured["timeout"]
assert timeout == 8 or getattr(timeout, "read", timeout) == 8 or float(timeout) == 8.0
def test_get_news_models_upstream_failure_returns_fallback(monkeypatch):
monkeypatch.setattr("keystore.get_api_key", _present_key)
async def boom_http_get(url, *, headers, timeout):
raise httpx.ConnectError("upstream down")
import settings_store
monkeypatch.setattr(settings_store, "_http_get", boom_http_get)
settings_store._models_cache = None
resp = _get("/api/news/models")
assert resp.status_code == 200
body = resp.json()
assert body["source"] == "fallback"
ids = [m["id"] for m in body["models"]]
assert "Hermes-4.3-36B" in ids
def test_put_settings_empty_returns_422():
resp = _put("/api/settings", {"summary_model": ""})
assert resp.status_code == 422
def test_put_settings_whitespace_only_returns_422():
resp = _put("/api/settings", {"summary_model": " "})
assert resp.status_code == 422
@requires_db
def test_put_settings_round_trip(clean_settings, monkeypatch):
monkeypatch.delenv("SUMMARY_MODEL", raising=False)
monkeypatch.delenv("NOUS_BASE_URL", raising=False)
put_resp = _put("/api/settings", {"summary_model": "google/gemini-2.5-flash"})
assert put_resp.status_code == 200
put_body = put_resp.json()
assert put_body["summary_model"] == "google/gemini-2.5-flash"
assert put_body["nous_base_url"] == "https://inference-api.nousresearch.com/v1"
get_resp = _get("/api/settings")
assert get_resp.status_code == 200
get_body = get_resp.json()
assert get_body["summary_model"] == "google/gemini-2.5-flash"
assert get_body["nous_base_url"] == "https://inference-api.nousresearch.com/v1"
assert set(get_body.keys()) == {"summary_model", "nous_base_url"}

View file

@ -1,87 +0,0 @@
"""GET /api/stats HUD counter contract (counts only, small, never 500)."""
from __future__ import annotations
import asyncio
import re
from datetime import timezone
import httpx
from main import app, _stats_counts
BASE = "http://test"
EXPECTED_KEYS = ("aircraft", "vessels", "trains", "cameras",
"fires", "quakes", "alerts", "timestamp")
async def _get(path: str) -> httpx.Response:
transport = httpx.ASGITransport(app=app)
async with httpx.AsyncClient(transport=transport, base_url=BASE) as client:
return await client.get(path)
def test_stats_200_all_keys_present():
resp = asyncio.run(_get("/api/stats"))
assert resp.status_code == 200
body = resp.json()
for key in EXPECTED_KEYS:
assert key in body, f"missing key {key}"
assert "max-age" in (resp.headers.get("cache-control") or "").lower()
def test_stats_counters_are_ints():
body = asyncio.run(_get("/api/stats")).json()
for key in EXPECTED_KEYS:
if key == "timestamp":
continue
assert isinstance(body[key], int), f"{key} is not an int: {body[key]!r}"
def test_stats_timestamp_is_iso8601_z():
body = asyncio.run(_get("/api/stats")).json()
ts = body["timestamp"]
# ISO8601 with a trailing Z (we normalize +00:00 -> Z).
assert isinstance(ts, str) and ts.endswith("Z")
assert re.match(r"^\d{4}-\d{2}-\d{2}T\d{2}:\d{2}:\d{2}", ts)
def test_stats_payload_is_tiny():
resp = asyncio.run(_get("/api/stats"))
assert len(resp.content) < 2048, "stats payload must be counts-only, not GeoJSON"
def test_stats_counts_reflect_last_known(monkeypatch):
"""aircraft/vessels/trains/alerts come from in-memory last-known state."""
import live_layers
monkeypatch.setattr(live_layers, "aircraft_last_known", {str(i): {} for i in range(7)})
monkeypatch.setattr(live_layers, "vessel_last_known", {str(i): {} for i in range(3)})
monkeypatch.setattr(live_layers, "train_count", 11)
monkeypatch.setattr(live_layers, "nws_alert_count", 5)
# _stats_counts imports the dicts/counters inside the function from live_layers,
# so monkeypatching the module attributes is what it observes.
from main import _stats_counts as fn
body = asyncio.run(fn())
assert body["aircraft"] == 7
assert body["vessels"] == 3
assert body["trains"] == 11
assert body["alerts"] == 5
def test_stats_db_failure_degrades_to_zero(monkeypatch):
"""A down DB yields zeros for the SQL-backed counters, never a 500."""
# Make the session factory raise synchronously so the try/except in
# _stats_counts degrades the SQL counters to zero (no dangling coroutine).
def _raise(*args, **kwargs):
raise RuntimeError("db down")
monkeypatch.setattr("main.async_session", _raise)
body = asyncio.run(_stats_counts())
assert body["cameras"] == 0
assert body["fires"] == 0
assert body["quakes"] == 0
assert isinstance(body["timestamp"], str)

View file

@ -1,55 +0,0 @@
"""ffmpeg snapshots stay off the request path (asyncio.create_task)."""
from __future__ import annotations
import asyncio
import bg_jobs
def test_bg_jobs_has_no_pps_cap():
assert not any(name.endswith("_PPS_CAP") for name in dir(bg_jobs))
def test_camera_preview_has_no_public_feed_probe():
import camera_preview
assert not hasattr(camera_preview, "probe_public_feed")
assert not hasattr(camera_preview, "_http_feed_url")
def test_ingest_routes_exclude_active_discovery():
from main import app
ingest = [
getattr(r, "path", "")
for r in app.routes
if getattr(r, "path", "").startswith("/api/ingest/")
]
assert "/api/ingest/fires" in ingest
assert all("scan" not in path for path in ingest)
def test_schedule_ffmpeg_snapshot_is_a_task_not_inline(monkeypatch):
calls = {"n": 0}
async def fake_grab(url, timeout=8.0):
calls["n"] += 1
await asyncio.sleep(5)
return b"\xff\xd8fakejpeg"
monkeypatch.setattr(bg_jobs, "_ffmpeg_grab", fake_grab)
bg_jobs._ffmpeg_tasks.clear()
bg_jobs._ffmpeg_cache.clear()
async def run():
task = bg_jobs.schedule_ffmpeg_snapshot("rtsp://10.0.0.1/")
assert isinstance(task, asyncio.Task)
assert not task.done()
task.cancel()
try:
await task
except (asyncio.CancelledError, Exception):
pass
asyncio.run(run())

View file

@ -1,21 +0,0 @@
"""Timeline bucket_hours + static Cache-Control."""
from __future__ import annotations
from pathlib import Path
ROOT = Path(__file__).resolve().parent.parent
def test_timeline_uses_bucket_hours():
src = (ROOT / "app/main.py").read_text()
fn = src.split("async def get_timeline")[1].split("async def sentiment_by_source")[0]
assert "bucket_hours" in fn
assert "date_trunc('hour'" not in fn or "bucket" in fn.lower()
# Must not ignore the query param.
assert ":bucket" in fn or "bucket_hours" in fn.split("text(")[1][:800]
def test_static_vendor_cache_control():
src = (ROOT / "app/main.py").read_text()
assert "max-age=31536000" in src or "immutable" in src.lower()

View file

@ -1,15 +0,0 @@
"""Camera list is slim (no URLs). Popup must fetch GET /api/cameras/{id}."""
from pathlib import Path
HTML = Path(__file__).resolve().parents[1] / "app/static/index.html"
def test_popupopen_fetches_camera_detail_row():
html = HTML.read_text()
start = html.index("map.on('popupopen'")
end = html.index("map.on('popupclose'")
block = html[start:end]
assert "fetch(" in block
assert "/api/cameras/" in block
assert "camPopupHtml(" in block

View file

@ -1,122 +0,0 @@
"""Tests for the chokepoint preset catalog + vessels ``src=`` filter.
- Span: every catalog box passes VesselAPI's ``|dLat|+|dLon| <= 4`` validator.
- Catalog: ``GET /api/map/chokepoints`` returns 200 with the documented shape.
- Vessels filter: ``GET /api/vessels?src=`` narrows the union store by provider.
"""
from __future__ import annotations
import asyncio
import httpx
from chokepoints import chokepoints
from live_layers import fetch_vessels, vessel_last_known
from main import app
from vesselapi import validate_bbox_span
BASE = "http://test"
async def _get(path: str) -> httpx.Response:
transport = httpx.ASGITransport(app=app)
async with httpx.AsyncClient(transport=transport, base_url=BASE) as client:
return await client.get(path)
# ── Span validation (VesselAPI rule) ──────────────────────────────────────
def test_all_catalog_boxes_within_span() -> None:
for preset in chokepoints():
minlat, minlon, maxlat, maxlon = (float(p) for p in preset["bbox"].split(","))
dlat = abs(maxlat - minlat)
dlon = abs(maxlon - minlon)
assert dlat + dlon <= 4.0, preset["id"]
validate_bbox_span(minlat, minlon, maxlat, maxlon) # no raise
# ── Catalog API contract ──────────────────────────────────────────────────
def test_chokepoints_catalog_shape() -> None:
resp = asyncio.run(_get("/api/map/chokepoints"))
assert resp.status_code == 200
body = resp.json()
assert set(body) == {"chokepoints"}
rows = body["chokepoints"]
assert [r["id"] for r in rows] == [
"hormuz", "bab_el_mandeb", "suez", "malacca", "taiwan",
]
for r in rows:
assert set(r) == {"id", "title", "bbox", "center", "zoom", "vesselapi"}
assert isinstance(r["center"], list) and len(r["center"]) == 2
assert r["zoom"] == 9
assert isinstance(r["vesselapi"], bool)
# bbox is minlat,minlon,maxlat,maxlon
minlat, minlon, maxlat, maxlon = (float(p) for p in r["bbox"].split(","))
assert minlat < maxlat and minlon < maxlon
def test_only_hormuz_is_vesselapi() -> None:
rows = chokepoints()
by_id = {r["id"]: r for r in rows}
assert by_id["hormuz"]["vesselapi"] is True
for cid in ("bab_el_mandeb", "suez", "malacca", "taiwan"):
assert by_id[cid]["vesselapi"] is False
# ── Vessels src= filter (mocked store) ────────────────────────────────────
def _seed_store() -> None:
vessel_last_known.clear()
vessel_last_known["422050100"] = {
"id": "422050100", "lat": 26.5, "lon": 56.3, "label": "HORMUZ STAR",
"extra": {"src": "vesselapi", "mmsi": "422050100"},
}
vessel_last_known["366001230"] = {
"id": "366001230", "lat": 35.0, "lon": -79.0, "label": "CONUS SHIP",
"extra": {"src": "aisstream", "mmsi": "366001230"},
}
vessel_last_known["366001231"] = {
"id": "366001231", "lat": 36.0, "lon": -78.0, "label": "CONUS SHIP 2",
"extra": {"src": "aisstream", "mmsi": "366001231"},
}
def test_fetch_vessels_src_filters() -> None:
_seed_store()
assert {v["id"] for v in asyncio.run(fetch_vessels(None, src="vesselapi"))} == {"422050100"}
assert {v["id"] for v in asyncio.run(fetch_vessels(None, src="aisstream"))} == {
"366001230", "366001231",
}
assert len(asyncio.run(fetch_vessels(None, src="all"))) == 3
assert len(asyncio.run(fetch_vessels(None))) == 3 # default all
def test_vessels_src_query_param(monkeypatch) -> None:
_seed_store()
async def _fake_fetch(bbox, limit, src=None):
rows = [
{"id": k, **{kk: v[kk] for kk in ("lat", "lon", "label", "extra")}}
for k, v in vessel_last_known.items()
]
if src and src != "all":
rows = [r for r in rows if (r.get("extra") or {}).get("src") == src]
return rows
monkeypatch.setattr("main.fetch_vessels", _fake_fetch)
body = asyncio.run(_get("/api/vessels?src=vesselapi")).json()
assert [r["id"] for r in body] == ["422050100"]
body = asyncio.run(_get("/api/vessels?src=aisstream")).json()
assert {r["id"] for r in body} == {"366001230", "366001231"}
body = asyncio.run(_get("/api/vessels?src=all")).json()
assert len(body) == 3
def test_vessels_src_rejects_bad_value() -> None:
resp = asyncio.run(_get("/api/vessels?src=marine-traffic"))
assert resp.status_code == 422

View file

@ -1,132 +0,0 @@
"""GET /api/conflicts — curated conflict-zone catalog + event-count roll-up.
No outbound HTTP: event counts come from geocoded rows already (or not) in the
DB, and the API tests monkeypatch ``main._fetch_geocoded_points`` so no database
is required for the contract checks.
"""
from datetime import datetime, timezone
import httpx
from conflicts import SEVERITIES, conflict_zones, zone_event_stats
from live_layers import overlay_catalog
from main import app
BASE = "http://test"
def _get(path: str, monkeypatch=None, points=None) -> httpx.Response:
import asyncio
async def run() -> httpx.Response:
if monkeypatch is not None:
async def fake():
return points or []
monkeypatch.setattr("main._fetch_geocoded_points", fake)
transport = httpx.ASGITransport(app=app)
async with httpx.AsyncClient(transport=transport, base_url=BASE) as client:
return await client.get(path)
return asyncio.run(run())
# ── Catalog shape ──────────────────────────────────────────────────────
def test_catalog_length():
zones = conflict_zones()
assert len(zones) == 13
def test_catalog_severity_enum():
zones = conflict_zones()
sevs = {z["severity"] for z in zones}
assert sevs.issubset(SEVERITIES)
# All three tiers are represented.
assert sevs == SEVERITIES
def test_catalog_fields_factual_and_complete():
zones = conflict_zones()
ids = [z["id"] for z in zones]
assert len(set(ids)) == len(ids) # unique ids
for z in zones:
assert z["label"]
assert z["description"].strip()
assert -90.0 <= z["lat"] <= 90.0
assert -180.0 <= z["lon"] <= 180.0
# internal-only bbox is well-formed: (min_lat, min_lon, max_lat, max_lon)
min_lat, min_lon, max_lat, max_lon = z["bbox"]
assert min_lat <= max_lat and min_lon <= max_lon
assert min_lat <= z["lat"] <= max_lat and min_lon <= z["lon"] <= max_lon
def test_overlay_catalog_has_conflicts():
entry = overlay_catalog()["conflicts"]
assert entry["kind"] == "points"
assert entry["endpoint"] == "/api/conflicts"
# ── Pure counting ──────────────────────────────────────────────────────
TS1 = datetime(2026, 8, 30, 12, 0, tzinfo=timezone.utc)
TS2 = datetime(2026, 8, 30, 13, 0, tzinfo=timezone.utc)
def test_zone_event_stats_counts_and_picks_latest():
bbox = (40.0, 20.0, 52.0, 40.0) # roughly Ukraine
points = [
(50.45, 30.52, TS1), # inside
(48.0, 25.0, TS2), # inside, later
(0.0, -60.0, TS1), # outside
(15.0, 45.0, TS2), # outside (lat ok, lon out)
]
count, latest = zone_event_stats(points, bbox)
assert count == 2
assert latest == TS2
def test_zone_event_stats_empty_bbox():
count, latest = zone_event_stats([], (0.0, 0.0, 1.0, 1.0))
assert count == 0
assert latest is None
# ── API contract (mocked map items, no DB) ─────────────────────────────
def test_conflicts_returns_catalog_with_mocked_counts(monkeypatch):
points = [
(50.45, 30.52, TS1), # Ukraine
(25.03, 121.56, TS2), # Taiwan Strait
(0.0, -60.0, TS1), # nowhere
]
resp = _get("/api/conflicts", monkeypatch=monkeypatch, points=points)
assert resp.status_code == 200
body = resp.json()
assert "zones" in body and "timestamp" in body
by_id = {z["id"]: z for z in body["zones"]}
assert len(body["zones"]) == 13
zone = by_id["ukraine"]
assert zone["eventCount"] == 1
assert zone["lastUpdated"] == TS1.isoformat().replace("+00:00", "Z")
assert zone["severity"] == "war"
assert by_id["taiwan_strait"]["eventCount"] == 1
assert by_id["gaza"]["eventCount"] == 0
# exact per-zone key contract the frontend consumes
assert set(zone.keys()) == {
"id", "label", "severity", "lat", "lon",
"description", "eventCount", "lastUpdated",
}
def test_conflicts_empty_db_yields_zero_counts(monkeypatch):
resp = _get("/api/conflicts", monkeypatch=monkeypatch, points=[])
assert resp.status_code == 200
body = resp.json()
assert all(z["eventCount"] == 0 for z in body["zones"])
assert all(z["lastUpdated"] is None for z in body["zones"])

View file

@ -1,66 +0,0 @@
"""Conflicts Leaflet overlay: default-off toggle, catalog fetch, no jitter."""
from __future__ import annotations
from pathlib import Path
ROOT = Path(__file__).resolve().parent.parent
HTML = (ROOT / "app/static/index.html").read_text()
def _fn(name: str, until: str | None = None) -> str:
chunk = HTML.split(f"function {name}", 1)[1]
if until:
chunk = chunk.split(until, 1)[0]
return chunk
def test_conflicts_toggle_default_off():
assert 'id="lp-conflicts-on"' in HTML
assert 'id="conflicts-layer"' in HTML
assert "> Conflicts<" in HTML or "> Conflicts</" in HTML
on = HTML.split('id="lp-conflicts-on"', 1)[1].split(">", 1)[0]
assert "checked" not in on
def test_conflicts_fetches_catalog_not_liveuamap():
js = _fn("loadConflicts", "/* ═══════════════ INITIAL LOAD")
assert "/api/conflicts" in js
assert "liveuamap.com" not in HTML.lower()
assert "Math.random" not in js
assert "jitter" not in js.lower()
def test_conflicts_not_refetched_on_moveend():
refresh = HTML.split("function refreshLiveOverlays", 1)[1].split(
"function addExtraAttrib", 1
)[0]
assert "loadConflicts" not in refresh
assert "probeConflicts" not in refresh
init = HTML.split("function initMap", 1)[1].split("function readMapPrefs", 1)[0]
assert "probeConflicts()" in init
assert "loadConflicts(true)" not in init
assert "paintConflicts()" not in init
def test_conflicts_hides_toggle_on_404():
js = _fn("loadConflicts", "/* ═══════════════ INITIAL LOAD")
assert "r.status === 404" in js
assert "hideConflictsToggle()" in js
hide = _fn("hideConflictsToggle", "function paintConflicts")
assert "row.hidden = true" in hide
assert "lp-conflicts-on" in hide
def test_conflicts_popup_and_severity_colors():
paint = _fn("paintConflicts", "async function probeConflicts")
assert "z.label" in paint
assert "z.description" in paint
assert "eventCount" in paint
assert "L.circleMarker" in paint
assert "z.lat == null || z.lon == null" in paint
assert "Number.isFinite(lat)" in paint
color = _fn("conflictSeverityColor", "function hideConflictsToggle")
assert "war" in color and "#ff2a6d" in color
assert "high" in color and "#fb923c" in color
assert "elevated" in color and "#facc15" in color

View file

@ -1,133 +0,0 @@
"""Generic event ingest: idempotency, USGS ids, GDELT DOC URL."""
from __future__ import annotations
import asyncio
from datetime import datetime, timezone
def test_event_dedup_key_prefers_url():
from sources import event_dedup_key
assert event_dedup_key({"url": "https://earthquake.usgs.gov/earthquakes/eventpage/ci1"}) == (
"https://earthquake.usgs.gov/earthquakes/eventpage/ci1"
)
assert event_dedup_key({"url": " "}) is None
assert event_dedup_key({}) is None
def test_usgs_feature_keeps_id_and_url():
from sources import parse_usgs_feature
feature = {
"id": "ci39818991",
"properties": {
"title": "M 2.1 - 5 km W of",
"url": "https://earthquake.usgs.gov/earthquakes/eventpage/ci39818991",
"place": "5 km W of",
"mag": 2.1,
"time": 1_700_000_000_000,
},
"geometry": {"coordinates": [-118.5, 34.1, 10.0]},
}
event = parse_usgs_feature(feature)
assert event["url"] == "https://earthquake.usgs.gov/earthquakes/eventpage/ci39818991"
assert event["raw"]["usgs_id"] == "ci39818991"
assert event["location_lat"] == 34.1
assert event["location_lon"] == -118.5
assert event["source_type"] == "earthquake"
def test_gdelt_uses_doc_api_and_query_param():
from sources import GDELT_API, gdelt_params
assert GDELT_API == "https://api.gdeltproject.org/api/v2/doc/doc"
params = gdelt_params(query="unrest", max_articles=50)
assert params["query"] == "unrest"
assert "search" not in params
assert params["mode"] == "ArtList"
assert params["format"] == "json"
assert int(params["maxrecords"]) == 50
def test_gdelt_default_query_when_empty():
from sources import gdelt_params
params = gdelt_params(query="", max_articles=25)
assert params["query"]
assert "unrest" in params["query"].lower() or "cyber" in params["query"].lower()
def test_parse_gdelt_articles_maps_doc_payload():
from sources import parse_gdelt_articles
payload = {
"articles": [
{
"url": "https://example.com/a",
"title": "Outage",
"seendate": "20240101T120000Z",
"domain": "example.com",
"language": "English",
"sourcecountry": "US",
}
]
}
events = parse_gdelt_articles(payload)
assert len(events) == 1
assert events[0]["source_type"] == "gdel-t2"
assert events[0]["url"] == "https://example.com/a"
assert events[0]["title"] == "Outage"
def test_ingest_event_skips_duplicate_url(monkeypatch):
"""Second insert with the same url must not hit events_table.insert."""
from ingestor import ingest_event
calls = {"insert": 0, "dedup": 0}
class _Result:
rowcount = 1
inserted_primary_key = ["evt-1"]
class _Session:
async def execute(self, stmt):
sql = str(stmt).lower()
if "event_dedup" in sql or "on conflict" in sql:
calls["dedup"] += 1
self_result = _Result()
if calls["dedup"] > 1:
self_result.rowcount = 0
return self_result
calls["insert"] += 1
return _Result()
async def commit(self):
return None
async def __aenter__(self):
return self
async def __aexit__(self, *a):
return False
import ingestor
monkeypatch.setattr(ingestor, "async_session", lambda: _Session())
msg = {
"source_type": "earthquake",
"title": "M 2.1",
"url": "https://earthquake.usgs.gov/earthquakes/eventpage/ci1",
"source_timestamp": datetime(2026, 1, 1, tzinfo=timezone.utc).isoformat(),
}
async def run():
first = await ingest_event(msg)
second = await ingest_event(msg)
return first, second
first, second = asyncio.run(run())
assert first is not None
assert second is None
assert calls["insert"] == 1

View file

@ -1,66 +0,0 @@
"""WFIGS/FIRMS × firefighting ADS-B correlation within 20 miles."""
from __future__ import annotations
from fire_aircraft import (
FIREFIGHTER_ICAO,
RADIUS_MILES,
correlate_aircraft_to_fires,
is_firefighter,
)
def _ac(hex_id, lat, lon, icao, **extra):
return {
"id": hex_id,
"lat": lat,
"lon": lon,
"label": hex_id,
"heading": 0,
"speed": 120,
"extra": {"type": icao, "hex": hex_id, **extra},
}
def _fire(name, lat, lon, **extra):
return {
"id": name,
"lat": lat,
"lon": lon,
"label": name,
"extra": extra,
}
def test_air_tractor_is_firefighter_airliner_is_not():
assert is_firefighter(_ac("aaa", 0, 0, "AT802")) is True
assert is_firefighter(_ac("bbb", 0, 0, "C130")) is True
assert is_firefighter(_ac("ccc", 0, 0, "B738")) is False
assert "AT802" in FIREFIGHTER_ICAO
def test_flags_tanker_within_20_miles_of_fire():
# ~10 miles north of a Piedmont fire
fire = _fire("Jones Gap", 35.00, -82.00)
tanker = _ac("acf001", 35.145, -82.00, "AT802")
airliner = _ac("a0b738", 35.145, -82.00, "B738")
far = _ac("acfar", 35.50, -82.00, "C130") # ~34 miles
hits = correlate_aircraft_to_fires([fire], [tanker, airliner, far])
assert RADIUS_MILES == 20.0
assert len(hits) == 1
h = hits[0]
assert h["aircraft_hex"] == "acf001"
assert h["fire_id"] == "Jones Gap"
assert h["aircraft_type"] == "AT802"
assert 0 < h["distance_mi"] <= 20.0
def test_persist_shape_has_reload_fields():
fire = _fire("Jones Gap", 35.00, -82.00, src="wfigs")
tanker = _ac("acf001", 35.10, -82.00, "S64")
hits = correlate_aircraft_to_fires([fire], [tanker])
assert set(hits[0]).issuperset({
"fire_id", "fire_lat", "fire_lon",
"aircraft_hex", "aircraft_type", "aircraft_lat", "aircraft_lon",
"distance_mi",
})

View file

@ -85,183 +85,3 @@ def test_ingest_fire_row_drops_malformed(clean_fires):
assert await ingest_fire_row(make_fire_msg(acq_time="garbage")) is False assert await ingest_fire_row(make_fire_msg(acq_time="garbage")) is False
asyncio.run(run()) asyncio.run(run())
class _InsertSession:
"""async_session stand-in: fire insert succeeds (rowcount=1)."""
def __init__(self):
self.rowcount = 1
async def execute(self, *a, **k):
return self
async def commit(self):
return None
async def __aenter__(self):
return self
async def __aexit__(self, *a):
return False
def test_ingest_fire_row_geofence_when_cache_empty(monkeypatch):
"""Ingester process has empty geofence cache; still notify on FIRMS insert."""
import geofence
import ingestor
import live_layers
geofence._cache.clear()
live_layers.aircraft_last_known.clear()
notified = []
async def fake_record(**kw):
notified.append(kw)
return 1
async def no_markers(*a, **k):
return []
monkeypatch.setattr(ingestor, "async_session", _InsertSession)
monkeypatch.setattr(geofence, "record_and_notify", fake_record)
monkeypatch.setattr("tracks.recent_markers", no_markers)
assert asyncio.run(ingest_fire_row(make_fire_msg())) is True
assert len(notified) == 1
assert notified[0]["source_kind"] == "firms"
assert notified[0]["lat"] == 39.45678
assert notified[0]["lon"] == -121.12345
def test_ingest_fire_row_correlates_from_hypertable_when_last_known_empty(monkeypatch):
"""FIRMS ingester has no ADS-B last-known; still correlate from aircraft_positions."""
import geofence
import ingestor
import live_layers
geofence._cache.clear()
live_layers.aircraft_last_known.clear()
correlated = []
async def fake_record(**kw):
return 0
async def fake_recent(kind, limit=2000):
assert kind == "aircraft"
return [{
"id": "acf001",
"lat": 39.45,
"lon": -121.12,
"extra": {"type": "AT802"},
}]
async def fake_corr(fires, aircraft):
correlated.append((fires, aircraft))
return aircraft
monkeypatch.setattr(ingestor, "async_session", _InsertSession)
monkeypatch.setattr(geofence, "record_and_notify", fake_record)
monkeypatch.setattr("tracks.recent_markers", fake_recent)
monkeypatch.setattr("fire_aircraft.correlate_and_notify", fake_corr)
assert asyncio.run(ingest_fire_row(make_fire_msg())) is True
assert len(correlated) == 1
assert correlated[0][1][0]["id"] == "acf001"
assert correlated[0][0][0]["lat"] == 39.45678
def test_ingest_fire_rows_one_execute_one_commit(monkeypatch):
"""93k FIRMS points must not be 93k commits."""
from ingestor import ingest_fire_rows
class _Session:
def __init__(self):
self.executes = 0
self.commits = 0
self.rowcount = 3
async def execute(self, *a, **k):
self.executes += 1
return self
async def commit(self):
self.commits += 1
async def __aenter__(self):
return self
async def __aexit__(self, *a):
return False
session = _Session()
import ingestor
monkeypatch.setattr(ingestor, "async_session", lambda: session)
msgs = [
make_fire_msg(latitude=39.1 + i * 0.01, longitude=-121.1)
for i in range(3)
]
async def no_corr(*a, **k):
return []
monkeypatch.setattr("fire_aircraft.correlate_and_notify", no_corr)
monkeypatch.setattr("geofence.record_and_notify", no_corr)
inserted = asyncio.run(ingest_fire_rows(msgs))
assert inserted == 3
assert session.executes == 1
assert session.commits == 1
def test_fire_insert_chunk_stays_under_asyncpg_bind_limit():
"""asyncpg caps bind params at 32767 — a 93k-row INSERT dies."""
from ingestor import FIRE_INSERT_CHUNK, FIRE_ROW_BIND_PARAMS
assert FIRE_INSERT_CHUNK * FIRE_ROW_BIND_PARAMS < 32767
assert FIRE_INSERT_CHUNK >= 500
def test_ingest_fire_rows_chunks_when_over_limit(monkeypatch):
from ingestor import ingest_fire_rows
import ingestor
class _Session:
def __init__(self):
self.executes = 0
self.commits = 0
self.rowcount = 0
async def execute(self, *a, **k):
self.executes += 1
self.rowcount = 2 if self.executes < 3 else 1
return self
async def commit(self):
self.commits += 1
async def __aenter__(self):
return self
async def __aexit__(self, *a):
return False
session = _Session()
monkeypatch.setattr(ingestor, "async_session", lambda: session)
monkeypatch.setattr(ingestor, "FIRE_INSERT_CHUNK", 2)
async def no_corr(*a, **k):
return []
monkeypatch.setattr("fire_aircraft.correlate_and_notify", no_corr)
monkeypatch.setattr("geofence.record_and_notify", no_corr)
msgs = [
make_fire_msg(latitude=39.1 + i * 0.01, longitude=-121.1)
for i in range(5)
]
inserted = asyncio.run(ingest_fire_rows(msgs))
assert inserted == 5
assert session.executes == 3
assert session.commits == 1

View file

@ -83,8 +83,6 @@ def test_ingest_fires_idles_without_map_key(monkeypatch):
def test_ingest_fires_uses_keystore_key(monkeypatch): def test_ingest_fires_uses_keystore_key(monkeypatch):
# Key saved via the dashboard Keys page (Postgres) is picked up. # Key saved via the dashboard Keys page (Postgres) is picked up.
from upstream_cache import firms_cache
firms_cache.clear()
monkeypatch.setenv("FIRMS_MAP_KEY", "") monkeypatch.setenv("FIRMS_MAP_KEY", "")
monkeypatch.setattr( monkeypatch.setattr(
"fire_sources.get_api_key", "fire_sources.get_api_key",
@ -123,7 +121,7 @@ def test_ingest_fires_uses_keystore_key(monkeypatch):
published.extend(points) published.extend(points)
return len(points) return len(points)
monkeypatch.setattr("fire_sources.persist_hotspots", fake_publish) monkeypatch.setattr("fire_sources.publish_fire_batch", fake_publish)
assert asyncio.run(ingest_fires()) == 10 # NOAA-20 + NOAA-21 dual-write assert asyncio.run(ingest_fires()) == 10 # NOAA-20 + NOAA-21 dual-write
assert "a" * 32 in captured["url"] assert "a" * 32 in captured["url"]
@ -131,100 +129,6 @@ def test_ingest_fires_uses_keystore_key(monkeypatch):
assert any("VIIRS_NOAA21_NRT" in u for u in captured["urls"]) assert any("VIIRS_NOAA21_NRT" in u for u in captured["urls"])
def _reset_firms_poll_state():
from upstream_cache import firms_cache
import fire_sources
firms_cache.clear()
if hasattr(fire_sources, "_csv_digest"):
fire_sources._csv_digest.clear()
if hasattr(fire_sources, "_seen_ids"):
fire_sources._seen_ids.clear()
def _fake_firms_http(monkeypatch, bodies_by_call: list[str] | None = None, body: str = SAMPLE_CSV):
hits = {"n": 0}
class FakeResp:
def __init__(self, text):
self.text = text
def raise_for_status(self):
pass
class FakeClient:
def __init__(self, **kw):
pass
async def __aenter__(self):
return self
async def __aexit__(self, *exc):
return False
async def get(self, url):
idx = hits["n"]
hits["n"] += 1
if bodies_by_call is not None:
text = bodies_by_call[min(idx, len(bodies_by_call) - 1)]
else:
text = body
return FakeResp(text)
monkeypatch.setenv("FIRMS_MAP_KEY", "k" * 32)
monkeypatch.setattr("fire_sources.FIRMS_DATASETS", ["VIIRS_NOAA20_NRT"])
monkeypatch.setattr("fire_sources.httpx.AsyncClient", FakeClient)
return hits
def test_ingest_fires_skips_unchanged_csv(monkeypatch):
"""Same FIRMS CSV must not be re-parsed into a 100k-row ON CONFLICT insert."""
_reset_firms_poll_state()
hits = _fake_firms_http(monkeypatch)
persisted = []
async def fake_persist(points):
persisted.append(len(points))
return len(points)
monkeypatch.setattr("fire_sources.persist_hotspots", fake_persist)
assert asyncio.run(ingest_fires()) == 5
assert persisted == [5]
firms_cache_hits = hits["n"]
persisted.clear()
assert asyncio.run(ingest_fires()) == 0
assert persisted == []
# TTL cache may skip HTTP; either way we must not persist again.
assert hits["n"] >= firms_cache_hits
def test_ingest_fires_persists_only_new_hotspots(monkeypatch):
"""When the CSV grows, persist the delta — not the whole 2-day dump."""
_reset_firms_poll_state()
extra = (
SAMPLE_CSV
+ "16.00000,-12.00000,340.00,0.40,0.40,2025-06-06,1500,N20,VIIRS,h,2.0NRT,310.00,8.00,D\n"
)
hits = _fake_firms_http(monkeypatch, bodies_by_call=[SAMPLE_CSV, extra])
persisted = []
async def fake_persist(points):
persisted.append([p["latitude"] for p in points])
return len(points)
monkeypatch.setattr("fire_sources.persist_hotspots", fake_persist)
from upstream_cache import firms_cache
assert asyncio.run(ingest_fires()) == 5
firms_cache.clear() # force the next poll to see the grown CSV
persisted.clear()
assert asyncio.run(ingest_fires()) == 1
assert persisted == [[16.0]]
assert hits["n"] == 2
def _async_return(value): def _async_return(value):
async def inner(): async def inner():
return value return value

View file

@ -1,86 +0,0 @@
"""HUD load-time: no market 404 poll, deferred overlays, WS backoff, nginx snippet."""
from __future__ import annotations
from pathlib import Path
ROOT = Path(__file__).resolve().parent.parent
HTML = (ROOT / "app/static/index.html").read_text()
def test_summarizer_dockerfile_copies_intel_modules():
df = (ROOT / "news/summerizer/Dockerfile").read_text()
assert "intel.py" in df
assert "nous_client.py" in df
def test_news_panel_pins_daily_recap():
assert "kind=daily_recap" in HTML
assert "DAILY RECAP" in HTML
def test_nginx_ws_snippet_has_upgrade_headers():
conf = (ROOT / "deploy/osint-ws.nginx.conf").read_text()
assert "proxy_http_version 1.1" in conf
assert "Upgrade" in conf
assert "Connection" in conf
assert "/ws/" in conf
def test_market_ticker_does_not_poll_unwired_endpoint():
assert "setInterval(probeMarket" not in HTML
assert "initMarketTicker()" not in HTML or "probeMarket();" not in HTML.split("function initMarketTicker")[1][:400]
def test_startup_defers_nonessential_overlays():
init = HTML.split("function initMap")[1].split("function readMapPrefs")[0]
# Must not fire all four DB loads + live overlays in the same tick.
assert "setTimeout" in init or "requestAnimationFrame" in init
def test_ws_reconnect_uses_backoff():
assert "setTimeout(connectLiveWs, 4000)" not in HTML
ws = HTML.split("function connectLiveWs")[1][:1200]
assert "backoff" in ws.lower() or "wsRetry" in ws or "wsDelay" in ws
def test_check_health_treats_degraded_status():
fn = HTML.split("async function checkHealth")[1].split("/* ═══════════════ NAV")[0]
assert "degraded" in fn.lower() or "d.status" in fn
def test_chokepoint_presets_in_toolbar():
assert 'id="chokepoint-btns"' in HTML
assert 'id="chokepoint-select"' in HTML
assert "loadChokepoints()" in HTML
assert "/api/map/chokepoints" in HTML
assert "function applyChokepoint" in HTML
for name in ("Hormuz", "Bab el-Mandeb", "Suez", "Malacca", "Taiwan"):
assert name in HTML
def test_chokepoint_skips_aisstream_subscribe_outside_conus():
load = HTML.split("async function loadVessels")[1].split("async function toggleStorms")[0]
assert "intersectsConus()" in load
assert "api/vessels/subscribe" in load
assert "src=${encodeURIComponent(vesselSrcPref)}" in load or "&src=" in load
apply = HTML.split("function applyChokepoint")[1].split("function currentBBox")[0]
assert "vesselapi" in apply
assert "lp-vessels-on" in apply
assert "lp-sentinel-on" in apply
assert "map.setView" in apply
assert "minlat,minlon,maxlat,maxlon" in HTML.split("function chokepointLeafletBounds")[1][:400]
def test_news_ticker_polls_more_often_than_summarizer_cycle():
assert "NEWS_REFRESH_MS" in HTML
# Summarizer is 15 min; ticker should refresh on a shorter cadence so
# lesser-news fills show up without waiting for the next brief.
line = [ln for ln in HTML.splitlines() if "NEWS_REFRESH_MS" in ln][0]
assert "900000" not in line
def test_phone_chokepoints_use_select_not_buttons():
mobile = HTML.split("@media (max-width: 820px)")[1].split("@media (prefers-reduced-motion")[0]
assert "#chokepoint-select { display: block; }" in mobile
assert ".chokepoint-btns { display: none; }" in mobile or "#chokepoint-label, .chokepoint-btns { display: none; }" in mobile

View file

@ -1,363 +0,0 @@
"""Geofence hit detection and WS alert routing (no Redis)."""
from __future__ import annotations
import asyncio
import geofence
from geofence import (
matching_geofences,
point_in_geojson,
validate_polygon_geojson,
)
from ws_manager import ConnectionManager
NC_BOX = {
"type": "Polygon",
"coordinates": [[
[-80.0, 35.0],
[-78.0, 35.0],
[-78.0, 36.0],
[-80.0, 36.0],
[-80.0, 35.0],
]],
}
def test_point_inside_polygon_hits():
assert point_in_geojson(-79.0, 35.5, NC_BOX) is True
def test_point_outside_polygon_misses():
assert point_in_geojson(-122.4, 37.7, NC_BOX) is False
def test_validate_rejects_non_polygon():
try:
validate_polygon_geojson({"type": "Point", "coordinates": [-79.0, 35.5]})
assert False, "expected ValueError"
except ValueError:
pass
def test_matching_geofences_only_active_hits():
fences = [
{"id": "a", "name": "NC", "active": True, "geojson": NC_BOX},
{"id": "b", "name": "off", "active": False, "geojson": NC_BOX},
]
hits = matching_geofences(-79.0, 35.5, fences)
assert [h["id"] for h in hits] == ["a"]
assert matching_geofences(-122.4, 37.7, fences) == []
FENCE_ID = "11111111-1111-1111-1111-111111111111"
NC_VIEW = (-80.0, 35.0, -78.0, 36.0)
SF_VIEW = (-123.0, 37.0, -121.0, 38.0)
def _alert_payload(gid=FENCE_ID):
return {
"geofence_id": gid,
"geofence_name": "NC",
"source_kind": "ais",
"entity_id": "366123456",
"lat": 35.5,
"lon": -79.0,
}
def test_geofence_alert_fans_out_only_to_viewport_clients():
mgr = ConnectionManager()
q_nc = mgr.register("nc")
q_sf = mgr.register("sf")
mgr.set_viewport("nc", NC_VIEW)
mgr.set_viewport("sf", SF_VIEW)
async def run():
n = await mgr.publish_point(
"geofence_alert", _alert_payload(), lat=35.5, lon=-79.0,
)
assert n == 1
msg = q_nc.get_nowait()
assert msg["type"] == "geofence_alert"
assert msg["payload"]["entity_id"] == "366123456"
assert q_sf.empty()
asyncio.run(run())
def test_off_viewport_watch_receives_geofence_alert():
mgr = ConnectionManager()
q_sf = mgr.register("sf")
mgr.set_viewport("sf", SF_VIEW)
mgr.set_watched_geofences("sf", [FENCE_ID])
async def run():
n = await mgr.publish_point(
"geofence_alert", _alert_payload(), lat=35.5, lon=-79.0,
)
assert n == 1
msg = q_sf.get_nowait()
assert msg["type"] == "geofence_alert"
assert msg["payload"]["geofence_id"] == FENCE_ID
asyncio.run(run())
def test_off_viewport_without_watch_does_not_receive_geofence_alert():
mgr = ConnectionManager()
q_sf = mgr.register("sf")
mgr.set_viewport("sf", SF_VIEW)
async def run():
n = await mgr.publish_point(
"geofence_alert", _alert_payload(), lat=35.5, lon=-79.0,
)
assert n == 0
assert q_sf.empty()
asyncio.run(run())
def test_on_viewport_receives_geofence_alert_without_watch():
mgr = ConnectionManager()
q_nc = mgr.register("nc")
mgr.set_viewport("nc", NC_VIEW)
async def run():
n = await mgr.publish_point(
"geofence_alert", _alert_payload(), lat=35.5, lon=-79.0,
)
assert n == 1
assert q_nc.get_nowait()["type"] == "geofence_alert"
asyncio.run(run())
def test_ais_stays_viewport_only_even_when_watching():
mgr = ConnectionManager()
q_sf = mgr.register("sf")
mgr.set_viewport("sf", SF_VIEW)
mgr.set_watched_geofences("sf", [FENCE_ID])
async def run():
n = await mgr.publish_point("ais", {"id": "366123456"}, lat=35.5, lon=-79.0)
assert n == 0
assert q_sf.empty()
asyncio.run(run())
def test_invalid_watch_uuids_ignored_empty_list_clears():
mgr = ConnectionManager()
q = mgr.register("sf")
mgr.set_viewport("sf", SF_VIEW)
mgr.set_watched_geofences("sf", ["not-a-uuid", FENCE_ID, "also-bad"])
async def run():
n = await mgr.publish_point(
"geofence_alert", _alert_payload(), lat=35.5, lon=-79.0,
)
assert n == 1
q.get_nowait()
mgr.set_watched_geofences("sf", [])
n2 = await mgr.publish_point(
"geofence_alert", _alert_payload(), lat=35.5, lon=-79.0,
)
assert n2 == 0
assert q.empty()
asyncio.run(run())
def test_unregister_clears_watched_geofences():
mgr = ConnectionManager()
q = mgr.register("sf")
mgr.set_viewport("sf", SF_VIEW)
mgr.set_watched_geofences("sf", [FENCE_ID])
mgr.unregister("sf")
q2 = mgr.register("sf")
mgr.set_viewport("sf", SF_VIEW)
async def run():
n = await mgr.publish_point(
"geofence_alert", _alert_payload(), lat=35.5, lon=-79.0,
)
assert n == 0
assert q2.empty()
asyncio.run(run())
def test_record_and_notify_queries_postgis_when_cache_empty(monkeypatch):
"""FIRMS ingest in the ingester has an empty in-process cache — still ST_Intersects."""
import geofence
geofence._cache.clear()
geofence._recent_hits.clear()
st_called = []
async def fake_st(lon, lat):
st_called.append((lon, lat))
return [{
"id": "11111111-1111-1111-1111-111111111111",
"name": "NC",
"geojson": NC_BOX,
"active": True,
}]
monkeypatch.setattr(geofence, "st_intersects", fake_st)
executed: list = []
class FakeSession:
async def execute(self, stmt, params=None):
executed.append(params or {})
return None
async def commit(self):
executed.append("commit")
async def __aenter__(self):
return self
async def __aexit__(self, *a):
return False
monkeypatch.setattr(geofence, "async_session", FakeSession)
async def run():
from ws_manager import manager
q = manager.register("nc")
manager.set_viewport("nc", (-80.0, 35.0, -78.0, 36.0))
n = await geofence.record_and_notify(
source_kind="firms", entity_id="35.5,-79.0,N",
lat=35.5, lon=-79.0, payload={"satellite": "N"},
)
msg = None if q.empty() else q.get_nowait()
manager.unregister("nc")
return n, msg
n, msg = asyncio.run(run())
assert st_called == [(-79.0, 35.5)]
assert n == 1
assert msg["type"] == "geofence_alert"
assert msg["payload"]["source_kind"] == "firms"
inserts = [p for p in executed if isinstance(p, dict)]
assert inserts and inserts[0]["source_kind"] == "firms"
assert "commit" in executed
def test_list_alerts_sql_filters(monkeypatch):
captured: dict = {}
class FakeResult:
def mappings(self):
return self
def all(self):
return []
class FakeSession:
async def execute(self, stmt, params=None):
captured["sql"] = str(stmt)
captured["params"] = params
return FakeResult()
async def __aenter__(self):
return self
async def __aexit__(self, *a):
return False
monkeypatch.setattr(geofence, "async_session", FakeSession)
from datetime import datetime, timezone
since = datetime(2026, 8, 28, tzinfo=timezone.utc)
until = datetime(2026, 8, 29, tzinfo=timezone.utc)
async def run():
return await geofence.list_alerts(
geofence_id=FENCE_ID, since=since, until=until,
source_kind="firms", limit=5,
)
assert asyncio.run(run()) == []
sql = captured["sql"].lower()
assert "geofence_id" in sql
assert "created_at >=" in sql
assert "created_at <=" in sql
assert "source_kind" in sql
assert captured["params"]["geofence_id"] == FENCE_ID
assert captured["params"]["source_kind"] == "firms"
assert captured["params"]["limit"] == 5
def test_alembic_fence_created_index_exists():
from pathlib import Path
text = Path(__file__).resolve().parent.parent.joinpath(
"alembic/versions/011_geofence_alerts_fence.py",
).read_text()
assert "ix_geofence_alerts_fence_created" in text
assert "010_bbox_gist" in text
def test_snapshot_at_404_when_fence_missing(monkeypatch):
geofence._cache.clear()
async def boom():
raise RuntimeError("db down")
monkeypatch.setattr(geofence, "refresh_cache", boom)
async def run():
from datetime import datetime, timezone
return await geofence.snapshot_at(
FENCE_ID, datetime(2026, 8, 28, 12, 4, tzinfo=timezone.utc),
)
assert asyncio.run(run()) is None
def test_snapshot_queries_st_intersects(monkeypatch):
geofence._cache[:] = [{
"id": FENCE_ID, "name": "NC", "geojson": NC_BOX, "active": True,
}]
sqls: list[str] = []
class FakeResult:
def mappings(self):
return self
def all(self):
return []
class FakeSession:
async def execute(self, stmt, params=None):
sqls.append(str(stmt))
return FakeResult()
async def __aenter__(self):
return self
async def __aexit__(self, *a):
return False
monkeypatch.setattr(geofence, "async_session", FakeSession)
async def run():
from datetime import datetime, timezone
return await geofence.snapshot_at(
FENCE_ID, datetime(2026, 8, 28, 12, 4, 30, tzinfo=timezone.utc),
)
body = asyncio.run(run())
assert body["aircraft"] == []
assert body["vessels"] == []
assert body["fires"] == []
blob = "\n".join(sqls).lower()
assert "st_intersects" in blob
assert "aircraft_tracks_1min" in blob
assert "vessel_tracks_1min" in blob
assert "from fires" in blob

View file

@ -1,56 +0,0 @@
"""Geofence layer panel: draw, watch, inbox, delete (HTML contract)."""
from __future__ import annotations
from pathlib import Path
ROOT = Path(__file__).resolve().parent.parent
HTML = (ROOT / "app/static/index.html").read_text()
def test_geofence_panel_has_list_and_delete_hook():
assert 'id="gf-draw"' in HTML
assert 'id="gf-list"' in HTML
assert "function deleteGeofence" in HTML
assert "method: 'DELETE'" in HTML or 'method: "DELETE"' in HTML
assert "/api/geofences/" in HTML
def test_load_geofences_renders_delete_controls():
js = HTML.split("async function loadGeofences", 1)[1].split(
"async function loadFireAircraftHits", 1
)[0]
assert "gf-list" in js
assert "deleteGeofence" in js
assert "onEachFeature" in js
assert "bindPopup" in js
def test_finish_cancel_draw_controls():
assert 'id="gf-finish"' in HTML
assert 'id="gf-cancel"' in HTML
assert "function cancelGeofenceDraw" in HTML
assert "function onGfClose" in HTML
def test_watch_geofences_ws_payload():
assert "watch_geofences" in HTML
assert "function sendWatchGeofences" in HTML
def test_geofence_alert_inbox():
assert 'id="gf-inbox"' in HTML
assert "/api/geofence-alerts" in HTML
assert "function loadGfInbox" in HTML
assert "function pushGfInbox" in HTML
def test_delete_geofence_still_present():
assert "function deleteGeofence" in HTML
assert "method: 'DELETE'" in HTML or 'method: "DELETE"' in HTML
def test_fence_dvr_at_endpoint():
assert "/at?timestamp=" in HTML or "/at?timestamp=${" in HTML
assert "function dvrScrubFence" in HTML
assert "gfSelectedId" in HTML

View file

@ -1,150 +0,0 @@
"""GPSJAM GPS-interference overlay: level mapping, CSV→GeoJSON, API contract."""
import asyncio
import httpx
from live_layers import gpsjam_csv_to_geojson, gpsjam_level, overlay_catalog, _cache
from main import app
BASE = "http://test"
# A valid H3 resolution-4 cell id (the payload hex column carries these).
HEX_A = "8400c57ffffffff"
CSV = (
"hex,count_good_aircraft,count_bad_aircraft\n"
f"{HEX_A},0,20\n" # 100*(20-1)/20 = 95 -> high
f"{HEX_A},8,2\n" # 100*(2-1)/10 = 10 -> medium
f"{HEX_A},98,2\n" # 100*(2-1)/100 = 1 -> low
f"{HEX_A},100,0\n" # bad == 0 -> dropped
)
async def _get(path: str) -> httpx.Response:
transport = httpx.ASGITransport(app=app)
async with httpx.AsyncClient(transport=transport, base_url=BASE) as client:
return await client.get(path)
def test_gpsjam_level_thresholds():
assert gpsjam_level(0.0) == "low"
assert gpsjam_level(2.0) == "low"
assert gpsjam_level(2.1) == "medium"
assert gpsjam_level(10.0) == "medium"
assert gpsjam_level(10.1) == "high"
assert gpsjam_level(95.0) == "high"
def test_gpsjam_csv_to_geojson_levels_and_drop_zero_bad():
fc = gpsjam_csv_to_geojson(CSV)
assert fc["type"] == "FeatureCollection"
assert len(fc["features"]) == 3 # bad==0 row dropped
levels = [f["properties"]["level"] for f in fc["features"]]
assert levels == ["high", "medium", "low"]
for f in fc["features"]:
geom = f["geometry"]
assert geom["type"] == "Polygon"
ring = geom["coordinates"][0]
assert len(ring) == 7 # 6 verts + closing point
assert ring[0] == ring[-1]
assert f["properties"]["hex"] == HEX_A
assert set(f["properties"]).issuperset({"level", "percent_bad", "good", "bad", "hex"})
def test_gpsjam_csv_skips_malformed_rows():
bad_csv = "hex,count_good_aircraft,count_bad_aircraft\n" \
",1,5\n" \
f"{HEX_A},x,5\n" \
f"{HEX_A},1,notanint\n" \
"not_a_cell,1,5\n"
fc = gpsjam_csv_to_geojson(bad_csv)
assert fc["features"] == []
def test_overlay_catalog_has_gpsjam_stub():
entry = overlay_catalog()["gpsjam"]
assert entry["kind"] == "geojson"
assert entry["endpoint"] == "/api/map/gpsjam"
assert "GPSJAM" in entry["attribution"]
def test_map_gpsjam_returns_featurecollection(monkeypatch):
async def fake_fetch(date):
return {"type": "FeatureCollection", "features": [{"type": "Feature"}]}
monkeypatch.setattr("main.fetch_gpsjam", fake_fetch)
resp = asyncio.run(_get("/api/map/gpsjam?date=2026-08-28"))
assert resp.status_code == 200
assert resp.json()["type"] == "FeatureCollection"
assert "max-age" in (resp.headers.get("cache-control") or "").lower()
def test_map_gpsjam_rejects_bad_date():
resp = asyncio.run(_get("/api/map/gpsjam?date=08-28-2026"))
assert resp.status_code == 422
def test_map_gpsjam_unavailable_on_404(monkeypatch):
import httpx as _httpx
async def fake_fetch(date):
exc = _httpx.HTTPStatusError(
"404", request=_httpx.Request("GET", "http://x"), response=_httpx.Response(404)
)
raise exc
monkeypatch.setattr("main.fetch_gpsjam", fake_fetch)
resp = asyncio.run(_get("/api/map/gpsjam?date=2026-08-28"))
assert resp.status_code == 200
body = resp.json()
assert body["error"] == "unavailable"
assert body["href"] == "https://gpsjam.org/"
def test_map_gpsjam_unavailable_on_empty_features(monkeypatch):
async def fake_fetch(date):
return {"type": "FeatureCollection", "features": []}
monkeypatch.setattr("main.fetch_gpsjam", fake_fetch)
resp = asyncio.run(_get("/api/map/gpsjam?date=2026-08-28"))
assert resp.status_code == 200
assert resp.json()["error"] == "unavailable"
def test_fetch_gpsjam_hits_http_once_within_ttl(monkeypatch):
_cache.clear()
hits = {"n": 0}
class FakeResp:
text = CSV
def raise_for_status(self):
pass
class FakeClient:
def __init__(self, **kw):
pass
async def __aenter__(self):
return self
async def __aexit__(self, *exc):
return False
async def get(self, url):
hits["n"] += 1
assert url == "https://gpsjam.org/data/2026-08-28-h3_4.csv"
return FakeResp()
monkeypatch.setattr("live_layers.httpx.AsyncClient", FakeClient)
monkeypatch.setattr("live_layers._http", None)
from live_layers import fetch_gpsjam
fc1 = asyncio.run(fetch_gpsjam("2026-08-28"))
fc2 = asyncio.run(fetch_gpsjam("2026-08-28"))
assert len(fc1["features"]) == 3
assert fc2 == fc1
assert hits["n"] == 1
_cache.clear()

View file

@ -1,33 +0,0 @@
"""Liveness stays up; readiness/freshness is explicit."""
from __future__ import annotations
import asyncio
import httpx
from main import app
BASE = "http://test"
async def _get(path: str) -> httpx.Response:
transport = httpx.ASGITransport(app=app)
async with httpx.AsyncClient(transport=transport, base_url=BASE) as client:
return await client.get(path)
def test_health_includes_checks_even_when_db_ok(monkeypatch):
"""HUD must be able to show degraded without docker killing the container."""
body = asyncio.run(_get("/api/health")).json()
assert "status" in body
assert "checks" in body
assert "db" in body["checks"]
def test_ready_endpoint_exists():
resp = asyncio.run(_get("/api/ready"))
assert resp.status_code in (200, 503)
body = resp.json()
assert "checks" in body
assert "status" in body

View file

@ -1,113 +0,0 @@
"""Quiet HUD chrome: VIIRS default, collapsed rail, no Orbitron/MKT dashes."""
from __future__ import annotations
from pathlib import Path
ROOT = Path(__file__).resolve().parent.parent
HTML = (ROOT / "app/static/index.html").read_text()
def _attr(html: str, elem_id: str) -> str:
chunk = html.split(f'id="{elem_id}"', 1)[1].split(">", 1)[0]
return chunk
def test_initmap_prefers_viirs_true_color():
init = HTML.split("async function initMap", 1)[1].split("function readMapPrefs", 1)[0]
assert "VIIRS_SNPP_CorrectedReflectance_TrueColor" in init
assert init.index("VIIRS_SNPP_CorrectedReflectance_TrueColor") < init.index(
"MODIS_Terra_CorrectedReflectance_TrueColor"
)
assert init.index("MODIS_Terra_CorrectedReflectance_TrueColor") < init.index(
"BlueMarble_ShadedRelief_Bathymetry"
)
def test_orbitron_gone():
assert "Orbitron" not in HTML
assert "IBM Plex Sans" in HTML
assert "IBM Plex Mono" in HTML
def test_lp_note_stripped_from_layer_list():
assert 'class="lp-note"' not in HTML
body = HTML.split('class="lp-body"', 1)[1].split("lp-legend", 1)[0]
assert "lp-note" not in body
def test_default_overlays_basemap_and_firms_only():
fires = _attr(HTML, "lp-fires-on")
assert "checked" in fires
for eid in (
"lp-cams-on",
"lp-blips-on",
"lp-news-on",
"lp-radar-on",
"lp-alerts-on",
"lp-perim-on",
"lp-ac-on",
"lp-trains-on",
"lp-storms-on",
):
assert "checked" not in _attr(HTML, eid), eid
def test_geofence_markup_before_cameras():
assert 'id="gf-draw"' in HTML
assert HTML.index('id="gf-draw"') < HTML.index('id="lp-cams-on"')
assert HTML.index('id="lp-base-on"') < HTML.index('id="gf-draw"')
def test_parent_geofence_hud_survives():
assert "watch_geofences" in HTML
assert "function deleteGeofence" in HTML
assert 'id="gf-finish"' in HTML
assert 'id="gf-cancel"' in HTML
assert 'id="gf-inbox"' in HTML
def test_layer_rail_collapsed_on_load():
head = HTML.split('class="lp-head"', 1)[1].split("</div>", 1)[0]
assert 'aria-expanded="false"' in head
assert 'id="layer-panel" class="collapsed"' in HTML
def test_market_ticker_hidden_no_poll():
mkt = HTML.split('class="ticker market"', 1)[1].split(">", 1)[0]
assert "hidden" in mkt
assert "setInterval(probeMarket" not in HTML
assert "setInterval(loadMarket" not in HTML
init = HTML.split("function initMarketTicker", 1)[1].split("function ", 1)[0]
assert "/api/market" in init or "404-poll" in init
assert "setInterval" not in init
def test_news_ticker_fills_news_only_dock():
css = HTML.split("</style>", 1)[0]
compact = css.replace(" ", "").replace("\n", "")
assert ".dock.news-only{height:32px;}" in compact
assert ".dock.news-only.ticker{height:100%;}" in compact
assert ".ticker{display:flex;align-items:stretch;height:50%;" in compact
def test_news_pins_are_circle_markers():
js = HTML.split("async function loadNewsPins", 1)[1].split("function refreshLiveOverlays", 1)[0]
assert "L.circleMarker" in js
assert "fillOpacity: 0.7" in js or "fillOpacity:0.7" in js
assert "rotate(45deg)" not in js
assert "L.divIcon" not in js
def test_chokepoint_buttons_not_in_toolbar_flow():
assert 'id="chokepoint-select"' in HTML
css = HTML.split("</style>", 1)[0]
assert ".chokepoint-btns { display: none; }" in css or ".chokepoint-btns{display:none" in css.replace(
" ", ""
)
def test_brand_is_osint_slash():
assert "GLOBAL SITUATIONAL AWARENESS TERMINAL" not in HTML
assert "OSINT" in HTML
assert 'class="accent">//</span>' in HTML

View file

@ -1,90 +0,0 @@
"""HUD: layer-rail stats, shortcuts, terminator, zoom-gated cams, SWPC chip."""
from __future__ import annotations
from pathlib import Path
ROOT = Path(__file__).resolve().parent.parent
HTML = (ROOT / "app/static/index.html").read_text()
def _fn(name: str, nxt: str | None = None) -> str:
start = HTML.index(f"function {name}")
if nxt:
return HTML[start : HTML.index(f"function {nxt}", start + 1)]
return HTML[start : start + 4000]
def test_stats_poll_uses_api_then_falls_back():
assert "/api/stats" in HTML
assert "30000" in HTML.split("pollLayerStats")[1][:2500] or "STATS_POLL_MS" in HTML
poll = HTML.split("async function pollLayerStats")[1].split("async function ")[0]
assert "404" in poll
assert "catch" in poll
ids = HTML.split("STATS_COUNT_IDS")[1].split("};")[0]
assert "aircraft" in ids and "cameras" in ids and "fires" in ids and "vessels" in ids
# Overlay loaders still write array lengths when stats is down.
assert "setLayerCount('lp-fires-count'" in HTML or 'setLayerCount("lp-fires-count"' in HTML
assert "setLayerCount('lp-cams-count'" in HTML or 'setLayerCount("lp-cams-count"' in HTML
assert "setLayerCount('lp-ac-count'" in HTML or 'setLayerCount("lp-ac-count"' in HTML
assert "setLayerCount('lp-vessels-count'" in HTML or 'setLayerCount("lp-vessels-count"' in HTML
def test_keyboard_shortcuts_do_not_steal_osiris_fs():
keys = HTML.split("function initHudKeys")[1].split("function ")[0]
assert "Escape" in keys
assert "cheat-sheet" in keys or "toggleCheatSheet" in keys
assert "mapResetView" in keys
assert "toggleLayerPanel" in keys or "closeLayerPanel" in keys
# Do not bind Osiris's conflicting F/S (flights vs fullscreen / search).
assert "e.key === 'f'" not in keys.lower()
assert "e.key === 's'" not in keys.lower()
assert "case 'f'" not in keys.lower()
assert "case 's'" not in keys.lower()
assert 'id="cheat-sheet"' in HTML
assert "?" in keys or "Shift" in keys
def test_terminator_toggle_defaults_off():
assert 'id="lp-terminator-on"' in HTML
row = HTML.split('id="lp-terminator-on"')[0][-120:] + HTML.split('id="lp-terminator-on"')[1][:80]
assert "checked" not in row.split(">")[0]
assert "function toggleTerminator" in HTML
assert "subsolarPoint" in HTML or "terminator" in HTML.lower()
def test_camera_thumbs_gated_at_zoom_12():
assert "CAM_THUMB_MIN_ZOOM" in HTML
assert "CAM_THUMB_MIN_ZOOM = 12" in HTML
thumb = _fn("camThumb", "camPopupHtml")
assert "camThumbsAllowed" in thumb or "CAM_THUMB_MIN_ZOOM" in thumb
assert "zoom in for preview" in HTML or "zoom for preview" in HTML
assert "preview unavailable" in HTML
# RTSP still proxy through snapshot; never emit rtsp hrefs.
src = _fn("camSourceLink", "youtubeId")
assert "rtsp://" in src
assert "href=" not in src.split("rtsp://")[1].split("return")[0] or "Never emit" in src
assert 'href="${esc(url)}"' in src or "href=\"${esc(url)}\"" in src
assert src.index("rtsp://") < src.index("href=")
def test_swpc_chip_browser_direct_correct_urls():
assert 'id="swpc-chip"' in HTML
assert "services.swpc.noaa.gov/json/planetary_k_index_1m.json" in HTML
assert "services.swpc.noaa.gov/json/goes/primary/xray-flares-latest.json" in HTML
assert "services.swpc.noaa.gov/products/alerts.json" in HTML
assert "services.swpc.noaa.gov/json/alerts.json" not in HTML
sw = HTML.split("async function pollSwpc")[1].split("async function ")[0]
assert "hidden" in sw
assert "kp_index" in sw
assert "90000" in HTML or "SWPC_POLL_MS" in HTML
def test_new_chrome_does_not_cover_mobile_layers_zoom():
mobile = HTML.split("@media (max-width: 820px)")[1].split("@media (prefers-reduced-motion")[0]
assert "#layer-panel" in mobile
assert ".leaflet-top.leaflet-right .leaflet-control-zoom" in mobile
assert 'id="cheat-sheet"' in HTML
cheat = HTML.split(".cheat-sheet")[1][:500]
assert "z-index" in cheat
assert "calc(100% - 96px)" in cheat or "96px" in cheat

View file

@ -1,145 +0,0 @@
"""GET /api/infrastructure — Overpass nuclear markers."""
import asyncio
import httpx
from live_layers import (
normalize_infra_element,
overlay_catalog,
overpass_nuclear_to_markers,
_cache,
)
from main import app
BASE = "http://test"
OVERPASS = {
"version": 0.6,
"generator": "Overpass API",
"elements": [
{
"type": "node",
"id": 12345,
"lat": 44.0,
"lon": -1.5,
"tags": {"name": "Test NPP", "operator": "EDF", "plant:source": "nuclear"},
},
{
"type": "way",
"id": 67890,
"center": {"lat": 43.5, "lon": -1.25},
"tags": {"name": "Test Plant Way", "plant:source": "nuclear"},
},
{
"type": "relation",
"id": 999,
"center": {"lat": 43.0, "lon": -1.0},
"tags": {},
},
],
}
async def _get(path: str) -> httpx.Response:
transport = httpx.ASGITransport(app=app)
async with httpx.AsyncClient(transport=transport, base_url=BASE) as client:
return await client.get(path)
def test_normalize_node_to_marker():
m = normalize_infra_element(OVERPASS["elements"][0], "nuclear")
assert m["id"] == "node/12345"
assert m["name"] == "Test NPP"
assert m["lat"] == 44.0
assert m["lon"] == -1.5
assert m["type"] == "nuclear"
assert m["extra"]["operator"] == "EDF"
assert "name" not in m["extra"]
def test_way_center_and_unnamed_fallback():
way = normalize_infra_element(OVERPASS["elements"][1], "nuclear")
assert way["lat"] == 43.5
assert way["lon"] == -1.25
rel = normalize_infra_element(OVERPASS["elements"][2], "nuclear")
assert rel["name"] == "relation/999"
def test_overpass_json_to_markers():
markers = overpass_nuclear_to_markers(OVERPASS)
assert len(markers) == 3
assert markers[0]["id"] == "node/12345"
def test_missing_bbox_400():
resp = asyncio.run(_get("/api/infrastructure?types=nuclear"))
assert resp.status_code == 400
def test_unknown_type_422():
resp = asyncio.run(_get("/api/infrastructure?types=military&bbox=-2,43,-1,44"))
assert resp.status_code == 422
def test_map_infrastructure_returns_markers(monkeypatch):
async def fake_fetch(types, bbox):
return [
{"id": "node/1", "name": "X", "lat": 1.0, "lon": 2.0,
"type": "nuclear", "extra": {}}
]
monkeypatch.setattr("main.fetch_infrastructure", fake_fetch)
resp = asyncio.run(_get("/api/infrastructure?types=nuclear&bbox=-2,43,-1,44"))
assert resp.status_code == 200
body = resp.json()
assert body[0]["name"] == "X"
assert body[0]["type"] == "nuclear"
assert "max-age" in (resp.headers.get("cache-control") or "").lower()
def test_overlay_catalog_has_infra_nuclear():
entry = overlay_catalog()["infra_nuclear"]
assert entry["kind"] == "points"
assert "nuclear" in entry["endpoint"]
def test_fetch_infrastructure_cache_hit_no_refetch(monkeypatch):
_cache.clear()
hits = {"n": 0}
class FakeResp:
def raise_for_status(self):
pass
def json(self):
return OVERPASS
class FakeClient:
def __init__(self, **kw):
pass
async def __aenter__(self):
return self
async def __aexit__(self, *exc):
return False
async def post(self, url, data=None, timeout=None):
hits["n"] += 1
assert "overpass-api.de" in url
assert "plant:source" in data["data"]
assert "nuclear" in data["data"]
return FakeResp()
monkeypatch.setattr("live_layers.httpx.AsyncClient", FakeClient)
monkeypatch.setattr("live_layers._http", None)
from live_layers import fetch_infrastructure
m1 = asyncio.run(fetch_infrastructure("nuclear", "-2,43,-1,44"))
m2 = asyncio.run(fetch_infrastructure("nuclear", "-2,43,-1,44"))
assert len(m1) == 3
assert m2 == m1
assert hits["n"] == 1
_cache.clear()

View file

@ -1,67 +0,0 @@
"""SSRF guard on ingest triggers + PATCH /api/sources allowlist."""
from __future__ import annotations
import asyncio
import httpx
import pytest
from pydantic import ValidationError
from main import app
BASE = "http://test"
LINK_LOCAL_META = "http://169.254.169.254/latest/meta-data/"
LOOPBACK = "http://127.0.0.1/secret"
async def _req(method: str, path: str, **kw) -> httpx.Response:
transport = httpx.ASGITransport(app=app)
async with httpx.AsyncClient(transport=transport, base_url=BASE) as client:
return await client.request(method, path, **kw)
def test_rss_ingest_rejects_link_local_metadata_url(monkeypatch):
called = {"n": 0}
async def _boom(*_a, **_k):
called["n"] += 1
raise AssertionError("ingest_rss_feed must not run for a private URL")
monkeypatch.setattr("main.ingest_rss_feed", _boom)
resp = asyncio.run(_req("POST", "/api/ingest/rss", params={"feed_url": LINK_LOCAL_META}))
assert resp.status_code == 400
assert called["n"] == 0
def test_gdelt_ingest_rejects_private_query_url(monkeypatch):
called = {"n": 0}
async def _boom(*_a, **_k):
called["n"] += 1
raise AssertionError("ingest_gdelt must not run for a private URL query")
monkeypatch.setattr("main.ingest_gdelt", _boom)
resp = asyncio.run(_req("POST", "/api/ingest/gdelt", params={"query": LOOPBACK}))
assert resp.status_code == 400
assert called["n"] == 0
def test_update_source_rejects_unknown_fields():
sid = "00000000-0000-0000-0000-000000000001"
resp = asyncio.run(_req("PATCH", f"/api/sources/{sid}", json={"enabled": True, "source_type": "rss"}))
assert resp.status_code == 422
def test_feed_source_update_allowlist_only():
from schemas import FeedSourceUpdate
payload = FeedSourceUpdate(name="n", url="https://example.com/rss", config={"k": 1}, enabled=False)
assert payload.model_dump(exclude_unset=True) == {
"name": "n",
"url": "https://example.com/rss",
"config": {"k": 1},
"enabled": False,
}
with pytest.raises(ValidationError):
FeedSourceUpdate.model_validate({"enabled": True, "id": "00000000-0000-0000-0000-000000000001"})

View file

@ -1,7 +1,5 @@
"""Unit tests for live map-layer mappers (aircraft, trains, AIS, WFIGS, Caltrans).""" """Unit tests for live map-layer mappers (aircraft, trains, AIS, WFIGS, Caltrans)."""
import json
from live_layers import ( from live_layers import (
MARKER_FIELDS, MARKER_FIELDS,
bbox_center_radius_nm, bbox_center_radius_nm,
@ -9,10 +7,7 @@ from live_layers import (
filter_points_bbox, filter_points_bbox,
parse_bbox, parse_bbox,
quantize_bbox, quantize_bbox,
pick_sentinel_feature,
rainviewer_tile_url, rainviewer_tile_url,
sign_cog_url,
sentinel1_tile_url,
slim_alert_properties, slim_alert_properties,
to_marker, to_marker,
transform_adsb_lol, transform_adsb_lol,
@ -20,14 +15,12 @@ from live_layers import (
transform_amtraker, transform_amtraker,
transform_nhc_storms, transform_nhc_storms,
transform_wfigs_incidents, transform_wfigs_incidents,
SENTINEL1_ATTRIBUTION,
TITILER_COG_TILES,
_cache, _cache,
_ttl_get, _ttl_get,
_wfigs_params, _wfigs_params,
) )
from camera_scraper import parse_caltrans_json, parse_udot_ibi_page, parse_odot_json, parse_mdot_json from camera_scraper import parse_caltrans_json
def test_parse_bbox_and_radius_clamps_to_150_nm(): def test_parse_bbox_and_radius_clamps_to_150_nm():
@ -259,186 +252,6 @@ def test_parse_caltrans_skips_oos_and_maps_jpeg_hls():
assert "rtsp://" not in cam["snapshot_url"].lower() assert "rtsp://" not in cam["snapshot_url"].lower()
# ── UDOT IBI 511 parser ──────────────────────────────────────────────────
def _udot_row(cam_id, lng, lat, **img_overrides):
img = {
"id": cam_id, "cameraSiteId": cam_id,
"imageUrl": f"/map/Cctv/{cam_id}", "disabled": False, "blocked": False,
}
img.update(img_overrides)
return {
"id": cam_id, "sourceId": "102771", "source": "ADX",
"roadway": "Unknown", "direction": "Unknown",
"location": "Freedom Blvd / 200 W @ 1100 N, PVO",
"latLng": {"geography": {
"coordinateSystemId": 4326,
"wellKnownText": f"POINT ({lng} {lat})"}},
"images": [img],
}
def _udot_page(rows):
import json
return json.dumps({"draw": 0, "recordsTotal": len(rows),
"recordsFiltered": len(rows), "data": rows})
def test_parse_udot_wkt_maps_lng_lat():
cams = parse_udot_ibi_page(_udot_page([_udot_row(112731, -111.66204, 40.24863)]))
assert len(cams) == 1
cam = cams[0]
# WKT is `POINT (lng lat)` — order must not be swapped.
assert cam["location_lat"] == 40.24863
assert cam["location_lon"] == -111.66204
assert cam["discovery_source"] == "udot"
assert cam["vendor"] == "UDOT"
assert cam["source_url"] == "https://prod-ut.ibi511.com/map/Cctv/112731"
assert cam["snapshot_url"] == cam["source_url"]
assert "rtsp://" not in cam["source_url"].lower()
assert cam["raw"]["udot_id"] == 112731
def test_parse_udot_skips_blocked_and_disabled():
rows = [
_udot_row(1, -111.0, 40.0),
_udot_row(2, -111.1, 40.1, blocked=True),
_udot_row(3, -111.2, 40.2, disabled=True),
]
rows.append(_udot_row(4, -111.3, 40.3))
rows[3]["images"] = [] # no images → drop
cams = parse_udot_ibi_page(_udot_page(rows))
assert [c["raw"]["udot_id"] for c in cams] == [1]
def test_parse_udot_drops_out_of_bbox():
rows = [
_udot_row(1, -111.0, 40.0), # inside Utah
_udot_row(2, -100.0, 40.0), # east of -108.9
_udot_row(3, -120.0, 40.0), # west of -114.2
_udot_row(4, -111.0, 44.0), # north of 42.1
_udot_row(5, -111.0, 30.0), # south of 36.9
]
cams = parse_udot_ibi_page(_udot_page(rows))
assert [c["raw"]["udot_id"] for c in cams] == [1]
def test_parse_udot_bad_payload_returns_empty():
import json
assert parse_udot_ibi_page("not json") == []
assert parse_udot_ibi_page(json.dumps({"data": None})) == []
assert parse_udot_ibi_page(json.dumps({"data": "nope"})) == []
def test_parse_udot_missing_wkt_skipped():
row = _udot_row(1, -111.0, 40.0)
row["latLng"] = {}
assert parse_udot_ibi_page(_udot_page([row])) == []
def test_parse_odot_tripcheck_keeps_valid_skips_missing_and_oob():
payload = """
{"features":[
{"attributes":{
"cameraId":277,"filename":"AstoriaUS101_pid392.jpg",
"latitude":46.18785,"longitude":-123.85347,
"route":"US101 ","title":"US101 at Astoria"
}},
{"attributes":{
"cameraId":200,"filename":"","latitude":45.0,"longitude":-122.0,
"route":"I-5","title":"missing filename"
}},
{"attributes":{
"cameraId":300,"filename":"nocal_pid1.jpg",
"latitude":40.0,"longitude":-122.0,
"route":"US97","title":"out of bbox"
}},
{"attributes":{
"cameraId":400,"filename":"badcoord_pid2.jpg",
"latitude":null,"longitude":-122.0,
"route":"OR22","title":"null coord"
}}
]}
"""
cams = parse_odot_json(payload, "www.tripcheck.com")
assert len(cams) == 1
cam = cams[0]
assert cam["discovery_source"] == "odot"
assert cam["snapshot_url"] == (
"https://tripcheck.com/RoadCams/cams/AstoriaUS101_pid392.jpg")
assert cam["source_url"] == cam["snapshot_url"]
assert cam["location_lat"] == 46.18785
assert cam["location_lon"] == -123.85347
assert "US101 at Astoria" in cam["location_name"]
assert cam["vendor"] == "ODOT"
assert cam["device_type"] == "http"
assert "rtsp://" not in cam["snapshot_url"].lower()
def test_parse_odot_tripcheck_handles_malformed():
assert parse_odot_json("not json", "www.tripcheck.com") == []
assert parse_odot_json('{"features":null}', "www.tripcheck.com") == []
def test_parse_mdot_extracts_html_fields_and_bbox_filters():
rows = [
# In-bbox, full fields.
{
"route": "11 Mile",
"county": 'Wayne County <a href="/MiDrive/map?cameras=true&lat=42.491304&lon=-83.04479&zoom=15&id=1129"target="_blank">Go to</a>',
"location": " @ Mound NB",
"direction": "Traffic closest to camera is traveling north.",
"image": '<img alt="x" class="cameraImageForActivePane" id="1129Img" src="https://micamerasimages.net/thumbs/semtoc_cam_253.flv.jpg?item=1" height="170" width="250" onerror="cameraImageBroken(this)">',
},
# Out of bbox (lat 50) → drop.
{
"route": "Far",
"county": 'Nowhere <a href="/MiDrive/map?lat=50.0&lon=-83.0&zoom=15&id=9999">Go to</a>',
"location": "",
"image": '<img src="https://micamerasimages.net/thumbs/x.jpg">',
},
# Missing coordinates → drop.
{
"route": "NoCoords",
"county": 'Somewhere <a href="/MiDrive/map?zoom=15&id=8888">Go to</a>',
"location": "",
"image": '<img src="https://micamerasimages.net/thumbs/y.jpg">',
},
# Missing image → drop.
{
"route": "NoImage",
"county": 'Kent <a href="/MiDrive/map?lat=42.8841&lon=-85.6646&zoom=15&id=2113">Go to</a>',
"location": " @ Division",
"image": "",
},
# RTSP image src → drop.
{
"route": "Rtsp",
"county": 'Wayne <a href="/MiDrive/map?lat=42.4&lon=-83.1&zoom=15&id=1234">Go to</a>',
"location": "",
"image": '<img src="rtsp://10.0.0.1/stream">',
},
]
cams = parse_mdot_json(json.dumps(rows), "mdotjboss.state.mi.us")
assert len(cams) == 1
cam = cams[0]
assert cam["discovery_source"] == "mdot"
assert cam["location_lat"] == 42.491304
assert cam["location_lon"] == -83.04479
assert cam["snapshot_url"] == "https://micamerasimages.net/thumbs/semtoc_cam_253.flv.jpg?item=1"
assert cam["source_url"] == "https://mdotjboss.state.mi.us/MiDrive/camera/1129"
assert cam["device_type"] == "http"
assert cam["vendor"] == "MDOT"
assert "11 Mile @ Mound NB" in cam["location_name"]
assert "Wayne County" in cam["location_name"]
def test_parse_mdot_handles_malformed_payload():
assert parse_mdot_json("not json", "mdot") == []
assert parse_mdot_json('{"not": "a list"}', "mdot") == []
assert parse_mdot_json("[]", "mdot") == []
def test_quantize_bbox_stable_under_jitter(): def test_quantize_bbox_stable_under_jitter():
a = quantize_bbox(*parse_bbox("-78.7912,35.7711,-78.6101,35.9102")) a = quantize_bbox(*parse_bbox("-78.7912,35.7711,-78.6101,35.9102"))
b = quantize_bbox(*parse_bbox("-78.7900,35.7700,-78.6110,35.9090")) b = quantize_bbox(*parse_bbox("-78.7900,35.7700,-78.6110,35.9090"))
@ -549,440 +362,3 @@ def test_wfigs_params_requests_simplified_geometry():
# Envelope is the quantized cell, not the raw pan box. # Envelope is the quantized cell, not the raw pan box.
geom = params["geometry"] geom = params["geometry"]
assert geom != "-84.5,33.8,-75.4,36.6" assert geom != "-84.5,33.8,-75.4,36.6"
def test_transform_adsb_lol_flags_military_from_dbflags():
payload = {
"ac": [
{
"hex": "ae01ab",
"flight": "RCH123 ",
"r": "04-1234",
"t": "C17",
"lat": 35.1,
"lon": -77.9,
"alt_baro": 24000,
"gs": 410,
"track": 90,
"squawk": "5101",
"emergency": "none",
"category": "A5",
"dbFlags": 1,
"baro_rate": 64,
"alt_geom": 24500,
"desc": "Boeing C-17A Globemaster III",
"ownOp": "USAF",
},
{
"hex": "a1b2c3",
"flight": "AAL123",
"r": "N123AA",
"t": "B738",
"lat": 35.88,
"lon": -78.79,
"alt_baro": 32000,
"gs": 430,
"track": 87,
"squawk": "1200",
"emergency": "none",
"category": "A3",
},
]
}
rows = {r["id"]: r for r in transform_adsb_lol(payload)}
mil = rows["ae01ab"]["extra"]
civ = rows["a1b2c3"]["extra"]
assert mil["role"] == "military"
assert mil["role_src"] == "dbFlags"
assert mil["emitter"] == "heavy"
assert mil["desc"] == "Boeing C-17A Globemaster III"
assert mil["ownOp"] == "USAF"
assert mil["vs"] == 64
assert mil["alt_geom"] == 24500
assert civ["role"] == "civilian"
assert civ["emitter"] == "large"
def test_transform_adsb_lol_military_from_icao_type_and_hex():
payload = {
"ac": [
{"hex": "3b76aa", "flight": "FAF123", "t": "F16", "lat": 1, "lon": 2, "category": "A1"},
{"hex": "ae1234", "flight": "BOXER1", "t": "C172", "lat": 1, "lon": 2, "category": "A1"},
]
}
rows = {r["id"]: r for r in transform_adsb_lol(payload)}
assert rows["3b76aa"]["extra"]["role"] == "military"
assert rows["3b76aa"]["extra"]["role_src"] == "type"
assert rows["ae1234"]["extra"]["role"] == "military"
assert rows["ae1234"]["extra"]["role_src"] == "hex"
def test_transform_ais_static_classifies_military_and_cargo():
mil = transform_ais_frame({
"MessageType": "ShipStaticData",
"MetaData": {"MMSI": 338123456, "ShipName": "USNS BOB", "Latitude": 32.7, "Longitude": -117.2},
"Message": {"ShipStaticData": {
"Type": 35, "CallSign": "NBXX", "ImoNumber": 0,
"Destination": "SAN DIEGO", "MaximumStaticDraught": 8.2,
"Dimension": {"A": 80, "B": 20, "C": 8, "D": 8},
"Eta": {"Month": 8, "Day": 29, "Hour": 14, "Minute": 0},
}},
})
cargo = transform_ais_frame({
"MessageType": "ShipStaticData",
"MetaData": {"MMSI": 477123456, "ShipName": "EVER GIVEN", "Latitude": 36.9, "Longitude": -76.3},
"Message": {"ShipStaticData": {
"Type": 70, "CallSign": "VRXX", "ImoNumber": 9811000,
"Destination": "NORFOLK", "MaximumStaticDraught": 14.5,
"Dimension": {"A": 200, "B": 150, "C": 20, "D": 20},
}},
})
assert mil is not None and cargo is not None
assert mil["extra"]["role"] == "military"
assert mil["extra"]["kind"] == "military"
assert mil["extra"]["callsign"] == "NBXX"
assert mil["extra"]["length"] == 100
assert mil["extra"]["beam"] == 16
assert mil["extra"]["dest"] == "SAN DIEGO"
assert mil["extra"]["country"] == "United States"
assert cargo["extra"]["role"] == "civilian"
assert cargo["extra"]["kind"] == "cargo"
assert cargo["extra"]["imo"] == 9811000
def test_transform_ais_position_decodes_navstat():
row = transform_ais_frame({
"MessageType": "PositionReport",
"MetaData": {"MMSI": 366912810, "ShipName": "EVER GIVEN", "latitude": 36.9, "longitude": -76.3},
"Message": {"PositionReport": {"Sog": 0.1, "Cog": 88.0, "TrueHeading": 90, "NavigationalStatus": 5}},
})
assert row is not None
assert row["extra"]["nav"] == "moored"
assert row["extra"]["navstat"] == 5
def test_nws_alerts_does_not_send_bbox_param(monkeypatch):
"""api.weather.gov/alerts/active 400s on bbox — clip locally instead."""
import asyncio
from live_layers import fetch_weather_alerts, _cache
seen = []
async def fake_get(url, params=None):
seen.append((url, dict(params or {})))
if "weather.gov" in url:
return {
"type": "FeatureCollection",
"features": [{
"type": "Feature",
"properties": {"event": "Tornado Warning", "severity": "Extreme"},
"geometry": {"type": "Point", "coordinates": [-78.7, 35.8]},
}],
}
return {"type": "FeatureCollection", "features": []}
monkeypatch.setattr("live_layers._get_json", fake_get)
_cache.clear()
fc = asyncio.run(fetch_weather_alerts(None, "-79.0,35.5,-78.0,36.0"))
nws_calls = [p for u, p in seen if "weather.gov" in u]
assert nws_calls, "NWS should still be fetched"
assert "bbox" not in nws_calls[0]
assert fc.get("nws_ok") is True
assert len(fc["features"]) == 1
def test_nws_alerts_failure_is_flagged(monkeypatch):
import asyncio
from live_layers import fetch_weather_alerts, _cache
async def fake_get(url, params=None):
if "weather.gov" in url:
raise RuntimeError("400 Bad Request")
return {"type": "FeatureCollection", "features": []}
monkeypatch.setattr("live_layers._get_json", fake_get)
_cache.clear()
fc = asyncio.run(fetch_weather_alerts(None, None))
assert fc.get("nws_ok") is False
def test_fetch_aircraft_get_path_does_not_persist(monkeypatch):
"""GET /api/aircraft must serve last-known without track/geofence writes."""
import asyncio
from live_layers import (
aircraft_last_known, fetch_aircraft, persist_aircraft_snapshot, _cache,
)
aircraft_last_known.clear()
aircraft_last_known["abc"] = {
"id": "abc", "lat": 35.8, "lon": -78.7, "heading": 90, "speed": 400,
"label": "ABC", "extra": {},
}
writes = {"n": 0}
async def boom(*a, **k):
writes["n"] += 1
raise AssertionError("GET path must not persist")
monkeypatch.setattr("tracks.record_position", boom)
monkeypatch.setattr("geofence.record_and_notify", boom)
_cache.clear()
rows = asyncio.run(fetch_aircraft("-79,35,-78,36", persist=False))
assert writes["n"] == 0
assert any(r["id"] == "abc" for r in rows)
def test_persist_aircraft_snapshot_writes_tracks(monkeypatch):
import asyncio
from live_layers import persist_aircraft_snapshot
recorded = []
async def fake_record(kind, marker):
recorded.append((kind, marker["id"]))
return True
async def fake_gf(**kw):
return 0
monkeypatch.setattr("tracks.record_position", fake_record)
monkeypatch.setattr("geofence.record_and_notify", fake_gf)
monkeypatch.setattr("ws_manager.manager.has_clients", lambda: False)
rows = [{
"id": "abc", "lat": 35.8, "lon": -78.7, "heading": 90, "speed": 400,
"label": "ABC", "extra": {},
}]
asyncio.run(persist_aircraft_snapshot(rows))
assert recorded == [("aircraft", "abc")]
# ── Planespotters.net photo lookup ────────────────────────────────────────
def test_normalize_planespotter_photo_prefers_large_thumbnail():
from live_layers import _normalize_planespotter_photo
out = _normalize_planespotter_photo({
"id": "1053982",
"thumbnail": {"src": "https://t.plnspttrs.net/x_t.jpg", "size": {"width": 200, "height": 141}},
"thumbnail_large": {"src": "https://t.plnspttrs.net/x_280.jpg", "size": {"width": 395, "height": 280}},
"link": "https://www.planespotters.net/photo/1053982/foo",
"photographer": "Günther Feniuk",
})
assert out["id"] == "1053982"
assert out["src"] == "https://t.plnspttrs.net/x_280.jpg"
assert out["width"] == 395
assert out["height"] == 280
assert out["photographer"] == "Günther Feniuk"
assert "planespotters.net" in out["link"]
def test_normalize_planespotter_photo_empty_or_malformed_returns_none():
from live_layers import _normalize_planespotter_photo
assert _normalize_planespotter_photo({}) is None
assert _normalize_planespotter_photo({"thumbnail": {}}) is None
assert _normalize_planespotter_photo(None) is None
assert _normalize_planespotter_photo("not-a-dict") is None
def test_fetch_planespotters_photo_hex_builds_url_and_normalizes(monkeypatch):
import asyncio
from live_layers import fetch_planespotters_photo, _cache
seen = []
async def fake_get(url, params=None, headers=None):
seen.append((url, (headers or {}).get("User-Agent", "")))
return {"photos": [{
"id": "1", "thumbnail": {"src": "https://t.plnspttrs.net/a_t.jpg"},
"thumbnail_large": {"src": "https://t.plnspttrs.net/a_280.jpg"},
"link": "https://www.planespotters.net/photo/1/x", "photographer": "A",
}]}
monkeypatch.setattr("live_layers._get_json", fake_get)
_cache.clear()
out = asyncio.run(fetch_planespotters_photo(hex_code="e8027e"))
assert out["src"] == "https://t.plnspttrs.net/a_280.jpg"
assert seen[0][0] == "https://api.planespotters.net/pub/photos/hex/e8027e"
assert "@" in seen[0][1] or "http" in seen[0][1]
def test_fetch_planespotters_photo_reg_fallback_and_no_result(monkeypatch):
import asyncio
from live_layers import fetch_planespotters_photo, _cache
seen = []
async def fake_get(url, params=None, headers=None):
seen.append(url)
return {"photos": []}
monkeypatch.setattr("live_layers._get_json", fake_get)
_cache.clear()
assert asyncio.run(fetch_planespotters_photo(reg="D-ABCD")) is None
assert seen == ["https://api.planespotters.net/pub/photos/reg/D-ABCD"]
# no hex and no reg → no upstream call at all
assert asyncio.run(fetch_planespotters_photo()) is None
def test_planespotters_headers_add_contact_when_ua_is_generic(monkeypatch):
import live_layers
monkeypatch.setattr(live_layers, "OSINT_USER_AGENT", "osint-dashboard/1.0 (self-hosted)")
ua = live_layers._planespotters_headers()["User-Agent"]
assert "osint-dashboard" in ua
assert "@" in ua
# ── Sentinel-1 SAR (Planetary Computer STAC → signed COG template) ────────
def test_sign_cog_url_appends_token():
# PC returns the token pre-encoded as a query string; append verbatim.
assert sign_cog_url("https://blob.example/x.tif", "st=s&se=e&sig=x%3D") == \
"https://blob.example/x.tif?st=s&se=e&sig=x%3D"
# Existing query string → append with &
assert sign_cog_url("https://blob.example/x.tif?foo=1", "st=s&sig=x") == \
"https://blob.example/x.tif?foo=1&st=s&sig=x"
def test_sentinel1_tile_url_contains_titiler_rescale_and_cfastie():
signed = "https://blob.example/x.tif?token=secret"
url = sentinel1_tile_url(signed)
assert url.startswith(TITILER_COG_TILES + "?")
assert "WebMercatorQuad/{z}/{x}/{y}?" in url
assert "url=https%3A%2F%2Fblob.example%2Fx.tif%3Ftoken%3Dsecret" in url
assert "rescale=0%2C500" in url
assert "colormap_name=cfastie" in url
def test_sentinel1_tile_url_is_same_origin_relative():
# Self-hosted TiTiler: the browser must hit the Pi's nginx vhost, not
# titiler.xyz or a raw host:port. The template is a root-relative path.
url = sentinel1_tile_url("https://blob.example/x.tif")
assert url.startswith("/titiler/cog/tiles/WebMercatorQuad/")
assert "://" not in url
assert "titiler.xyz" not in url
def _stac_feature(assets: dict) -> dict:
return {
"type": "Feature",
"id": "S1A_IW_GRDH_1SDV_20240820T000000",
"properties": {"datetime": "2024-08-20T00:00:00Z"},
"assets": assets,
}
def test_fetch_sentinel1_vv_signed_tile_url(monkeypatch):
import asyncio
from live_layers import fetch_sentinel1, _cache
calls = []
async def fake_post(url, json=None, headers=None):
calls.append(("post", url, json))
return {"features": [_stac_feature({
"vv": {"href": "https://blob.example/grd-vv.tif"},
})]}
async def fake_get(url, params=None, headers=None):
calls.append(("get", url))
return {"token": "sig=abc123"}
monkeypatch.setattr("live_layers._post_json", fake_post)
monkeypatch.setattr("live_layers._get_json", fake_get)
_cache.clear()
out = asyncio.run(fetch_sentinel1("-80,35,-79,36"))
assert out["id"] == "sentinel-1-sar"
assert out["kind"] == "raster"
assert out["polarization"] == "vv"
assert out["opacity"] == 0.8
assert out["itemId"].startswith("S1A")
assert out["attribution"] == SENTINEL1_ATTRIBUTION
assert "WebMercatorQuad/{z}/{x}/{y}?" in out["tileUrl"]
assert "rescale=0%2C500" in out["tileUrl"]
assert "colormap_name=cfastie" in out["tileUrl"]
# SAS token "sig=abc123" is appended top-level, then the whole COG URL is
# percent-encoded again as a query param (=> sig%3Dabc123).
assert "sig%3Dabc123" in out["tileUrl"]
# STAC search payload shape
post_url, post_json = calls[0][1], calls[0][2]
assert post_url.endswith("/api/stac/v1/search")
assert post_json["collections"] == ["sentinel-1-grd"]
assert post_json["limit"] >= 1
assert post_json["sortby"][0]["direction"] == "desc"
assert "bbox" in out
def test_fetch_sentinel1_uses_hh_when_vv_missing(monkeypatch):
import asyncio
from live_layers import fetch_sentinel1, _cache
async def fake_post(url, json=None, headers=None):
return {"features": [_stac_feature({
"hh": {"href": "https://blob.example/grd-hh.tif"},
})]}
async def fake_get(url, params=None, headers=None):
return {"token": "tok"}
monkeypatch.setattr("live_layers._post_json", fake_post)
monkeypatch.setattr("live_layers._get_json", fake_get)
_cache.clear()
out = asyncio.run(fetch_sentinel1("-80,35,-79,36"))
assert out["polarization"] == "hh"
assert "url=https%3A%2F%2Fblob.example%2Fgrd-hh.tif" in out["tileUrl"]
def test_fetch_sentinel1_none_on_empty_features(monkeypatch):
import asyncio
from live_layers import fetch_sentinel1, _cache
async def fake_post(url, json=None, headers=None):
return {"features": []}
monkeypatch.setattr("live_layers._post_json", fake_post)
_cache.clear()
assert asyncio.run(fetch_sentinel1("-80,35,-79,36")) is None
def test_fetch_sentinel1_none_when_no_vv_or_hh(monkeypatch):
import asyncio
from live_layers import fetch_sentinel1, _cache
async def fake_post(url, json=None, headers=None):
return {"features": [_stac_feature({"thumbnail": {"href": "https://x"}})]}
monkeypatch.setattr("live_layers._post_json", fake_post)
_cache.clear()
assert asyncio.run(fetch_sentinel1("-80,35,-79,36")) is None
def test_pick_sentinel_feature_prefers_scene_covering_center():
features = [
{"id": "far", "bbox": [10.0, 10.0, 12.0, 12.0]},
{"id": "cover", "bbox": [-80.5, 34.5, -78.5, 36.5]},
{"id": "also-far", "bbox": [-10.0, 0.0, -8.0, 2.0]},
]
picked = pick_sentinel_feature(features, -79.5, 35.5)
assert picked["id"] == "cover"
def test_pick_sentinel_feature_falls_back_to_first_when_none_cover():
features = [
{"id": "a", "bbox": [10.0, 10.0, 12.0, 12.0]},
{"id": "b", "bbox": [20.0, 20.0, 22.0, 22.0]},
]
assert pick_sentinel_feature(features, -79.5, 35.5)["id"] == "a"
assert pick_sentinel_feature([], -79.5, 35.5) is None

View file

@ -1,128 +0,0 @@
"""NASA EONET + CISA KEV parsers (no network)."""
from __future__ import annotations
def test_parse_eonet_keeps_stable_ids_and_points():
from sources import parse_eonet_events
payload = {
"events": [
{
"id": "EONET_6363",
"title": "Etna Volcano",
"categories": [{"id": "volcanoes", "title": "Volcanoes"}],
"geometry": [
{"date": "2024-01-01T00:00:00Z", "type": "Point", "coordinates": [15.0, 37.7]},
],
"link": "https://eonet.gsfc.nasa.gov/api/v3/events/EONET_6363",
},
{
"id": "EONET_skip",
"title": "No geometry",
"categories": [],
"geometry": [],
"link": "https://eonet.gsfc.nasa.gov/api/v3/events/EONET_skip",
},
]
}
events = parse_eonet_events(payload)
assert len(events) == 1
ev = events[0]
assert ev["url"] == "https://eonet.gsfc.nasa.gov/api/v3/events/EONET_6363"
assert ev["source_type"] == "disaster"
assert ev["location_lat"] == 37.7
assert ev["location_lon"] == 15.0
assert ev["raw"]["eonet_id"] == "EONET_6363"
assert "volcanoes" in ev["tags"]
def test_parse_cisa_kev_emits_cve_url_no_coords():
from sources import parse_cisa_kev
payload = {
"vulnerabilities": [
{
"cveID": "CVE-2024-1234",
"vendorProject": "Acme",
"product": "Widget",
"vulnerabilityName": "RCE",
"dateAdded": "2024-06-01",
"shortDescription": "Remote code execution",
"requiredAction": "Apply updates",
"dueDate": "2024-06-22",
"knownRansomwareCampaignUse": "Known",
}
]
}
events = parse_cisa_kev(payload)
assert len(events) == 1
ev = events[0]
assert ev["url"] == "https://nvd.nist.gov/vuln/detail/CVE-2024-1234"
assert ev["location_lat"] is None
assert ev["location_lon"] is None
assert "cisa-kev" in ev["tags"]
assert "CVE-2024-1234" in ev["tags"]
assert ev["raw"]["cveID"] == "CVE-2024-1234"
def test_ingest_cisa_kev_does_not_republish_known_nist_urls(monkeypatch):
"""Producer must not push the whole KEV catalog to NATS every cycle."""
import asyncio
from sources import ingest_cisa_kev
payload = {
"vulnerabilities": [
{
"cveID": "CVE-2024-1111",
"vulnerabilityName": "old",
"dateAdded": "2024-01-01",
"shortDescription": "already in db",
},
{
"cveID": "CVE-2024-2222",
"vulnerabilityName": "new",
"dateAdded": "2024-06-01",
"shortDescription": "not in db yet",
},
]
}
class FakeResp:
def raise_for_status(self):
pass
def json(self):
return payload
class FakeClient:
def __init__(self, **kw):
pass
async def __aenter__(self):
return self
async def __aexit__(self, *exc):
return False
async def get(self, url):
return FakeResp()
published: list[str] = []
async def fake_publish(subject, event):
published.append(event["url"])
known = {"https://nvd.nist.gov/vuln/detail/CVE-2024-1111"}
async def fake_existing(urls):
return {u for u in urls if u in known}
monkeypatch.setattr("sources.httpx.AsyncClient", FakeClient)
monkeypatch.setattr("sources.publish_event", fake_publish)
monkeypatch.setattr("sources.existing_event_urls", fake_existing, raising=False)
n = asyncio.run(ingest_cisa_kev())
assert n == 1
assert published == ["https://nvd.nist.gov/vuln/detail/CVE-2024-2222"]

View file

@ -1,42 +0,0 @@
"""News spider/pipeline: skip audio, use pubDate, don't log dupes as errors."""
from __future__ import annotations
import sys
from pathlib import Path
ROOT = Path(__file__).resolve().parent.parent
_SCRAPER = ROOT / "news/scraper"
if str(_SCRAPER) not in sys.path:
sys.path.insert(0, str(_SCRAPER))
def test_is_audio_enclosure():
from newsScraper.feed_util import is_audio_url
assert is_audio_url("https://cdn.example/podcast.mp3") is True
assert is_audio_url("https://cdn.example/show.m4a?x=1") is True
assert is_audio_url("https://www.example.com/world/story") is False
def test_article_timestamp_prefers_pubdate():
from newsScraper.feed_util import article_timestamp
ts = article_timestamp("Tue, 01 Apr 2025 12:00:00 GMT")
assert ts.tzinfo is not None
assert ts.year == 2025
assert ts.month == 4
assert ts.day == 1
def test_pipeline_does_not_wrap_dropitem_as_error():
src = (ROOT / "news/scraper/newsScraper/pipelines.py").read_text()
assert "except DropItem" in src
assert "seen_urls.add" in src or "self.seen_urls.add" in src
def test_spider_skips_audio_before_request():
src = (ROOT / "news/scraper/newsScraper/spiders/news_spider.py").read_text()
assert "is_audio_url" in src
assert "article_timestamp" in src
assert "datetime.datetime.now()" not in src

View file

@ -1,65 +0,0 @@
"""Phase 1 guardrails: compose limits, 500ms debounce, no extra brokers."""
from __future__ import annotations
from pathlib import Path
ROOT = Path(__file__).resolve().parent.parent
def test_compose_memory_limits_and_shared_buffers():
text = (ROOT / "docker-compose.yml").read_text()
assert "shared_buffers=2GB" in text
assert "shared_preload_libraries=timescaledb" in text
assert "memory: 3G" in text or "memory: 3GB" in text
assert "memory: 2G" in text or "memory: 2GB" in text
def test_map_moveend_debounced_500ms():
html = (ROOT / "app" / "static" / "index.html").read_text()
assert "map.on('moveend'" in html
# Existing 300ms debounce must be 500ms so pans don't spam bbox POSTs/WS.
assert "}, 500);" in html
assert "}, 300);" not in html.split("map.on('moveend'")[1][:800]
def test_no_redis_kafka_celery():
req = (ROOT / "app" / "requirements.txt").read_text().lower()
compose = (ROOT / "docker-compose.yml").read_text().lower()
for blob in (req, compose):
assert "redis" not in blob
assert "kafka" not in blob
assert "celery" not in blob
assert "cachetools" in req
def test_titiler_image_pinned_by_digest():
text = (ROOT / "docker-compose.yml").read_text()
assert (
"ghcr.io/developmentseed/titiler:latest@sha256:"
"1809958d063543e3ec858259536002b2de78e9f8f09a22a8d9591bdc2b550b14"
in text
)
# Unpinned :latest would drift on every pull.
for line in text.splitlines():
if "titiler" in line.lower() and "image:" in line:
assert "@sha256:" in line
def test_uvicorn_single_worker_guard():
text = (ROOT / "app" / "main.py").read_text()
main_block = text.split('if __name__ == "__main__":', 1)[1]
assert "workers=1" in main_block
def test_bbox_gist_migration_keeps_btree_and_adds_gist():
text = (ROOT / "alembic" / "versions" / "010_bbox_gist.py").read_text()
assert "down_revision" in text and "009_vessels" in text
assert "ix_events_geom_gist" in text
assert "ix_fires_geom_gist" in text
assert "ST_MakePoint(location_lon, location_lat)" in text
assert "ST_MakePoint(longitude, latitude)" in text
assert "USING gist" in text
models = (ROOT / "app" / "models.py").read_text()
assert 'Index("ix_events_location"' in models
assert 'Index("ix_fires_bbox"' in models

Some files were not shown because too many files have changed in this diff Show more