From c10b617f1fd71b1476906ee4879bccf40a2daccf Mon Sep 17 00:00:00 2001 From: Sirius DevOps Date: Fri, 28 Aug 2026 19:49:51 -0400 Subject: [PATCH] Revert "feat: rebuild news summarizer on local Ollama" This reverts commit 58ff43bdee2af8f8ac792cdba88717073e74f0da. --- .env.example | 14 +-- app/keystore.py | 7 +- app/main.py | 10 +- app/schemas.py | 6 +- app/settings_store.py | 54 +++++++--- app/static/index.html | 11 +- docker-compose.yml | 3 +- docs/news.md | 73 ++++++------- news/summerizer/nous_client.py | 36 +++++++ news/summerizer/ollama_client.py | 102 ------------------ news/summerizer/run_news_summarizer.py | 18 +++- news/summerizer/summarizer.py | 64 +++++++---- news/summerizer/tests/test_nous_client.py | 97 +++++++++++++++++ news/summerizer/tests/test_ollama_client.py | 111 -------------------- tests/test_api_settings.py | 67 ++++++++---- 15 files changed, 338 insertions(+), 335 deletions(-) create mode 100644 news/summerizer/nous_client.py delete mode 100644 news/summerizer/ollama_client.py create mode 100644 news/summerizer/tests/test_nous_client.py delete mode 100644 news/summerizer/tests/test_ollama_client.py diff --git a/.env.example b/.env.example index 2ba3877..ac898a9 100644 --- a/.env.example +++ b/.env.example @@ -60,7 +60,7 @@ FIRMS_INTERVAL=900 INGEST_FIRES=1 # ── API keys (managed from the dashboard UI) ────────────────────────────── -# Keys such as TELEGRAM_TOKEN are stored in the Postgres +# Keys such as NOUS_API_KEY and TELEGRAM_TOKEN are stored in the Postgres # `api_keys` table and managed from the dashboard's "Keys" tab # (GET/POST/DELETE /api/keys/{name}) — see app/keystore.py. The FIRMS ingestor # currently reads FIRMS_MAP_KEY from .env (above); wiring the Keys-UI store as @@ -68,15 +68,17 @@ INGEST_FIRES=1 # ── News pipeline (scraper + summarizer, profile `ingest`) ──────────────── # Hourly: the scraper crawls 257 RSS sources at minute :00 and the summarizer -# runs the local Ollama map-reduce at minute :05, both writing to the shared osint-db +# runs the Nous map-reduce at minute :05, both writing to the shared osint-db # (tables `articles` + `article_summaries`, created by alembic 003_news). # Consume via GET /api/news and GET /api/news/summaries. -# LLM_URL is the Ollama origin (Pi container cannot use 127.0.0.1 for a laptop -# GPU). Example: LLM_URL=http://100.91.7.84:11434 -# LLM_URL= +# NOUS_API_KEY is also (preferably) set in the Keys UI; env is an override. +# Unset in both env and api_keys = summarizer logs and idles (never crashes). +NOUS_API_KEY= +NOUS_BASE_URL=https://inference-api.nousresearch.com/v1 +# Optional LLM knobs # SUMMARY_MODEL is an optional override. Leave unset so Settings # (app_settings.SUMMARY_MODEL) can reach the summarizer. Code default -# qwen3:30b-a3b remains after a Postgres miss. Env wins when set. +# Hermes-4.3-36B remains after a Postgres miss. Env wins when set. # SUMMARY_MODEL= NEWS_BATCH_SIZE=50 SUMMARY_WINDOW_HOURS=1 diff --git a/app/keystore.py b/app/keystore.py index a701fb7..0eb49a4 100644 --- a/app/keystore.py +++ b/app/keystore.py @@ -57,6 +57,11 @@ KEY_REGISTRY: dict[str, dict] = { "pattern": r"^[0-9a-fA-F]{32}$", "example": "32-char hex string (e.g. 5f3c…9a02)", }, + "NOUS_API_KEY": { + "description": "Nous Portal API key — hourly news summarizer (inference-api.nousresearch.com).", + "pattern": r"^.{16,}$", + "example": "key from https://portal.nousresearch.com (API keys page)", + }, "TELEGRAM_TOKEN": { "description": "Telegram bot token — push alert notifications to a channel.", "pattern": r"^\d{8,10}:[0-9A-Za-z_-]{35}$", @@ -209,7 +214,7 @@ async def get_api_key(name: str) -> str | None: """Read a stored key value — used by ingest services, never by the API. Returns the raw value (or None when unset) so producers can pass it to - external APIs (FIRMS, Telegram, …). Reads live from Postgres, so a + external APIs (FIRMS, Nous, Telegram, …). Reads live from Postgres, so a key set via the dashboard is picked up on the next poll — no restart. """ await ensure_api_keys_table() diff --git a/app/main.py b/app/main.py index df90401..a2e47ee 100644 --- a/app/main.py +++ b/app/main.py @@ -605,7 +605,7 @@ async def list_documents( async def list_api_keys(): """List known API keys with set/missing status — masked, never raw. - Registered keys (FIRMS_MAP_KEY, TELEGRAM_TOKEN) + Registered keys (FIRMS_MAP_KEY, NOUS_API_KEY, TELEGRAM_TOKEN) are always included. Any extra stored keys are appended. """ return await list_keys() @@ -638,13 +638,13 @@ async def remove_api_key(name: str): @app.get("/api/settings", response_model=SettingsOut) async def get_settings(): - """Summarizer model + read-only Ollama URL.""" + """Summarizer model + read-only Nous base URL.""" return await get_app_settings() @app.put("/api/settings", response_model=SettingsOut) async def put_settings(payload: SettingsIn): - """Persist SUMMARY_MODEL. ``llm_url`` is ignored even if sent.""" + """Persist SUMMARY_MODEL. ``nous_base_url`` is ignored even if sent.""" try: return await set_summary_model(payload.summary_model) except SettingsError as exc: @@ -1084,7 +1084,7 @@ async def camera_hls_segment(camera_id: UUID, u: str = Query(..., min_length=8)) # ── News pipeline (scraper + summarizer) ────────────────────────────────── # Backing data for the frontend news panel. Written by the vendored -# news-scraper (hourly Scrapy crawl) and news-summarizer (hourly Ollama +# news-scraper (hourly Scrapy crawl) and news-summarizer (hourly Gemini # map-reduce) services into the shared osint-db. @app.get("/api/news", response_model=list[NewsArticleOut]) @@ -1249,7 +1249,7 @@ async def list_news_map( @app.get("/api/news/models", response_model=NewsModelsOut) async def list_news_models(): - """Ollama model catalog for the summarizer selector. Never 502s.""" + """Nous model catalog for the summarizer selector. Never 502s.""" return await list_models() diff --git a/app/schemas.py b/app/schemas.py index 8fd4c68..b61e296 100644 --- a/app/schemas.py +++ b/app/schemas.py @@ -316,7 +316,7 @@ class NewsModelsOut(BaseModel): class SettingsIn(BaseModel): - """Body for PUT /api/settings. ``llm_url`` is not writable.""" + """Body for PUT /api/settings. ``nous_base_url`` is not writable.""" summary_model: str = Field(..., min_length=1, max_length=128) @@ -332,10 +332,10 @@ class SettingsIn(BaseModel): class SettingsOut(BaseModel): - """Current summarizer settings. ``llm_url`` is read-only.""" + """Current summarizer settings. ``nous_base_url`` is read-only.""" summary_model: str - llm_url: str + nous_base_url: str # ─── Aggregations ──────────────────────────────────────────────────────── diff --git a/app/settings_store.py b/app/settings_store.py index 44434dc..314db42 100644 --- a/app/settings_store.py +++ b/app/settings_store.py @@ -17,10 +17,11 @@ from datetime import datetime, timezone import httpx from sqlalchemy import Column, DateTime, String, Table, Text, func, select, text +import keystore from database import async_session, engine, metadata -DEFAULT_LLM_URL = "http://127.0.0.1:11434" -DEFAULT_SUMMARY_MODEL = "qwen3:30b-a3b" +DEFAULT_NOUS_BASE_URL = "https://inference-api.nousresearch.com/v1" +DEFAULT_SUMMARY_MODEL = "Hermes-4.3-36B" SETTING_SUMMARY_MODEL = "SUMMARY_MODEL" ALLOWED_SETTINGS = frozenset({SETTING_SUMMARY_MODEL}) MODELS_CACHE_TTL_S = 600.0 @@ -28,7 +29,12 @@ MODELS_TIMEOUT_S = 8.0 DEFAULT_MODELS_USER_AGENT = "osint-dashboard-news-summarizer" FALLBACK_MODELS = [ - "qwen3:30b-a3b", + "Hermes-4.3-36B", + "Hermes-4-70B", + "google/gemini-2.5-flash", + "anthropic/claude-haiku-4.5", + "openai/gpt-4.1-mini", + "x-ai/grok-4", ] app_settings = Table( @@ -71,14 +77,10 @@ async def ensure_app_settings_table() -> None: _ensured = True -def llm_url() -> str: - """Read-only Ollama origin (env, never writable from the UI).""" - raw = (os.getenv("LLM_URL") or "").strip().rstrip("/") - for suffix in ("/api/generate", "/api/chat", "/api", "/v1"): - if raw.endswith(suffix): - raw = raw[: -len(suffix)].rstrip("/") - break - return raw or DEFAULT_LLM_URL +def nous_base_url() -> str: + """Read-only Nous inference base URL (env, never writable from the UI).""" + raw = (os.getenv("NOUS_BASE_URL") or "").strip().rstrip("/") + return raw or DEFAULT_NOUS_BASE_URL def _validate_summary_model(value: str) -> str: @@ -89,7 +91,7 @@ def _validate_summary_model(value: str) -> str: async def get_summary_model() -> str: - """Stored SUMMARY_MODEL, else env, else qwen3:30b-a3b.""" + """Stored SUMMARY_MODEL, else env, else Hermes-4.3-36B.""" await ensure_app_settings_table() async with async_session() as session: row = ( @@ -133,7 +135,7 @@ async def set_summary_model(value: str) -> dict: async def get_app_settings() -> dict: return { "summary_model": await get_summary_model(), - "llm_url": llm_url(), + "nous_base_url": nous_base_url(), } @@ -144,6 +146,18 @@ def _fallback_payload() -> dict: } +async def _nous_api_key() -> str | None: + """Keystore first, then env. Any lookup failure is treated as missing.""" + try: + stored = await keystore.get_api_key("NOUS_API_KEY") + except Exception: + stored = None + if stored and str(stored).strip(): + return str(stored).strip() + env = (os.getenv("NOUS_API_KEY") or "").strip() + return env or None + + def _models_user_agent() -> str: return os.getenv("OSINT_USER_AGENT") or DEFAULT_MODELS_USER_AGENT @@ -160,7 +174,7 @@ def _parse_models_payload(body: object) -> list[dict[str, str]]: if isinstance(item, str) and item.strip(): out.append({"id": item.strip()}) elif isinstance(item, dict): - mid = item.get("id") or item.get("name") or item.get("model") + mid = item.get("id") or item.get("name") if mid: out.append({"id": str(mid)}) return out @@ -172,15 +186,23 @@ async def _http_get(url: str, *, headers: dict[str, str], timeout: float) -> htt async def list_models() -> dict: - """Live ``GET {llm_url}/api/tags``. Never 502 — fallback on failure.""" + """Live ``GET {base}/models`` when a key is present; otherwise curated fallback. + + Never raises to the caller for missing key or upstream failure. + """ global _models_cache + key = await _nous_api_key() + if not key: + return _fallback_payload() + now = time.monotonic() hit = _models_cache if hit and now - hit[0] < MODELS_CACHE_TTL_S: return hit[1] - url = f"{llm_url()}/api/tags" + url = f"{nous_base_url()}/models" headers = { + "Authorization": f"Bearer {key}", "User-Agent": _models_user_agent(), "Accept": "application/json", } diff --git a/app/static/index.html b/app/static/index.html index c4f8190..db4fb01 100644 --- a/app/static/index.html +++ b/app/static/index.html @@ -840,7 +840,7 @@
Executive Summary WAITING
-
No executive summary yet — the Ollama summarizer is idle until a model is chosen in Settings and LLM_URL is reachable.
+
No executive summary yet — the Nous summarizer is idle until NOUS_API_KEY is set on the Pi and a model is chosen in Settings.
@@ -968,7 +968,7 @@

API Keys

-

Keys used by ingest services (FIRMS, Telegram, AIS, …). Stored in Postgres, never shown again — only status + last 4 chars.

+

Keys used by ingest services (FIRMS, Nous Portal, Telegram, AIS, …). Stored in Postgres, never shown again — only status + last 4 chars.

@@ -997,11 +997,11 @@

Takes effect on the next hourly run (:05). No container restart.

- Local Ollama (LLM_URL) + Nous Portal (inference-api.nousresearch.com)
- none — Ollama on the GPU host + set in API Keys as NOUS_API_KEY (never shown here)
@@ -1370,7 +1370,7 @@ function renderNewsSummary(summaries) { const badgeEl = document.getElementById('ns-badge'); if (!bodyEl) return; if (!summaries || !summaries.length) { - bodyEl.innerHTML = '
No executive summary yet — the Ollama summarizer is idle until a model is chosen in Settings and LLM_URL is reachable.
'; + bodyEl.innerHTML = '
No executive summary yet — the Nous summarizer is idle until NOUS_API_KEY is set on the Pi and a model is chosen in Settings.
'; timeEl.textContent = '—'; badgeEl.textContent = 'WAITING'; return; @@ -1547,6 +1547,7 @@ function sentimentBadge(label) { /* ═══════════════ API KEYS ═══════════════ */ const KEY_PATTERNS = { 'FIRMS_MAP_KEY': /^[0-9a-fA-F]{32}$/, + 'NOUS_API_KEY': /^.{16,}$/, 'TELEGRAM_TOKEN': /^\d{8,10}:[0-9A-Za-z_-]{35}$/, }; async function loadKeys() { diff --git a/docker-compose.yml b/docker-compose.yml index 400c6b8..aa674f7 100644 --- a/docker-compose.yml +++ b/docker-compose.yml @@ -224,7 +224,8 @@ services: DB_HOST: db DB_PORT: ${DB_PORT:-5432} DB_NAME: ${DB_NAME:-osint_data} - LLM_URL: ${LLM_URL:-} + NOUS_API_KEY: ${NOUS_API_KEY:-} + NOUS_BASE_URL: ${NOUS_BASE_URL:-https://inference-api.nousresearch.com/v1} SUMMARY_MODEL: ${SUMMARY_MODEL:-} OSINT_USER_AGENT: ${OSINT_USER_AGENT:-osint-dashboard-news-summarizer} BATCH_SIZE: ${NEWS_BATCH_SIZE:-50} diff --git a/docs/news.md b/docs/news.md index 41e5ced..43a39c0 100644 --- a/docs/news.md +++ b/docs/news.md @@ -1,12 +1,11 @@ -# News pipeline — scraper + local Ollama summarizer +# News pipeline — scraper + Nous Portal summarizer The OSINT dashboard ingests ~257 global news RSS sources hourly and produces an English LLM brief plus flagged ticker/map rows. Both services were vendored from the upstream `~/Projects/newsPipeline` project and re-integrated here to replace the old k8s CronJob choreography with in-compose scheduling against -the EXISTING osint-db — **no second Postgres**. The LLM is **local Ollama** -(`LLM_URL`, default `http://127.0.0.1:11434`) — the newsPipeline `local_llm` -path (`POST /api/generate`). No API key. +the EXISTING osint-db — **no second Postgres**. The LLM is **Nous Portal** +(`inference-api.nousresearch.com`) — not Gemini. ## Architecture @@ -17,7 +16,7 @@ path (`POST /api/generate`). No API key. news-scraper (Scrapy, hourly :00) ──► articles table (osint-db) │ │ │ ▼ -news-summarizer (Ollama map-reduce, :05) ──► article_summaries + news_items +news-summarizer (Nous Portal map-reduce, :05) ──► article_summaries + news_items │ ▼ GET /api/news · /api/news/summaries · /api/news/ticker · /api/news/map @@ -47,8 +46,8 @@ feeds (`GET /api/news` exact key set is unchanged on purpose). (`ON CONFLICT (url) DO NOTHING`). 2. **Summarizer** — `news/summerizer/run_news_summarizer.py` runs `summarizer.py` at :05 past each hour. It reads articles from the last - `SUMMARY_WINDOW_HOURS`, map-reduces them through local Ollama - (`SUMMARY_MODEL` / Settings, default `qwen3:30b-a3b`), writes the English + `SUMMARY_WINDOW_HOURS`, map-reduces them through Nous Portal + (`SUMMARY_MODEL` / Settings, default `Hermes-4.3-36B`), writes the English brief to `article_summaries` (column `model` is the LLM id), and flagged ticker/map rows to `news_items`. @@ -68,16 +67,17 @@ Container startup order doesn't matter. ## Keys and Settings -- **No API key.** Ollama on the GPU host; the summarizer container calls - `LLM_URL` (`POST /api/generate`, `GET /api/tags`). -- **Idle without Ollama** — if `/api/tags` is down or the chosen model is not - pulled, the summarizer logs and idles (never crashes). News intel APIs - return `[]`. +- **`NOUS_API_KEY`** — paste in the dashboard **Keys** UI (`api_keys` / + `keystore.KEY_REGISTRY`). Env / `.env` is an **override** (env wins, same + as FIRMS). Never returned by any API; never emitted into `index.html`; + never proxied from the browser. +- **Idle without a key** — if env is unset **and** the keystore row is empty, + the summarizer logs and idles (never crashes). News intel APIs return `[]`. - **Model** — non-secret. Settings UI model selector `PUT /api/settings` `{ "summary_model": "…" }` stores `SUMMARY_MODEL` in `app_settings` (1–128 - chars). `GET /api/settings` echoes `{summary_model, llm_url}`. - `llm_url` is read-only. Default `qwen3:30b-a3b`. Live catalog is - best-effort `GET /api/news/models` (`GET {LLM_URL}/api/tags`). + chars). `GET /api/settings` echoes `{summary_model, nous_base_url}`. + `nous_base_url` is read-only. Default `Hermes-4.3-36B`. Live catalog is + best-effort `GET /api/news/models`. ## Endpoints @@ -192,15 +192,15 @@ Never 502s. `{ "source": "live"|"fallback", "models": [{"id": "…"}] }`. ### GET /api/settings · PUT /api/settings ```json -{ "summary_model": "qwen3:30b-a3b", "llm_url": "http://127.0.0.1:11434" } +{ "summary_model": "Hermes-4.3-36B", "nous_base_url": "https://inference-api.nousresearch.com/v1" } ``` -PUT body is `{ "summary_model": "<1–128 char id>" }`. `llm_url` is +PUT body is `{ "summary_model": "<1–128 char id>" }`. `nous_base_url` is ignored even if sent. ## Reduce JSON contract -Reduce phase (`format: json` on Ollama generate, English only) must be a single +Reduce phase (`response_format: json_object`, English only) must be a single object. Parser (`intel.parse_reduce_json`) strips `` and markdown json fences, then brace-slices: @@ -234,8 +234,9 @@ lands in `article_summaries.summary_text`. | Var | Default | Notes | |---|---|---| -| `LLM_URL` | `http://127.0.0.1:11434` | Ollama origin. Pi container must point at the GPU host (e.g. Tailscale). Compose leaves this empty so `.env` wins. | -| `SUMMARY_MODEL` | `qwen3:30b-a3b` | Leave compose unset so Settings → `app_settings.SUMMARY_MODEL` reaches the worker. Env wins when set. | +| `NOUS_API_KEY` | *(blank)* | **Required for summaries.** Prefer Keys UI; env overrides. Unset in **both** env and `api_keys` = summarizer logs and idles (never crashes); APIs return `[]`. | +| `NOUS_BASE_URL` | `https://inference-api.nousresearch.com/v1` | Read-only in Settings. | +| `SUMMARY_MODEL` | `Hermes-4.3-36B` | Compose default. Operator-facing choice is Settings → `app_settings.SUMMARY_MODEL`. | | `NEWS_BATCH_SIZE` | `50` | Articles per map-phase batch (compose maps to container `BATCH_SIZE`). | | `SUMMARY_WINDOW_HOURS` | `1` | How far back the summarizer looks for new articles. | | `INCLUDE_FUTURES` | `0` | Legacy futures-prices coupling (upstream pipeline). OFF for OSINT; set `1` + install `yfinance` to enable. | @@ -245,14 +246,15 @@ lands in `article_summaries.summary_text`. | `NEWS_SUMMARIZE_RUN_ON_START` | `1` | Run one summarize immediately on container start. | | `NEWS_SUMMARIZE_FORCE` | `0` | `1` ignores the current-UTC-hour idempotency skip (double-pins on recreate). | | `NEWS_LOG_LEVEL` | `INFO` | Scrapy log level. | -| `OSINT_USER_AGENT` | `osint-dashboard-news-summarizer` | Sent on every outbound Ollama call. | +| `OSINT_USER_AGENT` | `osint-dashboard-news-summarizer` | Sent on every outbound Nous call. | | `TELEGRAM_TOKEN` / `TELEGRAM_CHAT_ID` | *(blank)* | Reserved for the (out-of-scope) Telegram delivery bot. | DB_* for both services is mapped to the shared osint-db credentials (`DB_HOST=db`, same `DB_USER/DB_PASSWORD/DB_NAME` as the rest of the stack). -Ollama generate: `POST {LLM_URL}/api/generate` via `news/summerizer/ollama_client.py` -(`httpx`, no `ollama` Python SDK). No API key. Reduce uses `format: json`. +Nous chat: `POST {NOUS_BASE_URL}/chat/completions` via `news/summerizer/nous_client.py` +(`httpx`, no `openai` SDK). Auth is a Bearer token from `NOUS_API_KEY`. +No Hermes-4 reasoning system prompt. Reduce uses `json_mode=True`. ## Prompts @@ -266,7 +268,7 @@ language is gated behind `INCLUDE_FUTURES=1`. ```bash PYTHONPATH=news/summerizer pytest news/summerizer/tests -v -# intel + ollama_client tests PASS (no network) +# intel + nous_client tests PASS (no network) PYTHONPATH=app pytest tests/test_api_news.py \ tests/test_api_settings.py tests/test_api_live_layers.py -v @@ -278,17 +280,16 @@ PYTHONPATH=app pytest tests/test_api_news.py \ After deploy / compose rebuild of `news-summarizer` on the Pi: -1. Run Ollama on the GPU host and pull the model (`ollama pull qwen3:30b-a3b`). -2. Set `LLM_URL` on the Pi to that host (Tailscale IP + `:11434`). -3. Settings: pick a model → Save → `GET /api/settings` echoes it. -4. `docker compose --profile ingest logs -f news-summarizer` — next run (or +1. Keys UI: save `NOUS_API_KEY` → status `****last4`. +2. Settings: pick a model → Save → `GET /api/settings` echoes it. +3. `docker compose --profile ingest logs -f news-summarizer` — next run (or `NEWS_SUMMARIZE_RUN_ON_START=1` recreate) logs `Processing N articles with `. -5. `curl -s localhost:8000/api/news/summaries?limit=1` — English `summary_text`, `model` set. -6. `curl -s localhost:8000/api/news/ticker` — flagged headlines only. -7. `curl -s localhost:8000/api/news/map` — only rows with lat/lon. -8. HUD: NEWS ticker scrolls flagged items; map overlay pins popup with location. -9. Ollama down / model missing → summarizer logs idle, APIs return `[]`, no crash. +4. `curl -s localhost:8000/api/news/summaries?limit=1` — English `summary_text`, `model` set. +5. `curl -s localhost:8000/api/news/ticker` — flagged headlines only. +6. `curl -s localhost:8000/api/news/map` — only rows with lat/lon. +7. HUD: NEWS ticker scrolls flagged items; map overlay pins popup with location. +8. Unset key + empty keystore → summarizer logs idle, APIs return `[]`, no crash. -**Operator action after merge:** start Ollama on the laptop GPU, set `LLM_URL` -in Pi `.env`, pick a model in Settings if the default `qwen3:30b-a3b` is not -wanted; rebuild `osint-news-summarizer` on the Pi (`pi-app-deploy` / compose). +**Operator action after merge:** paste a Nous Portal API key in API Keys; pick +a model in Settings if the default `Hermes-4.3-36B` is not wanted; rebuild +`osint-news-summarizer` on the Pi (`pi-app-deploy` / compose). diff --git a/news/summerizer/nous_client.py b/news/summerizer/nous_client.py new file mode 100644 index 0000000..da0b2c4 --- /dev/null +++ b/news/summerizer/nous_client.py @@ -0,0 +1,36 @@ +"""HTTP client for the Nous inference chat completions API.""" + +from __future__ import annotations + +import os + +import httpx + +_DEFAULT_UA = "osint-dashboard-news-summarizer" +_DEFAULT_BASE = "https://inference-api.nousresearch.com/v1" + + +def chat(prompt, *, api_key, model, base_url, json_mode=False) -> str: + resolved = (base_url or os.environ.get("NOUS_BASE_URL", _DEFAULT_BASE)).rstrip("/") + url = f"{resolved}/chat/completions" + headers = { + "Authorization": f"Bearer {api_key}", + "User-Agent": os.environ.get("OSINT_USER_AGENT") or _DEFAULT_UA, + } + payload = { + "model": model, + "messages": [{"role": "user", "content": prompt}], + "temperature": 0.2, + "max_tokens": 4096, + } + if json_mode: + payload["response_format"] = {"type": "json_object"} + try: + with httpx.Client(timeout=60.0) as client: + resp = client.post(url, headers=headers, json=payload) + if resp.status_code == 401 or resp.status_code >= 500: + return "" + data = resp.json() + return data["choices"][0]["message"]["content"] + except Exception: + return "" diff --git a/news/summerizer/ollama_client.py b/news/summerizer/ollama_client.py deleted file mode 100644 index 057b437..0000000 --- a/news/summerizer/ollama_client.py +++ /dev/null @@ -1,102 +0,0 @@ -"""HTTP client for a local Ollama generate API. - -Matches the newsPipeline local_llm summarizer: POST /api/generate, -GET /api/tags. No API key. ``LLM_URL`` may be a host, ``/api``, or -``/api/generate`` — we normalize to the origin. -""" - -from __future__ import annotations - -import os - -import httpx - -_DEFAULT_UA = "osint-dashboard-news-summarizer" -_DEFAULT_BASE = "http://127.0.0.1:11434" -_GENERATE_TIMEOUT = 300.0 -_TAGS_TIMEOUT = 8.0 - -_GENERATE_OPTIONS = { - "num_predict": 4096, - "temperature": 0.6, - "top_p": 0.9, - "top_k": 40, - "num_ctx": 32768, - "repeat_penalty": 1.1, -} - - -def normalize_base(url: str | None) -> str: - """Strip path suffixes so LLM_URL variants share one origin.""" - raw = (url or os.environ.get("LLM_URL", "") or _DEFAULT_BASE).strip() - raw = raw.rstrip("/") - for suffix in ("/api/generate", "/api/chat", "/api", "/v1"): - if raw.endswith(suffix): - raw = raw[: -len(suffix)].rstrip("/") - break - return raw or _DEFAULT_BASE - - -def _headers() -> dict[str, str]: - return { - "User-Agent": os.environ.get("OSINT_USER_AGENT") or _DEFAULT_UA, - "Accept": "application/json", - } - - -def chat(prompt: str, *, model: str, base_url: str | None = None, json_mode: bool = False) -> str: - """POST /api/generate. Returns response text, or \"\" on any failure.""" - if not (model or "").strip(): - return "" - origin = normalize_base(base_url) - url = f"{origin}/api/generate" - payload: dict = { - "model": model.strip(), - "prompt": prompt, - "stream": False, - "options": dict(_GENERATE_OPTIONS), - } - if json_mode: - payload["format"] = "json" - try: - with httpx.Client(timeout=_GENERATE_TIMEOUT) as client: - resp = client.post(url, headers=_headers(), json=payload) - if resp.status_code == 401 or resp.status_code >= 500: - return "" - data = resp.json() - text = data.get("response") if isinstance(data, dict) else None - return text if isinstance(text, str) else "" - except Exception: - return "" - - -def list_tags(base_url: str | None = None) -> list[str]: - """GET /api/tags model names. Empty list on failure.""" - origin = normalize_base(base_url) - url = f"{origin}/api/tags" - try: - with httpx.Client(timeout=_TAGS_TIMEOUT) as client: - resp = client.get(url, headers=_headers()) - if resp.status_code != 200: - return [] - models = (resp.json() or {}).get("models") or [] - names: list[str] = [] - for item in models: - if isinstance(item, dict): - name = item.get("name") or item.get("model") or "" - if name: - names.append(str(name)) - return names - except Exception: - return [] - - -def ollama_ready(model: str, *, base_url: str | None = None) -> bool: - """True when /api/tags is up and ``model`` is present (prefix match).""" - if not (model or "").strip(): - return False - names = list_tags(base_url) - if not names: - return False - want = model.strip() - return any(n == want or n.startswith(want) for n in names) diff --git a/news/summerizer/run_news_summarizer.py b/news/summerizer/run_news_summarizer.py index 483ea9b..91740c6 100644 --- a/news/summerizer/run_news_summarizer.py +++ b/news/summerizer/run_news_summarizer.py @@ -1,11 +1,16 @@ #!/usr/bin/env python3 """Scheduler loop for the news summarizer — hourly summarize at minute :05. +Replaces the k8s CronJob (`5 * * * *`) with an in-compose loop. Runs once on +boot (catches up on any articles scraped since the last summary), then fires +at each :NEWS_SUMMARIZE_MINUTE wall-clock boundary. + +The loop is serial, so a slow LLM pass never overlaps the next run. + Env (all optional, 12-factor): NEWS_SUMMARIZE_MINUTE minute of the hour to fire (default 5) NEWS_SUMMARIZE_RUN_ON_START "1" to summarize once immediately on boot (default 1) - LLM_URL Ollama origin; default http://127.0.0.1:11434 - SUMMARY_MODEL Ollama tag (else app_settings) + NOUS_API_KEY optional in env; Keys UI / api_keys also works """ from __future__ import annotations @@ -41,10 +46,13 @@ def run_summarize() -> None: def main() -> None: - llm_url = os.getenv("LLM_URL", "").strip() or "http://127.0.0.1:11434" + if not os.getenv("NOUS_API_KEY", "").strip(): + logger.warning( + "NOUS_API_KEY unset in env — will read api_keys on each run; idle if both empty" + ) logger.info( - "news summarizer loop starting (minute=%s, run_on_start=%s, llm_url=%s)", - MINUTE, RUN_ON_START, llm_url, + "news summarizer loop starting (minute=%s, run_on_start=%s)", + MINUTE, RUN_ON_START, ) if RUN_ON_START: run_summarize() diff --git a/news/summerizer/summarizer.py b/news/summerizer/summarizer.py index 196f3b9..ea82a37 100644 --- a/news/summerizer/summarizer.py +++ b/news/summerizer/summarizer.py @@ -1,17 +1,19 @@ #!/usr/bin/env python3 -"""News summarizer — local Ollama map-reduce into brief/ticker/map. +"""News summarizer — Nous map-reduce of scraped articles into brief/ticker/map. Reads articles scraped within the last hour from the shared `articles` table, -maps them with a local Ollama model (per-article English fact blocks), reduces -to one JSON object (summary_en + ticker + map_items), and stores the brief in +maps them with Nous (per-article English fact blocks), reduces to one JSON +object (summary_en + ticker + map_items), and stores the brief in `article_summaries` plus flagged rows in `news_items`. Tables live in the EXISTING osint-db (alembic 003_news + 005_news_items, idempotent). -LLM is the newsPipeline local_llm path: Ollama POST /api/generate. No API key. +Everything is env-driven (12-factor). Secrets/config are resolved at the start +of each summarize_news() — env wins, else api_keys / app_settings: DB_HOST / DB_NAME / DB_USER / DB_PASSWORD / DB_PORT PostgreSQL (osint-db) - LLM_URL Ollama origin (default http://127.0.0.1:11434) - SUMMARY_MODEL Ollama tag (else app_settings; default qwen3:30b-a3b) + NOUS_API_KEY Nous Portal key (else api_keys.name='NOUS_API_KEY') + NOUS_BASE_URL default https://inference-api.nousresearch.com/v1 + SUMMARY_MODEL default Hermes-4.3-36B (else app_settings) BATCH_SIZE articles per map-phase batch (default 50) SUMMARY_WINDOW_HOURS look-back window in hours (default 1) OSINT_USER_AGENT default osint-dashboard-news-summarizer @@ -19,6 +21,11 @@ LLM is the newsPipeline local_llm path: Ollama POST /api/generate. No API key. SUMMARY_PROMPT override reduce-phase prompt (uses {final_input}) NEWS_SUMMARIZE_FORCE "1" to ignore the current-UTC-hour idempotency skip INCLUDE_FUTURES "1" to prepend live futures prices (default 0) + +The futures/markets coupling from the original pipeline is gated behind +INCLUDE_FUTURES and OFF by default — it is irrelevant to the OSINT dashboard +and pulled yfinance into the image. Re-enable by installing yfinance and +setting INCLUDE_FUTURES=1. """ from __future__ import annotations @@ -30,7 +37,7 @@ from datetime import datetime import psycopg2 from intel import parse_reduce_json, select_map, select_ticker -from ollama_client import chat, ollama_ready +from nous_client import chat logging.basicConfig(level=logging.INFO, format="%(asctime)s %(levelname)s %(message)s") logger = logging.getLogger("news.summarizer") @@ -44,8 +51,8 @@ DB_CONFIG = { "port": int(os.getenv("DB_PORT", "5432")), } -DEFAULT_LLM_URL = "http://127.0.0.1:11434" -DEFAULT_SUMMARY_MODEL = "qwen3:30b-a3b" +DEFAULT_NOUS_BASE_URL = "https://inference-api.nousresearch.com/v1" +DEFAULT_SUMMARY_MODEL = "Hermes-4.3-36B" BATCH_SIZE = int(os.getenv("BATCH_SIZE", "50")) SUMMARY_WINDOW_HOURS = int(os.getenv("SUMMARY_WINDOW_HOURS", "1")) INCLUDE_FUTURES = os.getenv("INCLUDE_FUTURES", "0").lower() in ("1", "true", "yes") @@ -127,6 +134,20 @@ def _kv(conn, table, name) -> str: return (row[0] or "").strip() if row else "" +def resolve_api_key() -> str: + env = os.getenv("NOUS_API_KEY", "").strip() + if env: + return env + try: + conn = psycopg2.connect(**DB_CONFIG) + try: + return _kv(conn, "api_keys", "NOUS_API_KEY") + finally: + conn.close() + except Exception: # noqa: BLE001 + return "" + + def resolve_model() -> str: env = os.getenv("SUMMARY_MODEL", "").strip() if env: @@ -143,15 +164,15 @@ def resolve_model() -> str: def resolve_base_url() -> str: - return os.getenv("LLM_URL", DEFAULT_LLM_URL).strip() or DEFAULT_LLM_URL + return os.getenv("NOUS_BASE_URL", DEFAULT_NOUS_BASE_URL).strip() or DEFAULT_NOUS_BASE_URL -def call_llm(prompt: str, *, model: str, base_url: str, json_mode: bool = False) -> str: - """Send a prompt to Ollama generate and return the text (\"\" on failure).""" - if not model: - logger.warning("SUMMARY_MODEL empty — skipping LLM call") +def call_llm(prompt: str, *, api_key: str, model: str, base_url: str, json_mode: bool = False) -> str: + """Send a prompt to Nous chat completions and return the text (\"\" on failure).""" + if not api_key: + logger.warning("NOUS_API_KEY not set — skipping LLM call") return "" - return chat(prompt, model=model, base_url=base_url, json_mode=json_mode) + return chat(prompt, api_key=api_key, model=model, base_url=base_url, json_mode=json_mode) # ── Futures (legacy, gated) ──────────────────────────────────────────────── @@ -405,16 +426,11 @@ def summarize_news() -> None: ) return + api_key = resolve_api_key() model = resolve_model() base_url = resolve_base_url() - if not model: - logger.warning("SUMMARY_MODEL empty — idle this run") - return - if not ollama_ready(model, base_url=base_url): - logger.warning( - "Ollama not ready at %s (model %s missing or /api/tags down) — idle this run", - base_url, model, - ) + if not api_key: + logger.warning("NOUS_API_KEY unset in env and api_keys — idle this run") return articles = get_recent_news() @@ -436,6 +452,7 @@ def summarize_news() -> None: ) summary = call_llm( build_map_prompt(batch), + api_key=api_key, model=model, base_url=base_url, json_mode=False, @@ -451,6 +468,7 @@ def summarize_news() -> None: logger.info("reduce phase over %d partial summaries", len(partial_summaries)) master_raw = call_llm( build_master_prompt(final_input), + api_key=api_key, model=model, base_url=base_url, json_mode=True, diff --git a/news/summerizer/tests/test_nous_client.py b/news/summerizer/tests/test_nous_client.py new file mode 100644 index 0000000..002faa4 --- /dev/null +++ b/news/summerizer/tests/test_nous_client.py @@ -0,0 +1,97 @@ +from unittest.mock import MagicMock + +import httpx +from nous_client import chat + +DEFAULT_UA = "osint-dashboard-news-summarizer" +BASE = "https://inference-api.nousresearch.com/v1" + + +def _ok_response(content="hello"): + resp = MagicMock() + resp.status_code = 200 + resp.json.return_value = {"choices": [{"message": {"content": content}}]} + return resp + + +def _install_fake(monkeypatch, post_impl): + captured = {} + + class FakeClient: + def __init__(self, timeout=None, **kwargs): + captured["timeout"] = timeout + + def __enter__(self): + return self + + def __exit__(self, *exc): + return False + + def post(self, url, *, headers=None, json=None, **kwargs): + captured["url"] = url + captured["headers"] = headers + captured["json"] = json + return post_impl(url, headers, json) + + monkeypatch.setattr(httpx, "Client", FakeClient) + return captured + + +def test_posts_chat_completions_with_auth_body_and_returns_content(monkeypatch): + captured = _install_fake(monkeypatch, lambda *a: _ok_response("the-content")) + out = chat( + "summarize this", + api_key="secret-key", + model="hermes-3", + base_url=BASE, + ) + assert out == "the-content" + assert captured["url"] == f"{BASE}/chat/completions" + assert captured["headers"]["Authorization"] == "Bearer secret-key" + assert captured["headers"]["User-Agent"] == DEFAULT_UA + assert captured["json"]["model"] == "hermes-3" + assert captured["json"]["messages"] == [{"role": "user", "content": "summarize this"}] + assert captured["json"]["temperature"] == 0.2 + assert captured["json"]["max_tokens"] == 4096 + assert "response_format" not in captured["json"] + + +def test_user_agent_equals_osint_user_agent_env(monkeypatch): + monkeypatch.setenv("OSINT_USER_AGENT", "custom-ua/2.0") + captured = _install_fake(monkeypatch, lambda *a: _ok_response("ok")) + chat("p", api_key="k", model="m", base_url=BASE) + assert captured["headers"]["User-Agent"] == "custom-ua/2.0" + + +def test_json_mode_sets_response_format(monkeypatch): + captured = _install_fake(monkeypatch, lambda *a: _ok_response("{}")) + chat("p", api_key="k", model="m", base_url=BASE, json_mode=True) + assert captured["json"]["response_format"] == {"type": "json_object"} + + +def test_401_returns_empty_string(monkeypatch): + def post_impl(*a): + resp = MagicMock() + resp.status_code = 401 + return resp + + _install_fake(monkeypatch, post_impl) + assert chat("p", api_key="bad", model="m", base_url=BASE) == "" + + +def test_5xx_returns_empty_string(monkeypatch): + def post_impl(*a): + resp = MagicMock() + resp.status_code = 503 + return resp + + _install_fake(monkeypatch, post_impl) + assert chat("p", api_key="k", model="m", base_url=BASE) == "" + + +def test_timeout_returns_empty_string(monkeypatch): + def post_impl(*a): + raise httpx.TimeoutException("timed out") + + _install_fake(monkeypatch, post_impl) + assert chat("p", api_key="k", model="m", base_url=BASE) == "" diff --git a/news/summerizer/tests/test_ollama_client.py b/news/summerizer/tests/test_ollama_client.py deleted file mode 100644 index 8339c86..0000000 --- a/news/summerizer/tests/test_ollama_client.py +++ /dev/null @@ -1,111 +0,0 @@ -from unittest.mock import MagicMock - -import httpx -from ollama_client import chat, list_tags, normalize_base, ollama_ready - -DEFAULT_UA = "osint-dashboard-news-summarizer" -BASE = "http://127.0.0.1:11434" - - -def _ok_response(content="hello"): - resp = MagicMock() - resp.status_code = 200 - resp.json.return_value = {"response": content} - return resp - - -def _install_fake(monkeypatch, post_impl=None, get_impl=None): - captured = {} - - class FakeClient: - def __init__(self, timeout=None, **kwargs): - captured["timeout"] = timeout - - def __enter__(self): - return self - - def __exit__(self, *exc): - return False - - def post(self, url, *, headers=None, json=None, **kwargs): - captured["url"] = url - captured["headers"] = headers - captured["json"] = json - if post_impl is None: - raise AssertionError("post not expected") - return post_impl(url, headers, json) - - def get(self, url, *, headers=None, **kwargs): - captured["url"] = url - captured["headers"] = headers - if get_impl is None: - raise AssertionError("get not expected") - return get_impl(url, headers) - - monkeypatch.setattr(httpx, "Client", FakeClient) - return captured - - -def test_normalize_base_strips_generate_path(): - assert normalize_base("http://ollama:11434/api/generate") == "http://ollama:11434" - assert normalize_base("http://ollama:11434/api/") == "http://ollama:11434" - assert normalize_base("http://ollama:11434") == "http://ollama:11434" - - -def test_posts_generate_and_returns_response(monkeypatch): - captured = _install_fake(monkeypatch, post_impl=lambda *a: _ok_response("the-content")) - out = chat("summarize this", model="qwen3:30b-a3b", base_url=BASE) - assert out == "the-content" - assert captured["url"] == f"{BASE}/api/generate" - assert captured["headers"]["User-Agent"] == DEFAULT_UA - assert captured["json"]["model"] == "qwen3:30b-a3b" - assert captured["json"]["prompt"] == "summarize this" - assert captured["json"]["stream"] is False - assert "format" not in captured["json"] - assert captured["json"]["options"]["num_ctx"] == 32768 - assert captured["timeout"] == 300.0 - - -def test_json_mode_sets_format_json(monkeypatch): - captured = _install_fake(monkeypatch, post_impl=lambda *a: _ok_response("{}")) - chat("p", model="m", base_url=BASE, json_mode=True) - assert captured["json"]["format"] == "json" - - -def test_empty_model_returns_empty(monkeypatch): - assert chat("p", model="", base_url=BASE) == "" - - -def test_5xx_returns_empty_string(monkeypatch): - def post_impl(*a): - resp = MagicMock() - resp.status_code = 503 - return resp - - _install_fake(monkeypatch, post_impl=post_impl) - assert chat("p", model="m", base_url=BASE) == "" - - -def test_timeout_returns_empty_string(monkeypatch): - def post_impl(*a): - raise httpx.TimeoutException("timed out") - - _install_fake(monkeypatch, post_impl=post_impl) - assert chat("p", model="m", base_url=BASE) == "" - - -def test_list_tags_and_ready(monkeypatch): - def get_impl(*a): - resp = MagicMock() - resp.status_code = 200 - resp.json.return_value = { - "models": [{"name": "qwen3:30b-a3b"}, {"name": "llama3.2:latest"}], - } - return resp - - captured = _install_fake(monkeypatch, get_impl=get_impl) - names = list_tags(BASE) - assert names == ["qwen3:30b-a3b", "llama3.2:latest"] - assert captured["url"] == f"{BASE}/api/tags" - assert ollama_ready("qwen3:30b-a3b", base_url=BASE) is True - assert ollama_ready("missing", base_url=BASE) is False diff --git a/tests/test_api_settings.py b/tests/test_api_settings.py index 5d2ec40..6dad2f5 100644 --- a/tests/test_api_settings.py +++ b/tests/test_api_settings.py @@ -64,26 +64,31 @@ def _put(path: str, json: dict) -> httpx.Response: return _request("PUT", path, json=json) -def test_get_news_models_ollama_down_returns_fallback(monkeypatch): - async def boom_http_get(url, *, headers, timeout): - raise httpx.ConnectError("ollama down") - - import settings_store - monkeypatch.setattr(settings_store, "_http_get", boom_http_get) - settings_store._models_cache = None +def test_get_news_models_without_key_returns_fallback(monkeypatch): + monkeypatch.delenv("NOUS_API_KEY", raising=False) + monkeypatch.setattr("keystore.get_api_key", _missing_key) resp = _get("/api/news/models") assert resp.status_code == 200 body = resp.json() assert body["source"] == "fallback" ids = [m["id"] for m in body["models"]] + assert "Hermes-4.3-36B" in ids from settings_store import FALLBACK_MODELS assert ids == FALLBACK_MODELS - assert "qwen3:30b-a3b" in ids -def test_get_news_models_live_from_ollama_tags(monkeypatch): - monkeypatch.setenv("LLM_URL", "http://ollama:11434") +async def _missing_key(name: str): + return None + + +async def _present_key(name: str): + return "test-nous-api-key-1234" + + +def test_get_news_models_live_from_upstream(monkeypatch): + monkeypatch.setattr("keystore.get_api_key", _present_key) + monkeypatch.setenv("NOUS_BASE_URL", "https://inference-api.nousresearch.com/v1") monkeypatch.delenv("OSINT_USER_AGENT", raising=False) captured: dict = {} @@ -95,7 +100,7 @@ def test_get_news_models_live_from_ollama_tags(monkeypatch): return None def json(self): - return {"models": [{"name": "qwen3:30b-a3b"}, {"name": "llama3.2:latest"}]} + return {"data": [{"id": "live-model-a"}, {"id": "Hermes-4.3-36B"}]} async def fake_http_get(url, *, headers, timeout): captured["url"] = url @@ -112,10 +117,30 @@ def test_get_news_models_live_from_ollama_tags(monkeypatch): body = resp.json() assert body["source"] == "live" ids = [m["id"] for m in body["models"]] - assert ids == ["qwen3:30b-a3b", "llama3.2:latest"] - assert captured["url"] == "http://ollama:11434/api/tags" + assert ids == ["live-model-a", "Hermes-4.3-36B"] + assert captured["url"] == "https://inference-api.nousresearch.com/v1/models" assert captured["headers"]["User-Agent"] == "osint-dashboard-news-summarizer" - assert "Authorization" not in captured["headers"] + assert captured["headers"]["Authorization"] == "Bearer test-nous-api-key-1234" + timeout = captured["timeout"] + assert timeout == 8 or getattr(timeout, "read", timeout) == 8 or float(timeout) == 8.0 + + +def test_get_news_models_upstream_failure_returns_fallback(monkeypatch): + monkeypatch.setattr("keystore.get_api_key", _present_key) + + async def boom_http_get(url, *, headers, timeout): + raise httpx.ConnectError("upstream down") + + import settings_store + monkeypatch.setattr(settings_store, "_http_get", boom_http_get) + settings_store._models_cache = None + + resp = _get("/api/news/models") + assert resp.status_code == 200 + body = resp.json() + assert body["source"] == "fallback" + ids = [m["id"] for m in body["models"]] + assert "Hermes-4.3-36B" in ids def test_put_settings_empty_returns_422(): @@ -131,17 +156,17 @@ def test_put_settings_whitespace_only_returns_422(): @requires_db def test_put_settings_round_trip(clean_settings, monkeypatch): monkeypatch.delenv("SUMMARY_MODEL", raising=False) - monkeypatch.delenv("LLM_URL", raising=False) + monkeypatch.delenv("NOUS_BASE_URL", raising=False) - put_resp = _put("/api/settings", {"summary_model": "qwen3:30b-a3b"}) + put_resp = _put("/api/settings", {"summary_model": "google/gemini-2.5-flash"}) assert put_resp.status_code == 200 put_body = put_resp.json() - assert put_body["summary_model"] == "qwen3:30b-a3b" - assert put_body["llm_url"] == "http://127.0.0.1:11434" + assert put_body["summary_model"] == "google/gemini-2.5-flash" + assert put_body["nous_base_url"] == "https://inference-api.nousresearch.com/v1" get_resp = _get("/api/settings") assert get_resp.status_code == 200 get_body = get_resp.json() - assert get_body["summary_model"] == "qwen3:30b-a3b" - assert get_body["llm_url"] == "http://127.0.0.1:11434" - assert set(get_body.keys()) == {"summary_model", "llm_url"} + assert get_body["summary_model"] == "google/gemini-2.5-flash" + assert get_body["nous_base_url"] == "https://inference-api.nousresearch.com/v1" + assert set(get_body.keys()) == {"summary_model", "nous_base_url"}