From 69f1a6d44f1e492f4d00dc3d71915e07654b0fa2 Mon Sep 17 00:00:00 2001 From: Marc Shade Date: Tue, 24 Feb 2026 08:41:40 -0500 Subject: [PATCH] =?UTF-8?q?feat:=20NLP=20intelligence=20=E2=80=94=20entity?= =?UTF-8?q?=20extraction,=20classification,=20clustering,=20spikes=20(+4?= =?UTF-8?q?=20=3D=2064=20tools)?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Phase 10: News Intelligence & NLP - intel_extract_entities: Regex-based NER for countries, leaders, orgs, companies, CVEs, APT groups (28 leaders, 41 orgs, 25 companies, 36 APTs) - intel_classify_event: Keyword-based threat classification into 14 categories with severity scoring (1-10) - intel_news_clusters: Jaccard similarity topic clustering for news articles with keyword extraction - intel_keyword_spikes: Welford's algorithm baseline comparison with CVE/APT pattern extraction from headlines No new dependencies — all pure Python with regex and SQLite. Co-Authored-By: Claude Opus 4.6 --- src/world_intel_mcp/analysis/classifier.py | 198 +++++++++++++++++++++ src/world_intel_mcp/analysis/clustering.py | 74 ++++++++ src/world_intel_mcp/analysis/entities.py | 174 ++++++++++++++++++ src/world_intel_mcp/analysis/spikes.py | 187 +++++++++++++++++++ src/world_intel_mcp/config/entities.py | 132 ++++++++++++++ src/world_intel_mcp/server.py | 70 ++++++++ 6 files changed, 835 insertions(+) create mode 100644 src/world_intel_mcp/analysis/classifier.py create mode 100644 src/world_intel_mcp/analysis/entities.py create mode 100644 src/world_intel_mcp/analysis/spikes.py create mode 100644 src/world_intel_mcp/config/entities.py diff --git a/src/world_intel_mcp/analysis/classifier.py b/src/world_intel_mcp/analysis/classifier.py new file mode 100644 index 0000000..a3b88ee --- /dev/null +++ b/src/world_intel_mcp/analysis/classifier.py @@ -0,0 +1,198 @@ +"""Keyword-based event threat classification. + +Fast classifier with 14 categories and severity scoring. No ML deps. +Designed as the first pass; an LLM can refine confidence later. +""" + +from __future__ import annotations + +import re +from datetime import datetime, timezone + +# --------------------------------------------------------------------------- +# Category keyword patterns (ordered by specificity) +# --------------------------------------------------------------------------- + +CATEGORIES: dict[str, dict] = { + "nuclear": { + "keywords": [ + "nuclear", "atomic", "uranium", "plutonium", "warhead", + "icbm", "ballistic missile", "nuclear test", "radiation", + "enrichment", "nonproliferation", "iaea", "dirty bomb", + ], + "severity_base": 9, + }, + "terrorism": { + "keywords": [ + "terrorist", "terrorism", "suicide bomb", "car bomb", "ied", + "hostage", "kidnapping", "beheading", "mass shooting", + "lone wolf", "radicalization", "jihad", "extremist", + ], + "severity_base": 9, + }, + "military": { + "keywords": [ + "military", "troops", "soldier", "artillery", "airstrike", + "bombing", "invasion", "offensive", "deployed", "nato", + "warship", "submarine", "fighter jet", "drone strike", + "mobilization", "ceasefire", "frontline", "casualties", + ], + "severity_base": 8, + }, + "cyber": { + "keywords": [ + "cyber", "hack", "ransomware", "malware", "phishing", + "data breach", "ddos", "vulnerability", "exploit", "cve-", + "zero-day", "apt", "botnet", "credential", "encryption", + ], + "severity_base": 7, + }, + "political": { + "keywords": [ + "election", "coup", "protest", "demonstration", "sanctions", + "embargo", "diplomatic", "parliament", "congress", + "impeachment", "referendum", "regime", "martial law", + "authoritarian", "democracy", "opposition", + ], + "severity_base": 6, + }, + "economic": { + "keywords": [ + "recession", "inflation", "default", "debt crisis", + "currency collapse", "bank run", "stock crash", "layoffs", + "trade war", "tariff", "supply chain", "shortage", + ], + "severity_base": 6, + }, + "health": { + "keywords": [ + "pandemic", "epidemic", "outbreak", "virus", "pathogen", + "vaccine", "quarantine", "who emergency", "mortality", + "infection", "ebola", "bird flu", "h5n1", "mpox", + ], + "severity_base": 7, + }, + "climate": { + "keywords": [ + "hurricane", "typhoon", "cyclone", "earthquake", "tsunami", + "flood", "drought", "wildfire", "heatwave", "glacier", + "sea level", "emission", "carbon", "climate change", + ], + "severity_base": 6, + }, + "infrastructure": { + "keywords": [ + "power outage", "blackout", "pipeline", "cable cut", + "internet outage", "grid failure", "dam collapse", + "bridge collapse", "port blockade", "supply disruption", + ], + "severity_base": 7, + }, + "maritime": { + "keywords": [ + "shipping", "strait", "piracy", "naval blockade", "vessel", + "cargo ship", "tanker", "port", "chokepoint", "seafarer", + "coast guard", "maritime security", + ], + "severity_base": 5, + }, + "aviation": { + "keywords": [ + "airspace", "no-fly zone", "intercept", "air defense", + "airline", "crash", "hijack", "flight ban", + ], + "severity_base": 6, + }, + "energy": { + "keywords": [ + "oil price", "opec", "gas pipeline", "lng", "refinery", + "energy crisis", "power grid", "renewable", "solar", + "wind farm", "battery", "hydrogen", + ], + "severity_base": 5, + }, + "social_unrest": { + "keywords": [ + "riot", "looting", "tear gas", "water cannon", "curfew", + "state of emergency", "civil disobedience", "strike", + "labor dispute", "food riot", "uprising", + ], + "severity_base": 7, + }, + "space": { + "keywords": [ + "satellite", "space debris", "solar flare", "geomagnetic", + "space weather", "orbit", "launch", "rocket", "asteroid", + ], + "severity_base": 4, + }, +} + +# Severity modifiers — keywords that bump severity up +_HIGH_SEVERITY: list[str] = [ + "killed", "dead", "death toll", "massacre", "war crime", "genocide", + "nuclear", "catastrophic", "critical", "emergency", "mass casualty", + "destroyed", "collapse", "unprecedented", "worst", +] + +_MODERATE_SEVERITY: list[str] = [ + "injured", "wounded", "escalation", "crisis", "threat", "warning", + "attack", "strike", "explosion", "breach", "violation", +] + + +def classify_event(text: str) -> dict: + """Classify text into threat categories with severity scoring. + + Returns dict with primary category, severity (1-10), confidence, + all matched categories, and matched keywords. + """ + text_lower = text.lower() + matches: list[dict] = [] + + for cat_name, cat_info in CATEGORIES.items(): + matched_kw = [kw for kw in cat_info["keywords"] if kw in text_lower] + if matched_kw: + # Confidence scales with number of keyword hits + confidence = min(1.0, len(matched_kw) * 0.2 + 0.3) + matches.append({ + "category": cat_name, + "keywords_matched": matched_kw, + "keyword_count": len(matched_kw), + "severity_base": cat_info["severity_base"], + "confidence": round(confidence, 2), + }) + + # Sort by keyword count (most specific match first) + matches.sort(key=lambda m: m["keyword_count"], reverse=True) + + # Compute severity modifier + severity_mod = 0 + high_hits = [kw for kw in _HIGH_SEVERITY if kw in text_lower] + mod_hits = [kw for kw in _MODERATE_SEVERITY if kw in text_lower] + if high_hits: + severity_mod = 2 + elif mod_hits: + severity_mod = 1 + + primary = matches[0] if matches else None + severity = min(10, (primary["severity_base"] + severity_mod)) if primary else 0 + + return { + "primary_category": primary["category"] if primary else "unclassified", + "severity": severity, + "confidence": primary["confidence"] if primary else 0.0, + "all_categories": [ + {"category": m["category"], "confidence": m["confidence"], "keywords": m["keywords_matched"]} + for m in matches + ], + "category_count": len(matches), + "severity_modifiers": high_hits + mod_hits, + "source": "keyword-classifier", + "timestamp": datetime.now(timezone.utc).isoformat(), + } + + +async def fetch_classify_event(fetcher, text: str) -> dict: + """Classify a text event. Thin async wrapper for MCP dispatch.""" + return classify_event(text) diff --git a/src/world_intel_mcp/analysis/clustering.py b/src/world_intel_mcp/analysis/clustering.py index 30b0e62..9fcd8c9 100644 --- a/src/world_intel_mcp/analysis/clustering.py +++ b/src/world_intel_mcp/analysis/clustering.py @@ -96,3 +96,77 @@ def cluster_articles( # Sort by member count descending clusters.sort(key=lambda c: c["member_count"], reverse=True) return clusters + + +# --------------------------------------------------------------------------- +# MCP-facing async wrapper +# --------------------------------------------------------------------------- + +async def fetch_news_clusters( + fetcher, + category: str | None = None, + limit: int = 100, + threshold: float = 0.25, +) -> dict: + """Fetch recent news and cluster by topic similarity. + + Args: + fetcher: HTTP fetcher instance. + category: Optional RSS feed category filter. + limit: Max news items to fetch. + threshold: Jaccard similarity threshold (0.0 - 1.0). + + Returns: + Dict with clusters[], cluster_count, total_items, singleton_count, source. + """ + from datetime import datetime, timezone + from ..sources import news + + feed_data = await news.fetch_news_feed(fetcher, category=category, limit=limit) + items = feed_data.get("items", []) + + raw_clusters = cluster_articles(items, similarity_threshold=threshold) + + out_clusters = [] + for c in raw_clusters: + if c["member_count"] <= 1: + continue + rep = c["representative"] + # Collect keywords from all member titles + all_tokens: set[str] = set() + member_items = [] + for idx in c["member_indices"]: + art = items[idx] + all_tokens |= _tokenize(art.get("title", "")) + member_items.append({ + "title": art.get("title", ""), + "source": art.get("source_name", art.get("source", "")), + "link": art.get("link", ""), + }) + + # Top keywords by frequency + word_freq: dict[str, int] = {} + for idx in c["member_indices"]: + for w in _tokenize(items[idx].get("title", "")): + word_freq[w] = word_freq.get(w, 0) + 1 + top_kw = sorted(word_freq, key=word_freq.get, reverse=True)[:8] # type: ignore[arg-type] + + out_clusters.append({ + "headline": rep.get("title", ""), + "size": c["member_count"], + "keywords": top_kw, + "sources": list({m["source"] for m in member_items if m["source"]}), + "items": member_items[:5], + }) + + singletons = sum(1 for c in raw_clusters if c["member_count"] == 1) + + return { + "clusters": out_clusters, + "cluster_count": len(out_clusters), + "total_items": len(items), + "singleton_count": singletons, + "threshold": threshold, + "source": "jaccard-clustering", + "timestamp": datetime.now(timezone.utc).isoformat(), + } diff --git a/src/world_intel_mcp/analysis/entities.py b/src/world_intel_mcp/analysis/entities.py new file mode 100644 index 0000000..1549689 --- /dev/null +++ b/src/world_intel_mcp/analysis/entities.py @@ -0,0 +1,174 @@ +"""Named entity extraction from text. + +Regex-based NER for countries, leaders, organizations, companies, CVEs, +and APT groups. No ML dependencies — uses reference data from config/. +""" + +from __future__ import annotations + +import re +from datetime import datetime, timezone + +from ..config.countries import TIER1_COUNTRIES +from ..config.entities import LEADERS, ORGANIZATIONS, COMPANIES, APT_GROUPS + + +# Pre-compile patterns +_CVE_RE = re.compile(r"CVE-\d{4}-\d{4,}", re.IGNORECASE) +_APT_RE = re.compile( + r"\b(?:" + "|".join(re.escape(a) for a in sorted(APT_GROUPS, key=len, reverse=True)) + r")\b", + re.IGNORECASE, +) + +# Build country keyword → iso3 lookup (lowercase) +_COUNTRY_KW: dict[str, str] = {} +for _iso3, _info in TIER1_COUNTRIES.items(): + for _kw in _info["keywords"]: + _COUNTRY_KW[_kw.lower()] = _iso3 + _COUNTRY_KW[_info["name"].lower()] = _iso3 + _COUNTRY_KW[_iso3.lower()] = _iso3 + +# Build leader name lookup (lowercase) +_LEADER_KW: dict[str, dict] = {k.lower(): v for k, v in LEADERS.items()} + +# Build org lookup (lowercase) +_ORG_KW: dict[str, dict] = {k.lower(): v for k, v in ORGANIZATIONS.items()} + +# Build company lookup (lowercase) +_COMPANY_KW: dict[str, dict] = {k.lower(): v for k, v in COMPANIES.items()} + + +def extract_entities(text: str) -> dict: + """Extract named entities from text. + + Returns dict with entities grouped by type, plus counts and metadata. + """ + text_lower = text.lower() + words_set = set(text_lower.split()) + + countries: list[dict] = [] + leaders: list[dict] = [] + organizations: list[dict] = [] + companies: list[dict] = [] + cves: list[str] = [] + apts: list[str] = [] + + seen_countries: set[str] = set() + seen_leaders: set[str] = set() + seen_orgs: set[str] = set() + seen_companies: set[str] = set() + + # Countries (keyword search in text) + for kw, iso3 in _COUNTRY_KW.items(): + if iso3 in seen_countries: + continue + if kw in text_lower: + seen_countries.add(iso3) + info = TIER1_COUNTRIES[iso3] + countries.append({ + "iso3": iso3, + "name": info["name"], + "baseline_risk": info["baseline_risk"], + }) + + # Leaders (match longest first to avoid partial matches) + for kw in sorted(_LEADER_KW.keys(), key=len, reverse=True): + info = _LEADER_KW[kw] + if info["name"] in seen_leaders: + continue + if kw in text_lower: + seen_leaders.add(info["name"]) + leaders.append({ + "name": info["name"], + "title": info["title"], + "country": info["country"], + }) + + # Organizations + for kw in sorted(_ORG_KW.keys(), key=len, reverse=True): + info = _ORG_KW[kw] + if info["abbrev"] in seen_orgs: + continue + # For short abbreviations (2-4 chars), require word boundary + if len(kw) <= 4: + if kw in words_set: + seen_orgs.add(info["abbrev"]) + organizations.append({ + "name": info["abbrev"], + "type": info["type"], + }) + elif kw in text_lower: + seen_orgs.add(info["abbrev"]) + organizations.append({ + "name": info["abbrev"], + "type": info["type"], + }) + + # Companies + for kw in sorted(_COMPANY_KW.keys(), key=len, reverse=True): + info = _COMPANY_KW[kw] + if kw in seen_companies: + continue + if kw in text_lower: + seen_companies.add(kw) + companies.append({ + "name": kw.title(), + "ticker": info["ticker"], + "sector": info["sector"], + }) + + # CVEs + cves = list(set(_CVE_RE.findall(text))) + + # APTs + apt_matches = _APT_RE.findall(text) + apts = list(set(m.lower() for m in apt_matches)) + + total = len(countries) + len(leaders) + len(organizations) + len(companies) + len(cves) + len(apts) + + return { + "entities": { + "countries": countries, + "leaders": leaders, + "organizations": organizations, + "companies": companies, + "cves": cves, + "apt_groups": apts, + }, + "by_type": { + "countries": len(countries), + "leaders": len(leaders), + "organizations": len(organizations), + "companies": len(companies), + "cves": len(cves), + "apt_groups": len(apts), + }, + "total_entities": total, + "source": "regex-ner", + "timestamp": datetime.now(timezone.utc).isoformat(), + } + + +async def fetch_entity_extraction(fetcher, text: str | None = None, use_news: bool = True) -> dict: + """Extract entities from provided text or recent news headlines. + + If text is provided, extract from that. Otherwise fetch recent news + headlines and extract entities from the combined text. + """ + if text: + return extract_entities(text) + + if use_news: + from ..sources import news + feed_data = await news.fetch_news_feed(fetcher, limit=100) + items = feed_data.get("items", []) + combined = " ".join( + (item.get("title", "") + " " + item.get("summary", "")) + for item in items + ) + result = extract_entities(combined) + result["input_source"] = "news_feed" + result["items_analyzed"] = len(items) + return result + + return extract_entities("") diff --git a/src/world_intel_mcp/analysis/spikes.py b/src/world_intel_mcp/analysis/spikes.py new file mode 100644 index 0000000..5c78e38 --- /dev/null +++ b/src/world_intel_mcp/analysis/spikes.py @@ -0,0 +1,187 @@ +"""Keyword spike detection with CVE/APT extraction. + +Compares recent keyword frequencies against stored baselines to detect +abnormal surges. Uses SQLite for baseline persistence (like temporal.py). +""" + +from __future__ import annotations + +import logging +import os +import re +import sqlite3 +from datetime import datetime, timezone + +logger = logging.getLogger("world-intel-mcp.analysis.spikes") + +_DB_PATH = os.path.join( + os.path.expanduser("~"), ".cache", "world-intel-mcp", "keyword_spikes.db" +) + +_CVE_RE = re.compile(r"CVE-\d{4}-\d{4,}", re.IGNORECASE) +_APT_RE = re.compile( + r"\b(APT\d{1,3}|Lazarus|Fancy Bear|Cozy Bear|Sandworm|Turla|Kimsuky|" + r"Volt Typhoon|Salt Typhoon|Midnight Blizzard|Scattered Spider|" + r"LockBit|BlackCat|ALPHV|CL0P|Black Basta|Charming Kitten)\b", + re.IGNORECASE, +) + + +class KeywordSpikeDetector: + """Detects keyword frequency spikes against rolling baselines.""" + + def __init__(self, db_path: str = _DB_PATH): + os.makedirs(os.path.dirname(db_path), exist_ok=True) + self._conn = sqlite3.connect(db_path) + self._conn.execute( + """CREATE TABLE IF NOT EXISTS kw_baselines ( + keyword TEXT PRIMARY KEY, + count INTEGER NOT NULL DEFAULT 0, + mean REAL NOT NULL DEFAULT 0.0, + m2 REAL NOT NULL DEFAULT 0.0, + updated_at TEXT NOT NULL + )""" + ) + self._conn.commit() + + def _get_baseline(self, keyword: str) -> tuple[int, float, float]: + """Get (count, mean, m2) for a keyword.""" + row = self._conn.execute( + "SELECT count, mean, m2 FROM kw_baselines WHERE keyword = ?", + (keyword,), + ).fetchone() + return row if row else (0, 0.0, 0.0) + + def _update_baseline(self, keyword: str, value: float) -> None: + """Welford's online update for running mean/variance.""" + n, mean, m2 = self._get_baseline(keyword) + n += 1 + delta = value - mean + mean += delta / n + delta2 = value - mean + m2 += delta * delta2 + now = datetime.now(timezone.utc).isoformat() + self._conn.execute( + """INSERT INTO kw_baselines (keyword, count, mean, m2, updated_at) + VALUES (?, ?, ?, ?, ?) + ON CONFLICT(keyword) DO UPDATE SET + count=?, mean=?, m2=?, updated_at=?""", + (keyword, n, mean, m2, now, n, mean, m2, now), + ) + self._conn.commit() + + def detect_spikes( + self, + keyword_counts: dict[str, int], + z_threshold: float = 2.0, + ) -> list[dict]: + """Compare current keyword counts against baselines. + + Args: + keyword_counts: {keyword: count} from current window. + z_threshold: Z-score threshold for spike detection. + + Returns: + List of spike dicts sorted by z-score descending. + """ + spikes: list[dict] = [] + + for kw, current_count in keyword_counts.items(): + n, mean, m2 = self._get_baseline(kw) + + if n < 3: + # Not enough data for meaningful z-score + self._update_baseline(kw, float(current_count)) + continue + + variance = m2 / n + stddev = variance ** 0.5 + + if stddev < 0.1: + # Near-zero variance — use ratio instead + ratio = current_count / max(mean, 0.1) + if ratio > 3.0: + spikes.append({ + "keyword": kw, + "current_count": current_count, + "baseline_mean": round(mean, 2), + "ratio": round(ratio, 2), + "z_score": None, + "detection": "ratio", + }) + else: + z = (current_count - mean) / stddev + if z >= z_threshold: + spikes.append({ + "keyword": kw, + "current_count": current_count, + "baseline_mean": round(mean, 2), + "stddev": round(stddev, 2), + "z_score": round(z, 2), + "ratio": round(current_count / max(mean, 0.1), 2), + "detection": "z_score", + }) + + # Update baseline + self._update_baseline(kw, float(current_count)) + + spikes.sort(key=lambda s: s.get("z_score") or s.get("ratio", 0), reverse=True) + return spikes + + +# Module-level singleton +_detector: KeywordSpikeDetector | None = None + + +def _ensure_detector() -> KeywordSpikeDetector: + global _detector + if _detector is None: + _detector = KeywordSpikeDetector() + return _detector + + +async def fetch_keyword_spikes( + fetcher, + min_count: int = 3, + z_threshold: float = 2.0, +) -> dict: + """Fetch trending keywords and detect spikes against baselines. + + Also extracts CVE identifiers and APT group mentions from headlines. + """ + from ..sources import news + + # Get current keyword frequencies + kw_data = await news.fetch_trending_keywords(fetcher, min_count=min_count) + keywords = kw_data.get("keywords", []) + + # Build count dict + kw_counts: dict[str, int] = {} + for item in keywords: + kw_counts[item["keyword"]] = item["count"] + + # Detect spikes + detector = _ensure_detector() + spikes = detector.detect_spikes(kw_counts, z_threshold=z_threshold) + + # Extract CVEs and APTs from all recent headlines + feed_data = await news.fetch_news_feed(fetcher, limit=100) + all_text = " ".join( + (it.get("title", "") + " " + it.get("summary", "")) + for it in feed_data.get("items", []) + ) + cve_mentions = list(set(_CVE_RE.findall(all_text))) + apt_mentions = list(set(m.lower() for m in _APT_RE.findall(all_text))) + + return { + "spikes": spikes, + "spike_count": len(spikes), + "keywords_analyzed": len(kw_counts), + "z_threshold": z_threshold, + "cve_mentions": sorted(cve_mentions), + "apt_mentions": sorted(apt_mentions), + "cve_count": len(cve_mentions), + "apt_count": len(apt_mentions), + "source": "keyword-spike-detector", + "timestamp": datetime.now(timezone.utc).isoformat(), + } diff --git a/src/world_intel_mcp/config/entities.py b/src/world_intel_mcp/config/entities.py new file mode 100644 index 0000000..d24f6d6 --- /dev/null +++ b/src/world_intel_mcp/config/entities.py @@ -0,0 +1,132 @@ +"""Entity reference data for NER extraction. + +Pure data module — no I/O, no external dependencies. +Maps leaders, organizations, companies to normalized forms. +""" + +from __future__ import annotations + +# --------------------------------------------------------------------------- +# World leaders (name variants → normalized key) +# --------------------------------------------------------------------------- +LEADERS: dict[str, dict] = { + "biden": {"name": "Joe Biden", "title": "President", "country": "USA"}, + "joe biden": {"name": "Joe Biden", "title": "President", "country": "USA"}, + "trump": {"name": "Donald Trump", "title": "Former President", "country": "USA"}, + "donald trump": {"name": "Donald Trump", "title": "Former President", "country": "USA"}, + "xi jinping": {"name": "Xi Jinping", "title": "President", "country": "CHN"}, + "xi": {"name": "Xi Jinping", "title": "President", "country": "CHN"}, + "putin": {"name": "Vladimir Putin", "title": "President", "country": "RUS"}, + "vladimir putin": {"name": "Vladimir Putin", "title": "President", "country": "RUS"}, + "zelensky": {"name": "Volodymyr Zelensky", "title": "President", "country": "UKR"}, + "zelenskyy": {"name": "Volodymyr Zelensky", "title": "President", "country": "UKR"}, + "macron": {"name": "Emmanuel Macron", "title": "President", "country": "FRA"}, + "starmer": {"name": "Keir Starmer", "title": "Prime Minister", "country": "GBR"}, + "scholz": {"name": "Olaf Scholz", "title": "Chancellor", "country": "DEU"}, + "modi": {"name": "Narendra Modi", "title": "Prime Minister", "country": "IND"}, + "narendra modi": {"name": "Narendra Modi", "title": "Prime Minister", "country": "IND"}, + "netanyahu": {"name": "Benjamin Netanyahu", "title": "Prime Minister", "country": "ISR"}, + "khamenei": {"name": "Ali Khamenei", "title": "Supreme Leader", "country": "IRN"}, + "kim jong un": {"name": "Kim Jong Un", "title": "Supreme Leader", "country": "PRK"}, + "kim jong-un": {"name": "Kim Jong Un", "title": "Supreme Leader", "country": "PRK"}, + "erdogan": {"name": "Recep Tayyip Erdogan", "title": "President", "country": "TUR"}, + "kishida": {"name": "Fumio Kishida", "title": "Prime Minister", "country": "JPN"}, + "trudeau": {"name": "Justin Trudeau", "title": "Prime Minister", "country": "CAN"}, + "lula": {"name": "Luiz Inacio Lula da Silva", "title": "President", "country": "BRA"}, + "al-assad": {"name": "Bashar al-Assad", "title": "Former President", "country": "SYR"}, + "marcos": {"name": "Ferdinand Marcos Jr.", "title": "President", "country": "PHL"}, + "albanese": {"name": "Anthony Albanese", "title": "Prime Minister", "country": "AUS"}, + "milei": {"name": "Javier Milei", "title": "President", "country": "ARG"}, + "meloni": {"name": "Giorgia Meloni", "title": "Prime Minister", "country": "ITA"}, +} + +# --------------------------------------------------------------------------- +# International organizations +# --------------------------------------------------------------------------- +ORGANIZATIONS: dict[str, dict] = { + "united nations": {"abbrev": "UN", "type": "intl_org"}, + "un": {"abbrev": "UN", "type": "intl_org"}, + "nato": {"abbrev": "NATO", "type": "military_alliance"}, + "european union": {"abbrev": "EU", "type": "political_bloc"}, + "eu": {"abbrev": "EU", "type": "political_bloc"}, + "who": {"abbrev": "WHO", "type": "intl_org"}, + "world health organization": {"abbrev": "WHO", "type": "intl_org"}, + "imf": {"abbrev": "IMF", "type": "financial"}, + "international monetary fund": {"abbrev": "IMF", "type": "financial"}, + "world bank": {"abbrev": "WB", "type": "financial"}, + "opec": {"abbrev": "OPEC", "type": "energy"}, + "iaea": {"abbrev": "IAEA", "type": "nuclear"}, + "international atomic energy agency": {"abbrev": "IAEA", "type": "nuclear"}, + "red cross": {"abbrev": "ICRC", "type": "humanitarian"}, + "icrc": {"abbrev": "ICRC", "type": "humanitarian"}, + "unhcr": {"abbrev": "UNHCR", "type": "humanitarian"}, + "unicef": {"abbrev": "UNICEF", "type": "humanitarian"}, + "asean": {"abbrev": "ASEAN", "type": "political_bloc"}, + "african union": {"abbrev": "AU", "type": "political_bloc"}, + "brics": {"abbrev": "BRICS", "type": "political_bloc"}, + "g7": {"abbrev": "G7", "type": "political_bloc"}, + "g20": {"abbrev": "G20", "type": "political_bloc"}, + "wto": {"abbrev": "WTO", "type": "trade"}, + "world trade organization": {"abbrev": "WTO", "type": "trade"}, + "hamas": {"abbrev": "Hamas", "type": "militant"}, + "hezbollah": {"abbrev": "Hezbollah", "type": "militant"}, + "houthis": {"abbrev": "Houthis", "type": "militant"}, + "isis": {"abbrev": "ISIS", "type": "militant"}, + "islamic state": {"abbrev": "ISIS", "type": "militant"}, + "al-qaeda": {"abbrev": "AQ", "type": "militant"}, + "al qaeda": {"abbrev": "AQ", "type": "militant"}, + "taliban": {"abbrev": "Taliban", "type": "militant"}, + "wagner": {"abbrev": "Wagner", "type": "pmc"}, + "wagner group": {"abbrev": "Wagner", "type": "pmc"}, + "cia": {"abbrev": "CIA", "type": "intelligence"}, + "fbi": {"abbrev": "FBI", "type": "intelligence"}, + "mossad": {"abbrev": "Mossad", "type": "intelligence"}, + "mi6": {"abbrev": "MI6", "type": "intelligence"}, + "fsb": {"abbrev": "FSB", "type": "intelligence"}, + "pentagon": {"abbrev": "DoD", "type": "military"}, + "department of defense": {"abbrev": "DoD", "type": "military"}, +} + +# --------------------------------------------------------------------------- +# Major companies (defense, tech, energy) +# --------------------------------------------------------------------------- +COMPANIES: dict[str, dict] = { + "lockheed martin": {"ticker": "LMT", "sector": "defense"}, + "raytheon": {"ticker": "RTX", "sector": "defense"}, + "northrop grumman": {"ticker": "NOC", "sector": "defense"}, + "boeing": {"ticker": "BA", "sector": "defense"}, + "general dynamics": {"ticker": "GD", "sector": "defense"}, + "bae systems": {"ticker": "BA.L", "sector": "defense"}, + "rheinmetall": {"ticker": "RHM.DE", "sector": "defense"}, + "apple": {"ticker": "AAPL", "sector": "tech"}, + "google": {"ticker": "GOOGL", "sector": "tech"}, + "alphabet": {"ticker": "GOOGL", "sector": "tech"}, + "microsoft": {"ticker": "MSFT", "sector": "tech"}, + "amazon": {"ticker": "AMZN", "sector": "tech"}, + "meta": {"ticker": "META", "sector": "tech"}, + "nvidia": {"ticker": "NVDA", "sector": "tech"}, + "openai": {"ticker": None, "sector": "ai"}, + "anthropic": {"ticker": None, "sector": "ai"}, + "deepmind": {"ticker": None, "sector": "ai"}, + "aramco": {"ticker": "2222.SR", "sector": "energy"}, + "exxonmobil": {"ticker": "XOM", "sector": "energy"}, + "chevron": {"ticker": "CVX", "sector": "energy"}, + "shell": {"ticker": "SHEL", "sector": "energy"}, + "bp": {"ticker": "BP", "sector": "energy"}, + "gazprom": {"ticker": None, "sector": "energy"}, + "tsmc": {"ticker": "TSM", "sector": "semiconductor"}, + "samsung": {"ticker": "005930.KS", "sector": "tech"}, +} + +# --------------------------------------------------------------------------- +# APT / threat actor names +# --------------------------------------------------------------------------- +APT_GROUPS: set[str] = { + "apt28", "apt29", "apt30", "apt31", "apt33", "apt34", "apt35", "apt38", + "apt40", "apt41", "lazarus", "lazarus group", "fancy bear", "cozy bear", + "sandworm", "turla", "equation group", "kimsuky", "charming kitten", + "double dragon", "stone panda", "volt typhoon", "salt typhoon", + "midnight blizzard", "star blizzard", "forest blizzard", + "scattered spider", "lapsus$", "lockbit", "blackcat", "alphv", + "cl0p", "conti", "revil", "darkside", "black basta", +} diff --git a/src/world_intel_mcp/server.py b/src/world_intel_mcp/server.py index 769db1f..7ec5a12 100644 --- a/src/world_intel_mcp/server.py +++ b/src/world_intel_mcp/server.py @@ -16,6 +16,7 @@ Phase 6: Military & infrastructure intelligence (+6 = 45 tools). Phase 7: Health, sanctions, elections, shipping, social, nuclear, alerts, trends (+10 = 55 tools). Phase 8: Service status monitoring, RSS expansion (80+ feeds, 14 categories) (+1 = 56 tools). Phase 9: Geospatial datasets — military bases, ports, pipelines, nuclear facilities (+4 = 60 tools). +Phase 10: NLP intelligence — entity extraction, event classification, news clustering, keyword spikes (+4 = 64 tools). """ import asyncio @@ -635,6 +636,51 @@ TOOLS: list[Tool] = [ }, }, ), + # --- NLP Intelligence (4 tools) --- + Tool( + name="intel_extract_entities", + description="Extract named entities (countries, leaders, organizations, companies, CVEs, APT groups) from text or recent news headlines.", + inputSchema={ + "type": "object", + "properties": { + "text": {"type": "string", "description": "Text to analyze. If omitted, analyzes recent news headlines."}, + }, + }, + ), + Tool( + name="intel_classify_event", + description="Classify text into threat categories (military, terrorism, cyber, political, economic, health, climate, nuclear, etc.) with severity scoring (1-10).", + inputSchema={ + "type": "object", + "properties": { + "text": {"type": "string", "description": "Event text or headline to classify."}, + }, + "required": ["text"], + }, + ), + Tool( + name="intel_news_clusters", + description="Cluster recent news articles by topic similarity using Jaccard coefficient. Groups related stories and extracts top keywords per cluster.", + inputSchema={ + "type": "object", + "properties": { + "category": {"type": "string", "description": "RSS feed category filter (geopolitics, security, military, etc.)"}, + "limit": {"type": "integer", "description": "Max news items to cluster (default: 100)"}, + "threshold": {"type": "number", "description": "Similarity threshold 0.0-1.0 (default: 0.25)"}, + }, + }, + ), + Tool( + name="intel_keyword_spikes", + description="Detect trending keyword spikes against historical baselines using Welford's algorithm. Extracts CVE identifiers and APT group mentions.", + inputSchema={ + "type": "object", + "properties": { + "min_count": {"type": "integer", "description": "Minimum keyword frequency to consider (default: 3)"}, + "z_threshold": {"type": "number", "description": "Z-score threshold for spike detection (default: 2.0)"}, + }, + }, + ), # --- System (1 tool) --- Tool( name="intel_status", @@ -902,6 +948,29 @@ async def _dispatch(name: str, arguments: dict[str, Any]) -> Any: status=arguments.get("status"), ) + # NLP Intelligence + case "intel_extract_entities": + from .analysis.entities import fetch_entity_extraction + return await fetch_entity_extraction(fetcher, text=arguments.get("text")) + case "intel_classify_event": + from .analysis.classifier import fetch_classify_event + return await fetch_classify_event(fetcher, text=arguments["text"]) + case "intel_news_clusters": + from .analysis.clustering import fetch_news_clusters + return await fetch_news_clusters( + fetcher, + category=arguments.get("category"), + limit=arguments.get("limit", 100), + threshold=arguments.get("threshold", 0.25), + ) + case "intel_keyword_spikes": + from .analysis.spikes import fetch_keyword_spikes + return await fetch_keyword_spikes( + fetcher, + min_count=arguments.get("min_count", 3), + z_threshold=arguments.get("z_threshold", 2.0), + ) + # System case "intel_status": return { @@ -932,6 +1001,7 @@ async def _dispatch(name: str, arguments: dict[str, Any]) -> Any: "nuclear": ["usgs-nuclear-monitor"], "service_status": ["aws", "azure", "gcp", "cloudflare", "github"], "geospatial": ["static-datasets (bases, ports, pipelines, nuclear)"], + "nlp": ["regex-ner", "keyword-classifier", "jaccard-clustering", "keyword-spike-detector"], }, }