"""Wikipedia integration: location summaries, free-text search (for the "search via Wikipedia" text-selection feature), and best-effort national parliament composition diagrams. Everything here is a thin, cached proxy around Wikipedia's public REST/action APIs so the frontend never talks to wikipedia.org directly (avoids CORS and lets us cache/rate-limit centrally). """ import time from functools import lru_cache import httpx import yaml from .config import APP_DIR REST_SUMMARY = "https://en.wikipedia.org/api/rest_v1/page/summary/{title}" OPENSEARCH = "https://en.wikipedia.org/w/api.php" HEADERS = {"User-Agent": "NewsAtlas/1.0 (self-hosted news globe; contact: admin@localhost)"} _PARLIAMENTS_FILE = APP_DIR / "data" / "parliaments.yaml" _CACHE_TTL = 6 * 3600 # seconds _summary_cache: dict[str, tuple[float, dict | None]] = {} _parliament_resolve_cache: dict[str, tuple[float, str | None]] = {} @lru_cache(maxsize=1) def _parliament_map() -> dict[str, str | None]: with open(_PARLIAMENTS_FILE, encoding="utf-8") as f: return yaml.safe_load(f) or {} def _cache_get(cache: dict, key: str): entry = cache.get(key) if entry and time.time() - entry[0] < _CACHE_TTL: return entry[1] return "MISS" async def get_summary(title: str) -> dict | None: cached = _cache_get(_summary_cache, title) if cached != "MISS": return cached result = None async with httpx.AsyncClient(headers=HEADERS, timeout=10) as client: resp = await client.get(REST_SUMMARY.format(title=title.replace(" ", "_"))) if resp.status_code == 200: data = resp.json() if data.get("type") != "disambiguation": result = { "title": data.get("title"), "extract": data.get("extract"), "thumbnail": (data.get("thumbnail") or {}).get("source"), "original_image": (data.get("originalimage") or {}).get("source"), "page_url": (data.get("content_urls", {}).get("desktop") or {}).get("page"), } _summary_cache[title] = (time.time(), result) return result async def search(query: str, limit: int = 5) -> list[dict]: if not query or not query.strip(): return [] params = { "action": "opensearch", "search": query.strip(), "limit": str(limit), "namespace": "0", "format": "json", } async with httpx.AsyncClient(headers=HEADERS, timeout=10) as client: resp = await client.get(OPENSEARCH, params=params) resp.raise_for_status() _, titles, snippets, urls = resp.json() return [ {"title": t, "snippet": s, "page_url": u} for t, s, u in zip(titles, snippets, urls) ] async def _resolve_parliament_title(country: str) -> str | None: mapping = _parliament_map() if country in mapping: return mapping[country] # may be explicit None -> "no legislature" cached = _cache_get(_parliament_resolve_cache, country) if cached != "MISS": return cached results = await search(f"{country} parliament", limit=1) resolved = results[0]["title"] if results else None _parliament_resolve_cache[country] = (time.time(), resolved) return resolved async def get_parliament(country: str) -> dict | None: title = await _resolve_parliament_title(country) if not title: return None summary = await get_summary(title) if not summary: return None diagram = summary.get("original_image") or summary.get("thumbnail") if not diagram: return None return { "country": country, "legislature": summary["title"], "diagram_url": diagram, "extract": summary.get("extract"), "page_url": summary.get("page_url"), }