diff --git a/config.toml b/config.toml index f8a9e97..b4aa363 100644 --- a/config.toml +++ b/config.toml @@ -80,3 +80,15 @@ enable-game-info = true # Ops (SPEC-012): consecutive OpenAI failures before a staff alert # api-error-alert-threshold = 5 # Backups: cron runs deploy/backup_db.py daily -> ~/backups// (keep 14) + +# News digest (SPEC-013) — `python -m fjerkroa_bot.news --config X.toml` via cron; +# writes the {news} file. Feeds are [url, label] pairs (RSS or Atom): +# news = "news_feed.txt" +# news-per-feed = 3 +# news-max-items = 15 +# news-feeds = [ +# ["https://blog.playstation.com/feed/", "PS"], +# ["https://kotaku.com/rss", "Kotaku"], +# ["https://www.pushsquare.com/feeds/latest", "Push"], +# ["https://mein-mmo.de/feed/", "MeinMMO"], +# ] diff --git a/fjerkroa_bot/news.py b/fjerkroa_bot/news.py new file mode 100644 index 0000000..7706cdf --- /dev/null +++ b/fjerkroa_bot/news.py @@ -0,0 +1,153 @@ +"""News digest fetcher (SPEC-013, FDB-012 news rewrite). + +Replaces the broken pre-1.0-openai `news_feed.py`. Fetches configured +RSS/Atom feeds (stdlib, no feedparser dep), builds a compact sanitized +headline digest, and writes it to the `{news}` file the responder +injects (AIResponder.message). Feeds are external input: titles are +sanitized (SAF-03) and each feed URL is SSRF-guarded before fetching. + +CLI: python -m fjerkroa_bot.news --config kroa.toml +""" + +import argparse +import logging +import sys +import time +from typing import Any, Dict, List, Optional, Tuple + +import defusedxml.ElementTree as ElementTree # hardened XML: feeds are untrusted (XXE/billion-laughs) + +from .ai_responder import sanitize_external_text + +DEFAULT_PER_FEED = 3 +DEFAULT_MAX_ITEMS = 15 +FETCH_TIMEOUT_S = 15 +_ATOM = "{http://www.w3.org/2005/Atom}" + + +def parse_feed(data: bytes, source: str = "") -> List[Dict[str, str]]: + """Parse RSS or Atom bytes into [{title, link, source}] (tolerant).""" + try: + root = ElementTree.fromstring(data) + except Exception as err: + # malformed XML or a blocked entity/DTD attack — tolerate, never raise (NEWS-01) + logging.warning(f"news: unparseable/unsafe feed {source!r}: {err!r}") + return [] + items: List[Dict[str, str]] = [] + # RSS: <link/> + for item in root.iter("item"): + title = (item.findtext("title") or "").strip() + link = (item.findtext("link") or "").strip() + if title: + items.append({"title": title, "link": link, "source": source}) + # Atom: <feed><entry><title/><link href=/> + for entry in root.iter(f"{_ATOM}entry"): + title = (entry.findtext(f"{_ATOM}title") or "").strip() + link_el = entry.find(f"{_ATOM}link") + link = link_el.get("href", "") if link_el is not None else "" + if title: + items.append({"title": title, "link": link, "source": source}) + return items + + +def render_digest(items: List[Dict[str, str]], max_items: int = DEFAULT_MAX_ITEMS) -> str: + """Compact sanitized digest for the {news} prompt slot.""" + lines = [] + for item in items[:max_items]: + title = sanitize_external_text(item["title"], 200) + source = item.get("source", "") + link = item.get("link", "") + prefix = f"[{source}] " if source else "" + lines.append(f"- {prefix}{title}" + (f" ({link})" if link else "")) + return "\n".join(lines) + + +class NewsFetcher: + def __init__(self, guard, fetch_bytes) -> None: + # injected so tests need no network; production wires aiohttp + guard_url + self._guard = guard + self._fetch_bytes = fetch_bytes + + async def collect(self, feeds: List[Tuple[str, str]], per_feed: int) -> List[Dict[str, str]]: + """feeds = [(url, label)]; returns deduped items, order preserved.""" + seen = set() + out: List[Dict[str, str]] = [] + for url, label in feeds: + reason = self._guard(url) + if reason: + logging.warning(f"news: skipping feed {label} — {reason}") + continue + try: + data = await self._fetch_bytes(url) + except Exception as err: + logging.warning(f"news: fetch failed for {label}: {repr(err)}") + continue + for item in parse_feed(data, label)[:per_feed]: + key = item["title"] + if key not in seen: + seen.add(key) + out.append(item) + return out + + +def _feeds_from_config(config: Dict[str, Any]) -> List[Tuple[str, str]]: + """news-feeds = [["url", "label"], ...] or ["url", ...].""" + feeds = [] + for entry in config.get("news-feeds", []): + if isinstance(entry, (list, tuple)): + feeds.append((str(entry[0]), str(entry[1]) if len(entry) > 1 else "")) + else: + feeds.append((str(entry), "")) + return feeds + + +async def _aiohttp_fetch(url: str) -> bytes: + import aiohttp + + from .httpread import read_capped + + timeout = aiohttp.ClientTimeout(total=FETCH_TIMEOUT_S) + async with aiohttp.ClientSession(timeout=timeout, headers={"User-Agent": "Mozilla/5.0 (compatible; FjerkroaBot-news/1.0)"}) as session: + async with session.get(url) as response: + response.raise_for_status() + return await read_capped(response, 4 * 1024 * 1024) + + +async def run(config: Dict[str, Any]) -> Optional[str]: + from .url_reader import guard_url + + out_path = config.get("news") + if not out_path: + logging.error("news: no `news` output path in config") + return None + feeds = _feeds_from_config(config) + if not feeds: + logging.error("news: no `news-feeds` configured") + return None + fetcher = NewsFetcher(guard_url, _aiohttp_fetch) + items = await fetcher.collect(feeds, int(config.get("news-per-feed", DEFAULT_PER_FEED))) + digest = render_digest(items, int(config.get("news-max-items", DEFAULT_MAX_ITEMS))) + header = f"News as of {time.strftime('%Y-%m-%d %H:%M UTC', time.gmtime())}:\n" + with open(out_path, "w", encoding="utf-8") as fd: + fd.write(header + digest + "\n") + logging.info(f"news: wrote {len(items)} items to {out_path}") + return out_path + + +def main() -> int: + import asyncio + + import tomlkit + + logging.basicConfig(level=logging.INFO, format="%(asctime)s [%(levelname)s] %(message)s") + parser = argparse.ArgumentParser(description="Fetch RSS/Atom feeds into the {news} digest file") + parser.add_argument("--config", required=True) + args = parser.parse_args() + with open(args.config, encoding="utf-8") as fd: + config = tomlkit.load(fd) + result = asyncio.run(run(config)) + return 0 if result else 1 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/pyproject.toml b/pyproject.toml index c6ed356..fb98cb4 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -15,6 +15,7 @@ dependencies = [ "tomlkit>=0.13", "watchdog>=6", "requests>=2.32", + "defusedxml>=0.7", ] [project.scripts] diff --git a/specs/SPEC-013-news.md b/specs/SPEC-013-news.md new file mode 100644 index 0000000..5c10473 --- /dev/null +++ b/specs/SPEC-013-news.md @@ -0,0 +1,25 @@ +# SPEC-013 — News digest + +Replaces the broken pre-1.0-openai `news_feed.py`. A CLI +(`python -m fjerkroa_bot.news --config <cfg>`) fetches the +`news-feeds` and writes a compact digest to the `news` file that +`AIResponder.message` injects into the `{news}` slot. Feeds are +external input and operator-configured. + +### NEWS-01 — RSS and Atom parse to items (coverage: test) + +`parse_feed(bytes, label)` extracts `{title, link, source}` from both +RSS (`<item>`) and Atom (`<entry>`) documents, tolerates malformed +XML (returns an empty list, logs), and never raises. + +### NEWS-02 — Digest is sanitized and bounded (coverage: test) + +`render_digest` caps at `news-max-items`, and every headline passes +`sanitize_external_text` (SAF-03) — a feed cannot inject `@everyone` +or control characters into the prompt via a headline. + +### NEWS-03 — Feeds are SSRF-guarded and deduped (coverage: test) + +`NewsFetcher.collect` skips any feed URL the SSRF guard rejects, +skips feeds that fail to fetch (one bad feed never sinks the run), +and drops duplicate headlines across feeds. diff --git a/tests/test_spec_news.py b/tests/test_spec_news.py new file mode 100644 index 0000000..057304b --- /dev/null +++ b/tests/test_spec_news.py @@ -0,0 +1,75 @@ +"""Unit coverage for SPEC-013 news digest (NEWS-01..03).""" + +import unittest +from unittest.mock import AsyncMock + +from fjerkroa_bot.news import NewsFetcher, parse_feed, render_digest + +RSS = b"""<?xml version="1.0"?><rss><channel> +<item><title>Game X releasedhttps://ex.com/x +Patch Y noteshttps://ex.com/y +""" + +ATOM = b""" +Atom headline +""" + + +class TestParse(unittest.TestCase): + def test_rss(self): + """NEWS-01: RSS items parsed with title + link.""" + items = parse_feed(RSS, "Src") + self.assertEqual([i["title"] for i in items], ["Game X released", "Patch Y notes"]) + self.assertEqual(items[0]["link"], "https://ex.com/x") + self.assertEqual(items[0]["source"], "Src") + + def test_atom(self): + """NEWS-01: Atom entries parsed with href link.""" + items = parse_feed(ATOM, "A") + self.assertEqual(items[0]["title"], "Atom headline") + self.assertEqual(items[0]["link"], "https://ex.com/a") + + def test_malformed_never_raises(self): + """NEWS-01: garbage XML returns [] without raising.""" + self.assertEqual(parse_feed(b" dedup + + fetcher = NewsFetcher(guard, fetch) + feeds = [ + ("https://a.com/feed", "A"), + ("https://internal/feed", "Internal"), # SSRF-skipped + ("https://boom.com/feed", "Boom"), # fetch fails + ("https://b.com/feed", "B"), # same RSS -> dup titles dropped + ] + items = await fetcher.collect(feeds, per_feed=5) + titles = [i["title"] for i in items] + self.assertEqual(titles, ["Game X released", "Patch Y notes"]) # deduped, internal+boom skipped + + async def test_per_feed_limit(self): + """NEWS-03: per-feed cap honored.""" + fetcher = NewsFetcher(lambda u: None, AsyncMock(return_value=RSS)) + items = await fetcher.collect([("https://a.com", "A")], per_feed=1) + self.assertEqual(len(items), 1) diff --git a/uv.lock b/uv.lock index 919872e..1ba0364 100644 --- a/uv.lock +++ b/uv.lock @@ -627,6 +627,7 @@ version = "3.0.0" source = { editable = "." } dependencies = [ { name = "aiohttp" }, + { name = "defusedxml" }, { name = "discord-py" }, { name = "openai" }, { name = "requests" }, @@ -656,6 +657,7 @@ dev = [ [package.metadata] requires-dist = [ { name = "aiohttp", specifier = ">=3.12" }, + { name = "defusedxml", specifier = ">=0.7" }, { name = "discord-py", specifier = ">=2.5,<3" }, { name = "openai", specifier = ">=2.45" }, { name = "requests", specifier = ">=2.32" },