news digest (spec-013): rss/atom -> {news} file via cron
Replaces the broken pre-1.0-openai news_feed.py. Stdlib parsing with defusedxml (feeds are untrusted XML), titles sanitized (SAF-03), feed URLs SSRF-guarded. CLI: python -m fjerkroa_bot.news --config <cfg>.
This commit is contained in:
+12
@@ -80,3 +80,15 @@ enable-game-info = true
|
|||||||
# Ops (SPEC-012): consecutive OpenAI failures before a staff alert
|
# Ops (SPEC-012): consecutive OpenAI failures before a staff alert
|
||||||
# api-error-alert-threshold = 5
|
# api-error-alert-threshold = 5
|
||||||
# Backups: cron runs deploy/backup_db.py daily -> ~/backups/<bot>/ (keep 14)
|
# Backups: cron runs deploy/backup_db.py daily -> ~/backups/<bot>/ (keep 14)
|
||||||
|
|
||||||
|
# News digest (SPEC-013) — `python -m fjerkroa_bot.news --config X.toml` via cron;
|
||||||
|
# writes the {news} file. Feeds are [url, label] pairs (RSS or Atom):
|
||||||
|
# news = "news_feed.txt"
|
||||||
|
# news-per-feed = 3
|
||||||
|
# news-max-items = 15
|
||||||
|
# news-feeds = [
|
||||||
|
# ["https://blog.playstation.com/feed/", "PS"],
|
||||||
|
# ["https://kotaku.com/rss", "Kotaku"],
|
||||||
|
# ["https://www.pushsquare.com/feeds/latest", "Push"],
|
||||||
|
# ["https://mein-mmo.de/feed/", "MeinMMO"],
|
||||||
|
# ]
|
||||||
|
|||||||
@@ -0,0 +1,153 @@
|
|||||||
|
"""News digest fetcher (SPEC-013, FDB-012 news rewrite).
|
||||||
|
|
||||||
|
Replaces the broken pre-1.0-openai `news_feed.py`. Fetches configured
|
||||||
|
RSS/Atom feeds (stdlib, no feedparser dep), builds a compact sanitized
|
||||||
|
headline digest, and writes it to the `{news}` file the responder
|
||||||
|
injects (AIResponder.message). Feeds are external input: titles are
|
||||||
|
sanitized (SAF-03) and each feed URL is SSRF-guarded before fetching.
|
||||||
|
|
||||||
|
CLI: python -m fjerkroa_bot.news --config kroa.toml
|
||||||
|
"""
|
||||||
|
|
||||||
|
import argparse
|
||||||
|
import logging
|
||||||
|
import sys
|
||||||
|
import time
|
||||||
|
from typing import Any, Dict, List, Optional, Tuple
|
||||||
|
|
||||||
|
import defusedxml.ElementTree as ElementTree # hardened XML: feeds are untrusted (XXE/billion-laughs)
|
||||||
|
|
||||||
|
from .ai_responder import sanitize_external_text
|
||||||
|
|
||||||
|
DEFAULT_PER_FEED = 3
|
||||||
|
DEFAULT_MAX_ITEMS = 15
|
||||||
|
FETCH_TIMEOUT_S = 15
|
||||||
|
_ATOM = "{http://www.w3.org/2005/Atom}"
|
||||||
|
|
||||||
|
|
||||||
|
def parse_feed(data: bytes, source: str = "") -> List[Dict[str, str]]:
|
||||||
|
"""Parse RSS or Atom bytes into [{title, link, source}] (tolerant)."""
|
||||||
|
try:
|
||||||
|
root = ElementTree.fromstring(data)
|
||||||
|
except Exception as err:
|
||||||
|
# malformed XML or a blocked entity/DTD attack — tolerate, never raise (NEWS-01)
|
||||||
|
logging.warning(f"news: unparseable/unsafe feed {source!r}: {err!r}")
|
||||||
|
return []
|
||||||
|
items: List[Dict[str, str]] = []
|
||||||
|
# RSS: <rss><channel><item><title/><link/>
|
||||||
|
for item in root.iter("item"):
|
||||||
|
title = (item.findtext("title") or "").strip()
|
||||||
|
link = (item.findtext("link") or "").strip()
|
||||||
|
if title:
|
||||||
|
items.append({"title": title, "link": link, "source": source})
|
||||||
|
# Atom: <feed><entry><title/><link href=/>
|
||||||
|
for entry in root.iter(f"{_ATOM}entry"):
|
||||||
|
title = (entry.findtext(f"{_ATOM}title") or "").strip()
|
||||||
|
link_el = entry.find(f"{_ATOM}link")
|
||||||
|
link = link_el.get("href", "") if link_el is not None else ""
|
||||||
|
if title:
|
||||||
|
items.append({"title": title, "link": link, "source": source})
|
||||||
|
return items
|
||||||
|
|
||||||
|
|
||||||
|
def render_digest(items: List[Dict[str, str]], max_items: int = DEFAULT_MAX_ITEMS) -> str:
|
||||||
|
"""Compact sanitized digest for the {news} prompt slot."""
|
||||||
|
lines = []
|
||||||
|
for item in items[:max_items]:
|
||||||
|
title = sanitize_external_text(item["title"], 200)
|
||||||
|
source = item.get("source", "")
|
||||||
|
link = item.get("link", "")
|
||||||
|
prefix = f"[{source}] " if source else ""
|
||||||
|
lines.append(f"- {prefix}{title}" + (f" ({link})" if link else ""))
|
||||||
|
return "\n".join(lines)
|
||||||
|
|
||||||
|
|
||||||
|
class NewsFetcher:
|
||||||
|
def __init__(self, guard, fetch_bytes) -> None:
|
||||||
|
# injected so tests need no network; production wires aiohttp + guard_url
|
||||||
|
self._guard = guard
|
||||||
|
self._fetch_bytes = fetch_bytes
|
||||||
|
|
||||||
|
async def collect(self, feeds: List[Tuple[str, str]], per_feed: int) -> List[Dict[str, str]]:
|
||||||
|
"""feeds = [(url, label)]; returns deduped items, order preserved."""
|
||||||
|
seen = set()
|
||||||
|
out: List[Dict[str, str]] = []
|
||||||
|
for url, label in feeds:
|
||||||
|
reason = self._guard(url)
|
||||||
|
if reason:
|
||||||
|
logging.warning(f"news: skipping feed {label} — {reason}")
|
||||||
|
continue
|
||||||
|
try:
|
||||||
|
data = await self._fetch_bytes(url)
|
||||||
|
except Exception as err:
|
||||||
|
logging.warning(f"news: fetch failed for {label}: {repr(err)}")
|
||||||
|
continue
|
||||||
|
for item in parse_feed(data, label)[:per_feed]:
|
||||||
|
key = item["title"]
|
||||||
|
if key not in seen:
|
||||||
|
seen.add(key)
|
||||||
|
out.append(item)
|
||||||
|
return out
|
||||||
|
|
||||||
|
|
||||||
|
def _feeds_from_config(config: Dict[str, Any]) -> List[Tuple[str, str]]:
|
||||||
|
"""news-feeds = [["url", "label"], ...] or ["url", ...]."""
|
||||||
|
feeds = []
|
||||||
|
for entry in config.get("news-feeds", []):
|
||||||
|
if isinstance(entry, (list, tuple)):
|
||||||
|
feeds.append((str(entry[0]), str(entry[1]) if len(entry) > 1 else ""))
|
||||||
|
else:
|
||||||
|
feeds.append((str(entry), ""))
|
||||||
|
return feeds
|
||||||
|
|
||||||
|
|
||||||
|
async def _aiohttp_fetch(url: str) -> bytes:
|
||||||
|
import aiohttp
|
||||||
|
|
||||||
|
from .httpread import read_capped
|
||||||
|
|
||||||
|
timeout = aiohttp.ClientTimeout(total=FETCH_TIMEOUT_S)
|
||||||
|
async with aiohttp.ClientSession(timeout=timeout, headers={"User-Agent": "Mozilla/5.0 (compatible; FjerkroaBot-news/1.0)"}) as session:
|
||||||
|
async with session.get(url) as response:
|
||||||
|
response.raise_for_status()
|
||||||
|
return await read_capped(response, 4 * 1024 * 1024)
|
||||||
|
|
||||||
|
|
||||||
|
async def run(config: Dict[str, Any]) -> Optional[str]:
|
||||||
|
from .url_reader import guard_url
|
||||||
|
|
||||||
|
out_path = config.get("news")
|
||||||
|
if not out_path:
|
||||||
|
logging.error("news: no `news` output path in config")
|
||||||
|
return None
|
||||||
|
feeds = _feeds_from_config(config)
|
||||||
|
if not feeds:
|
||||||
|
logging.error("news: no `news-feeds` configured")
|
||||||
|
return None
|
||||||
|
fetcher = NewsFetcher(guard_url, _aiohttp_fetch)
|
||||||
|
items = await fetcher.collect(feeds, int(config.get("news-per-feed", DEFAULT_PER_FEED)))
|
||||||
|
digest = render_digest(items, int(config.get("news-max-items", DEFAULT_MAX_ITEMS)))
|
||||||
|
header = f"News as of {time.strftime('%Y-%m-%d %H:%M UTC', time.gmtime())}:\n"
|
||||||
|
with open(out_path, "w", encoding="utf-8") as fd:
|
||||||
|
fd.write(header + digest + "\n")
|
||||||
|
logging.info(f"news: wrote {len(items)} items to {out_path}")
|
||||||
|
return out_path
|
||||||
|
|
||||||
|
|
||||||
|
def main() -> int:
|
||||||
|
import asyncio
|
||||||
|
|
||||||
|
import tomlkit
|
||||||
|
|
||||||
|
logging.basicConfig(level=logging.INFO, format="%(asctime)s [%(levelname)s] %(message)s")
|
||||||
|
parser = argparse.ArgumentParser(description="Fetch RSS/Atom feeds into the {news} digest file")
|
||||||
|
parser.add_argument("--config", required=True)
|
||||||
|
args = parser.parse_args()
|
||||||
|
with open(args.config, encoding="utf-8") as fd:
|
||||||
|
config = tomlkit.load(fd)
|
||||||
|
result = asyncio.run(run(config))
|
||||||
|
return 0 if result else 1
|
||||||
|
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
sys.exit(main())
|
||||||
@@ -15,6 +15,7 @@ dependencies = [
|
|||||||
"tomlkit>=0.13",
|
"tomlkit>=0.13",
|
||||||
"watchdog>=6",
|
"watchdog>=6",
|
||||||
"requests>=2.32",
|
"requests>=2.32",
|
||||||
|
"defusedxml>=0.7",
|
||||||
]
|
]
|
||||||
|
|
||||||
[project.scripts]
|
[project.scripts]
|
||||||
|
|||||||
@@ -0,0 +1,25 @@
|
|||||||
|
# SPEC-013 — News digest
|
||||||
|
|
||||||
|
Replaces the broken pre-1.0-openai `news_feed.py`. A CLI
|
||||||
|
(`python -m fjerkroa_bot.news --config <cfg>`) fetches the
|
||||||
|
`news-feeds` and writes a compact digest to the `news` file that
|
||||||
|
`AIResponder.message` injects into the `{news}` slot. Feeds are
|
||||||
|
external input and operator-configured.
|
||||||
|
|
||||||
|
### NEWS-01 — RSS and Atom parse to items (coverage: test)
|
||||||
|
|
||||||
|
`parse_feed(bytes, label)` extracts `{title, link, source}` from both
|
||||||
|
RSS (`<item>`) and Atom (`<entry>`) documents, tolerates malformed
|
||||||
|
XML (returns an empty list, logs), and never raises.
|
||||||
|
|
||||||
|
### NEWS-02 — Digest is sanitized and bounded (coverage: test)
|
||||||
|
|
||||||
|
`render_digest` caps at `news-max-items`, and every headline passes
|
||||||
|
`sanitize_external_text` (SAF-03) — a feed cannot inject `@everyone`
|
||||||
|
or control characters into the prompt via a headline.
|
||||||
|
|
||||||
|
### NEWS-03 — Feeds are SSRF-guarded and deduped (coverage: test)
|
||||||
|
|
||||||
|
`NewsFetcher.collect` skips any feed URL the SSRF guard rejects,
|
||||||
|
skips feeds that fail to fetch (one bad feed never sinks the run),
|
||||||
|
and drops duplicate headlines across feeds.
|
||||||
@@ -0,0 +1,75 @@
|
|||||||
|
"""Unit coverage for SPEC-013 news digest (NEWS-01..03)."""
|
||||||
|
|
||||||
|
import unittest
|
||||||
|
from unittest.mock import AsyncMock
|
||||||
|
|
||||||
|
from fjerkroa_bot.news import NewsFetcher, parse_feed, render_digest
|
||||||
|
|
||||||
|
RSS = b"""<?xml version="1.0"?><rss><channel>
|
||||||
|
<item><title>Game X released</title><link>https://ex.com/x</link></item>
|
||||||
|
<item><title>Patch Y notes</title><link>https://ex.com/y</link></item>
|
||||||
|
</channel></rss>"""
|
||||||
|
|
||||||
|
ATOM = b"""<?xml version="1.0"?><feed xmlns="http://www.w3.org/2005/Atom">
|
||||||
|
<entry><title>Atom headline</title><link href="https://ex.com/a"/></entry>
|
||||||
|
</feed>"""
|
||||||
|
|
||||||
|
|
||||||
|
class TestParse(unittest.TestCase):
|
||||||
|
def test_rss(self):
|
||||||
|
"""NEWS-01: RSS items parsed with title + link."""
|
||||||
|
items = parse_feed(RSS, "Src")
|
||||||
|
self.assertEqual([i["title"] for i in items], ["Game X released", "Patch Y notes"])
|
||||||
|
self.assertEqual(items[0]["link"], "https://ex.com/x")
|
||||||
|
self.assertEqual(items[0]["source"], "Src")
|
||||||
|
|
||||||
|
def test_atom(self):
|
||||||
|
"""NEWS-01: Atom entries parsed with href link."""
|
||||||
|
items = parse_feed(ATOM, "A")
|
||||||
|
self.assertEqual(items[0]["title"], "Atom headline")
|
||||||
|
self.assertEqual(items[0]["link"], "https://ex.com/a")
|
||||||
|
|
||||||
|
def test_malformed_never_raises(self):
|
||||||
|
"""NEWS-01: garbage XML returns [] without raising."""
|
||||||
|
self.assertEqual(parse_feed(b"<not xml", "bad"), [])
|
||||||
|
self.assertEqual(parse_feed(b"", "empty"), [])
|
||||||
|
|
||||||
|
|
||||||
|
class TestDigest(unittest.TestCase):
|
||||||
|
def test_sanitized_and_capped(self):
|
||||||
|
"""NEWS-02: headlines sanitized, item count capped."""
|
||||||
|
items = [{"title": "@everyone big news \x00", "link": "", "source": "S"} for _ in range(20)]
|
||||||
|
digest = render_digest(items, max_items=5)
|
||||||
|
self.assertEqual(digest.count("\n"), 4) # 5 lines
|
||||||
|
self.assertNotIn("@everyone", digest)
|
||||||
|
self.assertNotIn("\x00", digest)
|
||||||
|
|
||||||
|
|
||||||
|
class TestCollect(unittest.IsolatedAsyncioTestCase):
|
||||||
|
async def test_ssrf_skip_and_dedup(self):
|
||||||
|
"""NEWS-03: guarded feed skipped, dup titles dropped, bad fetch survived."""
|
||||||
|
|
||||||
|
def guard(url):
|
||||||
|
return "refused" if "internal" in url else None
|
||||||
|
|
||||||
|
async def fetch(url):
|
||||||
|
if "boom" in url:
|
||||||
|
raise ValueError("boom")
|
||||||
|
return RSS # same content from two feeds -> dedup
|
||||||
|
|
||||||
|
fetcher = NewsFetcher(guard, fetch)
|
||||||
|
feeds = [
|
||||||
|
("https://a.com/feed", "A"),
|
||||||
|
("https://internal/feed", "Internal"), # SSRF-skipped
|
||||||
|
("https://boom.com/feed", "Boom"), # fetch fails
|
||||||
|
("https://b.com/feed", "B"), # same RSS -> dup titles dropped
|
||||||
|
]
|
||||||
|
items = await fetcher.collect(feeds, per_feed=5)
|
||||||
|
titles = [i["title"] for i in items]
|
||||||
|
self.assertEqual(titles, ["Game X released", "Patch Y notes"]) # deduped, internal+boom skipped
|
||||||
|
|
||||||
|
async def test_per_feed_limit(self):
|
||||||
|
"""NEWS-03: per-feed cap honored."""
|
||||||
|
fetcher = NewsFetcher(lambda u: None, AsyncMock(return_value=RSS))
|
||||||
|
items = await fetcher.collect([("https://a.com", "A")], per_feed=1)
|
||||||
|
self.assertEqual(len(items), 1)
|
||||||
@@ -627,6 +627,7 @@ version = "3.0.0"
|
|||||||
source = { editable = "." }
|
source = { editable = "." }
|
||||||
dependencies = [
|
dependencies = [
|
||||||
{ name = "aiohttp" },
|
{ name = "aiohttp" },
|
||||||
|
{ name = "defusedxml" },
|
||||||
{ name = "discord-py" },
|
{ name = "discord-py" },
|
||||||
{ name = "openai" },
|
{ name = "openai" },
|
||||||
{ name = "requests" },
|
{ name = "requests" },
|
||||||
@@ -656,6 +657,7 @@ dev = [
|
|||||||
[package.metadata]
|
[package.metadata]
|
||||||
requires-dist = [
|
requires-dist = [
|
||||||
{ name = "aiohttp", specifier = ">=3.12" },
|
{ name = "aiohttp", specifier = ">=3.12" },
|
||||||
|
{ name = "defusedxml", specifier = ">=0.7" },
|
||||||
{ name = "discord-py", specifier = ">=2.5,<3" },
|
{ name = "discord-py", specifier = ">=2.5,<3" },
|
||||||
{ name = "openai", specifier = ">=2.45" },
|
{ name = "openai", specifier = ">=2.45" },
|
||||||
{ name = "requests", specifier = ">=2.32" },
|
{ name = "requests", specifier = ">=2.32" },
|
||||||
|
|||||||
Reference in New Issue
Block a user