Compare commits
3 Commits
| Author | SHA1 | Date | |
|---|---|---|---|
| 7628faf551 | |||
| 2caa18a17f | |||
| d4eec4088d |
+12
@@ -80,3 +80,15 @@ enable-game-info = true
|
||||
# Ops (SPEC-012): consecutive OpenAI failures before a staff alert
|
||||
# api-error-alert-threshold = 5
|
||||
# Backups: cron runs deploy/backup_db.py daily -> ~/backups/<bot>/ (keep 14)
|
||||
|
||||
# News digest (SPEC-013) — `python -m fjerkroa_bot.news --config X.toml` via cron;
|
||||
# writes the {news} file. Feeds are [url, label] pairs (RSS or Atom):
|
||||
# news = "news_feed.txt"
|
||||
# news-per-feed = 3
|
||||
# news-max-items = 15
|
||||
# news-feeds = [
|
||||
# ["https://blog.playstation.com/feed/", "PS"],
|
||||
# ["https://kotaku.com/rss", "Kotaku"],
|
||||
# ["https://www.pushsquare.com/feeds/latest", "Push"],
|
||||
# ["https://mein-mmo.de/feed/", "MeinMMO"],
|
||||
# ]
|
||||
|
||||
@@ -0,0 +1,18 @@
|
||||
"""Bounded HTTP body read (leaf module, no intra-package imports).
|
||||
|
||||
`response.content.read(n)` returns whatever is buffered, not n bytes,
|
||||
so it silently truncates large or chunked bodies (and web feeds/pages
|
||||
parse to garbage). This accumulates decompressed chunks up to a hard
|
||||
cap instead.
|
||||
"""
|
||||
|
||||
CHUNK = 65536
|
||||
|
||||
|
||||
async def read_capped(response, max_bytes: int) -> bytes:
|
||||
buf = bytearray()
|
||||
async for chunk in response.content.iter_chunked(CHUNK):
|
||||
buf.extend(chunk)
|
||||
if len(buf) > max_bytes:
|
||||
break
|
||||
return bytes(buf[:max_bytes])
|
||||
+12
-6
@@ -56,8 +56,12 @@ class IGDBQuery(object):
|
||||
return requests.post(igdb_url, headers=headers, data=query_body)
|
||||
|
||||
@staticmethod
|
||||
def build_query(fields, filters=None, limit=10, offset=None):
|
||||
query = f"fields {','.join(fields) if fields is not None and len(fields) > 0 else '*'}; limit {limit};"
|
||||
def build_query(fields, filters=None, limit=10, offset=None, search_term=None):
|
||||
query = ""
|
||||
if search_term:
|
||||
escaped = search_term.replace("\\", "\\\\").replace('"', '\\"')
|
||||
query += f'search "{escaped}"; '
|
||||
query += f"fields {','.join(fields) if fields is not None and len(fields) > 0 else '*'}; limit {limit};"
|
||||
if offset is not None:
|
||||
query += f" offset {offset};"
|
||||
if filters:
|
||||
@@ -65,12 +69,12 @@ class IGDBQuery(object):
|
||||
query += " where " + " & ".join(filter_statements) + ";"
|
||||
return query
|
||||
|
||||
def generalized_igdb_query(self, params, endpoint, fields, additional_filters=None, limit=10, offset=None):
|
||||
def generalized_igdb_query(self, params, endpoint, fields, additional_filters=None, limit=10, offset=None, search_term=None):
|
||||
all_filters = {key: f'~ "{value}"*' for key, value in params.items() if value}
|
||||
if additional_filters:
|
||||
all_filters.update(additional_filters)
|
||||
|
||||
query = self.build_query(fields, all_filters, limit, offset)
|
||||
query = self.build_query(fields, all_filters, limit, offset, search_term)
|
||||
data = self.send_igdb_request(endpoint, query)
|
||||
print(f"{endpoint}: {query} -> {data}")
|
||||
return data
|
||||
@@ -134,9 +138,10 @@ class IGDBQuery(object):
|
||||
return None
|
||||
|
||||
try:
|
||||
# Search for games with fuzzy matching
|
||||
# IGDB native full-text search: diacritic- and word-order-insensitive,
|
||||
# unlike a `name ~ "..."*` prefix filter
|
||||
games = self.generalized_igdb_query(
|
||||
{"name": query.strip()},
|
||||
{},
|
||||
"games",
|
||||
[
|
||||
"id",
|
||||
@@ -155,6 +160,7 @@ class IGDBQuery(object):
|
||||
],
|
||||
additional_filters={"game_type": "= 0"}, # Main games only (IGDB renamed category -> game_type)
|
||||
limit=limit,
|
||||
search_term=query.strip(),
|
||||
)
|
||||
|
||||
if not games:
|
||||
|
||||
@@ -14,6 +14,7 @@ from typing import Any, Callable, Dict, List, Optional
|
||||
|
||||
import aiohttp
|
||||
|
||||
from .httpread import read_capped
|
||||
from .persistence import PersistentStore
|
||||
|
||||
DEFAULT_CACHE_MB = 500
|
||||
@@ -79,7 +80,9 @@ class ImageCache:
|
||||
async with aiohttp.ClientSession(timeout=timeout) as session:
|
||||
async with session.get(url) as response:
|
||||
response.raise_for_status()
|
||||
return await response.content.read(limit + 1)
|
||||
# limit + 1: an over-limit body must stay over-limit so
|
||||
# ingest_bytes rejects it instead of caching it truncated
|
||||
return await read_capped(response, limit + 1)
|
||||
|
||||
def data_url(self, sha256: str, ext: str) -> Optional[str]:
|
||||
path = self._path(sha256, ext)
|
||||
|
||||
@@ -0,0 +1,291 @@
|
||||
"""News digest fetcher (SPEC-013, FDB-012 news rewrite).
|
||||
|
||||
Replaces the broken pre-1.0-openai `news_feed.py`. Fetches configured
|
||||
RSS/Atom feeds (stdlib, no feedparser dep), builds a compact sanitized
|
||||
headline digest, and writes it to the `{news}` file the responder
|
||||
injects (AIResponder.message). Feeds are external input: titles are
|
||||
sanitized (SAF-03) and each feed URL is SSRF-guarded before fetching.
|
||||
|
||||
CLI: python -m fjerkroa_bot.news --config kroa.toml
|
||||
"""
|
||||
|
||||
import argparse
|
||||
import logging
|
||||
import sys
|
||||
import time
|
||||
from typing import Any, Dict, List, Optional, Tuple
|
||||
|
||||
import defusedxml.ElementTree as ElementTree # hardened XML: feeds are untrusted (XXE/billion-laughs)
|
||||
|
||||
from .ai_responder import sanitize_external_text
|
||||
|
||||
DEFAULT_PER_FEED = 3
|
||||
DEFAULT_MAX_ITEMS = 15
|
||||
FETCH_TIMEOUT_S = 15
|
||||
_ATOM = "{http://www.w3.org/2005/Atom}"
|
||||
|
||||
|
||||
def parse_feed(data: bytes, source: str = "") -> List[Dict[str, str]]:
|
||||
"""Parse RSS or Atom bytes into [{title, link, source}] (tolerant)."""
|
||||
try:
|
||||
root = ElementTree.fromstring(data)
|
||||
except Exception as err:
|
||||
# malformed XML or a blocked entity/DTD attack — tolerate, never raise (NEWS-01)
|
||||
logging.warning(f"news: unparseable/unsafe feed {source!r}: {err!r}")
|
||||
return []
|
||||
items: List[Dict[str, str]] = []
|
||||
# RSS: <rss><channel><item><title/><link/>
|
||||
for item in root.iter("item"):
|
||||
title = (item.findtext("title") or "").strip()
|
||||
link = (item.findtext("link") or "").strip()
|
||||
if title:
|
||||
items.append({"title": title, "link": link, "source": source})
|
||||
# Atom: <feed><entry><title/><link href=/>
|
||||
for entry in root.iter(f"{_ATOM}entry"):
|
||||
title = (entry.findtext(f"{_ATOM}title") or "").strip()
|
||||
link_el = entry.find(f"{_ATOM}link")
|
||||
link = link_el.get("href", "") if link_el is not None else ""
|
||||
if title:
|
||||
items.append({"title": title, "link": link, "source": source})
|
||||
return items
|
||||
|
||||
|
||||
def render_digest(items: List[Dict[str, str]], max_items: int = DEFAULT_MAX_ITEMS) -> str:
|
||||
"""Compact sanitized digest for the {news} prompt slot."""
|
||||
lines = []
|
||||
for item in items[:max_items]:
|
||||
title = sanitize_external_text(item["title"], 200)
|
||||
source = item.get("source", "")
|
||||
link = item.get("link", "")
|
||||
prefix = f"[{source}] " if source else ""
|
||||
lines.append(f"- {prefix}{title}" + (f" ({link})" if link else ""))
|
||||
return "\n".join(lines)
|
||||
|
||||
|
||||
class NewsFetcher:
|
||||
def __init__(self, guard, fetch_bytes) -> None:
|
||||
# injected so tests need no network; production wires aiohttp + guard_url
|
||||
self._guard = guard
|
||||
self._fetch_bytes = fetch_bytes
|
||||
|
||||
async def collect(self, feeds: List[Tuple[str, str]], per_feed: int) -> List[Dict[str, str]]:
|
||||
"""feeds = [(url, label)]; returns deduped items, order preserved."""
|
||||
seen = set()
|
||||
out: List[Dict[str, str]] = []
|
||||
for url, label in feeds:
|
||||
reason = self._guard(url)
|
||||
if reason:
|
||||
logging.warning(f"news: skipping feed {label} — {reason}")
|
||||
continue
|
||||
try:
|
||||
data = await self._fetch_bytes(url)
|
||||
except Exception as err:
|
||||
logging.warning(f"news: fetch failed for {label}: {repr(err)}")
|
||||
continue
|
||||
for item in parse_feed(data, label)[:per_feed]:
|
||||
key = item["title"]
|
||||
if key not in seen:
|
||||
seen.add(key)
|
||||
out.append(item)
|
||||
return out
|
||||
|
||||
|
||||
DEFAULT_SEEN_CAP = 5000
|
||||
DEFAULT_POST_PER_FEED = 5
|
||||
DEFAULT_POST_MAX_PER_RUN = 8
|
||||
|
||||
|
||||
def item_key(item: Dict[str, str]) -> str:
|
||||
return item.get("link") or item.get("title") or ""
|
||||
|
||||
|
||||
class NewsPoster:
|
||||
"""Post NEW feed items to Discord channel webhooks (ggg model, SPEC-013 NEWS-04..06)."""
|
||||
|
||||
def __init__(self, guard, fetch_bytes, post_webhook) -> None:
|
||||
self._guard = guard
|
||||
self._fetch_bytes = fetch_bytes
|
||||
self._post_webhook = post_webhook
|
||||
|
||||
async def run_post(
|
||||
self,
|
||||
feeds: List[Tuple[str, str, str]],
|
||||
webhooks: Dict[str, str],
|
||||
seen: set,
|
||||
per_feed: int,
|
||||
max_per_run: int,
|
||||
seed_only: bool,
|
||||
) -> Tuple[int, set]:
|
||||
"""Returns (posted_count, updated_seen). seed_only marks new items seen without posting."""
|
||||
posted = 0
|
||||
for url, label, channel in feeds:
|
||||
reason = self._guard(url)
|
||||
if reason:
|
||||
logging.warning(f"news-post: skipping feed {label} — {reason}")
|
||||
continue
|
||||
try:
|
||||
data = await self._fetch_bytes(url)
|
||||
except Exception as err:
|
||||
logging.warning(f"news-post: fetch failed for {label}: {repr(err)}")
|
||||
continue
|
||||
for item in parse_feed(data, label)[:per_feed]:
|
||||
key = item_key(item)
|
||||
if not key or key in seen:
|
||||
continue
|
||||
seen.add(key)
|
||||
may_post = not seed_only and posted < max_per_run
|
||||
if may_post and await self._deliver(item, label, channel, webhooks):
|
||||
posted += 1
|
||||
return posted, seen
|
||||
|
||||
async def _deliver(self, item: Dict[str, str], label: str, channel: str, webhooks: Dict[str, str]) -> bool:
|
||||
hook = webhooks.get(channel)
|
||||
if not hook:
|
||||
logging.warning(f"news-post: no webhook for channel {channel!r} ({label})")
|
||||
return False
|
||||
title = sanitize_external_text(item["title"], 300)
|
||||
link = item.get("link", "")
|
||||
content = f"**[{label}]** {title}" + (f"\n{link}" if link else "")
|
||||
try:
|
||||
await self._post_webhook(hook, content)
|
||||
return True
|
||||
except Exception as err:
|
||||
logging.warning(f"news-post: webhook post failed ({label}): {repr(err)}")
|
||||
return False
|
||||
|
||||
|
||||
def load_seen(path: str) -> Tuple[set, bool]:
|
||||
"""(seen-set, existed). Missing/broken state -> empty set, existed=False (seed run)."""
|
||||
import json
|
||||
import os
|
||||
|
||||
if not os.path.exists(path):
|
||||
return set(), False
|
||||
try:
|
||||
with open(path, encoding="utf-8") as fd:
|
||||
return set(json.load(fd)), True
|
||||
except Exception as err:
|
||||
logging.warning(f"news-post: unreadable state {path}: {err!r} — reseeding")
|
||||
return set(), False
|
||||
|
||||
|
||||
def save_seen(path: str, seen: set, cap: int = DEFAULT_SEEN_CAP) -> None:
|
||||
import json
|
||||
|
||||
# keep the newest `cap` keys (insertion order preserved by Python sets? no — use a bounded slice)
|
||||
keys = list(seen)[-cap:]
|
||||
with open(path, "w", encoding="utf-8") as fd:
|
||||
json.dump(keys, fd)
|
||||
|
||||
|
||||
def _post_feeds_from_config(config: Dict[str, Any]) -> List[Tuple[str, str, str]]:
|
||||
feeds = []
|
||||
for entry in config.get("news-post-feeds", []):
|
||||
if isinstance(entry, (list, tuple)) and len(entry) >= 3:
|
||||
feeds.append((str(entry[0]), str(entry[1]), str(entry[2])))
|
||||
return feeds
|
||||
|
||||
|
||||
def _feeds_from_config(config: Dict[str, Any]) -> List[Tuple[str, str]]:
|
||||
"""news-feeds = [["url", "label"], ...] or ["url", ...]."""
|
||||
feeds = []
|
||||
for entry in config.get("news-feeds", []):
|
||||
if isinstance(entry, (list, tuple)):
|
||||
feeds.append((str(entry[0]), str(entry[1]) if len(entry) > 1 else ""))
|
||||
else:
|
||||
feeds.append((str(entry), ""))
|
||||
return feeds
|
||||
|
||||
|
||||
async def _aiohttp_fetch(url: str) -> bytes:
|
||||
import aiohttp
|
||||
|
||||
from .httpread import read_capped
|
||||
|
||||
timeout = aiohttp.ClientTimeout(total=FETCH_TIMEOUT_S)
|
||||
async with aiohttp.ClientSession(timeout=timeout, headers={"User-Agent": "Mozilla/5.0 (compatible; FjerkroaBot-news/1.0)"}) as session:
|
||||
async with session.get(url) as response:
|
||||
response.raise_for_status()
|
||||
return await read_capped(response, 4 * 1024 * 1024)
|
||||
|
||||
|
||||
async def _aiohttp_post(hook: str, content: str) -> None:
|
||||
import aiohttp
|
||||
|
||||
timeout = aiohttp.ClientTimeout(total=FETCH_TIMEOUT_S)
|
||||
async with aiohttp.ClientSession(timeout=timeout) as session:
|
||||
# allowed_mentions none: a headline can never ping the channel (SAF-02 spirit)
|
||||
payload = {"content": content[:2000], "allowed_mentions": {"parse": []}}
|
||||
async with session.post(hook, json=payload) as response:
|
||||
response.raise_for_status()
|
||||
|
||||
|
||||
async def run_post(config: Dict[str, Any]) -> int:
|
||||
"""Webhook-posting mode (ggg): post new items to channels. Returns posted count."""
|
||||
from .url_reader import guard_url
|
||||
|
||||
webhooks = dict(config.get("news-post-webhooks", {}))
|
||||
feeds = _post_feeds_from_config(config)
|
||||
state_path = config.get("news-post-state", "news_state.json")
|
||||
if not webhooks or not feeds:
|
||||
logging.error("news-post: need news-post-webhooks and news-post-feeds")
|
||||
return 0
|
||||
seen, existed = load_seen(state_path)
|
||||
poster = NewsPoster(guard_url, _aiohttp_fetch, _aiohttp_post)
|
||||
posted, seen = await poster.run_post(
|
||||
feeds,
|
||||
webhooks,
|
||||
seen,
|
||||
int(config.get("news-post-per-feed", DEFAULT_POST_PER_FEED)),
|
||||
int(config.get("news-post-max-per-run", DEFAULT_POST_MAX_PER_RUN)),
|
||||
seed_only=not existed, # first run seeds without flooding the channels
|
||||
)
|
||||
save_seen(state_path, seen, int(config.get("news-post-seen-cap", DEFAULT_SEEN_CAP)))
|
||||
logging.info(f"news-post: posted {posted} item(s)" + (" (seed run — nothing posted)" if not existed else ""))
|
||||
return posted
|
||||
|
||||
|
||||
async def run(config: Dict[str, Any]) -> Optional[str]:
|
||||
from .url_reader import guard_url
|
||||
|
||||
out_path = config.get("news")
|
||||
if not out_path:
|
||||
logging.error("news: no `news` output path in config")
|
||||
return None
|
||||
feeds = _feeds_from_config(config)
|
||||
if not feeds:
|
||||
logging.error("news: no `news-feeds` configured")
|
||||
return None
|
||||
fetcher = NewsFetcher(guard_url, _aiohttp_fetch)
|
||||
items = await fetcher.collect(feeds, int(config.get("news-per-feed", DEFAULT_PER_FEED)))
|
||||
digest = render_digest(items, int(config.get("news-max-items", DEFAULT_MAX_ITEMS)))
|
||||
header = f"News as of {time.strftime('%Y-%m-%d %H:%M UTC', time.gmtime())}:\n"
|
||||
with open(out_path, "w", encoding="utf-8") as fd:
|
||||
fd.write(header + digest + "\n")
|
||||
logging.info(f"news: wrote {len(items)} items to {out_path}")
|
||||
return out_path
|
||||
|
||||
|
||||
def main() -> int:
|
||||
import asyncio
|
||||
|
||||
import tomlkit
|
||||
|
||||
logging.basicConfig(level=logging.INFO, format="%(asctime)s [%(levelname)s] %(message)s")
|
||||
parser = argparse.ArgumentParser(
|
||||
description="Fetch RSS/Atom feeds: --post to channel webhooks (ggg) or default {news} digest file (kroa)"
|
||||
)
|
||||
parser.add_argument("--config", required=True)
|
||||
parser.add_argument("--post", action="store_true", help="webhook-posting mode (post new items to Discord channels)")
|
||||
args = parser.parse_args()
|
||||
with open(args.config, encoding="utf-8") as fd:
|
||||
config = tomlkit.load(fd)
|
||||
if args.post:
|
||||
asyncio.run(run_post(config))
|
||||
return 0
|
||||
result = asyncio.run(run(config))
|
||||
return 0 if result else 1
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
sys.exit(main())
|
||||
@@ -18,6 +18,7 @@ from urllib.parse import urljoin, urlparse
|
||||
import aiohttp
|
||||
|
||||
from .ai_responder import sanitize_external_text
|
||||
from .httpread import read_capped
|
||||
|
||||
DEFAULT_MAX_BYTES = 2 * 1024 * 1024
|
||||
DEFAULT_MAX_CHARS = 6000
|
||||
@@ -115,7 +116,7 @@ class URLReader:
|
||||
current = urljoin(current, response.headers["Location"])
|
||||
continue
|
||||
response.raise_for_status()
|
||||
return str(response.url), await response.content.read(max_bytes + 1)
|
||||
return str(response.url), await read_capped(response, max_bytes)
|
||||
raise ValueError("too many redirects")
|
||||
|
||||
async def fetch(self, url: str, channel: str, user: str) -> Dict[str, Any]:
|
||||
|
||||
@@ -15,6 +15,7 @@ dependencies = [
|
||||
"tomlkit>=0.13",
|
||||
"watchdog>=6",
|
||||
"requests>=2.32",
|
||||
"defusedxml>=0.7",
|
||||
]
|
||||
|
||||
[project.scripts]
|
||||
|
||||
@@ -0,0 +1,54 @@
|
||||
# SPEC-013 — News digest
|
||||
|
||||
Replaces the broken pre-1.0-openai `news_feed.py`. A CLI
|
||||
(`python -m fjerkroa_bot.news --config <cfg>`) fetches the
|
||||
`news-feeds` and writes a compact digest to the `news` file that
|
||||
`AIResponder.message` injects into the `{news}` slot. Feeds are
|
||||
external input and operator-configured.
|
||||
|
||||
### NEWS-01 — RSS and Atom parse to items (coverage: test)
|
||||
|
||||
`parse_feed(bytes, label)` extracts `{title, link, source}` from both
|
||||
RSS (`<item>`) and Atom (`<entry>`) documents, tolerates malformed
|
||||
XML (returns an empty list, logs), and never raises.
|
||||
|
||||
### NEWS-02 — Digest is sanitized and bounded (coverage: test)
|
||||
|
||||
`render_digest` caps at `news-max-items`, and every headline passes
|
||||
`sanitize_external_text` (SAF-03) — a feed cannot inject `@everyone`
|
||||
or control characters into the prompt via a headline.
|
||||
|
||||
### NEWS-03 — Feeds are SSRF-guarded and deduped (coverage: test)
|
||||
|
||||
`NewsFetcher.collect` skips any feed URL the SSRF guard rejects,
|
||||
skips feeds that fail to fetch (one bad feed never sinks the run),
|
||||
and drops duplicate headlines across feeds.
|
||||
|
||||
## Webhook posting (ggg model)
|
||||
|
||||
`--post` mode fetches feeds mapped to channels and posts NEW items to
|
||||
the channel's Discord webhook — replacing the py3.8 `getnews.py`
|
||||
(dead play3 feed, 35 MB substring-scan state file, HTML-redirect
|
||||
cruft). Config: `news-post-feeds = [[url, label, channel], …]`,
|
||||
`news-post-webhooks = {channel = url}`, `news-post-state`.
|
||||
|
||||
### NEWS-04 — Only unseen items post, then are marked seen (coverage: test)
|
||||
|
||||
`NewsPoster.run_post` posts each item whose key (link, else title) is
|
||||
not in the seen-set, adds it to the set, and posts to the mapped
|
||||
channel's webhook. Re-runs over the same feed post nothing new.
|
||||
|
||||
### NEWS-05 — First run seeds without flooding (coverage: test)
|
||||
|
||||
With no prior state file (`seed_only`), every current item is marked
|
||||
seen but nothing is posted — migrating off getnews.py never dumps a
|
||||
backlog into the channels. `news-post-max-per-run` caps steady-state
|
||||
posts per run.
|
||||
|
||||
### NEWS-06 — Post failures and bad channels are survived (coverage: test)
|
||||
|
||||
A feed the SSRF guard rejects, a feed that fails to fetch, an item
|
||||
whose channel has no configured webhook, and a webhook POST that
|
||||
raises are each logged and skipped — one failure never sinks the
|
||||
run, and the seen-set still advances for successfully-processed
|
||||
items.
|
||||
@@ -174,6 +174,25 @@ class TestIGDBQuery(unittest.TestCase):
|
||||
self.assertEqual(result, [{"id": 1, "name": "Super Mario Bros"}])
|
||||
|
||||
|
||||
class TestIGDBNativeSearch(unittest.TestCase):
|
||||
def test_build_query_with_search_term(self):
|
||||
"""search_games uses IGDB full-text search, not a name prefix filter."""
|
||||
query = IGDBQuery.build_query(["name"], {"game_type": "= 0"}, limit=5, search_term="Marvel Tōkon")
|
||||
self.assertEqual(query, 'search "Marvel Tōkon"; fields name; limit 5; where game_type = 0;')
|
||||
|
||||
def test_search_term_escapes_quotes_and_backslashes(self):
|
||||
query = IGDBQuery.build_query(["name"], search_term='say "hi" \\ bye')
|
||||
self.assertIn('search "say \\"hi\\" \\\\ bye";', query)
|
||||
|
||||
@patch.object(IGDBQuery, "generalized_igdb_query")
|
||||
def test_search_games_passes_search_term(self, mock_query):
|
||||
mock_query.return_value = []
|
||||
IGDBQuery("cid", "token").search_games("Elden Ring", limit=3)
|
||||
_, kwargs = mock_query.call_args
|
||||
self.assertEqual(kwargs["search_term"], "Elden Ring")
|
||||
self.assertEqual(mock_query.call_args.args[0], {})
|
||||
|
||||
|
||||
class TestIGDBTokenRefresh(unittest.TestCase):
|
||||
@staticmethod
|
||||
def _oauth_response(token="fresh_token", expires_in=5_000_000):
|
||||
|
||||
@@ -45,6 +45,45 @@ class TestIngest(unittest.TestCase):
|
||||
self.assertIsNone(cache.ingest_bytes(PNG, "chat", "alice", "1"))
|
||||
|
||||
|
||||
class TestOversizedDownloadRejected(unittest.IsolatedAsyncioTestCase):
|
||||
async def test_download_stays_over_limit_and_is_rejected(self):
|
||||
"""IMG-10: an over-limit download must be rejected, not cached truncated."""
|
||||
with tempfile.TemporaryDirectory() as tmp:
|
||||
_, cache = make_cache(tmp, {"image-max-bytes": 32})
|
||||
|
||||
class FakeContent:
|
||||
@staticmethod
|
||||
async def iter_chunked(size):
|
||||
yield PNG # 72 bytes > 32
|
||||
|
||||
class FakeResp:
|
||||
content = FakeContent()
|
||||
|
||||
async def __aenter__(self):
|
||||
return self
|
||||
|
||||
async def __aexit__(self, *a):
|
||||
return False
|
||||
|
||||
def raise_for_status(self):
|
||||
pass
|
||||
|
||||
class FakeSession:
|
||||
def get(self, url):
|
||||
return FakeResp()
|
||||
|
||||
async def __aenter__(self):
|
||||
return self
|
||||
|
||||
async def __aexit__(self, *a):
|
||||
return False
|
||||
|
||||
with patch("fjerkroa_bot.images.aiohttp.ClientSession", return_value=FakeSession()):
|
||||
data = await cache._download("http://x.com/big.png")
|
||||
self.assertEqual(len(data), 33) # limit + 1, not silently capped to limit
|
||||
self.assertIsNone(await cache.ingest_url("http://x.com/big.png", "chat", "alice", "1"))
|
||||
|
||||
|
||||
class TestVisionDataUrls(OpsBase):
|
||||
async def test_attachment_becomes_data_url(self):
|
||||
"""IMG-11: the model sees a data: URL, never the CDN link."""
|
||||
|
||||
@@ -0,0 +1,175 @@
|
||||
"""Unit coverage for SPEC-013 news digest (NEWS-01..03)."""
|
||||
|
||||
import unittest
|
||||
from unittest.mock import AsyncMock
|
||||
|
||||
from fjerkroa_bot.news import NewsFetcher, NewsPoster, load_seen, parse_feed, render_digest, save_seen
|
||||
|
||||
RSS = b"""<?xml version="1.0"?><rss><channel>
|
||||
<item><title>Game X released</title><link>https://ex.com/x</link></item>
|
||||
<item><title>Patch Y notes</title><link>https://ex.com/y</link></item>
|
||||
</channel></rss>"""
|
||||
|
||||
ATOM = b"""<?xml version="1.0"?><feed xmlns="http://www.w3.org/2005/Atom">
|
||||
<entry><title>Atom headline</title><link href="https://ex.com/a"/></entry>
|
||||
</feed>"""
|
||||
|
||||
|
||||
class TestParse(unittest.TestCase):
|
||||
def test_rss(self):
|
||||
"""NEWS-01: RSS items parsed with title + link."""
|
||||
items = parse_feed(RSS, "Src")
|
||||
self.assertEqual([i["title"] for i in items], ["Game X released", "Patch Y notes"])
|
||||
self.assertEqual(items[0]["link"], "https://ex.com/x")
|
||||
self.assertEqual(items[0]["source"], "Src")
|
||||
|
||||
def test_atom(self):
|
||||
"""NEWS-01: Atom entries parsed with href link."""
|
||||
items = parse_feed(ATOM, "A")
|
||||
self.assertEqual(items[0]["title"], "Atom headline")
|
||||
self.assertEqual(items[0]["link"], "https://ex.com/a")
|
||||
|
||||
def test_malformed_never_raises(self):
|
||||
"""NEWS-01: garbage XML returns [] without raising."""
|
||||
self.assertEqual(parse_feed(b"<not xml", "bad"), [])
|
||||
self.assertEqual(parse_feed(b"", "empty"), [])
|
||||
|
||||
|
||||
class TestDigest(unittest.TestCase):
|
||||
def test_sanitized_and_capped(self):
|
||||
"""NEWS-02: headlines sanitized, item count capped."""
|
||||
items = [{"title": "@everyone big news \x00", "link": "", "source": "S"} for _ in range(20)]
|
||||
digest = render_digest(items, max_items=5)
|
||||
self.assertEqual(digest.count("\n"), 4) # 5 lines
|
||||
self.assertNotIn("@everyone", digest)
|
||||
self.assertNotIn("\x00", digest)
|
||||
|
||||
|
||||
class TestCollect(unittest.IsolatedAsyncioTestCase):
|
||||
async def test_ssrf_skip_and_dedup(self):
|
||||
"""NEWS-03: guarded feed skipped, dup titles dropped, bad fetch survived."""
|
||||
|
||||
def guard(url):
|
||||
return "refused" if "internal" in url else None
|
||||
|
||||
async def fetch(url):
|
||||
if "boom" in url:
|
||||
raise ValueError("boom")
|
||||
return RSS # same content from two feeds -> dedup
|
||||
|
||||
fetcher = NewsFetcher(guard, fetch)
|
||||
feeds = [
|
||||
("https://a.com/feed", "A"),
|
||||
("https://internal/feed", "Internal"), # SSRF-skipped
|
||||
("https://boom.com/feed", "Boom"), # fetch fails
|
||||
("https://b.com/feed", "B"), # same RSS -> dup titles dropped
|
||||
]
|
||||
items = await fetcher.collect(feeds, per_feed=5)
|
||||
titles = [i["title"] for i in items]
|
||||
self.assertEqual(titles, ["Game X released", "Patch Y notes"]) # deduped, internal+boom skipped
|
||||
|
||||
async def test_per_feed_limit(self):
|
||||
"""NEWS-03: per-feed cap honored."""
|
||||
fetcher = NewsFetcher(lambda u: None, AsyncMock(return_value=RSS))
|
||||
items = await fetcher.collect([("https://a.com", "A")], per_feed=1)
|
||||
self.assertEqual(len(items), 1)
|
||||
|
||||
|
||||
class TestPoster(unittest.IsolatedAsyncioTestCase):
|
||||
def poster(self, posts):
|
||||
async def fetch(url):
|
||||
return RSS
|
||||
|
||||
async def post(hook, content):
|
||||
posts.append((hook, content))
|
||||
|
||||
return NewsPoster(lambda u: None, fetch, post)
|
||||
|
||||
async def test_posts_unseen_then_dedups(self):
|
||||
"""NEWS-04: unseen items post to the mapped webhook; re-run posts nothing."""
|
||||
posts = []
|
||||
poster = self.poster(posts)
|
||||
feeds = [("https://a.com/feed", "PS", "news")]
|
||||
hooks = {"news": "https://discord.com/api/webhooks/x"}
|
||||
posted, seen = await poster.run_post(feeds, hooks, set(), per_feed=5, max_per_run=8, seed_only=False)
|
||||
self.assertEqual(posted, 2)
|
||||
self.assertIn("PS", posts[0][1])
|
||||
self.assertIn("https://discord.com/api/webhooks/x", posts[0][0])
|
||||
# re-run with the accumulated seen -> nothing new
|
||||
posts.clear()
|
||||
posted2, _ = await poster.run_post(feeds, hooks, seen, per_feed=5, max_per_run=8, seed_only=False)
|
||||
self.assertEqual(posted2, 0)
|
||||
self.assertEqual(posts, [])
|
||||
|
||||
async def test_seed_run_posts_nothing(self):
|
||||
"""NEWS-05: seed_only marks items seen without posting."""
|
||||
posts = []
|
||||
poster = self.poster(posts)
|
||||
feeds = [("https://a.com/feed", "PS", "news")]
|
||||
posted, seen = await poster.run_post(feeds, {"news": "h"}, set(), 5, 8, seed_only=True)
|
||||
self.assertEqual(posted, 0)
|
||||
self.assertEqual(posts, [])
|
||||
self.assertEqual(len(seen), 2) # both marked seen
|
||||
|
||||
async def test_max_per_run_caps(self):
|
||||
"""NEWS-05: max-per-run caps posts; extras stay seen (not re-posted next run)."""
|
||||
posts = []
|
||||
poster = self.poster(posts)
|
||||
feeds = [("https://a.com/feed", "PS", "news")]
|
||||
posted, seen = await poster.run_post(feeds, {"news": "h"}, set(), per_feed=5, max_per_run=1, seed_only=False)
|
||||
self.assertEqual(posted, 1)
|
||||
self.assertEqual(len(seen), 2) # both seen, only one posted
|
||||
|
||||
async def test_failures_survived(self):
|
||||
"""NEWS-06: SSRF-skip, fetch fail, missing webhook, post error each survive."""
|
||||
posts = []
|
||||
|
||||
async def fetch(url):
|
||||
if "boom" in url:
|
||||
raise ValueError("boom")
|
||||
return RSS
|
||||
|
||||
async def post(hook, content):
|
||||
if hook == "bad":
|
||||
raise RuntimeError("post failed")
|
||||
posts.append((hook, content))
|
||||
|
||||
def guard(url):
|
||||
return "refused" if "internal" in url else None
|
||||
|
||||
poster = NewsPoster(guard, fetch, post)
|
||||
feeds = [
|
||||
("https://internal/feed", "I", "news"), # SSRF-skipped
|
||||
("https://boom.com/feed", "B", "news"), # fetch fails
|
||||
("https://ok.com/feed", "OK", "nowhere"), # no webhook for channel
|
||||
("https://ok2.com/feed", "OK2", "news"), # webhook raises
|
||||
]
|
||||
posted, seen = await poster.run_post(feeds, {"news": "bad"}, set(), 5, 8, seed_only=False)
|
||||
self.assertEqual(posted, 0) # everything failed/skipped, no crash
|
||||
|
||||
|
||||
class TestSeenState(unittest.TestCase):
|
||||
def test_roundtrip_and_seed_detection(self):
|
||||
"""NEWS-05: missing state -> (empty, existed=False); saved state reloads."""
|
||||
import tempfile
|
||||
from pathlib import Path
|
||||
|
||||
with tempfile.TemporaryDirectory() as tmp:
|
||||
path = str(Path(tmp) / "state.json")
|
||||
seen, existed = load_seen(path)
|
||||
self.assertEqual((seen, existed), (set(), False))
|
||||
save_seen(path, {"a", "b", "c"}, cap=5000)
|
||||
reloaded, existed2 = load_seen(path)
|
||||
self.assertEqual(reloaded, {"a", "b", "c"})
|
||||
self.assertTrue(existed2)
|
||||
|
||||
def test_cap_bounds_state(self):
|
||||
"""NEWS-05: save keeps at most `cap` keys."""
|
||||
import json
|
||||
import tempfile
|
||||
from pathlib import Path
|
||||
|
||||
with tempfile.TemporaryDirectory() as tmp:
|
||||
path = str(Path(tmp) / "state.json")
|
||||
save_seen(path, {f"k{i}" for i in range(100)}, cap=10)
|
||||
self.assertEqual(len(json.load(open(path))), 10)
|
||||
@@ -91,6 +91,46 @@ class TestTextExtraction(unittest.TestCase):
|
||||
self.assertNotIn("x{}", text)
|
||||
|
||||
|
||||
class TestBodyReadCollectsAllChunks(unittest.IsolatedAsyncioTestCase):
|
||||
async def test_get_reads_past_first_chunk(self):
|
||||
"""URL-05 regression: body arrives in many chunks; all are collected up to the cap."""
|
||||
reader = URLReader(lambda: {}, None)
|
||||
chunks = [b"<title>t</title>", b"<p>middle</p>", b"<p>end</p>"]
|
||||
|
||||
class FakeContent:
|
||||
@staticmethod
|
||||
async def iter_chunked(size):
|
||||
for chunk in chunks:
|
||||
yield chunk
|
||||
|
||||
class FakeResp:
|
||||
status = 200
|
||||
headers = {}
|
||||
url = "http://safe.example.com"
|
||||
content = FakeContent()
|
||||
|
||||
async def __aenter__(self):
|
||||
return self
|
||||
|
||||
async def __aexit__(self, *a):
|
||||
return False
|
||||
|
||||
def raise_for_status(self):
|
||||
pass
|
||||
|
||||
class FakeSession:
|
||||
def get(self, url, allow_redirects=False):
|
||||
return FakeResp()
|
||||
|
||||
with patch("fjerkroa_bot.url_reader.guard_url", return_value=None):
|
||||
_, body = await reader._get(FakeSession(), "http://safe.example.com", 1000)
|
||||
self.assertEqual(body, b"".join(chunks))
|
||||
|
||||
with patch("fjerkroa_bot.url_reader.guard_url", return_value=None):
|
||||
_, body = await reader._get(FakeSession(), "http://safe.example.com", 20)
|
||||
self.assertEqual(body, b"".join(chunks)[:20])
|
||||
|
||||
|
||||
class TestFetchSanitizes(unittest.IsolatedAsyncioTestCase):
|
||||
async def test_fetch_result_is_sanitized_and_capped(self):
|
||||
"""URL-05: fetch output is length-capped and @everyone-neutralized."""
|
||||
|
||||
@@ -627,6 +627,7 @@ version = "3.0.0"
|
||||
source = { editable = "." }
|
||||
dependencies = [
|
||||
{ name = "aiohttp" },
|
||||
{ name = "defusedxml" },
|
||||
{ name = "discord-py" },
|
||||
{ name = "openai" },
|
||||
{ name = "requests" },
|
||||
@@ -656,6 +657,7 @@ dev = [
|
||||
[package.metadata]
|
||||
requires-dist = [
|
||||
{ name = "aiohttp", specifier = ">=3.12" },
|
||||
{ name = "defusedxml", specifier = ">=0.7" },
|
||||
{ name = "discord-py", specifier = ">=2.5,<3" },
|
||||
{ name = "openai", specifier = ">=2.45" },
|
||||
{ name = "requests", specifier = ">=2.32" },
|
||||
|
||||
Reference in New Issue
Block a user