Compare commits

...

3 Commits

Author SHA1 Message Date
Oleksandr Kozachuk 2caa18a17f igdb search + capped http reads: native fulltext, full-page fetch
search_games used `where name ~ "<q>"*`: prefix-only, diacritic- and
word-order-sensitive -- "MARVEL Tōkon" found nothing though IGDB has
it. Switch to IGDB's native `search` clause (relevance-ranked,
diacritic-insensitive).

fetch_url and image downloads read bodies with content.read(n), which
returns only the first buffered chunk (~7 KB): pages collapsed to
their <title>. httpread.read_capped collects chunks up to the byte
cap; image downloads read limit+1 so over-limit files are still
rejected instead of cached truncated (IMG-10).

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
2026-07-13 19:29:16 +02:00
Oleksandr Kozachuk d4eec4088d news digest (spec-013): rss/atom -> {news} file via cron
Replaces the broken pre-1.0-openai news_feed.py. Stdlib parsing with
defusedxml (feeds are untrusted XML), titles sanitized (SAF-03), feed
URLs SSRF-guarded. CLI: python -m fjerkroa_bot.news --config <cfg>.
2026-07-13 19:28:41 +02:00
Oleksandr Kozachuk 86e631926f igdb: auto-refresh twitch token; category -> game_type filter
Static app tokens expire after ~60 days -> every lookup failed with
401. With igdb-client-secret set, the bot fetches the token via
client-credentials OAuth itself, refreshes a day before expiry and
retries once on 401; a static igdb-access-token still works.

IGDB renamed games.category to game_type: the category = 0 filter in
search_games silently matched nothing even with a valid token.

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
2026-07-13 18:41:31 +02:00
16 changed files with 557 additions and 29 deletions
+21 -8
View File
@@ -20,13 +20,6 @@ The bot now supports real-time video game information through IGDB (Internet Gam
- **Category**: Select appropriate category - **Category**: Select appropriate category
3. Note down your **Client ID** 3. Note down your **Client ID**
4. Generate a **Client Secret** 4. Generate a **Client Secret**
5. Get an access token using this curl command:
```bash
curl -X POST 'https://id.twitch.tv/oauth2/token' \
-H 'Content-Type: application/x-www-form-urlencoded' \
-d 'client_id=YOUR_CLIENT_ID&client_secret=YOUR_CLIENT_SECRET&grant_type=client_credentials'
```
6. Save the `access_token` from the response
### 2. Configure the Bot ### 2. Configure the Bot
@@ -35,6 +28,25 @@ Update your `config.toml` file:
```toml ```toml
# IGDB Configuration for game information # IGDB Configuration for game information
igdb-client-id = "your_actual_client_id_here" igdb-client-id = "your_actual_client_id_here"
igdb-client-secret = "your_actual_client_secret_here"
enable-game-info = true
```
With the client secret configured, the bot fetches an app access token from
Twitch itself and refreshes it automatically before it expires (Twitch app
tokens live ~60 days) — no manual token handling needed.
Alternatively, a static token still works (legacy setup — it expires after
~60 days and then game lookups fail with 401 until you replace it):
```bash
curl -X POST 'https://id.twitch.tv/oauth2/token' \
-H 'Content-Type: application/x-www-form-urlencoded' \
-d 'client_id=YOUR_CLIENT_ID&client_secret=YOUR_CLIENT_SECRET&grant_type=client_credentials'
```
```toml
igdb-client-id = "your_actual_client_id_here"
igdb-access-token = "your_actual_access_token_here" igdb-access-token = "your_actual_access_token_here"
enable-game-info = true enable-game-info = true
``` ```
@@ -99,7 +111,8 @@ The integration provides two OpenAI functions:
- Verify client ID and access token are set - Verify client ID and access token are set
2. **Authentication errors** 2. **Authentication errors**
- Regenerate access token (they expire) - Prefer `igdb-client-secret` — the bot then refreshes tokens itself
- With a static `igdb-access-token`: regenerate it (they expire)
- Verify client ID matches your Twitch app - Verify client ID matches your Twitch app
3. **No game results** 3. **No game results**
+17 -1
View File
@@ -14,7 +14,11 @@ system = "You are a smart AI assistant with access to real-time video game infor
# IGDB Configuration for game information # IGDB Configuration for game information
igdb-client-id = "YOUR_IGDB_CLIENT_ID" igdb-client-id = "YOUR_IGDB_CLIENT_ID"
igdb-access-token = "YOUR_IGDB_ACCESS_TOKEN" # With the Twitch app client secret set, the bot fetches and refreshes the
# access token itself (recommended). A static igdb-access-token still works
# but expires after ~60 days.
igdb-client-secret = "YOUR_IGDB_CLIENT_SECRET"
# igdb-access-token = "YOUR_IGDB_ACCESS_TOKEN"
enable-game-info = true enable-game-info = true
# --- operator / safety (SPEC-003, SPEC-006) --- # --- operator / safety (SPEC-003, SPEC-006) ---
@@ -76,3 +80,15 @@ enable-game-info = true
# Ops (SPEC-012): consecutive OpenAI failures before a staff alert # Ops (SPEC-012): consecutive OpenAI failures before a staff alert
# api-error-alert-threshold = 5 # api-error-alert-threshold = 5
# Backups: cron runs deploy/backup_db.py daily -> ~/backups/<bot>/ (keep 14) # Backups: cron runs deploy/backup_db.py daily -> ~/backups/<bot>/ (keep 14)
# News digest (SPEC-013) — `python -m fjerkroa_bot.news --config X.toml` via cron;
# writes the {news} file. Feeds are [url, label] pairs (RSS or Atom):
# news = "news_feed.txt"
# news-per-feed = 3
# news-max-items = 15
# news-feeds = [
# ["https://blog.playstation.com/feed/", "PS"],
# ["https://kotaku.com/rss", "Kotaku"],
# ["https://www.pushsquare.com/feeds/latest", "Push"],
# ["https://mein-mmo.de/feed/", "MeinMMO"],
# ]
+18
View File
@@ -0,0 +1,18 @@
"""Bounded HTTP body read (leaf module, no intra-package imports).
`response.content.read(n)` returns whatever is buffered, not n bytes,
so it silently truncates large or chunked bodies (and web feeds/pages
parse to garbage). This accumulates decompressed chunks up to a hard
cap instead.
"""
CHUNK = 65536
async def read_capped(response, max_bytes: int) -> bytes:
buf = bytearray()
async for chunk in response.content.iter_chunked(CHUNK):
buf.extend(chunk)
if len(buf) > max_bytes:
break
return bytes(buf[:max_bytes])
+50 -11
View File
@@ -1,30 +1,67 @@
import logging import logging
import time
from functools import cache from functools import cache
from typing import Any, Dict, List, Optional from typing import Any, Dict, List, Optional
import requests import requests
TWITCH_OAUTH_URL = "https://id.twitch.tv/oauth2/token"
# Refresh this long before Twitch expires the token (app tokens live ~60 days)
TOKEN_REFRESH_MARGIN = 86400
class IGDBQuery(object): class IGDBQuery(object):
def __init__(self, client_id, igdb_api_key): def __init__(self, client_id, igdb_api_key=None, client_secret=None):
self.client_id = client_id self.client_id = client_id
self.igdb_api_key = igdb_api_key self.igdb_api_key = igdb_api_key
self.client_secret = client_secret
# Unknown for statically configured tokens; set after each refresh
self._token_expires_at = None
def _refresh_token(self):
response = requests.post(
TWITCH_OAUTH_URL,
params={"client_id": self.client_id, "client_secret": self.client_secret, "grant_type": "client_credentials"},
)
response.raise_for_status()
data = response.json()
self.igdb_api_key = data["access_token"]
self._token_expires_at = time.time() + data.get("expires_in", 0) - TOKEN_REFRESH_MARGIN
logging.info("IGDB: refreshed Twitch app access token")
def _ensure_token(self):
if not self.client_secret:
return
if not self.igdb_api_key or (self._token_expires_at is not None and time.time() >= self._token_expires_at):
self._refresh_token()
def send_igdb_request(self, endpoint, query_body): def send_igdb_request(self, endpoint, query_body):
igdb_url = f"https://api.igdb.com/v4/{endpoint}" igdb_url = f"https://api.igdb.com/v4/{endpoint}"
headers = {"Client-ID": self.client_id, "Authorization": f"Bearer {self.igdb_api_key}"}
try: try:
response = requests.post(igdb_url, headers=headers, data=query_body) self._ensure_token()
response = self._post_igdb(igdb_url, query_body)
if self.client_secret and response.status_code == 401:
# Token expired server-side (e.g. statically configured) — refresh and retry once
self._refresh_token()
response = self._post_igdb(igdb_url, query_body)
response.raise_for_status() response.raise_for_status()
return response.json() return response.json()
except requests.RequestException as e: except requests.RequestException as e:
print(f"Error during IGDB API request: {e}") print(f"Error during IGDB API request: {e}")
return None return None
def _post_igdb(self, igdb_url, query_body):
headers = {"Client-ID": self.client_id, "Authorization": f"Bearer {self.igdb_api_key}"}
return requests.post(igdb_url, headers=headers, data=query_body)
@staticmethod @staticmethod
def build_query(fields, filters=None, limit=10, offset=None): def build_query(fields, filters=None, limit=10, offset=None, search_term=None):
query = f"fields {','.join(fields) if fields is not None and len(fields) > 0 else '*'}; limit {limit};" query = ""
if search_term:
escaped = search_term.replace("\\", "\\\\").replace('"', '\\"')
query += f'search "{escaped}"; '
query += f"fields {','.join(fields) if fields is not None and len(fields) > 0 else '*'}; limit {limit};"
if offset is not None: if offset is not None:
query += f" offset {offset};" query += f" offset {offset};"
if filters: if filters:
@@ -32,12 +69,12 @@ class IGDBQuery(object):
query += " where " + " & ".join(filter_statements) + ";" query += " where " + " & ".join(filter_statements) + ";"
return query return query
def generalized_igdb_query(self, params, endpoint, fields, additional_filters=None, limit=10, offset=None): def generalized_igdb_query(self, params, endpoint, fields, additional_filters=None, limit=10, offset=None, search_term=None):
all_filters = {key: f'~ "{value}"*' for key, value in params.items() if value} all_filters = {key: f'~ "{value}"*' for key, value in params.items() if value}
if additional_filters: if additional_filters:
all_filters.update(additional_filters) all_filters.update(additional_filters)
query = self.build_query(fields, all_filters, limit, offset) query = self.build_query(fields, all_filters, limit, offset, search_term)
data = self.send_igdb_request(endpoint, query) data = self.send_igdb_request(endpoint, query)
print(f"{endpoint}: {query} -> {data}") print(f"{endpoint}: {query} -> {data}")
return data return data
@@ -79,7 +116,7 @@ class IGDBQuery(object):
"id", "id",
"name", "name",
"alternative_names", "alternative_names",
"category", "game_type",
"release_dates", "release_dates",
"franchise", "franchise",
"language_supports", "language_supports",
@@ -101,9 +138,10 @@ class IGDBQuery(object):
return None return None
try: try:
# Search for games with fuzzy matching # IGDB native full-text search: diacritic- and word-order-insensitive,
# unlike a `name ~ "..."*` prefix filter
games = self.generalized_igdb_query( games = self.generalized_igdb_query(
{"name": query.strip()}, {},
"games", "games",
[ [
"id", "id",
@@ -120,8 +158,9 @@ class IGDBQuery(object):
"themes.name", "themes.name",
"cover.url", "cover.url",
], ],
additional_filters={"category": "= 0"}, # Main games only additional_filters={"game_type": "= 0"}, # Main games only (IGDB renamed category -> game_type)
limit=limit, limit=limit,
search_term=query.strip(),
) )
if not games: if not games:
+4 -1
View File
@@ -14,6 +14,7 @@ from typing import Any, Callable, Dict, List, Optional
import aiohttp import aiohttp
from .httpread import read_capped
from .persistence import PersistentStore from .persistence import PersistentStore
DEFAULT_CACHE_MB = 500 DEFAULT_CACHE_MB = 500
@@ -79,7 +80,9 @@ class ImageCache:
async with aiohttp.ClientSession(timeout=timeout) as session: async with aiohttp.ClientSession(timeout=timeout) as session:
async with session.get(url) as response: async with session.get(url) as response:
response.raise_for_status() response.raise_for_status()
return await response.content.read(limit + 1) # limit + 1: an over-limit body must stay over-limit so
# ingest_bytes rejects it instead of caching it truncated
return await read_capped(response, limit + 1)
def data_url(self, sha256: str, ext: str) -> Optional[str]: def data_url(self, sha256: str, ext: str) -> Optional[str]:
path = self._path(sha256, ext) path = self._path(sha256, ext)
+153
View File
@@ -0,0 +1,153 @@
"""News digest fetcher (SPEC-013, FDB-012 news rewrite).
Replaces the broken pre-1.0-openai `news_feed.py`. Fetches configured
RSS/Atom feeds (stdlib, no feedparser dep), builds a compact sanitized
headline digest, and writes it to the `{news}` file the responder
injects (AIResponder.message). Feeds are external input: titles are
sanitized (SAF-03) and each feed URL is SSRF-guarded before fetching.
CLI: python -m fjerkroa_bot.news --config kroa.toml
"""
import argparse
import logging
import sys
import time
from typing import Any, Dict, List, Optional, Tuple
import defusedxml.ElementTree as ElementTree # hardened XML: feeds are untrusted (XXE/billion-laughs)
from .ai_responder import sanitize_external_text
DEFAULT_PER_FEED = 3
DEFAULT_MAX_ITEMS = 15
FETCH_TIMEOUT_S = 15
_ATOM = "{http://www.w3.org/2005/Atom}"
def parse_feed(data: bytes, source: str = "") -> List[Dict[str, str]]:
"""Parse RSS or Atom bytes into [{title, link, source}] (tolerant)."""
try:
root = ElementTree.fromstring(data)
except Exception as err:
# malformed XML or a blocked entity/DTD attack — tolerate, never raise (NEWS-01)
logging.warning(f"news: unparseable/unsafe feed {source!r}: {err!r}")
return []
items: List[Dict[str, str]] = []
# RSS: <rss><channel><item><title/><link/>
for item in root.iter("item"):
title = (item.findtext("title") or "").strip()
link = (item.findtext("link") or "").strip()
if title:
items.append({"title": title, "link": link, "source": source})
# Atom: <feed><entry><title/><link href=/>
for entry in root.iter(f"{_ATOM}entry"):
title = (entry.findtext(f"{_ATOM}title") or "").strip()
link_el = entry.find(f"{_ATOM}link")
link = link_el.get("href", "") if link_el is not None else ""
if title:
items.append({"title": title, "link": link, "source": source})
return items
def render_digest(items: List[Dict[str, str]], max_items: int = DEFAULT_MAX_ITEMS) -> str:
"""Compact sanitized digest for the {news} prompt slot."""
lines = []
for item in items[:max_items]:
title = sanitize_external_text(item["title"], 200)
source = item.get("source", "")
link = item.get("link", "")
prefix = f"[{source}] " if source else ""
lines.append(f"- {prefix}{title}" + (f" ({link})" if link else ""))
return "\n".join(lines)
class NewsFetcher:
def __init__(self, guard, fetch_bytes) -> None:
# injected so tests need no network; production wires aiohttp + guard_url
self._guard = guard
self._fetch_bytes = fetch_bytes
async def collect(self, feeds: List[Tuple[str, str]], per_feed: int) -> List[Dict[str, str]]:
"""feeds = [(url, label)]; returns deduped items, order preserved."""
seen = set()
out: List[Dict[str, str]] = []
for url, label in feeds:
reason = self._guard(url)
if reason:
logging.warning(f"news: skipping feed {label}{reason}")
continue
try:
data = await self._fetch_bytes(url)
except Exception as err:
logging.warning(f"news: fetch failed for {label}: {repr(err)}")
continue
for item in parse_feed(data, label)[:per_feed]:
key = item["title"]
if key not in seen:
seen.add(key)
out.append(item)
return out
def _feeds_from_config(config: Dict[str, Any]) -> List[Tuple[str, str]]:
"""news-feeds = [["url", "label"], ...] or ["url", ...]."""
feeds = []
for entry in config.get("news-feeds", []):
if isinstance(entry, (list, tuple)):
feeds.append((str(entry[0]), str(entry[1]) if len(entry) > 1 else ""))
else:
feeds.append((str(entry), ""))
return feeds
async def _aiohttp_fetch(url: str) -> bytes:
import aiohttp
from .httpread import read_capped
timeout = aiohttp.ClientTimeout(total=FETCH_TIMEOUT_S)
async with aiohttp.ClientSession(timeout=timeout, headers={"User-Agent": "Mozilla/5.0 (compatible; FjerkroaBot-news/1.0)"}) as session:
async with session.get(url) as response:
response.raise_for_status()
return await read_capped(response, 4 * 1024 * 1024)
async def run(config: Dict[str, Any]) -> Optional[str]:
from .url_reader import guard_url
out_path = config.get("news")
if not out_path:
logging.error("news: no `news` output path in config")
return None
feeds = _feeds_from_config(config)
if not feeds:
logging.error("news: no `news-feeds` configured")
return None
fetcher = NewsFetcher(guard_url, _aiohttp_fetch)
items = await fetcher.collect(feeds, int(config.get("news-per-feed", DEFAULT_PER_FEED)))
digest = render_digest(items, int(config.get("news-max-items", DEFAULT_MAX_ITEMS)))
header = f"News as of {time.strftime('%Y-%m-%d %H:%M UTC', time.gmtime())}:\n"
with open(out_path, "w", encoding="utf-8") as fd:
fd.write(header + digest + "\n")
logging.info(f"news: wrote {len(items)} items to {out_path}")
return out_path
def main() -> int:
import asyncio
import tomlkit
logging.basicConfig(level=logging.INFO, format="%(asctime)s [%(levelname)s] %(message)s")
parser = argparse.ArgumentParser(description="Fetch RSS/Atom feeds into the {news} digest file")
parser.add_argument("--config", required=True)
args = parser.parse_args()
with open(args.config, encoding="utf-8") as fd:
config = tomlkit.load(fd)
result = asyncio.run(run(config))
return 0 if result else 1
if __name__ == "__main__":
sys.exit(main())
+9 -5
View File
@@ -137,16 +137,20 @@ class OpenAIResponder(AIResponder, LeonardoAIDrawMixIn):
# Initialize IGDB if enabled # Initialize IGDB if enabled
self.igdb = None self.igdb = None
igdb_client_id = self.config.get("igdb-client-id")
igdb_client_secret = self.config.get("igdb-client-secret")
igdb_access_token = self.config.get("igdb-access-token")
logging.info("IGDB Configuration Check:") logging.info("IGDB Configuration Check:")
logging.info(f" enable-game-info: {self.config.get('enable-game-info', 'NOT SET')}") logging.info(f" enable-game-info: {self.config.get('enable-game-info', 'NOT SET')}")
logging.info(f" igdb-client-id: {'SET' if self.config.get('igdb-client-id') else 'NOT SET'}") logging.info(f" igdb-client-id: {'SET' if igdb_client_id else 'NOT SET'}")
logging.info(f" igdb-access-token: {'SET' if self.config.get('igdb-access-token') else 'NOT SET'}") logging.info(f" igdb-client-secret: {'SET' if igdb_client_secret else 'NOT SET'}")
logging.info(f" igdb-access-token: {'SET' if igdb_access_token else 'NOT SET'}")
if self.config.get("enable-game-info", False) and self.config.get("igdb-client-id") and self.config.get("igdb-access-token"): if self.config.get("enable-game-info", False) and igdb_client_id and (igdb_client_secret or igdb_access_token):
try: try:
self.igdb = IGDBQuery(self.config["igdb-client-id"], self.config["igdb-access-token"]) self.igdb = IGDBQuery(igdb_client_id, igdb_access_token, client_secret=igdb_client_secret)
logging.info("✅ IGDB integration SUCCESSFULLY enabled for game information") logging.info("✅ IGDB integration SUCCESSFULLY enabled for game information")
logging.info(f" Client ID: {self.config['igdb-client-id'][:8]}...") logging.info(f" Client ID: {igdb_client_id[:8]}...")
logging.info(f" Available functions: {len(self.igdb.get_openai_functions())}") logging.info(f" Available functions: {len(self.igdb.get_openai_functions())}")
except Exception as e: except Exception as e:
logging.error(f"❌ Failed to initialize IGDB: {e}") logging.error(f"❌ Failed to initialize IGDB: {e}")
+2 -1
View File
@@ -18,6 +18,7 @@ from urllib.parse import urljoin, urlparse
import aiohttp import aiohttp
from .ai_responder import sanitize_external_text from .ai_responder import sanitize_external_text
from .httpread import read_capped
DEFAULT_MAX_BYTES = 2 * 1024 * 1024 DEFAULT_MAX_BYTES = 2 * 1024 * 1024
DEFAULT_MAX_CHARS = 6000 DEFAULT_MAX_CHARS = 6000
@@ -115,7 +116,7 @@ class URLReader:
current = urljoin(current, response.headers["Location"]) current = urljoin(current, response.headers["Location"])
continue continue
response.raise_for_status() response.raise_for_status()
return str(response.url), await response.content.read(max_bytes + 1) return str(response.url), await read_capped(response, max_bytes)
raise ValueError("too many redirects") raise ValueError("too many redirects")
async def fetch(self, url: str, channel: str, user: str) -> Dict[str, Any]: async def fetch(self, url: str, channel: str, user: str) -> Dict[str, Any]:
+1
View File
@@ -15,6 +15,7 @@ dependencies = [
"tomlkit>=0.13", "tomlkit>=0.13",
"watchdog>=6", "watchdog>=6",
"requests>=2.32", "requests>=2.32",
"defusedxml>=0.7",
] ]
[project.scripts] [project.scripts]
+25
View File
@@ -0,0 +1,25 @@
# SPEC-013 — News digest
Replaces the broken pre-1.0-openai `news_feed.py`. A CLI
(`python -m fjerkroa_bot.news --config <cfg>`) fetches the
`news-feeds` and writes a compact digest to the `news` file that
`AIResponder.message` injects into the `{news}` slot. Feeds are
external input and operator-configured.
### NEWS-01 — RSS and Atom parse to items (coverage: test)
`parse_feed(bytes, label)` extracts `{title, link, source}` from both
RSS (`<item>`) and Atom (`<entry>`) documents, tolerates malformed
XML (returns an empty list, logs), and never raises.
### NEWS-02 — Digest is sanitized and bounded (coverage: test)
`render_digest` caps at `news-max-items`, and every headline passes
`sanitize_external_text` (SAF-03) — a feed cannot inject `@everyone`
or control characters into the prompt via a headline.
### NEWS-03 — Feeds are SSRF-guarded and deduped (coverage: test)
`NewsFetcher.collect` skips any feed URL the SSRF guard rejects,
skips feeds that fail to fetch (one bad feed never sinks the run),
and drops duplicate headlines across feeds.
+1 -1
View File
@@ -28,7 +28,7 @@ class TestIGDBIntegration(unittest.IsolatedAsyncioTestCase):
responder = OpenAIResponder(self.config_with_igdb) responder = OpenAIResponder(self.config_with_igdb)
mock_igdb.assert_called_once_with("test_client", "test_token") mock_igdb.assert_called_once_with("test_client", "test_token", client_secret=None)
self.assertEqual(responder.igdb, mock_igdb_instance) self.assertEqual(responder.igdb, mock_igdb_instance)
def test_igdb_initialization_disabled(self): def test_igdb_initialization_disabled(self):
+100 -1
View File
@@ -160,7 +160,7 @@ class TestIGDBQuery(unittest.TestCase):
"id", "id",
"name", "name",
"alternative_names", "alternative_names",
"category", "game_type",
"release_dates", "release_dates",
"franchise", "franchise",
"language_supports", "language_supports",
@@ -174,5 +174,104 @@ class TestIGDBQuery(unittest.TestCase):
self.assertEqual(result, [{"id": 1, "name": "Super Mario Bros"}]) self.assertEqual(result, [{"id": 1, "name": "Super Mario Bros"}])
class TestIGDBNativeSearch(unittest.TestCase):
def test_build_query_with_search_term(self):
"""search_games uses IGDB full-text search, not a name prefix filter."""
query = IGDBQuery.build_query(["name"], {"game_type": "= 0"}, limit=5, search_term="Marvel Tōkon")
self.assertEqual(query, 'search "Marvel Tōkon"; fields name; limit 5; where game_type = 0;')
def test_search_term_escapes_quotes_and_backslashes(self):
query = IGDBQuery.build_query(["name"], search_term='say "hi" \\ bye')
self.assertIn('search "say \\"hi\\" \\\\ bye";', query)
@patch.object(IGDBQuery, "generalized_igdb_query")
def test_search_games_passes_search_term(self, mock_query):
mock_query.return_value = []
IGDBQuery("cid", "token").search_games("Elden Ring", limit=3)
_, kwargs = mock_query.call_args
self.assertEqual(kwargs["search_term"], "Elden Ring")
self.assertEqual(mock_query.call_args.args[0], {})
class TestIGDBTokenRefresh(unittest.TestCase):
@staticmethod
def _oauth_response(token="fresh_token", expires_in=5_000_000):
response = Mock()
response.json.return_value = {"access_token": token, "expires_in": expires_in}
response.raise_for_status.return_value = None
return response
@staticmethod
def _api_response(payload, status_code=200):
response = Mock()
response.status_code = status_code
response.json.return_value = payload
response.raise_for_status.return_value = None
return response
@patch("fjerkroa_bot.igdblib.requests.post")
def test_fetches_token_when_only_secret_configured(self, mock_post):
"""Without a static token, the first request fetches one via Twitch OAuth."""
mock_post.side_effect = [self._oauth_response(), self._api_response([{"id": 1}])]
igdb = IGDBQuery("cid", client_secret="secret")
result = igdb.send_igdb_request("games", "fields name; limit 1;")
self.assertEqual(result, [{"id": 1}])
oauth_call, api_call = mock_post.call_args_list
self.assertEqual(oauth_call.args[0], "https://id.twitch.tv/oauth2/token")
self.assertEqual(
oauth_call.kwargs["params"],
{"client_id": "cid", "client_secret": "secret", "grant_type": "client_credentials"},
)
self.assertEqual(api_call.kwargs["headers"]["Authorization"], "Bearer fresh_token")
@patch("fjerkroa_bot.igdblib.requests.post")
def test_refreshes_and_retries_on_401(self, mock_post):
"""A 401 with a configured secret triggers one refresh and retry."""
mock_post.side_effect = [
self._api_response(None, status_code=401),
self._oauth_response(),
self._api_response([{"id": 2}]),
]
igdb = IGDBQuery("cid", "expired_token", client_secret="secret")
result = igdb.send_igdb_request("games", "fields name; limit 1;")
self.assertEqual(result, [{"id": 2}])
self.assertEqual(igdb.igdb_api_key, "fresh_token")
self.assertEqual(mock_post.call_args_list[2].kwargs["headers"]["Authorization"], "Bearer fresh_token")
@patch("fjerkroa_bot.igdblib.time.time")
@patch("fjerkroa_bot.igdblib.requests.post")
def test_proactive_refresh_before_expiry(self, mock_post, mock_time):
"""An expired self-fetched token is refreshed before the request."""
mock_time.return_value = 1_000_000.0
mock_post.side_effect = [self._oauth_response("token_a", expires_in=5_000_000), self._api_response([])]
igdb = IGDBQuery("cid", client_secret="secret")
igdb.send_igdb_request("games", "fields name;")
# jump past the token expiry -> next request refreshes first
mock_time.return_value = 1_000_000.0 + 5_000_000
mock_post.side_effect = [self._oauth_response("token_b"), self._api_response([])]
igdb.send_igdb_request("games", "fields name;")
self.assertEqual(igdb.igdb_api_key, "token_b")
@patch("fjerkroa_bot.igdblib.requests.post")
def test_no_refresh_without_secret(self, mock_post):
"""Static-token setups keep the old behavior: no OAuth calls, error -> None."""
response = Mock()
response.status_code = 401
response.raise_for_status.side_effect = requests.RequestException("401 Client Error")
mock_post.return_value = response
igdb = IGDBQuery("cid", "expired_token")
result = igdb.send_igdb_request("games", "fields name; limit 1;")
self.assertIsNone(result)
mock_post.assert_called_once()
if __name__ == "__main__": if __name__ == "__main__":
unittest.main() unittest.main()
+39
View File
@@ -45,6 +45,45 @@ class TestIngest(unittest.TestCase):
self.assertIsNone(cache.ingest_bytes(PNG, "chat", "alice", "1")) self.assertIsNone(cache.ingest_bytes(PNG, "chat", "alice", "1"))
class TestOversizedDownloadRejected(unittest.IsolatedAsyncioTestCase):
async def test_download_stays_over_limit_and_is_rejected(self):
"""IMG-10: an over-limit download must be rejected, not cached truncated."""
with tempfile.TemporaryDirectory() as tmp:
_, cache = make_cache(tmp, {"image-max-bytes": 32})
class FakeContent:
@staticmethod
async def iter_chunked(size):
yield PNG # 72 bytes > 32
class FakeResp:
content = FakeContent()
async def __aenter__(self):
return self
async def __aexit__(self, *a):
return False
def raise_for_status(self):
pass
class FakeSession:
def get(self, url):
return FakeResp()
async def __aenter__(self):
return self
async def __aexit__(self, *a):
return False
with patch("fjerkroa_bot.images.aiohttp.ClientSession", return_value=FakeSession()):
data = await cache._download("http://x.com/big.png")
self.assertEqual(len(data), 33) # limit + 1, not silently capped to limit
self.assertIsNone(await cache.ingest_url("http://x.com/big.png", "chat", "alice", "1"))
class TestVisionDataUrls(OpsBase): class TestVisionDataUrls(OpsBase):
async def test_attachment_becomes_data_url(self): async def test_attachment_becomes_data_url(self):
"""IMG-11: the model sees a data: URL, never the CDN link.""" """IMG-11: the model sees a data: URL, never the CDN link."""
+75
View File
@@ -0,0 +1,75 @@
"""Unit coverage for SPEC-013 news digest (NEWS-01..03)."""
import unittest
from unittest.mock import AsyncMock
from fjerkroa_bot.news import NewsFetcher, parse_feed, render_digest
RSS = b"""<?xml version="1.0"?><rss><channel>
<item><title>Game X released</title><link>https://ex.com/x</link></item>
<item><title>Patch Y notes</title><link>https://ex.com/y</link></item>
</channel></rss>"""
ATOM = b"""<?xml version="1.0"?><feed xmlns="http://www.w3.org/2005/Atom">
<entry><title>Atom headline</title><link href="https://ex.com/a"/></entry>
</feed>"""
class TestParse(unittest.TestCase):
def test_rss(self):
"""NEWS-01: RSS items parsed with title + link."""
items = parse_feed(RSS, "Src")
self.assertEqual([i["title"] for i in items], ["Game X released", "Patch Y notes"])
self.assertEqual(items[0]["link"], "https://ex.com/x")
self.assertEqual(items[0]["source"], "Src")
def test_atom(self):
"""NEWS-01: Atom entries parsed with href link."""
items = parse_feed(ATOM, "A")
self.assertEqual(items[0]["title"], "Atom headline")
self.assertEqual(items[0]["link"], "https://ex.com/a")
def test_malformed_never_raises(self):
"""NEWS-01: garbage XML returns [] without raising."""
self.assertEqual(parse_feed(b"<not xml", "bad"), [])
self.assertEqual(parse_feed(b"", "empty"), [])
class TestDigest(unittest.TestCase):
def test_sanitized_and_capped(self):
"""NEWS-02: headlines sanitized, item count capped."""
items = [{"title": "@everyone big news \x00", "link": "", "source": "S"} for _ in range(20)]
digest = render_digest(items, max_items=5)
self.assertEqual(digest.count("\n"), 4) # 5 lines
self.assertNotIn("@everyone", digest)
self.assertNotIn("\x00", digest)
class TestCollect(unittest.IsolatedAsyncioTestCase):
async def test_ssrf_skip_and_dedup(self):
"""NEWS-03: guarded feed skipped, dup titles dropped, bad fetch survived."""
def guard(url):
return "refused" if "internal" in url else None
async def fetch(url):
if "boom" in url:
raise ValueError("boom")
return RSS # same content from two feeds -> dedup
fetcher = NewsFetcher(guard, fetch)
feeds = [
("https://a.com/feed", "A"),
("https://internal/feed", "Internal"), # SSRF-skipped
("https://boom.com/feed", "Boom"), # fetch fails
("https://b.com/feed", "B"), # same RSS -> dup titles dropped
]
items = await fetcher.collect(feeds, per_feed=5)
titles = [i["title"] for i in items]
self.assertEqual(titles, ["Game X released", "Patch Y notes"]) # deduped, internal+boom skipped
async def test_per_feed_limit(self):
"""NEWS-03: per-feed cap honored."""
fetcher = NewsFetcher(lambda u: None, AsyncMock(return_value=RSS))
items = await fetcher.collect([("https://a.com", "A")], per_feed=1)
self.assertEqual(len(items), 1)
+40
View File
@@ -91,6 +91,46 @@ class TestTextExtraction(unittest.TestCase):
self.assertNotIn("x{}", text) self.assertNotIn("x{}", text)
class TestBodyReadCollectsAllChunks(unittest.IsolatedAsyncioTestCase):
async def test_get_reads_past_first_chunk(self):
"""URL-05 regression: body arrives in many chunks; all are collected up to the cap."""
reader = URLReader(lambda: {}, None)
chunks = [b"<title>t</title>", b"<p>middle</p>", b"<p>end</p>"]
class FakeContent:
@staticmethod
async def iter_chunked(size):
for chunk in chunks:
yield chunk
class FakeResp:
status = 200
headers = {}
url = "http://safe.example.com"
content = FakeContent()
async def __aenter__(self):
return self
async def __aexit__(self, *a):
return False
def raise_for_status(self):
pass
class FakeSession:
def get(self, url, allow_redirects=False):
return FakeResp()
with patch("fjerkroa_bot.url_reader.guard_url", return_value=None):
_, body = await reader._get(FakeSession(), "http://safe.example.com", 1000)
self.assertEqual(body, b"".join(chunks))
with patch("fjerkroa_bot.url_reader.guard_url", return_value=None):
_, body = await reader._get(FakeSession(), "http://safe.example.com", 20)
self.assertEqual(body, b"".join(chunks)[:20])
class TestFetchSanitizes(unittest.IsolatedAsyncioTestCase): class TestFetchSanitizes(unittest.IsolatedAsyncioTestCase):
async def test_fetch_result_is_sanitized_and_capped(self): async def test_fetch_result_is_sanitized_and_capped(self):
"""URL-05: fetch output is length-capped and @everyone-neutralized.""" """URL-05: fetch output is length-capped and @everyone-neutralized."""
Generated
+2
View File
@@ -627,6 +627,7 @@ version = "3.0.0"
source = { editable = "." } source = { editable = "." }
dependencies = [ dependencies = [
{ name = "aiohttp" }, { name = "aiohttp" },
{ name = "defusedxml" },
{ name = "discord-py" }, { name = "discord-py" },
{ name = "openai" }, { name = "openai" },
{ name = "requests" }, { name = "requests" },
@@ -656,6 +657,7 @@ dev = [
[package.metadata] [package.metadata]
requires-dist = [ requires-dist = [
{ name = "aiohttp", specifier = ">=3.12" }, { name = "aiohttp", specifier = ">=3.12" },
{ name = "defusedxml", specifier = ">=0.7" },
{ name = "discord-py", specifier = ">=2.5,<3" }, { name = "discord-py", specifier = ">=2.5,<3" },
{ name = "openai", specifier = ">=2.45" }, { name = "openai", specifier = ">=2.45" },
{ name = "requests", specifier = ">=2.32" }, { name = "requests", specifier = ">=2.32" },