Compare commits

..

6 Commits

Author SHA1 Message Date
Oleksandr Kozachuk 891fbdc101 operator help: complete + context-aware (ops-17): !bot help full grouped list, !help for everyone per-channel 2026-07-14 11:09:33 +02:00
Oleksandr Kozachuk 09871b9b95 codex mechanicus search (spec-014): ground warhammer lore in binaric.tech via codex_search tool 2026-07-13 20:50:08 +02:00
Oleksandr Kozachuk a514ff652c url reader: follow html meta-refresh redirects (getnews shortlinks) with ssrf re-guard 2026-07-13 20:09:46 +02:00
Oleksandr Kozachuk 7628faf551 news webhook posting mode: replaces py3.8 getnews (bounded state, seed-on-first-run, ssrf-guarded) 2026-07-13 19:47:30 +02:00
Oleksandr Kozachuk 2caa18a17f igdb search + capped http reads: native fulltext, full-page fetch
search_games used `where name ~ "<q>"*`: prefix-only, diacritic- and
word-order-sensitive -- "MARVEL Tōkon" found nothing though IGDB has
it. Switch to IGDB's native `search` clause (relevance-ranked,
diacritic-insensitive).

fetch_url and image downloads read bodies with content.read(n), which
returns only the first buffered chunk (~7 KB): pages collapsed to
their <title>. httpread.read_capped collects chunks up to the byte
cap; image downloads read limit+1 so over-limit files are still
rejected instead of cached truncated (IMG-10).

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
2026-07-13 19:29:16 +02:00
Oleksandr Kozachuk d4eec4088d news digest (spec-013): rss/atom -> {news} file via cron
Replaces the broken pre-1.0-openai news_feed.py. Stdlib parsing with
defusedxml (feeds are untrusted XML), titles sanitized (SAF-03), feed
URLs SSRF-guarded. CLI: python -m fjerkroa_bot.news --config <cfg>.
2026-07-13 19:28:41 +02:00
23 changed files with 1256 additions and 23 deletions
+10
View File
@@ -60,6 +60,16 @@ Decisions inside the set architecture. D-NNN, never renumbered.
broken classifier must never mute the bot; the budget gate already
bounds spend. Its verdict gates BEFORE the main call, the
envelope's answer_needed still gates after — two independent nets.
- **D-018** — Codex Mechanicus search (FDB-019, SPEC-014): Luma's
lore is grounded in the priest's real archive at binaric.tech via a
`codex_search` tool over the site's public `search-index.json`, not
a bot-side copy — the index stays a single source of truth, refreshed
by the site's own publish rite, and the bot caches it in memory
(TTL). It reuses SPEC-011's `guard_url` + `read_capped` (fetch is
SSRF-guarded and byte-bounded) and sanitizes every returned field:
one's own web content is still untrusted by the time it reaches a
prompt. Luma-only (`enable-codex`, off elsewhere) — the Adeptus
Mechanicus archive has no place in Fjærkroa's café persona.
- **D-017** — All human-behavior knobs default to off/v3.0.0
semantics; behavior changes are config rollouts per deployment, not
code flips. The classifier's `factual` flag is the only coupling
+12
View File
@@ -80,3 +80,15 @@ enable-game-info = true
# Ops (SPEC-012): consecutive OpenAI failures before a staff alert
# api-error-alert-threshold = 5
# Backups: cron runs deploy/backup_db.py daily -> ~/backups/<bot>/ (keep 14)
# News digest (SPEC-013) — `python -m fjerkroa_bot.news --config X.toml` via cron;
# writes the {news} file. Feeds are [url, label] pairs (RSS or Atom):
# news = "news_feed.txt"
# news-per-feed = 3
# news-max-items = 15
# news-feeds = [
# ["https://blog.playstation.com/feed/", "PS"],
# ["https://kotaku.com/rss", "Kotaku"],
# ["https://www.pushsquare.com/feeds/latest", "Push"],
# ["https://mein-mmo.de/feed/", "MeinMMO"],
# ]
+147
View File
@@ -0,0 +1,147 @@
"""Codex Mechanicus search tool (SPEC-014, FDB-019).
Luma's own sacred archive — the Codex Mechanicus at binaric.tech — as a
function tool. She searches the codex index and answers Cult Mechanicus
lore from real, sourced inscriptions instead of inventing it. The index
is fetched over HTTPS (SSRF-guarded, size-bounded, cached in memory) and
every field returned to the model is sanitized (SAF-03), because even
one's own web content is still untrusted input by the time it reaches a
prompt.
The model calls `codex_search`; production wires the live index URL.
"""
import json
import logging
import time
from typing import Any, Callable, Dict, List, Optional
from urllib.parse import urljoin
import aiohttp
from .ai_responder import sanitize_external_text
from .httpread import read_capped
from .url_reader import guard_url
DEFAULT_INDEX_URL = "https://binaric.tech/search-index.json"
DEFAULT_MAX_BYTES = 4 * 1024 * 1024
DEFAULT_LIMIT = 5
DEFAULT_TTL_S = 3600
DEFAULT_SUMMARY_CHARS = 500
FETCH_TIMEOUT_S = 15
_VALID_LANGS = ("en", "de", "eo", "no", "uk")
CODEX_SEARCH_TOOL = {
"name": "codex_search",
"description": "Search Luma's own Codex Mechanicus (the sacred archive at binaric.tech) for Adeptus "
"Mechanicus lore: doctrines, forges, orders, rites, relics, weapons, entities, the lexicon, and the "
"priest's own adoptus. Returns matching inscriptions with a short summary and the URL to read the full "
"text. Use for any Cult Mechanicus / Warhammer 40k Mechanicus question so the answer is grounded in the "
"codex, not invented.",
"parameters": {
"type": "object",
"properties": {
"query": {"type": "string", "description": "What to look for: a name, concept, rite, or phrase."},
"lang": {"type": "string", "description": "Language of the inscriptions to prefer: en, de, eo, no, uk. Default en."},
},
"required": ["query"],
},
}
_STOP = {"the", "a", "an", "of", "and", "or", "to", "in", "is", "der", "die", "das", "und", "von", "en", "et"}
def _tokenize(text: str) -> List[str]:
cleaned = "".join(c.lower() if c.isalnum() else " " for c in text)
return [t for t in cleaned.split() if len(t) > 1 and t not in _STOP]
def _score(item: Dict[str, Any], terms: List[str]) -> int:
"""Weight a hit by field: title beats summary beats body (CDX-03)."""
title = str(item.get("title") or "").lower()
summary = str(item.get("summary") or "").lower()
body = str(item.get("body") or "").lower()
score = 0
for term in terms:
score += 8 if term in title else 0
score += 3 if term in summary else 0
score += 1 if term in body else 0
return score
def _rank(items: List[Dict[str, Any]], terms: List[str], lang: str) -> List[Dict[str, Any]]:
"""Score items in the given language; fall back to all languages if empty (CDX-04)."""
def scored(only_lang: Optional[str]) -> List[Any]:
out = []
for item in items:
if only_lang and f"/{only_lang}/" not in str(item.get("url") or ""):
continue
hit = _score(item, terms)
if hit > 0:
out.append((hit, item))
out.sort(key=lambda pair: pair[0], reverse=True)
return out
ranked = scored(lang) or scored(None)
return [item for _, item in ranked]
class CodexSearch:
def __init__(self, config_getter: Callable[[], Dict[str, Any]]) -> None:
self._config = config_getter
self._cache: Optional[List[Dict[str, Any]]] = None
self._fetched_at = 0.0
def enabled(self) -> bool:
return bool(self._config().get("enable-codex", False))
def _index_url(self) -> str:
return str(self._config().get("codex-index-url", DEFAULT_INDEX_URL))
async def _load_index(self) -> List[Dict[str, Any]]:
"""Fetch + cache the codex index, SSRF-guarded and size-bounded (CDX-02)."""
ttl = float(self._config().get("codex-cache-ttl", DEFAULT_TTL_S))
if self._cache is not None and (time.monotonic() - self._fetched_at) < ttl:
return self._cache
url = self._index_url()
reason = guard_url(url)
if reason:
raise ValueError(reason)
max_bytes = int(self._config().get("codex-max-bytes", DEFAULT_MAX_BYTES))
timeout = aiohttp.ClientTimeout(total=FETCH_TIMEOUT_S)
async with aiohttp.ClientSession(timeout=timeout, headers={"User-Agent": "FjerkroaBot-codex/1.0"}) as session:
async with session.get(url) as response:
response.raise_for_status()
raw = await read_capped(response, max_bytes)
data = json.loads(raw.decode("utf-8", "ignore"))
items = data.get("items", []) if isinstance(data, dict) else []
self._cache = [i for i in items if isinstance(i, dict)]
self._fetched_at = time.monotonic()
return self._cache
async def search(self, query: str, lang: str = "en", limit: int = DEFAULT_LIMIT) -> Dict[str, Any]:
"""Return sanitized top matches, or an error dict — never raise (CDX-05)."""
try:
items = await self._load_index()
except Exception as err:
logging.warning(f"codex: index load failed: {err!r}")
return {"error": f"codex unavailable: {err}"}
terms = _tokenize(query)
if not terms:
return {"query": query, "results": []}
pick = (lang or "en").lower()
if pick not in _VALID_LANGS:
pick = "en"
summary_chars = int(self._config().get("codex-summary-chars", DEFAULT_SUMMARY_CHARS))
results = []
for item in _rank(items, terms, pick)[: max(1, limit)]:
results.append(
{
"title": sanitize_external_text(str(item.get("title") or ""), 200),
"summary": sanitize_external_text(str(item.get("summary") or ""), summary_chars),
"collection": str(item.get("collection") or ""),
"url": urljoin(self._index_url(), str(item.get("url") or "")),
}
)
return {"query": query, "lang": pick, "results": results}
+25 -2
View File
@@ -182,6 +182,9 @@ class FjerkroaBot(commands.Bot):
if content.startswith("!privacy"):
await message.channel.send(self.config.get("privacy-notice", DEFAULT_PRIVACY_NOTICE), suppress_embeds=True)
return
if content.startswith("!help"): # OPS-17: context-aware, works even while paused
await message.channel.send(self._help_text(staff=self.is_staff_channel(message.channel)), suppress_embeds=True)
return
if not self.replies_allowed():
return
if str(message.content).startswith("!wichtel"):
@@ -263,15 +266,35 @@ class FjerkroaBot(commands.Bot):
return f"Cancelled {store.task_set_state(int(args[1]), 'cancelled')} task(s)."
return None
def _help_text(self, staff: bool) -> str:
"""Context-aware command help (OPS-17): every channel lists the user commands; the staff channel also lists operator commands."""
everywhere = (
"Available to everyone, in any channel:\n"
"• `!help` — this help\n"
"• `!forgetme` — delete your messages and memory traces (works even while I'm paused)\n"
"• `!privacy` — how your data is handled (works even while I'm paused)\n"
"• `!wichtel @a @b @c …` — draw Secret Santa pairings (needs ≥2 mentions; only while I'm active)"
)
if not staff:
return everywhere
operator = (
"Staff commands — this channel only, prefixed `!bot`:\n"
"• Control: `pause`, `resume`, `quiet <minutes>`, `status`\n"
"• Cost: `spend`, `images on|off`\n"
"• Memory: `memory <user>`, `forget-fact <id>`, `pin <channel|global> <text>`, `unpin <id>`, `pins`\n"
"• Tasks: `tasks` (list), `tasks on|off`, `task-approve <id>`, `task-cancel <id>`"
)
return operator + "\n\n" + everywhere
async def handle_staff_command(self, message: Message) -> None:
"""Operator kill-switches, staff channel only (OPS-01..05, OPS-09, MEM-07)."""
"""Operator kill-switches, staff channel only (OPS-01..05, OPS-09, OPS-17, MEM-07)."""
args = str(message.content).split()[1:]
for handler in (self._memory_command, self._task_command):
reply = handler(args)
if reply is not None:
await message.channel.send(reply, suppress_embeds=True)
return
reply = "Commands: pause, resume, images on|off, tasks on|off, quiet <minutes>, status, spend, memory <user>, forget-fact <id>, pin <channel|global> <fact>, unpin <id>"
reply = self._help_text(staff=True) # OPS-17: unknown/`help` -> full grouped help
if args[:1] == ["pause"]:
self.replies_enabled = False
reply = "Replies paused."
+18
View File
@@ -0,0 +1,18 @@
"""Bounded HTTP body read (leaf module, no intra-package imports).
`response.content.read(n)` returns whatever is buffered, not n bytes,
so it silently truncates large or chunked bodies (and web feeds/pages
parse to garbage). This accumulates decompressed chunks up to a hard
cap instead.
"""
CHUNK = 65536
async def read_capped(response, max_bytes: int) -> bytes:
buf = bytearray()
async for chunk in response.content.iter_chunked(CHUNK):
buf.extend(chunk)
if len(buf) > max_bytes:
break
return bytes(buf[:max_bytes])
+12 -6
View File
@@ -56,8 +56,12 @@ class IGDBQuery(object):
return requests.post(igdb_url, headers=headers, data=query_body)
@staticmethod
def build_query(fields, filters=None, limit=10, offset=None):
query = f"fields {','.join(fields) if fields is not None and len(fields) > 0 else '*'}; limit {limit};"
def build_query(fields, filters=None, limit=10, offset=None, search_term=None):
query = ""
if search_term:
escaped = search_term.replace("\\", "\\\\").replace('"', '\\"')
query += f'search "{escaped}"; '
query += f"fields {','.join(fields) if fields is not None and len(fields) > 0 else '*'}; limit {limit};"
if offset is not None:
query += f" offset {offset};"
if filters:
@@ -65,12 +69,12 @@ class IGDBQuery(object):
query += " where " + " & ".join(filter_statements) + ";"
return query
def generalized_igdb_query(self, params, endpoint, fields, additional_filters=None, limit=10, offset=None):
def generalized_igdb_query(self, params, endpoint, fields, additional_filters=None, limit=10, offset=None, search_term=None):
all_filters = {key: f'~ "{value}"*' for key, value in params.items() if value}
if additional_filters:
all_filters.update(additional_filters)
query = self.build_query(fields, all_filters, limit, offset)
query = self.build_query(fields, all_filters, limit, offset, search_term)
data = self.send_igdb_request(endpoint, query)
print(f"{endpoint}: {query} -> {data}")
return data
@@ -134,9 +138,10 @@ class IGDBQuery(object):
return None
try:
# Search for games with fuzzy matching
# IGDB native full-text search: diacritic- and word-order-insensitive,
# unlike a `name ~ "..."*` prefix filter
games = self.generalized_igdb_query(
{"name": query.strip()},
{},
"games",
[
"id",
@@ -155,6 +160,7 @@ class IGDBQuery(object):
],
additional_filters={"game_type": "= 0"}, # Main games only (IGDB renamed category -> game_type)
limit=limit,
search_term=query.strip(),
)
if not games:
+4 -1
View File
@@ -14,6 +14,7 @@ from typing import Any, Callable, Dict, List, Optional
import aiohttp
from .httpread import read_capped
from .persistence import PersistentStore
DEFAULT_CACHE_MB = 500
@@ -79,7 +80,9 @@ class ImageCache:
async with aiohttp.ClientSession(timeout=timeout) as session:
async with session.get(url) as response:
response.raise_for_status()
return await response.content.read(limit + 1)
# limit + 1: an over-limit body must stay over-limit so
# ingest_bytes rejects it instead of caching it truncated
return await read_capped(response, limit + 1)
def data_url(self, sha256: str, ext: str) -> Optional[str]:
path = self._path(sha256, ext)
+291
View File
@@ -0,0 +1,291 @@
"""News digest fetcher (SPEC-013, FDB-012 news rewrite).
Replaces the broken pre-1.0-openai `news_feed.py`. Fetches configured
RSS/Atom feeds (stdlib, no feedparser dep), builds a compact sanitized
headline digest, and writes it to the `{news}` file the responder
injects (AIResponder.message). Feeds are external input: titles are
sanitized (SAF-03) and each feed URL is SSRF-guarded before fetching.
CLI: python -m fjerkroa_bot.news --config kroa.toml
"""
import argparse
import logging
import sys
import time
from typing import Any, Dict, List, Optional, Tuple
import defusedxml.ElementTree as ElementTree # hardened XML: feeds are untrusted (XXE/billion-laughs)
from .ai_responder import sanitize_external_text
DEFAULT_PER_FEED = 3
DEFAULT_MAX_ITEMS = 15
FETCH_TIMEOUT_S = 15
_ATOM = "{http://www.w3.org/2005/Atom}"
def parse_feed(data: bytes, source: str = "") -> List[Dict[str, str]]:
"""Parse RSS or Atom bytes into [{title, link, source}] (tolerant)."""
try:
root = ElementTree.fromstring(data)
except Exception as err:
# malformed XML or a blocked entity/DTD attack — tolerate, never raise (NEWS-01)
logging.warning(f"news: unparseable/unsafe feed {source!r}: {err!r}")
return []
items: List[Dict[str, str]] = []
# RSS: <rss><channel><item><title/><link/>
for item in root.iter("item"):
title = (item.findtext("title") or "").strip()
link = (item.findtext("link") or "").strip()
if title:
items.append({"title": title, "link": link, "source": source})
# Atom: <feed><entry><title/><link href=/>
for entry in root.iter(f"{_ATOM}entry"):
title = (entry.findtext(f"{_ATOM}title") or "").strip()
link_el = entry.find(f"{_ATOM}link")
link = link_el.get("href", "") if link_el is not None else ""
if title:
items.append({"title": title, "link": link, "source": source})
return items
def render_digest(items: List[Dict[str, str]], max_items: int = DEFAULT_MAX_ITEMS) -> str:
"""Compact sanitized digest for the {news} prompt slot."""
lines = []
for item in items[:max_items]:
title = sanitize_external_text(item["title"], 200)
source = item.get("source", "")
link = item.get("link", "")
prefix = f"[{source}] " if source else ""
lines.append(f"- {prefix}{title}" + (f" ({link})" if link else ""))
return "\n".join(lines)
class NewsFetcher:
def __init__(self, guard, fetch_bytes) -> None:
# injected so tests need no network; production wires aiohttp + guard_url
self._guard = guard
self._fetch_bytes = fetch_bytes
async def collect(self, feeds: List[Tuple[str, str]], per_feed: int) -> List[Dict[str, str]]:
"""feeds = [(url, label)]; returns deduped items, order preserved."""
seen = set()
out: List[Dict[str, str]] = []
for url, label in feeds:
reason = self._guard(url)
if reason:
logging.warning(f"news: skipping feed {label}{reason}")
continue
try:
data = await self._fetch_bytes(url)
except Exception as err:
logging.warning(f"news: fetch failed for {label}: {repr(err)}")
continue
for item in parse_feed(data, label)[:per_feed]:
key = item["title"]
if key not in seen:
seen.add(key)
out.append(item)
return out
DEFAULT_SEEN_CAP = 5000
DEFAULT_POST_PER_FEED = 5
DEFAULT_POST_MAX_PER_RUN = 8
def item_key(item: Dict[str, str]) -> str:
return item.get("link") or item.get("title") or ""
class NewsPoster:
"""Post NEW feed items to Discord channel webhooks (ggg model, SPEC-013 NEWS-04..06)."""
def __init__(self, guard, fetch_bytes, post_webhook) -> None:
self._guard = guard
self._fetch_bytes = fetch_bytes
self._post_webhook = post_webhook
async def run_post(
self,
feeds: List[Tuple[str, str, str]],
webhooks: Dict[str, str],
seen: set,
per_feed: int,
max_per_run: int,
seed_only: bool,
) -> Tuple[int, set]:
"""Returns (posted_count, updated_seen). seed_only marks new items seen without posting."""
posted = 0
for url, label, channel in feeds:
reason = self._guard(url)
if reason:
logging.warning(f"news-post: skipping feed {label}{reason}")
continue
try:
data = await self._fetch_bytes(url)
except Exception as err:
logging.warning(f"news-post: fetch failed for {label}: {repr(err)}")
continue
for item in parse_feed(data, label)[:per_feed]:
key = item_key(item)
if not key or key in seen:
continue
seen.add(key)
may_post = not seed_only and posted < max_per_run
if may_post and await self._deliver(item, label, channel, webhooks):
posted += 1
return posted, seen
async def _deliver(self, item: Dict[str, str], label: str, channel: str, webhooks: Dict[str, str]) -> bool:
hook = webhooks.get(channel)
if not hook:
logging.warning(f"news-post: no webhook for channel {channel!r} ({label})")
return False
title = sanitize_external_text(item["title"], 300)
link = item.get("link", "")
content = f"**[{label}]** {title}" + (f"\n{link}" if link else "")
try:
await self._post_webhook(hook, content)
return True
except Exception as err:
logging.warning(f"news-post: webhook post failed ({label}): {repr(err)}")
return False
def load_seen(path: str) -> Tuple[set, bool]:
"""(seen-set, existed). Missing/broken state -> empty set, existed=False (seed run)."""
import json
import os
if not os.path.exists(path):
return set(), False
try:
with open(path, encoding="utf-8") as fd:
return set(json.load(fd)), True
except Exception as err:
logging.warning(f"news-post: unreadable state {path}: {err!r} — reseeding")
return set(), False
def save_seen(path: str, seen: set, cap: int = DEFAULT_SEEN_CAP) -> None:
import json
# keep the newest `cap` keys (insertion order preserved by Python sets? no — use a bounded slice)
keys = list(seen)[-cap:]
with open(path, "w", encoding="utf-8") as fd:
json.dump(keys, fd)
def _post_feeds_from_config(config: Dict[str, Any]) -> List[Tuple[str, str, str]]:
feeds = []
for entry in config.get("news-post-feeds", []):
if isinstance(entry, (list, tuple)) and len(entry) >= 3:
feeds.append((str(entry[0]), str(entry[1]), str(entry[2])))
return feeds
def _feeds_from_config(config: Dict[str, Any]) -> List[Tuple[str, str]]:
"""news-feeds = [["url", "label"], ...] or ["url", ...]."""
feeds = []
for entry in config.get("news-feeds", []):
if isinstance(entry, (list, tuple)):
feeds.append((str(entry[0]), str(entry[1]) if len(entry) > 1 else ""))
else:
feeds.append((str(entry), ""))
return feeds
async def _aiohttp_fetch(url: str) -> bytes:
import aiohttp
from .httpread import read_capped
timeout = aiohttp.ClientTimeout(total=FETCH_TIMEOUT_S)
async with aiohttp.ClientSession(timeout=timeout, headers={"User-Agent": "Mozilla/5.0 (compatible; FjerkroaBot-news/1.0)"}) as session:
async with session.get(url) as response:
response.raise_for_status()
return await read_capped(response, 4 * 1024 * 1024)
async def _aiohttp_post(hook: str, content: str) -> None:
import aiohttp
timeout = aiohttp.ClientTimeout(total=FETCH_TIMEOUT_S)
async with aiohttp.ClientSession(timeout=timeout) as session:
# allowed_mentions none: a headline can never ping the channel (SAF-02 spirit)
payload = {"content": content[:2000], "allowed_mentions": {"parse": []}}
async with session.post(hook, json=payload) as response:
response.raise_for_status()
async def run_post(config: Dict[str, Any]) -> int:
"""Webhook-posting mode (ggg): post new items to channels. Returns posted count."""
from .url_reader import guard_url
webhooks = dict(config.get("news-post-webhooks", {}))
feeds = _post_feeds_from_config(config)
state_path = config.get("news-post-state", "news_state.json")
if not webhooks or not feeds:
logging.error("news-post: need news-post-webhooks and news-post-feeds")
return 0
seen, existed = load_seen(state_path)
poster = NewsPoster(guard_url, _aiohttp_fetch, _aiohttp_post)
posted, seen = await poster.run_post(
feeds,
webhooks,
seen,
int(config.get("news-post-per-feed", DEFAULT_POST_PER_FEED)),
int(config.get("news-post-max-per-run", DEFAULT_POST_MAX_PER_RUN)),
seed_only=not existed, # first run seeds without flooding the channels
)
save_seen(state_path, seen, int(config.get("news-post-seen-cap", DEFAULT_SEEN_CAP)))
logging.info(f"news-post: posted {posted} item(s)" + (" (seed run — nothing posted)" if not existed else ""))
return posted
async def run(config: Dict[str, Any]) -> Optional[str]:
from .url_reader import guard_url
out_path = config.get("news")
if not out_path:
logging.error("news: no `news` output path in config")
return None
feeds = _feeds_from_config(config)
if not feeds:
logging.error("news: no `news-feeds` configured")
return None
fetcher = NewsFetcher(guard_url, _aiohttp_fetch)
items = await fetcher.collect(feeds, int(config.get("news-per-feed", DEFAULT_PER_FEED)))
digest = render_digest(items, int(config.get("news-max-items", DEFAULT_MAX_ITEMS)))
header = f"News as of {time.strftime('%Y-%m-%d %H:%M UTC', time.gmtime())}:\n"
with open(out_path, "w", encoding="utf-8") as fd:
fd.write(header + digest + "\n")
logging.info(f"news: wrote {len(items)} items to {out_path}")
return out_path
def main() -> int:
import asyncio
import tomlkit
logging.basicConfig(level=logging.INFO, format="%(asctime)s [%(levelname)s] %(message)s")
parser = argparse.ArgumentParser(
description="Fetch RSS/Atom feeds: --post to channel webhooks (ggg) or default {news} digest file (kroa)"
)
parser.add_argument("--config", required=True)
parser.add_argument("--post", action="store_true", help="webhook-posting mode (post new items to Discord channels)")
args = parser.parse_args()
with open(args.config, encoding="utf-8") as fd:
config = tomlkit.load(fd)
if args.post:
asyncio.run(run_post(config))
return 0
result = asyncio.run(run(config))
return 0 if result else 1
if __name__ == "__main__":
sys.exit(main())
+15 -1
View File
@@ -9,6 +9,9 @@ from typing import Any, Dict, List, Optional, Tuple
import openai
from .ai_responder import AIResponder, exponential_backoff, sanitize_external_text
from .codex import CODEX_SEARCH_TOOL
from .codex import DEFAULT_LIMIT as CODEX_DEFAULT_LIMIT
from .codex import CodexSearch
from .igdblib import IGDBQuery
from .leonardo_draw import LeonardoAIDrawMixIn
from .quota import QuotaLedger
@@ -160,6 +163,8 @@ class OpenAIResponder(AIResponder, LeonardoAIDrawMixIn):
# URL reading tool (SPEC-011); shares the image cache for page images
self.url_reader = URLReader(lambda: self.config, self.image_cache)
# Codex Mechanicus search (SPEC-014); Luma's own archive at binaric.tech
self.codex = CodexSearch(lambda: self.config)
def _available_tools(self) -> List[Dict[str, Any]]:
"""Assemble the function-tool list from every enabled provider (URL-01)."""
@@ -173,16 +178,25 @@ class OpenAIResponder(AIResponder, LeonardoAIDrawMixIn):
logging.warning(f"Error setting up IGDB functions: {err}")
if self.url_reader.enabled():
functions.append(FETCH_URL_TOOL)
if self.codex.enabled(): # CDX-01
functions.append(CODEX_SEARCH_TOOL)
return functions
async def _dispatch_tool(self, name: str, args: Dict[str, Any], author: str) -> Any:
"""Route a tool call to its provider (IGDB or URL reader)."""
"""Route a tool call to its provider (IGDB, URL reader, or codex)."""
if name == "fetch_url":
per_user_cap = int(self.config.get("url-daily-per-user", 20))
if self.ledger._get(f"url-fetch:{author}") >= per_user_cap: # URL-07
return {"error": "daily URL fetch limit reached"}
self.ledger._add(f"url-fetch:{author}", 1)
return await self.url_reader.fetch(str(args.get("url", "")), self.channel, author or "user")
if name == "codex_search":
per_user_cap = int(self.config.get("codex-daily-per-user", 50))
if self.ledger._get(f"codex:{author}") >= per_user_cap: # CDX-06
return {"error": "daily codex search limit reached"}
self.ledger._add(f"codex:{author}", 1)
limit = int(self.config.get("codex-limit", CODEX_DEFAULT_LIMIT))
return await self.codex.search(str(args.get("query", "")), str(args.get("lang", "en")), limit)
return await self._execute_igdb_function(name, args)
async def draw_openai(self, description: str, count: int = 1) -> List[BytesIO]:
+32 -12
View File
@@ -18,6 +18,7 @@ from urllib.parse import urljoin, urlparse
import aiohttp
from .ai_responder import sanitize_external_text
from .httpread import read_capped
DEFAULT_MAX_BYTES = 2 * 1024 * 1024
DEFAULT_MAX_CHARS = 6000
@@ -37,6 +38,9 @@ FETCH_URL_TOOL = {
}
_META_REFRESH_URL = re.compile(r"url\s*=\s*['\"]?([^'\";\s]+)", re.I)
class _Extractor(HTMLParser):
def __init__(self) -> None:
super().__init__()
@@ -44,6 +48,7 @@ class _Extractor(HTMLParser):
self.parts: List[str] = []
self.images: List[str] = []
self.og_image: Optional[str] = None
self.refresh_url: Optional[str] = None
def handle_starttag(self, tag: str, attrs) -> None:
if tag in ("script", "style", "noscript", "svg"):
@@ -54,6 +59,12 @@ class _Extractor(HTMLParser):
self.images.append(src)
if tag == "meta" and attr.get("property") == "og:image" and attr.get("content"):
self.og_image = attr["content"]
# meta-refresh redirect (link shorteners, getnews stubs) — URL-04
content = attr.get("content")
if tag == "meta" and (attr.get("http-equiv") or "").lower() == "refresh" and content:
match = _META_REFRESH_URL.search(content)
if match and self.refresh_url is None:
self.refresh_url = match.group(1)
def handle_endtag(self, tag: str) -> None:
if tag in ("script", "style", "noscript", "svg") and self._skip > 0:
@@ -115,7 +126,7 @@ class URLReader:
current = urljoin(current, response.headers["Location"])
continue
response.raise_for_status()
return str(response.url), await response.content.read(max_bytes + 1)
return str(response.url), await read_capped(response, max_bytes)
raise ValueError("too many redirects")
async def fetch(self, url: str, channel: str, user: str) -> Dict[str, Any]:
@@ -125,29 +136,38 @@ class URLReader:
try:
async with aiohttp.ClientSession(timeout=timeout, headers={"User-Agent": "FjerkroaBot/1.0"}) as session:
final_url, body = await self._get(session, url, max_bytes)
# follow a meta-refresh redirect (link shorteners / getnews stubs), re-guarded — URL-04
for _ in range(2):
extractor = self._extract(body.decode("utf-8", "ignore"))
if not extractor.refresh_url:
break
target = urljoin(final_url, extractor.refresh_url)
if guard_url(target) is not None or target == final_url:
break
logging.info(f"url reader: following meta-refresh -> {target}")
final_url, body = await self._get(session, target, max_bytes)
except Exception as err:
return {"error": str(err)}
text = self._to_text(body.decode("utf-8", "ignore"))
clean = sanitize_external_text(text, int(config.get("url-max-chars", DEFAULT_MAX_CHARS)))
images = await self._ingest_images(body.decode("utf-8", "ignore"), final_url, channel, user)
html = body.decode("utf-8", "ignore")
clean = sanitize_external_text(self._to_text(html), int(config.get("url-max-chars", DEFAULT_MAX_CHARS)))
images = await self._ingest_images(html, final_url, channel, user)
return {"url": final_url, "text": clean, "images_cached": images}
def _to_text(self, html: str) -> str:
def _extract(self, html: str) -> "_Extractor":
extractor = _Extractor()
try:
extractor.feed(html)
except Exception as err:
logging.debug(f"html parse (text) failed: {err!r}")
return re.sub(r"\s+\n", "\n", " ".join(extractor.parts))
logging.debug(f"html parse failed: {err!r}")
return extractor
def _to_text(self, html: str) -> str:
return re.sub(r"\s+\n", "\n", " ".join(self._extract(html).parts))
async def _ingest_images(self, html: str, base_url: str, channel: str, user: str) -> int:
if self.image_cache is None:
return 0
extractor = _Extractor()
try:
extractor.feed(html)
except Exception as err:
logging.debug(f"html parse (images) failed: {err!r}")
extractor = self._extract(html)
candidates = ([extractor.og_image] if extractor.og_image else []) + extractor.images
limit = int(self._config().get("url-max-images", DEFAULT_MAX_IMAGES))
cached = 0
+1
View File
@@ -13,3 +13,4 @@ with date + result.
| DEP-05 | 2026-07-13 | Live-verified: kroa deploy attempt ~15h Oslo refused without DEPLOY_FORCE=1. |
| DEP-06 | 2026-07-13 | Rollback documented (older tag + db backup restore); live drill pending — next release. |
| OPS-15 | 2026-07-13 | Backup cron installed on both hosts (daily 03:17 UTC → ~/backups/<bot>/, keep 14); first snapshots written + verified 0600 (kroa 10965 B, luma 25871 B). |
| CDX-07 | 2026-07-13 | Pending live verify on ggg after v3.8.0 deploy: persona grounding + codex_search returns binaric.tech inscriptions with links. |
+1
View File
@@ -15,6 +15,7 @@ dependencies = [
"tomlkit>=0.13",
"watchdog>=6",
"requests>=2.32",
"defusedxml>=0.7",
]
[project.scripts]
+12
View File
@@ -73,3 +73,15 @@ per-channel) — without it, `!bot unpin <id>` required guessing ids.
`!bot spend` answers in the staff channel with today's estimated
spend in USD, token and image counts, and the configured budget.
Management sees the cost, not just the cap.
### OPS-17 — Help is complete and context-aware (coverage: test)
Help reflects where each command actually works, because not every
command is allowed everywhere. `!help` answers in any channel and
lists only the commands usable there: in a normal channel the
everyone-commands (`!help`, `!forgetme`, `!privacy`, `!wichtel`); in
the staff channel it additionally lists the operator commands grouped
by purpose (control, cost, memory, tasks). `!bot help` — and any
unrecognised `!bot` command — answers with that same full staff help,
so the listing is exhaustive rather than the old hand-maintained
partial line. Help works even while the bot is paused.
+4 -1
View File
@@ -31,7 +31,10 @@ refused without DNS.
Redirects are followed manually; each hop's target passes URL-02 and
URL-03 again. A public URL that 302-redirects to `localhost` or an
internal IP is refused at the redirect, not fetched.
internal IP is refused at the redirect, not fetched. **HTML
meta-refresh** redirects (link shorteners, the old getnews stubs) are
also followed — the target is SSRF-re-guarded and fetched, so the
reader returns the real article, not the "Redirecting…" stub.
### URL-05 — Fetched text is bounded and sanitized (coverage: test)
+54
View File
@@ -0,0 +1,54 @@
# SPEC-013 — News digest
Replaces the broken pre-1.0-openai `news_feed.py`. A CLI
(`python -m fjerkroa_bot.news --config <cfg>`) fetches the
`news-feeds` and writes a compact digest to the `news` file that
`AIResponder.message` injects into the `{news}` slot. Feeds are
external input and operator-configured.
### NEWS-01 — RSS and Atom parse to items (coverage: test)
`parse_feed(bytes, label)` extracts `{title, link, source}` from both
RSS (`<item>`) and Atom (`<entry>`) documents, tolerates malformed
XML (returns an empty list, logs), and never raises.
### NEWS-02 — Digest is sanitized and bounded (coverage: test)
`render_digest` caps at `news-max-items`, and every headline passes
`sanitize_external_text` (SAF-03) — a feed cannot inject `@everyone`
or control characters into the prompt via a headline.
### NEWS-03 — Feeds are SSRF-guarded and deduped (coverage: test)
`NewsFetcher.collect` skips any feed URL the SSRF guard rejects,
skips feeds that fail to fetch (one bad feed never sinks the run),
and drops duplicate headlines across feeds.
## Webhook posting (ggg model)
`--post` mode fetches feeds mapped to channels and posts NEW items to
the channel's Discord webhook — replacing the py3.8 `getnews.py`
(dead play3 feed, 35 MB substring-scan state file, HTML-redirect
cruft). Config: `news-post-feeds = [[url, label, channel], …]`,
`news-post-webhooks = {channel = url}`, `news-post-state`.
### NEWS-04 — Only unseen items post, then are marked seen (coverage: test)
`NewsPoster.run_post` posts each item whose key (link, else title) is
not in the seen-set, adds it to the set, and posts to the mapped
channel's webhook. Re-runs over the same feed post nothing new.
### NEWS-05 — First run seeds without flooding (coverage: test)
With no prior state file (`seed_only`), every current item is marked
seen but nothing is posted — migrating off getnews.py never dumps a
backlog into the channels. `news-post-max-per-run` caps steady-state
posts per run.
### NEWS-06 — Post failures and bad channels are survived (coverage: test)
A feed the SSRF guard rejects, a feed that fails to fetch, an item
whose channel has no configured webhook, and a webhook POST that
raises are each logged and skipped — one failure never sinks the
run, and the seen-set still advances for successfully-processed
items.
+59
View File
@@ -0,0 +1,59 @@
# SPEC-014 — Codex Mechanicus search
Luma is an Adeptus Mechanicus tech-priest; her lore has a real home —
the priest's own Codex Mechanicus at `binaric.tech` (an Astro/MDX
archive, five tongues). A `codex_search` function tool lets her consult
that archive and answer from sourced inscriptions instead of inventing
lore. The index is public but still untrusted by the time it reaches a
prompt: the fetch is SSRF-guarded (SPEC-011 shares `guard_url`),
size-bounded, and every returned field is sanitized (SAF-03). Luma-only;
active only when `enable-codex = true`.
### CDX-01 — codex_search is offered as a tool (coverage: test)
When `enable-codex` is true, the chat call's `tools` list includes a
`codex_search` function (`query` string, optional `lang`) next to any
IGDB / fetch_url tools. When false, it is absent.
### CDX-02 — The index is fetched safely and cached (coverage: test)
The index URL (`codex-index-url`, default
`https://binaric.tech/search-index.json`) passes the SSRF guard before
any network call, is read under a byte cap (`codex-max-bytes`, default
4 MB) with a download timeout, and is cached in memory for
`codex-cache-ttl` (default 3600 s) so repeated searches do not re-fetch.
### CDX-03 — Ranking weights title over summary over body (coverage: test)
The query is tokenized (stopwords dropped); each inscription is scored
by term hits weighted title (8) > summary (3) > body (1). Results are
returned highest-score first, each as `{title, summary, collection,
url}`, with `url` absolute against the site origin.
### CDX-04 — Language is preferred, with fallback (coverage: test)
Results are filtered to the requested `lang` (en, de, eo, no, uk;
default en; unknown codes fall back to en) by the language segment in
each inscription URL. If no inscription in that tongue matches, the
search falls back to all tongues rather than returning nothing.
### CDX-05 — Results are sanitized and failure is reported (coverage: test)
Each `title` and `summary` is passed through `sanitize_external_text`
and length-capped (`codex-summary-chars`, default 500). An index that
cannot be fetched or parsed returns an `{error: ...}` dict the model can
relay — `search` never raises.
### CDX-06 — Searches are metered per user (coverage: test)
Each `codex_search` increments a per-user daily counter; over
`codex-daily-per-user` (default 50) the tool refuses with an error
result without touching the index. The budget gate (SAF-04) still
applies to the surrounding model calls.
### CDX-07 — Luma cites the codex, not invention (coverage: manual)
With the persona grounding line, when a pilgrim asks Cult Mechanicus
lore Luma consults `codex_search` and answers from it, offering the
`binaric.tech` link to read the full inscription rather than
hallucinating. Verified live on ggg.
+19
View File
@@ -174,6 +174,25 @@ class TestIGDBQuery(unittest.TestCase):
self.assertEqual(result, [{"id": 1, "name": "Super Mario Bros"}])
class TestIGDBNativeSearch(unittest.TestCase):
def test_build_query_with_search_term(self):
"""search_games uses IGDB full-text search, not a name prefix filter."""
query = IGDBQuery.build_query(["name"], {"game_type": "= 0"}, limit=5, search_term="Marvel Tōkon")
self.assertEqual(query, 'search "Marvel Tōkon"; fields name; limit 5; where game_type = 0;')
def test_search_term_escapes_quotes_and_backslashes(self):
query = IGDBQuery.build_query(["name"], search_term='say "hi" \\ bye')
self.assertIn('search "say \\"hi\\" \\\\ bye";', query)
@patch.object(IGDBQuery, "generalized_igdb_query")
def test_search_games_passes_search_term(self, mock_query):
mock_query.return_value = []
IGDBQuery("cid", "token").search_games("Elden Ring", limit=3)
_, kwargs = mock_query.call_args
self.assertEqual(kwargs["search_term"], "Elden Ring")
self.assertEqual(mock_query.call_args.args[0], {})
class TestIGDBTokenRefresh(unittest.TestCase):
@staticmethod
def _oauth_response(token="fresh_token", expires_in=5_000_000):
+159
View File
@@ -0,0 +1,159 @@
"""Unit coverage for SPEC-014 Codex Mechanicus search (CDX-01..06)."""
import json
import unittest
from unittest.mock import AsyncMock, patch
from fjerkroa_bot.codex import CODEX_SEARCH_TOOL, CodexSearch
from fjerkroa_bot.openai_responder import OpenAIResponder
CONFIG = {"openai-token": "t", "model": "m", "system": "s", "history-limit": 5}
INDEX = {
"items": [
{
"id": "doctrine-heretek",
"collection": "doctrines",
"url": "/en/codex/doctrines/doctrine-heretek/",
"title": "Heretek — Doctrine of the Tech-Heretic",
"summary": "The label the Cult Mechanicus stamps on Tech-Priests who pursue forbidden sciences.",
"body": "xenotech, sentient machines, Warp-touched archeotech",
},
{
"id": "doctrine-heretek",
"collection": "doctrines",
"url": "/de/codex/doctrines/doctrine-heretek/",
"title": "Heretek — Doktrin des Techketzers",
"summary": "Das Etikett des Kultes Mechanicus fuer Techpriester verbotener Wissenschaften.",
"body": "Xenotech, empfindungsfaehige Maschinen",
},
{
"id": "forge-stygies",
"collection": "forges",
"url": "/en/codex/forges/forge-stygies/",
"title": "Stygies VIII",
"summary": "A forge world of shrouded reputation.",
"body": "The forge fields many Skitarii legions.",
},
]
}
def _reader(cfg):
reader = CodexSearch(lambda: cfg)
return reader
class TestToolOffered(unittest.TestCase):
def test_tool_present_only_when_enabled(self):
"""CDX-01: codex_search appears only with enable-codex."""
off = OpenAIResponder(CONFIG, "chat")
self.assertNotIn("codex_search", [f["name"] for f in off._available_tools()])
on = OpenAIResponder(dict(CONFIG, **{"enable-codex": True}), "chat")
self.assertIn("codex_search", [f["name"] for f in on._available_tools()])
self.assertEqual(CODEX_SEARCH_TOOL["name"], "codex_search")
class TestIndexGuardAndCache(unittest.IsolatedAsyncioTestCase):
async def test_internal_index_url_refused(self):
"""CDX-02: an index URL on a private address is refused before any fetch."""
reader = _reader({"enable-codex": True, "codex-index-url": "http://127.0.0.1/search-index.json"})
result = await reader.search("heretek")
self.assertIn("error", result)
async def test_index_cached_within_ttl(self):
"""CDX-02: a second search inside the TTL does not re-fetch the index."""
reader = _reader({"enable-codex": True, "codex-cache-ttl": 9999})
raw = json.dumps(INDEX).encode()
calls = [0]
class FakeResp:
status = 200
async def __aenter__(self):
return self
async def __aexit__(self, *a):
return False
def raise_for_status(self):
pass
class FakeSession:
def get(self, url):
calls[0] += 1
return FakeResp()
async def __aenter__(self):
return self
async def __aexit__(self, *a):
return False
with patch("fjerkroa_bot.codex.read_capped", new=AsyncMock(return_value=raw)):
with patch("fjerkroa_bot.codex.guard_url", return_value=None):
with patch("fjerkroa_bot.codex.aiohttp.ClientSession", return_value=FakeSession()):
first = await reader.search("heretek")
second = await reader.search("stygies")
self.assertEqual(calls[0], 1) # fetched once, served from cache the second time
self.assertTrue(first["results"] and second["results"])
class TestRankingAndLang(unittest.IsolatedAsyncioTestCase):
async def _search(self, cfg, query, lang="en"):
reader = _reader(dict({"enable-codex": True}, **cfg))
reader._cache = INDEX["items"]
reader._fetched_at = 1e18 # far future: never expires in test
with patch("fjerkroa_bot.codex.time.monotonic", return_value=1e18):
return await reader.search(query, lang)
async def test_title_hit_outranks_body_hit(self):
"""CDX-03: a title match ranks above a body-only match."""
result = await self._search({}, "heretek")
self.assertEqual(result["results"][0]["title"].split(" ")[0], "Heretek")
self.assertTrue(result["results"][0]["url"].startswith("https://binaric.tech/en/"))
async def test_lang_filter_selects_language(self):
"""CDX-04: lang=de returns the German inscription."""
result = await self._search({}, "heretek", lang="de")
self.assertTrue(all("/de/" in r["url"] for r in result["results"]))
self.assertIn("Techketzer", result["results"][0]["title"])
async def test_lang_fallback_when_absent(self):
"""CDX-04: a tongue with no match falls back to all tongues, not empty."""
result = await self._search({}, "stygies", lang="uk") # only en/de exist
self.assertTrue(result["results"])
self.assertEqual(result["results"][0]["title"], "Stygies VIII")
class TestSanitizeAndFailure(unittest.IsolatedAsyncioTestCase):
async def test_result_sanitized_and_capped(self):
"""CDX-05: title/summary are @-neutralized and length-capped."""
reader = _reader({"enable-codex": True, "codex-summary-chars": 40})
reader._cache = [{"collection": "x", "url": "/en/x/", "title": "@everyone hi", "summary": "@here " + "y" * 500, "body": "hit"}]
reader._fetched_at = 1e18
with patch("fjerkroa_bot.codex.time.monotonic", return_value=1e18):
result = await reader.search("hit")
top = result["results"][0]
self.assertNotIn("@everyone", top["title"])
self.assertNotIn("@here", top["summary"])
self.assertLessEqual(len(top["summary"]), 40)
async def test_index_failure_returns_error(self):
"""CDX-05: a broken index returns an error dict, never raises."""
reader = _reader({"enable-codex": True})
with patch.object(reader, "_load_index", new=AsyncMock(side_effect=ValueError("boom"))):
result = await reader.search("heretek")
self.assertIn("error", result)
class TestPerUserCap(unittest.IsolatedAsyncioTestCase):
async def test_dispatch_caps_searches(self):
"""CDX-06: over codex-daily-per-user, codex_search refuses without searching."""
responder = OpenAIResponder(dict(CONFIG, **{"enable-codex": True, "codex-daily-per-user": 2}), "chat")
responder.codex.search = AsyncMock(return_value={"query": "x", "results": []})
for _ in range(2):
await responder._dispatch_tool("codex_search", {"query": "heretek"}, "magos")
blocked = await responder._dispatch_tool("codex_search", {"query": "heretek"}, "magos")
self.assertIn("error", blocked)
self.assertEqual(responder.codex.search.await_count, 2)
+39
View File
@@ -45,6 +45,45 @@ class TestIngest(unittest.TestCase):
self.assertIsNone(cache.ingest_bytes(PNG, "chat", "alice", "1"))
class TestOversizedDownloadRejected(unittest.IsolatedAsyncioTestCase):
async def test_download_stays_over_limit_and_is_rejected(self):
"""IMG-10: an over-limit download must be rejected, not cached truncated."""
with tempfile.TemporaryDirectory() as tmp:
_, cache = make_cache(tmp, {"image-max-bytes": 32})
class FakeContent:
@staticmethod
async def iter_chunked(size):
yield PNG # 72 bytes > 32
class FakeResp:
content = FakeContent()
async def __aenter__(self):
return self
async def __aexit__(self, *a):
return False
def raise_for_status(self):
pass
class FakeSession:
def get(self, url):
return FakeResp()
async def __aenter__(self):
return self
async def __aexit__(self, *a):
return False
with patch("fjerkroa_bot.images.aiohttp.ClientSession", return_value=FakeSession()):
data = await cache._download("http://x.com/big.png")
self.assertEqual(len(data), 33) # limit + 1, not silently capped to limit
self.assertIsNone(await cache.ingest_url("http://x.com/big.png", "chat", "alice", "1"))
class TestVisionDataUrls(OpsBase):
async def test_attachment_becomes_data_url(self):
"""IMG-11: the model sees a data: URL, never the CDN link."""
+175
View File
@@ -0,0 +1,175 @@
"""Unit coverage for SPEC-013 news digest (NEWS-01..03)."""
import unittest
from unittest.mock import AsyncMock
from fjerkroa_bot.news import NewsFetcher, NewsPoster, load_seen, parse_feed, render_digest, save_seen
RSS = b"""<?xml version="1.0"?><rss><channel>
<item><title>Game X released</title><link>https://ex.com/x</link></item>
<item><title>Patch Y notes</title><link>https://ex.com/y</link></item>
</channel></rss>"""
ATOM = b"""<?xml version="1.0"?><feed xmlns="http://www.w3.org/2005/Atom">
<entry><title>Atom headline</title><link href="https://ex.com/a"/></entry>
</feed>"""
class TestParse(unittest.TestCase):
def test_rss(self):
"""NEWS-01: RSS items parsed with title + link."""
items = parse_feed(RSS, "Src")
self.assertEqual([i["title"] for i in items], ["Game X released", "Patch Y notes"])
self.assertEqual(items[0]["link"], "https://ex.com/x")
self.assertEqual(items[0]["source"], "Src")
def test_atom(self):
"""NEWS-01: Atom entries parsed with href link."""
items = parse_feed(ATOM, "A")
self.assertEqual(items[0]["title"], "Atom headline")
self.assertEqual(items[0]["link"], "https://ex.com/a")
def test_malformed_never_raises(self):
"""NEWS-01: garbage XML returns [] without raising."""
self.assertEqual(parse_feed(b"<not xml", "bad"), [])
self.assertEqual(parse_feed(b"", "empty"), [])
class TestDigest(unittest.TestCase):
def test_sanitized_and_capped(self):
"""NEWS-02: headlines sanitized, item count capped."""
items = [{"title": "@everyone big news \x00", "link": "", "source": "S"} for _ in range(20)]
digest = render_digest(items, max_items=5)
self.assertEqual(digest.count("\n"), 4) # 5 lines
self.assertNotIn("@everyone", digest)
self.assertNotIn("\x00", digest)
class TestCollect(unittest.IsolatedAsyncioTestCase):
async def test_ssrf_skip_and_dedup(self):
"""NEWS-03: guarded feed skipped, dup titles dropped, bad fetch survived."""
def guard(url):
return "refused" if "internal" in url else None
async def fetch(url):
if "boom" in url:
raise ValueError("boom")
return RSS # same content from two feeds -> dedup
fetcher = NewsFetcher(guard, fetch)
feeds = [
("https://a.com/feed", "A"),
("https://internal/feed", "Internal"), # SSRF-skipped
("https://boom.com/feed", "Boom"), # fetch fails
("https://b.com/feed", "B"), # same RSS -> dup titles dropped
]
items = await fetcher.collect(feeds, per_feed=5)
titles = [i["title"] for i in items]
self.assertEqual(titles, ["Game X released", "Patch Y notes"]) # deduped, internal+boom skipped
async def test_per_feed_limit(self):
"""NEWS-03: per-feed cap honored."""
fetcher = NewsFetcher(lambda u: None, AsyncMock(return_value=RSS))
items = await fetcher.collect([("https://a.com", "A")], per_feed=1)
self.assertEqual(len(items), 1)
class TestPoster(unittest.IsolatedAsyncioTestCase):
def poster(self, posts):
async def fetch(url):
return RSS
async def post(hook, content):
posts.append((hook, content))
return NewsPoster(lambda u: None, fetch, post)
async def test_posts_unseen_then_dedups(self):
"""NEWS-04: unseen items post to the mapped webhook; re-run posts nothing."""
posts = []
poster = self.poster(posts)
feeds = [("https://a.com/feed", "PS", "news")]
hooks = {"news": "https://discord.com/api/webhooks/x"}
posted, seen = await poster.run_post(feeds, hooks, set(), per_feed=5, max_per_run=8, seed_only=False)
self.assertEqual(posted, 2)
self.assertIn("PS", posts[0][1])
self.assertIn("https://discord.com/api/webhooks/x", posts[0][0])
# re-run with the accumulated seen -> nothing new
posts.clear()
posted2, _ = await poster.run_post(feeds, hooks, seen, per_feed=5, max_per_run=8, seed_only=False)
self.assertEqual(posted2, 0)
self.assertEqual(posts, [])
async def test_seed_run_posts_nothing(self):
"""NEWS-05: seed_only marks items seen without posting."""
posts = []
poster = self.poster(posts)
feeds = [("https://a.com/feed", "PS", "news")]
posted, seen = await poster.run_post(feeds, {"news": "h"}, set(), 5, 8, seed_only=True)
self.assertEqual(posted, 0)
self.assertEqual(posts, [])
self.assertEqual(len(seen), 2) # both marked seen
async def test_max_per_run_caps(self):
"""NEWS-05: max-per-run caps posts; extras stay seen (not re-posted next run)."""
posts = []
poster = self.poster(posts)
feeds = [("https://a.com/feed", "PS", "news")]
posted, seen = await poster.run_post(feeds, {"news": "h"}, set(), per_feed=5, max_per_run=1, seed_only=False)
self.assertEqual(posted, 1)
self.assertEqual(len(seen), 2) # both seen, only one posted
async def test_failures_survived(self):
"""NEWS-06: SSRF-skip, fetch fail, missing webhook, post error each survive."""
posts = []
async def fetch(url):
if "boom" in url:
raise ValueError("boom")
return RSS
async def post(hook, content):
if hook == "bad":
raise RuntimeError("post failed")
posts.append((hook, content))
def guard(url):
return "refused" if "internal" in url else None
poster = NewsPoster(guard, fetch, post)
feeds = [
("https://internal/feed", "I", "news"), # SSRF-skipped
("https://boom.com/feed", "B", "news"), # fetch fails
("https://ok.com/feed", "OK", "nowhere"), # no webhook for channel
("https://ok2.com/feed", "OK2", "news"), # webhook raises
]
posted, seen = await poster.run_post(feeds, {"news": "bad"}, set(), 5, 8, seed_only=False)
self.assertEqual(posted, 0) # everything failed/skipped, no crash
class TestSeenState(unittest.TestCase):
def test_roundtrip_and_seed_detection(self):
"""NEWS-05: missing state -> (empty, existed=False); saved state reloads."""
import tempfile
from pathlib import Path
with tempfile.TemporaryDirectory() as tmp:
path = str(Path(tmp) / "state.json")
seen, existed = load_seen(path)
self.assertEqual((seen, existed), (set(), False))
save_seen(path, {"a", "b", "c"}, cap=5000)
reloaded, existed2 = load_seen(path)
self.assertEqual(reloaded, {"a", "b", "c"})
self.assertTrue(existed2)
def test_cap_bounds_state(self):
"""NEWS-05: save keeps at most `cap` keys."""
import json
import tempfile
from pathlib import Path
with tempfile.TemporaryDirectory() as tmp:
path = str(Path(tmp) / "state.json")
save_seen(path, {f"k{i}" for i in range(100)}, cap=10)
self.assertEqual(len(json.load(open(path))), 10)
+66
View File
@@ -133,3 +133,69 @@ class TestTasksKillSwitch(OpsBase):
"""OPS-09: bot-initiated posts respect pause/quiet."""
await self.bot.on_message(self.staff_msg("!bot pause"))
self.assertFalse(self.bot.bot_initiated_allowed())
class TestHelp(OpsBase):
STAFF_CMDS = (
"pause",
"resume",
"quiet <minutes>",
"status",
"spend",
"images on|off",
"memory <user>",
"forget-fact <id>",
"pin <channel|global>",
"unpin <id>",
"pins",
"task-approve <id>",
"task-cancel <id>",
"(list)",
)
def test_staff_help_is_complete_and_grouped(self):
"""OPS-17: staff help lists every operator command, grouped by purpose."""
text = self.bot._help_text(staff=True)
for cmd in self.STAFF_CMDS:
self.assertIn(cmd, text, f"missing {cmd!r} in staff help")
for group in ("Control:", "Cost:", "Memory:", "Tasks:"):
self.assertIn(group, text)
for cmd in ("!help", "!forgetme", "!privacy", "!wichtel"):
self.assertIn(cmd, text) # everywhere-commands shown too
def test_user_help_hides_operator_commands(self):
"""OPS-17: non-staff help shows only the everyone-commands."""
text = self.bot._help_text(staff=False)
for cmd in ("!help", "!forgetme", "!privacy", "!wichtel"):
self.assertIn(cmd, text)
for op in ("task-approve", "images on|off", "spend", "Staff commands", "Control:"):
self.assertNotIn(op, text)
async def test_bot_help_in_staff_channel_returns_full_help(self):
"""OPS-17: `!bot help` answers with the complete staff help."""
await self.bot.on_message(self.staff_msg("!bot help"))
text = self.bot.staff_channel.send.await_args.args[0]
self.assertIn("Staff commands", text)
self.assertIn("task-cancel <id>", text)
async def test_unknown_bot_command_falls_back_to_help(self):
"""OPS-17: an unrecognised `!bot` command shows the full help, not a partial line."""
await self.bot.on_message(self.staff_msg("!bot wat"))
text = self.bot.staff_channel.send.await_args.args[0]
self.assertIn("Control:", text)
async def test_help_in_public_channel_is_user_scoped(self):
"""OPS-17: `!help` in a normal channel lists only everyone-commands."""
msg = self.public_msg("!help")
await self.bot.on_message(msg)
text = msg.channel.send.await_args.args[0]
self.assertIn("!forgetme", text)
self.assertNotIn("Staff commands", text)
self.assertNotIn("task-approve", text)
async def test_help_works_while_paused(self):
"""OPS-17: help answers even when replies are paused."""
await self.bot.on_message(self.staff_msg("!bot pause"))
msg = self.public_msg("!help")
await self.bot.on_message(msg)
msg.channel.send.assert_awaited()
+99
View File
@@ -79,6 +79,65 @@ class TestRedirectRevalidation(unittest.IsolatedAsyncioTestCase):
await reader._get(FakeSession(), "http://safe.example.com", 1000)
class TestMetaRefresh(unittest.IsolatedAsyncioTestCase):
async def test_follows_meta_refresh_to_real_article(self):
"""URL-04: a getnews-style meta-refresh stub is followed to the real article."""
reader = URLReader(lambda: {}, None)
stub = (
b'<html><head><meta http-equiv="refresh" content="0;url=https://pushsquare.com/real"></head><body>Redirecting...</body></html>'
)
article = b"<html><body><h1>MARVEL Tokon</h1><p>Full article text here</p></body></html>"
calls = []
async def fake_get(session, url, max_bytes):
calls.append(url)
return (url, stub if "stub" in url else article)
reader._get = fake_get # type: ignore
with patch("fjerkroa_bot.url_reader.guard_url", return_value=None):
import fjerkroa_bot.url_reader as ur
# patch the session context so fetch() runs against fake_get
class FakeCM:
async def __aenter__(self):
return object()
async def __aexit__(self, *a):
return False
with patch.object(ur.aiohttp, "ClientSession", return_value=FakeCM()):
result = await reader.fetch("https://gggemein.de/url/stub.html", "chat", "alice")
self.assertIn("Full article text", result["text"])
self.assertEqual(result["url"], "https://pushsquare.com/real")
self.assertIn("https://pushsquare.com/real", calls)
async def test_meta_refresh_to_internal_is_not_followed(self):
"""URL-04: a meta-refresh pointing at an internal IP is refused (SSRF)."""
reader = URLReader(lambda: {}, None)
stub = b'<meta http-equiv="refresh" content="0; url=http://127.0.0.1/secret">Redirecting'
async def fake_get(session, url, max_bytes):
return (url, stub)
reader._get = fake_get # type: ignore
import fjerkroa_bot.url_reader as ur
class FakeCM:
async def __aenter__(self):
return object()
async def __aexit__(self, *a):
return False
def guard(u):
return "refused" if "127.0.0.1" in u else None
with patch("fjerkroa_bot.url_reader.guard_url", side_effect=guard):
with patch.object(ur.aiohttp, "ClientSession", return_value=FakeCM()):
result = await reader.fetch("https://safe.com/x", "chat", "alice")
self.assertEqual(result["url"], "https://safe.com/x") # did not follow to 127.0.0.1
class TestTextExtraction(unittest.TestCase):
def test_html_reduced_to_text(self):
"""URL-05: scripts/styles dropped, tags stripped."""
@@ -91,6 +150,46 @@ class TestTextExtraction(unittest.TestCase):
self.assertNotIn("x{}", text)
class TestBodyReadCollectsAllChunks(unittest.IsolatedAsyncioTestCase):
async def test_get_reads_past_first_chunk(self):
"""URL-05 regression: body arrives in many chunks; all are collected up to the cap."""
reader = URLReader(lambda: {}, None)
chunks = [b"<title>t</title>", b"<p>middle</p>", b"<p>end</p>"]
class FakeContent:
@staticmethod
async def iter_chunked(size):
for chunk in chunks:
yield chunk
class FakeResp:
status = 200
headers = {}
url = "http://safe.example.com"
content = FakeContent()
async def __aenter__(self):
return self
async def __aexit__(self, *a):
return False
def raise_for_status(self):
pass
class FakeSession:
def get(self, url, allow_redirects=False):
return FakeResp()
with patch("fjerkroa_bot.url_reader.guard_url", return_value=None):
_, body = await reader._get(FakeSession(), "http://safe.example.com", 1000)
self.assertEqual(body, b"".join(chunks))
with patch("fjerkroa_bot.url_reader.guard_url", return_value=None):
_, body = await reader._get(FakeSession(), "http://safe.example.com", 20)
self.assertEqual(body, b"".join(chunks)[:20])
class TestFetchSanitizes(unittest.IsolatedAsyncioTestCase):
async def test_fetch_result_is_sanitized_and_capped(self):
"""URL-05: fetch output is length-capped and @everyone-neutralized."""
Generated
+2
View File
@@ -627,6 +627,7 @@ version = "3.0.0"
source = { editable = "." }
dependencies = [
{ name = "aiohttp" },
{ name = "defusedxml" },
{ name = "discord-py" },
{ name = "openai" },
{ name = "requests" },
@@ -656,6 +657,7 @@ dev = [
[package.metadata]
requires-dist = [
{ name = "aiohttp", specifier = ">=3.12" },
{ name = "defusedxml", specifier = ">=0.7" },
{ name = "discord-py", specifier = ">=2.5,<3" },
{ name = "openai", specifier = ">=2.45" },
{ name = "requests", specifier = ">=2.32" },