Compare commits
7 Commits
| Author | SHA1 | Date | |
|---|---|---|---|
| 5d01400638 | |||
| 891fbdc101 | |||
| 09871b9b95 | |||
| a514ff652c | |||
| 7628faf551 | |||
| 2caa18a17f | |||
| d4eec4088d |
@@ -60,6 +60,28 @@ Decisions inside the set architecture. D-NNN, never renumbered.
|
||||
broken classifier must never mute the bot; the budget gate already
|
||||
bounds spend. Its verdict gates BEFORE the main call, the
|
||||
envelope's answer_needed still gates after — two independent nets.
|
||||
- **D-019** — News memory + on-demand tool (SPEC-013 NEWS-07..12):
|
||||
the news pipeline now carries item summaries (feed descriptions,
|
||||
HTML-stripped) and persists every fetched item into a deduped `news`
|
||||
table (schema v6), pruned to a rolling window (`news-keep`). Both the
|
||||
kroa digest run and the ggg posting run write to it, so the store is
|
||||
a single searchable source across both models. A `get_news` tool
|
||||
reads that store (topic/source-filtered, metered, sanitized) rather
|
||||
than re-fetching feeds live: the ambient `{news}` digest stays a
|
||||
small always-on snapshot, while the tool gives unbounded on-demand
|
||||
reach without a fresh network round-trip per call. The store is the
|
||||
same `bot.db` (WAL) the bot uses; the cron process opens it
|
||||
independently — concurrent reader/writer is what WAL is for.
|
||||
- **D-018** — Codex Mechanicus search (FDB-019, SPEC-014): Luma's
|
||||
lore is grounded in the priest's real archive at binaric.tech via a
|
||||
`codex_search` tool over the site's public `search-index.json`, not
|
||||
a bot-side copy — the index stays a single source of truth, refreshed
|
||||
by the site's own publish rite, and the bot caches it in memory
|
||||
(TTL). It reuses SPEC-011's `guard_url` + `read_capped` (fetch is
|
||||
SSRF-guarded and byte-bounded) and sanitizes every returned field:
|
||||
one's own web content is still untrusted by the time it reaches a
|
||||
prompt. Luma-only (`enable-codex`, off elsewhere) — the Adeptus
|
||||
Mechanicus archive has no place in Fjærkroa's café persona.
|
||||
- **D-017** — All human-behavior knobs default to off/v3.0.0
|
||||
semantics; behavior changes are config rollouts per deployment, not
|
||||
code flips. The classifier's `factual` flag is the only coupling
|
||||
|
||||
+12
@@ -80,3 +80,15 @@ enable-game-info = true
|
||||
# Ops (SPEC-012): consecutive OpenAI failures before a staff alert
|
||||
# api-error-alert-threshold = 5
|
||||
# Backups: cron runs deploy/backup_db.py daily -> ~/backups/<bot>/ (keep 14)
|
||||
|
||||
# News digest (SPEC-013) — `python -m fjerkroa_bot.news --config X.toml` via cron;
|
||||
# writes the {news} file. Feeds are [url, label] pairs (RSS or Atom):
|
||||
# news = "news_feed.txt"
|
||||
# news-per-feed = 3
|
||||
# news-max-items = 15
|
||||
# news-feeds = [
|
||||
# ["https://blog.playstation.com/feed/", "PS"],
|
||||
# ["https://kotaku.com/rss", "Kotaku"],
|
||||
# ["https://www.pushsquare.com/feeds/latest", "Push"],
|
||||
# ["https://mein-mmo.de/feed/", "MeinMMO"],
|
||||
# ]
|
||||
|
||||
@@ -0,0 +1,147 @@
|
||||
"""Codex Mechanicus search tool (SPEC-014, FDB-019).
|
||||
|
||||
Luma's own sacred archive — the Codex Mechanicus at binaric.tech — as a
|
||||
function tool. She searches the codex index and answers Cult Mechanicus
|
||||
lore from real, sourced inscriptions instead of inventing it. The index
|
||||
is fetched over HTTPS (SSRF-guarded, size-bounded, cached in memory) and
|
||||
every field returned to the model is sanitized (SAF-03), because even
|
||||
one's own web content is still untrusted input by the time it reaches a
|
||||
prompt.
|
||||
|
||||
The model calls `codex_search`; production wires the live index URL.
|
||||
"""
|
||||
|
||||
import json
|
||||
import logging
|
||||
import time
|
||||
from typing import Any, Callable, Dict, List, Optional
|
||||
from urllib.parse import urljoin
|
||||
|
||||
import aiohttp
|
||||
|
||||
from .ai_responder import sanitize_external_text
|
||||
from .httpread import read_capped
|
||||
from .url_reader import guard_url
|
||||
|
||||
DEFAULT_INDEX_URL = "https://binaric.tech/search-index.json"
|
||||
DEFAULT_MAX_BYTES = 4 * 1024 * 1024
|
||||
DEFAULT_LIMIT = 5
|
||||
DEFAULT_TTL_S = 3600
|
||||
DEFAULT_SUMMARY_CHARS = 500
|
||||
FETCH_TIMEOUT_S = 15
|
||||
_VALID_LANGS = ("en", "de", "eo", "no", "uk")
|
||||
|
||||
CODEX_SEARCH_TOOL = {
|
||||
"name": "codex_search",
|
||||
"description": "Search Luma's own Codex Mechanicus (the sacred archive at binaric.tech) for Adeptus "
|
||||
"Mechanicus lore: doctrines, forges, orders, rites, relics, weapons, entities, the lexicon, and the "
|
||||
"priest's own adoptus. Returns matching inscriptions with a short summary and the URL to read the full "
|
||||
"text. Use for any Cult Mechanicus / Warhammer 40k Mechanicus question so the answer is grounded in the "
|
||||
"codex, not invented.",
|
||||
"parameters": {
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"query": {"type": "string", "description": "What to look for: a name, concept, rite, or phrase."},
|
||||
"lang": {"type": "string", "description": "Language of the inscriptions to prefer: en, de, eo, no, uk. Default en."},
|
||||
},
|
||||
"required": ["query"],
|
||||
},
|
||||
}
|
||||
|
||||
_STOP = {"the", "a", "an", "of", "and", "or", "to", "in", "is", "der", "die", "das", "und", "von", "en", "et"}
|
||||
|
||||
|
||||
def _tokenize(text: str) -> List[str]:
|
||||
cleaned = "".join(c.lower() if c.isalnum() else " " for c in text)
|
||||
return [t for t in cleaned.split() if len(t) > 1 and t not in _STOP]
|
||||
|
||||
|
||||
def _score(item: Dict[str, Any], terms: List[str]) -> int:
|
||||
"""Weight a hit by field: title beats summary beats body (CDX-03)."""
|
||||
title = str(item.get("title") or "").lower()
|
||||
summary = str(item.get("summary") or "").lower()
|
||||
body = str(item.get("body") or "").lower()
|
||||
score = 0
|
||||
for term in terms:
|
||||
score += 8 if term in title else 0
|
||||
score += 3 if term in summary else 0
|
||||
score += 1 if term in body else 0
|
||||
return score
|
||||
|
||||
|
||||
def _rank(items: List[Dict[str, Any]], terms: List[str], lang: str) -> List[Dict[str, Any]]:
|
||||
"""Score items in the given language; fall back to all languages if empty (CDX-04)."""
|
||||
|
||||
def scored(only_lang: Optional[str]) -> List[Any]:
|
||||
out = []
|
||||
for item in items:
|
||||
if only_lang and f"/{only_lang}/" not in str(item.get("url") or ""):
|
||||
continue
|
||||
hit = _score(item, terms)
|
||||
if hit > 0:
|
||||
out.append((hit, item))
|
||||
out.sort(key=lambda pair: pair[0], reverse=True)
|
||||
return out
|
||||
|
||||
ranked = scored(lang) or scored(None)
|
||||
return [item for _, item in ranked]
|
||||
|
||||
|
||||
class CodexSearch:
|
||||
def __init__(self, config_getter: Callable[[], Dict[str, Any]]) -> None:
|
||||
self._config = config_getter
|
||||
self._cache: Optional[List[Dict[str, Any]]] = None
|
||||
self._fetched_at = 0.0
|
||||
|
||||
def enabled(self) -> bool:
|
||||
return bool(self._config().get("enable-codex", False))
|
||||
|
||||
def _index_url(self) -> str:
|
||||
return str(self._config().get("codex-index-url", DEFAULT_INDEX_URL))
|
||||
|
||||
async def _load_index(self) -> List[Dict[str, Any]]:
|
||||
"""Fetch + cache the codex index, SSRF-guarded and size-bounded (CDX-02)."""
|
||||
ttl = float(self._config().get("codex-cache-ttl", DEFAULT_TTL_S))
|
||||
if self._cache is not None and (time.monotonic() - self._fetched_at) < ttl:
|
||||
return self._cache
|
||||
url = self._index_url()
|
||||
reason = guard_url(url)
|
||||
if reason:
|
||||
raise ValueError(reason)
|
||||
max_bytes = int(self._config().get("codex-max-bytes", DEFAULT_MAX_BYTES))
|
||||
timeout = aiohttp.ClientTimeout(total=FETCH_TIMEOUT_S)
|
||||
async with aiohttp.ClientSession(timeout=timeout, headers={"User-Agent": "FjerkroaBot-codex/1.0"}) as session:
|
||||
async with session.get(url) as response:
|
||||
response.raise_for_status()
|
||||
raw = await read_capped(response, max_bytes)
|
||||
data = json.loads(raw.decode("utf-8", "ignore"))
|
||||
items = data.get("items", []) if isinstance(data, dict) else []
|
||||
self._cache = [i for i in items if isinstance(i, dict)]
|
||||
self._fetched_at = time.monotonic()
|
||||
return self._cache
|
||||
|
||||
async def search(self, query: str, lang: str = "en", limit: int = DEFAULT_LIMIT) -> Dict[str, Any]:
|
||||
"""Return sanitized top matches, or an error dict — never raise (CDX-05)."""
|
||||
try:
|
||||
items = await self._load_index()
|
||||
except Exception as err:
|
||||
logging.warning(f"codex: index load failed: {err!r}")
|
||||
return {"error": f"codex unavailable: {err}"}
|
||||
terms = _tokenize(query)
|
||||
if not terms:
|
||||
return {"query": query, "results": []}
|
||||
pick = (lang or "en").lower()
|
||||
if pick not in _VALID_LANGS:
|
||||
pick = "en"
|
||||
summary_chars = int(self._config().get("codex-summary-chars", DEFAULT_SUMMARY_CHARS))
|
||||
results = []
|
||||
for item in _rank(items, terms, pick)[: max(1, limit)]:
|
||||
results.append(
|
||||
{
|
||||
"title": sanitize_external_text(str(item.get("title") or ""), 200),
|
||||
"summary": sanitize_external_text(str(item.get("summary") or ""), summary_chars),
|
||||
"collection": str(item.get("collection") or ""),
|
||||
"url": urljoin(self._index_url(), str(item.get("url") or "")),
|
||||
}
|
||||
)
|
||||
return {"query": query, "lang": pick, "results": results}
|
||||
@@ -182,6 +182,9 @@ class FjerkroaBot(commands.Bot):
|
||||
if content.startswith("!privacy"):
|
||||
await message.channel.send(self.config.get("privacy-notice", DEFAULT_PRIVACY_NOTICE), suppress_embeds=True)
|
||||
return
|
||||
if content.startswith("!help"): # OPS-17: context-aware, works even while paused
|
||||
await message.channel.send(self._help_text(staff=self.is_staff_channel(message.channel)), suppress_embeds=True)
|
||||
return
|
||||
if not self.replies_allowed():
|
||||
return
|
||||
if str(message.content).startswith("!wichtel"):
|
||||
@@ -263,15 +266,35 @@ class FjerkroaBot(commands.Bot):
|
||||
return f"Cancelled {store.task_set_state(int(args[1]), 'cancelled')} task(s)."
|
||||
return None
|
||||
|
||||
def _help_text(self, staff: bool) -> str:
|
||||
"""Context-aware command help (OPS-17): every channel lists the user commands; the staff channel also lists operator commands."""
|
||||
everywhere = (
|
||||
"Available to everyone, in any channel:\n"
|
||||
"• `!help` — this help\n"
|
||||
"• `!forgetme` — delete your messages and memory traces (works even while I'm paused)\n"
|
||||
"• `!privacy` — how your data is handled (works even while I'm paused)\n"
|
||||
"• `!wichtel @a @b @c …` — draw Secret Santa pairings (needs ≥2 mentions; only while I'm active)"
|
||||
)
|
||||
if not staff:
|
||||
return everywhere
|
||||
operator = (
|
||||
"Staff commands — this channel only, prefixed `!bot`:\n"
|
||||
"• Control: `pause`, `resume`, `quiet <minutes>`, `status`\n"
|
||||
"• Cost: `spend`, `images on|off`\n"
|
||||
"• Memory: `memory <user>`, `forget-fact <id>`, `pin <channel|global> <text>`, `unpin <id>`, `pins`\n"
|
||||
"• Tasks: `tasks` (list), `tasks on|off`, `task-approve <id>`, `task-cancel <id>`"
|
||||
)
|
||||
return operator + "\n\n" + everywhere
|
||||
|
||||
async def handle_staff_command(self, message: Message) -> None:
|
||||
"""Operator kill-switches, staff channel only (OPS-01..05, OPS-09, MEM-07)."""
|
||||
"""Operator kill-switches, staff channel only (OPS-01..05, OPS-09, OPS-17, MEM-07)."""
|
||||
args = str(message.content).split()[1:]
|
||||
for handler in (self._memory_command, self._task_command):
|
||||
reply = handler(args)
|
||||
if reply is not None:
|
||||
await message.channel.send(reply, suppress_embeds=True)
|
||||
return
|
||||
reply = "Commands: pause, resume, images on|off, tasks on|off, quiet <minutes>, status, spend, memory <user>, forget-fact <id>, pin <channel|global> <fact>, unpin <id>"
|
||||
reply = self._help_text(staff=True) # OPS-17: unknown/`help` -> full grouped help
|
||||
if args[:1] == ["pause"]:
|
||||
self.replies_enabled = False
|
||||
reply = "Replies paused."
|
||||
|
||||
@@ -0,0 +1,18 @@
|
||||
"""Bounded HTTP body read (leaf module, no intra-package imports).
|
||||
|
||||
`response.content.read(n)` returns whatever is buffered, not n bytes,
|
||||
so it silently truncates large or chunked bodies (and web feeds/pages
|
||||
parse to garbage). This accumulates decompressed chunks up to a hard
|
||||
cap instead.
|
||||
"""
|
||||
|
||||
CHUNK = 65536
|
||||
|
||||
|
||||
async def read_capped(response, max_bytes: int) -> bytes:
|
||||
buf = bytearray()
|
||||
async for chunk in response.content.iter_chunked(CHUNK):
|
||||
buf.extend(chunk)
|
||||
if len(buf) > max_bytes:
|
||||
break
|
||||
return bytes(buf[:max_bytes])
|
||||
+12
-6
@@ -56,8 +56,12 @@ class IGDBQuery(object):
|
||||
return requests.post(igdb_url, headers=headers, data=query_body)
|
||||
|
||||
@staticmethod
|
||||
def build_query(fields, filters=None, limit=10, offset=None):
|
||||
query = f"fields {','.join(fields) if fields is not None and len(fields) > 0 else '*'}; limit {limit};"
|
||||
def build_query(fields, filters=None, limit=10, offset=None, search_term=None):
|
||||
query = ""
|
||||
if search_term:
|
||||
escaped = search_term.replace("\\", "\\\\").replace('"', '\\"')
|
||||
query += f'search "{escaped}"; '
|
||||
query += f"fields {','.join(fields) if fields is not None and len(fields) > 0 else '*'}; limit {limit};"
|
||||
if offset is not None:
|
||||
query += f" offset {offset};"
|
||||
if filters:
|
||||
@@ -65,12 +69,12 @@ class IGDBQuery(object):
|
||||
query += " where " + " & ".join(filter_statements) + ";"
|
||||
return query
|
||||
|
||||
def generalized_igdb_query(self, params, endpoint, fields, additional_filters=None, limit=10, offset=None):
|
||||
def generalized_igdb_query(self, params, endpoint, fields, additional_filters=None, limit=10, offset=None, search_term=None):
|
||||
all_filters = {key: f'~ "{value}"*' for key, value in params.items() if value}
|
||||
if additional_filters:
|
||||
all_filters.update(additional_filters)
|
||||
|
||||
query = self.build_query(fields, all_filters, limit, offset)
|
||||
query = self.build_query(fields, all_filters, limit, offset, search_term)
|
||||
data = self.send_igdb_request(endpoint, query)
|
||||
print(f"{endpoint}: {query} -> {data}")
|
||||
return data
|
||||
@@ -134,9 +138,10 @@ class IGDBQuery(object):
|
||||
return None
|
||||
|
||||
try:
|
||||
# Search for games with fuzzy matching
|
||||
# IGDB native full-text search: diacritic- and word-order-insensitive,
|
||||
# unlike a `name ~ "..."*` prefix filter
|
||||
games = self.generalized_igdb_query(
|
||||
{"name": query.strip()},
|
||||
{},
|
||||
"games",
|
||||
[
|
||||
"id",
|
||||
@@ -155,6 +160,7 @@ class IGDBQuery(object):
|
||||
],
|
||||
additional_filters={"game_type": "= 0"}, # Main games only (IGDB renamed category -> game_type)
|
||||
limit=limit,
|
||||
search_term=query.strip(),
|
||||
)
|
||||
|
||||
if not games:
|
||||
|
||||
@@ -14,6 +14,7 @@ from typing import Any, Callable, Dict, List, Optional
|
||||
|
||||
import aiohttp
|
||||
|
||||
from .httpread import read_capped
|
||||
from .persistence import PersistentStore
|
||||
|
||||
DEFAULT_CACHE_MB = 500
|
||||
@@ -79,7 +80,9 @@ class ImageCache:
|
||||
async with aiohttp.ClientSession(timeout=timeout) as session:
|
||||
async with session.get(url) as response:
|
||||
response.raise_for_status()
|
||||
return await response.content.read(limit + 1)
|
||||
# limit + 1: an over-limit body must stay over-limit so
|
||||
# ingest_bytes rejects it instead of caching it truncated
|
||||
return await read_capped(response, limit + 1)
|
||||
|
||||
def data_url(self, sha256: str, ext: str) -> Optional[str]:
|
||||
path = self._path(sha256, ext)
|
||||
|
||||
@@ -0,0 +1,391 @@
|
||||
"""News digest fetcher (SPEC-013, FDB-012 news rewrite).
|
||||
|
||||
Replaces the broken pre-1.0-openai `news_feed.py`. Fetches configured
|
||||
RSS/Atom feeds (stdlib, no feedparser dep), builds a compact sanitized
|
||||
headline digest, and writes it to the `{news}` file the responder
|
||||
injects (AIResponder.message). Feeds are external input: titles are
|
||||
sanitized (SAF-03) and each feed URL is SSRF-guarded before fetching.
|
||||
|
||||
CLI: python -m fjerkroa_bot.news --config kroa.toml
|
||||
"""
|
||||
|
||||
import argparse
|
||||
import logging
|
||||
import re
|
||||
import sys
|
||||
import time
|
||||
from html import unescape
|
||||
from typing import Any, Dict, List, Optional, Tuple
|
||||
|
||||
import defusedxml.ElementTree as ElementTree # hardened XML: feeds are untrusted (XXE/billion-laughs)
|
||||
|
||||
from .ai_responder import sanitize_external_text
|
||||
|
||||
DEFAULT_PER_FEED = 3
|
||||
DEFAULT_MAX_ITEMS = 15
|
||||
DEFAULT_SUMMARY_CHARS = 200
|
||||
DEFAULT_NEWS_KEEP = 400
|
||||
FETCH_TIMEOUT_S = 15
|
||||
_ATOM = "{http://www.w3.org/2005/Atom}"
|
||||
_TAG_RE = re.compile(r"<[^>]+>")
|
||||
|
||||
|
||||
def _clean_summary(raw: str, max_len: int = 300) -> str:
|
||||
"""Strip HTML, unescape entities, collapse whitespace (feed descriptions are often HTML)."""
|
||||
text = unescape(_TAG_RE.sub(" ", raw or ""))
|
||||
return re.sub(r"\s+", " ", text).strip()[:max_len]
|
||||
|
||||
|
||||
def parse_feed(data: bytes, source: str = "") -> List[Dict[str, str]]:
|
||||
"""Parse RSS or Atom bytes into [{title, link, source}] (tolerant)."""
|
||||
try:
|
||||
root = ElementTree.fromstring(data)
|
||||
except Exception as err:
|
||||
# malformed XML or a blocked entity/DTD attack — tolerate, never raise (NEWS-01)
|
||||
logging.warning(f"news: unparseable/unsafe feed {source!r}: {err!r}")
|
||||
return []
|
||||
items: List[Dict[str, str]] = []
|
||||
# RSS: <rss><channel><item><title/><link/><description/>
|
||||
for item in root.iter("item"):
|
||||
title = (item.findtext("title") or "").strip()
|
||||
link = (item.findtext("link") or "").strip()
|
||||
summary = _clean_summary(item.findtext("description") or "")
|
||||
if title:
|
||||
items.append({"title": title, "link": link, "source": source, "summary": summary})
|
||||
# Atom: <feed><entry><title/><link href=/><summary|content/>
|
||||
for entry in root.iter(f"{_ATOM}entry"):
|
||||
title = (entry.findtext(f"{_ATOM}title") or "").strip()
|
||||
link_el = entry.find(f"{_ATOM}link")
|
||||
link = link_el.get("href", "") if link_el is not None else ""
|
||||
summary = _clean_summary(entry.findtext(f"{_ATOM}summary") or entry.findtext(f"{_ATOM}content") or "")
|
||||
if title:
|
||||
items.append({"title": title, "link": link, "source": source, "summary": summary})
|
||||
return items
|
||||
|
||||
|
||||
def render_digest(items: List[Dict[str, str]], max_items: int = DEFAULT_MAX_ITEMS, summary_chars: int = DEFAULT_SUMMARY_CHARS) -> str:
|
||||
"""Compact sanitized digest for the {news} prompt slot (title + short summary + link)."""
|
||||
lines = []
|
||||
for item in items[:max_items]:
|
||||
title = sanitize_external_text(item["title"], 200)
|
||||
source = item.get("source", "")
|
||||
link = item.get("link", "")
|
||||
summary = sanitize_external_text(item.get("summary", ""), summary_chars) if summary_chars else ""
|
||||
prefix = f"[{source}] " if source else ""
|
||||
line = f"- {prefix}{title}"
|
||||
if summary:
|
||||
line += f" — {summary}"
|
||||
if link:
|
||||
line += f" ({link})"
|
||||
lines.append(line)
|
||||
return "\n".join(lines)
|
||||
|
||||
|
||||
class NewsFetcher:
|
||||
def __init__(self, guard, fetch_bytes) -> None:
|
||||
# injected so tests need no network; production wires aiohttp + guard_url
|
||||
self._guard = guard
|
||||
self._fetch_bytes = fetch_bytes
|
||||
|
||||
async def collect(self, feeds: List[Tuple[str, str]], per_feed: int) -> List[Dict[str, str]]:
|
||||
"""feeds = [(url, label)]; returns deduped items, order preserved."""
|
||||
seen = set()
|
||||
out: List[Dict[str, str]] = []
|
||||
for url, label in feeds:
|
||||
reason = self._guard(url)
|
||||
if reason:
|
||||
logging.warning(f"news: skipping feed {label} — {reason}")
|
||||
continue
|
||||
try:
|
||||
data = await self._fetch_bytes(url)
|
||||
except Exception as err:
|
||||
logging.warning(f"news: fetch failed for {label}: {repr(err)}")
|
||||
continue
|
||||
for item in parse_feed(data, label)[:per_feed]:
|
||||
key = item["title"]
|
||||
if key not in seen:
|
||||
seen.add(key)
|
||||
out.append(item)
|
||||
return out
|
||||
|
||||
|
||||
DEFAULT_SEEN_CAP = 5000
|
||||
DEFAULT_POST_PER_FEED = 5
|
||||
DEFAULT_POST_MAX_PER_RUN = 8
|
||||
|
||||
|
||||
def item_key(item: Dict[str, str]) -> str:
|
||||
return item.get("link") or item.get("title") or ""
|
||||
|
||||
|
||||
class NewsPoster:
|
||||
"""Post NEW feed items to Discord channel webhooks (ggg model, SPEC-013 NEWS-04..06)."""
|
||||
|
||||
def __init__(self, guard, fetch_bytes, post_webhook, store: Any = None) -> None:
|
||||
self._guard = guard
|
||||
self._fetch_bytes = fetch_bytes
|
||||
self._post_webhook = post_webhook
|
||||
self._store = store
|
||||
|
||||
async def run_post(
|
||||
self,
|
||||
feeds: List[Tuple[str, str, str]],
|
||||
webhooks: Dict[str, str],
|
||||
seen: set,
|
||||
per_feed: int,
|
||||
max_per_run: int,
|
||||
seed_only: bool,
|
||||
keep: int = DEFAULT_NEWS_KEEP,
|
||||
) -> Tuple[int, set]:
|
||||
"""Returns (posted_count, updated_seen). seed_only marks new items seen without posting."""
|
||||
posted = 0
|
||||
harvested: List[Dict[str, str]] = []
|
||||
for url, label, channel in feeds:
|
||||
reason = self._guard(url)
|
||||
if reason:
|
||||
logging.warning(f"news-post: skipping feed {label} — {reason}")
|
||||
continue
|
||||
try:
|
||||
data = await self._fetch_bytes(url)
|
||||
except Exception as err:
|
||||
logging.warning(f"news-post: fetch failed for {label}: {repr(err)}")
|
||||
continue
|
||||
for item in parse_feed(data, label)[:per_feed]:
|
||||
harvested.append(item) # NEWS-09: everything parsed feeds the searchable store
|
||||
key = item_key(item)
|
||||
if not key or key in seen:
|
||||
continue
|
||||
seen.add(key)
|
||||
may_post = not seed_only and posted < max_per_run
|
||||
if may_post and await self._deliver(item, label, channel, webhooks):
|
||||
posted += 1
|
||||
if self._store is not None and harvested:
|
||||
self._store.add_news_items(harvested)
|
||||
self._store.prune_news(keep)
|
||||
return posted, seen
|
||||
|
||||
async def _deliver(self, item: Dict[str, str], label: str, channel: str, webhooks: Dict[str, str]) -> bool:
|
||||
hook = webhooks.get(channel)
|
||||
if not hook:
|
||||
logging.warning(f"news-post: no webhook for channel {channel!r} ({label})")
|
||||
return False
|
||||
title = sanitize_external_text(item["title"], 300)
|
||||
link = item.get("link", "")
|
||||
content = f"**[{label}]** {title}" + (f"\n{link}" if link else "")
|
||||
try:
|
||||
await self._post_webhook(hook, content)
|
||||
return True
|
||||
except Exception as err:
|
||||
logging.warning(f"news-post: webhook post failed ({label}): {repr(err)}")
|
||||
return False
|
||||
|
||||
|
||||
# --- news memory + on-demand retrieval tool (SPEC-013 NEWS-09..12) ---
|
||||
|
||||
GET_NEWS_TOOL = {
|
||||
"name": "get_news",
|
||||
"description": "Fetch recent real-world news the bot has collected from its RSS feeds (local, national, world, sport, "
|
||||
"culture). Use when someone asks what is new or what is happening, optionally about a topic or from a particular "
|
||||
"source. Returns headlines with a short summary and a link to read more.",
|
||||
"parameters": {
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"topic": {"type": "string", "description": "Optional keywords to filter by, e.g. 'Nordland', 'football', 'weather'."},
|
||||
"source": {"type": "string", "description": "Optional source label, e.g. 'NRK', 'Aftenposten', 'Verden', 'Sport'."},
|
||||
"limit": {"type": "integer", "description": "How many items to return (default 10, max 30)."},
|
||||
},
|
||||
"required": [],
|
||||
},
|
||||
}
|
||||
|
||||
|
||||
def _news_terms(topic: Optional[str]) -> List[str]:
|
||||
return [t for t in re.split(r"\W+", (topic or "").lower()) if len(t) > 1][:5]
|
||||
|
||||
|
||||
def query_news(
|
||||
store: Any, topic: Optional[str] = None, source: Optional[str] = None, limit: int = 10, summary_chars: int = DEFAULT_SUMMARY_CHARS
|
||||
) -> Dict[str, Any]:
|
||||
"""Retrieve stored news for the get_news tool: term/source-filtered, sanitized (NEWS-11)."""
|
||||
if store is None:
|
||||
return {"error": "news store unavailable"}
|
||||
limit = max(1, min(int(limit or 10), 30))
|
||||
src = (str(source).strip() or None) if source else None
|
||||
terms = _news_terms(topic)
|
||||
try:
|
||||
rows = store.search_news(terms, limit, src) if terms else store.recent_news(limit, src)
|
||||
except Exception as err:
|
||||
logging.warning(f"news: query failed: {err!r}")
|
||||
return {"error": "news lookup failed"}
|
||||
results = [
|
||||
{
|
||||
"source": row.get("source", ""),
|
||||
"title": sanitize_external_text(row.get("title", ""), 200),
|
||||
"summary": sanitize_external_text(row.get("summary", ""), summary_chars),
|
||||
"link": row.get("link", ""),
|
||||
}
|
||||
for row in rows
|
||||
]
|
||||
return {"topic": topic or "", "source": src or "", "results": results}
|
||||
|
||||
|
||||
def _open_store(config: Dict[str, Any]) -> Any:
|
||||
directory = config.get("history-directory")
|
||||
if not directory:
|
||||
return None
|
||||
from pathlib import Path
|
||||
|
||||
from .persistence import PersistentStore
|
||||
|
||||
return PersistentStore(Path(str(directory)).expanduser() / "bot.db")
|
||||
|
||||
|
||||
def persist_news(config: Dict[str, Any], items: List[Dict[str, str]]) -> int:
|
||||
"""Upsert fetched items into the news store, prune to the rolling window (NEWS-09)."""
|
||||
store = _open_store(config)
|
||||
if store is None or not items:
|
||||
return 0
|
||||
added = store.add_news_items(items)
|
||||
store.prune_news(int(config.get("news-keep", DEFAULT_NEWS_KEEP)))
|
||||
return added
|
||||
|
||||
|
||||
def load_seen(path: str) -> Tuple[set, bool]:
|
||||
"""(seen-set, existed). Missing/broken state -> empty set, existed=False (seed run)."""
|
||||
import json
|
||||
import os
|
||||
|
||||
if not os.path.exists(path):
|
||||
return set(), False
|
||||
try:
|
||||
with open(path, encoding="utf-8") as fd:
|
||||
return set(json.load(fd)), True
|
||||
except Exception as err:
|
||||
logging.warning(f"news-post: unreadable state {path}: {err!r} — reseeding")
|
||||
return set(), False
|
||||
|
||||
|
||||
def save_seen(path: str, seen: set, cap: int = DEFAULT_SEEN_CAP) -> None:
|
||||
import json
|
||||
|
||||
# keep the newest `cap` keys (insertion order preserved by Python sets? no — use a bounded slice)
|
||||
keys = list(seen)[-cap:]
|
||||
with open(path, "w", encoding="utf-8") as fd:
|
||||
json.dump(keys, fd)
|
||||
|
||||
|
||||
def _post_feeds_from_config(config: Dict[str, Any]) -> List[Tuple[str, str, str]]:
|
||||
feeds = []
|
||||
for entry in config.get("news-post-feeds", []):
|
||||
if isinstance(entry, (list, tuple)) and len(entry) >= 3:
|
||||
feeds.append((str(entry[0]), str(entry[1]), str(entry[2])))
|
||||
return feeds
|
||||
|
||||
|
||||
def _feeds_from_config(config: Dict[str, Any]) -> List[Tuple[str, str]]:
|
||||
"""news-feeds = [["url", "label"], ...] or ["url", ...]."""
|
||||
feeds = []
|
||||
for entry in config.get("news-feeds", []):
|
||||
if isinstance(entry, (list, tuple)):
|
||||
feeds.append((str(entry[0]), str(entry[1]) if len(entry) > 1 else ""))
|
||||
else:
|
||||
feeds.append((str(entry), ""))
|
||||
return feeds
|
||||
|
||||
|
||||
async def _aiohttp_fetch(url: str) -> bytes:
|
||||
import aiohttp
|
||||
|
||||
from .httpread import read_capped
|
||||
|
||||
timeout = aiohttp.ClientTimeout(total=FETCH_TIMEOUT_S)
|
||||
async with aiohttp.ClientSession(timeout=timeout, headers={"User-Agent": "Mozilla/5.0 (compatible; FjerkroaBot-news/1.0)"}) as session:
|
||||
async with session.get(url) as response:
|
||||
response.raise_for_status()
|
||||
return await read_capped(response, 4 * 1024 * 1024)
|
||||
|
||||
|
||||
async def _aiohttp_post(hook: str, content: str) -> None:
|
||||
import aiohttp
|
||||
|
||||
timeout = aiohttp.ClientTimeout(total=FETCH_TIMEOUT_S)
|
||||
async with aiohttp.ClientSession(timeout=timeout) as session:
|
||||
# allowed_mentions none: a headline can never ping the channel (SAF-02 spirit)
|
||||
payload = {"content": content[:2000], "allowed_mentions": {"parse": []}}
|
||||
async with session.post(hook, json=payload) as response:
|
||||
response.raise_for_status()
|
||||
|
||||
|
||||
async def run_post(config: Dict[str, Any]) -> int:
|
||||
"""Webhook-posting mode (ggg): post new items to channels. Returns posted count."""
|
||||
from .url_reader import guard_url
|
||||
|
||||
webhooks = dict(config.get("news-post-webhooks", {}))
|
||||
feeds = _post_feeds_from_config(config)
|
||||
state_path = config.get("news-post-state", "news_state.json")
|
||||
if not webhooks or not feeds:
|
||||
logging.error("news-post: need news-post-webhooks and news-post-feeds")
|
||||
return 0
|
||||
seen, existed = load_seen(state_path)
|
||||
poster = NewsPoster(guard_url, _aiohttp_fetch, _aiohttp_post, store=_open_store(config))
|
||||
posted, seen = await poster.run_post(
|
||||
feeds,
|
||||
webhooks,
|
||||
seen,
|
||||
int(config.get("news-post-per-feed", DEFAULT_POST_PER_FEED)),
|
||||
int(config.get("news-post-max-per-run", DEFAULT_POST_MAX_PER_RUN)),
|
||||
seed_only=not existed, # first run seeds without flooding the channels
|
||||
keep=int(config.get("news-keep", DEFAULT_NEWS_KEEP)),
|
||||
)
|
||||
save_seen(state_path, seen, int(config.get("news-post-seen-cap", DEFAULT_SEEN_CAP)))
|
||||
logging.info(f"news-post: posted {posted} item(s)" + (" (seed run — nothing posted)" if not existed else ""))
|
||||
return posted
|
||||
|
||||
|
||||
async def run(config: Dict[str, Any]) -> Optional[str]:
|
||||
from .url_reader import guard_url
|
||||
|
||||
out_path = config.get("news")
|
||||
if not out_path:
|
||||
logging.error("news: no `news` output path in config")
|
||||
return None
|
||||
feeds = _feeds_from_config(config)
|
||||
if not feeds:
|
||||
logging.error("news: no `news-feeds` configured")
|
||||
return None
|
||||
fetcher = NewsFetcher(guard_url, _aiohttp_fetch)
|
||||
items = await fetcher.collect(feeds, int(config.get("news-per-feed", DEFAULT_PER_FEED)))
|
||||
persist_news(config, items) # NEWS-09: feed the searchable rolling store for get_news
|
||||
digest = render_digest(
|
||||
items, int(config.get("news-max-items", DEFAULT_MAX_ITEMS)), int(config.get("news-summary-chars", DEFAULT_SUMMARY_CHARS))
|
||||
)
|
||||
header = f"News as of {time.strftime('%Y-%m-%d %H:%M UTC', time.gmtime())}:\n"
|
||||
with open(out_path, "w", encoding="utf-8") as fd:
|
||||
fd.write(header + digest + "\n")
|
||||
logging.info(f"news: wrote {len(items)} items to {out_path}")
|
||||
return out_path
|
||||
|
||||
|
||||
def main() -> int:
|
||||
import asyncio
|
||||
|
||||
import tomlkit
|
||||
|
||||
logging.basicConfig(level=logging.INFO, format="%(asctime)s [%(levelname)s] %(message)s")
|
||||
parser = argparse.ArgumentParser(
|
||||
description="Fetch RSS/Atom feeds: --post to channel webhooks (ggg) or default {news} digest file (kroa)"
|
||||
)
|
||||
parser.add_argument("--config", required=True)
|
||||
parser.add_argument("--post", action="store_true", help="webhook-posting mode (post new items to Discord channels)")
|
||||
args = parser.parse_args()
|
||||
with open(args.config, encoding="utf-8") as fd:
|
||||
config = tomlkit.load(fd)
|
||||
if args.post:
|
||||
asyncio.run(run_post(config))
|
||||
return 0
|
||||
result = asyncio.run(run(config))
|
||||
return 0 if result else 1
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
sys.exit(main())
|
||||
@@ -9,8 +9,12 @@ from typing import Any, Dict, List, Optional, Tuple
|
||||
import openai
|
||||
|
||||
from .ai_responder import AIResponder, exponential_backoff, sanitize_external_text
|
||||
from .codex import CODEX_SEARCH_TOOL
|
||||
from .codex import DEFAULT_LIMIT as CODEX_DEFAULT_LIMIT
|
||||
from .codex import CodexSearch
|
||||
from .igdblib import IGDBQuery
|
||||
from .leonardo_draw import LeonardoAIDrawMixIn
|
||||
from .news import GET_NEWS_TOOL, query_news
|
||||
from .quota import QuotaLedger
|
||||
from .url_reader import FETCH_URL_TOOL, URLReader
|
||||
|
||||
@@ -160,6 +164,8 @@ class OpenAIResponder(AIResponder, LeonardoAIDrawMixIn):
|
||||
|
||||
# URL reading tool (SPEC-011); shares the image cache for page images
|
||||
self.url_reader = URLReader(lambda: self.config, self.image_cache)
|
||||
# Codex Mechanicus search (SPEC-014); Luma's own archive at binaric.tech
|
||||
self.codex = CodexSearch(lambda: self.config)
|
||||
|
||||
def _available_tools(self) -> List[Dict[str, Any]]:
|
||||
"""Assemble the function-tool list from every enabled provider (URL-01)."""
|
||||
@@ -173,16 +179,34 @@ class OpenAIResponder(AIResponder, LeonardoAIDrawMixIn):
|
||||
logging.warning(f"Error setting up IGDB functions: {err}")
|
||||
if self.url_reader.enabled():
|
||||
functions.append(FETCH_URL_TOOL)
|
||||
if self.codex.enabled(): # CDX-01
|
||||
functions.append(CODEX_SEARCH_TOOL)
|
||||
if self.config.get("enable-news-tool", False) and self.store is not None: # NEWS-10
|
||||
functions.append(GET_NEWS_TOOL)
|
||||
return functions
|
||||
|
||||
async def _dispatch_tool(self, name: str, args: Dict[str, Any], author: str) -> Any:
|
||||
"""Route a tool call to its provider (IGDB or URL reader)."""
|
||||
"""Route a tool call to its provider (IGDB, URL reader, or codex)."""
|
||||
if name == "fetch_url":
|
||||
per_user_cap = int(self.config.get("url-daily-per-user", 20))
|
||||
if self.ledger._get(f"url-fetch:{author}") >= per_user_cap: # URL-07
|
||||
return {"error": "daily URL fetch limit reached"}
|
||||
self.ledger._add(f"url-fetch:{author}", 1)
|
||||
return await self.url_reader.fetch(str(args.get("url", "")), self.channel, author or "user")
|
||||
if name == "codex_search":
|
||||
per_user_cap = int(self.config.get("codex-daily-per-user", 50))
|
||||
if self.ledger._get(f"codex:{author}") >= per_user_cap: # CDX-06
|
||||
return {"error": "daily codex search limit reached"}
|
||||
self.ledger._add(f"codex:{author}", 1)
|
||||
limit = int(self.config.get("codex-limit", CODEX_DEFAULT_LIMIT))
|
||||
return await self.codex.search(str(args.get("query", "")), str(args.get("lang", "en")), limit)
|
||||
if name == "get_news":
|
||||
per_user_cap = int(self.config.get("news-daily-per-user", 30))
|
||||
if self.ledger._get(f"news:{author}") >= per_user_cap: # NEWS-12
|
||||
return {"error": "daily news lookup limit reached"}
|
||||
self.ledger._add(f"news:{author}", 1)
|
||||
summary_chars = int(self.config.get("news-summary-chars", 200))
|
||||
return query_news(self.store, args.get("topic"), args.get("source"), args.get("limit", 10), summary_chars)
|
||||
return await self._execute_igdb_function(name, args)
|
||||
|
||||
async def draw_openai(self, description: str, count: int = 1) -> List[BytesIO]:
|
||||
|
||||
@@ -13,7 +13,7 @@ from contextlib import closing
|
||||
from pathlib import Path
|
||||
from typing import Any, Dict, List, Optional
|
||||
|
||||
SCHEMA_VERSION = 5
|
||||
SCHEMA_VERSION = 6
|
||||
|
||||
|
||||
class PersistentStore:
|
||||
@@ -72,6 +72,13 @@ class PersistentStore:
|
||||
" due_at TEXT NOT NULL, payload TEXT NOT NULL, state TEXT NOT NULL DEFAULT 'queued',"
|
||||
" created_at TEXT NOT NULL DEFAULT (datetime('now')), executed_at TEXT)"
|
||||
)
|
||||
if version < 6:
|
||||
# News memory (SPEC-013 NEWS-09): deduped rolling store of fetched items
|
||||
conn.execute(
|
||||
"CREATE TABLE IF NOT EXISTS news (id INTEGER PRIMARY KEY, dedup_key TEXT UNIQUE NOT NULL,"
|
||||
" source TEXT NOT NULL DEFAULT '', title TEXT NOT NULL, link TEXT NOT NULL DEFAULT '',"
|
||||
" summary TEXT NOT NULL DEFAULT '', first_seen TEXT NOT NULL DEFAULT (datetime('now')))"
|
||||
)
|
||||
if version < SCHEMA_VERSION:
|
||||
conn.execute(f"PRAGMA user_version = {SCHEMA_VERSION}")
|
||||
os.chmod(self.db_path, 0o600) # conversation data (PER-04)
|
||||
@@ -115,6 +122,64 @@ class PersistentStore:
|
||||
row = conn.execute("SELECT value FROM usage WHERE day = ? AND key = ?", (day, key)).fetchone()
|
||||
return float(row[0]) if row else 0.0
|
||||
|
||||
# --- news memory (SPEC-013 NEWS-09..12) ---
|
||||
|
||||
def add_news_items(self, items: List[Dict[str, Any]]) -> int:
|
||||
"""Insert deduped news rows (by link or title); returns how many were new (NEWS-09)."""
|
||||
added = 0
|
||||
with closing(self._connect()) as conn, conn:
|
||||
for item in items:
|
||||
title = str(item.get("title") or "").strip()
|
||||
key = (str(item.get("link") or "").strip()) or title
|
||||
if not title or not key:
|
||||
continue
|
||||
cursor = conn.execute(
|
||||
"INSERT OR IGNORE INTO news (dedup_key, source, title, link, summary) VALUES (?, ?, ?, ?, ?)",
|
||||
(key, str(item.get("source") or ""), title, str(item.get("link") or ""), str(item.get("summary") or "")),
|
||||
)
|
||||
added += cursor.rowcount
|
||||
return added
|
||||
|
||||
def recent_news(self, limit: int = 20, source: Optional[str] = None) -> List[Dict[str, Any]]:
|
||||
sql = "SELECT source, title, link, summary FROM news"
|
||||
params: List[Any] = []
|
||||
if source:
|
||||
sql += " WHERE source = ?"
|
||||
params.append(source)
|
||||
sql += " ORDER BY id DESC LIMIT ?"
|
||||
params.append(int(limit))
|
||||
with closing(self._connect()) as conn:
|
||||
rows = conn.execute(sql, params).fetchall()
|
||||
return [{"source": r[0], "title": r[1], "link": r[2], "summary": r[3]} for r in rows]
|
||||
|
||||
def search_news(self, terms: List[str], limit: int = 20, source: Optional[str] = None) -> List[Dict[str, Any]]:
|
||||
"""Rows where every term appears in title or summary; optional source filter (NEWS-11)."""
|
||||
params: List[Any] = []
|
||||
clauses = []
|
||||
for term in terms:
|
||||
clauses.append("(title LIKE ? OR summary LIKE ? OR source LIKE ?)")
|
||||
like = f"%{term}%"
|
||||
params += [like, like, like]
|
||||
where = " AND ".join(clauses) if clauses else "1=1"
|
||||
if source:
|
||||
where = f"({where}) AND source = ?"
|
||||
params.append(source)
|
||||
params.append(int(limit))
|
||||
sql = f"SELECT source, title, link, summary FROM news WHERE {where} ORDER BY id DESC LIMIT ?" # nosec B608 - fixed templates; values parameterised
|
||||
with closing(self._connect()) as conn:
|
||||
rows = conn.execute(sql, params).fetchall()
|
||||
return [{"source": r[0], "title": r[1], "link": r[2], "summary": r[3]} for r in rows]
|
||||
|
||||
def prune_news(self, keep: int) -> int:
|
||||
"""Keep the newest `keep` rows, delete the rest (rolling window, NEWS-09)."""
|
||||
with closing(self._connect()) as conn, conn:
|
||||
cursor = conn.execute("DELETE FROM news WHERE id NOT IN (SELECT id FROM news ORDER BY id DESC LIMIT ?)", (int(keep),))
|
||||
return cursor.rowcount
|
||||
|
||||
def news_count(self) -> int:
|
||||
with closing(self._connect()) as conn:
|
||||
return int(conn.execute("SELECT COUNT(*) FROM news").fetchone()[0])
|
||||
|
||||
def delete_history_of_user(self, user: str) -> int:
|
||||
"""Remove persisted rows carrying this user's messages (SAF-08)."""
|
||||
with closing(self._connect()) as conn, conn:
|
||||
|
||||
+32
-12
@@ -18,6 +18,7 @@ from urllib.parse import urljoin, urlparse
|
||||
import aiohttp
|
||||
|
||||
from .ai_responder import sanitize_external_text
|
||||
from .httpread import read_capped
|
||||
|
||||
DEFAULT_MAX_BYTES = 2 * 1024 * 1024
|
||||
DEFAULT_MAX_CHARS = 6000
|
||||
@@ -37,6 +38,9 @@ FETCH_URL_TOOL = {
|
||||
}
|
||||
|
||||
|
||||
_META_REFRESH_URL = re.compile(r"url\s*=\s*['\"]?([^'\";\s]+)", re.I)
|
||||
|
||||
|
||||
class _Extractor(HTMLParser):
|
||||
def __init__(self) -> None:
|
||||
super().__init__()
|
||||
@@ -44,6 +48,7 @@ class _Extractor(HTMLParser):
|
||||
self.parts: List[str] = []
|
||||
self.images: List[str] = []
|
||||
self.og_image: Optional[str] = None
|
||||
self.refresh_url: Optional[str] = None
|
||||
|
||||
def handle_starttag(self, tag: str, attrs) -> None:
|
||||
if tag in ("script", "style", "noscript", "svg"):
|
||||
@@ -54,6 +59,12 @@ class _Extractor(HTMLParser):
|
||||
self.images.append(src)
|
||||
if tag == "meta" and attr.get("property") == "og:image" and attr.get("content"):
|
||||
self.og_image = attr["content"]
|
||||
# meta-refresh redirect (link shorteners, getnews stubs) — URL-04
|
||||
content = attr.get("content")
|
||||
if tag == "meta" and (attr.get("http-equiv") or "").lower() == "refresh" and content:
|
||||
match = _META_REFRESH_URL.search(content)
|
||||
if match and self.refresh_url is None:
|
||||
self.refresh_url = match.group(1)
|
||||
|
||||
def handle_endtag(self, tag: str) -> None:
|
||||
if tag in ("script", "style", "noscript", "svg") and self._skip > 0:
|
||||
@@ -115,7 +126,7 @@ class URLReader:
|
||||
current = urljoin(current, response.headers["Location"])
|
||||
continue
|
||||
response.raise_for_status()
|
||||
return str(response.url), await response.content.read(max_bytes + 1)
|
||||
return str(response.url), await read_capped(response, max_bytes)
|
||||
raise ValueError("too many redirects")
|
||||
|
||||
async def fetch(self, url: str, channel: str, user: str) -> Dict[str, Any]:
|
||||
@@ -125,29 +136,38 @@ class URLReader:
|
||||
try:
|
||||
async with aiohttp.ClientSession(timeout=timeout, headers={"User-Agent": "FjerkroaBot/1.0"}) as session:
|
||||
final_url, body = await self._get(session, url, max_bytes)
|
||||
# follow a meta-refresh redirect (link shorteners / getnews stubs), re-guarded — URL-04
|
||||
for _ in range(2):
|
||||
extractor = self._extract(body.decode("utf-8", "ignore"))
|
||||
if not extractor.refresh_url:
|
||||
break
|
||||
target = urljoin(final_url, extractor.refresh_url)
|
||||
if guard_url(target) is not None or target == final_url:
|
||||
break
|
||||
logging.info(f"url reader: following meta-refresh -> {target}")
|
||||
final_url, body = await self._get(session, target, max_bytes)
|
||||
except Exception as err:
|
||||
return {"error": str(err)}
|
||||
text = self._to_text(body.decode("utf-8", "ignore"))
|
||||
clean = sanitize_external_text(text, int(config.get("url-max-chars", DEFAULT_MAX_CHARS)))
|
||||
images = await self._ingest_images(body.decode("utf-8", "ignore"), final_url, channel, user)
|
||||
html = body.decode("utf-8", "ignore")
|
||||
clean = sanitize_external_text(self._to_text(html), int(config.get("url-max-chars", DEFAULT_MAX_CHARS)))
|
||||
images = await self._ingest_images(html, final_url, channel, user)
|
||||
return {"url": final_url, "text": clean, "images_cached": images}
|
||||
|
||||
def _to_text(self, html: str) -> str:
|
||||
def _extract(self, html: str) -> "_Extractor":
|
||||
extractor = _Extractor()
|
||||
try:
|
||||
extractor.feed(html)
|
||||
except Exception as err:
|
||||
logging.debug(f"html parse (text) failed: {err!r}")
|
||||
return re.sub(r"\s+\n", "\n", " ".join(extractor.parts))
|
||||
logging.debug(f"html parse failed: {err!r}")
|
||||
return extractor
|
||||
|
||||
def _to_text(self, html: str) -> str:
|
||||
return re.sub(r"\s+\n", "\n", " ".join(self._extract(html).parts))
|
||||
|
||||
async def _ingest_images(self, html: str, base_url: str, channel: str, user: str) -> int:
|
||||
if self.image_cache is None:
|
||||
return 0
|
||||
extractor = _Extractor()
|
||||
try:
|
||||
extractor.feed(html)
|
||||
except Exception as err:
|
||||
logging.debug(f"html parse (images) failed: {err!r}")
|
||||
extractor = self._extract(html)
|
||||
candidates = ([extractor.og_image] if extractor.og_image else []) + extractor.images
|
||||
limit = int(self._config().get("url-max-images", DEFAULT_MAX_IMAGES))
|
||||
cached = 0
|
||||
|
||||
@@ -13,3 +13,4 @@ with date + result.
|
||||
| DEP-05 | 2026-07-13 | Live-verified: kroa deploy attempt ~15h Oslo refused without DEPLOY_FORCE=1. |
|
||||
| DEP-06 | 2026-07-13 | Rollback documented (older tag + db backup restore); live drill pending — next release. |
|
||||
| OPS-15 | 2026-07-13 | Backup cron installed on both hosts (daily 03:17 UTC → ~/backups/<bot>/, keep 14); first snapshots written + verified 0600 (kroa 10965 B, luma 25871 B). |
|
||||
| CDX-07 | 2026-07-13 | Pending live verify on ggg after v3.8.0 deploy: persona grounding + codex_search returns binaric.tech inscriptions with links. |
|
||||
|
||||
@@ -15,6 +15,7 @@ dependencies = [
|
||||
"tomlkit>=0.13",
|
||||
"watchdog>=6",
|
||||
"requests>=2.32",
|
||||
"defusedxml>=0.7",
|
||||
]
|
||||
|
||||
[project.scripts]
|
||||
|
||||
@@ -73,3 +73,15 @@ per-channel) — without it, `!bot unpin <id>` required guessing ids.
|
||||
`!bot spend` answers in the staff channel with today's estimated
|
||||
spend in USD, token and image counts, and the configured budget.
|
||||
Management sees the cost, not just the cap.
|
||||
|
||||
### OPS-17 — Help is complete and context-aware (coverage: test)
|
||||
|
||||
Help reflects where each command actually works, because not every
|
||||
command is allowed everywhere. `!help` answers in any channel and
|
||||
lists only the commands usable there: in a normal channel the
|
||||
everyone-commands (`!help`, `!forgetme`, `!privacy`, `!wichtel`); in
|
||||
the staff channel it additionally lists the operator commands grouped
|
||||
by purpose (control, cost, memory, tasks). `!bot help` — and any
|
||||
unrecognised `!bot` command — answers with that same full staff help,
|
||||
so the listing is exhaustive rather than the old hand-maintained
|
||||
partial line. Help works even while the bot is paused.
|
||||
|
||||
@@ -31,7 +31,10 @@ refused without DNS.
|
||||
|
||||
Redirects are followed manually; each hop's target passes URL-02 and
|
||||
URL-03 again. A public URL that 302-redirects to `localhost` or an
|
||||
internal IP is refused at the redirect, not fetched.
|
||||
internal IP is refused at the redirect, not fetched. **HTML
|
||||
meta-refresh** redirects (link shorteners, the old getnews stubs) are
|
||||
also followed — the target is SSRF-re-guarded and fetched, so the
|
||||
reader returns the real article, not the "Redirecting…" stub.
|
||||
|
||||
### URL-05 — Fetched text is bounded and sanitized (coverage: test)
|
||||
|
||||
|
||||
@@ -0,0 +1,100 @@
|
||||
# SPEC-013 — News digest
|
||||
|
||||
Replaces the broken pre-1.0-openai `news_feed.py`. A CLI
|
||||
(`python -m fjerkroa_bot.news --config <cfg>`) fetches the
|
||||
`news-feeds` and writes a compact digest to the `news` file that
|
||||
`AIResponder.message` injects into the `{news}` slot. Feeds are
|
||||
external input and operator-configured.
|
||||
|
||||
### NEWS-01 — RSS and Atom parse to items (coverage: test)
|
||||
|
||||
`parse_feed(bytes, label)` extracts `{title, link, source}` from both
|
||||
RSS (`<item>`) and Atom (`<entry>`) documents, tolerates malformed
|
||||
XML (returns an empty list, logs), and never raises.
|
||||
|
||||
### NEWS-02 — Digest is sanitized and bounded (coverage: test)
|
||||
|
||||
`render_digest` caps at `news-max-items`, and every headline passes
|
||||
`sanitize_external_text` (SAF-03) — a feed cannot inject `@everyone`
|
||||
or control characters into the prompt via a headline.
|
||||
|
||||
### NEWS-03 — Feeds are SSRF-guarded and deduped (coverage: test)
|
||||
|
||||
`NewsFetcher.collect` skips any feed URL the SSRF guard rejects,
|
||||
skips feeds that fail to fetch (one bad feed never sinks the run),
|
||||
and drops duplicate headlines across feeds.
|
||||
|
||||
## Webhook posting (ggg model)
|
||||
|
||||
`--post` mode fetches feeds mapped to channels and posts NEW items to
|
||||
the channel's Discord webhook — replacing the py3.8 `getnews.py`
|
||||
(dead play3 feed, 35 MB substring-scan state file, HTML-redirect
|
||||
cruft). Config: `news-post-feeds = [[url, label, channel], …]`,
|
||||
`news-post-webhooks = {channel = url}`, `news-post-state`.
|
||||
|
||||
### NEWS-04 — Only unseen items post, then are marked seen (coverage: test)
|
||||
|
||||
`NewsPoster.run_post` posts each item whose key (link, else title) is
|
||||
not in the seen-set, adds it to the set, and posts to the mapped
|
||||
channel's webhook. Re-runs over the same feed post nothing new.
|
||||
|
||||
### NEWS-05 — First run seeds without flooding (coverage: test)
|
||||
|
||||
With no prior state file (`seed_only`), every current item is marked
|
||||
seen but nothing is posted — migrating off getnews.py never dumps a
|
||||
backlog into the channels. `news-post-max-per-run` caps steady-state
|
||||
posts per run.
|
||||
|
||||
### NEWS-06 — Post failures and bad channels are survived (coverage: test)
|
||||
|
||||
A feed the SSRF guard rejects, a feed that fails to fetch, an item
|
||||
whose channel has no configured webhook, and a webhook POST that
|
||||
raises are each logged and skipped — one failure never sinks the
|
||||
run, and the seen-set still advances for successfully-processed
|
||||
items.
|
||||
|
||||
### NEWS-07 — Item summaries are extracted (coverage: test)
|
||||
|
||||
`parse_feed` also captures each item's short description — RSS
|
||||
`<description>`, Atom `<summary>` or `<content>` — with HTML stripped,
|
||||
entities unescaped, and whitespace collapsed, so an item carries what
|
||||
it is about, not only a headline. Missing descriptions yield an empty
|
||||
summary, never an error.
|
||||
|
||||
### NEWS-08 — The digest carries summaries (coverage: test)
|
||||
|
||||
`render_digest` appends the sanitized, length-capped
|
||||
(`news-summary-chars`, default 200) summary after each headline, so
|
||||
the bot's ambient `{news}` context knows the gist of each story, not
|
||||
just its title. A zero cap restores the title-only digest.
|
||||
|
||||
### NEWS-09 — Fetched news is stored, deduped, and rolled over (coverage: test)
|
||||
|
||||
Both the digest run (kroa) and the posting run (ggg) upsert every
|
||||
fetched item into a `news` table keyed by link (or title), so the same
|
||||
story is stored once. After each run the store is pruned to the newest
|
||||
`news-keep` rows (default 400), a rolling window that bounds growth
|
||||
while keeping recent history searchable.
|
||||
|
||||
### NEWS-10 — get_news is offered as a tool (coverage: test)
|
||||
|
||||
When `enable-news-tool` is true and a store is configured, the chat
|
||||
call's `tools` list includes a `get_news` function (optional `topic`,
|
||||
`source`, `limit`) next to the other tools. Without a store or the
|
||||
flag it is absent.
|
||||
|
||||
### NEWS-11 — get_news retrieves filtered, sanitized items (coverage: test)
|
||||
|
||||
`get_news` returns recent stored items, newest first, optionally
|
||||
narrowed by `topic` (every keyword must appear in the title, summary,
|
||||
or source label — so `topic: "Nordland"` finds items from that source)
|
||||
and/or an exact `source`; `limit` is clamped to 1..30. Each result's title and
|
||||
summary are passed through `sanitize_external_text`. The bot can then
|
||||
`fetch_url` a returned link for the full article.
|
||||
|
||||
### NEWS-12 — get_news is metered per user (coverage: test)
|
||||
|
||||
Each `get_news` call increments a per-user daily counter; over
|
||||
`news-daily-per-user` (default 30) the tool refuses with an error
|
||||
result without touching the store. The budget gate (SAF-04) still
|
||||
applies to the surrounding model calls.
|
||||
@@ -0,0 +1,59 @@
|
||||
# SPEC-014 — Codex Mechanicus search
|
||||
|
||||
Luma is an Adeptus Mechanicus tech-priest; her lore has a real home —
|
||||
the priest's own Codex Mechanicus at `binaric.tech` (an Astro/MDX
|
||||
archive, five tongues). A `codex_search` function tool lets her consult
|
||||
that archive and answer from sourced inscriptions instead of inventing
|
||||
lore. The index is public but still untrusted by the time it reaches a
|
||||
prompt: the fetch is SSRF-guarded (SPEC-011 shares `guard_url`),
|
||||
size-bounded, and every returned field is sanitized (SAF-03). Luma-only;
|
||||
active only when `enable-codex = true`.
|
||||
|
||||
### CDX-01 — codex_search is offered as a tool (coverage: test)
|
||||
|
||||
When `enable-codex` is true, the chat call's `tools` list includes a
|
||||
`codex_search` function (`query` string, optional `lang`) next to any
|
||||
IGDB / fetch_url tools. When false, it is absent.
|
||||
|
||||
### CDX-02 — The index is fetched safely and cached (coverage: test)
|
||||
|
||||
The index URL (`codex-index-url`, default
|
||||
`https://binaric.tech/search-index.json`) passes the SSRF guard before
|
||||
any network call, is read under a byte cap (`codex-max-bytes`, default
|
||||
4 MB) with a download timeout, and is cached in memory for
|
||||
`codex-cache-ttl` (default 3600 s) so repeated searches do not re-fetch.
|
||||
|
||||
### CDX-03 — Ranking weights title over summary over body (coverage: test)
|
||||
|
||||
The query is tokenized (stopwords dropped); each inscription is scored
|
||||
by term hits weighted title (8) > summary (3) > body (1). Results are
|
||||
returned highest-score first, each as `{title, summary, collection,
|
||||
url}`, with `url` absolute against the site origin.
|
||||
|
||||
### CDX-04 — Language is preferred, with fallback (coverage: test)
|
||||
|
||||
Results are filtered to the requested `lang` (en, de, eo, no, uk;
|
||||
default en; unknown codes fall back to en) by the language segment in
|
||||
each inscription URL. If no inscription in that tongue matches, the
|
||||
search falls back to all tongues rather than returning nothing.
|
||||
|
||||
### CDX-05 — Results are sanitized and failure is reported (coverage: test)
|
||||
|
||||
Each `title` and `summary` is passed through `sanitize_external_text`
|
||||
and length-capped (`codex-summary-chars`, default 500). An index that
|
||||
cannot be fetched or parsed returns an `{error: ...}` dict the model can
|
||||
relay — `search` never raises.
|
||||
|
||||
### CDX-06 — Searches are metered per user (coverage: test)
|
||||
|
||||
Each `codex_search` increments a per-user daily counter; over
|
||||
`codex-daily-per-user` (default 50) the tool refuses with an error
|
||||
result without touching the index. The budget gate (SAF-04) still
|
||||
applies to the surrounding model calls.
|
||||
|
||||
### CDX-07 — Luma cites the codex, not invention (coverage: manual)
|
||||
|
||||
With the persona grounding line, when a pilgrim asks Cult Mechanicus
|
||||
lore Luma consults `codex_search` and answers from it, offering the
|
||||
`binaric.tech` link to read the full inscription rather than
|
||||
hallucinating. Verified live on ggg.
|
||||
@@ -174,6 +174,25 @@ class TestIGDBQuery(unittest.TestCase):
|
||||
self.assertEqual(result, [{"id": 1, "name": "Super Mario Bros"}])
|
||||
|
||||
|
||||
class TestIGDBNativeSearch(unittest.TestCase):
|
||||
def test_build_query_with_search_term(self):
|
||||
"""search_games uses IGDB full-text search, not a name prefix filter."""
|
||||
query = IGDBQuery.build_query(["name"], {"game_type": "= 0"}, limit=5, search_term="Marvel Tōkon")
|
||||
self.assertEqual(query, 'search "Marvel Tōkon"; fields name; limit 5; where game_type = 0;')
|
||||
|
||||
def test_search_term_escapes_quotes_and_backslashes(self):
|
||||
query = IGDBQuery.build_query(["name"], search_term='say "hi" \\ bye')
|
||||
self.assertIn('search "say \\"hi\\" \\\\ bye";', query)
|
||||
|
||||
@patch.object(IGDBQuery, "generalized_igdb_query")
|
||||
def test_search_games_passes_search_term(self, mock_query):
|
||||
mock_query.return_value = []
|
||||
IGDBQuery("cid", "token").search_games("Elden Ring", limit=3)
|
||||
_, kwargs = mock_query.call_args
|
||||
self.assertEqual(kwargs["search_term"], "Elden Ring")
|
||||
self.assertEqual(mock_query.call_args.args[0], {})
|
||||
|
||||
|
||||
class TestIGDBTokenRefresh(unittest.TestCase):
|
||||
@staticmethod
|
||||
def _oauth_response(token="fresh_token", expires_in=5_000_000):
|
||||
|
||||
@@ -0,0 +1,159 @@
|
||||
"""Unit coverage for SPEC-014 Codex Mechanicus search (CDX-01..06)."""
|
||||
|
||||
import json
|
||||
import unittest
|
||||
from unittest.mock import AsyncMock, patch
|
||||
|
||||
from fjerkroa_bot.codex import CODEX_SEARCH_TOOL, CodexSearch
|
||||
from fjerkroa_bot.openai_responder import OpenAIResponder
|
||||
|
||||
CONFIG = {"openai-token": "t", "model": "m", "system": "s", "history-limit": 5}
|
||||
|
||||
INDEX = {
|
||||
"items": [
|
||||
{
|
||||
"id": "doctrine-heretek",
|
||||
"collection": "doctrines",
|
||||
"url": "/en/codex/doctrines/doctrine-heretek/",
|
||||
"title": "Heretek — Doctrine of the Tech-Heretic",
|
||||
"summary": "The label the Cult Mechanicus stamps on Tech-Priests who pursue forbidden sciences.",
|
||||
"body": "xenotech, sentient machines, Warp-touched archeotech",
|
||||
},
|
||||
{
|
||||
"id": "doctrine-heretek",
|
||||
"collection": "doctrines",
|
||||
"url": "/de/codex/doctrines/doctrine-heretek/",
|
||||
"title": "Heretek — Doktrin des Techketzers",
|
||||
"summary": "Das Etikett des Kultes Mechanicus fuer Techpriester verbotener Wissenschaften.",
|
||||
"body": "Xenotech, empfindungsfaehige Maschinen",
|
||||
},
|
||||
{
|
||||
"id": "forge-stygies",
|
||||
"collection": "forges",
|
||||
"url": "/en/codex/forges/forge-stygies/",
|
||||
"title": "Stygies VIII",
|
||||
"summary": "A forge world of shrouded reputation.",
|
||||
"body": "The forge fields many Skitarii legions.",
|
||||
},
|
||||
]
|
||||
}
|
||||
|
||||
|
||||
def _reader(cfg):
|
||||
reader = CodexSearch(lambda: cfg)
|
||||
return reader
|
||||
|
||||
|
||||
class TestToolOffered(unittest.TestCase):
|
||||
def test_tool_present_only_when_enabled(self):
|
||||
"""CDX-01: codex_search appears only with enable-codex."""
|
||||
off = OpenAIResponder(CONFIG, "chat")
|
||||
self.assertNotIn("codex_search", [f["name"] for f in off._available_tools()])
|
||||
on = OpenAIResponder(dict(CONFIG, **{"enable-codex": True}), "chat")
|
||||
self.assertIn("codex_search", [f["name"] for f in on._available_tools()])
|
||||
self.assertEqual(CODEX_SEARCH_TOOL["name"], "codex_search")
|
||||
|
||||
|
||||
class TestIndexGuardAndCache(unittest.IsolatedAsyncioTestCase):
|
||||
async def test_internal_index_url_refused(self):
|
||||
"""CDX-02: an index URL on a private address is refused before any fetch."""
|
||||
reader = _reader({"enable-codex": True, "codex-index-url": "http://127.0.0.1/search-index.json"})
|
||||
result = await reader.search("heretek")
|
||||
self.assertIn("error", result)
|
||||
|
||||
async def test_index_cached_within_ttl(self):
|
||||
"""CDX-02: a second search inside the TTL does not re-fetch the index."""
|
||||
reader = _reader({"enable-codex": True, "codex-cache-ttl": 9999})
|
||||
raw = json.dumps(INDEX).encode()
|
||||
calls = [0]
|
||||
|
||||
class FakeResp:
|
||||
status = 200
|
||||
|
||||
async def __aenter__(self):
|
||||
return self
|
||||
|
||||
async def __aexit__(self, *a):
|
||||
return False
|
||||
|
||||
def raise_for_status(self):
|
||||
pass
|
||||
|
||||
class FakeSession:
|
||||
def get(self, url):
|
||||
calls[0] += 1
|
||||
return FakeResp()
|
||||
|
||||
async def __aenter__(self):
|
||||
return self
|
||||
|
||||
async def __aexit__(self, *a):
|
||||
return False
|
||||
|
||||
with patch("fjerkroa_bot.codex.read_capped", new=AsyncMock(return_value=raw)):
|
||||
with patch("fjerkroa_bot.codex.guard_url", return_value=None):
|
||||
with patch("fjerkroa_bot.codex.aiohttp.ClientSession", return_value=FakeSession()):
|
||||
first = await reader.search("heretek")
|
||||
second = await reader.search("stygies")
|
||||
self.assertEqual(calls[0], 1) # fetched once, served from cache the second time
|
||||
self.assertTrue(first["results"] and second["results"])
|
||||
|
||||
|
||||
class TestRankingAndLang(unittest.IsolatedAsyncioTestCase):
|
||||
async def _search(self, cfg, query, lang="en"):
|
||||
reader = _reader(dict({"enable-codex": True}, **cfg))
|
||||
reader._cache = INDEX["items"]
|
||||
reader._fetched_at = 1e18 # far future: never expires in test
|
||||
with patch("fjerkroa_bot.codex.time.monotonic", return_value=1e18):
|
||||
return await reader.search(query, lang)
|
||||
|
||||
async def test_title_hit_outranks_body_hit(self):
|
||||
"""CDX-03: a title match ranks above a body-only match."""
|
||||
result = await self._search({}, "heretek")
|
||||
self.assertEqual(result["results"][0]["title"].split(" ")[0], "Heretek")
|
||||
self.assertTrue(result["results"][0]["url"].startswith("https://binaric.tech/en/"))
|
||||
|
||||
async def test_lang_filter_selects_language(self):
|
||||
"""CDX-04: lang=de returns the German inscription."""
|
||||
result = await self._search({}, "heretek", lang="de")
|
||||
self.assertTrue(all("/de/" in r["url"] for r in result["results"]))
|
||||
self.assertIn("Techketzer", result["results"][0]["title"])
|
||||
|
||||
async def test_lang_fallback_when_absent(self):
|
||||
"""CDX-04: a tongue with no match falls back to all tongues, not empty."""
|
||||
result = await self._search({}, "stygies", lang="uk") # only en/de exist
|
||||
self.assertTrue(result["results"])
|
||||
self.assertEqual(result["results"][0]["title"], "Stygies VIII")
|
||||
|
||||
|
||||
class TestSanitizeAndFailure(unittest.IsolatedAsyncioTestCase):
|
||||
async def test_result_sanitized_and_capped(self):
|
||||
"""CDX-05: title/summary are @-neutralized and length-capped."""
|
||||
reader = _reader({"enable-codex": True, "codex-summary-chars": 40})
|
||||
reader._cache = [{"collection": "x", "url": "/en/x/", "title": "@everyone hi", "summary": "@here " + "y" * 500, "body": "hit"}]
|
||||
reader._fetched_at = 1e18
|
||||
with patch("fjerkroa_bot.codex.time.monotonic", return_value=1e18):
|
||||
result = await reader.search("hit")
|
||||
top = result["results"][0]
|
||||
self.assertNotIn("@everyone", top["title"])
|
||||
self.assertNotIn("@here", top["summary"])
|
||||
self.assertLessEqual(len(top["summary"]), 40)
|
||||
|
||||
async def test_index_failure_returns_error(self):
|
||||
"""CDX-05: a broken index returns an error dict, never raises."""
|
||||
reader = _reader({"enable-codex": True})
|
||||
with patch.object(reader, "_load_index", new=AsyncMock(side_effect=ValueError("boom"))):
|
||||
result = await reader.search("heretek")
|
||||
self.assertIn("error", result)
|
||||
|
||||
|
||||
class TestPerUserCap(unittest.IsolatedAsyncioTestCase):
|
||||
async def test_dispatch_caps_searches(self):
|
||||
"""CDX-06: over codex-daily-per-user, codex_search refuses without searching."""
|
||||
responder = OpenAIResponder(dict(CONFIG, **{"enable-codex": True, "codex-daily-per-user": 2}), "chat")
|
||||
responder.codex.search = AsyncMock(return_value={"query": "x", "results": []})
|
||||
for _ in range(2):
|
||||
await responder._dispatch_tool("codex_search", {"query": "heretek"}, "magos")
|
||||
blocked = await responder._dispatch_tool("codex_search", {"query": "heretek"}, "magos")
|
||||
self.assertIn("error", blocked)
|
||||
self.assertEqual(responder.codex.search.await_count, 2)
|
||||
@@ -45,6 +45,45 @@ class TestIngest(unittest.TestCase):
|
||||
self.assertIsNone(cache.ingest_bytes(PNG, "chat", "alice", "1"))
|
||||
|
||||
|
||||
class TestOversizedDownloadRejected(unittest.IsolatedAsyncioTestCase):
|
||||
async def test_download_stays_over_limit_and_is_rejected(self):
|
||||
"""IMG-10: an over-limit download must be rejected, not cached truncated."""
|
||||
with tempfile.TemporaryDirectory() as tmp:
|
||||
_, cache = make_cache(tmp, {"image-max-bytes": 32})
|
||||
|
||||
class FakeContent:
|
||||
@staticmethod
|
||||
async def iter_chunked(size):
|
||||
yield PNG # 72 bytes > 32
|
||||
|
||||
class FakeResp:
|
||||
content = FakeContent()
|
||||
|
||||
async def __aenter__(self):
|
||||
return self
|
||||
|
||||
async def __aexit__(self, *a):
|
||||
return False
|
||||
|
||||
def raise_for_status(self):
|
||||
pass
|
||||
|
||||
class FakeSession:
|
||||
def get(self, url):
|
||||
return FakeResp()
|
||||
|
||||
async def __aenter__(self):
|
||||
return self
|
||||
|
||||
async def __aexit__(self, *a):
|
||||
return False
|
||||
|
||||
with patch("fjerkroa_bot.images.aiohttp.ClientSession", return_value=FakeSession()):
|
||||
data = await cache._download("http://x.com/big.png")
|
||||
self.assertEqual(len(data), 33) # limit + 1, not silently capped to limit
|
||||
self.assertIsNone(await cache.ingest_url("http://x.com/big.png", "chat", "alice", "1"))
|
||||
|
||||
|
||||
class TestVisionDataUrls(OpsBase):
|
||||
async def test_attachment_becomes_data_url(self):
|
||||
"""IMG-11: the model sees a data: URL, never the CDN link."""
|
||||
|
||||
@@ -0,0 +1,317 @@
|
||||
"""Unit coverage for SPEC-013 news digest + memory + tool (NEWS-01..12)."""
|
||||
|
||||
import tempfile
|
||||
import unittest
|
||||
from pathlib import Path
|
||||
from unittest.mock import AsyncMock
|
||||
|
||||
from fjerkroa_bot.news import (
|
||||
GET_NEWS_TOOL,
|
||||
NewsFetcher,
|
||||
NewsPoster,
|
||||
load_seen,
|
||||
parse_feed,
|
||||
query_news,
|
||||
render_digest,
|
||||
save_seen,
|
||||
)
|
||||
from fjerkroa_bot.openai_responder import OpenAIResponder
|
||||
from fjerkroa_bot.persistence import PersistentStore
|
||||
|
||||
CONFIG = {"openai-token": "t", "model": "m", "system": "s", "history-limit": 5}
|
||||
|
||||
RSS = b"""<?xml version="1.0"?><rss><channel>
|
||||
<item><title>Game X released</title><link>https://ex.com/x</link></item>
|
||||
<item><title>Patch Y notes</title><link>https://ex.com/y</link></item>
|
||||
</channel></rss>"""
|
||||
|
||||
ATOM = b"""<?xml version="1.0"?><feed xmlns="http://www.w3.org/2005/Atom">
|
||||
<entry><title>Atom headline</title><link href="https://ex.com/a"/></entry>
|
||||
</feed>"""
|
||||
|
||||
|
||||
class TestParse(unittest.TestCase):
|
||||
def test_rss(self):
|
||||
"""NEWS-01: RSS items parsed with title + link."""
|
||||
items = parse_feed(RSS, "Src")
|
||||
self.assertEqual([i["title"] for i in items], ["Game X released", "Patch Y notes"])
|
||||
self.assertEqual(items[0]["link"], "https://ex.com/x")
|
||||
self.assertEqual(items[0]["source"], "Src")
|
||||
|
||||
def test_atom(self):
|
||||
"""NEWS-01: Atom entries parsed with href link."""
|
||||
items = parse_feed(ATOM, "A")
|
||||
self.assertEqual(items[0]["title"], "Atom headline")
|
||||
self.assertEqual(items[0]["link"], "https://ex.com/a")
|
||||
|
||||
def test_malformed_never_raises(self):
|
||||
"""NEWS-01: garbage XML returns [] without raising."""
|
||||
self.assertEqual(parse_feed(b"<not xml", "bad"), [])
|
||||
self.assertEqual(parse_feed(b"", "empty"), [])
|
||||
|
||||
|
||||
class TestDigest(unittest.TestCase):
|
||||
def test_sanitized_and_capped(self):
|
||||
"""NEWS-02: headlines sanitized, item count capped."""
|
||||
items = [{"title": "@everyone big news \x00", "link": "", "source": "S"} for _ in range(20)]
|
||||
digest = render_digest(items, max_items=5)
|
||||
self.assertEqual(digest.count("\n"), 4) # 5 lines
|
||||
self.assertNotIn("@everyone", digest)
|
||||
self.assertNotIn("\x00", digest)
|
||||
|
||||
|
||||
class TestCollect(unittest.IsolatedAsyncioTestCase):
|
||||
async def test_ssrf_skip_and_dedup(self):
|
||||
"""NEWS-03: guarded feed skipped, dup titles dropped, bad fetch survived."""
|
||||
|
||||
def guard(url):
|
||||
return "refused" if "internal" in url else None
|
||||
|
||||
async def fetch(url):
|
||||
if "boom" in url:
|
||||
raise ValueError("boom")
|
||||
return RSS # same content from two feeds -> dedup
|
||||
|
||||
fetcher = NewsFetcher(guard, fetch)
|
||||
feeds = [
|
||||
("https://a.com/feed", "A"),
|
||||
("https://internal/feed", "Internal"), # SSRF-skipped
|
||||
("https://boom.com/feed", "Boom"), # fetch fails
|
||||
("https://b.com/feed", "B"), # same RSS -> dup titles dropped
|
||||
]
|
||||
items = await fetcher.collect(feeds, per_feed=5)
|
||||
titles = [i["title"] for i in items]
|
||||
self.assertEqual(titles, ["Game X released", "Patch Y notes"]) # deduped, internal+boom skipped
|
||||
|
||||
async def test_per_feed_limit(self):
|
||||
"""NEWS-03: per-feed cap honored."""
|
||||
fetcher = NewsFetcher(lambda u: None, AsyncMock(return_value=RSS))
|
||||
items = await fetcher.collect([("https://a.com", "A")], per_feed=1)
|
||||
self.assertEqual(len(items), 1)
|
||||
|
||||
|
||||
class TestPoster(unittest.IsolatedAsyncioTestCase):
|
||||
def poster(self, posts):
|
||||
async def fetch(url):
|
||||
return RSS
|
||||
|
||||
async def post(hook, content):
|
||||
posts.append((hook, content))
|
||||
|
||||
return NewsPoster(lambda u: None, fetch, post)
|
||||
|
||||
async def test_posts_unseen_then_dedups(self):
|
||||
"""NEWS-04: unseen items post to the mapped webhook; re-run posts nothing."""
|
||||
posts = []
|
||||
poster = self.poster(posts)
|
||||
feeds = [("https://a.com/feed", "PS", "news")]
|
||||
hooks = {"news": "https://discord.com/api/webhooks/x"}
|
||||
posted, seen = await poster.run_post(feeds, hooks, set(), per_feed=5, max_per_run=8, seed_only=False)
|
||||
self.assertEqual(posted, 2)
|
||||
self.assertIn("PS", posts[0][1])
|
||||
self.assertIn("https://discord.com/api/webhooks/x", posts[0][0])
|
||||
# re-run with the accumulated seen -> nothing new
|
||||
posts.clear()
|
||||
posted2, _ = await poster.run_post(feeds, hooks, seen, per_feed=5, max_per_run=8, seed_only=False)
|
||||
self.assertEqual(posted2, 0)
|
||||
self.assertEqual(posts, [])
|
||||
|
||||
async def test_seed_run_posts_nothing(self):
|
||||
"""NEWS-05: seed_only marks items seen without posting."""
|
||||
posts = []
|
||||
poster = self.poster(posts)
|
||||
feeds = [("https://a.com/feed", "PS", "news")]
|
||||
posted, seen = await poster.run_post(feeds, {"news": "h"}, set(), 5, 8, seed_only=True)
|
||||
self.assertEqual(posted, 0)
|
||||
self.assertEqual(posts, [])
|
||||
self.assertEqual(len(seen), 2) # both marked seen
|
||||
|
||||
async def test_max_per_run_caps(self):
|
||||
"""NEWS-05: max-per-run caps posts; extras stay seen (not re-posted next run)."""
|
||||
posts = []
|
||||
poster = self.poster(posts)
|
||||
feeds = [("https://a.com/feed", "PS", "news")]
|
||||
posted, seen = await poster.run_post(feeds, {"news": "h"}, set(), per_feed=5, max_per_run=1, seed_only=False)
|
||||
self.assertEqual(posted, 1)
|
||||
self.assertEqual(len(seen), 2) # both seen, only one posted
|
||||
|
||||
async def test_failures_survived(self):
|
||||
"""NEWS-06: SSRF-skip, fetch fail, missing webhook, post error each survive."""
|
||||
posts = []
|
||||
|
||||
async def fetch(url):
|
||||
if "boom" in url:
|
||||
raise ValueError("boom")
|
||||
return RSS
|
||||
|
||||
async def post(hook, content):
|
||||
if hook == "bad":
|
||||
raise RuntimeError("post failed")
|
||||
posts.append((hook, content))
|
||||
|
||||
def guard(url):
|
||||
return "refused" if "internal" in url else None
|
||||
|
||||
poster = NewsPoster(guard, fetch, post)
|
||||
feeds = [
|
||||
("https://internal/feed", "I", "news"), # SSRF-skipped
|
||||
("https://boom.com/feed", "B", "news"), # fetch fails
|
||||
("https://ok.com/feed", "OK", "nowhere"), # no webhook for channel
|
||||
("https://ok2.com/feed", "OK2", "news"), # webhook raises
|
||||
]
|
||||
posted, seen = await poster.run_post(feeds, {"news": "bad"}, set(), 5, 8, seed_only=False)
|
||||
self.assertEqual(posted, 0) # everything failed/skipped, no crash
|
||||
|
||||
|
||||
class TestSeenState(unittest.TestCase):
|
||||
def test_roundtrip_and_seed_detection(self):
|
||||
"""NEWS-05: missing state -> (empty, existed=False); saved state reloads."""
|
||||
import tempfile
|
||||
from pathlib import Path
|
||||
|
||||
with tempfile.TemporaryDirectory() as tmp:
|
||||
path = str(Path(tmp) / "state.json")
|
||||
seen, existed = load_seen(path)
|
||||
self.assertEqual((seen, existed), (set(), False))
|
||||
save_seen(path, {"a", "b", "c"}, cap=5000)
|
||||
reloaded, existed2 = load_seen(path)
|
||||
self.assertEqual(reloaded, {"a", "b", "c"})
|
||||
self.assertTrue(existed2)
|
||||
|
||||
def test_cap_bounds_state(self):
|
||||
"""NEWS-05: save keeps at most `cap` keys."""
|
||||
import json
|
||||
|
||||
with tempfile.TemporaryDirectory() as tmp:
|
||||
path = str(Path(tmp) / "state.json")
|
||||
save_seen(path, {f"k{i}" for i in range(100)}, cap=10)
|
||||
self.assertEqual(len(json.load(open(path))), 10)
|
||||
|
||||
|
||||
RSS_DESC = b"""<?xml version="1.0"?><rss><channel>
|
||||
<item><title>Storm hits coast</title><link>https://ex.com/s</link>
|
||||
<description><p>Heavy <b>wind</b> expected</p></description></item>
|
||||
</channel></rss>"""
|
||||
|
||||
ATOM_SUM = b"""<?xml version="1.0"?><feed xmlns="http://www.w3.org/2005/Atom">
|
||||
<entry><title>Atom T</title><link href="https://ex.com/a"/><summary>Short gist here</summary></entry>
|
||||
</feed>"""
|
||||
|
||||
|
||||
class TestSummaries(unittest.TestCase):
|
||||
def test_rss_description_stripped(self):
|
||||
"""NEWS-07: RSS description parsed, HTML stripped, entities unescaped, whitespace collapsed."""
|
||||
items = parse_feed(RSS_DESC, "S")
|
||||
self.assertEqual(items[0]["summary"], "Heavy wind expected")
|
||||
|
||||
def test_atom_summary(self):
|
||||
"""NEWS-07: Atom summary collapsed to clean text."""
|
||||
items = parse_feed(ATOM_SUM, "A")
|
||||
self.assertEqual(items[0]["summary"], "Short gist here")
|
||||
|
||||
def test_missing_description_is_empty(self):
|
||||
"""NEWS-07: no description -> empty summary, never an error."""
|
||||
self.assertEqual(parse_feed(RSS, "S")[0]["summary"], "")
|
||||
|
||||
def test_digest_carries_summary(self):
|
||||
"""NEWS-08: digest appends the sanitized capped summary; zero cap = title only."""
|
||||
items = [{"title": "T", "link": "https://ex.com/x", "source": "NRK", "summary": "the gist of it"}]
|
||||
digest = render_digest(items, 10, 100)
|
||||
self.assertIn("[NRK]", digest)
|
||||
self.assertIn("the gist of it", digest)
|
||||
self.assertNotIn("the gist", render_digest(items, 10, 0)) # zero cap -> title only
|
||||
|
||||
|
||||
class NewsStoreBase(unittest.TestCase):
|
||||
def setUp(self):
|
||||
self.tmp = tempfile.TemporaryDirectory()
|
||||
self.addCleanup(self.tmp.cleanup)
|
||||
self.store = PersistentStore(Path(self.tmp.name) / "bot.db")
|
||||
|
||||
|
||||
class TestNewsStore(NewsStoreBase):
|
||||
def test_dedup_and_rolling_window(self):
|
||||
"""NEWS-09: items deduped by link; prune keeps the newest N."""
|
||||
first = [
|
||||
{"title": "A", "link": "L1", "source": "S", "summary": "sa"},
|
||||
{"title": "B", "link": "L2", "source": "S", "summary": "sb"},
|
||||
]
|
||||
self.assertEqual(self.store.add_news_items(first), 2)
|
||||
self.assertEqual(self.store.add_news_items([dict(first[0])]), 0) # dup link ignored
|
||||
self.assertEqual(self.store.news_count(), 2)
|
||||
self.store.prune_news(1)
|
||||
self.assertEqual(self.store.news_count(), 1)
|
||||
self.assertEqual(self.store.recent_news(5)[0]["title"], "B") # newest survives
|
||||
|
||||
def test_dedup_by_title_when_no_link(self):
|
||||
"""NEWS-09: linkless items dedup on title."""
|
||||
self.store.add_news_items([{"title": "Same", "link": "", "source": "S", "summary": ""}])
|
||||
self.store.add_news_items([{"title": "Same", "link": "", "source": "S", "summary": ""}])
|
||||
self.assertEqual(self.store.news_count(), 1)
|
||||
|
||||
|
||||
class TestQueryNews(NewsStoreBase):
|
||||
def seed(self):
|
||||
self.store.add_news_items(
|
||||
[
|
||||
{"title": "Nordland storm", "link": "L1", "source": "Nordland", "summary": "strong wind on the coast"},
|
||||
{"title": "Oslo budget", "link": "L2", "source": "NRK", "summary": "@everyone spending plan"},
|
||||
{"title": "Sport result", "link": "L3", "source": "Sport", "summary": "the match ended"},
|
||||
]
|
||||
)
|
||||
|
||||
def test_topic_filter(self):
|
||||
"""NEWS-11: topic keywords must appear in title or summary."""
|
||||
self.seed()
|
||||
res = query_news(self.store, topic="storm")
|
||||
self.assertEqual([r["title"] for r in res["results"]], ["Nordland storm"])
|
||||
|
||||
def test_topic_matches_source_label(self):
|
||||
"""NEWS-11: topic also matches the source label, so 'Nordland' finds regional items."""
|
||||
self.store.add_news_items([{"title": "Ferry delayed", "link": "LX", "source": "Nordland", "summary": "boat late"}])
|
||||
res = query_news(self.store, topic="Nordland")
|
||||
self.assertTrue(any(r["link"] == "LX" for r in res["results"])) # matched via source, not title/summary
|
||||
|
||||
def test_source_filter_and_sanitize(self):
|
||||
"""NEWS-11: source narrows results; title/summary are sanitized."""
|
||||
self.seed()
|
||||
res = query_news(self.store, source="NRK")
|
||||
self.assertTrue(res["results"] and all(r["source"] == "NRK" for r in res["results"]))
|
||||
self.assertNotIn("@everyone", res["results"][0]["summary"])
|
||||
|
||||
def test_limit_clamped_and_no_store(self):
|
||||
"""NEWS-11: limit clamps to 1..30; a missing store returns an error."""
|
||||
self.seed()
|
||||
self.assertLessEqual(len(query_news(self.store, limit=999)["results"]), 30)
|
||||
self.assertGreaterEqual(len(query_news(self.store, limit=0)["results"]), 1)
|
||||
self.assertIn("error", query_news(None))
|
||||
|
||||
|
||||
class TestNewsTool(unittest.IsolatedAsyncioTestCase):
|
||||
def setUp(self):
|
||||
self.tmp = tempfile.TemporaryDirectory()
|
||||
self.addCleanup(self.tmp.cleanup)
|
||||
|
||||
def _responder(self, **extra):
|
||||
cfg = dict(CONFIG, **{"history-directory": self.tmp.name}, **extra)
|
||||
return OpenAIResponder(cfg, "chat")
|
||||
|
||||
def test_tool_offered_needs_flag_and_store(self):
|
||||
"""NEWS-10: get_news offered only with enable-news-tool AND a store."""
|
||||
no_store = OpenAIResponder(dict(CONFIG, **{"enable-news-tool": True}), "chat")
|
||||
self.assertIsNone(no_store.store)
|
||||
self.assertNotIn("get_news", [f["name"] for f in no_store._available_tools()])
|
||||
flag_off = self._responder()
|
||||
self.assertNotIn("get_news", [f["name"] for f in flag_off._available_tools()])
|
||||
on = self._responder(**{"enable-news-tool": True})
|
||||
self.assertIn("get_news", [f["name"] for f in on._available_tools()])
|
||||
self.assertEqual(GET_NEWS_TOOL["name"], "get_news")
|
||||
|
||||
async def test_dispatch_caps_news(self):
|
||||
"""NEWS-12: over news-daily-per-user, get_news refuses without querying."""
|
||||
responder = self._responder(**{"enable-news-tool": True, "news-daily-per-user": 2})
|
||||
responder.store.add_news_items([{"title": "x", "link": "l", "source": "s", "summary": "y"}])
|
||||
for _ in range(2):
|
||||
self.assertIn("results", await responder._dispatch_tool("get_news", {}, "alice"))
|
||||
blocked = await responder._dispatch_tool("get_news", {}, "alice")
|
||||
self.assertIn("error", blocked)
|
||||
@@ -133,3 +133,69 @@ class TestTasksKillSwitch(OpsBase):
|
||||
"""OPS-09: bot-initiated posts respect pause/quiet."""
|
||||
await self.bot.on_message(self.staff_msg("!bot pause"))
|
||||
self.assertFalse(self.bot.bot_initiated_allowed())
|
||||
|
||||
|
||||
class TestHelp(OpsBase):
|
||||
STAFF_CMDS = (
|
||||
"pause",
|
||||
"resume",
|
||||
"quiet <minutes>",
|
||||
"status",
|
||||
"spend",
|
||||
"images on|off",
|
||||
"memory <user>",
|
||||
"forget-fact <id>",
|
||||
"pin <channel|global>",
|
||||
"unpin <id>",
|
||||
"pins",
|
||||
"task-approve <id>",
|
||||
"task-cancel <id>",
|
||||
"(list)",
|
||||
)
|
||||
|
||||
def test_staff_help_is_complete_and_grouped(self):
|
||||
"""OPS-17: staff help lists every operator command, grouped by purpose."""
|
||||
text = self.bot._help_text(staff=True)
|
||||
for cmd in self.STAFF_CMDS:
|
||||
self.assertIn(cmd, text, f"missing {cmd!r} in staff help")
|
||||
for group in ("Control:", "Cost:", "Memory:", "Tasks:"):
|
||||
self.assertIn(group, text)
|
||||
for cmd in ("!help", "!forgetme", "!privacy", "!wichtel"):
|
||||
self.assertIn(cmd, text) # everywhere-commands shown too
|
||||
|
||||
def test_user_help_hides_operator_commands(self):
|
||||
"""OPS-17: non-staff help shows only the everyone-commands."""
|
||||
text = self.bot._help_text(staff=False)
|
||||
for cmd in ("!help", "!forgetme", "!privacy", "!wichtel"):
|
||||
self.assertIn(cmd, text)
|
||||
for op in ("task-approve", "images on|off", "spend", "Staff commands", "Control:"):
|
||||
self.assertNotIn(op, text)
|
||||
|
||||
async def test_bot_help_in_staff_channel_returns_full_help(self):
|
||||
"""OPS-17: `!bot help` answers with the complete staff help."""
|
||||
await self.bot.on_message(self.staff_msg("!bot help"))
|
||||
text = self.bot.staff_channel.send.await_args.args[0]
|
||||
self.assertIn("Staff commands", text)
|
||||
self.assertIn("task-cancel <id>", text)
|
||||
|
||||
async def test_unknown_bot_command_falls_back_to_help(self):
|
||||
"""OPS-17: an unrecognised `!bot` command shows the full help, not a partial line."""
|
||||
await self.bot.on_message(self.staff_msg("!bot wat"))
|
||||
text = self.bot.staff_channel.send.await_args.args[0]
|
||||
self.assertIn("Control:", text)
|
||||
|
||||
async def test_help_in_public_channel_is_user_scoped(self):
|
||||
"""OPS-17: `!help` in a normal channel lists only everyone-commands."""
|
||||
msg = self.public_msg("!help")
|
||||
await self.bot.on_message(msg)
|
||||
text = msg.channel.send.await_args.args[0]
|
||||
self.assertIn("!forgetme", text)
|
||||
self.assertNotIn("Staff commands", text)
|
||||
self.assertNotIn("task-approve", text)
|
||||
|
||||
async def test_help_works_while_paused(self):
|
||||
"""OPS-17: help answers even when replies are paused."""
|
||||
await self.bot.on_message(self.staff_msg("!bot pause"))
|
||||
msg = self.public_msg("!help")
|
||||
await self.bot.on_message(msg)
|
||||
msg.channel.send.assert_awaited()
|
||||
|
||||
@@ -79,6 +79,65 @@ class TestRedirectRevalidation(unittest.IsolatedAsyncioTestCase):
|
||||
await reader._get(FakeSession(), "http://safe.example.com", 1000)
|
||||
|
||||
|
||||
class TestMetaRefresh(unittest.IsolatedAsyncioTestCase):
|
||||
async def test_follows_meta_refresh_to_real_article(self):
|
||||
"""URL-04: a getnews-style meta-refresh stub is followed to the real article."""
|
||||
reader = URLReader(lambda: {}, None)
|
||||
stub = (
|
||||
b'<html><head><meta http-equiv="refresh" content="0;url=https://pushsquare.com/real"></head><body>Redirecting...</body></html>'
|
||||
)
|
||||
article = b"<html><body><h1>MARVEL Tokon</h1><p>Full article text here</p></body></html>"
|
||||
calls = []
|
||||
|
||||
async def fake_get(session, url, max_bytes):
|
||||
calls.append(url)
|
||||
return (url, stub if "stub" in url else article)
|
||||
|
||||
reader._get = fake_get # type: ignore
|
||||
with patch("fjerkroa_bot.url_reader.guard_url", return_value=None):
|
||||
import fjerkroa_bot.url_reader as ur
|
||||
|
||||
# patch the session context so fetch() runs against fake_get
|
||||
class FakeCM:
|
||||
async def __aenter__(self):
|
||||
return object()
|
||||
|
||||
async def __aexit__(self, *a):
|
||||
return False
|
||||
|
||||
with patch.object(ur.aiohttp, "ClientSession", return_value=FakeCM()):
|
||||
result = await reader.fetch("https://gggemein.de/url/stub.html", "chat", "alice")
|
||||
self.assertIn("Full article text", result["text"])
|
||||
self.assertEqual(result["url"], "https://pushsquare.com/real")
|
||||
self.assertIn("https://pushsquare.com/real", calls)
|
||||
|
||||
async def test_meta_refresh_to_internal_is_not_followed(self):
|
||||
"""URL-04: a meta-refresh pointing at an internal IP is refused (SSRF)."""
|
||||
reader = URLReader(lambda: {}, None)
|
||||
stub = b'<meta http-equiv="refresh" content="0; url=http://127.0.0.1/secret">Redirecting'
|
||||
|
||||
async def fake_get(session, url, max_bytes):
|
||||
return (url, stub)
|
||||
|
||||
reader._get = fake_get # type: ignore
|
||||
import fjerkroa_bot.url_reader as ur
|
||||
|
||||
class FakeCM:
|
||||
async def __aenter__(self):
|
||||
return object()
|
||||
|
||||
async def __aexit__(self, *a):
|
||||
return False
|
||||
|
||||
def guard(u):
|
||||
return "refused" if "127.0.0.1" in u else None
|
||||
|
||||
with patch("fjerkroa_bot.url_reader.guard_url", side_effect=guard):
|
||||
with patch.object(ur.aiohttp, "ClientSession", return_value=FakeCM()):
|
||||
result = await reader.fetch("https://safe.com/x", "chat", "alice")
|
||||
self.assertEqual(result["url"], "https://safe.com/x") # did not follow to 127.0.0.1
|
||||
|
||||
|
||||
class TestTextExtraction(unittest.TestCase):
|
||||
def test_html_reduced_to_text(self):
|
||||
"""URL-05: scripts/styles dropped, tags stripped."""
|
||||
@@ -91,6 +150,46 @@ class TestTextExtraction(unittest.TestCase):
|
||||
self.assertNotIn("x{}", text)
|
||||
|
||||
|
||||
class TestBodyReadCollectsAllChunks(unittest.IsolatedAsyncioTestCase):
|
||||
async def test_get_reads_past_first_chunk(self):
|
||||
"""URL-05 regression: body arrives in many chunks; all are collected up to the cap."""
|
||||
reader = URLReader(lambda: {}, None)
|
||||
chunks = [b"<title>t</title>", b"<p>middle</p>", b"<p>end</p>"]
|
||||
|
||||
class FakeContent:
|
||||
@staticmethod
|
||||
async def iter_chunked(size):
|
||||
for chunk in chunks:
|
||||
yield chunk
|
||||
|
||||
class FakeResp:
|
||||
status = 200
|
||||
headers = {}
|
||||
url = "http://safe.example.com"
|
||||
content = FakeContent()
|
||||
|
||||
async def __aenter__(self):
|
||||
return self
|
||||
|
||||
async def __aexit__(self, *a):
|
||||
return False
|
||||
|
||||
def raise_for_status(self):
|
||||
pass
|
||||
|
||||
class FakeSession:
|
||||
def get(self, url, allow_redirects=False):
|
||||
return FakeResp()
|
||||
|
||||
with patch("fjerkroa_bot.url_reader.guard_url", return_value=None):
|
||||
_, body = await reader._get(FakeSession(), "http://safe.example.com", 1000)
|
||||
self.assertEqual(body, b"".join(chunks))
|
||||
|
||||
with patch("fjerkroa_bot.url_reader.guard_url", return_value=None):
|
||||
_, body = await reader._get(FakeSession(), "http://safe.example.com", 20)
|
||||
self.assertEqual(body, b"".join(chunks)[:20])
|
||||
|
||||
|
||||
class TestFetchSanitizes(unittest.IsolatedAsyncioTestCase):
|
||||
async def test_fetch_result_is_sanitized_and_capped(self):
|
||||
"""URL-05: fetch output is length-capped and @everyone-neutralized."""
|
||||
|
||||
@@ -627,6 +627,7 @@ version = "3.0.0"
|
||||
source = { editable = "." }
|
||||
dependencies = [
|
||||
{ name = "aiohttp" },
|
||||
{ name = "defusedxml" },
|
||||
{ name = "discord-py" },
|
||||
{ name = "openai" },
|
||||
{ name = "requests" },
|
||||
@@ -656,6 +657,7 @@ dev = [
|
||||
[package.metadata]
|
||||
requires-dist = [
|
||||
{ name = "aiohttp", specifier = ">=3.12" },
|
||||
{ name = "defusedxml", specifier = ">=0.7" },
|
||||
{ name = "discord-py", specifier = ">=2.5,<3" },
|
||||
{ name = "openai", specifier = ">=2.45" },
|
||||
{ name = "requests", specifier = ">=2.32" },
|
||||
|
||||
Reference in New Issue
Block a user