news memory + get_news tool (news-07..12): feed summaries, deduped rolling store (schema v6), on-demand filtered retrieval

This commit is contained in:
Oleksandr Kozachuk
2026-07-14 11:34:23 +02:00
parent 891fbdc101
commit 5d01400638
6 changed files with 390 additions and 15 deletions
+110 -10
View File
@@ -11,8 +11,10 @@ CLI: python -m fjerkroa_bot.news --config kroa.toml
import argparse
import logging
import re
import sys
import time
from html import unescape
from typing import Any, Dict, List, Optional, Tuple
import defusedxml.ElementTree as ElementTree # hardened XML: feeds are untrusted (XXE/billion-laughs)
@@ -21,8 +23,17 @@ from .ai_responder import sanitize_external_text
DEFAULT_PER_FEED = 3
DEFAULT_MAX_ITEMS = 15
DEFAULT_SUMMARY_CHARS = 200
DEFAULT_NEWS_KEEP = 400
FETCH_TIMEOUT_S = 15
_ATOM = "{http://www.w3.org/2005/Atom}"
_TAG_RE = re.compile(r"<[^>]+>")
def _clean_summary(raw: str, max_len: int = 300) -> str:
"""Strip HTML, unescape entities, collapse whitespace (feed descriptions are often HTML)."""
text = unescape(_TAG_RE.sub(" ", raw or ""))
return re.sub(r"\s+", " ", text).strip()[:max_len]
def parse_feed(data: bytes, source: str = "") -> List[Dict[str, str]]:
@@ -34,31 +45,39 @@ def parse_feed(data: bytes, source: str = "") -> List[Dict[str, str]]:
logging.warning(f"news: unparseable/unsafe feed {source!r}: {err!r}")
return []
items: List[Dict[str, str]] = []
# RSS: <rss><channel><item><title/><link/>
# RSS: <rss><channel><item><title/><link/><description/>
for item in root.iter("item"):
title = (item.findtext("title") or "").strip()
link = (item.findtext("link") or "").strip()
summary = _clean_summary(item.findtext("description") or "")
if title:
items.append({"title": title, "link": link, "source": source})
# Atom: <feed><entry><title/><link href=/>
items.append({"title": title, "link": link, "source": source, "summary": summary})
# Atom: <feed><entry><title/><link href=/><summary|content/>
for entry in root.iter(f"{_ATOM}entry"):
title = (entry.findtext(f"{_ATOM}title") or "").strip()
link_el = entry.find(f"{_ATOM}link")
link = link_el.get("href", "") if link_el is not None else ""
summary = _clean_summary(entry.findtext(f"{_ATOM}summary") or entry.findtext(f"{_ATOM}content") or "")
if title:
items.append({"title": title, "link": link, "source": source})
items.append({"title": title, "link": link, "source": source, "summary": summary})
return items
def render_digest(items: List[Dict[str, str]], max_items: int = DEFAULT_MAX_ITEMS) -> str:
"""Compact sanitized digest for the {news} prompt slot."""
def render_digest(items: List[Dict[str, str]], max_items: int = DEFAULT_MAX_ITEMS, summary_chars: int = DEFAULT_SUMMARY_CHARS) -> str:
"""Compact sanitized digest for the {news} prompt slot (title + short summary + link)."""
lines = []
for item in items[:max_items]:
title = sanitize_external_text(item["title"], 200)
source = item.get("source", "")
link = item.get("link", "")
summary = sanitize_external_text(item.get("summary", ""), summary_chars) if summary_chars else ""
prefix = f"[{source}] " if source else ""
lines.append(f"- {prefix}{title}" + (f" ({link})" if link else ""))
line = f"- {prefix}{title}"
if summary:
line += f"{summary}"
if link:
line += f" ({link})"
lines.append(line)
return "\n".join(lines)
@@ -102,10 +121,11 @@ def item_key(item: Dict[str, str]) -> str:
class NewsPoster:
"""Post NEW feed items to Discord channel webhooks (ggg model, SPEC-013 NEWS-04..06)."""
def __init__(self, guard, fetch_bytes, post_webhook) -> None:
def __init__(self, guard, fetch_bytes, post_webhook, store: Any = None) -> None:
self._guard = guard
self._fetch_bytes = fetch_bytes
self._post_webhook = post_webhook
self._store = store
async def run_post(
self,
@@ -115,9 +135,11 @@ class NewsPoster:
per_feed: int,
max_per_run: int,
seed_only: bool,
keep: int = DEFAULT_NEWS_KEEP,
) -> Tuple[int, set]:
"""Returns (posted_count, updated_seen). seed_only marks new items seen without posting."""
posted = 0
harvested: List[Dict[str, str]] = []
for url, label, channel in feeds:
reason = self._guard(url)
if reason:
@@ -129,6 +151,7 @@ class NewsPoster:
logging.warning(f"news-post: fetch failed for {label}: {repr(err)}")
continue
for item in parse_feed(data, label)[:per_feed]:
harvested.append(item) # NEWS-09: everything parsed feeds the searchable store
key = item_key(item)
if not key or key in seen:
continue
@@ -136,6 +159,9 @@ class NewsPoster:
may_post = not seed_only and posted < max_per_run
if may_post and await self._deliver(item, label, channel, webhooks):
posted += 1
if self._store is not None and harvested:
self._store.add_news_items(harvested)
self._store.prune_news(keep)
return posted, seen
async def _deliver(self, item: Dict[str, str], label: str, channel: str, webhooks: Dict[str, str]) -> bool:
@@ -154,6 +180,76 @@ class NewsPoster:
return False
# --- news memory + on-demand retrieval tool (SPEC-013 NEWS-09..12) ---
GET_NEWS_TOOL = {
"name": "get_news",
"description": "Fetch recent real-world news the bot has collected from its RSS feeds (local, national, world, sport, "
"culture). Use when someone asks what is new or what is happening, optionally about a topic or from a particular "
"source. Returns headlines with a short summary and a link to read more.",
"parameters": {
"type": "object",
"properties": {
"topic": {"type": "string", "description": "Optional keywords to filter by, e.g. 'Nordland', 'football', 'weather'."},
"source": {"type": "string", "description": "Optional source label, e.g. 'NRK', 'Aftenposten', 'Verden', 'Sport'."},
"limit": {"type": "integer", "description": "How many items to return (default 10, max 30)."},
},
"required": [],
},
}
def _news_terms(topic: Optional[str]) -> List[str]:
return [t for t in re.split(r"\W+", (topic or "").lower()) if len(t) > 1][:5]
def query_news(
store: Any, topic: Optional[str] = None, source: Optional[str] = None, limit: int = 10, summary_chars: int = DEFAULT_SUMMARY_CHARS
) -> Dict[str, Any]:
"""Retrieve stored news for the get_news tool: term/source-filtered, sanitized (NEWS-11)."""
if store is None:
return {"error": "news store unavailable"}
limit = max(1, min(int(limit or 10), 30))
src = (str(source).strip() or None) if source else None
terms = _news_terms(topic)
try:
rows = store.search_news(terms, limit, src) if terms else store.recent_news(limit, src)
except Exception as err:
logging.warning(f"news: query failed: {err!r}")
return {"error": "news lookup failed"}
results = [
{
"source": row.get("source", ""),
"title": sanitize_external_text(row.get("title", ""), 200),
"summary": sanitize_external_text(row.get("summary", ""), summary_chars),
"link": row.get("link", ""),
}
for row in rows
]
return {"topic": topic or "", "source": src or "", "results": results}
def _open_store(config: Dict[str, Any]) -> Any:
directory = config.get("history-directory")
if not directory:
return None
from pathlib import Path
from .persistence import PersistentStore
return PersistentStore(Path(str(directory)).expanduser() / "bot.db")
def persist_news(config: Dict[str, Any], items: List[Dict[str, str]]) -> int:
"""Upsert fetched items into the news store, prune to the rolling window (NEWS-09)."""
store = _open_store(config)
if store is None or not items:
return 0
added = store.add_news_items(items)
store.prune_news(int(config.get("news-keep", DEFAULT_NEWS_KEEP)))
return added
def load_seen(path: str) -> Tuple[set, bool]:
"""(seen-set, existed). Missing/broken state -> empty set, existed=False (seed run)."""
import json
@@ -231,7 +327,7 @@ async def run_post(config: Dict[str, Any]) -> int:
logging.error("news-post: need news-post-webhooks and news-post-feeds")
return 0
seen, existed = load_seen(state_path)
poster = NewsPoster(guard_url, _aiohttp_fetch, _aiohttp_post)
poster = NewsPoster(guard_url, _aiohttp_fetch, _aiohttp_post, store=_open_store(config))
posted, seen = await poster.run_post(
feeds,
webhooks,
@@ -239,6 +335,7 @@ async def run_post(config: Dict[str, Any]) -> int:
int(config.get("news-post-per-feed", DEFAULT_POST_PER_FEED)),
int(config.get("news-post-max-per-run", DEFAULT_POST_MAX_PER_RUN)),
seed_only=not existed, # first run seeds without flooding the channels
keep=int(config.get("news-keep", DEFAULT_NEWS_KEEP)),
)
save_seen(state_path, seen, int(config.get("news-post-seen-cap", DEFAULT_SEEN_CAP)))
logging.info(f"news-post: posted {posted} item(s)" + (" (seed run — nothing posted)" if not existed else ""))
@@ -258,7 +355,10 @@ async def run(config: Dict[str, Any]) -> Optional[str]:
return None
fetcher = NewsFetcher(guard_url, _aiohttp_fetch)
items = await fetcher.collect(feeds, int(config.get("news-per-feed", DEFAULT_PER_FEED)))
digest = render_digest(items, int(config.get("news-max-items", DEFAULT_MAX_ITEMS)))
persist_news(config, items) # NEWS-09: feed the searchable rolling store for get_news
digest = render_digest(
items, int(config.get("news-max-items", DEFAULT_MAX_ITEMS)), int(config.get("news-summary-chars", DEFAULT_SUMMARY_CHARS))
)
header = f"News as of {time.strftime('%Y-%m-%d %H:%M UTC', time.gmtime())}:\n"
with open(out_path, "w", encoding="utf-8") as fd:
fd.write(header + digest + "\n")
+10
View File
@@ -14,6 +14,7 @@ from .codex import DEFAULT_LIMIT as CODEX_DEFAULT_LIMIT
from .codex import CodexSearch
from .igdblib import IGDBQuery
from .leonardo_draw import LeonardoAIDrawMixIn
from .news import GET_NEWS_TOOL, query_news
from .quota import QuotaLedger
from .url_reader import FETCH_URL_TOOL, URLReader
@@ -180,6 +181,8 @@ class OpenAIResponder(AIResponder, LeonardoAIDrawMixIn):
functions.append(FETCH_URL_TOOL)
if self.codex.enabled(): # CDX-01
functions.append(CODEX_SEARCH_TOOL)
if self.config.get("enable-news-tool", False) and self.store is not None: # NEWS-10
functions.append(GET_NEWS_TOOL)
return functions
async def _dispatch_tool(self, name: str, args: Dict[str, Any], author: str) -> Any:
@@ -197,6 +200,13 @@ class OpenAIResponder(AIResponder, LeonardoAIDrawMixIn):
self.ledger._add(f"codex:{author}", 1)
limit = int(self.config.get("codex-limit", CODEX_DEFAULT_LIMIT))
return await self.codex.search(str(args.get("query", "")), str(args.get("lang", "en")), limit)
if name == "get_news":
per_user_cap = int(self.config.get("news-daily-per-user", 30))
if self.ledger._get(f"news:{author}") >= per_user_cap: # NEWS-12
return {"error": "daily news lookup limit reached"}
self.ledger._add(f"news:{author}", 1)
summary_chars = int(self.config.get("news-summary-chars", 200))
return query_news(self.store, args.get("topic"), args.get("source"), args.get("limit", 10), summary_chars)
return await self._execute_igdb_function(name, args)
async def draw_openai(self, description: str, count: int = 1) -> List[BytesIO]:
+66 -1
View File
@@ -13,7 +13,7 @@ from contextlib import closing
from pathlib import Path
from typing import Any, Dict, List, Optional
SCHEMA_VERSION = 5
SCHEMA_VERSION = 6
class PersistentStore:
@@ -72,6 +72,13 @@ class PersistentStore:
" due_at TEXT NOT NULL, payload TEXT NOT NULL, state TEXT NOT NULL DEFAULT 'queued',"
" created_at TEXT NOT NULL DEFAULT (datetime('now')), executed_at TEXT)"
)
if version < 6:
# News memory (SPEC-013 NEWS-09): deduped rolling store of fetched items
conn.execute(
"CREATE TABLE IF NOT EXISTS news (id INTEGER PRIMARY KEY, dedup_key TEXT UNIQUE NOT NULL,"
" source TEXT NOT NULL DEFAULT '', title TEXT NOT NULL, link TEXT NOT NULL DEFAULT '',"
" summary TEXT NOT NULL DEFAULT '', first_seen TEXT NOT NULL DEFAULT (datetime('now')))"
)
if version < SCHEMA_VERSION:
conn.execute(f"PRAGMA user_version = {SCHEMA_VERSION}")
os.chmod(self.db_path, 0o600) # conversation data (PER-04)
@@ -115,6 +122,64 @@ class PersistentStore:
row = conn.execute("SELECT value FROM usage WHERE day = ? AND key = ?", (day, key)).fetchone()
return float(row[0]) if row else 0.0
# --- news memory (SPEC-013 NEWS-09..12) ---
def add_news_items(self, items: List[Dict[str, Any]]) -> int:
"""Insert deduped news rows (by link or title); returns how many were new (NEWS-09)."""
added = 0
with closing(self._connect()) as conn, conn:
for item in items:
title = str(item.get("title") or "").strip()
key = (str(item.get("link") or "").strip()) or title
if not title or not key:
continue
cursor = conn.execute(
"INSERT OR IGNORE INTO news (dedup_key, source, title, link, summary) VALUES (?, ?, ?, ?, ?)",
(key, str(item.get("source") or ""), title, str(item.get("link") or ""), str(item.get("summary") or "")),
)
added += cursor.rowcount
return added
def recent_news(self, limit: int = 20, source: Optional[str] = None) -> List[Dict[str, Any]]:
sql = "SELECT source, title, link, summary FROM news"
params: List[Any] = []
if source:
sql += " WHERE source = ?"
params.append(source)
sql += " ORDER BY id DESC LIMIT ?"
params.append(int(limit))
with closing(self._connect()) as conn:
rows = conn.execute(sql, params).fetchall()
return [{"source": r[0], "title": r[1], "link": r[2], "summary": r[3]} for r in rows]
def search_news(self, terms: List[str], limit: int = 20, source: Optional[str] = None) -> List[Dict[str, Any]]:
"""Rows where every term appears in title or summary; optional source filter (NEWS-11)."""
params: List[Any] = []
clauses = []
for term in terms:
clauses.append("(title LIKE ? OR summary LIKE ? OR source LIKE ?)")
like = f"%{term}%"
params += [like, like, like]
where = " AND ".join(clauses) if clauses else "1=1"
if source:
where = f"({where}) AND source = ?"
params.append(source)
params.append(int(limit))
sql = f"SELECT source, title, link, summary FROM news WHERE {where} ORDER BY id DESC LIMIT ?" # nosec B608 - fixed templates; values parameterised
with closing(self._connect()) as conn:
rows = conn.execute(sql, params).fetchall()
return [{"source": r[0], "title": r[1], "link": r[2], "summary": r[3]} for r in rows]
def prune_news(self, keep: int) -> int:
"""Keep the newest `keep` rows, delete the rest (rolling window, NEWS-09)."""
with closing(self._connect()) as conn, conn:
cursor = conn.execute("DELETE FROM news WHERE id NOT IN (SELECT id FROM news ORDER BY id DESC LIMIT ?)", (int(keep),))
return cursor.rowcount
def news_count(self) -> int:
with closing(self._connect()) as conn:
return int(conn.execute("SELECT COUNT(*) FROM news").fetchone()[0])
def delete_history_of_user(self, user: str) -> int:
"""Remove persisted rows carrying this user's messages (SAF-08)."""
with closing(self._connect()) as conn, conn: