Compare commits

...

2 Commits

6 changed files with 45 additions and 19 deletions
+1 -1
View File
@@ -1,7 +1,7 @@
"""Codex Mechanicus search tool (SPEC-014, FDB-019). """Codex Mechanicus search tool (SPEC-014, FDB-019).
Luma's own sacred archive — the Codex Mechanicus at binaric.tech — as a Luma's own sacred archive — the Codex Mechanicus at binaric.tech — as a
function tool. She searches the codex index and answers Cult Mechanicus function tool. He searches the codex index and answers Cult Mechanicus
lore from real, sourced inscriptions instead of inventing it. The index lore from real, sourced inscriptions instead of inventing it. The index
is fetched over HTTPS (SSRF-guarded, size-bounded, cached in memory) and is fetched over HTTPS (SSRF-guarded, size-bounded, cached in memory) and
every field returned to the model is sanitized (SAF-03), because even every field returned to the model is sanitized (SAF-03), because even
+20 -12
View File
@@ -27,6 +27,7 @@ DEFAULT_SUMMARY_CHARS = 200
DEFAULT_NEWS_KEEP = 400 DEFAULT_NEWS_KEEP = 400
FETCH_TIMEOUT_S = 15 FETCH_TIMEOUT_S = 15
_ATOM = "{http://www.w3.org/2005/Atom}" _ATOM = "{http://www.w3.org/2005/Atom}"
_RSS1 = "{http://purl.org/rss/1.0/}" # RSS 1.0 / RDF (e.g. 4gamer.net) namespaces <item>/<title>/<link>
_TAG_RE = re.compile(r"<[^>]+>") _TAG_RE = re.compile(r"<[^>]+>")
@@ -36,22 +37,28 @@ def _clean_summary(raw: str, max_len: int = 300) -> str:
return re.sub(r"\s+", " ", text).strip()[:max_len] return re.sub(r"\s+", " ", text).strip()[:max_len]
def _rss_items(root: Any, ns: str, source: str) -> List[Dict[str, str]]:
"""RSS 2.0 (ns='') and RSS 1.0/RDF (ns=_RSS1) both use <item><title><link><description>."""
out: List[Dict[str, str]] = []
for item in root.iter(f"{ns}item"):
title = (item.findtext(f"{ns}title") or "").strip()
link = (item.findtext(f"{ns}link") or "").strip()
summary = _clean_summary(item.findtext(f"{ns}description") or "")
if title:
out.append({"title": title, "link": link, "source": source, "summary": summary})
return out
def parse_feed(data: bytes, source: str = "") -> List[Dict[str, str]]: def parse_feed(data: bytes, source: str = "") -> List[Dict[str, str]]:
"""Parse RSS or Atom bytes into [{title, link, source}] (tolerant).""" """Parse RSS 2.0, RSS 1.0/RDF, or Atom bytes into [{title, link, source, summary}] (tolerant)."""
try: try:
root = ElementTree.fromstring(data) root = ElementTree.fromstring(data)
except Exception as err: except Exception as err:
# malformed XML or a blocked entity/DTD attack — tolerate, never raise (NEWS-01) # malformed XML or a blocked entity/DTD attack — tolerate, never raise (NEWS-01)
logging.warning(f"news: unparseable/unsafe feed {source!r}: {err!r}") logging.warning(f"news: unparseable/unsafe feed {source!r}: {err!r}")
return [] return []
items: List[Dict[str, str]] = [] # RSS 2.0 (unqualified) + RSS 1.0/RDF (namespaced, e.g. 4gamer) share <item><title><link><description>
# RSS: <rss><channel><item><title/><link/><description/> items: List[Dict[str, str]] = _rss_items(root, "", source) + _rss_items(root, _RSS1, source)
for item in root.iter("item"):
title = (item.findtext("title") or "").strip()
link = (item.findtext("link") or "").strip()
summary = _clean_summary(item.findtext("description") or "")
if title:
items.append({"title": title, "link": link, "source": source, "summary": summary})
# Atom: <feed><entry><title/><link href=/><summary|content/> # Atom: <feed><entry><title/><link href=/><summary|content/>
for entry in root.iter(f"{_ATOM}entry"): for entry in root.iter(f"{_ATOM}entry"):
title = (entry.findtext(f"{_ATOM}title") or "").strip() title = (entry.findtext(f"{_ATOM}title") or "").strip()
@@ -184,9 +191,10 @@ class NewsPoster:
GET_NEWS_TOOL = { GET_NEWS_TOOL = {
"name": "get_news", "name": "get_news",
"description": "Fetch recent real-world news the bot has collected from its RSS feeds (local, national, world, sport, " "description": "Fetch news the bot has collected from its RSS feeds — this is the SAME news that gets posted in the "
"culture). Use when someone asks what is new or what is happening, optionally about a topic or from a particular " "server's news channels (e.g. #news, #newsjp / ニュース). Use this FIRST, before web_search, for anything about "
"source. Returns headlines with a short summary and a link to read more.", "current news or about something someone saw in a news channel; filter by topic (a keyword, also matches the source "
"label) or by source. Returns headlines with a short summary and a link; follow up with fetch_url for the full text.",
"parameters": { "parameters": {
"type": "object", "type": "object",
"properties": { "properties": {
+4 -3
View File
@@ -24,9 +24,10 @@ FETCH_TIMEOUT_S = 15
WEB_SEARCH_TOOL = { WEB_SEARCH_TOOL = {
"name": "web_search", "name": "web_search",
"description": "Search the open web for current information when the user asks you to look something up and it is not " "description": "Search the open web for general information. Use ONLY when the answer is not in your own sources: for "
"covered by game info (IGDB), the Codex, the news store, or a URL they pasted. Returns result titles, URLs, and a short " "the server's news use get_news, for Adeptus Mechanicus / Warhammer 40k lore use codex_search, for video-game facts use "
"snippet; follow up with fetch_url on a result link for the full article.", "the game tools, for a specific URL someone pasted use fetch_url. Returns result titles, URLs, and a short snippet; "
"follow up with fetch_url on a result link for the full article.",
"parameters": { "parameters": {
"type": "object", "type": "object",
"properties": { "properties": {
+1 -1
View File
@@ -6,7 +6,7 @@ Replaces the broken pre-1.0-openai `news_feed.py`. A CLI
`AIResponder.message` injects into the `{news}` slot. Feeds are `AIResponder.message` injects into the `{news}` slot. Feeds are
external input and operator-configured. external input and operator-configured.
### NEWS-01 — RSS and Atom parse to items (coverage: test) ### NEWS-01 — RSS (2.0 and 1.0/RDF) and Atom parse to items (coverage: test)
`parse_feed(bytes, label)` extracts `{title, link, source}` from both `parse_feed(bytes, label)` extracts `{title, link, source}` from both
RSS (`<item>`) and Atom (`<entry>`) documents, tolerates malformed RSS (`<item>`) and Atom (`<entry>`) documents, tolerates malformed
+2 -2
View File
@@ -1,8 +1,8 @@
# SPEC-014 — Codex Mechanicus search # SPEC-014 — Codex Mechanicus search
Luma is an Adeptus Mechanicus tech-priest; her lore has a real home — Luma is an Adeptus Mechanicus tech-priest; his lore has a real home —
the priest's own Codex Mechanicus at `binaric.tech` (an Astro/MDX the priest's own Codex Mechanicus at `binaric.tech` (an Astro/MDX
archive, five tongues). A `codex_search` function tool lets her consult archive, five tongues). A `codex_search` function tool lets him consult
that archive and answer from sourced inscriptions instead of inventing that archive and answer from sourced inscriptions instead of inventing
lore. The index is public but still untrusted by the time it reaches a lore. The index is public but still untrusted by the time it reaches a
prompt: the fetch is SSRF-guarded (SPEC-011 shares `guard_url`), prompt: the fetch is SSRF-guarded (SPEC-011 shares `guard_url`),
+17
View File
@@ -29,6 +29,16 @@ ATOM = b"""<?xml version="1.0"?><feed xmlns="http://www.w3.org/2005/Atom">
<entry><title>Atom headline</title><link href="https://ex.com/a"/></entry> <entry><title>Atom headline</title><link href="https://ex.com/a"/></entry>
</feed>""" </feed>"""
RSS1 = (
'<?xml version="1.0" encoding="UTF-8"?>'
'<rdf:RDF xmlns:rdf="http://www.w3.org/1999/02/22-rdf-syntax-ns#" xmlns="http://purl.org/rss/1.0/">'
'<channel rdf:about="https://ex.jp"><title>Feed</title></channel>'
'<item rdf:about="https://ex.jp/1"><title>ゲームニュース</title><link>https://ex.jp/1</link>'
"<description>本文ここ</description></item>"
'<item rdf:about="https://ex.jp/2"><title>Second</title><link>https://ex.jp/2</link></item>'
"</rdf:RDF>"
).encode("utf-8")
class TestParse(unittest.TestCase): class TestParse(unittest.TestCase):
def test_rss(self): def test_rss(self):
@@ -44,6 +54,13 @@ class TestParse(unittest.TestCase):
self.assertEqual(items[0]["title"], "Atom headline") self.assertEqual(items[0]["title"], "Atom headline")
self.assertEqual(items[0]["link"], "https://ex.com/a") self.assertEqual(items[0]["link"], "https://ex.com/a")
def test_rss1_rdf(self):
"""NEWS-01: RSS 1.0/RDF (namespaced <item>, e.g. 4gamer.net) parses like RSS 2.0."""
items = parse_feed(RSS1, "JP")
self.assertEqual([i["title"] for i in items], ["ゲームニュース", "Second"])
self.assertEqual(items[0]["link"], "https://ex.jp/1")
self.assertEqual(items[0]["summary"], "本文ここ")
def test_malformed_never_raises(self): def test_malformed_never_raises(self):
"""NEWS-01: garbage XML returns [] without raising.""" """NEWS-01: garbage XML returns [] without raising."""
self.assertEqual(parse_feed(b"<not xml", "bad"), []) self.assertEqual(parse_feed(b"<not xml", "bad"), [])