Compare commits
2 Commits
| Author | SHA1 | Date | |
|---|---|---|---|
| 6b61ed6175 | |||
| 80000528d3 |
@@ -1,7 +1,7 @@
|
||||
"""Codex Mechanicus search tool (SPEC-014, FDB-019).
|
||||
|
||||
Luma's own sacred archive — the Codex Mechanicus at binaric.tech — as a
|
||||
function tool. She searches the codex index and answers Cult Mechanicus
|
||||
function tool. He searches the codex index and answers Cult Mechanicus
|
||||
lore from real, sourced inscriptions instead of inventing it. The index
|
||||
is fetched over HTTPS (SSRF-guarded, size-bounded, cached in memory) and
|
||||
every field returned to the model is sanitized (SAF-03), because even
|
||||
|
||||
+20
-12
@@ -27,6 +27,7 @@ DEFAULT_SUMMARY_CHARS = 200
|
||||
DEFAULT_NEWS_KEEP = 400
|
||||
FETCH_TIMEOUT_S = 15
|
||||
_ATOM = "{http://www.w3.org/2005/Atom}"
|
||||
_RSS1 = "{http://purl.org/rss/1.0/}" # RSS 1.0 / RDF (e.g. 4gamer.net) namespaces <item>/<title>/<link>
|
||||
_TAG_RE = re.compile(r"<[^>]+>")
|
||||
|
||||
|
||||
@@ -36,22 +37,28 @@ def _clean_summary(raw: str, max_len: int = 300) -> str:
|
||||
return re.sub(r"\s+", " ", text).strip()[:max_len]
|
||||
|
||||
|
||||
def _rss_items(root: Any, ns: str, source: str) -> List[Dict[str, str]]:
|
||||
"""RSS 2.0 (ns='') and RSS 1.0/RDF (ns=_RSS1) both use <item><title><link><description>."""
|
||||
out: List[Dict[str, str]] = []
|
||||
for item in root.iter(f"{ns}item"):
|
||||
title = (item.findtext(f"{ns}title") or "").strip()
|
||||
link = (item.findtext(f"{ns}link") or "").strip()
|
||||
summary = _clean_summary(item.findtext(f"{ns}description") or "")
|
||||
if title:
|
||||
out.append({"title": title, "link": link, "source": source, "summary": summary})
|
||||
return out
|
||||
|
||||
|
||||
def parse_feed(data: bytes, source: str = "") -> List[Dict[str, str]]:
|
||||
"""Parse RSS or Atom bytes into [{title, link, source}] (tolerant)."""
|
||||
"""Parse RSS 2.0, RSS 1.0/RDF, or Atom bytes into [{title, link, source, summary}] (tolerant)."""
|
||||
try:
|
||||
root = ElementTree.fromstring(data)
|
||||
except Exception as err:
|
||||
# malformed XML or a blocked entity/DTD attack — tolerate, never raise (NEWS-01)
|
||||
logging.warning(f"news: unparseable/unsafe feed {source!r}: {err!r}")
|
||||
return []
|
||||
items: List[Dict[str, str]] = []
|
||||
# RSS: <rss><channel><item><title/><link/><description/>
|
||||
for item in root.iter("item"):
|
||||
title = (item.findtext("title") or "").strip()
|
||||
link = (item.findtext("link") or "").strip()
|
||||
summary = _clean_summary(item.findtext("description") or "")
|
||||
if title:
|
||||
items.append({"title": title, "link": link, "source": source, "summary": summary})
|
||||
# RSS 2.0 (unqualified) + RSS 1.0/RDF (namespaced, e.g. 4gamer) share <item><title><link><description>
|
||||
items: List[Dict[str, str]] = _rss_items(root, "", source) + _rss_items(root, _RSS1, source)
|
||||
# Atom: <feed><entry><title/><link href=/><summary|content/>
|
||||
for entry in root.iter(f"{_ATOM}entry"):
|
||||
title = (entry.findtext(f"{_ATOM}title") or "").strip()
|
||||
@@ -184,9 +191,10 @@ class NewsPoster:
|
||||
|
||||
GET_NEWS_TOOL = {
|
||||
"name": "get_news",
|
||||
"description": "Fetch recent real-world news the bot has collected from its RSS feeds (local, national, world, sport, "
|
||||
"culture). Use when someone asks what is new or what is happening, optionally about a topic or from a particular "
|
||||
"source. Returns headlines with a short summary and a link to read more.",
|
||||
"description": "Fetch news the bot has collected from its RSS feeds — this is the SAME news that gets posted in the "
|
||||
"server's news channels (e.g. #news, #newsjp / ニュース). Use this FIRST, before web_search, for anything about "
|
||||
"current news or about something someone saw in a news channel; filter by topic (a keyword, also matches the source "
|
||||
"label) or by source. Returns headlines with a short summary and a link; follow up with fetch_url for the full text.",
|
||||
"parameters": {
|
||||
"type": "object",
|
||||
"properties": {
|
||||
|
||||
@@ -24,9 +24,10 @@ FETCH_TIMEOUT_S = 15
|
||||
|
||||
WEB_SEARCH_TOOL = {
|
||||
"name": "web_search",
|
||||
"description": "Search the open web for current information when the user asks you to look something up and it is not "
|
||||
"covered by game info (IGDB), the Codex, the news store, or a URL they pasted. Returns result titles, URLs, and a short "
|
||||
"snippet; follow up with fetch_url on a result link for the full article.",
|
||||
"description": "Search the open web for general information. Use ONLY when the answer is not in your own sources: for "
|
||||
"the server's news use get_news, for Adeptus Mechanicus / Warhammer 40k lore use codex_search, for video-game facts use "
|
||||
"the game tools, for a specific URL someone pasted use fetch_url. Returns result titles, URLs, and a short snippet; "
|
||||
"follow up with fetch_url on a result link for the full article.",
|
||||
"parameters": {
|
||||
"type": "object",
|
||||
"properties": {
|
||||
|
||||
@@ -6,7 +6,7 @@ Replaces the broken pre-1.0-openai `news_feed.py`. A CLI
|
||||
`AIResponder.message` injects into the `{news}` slot. Feeds are
|
||||
external input and operator-configured.
|
||||
|
||||
### NEWS-01 — RSS and Atom parse to items (coverage: test)
|
||||
### NEWS-01 — RSS (2.0 and 1.0/RDF) and Atom parse to items (coverage: test)
|
||||
|
||||
`parse_feed(bytes, label)` extracts `{title, link, source}` from both
|
||||
RSS (`<item>`) and Atom (`<entry>`) documents, tolerates malformed
|
||||
|
||||
@@ -1,8 +1,8 @@
|
||||
# SPEC-014 — Codex Mechanicus search
|
||||
|
||||
Luma is an Adeptus Mechanicus tech-priest; her lore has a real home —
|
||||
Luma is an Adeptus Mechanicus tech-priest; his lore has a real home —
|
||||
the priest's own Codex Mechanicus at `binaric.tech` (an Astro/MDX
|
||||
archive, five tongues). A `codex_search` function tool lets her consult
|
||||
archive, five tongues). A `codex_search` function tool lets him consult
|
||||
that archive and answer from sourced inscriptions instead of inventing
|
||||
lore. The index is public but still untrusted by the time it reaches a
|
||||
prompt: the fetch is SSRF-guarded (SPEC-011 shares `guard_url`),
|
||||
|
||||
@@ -29,6 +29,16 @@ ATOM = b"""<?xml version="1.0"?><feed xmlns="http://www.w3.org/2005/Atom">
|
||||
<entry><title>Atom headline</title><link href="https://ex.com/a"/></entry>
|
||||
</feed>"""
|
||||
|
||||
RSS1 = (
|
||||
'<?xml version="1.0" encoding="UTF-8"?>'
|
||||
'<rdf:RDF xmlns:rdf="http://www.w3.org/1999/02/22-rdf-syntax-ns#" xmlns="http://purl.org/rss/1.0/">'
|
||||
'<channel rdf:about="https://ex.jp"><title>Feed</title></channel>'
|
||||
'<item rdf:about="https://ex.jp/1"><title>ゲームニュース</title><link>https://ex.jp/1</link>'
|
||||
"<description>本文ここ</description></item>"
|
||||
'<item rdf:about="https://ex.jp/2"><title>Second</title><link>https://ex.jp/2</link></item>'
|
||||
"</rdf:RDF>"
|
||||
).encode("utf-8")
|
||||
|
||||
|
||||
class TestParse(unittest.TestCase):
|
||||
def test_rss(self):
|
||||
@@ -44,6 +54,13 @@ class TestParse(unittest.TestCase):
|
||||
self.assertEqual(items[0]["title"], "Atom headline")
|
||||
self.assertEqual(items[0]["link"], "https://ex.com/a")
|
||||
|
||||
def test_rss1_rdf(self):
|
||||
"""NEWS-01: RSS 1.0/RDF (namespaced <item>, e.g. 4gamer.net) parses like RSS 2.0."""
|
||||
items = parse_feed(RSS1, "JP")
|
||||
self.assertEqual([i["title"] for i in items], ["ゲームニュース", "Second"])
|
||||
self.assertEqual(items[0]["link"], "https://ex.jp/1")
|
||||
self.assertEqual(items[0]["summary"], "本文ここ")
|
||||
|
||||
def test_malformed_never_raises(self):
|
||||
"""NEWS-01: garbage XML returns [] without raising."""
|
||||
self.assertEqual(parse_feed(b"<not xml", "bad"), [])
|
||||
|
||||
Reference in New Issue
Block a user