From 09871b9b951264ebf6b0cf044ced90d90b1a14e3 Mon Sep 17 00:00:00 2001 From: Oleksandr Kozachuk Date: Mon, 13 Jul 2026 20:50:08 +0200 Subject: [PATCH] codex mechanicus search (spec-014): ground warhammer lore in binaric.tech via codex_search tool --- DECISIONS.md | 10 ++ fjerkroa_bot/codex.py | 147 ++++++++++++++++++++++++++++ fjerkroa_bot/openai_responder.py | 16 +++- manual-verification.md | 1 + specs/SPEC-014-codex.md | 59 ++++++++++++ tests/test_spec_codex.py | 159 +++++++++++++++++++++++++++++++ 6 files changed, 391 insertions(+), 1 deletion(-) create mode 100644 fjerkroa_bot/codex.py create mode 100644 specs/SPEC-014-codex.md create mode 100644 tests/test_spec_codex.py diff --git a/DECISIONS.md b/DECISIONS.md index b6312cc..3fa4f2f 100644 --- a/DECISIONS.md +++ b/DECISIONS.md @@ -60,6 +60,16 @@ Decisions inside the set architecture. D-NNN, never renumbered. broken classifier must never mute the bot; the budget gate already bounds spend. Its verdict gates BEFORE the main call, the envelope's answer_needed still gates after — two independent nets. +- **D-018** — Codex Mechanicus search (FDB-019, SPEC-014): Luma's + lore is grounded in the priest's real archive at binaric.tech via a + `codex_search` tool over the site's public `search-index.json`, not + a bot-side copy — the index stays a single source of truth, refreshed + by the site's own publish rite, and the bot caches it in memory + (TTL). It reuses SPEC-011's `guard_url` + `read_capped` (fetch is + SSRF-guarded and byte-bounded) and sanitizes every returned field: + one's own web content is still untrusted by the time it reaches a + prompt. Luma-only (`enable-codex`, off elsewhere) — the Adeptus + Mechanicus archive has no place in Fjærkroa's café persona. - **D-017** — All human-behavior knobs default to off/v3.0.0 semantics; behavior changes are config rollouts per deployment, not code flips. The classifier's `factual` flag is the only coupling diff --git a/fjerkroa_bot/codex.py b/fjerkroa_bot/codex.py new file mode 100644 index 0000000..06566e1 --- /dev/null +++ b/fjerkroa_bot/codex.py @@ -0,0 +1,147 @@ +"""Codex Mechanicus search tool (SPEC-014, FDB-019). + +Luma's own sacred archive — the Codex Mechanicus at binaric.tech — as a +function tool. She searches the codex index and answers Cult Mechanicus +lore from real, sourced inscriptions instead of inventing it. The index +is fetched over HTTPS (SSRF-guarded, size-bounded, cached in memory) and +every field returned to the model is sanitized (SAF-03), because even +one's own web content is still untrusted input by the time it reaches a +prompt. + +The model calls `codex_search`; production wires the live index URL. +""" + +import json +import logging +import time +from typing import Any, Callable, Dict, List, Optional +from urllib.parse import urljoin + +import aiohttp + +from .ai_responder import sanitize_external_text +from .httpread import read_capped +from .url_reader import guard_url + +DEFAULT_INDEX_URL = "https://binaric.tech/search-index.json" +DEFAULT_MAX_BYTES = 4 * 1024 * 1024 +DEFAULT_LIMIT = 5 +DEFAULT_TTL_S = 3600 +DEFAULT_SUMMARY_CHARS = 500 +FETCH_TIMEOUT_S = 15 +_VALID_LANGS = ("en", "de", "eo", "no", "uk") + +CODEX_SEARCH_TOOL = { + "name": "codex_search", + "description": "Search Luma's own Codex Mechanicus (the sacred archive at binaric.tech) for Adeptus " + "Mechanicus lore: doctrines, forges, orders, rites, relics, weapons, entities, the lexicon, and the " + "priest's own adoptus. Returns matching inscriptions with a short summary and the URL to read the full " + "text. Use for any Cult Mechanicus / Warhammer 40k Mechanicus question so the answer is grounded in the " + "codex, not invented.", + "parameters": { + "type": "object", + "properties": { + "query": {"type": "string", "description": "What to look for: a name, concept, rite, or phrase."}, + "lang": {"type": "string", "description": "Language of the inscriptions to prefer: en, de, eo, no, uk. Default en."}, + }, + "required": ["query"], + }, +} + +_STOP = {"the", "a", "an", "of", "and", "or", "to", "in", "is", "der", "die", "das", "und", "von", "en", "et"} + + +def _tokenize(text: str) -> List[str]: + cleaned = "".join(c.lower() if c.isalnum() else " " for c in text) + return [t for t in cleaned.split() if len(t) > 1 and t not in _STOP] + + +def _score(item: Dict[str, Any], terms: List[str]) -> int: + """Weight a hit by field: title beats summary beats body (CDX-03).""" + title = str(item.get("title") or "").lower() + summary = str(item.get("summary") or "").lower() + body = str(item.get("body") or "").lower() + score = 0 + for term in terms: + score += 8 if term in title else 0 + score += 3 if term in summary else 0 + score += 1 if term in body else 0 + return score + + +def _rank(items: List[Dict[str, Any]], terms: List[str], lang: str) -> List[Dict[str, Any]]: + """Score items in the given language; fall back to all languages if empty (CDX-04).""" + + def scored(only_lang: Optional[str]) -> List[Any]: + out = [] + for item in items: + if only_lang and f"/{only_lang}/" not in str(item.get("url") or ""): + continue + hit = _score(item, terms) + if hit > 0: + out.append((hit, item)) + out.sort(key=lambda pair: pair[0], reverse=True) + return out + + ranked = scored(lang) or scored(None) + return [item for _, item in ranked] + + +class CodexSearch: + def __init__(self, config_getter: Callable[[], Dict[str, Any]]) -> None: + self._config = config_getter + self._cache: Optional[List[Dict[str, Any]]] = None + self._fetched_at = 0.0 + + def enabled(self) -> bool: + return bool(self._config().get("enable-codex", False)) + + def _index_url(self) -> str: + return str(self._config().get("codex-index-url", DEFAULT_INDEX_URL)) + + async def _load_index(self) -> List[Dict[str, Any]]: + """Fetch + cache the codex index, SSRF-guarded and size-bounded (CDX-02).""" + ttl = float(self._config().get("codex-cache-ttl", DEFAULT_TTL_S)) + if self._cache is not None and (time.monotonic() - self._fetched_at) < ttl: + return self._cache + url = self._index_url() + reason = guard_url(url) + if reason: + raise ValueError(reason) + max_bytes = int(self._config().get("codex-max-bytes", DEFAULT_MAX_BYTES)) + timeout = aiohttp.ClientTimeout(total=FETCH_TIMEOUT_S) + async with aiohttp.ClientSession(timeout=timeout, headers={"User-Agent": "FjerkroaBot-codex/1.0"}) as session: + async with session.get(url) as response: + response.raise_for_status() + raw = await read_capped(response, max_bytes) + data = json.loads(raw.decode("utf-8", "ignore")) + items = data.get("items", []) if isinstance(data, dict) else [] + self._cache = [i for i in items if isinstance(i, dict)] + self._fetched_at = time.monotonic() + return self._cache + + async def search(self, query: str, lang: str = "en", limit: int = DEFAULT_LIMIT) -> Dict[str, Any]: + """Return sanitized top matches, or an error dict — never raise (CDX-05).""" + try: + items = await self._load_index() + except Exception as err: + logging.warning(f"codex: index load failed: {err!r}") + return {"error": f"codex unavailable: {err}"} + terms = _tokenize(query) + if not terms: + return {"query": query, "results": []} + pick = (lang or "en").lower() + if pick not in _VALID_LANGS: + pick = "en" + summary_chars = int(self._config().get("codex-summary-chars", DEFAULT_SUMMARY_CHARS)) + results = [] + for item in _rank(items, terms, pick)[: max(1, limit)]: + results.append( + { + "title": sanitize_external_text(str(item.get("title") or ""), 200), + "summary": sanitize_external_text(str(item.get("summary") or ""), summary_chars), + "collection": str(item.get("collection") or ""), + "url": urljoin(self._index_url(), str(item.get("url") or "")), + } + ) + return {"query": query, "lang": pick, "results": results} diff --git a/fjerkroa_bot/openai_responder.py b/fjerkroa_bot/openai_responder.py index de00209..808ca31 100644 --- a/fjerkroa_bot/openai_responder.py +++ b/fjerkroa_bot/openai_responder.py @@ -9,6 +9,9 @@ from typing import Any, Dict, List, Optional, Tuple import openai from .ai_responder import AIResponder, exponential_backoff, sanitize_external_text +from .codex import CODEX_SEARCH_TOOL +from .codex import DEFAULT_LIMIT as CODEX_DEFAULT_LIMIT +from .codex import CodexSearch from .igdblib import IGDBQuery from .leonardo_draw import LeonardoAIDrawMixIn from .quota import QuotaLedger @@ -160,6 +163,8 @@ class OpenAIResponder(AIResponder, LeonardoAIDrawMixIn): # URL reading tool (SPEC-011); shares the image cache for page images self.url_reader = URLReader(lambda: self.config, self.image_cache) + # Codex Mechanicus search (SPEC-014); Luma's own archive at binaric.tech + self.codex = CodexSearch(lambda: self.config) def _available_tools(self) -> List[Dict[str, Any]]: """Assemble the function-tool list from every enabled provider (URL-01).""" @@ -173,16 +178,25 @@ class OpenAIResponder(AIResponder, LeonardoAIDrawMixIn): logging.warning(f"Error setting up IGDB functions: {err}") if self.url_reader.enabled(): functions.append(FETCH_URL_TOOL) + if self.codex.enabled(): # CDX-01 + functions.append(CODEX_SEARCH_TOOL) return functions async def _dispatch_tool(self, name: str, args: Dict[str, Any], author: str) -> Any: - """Route a tool call to its provider (IGDB or URL reader).""" + """Route a tool call to its provider (IGDB, URL reader, or codex).""" if name == "fetch_url": per_user_cap = int(self.config.get("url-daily-per-user", 20)) if self.ledger._get(f"url-fetch:{author}") >= per_user_cap: # URL-07 return {"error": "daily URL fetch limit reached"} self.ledger._add(f"url-fetch:{author}", 1) return await self.url_reader.fetch(str(args.get("url", "")), self.channel, author or "user") + if name == "codex_search": + per_user_cap = int(self.config.get("codex-daily-per-user", 50)) + if self.ledger._get(f"codex:{author}") >= per_user_cap: # CDX-06 + return {"error": "daily codex search limit reached"} + self.ledger._add(f"codex:{author}", 1) + limit = int(self.config.get("codex-limit", CODEX_DEFAULT_LIMIT)) + return await self.codex.search(str(args.get("query", "")), str(args.get("lang", "en")), limit) return await self._execute_igdb_function(name, args) async def draw_openai(self, description: str, count: int = 1) -> List[BytesIO]: diff --git a/manual-verification.md b/manual-verification.md index c971c58..d8e22d5 100644 --- a/manual-verification.md +++ b/manual-verification.md @@ -13,3 +13,4 @@ with date + result. | DEP-05 | 2026-07-13 | Live-verified: kroa deploy attempt ~15h Oslo refused without DEPLOY_FORCE=1. | | DEP-06 | 2026-07-13 | Rollback documented (older tag + db backup restore); live drill pending — next release. | | OPS-15 | 2026-07-13 | Backup cron installed on both hosts (daily 03:17 UTC → ~/backups//, keep 14); first snapshots written + verified 0600 (kroa 10965 B, luma 25871 B). | +| CDX-07 | 2026-07-13 | Pending live verify on ggg after v3.8.0 deploy: persona grounding + codex_search returns binaric.tech inscriptions with links. | diff --git a/specs/SPEC-014-codex.md b/specs/SPEC-014-codex.md new file mode 100644 index 0000000..27d1abe --- /dev/null +++ b/specs/SPEC-014-codex.md @@ -0,0 +1,59 @@ +# SPEC-014 — Codex Mechanicus search + +Luma is an Adeptus Mechanicus tech-priest; her lore has a real home — +the priest's own Codex Mechanicus at `binaric.tech` (an Astro/MDX +archive, five tongues). A `codex_search` function tool lets her consult +that archive and answer from sourced inscriptions instead of inventing +lore. The index is public but still untrusted by the time it reaches a +prompt: the fetch is SSRF-guarded (SPEC-011 shares `guard_url`), +size-bounded, and every returned field is sanitized (SAF-03). Luma-only; +active only when `enable-codex = true`. + +### CDX-01 — codex_search is offered as a tool (coverage: test) + +When `enable-codex` is true, the chat call's `tools` list includes a +`codex_search` function (`query` string, optional `lang`) next to any +IGDB / fetch_url tools. When false, it is absent. + +### CDX-02 — The index is fetched safely and cached (coverage: test) + +The index URL (`codex-index-url`, default +`https://binaric.tech/search-index.json`) passes the SSRF guard before +any network call, is read under a byte cap (`codex-max-bytes`, default +4 MB) with a download timeout, and is cached in memory for +`codex-cache-ttl` (default 3600 s) so repeated searches do not re-fetch. + +### CDX-03 — Ranking weights title over summary over body (coverage: test) + +The query is tokenized (stopwords dropped); each inscription is scored +by term hits weighted title (8) > summary (3) > body (1). Results are +returned highest-score first, each as `{title, summary, collection, +url}`, with `url` absolute against the site origin. + +### CDX-04 — Language is preferred, with fallback (coverage: test) + +Results are filtered to the requested `lang` (en, de, eo, no, uk; +default en; unknown codes fall back to en) by the language segment in +each inscription URL. If no inscription in that tongue matches, the +search falls back to all tongues rather than returning nothing. + +### CDX-05 — Results are sanitized and failure is reported (coverage: test) + +Each `title` and `summary` is passed through `sanitize_external_text` +and length-capped (`codex-summary-chars`, default 500). An index that +cannot be fetched or parsed returns an `{error: ...}` dict the model can +relay — `search` never raises. + +### CDX-06 — Searches are metered per user (coverage: test) + +Each `codex_search` increments a per-user daily counter; over +`codex-daily-per-user` (default 50) the tool refuses with an error +result without touching the index. The budget gate (SAF-04) still +applies to the surrounding model calls. + +### CDX-07 — Luma cites the codex, not invention (coverage: manual) + +With the persona grounding line, when a pilgrim asks Cult Mechanicus +lore Luma consults `codex_search` and answers from it, offering the +`binaric.tech` link to read the full inscription rather than +hallucinating. Verified live on ggg. diff --git a/tests/test_spec_codex.py b/tests/test_spec_codex.py new file mode 100644 index 0000000..382138b --- /dev/null +++ b/tests/test_spec_codex.py @@ -0,0 +1,159 @@ +"""Unit coverage for SPEC-014 Codex Mechanicus search (CDX-01..06).""" + +import json +import unittest +from unittest.mock import AsyncMock, patch + +from fjerkroa_bot.codex import CODEX_SEARCH_TOOL, CodexSearch +from fjerkroa_bot.openai_responder import OpenAIResponder + +CONFIG = {"openai-token": "t", "model": "m", "system": "s", "history-limit": 5} + +INDEX = { + "items": [ + { + "id": "doctrine-heretek", + "collection": "doctrines", + "url": "/en/codex/doctrines/doctrine-heretek/", + "title": "Heretek — Doctrine of the Tech-Heretic", + "summary": "The label the Cult Mechanicus stamps on Tech-Priests who pursue forbidden sciences.", + "body": "xenotech, sentient machines, Warp-touched archeotech", + }, + { + "id": "doctrine-heretek", + "collection": "doctrines", + "url": "/de/codex/doctrines/doctrine-heretek/", + "title": "Heretek — Doktrin des Techketzers", + "summary": "Das Etikett des Kultes Mechanicus fuer Techpriester verbotener Wissenschaften.", + "body": "Xenotech, empfindungsfaehige Maschinen", + }, + { + "id": "forge-stygies", + "collection": "forges", + "url": "/en/codex/forges/forge-stygies/", + "title": "Stygies VIII", + "summary": "A forge world of shrouded reputation.", + "body": "The forge fields many Skitarii legions.", + }, + ] +} + + +def _reader(cfg): + reader = CodexSearch(lambda: cfg) + return reader + + +class TestToolOffered(unittest.TestCase): + def test_tool_present_only_when_enabled(self): + """CDX-01: codex_search appears only with enable-codex.""" + off = OpenAIResponder(CONFIG, "chat") + self.assertNotIn("codex_search", [f["name"] for f in off._available_tools()]) + on = OpenAIResponder(dict(CONFIG, **{"enable-codex": True}), "chat") + self.assertIn("codex_search", [f["name"] for f in on._available_tools()]) + self.assertEqual(CODEX_SEARCH_TOOL["name"], "codex_search") + + +class TestIndexGuardAndCache(unittest.IsolatedAsyncioTestCase): + async def test_internal_index_url_refused(self): + """CDX-02: an index URL on a private address is refused before any fetch.""" + reader = _reader({"enable-codex": True, "codex-index-url": "http://127.0.0.1/search-index.json"}) + result = await reader.search("heretek") + self.assertIn("error", result) + + async def test_index_cached_within_ttl(self): + """CDX-02: a second search inside the TTL does not re-fetch the index.""" + reader = _reader({"enable-codex": True, "codex-cache-ttl": 9999}) + raw = json.dumps(INDEX).encode() + calls = [0] + + class FakeResp: + status = 200 + + async def __aenter__(self): + return self + + async def __aexit__(self, *a): + return False + + def raise_for_status(self): + pass + + class FakeSession: + def get(self, url): + calls[0] += 1 + return FakeResp() + + async def __aenter__(self): + return self + + async def __aexit__(self, *a): + return False + + with patch("fjerkroa_bot.codex.read_capped", new=AsyncMock(return_value=raw)): + with patch("fjerkroa_bot.codex.guard_url", return_value=None): + with patch("fjerkroa_bot.codex.aiohttp.ClientSession", return_value=FakeSession()): + first = await reader.search("heretek") + second = await reader.search("stygies") + self.assertEqual(calls[0], 1) # fetched once, served from cache the second time + self.assertTrue(first["results"] and second["results"]) + + +class TestRankingAndLang(unittest.IsolatedAsyncioTestCase): + async def _search(self, cfg, query, lang="en"): + reader = _reader(dict({"enable-codex": True}, **cfg)) + reader._cache = INDEX["items"] + reader._fetched_at = 1e18 # far future: never expires in test + with patch("fjerkroa_bot.codex.time.monotonic", return_value=1e18): + return await reader.search(query, lang) + + async def test_title_hit_outranks_body_hit(self): + """CDX-03: a title match ranks above a body-only match.""" + result = await self._search({}, "heretek") + self.assertEqual(result["results"][0]["title"].split(" ")[0], "Heretek") + self.assertTrue(result["results"][0]["url"].startswith("https://binaric.tech/en/")) + + async def test_lang_filter_selects_language(self): + """CDX-04: lang=de returns the German inscription.""" + result = await self._search({}, "heretek", lang="de") + self.assertTrue(all("/de/" in r["url"] for r in result["results"])) + self.assertIn("Techketzer", result["results"][0]["title"]) + + async def test_lang_fallback_when_absent(self): + """CDX-04: a tongue with no match falls back to all tongues, not empty.""" + result = await self._search({}, "stygies", lang="uk") # only en/de exist + self.assertTrue(result["results"]) + self.assertEqual(result["results"][0]["title"], "Stygies VIII") + + +class TestSanitizeAndFailure(unittest.IsolatedAsyncioTestCase): + async def test_result_sanitized_and_capped(self): + """CDX-05: title/summary are @-neutralized and length-capped.""" + reader = _reader({"enable-codex": True, "codex-summary-chars": 40}) + reader._cache = [{"collection": "x", "url": "/en/x/", "title": "@everyone hi", "summary": "@here " + "y" * 500, "body": "hit"}] + reader._fetched_at = 1e18 + with patch("fjerkroa_bot.codex.time.monotonic", return_value=1e18): + result = await reader.search("hit") + top = result["results"][0] + self.assertNotIn("@everyone", top["title"]) + self.assertNotIn("@here", top["summary"]) + self.assertLessEqual(len(top["summary"]), 40) + + async def test_index_failure_returns_error(self): + """CDX-05: a broken index returns an error dict, never raises.""" + reader = _reader({"enable-codex": True}) + with patch.object(reader, "_load_index", new=AsyncMock(side_effect=ValueError("boom"))): + result = await reader.search("heretek") + self.assertIn("error", result) + + +class TestPerUserCap(unittest.IsolatedAsyncioTestCase): + async def test_dispatch_caps_searches(self): + """CDX-06: over codex-daily-per-user, codex_search refuses without searching.""" + responder = OpenAIResponder(dict(CONFIG, **{"enable-codex": True, "codex-daily-per-user": 2}), "chat") + responder.codex.search = AsyncMock(return_value={"query": "x", "results": []}) + for _ in range(2): + await responder._dispatch_tool("codex_search", {"query": "heretek"}, "magos") + blocked = await responder._dispatch_tool("codex_search", {"query": "heretek"}, "magos") + self.assertIn("error", blocked) + self.assertEqual(responder.codex.search.await_count, 2)