codex mechanicus search (spec-014): ground warhammer lore in binaric.tech via codex_search tool
This commit is contained in:
@@ -0,0 +1,147 @@
|
||||
"""Codex Mechanicus search tool (SPEC-014, FDB-019).
|
||||
|
||||
Luma's own sacred archive — the Codex Mechanicus at binaric.tech — as a
|
||||
function tool. She searches the codex index and answers Cult Mechanicus
|
||||
lore from real, sourced inscriptions instead of inventing it. The index
|
||||
is fetched over HTTPS (SSRF-guarded, size-bounded, cached in memory) and
|
||||
every field returned to the model is sanitized (SAF-03), because even
|
||||
one's own web content is still untrusted input by the time it reaches a
|
||||
prompt.
|
||||
|
||||
The model calls `codex_search`; production wires the live index URL.
|
||||
"""
|
||||
|
||||
import json
|
||||
import logging
|
||||
import time
|
||||
from typing import Any, Callable, Dict, List, Optional
|
||||
from urllib.parse import urljoin
|
||||
|
||||
import aiohttp
|
||||
|
||||
from .ai_responder import sanitize_external_text
|
||||
from .httpread import read_capped
|
||||
from .url_reader import guard_url
|
||||
|
||||
DEFAULT_INDEX_URL = "https://binaric.tech/search-index.json"
|
||||
DEFAULT_MAX_BYTES = 4 * 1024 * 1024
|
||||
DEFAULT_LIMIT = 5
|
||||
DEFAULT_TTL_S = 3600
|
||||
DEFAULT_SUMMARY_CHARS = 500
|
||||
FETCH_TIMEOUT_S = 15
|
||||
_VALID_LANGS = ("en", "de", "eo", "no", "uk")
|
||||
|
||||
CODEX_SEARCH_TOOL = {
|
||||
"name": "codex_search",
|
||||
"description": "Search Luma's own Codex Mechanicus (the sacred archive at binaric.tech) for Adeptus "
|
||||
"Mechanicus lore: doctrines, forges, orders, rites, relics, weapons, entities, the lexicon, and the "
|
||||
"priest's own adoptus. Returns matching inscriptions with a short summary and the URL to read the full "
|
||||
"text. Use for any Cult Mechanicus / Warhammer 40k Mechanicus question so the answer is grounded in the "
|
||||
"codex, not invented.",
|
||||
"parameters": {
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"query": {"type": "string", "description": "What to look for: a name, concept, rite, or phrase."},
|
||||
"lang": {"type": "string", "description": "Language of the inscriptions to prefer: en, de, eo, no, uk. Default en."},
|
||||
},
|
||||
"required": ["query"],
|
||||
},
|
||||
}
|
||||
|
||||
_STOP = {"the", "a", "an", "of", "and", "or", "to", "in", "is", "der", "die", "das", "und", "von", "en", "et"}
|
||||
|
||||
|
||||
def _tokenize(text: str) -> List[str]:
|
||||
cleaned = "".join(c.lower() if c.isalnum() else " " for c in text)
|
||||
return [t for t in cleaned.split() if len(t) > 1 and t not in _STOP]
|
||||
|
||||
|
||||
def _score(item: Dict[str, Any], terms: List[str]) -> int:
|
||||
"""Weight a hit by field: title beats summary beats body (CDX-03)."""
|
||||
title = str(item.get("title") or "").lower()
|
||||
summary = str(item.get("summary") or "").lower()
|
||||
body = str(item.get("body") or "").lower()
|
||||
score = 0
|
||||
for term in terms:
|
||||
score += 8 if term in title else 0
|
||||
score += 3 if term in summary else 0
|
||||
score += 1 if term in body else 0
|
||||
return score
|
||||
|
||||
|
||||
def _rank(items: List[Dict[str, Any]], terms: List[str], lang: str) -> List[Dict[str, Any]]:
|
||||
"""Score items in the given language; fall back to all languages if empty (CDX-04)."""
|
||||
|
||||
def scored(only_lang: Optional[str]) -> List[Any]:
|
||||
out = []
|
||||
for item in items:
|
||||
if only_lang and f"/{only_lang}/" not in str(item.get("url") or ""):
|
||||
continue
|
||||
hit = _score(item, terms)
|
||||
if hit > 0:
|
||||
out.append((hit, item))
|
||||
out.sort(key=lambda pair: pair[0], reverse=True)
|
||||
return out
|
||||
|
||||
ranked = scored(lang) or scored(None)
|
||||
return [item for _, item in ranked]
|
||||
|
||||
|
||||
class CodexSearch:
|
||||
def __init__(self, config_getter: Callable[[], Dict[str, Any]]) -> None:
|
||||
self._config = config_getter
|
||||
self._cache: Optional[List[Dict[str, Any]]] = None
|
||||
self._fetched_at = 0.0
|
||||
|
||||
def enabled(self) -> bool:
|
||||
return bool(self._config().get("enable-codex", False))
|
||||
|
||||
def _index_url(self) -> str:
|
||||
return str(self._config().get("codex-index-url", DEFAULT_INDEX_URL))
|
||||
|
||||
async def _load_index(self) -> List[Dict[str, Any]]:
|
||||
"""Fetch + cache the codex index, SSRF-guarded and size-bounded (CDX-02)."""
|
||||
ttl = float(self._config().get("codex-cache-ttl", DEFAULT_TTL_S))
|
||||
if self._cache is not None and (time.monotonic() - self._fetched_at) < ttl:
|
||||
return self._cache
|
||||
url = self._index_url()
|
||||
reason = guard_url(url)
|
||||
if reason:
|
||||
raise ValueError(reason)
|
||||
max_bytes = int(self._config().get("codex-max-bytes", DEFAULT_MAX_BYTES))
|
||||
timeout = aiohttp.ClientTimeout(total=FETCH_TIMEOUT_S)
|
||||
async with aiohttp.ClientSession(timeout=timeout, headers={"User-Agent": "FjerkroaBot-codex/1.0"}) as session:
|
||||
async with session.get(url) as response:
|
||||
response.raise_for_status()
|
||||
raw = await read_capped(response, max_bytes)
|
||||
data = json.loads(raw.decode("utf-8", "ignore"))
|
||||
items = data.get("items", []) if isinstance(data, dict) else []
|
||||
self._cache = [i for i in items if isinstance(i, dict)]
|
||||
self._fetched_at = time.monotonic()
|
||||
return self._cache
|
||||
|
||||
async def search(self, query: str, lang: str = "en", limit: int = DEFAULT_LIMIT) -> Dict[str, Any]:
|
||||
"""Return sanitized top matches, or an error dict — never raise (CDX-05)."""
|
||||
try:
|
||||
items = await self._load_index()
|
||||
except Exception as err:
|
||||
logging.warning(f"codex: index load failed: {err!r}")
|
||||
return {"error": f"codex unavailable: {err}"}
|
||||
terms = _tokenize(query)
|
||||
if not terms:
|
||||
return {"query": query, "results": []}
|
||||
pick = (lang or "en").lower()
|
||||
if pick not in _VALID_LANGS:
|
||||
pick = "en"
|
||||
summary_chars = int(self._config().get("codex-summary-chars", DEFAULT_SUMMARY_CHARS))
|
||||
results = []
|
||||
for item in _rank(items, terms, pick)[: max(1, limit)]:
|
||||
results.append(
|
||||
{
|
||||
"title": sanitize_external_text(str(item.get("title") or ""), 200),
|
||||
"summary": sanitize_external_text(str(item.get("summary") or ""), summary_chars),
|
||||
"collection": str(item.get("collection") or ""),
|
||||
"url": urljoin(self._index_url(), str(item.get("url") or "")),
|
||||
}
|
||||
)
|
||||
return {"query": query, "lang": pick, "results": results}
|
||||
@@ -9,6 +9,9 @@ from typing import Any, Dict, List, Optional, Tuple
|
||||
import openai
|
||||
|
||||
from .ai_responder import AIResponder, exponential_backoff, sanitize_external_text
|
||||
from .codex import CODEX_SEARCH_TOOL
|
||||
from .codex import DEFAULT_LIMIT as CODEX_DEFAULT_LIMIT
|
||||
from .codex import CodexSearch
|
||||
from .igdblib import IGDBQuery
|
||||
from .leonardo_draw import LeonardoAIDrawMixIn
|
||||
from .quota import QuotaLedger
|
||||
@@ -160,6 +163,8 @@ class OpenAIResponder(AIResponder, LeonardoAIDrawMixIn):
|
||||
|
||||
# URL reading tool (SPEC-011); shares the image cache for page images
|
||||
self.url_reader = URLReader(lambda: self.config, self.image_cache)
|
||||
# Codex Mechanicus search (SPEC-014); Luma's own archive at binaric.tech
|
||||
self.codex = CodexSearch(lambda: self.config)
|
||||
|
||||
def _available_tools(self) -> List[Dict[str, Any]]:
|
||||
"""Assemble the function-tool list from every enabled provider (URL-01)."""
|
||||
@@ -173,16 +178,25 @@ class OpenAIResponder(AIResponder, LeonardoAIDrawMixIn):
|
||||
logging.warning(f"Error setting up IGDB functions: {err}")
|
||||
if self.url_reader.enabled():
|
||||
functions.append(FETCH_URL_TOOL)
|
||||
if self.codex.enabled(): # CDX-01
|
||||
functions.append(CODEX_SEARCH_TOOL)
|
||||
return functions
|
||||
|
||||
async def _dispatch_tool(self, name: str, args: Dict[str, Any], author: str) -> Any:
|
||||
"""Route a tool call to its provider (IGDB or URL reader)."""
|
||||
"""Route a tool call to its provider (IGDB, URL reader, or codex)."""
|
||||
if name == "fetch_url":
|
||||
per_user_cap = int(self.config.get("url-daily-per-user", 20))
|
||||
if self.ledger._get(f"url-fetch:{author}") >= per_user_cap: # URL-07
|
||||
return {"error": "daily URL fetch limit reached"}
|
||||
self.ledger._add(f"url-fetch:{author}", 1)
|
||||
return await self.url_reader.fetch(str(args.get("url", "")), self.channel, author or "user")
|
||||
if name == "codex_search":
|
||||
per_user_cap = int(self.config.get("codex-daily-per-user", 50))
|
||||
if self.ledger._get(f"codex:{author}") >= per_user_cap: # CDX-06
|
||||
return {"error": "daily codex search limit reached"}
|
||||
self.ledger._add(f"codex:{author}", 1)
|
||||
limit = int(self.config.get("codex-limit", CODEX_DEFAULT_LIMIT))
|
||||
return await self.codex.search(str(args.get("query", "")), str(args.get("lang", "en")), limit)
|
||||
return await self._execute_igdb_function(name, args)
|
||||
|
||||
async def draw_openai(self, description: str, count: int = 1) -> List[BytesIO]:
|
||||
|
||||
Reference in New Issue
Block a user