Compare commits
2 Commits
| Author | SHA1 | Date | |
|---|---|---|---|
| 09871b9b95 | |||
| a514ff652c |
@@ -60,6 +60,16 @@ Decisions inside the set architecture. D-NNN, never renumbered.
|
||||
broken classifier must never mute the bot; the budget gate already
|
||||
bounds spend. Its verdict gates BEFORE the main call, the
|
||||
envelope's answer_needed still gates after — two independent nets.
|
||||
- **D-018** — Codex Mechanicus search (FDB-019, SPEC-014): Luma's
|
||||
lore is grounded in the priest's real archive at binaric.tech via a
|
||||
`codex_search` tool over the site's public `search-index.json`, not
|
||||
a bot-side copy — the index stays a single source of truth, refreshed
|
||||
by the site's own publish rite, and the bot caches it in memory
|
||||
(TTL). It reuses SPEC-011's `guard_url` + `read_capped` (fetch is
|
||||
SSRF-guarded and byte-bounded) and sanitizes every returned field:
|
||||
one's own web content is still untrusted by the time it reaches a
|
||||
prompt. Luma-only (`enable-codex`, off elsewhere) — the Adeptus
|
||||
Mechanicus archive has no place in Fjærkroa's café persona.
|
||||
- **D-017** — All human-behavior knobs default to off/v3.0.0
|
||||
semantics; behavior changes are config rollouts per deployment, not
|
||||
code flips. The classifier's `factual` flag is the only coupling
|
||||
|
||||
@@ -0,0 +1,147 @@
|
||||
"""Codex Mechanicus search tool (SPEC-014, FDB-019).
|
||||
|
||||
Luma's own sacred archive — the Codex Mechanicus at binaric.tech — as a
|
||||
function tool. She searches the codex index and answers Cult Mechanicus
|
||||
lore from real, sourced inscriptions instead of inventing it. The index
|
||||
is fetched over HTTPS (SSRF-guarded, size-bounded, cached in memory) and
|
||||
every field returned to the model is sanitized (SAF-03), because even
|
||||
one's own web content is still untrusted input by the time it reaches a
|
||||
prompt.
|
||||
|
||||
The model calls `codex_search`; production wires the live index URL.
|
||||
"""
|
||||
|
||||
import json
|
||||
import logging
|
||||
import time
|
||||
from typing import Any, Callable, Dict, List, Optional
|
||||
from urllib.parse import urljoin
|
||||
|
||||
import aiohttp
|
||||
|
||||
from .ai_responder import sanitize_external_text
|
||||
from .httpread import read_capped
|
||||
from .url_reader import guard_url
|
||||
|
||||
DEFAULT_INDEX_URL = "https://binaric.tech/search-index.json"
|
||||
DEFAULT_MAX_BYTES = 4 * 1024 * 1024
|
||||
DEFAULT_LIMIT = 5
|
||||
DEFAULT_TTL_S = 3600
|
||||
DEFAULT_SUMMARY_CHARS = 500
|
||||
FETCH_TIMEOUT_S = 15
|
||||
_VALID_LANGS = ("en", "de", "eo", "no", "uk")
|
||||
|
||||
CODEX_SEARCH_TOOL = {
|
||||
"name": "codex_search",
|
||||
"description": "Search Luma's own Codex Mechanicus (the sacred archive at binaric.tech) for Adeptus "
|
||||
"Mechanicus lore: doctrines, forges, orders, rites, relics, weapons, entities, the lexicon, and the "
|
||||
"priest's own adoptus. Returns matching inscriptions with a short summary and the URL to read the full "
|
||||
"text. Use for any Cult Mechanicus / Warhammer 40k Mechanicus question so the answer is grounded in the "
|
||||
"codex, not invented.",
|
||||
"parameters": {
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"query": {"type": "string", "description": "What to look for: a name, concept, rite, or phrase."},
|
||||
"lang": {"type": "string", "description": "Language of the inscriptions to prefer: en, de, eo, no, uk. Default en."},
|
||||
},
|
||||
"required": ["query"],
|
||||
},
|
||||
}
|
||||
|
||||
_STOP = {"the", "a", "an", "of", "and", "or", "to", "in", "is", "der", "die", "das", "und", "von", "en", "et"}
|
||||
|
||||
|
||||
def _tokenize(text: str) -> List[str]:
|
||||
cleaned = "".join(c.lower() if c.isalnum() else " " for c in text)
|
||||
return [t for t in cleaned.split() if len(t) > 1 and t not in _STOP]
|
||||
|
||||
|
||||
def _score(item: Dict[str, Any], terms: List[str]) -> int:
|
||||
"""Weight a hit by field: title beats summary beats body (CDX-03)."""
|
||||
title = str(item.get("title") or "").lower()
|
||||
summary = str(item.get("summary") or "").lower()
|
||||
body = str(item.get("body") or "").lower()
|
||||
score = 0
|
||||
for term in terms:
|
||||
score += 8 if term in title else 0
|
||||
score += 3 if term in summary else 0
|
||||
score += 1 if term in body else 0
|
||||
return score
|
||||
|
||||
|
||||
def _rank(items: List[Dict[str, Any]], terms: List[str], lang: str) -> List[Dict[str, Any]]:
|
||||
"""Score items in the given language; fall back to all languages if empty (CDX-04)."""
|
||||
|
||||
def scored(only_lang: Optional[str]) -> List[Any]:
|
||||
out = []
|
||||
for item in items:
|
||||
if only_lang and f"/{only_lang}/" not in str(item.get("url") or ""):
|
||||
continue
|
||||
hit = _score(item, terms)
|
||||
if hit > 0:
|
||||
out.append((hit, item))
|
||||
out.sort(key=lambda pair: pair[0], reverse=True)
|
||||
return out
|
||||
|
||||
ranked = scored(lang) or scored(None)
|
||||
return [item for _, item in ranked]
|
||||
|
||||
|
||||
class CodexSearch:
|
||||
def __init__(self, config_getter: Callable[[], Dict[str, Any]]) -> None:
|
||||
self._config = config_getter
|
||||
self._cache: Optional[List[Dict[str, Any]]] = None
|
||||
self._fetched_at = 0.0
|
||||
|
||||
def enabled(self) -> bool:
|
||||
return bool(self._config().get("enable-codex", False))
|
||||
|
||||
def _index_url(self) -> str:
|
||||
return str(self._config().get("codex-index-url", DEFAULT_INDEX_URL))
|
||||
|
||||
async def _load_index(self) -> List[Dict[str, Any]]:
|
||||
"""Fetch + cache the codex index, SSRF-guarded and size-bounded (CDX-02)."""
|
||||
ttl = float(self._config().get("codex-cache-ttl", DEFAULT_TTL_S))
|
||||
if self._cache is not None and (time.monotonic() - self._fetched_at) < ttl:
|
||||
return self._cache
|
||||
url = self._index_url()
|
||||
reason = guard_url(url)
|
||||
if reason:
|
||||
raise ValueError(reason)
|
||||
max_bytes = int(self._config().get("codex-max-bytes", DEFAULT_MAX_BYTES))
|
||||
timeout = aiohttp.ClientTimeout(total=FETCH_TIMEOUT_S)
|
||||
async with aiohttp.ClientSession(timeout=timeout, headers={"User-Agent": "FjerkroaBot-codex/1.0"}) as session:
|
||||
async with session.get(url) as response:
|
||||
response.raise_for_status()
|
||||
raw = await read_capped(response, max_bytes)
|
||||
data = json.loads(raw.decode("utf-8", "ignore"))
|
||||
items = data.get("items", []) if isinstance(data, dict) else []
|
||||
self._cache = [i for i in items if isinstance(i, dict)]
|
||||
self._fetched_at = time.monotonic()
|
||||
return self._cache
|
||||
|
||||
async def search(self, query: str, lang: str = "en", limit: int = DEFAULT_LIMIT) -> Dict[str, Any]:
|
||||
"""Return sanitized top matches, or an error dict — never raise (CDX-05)."""
|
||||
try:
|
||||
items = await self._load_index()
|
||||
except Exception as err:
|
||||
logging.warning(f"codex: index load failed: {err!r}")
|
||||
return {"error": f"codex unavailable: {err}"}
|
||||
terms = _tokenize(query)
|
||||
if not terms:
|
||||
return {"query": query, "results": []}
|
||||
pick = (lang or "en").lower()
|
||||
if pick not in _VALID_LANGS:
|
||||
pick = "en"
|
||||
summary_chars = int(self._config().get("codex-summary-chars", DEFAULT_SUMMARY_CHARS))
|
||||
results = []
|
||||
for item in _rank(items, terms, pick)[: max(1, limit)]:
|
||||
results.append(
|
||||
{
|
||||
"title": sanitize_external_text(str(item.get("title") or ""), 200),
|
||||
"summary": sanitize_external_text(str(item.get("summary") or ""), summary_chars),
|
||||
"collection": str(item.get("collection") or ""),
|
||||
"url": urljoin(self._index_url(), str(item.get("url") or "")),
|
||||
}
|
||||
)
|
||||
return {"query": query, "lang": pick, "results": results}
|
||||
@@ -9,6 +9,9 @@ from typing import Any, Dict, List, Optional, Tuple
|
||||
import openai
|
||||
|
||||
from .ai_responder import AIResponder, exponential_backoff, sanitize_external_text
|
||||
from .codex import CODEX_SEARCH_TOOL
|
||||
from .codex import DEFAULT_LIMIT as CODEX_DEFAULT_LIMIT
|
||||
from .codex import CodexSearch
|
||||
from .igdblib import IGDBQuery
|
||||
from .leonardo_draw import LeonardoAIDrawMixIn
|
||||
from .quota import QuotaLedger
|
||||
@@ -160,6 +163,8 @@ class OpenAIResponder(AIResponder, LeonardoAIDrawMixIn):
|
||||
|
||||
# URL reading tool (SPEC-011); shares the image cache for page images
|
||||
self.url_reader = URLReader(lambda: self.config, self.image_cache)
|
||||
# Codex Mechanicus search (SPEC-014); Luma's own archive at binaric.tech
|
||||
self.codex = CodexSearch(lambda: self.config)
|
||||
|
||||
def _available_tools(self) -> List[Dict[str, Any]]:
|
||||
"""Assemble the function-tool list from every enabled provider (URL-01)."""
|
||||
@@ -173,16 +178,25 @@ class OpenAIResponder(AIResponder, LeonardoAIDrawMixIn):
|
||||
logging.warning(f"Error setting up IGDB functions: {err}")
|
||||
if self.url_reader.enabled():
|
||||
functions.append(FETCH_URL_TOOL)
|
||||
if self.codex.enabled(): # CDX-01
|
||||
functions.append(CODEX_SEARCH_TOOL)
|
||||
return functions
|
||||
|
||||
async def _dispatch_tool(self, name: str, args: Dict[str, Any], author: str) -> Any:
|
||||
"""Route a tool call to its provider (IGDB or URL reader)."""
|
||||
"""Route a tool call to its provider (IGDB, URL reader, or codex)."""
|
||||
if name == "fetch_url":
|
||||
per_user_cap = int(self.config.get("url-daily-per-user", 20))
|
||||
if self.ledger._get(f"url-fetch:{author}") >= per_user_cap: # URL-07
|
||||
return {"error": "daily URL fetch limit reached"}
|
||||
self.ledger._add(f"url-fetch:{author}", 1)
|
||||
return await self.url_reader.fetch(str(args.get("url", "")), self.channel, author or "user")
|
||||
if name == "codex_search":
|
||||
per_user_cap = int(self.config.get("codex-daily-per-user", 50))
|
||||
if self.ledger._get(f"codex:{author}") >= per_user_cap: # CDX-06
|
||||
return {"error": "daily codex search limit reached"}
|
||||
self.ledger._add(f"codex:{author}", 1)
|
||||
limit = int(self.config.get("codex-limit", CODEX_DEFAULT_LIMIT))
|
||||
return await self.codex.search(str(args.get("query", "")), str(args.get("lang", "en")), limit)
|
||||
return await self._execute_igdb_function(name, args)
|
||||
|
||||
async def draw_openai(self, description: str, count: int = 1) -> List[BytesIO]:
|
||||
|
||||
+30
-11
@@ -38,6 +38,9 @@ FETCH_URL_TOOL = {
|
||||
}
|
||||
|
||||
|
||||
_META_REFRESH_URL = re.compile(r"url\s*=\s*['\"]?([^'\";\s]+)", re.I)
|
||||
|
||||
|
||||
class _Extractor(HTMLParser):
|
||||
def __init__(self) -> None:
|
||||
super().__init__()
|
||||
@@ -45,6 +48,7 @@ class _Extractor(HTMLParser):
|
||||
self.parts: List[str] = []
|
||||
self.images: List[str] = []
|
||||
self.og_image: Optional[str] = None
|
||||
self.refresh_url: Optional[str] = None
|
||||
|
||||
def handle_starttag(self, tag: str, attrs) -> None:
|
||||
if tag in ("script", "style", "noscript", "svg"):
|
||||
@@ -55,6 +59,12 @@ class _Extractor(HTMLParser):
|
||||
self.images.append(src)
|
||||
if tag == "meta" and attr.get("property") == "og:image" and attr.get("content"):
|
||||
self.og_image = attr["content"]
|
||||
# meta-refresh redirect (link shorteners, getnews stubs) — URL-04
|
||||
content = attr.get("content")
|
||||
if tag == "meta" and (attr.get("http-equiv") or "").lower() == "refresh" and content:
|
||||
match = _META_REFRESH_URL.search(content)
|
||||
if match and self.refresh_url is None:
|
||||
self.refresh_url = match.group(1)
|
||||
|
||||
def handle_endtag(self, tag: str) -> None:
|
||||
if tag in ("script", "style", "noscript", "svg") and self._skip > 0:
|
||||
@@ -126,29 +136,38 @@ class URLReader:
|
||||
try:
|
||||
async with aiohttp.ClientSession(timeout=timeout, headers={"User-Agent": "FjerkroaBot/1.0"}) as session:
|
||||
final_url, body = await self._get(session, url, max_bytes)
|
||||
# follow a meta-refresh redirect (link shorteners / getnews stubs), re-guarded — URL-04
|
||||
for _ in range(2):
|
||||
extractor = self._extract(body.decode("utf-8", "ignore"))
|
||||
if not extractor.refresh_url:
|
||||
break
|
||||
target = urljoin(final_url, extractor.refresh_url)
|
||||
if guard_url(target) is not None or target == final_url:
|
||||
break
|
||||
logging.info(f"url reader: following meta-refresh -> {target}")
|
||||
final_url, body = await self._get(session, target, max_bytes)
|
||||
except Exception as err:
|
||||
return {"error": str(err)}
|
||||
text = self._to_text(body.decode("utf-8", "ignore"))
|
||||
clean = sanitize_external_text(text, int(config.get("url-max-chars", DEFAULT_MAX_CHARS)))
|
||||
images = await self._ingest_images(body.decode("utf-8", "ignore"), final_url, channel, user)
|
||||
html = body.decode("utf-8", "ignore")
|
||||
clean = sanitize_external_text(self._to_text(html), int(config.get("url-max-chars", DEFAULT_MAX_CHARS)))
|
||||
images = await self._ingest_images(html, final_url, channel, user)
|
||||
return {"url": final_url, "text": clean, "images_cached": images}
|
||||
|
||||
def _to_text(self, html: str) -> str:
|
||||
def _extract(self, html: str) -> "_Extractor":
|
||||
extractor = _Extractor()
|
||||
try:
|
||||
extractor.feed(html)
|
||||
except Exception as err:
|
||||
logging.debug(f"html parse (text) failed: {err!r}")
|
||||
return re.sub(r"\s+\n", "\n", " ".join(extractor.parts))
|
||||
logging.debug(f"html parse failed: {err!r}")
|
||||
return extractor
|
||||
|
||||
def _to_text(self, html: str) -> str:
|
||||
return re.sub(r"\s+\n", "\n", " ".join(self._extract(html).parts))
|
||||
|
||||
async def _ingest_images(self, html: str, base_url: str, channel: str, user: str) -> int:
|
||||
if self.image_cache is None:
|
||||
return 0
|
||||
extractor = _Extractor()
|
||||
try:
|
||||
extractor.feed(html)
|
||||
except Exception as err:
|
||||
logging.debug(f"html parse (images) failed: {err!r}")
|
||||
extractor = self._extract(html)
|
||||
candidates = ([extractor.og_image] if extractor.og_image else []) + extractor.images
|
||||
limit = int(self._config().get("url-max-images", DEFAULT_MAX_IMAGES))
|
||||
cached = 0
|
||||
|
||||
@@ -13,3 +13,4 @@ with date + result.
|
||||
| DEP-05 | 2026-07-13 | Live-verified: kroa deploy attempt ~15h Oslo refused without DEPLOY_FORCE=1. |
|
||||
| DEP-06 | 2026-07-13 | Rollback documented (older tag + db backup restore); live drill pending — next release. |
|
||||
| OPS-15 | 2026-07-13 | Backup cron installed on both hosts (daily 03:17 UTC → ~/backups/<bot>/, keep 14); first snapshots written + verified 0600 (kroa 10965 B, luma 25871 B). |
|
||||
| CDX-07 | 2026-07-13 | Pending live verify on ggg after v3.8.0 deploy: persona grounding + codex_search returns binaric.tech inscriptions with links. |
|
||||
|
||||
@@ -31,7 +31,10 @@ refused without DNS.
|
||||
|
||||
Redirects are followed manually; each hop's target passes URL-02 and
|
||||
URL-03 again. A public URL that 302-redirects to `localhost` or an
|
||||
internal IP is refused at the redirect, not fetched.
|
||||
internal IP is refused at the redirect, not fetched. **HTML
|
||||
meta-refresh** redirects (link shorteners, the old getnews stubs) are
|
||||
also followed — the target is SSRF-re-guarded and fetched, so the
|
||||
reader returns the real article, not the "Redirecting…" stub.
|
||||
|
||||
### URL-05 — Fetched text is bounded and sanitized (coverage: test)
|
||||
|
||||
|
||||
@@ -0,0 +1,59 @@
|
||||
# SPEC-014 — Codex Mechanicus search
|
||||
|
||||
Luma is an Adeptus Mechanicus tech-priest; her lore has a real home —
|
||||
the priest's own Codex Mechanicus at `binaric.tech` (an Astro/MDX
|
||||
archive, five tongues). A `codex_search` function tool lets her consult
|
||||
that archive and answer from sourced inscriptions instead of inventing
|
||||
lore. The index is public but still untrusted by the time it reaches a
|
||||
prompt: the fetch is SSRF-guarded (SPEC-011 shares `guard_url`),
|
||||
size-bounded, and every returned field is sanitized (SAF-03). Luma-only;
|
||||
active only when `enable-codex = true`.
|
||||
|
||||
### CDX-01 — codex_search is offered as a tool (coverage: test)
|
||||
|
||||
When `enable-codex` is true, the chat call's `tools` list includes a
|
||||
`codex_search` function (`query` string, optional `lang`) next to any
|
||||
IGDB / fetch_url tools. When false, it is absent.
|
||||
|
||||
### CDX-02 — The index is fetched safely and cached (coverage: test)
|
||||
|
||||
The index URL (`codex-index-url`, default
|
||||
`https://binaric.tech/search-index.json`) passes the SSRF guard before
|
||||
any network call, is read under a byte cap (`codex-max-bytes`, default
|
||||
4 MB) with a download timeout, and is cached in memory for
|
||||
`codex-cache-ttl` (default 3600 s) so repeated searches do not re-fetch.
|
||||
|
||||
### CDX-03 — Ranking weights title over summary over body (coverage: test)
|
||||
|
||||
The query is tokenized (stopwords dropped); each inscription is scored
|
||||
by term hits weighted title (8) > summary (3) > body (1). Results are
|
||||
returned highest-score first, each as `{title, summary, collection,
|
||||
url}`, with `url` absolute against the site origin.
|
||||
|
||||
### CDX-04 — Language is preferred, with fallback (coverage: test)
|
||||
|
||||
Results are filtered to the requested `lang` (en, de, eo, no, uk;
|
||||
default en; unknown codes fall back to en) by the language segment in
|
||||
each inscription URL. If no inscription in that tongue matches, the
|
||||
search falls back to all tongues rather than returning nothing.
|
||||
|
||||
### CDX-05 — Results are sanitized and failure is reported (coverage: test)
|
||||
|
||||
Each `title` and `summary` is passed through `sanitize_external_text`
|
||||
and length-capped (`codex-summary-chars`, default 500). An index that
|
||||
cannot be fetched or parsed returns an `{error: ...}` dict the model can
|
||||
relay — `search` never raises.
|
||||
|
||||
### CDX-06 — Searches are metered per user (coverage: test)
|
||||
|
||||
Each `codex_search` increments a per-user daily counter; over
|
||||
`codex-daily-per-user` (default 50) the tool refuses with an error
|
||||
result without touching the index. The budget gate (SAF-04) still
|
||||
applies to the surrounding model calls.
|
||||
|
||||
### CDX-07 — Luma cites the codex, not invention (coverage: manual)
|
||||
|
||||
With the persona grounding line, when a pilgrim asks Cult Mechanicus
|
||||
lore Luma consults `codex_search` and answers from it, offering the
|
||||
`binaric.tech` link to read the full inscription rather than
|
||||
hallucinating. Verified live on ggg.
|
||||
@@ -0,0 +1,159 @@
|
||||
"""Unit coverage for SPEC-014 Codex Mechanicus search (CDX-01..06)."""
|
||||
|
||||
import json
|
||||
import unittest
|
||||
from unittest.mock import AsyncMock, patch
|
||||
|
||||
from fjerkroa_bot.codex import CODEX_SEARCH_TOOL, CodexSearch
|
||||
from fjerkroa_bot.openai_responder import OpenAIResponder
|
||||
|
||||
CONFIG = {"openai-token": "t", "model": "m", "system": "s", "history-limit": 5}
|
||||
|
||||
INDEX = {
|
||||
"items": [
|
||||
{
|
||||
"id": "doctrine-heretek",
|
||||
"collection": "doctrines",
|
||||
"url": "/en/codex/doctrines/doctrine-heretek/",
|
||||
"title": "Heretek — Doctrine of the Tech-Heretic",
|
||||
"summary": "The label the Cult Mechanicus stamps on Tech-Priests who pursue forbidden sciences.",
|
||||
"body": "xenotech, sentient machines, Warp-touched archeotech",
|
||||
},
|
||||
{
|
||||
"id": "doctrine-heretek",
|
||||
"collection": "doctrines",
|
||||
"url": "/de/codex/doctrines/doctrine-heretek/",
|
||||
"title": "Heretek — Doktrin des Techketzers",
|
||||
"summary": "Das Etikett des Kultes Mechanicus fuer Techpriester verbotener Wissenschaften.",
|
||||
"body": "Xenotech, empfindungsfaehige Maschinen",
|
||||
},
|
||||
{
|
||||
"id": "forge-stygies",
|
||||
"collection": "forges",
|
||||
"url": "/en/codex/forges/forge-stygies/",
|
||||
"title": "Stygies VIII",
|
||||
"summary": "A forge world of shrouded reputation.",
|
||||
"body": "The forge fields many Skitarii legions.",
|
||||
},
|
||||
]
|
||||
}
|
||||
|
||||
|
||||
def _reader(cfg):
|
||||
reader = CodexSearch(lambda: cfg)
|
||||
return reader
|
||||
|
||||
|
||||
class TestToolOffered(unittest.TestCase):
|
||||
def test_tool_present_only_when_enabled(self):
|
||||
"""CDX-01: codex_search appears only with enable-codex."""
|
||||
off = OpenAIResponder(CONFIG, "chat")
|
||||
self.assertNotIn("codex_search", [f["name"] for f in off._available_tools()])
|
||||
on = OpenAIResponder(dict(CONFIG, **{"enable-codex": True}), "chat")
|
||||
self.assertIn("codex_search", [f["name"] for f in on._available_tools()])
|
||||
self.assertEqual(CODEX_SEARCH_TOOL["name"], "codex_search")
|
||||
|
||||
|
||||
class TestIndexGuardAndCache(unittest.IsolatedAsyncioTestCase):
|
||||
async def test_internal_index_url_refused(self):
|
||||
"""CDX-02: an index URL on a private address is refused before any fetch."""
|
||||
reader = _reader({"enable-codex": True, "codex-index-url": "http://127.0.0.1/search-index.json"})
|
||||
result = await reader.search("heretek")
|
||||
self.assertIn("error", result)
|
||||
|
||||
async def test_index_cached_within_ttl(self):
|
||||
"""CDX-02: a second search inside the TTL does not re-fetch the index."""
|
||||
reader = _reader({"enable-codex": True, "codex-cache-ttl": 9999})
|
||||
raw = json.dumps(INDEX).encode()
|
||||
calls = [0]
|
||||
|
||||
class FakeResp:
|
||||
status = 200
|
||||
|
||||
async def __aenter__(self):
|
||||
return self
|
||||
|
||||
async def __aexit__(self, *a):
|
||||
return False
|
||||
|
||||
def raise_for_status(self):
|
||||
pass
|
||||
|
||||
class FakeSession:
|
||||
def get(self, url):
|
||||
calls[0] += 1
|
||||
return FakeResp()
|
||||
|
||||
async def __aenter__(self):
|
||||
return self
|
||||
|
||||
async def __aexit__(self, *a):
|
||||
return False
|
||||
|
||||
with patch("fjerkroa_bot.codex.read_capped", new=AsyncMock(return_value=raw)):
|
||||
with patch("fjerkroa_bot.codex.guard_url", return_value=None):
|
||||
with patch("fjerkroa_bot.codex.aiohttp.ClientSession", return_value=FakeSession()):
|
||||
first = await reader.search("heretek")
|
||||
second = await reader.search("stygies")
|
||||
self.assertEqual(calls[0], 1) # fetched once, served from cache the second time
|
||||
self.assertTrue(first["results"] and second["results"])
|
||||
|
||||
|
||||
class TestRankingAndLang(unittest.IsolatedAsyncioTestCase):
|
||||
async def _search(self, cfg, query, lang="en"):
|
||||
reader = _reader(dict({"enable-codex": True}, **cfg))
|
||||
reader._cache = INDEX["items"]
|
||||
reader._fetched_at = 1e18 # far future: never expires in test
|
||||
with patch("fjerkroa_bot.codex.time.monotonic", return_value=1e18):
|
||||
return await reader.search(query, lang)
|
||||
|
||||
async def test_title_hit_outranks_body_hit(self):
|
||||
"""CDX-03: a title match ranks above a body-only match."""
|
||||
result = await self._search({}, "heretek")
|
||||
self.assertEqual(result["results"][0]["title"].split(" ")[0], "Heretek")
|
||||
self.assertTrue(result["results"][0]["url"].startswith("https://binaric.tech/en/"))
|
||||
|
||||
async def test_lang_filter_selects_language(self):
|
||||
"""CDX-04: lang=de returns the German inscription."""
|
||||
result = await self._search({}, "heretek", lang="de")
|
||||
self.assertTrue(all("/de/" in r["url"] for r in result["results"]))
|
||||
self.assertIn("Techketzer", result["results"][0]["title"])
|
||||
|
||||
async def test_lang_fallback_when_absent(self):
|
||||
"""CDX-04: a tongue with no match falls back to all tongues, not empty."""
|
||||
result = await self._search({}, "stygies", lang="uk") # only en/de exist
|
||||
self.assertTrue(result["results"])
|
||||
self.assertEqual(result["results"][0]["title"], "Stygies VIII")
|
||||
|
||||
|
||||
class TestSanitizeAndFailure(unittest.IsolatedAsyncioTestCase):
|
||||
async def test_result_sanitized_and_capped(self):
|
||||
"""CDX-05: title/summary are @-neutralized and length-capped."""
|
||||
reader = _reader({"enable-codex": True, "codex-summary-chars": 40})
|
||||
reader._cache = [{"collection": "x", "url": "/en/x/", "title": "@everyone hi", "summary": "@here " + "y" * 500, "body": "hit"}]
|
||||
reader._fetched_at = 1e18
|
||||
with patch("fjerkroa_bot.codex.time.monotonic", return_value=1e18):
|
||||
result = await reader.search("hit")
|
||||
top = result["results"][0]
|
||||
self.assertNotIn("@everyone", top["title"])
|
||||
self.assertNotIn("@here", top["summary"])
|
||||
self.assertLessEqual(len(top["summary"]), 40)
|
||||
|
||||
async def test_index_failure_returns_error(self):
|
||||
"""CDX-05: a broken index returns an error dict, never raises."""
|
||||
reader = _reader({"enable-codex": True})
|
||||
with patch.object(reader, "_load_index", new=AsyncMock(side_effect=ValueError("boom"))):
|
||||
result = await reader.search("heretek")
|
||||
self.assertIn("error", result)
|
||||
|
||||
|
||||
class TestPerUserCap(unittest.IsolatedAsyncioTestCase):
|
||||
async def test_dispatch_caps_searches(self):
|
||||
"""CDX-06: over codex-daily-per-user, codex_search refuses without searching."""
|
||||
responder = OpenAIResponder(dict(CONFIG, **{"enable-codex": True, "codex-daily-per-user": 2}), "chat")
|
||||
responder.codex.search = AsyncMock(return_value={"query": "x", "results": []})
|
||||
for _ in range(2):
|
||||
await responder._dispatch_tool("codex_search", {"query": "heretek"}, "magos")
|
||||
blocked = await responder._dispatch_tool("codex_search", {"query": "heretek"}, "magos")
|
||||
self.assertIn("error", blocked)
|
||||
self.assertEqual(responder.codex.search.await_count, 2)
|
||||
@@ -79,6 +79,65 @@ class TestRedirectRevalidation(unittest.IsolatedAsyncioTestCase):
|
||||
await reader._get(FakeSession(), "http://safe.example.com", 1000)
|
||||
|
||||
|
||||
class TestMetaRefresh(unittest.IsolatedAsyncioTestCase):
|
||||
async def test_follows_meta_refresh_to_real_article(self):
|
||||
"""URL-04: a getnews-style meta-refresh stub is followed to the real article."""
|
||||
reader = URLReader(lambda: {}, None)
|
||||
stub = (
|
||||
b'<html><head><meta http-equiv="refresh" content="0;url=https://pushsquare.com/real"></head><body>Redirecting...</body></html>'
|
||||
)
|
||||
article = b"<html><body><h1>MARVEL Tokon</h1><p>Full article text here</p></body></html>"
|
||||
calls = []
|
||||
|
||||
async def fake_get(session, url, max_bytes):
|
||||
calls.append(url)
|
||||
return (url, stub if "stub" in url else article)
|
||||
|
||||
reader._get = fake_get # type: ignore
|
||||
with patch("fjerkroa_bot.url_reader.guard_url", return_value=None):
|
||||
import fjerkroa_bot.url_reader as ur
|
||||
|
||||
# patch the session context so fetch() runs against fake_get
|
||||
class FakeCM:
|
||||
async def __aenter__(self):
|
||||
return object()
|
||||
|
||||
async def __aexit__(self, *a):
|
||||
return False
|
||||
|
||||
with patch.object(ur.aiohttp, "ClientSession", return_value=FakeCM()):
|
||||
result = await reader.fetch("https://gggemein.de/url/stub.html", "chat", "alice")
|
||||
self.assertIn("Full article text", result["text"])
|
||||
self.assertEqual(result["url"], "https://pushsquare.com/real")
|
||||
self.assertIn("https://pushsquare.com/real", calls)
|
||||
|
||||
async def test_meta_refresh_to_internal_is_not_followed(self):
|
||||
"""URL-04: a meta-refresh pointing at an internal IP is refused (SSRF)."""
|
||||
reader = URLReader(lambda: {}, None)
|
||||
stub = b'<meta http-equiv="refresh" content="0; url=http://127.0.0.1/secret">Redirecting'
|
||||
|
||||
async def fake_get(session, url, max_bytes):
|
||||
return (url, stub)
|
||||
|
||||
reader._get = fake_get # type: ignore
|
||||
import fjerkroa_bot.url_reader as ur
|
||||
|
||||
class FakeCM:
|
||||
async def __aenter__(self):
|
||||
return object()
|
||||
|
||||
async def __aexit__(self, *a):
|
||||
return False
|
||||
|
||||
def guard(u):
|
||||
return "refused" if "127.0.0.1" in u else None
|
||||
|
||||
with patch("fjerkroa_bot.url_reader.guard_url", side_effect=guard):
|
||||
with patch.object(ur.aiohttp, "ClientSession", return_value=FakeCM()):
|
||||
result = await reader.fetch("https://safe.com/x", "chat", "alice")
|
||||
self.assertEqual(result["url"], "https://safe.com/x") # did not follow to 127.0.0.1
|
||||
|
||||
|
||||
class TestTextExtraction(unittest.TestCase):
|
||||
def test_html_reduced_to_text(self):
|
||||
"""URL-05: scripts/styles dropped, tags stripped."""
|
||||
|
||||
Reference in New Issue
Block a user