smarter: factual-model routing (beh-10), fetch_url main-content extraction + 8k cap (url-08), get_weather via met.no (spec-016)

This commit is contained in:
Oleksandr Kozachuk
2026-07-17 19:15:18 +02:00
parent f8b9bc75ee
commit e0b97363c9
11 changed files with 404 additions and 6 deletions
+4
View File
@@ -105,6 +105,7 @@ class AIMessage(AIMessageBase):
self.channel = channel
self.direct = direct
self.historise_question = historise_question
self.factual = False # classifier verdict; may route to factual-model (BEH-10)
self.vars = ["user", "message", "channel", "direct", "historise_question"]
@@ -340,6 +341,9 @@ class AIResponder(AIResponderBase):
# Get the history limit from the configuration
limit = self.config["history-limit"]
# Factual verdict routes this call to factual-model if configured (BEH-10)
self._factual = bool(getattr(message, "factual", False))
# Check if a short path applies, return an empty AIResponse if it does
if self.short_path(message, limit):
await self._persist_history()
+3
View File
@@ -701,6 +701,9 @@ class FjerkroaBot(commands.Bot):
# Get the AI responder based on the channel name
airesponder = self.get_ai_responder(channel_name)
# Classifier verdict rides along: factual questions may use factual-model (BEH-10)
message.factual = factual
# Send the user message to the AI responder, with typing indicators.
# A raised call = a broken API path (cf. the gpt-5.6 tools incident):
# count it, alert staff at threshold, never crash the handler (OPS-16).
+12
View File
@@ -17,6 +17,7 @@ from .leonardo_draw import LeonardoAIDrawMixIn
from .news import GET_NEWS_TOOL, query_news
from .quota import QuotaLedger
from .url_reader import FETCH_URL_TOOL, URLReader
from .weather import GET_WEATHER_TOOL, Weather
from .websearch import DEFAULT_RESULTS as WEB_DEFAULT_RESULTS
from .websearch import WEB_SEARCH_TOOL, WebSearch
@@ -170,6 +171,7 @@ class OpenAIResponder(AIResponder, LeonardoAIDrawMixIn):
self.codex = CodexSearch(lambda: self.config)
# Web search (SPEC-015) via Exa; general "look it up" beyond fetch_url/news/codex
self.web_search = WebSearch(lambda: self.config)
self.weather = Weather(lambda: self.config)
def _available_tools(self) -> List[Dict[str, Any]]:
"""Assemble the function-tool list from every enabled provider (URL-01)."""
@@ -189,6 +191,8 @@ class OpenAIResponder(AIResponder, LeonardoAIDrawMixIn):
functions.append(GET_NEWS_TOOL)
if self.web_search.enabled(): # WEB-01
functions.append(WEB_SEARCH_TOOL)
if self.weather.enabled(): # WEA-01
functions.append(GET_WEATHER_TOOL)
return functions
async def _dispatch_tool(self, name: str, args: Dict[str, Any], author: str) -> Any:
@@ -219,6 +223,12 @@ class OpenAIResponder(AIResponder, LeonardoAIDrawMixIn):
return {"error": "daily web search limit reached"}
self.ledger._add(f"web:{author}", 1)
return await self.web_search.search(str(args.get("query", "")), int(args.get("num_results", WEB_DEFAULT_RESULTS)))
if name == "get_weather":
per_user_cap = int(self.config.get("weather-daily-per-user", 30))
if self.ledger._get(f"weather:{author}") >= per_user_cap: # WEA-04
return {"error": "daily weather lookup limit reached"}
self.ledger._add(f"weather:{author}", 1)
return await self.weather.forecast(args.get("location"))
return await self._execute_igdb_function(name, args)
async def draw_openai(self, description: str, count: int = 1) -> List[BytesIO]:
@@ -292,6 +302,8 @@ class OpenAIResponder(AIResponder, LeonardoAIDrawMixIn):
model = self.config["model-vision"]
else:
messages[-1]["content"] = messages[-1]["content"][0]["text"]
if getattr(self, "_factual", False) and "factual-model" in self.config:
model = self.config["factual-model"] # BEH-10: facts get the stronger tier
if self._use_retry_model and "retry-model" in self.config:
model = self.config["retry-model"]
except (KeyError, IndexError, TypeError) as e:
+41 -6
View File
@@ -21,7 +21,7 @@ from .ai_responder import sanitize_external_text
from .httpread import read_capped
DEFAULT_MAX_BYTES = 2 * 1024 * 1024
DEFAULT_MAX_CHARS = 6000
DEFAULT_MAX_CHARS = 8000 # URL-08: budget goes to content now, not chrome
DEFAULT_MAX_IMAGES = 2
FETCH_TIMEOUT_S = 15
MAX_REDIRECTS = 5
@@ -41,18 +41,37 @@ FETCH_URL_TOOL = {
_META_REFRESH_URL = re.compile(r"url\s*=\s*['\"]?([^'\";\s]+)", re.I)
_SKIP_TAGS = ("script", "style", "noscript", "svg", "nav", "header", "footer", "aside", "form", "select", "button")
_BLOCK_TAGS = ("p", "li", "div", "section", "article", "td", "ul", "ol", "table", "h1", "h2", "h3", "h4", "h5", "h6")
_LINK_DENSITY_MAX = 0.6 # boilerplate: block mostly link text ... (URL-08)
_LINK_BLOCK_MAX_CHARS = 200 # ... AND short (menus, related lists); long linky paragraphs survive
class _Extractor(HTMLParser):
def __init__(self) -> None:
super().__init__()
self._skip = 0
self.parts: List[str] = []
self._links = 0
self._buf: List[str] = []
self._buf_link_chars = 0
self.blocks: List[Tuple[str, int]] = [] # (text, chars inside <a>)
self.images: List[str] = []
self.og_image: Optional[str] = None
self.refresh_url: Optional[str] = None
def _flush(self) -> None:
text = " ".join(self._buf).strip()
if text:
self.blocks.append((text, self._buf_link_chars))
self._buf, self._buf_link_chars = [], 0
def handle_starttag(self, tag: str, attrs) -> None:
if tag in ("script", "style", "noscript", "svg"):
if tag in _SKIP_TAGS:
self._skip += 1
if tag == "a":
self._links += 1
if tag in _BLOCK_TAGS:
self._flush()
attr = dict(attrs)
src = attr.get("src")
if tag == "img" and src:
@@ -67,12 +86,28 @@ class _Extractor(HTMLParser):
self.refresh_url = match.group(1)
def handle_endtag(self, tag: str) -> None:
if tag in ("script", "style", "noscript", "svg") and self._skip > 0:
if tag in _SKIP_TAGS and self._skip > 0:
self._skip -= 1
if tag == "a" and self._links > 0:
self._links -= 1
if tag in _BLOCK_TAGS:
self._flush()
def handle_data(self, data: str) -> None:
if self._skip == 0 and data.strip():
self.parts.append(data.strip())
self._buf.append(data.strip())
if self._links > 0:
self._buf_link_chars += len(data.strip())
def content_parts(self) -> List[str]:
"""Blocks minus boilerplate: short blocks dominated by link text are chrome (URL-08)."""
self._flush()
out = []
for text, link_chars in self.blocks:
if link_chars / max(1, len(text)) > _LINK_DENSITY_MAX and len(text) < _LINK_BLOCK_MAX_CHARS:
continue
out.append(text)
return out
def _ip_is_public(ip_str: str) -> bool:
@@ -162,7 +197,7 @@ class URLReader:
return extractor
def _to_text(self, html: str) -> str:
return re.sub(r"\s+\n", "\n", " ".join(self._extract(html).parts))
return re.sub(r"\s+\n", "\n", " ".join(self._extract(html).content_parts()))
async def _ingest_images(self, html: str, base_url: str, channel: str, user: str) -> int:
if self.image_cache is None:
+111
View File
@@ -0,0 +1,111 @@
"""Weather tool via MET Norway Locationforecast (SPEC-016).
A `get_weather` function tool: both personas talk about weather (the
sea over the skerries, rain on patch day) but had to guess it. The
free api.met.no compact forecast grounds it. Locations are
host-configured `[name, lat, lon]` entries the model picks by name
and never supplies coordinates or URLs, so there is no SSRF surface.
"""
import logging
from typing import Any, Callable, Dict, List, Optional, Tuple
import aiohttp
from .ai_responder import sanitize_external_text
MET_COMPACT_URL = "https://api.met.no/weatherapi/locationforecast/2.0/compact"
USER_AGENT = "fjerkroa-discord-bot/3 (https://fjerkroa.no)"
FETCH_TIMEOUT_S = 15
FORECAST_POINT_INDICES = (6, 12, 24) # hourly series: ~6h/12h/24h ahead
GET_WEATHER_TOOL = {
"name": "get_weather",
"description": "Current weather and a short forecast for the configured local places. Use this whenever weather comes "
"up in conversation — never guess or invent weather. Returns current temperature (°C), wind (m/s) and conditions, "
"plus a few forecast points.",
"parameters": {
"type": "object",
"properties": {
"location": {"type": "string", "description": "Place name to look up; omit for the default (first configured) place."},
},
"required": [],
},
}
def _reduce(data: Any, name: str) -> Dict[str, Any]:
"""Compact MET timeseries -> {location, now, forecast[]} (WEA-02). Nothing else reaches the prompt."""
series = data.get("properties", {}).get("timeseries", []) if isinstance(data, dict) else []
if not series:
return {"error": "weather data unavailable"}
def point(entry: Dict[str, Any]) -> Dict[str, Any]:
details = entry.get("data", {}).get("instant", {}).get("details", {})
hour = entry.get("data", {}).get("next_1_hours", {}) or entry.get("data", {}).get("next_6_hours", {})
out: Dict[str, Any] = {
"time": str(entry.get("time", "")),
"temp_c": details.get("air_temperature"),
"wind_ms": details.get("wind_speed"),
}
symbol = hour.get("summary", {}).get("symbol_code")
if symbol:
out["conditions"] = str(symbol)
precip = hour.get("details", {}).get("precipitation_amount")
if precip is not None:
out["precip_mm"] = precip
return out
forecast = [point(series[i]) for i in FORECAST_POINT_INDICES if i < len(series)]
return {"location": sanitize_external_text(name, 80), "now": point(series[0]), "forecast": forecast}
class Weather:
def __init__(self, config_getter: Callable[[], Dict[str, Any]]) -> None:
self._config = config_getter
def _locations(self) -> List[Tuple[str, float, float]]:
out: List[Tuple[str, float, float]] = []
for entry in self._config().get("weather-locations", []):
try:
name, lat, lon = entry[0], float(entry[1]), float(entry[2])
out.append((str(name), lat, lon))
except (TypeError, ValueError, IndexError):
logging.warning(f"weather: bad location entry {entry!r}")
return out
def enabled(self) -> bool:
return bool(self._config().get("enable-weather", False)) and bool(self._locations())
def _pick(self, location: Optional[str]) -> Optional[Tuple[str, float, float]]:
"""Case-insensitive substring match; unknown/absent = first configured (WEA-03)."""
entries = self._locations()
if not entries:
return None
wanted = (location or "").strip().casefold()
if wanted:
for entry in entries:
if wanted in entry[0].casefold():
return entry
return entries[0]
async def _fetch_json(self, lat: float, lon: float) -> Any:
timeout = aiohttp.ClientTimeout(total=FETCH_TIMEOUT_S)
params = {"lat": f"{lat:.4f}", "lon": f"{lon:.4f}"}
async with aiohttp.ClientSession(timeout=timeout, headers={"User-Agent": USER_AGENT}) as session:
async with session.get(MET_COMPACT_URL, params=params) as response:
response.raise_for_status()
return await response.json()
async def forecast(self, location: Optional[str] = None) -> Dict[str, Any]:
"""Return a compact forecast, or an error dict — never raise (WEA-04)."""
picked = self._pick(location)
if picked is None:
return {"error": "weather unavailable: no locations configured"}
name, lat, lon = picked
try:
data = await self._fetch_json(lat, lon)
except Exception as err:
logging.warning(f"weather fetch failed: {err!r}")
return {"error": "weather lookup failed"}
return _reduce(data, name)