From f3c25de310271685ff4989bd61633d2c5900388b Mon Sep 17 00:00:00 2001 From: Oleksandr Kozachuk Date: Tue, 21 Jul 2026 13:01:21 +0200 Subject: [PATCH] =?UTF-8?q?url-09:=20fetched=20images=20become=20vision=20?= =?UTF-8?q?input=20=E2=80=94=20fetch=5Furl=20og:image/body=20images=20ride?= =?UTF-8?q?=20as=20input=5Fimage,=20direct=20image=20urls=20ingested,=20da?= =?UTF-8?q?ta=20urls=20never=20in=20json=20tool=20text?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- fjerkroa_bot/openai_responder.py | 23 ++++++ fjerkroa_bot/url_reader.py | 44 +++++++---- specs/SPEC-011-url-reading.md | 17 ++++ tests/test_spec_responses.py | 31 ++++++++ tests/test_spec_url.py | 129 ++++++++++++++++++++++++++++--- 5 files changed, 222 insertions(+), 22 deletions(-) diff --git a/fjerkroa_bot/openai_responder.py b/fjerkroa_bot/openai_responder.py index be3c1aa..b48fc8a 100644 --- a/fjerkroa_bot/openai_responder.py +++ b/fjerkroa_bot/openai_responder.py @@ -329,6 +329,17 @@ class OpenAIResponder(AIResponder, LeonardoAIDrawMixIn): items.append(data) return items + @staticmethod + def _split_vision(function_result: Any) -> Tuple[Any, List[str]]: + """Detach cached image data URLs from a tool result (URL-09) — they + ride to the model as image input, never as JSON text (a base64 data + URL would blow the 8000-char sanitizer cap).""" + if isinstance(function_result, dict) and function_result.get("vision"): + return function_result, [str(url) for url in function_result.pop("vision")] + if isinstance(function_result, dict): + function_result.pop("vision", None) + return function_result, [] + @staticmethod def _responses_refused(result: Any) -> bool: for item in getattr(result, "output", []) or []: @@ -381,6 +392,7 @@ class OpenAIResponder(AIResponder, LeonardoAIDrawMixIn): function_args = json.loads(call.arguments) if call.arguments else {} logging.info(f"🔧 Executing tool: {call.name} with args: {function_args}") function_result = await self._dispatch_tool(call.name, function_args, author or "") + function_result, vision = self._split_vision(function_result) logging.info(f"🔧 Tool result: {type(function_result)} - {str(function_result)[:200]}...") context.append( { @@ -390,6 +402,10 @@ class OpenAIResponder(AIResponder, LeonardoAIDrawMixIn): "output": sanitize_external_text(json.dumps(function_result), 8000) if function_result else "No results found", } ) + if vision: + # fetched images become sight, not text (URL-09) + logging.info(f"🔧 Tool returned {len(vision)} image(s) — attached as vision input") + context.append({"role": "user", "content": [{"type": "input_image", "image_url": url} for url in vision]}) kwargs["input"] = context rounds -= 1 if rounds <= 0: @@ -504,6 +520,7 @@ class OpenAIResponder(AIResponder, LeonardoAIDrawMixIn): # Route to the right provider (IGDB or URL reader) function_result = await self._dispatch_tool(function_name, function_args, self._last_author(messages) or "") + function_result, vision = self._split_vision(function_result) logging.info(f"🔧 Tool result: {type(function_result)} - {str(function_result)[:200]}...") @@ -517,6 +534,12 @@ class OpenAIResponder(AIResponder, LeonardoAIDrawMixIn): ), } ) + if vision: + # fetched images become sight, not text (URL-09) + logging.info(f"🔧 Tool returned {len(vision)} image(s) — attached as vision input") + messages.append( + {"role": "user", "content": [{"type": "image_url", "image_url": {"url": url}} for url in vision]} + ) # Get final response after function execution - remove tools for final call final_chat_kwargs = { diff --git a/fjerkroa_bot/url_reader.py b/fjerkroa_bot/url_reader.py index 8a56a93..c628d2d 100644 --- a/fjerkroa_bot/url_reader.py +++ b/fjerkroa_bot/url_reader.py @@ -149,7 +149,7 @@ class URLReader: def enabled(self) -> bool: return bool(self._config().get("enable-url-reading", False)) - async def _get(self, session, url: str, max_bytes: int) -> Tuple[str, bytes]: + async def _get(self, session, url: str, max_bytes: int) -> Tuple[str, bytes, str]: """Manual redirect handling so every hop is re-guarded (URL-04).""" current = url for _ in range(MAX_REDIRECTS): @@ -161,7 +161,8 @@ class URLReader: current = urljoin(current, response.headers["Location"]) continue response.raise_for_status() - return str(response.url), await read_capped(response, max_bytes) + content_type = str(response.headers.get("Content-Type", "")).split(";")[0].strip().lower() + return str(response.url), await read_capped(response, max_bytes), content_type raise ValueError("too many redirects") async def fetch(self, url: str, channel: str, user: str) -> Dict[str, Any]: @@ -170,9 +171,11 @@ class URLReader: timeout = aiohttp.ClientTimeout(total=FETCH_TIMEOUT_S) try: async with aiohttp.ClientSession(timeout=timeout, headers={"User-Agent": "FjerkroaBot/1.0"}) as session: - final_url, body = await self._get(session, url, max_bytes) + final_url, body, content_type = await self._get(session, url, max_bytes) # follow a meta-refresh redirect (link shorteners / getnews stubs), re-guarded — URL-04 for _ in range(2): + if content_type.startswith("image/"): + break extractor = self._extract(body.decode("utf-8", "ignore")) if not extractor.refresh_url: break @@ -180,13 +183,21 @@ class URLReader: if guard_url(target) is not None or target == final_url: break logging.info(f"url reader: following meta-refresh -> {target}") - final_url, body = await self._get(session, target, max_bytes) + final_url, body, content_type = await self._get(session, target, max_bytes) except Exception as err: return {"error": str(err)} + # a URL that IS an image: cache it and hand it over as sight (URL-09) + if content_type.startswith("image/"): + vision = [] + if self.image_cache is not None and len(body) < max_bytes: # >= cap means possibly truncated + sha = self.image_cache.ingest_bytes(body, channel, user, None) + data_url = self._cached_data_url(sha, channel) if sha else None + vision = [data_url] if data_url else [] + return {"url": final_url, "text": "(image)", "images_cached": len(vision), "vision": vision} html = body.decode("utf-8", "ignore") clean = sanitize_external_text(self._to_text(html), int(config.get("url-max-chars", DEFAULT_MAX_CHARS))) - images = await self._ingest_images(html, final_url, channel, user) - return {"url": final_url, "text": clean, "images_cached": images} + vision = await self._ingest_images(html, final_url, channel, user) + return {"url": final_url, "text": clean, "images_cached": len(vision), "vision": vision} def _extract(self, html: str) -> "_Extractor": extractor = _Extractor() @@ -199,20 +210,27 @@ class URLReader: def _to_text(self, html: str) -> str: return re.sub(r"\s+\n", "\n", " ".join(self._extract(html).content_parts())) - async def _ingest_images(self, html: str, base_url: str, channel: str, user: str) -> int: + async def _ingest_images(self, html: str, base_url: str, channel: str, user: str) -> List[str]: + """Cache page images and return their data URLs for vision input (URL-09).""" if self.image_cache is None: - return 0 + return [] extractor = self._extract(html) candidates = ([extractor.og_image] if extractor.og_image else []) + extractor.images limit = int(self._config().get("url-max-images", DEFAULT_MAX_IMAGES)) - cached = 0 + data_urls: List[str] = [] for src in candidates: - if cached >= limit: + if len(data_urls) >= limit: break absolute = urljoin(base_url, src) if guard_url(absolute) is not None: continue sha = await self.image_cache.ingest_url(absolute, channel, user, None) - if sha is not None: - cached += 1 - return cached + data_url = self._cached_data_url(sha, channel) if sha else None + if data_url: + data_urls.append(data_url) + return data_urls + + def _cached_data_url(self, sha: str, channel: str) -> Optional[str]: + recent = self.image_cache.recent(channel, 8) + ext = next((row["ext"] for row in recent if row["sha256"] == sha), None) + return self.image_cache.data_url(sha, ext) if ext else None diff --git a/specs/SPEC-011-url-reading.md b/specs/SPEC-011-url-reading.md index c7c21b5..af9ad3b 100644 --- a/specs/SPEC-011-url-reading.md +++ b/specs/SPEC-011-url-reading.md @@ -69,3 +69,20 @@ block's characters inside `` and the block shorter than 200 chars boilerplate and removed. Body paragraphs with inline links survive. The default `url-max-chars` cap rises to 8000 now that the budget is spent on content, not chrome. + +### URL-09 — Fetched images become vision input (coverage: test) + +The images the URL reader already caches from a fetched page +(`og:image` first, then body images, `url-max-images` cap, every +candidate SSRF-guarded and magic-byte-sniffed by the image cache) now +travel to the model as image input alongside the tool result — the +model sees the picture, not just an `images_cached` count. A URL +whose response is itself an image (content-type `image/*`) is +ingested directly and returns text `(image)`; a body at the byte cap +is treated as possibly truncated and not ingested. The data URLs +ride in a `vision` key that the responder detaches before the JSON +tool text is built (a base64 data URL would blow the 8000-char +sanitizer cap): they are appended as `input_image` items on the +Responses path and as `image_url` parts on the legacy path. Vision +is per-turn — nothing extra is historised; the file stays in the +image cache for later `picture_edit` (IMG-15). diff --git a/tests/test_spec_responses.py b/tests/test_spec_responses.py index 8316adc..8374f41 100644 --- a/tests/test_spec_responses.py +++ b/tests/test_spec_responses.py @@ -163,3 +163,34 @@ class TestResponsesPath(unittest.IsolatedAsyncioTestCase): self.assertEqual(items[0]["content"][0], {"type": "input_text", "text": "look"}) self.assertEqual(items[0]["content"][1], {"type": "input_image", "image_url": "data:x"}) self.assertEqual(len(items), 2) # tool row dropped + + +class TestToolVisionInjection(unittest.IsolatedAsyncioTestCase): + def _responder(self, **extra): + return OpenAIResponder(dict(CONFIG, **extra), "chat") + + async def test_tool_vision_images_attached_as_input_image(self): + """URL-09: a tool result's vision data URLs become input_image items; never JSON text.""" + responder = self._responder(**{"enable-news-tool": True}) + responder.store = Mock() + responder._dispatch_tool = AsyncMock( + return_value={"url": "u", "text": "t", "images_cached": 1, "vision": ["data:image/png;base64,AAA"]} + ) + first = _response([_call_item("fetch_url", {"url": "https://xkcd.com/1"}, "call-2")]) + second = _response([_msg_item()], envelope(answer="seen", answer_needed=True)) + with patch("fjerkroa_bot.openai_responder.openai_responses", new_callable=AsyncMock) as responses_mock: + responses_mock.side_effect = [first, second] + answer, _ = await responder.chat([{"role": "user", "content": "look at this"}], 10) + self.assertEqual(json.loads(answer["content"])["answer"], "seen") + followup = responses_mock.await_args_list[1].kwargs["input"] + image_parts = [ + part + for item in followup + if isinstance(item, dict) and isinstance(item.get("content"), list) + for part in item["content"] + if part.get("type") == "input_image" + ] + self.assertEqual(image_parts[0]["image_url"], "data:image/png;base64,AAA") + outputs = [item for item in followup if isinstance(item, dict) and item.get("type") == "function_call_output"] + self.assertNotIn("data:image", outputs[0]["output"]) # data URL never in JSON tool text + self.assertNotIn("vision", outputs[0]["output"]) diff --git a/tests/test_spec_url.py b/tests/test_spec_url.py index 57015fd..bb45b89 100644 --- a/tests/test_spec_url.py +++ b/tests/test_spec_url.py @@ -1,11 +1,14 @@ -"""Unit coverage for SPEC-011 URL reading (URL-01..07).""" +"""Unit coverage for SPEC-011 URL reading (URL-01..09).""" +import json import unittest -from unittest.mock import AsyncMock, patch +from unittest.mock import AsyncMock, Mock, patch from fjerkroa_bot.openai_responder import OpenAIResponder from fjerkroa_bot.url_reader import FETCH_URL_TOOL, URLReader, guard_url +from .test_bdd_envelope import envelope + CONFIG = {"openai-token": "t", "model": "m", "system": "s", "history-limit": 5} @@ -91,7 +94,7 @@ class TestMetaRefresh(unittest.IsolatedAsyncioTestCase): async def fake_get(session, url, max_bytes): calls.append(url) - return (url, stub if "stub" in url else article) + return (url, stub if "stub" in url else article, "text/html") reader._get = fake_get # type: ignore with patch("fjerkroa_bot.url_reader.guard_url", return_value=None): @@ -117,7 +120,7 @@ class TestMetaRefresh(unittest.IsolatedAsyncioTestCase): stub = b'Redirecting' async def fake_get(session, url, max_bytes): - return (url, stub) + return (url, stub, "text/html") reader._get = fake_get # type: ignore import fjerkroa_bot.url_reader as ur @@ -207,11 +210,11 @@ class TestBodyReadCollectsAllChunks(unittest.IsolatedAsyncioTestCase): return FakeResp() with patch("fjerkroa_bot.url_reader.guard_url", return_value=None): - _, body = await reader._get(FakeSession(), "http://safe.example.com", 1000) + _, body, _ = await reader._get(FakeSession(), "http://safe.example.com", 1000) self.assertEqual(body, b"".join(chunks)) with patch("fjerkroa_bot.url_reader.guard_url", return_value=None): - _, body = await reader._get(FakeSession(), "http://safe.example.com", 20) + _, body, _ = await reader._get(FakeSession(), "http://safe.example.com", 20) self.assertEqual(body, b"".join(chunks)[:20]) @@ -220,7 +223,7 @@ class TestFetchSanitizes(unittest.IsolatedAsyncioTestCase): """URL-05: fetch output is length-capped and @everyone-neutralized.""" reader = URLReader(lambda: {"url-max-chars": 50}, None) payload = ("

@everyone " + "x" * 5000 + "

").encode() - with patch.object(reader, "_get", new=AsyncMock(return_value=("http://x.com", payload))): + with patch.object(reader, "_get", new=AsyncMock(return_value=("http://x.com", payload, "text/html"))): result = await reader.fetch("http://x.com", "chat", "alice") self.assertLessEqual(len(result["text"]), 50) self.assertNotIn("@everyone", result["text"]) @@ -238,6 +241,8 @@ class TestImageIngest(unittest.IsolatedAsyncioTestCase): """URL-06: og:image + ingested (cap honored), internal srcs skipped.""" cache = type("C", (), {})() cache.ingest_url = AsyncMock(side_effect=["sha1", "sha2", "sha3"]) + cache.recent = Mock(return_value=[{"sha256": "sha1", "ext": "jpg"}, {"sha256": "sha2", "ext": "png"}]) + cache.data_url = Mock(side_effect=lambda sha, ext: f"data:image/{ext};base64,{sha}") reader = URLReader(lambda: {"url-max-images": 2}, cache) html = ( '' @@ -249,8 +254,9 @@ class TestImageIngest(unittest.IsolatedAsyncioTestCase): return "refused" if "127.0.0.1" in url else None with patch("fjerkroa_bot.url_reader.guard_url", side_effect=fake_guard): - count = await reader._ingest_images(html, "https://example.com", "chat", "alice") - self.assertEqual(count, 2) # og:image + first public img, cap 2 + data_urls = await reader._ingest_images(html, "https://example.com", "chat", "alice") + self.assertEqual(len(data_urls), 2) # og:image + first public img, cap 2 + self.assertEqual(data_urls[0], "data:image/jpg;base64,sha1") # URL-09: data URLs for vision ingested = [call.args[0] for call in cache.ingest_url.await_args_list] self.assertNotIn("http://127.0.0.1/internal.png", ingested) @@ -265,3 +271,108 @@ class TestPerUserCap(unittest.IsolatedAsyncioTestCase): blocked = await responder._dispatch_tool("fetch_url", {"url": "http://x.com"}, "alice") self.assertIn("error", blocked) self.assertEqual(responder.url_reader.fetch.await_count, 2) + + +class TestFetchedImagesBecomeVision(unittest.IsolatedAsyncioTestCase): + """URL-09: fetch results carry vision data URLs, direct image URLs are ingested.""" + + @staticmethod + def _cache(): + cache = Mock() + cache.ingest_url = AsyncMock(return_value="abc123") + cache.ingest_bytes = Mock(return_value="abc123") + cache.recent = Mock(return_value=[{"sha256": "abc123", "ext": "png"}]) + cache.data_url = Mock(return_value="data:image/png;base64,AAA") + return cache + + @staticmethod + def _session_cm(): + import fjerkroa_bot.url_reader as ur + + class FakeCM: + async def __aenter__(self): + return object() + + async def __aexit__(self, *a): + return False + + return patch.object(ur.aiohttp, "ClientSession", return_value=FakeCM()) + + async def test_html_page_vision_data_urls(self): + """URL-09: og:image lands in the result's vision list, count matches.""" + reader = URLReader(lambda: {}, self._cache()) + html = b'

Comic of the day, longer text.

' + + async def fake_get(session, url, max_bytes): + return (url, html, "text/html") + + reader._get = fake_get # type: ignore + with patch("fjerkroa_bot.url_reader.guard_url", return_value=None): + with self._session_cm(): + result = await reader.fetch("https://xkcd.com/1234", "chat", "alice") + self.assertEqual(result["vision"], ["data:image/png;base64,AAA"]) + self.assertEqual(result["images_cached"], 1) + + async def test_direct_image_url_ingested(self): + """URL-09: content-type image/* -> direct ingest, text '(image)'.""" + cache = self._cache() + reader = URLReader(lambda: {}, cache) + + async def fake_get(session, url, max_bytes): + return (url, b"\x89PNG-bytes", "image/png") + + reader._get = fake_get # type: ignore + with patch("fjerkroa_bot.url_reader.guard_url", return_value=None): + with self._session_cm(): + result = await reader.fetch("https://imgs.xkcd.com/comics/x.png", "chat", "alice") + self.assertEqual(result["text"], "(image)") + self.assertEqual(result["vision"], ["data:image/png;base64,AAA"]) + cache.ingest_bytes.assert_called_once() + + async def test_capped_image_body_not_ingested(self): + """URL-09: an image body at the byte cap may be truncated - not ingested.""" + cache = self._cache() + reader = URLReader(lambda: {"url-max-bytes": 10}, cache) + + async def fake_get(session, url, max_bytes): + return (url, b"0123456789", "image/png") # len == cap + + reader._get = fake_get # type: ignore + with patch("fjerkroa_bot.url_reader.guard_url", return_value=None): + with self._session_cm(): + result = await reader.fetch("https://x.com/big.png", "chat", "alice") + self.assertEqual(result["vision"], []) + cache.ingest_bytes.assert_not_called() + + +class TestLegacyPathVision(unittest.IsolatedAsyncioTestCase): + async def test_legacy_tool_loop_appends_image_message(self): + """URL-09: legacy path - vision data URLs become an image_url user message; never JSON text.""" + responder = OpenAIResponder(dict(CONFIG, **{"enable-url-reading": True}), "chat") + responder._dispatch_tool = AsyncMock( + return_value={"url": "u", "text": "t", "images_cached": 1, "vision": ["data:image/png;base64,AAA"]} + ) + func = Mock() + func.name = "fetch_url" + func.arguments = json.dumps({"url": "https://xkcd.com/1"}) + call = Mock(id="tc1", type="function", function=func) + first_msg = Mock(content=None, role="assistant", tool_calls=[call], refusal=None) + first = Mock(choices=[Mock(message=first_msg)], usage=None) + final_msg = Mock(content=envelope(answer="seen", answer_needed=True), role="assistant", tool_calls=None, refusal=None) + final = Mock(choices=[Mock(message=final_msg)], usage=None) + with patch("fjerkroa_bot.openai_responder.openai_chat", new_callable=AsyncMock) as chat_mock: + chat_mock.side_effect = [first, final] + answer, _ = await responder.chat([{"role": "user", "content": "look at this"}], 10) + self.assertEqual(json.loads(answer["content"])["answer"], "seen") + final_messages = chat_mock.await_args_list[1].kwargs["messages"] + image_parts = [ + part + for msg in final_messages + if isinstance(msg.get("content"), list) + for part in msg["content"] + if part.get("type") == "image_url" + ] + self.assertEqual(image_parts[0]["image_url"]["url"], "data:image/png;base64,AAA") + tool_texts = [msg["content"] for msg in final_messages if msg.get("role") == "tool"] + self.assertNotIn("data:image", tool_texts[0]) # data URL never in JSON tool text + self.assertNotIn("vision", tool_texts[0])