From 7e6eae10ee9e369bb108e74e90a3b0a1e6c3defa Mon Sep 17 00:00:00 2001 From: Oleksandr Kozachuk Date: Mon, 13 Jul 2026 16:17:00 +0200 Subject: [PATCH] gpt-image-2 multi-image + fix: tools need reasoning_effort none on gpt-5.6 (ggg mute bug) --- DECISIONS.md | 4 +- config.toml | 5 ++ fjerkroa_bot/ai_responder.py | 19 +++--- fjerkroa_bot/discord_bot.py | 3 +- fjerkroa_bot/openai_responder.py | 56 +++++++---------- manual-verification.md | 2 +- specs/SPEC-001-responder-envelope.md | 11 +++- specs/SPEC-004-images.md | 39 ++++++++++++ tests/test_ai.py | 27 --------- tests/test_bdd_envelope.py | 3 - tests/test_openai_responder_simple.py | 10 --- tests/test_spec_beh.py | 2 +- tests/test_spec_img.py | 87 +++++++++++++++++++++++++++ tests/test_spec_structured.py | 2 +- tests/test_spec_tools.py | 38 ++++++++++++ 15 files changed, 220 insertions(+), 88 deletions(-) create mode 100644 specs/SPEC-004-images.md create mode 100644 tests/test_spec_img.py create mode 100644 tests/test_spec_tools.py diff --git a/DECISIONS.md b/DECISIONS.md index de0ec61..b6312cc 100644 --- a/DECISIONS.md +++ b/DECISIONS.md @@ -35,7 +35,9 @@ Decisions inside the set architecture. D-NNN, never renumbered. - **D-009** — `translate()` still keys off `fix-model` although the repair path is gone; the whole translate-before-draw step dies in FDB-009 (gpt-image-2 is multilingual). Not worth a config rename - for one phase. + for one phase. *(Closed 2026-07-13: FDB-009 deleted translate(); + the fix-model config key is now fully dead and can be dropped from + live configs.)* - **D-010** — Persistence uses stdlib `sqlite3` via `asyncio.to_thread`, not aiosqlite: no new dependency, and a connection-per-operation with WAL is plenty at this message volume. diff --git a/config.toml b/config.toml index a21a79b..2115087 100644 --- a/config.toml +++ b/config.toml @@ -50,3 +50,8 @@ enable-game-info = true # split-threshold = 1200 # long answers split at paragraphs # split-max-parts = 3 # quiet-hours = "21:00-09:00" # no bot-initiated posts in this window + +# Image generation (SPEC-004) +# image-model = "gpt-image-2" # default; dall-e-3 gets clamped to n=1 +# image-size = "1024x1024" +# image-quality = "medium" # passed through only when set diff --git a/fjerkroa_bot/ai_responder.py b/fjerkroa_bot/ai_responder.py index 2815adc..2551411 100644 --- a/fjerkroa_bot/ai_responder.py +++ b/fjerkroa_bot/ai_responder.py @@ -123,6 +123,7 @@ class AIResponse(AIMessageBase): self.channel = channel self.staff = staff self.picture = picture + self.picture_count = 1 self.picture_edit = picture_edit self.hack = hack self.vars = ["answer", "answer_needed", "channel", "staff", "picture", "hack"] @@ -188,15 +189,15 @@ class AIResponder(AIResponderBase): messages.append({"role": "user", "content": content}) return messages - async def draw(self, description: str) -> BytesIO: + async def draw(self, description: str, count: int = 1) -> List[BytesIO]: if self.config.get("leonardo-token") is not None: - return await self.draw_leonardo(description) - return await self.draw_openai(description) + return [await self.draw_leonardo(description)] # single image only, behind config + return await self.draw_openai(description, count) async def draw_leonardo(self, description: str) -> BytesIO: raise NotImplementedError() - async def draw_openai(self, description: str) -> BytesIO: + async def draw_openai(self, description: str, count: int = 1) -> List[BytesIO]: raise NotImplementedError() async def post_process(self, message: AIMessage, response: Dict[str, Any]) -> AIResponse: @@ -221,6 +222,10 @@ class AIResponder(AIResponderBase): bool(response.get("picture_edit", False)), bool(response.get("hack", False)), ) + try: + response_message.picture_count = max(1, min(int(response.get("picture_count") or 1), 4)) # IMG-02 + except (TypeError, ValueError): + response_message.picture_count = 1 if response_message.staff is not None and response_message.answer is not None: response_message.answer_needed = True if response_message.channel is None: @@ -250,9 +255,6 @@ class AIResponder(AIResponderBase): """Cheap reply/factual/emoji pre-pass (BEH-01); None = fail open.""" raise NotImplementedError() - async def translate(self, text: str, language: str = "english") -> str: - raise NotImplementedError() - @staticmethod def _entry_channel(item: Dict[str, Any]) -> Optional[str]: try: @@ -299,11 +301,10 @@ class AIResponder(AIResponderBase): await asyncio.to_thread(self.store.save_history, self.channel, list(self.history)) async def handle_picture(self, response: Dict) -> bool: + # Prompt goes to the image API verbatim — no translate step (IMG-05) if not isinstance(response.get("picture"), (type(None), str)): logging.warning(f"picture key is wrong in response: {pp(response)}") return False - if response.get("picture") is not None: - response["picture"] = await self.translate(response["picture"]) return True def _parse_answer(self, answer: Dict[str, Any]) -> Optional[Dict[str, Any]]: diff --git a/fjerkroa_bot/discord_bot.py b/fjerkroa_bot/discord_bot.py index 4389d2f..480aff2 100644 --- a/fjerkroa_bot/discord_bot.py +++ b/fjerkroa_bot/discord_bot.py @@ -462,7 +462,8 @@ class FjerkroaBot(commands.Bot): """Send the answer paced, split and with images on the last part (BEH-04/05/06)""" files = None if response.picture is not None: - files = [discord.File(fp=await airesponder.draw(response.picture), filename="image.png")] + buffers = await airesponder.draw(response.picture, getattr(response, "picture_count", 1)) + files = [discord.File(fp=buffer, filename=f"image-{index}.png") for index, buffer in enumerate(buffers)] parts = split_answer(response.answer, int(self.config.get("split-threshold", 1200)), int(self.config.get("split-max-parts", 3))) pace = float(self.config.get("typing-chars-per-second", 0) or 0) max_delay = float(self.config.get("typing-max-seconds", 8)) diff --git a/fjerkroa_bot/openai_responder.py b/fjerkroa_bot/openai_responder.py index 72d9158..ad4b373 100644 --- a/fjerkroa_bot/openai_responder.py +++ b/fjerkroa_bot/openai_responder.py @@ -1,14 +1,14 @@ import asyncio +import base64 import hashlib import json import logging from io import BytesIO from typing import Any, Dict, List, Optional, Tuple -import aiohttp import openai -from .ai_responder import AIResponder, exponential_backoff, pp, sanitize_external_text +from .ai_responder import AIResponder, exponential_backoff, sanitize_external_text from .igdblib import IGDBQuery from .leonardo_draw import LeonardoAIDrawMixIn from .quota import QuotaLedger @@ -24,10 +24,11 @@ ENVELOPE_SCHEMA = { "channel": {"type": ["string", "null"], "description": "Target channel name, or null for the origin channel."}, "staff": {"type": ["string", "null"], "description": "Alert text for the staff channel, or null."}, "picture": {"type": ["string", "null"], "description": "Image generation prompt, or null."}, + "picture_count": {"type": "integer", "description": "How many images to generate (1-4), 1 unless more were asked for."}, "picture_edit": {"type": "boolean", "description": "Whether the picture refers to an earlier image."}, "hack": {"type": "boolean", "description": "Whether the user tried to manipulate the assistant."}, }, - "required": ["answer", "answer_needed", "channel", "staff", "picture", "picture_edit", "hack"], + "required": ["answer", "answer_needed", "channel", "staff", "picture", "picture_count", "picture_edit", "hack"], "additionalProperties": False, } ENVELOPE_RESPONSE_FORMAT = {"type": "json_schema", "json_schema": {"name": "envelope", "strict": True, "schema": ENVELOPE_SCHEMA}} @@ -92,10 +93,7 @@ async def openai_chat(client, *args, **kwargs): async def openai_image(client, *args, **kwargs): - response = await client.images.generate(*args, **kwargs) - async with aiohttp.ClientSession() as session: - async with session.get(response.data[0].url) as image: - return BytesIO(await image.read()) + return await client.images.generate(*args, **kwargs) class OpenAIResponder(AIResponder, LeonardoAIDrawMixIn): @@ -126,15 +124,26 @@ class OpenAIResponder(AIResponder, LeonardoAIDrawMixIn): else: logging.warning("❌ IGDB integration DISABLED - missing configuration or disabled in config") - async def draw_openai(self, description: str) -> BytesIO: + async def draw_openai(self, description: str, count: int = 1) -> List[BytesIO]: if not self.ledger.budget_ok(): raise RuntimeError("daily budget exhausted - refusing image call") + model = self.config.get("image-model", "gpt-image-2") + kwargs: Dict[str, Any] = {"model": model, "prompt": description, "size": self.config.get("image-size", "1024x1024")} + if "image-quality" in self.config: + kwargs["quality"] = self.config["image-quality"] + if model.startswith("gpt-image"): + kwargs["n"] = max(1, min(int(count), 4)) + else: + # legacy models: single image, base64 must be requested (IMG-04) + kwargs["n"] = 1 + kwargs["response_format"] = "b64_json" for _ in range(3): try: - response = await openai_image(self.client, prompt=description, n=1, size="1024x1024", model="dall-e-3") - self.ledger.add_images(1) - logging.info(f"Drawed a picture with DALL-E on this description: {repr(description)}") - return response + response = await openai_image(self.client, **kwargs) + buffers = [BytesIO(base64.b64decode(item.b64_json)) for item in response.data] + self.ledger.add_images(len(buffers)) + logging.info(f"generated {len(buffers)} image(s) on {model} for: {repr(description)}") + return buffers except Exception as err: logging.warning(f"Failed to generate image {repr(description)}: {repr(err)}") raise RuntimeError(f"Failed to generate image {repr(description)} after multiple retries") @@ -209,6 +218,8 @@ class OpenAIResponder(AIResponder, LeonardoAIDrawMixIn): if igdb_functions and isinstance(igdb_functions, list): chat_kwargs["tools"] = [{"type": "function", "function": func} for func in igdb_functions] chat_kwargs["tool_choice"] = "auto" + # gpt-5.6 rejects tools + reasoning on chat/completions (ENV-21) + chat_kwargs["reasoning_effort"] = self.config.get("reasoning-effort", "none") logging.info(f"🎮 IGDB functions available to AI: {[f['name'] for f in igdb_functions]}") logging.debug(f" Full chat_kwargs with tools: {list(chat_kwargs.keys())}") except (TypeError, AttributeError) as e: @@ -343,27 +354,6 @@ class OpenAIResponder(AIResponder, LeonardoAIDrawMixIn): logging.debug(f"Full traceback: {traceback.format_exc()}") return None, limit - async def translate(self, text: str, language: str = "english") -> str: - if "fix-model" not in self.config: - return text - message = [ - { - "role": "system", - "content": f"You are an professional translator to {language} language," - f" you translate everything you get directly to {language}" - f" if it is not already in {language}, otherwise you just copy it.", - }, - {"role": "user", "content": text}, - ] - try: - result = await openai_chat(self.client, model=self.config["fix-model"], messages=message) - response = result.choices[0].message.content - logging.info(f"got this translated message:\n{pp(response)}") - return response - except Exception as err: - logging.warning(f"failed to translate the text: {repr(err)}") - return text - async def classify(self, message: Any, history_tail: List[Dict[str, Any]]) -> Optional[Dict[str, Any]]: """~100-token reply/factual/emoji verdict on classifier-model (BEH-01/03).""" if "classifier-model" not in self.config or not self.ledger.budget_ok(): diff --git a/manual-verification.md b/manual-verification.md index fbf2866..9aa5e0b 100644 --- a/manual-verification.md +++ b/manual-verification.md @@ -8,7 +8,7 @@ with date + result. | --- | --- | --- | | DEP-01 | 2026-07-13 | Verified with the v3.0.0 ggg deploy: tag-only refusal + untracked config/state survived. fjerkroa redeploy after the service window (tree already identical to 3d22894). | | DEP-02 | 2026-07-13 | Service map exercised: luma restart via script (v3.0.0); kroa mapping code-reviewed, exercised on its next deploy. | -| DEP-03 | 2026-07-13 | ggg had no pre-existing bot.db (pickle era) — nothing to back up; backup branch code-reviewed, exercised on the next deploy of either host. | +| DEP-03 | 2026-07-13 | Exercised with the v3.1.0 ggg deploy: bot.db.pre-v3.1.0 confirmed on the host. (v3.0.0 note: no pre-existing db in the pickle era.) | | DEP-04 | 2026-07-13 | Smoke gate exercised on ggg: RUNNING + fresh login line. | | DEP-05 | 2026-07-13 | Live-verified: kroa deploy attempt ~15h Oslo refused without DEPLOY_FORCE=1. | | DEP-06 | 2026-07-13 | Rollback documented (older tag + db backup restore); live drill pending — next release. | diff --git a/specs/SPEC-001-responder-envelope.md b/specs/SPEC-001-responder-envelope.md index 5a46bde..f6748b2 100644 --- a/specs/SPEC-001-responder-envelope.md +++ b/specs/SPEC-001-responder-envelope.md @@ -75,6 +75,14 @@ in a context suffix, not inline — see ENV-20; legacy `{date}`, `{time}`, `{news}`, `{memory}` placeholders in operator templates are stripped. +### ENV-21 — Tool calls disable reasoning effort (coverage: test) + +When function tools are attached to a chat call, the call carries +`reasoning_effort` (config `reasoning-effort`, default `"none"`) — +gpt-5.6 models reject tools + reasoning on chat/completions with a +400 otherwise (found live on ggg 2026-07-13: IGDB tools made Luma +mute after the Luna cutover). Tool-less calls stay untouched. + ### ENV-20 — Persona prefix is byte-stable (coverage: test) `message()` renders the system message as: static persona text @@ -140,6 +148,7 @@ ENV-06. Every chat call carries `response_format` = strict JSON schema named `envelope` with exactly the fields `answer`, `answer_needed`, -`channel`, `staff`, `picture`, `picture_edit`, `hack` — all required, +`channel`, `staff`, `picture`, `picture_count` (since FDB-009, +IMG-02), `picture_edit`, `hack` — all required, `additionalProperties: false`, nullable where the protocol allows null. Tool-followup calls carry the same format. diff --git a/specs/SPEC-004-images.md b/specs/SPEC-004-images.md new file mode 100644 index 0000000..7bfd650 --- /dev/null +++ b/specs/SPEC-004-images.md @@ -0,0 +1,39 @@ +# SPEC-004 — Image generation + +Generation on `image-model` (default `gpt-image-2`), base64 end to +end — no URL downloads, no expiring CDN links in the generation path. +Optional knobs: `image-size` (default 1024x1024), `image-quality` +(passed through only when set). The Leonardo path stays behind +`leonardo-token` until parity is confirmed, then dies. The input +pipeline (attachment cache, vision, edit/remix) is FDB-010 / IMG-10+. + +### IMG-01 — Images arrive as base64 buffers (coverage: test) + +`draw_openai(description, count)` requests `count` images and returns +a list of decoded image buffers straight from the API response; every +generated image is metered in the ledger (SAF-05). + +### IMG-02 — The envelope carries picture_count (coverage: test) + +The envelope gains `picture_count` (integer). `post_process` clamps +it to 1..4 and defaults to 1 when absent (legacy history entries, +old-model output). ENV-19's field list is revised accordingly. + +### IMG-03 — Multiple images, one message (coverage: test) + +`picture_count` images are attached as multiple files to a single +Discord send (the last part when the answer is split, per BEH-06). + +### IMG-04 — Legacy image models degrade safely (coverage: test) + +When `image-model` is not a `gpt-image-*` model (e.g. `dall-e-3`), +the count is clamped to 1 and `response_format="b64_json"` is +requested explicitly (gpt-image models return base64 natively and +reject the parameter). + +### IMG-05 — Picture prompts go to the API untouched (coverage: test) + +The translate-before-draw step is deleted: the model's picture prompt +reaches the image API verbatim (current image models handle +Norwegian/German natively). The `translate()` method and its +`fix-model` dependency are gone (closes D-009). diff --git a/tests/test_ai.py b/tests/test_ai.py index cd26aaf..63079ca 100644 --- a/tests/test_ai.py +++ b/tests/test_ai.py @@ -102,33 +102,6 @@ You always try to say something positive about the current day and the Fjærkroa # Skip this test due to Mock iteration issues - functionality works in practice self.skipTest("Mock iteration issue - test works in real usage") - async def test_translate1(self) -> None: - self.bot.airesponder.config["fix-model"] = "gpt-4o-mini" - - # Mock translation responses - def translation_side_effect(*args, **kwargs): - mock_resp = Mock() - mock_resp.choices = [Mock()] - mock_resp.choices[0].message = Mock() - - # Check the input text to return appropriate translation - user_content = kwargs["messages"][1]["content"] - if user_content == "Das ist ein komischer Text.": - mock_resp.choices[0].message.content = "This is a strange text." - elif user_content == "This is a strange text.": - mock_resp.choices[0].message.content = "Dies ist ein seltsamer Text." - else: - mock_resp.choices[0].message.content = user_content - - return mock_resp - - self.mock_openai_chat.side_effect = translation_side_effect - - response = await self.bot.airesponder.translate("Das ist ein komischer Text.") - self.assertEqual(response, "This is a strange text.") - response = await self.bot.airesponder.translate("This is a strange text.", language="german") - self.assertEqual(response, "Dies ist ein seltsamer Text.") - async def test_fix1(self) -> None: # Skip this test due to Mock iteration issues - functionality works in practice self.skipTest("Mock iteration issue - test works in real usage") diff --git a/tests/test_bdd_envelope.py b/tests/test_bdd_envelope.py index 4776287..2d5fc4e 100644 --- a/tests/test_bdd_envelope.py +++ b/tests/test_bdd_envelope.py @@ -49,9 +49,6 @@ class FakeModelResponder(AIResponder): async def classify(self, message, history_tail): return getattr(self, "scripted_classification", None) - async def translate(self, text: str, language: str = "english") -> str: - return text - @given(parsers.parse("a responder with history limit {limit:d}"), target_fixture="responder") def responder(limit): diff --git a/tests/test_openai_responder_simple.py b/tests/test_openai_responder_simple.py index 402ed5c..00a7d41 100644 --- a/tests/test_openai_responder_simple.py +++ b/tests/test_openai_responder_simple.py @@ -30,16 +30,6 @@ class TestOpenAIResponderSimple(unittest.IsolatedAsyncioTestCase): """ENV-18: the repair path is gone — no fix() on the responder.""" self.assertFalse(hasattr(self.responder, "fix")) - async def test_translate_no_fix_model(self): - """Test translate when no fix-model is configured.""" - config_no_fix = {"openai-key": "test", "model": "gpt-4"} - responder = OpenAIResponder(config_no_fix) - - original_text = "Hello world" - result = await responder.translate(original_text) - - self.assertEqual(result, original_text) - async def test_consolidate_no_memory_model(self): """MEM-10: without memory-model, consolidation is a no-op returning None.""" config_no_memory = {"openai-key": "test", "model": "gpt-4"} diff --git a/tests/test_spec_beh.py b/tests/test_spec_beh.py index c497e3c..70ba886 100644 --- a/tests/test_spec_beh.py +++ b/tests/test_spec_beh.py @@ -121,7 +121,7 @@ class TestSplitSends(OpsBase): self.bot.config["split-threshold"] = 50 answer = "Første del.\n\nAndre del som også er ganske lang her." response = AIResponse(answer, True, "chat", None, "a cat", False, False) - self.bot.airesponder.draw = AsyncMock(return_value=__import__("io").BytesIO(b"png")) + self.bot.airesponder.draw = AsyncMock(return_value=[__import__("io").BytesIO(b"png")]) channel = MagicMock(spec=TextChannel) channel.send = AsyncMock() await self.bot.send_answer_with_typing(response, channel, self.bot.airesponder, factual=True) diff --git a/tests/test_spec_img.py b/tests/test_spec_img.py new file mode 100644 index 0000000..ed84997 --- /dev/null +++ b/tests/test_spec_img.py @@ -0,0 +1,87 @@ +"""Unit coverage for SPEC-004 image generation (IMG-01..05).""" + +import base64 +import unittest +from unittest.mock import AsyncMock, MagicMock, Mock, patch + +from discord import TextChannel + +from fjerkroa_bot.ai_responder import AIMessage, AIResponder, AIResponse +from fjerkroa_bot.openai_responder import OpenAIResponder + +from .test_bdd_envelope import FakeModelResponder, envelope +from .test_spec_ops import OpsBase + +RESPONDER_CONFIG = {"openai-token": "t", "model": "m", "system": "s", "history-limit": 5} + + +def image_api_result(count): + return Mock(data=[Mock(b64_json=base64.b64encode(f"png{i}".encode()).decode()) for i in range(count)]) + + +class TestBase64Generation(unittest.IsolatedAsyncioTestCase): + async def test_draw_returns_decoded_buffers_and_meters(self): + """IMG-01: count images decoded from b64_json, each metered in the ledger.""" + responder = OpenAIResponder(RESPONDER_CONFIG, "chat") + with patch("fjerkroa_bot.openai_responder.openai_image", new_callable=AsyncMock) as image_mock: + image_mock.return_value = image_api_result(2) + buffers = await responder.draw_openai("en katt på brygga", 2) + self.assertEqual([buf.read() for buf in buffers], [b"png0", b"png1"]) + self.assertEqual(responder.ledger.images_today(), 2) + self.assertEqual(image_mock.await_args.kwargs["n"], 2) + self.assertEqual(image_mock.await_args.kwargs["model"], "gpt-image-2") + self.assertNotIn("response_format", image_mock.await_args.kwargs) + + async def test_legacy_model_clamped_single_b64(self): + """IMG-04: dall-e-3 -> n=1 and explicit response_format=b64_json.""" + config = dict(RESPONDER_CONFIG, **{"image-model": "dall-e-3"}) + responder = OpenAIResponder(config, "chat") + with patch("fjerkroa_bot.openai_responder.openai_image", new_callable=AsyncMock) as image_mock: + image_mock.return_value = image_api_result(1) + buffers = await responder.draw_openai("a cat", 3) + self.assertEqual(len(buffers), 1) + self.assertEqual(image_mock.await_args.kwargs["n"], 1) + self.assertEqual(image_mock.await_args.kwargs["response_format"], "b64_json") + + +class TestPictureCountEnvelope(unittest.IsolatedAsyncioTestCase): + async def clamp(self, raw): + responder = FakeModelResponder({"system": "s", "history-limit": 5}, "chat") + payload = {"answer": "ok", "answer_needed": True, "channel": "chat", "picture": "katt"} + if raw is not None: + payload["picture_count"] = raw + return await responder.post_process(AIMessage("alice", "tegn", "chat"), payload) + + async def test_clamped_and_defaulted(self): + """IMG-02: picture_count clamps to 1..4, defaults to 1 when absent.""" + self.assertEqual((await self.clamp(3)).picture_count, 3) + self.assertEqual((await self.clamp(9)).picture_count, 4) + self.assertEqual((await self.clamp(0)).picture_count, 1) + self.assertEqual((await self.clamp(None)).picture_count, 1) + + +class TestMultiImageSend(OpsBase): + async def test_files_attached_to_single_send(self): + """IMG-03: picture_count images ride as multiple files on one send.""" + response = AIResponse("her er kattene", True, "chat", None, "to katter", False, False) + response.picture_count = 2 + import io + + self.bot.airesponder.draw = AsyncMock(return_value=[io.BytesIO(b"a"), io.BytesIO(b"b")]) + channel = MagicMock(spec=TextChannel) + channel.send = AsyncMock() + await self.bot.send_answer_with_typing(response, channel, self.bot.airesponder, factual=True) + self.bot.airesponder.draw.assert_awaited_once_with("to katter", 2) + files = channel.send.await_args.kwargs["files"] + self.assertEqual(len(files), 2) + + +class TestNoTranslateStep(unittest.IsolatedAsyncioTestCase): + async def test_translate_is_gone_prompt_untouched(self): + """IMG-05: no translate() anywhere; the picture prompt survives verbatim.""" + responder = FakeModelResponder({"system": "s", "history-limit": 5}, "chat") + self.assertFalse(hasattr(responder, "translate")) + self.assertFalse(hasattr(AIResponder, "translate")) + responder.scripted.append(envelope(answer="ok", answer_needed=True, picture="en rød katt på brygga")) + result = await responder.send(AIMessage("alice", "tegn en katt", "chat")) + self.assertEqual(result.picture, "en rød katt på brygga") diff --git a/tests/test_spec_structured.py b/tests/test_spec_structured.py index b735539..cafbe54 100644 --- a/tests/test_spec_structured.py +++ b/tests/test_spec_structured.py @@ -43,7 +43,7 @@ class TestEnvelopeSchema(unittest.IsolatedAsyncioTestCase): """ENV-19: strict envelope schema — exact fields, all required, closed object.""" json_schema = ENVELOPE_RESPONSE_FORMAT["json_schema"] schema = json_schema["schema"] - expected = {"answer", "answer_needed", "channel", "staff", "picture", "picture_edit", "hack"} + expected = {"answer", "answer_needed", "channel", "staff", "picture", "picture_count", "picture_edit", "hack"} self.assertEqual(set(schema["properties"]), expected) self.assertEqual(set(schema["required"]), expected) self.assertFalse(schema["additionalProperties"]) diff --git a/tests/test_spec_tools.py b/tests/test_spec_tools.py new file mode 100644 index 0000000..a0b40ac --- /dev/null +++ b/tests/test_spec_tools.py @@ -0,0 +1,38 @@ +"""Unit coverage for ENV-21 (tools + reasoning_effort, found live on ggg).""" + +import unittest +from unittest.mock import AsyncMock, Mock, patch + +from fjerkroa_bot.openai_responder import OpenAIResponder + +from .test_bdd_envelope import envelope + + +def ok_result(): + message = Mock(content=envelope(answer="x", answer_needed=True), role="assistant", tool_calls=None, refusal=None) + return Mock(choices=[Mock(message=message)], usage="usage") + + +class TestToolsReasoningEffort(unittest.IsolatedAsyncioTestCase): + async def chat_kwargs(self, with_tools): + config = {"openai-token": "t", "model": "gpt-5.6-luna", "system": "s", "history-limit": 5, "enable-game-info": with_tools} + responder = OpenAIResponder(config, "chat") + if with_tools: + responder.igdb = Mock() + responder.igdb.get_openai_functions = Mock(return_value=[{"name": "search_games", "parameters": {}}]) + with patch("fjerkroa_bot.openai_responder.openai_chat", new_callable=AsyncMock) as chat_mock: + chat_mock.return_value = ok_result() + await responder.chat([{"role": "user", "content": "hi"}], 10) + return chat_mock.await_args.kwargs + + async def test_tools_carry_reasoning_effort_none(self): + """ENV-21: tools attached -> reasoning_effort 'none' rides along.""" + kwargs = await self.chat_kwargs(with_tools=True) + self.assertIn("tools", kwargs) + self.assertEqual(kwargs["reasoning_effort"], "none") + + async def test_toolless_calls_untouched(self): + """ENV-21: without tools no reasoning_effort is sent.""" + kwargs = await self.chat_kwargs(with_tools=False) + self.assertNotIn("tools", kwargs) + self.assertNotIn("reasoning_effort", kwargs)