Compare commits

...

1 Commits

Author SHA1 Message Date
Oleksandr Kozachuk 7e6eae10ee gpt-image-2 multi-image + fix: tools need reasoning_effort none on gpt-5.6 (ggg mute bug) 2026-07-13 16:17:00 +02:00
15 changed files with 220 additions and 88 deletions
+3 -1
View File
@@ -35,7 +35,9 @@ Decisions inside the set architecture. D-NNN, never renumbered.
- **D-009** — `translate()` still keys off `fix-model` although the
repair path is gone; the whole translate-before-draw step dies in
FDB-009 (gpt-image-2 is multilingual). Not worth a config rename
for one phase.
for one phase. *(Closed 2026-07-13: FDB-009 deleted translate();
the fix-model config key is now fully dead and can be dropped from
live configs.)*
- **D-010** — Persistence uses stdlib `sqlite3` via
`asyncio.to_thread`, not aiosqlite: no new dependency, and a
connection-per-operation with WAL is plenty at this message volume.
+5
View File
@@ -50,3 +50,8 @@ enable-game-info = true
# split-threshold = 1200 # long answers split at paragraphs
# split-max-parts = 3
# quiet-hours = "21:00-09:00" # no bot-initiated posts in this window
# Image generation (SPEC-004)
# image-model = "gpt-image-2" # default; dall-e-3 gets clamped to n=1
# image-size = "1024x1024"
# image-quality = "medium" # passed through only when set
+10 -9
View File
@@ -123,6 +123,7 @@ class AIResponse(AIMessageBase):
self.channel = channel
self.staff = staff
self.picture = picture
self.picture_count = 1
self.picture_edit = picture_edit
self.hack = hack
self.vars = ["answer", "answer_needed", "channel", "staff", "picture", "hack"]
@@ -188,15 +189,15 @@ class AIResponder(AIResponderBase):
messages.append({"role": "user", "content": content})
return messages
async def draw(self, description: str) -> BytesIO:
async def draw(self, description: str, count: int = 1) -> List[BytesIO]:
if self.config.get("leonardo-token") is not None:
return await self.draw_leonardo(description)
return await self.draw_openai(description)
return [await self.draw_leonardo(description)] # single image only, behind config
return await self.draw_openai(description, count)
async def draw_leonardo(self, description: str) -> BytesIO:
raise NotImplementedError()
async def draw_openai(self, description: str) -> BytesIO:
async def draw_openai(self, description: str, count: int = 1) -> List[BytesIO]:
raise NotImplementedError()
async def post_process(self, message: AIMessage, response: Dict[str, Any]) -> AIResponse:
@@ -221,6 +222,10 @@ class AIResponder(AIResponderBase):
bool(response.get("picture_edit", False)),
bool(response.get("hack", False)),
)
try:
response_message.picture_count = max(1, min(int(response.get("picture_count") or 1), 4)) # IMG-02
except (TypeError, ValueError):
response_message.picture_count = 1
if response_message.staff is not None and response_message.answer is not None:
response_message.answer_needed = True
if response_message.channel is None:
@@ -250,9 +255,6 @@ class AIResponder(AIResponderBase):
"""Cheap reply/factual/emoji pre-pass (BEH-01); None = fail open."""
raise NotImplementedError()
async def translate(self, text: str, language: str = "english") -> str:
raise NotImplementedError()
@staticmethod
def _entry_channel(item: Dict[str, Any]) -> Optional[str]:
try:
@@ -299,11 +301,10 @@ class AIResponder(AIResponderBase):
await asyncio.to_thread(self.store.save_history, self.channel, list(self.history))
async def handle_picture(self, response: Dict) -> bool:
# Prompt goes to the image API verbatim — no translate step (IMG-05)
if not isinstance(response.get("picture"), (type(None), str)):
logging.warning(f"picture key is wrong in response: {pp(response)}")
return False
if response.get("picture") is not None:
response["picture"] = await self.translate(response["picture"])
return True
def _parse_answer(self, answer: Dict[str, Any]) -> Optional[Dict[str, Any]]:
+2 -1
View File
@@ -462,7 +462,8 @@ class FjerkroaBot(commands.Bot):
"""Send the answer paced, split and with images on the last part (BEH-04/05/06)"""
files = None
if response.picture is not None:
files = [discord.File(fp=await airesponder.draw(response.picture), filename="image.png")]
buffers = await airesponder.draw(response.picture, getattr(response, "picture_count", 1))
files = [discord.File(fp=buffer, filename=f"image-{index}.png") for index, buffer in enumerate(buffers)]
parts = split_answer(response.answer, int(self.config.get("split-threshold", 1200)), int(self.config.get("split-max-parts", 3)))
pace = float(self.config.get("typing-chars-per-second", 0) or 0)
max_delay = float(self.config.get("typing-max-seconds", 8))
+23 -33
View File
@@ -1,14 +1,14 @@
import asyncio
import base64
import hashlib
import json
import logging
from io import BytesIO
from typing import Any, Dict, List, Optional, Tuple
import aiohttp
import openai
from .ai_responder import AIResponder, exponential_backoff, pp, sanitize_external_text
from .ai_responder import AIResponder, exponential_backoff, sanitize_external_text
from .igdblib import IGDBQuery
from .leonardo_draw import LeonardoAIDrawMixIn
from .quota import QuotaLedger
@@ -24,10 +24,11 @@ ENVELOPE_SCHEMA = {
"channel": {"type": ["string", "null"], "description": "Target channel name, or null for the origin channel."},
"staff": {"type": ["string", "null"], "description": "Alert text for the staff channel, or null."},
"picture": {"type": ["string", "null"], "description": "Image generation prompt, or null."},
"picture_count": {"type": "integer", "description": "How many images to generate (1-4), 1 unless more were asked for."},
"picture_edit": {"type": "boolean", "description": "Whether the picture refers to an earlier image."},
"hack": {"type": "boolean", "description": "Whether the user tried to manipulate the assistant."},
},
"required": ["answer", "answer_needed", "channel", "staff", "picture", "picture_edit", "hack"],
"required": ["answer", "answer_needed", "channel", "staff", "picture", "picture_count", "picture_edit", "hack"],
"additionalProperties": False,
}
ENVELOPE_RESPONSE_FORMAT = {"type": "json_schema", "json_schema": {"name": "envelope", "strict": True, "schema": ENVELOPE_SCHEMA}}
@@ -92,10 +93,7 @@ async def openai_chat(client, *args, **kwargs):
async def openai_image(client, *args, **kwargs):
response = await client.images.generate(*args, **kwargs)
async with aiohttp.ClientSession() as session:
async with session.get(response.data[0].url) as image:
return BytesIO(await image.read())
return await client.images.generate(*args, **kwargs)
class OpenAIResponder(AIResponder, LeonardoAIDrawMixIn):
@@ -126,15 +124,26 @@ class OpenAIResponder(AIResponder, LeonardoAIDrawMixIn):
else:
logging.warning("❌ IGDB integration DISABLED - missing configuration or disabled in config")
async def draw_openai(self, description: str) -> BytesIO:
async def draw_openai(self, description: str, count: int = 1) -> List[BytesIO]:
if not self.ledger.budget_ok():
raise RuntimeError("daily budget exhausted - refusing image call")
model = self.config.get("image-model", "gpt-image-2")
kwargs: Dict[str, Any] = {"model": model, "prompt": description, "size": self.config.get("image-size", "1024x1024")}
if "image-quality" in self.config:
kwargs["quality"] = self.config["image-quality"]
if model.startswith("gpt-image"):
kwargs["n"] = max(1, min(int(count), 4))
else:
# legacy models: single image, base64 must be requested (IMG-04)
kwargs["n"] = 1
kwargs["response_format"] = "b64_json"
for _ in range(3):
try:
response = await openai_image(self.client, prompt=description, n=1, size="1024x1024", model="dall-e-3")
self.ledger.add_images(1)
logging.info(f"Drawed a picture with DALL-E on this description: {repr(description)}")
return response
response = await openai_image(self.client, **kwargs)
buffers = [BytesIO(base64.b64decode(item.b64_json)) for item in response.data]
self.ledger.add_images(len(buffers))
logging.info(f"generated {len(buffers)} image(s) on {model} for: {repr(description)}")
return buffers
except Exception as err:
logging.warning(f"Failed to generate image {repr(description)}: {repr(err)}")
raise RuntimeError(f"Failed to generate image {repr(description)} after multiple retries")
@@ -209,6 +218,8 @@ class OpenAIResponder(AIResponder, LeonardoAIDrawMixIn):
if igdb_functions and isinstance(igdb_functions, list):
chat_kwargs["tools"] = [{"type": "function", "function": func} for func in igdb_functions]
chat_kwargs["tool_choice"] = "auto"
# gpt-5.6 rejects tools + reasoning on chat/completions (ENV-21)
chat_kwargs["reasoning_effort"] = self.config.get("reasoning-effort", "none")
logging.info(f"🎮 IGDB functions available to AI: {[f['name'] for f in igdb_functions]}")
logging.debug(f" Full chat_kwargs with tools: {list(chat_kwargs.keys())}")
except (TypeError, AttributeError) as e:
@@ -343,27 +354,6 @@ class OpenAIResponder(AIResponder, LeonardoAIDrawMixIn):
logging.debug(f"Full traceback: {traceback.format_exc()}")
return None, limit
async def translate(self, text: str, language: str = "english") -> str:
if "fix-model" not in self.config:
return text
message = [
{
"role": "system",
"content": f"You are an professional translator to {language} language,"
f" you translate everything you get directly to {language}"
f" if it is not already in {language}, otherwise you just copy it.",
},
{"role": "user", "content": text},
]
try:
result = await openai_chat(self.client, model=self.config["fix-model"], messages=message)
response = result.choices[0].message.content
logging.info(f"got this translated message:\n{pp(response)}")
return response
except Exception as err:
logging.warning(f"failed to translate the text: {repr(err)}")
return text
async def classify(self, message: Any, history_tail: List[Dict[str, Any]]) -> Optional[Dict[str, Any]]:
"""~100-token reply/factual/emoji verdict on classifier-model (BEH-01/03)."""
if "classifier-model" not in self.config or not self.ledger.budget_ok():
+1 -1
View File
@@ -8,7 +8,7 @@ with date + result.
| --- | --- | --- |
| DEP-01 | 2026-07-13 | Verified with the v3.0.0 ggg deploy: tag-only refusal + untracked config/state survived. fjerkroa redeploy after the service window (tree already identical to 3d22894). |
| DEP-02 | 2026-07-13 | Service map exercised: luma restart via script (v3.0.0); kroa mapping code-reviewed, exercised on its next deploy. |
| DEP-03 | 2026-07-13 | ggg had no pre-existing bot.db (pickle era) — nothing to back up; backup branch code-reviewed, exercised on the next deploy of either host. |
| DEP-03 | 2026-07-13 | Exercised with the v3.1.0 ggg deploy: bot.db.pre-v3.1.0 confirmed on the host. (v3.0.0 note: no pre-existing db in the pickle era.) |
| DEP-04 | 2026-07-13 | Smoke gate exercised on ggg: RUNNING + fresh login line. |
| DEP-05 | 2026-07-13 | Live-verified: kroa deploy attempt ~15h Oslo refused without DEPLOY_FORCE=1. |
| DEP-06 | 2026-07-13 | Rollback documented (older tag + db backup restore); live drill pending — next release. |
+10 -1
View File
@@ -75,6 +75,14 @@ in a context suffix, not inline — see ENV-20; legacy `{date}`,
`{time}`, `{news}`, `{memory}` placeholders in operator templates are
stripped.
### ENV-21 — Tool calls disable reasoning effort (coverage: test)
When function tools are attached to a chat call, the call carries
`reasoning_effort` (config `reasoning-effort`, default `"none"`) —
gpt-5.6 models reject tools + reasoning on chat/completions with a
400 otherwise (found live on ggg 2026-07-13: IGDB tools made Luma
mute after the Luna cutover). Tool-less calls stay untouched.
### ENV-20 — Persona prefix is byte-stable (coverage: test)
`message()` renders the system message as: static persona text
@@ -140,6 +148,7 @@ ENV-06.
Every chat call carries `response_format` = strict JSON schema named
`envelope` with exactly the fields `answer`, `answer_needed`,
`channel`, `staff`, `picture`, `picture_edit`, `hack` — all required,
`channel`, `staff`, `picture`, `picture_count` (since FDB-009,
IMG-02), `picture_edit`, `hack` — all required,
`additionalProperties: false`, nullable where the protocol allows
null. Tool-followup calls carry the same format.
+39
View File
@@ -0,0 +1,39 @@
# SPEC-004 — Image generation
Generation on `image-model` (default `gpt-image-2`), base64 end to
end — no URL downloads, no expiring CDN links in the generation path.
Optional knobs: `image-size` (default 1024x1024), `image-quality`
(passed through only when set). The Leonardo path stays behind
`leonardo-token` until parity is confirmed, then dies. The input
pipeline (attachment cache, vision, edit/remix) is FDB-010 / IMG-10+.
### IMG-01 — Images arrive as base64 buffers (coverage: test)
`draw_openai(description, count)` requests `count` images and returns
a list of decoded image buffers straight from the API response; every
generated image is metered in the ledger (SAF-05).
### IMG-02 — The envelope carries picture_count (coverage: test)
The envelope gains `picture_count` (integer). `post_process` clamps
it to 1..4 and defaults to 1 when absent (legacy history entries,
old-model output). ENV-19's field list is revised accordingly.
### IMG-03 — Multiple images, one message (coverage: test)
`picture_count` images are attached as multiple files to a single
Discord send (the last part when the answer is split, per BEH-06).
### IMG-04 — Legacy image models degrade safely (coverage: test)
When `image-model` is not a `gpt-image-*` model (e.g. `dall-e-3`),
the count is clamped to 1 and `response_format="b64_json"` is
requested explicitly (gpt-image models return base64 natively and
reject the parameter).
### IMG-05 — Picture prompts go to the API untouched (coverage: test)
The translate-before-draw step is deleted: the model's picture prompt
reaches the image API verbatim (current image models handle
Norwegian/German natively). The `translate()` method and its
`fix-model` dependency are gone (closes D-009).
-27
View File
@@ -102,33 +102,6 @@ You always try to say something positive about the current day and the Fjærkroa
# Skip this test due to Mock iteration issues - functionality works in practice
self.skipTest("Mock iteration issue - test works in real usage")
async def test_translate1(self) -> None:
self.bot.airesponder.config["fix-model"] = "gpt-4o-mini"
# Mock translation responses
def translation_side_effect(*args, **kwargs):
mock_resp = Mock()
mock_resp.choices = [Mock()]
mock_resp.choices[0].message = Mock()
# Check the input text to return appropriate translation
user_content = kwargs["messages"][1]["content"]
if user_content == "Das ist ein komischer Text.":
mock_resp.choices[0].message.content = "This is a strange text."
elif user_content == "This is a strange text.":
mock_resp.choices[0].message.content = "Dies ist ein seltsamer Text."
else:
mock_resp.choices[0].message.content = user_content
return mock_resp
self.mock_openai_chat.side_effect = translation_side_effect
response = await self.bot.airesponder.translate("Das ist ein komischer Text.")
self.assertEqual(response, "This is a strange text.")
response = await self.bot.airesponder.translate("This is a strange text.", language="german")
self.assertEqual(response, "Dies ist ein seltsamer Text.")
async def test_fix1(self) -> None:
# Skip this test due to Mock iteration issues - functionality works in practice
self.skipTest("Mock iteration issue - test works in real usage")
-3
View File
@@ -49,9 +49,6 @@ class FakeModelResponder(AIResponder):
async def classify(self, message, history_tail):
return getattr(self, "scripted_classification", None)
async def translate(self, text: str, language: str = "english") -> str:
return text
@given(parsers.parse("a responder with history limit {limit:d}"), target_fixture="responder")
def responder(limit):
-10
View File
@@ -30,16 +30,6 @@ class TestOpenAIResponderSimple(unittest.IsolatedAsyncioTestCase):
"""ENV-18: the repair path is gone — no fix() on the responder."""
self.assertFalse(hasattr(self.responder, "fix"))
async def test_translate_no_fix_model(self):
"""Test translate when no fix-model is configured."""
config_no_fix = {"openai-key": "test", "model": "gpt-4"}
responder = OpenAIResponder(config_no_fix)
original_text = "Hello world"
result = await responder.translate(original_text)
self.assertEqual(result, original_text)
async def test_consolidate_no_memory_model(self):
"""MEM-10: without memory-model, consolidation is a no-op returning None."""
config_no_memory = {"openai-key": "test", "model": "gpt-4"}
+1 -1
View File
@@ -121,7 +121,7 @@ class TestSplitSends(OpsBase):
self.bot.config["split-threshold"] = 50
answer = "Første del.\n\nAndre del som også er ganske lang her."
response = AIResponse(answer, True, "chat", None, "a cat", False, False)
self.bot.airesponder.draw = AsyncMock(return_value=__import__("io").BytesIO(b"png"))
self.bot.airesponder.draw = AsyncMock(return_value=[__import__("io").BytesIO(b"png")])
channel = MagicMock(spec=TextChannel)
channel.send = AsyncMock()
await self.bot.send_answer_with_typing(response, channel, self.bot.airesponder, factual=True)
+87
View File
@@ -0,0 +1,87 @@
"""Unit coverage for SPEC-004 image generation (IMG-01..05)."""
import base64
import unittest
from unittest.mock import AsyncMock, MagicMock, Mock, patch
from discord import TextChannel
from fjerkroa_bot.ai_responder import AIMessage, AIResponder, AIResponse
from fjerkroa_bot.openai_responder import OpenAIResponder
from .test_bdd_envelope import FakeModelResponder, envelope
from .test_spec_ops import OpsBase
RESPONDER_CONFIG = {"openai-token": "t", "model": "m", "system": "s", "history-limit": 5}
def image_api_result(count):
return Mock(data=[Mock(b64_json=base64.b64encode(f"png{i}".encode()).decode()) for i in range(count)])
class TestBase64Generation(unittest.IsolatedAsyncioTestCase):
async def test_draw_returns_decoded_buffers_and_meters(self):
"""IMG-01: count images decoded from b64_json, each metered in the ledger."""
responder = OpenAIResponder(RESPONDER_CONFIG, "chat")
with patch("fjerkroa_bot.openai_responder.openai_image", new_callable=AsyncMock) as image_mock:
image_mock.return_value = image_api_result(2)
buffers = await responder.draw_openai("en katt på brygga", 2)
self.assertEqual([buf.read() for buf in buffers], [b"png0", b"png1"])
self.assertEqual(responder.ledger.images_today(), 2)
self.assertEqual(image_mock.await_args.kwargs["n"], 2)
self.assertEqual(image_mock.await_args.kwargs["model"], "gpt-image-2")
self.assertNotIn("response_format", image_mock.await_args.kwargs)
async def test_legacy_model_clamped_single_b64(self):
"""IMG-04: dall-e-3 -> n=1 and explicit response_format=b64_json."""
config = dict(RESPONDER_CONFIG, **{"image-model": "dall-e-3"})
responder = OpenAIResponder(config, "chat")
with patch("fjerkroa_bot.openai_responder.openai_image", new_callable=AsyncMock) as image_mock:
image_mock.return_value = image_api_result(1)
buffers = await responder.draw_openai("a cat", 3)
self.assertEqual(len(buffers), 1)
self.assertEqual(image_mock.await_args.kwargs["n"], 1)
self.assertEqual(image_mock.await_args.kwargs["response_format"], "b64_json")
class TestPictureCountEnvelope(unittest.IsolatedAsyncioTestCase):
async def clamp(self, raw):
responder = FakeModelResponder({"system": "s", "history-limit": 5}, "chat")
payload = {"answer": "ok", "answer_needed": True, "channel": "chat", "picture": "katt"}
if raw is not None:
payload["picture_count"] = raw
return await responder.post_process(AIMessage("alice", "tegn", "chat"), payload)
async def test_clamped_and_defaulted(self):
"""IMG-02: picture_count clamps to 1..4, defaults to 1 when absent."""
self.assertEqual((await self.clamp(3)).picture_count, 3)
self.assertEqual((await self.clamp(9)).picture_count, 4)
self.assertEqual((await self.clamp(0)).picture_count, 1)
self.assertEqual((await self.clamp(None)).picture_count, 1)
class TestMultiImageSend(OpsBase):
async def test_files_attached_to_single_send(self):
"""IMG-03: picture_count images ride as multiple files on one send."""
response = AIResponse("her er kattene", True, "chat", None, "to katter", False, False)
response.picture_count = 2
import io
self.bot.airesponder.draw = AsyncMock(return_value=[io.BytesIO(b"a"), io.BytesIO(b"b")])
channel = MagicMock(spec=TextChannel)
channel.send = AsyncMock()
await self.bot.send_answer_with_typing(response, channel, self.bot.airesponder, factual=True)
self.bot.airesponder.draw.assert_awaited_once_with("to katter", 2)
files = channel.send.await_args.kwargs["files"]
self.assertEqual(len(files), 2)
class TestNoTranslateStep(unittest.IsolatedAsyncioTestCase):
async def test_translate_is_gone_prompt_untouched(self):
"""IMG-05: no translate() anywhere; the picture prompt survives verbatim."""
responder = FakeModelResponder({"system": "s", "history-limit": 5}, "chat")
self.assertFalse(hasattr(responder, "translate"))
self.assertFalse(hasattr(AIResponder, "translate"))
responder.scripted.append(envelope(answer="ok", answer_needed=True, picture="en rød katt på brygga"))
result = await responder.send(AIMessage("alice", "tegn en katt", "chat"))
self.assertEqual(result.picture, "en rød katt på brygga")
+1 -1
View File
@@ -43,7 +43,7 @@ class TestEnvelopeSchema(unittest.IsolatedAsyncioTestCase):
"""ENV-19: strict envelope schema — exact fields, all required, closed object."""
json_schema = ENVELOPE_RESPONSE_FORMAT["json_schema"]
schema = json_schema["schema"]
expected = {"answer", "answer_needed", "channel", "staff", "picture", "picture_edit", "hack"}
expected = {"answer", "answer_needed", "channel", "staff", "picture", "picture_count", "picture_edit", "hack"}
self.assertEqual(set(schema["properties"]), expected)
self.assertEqual(set(schema["required"]), expected)
self.assertFalse(schema["additionalProperties"])
+38
View File
@@ -0,0 +1,38 @@
"""Unit coverage for ENV-21 (tools + reasoning_effort, found live on ggg)."""
import unittest
from unittest.mock import AsyncMock, Mock, patch
from fjerkroa_bot.openai_responder import OpenAIResponder
from .test_bdd_envelope import envelope
def ok_result():
message = Mock(content=envelope(answer="x", answer_needed=True), role="assistant", tool_calls=None, refusal=None)
return Mock(choices=[Mock(message=message)], usage="usage")
class TestToolsReasoningEffort(unittest.IsolatedAsyncioTestCase):
async def chat_kwargs(self, with_tools):
config = {"openai-token": "t", "model": "gpt-5.6-luna", "system": "s", "history-limit": 5, "enable-game-info": with_tools}
responder = OpenAIResponder(config, "chat")
if with_tools:
responder.igdb = Mock()
responder.igdb.get_openai_functions = Mock(return_value=[{"name": "search_games", "parameters": {}}])
with patch("fjerkroa_bot.openai_responder.openai_chat", new_callable=AsyncMock) as chat_mock:
chat_mock.return_value = ok_result()
await responder.chat([{"role": "user", "content": "hi"}], 10)
return chat_mock.await_args.kwargs
async def test_tools_carry_reasoning_effort_none(self):
"""ENV-21: tools attached -> reasoning_effort 'none' rides along."""
kwargs = await self.chat_kwargs(with_tools=True)
self.assertIn("tools", kwargs)
self.assertEqual(kwargs["reasoning_effort"], "none")
async def test_toolless_calls_untouched(self):
"""ENV-21: without tools no reasoning_effort is sent."""
kwargs = await self.chat_kwargs(with_tools=False)
self.assertNotIn("tools", kwargs)
self.assertNotIn("reasoning_effort", kwargs)