diff --git a/DECISIONS.md b/DECISIONS.md index d846953..d5bb0ba 100644 --- a/DECISIONS.md +++ b/DECISIONS.md @@ -70,6 +70,16 @@ Decisions inside the set architecture. D-NNN, never renumbered. depth); each is independently skippable when it has no data, so a deployment without a budget or store still runs the others. Opt-in (`enable-monitoring`) like every other operational rollout. +- **D-021** — Responses API behind `use-responses-api` (FDB-028, + ENV-22..24, resolves D-006): the responder path can use + `/v1/responses`, which allows tools + `reasoning_effort` (the + chat/completions 400 from ENV-21) and keeps one chain of thought + across tool rounds. Stateless by choice: `store=false` + + encrypted reasoning items passed back — GDPR posture unchanged, no + server-side conversation retention. Flag defaults off; rollback is + a config toggle (hot-reload), not a deploy. Classifier / + consolidation / task-gen stay on chat/completions (no tools, no + reasoning need — not worth the churn). - **D-020** — Web search via Exa (FDB-022, SPEC-015): a `web_search` tool alongside fetch_url/IGDB/codex/get_news, filling the "look it up on the open web" gap. Exa (not a raw search-engine scrape) because it diff --git a/fjerkroa_bot/openai_responder.py b/fjerkroa_bot/openai_responder.py index 2397383..042a2aa 100644 --- a/fjerkroa_bot/openai_responder.py +++ b/fjerkroa_bot/openai_responder.py @@ -40,6 +40,9 @@ ENVELOPE_SCHEMA = { "additionalProperties": False, } ENVELOPE_RESPONSE_FORMAT = {"type": "json_schema", "json_schema": {"name": "envelope", "strict": True, "schema": ENVELOPE_SCHEMA}} +# Same schema in the Responses API shape (ENV-22): text.format is flat, not nested under json_schema +ENVELOPE_TEXT_FORMAT = {"format": {"type": "json_schema", "name": "envelope", "strict": True, "schema": ENVELOPE_SCHEMA}} +DEFAULT_RESPONSES_TOOL_ROUNDS = 4 # Consolidation output (SPEC-002 MEM-02/03): new self-authored facts + one episode summary CONSOLIDATION_SCHEMA = { @@ -125,6 +128,10 @@ async def openai_chat(client, *args, **kwargs): return await client.chat.completions.create(*args, **kwargs) +async def openai_responses(client, *args, **kwargs): + return await client.responses.create(*args, **kwargs) + + async def openai_image(client, *args, **kwargs): return await client.images.generate(*args, **kwargs) @@ -269,9 +276,103 @@ class OpenAIResponder(AIResponder, LeonardoAIDrawMixIn): usage = getattr(result, "usage", None) prompt_tokens = getattr(usage, "prompt_tokens", None) completion_tokens = getattr(usage, "completion_tokens", None) + if not isinstance(prompt_tokens, int): # Responses API names them input/output (ENV-22) + prompt_tokens = getattr(usage, "input_tokens", None) + if not isinstance(completion_tokens, int): + completion_tokens = getattr(usage, "output_tokens", None) if isinstance(prompt_tokens, int) and isinstance(completion_tokens, int): self.ledger.add_tokens(prompt_tokens, completion_tokens) + @staticmethod + def _responses_input(messages: List[Dict[str, Any]]) -> List[Dict[str, Any]]: + """Chat-format history -> Responses input items; vision parts become input_image (ENV-22).""" + items: List[Dict[str, Any]] = [] + for msg in messages: + role = msg.get("role") + if role == "tool": + continue + content = msg.get("content") + if isinstance(content, list): + parts: List[Dict[str, Any]] = [] + for part in content: + if part.get("type") == "text": + parts.append({"type": "input_text", "text": part.get("text", "")}) + elif part.get("type") == "image_url": + parts.append({"type": "input_image", "image_url": part.get("image_url", {}).get("url", "")}) + items.append({"role": role, "content": parts}) + else: + items.append({"role": role, "content": str(content)}) + return items + + @staticmethod + def _responses_refused(result: Any) -> bool: + for item in getattr(result, "output", []) or []: + if getattr(item, "type", None) == "message": + for part in getattr(item, "content", []) or []: + if getattr(part, "type", None) == "refusal": + return True + return False + + async def _chat_via_responses(self, messages: List[Dict[str, Any]], limit: int, model: str) -> Tuple[Optional[Dict[str, Any]], int]: + """Responder call via /v1/responses: tools + reasoning allowed, stateless with encrypted reasoning (ENV-22/23).""" + context: List[Any] = self._responses_input(messages) + kwargs: Dict[str, Any] = { + "model": model, + "input": context, + "text": ENVELOPE_TEXT_FORMAT, + "store": False, # nothing retained server-side (ENV-23) + "include": ["reasoning.encrypted_content"], + "reasoning": {"effort": str(self.config.get("reasoning-effort", "none"))}, + } + author = self._last_author(messages) + if author: + # hashed, never the raw Discord name (SAF-10) + kwargs["safety_identifier"] = "discord-" + hashlib.sha256(author.encode()).hexdigest()[:16] + available_tools = self._available_tools() + if available_tools: + kwargs["tools"] = [{"type": "function", **func} for func in available_tools] + kwargs["tool_choice"] = "auto" + logging.info(f"🔧 Tools available to AI: {[func['name'] for func in available_tools]}") + + rounds = int(self.config.get("responses-tool-rounds", DEFAULT_RESPONSES_TOOL_ROUNDS)) + for _ in range(max(1, rounds) + 1): + result = await openai_responses(self.client, **kwargs) + self._record_usage(result) + if self._responses_refused(result): + logging.warning("model refused (responses path)") # ENV-24 + return None, limit + calls = [item for item in (getattr(result, "output", []) or []) if getattr(item, "type", None) == "function_call"] + if not calls or "tools" not in kwargs: + answer = {"content": getattr(result, "output_text", None) or "", "role": "assistant"} + self.rate_limit_backoff = exponential_backoff() + self._use_retry_model = False + logging.info(f"generated response {getattr(result, 'usage', None)}: {repr(answer)}") + return answer, limit + tool_names = [call.name for call in calls] + logging.info(f"🔧 OpenAI requested function calls: {tool_names}") + # Pass ALL output items back — reasoning items keep the chain of thought (ENV-23) + context = context + [item if isinstance(item, dict) else item.model_dump() for item in result.output] + for call in calls: + function_args = json.loads(call.arguments) if call.arguments else {} + logging.info(f"🔧 Executing tool: {call.name} with args: {function_args}") + function_result = await self._dispatch_tool(call.name, function_args, author or "") + logging.info(f"🔧 Tool result: {type(function_result)} - {str(function_result)[:200]}...") + context.append( + { + "type": "function_call_output", + "call_id": call.call_id, + # tool text is external input — sanitize before prompting (SAF-03) + "output": sanitize_external_text(json.dumps(function_result), 8000) if function_result else "No results found", + } + ) + kwargs["input"] = context + rounds -= 1 + if rounds <= 0: + # loop exhausted: force a tool-less final answer (ENV-23) + kwargs.pop("tools", None) + kwargs.pop("tool_choice", None) + return None, limit + async def chat(self, messages: List[Dict[str, Any]], limit: int) -> Tuple[Optional[Dict[str, Any]], int]: # Safety check for mock objects in tests if not isinstance(messages, list) or len(messages) == 0: @@ -310,6 +411,9 @@ class OpenAIResponder(AIResponder, LeonardoAIDrawMixIn): logging.warning(f"Error accessing message content: {e}") return None, limit try: + if bool(self.config.get("use-responses-api", False)): + return await self._chat_via_responses(messages, limit, model) # ENV-22 + # Prepare function calls if IGDB is enabled chat_kwargs = { "model": model, diff --git a/specs/SPEC-001-responder-envelope.md b/specs/SPEC-001-responder-envelope.md index f6748b2..e3a134f 100644 --- a/specs/SPEC-001-responder-envelope.md +++ b/specs/SPEC-001-responder-envelope.md @@ -152,3 +152,32 @@ Every chat call carries `response_format` = strict JSON schema named IMG-02), `picture_edit`, `hack` — all required, `additionalProperties: false`, nullable where the protocol allows null. Tool-followup calls carry the same format. + +### ENV-22 — Responses API path behind a flag (coverage: test) + +With `use-responses-api = true`, responder chat calls go to +`/v1/responses` instead of chat/completions: same model selection +(default / vision / factual / retry), the same strict envelope schema +(as `text.format`), tools in the flat Responses shape, and +`reasoning` = config `reasoning-effort` — tools + reasoning are +allowed here (the chat/completions 400 from ENV-21 does not apply). +Flag off (default) = the ENV-21 path, byte-identical behavior. +Classifier, consolidation and task-proposal calls stay on +chat/completions. + +### ENV-23 — Responses tool loop is stateless and keeps reasoning (coverage: test) + +The Responses path runs with `store=false` and +`include=["reasoning.encrypted_content"]` (nothing retained +server-side). On a function call, ALL output items — including +reasoning items — are passed back as input together with one +`function_call_output` per call (matched by `call_id`, result +sanitized per SAF-03), so the model continues one chain of thought +across tool rounds. Up to `responses-tool-rounds` (default 4) rounds +may call tools; an exhausted loop forces a final tool-less answer. + +### ENV-24 — Responses refusals are failed attempts (coverage: test) + +A refusal content part in the Responses output yields no answer +(backoff + retry per ENV-12/ENV-18), exactly like the +chat/completions path. diff --git a/tests/test_spec_responses.py b/tests/test_spec_responses.py new file mode 100644 index 0000000..05de954 --- /dev/null +++ b/tests/test_spec_responses.py @@ -0,0 +1,156 @@ +"""Unit coverage for the Responses API path (ENV-22..24, D-021).""" + +import json +import unittest +from unittest.mock import AsyncMock, Mock, patch + +from fjerkroa_bot.openai_responder import ENVELOPE_TEXT_FORMAT, OpenAIResponder + +from .test_bdd_envelope import envelope + +CONFIG = { + "openai-token": "t", + "model": "main-model", + "system": "s", + "history-limit": 5, + "use-responses-api": True, + "reasoning-effort": "medium", +} + + +def _msg_item(): + part = Mock() + part.type = "output_text" + item = Mock() + item.type = "message" + item.content = [part] + item.model_dump = lambda: {"type": "message"} + return item + + +def _refusal_item(): + part = Mock() + part.type = "refusal" + item = Mock() + item.type = "message" + item.content = [part] + return item + + +def _reasoning_item(): + item = Mock() + item.type = "reasoning" + item.model_dump = lambda: {"type": "reasoning", "encrypted_content": "opaque-cot"} + return item + + +def _call_item(name, args, call_id="call-1"): + item = Mock() + item.type = "function_call" + item.name = name + item.arguments = json.dumps(args) + item.call_id = call_id + item.model_dump = lambda: {"type": "function_call", "name": name, "arguments": json.dumps(args), "call_id": call_id} + return item + + +def _response(output, text=""): + result = Mock() + result.output = output + result.output_text = text + result.usage = Mock(prompt_tokens=None, completion_tokens=None, input_tokens=5, output_tokens=7) + return result + + +class TestResponsesPath(unittest.IsolatedAsyncioTestCase): + def _responder(self, **extra): + return OpenAIResponder(dict(CONFIG, **extra), "chat") + + async def test_flag_routes_to_responses_with_reasoning(self): + """ENV-22: flag on -> /v1/responses with envelope text.format, reasoning from config, stateless kwargs.""" + responder = self._responder() + with patch("fjerkroa_bot.openai_responder.openai_responses", new_callable=AsyncMock) as responses_mock: + with patch("fjerkroa_bot.openai_responder.openai_chat", new_callable=AsyncMock) as chat_mock: + responses_mock.return_value = _response([_msg_item()], envelope(answer="hi", answer_needed=True)) + answer, _ = await responder.chat([{"role": "user", "content": "hei"}], 10) + chat_mock.assert_not_awaited() + self.assertEqual(json.loads(answer["content"])["answer"], "hi") + kwargs = responses_mock.await_args.kwargs + self.assertEqual(kwargs["text"], ENVELOPE_TEXT_FORMAT) + self.assertEqual(kwargs["reasoning"], {"effort": "medium"}) + self.assertFalse(kwargs["store"]) # ENV-23 + self.assertIn("reasoning.encrypted_content", kwargs["include"]) + + async def test_flag_off_stays_on_chat_completions(self): + """ENV-22: flag off (default) -> openai_responses never called.""" + from .test_spec_structured import ok_result + + responder = OpenAIResponder({k: v for k, v in CONFIG.items() if k != "use-responses-api"}, "chat") + with patch("fjerkroa_bot.openai_responder.openai_responses", new_callable=AsyncMock) as responses_mock: + with patch("fjerkroa_bot.openai_responder.openai_chat", new_callable=AsyncMock) as chat_mock: + chat_mock.return_value = ok_result() + await responder.chat([{"role": "user", "content": "hei"}], 10) + responses_mock.assert_not_awaited() + chat_mock.assert_awaited() + + async def test_tools_flat_shape(self): + """ENV-22: tools are sent in the flat Responses shape (name at top level).""" + responder = self._responder(**{"enable-news-tool": True}) + responder.store = Mock() # store present -> get_news offered + with patch("fjerkroa_bot.openai_responder.openai_responses", new_callable=AsyncMock) as responses_mock: + responses_mock.return_value = _response([_msg_item()], envelope(answer="x", answer_needed=True)) + await responder.chat([{"role": "user", "content": "hei"}], 10) + tools = responses_mock.await_args.kwargs["tools"] + self.assertTrue(all(tool["type"] == "function" and "name" in tool and "function" not in tool for tool in tools)) + + async def test_tool_loop_passes_reasoning_and_outputs_back(self): + """ENV-23: function_call -> dispatch; next call carries reasoning item + function_call_output.""" + responder = self._responder(**{"enable-news-tool": True}) + responder.store = Mock() + responder._dispatch_tool = AsyncMock(return_value={"results": ["ok"]}) + first = _response([_reasoning_item(), _call_item("get_news", {"topic": "x"}, "call-9")]) + second = _response([_msg_item()], envelope(answer="done", answer_needed=True)) + with patch("fjerkroa_bot.openai_responder.openai_responses", new_callable=AsyncMock) as responses_mock: + responses_mock.side_effect = [first, second] + answer, _ = await responder.chat([{"role": "user", "content": "news?"}], 10) + self.assertEqual(json.loads(answer["content"])["answer"], "done") + responder._dispatch_tool.assert_awaited_once() + followup_input = responses_mock.await_args_list[1].kwargs["input"] + self.assertIn({"type": "reasoning", "encrypted_content": "opaque-cot"}, followup_input) + outputs = [item for item in followup_input if isinstance(item, dict) and item.get("type") == "function_call_output"] + self.assertEqual(len(outputs), 1) + self.assertEqual(outputs[0]["call_id"], "call-9") + + async def test_exhausted_rounds_force_toolless_answer(self): + """ENV-23: after responses-tool-rounds rounds the final call drops tools.""" + responder = self._responder(**{"enable-news-tool": True, "responses-tool-rounds": 1}) + responder.store = Mock() + responder._dispatch_tool = AsyncMock(return_value={"results": []}) + looping = _response([_call_item("get_news", {}, "c")]) + final = _response([_msg_item()], envelope(answer="forced", answer_needed=True)) + with patch("fjerkroa_bot.openai_responder.openai_responses", new_callable=AsyncMock) as responses_mock: + responses_mock.side_effect = [looping, final] + answer, _ = await responder.chat([{"role": "user", "content": "go"}], 10) + self.assertEqual(json.loads(answer["content"])["answer"], "forced") + self.assertNotIn("tools", responses_mock.await_args_list[1].kwargs) + + async def test_refusal_is_failed_attempt(self): + """ENV-24: a refusal part -> no answer.""" + responder = self._responder() + with patch("fjerkroa_bot.openai_responder.openai_responses", new_callable=AsyncMock) as responses_mock: + responses_mock.return_value = _response([_refusal_item()]) + answer, _ = await responder.chat([{"role": "user", "content": "hei"}], 10) + self.assertIsNone(answer) + + async def test_vision_parts_mapped(self): + """ENV-22: chat-format image parts become input_image items.""" + items = OpenAIResponder._responses_input( + [ + {"role": "user", "content": [{"type": "text", "text": "look"}, {"type": "image_url", "image_url": {"url": "data:x"}}]}, + {"role": "tool", "content": "dropped"}, + {"role": "assistant", "content": "{}"}, + ] + ) + self.assertEqual(items[0]["content"][0], {"type": "input_text", "text": "look"}) + self.assertEqual(items[0]["content"][1], {"type": "input_image", "image_url": "data:x"}) + self.assertEqual(len(items), 2) # tool row dropped