responses api behind use-responses-api flag (env-22..24, d-021) — tools + reasoning combinable, stateless with encrypted reasoning, multi-round tool loop
This commit is contained in:
@@ -40,6 +40,9 @@ ENVELOPE_SCHEMA = {
|
||||
"additionalProperties": False,
|
||||
}
|
||||
ENVELOPE_RESPONSE_FORMAT = {"type": "json_schema", "json_schema": {"name": "envelope", "strict": True, "schema": ENVELOPE_SCHEMA}}
|
||||
# Same schema in the Responses API shape (ENV-22): text.format is flat, not nested under json_schema
|
||||
ENVELOPE_TEXT_FORMAT = {"format": {"type": "json_schema", "name": "envelope", "strict": True, "schema": ENVELOPE_SCHEMA}}
|
||||
DEFAULT_RESPONSES_TOOL_ROUNDS = 4
|
||||
|
||||
# Consolidation output (SPEC-002 MEM-02/03): new self-authored facts + one episode summary
|
||||
CONSOLIDATION_SCHEMA = {
|
||||
@@ -125,6 +128,10 @@ async def openai_chat(client, *args, **kwargs):
|
||||
return await client.chat.completions.create(*args, **kwargs)
|
||||
|
||||
|
||||
async def openai_responses(client, *args, **kwargs):
|
||||
return await client.responses.create(*args, **kwargs)
|
||||
|
||||
|
||||
async def openai_image(client, *args, **kwargs):
|
||||
return await client.images.generate(*args, **kwargs)
|
||||
|
||||
@@ -269,9 +276,103 @@ class OpenAIResponder(AIResponder, LeonardoAIDrawMixIn):
|
||||
usage = getattr(result, "usage", None)
|
||||
prompt_tokens = getattr(usage, "prompt_tokens", None)
|
||||
completion_tokens = getattr(usage, "completion_tokens", None)
|
||||
if not isinstance(prompt_tokens, int): # Responses API names them input/output (ENV-22)
|
||||
prompt_tokens = getattr(usage, "input_tokens", None)
|
||||
if not isinstance(completion_tokens, int):
|
||||
completion_tokens = getattr(usage, "output_tokens", None)
|
||||
if isinstance(prompt_tokens, int) and isinstance(completion_tokens, int):
|
||||
self.ledger.add_tokens(prompt_tokens, completion_tokens)
|
||||
|
||||
@staticmethod
|
||||
def _responses_input(messages: List[Dict[str, Any]]) -> List[Dict[str, Any]]:
|
||||
"""Chat-format history -> Responses input items; vision parts become input_image (ENV-22)."""
|
||||
items: List[Dict[str, Any]] = []
|
||||
for msg in messages:
|
||||
role = msg.get("role")
|
||||
if role == "tool":
|
||||
continue
|
||||
content = msg.get("content")
|
||||
if isinstance(content, list):
|
||||
parts: List[Dict[str, Any]] = []
|
||||
for part in content:
|
||||
if part.get("type") == "text":
|
||||
parts.append({"type": "input_text", "text": part.get("text", "")})
|
||||
elif part.get("type") == "image_url":
|
||||
parts.append({"type": "input_image", "image_url": part.get("image_url", {}).get("url", "")})
|
||||
items.append({"role": role, "content": parts})
|
||||
else:
|
||||
items.append({"role": role, "content": str(content)})
|
||||
return items
|
||||
|
||||
@staticmethod
|
||||
def _responses_refused(result: Any) -> bool:
|
||||
for item in getattr(result, "output", []) or []:
|
||||
if getattr(item, "type", None) == "message":
|
||||
for part in getattr(item, "content", []) or []:
|
||||
if getattr(part, "type", None) == "refusal":
|
||||
return True
|
||||
return False
|
||||
|
||||
async def _chat_via_responses(self, messages: List[Dict[str, Any]], limit: int, model: str) -> Tuple[Optional[Dict[str, Any]], int]:
|
||||
"""Responder call via /v1/responses: tools + reasoning allowed, stateless with encrypted reasoning (ENV-22/23)."""
|
||||
context: List[Any] = self._responses_input(messages)
|
||||
kwargs: Dict[str, Any] = {
|
||||
"model": model,
|
||||
"input": context,
|
||||
"text": ENVELOPE_TEXT_FORMAT,
|
||||
"store": False, # nothing retained server-side (ENV-23)
|
||||
"include": ["reasoning.encrypted_content"],
|
||||
"reasoning": {"effort": str(self.config.get("reasoning-effort", "none"))},
|
||||
}
|
||||
author = self._last_author(messages)
|
||||
if author:
|
||||
# hashed, never the raw Discord name (SAF-10)
|
||||
kwargs["safety_identifier"] = "discord-" + hashlib.sha256(author.encode()).hexdigest()[:16]
|
||||
available_tools = self._available_tools()
|
||||
if available_tools:
|
||||
kwargs["tools"] = [{"type": "function", **func} for func in available_tools]
|
||||
kwargs["tool_choice"] = "auto"
|
||||
logging.info(f"🔧 Tools available to AI: {[func['name'] for func in available_tools]}")
|
||||
|
||||
rounds = int(self.config.get("responses-tool-rounds", DEFAULT_RESPONSES_TOOL_ROUNDS))
|
||||
for _ in range(max(1, rounds) + 1):
|
||||
result = await openai_responses(self.client, **kwargs)
|
||||
self._record_usage(result)
|
||||
if self._responses_refused(result):
|
||||
logging.warning("model refused (responses path)") # ENV-24
|
||||
return None, limit
|
||||
calls = [item for item in (getattr(result, "output", []) or []) if getattr(item, "type", None) == "function_call"]
|
||||
if not calls or "tools" not in kwargs:
|
||||
answer = {"content": getattr(result, "output_text", None) or "", "role": "assistant"}
|
||||
self.rate_limit_backoff = exponential_backoff()
|
||||
self._use_retry_model = False
|
||||
logging.info(f"generated response {getattr(result, 'usage', None)}: {repr(answer)}")
|
||||
return answer, limit
|
||||
tool_names = [call.name for call in calls]
|
||||
logging.info(f"🔧 OpenAI requested function calls: {tool_names}")
|
||||
# Pass ALL output items back — reasoning items keep the chain of thought (ENV-23)
|
||||
context = context + [item if isinstance(item, dict) else item.model_dump() for item in result.output]
|
||||
for call in calls:
|
||||
function_args = json.loads(call.arguments) if call.arguments else {}
|
||||
logging.info(f"🔧 Executing tool: {call.name} with args: {function_args}")
|
||||
function_result = await self._dispatch_tool(call.name, function_args, author or "")
|
||||
logging.info(f"🔧 Tool result: {type(function_result)} - {str(function_result)[:200]}...")
|
||||
context.append(
|
||||
{
|
||||
"type": "function_call_output",
|
||||
"call_id": call.call_id,
|
||||
# tool text is external input — sanitize before prompting (SAF-03)
|
||||
"output": sanitize_external_text(json.dumps(function_result), 8000) if function_result else "No results found",
|
||||
}
|
||||
)
|
||||
kwargs["input"] = context
|
||||
rounds -= 1
|
||||
if rounds <= 0:
|
||||
# loop exhausted: force a tool-less final answer (ENV-23)
|
||||
kwargs.pop("tools", None)
|
||||
kwargs.pop("tool_choice", None)
|
||||
return None, limit
|
||||
|
||||
async def chat(self, messages: List[Dict[str, Any]], limit: int) -> Tuple[Optional[Dict[str, Any]], int]:
|
||||
# Safety check for mock objects in tests
|
||||
if not isinstance(messages, list) or len(messages) == 0:
|
||||
@@ -310,6 +411,9 @@ class OpenAIResponder(AIResponder, LeonardoAIDrawMixIn):
|
||||
logging.warning(f"Error accessing message content: {e}")
|
||||
return None, limit
|
||||
try:
|
||||
if bool(self.config.get("use-responses-api", False)):
|
||||
return await self._chat_via_responses(messages, limit, model) # ENV-22
|
||||
|
||||
# Prepare function calls if IGDB is enabled
|
||||
chat_kwargs = {
|
||||
"model": model,
|
||||
|
||||
Reference in New Issue
Block a user