Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
13 changes: 7 additions & 6 deletions docs/tools.md
Original file line number Diff line number Diff line change
Expand Up @@ -255,12 +255,13 @@ the same request against an open model runs on the sandbox, with the same
provider with no Messages API of its own it is dropped rather than refused,
because a beta names an Anthropic feature that provider was never going to
serve, and refusing it would make the request fail purely because its model
changed. On
Responses a claimed `code_interpreter` is answered with a `code_interpreter_call`
item. Chat Completions has no native shape, so a claimed declaration there
resolves inside the tool loop and only the final message is returned. Nothing
runs natively on Chat Completions, and the bare `code_execution` form is no
provider's, so under `auto` both always run on the sandbox.
changed. On Responses a claimed `code_interpreter` is answered with a
`code_interpreter_call` item per run, in the order the calls ran among the
gateway's other native items (a `web_search_call`, say). Chat Completions has no
native shape, so a claimed declaration there resolves inside the tool loop and
only the final message is returned. Nothing runs natively on Chat Completions,
and the bare `code_execution` form is no provider's, so under `auto` both always
run on the sandbox.

The executor also decides where an attached file goes. Code running on Otari's
sandbox is given the file from Otari's own store, and code running in the
Expand Down
24 changes: 3 additions & 21 deletions src/gateway/api/routes/_pipeline.py
Original file line number Diff line number Diff line change
Expand Up @@ -249,7 +249,6 @@
claim_web_declarations,
declares_code_execution,
extract_web_tools,
native_code_execution_dialect,
native_rendering,
read_web_search_max_uses,
web_search_intercept_enabled,
Expand Down Expand Up @@ -2471,30 +2470,13 @@ def native_tools(self, dialect: Dialect) -> frozenset[str]:

A caller who declared a tool in a provider's words is owed that provider's
items back, and each tool's registry entry decides whether its declaration
asks for them. Code execution answers from :attr:`native_code_execution_dialect`
instead, because the dialect loops still build its blocks themselves.
asks for them.
"""
names = {
return frozenset(
name
for name, entry in self.declared_gateway_tools.items()
if (rendering := native_rendering(name, dialect)) is not None and rendering.declared(entry)
}
if self.use_sandbox and self.native_code_execution_dialect == dialect:
names.add(CODE_EXECUTION_TOOL_NAME)
return frozenset(names)

@property
def native_code_execution_dialect(self) -> Dialect | None:
"""The wire format whose native code-execution blocks this request expects.

Set only when the gateway runs a declaration made in a provider's own
vocabulary: the caller asked in Anthropic's or OpenAI's words and its SDK
will look for that provider's result shape, so the loop answers in it.
``None`` for ``otari_code_execution``, whose callers get the plain result.
"""
if not self.use_sandbox:
return None
return native_code_execution_dialect(self.sandbox_tool_entry)
)

@property
def max_web_search_uses(self) -> int | None:
Expand Down
9 changes: 7 additions & 2 deletions src/gateway/api/routes/responses.py
Original file line number Diff line number Diff line change
Expand Up @@ -61,13 +61,18 @@
from gateway.services.log_writer import LogWriter
from gateway.services.mcp_loop import ToolBackend
from gateway.services.mcp_loop_responses import (
CODE_INTERPRETER_CALL_ID_PREFIX,
MAX_TOOL_ITERATIONS_CAP,
responses_tool_loop,
responses_tool_loop_stream,
)
from gateway.services.tool_format import inject_purpose_hints_responses, openai_to_responses_tools
from gateway.services.tools import CODE_EXECUTION_HEADER, WEB_SEARCH_HEADER, Dialect, ToolUseBudget
from gateway.services.tools import (
CODE_EXECUTION_HEADER,
CODE_INTERPRETER_CALL_ID_PREFIX,
WEB_SEARCH_HEADER,
Dialect,
ToolUseBudget,
)
from gateway.streaming import RESPONSES_STREAM_FORMAT, StreamFormat
from gateway.types.attempt import Attempt
from gateway.types.normalization_target import NormalizationTarget
Expand Down
63 changes: 3 additions & 60 deletions src/gateway/services/mcp_loop_messages.py
Original file line number Diff line number Diff line change
Expand Up @@ -18,15 +18,8 @@
import uuid
from collections.abc import AsyncGenerator, AsyncIterator, Callable
from contextlib import aclosing
from typing import TYPE_CHECKING, Any, Literal, Protocol, TypedDict, cast, runtime_checkable

from anthropic.types import (
CodeExecutionOutputBlock,
CodeExecutionResultBlock,
CodeExecutionToolResultBlock,
CodeExecutionToolResultError,
ServerToolUseBlock,
)
from typing import TYPE_CHECKING, Any, Protocol, TypedDict, runtime_checkable

from anthropic.types.beta import BetaMCPToolResultBlock, BetaMCPToolUseBlock
from anthropic.types.beta.beta_container import BetaContainer
from any_llm import amessages
Expand All @@ -45,12 +38,10 @@
MaxToolIterationsExceeded,
ToolBackend,
)
from gateway.services.sandbox_backend import CODE_EXECUTION_TOOL_NAME, CodeExecution
from gateway.services.tool_format import openai_to_anthropic_tools
from gateway.services.tool_usage import is_tool_error
from gateway.services.tools import (
MAX_USES_EXCEEDED_ERROR,
SERVER_TOOL_USE_ID_PREFIX,
Dialect,
NativeCall,
ToolUseBudget,
Expand Down Expand Up @@ -126,59 +117,11 @@ def _max_uses_exceeded_result(call: NativeCall, native: _NativeSink | None) -> d
return {"type": "tool_result", "tool_use_id": call.id, "content": MAX_USES_EXCEEDED_ERROR}


def _native_code_execution_blocks(execution: CodeExecution) -> list[Any]:
"""A ``server_tool_use`` / ``code_execution_tool_result`` pair for one gateway execution.

Emitted for a caller that declared code execution in Anthropic's own
vocabulary and whose request the gateway's sandbox ran instead. The result
block is the contract's own shape, which mirrors Anthropic's, so a client
parsing Anthropic responses reads it with no translation. A call the backend
never answered is reported in the vocabulary's error shape rather than
dropped, because the model was told about the failure and the client should
see the same story.
"""
tool_use_id = f"{SERVER_TOOL_USE_ID_PREFIX}{uuid.uuid4().hex}"
content: CodeExecutionResultBlock | CodeExecutionToolResultError
if execution.result is None:
content = CodeExecutionToolResultError(type="code_execution_tool_result_error", error_code="unavailable")
else:
result = execution.result.content
content = CodeExecutionResultBlock(
type="code_execution_result",
stdout=result.stdout,
stderr=result.stderr,
return_code=result.return_code if result.return_code is not None else 0,
# The ids are the ones ``/v1/files`` serves, not the sandbox's own: a
# produced file that was not stored has no id the caller could use.
content=[
CodeExecutionOutputBlock(type="code_execution_output", file_id=file_id)
for file_id in execution.file_ids.values()
],
)
return [
ServerToolUseBlock(
id=tool_use_id,
name=cast('Literal["code_execution"]', CODE_EXECUTION_TOOL_NAME),
input={"code": execution.code},
type="server_tool_use",
),
CodeExecutionToolResultBlock(tool_use_id=tool_use_id, type="code_execution_tool_result", content=content),
]


def _native_blocks_for_call(pool: ToolBackend, call: NativeCall) -> list[Any]:
"""Native blocks describing one gateway tool call, if its tool has any in this dialect.

A code execution contributes blocks whether or not the program succeeded, because
a non-zero exit is a result the vocabulary can carry and a call the backend never
ran has an error shape of its own. An MCP call has no Anthropic block that would
be honest to emit and stays invisible.
An MCP call has no Anthropic block that would be honest to emit and stays invisible.
"""
if call.name == CODE_EXECUTION_TOOL_NAME:
take_executions = getattr(pool, "take_executions", None)
if take_executions is None:
return []
return [block for execution in take_executions() for block in _native_code_execution_blocks(execution)]
rendering = native_rendering(call.name, Dialect.MESSAGES)
return rendering.ran(call, pool) if rendering is not None else []

Expand Down
Loading
Loading