diff --git a/src/marsys/models/adapters/google.py b/src/marsys/models/adapters/google.py index 7619f869..ef0cf717 100644 --- a/src/marsys/models/adapters/google.py +++ b/src/marsys/models/adapters/google.py @@ -244,13 +244,18 @@ def format_request_payload(self, messages: List[Dict], **kwargs) -> Dict[str, An # Add native function calling support if kwargs.get("tools"): - if any(isinstance(t, dict) and t.get("defer_loading") for t in kwargs["tools"]): + if any( + isinstance(t, dict) and (t.get("defer_loading") or t.get("namespace")) + for t in kwargs["tools"] + ): # Google has no deferred-tool-loading feature; the rebuild below already drops the - # per-tool defer_loading flag, so warn (not a silent drop) and fall back to eager. + # per-tool defer_loading flag and the namespace label that rides beside it, so + # warn (not a silent drop) and fall back to eager loading with every tool flat. import warnings warnings.warn( "Google models do not support deferred tool loading; the per-tool " - "defer_loading flag is ignored and all tools are loaded eagerly." + "defer_loading flag and its namespace label are ignored and all tools are " + "loaded eagerly." ) # Convert OpenAI format tools to Google format google_tools = [] diff --git a/src/marsys/models/adapters/openai.py b/src/marsys/models/adapters/openai.py index 12b7cd16..1ae5fa59 100644 --- a/src/marsys/models/adapters/openai.py +++ b/src/marsys/models/adapters/openai.py @@ -3,7 +3,7 @@ import re import time import warnings -from typing import Any, Callable, Collection, Dict, List, Optional +from typing import Any, Callable, Collection, Dict, List, Optional, Tuple from marsys.models.adapters.base import ( CACHE_EXEMPT_KEY, @@ -102,6 +102,154 @@ _GENERATION_RE = re.compile(r"^gpt-(\d+)(?:\.(\d+))?") +def _generation(model_lower: str) -> Optional[Tuple[int, int]]: + """``(major, minor)`` read off a model name, or None when the name is not GPT-shaped. + + One reading for every feature gated on a generation, so the gates differ in their FLOOR and + in nothing else — which is the only thing that should ever separate two of them.""" + match = _GENERATION_RE.match(model_lower or "") + if not match: + return None + return int(match.group(1)), int(match.group(2) or 0) + + +# --- hosted tool search ------------------------------------------------------------- +# +# The provider runs the search itself: the request carries deferred functions, grouped in +# namespace containers or standing alone, plus the `tool_search` built-in; the model searches, +# the provider injects the loaded definitions at the end of the context, and the model calls in +# the same reply. + +# The generation that serves it. Its OWN floor, two generations below the explicit prompt-cache +# one beside it, because the two features shipped apart: one gate for both would either withhold +# the search from models that serve it or send cache fields to models that answer them with a 400. +_HOSTED_TOOL_SEARCH_MIN_VERSION = (5, 4) + +# The providers whose endpoints were measured to serve it. The factory routes an unrecognized +# provider to this adapter, so a third-party OpenAI-compatible endpoint behind a GPT-5-shaped +# model name would otherwise be told it serves a feature nobody asked it about. +_HOSTED_TOOL_SEARCH_PROVIDERS = frozenset({"openai", "azure"}) + +# The reply items a hosted search emits: the call, and its output carrying the definitions the +# search loaded. Kept off the reply and replayed on the next request verbatim. +_TOOL_SEARCH_ITEM_TYPES = frozenset({"tool_search_call", "tool_search_output"}) + +# The key a caller puts on a Chat-Completions tool dict to say which container the rendered +# function belongs in: `{"name": ..., "description": ...}`. The grouping travels per tool rather +# than as a container the caller builds, so the legs that must not see it strip one key off a flat +# dict — the move they already make for `defer_loading` — instead of unwrapping a tool type they +# have no shape for. +NAMESPACE_LABEL_KEY = "namespace" + + +def supports_hosted_tool_search(provider: Optional[str], model_lower: str) -> bool: + """Whether this leg serves the provider-run tool search over deferred functions. + + Carries the same honesty caveat as the prompt-cache gate above, and it bites harder here: on + Azure the model field is an operator-chosen deployment label rather than a model name, so a + deployment named after a model it does not serve lies to this check — and the failure is a 400 + on `tools.defer_loading` for every call rather than a price. Ask the deployment before turning + the feature on for it. + """ + if provider not in _HOSTED_TOOL_SEARCH_PROVIDERS: + return False + generation = _generation(model_lower) + return generation is not None and generation >= _HOSTED_TOOL_SEARCH_MIN_VERSION + + +def _namespace_label(tool: Any) -> Optional[Dict[str, Any]]: + """The namespace label on a tool dict, or None. A label with no name names no container, so + it is no label at all.""" + if not isinstance(tool, dict): + return None + label = tool.get(NAMESPACE_LABEL_KEY) + return label if isinstance(label, dict) and label.get("name") else None + + +def _convert_tools_for_responses(tools: Collection[Any]) -> List[Any]: + """The Responses `tools` array for a caller's tool list. + + Three jobs in one pass, because they read the same dicts. A Chat-Completions `function` tool + flattens to the internally-tagged Responses shape. A tool carrying a namespace label goes + inside a `namespace` container instead of standing at the top level — one container per + distinct label name, placed where its first member appeared, members in the order given, and + the container never carries the deferral flag its members do. And a request that defers + anything gains the hosted-search built-in: the provider refuses a deferred tool without it + ("Deferred tools require tools.tool_search") whether the flag sat at the top level or inside a + container, so reading only the top level makes a namespace request a refusal rather than a + degradation. + + The caller's dicts are never mutated — a converted function is a new dict and a container is + built here — so a tool array a caller holds frozen across a turn stays what it was. + """ + converted: List[Any] = [] + containers: Dict[str, Dict[str, Any]] = {} + any_deferred = False + for tool in tools: + if not isinstance(tool, dict): + converted.append(tool) + continue + if tool.get("type") == "function" and "function" in tool: + # Convert from Chat Completions format (externally tagged) + func = tool["function"] + rendered: Any = { + "type": "function", + "name": func.get("name"), + "description": func.get("description"), + "parameters": func.get("parameters"), + # Note: strict is true by default in Responses API + } + if tool.get("defer_loading"): + rendered["defer_loading"] = True + any_deferred = True + else: + # Already in Responses API format or other tool type + rendered = ( + {k: v for k, v in tool.items() if k != NAMESPACE_LABEL_KEY} + if NAMESPACE_LABEL_KEY in tool + else tool + ) + if tool.get("defer_loading"): + any_deferred = True + label = _namespace_label(tool) + if label is None: + converted.append(rendered) + continue + container = containers.get(label["name"]) + if container is None: + container = { + "type": "namespace", + "name": label["name"], + "description": label.get("description", ""), + "tools": [], + } + containers[label["name"]] = container + converted.append(container) + container["tools"].append(rendered) + if any_deferred and not any( + isinstance(t, dict) and t.get("type") == "tool_search" for t in converted + ): + # Auto-add the Responses tool-search built-in so deferred tools are discoverable + # (gpt-5.4+). Suppressed if the caller supplied their own. + converted.append({"type": "tool_search"}) + return converted + + +def _replayed_provider_items(msg: Dict[str, Any]) -> List[Dict[str, Any]]: + """This assistant row's provider items that belong back in the Responses input array. + + Type-discriminated, the way every leg reads this channel: the field is shared with the other + providers' opaque blocks (extended-thinking blocks, thought signatures) and each payload + builder re-emits only the types its own endpoint emitted. Anything else on it is another + leg's and is left where it is. + """ + return [ + item + for item in (msg.get("reasoning_details") or []) + if isinstance(item, dict) and item.get("type") in _TOOL_SEARCH_ITEM_TYPES + ] + + def supports_explicit_prompt_cache(model_lower: str) -> bool: """Whether a model name is GPT-5.6 or later, the generation that serves the fields. @@ -113,12 +261,8 @@ def supports_explicit_prompt_cache(model_lower: str) -> bool: mistake for a cost problem. A 5.6 model behind an older-shaped name simply keeps today's behaviour and pays today's price. """ - match = _GENERATION_RE.match(model_lower or "") - if not match: - return False - major = int(match.group(1)) - minor = int(match.group(2) or 0) - return (major, minor) >= _EXPLICIT_PROMPT_CACHE_MIN_VERSION + generation = _generation(model_lower) + return generation is not None and generation >= _EXPLICIT_PROMPT_CACHE_MIN_VERSION def _blocks_with_breakpoint( @@ -369,6 +513,16 @@ def format_request_payload(self, messages: List[Dict], **kwargs) -> Dict[str, An role = msg.get("role") item_start = len(converted_messages) + # The provider's own items from this reply go back first: a hosted search's call and + # its output, the output carrying the definitions the search loaded. Ahead of the + # row's own message and calls because that is the order the provider emitted them in, + # and because the definitions have to reach the model's context before the call that + # uses them. Dropping them instead costs the model a second search of the same family + # the next time it wants the tool, and keeps the loaded definitions out of the cached + # prefix for good. + if role == "assistant": + converted_messages.extend(_replayed_provider_items(msg)) + # Handle assistant messages with tool_calls -> function_call items if role == "assistant" and msg.get("tool_calls"): # First add any text content as a message @@ -381,12 +535,17 @@ def format_request_payload(self, messages: List[Dict], **kwargs) -> Dict[str, An # Convert each tool_call to a function_call item for tc in msg["tool_calls"]: func = tc.get("function", {}) - converted_messages.append({ + call_item = { "type": "function_call", "call_id": tc.get("id"), "name": func.get("name"), "arguments": func.get("arguments", "{}") - }) + } + # The container the provider's search loaded this function from, when it named + # one. The item that goes back is the provider's own, so it goes back whole. + if tc.get(NAMESPACE_LABEL_KEY): + call_item[NAMESPACE_LABEL_KEY] = tc[NAMESPACE_LABEL_KEY] + converted_messages.append(call_item) # Handle tool role messages -> function_call_output items elif role == "tool": converted_messages.append({ @@ -497,44 +656,13 @@ def format_request_payload(self, messages: List[Dict], **kwargs) -> Dict[str, An # Handle tools - Responses API uses flattened structure (internally tagged) # Converts externally tagged format to internally tagged format. - # A per-tool ``defer_loading: true`` rides the Chat-Completions tool dict top-level - # (deferred tool loading); it maps onto the flat Responses tool and triggers the - # ``tool_search`` built-in so deferred tools are discovered on demand (their schemas stay - # out of the cached prefix). Nothing deferred → byte-identical to before. + # A per-tool ``defer_loading: true`` and a per-tool ``namespace`` label ride the + # Chat-Completions tool dict top-level; they map onto the flat Responses tool, its + # container, and the ``tool_search`` built-in, so deferred tools are discovered on demand + # and their schemas stay out of the cached prefix. Nothing deferred and nothing labelled → + # byte-identical to before. if kwargs.get("tools"): - tools = kwargs["tools"] - converted_tools = [] - any_deferred = False - for tool in tools: - if isinstance(tool, dict): - if tool.get("type") == "function" and "function" in tool: - # Convert from Chat Completions format (externally tagged) - func = tool["function"] - converted = { - "type": "function", - "name": func.get("name"), - "description": func.get("description"), - "parameters": func.get("parameters"), - # Note: strict is true by default in Responses API - } - if tool.get("defer_loading"): - converted["defer_loading"] = True - any_deferred = True - converted_tools.append(converted) - else: - # Already in Responses API format or other tool type - converted_tools.append(tool) - if isinstance(tool, dict) and tool.get("defer_loading"): - any_deferred = True - else: - converted_tools.append(tool) - if any_deferred and not any( - isinstance(t, dict) and t.get("type") == "tool_search" for t in converted_tools - ): - # Auto-add the Responses tool-search built-in so deferred tools are discoverable - # (gpt-5.4+). Suppressed if the caller supplied their own. - converted_tools.append({"type": "tool_search"}) - payload["tools"] = converted_tools + payload["tools"] = _convert_tools_for_responses(kwargs["tools"]) # Handle OpenAI reasoning (effort-based for all models via Responses API). # An explicit `reasoning_effort` wins; failing that, a caller's thinking budget @@ -743,6 +871,11 @@ def harmonize_response( finish_reason = None reasoning_data = None tool_calls = [] + # The hosted search's own output items, kept whole and in arrival order. They ride the + # opaque provider-items channel the other legs already use for the blocks their endpoints + # demand back verbatim; the next request replays them, which is what puts a loaded + # definition into the conversation body and, from the call after, into the cached prefix. + provider_items: List[Dict[str, Any]] = [] # Parse output array from Responses API output_array = raw_response.get("output", []) @@ -799,9 +932,15 @@ def harmonize_response( "name": item.get("name", ""), "arguments": item.get("arguments", "") }, + # Present only when the provider loaded this function out of a namespace. + namespace=item.get("namespace") or None, ) ) + # The hosted search's two items, kept verbatim for the replay (see ``provider_items``). + elif item_type in _TOOL_SEARCH_ITEM_TYPES: + provider_items.append(item) + # Responses API uses input_tokens/output_tokens (not prompt/ # completion_tokens) and nests reasoning in output_tokens_details. # Fall back to chat-completions names for endpoint compat. @@ -881,6 +1020,7 @@ def harmonize_response( content=content, tool_calls=tool_calls, reasoning=reasoning_data, + reasoning_details=provider_items or None, metadata=metadata, ) diff --git a/src/marsys/models/adapters/openrouter.py b/src/marsys/models/adapters/openrouter.py index 2801ea36..26cca84b 100644 --- a/src/marsys/models/adapters/openrouter.py +++ b/src/marsys/models/adapters/openrouter.py @@ -186,18 +186,23 @@ def format_request_payload(self, messages: List[Dict], **kwargs) -> Dict[str, An if kwargs.get("tools"): tools = kwargs["tools"] - if any(isinstance(t, dict) and t.get("defer_loading") for t in tools): + if any( + isinstance(t, dict) and (t.get("defer_loading") or t.get("namespace")) + for t in tools + ): # OpenRouter has no deferred-tool-loading feature and forwards `tools` verbatim, - # so a defer_loading marker would reach the wire (possible 400). Strip it and fall - # back to eager loading. Warn — a SILENT behavior change is what the additive - # contract forbids. + # so a defer_loading marker, or the namespace label that rides beside it to say + # which container a deferred function belongs in, would reach the wire (possible + # 400). Strip both and fall back to eager loading with every tool at the top + # level. Warn — a SILENT behavior change is what the additive contract forbids. import warnings warnings.warn( "OpenRouter does not support deferred tool loading; the per-tool defer_loading " - "flag is stripped and all tools are loaded eagerly." + "flag and its namespace label are stripped and all tools are loaded eagerly." ) tools = [ - {k: v for k, v in t.items() if k != "defer_loading"} if isinstance(t, dict) else t + {k: v for k, v in t.items() if k not in ("defer_loading", "namespace")} + if isinstance(t, dict) else t for t in tools ] payload["tools"] = tools diff --git a/src/marsys/models/response_models.py b/src/marsys/models/response_models.py index 6e17b882..a854e6b3 100644 --- a/src/marsys/models/response_models.py +++ b/src/marsys/models/response_models.py @@ -6,7 +6,14 @@ from datetime import datetime from typing import Any, Dict, List, Optional, Union -from pydantic import BaseModel, Field, field_validator, model_validator, ConfigDict +from pydantic import ( + BaseModel, + ConfigDict, + Field, + field_validator, + model_serializer, + model_validator, +) class ToolCall(BaseModel): @@ -14,7 +21,11 @@ class ToolCall(BaseModel): id: str type: str = "function" function: Dict[str, Any] = Field(default_factory=dict) - + # The container the provider loaded this function out of, when the leg names one: a hosted + # tool search returns the namespace it searched and stamps the call with it. Carried so the + # item can go back to that provider whole on the next request. + namespace: Optional[str] = None + @field_validator('function') @classmethod def validate_function(cls, v): @@ -25,6 +36,19 @@ def validate_function(cls, v): v['arguments'] = {} return v + @model_serializer(mode="wrap") + def _omit_absent_namespace(self, handler): + """Drop ``namespace`` from the serialized form when the provider named none. + + Every leg but one sends no namespace, and a tool call's serialized form becomes a durable + conversation row on the way back: a key that is always there and always null would change + those bytes on every call of every leg, which is a cache re-write for a field nobody + reads. Absent means absent.""" + data = handler(self) + if isinstance(data, dict) and data.get("namespace") is None: + data.pop("namespace", None) + return data + class UsageInfo(BaseModel): """Token usage information. @@ -111,12 +135,17 @@ class HarmonizedResponse(BaseModel): tool_calls: List[ToolCall] = Field(default_factory=list) reasoning: Optional[str] = None # For o1 models or reasoning traces thinking: Optional[str] = None # For thinking/planning content - # Opaque provider reasoning blocks that must round-trip VERBATIM on the - # next request: Gemini 3 thought signatures ({"type": "text"|"function_call", - # "thought_signature": ...}) and Anthropic extended-thinking blocks + # Opaque provider-emitted blocks that must round-trip VERBATIM on the next + # request: Gemini 3 thought signatures ({"type": "text"|"function_call", + # "thought_signature": ...}), Anthropic extended-thinking blocks # ({"type": "thinking", "thinking", "signature"} / {"type": - # "redacted_thinking", "data"}). Type-discriminated — each provider's - # payload builder re-emits only its own block types. + # "redacted_thinking", "data"}), and the OpenAI Responses hosted tool search's + # own two output items ({"type": "tool_search_call"} and + # {"type": "tool_search_output"}, the output carrying the definitions the + # search loaded). Type-discriminated — each provider's payload builder + # re-emits only its own block types. The name says reasoning and the channel + # means provider output the next request owes back; renaming it reaches four + # adapters, the agent memory and the tracing, and every row already written. reasoning_details: Optional[List[Dict[str, Any]]] = None metadata: ResponseMetadata diff --git a/tests/models/test_hosted_tool_search.py b/tests/models/test_hosted_tool_search.py new file mode 100644 index 00000000..1b2e14b9 --- /dev/null +++ b/tests/models/test_hosted_tool_search.py @@ -0,0 +1,529 @@ +"""Hosted tool search on the OpenAI Responses legs — namespaces, the built-in, and the round trip. + +The provider runs the search itself. The request carries a `namespace` container per family, its +members deferred, and the `tool_search` built-in beside them; the model searches, the provider +injects the loaded definitions at the end of the context and the model calls in the same reply; +the reply's two search items come back on the next request so the loaded definitions ride the +conversation body instead of being searched for again. + +Three things here are measured facts rather than preferences, and each has a test that would fail +loudly if the adapter forgot them. The provider REFUSES a deferred tool with no built-in beside it +("Invalid Value: 'tools.defer_loading'. Deferred tools require tools.tool_search.", HTTP 400) — +and the flag never sits on a container, so reading only the top level turns the namespace request +into a refusal. The search items are dropped by a parser that knows only reasoning, message and +function_call, and dropping them costs a second search of the whole family at the next need. And +the function call comes back stamped with the container it was loaded from. + +No network anywhere in this file. +""" +import pytest + +from marsys.models.adapters.anthropic import AnthropicAdapter +from marsys.models.adapters.anthropic_oauth import AnthropicOAuthAdapter +from marsys.models.adapters.azure import AzureOpenAIAdapter +from marsys.models.adapters.google import GoogleAdapter +from marsys.models.adapters.openai import ( + AsyncOpenAIAdapter, + OpenAIAdapter, + supports_explicit_prompt_cache, + supports_hosted_tool_search, +) +from marsys.models.adapters.openrouter import OpenRouterAdapter +from marsys.models.response_models import ToolCall + +MSGS = [{"role": "user", "content": "hi"}] +LEDGER = {"name": "ledger", "description": "the books: close a quarter, reopen a period"} +PEOPLE = {"name": "people", "description": "who works here and what they may see"} + + +def _tool(name, *, defer=False, label=None): + """A Chat-Completions tool dict the way a caller hands one to the builder.""" + tool = { + "type": "function", + "function": { + "name": name, + "description": f"{name} description", + "parameters": {"type": "object", "properties": {}}, + }, + } + if defer: + tool["defer_loading"] = True + if label is not None: + tool["namespace"] = label + return tool + + +@pytest.fixture +def openai(): + return OpenAIAdapter(model_name="gpt-5.6-terra", api_key="x", base_url="https://api.openai.com/v1") + + +@pytest.fixture +def azure(): + return AzureOpenAIAdapter( + model_name="gpt-5.6-terra", api_key="x", base_url="https://example.invalid/openai/v1" + ) + + +@pytest.fixture +def openrouter(): + return OpenRouterAdapter( + model_name="anthropic/claude-sonnet-4-6", api_key="x", + base_url="https://openrouter.ai/api/v1", + ) + + +@pytest.fixture +def google(): + return GoogleAdapter( + model_name="gemini-3.5-flash", api_key="x", + base_url="https://generativelanguage.googleapis.com", + ) + + +@pytest.fixture +def anthropic(): + return AnthropicAdapter( + model_name="claude-sonnet-4-6", api_key="x", base_url="https://api.anthropic.com/v1" + ) + + +@pytest.fixture +def anthropic_oauth(monkeypatch): + monkeypatch.setattr( + AnthropicOAuthAdapter, "_load_claude_credentials", + lambda self, path=None: {"access_token": "fake"}, + ) + return AnthropicOAuthAdapter(model_name="claude-sonnet-4-6", auto_refresh=False) + + +def _containers(tools): + return [t for t in tools if isinstance(t, dict) and t.get("type") == "namespace"] + + +def _by_name(tools, name): + return next((t for t in tools if isinstance(t, dict) and t.get("name") == name), None) + + +def _built_ins(tools): + return [t for t in tools if isinstance(t, dict) and t.get("type") == "tool_search"] + + +# --- the namespace container --------------------------------------------------------- + + +@pytest.mark.parametrize("leg", ["openai", "azure"]) +def test_labelled_tools_group_into_one_container_per_label(request, leg): + adapter = request.getfixturevalue(leg) + tools = adapter.format_request_payload( + MSGS, + tools=[ + _tool("ledger_close", defer=True, label=LEDGER), + _tool("people_list", defer=True, label=PEOPLE), + _tool("ledger_reopen", defer=True, label=LEDGER), + ], + )["tools"] + containers = _containers(tools) + assert [c["name"] for c in containers] == ["ledger", "people"] # first-appearance order + assert [f["name"] for f in containers[0]["tools"]] == ["ledger_close", "ledger_reopen"] + assert [f["name"] for f in containers[1]["tools"]] == ["people_list"] + + +def test_a_container_sits_where_its_first_member_appeared(openai): + tools = openai.format_request_payload( + MSGS, + tools=[ + _tool("always_on"), + _tool("ledger_close", defer=True, label=LEDGER), + _tool("also_always_on"), + _tool("ledger_reopen", defer=True, label=LEDGER), + ], + )["tools"] + kinds = [(t.get("type"), t.get("name")) for t in tools] + assert kinds[:3] == [ + ("function", "always_on"), + ("namespace", "ledger"), + ("function", "also_always_on"), + ] + + +def test_members_keep_the_flag_and_the_container_carries_none(openai): + tools = openai.format_request_payload( + MSGS, tools=[_tool("ledger_close", defer=True, label=LEDGER)] + )["tools"] + container = _containers(tools)[0] + assert "defer_loading" not in container + assert container["tools"][0]["defer_loading"] is True + + +def test_the_container_takes_its_name_and_description_from_the_label(openai): + container = _containers( + openai.format_request_payload( + MSGS, tools=[_tool("ledger_close", defer=True, label=LEDGER)] + )["tools"] + )[0] + assert container["name"] == LEDGER["name"] + assert container["description"] == LEDGER["description"] + + +def test_a_member_renders_as_a_flat_responses_function(openai): + member = _containers( + openai.format_request_payload( + MSGS, tools=[_tool("ledger_close", defer=True, label=LEDGER)] + )["tools"] + )[0]["tools"][0] + assert member["type"] == "function" + assert member["name"] == "ledger_close" + assert member["description"] == "ledger_close description" + assert member["parameters"] == {"type": "object", "properties": {}} + assert "function" not in member # flattened, not the externally-tagged shape + assert "namespace" not in member # the label named the container, it does not ride the member + + +def test_the_callers_tool_dicts_are_not_mutated(openai): + sent = [ + _tool("ledger_close", defer=True, label=LEDGER), + _tool("always_on"), + ] + before = [dict(t) for t in sent] + openai.format_request_payload(MSGS, tools=sent) + assert sent == before + assert sent[0]["namespace"] is LEDGER # the label object itself is untouched + + +def test_a_label_without_a_name_names_no_container(openai): + tools = openai.format_request_payload( + MSGS, tools=[_tool("ledger_close", defer=True, label={"description": "no name"})] + )["tools"] + assert _containers(tools) == [] + assert _by_name(tools, "ledger_close")["defer_loading"] is True + + +# --- the built-in, which the provider refuses the request without --------------------- + + +def test_the_built_in_is_added_when_a_container_member_is_deferred(openai): + tools = openai.format_request_payload( + MSGS, tools=[_tool("ledger_close", defer=True, label=LEDGER)] + )["tools"] + assert len(_built_ins(tools)) == 1 + + +def test_the_built_in_is_added_when_a_top_level_tool_is_deferred(openai): + tools = openai.format_request_payload(MSGS, tools=[_tool("ledger_close", defer=True)])["tools"] + assert len(_built_ins(tools)) == 1 + + +def test_a_caller_supplied_built_in_is_not_duplicated(openai): + tools = openai.format_request_payload( + MSGS, + tools=[_tool("ledger_close", defer=True, label=LEDGER), {"type": "tool_search"}], + )["tools"] + assert len(_built_ins(tools)) == 1 + + +def test_a_container_with_nothing_deferred_adds_no_built_in(openai): + tools = openai.format_request_payload( + MSGS, tools=[_tool("ledger_close", label=LEDGER)] + )["tools"] + assert _containers(tools)[0]["tools"][0].get("defer_loading") is None + assert _built_ins(tools) == [] + + +def test_nothing_labelled_and_nothing_deferred_renders_as_before(openai): + tools = openai.format_request_payload(MSGS, tools=[_tool("a"), _tool("b")])["tools"] + assert tools == [ + { + "type": "function", + "name": name, + "description": f"{name} description", + "parameters": {"type": "object", "properties": {}}, + } + for name in ("a", "b") + ] + + +# --- the legs that must never see the label ------------------------------------------ + + +def test_openrouter_strips_the_label_and_warns(openrouter): + with pytest.warns(Warning, match="namespace label"): + tools = openrouter.format_request_payload( + MSGS, tools=[_tool("ledger_close", defer=True, label=LEDGER)] + )["tools"] + assert all("namespace" not in t and "defer_loading" not in t for t in tools) + + +def test_openrouter_strips_a_label_that_arrives_without_the_flag(openrouter): + with pytest.warns(Warning, match="namespace label"): + tools = openrouter.format_request_payload( + MSGS, tools=[_tool("ledger_close", label=LEDGER)] + )["tools"] + assert all("namespace" not in t for t in tools) + + +def test_google_warns_and_sends_no_label(google): + with pytest.warns(Warning, match="namespace label"): + payload = google.format_request_payload( + MSGS, tools=[_tool("ledger_close", defer=True, label=LEDGER)] + ) + wire = repr(payload["tools"]) + assert "namespace" not in wire and "defer_loading" not in wire + + +@pytest.mark.parametrize("leg", ["anthropic", "anthropic_oauth"]) +def test_the_anthropic_builders_drop_the_label_and_keep_their_mapping(request, leg): + adapter = request.getfixturevalue(leg) + tools = adapter.format_request_payload( + MSGS, tools=[_tool("ledger_close", defer=True, label=LEDGER), _tool("core_tool")] + )["tools"] + assert all("namespace" not in t for t in tools) + assert _by_name(tools, "ledger_close")["defer_loading"] is True + assert "defer_loading" not in _by_name(tools, "core_tool") + assert any(t.get("type") == "tool_search_tool_regex_20251119" for t in tools) + + +# --- the parser ---------------------------------------------------------------------- + + +def _reply(output): + return {"id": "resp_1", "model": "gpt-5.6-terra", "output": output, "usage": {}} + + +SEARCH_CALL = {"type": "tool_search_call", "id": "ts_1", "status": "completed", "paths": ["ledger"]} +SEARCH_OUTPUT = { + "type": "tool_search_output", + "id": "tso_1", + "status": "completed", + "tools": [{"type": "namespace", "name": "ledger", "tools": [{"name": "ledger_close"}]}], +} + + +def test_the_search_items_are_kept_verbatim_and_in_order(openai): + resp = openai.harmonize_response( + _reply([ + {"type": "reasoning", "summary": ["thinking"]}, + SEARCH_CALL, + SEARCH_OUTPUT, + {"type": "function_call", "call_id": "c1", "name": "ledger_close", + "arguments": "{}", "namespace": "ledger"}, + ]), + request_start_time=0.0, + ) + assert resp.reasoning_details == [SEARCH_CALL, SEARCH_OUTPUT] + + +def test_reasoning_message_and_function_call_items_parse_as_before(openai): + resp = openai.harmonize_response( + _reply([ + {"type": "reasoning", "summary": ["because"]}, + {"type": "message", "role": "assistant", "status": "completed", + "content": [{"type": "output_text", "text": "done"}]}, + {"type": "function_call", "call_id": "c1", "name": "ledger_close", "arguments": "{}"}, + ]), + request_start_time=0.0, + ) + assert resp.reasoning == "because" + assert resp.content == "done" + assert resp.metadata.finish_reason == "stop" + assert resp.tool_calls[0].id == "c1" + assert resp.tool_calls[0].function == {"name": "ledger_close", "arguments": "{}"} + assert resp.reasoning_details is None + + +def test_the_calls_namespace_is_read_when_the_provider_sent_one(openai): + resp = openai.harmonize_response( + _reply([ + SEARCH_CALL, SEARCH_OUTPUT, + {"type": "function_call", "call_id": "c1", "name": "ledger_close", + "arguments": "{}", "namespace": "ledger"}, + ]), + request_start_time=0.0, + ) + assert resp.tool_calls[0].namespace == "ledger" + + +def test_a_call_without_a_namespace_has_none_and_serializes_as_before(openai): + resp = openai.harmonize_response( + _reply([ + {"type": "function_call", "call_id": "c1", "name": "ledger_close", "arguments": "{}"} + ]), + request_start_time=0.0, + ) + assert resp.tool_calls[0].namespace is None + assert resp.tool_calls[0].model_dump() == { + "id": "c1", "type": "function", + "function": {"name": "ledger_close", "arguments": "{}"}, + } + + +def test_a_namespaced_call_serializes_with_it(): + call = ToolCall( + id="c1", function={"name": "ledger_close", "arguments": "{}"}, namespace="ledger" + ) + assert call.model_dump()["namespace"] == "ledger" + + +# --- the replay ---------------------------------------------------------------------- + + +def _assistant_row(**extra): + row = { + "role": "assistant", + "content": "", + "tool_calls": [ + {"id": "c1", "type": "function", + "function": {"name": "ledger_close", "arguments": "{}"}}, + ], + } + row.update(extra) + return row + + +def test_the_search_items_are_replayed_ahead_of_the_rows_message_and_calls(openai): + payload = openai.format_request_payload([ + {"role": "user", "content": "close the quarter"}, + _assistant_row(content="on it", reasoning_details=[SEARCH_CALL, SEARCH_OUTPUT]), + {"role": "tool", "tool_call_id": "c1", "content": "{}"}, + ]) + types = [item.get("type") or item.get("role") for item in payload["input"]] + assert types == [ + "user", "tool_search_call", "tool_search_output", "assistant", + "function_call", "function_call_output", + ] + assert payload["input"][1] == SEARCH_CALL + assert payload["input"][2] == SEARCH_OUTPUT + + +def test_a_replayed_call_carries_its_namespace(openai): + row = _assistant_row(reasoning_details=[SEARCH_CALL, SEARCH_OUTPUT]) + row["tool_calls"][0]["namespace"] = "ledger" + payload = openai.format_request_payload([{"role": "user", "content": "go"}, row]) + call = next(i for i in payload["input"] if i.get("type") == "function_call") + assert call["namespace"] == "ledger" + + +def test_a_call_the_provider_sent_without_a_namespace_replays_without_one(openai): + payload = openai.format_request_payload([{"role": "user", "content": "go"}, _assistant_row()]) + call = next(i for i in payload["input"] if i.get("type") == "function_call") + assert call == { + "type": "function_call", "call_id": "c1", "name": "ledger_close", "arguments": "{}", + } + + +def test_another_legs_blocks_on_the_channel_are_not_replayed_here(openai): + payload = openai.format_request_payload([ + {"role": "user", "content": "go"}, + _assistant_row(reasoning_details=[ + {"type": "thinking", "thinking": "…", "signature": "sig"}, + SEARCH_CALL, + {"type": "text", "thought_signature": "sig"}, + ]), + ]) + types = [item.get("type") for item in payload["input"]] + assert types.count("tool_search_call") == 1 + assert "thinking" not in types and "text" not in types + + +def test_a_text_answering_reply_replays_its_search_items_too(openai): + """A reply that searches and then answers carries no tool call, and its loaded definitions + are exactly what the next request must not lose.""" + payload = openai.format_request_payload([ + {"role": "user", "content": "what do we owe?"}, + {"role": "assistant", "content": "about forty", "reasoning_details": [SEARCH_CALL, SEARCH_OUTPUT]}, + {"role": "user", "content": "and last quarter?"}, + ]) + types = [item.get("type") or item.get("role") for item in payload["input"]] + assert types == ["user", "tool_search_call", "tool_search_output", "assistant", "user"] + + +# --- the round trip ------------------------------------------------------------------ + + +def test_the_round_trip_grows_at_the_end_with_a_byte_identical_tools_array(openai): + tools = [ + _tool("workspace_read"), + _tool("ledger_close", defer=True, label=LEDGER), + _tool("ledger_reopen", defer=True, label=LEDGER), + ] + first = openai.format_request_payload( + [{"role": "user", "content": "close the quarter"}], tools=tools + ) + reply = openai.harmonize_response( + _reply([ + SEARCH_CALL, SEARCH_OUTPUT, + {"type": "function_call", "call_id": "c1", "name": "ledger_close", + "arguments": "{}", "namespace": "ledger"}, + ]), + request_start_time=0.0, + ) + second = openai.format_request_payload( + [ + {"role": "user", "content": "close the quarter"}, + {"role": "assistant", "content": "", + "tool_calls": [tc.model_dump() for tc in reply.tool_calls], + "reasoning_details": reply.reasoning_details}, + {"role": "tool", "tool_call_id": "c1", "content": "closed"}, + ], + tools=tools, + ) + assert second["input"][: len(first["input"])] == first["input"] + assert [i.get("type") for i in second["input"][len(first["input"]):]] == [ + "tool_search_call", "tool_search_output", "function_call", "function_call_output", + ] + assert second["input"][-2]["namespace"] == "ledger" + assert second["tools"] == first["tools"] + + +# --- the predicate ------------------------------------------------------------------- + + +@pytest.mark.parametrize("provider", ["openai", "azure"]) +@pytest.mark.parametrize("model", ["gpt-5.4", "gpt-5.5", "gpt-5.6-terra", "gpt-6"]) +def test_the_predicate_answers_true_for_the_served_legs(provider, model): + assert supports_hosted_tool_search(provider, model) is True + + +@pytest.mark.parametrize("model", ["gpt-5.3", "gpt-5", "gpt-4.1", "o3", "claude-sonnet-4-6", ""]) +def test_the_predicate_answers_false_below_the_floor(model): + assert supports_hosted_tool_search("openai", model) is False + + +@pytest.mark.parametrize("provider", ["openrouter", "google", "anthropic", None]) +def test_the_predicate_answers_false_off_the_served_providers(provider): + assert supports_hosted_tool_search(provider, "gpt-5.6-terra") is False + + +@pytest.mark.parametrize("model", ["gpt-5.4", "gpt-5.5"]) +def test_the_two_gates_keep_their_own_floors(model): + assert supports_hosted_tool_search("openai", model) is True + assert supports_explicit_prompt_cache(model) is False + + +def test_the_predicate_states_the_azure_deployment_caveat(): + assert "deployment" in (supports_hosted_tool_search.__doc__ or "") + + +# --- byte identity everywhere nothing is labelled and nothing deferred ---------------- + + +@pytest.mark.parametrize("adapter_type", [OpenAIAdapter, AsyncOpenAIAdapter]) +def test_a_plain_request_is_untouched_by_any_of_this(adapter_type): + adapter = adapter_type( + model_name="gpt-5.6-terra", api_key="x", base_url="https://api.openai.com/v1" + ) + payload = adapter.format_request_payload( + [ + {"role": "user", "content": "hi"}, + _assistant_row(content="calling"), + {"role": "tool", "tool_call_id": "c1", "content": "{}"}, + ], + tools=[_tool("a")], + ) + assert [i.get("type") or i.get("role") for i in payload["input"]] == [ + "user", "assistant", "function_call", "function_call_output", + ] + assert payload["tools"] == [{ + "type": "function", "name": "a", "description": "a description", + "parameters": {"type": "object", "properties": {}}, + }]