From dfbb9650255479f87e39dbe3bfdc0c898a9e8e5e Mon Sep 17 00:00:00 2001 From: rezaho Date: Sat, 12 Sep 2026 17:17:37 +0200 Subject: [PATCH 1/2] Group deferred tools into namespaces and keep the provider's search items MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The Responses legs serve a hosted tool search: the request carries a namespace container per family with its members deferred and the tool_search built-in beside them, the model searches, and the provider injects the loaded definitions at the end of the context. Three things stood between this adapter and that shape, and a live probe of an Azure gpt-5.6 deployment measured each of them. The built-in was added only when defer_loading sat at a tool's top level. The flag never sits on a container, so a namespace request went out without the built-in and came back HTTP 400, "Invalid Value: 'tools.defer_loading'. Deferred tools require tools.tool_search." — a refusal, not a degradation. Deferral is now read wherever it sits. The parser knew reasoning, message and function_call, so the search's own two items were dropped and with them the definitions the search had loaded. They ride the opaque provider-items channel the Anthropic and Google legs already use for the blocks their endpoints demand back verbatim, and the request builder replays them ahead of the row's message and calls — including for a reply that searches and then answers in text, which carries no tool call to hang them on. Without the replay the model searches the whole family again at its next need; with it the loaded definition rides the conversation body into the cached prefix. The function call came back stamped with the container it was loaded from and that stamp was dropped too. It rides the tool call and goes back on the replayed item, because the item that goes back is the provider's own. Absent when the provider named none, so a call on every other leg serializes exactly as it did. The grouping travels as a per-tool label rather than a container the caller builds: the three legs that must not see it strip one key off a flat dict, the move they already make for defer_loading, instead of unwrapping a tool type they have no shape for. OpenRouter strips and warns, Google warns, and the Anthropic builders drop it by construction while keeping their own deferred-loading mapping. The hosted-search predicate is its own function with its own floor. Tool search is served from gpt-5.4 while the explicit prompt-cache markers beside it need 5.6, so one gate for both would either withhold the search from two generations that serve it or send cache fields to a model that answers them with a 400. Depends on #62: written on top of it, since both change the request builder and the response harmonizer. With nothing labelled and nothing deferred every payload on every leg is byte-identical to before. --- src/marsys/models/adapters/google.py | 11 +- src/marsys/models/adapters/openai.py | 215 +++++++-- src/marsys/models/adapters/openrouter.py | 17 +- src/marsys/models/response_models.py | 43 +- tests/models/test_hosted_tool_search.py | 529 +++++++++++++++++++++++ 5 files changed, 760 insertions(+), 55 deletions(-) create mode 100644 tests/models/test_hosted_tool_search.py diff --git a/src/marsys/models/adapters/google.py b/src/marsys/models/adapters/google.py index 7619f869..ef0cf717 100644 --- a/src/marsys/models/adapters/google.py +++ b/src/marsys/models/adapters/google.py @@ -244,13 +244,18 @@ def format_request_payload(self, messages: List[Dict], **kwargs) -> Dict[str, An # Add native function calling support if kwargs.get("tools"): - if any(isinstance(t, dict) and t.get("defer_loading") for t in kwargs["tools"]): + if any( + isinstance(t, dict) and (t.get("defer_loading") or t.get("namespace")) + for t in kwargs["tools"] + ): # Google has no deferred-tool-loading feature; the rebuild below already drops the - # per-tool defer_loading flag, so warn (not a silent drop) and fall back to eager. + # per-tool defer_loading flag and the namespace label that rides beside it, so + # warn (not a silent drop) and fall back to eager loading with every tool flat. import warnings warnings.warn( "Google models do not support deferred tool loading; the per-tool " - "defer_loading flag is ignored and all tools are loaded eagerly." + "defer_loading flag and its namespace label are ignored and all tools are " + "loaded eagerly." ) # Convert OpenAI format tools to Google format google_tools = [] diff --git a/src/marsys/models/adapters/openai.py b/src/marsys/models/adapters/openai.py index 12b7cd16..7b2f9871 100644 --- a/src/marsys/models/adapters/openai.py +++ b/src/marsys/models/adapters/openai.py @@ -102,6 +102,147 @@ _GENERATION_RE = re.compile(r"^gpt-(\d+)(?:\.(\d+))?") +# --- hosted tool search ------------------------------------------------------------- +# +# The provider runs the search itself: the request carries deferred functions, grouped in +# namespace containers or standing alone, plus the `tool_search` built-in; the model searches, +# the provider injects the loaded definitions at the end of the context, and the model calls in +# the same reply. + +# The generation that serves it. Its OWN floor, two generations below the explicit prompt-cache +# one beside it, because the two features shipped apart: one gate for both would either withhold +# the search from models that serve it or send cache fields to models that answer them with a 400. +_HOSTED_TOOL_SEARCH_MIN_VERSION = (5, 4) + +# The providers whose endpoints were measured to serve it. The factory routes an unrecognized +# provider to this adapter, so a third-party OpenAI-compatible endpoint behind a GPT-5-shaped +# model name would otherwise be told it serves a feature nobody asked it about. +_HOSTED_TOOL_SEARCH_PROVIDERS = frozenset({"openai", "azure"}) + +# The reply items a hosted search emits: the call, and its output carrying the definitions the +# search loaded. Kept off the reply and replayed on the next request verbatim. +_TOOL_SEARCH_ITEM_TYPES = frozenset({"tool_search_call", "tool_search_output"}) + +# The key a caller puts on a Chat-Completions tool dict to say which container the rendered +# function belongs in: `{"name": ..., "description": ...}`. The grouping travels per tool rather +# than as a container the caller builds, so the legs that must not see it strip one key off a flat +# dict — the move they already make for `defer_loading` — instead of unwrapping a tool type they +# have no shape for. +NAMESPACE_LABEL_KEY = "namespace" + + +def supports_hosted_tool_search(provider: Optional[str], model_lower: str) -> bool: + """Whether this leg serves the provider-run tool search over deferred functions. + + Carries the same honesty caveat as the prompt-cache gate above, and it bites harder here: on + Azure the model field is an operator-chosen deployment label rather than a model name, so a + deployment named after a model it does not serve lies to this check — and the failure is a 400 + on `tools.defer_loading` for every call rather than a price. Ask the deployment before turning + the feature on for it. + """ + if provider not in _HOSTED_TOOL_SEARCH_PROVIDERS: + return False + match = _GENERATION_RE.match(model_lower or "") + if not match: + return False + major = int(match.group(1)) + minor = int(match.group(2) or 0) + return (major, minor) >= _HOSTED_TOOL_SEARCH_MIN_VERSION + + +def _namespace_label(tool: Any) -> Optional[Dict[str, Any]]: + """The namespace label on a tool dict, or None. A label with no name names no container, so + it is no label at all.""" + if not isinstance(tool, dict): + return None + label = tool.get(NAMESPACE_LABEL_KEY) + return label if isinstance(label, dict) and label.get("name") else None + + +def _convert_tools_for_responses(tools: Collection[Any]) -> List[Any]: + """The Responses `tools` array for a caller's tool list. + + Three jobs in one pass, because they read the same dicts. A Chat-Completions `function` tool + flattens to the internally-tagged Responses shape. A tool carrying a namespace label goes + inside a `namespace` container instead of standing at the top level — one container per + distinct label name, placed where its first member appeared, members in the order given, and + the container never carries the deferral flag its members do. And a request that defers + anything gains the hosted-search built-in: the provider refuses a deferred tool without it + ("Deferred tools require tools.tool_search") whether the flag sat at the top level or inside a + container, so reading only the top level makes a namespace request a refusal rather than a + degradation. + + The caller's dicts are never mutated — a converted function is a new dict and a container is + built here — so a tool array a caller holds frozen across a turn stays what it was. + """ + converted: List[Any] = [] + containers: Dict[str, Dict[str, Any]] = {} + any_deferred = False + for tool in tools: + if not isinstance(tool, dict): + converted.append(tool) + continue + if tool.get("type") == "function" and "function" in tool: + # Convert from Chat Completions format (externally tagged) + func = tool["function"] + rendered: Any = { + "type": "function", + "name": func.get("name"), + "description": func.get("description"), + "parameters": func.get("parameters"), + # Note: strict is true by default in Responses API + } + if tool.get("defer_loading"): + rendered["defer_loading"] = True + any_deferred = True + else: + # Already in Responses API format or other tool type + rendered = ( + {k: v for k, v in tool.items() if k != NAMESPACE_LABEL_KEY} + if NAMESPACE_LABEL_KEY in tool + else tool + ) + if tool.get("defer_loading"): + any_deferred = True + label = _namespace_label(tool) + if label is None: + converted.append(rendered) + continue + container = containers.get(label["name"]) + if container is None: + container = { + "type": "namespace", + "name": label["name"], + "description": label.get("description", ""), + "tools": [], + } + containers[label["name"]] = container + converted.append(container) + container["tools"].append(rendered) + if any_deferred and not any( + isinstance(t, dict) and t.get("type") == "tool_search" for t in converted + ): + # Auto-add the Responses tool-search built-in so deferred tools are discoverable + # (gpt-5.4+). Suppressed if the caller supplied their own. + converted.append({"type": "tool_search"}) + return converted + + +def _replayed_provider_items(msg: Dict[str, Any]) -> List[Dict[str, Any]]: + """This assistant row's provider items that belong back in the Responses input array. + + Type-discriminated, the way every leg reads this channel: the field is shared with the other + providers' opaque blocks (extended-thinking blocks, thought signatures) and each payload + builder re-emits only the types its own endpoint emitted. Anything else on it is another + leg's and is left where it is. + """ + return [ + item + for item in (msg.get("reasoning_details") or []) + if isinstance(item, dict) and item.get("type") in _TOOL_SEARCH_ITEM_TYPES + ] + + def supports_explicit_prompt_cache(model_lower: str) -> bool: """Whether a model name is GPT-5.6 or later, the generation that serves the fields. @@ -369,6 +510,16 @@ def format_request_payload(self, messages: List[Dict], **kwargs) -> Dict[str, An role = msg.get("role") item_start = len(converted_messages) + # The provider's own items from this reply go back first: a hosted search's call and + # its output, the output carrying the definitions the search loaded. Ahead of the + # row's own message and calls because that is the order the provider emitted them in, + # and because the definitions have to reach the model's context before the call that + # uses them. Dropping them instead costs the model a second search of the same family + # the next time it wants the tool, and keeps the loaded definitions out of the cached + # prefix for good. + if role == "assistant": + converted_messages.extend(_replayed_provider_items(msg)) + # Handle assistant messages with tool_calls -> function_call items if role == "assistant" and msg.get("tool_calls"): # First add any text content as a message @@ -381,12 +532,17 @@ def format_request_payload(self, messages: List[Dict], **kwargs) -> Dict[str, An # Convert each tool_call to a function_call item for tc in msg["tool_calls"]: func = tc.get("function", {}) - converted_messages.append({ + call_item = { "type": "function_call", "call_id": tc.get("id"), "name": func.get("name"), "arguments": func.get("arguments", "{}") - }) + } + # The container the provider's search loaded this function from, when it named + # one. The item that goes back is the provider's own, so it goes back whole. + if tc.get(NAMESPACE_LABEL_KEY): + call_item[NAMESPACE_LABEL_KEY] = tc[NAMESPACE_LABEL_KEY] + converted_messages.append(call_item) # Handle tool role messages -> function_call_output items elif role == "tool": converted_messages.append({ @@ -497,44 +653,13 @@ def format_request_payload(self, messages: List[Dict], **kwargs) -> Dict[str, An # Handle tools - Responses API uses flattened structure (internally tagged) # Converts externally tagged format to internally tagged format. - # A per-tool ``defer_loading: true`` rides the Chat-Completions tool dict top-level - # (deferred tool loading); it maps onto the flat Responses tool and triggers the - # ``tool_search`` built-in so deferred tools are discovered on demand (their schemas stay - # out of the cached prefix). Nothing deferred → byte-identical to before. + # A per-tool ``defer_loading: true`` and a per-tool ``namespace`` label ride the + # Chat-Completions tool dict top-level; they map onto the flat Responses tool, its + # container, and the ``tool_search`` built-in, so deferred tools are discovered on demand + # and their schemas stay out of the cached prefix. Nothing deferred and nothing labelled → + # byte-identical to before. if kwargs.get("tools"): - tools = kwargs["tools"] - converted_tools = [] - any_deferred = False - for tool in tools: - if isinstance(tool, dict): - if tool.get("type") == "function" and "function" in tool: - # Convert from Chat Completions format (externally tagged) - func = tool["function"] - converted = { - "type": "function", - "name": func.get("name"), - "description": func.get("description"), - "parameters": func.get("parameters"), - # Note: strict is true by default in Responses API - } - if tool.get("defer_loading"): - converted["defer_loading"] = True - any_deferred = True - converted_tools.append(converted) - else: - # Already in Responses API format or other tool type - converted_tools.append(tool) - if isinstance(tool, dict) and tool.get("defer_loading"): - any_deferred = True - else: - converted_tools.append(tool) - if any_deferred and not any( - isinstance(t, dict) and t.get("type") == "tool_search" for t in converted_tools - ): - # Auto-add the Responses tool-search built-in so deferred tools are discoverable - # (gpt-5.4+). Suppressed if the caller supplied their own. - converted_tools.append({"type": "tool_search"}) - payload["tools"] = converted_tools + payload["tools"] = _convert_tools_for_responses(kwargs["tools"]) # Handle OpenAI reasoning (effort-based for all models via Responses API). # An explicit `reasoning_effort` wins; failing that, a caller's thinking budget @@ -743,6 +868,11 @@ def harmonize_response( finish_reason = None reasoning_data = None tool_calls = [] + # The hosted search's own output items, kept whole and in arrival order. They ride the + # opaque provider-items channel the other legs already use for the blocks their endpoints + # demand back verbatim; the next request replays them, which is what puts a loaded + # definition into the conversation body and, from the call after, into the cached prefix. + provider_items: List[Dict[str, Any]] = [] # Parse output array from Responses API output_array = raw_response.get("output", []) @@ -799,9 +929,15 @@ def harmonize_response( "name": item.get("name", ""), "arguments": item.get("arguments", "") }, + # Present only when the provider loaded this function out of a namespace. + namespace=item.get("namespace") or None, ) ) + # The hosted search's two items, kept verbatim for the replay (see ``provider_items``). + elif item_type in _TOOL_SEARCH_ITEM_TYPES: + provider_items.append(item) + # Responses API uses input_tokens/output_tokens (not prompt/ # completion_tokens) and nests reasoning in output_tokens_details. # Fall back to chat-completions names for endpoint compat. @@ -881,6 +1017,7 @@ def harmonize_response( content=content, tool_calls=tool_calls, reasoning=reasoning_data, + reasoning_details=provider_items or None, metadata=metadata, ) diff --git a/src/marsys/models/adapters/openrouter.py b/src/marsys/models/adapters/openrouter.py index 2801ea36..26cca84b 100644 --- a/src/marsys/models/adapters/openrouter.py +++ b/src/marsys/models/adapters/openrouter.py @@ -186,18 +186,23 @@ def format_request_payload(self, messages: List[Dict], **kwargs) -> Dict[str, An if kwargs.get("tools"): tools = kwargs["tools"] - if any(isinstance(t, dict) and t.get("defer_loading") for t in tools): + if any( + isinstance(t, dict) and (t.get("defer_loading") or t.get("namespace")) + for t in tools + ): # OpenRouter has no deferred-tool-loading feature and forwards `tools` verbatim, - # so a defer_loading marker would reach the wire (possible 400). Strip it and fall - # back to eager loading. Warn — a SILENT behavior change is what the additive - # contract forbids. + # so a defer_loading marker, or the namespace label that rides beside it to say + # which container a deferred function belongs in, would reach the wire (possible + # 400). Strip both and fall back to eager loading with every tool at the top + # level. Warn — a SILENT behavior change is what the additive contract forbids. import warnings warnings.warn( "OpenRouter does not support deferred tool loading; the per-tool defer_loading " - "flag is stripped and all tools are loaded eagerly." + "flag and its namespace label are stripped and all tools are loaded eagerly." ) tools = [ - {k: v for k, v in t.items() if k != "defer_loading"} if isinstance(t, dict) else t + {k: v for k, v in t.items() if k not in ("defer_loading", "namespace")} + if isinstance(t, dict) else t for t in tools ] payload["tools"] = tools diff --git a/src/marsys/models/response_models.py b/src/marsys/models/response_models.py index 6e17b882..a854e6b3 100644 --- a/src/marsys/models/response_models.py +++ b/src/marsys/models/response_models.py @@ -6,7 +6,14 @@ from datetime import datetime from typing import Any, Dict, List, Optional, Union -from pydantic import BaseModel, Field, field_validator, model_validator, ConfigDict +from pydantic import ( + BaseModel, + ConfigDict, + Field, + field_validator, + model_serializer, + model_validator, +) class ToolCall(BaseModel): @@ -14,7 +21,11 @@ class ToolCall(BaseModel): id: str type: str = "function" function: Dict[str, Any] = Field(default_factory=dict) - + # The container the provider loaded this function out of, when the leg names one: a hosted + # tool search returns the namespace it searched and stamps the call with it. Carried so the + # item can go back to that provider whole on the next request. + namespace: Optional[str] = None + @field_validator('function') @classmethod def validate_function(cls, v): @@ -25,6 +36,19 @@ def validate_function(cls, v): v['arguments'] = {} return v + @model_serializer(mode="wrap") + def _omit_absent_namespace(self, handler): + """Drop ``namespace`` from the serialized form when the provider named none. + + Every leg but one sends no namespace, and a tool call's serialized form becomes a durable + conversation row on the way back: a key that is always there and always null would change + those bytes on every call of every leg, which is a cache re-write for a field nobody + reads. Absent means absent.""" + data = handler(self) + if isinstance(data, dict) and data.get("namespace") is None: + data.pop("namespace", None) + return data + class UsageInfo(BaseModel): """Token usage information. @@ -111,12 +135,17 @@ class HarmonizedResponse(BaseModel): tool_calls: List[ToolCall] = Field(default_factory=list) reasoning: Optional[str] = None # For o1 models or reasoning traces thinking: Optional[str] = None # For thinking/planning content - # Opaque provider reasoning blocks that must round-trip VERBATIM on the - # next request: Gemini 3 thought signatures ({"type": "text"|"function_call", - # "thought_signature": ...}) and Anthropic extended-thinking blocks + # Opaque provider-emitted blocks that must round-trip VERBATIM on the next + # request: Gemini 3 thought signatures ({"type": "text"|"function_call", + # "thought_signature": ...}), Anthropic extended-thinking blocks # ({"type": "thinking", "thinking", "signature"} / {"type": - # "redacted_thinking", "data"}). Type-discriminated — each provider's - # payload builder re-emits only its own block types. + # "redacted_thinking", "data"}), and the OpenAI Responses hosted tool search's + # own two output items ({"type": "tool_search_call"} and + # {"type": "tool_search_output"}, the output carrying the definitions the + # search loaded). Type-discriminated — each provider's payload builder + # re-emits only its own block types. The name says reasoning and the channel + # means provider output the next request owes back; renaming it reaches four + # adapters, the agent memory and the tracing, and every row already written. reasoning_details: Optional[List[Dict[str, Any]]] = None metadata: ResponseMetadata diff --git a/tests/models/test_hosted_tool_search.py b/tests/models/test_hosted_tool_search.py new file mode 100644 index 00000000..1b2e14b9 --- /dev/null +++ b/tests/models/test_hosted_tool_search.py @@ -0,0 +1,529 @@ +"""Hosted tool search on the OpenAI Responses legs — namespaces, the built-in, and the round trip. + +The provider runs the search itself. The request carries a `namespace` container per family, its +members deferred, and the `tool_search` built-in beside them; the model searches, the provider +injects the loaded definitions at the end of the context and the model calls in the same reply; +the reply's two search items come back on the next request so the loaded definitions ride the +conversation body instead of being searched for again. + +Three things here are measured facts rather than preferences, and each has a test that would fail +loudly if the adapter forgot them. The provider REFUSES a deferred tool with no built-in beside it +("Invalid Value: 'tools.defer_loading'. Deferred tools require tools.tool_search.", HTTP 400) — +and the flag never sits on a container, so reading only the top level turns the namespace request +into a refusal. The search items are dropped by a parser that knows only reasoning, message and +function_call, and dropping them costs a second search of the whole family at the next need. And +the function call comes back stamped with the container it was loaded from. + +No network anywhere in this file. +""" +import pytest + +from marsys.models.adapters.anthropic import AnthropicAdapter +from marsys.models.adapters.anthropic_oauth import AnthropicOAuthAdapter +from marsys.models.adapters.azure import AzureOpenAIAdapter +from marsys.models.adapters.google import GoogleAdapter +from marsys.models.adapters.openai import ( + AsyncOpenAIAdapter, + OpenAIAdapter, + supports_explicit_prompt_cache, + supports_hosted_tool_search, +) +from marsys.models.adapters.openrouter import OpenRouterAdapter +from marsys.models.response_models import ToolCall + +MSGS = [{"role": "user", "content": "hi"}] +LEDGER = {"name": "ledger", "description": "the books: close a quarter, reopen a period"} +PEOPLE = {"name": "people", "description": "who works here and what they may see"} + + +def _tool(name, *, defer=False, label=None): + """A Chat-Completions tool dict the way a caller hands one to the builder.""" + tool = { + "type": "function", + "function": { + "name": name, + "description": f"{name} description", + "parameters": {"type": "object", "properties": {}}, + }, + } + if defer: + tool["defer_loading"] = True + if label is not None: + tool["namespace"] = label + return tool + + +@pytest.fixture +def openai(): + return OpenAIAdapter(model_name="gpt-5.6-terra", api_key="x", base_url="https://api.openai.com/v1") + + +@pytest.fixture +def azure(): + return AzureOpenAIAdapter( + model_name="gpt-5.6-terra", api_key="x", base_url="https://example.invalid/openai/v1" + ) + + +@pytest.fixture +def openrouter(): + return OpenRouterAdapter( + model_name="anthropic/claude-sonnet-4-6", api_key="x", + base_url="https://openrouter.ai/api/v1", + ) + + +@pytest.fixture +def google(): + return GoogleAdapter( + model_name="gemini-3.5-flash", api_key="x", + base_url="https://generativelanguage.googleapis.com", + ) + + +@pytest.fixture +def anthropic(): + return AnthropicAdapter( + model_name="claude-sonnet-4-6", api_key="x", base_url="https://api.anthropic.com/v1" + ) + + +@pytest.fixture +def anthropic_oauth(monkeypatch): + monkeypatch.setattr( + AnthropicOAuthAdapter, "_load_claude_credentials", + lambda self, path=None: {"access_token": "fake"}, + ) + return AnthropicOAuthAdapter(model_name="claude-sonnet-4-6", auto_refresh=False) + + +def _containers(tools): + return [t for t in tools if isinstance(t, dict) and t.get("type") == "namespace"] + + +def _by_name(tools, name): + return next((t for t in tools if isinstance(t, dict) and t.get("name") == name), None) + + +def _built_ins(tools): + return [t for t in tools if isinstance(t, dict) and t.get("type") == "tool_search"] + + +# --- the namespace container --------------------------------------------------------- + + +@pytest.mark.parametrize("leg", ["openai", "azure"]) +def test_labelled_tools_group_into_one_container_per_label(request, leg): + adapter = request.getfixturevalue(leg) + tools = adapter.format_request_payload( + MSGS, + tools=[ + _tool("ledger_close", defer=True, label=LEDGER), + _tool("people_list", defer=True, label=PEOPLE), + _tool("ledger_reopen", defer=True, label=LEDGER), + ], + )["tools"] + containers = _containers(tools) + assert [c["name"] for c in containers] == ["ledger", "people"] # first-appearance order + assert [f["name"] for f in containers[0]["tools"]] == ["ledger_close", "ledger_reopen"] + assert [f["name"] for f in containers[1]["tools"]] == ["people_list"] + + +def test_a_container_sits_where_its_first_member_appeared(openai): + tools = openai.format_request_payload( + MSGS, + tools=[ + _tool("always_on"), + _tool("ledger_close", defer=True, label=LEDGER), + _tool("also_always_on"), + _tool("ledger_reopen", defer=True, label=LEDGER), + ], + )["tools"] + kinds = [(t.get("type"), t.get("name")) for t in tools] + assert kinds[:3] == [ + ("function", "always_on"), + ("namespace", "ledger"), + ("function", "also_always_on"), + ] + + +def test_members_keep_the_flag_and_the_container_carries_none(openai): + tools = openai.format_request_payload( + MSGS, tools=[_tool("ledger_close", defer=True, label=LEDGER)] + )["tools"] + container = _containers(tools)[0] + assert "defer_loading" not in container + assert container["tools"][0]["defer_loading"] is True + + +def test_the_container_takes_its_name_and_description_from_the_label(openai): + container = _containers( + openai.format_request_payload( + MSGS, tools=[_tool("ledger_close", defer=True, label=LEDGER)] + )["tools"] + )[0] + assert container["name"] == LEDGER["name"] + assert container["description"] == LEDGER["description"] + + +def test_a_member_renders_as_a_flat_responses_function(openai): + member = _containers( + openai.format_request_payload( + MSGS, tools=[_tool("ledger_close", defer=True, label=LEDGER)] + )["tools"] + )[0]["tools"][0] + assert member["type"] == "function" + assert member["name"] == "ledger_close" + assert member["description"] == "ledger_close description" + assert member["parameters"] == {"type": "object", "properties": {}} + assert "function" not in member # flattened, not the externally-tagged shape + assert "namespace" not in member # the label named the container, it does not ride the member + + +def test_the_callers_tool_dicts_are_not_mutated(openai): + sent = [ + _tool("ledger_close", defer=True, label=LEDGER), + _tool("always_on"), + ] + before = [dict(t) for t in sent] + openai.format_request_payload(MSGS, tools=sent) + assert sent == before + assert sent[0]["namespace"] is LEDGER # the label object itself is untouched + + +def test_a_label_without_a_name_names_no_container(openai): + tools = openai.format_request_payload( + MSGS, tools=[_tool("ledger_close", defer=True, label={"description": "no name"})] + )["tools"] + assert _containers(tools) == [] + assert _by_name(tools, "ledger_close")["defer_loading"] is True + + +# --- the built-in, which the provider refuses the request without --------------------- + + +def test_the_built_in_is_added_when_a_container_member_is_deferred(openai): + tools = openai.format_request_payload( + MSGS, tools=[_tool("ledger_close", defer=True, label=LEDGER)] + )["tools"] + assert len(_built_ins(tools)) == 1 + + +def test_the_built_in_is_added_when_a_top_level_tool_is_deferred(openai): + tools = openai.format_request_payload(MSGS, tools=[_tool("ledger_close", defer=True)])["tools"] + assert len(_built_ins(tools)) == 1 + + +def test_a_caller_supplied_built_in_is_not_duplicated(openai): + tools = openai.format_request_payload( + MSGS, + tools=[_tool("ledger_close", defer=True, label=LEDGER), {"type": "tool_search"}], + )["tools"] + assert len(_built_ins(tools)) == 1 + + +def test_a_container_with_nothing_deferred_adds_no_built_in(openai): + tools = openai.format_request_payload( + MSGS, tools=[_tool("ledger_close", label=LEDGER)] + )["tools"] + assert _containers(tools)[0]["tools"][0].get("defer_loading") is None + assert _built_ins(tools) == [] + + +def test_nothing_labelled_and_nothing_deferred_renders_as_before(openai): + tools = openai.format_request_payload(MSGS, tools=[_tool("a"), _tool("b")])["tools"] + assert tools == [ + { + "type": "function", + "name": name, + "description": f"{name} description", + "parameters": {"type": "object", "properties": {}}, + } + for name in ("a", "b") + ] + + +# --- the legs that must never see the label ------------------------------------------ + + +def test_openrouter_strips_the_label_and_warns(openrouter): + with pytest.warns(Warning, match="namespace label"): + tools = openrouter.format_request_payload( + MSGS, tools=[_tool("ledger_close", defer=True, label=LEDGER)] + )["tools"] + assert all("namespace" not in t and "defer_loading" not in t for t in tools) + + +def test_openrouter_strips_a_label_that_arrives_without_the_flag(openrouter): + with pytest.warns(Warning, match="namespace label"): + tools = openrouter.format_request_payload( + MSGS, tools=[_tool("ledger_close", label=LEDGER)] + )["tools"] + assert all("namespace" not in t for t in tools) + + +def test_google_warns_and_sends_no_label(google): + with pytest.warns(Warning, match="namespace label"): + payload = google.format_request_payload( + MSGS, tools=[_tool("ledger_close", defer=True, label=LEDGER)] + ) + wire = repr(payload["tools"]) + assert "namespace" not in wire and "defer_loading" not in wire + + +@pytest.mark.parametrize("leg", ["anthropic", "anthropic_oauth"]) +def test_the_anthropic_builders_drop_the_label_and_keep_their_mapping(request, leg): + adapter = request.getfixturevalue(leg) + tools = adapter.format_request_payload( + MSGS, tools=[_tool("ledger_close", defer=True, label=LEDGER), _tool("core_tool")] + )["tools"] + assert all("namespace" not in t for t in tools) + assert _by_name(tools, "ledger_close")["defer_loading"] is True + assert "defer_loading" not in _by_name(tools, "core_tool") + assert any(t.get("type") == "tool_search_tool_regex_20251119" for t in tools) + + +# --- the parser ---------------------------------------------------------------------- + + +def _reply(output): + return {"id": "resp_1", "model": "gpt-5.6-terra", "output": output, "usage": {}} + + +SEARCH_CALL = {"type": "tool_search_call", "id": "ts_1", "status": "completed", "paths": ["ledger"]} +SEARCH_OUTPUT = { + "type": "tool_search_output", + "id": "tso_1", + "status": "completed", + "tools": [{"type": "namespace", "name": "ledger", "tools": [{"name": "ledger_close"}]}], +} + + +def test_the_search_items_are_kept_verbatim_and_in_order(openai): + resp = openai.harmonize_response( + _reply([ + {"type": "reasoning", "summary": ["thinking"]}, + SEARCH_CALL, + SEARCH_OUTPUT, + {"type": "function_call", "call_id": "c1", "name": "ledger_close", + "arguments": "{}", "namespace": "ledger"}, + ]), + request_start_time=0.0, + ) + assert resp.reasoning_details == [SEARCH_CALL, SEARCH_OUTPUT] + + +def test_reasoning_message_and_function_call_items_parse_as_before(openai): + resp = openai.harmonize_response( + _reply([ + {"type": "reasoning", "summary": ["because"]}, + {"type": "message", "role": "assistant", "status": "completed", + "content": [{"type": "output_text", "text": "done"}]}, + {"type": "function_call", "call_id": "c1", "name": "ledger_close", "arguments": "{}"}, + ]), + request_start_time=0.0, + ) + assert resp.reasoning == "because" + assert resp.content == "done" + assert resp.metadata.finish_reason == "stop" + assert resp.tool_calls[0].id == "c1" + assert resp.tool_calls[0].function == {"name": "ledger_close", "arguments": "{}"} + assert resp.reasoning_details is None + + +def test_the_calls_namespace_is_read_when_the_provider_sent_one(openai): + resp = openai.harmonize_response( + _reply([ + SEARCH_CALL, SEARCH_OUTPUT, + {"type": "function_call", "call_id": "c1", "name": "ledger_close", + "arguments": "{}", "namespace": "ledger"}, + ]), + request_start_time=0.0, + ) + assert resp.tool_calls[0].namespace == "ledger" + + +def test_a_call_without_a_namespace_has_none_and_serializes_as_before(openai): + resp = openai.harmonize_response( + _reply([ + {"type": "function_call", "call_id": "c1", "name": "ledger_close", "arguments": "{}"} + ]), + request_start_time=0.0, + ) + assert resp.tool_calls[0].namespace is None + assert resp.tool_calls[0].model_dump() == { + "id": "c1", "type": "function", + "function": {"name": "ledger_close", "arguments": "{}"}, + } + + +def test_a_namespaced_call_serializes_with_it(): + call = ToolCall( + id="c1", function={"name": "ledger_close", "arguments": "{}"}, namespace="ledger" + ) + assert call.model_dump()["namespace"] == "ledger" + + +# --- the replay ---------------------------------------------------------------------- + + +def _assistant_row(**extra): + row = { + "role": "assistant", + "content": "", + "tool_calls": [ + {"id": "c1", "type": "function", + "function": {"name": "ledger_close", "arguments": "{}"}}, + ], + } + row.update(extra) + return row + + +def test_the_search_items_are_replayed_ahead_of_the_rows_message_and_calls(openai): + payload = openai.format_request_payload([ + {"role": "user", "content": "close the quarter"}, + _assistant_row(content="on it", reasoning_details=[SEARCH_CALL, SEARCH_OUTPUT]), + {"role": "tool", "tool_call_id": "c1", "content": "{}"}, + ]) + types = [item.get("type") or item.get("role") for item in payload["input"]] + assert types == [ + "user", "tool_search_call", "tool_search_output", "assistant", + "function_call", "function_call_output", + ] + assert payload["input"][1] == SEARCH_CALL + assert payload["input"][2] == SEARCH_OUTPUT + + +def test_a_replayed_call_carries_its_namespace(openai): + row = _assistant_row(reasoning_details=[SEARCH_CALL, SEARCH_OUTPUT]) + row["tool_calls"][0]["namespace"] = "ledger" + payload = openai.format_request_payload([{"role": "user", "content": "go"}, row]) + call = next(i for i in payload["input"] if i.get("type") == "function_call") + assert call["namespace"] == "ledger" + + +def test_a_call_the_provider_sent_without_a_namespace_replays_without_one(openai): + payload = openai.format_request_payload([{"role": "user", "content": "go"}, _assistant_row()]) + call = next(i for i in payload["input"] if i.get("type") == "function_call") + assert call == { + "type": "function_call", "call_id": "c1", "name": "ledger_close", "arguments": "{}", + } + + +def test_another_legs_blocks_on_the_channel_are_not_replayed_here(openai): + payload = openai.format_request_payload([ + {"role": "user", "content": "go"}, + _assistant_row(reasoning_details=[ + {"type": "thinking", "thinking": "…", "signature": "sig"}, + SEARCH_CALL, + {"type": "text", "thought_signature": "sig"}, + ]), + ]) + types = [item.get("type") for item in payload["input"]] + assert types.count("tool_search_call") == 1 + assert "thinking" not in types and "text" not in types + + +def test_a_text_answering_reply_replays_its_search_items_too(openai): + """A reply that searches and then answers carries no tool call, and its loaded definitions + are exactly what the next request must not lose.""" + payload = openai.format_request_payload([ + {"role": "user", "content": "what do we owe?"}, + {"role": "assistant", "content": "about forty", "reasoning_details": [SEARCH_CALL, SEARCH_OUTPUT]}, + {"role": "user", "content": "and last quarter?"}, + ]) + types = [item.get("type") or item.get("role") for item in payload["input"]] + assert types == ["user", "tool_search_call", "tool_search_output", "assistant", "user"] + + +# --- the round trip ------------------------------------------------------------------ + + +def test_the_round_trip_grows_at_the_end_with_a_byte_identical_tools_array(openai): + tools = [ + _tool("workspace_read"), + _tool("ledger_close", defer=True, label=LEDGER), + _tool("ledger_reopen", defer=True, label=LEDGER), + ] + first = openai.format_request_payload( + [{"role": "user", "content": "close the quarter"}], tools=tools + ) + reply = openai.harmonize_response( + _reply([ + SEARCH_CALL, SEARCH_OUTPUT, + {"type": "function_call", "call_id": "c1", "name": "ledger_close", + "arguments": "{}", "namespace": "ledger"}, + ]), + request_start_time=0.0, + ) + second = openai.format_request_payload( + [ + {"role": "user", "content": "close the quarter"}, + {"role": "assistant", "content": "", + "tool_calls": [tc.model_dump() for tc in reply.tool_calls], + "reasoning_details": reply.reasoning_details}, + {"role": "tool", "tool_call_id": "c1", "content": "closed"}, + ], + tools=tools, + ) + assert second["input"][: len(first["input"])] == first["input"] + assert [i.get("type") for i in second["input"][len(first["input"]):]] == [ + "tool_search_call", "tool_search_output", "function_call", "function_call_output", + ] + assert second["input"][-2]["namespace"] == "ledger" + assert second["tools"] == first["tools"] + + +# --- the predicate ------------------------------------------------------------------- + + +@pytest.mark.parametrize("provider", ["openai", "azure"]) +@pytest.mark.parametrize("model", ["gpt-5.4", "gpt-5.5", "gpt-5.6-terra", "gpt-6"]) +def test_the_predicate_answers_true_for_the_served_legs(provider, model): + assert supports_hosted_tool_search(provider, model) is True + + +@pytest.mark.parametrize("model", ["gpt-5.3", "gpt-5", "gpt-4.1", "o3", "claude-sonnet-4-6", ""]) +def test_the_predicate_answers_false_below_the_floor(model): + assert supports_hosted_tool_search("openai", model) is False + + +@pytest.mark.parametrize("provider", ["openrouter", "google", "anthropic", None]) +def test_the_predicate_answers_false_off_the_served_providers(provider): + assert supports_hosted_tool_search(provider, "gpt-5.6-terra") is False + + +@pytest.mark.parametrize("model", ["gpt-5.4", "gpt-5.5"]) +def test_the_two_gates_keep_their_own_floors(model): + assert supports_hosted_tool_search("openai", model) is True + assert supports_explicit_prompt_cache(model) is False + + +def test_the_predicate_states_the_azure_deployment_caveat(): + assert "deployment" in (supports_hosted_tool_search.__doc__ or "") + + +# --- byte identity everywhere nothing is labelled and nothing deferred ---------------- + + +@pytest.mark.parametrize("adapter_type", [OpenAIAdapter, AsyncOpenAIAdapter]) +def test_a_plain_request_is_untouched_by_any_of_this(adapter_type): + adapter = adapter_type( + model_name="gpt-5.6-terra", api_key="x", base_url="https://api.openai.com/v1" + ) + payload = adapter.format_request_payload( + [ + {"role": "user", "content": "hi"}, + _assistant_row(content="calling"), + {"role": "tool", "tool_call_id": "c1", "content": "{}"}, + ], + tools=[_tool("a")], + ) + assert [i.get("type") or i.get("role") for i in payload["input"]] == [ + "user", "assistant", "function_call", "function_call_output", + ] + assert payload["tools"] == [{ + "type": "function", "name": "a", "description": "a description", + "parameters": {"type": "object", "properties": {}}, + }] From ab0c3440454c5634bcd05482a71cc37faeaa7239 Mon Sep 17 00:00:00 2001 From: rezaho Date: Sat, 12 Sep 2026 19:27:28 +0200 Subject: [PATCH 2/2] Read a model's generation once for both adapter gates MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The hosted-tool-search predicate and the explicit prompt-cache gate parsed the same model name with the same four lines each. They share one reader now, so the floor is the only thing that separates them — which is the only thing that should. --- src/marsys/models/adapters/openai.py | 29 +++++++++++++++------------- 1 file changed, 16 insertions(+), 13 deletions(-) diff --git a/src/marsys/models/adapters/openai.py b/src/marsys/models/adapters/openai.py index 7b2f9871..1ae5fa59 100644 --- a/src/marsys/models/adapters/openai.py +++ b/src/marsys/models/adapters/openai.py @@ -3,7 +3,7 @@ import re import time import warnings -from typing import Any, Callable, Collection, Dict, List, Optional +from typing import Any, Callable, Collection, Dict, List, Optional, Tuple from marsys.models.adapters.base import ( CACHE_EXEMPT_KEY, @@ -102,6 +102,17 @@ _GENERATION_RE = re.compile(r"^gpt-(\d+)(?:\.(\d+))?") +def _generation(model_lower: str) -> Optional[Tuple[int, int]]: + """``(major, minor)`` read off a model name, or None when the name is not GPT-shaped. + + One reading for every feature gated on a generation, so the gates differ in their FLOOR and + in nothing else — which is the only thing that should ever separate two of them.""" + match = _GENERATION_RE.match(model_lower or "") + if not match: + return None + return int(match.group(1)), int(match.group(2) or 0) + + # --- hosted tool search ------------------------------------------------------------- # # The provider runs the search itself: the request carries deferred functions, grouped in @@ -142,12 +153,8 @@ def supports_hosted_tool_search(provider: Optional[str], model_lower: str) -> bo """ if provider not in _HOSTED_TOOL_SEARCH_PROVIDERS: return False - match = _GENERATION_RE.match(model_lower or "") - if not match: - return False - major = int(match.group(1)) - minor = int(match.group(2) or 0) - return (major, minor) >= _HOSTED_TOOL_SEARCH_MIN_VERSION + generation = _generation(model_lower) + return generation is not None and generation >= _HOSTED_TOOL_SEARCH_MIN_VERSION def _namespace_label(tool: Any) -> Optional[Dict[str, Any]]: @@ -254,12 +261,8 @@ def supports_explicit_prompt_cache(model_lower: str) -> bool: mistake for a cost problem. A 5.6 model behind an older-shaped name simply keeps today's behaviour and pays today's price. """ - match = _GENERATION_RE.match(model_lower or "") - if not match: - return False - major = int(match.group(1)) - minor = int(match.group(2) or 0) - return (major, minor) >= _EXPLICIT_PROMPT_CACHE_MIN_VERSION + generation = _generation(model_lower) + return generation is not None and generation >= _EXPLICIT_PROMPT_CACHE_MIN_VERSION def _blocks_with_breakpoint(