Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
11 changes: 8 additions & 3 deletions src/marsys/models/adapters/google.py
Original file line number Diff line number Diff line change
Expand Up @@ -244,13 +244,18 @@ def format_request_payload(self, messages: List[Dict], **kwargs) -> Dict[str, An

# Add native function calling support
if kwargs.get("tools"):
if any(isinstance(t, dict) and t.get("defer_loading") for t in kwargs["tools"]):
if any(
isinstance(t, dict) and (t.get("defer_loading") or t.get("namespace"))
for t in kwargs["tools"]
):
# Google has no deferred-tool-loading feature; the rebuild below already drops the
# per-tool defer_loading flag, so warn (not a silent drop) and fall back to eager.
# per-tool defer_loading flag and the namespace label that rides beside it, so
# warn (not a silent drop) and fall back to eager loading with every tool flat.
import warnings
warnings.warn(
"Google models do not support deferred tool loading; the per-tool "
"defer_loading flag is ignored and all tools are loaded eagerly."
"defer_loading flag and its namespace label are ignored and all tools are "
"loaded eagerly."
)
# Convert OpenAI format tools to Google format
google_tools = []
Expand Down
232 changes: 186 additions & 46 deletions src/marsys/models/adapters/openai.py
Original file line number Diff line number Diff line change
Expand Up @@ -3,7 +3,7 @@
import re
import time
import warnings
from typing import Any, Callable, Collection, Dict, List, Optional
from typing import Any, Callable, Collection, Dict, List, Optional, Tuple

from marsys.models.adapters.base import (
CACHE_EXEMPT_KEY,
Expand Down Expand Up @@ -102,6 +102,154 @@
_GENERATION_RE = re.compile(r"^gpt-(\d+)(?:\.(\d+))?")


def _generation(model_lower: str) -> Optional[Tuple[int, int]]:
"""``(major, minor)`` read off a model name, or None when the name is not GPT-shaped.

One reading for every feature gated on a generation, so the gates differ in their FLOOR and
in nothing else — which is the only thing that should ever separate two of them."""
match = _GENERATION_RE.match(model_lower or "")
if not match:
return None
return int(match.group(1)), int(match.group(2) or 0)


# --- hosted tool search -------------------------------------------------------------
#
# The provider runs the search itself: the request carries deferred functions, grouped in
# namespace containers or standing alone, plus the `tool_search` built-in; the model searches,
# the provider injects the loaded definitions at the end of the context, and the model calls in
# the same reply.

# The generation that serves it. Its OWN floor, two generations below the explicit prompt-cache
# one beside it, because the two features shipped apart: one gate for both would either withhold
# the search from models that serve it or send cache fields to models that answer them with a 400.
_HOSTED_TOOL_SEARCH_MIN_VERSION = (5, 4)

# The providers whose endpoints were measured to serve it. The factory routes an unrecognized
# provider to this adapter, so a third-party OpenAI-compatible endpoint behind a GPT-5-shaped
# model name would otherwise be told it serves a feature nobody asked it about.
_HOSTED_TOOL_SEARCH_PROVIDERS = frozenset({"openai", "azure"})

# The reply items a hosted search emits: the call, and its output carrying the definitions the
# search loaded. Kept off the reply and replayed on the next request verbatim.
_TOOL_SEARCH_ITEM_TYPES = frozenset({"tool_search_call", "tool_search_output"})

# The key a caller puts on a Chat-Completions tool dict to say which container the rendered
# function belongs in: `{"name": ..., "description": ...}`. The grouping travels per tool rather
# than as a container the caller builds, so the legs that must not see it strip one key off a flat
# dict — the move they already make for `defer_loading` — instead of unwrapping a tool type they
# have no shape for.
NAMESPACE_LABEL_KEY = "namespace"


def supports_hosted_tool_search(provider: Optional[str], model_lower: str) -> bool:
"""Whether this leg serves the provider-run tool search over deferred functions.

Carries the same honesty caveat as the prompt-cache gate above, and it bites harder here: on
Azure the model field is an operator-chosen deployment label rather than a model name, so a
deployment named after a model it does not serve lies to this check — and the failure is a 400
on `tools.defer_loading` for every call rather than a price. Ask the deployment before turning
the feature on for it.
"""
if provider not in _HOSTED_TOOL_SEARCH_PROVIDERS:
return False
generation = _generation(model_lower)
return generation is not None and generation >= _HOSTED_TOOL_SEARCH_MIN_VERSION


def _namespace_label(tool: Any) -> Optional[Dict[str, Any]]:
"""The namespace label on a tool dict, or None. A label with no name names no container, so
it is no label at all."""
if not isinstance(tool, dict):
return None
label = tool.get(NAMESPACE_LABEL_KEY)
return label if isinstance(label, dict) and label.get("name") else None


def _convert_tools_for_responses(tools: Collection[Any]) -> List[Any]:
"""The Responses `tools` array for a caller's tool list.

Three jobs in one pass, because they read the same dicts. A Chat-Completions `function` tool
flattens to the internally-tagged Responses shape. A tool carrying a namespace label goes
inside a `namespace` container instead of standing at the top level — one container per
distinct label name, placed where its first member appeared, members in the order given, and
the container never carries the deferral flag its members do. And a request that defers
anything gains the hosted-search built-in: the provider refuses a deferred tool without it
("Deferred tools require tools.tool_search") whether the flag sat at the top level or inside a
container, so reading only the top level makes a namespace request a refusal rather than a
degradation.

The caller's dicts are never mutated — a converted function is a new dict and a container is
built here — so a tool array a caller holds frozen across a turn stays what it was.
"""
converted: List[Any] = []
containers: Dict[str, Dict[str, Any]] = {}
any_deferred = False
for tool in tools:
if not isinstance(tool, dict):
converted.append(tool)
continue
if tool.get("type") == "function" and "function" in tool:
# Convert from Chat Completions format (externally tagged)
func = tool["function"]
rendered: Any = {
"type": "function",
"name": func.get("name"),
"description": func.get("description"),
"parameters": func.get("parameters"),
# Note: strict is true by default in Responses API
}
if tool.get("defer_loading"):
rendered["defer_loading"] = True
any_deferred = True
else:
# Already in Responses API format or other tool type
rendered = (
{k: v for k, v in tool.items() if k != NAMESPACE_LABEL_KEY}
if NAMESPACE_LABEL_KEY in tool
else tool
)
if tool.get("defer_loading"):
any_deferred = True
label = _namespace_label(tool)
if label is None:
converted.append(rendered)
continue
container = containers.get(label["name"])
if container is None:
container = {
"type": "namespace",
"name": label["name"],
"description": label.get("description", ""),
"tools": [],
}
containers[label["name"]] = container
converted.append(container)
container["tools"].append(rendered)
if any_deferred and not any(
isinstance(t, dict) and t.get("type") == "tool_search" for t in converted
):
# Auto-add the Responses tool-search built-in so deferred tools are discoverable
# (gpt-5.4+). Suppressed if the caller supplied their own.
converted.append({"type": "tool_search"})
return converted


def _replayed_provider_items(msg: Dict[str, Any]) -> List[Dict[str, Any]]:
"""This assistant row's provider items that belong back in the Responses input array.

Type-discriminated, the way every leg reads this channel: the field is shared with the other
providers' opaque blocks (extended-thinking blocks, thought signatures) and each payload
builder re-emits only the types its own endpoint emitted. Anything else on it is another
leg's and is left where it is.
"""
return [
item
for item in (msg.get("reasoning_details") or [])
if isinstance(item, dict) and item.get("type") in _TOOL_SEARCH_ITEM_TYPES
]


def supports_explicit_prompt_cache(model_lower: str) -> bool:
"""Whether a model name is GPT-5.6 or later, the generation that serves the fields.

Expand All @@ -113,12 +261,8 @@ def supports_explicit_prompt_cache(model_lower: str) -> bool:
mistake for a cost problem. A 5.6 model behind an older-shaped name simply keeps
today's behaviour and pays today's price.
"""
match = _GENERATION_RE.match(model_lower or "")
if not match:
return False
major = int(match.group(1))
minor = int(match.group(2) or 0)
return (major, minor) >= _EXPLICIT_PROMPT_CACHE_MIN_VERSION
generation = _generation(model_lower)
return generation is not None and generation >= _EXPLICIT_PROMPT_CACHE_MIN_VERSION


def _blocks_with_breakpoint(
Expand Down Expand Up @@ -369,6 +513,16 @@ def format_request_payload(self, messages: List[Dict], **kwargs) -> Dict[str, An
role = msg.get("role")
item_start = len(converted_messages)

# The provider's own items from this reply go back first: a hosted search's call and
# its output, the output carrying the definitions the search loaded. Ahead of the
# row's own message and calls because that is the order the provider emitted them in,
# and because the definitions have to reach the model's context before the call that
# uses them. Dropping them instead costs the model a second search of the same family
# the next time it wants the tool, and keeps the loaded definitions out of the cached
# prefix for good.
if role == "assistant":
converted_messages.extend(_replayed_provider_items(msg))

# Handle assistant messages with tool_calls -> function_call items
if role == "assistant" and msg.get("tool_calls"):
# First add any text content as a message
Expand All @@ -381,12 +535,17 @@ def format_request_payload(self, messages: List[Dict], **kwargs) -> Dict[str, An
# Convert each tool_call to a function_call item
for tc in msg["tool_calls"]:
func = tc.get("function", {})
converted_messages.append({
call_item = {
"type": "function_call",
"call_id": tc.get("id"),
"name": func.get("name"),
"arguments": func.get("arguments", "{}")
})
}
# The container the provider's search loaded this function from, when it named
# one. The item that goes back is the provider's own, so it goes back whole.
if tc.get(NAMESPACE_LABEL_KEY):
call_item[NAMESPACE_LABEL_KEY] = tc[NAMESPACE_LABEL_KEY]
converted_messages.append(call_item)
# Handle tool role messages -> function_call_output items
elif role == "tool":
converted_messages.append({
Expand Down Expand Up @@ -497,44 +656,13 @@ def format_request_payload(self, messages: List[Dict], **kwargs) -> Dict[str, An

# Handle tools - Responses API uses flattened structure (internally tagged)
# Converts externally tagged format to internally tagged format.
# A per-tool ``defer_loading: true`` rides the Chat-Completions tool dict top-level
# (deferred tool loading); it maps onto the flat Responses tool and triggers the
# ``tool_search`` built-in so deferred tools are discovered on demand (their schemas stay
# out of the cached prefix). Nothing deferred → byte-identical to before.
# A per-tool ``defer_loading: true`` and a per-tool ``namespace`` label ride the
# Chat-Completions tool dict top-level; they map onto the flat Responses tool, its
# container, and the ``tool_search`` built-in, so deferred tools are discovered on demand
# and their schemas stay out of the cached prefix. Nothing deferred and nothing labelled →
# byte-identical to before.
if kwargs.get("tools"):
tools = kwargs["tools"]
converted_tools = []
any_deferred = False
for tool in tools:
if isinstance(tool, dict):
if tool.get("type") == "function" and "function" in tool:
# Convert from Chat Completions format (externally tagged)
func = tool["function"]
converted = {
"type": "function",
"name": func.get("name"),
"description": func.get("description"),
"parameters": func.get("parameters"),
# Note: strict is true by default in Responses API
}
if tool.get("defer_loading"):
converted["defer_loading"] = True
any_deferred = True
converted_tools.append(converted)
else:
# Already in Responses API format or other tool type
converted_tools.append(tool)
if isinstance(tool, dict) and tool.get("defer_loading"):
any_deferred = True
else:
converted_tools.append(tool)
if any_deferred and not any(
isinstance(t, dict) and t.get("type") == "tool_search" for t in converted_tools
):
# Auto-add the Responses tool-search built-in so deferred tools are discoverable
# (gpt-5.4+). Suppressed if the caller supplied their own.
converted_tools.append({"type": "tool_search"})
payload["tools"] = converted_tools
payload["tools"] = _convert_tools_for_responses(kwargs["tools"])

# Handle OpenAI reasoning (effort-based for all models via Responses API).
# An explicit `reasoning_effort` wins; failing that, a caller's thinking budget
Expand Down Expand Up @@ -743,6 +871,11 @@ def harmonize_response(
finish_reason = None
reasoning_data = None
tool_calls = []
# The hosted search's own output items, kept whole and in arrival order. They ride the
# opaque provider-items channel the other legs already use for the blocks their endpoints
# demand back verbatim; the next request replays them, which is what puts a loaded
# definition into the conversation body and, from the call after, into the cached prefix.
provider_items: List[Dict[str, Any]] = []

# Parse output array from Responses API
output_array = raw_response.get("output", [])
Expand Down Expand Up @@ -799,9 +932,15 @@ def harmonize_response(
"name": item.get("name", ""),
"arguments": item.get("arguments", "")
},
# Present only when the provider loaded this function out of a namespace.
namespace=item.get("namespace") or None,
)
)

# The hosted search's two items, kept verbatim for the replay (see ``provider_items``).
elif item_type in _TOOL_SEARCH_ITEM_TYPES:
provider_items.append(item)

# Responses API uses input_tokens/output_tokens (not prompt/
# completion_tokens) and nests reasoning in output_tokens_details.
# Fall back to chat-completions names for endpoint compat.
Expand Down Expand Up @@ -881,6 +1020,7 @@ def harmonize_response(
content=content,
tool_calls=tool_calls,
reasoning=reasoning_data,
reasoning_details=provider_items or None,
metadata=metadata,
)

Expand Down
17 changes: 11 additions & 6 deletions src/marsys/models/adapters/openrouter.py
Original file line number Diff line number Diff line change
Expand Up @@ -186,18 +186,23 @@ def format_request_payload(self, messages: List[Dict], **kwargs) -> Dict[str, An

if kwargs.get("tools"):
tools = kwargs["tools"]
if any(isinstance(t, dict) and t.get("defer_loading") for t in tools):
if any(
isinstance(t, dict) and (t.get("defer_loading") or t.get("namespace"))
for t in tools
):
# OpenRouter has no deferred-tool-loading feature and forwards `tools` verbatim,
# so a defer_loading marker would reach the wire (possible 400). Strip it and fall
# back to eager loading. Warn — a SILENT behavior change is what the additive
# contract forbids.
# so a defer_loading marker, or the namespace label that rides beside it to say
# which container a deferred function belongs in, would reach the wire (possible
# 400). Strip both and fall back to eager loading with every tool at the top
# level. Warn — a SILENT behavior change is what the additive contract forbids.
import warnings
warnings.warn(
"OpenRouter does not support deferred tool loading; the per-tool defer_loading "
"flag is stripped and all tools are loaded eagerly."
"flag and its namespace label are stripped and all tools are loaded eagerly."
)
tools = [
{k: v for k, v in t.items() if k != "defer_loading"} if isinstance(t, dict) else t
{k: v for k, v in t.items() if k not in ("defer_loading", "namespace")}
if isinstance(t, dict) else t
for t in tools
]
payload["tools"] = tools
Expand Down
Loading
Loading