Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
101 changes: 98 additions & 3 deletions src/marsys/models/adapters/openai.py
Original file line number Diff line number Diff line change
Expand Up @@ -99,6 +99,29 @@
# content.
_BREAKPOINT_ITEM_ROLES = frozenset({"system", "developer", "user"})


# --- reasoning summaries ------------------------------------------------------------
#
# The generation documented to serve a provider-authored reasoning summary. Azure's
# reasoning page marks the field on all twenty-one gpt-5.x deployments, carries no such
# row at all for the gpt-6 family, and splits the o-series three-for-six; only the
# gpt-5 generation is written down here because only it is uniformly documented, and
# opening it upward is a measurement rather than a guess.
_REASONING_SUMMARY_GENERATION = 5

# The providers whose endpoints were measured to serve the field, on the same reasoning
# as the prompt-cache set above: the factory routes an unrecognized provider to this
# adapter, so a third-party OpenAI-compatible endpoint behind a gpt-5-shaped model name
# would otherwise be asked for a field nobody has asked it for.
_REASONING_SUMMARY_PROVIDERS = frozenset({"openai", "azure"})

# The word sent when a served leg says nothing else. Measured against a live Azure
# resource over eleven streaming requests: on one deployment `detailed` came back with a
# summary three times out of three and `auto` twice out of four, at no measurable
# difference in reasoning or output tokens. Arrival is not guaranteed either way, which
# the provider documents, so a request that draws no summary is not a fault.
_DEFAULT_REASONING_SUMMARY = "detailed"

_GENERATION_RE = re.compile(r"^gpt-(\d+)(?:\.(\d+))?")


Expand Down Expand Up @@ -265,6 +288,45 @@ def supports_explicit_prompt_cache(model_lower: str) -> bool:
return generation is not None and generation >= _EXPLICIT_PROMPT_CACHE_MIN_VERSION


def supports_reasoning_summary(model_lower: str) -> bool:
"""Whether a model name is one the provider documents as serving a summary.

Closed at the gpt-5 generation, which is the whole of what the two provider pages
state: every gpt-5.x deployment is marked, the gpt-6 family has no such row, and the
o-series is marked on three of six names this regex cannot match anyway.

Reads the name the way the caching rule beside it does, and fails in the one
direction that is worth naming here. On Azure the name is an operator-chosen
deployment label, so a deployment renamed away from the model it serves stops being
asked for a summary, and the person watching then sees the generic cue for every
turn, forever, with no error anywhere to read. The pilots' deployments are named
`gpt-5.6-sol` and `gpt-5.6-terra`, which this matches. The other direction is loud
and cheap by comparison: a name shaped like the generation on a leg that does not
serve the field takes a 400 on its first call.
"""
generation = _generation(model_lower)
return generation is not None and generation[0] == _REASONING_SUMMARY_GENERATION


def _reasoning_parts_text(parts: List[Any]) -> str:
"""A reasoning item's ``summary`` or ``content`` list read as text, one part a line.

The wire part is an object, ``{"type": "summary_text", "text": ...}`` on a summary
and ``{"type": "reasoning_text", "text": ...}`` on the content list, so rendering a
part with ``str()`` puts a Python dict repr where the model's own words belong. A
plain string part is its own text and stays readable, which is the shape older
fixtures and re-hosted surfaces still hand over.
"""
texts = []
for part in parts:
if not part:
continue
text = part.get("text") if isinstance(part, dict) else part
if text:
texts.append(str(text))
return "\n".join(texts)


def _blocks_with_breakpoint(
blocks: List[Any],
) -> Optional[List[Any]]:
Expand Down Expand Up @@ -667,13 +729,20 @@ def format_request_payload(self, messages: List[Dict], **kwargs) -> Dict[str, An
# Handle OpenAI reasoning (effort-based for all models via Responses API).
# An explicit `reasoning_effort` wins; failing that, a caller's thinking budget
# selects the bucket, so the one knob this stack exposes reaches this leg too.
# The summary rides this object and never creates it: a request that asks for no
# thinking must not be told to show its thinking, and there is one creation site
# for `reasoning` on this builder.
reasoning_effort = kwargs.get("reasoning_effort")
if not reasoning_effort:
reasoning_effort = thinking_budget_to_effort(kwargs.get("thinking_budget"))
if reasoning_effort and reasoning_effort.lower() in ["minimal", "low", "medium", "high"]:
payload["reasoning"] = {
reasoning = {
"effort": self._served_effort(reasoning_effort.lower(), model_lower)
}
summary = self._reasoning_summary(model_lower, kwargs.get("reasoning_summary"))
if summary:
reasoning["summary"] = summary
payload["reasoning"] = reasoning

if kwargs.get("prompt_cache_key") is not None:
payload["prompt_cache_key"] = kwargs["prompt_cache_key"]
Expand Down Expand Up @@ -722,6 +791,7 @@ def format_request_payload(self, messages: List[Dict], **kwargs) -> Dict[str, An
"tool_choice",
"parallel_tool_calls",
"reasoning_effort", # Converted to reasoning.effort
"reasoning_summary", # Converted to reasoning.summary
# Streaming and logging
"stream",
"stream_options",
Expand Down Expand Up @@ -770,6 +840,31 @@ def _served_effort(self, effort: str, model_lower: str) -> str:
"""Allow re-hosted surfaces to override the shared model compatibility rule."""
return served_reasoning_effort(effort, model_lower)

def _reasoning_summary(self, model_lower: str, requested: Any) -> Optional[str]:
"""The summary word this request carries, or ``None`` for no ``summary`` field.

The caller's three states and the gate resolve together here, so the payload
sets the effort and the summary side by side and decides nothing in place.

An explicit ``reasoning_summary`` wins in both directions, because the gate
exists to protect the default from legs nobody measured and a caller who names a
word knows their own leg: a string is passed through untouched, and ``False``
sends nothing at all while leaving the effort exactly as it was. Absent, the
gate answers: the provider the factory stamped (falling back to this class's own
name, the way the caching gate beside it does), then the documented generation.
Returns the value rather than a verdict, following ``_served_effort``, so a
re-hosted surface whose table splits inside the generation overrides one method
instead of growing a second conditional in the builder.
"""
if requested is not None:
return requested or None
provider = getattr(self, "provider", None) or self._provider_name()
if provider not in _REASONING_SUMMARY_PROVIDERS:
return None
if not supports_reasoning_summary(model_lower):
return None
return _DEFAULT_REASONING_SUMMARY

def _supports_explicit_prompt_cache(self, model_lower: str) -> bool:
"""Whether this request may carry the explicit prompt-cache fields.

Expand Down Expand Up @@ -890,9 +985,9 @@ def harmonize_response(

# Prefer summary (key insights) over detailed content
if summary:
reasoning_data = "\n".join(str(s) for s in summary if s)
reasoning_data = _reasoning_parts_text(summary)
elif content_array:
reasoning_data = "\n".join(str(c) for c in content_array if c)
reasoning_data = _reasoning_parts_text(content_array)
else:
reasoning_data = None

Expand Down
15 changes: 13 additions & 2 deletions tests/models/test_adapter_streaming.py
Original file line number Diff line number Diff line change
Expand Up @@ -348,7 +348,10 @@ def test_foreign_reasoning_details_are_not_emitted_as_anthropic_blocks():
{"type": "response.completed", "response": {
"id": "resp_1", "model": "gpt-test",
"output": [
{"type": "reasoning", "content": [], "summary": ["Weighing options."]},
# The wire shape: a summary is a list of objects, not of strings, which is
# what the terminal object carries once the request asks for one.
{"type": "reasoning", "content": [],
"summary": [{"type": "summary_text", "text": "Weighing options."}]},
{"type": "message", "role": "assistant", "status": "completed",
"content": [{"type": "output_text", "text": "Hello world."}]},
],
Expand All @@ -368,7 +371,15 @@ def test_responses_accumulator_taps_and_captures_the_terminal_object():
("text_delta", "Hello "),
("text_delta", "world."),
]
assert acc.to_rest_response()["id"] == "resp_1"
rest = acc.to_rest_response()
assert rest["id"] == "resp_1"
# The rebuilt REST shape carries the terminal reasoning item as it arrived, so it is
# read back through the same harmonizer the non-streaming path uses: an accumulator
# that dropped or flattened the summary part would hand over a dict repr here.
harmonized = AsyncOpenAIAdapter(
model_name="gpt-test", api_key="k", base_url="https://api.openai.com/v1",
).harmonize_response(rest, request_start_time=0.0)
assert harmonized.reasoning == "Weighing options."


def test_responses_failed_event_is_terminal():
Expand Down
38 changes: 28 additions & 10 deletions tests/models/test_azure_openai_leg.py
Original file line number Diff line number Diff line change
Expand Up @@ -323,16 +323,21 @@ def test_a_thinking_budget_selects_a_reasoning_effort(budget, expected):
def test_the_configured_thinking_budget_reaches_this_leg():
"""The only deliberation knob this stack exposes is a token budget. Unmapped, a
caller's setting is inert and every call runs at the provider default (`medium`),
which is indistinguishable from the knob working."""
which is indistinguishable from the knob working.

The object carries a summary beside the effort because this deployment's generation
is documented to serve one and the request is the only place it can be asked for."""
payload = _azure().format_request_payload(MESSAGES, thinking_budget=32768)
assert payload["reasoning"] == {"effort": "high"}
assert payload["reasoning"] == {"effort": "high", "summary": "detailed"}


def test_an_explicit_effort_beats_the_budget():
"""The object carries a summary beside the effort for the same reason as above: this
deployment's generation is documented to serve one, whichever route set the effort."""
payload = _azure().format_request_payload(
MESSAGES, thinking_budget=32768, reasoning_effort="low"
)
assert payload["reasoning"] == {"effort": "low"}
assert payload["reasoning"] == {"effort": "low", "summary": "detailed"}


def test_thinking_off_sends_no_reasoning_block():
Expand All @@ -350,31 +355,44 @@ def test_the_smallest_budget_asks_for_an_effort_this_surface_actually_serves():
"""`minimal` is a 400 here — the endpoint's own reply lists none / low / medium /
high / xhigh / max — so the smallest bucket has to arrive as the smallest this
surface has. It must still ask for reasoning: a positive budget is "think a little",
and `none` would answer a question nobody asked."""
and `none` would answer a question nobody asked.

The summary rides along for the same reason as above; the two fields are resolved
independently and the effort substitution is what this case is about."""
payload = _azure().format_request_payload(MESSAGES, thinking_budget=512)
assert payload["reasoning"] == {"effort": "low"}
assert payload["reasoning"] == {"effort": "low", "summary": "detailed"}


def test_an_explicit_minimal_is_substituted_too():
"""The caller who names the effort outright is on the same endpoint as the one who
named a budget, and it rejects the value for both of them."""
named a budget, and it rejects the value for both of them.

The summary rides along here too, resolved independently of the substitution this
case is about."""
payload = _azure().format_request_payload(MESSAGES, reasoning_effort="minimal")
assert payload["reasoning"] == {"effort": "low"}
assert payload["reasoning"] == {"effort": "low", "summary": "detailed"}


def test_the_first_party_leg_still_sends_minimal():
"""The control, and the scope line: `minimal` is served by OpenAI's own endpoint and
the substitution above belongs to this re-hosting surface, not to the shared payload
builder. A run of this file that changed the first-party leg would be a silent change
to every OpenAI caller in the stack."""
to every OpenAI caller in the stack.

Both names carry a summary: the gate is the model generation and the provider, and a
hand-built `OpenAIAdapter` names OpenAI's own endpoint by its class. `gpt-5.6-codex`
is a fixture name here rather than a deployment either provider publishes, kept
because it is the one case that exercises the codex effort exception; it pins the
exception and the new default together on one row."""
payload = OpenAIAdapter(
model_name="gpt-5.6", api_key="k", base_url="https://api.openai.com/v1"
).format_request_payload(MESSAGES, thinking_budget=512)
assert payload["reasoning"] == {"effort": "minimal"}
assert payload["reasoning"] == {"effort": "minimal", "summary": "detailed"}
codex = OpenAIAdapter(
model_name="gpt-5.6-codex", api_key="k", base_url="https://api.openai.com/v1"
).format_request_payload(MESSAGES, thinking_budget=512)
assert codex["reasoning"] == {"effort": "low"} # the pre-existing codex exception
# the pre-existing codex effort exception, beside the summary the generation serves
assert codex["reasoning"] == {"effort": "low", "summary": "detailed"}


# --- the meter ---------------------------------------------------------------
Expand Down
6 changes: 6 additions & 0 deletions tests/models/test_openai_minimal_effort.py
Original file line number Diff line number Diff line change
Expand Up @@ -91,9 +91,15 @@ def post(url, *, json, **kwargs):
payload = captured[0]
if expected == "minimal" and model_name != "gpt-5":
expected = "low"
# The OAuth leg's object exists with or without an effort and always says `auto`.
# The api-key leg's object exists only when an effort resolved, and then carries the
# summary its model generation is documented to serve; every model name here is that
# generation, so the two rows that resolve no effort still send nothing at all.
reasoning = {"summary": "auto"} if oauth else {}
if expected is not None:
reasoning["effort"] = expected
if not oauth:
reasoning["summary"] = "detailed"
assert payload.get("reasoning", {}) == reasoning
assert payload["prompt_cache_key"] == "install:owner"
if oauth:
Expand Down
Loading
Loading