From 556ec5af56137050fdd06610ef3477748bfb9a08 Mon Sep 17 00:00:00 2001 From: rishu685 Date: Thu, 3 Sep 2026 16:15:19 +0530 Subject: [PATCH 1/3] feat(engine): support MCP structuredContent and artifacts in tool results (#134) - Introduce NormalizedToolResult domain model and normalize_tool_result function - Extract model-facing text alongside structuredContent and artifact metadata for MCP and local tools - Update ToolExecutionRecord and ToolExecutionManager to persist structured output across idempotent replays - Expose structured and artifact fields on ToolResultContext for transform_tool_result hooks - Add comprehensive test suite in test_tool_result_normalization.py --- .../in_memory_tool_execution_repository.py | 10 +- src/agent_engine/approvals/models.py | 1 + .../approvals/tool_execution_manager.py | 17 ++- .../approvals/tool_execution_repository.py | 9 +- .../engine/langgraph/tools/tool_invoker.py | 22 ++- .../engine/langgraph/tools/tool_result.py | 106 ++++++++++++++ src/agent_engine/runtime/hooks/models.py | 13 +- .../engine/test_tool_result_normalization.py | 135 ++++++++++++++++++ 8 files changed, 302 insertions(+), 11 deletions(-) create mode 100644 src/agent_engine/engine/langgraph/tools/tool_result.py create mode 100644 tests/engine/test_tool_result_normalization.py diff --git a/src/agent_engine/approvals/in_memory_tool_execution_repository.py b/src/agent_engine/approvals/in_memory_tool_execution_repository.py index 99c9cd64..eda81656 100644 --- a/src/agent_engine/approvals/in_memory_tool_execution_repository.py +++ b/src/agent_engine/approvals/in_memory_tool_execution_repository.py @@ -3,6 +3,7 @@ from __future__ import annotations import asyncio +from typing import Any from agent_engine.approvals.models import ToolExecutionRecord from agent_engine.approvals.tool_execution_repository import ToolExecutionRepository @@ -25,9 +26,16 @@ async def start(self, record: ToolExecutionRecord) -> tuple[ToolExecutionRecord, self._records[record.execution_id] = record return record, True - async def complete(self, execution_id: str, status: str, result: str) -> None: + async def complete( + self, + execution_id: str, + status: str, + result: str, + structured: Any | None = None, + ) -> None: async with self._lock: record = self._records.get(execution_id) if record is not None: record.status = status record.result = result + record.structured = structured diff --git a/src/agent_engine/approvals/models.py b/src/agent_engine/approvals/models.py index d6fca2c7..37bfb8a0 100644 --- a/src/agent_engine/approvals/models.py +++ b/src/agent_engine/approvals/models.py @@ -156,4 +156,5 @@ class ToolExecutionRecord: tool_name: str status: str = "started" # started | succeeded | failed result: str | None = None + structured: Any | None = None created_at: float = field(default_factory=time.time) diff --git a/src/agent_engine/approvals/tool_execution_manager.py b/src/agent_engine/approvals/tool_execution_manager.py index d1e13a11..bd4fb406 100644 --- a/src/agent_engine/approvals/tool_execution_manager.py +++ b/src/agent_engine/approvals/tool_execution_manager.py @@ -3,6 +3,7 @@ from __future__ import annotations import hashlib +from typing import Any from agent_engine.approvals.models import ToolExecutionRecord from agent_engine.approvals.tool_execution_repository import ToolExecutionRepository @@ -54,6 +55,18 @@ async def begin_execution( ) return created - async def finish_execution(self, execution_id: str, *, status: str, result: str) -> None: + async def finish_execution( + self, + execution_id: str, + *, + status: str, + result: str, + structured: Any | None = None, + ) -> None: if self._executions is not None: - await self._executions.complete(execution_id, status=status, result=result) + await self._executions.complete( + execution_id, + status=status, + result=result, + structured=structured, + ) diff --git a/src/agent_engine/approvals/tool_execution_repository.py b/src/agent_engine/approvals/tool_execution_repository.py index df5bf913..98eb12d9 100644 --- a/src/agent_engine/approvals/tool_execution_repository.py +++ b/src/agent_engine/approvals/tool_execution_repository.py @@ -3,6 +3,7 @@ from __future__ import annotations from abc import ABC, abstractmethod +from typing import Any from agent_engine.approvals.models import ToolExecutionRecord @@ -17,5 +18,11 @@ async def start(self, record: ToolExecutionRecord) -> tuple[ToolExecutionRecord, raise NotImplementedError @abstractmethod - async def complete(self, execution_id: str, status: str, result: str) -> None: + async def complete( + self, + execution_id: str, + status: str, + result: str, + structured: Any | None = None, + ) -> None: raise NotImplementedError diff --git a/src/agent_engine/engine/langgraph/tools/tool_invoker.py b/src/agent_engine/engine/langgraph/tools/tool_invoker.py index aeeb3f9b..f9550ba9 100644 --- a/src/agent_engine/engine/langgraph/tools/tool_invoker.py +++ b/src/agent_engine/engine/langgraph/tools/tool_invoker.py @@ -29,6 +29,10 @@ from agent_engine.core.spec import AgentSpec from agent_engine.engine.langgraph.tools.agent_tool_binding import AgentToolBinding from agent_engine.engine.langgraph.tools.tool_gate import DenyTool, ExecuteTool, ToolGate +from agent_engine.engine.langgraph.tools.tool_result import ( + NormalizedToolResult, + normalize_tool_result, +) from agent_engine.logging_config import log from agent_engine.runtime.execution_limiter import ( ExecutionLimitExceeded, @@ -321,25 +325,33 @@ async def _record_success(self, call: _ToolCall, result: object, latency_ms: int current_run_context.get(), self._call_context(call, "succeeded", latency_ms), ) - result_text = await self._transform_result(call, _extract_result_text(result), latency_ms) + normalized = normalize_tool_result(result) + result_text = await self._transform_result(call, normalized, latency_ms) await self._execution_manager.finish_execution( - call.exec_id, status="succeeded", result=result_text + call.exec_id, + status="succeeded", + result=result_text, + structured=normalized.structured, ) return result_text - async def _transform_result(self, call: _ToolCall, result_text: str, latency_ms: int) -> str: + async def _transform_result( + self, call: _ToolCall, normalized: NormalizedToolResult, latency_ms: int + ) -> str: """Let ``transform_tool_result`` hooks reshape the result (e.g. truncate oversized MCP output). The context is only built when such a hook exists. """ if not self._hook_manager.has("transform_tool_result"): - return result_text + return normalized.text transformed = await self._hook_manager.run_transform_tool_result( current_run_context.get(), ToolResultContext( agent_id=self._spec.id, tool_name=call.name, provider=call.provider, - result=result_text, + result=normalized.text, + structured=normalized.structured, + artifact=normalized.artifact, server_id=call.server_id, latency_ms=latency_ms, ), diff --git a/src/agent_engine/engine/langgraph/tools/tool_result.py b/src/agent_engine/engine/langgraph/tools/tool_result.py new file mode 100644 index 00000000..3220bbb6 --- /dev/null +++ b/src/agent_engine/engine/langgraph/tools/tool_result.py @@ -0,0 +1,106 @@ +"""Provider-agnostic normalized representation for tool execution outcomes.""" + +from __future__ import annotations + +import json +import logging +from dataclasses import asdict, dataclass, is_dataclass +from typing import Any + +from langchain_core.messages import ToolMessage + +logger = logging.getLogger(__name__) + + +def _as_text(content: Any) -> str: + if isinstance(content, str): + return content + if isinstance(content, list): + parts: list[str] = [] + for item in content: + if isinstance(item, str): + parts.append(item) + elif isinstance(item, dict) and "text" in item: + parts.append(str(item["text"])) + return "".join(parts) + return str(content) if content is not None else "" + + +@dataclass(frozen=True) +class NormalizedToolResult: + """Provider-agnostic representation of a tool execution outcome. + + Separates model-facing text content from machine-readable structured output + and optional artifact metadata, preserving MCP structuredContent without + collapsing everything to a plain string. + """ + + text: str + structured: Any | None = None + artifact: Any | None = None + + +def normalize_tool_result(raw_result: Any) -> NormalizedToolResult: + """Normalize a raw tool execution return value into a NormalizedToolResult. + + Supports: + - String output (local tools) + - LangChain ToolMessage (carrying MCP content blocks and artifact.structuredContent) + - Dicts with content / structuredContent + - Local Python objects / dicts / Pydantic models + """ + if isinstance(raw_result, NormalizedToolResult): + return raw_result + + if isinstance(raw_result, str): + return NormalizedToolResult(text=raw_result) + + text = "" + structured: Any | None = None + artifact: Any | None = None + + try: + if isinstance(raw_result, ToolMessage): + text = _as_text(raw_result.content) + artifact = getattr(raw_result, "artifact", None) + if artifact is not None: + if isinstance(artifact, dict) and "structuredContent" in artifact: + structured = artifact["structuredContent"] + elif hasattr(artifact, "structuredContent"): + structured = artifact.structuredContent + elif isinstance(artifact, (dict, list)): + structured = artifact + else: + structured = artifact + elif isinstance(raw_result, dict): + artifact = raw_result.get("artifact") + if "content" in raw_result: + text = _as_text(raw_result["content"]) + if "structuredContent" in raw_result: + structured = raw_result["structuredContent"] + elif "content" not in raw_result and "artifact" not in raw_result: + structured = raw_result + + if not text and structured is not None: + try: + text = json.dumps(structured) + except Exception: + text = str(structured) + elif is_dataclass(raw_result) and not isinstance(raw_result, type): + structured = asdict(raw_result) + text = json.dumps(structured) + elif hasattr(raw_result, "model_dump") and callable(raw_result.model_dump): + structured = raw_result.model_dump() + text = json.dumps(structured) + elif hasattr(raw_result, "dict") and callable(raw_result.dict): + structured = raw_result.dict() + text = json.dumps(structured) + else: + text = str(raw_result) if raw_result is not None else "" + except Exception as exc: + logger.warning("Failed to extract structured content from tool result: %s", exc) + text = _as_text(raw_result) if raw_result is not None else "" + structured = None + artifact = None + + return NormalizedToolResult(text=text, structured=structured, artifact=artifact) diff --git a/src/agent_engine/runtime/hooks/models.py b/src/agent_engine/runtime/hooks/models.py index 8df0146a..25afeceb 100644 --- a/src/agent_engine/runtime/hooks/models.py +++ b/src/agent_engine/runtime/hooks/models.py @@ -246,10 +246,19 @@ class ToolResultContext: tool_name: str provider: ToolProvider result: str + structured: Any | None = None + artifact: Any | None = None server_id: str | None = None latency_ms: int | None = None metadata: dict[str, object] = field(default_factory=dict) - def with_result(self, result: str) -> ToolResultContext: + def with_result( + self, + result: str, + structured: Any | None = None, + artifact: Any | None = None, + ) -> ToolResultContext: """Return a copy with ``result`` replaced (immutable update).""" - return dataclasses.replace(self, result=result) + st = structured if structured is not None else self.structured + art = artifact if artifact is not None else self.artifact + return dataclasses.replace(self, result=result, structured=st, artifact=art) diff --git a/tests/engine/test_tool_result_normalization.py b/tests/engine/test_tool_result_normalization.py new file mode 100644 index 00000000..35513e73 --- /dev/null +++ b/tests/engine/test_tool_result_normalization.py @@ -0,0 +1,135 @@ +"""Tests for tool result normalization and MCP structuredContent support.""" + +from __future__ import annotations + +from dataclasses import dataclass + +import pytest +from langchain_core.messages import ToolMessage + +from agent_engine.approvals.in_memory_tool_execution_repository import ( + InMemoryToolExecutionRepository, +) +from agent_engine.approvals.tool_execution_manager import ToolExecutionManager +from agent_engine.engine.langgraph.tools.tool_result import ( + normalize_tool_result, +) +from agent_engine.runtime.hooks.models import ToolResultContext + + +def test_normalize_plain_string_tool() -> None: + res = normalize_tool_result("hello world") + assert res.text == "hello world" + assert res.structured is None + assert res.artifact is None + + +def test_normalize_mcp_text_and_structured_content() -> None: + msg = ToolMessage( + content=[{"type": "text", "text": "Found 2 invoices"}], + tool_call_id="call_123", + artifact={ + "structuredContent": { + "invoices": [ + {"id": "INV-123", "amount": 500}, + {"id": "INV-456", "amount": 800}, + ] + } + }, + ) + res = normalize_tool_result(msg) + assert res.text == "Found 2 invoices" + assert res.structured == { + "invoices": [ + {"id": "INV-123", "amount": 500}, + {"id": "INV-456", "amount": 800}, + ] + } + assert res.artifact == { + "structuredContent": { + "invoices": [ + {"id": "INV-123", "amount": 500}, + {"id": "INV-456", "amount": 800}, + ] + } + } + + +def test_normalize_multiple_text_blocks_and_structured_content() -> None: + msg = ToolMessage( + content=[ + {"type": "text", "text": "Header line.\n"}, + {"type": "text", "text": "Detail line."}, + ], + tool_call_id="call_456", + artifact={"structuredContent": {"count": 42}}, + ) + res = normalize_tool_result(msg) + assert res.text == "Header line.\nDetail line." + assert res.structured == {"count": 42} + + +def test_normalize_structured_only_result() -> None: + data = {"structuredContent": {"balance": 1250}} + res = normalize_tool_result(data) + assert res.structured == {"balance": 1250} + assert "1250" in res.text + + +def test_normalize_local_structured_dict() -> None: + local_data = {"status": "ok", "items": [1, 2, 3]} + res = normalize_tool_result(local_data) + assert res.structured == {"status": "ok", "items": [1, 2, 3]} + assert '{"status": "ok", "items": [1, 2, 3]}' in res.text + + +@dataclass +class CustomToolOutput: + status: str + count: int + + +def test_normalize_local_dataclass() -> None: + output = CustomToolOutput(status="success", count=5) + res = normalize_tool_result(output) + assert res.structured == {"status": "success", "count": 5} + assert "success" in res.text + + +@pytest.mark.asyncio +async def test_execution_manager_persists_and_restores_structured_result() -> None: + repo = InMemoryToolExecutionRepository() + manager = ToolExecutionManager(execution_repository=repo) + + exec_id = "exec_test_123" + await manager.begin_execution( + exec_id, tool_call_id="tc_1", run_id="run_1", tool_name="test_tool" + ) + await manager.finish_execution( + exec_id, + status="succeeded", + result="Found items", + structured={"items": ["a", "b"]}, + ) + + record = await manager.already_executed(exec_id) + assert record is not None + assert record.result == "Found items" + assert record.structured == {"items": ["a", "b"]} + + +def test_tool_result_context_with_result_preserves_structured() -> None: + ctx = ToolResultContext( + agent_id="a1", + tool_name="t1", + provider="mcp", + result="Long raw output text", + structured={"data": [1, 2, 3]}, + artifact={"meta": "v1"}, + ) + + # Truncate text + updated = ctx.with_result("Truncated text") + assert updated.result == "Truncated text" + assert updated.structured == {"data": [1, 2, 3]} + assert updated.artifact == {"meta": "v1"} From 6889709e3a0bc85ec9a4731846e61e7fbf4b45eb Mon Sep 17 00:00:00 2001 From: rishu685 Date: Fri, 4 Sep 2026 21:52:42 +0530 Subject: [PATCH 2/3] fix(tools): address PR #136 review feedback on replay, fallbacks, and hooks --- .../in_memory_tool_execution_repository.py | 2 + src/agent_engine/approvals/models.py | 1 + .../approvals/tool_execution_manager.py | 2 + .../approvals/tool_execution_repository.py | 1 + .../engine/langgraph/tools/tool_invoker.py | 29 ++++++++++----- .../engine/langgraph/tools/tool_result.py | 27 ++++++-------- src/agent_engine/runtime/hooks/models.py | 11 ++++-- .../engine/test_tool_result_normalization.py | 37 +++++++++++++++++-- 8 files changed, 78 insertions(+), 32 deletions(-) diff --git a/src/agent_engine/approvals/in_memory_tool_execution_repository.py b/src/agent_engine/approvals/in_memory_tool_execution_repository.py index eda81656..bc0eced6 100644 --- a/src/agent_engine/approvals/in_memory_tool_execution_repository.py +++ b/src/agent_engine/approvals/in_memory_tool_execution_repository.py @@ -32,6 +32,7 @@ async def complete( status: str, result: str, structured: Any | None = None, + artifact: Any | None = None, ) -> None: async with self._lock: record = self._records.get(execution_id) @@ -39,3 +40,4 @@ async def complete( record.status = status record.result = result record.structured = structured + record.artifact = artifact diff --git a/src/agent_engine/approvals/models.py b/src/agent_engine/approvals/models.py index 37bfb8a0..ab585064 100644 --- a/src/agent_engine/approvals/models.py +++ b/src/agent_engine/approvals/models.py @@ -157,4 +157,5 @@ class ToolExecutionRecord: status: str = "started" # started | succeeded | failed result: str | None = None structured: Any | None = None + artifact: Any | None = None created_at: float = field(default_factory=time.time) diff --git a/src/agent_engine/approvals/tool_execution_manager.py b/src/agent_engine/approvals/tool_execution_manager.py index bd4fb406..48a1bbbc 100644 --- a/src/agent_engine/approvals/tool_execution_manager.py +++ b/src/agent_engine/approvals/tool_execution_manager.py @@ -62,6 +62,7 @@ async def finish_execution( status: str, result: str, structured: Any | None = None, + artifact: Any | None = None, ) -> None: if self._executions is not None: await self._executions.complete( @@ -69,4 +70,5 @@ async def finish_execution( status=status, result=result, structured=structured, + artifact=artifact, ) diff --git a/src/agent_engine/approvals/tool_execution_repository.py b/src/agent_engine/approvals/tool_execution_repository.py index 98eb12d9..79eb0d2b 100644 --- a/src/agent_engine/approvals/tool_execution_repository.py +++ b/src/agent_engine/approvals/tool_execution_repository.py @@ -24,5 +24,6 @@ async def complete( status: str, result: str, structured: Any | None = None, + artifact: Any | None = None, ) -> None: raise NotImplementedError diff --git a/src/agent_engine/engine/langgraph/tools/tool_invoker.py b/src/agent_engine/engine/langgraph/tools/tool_invoker.py index f9550ba9..0386a224 100644 --- a/src/agent_engine/engine/langgraph/tools/tool_invoker.py +++ b/src/agent_engine/engine/langgraph/tools/tool_invoker.py @@ -189,7 +189,7 @@ async def invoke(self, tc: dict[str, Any]) -> str: cached = await self._cached_result(call) if cached is not None: - return cached + return cached.text return await self._execute(call) @@ -239,7 +239,7 @@ def _enforce_limits(self, call: _ToolCall) -> str | None: return blocked_message(exc) return None - async def _cached_result(self, call: _ToolCall) -> str | None: + async def _cached_result(self, call: _ToolCall) -> NormalizedToolResult | None: """Return a prior successful result for this exact call, if any. Guards against a second side effect when a graph re-entry after resume @@ -260,7 +260,11 @@ async def _cached_result(self, call: _ToolCall) -> str | None: execution_id=call.exec_id, ) await self._usage.record_success(call.identity) - return cached.result + return NormalizedToolResult( + text=cached.result, + structured=cached.structured, + artifact=getattr(cached, "artifact", None), + ) async def _execute(self, call: _ToolCall) -> str: """Run the provider exactly once, wrapped in the idempotency ledger and the @@ -326,23 +330,24 @@ async def _record_success(self, call: _ToolCall, result: object, latency_ms: int self._call_context(call, "succeeded", latency_ms), ) normalized = normalize_tool_result(result) - result_text = await self._transform_result(call, normalized, latency_ms) + final_result = await self._transform_result(call, normalized, latency_ms) await self._execution_manager.finish_execution( call.exec_id, status="succeeded", - result=result_text, - structured=normalized.structured, + result=final_result.text, + structured=final_result.structured, + artifact=final_result.artifact, ) - return result_text + return final_result.text async def _transform_result( self, call: _ToolCall, normalized: NormalizedToolResult, latency_ms: int - ) -> str: + ) -> NormalizedToolResult: """Let ``transform_tool_result`` hooks reshape the result (e.g. truncate oversized MCP output). The context is only built when such a hook exists. """ if not self._hook_manager.has("transform_tool_result"): - return normalized.text + return normalized transformed = await self._hook_manager.run_transform_tool_result( current_run_context.get(), ToolResultContext( @@ -356,7 +361,11 @@ async def _transform_result( latency_ms=latency_ms, ), ) - return transformed.result + return NormalizedToolResult( + text=transformed.result, + structured=transformed.structured, + artifact=transformed.artifact, + ) def _call_context( self, diff --git a/src/agent_engine/engine/langgraph/tools/tool_result.py b/src/agent_engine/engine/langgraph/tools/tool_result.py index 3220bbb6..1f31933a 100644 --- a/src/agent_engine/engine/langgraph/tools/tool_result.py +++ b/src/agent_engine/engine/langgraph/tools/tool_result.py @@ -68,39 +68,36 @@ def normalize_tool_result(raw_result: Any) -> NormalizedToolResult: structured = artifact["structuredContent"] elif hasattr(artifact, "structuredContent"): structured = artifact.structuredContent - elif isinstance(artifact, (dict, list)): - structured = artifact - else: - structured = artifact elif isinstance(raw_result, dict): artifact = raw_result.get("artifact") if "content" in raw_result: text = _as_text(raw_result["content"]) if "structuredContent" in raw_result: structured = raw_result["structuredContent"] + elif isinstance(artifact, dict) and "structuredContent" in artifact: + structured = artifact["structuredContent"] elif "content" not in raw_result and "artifact" not in raw_result: structured = raw_result - - if not text and structured is not None: - try: - text = json.dumps(structured) - except Exception: - text = str(structured) elif is_dataclass(raw_result) and not isinstance(raw_result, type): structured = asdict(raw_result) - text = json.dumps(structured) elif hasattr(raw_result, "model_dump") and callable(raw_result.model_dump): structured = raw_result.model_dump() - text = json.dumps(structured) elif hasattr(raw_result, "dict") and callable(raw_result.dict): structured = raw_result.dict() - text = json.dumps(structured) + elif isinstance(raw_result, (bytes, bytearray)): + text = raw_result.decode("utf-8", errors="replace") else: - text = str(raw_result) if raw_result is not None else "" + text = "" except Exception as exc: logger.warning("Failed to extract structured content from tool result: %s", exc) - text = _as_text(raw_result) if raw_result is not None else "" + text = "" structured = None artifact = None + if not text.strip() and structured is not None: + try: + text = json.dumps(structured, default=str) + except Exception: + text = "" + return NormalizedToolResult(text=text, structured=structured, artifact=artifact) diff --git a/src/agent_engine/runtime/hooks/models.py b/src/agent_engine/runtime/hooks/models.py index 25afeceb..9638c831 100644 --- a/src/agent_engine/runtime/hooks/models.py +++ b/src/agent_engine/runtime/hooks/models.py @@ -228,6 +228,9 @@ class ToolCallContext: metadata: dict[str, object] = field(default_factory=dict) +_UNSET = object() + + @dataclass(frozen=True) class ToolResultContext: """A successful tool call's result, passed to ``transform_tool_result`` hooks @@ -255,10 +258,10 @@ class ToolResultContext: def with_result( self, result: str, - structured: Any | None = None, - artifact: Any | None = None, + structured: Any | None = _UNSET, + artifact: Any | None = _UNSET, ) -> ToolResultContext: """Return a copy with ``result`` replaced (immutable update).""" - st = structured if structured is not None else self.structured - art = artifact if artifact is not None else self.artifact + st = self.structured if structured is _UNSET else structured + art = self.artifact if artifact is _UNSET else artifact return dataclasses.replace(self, result=result, structured=st, artifact=art) diff --git a/tests/engine/test_tool_result_normalization.py b/tests/engine/test_tool_result_normalization.py index 35513e73..c0f99aa6 100644 --- a/tests/engine/test_tool_result_normalization.py +++ b/tests/engine/test_tool_result_normalization.py @@ -97,7 +97,7 @@ def test_normalize_local_dataclass() -> None: @pytest.mark.asyncio -async def test_execution_manager_persists_and_restores_structured_result() -> None: +async def test_execution_manager_persists_and_restores_structured_and_artifact() -> None: repo = InMemoryToolExecutionRepository() manager = ToolExecutionManager(execution_repository=repo) @@ -110,15 +110,40 @@ async def test_execution_manager_persists_and_restores_structured_result() -> No status="succeeded", result="Found items", structured={"items": ["a", "b"]}, + artifact={"meta": "v1"}, ) record = await manager.already_executed(exec_id) assert record is not None assert record.result == "Found items" assert record.structured == {"items": ["a", "b"]} + assert record.artifact == {"meta": "v1"} + + +def test_normalize_structured_only_tool_message() -> None: + msg = ToolMessage( + content=[], + tool_call_id="call_structured_only", + artifact={"structuredContent": {"balance": 1250}}, + ) + res = normalize_tool_result(msg) + assert res.structured == {"balance": 1250} + assert res.text == '{"balance": 1250}' -def test_tool_result_context_with_result_preserves_structured() -> None: +def test_normalize_auxiliary_artifact_not_conflated_with_structured() -> None: + msg = ToolMessage( + content="Generated report PDF", + tool_call_id="call_aux_artifact", + artifact={"file_id": "abc_123", "mime_type": "application/pdf"}, + ) + res = normalize_tool_result(msg) + assert res.text == "Generated report PDF" + assert res.structured is None + assert res.artifact == {"file_id": "abc_123", "mime_type": "application/pdf"} + + +def test_tool_result_context_with_result_preserves_and_clears_structured() -> None: ctx = ToolResultContext( agent_id="a1", tool_name="t1", @@ -128,8 +153,14 @@ def test_tool_result_context_with_result_preserves_structured() -> None: artifact={"meta": "v1"}, ) - # Truncate text + # Truncate text while preserving structured and artifact updated = ctx.with_result("Truncated text") assert updated.result == "Truncated text" assert updated.structured == {"data": [1, 2, 3]} assert updated.artifact == {"meta": "v1"} + + # Explicitly clear structured and artifact + cleared = ctx.with_result("Cleared text", structured=None, artifact=None) + assert cleared.result == "Cleared text" + assert cleared.structured is None + assert cleared.artifact is None From 939f2b88cfd8c8d256ef6e6e339739fa7c065c0e Mon Sep 17 00:00:00 2001 From: rishu685 Date: Sat, 5 Sep 2026 11:18:06 +0530 Subject: [PATCH 3/3] style: format models.py with ruff format --- .../approvals/tool_execution_repository.py | 1 - src/agent_engine/runtime/hooks/models.py | 4 +-- .../engine/test_tool_result_normalization.py | 25 +++++++++++-------- 3 files changed, 16 insertions(+), 14 deletions(-) diff --git a/src/agent_engine/approvals/tool_execution_repository.py b/src/agent_engine/approvals/tool_execution_repository.py index 96151510..d7eefb59 100644 --- a/src/agent_engine/approvals/tool_execution_repository.py +++ b/src/agent_engine/approvals/tool_execution_repository.py @@ -3,7 +3,6 @@ from __future__ import annotations from abc import ABC, abstractmethod -from typing import Any from agent_engine.approvals.models import ToolExecutionRecord, ToolExecutionStatus from agent_engine.runtime.tool_results import PersistedToolResult diff --git a/src/agent_engine/runtime/hooks/models.py b/src/agent_engine/runtime/hooks/models.py index c3b9ccad..014bb0d1 100644 --- a/src/agent_engine/runtime/hooks/models.py +++ b/src/agent_engine/runtime/hooks/models.py @@ -272,9 +272,7 @@ def with_result( """Return a copy with ``result`` replaced (immutable update).""" st = self.structured_result if structured is _UNSET else structured art = self.artifact if artifact is _UNSET else artifact - return dataclasses.replace( - self, result=result, structured_result=st, artifact=art - ) + return dataclasses.replace(self, result=result, structured_result=st, artifact=art) def with_structured_result(self, structured_result: JsonValue | None) -> ToolResultContext: """Return a copy with the machine-readable result replaced.""" diff --git a/tests/engine/test_tool_result_normalization.py b/tests/engine/test_tool_result_normalization.py index c0f99aa6..3f6027b9 100644 --- a/tests/engine/test_tool_result_normalization.py +++ b/tests/engine/test_tool_result_normalization.py @@ -10,11 +10,13 @@ from agent_engine.approvals.in_memory_tool_execution_repository import ( InMemoryToolExecutionRepository, ) +from agent_engine.approvals.models import ToolExecutionStatus from agent_engine.approvals.tool_execution_manager import ToolExecutionManager from agent_engine.engine.langgraph.tools.tool_result import ( normalize_tool_result, ) from agent_engine.runtime.hooks.models import ToolResultContext +from agent_engine.runtime.tool_results import NormalizedToolResult def test_normalize_plain_string_tool() -> None: @@ -105,19 +107,22 @@ async def test_execution_manager_persists_and_restores_structured_and_artifact() await manager.begin_execution( exec_id, tool_call_id="tc_1", run_id="run_1", tool_name="test_tool" ) - await manager.finish_execution( - exec_id, - status="succeeded", - result="Found items", + result = NormalizedToolResult( + text="Found items", structured={"items": ["a", "b"]}, artifact={"meta": "v1"}, ) + await manager.finish_execution( + exec_id, + status=ToolExecutionStatus.SUCCEEDED, + result=result, + ) - record = await manager.already_executed(exec_id) - assert record is not None - assert record.result == "Found items" - assert record.structured == {"items": ["a", "b"]} - assert record.artifact == {"meta": "v1"} + restored = await manager.restored_result(exec_id) + assert restored is not None + assert restored.text == "Found items" + assert restored.structured == {"items": ["a", "b"]} + assert restored.artifact == {"meta": "v1"} def test_normalize_structured_only_tool_message() -> None: @@ -149,7 +154,7 @@ def test_tool_result_context_with_result_preserves_and_clears_structured() -> No tool_name="t1", provider="mcp", result="Long raw output text", - structured={"data": [1, 2, 3]}, + structured_result={"data": [1, 2, 3]}, artifact={"meta": "v1"}, )