From f049cb58c3cd5ed31b9a4f1bbb757e73ddd8b716 Mon Sep 17 00:00:00 2001 From: Manohar Paturi <186662190+ManoharPaturi@users.noreply.github.com> Date: Thu, 24 Sep 2026 21:01:15 +0530 Subject: [PATCH 1/4] Python: filter non-assistant messages from workflow agent responses When running workflows converted to an agent via workflow.as_agent(), WorkflowAgent._convert_workflow_events_to_agent_response and _convert_workflow_event_to_agent_response_updates forwarded messages without checking their role. When underlying executors or orchestrations such as GroupChat yield conversation history containing user inputs, system instructions, or tool outputs, these non-assistant messages were re-emitted as caller-facing agent responses and streaming chunks. This change filters out non-assistant messages across AgentResponse, list[Message], Message, and AgentResponseUpdate workflow outputs so that WorkflowAgent exclusively returns assistant output. If an event contains only non-assistant messages, it is safely excluded from agent output without altering internal workflow execution. Signed-off-by: Manohar Paturi <186662190+ManoharPaturi@users.noreply.github.com> --- .../core/agent_framework/_workflows/_agent.py | 50 +++-- .../tests/workflow/test_workflow_agent.py | 194 +++++++++++++++++- 2 files changed, 227 insertions(+), 17 deletions(-) diff --git a/python/packages/core/agent_framework/_workflows/_agent.py b/python/packages/core/agent_framework/_workflows/_agent.py index f4aa653310d..82293770a39 100644 --- a/python/packages/core/agent_framework/_workflows/_agent.py +++ b/python/packages/core/agent_framework/_workflows/_agent.py @@ -587,23 +587,32 @@ def _convert_workflow_events_to_agent_response( ) if isinstance(data, AgentResponse): - messages.extend(data.messages) - raw_representations.append(data.raw_representation) - merged_usage = add_usage_details(merged_usage, data.usage_details) - latest_created_at = ( - data.created_at - if not latest_created_at - else max(latest_created_at, data.created_at) - if data.created_at - else latest_created_at - ) + # Filter to only assistant messages — system, tool, and user messages + # are intentionally excluded. System prompts and tool results are + # internal workflow artifacts; user messages would be re-emitted + # (e.g., from GroupChat orchestrators that include full conversation history). + assistant_messages = [msg for msg in data.messages if msg.role == "assistant"] + if assistant_messages: + messages.extend(assistant_messages) + raw_representations.append(data.raw_representation) + merged_usage = add_usage_details(merged_usage, data.usage_details) + latest_created_at = ( + data.created_at + if not latest_created_at + else max(latest_created_at, data.created_at) + if data.created_at + else latest_created_at + ) elif isinstance(data, Message): - messages.append(data) - raw_representations.append(data.raw_representation) + if data.role == "assistant": + messages.append(data) + raw_representations.append(data.raw_representation) elif is_instance_of(data, list[Message]): chat_messages = cast(list[Message], data) - messages.extend(chat_messages) - raw_representations.append(data) + assistant_messages = [msg for msg in chat_messages if msg.role == "assistant"] + if assistant_messages: + messages.extend(assistant_messages) + raw_representations.append(data) else: contents = self._extract_contents(data) if not contents: @@ -654,6 +663,9 @@ def _convert_workflow_event_to_agent_response_updates( executor_id = event.executor_id if isinstance(data, AgentResponseUpdate): + # Filter out non-assistant updates (e.g. user input echoed back) + if data.role is not None and data.role != "assistant": + return [] # Construct a fresh AgentResponseUpdate so we don't mutate a payload # that AgentExecutor still holds a reference to in its `updates` list. return [ @@ -676,9 +688,11 @@ def _convert_workflow_event_to_agent_response_updates( ) ] if isinstance(data, AgentResponse): - # Convert each message in AgentResponse to an AgentResponseUpdate + # Convert each assistant message in AgentResponse to an AgentResponseUpdate updates: list[AgentResponseUpdate] = [] for msg in data.messages: + if msg.role != "assistant": + continue updates.append( AgentResponseUpdate( contents=list(msg.contents), @@ -698,6 +712,8 @@ def _convert_workflow_event_to_agent_response_updates( updates[-1].additional_properties = dict(data.additional_properties) return updates if isinstance(data, Message): + if data.role != "assistant": + return [] return [ AgentResponseUpdate( contents=list(data.contents), @@ -710,10 +726,12 @@ def _convert_workflow_event_to_agent_response_updates( ) ] if is_instance_of(data, list[Message]): - # Convert each Message to an AgentResponseUpdate + # Convert each assistant Message to an AgentResponseUpdate chat_messages = cast(list[Message], data) updates = [] for msg in chat_messages: + if msg.role != "assistant": + continue updates.append( AgentResponseUpdate( contents=list(msg.contents), diff --git a/python/packages/core/tests/workflow/test_workflow_agent.py b/python/packages/core/tests/workflow/test_workflow_agent.py index 0ef70ca2f6e..a09d8f52ecc 100644 --- a/python/packages/core/tests/workflow/test_workflow_agent.py +++ b/python/packages/core/tests/workflow/test_workflow_agent.py @@ -1105,7 +1105,7 @@ async def test_workflow_as_agent_yield_output_with_list_of_chat_messages(self) - async def list_yielding_executor(messages: list[Message], ctx: WorkflowContext[Never, list[Message]]) -> None: # type: ignore[valid-type] # Yield a list of Messages (as SequentialBuilder does) msg_list = [ - Message(role="user", contents=["first message"]), + Message(role="assistant", contents=["first message"]), Message(role="assistant", contents=["second message"]), Message( role="assistant", @@ -2597,3 +2597,195 @@ async def start(messages: list[Message], ctx: WorkflowContext[AgentExecutorReque pending = await workflow._runner_context.get_pending_request_info_events() # The agent's approval id is used as the workflow's pending request id. assert list(pending.keys()) == [approval_id] + + async def test_workflow_as_agent_filters_non_assistant_messages_from_agent_response(self) -> None: + """Verify WorkflowAgent filters user, system, and tool messages from AgentResponse.""" + + @executor + async def mixed_agent_response_executor( + messages: list[Message], ctx: WorkflowContext[Never, AgentResponse] + ) -> None: + response = AgentResponse( + messages=[ + Message(role="system", contents=["System instructions"]), + Message(role="user", contents=["User input question"]), + Message(role="assistant", contents=["Assistant answer"], author_name="Teacher"), + Message(role="tool", contents=["Tool execution result"]), + ] + ) + await ctx.yield_output(response) + + workflow = WorkflowBuilder(start_executor=mixed_agent_response_executor).build() + agent = workflow.as_agent("mixed-response-agent") + + # Test streaming path + updates: list[AgentResponseUpdate] = [] + async for chunk in agent.run("hello", stream=True): + updates.append(chunk) + + assert len(updates) == 1 + assert updates[0].role == "assistant" + assert updates[0].text == "Assistant answer" + assert updates[0].author_name == "Teacher" + + # Test non-streaming path + result = await agent.run("hello") + assert len(result.messages) == 1 + assert result.messages[0].role == "assistant" + assert result.messages[0].text == "Assistant answer" + assert result.messages[0].author_name == "Teacher" + + async def test_workflow_as_agent_filters_non_assistant_messages_from_list_of_messages(self) -> None: + """Verify WorkflowAgent filters user, system, and tool messages from list[Message].""" + + @executor + async def mixed_list_executor(messages: list[Message], ctx: WorkflowContext[Never, list[Message]]) -> None: + await ctx.yield_output([ + Message(role="user", contents=["what is 2+2?"]), + Message(role="assistant", contents=["4"], author_name="Maths"), + Message(role="system", contents=["system prompt"]), + Message(role="assistant", contents=["four"], author_name="English"), + ]) + + workflow = WorkflowBuilder(start_executor=mixed_list_executor).build() + agent = workflow.as_agent("mixed-list-agent") + + # Test streaming path + updates: list[AgentResponseUpdate] = [] + async for chunk in agent.run("calc", stream=True): + updates.append(chunk) + + assert len(updates) == 2 + for update in updates: + assert update.role == "assistant" + assert updates[0].author_name == "Maths" + assert updates[0].text == "4" + assert updates[1].author_name == "English" + assert updates[1].text == "four" + + # Test non-streaming path + result = await agent.run("calc") + assert len(result.messages) == 2 + for message in result.messages: + assert message.role == "assistant" + assert result.messages[0].author_name == "Maths" + assert result.messages[0].text == "4" + assert result.messages[1].author_name == "English" + assert result.messages[1].text == "four" + + async def test_workflow_as_agent_filters_single_non_assistant_message(self) -> None: + """Verify WorkflowAgent filters a single Message when role is not assistant.""" + + @executor + async def user_message_executor(messages: list[Message], ctx: WorkflowContext[Never, Message]) -> None: + await ctx.yield_output(Message(role="user", contents=["echoed user message"])) + + workflow = WorkflowBuilder(start_executor=user_message_executor).build() + agent = workflow.as_agent("user-msg-agent") + + # Streaming should yield no updates + updates: list[AgentResponseUpdate] = [] + async for chunk in agent.run("test", stream=True): + updates.append(chunk) + assert len(updates) == 0 + + # Non-streaming should produce empty messages list + result = await agent.run("test") + assert len(result.messages) == 0 + + async def test_workflow_as_agent_filters_user_agent_response_update(self) -> None: + """Verify WorkflowAgent drops AgentResponseUpdate when role is user.""" + + @executor + async def update_yielding_executor( + messages: list[Message], ctx: WorkflowContext[Never, AgentResponseUpdate] + ) -> None: + await ctx.yield_output(AgentResponseUpdate(contents=[Content.from_text(text="echo")], role="user")) + await ctx.yield_output(AgentResponseUpdate(contents=[Content.from_text(text="answer")], role="assistant")) + + workflow = WorkflowBuilder(start_executor=update_yielding_executor).build() + agent = workflow.as_agent("update-agent") + + updates: list[AgentResponseUpdate] = [] + async for chunk in agent.run("test", stream=True): + updates.append(chunk) + + assert len(updates) == 1 + assert updates[0].role == "assistant" + assert updates[0].text == "answer" + + async def test_workflow_as_agent_empty_after_filtering(self) -> None: + """Verify WorkflowAgent handles all non-assistant messages without crashing.""" + + @executor + async def non_assistant_only_executor( + messages: list[Message], ctx: WorkflowContext[Never, AgentResponse] + ) -> None: + response = AgentResponse( + messages=[ + Message(role="user", contents=["user msg"]), + Message(role="system", contents=["system msg"]), + Message(role="tool", contents=["tool msg"]), + ] + ) + await ctx.yield_output(response) + + workflow = WorkflowBuilder(start_executor=non_assistant_only_executor).build() + agent = workflow.as_agent("all-filtered-agent") + + result = await agent.run("test") + assert len(result.messages) == 0 + assert len(result.raw_representation) == 0 + + async def test_workflow_as_agent_multi_turn_user_input_not_compounded(self) -> None: + """Verify user messages in conversation history do not compound into responses across turns.""" + + class HistoryYieldingExecutor(Executor): + @handler + async def handle_messages( + self, + messages: list[Message], + ctx: WorkflowContext[Never, AgentResponse], + ) -> None: + user_text = messages[-1].text or "" + # Simulates orchestrators that include full conversation history in output + full_history = [ + Message(role="user", contents=[user_text]), + Message(role="assistant", contents=[f"Answer: {user_text}"], author_name="Agent"), + ] + await ctx.yield_output(AgentResponse(messages=full_history)) + + workflow = WorkflowBuilder(start_executor=HistoryYieldingExecutor(id="history-exec")).build() + agent = workflow.as_agent("history-agent") + session = AgentSession() + + # Turn 1 non-streaming + resp1 = await agent.run("first_query", session=session) + assert len(resp1.messages) == 1 + assert resp1.messages[0].role == "assistant" + assert resp1.text == "Answer: first_query" + assert "first_query" not in (resp1.text.replace("Answer: first_query", "")) + + # Turn 2 non-streaming: first_query must not bleed into turn 2 + resp2 = await agent.run("second_query", session=session) + assert len(resp2.messages) == 1 + assert resp2.messages[0].role == "assistant" + assert resp2.text == "Answer: second_query" + + # Streaming check + streaming_agent = workflow.as_agent("streaming-history-agent") + streaming_session = AgentSession() + + chunks1: list[AgentResponseUpdate] = [] + async for chunk in streaming_agent.run("stream_q1", stream=True, session=streaming_session): + chunks1.append(chunk) + assert len(chunks1) == 1 + assert chunks1[0].role == "assistant" + assert chunks1[0].text == "Answer: stream_q1" + + chunks2: list[AgentResponseUpdate] = [] + async for chunk in streaming_agent.run("stream_q2", stream=True, session=streaming_session): + chunks2.append(chunk) + assert len(chunks2) == 1 + assert chunks2[0].role == "assistant" + assert chunks2[0].text == "Answer: stream_q2" From a6eff9e8e7463e3cd17452975fc9e99352526a3e Mon Sep 17 00:00:00 2001 From: Manohar Paturi <186662190+ManoharPaturi@users.noreply.github.com> Date: Fri, 25 Sep 2026 18:36:51 +0530 Subject: [PATCH 2/4] Fix test typing checks in workflow agent tests Signed-off-by: Manohar Paturi <186662190+ManoharPaturi@users.noreply.github.com> --- .../tests/workflow/test_workflow_agent.py | 23 +++++++++++++------ 1 file changed, 16 insertions(+), 7 deletions(-) diff --git a/python/packages/core/tests/workflow/test_workflow_agent.py b/python/packages/core/tests/workflow/test_workflow_agent.py index a09d8f52ecc..d04f01499af 100644 --- a/python/packages/core/tests/workflow/test_workflow_agent.py +++ b/python/packages/core/tests/workflow/test_workflow_agent.py @@ -2603,7 +2603,8 @@ async def test_workflow_as_agent_filters_non_assistant_messages_from_agent_respo @executor async def mixed_agent_response_executor( - messages: list[Message], ctx: WorkflowContext[Never, AgentResponse] + messages: list[Message], + ctx: WorkflowContext[Never, AgentResponse], # type: ignore[valid-type] ) -> None: response = AgentResponse( messages=[ @@ -2639,7 +2640,10 @@ async def test_workflow_as_agent_filters_non_assistant_messages_from_list_of_mes """Verify WorkflowAgent filters user, system, and tool messages from list[Message].""" @executor - async def mixed_list_executor(messages: list[Message], ctx: WorkflowContext[Never, list[Message]]) -> None: + async def mixed_list_executor( + messages: list[Message], + ctx: WorkflowContext[Never, list[Message]], # type: ignore[valid-type] + ) -> None: await ctx.yield_output([ Message(role="user", contents=["what is 2+2?"]), Message(role="assistant", contents=["4"], author_name="Maths"), @@ -2677,7 +2681,10 @@ async def test_workflow_as_agent_filters_single_non_assistant_message(self) -> N """Verify WorkflowAgent filters a single Message when role is not assistant.""" @executor - async def user_message_executor(messages: list[Message], ctx: WorkflowContext[Never, Message]) -> None: + async def user_message_executor( + messages: list[Message], + ctx: WorkflowContext[Never, Message], # type: ignore[valid-type] + ) -> None: await ctx.yield_output(Message(role="user", contents=["echoed user message"])) workflow = WorkflowBuilder(start_executor=user_message_executor).build() @@ -2698,7 +2705,8 @@ async def test_workflow_as_agent_filters_user_agent_response_update(self) -> Non @executor async def update_yielding_executor( - messages: list[Message], ctx: WorkflowContext[Never, AgentResponseUpdate] + messages: list[Message], + ctx: WorkflowContext[Never, AgentResponseUpdate], # type: ignore[valid-type] ) -> None: await ctx.yield_output(AgentResponseUpdate(contents=[Content.from_text(text="echo")], role="user")) await ctx.yield_output(AgentResponseUpdate(contents=[Content.from_text(text="answer")], role="assistant")) @@ -2719,7 +2727,8 @@ async def test_workflow_as_agent_empty_after_filtering(self) -> None: @executor async def non_assistant_only_executor( - messages: list[Message], ctx: WorkflowContext[Never, AgentResponse] + messages: list[Message], + ctx: WorkflowContext[Never, AgentResponse], # type: ignore[valid-type] ) -> None: response = AgentResponse( messages=[ @@ -2735,7 +2744,7 @@ async def non_assistant_only_executor( result = await agent.run("test") assert len(result.messages) == 0 - assert len(result.raw_representation) == 0 + assert not result.raw_representation async def test_workflow_as_agent_multi_turn_user_input_not_compounded(self) -> None: """Verify user messages in conversation history do not compound into responses across turns.""" @@ -2745,7 +2754,7 @@ class HistoryYieldingExecutor(Executor): async def handle_messages( self, messages: list[Message], - ctx: WorkflowContext[Never, AgentResponse], + ctx: WorkflowContext[Never, AgentResponse], # type: ignore[valid-type] ) -> None: user_text = messages[-1].text or "" # Simulates orchestrators that include full conversation history in output From cf2a1c0967eaa4621d0e31990cae8deb36448149 Mon Sep 17 00:00:00 2001 From: Manohar Paturi <186662190+ManoharPaturi@users.noreply.github.com> Date: Mon, 28 Sep 2026 19:00:21 +0530 Subject: [PATCH 3/4] fix(workflows): keep non-assistant messages out of raw_representation the list[Message] branch appended the whole yielded list as the raw_representation even when non-assistant entries were filtered out of the public messages, so user/system/tool content stayed reachable through the non-streaming AgentResponse payload. when the list is filtered, append each assistant message's own raw_representation instead; unfiltered lists keep the original object. Signed-off-by: Manohar Paturi <186662190+ManoharPaturi@users.noreply.github.com> --- .../packages/core/agent_framework/_workflows/_agent.py | 9 ++++++++- .../packages/core/tests/workflow/test_workflow_agent.py | 5 +++++ 2 files changed, 13 insertions(+), 1 deletion(-) diff --git a/python/packages/core/agent_framework/_workflows/_agent.py b/python/packages/core/agent_framework/_workflows/_agent.py index 82293770a39..38fa8ea0a8d 100644 --- a/python/packages/core/agent_framework/_workflows/_agent.py +++ b/python/packages/core/agent_framework/_workflows/_agent.py @@ -612,7 +612,14 @@ def _convert_workflow_events_to_agent_response( assistant_messages = [msg for msg in chat_messages if msg.role == "assistant"] if assistant_messages: messages.extend(assistant_messages) - raw_representations.append(data) + # raw_representation of a filtered list must not leak the + # non-assistant entries the public messages list dropped. + if len(assistant_messages) == len(chat_messages): + raw_representations.append(data) + else: + raw_representations.extend( + msg.raw_representation for msg in assistant_messages + ) else: contents = self._extract_contents(data) if not contents: diff --git a/python/packages/core/tests/workflow/test_workflow_agent.py b/python/packages/core/tests/workflow/test_workflow_agent.py index d04f01499af..0d4299ef543 100644 --- a/python/packages/core/tests/workflow/test_workflow_agent.py +++ b/python/packages/core/tests/workflow/test_workflow_agent.py @@ -2677,6 +2677,11 @@ async def mixed_list_executor( assert result.messages[1].author_name == "English" assert result.messages[1].text == "four" + # raw_representation of the non-streaming result must not leak the + # filtered-out user/system messages through the public payload. + for rep in result.raw_representation or []: + assert rep is None or (isinstance(rep, Message) and rep.role == "assistant") + async def test_workflow_as_agent_filters_single_non_assistant_message(self) -> None: """Verify WorkflowAgent filters a single Message when role is not assistant.""" From f320aed5165f9da51327b6e9fa237efe191bdc09 Mon Sep 17 00:00:00 2001 From: Manohar Paturi <186662190+ManoharPaturi@users.noreply.github.com> Date: Tue, 29 Sep 2026 19:56:54 +0530 Subject: [PATCH 4/4] fix(workflows): drop assistant function calls orphaned by role filtering MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Role-only filtering kept assistant(function_call) while dropping its tool(function_result), producing an invalid call-without-result transcript that providers reject on replay when the response is persisted through a session. Assistant messages whose contents are entirely function-call envelopes are now dropped alongside the non-assistant messages, so the public response carries only user-facing assistant output with no dangling calls. Regression test covers call → tool result → final. Signed-off-by: Manohar Paturi <186662190+ManoharPaturi@users.noreply.github.com> --- .../core/agent_framework/_workflows/_agent.py | 52 ++++++++++++++----- .../tests/workflow/test_workflow_agent.py | 43 ++++++++++++++- 2 files changed, 81 insertions(+), 14 deletions(-) diff --git a/python/packages/core/agent_framework/_workflows/_agent.py b/python/packages/core/agent_framework/_workflows/_agent.py index 38fa8ea0a8d..ea96af26da9 100644 --- a/python/packages/core/agent_framework/_workflows/_agent.py +++ b/python/packages/core/agent_framework/_workflows/_agent.py @@ -3,12 +3,19 @@ from __future__ import annotations import logging -import sys import uuid from collections.abc import AsyncIterable, Awaitable, Callable, Mapping, Sequence from dataclasses import dataclass from datetime import datetime, timezone -from typing import TYPE_CHECKING, Any, ClassVar, Literal, cast, overload +from typing import ( + TYPE_CHECKING, + Any, + ClassVar, + Literal, + TypedDict, # pragma: no cover + cast, + overload, +) from .._agents import BaseAgent from .._sessions import ( @@ -41,17 +48,30 @@ from ._message_utils import normalize_messages_input from ._typing_utils import is_instance_of, is_type_compatible -if sys.version_info >= (3, 11): - from typing import TypedDict # pragma: no cover -else: - from typing_extensions import TypedDict # pragma: no cover - if TYPE_CHECKING: from ._workflow import Workflow, WorkflowInvocationKwargs logger = logging.getLogger(__name__) +def _is_orphaned_function_call(message: Message) -> bool: + """Drop a call whose result cannot follow it. + + True when an assistant message is only a function call whose tool result + is not assistant-role, so keeping the call alone would produce an invalid + (unpaired) transcript for providers that validate call/result history. + """ + if message.role != "assistant": + return False + contents = list(getattr(message, "contents", []) or []) + if not contents: + return False + # Orphaned only when every content is a function-call envelope + return all( + getattr(content, "type", None) in ("function_call", "function_approval_response") for content in contents + ) + + class WorkflowAgent(BaseAgent): """An `Agent` subclass that wraps a workflow and exposes it as an agent.""" @@ -591,7 +611,13 @@ def _convert_workflow_events_to_agent_response( # are intentionally excluded. System prompts and tool results are # internal workflow artifacts; user messages would be re-emitted # (e.g., from GroupChat orchestrators that include full conversation history). - assistant_messages = [msg for msg in data.messages if msg.role == "assistant"] + # Assistant messages that are bare function calls are also dropped + # when their tool result is not assistant-role: a call without its + # result is an invalid transcript for providers that validate + # call/result pairing on replay. + assistant_messages = [ + msg for msg in data.messages if msg.role == "assistant" and not _is_orphaned_function_call(msg) + ] if assistant_messages: messages.extend(assistant_messages) raw_representations.append(data.raw_representation) @@ -609,7 +635,11 @@ def _convert_workflow_events_to_agent_response( raw_representations.append(data.raw_representation) elif is_instance_of(data, list[Message]): chat_messages = cast(list[Message], data) - assistant_messages = [msg for msg in chat_messages if msg.role == "assistant"] + # Keep tool results that pair with surviving assistant calls so the + # transcript stays valid; drop user/system and orphaned calls. + assistant_messages = [ + msg for msg in chat_messages if msg.role == "assistant" and not _is_orphaned_function_call(msg) + ] if assistant_messages: messages.extend(assistant_messages) # raw_representation of a filtered list must not leak the @@ -617,9 +647,7 @@ def _convert_workflow_events_to_agent_response( if len(assistant_messages) == len(chat_messages): raw_representations.append(data) else: - raw_representations.extend( - msg.raw_representation for msg in assistant_messages - ) + raw_representations.extend(msg.raw_representation for msg in assistant_messages) else: contents = self._extract_contents(data) if not contents: diff --git a/python/packages/core/tests/workflow/test_workflow_agent.py b/python/packages/core/tests/workflow/test_workflow_agent.py index 0d4299ef543..a04c572a21f 100644 --- a/python/packages/core/tests/workflow/test_workflow_agent.py +++ b/python/packages/core/tests/workflow/test_workflow_agent.py @@ -3,10 +3,9 @@ import uuid from collections.abc import Awaitable, Sequence from dataclasses import dataclass -from typing import Any, Literal, cast, overload +from typing import Any, Literal, Never, assert_type, cast, overload import pytest -from typing_extensions import Never, assert_type from agent_framework import ( AgentExecutorRequest, @@ -2682,6 +2681,46 @@ async def mixed_list_executor( for rep in result.raw_representation or []: assert rep is None or (isinstance(rep, Message) and rep.role == "assistant") + async def test_workflow_as_agent_drops_orphaned_function_calls(self) -> None: + """assistant(function_call) whose tool result is tool-role must not survive filtering. + + Keeping the call without its result would produce an invalid transcript for + providers that validate call/result pairing on replay (eavanvalkenburg's review). + """ + + @executor + async def tool_transcript_executor( + messages: list[Message], + ctx: WorkflowContext[Never, list[Message]], # type: ignore[valid-type] + ) -> None: + await ctx.yield_output([ + Message( + role="assistant", + contents=[ + Content.from_function_call(call_id="call-1", name="get_weather", arguments={"city": "Paris"}), + ], + ), + Message( + role="tool", + contents=[ + Content.from_function_result(call_id="call-1", result="18C"), + ], + ), + Message(role="assistant", contents=[Content.from_text("It is 18C in Paris.")]), + ]) + + workflow = WorkflowBuilder(start_executor=tool_transcript_executor).build() + agent = workflow.as_agent("tool-transcript-agent") + + result = await agent.run("weather") + + # The orphaned call is dropped; the final user-facing answer survives. + assert all(msg.role == "assistant" for msg in result.messages) + assert not any( + getattr(content, "type", None) == "function_call" for msg in result.messages for content in msg.contents + ) + assert any("18C" in (getattr(content, "text", "") or "") for msg in result.messages for content in msg.contents) + async def test_workflow_as_agent_filters_single_non_assistant_message(self) -> None: """Verify WorkflowAgent filters a single Message when role is not assistant."""