feat: forward tool-result images to the MCP caller in send_message results

fast-agent's agent.send() returns only the final assistant text, so
ImageContent produced by downstream tools during the agentic loop
(playwright screenshots, rommie desktop captures) reached the agent's own
vision model but never crossed the MCP boundary — Daedalus and lead agents
saw text-only results.

A per-request after_tool_call hook (pallas.image_passthrough, same
composition pattern as assistant_stream / loop_guard) collects every
ImageContent block from the turn's tool results; send_message then returns
a FastMCP ToolResult of [final text, *images]. Turns with no images return
the plain string — wire shape unchanged (the str-only output schema is
dropped so the union return passes through cleanly; no consumer read
structuredContent). Images cascade hop-by-hop up delegation chains with no
extra wiring: verified live playwright → dolores → harper → MCP client,
image intact at each hop.

New per-agent agents.yaml knob max_result_images (default 8, keeps most
recent, 0 disables) and pallas_result_images_total counter. Version 0.7.0.

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
This commit is contained in:
2026-08-01 22:47:25 -04:00
parent 54d639c13a
commit 1b05504207
7 changed files with 359 additions and 5 deletions

146
pallas/image_passthrough.py Normal file
View File

@@ -0,0 +1,146 @@
"""Tool-result image passthrough for ``send_message``.
fast-agent's ``agent.send()`` returns only the final assistant text, so
``ImageContent`` produced by downstream tools during the agentic loop —
playwright screenshots, rommie desktop captures — reaches the agent's own
vision model but never crosses the MCP boundary. The caller (Daedalus, or a
lead agent driving a sub-agent) sees a text-only ``CallToolResult`` and the
live stream reduces result images to a ``"+N images"`` preview.
This module fixes that by installing a per-request ``after_tool_call`` hook
(same composition pattern as ``assistant_stream`` / ``loop_guard``) that
collects every ``ImageContent`` block from the turn's tool results. At end of
turn ``multimodal_server.send_message`` appends the collected images to the
final ``CallToolResult`` via FastMCP's ``ToolResult`` passthrough, after the
assistant's text block. Daedalus already renders image blocks in final
results; a lead agent calling a sub-agent receives them as ordinary
tool-result content, so images cascade hop-by-hop up the delegation chain.
Only the most recent ``max_images`` are forwarded — a long loop that grabs a
screenshot per iteration would otherwise balloon the HTTP response — and the
final state is what the screenshots are usually for. Dropped images are
logged; forwarded ones are counted in ``pallas_result_images_total``.
"""
from __future__ import annotations
import logging
from typing import Any
from fast_agent.agents.tool_runner import ToolRunnerHooks
from fast_agent.types import PromptMessageExtended
from fastmcp.tools import ToolResult
from mcp.types import ImageContent, TextContent
from pallas.assistant_stream import _merge_hooks
logger = logging.getLogger("pallas.image_passthrough")
DEFAULT_MAX_IMAGES = 8
class ImageCollector:
"""Per-request accumulator of tool-result ``ImageContent`` blocks."""
def __init__(
self,
*,
agent_name: str,
conversation_id: str | None,
max_images: int = DEFAULT_MAX_IMAGES,
) -> None:
self._agent_name = agent_name
self._conversation_id = conversation_id
self._max_images = max_images
self._images: list[ImageContent] = []
def as_after_tool_call_hook(self):
async def _hook(_runner: Any, message: PromptMessageExtended) -> None:
try:
for result in (message.tool_results or {}).values():
for block in getattr(result, "content", None) or []:
if isinstance(block, ImageContent):
self._images.append(block)
except Exception: # never let collection break a live turn
logger.warning(
"image_passthrough collection failed",
exc_info=True,
extra={
"agent": self._agent_name,
"conversation_id": self._conversation_id,
},
)
return _hook
def collected(self) -> list[ImageContent]:
"""Return the images to forward — the most recent ``max_images``.
When the loop produced more than the cap, the oldest are dropped:
screenshots are usually taken to show the *final* state, and the
intermediate ones were already summarized on the live stream.
"""
if len(self._images) <= self._max_images:
return list(self._images)
dropped = len(self._images) - self._max_images
logger.warning(
"image_passthrough dropping oldest tool-result images",
extra={
"agent": self._agent_name,
"conversation_id": self._conversation_id,
"collected": len(self._images),
"forwarded": self._max_images,
"dropped": dropped,
},
)
return self._images[-self._max_images :]
def build_result(text: str, images: list[ImageContent]) -> str | ToolResult:
"""Assemble ``send_message``'s return value.
With no images the plain string is returned so the wire shape is
byte-identical to previous releases. With images, an explicit FastMCP
``ToolResult`` carries ``[text, *images]`` content blocks — FastMCP
passes it through verbatim, bypassing return-annotation conversion.
"""
if not images:
return text
content: list[TextContent | ImageContent] = []
if text:
content.append(TextContent(type="text", text=text))
content.extend(images)
return ToolResult(content=content)
def install_for_request(
agent: Any,
*,
agent_name: str,
conversation_id: str | None,
max_images: int = DEFAULT_MAX_IMAGES,
):
"""Install the image collector on a request-scoped agent instance.
Returns ``(collector, restore)``. Call ``restore`` in a ``finally`` like
the other per-request hooks. A non-positive ``max_images`` disables
passthrough entirely (returns ``(None, no-op)``).
"""
if max_images is None or max_images < 1:
return None, lambda: None
collector = ImageCollector(
agent_name=agent_name,
conversation_id=conversation_id,
max_images=max_images,
)
extra = ToolRunnerHooks(after_tool_call=collector.as_after_tool_call_hook())
previous = getattr(agent, "tool_runner_hooks", None)
agent.tool_runner_hooks = _merge_hooks(previous, extra)
def restore() -> None:
agent.tool_runner_hooks = previous
return collector, restore

View File

@@ -138,6 +138,13 @@ agent_loop_aborted_total = Counter(
registry=REGISTRY,
)
result_images_total = Counter(
"pallas_result_images_total",
"Tool-result images forwarded to the MCP caller in send_message results",
labelnames=["agent"],
registry=REGISTRY,
)
# ── Helpers ──────────────────────────────────────────────────────────────────
@@ -147,6 +154,11 @@ def record_loop_abort(agent: str, reason: str) -> None:
agent_loop_aborted_total.labels(agent=agent, reason=reason).inc()
def record_result_images(agent: str, count: int) -> None:
"""Record tool-result images forwarded in a send_message result."""
result_images_total.labels(agent=agent).inc(count)
def set_agent_info(agents: dict[str, dict]) -> None:
"""Record the deployment's configured agents (called once at startup)."""
for name, agent in agents.items():

View File

@@ -9,7 +9,10 @@ Overrides register_agent_tools to:
so callers own conversation state and seed it on every turn,
* accept an optional ``conversation_id`` string that is recorded in
structured logs and progress notification metadata for end-to-end
trace correlation.
trace correlation,
* append images produced by downstream tools during the turn (playwright
screenshots, rommie desktop captures) to the final ``CallToolResult``
so they reach the MCP caller — see ``pallas.image_passthrough``.
Drop-in replacement for AgentMCPServer. When combined with
``instance_scope="request"`` (the Pallas default), this gives a fully
@@ -29,11 +32,17 @@ from fast_agent.mcp.server import AgentMCPServer
from fast_agent.types import PromptMessageExtended, RequestParams
from pallas.assistant_stream import install_for_request as _install_assistant_stream
from pallas.image_passthrough import (
DEFAULT_MAX_IMAGES,
build_result as _build_image_result,
install_for_request as _install_image_passthrough,
)
from pallas.loop_guard import install_for_request as _install_loop_guard
from pallas.progress import EnrichedMCPToolProgressManager
from pallas import metrics as _pallas_metrics
from fastmcp import Context as MCPContext
from fastmcp.prompts import Message
from fastmcp.tools import ToolResult
from mcp.types import ImageContent, TextContent
from prometheus_client import CONTENT_TYPE_LATEST, generate_latest
from starlette.responses import JSONResponse, Response
@@ -188,7 +197,7 @@ class MultimodalAgentMCPServer(AgentMCPServer):
images: list[dict] | None = None,
history: list[dict] | None = None,
conversation_id: str | None = None,
) -> str:
) -> str | ToolResult:
"""Send a single turn to the agent.
Parameters
@@ -209,6 +218,12 @@ class MultimodalAgentMCPServer(AgentMCPServer):
conversation_id:
Optional opaque identifier, logged for trace correlation.
Pallas does not interpret it.
Returns the assistant's final text. When downstream tools
produced images this turn (screenshots etc.), returns a
``ToolResult`` whose content is the text block followed by the
most recent ``max_result_images`` image blocks, so they reach
the MCP caller instead of dying inside ``message_history``.
"""
report_progress = self._build_progress_reporter(ctx)
request_params = RequestParams(
@@ -245,8 +260,20 @@ class MultimodalAgentMCPServer(AgentMCPServer):
conversation_id=conversation_id,
threshold=self._request_limits.get("loop_repeat_threshold", 3),
)
# Collect tool-result images (screenshots) so the final
# CallToolResult can carry them to the caller — fast-agent's
# send() return value is text-only.
image_collector, restore_images = _install_image_passthrough(
agent,
agent_name=agent_name,
conversation_id=conversation_id,
max_images=self._request_limits.get(
"max_result_images", DEFAULT_MAX_IMAGES
),
)
def restore_hooks() -> None:
restore_images()
restore_guard()
restore_stream()
try:
@@ -314,7 +341,25 @@ class MultimodalAgentMCPServer(AgentMCPServer):
return await execute_send()
try:
return await asyncio.wait_for(_dispatch(), timeout=turn_timeout)
response = await asyncio.wait_for(
_dispatch(), timeout=turn_timeout
)
images = (
image_collector.collected() if image_collector else []
)
if images:
_pallas_metrics.record_result_images(
agent_name, len(images)
)
logger.debug(
f"Forwarding {len(images)} tool-result image(s) "
f"from agent '{agent_name}'",
name="result_images_forwarded",
agent=agent_name,
image_count=len(images),
conversation_id=conversation_id,
)
return _build_image_result(response, images)
except asyncio.TimeoutError:
logger.warning(
f"Agent '{agent_name}' turn exceeded {turn_timeout}s wall-clock limit",

View File

@@ -67,6 +67,7 @@ def _build_agents_table(config: dict) -> dict[str, dict]:
"streaming_timeout": agent.get("streaming_timeout"),
"turn_timeout": agent.get("turn_timeout"),
"loop_repeat_threshold": agent.get("loop_repeat_threshold"),
"max_result_images": agent.get("max_result_images"),
}
for name, agent in config["agents"].items()
}
@@ -271,6 +272,7 @@ async def _start_agent(name: str, agents: dict[str, dict]) -> None:
"streaming_timeout",
"turn_timeout",
"loop_repeat_threshold",
"max_result_images",
)
if entry.get(k) is not None
}