diff --git a/docs/bedrock.md b/docs/bedrock.md index 9be4b00..023ae12 100644 --- a/docs/bedrock.md +++ b/docs/bedrock.md @@ -218,7 +218,7 @@ anthropic: That's the whole configuration. Pallas auto-detects the `bedrock-mantle` hostname in `anthropic.base_url` at startup and installs -two compatibility shims so fast-agent's default request shape matches +four compatibility shims so fast-agent's default request shape matches what Mantle expects (see `pallas/mantle_shims.py`): 1. **Wire-name prefix** — re-adds the `anthropic.` prefix that fast-agent's @@ -233,6 +233,25 @@ what Mantle expects (see `pallas/mantle_shims.py`): Input should be a valid dictionary or object"`, which would otherwise break the MCP tool-use loop on the second turn. +3. **Fine-grained tool streaming opt-out** — stops fast-agent sending the + `fine-grained-tool-streaming-2025-05-14` beta. Under that beta an + output-token cutoff mid-`tool_use` block ends the stream without + `content_block_stop`, which fast-agent surfaces as + `Streaming completed but tool call never finished` and then retries + into the same wall (~700 s agent failures on large tool bodies). + Without the beta a cutoff closes blocks properly and lands in + fast-agent's graceful `stop_reason=max_tokens` handling. + +4. **`max_tokens` clamp** — Mantle enforces a server-side output ceiling + of 20 000 tokens per response regardless of the requested + `max_tokens` (fast-agent asks for the model's full 128 000). The shim + clamps default `maxTokens` to `MANTLE_MAX_OUTPUT_TOKENS` (20 000) so + the model stops gracefully at the limit instead of the gateway + cutting the stream. A single agent turn — thinking, prose, and tool + input combined — cannot exceed this on Mantle; agents that must emit + more output in one turn need to split the work (e.g. patch-style + edits instead of full-document rewrites). + The Anthropic SDK appends `/v1/messages` to `base_url` automatically. **Feature support.** Mantle accepts the same Messages API request shape diff --git a/pallas/mantle_shims.py b/pallas/mantle_shims.py index 004eab9..3e75ec1 100644 --- a/pallas/mantle_shims.py +++ b/pallas/mantle_shims.py @@ -24,7 +24,23 @@ before the wire traffic is valid: Upstream SDK tracker: https://github.com/anthropics/anthropic-sdk-python/issues/1454 -Both shims are idempotent and may be installed at process startup before any +3. **Fine-grained tool streaming truncation.** Fast-agent unconditionally + sends the ``fine-grained-tool-streaming-2025-05-14`` beta on tool-bearing + requests. Under that beta, an output-token cutoff mid-``tool_use`` block + ends the stream *without* ``content_block_stop``, which fast-agent's + stream accounting surfaces as ``Streaming completed but tool call never + finished`` — followed by a full retry ladder against the same wall + (observed as ~700 s agent failures on large ``revise_workspace_file`` + bodies). We disable that one beta so a cutoff closes blocks properly and + lands in fast-agent's graceful ``stop_reason=max_tokens`` handling. + +4. **Output-token ceiling.** Mantle clamps ``max_tokens`` to 20 000 + server-side (streams complete at exactly 20 000 output tokens regardless + of the requested 128 000). We clamp the request to that ceiling so the + *model* stops gracefully at the limit — emitting proper block-stop events + and ``stop_reason`` — instead of being cut off by the gateway's clamp. + +All shims are idempotent and may be installed at process startup before any fast-agent ``FastAgent`` instance is constructed. """ from __future__ import annotations @@ -129,12 +145,90 @@ def install_tool_use_caller_strip() -> None: logger.info("Mantle tool_use.caller strip shim installed") +# ── Shim 3: disable fine-grained tool streaming ────────────────────────────── + +_beta_opt_out_installed = False + + +def install_fine_grained_tool_streaming_opt_out() -> None: + """Stop fast-agent requesting the fine-grained tool streaming beta. + + Under that beta a ``max_tokens`` cutoff mid-``tool_use`` ends the stream + without ``content_block_stop``; fast-agent then raises + ``Streaming completed but tool call never finished`` and burns its whole + retry ladder against the same ceiling. Without the beta the cutoff closes + blocks properly and fast-agent's ``stop_reason=max_tokens`` handling + applies. Safe to call more than once. + """ + global _beta_opt_out_installed + if _beta_opt_out_installed: + return + + from fast_agent.llm.provider.anthropic.llm_anthropic import AnthropicLLM + + original_supports = AnthropicLLM.supports_direct_anthropic_beta + + def patched_supports(self: Any, feature: str) -> bool: + if feature == "fine_grained_tool_streaming": + return False + return original_supports(self, feature) + + AnthropicLLM.supports_direct_anthropic_beta = patched_supports # type: ignore[method-assign] + + _beta_opt_out_installed = True + logger.info("Mantle fine-grained tool streaming opt-out shim installed") + + +# ── Shim 4: clamp max_tokens to Mantle's output ceiling ────────────────────── + +# Observed server-side clamp: Mantle streams stop at exactly 20 000 output +# tokens however large the requested max_tokens. Requesting the ceiling +# explicitly makes the model stop gracefully (proper block close + stop_reason) +# instead of the gateway cutting the stream at its own limit. +MANTLE_MAX_OUTPUT_TOKENS = 20_000 + +_max_tokens_clamp_installed = False + + +def install_max_tokens_clamp() -> None: + """Clamp default ``maxTokens`` to Mantle's output ceiling. + + Wraps ``AnthropicLLM._initialize_default_params`` so every agent's + default request params carry an explicit ``maxTokens`` no higher than the + ceiling. Per-request ``RequestParams`` overrides bypass this — none of + our agent modules set one. Safe to call more than once. + """ + global _max_tokens_clamp_installed + if _max_tokens_clamp_installed: + return + + from fast_agent.llm.provider.anthropic.llm_anthropic import AnthropicLLM + + original_init = AnthropicLLM._initialize_default_params # noqa: SLF001 + + def patched_init(self: Any, kwargs: dict) -> Any: + params = original_init(self, kwargs) + if params.maxTokens is None or params.maxTokens > MANTLE_MAX_OUTPUT_TOKENS: + params.maxTokens = MANTLE_MAX_OUTPUT_TOKENS + return params + + AnthropicLLM._initialize_default_params = patched_init # noqa: SLF001 + + _max_tokens_clamp_installed = True + logger.info( + "Mantle max_tokens clamp shim installed (ceiling %d)", + MANTLE_MAX_OUTPUT_TOKENS, + ) + + # ── Orchestrator ───────────────────────────────────────────────────────────── def install_all() -> None: """Install all Mantle shims. Call once at process startup.""" install_wire_name_prefix() install_tool_use_caller_strip() + install_fine_grained_tool_streaming_opt_out() + install_max_tokens_clamp() def maybe_install(anthropic_base_url: str | None) -> bool: diff --git a/tests/test_mantle_shims.py b/tests/test_mantle_shims.py index 33aab44..cae2fd6 100644 --- a/tests/test_mantle_shims.py +++ b/tests/test_mantle_shims.py @@ -114,34 +114,97 @@ def test_install_tool_use_caller_strip_is_idempotent() -> None: mantle_shims.install_tool_use_caller_strip() # must not raise or re-wrap +# ── install_fine_grained_tool_streaming_opt_out ────────────────────────────── + +def test_fine_grained_beta_opt_out() -> None: + from fast_agent.llm.provider.anthropic.llm_anthropic import AnthropicLLM + + mantle_shims.install_fine_grained_tool_streaming_opt_out() + + # The patched method never touches self, so a bare object suffices. + stub = object() + assert AnthropicLLM.supports_direct_anthropic_beta(stub, "fine_grained_tool_streaming") is False + # Every other beta keeps the base-class answer (True). + assert AnthropicLLM.supports_direct_anthropic_beta(stub, "interleaved_thinking") is True + assert AnthropicLLM.supports_direct_anthropic_beta(stub, "long_context") is True + + +def test_fine_grained_beta_opt_out_is_idempotent() -> None: + mantle_shims.install_fine_grained_tool_streaming_opt_out() + mantle_shims.install_fine_grained_tool_streaming_opt_out() # must not re-wrap + + from fast_agent.llm.provider.anthropic.llm_anthropic import AnthropicLLM + + assert ( + AnthropicLLM.supports_direct_anthropic_beta(object(), "fine_grained_tool_streaming") + is False + ) + + +# ── install_max_tokens_clamp ───────────────────────────────────────────────── + +@pytest.mark.parametrize( + "initial,expected", + [ + (128000, mantle_shims.MANTLE_MAX_OUTPUT_TOKENS), # over the ceiling → clamped + (None, mantle_shims.MANTLE_MAX_OUTPUT_TOKENS), # unset → pinned to ceiling + (4096, 4096), # under the ceiling → untouched + ], +) +def test_max_tokens_clamp( + monkeypatch: pytest.MonkeyPatch, initial: int | None, expected: int +) -> None: + from fast_agent.llm.provider.anthropic.llm_anthropic import AnthropicLLM + from fast_agent.types import RequestParams + + # Stub the underlying initializer, then force a fresh wrap around it. + monkeypatch.setattr( + AnthropicLLM, + "_initialize_default_params", + lambda self, kwargs: RequestParams(maxTokens=initial), + ) + monkeypatch.setattr(mantle_shims, "_max_tokens_clamp_installed", False) + mantle_shims.install_max_tokens_clamp() + + params = AnthropicLLM._initialize_default_params(object(), {}) + assert params.maxTokens == expected + + +def test_max_tokens_clamp_is_idempotent() -> None: + mantle_shims.install_max_tokens_clamp() + mantle_shims.install_max_tokens_clamp() # must not raise or re-wrap + + # ── maybe_install ──────────────────────────────────────────────────────────── -def test_maybe_install_installs_when_mantle(monkeypatch: pytest.MonkeyPatch) -> None: +_INSTALLER_NAMES = [ + ("install_wire_name_prefix", "wire"), + ("install_tool_use_caller_strip", "tool_use"), + ("install_fine_grained_tool_streaming_opt_out", "beta_opt_out"), + ("install_max_tokens_clamp", "max_tokens"), +] + + +def _patch_installers(monkeypatch: pytest.MonkeyPatch) -> list[str]: calls: list[str] = [] - monkeypatch.setattr( - mantle_shims, "install_wire_name_prefix", - lambda: calls.append("wire"), - ) - monkeypatch.setattr( - mantle_shims, "install_tool_use_caller_strip", - lambda: calls.append("tool_use"), - ) + for attr, label in _INSTALLER_NAMES: + monkeypatch.setattr( + mantle_shims, attr, + lambda label=label: calls.append(label), + ) + return calls + + +def test_maybe_install_installs_when_mantle(monkeypatch: pytest.MonkeyPatch) -> None: + calls = _patch_installers(monkeypatch) installed = mantle_shims.maybe_install("https://bedrock-mantle.us-east-1.api.aws/anthropic") assert installed is True - assert calls == ["wire", "tool_use"] + assert calls == ["wire", "tool_use", "beta_opt_out", "max_tokens"] def test_maybe_install_noop_for_non_mantle(monkeypatch: pytest.MonkeyPatch) -> None: - calls: list[str] = [] - monkeypatch.setattr( - mantle_shims, "install_wire_name_prefix", - lambda: calls.append("wire"), - ) - monkeypatch.setattr( - mantle_shims, "install_tool_use_caller_strip", - lambda: calls.append("tool_use"), - ) + calls = _patch_installers(monkeypatch) assert mantle_shims.maybe_install("https://api.anthropic.com") is False assert mantle_shims.maybe_install(None) is False