Merge pull request '🐾 fix(mantle): survive Mantle's 20k output ceiling on tool-heavy turns' (#8) from feature/mantle-max-tokens-shim into main
Reviewed-on: #8
This commit was merged in pull request #8.
This commit is contained in:
@@ -218,7 +218,7 @@ anthropic:
|
||||
|
||||
That's the whole configuration. Pallas auto-detects the
|
||||
`bedrock-mantle` hostname in `anthropic.base_url` at startup and installs
|
||||
two compatibility shims so fast-agent's default request shape matches
|
||||
four compatibility shims so fast-agent's default request shape matches
|
||||
what Mantle expects (see `pallas/mantle_shims.py`):
|
||||
|
||||
1. **Wire-name prefix** — re-adds the `anthropic.` prefix that fast-agent's
|
||||
@@ -233,6 +233,25 @@ what Mantle expects (see `pallas/mantle_shims.py`):
|
||||
Input should be a valid dictionary or object"`, which would otherwise
|
||||
break the MCP tool-use loop on the second turn.
|
||||
|
||||
3. **Fine-grained tool streaming opt-out** — stops fast-agent sending the
|
||||
`fine-grained-tool-streaming-2025-05-14` beta. Under that beta an
|
||||
output-token cutoff mid-`tool_use` block ends the stream without
|
||||
`content_block_stop`, which fast-agent surfaces as
|
||||
`Streaming completed but tool call never finished` and then retries
|
||||
into the same wall (~700 s agent failures on large tool bodies).
|
||||
Without the beta a cutoff closes blocks properly and lands in
|
||||
fast-agent's graceful `stop_reason=max_tokens` handling.
|
||||
|
||||
4. **`max_tokens` clamp** — Mantle enforces a server-side output ceiling
|
||||
of 20 000 tokens per response regardless of the requested
|
||||
`max_tokens` (fast-agent asks for the model's full 128 000). The shim
|
||||
clamps default `maxTokens` to `MANTLE_MAX_OUTPUT_TOKENS` (20 000) so
|
||||
the model stops gracefully at the limit instead of the gateway
|
||||
cutting the stream. A single agent turn — thinking, prose, and tool
|
||||
input combined — cannot exceed this on Mantle; agents that must emit
|
||||
more output in one turn need to split the work (e.g. patch-style
|
||||
edits instead of full-document rewrites).
|
||||
|
||||
The Anthropic SDK appends `/v1/messages` to `base_url` automatically.
|
||||
|
||||
**Feature support.** Mantle accepts the same Messages API request shape
|
||||
|
||||
@@ -24,7 +24,23 @@ before the wire traffic is valid:
|
||||
|
||||
Upstream SDK tracker: https://github.com/anthropics/anthropic-sdk-python/issues/1454
|
||||
|
||||
Both shims are idempotent and may be installed at process startup before any
|
||||
3. **Fine-grained tool streaming truncation.** Fast-agent unconditionally
|
||||
sends the ``fine-grained-tool-streaming-2025-05-14`` beta on tool-bearing
|
||||
requests. Under that beta, an output-token cutoff mid-``tool_use`` block
|
||||
ends the stream *without* ``content_block_stop``, which fast-agent's
|
||||
stream accounting surfaces as ``Streaming completed but tool call never
|
||||
finished`` — followed by a full retry ladder against the same wall
|
||||
(observed as ~700 s agent failures on large ``revise_workspace_file``
|
||||
bodies). We disable that one beta so a cutoff closes blocks properly and
|
||||
lands in fast-agent's graceful ``stop_reason=max_tokens`` handling.
|
||||
|
||||
4. **Output-token ceiling.** Mantle clamps ``max_tokens`` to 20 000
|
||||
server-side (streams complete at exactly 20 000 output tokens regardless
|
||||
of the requested 128 000). We clamp the request to that ceiling so the
|
||||
*model* stops gracefully at the limit — emitting proper block-stop events
|
||||
and ``stop_reason`` — instead of being cut off by the gateway's clamp.
|
||||
|
||||
All shims are idempotent and may be installed at process startup before any
|
||||
fast-agent ``FastAgent`` instance is constructed.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
@@ -129,12 +145,90 @@ def install_tool_use_caller_strip() -> None:
|
||||
logger.info("Mantle tool_use.caller strip shim installed")
|
||||
|
||||
|
||||
# ── Shim 3: disable fine-grained tool streaming ──────────────────────────────
|
||||
|
||||
_beta_opt_out_installed = False
|
||||
|
||||
|
||||
def install_fine_grained_tool_streaming_opt_out() -> None:
|
||||
"""Stop fast-agent requesting the fine-grained tool streaming beta.
|
||||
|
||||
Under that beta a ``max_tokens`` cutoff mid-``tool_use`` ends the stream
|
||||
without ``content_block_stop``; fast-agent then raises
|
||||
``Streaming completed but tool call never finished`` and burns its whole
|
||||
retry ladder against the same ceiling. Without the beta the cutoff closes
|
||||
blocks properly and fast-agent's ``stop_reason=max_tokens`` handling
|
||||
applies. Safe to call more than once.
|
||||
"""
|
||||
global _beta_opt_out_installed
|
||||
if _beta_opt_out_installed:
|
||||
return
|
||||
|
||||
from fast_agent.llm.provider.anthropic.llm_anthropic import AnthropicLLM
|
||||
|
||||
original_supports = AnthropicLLM.supports_direct_anthropic_beta
|
||||
|
||||
def patched_supports(self: Any, feature: str) -> bool:
|
||||
if feature == "fine_grained_tool_streaming":
|
||||
return False
|
||||
return original_supports(self, feature)
|
||||
|
||||
AnthropicLLM.supports_direct_anthropic_beta = patched_supports # type: ignore[method-assign]
|
||||
|
||||
_beta_opt_out_installed = True
|
||||
logger.info("Mantle fine-grained tool streaming opt-out shim installed")
|
||||
|
||||
|
||||
# ── Shim 4: clamp max_tokens to Mantle's output ceiling ──────────────────────
|
||||
|
||||
# Observed server-side clamp: Mantle streams stop at exactly 20 000 output
|
||||
# tokens however large the requested max_tokens. Requesting the ceiling
|
||||
# explicitly makes the model stop gracefully (proper block close + stop_reason)
|
||||
# instead of the gateway cutting the stream at its own limit.
|
||||
MANTLE_MAX_OUTPUT_TOKENS = 20_000
|
||||
|
||||
_max_tokens_clamp_installed = False
|
||||
|
||||
|
||||
def install_max_tokens_clamp() -> None:
|
||||
"""Clamp default ``maxTokens`` to Mantle's output ceiling.
|
||||
|
||||
Wraps ``AnthropicLLM._initialize_default_params`` so every agent's
|
||||
default request params carry an explicit ``maxTokens`` no higher than the
|
||||
ceiling. Per-request ``RequestParams`` overrides bypass this — none of
|
||||
our agent modules set one. Safe to call more than once.
|
||||
"""
|
||||
global _max_tokens_clamp_installed
|
||||
if _max_tokens_clamp_installed:
|
||||
return
|
||||
|
||||
from fast_agent.llm.provider.anthropic.llm_anthropic import AnthropicLLM
|
||||
|
||||
original_init = AnthropicLLM._initialize_default_params # noqa: SLF001
|
||||
|
||||
def patched_init(self: Any, kwargs: dict) -> Any:
|
||||
params = original_init(self, kwargs)
|
||||
if params.maxTokens is None or params.maxTokens > MANTLE_MAX_OUTPUT_TOKENS:
|
||||
params.maxTokens = MANTLE_MAX_OUTPUT_TOKENS
|
||||
return params
|
||||
|
||||
AnthropicLLM._initialize_default_params = patched_init # noqa: SLF001
|
||||
|
||||
_max_tokens_clamp_installed = True
|
||||
logger.info(
|
||||
"Mantle max_tokens clamp shim installed (ceiling %d)",
|
||||
MANTLE_MAX_OUTPUT_TOKENS,
|
||||
)
|
||||
|
||||
|
||||
# ── Orchestrator ─────────────────────────────────────────────────────────────
|
||||
|
||||
def install_all() -> None:
|
||||
"""Install all Mantle shims. Call once at process startup."""
|
||||
install_wire_name_prefix()
|
||||
install_tool_use_caller_strip()
|
||||
install_fine_grained_tool_streaming_opt_out()
|
||||
install_max_tokens_clamp()
|
||||
|
||||
|
||||
def maybe_install(anthropic_base_url: str | None) -> bool:
|
||||
|
||||
@@ -114,34 +114,97 @@ def test_install_tool_use_caller_strip_is_idempotent() -> None:
|
||||
mantle_shims.install_tool_use_caller_strip() # must not raise or re-wrap
|
||||
|
||||
|
||||
# ── install_fine_grained_tool_streaming_opt_out ──────────────────────────────
|
||||
|
||||
def test_fine_grained_beta_opt_out() -> None:
|
||||
from fast_agent.llm.provider.anthropic.llm_anthropic import AnthropicLLM
|
||||
|
||||
mantle_shims.install_fine_grained_tool_streaming_opt_out()
|
||||
|
||||
# The patched method never touches self, so a bare object suffices.
|
||||
stub = object()
|
||||
assert AnthropicLLM.supports_direct_anthropic_beta(stub, "fine_grained_tool_streaming") is False
|
||||
# Every other beta keeps the base-class answer (True).
|
||||
assert AnthropicLLM.supports_direct_anthropic_beta(stub, "interleaved_thinking") is True
|
||||
assert AnthropicLLM.supports_direct_anthropic_beta(stub, "long_context") is True
|
||||
|
||||
|
||||
def test_fine_grained_beta_opt_out_is_idempotent() -> None:
|
||||
mantle_shims.install_fine_grained_tool_streaming_opt_out()
|
||||
mantle_shims.install_fine_grained_tool_streaming_opt_out() # must not re-wrap
|
||||
|
||||
from fast_agent.llm.provider.anthropic.llm_anthropic import AnthropicLLM
|
||||
|
||||
assert (
|
||||
AnthropicLLM.supports_direct_anthropic_beta(object(), "fine_grained_tool_streaming")
|
||||
is False
|
||||
)
|
||||
|
||||
|
||||
# ── install_max_tokens_clamp ─────────────────────────────────────────────────
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"initial,expected",
|
||||
[
|
||||
(128000, mantle_shims.MANTLE_MAX_OUTPUT_TOKENS), # over the ceiling → clamped
|
||||
(None, mantle_shims.MANTLE_MAX_OUTPUT_TOKENS), # unset → pinned to ceiling
|
||||
(4096, 4096), # under the ceiling → untouched
|
||||
],
|
||||
)
|
||||
def test_max_tokens_clamp(
|
||||
monkeypatch: pytest.MonkeyPatch, initial: int | None, expected: int
|
||||
) -> None:
|
||||
from fast_agent.llm.provider.anthropic.llm_anthropic import AnthropicLLM
|
||||
from fast_agent.types import RequestParams
|
||||
|
||||
# Stub the underlying initializer, then force a fresh wrap around it.
|
||||
monkeypatch.setattr(
|
||||
AnthropicLLM,
|
||||
"_initialize_default_params",
|
||||
lambda self, kwargs: RequestParams(maxTokens=initial),
|
||||
)
|
||||
monkeypatch.setattr(mantle_shims, "_max_tokens_clamp_installed", False)
|
||||
mantle_shims.install_max_tokens_clamp()
|
||||
|
||||
params = AnthropicLLM._initialize_default_params(object(), {})
|
||||
assert params.maxTokens == expected
|
||||
|
||||
|
||||
def test_max_tokens_clamp_is_idempotent() -> None:
|
||||
mantle_shims.install_max_tokens_clamp()
|
||||
mantle_shims.install_max_tokens_clamp() # must not raise or re-wrap
|
||||
|
||||
|
||||
# ── maybe_install ────────────────────────────────────────────────────────────
|
||||
|
||||
def test_maybe_install_installs_when_mantle(monkeypatch: pytest.MonkeyPatch) -> None:
|
||||
_INSTALLER_NAMES = [
|
||||
("install_wire_name_prefix", "wire"),
|
||||
("install_tool_use_caller_strip", "tool_use"),
|
||||
("install_fine_grained_tool_streaming_opt_out", "beta_opt_out"),
|
||||
("install_max_tokens_clamp", "max_tokens"),
|
||||
]
|
||||
|
||||
|
||||
def _patch_installers(monkeypatch: pytest.MonkeyPatch) -> list[str]:
|
||||
calls: list[str] = []
|
||||
monkeypatch.setattr(
|
||||
mantle_shims, "install_wire_name_prefix",
|
||||
lambda: calls.append("wire"),
|
||||
)
|
||||
monkeypatch.setattr(
|
||||
mantle_shims, "install_tool_use_caller_strip",
|
||||
lambda: calls.append("tool_use"),
|
||||
)
|
||||
for attr, label in _INSTALLER_NAMES:
|
||||
monkeypatch.setattr(
|
||||
mantle_shims, attr,
|
||||
lambda label=label: calls.append(label),
|
||||
)
|
||||
return calls
|
||||
|
||||
|
||||
def test_maybe_install_installs_when_mantle(monkeypatch: pytest.MonkeyPatch) -> None:
|
||||
calls = _patch_installers(monkeypatch)
|
||||
|
||||
installed = mantle_shims.maybe_install("https://bedrock-mantle.us-east-1.api.aws/anthropic")
|
||||
assert installed is True
|
||||
assert calls == ["wire", "tool_use"]
|
||||
assert calls == ["wire", "tool_use", "beta_opt_out", "max_tokens"]
|
||||
|
||||
|
||||
def test_maybe_install_noop_for_non_mantle(monkeypatch: pytest.MonkeyPatch) -> None:
|
||||
calls: list[str] = []
|
||||
monkeypatch.setattr(
|
||||
mantle_shims, "install_wire_name_prefix",
|
||||
lambda: calls.append("wire"),
|
||||
)
|
||||
monkeypatch.setattr(
|
||||
mantle_shims, "install_tool_use_caller_strip",
|
||||
lambda: calls.append("tool_use"),
|
||||
)
|
||||
calls = _patch_installers(monkeypatch)
|
||||
|
||||
assert mantle_shims.maybe_install("https://api.anthropic.com") is False
|
||||
assert mantle_shims.maybe_install(None) is False
|
||||
|
||||
Reference in New Issue
Block a user