Compare commits
5 Commits
fix/tool-d
...
84e026c2e7
| Author | SHA1 | Date | |
|---|---|---|---|
| 84e026c2e7 | |||
| d8703a1ad6 | |||
| 537f3c7963 | |||
| b5a3aa214b | |||
| d528192bee |
@@ -218,7 +218,7 @@ anthropic:
|
|||||||
|
|
||||||
That's the whole configuration. Pallas auto-detects the
|
That's the whole configuration. Pallas auto-detects the
|
||||||
`bedrock-mantle` hostname in `anthropic.base_url` at startup and installs
|
`bedrock-mantle` hostname in `anthropic.base_url` at startup and installs
|
||||||
two compatibility shims so fast-agent's default request shape matches
|
four compatibility shims so fast-agent's default request shape matches
|
||||||
what Mantle expects (see `pallas/mantle_shims.py`):
|
what Mantle expects (see `pallas/mantle_shims.py`):
|
||||||
|
|
||||||
1. **Wire-name prefix** — re-adds the `anthropic.` prefix that fast-agent's
|
1. **Wire-name prefix** — re-adds the `anthropic.` prefix that fast-agent's
|
||||||
@@ -233,6 +233,25 @@ what Mantle expects (see `pallas/mantle_shims.py`):
|
|||||||
Input should be a valid dictionary or object"`, which would otherwise
|
Input should be a valid dictionary or object"`, which would otherwise
|
||||||
break the MCP tool-use loop on the second turn.
|
break the MCP tool-use loop on the second turn.
|
||||||
|
|
||||||
|
3. **Fine-grained tool streaming opt-out** — stops fast-agent sending the
|
||||||
|
`fine-grained-tool-streaming-2025-05-14` beta. Under that beta an
|
||||||
|
output-token cutoff mid-`tool_use` block ends the stream without
|
||||||
|
`content_block_stop`, which fast-agent surfaces as
|
||||||
|
`Streaming completed but tool call never finished` and then retries
|
||||||
|
into the same wall (~700 s agent failures on large tool bodies).
|
||||||
|
Without the beta a cutoff closes blocks properly and lands in
|
||||||
|
fast-agent's graceful `stop_reason=max_tokens` handling.
|
||||||
|
|
||||||
|
4. **`max_tokens` clamp** — Mantle enforces a server-side output ceiling
|
||||||
|
of 20 000 tokens per response regardless of the requested
|
||||||
|
`max_tokens` (fast-agent asks for the model's full 128 000). The shim
|
||||||
|
clamps default `maxTokens` to `MANTLE_MAX_OUTPUT_TOKENS` (20 000) so
|
||||||
|
the model stops gracefully at the limit instead of the gateway
|
||||||
|
cutting the stream. A single agent turn — thinking, prose, and tool
|
||||||
|
input combined — cannot exceed this on Mantle; agents that must emit
|
||||||
|
more output in one turn need to split the work (e.g. patch-style
|
||||||
|
edits instead of full-document rewrites).
|
||||||
|
|
||||||
The Anthropic SDK appends `/v1/messages` to `base_url` automatically.
|
The Anthropic SDK appends `/v1/messages` to `base_url` automatically.
|
||||||
|
|
||||||
**Feature support.** Mantle accepts the same Messages API request shape
|
**Feature support.** Mantle accepts the same Messages API request shape
|
||||||
|
|||||||
@@ -24,7 +24,23 @@ before the wire traffic is valid:
|
|||||||
|
|
||||||
Upstream SDK tracker: https://github.com/anthropics/anthropic-sdk-python/issues/1454
|
Upstream SDK tracker: https://github.com/anthropics/anthropic-sdk-python/issues/1454
|
||||||
|
|
||||||
Both shims are idempotent and may be installed at process startup before any
|
3. **Fine-grained tool streaming truncation.** Fast-agent unconditionally
|
||||||
|
sends the ``fine-grained-tool-streaming-2025-05-14`` beta on tool-bearing
|
||||||
|
requests. Under that beta, an output-token cutoff mid-``tool_use`` block
|
||||||
|
ends the stream *without* ``content_block_stop``, which fast-agent's
|
||||||
|
stream accounting surfaces as ``Streaming completed but tool call never
|
||||||
|
finished`` — followed by a full retry ladder against the same wall
|
||||||
|
(observed as ~700 s agent failures on large ``revise_workspace_file``
|
||||||
|
bodies). We disable that one beta so a cutoff closes blocks properly and
|
||||||
|
lands in fast-agent's graceful ``stop_reason=max_tokens`` handling.
|
||||||
|
|
||||||
|
4. **Output-token ceiling.** Mantle clamps ``max_tokens`` to 20 000
|
||||||
|
server-side (streams complete at exactly 20 000 output tokens regardless
|
||||||
|
of the requested 128 000). We clamp the request to that ceiling so the
|
||||||
|
*model* stops gracefully at the limit — emitting proper block-stop events
|
||||||
|
and ``stop_reason`` — instead of being cut off by the gateway's clamp.
|
||||||
|
|
||||||
|
All shims are idempotent and may be installed at process startup before any
|
||||||
fast-agent ``FastAgent`` instance is constructed.
|
fast-agent ``FastAgent`` instance is constructed.
|
||||||
"""
|
"""
|
||||||
from __future__ import annotations
|
from __future__ import annotations
|
||||||
@@ -129,12 +145,104 @@ def install_tool_use_caller_strip() -> None:
|
|||||||
logger.info("Mantle tool_use.caller strip shim installed")
|
logger.info("Mantle tool_use.caller strip shim installed")
|
||||||
|
|
||||||
|
|
||||||
|
# ── Shim 3: disable fine-grained tool streaming ──────────────────────────────
|
||||||
|
|
||||||
|
_beta_opt_out_installed = False
|
||||||
|
|
||||||
|
|
||||||
|
def install_fine_grained_tool_streaming_opt_out() -> None:
|
||||||
|
"""Stop fast-agent requesting the fine-grained tool streaming beta.
|
||||||
|
|
||||||
|
Under that beta a ``max_tokens`` cutoff mid-``tool_use`` ends the stream
|
||||||
|
without ``content_block_stop``; fast-agent then raises
|
||||||
|
``Streaming completed but tool call never finished`` and burns its whole
|
||||||
|
retry ladder against the same ceiling. Without the beta the cutoff closes
|
||||||
|
blocks properly and fast-agent's ``stop_reason=max_tokens`` handling
|
||||||
|
applies. Safe to call more than once.
|
||||||
|
"""
|
||||||
|
global _beta_opt_out_installed
|
||||||
|
if _beta_opt_out_installed:
|
||||||
|
return
|
||||||
|
|
||||||
|
from fast_agent.llm.provider.anthropic.llm_anthropic import AnthropicLLM
|
||||||
|
|
||||||
|
original_supports = AnthropicLLM.supports_direct_anthropic_beta
|
||||||
|
|
||||||
|
def patched_supports(self: Any, feature: str) -> bool:
|
||||||
|
if feature == "fine_grained_tool_streaming":
|
||||||
|
return False
|
||||||
|
return original_supports(self, feature)
|
||||||
|
|
||||||
|
AnthropicLLM.supports_direct_anthropic_beta = patched_supports # type: ignore[method-assign]
|
||||||
|
|
||||||
|
_beta_opt_out_installed = True
|
||||||
|
logger.info("Mantle fine-grained tool streaming opt-out shim installed")
|
||||||
|
|
||||||
|
|
||||||
|
# ── Shim 4: clamp max_tokens to Mantle's output ceiling ──────────────────────
|
||||||
|
|
||||||
|
# Observed server-side clamp: Mantle streams stop at exactly 20 000 output
|
||||||
|
# tokens however large the requested max_tokens. Requesting the ceiling
|
||||||
|
# explicitly makes the model stop gracefully (proper block close + stop_reason)
|
||||||
|
# instead of the gateway cutting the stream at its own limit.
|
||||||
|
MANTLE_MAX_OUTPUT_TOKENS = 20_000
|
||||||
|
|
||||||
|
_max_tokens_clamp_installed = False
|
||||||
|
|
||||||
|
|
||||||
|
def install_max_tokens_clamp() -> None:
|
||||||
|
"""Clamp default ``maxTokens`` to Mantle's output ceiling.
|
||||||
|
|
||||||
|
Wraps ``AnthropicLLM._initialize_default_params`` so every agent's
|
||||||
|
default request params carry an explicit ``maxTokens`` no higher than the
|
||||||
|
ceiling. Per-request ``RequestParams`` overrides bypass this — none of
|
||||||
|
our agent modules set one. Safe to call more than once.
|
||||||
|
"""
|
||||||
|
global _max_tokens_clamp_installed
|
||||||
|
if _max_tokens_clamp_installed:
|
||||||
|
return
|
||||||
|
|
||||||
|
from fast_agent.llm.provider.anthropic.llm_anthropic import AnthropicLLM
|
||||||
|
|
||||||
|
original_init = AnthropicLLM._initialize_default_params # noqa: SLF001
|
||||||
|
|
||||||
|
def patched_init(self: Any, kwargs: dict) -> Any:
|
||||||
|
params = original_init(self, kwargs)
|
||||||
|
if params.maxTokens is None or params.maxTokens > MANTLE_MAX_OUTPUT_TOKENS:
|
||||||
|
params.maxTokens = MANTLE_MAX_OUTPUT_TOKENS
|
||||||
|
return params
|
||||||
|
|
||||||
|
AnthropicLLM._initialize_default_params = patched_init # noqa: SLF001
|
||||||
|
|
||||||
|
_max_tokens_clamp_installed = True
|
||||||
|
logger.info(
|
||||||
|
"Mantle max_tokens clamp shim installed (ceiling %d)",
|
||||||
|
MANTLE_MAX_OUTPUT_TOKENS,
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
# ── Orchestrator ─────────────────────────────────────────────────────────────
|
# ── Orchestrator ─────────────────────────────────────────────────────────────
|
||||||
|
|
||||||
def install_all() -> None:
|
def install_all() -> None:
|
||||||
"""Install all Mantle shims. Call once at process startup."""
|
"""Install all Mantle shims. Call once at process startup.
|
||||||
|
|
||||||
|
``install_max_tokens_clamp`` is deliberately NOT installed. Its 20 000
|
||||||
|
ceiling was an empirical observation, never a documented Mantle limit,
|
||||||
|
and re-investigation could not establish what enforces it: fast-agent
|
||||||
|
carries no 20 000 default anywhere, no model overlay is configured, and
|
||||||
|
``ModelDatabase`` reports ``max_output_tokens=128000`` for opus-4-8.
|
||||||
|
Hardcoding the constant would cement a ceiling we cannot prove and would
|
||||||
|
silently truncate turns that might otherwise complete. The shim is kept
|
||||||
|
below so it can be re-enabled if the limit is ever confirmed.
|
||||||
|
|
||||||
|
The fine-grained-tool-streaming opt-out is what actually fixes the
|
||||||
|
``Streaming completed but tool call never finished`` crash loop: it costs
|
||||||
|
no output length, it only lets a cutoff close its blocks properly and
|
||||||
|
land in fast-agent's graceful ``stop_reason=max_tokens`` handling.
|
||||||
|
"""
|
||||||
install_wire_name_prefix()
|
install_wire_name_prefix()
|
||||||
install_tool_use_caller_strip()
|
install_tool_use_caller_strip()
|
||||||
|
install_fine_grained_tool_streaming_opt_out()
|
||||||
|
|
||||||
|
|
||||||
def maybe_install(anthropic_base_url: str | None) -> bool:
|
def maybe_install(anthropic_base_url: str | None) -> bool:
|
||||||
|
|||||||
@@ -114,34 +114,99 @@ def test_install_tool_use_caller_strip_is_idempotent() -> None:
|
|||||||
mantle_shims.install_tool_use_caller_strip() # must not raise or re-wrap
|
mantle_shims.install_tool_use_caller_strip() # must not raise or re-wrap
|
||||||
|
|
||||||
|
|
||||||
|
# ── install_fine_grained_tool_streaming_opt_out ──────────────────────────────
|
||||||
|
|
||||||
|
def test_fine_grained_beta_opt_out() -> None:
|
||||||
|
from fast_agent.llm.provider.anthropic.llm_anthropic import AnthropicLLM
|
||||||
|
|
||||||
|
mantle_shims.install_fine_grained_tool_streaming_opt_out()
|
||||||
|
|
||||||
|
# The patched method never touches self, so a bare object suffices.
|
||||||
|
stub = object()
|
||||||
|
assert AnthropicLLM.supports_direct_anthropic_beta(stub, "fine_grained_tool_streaming") is False
|
||||||
|
# Every other beta keeps the base-class answer (True).
|
||||||
|
assert AnthropicLLM.supports_direct_anthropic_beta(stub, "interleaved_thinking") is True
|
||||||
|
assert AnthropicLLM.supports_direct_anthropic_beta(stub, "long_context") is True
|
||||||
|
|
||||||
|
|
||||||
|
def test_fine_grained_beta_opt_out_is_idempotent() -> None:
|
||||||
|
mantle_shims.install_fine_grained_tool_streaming_opt_out()
|
||||||
|
mantle_shims.install_fine_grained_tool_streaming_opt_out() # must not re-wrap
|
||||||
|
|
||||||
|
from fast_agent.llm.provider.anthropic.llm_anthropic import AnthropicLLM
|
||||||
|
|
||||||
|
assert (
|
||||||
|
AnthropicLLM.supports_direct_anthropic_beta(object(), "fine_grained_tool_streaming")
|
||||||
|
is False
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
# ── install_max_tokens_clamp ─────────────────────────────────────────────────
|
||||||
|
|
||||||
|
@pytest.mark.parametrize(
|
||||||
|
"initial,expected",
|
||||||
|
[
|
||||||
|
(128000, mantle_shims.MANTLE_MAX_OUTPUT_TOKENS), # over the ceiling → clamped
|
||||||
|
(None, mantle_shims.MANTLE_MAX_OUTPUT_TOKENS), # unset → pinned to ceiling
|
||||||
|
(4096, 4096), # under the ceiling → untouched
|
||||||
|
],
|
||||||
|
)
|
||||||
|
def test_max_tokens_clamp(
|
||||||
|
monkeypatch: pytest.MonkeyPatch, initial: int | None, expected: int
|
||||||
|
) -> None:
|
||||||
|
from fast_agent.llm.provider.anthropic.llm_anthropic import AnthropicLLM
|
||||||
|
from fast_agent.types import RequestParams
|
||||||
|
|
||||||
|
# Stub the underlying initializer, then force a fresh wrap around it.
|
||||||
|
monkeypatch.setattr(
|
||||||
|
AnthropicLLM,
|
||||||
|
"_initialize_default_params",
|
||||||
|
lambda self, kwargs: RequestParams(maxTokens=initial),
|
||||||
|
)
|
||||||
|
monkeypatch.setattr(mantle_shims, "_max_tokens_clamp_installed", False)
|
||||||
|
mantle_shims.install_max_tokens_clamp()
|
||||||
|
|
||||||
|
params = AnthropicLLM._initialize_default_params(object(), {})
|
||||||
|
assert params.maxTokens == expected
|
||||||
|
|
||||||
|
|
||||||
|
def test_max_tokens_clamp_is_idempotent() -> None:
|
||||||
|
mantle_shims.install_max_tokens_clamp()
|
||||||
|
mantle_shims.install_max_tokens_clamp() # must not raise or re-wrap
|
||||||
|
|
||||||
|
|
||||||
# ── maybe_install ────────────────────────────────────────────────────────────
|
# ── maybe_install ────────────────────────────────────────────────────────────
|
||||||
|
|
||||||
def test_maybe_install_installs_when_mantle(monkeypatch: pytest.MonkeyPatch) -> None:
|
_INSTALLER_NAMES = [
|
||||||
|
("install_wire_name_prefix", "wire"),
|
||||||
|
("install_tool_use_caller_strip", "tool_use"),
|
||||||
|
("install_fine_grained_tool_streaming_opt_out", "beta_opt_out"),
|
||||||
|
("install_max_tokens_clamp", "max_tokens"),
|
||||||
|
]
|
||||||
|
|
||||||
|
|
||||||
|
def _patch_installers(monkeypatch: pytest.MonkeyPatch) -> list[str]:
|
||||||
calls: list[str] = []
|
calls: list[str] = []
|
||||||
monkeypatch.setattr(
|
for attr, label in _INSTALLER_NAMES:
|
||||||
mantle_shims, "install_wire_name_prefix",
|
monkeypatch.setattr(
|
||||||
lambda: calls.append("wire"),
|
mantle_shims, attr,
|
||||||
)
|
lambda label=label: calls.append(label),
|
||||||
monkeypatch.setattr(
|
)
|
||||||
mantle_shims, "install_tool_use_caller_strip",
|
return calls
|
||||||
lambda: calls.append("tool_use"),
|
|
||||||
)
|
|
||||||
|
def test_maybe_install_installs_when_mantle(monkeypatch: pytest.MonkeyPatch) -> None:
|
||||||
|
calls = _patch_installers(monkeypatch)
|
||||||
|
|
||||||
installed = mantle_shims.maybe_install("https://bedrock-mantle.us-east-1.api.aws/anthropic")
|
installed = mantle_shims.maybe_install("https://bedrock-mantle.us-east-1.api.aws/anthropic")
|
||||||
assert installed is True
|
assert installed is True
|
||||||
assert calls == ["wire", "tool_use"]
|
# "max_tokens" is intentionally absent: the 20 000 clamp is not installed
|
||||||
|
# by default because the ceiling was never confirmed. See install_all().
|
||||||
|
assert calls == ["wire", "tool_use", "beta_opt_out"]
|
||||||
|
|
||||||
|
|
||||||
def test_maybe_install_noop_for_non_mantle(monkeypatch: pytest.MonkeyPatch) -> None:
|
def test_maybe_install_noop_for_non_mantle(monkeypatch: pytest.MonkeyPatch) -> None:
|
||||||
calls: list[str] = []
|
calls = _patch_installers(monkeypatch)
|
||||||
monkeypatch.setattr(
|
|
||||||
mantle_shims, "install_wire_name_prefix",
|
|
||||||
lambda: calls.append("wire"),
|
|
||||||
)
|
|
||||||
monkeypatch.setattr(
|
|
||||||
mantle_shims, "install_tool_use_caller_strip",
|
|
||||||
lambda: calls.append("tool_use"),
|
|
||||||
)
|
|
||||||
|
|
||||||
assert mantle_shims.maybe_install("https://api.anthropic.com") is False
|
assert mantle_shims.maybe_install("https://api.anthropic.com") is False
|
||||||
assert mantle_shims.maybe_install(None) is False
|
assert mantle_shims.maybe_install(None) is False
|
||||||
|
|||||||
Reference in New Issue
Block a user