diff --git a/agent/agent_runtime_helpers.py b/agent/agent_runtime_helpers.py index e64fd9ad72309..1d74670952f20 100644 --- a/agent/agent_runtime_helpers.py +++ b/agent/agent_runtime_helpers.py @@ -2215,13 +2215,26 @@ def anthropic_prompt_cache_policy( # api.minimax.io/anthropic / api.minimaxi.com/anthropic) get the # same cost reduction as Claude traffic. # Docs: https://platform.minimax.io/docs/api-reference/anthropic-api-compatible-cache + # + # MiniMax-M3 is intentionally excluded: M3 ships server-side automatic + # prefix caching on this wire format (content-keyed, no marker needed — + # see https://platform.minimax.io/docs/api-reference/text-prompt-caching), + # and cache_control markers are NOT on its explicit-cache support list + # (M2.7/M2.5/M2.1/M2 only). Emitting markers on M3 wasted serialization + # overhead, risked perturbing the server-side prefix hash, and gave users + # a false sense of explicit-cache savings. Empirically verified against + # api.minimaxi.com/anthropic/v1/messages with MiniMax-M3[1m]: identical + # system prompt hit-rate with and without markers; cache_read field has + # a +128 floor and cache_creation is always 0, so the marker path is + # neither observable nor billable for M3 users. if is_anthropic_wire: is_minimax_provider = provider_lower in {"minimax", "minimax-cn"} is_minimax_host = ( base_url_host_matches(eff_base_url, "api.minimax.io") or base_url_host_matches(eff_base_url, "api.minimaxi.com") ) - if is_minimax_provider or is_minimax_host: + is_minimax_m3 = "minimax-m3" in model_lower + if (is_minimax_provider or is_minimax_host) and not is_minimax_m3: return True, True # Qwen/Alibaba on OpenCode (Zen/Go) and native DashScope: OpenAI-wire diff --git a/agent/usage_pricing.py b/agent/usage_pricing.py index 04d34d13372c9..8b6875028d7f3 100644 --- a/agent/usage_pricing.py +++ b/agent/usage_pricing.py @@ -1,5 +1,6 @@ from __future__ import annotations +import logging import re from dataclasses import dataclass from datetime import datetime, timezone @@ -9,6 +10,8 @@ from typing import Any, Dict, Literal, Optional from agent.model_metadata import fetch_endpoint_model_metadata, fetch_model_metadata from utils import base_url_host_matches +logger = logging.getLogger(__name__) + DEFAULT_PRICING = {"input": 0.0, "output": 0.0} _ZERO = Decimal("0") @@ -1207,6 +1210,7 @@ def normalize_usage( *, provider: Optional[str] = None, api_mode: Optional[str] = None, + debug: bool = False, ) -> CanonicalUsage: """Normalize raw API response usage into canonical token buckets. @@ -1288,6 +1292,26 @@ def normalize_usage( getattr(completion_details, "reasoning_tokens", 0) ) + # NOTE: opt-in cache observability for MiniMax-M3 (and similar providers + # whose usage.cache_read_input_tokens carries a constant +128 floor and + # whose usage.cache_creation_input_tokens is always 0). See + # https://platform.minimax.io/docs/api-reference/text-prompt-caching + # (Automatic Caching table). On M3, the cache_read field is NOT a + # reliable hit signal; the only signal that survives is the input_tokens + # drop between consecutive calls. This debug block logs the + # observable-only fields so an operator can confirm cache is working + # without relying on the misleading cache_read number. + if debug and (mode == "anthropic_messages" or provider_name in {"minimax", "minimax-cn"}): + logger.debug( + "cache_observability provider=%s mode=%s input_tokens=%s " + "output_tokens=%s cache_read_tokens=%s cache_write_tokens=%s " + "(note: cache_read on this provider carries a +128 constant " + "floor and is not a reliable hit signal — track input_tokens " + "drops across calls instead)", + provider_name, mode, input_tokens, output_tokens, + cache_read_tokens, cache_write_tokens, + ) + return CanonicalUsage( input_tokens=input_tokens, output_tokens=output_tokens, diff --git a/tests/agent/test_usage_pricing.py b/tests/agent/test_usage_pricing.py index 3d5228d19ae6d..0ae4187491e6f 100644 --- a/tests/agent/test_usage_pricing.py +++ b/tests/agent/test_usage_pricing.py @@ -312,3 +312,85 @@ def test_vertex_default_model_estimates_cached_usage(monkeypatch): assert result.status == "estimated" assert result.amount_usd is not None and result.amount_usd > 0 + + +def test_normalize_usage_debug_off_emits_no_log(caplog): + """The opt-in cache observability block must NOT log when debug=False + (the default). Production callers should never see this log line — + it is only useful when an operator explicitly opts in to investigate + cache behavior on MiniMax-M3 or any other provider whose + cache_read_input_tokens carries a misleading constant offset. + """ + usage = SimpleNamespace( + input_tokens=53, + output_tokens=10, + cache_read_input_tokens=128, + cache_creation_input_tokens=0, + ) + + with caplog.at_level("DEBUG", logger="agent.usage_pricing"): + normalize_usage(usage, provider="minimax-cn", api_mode="anthropic_messages") + + assert all("cache_observability" not in rec.message for rec in caplog.records) + + +def test_normalize_usage_debug_on_minimax_logs_cache_observability(caplog): + """When debug=True is passed and the provider is MiniMax/M3, + normalize_usage emits a debug log line that records the observable-only + fields (input_tokens, output_tokens, cache_read_tokens, + cache_write_tokens) so an operator can see real cache behavior + without trusting the misleading cache_read number. + """ + usage = SimpleNamespace( + input_tokens=1, + output_tokens=11, + cache_read_input_tokens=8594, + cache_creation_input_tokens=0, + ) + + with caplog.at_level("DEBUG", logger="agent.usage_pricing"): + normalize_usage( + usage, + provider="minimax-cn", + api_mode="anthropic_messages", + debug=True, + ) + + cache_obs_records = [r for r in caplog.records if "cache_observability" in r.message] + assert len(cache_obs_records) == 1 + record = cache_obs_records[0] + assert "input_tokens=1" in record.message + assert "output_tokens=11" in record.message + assert "cache_read_tokens=8594" in record.message + assert "cache_write_tokens=0" in record.message + assert "+128 constant floor" in record.message + + +def test_normalize_usage_debug_on_claude_also_logs_cache_observability(caplog): + """The opt-in observability block fires for every anthropic_messages + response — not just MiniMax — because the input_tokens-vs-cache_read + framing is useful diagnostic information on any Anthropic-compatible + wire (e.g. OpenRouter Claude, Bedrock Claude, GLM Claude) where a + provider's cache_read_input_tokens semantics may differ from + Anthropic's native contract. The block is debug-level and off by + default, so emitting it for Claude traffic has zero production cost. + """ + usage = SimpleNamespace( + input_tokens=100, + output_tokens=20, + cache_read_input_tokens=50, + cache_creation_input_tokens=10, + ) + + with caplog.at_level("DEBUG", logger="agent.usage_pricing"): + normalize_usage( + usage, + provider="anthropic", + api_mode="anthropic_messages", + debug=True, + ) + + cache_obs_records = [r for r in caplog.records if "cache_observability" in r.message] + assert len(cache_obs_records) == 1 + assert "input_tokens=100" in cache_obs_records[0].message + assert "cache_read_tokens=50" in cache_obs_records[0].message diff --git a/tests/run_agent/test_anthropic_prompt_cache_policy.py b/tests/run_agent/test_anthropic_prompt_cache_policy.py index 2c4b44e9c0932..2852393e70593 100644 --- a/tests/run_agent/test_anthropic_prompt_cache_policy.py +++ b/tests/run_agent/test_anthropic_prompt_cache_policy.py @@ -211,6 +211,61 @@ class TestMiniMaxAnthropicWire: ) assert agent._anthropic_prompt_cache_policy() == (False, False) + def test_minimax_m3_on_provider_minimax_does_not_cache(self): + # MiniMax-M3 uses server-side automatic prefix caching on the + # /anthropic wire (content-keyed, no marker needed). M3 is NOT on + # MiniMax's explicit-cache support list (which covers only M2.7 / + # M2.5 / M2.1 / M2), and emitting cache_control markers on M3 is + # neither observable nor billable — it only wastes serialization + # overhead and risks perturbing the server-side prefix hash. Marker + # path must stay off for M3 so the response.usage fields reflect + # server-side automatic caching without interference. + agent = _make_agent( + provider="minimax", + base_url="https://api.minimax.io/anthropic", + api_mode="anthropic_messages", + model="MiniMax-M3[1m]", + ) + assert agent._anthropic_prompt_cache_policy() == (False, False) + + def test_minimax_m3_on_china_endpoint_does_not_cache(self): + # Mirror of the above against the China-region host. The + # M3-vs-M2 substring guard must trigger on the model name + # regardless of which MiniMax host the user picks. + agent = _make_agent( + provider="minimax-cn", + base_url="https://api.minimaxi.com/anthropic", + api_mode="anthropic_messages", + model="MiniMax-M3", + ) + assert agent._anthropic_prompt_cache_policy() == (False, False) + + def test_minimax_m3_via_custom_provider_does_not_cache(self): + # When the user wires a custom provider manually at MiniMax's + # Anthropic URL with M3, host-match alone must NOT bypass the + # M3-specific opt-out. + agent = _make_agent( + provider="custom", + base_url="https://api.minimaxi.com/anthropic", + api_mode="anthropic_messages", + model="MiniMax-M3[1m]", + ) + assert agent._anthropic_prompt_cache_policy() == (False, False) + + def test_minimax_m27_still_caches_after_m3_opt_out(self): + # Regression guard: the M3 substring check must not collide with + # M2.7 / M2.5 / M2.1 / M2 model names. "minimax-m3" is not a + # substring of "minimax-m2.7" etc., but pin this with a test so a + # future "startswith minimax-m" loosening can't silently drop the + # M2.x cache_control path. + agent = _make_agent( + provider="minimax", + base_url="https://api.minimax.io/anthropic", + api_mode="anthropic_messages", + model="MiniMax-M2.7", + ) + assert agent._anthropic_prompt_cache_policy() == (True, True) + class TestOpenAIWireFormatOnCustomProvider: """A custom provider using chat_completions (OpenAI wire) should NOT get caching."""