From 8ee51747fe52fefe428e3d174a43bd2001e5bdde Mon Sep 17 00:00:00 2001 From: 686f6c61 Date: Sat, 1 Aug 2026 14:30:30 +0200 Subject: [PATCH] fix(agent): consolidate cache-disable stubs with blank_cache_policy_stub Absorb the useful deltas from the parallel #76121 approach: a single blank_cache_policy_stub factory so _cache_disabled cannot be left off hand-rolled SimpleNamespaces, and pin the live agent disable onto MoA advisor fan-out and one-shot aggregate_moa_context decoration so those paths track conversation state rather than a fresh config re-read. Keeps the earlier tri-state prepared-aggregator no-agent fix. Adds factory and synthesis/advisor regressions. Coordinates with #76121 / #76085. Co-authored-by: JoaoMarcos44 <87440198+JoaoMarcos44@users.noreply.github.com> --- agent/agent_runtime_helpers.py | 37 +++++-- agent/moa_loop.py | 44 +++++--- tests/agent/test_cache_disabled_on_stubs.py | 109 ++++++++++++++++++++ 3 files changed, 163 insertions(+), 27 deletions(-) diff --git a/agent/agent_runtime_helpers.py b/agent/agent_runtime_helpers.py index 88423be486e02..20e8295cc1cd7 100644 --- a/agent/agent_runtime_helpers.py +++ b/agent/agent_runtime_helpers.py @@ -1878,6 +1878,30 @@ def prompt_caching_disabled_from_config() -> bool: return str(ttl).lower() in ("off", "false", "disabled", "no", "none") +def blank_cache_policy_stub(cache_disabled: Optional[bool] = None): + """Build the destination-identity-blank stub for ``anthropic_prompt_cache_policy``. + + Single sanctioned constructor for that stub. Callers that resolve cache + policy against a destination identified out-of-band (not a live + ``AIAgent``) must go through here so ``_cache_disabled`` is never left + off a hand-rolled ``SimpleNamespace`` (#76085). + + When ``cache_disabled`` is omitted, falls back to the global config so + stub paths without an agent snapshot still honor an operator disable. + """ + from types import SimpleNamespace + + if cache_disabled is None: + cache_disabled = prompt_caching_disabled_from_config() + return SimpleNamespace( + provider="", + base_url="", + api_mode="", + model="", + _cache_disabled=bool(cache_disabled), + ) + + def plan_cache_sections_for_destination( messages: list, tools: Optional[list], @@ -1905,23 +1929,13 @@ def plan_cache_sections_for_destination( consulted so MoA/auxiliary paths cannot re-enable markers after the user turned caching off (#76085). """ - from types import SimpleNamespace - from agent.prompt_caching import ( build_prompt_cache_plan, strip_anthropic_cache_control, strip_anthropic_tool_cache_control, ) - if cache_disabled is None: - cache_disabled = prompt_caching_disabled_from_config() - stub = SimpleNamespace( - provider="", - base_url="", - api_mode="", - model="", - _cache_disabled=bool(cache_disabled), - ) + stub = blank_cache_policy_stub(cache_disabled) should_cache, native_layout = anthropic_prompt_cache_policy( stub, provider=provider, @@ -3931,6 +3945,7 @@ __all__ = [ "extract_reasoning", "dump_api_request_debug", "prompt_caching_disabled_from_config", + "blank_cache_policy_stub", "plan_cache_sections_for_destination", "anthropic_prompt_cache_policy", "create_openai_client", diff --git a/agent/moa_loop.py b/agent/moa_loop.py index cefef3114d75d..84fa4a5acf366 100644 --- a/agent/moa_loop.py +++ b/agent/moa_loop.py @@ -404,27 +404,21 @@ def _maybe_apply_moa_cache_control( blank-agent pattern (#76085). """ try: - from types import SimpleNamespace - from agent.agent_runtime_helpers import ( anthropic_prompt_cache_policy, - prompt_caching_disabled_from_config, + blank_cache_policy_stub, ) from agent.prompt_caching import apply_anthropic_cache_control - if cache_disabled is None: - cache_disabled = prompt_caching_disabled_from_config() + # Prefer an explicit kwarg, then a snapshot on the runtime dict + # (threaded from the live agent), else config via the stub factory. + if cache_disabled is None and "_cache_disabled" in runtime: + cache_disabled = runtime.get("_cache_disabled") # The policy function reads agent.* only as fallbacks for kwargs we - # don't pass; provide a stub so the slot is judged purely on its own - # resolved runtime (plus the operator disable flag). - stub = SimpleNamespace( - provider="", - base_url="", - api_mode="", - model="", - _cache_disabled=bool(cache_disabled), - ) + # don't pass; blank_cache_policy_stub is the only sanctioned stub + # so _cache_disabled cannot be left off again (#76085). + stub = blank_cache_policy_stub(cache_disabled) should_cache, native_layout = anthropic_prompt_cache_policy( stub, provider=runtime.get("provider") or "", @@ -450,6 +444,7 @@ def _run_reference( max_tokens: int | None = None, reference_timeout: float | None = None, context_length_cache: Any = None, + cache_disabled: bool | None = None, ) -> tuple[str, str, Any]: """Call one reference model and return ``(label, text, accounting)``. @@ -511,7 +506,12 @@ def _run_reference( # caching is opt-in per request. OpenAI-family advisors are untouched # (their caching is automatic; markers are ignored harmlessly, but we # only decorate when the policy says the route honors them). - messages = _maybe_apply_moa_cache_control(messages, runtime) + # Pin the live agent disable onto the runtime so advisor decoration + # tracks conversation state, not a fresh config re-read (#76085). + cache_runtime = runtime + if cache_disabled is not None: + cache_runtime = {**runtime, "_cache_disabled": cache_disabled} + messages = _maybe_apply_moa_cache_control(messages, cache_runtime) # Per-slot max_tokens takes precedence over the preset-level # reference_max_tokens passed in by the caller. This lets each # reference model have its own output cap independently. @@ -815,6 +815,9 @@ def _run_references_parallel( # instead of re-probing metadata sources per reference (dict get/set is # GIL-atomic; a rare duplicate probe on a first-use race is harmless). _ctx_len_cache: dict[tuple[str, str], int | None] = {} + cache_disabled = ( + getattr(agent, "_cache_disabled", None) if agent is not None else None + ) try: for idx, slot in enumerate(reference_models): if slot.get("provider") == "moa": @@ -833,6 +836,7 @@ def _run_references_parallel( max_tokens=max_tokens, reference_timeout=reference_timeout, context_length_cache=_ctx_len_cache, + cache_disabled=cache_disabled, ) ] = idx @@ -1280,6 +1284,14 @@ def aggregate_moa_context( agg_label = _slot_label(aggregator) agg_runtime = _slot_runtime(aggregator) + # Pin the live agent disable onto synthesis decoration so mid-session + # config flips cannot re-enable markers on this path alone (#76085). + agg_cache_runtime = agg_runtime + if agent is not None: + agg_cache_runtime = { + **agg_runtime, + "_cache_disabled": getattr(agent, "_cache_disabled", None), + } try: # Same cache_control decoration as _run_reference's advisor calls # (see _maybe_apply_moa_cache_control) — this synthesis call is a @@ -1292,7 +1304,7 @@ def aggregate_moa_context( # breakpoints, even when the resolved aggregator slot is a # cache-honoring route (e.g. Claude on OpenRouter/native Anthropic). agg_messages = _maybe_apply_moa_cache_control( - [{"role": "user", "content": synth_prompt}], agg_runtime + [{"role": "user", "content": synth_prompt}], agg_cache_runtime ) response = call_llm( task="moa_aggregator", diff --git a/tests/agent/test_cache_disabled_on_stubs.py b/tests/agent/test_cache_disabled_on_stubs.py index cd0844071d7dc..5f685ca097be5 100644 --- a/tests/agent/test_cache_disabled_on_stubs.py +++ b/tests/agent/test_cache_disabled_on_stubs.py @@ -258,3 +258,112 @@ class TestPreparedAggregatorNoAgentConfigOff: ) assert not _has_cache_control(calls[0].get("messages") or []) assert tools == canonical_tools + + +class TestBlankCachePolicyStubFactory: + def test_factory_sets_cache_disabled_from_config(self): + from agent.agent_runtime_helpers import blank_cache_policy_stub + + with patch( + "hermes_cli.config.load_config_readonly", + return_value={"prompt_caching": {"cache_ttl": "off"}}, + ): + stub = blank_cache_policy_stub() + assert stub._cache_disabled is True + + def test_factory_honors_explicit_false(self): + from agent.agent_runtime_helpers import blank_cache_policy_stub + + with patch( + "hermes_cli.config.load_config_readonly", + return_value={"prompt_caching": {"cache_ttl": "off"}}, + ): + stub = blank_cache_policy_stub(False) + assert stub._cache_disabled is False + + +class TestOneShotSynthesisAgentDisable: + """aggregate_moa_context must pin agent._cache_disabled onto decoration + so the one-shot synthesis path cannot re-enable markers mid-session. + """ + + def test_synthesis_untouched_when_agent_disables_cache(self): + from agent import moa_loop + + calls = [] + with ( + patch.object( + moa_loop, + "call_llm", + side_effect=lambda **kwargs: calls.append(kwargs) or SimpleNamespace( + choices=[SimpleNamespace( + message=SimpleNamespace(content="synth", tool_calls=[]), + finish_reason="stop", + )], + usage=None, + model="fake", + ), + ), + patch.object( + moa_loop, + "_run_references_parallel", + return_value=[("advisor-a", "advice from a", None)], + ), + patch.object( + moa_loop, + "_slot_runtime", + return_value={ + "provider": "anthropic", + "model": "claude-opus-4.8", + "base_url": "", + "api_mode": "anthropic_messages", + }, + ), + # Config would enable caching; agent snapshot must win. + patch( + "hermes_cli.config.load_config_readonly", + return_value={"prompt_caching": {"cache_ttl": "5m"}}, + ), + ): + moa_loop.aggregate_moa_context( + user_prompt="what should I do next?", + api_messages=[{"role": "user", "content": "help me plan"}], + reference_models=[{"provider": "openrouter", "model": "openai/gpt-5.5"}], + aggregator={"provider": "anthropic", "model": "claude-opus-4.8"}, + agent=SimpleNamespace(_cache_disabled=True), + ) + + assert calls, "synthesis must still call the LLM" + synth_msgs = calls[0].get("messages") or [] + assert not _has_cache_control(synth_msgs), ( + "agent._cache_disabled must keep the one-shot synthesis " + "message undecorated even on a cache-honoring route" + ) + + +class TestAdvisorRuntimeDisable: + def test_maybe_apply_honors_runtime_cache_disabled_snapshot(self): + from agent.moa_loop import _maybe_apply_moa_cache_control + + messages = [ + {"role": "system", "content": "advisor"}, + {"role": "user", "content": "review"}, + {"role": "assistant", "content": "a1"}, + {"role": "user", "content": "again"}, + ] + with patch( + "hermes_cli.config.load_config_readonly", + return_value={"prompt_caching": {"cache_ttl": "5m"}}, + ): + out = _maybe_apply_moa_cache_control( + messages, + { + "provider": "anthropic", + "model": "claude-sonnet-4-6", + "base_url": "https://api.anthropic.com", + "api_mode": "anthropic_messages", + "_cache_disabled": True, + }, + ) + assert not _has_cache_control(out) + assert out == messages or not _has_cache_control(out)