fix(agent): make auxiliary auto-chain fallback configurable and free-only (#75803)

This commit is contained in:
webtecnica 2026-08-01 01:08:56 -03:00 committed by kshitij
parent f86693c2f9
commit c19c63d9e6
3 changed files with 209 additions and 2 deletions

View File

@ -14,6 +14,16 @@ Resolution order for text tasks (auto mode):
6. Direct API-key providers (z.ai/GLM, Kimi/Moonshot, MiniMax, MiniMax-CN) 6. Direct API-key providers (z.ai/GLM, Kimi/Moonshot, MiniMax, MiniMax-CN)
7. None 7. None
OpenRouter fallback cost guard (issue #75803):
The step-2 OpenRouter fallback model is ``auxiliary.openrouter_model`` in
config.yaml (default: google/gemini-3.6-flash a PAID model). Because
auxiliary traffic is background-shaped (compression, title generation,
session search, vision, web extract), a paid fallback must never be
silent: set ``auxiliary.free_only: true`` to restrict the fallback to
``:free`` SKUs (skipping OpenRouter entirely when the model is not a
``:free`` id), and a one-time WARNING is logged whenever a non-``:free``
model is engaged so no cash lane is reachable without the user knowing.
Resolution order for vision/multimodal tasks (auto mode): Resolution order for vision/multimodal tasks (auto mode):
1. Selected main provider, if it is one of the supported vision backends below 1. Selected main provider, if it is one of the supported vision backends below
2. OpenRouter 2. OpenRouter
@ -2159,8 +2169,104 @@ def _resolve_api_key_provider() -> Tuple[Optional[OpenAI], Optional[str]]:
# ── Provider resolution helpers ───────────────────────────────────────────── # ── Provider resolution helpers ─────────────────────────────────────────────
# ── Issue #75803: paid-lane guard for the OpenRouter auxiliary fallback ─────
# The auto-chain's step-2 OpenRouter fallback historically used a hardcoded
# PAID model (_OPENROUTER_MODEL) with no config surface and no free marker.
# A user whose fallback ladder is deliberately :free/local could still have
# background auxiliary traffic (compression, title generation, session
# search, vision, web extract) land on a real-money OpenRouter lane.
#
# Guards (all best-effort, config-driven):
# * auxiliary.free_only: true — the OpenRouter fallback is skipped unless
# the resolved model is a :free SKU. This is the opt-out.
# * auxiliary.openrouter_model — replaces the hardcoded fallback model
# (e.g. "nvidia/nemotron-3-ultra-550b-a55b:free"). This is the config
# surface.
# * A one-time WARNING is logged whenever the fallback would serve a
# non-:free model, so the paid lane is never silent.
_paid_lane_warned: set = set()
def _is_free_model(model: Optional[str]) -> bool:
"""True when ``model`` is an OpenRouter free SKU (``:free`` suffix)."""
return bool(model) and str(model).strip().endswith(":free")
def _aux_free_only() -> bool:
"""Read ``auxiliary.free_only`` from config.yaml (default False).
When enabled, the auxiliary auto-chain's OpenRouter fallback refuses to
engage a PAID model: if the resolved fallback model is not a ``:free``
SKU, the OpenRouter step is skipped entirely with an explicit WARNING.
Best-effort any config-read failure falls back to the default.
"""
try:
from hermes_cli.config import cfg_get, load_config
val = cfg_get(load_config(), "auxiliary", "free_only", default=False)
return bool(val)
except Exception:
return False
def _aux_openrouter_model() -> str:
"""Resolve the auxiliary OpenRouter fallback model.
``auxiliary.openrouter_model`` in config.yaml wins; otherwise the module
default ``_OPENROUTER_MODEL`` is used. Best-effort a config-read
failure falls back to the default.
"""
try:
from hermes_cli.config import cfg_get, load_config
val = cfg_get(load_config(), "auxiliary", "openrouter_model")
if isinstance(val, str) and val.strip():
return val.strip()
except Exception:
pass
return _OPENROUTER_MODEL
def _warn_paid_lane_once(model: str) -> None:
"""Log a WARNING the first time a non-:free OpenRouter model is engaged.
Auxiliary traffic is background-shaped (compression and title generation
fire inside cron and idle sessions), so a paid lane must never be silent.
Deduplicated per model id to avoid log spam on every aux call.
"""
if model in _paid_lane_warned:
return
_paid_lane_warned.add(model)
logger.warning(
"Auxiliary client: PAID lane engaged for auxiliary task — OpenRouter "
"fallback model %r is not a :free SKU and may incur real spend. Set "
"auxiliary.free_only: true to restrict auxiliary fallbacks to free "
"models, or auxiliary.openrouter_model to a :free model.",
model,
)
def _try_openrouter(explicit_api_key: str = None, model: str = None) -> Tuple[Optional[OpenAI], Optional[str]]: def _try_openrouter(explicit_api_key: str = None, model: str = None) -> Tuple[Optional[OpenAI], Optional[str]]:
# Issue #75803: never let the OpenRouter fallback silently engage a PAID
# lane. The fallback model is configurable (auxiliary.openrouter_model)
# and auxiliary.free_only: true skips OpenRouter unless the model is a
# :free SKU. Explicit caller models (auxiliary.<task>.model) go through
# the same guard so the opt-out covers every OpenRouter aux path.
or_model = model or _aux_openrouter_model()
if _aux_free_only() and not _is_free_model(or_model):
logger.warning(
"Auxiliary client: auxiliary.free_only is enabled but the "
"OpenRouter fallback model %r is not a :free SKU — skipping the "
"OpenRouter fallback. Set auxiliary.openrouter_model to a :free "
"model (e.g. nvidia/nemotron-3-ultra-550b-a55b:free) or disable "
"auxiliary.free_only.",
or_model,
)
_mark_provider_unhealthy("openrouter", ttl=60)
return None, None
if not _is_free_model(or_model):
_warn_paid_lane_once(or_model)
pool_present, entry = _select_pool_entry("openrouter") pool_present, entry = _select_pool_entry("openrouter")
if pool_present: if pool_present:
or_key = explicit_api_key or _pool_runtime_api_key(entry) or_key = explicit_api_key or _pool_runtime_api_key(entry)
@ -2168,7 +2274,7 @@ def _try_openrouter(explicit_api_key: str = None, model: str = None) -> Tuple[Op
base_url = _pool_runtime_base_url(entry, OPENROUTER_BASE_URL) or OPENROUTER_BASE_URL base_url = _pool_runtime_base_url(entry, OPENROUTER_BASE_URL) or OPENROUTER_BASE_URL
logger.debug("Auxiliary client: OpenRouter via pool") logger.debug("Auxiliary client: OpenRouter via pool")
return _create_openai_client(api_key=or_key, base_url=base_url, return _create_openai_client(api_key=or_key, base_url=base_url,
default_headers=build_or_headers()), model or _OPENROUTER_MODEL default_headers=build_or_headers()), or_model
# Pool exists but is exhausted (no usable runtime key) — fall through to # Pool exists but is exhausted (no usable runtime key) — fall through to
# the OPENROUTER_API_KEY env-var path rather than failing outright. # the OPENROUTER_API_KEY env-var path rather than failing outright.
logger.debug("Auxiliary client: OpenRouter pool exhausted, trying OPENROUTER_API_KEY") logger.debug("Auxiliary client: OpenRouter pool exhausted, trying OPENROUTER_API_KEY")
@ -2179,7 +2285,7 @@ def _try_openrouter(explicit_api_key: str = None, model: str = None) -> Tuple[Op
return None, None return None, None
logger.debug("Auxiliary client: OpenRouter") logger.debug("Auxiliary client: OpenRouter")
return _create_openai_client(api_key=or_key, base_url=OPENROUTER_BASE_URL, return _create_openai_client(api_key=or_key, base_url=OPENROUTER_BASE_URL,
default_headers=build_or_headers()), model or _OPENROUTER_MODEL default_headers=build_or_headers()), or_model
def _describe_openrouter_unavailable() -> str: def _describe_openrouter_unavailable() -> str:

View File

@ -802,6 +802,20 @@ DEFAULT_CONFIG = {
# not a meaningful recovery, so an unretried blip silently loses the # not a meaningful recovery, so an unretried blip silently loses the
# call. # call.
"transient_retries": 2, "transient_retries": 2,
# Restrict the auxiliary auto-chain's OpenRouter fallback to free
# (:free) SKUs. When true, the OpenRouter step is skipped entirely
# unless the resolved fallback model ends in ":free" — a PAID lane
# is never engaged for background auxiliary traffic (compression,
# title generation, session search, vision, web extract) even when
# OPENROUTER_API_KEY is present. Default false keeps the historical
# paid fallback for users who want it.
"free_only": False,
# Override the auxiliary auto-chain's OpenRouter fallback model
# (default: google/gemini-3.6-flash, a PAID model). Set e.g.
# "nvidia/nemotron-3-ultra-550b-a55b:free" together with
# free_only: true to keep auxiliary traffic free-only. A one-time
# WARNING is logged whenever a non-":free" model is engaged.
"openrouter_model": "",
# Endpoints that reject NON-streaming chat requests outright (e.g. # Endpoints that reject NON-streaming chat requests outright (e.g.
# Tencent Copilot returns HTTP 400 "Non-stream chat request is # Tencent Copilot returns HTTP 400 "Non-stream chat request is
# currently not supported"). Auxiliary calls to a matching endpoint # currently not supported"). Auxiliary calls to a matching endpoint

View File

@ -989,6 +989,93 @@ class TestExplicitProviderRouting:
assert mock_openai.call_args.kwargs["base_url"] == OPENROUTER_BASE_URL assert mock_openai.call_args.kwargs["base_url"] == OPENROUTER_BASE_URL
class TestOpenRouterPaidLaneGuard:
"""Issue #75803: auxiliary auto-chain OpenRouter fallback must be
configurable and never silently engage a PAID model."""
def test_free_only_skips_paid_default_model(self, monkeypatch):
"""free_only=true + default (paid) model → OpenRouter skipped."""
monkeypatch.setenv("OPENROUTER_API_KEY", "or-key")
with patch("agent.auxiliary_client._select_pool_entry", return_value=(False, None)), \
patch("hermes_cli.config.load_config", return_value={"auxiliary": {"free_only": True}}), \
patch("agent.auxiliary_client.OpenAI") as mock_openai:
client, model = _try_openrouter()
assert client is None
assert model is None
mock_openai.assert_not_called()
def test_free_only_allows_free_model(self, monkeypatch):
"""free_only=true + :free model → OpenRouter used with that model."""
monkeypatch.setenv("OPENROUTER_API_KEY", "or-key")
with patch("agent.auxiliary_client._select_pool_entry", return_value=(False, None)), \
patch("hermes_cli.config.load_config",
return_value={"auxiliary": {"free_only": True,
"openrouter_model": "nvidia/nemotron-3-ultra-550b-a55b:free"}}), \
patch("agent.auxiliary_client.OpenAI") as mock_openai:
mock_client = MagicMock(name="openrouter_client")
mock_openai.return_value = mock_client
client, model = _try_openrouter()
assert client is mock_client
assert model == "nvidia/nemotron-3-ultra-550b-a55b:free"
def test_configured_model_overrides_hardcoded_default(self, monkeypatch):
"""auxiliary.openrouter_model replaces _OPENROUTER_MODEL."""
monkeypatch.setenv("OPENROUTER_API_KEY", "or-key")
with patch("agent.auxiliary_client._select_pool_entry", return_value=(False, None)), \
patch("hermes_cli.config.load_config",
return_value={"auxiliary": {"openrouter_model": "some/vendor-model"}}), \
patch("agent.auxiliary_client.OpenAI") as mock_openai:
mock_client = MagicMock(name="openrouter_client")
mock_openai.return_value = mock_client
client, model = _try_openrouter()
assert client is mock_client
assert model == "some/vendor-model"
def test_explicit_caller_model_respects_free_only(self, monkeypatch):
"""Auxiliary.<task>.model (explicit) is also gated by free_only."""
monkeypatch.setenv("OPENROUTER_API_KEY", "or-key")
with patch("agent.auxiliary_client._select_pool_entry", return_value=(False, None)), \
patch("hermes_cli.config.load_config", return_value={"auxiliary": {"free_only": True}}), \
patch("agent.auxiliary_client.OpenAI") as mock_openai:
client, model = _try_openrouter(model="google/gemini-3.6-flash")
assert client is None
assert model is None
mock_openai.assert_not_called()
def test_paid_lane_warns_once(self, monkeypatch, caplog):
"""Engaging the default paid model logs a WARNING (once per model)."""
import logging
from agent.auxiliary_client import _paid_lane_warned
_paid_lane_warned.discard(_OPENROUTER_MODEL)
monkeypatch.setenv("OPENROUTER_API_KEY", "or-key")
with patch("agent.auxiliary_client._select_pool_entry", return_value=(False, None)), \
patch("hermes_cli.config.load_config", return_value={"auxiliary": {}}), \
patch("agent.auxiliary_client.OpenAI") as mock_openai:
mock_client = MagicMock(name="openrouter_client")
mock_openai.return_value = mock_client
with caplog.at_level(logging.WARNING, logger="agent.auxiliary_client"):
client, model = _try_openrouter()
assert client is mock_client
assert model == _OPENROUTER_MODEL
assert any("PAID lane engaged" in r.getMessage() for r in caplog.records)
# Second call logs nothing new.
with patch("agent.auxiliary_client._select_pool_entry", return_value=(False, None)), \
patch("hermes_cli.config.load_config", return_value={"auxiliary": {}}), \
patch("agent.auxiliary_client.OpenAI") as mock_openai:
caplog.clear()
with caplog.at_level(logging.WARNING, logger="agent.auxiliary_client"):
_try_openrouter()
assert not any("PAID lane engaged" in r.getMessage() for r in caplog.records)
_paid_lane_warned.discard(_OPENROUTER_MODEL)
def test_is_free_model(self):
from agent.auxiliary_client import _is_free_model
assert _is_free_model("nvidia/nemotron-3-ultra-550b-a55b:free")
assert not _is_free_model("google/gemini-3.6-flash")
assert not _is_free_model("")
assert not _is_free_model(None)
class TestGetTextAuxiliaryClient: class TestGetTextAuxiliaryClient:
"""Test the full resolution chain for get_text_auxiliary_client.""" """Test the full resolution chain for get_text_auxiliary_client."""