diff --git a/agent/model_metadata.py b/agent/model_metadata.py index 296fe0aedcab3..757a306cde78b 100644 --- a/agent/model_metadata.py +++ b/agent/model_metadata.py @@ -2429,21 +2429,27 @@ def get_model_context_length( if context_length is not None: return context_length if not _is_known_provider_base_url(base_url): - # 2b. Ollama native /api/show — any URL might be an Ollama server - # (local, cloud, or custom hosting). Non-Ollama servers return - # 404/405 quickly. Fall through on failure. - ctx = _query_ollama_api_show(model, base_url, api_key=api_key) - if ctx is not None: - if not _skip_persistent_context_cache(base_url, provider): - save_context_length(model, base_url, ctx) - return ctx - # 3. Try querying local server directly + # For local endpoints, run the probe that respects configured + # Modelfile context values first. _query_local_context_length + # prefers num_ctx from Modelfile, while _query_ollama_api_show + # returns the GGUF training max first which can be larger and + # would create a false-safe window for compression (#63122). + # Non-local endpoints preserve the existing GGUF-first behavior. if is_local_endpoint(base_url): local_ctx = _query_local_context_length(model, base_url, api_key=api_key) if local_ctx and local_ctx > 0: if not _skip_persistent_context_cache(base_url, provider): _maybe_cache_local_context_length(model, base_url, local_ctx) return local_ctx + # 2b. Ollama native /api/show — non-local endpoints preserve + # the existing generic /api/show GGUF-first behavior. + # Non-Ollama servers return 404/405 quickly. + ctx = _query_ollama_api_show(model, base_url, api_key=api_key) + if ctx is not None: + if not _skip_persistent_context_cache(base_url, provider): + save_context_length(model, base_url, ctx) + return ctx + # 3. Probe-down fallback after endpoint-specific detection failed logger.info( "Could not detect context length for model %r at %s — " "defaulting to %s tokens (probe-down). Set model.context_length " diff --git a/tests/agent/test_model_metadata.py b/tests/agent/test_model_metadata.py index 7a29178a40a06..a41c739127e9a 100644 --- a/tests/agent/test_model_metadata.py +++ b/tests/agent/test_model_metadata.py @@ -1183,6 +1183,73 @@ class TestGetModelContextLength: f"Expected {DEFAULT_FALLBACK_CONTEXT}, got {ctx3}" ) + # ── Local vs non-local Ollama context resolution (#63122) ────────── + + @patch("agent.model_metadata.get_cached_context_length", return_value=None) + @patch("agent.model_metadata.fetch_model_metadata", return_value={}) + @patch("agent.model_metadata._resolve_endpoint_context_length", return_value=None) + @patch("agent.model_metadata._query_ollama_api_show", return_value=131072) + @patch("agent.model_metadata._query_local_context_length", return_value=32768) + @patch("agent.model_metadata.is_local_endpoint", return_value=True) + @patch("agent.model_metadata.save_context_length") + @patch("agent.model_metadata._maybe_cache_local_context_length") + def test_local_ollama_prefers_num_ctx_over_gguf( + self, + mock_maybe_cache, mock_save, + mock_is_local, mock_local_ctx, + mock_ollama_show, mock_resolve_ep, + mock_fetch, mock_cache, + ): + """Local Ollama: _query_local_context_length (num_ctx-first) must + win over _query_ollama_api_show (GGUF-first). The configured + Modelfile num_ctx is the context value the local probe prefers; + the GGUF training max can be larger and would create a false-safe + window for compression (#63122).""" + result = get_model_context_length( + "my-model", + base_url="http://localhost:11434", + ) + assert result == 32768, ( + f"Expected configured Modelfile num_ctx (32768), got {result}. " + "Local Ollama must prefer num_ctx over GGUF training max." + ) + # The non-local-oriented probe must NOT fire when local probe succeeds + mock_ollama_show.assert_not_called() + # The local probe MUST be called exactly once + mock_local_ctx.assert_called_once() + + @patch("agent.model_metadata.get_cached_context_length", return_value=None) + @patch("agent.model_metadata.fetch_model_metadata", return_value={}) + @patch("agent.model_metadata._resolve_endpoint_context_length", return_value=None) + @patch("agent.model_metadata._query_ollama_api_show", return_value=131072) + @patch("agent.model_metadata._query_local_context_length", return_value=None) + @patch("agent.model_metadata.is_local_endpoint", return_value=False) + @patch("agent.model_metadata.save_context_length") + @patch("agent.model_metadata._maybe_cache_local_context_length") + def test_non_local_custom_ollama_preserves_gguf_first( + self, + mock_maybe_cache, mock_save, + mock_is_local, mock_local_ctx, + mock_ollama_show, mock_resolve_ep, + mock_fetch, mock_cache, + ): + """Non-local custom Ollama: GGUF-first ordering must be preserved. + A non-local endpoint should use _query_ollama_api_show (which + prefers model_info.context_length) — this preserves the existing + GGUF-first behavior for non-local Ollama endpoints.""" + result = get_model_context_length( + "my-model", + base_url="http://ollama.example.com:11434", + ) + assert result == 131072, ( + f"Expected GGUF training max (131072), got {result}. " + "Non-local Ollama must preserve GGUF-first ordering." + ) + # The local probe must NOT be called for non-local endpoints + mock_local_ctx.assert_not_called() + # The non-local probe MUST be called + mock_ollama_show.assert_called_once() + # ========================================================================= # Bedrock context resolution — must run BEFORE custom-endpoint probe