From 86f55b3a4fc6dd78fda937d4a39664798be4e001 Mon Sep 17 00:00:00 2001 From: Paul Zabalegui Date: Wed, 10 Dec 2025 14:32:01 +0100 Subject: [PATCH] feat: Add Ollama Cloud integration with ollama_cloud/ prefix - Added support for Ollama Cloud models via AsyncOpenAI - Models use ollama_cloud/ prefix (e.g., ollama_cloud/gpt-oss:120b) - Reads OLLAMA_API_KEY and OLLAMA_API_BASE for cloud authentication - Added predefined Ollama Cloud models to /model-show - Filtered obsolete ollama/*-cloud models from LiteLLM database - Updated authentication headers in completer, banner, and toolbar - Added concise English documentation in docs/providers/ - Adapted to new global model cache architecture from #371 Modified files: - src/cai/sdk/agents/models/openai_chatcompletions.py - src/cai/repl/commands/model.py (adapted to global cache) - src/cai/util.py - src/cai/repl/commands/completer.py - src/cai/repl/ui/banner.py - src/cai/repl/ui/toolbar.py - docs/providers/ollama.md - docs/providers/ollama_cloud.md All existing functionality preserved (backward compatible). --- docs/providers/ollama.md | 27 +++++++ docs/providers/ollama_cloud.md | 79 +++++++++++++++++++ src/cai/repl/commands/completer.py | 12 ++- src/cai/repl/commands/model.py | 18 ++++- src/cai/repl/ui/banner.py | 15 +++- src/cai/repl/ui/toolbar.py | 17 +++- .../agents/models/openai_chatcompletions.py | 63 ++++++++++++++- src/cai/util.py | 20 ++++- 8 files changed, 237 insertions(+), 14 deletions(-) create mode 100644 docs/providers/ollama_cloud.md diff --git a/docs/providers/ollama.md b/docs/providers/ollama.md index 1ff39758..34f32aec 100644 --- a/docs/providers/ollama.md +++ b/docs/providers/ollama.md @@ -1,5 +1,7 @@ # Ollama Configuration +## Ollama Local (Self-hosted) + #### [Ollama Integration](https://ollama.com/) For local models using Ollama, add the following to your .env: @@ -9,3 +11,28 @@ OLLAMA_API_BASE=http://localhost:8000/v1 # note, maybe you have a different endp ``` Make sure that the Ollama server is running and accessible at the specified base URL. You can swap the model with any other supported by your local Ollama instance. + +## Ollama Cloud + +For cloud models using Ollama Cloud (no GPU required), add the following to your .env: + +```bash +# API Key from ollama.com +OLLAMA_API_KEY=your_api_key_here +OLLAMA_API_BASE=https://ollama.com + +# Cloud model (note the ollama_cloud/ prefix) +CAI_MODEL=ollama_cloud/gpt-oss:120b +``` + +**Requirements:** +1. Create an account at [ollama.com](https://ollama.com) +2. Generate an API key from your profile +3. Use models with `ollama_cloud/` prefix (e.g., `ollama_cloud/gpt-oss:120b`) + +**Key differences:** +- Prefix: `ollama_cloud/` (cloud) vs `ollama/` (local) +- API Key: Required for cloud, not needed for local +- Endpoint: `https://ollama.com/v1` (cloud) vs `http://localhost:8000/v1` (local) + +See [Ollama Cloud documentation](ollama_cloud.md) for detailed setup instructions. diff --git a/docs/providers/ollama_cloud.md b/docs/providers/ollama_cloud.md new file mode 100644 index 00000000..6498731e --- /dev/null +++ b/docs/providers/ollama_cloud.md @@ -0,0 +1,79 @@ +# Ollama Cloud + +Run large language models without local GPU using Ollama's cloud service. + +## Quick Start + +### 1. Get API Key + +- Create account at [ollama.com](https://ollama.com) +- Generate API key from your profile + +### 2. Configure `.env` + +```bash +OLLAMA_API_KEY=your_api_key_here +OLLAMA_API_BASE=https://ollama.com +CAI_MODEL=ollama_cloud/gpt-oss:120b +``` + +### 3. Run + +```bash +cai +``` + +## Available Models + +View in CAI with `/model-show` under "Ollama Cloud" category: + +- `ollama_cloud/gpt-oss:120b` - General purpose 120B model +- `ollama_cloud/llama3.3:70b` - Llama 3.3 70B +- `ollama_cloud/qwen2.5:72b` - Qwen 2.5 72B +- `ollama_cloud/deepseek-v3:671b` - DeepSeek V3 671B + +More models at [ollama.com/library](https://ollama.com/library). + +## Model Selection + +```bash +# By name +CAI> /model ollama_cloud/gpt-oss:120b + +# By number (after /model-show) +CAI> /model 3 +``` + +## Local vs Cloud + +| Feature | Local | Cloud | +|---------|-------|-------| +| Prefix | `ollama/` | `ollama_cloud/` | +| API Key | Not required | Required | +| Endpoint | `http://localhost:8000/v1` | `https://ollama.com/v1` | +| GPU | Required | Not required | + +## Troubleshooting + +**Unauthorized error**: Verify `OLLAMA_API_KEY` is set correctly + +**Path not found**: Ensure `OLLAMA_API_BASE=https://ollama.com` (without `/v1`) + +**Model not listed**: Check model prefix is `ollama_cloud/`, not `ollama/` + +## Validation + +Test connection with curl: + +```bash +curl https://ollama.com/v1/chat/completions \ + -H "Authorization: Bearer $OLLAMA_API_KEY" \ + -H "Content-Type: application/json" \ + -d '{"model": "gpt-oss:120b", "messages": [{"role": "user", "content": "test"}]}' +``` + +## References + +- [Ollama Cloud Docs](https://ollama.com/docs/cloud) +- [Model Library](https://ollama.com/library) +- [Get API Key](https://ollama.com/settings/keys) diff --git a/src/cai/repl/commands/completer.py b/src/cai/repl/commands/completer.py index f655d9c7..0c313335 100644 --- a/src/cai/repl/commands/completer.py +++ b/src/cai/repl/commands/completer.py @@ -184,8 +184,18 @@ class FuzzyCommandCompleter(Completer): try: # Get Ollama models with a short timeout to prevent hanging api_base = get_ollama_api_base() + + # Add authentication headers for Ollama Cloud if using OPENAI_BASE_URL + headers = {} + if "ollama.com" in api_base: + api_key = os.getenv("OPENAI_API_KEY") + if api_key: + headers["Authorization"] = f"Bearer {api_key}" + response = requests.get( - f"{api_base.replace('/v1', '')}/api/tags", timeout=0.5) + f"{api_base.replace('/v1', '')}/api/tags", + headers=headers, + timeout=0.5) if response.status_code == 200: data = response.json() diff --git a/src/cai/repl/commands/model.py b/src/cai/repl/commands/model.py index a16bc5d4..5981eee9 100644 --- a/src/cai/repl/commands/model.py +++ b/src/cai/repl/commands/model.py @@ -175,7 +175,11 @@ def load_all_available_models() -> tuple[List[str], List[Dict[str, Any]]]: try: response = requests.get(LITELLM_URL, timeout=5) if response.status_code == 200: - litellm_names = sorted(response.json().keys()) + # Filter out obsolete Ollama Cloud models (replaced by ollama_cloud/ prefix) + litellm_names = [ + model_name for model_name in sorted(response.json().keys()) + if not (model_name.startswith("ollama/") and "-cloud" in model_name) + ] except Exception: # pylint: disable=broad-except pass @@ -184,7 +188,17 @@ def load_all_available_models() -> tuple[List[str], List[Dict[str, Any]]]: ollama_names = [] try: api_base = get_ollama_api_base() - response = requests.get(f"{api_base.replace('/v1', '')}/api/tags", timeout=1) + ollama_base = api_base.replace('/v1', '') + + # Add authentication headers for Ollama Cloud if needed + headers = {} + is_cloud = "ollama.com" in api_base + timeout = 5 if is_cloud else 1 # Cloud needs more time + + if is_cloud: + headers = get_ollama_auth_headers() + + response = requests.get(f"{ollama_base}/api/tags", headers=headers, timeout=timeout) if response.status_code == 200: data = response.json() ollama_data = data.get('models', data.get('items', [])) diff --git a/src/cai/repl/ui/banner.py b/src/cai/repl/ui/banner.py index 0050b992..a5e40140 100644 --- a/src/cai/repl/ui/banner.py +++ b/src/cai/repl/ui/banner.py @@ -76,12 +76,19 @@ def get_supported_models_count(): # Try to get Ollama models count try: - ollama_api_base = os.getenv( - "OLLAMA_API_BASE", - "http://host.docker.internal:8000/v1" - ) + from cai.util import get_ollama_api_base + ollama_api_base = get_ollama_api_base() + + # Add authentication headers for Ollama Cloud if using OPENAI_BASE_URL + headers = {} + if "ollama.com" in ollama_api_base: + api_key = os.getenv("OPENAI_API_KEY") + if api_key: + headers["Authorization"] = f"Bearer {api_key}" + ollama_response = requests.get( f"{ollama_api_base.replace('/v1', '')}/api/tags", + headers=headers, timeout=1 ) diff --git a/src/cai/repl/ui/toolbar.py b/src/cai/repl/ui/toolbar.py index 06b1e70f..fd6fd282 100644 --- a/src/cai/repl/ui/toolbar.py +++ b/src/cai/repl/ui/toolbar.py @@ -96,11 +96,20 @@ def update_toolbar_in_background(): ollama_status = "unavailable" try: # Get Ollama models with a short timeout to prevent hanging - api_base = os.getenv( - "OLLAMA_API_BASE", - "http://host.docker.internal:8000/v1") + from cai.util import get_ollama_api_base + api_base = get_ollama_api_base() + + # Add authentication headers for Ollama Cloud if using OPENAI_BASE_URL + headers = {} + if "ollama.com" in api_base: + api_key = os.getenv("OPENAI_API_KEY") + if api_key: + headers["Authorization"] = f"Bearer {api_key}" + response = requests.get( - f"{api_base.replace('/v1', '')}/api/tags", timeout=0.5) + f"{api_base.replace('/v1', '')}/api/tags", + headers=headers, + timeout=0.5) if response.status_code == 200: data = response.json() diff --git a/src/cai/sdk/agents/models/openai_chatcompletions.py b/src/cai/sdk/agents/models/openai_chatcompletions.py index 7e88f438..8931edd6 100644 --- a/src/cai/sdk/agents/models/openai_chatcompletions.py +++ b/src/cai/sdk/agents/models/openai_chatcompletions.py @@ -2708,7 +2708,23 @@ class OpenAIChatCompletionsModel(Model): provider = model_str.split("/")[0] # Apply provider-specific configurations - if provider == "deepseek": + if provider == "ollama_cloud": + # Ollama Cloud configuration + ollama_api_key = os.getenv("OLLAMA_API_KEY") + ollama_api_base = os.getenv("OLLAMA_API_BASE", "https://ollama.com") + + if ollama_api_key: + kwargs["api_key"] = ollama_api_key + if ollama_api_base: + kwargs["api_base"] = ollama_api_base + + # Drop params not supported by Ollama + litellm.drop_params = True + kwargs.pop("parallel_tool_calls", None) + kwargs.pop("store", None) + if not converted_tools: + kwargs.pop("tool_choice", None) + elif provider == "deepseek": litellm.drop_params = True kwargs.pop("parallel_tool_calls", None) kwargs.pop("store", None) # DeepSeek doesn't support store parameter @@ -2846,6 +2862,51 @@ class OpenAIChatCompletionsModel(Model): max_retries = 3 retry_count = 0 + # Check if this is Ollama Cloud (ollama_cloud/ prefix) + # Ollama Cloud is OpenAI-compatible, so we bypass LiteLLM to avoid parsing issues + is_ollama_cloud = "ollama_cloud/" in model_str + + if is_ollama_cloud: + # Use AsyncOpenAI client directly for Ollama Cloud + # Ollama Cloud is fully OpenAI-compatible at /v1/chat/completions + try: + # Configure the client with Ollama Cloud settings + ollama_api_key = os.getenv("OLLAMA_API_KEY") or os.getenv("OPENAI_API_KEY") + ollama_base_url = os.getenv("OLLAMA_API_BASE", "https://ollama.com") + + # Ensure the URL has /v1 for OpenAI compatibility + if not ollama_base_url.endswith("/v1"): + ollama_base_url = f"{ollama_base_url}/v1" + + # Create a temporary client configured for Ollama Cloud + ollama_client = AsyncOpenAI( + api_key=ollama_api_key, + base_url=ollama_base_url + ) + + # Remove the ollama_cloud/ prefix from the model name + clean_model = kwargs["model"].replace("ollama_cloud/", "") + kwargs["model"] = clean_model + + # Remove LiteLLM-specific parameters + kwargs.pop("extra_headers", None) + kwargs.pop("api_key", None) + kwargs.pop("api_base", None) + kwargs.pop("custom_llm_provider", None) + + # Call Ollama Cloud using OpenAI-compatible API + if stream: + return await ollama_client.chat.completions.create(**kwargs) + else: + return await ollama_client.chat.completions.create(**kwargs) + + except Exception as e: + # If Ollama Cloud fails, raise with helpful message + raise Exception( + f"Error connecting to Ollama Cloud: {str(e)}\n" + f"Verify OLLAMA_API_KEY and OLLAMA_API_BASE are configured correctly." + ) from e + while retry_count < max_retries: try: if self.is_ollama: diff --git a/src/cai/util.py b/src/cai/util.py index a935e944..c587ffd7 100644 --- a/src/cai/util.py +++ b/src/cai/util.py @@ -744,8 +744,24 @@ install_pretty() def get_ollama_api_base(): - """Get the Ollama API base URL from environment variable or default to localhost:8000.""" - return os.environ.get("OLLAMA_API_BASE", "http://localhost:8000/v1") + """Get the Ollama API base URL from environment variable or default to localhost:8000. + + Supports both: + - OLLAMA_API_BASE: For local Ollama instances (e.g., http://localhost:8000/v1) + - OPENAI_BASE_URL: For Ollama Cloud or other OpenAI-compatible services (e.g., https://ollama.com/api/v1) + """ + # First check OLLAMA_API_BASE for local Ollama + ollama_base = os.environ.get("OLLAMA_API_BASE") + if ollama_base: + return ollama_base + + # Then check OPENAI_BASE_URL for Ollama Cloud or other services + openai_base = os.environ.get("OPENAI_BASE_URL") + if openai_base and "ollama.com" in openai_base: + return openai_base + + # Default to local Ollama + return "http://localhost:8000/v1" def load_prompt_template(template_path):