From 0bfb69e1a1357af42e1c6ad0b54ffda4e2643fe4 Mon Sep 17 00:00:00 2001 From: bitcoinaudio Date: Mon, 27 Jul 2026 16:53:04 +0000 Subject: [PATCH] feat(buzz-agent): first-class ollama provider alias + local-models docs Local models already worked through the generic OpenAI-compatible provider, but the path was undocumented and required a dummy API key plus a magic base URL. The word 'ollama' appeared nowhere in the repo, leading operators to set model env vars on the relay (which never calls an LLM) and conclude local models were unsupported. - BUZZ_AGENT_PROVIDER=ollama maps onto the OpenAI-compatible path with local-friendly defaults: no API key required (placeholder sent), OPENAI_COMPAT_BASE_URL defaults to http://localhost:11434/v1 - docs/local-models.md: where LLM config actually lives (agent process, not the relay), Ollama quickstart, generic openai-compat recipe, endpoint verification, context-window tuning, desktop steps - README: ollama run example + env table updates Closes #3145 Signed-off-by: bitcoinaudio --- crates/buzz-agent/README.md | 13 ++-- crates/buzz-agent/src/config.rs | 115 +++++++++++++++++++++++++++++++- docs/local-models.md | 80 ++++++++++++++++++++++ 3 files changed, 201 insertions(+), 7 deletions(-) create mode 100644 docs/local-models.md diff --git a/crates/buzz-agent/README.md b/crates/buzz-agent/README.md index a2504db451..892f2c1a8a 100644 --- a/crates/buzz-agent/README.md +++ b/crates/buzz-agent/README.md @@ -50,6 +50,11 @@ OPENAI_COMPAT_MODEL=gpt-5 \ OPENAI_COMPAT_BASE_URL=https://api.openai.com/v1 \ ./target/release/buzz-agent +# Or a local Ollama server (no API key needed) +BUZZ_AGENT_PROVIDER=ollama \ +OPENAI_COMPAT_MODEL=qwen2.5:7b-instruct \ + ./target/release/buzz-agent + # Or Databricks model serving via OAuth 2.0 PKCE BUZZ_AGENT_PROVIDER=databricks \ DATABRICKS_HOST=https://dbc-...cloud.databricks.com \ @@ -129,14 +134,14 @@ Everything is environment variables. No flags, no config files. (We are a subpro | Variable | Default | Notes | |---|---|---| -| `BUZZ_AGENT_PROVIDER` | — | Required. `anthropic`, `openai`, `databricks`, or `databricks_v2`. No implicit fallback — the agent errors at startup when this is unset. | +| `BUZZ_AGENT_PROVIDER` | — | Required. `anthropic`, `openai`, `openai-compat`, `ollama`, `databricks`, or `databricks_v2`. No implicit fallback — the agent errors at startup when this is unset. | | `ANTHROPIC_API_KEY` | — | Required when provider=anthropic. | | `ANTHROPIC_MODEL` | — | Required when provider=anthropic. | | `ANTHROPIC_BASE_URL` | `https://api.anthropic.com` | | | `ANTHROPIC_API_VERSION` | `2023-06-01` | | -| `OPENAI_COMPAT_API_KEY` | — | Required when provider=openai. | -| `OPENAI_COMPAT_MODEL` | — | Required when provider=openai. | -| `OPENAI_COMPAT_BASE_URL` | `https://api.openai.com/v1` | Point at vLLM, llama.cpp, OpenRouter, Ollama, etc. | +| `OPENAI_COMPAT_API_KEY` | — | Required when provider=openai/openai-compat. Optional when provider=ollama (Ollama ignores it; a placeholder is sent when unset). | +| `OPENAI_COMPAT_MODEL` | — | Required when provider=openai/openai-compat/ollama. For Ollama, use a pulled model name from `ollama list` (e.g. `qwen2.5:7b-instruct`). | +| `OPENAI_COMPAT_BASE_URL` | `https://api.openai.com/v1` (`http://localhost:11434/v1` for provider=ollama) | Point at vLLM, llama.cpp, OpenRouter, Ollama, etc. Set this for a remote Ollama host. | | `OPENAI_COMPAT_API` | `auto` | `auto` \| `chat` \| `responses`. `auto` picks Responses for `*.openai.com`, Chat Completions everywhere else. | | `DATABRICKS_HOST` | — | Required when provider=databricks or provider=databricks_v2. | | `DATABRICKS_MODEL` | — | Required when provider=databricks or provider=databricks_v2. | diff --git a/crates/buzz-agent/src/config.rs b/crates/buzz-agent/src/config.rs index f3464fb903..260368719c 100644 --- a/crates/buzz-agent/src/config.rs +++ b/crates/buzz-agent/src/config.rs @@ -658,9 +658,15 @@ pub const HANDOFF_MAX_TOOL_NAMES: usize = 20; const DEFAULT_SYSTEM_PROMPT: &str = "You are buzz-agent. Use the provided tools to act. Tool calls are your only output."; +/// Default OpenAI-compatible endpoint for a local Ollama server, used when +/// `BUZZ_AGENT_PROVIDER=ollama` and `OPENAI_COMPAT_BASE_URL` is unset. +pub const OLLAMA_DEFAULT_BASE_URL: &str = "http://localhost:11434/v1"; + #[derive(Debug, Clone, Copy, PartialEq)] pub enum Provider { Anthropic, + /// Also selected by the `ollama` alias: Ollama serves an OpenAI-compatible + /// API, so it rides the same dispatch with local-friendly defaults. OpenAi, /// Databricks model serving. Routes to `{base_url}/serving-endpoints/{model}/invocations` /// with a dynamically-acquired bearer (OAuth 2.0 PKCE, or static `DATABRICKS_TOKEN`). @@ -772,13 +778,19 @@ impl Config { OpenAiApi::Auto, // unused for Anthropic ), Provider::OpenAi => ( - req("OPENAI_COMPAT_API_KEY")?, + resolve_openai_api_key( + env("BUZZ_AGENT_PROVIDER").as_deref(), + env("OPENAI_COMPAT_API_KEY"), + )?, resolve_model( buzz_agent_model.as_deref(), env("OPENAI_COMPAT_MODEL").as_deref(), ) .ok_or_else(|| "config: OPENAI_COMPAT_MODEL required".to_string())?, - env_or("OPENAI_COMPAT_BASE_URL", "https://api.openai.com/v1"), + resolve_openai_base_url( + env("BUZZ_AGENT_PROVIDER").as_deref(), + env("OPENAI_COMPAT_BASE_URL"), + ), parse_openai_api(env("OPENAI_COMPAT_API").as_deref())?, ), Provider::Databricks | Provider::DatabricksV2 => ( @@ -987,6 +999,39 @@ fn present_nonempty(v: Option<&str>) -> bool { v.map(str::trim).is_some_and(|s| !s.is_empty()) } +/// True when `BUZZ_AGENT_PROVIDER` selects the `ollama` alias. +fn is_ollama_provider(requested: Option<&str>) -> bool { + requested + .map(str::trim) + .is_some_and(|s| s.eq_ignore_ascii_case("ollama")) +} + +/// Resolve the OpenAI-compatible API key. Ollama ignores the key but the +/// header must exist, so the alias supplies a placeholder when unset; every +/// other OpenAI-compatible provider still requires a real key. +fn resolve_openai_api_key( + requested_provider: Option<&str>, + key: Option, +) -> Result { + match key { + Some(k) if present_nonempty(Some(&k)) => Ok(k), + _ if is_ollama_provider(requested_provider) => Ok("ollama".to_owned()), + _ => Err("config: OPENAI_COMPAT_API_KEY required".into()), + } +} + +/// Resolve the OpenAI-compatible base URL, defaulting to the well-known local +/// Ollama endpoint for the `ollama` alias and api.openai.com otherwise. +fn resolve_openai_base_url(requested_provider: Option<&str>, base_url: Option) -> String { + base_url.unwrap_or_else(|| { + if is_ollama_provider(requested_provider) { + OLLAMA_DEFAULT_BASE_URL.to_owned() + } else { + "https://api.openai.com/v1".to_owned() + } + }) +} + fn resolve_provider( requested: Option<&str>, anthropic_key: Option<&str>, @@ -1004,6 +1049,8 @@ fn resolve_provider( "openai" | "openai-compat" => Err( "config: OPENAI_COMPAT_API_KEY required".into(), ), + // Ollama speaks the OpenAI-compatible API and needs no key. + "ollama" => Ok(Provider::OpenAi), "databricks" => Ok(Provider::Databricks), "databricks_v2" | "databricks-v2" => Ok(Provider::DatabricksV2), _ => Err(format!( @@ -1012,7 +1059,7 @@ fn resolve_provider( } } None => Err( - "config: BUZZ_AGENT_PROVIDER is required — set it to your provider (e.g. anthropic, openai, databricks)".into(), + "config: BUZZ_AGENT_PROVIDER is required — set it to your provider (e.g. anthropic, openai, ollama, databricks)".into(), ), } } @@ -1271,6 +1318,68 @@ mod tests { assert!(err.contains("BUZZ_AGENT_PROVIDER=OpenAIish")); } + #[test] + fn resolve_provider_ollama_needs_no_api_key() { + // The ollama alias maps onto the OpenAI-compatible path and never + // requires a key, present or absent. + assert_eq!( + resolve_provider(Some("ollama"), None, None).unwrap(), + Provider::OpenAi + ); + assert_eq!( + resolve_provider(Some("Ollama"), None, Some("anything")).unwrap(), + Provider::OpenAi + ); + } + + #[test] + fn resolve_openai_api_key_ollama_defaults_to_placeholder() { + assert_eq!( + resolve_openai_api_key(Some("ollama"), None).unwrap(), + "ollama" + ); + // Whitespace-only keys are treated as unset. + assert_eq!( + resolve_openai_api_key(Some("ollama"), Some(" ".to_owned())).unwrap(), + "ollama" + ); + // An explicit key always wins. + assert_eq!( + resolve_openai_api_key(Some("ollama"), Some("sk-real".to_owned())).unwrap(), + "sk-real" + ); + } + + #[test] + fn resolve_openai_api_key_non_ollama_still_required() { + let err = resolve_openai_api_key(Some("openai"), None).unwrap_err(); + assert!(err.contains("OPENAI_COMPAT_API_KEY required"), "{err}"); + let err = resolve_openai_api_key(Some("openai-compat"), Some(String::new())).unwrap_err(); + assert!(err.contains("OPENAI_COMPAT_API_KEY required"), "{err}"); + assert_eq!( + resolve_openai_api_key(Some("openai"), Some("sk-openai".to_owned())).unwrap(), + "sk-openai" + ); + } + + #[test] + fn resolve_openai_base_url_ollama_defaults_to_local_endpoint() { + assert_eq!( + resolve_openai_base_url(Some("ollama"), None), + OLLAMA_DEFAULT_BASE_URL + ); + // Explicit override wins for remote Ollama servers. + assert_eq!( + resolve_openai_base_url(Some("ollama"), Some("http://gpu-box:11434/v1".to_owned())), + "http://gpu-box:11434/v1" + ); + // Non-ollama default unchanged. + assert_eq!( + resolve_openai_base_url(Some("openai"), None), + "https://api.openai.com/v1" + ); + } + #[test] fn is_openai_host_matrix() { // Lookalike-safe: `api.openai.com.evil.example` and malformed URLs diff --git a/docs/local-models.md b/docs/local-models.md new file mode 100644 index 0000000000..fa64944746 --- /dev/null +++ b/docs/local-models.md @@ -0,0 +1,80 @@ +# Local Models (Ollama and other OpenAI-compatible servers) + +Buzz agents can run entirely against local models. `buzz-agent` speaks to any +OpenAI-compatible endpoint, and Ollama is a first-class provider alias. + +## Where the LLM config lives + +Model selection is a property of the **agent process**, not the relay. +`buzz-relay` never calls an LLM — setting `OPENAI_COMPAT_*` / `OLLAMA_*` env +vars on the relay container or in the relay's `.env` has no effect. + +Agents are spawned by an ACP harness: + +- the desktop app (managed agents, configured in Settings → Agents), +- `sprig` / `buzz-acp` on a server, or +- `buzz-agent` run standalone for testing. + +Whichever process spawns the agent is where these env vars belong. + +## Ollama (quickest path) + +```bash +ollama pull qwen2.5:7b-instruct # any model; tool-calling models work best + +BUZZ_AGENT_PROVIDER=ollama \ +OPENAI_COMPAT_MODEL=qwen2.5:7b-instruct \ + ./target/release/buzz-agent +``` + +With `BUZZ_AGENT_PROVIDER=ollama`: + +- `OPENAI_COMPAT_BASE_URL` defaults to `http://localhost:11434/v1` — set it + only for a remote Ollama host (e.g. `http://gpu-box:11434/v1`). +- `OPENAI_COMPAT_API_KEY` is optional — Ollama ignores it, so the agent sends + a placeholder when unset. +- `OPENAI_COMPAT_API` stays at its `auto` default, which selects Chat + Completions for any non-`*.openai.com` host. + +## Any other OpenAI-compatible server + +vLLM, llama.cpp (`llama-server`), OpenRouter, LM Studio, etc. use the generic +provider: + +```bash +BUZZ_AGENT_PROVIDER=openai \ +OPENAI_COMPAT_BASE_URL=http://localhost:8080/v1 \ +OPENAI_COMPAT_MODEL= \ +OPENAI_COMPAT_API_KEY= \ + ./target/release/buzz-agent +``` + +## Verifying the endpoint before launching + +```bash +# Model catalog (Ollama) +curl http://localhost:11434/api/tags + +# OpenAI-compatible chat round-trip +curl http://localhost:11434/v1/chat/completions \ + -H 'Content-Type: application/json' \ + -d '{"model":"qwen2.5:7b-instruct","messages":[{"role":"user","content":"hi"}]}' +``` + +## Tuning for local models + +Local models usually have smaller context windows than the 200k default. +Lower the handoff threshold so the agent compacts history before overflowing +the window: + +```bash +BUZZ_AGENT_MAX_CONTEXT_TOKENS=32768 # match the model's real window +BUZZ_AGENT_MAX_OUTPUT_TOKENS=4096 +``` + +## Desktop app + +In Settings → Agents, pick the **Buzz Agent** harness with provider +**OpenAI-compatible**, set the base URL to `http://:11434/v1` +and the model to a pulled Ollama model. The API key field requires a value — +any non-empty string works for Ollama.