Compare commits

...

5 Commits

Author SHA1 Message Date
Darshil
fa5728817f style(docs): format ollama provider doc (#9870) 2026-02-05 16:26:52 -08:00
Darshil
6b042a2187 fix(ollama): wire streaming defaults, env auth, docs/tests (#9870) (thanks @rafelbev) 2026-02-05 16:25:26 -08:00
Raphael Borg Ellul Vincenti
5a60470331 chore: remove unused vi import from ollama test 2026-02-05 16:23:44 -08:00
Raphael Borg Ellul Vincenti
b60d9d9476 docs(ollama): use gpt-oss:20b as primary example
Updates documentation to use gpt-oss:20b as the primary example model
since it supports tool calling. The model examples now show:

- gpt-oss:20b as the primary recommended model (tool-capable)
- llama3.3 and qwen2.5-coder:32b as additional options

This provides users with a clear, working example that supports
OpenClaw's tool calling features.
2026-02-05 16:23:44 -08:00
Raphael Borg Ellul Vincenti
84cece580f fix(ollama): add streaming config and fix OLLAMA_API_KEY env var support
Adds configurable streaming parameter to model configuration and sets streaming
to false by default for Ollama models. This addresses the corrupted response
issue caused by upstream SDK bug badlogic/pi-mono#1205 where interleaved
content/reasoning deltas in streaming responses cause garbled output.

Changes:
- Add streaming param to AgentModelEntryConfig type
- Set streaming: false default for Ollama models
- Add OLLAMA_API_KEY to envMap (was missing, preventing env var auth)
- Document streaming configuration in Ollama provider docs
- Add tests for Ollama model configuration

Users can now configure streaming per-model and Ollama authentication
via OLLAMA_API_KEY environment variable works correctly.

Fixes #8839
Related: badlogic/pi-mono#1205
2026-02-05 16:23:44 -08:00
9 changed files with 216 additions and 8 deletions

View File

@@ -40,6 +40,7 @@ Docs: https://docs.openclaw.ai
- CLI: resolve bundled Chrome extension assets by walking up to the nearest assets directory; add resolver and clipboard tests. (#8914) Thanks @kelvinCB.
- Tests: stabilize Windows ACL coverage with deterministic os.userInfo mocking. (#9335) Thanks @M00N7682.
- Exec approvals: coerce bare string allowlist entries to objects to prevent allowlist corruption. (#9903, fixes #9790) Thanks @mcaxtr.
- Ollama: default embedded runs to non-streaming with `params.streaming` overrides, wire `OLLAMA_API_KEY` env auth mapping, and update docs/tests for streaming guidance. (#9870, fixes #8839) Thanks @rafelbev.
- Heartbeat: allow explicit accountId routing for multi-account channels. (#8702) Thanks @lsh411.
- TUI/Gateway: handle non-streaming finals, refresh history for non-local chat runs, and avoid event gap warnings for targeted tool streams. (#8432) Thanks @gumadeiras.
- Shell completion: auto-detect and migrate slow dynamic patterns to cached files for faster terminal startup; add completion health checks to doctor/update/onboard.

View File

@@ -17,6 +17,8 @@ Ollama is a local LLM runtime that makes it easy to run open-source models on yo
2. Pull a model:
```bash
ollama pull gpt-oss:20b
# or
ollama pull llama3.3
# or
ollama pull qwen2.5-coder:32b
@@ -40,7 +42,7 @@ openclaw config set models.providers.ollama.apiKey "ollama-local"
{
agents: {
defaults: {
model: { primary: "ollama/llama3.3" },
model: { primary: "ollama/gpt-oss:20b" },
},
},
}
@@ -105,8 +107,8 @@ Use explicit config when:
api: "openai-completions",
models: [
{
id: "llama3.3",
name: "Llama 3.3",
id: "gpt-oss:20b",
name: "GPT-OSS 20B",
reasoning: false,
input: ["text"],
cost: { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 },
@@ -148,8 +150,8 @@ Once configured, all your Ollama models are available:
agents: {
defaults: {
model: {
primary: "ollama/llama3.3",
fallbacks: ["ollama/qwen2.5-coder:32b"],
primary: "ollama/gpt-oss:20b",
fallbacks: ["ollama/llama3.3", "ollama/qwen2.5-coder:32b"],
},
},
},
@@ -170,6 +172,46 @@ ollama pull deepseek-r1:32b
Ollama is free and runs locally, so all model costs are set to $0.
### Streaming Configuration
Due to a [known upstream SDK issue](https://github.com/badlogic/pi-mono/issues/1205) with some Ollama streaming delta payloads, OpenClaw defaults Ollama runs to non-streaming mode. This avoids corrupted responses when tool-capable models emit interleaved content and reasoning chunks.
To opt in to streaming for a specific Ollama model, configure:
```json5
{
agents: {
defaults: {
models: {
"ollama/gpt-oss:20b": {
params: {
streaming: true,
},
},
},
},
},
}
```
You can also disable streaming for other providers via the same `params.streaming` key:
```json5
{
agents: {
defaults: {
models: {
"openai/gpt-4": {
params: {
streaming: false,
},
},
},
},
},
}
```
### Context windows
For auto-discovered models, OpenClaw uses the context window reported by Ollama when available, otherwise it defaults to `8192`. You can override `contextWindow` and `maxTokens` in explicit provider config.
@@ -201,7 +243,8 @@ To add models:
```bash
ollama list # See what's installed
ollama pull llama3.3 # Pull a model
ollama pull gpt-oss:20b # Pull a tool-capable model
ollama pull llama3.3 # Or another model
```
### Connection refused
@@ -216,6 +259,28 @@ ps aux | grep ollama
ollama serve
```
### Corrupted responses or tool names in output
If you see fragmented text or raw tool names in responses, check whether streaming was manually enabled for your Ollama model.
Set the model back to non-streaming:
```json5
{
agents: {
defaults: {
models: {
"ollama/gpt-oss:20b": {
params: {
streaming: false,
},
},
},
},
},
}
```
## See Also
- [Model Providers](/concepts/model-providers) - Overview of all providers

View File

@@ -301,6 +301,7 @@ export function resolveEnvApiKey(provider: string): EnvApiKeyResult | null {
venice: "VENICE_API_KEY",
mistral: "MISTRAL_API_KEY",
opencode: "OPENCODE_API_KEY",
ollama: "OLLAMA_API_KEY",
};
const envVar = envMap[normalized];
if (!envVar) {

View File

@@ -12,4 +12,19 @@ describe("Ollama provider", () => {
// Ollama requires explicit configuration via OLLAMA_API_KEY env var or profile
expect(providers?.ollama).toBeUndefined();
});
it("should include ollama when OLLAMA_API_KEY is configured", async () => {
const agentDir = mkdtempSync(join(tmpdir(), "openclaw-test-"));
process.env.OLLAMA_API_KEY = "ollama-local";
try {
const providers = await resolveImplicitProviders({ agentDir });
// resolveImplicitProviders stores the env var name for late runtime resolution.
expect(providers?.ollama).toBeDefined();
expect(providers?.ollama?.apiKey).toBe("OLLAMA_API_KEY");
} finally {
delete process.env.OLLAMA_API_KEY;
}
});
});

View File

@@ -125,6 +125,11 @@ async function discoverOllamaModels(): Promise<ModelDefinitionConfig[]> {
cost: OLLAMA_DEFAULT_COST,
contextWindow: OLLAMA_DEFAULT_CONTEXT_WINDOW,
maxTokens: OLLAMA_DEFAULT_MAX_TOKENS,
// Disable streaming by default for Ollama to avoid SDK issue #1205
// See: https://github.com/badlogic/pi-mono/issues/1205
params: {
streaming: false,
},
};
});
} catch (error) {

View File

@@ -91,4 +91,108 @@ describe("applyExtraParamsToAgent", () => {
"X-Custom": "1",
});
});
it("defaults Ollama models to non-streaming", () => {
const calls: Array<Record<string, unknown> | undefined> = [];
const baseStreamFn: StreamFn = (_model, _context, options) => {
calls.push(options as Record<string, unknown> | undefined);
return new AssistantMessageEventStream();
};
const agent = { streamFn: baseStreamFn };
applyExtraParamsToAgent(agent, undefined, "ollama", "gpt-oss:20b");
const model = {
api: "openai-completions",
provider: "ollama",
id: "gpt-oss:20b",
} as Model<"openai-completions">;
const context: Context = { messages: [] };
void agent.streamFn?.(model, context, {});
expect(calls).toHaveLength(1);
expect(calls[0]?.streaming).toBe(false);
});
it("allows per-model Ollama streaming override from config", () => {
const calls: Array<Record<string, unknown> | undefined> = [];
const baseStreamFn: StreamFn = (_model, _context, options) => {
calls.push(options as Record<string, unknown> | undefined);
return new AssistantMessageEventStream();
};
const agent = { streamFn: baseStreamFn };
applyExtraParamsToAgent(
agent,
{
agents: {
defaults: {
models: {
"ollama/gpt-oss:20b": {
params: {
streaming: true,
},
},
},
},
},
},
"ollama",
"gpt-oss:20b",
);
const model = {
api: "openai-completions",
provider: "ollama",
id: "gpt-oss:20b",
} as Model<"openai-completions">;
const context: Context = { messages: [] };
void agent.streamFn?.(model, context, {});
expect(calls).toHaveLength(1);
expect(calls[0]?.streaming).toBe(true);
});
it("applies per-run overrides after model defaults", () => {
const calls: Array<Record<string, unknown> | undefined> = [];
const baseStreamFn: StreamFn = (_model, _context, options) => {
calls.push(options as Record<string, unknown> | undefined);
return new AssistantMessageEventStream();
};
const agent = { streamFn: baseStreamFn };
applyExtraParamsToAgent(
agent,
{
agents: {
defaults: {
models: {
"ollama/gpt-oss:20b": {
params: {
streaming: false,
},
},
},
},
},
},
"ollama",
"gpt-oss:20b",
{ streaming: true },
);
const model = {
api: "openai-completions",
provider: "ollama",
id: "gpt-oss:20b",
} as Model<"openai-completions">;
const context: Context = { messages: [] };
void agent.streamFn?.(model, context, {});
expect(calls).toHaveLength(1);
expect(calls[0]?.streaming).toBe(true);
});
});

View File

@@ -11,7 +11,7 @@ const OPENROUTER_APP_HEADERS: Record<string, string> = {
/**
* Resolve provider-specific extra params from model config.
* Used to pass through stream params like temperature/maxTokens.
* Used to pass through stream params like temperature/maxTokens/streaming.
*
* @internal Exported for testing only
*/
@@ -28,8 +28,17 @@ export function resolveExtraParams(params: {
type CacheRetention = "none" | "short" | "long";
type CacheRetentionStreamOptions = Partial<SimpleStreamOptions> & {
cacheRetention?: CacheRetention;
streaming?: boolean;
};
function resolveProviderDefaultExtraParams(provider: string): Record<string, unknown> | undefined {
// Ollama streaming is disabled by default due to upstream SDK stream-delta interleaving issues.
if (provider.trim().toLowerCase() === "ollama") {
return { streaming: false };
}
return undefined;
}
/**
* Resolve cacheRetention from extraParams, supporting both new `cacheRetention`
* and legacy `cacheControlTtl` values for backwards compatibility.
@@ -80,6 +89,9 @@ function createStreamFnWithExtraParams(
if (typeof extraParams.maxTokens === "number") {
streamParams.maxTokens = extraParams.maxTokens;
}
if (typeof extraParams.streaming === "boolean") {
streamParams.streaming = extraParams.streaming;
}
const cacheRetention = resolveCacheRetention(extraParams, provider);
if (cacheRetention) {
streamParams.cacheRetention = cacheRetention;
@@ -141,7 +153,8 @@ export function applyExtraParamsToAgent(
Object.entries(extraParamsOverride).filter(([, value]) => value !== undefined),
)
: undefined;
const merged = Object.assign({}, extraParams, override);
const providerDefaults = resolveProviderDefaultExtraParams(provider);
const merged = Object.assign({}, providerDefaults, extraParams, override);
const wrappedStreamFn = createStreamFnWithExtraParams(agent.streamFn, merged, provider);
if (wrappedStreamFn) {

View File

@@ -16,6 +16,8 @@ export type AgentModelEntryConfig = {
alias?: string;
/** Provider-specific API parameters (e.g., GLM-4.7 thinking mode). */
params?: Record<string, unknown>;
/** Enable streaming for this model (default: true, false for Ollama to avoid SDK issue #1205). */
streaming?: boolean;
};
export type AgentModelListConfig = {

View File

@@ -37,6 +37,8 @@ export const AgentDefaultsSchema = z
alias: z.string().optional(),
/** Provider-specific API parameters (e.g., GLM-4.7 thinking mode). */
params: z.record(z.string(), z.unknown()).optional(),
/** Enable streaming for this model (default: true, false for Ollama to avoid SDK issue #1205). */
streaming: z.boolean().optional(),
})
.strict(),
)