fix(ollama): resolve context window per model via POST /api/show (#1437)

The Ollama integration never applied a correct context window, so agents
with real prompts (20k-100k tokens) were rejected with HTTP 400
exceed_context_size_error against a 4096-token default.

Three root causes fixed:

1. FetchOllamaModelContext issued a GET to /api/show, which Ollama answers
   with 405 (the endpoint is POST-only). Now POSTs {"model": "<name>"}.

2. The response parser expected a flat model_info.context_length, but a real
   Ollama server namespaces the key by architecture (gemma4.context_length,
   qwen3.context_length, ...). extractContextLength now matches "context_length"
   or any "*.context_length" key.

3. num_ctx was resolved once at startup for a hardcoded "llama3.3" model and
   never for the model an agent actually uses. Resolution now happens per
   request for the real model inside OllamaProvider.resolveNumCtx, cached under
   an RWMutex, with an explicit settings override winning and the fetched value
   bounded by OllamaDefaultNumCtx so an enormous advertised window (Qwen3.5
   reports 262144) cannot balloon the KV cache beyond VRAM.

Also classify Ollama's "exceed_context_size" 400 as a context-overflow error so
the pipeline's emergency-compaction+retry path (Issue 958) engages gracefully
instead of surfacing a raw error.

Co-authored-by: Bruno Clermont <bruno.clermont@gmail.com>
Co-authored-by: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
This commit is contained in:
authored and GitHub committed 2026-07-15 08:24:31 +07:00
1 parent 6f91b937c1
commit 4f558e2c3b
7 files changed
+177 -40

No files matched your search

+11 -17
View File
@@ -316,7 +316,7 @@ func registerProvidersFromDB(registry *providers.Registry, provStore store.Provi
if host == "" {
host = "http://localhost:11434"
}
numCtx := resolveOllamaNumCtx(&p, config.DockerLocalhost(host), "")
numCtx := resolveOllamaNumCtx(&p)
prov := providers.NewOllamaProvider(p.Name, config.DockerLocalhost(host), "llama3.3", numCtx, nil).
WithThinkingEnabled(store.ParseThinkingEnabled(p.Settings))
registry.RegisterForTenant(p.TenantID, prov)
@@ -397,7 +397,7 @@ func registerProvidersFromDB(registry *providers.Registry, provStore store.Provi
if base == "" {
base = "https://ollama.com"
}
numCtx := resolveOllamaNumCtx(&p, base, p.APIKey)
numCtx := resolveOllamaNumCtx(&p)
prov := providers.NewOllamaProvider(p.Name, base, "llama3.3", numCtx, nil).
WithThinkingEnabled(store.ParseThinkingEnabled(p.Settings))
registry.RegisterForTenant(p.TenantID, prov)
@@ -466,24 +466,18 @@ func openAIProviderDefaults(providerType, apiBase string) (string, string) {
}
}
// resolveOllamaNumCtx returns the num_ctx to use for an Ollama provider, or nil
// when the built-in default should be used (provider handles it internally).
// Priority:
// 1. User-configured num_ctx from provider settings JSONB (explicit override wins).
// 2. Value queried from Ollama /api/show for the provider's default model.
// 3. nil when neither is available (OllamaProvider omits options.num_ctx, using Ollama's default).
func resolveOllamaNumCtx(p *store.LLMProviderData, apiBase, apiKey string) *int {
// resolveOllamaNumCtx returns the operator-configured num_ctx for an Ollama
// provider, or nil to let the provider resolve it per model at request time.
//
// Only the explicit settings JSONB override is honoured here. Probing /api/show
// at startup cannot work: the model an agent will use is not known until it
// sends a request, so the probe had to guess a model name, and a wrong guess
// resolved to nothing. OllamaProvider.resolveNumCtx does the lookup against the
// real model instead, and caches it.
func resolveOllamaNumCtx(p *store.LLMProviderData) *int {
if s := store.ParseOllamaSettings(p.Settings); s != nil {
return s.NumCtx
}
// Query the Ollama API for the model's native context length.
// Use a short timeout so startup is not blocked by a slow/absent Ollama server.
ctx, cancel := context.WithTimeout(context.Background(), 5*time.Second)
defer cancel()
numCtx := providers.FetchOllamaModelContext(ctx, apiBase, "llama3.3", apiKey)
if numCtx != providers.OllamaDefaultNumCtx {
return &numCtx
}
return nil
}