From 462ca7c112d3cfb10593898db70b324ccd9b023f Mon Sep 17 00:00:00 2001 From: teknium1 <127238744+teknium1@users.noreply.github.com> Date: Tue, 15 Sep 2026 11:39:13 -0700 Subject: [PATCH] docs(config): explicit ollama_num_ctx is never capped by context_length agent/agent_init.py::_configure_ollama_num_ctx caps only the auto-detected value (the cap is skipped when an explicit override is set). The salvaged example wording said context_length caps the explicit value too, which would send readers chasing a cap that does not apply. --- cli-config.yaml.example | 5 +++-- 1 file changed, 3 insertions(+), 2 deletions(-) diff --git a/cli-config.yaml.example b/cli-config.yaml.example index 4f4b323d6f..dfaeed15d8 100644 --- a/cli-config.yaml.example +++ b/cli-config.yaml.example @@ -113,8 +113,9 @@ model: # # ollama_num_ctx: Ollama only — the num_ctx sent on every chat request. # Hermes auto-detects the model's window and sends it (Ollama otherwise - # defaults to 2048); set this to cap VRAM use. context_length, if set, - # caps it further. + # defaults to 2048); context_length, if set, caps the detected value. + # An explicit ollama_num_ctx is sent as-is (never capped) — set it to + # pin VRAM use. # # ollama_num_ctx: 32768 #