From df960b2ed92f9e45fc4984d19ab65c55074bb1d7 Mon Sep 17 00:00:00 2001 From: Henry J Webster Date: Tue, 4 Aug 2026 00:02:54 -0500 Subject: kusanagi/ollama: double default context, add 1h keep-alive MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The 12B model at 32k context uses only 7.7 GB of the 7900 XT's 20 GB, so there is ample VRAM headroom for 64k with the q8_0 KV cache — agentic coding tools are context-hungry and 32k is the practical bottleneck. The keep-alive stops the default 5-minute idle unload from adding a reload stall to every resumed session on a single-user workstation. Update flake to get newer packages. Assisted-by: Claude Code:claude-fable-5 --- machines/kusanagi/default.nix | 5 ++++- 1 file changed, 4 insertions(+), 1 deletion(-) (limited to 'machines/kusanagi/default.nix') diff --git a/machines/kusanagi/default.nix b/machines/kusanagi/default.nix index ff786f6..1c8ccb1 100644 --- a/machines/kusanagi/default.nix +++ b/machines/kusanagi/default.nix @@ -228,8 +228,11 @@ # KV-cache quantization (below) is a no-op without flash attention; ollama # falls back to f16 KV cache, inflating VRAM use and forcing CPU offload. OLLAMA_FLASH_ATTENTION = "1"; - OLLAMA_CONTEXT_LENGTH = "32768"; + OLLAMA_CONTEXT_LENGTH = "65536"; OLLAMA_KV_CACHE_TYPE = "q8_0"; + # Keep the model resident between agent turns; the default 5-minute + # keep-alive forces a multi-second reload after any pause. + OLLAMA_KEEP_ALIVE = "1h"; }; }; -- cgit v1.3