summaryrefslogtreecommitdiff
path: root/machines
diff options
context:
space:
mode:
authorHenry J Webster <hwebs@hwebs.info>2026-08-04 00:02:54 -0500
committerHenry J. Webster <hwebs@hwebs.info>2026-08-04 16:05:20 -0500
commitdf960b2ed92f9e45fc4984d19ab65c55074bb1d7 (patch)
tree702ad999126a55f93122277c986a1bd6974a47be /machines
parent8f9fbf2f4a269eae143c9e279afc6608ce64b7d0 (diff)
kusanagi/ollama: double default context, add 1h keep-alive
The 12B model at 32k context uses only 7.7 GB of the 7900 XT's 20 GB, so there is ample VRAM headroom for 64k with the q8_0 KV cache — agentic coding tools are context-hungry and 32k is the practical bottleneck. The keep-alive stops the default 5-minute idle unload from adding a reload stall to every resumed session on a single-user workstation. Update flake to get newer packages. Assisted-by: Claude Code:claude-fable-5
Diffstat (limited to 'machines')
-rw-r--r--machines/kusanagi/default.nix5
1 files changed, 4 insertions, 1 deletions
diff --git a/machines/kusanagi/default.nix b/machines/kusanagi/default.nix
index ff786f6..1c8ccb1 100644
--- a/machines/kusanagi/default.nix
+++ b/machines/kusanagi/default.nix
@@ -228,8 +228,11 @@
# KV-cache quantization (below) is a no-op without flash attention; ollama
# falls back to f16 KV cache, inflating VRAM use and forcing CPU offload.
OLLAMA_FLASH_ATTENTION = "1";
- OLLAMA_CONTEXT_LENGTH = "32768";
+ OLLAMA_CONTEXT_LENGTH = "65536";
OLLAMA_KV_CACHE_TYPE = "q8_0";
+ # Keep the model resident between agent turns; the default 5-minute
+ # keep-alive forces a multi-second reload after any pause.
+ OLLAMA_KEEP_ALIVE = "1h";
};
};