diff --git a/ollama/docker-compose.yml b/ollama/docker-compose.yml index 3cbbc0b..18bc455 100644 --- a/ollama/docker-compose.yml +++ b/ollama/docker-compose.yml @@ -10,6 +10,13 @@ services: - NVIDIA_DRIVER_CAPABILITIES=compute,utility - TZ=America/New_York - OLLAMA_HOST=0.0.0.0 + # This GPU (6GB) is right at the edge for qwen2.5:7b at 16384 context — + # KV cache size was observed varying 224MB-896MB between loads with flash + # attention off, and the larger figure tips offload to 25/29 layers + # instead of 29/29, causing multi-minute DJ-agent latency. Flash + # attention + an explicitly quantized KV cache keep it small and stable. + - OLLAMA_FLASH_ATTENTION=1 + - OLLAMA_KV_CACHE_TYPE=q8_0 volumes: - /srv/ollama:/root/.ollama runtime: nvidia