services: ollama: container_name: ollama image: ollama/ollama:latest restart: unless-stopped ports: - 127.0.0.1:11434:11434 environment: - NVIDIA_VISIBLE_DEVICES=all - NVIDIA_DRIVER_CAPABILITIES=compute,utility - TZ=America/New_York - OLLAMA_HOST=0.0.0.0 # This GPU (6GB) is right at the edge for qwen2.5:7b at 16384 context — # KV cache size was observed varying 224MB-896MB between loads with flash # attention off, and the larger figure tips offload to 25/29 layers # instead of 29/29, causing multi-minute DJ-agent latency. Flash # attention + an explicitly quantized KV cache keep it small and stable. - OLLAMA_FLASH_ATTENTION=1 - OLLAMA_KV_CACHE_TYPE=q8_0 # Server-level default context, kept in step with SUB/WAVE's own # llm.numCtx (settings.json) -- 11264 is the largest value that still # fits the full 29-layer GPU offload on this 6GB card with margin, while # staying above the ~7.5k-token peak seen in real multi-turn DJ-agent # calls (a lower ceiling truncates the front of the prompt -- including # the tool definitions -- and the agent stops calling `done`, #291). - OLLAMA_CONTEXT_LENGTH=11264 volumes: - /srv/ollama:/root/.ollama runtime: nvidia networks: - npm-network # Optional: Web UI for Ollama open-webui: container_name: open-webui image: ghcr.io/open-webui/open-webui:latest restart: unless-stopped ports: - 3000:8080 environment: - OLLAMA_BASE_URL=http://ollama:11434 - TZ=America/New_York - ENABLE_OAUTH_SIGNUP=true - OAUTH_MERGE_ACCOUNTS_BY_EMAIL=true - OAUTH_PROVIDER_NAME=Authelia - OPENID_PROVIDER_URL=https://auth.kolpacksoftware.com/.well-known/openid-configuration - OAUTH_CLIENT_ID=open-webui - OAUTH_CLIENT_SECRET=${AUTHELIA_OIDC_CLIENT_SECRET_OPEN_WEBUI} - OAUTH_TOKEN_ENDPOINT_AUTH_METHOD=client_secret_post - WEBUI_SECRET_KEY=${WEBUI_SECRET_KEY} volumes: - /srv/open-webui:/app/backend/data depends_on: - ollama networks: - npm-network networks: npm-network: external: true