{ pkgs, ... }: { services.ollama = { enable = true; # `services.ollama.acceleration` does not exist (that option only lives on # services.tabby); GPU backend selection for ollama is done by package choice. package = pkgs.ollama-rocm; # Same "don't expose services directly" pattern as the rest of the homelab: # bind to localhost only. If this needs to be reachable from other devices # later, put it behind Traefik + Headscale like the other services instead # of opening it up directly. host = "127.0.0.1"; port = 11434; environmentVariables = { # Ollama is on-demand by design: it loads a model into VRAM on the first # request and unloads it after this much idle time. 5m is already the # ollama default, set explicitly so it's not an implicit behavior. OLLAMA_KEEP_ALIVE = "5m"; }; # RDNA4 (gfx1200/gfx1201, incl. the RX 9060 XT) has official ROCm support # since ROCm 6.4.4. Verified on this host (2026-08-30) with rocmPackages.clr # 7.2.3: `rocminfo` correctly reports "Name: gfx1200" and `rocm-smi # --showmeminfo vram` reports the full 16GiB, so the "detected with 0 VRAM, # falls back to CPU" issue does not reproduce here and rocmOverrideGfx is # not needed. If a future nixpkgs bump regresses this, re-check with: # nix shell nixpkgs#rocmPackages.rocminfo -c rocminfo | grep -i gfx # nix shell nixpkgs#rocmPackages.rocm-smi -c rocm-smi --showmeminfo vram # and only then set rocmOverrideGfx = "12.0.0"; (gfx1200) if VRAM reports 0. # rocmOverrideGfx = "12.0.0"; }; }