36 lines
1.6 KiB
Nix
36 lines
1.6 KiB
Nix
{ pkgs, ... }:
|
|
|
|
{
|
|
services.ollama = {
|
|
enable = true;
|
|
# `services.ollama.acceleration` does not exist (that option only lives on
|
|
# services.tabby); GPU backend selection for ollama is done by package choice.
|
|
package = pkgs.ollama-rocm;
|
|
|
|
# Same "don't expose services directly" pattern as the rest of the homelab:
|
|
# bind to localhost only. If this needs to be reachable from other devices
|
|
# later, put it behind Traefik + Headscale like the other services instead
|
|
# of opening it up directly.
|
|
host = "127.0.0.1";
|
|
port = 11434;
|
|
|
|
environmentVariables = {
|
|
# Ollama is on-demand by design: it loads a model into VRAM on the first
|
|
# request and unloads it after this much idle time. 5m is already the
|
|
# ollama default, set explicitly so it's not an implicit behavior.
|
|
OLLAMA_KEEP_ALIVE = "5m";
|
|
};
|
|
|
|
# RDNA4 (gfx1200/gfx1201, incl. the RX 9060 XT) has official ROCm support
|
|
# since ROCm 6.4.4. Verified on this host (2026-08-30) with rocmPackages.clr
|
|
# 7.2.3: `rocminfo` correctly reports "Name: gfx1200" and `rocm-smi
|
|
# --showmeminfo vram` reports the full 16GiB, so the "detected with 0 VRAM,
|
|
# falls back to CPU" issue does not reproduce here and rocmOverrideGfx is
|
|
# not needed. If a future nixpkgs bump regresses this, re-check with:
|
|
# nix shell nixpkgs#rocmPackages.rocminfo -c rocminfo | grep -i gfx
|
|
# nix shell nixpkgs#rocmPackages.rocm-smi -c rocm-smi --showmeminfo vram
|
|
# and only then set rocmOverrideGfx = "12.0.0"; (gfx1200) if VRAM reports 0.
|
|
# rocmOverrideGfx = "12.0.0";
|
|
};
|
|
}
|