Files

36 lines
1.6 KiB
Nix

{ pkgs, ... }:
{
services.ollama = {
enable = true;
# `services.ollama.acceleration` does not exist (that option only lives on
# services.tabby); GPU backend selection for ollama is done by package choice.
package = pkgs.ollama-rocm;
# Same "don't expose services directly" pattern as the rest of the homelab:
# bind to localhost only. If this needs to be reachable from other devices
# later, put it behind Traefik + Headscale like the other services instead
# of opening it up directly.
host = "127.0.0.1";
port = 11434;
environmentVariables = {
# Ollama is on-demand by design: it loads a model into VRAM on the first
# request and unloads it after this much idle time. 5m is already the
# ollama default, set explicitly so it's not an implicit behavior.
OLLAMA_KEEP_ALIVE = "5m";
};
# RDNA4 (gfx1200/gfx1201, incl. the RX 9060 XT) has official ROCm support
# since ROCm 6.4.4. Verified on this host (2026-08-30) with rocmPackages.clr
# 7.2.3: `rocminfo` correctly reports "Name: gfx1200" and `rocm-smi
# --showmeminfo vram` reports the full 16GiB, so the "detected with 0 VRAM,
# falls back to CPU" issue does not reproduce here and rocmOverrideGfx is
# not needed. If a future nixpkgs bump regresses this, re-check with:
# nix shell nixpkgs#rocmPackages.rocminfo -c rocminfo | grep -i gfx
# nix shell nixpkgs#rocmPackages.rocm-smi -c rocm-smi --showmeminfo vram
# and only then set rocmOverrideGfx = "12.0.0"; (gfx1200) if VRAM reports 0.
# rocmOverrideGfx = "12.0.0";
};
}