refactor(llama-router): extract atlas llm setup into reusable module

Assisted-by: pi (opus-4.8)
This commit is contained in:
Gabriel Fontes
2026-06-27 00:44:12 -03:00
parent 514f0a771b
commit fe0fcaeb05
3 changed files with 98 additions and 46 deletions
+6 -46
View File
@@ -1,9 +1,4 @@
{
lib,
outputs,
pkgs,
...
}: let
{outputs, ...}: let
port = 18080;
models = {
# flash-attn + q8_0 KV cache halves the KV footprint to stay under 8GB VRAM.
@@ -42,6 +37,11 @@
};
};
in {
services.llama-router = {
enable = true;
inherit port models;
};
services.nginx.virtualHosts."llm.m7.rs" = {
locations."/" = {
proxyPass = "http://localhost:${toString port}";
@@ -54,44 +54,4 @@ in {
'';
};
};
users.users.llama = {
isSystemUser = true;
group = "llama";
home = "/var/lib/llama";
};
users.groups.llama = {};
systemd.services.llama-cpp-router = {
description = "Local llama.cpp router API";
after = ["network-online.target"];
wants = ["network-online.target"];
wantedBy = ["multi-user.target"];
environment = {
XDG_CACHE_HOME = "/var/lib/llama";
XDG_DATA_HOME = "/var/lib/llama";
};
path = [pkgs.llama-cpp-vulkan];
script = lib.concatStringsSep " " [
"llama-server"
# No --models-dir: all models come from --models-preset below.
# The preset parser reads INI values verbatim (no quote stripping), so use
# plain INI, not TOML - quoted strings would keep their literal quotes.
"--models-preset ${(pkgs.formats.ini {}).generate "llama-cpp-models.ini" models}"
"--models-max 1"
"--host 127.0.0.1"
"--port ${toString port}"
];
serviceConfig = {
User = "llama";
Group = "llama";
StateDirectory = "llama";
RuntimeDirectory = "llama";
SupplementaryGroups = ["render" "video"];
Restart = "on-failure";
RestartSec = 2;
# llama-server can dawdle on shutdown; SIGKILL it after 15s.
TimeoutStopSec = 15;
};
};
}
+1
View File
@@ -4,4 +4,5 @@
opencode = import ./opencode.nix;
openrgb = import ./openrgb.nix;
nix-registry-prometheus-exporter = import ./nix-registry-prometheus-exporter.nix;
llama-router = import ./llama-router.nix;
}
+91
View File
@@ -0,0 +1,91 @@
{
pkgs,
lib,
config,
...
}: let
cfg = config.services.llama-router;
in {
options.services.llama-router = {
enable = lib.mkEnableOption "the local llama.cpp router API";
package = lib.mkPackageOption pkgs "llama-cpp-vulkan" {};
models = lib.mkOption {
type = (pkgs.formats.ini {}).type;
default = {};
description = ''
Model presets to serve, as an attrset mapping each model name to its
llama-server settings (e.g. `hf`, `ctx-size`, `n-gpu-layers`).
'';
};
host = lib.mkOption {
type = lib.types.str;
default = "127.0.0.1";
description = "Address llama-server binds to.";
};
port = lib.mkOption {
type = lib.types.port;
default = 18080;
description = "Port llama-server listens on.";
};
user = lib.mkOption {
type = lib.types.str;
default = "llama";
description = "User to run the router service as.";
};
group = lib.mkOption {
type = lib.types.str;
default = "llama";
description = "Group to run the router service as.";
};
};
config = lib.mkIf cfg.enable {
users.users = lib.mkIf (cfg.user == "llama") {
llama = {
isSystemUser = true;
group = cfg.group;
home = "/var/lib/llama";
};
};
users.groups = lib.mkIf (cfg.group == "llama") {
llama = {};
};
systemd.services.llama-cpp-router = {
description = "Local llama.cpp router API";
after = ["network-online.target"];
wants = ["network-online.target"];
wantedBy = ["multi-user.target"];
environment = {
XDG_CACHE_HOME = "/var/lib/llama";
XDG_DATA_HOME = "/var/lib/llama";
};
path = [cfg.package];
script = lib.concatStringsSep " " [
"llama-server"
# No --models-dir: all models come from --models-preset below.
"--models-preset ${(pkgs.formats.ini {}).generate "llama-cpp-models.ini" cfg.models}"
"--models-max 1"
"--host ${cfg.host}"
"--port ${toString cfg.port}"
];
serviceConfig = {
User = cfg.user;
Group = cfg.group;
StateDirectory = "llama";
RuntimeDirectory = "llama";
SupplementaryGroups = ["render" "video"];
Restart = "on-failure";
RestartSec = 2;
# llama-server can dawdle on shutdown; SIGKILL it after 15s.
TimeoutStopSec = 15;
};
};
};
}