Files
TheaninovOS/modules/nixos/usecases/localai.nix
T
2026-10-04 13:08:43 +02:00

160 lines
4.0 KiB
Nix

{
config,
lib,
pkgs,
username,
...
}:
with lib;
let
cfg = config.usecases.localai;
models-preset = pkgs.writeText "presets.ini" (
pkgs.lib.generators.toINI { } {
"Qwen3.8-Flash" = {
# model = "/mnt/llms/Qwen3.8-Flash/Qwen3.8-Flash-Next-GSQ-RCO-IQ3_XXS-00001-of-00002.gguf";
model = "/mnt/llms/Qwen3.8-Flash/Qwen3.8-Flash-Next-GSQ-RCO-IQ3_S-00001-of-00002.gguf";
cpu-range = "0-5";
cpu-strict = 1;
threads = 6;
threads-batch = 12;
cpu-range-batch = "0-11";
parallel = 1;
cpu-moe = true;
load-mode = "mmap";
lazy-mode = "on";
flash-attn = "on";
mmproj = "/mnt/llms/Qwen3.8-Flash/mmproj-Qwen3.8-Flash-Next-BF16.gguf";
no-mmproj-offload = true;
# moe-cache-profile = "/mnt/llms/Qwen3.8-Flash/trace/qwen-merged.csv";
# moe-cache-slots = 48;
/*
spec-type = "draft-mtp";
spec-draft-model = "/mnt/llms/Qwen3.8-Flash/mtp-Qwen3.8-Flash-Next-shared-Q8_0.gguf";
spec-draft-n-max = 2;
spec-draft-ngl = 99;
spec-draft-threads = 6;
spec-draft-cpu-range = "6-11";
spec-draft-cpu-strict = 1;
spec-draft-type-k = "q8_0";
spec-draft-type-v = "q8_0";
*/
cache-reuse = 256;
fit = "on";
#no-sched-async-cpu = true;
cache-ram = 16384;
cache-idle-slots = true;
kv-offload = true;
kv-unified = true;
cache-type-k = "q8_0";
cache-type-v = "q8_0";
ctx-size = 262144; # 150000; # 131072;
batch-size = 2048;
ubatch-size = 512;
jinja = true;
temperature = 1.0;
top-p = 0.95;
top-k = 20;
min-p = 0.0;
presence-penalty = 0.0;
repeat-penalty = 1.0;
};
}
);
llama-moe-trace = pkgs.writeShellApplication {
name = "llama-moe-trace";
runtimeEnv.LD_LIBRARY_PATH = pkgs.lib.makeLibraryPath [
pkgs.gcc.cc.lib
];
text = ''
exec ${pkgs.llama-cpp-codacus}/bin/llama-moe-trace "$@"
'';
};
create-cache = pkgs.writeShellApplication {
name = "llama-create-moe-cache";
runtimeInputs = [
pkgs.matugen
pkgs.awww
pkgs.zenity
pkgs.sunwait
];
runtimeEnv = {
HSA_OVERRIDE_GFX_VERSION = "10.3.0";
HIP_VISIBLE_DEVICES = 0;
ROC_ENABLE_PRE_VEGA = 1;
SERVER = "localhost:9931";
MODEL = "Qwen3.8-Flash";
BIN = pkgs.lib.getExe llama-moe-trace;
TRACE_DIR = "/mnt/llms/Qwen3.8-Flash/trace";
};
text = builtins.readFile ./localai_trace.sh;
};
in
{
options.usecases.localai = {
enable = mkEnableOption "Enable local LLM services";
};
config = mkIf cfg.enable {
home-manager.users.${username}.home.packages = [
create-cache
pkgs.llama-cpp-codacus
pkgs.opencode
pkgs.github-copilot-cli
];
systemd.services.llama-cpp.serviceConfig.Environment = [
"HSA_OVERRIDE_GFX_VERSION=10.3.0"
"HIP_VISIBLE_DEVICES=0"
"ROC_ENABLE_PRE_VEGA=1"
];
services = {
llama-cpp = {
enable = true;
package = pkgs.llama-cpp-codacus;
settings = {
host = "127.0.0.1";
port = 9931;
models-max = 1;
inherit models-preset;
};
};
/*
open-webui = {
enable = true;
port = 57461;
environment = {
ANONYMIZED_TELEMETRY = "False";
DO_NOT_TRACK = "True";
SCARF_NO_ANALYTICS = "True";
WEBUI_AUTH = "False";
};
};
*/
};
/*
virtualisation.oci-containers.containers.open-terminal =
let
xdg = config.home-manager.users.${username}.xdg;
in
{
podman.user = "${username}";
image = "ghcr.io/open-webui/open-terminal";
volumes = [ "${xdg.dataHome}/open-terminal:/home/user" ];
ports = [ "54183:8000" ];
extraOptions = [ "--env-file=${xdg.configHome}/open-terminal.env" ];
};
*/
};
}