update llama-swap defs

This commit is contained in:
2026-01-02 19:49:00 -05:00
parent f3ceb57e5e
commit 685d12dabd
2 changed files with 25 additions and 7 deletions

View File

@@ -62,14 +62,15 @@ in
}; };
}; };
}; };
mcp = { lsp = {
gopls = { starlark = {
type = "local";
command = [ command = [
"${pkgs.gopls}/bin/gopls" "${pkgs.pyright}/bin/pyright-langserver"
"mcp" "--stdio"
];
extensions = [
".star"
]; ];
enabled = true;
}; };
}; };
}; };

View File

@@ -112,6 +112,7 @@ in
-fit off \ -fit off \
-dev CUDA0 -dev CUDA0
''; '';
env = [ "GGML_CUDA_ENABLE_UNIFIED_MEMORY=1" ];
}; };
# https://huggingface.co/mradermacher/gpt-oss-20b-heretic-v2-i1-GGUF/tree/main # https://huggingface.co/mradermacher/gpt-oss-20b-heretic-v2-i1-GGUF/tree/main
@@ -128,6 +129,7 @@ in
--top-k 40 \ --top-k 40 \
-dev CUDA0 -dev CUDA0
''; '';
env = [ "GGML_CUDA_ENABLE_UNIFIED_MEMORY=1" ];
}; };
# https://huggingface.co/mradermacher/GPT-OSS-Cybersecurity-20B-Merged-i1-GGUF/tree/main # https://huggingface.co/mradermacher/GPT-OSS-Cybersecurity-20B-Merged-i1-GGUF/tree/main
@@ -143,8 +145,12 @@ in
--top-k 40 \ --top-k 40 \
-dev CUDA0 -dev CUDA0
''; '';
env = [ "GGML_CUDA_ENABLE_UNIFIED_MEMORY=1" ];
}; };
# https://huggingface.co/shb777/Llama-3.3-8B-Instruct-GGUF/tree/main
# llama-server --host 0.0.0.0 --port 8081 -m /mnt/ssd/Models/Llama/llama-3.3-8b-instruct-q6_k.gguf -c 131072 -dev CUDA0 -fit off
# https://huggingface.co/unsloth/Qwen3-Next-80B-A3B-Instruct-GGUF/tree/main # https://huggingface.co/unsloth/Qwen3-Next-80B-A3B-Instruct-GGUF/tree/main
"qwen3-next-80b-instruct" = { "qwen3-next-80b-instruct" = {
name = "Qwen3 Next (80B) - Instruct"; name = "Qwen3 Next (80B) - Instruct";
@@ -162,6 +168,7 @@ in
-ctv q8_0 \ -ctv q8_0 \
-fit off -fit off
''; '';
env = [ "GGML_CUDA_ENABLE_UNIFIED_MEMORY=1" ];
}; };
# https://huggingface.co/unsloth/Qwen3-30B-A3B-Instruct-2507-GGUF/tree/main # https://huggingface.co/unsloth/Qwen3-30B-A3B-Instruct-2507-GGUF/tree/main
@@ -181,6 +188,7 @@ in
-ctv q8_0 \ -ctv q8_0 \
-ts 70,30 -ts 70,30
''; '';
env = [ "GGML_CUDA_ENABLE_UNIFIED_MEMORY=1" ];
}; };
# https://huggingface.co/unsloth/Qwen3-Coder-30B-A3B-Instruct-GGUF/tree/main # https://huggingface.co/unsloth/Qwen3-Coder-30B-A3B-Instruct-GGUF/tree/main
@@ -200,6 +208,7 @@ in
-ctv q8_0 \ -ctv q8_0 \
-ts 70,30 -ts 70,30
''; '';
env = [ "GGML_CUDA_ENABLE_UNIFIED_MEMORY=1" ];
}; };
# https://huggingface.co/unsloth/Qwen3-30B-A3B-Thinking-2507-GGUF/tree/main # https://huggingface.co/unsloth/Qwen3-30B-A3B-Thinking-2507-GGUF/tree/main
@@ -219,6 +228,7 @@ in
-ctv q8_0 \ -ctv q8_0 \
-ts 70,30 -ts 70,30
''; '';
env = [ "GGML_CUDA_ENABLE_UNIFIED_MEMORY=1" ];
}; };
# https://huggingface.co/unsloth/Nemotron-3-Nano-30B-A3B-GGUF/tree/main # https://huggingface.co/unsloth/Nemotron-3-Nano-30B-A3B-GGUF/tree/main
@@ -233,6 +243,7 @@ in
--top-p 0.95 \ --top-p 0.95 \
-fit off -fit off
''; '';
env = [ "GGML_CUDA_ENABLE_UNIFIED_MEMORY=1" ];
}; };
# https://huggingface.co/unsloth/Qwen3-VL-8B-Instruct-GGUF/tree/main # https://huggingface.co/unsloth/Qwen3-VL-8B-Instruct-GGUF/tree/main
@@ -253,6 +264,7 @@ in
-fit off \ -fit off \
-dev CUDA1 -dev CUDA1
''; '';
env = [ "GGML_CUDA_ENABLE_UNIFIED_MEMORY=1" ];
}; };
# https://huggingface.co/unsloth/Qwen2.5-Coder-7B-Instruct-128K-GGUF/tree/main # https://huggingface.co/unsloth/Qwen2.5-Coder-7B-Instruct-128K-GGUF/tree/main
@@ -267,6 +279,7 @@ in
-fit off \ -fit off \
-dev CUDA1 -dev CUDA1
''; '';
env = [ "GGML_CUDA_ENABLE_UNIFIED_MEMORY=1" ];
}; };
# https://huggingface.co/unsloth/Qwen2.5-Coder-3B-Instruct-128K-GGUF/tree/main # https://huggingface.co/unsloth/Qwen2.5-Coder-3B-Instruct-128K-GGUF/tree/main
@@ -280,6 +293,7 @@ in
-fit off \ -fit off \
-dev CUDA1 -dev CUDA1
''; '';
env = [ "GGML_CUDA_ENABLE_UNIFIED_MEMORY=1" ];
}; };
# https://huggingface.co/unsloth/Qwen3-4B-Instruct-2507-GGUF/tree/main # https://huggingface.co/unsloth/Qwen3-4B-Instruct-2507-GGUF/tree/main
@@ -295,6 +309,7 @@ in
-ctv q8_0 \ -ctv q8_0 \
-dev CUDA1 -dev CUDA1
''; '';
env = [ "GGML_CUDA_ENABLE_UNIFIED_MEMORY=1" ];
}; };
"z-image-turbo" = { "z-image-turbo" = {
@@ -311,6 +326,7 @@ in
--steps 9 \ --steps 9 \
--rng cuda --rng cuda
''; '';
env = [ "GGML_CUDA_ENABLE_UNIFIED_MEMORY=1" ];
}; };
"qwen-image-edit" = { "qwen-image-edit" = {
@@ -329,13 +345,14 @@ in
--steps 9 \ --steps 9 \
--rng cuda --rng cuda
''; '';
env = [ "GGML_CUDA_ENABLE_UNIFIED_MEMORY=1" ];
}; };
}; };
groups = { groups = {
shared = { shared = {
swap = true; swap = true;
exclusive = true; exclusive = false;
members = [ members = [
"nemotron-3-nano-30b-thinking" "nemotron-3-nano-30b-thinking"
"qwen3-30b-2507-instruct" "qwen3-30b-2507-instruct"