1amends "../../config/Service.pkl"
2
3import "../../config/site.pkl" as site
4import "pkl:json"
5
6local model = (new json.Parser {}).parse(read("catalog.json")).models[0]
7
8meta { name = "Local AI"; launcher = false; access = "admin" }
9enabled = (read?("env:STUDIO_LOCAL_AI") ?? "false") == "true"
10rollout = "simple"
11stageIsolation = "fresh"
12healthyDeadline = "30m"
13healthRestartGrace = "30m"
14secrets { ["api_key"] {} }
15
16container {
17 image = "ghcr.io/ggml-org/llama.cpp@sha256:e8318a7b3988f9ca57b28c03486569e60def08f2b7b060550ebfa046e175e6cb"
18 cpu = 4000
19 memory = 16384
20 entrypoint = "/usr/bin/python3"
21 http {
22 containerPort = 8080
23 subdomain = "ai"
24 checkPath = "/health"
25 }
26 volumes {
27 ["/config/gateway.py"] { config = "gateway.py" }
28 ["/models"] { src = "\(site.mediaRoot)/AI/LLM/Qwen3.8-27B"; readOnly = true }
29 // The host's NVIDIA libraries link to immutable Nix store paths.
30 ["/nvidia/lib"] { src = "/run/opengl-driver/lib"; readOnly = true }
31 ["/nix/store"] { src = "/nix/store"; readOnly = true }
32 }
33 devices { "/dev/nvidia0"; "/dev/nvidiactl"; "/dev/nvidia-uvm" }
34 env {
35 ["LD_LIBRARY_PATH"] = "/nvidia/lib:/app"
36 ["LLAMA_API_KEY"] = "${secret.own.api_key}"
37 }
38 args {
39 "/config/gateway.py"
40 "--model"; "/models/Qwen3.8-27B-UD-Q4_K_M.gguf"
41 "--mmproj"; "/models/mmproj-F16.gguf"
42 "--alias"; model.slug
43 "--host"; "127.0.0.1"
44 "--port"; "8081"
45 "--n-gpu-layers"; "99"
46 "--ctx-size"; model.context_window.toString()
47 "--parallel"; "1"
48 "--flash-attn"; "on"
49 "--cache-type-k"; "q8_0"
50 "--cache-type-v"; "q8_0"
51 "--cache-ram"; "8192"
52 "--jinja"
53 "--timeout"; "28800"
54 "--sse-ping-interval"; "10"
55 "--metrics"
56 "--temp"; "1.0"
57 "--top-p"; "0.95"
58 "--top-k"; "20"
59 "--min-p"; "0"
60 "--chat-template-kwargs"; "{\"enable_thinking\": true, \"reasoning_effort\": \"medium\", \"preserve_thinking\": true}"
61 }
62}