llamacppctl/llama.cpp.config.example

87 lines
2.8 KiB
Text
Raw Permalink Normal View History

# llama.cpp.config.example
#
# Copy this file to llama.cpp.config and adjust it for your host before
# running llamacppctl. See docs/SECURITY_AND_OPERATIONS.md for details on
# every field and on the security model for --system-url/--prompt-url.
#
# Section types:
# [default] global defaults, merged over the built-in fallbacks
# [model.<name>] a model profile, selected via --profile <name>
# [prompt.<name>] a system-prompt profile, selected via --system-prompt-profile <name>
#
# container_name is MANDATORY and must be unique per profile you intend to
# run concurrently: it is the single identity anchor for Docker naming,
# locking (--change), and --stop/--check targeting.
[default]
# Pinned by digest for reproducibility. The :server-cuda tag is a moving target;
# to update, pull it, read the new digest, and replace the pin below.
image = ghcr.io/ggml-org/llama.cpp@sha256:5535de118ed457f761cbfeacd7e10fef31cb391ca7cac1d5c78b11d28fcf88e6
# hf_home unterstützt Environment-Variablen und ~, z. B. hf_home = ${HF_HOME}
hf_home = /srv/models
model_path = qwen3/default.gguf
container_name = llama_cpp_server
host_port = 8001
container_port = 8000
model_alias = default_llm
gpu_device = 0
restart_policy = unless-stopped
ctx_size = 262144
n_predict = 16384
temp = 0.65
top_p = 0.80
top_k = 20
min_p = 0.01
repeat_penalty = 1.05
main_gpu = 0
ngl = 999
fa = true
kv_unified = true
jinja = true
reasoning = on
no_context_shift = true
cache_type_k = q4_0
cache_type_v = q4_0
batch_size = 1024
ubatch_size = 512
parallel = 1
cont_batching = true
health_endpoint = /health
models_endpoint = /v1/models
chat_endpoint = /v1/chat/completions
timeout = 300
poll_interval = 2
# Port nur auf localhost veröffentlichen (Default). expose = true bindet auf
# alle Interfaces (LAN) -> dann unbedingt api_key setzen.
expose = false
# api_key: leer = keine Authentifizierung. Gesetzt -> Server verlangt ihn und
# das Tool sendet ihn als Bearer-Token.
api_key =
# Chat-Antwortbudget (wirkt auf --chat / --start-Antwort, nicht auf den Container).
# Reasoning-Modelle brauchen viel Budget; für lange Texte hochsetzen.
max_tokens = 2048
# chat_temperature leer lassen -> die Server-Temperatur (temp) gilt.
chat_temperature =
# Example second model profile, pinned to the second GPU (e.g. RTX 3090 #2)
# with its own port and container name so it can run alongside [default].
[model.qwen35b]
model_path = qwen3/Qwen3-35B-A3B-Q4_K_M.gguf
model_alias = qwen35b
container_name = llama_cpp_qwen35b
host_port = 8002
gpu_device = 1
[model.deepseek]
model_path = deepseek/deepseek-r1-q4.gguf
model_alias = deepseek
container_name = llama_cpp_deepseek
host_port = 8003
gpu_device = 1
ctx_size = 131072
[prompt.concise]
system_prompt = Du antwortest kurz, präzise und technisch.
[prompt.coding]
system_prompt = Du bist ein erfahrener Linux-, Python- und LLM-Infrastruktur-Engineer.