# llama.cpp.config.example # # Copy this file to llama.cpp.config and adjust it for your host before # running llamacppctl. See docs/SECURITY_AND_OPERATIONS.md for details on # every field and on the security model for --system-url/--prompt-url. # # Section types: # [default] global defaults, merged over the built-in fallbacks # [model.] a model profile, selected via --profile # [prompt.] a system-prompt profile, selected via --system-prompt-profile # # container_name is MANDATORY and must be unique per profile you intend to # run concurrently: it is the single identity anchor for Docker naming, # locking (--change), and --stop/--check targeting. [default] image = ghcr.io/ggml-org/llama.cpp:server-cuda # hf_home unterstützt Environment-Variablen und ~, z. B. hf_home = ${HF_HOME} hf_home = /srv/models model_path = qwen3/default.gguf container_name = llama_cpp_server host_port = 8001 container_port = 8000 model_alias = default_llm gpu_device = 0 restart_policy = unless-stopped ctx_size = 262144 n_predict = 16384 temp = 0.65 top_p = 0.80 top_k = 20 min_p = 0.01 repeat_penalty = 1.05 main_gpu = 0 ngl = 999 fa = true kv_unified = true jinja = true reasoning = on no_context_shift = true cache_type_k = q4_0 cache_type_v = q4_0 batch_size = 1024 ubatch_size = 512 parallel = 1 cont_batching = true health_endpoint = /health models_endpoint = /v1/models chat_endpoint = /v1/chat/completions timeout = 300 poll_interval = 2 # Port nur auf localhost veröffentlichen (Default). expose = true bindet auf # alle Interfaces (LAN) -> dann unbedingt api_key setzen. expose = false # api_key: leer = keine Authentifizierung. Gesetzt -> Server verlangt ihn und # das Tool sendet ihn als Bearer-Token. api_key = # Chat-Antwortbudget (wirkt auf --chat / --start-Antwort, nicht auf den Container). # Reasoning-Modelle brauchen viel Budget; für lange Texte hochsetzen. max_tokens = 2048 # chat_temperature leer lassen -> die Server-Temperatur (temp) gilt. chat_temperature = # Example second model profile, pinned to the second GPU (e.g. RTX 3090 #2) # with its own port and container name so it can run alongside [default]. [model.qwen35] model_path = qwen3/Qwen3-35B-A3B-Q4_K_M.gguf model_alias = qwen35 container_name = llama_cpp_qwen35 host_port = 8002 gpu_device = 1 [model.deepseek] model_path = deepseek/deepseek-r1-q4.gguf model_alias = deepseek container_name = llama_cpp_deepseek host_port = 8003 gpu_device = 1 ctx_size = 131072 [prompt.concise] system_prompt = Du antwortest kurz, präzise und technisch. [prompt.coding] system_prompt = Du bist ein erfahrener Linux-, Python- und LLM-Infrastruktur-Engineer.