# llama.cpp.config.example # # Copy this file to llama.cpp.config and adjust it for your host before # running llamacppctl. See docs/SECURITY_AND_OPERATIONS.md for details on # every field and on the security model for --system-url/--prompt-url. # # Section types: # [default] global defaults, merged over the built-in fallbacks # [model.] a model profile, selected via --profile # [prompt.] a system-prompt profile, selected via --system-prompt-profile # # container_name is MANDATORY and must be unique per profile you intend to # run concurrently: it is the single identity anchor for Docker naming, # locking (--change), and --stop/--check targeting. [default] image = ghcr.io/ggml-org/llama.cpp:server-cuda hf_home = ${HF_HOME} model_path = models/qwen3/Qwen3.6-35B-A3B-Uncensored-HauhauCS-Aggressive-Q4_K_M.gguf container_name = llama_cpp_server host_port = 8001 container_port = 8000 model_alias = default_llm gpu_device = 1 restart_policy = unless-stopped # Groß-Profil für literarische Texte / durchdachte Reden: volles # 256k-Kontextfenster (= n_ctx_train des Modells) und sehr lange Antworten. # Empirisch gemessen passt der q4_0-KV-Cache dafür knapp auf EINE 24-GB-3090 # (~22,6 GB inkl. 21-GB-Modell). Bei Bedarf auf 131072/65536 senken. ctx_size = 262144 n_predict = 32768 temp = 0.65 top_p = 0.80 top_k = 20 min_p = 0.01 repeat_penalty = 1.05 main_gpu = 0 ngl = 999 fa = true kv_unified = true jinja = true reasoning = on no_context_shift = true cache_type_k = q4_0 cache_type_v = q4_0 batch_size = 1024 ubatch_size = 512 parallel = 1 cont_batching = true health_endpoint = /health models_endpoint = /v1/models chat_endpoint = /v1/chat/completions timeout = 300 poll_interval = 2 # Chat-Antwortbudget (Reasoning-Modell braucht viel; für lange Texte hoch): max_tokens = 32768 # chat_temperature leer lassen -> Server-Temp (temp oben) gilt. chat_temperature = # Modell-Profile: erben ALLES aus [default] (Port 8001, Container # llama_cpp_server, Alias default_llm, gpu_device 1, ctx_size 262144) und # überschreiben NUR das Modell. Es läuft also immer ein Modell auf dem # Standard-Port; --start/--change tauscht es aus. Wer Modelle GLEICHZEITIG # betreiben will, gibt dem Profil zusätzlich einen eigenen container_name + # host_port (Muster siehe llama.cpp.config.example). [model.carnice] model_path = models/qwen3/Carnice-Qwen3.6-MoE-35B-A3B-Q4_K_M.gguf [model.qwen27] model_path = models/qwen3/Qwen3.6-27B-Uncensored-HauhauCS-Aggressive-IQ4_XS.gguf [model.qwopus] model_path = models/qwen3/Qwopus3.6-35B-A3B-v1-Q4_K_M.gguf [prompt.concise] system_prompt = Du antwortest kurz, präzise und technisch. [prompt.coding] system_prompt = Du bist ein erfahrener Linux-, Python- und LLM-Infrastruktur-Engineer.