Rename the size-numbered profiles so the digits read as parameter count, not Qwen version: qwen35 -> qwen35b (llama.cpp.config.example, build_archive smoke test, SECURITY_AND_OPERATIONS.md) and qwen27 -> qwen27b (llama.cpp.config). Drop the misleading `--profile qwen35` from the README/man/install examples: that profile only exists in the .example file, so pasted commands failed against the real config. Primary examples now omit --profile (using the [default] section, which always resolves), with one example plus a note showing how to select a real [model.<name>] profile. Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
79 lines
2.7 KiB
Text
79 lines
2.7 KiB
Text
# llama.cpp.config.example
|
|
#
|
|
# Copy this file to llama.cpp.config and adjust it for your host before
|
|
# running llamacppctl. See docs/SECURITY_AND_OPERATIONS.md for details on
|
|
# every field and on the security model for --system-url/--prompt-url.
|
|
#
|
|
# Section types:
|
|
# [default] global defaults, merged over the built-in fallbacks
|
|
# [model.<name>] a model profile, selected via --profile <name>
|
|
# [prompt.<name>] a system-prompt profile, selected via --system-prompt-profile <name>
|
|
#
|
|
# container_name is MANDATORY and must be unique per profile you intend to
|
|
# run concurrently: it is the single identity anchor for Docker naming,
|
|
# locking (--change), and --stop/--check targeting.
|
|
|
|
[default]
|
|
image = ghcr.io/ggml-org/llama.cpp:server-cuda
|
|
hf_home = ${HF_HOME}
|
|
model_path = models/qwen3/Qwen3.6-35B-A3B-Uncensored-HauhauCS-Aggressive-Q4_K_M.gguf
|
|
container_name = llama_cpp_server
|
|
host_port = 8001
|
|
container_port = 8000
|
|
model_alias = default_llm
|
|
gpu_device = 1
|
|
restart_policy = unless-stopped
|
|
# Groß-Profil für literarische Texte / durchdachte Reden: volles
|
|
# 256k-Kontextfenster (= n_ctx_train des Modells) und sehr lange Antworten.
|
|
# Empirisch gemessen passt der q4_0-KV-Cache dafür knapp auf EINE 24-GB-3090
|
|
# (~22,6 GB inkl. 21-GB-Modell). Bei Bedarf auf 131072/65536 senken.
|
|
ctx_size = 262144
|
|
n_predict = 32768
|
|
temp = 0.65
|
|
top_p = 0.80
|
|
top_k = 20
|
|
min_p = 0.01
|
|
repeat_penalty = 1.05
|
|
main_gpu = 0
|
|
ngl = 999
|
|
fa = true
|
|
kv_unified = true
|
|
jinja = true
|
|
reasoning = on
|
|
no_context_shift = true
|
|
cache_type_k = q4_0
|
|
cache_type_v = q4_0
|
|
batch_size = 1024
|
|
ubatch_size = 512
|
|
parallel = 1
|
|
cont_batching = true
|
|
health_endpoint = /health
|
|
models_endpoint = /v1/models
|
|
chat_endpoint = /v1/chat/completions
|
|
timeout = 300
|
|
poll_interval = 2
|
|
# Chat-Antwortbudget (Reasoning-Modell braucht viel; für lange Texte hoch):
|
|
max_tokens = 32768
|
|
# chat_temperature leer lassen -> Server-Temp (temp oben) gilt.
|
|
chat_temperature =
|
|
|
|
# Modell-Profile: erben ALLES aus [default] (Port 8001, Container
|
|
# llama_cpp_server, Alias default_llm, gpu_device 1, ctx_size 262144) und
|
|
# überschreiben NUR das Modell. Es läuft also immer ein Modell auf dem
|
|
# Standard-Port; --start/--change tauscht es aus. Wer Modelle GLEICHZEITIG
|
|
# betreiben will, gibt dem Profil zusätzlich einen eigenen container_name +
|
|
# host_port (Muster siehe llama.cpp.config.example).
|
|
[model.carnice]
|
|
model_path = models/qwen3/Carnice-Qwen3.6-MoE-35B-A3B-Q4_K_M.gguf
|
|
|
|
[model.qwen27b]
|
|
model_path = models/qwen3/Qwen3.6-27B-Uncensored-HauhauCS-Aggressive-IQ4_XS.gguf
|
|
|
|
[model.qwopus]
|
|
model_path = models/qwen3/Qwopus3.6-35B-A3B-v1-Q4_K_M.gguf
|
|
|
|
[prompt.concise]
|
|
system_prompt = Du antwortest kurz, präzise und technisch.
|
|
|
|
[prompt.coding]
|
|
system_prompt = Du bist ein erfahrener Linux-, Python- und LLM-Infrastruktur-Engineer.
|