Compare commits
2 commits
d72361cbae
...
9411c83a45
| Author | SHA1 | Date | |
|---|---|---|---|
| 9411c83a45 | |||
| c553654aae |
9 changed files with 49 additions and 6 deletions
|
|
@ -6,6 +6,24 @@ on:
|
||||||
pull_request:
|
pull_request:
|
||||||
|
|
||||||
jobs:
|
jobs:
|
||||||
|
lint:
|
||||||
|
runs-on: ubuntu-latest
|
||||||
|
steps:
|
||||||
|
- uses: actions/checkout@v4
|
||||||
|
|
||||||
|
- uses: actions/setup-python@v5
|
||||||
|
with:
|
||||||
|
python-version: "3.12"
|
||||||
|
|
||||||
|
- name: Install package with dev extras
|
||||||
|
run: pip install -e ".[dev]"
|
||||||
|
|
||||||
|
- name: Ruff (lint)
|
||||||
|
run: ruff check src/ tests/
|
||||||
|
|
||||||
|
- name: Mypy (type check)
|
||||||
|
run: mypy
|
||||||
|
|
||||||
test:
|
test:
|
||||||
runs-on: ubuntu-latest
|
runs-on: ubuntu-latest
|
||||||
strategy:
|
strategy:
|
||||||
|
|
|
||||||
|
|
@ -34,7 +34,8 @@ testbar von der Orchestrierung.
|
||||||
|
|
||||||
1. **Eingebaute Defaults** (`config.builtin_defaults()`) — spiegeln die
|
1. **Eingebaute Defaults** (`config.builtin_defaults()`) — spiegeln die
|
||||||
Standardwerte der ursprünglichen Shell-Skripte wider (Image
|
Standardwerte der ursprünglichen Shell-Skripte wider (Image
|
||||||
`ghcr.io/ggml-org/llama.cpp:server-cuda`, `host_port=8001`,
|
`ghcr.io/ggml-org/llama.cpp` per Digest gepinnt für Reproduzierbarkeit,
|
||||||
|
`host_port=8001`,
|
||||||
`container_name=va_llm`, `gpu_device=1`, `jinja/fa/kv_unified/
|
`container_name=va_llm`, `gpu_device=1`, `jinja/fa/kv_unified/
|
||||||
cont_batching/no_context_shift=true`, `reasoning=on`,
|
cont_batching/no_context_shift=true`, `reasoning=on`,
|
||||||
`cache_type_k/v=q4_0`, `batch_size=1024`, `ubatch_size=512`,
|
`cache_type_k/v=q4_0`, `batch_size=1024`, `ubatch_size=512`,
|
||||||
|
|
|
||||||
|
|
@ -14,7 +14,9 @@
|
||||||
# locking (--change), and --stop/--check targeting.
|
# locking (--change), and --stop/--check targeting.
|
||||||
|
|
||||||
[default]
|
[default]
|
||||||
image = ghcr.io/ggml-org/llama.cpp:server-cuda
|
# Pinned by digest for reproducibility. The :server-cuda tag is a moving target;
|
||||||
|
# to update, pull it, read the new digest, and replace the pin below.
|
||||||
|
image = ghcr.io/ggml-org/llama.cpp@sha256:5535de118ed457f761cbfeacd7e10fef31cb391ca7cac1d5c78b11d28fcf88e6
|
||||||
# hf_home unterstützt Environment-Variablen und ~, z. B. hf_home = ${HF_HOME}
|
# hf_home unterstützt Environment-Variablen und ~, z. B. hf_home = ${HF_HOME}
|
||||||
hf_home = /srv/models
|
hf_home = /srv/models
|
||||||
model_path = qwen3/default.gguf
|
model_path = qwen3/default.gguf
|
||||||
|
|
|
||||||
|
|
@ -17,6 +17,9 @@ dependencies = [
|
||||||
[project.optional-dependencies]
|
[project.optional-dependencies]
|
||||||
dev = [
|
dev = [
|
||||||
"pytest>=7.4",
|
"pytest>=7.4",
|
||||||
|
"ruff>=0.6",
|
||||||
|
"mypy>=1.11",
|
||||||
|
"types-requests",
|
||||||
]
|
]
|
||||||
|
|
||||||
[project.scripts]
|
[project.scripts]
|
||||||
|
|
@ -27,3 +30,13 @@ where = ["src"]
|
||||||
|
|
||||||
[tool.setuptools.package-data]
|
[tool.setuptools.package-data]
|
||||||
llamacppctl = ["py.typed"]
|
llamacppctl = ["py.typed"]
|
||||||
|
|
||||||
|
[tool.ruff]
|
||||||
|
target-version = "py310"
|
||||||
|
line-length = 100
|
||||||
|
|
||||||
|
[tool.mypy]
|
||||||
|
python_version = "3.10"
|
||||||
|
files = ["src/llamacppctl"]
|
||||||
|
warn_unused_ignores = true
|
||||||
|
warn_redundant_casts = true
|
||||||
|
|
|
||||||
|
|
@ -2,3 +2,6 @@
|
||||||
# in pyproject.toml).
|
# in pyproject.toml).
|
||||||
-r requirements.txt
|
-r requirements.txt
|
||||||
pytest>=7.4
|
pytest>=7.4
|
||||||
|
ruff>=0.6
|
||||||
|
mypy>=1.11
|
||||||
|
types-requests
|
||||||
|
|
|
||||||
|
|
@ -36,7 +36,10 @@ class ConfigError(ValueError):
|
||||||
def builtin_defaults() -> dict:
|
def builtin_defaults() -> dict:
|
||||||
"""Mirrors the original start-llm-server.sh / status-llm-server.sh defaults."""
|
"""Mirrors the original start-llm-server.sh / status-llm-server.sh defaults."""
|
||||||
return {
|
return {
|
||||||
"image": "ghcr.io/ggml-org/llama.cpp:server-cuda",
|
# Pinned by digest for reproducibility (the :server-cuda tag is a moving
|
||||||
|
# target). To update: docker pull ghcr.io/ggml-org/llama.cpp:server-cuda,
|
||||||
|
# read the new digest, and bump it here (or override `image` in the config).
|
||||||
|
"image": "ghcr.io/ggml-org/llama.cpp@sha256:5535de118ed457f761cbfeacd7e10fef31cb391ca7cac1d5c78b11d28fcf88e6",
|
||||||
"hf_home": "/models",
|
"hf_home": "/models",
|
||||||
"model_path": "qwen3/default.gguf",
|
"model_path": "qwen3/default.gguf",
|
||||||
"container_name": "va_llm",
|
"container_name": "va_llm",
|
||||||
|
|
|
||||||
|
|
@ -11,7 +11,6 @@ import shlex
|
||||||
import subprocess
|
import subprocess
|
||||||
from dataclasses import dataclass
|
from dataclasses import dataclass
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
from typing import Optional
|
|
||||||
|
|
||||||
from .schema import ServerConfig
|
from .schema import ServerConfig
|
||||||
|
|
||||||
|
|
|
||||||
|
|
@ -8,6 +8,7 @@ from __future__ import annotations
|
||||||
|
|
||||||
import fcntl
|
import fcntl
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
|
from typing import TextIO
|
||||||
|
|
||||||
|
|
||||||
class LockError(RuntimeError):
|
class LockError(RuntimeError):
|
||||||
|
|
@ -17,7 +18,7 @@ class LockError(RuntimeError):
|
||||||
class FileLock:
|
class FileLock:
|
||||||
def __init__(self, path: Path):
|
def __init__(self, path: Path):
|
||||||
self.path = Path(path)
|
self.path = Path(path)
|
||||||
self.fd = None
|
self.fd: TextIO | None = None
|
||||||
|
|
||||||
def __enter__(self) -> "FileLock":
|
def __enter__(self) -> "FileLock":
|
||||||
self.path.parent.mkdir(parents=True, exist_ok=True)
|
self.path.parent.mkdir(parents=True, exist_ok=True)
|
||||||
|
|
|
||||||
|
|
@ -202,7 +202,10 @@ def _pin_dns(host: str, allowed_ips: list):
|
||||||
)
|
)
|
||||||
return results
|
return results
|
||||||
|
|
||||||
socket.getaddrinfo = pinned
|
# Intentional monkeypatch: pin DNS to the pre-validated addresses for the
|
||||||
|
# duration of the request (SSRF/rebinding defense). Signature differs from
|
||||||
|
# the stdlib function, hence the targeted ignore.
|
||||||
|
socket.getaddrinfo = pinned # type: ignore[assignment]
|
||||||
try:
|
try:
|
||||||
yield
|
yield
|
||||||
finally:
|
finally:
|
||||||
|
|
|
||||||
Loading…
Add table
Add a link
Reference in a new issue