528b16676e
gates / consistency-and-conventions (push) Successful in 2m16s
Adversarial review of the AI tools (commits 387f23f/0856b25) found 2 BLOCKING + 5 REQUIRED defects; all fixed: - pos-ai-hf --include/--exclude: bash-case glob filtering (array-safe, no jq regex interpolation, composes gguf->filename->include->exclude) - pos-ai-server: ExecStart rebuilt as single-line properly-quoted command (systemd_quote for executable + model path; systemd-analyze verify rc=0) - --branch/--revision aliased (last wins), dead BRANCH variable removed - parallel download drains all jobs: per-pid wait, honest 'X of Y files, N failed' summary, rc=1 on partial failure, no .hf-meta for half-downloaded models, EXIT-trap temp cleanup - detect_llama_version guarded; validate_requested_flags errors on unsupported explicit flags with version-aware message - pos ai hf cache [status|clear]: real implementation, fail-closed confirm - new bin/pos-ai-llamacpp thin forwarder + llamacpp shorthand in bin/pos-ai (pos ai llamacpp <subcmd> = pos ai --provider llamacpp <subcmd>) - docs synced: bin/pos-ai usage(), DOC/POS.md AI_PROVIDER row, howto/ai.md (adapter list, --provider backends, shorthand, providers table); gen regenerated (tree/dispatch/completions) Verified: bash -n all bin/pos*; make gen idempotent; make check green; make lint 0 FAIL, 0 WARN. Reviewer acceptance: APPROVE_WITH_NOTES (0 REQUIRED). Audit deliverables + agent reports included for context.
660 lines
24 KiB
Bash
Executable File
660 lines
24 KiB
Bash
Executable File
#!/usr/bin/env bash
|
|
set -euo pipefail
|
|
# POS: ai server — llama.cpp local inference server (start, stop, status, models, logs)
|
|
# POS_SUBCMDS: start stop status models logs
|
|
# POS_FLAGS: --port --host --model --ctx --gpu --threads --gpu-layers --gpu-threads --tensor-split --n-gpu-layers --batch-size --ubatch-size --temperature --top-k --top-p --repetition-penalty --mmap --mlock --kv-cache --ctx-size --metrics --health --slots
|
|
# POS_DEPS: curl jq
|
|
|
|
source "$(dirname "$0")/../lib/common.sh" 2>/dev/null || source "$(dirname "$0")/common.sh"
|
|
|
|
# ── Dependencies (before --help) ───────────────────────────────
|
|
command -v curl &>/dev/null || err "curl not found (install curl)"
|
|
command -v jq &>/dev/null || err "jq not found (install jq)"
|
|
|
|
# ── Config / seams ─────────────────────────────────────────────
|
|
CONFIG_FILE="${CONFIG_FILE:-$HOME/.config/linux_post_install/ai.env}"
|
|
USER_SYSTEMD_DIR="${USER_SYSTEMD_DIR:-${XDG_CONFIG_HOME:-$HOME/.config}/systemd/user}"
|
|
SERVICE="pos-ai-server.service"
|
|
HF_DOWNLOAD_DIR="${HF_DOWNLOAD_DIR:-$HOME/.local/share/linux_post_install/ai/models}"
|
|
|
|
# ── Config loader (env-var precedence, same pattern as pos-ai-hf) ──
|
|
load_config() {
|
|
[ -f "$CONFIG_FILE" ] || return 0
|
|
local k v
|
|
while IFS='=' read -r k v; do
|
|
[ -n "$k" ] || continue
|
|
case "$k" in
|
|
\#*) continue ;;
|
|
esac
|
|
v="${v%\"}"; v="${v#\"}"; v="${v%\'}"; v="${v#\'}"
|
|
v="${v//$'\r'/}"
|
|
if [ -z "${!k:-}" ]; then
|
|
export "$k"="$v"
|
|
fi
|
|
done < <(grep -E '^[A-Z_]+=' "$CONFIG_FILE" || true)
|
|
}
|
|
|
|
load_config
|
|
|
|
# ── Binary detection ───────────────────────────────────────────
|
|
find_llamacpp() {
|
|
local candidates=("llama-server" "llama.cpp/server" "server" "llama-server-cuda")
|
|
local bin
|
|
for bin in "${candidates[@]}"; do
|
|
command -v "$bin" &>/dev/null && { echo "$bin"; return 0; }
|
|
done
|
|
return 1
|
|
}
|
|
|
|
# ── Version detection ──────────────────────────────────────────
|
|
# detect_llama_version <binary> → X.Y.Z or "unknown". Guarded: a missing
|
|
# binary or unreadable --version output yields "unknown", never an errexit.
|
|
detect_llama_version() {
|
|
local bin="${1:-llama-server}"
|
|
command -v "$bin" &>/dev/null || { echo "unknown"; return 0; }
|
|
local version
|
|
version="$("$bin" --version 2>/dev/null | grep -oE '[0-9]+\.[0-9]+\.[0-9]+' | head -1 || true)"
|
|
[ -n "$version" ] || version="unknown"
|
|
echo "$version"
|
|
}
|
|
|
|
# ── Validate explicitly requested flags ────────────────────────
|
|
# validate_requested_flags <binary> <version> <flag...> — for every flag the
|
|
# user explicitly requested, check its token appears in the binary's --help
|
|
# output and err (version-aware) on the first unsupported one. If --help
|
|
# cannot be read, warn once and proceed instead of hard-failing.
|
|
validate_requested_flags() {
|
|
local bin="$1" version="$2"
|
|
shift 2
|
|
[ $# -gt 0 ] || return 0
|
|
|
|
local help_text
|
|
help_text="$("$bin" --help 2>/dev/null)" || {
|
|
warn "Cannot obtain llama-server --help output — skipping flag validation"
|
|
return 0
|
|
}
|
|
|
|
local seen=() flag
|
|
for flag in "$@"; do
|
|
case " ${seen[*]:-} " in
|
|
*" $flag "*) continue ;; # dedupe alias-mapped flags (e.g. --gpu → --n-gpu-layers)
|
|
esac
|
|
seen+=("$flag")
|
|
if ! printf '%s' "$help_text" | grep -qF -- "$flag"; then
|
|
err "installed llama.cpp ${version} does not expose ${flag} — remove it or upgrade llama.cpp"
|
|
fi
|
|
done
|
|
}
|
|
|
|
# ── GPU detection ──────────────────────────────────────────────
|
|
detect_gpu() {
|
|
if command -v nvidia-smi &>/dev/null && nvidia-smi &>/dev/null 2>&1; then
|
|
echo "cuda"
|
|
else
|
|
echo "cpu"
|
|
fi
|
|
}
|
|
|
|
resolve_gpu_layers() {
|
|
local configured="${LLAMACPP_GPU_LAYERS:-}"
|
|
if [ -n "$configured" ] && [ "$configured" != "-1" ]; then
|
|
echo "$configured"
|
|
return
|
|
fi
|
|
# Auto-detect
|
|
local gpu
|
|
gpu="$(detect_gpu)"
|
|
case "$gpu" in
|
|
cuda) echo "-1" ;;
|
|
*) echo "0" ;;
|
|
esac
|
|
}
|
|
|
|
# ── Human-readable size ────────────────────────────────────────
|
|
human_size() {
|
|
local bytes="$1"
|
|
if [ "$bytes" -ge 1073741824 ]; then
|
|
awk "BEGIN { printf \"%.1f GB\", $bytes / 1073741824 }"
|
|
elif [ "$bytes" -ge 1048576 ]; then
|
|
awk "BEGIN { printf \"%.1f MB\", $bytes / 1048576 }"
|
|
elif [ "$bytes" -ge 1024 ]; then
|
|
awk "BEGIN { printf \"%.1f KB\", $bytes / 1024 }"
|
|
else
|
|
printf '%d B' "$bytes"
|
|
fi
|
|
}
|
|
|
|
# ── Health check ───────────────────────────────────────────────
|
|
check_health() {
|
|
local port="${LLAMACPP_PORT:-8088}"
|
|
local resp
|
|
resp="$(curl -sf "http://127.0.0.1:$port/health" 2>/dev/null)" || { echo "not running"; return 1; }
|
|
local status
|
|
status="$(printf '%s' "$resp" | jq -r '.status // "unknown"' 2>/dev/null)"
|
|
echo "$status"
|
|
}
|
|
|
|
# ── Interactive model picker (reads /dev/tty, not stdin) ───────
|
|
pick_model() {
|
|
local models=() i
|
|
while IFS= read -r f; do
|
|
[ -f "$f" ] || continue
|
|
models+=("$f")
|
|
done < <(find "$HF_DOWNLOAD_DIR" -name '*.gguf' -type f 2>/dev/null | sort)
|
|
|
|
[ ${#models[@]} -gt 0 ] || err "No GGUF models found — run 'pos ai hf download <repo> --gguf'"
|
|
|
|
echo "Available models:"
|
|
for ((i = 0; i < ${#models[@]}; i++)); do
|
|
local name size
|
|
name="$(basename "${models[$i]}")"
|
|
size="$(stat -c%s "${models[$i]}" 2>/dev/null || echo 0)"
|
|
printf ' %2d) %-50s %s\n' "$((i + 1))" "$name" "$(human_size "$size")"
|
|
done
|
|
echo
|
|
local choice
|
|
printf 'Pick a model [1-%d]: ' "${#models[@]}"
|
|
IFS= read -r choice </dev/tty || choice=""
|
|
[[ "$choice" =~ ^[0-9]+$ ]] && [ "$choice" -ge 1 ] && [ "$choice" -le "${#models[@]}" ] || err "Invalid selection"
|
|
printf '%s' "${models[$((choice - 1))]}"
|
|
}
|
|
|
|
# ── Model resolution ───────────────────────────────────────────
|
|
resolve_model() {
|
|
local explicit="${1:-}"
|
|
# 1. Explicit argument
|
|
if [ -n "$explicit" ]; then
|
|
# Absolute path
|
|
if [[ "$explicit" == /* ]]; then
|
|
[ -f "$explicit" ] || err "Model not found: $explicit"
|
|
printf '%s' "$explicit"
|
|
return
|
|
fi
|
|
# Relative to HF_DOWNLOAD_DIR
|
|
local candidate="$HF_DOWNLOAD_DIR/$explicit"
|
|
if [ -f "$candidate" ]; then
|
|
printf '%s' "$candidate"
|
|
return
|
|
fi
|
|
# Also try with the name as-is (could be a relative path)
|
|
[ -f "$explicit" ] && { printf '%s' "$explicit"; return; }
|
|
err "Model not found: $explicit (also searched $HF_DOWNLOAD_DIR)"
|
|
fi
|
|
# 2. Config
|
|
if [ -n "${LLAMACPP_MODEL:-}" ]; then
|
|
[ -f "$LLAMACPP_MODEL" ] || err "Configured model not found: $LLAMACPP_MODEL"
|
|
printf '%s' "$LLAMACPP_MODEL"
|
|
return
|
|
fi
|
|
# 3. Interactive pick (only on TTY)
|
|
if [ -t 0 ] || [ -w /dev/tty ]; then
|
|
local picked
|
|
picked="$(pick_model)"
|
|
printf '%s' "$picked"
|
|
return
|
|
fi
|
|
err "No model specified and no LLAMACPP_MODEL configured — run 'pos ai server start <model>' or set LLAMACPP_MODEL in ai.env"
|
|
}
|
|
|
|
# ── Usage ──────────────────────────────────────────────────────
|
|
usage() {
|
|
cat <<'EOF'
|
|
Usage: pos ai server <command> [args]
|
|
|
|
Manage a local llama.cpp inference server via systemd user service.
|
|
|
|
Commands:
|
|
start [model] Start the server (model: argument, config, or interactive pick)
|
|
stop Stop and disable the server
|
|
status Show service state, config, and health
|
|
models List available GGUF files
|
|
logs [lines] Show recent server logs
|
|
|
|
Options:
|
|
--port <port> Server port (default: 8088)
|
|
--host <addr> Bind address (default: 127.0.0.1)
|
|
--model <path> Model path (overrides argument and config)
|
|
--ctx <size> Context window size (default: 4096)
|
|
--gpu <layers> GPU layers: -1=auto, 0=CPU, N=explicit (default: -1)
|
|
--threads <n> CPU threads (default: nproc)
|
|
--gpu-layers <n> GPU layers (overrides --gpu)
|
|
--gpu-threads <n> GPU threads (default: auto)
|
|
--tensor-split <n> Tensor split configuration
|
|
--n-gpu-layers <n> GPU layers (alternative to --gpu)
|
|
--batch-size <n> Batch size for processing
|
|
--ubatch-size <n> UBatch size for processing
|
|
--temperature <n> Sampling temperature (default: 0.8)
|
|
--top-k <n> Top-K sampling parameter
|
|
--top-p <n> Top-P sampling parameter
|
|
--repetition-penalty <n> Repetition penalty for sampling
|
|
--mmap Use memory mapping
|
|
--mlock Lock memory
|
|
--kv-cache <size> KV cache size
|
|
--ctx-size <n> Context window size (alternative to --ctx)
|
|
--metrics Enable metrics endpoint
|
|
--health Enable health endpoint
|
|
--slots <n> Concurrent request slots
|
|
|
|
Examples:
|
|
pos ai server start mistral-7b-v0.1.Q4_K_M.gguf
|
|
pos ai server start /path/to/model.gguf --port 9090 --gpu 0
|
|
pos ai server status
|
|
pos ai server logs 50
|
|
pos ai server models
|
|
pos ai server stop
|
|
pos ai server start --model model.gguf --gpu-layers 35 --ctx-size 4096 --temperature 0.7
|
|
pos ai server start --model model.gguf --mmap --mlock --batch-size 512
|
|
|
|
Config (~/.config/linux_post_install/ai.env):
|
|
LLAMACPP_PORT Server port (default 8088)
|
|
LLAMACPP_HOST Bind address (default 127.0.0.1)
|
|
LLAMACPP_MODEL Default model path (GGUF file)
|
|
LLAMACPP_CTX_SIZE Context window size (default 4096)
|
|
LLAMACPP_GPU_LAYERS GPU layers: -1=auto, 0=CPU only (default -1)
|
|
LLAMACPP_THREADS CPU threads (default: nproc)
|
|
|
|
Requires: llama-server binary (install llama.cpp: https://github.com/ggerganov/llama.cpp)
|
|
EOF
|
|
exit 0
|
|
}
|
|
|
|
# ── Parse flags ────────────────────────────────────────────────
|
|
PORT="${LLAMACPP_PORT:-8088}"
|
|
HOST="${LLAMACPP_HOST:-127.0.0.1}"
|
|
CTX_SIZE="${LLAMACPP_CTX_SIZE:-4096}"
|
|
GPU_LAYERS="${LLAMACPP_GPU_LAYERS:--1}"
|
|
THREADS="${LLAMACPP_THREADS:-}"
|
|
MODEL_ARG=""
|
|
SUBCMD=""
|
|
SUBCMD_ARGS=()
|
|
|
|
# New GPU and performance options
|
|
GPU_LAYERS_FLAG=""
|
|
GPU_THREADS=""
|
|
TENSOR_SPLIT=""
|
|
BATCH_SIZE=""
|
|
UBATCH_SIZE=""
|
|
TEMPERATURE=""
|
|
TOP_K=""
|
|
TOP_P=""
|
|
REPETITION_PENALTY=""
|
|
MAPPING=""
|
|
LOCKING=""
|
|
KV_CACHE_SIZE=""
|
|
METRICS=""
|
|
HEALTH=""
|
|
SLOTS=""
|
|
|
|
# Canonical flag tokens the user explicitly requested (defaults excluded) —
|
|
# validated against the installed binary's --help in cmd_start.
|
|
REQUESTED_FLAGS=()
|
|
|
|
while [ $# -gt 0 ]; do
|
|
case "$1" in
|
|
-h|--help) usage ;;
|
|
--port)
|
|
[ $# -ge 2 ] || err "--port requires a value"
|
|
PORT="$2"; REQUESTED_FLAGS+=("--port"); shift 2 ;;
|
|
--host)
|
|
[ $# -ge 2 ] || err "--host requires a value"
|
|
HOST="$2"; REQUESTED_FLAGS+=("--host"); shift 2 ;;
|
|
--model)
|
|
[ $# -ge 2 ] || err "--model requires a value"
|
|
MODEL_ARG="$2"; REQUESTED_FLAGS+=("--model"); shift 2 ;;
|
|
--ctx)
|
|
[ $# -ge 2 ] || err "--ctx requires a value"
|
|
CTX_SIZE="$2"; REQUESTED_FLAGS+=("--ctx-size"); shift 2 ;;
|
|
--gpu)
|
|
[ $# -ge 2 ] || err "--gpu requires a value"
|
|
GPU_LAYERS="$2"; REQUESTED_FLAGS+=("--n-gpu-layers"); shift 2 ;;
|
|
--threads)
|
|
[ $# -ge 2 ] || err "--threads requires a value"
|
|
THREADS="$2"; REQUESTED_FLAGS+=("--threads"); shift 2 ;;
|
|
--gpu-layers)
|
|
[ $# -ge 2 ] || err "--gpu-layers requires a value"
|
|
GPU_LAYERS_FLAG="$2"; REQUESTED_FLAGS+=("--n-gpu-layers"); shift 2 ;;
|
|
--gpu-threads)
|
|
[ $# -ge 2 ] || err "--gpu-threads requires a value"
|
|
GPU_THREADS="$2"; REQUESTED_FLAGS+=("--gpu-threads"); shift 2 ;;
|
|
--tensor-split)
|
|
[ $# -ge 2 ] || err "--tensor-split requires a value"
|
|
TENSOR_SPLIT="$2"; REQUESTED_FLAGS+=("--tensor-split"); shift 2 ;;
|
|
--n-gpu-layers)
|
|
[ $# -ge 2 ] || err "--n-gpu-layers requires a value"
|
|
GPU_LAYERS_FLAG="$2"; REQUESTED_FLAGS+=("--n-gpu-layers"); shift 2 ;;
|
|
--batch-size)
|
|
[ $# -ge 2 ] || err "--batch-size requires a value"
|
|
BATCH_SIZE="$2"; REQUESTED_FLAGS+=("--batch-size"); shift 2 ;;
|
|
--ubatch-size)
|
|
[ $# -ge 2 ] || err "--ubatch-size requires a value"
|
|
UBATCH_SIZE="$2"; REQUESTED_FLAGS+=("--ubatch-size"); shift 2 ;;
|
|
--temperature)
|
|
[ $# -ge 2 ] || err "--temperature requires a value"
|
|
TEMPERATURE="$2"; REQUESTED_FLAGS+=("--temperature"); shift 2 ;;
|
|
--top-k)
|
|
[ $# -ge 2 ] || err "--top-k requires a value"
|
|
TOP_K="$2"; REQUESTED_FLAGS+=("--top-k"); shift 2 ;;
|
|
--top-p)
|
|
[ $# -ge 2 ] || err "--top-p requires a value"
|
|
TOP_P="$2"; REQUESTED_FLAGS+=("--top-p"); shift 2 ;;
|
|
--repetition-penalty)
|
|
[ $# -ge 2 ] || err "--repetition-penalty requires a value"
|
|
REPETITION_PENALTY="$2"; REQUESTED_FLAGS+=("--repetition-penalty"); shift 2 ;;
|
|
--mmap)
|
|
MAPPING="true"; REQUESTED_FLAGS+=("--mmap"); shift ;;
|
|
--mlock)
|
|
LOCKING="true"; REQUESTED_FLAGS+=("--mlock"); shift ;;
|
|
--kv-cache)
|
|
[ $# -ge 2 ] || err "--kv-cache requires a value"
|
|
KV_CACHE_SIZE="$2"; REQUESTED_FLAGS+=("--kv-cache"); shift 2 ;;
|
|
--ctx-size)
|
|
[ $# -ge 2 ] || err "--ctx-size requires a value"
|
|
CTX_SIZE="$2"; REQUESTED_FLAGS+=("--ctx-size"); shift 2 ;;
|
|
--metrics)
|
|
METRICS="true"; REQUESTED_FLAGS+=("--metrics"); shift ;;
|
|
--health)
|
|
HEALTH="true"; REQUESTED_FLAGS+=("--health"); shift ;;
|
|
--slots)
|
|
[ $# -ge 2 ] || err "--slots requires a value"
|
|
SLOTS="$2"; REQUESTED_FLAGS+=("--slots"); shift 2 ;;
|
|
-*)
|
|
err "Unknown option '$1' (see --help)" ;;
|
|
*)
|
|
if [ -z "$SUBCMD" ]; then
|
|
SUBCMD="$1"
|
|
else
|
|
SUBCMD_ARGS+=("$1")
|
|
fi
|
|
shift ;;
|
|
esac
|
|
done
|
|
|
|
# Apply flag overrides back to config defaults (flags > env > file default)
|
|
LLAMACPP_PORT="$PORT"
|
|
LLAMACPP_HOST="$HOST"
|
|
LLAMACPP_CTX_SIZE="$CTX_SIZE"
|
|
LLAMACPP_GPU_LAYERS="$GPU_LAYERS"
|
|
if [ -z "$THREADS" ]; then
|
|
THREADS="$(nproc 2>/dev/null || echo 4)"
|
|
fi
|
|
LLAMACPP_THREADS="$THREADS"
|
|
|
|
# ── Subcommands ────────────────────────────────────────────────
|
|
|
|
# systemd_quote <value> — wrap a path in double quotes for systemd's
|
|
# ExecStart word-splitting (systemd.service(5)), escaping embedded `"` as
|
|
# `\"`. Only tokens that may legally contain spaces need this (binary and
|
|
# model path); plain numeric/flag tokens like `--port 8088` stay unquoted.
|
|
systemd_quote() {
|
|
local value="$1"
|
|
value="${value//\"/\\\"}"
|
|
printf '"%s"' "$value"
|
|
}
|
|
|
|
cmd_start() {
|
|
# Resolve the llama-server binary
|
|
local llamacpp_bin
|
|
llamacpp_bin="$(find_llamacpp)" || err "llama-server not found — install llama.cpp (https://github.com/ggerganov/llama.cpp)"
|
|
local llamacpp_full
|
|
llamacpp_full="$(command -v "$llamacpp_bin")"
|
|
|
|
# Detect version (guarded — never crashes; returns "unknown" when
|
|
# unreadable, then basic defaults are used)
|
|
local version
|
|
version="$(detect_llama_version "$llamacpp_bin")"
|
|
|
|
# Validate explicitly requested flags against this binary's --help
|
|
if [ "${#REQUESTED_FLAGS[@]}" -gt 0 ]; then
|
|
validate_requested_flags "$llamacpp_bin" "$version" "${REQUESTED_FLAGS[@]}"
|
|
fi
|
|
|
|
# Resolve model
|
|
local explicit_model="${SUBCMD_ARGS[0]:-}"
|
|
# Flag --model takes precedence over positional arg
|
|
[ -n "$MODEL_ARG" ] && explicit_model="$MODEL_ARG"
|
|
local model
|
|
model="$(resolve_model "$explicit_model")"
|
|
|
|
# Resolve GPU layers
|
|
local gpu_layers
|
|
gpu_layers="$(resolve_gpu_layers)"
|
|
# Use the flag value if provided, otherwise use resolved value
|
|
[ -n "$GPU_LAYERS_FLAG" ] && gpu_layers="$GPU_LAYERS_FLAG"
|
|
|
|
# Warn if no GPU detected and auto-detect resolved to CPU
|
|
if [ "$gpu_layers" = "0" ] && [ "${LLAMACPP_GPU_LAYERS:--1}" = "-1" ]; then
|
|
warn "No NVIDIA GPU detected — running in CPU mode"
|
|
fi
|
|
|
|
# Check port availability (best-effort)
|
|
if command -v ss &>/dev/null; then
|
|
if ss -tlnp 2>/dev/null | grep -q ":${PORT} "; then
|
|
# Port might be our own old instance — only warn
|
|
warn "Port $PORT may already be in use — check with 'ss -tlnp'"
|
|
fi
|
|
fi
|
|
|
|
# Build ONE command line: binary + model + ALL resolved flags. A single
|
|
# string keeps the systemd unit's ExecStart on one line (systemd requires
|
|
# trailing `\` for multi-line continuations) and makes dry-run show
|
|
# exactly what the unit will contain. systemd splits ExecStart on
|
|
# unquoted whitespace, so the binary and the model path — the only tokens
|
|
# that may contain spaces — are systemd_quote()d; plain flag/number
|
|
# tokens stay unquoted.
|
|
local exec_cmd
|
|
exec_cmd="$(systemd_quote "$llamacpp_full") -m $(systemd_quote "$model") --port $PORT --host $HOST"
|
|
exec_cmd+=" --n-gpu-layers $gpu_layers"
|
|
exec_cmd+=" --ctx-size $CTX_SIZE"
|
|
exec_cmd+=" --threads $THREADS"
|
|
if [ -n "$GPU_THREADS" ]; then
|
|
exec_cmd+=" --gpu-threads $GPU_THREADS"
|
|
fi
|
|
if [ -n "$TENSOR_SPLIT" ]; then
|
|
exec_cmd+=" --tensor-split $TENSOR_SPLIT"
|
|
fi
|
|
if [ -n "$BATCH_SIZE" ]; then
|
|
exec_cmd+=" --batch-size $BATCH_SIZE"
|
|
fi
|
|
if [ -n "$UBATCH_SIZE" ]; then
|
|
exec_cmd+=" --ubatch-size $UBATCH_SIZE"
|
|
fi
|
|
if [ -n "$TEMPERATURE" ]; then
|
|
exec_cmd+=" --temperature $TEMPERATURE"
|
|
fi
|
|
if [ -n "$TOP_K" ]; then
|
|
exec_cmd+=" --top-k $TOP_K"
|
|
fi
|
|
if [ -n "$TOP_P" ]; then
|
|
exec_cmd+=" --top-p $TOP_P"
|
|
fi
|
|
if [ -n "$REPETITION_PENALTY" ]; then
|
|
exec_cmd+=" --repetition-penalty $REPETITION_PENALTY"
|
|
fi
|
|
if [ -n "$MAPPING" ]; then
|
|
exec_cmd+=" --mmap"
|
|
fi
|
|
if [ -n "$LOCKING" ]; then
|
|
exec_cmd+=" --mlock"
|
|
fi
|
|
if [ -n "$KV_CACHE_SIZE" ]; then
|
|
exec_cmd+=" --kv-cache $KV_CACHE_SIZE"
|
|
fi
|
|
if [ -n "$METRICS" ]; then
|
|
exec_cmd+=" --metrics"
|
|
fi
|
|
if [ -n "$HEALTH" ]; then
|
|
exec_cmd+=" --health"
|
|
fi
|
|
if [ -n "$SLOTS" ]; then
|
|
exec_cmd+=" --slots $SLOTS"
|
|
fi
|
|
|
|
if [ "${DRY_RUN:-0}" -eq 1 ]; then
|
|
log "(dry-run) generate systemd unit $USER_SYSTEMD_DIR/$SERVICE"
|
|
log "(dry-run) ExecStart: $exec_cmd"
|
|
log "(dry-run) systemctl --user daemon-reload && enable --now $SERVICE"
|
|
return 0
|
|
fi
|
|
|
|
# Generate systemd unit — ExecStart is a single line with the full command
|
|
mkdir -p "$USER_SYSTEMD_DIR"
|
|
cat > "$USER_SYSTEMD_DIR/$SERVICE" <<EOF
|
|
[Unit]
|
|
Description=pos llama.cpp inference server (linux-post-install)
|
|
After=network-online.target
|
|
|
|
[Service]
|
|
Type=simple
|
|
ExecStart=$exec_cmd
|
|
Restart=on-failure
|
|
RestartSec=5
|
|
TimeoutStopSec=10
|
|
KillMode=control-group
|
|
EnvironmentFile=-%h/.config/linux_post_install/ai.env
|
|
|
|
[Install]
|
|
WantedBy=default.target
|
|
EOF
|
|
chmod 644 "$USER_SYSTEMD_DIR/$SERVICE"
|
|
|
|
# Enable and start
|
|
systemctl --user daemon-reload
|
|
systemctl --user enable --now "$SERVICE"
|
|
|
|
log "Server starting — model: $(basename "$model"), port: $PORT"
|
|
|
|
# Linger warning
|
|
if command -v loginctl >/dev/null 2>&1; then
|
|
if ! loginctl show-user "$(id -un)" 2>/dev/null | grep -q '^Linger=yes'; then
|
|
warn "enable linger so the server survives logout: sudo loginctl enable-linger $(id -un)"
|
|
fi
|
|
fi
|
|
|
|
# Health check (wait briefly)
|
|
sleep 2
|
|
local health
|
|
health="$(check_health)" || true
|
|
if [ "$health" != "not running" ]; then
|
|
ok "Server healthy (status: $health)"
|
|
else
|
|
warn "Server may not be ready yet — check with 'pos ai server status'"
|
|
fi
|
|
}
|
|
|
|
cmd_stop() {
|
|
if [ ! -f "$USER_SYSTEMD_DIR/$SERVICE" ]; then
|
|
warn "No llama.cpp server service installed ($SERVICE)"
|
|
return 0
|
|
fi
|
|
if [ "${DRY_RUN:-0}" -eq 1 ]; then
|
|
log "(dry-run) systemctl --user disable --now $SERVICE; remove unit"
|
|
else
|
|
systemctl --user disable --now "$SERVICE" 2>/dev/null || true
|
|
rm -f "$USER_SYSTEMD_DIR/$SERVICE"
|
|
systemctl --user daemon-reload
|
|
fi
|
|
log "llama.cpp server stopped and removed"
|
|
}
|
|
|
|
cmd_status() {
|
|
# llama-server must be present for the version probe below — same
|
|
# actionable deps message as `start`
|
|
if ! find_llamacpp >/dev/null 2>&1; then
|
|
err "llama-server not found — install llama.cpp (https://github.com/ggerganov/llama.cpp)"
|
|
fi
|
|
|
|
# Service state
|
|
local svc_state="stopped"
|
|
if systemctl --user is-active "$SERVICE" &>/dev/null; then
|
|
svc_state="running"
|
|
fi
|
|
printf 'service: %s\n' "$svc_state"
|
|
|
|
# Model (from health endpoint if running)
|
|
if [ "$svc_state" = "running" ]; then
|
|
local models_resp
|
|
models_resp="$(curl -sf "http://127.0.0.1:$PORT/v1/models" 2>/dev/null)" || true
|
|
local model_id
|
|
model_id="$(printf '%s' "$models_resp" | jq -r '.data[0].id // "unknown"' 2>/dev/null)" || model_id="unknown"
|
|
printf 'model: %s\n' "$model_id"
|
|
else
|
|
printf 'model: (not loaded)\n'
|
|
fi
|
|
|
|
# Config
|
|
printf 'port: %s\n' "$PORT"
|
|
printf 'host: %s\n' "$HOST"
|
|
|
|
# GPU
|
|
local gpu_type
|
|
gpu_type="$(detect_gpu)"
|
|
printf 'gpu: %s (%s layers)\n' "${gpu_type^^}" "$GPU_LAYERS"
|
|
|
|
printf 'context: %s\n' "$CTX_SIZE"
|
|
printf 'threads: %s\n' "$THREADS"
|
|
|
|
# Autostart
|
|
if systemctl --user is-enabled "$SERVICE" &>/dev/null; then
|
|
printf 'autostart: enabled\n'
|
|
else
|
|
printf 'autostart: disabled\n'
|
|
fi
|
|
|
|
# Endpoint
|
|
printf 'endpoint: http://%s:%s\n' "$HOST" "$PORT"
|
|
|
|
# Health
|
|
if [ "$svc_state" = "running" ]; then
|
|
local health
|
|
health="$(check_health)" || health="not responding"
|
|
printf 'health: %s\n' "$health"
|
|
else
|
|
printf 'health: not running\n'
|
|
fi
|
|
|
|
# Version info (probe the resolved binary; "unknown" if unreadable)
|
|
local llamacpp_bin version
|
|
llamacpp_bin="$(find_llamacpp)"
|
|
version="$(detect_llama_version "$llamacpp_bin")"
|
|
printf 'version: %s\n' "$version"
|
|
}
|
|
|
|
cmd_models() {
|
|
local dir="${HF_DOWNLOAD_DIR}"
|
|
[ -d "$dir" ] || { warn "No models directory — run 'pos ai hf download' first"; return 0; }
|
|
|
|
local found=0
|
|
echo "Available GGUF models:"
|
|
while IFS= read -r gguf; do
|
|
[ -f "$gguf" ] || continue
|
|
found=1
|
|
local name size
|
|
name="$(basename "$gguf")"
|
|
local dir_name
|
|
dir_name="$(basename "$(dirname "$gguf")")"
|
|
size="$(stat -c%s "$gguf" 2>/dev/null || echo 0)"
|
|
local hsize
|
|
hsize="$(human_size "$size")"
|
|
printf ' %-50s %s\n' "$dir_name/$name" "$hsize"
|
|
done < <(find "$dir" -name '*.gguf' -type f 2>/dev/null | sort)
|
|
|
|
[ "$found" -eq 0 ] && warn "No .gguf files found — download with 'pos ai hf download <repo> --gguf'"
|
|
}
|
|
|
|
cmd_logs() {
|
|
local lines="${SUBCMD_ARGS[0]:-50}"
|
|
[[ "$lines" =~ ^[0-9]+$ ]] || err "lines must be a number"
|
|
journalctl --user -u "$SERVICE" -n "$lines" --no-pager 2>/dev/null || warn "No logs found — server may not have been started"
|
|
}
|
|
|
|
# ── Dispatch ───────────────────────────────────────────────────
|
|
case "${SUBCMD:-}" in
|
|
"") usage ;;
|
|
start) cmd_start ;;
|
|
stop) cmd_stop ;;
|
|
status) cmd_status ;;
|
|
models) cmd_models ;;
|
|
logs) cmd_logs ;;
|
|
*) err "Unknown subcommand '$SUBCMD' (see --help)" ;;
|
|
esac
|