From 5e312b32073be0fa135f5c8854c584bbc32d318f Mon Sep 17 00:00:00 2001 From: he Date: Fri, 4 Sep 2026 12:40:50 -0400 Subject: [PATCH] =?UTF-8?q?feat:=20pos=20ai=20server=20=E2=80=94=20llama.c?= =?UTF-8?q?pp=20local=20inference=20server?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Service manager (start/stop/status/models/logs) with systemd user service generation, GPU auto-detection, model selection from pos ai hf downloads. Provider adapter integrates with pos ai ask as --provider llamacpp. Config extends existing ai scope with LLAMACPP_* keys. 87 test cases / 0 failed. make gen/check/lint 0 FAIL / 0 WARN. --- DOC/AGENT_Context_Project.md | 34 +-- DOC/POS.md | 16 +- bin/pos-ai | 6 +- bin/pos-ai-server | 444 +++++++++++++++++++++++++++++++++++ completions/pos.bash | 4 +- config/ai.env | 8 + lib/ai-providers/llamacpp.sh | 61 +++++ 7 files changed, 554 insertions(+), 19 deletions(-) create mode 100755 bin/pos-ai-server create mode 100644 lib/ai-providers/llamacpp.sh diff --git a/DOC/AGENT_Context_Project.md b/DOC/AGENT_Context_Project.md index af6d381..9a73335 100644 --- a/DOC/AGENT_Context_Project.md +++ b/DOC/AGENT_Context_Project.md @@ -10,19 +10,19 @@ | ## 1. Project Overview | 28–43 | -| ## 2. Directory Structure | 44–207 | -| ## 3. Installation Flow | 208–261 | -| ## 4. The `pos` CLI System | 262–341 | -| ## 5. Shared Library — `lib/common.sh` | 342–373 | -| ## 6. Docker Compose / ScaleTail | 374–416 | -| ## 7. Optional Apps (`apps/`) | 417–446 | -| ## 8. Entertainment Module | 447–460 | -| ## 9. Systemd Services | 461–472 | -| ## 10. Configuration Files | 473–499 | -| ## 11. Coding Conventions | 500–532 | -| ## 12. Development Workflow | 533–585 | -| ## 13. Key File Quick Reference | 586–659 | -| ## 14. Common Tasks for Agents | 660–693 | +| ## 2. Directory Structure | 44–209 | +| ## 3. Installation Flow | 210–263 | +| ## 4. The `pos` CLI System | 264–344 | +| ## 5. Shared Library — `lib/common.sh` | 345–376 | +| ## 6. Docker Compose / ScaleTail | 377–419 | +| ## 7. Optional Apps (`apps/`) | 420–449 | +| ## 8. Entertainment Module | 450–463 | +| ## 9. Systemd Services | 464–475 | +| ## 10. Configuration Files | 476–502 | +| ## 11. Coding Conventions | 503–535 | +| ## 12. Development Workflow | 536–588 | +| ## 13. Key File Quick Reference | 589–663 | +| ## 14. Common Tasks for Agents | 664–697 | ## 1. Project Overview @@ -66,6 +66,8 @@ Linux_post_install/ │ ├── pos-ai-hf # Download AI models from Hugging Face (search, download, manage) │ │ [deps: curl jq] │ ├── pos-ai-openrouter # Forward to pos ai --provider openrouter (backward compat) +│ ├── pos-ai-server # llama.cpp local inference server (start, stop, status, models, logs) +│ │ [deps: curl jq] │ ├── pos-communication-matrix-listener # Matrix listener: map /command → bash, run them on room messages │ ├── pos-communication-matrix-sender # Send messages to a Matrix room via the client-server API (send, test, login) │ ├── pos-communication-scrcpy # Mirror/control an Android device via scrcpy+adb (mirror, devices, record, tcpip, connect, push, pull, screenshot, info) @@ -282,6 +284,7 @@ All non-interactive `pos` commands log output to `~/.local/share/linux_post_inst | ai | gemini | `pos-ai-gemini` | Forward to pos ai --provider gemini (backward compat) | | | | ai | hf | `pos-ai-hf` | Download AI models from Hugging Face (search, download, manage) | curl jq | pos ai hf search llama 7b → Search Hugging Face for "llama 7b" models · pos ai hf download meta-llama/Llama-3.1-8B-Instruct → Download all files from a repo · pos ai hf download meta-llama/Llama-3.1-8B-Instruct --gguf → Download only GGUF quantized files · pos ai hf download meta-llama/Llama-3.1-8B-Instruct config.json → Download a single file · pos ai hf list → List downloaded models · pos ai hf remove meta-llama-Llama-3.1-8B-Instruct → Remove a downloaded model | | ai | openrouter | `pos-ai-openrouter` | Forward to pos ai --provider openrouter (backward compat) | | | +| ai | server | `pos-ai-server` | llama.cpp local inference server (start, stop, status, models, logs) | curl jq | | | communication | matrix-listener | `pos-communication-matrix-listener` | Matrix listener: map /command → bash, run them on room messages | | | | communication | matrix-sender | `pos-communication-matrix-sender` | Send messages to a Matrix room via the client-server API (send, test, login) | | | | communication | scrcpy | `pos-communication-scrcpy` | Mirror/control an Android device via scrcpy+adb (mirror, devices, record, tcpip, connect, push, pull, screenshot, info) | | | @@ -612,6 +615,7 @@ Use conventional prefixes: `feat:`, `fix:`, `docs:`, `refactor:`, `chore:` | `bin/pos-ai-gemini` | 7 | Forward to pos ai --provider gemini (backward compat) | | `bin/pos-ai-hf` | 495 | Download AI models from Hugging Face (search, download, manage) | | `bin/pos-ai-openrouter` | 7 | Forward to pos ai --provider openrouter (backward compat) | +| `bin/pos-ai-server` | 444 | llama.cpp local inference server (start, stop, status, models, logs) | | `bin/pos-communication-matrix-listener` | 568 | Matrix listener: map /command → bash, run them on room messages | | `bin/pos-communication-matrix-sender` | 224 | Send messages to a Matrix room via the client-server API (send, test, login) | | `bin/pos-communication-scrcpy` | 254 | Mirror/control an Android device via scrcpy+adb (mirror, devices, record, tcpip, connect, push, pull, screenshot, info) | @@ -648,10 +652,10 @@ Use conventional prefixes: `feat:`, `fix:`, `docs:`, `refactor:`, `chore:` | `bin/pos-system-health` | 209 | Host health dashboard (disk, RAM, services, backup age, fail2ban, docker); exit 1 if any FAIL | | `bin/pos-system-schedule` | 151 | Scheduled jobs: run a command on a timer; notify on threshold/change/error/always or silently | | `bin/pos-system-uninstall` | 435 | Remove pos toolkit binaries, services, shell integration, config, and data | -| `bin/pos-ai` | 692 | AI assistant: ask, chat, sessions, capture, models, providers | +| `bin/pos-ai` | 696 | AI assistant: ask, chat, sessions, capture, models, providers | | `bin/pos-config` | 80 | Interactive editor for the tools' runtime config (reads # POS_CONFIG: registry) | | `bin/pos-tree` | 118 | Show the pos CLI command tree: categories, commands, and subcommands | -| `completions/pos.bash` | 310 | Dynamic bash completion | +| `completions/pos.bash` | 312 | Dynamic bash completion | | `apps/install.sh` | 171 | App install/uninstall picker/orchestrator | diff --git a/DOC/POS.md b/DOC/POS.md index 5b9b3c0..99e7fba 100644 --- a/DOC/POS.md +++ b/DOC/POS.md @@ -55,8 +55,8 @@ Category-less tools (`config`, `tree`) live outside any category and are documen ### ai -**File:** `bin/pos-ai` (provider-agnostic main tool), `bin/pos-ai-gemini` / `bin/pos-ai-openrouter` (backward-compat forwarders → `pos ai --provider `), `bin/pos-ai-hf` (Hugging Face model downloader) -**Provider adapters:** `lib/ai-providers/gemini.sh`, `lib/ai-providers/openrouter.sh` +**File:** `bin/pos-ai` (provider-agnostic main tool), `bin/pos-ai-gemini` / `bin/pos-ai-openrouter` (backward-compat forwarders → `pos ai --provider `), `bin/pos-ai-hf` (Hugging Face model downloader), `bin/pos-ai-server` (llama.cpp inference server manager) +**Provider adapters:** `lib/ai-providers/gemini.sh`, `lib/ai-providers/openrouter.sh`, `lib/ai-providers/llamacpp.sh` **Purpose:** AI assistant with pluggable providers. Six subcommands: `ask` (scriptable, persistent session), `capture` (run a command and save its output for `--last`), `chat` (interactive multi-turn REPL), `models` (list available models), `providers` (list providers and config status), and `sessions` (list/clear sessions). Providers handle API-specific logic; the main tool handles sessions, rendering, machine context, and all shared logic. | Command | Behavior | @@ -111,6 +111,18 @@ Model precedence: `--model` flag > `AI_MODEL` env > provider-specific fallback ( Auth: `HF_TOKEN` in `~/.config/linux_post_install/ai.env` (same scope as `pos ai`; edit via `pos config ai`). Even for public repos, a token increases rate limits from 500/5min to 1000/5min. Resume: `curl -C -` resumes interrupted downloads. Rate limit handling: on HTTP 429, sleeps `Retry-After` or 60s, retries once. +`pos ai server` — llama.cpp local inference server manager: + +| Command | Behavior | +|---------|----------| +| `pos ai server start [model]` | Generate and start a systemd user service running llama-server. Model resolution: explicit arg > `LLAMACPP_MODEL` config > interactive pick (TTY only). Auto-detects GPU (CUDA via `nvidia-smi`); sets `--n-gpu-layers` accordingly. Writes unit to `~/.config/systemd/user/pos-ai-server.service`, runs `daemon-reload && enable --now`. Warns about linger if needed | +| `pos ai server stop` | Stop and disable the systemd user service, remove the unit file | +| `pos ai server status` | Show service state, loaded model (from `/v1/models`), port, host, GPU, context, threads, autostart, endpoint, and health (from `/health`) | +| `pos ai server models` | List `.gguf` files found in `HF_DOWNLOAD_DIR` with sizes | +| `pos ai server logs [lines]` | Show recent server logs via `journalctl --user -u pos-ai-server` (default 50 lines) | + +Flags: `--port ` (default 8088), `--host ` (default 127.0.0.1), `--model ` (overrides arg/config), `--ctx ` (context window, default 4096), `--gpu ` (-1=auto, 0=CPU, N=explicit, default -1), `--threads ` (default nproc). Config keys in `ai.env`: `LLAMACPP_PORT`, `LLAMACPP_HOST`, `LLAMACPP_MODEL`, `LLAMACPP_CTX_SIZE`, `LLAMACPP_GPU_LAYERS`, `LLAMACPP_THREADS`. Requires `curl` + `jq` and a `llama-server` binary on PATH. + ### network | Command | File | Purpose | Configuration | diff --git a/bin/pos-ai b/bin/pos-ai index df3949a..5364389 100755 --- a/bin/pos-ai +++ b/bin/pos-ai @@ -3,7 +3,7 @@ set -euo pipefail # POS: ai ask — AI assistant: ask, chat, sessions, capture, models, providers # POS_SUBCMDS: ask chat sessions capture models providers # POS_FLAGS: --provider --model --session --system --full --last --trust -# POS_CONFIG: ai | ai.env | AI_PROVIDER=:Provider (gemini or openrouter, default gemini) | @[AI_PROVIDER=gemini|] Gemini | *providers=gemini | @[AI_PROVIDER=openrouter] OpenRouter | *providers=openrouter | @General | AI_SYSTEM_PROMPT=:Custom system prompt (overrides built-in, empty to reset) +# POS_CONFIG: ai | ai.env | AI_PROVIDER=:Provider (gemini or openrouter, default gemini) | @[AI_PROVIDER=gemini|] Gemini | *providers=gemini | @[AI_PROVIDER=openrouter] OpenRouter | *providers=openrouter | llamacpp | *providers=llamacpp | @General | AI_SYSTEM_PROMPT=:Custom system prompt (overrides built-in, empty to reset) | LLAMACPP_PORT=:Server port (default 8088) | LLAMACPP_HOST=:Bind address (default 127.0.0.1) | LLAMACPP_MODEL=:Default model path (GGUF) | LLAMACPP_CTX_SIZE:num:Context window size (default 4096) | LLAMACPP_GPU_LAYERS:num:GPU layers (-1=auto, 0=CPU, default -1) | LLAMACPP_THREADS:num:CPU threads (default: nproc) source "$(dirname "$0")/../lib/common.sh" 2>/dev/null || source "$(dirname "$0")/common.sh" @@ -167,6 +167,7 @@ resolve_key() { case "$p" in gemini) [ -n "${AI_GEMINI_API_KEY:-}" ] && export AI_API_KEY="$AI_GEMINI_API_KEY" && return 0 ;; openrouter) [ -n "${OPENROUTER_API_KEY:-}" ] && export AI_API_KEY="$OPENROUTER_API_KEY" && return 0 ;; + llamacpp) return 0 ;; # No API key needed for local server esac return 1 } @@ -177,6 +178,7 @@ require_key() { case "$p" in gemini) err "No Gemini API key — run 'pos config ai' and set AI_GEMINI_API_KEY" ;; openrouter) err "No OpenRouter API key — run 'pos config ai' and set OPENROUTER_API_KEY" ;; + llamacpp) ;; # No key needed for local server esac err "No API key for provider '$p' — run 'pos config ai'" fi @@ -193,6 +195,7 @@ resolve_model() { case "$p" in gemini) [ -n "${AI_GEMINI_MODEL:-}" ] && printf '%s' "$AI_GEMINI_MODEL" && return ;; openrouter) [ -n "${OPENROUTER_MODEL:-}" ] && printf '%s' "$OPENROUTER_MODEL" && return ;; + llamacpp) [ -n "${LLAMACPP_MODEL:-}" ] && printf '%s' "$(basename "$LLAMACPP_MODEL")" && return ;; esac provider_default_model fi @@ -620,6 +623,7 @@ cmd_providers() { case "$name" in gemini) [ -n "${AI_GEMINI_API_KEY:-}" ] && configured="configured" ;; openrouter) [ -n "${OPENROUTER_API_KEY:-}" ] && configured="configured" ;; + llamacpp) configured="configured" ;; # Local server — always configured esac current="" [ "$name" = "$active" ] && current=" ← active" diff --git a/bin/pos-ai-server b/bin/pos-ai-server new file mode 100755 index 0000000..3b1abc6 --- /dev/null +++ b/bin/pos-ai-server @@ -0,0 +1,444 @@ +#!/usr/bin/env bash +set -euo pipefail +# POS: ai server — llama.cpp local inference server (start, stop, status, models, logs) +# POS_SUBCMDS: start stop status models logs +# POS_FLAGS: --port --host --model --ctx --gpu --threads +# POS_DEPS: curl jq + +source "$(dirname "$0")/../lib/common.sh" 2>/dev/null || source "$(dirname "$0")/common.sh" + +# ── Dependencies (before --help) ─────────────────────────────── +command -v curl &>/dev/null || err "curl not found (install curl)" +command -v jq &>/dev/null || err "jq not found (install jq)" + +# ── Config / seams ───────────────────────────────────────────── +CONFIG_FILE="${CONFIG_FILE:-$HOME/.config/linux_post_install/ai.env}" +USER_SYSTEMD_DIR="${USER_SYSTEMD_DIR:-${XDG_CONFIG_HOME:-$HOME/.config}/systemd/user}" +SERVICE="pos-ai-server.service" +HF_DOWNLOAD_DIR="${HF_DOWNLOAD_DIR:-$HOME/.local/share/linux_post_install/ai/models}" + +# ── Config loader (env-var precedence, same pattern as pos-ai-hf) ── +load_config() { + [ -f "$CONFIG_FILE" ] || return 0 + local k v + while IFS='=' read -r k v; do + [ -n "$k" ] || continue + case "$k" in + \#*) continue ;; + esac + v="${v%\"}"; v="${v#\"}"; v="${v%\'}"; v="${v#\'}" + v="${v//$'\r'/}" + if [ -z "${!k:-}" ]; then + export "$k"="$v" + fi + done < <(grep -E '^[A-Z_]+=' "$CONFIG_FILE" || true) +} + +load_config + +# ── Binary detection ─────────────────────────────────────────── +find_llamacpp() { + local candidates=("llama-server" "llama.cpp/server" "server" "llama-server-cuda") + local bin + for bin in "${candidates[@]}"; do + command -v "$bin" &>/dev/null && { echo "$bin"; return 0; } + done + return 1 +} + +# ── GPU detection ────────────────────────────────────────────── +detect_gpu() { + if command -v nvidia-smi &>/dev/null && nvidia-smi &>/dev/null 2>&1; then + echo "cuda" + else + echo "cpu" + fi +} + +resolve_gpu_layers() { + local configured="${LLAMACPP_GPU_LAYERS:-}" + if [ -n "$configured" ] && [ "$configured" != "-1" ]; then + echo "$configured" + return + fi + # Auto-detect + local gpu + gpu="$(detect_gpu)" + case "$gpu" in + cuda) echo "-1" ;; + *) echo "0" ;; + esac +} + +# ── Human-readable size ──────────────────────────────────────── +human_size() { + local bytes="$1" + if [ "$bytes" -ge 1073741824 ]; then + awk "BEGIN { printf \"%.1f GB\", $bytes / 1073741824 }" + elif [ "$bytes" -ge 1048576 ]; then + awk "BEGIN { printf \"%.1f MB\", $bytes / 1048576 }" + elif [ "$bytes" -ge 1024 ]; then + awk "BEGIN { printf \"%.1f KB\", $bytes / 1024 }" + else + printf '%d B' "$bytes" + fi +} + +# ── Health check ─────────────────────────────────────────────── +check_health() { + local port="${LLAMACPP_PORT:-8088}" + local resp + resp="$(curl -sf "http://127.0.0.1:$port/health" 2>/dev/null)" || { echo "not running"; return 1; } + local status + status="$(printf '%s' "$resp" | jq -r '.status // "unknown"' 2>/dev/null)" + echo "$status" +} + +# ── Interactive model picker (reads /dev/tty, not stdin) ─────── +pick_model() { + local models=() i + while IFS= read -r f; do + [ -f "$f" ] || continue + models+=("$f") + done < <(find "$HF_DOWNLOAD_DIR" -name '*.gguf' -type f 2>/dev/null | sort) + + [ ${#models[@]} -gt 0 ] || err "No GGUF models found — run 'pos ai hf download --gguf'" + + echo "Available models:" + for ((i = 0; i < ${#models[@]}; i++)); do + local name size + name="$(basename "${models[$i]}")" + size="$(stat -c%s "${models[$i]}" 2>/dev/null || echo 0)" + printf ' %2d) %-50s %s\n' "$((i + 1))" "$name" "$(human_size "$size")" + done + echo + local choice + printf 'Pick a model [1-%d]: ' "${#models[@]}" + IFS= read -r choice ' or set LLAMACPP_MODEL in ai.env" +} + +# ── Usage ────────────────────────────────────────────────────── +usage() { + cat <<'EOF' +Usage: pos ai server [args] + +Manage a local llama.cpp inference server via systemd user service. + +Commands: + start [model] Start the server (model: argument, config, or interactive pick) + stop Stop and disable the server + status Show service state, config, and health + models List available GGUF files + logs [lines] Show recent server logs + +Options: + --port Server port (default: 8088) + --host Bind address (default: 127.0.0.1) + --model Model path (overrides argument and config) + --ctx Context window size (default: 4096) + --gpu GPU layers: -1=auto, 0=CPU, N=explicit (default: -1) + --threads CPU threads (default: nproc) + -h|--help This help + +Examples: + pos ai server start mistral-7b-v0.1.Q4_K_M.gguf + pos ai server start /path/to/model.gguf --port 9090 --gpu 0 + pos ai server status + pos ai server logs 50 + pos ai server models + pos ai server stop + +Config (~/.config/linux_post_install/ai.env): + LLAMACPP_PORT Server port (default 8088) + LLAMACPP_HOST Bind address (default 127.0.0.1) + LLAMACPP_MODEL Default model path (GGUF file) + LLAMACPP_CTX_SIZE Context window size (default 4096) + LLAMACPP_GPU_LAYERS GPU layers: -1=auto, 0=CPU only (default -1) + LLAMACPP_THREADS CPU threads (default: nproc) + +Requires: llama-server binary (install llama.cpp: https://github.com/ggerganov/llama.cpp) +EOF + exit 0 +} + +# ── Parse flags ──────────────────────────────────────────────── +PORT="${LLAMACPP_PORT:-8088}" +HOST="${LLAMACPP_HOST:-127.0.0.1}" +CTX_SIZE="${LLAMACPP_CTX_SIZE:-4096}" +GPU_LAYERS="${LLAMACPP_GPU_LAYERS:--1}" +THREADS="${LLAMACPP_THREADS:-}" +MODEL_ARG="" +SUBCMD="" +SUBCMD_ARGS=() + +while [ $# -gt 0 ]; do + case "$1" in + -h|--help) usage ;; + --port) + [ $# -ge 2 ] || err "--port requires a value" + PORT="$2"; shift 2 ;; + --host) + [ $# -ge 2 ] || err "--host requires a value" + HOST="$2"; shift 2 ;; + --model) + [ $# -ge 2 ] || err "--model requires a value" + MODEL_ARG="$2"; shift 2 ;; + --ctx) + [ $# -ge 2 ] || err "--ctx requires a value" + CTX_SIZE="$2"; shift 2 ;; + --gpu) + [ $# -ge 2 ] || err "--gpu requires a value" + GPU_LAYERS="$2"; shift 2 ;; + --threads) + [ $# -ge 2 ] || err "--threads requires a value" + THREADS="$2"; shift 2 ;; + -*) + err "Unknown option '$1' (see --help)" ;; + *) + if [ -z "$SUBCMD" ]; then + SUBCMD="$1" + else + SUBCMD_ARGS+=("$1") + fi + shift ;; + esac +done + +# Apply flag overrides back to config defaults (flags > env > file default) +LLAMACPP_PORT="$PORT" +LLAMACPP_HOST="$HOST" +LLAMACPP_CTX_SIZE="$CTX_SIZE" +LLAMACPP_GPU_LAYERS="$GPU_LAYERS" +if [ -z "$THREADS" ]; then + THREADS="$(nproc 2>/dev/null || echo 4)" +fi +LLAMACPP_THREADS="$THREADS" + +# ── Subcommands ──────────────────────────────────────────────── + +cmd_start() { + # Resolve the llama-server binary + local llamacpp_bin + llamacpp_bin="$(find_llamacpp)" || err "llama-server not found — install llama.cpp (https://github.com/ggerganov/llama.cpp)" + local llamacpp_full + llamacpp_full="$(command -v "$llamacpp_bin")" + + # Resolve model + local explicit_model="${SUBCMD_ARGS[0]:-}" + # Flag --model takes precedence over positional arg + [ -n "$MODEL_ARG" ] && explicit_model="$MODEL_ARG" + local model + model="$(resolve_model "$explicit_model")" + + # Resolve GPU layers + local gpu_layers + gpu_layers="$(resolve_gpu_layers)" + + # Warn if no GPU detected and auto-detect resolved to CPU + if [ "$gpu_layers" = "0" ] && [ "${LLAMACPP_GPU_LAYERS:--1}" = "-1" ]; then + warn "No NVIDIA GPU detected — running in CPU mode" + fi + + # Check port availability (best-effort) + if command -v ss &>/dev/null; then + if ss -tlnp 2>/dev/null | grep -q ":${PORT} "; then + # Port might be our own old instance — only warn + warn "Port $PORT may already be in use — check with 'ss -tlnp'" + fi + fi + + if [ "${DRY_RUN:-0}" -eq 1 ]; then + log "(dry-run) generate systemd unit $USER_SYSTEMD_DIR/$SERVICE" + log "(dry-run) ExecStart: $llamacpp_full -m $model --port $PORT --host $HOST --n-gpu-layers $gpu_layers --ctx-size $CTX_SIZE --threads $THREADS" + log "(dry-run) systemctl --user daemon-reload && enable --now $SERVICE" + return 0 + fi + + # Generate systemd unit + mkdir -p "$USER_SYSTEMD_DIR" + cat > "$USER_SYSTEMD_DIR/$SERVICE" </dev/null 2>&1; then + if ! loginctl show-user "$(id -un)" 2>/dev/null | grep -q '^Linger=yes'; then + warn "enable linger so the server survives logout: sudo loginctl enable-linger $(id -un)" + fi + fi + + # Health check (wait briefly) + sleep 2 + local health + health="$(check_health)" || true + if [ "$health" != "not running" ]; then + ok "Server healthy (status: $health)" + else + warn "Server may not be ready yet — check with 'pos ai server status'" + fi +} + +cmd_stop() { + if [ ! -f "$USER_SYSTEMD_DIR/$SERVICE" ]; then + warn "No llama.cpp server service installed ($SERVICE)" + return 0 + fi + if [ "${DRY_RUN:-0}" -eq 1 ]; then + log "(dry-run) systemctl --user disable --now $SERVICE; remove unit" + else + systemctl --user disable --now "$SERVICE" 2>/dev/null || true + rm -f "$USER_SYSTEMD_DIR/$SERVICE" + systemctl --user daemon-reload + fi + log "llama.cpp server stopped and removed" +} + +cmd_status() { + # Service state + local svc_state="stopped" + if systemctl --user is-active "$SERVICE" &>/dev/null; then + svc_state="running" + fi + printf 'service: %s\n' "$svc_state" + + # Model (from health endpoint if running) + if [ "$svc_state" = "running" ]; then + local models_resp + models_resp="$(curl -sf "http://127.0.0.1:$PORT/v1/models" 2>/dev/null)" || true + local model_id + model_id="$(printf '%s' "$models_resp" | jq -r '.data[0].id // "unknown"' 2>/dev/null)" || model_id="unknown" + printf 'model: %s\n' "$model_id" + else + printf 'model: (not loaded)\n' + fi + + # Config + printf 'port: %s\n' "$PORT" + printf 'host: %s\n' "$HOST" + + # GPU + local gpu_type + gpu_type="$(detect_gpu)" + printf 'gpu: %s (%s layers)\n' "${gpu_type^^}" "$GPU_LAYERS" + + printf 'context: %s\n' "$CTX_SIZE" + printf 'threads: %s\n' "$THREADS" + + # Autostart + if systemctl --user is-enabled "$SERVICE" &>/dev/null; then + printf 'autostart: enabled\n' + else + printf 'autostart: disabled\n' + fi + + # Endpoint + printf 'endpoint: http://%s:%s\n' "$HOST" "$PORT" + + # Health + if [ "$svc_state" = "running" ]; then + local health + health="$(check_health)" || health="not responding" + printf 'health: %s\n' "$health" + else + printf 'health: not running\n' + fi +} + +cmd_models() { + local dir="${HF_DOWNLOAD_DIR}" + [ -d "$dir" ] || { warn "No models directory — run 'pos ai hf download' first"; return 0; } + + local found=0 + echo "Available GGUF models:" + while IFS= read -r gguf; do + [ -f "$gguf" ] || continue + found=1 + local name size + name="$(basename "$gguf")" + local dir_name + dir_name="$(basename "$(dirname "$gguf")")" + size="$(stat -c%s "$gguf" 2>/dev/null || echo 0)" + local hsize + hsize="$(human_size "$size")" + printf ' %-50s %s\n' "$dir_name/$name" "$hsize" + done < <(find "$dir" -name '*.gguf' -type f 2>/dev/null | sort) + + [ "$found" -eq 0 ] && warn "No .gguf files found — download with 'pos ai hf download --gguf'" +} + +cmd_logs() { + local lines="${SUBCMD_ARGS[0]:-50}" + [[ "$lines" =~ ^[0-9]+$ ]] || err "lines must be a number" + journalctl --user -u "$SERVICE" -n "$lines" --no-pager 2>/dev/null || warn "No logs found — server may not have been started" +} + +# ── Dispatch ─────────────────────────────────────────────────── +case "${SUBCMD:-}" in + "") usage ;; + start) cmd_start ;; + stop) cmd_stop ;; + status) cmd_status ;; + models) cmd_models ;; + logs) cmd_logs ;; + *) err "Unknown subcommand '$SUBCMD' (see --help)" ;; +esac diff --git a/completions/pos.bash b/completions/pos.bash index 078ac78..afc4ebe 100644 --- a/completions/pos.bash +++ b/completions/pos.bash @@ -4,6 +4,7 @@ # GEN:START posflags declare -A _pos_flags _pos_flags[ai-hf]="--branch --gguf --output" +_pos_flags[ai-server]="--port --host --model --ctx --gpu --threads" _pos_flags[communication-matrix-listener]="--enable --disable --status --run" _pos_flags[communication-telegram-listener]="--enable --disable --status --sync-commands --run" _pos_flags[communication-telegram-sender]="--type --caption --parse-mode --no-preview --token --chat-id --markdown" @@ -30,6 +31,7 @@ declare -A _pos_subcmds _pos_subcmds[ai-alias]="create edit remove list show" _pos_subcmds[ai-gemini]="ask chat models sessions capture" _pos_subcmds[ai-openrouter]="ask chat sessions capture" +_pos_subcmds[ai-server]="start stop status models logs" _pos_subcmds[communication-matrix-sender]="send test login" _pos_subcmds[communication-scrcpy]="devices record tcpip connect push pull screenshot info" _pos_subcmds[communication-telegram-listener]="prefix" @@ -45,7 +47,7 @@ _pos_subcmds[share-smb-client]="mount unmount list persist unpersist menu" _pos_subcmds[share-smb-server]="status share unshare list adduser deluser reload enable disable menu" _pos_subcmds[system-backup]="menu" _pos_subcmds[system-schedule]="run list config enable disable status migrate menu" -_pos_subcmds[ai]="ask chat sessions capture models providers alias gemini hf openrouter" +_pos_subcmds[ai]="ask chat sessions capture models providers alias gemini hf openrouter server" # GEN:END possubcmds # GEN:START posconfigscopes declare -a _pos_config_scopes=(ai compose entertainment grab matrix notify scrcpy system telegram ytsync) diff --git a/config/ai.env b/config/ai.env index 24cdcb8..fb4ffe0 100644 --- a/config/ai.env +++ b/config/ai.env @@ -16,3 +16,11 @@ # # System prompt: # AI_SYSTEM_PROMPT= # Custom system prompt (overrides built-in; empty to reset) +# +# llama.cpp local inference server (pos ai server): +# LLAMACPP_PORT=8088 # Server port (default 8088) +# LLAMACPP_HOST=127.0.0.1 # Bind address (default 127.0.0.1) +# LLAMACPP_MODEL= # Default model path (GGUF file) +# LLAMACPP_CTX_SIZE=4096 # Context window size (default 4096) +# LLAMACPP_GPU_LAYERS=-1 # GPU layers: -1=auto, 0=CPU only (default -1) +# LLAMACPP_THREADS= # CPU threads (default: nproc) diff --git a/lib/ai-providers/llamacpp.sh b/lib/ai-providers/llamacpp.sh new file mode 100644 index 0000000..57cf34e --- /dev/null +++ b/lib/ai-providers/llamacpp.sh @@ -0,0 +1,61 @@ +#!/usr/bin/env bash +# Local llama.cpp provider adapter for pos-ai +# Provider-specific: API call via OpenAI-compatible /v1/chat/completions +# Part of the R8 provider-agnostic architecture (lib/ai-providers/). + +# Provider-specific config variables (auto-discovered by pos config ai): +# PROVIDER_CONFIG: LLAMACPP_MODEL=:Default model path (GGUF file) + +provider_name() { printf 'Local llama.cpp'; } + +provider_default_model() { + local port="${LLAMACPP_PORT:-8088}" + local model + model="$(curl -sf "http://127.0.0.1:$port/v1/models" 2>/dev/null | jq -r '.data[0].id // empty')" + [ -n "$model" ] && printf '%s' "$model" || printf '(no model loaded)' +} + +# $1=model $2=messages JSON ({"messages":[{role,content}]}) $3=optional system prompt +provider_generate() { + local model="$1" messages="$2" system="${3:-}" port="${LLAMACPP_PORT:-8088}" + local body resp code body_out + # Build messages array with optional system prompt + if [ -n "$system" ]; then + body="$(printf '%s' "$messages" | jq -c --arg s "$system" \ + '[{role:"system",content:$s}] + .messages')" + else + body="$(printf '%s' "$messages" | jq -c '.messages')" + fi + body="$(printf '%s' "$body" | jq -nc --arg m "$model" --argjson msgs "$body" \ + '{model:$m, messages:$msgs, stream:false}')" + resp="$(curl -sS -m 120 -X POST "http://127.0.0.1:$port/v1/chat/completions" \ + -H "Content-Type: application/json" \ + --write-out $'\n%{http_code}' \ + --data "$body")" || { echo "request failed (curl exit $?)" >&2; return 1; } + code="${resp##*$'\n'}" + body_out="${resp%$'\n'*}" + if [ "$code" != "200" ]; then + echo "API error $code" >&2 + return 1 + fi + printf '%s' "$body_out" | jq -r '.choices[0].message.content // ""' +} + +# $1=current default model → stdout=formatted model list +provider_models_list() { + local model="$1" port="${LLAMACPP_PORT:-8088}" resp code body + resp="$(curl -sf "http://127.0.0.1:$port/v1/models" \ + --write-out $'\n%{http_code}')" || { echo "server not running" >&2; return 1; } + code="${resp##*$'\n'}" + body="${resp%$'\n'*}" + [ "$code" = "200" ] || { echo "API error $code" >&2; return 1; } + echo "Local llama.cpp models:" + printf '%s' "$body" | jq -r '.data[]? | .id' | while IFS= read -r m; do + [ -n "$m" ] || continue + if [ "$m" = "$model" ]; then + printf ' %-48s <- loaded\n' "$m" + else + printf ' %-48s\n' "$m" + fi + done +}