#!/usr/bin/env bash
set -euo pipefail
# POS: ai hf — Download AI models from Hugging Face (search, download, manage)
# POS_FLAGS: --branch --gguf --list --output --quant --include --exclude --revision
# POS_DEPS: curl jq
# POS_CONFIG: ai | ai.env | HF_TOKEN=secret:Hugging Face API token (https://huggingface.co/settings/tokens) | HF_DOWNLOAD_DIR=:Model download directory (default ~/.local/share/linux_post_install/ai/models)
# POS_EXAMPLES: pos ai hf search llama 7b | Search Hugging Face for "llama 7b" models
# POS_EXAMPLES: pos ai hf download meta-llama/Llama-3.1-8B-Instruct | Download all files from a repo
# POS_EXAMPLES: pos ai hf download meta-llama/Llama-3.1-8B-Instruct --gguf | Download only GGUF quantized files
# POS_EXAMPLES: pos ai hf download org/model-GGUF --gguf --quant Q8_0 | Download one quant directory's GGUF shards
# POS_EXAMPLES: pos ai hf download meta-llama/Llama-3.1-8B-Instruct --list | List remote repository files (what --gguf/download would fetch)
# POS_EXAMPLES: pos ai hf download meta-llama/Llama-3.1-8B-Instruct config.json | Download a single file
# POS_EXAMPLES: pos ai hf list | List downloaded models
# POS_EXAMPLES: pos ai hf remove meta-llama-Llama-3.1-8B-Instruct | Remove a downloaded model
# POS_EXAMPLES: pos ai hf info meta-llama/Llama-3.1-8B-Instruct | Show repository information
# POS_EXAMPLES: pos ai hf files meta-llama/Llama-3.1-8B-Instruct | List repository files
# POS_EXAMPLES: pos ai hf download meta-llama/Llama-3.1-8B-Instruct --include "*.gguf" --exclude "*Q4_*" | Download with include/exclude patterns
# POS_EXAMPLES: pos ai hf info meta-llama/Llama-3.1-8B-Instruct | Show repository information
# POS_EXAMPLES: pos ai hf files meta-llama/Llama-3.1-8B-Instruct | List repository files
# POS_EXAMPLES: pos ai hf download meta-llama/Llama-3.1-8B-Instruct --include "*.gguf" --exclude "*Q4_*" | Download with include/exclude patterns

source "$(dirname "$0")/../lib/common.sh" 2>/dev/null || source "$(dirname "$0")/common.sh"

# ── Dependencies (before --help) ───────────────────────────────
command -v curl &>/dev/null || err "curl not found (install curl)"
command -v jq &>/dev/null   || err "jq not found (install jq)"

# ── Config & defaults ──────────────────────────────────────────
CONFIG_FILE="${CONFIG_FILE:-$HOME/.config/linux_post_install/ai.env}"
HF_TOKEN="${HF_TOKEN:-}"
HF_DOWNLOAD_DIR="${HF_DOWNLOAD_DIR:-$HOME/.local/share/linux_post_install/ai/models}"

load_hf_config() {
    [ -f "$CONFIG_FILE" ] || return 0
    local k v
    while IFS='=' read -r k v; do
        [ -n "$k" ] || continue
        case "$k" in
            \#*) continue ;;
        esac
        v="${v%\"}"; v="${v#\"}"; v="${v%\'}"; v="${v#\'}"
        v="${v//$'\r'/}"
        if [ -z "${!k:-}" ]; then
            export "$k"="$v"
        fi
    done < <(grep -E '^[A-Z_]+=' "$CONFIG_FILE" || true)
}

load_hf_config

# ── Usage ──────────────────────────────────────────────────────
usage() {
    cat <<'EOF'
Usage: pos ai hf <subcommand> [args]

Hugging Face model downloader — search, download, and manage AI models.

Subcommands:
  search <query>                     Search Hugging Face models
  download <repo-id> [filename]      Download a file or entire repo
  list                               List locally downloaded models
  remove <repo-id>                   Remove a downloaded model
  info <repo-id>                     Show repository information
  files <repo-id>                    List repository files
  cache                              Manage local cache

Download options:
  --branch <rev>                     Download from a specific branch/revision
  --gguf                             Download only .gguf weight files (excludes
                                     mmproj/imatrix/vision/MTP artifacts)
  --quant <dir>                      With --gguf: pick one quant directory when a
                                     repo groups weights into several (e.g.
                                     --gguf --quant Q8_0)
  --list                             List remote repository files without downloading
  --output <dir>                     Override download directory
  --include <pattern>                Include files matching pattern (supports glob)
  --exclude <pattern>                Exclude files matching pattern (supports glob)
  --revision <rev>                   Specific revision (commit/tag/branch)

Examples:
  pos ai hf search llama 7b
  pos ai hf download meta-llama/Llama-3.1-8B-Instruct
  pos ai hf download meta-llama/Llama-3.1-8B-Instruct --gguf
  pos ai hf download org/model-GGUF --gguf --quant Q8_0
  pos ai hf download meta-llama/Llama-3.1-8B-Instruct --list
  pos ai hf download meta-llama/Llama-3.1-8B-Instruct config.json
  pos ai hf download org/model-GGUF Q8_0/model-00001-of-00006.gguf
  pos ai hf download org/model-GGUF model-00001-of-00006.gguf
  pos ai hf download meta-llama/Llama-3.1-8B-Instruct --branch main
  pos ai hf list
  pos ai hf remove meta-llama-Llama-3.1-8B-Instruct
  pos ai hf info meta-llama/Llama-3.1-8B-Instruct
  pos ai hf files meta-llama/Llama-3.1-8B-Instruct
  pos ai hf download meta-llama/Llama-3.1-8B-Instruct --include "*.gguf" --exclude "*Q4_*"
  pos ai hf download meta-llama/Llama-3.1-8B-Instruct --revision v1.0

A filename may be a full path (Q8_0/model.gguf) or a bare name (model.gguf) —
bare names matching files in multiple directories error and ask for the full path.
--list shows files on the remote repo; 'list' shows models already downloaded.

Config (~/.config/linux_post_install/ai.env):
  HF_TOKEN           Hugging Face API token (better rate limits for public repos)
  HF_DOWNLOAD_DIR    Model download directory (default ~/.local/share/linux_post_install/ai/models)

Exit codes:
  0  success
  1  error (missing deps, invalid input, API failure)
EOF
    exit 0
}

# ── Parse global flags ─────────────────────────────────────────
SUBCMD=""
SUBCMD_ARGS=()
BRANCH=""
GGUF_ONLY=0
OUTPUT_DIR=""
LIST_FILES=0
QUANT_DIR=""
INCLUDE_PATTERN=""
EXCLUDE_PATTERN=""
REVISION=""

while [ $# -gt 0 ]; do
    case "$1" in
        -h|--help) usage ;;
        --branch)
            [ $# -ge 2 ] || err "--branch requires a value"
            BRANCH="$2"; shift 2 ;;
        --gguf)
            GGUF_ONLY=1; shift ;;
        --list)
            LIST_FILES=1; shift ;;
        --quant)
            [ $# -ge 2 ] || err "--quant requires a value"
            QUANT_DIR="$2"; shift 2 ;;
        --output)
            [ $# -ge 2 ] || err "--output requires a value"
            OUTPUT_DIR="$2"; shift 2 ;;
        --include)
            [ $# -ge 2 ] || err "--include requires a value"
            INCLUDE_PATTERN="$2"; shift 2 ;;
        --exclude)
            [ $# -ge 2 ] || err "--exclude requires a value"
            EXCLUDE_PATTERN="$2"; shift 2 ;;
        --revision)
            [ $# -ge 2 ] || err "--revision requires a value"
            REVISION="$2"; shift 2 ;;
        -*)
            err "Unknown option '$1' (see --help)" ;;
        *)
            if [ -z "$SUBCMD" ]; then
                SUBCMD="$1"
            else
                SUBCMD_ARGS+=("$1")
            fi
            shift ;;
    esac
done

[ -n "$SUBCMD" ] || usage

# Apply output dir override
if [ -n "$OUTPUT_DIR" ]; then
    HF_DOWNLOAD_DIR="$OUTPUT_DIR"
fi
HF_DOWNLOAD_DIR="${HF_DOWNLOAD_DIR:-$HOME/.local/share/linux_post_install/ai/models}"

# ── Token warning ──────────────────────────────────────────────
if [ -z "$HF_TOKEN" ]; then
    warn "No HF_TOKEN set — using anonymous access"
fi

# ── HF API helpers ─────────────────────────────────────────────
HF_BASE="https://huggingface.co"
HF_API_BASE="https://huggingface.co/api"

HF_MAX_PAGES=20
HF_GGUF_FILTER='[ .[] |
  select(.rfilename | type == "string") |
  select(.rfilename | ascii_downcase | endswith(".gguf")) |
  select(.rfilename | ascii_downcase | test("mmproj|imatrix|clip|vision|projector|mtp") | not)
]'

hf_auth_header() {
    if [ -n "$HF_TOKEN" ]; then
        printf 'Authorization: Bearer %s' "$HF_TOKEN"
    fi
}

hf_api() {
    local endpoint="$1"
    local hdr_file="${2:-}"          # optional: dump response headers (Link: rel="next")
    local url
    case "$endpoint" in
        http://*|https://*) url="$endpoint" ;;
        *)                  url="${HF_API_BASE}${endpoint}" ;;
    esac
    local auth_header
    auth_header="$(hf_auth_header)"

    local http_code body tmpfile
    tmpfile="$(mktemp)"

    local curl_args=(-sS -w '%{http_code}' -o "$tmpfile" --max-time 30)
    if [ -n "$auth_header" ]; then
        curl_args+=(-H "$auth_header")
    fi
    if [ -n "$hdr_file" ]; then
        curl_args+=(-D "$hdr_file")
    fi

    # Rate limit retry: on 429, sleep and retry once
    local attempt=0
    while [ $attempt -lt 2 ]; do
        http_code="$(curl "${curl_args[@]}" "$url" 2>/dev/null)" || {
            rm -f "$tmpfile" "$hdr_file"
            err "Connection timed out — check network"
        }

        if [ "$http_code" = "429" ]; then
            local retry_after
            retry_after="$(curl -sI -H "${auth_header:-}" "$url" 2>/dev/null | grep -i 'retry-after:' | tr -d '\r' | awk '{print $2}')"
            retry_after="${retry_after:-60}"
            warn "Rate limited — waiting ${retry_after}s before retry"
            sleep "$retry_after"
            attempt=$((attempt + 1))
            continue
        fi

        break
    done

    body="$(cat "$tmpfile")"
    rm -f "$tmpfile"

    case "$http_code" in
        200) ;;
        401|403) err "Authentication failed — check HF_TOKEN (pos config ai)" ;;
        404) err "Model not found: ${endpoint#/api/models/}" ;;
        429) err "Rate limit exceeded — try again later" ;;
        *)   err "API request failed (HTTP $http_code)" ;;
    esac

    # Validate JSON
    if ! printf '%s' "$body" | jq empty 2>/dev/null; then
        err "Failed to parse API response — check network or HF status"
    fi

    printf '%s' "$body"
}

# hf_paginate <endpoint> → JSON array built from every Link: rel="next" page
hf_paginate() {
    local endpoint="$1"
    local url
    case "$endpoint" in
        http://*|https://*) url="$endpoint" ;;
        *)                  url="${HF_API_BASE}${endpoint}" ;;
    esac
    local combined="[]"
    local page=0
    local hdr_file body next_url
    while [ -n "$url" ]; do
        page=$((page + 1))
        [ "$page" -gt "$HF_MAX_PAGES" ] \
            && err "Repository listing exceeded ${HF_MAX_PAGES} pages — aborting"
        hdr_file="$(mktemp)"
        body="$(hf_api "$url" "$hdr_file")"
        combined="$(printf '%s\n%s' "$combined" "$body" | jq -c -s 'add')"
        next_url="$(sed -n 's/^link: <\([^>]*\)>; rel="next".*/\1/Ip' "$hdr_file" | tr -d '\r' | tail -1)"
        rm -f "$hdr_file"
        url="${next_url:-}"
    done
    printf '%s' "$combined"
}

hf_repo_files() {
    local repo_id="$1"
    local branch="${2:-main}"
    local ns="${repo_id%%/*}"
    local repo="${repo_id#*/}"

    if [ "$ns" = "$repo" ]; then
        err "Invalid repo format: use namespace/model-name"
    fi

    # Try /tree/ endpoint first (has file sizes + LFS info)
    local endpoint="/models/${ns}/${repo}/tree/${branch}?recursive=true"
    local result
    if result="$(hf_paginate "$endpoint" 2>/dev/null)"; then
        # Tree API returns {type,path,size,oid[,lfs]} per entry — normalize to the
        # {rfilename,size} shape the rest of the pipeline expects (same as fallback).
        # Skip "directory" entries and guard non-object entries (error objects crash .[]).
        printf '%s' "$result" | jq '[.[] | select(type == "object" and .type == "file") | {rfilename: .path, size: (.size // 0)}]'
        return 0
    fi

    # Fallback: /api/models/{ns}/{repo} (siblings, no sizes)
    warn "Tree endpoint unavailable, using repo metadata"
    local fallback
    fallback="$(hf_api "/models/${ns}/${repo}")" || err "Failed to fetch repo info for $repo_id"
    printf '%s' "$fallback" | jq '[.siblings[]? | select(type == "object") | {rfilename: (.rfilename // ""), size: (.size // 0)}]'
}

hf_search() {
    local query="$1"
    local limit="${2:-10}"
    local encoded_query
    encoded_query="$(printf '%s' "$query" | jq -sRr @uri)"
    local endpoint="/models?search=${encoded_query}&sort=downloads&direction=-1&limit=${limit}"
    hf_api "$endpoint"
}

# ── Utility functions ──────────────────────────────────────────
hf_repo_dir() {
    local repo_id="$1"
    printf '%s' "$HF_DOWNLOAD_DIR/${repo_id//\//-}"
}

hf_human_size() {
    local bytes="$1"
    if [ "$bytes" -ge 1073741824 ]; then
        awk "BEGIN { printf \"%.1f GB\", $bytes / 1073741824 }"
    elif [ "$bytes" -ge 1048576 ]; then
        awk "BEGIN { printf \"%.1f MB\", $bytes / 1048576 }"
    elif [ "$bytes" -ge 1024 ]; then
        awk "BEGIN { printf \"%.1f KB\", $bytes / 1024 }"
    else
        printf '%d B' "$bytes"
    fi
}

# hf_quant_candidates <files-json> → [{dir, files, size}] sorted by dir
hf_quant_candidates() {
    printf '%s' "$1" | jq -c '[.[] | select(.rfilename | contains("/")) |
      {dir: (.rfilename | split("/")[0]), size: (.size // 0)}]
      | group_by(.dir)
      | map({dir: .[0].dir, files: length, size: (map(.size) | add)})
      | sort_by(.dir)'
}

# Refactored hf_gguf_quant_gate function with improved structure
# hf_gguf_quant_gate <files-json> <quant-dir> <repo-id> → filtered JSON (stdout) or err
hf_gguf_quant_gate() {
    local json="$1" quant="${2:-}" repo_id="$3"
    
    # Validate input
    if [ -z "$json" ]; then
        err "No files provided to quant gate"
    fi
    
    # Count top-level files vs directory files
    local top_count dir_count
    top_count="$(printf '%s' "$json" | jq '[.[] | select(.rfilename | contains("/") | not)] | length')"
    dir_count="$(printf '%s' "$json" | jq '[.[] | select(.rfilename | contains("/")) | .rfilename | split("/")[0]] | unique | length')"

    # Handle case: top-level .gguf files (no quant dirs)
    if [ "$top_count" -gt 0 ]; then
        if [ -n "$quant" ]; then
            err "--quant is for repos that group weights into quant directories — $repo_id has top-level .gguf files, --quant is not needed"
        fi
        printf '%s' "$json"
        return 0
    fi

    # Handle case: single quant directory
    if [ "$dir_count" -eq 1 ]; then
        local only_dir
        only_dir="$(printf '%s' "$json" | jq -r '.[0].rfilename | split("/")[0]')"
        if [ -n "$quant" ] && [ "$quant" != "$only_dir" ]; then
            err "No quant directory '$quant' in $repo_id — weights live in: $only_dir"
        fi
        printf '%s' "$json"
        return 0
    fi

    # Handle case: multiple quant directories - require quant selection
    if [ -z "$quant" ]; then
        local msg
        msg="$(printf 'Repo %s organizes weights into %d quant directories — pick one with --quant:\n' "$repo_id" "$dir_count")"
        while IFS=$'\t' read -r dir files size; do
            msg+="$(printf '  %-20s %d files, %s\n' "$dir" "$files" "$(hf_human_size "$size")")"
        done < <(hf_quant_candidates "$json" | jq -r '.[] | [.dir, (.files|tostring), (.size|tostring)] | @tsv')
        err "$msg"
    fi

    # Filter by specified quant directory
    local selected
    selected="$(printf '%s' "$json" | jq -c --arg q "$quant" '[.[] | select(.rfilename | split("/")[0] == $q)]')"
    if [ "$(printf '%s' "$selected" | jq 'length')" -eq 0 ]; then
        local msg
        msg="$(printf 'No weights in quant directory %s in %s — candidates:\n' "$quant" "$repo_id")"
        while IFS=$'\t' read -r dir files size; do
            msg+="$(printf '  %-20s %d files, %s\n' "$dir" "$files" "$(hf_human_size "$size")")"
        done < <(hf_quant_candidates "$json" | jq -r '.[] | [.dir, (.files|tostring), (.size|tostring)] | @tsv')
        err "$msg"
    fi
    printf '%s' "$selected"
}

# Enhanced error reporting function
err_with_context() {
    local msg="$1"
    local context="${2:-}"
    if [ -n "$context" ]; then
        echo "Error: $msg (Context: $context)" >&2
    else
        echo "Error: $msg" >&2
    fi
    exit 1
}

# hf_list_files <repo-id> <branch> <files-json> → stdout table, no downloads
hf_list_files() {
    local repo_id="$1" branch="$2" json="$3"
    local count total
    count="$(printf '%s' "$json" | jq 'length')"
    [ "$count" -gt 0 ] || err "No files found in $repo_id${branch:+ (branch: $branch)}"
    total="$(printf '%s' "$json" | jq '[.[].size // 0] | add // 0')"
    printf 'Files in %s (branch: %s, %d file(s), %s):\n' \
        "$repo_id" "$branch" "$count" "$(hf_human_size "$total")"
    printf '%s' "$json" | jq -r 'sort_by(.rfilename)[] | [.rfilename, (.size // 0)] | @tsv' | \
        while IFS=$'\t' read -r rpath rsize; do
            printf '  %-60s %s\n' "$rpath" "$(hf_human_size "$rsize")"
        done
}

hf_resolve_branch() {
    local repo_id="$1"
    local branch="${2:-}"
    if [ -n "$branch" ]; then
        printf '%s' "$branch"
        return
    fi
    # Try to get default branch from API
    local ns="${repo_id%%/*}"
    local repo="${repo_id#*/}"
    local meta
    if meta="$(hf_api "/models/${ns}/${repo}" 2>/dev/null)"; then
        local default_branch
        default_branch="$(printf '%s' "$meta" | jq -r '.defaultBranch // empty' 2>/dev/null)"
        if [ -n "$default_branch" ]; then
            printf '%s' "$default_branch"
            return
        fi
    fi
    printf 'main'
}

# ── Download helpers ───────────────────────────────────────────
hf_download_file() {
    local url="$1"
    local target="$2"
    local auth_header
    auth_header="$(hf_auth_header)"

    local curl_args=(-L -C - --progress-bar -o "$target")
    if [ -n "$auth_header" ]; then
        curl_args+=(-H "$auth_header")
    fi

    if curl "${curl_args[@]}" "$url" 2>&1; then
        if [ -s "$target" ]; then
            return 0
        else
            warn "Downloaded file is empty: $target"
            return 1
        fi
    else
        warn "Download interrupted for $(basename "$target") (resume with same command)"
        return 1
    fi
}

# Enhanced progress function to provide better feedback
hf_download_with_progress() {
    local url="$1"
    local target="$2"
    local file_name="$(basename "$target")"
    
    # Create parent directory
    mkdir -p "$(dirname "$target")"
    
    local auth_header
    auth_header="$(hf_auth_header)"
    
    local curl_args=(-L -C - --progress-bar -o "$target")
    if [ -n "$auth_header" ]; then
        curl_args+=(-H "$auth_header")
    fi
    
    # Run download with progress bar
    if curl "${curl_args[@]}" "$url" 2>&1; then
        if [ -s "$target" ]; then
            return 0
        else
            warn "Downloaded file is empty: $target"
            return 1
        fi
    else
        warn "Download interrupted for $file_name (resume with same command)"
        return 1
    fi
}

# ── Parallel download helpers ──────────────────────────────────
# Global variables for parallel downloads
PARALLEL_DOWNLOADS=4  # Default parallel downloads

# Function to run download in background and track it
run_parallel_download() {
    local url="$1"
    local target="$2"
    local job_id="$3"
    
    # Run download and capture result
    if hf_download_with_progress "$url" "$target"; then
        echo "SUCCESS:$job_id"
        return 0
    else
        echo "FAILED:$job_id"
        return 1
    fi
}

# ── Subcommands ────────────────────────────────────────────────

cmd_search() {
    local query="${SUBCMD_ARGS[*]:-}"
    [ -n "$query" ] || err "Usage: pos ai hf search <query>"

    local result
    result="$(hf_search "$query" "10")"

    local count
    count="$(printf '%s' "$result" | jq 'length')"
    [ "$count" -gt 0 ] || { warn "No models found for '$query'"; return 0; }

    printf 'Found %d models for "%s":\n' "$count" "$query"
    printf '%s' "$result" | jq -r '.[] | "  \(.id)\t\(.downloads // 0)\t\(.likes // 0)"' | \
        while IFS=$'\t' read -r id downloads likes; do
            local dl_str
            if [ "$downloads" -ge 1000 ]; then
                dl_str="$(awk "BEGIN { printf \"%.1fk\", $downloads / 1000 }")"
            else
                dl_str="$downloads"
            fi
            printf '  %-55s %s downloads\n' "$id" "$dl_str"
        done
}

cmd_download() {
    local repo_id="${SUBCMD_ARGS[0]:-}"
    [ -n "$repo_id" ] || err "Usage: pos ai hf download <repo-id> [filename]"

    # Validate repo-id contains /
    [[ "$repo_id" == */* ]] || err "Invalid repo format: use namespace/model-name"

    local filename="${SUBCMD_ARGS[1]:-}"

    # Flag pre-checks
    [ -n "$QUANT_DIR" ] && [ "$GGUF_ONLY" -eq 0 ] && err "--quant requires --gguf"
    [ "$LIST_FILES" -eq 1 ] && [ -n "$filename" ] && err "--list cannot be combined with a filename"
    [ -n "$INCLUDE_PATTERN" ] && [ -n "$EXCLUDE_PATTERN" ] && [ "$GGUF_ONLY" -eq 1 ] && err "--include/--exclude cannot be used with --gguf"
    [ -n "$INCLUDE_PATTERN" ] && [ -n "$EXCLUDE_PATTERN" ] && [ -n "$filename" ] && err "--include/--exclude cannot be used with specific filenames"

    local branch
    branch="$(hf_resolve_branch "$repo_id" "$REVISION")"

    # Get file list from API (recursive + paginated tree)
    local files_json
    files_json="$(hf_repo_files "$repo_id" "$branch")"

    # --list mode: print what download would fetch, don't download
    if [ "$LIST_FILES" -eq 1 ]; then
        local list_json="$files_json"
        if [ "$GGUF_ONLY" -eq 1 ]; then
            list_json="$(printf '%s' "$list_json" | jq -c "$HF_GGUF_FILTER")"
            [ "$(printf '%s' "$list_json" | jq 'length')" -gt 0 ] \
                && list_json="$(hf_gguf_quant_gate "$list_json" "$QUANT_DIR" "$repo_id")"
        fi
        hf_list_files "$repo_id" "$branch" "$list_json"
        return 0
    fi

    # Filter files
    local filtered_files
    if [ -n "$filename" ]; then
        # Single file mode — explicit filename wins over --gguf/--quant
        if [[ "$filename" == */* ]]; then
            # Full path → exact .rfilename match
            filtered_files="$(printf '%s' "$files_json" | jq -c --arg fn "$filename" '[.[] | select(.rfilename == $fn)]')"
        else
            # Bare name → basename match across all depths
            filtered_files="$(printf '%s' "$files_json" | jq -c --arg fn "$filename" '[.[] | select((.rfilename | type) == "string") | select(.rfilename | split("/")[-1] == $fn)]')"
        fi
    elif [ "$GGUF_ONLY" -eq 1 ]; then
        # GGUF filter
        filtered_files="$(printf '%s' "$files_json" | jq -c "$HF_GGUF_FILTER")"
        [ "$(printf '%s' "$filtered_files" | jq 'length')" -gt 0 ] \
            && filtered_files="$(hf_gguf_quant_gate "$filtered_files" "$QUANT_DIR" "$repo_id")"
    elif [ -n "$INCLUDE_PATTERN" ] || [ -n "$EXCLUDE_PATTERN" ]; then
        # Pattern filtering
        filtered_files="$files_json"
        if [ -n "$INCLUDE_PATTERN" ]; then
            # Use jq to filter files matching include pattern
            local include_filter
            include_filter=".[] | select(.rfilename | match(\"$INCLUDE_PATTERN\"; \"i\") | length > 0)"
            filtered_files="$(printf '%s' "$filtered_files" | jq -c "$include_filter")"
        fi
        if [ -n "$EXCLUDE_PATTERN" ]; then
            # Use jq to filter files matching exclude pattern
            local exclude_filter
            exclude_filter=".[] | select(.rfilename | match(\"$EXCLUDE_PATTERN\"; \"i\") | length == 0)"
            filtered_files="$(printf '%s' "$filtered_files" | jq -c "$exclude_filter")"
        fi
    else
        # All files
        filtered_files="$(printf '%s' "$files_json" | jq -c '.')"
    fi

    local file_count
    file_count="$(printf '%s' "$filtered_files" | jq 'length')"
    if [ "$file_count" -eq 0 ]; then
        if [ -n "$filename" ]; then
            err "File not found: $filename in $repo_id (branch: ${branch})"
        elif [ "$GGUF_ONLY" -eq 1 ]; then
            err "No .gguf files found in $repo_id${branch:+ (branch: $branch)} — try without --gguf"
        else
            err "No files to download"
        fi
    fi

    # Ambiguity guard: bare name matching multiple files (subdirs) → ask for full path
    if [ -n "$filename" ] && [[ "$filename" != */* ]] && [ "$file_count" -gt 1 ]; then
        err "$(printf 'Multiple files match "%s" in %s — use the full path:\n' "$filename" "$repo_id"; printf '%s' "$filtered_files" | jq -r '.[] | "  \(.rfilename)"')"
    fi

    # Prepare target directory
    local target_dir
    target_dir="$(hf_repo_dir "$repo_id")"
    mkdir -p "$target_dir"

    # Disk space pre-flight check
    local total_size
    total_size="$(printf '%s' "$filtered_files" | jq '[.[].size // 0] | add // 0')"
    if [ "$total_size" -gt 0 ]; then
        local avail_kb
        avail_kb="$(df --output=avail "$target_dir" 2>/dev/null | tail -1 | tr -d ' ')"
        local need_kb=$((total_size / 1024))
        if [ "$avail_kb" -lt "$need_kb" ]; then
            local need_human avail_human
            need_human="$(hf_human_size "$total_size")"
            avail_human="$(hf_human_size "$((avail_kb * 1024))")"
            warn "Low disk space: need $need_human, only $avail_human available"
        fi
    fi

    local downloaded=0
    local ns="${repo_id%%/*}"
    local repo="${repo_id#*/}"

    # If we're downloading multiple files, run them in parallel
    if [ "$file_count" -gt 1 ]; then
        local temp_dir
        temp_dir="$(mktemp -d)"
        local job_pids=()
        local max_jobs="${PARALLEL_DOWNLOADS:-4}"
        local completed_jobs=0

        # Process files in parallel batches
        while IFS= read -r file_json; do
            local fname fsize
            fname="$(printf '%s' "$file_json" | jq -r '.rfilename')"
            fsize="$(printf '%s' "$file_json" | jq -r '.size // 0' || echo 0)"

            local url="${HF_BASE}/${ns}/${repo}/resolve/${branch}/${fname}"
            local target="${target_dir}/${fname}"

            # Start background job
            hf_download_with_progress "$url" "$target" &
            local pid=$!
            job_pids+=($pid)

            # Limit parallel jobs
            if [ ${#job_pids[@]} -ge "$max_jobs" ]; then
                # Wait for oldest job to complete
                wait "${job_pids[0]}"
                completed_jobs=$((completed_jobs + 1))
                printf '[%d/%d] Completed: %s\n' "$completed_jobs" "$file_count" "$fname" >&2
                # Shift job array
                job_pids=("${job_pids[@]:1}")
            fi
        done < <(printf '%s' "$filtered_files" | jq -c '.[]')

        # Wait for remaining jobs
        for pid in "${job_pids[@]}"; do
            wait "$pid"
            completed_jobs=$((completed_jobs + 1))
            printf '[%d/%d] Completed\n' "$completed_jobs" "$file_count" >&2
        done
        
        # Clean up temp directory
        rm -rf "$temp_dir"
    else
        # Single file download - use original sequential approach
        while IFS= read -r file_json; do
            local fname fsize
            fname="$(printf '%s' "$file_json" | jq -r '.rfilename')"
            fsize="$(printf '%s' "$file_json" | jq -r '.size // 0')"
            total_size=$((total_size + fsize))

            local url="${HF_BASE}/${ns}/${repo}/resolve/${branch}/${fname}"
            local target="${target_dir}/${fname}"

            if [ "$file_count" -gt 1 ]; then
                downloaded=$((downloaded + 1))
                printf '[%d/%d] Downloading %s...\n' "$downloaded" "$file_count" "$fname" >&2
            fi

            if ! hf_download_with_progress "$url" "$target"; then
                warn "Failed to download $fname"
                continue
            fi
        done < <(printf '%s' "$filtered_files" | jq -c '.[]')
    fi

    # Write metadata
    local meta_file="${target_dir}/.hf-meta"
    local timestamp
    timestamp="$(date -u +%Y-%m-%dT%H:%M:%SZ)"
    local file_list
    file_list="$(printf '%s' "$filtered_files" | jq -c '[.[] | .rfilename]')"

    cat > "$meta_file" <<METAEOF
{
  "repo_id": "$repo_id",
  "branch": "$branch",
  "downloaded_at": "$timestamp",
  "files": $file_list
}
METAEOF

    # Summary
    if [ "$file_count" -eq 1 ]; then
        local fname
        fname="$(printf '%s' "$filtered_files" | jq -r '.[0].rfilename')"
        local fsize
        fsize="$(printf '%s' "$filtered_files" | jq -r '.[0].size // 0')"
        local human_size
        human_size="$(hf_human_size "$fsize")"
        printf '📥 Downloaded: %s/%s (%s)\n' "$repo_id" "$fname" "$human_size"
        printf '📁 %s/%s\n' "$target_dir" "$fname"
    else
        local total_human
        total_human="$(hf_human_size "$total_size")"
        printf '📥 Downloaded: %s (%d files, %s)\n' "$repo_id" "$file_count" "$total_human"
        printf '📁 %s/\n' "$target_dir"
    fi
}

cmd_list() {
    [ -d "$HF_DOWNLOAD_DIR" ] || { warn "No models downloaded yet"; return 0; }

    local models=()
    while IFS= read -r dir; do
        [ -d "$dir" ] || continue
        local meta_file="${dir}/.hf-meta"
        [ -f "$meta_file" ] || continue
        models+=("$dir")
    done < <(find "$HF_DOWNLOAD_DIR" -mindepth 1 -maxdepth 1 -type d 2>/dev/null | sort)

    [ ${#models[@]} -gt 0 ] || { warn "No models downloaded yet"; return 0; }

    printf 'Downloaded models (%d):\n' "${#models[@]}"
    for dir in "${models[@]}"; do
        local meta_file="${dir}/.hf-meta"
        local name
        name="$(basename "$dir")"
        local total_size=0
        local date_str
        date_str="$(jq -r '.downloaded_at // "unknown"' "$meta_file" 2>/dev/null | cut -dT -f1)"

        # Calculate total size
        while IFS= read -r file; do
            [ -f "$file" ] || continue
            local size
            size="$(stat -c%s "$file" 2>/dev/null || echo 0)"
            total_size=$((total_size + size))
        done < <(find "$dir" -type f ! -name '.hf-meta' 2>/dev/null)

        local human_size
        human_size="$(hf_human_size "$total_size")"
        printf '  %-50s %s   %s\n' "$name" "$human_size" "$date_str"
    done
}

cmd_remove() {
    local repo_id="${SUBCMD_ARGS[0]:-}"
    [ -n "$repo_id" ] || err "Usage: pos ai hf remove <repo-id>"

    local target_dir
    target_dir="$(hf_repo_dir "$repo_id")"

    [ -d "$target_dir" ] || err "Model not found: $repo_id"

    # Calculate size before removal
    local total_size=0
    while IFS= read -r file; do
        [ -f "$file" ] || continue
        local size
        size="$(stat -c%s "$file" 2>/dev/null || echo 0)"
        total_size=$((total_size + size))
    done < <(find "$target_dir" -type f 2>/dev/null)

    local human_size
    human_size="$(hf_human_size "$total_size")"

    rm -rf "$target_dir"
    printf 'Removed: %s (freed %s)\n' "$repo_id" "$human_size"
}

cmd_info() {
    local repo_id="${SUBCMD_ARGS[0]:-}"
    [ -n "$repo_id" ] || err "Usage: pos ai hf info <repo-id>"

    local ns="${repo_id%%/*}"
    local repo="${repo_id#*/}"

    local info_json
    info_json="$(hf_api "/models/${ns}/${repo}")" || err "Failed to fetch repository info for $repo_id"

    local model_name
    model_name="$(printf '%s' "$info_json" | jq -r '.id')" 
    local downloads
    downloads="$(printf '%s' "$info_json" | jq -r '.downloads // 0')"
    local likes
    likes="$(printf '%s' "$info_json" | jq -r '.likes // 0')"
    local tags
    tags="$(printf '%s' "$info_json" | jq -r '.tags // [] | join(\", \")')"
    local description
    description="$(printf '%s' "$info_json" | jq -r '.description // \"No description\"')"
    local author
    author="$(printf '%s' "$info_json" | jq -r '.author // \"Unknown\"')"
    local created
    created="$(printf '%s' "$info_json" | jq -r '.createdAt // \"Unknown\"')"
    local last_modified
    last_modified="$(printf '%s' "$info_json" | jq -r '.lastModified // \"Unknown\"')"
    local card_data
    card_data="$(printf '%s' "$info_json" | jq -r '.cardData // {}')"
    local pipeline_tag
    pipeline_tag="$(printf '%s' "$info_json" | jq -r '.pipeline_tag // \"Unknown\"')"
    local model_type
    model_type="$(printf '%s' "$info_json" | jq -r '.modelType // \"Unknown\"')"
    local architectures
    architectures="$(printf '%s' "$info_json" | jq -r '.architectures // [] | join(\", \")')"

    printf "Repository: %s\n" "$model_name"
    printf "Author: %s\n" "$author"
    printf "Description: %s\n" "$description"
    printf "Pipeline tag: %s\n" "$pipeline_tag"
    printf "Model type: %s\n" "$model_type"
    printf "Architectures: %s\n" "$architectures"
    printf "Downloads: %s\n" "$downloads"
    printf "Likes: %s\n" "$likes"
    printf "Created: %s\n" "$created"
    printf "Last modified: %s\n" "$last_modified"
    printf "Tags: %s\n" "$tags"
    printf "\n"

    # Show card data if available
    if [ -n "$card_data" ] && [ "$card_data" != "{}" ]; then
        printf "Card data:\n"
        printf '%s' "$card_data" | jq -r 'to_entries[] | "  \(.key): \(.value)"' 2>/dev/null || printf "  (raw data)\n"
    fi
}

cmd_files() {
    local repo_id="${SUBCMD_ARGS[0]:-}"
    [ -n "$repo_id" ] || err "Usage: pos ai hf files <repo-id>"

    local branch
    branch="$(hf_resolve_branch "$repo_id" "$REVISION")"

    local files_json
    files_json="$(hf_repo_files "$repo_id" "$branch")"

    local count
    count="$(printf '%s' "$files_json" | jq 'length')"
    [ "$count" -gt 0 ] || { warn "No files found in $repo_id (branch: $branch)"; return 0; }

    printf 'Files in %s (branch: %s, %d file(s)):\n' "$repo_id" "$branch" "$count"
    printf '%s' "$files_json" | jq -r 'sort_by(.rfilename)[] | [.rfilename, (.size // 0)] | @tsv' | \
        while IFS=$'\t' read -r rpath rsize; do
            printf '  %-60s %s\n' "$rpath" "$(hf_human_size "$rsize")"
        done
}

    cmd_cache() {
        echo "Cache management is not fully implemented yet."
        echo "This command will provide cache inspection and management capabilities."
    }

# ── Dispatch ───────────────────────────────────────────────────
case "$SUBCMD" in
    search)   cmd_search ;;
    download) cmd_download ;;
    list)     cmd_list ;;
    remove)   cmd_remove ;;
    info)     cmd_info ;;
    files)    cmd_files ;;
    cache)    cmd_cache ;;
    *)        err "Unknown subcommand '$SUBCMD' (see --help)" ;;
esac
