Running a GPU-accelerated LLM on FreeBSD (2-line howto)

The associate launcher with downloader above,

Python:
cat myzsh
#!/bin/sh

# I was created using A.I.
# I Run on FreeBSD
# System Specs: 12GB VRAM (No Spillover), 32GB System RAM

# ==========================================
# CONFIGURATION & PARAMETERS
# ==========================================
BASE_DIR="/home/x/hugging"
HOST="127.0.0.1"
PORT="8123"

# ==========================================
# SEPARATE UTILITY FUNCTIONS
# ==========================================

# Cleanly terminate existing llama-server instances
kill_existing_server() {
    echo "Stopping any existing llama-server instances..."
    killall llama-server >/dev/null 2>&1
    sleep 2
    killall -9 llama-server >/dev/null 2>&1
    echo "Sleeping 6 seconds for port cleanup..."
    sleep 6
    echo "Done sleeping."
}

# Prompt user for the VRAM Tier directory
select_vram_tier() {
    while true; do
        printf "Select VRAM Tier Profile (8, 10, 12, 14): "
        read -r TIER
        case "$TIER" in
            8|10|12|14)
                TARGET_DIR="${BASE_DIR}/${TIER}"
                if [ ! -d "$TARGET_DIR" ]; then
                    echo "❌ Directory $TARGET_DIR does not exist."
                    continue
                fi
                return 0
                ;;
            *)
                echo "Invalid profile tier choice. Choose 8, 10, 12, or 14."
                ;;
        esac
    done
}

# Scan directory and choose an available model
select_model_file() {
    echo ""
    echo "Searching available models in: $TARGET_DIR"
    echo "--------------------------------------------------------"
   
    count=0
    set --
   
    for file in "$TARGET_DIR"/*.gguf; do
        if [ -f "$file" ]; then
            count=$((count + 1))
            set -- "$@" "$file"
            filename=$(basename "$file")
            filesize=$(du -h "$file" | cut -f1)
            echo "  $count) $filename ($filesize)"
        fi
    done

    if [ "$count" -eq 0 ]; then
        echo "❌ No GGUF binaries found inside $TARGET_DIR"
        exit 1
    fi

    while true; do
        printf "\nPlease choose a model [1-$count]: "
        read -r choice
        if echo "$choice" | egrep -q '^[0-9]+$' && [ "$choice" -ge 1 ] && [ "$choice" -le "$count" ]; then
            eval "MODEL_PATH=\${$choice}"
            return 0
        else
            echo "Invalid selection choice. Try again."
        fi
    done
}

# Key-Value Pair Lookup Database for Specific Model Names
# Maps to: ctx_size, cache_ram, ubatch_size, threads, threads_batch, temp
lookup_model_profile() {
    case "$1" in
        # --- Heavy Tiers (~10GB+ files) ---
        *DeepSeek-Coder-V2*Q5_K_M*|*DeepSeek-Coder-V2*Q8_0*|*Qwen2.5-Coder-14B*Q8_0*)
            echo "2048 512 512 4 4 0.3"
            ;;
        # --- Mid Tiers (~8.5GB - 10GB files) ---
        *Qwen2.5-Coder-14B*Q4_K_M*|*Qwen2.5-Coder-14B*Q5_K_M*|*Qwen3-14B*Q4_K_M*|*DeepSeek-Coder-V2*IQ4_XS*|*Qwen3.6-35B-A3B*UD-IQ3_S*|*Qwen3.6-35B-A3B*UD-IQ4_XS*)
            echo "8192 2048 512 4 4 0.3"
            ;;
        # --- Light Tiers (~5GB - 8GB files) ---
        *Qwen2.5-Coder-7B*Q8_0*|*DeepSeek-Coder-6.7B*Q8_0*|*Qwen3-8B*Q4_K_M*|*Qwen2.5-Coder-7B*Q5_K_M*|*Qwen2.5-Coder-7B-instruct*)
            echo "16384 4096 1024 6 4 0.3"
            ;;
        # --- Ultra Light Tiers (< 5GB files) ---
        *DeepSeek-Coder-6.7B*Q5_K_M*|*deepseek-coder:6.7b*|*Qwen3-4B*Q4_K_M*|*Qwen2.5-Coder-3B*|*DeepSeek-R1-Distill-Qwen-1.5B*|*deepseek-r1:1.5b*)
            echo "32768 8192 1024 6 4 0.3"
            ;;
        *)
            echo "NOT_FOUND"
            ;;
    esac
}

# Fallback Key-Value Pair Database using Size Category
# Maps to: ctx_size, cache_ram, ubatch_size, threads, threads_batch, temp
lookup_size_profile() {
    FILE_SIZE_GB=$1
    if [ "$FILE_SIZE_GB" -ge 10 ]; then
        echo "2048 512 512 4 4 0.3"
    elif [ "$FILE_SIZE_GB" -ge 8 ]; then
        echo "8192 2048 512 4 4 0.3"
    elif [ "$FILE_SIZE_GB" -ge 5 ]; then
        echo "16384 4096 1024 6 4 0.3"
    else
        echo "32768 8192 1024 6 4 0.3"
    fi
}

# Allocate context bounds: Try explicit key-value first, then size profile fallback
tune_vram_memory() {
    FILENAME=$(basename "$MODEL_PATH")
    echo "🔍 Analyzing model parameter mapping for: $FILENAME"

    # Step 1: Look up by model name key
    PROFILE_DATA=$(lookup_model_profile "$FILENAME")

    # Step 2: Fallback to lookup by size category key if name isn't matched
    if [ "$PROFILE_DATA" = "NOT_FOUND" ]; then
        echo "⚠️ No explicit profile name matched. Evaluating via size category key..."
        FILE_SIZE_KB=$(stat -f %z "$MODEL_PATH")
        FILE_SIZE_GB=$((FILE_SIZE_KB / 1024 / 1024 / 1024))
        echo "📊 Dynamic calculation: ${FILE_SIZE_GB} GB footprint discovered."
        PROFILE_DATA=$(lookup_size_profile "$FILE_SIZE_GB")
    else
        echo "🎯 Profile Match: Named model pair discovered."
    fi

    # Step 3: Extract values out of the selected key-value data list
    set -- $PROFILE_DATA
    CTX_SIZE=$1
    CACHE_RAM=$2
    UBATCH_SIZE=$3
    THREADS=$4
    THREADS_BATCH=$5
    TEMP=$6
}

# Execute server process loop
launch_llama_server() {
    echo "--------------------------------------------------------"
    echo "🚀 Launching Llama Server Instance"
    echo "📌 Context Size:   $CTX_SIZE"
    echo "📌 Cache RAM:     $CACHE_RAM"
    echo "📌 Threads:       $THREADS (Batch: $THREADS_BATCH)"
    echo "📌 Temperature:   $TEMP"
    echo "📌 Model:         $(basename "$MODEL_PATH")"
    echo "--------------------------------------------------------"

    llama-server \
            --ctx-size    "$CTX_SIZE" \
            --cache-ram   "$CACHE_RAM" \
            --ubatch-size "$UBATCH_SIZE" \
            --threads-batch "$THREADS_BATCH" \
            --threads     "$THREADS" \
            --host        "$HOST" \
            --port        "$PORT" \
            --temp        "$TEMP" \
            --model       "$MODEL_PATH" \
            --no-warmup \
            --parallel 1 &
           
    sleep 5
    firefox "http://${HOST}:${PORT}"
}

# ==========================================
# MAIN EXECUTION ROUTINE
# ==========================================
main() {
    kill_existing_server
    select_vram_tier
    select_model_file
    tune_vram_memory
    launch_llama_server
}

main

Note to self, never use code where you used A.I. help which , you don't fully understand (read don't become lazy) , and have not manually verified.
 
The associate launcher with downloader above,

Python:
cat myzsh
#!/bin/sh

# I was created using A.I.
# I Run on FreeBSD
# System Specs: 12GB VRAM (No Spillover), 32GB System RAM

# ==========================================
# CONFIGURATION & PARAMETERS
# ==========================================
BASE_DIR="/home/x/hugging"
HOST="127.0.0.1"
PORT="8123"

# ==========================================
# SEPARATE UTILITY FUNCTIONS
# ==========================================

# Cleanly terminate existing llama-server instances
kill_existing_server() {
    echo "Stopping any existing llama-server instances..."
    killall llama-server >/dev/null 2>&1
    sleep 2
    killall -9 llama-server >/dev/null 2>&1
    echo "Sleeping 6 seconds for port cleanup..."
    sleep 6
    echo "Done sleeping."
}

# Prompt user for the VRAM Tier directory
select_vram_tier() {
    while true; do
        printf "Select VRAM Tier Profile (8, 10, 12, 14): "
        read -r TIER
        case "$TIER" in
            8|10|12|14)
                TARGET_DIR="${BASE_DIR}/${TIER}"
                if [ ! -d "$TARGET_DIR" ]; then
                    echo "❌ Directory $TARGET_DIR does not exist."
                    continue
                fi
                return 0
                ;;
            *)
                echo "Invalid profile tier choice. Choose 8, 10, 12, or 14."
                ;;
        esac
    done
}

# Scan directory and choose an available model
select_model_file() {
    echo ""
    echo "Searching available models in: $TARGET_DIR"
    echo "--------------------------------------------------------"
  
    count=0
    set --
  
    for file in "$TARGET_DIR"/*.gguf; do
        if [ -f "$file" ]; then
            count=$((count + 1))
            set -- "$@" "$file"
            filename=$(basename "$file")
            filesize=$(du -h "$file" | cut -f1)
            echo "  $count) $filename ($filesize)"
        fi
    done

    if [ "$count" -eq 0 ]; then
        echo "❌ No GGUF binaries found inside $TARGET_DIR"
        exit 1
    fi

    while true; do
        printf "\nPlease choose a model [1-$count]: "
        read -r choice
        if echo "$choice" | egrep -q '^[0-9]+$' && [ "$choice" -ge 1 ] && [ "$choice" -le "$count" ]; then
            eval "MODEL_PATH=\${$choice}"
            return 0
        else
            echo "Invalid selection choice. Try again."
        fi
    done
}

# Key-Value Pair Lookup Database for Specific Model Names
# Maps to: ctx_size, cache_ram, ubatch_size, threads, threads_batch, temp
lookup_model_profile() {
    case "$1" in
        # --- Heavy Tiers (~10GB+ files) ---
        *DeepSeek-Coder-V2*Q5_K_M*|*DeepSeek-Coder-V2*Q8_0*|*Qwen2.5-Coder-14B*Q8_0*)
            echo "2048 512 512 4 4 0.3"
            ;;
        # --- Mid Tiers (~8.5GB - 10GB files) ---
        *Qwen2.5-Coder-14B*Q4_K_M*|*Qwen2.5-Coder-14B*Q5_K_M*|*Qwen3-14B*Q4_K_M*|*DeepSeek-Coder-V2*IQ4_XS*|*Qwen3.6-35B-A3B*UD-IQ3_S*|*Qwen3.6-35B-A3B*UD-IQ4_XS*)
            echo "8192 2048 512 4 4 0.3"
            ;;
        # --- Light Tiers (~5GB - 8GB files) ---
        *Qwen2.5-Coder-7B*Q8_0*|*DeepSeek-Coder-6.7B*Q8_0*|*Qwen3-8B*Q4_K_M*|*Qwen2.5-Coder-7B*Q5_K_M*|*Qwen2.5-Coder-7B-instruct*)
            echo "16384 4096 1024 6 4 0.3"
            ;;
        # --- Ultra Light Tiers (< 5GB files) ---
        *DeepSeek-Coder-6.7B*Q5_K_M*|*deepseek-coder:6.7b*|*Qwen3-4B*Q4_K_M*|*Qwen2.5-Coder-3B*|*DeepSeek-R1-Distill-Qwen-1.5B*|*deepseek-r1:1.5b*)
            echo "32768 8192 1024 6 4 0.3"
            ;;
        *)
            echo "NOT_FOUND"
            ;;
    esac
}

# Fallback Key-Value Pair Database using Size Category
# Maps to: ctx_size, cache_ram, ubatch_size, threads, threads_batch, temp
lookup_size_profile() {
    FILE_SIZE_GB=$1
    if [ "$FILE_SIZE_GB" -ge 10 ]; then
        echo "2048 512 512 4 4 0.3"
    elif [ "$FILE_SIZE_GB" -ge 8 ]; then
        echo "8192 2048 512 4 4 0.3"
    elif [ "$FILE_SIZE_GB" -ge 5 ]; then
        echo "16384 4096 1024 6 4 0.3"
    else
        echo "32768 8192 1024 6 4 0.3"
    fi
}

# Allocate context bounds: Try explicit key-value first, then size profile fallback
tune_vram_memory() {
    FILENAME=$(basename "$MODEL_PATH")
    echo "🔍 Analyzing model parameter mapping for: $FILENAME"

    # Step 1: Look up by model name key
    PROFILE_DATA=$(lookup_model_profile "$FILENAME")

    # Step 2: Fallback to lookup by size category key if name isn't matched
    if [ "$PROFILE_DATA" = "NOT_FOUND" ]; then
        echo "⚠️ No explicit profile name matched. Evaluating via size category key..."
        FILE_SIZE_KB=$(stat -f %z "$MODEL_PATH")
        FILE_SIZE_GB=$((FILE_SIZE_KB / 1024 / 1024 / 1024))
        echo "📊 Dynamic calculation: ${FILE_SIZE_GB} GB footprint discovered."
        PROFILE_DATA=$(lookup_size_profile "$FILE_SIZE_GB")
    else
        echo "🎯 Profile Match: Named model pair discovered."
    fi

    # Step 3: Extract values out of the selected key-value data list
    set -- $PROFILE_DATA
    CTX_SIZE=$1
    CACHE_RAM=$2
    UBATCH_SIZE=$3
    THREADS=$4
    THREADS_BATCH=$5
    TEMP=$6
}

# Execute server process loop
launch_llama_server() {
    echo "--------------------------------------------------------"
    echo "🚀 Launching Llama Server Instance"
    echo "📌 Context Size:   $CTX_SIZE"
    echo "📌 Cache RAM:     $CACHE_RAM"
    echo "📌 Threads:       $THREADS (Batch: $THREADS_BATCH)"
    echo "📌 Temperature:   $TEMP"
    echo "📌 Model:         $(basename "$MODEL_PATH")"
    echo "--------------------------------------------------------"

    llama-server \
            --ctx-size    "$CTX_SIZE" \
            --cache-ram   "$CACHE_RAM" \
            --ubatch-size "$UBATCH_SIZE" \
            --threads-batch "$THREADS_BATCH" \
            --threads     "$THREADS" \
            --host        "$HOST" \
            --port        "$PORT" \
            --temp        "$TEMP" \
            --model       "$MODEL_PATH" \
            --no-warmup \
            --parallel 1 &
          
    sleep 5
    firefox "http://${HOST}:${PORT}"
}

# ==========================================
# MAIN EXECUTION ROUTINE
# ==========================================
main() {
    kill_existing_server
    select_vram_tier
    select_model_file
    tune_vram_memory
    launch_llama_server
}

main

Note to self, never use code where you used A.I. help which , you don't fully understand (read don't become lazy) , and have not manually verified.
Noob question.
Is there some nice Debug Tool for shell scripts? Like you go line by line manually and it shows you the output (simulated preferably)?
 
Back
Top