Running a GPU-accelerated LLM on FreeBSD (2-line howto)

The associate launcher with downloader above,

Python:
cat myzsh
#!/bin/sh

# I was created using A.I.
# I Run on FreeBSD
# System Specs: 12GB VRAM (No Spillover), 32GB System RAM

# ==========================================
# CONFIGURATION & PARAMETERS
# ==========================================
BASE_DIR="/home/x/hugging"
HOST="127.0.0.1"
PORT="8123"

# ==========================================
# SEPARATE UTILITY FUNCTIONS
# ==========================================

# Cleanly terminate existing llama-server instances
kill_existing_server() {
    echo "Stopping any existing llama-server instances..."
    killall llama-server >/dev/null 2>&1
    sleep 2
    killall -9 llama-server >/dev/null 2>&1
    echo "Sleeping 6 seconds for port cleanup..."
    sleep 6
    echo "Done sleeping."
}

# Prompt user for the VRAM Tier directory
select_vram_tier() {
    while true; do
        printf "Select VRAM Tier Profile (8, 10, 12, 14): "
        read -r TIER
        case "$TIER" in
            8|10|12|14)
                TARGET_DIR="${BASE_DIR}/${TIER}"
                if [ ! -d "$TARGET_DIR" ]; then
                    echo "❌ Directory $TARGET_DIR does not exist."
                    continue
                fi
                return 0
                ;;
            *)
                echo "Invalid profile tier choice. Choose 8, 10, 12, or 14."
                ;;
        esac
    done
}

# Scan directory and choose an available model
select_model_file() {
    echo ""
    echo "Searching available models in: $TARGET_DIR"
    echo "--------------------------------------------------------"
  
    count=0
    set --
  
    for file in "$TARGET_DIR"/*.gguf; do
        if [ -f "$file" ]; then
            count=$((count + 1))
            set -- "$@" "$file"
            filename=$(basename "$file")
            filesize=$(du -h "$file" | cut -f1)
            echo "  $count) $filename ($filesize)"
        fi
    done

    if [ "$count" -eq 0 ]; then
        echo "❌ No GGUF binaries found inside $TARGET_DIR"
        exit 1
    fi

    while true; do
        printf "\nPlease choose a model [1-$count]: "
        read -r choice
        if echo "$choice" | egrep -q '^[0-9]+$' && [ "$choice" -ge 1 ] && [ "$choice" -le "$count" ]; then
            eval "MODEL_PATH=\${$choice}"
            return 0
        else
            echo "Invalid selection choice. Try again."
        fi
    done
}

# Key-Value Pair Lookup Database for Specific Model Names
# Maps to: ctx_size, cache_ram, ubatch_size, threads, threads_batch, temp
lookup_model_profile() {
    case "$1" in
        # --- Heavy Tiers (~10GB+ files) ---
        *DeepSeek-Coder-V2*Q5_K_M*|*DeepSeek-Coder-V2*Q8_0*|*Qwen2.5-Coder-14B*Q8_0*)
            echo "2048 512 512 4 4 0.3"
            ;;
        # --- Mid Tiers (~8.5GB - 10GB files) ---
        *Qwen2.5-Coder-14B*Q4_K_M*|*Qwen2.5-Coder-14B*Q5_K_M*|*Qwen3-14B*Q4_K_M*|*DeepSeek-Coder-V2*IQ4_XS*|*Qwen3.6-35B-A3B*UD-IQ3_S*|*Qwen3.6-35B-A3B*UD-IQ4_XS*)
            echo "8192 2048 512 4 4 0.3"
            ;;
        # --- Light Tiers (~5GB - 8GB files) ---
        *Qwen2.5-Coder-7B*Q8_0*|*DeepSeek-Coder-6.7B*Q8_0*|*Qwen3-8B*Q4_K_M*|*Qwen2.5-Coder-7B*Q5_K_M*|*Qwen2.5-Coder-7B-instruct*)
            echo "16384 4096 1024 6 4 0.3"
            ;;
        # --- Ultra Light Tiers (< 5GB files) ---
        *DeepSeek-Coder-6.7B*Q5_K_M*|*deepseek-coder:6.7b*|*Qwen3-4B*Q4_K_M*|*Qwen2.5-Coder-3B*|*DeepSeek-R1-Distill-Qwen-1.5B*|*deepseek-r1:1.5b*)
            echo "32768 8192 1024 6 4 0.3"
            ;;
        *)
            echo "NOT_FOUND"
            ;;
    esac
}

# Fallback Key-Value Pair Database using Size Category
# Maps to: ctx_size, cache_ram, ubatch_size, threads, threads_batch, temp
lookup_size_profile() {
    FILE_SIZE_GB=$1
    if [ "$FILE_SIZE_GB" -ge 10 ]; then
        echo "2048 512 512 4 4 0.3"
    elif [ "$FILE_SIZE_GB" -ge 8 ]; then
        echo "8192 2048 512 4 4 0.3"
    elif [ "$FILE_SIZE_GB" -ge 5 ]; then
        echo "16384 4096 1024 6 4 0.3"
    else
        echo "32768 8192 1024 6 4 0.3"
    fi
}

# Allocate context bounds: Try explicit key-value first, then size profile fallback
tune_vram_memory() {
    FILENAME=$(basename "$MODEL_PATH")
    echo "🔍 Analyzing model parameter mapping for: $FILENAME"

    # Step 1: Look up by model name key
    PROFILE_DATA=$(lookup_model_profile "$FILENAME")

    # Step 2: Fallback to lookup by size category key if name isn't matched
    if [ "$PROFILE_DATA" = "NOT_FOUND" ]; then
        echo "⚠️ No explicit profile name matched. Evaluating via size category key..."
        FILE_SIZE_KB=$(stat -f %z "$MODEL_PATH")
        FILE_SIZE_GB=$((FILE_SIZE_KB / 1024 / 1024 / 1024))
        echo "📊 Dynamic calculation: ${FILE_SIZE_GB} GB footprint discovered."
        PROFILE_DATA=$(lookup_size_profile "$FILE_SIZE_GB")
    else
        echo "🎯 Profile Match: Named model pair discovered."
    fi

    # Step 3: Extract values out of the selected key-value data list
    set -- $PROFILE_DATA
    CTX_SIZE=$1
    CACHE_RAM=$2
    UBATCH_SIZE=$3
    THREADS=$4
    THREADS_BATCH=$5
    TEMP=$6
}

# Execute server process loop
launch_llama_server() {
    echo "--------------------------------------------------------"
    echo "🚀 Launching Llama Server Instance"
    echo "📌 Context Size:   $CTX_SIZE"
    echo "📌 Cache RAM:     $CACHE_RAM"
    echo "📌 Threads:       $THREADS (Batch: $THREADS_BATCH)"
    echo "📌 Temperature:   $TEMP"
    echo "📌 Model:         $(basename "$MODEL_PATH")"
    echo "--------------------------------------------------------"

    llama-server \
            --ctx-size    "$CTX_SIZE" \
            --cache-ram   "$CACHE_RAM" \
            --ubatch-size "$UBATCH_SIZE" \
            --threads-batch "$THREADS_BATCH" \
            --threads     "$THREADS" \
            --host        "$HOST" \
            --port        "$PORT" \
            --temp        "$TEMP" \
            --model       "$MODEL_PATH" \
            --no-warmup \
            --parallel 1 &
          
    sleep 5
    firefox "http://${HOST}:${PORT}"
}

# ==========================================
# MAIN EXECUTION ROUTINE
# ==========================================
main() {
    kill_existing_server
    select_vram_tier
    select_model_file
    tune_vram_memory
    launch_llama_server
}

main

Note to self, never use code where you used A.I. help which , you don't fully understand (read don't become lazy) , and have not manually verified.
Noob question.
Is there some nice Debug Tool for shell scripts? Like you go line by line manually and it shows you the output (simulated preferably)?
 
@cracauer, you might like this one-liner for 12GB VRAM (no spill)

(2,048 tokens)
Code:
llama-server --ctx-size 2048 --ubatch-size 128 --threads-batch 4 --threads 4 --host 127.0.0.1 --port 8123 --temp 0.3 --model /home/x/hugging/12/Qwen3.6-35B-A3B-UD-IQ3_S.gguf --no-warmup --jinja --parallel 1 --no-mmap --fit off --n-gpu-layers all --cache-ram 4096 --flash-attn on --batch-size 128 --cpu-moe

When i need more tokens i simply do,
Code:
ollama run deepseek-coder-v2:16b-lite-instruct-q5_K_M
 
Yes, controlling the context size is important. Think about what you need when picking an LLM for your video card. If you need more context size you might have to use a smaller model on the same video card.

Alain, why do you use no-mmap?
 
For those that want to try AI models locally with Linuxlator and CUDA on FreeBSD have a look at cracauer@ 's other thread at this link from post #30 and upward.

I have put 11 scripts there over posts #32-#35 that does this:


mainmenu.png
modelslist.png

contextoptions.png
aiderclaude.png


/grandpa
 
first thanks for toturial
second

8GB VRAM (RTX 4060)
32GB DDR4 RAM

i want fastest model i can find find for mcp use (should be very good at this) and computer use (not coding)

qwen 35b-a3b is nice at speed but stupid at computer use
I think there is a nemotron variant that fits 8 GB.
 
Code:
 find /home/x/hugging | grep gguf$
/home/x/hugging/Qwen2.5-32B-Instruct-Q4_K_M.gguf
/home/x/hugging/8/Qwen2.5-Coder-7B-Instruct-Q5_K_M.gguf
/home/x/hugging/8/DeepSeek-R1-Distill-Qwen-1.5B-Q8_0.gguf
/home/x/hugging/8/Qwen2.5-Coder-7B-Instruct-Q4_K_M.gguf
/home/x/hugging/10/Qwen2.5-Coder-14B-Instruct-Q4_K_M.gguf
/home/x/hugging/10/Qwen2.5-Coder-7B-Instruct-Q8_0.gguf
/home/x/hugging/14/Qwen3.6-35B-A3B-UD-IQ4_XS.gguf
/home/x/hugging/14/Qwen2.5-Coder-14B-Instruct-Q5_K_M.gguf
/home/x/hugging/Qwen3.6-27B-Fable-Fus-711-UnHeretic-NM-DAU-NEO-MAX-NEO-MTP-Q4_K_M.gguf
/home/x/hugging/16/Qwen3.6-35B-A3B-UD-IQ4_XS.gguf
/home/x/hugging/16/Qwen3.8-27B-UD-IQ2_S.gguf
/home/x/hugging/16/Qwen2.5-Coder-14B-Instruct-Q4_K_M.gguf
/home/x/hugging/16/Qwen2.5-Coder-7B-Instruct-Q8_0.gguf
/home/x/hugging/16/Qwen2.5-Coder-14B-Instruct-Q5_K_M.gguf
/home/x/hugging/16/Qwen2.5-Coder-7B-Instruct-Q6_K.gguf
/home/x/hugging/16/Qwen3.8-27B-UD-IQ3_S.gguf
/home/x/hugging/12/Qwen2.5-Coder-7B-Instruct-Q8_0.gguf
/home/x/hugging/12/Qwen2.5-Coder-14B-Instruct-Q4_K_M.gguf
/home/x/hugging/12/Qwen3.6-35B-A3B-UD-IQ3_S.gguf
x@myfreebsd:/mnt/MYWINZFS/data $ ls | grep gguf
Qwen3.6-27B-Fable-Fus-711-UnHeretic-NM-DAU-NEO-MAX-NEO-MTP-Q4_K_M.gguf
Qwen3.6-35B-A3B-MTP-UD-Q4_K_XL.gguf
Qwen3.6-35B-A3B-UD-Q4_K_XL.gguf
Qwen3.6-35B-A3B-UD-Q4_K_XL.gguf.aria2
 
For a rough and crude guide for how to get as much as possible out of a model take a look at this table in cracauer@ 's other thread.

Basically if you want to run a model with high compression, like IQ2, then aim for a model with 70B+ parameters. Supposedly the number of parameters trained will compensate for the level of quantization.

Another example - for a model with Q8_0 then 1-8B parameters is fine.

/grandpa
 
Back
Top