Running a GPU-accelerated LLM on FreeBSD (2-line howto)

The associate launcher with downloader above,

Python:
cat myzsh
#!/bin/sh

# I was created using A.I.
# I Run on FreeBSD
# System Specs: 12GB VRAM (No Spillover), 32GB System RAM

# ==========================================
# CONFIGURATION & PARAMETERS
# ==========================================
BASE_DIR="/home/x/hugging"
HOST="127.0.0.1"
PORT="8123"

# ==========================================
# SEPARATE UTILITY FUNCTIONS
# ==========================================

# Cleanly terminate existing llama-server instances
kill_existing_server() {
    echo "Stopping any existing llama-server instances..."
    killall llama-server >/dev/null 2>&1
    sleep 2
    killall -9 llama-server >/dev/null 2>&1
    echo "Sleeping 6 seconds for port cleanup..."
    sleep 6
    echo "Done sleeping."
}

# Prompt user for the VRAM Tier directory
select_vram_tier() {
    while true; do
        printf "Select VRAM Tier Profile (8, 10, 12, 14): "
        read -r TIER
        case "$TIER" in
            8|10|12|14)
                TARGET_DIR="${BASE_DIR}/${TIER}"
                if [ ! -d "$TARGET_DIR" ]; then
                    echo "❌ Directory $TARGET_DIR does not exist."
                    continue
                fi
                return 0
                ;;
            *)
                echo "Invalid profile tier choice. Choose 8, 10, 12, or 14."
                ;;
        esac
    done
}

# Scan directory and choose an available model
select_model_file() {
    echo ""
    echo "Searching available models in: $TARGET_DIR"
    echo "--------------------------------------------------------"
   
    count=0
    set --
   
    for file in "$TARGET_DIR"/*.gguf; do
        if [ -f "$file" ]; then
            count=$((count + 1))
            set -- "$@" "$file"
            filename=$(basename "$file")
            filesize=$(du -h "$file" | cut -f1)
            echo "  $count) $filename ($filesize)"
        fi
    done

    if [ "$count" -eq 0 ]; then
        echo "❌ No GGUF binaries found inside $TARGET_DIR"
        exit 1
    fi

    while true; do
        printf "\nPlease choose a model [1-$count]: "
        read -r choice
        if echo "$choice" | egrep -q '^[0-9]+$' && [ "$choice" -ge 1 ] && [ "$choice" -le "$count" ]; then
            eval "MODEL_PATH=\${$choice}"
            return 0
        else
            echo "Invalid selection choice. Try again."
        fi
    done
}

# Key-Value Pair Lookup Database for Specific Model Names
# Maps to: ctx_size, cache_ram, ubatch_size, threads, threads_batch, temp
lookup_model_profile() {
    case "$1" in
        # --- Heavy Tiers (~10GB+ files) ---
        *DeepSeek-Coder-V2*Q5_K_M*|*DeepSeek-Coder-V2*Q8_0*|*Qwen2.5-Coder-14B*Q8_0*)
            echo "2048 512 512 4 4 0.3"
            ;;
        # --- Mid Tiers (~8.5GB - 10GB files) ---
        *Qwen2.5-Coder-14B*Q4_K_M*|*Qwen2.5-Coder-14B*Q5_K_M*|*Qwen3-14B*Q4_K_M*|*DeepSeek-Coder-V2*IQ4_XS*|*Qwen3.6-35B-A3B*UD-IQ3_S*|*Qwen3.6-35B-A3B*UD-IQ4_XS*)
            echo "8192 2048 512 4 4 0.3"
            ;;
        # --- Light Tiers (~5GB - 8GB files) ---
        *Qwen2.5-Coder-7B*Q8_0*|*DeepSeek-Coder-6.7B*Q8_0*|*Qwen3-8B*Q4_K_M*|*Qwen2.5-Coder-7B*Q5_K_M*|*Qwen2.5-Coder-7B-instruct*)
            echo "16384 4096 1024 6 4 0.3"
            ;;
        # --- Ultra Light Tiers (< 5GB files) ---
        *DeepSeek-Coder-6.7B*Q5_K_M*|*deepseek-coder:6.7b*|*Qwen3-4B*Q4_K_M*|*Qwen2.5-Coder-3B*|*DeepSeek-R1-Distill-Qwen-1.5B*|*deepseek-r1:1.5b*)
            echo "32768 8192 1024 6 4 0.3"
            ;;
        *)
            echo "NOT_FOUND"
            ;;
    esac
}

# Fallback Key-Value Pair Database using Size Category
# Maps to: ctx_size, cache_ram, ubatch_size, threads, threads_batch, temp
lookup_size_profile() {
    FILE_SIZE_GB=$1
    if [ "$FILE_SIZE_GB" -ge 10 ]; then
        echo "2048 512 512 4 4 0.3"
    elif [ "$FILE_SIZE_GB" -ge 8 ]; then
        echo "8192 2048 512 4 4 0.3"
    elif [ "$FILE_SIZE_GB" -ge 5 ]; then
        echo "16384 4096 1024 6 4 0.3"
    else
        echo "32768 8192 1024 6 4 0.3"
    fi
}

# Allocate context bounds: Try explicit key-value first, then size profile fallback
tune_vram_memory() {
    FILENAME=$(basename "$MODEL_PATH")
    echo "🔍 Analyzing model parameter mapping for: $FILENAME"

    # Step 1: Look up by model name key
    PROFILE_DATA=$(lookup_model_profile "$FILENAME")

    # Step 2: Fallback to lookup by size category key if name isn't matched
    if [ "$PROFILE_DATA" = "NOT_FOUND" ]; then
        echo "⚠️ No explicit profile name matched. Evaluating via size category key..."
        FILE_SIZE_KB=$(stat -f %z "$MODEL_PATH")
        FILE_SIZE_GB=$((FILE_SIZE_KB / 1024 / 1024 / 1024))
        echo "📊 Dynamic calculation: ${FILE_SIZE_GB} GB footprint discovered."
        PROFILE_DATA=$(lookup_size_profile "$FILE_SIZE_GB")
    else
        echo "🎯 Profile Match: Named model pair discovered."
    fi

    # Step 3: Extract values out of the selected key-value data list
    set -- $PROFILE_DATA
    CTX_SIZE=$1
    CACHE_RAM=$2
    UBATCH_SIZE=$3
    THREADS=$4
    THREADS_BATCH=$5
    TEMP=$6
}

# Execute server process loop
launch_llama_server() {
    echo "--------------------------------------------------------"
    echo "🚀 Launching Llama Server Instance"
    echo "📌 Context Size:   $CTX_SIZE"
    echo "📌 Cache RAM:     $CACHE_RAM"
    echo "📌 Threads:       $THREADS (Batch: $THREADS_BATCH)"
    echo "📌 Temperature:   $TEMP"
    echo "📌 Model:         $(basename "$MODEL_PATH")"
    echo "--------------------------------------------------------"

    llama-server \
            --ctx-size    "$CTX_SIZE" \
            --cache-ram   "$CACHE_RAM" \
            --ubatch-size "$UBATCH_SIZE" \
            --threads-batch "$THREADS_BATCH" \
            --threads     "$THREADS" \
            --host        "$HOST" \
            --port        "$PORT" \
            --temp        "$TEMP" \
            --model       "$MODEL_PATH" \
            --no-warmup \
            --parallel 1 &
           
    sleep 5
    firefox "http://${HOST}:${PORT}"
}

# ==========================================
# MAIN EXECUTION ROUTINE
# ==========================================
main() {
    kill_existing_server
    select_vram_tier
    select_model_file
    tune_vram_memory
    launch_llama_server
}

main

Note to self, never use code where you used A.I. help which , you don't fully understand (read don't become lazy) , and have not manually verified.
 
The associate launcher with downloader above,

Python:
cat myzsh
#!/bin/sh

# I was created using A.I.
# I Run on FreeBSD
# System Specs: 12GB VRAM (No Spillover), 32GB System RAM

# ==========================================
# CONFIGURATION & PARAMETERS
# ==========================================
BASE_DIR="/home/x/hugging"
HOST="127.0.0.1"
PORT="8123"

# ==========================================
# SEPARATE UTILITY FUNCTIONS
# ==========================================

# Cleanly terminate existing llama-server instances
kill_existing_server() {
    echo "Stopping any existing llama-server instances..."
    killall llama-server >/dev/null 2>&1
    sleep 2
    killall -9 llama-server >/dev/null 2>&1
    echo "Sleeping 6 seconds for port cleanup..."
    sleep 6
    echo "Done sleeping."
}

# Prompt user for the VRAM Tier directory
select_vram_tier() {
    while true; do
        printf "Select VRAM Tier Profile (8, 10, 12, 14): "
        read -r TIER
        case "$TIER" in
            8|10|12|14)
                TARGET_DIR="${BASE_DIR}/${TIER}"
                if [ ! -d "$TARGET_DIR" ]; then
                    echo "❌ Directory $TARGET_DIR does not exist."
                    continue
                fi
                return 0
                ;;
            *)
                echo "Invalid profile tier choice. Choose 8, 10, 12, or 14."
                ;;
        esac
    done
}

# Scan directory and choose an available model
select_model_file() {
    echo ""
    echo "Searching available models in: $TARGET_DIR"
    echo "--------------------------------------------------------"
  
    count=0
    set --
  
    for file in "$TARGET_DIR"/*.gguf; do
        if [ -f "$file" ]; then
            count=$((count + 1))
            set -- "$@" "$file"
            filename=$(basename "$file")
            filesize=$(du -h "$file" | cut -f1)
            echo "  $count) $filename ($filesize)"
        fi
    done

    if [ "$count" -eq 0 ]; then
        echo "❌ No GGUF binaries found inside $TARGET_DIR"
        exit 1
    fi

    while true; do
        printf "\nPlease choose a model [1-$count]: "
        read -r choice
        if echo "$choice" | egrep -q '^[0-9]+$' && [ "$choice" -ge 1 ] && [ "$choice" -le "$count" ]; then
            eval "MODEL_PATH=\${$choice}"
            return 0
        else
            echo "Invalid selection choice. Try again."
        fi
    done
}

# Key-Value Pair Lookup Database for Specific Model Names
# Maps to: ctx_size, cache_ram, ubatch_size, threads, threads_batch, temp
lookup_model_profile() {
    case "$1" in
        # --- Heavy Tiers (~10GB+ files) ---
        *DeepSeek-Coder-V2*Q5_K_M*|*DeepSeek-Coder-V2*Q8_0*|*Qwen2.5-Coder-14B*Q8_0*)
            echo "2048 512 512 4 4 0.3"
            ;;
        # --- Mid Tiers (~8.5GB - 10GB files) ---
        *Qwen2.5-Coder-14B*Q4_K_M*|*Qwen2.5-Coder-14B*Q5_K_M*|*Qwen3-14B*Q4_K_M*|*DeepSeek-Coder-V2*IQ4_XS*|*Qwen3.6-35B-A3B*UD-IQ3_S*|*Qwen3.6-35B-A3B*UD-IQ4_XS*)
            echo "8192 2048 512 4 4 0.3"
            ;;
        # --- Light Tiers (~5GB - 8GB files) ---
        *Qwen2.5-Coder-7B*Q8_0*|*DeepSeek-Coder-6.7B*Q8_0*|*Qwen3-8B*Q4_K_M*|*Qwen2.5-Coder-7B*Q5_K_M*|*Qwen2.5-Coder-7B-instruct*)
            echo "16384 4096 1024 6 4 0.3"
            ;;
        # --- Ultra Light Tiers (< 5GB files) ---
        *DeepSeek-Coder-6.7B*Q5_K_M*|*deepseek-coder:6.7b*|*Qwen3-4B*Q4_K_M*|*Qwen2.5-Coder-3B*|*DeepSeek-R1-Distill-Qwen-1.5B*|*deepseek-r1:1.5b*)
            echo "32768 8192 1024 6 4 0.3"
            ;;
        *)
            echo "NOT_FOUND"
            ;;
    esac
}

# Fallback Key-Value Pair Database using Size Category
# Maps to: ctx_size, cache_ram, ubatch_size, threads, threads_batch, temp
lookup_size_profile() {
    FILE_SIZE_GB=$1
    if [ "$FILE_SIZE_GB" -ge 10 ]; then
        echo "2048 512 512 4 4 0.3"
    elif [ "$FILE_SIZE_GB" -ge 8 ]; then
        echo "8192 2048 512 4 4 0.3"
    elif [ "$FILE_SIZE_GB" -ge 5 ]; then
        echo "16384 4096 1024 6 4 0.3"
    else
        echo "32768 8192 1024 6 4 0.3"
    fi
}

# Allocate context bounds: Try explicit key-value first, then size profile fallback
tune_vram_memory() {
    FILENAME=$(basename "$MODEL_PATH")
    echo "🔍 Analyzing model parameter mapping for: $FILENAME"

    # Step 1: Look up by model name key
    PROFILE_DATA=$(lookup_model_profile "$FILENAME")

    # Step 2: Fallback to lookup by size category key if name isn't matched
    if [ "$PROFILE_DATA" = "NOT_FOUND" ]; then
        echo "⚠️ No explicit profile name matched. Evaluating via size category key..."
        FILE_SIZE_KB=$(stat -f %z "$MODEL_PATH")
        FILE_SIZE_GB=$((FILE_SIZE_KB / 1024 / 1024 / 1024))
        echo "📊 Dynamic calculation: ${FILE_SIZE_GB} GB footprint discovered."
        PROFILE_DATA=$(lookup_size_profile "$FILE_SIZE_GB")
    else
        echo "🎯 Profile Match: Named model pair discovered."
    fi

    # Step 3: Extract values out of the selected key-value data list
    set -- $PROFILE_DATA
    CTX_SIZE=$1
    CACHE_RAM=$2
    UBATCH_SIZE=$3
    THREADS=$4
    THREADS_BATCH=$5
    TEMP=$6
}

# Execute server process loop
launch_llama_server() {
    echo "--------------------------------------------------------"
    echo "🚀 Launching Llama Server Instance"
    echo "📌 Context Size:   $CTX_SIZE"
    echo "📌 Cache RAM:     $CACHE_RAM"
    echo "📌 Threads:       $THREADS (Batch: $THREADS_BATCH)"
    echo "📌 Temperature:   $TEMP"
    echo "📌 Model:         $(basename "$MODEL_PATH")"
    echo "--------------------------------------------------------"

    llama-server \
            --ctx-size    "$CTX_SIZE" \
            --cache-ram   "$CACHE_RAM" \
            --ubatch-size "$UBATCH_SIZE" \
            --threads-batch "$THREADS_BATCH" \
            --threads     "$THREADS" \
            --host        "$HOST" \
            --port        "$PORT" \
            --temp        "$TEMP" \
            --model       "$MODEL_PATH" \
            --no-warmup \
            --parallel 1 &
          
    sleep 5
    firefox "http://${HOST}:${PORT}"
}

# ==========================================
# MAIN EXECUTION ROUTINE
# ==========================================
main() {
    kill_existing_server
    select_vram_tier
    select_model_file
    tune_vram_memory
    launch_llama_server
}

main

Note to self, never use code where you used A.I. help which , you don't fully understand (read don't become lazy) , and have not manually verified.
Noob question.
Is there some nice Debug Tool for shell scripts? Like you go line by line manually and it shows you the output (simulated preferably)?
 
@cracauer, you might like this one-liner for 12GB VRAM (no spill)

(2,048 tokens)
Code:
llama-server --ctx-size 2048 --ubatch-size 128 --threads-batch 4 --threads 4 --host 127.0.0.1 --port 8123 --temp 0.3 --model /home/x/hugging/12/Qwen3.6-35B-A3B-UD-IQ3_S.gguf --no-warmup --jinja --parallel 1 --no-mmap --fit off --n-gpu-layers all --cache-ram 4096 --flash-attn on --batch-size 128 --cpu-moe

When i need more tokens i simply do,
Code:
ollama run deepseek-coder-v2:16b-lite-instruct-q5_K_M
 
Yes, controlling the context size is important. Think about what you need when picking an LLM for your video card. If you need more context size you might have to use a smaller model on the same video card.

Alain, why do you use no-mmap?
 
no-mmap no even a "copy in ram".

My latest launcher script , it explains itself,
cat Main.scala
Code:
import cats.effect.{IO, IOApp, Resource}
import scala.sys.process._
import scala.util.control.NonFatal
import java.io.{FileWriter, PrintWriter, BufferedReader, InputStreamReader, IOException}
import scala.concurrent.duration._
import java.time.LocalDateTime
import java.time.format.DateTimeFormatter

object Main extends IOApp.Simple {

  val myport = "8123"
  val myhost = "127.0.0.1"
  val myprograms: Vector[(String, Vector[String])] = Vector(
    "chrome" -> Vector(s"http://$myhost:$myport")
  )

  // Base path tracking where your model subfolders are stored
  val myhuggingpath: String = "/home/x/hugging"
  // Custom conditional parameters injected dynamically into the configuration baseline
  val DYNAMIC_MMAP_PARAM = " --no-mmap "
  val VRAM_MANAGEMENT_PARAMS = " --fit off "
  val STABILITY_PARAMS = " --cache-ram 4096 --kv-unified "
  val EXTRA_SERVER_PARAMS = " --no-warmup --parallel 1 "

  // batch -> 1024 ; threads -> 10 
  // Baseline parameters shared across model targets
  val myidentical: String =
    s""" --ctx-size   32768
         --batch-size  1024
         --ubatch-size 1024
         --threads-batch 10
         --threads       10
         --parallel       1
         --n-gpu-layers all
         --host     $myhost
         --port     $myport
         --no-warmup
         --jinja
         --verbose
         $DYNAMIC_MMAP_PARAM $VRAM_MANAGEMENT_PARAMS $STABILITY_PARAMS $EXTRA_SERVER_PARAMS 
        """

  // Cleaned collection mapping incorporating path interpolation for file lookups
  val mymodels: Vector[(String, String)] = Vector(
    "DeepSeek-R1-Distill-Qwen-1.5B-Q8_0" -> s" --temp 0.3 --model $myhuggingpath/8/DeepSeek-R1-Distill-Qwen-1.5B-Q8_0.gguf",
    "Qwen2.5-Coder-7B-Instruct-Q4_K_M"  ->  s" --temp 0.3 --model $myhuggingpath/8/Qwen2.5-Coder-7B-Instruct-Q4_K_M.gguf",
    "Qwen2.5-Coder-7B-Instruct-Q5_K_M"  ->  s" --temp 0.3 --model $myhuggingpath/8/Qwen2.5-Coder-7B-Instruct-Q5_K_M.gguf",
    "Qwen3.6-35B-A3B-UD-IQ3_S"          ->  s" --temp 0.3 --cpu-moe --model $myhuggingpath/12/Qwen3.6-35B-A3B-UD-IQ3_S.gguf",
    "Qwen3.6-35B-A3B-UD-IQ4_XS"         ->  
    s""" --spec-draft-p-min 0.75 --spec-draft-n-max 2 --spec-type draft-mtp --reasoning on --reasoning-budget -1 --chat-template-kwargs "{\\"preserve_thinking\\":true}" --n-gpu-layers 40 --temp 0.6 --presence-penalty 0.0 --repeat-penalty 1.0 --cpu-moe --model $myhuggingpath/16/Qwen3.6-35B-A3B-UD-IQ4_XS.gguf """,
    // s" --temp 0.3 --cpu-moe --model $myhuggingpath/16/Qwen3.6-35B-A3B-UD-IQ4_XS.gguf ",
    "Qwen3.8-27B-UD-IQ3_S"              ->  s" --temp 0.3 --cpu-moe --model $myhuggingpath/16/Qwen3.8-27B-UD-IQ3_S.gguf "
  )

  def isBinaryAvailable(binary: String): Boolean = {
    try {
      Seq("which", binary).!!
      true
    } catch {
      case _: Exception => false
    }
  }

  // Purely functional Resource management for the file logging handle
  def createLogWriter: Resource[IO, PrintWriter] = {
    Resource.make(IO.blocking(new PrintWriter(new FileWriter("mynewlog.txt", true))))(writer => 
      IO.blocking(writer.close())
    )
  }



  def launchProcessInBackground(binaryName: String, argumentTokens: Seq[String], logLabel: String, printWriter: PrintWriter): IO[Process] = IO.blocking {
    if (!isBinaryAvailable(binaryName)) {
      throw new java.io.FileNotFoundException(s"Binary '$binaryName' not found in PATH.")
    }

    val commandArguments = binaryName +: argumentTokens
    println(s"[$logLabel Debug] Spawning command: ${commandArguments.mkString(" ")}")
  
    val processBuilder = Process(commandArguments)
    val processIO = new ProcessIO(
      stdin => stdin.close(),
      stdout => {
        val reader = new BufferedReader(new InputStreamReader(stdout))
        try {
          var line: String = null
          while ({ line = reader.readLine(); line != null }) {
            printWriter.synchronized {
              printWriter.println(s"[$logLabel stdout] $line")
              printWriter.flush()
            }
          }
        } catch { case _: IOException => } finally { reader.close() }
      },
      stderr => {
        val reader = new BufferedReader(new InputStreamReader(stderr))
        try {
          var line: String = null
          while ({ line = reader.readLine(); line != null }) {
            printWriter.synchronized {
              printWriter.println(s"[$logLabel stderr] $line")
              printWriter.flush()
            }
          }
        } catch { case _: IOException => } finally { reader.close() }
      },
      // 💡 Pass true as the 4th parameter to flag all background I/O pumps as daemon threads
      daemonizeThreads = true
    )
    processBuilder.run(processIO)
  }

  def launchLlamaServerInBackground(uniqueParams: String, writer: PrintWriter): IO[Process] = {
    val fullParamString = s"$uniqueParams $myidentical"
    val tokens = fullParamString.trim.split("\\s+").toIndexedSeq
    launchProcessInBackground("llama-server", tokens, "Llama Server", writer)
  }

  def launchChromeInBackground(writer: PrintWriter): IO[Process] = {
    val chromeConfigOption = myprograms.find(p => p._1 == "chrome")
    val argumentsVector = chromeConfigOption.map(p => p._2).getOrElse(Vector.empty)
    launchProcessInBackground("chrome", argumentsVector, "Chrome", writer)
  }

  val displayMenu: IO[Unit] = IO.defer {
    val menuLines = mymodels.zipWithIndex.map { case ((name, _), index) =>
      s"${index + 1}. $name"
    }.mkString("\n")
    
    IO.println("\n=== Available Models ===") *>
    IO.println(menuLines) *>
    IO.print("\nSelect a model (1-6): ")
  }

  val handleSelection: IO[String] = for {
    _     <- displayMenu
    input <- IO.readLine
    params <- scala.util.Try(input.trim.toInt).toOption match {
      case Some(choice) if choice >= 1 && choice <= mymodels.length =>
        val (modelName, modelParams) = mymodels(choice - 1)
        IO.println(s"\nSelected Model: $modelName").as(modelParams)
      case _ =>
        IO.println("\nInvalid choice selection. Exiting process.") *> 
        IO.raiseError(new IllegalArgumentException("Selection out of bounds"))
    }
  } yield params

  def printLogTail(): IO[Unit] = IO.blocking {
    println("\n--- [START OF LOG TAIL FROM MYNEWLOG.TXT] ---")
    try {
      val source = scala.io.Source.fromFile("mynewlog.txt")
      val lines = source.getLines().toVector
      source.close()
      lines.takeRight(15).foreach(println)
    } catch {
      case _: Exception => println("Could not read mynewlog.txt file.")
    }
    println("--- [END OF LOG TAIL] ---\n")
  }

  def getCurrentTimeString(pattern: String): String = {
    LocalDateTime.now().format(DateTimeFormatter.ofPattern(pattern))
  }

  def checkSockstatPortListening(port: String): Boolean = {
    try {
      val output = Seq("sockstat", "-4l").!!
      output.contains(s":$port")
    } catch {
      case _: Exception => false
    }
  }

  def checkLogForListeningString(): Boolean = {
    try {
      val fileSource = scala.io.Source.fromFile("mynewlog.txt")
      val hasMatch = fileSource.getLines().exists(_.contains("server is listening on"))
      fileSource.close()
      hasMatch
    } catch {
      case _: Exception => false
    }
  }

  def executeServerCheckSequence(process: Process, currentLoop: Int): IO[Boolean] = {
    if (currentLoop > 35) IO.pure(false)
    else {
      IO.blocking(process.isAlive()).flatMap {
        case false => IO.pure(false)
        case true =>
          val logTimestamp = getCurrentTimeString("HH:mm-ss")
          IO.println(s"counting $currentLoop $logTimestamp") *> {
            if (checkLogForListeningString()) {
              val subTimestamp = getCurrentTimeString("HH-mm-ss")
              IO.println(s"we entered second if $subTimestamp") *> IO.blocking(checkSockstatPortListening(myport)).flatMap {
                case true  => IO.pure(true)
                case false => IO.sleep(3.seconds) *> executeServerCheckSequence(process, currentLoop + 1)
              }
            } else {
              IO.sleep(3.seconds) *> executeServerCheckSequence(process, currentLoop + 1)
            }
          }
      }
    }
  }

  def safelyTerminateProcess(process: scala.sys.process.Process, label: String): IO[Unit] = IO.blocking {
    try {
      if (process.isAlive()) {
        process.destroy() // Soft kill (SIGTERM)
        Thread.sleep(500) 
        
        if (process.isAlive()) {
          println(s"[$label Warning] Process did not respond to SIGTERM. Escalating to SIGKILL...")
          // Modern, type-safe Java 9+ process handle destruction replacing reflection hooks
          process match {
            case wrapped: java.lang.Process => 
              wrapped.destroyForcibly()
            case ch: Any =>
              // Uses standard Java 9 ProcessHandle API to scale across platforms cleanly
              val handleField = ch.getClass.getDeclaredFields.find(f => f.getType == classOf[java.lang.Process] || f.getName == "p")
              handleField.foreach { f =>
                f.setAccessible(true)
                f.get(ch).asInstanceOf[java.lang.Process].destroyForcibly()
              }
          }
        }
      }
    } catch {
      case _: Exception => 
    }
  }

  /**
   * Extracted logic for successful server boot.
   * Manages Chrome startup, keeps runtime active, and captures keyboard interrupts.
   */
  def launchClientAndManageLifecycle(llamaProcess: Process, writer: PrintWriter): IO[Unit] = for {
    _             <- IO.sleep(2.seconds)
    finalCheck    <- IO.blocking(checkSockstatPortListening(myport))
    _             <- if (finalCheck) {
      for {
        _             <- IO.println(s"🎉 Server spawned successfully and responding on port $myport!")
        _             <- IO.println("Launching Chrome in the background...")
        chromeProcess <- launchChromeInBackground(writer)
        _             <- IO.println("Everything is now running. Press [ENTER] in this terminal to shut down both systems cleanly...")
        _             <- IO.readLine
        _             <- IO.println("Terminating background processes...")
        _             <- safelyTerminateProcess(chromeProcess, "Chrome")
      } yield ()
    } else {
      IO.println(s"[ERROR] Server failed port binding verification on port $myport.")
    }
  } yield ()

  /**
   * Extracted logic for server boot failure.
   * Prints full debug text summaries, pipes log files, and tears down lingering threads safely.
   */
  def handleServerCrashAndTeardown(): IO[Unit] = for {
    _ <- IO.println("\n==========================================================================")
    _ <- IO.println("[CRITICAL CRASH] llama-server stopped unexpectedly or timed out during boot!")
    _ <- IO.println("Here are the latest log messages explaining why it failed:")
    _ <- printLogTail()
    _ <- IO.println("Please check if the model file path exists or if the parameters are valid.")
    _ <- IO.println("==========================================================================\n")
  } yield ()

  val programWorkflow: IO[Unit] = createLogWriter.use { writer =>
    for {
      // Initial cleanup sweep to free port 8123 from previous broken runs
      _             <- IO.blocking { 
                         try { Seq("pkill", "-f", "llama-server").! } catch { case _: Exception => 0 } 
                     }
      modelParams   <- handleSelection
      _             <- IO.println("Launching llama-server in the background...")
      llamaProcess  <- launchLlamaServerInBackground(modelParams, writer)
      _             <- IO.println("Checking server status...")
      isServerReady <- executeServerCheckSequence(llamaProcess, 1)
      _             <- if (isServerReady) launchClientAndManageLifecycle(llamaProcess, writer)
                       else handleServerCrashAndTeardown()
      
      _             <- safelyTerminateProcess(llamaProcess, "Llama Server")
      _             <- IO.println("Sleeping for 1 second before final exit...")
      _             <- IO.sleep(1.second)
      _             <- IO.println("Shutdown complete.")
    } yield ()
  }

  override val run: IO[Unit] = programWorkflow
}
 
a little bit of optimisation for 12GB VRAM,
sh:
Selected Model: Qwen3.6-35B-A3B-UD-IQ4_XS
Launching llama-server in the background...
[Llama Server Debug] Spawning command: llama-server --ctx-size 32768 --batch-size 1024 --ubatch-size 1024 --threads-batch 10 --threads 10 --n-gpu-layers all --host 127.0.0.1 --port 8123 --jinja --verbose --no-mmap --fit off --cache-ram 4096 --kv-unified --no-warmup --parallel 1 --n-gpu-layers 40 --temp 0.6 --cpu-moe --model /home/x/hugging/16/Qwen3.6-35B-A3B-UD-IQ4_XS.gguf --spec-draft-p-min 0.75 --spec-draft-n-max 2 --spec-type draft-mtp --reasoning on --reasoning-budget -1 --chat-template-kwargs "{\"preserve_thinking\":true}" --n-gpu-layers 40 --presence-penalty 0.0 --repeat-penalty 1.0
 
Back
Top