#!/bin/sh # Serve Qwen 3.8 27B (4-bit, the official baseline build) for the harness on the Mac mini. Run this on the MacBook. # Same server flags as train/serve.sh (the official baseline); only the host differs (0.0.0.0) and caffeinate keeps the # MacBook awake as long as the server process lives (caffeinate -w ). Usage: ~/serve_remote.sh Stop: kill $(cat ~/qwen_server.pid) MODEL="$HOME/models/Qwen3.8-27B-4bit" [ -f "$MODEL/config.json" ] || { echo "model not found: $MODEL" >&2; exit 1; } # the mlx_lm server: $MLX_SERVER, else ~/qwen-venv (mlx-lm 0.32.0, see docs/remote-model.md), else the harness venv, else PATH for c in "$MLX_SERVER" "$HOME/qwen-venv/bin/mlx_lm.server" "$HOME/projects/abap-llm/harness/train/.venv/bin/mlx_lm.server" "$(command -v mlx_lm.server)"; do if [ -n "$c" ] && [ -x "$c" ]; then SERVER="$c"; break; fi done [ -n "$SERVER" ] || { echo "mlx_lm.server not found: create ~/qwen-venv with mlx-lm 0.32.0 (docs/remote-model.md)" >&2; exit 1; } # the baseline flags need mlx-lm 0.32.x; an older server does not know them "$SERVER" --help 2>&1 | grep -q -- "--prompt-cache-bytes" || { echo "$SERVER is too old (no --prompt-cache-bytes). Install mlx-lm 0.32.0 in ~/qwen-venv (docs/remote-model.md)" >&2; exit 1; } echo "server: $SERVER" "$SERVER" \ --model "$MODEL" \ --host 0.0.0.0 --port 8080 \ --temp 0.2 --top-p 0.95 --top-k 20 --min-p 0 \ --max-tokens 32768 \ --prompt-cache-size 4 --prompt-cache-bytes 6000000000 \ --chat-template-args '{"enable_thinking": true, "reasoning_effort": "medium"}' & SP=$! echo "$SP" > "$HOME/qwen_server.pid" # the pid of the server itself exec caffeinate -dimsu -w "$SP" # keeps the Mac awake until the server process ends