25 lines
1.3 KiB
Bash
Executable File
25 lines
1.3 KiB
Bash
Executable File
#!/bin/sh
|
|
# Serve Qwen 3.8 27B (4-bit, the official baseline build) for the harness on the Mac mini. Run this on the MacBook.
|
|
# Same server flags as train/serve.sh (the official baseline); only the host differs (0.0.0.0) and caffeinate keeps the
|
|
# MacBook awake as long as the server process lives (caffeinate -w <server pid>). Usage: ~/serve_remote.sh Stop: kill $(cat ~/qwen_server.pid)
|
|
MODEL="$HOME/models/Qwen3.8-27B-4bit"
|
|
[ -f "$MODEL/config.json" ] || { echo "model not found: $MODEL" >&2; exit 1; }
|
|
# the mlx_lm of the harness venv if the repo is on this Mac, else the one on PATH
|
|
if [ -x "$HOME/projects/abap-llm/harness/train/.venv/bin/mlx_lm.server" ]; then
|
|
SERVER="$HOME/projects/abap-llm/harness/train/.venv/bin/mlx_lm.server"
|
|
elif command -v mlx_lm.server >/dev/null 2>&1; then
|
|
SERVER="$(command -v mlx_lm.server)"
|
|
else
|
|
echo "mlx_lm.server not found (pip install mlx-lm==0.32.0)" >&2; exit 1
|
|
fi
|
|
"$SERVER" \
|
|
--model "$MODEL" \
|
|
--host 0.0.0.0 --port 8080 \
|
|
--temp 0.2 --top-p 0.95 --top-k 20 --min-p 0 \
|
|
--max-tokens 32768 \
|
|
--prompt-cache-size 4 --prompt-cache-bytes 6000000000 \
|
|
--chat-template-args '{"enable_thinking": true, "reasoning_effort": "medium"}' &
|
|
SP=$!
|
|
echo "$SP" > "$HOME/qwen_server.pid" # the pid of the server itself
|
|
exec caffeinate -dimsu -w "$SP" # keeps the Mac awake until the server process ends
|