#!/bin/sh # Serve the base model (no adapter) with mlx_lm.server. Settings: train/README.md. # Usage: train/serve.sh [--adapter-path PATH] # Prompt cache limit: without it the cache grew to 7.9 GB in 15 min (2026-10-03); A4H needs 22 GB. cd "$(dirname "$0")/.." exec train/.venv/bin/mlx_lm.server \ --model "$HOME/models/Qwen3.8-27B-4bit" \ --host 127.0.0.1 --port 8080 \ --temp 0.2 --top-p 0.95 --top-k 20 --min-p 0 \ --max-tokens 32768 \ --prompt-cache-size 4 --prompt-cache-bytes 6000000000 \ --chat-template-args '{"enable_thinking": true, "reasoning_effort": "medium"}' \ "$@"