From a2ba9e7b44262bb088efc78d17424a018b6345cc Mon Sep 17 00:00:00 2001 From: Kral Date: Sun, 4 Oct 2026 17:02:59 +0200 Subject: [PATCH] serve.sh back to Qwen 3.8 27B 4-bit Co-Authored-By: Claude Sonnet 5.5 Claude-Session: https://claude.ai/code/session_014aUaQeLnwbb1zTpN7kHeat --- train/serve.sh | 6 ++++-- 1 file changed, 4 insertions(+), 2 deletions(-) diff --git a/train/serve.sh b/train/serve.sh index 68db2b2..1e09182 100755 --- a/train/serve.sh +++ b/train/serve.sh @@ -1,12 +1,14 @@ #!/bin/sh -# Base model: Devstral Small 2 (no thinking mode, no chat-template args). Serve the base model (no adapter) with mlx_lm.server. Settings: train/README.md. +# Serve the base model (no adapter) with mlx_lm.server. Settings: train/README.md. # Usage: train/serve.sh [--adapter-path PATH] +# Base model: Qwen 3.8 27B (decision 2026-10-04). Thinking is switched off per request by train/baseline.py. # Prompt cache limit: without it the cache grew to 7.9 GB in 15 min (2026-10-03); A4H needs 22 GB. cd "$(dirname "$0")/.." exec train/.venv/bin/mlx_lm.server \ - --model "$HOME/models/Devstral-Small-2-24B-4bit" \ + --model "$HOME/models/Qwen3.8-27B-4bit" \ --host 127.0.0.1 --port 8080 \ --temp 0.2 --top-p 0.95 --top-k 20 --min-p 0 \ --max-tokens 32768 \ --prompt-cache-size 4 --prompt-cache-bytes 6000000000 \ + --chat-template-args '{"enable_thinking": true, "reasoning_effort": "medium"}' \ "$@"