934476ca00ac68ac9c386779521f8b6b9a3ae1b4

Author
TheEdgeOfRage <git@theedgeofrage.com>
Committer
TheEdgeOfRage <git@theedgeofrage.com>
Date

Message

Switch to Q3 only quantization on Qwen3.8

Diff

 1diff --git a/dot_config/llama-server/models.ini b/dot_config/llama-server/models.ini
 2index 6145205034bfc7dac195c91b9a99d13ecf0abe88..5ea32f3f29505711d40cb3ade2c573fd9972a629 100644
 3--- a/dot_config/llama-server/models.ini
 4+++ b/dot_config/llama-server/models.ini
 5@@ -11,13 +11,13 @@ alias = f2llm-v2-330m
 6 n-gpu-layers = 99
 7 embedding = true
 8 
 9-[unsloth/Qwen3.8-27B-GGUF:UD-Q2_K_XL]
10-hf = unsloth/Qwen3.8-27B-GGUF:UD-Q2_K_XL
11-alias = qwen3.8-27b-q2,reviewer
12+[unsloth/Qwen3.8-27B-GGUF:UD-IQ3_S]
13+hf = unsloth/Qwen3.8-27B-GGUF:UD-IQ3_S
14+alias = qwen3.8-27b-q3,reviewer
15 n-gpu-layers = 99
16 flash-attn = on
17 jinja = true
18-ctx-size = 16384
19+ctx-size = 131072
20 temp = 0.6
21 top-p = 0.95
22 top-k = 20
23@@ -30,24 +30,6 @@ presence-penalty = 1.5
24 reasoning = on
25 chat-template-kwargs = {"reasoning_effort":"low"}
26 
27-[unsloth/Qwen3.8-27B-GGUF:UD-Q4_K_S]
28-hf = unsloth/Qwen3.8-27B-GGUF:UD-Q4_K_S
29-alias = qwen3.8-27b
30-n-gpu-layers = 99
31-flash-attn = on
32-jinja = true
33-ctx-size = 65536
34-temp = 0.9
35-top-p = 0.95
36-top-k = 20
37-min-p = 0.0
38-parallel = 1
39-spec-type = draft-mtp
40-spec-draft-n-max = 2
41-repeat-penalty = 1.0
42-reasoning = on
43-chat-template-kwargs = {"reasoning_effort":"medium"}
44-
45 [unsloth/gemma-4-26B-A4B-it-qat-GGUF:Q4_K_XL]
46 hf = unsloth/gemma-4-26B-A4B-it-qat-GGUF:UD-Q4_K_XL
47 alias = gemma4-26b-a4b
48@@ -56,6 +38,8 @@ n-gpu-layers-draft = 99
49 ngl=999
50 flash-attn = on
51 ctx-size = 131072
52+batch-size = 2048
53+ubatch-size = 2048
54 temp = 0.85
55 top-p = 0.9
56 top-k = 32