adee7b8fbe07b9541caccb1453e8a00260bdd0c9

Author
TheEdgeOfRage <git@theedgeofrage.com>
Committer
TheEdgeOfRage <git@theedgeofrage.com>
Date

Message

Switch to single-instance llama-server with per-prompt-family slots

Diff

This diff is truncated to protect this page.

  1diff --git a/Makefile b/Makefile
  2index b7f73c7b6072a79887f898347159ec469dd775d7..e130a4cda6c4de483d121b79efd3afe7fe317834 100644
  3--- a/Makefile
  4+++ b/Makefile
  5@@ -1,4 +1,4 @@
  6-.PHONY: lint fmt build run audiocpp fetch-models
  7+.PHONY: lint fmt build run llama audiocpp fetch-models
  8 
  9 lint:
 10 	golangci-lint run ./...
 11@@ -12,6 +12,27 @@ build:
 12 run:
 13 	go run ./cmd/jp
 14 
 15+llama:
 16+	llama-server \
 17+		--host 0.0.0.0 \
 18+		--port 9931 \
 19+		--hf-repo unsloth/gemma-4-12B-it-qat-GGUF:UD-Q4_K_XL \
 20+		--n-gpu-layers 99 \
 21+		--n-gpu-layers-draft 99 \
 22+		--fit off \
 23+		--flash-attn on \
 24+		--parallel 5 \
 25+		--kv-unified \
 26+		--ctx-size 32768 \
 27+		--batch-size 1024 \
 28+		--ubatch-size 1024 \
 29+		--temp 1.0 \
 30+		--top-p 0.95 \
 31+		--top-k 64 \
 32+		--spec-type draft-mtp \
 33+		--spec-draft-n-max 4 \
 34+		--reasoning off
 35+
 36 audiocpp:
 37 	audiocpp_server --config audio.cpp.json
 38 
 39diff --git a/README.md b/README.md
 40index 55a9f65099d8e8f6e5853568edcd0a6a05c163a1..e908af65c42f75eb50d7303c5720cd9a3e3c1de4 100644
 41--- a/README.md
 42+++ b/README.md
 43@@ -30,9 +30,8 @@ go build ./...             # or: make build
 44 jp                         # or: make run (go run ./cmd/jp)
 45 ```
 46 
 47-On startup the game preloads the LLM model, runs a bounded readiness check
 48-against the audio service, then warms each conversation's system prompt once so first
 49-replies are fast. A failed step prints one error line naming the affected
 50+On startup the game runs a bounded readiness check against the audio service,
 51+then warms each conversation's system prompt once so first replies are fast. A failed step prints one error line naming the affected
 52 service and its configured URL; startup continues either way. The game never
 53 launches a service to make a check pass.
 54 
 55@@ -44,10 +43,9 @@ which win over defaults. Defaults match a local setup.
 56 | Setting          | Flag                 | Env var               | Default                                   | Use                                             |
 57 | ---------------- | -------------------- | --------------------- | ----------------------------------------- | ----------------------------------------------- |
 58 | LLM router URL   | `--url`              | `JP_LLM_BASE_URL`     | `https://llama.home.theedgeofrage.com`    | llama.cpp router; `POST /v1/chat/completions`   |
 59-| LLM model        | `--model`            | `MODEL`               | `jp`                                      | model name sent in every LLM request            |
 60 | Temperature      | `--temperature`      | `TEMPERATURE`         | `1.0`                                     | sampling temperature                            |
 61 | Max tokens       | `--max-tokens`       | `MAX_TOKENS`          | `256`                                     | max output tokens per reply                     |
 62-| Thinking mode    | `--enable-thinking`  | `ENABLE_THINKING`     | `false`                                   | Qwen3 thinking (`chat_template_kwargs.enable_thinking`) |
 63+| Thinking mode    | `--enable-thinking`  | `ENABLE_THINKING`     | `false`                                   | model thinking (`chat_template_kwargs.enable_thinking`) |
 64 | Audio base URL   | `--audio-url`        | `JP_AUDIO_BASE_URL`   | `http://127.0.0.1:8080`                  | audio.cpp server (TTS + ASR)                    |
 65 | Scenario brief   | `--scenario`         | `JP_SCENARIO_PATH`    | `assets/scenarios/small_city.md`          | plain-text scenario brief file                  |
 66 
 67diff --git a/cmd/jp/main.go b/cmd/jp/main.go
 68index 9176f6b1d4b6d449d04bc5631d2886964fb9e7d0..b06b8a50dc65a2068f7cb7f0fd5f9bcb1ef83b1b 100644
 69--- a/cmd/jp/main.go
 70+++ b/cmd/jp/main.go
 71@@ -27,9 +27,15 @@ const requestTimeout = 10 * time.Second
 72 const warmupTimeout = 60 * time.Second
 73 
 74 // llama-server slot assignments and the capture command are fixed for this
 75-// deployment; they are not user-configurable.
 76-const gameSlot = 0
 77-const scratchSlot = 1
 78+// deployment; they are not user-configurable. One slot per prompt family keeps
 79+// each system prompt cached in the server's unified KV pool.
 80+const (
 81+	gameSlot       = 0
 82+	judgeSlot      = 1
 83+	compactionSlot = 2
 84+	sheetSlot      = 3
 85+	scratchSlot    = 4
 86+)
 87 const recordCommand = "arecord"
 88 
 89 func main() {
 90@@ -41,34 +47,39 @@ func main() {
 91 	}
 92 
 93 	hc := &http.Client{Timeout: requestTimeout}
 94-	gameClient := llm.NewClient(cfg, hc)
 95-	gameClient.Slot = gameSlot
 96-	scratchClient := llm.NewClient(cfg, hc)
 97-	scratchClient.Slot = scratchSlot
 98-
 99-	llmUp := true
100-	if err := gameClient.LoadModel(); err != nil {
101-		fmt.Fprintf(os.Stderr, "jp: LLM unavailable: %v\n", err)
102-		llmUp = false
103-	}
104+	gameClient := newLLMClient(cfg, hc, gameSlot)
105+	judgeClient := newLLMClient(cfg, hc, judgeSlot)
106+	compactionClient := newLLMClient(cfg, hc, compactionSlot)
107+	sheetClient := newLLMClient(cfg, hc, sheetSlot)
108+	scratchClient := newLLMClient(cfg, hc, scratchSlot)
109+
110 	if err := availability.CheckAudio(hc, cfg.AudioBaseURL, []string{tts.ModelName, stt.ASRModelName}); err != nil {
111 		fmt.Fprintf(os.Stderr, "jp: audio service unavailable: %v\n", err)
112 	}
113 
114-	if llmUp {
115-		ctx, cancel := context.WithTimeout(context.Background(), warmupTimeout)
116-		err := gameClient.Warmup(ctx, []string{llm.GameSystemPrompt(sc.Brief)})
117-		if err == nil {
118-			err = scratchClient.Warmup(ctx, []string{llm.JudgeSystemPrompt(), llm.CompactionPrompt(), llm.SheetSystemPrompt(), llm.ScratchSystemPrompt()})
119-		}
120-		cancel()
121-		if err != nil {
122-			fmt.Fprintf(os.Stderr, "jp: LLM warmup failed: %v\n", err)
123+	warmups := []struct {
124+		client *llm.Client
125+		prompt string
126+	}{
127+		{gameClient, llm.GameSystemPrompt(sc.Brief)},
128+		{judgeClient, llm.JudgeSystemPrompt()},
129+		{compactionClient, llm.CompactionPrompt()},
130+		{sheetClient, llm.SheetSystemPrompt()},
131+		{scratchClient, llm.ScratchSystemPrompt()},
132+	}
133+	ctx, cancel := context.WithTimeout(context.Background(), warmupTimeout)
134+	for _, w := range warmups {
135+		if err = w.client.Warmup(ctx, []string{w.prompt}); err != nil {
136+			break
137 		}
138 	}
139+	cancel()
140+	if err != nil {
141+		fmt.Fprintf(os.Stderr, "jp: LLM warmup failed: %v\n", err)
142+	}
143 
144 	state := game.NewState(sc.Brief)
145-	orch := buildOrchestrator(cfg, state, hc, gameClient, scratchClient)
146+	orch := buildOrchestrator(cfg, state, hc, gameClient, judgeClient, compactionClient, sheetClient, scratchClient)
147 
148 	m := ui.NewModel(state, orch, int(recordCap.Seconds()))
149 	if _, err := tea.NewProgram(m).Run(); err != nil {
150@@ -76,11 +87,11 @@ func main() {
151 	}
152 }
153 
154-func buildOrchestrator(cfg *config.Config, state *game.State, hc *http.Client, gameClient, scratchClient *llm.Client) *game.Orchestrator {
155+func buildOrchestrator(cfg *config.Config, state *game.State, hc *http.Client, gameClient, judgeClient, compactionClient, sheetClient, scratchClient *llm.Client) *game.Orchestrator {
156 	gameModel := &adapters.GameModel{Client: gameClient}
157-	compactor := &adapters.Compactor{Client: scratchClient}
158-	judgeModel := &adapters.JudgeModel{Client: scratchClient}
159-	sheetModel := &adapters.SheetModel{Client: scratchClient}
160+	compactor := &adapters.Compactor{Client: compactionClient}
161+	judgeModel := &adapters.JudgeModel{Client: judgeClient}
162+	sheetModel := &adapters.SheetModel{Client: sheetClient}
163 	askModel := &adapters.ScratchModel{Client: scratchClient}
164 
165 	speechIn := &adapters.SpeechInput{
166@@ -95,6 +106,12 @@ func buildOrchestrator(cfg *config.Config, state *game.State, hc *http.Client, g
167 	return game.NewOrchestrator(state, gameModel, judgeModel, compactor, sheetModel, askModel, speechIn, speechOut)
168 }
169 
170+func newLLMClient(cfg *config.Config, hc *http.Client, slot int) *llm.Client {
171diff --git a/docs/services.md b/docs/services.md
172index 31b9d573e473527c9152765edf27c64cd0b51880..4db6f168136c5e6707d03ac344abb667154c9ae1 100644
173--- a/docs/services.md
174+++ b/docs/services.md
175@@ -14,50 +14,37 @@ match the local setup:
176 | LLM router URL   | `--url`            | `JP_LLM_BASE_URL`   | `https://llama.home.theedgeofrage.com` | `POST /v1/chat/completions`                                       |
177 | Audio base URL   | `--audio-url`      | `JP_AUDIO_BASE_URL` | `http://127.0.0.1:8080`                | TTS `POST /v1/audio/speech` + ASR `POST /v1/audio/transcriptions` |
178 
179-At startup, `jp` first preloads the LLM model with
180-`POST {JP_LLM_BASE_URL}/models/load` (30s timeout) before anything else loads,
181-then makes only safe, bounded HTTP readiness checks (a few seconds each, no
182-inference requests). It then sends one best-effort warmup
183+At startup, `jp` makes only safe, bounded HTTP readiness checks (a few seconds
184+each, no inference requests). It then sends one best-effort warmup
185 completion per system prompt (`max_tokens: 1`) so the server's prompt cache
186 keeps those prefixes hot. It stays silent when every service answers; a down
187 service prints one error line naming the service and its URL. The game never
188 launches or supervises a service to make a check pass.
189 
190-## LLM - llama.cpp (Unsloth Qwen3-8B)
191+## LLM - llama.cpp (Unsloth Gemma 4 12B)
192 
193-llama.cpp server in router mode (started without a model argument) at
194-`JP_LLM_BASE_URL`, with OpenAI-compatible chat completions at
195-`POST {JP_LLM_BASE_URL}/v1/chat/completions`.
196+llama.cpp server in single-instance mode at `JP_LLM_BASE_URL`, with
197+OpenAI-compatible chat completions at `POST {JP_LLM_BASE_URL}/v1/chat/completions`.
198 
199-Model: `unsloth/Qwen3-8B-GGUF:UD-Q4_K_XL`, resolvable by the router as `jp`. The client
200-sends OpenAI-compatible requests with `model` set to `jp`, sends
201-`chat_template_kwargs.enable_thinking` per the `--enable-thinking` flag
202-(default false, which disables Qwen3 thinking), sets `cache_prompt: true` so
203-the server keeps prompt prefixes cached between turns, pins each request to a
204+Model: `unsloth/gemma-4-12B-it-qat-GGUF:UD-Q4_K_XL`. The client sends
205+OpenAI-compatible requests without a model field (the single-instance server
206+runs exactly one model), sends `chat_template_kwargs.enable_thinking` per the
207+`--enable-thinking` flag (default false), sets `cache_prompt: true` so the
208+server keeps prompt prefixes cached between turns, pins each request to a
209 llama-server slot via `id_slot` (see below), and reads streaming SSE. The model
210 chat template (`--jinja`) is used; ChatML is not built manually by the client.
211 
212 ### Slot pinning
213 
214 Each request body carries an `id_slot` field that assigns the task to a
215-specific llama-server slot, bypassing the server's LRU/similarity
216-auto-selection. The game uses two fixed slots so scratch requests can never
217-evict the game loop's cached context: game-loop requests pin `id_slot: 0`, and
218-scratch requests (judge, compaction, character sheets; warmup included) pin
219-`id_slot: 1`. These are hardcoded constants, not user configuration.
220diff --git a/internal/config/config.go b/internal/config/config.go
221index ddad64a789baa1505ef1db0947f5fac09b58c852..33da5bcf950a5930172f5512d228567490338c70 100644
222--- a/internal/config/config.go
223+++ b/internal/config/config.go
224@@ -13,7 +13,6 @@ import (
225 
226 type LLMConfig struct {
227 	BaseURL        string  `long:"url" env:"JP_LLM_BASE_URL" default:"https://llama.home.theedgeofrage.com" description:"llama.cpp router base URL"`
228-	Model          string  `long:"model" env:"MODEL" default:"jp"`
229 	Temperature    float64 `long:"temperature" env:"TEMPERATURE" default:"1.0"`
230 	MaxTokens      int     `long:"max-tokens" env:"MAX_TOKENS" default:"256"`
231 	EnableThinking bool    `long:"enable-thinking" env:"ENABLE_THINKING"`
232diff --git a/internal/llm/client.go b/internal/llm/client.go
233index c13cc8c70c26bc30c075a04836496c1ee958bb0b..cd463c8e3b1c422d08965d77b29eb3778eac880a 100644
234--- a/internal/llm/client.go
235+++ b/internal/llm/client.go
236@@ -14,7 +14,6 @@ import (
237 	"japanese/internal/config"
238 	"net/http"
239 	"strings"
240-	"time"
241 )
242 
243 type Role string
244@@ -33,7 +32,6 @@ type Message struct {
245 type Client struct {
246 	baseURL        string
247 	httpClient     *http.Client
248-	Model          string
249 	Temperature    float64
250 	MaxTokens      int
251 	EnableThinking bool
252@@ -49,7 +47,6 @@ func NewClient(cfg *config.Config, hc *http.Client) *Client {
253 	return &Client{
254 		httpClient:     hc,
255 		baseURL:        cfg.LLMConfig.BaseURL,
256-		Model:          cfg.LLMConfig.Model,
257 		Temperature:    cfg.LLMConfig.Temperature,
258 		MaxTokens:      cfg.LLMConfig.MaxTokens,
259 		EnableThinking: cfg.LLMConfig.EnableThinking,
260@@ -57,7 +54,6 @@ func NewClient(cfg *config.Config, hc *http.Client) *Client {
261 }
262 
263 type chatRequest struct {
264-	Model              string         `json:"model"`
265 	Messages           []Message      `json:"messages"`
266 	ChatTemplateKwargs map[string]any `json:"chat_template_kwargs"`
267 	Temperature        float64        `json:"temperature"`
268@@ -100,7 +96,6 @@ func (c *Client) chat(ctx context.Context, msgs []Message, maxTokens int) (strin
269 		return "", fmt.Errorf("llm: no base URL configured")
270 	}
271 	body := chatRequest{
272-		Model:              c.Model,
273 		Messages:           msgs,
274 		ChatTemplateKwargs: map[string]any{"enable_thinking": c.EnableThinking},
275 		Temperature:        c.Temperature,
276@@ -139,47 +134,6 @@ func (c *Client) chat(ctx context.Context, msgs []Message, maxTokens int) (strin
277 	return text, nil
278 }
279 
280-// loadTimeout bounds the model preload request. Loading a large model into
281-// VRAM takes much longer than a normal API call.
282-const loadTimeout = 30 * time.Second
283-
284-// LoadModel asks the llama.cpp router to load the configured model so the
285-// first turn does not pay the loading cost. It is safe to call when the model
286-// is already loaded; the router keeps it loaded.
287-func (c *Client) LoadModel() error {
288-	if c.baseURL == "" {
289-		return fmt.Errorf("llm: no base URL configured")
290-	}
291-	ctx, cancel := context.WithTimeout(context.Background(), loadTimeout)
292-	defer cancel()
293-
294-	payload, err := json.Marshal(map[string]string{"model": c.Model})
295-	if err != nil {
296-		return fmt.Errorf("llm: encode request: %w", err)
297-	}
298-	url := strings.TrimRight(c.baseURL, "/") + "/models/load"
299-	req, err := http.NewRequestWithContext(ctx, http.MethodPost, url, bytes.NewReader(payload))
300-	if err != nil {
301-		return fmt.Errorf("llm: build request: %w", err)
302-	}
303-	req.Header.Set("Content-Type", "application/json")
304-
305-	resp, err := c.httpClient.Do(req)
306-	if err != nil {
307-		return fmt.Errorf("llm: request to %s: %w", url, err)
308-	}
309-	defer func() { _ = resp.Body.Close() }()
310-
311-	if resp.StatusCode < 200 || resp.StatusCode >= 300 {
312-		detail, _ := io.ReadAll(io.LimitReader(resp.Body, 4096))
313-		if strings.Contains(string(detail), "already running") {
314-			return nil
315-		}
316-		return fmt.Errorf("llm: %s returned HTTP %d: %s", url, resp.StatusCode, strings.TrimSpace(string(detail)))
317-	}
318-	return nil
319-}
320-
321 func readSSE(r io.Reader) (string, error) {
322 	var sb strings.Builder
323 	scanner := bufio.NewScanner(r)