adee7b8fbe07b9541caccb1453e8a00260bdd0c9
- Author
- TheEdgeOfRage <git@theedgeofrage.com>
- Committer
- TheEdgeOfRage <git@theedgeofrage.com>
- Date
Message
Diff
This diff is truncated to protect this page.
1diff --git a/Makefile b/Makefile
2index b7f73c7b6072a79887f898347159ec469dd775d7..e130a4cda6c4de483d121b79efd3afe7fe317834 100644
3--- a/Makefile
4+++ b/Makefile
5@@ -1,4 +1,4 @@
6-.PHONY: lint fmt build run audiocpp fetch-models
7+.PHONY: lint fmt build run llama audiocpp fetch-models
8
9 lint:
10 golangci-lint run ./...
11@@ -12,6 +12,27 @@ build:
12 run:
13 go run ./cmd/jp
14
15+llama:
16+ llama-server \
17+ --host 0.0.0.0 \
18+ --port 9931 \
19+ --hf-repo unsloth/gemma-4-12B-it-qat-GGUF:UD-Q4_K_XL \
20+ --n-gpu-layers 99 \
21+ --n-gpu-layers-draft 99 \
22+ --fit off \
23+ --flash-attn on \
24+ --parallel 5 \
25+ --kv-unified \
26+ --ctx-size 32768 \
27+ --batch-size 1024 \
28+ --ubatch-size 1024 \
29+ --temp 1.0 \
30+ --top-p 0.95 \
31+ --top-k 64 \
32+ --spec-type draft-mtp \
33+ --spec-draft-n-max 4 \
34+ --reasoning off
35+
36 audiocpp:
37 audiocpp_server --config audio.cpp.json
38
39diff --git a/README.md b/README.md
40index 55a9f65099d8e8f6e5853568edcd0a6a05c163a1..e908af65c42f75eb50d7303c5720cd9a3e3c1de4 100644
41--- a/README.md
42+++ b/README.md
43@@ -30,9 +30,8 @@ go build ./... # or: make build
44 jp # or: make run (go run ./cmd/jp)
45 ```
46
47-On startup the game preloads the LLM model, runs a bounded readiness check
48-against the audio service, then warms each conversation's system prompt once so first
49-replies are fast. A failed step prints one error line naming the affected
50+On startup the game runs a bounded readiness check against the audio service,
51+then warms each conversation's system prompt once so first replies are fast. A failed step prints one error line naming the affected
52 service and its configured URL; startup continues either way. The game never
53 launches a service to make a check pass.
54
55@@ -44,10 +43,9 @@ which win over defaults. Defaults match a local setup.
56 | Setting | Flag | Env var | Default | Use |
57 | ---------------- | -------------------- | --------------------- | ----------------------------------------- | ----------------------------------------------- |
58 | LLM router URL | `--url` | `JP_LLM_BASE_URL` | `https://llama.home.theedgeofrage.com` | llama.cpp router; `POST /v1/chat/completions` |
59-| LLM model | `--model` | `MODEL` | `jp` | model name sent in every LLM request |
60 | Temperature | `--temperature` | `TEMPERATURE` | `1.0` | sampling temperature |
61 | Max tokens | `--max-tokens` | `MAX_TOKENS` | `256` | max output tokens per reply |
62-| Thinking mode | `--enable-thinking` | `ENABLE_THINKING` | `false` | Qwen3 thinking (`chat_template_kwargs.enable_thinking`) |
63+| Thinking mode | `--enable-thinking` | `ENABLE_THINKING` | `false` | model thinking (`chat_template_kwargs.enable_thinking`) |
64 | Audio base URL | `--audio-url` | `JP_AUDIO_BASE_URL` | `http://127.0.0.1:8080` | audio.cpp server (TTS + ASR) |
65 | Scenario brief | `--scenario` | `JP_SCENARIO_PATH` | `assets/scenarios/small_city.md` | plain-text scenario brief file |
66
67diff --git a/cmd/jp/main.go b/cmd/jp/main.go
68index 9176f6b1d4b6d449d04bc5631d2886964fb9e7d0..b06b8a50dc65a2068f7cb7f0fd5f9bcb1ef83b1b 100644
69--- a/cmd/jp/main.go
70+++ b/cmd/jp/main.go
71@@ -27,9 +27,15 @@ const requestTimeout = 10 * time.Second
72 const warmupTimeout = 60 * time.Second
73
74 // llama-server slot assignments and the capture command are fixed for this
75-// deployment; they are not user-configurable.
76-const gameSlot = 0
77-const scratchSlot = 1
78+// deployment; they are not user-configurable. One slot per prompt family keeps
79+// each system prompt cached in the server's unified KV pool.
80+const (
81+ gameSlot = 0
82+ judgeSlot = 1
83+ compactionSlot = 2
84+ sheetSlot = 3
85+ scratchSlot = 4
86+)
87 const recordCommand = "arecord"
88
89 func main() {
90@@ -41,34 +47,39 @@ func main() {
91 }
92
93 hc := &http.Client{Timeout: requestTimeout}
94- gameClient := llm.NewClient(cfg, hc)
95- gameClient.Slot = gameSlot
96- scratchClient := llm.NewClient(cfg, hc)
97- scratchClient.Slot = scratchSlot
98-
99- llmUp := true
100- if err := gameClient.LoadModel(); err != nil {
101- fmt.Fprintf(os.Stderr, "jp: LLM unavailable: %v\n", err)
102- llmUp = false
103- }
104+ gameClient := newLLMClient(cfg, hc, gameSlot)
105+ judgeClient := newLLMClient(cfg, hc, judgeSlot)
106+ compactionClient := newLLMClient(cfg, hc, compactionSlot)
107+ sheetClient := newLLMClient(cfg, hc, sheetSlot)
108+ scratchClient := newLLMClient(cfg, hc, scratchSlot)
109+
110 if err := availability.CheckAudio(hc, cfg.AudioBaseURL, []string{tts.ModelName, stt.ASRModelName}); err != nil {
111 fmt.Fprintf(os.Stderr, "jp: audio service unavailable: %v\n", err)
112 }
113
114- if llmUp {
115- ctx, cancel := context.WithTimeout(context.Background(), warmupTimeout)
116- err := gameClient.Warmup(ctx, []string{llm.GameSystemPrompt(sc.Brief)})
117- if err == nil {
118- err = scratchClient.Warmup(ctx, []string{llm.JudgeSystemPrompt(), llm.CompactionPrompt(), llm.SheetSystemPrompt(), llm.ScratchSystemPrompt()})
119- }
120- cancel()
121- if err != nil {
122- fmt.Fprintf(os.Stderr, "jp: LLM warmup failed: %v\n", err)
123+ warmups := []struct {
124+ client *llm.Client
125+ prompt string
126+ }{
127+ {gameClient, llm.GameSystemPrompt(sc.Brief)},
128+ {judgeClient, llm.JudgeSystemPrompt()},
129+ {compactionClient, llm.CompactionPrompt()},
130+ {sheetClient, llm.SheetSystemPrompt()},
131+ {scratchClient, llm.ScratchSystemPrompt()},
132+ }
133+ ctx, cancel := context.WithTimeout(context.Background(), warmupTimeout)
134+ for _, w := range warmups {
135+ if err = w.client.Warmup(ctx, []string{w.prompt}); err != nil {
136+ break
137 }
138 }
139+ cancel()
140+ if err != nil {
141+ fmt.Fprintf(os.Stderr, "jp: LLM warmup failed: %v\n", err)
142+ }
143
144 state := game.NewState(sc.Brief)
145- orch := buildOrchestrator(cfg, state, hc, gameClient, scratchClient)
146+ orch := buildOrchestrator(cfg, state, hc, gameClient, judgeClient, compactionClient, sheetClient, scratchClient)
147
148 m := ui.NewModel(state, orch, int(recordCap.Seconds()))
149 if _, err := tea.NewProgram(m).Run(); err != nil {
150@@ -76,11 +87,11 @@ func main() {
151 }
152 }
153
154-func buildOrchestrator(cfg *config.Config, state *game.State, hc *http.Client, gameClient, scratchClient *llm.Client) *game.Orchestrator {
155+func buildOrchestrator(cfg *config.Config, state *game.State, hc *http.Client, gameClient, judgeClient, compactionClient, sheetClient, scratchClient *llm.Client) *game.Orchestrator {
156 gameModel := &adapters.GameModel{Client: gameClient}
157- compactor := &adapters.Compactor{Client: scratchClient}
158- judgeModel := &adapters.JudgeModel{Client: scratchClient}
159- sheetModel := &adapters.SheetModel{Client: scratchClient}
160+ compactor := &adapters.Compactor{Client: compactionClient}
161+ judgeModel := &adapters.JudgeModel{Client: judgeClient}
162+ sheetModel := &adapters.SheetModel{Client: sheetClient}
163 askModel := &adapters.ScratchModel{Client: scratchClient}
164
165 speechIn := &adapters.SpeechInput{
166@@ -95,6 +106,12 @@ func buildOrchestrator(cfg *config.Config, state *game.State, hc *http.Client, g
167 return game.NewOrchestrator(state, gameModel, judgeModel, compactor, sheetModel, askModel, speechIn, speechOut)
168 }
169
170+func newLLMClient(cfg *config.Config, hc *http.Client, slot int) *llm.Client {
171diff --git a/docs/services.md b/docs/services.md
172index 31b9d573e473527c9152765edf27c64cd0b51880..4db6f168136c5e6707d03ac344abb667154c9ae1 100644
173--- a/docs/services.md
174+++ b/docs/services.md
175@@ -14,50 +14,37 @@ match the local setup:
176 | LLM router URL | `--url` | `JP_LLM_BASE_URL` | `https://llama.home.theedgeofrage.com` | `POST /v1/chat/completions` |
177 | Audio base URL | `--audio-url` | `JP_AUDIO_BASE_URL` | `http://127.0.0.1:8080` | TTS `POST /v1/audio/speech` + ASR `POST /v1/audio/transcriptions` |
178
179-At startup, `jp` first preloads the LLM model with
180-`POST {JP_LLM_BASE_URL}/models/load` (30s timeout) before anything else loads,
181-then makes only safe, bounded HTTP readiness checks (a few seconds each, no
182-inference requests). It then sends one best-effort warmup
183+At startup, `jp` makes only safe, bounded HTTP readiness checks (a few seconds
184+each, no inference requests). It then sends one best-effort warmup
185 completion per system prompt (`max_tokens: 1`) so the server's prompt cache
186 keeps those prefixes hot. It stays silent when every service answers; a down
187 service prints one error line naming the service and its URL. The game never
188 launches or supervises a service to make a check pass.
189
190-## LLM - llama.cpp (Unsloth Qwen3-8B)
191+## LLM - llama.cpp (Unsloth Gemma 4 12B)
192
193-llama.cpp server in router mode (started without a model argument) at
194-`JP_LLM_BASE_URL`, with OpenAI-compatible chat completions at
195-`POST {JP_LLM_BASE_URL}/v1/chat/completions`.
196+llama.cpp server in single-instance mode at `JP_LLM_BASE_URL`, with
197+OpenAI-compatible chat completions at `POST {JP_LLM_BASE_URL}/v1/chat/completions`.
198
199-Model: `unsloth/Qwen3-8B-GGUF:UD-Q4_K_XL`, resolvable by the router as `jp`. The client
200-sends OpenAI-compatible requests with `model` set to `jp`, sends
201-`chat_template_kwargs.enable_thinking` per the `--enable-thinking` flag
202-(default false, which disables Qwen3 thinking), sets `cache_prompt: true` so
203-the server keeps prompt prefixes cached between turns, pins each request to a
204+Model: `unsloth/gemma-4-12B-it-qat-GGUF:UD-Q4_K_XL`. The client sends
205+OpenAI-compatible requests without a model field (the single-instance server
206+runs exactly one model), sends `chat_template_kwargs.enable_thinking` per the
207+`--enable-thinking` flag (default false), sets `cache_prompt: true` so the
208+server keeps prompt prefixes cached between turns, pins each request to a
209 llama-server slot via `id_slot` (see below), and reads streaming SSE. The model
210 chat template (`--jinja`) is used; ChatML is not built manually by the client.
211
212 ### Slot pinning
213
214 Each request body carries an `id_slot` field that assigns the task to a
215-specific llama-server slot, bypassing the server's LRU/similarity
216-auto-selection. The game uses two fixed slots so scratch requests can never
217-evict the game loop's cached context: game-loop requests pin `id_slot: 0`, and
218-scratch requests (judge, compaction, character sheets; warmup included) pin
219-`id_slot: 1`. These are hardcoded constants, not user configuration.
220diff --git a/internal/config/config.go b/internal/config/config.go
221index ddad64a789baa1505ef1db0947f5fac09b58c852..33da5bcf950a5930172f5512d228567490338c70 100644
222--- a/internal/config/config.go
223+++ b/internal/config/config.go
224@@ -13,7 +13,6 @@ import (
225
226 type LLMConfig struct {
227 BaseURL string `long:"url" env:"JP_LLM_BASE_URL" default:"https://llama.home.theedgeofrage.com" description:"llama.cpp router base URL"`
228- Model string `long:"model" env:"MODEL" default:"jp"`
229 Temperature float64 `long:"temperature" env:"TEMPERATURE" default:"1.0"`
230 MaxTokens int `long:"max-tokens" env:"MAX_TOKENS" default:"256"`
231 EnableThinking bool `long:"enable-thinking" env:"ENABLE_THINKING"`
232diff --git a/internal/llm/client.go b/internal/llm/client.go
233index c13cc8c70c26bc30c075a04836496c1ee958bb0b..cd463c8e3b1c422d08965d77b29eb3778eac880a 100644
234--- a/internal/llm/client.go
235+++ b/internal/llm/client.go
236@@ -14,7 +14,6 @@ import (
237 "japanese/internal/config"
238 "net/http"
239 "strings"
240- "time"
241 )
242
243 type Role string
244@@ -33,7 +32,6 @@ type Message struct {
245 type Client struct {
246 baseURL string
247 httpClient *http.Client
248- Model string
249 Temperature float64
250 MaxTokens int
251 EnableThinking bool
252@@ -49,7 +47,6 @@ func NewClient(cfg *config.Config, hc *http.Client) *Client {
253 return &Client{
254 httpClient: hc,
255 baseURL: cfg.LLMConfig.BaseURL,
256- Model: cfg.LLMConfig.Model,
257 Temperature: cfg.LLMConfig.Temperature,
258 MaxTokens: cfg.LLMConfig.MaxTokens,
259 EnableThinking: cfg.LLMConfig.EnableThinking,
260@@ -57,7 +54,6 @@ func NewClient(cfg *config.Config, hc *http.Client) *Client {
261 }
262
263 type chatRequest struct {
264- Model string `json:"model"`
265 Messages []Message `json:"messages"`
266 ChatTemplateKwargs map[string]any `json:"chat_template_kwargs"`
267 Temperature float64 `json:"temperature"`
268@@ -100,7 +96,6 @@ func (c *Client) chat(ctx context.Context, msgs []Message, maxTokens int) (strin
269 return "", fmt.Errorf("llm: no base URL configured")
270 }
271 body := chatRequest{
272- Model: c.Model,
273 Messages: msgs,
274 ChatTemplateKwargs: map[string]any{"enable_thinking": c.EnableThinking},
275 Temperature: c.Temperature,
276@@ -139,47 +134,6 @@ func (c *Client) chat(ctx context.Context, msgs []Message, maxTokens int) (strin
277 return text, nil
278 }
279
280-// loadTimeout bounds the model preload request. Loading a large model into
281-// VRAM takes much longer than a normal API call.
282-const loadTimeout = 30 * time.Second
283-
284-// LoadModel asks the llama.cpp router to load the configured model so the
285-// first turn does not pay the loading cost. It is safe to call when the model
286-// is already loaded; the router keeps it loaded.
287-func (c *Client) LoadModel() error {
288- if c.baseURL == "" {
289- return fmt.Errorf("llm: no base URL configured")
290- }
291- ctx, cancel := context.WithTimeout(context.Background(), loadTimeout)
292- defer cancel()
293-
294- payload, err := json.Marshal(map[string]string{"model": c.Model})
295- if err != nil {
296- return fmt.Errorf("llm: encode request: %w", err)
297- }
298- url := strings.TrimRight(c.baseURL, "/") + "/models/load"
299- req, err := http.NewRequestWithContext(ctx, http.MethodPost, url, bytes.NewReader(payload))
300- if err != nil {
301- return fmt.Errorf("llm: build request: %w", err)
302- }
303- req.Header.Set("Content-Type", "application/json")
304-
305- resp, err := c.httpClient.Do(req)
306- if err != nil {
307- return fmt.Errorf("llm: request to %s: %w", url, err)
308- }
309- defer func() { _ = resp.Body.Close() }()
310-
311- if resp.StatusCode < 200 || resp.StatusCode >= 300 {
312- detail, _ := io.ReadAll(io.LimitReader(resp.Body, 4096))
313- if strings.Contains(string(detail), "already running") {
314- return nil
315- }
316- return fmt.Errorf("llm: %s returned HTTP %d: %s", url, resp.StatusCode, strings.TrimSpace(string(detail)))
317- }
318- return nil
319-}
320-
321 func readSSE(r io.Reader) (string, error) {
322 var sb strings.Builder
323 scanner := bufio.NewScanner(r)