Diff
1diff --git a/README.md b/README.md
2index 948f9e2ae0bc92745799599bf8e09f41536d04c9..3df5ad7210f324d8f68e68d4e2ddb6559c693a7b 100644
3--- a/README.md
4+++ b/README.md
5@@ -30,10 +30,12 @@ go build ./... # or: make build
6 jp # or: make run (go run ./cmd/jp)
7 ```
8
9-On startup the game runs a bounded readiness check against the audio service,
10-then warms each conversation's system prompt once so first replies are fast. A failed step prints one error line naming the affected
11-service and its configured URL; startup continues either way. The game never
12-launches a service to make a check pass.
13+On startup the game waits up to 30s for the LLM model to finish loading
14+(polling every second), runs a bounded readiness check against the audio
15+service, then warms each conversation's system prompt once so first replies
16+are fast. A failed step prints one error line naming the affected service and
17+its configured URL; startup continues either way. The game never launches a
18+service to make a check pass.
19
20 ## Configuration
21
22diff --git a/cmd/jp/main.go b/cmd/jp/main.go
23index 9f43a4bad397744bc20e4807fee1b509a7818eba..c04981dbf9387a68551c63f0a2fa28b6ed3a8fc6 100644
24--- a/cmd/jp/main.go
25+++ b/cmd/jp/main.go
26@@ -24,6 +24,7 @@ import (
27 // (which enforces it) and to the UI (which shows it as the cap indicator).
28 const recordCap = 15 * time.Second
29 const requestTimeout = 10 * time.Second
30+const llmReadyTimeout = 30 * time.Second
31 const warmupTimeout = 30 * time.Second
32
33 // llama-server slot assignments and the capture command are fixed for this
34@@ -57,6 +58,12 @@ func main() {
35 fmt.Fprintf(os.Stderr, "jp: audio service unavailable: %v\n", err)
36 }
37
38+ readyCtx, cancelReady := context.WithTimeout(context.Background(), llmReadyTimeout)
39+ if err := gameClient.WaitReady(readyCtx); err != nil {
40+ fmt.Fprintf(os.Stderr, "jp: LLM service not ready: %v\n", err)
41+ }
42+ cancelReady()
43+
44 warmups := []struct {
45 client *llm.Client
46 prompt string
47diff --git a/docs/services.md b/docs/services.md
48index 4db6f168136c5e6707d03ac344abb667154c9ae1..882484ab8683cc8655af1132027efcd75c46eb81 100644
49--- a/docs/services.md
50+++ b/docs/services.md
51@@ -14,12 +14,14 @@ match the local setup:
52 | LLM router URL | `--url` | `JP_LLM_BASE_URL` | `https://llama.home.theedgeofrage.com` | `POST /v1/chat/completions` |
53 | Audio base URL | `--audio-url` | `JP_AUDIO_BASE_URL` | `http://127.0.0.1:8080` | TTS `POST /v1/audio/speech` + ASR `POST /v1/audio/transcriptions` |
54
55-At startup, `jp` makes only safe, bounded HTTP readiness checks (a few seconds
56-each, no inference requests). It then sends one best-effort warmup
57-completion per system prompt (`max_tokens: 1`) so the server's prompt cache
58-keeps those prefixes hot. It stays silent when every service answers; a down
59-service prints one error line naming the service and its URL. The game never
60-launches or supervises a service to make a check pass.
61+At startup, `jp` waits for the LLM router to accept completions: a minimal
62+chat completion (`max_tokens: 1`) polled once per second, up to 30s, so a
63+freshly started model has time to load. It then checks the audio server at
64+`/v1/models` (bounded, no inference requests) and sends one best-effort
65+warmup completion per system prompt (`max_tokens: 1`) so the server's prompt
66+cache keeps those prefixes hot. It stays silent when every service answers; a
67+down or slow service prints one error line naming the service and its URL.
68+The game never launches or supervises a service to make a check pass.
69
70 ## LLM - llama.cpp (Unsloth Gemma 4 12B)
71
72diff --git a/internal/llm/client.go b/internal/llm/client.go
73index cd463c8e3b1c422d08965d77b29eb3778eac880a..9c1ff27e52eb35f0997190e5852eeb02c76cae18 100644
74--- a/internal/llm/client.go
75+++ b/internal/llm/client.go
76@@ -14,6 +14,7 @@ import (
77 "japanese/internal/config"
78 "net/http"
79 "strings"
80+ "time"
81 )
82
83 type Role string
84@@ -78,6 +79,28 @@ func (c *Client) Generate(ctx context.Context, msgs []Message) (string, error) {
85 return c.chat(ctx, msgs, c.MaxTokens)
86 }
87
88+// WaitReady polls the router with a minimal completion once per second until
89+// one succeeds or ctx expires. A successful probe means the next real request
90+// (warmup, turn) will be served.
91+func (c *Client) WaitReady(ctx context.Context) error {
92+ if c.baseURL == "" {
93+ return fmt.Errorf("llm: no base URL configured")
94+ }
95+ ticker := time.NewTicker(time.Second)
96+ defer ticker.Stop()
97+ var lastErr error
98+ for {
99+ if _, lastErr = c.chat(ctx, []Message{{Role: RoleUser, Content: "Ready."}}, 1); lastErr == nil {
100+ return nil
101+ }
102+ select {
103+ case <-ctx.Done():
104+ return fmt.Errorf("llm: not ready at %s: %w", c.baseURL, lastErr)
105+ case <-ticker.C:
106+ }
107+ }
108+}
109+
110 // Warmup sends each system prompt once with a minimal user line and one output
111 // token so the server keeps those prompt prefixes in its KV cache before the
112 // first real turn. It is best-effort; callers report failures as warnings.