ff25f4b1f748def6c3aa6053fa9726e9cc347953

Author
TheEdgeOfRage <git@theedgeofrage.com>
Committer
TheEdgeOfRage <git@theedgeofrage.com>
Date

Message

Wait for LLM model to load at startup

Diff

  1diff --git a/README.md b/README.md
  2index 948f9e2ae0bc92745799599bf8e09f41536d04c9..3df5ad7210f324d8f68e68d4e2ddb6559c693a7b 100644
  3--- a/README.md
  4+++ b/README.md
  5@@ -30,10 +30,12 @@ go build ./...             # or: make build
  6 jp                         # or: make run (go run ./cmd/jp)
  7 ```
  8 
  9-On startup the game runs a bounded readiness check against the audio service,
 10-then warms each conversation's system prompt once so first replies are fast. A failed step prints one error line naming the affected
 11-service and its configured URL; startup continues either way. The game never
 12-launches a service to make a check pass.
 13+On startup the game waits up to 30s for the LLM model to finish loading
 14+(polling every second), runs a bounded readiness check against the audio
 15+service, then warms each conversation's system prompt once so first replies
 16+are fast. A failed step prints one error line naming the affected service and
 17+its configured URL; startup continues either way. The game never launches a
 18+service to make a check pass.
 19 
 20 ## Configuration
 21 
 22diff --git a/cmd/jp/main.go b/cmd/jp/main.go
 23index 9f43a4bad397744bc20e4807fee1b509a7818eba..c04981dbf9387a68551c63f0a2fa28b6ed3a8fc6 100644
 24--- a/cmd/jp/main.go
 25+++ b/cmd/jp/main.go
 26@@ -24,6 +24,7 @@ import (
 27 // (which enforces it) and to the UI (which shows it as the cap indicator).
 28 const recordCap = 15 * time.Second
 29 const requestTimeout = 10 * time.Second
 30+const llmReadyTimeout = 30 * time.Second
 31 const warmupTimeout = 30 * time.Second
 32 
 33 // llama-server slot assignments and the capture command are fixed for this
 34@@ -57,6 +58,12 @@ func main() {
 35 		fmt.Fprintf(os.Stderr, "jp: audio service unavailable: %v\n", err)
 36 	}
 37 
 38+	readyCtx, cancelReady := context.WithTimeout(context.Background(), llmReadyTimeout)
 39+	if err := gameClient.WaitReady(readyCtx); err != nil {
 40+		fmt.Fprintf(os.Stderr, "jp: LLM service not ready: %v\n", err)
 41+	}
 42+	cancelReady()
 43+
 44 	warmups := []struct {
 45 		client *llm.Client
 46 		prompt string
 47diff --git a/docs/services.md b/docs/services.md
 48index 4db6f168136c5e6707d03ac344abb667154c9ae1..882484ab8683cc8655af1132027efcd75c46eb81 100644
 49--- a/docs/services.md
 50+++ b/docs/services.md
 51@@ -14,12 +14,14 @@ match the local setup:
 52 | LLM router URL   | `--url`            | `JP_LLM_BASE_URL`   | `https://llama.home.theedgeofrage.com` | `POST /v1/chat/completions`                                       |
 53 | Audio base URL   | `--audio-url`      | `JP_AUDIO_BASE_URL` | `http://127.0.0.1:8080`                | TTS `POST /v1/audio/speech` + ASR `POST /v1/audio/transcriptions` |
 54 
 55-At startup, `jp` makes only safe, bounded HTTP readiness checks (a few seconds
 56-each, no inference requests). It then sends one best-effort warmup
 57-completion per system prompt (`max_tokens: 1`) so the server's prompt cache
 58-keeps those prefixes hot. It stays silent when every service answers; a down
 59-service prints one error line naming the service and its URL. The game never
 60-launches or supervises a service to make a check pass.
 61+At startup, `jp` waits for the LLM router to accept completions: a minimal
 62+chat completion (`max_tokens: 1`) polled once per second, up to 30s, so a
 63+freshly started model has time to load. It then checks the audio server at
 64+`/v1/models` (bounded, no inference requests) and sends one best-effort
 65+warmup completion per system prompt (`max_tokens: 1`) so the server's prompt
 66+cache keeps those prefixes hot. It stays silent when every service answers; a
 67+down or slow service prints one error line naming the service and its URL.
 68+The game never launches or supervises a service to make a check pass.
 69 
 70 ## LLM - llama.cpp (Unsloth Gemma 4 12B)
 71 
 72diff --git a/internal/llm/client.go b/internal/llm/client.go
 73index cd463c8e3b1c422d08965d77b29eb3778eac880a..9c1ff27e52eb35f0997190e5852eeb02c76cae18 100644
 74--- a/internal/llm/client.go
 75+++ b/internal/llm/client.go
 76@@ -14,6 +14,7 @@ import (
 77 	"japanese/internal/config"
 78 	"net/http"
 79 	"strings"
 80+	"time"
 81 )
 82 
 83 type Role string
 84@@ -78,6 +79,28 @@ func (c *Client) Generate(ctx context.Context, msgs []Message) (string, error) {
 85 	return c.chat(ctx, msgs, c.MaxTokens)
 86 }
 87 
 88+// WaitReady polls the router with a minimal completion once per second until
 89+// one succeeds or ctx expires. A successful probe means the next real request
 90+// (warmup, turn) will be served.
 91+func (c *Client) WaitReady(ctx context.Context) error {
 92+	if c.baseURL == "" {
 93+		return fmt.Errorf("llm: no base URL configured")
 94+	}
 95+	ticker := time.NewTicker(time.Second)
 96+	defer ticker.Stop()
 97+	var lastErr error
 98+	for {
 99+		if _, lastErr = c.chat(ctx, []Message{{Role: RoleUser, Content: "Ready."}}, 1); lastErr == nil {
100+			return nil
101+		}
102+		select {
103+		case <-ctx.Done():
104+			return fmt.Errorf("llm: not ready at %s: %w", c.baseURL, lastErr)
105+		case <-ticker.C:
106+		}
107+	}
108+}
109+
110 // Warmup sends each system prompt once with a minimal user line and one output
111 // token so the server keeps those prompt prefixes in its KV cache before the
112 // first real turn. It is best-effort; callers report failures as warnings.