06d4b8111a62254a6c44a8db4cab77b41ede91c5
- Author
- TheEdgeOfRage <git@theedgeofrage.com>
- Committer
- TheEdgeOfRage <git@theedgeofrage.com>
- Date
Message
Diff
This diff is truncated to protect this page.
1diff --git a/internal/availability/availability.go b/internal/availability/availability.go
2deleted file mode 100644
3index 49de61bcca61f963c68934d6b22e8ae767e115b8..0000000000000000000000000000000000000000
4--- a/internal/availability/availability.go
5+++ /dev/null
6@@ -1,63 +0,0 @@
7-// Package availability reports whether the configured model services are
8-// reachable using only safe, bounded HTTP checks. It never sends an inference
9-// request and never starts, stops, or reconfigures a service.
10-package availability
11-
12-import (
13- "encoding/json"
14- "fmt"
15- "io"
16- "net/http"
17- "strings"
18-)
19-
20-// CheckAudio verifies the audio.cpp server answers at /v1/models and lists
21-// every configured model id.
22-func CheckAudio(client *http.Client, base string, modelIDs []string) error {
23- url := strings.TrimRight(base, "/") + "/v1/models"
24- status, raw, err := doGet(client, url)
25- if err != nil {
26- return fmt.Errorf("GET %s: %w", url, err)
27- }
28- if status < 200 || status >= 300 {
29- return fmt.Errorf("GET %s: HTTP %d", url, status)
30- }
31-
32- var out struct {
33- Data []struct {
34- ID string `json:"id"`
35- } `json:"data"`
36- }
37- if err := json.Unmarshal(raw, &out); err != nil {
38- return fmt.Errorf("decode models from %s: %w", url, err)
39- }
40-
41- listed := make(map[string]bool, len(out.Data))
42- for _, m := range out.Data {
43- listed[m.ID] = true
44- }
45- var missing []string
46- for _, id := range modelIDs {
47- if !listed[id] {
48- missing = append(missing, id)
49- }
50- }
51- if len(missing) > 0 {
52- return fmt.Errorf("model(s) not listed by %s: %s", url, strings.Join(missing, ", "))
53- }
54- return nil
55-}
56-
57-// doGet issues a GET and returns the status code plus the fully read body.
58-func doGet(client *http.Client, target string) (int, []byte, error) {
59- resp, err := client.Get(target)
60- if err != nil {
61- return 0, nil, err
62- }
63- defer func() { _ = resp.Body.Close() }()
64- body, readErr := io.ReadAll(resp.Body)
65- if readErr != nil {
66- return 0, nil, readErr
67- }
68- return resp.StatusCode, body, nil
69-}
70diff --git a/internal/bootstrap/bootstrap.go b/internal/bootstrap/bootstrap.go
71index 14edc838b5c780bf7136e2dbdc2567670abb6fa6..712e023508c7dca46aac3af3344141f50134b9ea 100644
72--- a/internal/bootstrap/bootstrap.go
73+++ b/internal/bootstrap/bootstrap.go
74@@ -1,29 +1,23 @@
75 // Package bootstrap builds the service clients shared by the kaiwari entry points:
76-// scenario load, one LLM client per prompt slot, the audio availability check,
77-// the LLM readiness wait, and the prompt warmup. Warnings (unavailable audio,
78-// not-ready LLM, failed warmup) print to stderr; only a bad scenario file is
79-// returned as an error.
80+// scenario load, self-hosted model services (unless disabled), one LLM client per
81+// prompt slot, and the prompt warmup. Every failure is returned as an error.
82 package bootstrap
83
84 import (
85 "context"
86 "fmt"
87 "net/http"
88- "os"
89 "time"
90
91- "git.theedgeofrage.com/TheEdgeOfRage/kaiwari/internal/availability"
92 "git.theedgeofrage.com/TheEdgeOfRage/kaiwari/internal/config"
93 "git.theedgeofrage.com/TheEdgeOfRage/kaiwari/internal/llm"
94 "git.theedgeofrage.com/TheEdgeOfRage/kaiwari/internal/scenario"
95- "git.theedgeofrage.com/TheEdgeOfRage/kaiwari/internal/stt"
96- "git.theedgeofrage.com/TheEdgeOfRage/kaiwari/internal/tts"
97+ "git.theedgeofrage.com/TheEdgeOfRage/kaiwari/internal/services"
98 )
99
100 const (
101- requestTimeout = 10 * time.Second
102- llmReadyTimeout = 30 * time.Second
103- warmupTimeout = 30 * time.Second
104+ requestTimeout = 10 * time.Second
105+ warmupTimeout = 30 * time.Second
106
107 // llama-server slot assignments are fixed for this deployment; they are not
108 // user-configurable. One slot per prompt family keeps each system prompt
109@@ -46,32 +40,37 @@ type Services struct {
110 Compaction *llm.Client
111 Sheet *llm.Client
112 Scratch *llm.Client
113+
114+ manager *services.Manager
115 }
116
117-// New loads the scenario, builds one LLM client per slot, checks audio
118-// availability, waits for the game client, and warms up the prompt slots.
119+// New loads the scenario, starts the self-hosted model services unless disabled,
120+// builds one LLM client per slot, and warms up the prompt slots. Any failure
121+// tears down whatever was started and is returned as an error.
122 func New(cfg *config.Config) (*Services, error) {
123 sc, err := scenario.Load(cfg.Scenario)
124 if err != nil {
125 return nil, fmt.Errorf("load scenario: %w", err)
126 }
127
128- hc := &http.Client{Timeout: requestTimeout}
129- gameClient := newLLMClient(cfg, hc, gameSlot)
130- judgeClient := newLLMClient(cfg, hc, judgeSlot)
131- compactionClient := newLLMClient(cfg, hc, compactionSlot)
132- sheetClient := newLLMClient(cfg, hc, sheetSlot)
133- scratchClient := newLLMClient(cfg, hc, scratchSlot)
134-
135- if err := availability.CheckAudio(hc, cfg.AudioConfig.AudioBaseURL, []string{tts.ModelName, stt.ASRModelName}); err != nil {
136- fmt.Fprintf(os.Stderr, "kaiwari: audio service unavailable: %v\n", err)
137+ c := *cfg
138+ var manager *services.Manager
139+ if !c.DisableModelLoading {
140+ manager = services.NewManager()
141+ if err := manager.Start(context.Background()); err != nil {
142+ return nil, err
143+ }
144+ // Managed children bind loopback; pin the endpoints regardless of URL flags.
145+ c.LLMConfig.BaseURL = "http://" + services.LLMListen
146+ c.AudioConfig.AudioBaseURL = "http://" + services.AudioListen
147 }
148
149- readyCtx, cancelReady := context.WithTimeout(context.Background(), llmReadyTimeout)
150- if err := gameClient.WaitReady(readyCtx); err != nil {
151- fmt.Fprintf(os.Stderr, "kaiwari: LLM service not ready: %v\n", err)
152- }
153- cancelReady()
154+ hc := &http.Client{Timeout: requestTimeout}
155+ gameClient := newLLMClient(&c, hc, gameSlot)
156+ judgeClient := newLLMClient(&c, hc, judgeSlot)
157+ compactionClient := newLLMClient(&c, hc, compactionSlot)
158+ sheetClient := newLLMClient(&c, hc, sheetSlot)
159+ scratchClient := newLLMClient(&c, hc, scratchSlot)
160
161 warmups := []struct {
162 client *llm.Client
163@@ -91,11 +90,14 @@ func New(cfg *config.Config) (*Services, error) {
164 }
165 }
166 if warmErr != nil {
167- fmt.Fprintf(os.Stderr, "kaiwari: LLM warmup failed: %v\n", warmErr)
168+ if manager != nil {
169+ manager.Stop()
170+ }
171+ return nil, fmt.Errorf("llm warmup: %w", warmErr)
172 }
173
174diff --git a/internal/config/config.go b/internal/config/config.go
175index 766f52c641263bbde3ca4abc8be0f896d4819874..31895cf7dce354a04fe94641f5c13b20ef3c49d9 100644
176--- a/internal/config/config.go
177+++ b/internal/config/config.go
178@@ -27,7 +27,8 @@ type Config struct {
179 LLMConfig LLMConfig `group:"llm" namespace:"llm" env-namespace:"JP_LLM"`
180 AudioConfig AudioConfig `group:"audio" namespace:"audio" env-namespace:"JP_AUDIO"`
181
182- Scenario string `long:"scenario" env:"JP_SCENARIO_PATH" default:"assets/scenarios/small_city.md" description:"path to the scenario brief file"`
183+ Scenario string `long:"scenario" env:"JP_SCENARIO_PATH" default:"assets/scenarios/small_city.md" description:"path to the scenario brief file"`
184+ DisableModelLoading bool `long:"disable-model-loading" env:"JP_DISABLE_MODEL_LOADING" default:"false" description:"connect to external model services instead of spawning them"`
185 }
186
187 // ServerConfig is the kaiwari-server option group: the shared Config plus the HTTP
188diff --git a/internal/llm/client.go b/internal/llm/client.go
189index d78482b3650dac97ada2de312fd1573727b0b2ea..c4b0ca9c15758d08e22a5ae61cc358d99444fcb0 100644
190--- a/internal/llm/client.go
191+++ b/internal/llm/client.go
192@@ -14,7 +14,6 @@ import (
193 "io"
194 "net/http"
195 "strings"
196- "time"
197 )
198
199 type Role string
200@@ -79,31 +78,9 @@ func (c *Client) Generate(ctx context.Context, msgs []Message) (string, error) {
201 return c.chat(ctx, msgs, c.MaxTokens)
202 }
203
204-// WaitReady polls the router with a minimal completion once per second until
205-// one succeeds or ctx expires. A successful probe means the next real request
206-// (warmup, turn) will be served.
207-func (c *Client) WaitReady(ctx context.Context) error {
208- if c.baseURL == "" {
209- return fmt.Errorf("llm: no base URL configured")
210- }
211- ticker := time.NewTicker(time.Second)
212- defer ticker.Stop()
213- var lastErr error
214- for {
215- if _, lastErr = c.chat(ctx, []Message{{Role: RoleUser, Content: "Ready."}}, 1); lastErr == nil {
216- return nil
217- }
218- select {
219- case <-ctx.Done():
220- return fmt.Errorf("llm: not ready at %s: %w", c.baseURL, lastErr)
221- case <-ticker.C:
222- }
223- }
224-}
225-
226 // Warmup sends each system prompt once with a minimal user line and one output
227 // token so the server keeps those prompt prefixes in its KV cache before the
228-// first real turn. It is best-effort; callers report failures as warnings.
229+// first real turn.
230 func (c *Client) Warmup(ctx context.Context, systems []string) error {
231 for _, s := range systems {
232 msgs := []Message{{Role: RoleSystem, Content: s}, {Role: RoleUser, Content: "Ready."}}