06d4b8111a62254a6c44a8db4cab77b41ede91c5

Author
TheEdgeOfRage <git@theedgeofrage.com>
Committer
TheEdgeOfRage <git@theedgeofrage.com>
Date

Message

Wire services.Manager into bootstrap; drop availability checks

Diff

This diff is truncated to protect this page.

  1diff --git a/internal/availability/availability.go b/internal/availability/availability.go
  2deleted file mode 100644
  3index 49de61bcca61f963c68934d6b22e8ae767e115b8..0000000000000000000000000000000000000000
  4--- a/internal/availability/availability.go
  5+++ /dev/null
  6@@ -1,63 +0,0 @@
  7-// Package availability reports whether the configured model services are
  8-// reachable using only safe, bounded HTTP checks. It never sends an inference
  9-// request and never starts, stops, or reconfigures a service.
 10-package availability
 11-
 12-import (
 13-	"encoding/json"
 14-	"fmt"
 15-	"io"
 16-	"net/http"
 17-	"strings"
 18-)
 19-
 20-// CheckAudio verifies the audio.cpp server answers at /v1/models and lists
 21-// every configured model id.
 22-func CheckAudio(client *http.Client, base string, modelIDs []string) error {
 23-	url := strings.TrimRight(base, "/") + "/v1/models"
 24-	status, raw, err := doGet(client, url)
 25-	if err != nil {
 26-		return fmt.Errorf("GET %s: %w", url, err)
 27-	}
 28-	if status < 200 || status >= 300 {
 29-		return fmt.Errorf("GET %s: HTTP %d", url, status)
 30-	}
 31-
 32-	var out struct {
 33-		Data []struct {
 34-			ID string `json:"id"`
 35-		} `json:"data"`
 36-	}
 37-	if err := json.Unmarshal(raw, &out); err != nil {
 38-		return fmt.Errorf("decode models from %s: %w", url, err)
 39-	}
 40-
 41-	listed := make(map[string]bool, len(out.Data))
 42-	for _, m := range out.Data {
 43-		listed[m.ID] = true
 44-	}
 45-	var missing []string
 46-	for _, id := range modelIDs {
 47-		if !listed[id] {
 48-			missing = append(missing, id)
 49-		}
 50-	}
 51-	if len(missing) > 0 {
 52-		return fmt.Errorf("model(s) not listed by %s: %s", url, strings.Join(missing, ", "))
 53-	}
 54-	return nil
 55-}
 56-
 57-// doGet issues a GET and returns the status code plus the fully read body.
 58-func doGet(client *http.Client, target string) (int, []byte, error) {
 59-	resp, err := client.Get(target)
 60-	if err != nil {
 61-		return 0, nil, err
 62-	}
 63-	defer func() { _ = resp.Body.Close() }()
 64-	body, readErr := io.ReadAll(resp.Body)
 65-	if readErr != nil {
 66-		return 0, nil, readErr
 67-	}
 68-	return resp.StatusCode, body, nil
 69-}
 70diff --git a/internal/bootstrap/bootstrap.go b/internal/bootstrap/bootstrap.go
 71index 14edc838b5c780bf7136e2dbdc2567670abb6fa6..712e023508c7dca46aac3af3344141f50134b9ea 100644
 72--- a/internal/bootstrap/bootstrap.go
 73+++ b/internal/bootstrap/bootstrap.go
 74@@ -1,29 +1,23 @@
 75 // Package bootstrap builds the service clients shared by the kaiwari entry points:
 76-// scenario load, one LLM client per prompt slot, the audio availability check,
 77-// the LLM readiness wait, and the prompt warmup. Warnings (unavailable audio,
 78-// not-ready LLM, failed warmup) print to stderr; only a bad scenario file is
 79-// returned as an error.
 80+// scenario load, self-hosted model services (unless disabled), one LLM client per
 81+// prompt slot, and the prompt warmup. Every failure is returned as an error.
 82 package bootstrap
 83 
 84 import (
 85 	"context"
 86 	"fmt"
 87 	"net/http"
 88-	"os"
 89 	"time"
 90 
 91-	"git.theedgeofrage.com/TheEdgeOfRage/kaiwari/internal/availability"
 92 	"git.theedgeofrage.com/TheEdgeOfRage/kaiwari/internal/config"
 93 	"git.theedgeofrage.com/TheEdgeOfRage/kaiwari/internal/llm"
 94 	"git.theedgeofrage.com/TheEdgeOfRage/kaiwari/internal/scenario"
 95-	"git.theedgeofrage.com/TheEdgeOfRage/kaiwari/internal/stt"
 96-	"git.theedgeofrage.com/TheEdgeOfRage/kaiwari/internal/tts"
 97+	"git.theedgeofrage.com/TheEdgeOfRage/kaiwari/internal/services"
 98 )
 99 
100 const (
101-	requestTimeout  = 10 * time.Second
102-	llmReadyTimeout = 30 * time.Second
103-	warmupTimeout   = 30 * time.Second
104+	requestTimeout = 10 * time.Second
105+	warmupTimeout  = 30 * time.Second
106 
107 	// llama-server slot assignments are fixed for this deployment; they are not
108 	// user-configurable. One slot per prompt family keeps each system prompt
109@@ -46,32 +40,37 @@ type Services struct {
110 	Compaction *llm.Client
111 	Sheet      *llm.Client
112 	Scratch    *llm.Client
113+
114+	manager *services.Manager
115 }
116 
117-// New loads the scenario, builds one LLM client per slot, checks audio
118-// availability, waits for the game client, and warms up the prompt slots.
119+// New loads the scenario, starts the self-hosted model services unless disabled,
120+// builds one LLM client per slot, and warms up the prompt slots. Any failure
121+// tears down whatever was started and is returned as an error.
122 func New(cfg *config.Config) (*Services, error) {
123 	sc, err := scenario.Load(cfg.Scenario)
124 	if err != nil {
125 		return nil, fmt.Errorf("load scenario: %w", err)
126 	}
127 
128-	hc := &http.Client{Timeout: requestTimeout}
129-	gameClient := newLLMClient(cfg, hc, gameSlot)
130-	judgeClient := newLLMClient(cfg, hc, judgeSlot)
131-	compactionClient := newLLMClient(cfg, hc, compactionSlot)
132-	sheetClient := newLLMClient(cfg, hc, sheetSlot)
133-	scratchClient := newLLMClient(cfg, hc, scratchSlot)
134-
135-	if err := availability.CheckAudio(hc, cfg.AudioConfig.AudioBaseURL, []string{tts.ModelName, stt.ASRModelName}); err != nil {
136-		fmt.Fprintf(os.Stderr, "kaiwari: audio service unavailable: %v\n", err)
137+	c := *cfg
138+	var manager *services.Manager
139+	if !c.DisableModelLoading {
140+		manager = services.NewManager()
141+		if err := manager.Start(context.Background()); err != nil {
142+			return nil, err
143+		}
144+		// Managed children bind loopback; pin the endpoints regardless of URL flags.
145+		c.LLMConfig.BaseURL = "http://" + services.LLMListen
146+		c.AudioConfig.AudioBaseURL = "http://" + services.AudioListen
147 	}
148 
149-	readyCtx, cancelReady := context.WithTimeout(context.Background(), llmReadyTimeout)
150-	if err := gameClient.WaitReady(readyCtx); err != nil {
151-		fmt.Fprintf(os.Stderr, "kaiwari: LLM service not ready: %v\n", err)
152-	}
153-	cancelReady()
154+	hc := &http.Client{Timeout: requestTimeout}
155+	gameClient := newLLMClient(&c, hc, gameSlot)
156+	judgeClient := newLLMClient(&c, hc, judgeSlot)
157+	compactionClient := newLLMClient(&c, hc, compactionSlot)
158+	sheetClient := newLLMClient(&c, hc, sheetSlot)
159+	scratchClient := newLLMClient(&c, hc, scratchSlot)
160 
161 	warmups := []struct {
162 		client *llm.Client
163@@ -91,11 +90,14 @@ func New(cfg *config.Config) (*Services, error) {
164 		}
165 	}
166 	if warmErr != nil {
167-		fmt.Fprintf(os.Stderr, "kaiwari: LLM warmup failed: %v\n", warmErr)
168+		if manager != nil {
169+			manager.Stop()
170+		}
171+		return nil, fmt.Errorf("llm warmup: %w", warmErr)
172 	}
173 
174diff --git a/internal/config/config.go b/internal/config/config.go
175index 766f52c641263bbde3ca4abc8be0f896d4819874..31895cf7dce354a04fe94641f5c13b20ef3c49d9 100644
176--- a/internal/config/config.go
177+++ b/internal/config/config.go
178@@ -27,7 +27,8 @@ type Config struct {
179 	LLMConfig   LLMConfig   `group:"llm" namespace:"llm" env-namespace:"JP_LLM"`
180 	AudioConfig AudioConfig `group:"audio" namespace:"audio" env-namespace:"JP_AUDIO"`
181 
182-	Scenario string `long:"scenario" env:"JP_SCENARIO_PATH" default:"assets/scenarios/small_city.md" description:"path to the scenario brief file"`
183+	Scenario            string `long:"scenario" env:"JP_SCENARIO_PATH" default:"assets/scenarios/small_city.md" description:"path to the scenario brief file"`
184+	DisableModelLoading bool   `long:"disable-model-loading" env:"JP_DISABLE_MODEL_LOADING" default:"false" description:"connect to external model services instead of spawning them"`
185 }
186 
187 // ServerConfig is the kaiwari-server option group: the shared Config plus the HTTP
188diff --git a/internal/llm/client.go b/internal/llm/client.go
189index d78482b3650dac97ada2de312fd1573727b0b2ea..c4b0ca9c15758d08e22a5ae61cc358d99444fcb0 100644
190--- a/internal/llm/client.go
191+++ b/internal/llm/client.go
192@@ -14,7 +14,6 @@ import (
193 	"io"
194 	"net/http"
195 	"strings"
196-	"time"
197 )
198 
199 type Role string
200@@ -79,31 +78,9 @@ func (c *Client) Generate(ctx context.Context, msgs []Message) (string, error) {
201 	return c.chat(ctx, msgs, c.MaxTokens)
202 }
203 
204-// WaitReady polls the router with a minimal completion once per second until
205-// one succeeds or ctx expires. A successful probe means the next real request
206-// (warmup, turn) will be served.
207-func (c *Client) WaitReady(ctx context.Context) error {
208-	if c.baseURL == "" {
209-		return fmt.Errorf("llm: no base URL configured")
210-	}
211-	ticker := time.NewTicker(time.Second)
212-	defer ticker.Stop()
213-	var lastErr error
214-	for {
215-		if _, lastErr = c.chat(ctx, []Message{{Role: RoleUser, Content: "Ready."}}, 1); lastErr == nil {
216-			return nil
217-		}
218-		select {
219-		case <-ctx.Done():
220-			return fmt.Errorf("llm: not ready at %s: %w", c.baseURL, lastErr)
221-		case <-ticker.C:
222-		}
223-	}
224-}
225-
226 // Warmup sends each system prompt once with a minimal user line and one output
227 // token so the server keeps those prompt prefixes in its KV cache before the
228-// first real turn. It is best-effort; callers report failures as warnings.
229+// first real turn.
230 func (c *Client) Warmup(ctx context.Context, systems []string) error {
231 	for _, s := range systems {
232 		msgs := []Message{{Role: RoleSystem, Content: s}, {Role: RoleUser, Content: "Ready."}}