f647590db132a3863c193a5af63857804243be59

Author
TheEdgeOfRage <git@theedgeofrage.com>
Committer
TheEdgeOfRage <git@theedgeofrage.com>
Date

Message

Hardcode LLM slot assignments and recorder command

Diff

  1diff --git a/README.md b/README.md
  2index 564fd9d3dcbdfe8c8fc243daea3e8cc2dbc74559..55a9f65099d8e8f6e5853568edcd0a6a05c163a1 100644
  3--- a/README.md
  4+++ b/README.md
  5@@ -15,10 +15,8 @@ downloads a model for any of them. You run those services yourself (see
  6 ## Requirements
  7 
  8 - Go 1.27+
  9-- `arecord` (ALSA) for microphone capture. The default bare `arecord` records in
 10-  system-default format; set `--record-command "arecord -f S16_LE -r 16000 -c 1"`
 11-  to get the required 16 kHz mono S16_LE WAV. Any command works as long as it
 12-  writes a 16 kHz mono S16_LE WAV file to the path given as its last argument
 13+- `arecord` (ALSA) for microphone capture. The game invokes bare `arecord`
 14+  (16 kHz mono S16_LE WAV written to the path given as its last argument)
 15 - Two external model services, both run by you:
 16   - **LLM** — llama.cpp router (OpenAI-compatible `POST /v1/chat/completions`)
 17   - **Audio** — audio.cpp server (`JP_AUDIO_BASE_URL`): TTS speech endpoint
 18@@ -50,10 +48,7 @@ which win over defaults. Defaults match a local setup.
 19 | Temperature      | `--temperature`      | `TEMPERATURE`         | `1.0`                                     | sampling temperature                            |
 20 | Max tokens       | `--max-tokens`       | `MAX_TOKENS`          | `256`                                     | max output tokens per reply                     |
 21 | Thinking mode    | `--enable-thinking`  | `ENABLE_THINKING`     | `false`                                   | Qwen3 thinking (`chat_template_kwargs.enable_thinking`) |
 22-| LLM game slot    | `--llm-game-slot`    | `JP_LLM_GAME_SLOT`    | `0`                                       | llama-server slot for game-loop requests (`-1` = server decides) |
 23-| LLM scratch slot | `--llm-scratch-slot` | `JP_LLM_SCRATCH_SLOT` | `1`                                       | llama-server slot for judge, compaction, sheet, and ask requests (`-1` = server decides) |
 24 | Audio base URL   | `--audio-url`        | `JP_AUDIO_BASE_URL`   | `http://127.0.0.1:8080`                  | audio.cpp server (TTS + ASR)                    |
 25-| Recorder command | `--record-command`   | —                     | `arecord`                                 | 16 kHz mono S16_LE WAV capture (see note above) |
 26 | Scenario brief   | `--scenario`         | `JP_SCENARIO_PATH`    | `assets/scenarios/small_city.md`          | plain-text scenario brief file                  |
 27 
 28 Run `jp -help` for the full flag list.
 29diff --git a/cmd/jp/main.go b/cmd/jp/main.go
 30index 72308cd7fd897afea3293e57cf529a9c235f569d..9176f6b1d4b6d449d04bc5631d2886964fb9e7d0 100644
 31--- a/cmd/jp/main.go
 32+++ b/cmd/jp/main.go
 33@@ -26,6 +26,12 @@ const recordCap = 10 * time.Second
 34 const requestTimeout = 10 * time.Second
 35 const warmupTimeout = 60 * time.Second
 36 
 37+// llama-server slot assignments and the capture command are fixed for this
 38+// deployment; they are not user-configurable.
 39+const gameSlot = 0
 40+const scratchSlot = 1
 41+const recordCommand = "arecord"
 42+
 43 func main() {
 44 	cfg := config.ParseConfig()
 45 
 46@@ -36,9 +42,9 @@ func main() {
 47 
 48 	hc := &http.Client{Timeout: requestTimeout}
 49 	gameClient := llm.NewClient(cfg, hc)
 50-	gameClient.Slot = cfg.LLMConfig.GameSlot
 51+	gameClient.Slot = gameSlot
 52 	scratchClient := llm.NewClient(cfg, hc)
 53-	scratchClient.Slot = cfg.LLMConfig.ScratchSlot
 54+	scratchClient.Slot = scratchSlot
 55 
 56 	llmUp := true
 57 	if err := gameClient.LoadModel(); err != nil {
 58@@ -78,7 +84,7 @@ func buildOrchestrator(cfg *config.Config, state *game.State, hc *http.Client, g
 59 	askModel := &adapters.ScratchModel{Client: scratchClient}
 60 
 61 	speechIn := &adapters.SpeechInput{
 62-		Recorder: stt.NewRecorder(cfg.RecordCommand, recordCap),
 63+		Recorder: stt.NewRecorder(recordCommand, recordCap),
 64 		ASR:      stt.NewASRClient(cfg.AudioBaseURL, hc),
 65 	}
 66 	speechOut := &adapters.SpeechOutput{
 67diff --git a/docs/services.md b/docs/services.md
 68index b78903586fb8d0cf45c2d2c6e755468d6d7d920e..31b9d573e473527c9152765edf27c64cd0b51880 100644
 69--- a/docs/services.md
 70+++ b/docs/services.md
 71@@ -13,7 +13,6 @@ match the local setup:
 72 | ---------------- | ------------------ | ------------------- | -------------------------------------- | ----------------------------------------------------------------- |
 73 | LLM router URL   | `--url`            | `JP_LLM_BASE_URL`   | `https://llama.home.theedgeofrage.com` | `POST /v1/chat/completions`                                       |
 74 | Audio base URL   | `--audio-url`      | `JP_AUDIO_BASE_URL` | `http://127.0.0.1:8080`                | TTS `POST /v1/audio/speech` + ASR `POST /v1/audio/transcriptions` |
 75-| Recorder command | `--record-command` | -                   | `arecord`                              | 16kHz mono S16_LE WAV capture                                     |
 76 
 77 At startup, `jp` first preloads the LLM model with
 78 `POST {JP_LLM_BASE_URL}/models/load` (30s timeout) before anything else loads,
 79@@ -42,22 +41,15 @@ chat template (`--jinja`) is used; ChatML is not built manually by the client.
 80 
 81 Each request body carries an `id_slot` field that assigns the task to a
 82 specific llama-server slot, bypassing the server's LRU/similarity
 83-auto-selection. The game uses two slots so scratch requests can never evict
 84-the game loop's cached context:
 85-
 86-| Client use                                                     | Flag                 | Env var               | Default |
 87-| -------------------------------------------------------------- | -------------------- | --------------------- | ------- |
 88-| Game loop (world replies)                                      | `--llm-game-slot`    | `JP_LLM_GAME_SLOT`    | `0`     |
 89-| Scratch (judge, compaction, character sheets; warmup included) | `--llm-scratch-slot` | `JP_LLM_SCRATCH_SLOT` | `1`     |
 90+auto-selection. The game uses two fixed slots so scratch requests can never
 91+evict the game loop's cached context: game-loop requests pin `id_slot: 0`, and
 92+scratch requests (judge, compaction, character sheets; warmup included) pin
 93+`id_slot: 1`. These are hardcoded constants, not user configuration.
 94 
 95 Warmup follows the same split: the game system prompt warms on the game
 96 client, the judge, compaction, and sheet prompts warm on the scratch client,
 97 so initial cache placement is deterministic per slot.
 98 
 99-**Fallback:** if the router does not honor `id_slot`, set both env vars to `-1`
100-(`JP_LLM_GAME_SLOT=-1 JP_LLM_SCRATCH_SLOT=-1`). The server then auto-selects a
101-slot by prompt similarity (`-sps`, default 0.10) and falls back to LRU.
102-
103 At startup, before anything else loads, the game preloads the model with
104 `POST {JP_LLM_BASE_URL}/models/load` and body `{"model": "jp"}` (30s timeout).
105 A failed preload prints one error line and startup continues.
106diff --git a/internal/config/config.go b/internal/config/config.go
107index 0f128a53a4a2eb837a384afeba6cdfe4eb78a5c1..ddad64a789baa1505ef1db0947f5fac09b58c852 100644
108--- a/internal/config/config.go
109+++ b/internal/config/config.go
110@@ -17,16 +17,13 @@ type LLMConfig struct {
111 	Temperature    float64 `long:"temperature" env:"TEMPERATURE" default:"1.0"`
112 	MaxTokens      int     `long:"max-tokens" env:"MAX_TOKENS" default:"256"`
113 	EnableThinking bool    `long:"enable-thinking" env:"ENABLE_THINKING"`
114-	GameSlot       int     `long:"llm-game-slot" env:"JP_LLM_GAME_SLOT" default:"0" description:"llama-server slot for game-loop requests (-1 = server decides)"`
115-	ScratchSlot    int     `long:"llm-scratch-slot" env:"JP_LLM_SCRATCH_SLOT" default:"1" description:"llama-server slot for judge, compaction, sheet, and ask requests (-1 = server decides)"`
116 }
117 
118 // Config is the shared option group. Parsed with github.com/jessevdk/go-flags.
119 type Config struct {
120 	LLMConfig LLMConfig
121 
122-	AudioBaseURL  string `long:"audio-url" env:"JP_AUDIO_BASE_URL" default:"http://127.0.0.1:8080" description:"audio.cpp server base URL (TTS + ASR)"`
123-	RecordCommand string `long:"record-command" default:"arecord" description:"Recorder command for 16kHz mono S16_LE WAV capture"`
124+	AudioBaseURL string `long:"audio-url" env:"JP_AUDIO_BASE_URL" default:"http://127.0.0.1:8080" description:"audio.cpp server base URL (TTS + ASR)"`
125 
126 	Scenario string `long:"scenario" env:"JP_SCENARIO_PATH" default:"assets/scenarios/small_city.md" description:"path to the scenario brief file"`
127 }