Diff
1diff --git a/README.md b/README.md
2index 564fd9d3dcbdfe8c8fc243daea3e8cc2dbc74559..55a9f65099d8e8f6e5853568edcd0a6a05c163a1 100644
3--- a/README.md
4+++ b/README.md
5@@ -15,10 +15,8 @@ downloads a model for any of them. You run those services yourself (see
6 ## Requirements
7
8 - Go 1.27+
9-- `arecord` (ALSA) for microphone capture. The default bare `arecord` records in
10- system-default format; set `--record-command "arecord -f S16_LE -r 16000 -c 1"`
11- to get the required 16 kHz mono S16_LE WAV. Any command works as long as it
12- writes a 16 kHz mono S16_LE WAV file to the path given as its last argument
13+- `arecord` (ALSA) for microphone capture. The game invokes bare `arecord`
14+ (16 kHz mono S16_LE WAV written to the path given as its last argument)
15 - Two external model services, both run by you:
16 - **LLM** — llama.cpp router (OpenAI-compatible `POST /v1/chat/completions`)
17 - **Audio** — audio.cpp server (`JP_AUDIO_BASE_URL`): TTS speech endpoint
18@@ -50,10 +48,7 @@ which win over defaults. Defaults match a local setup.
19 | Temperature | `--temperature` | `TEMPERATURE` | `1.0` | sampling temperature |
20 | Max tokens | `--max-tokens` | `MAX_TOKENS` | `256` | max output tokens per reply |
21 | Thinking mode | `--enable-thinking` | `ENABLE_THINKING` | `false` | Qwen3 thinking (`chat_template_kwargs.enable_thinking`) |
22-| LLM game slot | `--llm-game-slot` | `JP_LLM_GAME_SLOT` | `0` | llama-server slot for game-loop requests (`-1` = server decides) |
23-| LLM scratch slot | `--llm-scratch-slot` | `JP_LLM_SCRATCH_SLOT` | `1` | llama-server slot for judge, compaction, sheet, and ask requests (`-1` = server decides) |
24 | Audio base URL | `--audio-url` | `JP_AUDIO_BASE_URL` | `http://127.0.0.1:8080` | audio.cpp server (TTS + ASR) |
25-| Recorder command | `--record-command` | — | `arecord` | 16 kHz mono S16_LE WAV capture (see note above) |
26 | Scenario brief | `--scenario` | `JP_SCENARIO_PATH` | `assets/scenarios/small_city.md` | plain-text scenario brief file |
27
28 Run `jp -help` for the full flag list.
29diff --git a/cmd/jp/main.go b/cmd/jp/main.go
30index 72308cd7fd897afea3293e57cf529a9c235f569d..9176f6b1d4b6d449d04bc5631d2886964fb9e7d0 100644
31--- a/cmd/jp/main.go
32+++ b/cmd/jp/main.go
33@@ -26,6 +26,12 @@ const recordCap = 10 * time.Second
34 const requestTimeout = 10 * time.Second
35 const warmupTimeout = 60 * time.Second
36
37+// llama-server slot assignments and the capture command are fixed for this
38+// deployment; they are not user-configurable.
39+const gameSlot = 0
40+const scratchSlot = 1
41+const recordCommand = "arecord"
42+
43 func main() {
44 cfg := config.ParseConfig()
45
46@@ -36,9 +42,9 @@ func main() {
47
48 hc := &http.Client{Timeout: requestTimeout}
49 gameClient := llm.NewClient(cfg, hc)
50- gameClient.Slot = cfg.LLMConfig.GameSlot
51+ gameClient.Slot = gameSlot
52 scratchClient := llm.NewClient(cfg, hc)
53- scratchClient.Slot = cfg.LLMConfig.ScratchSlot
54+ scratchClient.Slot = scratchSlot
55
56 llmUp := true
57 if err := gameClient.LoadModel(); err != nil {
58@@ -78,7 +84,7 @@ func buildOrchestrator(cfg *config.Config, state *game.State, hc *http.Client, g
59 askModel := &adapters.ScratchModel{Client: scratchClient}
60
61 speechIn := &adapters.SpeechInput{
62- Recorder: stt.NewRecorder(cfg.RecordCommand, recordCap),
63+ Recorder: stt.NewRecorder(recordCommand, recordCap),
64 ASR: stt.NewASRClient(cfg.AudioBaseURL, hc),
65 }
66 speechOut := &adapters.SpeechOutput{
67diff --git a/docs/services.md b/docs/services.md
68index b78903586fb8d0cf45c2d2c6e755468d6d7d920e..31b9d573e473527c9152765edf27c64cd0b51880 100644
69--- a/docs/services.md
70+++ b/docs/services.md
71@@ -13,7 +13,6 @@ match the local setup:
72 | ---------------- | ------------------ | ------------------- | -------------------------------------- | ----------------------------------------------------------------- |
73 | LLM router URL | `--url` | `JP_LLM_BASE_URL` | `https://llama.home.theedgeofrage.com` | `POST /v1/chat/completions` |
74 | Audio base URL | `--audio-url` | `JP_AUDIO_BASE_URL` | `http://127.0.0.1:8080` | TTS `POST /v1/audio/speech` + ASR `POST /v1/audio/transcriptions` |
75-| Recorder command | `--record-command` | - | `arecord` | 16kHz mono S16_LE WAV capture |
76
77 At startup, `jp` first preloads the LLM model with
78 `POST {JP_LLM_BASE_URL}/models/load` (30s timeout) before anything else loads,
79@@ -42,22 +41,15 @@ chat template (`--jinja`) is used; ChatML is not built manually by the client.
80
81 Each request body carries an `id_slot` field that assigns the task to a
82 specific llama-server slot, bypassing the server's LRU/similarity
83-auto-selection. The game uses two slots so scratch requests can never evict
84-the game loop's cached context:
85-
86-| Client use | Flag | Env var | Default |
87-| -------------------------------------------------------------- | -------------------- | --------------------- | ------- |
88-| Game loop (world replies) | `--llm-game-slot` | `JP_LLM_GAME_SLOT` | `0` |
89-| Scratch (judge, compaction, character sheets; warmup included) | `--llm-scratch-slot` | `JP_LLM_SCRATCH_SLOT` | `1` |
90+auto-selection. The game uses two fixed slots so scratch requests can never
91+evict the game loop's cached context: game-loop requests pin `id_slot: 0`, and
92+scratch requests (judge, compaction, character sheets; warmup included) pin
93+`id_slot: 1`. These are hardcoded constants, not user configuration.
94
95 Warmup follows the same split: the game system prompt warms on the game
96 client, the judge, compaction, and sheet prompts warm on the scratch client,
97 so initial cache placement is deterministic per slot.
98
99-**Fallback:** if the router does not honor `id_slot`, set both env vars to `-1`
100-(`JP_LLM_GAME_SLOT=-1 JP_LLM_SCRATCH_SLOT=-1`). The server then auto-selects a
101-slot by prompt similarity (`-sps`, default 0.10) and falls back to LRU.
102-
103 At startup, before anything else loads, the game preloads the model with
104 `POST {JP_LLM_BASE_URL}/models/load` and body `{"model": "jp"}` (30s timeout).
105 A failed preload prints one error line and startup continues.
106diff --git a/internal/config/config.go b/internal/config/config.go
107index 0f128a53a4a2eb837a384afeba6cdfe4eb78a5c1..ddad64a789baa1505ef1db0947f5fac09b58c852 100644
108--- a/internal/config/config.go
109+++ b/internal/config/config.go
110@@ -17,16 +17,13 @@ type LLMConfig struct {
111 Temperature float64 `long:"temperature" env:"TEMPERATURE" default:"1.0"`
112 MaxTokens int `long:"max-tokens" env:"MAX_TOKENS" default:"256"`
113 EnableThinking bool `long:"enable-thinking" env:"ENABLE_THINKING"`
114- GameSlot int `long:"llm-game-slot" env:"JP_LLM_GAME_SLOT" default:"0" description:"llama-server slot for game-loop requests (-1 = server decides)"`
115- ScratchSlot int `long:"llm-scratch-slot" env:"JP_LLM_SCRATCH_SLOT" default:"1" description:"llama-server slot for judge, compaction, sheet, and ask requests (-1 = server decides)"`
116 }
117
118 // Config is the shared option group. Parsed with github.com/jessevdk/go-flags.
119 type Config struct {
120 LLMConfig LLMConfig
121
122- AudioBaseURL string `long:"audio-url" env:"JP_AUDIO_BASE_URL" default:"http://127.0.0.1:8080" description:"audio.cpp server base URL (TTS + ASR)"`
123- RecordCommand string `long:"record-command" default:"arecord" description:"Recorder command for 16kHz mono S16_LE WAV capture"`
124+ AudioBaseURL string `long:"audio-url" env:"JP_AUDIO_BASE_URL" default:"http://127.0.0.1:8080" description:"audio.cpp server base URL (TTS + ASR)"`
125
126 Scenario string `long:"scenario" env:"JP_SCENARIO_PATH" default:"assets/scenarios/small_city.md" description:"path to the scenario brief file"`
127 }