Diff
1diff --git a/AGENTS.md b/AGENTS.md
2index cee212c23d6bfd87b65fdcbf16acf73c73156c4b..eb32202e17037fc18a40c3e7b15a0b590ae28461 100644
3--- a/AGENTS.md
4+++ b/AGENTS.md
5@@ -7,13 +7,13 @@ map, talks to people at locations through push-to-talk, and gets separate
6 fluency feedback on each line. The UI renders romaji only; kana and kanji are
7 never shown.
8
9-The app is an HTTP client for three externally managed model services:
10+The app is an HTTP client for two externally managed model services:
11
12 - **LLM** — llama.cpp router (OpenAI-compatible chat completions) for NPC
13 dialogue and the judge. Prompts live in `internal/llm/prompt.go`; strict
14 `FIELD|value` output contracts are parsed in `internal/llm/contract.go`.
15-- **STT** — whisper.cpp server. The raw Japanese transcript stays internal; it
16- is what the LLM sees as the player's lines.
17+- **STT** — Qwen3-ASR on the shared audio.cpp server. The raw Japanese
18+ transcript stays internal; it is what the LLM sees as the player's lines.
19 - **TTS** — OpenAI-compatible speech endpoint that plays the NPC's kana.
20
21 Turn flow: record → transcribe → judge + NPC reply in parallel → play kana →
22diff --git a/README.md b/README.md
23index ea58e313f560f1b00586a48aa0fefdce49e27b8c..bc7c006d3bff3b0737abd8179fea0d739c6fb182 100644
24--- a/README.md
25+++ b/README.md
26@@ -7,7 +7,7 @@ judge. Every person you talk to gets a hidden character sheet before their
27 first line so they stay consistent across visits. The interface shows romaji
28 only; it never displays kana or kanji.
29
30-The game is an HTTP client only. It connects to three externally managed model
31+The game is an HTTP client only. It connects to two externally managed model
32 services and **never** starts, stops, restarts, kills, reconfigures, or
33 downloads a model for any of them. You run those services yourself (see
34 [`docs/services.md`](docs/services.md) for operator reference commands).
35@@ -19,10 +19,11 @@ downloads a model for any of them. You run those services yourself (see
36 system-default format; set `--record-command "arecord -f S16_LE -r 16000 -c 1"`
37 to get the required 16 kHz mono S16_LE WAV. Any command works as long as it
38 writes a 16 kHz mono S16_LE WAV file to the path given as its last argument
39-- Three external model services, all run by you:
40+- Two external model services, all run by you:
41 - **LLM** — llama.cpp router (OpenAI-compatible `POST /v1/chat/completions`)
42- - **TTS** — OpenAI-compatible speech (`POST /audio/speech`, returns WAV)
43- - **STT** — Whisper inference (multipart `POST`, JSON `text` response)
44+ - **Audio** — audio.cpp server (`JP_AUDIO_BASE_URL`): TTS speech endpoint
45+ (`POST /v1/audio/speech`, returns WAV) and ASR transcriptions (multipart
46+ `POST /v1/audio/transcriptions`, JSON `text` response)
47
48 ## Build and run
49
50@@ -31,8 +32,8 @@ go build ./...
51 jp # or: go run ./cmd/jp
52 ```
53
54-On startup the game preloads the LLM model, runs bounded readiness checks
55-against TTS and STT, then warms each conversation's system prompt once so first
56+On startup the game preloads the LLM model, runs a bounded readiness check
57+against the audio service, then warms each conversation's system prompt once so first
58 replies are fast. A failed step prints one error line naming the affected
59 service and its configured URL; startup continues either way. The game never
60 launches a service to make a check pass.
61@@ -51,9 +52,7 @@ which win over defaults. Defaults match a local setup.
62 | Thinking mode | `--enable-thinking` | `ENABLE_THINKING` | `false` | Qwen3 thinking (`chat_template_kwargs.enable_thinking`) |
63 | LLM game slot | `--llm-game-slot` | `JP_LLM_GAME_SLOT` | `0` | llama-server slot for game-loop requests (`-1` = server decides) |
64 | LLM scratch slot | `--llm-scratch-slot` | `JP_LLM_SCRATCH_SLOT` | `1` | llama-server slot for judge, compaction, and sheet requests (`-1` = server decides) |
65-| TTS base URL | `--tts-url` | `JP_TTS_BASE_URL` | `http://127.0.0.1:8080/v1` | `POST /audio/speech` |
66-| STT server URL | `--stt-url` | `JP_STT_URL` | `http://127.0.0.1:8178` | multipart `POST /inference` |
67-| STT language | `--stt-language` | `JP_STT_LANGUAGE` | `ja` | Whisper request language (ISO 639-1) |
68+| Audio base URL | `--audio-url` | `JP_AUDIO_BASE_URL` | `http://127.0.0.1:8080` | audio.cpp server (TTS + ASR) |
69 | Recorder command | `--record-command` | — | `arecord` | 16 kHz mono S16_LE WAV capture (see note above) |
70 | Scenario brief | `--scenario` | `JP_SCENARIO_PATH` | `assets/scenarios/small_city.md` | plain-text scenario brief file |
71
72diff --git a/cmd/jp/main.go b/cmd/jp/main.go
73index c963a648cb69e449299a6b66b4a5602f1325c142..ae729cd8c6c9c756a3233d02b655a6374009426e 100644
74--- a/cmd/jp/main.go
75+++ b/cmd/jp/main.go
76@@ -45,11 +45,8 @@ func main() {
77 fmt.Fprintf(os.Stderr, "jp: LLM unavailable: %v\n", err)
78 llmUp = false
79 }
80- if err := availability.CheckTTS(hc, cfg.TTSBaseURL); err != nil {
81- fmt.Fprintf(os.Stderr, "jp: TTS unavailable: %v\n", err)
82- }
83- if err := availability.CheckSTT(hc, cfg.STTURL); err != nil {
84- fmt.Fprintf(os.Stderr, "jp: STT unavailable: %v\n", err)
85+ if err := availability.CheckAudio(hc, cfg.AudioBaseURL, []string{tts.ModelName, stt.ASRModelName}); err != nil {
86+ fmt.Fprintf(os.Stderr, "jp: audio service unavailable: %v\n", err)
87 }
88
89 if llmUp {
90@@ -81,10 +78,10 @@ func buildOrchestrator(cfg *config.Config, state *game.State, hc *http.Client, g
91
92 speechIn := &adapters.SpeechInput{
93 Recorder: stt.NewRecorder(cfg.RecordCommand, recordCap),
94- Whisper: stt.NewWhisperClient(cfg.STTURL, cfg.STTLanguage, hc),
95+ ASR: stt.NewASRClient(cfg.AudioBaseURL, hc),
96 }
97 speechOut := &adapters.SpeechOutput{
98- Client: tts.NewClient(cfg.TTSBaseURL, hc),
99+ Client: tts.NewClient(cfg.AudioBaseURL, hc),
100 Player: tts.NewPlayer(),
101 }
102
103diff --git a/docs/services.md b/docs/services.md
104index f943b76aed10ec84b36bc89ba616b0daebc4fa63..5e43f7e2eb0a434c86490104a8cc402942c39789 100644
105--- a/docs/services.md
106+++ b/docs/services.md
107@@ -1,6 +1,6 @@
108 # External services
109
110-The game is an HTTP client only. It connects to three externally managed model
111+The game is an HTTP client only. It connects to two externally managed model
112 services and **never** starts, stops, restarts, kills, reconfigures, or
113 downloads a model for any of them. Operators run these services themselves; the
114 game only talks to the configured endpoints over HTTP.
115@@ -9,13 +9,11 @@ All endpoints are configuration, not hard-coded process assumptions. Each can be
116 set with a flag (see `jp -help`) or an environment variable, and defaults
117 match the local setup:
118
119-| Service | Flag | Env var | Default | Use |
120-| ----------------- | ------------------ | ----------------- | ----------------------------------------- | --------------------------------- |
121-| LLM router URL | `--url` | `JP_LLM_BASE_URL` | `https://llama.home.theedgeofrage.com` | `POST /v1/chat/completions` |
122-| TTS base URL | `--tts-url` | `JP_TTS_BASE_URL` | `http://127.0.0.1:8080/v1` | `POST /audio/speech` |
123-| STT server URL | `--stt-url` | `JP_STT_URL` | `http://127.0.0.1:8178` | multipart `POST /inference` |
124-| STT language | `--stt-language` | `JP_STT_LANGUAGE` | `ja` | Whisper request language (ISO 639-1) |
125-| Recorder command | `--record-command` | - | `arecord` | 16kHz mono S16_LE WAV capture |
126+| Service | Flag | Env var | Default | Use |
127+| -------------- | ------------------ | ------------------- | ---------------------------------------- | ---------------------------------------------------------- |
128+| LLM router URL | `--url` | `JP_LLM_BASE_URL` | `https://llama.home.theedgeofrage.com` | `POST /v1/chat/completions` |
129+| Audio base URL | `--audio-url` | `JP_AUDIO_BASE_URL` | `http://127.0.0.1:8080` | TTS `POST /v1/audio/speech` + ASR `POST /v1/audio/transcriptions` |
130+| Recorder command | `--record-command` | - | `arecord` | 16kHz mono S16_LE WAV capture |
131
132 At startup, `jp` first preloads the LLM model with
133 `POST {JP_LLM_BASE_URL}/models/load` (30s timeout) before anything else loads,
134@@ -70,29 +68,28 @@ A failed preload prints one error line and startup continues.
135 > llama-server # router mode; `jp` must resolve to the Qwen3 model above
136 > ```
137
138-## STT - Whisper server (ggml-large-v3-turbo)
139+## Audio - audio.cpp (Qwen3-TTS + Qwen3-ASR)
140
141-Multilingual `ggml-large-v3-turbo.bin` served on `127.0.0.1:8178`. Each turn,
142-the game uploads a 16kHz mono S16_LE WAV as multipart form data to
143-`JP_STT_URL/inference` with the recording as a form file (`file=@recording.wav`). The
144-`language` field is included only when it is non-empty (default `ja`, which
145-forces Japanese transcription). The response is JSON with a `text` field.
146+One operator-run audio.cpp server at `JP_AUDIO_BASE_URL` (default
147+`http://127.0.0.1:8080`) serves both speech endpoints:
148
149-> **Operator-run only - the game never starts this.**
150->
151-> ```bash
152-> whisper-server --host 127.0.0.1 --port 8178 \
153-> --model ~/.local/share/llama-server/models/ggml-large-v3-turbo.bin \
154-> --language auto --threads 8
155-> ```
156+- **TTS** — OpenAI-compatible speech at `POST {base}/v1/audio/speech`, model
157+ `qwen3`, returning WAV bytes.
158+- **ASR** — multipart form (`file`, `model=qwen3-asr`, `language=ja`) to
159+ `POST {base}/v1/audio/transcriptions`, JSON `text` response.
160
161-The game does not invoke `whisper-server` or `whisper-cli`.
162+The server config file is `~/dev/tts/audio.cpp.qwen3.json`; it lists both
163+models: the existing qwen3 TTS entry plus a qwen3-asr entry (family
164+`qwen3_asr`, path `models/Qwen3-ASR-1.7B-GGUF/qwen3-asr-1.7b-q8_0.gguf`, task
165+`asr`, mode `offline`). Keep `lazy_load: true` and leave residency at the
166+default so both models stay resident — the game alternates TTS→STT every turn,
167+so LRU unloading would thrash.
168
169-## TTS - audio.cpp (Qwen3-TTS)
170-
171-OpenAI-compatible speech endpoint at `POST {JP_TTS_BASE_URL}/audio/speech`,
172-returning WAV bytes. The service uses the existing configuration in
173-`~/dev/tts/audio.cpp.qwen3.json` and the existing Qwen3-TTS 1.7B CustomVoice
174-GGUF. Reference only; the operator runs this service.
175+Whisper support was removed; whisper-server is no longer used. The game does not
176+invoke `whisper-server` or `whisper-cli`.
177
178 > **Operator-run only - the game never starts this.**
179+>
180+> ```bash
181+> audiocpp_server --config ~/dev/tts/audio.cpp.qwen3.json
182+> ```
183diff --git a/internal/adapters/speech.go b/internal/adapters/speech.go
184index 08ec9708b1b35b79b58a073e3e8396134905b7b6..e7a93bd3edcfa56b253f735ec229d93a5512afa0 100644
185--- a/internal/adapters/speech.go
186+++ b/internal/adapters/speech.go
187@@ -8,12 +8,12 @@ import (
188 "japanese/internal/tts"
189 )
190
191-// SpeechInput adapts the stt recorder and Whisper client to game.SpeechInput.
192-// Begin starts push-to-talk capture; End stops it, uploads the recording to
193-// Whisper, and returns the raw transcript.
194+// SpeechInput adapts the stt recorder and ASR client to game.SpeechInput.
195+// Begin starts push-to-talk capture; End stops it, uploads the recording for
196+// transcription, and returns the raw transcript.
197 type SpeechInput struct {
198 Recorder *stt.Recorder
199- Whisper *stt.WhisperClient
200+ ASR *stt.ASRClient
201
202 cur *stt.Recording
203 }
204@@ -35,7 +35,7 @@ func (s *SpeechInput) End(ctx context.Context) (string, error) {
205 if err != nil {
206 return "", fmt.Errorf("mic: %w", err)
207 }
208- raw, err := s.Whisper.Transcribe(ctx, path)
209+ raw, err := s.ASR.Transcribe(ctx, path)
210 if err != nil {
211 return "", err
212 }
213diff --git a/internal/availability/availability.go b/internal/availability/availability.go
214index 4ecd7dad2737ca35429c41c3bd2f1ecaebe3818e..49de61bcca61f963c68934d6b22e8ae767e115b8 100644
215--- a/internal/availability/availability.go
216+++ b/internal/availability/availability.go
217@@ -4,34 +4,48 @@
218 package availability
219
220 import (
221+ "encoding/json"
222 "fmt"
223 "io"
224 "net/http"
225 "strings"
226 )
227
228-// CheckTTS verifies the TTS server answers. A 404 on /models is fine: some
229-// servers do not expose it.
230-func CheckTTS(client *http.Client, base string) error {
231- return checkServer(client, strings.TrimRight(base, "/")+"/models")
232-}
233+// CheckAudio verifies the audio.cpp server answers at /v1/models and lists
234+// every configured model id.
235+func CheckAudio(client *http.Client, base string, modelIDs []string) error {
236+ url := strings.TrimRight(base, "/") + "/v1/models"
237+ status, raw, err := doGet(client, url)
238+ if err != nil {
239+ return fmt.Errorf("GET %s: %w", url, err)
240+ }
241+ if status < 200 || status >= 300 {
242+ return fmt.Errorf("GET %s: HTTP %d", url, status)
243+ }
244
245-// CheckSTT verifies the STT server answers.
246-func CheckSTT(client *http.Client, sttURL string) error {
247- return checkServer(client, sttURL)
248-}
249+ var out struct {
250+ Data []struct {
251+ ID string `json:"id"`
252+ } `json:"data"`
253+ }
254+ if err := json.Unmarshal(raw, &out); err != nil {
255+ return fmt.Errorf("decode models from %s: %w", url, err)
256+ }
257
258-// checkServer GETs target and reports an error unless the server answers with
259-// a 2xx or 404 status.
260-func checkServer(client *http.Client, target string) error {
261- status, _, err := doGet(client, target)
262- if err != nil {
263- return fmt.Errorf("GET %s: %w", target, err)
264+ listed := make(map[string]bool, len(out.Data))
265+ for _, m := range out.Data {
266+ listed[m.ID] = true
267+ }
268+ var missing []string
269+ for _, id := range modelIDs {
270+ if !listed[id] {
271+ missing = append(missing, id)
272+ }
273 }
274- if status == http.StatusNotFound || (status >= 200 && status < 300) {
275- return nil
276+ if len(missing) > 0 {
277+ return fmt.Errorf("model(s) not listed by %s: %s", url, strings.Join(missing, ", "))
278 }
279- return fmt.Errorf("GET %s: HTTP %d", target, status)
280+ return nil
281 }
282
283 // doGet issues a GET and returns the status code plus the fully read body.
284diff --git a/internal/config/config.go b/internal/config/config.go
285index 09d040f2f8774ce4665db4729b15430c85bdc054..c984aaebc6ee723cc06bae5e7a0666760cd55517 100644
286--- a/internal/config/config.go
287+++ b/internal/config/config.go
288@@ -25,9 +25,7 @@ type LLMConfig struct {
289 type Config struct {
290 LLMConfig LLMConfig
291
292- TTSBaseURL string `long:"tts-url" env:"JP_TTS_BASE_URL" default:"http://127.0.0.1:8080/v1" description:"OpenAI-compatible TTS base URL"`
293- STTURL string `long:"stt-url" env:"JP_STT_URL" default:"http://127.0.0.1:8178" description:"Whisper server URL"`
294- STTLanguage string `long:"stt-language" env:"JP_STT_LANGUAGE" default:"ja" description:"Whisper request language (ISO 639-1 code)"`
295+ AudioBaseURL string `long:"audio-url" env:"JP_AUDIO_BASE_URL" default:"http://127.0.0.1:8080" description:"audio.cpp server base URL (TTS + ASR)"`
296 RecordCommand string `long:"record-command" default:"arecord" description:"Recorder command for 16kHz mono S16_LE WAV capture"`
297
298 Scenario string `long:"scenario" env:"JP_SCENARIO_PATH" default:"assets/scenarios/small_city.md" description:"path to the scenario brief file"`
299diff --git a/internal/stt/whisper.go b/internal/stt/asr.go
300rename from internal/stt/whisper.go
301rename to internal/stt/asr.go
302index 1602b1a61eb48364dbe1e84cf8bfda4bc4685390..a5f316e433309e72537dba6f430edb714cfb9fee 100644
303--- a/internal/stt/whisper.go
304+++ b/internal/stt/asr.go
305@@ -1,4 +1,4 @@
306-// Package stt wraps push-to-talk recording and the Whisper multipart client. It
307+// Package stt wraps push-to-talk recording and the audio.cpp ASR client. It
308 // is an HTTP/process client only and never starts or manages any model service.
309 package stt
310
311@@ -14,34 +14,40 @@ import (
312 "strings"
313 )
314
315-// WhisperClient is a multipart client for the externally managed whisper.cpp
316-// server. It uploads a WAV file and returns the recognized text.
317-type WhisperClient struct {
318- URL string
319- Language string // optional; sent only when non-empty
320- HTTP *http.Client
321+// ASRModelName is the model id sent in transcription requests.
322+const ASRModelName = "qwen3-asr"
323+
324+// asrLanguage forces Japanese transcription.
325+const asrLanguage = "ja"
326+
327+// ASRClient is a multipart client for the externally managed audio.cpp
328+// server's OpenAI-style transcriptions endpoint. It uploads a WAV file and
329+// returns the recognized text.
330+type ASRClient struct {
331+ URL string
332+ HTTP *http.Client
333 }
334
335-// NewWhisperClient takes the Whisper server base URL and appends the
336-// /inference path for transcription requests.
337-func NewWhisperClient(baseURL, language string, hc *http.Client) *WhisperClient {
338+// NewASRClient takes the audio.cpp base URL and appends the
339+// /v1/audio/transcriptions path for transcription requests.
340+func NewASRClient(baseURL string, hc *http.Client) *ASRClient {
341 if hc == nil {
342 hc = &http.Client{}
343 }
344- return &WhisperClient{URL: strings.TrimRight(baseURL, "/") + "/inference", Language: language, HTTP: hc}
345+ return &ASRClient{URL: strings.TrimRight(baseURL, "/") + "/v1/audio/transcriptions", HTTP: hc}
346 }
347
348 // Transcribe uploads the WAV at wavPath and returns the recognized text. The
349 // file is removed after the request completes (success or failure). Only the
350 // given game-created path is ever deleted.
351-func (c *WhisperClient) Transcribe(ctx context.Context, wavPath string) (string, error) {
352+func (c *ASRClient) Transcribe(ctx context.Context, wavPath string) (string, error) {
353 data, err := os.ReadFile(wavPath)
354 if err != nil {
355 return "", fmt.Errorf("stt: read recording %s: %w", wavPath, err)
356 }
357 defer func() { _ = os.Remove(wavPath) }()
358
359- body, contentType, err := encodeMultipart(data, c.Language)
360+ body, contentType, err := encodeMultipart(data)
361 if err != nil {
362 return "", fmt.Errorf("stt: encode multipart: %w", err)
363 }
364@@ -74,7 +80,7 @@ func (c *WhisperClient) Transcribe(ctx context.Context, wavPath string) (string,
365 return out.Text, nil
366 }
367
368-func encodeMultipart(wav []byte, language string) (io.Reader, string, error) {
369+func encodeMultipart(wav []byte) (io.Reader, string, error) {
370 var buf bytes.Buffer
371 mw := multipart.NewWriter(&buf)
372
373@@ -85,10 +91,11 @@ func encodeMultipart(wav []byte, language string) (io.Reader, string, error) {
374 if _, err := fw.Write(wav); err != nil {
375 return nil, "", fmt.Errorf("write file field: %w", err)
376 }
377- if language != "" {
378- if err := mw.WriteField("language", language); err != nil {
379- return nil, "", fmt.Errorf("write language field: %w", err)
380- }
381+ if err := mw.WriteField("model", ASRModelName); err != nil {
382+ return nil, "", fmt.Errorf("write model field: %w", err)
383+ }
384+ if err := mw.WriteField("language", asrLanguage); err != nil {
385+ return nil, "", fmt.Errorf("write language field: %w", err)
386 }
387 if err := mw.Close(); err != nil {
388 return nil, "", fmt.Errorf("close multipart: %w", err)
389diff --git a/internal/stt/recorder.go b/internal/stt/recorder.go
390index dfc2588dd93d3dfbf769180bffce967b4bff979b..40eab3aa4a1e5772c6cbefa4c5de1b85691e4452 100644
391--- a/internal/stt/recorder.go
392+++ b/internal/stt/recorder.go
393@@ -89,7 +89,7 @@ func (r *Recorder) Start(ctx context.Context) (*Recording, error) {
394
395 // Stop requests a graceful stop (SIGINT, then SIGKILL after a short grace
396 // period), waits for the process to be reaped, and returns the temp WAV path.
397-// The file is left on disk for the Whisper step, which deletes it after
398+// The file is left on disk for the ASR step, which deletes it after
399 // uploading. If the recorder command exited on its own with an error, the temp
400 // file is removed and that error is returned. Stop is safe to call more than
401 // once.
402diff --git a/internal/tts/client.go b/internal/tts/client.go
403index 6c28149a94513e31d299d09061f6f1470127789f..2669997cbf4da9aaa4c1b1e96b76e46c6a559304 100644
404--- a/internal/tts/client.go
405+++ b/internal/tts/client.go
406@@ -11,8 +11,8 @@ import (
407 )
408
409 const (
410- speechPath = "/audio/speech"
411- modelName = "qwen3"
412+ speechPath = "/v1/audio/speech"
413+ ModelName = "qwen3"
414 defaultVoice = "Ono_Anna"
415 ttsLanguage = "Japanese"
416 speakInstructions = "Speak at a natural, conversational pace with clear articulation and natural pauses between sentences."
417@@ -20,8 +20,8 @@ const (
418 )
419
420 // Client is the audio.cpp Qwen3-TTS speech client. BaseURL is the configured
421-// TTS base (for example http://127.0.0.1:8080/v1); Speech posts to
422-// {BaseURL}/audio/speech.
423+// audio.cpp base (for example http://127.0.0.1:8080); Speech posts to
424+// {BaseURL}/v1/audio/speech.
425 type Client struct {
426 BaseURL string
427 HTTP *http.Client
428@@ -47,7 +47,7 @@ type speechRequest struct {
429 // Speech posts kana to the speech endpoint and returns the WAV bytes.
430 func (c *Client) Speech(ctx context.Context, kana string) ([]byte, error) {
431 body, err := json.Marshal(speechRequest{
432- Model: modelName,
433+ Model: ModelName,
434 Input: kana,
435 Voice: defaultVoice,
436 Language: ttsLanguage,