dcc625c17b52d6f87cc8026b0e1585b14a7553d5

Author
TheEdgeOfRage <git@theedgeofrage.com>
Committer
TheEdgeOfRage <git@theedgeofrage.com>
Date

Message

Add audio.cpp server config and model download script with Qwen3-ASR

Diff

  1diff --git a/.gitignore b/.gitignore
  2index 56a906fb053f0b58d95c7bdf5022910cea9264be..143a374b0efe170cd8fc78b299c1d8628e1e90c0 100644
  3--- a/.gitignore
  4+++ b/.gitignore
  5@@ -5,11 +5,12 @@
  6 
  7 # Local temporary output
  8 /tmp/
  9-
 10-.pi/
 11-
 12-# Local debug output
 13 jp_raw.log
 14 
 15+# Agent files
 16+.pi/
 17 /plan.md
 18 /PLAN.md
 19+
 20+# Model weights (fetched by download-qwen3.sh)
 21+/models/
 22diff --git a/audio.cpp.json b/audio.cpp.json
 23new file mode 100644
 24index 0000000000000000000000000000000000000000..32850d571fded96f6e7dc252d0ed8a9602eba391
 25--- /dev/null
 26+++ b/audio.cpp.json
 27@@ -0,0 +1,38 @@
 28+{
 29+  "host": "127.0.0.1",
 30+  "port": 8080,
 31+  "backend": "vulkan",
 32+  "threads": 8,
 33+  "lazy_load": true,
 34+  "log_request_body": false,
 35+  "models": [
 36+    {
 37+      "id": "qwen3",
 38+      "family": "qwen3_tts",
 39+      "path": "models/Qwen3-TTS-12Hz-1.7B-CustomVoice-GGUF/qwen3-tts-12hz-1.7b-customvoice-q8_0.gguf",
 40+      "task": "tts",
 41+      "mode": "offline",
 42+      "session_options": {
 43+        "qwen3_tts.mem_saver": true
 44+      },
 45+      "default_voice_preset": {
 46+        "voice_id": "Ryan"
 47+      },
 48+      "voice_presets": {
 49+        "Ryan": {
 50+          "voice_id": "Ryan"
 51+        },
 52+        "Ono_Anna": {
 53+          "voice_id": "Ono_Anna"
 54+        }
 55+      }
 56+    },
 57+    {
 58+      "id": "qwen3-asr",
 59+      "family": "qwen3_asr",
 60+      "path": "models/Qwen3-ASR-1.7B-GGUF/qwen3-asr-1.7b-q8_0.gguf",
 61+      "task": "asr",
 62+      "mode": "offline"
 63+    }
 64+  ]
 65+}
 66diff --git a/docs/services.md b/docs/services.md
 67index 5e43f7e2eb0a434c86490104a8cc402942c39789..b78903586fb8d0cf45c2d2c6e755468d6d7d920e 100644
 68--- a/docs/services.md
 69+++ b/docs/services.md
 70@@ -9,11 +9,11 @@ All endpoints are configuration, not hard-coded process assumptions. Each can be
 71 set with a flag (see `jp -help`) or an environment variable, and defaults
 72 match the local setup:
 73 
 74-| Service        | Flag               | Env var             | Default                                  | Use                                                        |
 75-| -------------- | ------------------ | ------------------- | ---------------------------------------- | ---------------------------------------------------------- |
 76-| LLM router URL | `--url`            | `JP_LLM_BASE_URL`   | `https://llama.home.theedgeofrage.com`   | `POST /v1/chat/completions`                                |
 77-| Audio base URL | `--audio-url`      | `JP_AUDIO_BASE_URL` | `http://127.0.0.1:8080`                 | TTS `POST /v1/audio/speech` + ASR `POST /v1/audio/transcriptions` |
 78-| Recorder command | `--record-command` | -                  | `arecord`                                | 16kHz mono S16_LE WAV capture                              |
 79+| Service          | Flag               | Env var             | Default                                | Use                                                               |
 80+| ---------------- | ------------------ | ------------------- | -------------------------------------- | ----------------------------------------------------------------- |
 81+| LLM router URL   | `--url`            | `JP_LLM_BASE_URL`   | `https://llama.home.theedgeofrage.com` | `POST /v1/chat/completions`                                       |
 82+| Audio base URL   | `--audio-url`      | `JP_AUDIO_BASE_URL` | `http://127.0.0.1:8080`                | TTS `POST /v1/audio/speech` + ASR `POST /v1/audio/transcriptions` |
 83+| Recorder command | `--record-command` | -                   | `arecord`                              | 16kHz mono S16_LE WAV capture                                     |
 84 
 85 At startup, `jp` first preloads the LLM model with
 86 `POST {JP_LLM_BASE_URL}/models/load` (30s timeout) before anything else loads,
 87@@ -45,10 +45,10 @@ specific llama-server slot, bypassing the server's LRU/similarity
 88 auto-selection. The game uses two slots so scratch requests can never evict
 89 the game loop's cached context:
 90 
 91-| Client use                                   | Flag                   | Env var               | Default |
 92-| -------------------------------------------- | ---------------------- | --------------------- | ------- |
 93-| Game loop (world replies)                    | `--llm-game-slot`      | `JP_LLM_GAME_SLOT`    | `0`     |
 94-| Scratch (judge, compaction, character sheets; warmup included) | `--llm-scratch-slot`   | `JP_LLM_SCRATCH_SLOT` | `1`     |
 95+| Client use                                                     | Flag                 | Env var               | Default |
 96+| -------------------------------------------------------------- | -------------------- | --------------------- | ------- |
 97+| Game loop (world replies)                                      | `--llm-game-slot`    | `JP_LLM_GAME_SLOT`    | `0`     |
 98+| Scratch (judge, compaction, character sheets; warmup included) | `--llm-scratch-slot` | `JP_LLM_SCRATCH_SLOT` | `1`     |
 99 
100 Warmup follows the same split: the game system prompt warms on the game
101 client, the judge, compaction, and sheet prompts warm on the scratch client,
102@@ -78,7 +78,7 @@ One operator-run audio.cpp server at `JP_AUDIO_BASE_URL` (default
103 - **ASR** — multipart form (`file`, `model=qwen3-asr`, `language=ja`) to
104   `POST {base}/v1/audio/transcriptions`, JSON `text` response.
105 
106-The server config file is `~/dev/tts/audio.cpp.qwen3.json`; it lists both
107+The server config file is `audio.cpp.json` at the repository root; it lists both
108 models: the existing qwen3 TTS entry plus a qwen3-asr entry (family
109 `qwen3_asr`, path `models/Qwen3-ASR-1.7B-GGUF/qwen3-asr-1.7b-q8_0.gguf`, task
110 `asr`, mode `offline`). Keep `lazy_load: true` and leave residency at the
111@@ -91,5 +91,5 @@ invoke `whisper-server` or `whisper-cli`.
112 > **Operator-run only - the game never starts this.**
113 >
114 > ```bash
115-> audiocpp_server --config ~/dev/tts/audio.cpp.qwen3.json
116+> audiocpp_server --config audio.cpp.json  # from the repository root
117 > ```
118diff --git a/download-qwen3.sh b/download-qwen3.sh
119new file mode 100755
120index 0000000000000000000000000000000000000000..a2ec789537ee6a2d86681346a6eb710e88727b94
121--- /dev/null
122+++ b/download-qwen3.sh
123@@ -0,0 +1,18 @@
124+#!/usr/bin/env bash
125+set -euo pipefail
126+
127+hf download audio-cpp/audio.cpp-gguf \
128+  Qwen3-TTS-12Hz-1.7B-CustomVoice-GGUF/qwen3-tts-12hz-1.7b-customvoice-q8_0.gguf \
129+  --revision dc6fecccc2b0c6bdda0a8b2f38fa61394fee0b9c \
130+  --local-dir models
131+printf '%s  %s\n' \
132+  '3cfaac8e9f13554f6daea3c5e0c53fede71ef5500cbaae7445e5fc3a5bb12e72' \
133+  'models/Qwen3-TTS-12Hz-1.7B-CustomVoice-GGUF/qwen3-tts-12hz-1.7b-customvoice-q8_0.gguf' | sha256sum --check
134+
135+hf download audio-cpp/audio.cpp-gguf \
136+  Qwen3-ASR-1.7B-GGUF/qwen3-asr-1.7b-q8_0.gguf \
137+  --revision 056144d2744697c9439bd32647279674dba0c964 \
138+  --local-dir models
139+printf '%s  %s\n' \
140+  'da4fc2ac7f24dee784d1684eb1f35836cdbf559519452ae11777670734c0a4f8' \
141+  'models/Qwen3-ASR-1.7B-GGUF/qwen3-asr-1.7b-q8_0.gguf' | sha256sum --check