client.go
3252 bytes
1package tts
2
3import (
4 "bytes"
5 "context"
6 "encoding/json"
7 "fmt"
8 "io"
9 "net/http"
10 "strings"
11 "time"
12)
13
14const (
15 speechPath = "/v1/audio/speech"
16 ModelName = "qwen3"
17
18 // TTS voice presets (must match the names in audio.cpp.json).
19 VoiceMale = "Ryan"
20 VoiceFemale = "Ono_Anna"
21 // DefaultVoice is used when a speaker's gender is unknown (ambient lines).
22 DefaultVoice = VoiceFemale
23
24 ttsLanguage = "Japanese"
25 speakInstructions = "Speak at a natural, conversational pace with clear articulation and natural pauses between sentences."
26 speechSeed = 1234
27
28 // speechTimeout bounds one synthesis call; TTS is fast and a stuck call should
29 // not hang a turn.
30 speechTimeout = 10 * time.Second
31)
32
33// Client is the audio.cpp Qwen3-TTS speech client. BaseURL is the configured
34// audio.cpp base (for example http://127.0.0.1:9932); Speech posts to
35// {BaseURL}/v1/audio/speech.
36type Client struct {
37 BaseURL string
38 HTTP *http.Client
39}
40
41func NewClient(baseURL string, hc *http.Client) *Client {
42 if hc == nil {
43 hc = &http.Client{}
44 }
45 return &Client{BaseURL: baseURL, HTTP: hc}
46}
47
48type speechRequest struct {
49 Model string `json:"model"`
50 Input string `json:"input"`
51 Voice string `json:"voice"`
52 Language string `json:"language"`
53 Instructions string `json:"instructions"`
54 Seed int `json:"seed"`
55 ResponseFormat string `json:"response_format"`
56}
57
58// VoiceForGender maps a character-sheet gender to a TTS voice preset.
59// Unknown or empty genders fall back to DefaultVoice.
60func VoiceForGender(gender string) string {
61 switch strings.ToLower(strings.TrimSpace(gender)) {
62 case "male":
63 return VoiceMale
64 case "female":
65 return VoiceFemale
66 default:
67 return DefaultVoice
68 }
69}
70
71// Speech posts kana to the speech endpoint using the given voice preset and
72// returns the WAV bytes. An empty voice falls back to DefaultVoice.
73func (c *Client) Speech(ctx context.Context, kana, voice string) ([]byte, error) {
74 ctx, cancel := context.WithTimeout(ctx, speechTimeout)
75 defer cancel()
76 if voice == "" {
77 voice = DefaultVoice
78 }
79 body, err := json.Marshal(speechRequest{
80 Model: ModelName,
81 Input: kana,
82 Voice: voice,
83 Language: ttsLanguage,
84 Instructions: speakInstructions,
85 Seed: speechSeed,
86 ResponseFormat: "wav",
87 })
88 if err != nil {
89 return nil, fmt.Errorf("tts: encode request: %w", err)
90 }
91
92 url := c.BaseURL + speechPath
93 req, err := http.NewRequestWithContext(ctx, http.MethodPost, url, bytes.NewReader(body))
94 if err != nil {
95 return nil, fmt.Errorf("tts: build request: %w", err)
96 }
97 req.Header.Set("Content-Type", "application/json")
98
99 resp, err := c.HTTP.Do(req)
100 if err != nil {
101 return nil, fmt.Errorf("tts: request to %s: %w", url, err)
102 }
103 defer func() { _ = resp.Body.Close() }()
104
105 raw, err := io.ReadAll(resp.Body)
106 if err != nil {
107 return nil, fmt.Errorf("tts: read response from %s: %w", url, err)
108 }
109 if resp.StatusCode < 200 || resp.StatusCode >= 300 {
110 return nil, fmt.Errorf("tts: %s returned HTTP %d: %s", url, resp.StatusCode, strings.TrimSpace(string(raw)))
111 }
112 if _, err := ParseWAV(raw); err != nil {
113 return nil, fmt.Errorf("%s returned an invalid WAV response: %w", url, err)
114 }
115 return raw, nil
116}