Parent directory

client.go

3252 bytes
  1package tts
  2
  3import (
  4	"bytes"
  5	"context"
  6	"encoding/json"
  7	"fmt"
  8	"io"
  9	"net/http"
 10	"strings"
 11	"time"
 12)
 13
 14const (
 15	speechPath = "/v1/audio/speech"
 16	ModelName  = "qwen3"
 17
 18	// TTS voice presets (must match the names in audio.cpp.json).
 19	VoiceMale   = "Ryan"
 20	VoiceFemale = "Ono_Anna"
 21	// DefaultVoice is used when a speaker's gender is unknown (ambient lines).
 22	DefaultVoice = VoiceFemale
 23
 24	ttsLanguage       = "Japanese"
 25	speakInstructions = "Speak at a natural, conversational pace with clear articulation and natural pauses between sentences."
 26	speechSeed        = 1234
 27
 28	// speechTimeout bounds one synthesis call; TTS is fast and a stuck call should
 29	// not hang a turn.
 30	speechTimeout = 10 * time.Second
 31)
 32
 33// Client is the audio.cpp Qwen3-TTS speech client. BaseURL is the configured
 34// audio.cpp base (for example http://127.0.0.1:9932); Speech posts to
 35// {BaseURL}/v1/audio/speech.
 36type Client struct {
 37	BaseURL string
 38	HTTP    *http.Client
 39}
 40
 41func NewClient(baseURL string, hc *http.Client) *Client {
 42	if hc == nil {
 43		hc = &http.Client{}
 44	}
 45	return &Client{BaseURL: baseURL, HTTP: hc}
 46}
 47
 48type speechRequest struct {
 49	Model          string `json:"model"`
 50	Input          string `json:"input"`
 51	Voice          string `json:"voice"`
 52	Language       string `json:"language"`
 53	Instructions   string `json:"instructions"`
 54	Seed           int    `json:"seed"`
 55	ResponseFormat string `json:"response_format"`
 56}
 57
 58// VoiceForGender maps a character-sheet gender to a TTS voice preset.
 59// Unknown or empty genders fall back to DefaultVoice.
 60func VoiceForGender(gender string) string {
 61	switch strings.ToLower(strings.TrimSpace(gender)) {
 62	case "male":
 63		return VoiceMale
 64	case "female":
 65		return VoiceFemale
 66	default:
 67		return DefaultVoice
 68	}
 69}
 70
 71// Speech posts kana to the speech endpoint using the given voice preset and
 72// returns the WAV bytes. An empty voice falls back to DefaultVoice.
 73func (c *Client) Speech(ctx context.Context, kana, voice string) ([]byte, error) {
 74	ctx, cancel := context.WithTimeout(ctx, speechTimeout)
 75	defer cancel()
 76	if voice == "" {
 77		voice = DefaultVoice
 78	}
 79	body, err := json.Marshal(speechRequest{
 80		Model:          ModelName,
 81		Input:          kana,
 82		Voice:          voice,
 83		Language:       ttsLanguage,
 84		Instructions:   speakInstructions,
 85		Seed:           speechSeed,
 86		ResponseFormat: "wav",
 87	})
 88	if err != nil {
 89		return nil, fmt.Errorf("tts: encode request: %w", err)
 90	}
 91
 92	url := c.BaseURL + speechPath
 93	req, err := http.NewRequestWithContext(ctx, http.MethodPost, url, bytes.NewReader(body))
 94	if err != nil {
 95		return nil, fmt.Errorf("tts: build request: %w", err)
 96	}
 97	req.Header.Set("Content-Type", "application/json")
 98
 99	resp, err := c.HTTP.Do(req)
100	if err != nil {
101		return nil, fmt.Errorf("tts: request to %s: %w", url, err)
102	}
103	defer func() { _ = resp.Body.Close() }()
104
105	raw, err := io.ReadAll(resp.Body)
106	if err != nil {
107		return nil, fmt.Errorf("tts: read response from %s: %w", url, err)
108	}
109	if resp.StatusCode < 200 || resp.StatusCode >= 300 {
110		return nil, fmt.Errorf("tts: %s returned HTTP %d: %s", url, resp.StatusCode, strings.TrimSpace(string(raw)))
111	}
112	if _, err := ParseWAV(raw); err != nil {
113		return nil, fmt.Errorf("%s returned an invalid WAV response: %w", url, err)
114	}
115	return raw, nil
116}