Diff
1diff --git a/cmd/jp/main.go b/cmd/jp/main.go
2index 46e1a61e7525ee3cb5784311d69363533796b8a8..64b1ce365014526ee1e4f6eb3b040cea64b71d43 100644
3--- a/cmd/jp/main.go
4+++ b/cmd/jp/main.go
5@@ -8,7 +8,6 @@ import (
6 "time"
7
8 "github.com/charmbracelet/bubbletea"
9- flags "github.com/jessevdk/go-flags"
10
11 "japanese/internal/adapters"
12 "japanese/internal/availability"
13@@ -26,17 +25,7 @@ import (
14 const recordCap = 10 * time.Second
15
16 func main() {
17- var cfg config.Config
18- parser := flags.NewParser(&cfg, flags.HelpFlag)
19- if _, err := parser.ParseArgs(os.Args[1:]); err != nil {
20- if e, ok := err.(*flags.Error); ok && e.Type == flags.ErrHelp {
21- parser.WriteHelp(os.Stdout)
22- os.Exit(0)
23- }
24- fmt.Fprintf(os.Stderr, "jp: %v\n", err)
25- parser.WriteHelp(os.Stderr)
26- os.Exit(1)
27- }
28+ cfg := config.ParseConfig()
29
30 data, err := os.ReadFile(cfg.MapPath)
31 if err != nil {
32@@ -67,17 +56,17 @@ func main() {
33
34 // checkingServices seeds the UI with one "checking…" row per configured
35 // service before any probe result has arrived.
36-func checkingServices(cfg config.Config) []availability.Service {
37+func checkingServices(cfg *config.Config) []availability.Service {
38 return []availability.Service{
39- {Name: "LLM", URL: cfg.LLMBaseURL},
40+ {Name: "LLM", URL: cfg.LLMConfig.BaseURL},
41 {Name: "TTS", URL: cfg.TTSBaseURL},
42 {Name: "STT", URL: cfg.STTURL},
43 }
44 }
45
46-func buildOrchestrator(cfg config.Config, state *game.State, personas []persona.Persona, playDone chan struct{}) *game.Orchestrator {
47- hc := &http.Client{Timeout: 90 * time.Second}
48- llmClient := llm.NewClient(cfg.LLMBaseURL, hc)
49+func buildOrchestrator(cfg *config.Config, state *game.State, personas []persona.Persona, playDone chan struct{}) *game.Orchestrator {
50+ hc := &http.Client{Timeout: 5 * time.Second}
51+ llmClient := llm.NewClient(cfg, hc)
52 npcModel := &adapters.NPCModel{Client: llmClient, Policy: llm.DefaultHistoryPolicy()}
53 judgeModel := &adapters.JudgeModel{Client: llmClient}
54
55diff --git a/internal/adapters/llm.go b/internal/adapters/llm.go
56index a2d85f4f3ce096aabcae00fd6fee3c583ef9aa2b..d98f1a30d0f379738c861f88d5ee5d311a903fd2 100644
57--- a/internal/adapters/llm.go
58+++ b/internal/adapters/llm.go
59@@ -25,7 +25,7 @@ func (m *NPCModel) Reply(ctx context.Context, req game.NPCRequest) (llm.NPCReply
60 policy = llm.DefaultHistoryPolicy()
61 }
62 msgs := llm.BuildNPCMessages(req.Persona, req.Situation, req.Transcript, req.History, policy)
63- raw, err := m.Client.Generate(ctx, msgs, llm.Qwen3Options())
64+ raw, err := m.Client.Generate(ctx, msgs)
65 if err != nil {
66 return llm.NPCReply{}, err
67 }
68@@ -40,7 +40,7 @@ type JudgeModel struct {
69
70 func (m *JudgeModel) Judge(ctx context.Context, req game.JudgeRequest) (llm.JudgeResult, error) {
71 msgs := llm.BuildJudgeMessages(req.Situation, req.Transcript)
72- raw, err := m.Client.Generate(ctx, msgs, llm.Qwen3Options())
73+ raw, err := m.Client.Generate(ctx, msgs)
74 if err != nil {
75 return llm.JudgeResult{}, err
76 }
77diff --git a/internal/availability/availability.go b/internal/availability/availability.go
78index 7d7f401e0def7d6a20e6bea7e43e9827c7aef4b3..96edb843e38fc9574c1a0d9564d3ce75076a8d27 100644
79--- a/internal/availability/availability.go
80+++ b/internal/availability/availability.go
81@@ -30,11 +30,11 @@ type Service struct {
82 // once per service, in completion order. Each probe keeps its bounded timeout
83 // and names the affected URL on failure, so one slow or dead service does not
84 // hold up the others. It returns immediately after launching the probes.
85-func CheckAllAsync(cfg config.Config, fn func(Service)) {
86+func CheckAllAsync(cfg *config.Config, fn func(Service)) {
87 client := &http.Client{Timeout: checkTimeout}
88 ctx := context.Background()
89 jobs := []func() Service{
90- func() Service { return checkLLM(ctx, client, cfg.LLMBaseURL) },
91+ func() Service { return checkLLM(ctx, client, cfg.LLMConfig.BaseURL) },
92 func() Service { return checkTTS(ctx, client, cfg.TTSBaseURL) },
93 func() Service { return checkSTT(ctx, client, cfg.STTURL) },
94 }
95diff --git a/internal/config/config.go b/internal/config/config.go
96index 6dcbc0c8b6acd3e32c2cf8bf212b1063aec4957d..a3d9ab5322ea2523203f0d55dc44c11ed095282e 100644
97--- a/internal/config/config.go
98+++ b/internal/config/config.go
99@@ -4,10 +4,25 @@
100 // defaults.
101 package config
102
103-// Config is the shared option group. Subcommands embed it in their own options
104-// struct and parse with github.com/jessevdk/go-flags.
105+import (
106+ "fmt"
107+ "os"
108+
109+ "github.com/jessevdk/go-flags"
110+)
111+
112+type LLMConfig struct {
113+ BaseURL string `long:"url" env:"JP_LLM_BASE_URL" default:"https://llama.home.theedgeofrage.com/v1" description:"OpenAI-compatible LLM base URL"`
114+ Model string `long:"model" env:"MODEL" default:"jp"`
115+ Temperature float64 `long:"temperature" env:"TEMPERATURE" default:"1.0"`
116+ MaxTokens int `long:"max-tokens" env:"MAX_TOKENS" default:"256"`
117+ EnableThinking bool `long:"enable-thinking" env:"ENABLE_THINKING"`
118+}
119+
120+// Config is the shared option group. Parsed with github.com/jessevdk/go-flags.
121 type Config struct {
122- LLMBaseURL string `long:"llm-url" env:"JP_LLM_BASE_URL" default:"https://llama.home.theedgeofrage.com/v1" description:"OpenAI-compatible LLM base URL"`
123+ LLMConfig LLMConfig
124+
125 TTSBaseURL string `long:"tts-url" env:"JP_TTS_BASE_URL" default:"http://127.0.0.1:8080/v1" description:"OpenAI-compatible TTS base URL"`
126 STTURL string `long:"stt-url" env:"JP_STT_URL" default:"http://127.0.0.1:8178/inference" description:"Whisper inference URL"`
127 STTLanguage string `long:"stt-language" env:"JP_STT_LANGUAGE" default:"auto" description:"Optional Whisper request language"`
128@@ -16,3 +31,19 @@ type Config struct {
129 MapPath string `long:"map" env:"JP_MAP_PATH" default:"assets/maps/city.json" description:"path to the city map JSON"`
130 Personas string `long:"persona-dir" env:"JP_PERSONA_DIR" default:"assets/personas" description:"directory of persona JSON files"`
131 }
132+
133+func ParseConfig() *Config {
134+ var cfg Config
135+ parser := flags.NewParser(&cfg, flags.HelpFlag)
136+ if _, err := parser.ParseArgs(os.Args[1:]); err != nil {
137+ if e, ok := err.(*flags.Error); ok && e.Type == flags.ErrHelp {
138+ parser.WriteHelp(os.Stdout)
139+ os.Exit(0)
140+ }
141+ fmt.Fprintf(os.Stderr, "jp: %v\n", err)
142+ parser.WriteHelp(os.Stderr)
143+ os.Exit(1)
144+ }
145+
146+ return &cfg
147+}
148diff --git a/internal/llm/client.go b/internal/llm/client.go
149index 0a239cebc69cad200c44eaacdc72e23d56a61803..b07869206dd36d6f58813746bdeab6002a746b9f 100644
150--- a/internal/llm/client.go
151+++ b/internal/llm/client.go
152@@ -11,12 +11,11 @@ import (
153 "encoding/json"
154 "fmt"
155 "io"
156+ "japanese/internal/config"
157 "net/http"
158 "strings"
159 )
160
161-const llmModel = "qwen3.8-27b-q3"
162-
163 type Role string
164
165 const (
166@@ -30,29 +29,27 @@ type Message struct {
167 Content string `json:"content"`
168 }
169
170-// Options controls a single chat request. The Qwen3 game defaults are produced
171-// by Qwen3Options.
172-type Options struct {
173+type Client struct {
174+ baseURL string
175+ httpClient *http.Client
176 Model string
177 Temperature float64
178 MaxTokens int
179 EnableThinking bool
180 }
181
182-func Qwen3Options() Options {
183- return Options{Model: llmModel, Temperature: 1.0, MaxTokens: 256, EnableThinking: false}
184-}
185-
186-type Client struct {
187- BaseURL string
188- HTTP *http.Client
189-}
190-
191-func NewClient(baseURL string, hc *http.Client) *Client {
192+func NewClient(cfg *config.Config, hc *http.Client) *Client {
193 if hc == nil {
194 hc = &http.Client{}
195 }
196- return &Client{BaseURL: baseURL, HTTP: hc}
197+ return &Client{
198+ httpClient: hc,
199+ baseURL: cfg.LLMConfig.BaseURL,
200+ Model: cfg.LLMConfig.Model,
201+ Temperature: cfg.LLMConfig.Temperature,
202+ MaxTokens: cfg.LLMConfig.MaxTokens,
203+ EnableThinking: cfg.LLMConfig.EnableThinking,
204+ }
205 }
206
207 type chatRequest struct {
208@@ -75,16 +72,16 @@ type sseDelta struct {
209 // Generate sends one streaming chat request and returns the full assistant
210 // content, consuming SSE data events until [DONE]. The context bounds the whole
211 // call, including the read.
212-func (c *Client) Generate(ctx context.Context, msgs []Message, opts Options) (string, error) {
213- if c.BaseURL == "" {
214+func (c *Client) Generate(ctx context.Context, msgs []Message) (string, error) {
215+ if c.baseURL == "" {
216 return "", fmt.Errorf("llm: no base URL configured")
217 }
218 body := chatRequest{
219- Model: opts.Model,
220+ Model: c.Model,
221 Messages: msgs,
222- ChatTemplateKwargs: map[string]any{"enable_thinking": opts.EnableThinking},
223- Temperature: opts.Temperature,
224- MaxTokens: opts.MaxTokens,
225+ ChatTemplateKwargs: map[string]any{"enable_thinking": c.EnableThinking},
226+ Temperature: c.Temperature,
227+ MaxTokens: c.MaxTokens,
228 Stream: true,
229 }
230 payload, err := json.Marshal(body)
231@@ -92,14 +89,14 @@ func (c *Client) Generate(ctx context.Context, msgs []Message, opts Options) (st
232 return "", fmt.Errorf("llm: encode request: %w", err)
233 }
234
235- url := strings.TrimRight(c.BaseURL, "/") + "/chat/completions"
236+ url := strings.TrimRight(c.baseURL, "/") + "/chat/completions"
237 req, err := http.NewRequestWithContext(ctx, http.MethodPost, url, bytes.NewReader(payload))
238 if err != nil {
239 return "", fmt.Errorf("llm: build request: %w", err)
240 }
241 req.Header.Set("Content-Type", "application/json")
242
243- resp, err := c.HTTP.Do(req)
244+ resp, err := c.httpClient.Do(req)
245 if err != nil {
246 return "", fmt.Errorf("llm: request to %s: %w", url, err)
247 }