497099b46b8da716e7bd62abb42eeb38b79e309d
- Author
- TheEdgeOfRage <git@theedgeofrage.com>
- Committer
- TheEdgeOfRage <git@theedgeofrage.com>
- Date
Message
Diff
This diff is truncated to protect this page.
1diff --git a/internal/llm/.gitkeep b/internal/llm/.gitkeep
2deleted file mode 100644
3index e69de29bb2d1d6434b8b29ae775ad8c2e48c5391..0000000000000000000000000000000000000000
4--- a/internal/llm/.gitkeep
5+++ /dev/null
6diff --git a/internal/llm/client.go b/internal/llm/client.go
7new file mode 100644
8index 0000000000000000000000000000000000000000..452c2373c621f6f0ecee160c84e68735df73e7ec
9--- /dev/null
10+++ b/internal/llm/client.go
11@@ -0,0 +1,166 @@
12+// Package llm is an OpenAI-compatible chat-completions client with the Qwen3
13+// no-thinking request options, bounded conversation history, and strict parsers
14+// for the NPC reply and judge output contracts. It is an HTTP client only and
15+// never starts or manages any model service.
16+package llm
17+
18+import (
19+ "bufio"
20+ "bytes"
21+ "context"
22+ "encoding/json"
23+ "fmt"
24+ "io"
25+ "net/http"
26+ "strings"
27+)
28+
29+type Role string
30+
31+const (
32+ RoleSystem Role = "system"
33+ RoleUser Role = "user"
34+ RoleAssistant Role = "assistant"
35+)
36+
37+type Message struct {
38+ Role Role `json:"role"`
39+ Content string `json:"content"`
40+}
41+
42+// Options controls a single chat request. The Qwen3 game defaults are produced
43+// by Qwen3Options; tests may flip Stream to exercise the non-streaming path.
44+type Options struct {
45+ Model string
46+ Temperature float64
47+ MaxTokens int
48+ EnableThinking bool
49+ Stream bool
50+}
51+
52+func Qwen3Options() Options {
53+ return Options{Model: "jp", Temperature: 0.4, MaxTokens: 256, EnableThinking: false, Stream: true}
54+}
55+
56+type Client struct {
57+ BaseURL string
58+ HTTP *http.Client
59+}
60+
61+func NewClient(baseURL string, hc *http.Client) *Client {
62+ if hc == nil {
63+ hc = &http.Client{}
64+ }
65+ return &Client{BaseURL: baseURL, HTTP: hc}
66+}
67+
68+type chatRequest struct {
69+ Model string `json:"model"`
70+ Messages []Message `json:"messages"`
71+ ChatTemplateKwargs map[string]any `json:"chat_template_kwargs"`
72+ Temperature float64 `json:"temperature"`
73+ MaxTokens int `json:"max_tokens"`
74+ Stream bool `json:"stream"`
75+}
76+
77+type chatResponse struct {
78+ Choices []struct {
79+ Message struct {
80+ Content string `json:"content"`
81+ } `json:"message"`
82+ } `json:"choices"`
83+}
84+
85+type sseDelta struct {
86+ Choices []struct {
87+ Delta struct {
88+ Content string `json:"content"`
89+ } `json:"delta"`
90+ } `json:"choices"`
91+}
92+
93+// Generate sends one chat request and returns the full assistant content. With
94+// Options.Stream it consumes SSE data events until [DONE]; otherwise it reads a
95+// single JSON body. The context bounds the whole call, including the read.
96+func (c *Client) Generate(ctx context.Context, msgs []Message, opts Options) (string, error) {
97+ if c.BaseURL == "" {
98+ return "", fmt.Errorf("llm: no base URL configured")
99+ }
100+ body := chatRequest{
101+ Model: opts.Model,
102+ Messages: msgs,
103+ ChatTemplateKwargs: map[string]any{"enable_thinking": opts.EnableThinking},
104+ Temperature: opts.Temperature,
105+ MaxTokens: opts.MaxTokens,
106+ Stream: opts.Stream,
107+ }
108+ payload, err := json.Marshal(body)
109+ if err != nil {
110+ return "", fmt.Errorf("llm: encode request: %w", err)
111diff --git a/internal/llm/client_test.go b/internal/llm/client_test.go
112new file mode 100644
113index 0000000000000000000000000000000000000000..3da59aeb1bf962e8bebdac1cb7445dc4a7313136
114--- /dev/null
115+++ b/internal/llm/client_test.go
116@@ -0,0 +1,143 @@
117+package llm
118+
119+import (
120+ "context"
121+ "encoding/json"
122+ "fmt"
123+ "net/http"
124+ "net/http/httptest"
125+ "strings"
126+ "testing"
127+)
128+
129+type capturedRequest struct {
130+ body []byte
131+ path string
132+}
133+
134+func newTestServer(t *testing.T, handler func(w http.ResponseWriter, r *http.Request, reqBody []byte)) (*Client, *capturedRequest) {
135+ t.Helper()
136+ captured := &capturedRequest{}
137+ srv := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
138+ var buf strings.Builder
139+ b := make([]byte, 4096)
140+ for {
141+ n, err := r.Body.Read(b)
142+ buf.Write(b[:n])
143+ if err != nil {
144+ break
145+ }
146+ }
147+ captured.body = []byte(buf.String())
148+ captured.path = r.URL.Path
149+ handler(w, r, captured.body)
150+ }))
151+ t.Cleanup(srv.Close)
152+ return NewClient(srv.URL, srv.Client()), captured
153+}
154+
155+func assertRequestShape(t *testing.T, body []byte, stream bool) {
156+ t.Helper()
157+ var req map[string]any
158+ if err := json.Unmarshal(body, &req); err != nil {
159+ t.Fatalf("request is not JSON: %v", err)
160+ }
161+ if req["model"] != "jp" {
162+ t.Errorf("model = %v, want jp", req["model"])
163+ }
164+ if req["temperature"] != 0.4 {
165+ t.Errorf("temperature = %v, want 0.4", req["temperature"])
166+ }
167+ if req["max_tokens"] != float64(256) {
168+ t.Errorf("max_tokens = %v, want 256", req["max_tokens"])
169+ }
170+ if got := req["stream"]; got != stream {
171+ t.Errorf("stream = %v, want %v", got, stream)
172+ }
173+ kwargs, ok := req["chat_template_kwargs"].(map[string]any)
174+ if !ok {
175+ t.Fatalf("chat_template_kwargs missing or wrong type: %v", req["chat_template_kwargs"])
176+ }
177+ if kwargs["enable_thinking"] != false {
178+ t.Errorf("enable_thinking = %v, want false", kwargs["enable_thinking"])
179+ }
180+ msgs, ok := req["messages"].([]any)
181+ if !ok || len(msgs) == 0 {
182+ t.Fatalf("messages missing: %v", req["messages"])
183+ }
184+ first := msgs[0].(map[string]any)
185+ if first["role"] != "system" {
186+ t.Errorf("first message role = %v, want system", first["role"])
187+ }
188+}
189+
190+func TestGenerateStreamingSSE(t *testing.T) {
191+ client, _ := newTestServer(t, func(w http.ResponseWriter, r *http.Request, body []byte) {
192+ assertRequestShape(t, body, true)
193+ w.Header().Set("Content-Type", "text/event-stream")
194+ fmt.Fprint(w, "data: {\"choices\":[{\"delta\":{\"content\":\"Konnichi\"}}]}\n\n")
195+ fmt.Fprint(w, ": keep-alive comment line should be ignored\n\n")
196+ fmt.Fprint(w, "data: {\"choices\":[{\"delta\":{\"content\":\"wa.\"}}]}\n\n")
197+ fmt.Fprint(w, "data: [DONE]\n\n")
198+ })
199+
200+ got, err := client.Generate(context.Background(), []Message{{Role: RoleSystem, Content: "p"}}, Qwen3Options())
201+ if err != nil {
202+ t.Fatalf("Generate: %v", err)
203+ }
204+ if got != "Konnichiwa." {
205+ t.Errorf("content = %q, want %q", got, "Konnichiwa.")
206+ }
207+}
208+
209+func TestGenerateNonStreaming(t *testing.T) {
210+ client, _ := newTestServer(t, func(w http.ResponseWriter, r *http.Request, body []byte) {
211+ assertRequestShape(t, body, false)
212+ w.Header().Set("Content-Type", "application/json")
213+ fmt.Fprint(w, `{"choices":[{"message":{"role":"assistant","content":"Hai."}}]}`)
214+ })
215+
216diff --git a/internal/llm/contract.go b/internal/llm/contract.go
217new file mode 100644
218index 0000000000000000000000000000000000000000..3320a38b3f7d0c80ffa61704e15321f0342f1719
219--- /dev/null
220+++ b/internal/llm/contract.go
221@@ -0,0 +1,146 @@
222+package llm
223+
224+import (
225+ "strconv"
226+ "strings"
227+)
228+
229+type ContractErrorKind int
230+
231+const (
232+ MissingField ContractErrorKind = iota
233+ DuplicateField
234+ EmptyValue
235+ InvalidFormat
236+ OutOfRange
237+)
238+
239+// ContractError is a recoverable, inspectable failure to parse a model output
240+// contract. Callers can branch on Kind and Field instead of panicking.
241+type ContractError struct {
242+ Kind ContractErrorKind
243+ Field string
244+}
245+
246+func (e *ContractError) Error() string {
247+ return "llm: bad contract field " + e.Field + ": " + kindString(e.Kind)
248+}
249+
250+func kindString(k ContractErrorKind) string {
251+ switch k {
252+ case MissingField:
253+ return "missing"
254+ case DuplicateField:
255+ return "duplicate"
256+ case EmptyValue:
257+ return "empty value"
258+ case InvalidFormat:
259+ return "invalid format"
260+ case OutOfRange:
261+ return "out of range"
262+ default:
263+ return "unknown"
264+ }
265+}
266+
267+var (
268+ npcFieldNames = []string{"ROMAJI", "KANA", "ENGLISH"}
269+ judgeFieldNames = []string{"SCORE", "FEEDBACK"}
270+)
271+
272+// parseFields splits a raw model reply into FIELD|value lines. Surrounding
273+// whitespace is tolerated and blank lines are skipped. A line without a pipe,
274+// an unknown field name, or a repeated field is a recoverable error.
275+func parseFields(raw string, names []string) (map[string]string, error) {
276+ valid := make(map[string]bool, len(names))
277+ for _, n := range names {
278+ valid[n] = true
279+ }
280+ out := make(map[string]string, len(names))
281+ for _, line := range strings.Split(raw, "\n") {
282+ trimmed := strings.TrimSpace(line)
283+ if trimmed == "" {
284+ continue
285+ }
286+ name, value, ok := strings.Cut(trimmed, "|")
287+ if !ok {
288+ return nil, &ContractError{Kind: InvalidFormat, Field: trimmed}
289+ }
290+ name = strings.TrimSpace(name)
291+ value = strings.TrimSpace(value)
292+ if !valid[name] {
293+ return nil, &ContractError{Kind: InvalidFormat, Field: name}
294+ }
295+ if _, dup := out[name]; dup {
296+ return nil, &ContractError{Kind: DuplicateField, Field: name}
297+ }
298+ out[name] = value
299+ }
300+ return out, nil
301+}
302+
303+func requireFields(vals map[string]string, names []string) error {
304+ for _, name := range names {
305+ v, ok := vals[name]
306+ if !ok {
307+ return &ContractError{Kind: MissingField, Field: name}
308+ }
309+ if v == "" {
310+ return &ContractError{Kind: EmptyValue, Field: name}
311+ }
312+ }
313+ return nil
314+}
315+
316+type NPCReply struct {
317+ Romaji string
318+ Kana string
319+ English string
320+}
321diff --git a/internal/llm/contract_test.go b/internal/llm/contract_test.go
322new file mode 100644
323index 0000000000000000000000000000000000000000..427ebc665037c4d9f6deffb2e86bec69d825dae2
324--- /dev/null
325+++ b/internal/llm/contract_test.go
326@@ -0,0 +1,143 @@
327+package llm
328+
329+import (
330+ "errors"
331+ "strings"
332+ "testing"
333+)
334+
335+func kindOf(t *testing.T, err error) ContractErrorKind {
336+ t.Helper()
337+ var ce *ContractError
338+ if !errors.As(err, &ce) {
339+ t.Fatalf("expected *ContractError, got %T (%v)", err, err)
340+ }
341+ return ce.Kind
342+}
343+
344+func TestParseNPCReplyHappy(t *testing.T) {
345+ raw := "ROMAJI|Konnichiwa.\nKANA|こんにちは。\nENGLISH|Hello."
346+ got, err := ParseNPCReply(raw)
347+ if err != nil {
348+ t.Fatalf("unexpected error: %v", err)
349+ }
350+ want := NPCReply{Romaji: "Konnichiwa.", Kana: "こんにちは。", English: "Hello."}
351+ if got != want {
352+ t.Errorf("got %+v, want %+v", got, want)
353+ }
354+}
355+
356+func TestParseNPCReplyWhitespaceTolerance(t *testing.T) {
357+ raw := " \n\tROMAJI| Konnichiwa. \n KANA| こんにちは。 \nENGLISH|Hello.\n "
358+ got, err := ParseNPCReply(raw)
359+ if err != nil {
360+ t.Fatalf("unexpected error: %v", err)
361+ }
362+ if got.Romaji != "Konnichiwa." || got.Kana != "こんにちは。" || got.English != "Hello." {
363+ t.Errorf("got %+v", got)
364+ }
365+}
366+
367+func TestParseNPCReplyMissingField(t *testing.T) {
368+ raw := "ROMAJI|Konnichiwa.\nENGLISH|Hello."
369+ _, err := ParseNPCReply(raw)
370+ if kindOf(t, err) != MissingField {
371+ t.Fatalf("want MissingField, got %v", err)
372+ }
373+}
374+
375+func TestParseNPCReplyDuplicateField(t *testing.T) {
376+ raw := "ROMAJI|a\nKANA|b\nENGLISH|c\nENGLISH|d"
377+ _, err := ParseNPCReply(raw)
378+ if kindOf(t, err) != DuplicateField {
379+ t.Fatalf("want DuplicateField, got %v", err)
380+ }
381+}
382+
383+func TestParseNPCReplyMalformedLine(t *testing.T) {
384+ raw := "ROMAJI|a\nthis line has no pipe\nKANA|b\nENGLISH|c"
385+ if kindOf(t, ParseNPCReplyErr(raw)) != InvalidFormat {
386+ t.Fatalf("want InvalidFormat")
387+ }
388+}
389+
390+func TestParseNPCReplyEmptyValue(t *testing.T) {
391+ raw := "ROMAJI|\nKANA|b\nENGLISH|c"
392+ if kindOf(t, ParseNPCReplyErr(raw)) != EmptyValue {
393+ t.Fatalf("want EmptyValue")
394+ }
395+}
396+
397+func ParseNPCReplyErr(raw string) error {
398+ _, err := ParseNPCReply(raw)
399+ return err
400+}
401+
402+func TestParseJudgeHappy(t *testing.T) {
403+ raw := "SCORE|82\nFEEDBACK|Fushin na hodo, zensei deshita."
404+ got, err := ParseJudge(raw)
405+ if err != nil {
406+ t.Fatalf("unexpected error: %v", err)
407+ }
408+ want := JudgeResult{Score: 82, Feedback: "Fushin na hodo, zensei deshita."}
409+ if got != want {
410+ t.Errorf("got %+v, want %+v", got, want)
411+ }
412+}
413+
414+func TestParseJudgeOutOfRangeHigh(t *testing.T) {
415+ if kindOf(t, ParseJudgeErr("SCORE|101\nFEEDBACK|x")) != OutOfRange {
416+ t.Fatalf("want OutOfRange for 101")
417+ }
418+}
419+
420+func TestParseJudgeOutOfRangeNegative(t *testing.T) {
421+ if kindOf(t, ParseJudgeErr("SCORE|-1\nFEEDBACK|x")) != OutOfRange {
422+ t.Fatalf("want OutOfRange for -1")
423+ }
424+}
425+
426diff --git a/internal/llm/history.go b/internal/llm/history.go
427new file mode 100644
428index 0000000000000000000000000000000000000000..fd4ed88a07179f1e4deb87bdbf05f8d49001b3e3
429--- /dev/null
430+++ b/internal/llm/history.go
431@@ -0,0 +1,57 @@
432+package llm
433+
434+import "unicode/utf8"
435+
436+// Turn is one exchange: the player's spoken line and the NPC's reply to it.
437+type Turn struct {
438+ User string
439+ Assistant string
440+}
441+
442+const (
443+ maxContextTokens = 8192
444+ charsPerTokenEstimate = 1
445+ defaultMaxTurns = 8
446+)
447+
448+// HistoryPolicy bounds how much prior conversation a request may carry. The
449+// char budget is a rough estimate of the whole prompt, derived from the model
450+// context at charsPerTokenEstimate (one token per rune keeps the estimate on
451+// the safe side for kana/kanji and leaves romaji far inside the limit).
452+type HistoryPolicy struct {
453+ MaxTurns int
454+ PromptCharBudget int
455+}
456+
457+func DefaultHistoryPolicy() HistoryPolicy {
458+ return HistoryPolicy{MaxTurns: defaultMaxTurns, PromptCharBudget: maxContextTokens * charsPerTokenEstimate}
459+}
460+
461+func runeCount(s string) int { return utf8.RuneCountInString(s) }
462+
463+// BoundHistory returns the newest turns of prior that fit within maxTurns turns
464+// and whose combined character count stays within charBudget. Turns keep their
465+// original order; when the budget is exceeded the oldest turns are dropped
466+// first, so the result honors whichever cap (turns or chars) is tighter.
467+func BoundHistory(prior []Turn, maxTurns int, charBudget int) []Turn {
468+ if maxTurns <= 0 || charBudget <= 0 || len(prior) == 0 {
469+ return nil
470+ }
471+ start := 0
472+ if n := len(prior); n > maxTurns {
473+ start = n - maxTurns
474+ }
475+ kept := prior[start:]
476+ used := 0
477+ for _, t := range kept {
478+ used += runeCount(t.User) + runeCount(t.Assistant)
479+ }
480+ for len(kept) > 0 && used > charBudget {
481+ used -= runeCount(kept[0].User) + runeCount(kept[0].Assistant)
482+ kept = kept[1:]
483+ }
484+ if len(kept) == 0 {
485+ return nil
486+ }
487+ return kept
488+}
489diff --git a/internal/llm/history_test.go b/internal/llm/history_test.go
490new file mode 100644
491index 0000000000000000000000000000000000000000..08c87b8037cbe93a85ac04966177a3ef3fb4076a
492--- /dev/null
493+++ b/internal/llm/history_test.go
494@@ -0,0 +1,78 @@
495+package llm
496+
497+import (
498+ "testing"
499+)
500+
501+func turn(user, asst string) Turn { return Turn{User: user, Assistant: asst} }
502+
503+func TestBoundHistoryCapsByTurns(t *testing.T) {
504+ prior := make([]Turn, 0, 10)
505+ for i := 0; i < 10; i++ {
506+ prior = append(prior, turn("u", "a"))
507+ }
508+ got := BoundHistory(prior, 3, 1_000_000)
509+ if len(got) != 3 {
510+ t.Fatalf("kept %d turns, want 3", len(got))
511+ }
512+ // newest three are kept: indices 7,8,9 -> all identical content here, so
513+ // verify count and that the tail is preserved by order (last == last input).
514+ if got[len(got)-1] != prior[9] {
515+ t.Errorf("last kept turn = %+v, want last input", got[len(got)-1])
516+ }
517+}
518+
519+func TestBoundHistoryCapsByCharBudget(t *testing.T) {
520+ prior := []Turn{
521+ turn("u0", "a0"),
522+ turn("u1", "a1"),
523+ turn("u2", "a2"),
524+ }
525+ // Each turn is 4 chars. A budget of 8 keeps at most the newest two.
526+ got := BoundHistory(prior, 10, 8)
527+ if len(got) != 2 {
528+ t.Fatalf("kept %d turns, want 2", len(got))
529+ }
530+ if got[0] != prior[1] || got[1] != prior[2] {
531+ t.Errorf("kept wrong turns: %+v", got)
532+ }
533+}
534+
535+func TestBoundHistoryTighterCapWins(t *testing.T) {
536+ prior := []Turn{turn("u0", "a0"), turn("u1", "a1")}
537+ // maxTurns=1 is tighter than the char budget.
538+ got := BoundHistory(prior, 1, 1_000_000)
539+ if len(got) != 1 || got[0] != prior[1] {
540+ t.Fatalf("kept %+v, want newest single turn", got)
541+ }
542+}
543+
544+func TestBoundHistoryDropsAllWhenOverBudget(t *testing.T) {
545+ prior := []Turn{turn("a very long user line that alone exceeds the budget", "x")}
546+ got := BoundHistory(prior, 5, 10)
547+ if len(got) != 0 {
548+ t.Fatalf("expected no turns kept, got %+v", got)
549+ }
550+}
551+
552+func TestBoundHistoryEmptyInputs(t *testing.T) {
553+ if got := BoundHistory(nil, 5, 100); len(got) != 0 {
554+ t.Errorf("nil prior should yield nil, got %+v", got)
555+ }
556+ if got := BoundHistory([]Turn{turn("u", "a")}, 0, 100); len(got) != 0 {
557+ t.Errorf("maxTurns=0 should yield nil, got %+v", got)
558+ }
559+ if got := BoundHistory([]Turn{turn("u", "a")}, 5, 0); len(got) != 0 {
560+ t.Errorf("charBudget=0 should yield nil, got %+v", got)
561+ }
562+}
563+
564+func TestDefaultHistoryPolicy(t *testing.T) {
565+ p := DefaultHistoryPolicy()
566+ if p.MaxTurns <= 0 || p.PromptCharBudget <= 0 {
567+ t.Fatalf("policy not sane: %+v", p)
568+ }
569+ if p.PromptCharBudget != maxContextTokens*charsPerTokenEstimate {
570+ t.Errorf("PromptCharBudget = %d, want derived from context", p.PromptCharBudget)
571+ }
572+}
573diff --git a/internal/llm/prompt.go b/internal/llm/prompt.go
574new file mode 100644
575index 0000000000000000000000000000000000000000..240d09a5699805ed657b7f20397451f8b0c0c463
576--- /dev/null
577+++ b/internal/llm/prompt.go
578@@ -0,0 +1,68 @@
579+package llm
580+
581+import (
582+ "fmt"
583+ "strings"
584+)
585+
586+// npcSystemTemplate describes a real situation only and fixes the three-field
587+// reply contract. It never names a game, quest, or scenario, never refers to
588+// grading, and tells the model to speak as a normal person rather than a
589+// teacher. The persona and situation are inserted as authored data.
590+const npcSystemTemplate = `You are %s.
591+
592+Situation: %s
593+
594+Speak naturally, as yourself, in Japanese. Keep your reply short and conversational. Do not teach language, correct mistakes, or mention any evaluation.
595+
596+Respond with exactly three lines and nothing else, in this order:
597+ROMAJI|<romaji of the sentence you say>
598+KANA|<the same sentence written in kana>
599+ENGLISH|<a natural English translation of that sentence>`
600+
601+// judgeSystemTemplate is persona-free and fixes the two-field judge contract.
602+// It scores fluency, naturalness, and fit to the situation, accepting any
603+// natural phrasing. It never names a game, quest, or scenario.
604diff --git a/internal/llm/prompt_test.go b/internal/llm/prompt_test.go
605new file mode 100644
606index 0000000000000000000000000000000000000000..63737b79defd23cf2dcecd2463feb3ceed868b8b
607--- /dev/null
608+++ b/internal/llm/prompt_test.go
609@@ -0,0 +1,64 @@
610+package llm
611+
612+import (
613+ "strings"
614+ "testing"
615+)
616+
617+func TestBuildNPCMessagesStructure(t *testing.T) {
618+ policy := HistoryPolicy{MaxTurns: 2, PromptCharBudget: 1_000_000}
619+ prior := []Turn{turn("u0", "a0"), turn("u1", "a1")}
620+ msgs := BuildNPCMessages("a ramen chef", "ordering at a counter", "transcript line", prior, policy)
621+
622+ if msgs[0].Role != RoleSystem {
623+ t.Fatalf("first message should be system, got %s", msgs[0].Role)
624+ }
625+ if !strings.Contains(msgs[0].Content, "a ramen chef") || !strings.Contains(msgs[0].Content, "ordering at a counter") {
626+ t.Errorf("system message missing persona/situation: %q", msgs[0].Content)
627+ }
628+ last := msgs[len(msgs)-1]
629+ if last.Role != RoleUser || last.Content != "transcript line" {
630+ t.Fatalf("last message should be the transcript, got %+v", last)
631+ }
632+ // system + 2 turns * 2 + transcript = 6
633+ if len(msgs) != 6 {
634+ t.Fatalf("got %d messages, want 6: %+v", len(msgs), msgs)
635+ }
636+}
637+
638+func TestBuildNPCMessagesBoundedHistory(t *testing.T) {
639+ policy := HistoryPolicy{MaxTurns: 1, PromptCharBudget: 1_000_000}
640+ prior := []Turn{turn("u0", "a0"), turn("u1", "a1")}
641+ msgs := BuildNPCMessages("p", "s", "now", prior, policy)
642+ // system + 1 turn * 2 + transcript = 4
643+ if len(msgs) != 4 {
644+ t.Fatalf("got %d messages, want 4 (bounded to 1 turn)", len(msgs))
645+ }
646+}
647+
648+func TestBuildNPCMessagesNoJudgeLeak(t *testing.T) {
649+ msgs := BuildNPCMessages("p", "s", "spoken line", nil, DefaultHistoryPolicy())
650+ joined := ""
651+ for _, m := range msgs {
652+ joined += m.Content + "\n"
653+ }
654+ if strings.Contains(joined, "SCORE") || strings.Contains(joined, "FEEDBACK") {
655+ t.Errorf("judge fields leaked into NPC prompt: %q", joined)
656+ }
657+}
658+
659+func TestBuildJudgeMessagesStructure(t *testing.T) {
660+ msgs := BuildJudgeMessages("at a station ticket window", "raw japanese line")
661+ if len(msgs) != 2 {
662+ t.Fatalf("got %d messages, want 2", len(msgs))
663+ }
664+ if msgs[0].Role != RoleSystem || msgs[1].Role != RoleUser {
665+ t.Fatalf("roles should be system then user: %+v", msgs)
666+ }
667+ if !strings.Contains(msgs[1].Content, "raw japanese line") {
668+ t.Errorf("judge user message missing transcript: %q", msgs[1].Content)
669+ }
670+ if strings.Contains(msgs[0].Content, "persona") || strings.Contains(strings.ToLower(msgs[0].Content), "npc") {
671+ t.Errorf("judge prompt should be persona-free: %q", msgs[0].Content)
672+ }
673+}