Parent directory

history.go

1245 bytes
 1package llm
 2
 3const (
 4	// maxContextTokens is the safe per-prompt working budget in tokens, kept at
 5	// about half the model's 32768-token context to leave headroom for output and
 6	// for charsPerTokenEstimate over-counting kana.
 7	maxContextTokens      = 16384
 8	charsPerTokenEstimate = 1 // conservative: count one token per rune
 9	// totalContextTokens caps the whole prompt (all recaps + sheets + open
10	// segment), set below the model's 32768-token context so a very long session
11	// drops old recaps instead of overflowing.
12	totalContextTokens = 24000
13)
14
15// HistoryPolicy bounds how much prior conversation a request may carry. The char
16// budgets are rough estimates of the prompt at charsPerTokenEstimate (one token
17// per rune keeps the estimate on the safe side for kana/kanji and leaves romaji
18// far inside the limit). PromptCharBudget triggers per-location compaction of the
19// open segment; TotalContextBudget triggers dropping the oldest closed recaps.
20type HistoryPolicy struct {
21	PromptCharBudget   int
22	TotalContextBudget int
23}
24
25func DefaultHistoryPolicy() HistoryPolicy {
26	return HistoryPolicy{
27		PromptCharBudget:   maxContextTokens * charsPerTokenEstimate,
28		TotalContextBudget: totalContextTokens * charsPerTokenEstimate,
29	}
30}