history.go
1245 bytes
1package llm
2
3const (
4 // maxContextTokens is the safe per-prompt working budget in tokens, kept at
5 // about half the model's 32768-token context to leave headroom for output and
6 // for charsPerTokenEstimate over-counting kana.
7 maxContextTokens = 16384
8 charsPerTokenEstimate = 1 // conservative: count one token per rune
9 // totalContextTokens caps the whole prompt (all recaps + sheets + open
10 // segment), set below the model's 32768-token context so a very long session
11 // drops old recaps instead of overflowing.
12 totalContextTokens = 24000
13)
14
15// HistoryPolicy bounds how much prior conversation a request may carry. The char
16// budgets are rough estimates of the prompt at charsPerTokenEstimate (one token
17// per rune keeps the estimate on the safe side for kana/kanji and leaves romaji
18// far inside the limit). PromptCharBudget triggers per-location compaction of the
19// open segment; TotalContextBudget triggers dropping the oldest closed recaps.
20type HistoryPolicy struct {
21 PromptCharBudget int
22 TotalContextBudget int
23}
24
25func DefaultHistoryPolicy() HistoryPolicy {
26 return HistoryPolicy{
27 PromptCharBudget: maxContextTokens * charsPerTokenEstimate,
28 TotalContextBudget: totalContextTokens * charsPerTokenEstimate,
29 }
30}