feat: separate ContextWindow from MaxTokens and implement token-oriented compaction
- Add ContextWindow field to AgentDefaults (separates context size from output limit) - Add CompactionConfig for flexible compaction settings: - reserve_tokens: tokens to reserve after compaction (default 20000) - reserve_tokens_floor: minimum reserve tokens (default 20000) - keep_recent_tokens: how many recent tokens to keep (default 20000) - Remove hardcoded message count threshold (>20 messages) - Use token-oriented compaction trigger: - Trigger: tokenEstimate > ContextWindow - reserveTokensFloor - For 200k context window: triggers at ~180k tokens (90% threshold) - Add summarizeSessionWithTokenLimit method: - Keeps last N tokens instead of fixed message count - More accurate for large context windows Related: Issue #653 (context_window bug)
This commit is contained in:
parent
4cc8b90da9
commit
9e048ac29a
3 changed files with 148 additions and 14 deletions
|
|
@ -88,6 +88,12 @@ func NewAgentInstance(
|
||||||
temperature = *defaults.Temperature
|
temperature = *defaults.Temperature
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// Resolve ContextWindow: use from config if set, otherwise use maxTokens as fallback
|
||||||
|
contextWindow := defaults.ContextWindow
|
||||||
|
if contextWindow == 0 {
|
||||||
|
contextWindow = maxTokens
|
||||||
|
}
|
||||||
|
|
||||||
// Resolve fallback candidates
|
// Resolve fallback candidates
|
||||||
modelCfg := providers.ModelConfig{
|
modelCfg := providers.ModelConfig{
|
||||||
Primary: model,
|
Primary: model,
|
||||||
|
|
@ -104,7 +110,7 @@ func NewAgentInstance(
|
||||||
MaxIterations: maxIter,
|
MaxIterations: maxIter,
|
||||||
MaxTokens: maxTokens,
|
MaxTokens: maxTokens,
|
||||||
Temperature: temperature,
|
Temperature: temperature,
|
||||||
ContextWindow: maxTokens,
|
ContextWindow: contextWindow,
|
||||||
Provider: provider,
|
Provider: provider,
|
||||||
Sessions: sessionsManager,
|
Sessions: sessionsManager,
|
||||||
ContextBuilder: contextBuilder,
|
ContextBuilder: contextBuilder,
|
||||||
|
|
|
||||||
|
|
@ -29,6 +29,11 @@ import (
|
||||||
"github.com/sipeed/picoclaw/pkg/utils"
|
"github.com/sipeed/picoclaw/pkg/utils"
|
||||||
)
|
)
|
||||||
|
|
||||||
|
const (
|
||||||
|
DEFAULT_COMPACTION_RESERVE_TOKENS_FLOOR = 20000
|
||||||
|
DEFAULT_COMPACTION_KEEP_RECENT_TOKENS = 20000
|
||||||
|
)
|
||||||
|
|
||||||
type AgentLoop struct {
|
type AgentLoop struct {
|
||||||
bus *bus.MessageBus
|
bus *bus.MessageBus
|
||||||
cfg *config.Config
|
cfg *config.Config
|
||||||
|
|
@ -748,12 +753,27 @@ func (al *AgentLoop) updateToolContexts(agent *AgentInstance, channel, chatID st
|
||||||
}
|
}
|
||||||
|
|
||||||
// maybeSummarize triggers summarization if the session history exceeds thresholds.
|
// maybeSummarize triggers summarization if the session history exceeds thresholds.
|
||||||
|
// Uses token-oriented approach instead of message count to better handle large context windows.
|
||||||
func (al *AgentLoop) maybeSummarize(agent *AgentInstance, sessionKey, channel, chatID string) {
|
func (al *AgentLoop) maybeSummarize(agent *AgentInstance, sessionKey, channel, chatID string) {
|
||||||
newHistory := agent.Sessions.GetHistory(sessionKey)
|
newHistory := agent.Sessions.GetHistory(sessionKey)
|
||||||
tokenEstimate := al.estimateTokens(newHistory)
|
tokenEstimate := al.estimateTokens(newHistory)
|
||||||
threshold := agent.ContextWindow * 75 / 100
|
|
||||||
|
|
||||||
if len(newHistory) > 20 || tokenEstimate > threshold {
|
// Get compaction settings from defaults (use configured values or defaults)
|
||||||
|
reserveTokensFloor := DEFAULT_COMPACTION_RESERVE_TOKENS_FLOOR
|
||||||
|
keepRecentTokens := DEFAULT_COMPACTION_KEEP_RECENT_TOKENS
|
||||||
|
|
||||||
|
if al.cfg != nil && al.cfg.Agents.Defaults.Compaction.ReserveTokensFloor > 0 {
|
||||||
|
reserveTokensFloor = al.cfg.Agents.Defaults.Compaction.ReserveTokensFloor
|
||||||
|
}
|
||||||
|
if al.cfg != nil && al.cfg.Agents.Defaults.Compaction.KeepRecentTokens > 0 {
|
||||||
|
keepRecentTokens = al.cfg.Agents.Defaults.Compaction.KeepRecentTokens
|
||||||
|
}
|
||||||
|
|
||||||
|
// Trigger compaction when we're close to the context window limit
|
||||||
|
// Threshold: ContextWindow - ReserveTokensFloor (e.g., for 200k context: trigger at 180k)
|
||||||
|
triggerThreshold := agent.ContextWindow - reserveTokensFloor
|
||||||
|
|
||||||
|
if tokenEstimate > triggerThreshold {
|
||||||
summarizeKey := agent.ID + ":" + sessionKey
|
summarizeKey := agent.ID + ":" + sessionKey
|
||||||
if _, loading := al.summarizing.LoadOrStore(summarizeKey, true); !loading {
|
if _, loading := al.summarizing.LoadOrStore(summarizeKey, true); !loading {
|
||||||
go func() {
|
go func() {
|
||||||
|
|
@ -765,7 +785,7 @@ func (al *AgentLoop) maybeSummarize(agent *AgentInstance, sessionKey, channel, c
|
||||||
Content: "Memory threshold reached. Optimizing conversation history...",
|
Content: "Memory threshold reached. Optimizing conversation history...",
|
||||||
})
|
})
|
||||||
}
|
}
|
||||||
al.summarizeSession(agent, sessionKey)
|
al.summarizeSessionWithTokenLimit(agent, sessionKey, keepRecentTokens)
|
||||||
}()
|
}()
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
@ -1020,6 +1040,106 @@ func (al *AgentLoop) summarizeBatch(
|
||||||
return response.Content, nil
|
return response.Content, nil
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// summarizeSessionWithTokenLimit summarizes conversation history using a token-oriented approach.
|
||||||
|
// Keeps the last keepRecentTokens tokens instead of a fixed number of messages.
|
||||||
|
func (al *AgentLoop) summarizeSessionWithTokenLimit(agent *AgentInstance, sessionKey string, keepRecentTokens int) {
|
||||||
|
ctx, cancel := context.WithTimeout(context.Background(), 120*time.Second)
|
||||||
|
defer cancel()
|
||||||
|
|
||||||
|
history := agent.Sessions.GetHistory(sessionKey)
|
||||||
|
summary := agent.Sessions.GetSummary(sessionKey)
|
||||||
|
|
||||||
|
// Find the split point based on token limit
|
||||||
|
// We iterate from the end, accumulating tokens until we hit the limit
|
||||||
|
keepStartIdx := len(history)
|
||||||
|
currentTokens := 0
|
||||||
|
|
||||||
|
for i := len(history) - 1; i >= 0; i-- {
|
||||||
|
msgTokens := al.estimateTokens([]providers.Message{history[i]})
|
||||||
|
if currentTokens+msgTokens > keepRecentTokens {
|
||||||
|
break
|
||||||
|
}
|
||||||
|
currentTokens += msgTokens
|
||||||
|
keepStartIdx = i
|
||||||
|
}
|
||||||
|
|
||||||
|
// If there's not enough to summarize, return
|
||||||
|
if keepStartIdx <= 1 {
|
||||||
|
return
|
||||||
|
}
|
||||||
|
|
||||||
|
toSummarize := history[:keepStartIdx]
|
||||||
|
if len(toSummarize) == 0 {
|
||||||
|
return
|
||||||
|
}
|
||||||
|
|
||||||
|
// Oversized Message Guard
|
||||||
|
maxMessageTokens := agent.ContextWindow / 2
|
||||||
|
validMessages := make([]providers.Message, 0)
|
||||||
|
omitted := false
|
||||||
|
|
||||||
|
for _, m := range toSummarize {
|
||||||
|
if m.Role != "user" && m.Role != "assistant" {
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
msgTokens := len(m.Content) / 2
|
||||||
|
if msgTokens > maxMessageTokens {
|
||||||
|
omitted = true
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
validMessages = append(validMessages, m)
|
||||||
|
}
|
||||||
|
|
||||||
|
if len(validMessages) == 0 {
|
||||||
|
return
|
||||||
|
}
|
||||||
|
|
||||||
|
// Multi-Part Summarization
|
||||||
|
var finalSummary string
|
||||||
|
if len(validMessages) > 10 {
|
||||||
|
mid := len(validMessages) / 2
|
||||||
|
part1 := validMessages[:mid]
|
||||||
|
part2 := validMessages[mid:]
|
||||||
|
|
||||||
|
s1, _ := al.summarizeBatch(ctx, agent, part1, "")
|
||||||
|
s2, _ := al.summarizeBatch(ctx, agent, part2, "")
|
||||||
|
|
||||||
|
mergePrompt := fmt.Sprintf(
|
||||||
|
"Merge these two conversation summaries into one cohesive summary:\n\n1: %s\n\n2: %s",
|
||||||
|
s1,
|
||||||
|
s2,
|
||||||
|
)
|
||||||
|
resp, err := agent.Provider.Chat(
|
||||||
|
ctx,
|
||||||
|
[]providers.Message{{Role: "user", Content: mergePrompt}},
|
||||||
|
nil,
|
||||||
|
agent.Model,
|
||||||
|
map[string]any{
|
||||||
|
"max_tokens": 1024,
|
||||||
|
"temperature": 0.3,
|
||||||
|
},
|
||||||
|
)
|
||||||
|
if err == nil {
|
||||||
|
finalSummary = resp.Content
|
||||||
|
} else {
|
||||||
|
finalSummary = s1 + " " + s2
|
||||||
|
}
|
||||||
|
} else {
|
||||||
|
finalSummary, _ = al.summarizeBatch(ctx, agent, validMessages, summary)
|
||||||
|
}
|
||||||
|
|
||||||
|
if omitted && finalSummary != "" {
|
||||||
|
finalSummary += "\n[Note: Some oversized messages were omitted from this summary for efficiency.]"
|
||||||
|
}
|
||||||
|
|
||||||
|
if finalSummary != "" {
|
||||||
|
agent.Sessions.SetSummary(sessionKey, finalSummary)
|
||||||
|
// Keep only the messages that weren't summarized
|
||||||
|
agent.Sessions.SetHistory(sessionKey, history[keepStartIdx:])
|
||||||
|
agent.Sessions.Save(sessionKey)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
// estimateTokens estimates the number of tokens in a message list.
|
// estimateTokens estimates the number of tokens in a message list.
|
||||||
// Uses a safe heuristic of 2.5 characters per token to account for CJK and other
|
// Uses a safe heuristic of 2.5 characters per token to account for CJK and other
|
||||||
// overheads better than the previous 3 chars/token.
|
// overheads better than the previous 3 chars/token.
|
||||||
|
|
|
||||||
|
|
@ -175,8 +175,16 @@ type AgentDefaults struct {
|
||||||
ImageModel string `json:"image_model,omitempty" env:"PICOCLAW_AGENTS_DEFAULTS_IMAGE_MODEL"`
|
ImageModel string `json:"image_model,omitempty" env:"PICOCLAW_AGENTS_DEFAULTS_IMAGE_MODEL"`
|
||||||
ImageModelFallbacks []string `json:"image_model_fallbacks,omitempty"`
|
ImageModelFallbacks []string `json:"image_model_fallbacks,omitempty"`
|
||||||
MaxTokens int `json:"max_tokens" env:"PICOCLAW_AGENTS_DEFAULTS_MAX_TOKENS"`
|
MaxTokens int `json:"max_tokens" env:"PICOCLAW_AGENTS_DEFAULTS_MAX_TOKENS"`
|
||||||
|
ContextWindow int `json:"context_window,omitempty" env:"PICOCLAW_AGENTS_DEFAULTS_CONTEXT_WINDOW"`
|
||||||
Temperature *float64 `json:"temperature,omitempty" env:"PICOCLAW_AGENTS_DEFAULTS_TEMPERATURE"`
|
Temperature *float64 `json:"temperature,omitempty" env:"PICOCLAW_AGENTS_DEFAULTS_TEMPERATURE"`
|
||||||
MaxToolIterations int `json:"max_tool_iterations" env:"PICOCLAW_AGENTS_DEFAULTS_MAX_TOOL_ITERATIONS"`
|
MaxToolIterations int `json:"max_tool_iterations" env:"PICOCLAW_AGENTS_DEFAULTS_MAX_TOOL_ITERATIONS"`
|
||||||
|
Compaction CompactionConfig `json:"compaction,omitempty"`
|
||||||
|
}
|
||||||
|
|
||||||
|
type CompactionConfig struct {
|
||||||
|
ReserveTokens int `json:"reserve_tokens,omitempty" env:"PICOCLAW_AGENTS_DEFAULTS_COMPACTION_RESERVE_TOKENS"`
|
||||||
|
ReserveTokensFloor int `json:"reserve_tokens_floor,omitempty" env:"PICOCLAW_AGENTS_DEFAULTS_COMPACTION_RESERVE_TOKENS_FLOOR"`
|
||||||
|
KeepRecentTokens int `json:"keep_recent_tokens,omitempty" env:"PICOCLAW_AGENTS_DEFAULTS_COMPACTION_KEEP_RECENT_TOKENS"`
|
||||||
}
|
}
|
||||||
|
|
||||||
type ChannelsConfig struct {
|
type ChannelsConfig struct {
|
||||||
|
|
|
||||||
Loading…
Add table
Reference in a new issue