fix: add sliding window to conversation history (issue #8)
- Add _trim_conversation_history() with message count + token limits - Configurable via CLOVER_MAX_HISTORY_MESSAGES (default: 20) - Configurable via CLOVER_MAX_HISTORY_TOKENS (default: 8000) - Trims oldest messages first, preserves last 6 minimum - Prevents OOM and degraded LLM quality on long sessions
This commit is contained in:
parent
b25fb3c1c6
commit
33258ce2ad
46
ai_agent.py
46
ai_agent.py
@ -34,6 +34,8 @@ class AIAgent:
|
|||||||
self.model_manager = ModelManager()
|
self.model_manager = ModelManager()
|
||||||
self.conversation_history = []
|
self.conversation_history = []
|
||||||
self.available_tools = self._setup_tools()
|
self.available_tools = self._setup_tools()
|
||||||
|
self.max_history_messages = self.config.get("max_history_messages", 20)
|
||||||
|
self.max_history_tokens = self.config.get("max_history_tokens", 8000)
|
||||||
|
|
||||||
def _setup_tools(self):
|
def _setup_tools(self):
|
||||||
"""Setup available tools for the AI agent"""
|
"""Setup available tools for the AI agent"""
|
||||||
@ -267,6 +269,47 @@ When using tools, be methodical and explain each step. Always test your creation
|
|||||||
except Exception as e:
|
except Exception as e:
|
||||||
return {"error": f"Tool execution failed: {str(e)}"}
|
return {"error": f"Tool execution failed: {str(e)}"}
|
||||||
|
|
||||||
|
def _trim_conversation_history(self):
|
||||||
|
"""
|
||||||
|
Trim conversation history to prevent unbounded memory growth.
|
||||||
|
Uses a sliding window approach keeping the most recent messages.
|
||||||
|
|
||||||
|
Preserves at least the last N messages (max_history_messages)
|
||||||
|
and stays within token budget (max_history_tokens).
|
||||||
|
"""
|
||||||
|
if len(self.conversation_history) <= self.max_history_messages:
|
||||||
|
return
|
||||||
|
|
||||||
|
# Rough token estimation: ~4 chars per token
|
||||||
|
def estimate_tokens(msg):
|
||||||
|
return len(msg.get("content", "")) // 4
|
||||||
|
|
||||||
|
# Calculate total tokens in history
|
||||||
|
total_tokens = sum(estimate_tokens(msg) for msg in self.conversation_history)
|
||||||
|
|
||||||
|
if total_tokens <= self.max_history_tokens:
|
||||||
|
# Within token budget, just apply message count limit
|
||||||
|
if len(self.conversation_history) > self.max_history_messages:
|
||||||
|
self.conversation_history = self.conversation_history[
|
||||||
|
-self.max_history_messages :
|
||||||
|
]
|
||||||
|
return
|
||||||
|
|
||||||
|
# Over token budget - trim from the front, keeping recent messages
|
||||||
|
while (
|
||||||
|
len(self.conversation_history) > 6
|
||||||
|
and sum(estimate_tokens(msg) for msg in self.conversation_history)
|
||||||
|
> self.max_history_tokens
|
||||||
|
):
|
||||||
|
# Remove pairs of messages (user + assistant) from the front
|
||||||
|
self.conversation_history = self.conversation_history[2:]
|
||||||
|
|
||||||
|
# Also enforce message count limit
|
||||||
|
if len(self.conversation_history) > self.max_history_messages:
|
||||||
|
self.conversation_history = self.conversation_history[
|
||||||
|
-self.max_history_messages :
|
||||||
|
]
|
||||||
|
|
||||||
def chat(self, user_message: str) -> str:
|
def chat(self, user_message: str) -> str:
|
||||||
"""
|
"""
|
||||||
Have a conversation with the user, using tools as needed
|
Have a conversation with the user, using tools as needed
|
||||||
@ -288,6 +331,9 @@ When using tools, be methodical and explain each step. Always test your creation
|
|||||||
|
|
||||||
# Prepare messages for AI
|
# Prepare messages for AI
|
||||||
messages = [{"role": "system", "content": self._create_system_prompt()}]
|
messages = [{"role": "system", "content": self._create_system_prompt()}]
|
||||||
|
|
||||||
|
# Trim history to prevent unbounded memory growth
|
||||||
|
self._trim_conversation_history()
|
||||||
messages.extend(self.conversation_history)
|
messages.extend(self.conversation_history)
|
||||||
|
|
||||||
# Get AI response
|
# Get AI response
|
||||||
|
|||||||
@ -43,6 +43,10 @@ def load_config():
|
|||||||
"project_root": os.getenv("CLOVER_PROJECT_ROOT", "."),
|
"project_root": os.getenv("CLOVER_PROJECT_ROOT", "."),
|
||||||
"summary_file": os.getenv("CLOVER_SUMMARY_FILE", "clover.md"),
|
"summary_file": os.getenv("CLOVER_SUMMARY_FILE", "clover.md"),
|
||||||
"structure_file": os.getenv("CLOVER_STRUCTURE_FILE", "structure.md"),
|
"structure_file": os.getenv("CLOVER_STRUCTURE_FILE", "structure.md"),
|
||||||
|
"max_history_messages": int(
|
||||||
|
os.getenv("CLOVER_MAX_HISTORY_MESSAGES", "20")
|
||||||
|
),
|
||||||
|
"max_history_tokens": int(os.getenv("CLOVER_MAX_HISTORY_TOKENS", "8000")),
|
||||||
}
|
}
|
||||||
|
|
||||||
# Debug output if enabled
|
# Debug output if enabled
|
||||||
|
|||||||
Loading…
x
Reference in New Issue
Block a user