How Coding Agents Manage Memory

Coding agents primarily utilize file-based rules along with in-session compaction; however, some agents like Claude Code and Cursor also depend on embedding-based cross-session memory. Although they can occasionally feature self-editing blocks, this approach is generally excessive unless the agent operates in long-lived sessions.

First, same codes of in-session approach — no persistence, no embeddings. It’s what most coding agents actually ship.

# compaction.py
from __future__ import annotations
from dataclasses import dataclass, field
from datetime import datetime
from typing import Literal, Any
import asyncio
PRUNE_MINIMUM = 20_000
PRUNE_PROTECT = 40_000
PRUNE_PROTECTED_TOOLS = {"skill"}
CHARS_PER_TOKEN = 4
def estimate_tokens(text: str) -> int:
return max(0, round(len(text or "") / CHARS_PER_TOKEN))
@dataclass
class TokenUsage:
input: int = 0
output: int = 0
cache_read: int = 0
cache_write: int = 0
@dataclass
class ModelLimit:
context: int = 0
input: int = 0
output: int = 0
@dataclass
class ToolState:
status: Literal["running", "completed", "error"]
output: str = ""
attachments: list[dict] = field(default_factory=list)
time: dict = field(default_factory=dict) # includes "compacted": timestamp | None
@dataclass
class ToolPart:
type: Literal["tool"] = "tool"
tool: str = ""
state: ToolState = field(default_factory=ToolState)
# ... other fields omitted for brevity
@dataclass
class Message:
role: Literal["user", "assistant"]
id: str = ""
parts: list[Any] = field(default_factory=list)
tokens: TokenUsage = field(default_factory=TokenUsage)
summary: bool = False # marks compaction summary messages
finish: str | None = None
def is_overflow(tokens: TokenUsage, model: ModelLimit, output_token_max: int) -> bool:
"""Direct port of SessionCompaction.isOverflow."""
context = model.context
if context == 0:
return False
count = tokens.input + tokens.cache_read + tokens.output
output = min(model.output, output_token_max) or output_token_max
usable = model.input or (context - output)
return count > usable
def prune(messages: list[Message]) -> list[ToolPart]:
"""
Walk backwards through tool calls. Preserve the most recent PRUNE_PROTECT
tokens of tool output. Mark older ones as compacted (output cleared).
Returns the list of parts that were pruned.
"""
total = 0
pruned = 0
to_prune: list[ToolPart] = []
turns = 0
for msg in reversed(messages):
if msg.role == "user":
turns += 1
if turns < 2:
continue
if msg.role == "assistant" and msg.summary:
break
for part in reversed(msg.parts):
if not isinstance(part, ToolPart):
continue
if part.state.status != "completed":
continue
if part.tool in PRUNE_PROTECTED_TOOLS:
continue
if part.state.time.get("compacted"):
break
estimate = estimate_tokens(part.state.output)
total += estimate
if total > PRUNE_PROTECT:
pruned += estimate
to_prune.append(part)
if pruned > PRUNE_MINIMUM:
for part in to_prune:
part.state.time["compacted"] = datetime.now().timestamp()
part.state.output = "[Old tool result content cleared]"
part.state.attachments = []
return to_prune
return []
def filter_compacted(messages: list[Message]) -> list[Message]:
"""
Port of MessageV2.filterCompacted: keep only messages after the last
completed compaction summary. Returns in chronological order.
"""
result: list[Message] = []
completed_parents: set[str] = set()
for msg in messages:
result.append(msg)
if (
msg.role == "user"
and msg.id in completed_parents
and any(getattr(p, "type", None) == "compaction" for p in msg.parts)
):
break
if msg.role == "assistant" and msg.summary and msg.finish:
completed_parents.add(getattr(msg, "parent_id", ""))
return result

Second, File-based persistent rules (Claude Code / OpenCode style). This is the most common “memory” in production coding agents.

# rules_loader.py
from pathlib import Path
import os
import httpx
LOCAL_RULE_FILES = ["AGENTS.md", "CLAUDE.md", "CONTEXT.md"]
GLOBAL_RULE_FILES = [
Path.home() / ".config" / "opencode" / "AGENTS.md",
Path.home() / ".claude" / "CLAUDE.md",
]
def find_up(filename: str, cwd: Path, root: Path) -> list[Path]:
"""Walk up from cwd to root looking for filename."""
matches = []
current = cwd.resolve()
while True:
candidate = current / filename
if candidate.exists():
matches.append(candidate)
break
if current == root.resolve() or current.parent == current:
break
current = current.parent
return matches
async def load_custom_instructions(
cwd: Path,
root: Path,
config_instructions: list[str] | None = None,
) -> list[str]:
"""Port of SystemPrompt.custom() — loads rule files + URLs each turn."""
paths: set[Path] = set()
urls: list[str] = []
# local rule files (first match wins per file)
for rule_file in LOCAL_RULE_FILES:
matches = find_up(rule_file, cwd, root)
if matches:
paths.add(matches[0])
break
# global rule files (first existing wins)
for gpath in GLOBAL_RULE_FILES:
if gpath.exists():
paths.add(gpath)
break
# config instructions: paths, globs, or URLs
for instruction in config_instructions or []:
if instruction.startswith(("https://", "http://")):
urls.append(instruction)
continue
if instruction.startswith("~/"):
instruction = str(Path.home() / instruction[2:])
p = Path(instruction)
if p.is_absolute():
matches = list(p.parent.glob(p.name))
paths.update(m for m in matches if m.is_file())
# else: glob resolution omitted for brevity
# load in parallel
async def read_file(p: Path) -> str:
try:
text = p.read_text()
return f"Instructions from: {p}\n{text}"
except Exception:
return ""
async def fetch_url(url: str) -> str:
try:
resp = await httpx.AsyncClient(timeout=5.0).get(url)
if resp.status_code == 200:
return f"Instructions from: {url}\n{resp.text}"
except Exception:
pass
return ""
results = await asyncio.gather(
*[read_file(p) for p in paths],
*[fetch_url(u) for u in urls],
)
return [r for r in results if r]

Third, Vector store recall (RAG-style memory — what LangChain/Mem0 do). This is where embedding stores enter. The pattern: extract “memories” from conversations, embed them, retrieve top-k relevant ones each turn. As is discussed in my previous blog, RAG is fading hence not a focus now.

Fourth, Summary + entity memory (Letta/MemGPT style). The most sophisticated approach: the model manages its own memory as structured blocks, not just a vector dump. The system prompt then includes memory.render() so the model always sees its current memory state, and the tools let it self-edit. This is how Letta works — the model decides what to remember rather than a separate extraction pipeline.

Leave a Reply