CHAPTER 05 · Prompt Caching and Context Management · 6 / 8
Code MVP: cache-aware ordering and a compactor
"""
chapter 05: caching-aware ordering and compaction.
Two tools: keep the prompt prefix stable (so caching keeps hitting), and
compact old history when it grows past a token budget.
Reuses estimate_items_tokens from Chapter 4.
"""
def estimate_tokens(text: str) -> int:
return max(1, len(str(text)) // 4)
def estimate_items_tokens(items: list) -> int:
return sum(estimate_tokens(item.get("content", "")) for item in items)
# --- 1. Keep mid-session changes cache-friendly by APPENDING, not editing ---
def apply_change_cache_safely(history: list, change_description: str) -> list:
"""A config change (new cwd, new sandbox mode) is added as a NEW item at
the end, so the existing prefix is untouched and stays cached."""
history.append({"role": "developer", "type": "message",
"content": f"<context_update>{change_description}</context_update>"})
return history
# --- 2. Compaction: summarize old turns once we cross a token budget ---
class Compactor:
def __init__(self, max_tokens: int = 4000, keep_recent: int = 4):
self.max_tokens = max_tokens # window budget (toy value)
self.keep_recent = keep_recent # always keep the last N items verbatim
def summarize(self, items: list) -> str:
"""Stand-in for a real summarization model call. A production harness
would ask the model to compress these items; here we just sketch it."""
kinds = {}
for it in items:
kinds[it.get("type", "message")] = kinds.get(it.get("type", "message"), 0) + 1
breakdown = ", ".join(f"{n} {k}" for k, n in kinds.items())
return f"[summary of {len(items)} earlier items: {breakdown}]"
def compact(self, history: list) -> list:
if estimate_items_tokens(history) <= self.max_tokens:
return history # under budget: nothing to do
# ALWAYS preserve the very first item (the original user request)...
first = history[:1]
# ...and the most recent items (freshest, least safe to summarize).
recent = history[-self.keep_recent:]
middle = history[1:-self.keep_recent] if len(history) > self.keep_recent + 1 else []
if not middle:
return history
summary_item = {"role": "developer", "type": "message",
"content": self.summarize(middle)}
# Oldest tool outputs are the bulk; replacing them frees the most space.
return first + [summary_item] + recent
if __name__ == "__main__":
history = [{"role": "user", "type": "message", "content": "fix the build"}]
# Simulate a long session: many bulky tool results pile up.
for i in range(30):
history.append({"role": "assistant", "type": "tool_call",
"content": f"run step {i}"})
history.append({"role": "tool", "type": "tool_result",
"content": "X" * 800}) # a big log (~200 tokens each)
print("before:", estimate_items_tokens(history), "tokens,", len(history), "items")
compacted = Compactor(max_tokens=4000, keep_recent=4).compact(history)
print("after :", estimate_items_tokens(compacted), "tokens,", len(compacted), "items")
print("first item preserved:", compacted[0]["content"])
Run it and the history collapses from thousands of tokens to a few hundred, while the original request ("fix the build") and the last few exchanges survive intact. That is compaction in miniature: throw away the bulky middle, keep the bookends.