Skip to slide
Chapter 5 · Prompt Caching and Context Management
33 / 142

CHAPTER 05 · Prompt Caching and Context Management · 6 / 8

Code MVP: cache-aware ordering and a compactor

"""
chapter 05: caching-aware ordering and compaction.
Two tools: keep the prompt prefix stable (so caching keeps hitting), and
compact old history when it grows past a token budget.
Reuses estimate_items_tokens from Chapter 4.
"""

def estimate_tokens(text: str) -> int:
    return max(1, len(str(text)) // 4)

def estimate_items_tokens(items: list) -> int:
    return sum(estimate_tokens(item.get("content", "")) for item in items)

# --- 1. Keep mid-session changes cache-friendly by APPENDING, not editing ---
def apply_change_cache_safely(history: list, change_description: str) -> list:
    """A config change (new cwd, new sandbox mode) is added as a NEW item at
    the end, so the existing prefix is untouched and stays cached."""
    history.append({"role": "developer", "type": "message",
                    "content": f"<context_update>{change_description}</context_update>"})
    return history

# --- 2. Compaction: summarize old turns once we cross a token budget ---
class Compactor:
    def __init__(self, max_tokens: int = 4000, keep_recent: int = 4):
        self.max_tokens = max_tokens      # window budget (toy value)
        self.keep_recent = keep_recent    # always keep the last N items verbatim

    def summarize(self, items: list) -> str:
        """Stand-in for a real summarization model call. A production harness
        would ask the model to compress these items; here we just sketch it."""
        kinds = {}
        for it in items:
            kinds[it.get("type", "message")] = kinds.get(it.get("type", "message"), 0) + 1
        breakdown = ", ".join(f"{n} {k}" for k, n in kinds.items())
        return f"[summary of {len(items)} earlier items: {breakdown}]"

    def compact(self, history: list) -> list:
        if estimate_items_tokens(history) <= self.max_tokens:
            return history                       # under budget: nothing to do

        # ALWAYS preserve the very first item (the original user request)...
        first = history[:1]
        # ...and the most recent items (freshest, least safe to summarize).
        recent = history[-self.keep_recent:]
        middle = history[1:-self.keep_recent] if len(history) > self.keep_recent + 1 else []

        if not middle:
            return history
        summary_item = {"role": "developer", "type": "message",
                        "content": self.summarize(middle)}
        # Oldest tool outputs are the bulk; replacing them frees the most space.
        return first + [summary_item] + recent

if __name__ == "__main__":
    history = [{"role": "user", "type": "message", "content": "fix the build"}]
    # Simulate a long session: many bulky tool results pile up.
    for i in range(30):
        history.append({"role": "assistant", "type": "tool_call",
                        "content": f"run step {i}"})
        history.append({"role": "tool", "type": "tool_result",
                        "content": "X" * 800})  # a big log (~200 tokens each)

    print("before:", estimate_items_tokens(history), "tokens,", len(history), "items")
    compacted = Compactor(max_tokens=4000, keep_recent=4).compact(history)
    print("after :", estimate_items_tokens(compacted), "tokens,", len(compacted), "items")
    print("first item preserved:", compacted[0]["content"])

Run it and the history collapses from thousands of tokens to a few hundred, while the original request ("fix the build") and the last few exchanges survive intact. That is compaction in miniature: throw away the bulky middle, keep the bookends.

← → arrow keys work too