CHAPTER 04 · Tokens, Context Windows, and the Quadratic Problem · 4 / 6
Code MVP: a token and cost estimator
"""
chapter 04: token and cost estimation.
A fast, approximate token counter plus a session cost model that makes
the quadratic growth visible. This becomes the capstone's budget meter.
"""
from dataclasses import dataclass
CHARS_PER_TOKEN = 4 # the standard rough heuristic used by real harnesses
def estimate_tokens(text: str) -> int:
"""Cheap approximation: ~4 characters per token."""
return max(1, len(text) // CHARS_PER_TOKEN)
def estimate_items_tokens(items: list) -> int:
"""Total tokens across a list of structured history items (Ch 3)."""
return sum(estimate_tokens(str(item.get("content", ""))) for item in items)
@dataclass
class CostModel:
input_rate: float = 1.0 # cost per 1k fresh input tokens (arbitrary units)
output_rate: float = 4.0 # output usually costs more than input
cached_rate: float = 0.1 # cached input is ~10x cheaper (Chapter 5)
def turn_cost(self, prompt_tokens: int, output_tokens: int,
cached_tokens: int = 0) -> float:
fresh = prompt_tokens - cached_tokens
return (fresh / 1000) * self.input_rate \
+ (cached_tokens / 1000) * self.cached_rate \
+ (output_tokens / 1000) * self.output_rate
def simulate_session(turns: int, tokens_added_per_turn: int = 1500,
output_per_turn: int = 300, use_cache: bool = False):
"""Show how total cost grows with session length."""
model = CostModel()
history_tokens = 0
total = 0.0
for t in range(1, turns + 1):
history_tokens += tokens_added_per_turn # the prompt keeps growing
# With caching, everything except this turn's new content is cached.
cached = history_tokens - tokens_added_per_turn if use_cache else 0
total += model.turn_cost(history_tokens, output_per_turn, cached)
return round(total, 2)
if __name__ == "__main__":
for n in (5, 10, 20, 50):
no_cache = simulate_session(n, use_cache=False)
cached = simulate_session(n, use_cache=True)
print(f"{n:>3} turns | no cache: {no_cache:>8} | with cache: {cached:>7}")
Run it and watch the "no cache" column grow far faster than the turn count, while the "with cache" column grows much more gently. That gap is the entire argument for Chapter 5. The numbers are in arbitrary units, but the shape is what matters: without caching, doubling the turns roughly quadruples the cost; with caching, the per-turn marginal cost grows slowly.