Skip to slide
Chapter 14 · Autonomous Task Agents: Manus and CodeAct
107 / 142

CHAPTER 14 · Autonomous Task Agents: Manus and CodeAct · 8 / 10

Code MVP: a CodeAct-style agent loop

"""
chapter 14: a CodeAct-style autonomous loop with file-based memory.
The model's "action" is a short Python snippet (CodeAct). A planner writes a
todo.md checklist; the loop executes one action per iteration and records
progress to disk. (Execution here is simplified; use Chapter 7's sandbox.)
"""
import os, io, contextlib

WORKSPACE = "/tmp/manus_demo"
os.makedirs(WORKSPACE, exist_ok=True)

# --- Planner: goal -> ordered steps, persisted as todo.md ---
def make_plan(goal: str) -> list:
    # A real planner asks the model; here we sketch a fixed decomposition.
    steps = [f"Gather info for: {goal}", "Analyze findings", "Write output.md"]
    with open(os.path.join(WORKSPACE, "todo.md"), "w") as f:
        f.write("\n".join(f"- [ ] {s}" for s in steps))
    return steps

def mark_done(step_index: int, steps: list):
    """Tick a step off in todo.md (file-based memory, survives context loss)."""
    lines = [f"- [{'x' if i <= step_index else ' '}] {s}" for i, s in enumerate(steps)]
    open(os.path.join(WORKSPACE, "todo.md"), "w").write("\n".join(lines))

# --- CodeAct executor: run the model's Python action, capture output ---
def execute_code_action(code: str) -> str:
    """Run a snippet and capture stdout. PRODUCTION: run in Chapter 7's sandbox,
    never with raw exec on the host."""
    buf = io.StringIO()
    try:
        with contextlib.redirect_stdout(buf):
            exec(code, {"WORKSPACE": WORKSPACE, "os": os})   # toy sandbox only
        return buf.getvalue() or "(no output)"
    except Exception as e:
        return f"ERROR: {e}"            # the agent reads this and self-corrects

# --- The loop: analyze -> plan -> execute -> observe, one action per step ---
def codeact_loop(goal: str, code_for_step):
    steps = make_plan(goal)
    event_stream = [{"type": "user", "content": goal}]
    for i, step in enumerate(steps):
        code = code_for_step(i, step)               # the model writes Python here
        event_stream.append({"type": "action", "content": code})
        observation = execute_code_action(code)     # ONE action, then observe
        event_stream.append({"type": "observation", "content": observation})
        mark_done(i, steps)                         # persist progress to disk
    return event_stream

if __name__ == "__main__":
    # Fake "model": returns a Python action for each step (CodeAct in action).
    def code_for_step(i, step):
        if i == 0:
            return "print('gathered: agents need a loop, tools, memory')"
        if i == 1:
            return "print('analysis: the loop is simple, the harness is hard')"
        return ("open(os.path.join(WORKSPACE, 'output.md'), 'w')"
                ".write('# Report\\nThe harness is the product.')\n"
                "print('wrote output.md')")
    stream = codeact_loop("explain coding agents", code_for_step)
    print("\n".join(f"{e['type']}: {e['content'][:60]}" for e in stream))
    print("\ntodo.md:\n" + open(os.path.join(WORKSPACE, "todo.md")).read())
    print("\noutput.md:\n" + open(os.path.join(WORKSPACE, "output.md")).read())

Run it: the planner writes a three-item todo.md, the loop executes one Python action per step (gather, analyze, write the report file), checks each item off on disk, and produces output.md. The todo.md and output.md persisting to disk is the whole point: the agent's progress and deliverable live in files, not in a context window that could overflow. (Note the giant caveat in the code: never exec model-written code on the host. Route it through Chapter 7's sandbox.)

← → arrow keys work too