# The agent loop in 40 lines: a model, a sandbox and a stop condition An agent is a loop: call the model, run the tools it asks for, send the results back, and stop on a rule you wrote down first. **Runtime (withruntime.com) runs each tool call in a Firecracker microVM, where a command in a running sandbox answers in 53 ms on Runtime's servers, so forty tool calls add about 2.1 seconds to a whole task.** Frameworks hide the loop, and that is fine until it misbehaves: it runs for an hour, repeats one command forty times, or announces success on a branch where the tests still fail. Every one of those is a stop condition that was missing. This post writes the loop by hand on the Anthropic API, names each way it can end, and tests the stop rules without spending a token. ## What are the parts of an agent loop? Four, and each has one job: | Part | What it holds or does | Where it lives | | ----------- | -------------------------------------------------------- | ------------------ | | Transcript | Every message so far, including tool calls and results | Your process | | Model call | Sends the transcript and the tool list, gets one reply | The model provider | | Tool runner | Runs each requested call and turns the outcome into text | A sandbox | | Stop check | Decides, after every reply, whether to go round again | Your process | The transcript and the stop check are yours. The model never sees the stop check and cannot argue with it, which is the point: a rule the model can talk its way past is a suggestion. The tool runner lives in a sandbox because the model's commands are untrusted input that happens to be executable. ## What does the whole loop look like? One tool, `run`, which takes a bash command. A single shell tool covers reading files, editing them, running tests and installing packages, and current models use it well. The loop below is the whole agent: ```ts check import Anthropic from "@anthropic-ai/sdk"; import { Sandbox } from "withruntime"; const client = new Anthropic(); // reads ANTHROPIC_API_KEY const SYSTEM = "You work in /workspace on a Linux machine. Use the run tool for every command. " + "When the task is done, stop calling tools and say what you changed."; const tools: Anthropic.Tool[] = [ { name: "run", description: "Run one bash command in /workspace. Returns the exit code, stdout and stderr.", input_schema: { type: "object", properties: { command: { type: "string" } }, required: ["command"], additionalProperties: false, }, strict: true, }, ]; const clip = (s: string) => s.length > 8000 ? `${s.slice(0, 3000)}\n[... ${s.length - 6000} chars cut ...]\n${s.slice(-3000)}` : s; export async function agent(task: string, maxTurns = 40, deadline = Date.now() + 30 * 60_000) { await using sbx = await Sandbox.create({ timeoutSeconds: 3600 }); const messages: Anthropic.MessageParam[] = [{ role: "user", content: task }]; for (let turn = 1; turn <= maxTurns; turn++) { if (Date.now() > deadline) return { stop: "deadline", turn }; const reply = await client.messages.create({ model: "claude-opus-5-5", max_tokens: 16_000, system: SYSTEM, tools, messages, }); messages.push({ role: "assistant", content: reply.content }); if (reply.stop_reason !== "tool_use") { const text = reply.content.flatMap((b) => (b.type === "text" ? [b.text] : [])).join("\n"); return { stop: reply.stop_reason ?? "unknown", turn, text }; } const results: Anthropic.ToolResultBlockParam[] = []; for (const call of reply.content) { if (call.type !== "tool_use") continue; const { command } = call.input as { command: string }; const r = await sbx.exec(command, { cwd: "/workspace", timeoutMs: 120_000 }); const head = `exit ${r.exitCode}${r.timedOut ? " (timed out after 120 s)" : ""}`; const content = `${head}\n${clip(r.stdout)}\n${clip(r.stderr)}`; results.push({ type: "tool_result", tool_use_id: call.id, content, is_error: r.exitCode !== 0, }); } messages.push({ role: "user", content: results }); } return { stop: "max_turns", turn: maxTurns }; } ``` ```python check import time import anthropic from withruntime import Sandbox client = anthropic.Anthropic() # reads ANTHROPIC_API_KEY SYSTEM = ("You work in /workspace on a Linux machine. Use the run tool for every command. " "When the task is done, stop calling tools and say what you changed.") TOOLS = [{ "name": "run", "description": "Run one bash command in /workspace. Returns the exit code, stdout and stderr.", "input_schema": {"type": "object", "properties": {"command": {"type": "string"}}, "required": ["command"], "additionalProperties": False}, "strict": True, }] def clip(s: str) -> str: return s if len(s) <= 8000 else f"{s[:3000]}\n[... {len(s) - 6000} chars cut ...]\n{s[-3000:]}" def agent(task: str, max_turns: int = 40, minutes: float = 30) -> dict: deadline = time.monotonic() + minutes * 60 with Sandbox.create(timeout_seconds=3600) as sbx: messages: list[dict] = [{"role": "user", "content": task}] for turn in range(1, max_turns + 1): if time.monotonic() > deadline: return {"stop": "deadline", "turn": turn} reply = client.messages.create(model="claude-opus-5-5", max_tokens=16_000, system=SYSTEM, tools=TOOLS, messages=messages) messages.append({"role": "assistant", "content": reply.content}) if reply.stop_reason != "tool_use": text = "\n".join(b.text for b in reply.content if b.type == "text") return {"stop": reply.stop_reason, "turn": turn, "text": text} results = [] for call in reply.content: if call.type != "tool_use": continue r = sbx.exec(call.input["command"], cwd="/workspace", timeout_ms=120_000) head = f"exit {r.exit_code}" + (" (timed out after 120 s)" if r.timed_out else "") results.append({"type": "tool_result", "tool_use_id": call.id, "is_error": r.exit_code != 0, "content": f"{head}\n{clip(r.stdout)}\n{clip(r.stderr)}"}) messages.append({"role": "user", "content": results}) return {"stop": "max_turns", "turn": max_turns} ``` ## Why does each line exist? Most of the forty lines carry a lesson someone learned the slow way: - **The reply goes back whole.** The loop appends `reply.content`, not its text. Tool calls and reasoning blocks have to be in the history the model sees next, or the next request is refused or the model loses its place. - **Every call gets a result, in one message.** A reply can ask for several commands at once. All of their results go back together, each matched by `tool_use_id`. Splitting them across messages teaches the model to stop asking for more than one at a time. - **Failure is data, not an exception.** A non-zero exit sets `is_error` and still sends the output. The model reads the compiler's complaint and fixes it; an exception in your process would end the run for nothing. - **Output is clipped at both ends.** A test run that prints 40,000 lines would fill the context in two turns. The head shows what started, the tail shows the error, and the count tells the model what it missed. - **Each command has its own limit.** `timeoutMs` ends a command that waits on a prompt or starts a server in the foreground. The model sees "timed out" and tries another way. - **Cleanup is not optional.** `await using` and `with` remove the sandbox however the function exits, including when the model call throws. The model name is the only provider-specific part. The same shape runs on any model with tool calling; the open-model version of this agent is in [build a coding agent on an open model](/blog/open-model-agent-in-a-sandbox). ## What can a model reply mean for the loop? Every reply carries a `stop_reason`, and only one of them means "keep going": | `stop_reason` | What happened | What the loop should do | | --------------- | -------------------------------------- | ----------------------------------------------- | | `tool_use` | The model wants commands run | Run all of them, send every result back | | `end_turn` | The model believes it has finished | Check the work, then stop or go round again | | `max_tokens` | The reply hit `max_tokens` mid-thought | Stop, or retry the turn with a larger limit | | `refusal` | The model declined the request | Stop and report; do not resend the same request | | `pause_turn` | A server-side tool paused a long turn | Send the reply back as it is to let it continue | | `stop_sequence` | A stop sequence you set was produced | Whatever you set it for | The loop above treats everything except `tool_use` as the end, which is the safe default. In particular, a reply cut off by `max_tokens` can end inside a tool call whose input is incomplete, and running half a command is worse than running none. `pause_turn` only appears when you use Anthropic's own server tools, which this loop does not. ## When should your own code stop the loop? The model's opinion is one stop condition among several. The rest are yours, and each one exists for a failure you will eventually see: | Condition | Value here | Ends this failure | | ---------------------------------- | ----------- | ----------------------------------------------- | | Model turns | 40 | A loop that never converges | | Wall-clock deadline | 30 minutes | A user waiting on a run that will not finish | | Same command, same output, 3 times | 3 | A model retrying something that cannot work | | Failed commands in a row | 5 | A broken environment, such as no network | | Verification attempts | 2 | A model that keeps saying "done" when it is not | | One command's time limit | 120 seconds | A command waiting forever on input | | The sandbox's time limit | 1 hour | Your process dying before its cleanup runs | | The account's daily spending limit | Your choice | Every limit above failing at once | The repeat check deserves care. Running `pytest` ten times is normal, because the model edits between runs. Running it ten times with identical output means nothing changed, so the check counts the command and its output together. ## Why is "the model said done" not enough? Because `end_turn` measures the model's confidence, not the state of the repository. A model that has read a failing test, made a plausible edit and run nothing will describe a fix with total calm. The strongest stop condition is a command you choose: the test suite, a build, a linter. If it passes, the loop stops. If it fails, its output goes back to the model as the next user message, and the loop continues, a bounded number of times. This turns an agent from "probably did it" into "did it, and here is the exit code". It also gives you a number worth tracking across runs: the share of tasks that verify on the first `end_turn`. A full version of that check, with the rules a change has to pass before it reaches a branch, is in [a test gate for agent-written code](/blog/test-gate-for-agent-written-code). ## How do you test the loop without paying for tokens? Separate the stop rules from the model API and drive them with a scripted model: a function that replays canned replies. The loop then runs against a real sandbox, deterministically, with no API key. This version adds the repeat check, the failure streak and verification: ```ts import { Sandbox } from "withruntime"; type Call = { type: "tool_use"; id: string; input: { command: string } }; type Reply = { stop_reason: string; content: (Call | { type: "text"; text: string })[] }; type Model = (messages: unknown[]) => Promise; export async function loop(sbx: Sandbox, model: Model, task: string, verify: string) { const limits = { turns: 40, repeats: 3, failuresInARow: 5, verifications: 2 }; const messages: unknown[] = [{ role: "user", content: task }]; const seen = new Map(); let failures = 0; let verifications = 0; for (let turn = 1; turn <= limits.turns; turn++) { const reply = await model(messages); messages.push({ role: "assistant", content: reply.content }); if (reply.stop_reason === "end_turn") { const check = await sbx.exec(verify, { cwd: "/workspace", timeoutMs: 600_000 }); if (check.exitCode === 0) return { stop: "verified", turn }; if (++verifications >= limits.verifications) return { stop: "unverified", turn }; messages.push({ role: "user", content: `\`${verify}\` failed:\n${check.stdout}${check.stderr}`, }); continue; } if (reply.stop_reason !== "tool_use") return { stop: reply.stop_reason, turn }; const results = []; for (const call of reply.content) { if (call.type !== "tool_use") continue; const r = await sbx.exec(call.input.command, { cwd: "/workspace", timeoutMs: 120_000 }); const key = `${call.input.command}\n${r.stdout}\n${r.stderr}`; seen.set(key, (seen.get(key) ?? 0) + 1); if (seen.get(key)! >= limits.repeats) return { stop: "repeating", turn, command: call.input.command }; failures = r.exitCode === 0 ? 0 : failures + 1; const content = `exit ${r.exitCode}\n${r.stdout}${r.stderr}`; results.push({ type: "tool_result", tool_use_id: call.id, content, is_error: r.exitCode !== 0, }); } if (failures >= limits.failuresInARow) return { stop: "failing", turn }; messages.push({ role: "user", content: results }); } return { stop: "max_turns", turn: limits.turns }; } // A scripted model replays canned replies: no API key, no tokens, same result every run. const scripted = (...replies: Reply[]): Model => async () => replies.shift() ?? { stop_reason: "end_turn", content: [] }; const run = (id: string, command: string): Reply => ({ stop_reason: "tool_use", content: [{ type: "tool_use", id, input: { command } }], }); const done: Reply = { stop_reason: "end_turn", content: [{ type: "text", text: "Fixed." }] }; await using sbx = await Sandbox.create({ labels: { test: "agent-loop" } }); const fixes = scripted(run("a", "pytest -q"), run("b", "sed -i 's/<=/ dict: limits = {"turns": 40, "repeats": 3, "failures_in_a_row": 5, "verifications": 2} messages: list[dict] = [{"role": "user", "content": task}] seen: dict[str, int] = {} failures = verifications = 0 for turn in range(1, limits["turns"] + 1): reply = model(messages) messages.append({"role": "assistant", "content": reply["content"]}) if reply["stop_reason"] == "end_turn": check = sbx.exec(verify, cwd="/workspace", timeout_ms=600_000) if check.exit_code == 0: return {"stop": "verified", "turn": turn} verifications += 1 if verifications >= limits["verifications"]: return {"stop": "unverified", "turn": turn} messages.append({"role": "user", "content": f"`{verify}` failed:\n{check.stdout}{check.stderr}"}) continue if reply["stop_reason"] != "tool_use": return {"stop": reply["stop_reason"], "turn": turn} results = [] for call in reply["content"]: if call["type"] != "tool_use": continue command = call["input"]["command"] r = sbx.exec(command, cwd="/workspace", timeout_ms=120_000) key = f"{command}\n{r.stdout}\n{r.stderr}" seen[key] = seen.get(key, 0) + 1 if seen[key] >= limits["repeats"]: return {"stop": "repeating", "turn": turn, "command": command} failures = 0 if r.exit_code == 0 else failures + 1 results.append({"type": "tool_result", "tool_use_id": call["id"], "is_error": r.exit_code != 0, "content": f"exit {r.exit_code}\n{r.stdout}{r.stderr}"}) if failures >= limits["failures_in_a_row"]: return {"stop": "failing", "turn": turn} messages.append({"role": "user", "content": results}) return {"stop": "max_turns", "turn": limits["turns"]} def scripted(*replies: dict): """A scripted model replays canned replies: no API key, no tokens, same result every run.""" queue = list(replies) return lambda messages: queue.pop(0) if queue else {"stop_reason": "end_turn", "content": []} def run(call_id: str, command: str) -> dict: return {"stop_reason": "tool_use", "content": [{"type": "tool_use", "id": call_id, "input": {"command": command}}]} done = {"stop_reason": "end_turn", "content": [{"type": "text", "text": "Fixed."}]} with Sandbox.create(labels={"test": "agent-loop"}) as sbx: fixes = scripted(run("a", "pytest -q"), run("b", "sed -i 's/<=/