feat: 2 blog posts (agent loops, context window), interactive Quiz component

This commit is contained in:
artale 2026-06-12 10:05:12 +02:00
parent cd4d4cd0e2
commit f96f1232ee
62 changed files with 758 additions and 249 deletions

File diff suppressed because one or more lines are too long

File diff suppressed because one or more lines are too long

File diff suppressed because one or more lines are too long

View File

@ -1 +0,0 @@
import{c as a,Q as t,j as o,m as n}from"./chunks/framework.BPKcPtvA.js";const d=JSON.parse('{"title":"Blog","description":"","frontmatter":{},"headers":[],"relativePath":"blog/index.md","filePath":"blog/index.md","lastUpdated":1780488472000}'),s={name:"blog/index.md"};function r(i,e,l,h,u,c){return t(),o("div",null,[...e[0]||(e[0]=[n("",25)])])}const p=a(s,[["render",r]]);export{d as __pageData,p as default};

File diff suppressed because one or more lines are too long

File diff suppressed because one or more lines are too long

File diff suppressed because one or more lines are too long

File diff suppressed because one or more lines are too long

File diff suppressed because one or more lines are too long

File diff suppressed because one or more lines are too long

File diff suppressed because one or more lines are too long

File diff suppressed because one or more lines are too long

File diff suppressed because one or more lines are too long

File diff suppressed because one or more lines are too long

File diff suppressed because one or more lines are too long

File diff suppressed because one or more lines are too long

File diff suppressed because one or more lines are too long

File diff suppressed because one or more lines are too long

File diff suppressed because one or more lines are too long

View File

@ -1 +1 @@
{"404.md":"BiCvjdaY","api-keys.md":"D2Kyj8T3","blog_index.md":"B5nL9faf","blog_posts_cascade-routing.md":"DvBM3TSf","blog_posts_choosing-security-level.md":"BYXRZEDN","blog_posts_mental-models.md":"BRY80gtq","blog_posts_repo-is-spec.md":"BxY1cXc_","blog_posts_security-ladder.md":"DQaqn6Yt","blog_posts_three-x-rule.md":"BHO6bhvz","blog_posts_verifier-pattern.md":"Gha_L_u5","blog_posts_vibe-vs-agentic.md":"7mduPfz1","blog_posts_what-is-an-agent.md":"BU2wUq_Y","blog_posts_why-multi-agent.md":"BVRIN2vH","buy.md":"qEifz7WE","certificate.md":"CqjbFJYQ","checkout.md":"Ccj6L__h","checkout_cancel.md":"DwL5mulX","checkout_success.md":"CYTg6xhL","downloads.md":"CEHJXSp0","free-preview.md":"C5BtRucn","getting-started.md":"Boo_V9xC","index.md":"BNS2TR1g","labs_index.md":"sAzXzfkI","labs_l1-first-agent.md":"BwU9yf-G","labs_l2-context.md":"BexGm8_s","labs_l2-multi-tool.md":"Bj3Mb-oj","labs_l3-verifier.md":"xilrGGap","labs_l3-whitelist-hook.md":"DJtbL66Y","labs_l4-agent-chain.md":"D89P8dwJ","labs_l4-multi-team.md":"DmSK8e2P","labs_l5-cicd.md":"Dz1Jl78z","labs_l5-observability.md":"BDpqVYjT","labs_l6-cost-optimization.md":"CabFK4GB","labs_l6-eval-harness.md":"CBwH6xOR","labs_l7-autoresearch.md":"BYlzPLYo","labs_l7-meta-agent.md":"iTswbOPg","modules_competitive-analysis.md":"BHMHacei","modules_curriculum.md":"D7UeKRfo","modules_debate.md":"DWctKMlA","modules_feynman.md":"DBw5sPBP","modules_field-manual.md":"hmt_NLf1","modules_m1-foundations.md":"DHFyTzWj","modules_m2-architecture.md":"DVowtmf9","modules_m3-safety.md":"RiQQ_HWX","modules_m4-orchestration.md":"DFLcAKBv","modules_m5-production.md":"DTkLIrwQ","modules_m6-economics.md":"HihVEOPb","modules_m7-advanced.md":"B_jVHqsw","modules_m8-capstone.md":"CqV39Gzl","modules_non-technical.md":"BnvuUCRo","modules_reference-stack.md":"D9FXitvn","modules_software-factory.md":"C5Yf8Zwe","modules_tool-reference.md":"B40mlgZJ","public_certificate_template.md":"Cg1kPB1b","resources.md":"DcUu1NrK","skills.md":"BX3RBeCK","troubleshooting.md":"B6difx2I","verify.md":"Cl5ZMWNd"}
{"404.md":"BiCvjdaY","api-keys.md":"D2Kyj8T3","blog_index.md":"ej0SMQFJ","blog_posts_agent-loops-complete-guide.md":"DcCNOI9M","blog_posts_cascade-routing.md":"DvBM3TSf","blog_posts_choosing-security-level.md":"BYXRZEDN","blog_posts_context-window-management.md":"Dr1RUj8G","blog_posts_mental-models.md":"BRY80gtq","blog_posts_repo-is-spec.md":"BxY1cXc_","blog_posts_security-ladder.md":"DQaqn6Yt","blog_posts_three-x-rule.md":"BHO6bhvz","blog_posts_verifier-pattern.md":"Gha_L_u5","blog_posts_vibe-vs-agentic.md":"7mduPfz1","blog_posts_what-is-an-agent.md":"BU2wUq_Y","blog_posts_why-multi-agent.md":"BVRIN2vH","buy.md":"qEifz7WE","certificate.md":"DZ26T6CI","checkout.md":"Ccj6L__h","checkout_cancel.md":"DwL5mulX","checkout_success.md":"CYTg6xhL","downloads.md":"CEHJXSp0","free-preview.md":"C5BtRucn","getting-started.md":"Boo_V9xC","index.md":"BNS2TR1g","labs_index.md":"sAzXzfkI","labs_l1-first-agent.md":"BwU9yf-G","labs_l2-context.md":"BexGm8_s","labs_l2-multi-tool.md":"Bj3Mb-oj","labs_l3-verifier.md":"xilrGGap","labs_l3-whitelist-hook.md":"DJtbL66Y","labs_l4-agent-chain.md":"D89P8dwJ","labs_l4-multi-team.md":"DmSK8e2P","labs_l5-cicd.md":"Dz1Jl78z","labs_l5-observability.md":"BDpqVYjT","labs_l6-cost-optimization.md":"CabFK4GB","labs_l6-eval-harness.md":"CBwH6xOR","labs_l7-autoresearch.md":"BYlzPLYo","labs_l7-meta-agent.md":"iTswbOPg","modules_competitive-analysis.md":"BHMHacei","modules_curriculum.md":"D7UeKRfo","modules_debate.md":"DWctKMlA","modules_feynman.md":"DBw5sPBP","modules_field-manual.md":"hmt_NLf1","modules_m1-foundations.md":"DHFyTzWj","modules_m2-architecture.md":"DVowtmf9","modules_m3-safety.md":"RiQQ_HWX","modules_m4-orchestration.md":"DFLcAKBv","modules_m5-production.md":"DTkLIrwQ","modules_m6-economics.md":"HihVEOPb","modules_m7-advanced.md":"B_jVHqsw","modules_m8-capstone.md":"CqV39Gzl","modules_non-technical.md":"BnvuUCRo","modules_reference-stack.md":"D9FXitvn","modules_software-factory.md":"C5Yf8Zwe","modules_tool-reference.md":"B40mlgZJ","public_certificate_template.md":"Cg1kPB1b","resources.md":"DcUu1NrK","skills.md":"BX3RBeCK","troubleshooting.md":"B6difx2I","verify.md":"Cl5ZMWNd"}

File diff suppressed because one or more lines are too long

File diff suppressed because one or more lines are too long

File diff suppressed because one or more lines are too long

File diff suppressed because one or more lines are too long

File diff suppressed because one or more lines are too long

File diff suppressed because one or more lines are too long

File diff suppressed because one or more lines are too long

File diff suppressed because one or more lines are too long

File diff suppressed because one or more lines are too long

File diff suppressed because one or more lines are too long

File diff suppressed because one or more lines are too long

File diff suppressed because one or more lines are too long

File diff suppressed because one or more lines are too long

File diff suppressed because one or more lines are too long

File diff suppressed because one or more lines are too long

File diff suppressed because one or more lines are too long

File diff suppressed because one or more lines are too long

File diff suppressed because one or more lines are too long

File diff suppressed because one or more lines are too long

File diff suppressed because one or more lines are too long

File diff suppressed because one or more lines are too long

File diff suppressed because one or more lines are too long

File diff suppressed because one or more lines are too long

File diff suppressed because one or more lines are too long

File diff suppressed because one or more lines are too long

File diff suppressed because one or more lines are too long

File diff suppressed because one or more lines are too long

File diff suppressed because one or more lines are too long

File diff suppressed because one or more lines are too long

File diff suppressed because one or more lines are too long

File diff suppressed because one or more lines are too long

File diff suppressed because one or more lines are too long

File diff suppressed because one or more lines are too long

File diff suppressed because one or more lines are too long

File diff suppressed because one or more lines are too long

File diff suppressed because one or more lines are too long

File diff suppressed because one or more lines are too long

View File

@ -0,0 +1,218 @@
<template>
<div class="quiz-block">
<div class="quiz-header">
<span class="quiz-badge">Quiz</span>
<span class="quiz-count">{{ current + 1 }} / {{ questions.length }}</span>
</div>
<div v-if="!finished" class="quiz-body">
<h4 class="quiz-q">{{ questions[current].q }}</h4>
<div class="quiz-options">
<button
v-for="(opt, i) in questions[current].opts"
:key="i"
class="quiz-opt"
:class="{
correct: answered && i === questions[current].ans,
wrong: answered && selected === i && i !== questions[current].ans,
disabled: answered
}"
:disabled="answered"
@click="answer(i)"
>
<span class="opt-letter">{{ ['A','B','C','D'][i] }}</span>
<span class="opt-text">{{ opt }}</span>
<span v-if="answered && i === questions[current].ans" class="opt-icon material-symbols-outlined">check_circle</span>
<span v-if="answered && selected === i && i !== questions[current].ans" class="opt-icon material-symbols-outlined">cancel</span>
</button>
</div>
<div v-if="answered" class="quiz-feedback" :class="{ correct: selected === questions[current].ans }">
<p>{{ selected === questions[current].ans ? 'Correct!' : questions[current].exp || 'Not quite.' }}</p>
<button class="quiz-next" @click="next">
{{ current < questions.length - 1 ? 'Next Question' : 'See Results' }}
</button>
</div>
</div>
<div v-else class="quiz-results">
<div class="result-score">{{ score }} / {{ questions.length }}</div>
<div class="result-pct">{{ Math.round(score / questions.length * 100) }}%</div>
<div class="result-msg">
<template v-if="score === questions.length">Perfect score. You've mastered this module.</template>
<template v-else-if="score >= questions.length * 0.7">Good work. Review the missed questions.</template>
<template v-else>Review the module material and try again.</template>
</div>
<button class="quiz-retry" @click="reset">Retry Quiz</button>
</div>
</div>
</template>
<script setup>
import { ref, computed } from 'vue'
const props = defineProps({
questions: {
type: Array,
required: true,
}
})
const current = ref(0)
const selected = ref(-1)
const answered = ref(false)
const score = ref(0)
const finished = ref(false)
function answer(i) {
if (answered.value) return
selected.value = i
answered.value = true
if (i === props.questions[current.value].ans) {
score.value++
}
}
function next() {
if (current.value < props.questions.length - 1) {
current.value++
selected.value = -1
answered.value = false
} else {
finished.value = true
}
}
function reset() {
current.value = 0
selected.value = -1
answered.value = false
score.value = 0
finished.value = false
}
</script>
<style scoped>
.quiz-block {
background: var(--vp-c-bg-soft);
border: 1px solid var(--vp-c-border);
border-radius: 12px;
overflow: hidden;
margin: 24px 0;
}
.quiz-header {
display: flex;
justify-content: space-between;
align-items: center;
padding: 12px 20px;
background: var(--vp-c-bg-mute);
border-bottom: 1px solid var(--vp-c-border);
}
.quiz-badge {
font-size: 10px;
font-weight: 700;
letter-spacing: 0.15em;
text-transform: uppercase;
color: var(--vp-c-brand-1);
}
.quiz-count {
font-size: 12px;
color: var(--vp-c-text-3);
}
.quiz-body { padding: 20px; }
.quiz-q {
font-size: 15px;
font-weight: 600;
margin-bottom: 16px;
line-height: 1.5;
}
.quiz-options { display: flex; flex-direction: column; gap: 8px; }
.quiz-opt {
display: flex;
align-items: center;
gap: 12px;
padding: 12px 16px;
background: var(--vp-c-bg);
border: 1px solid var(--vp-c-border);
border-radius: 8px;
cursor: pointer;
text-align: left;
font-size: 14px;
color: var(--vp-c-text-1);
transition: all 0.15s;
}
.quiz-opt:hover:not(.disabled) {
border-color: var(--vp-c-brand-1);
background: var(--vp-c-brand-soft);
}
.quiz-opt.disabled { cursor: default; }
.quiz-opt.correct {
border-color: #22c55e;
background: rgba(34,197,94,0.08);
}
.quiz-opt.wrong {
border-color: #ef4444;
background: rgba(239,68,68,0.08);
}
.opt-letter {
width: 24px;
height: 24px;
border-radius: 6px;
background: var(--vp-c-bg-mute);
display: flex;
align-items: center;
justify-content: center;
font-size: 12px;
font-weight: 700;
flex-shrink: 0;
}
.quiz-opt.correct .opt-letter { background: #22c55e; color: #fff; }
.quiz-opt.wrong .opt-letter { background: #ef4444; color: #fff; }
.opt-text { flex: 1; }
.opt-icon { font-size: 20px; color: #22c55e; }
.quiz-opt.wrong .opt-icon { color: #ef4444; }
.quiz-feedback {
margin-top: 16px;
padding: 16px;
background: rgba(239,68,68,0.06);
border: 1px solid rgba(239,68,68,0.15);
border-radius: 8px;
}
.quiz-feedback.correct {
background: rgba(34,197,94,0.06);
border-color: rgba(34,197,94,0.15);
}
.quiz-feedback p { font-size: 13px; color: var(--vp-c-text-2); margin: 0 0 12px; }
.quiz-next, .quiz-retry {
padding: 8px 20px;
background: var(--vp-c-brand-1);
color: #fff;
border: none;
border-radius: 6px;
font-size: 13px;
font-weight: 600;
cursor: pointer;
}
.quiz-next:hover, .quiz-retry:hover { background: var(--vp-c-brand-3); }
.quiz-results {
text-align: center;
padding: 40px 20px;
}
.result-score {
font-size: 48px;
font-weight: 800;
color: var(--vp-c-brand-1);
line-height: 1;
}
.result-pct {
font-size: 16px;
color: var(--vp-c-text-3);
margin-bottom: 12px;
}
.result-msg {
font-size: 14px;
color: var(--vp-c-text-2);
margin-bottom: 20px;
}
</style>

View File

@ -1,6 +1,7 @@
import DefaultTheme from 'vitepress/theme'
import Layout from './Layout.vue'
import LandingLayout from './LandingLayout.vue'
import Quiz from './Quiz.vue'
import './style.css'
export default {
@ -8,5 +9,6 @@ export default {
Layout,
enhanceApp({ app }) {
app.component('LandingLayout', LandingLayout)
app.component('Quiz', Quiz)
},
}

View File

@ -4,36 +4,40 @@ Essays on agentic engineering, security, multi-agent systems, and production dep
## Latest Posts
### [What Is an AI Agent, Really?](/blog/posts/what-is-an-agent) — June 2
LLM + Tools + Loop. The simplest correct explanation.
### [Context Window Management for AI Agents](/blog/posts/context-window-management) — June 18
Sliding windows, summarization, mental models, and the 80/20 rule of context budget allocation.
### [Why One Agent Is Not Enough](/blog/posts/why-multi-agent) — June 3
Context, capability, and reliability ceilings of single-agent systems.
### [Vibe Coding vs Agentic Engineering](/blog/posts/vibe-vs-agentic) — June 4
The 5 hard rules that separate production from prompt gambling.
### [The Repo Is the Spec](/blog/posts/repo-is-spec) — June 5
Why every instruction your agent needs must live in a file.
### [The 6-Level Security Ladder](/blog/posts/security-ladder) — June 6
How to stop your AI agents from destroying production. From ACIP to no-bash.
### [Choosing Your Security Level](/blog/posts/choosing-security-level) — June 7
Which L-level you need based on what your agent can access.
### [The Verifier Pattern](/blog/posts/verifier-pattern) — June 8
A read-only verification agent that catches mistakes before production.
### [Agent Memory: Mental Models](/blog/posts/mental-models) — June 9
How agents remember across sessions using self-maintained expertise files.
### [The 3x Rule of Agent Costs](/blog/posts/three-x-rule) — June 10
Why production agents cost 3x your prototype estimate.
### [Agent Loops: The Complete Guide](/blog/posts/agent-loops-complete-guide) — June 15
Three loop types, termination conditions, anti-patterns, and the 5 rules of production loops.
### [Cascade Routing: Cut API Costs by 66%](/blog/posts/cascade-routing) — June 11
Use cheap models for simple steps, expensive models for complex reasoning.
---
### [The 3x Rule of Agent Costs](/blog/posts/three-x-rule) — June 10
Why production agents cost 3x your prototype estimate.
### [Agent Memory: Mental Models](/blog/posts/mental-models) — June 9
How agents remember across sessions using self-maintained expertise files.
### [The Verifier Pattern](/blog/posts/verifier-pattern) — June 8
A read-only verification agent that catches mistakes before production.
### [Choosing Your Security Level](/blog/posts/choosing-security-level) — June 7
Which L-level you need based on what your agent can access.
### [The 6-Level Security Ladder](/blog/posts/security-ladder) — June 6
How to stop your AI agents from destroying production. From ACIP to no-bash.
### [The Repo Is the Spec](/blog/posts/repo-is-spec) — June 5
Why every instruction your agent needs must live in a file.
### [Vibe Coding vs Agentic Engineering](/blog/posts/vibe-vs-agentic) — June 4
The 5 hard rules that separate production from prompt gambling.
### [Why One Agent Is Not Enough](/blog/posts/why-multi-agent) — June 3
Context, capability, and reliability ceilings of single-agent systems.
### [What Is an AI Agent, Really?](/blog/posts/what-is-an-agent) — June 2
LLM + Tools + Loop. The simplest correct explanation.
*Posts are based on content from the [Agentic Engineering Course](/). Each topic has a corresponding module with labs and exercises.*

View File

@ -0,0 +1,146 @@
# Agent Loops: The Complete Guide
**June 15, 2026**
Every agent is a loop. The difference between a demo agent and a production agent is how well you control that loop.
---
## The Three Loop Types
### Type 1: Think → Act → Observe (Basic)
```
1. LLM decides what to do next (thinks)
2. Tool executes the decision (acts)
3. Result feeds back to LLM (observes)
4. Repeat until done
```
This is the simplest loop. Every lab in this course starts here. It works for single-step tasks where the agent calls one tool and returns an answer.
**Problem**: No bounded iteration. Without `MAX_ITERATIONS`, the agent loops forever on ambiguous tasks.
### Type 2: Plan → Execute → Verify (Guarded)
```
1. Agent plans the approach (tool selection + sequence)
2. Agent executes each step
3. Verifier agent checks each result
4. On failure: re-plan with new context
5. On success: proceed or terminate
```
The verifier is a second, simpler agent (or the same agent with a verification prompt) that checks output quality before the loop continues. This prevents the agent from confidently proceeding with wrong results.
### Type 3: Cascade (Multi-Model)
```
Step 1: Haiku (cheap) — bulk processing
→ Step 2: Sonnet (mid) — analysis
→ Step 3: Opus (premium) — synthesis, quality check
```
Each step uses a different model tier. Early steps are cheap and fast. Later steps are expensive but thorough. The cascade loop saves 60-80% on token costs compared to running everything through Opus.
---
## Termination Conditions
Every loop needs at least one termination condition. Production loops need three:
### 1. Content-Based Termination
The agent decides it's done:
```python
if response.stop_reason == "end_turn":
return response.text # Normal completion
elif response.stop_reason == "tool_use":
continue_loop() # Agent wants another turn
```
### 2. Hard Limit Termination
The loop has a maximum iteration count:
```python
MAX_ITERATIONS = 10
for i in range(MAX_ITERATIONS):
result = agent_step()
if result.is_done:
return result
return {"error": "Max iterations exceeded", "partial_result": result}
```
This is non-negotiable in production. Every lab includes it. Without it, a single bad prompt can cost you $50+ in runaway token usage.
### 3. Cost Budget Termination
The loop tracks cumulative cost and stops when the budget is spent:
```python
BUDGET_CENTS = 50
total_cost = 0
for i in range(MAX_ITERATIONS):
result = agent_step()
total_cost += result.cost_cents
if total_cost > BUDGET_CENTS:
return {"error": "Budget exceeded", "total_cost": total_cost}
if result.is_done:
return result
```
---
## Loop Anti-Patterns
### Grinding
The agent runs the same code repeatedly hoping for a different result:
```python
# BAD: no change between iterations
for i in range(100):
score = evaluate(agent_config)
if score > best_score:
best_score = score # same config, different random seed
```
**Fix**: Hash the agent configuration. If it hasn't changed, don't re-run.
### Hallucination Cascade
Each loop iteration builds on potentially wrong information from the previous step. By iteration 5, the agent's context is full of hallucinated facts, and it makes reasonable-looking decisions based on nonsense.
**Fix**: A verifier step after every tool call checks factual claims before they enter the context window.
### Infinite Loop by Design
Some tasks naturally loop (monitoring, polling). Without careful budgeting, these can run forever:
```python
# BAD: no cost tracking on long-running loops
while True:
data = check_api()
if data.alerts:
send_notification(data.alerts)
time.sleep(60)
```
**Fix**: Daily cost budget + max iterations even in "infinite" loops.
---
## The 5 Rules of Production Loops
1. **Always set MAX_ITERATIONS** — even in loops you expect to terminate naturally
2. **Always track cost per iteration** — you can't optimize what you don't measure
3. **Always verify intermediate results** — don't let bad context compound
4. **Always have a fallback** — what happens when max iterations is reached? Return partial results, don't crash
5. **Always log the loop** — every iteration should be recorded for debugging
---
*This is adapted from Module 1: Foundations of the [Agentic Engineering the Hard Way](/free-preview) course. Full course includes 65 lessons, 13 labs, and 20 skill kits.*

View File

@ -0,0 +1,141 @@
# Context Window Management for AI Agents
**June 18, 2026**
Your agent's context window is its working memory. Fill it with the wrong things, and the agent makes bad decisions. Fill it with too much, and you're burning $0.15 per call on irrelevant tokens.
---
## The Context Budget
Every token in the context window has a cost — literally (API pricing) and figuratively (attention dilution). The key insight: **not all tokens are equal**.
```
High-Value Tokens (always include):
├── System prompt (persona, rules, constraints)
├── Tool definitions (name, description, schema)
├── Current user request
└── Most recent tool results
Medium-Value Tokens (include if relevant):
├── Conversation history (last 3-5 exchanges)
├── Mental model files (agent-specific expertise)
└── Reference documents (API docs, style guides)
Low-Value Tokens (exclude in production):
├── Entire conversation history (use summaries instead)
├── Large reference files (link instead of inline)
├── Previous tool results that are no longer relevant
└── System messages older than the last 10 turns
```
---
## Sliding Window Strategy
The most common production approach. Keep the window focused on recent + important:
```python
class SlidingWindow:
def __init__(self, max_tokens=32000):
self.max_tokens = max_tokens
self.system_prompt = "" # always retained
self.messages = [] # conversation history
self.token_count = 0
def add_message(self, msg):
self.messages.append(msg)
self.token_count += count_tokens(msg)
# Trim to fit budget — remove oldest tool results first
while self.token_count > self.max_tokens and len(self.messages) > 3:
removed = self.messages.pop(1) # keep system + last user
self.token_count -= count_tokens(removed)
```
**Best for**: Chat-style agents, interactive coding assistants, support bots.
**Limitation**: Loses context from early in the conversation. If the user mentions something important 20 turns ago, the agent won't remember it.
---
## Summarization Strategy
Periodically summarize the conversation into condensed context:
```python
def summarize_context(messages):
prompt = f"Summarize this conversation in 3-5 sentences, \
preserving key decisions and user preferences: \
{messages[-20:]}"
summary = llm.call(prompt)
return {"role": "system", "content": f"[CONTEXT: {summary}]"}
```
**Pattern**:
```
Turns 1-10: full messages
Turn 11: summarize turns 1-10 → insert as system message
Turns 11-20: full messages (with summary in system prompt)
Turn 21: summarize turns 11-20 → update system summary
... repeat
```
**Best for**: Long-running research agents, complex multi-step tasks, customer support.
**Cost**: Each summarization costs ~50-100 tokens. That's $0.0003-0.0015 per summary — essentially free.
---
## Mental Model Strategy (Advanced)
Give the agent its own persistent memory that it reads and writes:
```python
# Agent reads this file at the start of every session
MENTAL_MODEL = "agent-expertise.md"
def load_mental_model():
if os.path.exists(MENTAL_MODEL):
return open(MENTAL_MODEL).read()
return ""
def save_observation(key, value):
with open(MENTAL_MODEL, "a") as f:
f.write(f"\n- {key}: {value}")
```
The agent builds expertise over time. This is the most token-efficient strategy because the agent curates what it remembers — it doesn't keep everything.
**Best for**: Specialized agents (code reviewers, security auditors, data analysts) that work on multiple disjoint tasks.
---
## Token Budget Allocation Formula
```
Budget = SystemPrompt(15%) + Tools(15%) + Conversation(40%) + ToolResults(30%)
If budget is tight:
1. Shorten tool descriptions (remove examples)
2. Summarize conversation history (keep last 3-5 full, summarize the rest)
3. Trim tool results to only the relevant sections
4. Remind the agent about mental models instead of re-reading them
```
---
## The 80/20 Rule of Context Management
80% of context problems come from 20% of the causes:
1. **Too much conversation history** — keep last 5 exchanges, summarize the rest
2. **Too many tool results** — only keep results that were directly used
3. **Redundant system instructions** — don't repeat the same rules in every system message
4. **Large reference files** — cite them, don't include them inline
Fix these four, and you solve most context window issues.
---
*This is adapted from Module 2: Architecture of the [Agentic Engineering the Hard Way](/free-preview) course. Full course includes 65 lessons, 13 labs, and 20 skill kits.*