Part F: Optimization
All four measurements in this part come from one script. Parts 1 and 3 are pure token arithmetic and fully real. Parts 2 and 4 drive a ScriptedLLM to exercise the loop and the harness; their token counts and call counts are real, their "decisions" and pass rates are scripted.
examples/m07_optimize.py
"""Module 7: context optimization, measured.
1. cache-aware ordering: shared-prefix tokens across consecutive tickets
2. just-in-time loading vs preloading: tokens and number of calls
3. compaction triggers over a long conversation
4. an A/B harness: did the added context improve the output?
Parts 2 and 4 drive a ScriptedLLM (rules, not a model) to exercise the plumbing;
pass --real to run part 4 with supportdesk.llm.chat and a real provider.
Run: PYTHONPATH=. python examples/m07_optimize.py [--real]
"""
from __future__ import annotations
import json
import math
import random
import re
import sys
from collections.abc import Callable
from pathlib import Path
from typing import Any
sys.path.insert(0, str(Path(__file__).resolve().parent)) # this module's own example files
from m07_context import (FEW_SHOT, SYSTEM, TOOLS, ContextBuilder, extractive_summary, # noqa: E402
tool_tokens)
from supportdesk.data import Ticket, get_article, load_tickets # noqa: E402
from supportdesk.kb_search import KBSearch # noqa: E402
from supportdesk.llm import ChatResult, ToolCall, Usage # noqa: E402
from supportdesk.pricing import PRICES, cost_usd # noqa: E402
from supportdesk.stand_in import ScriptedLLM # noqa: E402
from supportdesk.tokens import count_messages, count_tokens, encode # noqa: E402
KB = KBSearch()
# --- 1. cache-aware ordering ---------------------------------------------------------------------
def serialize(messages: list[dict[str, Any]], tools: list[dict[str, Any]], tools_first: bool) -> str:
"""Roughly how a provider flattens a request into one prompt. Tool placement differs by provider."""
body = "".join(f"<|{m['role']}|>{m['content']}<|end|>" for m in messages)
tool_text = f"<|tools|>{json.dumps(tools, separators=(',', ':'))}<|end|>" if tools else ""
return tool_text + body if tools_first else body.replace("<|end|>", "<|end|>" + tool_text, 1)
def common_prefix(a: list[int], b: list[int]) -> int:
n = 0
for x, y in zip(a, b):
if x != y:
break
n += 1
return n
def layout(ticket: Ticket, order: str, per_category_tools: bool) -> tuple[list[dict], list[dict]]:
"""Build the same content in two orders: 'stable_first' (cache-aware) or 'volatile_first'."""
category = ticket.gold["category"]
b = (ContextBuilder().system().examples()
.with_tools(category=category if per_category_tools else None)
.profile({"customer_tier": ticket.customer_tier, "language": ticket.language})
.knowledge(KB.search(ticket.text, k=3))
.state({"ticket": ticket.id, "step": "draft_reply"}).ticket(ticket))
ctx = b.build()
if order == "stable_first":
return ctx.messages, ctx.tools
volatile = ctx.messages[-1]["content"]
return [{"role": "system", "content": volatile}, {"role": "user", "content": ctx.stable_prefix}], ctx.tools
def cache_report(tickets: list[Ticket]) -> None:
print("1. Shared prefix between each ticket and the one before it (dev split, arrival order)")
print(f" {'layout':52} {'prompt':>6} {'shared':>6} {'share':>6} {'billed':>6}")
for order, per_cat, tools_first in [("volatile_first", False, False), ("stable_first", False, False),
("stable_first", True, False), ("stable_first", True, True),
("stable_first", False, True)]:
ids = [encode(serialize(*layout(t, order, per_cat), tools_first)) for t in tickets]
shared = [common_prefix(a, b) for a, b in zip(ids, ids[1:])]
mean_len = sum(map(len, ids[1:])) / len(shared)
label = f"{order}, {'category' if per_cat else 'all'} tools, {'tools first' if tools_first else 'tools after system'}"
mean_shared = sum(shared) / len(shared)
billed = mean_len - 0.5 * mean_shared # input tokens billed at full price, 50% cache discount
print(f" {label:52} {mean_len:>6.0f} {mean_shared:>6.0f} {mean_shared / mean_len:>6.0%} {billed:>6.0f}")
# --- 2. just-in-time loading vs preloading -------------------------------------------------------
SEARCH_TOOL = [t for t in TOOLS if t["function"]["name"] == "search_kb"]
def jit_responder(messages: list[dict[str, Any]], kwargs: dict[str, Any]) -> ChatResult:
"""Scripted: first turn asks search_kb (unless the ticket needs no lookup), second turn answers."""
if messages[-1]["role"] == "tool":
return ChatResult(text='{"reply": "Draft based on the article.", "cited_articles": [], "confidence": "medium"}')
ticket = messages[-1]["content"]
if "feature=True" in ticket: # stands in for "the model decided no lookup is needed"
return ChatResult(text='{"reply": "Thanks for the idea.", "cited_articles": [], "confidence": "low"}')
subject = re.search(r"Subject: (.*)", ticket).group(1)
return ChatResult(text="", tool_calls=[ToolCall("call_1", "search_kb", {"query": subject, "k": 1})])
def run_jit(ticket: Ticket, llm: ScriptedLLM) -> tuple[int, int]:
"""Returns (calls, input tokens including tool definitions) for the just-in-time loop."""
messages = [{"role": "system", "content": SYSTEM},
{"role": "user", "content": f"<ticket>{ticket.text}</ticket>\nfeature={ticket.gold['category'] == 'feature_request'}"}]
calls = tokens = 0
for _ in range(3):
result = llm(messages, tools=SEARCH_TOOL)
calls += 1
tokens += result.usage.input_tokens + tool_tokens(SEARCH_TOOL)
if not result.tool_calls:
break
messages.append(result.as_message())
for call in result.tool_calls:
hits = KB.search(call.arguments["query"], k=call.arguments.get("k", 1))
content = "\n".join(get_article(h.article_id).body for h in hits)
messages.append({"role": "tool", "tool_call_id": call.id, "content": content})
return calls, tokens
def jit_report(tickets: list[Ticket]) -> None:
print("\n2. Preload top-3 articles vs just-in-time search_kb (dev split, input tokens per ticket)")
llm = ScriptedLLM(responder=jit_responder) # plumbing only: it decides by rule when to search
rows = {"preload k=3": [0, 0], "preload k=1": [0, 0], "just-in-time": [0, 0]}
for t in tickets:
for k in (3, 1):
knowledge = "\n".join(get_article(h.article_id).body for h in KB.search(t.text, k=k))
msgs = [{"role": "system", "content": SYSTEM},
{"role": "user", "content": f"<knowledge>{knowledge}</knowledge>\n<ticket>{t.text}</ticket>"}]
rows[f"preload k={k}"][0] += 1
rows[f"preload k={k}"][1] += count_messages(msgs)
calls, tokens = run_jit(t, llm)
rows["just-in-time"][0] += calls
rows["just-in-time"][1] += tokens
print(f" {'strategy':14} {'calls':>6} {'input tokens':>13} {'per ticket':>11}")
for name, (calls, tokens) in rows.items():
print(f" {name:14} {calls:>6} {tokens:>13} {tokens / len(tickets):>11.0f}")
# --- 3. compaction triggers ----------------------------------------------------------------------
def long_conversation(turns: int, seed: int = 0) -> list[dict[str, str]]:
"""A synthetic multi-issue chat built from real tickets and article sentences."""
rng = random.Random(seed)
pool = load_tickets("dev")
convo = []
for i in range(turns):
t = rng.choice(pool)
convo.append({"role": "user", "content": t.body})
art = t.gold["kb_article"] or "feature-requests"
convo.append({"role": "assistant", "content": get_article(art).body.split("\n")[0]})
return convo
def compaction_report() -> None:
print("\n3. Compaction over a 30-exchange conversation (history tokens sent per turn)")
convo = long_conversation(30)
budget, keep = 1200, 6
for name, trigger in [("never", lambda tok, n: False),
("every 10 exchanges", lambda tok, n: n % 20 == 0),
("at 70% of 1200", lambda tok, n: tok > 0.7 * budget),
("at 95% of 1200", lambda tok, n: tok > 0.95 * budget)]:
history: list[dict[str, str]] = []
sent, peak, compactions, rewrites = 0, 0, 0, 0
for n, message in enumerate(convo):
history.append(message)
tokens = sum(count_tokens(m["content"]) for m in history)
if trigger(tokens, n + 1) and len(history) > keep:
summary = extractive_summary(history[:-keep])
history = [{"role": "user", "content": summary}] + history[-keep:]
compactions += 1
rewrites += 1 # the prefix changed, so the next call misses the prompt cache
tokens = sum(count_tokens(m["content"]) for m in history)
if message["role"] == "user": # a model call happens after each customer message
sent += tokens
peak = max(peak, tokens)
print(f" {name:18} peak {peak:>5} total sent {sent:>6} compactions {compactions:>2} cache resets {rewrites:>2}")
# --- 4. A/B harness --------------------------------------------------------------------------------
def extractive_answerer(messages: list[dict[str, Any]], kwargs: dict[str, Any]) -> str:
"""Scripted, not a model: cites the article whose text shares the most words with the ticket."""
user = messages[-1]["content"]
ticket = set(re.findall(r"[a-z]+", user.split("<ticket")[-1].lower()))
best, best_overlap = None, 0
for aid, body in re.findall(r'<article id="([^"]+)"[^>]*>\n(.*?)\n</article>', user, re.S):
overlap = len(ticket & set(re.findall(r"[a-z]+", body.lower())))
if overlap > best_overlap:
best, best_overlap = aid, overlap
return json.dumps({"reply": "...", "cited_articles": [best] if best else [], "confidence": "medium"})
def mcnemar_exact(b: int, c: int) -> float:
"""Two-sided exact p-value for b wins vs c losses among discordant pairs."""
n = b + c
if n == 0:
return 1.0
tail = sum(math.comb(n, i) for i in range(0, min(b, c) + 1)) / 2 ** n
return min(1.0, 2 * tail)
def ab_test(variant_a: Callable[[Ticket], Any], variant_b: Callable[[Ticket], Any], tickets: list[Ticket],
call: Callable[[list[dict]], str], judge: Callable[[Ticket, str], bool], model: str) -> None:
"""Run both context variants on every ticket; report paired wins, losses, p-value, tokens, cost."""
passes = {"A": [], "B": []}
tokens = {"A": 0, "B": 0}
for t in tickets:
for name, variant in (("A", variant_a), ("B", variant_b)):
ctx = variant(t)
passes[name].append(judge(t, call(ctx.messages)))
tokens[name] += ctx.total_tokens
wins = sum(b and not a for a, b in zip(passes["A"], passes["B"]))
losses = sum(a and not b for a, b in zip(passes["A"], passes["B"]))
n = len(tickets)
for name in ("A", "B"):
usage = Usage(input_tokens=tokens[name] // n)
print(f" variant {name}: pass {sum(passes[name])}/{n} mean prompt tokens {tokens[name] / n:.0f}"
f" input cost per 1,000 tickets on {model}: {cost_usd(usage, model) * 1000:.3f} USD")
print(f" B vs A: {wins} wins, {losses} losses, {n - wins - losses} ties;"
f" exact McNemar p = {mcnemar_exact(wins, losses):.3f}")
def draft_judge(ticket: Ticket, text: str) -> bool:
"""Pass if the draft parses and cites the gold article (unanswerable: cites nothing or says low)."""
try:
draft = json.loads(re.search(r"\{.*\}", text, re.S).group(0))
except (AttributeError, json.JSONDecodeError):
return False
gold = ticket.gold["kb_article"]
return gold in draft.get("cited_articles", []) if gold else draft.get("confidence") == "low"
def variant(k: int) -> Callable[[Ticket], Any]:
def build(t: Ticket):
return (ContextBuilder().system().examples().with_tools(category=t.gold["category"])
.profile({"customer_tier": t.customer_tier, "language": t.language})
.knowledge(KB.search(t.text, k=k)).ticket(t).build())
return build
if __name__ == "__main__":
dev = load_tickets("dev")
cache_report(dev)
jit_report(dev)
compaction_report()
answerable = [t for t in dev if t.gold["answerable"]]
model = "gemini-3.5-flash"
if "--real" in sys.argv:
from supportdesk.llm import chat, resolve
model = resolve()[1]
print(f"\n4. A/B on {len(answerable)} answerable dev tickets with a real model ({model})")
ab_test(variant(1), variant(3), answerable, lambda m: chat(m, temperature=0.0).text, draft_judge,
model if model in PRICES else "openai/gpt-oss-120b")
else:
print(f"\n4. A/B harness, knowledge k=1 (A) vs k=3 (B), {len(answerable)} answerable dev tickets,"
" ScriptedLLM extractive answerer (plumbing, not model quality)")
llm = ScriptedLLM(responder=extractive_answerer)
ab_test(variant(1), variant(3), answerable, lambda m: llm(m).text, draft_judge, model)
Code explained
- In simple words: four experiments that put numbers on four context design choices.
- What happens:
serialize(messages, tools, tools_first): flattens a request into one string roughly the way a provider's chat template does, with tools either before everything or right after the system message. Providers differ on this, so the function models both.common_prefix(a, b): the number of leading tokens two prompts share. Prompt caching (Module 2) reuses computation for exactly this shared prefix.layout(ticket, order, per_category_tools): builds the same content in cache-aware order (stable_first) or reversed (volatile_first), with all tools or per-category tools.cache_report: encodes every dev ticket's prompt in arrival order and averages the prefix it shares with the previous ticket's prompt. Thebilledcolumn applies a 50 percent discount to the shared tokens, Groq's documented cached-input discount.jit_responderandrun_jit: a scripted model that first callssearch_kbwith the ticket subject (unless the ticket is a feature request, standing in for "the model decided no lookup is needed") and answers on the second call.run_jitexecutes the tool withKBSearch, returns the article as atoolmessage (Module 6's loop), and counts calls and input tokens, including the tool definition.jit_report: compares that loop with preloading 1 or 3 articles into a single call.long_conversationandcompaction_report: build a 30-exchange conversation from real tickets and article sentences, then replay it under four compaction triggers, counting the peak history size, the total history tokens sent over the conversation, and how many times compaction rewrote the prefix (each rewrite is a prompt-cache miss).extractive_answerer,draft_judge,mcnemar_exact,ab_test,variant(k): the A/B harness. Each variant is a function from ticket toContext.ab_testruns both on every ticket, judges each draft, and reports passes, paired wins and losses, an exact McNemar p-value, mean prompt tokens, and input cost frompricing.py. With--real,callissupportdesk.llm.chat.
- Comes out: (token counts are deterministic; the run takes about 15 seconds)
1. Shared prefix between each ticket and the one before it (dev split, arrival order)
layout prompt shared share billed
volatile_first, all tools, tools after system 1379 18 1% 1370
stable_first, all tools, tools after system 1379 966 70% 896
stable_first, category tools, tools after system 1165 664 57% 833
stable_first, category tools, tools first 1165 291 25% 1020
stable_first, all tools, tools first 1379 966 70% 896
2. Preload top-3 articles vs just-in-time search_kb (dev split, input tokens per ticket)
strategy calls input tokens per ticket
preload k=3 48 24032 501
preload k=1 48 15066 314
just-in-time 92 30587 637
3. Compaction over a 30-exchange conversation (history tokens sent per turn)
never peak 1480 total sent 22146 compactions 0 cache resets 0
every 10 exchanges peak 994 total sent 16346 compactions 3 cache resets 3
at 70% of 1200 peak 829 total sent 14265 compactions 2 cache resets 2
at 95% of 1200 peak 1108 total sent 19170 compactions 2 cache resets 2
4. A/B harness, knowledge k=1 (A) vs k=3 (B), 42 answerable dev tickets, ScriptedLLM extractive answerer (plumbing, not model quality)
variant A: pass 35/42 mean prompt tokens 916 input cost per 1,000 tickets on gemini-3.5-flash: 1.373 USD
variant B: pass 30/42 mean prompt tokens 1137 input cost per 1,000 tickets on gemini-3.5-flash: 1.706 USD
B vs A: 0 wins, 5 losses, 37 ties; exact McNemar p = 0.062
The next four topics read this output section by section.
F.1 Cache-aware ordering: stable prefix, volatile suffix
Prompt caching reuses the provider's work on a prompt prefix it has seen recently, and bills those tokens at a discount. It only works for an exact token-for-token prefix match. So the rule is: everything identical across requests goes first, everything that varies goes last.
Section 1 of the output measures it on the 48 dev tickets:
- Volatile first (ticket and articles before the system instruction): consecutive prompts share 18 tokens, 1 percent. Caching is useless.
- Stable first, all tools: they share 966 of 1,379 tokens, 70 percent. That is the system instruction, the examples, and the tool list.
- Stable first, per-category tools after the system message: 664 of 1,165 shared, 57 percent. The prefix is shorter because the tool list changes from ticket to ticket, but the prompt is shorter too.
- Per-category tools rendered first: only 291 shared, 25 percent. If your provider puts tools ahead of the system prompt, varying the tool list destroys the cache for everything after it.
Which design is cheaper depends on the discount. With Groq's 50 percent, the billed column says per-category tools still win (833 vs 896 full-price-equivalent tokens). With a 90 percent discount (the ratio in pricing.py for gemini-3.5-flash: 0.15 vs 1.50 USD per million), the arithmetic flips: all tools come to 1,379 - 0.9 x 966 = about 510, per-category tools to 1,165 - 0.9 x 664 = about 567.
Two provider facts decide whether any of this applies, and both need checking against current docs:
- Minimum cacheable length. Groq's prompt caching docs (checked 21 September 2026) list
openai/gpt-oss-20bandopenai/gpt-oss-120bas supported, with a minimum prefix of 128 to 1,024 tokens depending on the model, a 50 percent discount on cached input, and expiry after 2 hours without use. Google's Gemini docs say implicit caching is on by default for Gemini 2.5 and newer, with a minimum of 4,096 tokens for Gemini 3.5 Flash. Our whole prompt is about 1,400 tokens, so on Gemini 3.5 Flash nothing would be cached. On Groq, a 966-token prefix may or may not qualify. - How hits are reported.
supportdesk.llm.chatalready readsusage.prompt_tokens_details.cached_tokensintoUsage.cached_tokens, andpricing.cost_usdbills those at the cached rate. Log it; it is the only way to know your ordering works in production.
Note that pricing.py currently lists Groq's cached-input price equal to its input price (no discount), while Groq's docs page describes a 50 percent discount. We used 50 percent above; verify both before relying on either.
| Situation | Use this | Why |
|---|---|---|
| Any production prompt | Stable content first (system, fixed examples, fixed tools), volatile last (profile, articles, state, ticket) | Only an exact prefix is cached; 70 percent vs 1 percent shared here |
| Multi-turn conversation | Append-only history after the stable prefix; per-turn material in the last message | Each turn then reuses the previous turn's whole prompt as its prefix |
| Prompt shorter than the provider's minimum cacheable length | Do not contort the design for caching | The discount never applies; optimize for tokens and quality instead |
| Tools rendered before the system prompt | A fixed tool list, or a few fixed tool sets | A varying list resets the cache at token zero |
F.2 Just-in-time loading vs preloading
Preloading puts retrieved articles into the first call. Just-in-time (JIT) loading gives the model a tool such as search_kb and lets it fetch what it needs, which is what Anthropic's post describes as keeping "lightweight identifiers" and loading data "at runtime using tools".
Section 2 of the output compares them on 48 dev tickets:
- Preload k = 3: 48 calls, 501 input tokens per ticket.
- Preload k = 1: 48 calls, 314 tokens per ticket.
- Just-in-time: 92 calls (44 tickets searched, 4 did not), 637 tokens per ticket.
JIT is the most expensive here, for a structural reason: the second call resends the system instruction, the ticket, and the tool definition, plus the tool call and result. It also doubles latency, because the two calls are sequential. JIT wins when the material is large relative to the base prompt, or when most requests need none of it. With our numbers, the base prompt is about 200 tokens and each article about 100, so preloading 3 articles costs about as much as one extra round trip. With 2,000-token articles, a large codebase, or a customer's entire order history, the balance flips: loading everything up front would cost thousands of tokens per call, and a JIT lookup costs only what is actually fetched.
| Situation | Use this | Why |
|---|---|---|
| Small corpus, short documents, most requests need them | Preload the top 1 to 3 | One call, lower latency, fewer tokens (501 vs 637 per ticket here) |
| Large documents or many possible sources | Just-in-time tools | Pay only for what the model fetches |
| Mixed: usually one article, sometimes more | Preload the top 1 and offer search_kb for more | Covers the common case in one call and the hard case in two |
| Latency-critical path (live chat) | Preload | JIT adds a full model round trip |
F.3 Compaction triggers and strategies
Compaction means replacing older history with a shorter summary when it grows too large. Anthropic's post defines it as "taking a conversation nearing the context window limit, summarizing its contents, and reinitiating a new context window with the summary." The question is when to trigger it. Section 3 of the output replays a 30-exchange conversation under four triggers, keeping the last 6 messages verbatim:
| Trigger | Peak history tokens | Total history tokens sent | Compactions (cache resets) |
|---|---|---|---|
| never | 1,480 | 22,146 | 0 |
| every 10 exchanges | 994 | 16,346 | 3 |
| at 70 percent of a 1,200-token budget | 829 | 14,265 | 2 |
| at 95 percent of a 1,200-token budget | 1,108 | 19,170 | 2 |
Reading it: compacting early (at 70 percent) sent the fewest tokens overall and kept the peak lowest, with only 2 compactions. Compacting late (95 percent) had the same number of compactions but let history sit near the ceiling for longer, so it sent 34 percent more tokens. A fixed schedule (every 10 exchanges) compacted more often than needed. Every compaction rewrites the prefix, so the next call misses the prompt cache; a trigger that fires too often trades token savings for cache misses.
What this table cannot tell you is what each compaction lost. The summarizer here is the extractive one from C.4, which we already know drops facts. Before choosing a trigger, measure summary fidelity: list the facts in a conversation, compact it, and count how many survive.
| Situation | Use this | Why |
|---|---|---|
| Long conversations with a known window | Token threshold well below the limit (for example 70 percent) | Fewest tokens sent, room for the next turn's retrieval |
| Topic changes (a new ticket from the same customer) | Reset: fresh history plus memories and state | Old topic is distraction, not context |
| Agent runs with many tool results | Clear old tool results first, summarize prose second | Tool output is bulky and rarely needed verbatim later |
| Facts must survive | Move them to the state scaffold before compacting | Structured fields are not summarized away |
F.4 Did the added context improve the output?
Every section in this module has a cost you can count. Whether it has a benefit needs a model and a test. The A/B harness in m07_optimize.py is the tool: two context variants, the same tickets, a judge, and paired statistics. Section 4 of the output runs it with k = 1 (A) vs k = 3 (B) articles, using the ScriptedLLM extractive answerer. That answerer is a word-overlap rule, not a model, so its pass rates say nothing about real drafts. What the run does show is the harness working: paired counting (0 wins, 5 losses, 37 ties), an exact McNemar p-value (0.062, so even this 5-to-0 split is not significant at 0.05 with 42 tickets), and the cost difference (1,137 vs 916 mean prompt tokens, about 1.71 vs 1.37 USD per 1,000 tickets in input on gemini-3.5-flash at pricing.py prices).
To get real numbers, set a provider key and run:
export LLM_PROVIDER=groq GROQ_API_KEY=... # or gemini / ollama, as in Module 1
python examples/m07_optimize.py --real
Code explained
- In simple words: run the same A/B comparison against a real model.
- What happens:
--realswaps the scripted answerer forsupportdesk.llm.chatat temperature 0 and judges each draft by whether itscited_articlescontains the gold article. It makes 2 calls per answerable dev ticket (84 calls) and prices them withpricing.py. - Comes out: the same four report lines with your model's pass rates. No real-model run was captured for this module, so no numbers are shown. Rerun it 2 or 3 times: if the pass counts move by several tickets between identical runs, that movement is your noise floor, and any A/B difference smaller than it is not a result.
| Situation | Use this | Why |
|---|---|---|
| Adding or removing a context section | Paired A/B on the same tickets, wins vs losses, exact test | Removes ticket difficulty from the comparison |
| Difference of a few tickets out of 40 | Treat as noise; collect more tickets or stop | With 40 tickets, even 5 to 0 gives p of about 0.06 |
| Quality equal, cost differs | Choose the cheaper context | Tokens are certain; an unmeasured benefit is not |
Module Lab
The Lab runs the whole pipeline over the 48 dev tickets: glossary-expanded retrieval, memories from the store, every context section under budget, a model call, DraftReply validation, and a report. Then it diagnoses the first failure by ablation.
examples/m07_lab.py
"""Module 7 Lab: the full context pipeline over the dev split.
For each ticket: expand the query and retrieve articles, load remembered facts,
assemble every section with the ContextBuilder, call the model, validate the
draft, and record tokens per section. Then report context recall, pass rate,
tokens, shared prefix, estimated cost, and diagnose the first failure by ablation.
Default model: a ScriptedLLM extractive answerer (plumbing only, not a model).
With --real it calls supportdesk.llm.chat on your configured provider.
Run: PYTHONPATH=. python examples/m07_lab.py [--real]
"""
from __future__ import annotations
import json
import re
import sys
from collections import defaultdict
from datetime import datetime
from pathlib import Path
sys.path.insert(0, str(Path(__file__).resolve().parent))
from m07_context import SYSTEM, ContextBuilder # noqa: E402
from m07_diagnose import article, diagnose # noqa: E402
from m07_memory import CONVERSATIONS, SCRIPTED_EXTRACTIONS, MemoryStore # noqa: E402
from m07_optimize import common_prefix, extractive_answerer, serialize # noqa: E402
from m07_retrieval import expand_query # noqa: E402
from pydantic import ValidationError # noqa: E402
from supportdesk.data import Ticket, load_tickets # noqa: E402
from supportdesk.kb_search import KBSearch # noqa: E402
from supportdesk.llm import Usage # noqa: E402
from supportdesk.pricing import PRICES, cost_usd # noqa: E402
from supportdesk.schemas import DraftReply # noqa: E402
from supportdesk.stand_in import ScriptedLLM # noqa: E402
from supportdesk.tokens import encode # noqa: E402
NOW = datetime(2026, 9, 21, 9, 0)
CUSTOMER_OF = {"T-1004": "cust-acme", "T-1041": "cust-acme", "T-1067": "cust-acme"} # demo mapping
def parse_draft(text: str) -> DraftReply | None:
match = re.search(r"\{.*\}", text, re.S)
if not match:
return None
try:
return DraftReply.model_validate(json.loads(match.group(0)))
except (json.JSONDecodeError, ValidationError):
return None
def main(real: bool) -> None:
kb = KBSearch()
store = MemoryStore()
seed = ScriptedLLM(replies=list(SCRIPTED_EXTRACTIONS[:3])) # scripted extractor, see m07_memory.py
for ticket_id, when, convo in CONVERSATIONS[:3]:
store.extract(convo, "cust-acme", ticket_id, when, seed)
if real:
from supportdesk.llm import chat, resolve
model = resolve()[1]
call = lambda msgs, tools: chat(msgs, tools=tools or None, temperature=0.0, max_tokens=600) # noqa: E731
else:
model = "gemini-3.5-flash"
stand_in = ScriptedLLM(responder=extractive_answerer)
call = lambda msgs, tools: stand_in(msgs, tools=tools) # noqa: E731
tickets = load_tickets("dev")
section_tokens: dict[str, list[int]] = defaultdict(list)
recall = passed = valid = 0
answerable_ok = {True: 0, False: 0}
prompts, failures, total = [], [], Usage()
for t in tickets:
hits = kb.search(expand_query(t), k=3)
customer = CUSTOMER_OF.get(t.id, f"cust-{t.id}")
memories = [m.line() for m in store.retrieve(customer, t.text, NOW)]
ctx = (ContextBuilder().system().examples().with_tools(category=t.gold["category"])
.profile({"customer_tier": t.customer_tier, "language": t.language}, memories)
.knowledge(hits)
.state({"ticket": t.id, "step": "draft_reply", "plan": ["read", "check policy", "draft", "cite"]})
.ticket(t).build())
for r in ctx.report:
section_tokens[r.name].append(r.tokens)
prompts.append(encode(serialize(ctx.messages, ctx.tools, tools_first=False)))
result = call(ctx.messages, ctx.tools)
total.input_tokens += result.usage.input_tokens if real else ctx.total_tokens # stand-in: our estimate
total.output_tokens += result.usage.output_tokens
total.cached_tokens += result.usage.cached_tokens
draft = parse_draft(result.text)
gold = t.gold["kb_article"]
recall += gold is None or gold in [h.article_id for h in hits]
valid += draft is not None
ok = draft is not None and (gold in draft.cited_articles if gold else draft.confidence == "low")
passed += ok
answerable_ok[gold is not None] += ok
if not ok:
failures.append((t, hits))
n = len(tickets)
print(f"Lab over {n} dev tickets ({'real model ' + model if real else 'ScriptedLLM extractive answerer: plumbing only'})")
print(f"\n{'section':10} {'mean':>6} {'max':>5}")
for name, values in section_tokens.items():
print(f"{name:10} {sum(values) / n:>6.0f} {max(values):>5}")
shared = [common_prefix(a, b) for a, b in zip(prompts, prompts[1:])]
print(f"\ncontext recall (gold article in context, unanswerable counted as ok): {recall}/{n}")
n_answerable = sum(t.gold["answerable"] for t in tickets)
print(f"valid DraftReply JSON: {valid}/{n}; answerable drafts citing the gold article: "
f"{answerable_ok[True]}/{n_answerable}; unanswerable drafts marked low confidence: "
f"{answerable_ok[False]}/{n - n_answerable}")
print(f"mean prompt {sum(map(len, prompts)) / n:.0f} tokens, mean shared prefix with previous ticket "
f"{sum(shared) / len(shared):.0f} tokens")
price_model = model if model in PRICES else "openai/gpt-oss-120b"
print(f"input+output tokens {total.input_tokens}+{total.output_tokens}; estimated cost on {price_model}: "
f"{cost_usd(total, price_model):.4f} USD for {n} tickets")
if failures:
t, hits = failures[0]
sections = {"system": [SYSTEM], "profile": [f"customer_tier: {t.customer_tier}"],
"knowledge": [article(h.article_id) for h in hits], "ticket": [t.text]}
def check(answer: str) -> bool:
draft = parse_draft(answer)
return draft is not None and t.gold["kb_article"] in draft.cited_articles
def rerun(messages: list[dict]) -> str:
return call(messages, None).text
d = diagnose(sections, rerun, check)
print(f"\nfirst failure {t.id} ({t.language}) gold={t.gold['kb_article']} retrieved={[h.article_id for h in hits]}")
print(f" verdict: {d.verdict} ({d.calls} calls)")
if __name__ == "__main__":
main(real="--real" in sys.argv)
Code explained
- In simple words: everything from this module, wired together and pointed at the dev split.
- What happens:
- The memory store is seeded from the first three scripted conversations of
m07_memory.py, so tickets mapped tocust-acmeget the pinned role and language plus relevant memories. - For each ticket:
expand_queryplusKBSearchretrieves 3 articles;store.retrievepicks memories;ContextBuilderassembles all sections with per-category tools; the model is called;parse_draftvalidates the reply againstDraftReplyfromsupportdesk/schemas.py. - It records tokens per section, whether the gold article reached the context (context recall), whether the draft cites it, and the serialized prompt for shared-prefix measurement.
- For the first failure it rebuilds the context as ablation sections and runs
diagnosewith the same model. - Without
--realthe model is the scripted extractive answerer; with--realit issupportdesk.llm.chaton your configured provider.
- The memory store is seeded from the first three scripted conversations of
- Comes out: (scripted run; token counts are real, draft pass rates are not model quality)
Lab over 48 dev tickets (ScriptedLLM extractive answerer: plumbing only)
section mean max
system 171 171
examples 369 369
tools 180 292
profile 22 118
knowledge 349 396
state 30 30
ticket 41 57
context recall (gold article in context, unanswerable counted as ok): 46/48
valid DraftReply JSON: 48/48; answerable drafts citing the gold article: 30/42; unanswerable drafts marked low confidence: 0/6
mean prompt 1191 tokens, mean shared prefix with previous ticket 664 tokens
input+output tokens 56546+1006; estimated cost on gemini-3.5-flash: 0.0939 USD for 48 tickets
first failure T-1001 (en) gold=billing-refunds retrieved=['billing-invoices', 'billing-plans', 'boards-automations']
verdict: no single removal fixes it: check retrieval (missing knowledge) or the model itself (7 calls)
What is real here: the section sizes (the stable system and examples at 540 tokens, knowledge the largest volatile section at 349 on average, the profile growing to 118 tokens only for the customer with memories), context recall of 46 of 48 (40 of 42 answerable tickets have their gold article in the context, plus the 6 unanswerable ones), the 664-token shared prefix, and the cost estimate of about 0.09 USD for 48 tickets on gemini-3.5-flash at pricing.py prices.
What is scripted: 30 of 42 citing the gold article, and 0 of 6 unanswerable tickets marked low confidence (the stand-in always says "medium"). The diagnosis is still informative, because it is about the context, not the model: T-1001's gold article never reached the context, so no ablation can fix it, and the verdict says to fix retrieval. That matches Part B, where T-1001 was the English miss the glossary did not touch.
Run it against a real model with python examples/m07_lab.py --real and compare three numbers with the scripted run: valid JSON rate, gold-citation rate on answerable tickets, and low-confidence rate on unanswerable ones. The gap between context recall (40 of 42) and gold citations is the part a better prompt or model can close; the 2 tickets missing from the context are the part only retrieval can close.
The module's tests (tests/test_m07_context.py, 14 tests) pin the behaviors that matter: every section within budget, whole articles only, an over-budget system instruction raises, the stable prefix is identical across tickets, the glossary rescues Japanese ticket T-1034, trust-then-recency conflict resolution, redaction that keeps invoice numbers (the bug from D.3), rejection of sensitive and malformed memories, forget and opt-out, TTL expiry, the three diagnostic verdicts, and the McNemar helper.
python -m pytest -q tests/test_m07_context.py
Code explained
- In simple words: run the Module 7 tests.
- What happens: pytest imports the example files through
sys.pathand runs 14 tests offline; no key needed. - Comes out:
.............. [100%]
14 passed in 4.40s
Project Milestone
After this module the Brightlane support assistant repository contains:
supportdesk/kb_search.py(canonical, introduced here) and a measured baseline: 52 of 62 top-1, 57 of 62 top-3, with non-English tickets at 3 of 8 top-1.examples/m07_retrieval.py: per-language evaluation and the glossary query expansion (56 of 62 top-1, 59 of 62 top-3; 4 gains, 0 losses, p = 0.125).examples/m07_context.py: theContextBuilderwith per-section budgets, cache-aware assembly, and a token report for every request.examples/m07_memory.py: a long-term memory store with provenance, trust-then-recency conflict resolution, TTLs, redaction, forget, opt-out, and export.examples/m07_diagnose.py: an ablation diagnostic for poisoning, clash, and distraction.examples/m07_optimize.py: shared-prefix measurement, JIT vs preload, compaction triggers, and the A/B harness with--real.examples/m07_rot.py: the TinyLM context-rot and poisoning measurements, plus a real-model harness.- Small section studies:
m07_setup.py,m07_order.py,m07_history.py,m07_tool_cost.py,m07_fewshot_cost.py,m07_state.py. examples/m07_lab.pyandtests/test_m07_context.py(14 tests passing).
Maya's team can now see, for any draft, exactly which articles, memories, and history went into it and how many tokens each took. When a draft is wrong, they can hand the case to the diagnostic instead of guessing.
Interview Questions
1. What is context engineering, and how is it different from prompt engineering? Prompt engineering is about wording the instruction. Context engineering is about everything in the window on each call: which documents, how much history, which tools, what memories, in what order and size. Most of a production prompt is assembled by code from data, so the engineering is in the selection, budgeting, and ordering logic, and in measuring its effect. In our assistant the hand-written instruction is 171 tokens of a roughly 1,200-token prompt.
2. The model has a 1-million-token window. Why not just include the whole help center and every past ticket? Three reasons. Cost and latency scale with input tokens on every call. Quality degrades: Chroma's 2025 context-rot study found performance "grows increasingly unreliable as input length grows" across 18 models, and a focused 300-token prompt beat a 113,000-token one on LongMemEval. And every irrelevant passage is a potential distractor or clash. Retrieve the few items that answer the question; offer a search tool for the rest.
3. How would you decide how many retrieved documents to include? Measure recall@k against tokens added on labeled data and stop where the curve flattens. On Brightlane, k = 1 gives 0.90 recall for 109 tokens, k = 3 gives 0.95 for 308, and each ticket rescued beyond k = 2 costs thousands of tokens across the dataset. Then confirm with an A/B test that the extra articles improve answers, because more recall can still mean more distraction.
4. Your retriever works well in English but poorly for Japanese tickets. How do you investigate? Break the hit rate down by language first (ours: 0 of 2 Japanese at top-1). Then look at what the query becomes after tokenization: our BM25 tokenizer keeps only [a-z0-9], so a Japanese ticket becomes ['subject', 'team']. The fix has to add searchable text: a glossary or translation step, or a multilingual embedding model. Validate with a paired comparison and a held-out split, and report the small n honestly.
5. What goes in the stable prefix and why does order matter for cost? The system instruction, fixed examples, and a fixed tool list go first; profile, articles, state, and the new message go last. Prompt caching only reuses an exact token prefix, so one volatile token early breaks it for everything after. We measured 70 percent shared prefix across consecutive tickets with stable-first ordering vs 1 percent with the order reversed. Check the provider's minimum cacheable length (4,096 tokens for Gemini 3.5 Flash per Google's docs) before redesigning around caching.
6. How do you handle conversation history in a long support thread? Drop content-free turns, keep the last few verbatim, summarize older turns once history passes a token threshold (70 percent of the budget worked better than 95 percent in our replay), and move facts that must survive (invoice numbers, dates, promises) into a structured state scaffold before compacting. Test the summarizer: our first-sentence summarizer dropped the customer's finance contact and kept "Thanks for reaching out."
7. What should an assistant remember long-term, and what should it never remember? Remember preferences, roles, and open issues that save the customer effort and cannot be looked up. Do not remember what a system of record owns (plan, price, invoices), because the copy goes stale and clashes with the source. Never store secrets such as card numbers, passwords, or codes; redact before extraction. Every memory needs a source and a timestamp, and the customer needs forget, opt-out, and export.
8. Two memories disagree: the customer said "reply in English" last week, and the extractor inferred "prefers Spanish" yesterday. Which wins? The customer's statement. Resolve by source trust first and recency second: a model inference should never override what a person said, and among equally trusted sources the newer wins. Keep the losing version inactive, not deleted, so you can explain decisions, unless the customer asks to forget.
9. A draft is wrong. How do you find out whether the context caused it? Ablate. Keep the context as structured sections, remove one section at a time and rerun, then one item at a time. One item whose removal fixes the answer points to poisoning (or clash, if it contradicts another section); several interchangeable removals point to distraction; no fix means the needed fact was never retrieved. With a real model, repeat each ablation several times because of sampling noise, and run it offline on collected failures because it multiplies calls.
10. When is just-in-time loading better than preloading context? When the material is large relative to the base prompt or often unnecessary. JIT costs an extra round trip that resends the whole prompt. On our small help center, preloading 3 articles cost 501 tokens per ticket vs 637 for JIT with 92 calls instead of 48. With long documents or many sources the balance flips. A common hybrid is to preload the top hit and offer a search tool.
11. How do you prove that adding a context section helped? Run both variants on the same tickets and count paired wins and losses, then use an exact test. With about 40 tickets, a 5-to-0 split still gives p of about 0.06, so small differences are noise. Measure the noise floor by rerunning the same variant. If quality is indistinguishable, choose the cheaper context, because its cost is certain.
12. Why did adding irrelevant text barely affect TinyLM when it was same-domain, but hurt it badly when it was off-domain? TinyLM memorized its templated corpus, so same-format conversations are familiar and it continues its learned pattern, while unfamiliar prose pushes it out of distribution (first-token probability fell from 0.998 to 0.364 at 64 tokens). Large models read and weigh context, so for them plausible, similar distractors are the dangerous kind, as Chroma found. The transferable lesson is that every token in the window shifts the prediction.
Other Tools and Providers
| Tool or provider | What it offers for context engineering | When to consider it |
|---|---|---|
Anthropic prompt caching (cache_control breakpoints) | Explicit cache markers on stable prefix blocks | You want control over exactly what is cached and for how long |
| OpenAI automatic prompt caching | Automatic prefix caching for longer prompts, reported in usage | You use OpenAI models and want caching without code changes |
| Gemini implicit and explicit context caching | Implicit caching by default on newer models; explicit caches for large shared content | Large shared documents reused across many requests |
| Groq prompt caching | Automatic caching on supported models (gpt-oss family) with a cached-input discount | You run the course's default provider |
| LangGraph / LangChain memory | Checkpointed state and memory stores for agent graphs | You are already building on LangChain components |
| LlamaIndex | Retrieval, chunking, and query pipelines over documents | Larger document collections than a 12-article help center |
| Letta (formerly MemGPT), Mem0, Zep | Managed long-term memory with extraction and retrieval | You want a memory service instead of your own store; still apply your own privacy rules |
| Multilingual embedding models (for example from the sentence-transformers family) | Vector retrieval that matches across languages | Keyword search with glossaries stops scaling (Part B.3) |
| Chroma, pgvector, Qdrant, Weaviate | Vector stores for retrieval and memory | You move retrieval or memory from keywords to embeddings |
| Chroma's context-rot and LongMemEval-style benchmarks | Published harnesses for long-context degradation | You want to test your model's long-context behavior beyond real_model_rot |
Coming Up in Module 8
Module 8 turns the assistant into an agent: a loop that perceives, decides, acts through tools, and observes, for many steps. Everything in this module becomes a per-step problem there. The state scaffold becomes the agent's working memory, compaction decides whether a 40-step run fits in the window, just-in-time loading becomes the agent's main way to learn, and the ablation diagnostic grows into trajectory evaluation. You will also meet the costs of autonomy: every extra step resends the context you just learned to budget.