Topic 4: Frontier Trade-offs
16 min read·21 Sept 2026
module13/frontier.py
python
# module13/frontier.py
"""The economics behind the frontier arguments: long context, caching, and agentic research."""
from __future__ import annotations
from dataclasses import dataclass
@dataclass
class ModelPrices:
"""Per million tokens. Replace with your provider's current numbers."""
input_usd: float = 0.30
cached_input_usd: float = 0.03 # cached prefixes are typically much cheaper
output_usd: float = 1.20
def long_context_vs_rag(corpus_tokens: int, queries_per_month: int, prices: ModelPrices,
rag_context_tokens: int = 2_000, output_tokens: int = 300,
rag_infra_usd_per_month: float = 200.0, cache_hit_rate: float = 0.0) -> dict:
"""Stuffing the whole corpus versus retrieving a small context, per month."""
def generation_cost(input_tokens: int, cached_share: float) -> float:
cached = input_tokens * cached_share
fresh = input_tokens - cached
per_query = (fresh / 1e6 * prices.input_usd + cached / 1e6 * prices.cached_input_usd
+ output_tokens / 1e6 * prices.output_usd)
return per_query * queries_per_month
stuffing = generation_cost(corpus_tokens, cache_hit_rate)
retrieval = generation_cost(rag_context_tokens, 0.0) + rag_infra_usd_per_month
return {"corpus_tokens": corpus_tokens, "queries_per_month": queries_per_month,
"long_context_usd": round(stuffing, 2), "rag_usd": round(retrieval, 2),
"cheaper": "long context" if stuffing < retrieval else "RAG",
"ratio": round(stuffing / retrieval, 2) if retrieval else float("inf")}
def cache_augmented_break_even(corpus_tokens: int, prices: ModelPrices,
rag_context_tokens: int = 2_000) -> dict:
"""With a cached prefix, how big can the corpus get before retrieval is cheaper per query?"""
rag_per_query = rag_context_tokens / 1e6 * prices.input_usd
cag_per_query = corpus_tokens / 1e6 * prices.cached_input_usd
break_even_tokens = rag_per_query / prices.cached_input_usd * 1e6
return {"cag_per_query_usd": round(cag_per_query, 6), "rag_per_query_usd": round(rag_per_query, 6),
"break_even_corpus_tokens": int(break_even_tokens),
"verdict": "cache the corpus" if corpus_tokens <= break_even_tokens else "retrieve"}
def agentic_research_cost(steps: int, tools_per_step: float, context_growth_tokens: int,
prices: ModelPrices, output_tokens_per_step: int = 400) -> dict:
"""Agent loops re-send a growing transcript: cost grows roughly with the square of the steps."""
total_input = sum(context_growth_tokens * step for step in range(1, steps + 1))
total_output = output_tokens_per_step * steps
cost = total_input / 1e6 * prices.input_usd + total_output / 1e6 * prices.output_usd
single_pass = (context_growth_tokens / 1e6 * prices.input_usd
+ output_tokens_per_step / 1e6 * prices.output_usd)
return {"steps": steps, "tool_calls": round(steps * tools_per_step, 1),
"input_tokens": total_input, "usd": round(cost, 4),
"single_pass_usd": round(single_pass, 4),
"multiplier_vs_single_pass": round(cost / single_pass, 1) if single_pass else 0.0}
STACK_STABILITY = {
"stable": [
"Chunking with stable ids and preserved context (Module 3)",
"Hybrid retrieval: lexical plus dense (Module 6)",
"Reranking a wide candidate pool (Module 8)",
"Grounded generation with citations and refusals (Module 9)",
"Evaluation sets, attribution, and gates (Module 11)",
"Permissions, audit, and observability (Module 12)",
],
"moving": [
"Which embedding model is best this quarter (Module 4)",
"Sparse and late-interaction variants and their serving costs",
"How much reranking a long-context model still needs",
"Agentic loops: capability rising, cost discipline lagging",
"Vision-based document retrieval replacing parsing (Module 10)",
"Graph construction cost versus its benefit",
],
"watch": [
"Context windows and their real (not advertised) effective length",
"Prompt caching prices: they change the long-context maths directly",
"Model-side retrieval and built-in file search from providers",
"Regulatory pressure on data residency and retention",
],
}