CourseRAG · Module 10: RAG Architectures and Patterns · part 55 of 82
Part 55 · Module 10: RAG Architectures and Patterns

Topic 8: Choosing Between Them

13 min read·21 Sept 2026
python
# module10/choose.py
"""Choosing a pattern: describe each one, describe your constraints, let the table decide."""
from __future__ import annotations

from dataclasses import dataclass, field


@dataclass(frozen=True)
class PatternSpec:
    name: str
    extra_llm_calls: int            # per query, on top of the answer call
    extra_latency_ms: int           # rough, per query
    index_cost: str                 # none | medium | high (one-time, per corpus)
    handles: frozenset              # query kinds it is good at
    needs: frozenset = frozenset()  # what it requires to be possible
    note: str = ""


SPECS = [
    PatternSpec("naive", 0, 0, "none", frozenset({"simple"}), note="baseline; fine for tiny, clean corpora"),
    PatternSpec("hybrid", 0, 20, "none", frozenset({"simple", "keyword", "paraphrase"}),
                note="the cheapest real upgrade: codes and synonyms both work"),
    PatternSpec("reranked", 0, 150, "none", frozenset({"simple", "keyword", "paraphrase", "lookalike"}),
                note="the usual production default"),
    PatternSpec("parent-document", 0, 160, "medium", frozenset({"simple", "procedure", "lookalike"}),
                note="precise matching, complete context"),
    PatternSpec("sentence-window", 0, 160, "medium", frozenset({"factoid", "dense-prose"}),
                note="pinpoint answers in long prose"),
    PatternSpec("hierarchical", 0, 170, "high", frozenset({"broad", "simple"}),
                note="summary layer narrows before detail"),
    PatternSpec("raptor", 0, 170, "high", frozenset({"broad", "thematic"}),
                note="answers 'overall' questions; summaries must be rebuilt when content changes"),
    PatternSpec("contextual", 0, 150, "high", frozenset({"lookalike", "paraphrase"}),
                note="index-time context; strong when chunks look alike"),
    PatternSpec("multi-query", 1, 500, "none", frozenset({"paraphrase", "vague"}),
                note="covers wording gaps"),
    PatternSpec("hyde", 1, 600, "none", frozenset({"paraphrase", "vague"}),
                note="can drift on codes; measure before enabling"),
    PatternSpec("step-back", 1, 500, "none", frozenset({"specific-case", "policy"}),
                note="finds the governing rule"),
    PatternSpec("decomposed", 1, 600, "none", frozenset({"compound", "multi-hop"}),
                note="each part gets represented"),
    PatternSpec("routed", 0, 120, "medium", frozenset({"heterogeneous"}),
                needs=frozenset({"several corpora"}), note="fewer, cleaner candidates"),
    PatternSpec("self-rag", 3, 1200, "none", frozenset({"mixed", "quality-critical"}),
                note="grades and critiques; expensive"),
    PatternSpec("crag", 1, 700, "none", frozenset({"gappy-corpus"}),
                needs=frozenset({"fallback source"}), note="knows when the corpus fails"),
    PatternSpec("adaptive", 1, 400, "none", frozenset({"mixed"}),
                note="cheap questions stay cheap"),
    PatternSpec("iterative", 2, 1500, "none", frozenset({"multi-hop", "research"}),
                note="needs a hard round limit"),
    PatternSpec("flare", 1, 800, "none", frozenset({"long-answer", "quality-critical"}),
                note="fixes weakly supported sentences"),
    PatternSpec("agentic", 4, 3000, "none", frozenset({"multi-hop", "tools", "open-ended"}),
                needs=frozenset({"tools"}), note="most flexible, least predictable"),
    PatternSpec("multi-agent", 5, 4000, "none", frozenset({"research", "report"}),
                note="roles help long tasks, not short lookups"),
    PatternSpec("graph", 1, 400, "high", frozenset({"entity", "relationship", "global"}),
                note="extraction cost per document; strong for 'how do X and Y relate'"),
    PatternSpec("structured", 1, 300, "none", frozenset({"aggregate", "exact-data"}),
                needs=frozenset({"database"}), note="counts and sums belong in SQL"),
    PatternSpec("structured+text", 1, 450, "none", frozenset({"mixed-data"}),
                needs=frozenset({"database"}), note="policy plus the customer's own data"),
    PatternSpec("conversational", 1, 400, "none", frozenset({"follow-up", "chat"}),
                note="memory, not just retrieval"),
    PatternSpec("long-context", 0, 900, "none", frozenset({"simple", "whole-document"}),
                needs=frozenset({"tiny corpus"}), note="cache the prefix or it gets expensive"),
    PatternSpec("federated", 0, 600, "none", frozenset({"multi-system"}),
                needs=frozenset({"remote retrievers"}), note="data stays where it lives"),
    PatternSpec("streaming", 0, 120, "none", frozenset({"fresh-data"}),
                note="live buffer in front of the index"),
]


@dataclass
class Constraints:
    query_kinds: frozenset = frozenset({"simple"})
    latency_budget_ms: int = 1500
    extra_llm_calls_allowed: int = 1
    index_build_budget: str = "medium"        # none | medium | high
    available: frozenset = frozenset()        # capabilities you actually have
    quality_critical: bool = False


COST_ORDER = {"none": 0, "medium": 1, "high": 2}


def recommend(constraints: Constraints, specs: list[PatternSpec] = SPECS) -> list[PatternSpec]:
    """Keep the patterns you can afford that cover your query kinds; cheapest first."""
    affordable = [
        s for s in specs
        if s.extra_latency_ms <= constraints.latency_budget_ms
        and s.extra_llm_calls <= constraints.extra_llm_calls_allowed
        and COST_ORDER[s.index_cost] <= COST_ORDER[constraints.index_build_budget]
        and s.needs <= constraints.available
        and (s.handles & constraints.query_kinds)
    ]
    return sorted(affordable, key=lambda s: (s.extra_llm_calls, s.extra_latency_ms,
                                             COST_ORDER[s.index_cost]))


def explain(constraints: Constraints, specs: list[PatternSpec] = SPECS) -> list[str]:
    """Why each pattern was rejected. Useful when the recommendation surprises you."""
    reasons = []
    for s in specs:
        why = []
        if s.extra_latency_ms > constraints.latency_budget_ms:
            why.append("too slow")
        if s.extra_llm_calls > constraints.extra_llm_calls_allowed:
            why.append("too many LLM calls")
        if COST_ORDER[s.index_cost] > COST_ORDER[constraints.index_build_budget]:
            why.append("index build too expensive")
        if not s.needs <= constraints.available:
            why.append(f"needs {sorted(s.needs - constraints.available)}")
        if not (s.handles & constraints.query_kinds):
            why.append("does not target these query kinds")
        if why:
            reasons.append(f"{s.name}: " + ", ".join(why))
    return reasons

Code explained: module10/choose.py

  • PatternSpec describes a pattern in the terms that matter for a decision: extra LLM calls per query, rough extra latency, index build cost, the query kinds it targets, and what it needs to be possible at all.

The rest of this course is yours to keep

This course is bought on its own, once, and stays readable afterwards, including the parts added to it later.