Topic 8: Choosing Between Them
13 min read·21 Sept 2026
python
# module10/choose.py
"""Choosing a pattern: describe each one, describe your constraints, let the table decide."""
from __future__ import annotations
from dataclasses import dataclass, field
@dataclass(frozen=True)
class PatternSpec:
name: str
extra_llm_calls: int # per query, on top of the answer call
extra_latency_ms: int # rough, per query
index_cost: str # none | medium | high (one-time, per corpus)
handles: frozenset # query kinds it is good at
needs: frozenset = frozenset() # what it requires to be possible
note: str = ""
SPECS = [
PatternSpec("naive", 0, 0, "none", frozenset({"simple"}), note="baseline; fine for tiny, clean corpora"),
PatternSpec("hybrid", 0, 20, "none", frozenset({"simple", "keyword", "paraphrase"}),
note="the cheapest real upgrade: codes and synonyms both work"),
PatternSpec("reranked", 0, 150, "none", frozenset({"simple", "keyword", "paraphrase", "lookalike"}),
note="the usual production default"),
PatternSpec("parent-document", 0, 160, "medium", frozenset({"simple", "procedure", "lookalike"}),
note="precise matching, complete context"),
PatternSpec("sentence-window", 0, 160, "medium", frozenset({"factoid", "dense-prose"}),
note="pinpoint answers in long prose"),
PatternSpec("hierarchical", 0, 170, "high", frozenset({"broad", "simple"}),
note="summary layer narrows before detail"),
PatternSpec("raptor", 0, 170, "high", frozenset({"broad", "thematic"}),
note="answers 'overall' questions; summaries must be rebuilt when content changes"),
PatternSpec("contextual", 0, 150, "high", frozenset({"lookalike", "paraphrase"}),
note="index-time context; strong when chunks look alike"),
PatternSpec("multi-query", 1, 500, "none", frozenset({"paraphrase", "vague"}),
note="covers wording gaps"),
PatternSpec("hyde", 1, 600, "none", frozenset({"paraphrase", "vague"}),
note="can drift on codes; measure before enabling"),
PatternSpec("step-back", 1, 500, "none", frozenset({"specific-case", "policy"}),
note="finds the governing rule"),
PatternSpec("decomposed", 1, 600, "none", frozenset({"compound", "multi-hop"}),
note="each part gets represented"),
PatternSpec("routed", 0, 120, "medium", frozenset({"heterogeneous"}),
needs=frozenset({"several corpora"}), note="fewer, cleaner candidates"),
PatternSpec("self-rag", 3, 1200, "none", frozenset({"mixed", "quality-critical"}),
note="grades and critiques; expensive"),
PatternSpec("crag", 1, 700, "none", frozenset({"gappy-corpus"}),
needs=frozenset({"fallback source"}), note="knows when the corpus fails"),
PatternSpec("adaptive", 1, 400, "none", frozenset({"mixed"}),
note="cheap questions stay cheap"),
PatternSpec("iterative", 2, 1500, "none", frozenset({"multi-hop", "research"}),
note="needs a hard round limit"),
PatternSpec("flare", 1, 800, "none", frozenset({"long-answer", "quality-critical"}),
note="fixes weakly supported sentences"),
PatternSpec("agentic", 4, 3000, "none", frozenset({"multi-hop", "tools", "open-ended"}),
needs=frozenset({"tools"}), note="most flexible, least predictable"),
PatternSpec("multi-agent", 5, 4000, "none", frozenset({"research", "report"}),
note="roles help long tasks, not short lookups"),
PatternSpec("graph", 1, 400, "high", frozenset({"entity", "relationship", "global"}),
note="extraction cost per document; strong for 'how do X and Y relate'"),
PatternSpec("structured", 1, 300, "none", frozenset({"aggregate", "exact-data"}),
needs=frozenset({"database"}), note="counts and sums belong in SQL"),
PatternSpec("structured+text", 1, 450, "none", frozenset({"mixed-data"}),
needs=frozenset({"database"}), note="policy plus the customer's own data"),
PatternSpec("conversational", 1, 400, "none", frozenset({"follow-up", "chat"}),
note="memory, not just retrieval"),
PatternSpec("long-context", 0, 900, "none", frozenset({"simple", "whole-document"}),
needs=frozenset({"tiny corpus"}), note="cache the prefix or it gets expensive"),
PatternSpec("federated", 0, 600, "none", frozenset({"multi-system"}),
needs=frozenset({"remote retrievers"}), note="data stays where it lives"),
PatternSpec("streaming", 0, 120, "none", frozenset({"fresh-data"}),
note="live buffer in front of the index"),
]
@dataclass
class Constraints:
query_kinds: frozenset = frozenset({"simple"})
latency_budget_ms: int = 1500
extra_llm_calls_allowed: int = 1
index_build_budget: str = "medium" # none | medium | high
available: frozenset = frozenset() # capabilities you actually have
quality_critical: bool = False
COST_ORDER = {"none": 0, "medium": 1, "high": 2}
def recommend(constraints: Constraints, specs: list[PatternSpec] = SPECS) -> list[PatternSpec]:
"""Keep the patterns you can afford that cover your query kinds; cheapest first."""
affordable = [
s for s in specs
if s.extra_latency_ms <= constraints.latency_budget_ms
and s.extra_llm_calls <= constraints.extra_llm_calls_allowed
and COST_ORDER[s.index_cost] <= COST_ORDER[constraints.index_build_budget]
and s.needs <= constraints.available
and (s.handles & constraints.query_kinds)
]
return sorted(affordable, key=lambda s: (s.extra_llm_calls, s.extra_latency_ms,
COST_ORDER[s.index_cost]))
def explain(constraints: Constraints, specs: list[PatternSpec] = SPECS) -> list[str]:
"""Why each pattern was rejected. Useful when the recommendation surprises you."""
reasons = []
for s in specs:
why = []
if s.extra_latency_ms > constraints.latency_budget_ms:
why.append("too slow")
if s.extra_llm_calls > constraints.extra_llm_calls_allowed:
why.append("too many LLM calls")
if COST_ORDER[s.index_cost] > COST_ORDER[constraints.index_build_budget]:
why.append("index build too expensive")
if not s.needs <= constraints.available:
why.append(f"needs {sorted(s.needs - constraints.available)}")
if not (s.handles & constraints.query_kinds):
why.append("does not target these query kinds")
if why:
reasons.append(f"{s.name}: " + ", ".join(why))
return reasonsCode explained: module10/choose.py
PatternSpecdescribes a pattern in the terms that matter for a decision: extra LLM calls per query, rough extra latency, index build cost, the query kinds it targets, and what it needs to be possible at all.