CourseRAG · Module 13: Advanced and Emerging Topics · part 73 of 82
Part 73 · Module 13: Advanced and Emerging Topics

Topic 4: Frontier Trade-offs

16 min read·21 Sept 2026

module13/frontier.py

python
# module13/frontier.py
"""The economics behind the frontier arguments: long context, caching, and agentic research."""
from __future__ import annotations

from dataclasses import dataclass


@dataclass
class ModelPrices:
    """Per million tokens. Replace with your provider's current numbers."""
    input_usd: float = 0.30
    cached_input_usd: float = 0.03          # cached prefixes are typically much cheaper
    output_usd: float = 1.20


def long_context_vs_rag(corpus_tokens: int, queries_per_month: int, prices: ModelPrices,
                        rag_context_tokens: int = 2_000, output_tokens: int = 300,
                        rag_infra_usd_per_month: float = 200.0, cache_hit_rate: float = 0.0) -> dict:
    """Stuffing the whole corpus versus retrieving a small context, per month."""
    def generation_cost(input_tokens: int, cached_share: float) -> float:
        cached = input_tokens * cached_share
        fresh = input_tokens - cached
        per_query = (fresh / 1e6 * prices.input_usd + cached / 1e6 * prices.cached_input_usd
                     + output_tokens / 1e6 * prices.output_usd)
        return per_query * queries_per_month

    stuffing = generation_cost(corpus_tokens, cache_hit_rate)
    retrieval = generation_cost(rag_context_tokens, 0.0) + rag_infra_usd_per_month
    return {"corpus_tokens": corpus_tokens, "queries_per_month": queries_per_month,
            "long_context_usd": round(stuffing, 2), "rag_usd": round(retrieval, 2),
            "cheaper": "long context" if stuffing < retrieval else "RAG",
            "ratio": round(stuffing / retrieval, 2) if retrieval else float("inf")}


def cache_augmented_break_even(corpus_tokens: int, prices: ModelPrices,
                               rag_context_tokens: int = 2_000) -> dict:
    """With a cached prefix, how big can the corpus get before retrieval is cheaper per query?"""
    rag_per_query = rag_context_tokens / 1e6 * prices.input_usd
    cag_per_query = corpus_tokens / 1e6 * prices.cached_input_usd
    break_even_tokens = rag_per_query / prices.cached_input_usd * 1e6
    return {"cag_per_query_usd": round(cag_per_query, 6), "rag_per_query_usd": round(rag_per_query, 6),
            "break_even_corpus_tokens": int(break_even_tokens),
            "verdict": "cache the corpus" if corpus_tokens <= break_even_tokens else "retrieve"}


def agentic_research_cost(steps: int, tools_per_step: float, context_growth_tokens: int,
                          prices: ModelPrices, output_tokens_per_step: int = 400) -> dict:
    """Agent loops re-send a growing transcript: cost grows roughly with the square of the steps."""
    total_input = sum(context_growth_tokens * step for step in range(1, steps + 1))
    total_output = output_tokens_per_step * steps
    cost = total_input / 1e6 * prices.input_usd + total_output / 1e6 * prices.output_usd
    single_pass = (context_growth_tokens / 1e6 * prices.input_usd
                   + output_tokens_per_step / 1e6 * prices.output_usd)
    return {"steps": steps, "tool_calls": round(steps * tools_per_step, 1),
            "input_tokens": total_input, "usd": round(cost, 4),
            "single_pass_usd": round(single_pass, 4),
            "multiplier_vs_single_pass": round(cost / single_pass, 1) if single_pass else 0.0}


STACK_STABILITY = {
    "stable": [
        "Chunking with stable ids and preserved context (Module 3)",
        "Hybrid retrieval: lexical plus dense (Module 6)",
        "Reranking a wide candidate pool (Module 8)",
        "Grounded generation with citations and refusals (Module 9)",
        "Evaluation sets, attribution, and gates (Module 11)",
        "Permissions, audit, and observability (Module 12)",
    ],
    "moving": [
        "Which embedding model is best this quarter (Module 4)",
        "Sparse and late-interaction variants and their serving costs",
        "How much reranking a long-context model still needs",
        "Agentic loops: capability rising, cost discipline lagging",
        "Vision-based document retrieval replacing parsing (Module 10)",
        "Graph construction cost versus its benefit",
    ],
    "watch": [
        "Context windows and their real (not advertised) effective length",
        "Prompt caching prices: they change the long-context maths directly",
        "Model-side retrieval and built-in file search from providers",
        "Regulatory pressure on data residency and retention",
    ],
}

The rest of this course is yours to keep

This course is bought on its own, once, and stays readable afterwards, including the parts added to it later.