Topic 8: Measure
4 min read·22 Sept 2026
The question worth measuring here is narrow: does a second replica help, and what does the balancer cost? examples/m09_loadtest.py sends 200 tools/call requests for search_notes at concurrency 10 against three setups, after 20 unmeasured warm-up requests:
python
"""200 search_notes calls at concurrency 10 against three setups. One machine: not a capacity test."""
from __future__ import annotations
import statistics
import time
import anyio
import httpx2
from examples.m09_fleet import fleet
REQUESTS, CONCURRENCY = 200, 10
HEADERS = {"Accept": "application/json, text/event-stream", "MCP-Protocol-Version": "2026-07-28",
"Mcp-Method": "tools/call", "Mcp-Name": "search_notes"}
def body(n: int) -> dict:
return {"jsonrpc": "2.0", "id": n, "method": "tools/call", "params": {
"name": "search_notes", "arguments": {"query": "sleep memory focus", "limit": 3},
"_meta": {"io.modelcontextprotocol/protocolVersion": "2026-07-28",
"io.modelcontextprotocol/clientCapabilities": {}}}}
async def load(url: str) -> tuple[list[float], float, int]:
latencies: list[float] = []
errors = 0
queue = iter(range(REQUESTS))
limits = httpx2.Limits(max_connections=CONCURRENCY)
async def worker(http: httpx2.AsyncClient) -> None:
nonlocal errors
for n in queue:
started = time.perf_counter()
response = await http.post(url, json=body(n), headers=HEADERS)
latencies.append((time.perf_counter() - started) * 1000)
errors += response.status_code != 200 or '"isError":true' in response.text
async with httpx2.AsyncClient(limits=limits, timeout=30) as http:
for n in range(20): # warm-up, not measured
await http.post(url, json=body(n), headers=HEADERS)
started = time.perf_counter()
async with anyio.create_task_group() as tg:
for _ in range(CONCURRENCY):
tg.start_soon(worker, http)
elapsed = time.perf_counter() - started
return latencies, elapsed, errors
def report(label: str, latencies: list[float], elapsed: float, errors: int) -> None:
p95 = statistics.quantiles(latencies, n=20)[-1]
print(f"{label:<28} median {statistics.median(latencies):6.1f} ms p95 {p95:6.1f} ms "
f"{len(latencies) / elapsed:6.0f} req/s errors {errors}")
def main() -> None:
print(f"{REQUESTS} requests, concurrency {CONCURRENCY}")
with fleet(replicas=1, balancer_port=None) as f:
report("1 replica, direct", *anyio.run(load, f.url))
with fleet(replicas=1, balancer_port=8090) as f:
report("1 replica, via balancer", *anyio.run(load, f.url))
with fleet(replicas=2, balancer_port=8090) as f:
report("2 replicas, via balancer", *anyio.run(load, f.url))
if __name__ == "__main__":
main()Code explained
- In simple words: ten workers take 200 requests off a shared queue as fast as they can, and we time each one.