CourseModel Context Protocol · Module 9: Production and Ecosystem · part 65 of 83
Part 65 · Module 9: Production and Ecosystem

Topic 8: Measure

4 min read·22 Sept 2026

The question worth measuring here is narrow: does a second replica help, and what does the balancer cost? examples/m09_loadtest.py sends 200 tools/call requests for search_notes at concurrency 10 against three setups, after 20 unmeasured warm-up requests:

python
"""200 search_notes calls at concurrency 10 against three setups. One machine: not a capacity test."""
from __future__ import annotations

import statistics
import time

import anyio
import httpx2

from examples.m09_fleet import fleet

REQUESTS, CONCURRENCY = 200, 10
HEADERS = {"Accept": "application/json, text/event-stream", "MCP-Protocol-Version": "2026-07-28",
           "Mcp-Method": "tools/call", "Mcp-Name": "search_notes"}


def body(n: int) -> dict:
    return {"jsonrpc": "2.0", "id": n, "method": "tools/call", "params": {
        "name": "search_notes", "arguments": {"query": "sleep memory focus", "limit": 3},
        "_meta": {"io.modelcontextprotocol/protocolVersion": "2026-07-28",
                  "io.modelcontextprotocol/clientCapabilities": {}}}}


async def load(url: str) -> tuple[list[float], float, int]:
    latencies: list[float] = []
    errors = 0
    queue = iter(range(REQUESTS))
    limits = httpx2.Limits(max_connections=CONCURRENCY)

    async def worker(http: httpx2.AsyncClient) -> None:
        nonlocal errors
        for n in queue:
            started = time.perf_counter()
            response = await http.post(url, json=body(n), headers=HEADERS)
            latencies.append((time.perf_counter() - started) * 1000)
            errors += response.status_code != 200 or '"isError":true' in response.text

    async with httpx2.AsyncClient(limits=limits, timeout=30) as http:
        for n in range(20):  # warm-up, not measured
            await http.post(url, json=body(n), headers=HEADERS)
        started = time.perf_counter()
        async with anyio.create_task_group() as tg:
            for _ in range(CONCURRENCY):
                tg.start_soon(worker, http)
        elapsed = time.perf_counter() - started
    return latencies, elapsed, errors


def report(label: str, latencies: list[float], elapsed: float, errors: int) -> None:
    p95 = statistics.quantiles(latencies, n=20)[-1]
    print(f"{label:<28} median {statistics.median(latencies):6.1f} ms   p95 {p95:6.1f} ms   "
          f"{len(latencies) / elapsed:6.0f} req/s   errors {errors}")


def main() -> None:
    print(f"{REQUESTS} requests, concurrency {CONCURRENCY}")
    with fleet(replicas=1, balancer_port=None) as f:
        report("1 replica, direct", *anyio.run(load, f.url))
    with fleet(replicas=1, balancer_port=8090) as f:
        report("1 replica, via balancer", *anyio.run(load, f.url))
    with fleet(replicas=2, balancer_port=8090) as f:
        report("2 replicas, via balancer", *anyio.run(load, f.url))


if __name__ == "__main__":
    main()

Code explained

  • In simple words: ten workers take 200 requests off a shared queue as fast as they can, and we time each one.

The rest of this course is yours to keep

This course is bought on its own, once, and stays readable afterwards, including the parts added to it later.