Topic 7: Measuring it
4 min read·22 Sept 2026
The template for this course asks for numbers, not adjectives. Two measurements matter for a capstone this size: how much context the tool definitions cost on every model call, and what each transport adds to a question. The script measures both with the scripted stand-in, so the times are host plus transport plus server, with no model time at all.
python
"""Measure the assembled assistant: tool-definition size and latency per transport.
Starts its own HTTP server with auth on port 8111. Uses the scripted stand-in
model, so the times are host plus server plus transport, with no LLM time.
"""
from __future__ import annotations
import json
import logging
import os
import socket
import statistics
import subprocess
import sys
import time
from pathlib import Path
import anyio
import httpx2
from mcp import Client, StdioServerParameters
from mcp.client.streamable_http import streamable_http_client
from examples.m11_eval_real import CASES
from examples.m11_scripted_model import ScriptedModel
from notes_assistant.auth import mint_dev_token
from notes_assistant.host import Host
from notes_assistant.server import build_server
from notes_assistant.store import NoteStore
REPO = Path(__file__).resolve().parent.parent
PORT = 8111
URL = f"http://127.0.0.1:{PORT}/mcp"
AUTH = {"NOTES_AUTH_KEY": "measure-key-0123456789abcdef0123456789", "NOTES_AUTH_ISSUER": "http://127.0.0.1:9000", "NOTES_RESOURCE_URL": URL}
ROUNDS = 5
async def time_questions(client: Client) -> tuple[float, int]:
"""Median milliseconds per question over ROUNDS passes of the five answerable questions."""
samples = []
for _ in range(ROUNDS):
for case in CASES[:5]:
started = time.perf_counter()
await Host({"notes": client}, chat_fn=ScriptedModel()).ask(case.question)
samples.append((time.perf_counter() - started) * 1000)
return statistics.median(samples), len(samples)
async def main() -> None:
store = NoteStore(REPO / "notes")
async with Client(build_server(store)) as client:
specs = await Host({"notes": client}).load_tools()
size = len(json.dumps(specs))
print(f"tool definitions sent to the model: {len(specs)} tools, {size} characters, about {size // 4} tokens (estimate: chars / 4)")
rows = []
started = time.perf_counter()
async with Client(build_server(store)) as client:
connect = (time.perf_counter() - started) * 1000
rows.append(("in-memory", connect, *await time_questions(client)))
params = StdioServerParameters(command=sys.executable, args=["-m", "notes_assistant.server"],
env={"PYTHONPATH": str(REPO), "NOTES_DIR": str(REPO / "notes"), "NOTES_LOG_LEVEL": "WARNING"})
started = time.perf_counter()
async with Client(params) as client:
connect = (time.perf_counter() - started) * 1000
rows.append(("stdio", connect, *await time_questions(client)))
env = {**os.environ, **AUTH, "PYTHONPATH": str(REPO), "NOTES_DIR": str(REPO / "notes"),
"NOTES_TRANSPORT": "streamable-http", "NOTES_PORT": str(PORT), "NOTES_LOG_LEVEL": "WARNING"}
server = subprocess.Popen([sys.executable, "-m", "notes_assistant.server"], env=env,
stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL)
try:
while socket.socket().connect_ex(("127.0.0.1", PORT)) != 0:
await anyio.sleep(0.1)
token = mint_dev_token(AUTH["NOTES_AUTH_KEY"], AUTH["NOTES_AUTH_ISSUER"], URL)
async with httpx2.AsyncClient(headers={"Authorization": f"Bearer {token}"}) as http_client:
started = time.perf_counter()
async with Client(streamable_http_client(URL, http_client=http_client)) as client:
connect = (time.perf_counter() - started) * 1000
rows.append(("HTTP + auth", connect, *await time_questions(client)))
finally:
server.terminate()
server.wait()
print(f"\n{'transport':<12} {'connect ms':>10} {'median ms per question':>23} {'questions':>10}")
for name, connect, median, count in rows:
print(f"{name:<12} {connect:>10.0f} {median:>23.1f} {count:>10}")
if __name__ == "__main__":
logging.basicConfig(level=logging.WARNING) # before MCPServer() installs its own INFO handler
anyio.run(main)Code explained