Testing MCP servers
The in-memory client is a test fixture that connects to your server in the same process, so each tool's promises become pytest checks.
View the code here
from typing import Annotated, Literal
from pydantic import BaseModel, Field
from mcp.server import MCPServer
from mcp.server.mcpserver.prompts.base import AssistantMessage, Message, UserMessage
from mcp.server.mcpserver.exceptions import ResourceNotFoundError, ToolError
from mcp.types import ToolAnnotations
mcp = MCPServer("Shop support", log_level="WARNING")
ORDERS = {
"A17": {"item": "blue mug", "status": "shipped", "total": 12.50},
"B42": {"item": "desk lamp", "status": "processing", "total": 48.00},
}
ARTICLES = {
"Where is my order?": "orders",
"Changing an order": "orders",
"How refunds work": "refunds",
"Refunds for damaged items": "refunds",
"Resetting your password": "account",
}
class Order(BaseModel):
id: str
item: str
status: Literal["processing", "shipped", "delivered"]
total: float
@mcp.tool(annotations=ToolAnnotations(read_only_hint=True))
def lookup_order(order_id: str) -> Order:
"""Look up an order by its id."""
if order_id not in ORDERS:
raise ToolError(f"Order not found.")
return Order(id=order_id, **ORDERS[order_id])
@mcp.tool()
def search_help(
query: Annotated[str, Field(description="Words to look for in the help articles.")],
topic: Literal["orders", "refunds", "account"] | None = None,
limit: Annotated[int, Field(ge=1, le=5)] = 3,
) -> str:
"""Search the help centre and return matching article titles."""
found = [title for title, t in ARTICLES.items() if query.lower() in title.lower() and topic in (None, t)]
return "; ".join(found[:limit]) or "No articles found."
@mcp.tool(
title="Refund an order",
annotations=ToolAnnotations(read_only_hint=False, destructive_hint=False, idempotent_hint=False),
)
def refund_order(order_id: str, reason: str) -> str:
"""Refund the full total of an order to the customer's card."""
if order_id not in ORDERS:
raise ToolError(f"No order with id {order_id!r}.")
return f"Refunded {ORDERS[order_id]['total']:.2f} for order {order_id}: {reason}."
@mcp.resource("policy://refunds", mime_type="text/markdown")
def refund_policy() -> str:
"""The shop's refund policy."""
return "# Refunds\n\nFull refund within 30 days of delivery. Damaged items: refund or replacement."
@mcp.resource("orders://{order_id}", mime_type="application/json")
def order_record(order_id: str) -> Order:
"""The full record for one order."""
if order_id not in ORDERS:
raise ResourceNotFoundError(f"No order with id {order_id!r}.")
return Order(id=order_id, **ORDERS[order_id])
@mcp.prompt(title="Reply to a customer")
def reply_to_customer(ticket: str, tone: str = "friendly") -> list[Message]:
"""Draft a reply to a support ticket."""
return [
UserMessage(f"Write a {tone} reply to this support ticket:\n\n{ticket}"),
AssistantMessage("Hello, and thank you for getting in touch."),
]
if __name__ == "__main__":
mcp.run(transport="streamable-http", port=8200)
Last updated: 29 Sep, 2026 · MCP 2.2
import pytest
from mcp import Client
from shop import mcp
@pytest.fixture
def anyio_backend():
return "asyncio"
@pytest.fixture
async def client():
async with Client(mcp, raise_exceptions=True) as connected:
yield connectedThe tests are async, so they need a runner. anyio, installed with mcp, includes a pytest plugin: @pytest.mark.anyio runs a test in an event loop, and the anyio_backend fixture picks asyncio.
The client fixture connects once per test. raise_exceptions=True matters only for failures outside a tool, such as a broken server setup: over this connection the SDK would otherwise replace their message with a generic Internal server error, and in a test you want the real one.
@pytest.mark.anyio
async def test_lookup_returns_the_order(client):
result = await client.call_tool("lookup_order", {"order_id": "A17"})
assert result.is_error is False
assert result.structured_content == {"id": "A17", "item": "blue mug", "status": "shipped", "total": 12.5}
@pytest.mark.anyio
async def test_unknown_order_tells_the_model_what_to_send(client):
result = await client.call_tool("lookup_order", {"order_id": "Z9"})
assert result.is_error is True
assert "Order ids look like A17" in result.content[0].text
@pytest.mark.anyio
async def test_search_limit_is_enforced(client):
result = await client.call_tool("search_help", {"query": "refund", "limit": 50})
assert result.is_error is TrueOne test for the promised data, one for the error text a model depends on, one for a limit. Comparing the whole structured_content catches a field that appears, disappears or changes type.
pytest -q... [100%] 3 passed in 0.35s
A test catching a change
Someone edits the error message to a shorter "Order not found.":
raise ToolError(f"Order not found.")pytest -q --tb=short.F. [100%]
================================= FAILURES =================================
_____________ test_unknown_order_tells_the_model_what_to_send ______________
test_shop.py:29: in test_unknown_order_tells_the_model_what_to_send
assert "Order ids look like A17" in result.content[0].text
E AssertionError: assert 'Order ids look like A17' in 'Error executing tool lookup_order: Order not found.'
E + where 'Error executing tool lookup_order: Order not found.' = TextContent(type='text', text='Error executing tool lookup_order: Order not found.', annotations=None, meta=None).text
========================= short test summary info ==========================
FAILED test_shop.py::test_unknown_order_tells_the_model_what_to_send - As...
1 failed, 2 passed in 0.40s- The test failed and named the missing text, the sentence that told a model what a valid order id looks like.
- A message written for a model is behaviour, so it earns a test like any other.
In-memory tests vs a live transport
| Client(mcp) | stdio or HTTP | |
|---|---|---|
| Speed | Fast, no ports | Slower, a real process |
| Catches | Tool logic and errors | Transport and start-up too |
| Use for | Most tests | A few end-to-end checks |
When to test in memory
- Every tool's promised data, error text and limits.
- A change that would break a message a model depends on.
- Fast checks in CI, with a few end-to-end runs over a real transport.
raise_exceptions=True is for the test client only. It surfaces the real error instead of a generic one, which you want in a test and not in production.Related
- Previous: Agent loop
- Next: Integrations
- Add a test that
list_toolsreturns exactlylookup_order,search_helpandrefund_order. - Test that
search_helpwithtopic="refunds"returns only refund articles. - Add a test for a resource read from the Resource templates lesson, using
pytest.raises(MCPError)for a missing order.
This is what real progress feels like.