Lesson 25 / 27
Testing Your Integration With a Stand-In Server
Run SDK code offline against a local fake to test retries, errors, streaming and tools.
Test your code, not the model
Calling the real API in every test is slow, costs money, needs keys in CI and is non-deterministic. Instead test the integration around the model with a fake server or a mocked client: point the SDK's base_url at a local server that returns canned replies. You can then assert on exactly what your code sends (headers, body, model name, max_tokens), simulate 429s, 5xx and timeouts to check retries and fallbacks, replay streams and tool-use exchanges, and run it all in CI for free. Keep a separate small suite against the real model (an evaluation set run on a schedule or before releases) to measure quality, because fakes cannot tell you whether answers are good. All the outputs in this course were produced this way.
The stand-in server (save as mock.py)
A small server using only the Python standard library. It returns canned replies for the Anthropic Messages shape and the OpenAI Chat Completions shape, can answer 429 on demand, streams server-sent events, and records every request in STATE["log"]. It is not a model and does not validate everything a real service would.
"""A tiny local stand-in for the Anthropic Messages API and the OpenAI Chat Completions API.
It returns canned replies, so SDK behaviour (headers, retries, errors, streaming, tools) can be
tested offline. It is NOT a model."""
import json, threading
from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer
STATE = {"fail_next": 0, "log": []}
def start(port=0):
class H(BaseHTTPRequestHandler):
def log_message(self, *a): pass
def _send(self, code, obj, headers=None):
body = json.dumps(obj).encode()
self.send_response(code)
self.send_header("content-type", "application/json")
for k, v in (headers or {}).items(): self.send_header(k, v)
self.send_header("content-length", str(len(body)))
self.end_headers(); self.wfile.write(body)
def do_POST(self):
n = int(self.headers.get("content-length", 0))
body = json.loads(self.rfile.read(n) or b"{}")
STATE["log"].append({"path": self.path, "headers": {k.lower(): v for k, v in self.headers.items()}, "body": body})
if STATE["fail_next"] > 0:
STATE["fail_next"] -= 1
return self._send(429, {"type": "error", "error": {"type": "rate_limit_error", "message": "slow down"}},
{"retry-after": "0"})
if self.path == "/v1/messages":
return self.anthropic(body)
if self.path == "/v1/chat/completions":
return self.openai(body)
self._send(404, {"error": {"message": "not found"}})
def anthropic(self, b):
if not b.get("max_tokens"):
return self._send(400, {"type": "error", "error": {"type": "invalid_request_error", "message": "max_tokens: Field required"}})
if b.get("stream"):
self.send_response(200); self.send_header("content-type", "text/event-stream"); self.end_headers()
def ev(name, data): self.wfile.write(f"event: {name}\ndata: {json.dumps(data)}\n\n".encode())
ev("message_start", {"type": "message_start", "message": {"id": "msg_1", "type": "message", "role": "assistant", "model": b["model"], "content": [], "stop_reason": None, "usage": {"input_tokens": 12, "output_tokens": 1}}})
ev("content_block_start", {"type": "content_block_start", "index": 0, "content_block": {"type": "text", "text": ""}})
for piece in ["Hel", "lo ", "there"]:
ev("content_block_delta", {"type": "content_block_delta", "index": 0, "delta": {"type": "text_delta", "text": piece}})
ev("content_block_stop", {"type": "content_block_stop", "index": 0})
ev("message_delta", {"type": "message_delta", "delta": {"stop_reason": "end_turn"}, "usage": {"output_tokens": 3}})
ev("message_stop", {"type": "message_stop"})
return
if b.get("tools"):
last = b["messages"][-1]
if last["role"] == "user" and isinstance(last["content"], str):
return self._send(200, {"id": "msg_2", "type": "message", "role": "assistant", "model": b["model"],
"content": [{"type": "tool_use", "id": "toolu_1", "name": b["tools"][0]["name"], "input": {"order_id": "481516"}}],
"stop_reason": "tool_use", "usage": {"input_tokens": 40, "output_tokens": 20}})
return self._send(200, {"id": "msg_3", "type": "message", "role": "assistant", "model": b["model"],
"content": [{"type": "text", "text": "Your order 481516 has shipped."}],
"stop_reason": "end_turn", "usage": {"input_tokens": 70, "output_tokens": 9}})
self._send(200, {"id": "msg_0", "type": "message", "role": "assistant", "model": b["model"],
"content": [{"type": "text", "text": "Paris is the capital of France."}],
"stop_reason": "end_turn", "usage": {"input_tokens": 14, "output_tokens": 8}})
def openai(self, b):
if b.get("stream"):
self.send_response(200); self.send_header("content-type", "text/event-stream"); self.end_headers()
for piece in ["Hel", "lo ", "there"]:
c = {"id": "c1", "object": "chat.completion.chunk", "created": 1, "model": b["model"], "choices": [{"index": 0, "delta": {"content": piece}, "finish_reason": None}]}
self.wfile.write(f"data: {json.dumps(c)}\n\n".encode())
end = {"id": "c1", "object": "chat.completion.chunk", "created": 1, "model": b["model"], "choices": [{"index": 0, "delta": {}, "finish_reason": "stop"}]}
self.wfile.write(f"data: {json.dumps(end)}\n\ndata: [DONE]\n\n".encode()); return
if b.get("tools") and b["messages"][-1]["role"] == "user":
msg = {"role": "assistant", "content": None, "tool_calls": [{"id": "call_1", "type": "function",
"function": {"name": b["tools"][0]["function"]["name"], "arguments": json.dumps({"order_id": "481516"})}}]}
fr = "tool_calls"
elif b.get("tools"):
msg, fr = {"role": "assistant", "content": "Your order 481516 has shipped."}, "stop"
else:
msg, fr = {"role": "assistant", "content": "Paris is the capital of France."}, "stop"
self._send(200, {"id": "chatcmpl-1", "object": "chat.completion", "created": 1, "model": b["model"],
"choices": [{"index": 0, "message": msg, "finish_reason": fr}],
"usage": {"prompt_tokens": 14, "completion_tokens": 8, "total_tokens": 22}})
srv = ThreadingHTTPServer(("127.0.0.1", port), H)
threading.Thread(target=srv.serve_forever, daemon=True).start()
return srv
A test for retries (illustrative)
Uses the stand-in server and asserts on what the SDK sent. Written in pytest style; not run as a test here.
import mock, anthropic
def test_retries_once_on_429_then_succeeds():
srv = mock.start(); base = f"http://127.0.0.1:{srv.server_address[1]}"
mock.STATE["log"].clear(); mock.STATE["fail_next"] = 1
client = anthropic.Anthropic(api_key="test", base_url=base, max_retries=2)
msg = client.messages.create(model="m", max_tokens=10, messages=[{"role": "user", "content": "hi"}])
assert msg.stop_reason == "end_turn"
assert len(mock.STATE["log"]) == 2 # first request got 429, second succeeded
assert mock.STATE["log"][-1]["body"]["max_tokens"] == 10Test the unhappy paths
Make the stand-in server fail, stall and send malformed data, and check your code degrades gracefully.
Quick check: Why test with a fake server instead of the real API in unit tests?
- Real APIs are forbidden in tests
- Fakes measure answer quality
- Tests are fast, free, deterministic and need no keys
- Fakes train the model
Answer
Tests are fast, free, deterministic and need no keys — Use fakes for your integration logic and a separate eval set for quality.