# Testing Your Integration With a Stand-In Server — Claude API / OpenAI API Basics

Source: https://www.geekswithgeeks.com/en/llm-apis/p-test

> Run SDK code offline against a local fake to test retries, errors, streaming and tools.

## Test your code, not the model

Calling the real API in every test is slow, costs money, needs keys in CI and is non-deterministic. Instead test the **integration** around the model with a **fake server** or a **mocked client**: point the SDK's `base_url` at a local server that returns canned replies. You can then assert on exactly what your code **sends** (headers, body, model name, `max_tokens`), simulate **429s, 5xx and timeouts** to check retries and fallbacks, replay **streams** and **tool-use** exchanges, and run it all in CI for free. Keep a **separate small suite** against the real model (an evaluation set run on a schedule or before releases) to measure quality, because fakes cannot tell you whether answers are good. All the outputs in this course were produced this way.

## The stand-in server (save as mock.py)

A small server using only the Python standard library. It returns canned replies for the Anthropic Messages shape and the OpenAI Chat Completions shape, can answer 429 on demand, streams server-sent events, and records every request in `STATE["log"]`. It is not a model and does not validate everything a real service would.

```python
"""A tiny local stand-in for the Anthropic Messages API and the OpenAI Chat Completions API.
It returns canned replies, so SDK behaviour (headers, retries, errors, streaming, tools) can be
tested offline. It is NOT a model."""
import json, threading
from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer

STATE = {"fail_next": 0, "log": []}

def start(port=0):
    class H(BaseHTTPRequestHandler):
        def log_message(self, *a): pass
        def _send(self, code, obj, headers=None):
            body = json.dumps(obj).encode()
            self.send_response(code)
            self.send_header("content-type", "application/json")
            for k, v in (headers or {}).items(): self.send_header(k, v)
            self.send_header("content-length", str(len(body)))
            self.end_headers(); self.wfile.write(body)
        def do_POST(self):
            n = int(self.headers.get("content-length", 0))
            body = json.loads(self.rfile.read(n) or b"{}")
            STATE["log"].append({"path": self.path, "headers": {k.lower(): v for k, v in self.headers.items()}, "body": body})
            if STATE["fail_next"] > 0:
                STATE["fail_next"] -= 1
                return self._send(429, {"type": "error", "error": {"type": "rate_limit_error", "message": "slow down"}},
                                  {"retry-after": "0"})
            if self.path == "/v1/messages":
                return self.anthropic(body)
            if self.path == "/v1/chat/completions":
                return self.openai(body)
            self._send(404, {"error": {"message": "not found"}})
        def anthropic(self, b):
            if not b.get("max_tokens"):
                return self._send(400, {"type": "error", "error": {"type": "invalid_request_error", "message": "max_tokens: Field required"}})
            if b.get("stream"):
                self.send_response(200); self.send_header("content-type", "text/event-stream"); self.end_headers()
                def ev(name, data): self.wfile.write(f"event: {name}\ndata: {json.dumps(data)}\n\n".encode())
                ev("message_start", {"type": "message_start", "message": {"id": "msg_1", "type": "message", "role": "assistant", "model": b["model"], "content": [], "stop_reason": None, "usage": {"input_tokens": 12, "output_tokens": 1}}})
                ev("content_block_start", {"type": "content_block_start", "index": 0, "content_block": {"type": "text", "text": ""}})
                for piece in ["Hel", "lo ", "there"]:
                    ev("content_block_delta", {"type": "content_block_delta", "index": 0, "delta": {"type": "text_delta", "text": piece}})
                ev("content_block_stop", {"type": "content_block_stop", "index": 0})
                ev("message_delta", {"type": "message_delta", "delta": {"stop_reason": "end_turn"}, "usage": {"output_tokens": 3}})
                ev("message_stop", {"type": "message_stop"})
                return
            if b.get("tools"):
                last = b["messages"][-1]
                if last["role"] == "user" and isinstance(last["content"], str):
                    return self._send(200, {"id": "msg_2", "type": "message", "role": "assistant", "model": b["model"],
                        "content": [{"type": "tool_use", "id": "toolu_1", "name": b["tools"][0]["name"], "input": {"order_id": "481516"}}],
                        "stop_reason": "tool_use", "usage": {"input_tokens": 40, "output_tokens": 20}})
                return self._send(200, {"id": "msg_3", "type": "message", "role": "assistant", "model": b["model"],
                    "content": [{"type": "text", "text": "Your order 481516 has shipped."}],
                    "stop_reason": "end_turn", "usage": {"input_tokens": 70, "output_tokens": 9}})
            self._send(200, {"id": "msg_0", "type": "message", "role": "assistant", "model": b["model"],
                "content": [{"type": "text", "text": "Paris is the capital of France."}],
                "stop_reason": "end_turn", "usage": {"input_tokens": 14, "output_tokens": 8}})
        def openai(self, b):
            if b.get("stream"):
                self.send_response(200); self.send_header("content-type", "text/event-stream"); self.end_headers()
                for piece in ["Hel", "lo ", "there"]:
                    c = {"id": "c1", "object": "chat.completion.chunk", "created": 1, "model": b["model"], "choices": [{"index": 0, "delta": {"content": piece}, "finish_reason": None}]}
                    self.wfile.write(f"data: {json.dumps(c)}\n\n".encode())
                end = {"id": "c1", "object": "chat.completion.chunk", "created": 1, "model": b["model"], "choices": [{"index": 0, "delta": {}, "finish_reason": "stop"}]}
                self.wfile.write(f"data: {json.dumps(end)}\n\ndata: [DONE]\n\n".encode()); return
            if b.get("tools") and b["messages"][-1]["role"] == "user":
                msg = {"role": "assistant", "content": None, "tool_calls": [{"id": "call_1", "type": "function",
                       "function": {"name": b["tools"][0]["function"]["name"], "arguments": json.dumps({"order_id": "481516"})}}]}
                fr = "tool_calls"
            elif b.get("tools"):
                msg, fr = {"role": "assistant", "content": "Your order 481516 has shipped."}, "stop"
            else:
                msg, fr = {"role": "assistant", "content": "Paris is the capital of France."}, "stop"
            self._send(200, {"id": "chatcmpl-1", "object": "chat.completion", "created": 1, "model": b["model"],
                "choices": [{"index": 0, "message": msg, "finish_reason": fr}],
                "usage": {"prompt_tokens": 14, "completion_tokens": 8, "total_tokens": 22}})
    srv = ThreadingHTTPServer(("127.0.0.1", port), H)
    threading.Thread(target=srv.serve_forever, daemon=True).start()
    return srv

```

## A test for retries (illustrative)

Uses the stand-in server and asserts on what the SDK sent. Written in pytest style; not run as a test here.

```python
import mock, anthropic

def test_retries_once_on_429_then_succeeds():
    srv = mock.start(); base = f"http://127.0.0.1:{srv.server_address[1]}"
    mock.STATE["log"].clear(); mock.STATE["fail_next"] = 1
    client = anthropic.Anthropic(api_key="test", base_url=base, max_retries=2)
    msg = client.messages.create(model="m", max_tokens=10, messages=[{"role": "user", "content": "hi"}])
    assert msg.stop_reason == "end_turn"
    assert len(mock.STATE["log"]) == 2            # first request got 429, second succeeded
    assert mock.STATE["log"][-1]["body"]["max_tokens"] == 10
```

## Test the unhappy paths

Make the stand-in server fail, stall and send malformed data, and check your code degrades gracefully.

**Quiz:** Why test with a fake server instead of the real API in unit tests?

- [ ] Real APIs are forbidden in tests
- [ ] Fakes measure answer quality
- [x] Tests are fast, free, deterministic and need no keys
- [ ] Fakes train the model

*Answer:* Tests are fast, free, deterministic and need no keys. Use fakes for your integration logic and a separate eval set for quality.
