HTTP Clients

Every LLM call you make travels over HTTP. Auth headers, retries on 429s, pagination, streaming — the client layer decides whether your app is robust or fragile.

▶ Watch this reel

What you'll learn

  1. requests vs httpx
  2. Auth & headers
  3. Retries & backoff
  4. Pagination & streaming

Remember this

requests vs httpx

Auth

Retries

Pagination & streaming

Code: The resilient LLM client, assembled

import os, httpx, tenacity

client = httpx.AsyncClient(
    base_url="https://api.llm-provider.com/v1",
    headers={"Authorization": f"Bearer {os.environ['LLM_KEY']}"},
    timeout=httpx.Timeout(60.0, connect=5.0),
    limits=httpx.Limits(max_connections=100),  # pool cap
)

def transient(e: BaseException) -> bool:
    return isinstance(e, httpx.HTTPStatusError) \
        and e.response.status_code in (429, 500, 502, 503)

@tenacity.retry(retry=tenacity.retry_if_exception(transient),
                wait=tenacity.wait_exponential(1, max=30)
                     + tenacity.wait_random(0, 1),
                stop=tenacity.stop_after_attempt(5), reraise=True)
async def complete(prompt: str) -> str:
    r = await client.post("/chat", json={"prompt": prompt})
    r.raise_for_status()
    return r.json()["text"]

async def complete_stream(prompt: str):
    async with client.stream("POST", "/chat",
                             json={"prompt": prompt}) as r:
        async for line in r.aiter_lines():
            if line.startswith("data: ") and line != "data: [DONE]":
                yield json.loads(line[6:])["token"]

# One client, two personalities: buffered complete + live stream