API Basics

Every LLM app is a loop: build messages, call, stream, count, retry. Master the loop — skip the pain.

▶ Watch this reel

What you'll learn

  1. Chat API anatomy
  2. Streaming responses
  3. Token counting & cost
  4. Rate limits & retries
  5. Official SDKs
  6. Batch API

Remember this

Chat API anatomy

Streaming

Tokens & cost

Rate limits & retries

SDKs

Batch API

Code: The production call loop — streaming, usage, retry

import random, time
from openai import OpenAI, RateLimitError

client = OpenAI()

def chat(messages, model="gpt-4o-mini", max_retries=4, **kw):
    """Call with retry/backoff + usage logging. The only wrapper you need."""
    for attempt in range(max_retries):
        try:
            return client.chat.completions.create(
                model=model, messages=messages,
                stream_options={"include_usage": True},
                **kw,
            )
        except RateLimitError as e:
            if attempt == max_retries - 1:
                raise
            wait = (2 ** attempt) + random.random()   # backoff + jitter
            ra = e.response.headers.get("Retry-After")
            if ra: wait = float(ra)
            print(f"429 — retry {attempt + 1} in {wait:.1f}s")
            time.sleep(wait)

# --- streaming render ------------------------------------------------
stream = chat(
    [{"role": "user", "content": "Explain chunking in 2 sentences"}],
    stream=True, max_tokens=120,
)
full = ""
for chunk in stream:
    piece = chunk.choices[0].delta.content or ""
    print(piece, end="", flush=True)   # render live
    full += piece
    if chunk.usage:
        print(f"\n[usage] {chunk.usage}")   # tokens for cost logging

# --- batch for bulk (sketch) -----------------------------------------
# requests.jsonl: {"custom_id":"row-1","method":"POST","url":"/v1/chat/completions","body":{...}}
# batch = client.batches.create(input_file_id=uploaded.id, endpoint="/v1/chat/completions", completion_window="24h")