Embedding Models

The embedding model is a 10-year mortgage on your data. Choose with the exit cost in mind.

▶ Watch this reel

What you'll learn

  1. Choosing a model
  2. Multilingual needs
  3. Cost & batching
  4. Re-embedding migrations

Remember this

Choosing

Multilingual (EN + HI)

Cost & batching

Migration (re-embedding)

Code: Embedding selection + backfill, done right

import numpy as np
from openai import OpenAI

client = OpenAI()

# --- 1. Your relevance eval (the ONLY decision metric) ------------
GOLDEN = [
    ("refund policy for damaged goods", "policy_returns_v2.pdf#3"),
    ("remote work allowance rules", "hr_policy_2026.pdf#7"),
]
DOCS = {id_: text for id_, text in load_corpus().items()}

def hit_at5(model: str) -> float:
    doc_ids = list(DOCS)
    doc_vecs = embed_all(DOCS.values(), model=model)     # batched
    hits = 0
    for q, want in GOLDEN:
        qv = embed_all([q], model=model)[0]
        sims = doc_vecs @ (qv / np.linalg.norm(qv))
        top5 = [doc_ids[i] for i in np.argsort(sims)[::-1][:5]]
        hits += want in top5
    return hits / len(GOLDEN)

for m in ["text-embedding-3-small", "text-embedding-3-large"]:
    print(m, hit_at5(m))     # pick the cheapest model at hit@5 == 1.0

# --- 2. Bulk backfill: batch + version ---------------------------
def embed_all(texts, model, batch=1000):
    vecs = []
    texts = list(texts)
    for i in range(0, len(texts), batch):
        r = client.embeddings.create(model=model, input=texts[i:i+batch])
        vecs.extend(d.embedding for d in r.data)
    return np.array(vecs)

# store per record: {text, vector, embed_model: "text-embedding-3-small",
#                    embed_version: 1, dims: 1536}