5. Hands-on: Chroma, Pinecone, pgvector¶
Intermediate · 13 min read
Every vector database offers the same handful of operations: create a collection, upsert records (id + vector + metadata), query with filters, fetch and delete. Learn them once in Chroma — which runs locally with no server — then see the same steps in Pinecone, pgvector and Qdrant.
5.1 Data and embeddings¶
The same stand-in embedder as in Embeddings in depth — swap in a real model for real use. We pass vectors explicitly, so you control exactly which model is used (Chroma can also embed for you).
import numpy as np
from sklearn.feature_extraction.text import TfidfVectorizer
CHUNKS = [
{"id": "refunds#0", "text": "Refunds are processed within 5 working days of approval.", "category": "refunds", "tenant": "acme"},
{"id": "refunds#1", "text": "Damaged items can be returned for a full refund.", "category": "refunds", "tenant": "acme"},
{"id": "shipping#0", "text": "Orders ship within 48 hours from our Pune warehouse.", "category": "shipping", "tenant": "acme"},
{"id": "shipping#1", "text": "Express delivery is available in metro cities.", "category": "shipping", "tenant": "acme"},
{"id": "account#0", "text": "Reset your password from Settings, then Security.", "category": "account", "tenant": "acme"},
{"id": "globex-refunds#0", "text": "Globex refunds take 10 days.", "category": "refunds", "tenant": "globex"},
]
tfidf = TfidfVectorizer(analyzer="char_wb", ngram_range=(3, 5)).fit([c["text"] for c in CHUNKS])
def embed(texts: list[str]) -> list[list[float]]:
v = tfidf.transform(texts).toarray()
return (v / np.maximum(np.linalg.norm(v, axis=1, keepdims=True), 1e-9)).tolist()
print(len(embed(["hello"])[0]), "dimensions")
5.2 Chroma: create, add, query¶
import chromadb
client = chromadb.EphemeralClient() # in memory; PersistentClient(path="./chroma") saves to disk
collection = client.create_collection(
name="support_docs",
embedding_function=None, # we supply vectors ourselves
configuration={"hnsw": {"space": "cosine"}}, # match the metric to the embedding model
)
collection.add(
ids=[c["id"] for c in CHUNKS],
embeddings=embed([c["text"] for c in CHUNKS]),
documents=[c["text"] for c in CHUNKS],
metadatas=[{"category": c["category"], "tenant": c["tenant"]} for c in CHUNKS],
)
print(collection.count(), "records")
def show(result):
for id_, doc, dist in zip(result["ids"][0], result["documents"][0], result["distances"][0]):
print(f"{1 - dist:.2f} {id_:18} {doc}") # cosine distance → similarity
show(collection.query(query_embeddings=embed(["how long do refunds take"]), n_results=3))
6 records
0.60 globex-refunds#0 Globex refunds take 10 days.
0.23 refunds#0 Refunds are processed within 5 working days of approval.
0.14 refunds#1 Damaged items can be returned for a full refund.
A chunk from another tenant (Globex) came back for an Acme user. That's what metadata filters are for.
5.3 Filters¶
acme_only = {"tenant": "acme"}
show(collection.query(query_embeddings=embed(["how long do refunds take"]), n_results=3, where=acme_only))
print("---")
show(collection.query(
query_embeddings=embed(["shipping and delivery time"]),
n_results=2,
where={"$and": [{"tenant": "acme"}, {"category": {"$in": ["shipping", "refunds"]}}]},
))
0.23 refunds#0 Refunds are processed within 5 working days of approval.
0.14 refunds#1 Damaged items can be returned for a full refund.
0.02 shipping#0 Orders ship within 48 hours from our Pune warehouse.
---
0.39 shipping#1 Express delivery is available in metro cities.
0.11 shipping#0 Orders ship within 48 hours from our Pune warehouse.
Operators like $eq, $ne, $gt, $lt, $in, $and, $or look almost the same in Pinecone and other databases.
In a real app, {"tenant": current_user.tenant} is added by your server to every query.
5.4 Update, fetch, delete¶
collection.upsert( # insert or replace by id
ids=["refunds#0"],
embeddings=embed(["Refunds are processed within 3 working days of approval."]),
documents=["Refunds are processed within 3 working days of approval."],
metadatas=[{"category": "refunds", "tenant": "acme"}],
)
print(collection.get(ids=["refunds#0"])["documents"])
collection.delete(where={"tenant": "globex"}) # e.g. a customer leaves: delete all their data
print(collection.count(), "records after deleting the globex tenant")
['Refunds are processed within 3 working days of approval.']
5 records after deleting the globex tenant
When a document changes, upsert its chunks under the same IDs and delete chunks that no longer exist — Vector DBs in production shows how to compute exactly which.
5.5 The same operations elsewhere¶
# no-run — pip install pinecone ; needs PINECONE_API_KEY
from pinecone import Pinecone, ServerlessSpec
pc = Pinecone() # reads PINECONE_API_KEY
if not pc.has_index("support-docs"):
pc.create_index(name="support-docs", dimension=1536, metric="cosine",
spec=ServerlessSpec(cloud="aws", region="us-east-1"))
index = pc.Index("support-docs")
# one namespace per tenant = isolation + easy per-tenant deletes
index.upsert(namespace="acme", vectors=[
{"id": c["id"], "values": vec, "metadata": {"text": c["text"], "category": c["category"]}}
for c, vec in zip(chunks, vectors)
])
res = index.query(namespace="acme", vector=query_vector, top_k=3, include_metadata=True,
filter={"category": {"$in": ["refunds", "shipping"]}})
for m in res.matches:
print(round(m.score, 2), m.id, m.metadata["text"])
index.delete(namespace="acme", ids=["refunds#1"])
index.delete(namespace="globex", delete_all=True) # drop a whole tenant
Notes: upsert in batches (~100–200 vectors per request); metadata has a size limit per record, so very long chunk text is often stored elsewhere and referenced by ID; serverless indexes bill for storage, reads and writes.
# no-run — PostgreSQL with the pgvector extension; pip install "psycopg[binary]" pgvector
import numpy as np
import psycopg
from pgvector.psycopg import register_vector
conn = psycopg.connect("postgresql://localhost/rag", autocommit=True)
conn.execute("CREATE EXTENSION IF NOT EXISTS vector")
register_vector(conn)
conn.execute("""
CREATE TABLE IF NOT EXISTS chunks (
id TEXT PRIMARY KEY, tenant TEXT NOT NULL, category TEXT,
content TEXT, embedding vector(1536))""")
conn.execute("CREATE INDEX IF NOT EXISTS chunks_hnsw ON chunks USING hnsw (embedding vector_cosine_ops)")
conn.execute("CREATE INDEX IF NOT EXISTS chunks_tenant ON chunks (tenant)")
with conn.cursor() as cur: # upsert
cur.executemany("""
INSERT INTO chunks (id, tenant, category, content, embedding) VALUES (%s, %s, %s, %s, %s)
ON CONFLICT (id) DO UPDATE SET content = EXCLUDED.content, embedding = EXCLUDED.embedding""",
[(c["id"], c["tenant"], c["category"], c["text"], np.array(v)) for c, v in zip(chunks, vectors)])
rows = conn.execute("""
SELECT id, content, 1 - (embedding <=> %s) AS similarity -- <=> is cosine distance
FROM chunks
WHERE tenant = %s AND category = ANY(%s)
ORDER BY embedding <=> %s
LIMIT 3""", (np.array(query_vector), "acme", ["refunds", "shipping"], np.array(query_vector))).fetchall()
The big advantage: it's just SQL — join chunks to your documents, users or permissions tables, use
transactions, and enforce tenants with row-level security. See also SQL for GenAI.
# no-run — pip install qdrant-client ; QdrantClient(":memory:") also runs locally with no server
from qdrant_client import QdrantClient
from qdrant_client.models import (Distance, FieldCondition, Filter, MatchValue,
PointStruct, VectorParams)
client = QdrantClient(url="http://localhost:6333")
client.create_collection("support_docs", vectors_config=VectorParams(size=1536, distance=Distance.COSINE))
client.create_payload_index("support_docs", field_name="tenant", field_schema="keyword") # fast filtering
client.upsert("support_docs", points=[
PointStruct(id=i, vector=v, payload={"chunk_id": c["id"], "tenant": c["tenant"], "text": c["text"]})
for i, (c, v) in enumerate(zip(chunks, vectors)) # ids must be integers or UUIDs
])
hits = client.query_points(
"support_docs", query=query_vector, limit=3,
query_filter=Filter(must=[FieldCondition(key="tenant", match=MatchValue(value="acme"))]),
).points
for h in hits:
print(round(h.score, 2), h.payload["text"])
5.6 Wrap it behind your own interface¶
Your RAG code shouldn't care which database is underneath. A small interface keeps it swappable (and testable with an in-memory fake):
from typing import Protocol
class VectorStore(Protocol):
def upsert(self, records: list[dict]) -> None: ...
def query(self, vector: list[float], k: int, tenant: str, filters: dict | None = None) -> list[dict]: ...
def delete(self, ids: list[str], tenant: str) -> None: ...
class ChromaStore:
def __init__(self, collection):
self.c = collection
def upsert(self, records):
self.c.upsert(ids=[r["id"] for r in records], embeddings=[r["vector"] for r in records],
documents=[r["text"] for r in records], metadatas=[r["metadata"] for r in records])
def query(self, vector, k, tenant, filters=None):
where = {"$and": [{"tenant": tenant}, filters]} if filters else {"tenant": tenant} # tenant ALWAYS applied
r = self.c.query(query_embeddings=[vector], n_results=k, where=where)
return [{"id": i, "text": d, "score": 1 - s}
for i, d, s in zip(r["ids"][0], r["documents"][0], r["distances"][0])]
def delete(self, ids, tenant):
self.c.delete(ids=ids, where={"tenant": tenant})
store: VectorStore = ChromaStore(collection)
for hit in store.query(embed(["password reset"])[0], k=2, tenant="acme"):
print(f"{hit['score']:.2f} {hit['text']}")
0.56 Reset your password from Settings, then Security.
0.02 Refunds are processed within 3 working days of approval.
Note how query requires a tenant and always applies it — no caller can forget the filter. Frameworks like
LangChain and LlamaIndex provide similar vector-store wrappers for dozens of databases.
Practice¶
- Switch to
chromadb.PersistentClient(path="./chroma"), restart Python, and confirm the data is still there. - Add a
PineconeStoreclass with the same three methods, using namespaces for tenants.
Next: Vector DBs in production — keeping the index correct, secure and affordable.