Complete ChromaDB Tutorial: Simple Vector Database for AI
ChromaDB is an open-source vector database designed to store and query embeddings easily. With its intuitive API, ChromaDB is perfect for building RAG applications, semantic search, and recommendation systems.
Why ChromaDB?
ChromaDB Advantages:- Simple API: Easy to learn and use
- Embedded mode: Can run in-memory or persistent
- Multi-modal: Support text, images, embeddings
- Integrations: LangChain, LlamaIndex, OpenAI
- No infrastructure: No server setup required
Installation
pip install chromadb
pip install chromadb-client # For client mode
With sentence-transformers
pip install sentence-transformers
Quick Start
1. Basic Usage
import chromadb
Create client (in-memory)
client = chromadb.Client()
Or persistent storage
client = chromadb.PersistentClient(path="./chromadb")
Create collection
collection = client.createcollection(name="mycollection")
Add documents
collection.add(
documents=["Python is a programming language", "Machine learning is AI"],
metadatas=[{"source": "doc1"}, {"source": "doc2"}],
ids=["id1", "id2"]
)
Query
results = collection.query(
querytexts=["What is Python?"],
nresults=2
)
print(results)
2. With Embeddings
import chromadb
from chromadb.utils import embeddingfunctions
Setup embedding function
sentencetransformeref = embeddingfunctions.SentenceTransformerEmbeddingFunction(
modelname="all-MiniLM-L6-v2"
)
Create collection with embedding function
collection = client.createcollection(
name="docs",
embeddingfunction=sentencetransformeref
)
Add documents (embeddings auto-generated)
collection.add(
documents=["Doc 1 content", "Doc 2 content"],
ids=["1", "2"]
)
Query
results = collection.query(
querytexts=["search query"],
nresults=5
)
Collections
1. Collection Operations
# Create
collection = client.createcollection("mycollection")
Get existing
collection = client.getcollection("mycollection")
Get or create
collection = client.getorcreatecollection("mycollection")
Delete
client.deletecollection("mycollection")
List all
collections = client.listcollections()
Count items
count = collection.count()
2. Collection with Custom Embedding
from chromadb.utils import embeddingfunctions
OpenAI embeddings
openaief = embeddingfunctions.OpenAIEmbeddingFunction(
apikey="your-api-key",
modelname="text-embedding-3-small"
)
Sentence Transformers
stef = embeddingfunctions.SentenceTransformerEmbeddingFunction(
modelname="all-mpnet-base-v2"
)
Hugging Face
hfef = embeddingfunctions.HuggingFaceEmbeddingFunction(
apikey="your-hf-token",
modelname="sentence-transformers/all-MiniLM-L6-v2"
)
collection = client.createcollection(
name="mydocs",
embeddingfunction=openaief,
metadata={"hnsw:space": "cosine"} # Distance metric
)
CRUD Operations
1. Add Documents
# Add with documents (auto-embed)
collection.add(
documents=["text 1", "text 2", "text 3"],
metadatas=[{"source": "a"}, {"source": "b"}, {"source": "c"}],
ids=["1", "2", "3"]
)
Add with pre-computed embeddings
collection.add(
embeddings=[[0.1, 0.2, ...], [0.3, 0.4, ...]],
metadatas=[{"key": "value"}],
ids=["1", "2"]
)
Add with documents and embeddings
collection.add(
documents=["text"],
embeddings=[[0.1, 0.2, ...]],
metadatas=[{"key": "value"}],
ids=["1"]
)
2. Query
# Basic query
results = collection.query(
querytexts=["search query"],
nresults=10
)
Query with embeddings
results = collection.query(
queryembeddings=[[0.1, 0.2, ...]],
nresults=10
)
Query with filter
results = collection.query(
querytexts=["query"],
nresults=5,
where={"source": "doc1"},
wheredocument={"$contains": "keyword"}
)
Include specific fields
results = collection.query(
querytexts=["query"],
include=["documents", "metadatas", "distances", "embeddings"]
)
3. Update and Delete
# Update
collection.update(
ids=["1"],
documents=["updated text"],
metadatas=[{"updated": True}]
)
Upsert (update or insert)
collection.upsert(
ids=["1", "newid"],
documents=["text 1", "new text"],
metadatas=[{"key": "val1"}, {"key": "val2"}]
)
Delete by ID
collection.delete(ids=["1", "2"])
Delete with filter
collection.delete(where={"source": "oldsource"})
4. Get Documents
# Get by ID
results = collection.get(ids=["1", "2"])
Get with filter
results = collection.get(
where={"source": "doc1"},
limit=10
)
Get all
results = collection.get()
Filtering
1. Where Filters (Metadata)
# Equality
results = collection.query(
querytexts=["query"],
where={"category": "tech"}
)
Comparison
results = collection.query(
querytexts=["query"],
where={"price": {"$gt": 100}}
)
Operators: $eq, $ne, $gt, $gte, $lt, $lte
Logical AND
results = collection.query(
querytexts=["query"],
where={
"$and": [
{"category": "tech"},
{"price": {"$lt": 500}}
]
}
)
Logical OR
results = collection.query(
querytexts=["query"],
where={
"$or": [
{"category": "tech"},
{"category": "science"}
]
}
)
IN operator
results = collection.query(
querytexts=["query"],
where={"category": {"$in": ["tech", "science", "ai"]}}
)
2. Where Document Filters
# Contains
results = collection.query(
querytexts=["query"],
wheredocument={"$contains": "machine learning"}
)
Not contains
results = collection.query(
querytexts=["query"],
wheredocument={"$notcontains": "deprecated"}
)
RAG Implementation
import chromadb
from chromadb.utils import embeddingfunctions
from openai import OpenAI
Setup
chromaclient = chromadb.PersistentClient(path="./ragdb")
openaiclient = OpenAI()
ef = embeddingfunctions.OpenAIEmbeddingFunction(
apikey="your-key",
modelname="text-embedding-3-small"
)
collection = chromaclient.getorcreatecollection(
name="knowledgebase",
embeddingfunction=ef
)
Add knowledge
documents = [
"Python was created by Guido van Rossum in 1991.",
"Machine learning is a subset of artificial intelligence.",
"ChromaDB is an open-source vector database.",
]
collection.add(
documents=documents,
ids=[f"doc{i}" for i in range(len(documents))]
)
def ragquery(question: str, nresults: int = 3) -> str:
# Retrieve
results = collection.query(
querytexts=[question],
nresults=nresults
)
context = "\n".join(results["documents"][0])
# Generate
response = openaiclient.chat.completions.create(
model="gpt-4o-mini",
messages=[
{"role": "system", "content": f"Answer based on context:\n{context}"},
{"role": "user", "content": question}
]
)
return response.choices[0].message.content
Usage
answer = ragquery("What is ChromaDB?")
print(answer)
LangChain Integration
from langchaincommunity.vectorstores import Chroma
from langchain
openai import OpenAIEmbeddings
from langchain.textsplitter import RecursiveCharacterTextSplitter
Setup
embeddings = OpenAIEmbeddings()
From documents
textsplitter = RecursiveCharacterTextSplitter(chunksize=1000)
texts = textsplitter.splittext("Your long document here...")
vectorstore = Chroma.fromtexts(
texts,
embeddings,
persistdirectory="./chromalangchain"
)
Query
docs = vectorstore.similaritysearch("query", k=5)
As retriever
retriever = vectorstore.asretriever(searchkwargs={"k": 5})
docs = retriever.invoke("query")
Client-Server Mode
1. Start Server
# Install server
pip install chromadb
Run server
chroma run --host 0.0.0.0 --port 8000 --path ./chromadata
2. Client Connection
import chromadb
HTTP client
client = chromadb.HttpClient(host="localhost", port=8000)
With authentication
client = chromadb.HttpClient(
host="localhost",
port=8000,
headers={"Authorization": "Bearer token"}
)
Usage same as before
collection = client.getorcreatecollection("mycollection")
Best Practices
1. Batching
# Batch add for performance
BATCHSIZE = 100
documents = ["doc1", "doc2", ...] # Many documents
for i in range(0, len(documents), BATCHSIZE):
batch = documents[i:i+BATCHSIZE]
collection.add(
documents=batch,
ids=[f"id{j}" for j in range(i, i+len(batch))]
)
2. Distance Functions
# Cosine similarity (default)
collection = client.createcollection(
name="cosinecollection",
metadata={"hnsw:space": "cosine"}
)
L2 (Euclidean)
collection = client.createcollection(
name="l2collection",
metadata={"hnsw:space": "l2"}
)
Inner product
collection = client.createcollection(
name="ipcollection",
metadata={"hnsw:space": "ip"}
)
Conclusion
ChromaDB is the ideal vector database for:
Key takeaways:
- Use persistent client for production
- Choose appropriate embedding function
- Leverage metadata filtering for precision
- Batch operations for large datasets