Tutorial Lengkap ChromaDB: Vector Database Sederhana untuk AI
ChromaDB adalah open-source vector database yang dirancang untuk menyimpan dan query embeddings dengan mudah. Dengan API yang intuitif, ChromaDB cocok untuk membangun aplikasi RAG, semantic search, dan recommendation systems.
Mengapa ChromaDB?
Keunggulan ChromaDB:- Simple API: Mudah dipelajari dan digunakan
- Embedded mode: Bisa berjalan in-memory atau persistent
- Multi-modal: Support text, images, embeddings
- Integrations: LangChain, LlamaIndex, OpenAI
- No infrastructure: Tidak perlu setup server
Instalasi
pip install chromadb
pip install chromadb-client # Untuk client mode
Dengan sentence-transformers
pip install sentence-transformers
Quick Start
1. Basic Usage
import chromadb
Create client (in-memory)
client = chromadb.Client()
Atau persistent storage
client = chromadb.PersistentClient(path="./chromadb")
Create collection
collection = client.createcollection(name="mycollection")
Add documents
collection.add(
documents=["Python is a programming language", "Machine learning is AI"],
metadatas=[{"source": "doc1"}, {"source": "doc2"}],
ids=["id1", "id2"]
)
Query
results = collection.query(
querytexts=["What is Python?"],
nresults=2
)
print(results)
2. Dengan Embeddings
import chromadb
from chromadb.utils import embeddingfunctions
Setup embedding function
sentencetransformeref = embeddingfunctions.SentenceTransformerEmbeddingFunction(
modelname="all-MiniLM-L6-v2"
)
Create collection dengan embedding function
collection = client.createcollection(
name="docs",
embeddingfunction=sentencetransformeref
)
Add documents (embeddings auto-generated)
collection.add(
documents=["Doc 1 content", "Doc 2 content"],
ids=["1", "2"]
)
Query
results = collection.query(
querytexts=["search query"],
nresults=5
)
Collections
1. Collection Operations
# Create
collection = client.createcollection("mycollection")
Get existing
collection = client.getcollection("mycollection")
Get or create
collection = client.getorcreatecollection("mycollection")
Delete
client.deletecollection("mycollection")
List all
collections = client.listcollections()
Count items
count = collection.count()
2. Collection dengan Custom Embedding
from chromadb.utils import embeddingfunctions
OpenAI embeddings
openaief = embeddingfunctions.OpenAIEmbeddingFunction(
apikey="your-api-key",
modelname="text-embedding-3-small"
)
Sentence Transformers
stef = embeddingfunctions.SentenceTransformerEmbeddingFunction(
modelname="all-mpnet-base-v2"
)
Hugging Face
hfef = embeddingfunctions.HuggingFaceEmbeddingFunction(
apikey="your-hf-token",
modelname="sentence-transformers/all-MiniLM-L6-v2"
)
collection = client.createcollection(
name="mydocs",
embeddingfunction=openaief,
metadata={"hnsw:space": "cosine"} # Distance metric
)
CRUD Operations
1. Add Documents
# Add dengan documents (auto-embed)
collection.add(
documents=["text 1", "text 2", "text 3"],
metadatas=[{"source": "a"}, {"source": "b"}, {"source": "c"}],
ids=["1", "2", "3"]
)
Add dengan pre-computed embeddings
collection.add(
embeddings=[[0.1, 0.2, ...], [0.3, 0.4, ...]],
metadatas=[{"key": "value"}],
ids=["1", "2"]
)
Add dengan documents dan embeddings
collection.add(
documents=["text"],
embeddings=[[0.1, 0.2, ...]],
metadatas=[{"key": "value"}],
ids=["1"]
)
2. Query
# Basic query
results = collection.query(
querytexts=["search query"],
nresults=10
)
Query dengan embeddings
results = collection.query(
queryembeddings=[[0.1, 0.2, ...]],
nresults=10
)
Query dengan filter
results = collection.query(
querytexts=["query"],
nresults=5,
where={"source": "doc1"},
wheredocument={"$contains": "keyword"}
)
Include specific fields
results = collection.query(
querytexts=["query"],
include=["documents", "metadatas", "distances", "embeddings"]
)
3. Update dan Delete
# Update
collection.update(
ids=["1"],
documents=["updated text"],
metadatas=[{"updated": True}]
)
Upsert (update or insert)
collection.upsert(
ids=["1", "newid"],
documents=["text 1", "new text"],
metadatas=[{"key": "val1"}, {"key": "val2"}]
)
Delete by ID
collection.delete(ids=["1", "2"])
Delete dengan filter
collection.delete(where={"source": "oldsource"})
4. Get Documents
# Get by ID
results = collection.get(ids=["1", "2"])
Get dengan filter
results = collection.get(
where={"source": "doc1"},
limit=10
)
Get all
results = collection.get()
Filtering
1. Where Filters (Metadata)
# Equality
results = collection.query(
querytexts=["query"],
where={"category": "tech"}
)
Comparison
results = collection.query(
querytexts=["query"],
where={"price": {"$gt": 100}}
)
Operators: $eq, $ne, $gt, $gte, $lt, $lte
Logical AND
results = collection.query(
querytexts=["query"],
where={
"$and": [
{"category": "tech"},
{"price": {"$lt": 500}}
]
}
)
Logical OR
results = collection.query(
querytexts=["query"],
where={
"$or": [
{"category": "tech"},
{"category": "science"}
]
}
)
IN operator
results = collection.query(
querytexts=["query"],
where={"category": {"$in": ["tech", "science", "ai"]}}
)
2. Where Document Filters
# Contains
results = collection.query(
querytexts=["query"],
wheredocument={"$contains": "machine learning"}
)
Not contains
results = collection.query(
querytexts=["query"],
wheredocument={"$notcontains": "deprecated"}
)
RAG Implementation
import chromadb
from chromadb.utils import embeddingfunctions
from openai import OpenAI
Setup
chromaclient = chromadb.PersistentClient(path="./ragdb")
openaiclient = OpenAI()
ef = embeddingfunctions.OpenAIEmbeddingFunction(
apikey="your-key",
modelname="text-embedding-3-small"
)
collection = chromaclient.getorcreatecollection(
name="knowledgebase",
embeddingfunction=ef
)
Add knowledge
documents = [
"Python was created by Guido van Rossum in 1991.",
"Machine learning is a subset of artificial intelligence.",
"ChromaDB is an open-source vector database.",
]
collection.add(
documents=documents,
ids=[f"doc{i}" for i in range(len(documents))]
)
def ragquery(question: str, nresults: int = 3) -> str:
# Retrieve
results = collection.query(
querytexts=[question],
nresults=nresults
)
context = "\n".join(results["documents"][0])
# Generate
response = openaiclient.chat.completions.create(
model="gpt-4o-mini",
messages=[
{"role": "system", "content": f"Answer based on context:\n{context}"},
{"role": "user", "content": question}
]
)
return response.choices[0].message.content
Usage
answer = ragquery("What is ChromaDB?")
print(answer)
LangChain Integration
from langchaincommunity.vectorstores import Chroma
from langchain
openai import OpenAIEmbeddings
from langchain.textsplitter import RecursiveCharacterTextSplitter
Setup
embeddings = OpenAIEmbeddings()
From documents
textsplitter = RecursiveCharacterTextSplitter(chunksize=1000)
texts = textsplitter.splittext("Your long document here...")
vectorstore = Chroma.fromtexts(
texts,
embeddings,
persistdirectory="./chromalangchain"
)
Query
docs = vectorstore.similaritysearch("query", k=5)
As retriever
retriever = vectorstore.asretriever(searchkwargs={"k": 5})
docs = retriever.invoke("query")
Client-Server Mode
1. Start Server
# Install server
pip install chromadb
Run server
chroma run --host 0.0.0.0 --port 8000 --path ./chromadata
2. Client Connection
import chromadb
HTTP client
client = chromadb.HttpClient(host="localhost", port=8000)
Dengan authentication
client = chromadb.HttpClient(
host="localhost",
port=8000,
headers={"Authorization": "Bearer token"}
)
Usage sama seperti sebelumnya
collection = client.getorcreatecollection("mycollection")
Best Practices
1. Batching
# Batch add untuk performa
BATCHSIZE = 100
documents = ["doc1", "doc2", ...] # Many documents
for i in range(0, len(documents), BATCHSIZE):
batch = documents[i:i+BATCHSIZE]
collection.add(
documents=batch,
ids=[f"id{j}" for j in range(i, i+len(batch))]
)
2. Distance Functions
# Cosine similarity (default)
collection = client.createcollection(
name="cosinecollection",
metadata={"hnsw:space": "cosine"}
)
L2 (Euclidean)
collection = client.createcollection(
name="l2collection",
metadata={"hnsw:space": "l2"}
)
Inner product
collection = client.createcollection(
name="ipcollection",
metadata={"hnsw:space": "ip"}
)
Kesimpulan
ChromaDB adalah vector database yang ideal untuk:
Key takeaways:
- Gunakan persistent client untuk production
- Pilih embedding function yang sesuai
- Leverage metadata filtering untuk precision
- Batch operations untuk large datasets