Pendahuluan
MLflow adalah platform open-source untuk mengelola end-to-end machine learning lifecycle. Dikembangkan oleh Databricks, MLflow membantu data scientist dan ML engineer untuk tracking experiments, packaging code, managing models, dan deploying ke production.
Mengapa MLflow?- Reproducibility: Track semua experiment dengan detail
- Collaboration: Share results dengan team
- Model versioning: Kelola berbagai versi model
- Deployment ready: Deploy model dengan mudah ke berbagai platform
- Framework agnostic: Bekerja dengan TensorFlow, PyTorch, Scikit-learn, dll
Komponen Utama MLflow
MLflow terdiri dari 4 komponen utama:
Instalasi dan Setup
Instalasi Dasar
# Install MLflow
pip install mlflow
Install dengan extras untuk berbagai backend
pip install mlflow[extras]
Verify instalasi
mlflow --version
Setup Database Backend (PostgreSQL)
Untuk production, gunakan database backend:
# Install dependencies
pip install psycopg2-binary
Setup PostgreSQL (contoh menggunakan Docker)
docker run -d \
--name mlflow-db \
-e POSTGRESUSER=mlflow \
-e POSTGRESPASSWORD=mlflow \
-e POSTGRESDB=mlflow \
-p 5432:5432 \
postgres:13
Setup Artifact Store (MinIO/S3)
# Install boto3 untuk S3 compatibility
pip install boto3
Setup MinIO (S3-compatible storage)
docker run -d \
--name mlflow-minio \
-p 9000:9000 \
-p 9001:9001 \
-e MINIOROOTUSER=minioadmin \
-e MINIOROOTPASSWORD=minioadmin \
minio/minio server /data --console-address ":9001"
Jalankan MLflow Server
# Development mode (local file store)
mlflow server --host 0.0.0.0 --port 5000
Production mode (dengan database dan S3)
mlflow server \
--backend-store-uri postgresql://mlflow:mlflow@localhost:5432/mlflow \
--default-artifact-root s3://mlflow-artifacts \
--host 0.0.0.0 \
--port 5000
Setup Environment Variables
Buat file .env:
# MLflow Tracking
MLFLOWTRACKINGURI=http://localhost:5000
S3/MinIO Configuration
AWSACCESSKEYID=minioadmin
AWSSECRETACCESSKEY=minioadmin
MLFLOWS3ENDPOINTURL=http://localhost:9000
MLflow Tracking: Experiment Tracking
Basic Tracking
import mlflow
import mlflow.sklearn
from sklearn.ensemble import RandomForestClassifier
from sklearn.datasets import loadiris
from sklearn.modelselection import traintestsplit
from sklearn.metrics import accuracyscore, f1score
Set tracking URI
mlflow.settrackinguri("http://localhost:5000")
Set experiment
mlflow.setexperiment("iris-classification")
Load data
iris = loadiris()
Xtrain, Xtest, ytrain, ytest = traintestsplit(
iris.data, iris.target, testsize=0.2, randomstate=42
)
Start MLflow run
with mlflow.startrun(runname="random-forest-v1") as run:
# Log parameters
params = {
"nestimators": 100,
"maxdepth": 5,
"randomstate": 42
}
mlflow.logparams(params)
# Train model
model = RandomForestClassifier(params)
model.fit(Xtrain, ytrain)
# Make predictions
ypred = model.predict(Xtest)
# Log metrics
metrics = {
"accuracy": accuracyscore(ytest, ypred),
"f1score": f1score(ytest, ypred, average="weighted")
}
mlflow.logmetrics(metrics)
# Log model
mlflow.sklearn.logmodel(
model,
"model",
registeredmodelname="iris-classifier"
)
# Log additional artifacts
import matplotlib.pyplot as plt
from sklearn.metrics import confusionmatrix
import seaborn as sns
# Create confusion matrix
cm = confusionmatrix(ytest, ypred)
plt.figure(figsize=(8, 6))
sns.heatmap(cm, annot=True, fmt="d", cmap="Blues")
plt.title("Confusion Matrix")
plt.ylabel("True Label")
plt.xlabel("Predicted Label")
plt.savefig("confusionmatrix.png")
mlflow.logartifact("confusionmatrix.png")
# Log tags
mlflow.settags({
"modeltype": "RandomForest",
"dataset": "iris",
"environment": "development"
})
print(f"Run ID: {run.info.runid}")
print(f"Accuracy: {metrics['accuracy']:.4f}")
Autologging
MLflow menyediakan autologging untuk framework populer:
import mlflow
import mlflow.sklearn
from sklearn.ensemble import GradientBoostingClassifier
Enable autologging untuk sklearn
mlflow.sklearn.autolog()
mlflow.setexperiment("iris-autolog")
with mlflow.startrun():
model = GradientBoostingClassifier(
nestimators=100,
learningrate=0.1,
maxdepth=3
)
model.fit(Xtrain, ytrain)
# Metrics, parameters, dan model otomatis di-log!
score = model.score(Xtest, ytest)
print(f"Test accuracy: {score:.4f}")
Tracking dengan PyTorch
import torch
import torch.nn as nn
import mlflow.pytorch
mlflow.pytorch.autolog()
class SimpleNet(nn.Module):
def init(self, inputsize, hiddensize, numclasses):
super(SimpleNet, self).init()
self.fc1 = nn.Linear(inputsize, hiddensize)
self.relu = nn.ReLU()
self.fc2 = nn.Linear(hiddensize, numclasses)
def forward(self, x):
out = self.fc1(x)
out = self.relu(out)
out = self.fc2(out)
return out
mlflow.setexperiment("pytorch-mnist")
with mlflow.startrun():
# Model configuration
inputsize = 784
hiddensize = 128
numclasses = 10
learningrate = 0.001
numepochs = 5
# Log hyperparameters
mlflow.logparams({
"inputsize": inputsize,
"hiddensize": hiddensize,
"learningrate": learningrate,
"numepochs": numepochs
})
model = SimpleNet(inputsize, hiddensize, numclasses)
criterion = nn.CrossEntropyLoss()
optimizer = torch.optim.Adam(model.parameters(), lr=learningrate)
# Training loop
for epoch in range(numepochs):
# Training code...
loss = 0.0 # placeholder
# Log metrics per epoch
mlflow.logmetric("trainloss", loss, step=epoch)
# Log model
mlflow.pytorch.logmodel(model, "model")
Nested Runs untuk Hyperparameter Tuning
import mlflow
from sklearn.modelselection import GridSearchCV
from sklearn.ensemble import RandomForestClassifier
mlflow.setexperiment("iris-hyperparameter-tuning")
Parameter grid
paramgrid = {
"nestimators": [50, 100, 200],
"maxdepth": [3, 5, 7],
"minsamplessplit": [2, 5, 10]
}
with mlflow.startrun(runname="grid-search-parent") as parentrun:
mlflow.logparam("searchmethod", "gridsearch")
for nest in paramgrid["nestimators"]:
for maxd in paramgrid["maxdepth"]:
for minsplit in paramgrid["minsamplessplit"]:
with mlflow.startrun(nested=True, runname=f"rf-{nest}-{maxd}-{minsplit}"):
params = {
"nestimators": nest,
"maxdepth": maxd,
"minsamplessplit": minsplit
}
mlflow.logparams(params)
model = RandomForestClassifier(params, randomstate=42)
model.fit(Xtrain, ytrain)
accuracy = model.score(Xtest, ytest)
mlflow.logmetric("accuracy", accuracy)
print(f"Params: {params}, Accuracy: {accuracy:.4f}")
MLflow Projects
MLflow Projects adalah format untuk packaging ML code.
Struktur Project
my-ml-project/
├── MLproject
├── conda.yaml
├── train.py
├── predict.py
└── data/
└── dataset.csv
File MLproject
name: iris-classifier-project
condaenv: conda.yaml
entrypoints:
main:
parameters:
nestimators: {type: int, default: 100}
maxdepth: {type: int, default: 5}
learningrate: {type: float, default: 0.1}
command: "python train.py --n-estimators {nestimators} --max-depth {maxdepth} --learning-rate {learningrate}"
predict:
parameters:
modeluri: {type: str}
inputdata: {type: str}
command: "python predict.py --model-uri {modeluri} --input-data {inputdata}"
File conda.yaml
name: iris-env
channels:
- defaults
- conda-forge
dependencies:
- python=3.9
- pip
- pip:
- mlflow>=2.0.0
- scikit-learn>=1.0.0
- pandas>=1.3.0
- numpy>=1.21.0
File train.py
import argparse
import mlflow
import mlflow.sklearn
from sklearn.ensemble import GradientBoostingClassifier
from sklearn.datasets import loadiris
from sklearn.modelselection import traintestsplit
def train(nestimators, maxdepth, learningrate):
mlflow.setexperiment("iris-project")
with mlflow.startrun():
# Log parameters
mlflow.logparams({
"nestimators": nestimators,
"maxdepth": maxdepth,
"learningrate": learningrate
})
# Load data
iris = loadiris()
Xtrain, Xtest, ytrain, ytest = traintestsplit(
iris.data, iris.target, testsize=0.2, randomstate=42
)
# Train model
model = GradientBoostingClassifier(
nestimators=nestimators,
maxdepth=maxdepth,
learningrate=learningrate,
randomstate=42
)
model.fit(Xtrain, ytrain)
# Evaluate
accuracy = model.score(Xtest, ytest)
mlflow.logmetric("accuracy", accuracy)
# Save model
mlflow.sklearn.logmodel(
model,
"model",
registeredmodelname="iris-gb-classifier"
)
print(f"Model trained with accuracy: {accuracy:.4f}")
return mlflow.activerun().info.runid
if name == "main":
parser = argparse.ArgumentParser()
parser.addargument("--n-estimators", type=int, default=100)
parser.addargument("--max-depth", type=int, default=5)
parser.addargument("--learning-rate", type=float, default=0.1)
args = parser.parseargs()
train(args.nestimators, args.maxdepth, args.learningrate)
Menjalankan MLflow Project
# Run dari direktori lokal
mlflow run . -P nestimators=200 -P maxdepth=7
Run dari Git repository
mlflow run https://github.com/username/ml-project.git -P nestimators=150
Run dengan environment variables
mlflow run . --env-manager=local -P maxdepth=10
MLflow Models
Model Flavors
MLflow mendukung berbagai format model:
import mlflow
Scikit-learn
mlflow.sklearn.logmodel(sklearnmodel, "model")
PyTorch
mlflow.pytorch.logmodel(pytorchmodel, "model")
TensorFlow
mlflow.tensorflow.logmodel(tfmodel, "model")
Keras
mlflow.keras.logmodel(kerasmodel, "model")
XGBoost
mlflow.xgboost.logmodel(xgbmodel, "model")
LightGBM
mlflow.lightgbm.logmodel(lgbmodel, "model")
Custom Python Model
import mlflow.pyfunc
import pandas as pd
class IrisPredictor(mlflow.pyfunc.PythonModel):
def init(self, preprocessor, model):
self.preprocessor = preprocessor
self.model = model
def predict(self, context, modelinput):
# Custom preprocessing
processed = self.preprocessor.transform(modelinput)
# Prediction
predictions = self.model.predict(processed)
# Custom postprocessing
classnames = ["setosa", "versicolor", "virginica"]
return pd.DataFrame({
"prediction": predictions,
"classname": [classnames[p] for p in predictions]
})
Log custom model
with mlflow.startrun():
custommodel = IrisPredictor(preprocessor, trainedmodel)
mlflow.pyfunc.logmodel(
artifactpath="custom-model",
pythonmodel=custommodel,
condaenv={
"name": "custom-env",
"channels": ["defaults"],
"dependencies": [
"python=3.9",
"pip",
{"pip": ["scikit-learn", "pandas"]}
]
}
)
Model Signature
from mlflow.models.signature import infersignature
import pandas as pd
Infer signature dari data
inputexample = Xtrain[:5]
signature = infersignature(inputexample, model.predict(inputexample))
Log model dengan signature
mlflow.sklearn.logmodel(
model,
"model",
signature=signature,
inputexample=inputexample
)
Loading dan Prediksi
import mlflow.pyfunc
Load model by runid
modeluri = f"runs:/{runid}/model"
loadedmodel = mlflow.pyfunc.loadmodel(modeluri)
Predict
import pandas as pd
data = pd.DataFrame({
"sepallength": [5.1, 6.2],
"sepalwidth": [3.5, 2.8],
"petallength": [1.4, 4.8],
"petalwidth": [0.2, 1.8]
})
predictions = loadedmodel.predict(data)
print(predictions)
Load dari model registry
modeluri = "models:/iris-classifier/Production"
productionmodel = mlflow.pyfunc.loadmodel(modeluri)
predictions = productionmodel.predict(data)
MLflow Model Registry
Model Registry adalah centralized model store untuk versioning dan lifecycle management.
Registrasi Model
import mlflow
from mlflow.tracking import MlflowClient
client = MlflowClient()
Method 1: Register saat logmodel
with mlflow.startrun():
mlflow.sklearn.logmodel(
model,
"model",
registeredmodelname="iris-random-forest"
)
Method 2: Register existing run
runid = "abc123def456"
modeluri = f"runs:/{runid}/model"
modelname = "iris-random-forest"
result = mlflow.registermodel(modeluri, modelname)
print(f"Model registered: {result.name}, version {result.version}")
Method 3: Menggunakan client API
modelversion = client.createmodelversion(
name="iris-random-forest",
source=modeluri,
runid=runid,
description="Random Forest model trained on full dataset"
)
Model Versioning
from mlflow.tracking import MlflowClient
client = MlflowClient()
List semua registered models
for rm in client.searchregisteredmodels():
print(f"Model: {rm.name}")
for mv in rm.latestversions:
print(f" Version {mv.version}: {mv.currentstage}")
Get model version details
modelname = "iris-random-forest"
version = 3
modelversion = client.getmodelversion(modelname, version)
print(f"Version: {modelversion.version}")
print(f"Stage: {modelversion.currentstage}")
print(f"Description: {modelversion.description}")
Update model version
client.updatemodelversion(
name=modelname,
version=version,
description="Updated model with better performance"
)
Model Stages dan Lifecycle
MLflow mendukung lifecycle stages:
from mlflow.tracking import MlflowClient
client = MlflowClient()
modelname = "iris-random-forest"
Transition to Staging
client.transitionmodelversionstage(
name=modelname,
version=2,
stage="Staging",
archiveexistingversions=True
)
Transition to Production
client.transitionmodelversionstage(
name=modelname,
version=3,
stage="Production",
archiveexistingversions=True # Archive previous Production versions
)
Archive old version
client.transitionmodelversionstage(
name=modelname,
version=1,
stage="Archived"
)
Get latest version by stage
latestproduction = client.getlatestversions(
name=modelname,
stages=["Production"]
)[0]
print(f"Production model version: {latestproduction.version}")
Model Annotations dan Tags
# Set tags pada model version
client.setmodelversiontag(
name="iris-random-forest",
version="3",
key="validationaccuracy",
value="0.95"
)
client.setmodelversiontag(
name="iris-random-forest",
version="3",
key="trainingdataset",
value="irisv2cleaned"
)
Set tags pada registered model
client.setregisteredmodeltag(
name="iris-random-forest",
key="task",
value="classification"
)
Delete tag
client.deletemodelversiontag(
name="iris-random-forest",
version="3",
key="oldtag"
)
Deployment
Serve Model via REST API
# Serve model dari run
mlflow models serve -m runs:/ID>/model -p 5001
Serve dari model registry
mlflow models serve -m models:/iris-classifier/Production -p 5001
Test endpoint
curl -X POST http://localhost:5001/invocations \
-H 'Content-Type: application/json' \
-d '{
"dataframesplit": {
"columns": ["sepallength", "sepalwidth", "petallength", "petalwidth"],
"data": [[5.1, 3.5, 1.4, 0.2], [6.2, 2.8, 4.8, 1.8]]
}
}'
Deploy ke Docker
# Build Docker image
mlflow models build-docker \
-m models:/iris-classifier/Production \
-n iris-classifier:latest
Run container
docker run -p 5001:8080 iris-classifier:latest
Deploy ke AWS SageMaker
# Install plugin
pip install mlflow[sagemaker]
Deploy ke SageMaker
mlflow sagemaker deploy \
-a iris-classifier \
-m models:/iris-classifier/Production \
--region-name us-east-1 \
--execution-role-arn arn:aws:iam::123456789:role/SageMakerRole
Deploy ke Azure ML
import mlflow.azureml
from azureml.core import Workspace
from azureml.core.webservice import AciWebservice
Azure workspace
workspace = Workspace.fromconfig()
Deployment config
aciconfig = AciWebservice.deployconfiguration(
cpucores=1,
memorygb=1,
authenabled=True
)
Deploy
modeluri = "models:/iris-classifier/Production"
service = mlflow.azureml.deploy(
modeluri=modeluri,
workspace=workspace,
deploymentconfig=aciconfig,
servicename="iris-classifier-service"
)
print(f"Service URL: {service.scoringuri}")
Batch Inference
import mlflow.pyfunc
import pandas as pd
Load model
model = mlflow.pyfunc.loadmodel("models:/iris-classifier/Production")
Load batch data
batchdata = pd.readcsv("batchinput.csv")
Predict
predictions = model.predict(batchdata)
Save results
results = pd.DataFrame({
"input": batchdata.todict(orient="records"),
"prediction": predictions
})
results.tocsv("predictions.csv", index=False)
print(f"Processed {len(predictions)} predictions")
Advanced Features
Model Comparison
import mlflow
from mlflow.tracking import MlflowClient
import pandas as pd
client = MlflowClient()
experiment = client.getexperimentbyname("iris-classification")
Search runs
runs = client.searchruns(
experimentids=[experiment.experimentid],
filterstring="metrics.accuracy > 0.9",
orderby=["metrics.accuracy DESC"],
maxresults=10
)
Compare runs
comparison = pd.DataFrame([
{
"runid": run.info.runid,
"accuracy": run.data.metrics.get("accuracy"),
"f1score": run.data.metrics.get("f1score"),
"nestimators": run.data.params.get("nestimators"),
"maxdepth": run.data.params.get("maxdepth")
}
for run in runs
])
print(comparison.sortvalues("accuracy", ascending=False))
MLflow dengan DVC (Data Version Control)
import mlflow
import dvc.api
with mlflow.startrun():
# Get data version dari DVC
dataurl = dvc.api.geturl(
path="data/train.csv",
repo="https://github.com/username/ml-project"
)
# Log data version
mlflow.logparam("dataversion", dataurl)
# Load data
with dvc.api.open(dataurl) as f:
data = pd.readcsv(f)
# Training code...
MLflow dengan Airflow
from airflow import DAG
from airflow.operators.python import PythonOperator
import mlflow
from datetime import datetime
def trainmodel():
mlflow.settrackinguri("http://mlflow-server:5000")
with mlflow.startrun():
# Training code
pass
def evaluatemodel():
# Evaluation code
pass
def deploymodel():
client = MlflowClient()
# Get best model
bestrun = client.searchruns(
experimentids=["1"],
orderby=["metrics.accuracy DESC"],
maxresults=1
)[0]
# Transition to production
modeluri = f"runs:/{bestrun.info.runid}/model"
mlflow.registermodel(modeluri, "iris-classifier")
client.transitionmodelversionstage(
name="iris-classifier",
version=latestversion,
stage="Production"
)
with DAG(
'mlpipeline',
startdate=datetime(2024, 1, 1),
scheduleinterval='@daily'
) as dag:
train = PythonOperator(taskid='train', pythoncallable=trainmodel)
evaluate = PythonOperator(taskid='evaluate', pythoncallable=evaluatemodel)
deploy = PythonOperator(taskid='deploy', pythoncallable=deploymodel)
train >> evaluate >> deploy
Best Practices
1. Experiment Organization
# Gunakan naming convention yang konsisten
mlflow.setexperiment(f"project-name/{model-type}/{version}")
Contoh:
mlflow.setexperiment("fraud-detection/xgboost/v2")
mlflow.setexperiment("recommendation/collaborative-filtering/prod")
2. Comprehensive Logging
with mlflow.startrun():
# Log semua parameters
mlflow.logparams({
"modeltype": "RandomForest",
"nestimators": 100,
"maxdepth": 5,
"randomstate": 42,
"trainsize": len(Xtrain),
"testsize": len(Xtest),
"featurecount": Xtrain.shape[1]
})
# Log multiple metrics
mlflow.logmetrics({
"trainaccuracy": trainacc,
"testaccuracy": testacc,
"trainloss": trainloss,
"testloss": testloss,
"precision": precision,
"recall": recall,
"f1score": f1
})
# Log artifacts
mlflow.logartifact("featureimportance.png")
mlflow.logartifact("confusionmatrix.png")
mlflow.logartifact("roccurve.png")
# Log dataset info
mlflow.logdict({
"features": featurenames,
"target": targetname,
"classes": classnames
}, "datasetinfo.json")
3. Model Validation Before Registry
def validatemodel(model, testdata, threshold=0.85):
accuracy = model.score(testdata['X'], testdata['y'])
return accuracy >= threshold
with mlflow.startrun() as run:
# Train model
model.fit(Xtrain, ytrain)
# Validate
if validatemodel(model, {"X": Xtest, "y": ytest}):
# Register only if meets threshold
mlflow.sklearn.logmodel(
model,
"model",
registeredmodelname="iris-classifier"
)
print("Model registered")
else:
print("Model did not meet quality threshold")
4. Environment Reproducibility
import mlflow
Log conda environment
condaenv = {
"name": "ml-env",
"channels": ["defaults", "conda-forge"],
"dependencies": [
"python=3.9",
"pip",
{"pip": [
f"scikit-learn=={sklearn.version}",
f"pandas=={pd.version}",
f"numpy=={np.version}"
]}
]
}
mlflow.sklearn.logmodel(
model,
"model",
condaenv=condaenv
)
5. Cleanup Old Runs
from mlflow.tracking import MlflowClient
from datetime import datetime, timedelta
client = MlflowClient()
Delete runs lebih dari 30 hari
cutoffdate = datetime.now() - timedelta(days=30)
experiment = client.getexperimentbyname("old-experiment")
runs = client.searchruns(
experimentids=[experiment.experimentid],
filterstring=f"attributes.starttime < {int(cutoffdate.timestamp() 1000)}"
)
for run in runs:
client.deleterun(run.info.runid)
print(f"Deleted run: {run.info.runid}")
Monitoring dan Debugging
Performance Tracking
import time
import mlflow
with mlflow.startrun():
# Track training time
starttime = time.time()
model.fit(Xtrain, ytrain)
trainingtime = time.time() - starttime
mlflow.logmetric("trainingtimeseconds", trainingtime)
# Track inference time
starttime = time.time()
predictions = model.predict(Xtest)
inferencetime = time.time() - starttime
mlflow.logmetric("inferencetimeseconds", inferencetime)
mlflow.logmetric("avgpredictiontimems",
(inferencetime / len(Xtest)) 1000)
Model Drift Detection
import mlflow
from scipy import stats
def checkdatadrift(referencedata, newdata, threshold=0.05):
driftdetected = False
driftfeatures = []
for column in referencedata.columns:
# Kolmogorov-Smirnov test
statistic, pvalue = stats.ks2samp(
referencedata[column],
newdata[column]
)
mlflow.logmetric(f"driftpvalue{column}", pvalue)
if pvalue < threshold:
driftdetected = True
driftfeatures.append(column)
mlflow.logparam("driftdetected", driftdetected)
if driftdetected:
mlflow.logparam("driftfeatures", driftfeatures)
return driftdetected, driftfeatures
with mlflow.startrun():
driftdetected, features = checkdatadrift(traindata, productiondata)
if driftdetected:
print(f"Data drift detected in features: {features}")
Kesimpulan
MLflow adalah platform yang powerful untuk managing ML lifecycle:
- Selalu log parameters, metrics, dan artifacts secara komprehensif
- Gunakan autologging untuk framework yang didukung
- Implement model validation sebelum production
- Manfaatkan Model Registry untuk versioning dan governance
- Setup monitoring untuk data drift dan model performance
- Gunakan tags dan descriptions untuk dokumentasi
- Integrate dengan CI/CD pipeline untuk automation
Dengan MLflow, tim ML dapat:
- Reproduce experiments dengan mudah
- Compare model performance secara sistematis
- Deploy models dengan konsisten
- Manage model lifecycle dari development hingga production
- Collaborate lebih efektif dalam team