Tutorial Lengkap Azure MLflow Integration: Experiment Tracking dan Model Management
Azure Machine Learning menyediakan integrasi MLflow native untuk experiment tracking, model versioning, dan deployment. Tutorial ini mencakup penggunaan MLflow dengan Azure ML untuk manajemen lifecycle ML yang komprehensif.
Mengapa MLflow di Azure?
Manfaat Utama:- Integrasi native: Konektivitas seamless Azure ML
- Open standard: Portable lintas platform
- Unified tracking: Experiments, models, artifacts
- Easy deployment: Deploy MLflow models langsung
- Collaboration: Berbagi experiments antar tim
- Tracking: Log experiments dan metrics
- Projects: Package ML code
- Models: Model versioning dan deployment
- Registry: Centralized model store
Prerequisites
pip install mlflow azureml-mlflow azure-ai-ml azure-identity
Azure CLI
az login
Setup
1. Koneksi ke Azure ML
from azure.ai.ml import MLClient
from azure.identity import DefaultAzureCredential
import mlflow
Koneksi ke workspace
mlclient = MLClient(
credential=DefaultAzureCredential(),
subscriptionid="your-subscription-id",
resourcegroupname="my-resource-group",
workspacename="my-ml-workspace"
)
Dapatkan MLflow tracking URI
trackinguri = mlclient.workspaces.get().mlflowtrackinguri
print(f"Tracking URI: {trackinguri}")
Set tracking URI
mlflow.settrackinguri(trackinguri)
2. Konfigurasi Authentication
import os
Set Azure credentials untuk MLflow
os.environ["AZURETENANTID"] = "your-tenant-id"
os.environ["AZURECLIENTID"] = "your-client-id"
os.environ["AZURECLIENTSECRET"] = "your-client-secret"
Atau gunakan DefaultAzureCredential
from azure.identity import DefaultAzureCredential
credential = DefaultAzureCredential()
Experiment Tracking
1. Buat dan Set Experiment
import mlflow
Set experiment
mlflow.setexperiment("my-ml-experiment")
Atau buat dengan tags
experiment = mlflow.createexperiment(
name="classification-experiment",
tags={
"team": "data-science",
"project": "customer-churn"
}
)
2. Log Parameters dan Metrics
import mlflow
from sklearn.ensemble import RandomForestClassifier
from sklearn.modelselection import traintestsplit
from sklearn.metrics import accuracyscore, f1score, precisionscore, recallscore
Mulai run
with mlflow.startrun(runname="random-forest-v1"):
# Log parameters
mlflow.logparam("nestimators", 100)
mlflow.logparam("maxdepth", 10)
mlflow.logparam("randomstate", 42)
# Train model
model = RandomForestClassifier(
nestimators=100,
maxdepth=10,
randomstate=42
)
model.fit(Xtrain, ytrain)
# Predictions
predictions = model.predict(Xtest)
# Log metrics
mlflow.logmetric("accuracy", accuracyscore(ytest, predictions))
mlflow.logmetric("f1score", f1score(ytest, predictions))
mlflow.logmetric("precision", precisionscore(ytest, predictions))
mlflow.logmetric("recall", recallscore(ytest, predictions))
print("Run selesai")
3. Log Artifacts
import matplotlib.pyplot as plt
from sklearn.metrics import confusionmatrix, ConfusionMatrixDisplay
with mlflow.startrun():
# Train dan predict
model.fit(Xtrain, ytrain)
predictions = model.predict(Xtest)
# Buat confusion matrix plot
cm = confusionmatrix(ytest, predictions)
disp = ConfusionMatrixDisplay(confusionmatrix=cm)
disp.plot()
plt.savefig("confusionmatrix.png")
# Log artifact
mlflow.logartifact("confusionmatrix.png")
# Log direktori artifacts
mlflow.logartifacts("./plots", artifactpath="visualizations")
# Log text file
with open("modelinfo.txt", "w") as f:
f.write(f"Model: RandomForest\nFeatures: {Xtrain.shape[1]}")
mlflow.logartifact("modelinfo.txt")
4. Autologging
import mlflow.sklearn
Aktifkan autologging untuk sklearn
mlflow.sklearn.autolog()
with mlflow.startrun():
model = RandomForestClassifier(nestimators=100, maxdepth=10)
model.fit(Xtrain, ytrain)
# Semua parameters, metrics, dan model dilog otomatis
Autologging untuk framework lain
mlflow.tensorflow.autolog()
mlflow.pytorch.autolog()
mlflow.xgboost.autolog()
mlflow.lightgbm.autolog()
Model Logging
1. Log Sklearn Model
import mlflow.sklearn
with mlflow.startrun():
# Train model
model = RandomForestClassifier(nestimators=100)
model.fit(Xtrain, ytrain)
# Log model
mlflow.sklearn.logmodel(
model,
artifactpath="model",
registeredmodelname="sklearn-classifier"
)
# Dapatkan run info
runid = mlflow.activerun().info.runid
print(f"Run ID: {runid}")
2. Log PyTorch Model
import mlflow.pytorch
import torch
import torch.nn as nn
class SimpleNN(nn.Module):
def init(self, inputsize, hiddensize, outputsize):
super().init()
self.fc1 = nn.Linear(inputsize, hiddensize)
self.fc2 = nn.Linear(hiddensize, outputsize)
self.relu = nn.ReLU()
def forward(self, x):
x = self.relu(self.fc1(x))
return self.fc2(x)
with mlflow.startrun():
# Buat dan train model
model = SimpleNN(10, 50, 2)
# ... kode training
# Log model
mlflow.pytorch.logmodel(
model,
artifactpath="pytorch-model",
registeredmodelname="pytorch-classifier"
)
3. Log dengan Signature
from mlflow.models.signature import infersignature
import pandas as pd
with mlflow.startrun():
# Train model
model.fit(Xtrain, ytrain)
predictions = model.predict(Xtest)
# Infer signature
signature = infersignature(Xtrain, predictions)
# Log dengan signature
mlflow.sklearn.logmodel(
model,
artifactpath="model",
signature=signature,
inputexample=Xtrain[:5]
)
Model Registry
1. Register Model
import mlflow
Register model dari run
modeluri = f"runs:/{runid}/model"
result = mlflow.registermodel(
modeluri=modeluri,
name="production-classifier"
)
print(f"Model version: {result.version}")
2. Kelola Model Versions
from mlflow.tracking import MlflowClient
client = MlflowClient()
Dapatkan detail model
model = client.getregisteredmodel("production-classifier")
print(f"Model: {model.name}")
print(f"Latest versions: {model.latestversions}")
Dapatkan versi spesifik
modelversion = client.getmodelversion(
name="production-classifier",
version="1"
)
print(f"Version: {modelversion.version}")
print(f"Stage: {modelversion.currentstage}")
3. Transisi Model Stage
# Transisi ke staging
client.transitionmodelversionstage(
name="production-classifier",
version="1",
stage="Staging"
)
Transisi ke production
client.transitionmodelversionstage(
name="production-classifier",
version="1",
stage="Production",
archiveexistingversions=True
)
Archive model
client.transitionmodelversionstage(
name="production-classifier",
version="1",
stage="Archived"
)
4. Model Aliases
# Set alias
client.setregisteredmodelalias(
name="production-classifier",
alias="champion",
version="2"
)
Dapatkan model berdasarkan alias
modeluri = "models:/production-classifier@champion"
model = mlflow.pyfunc.loadmodel(modeluri)
Hapus alias
client.deleteregisteredmodelalias(
name="production-classifier",
alias="champion"
)
Load dan Gunakan Models
1. Load Model dari Registry
import mlflow.pyfunc
Load model production terbaru
model = mlflow.pyfunc.loadmodel("models:/production-classifier/Production")
Load versi spesifik
model = mlflow.pyfunc.loadmodel("models:/production-classifier/1")
Load berdasarkan alias
model = mlflow.pyfunc.loadmodel("models:/production-classifier@champion")
Buat predictions
predictions = model.predict(Xtest)
2. Load Model dari Run
# Load dari run spesifik
modeluri = f"runs:/{runid}/model"
model = mlflow.sklearn.loadmodel(modeluri)
predictions = model.predict(Xtest)
Deploy MLflow Models
1. Deploy ke Azure ML Online Endpoint
from azure.ai.ml.entities import (
ManagedOnlineEndpoint,
ManagedOnlineDeployment,
Model
)
from azure.ai.ml.constants import AssetTypes
Register MLflow model di Azure ML
model = Model(
path=f"runs:/{runid}/model",
name="mlflow-classifier",
type=AssetTypes.MLFLOWMODEL
)
registeredmodel = mlclient.models.createorupdate(model)
Buat endpoint
endpoint = ManagedOnlineEndpoint(
name="mlflow-endpoint",
authmode="key"
)
mlclient.onlineendpoints.begincreateorupdate(endpoint).result()
Buat deployment (tidak perlu scoring script untuk MLflow models)
deployment = ManagedOnlineDeployment(
name="blue",
endpointname="mlflow-endpoint",
model=f"azureml:{registeredmodel.name}:{registeredmodel.version}",
instancetype="StandardDS3v2",
instancecount=1
)
mlclient.onlinedeployments.begincreateorupdate(deployment).result()
Set traffic
endpoint.traffic = {"blue": 100}
mlclient.onlineendpoints.begincreateorupdate(endpoint).result()
2. Test Model yang Dideploy
import json
Siapkan data
testdata = {"inputdata": Xtest[:5].tolist()}
Panggil endpoint
response = mlclient.onlineendpoints.invoke(
endpointname="mlflow-endpoint",
requestfile=json.dumps(testdata)
)
print(f"Predictions: {response}")
Query Experiments
1. Search Runs
import mlflow
Search semua runs
runs = mlflow.searchruns(
experimentnames=["my-ml-experiment"]
)
print(runs[["runid", "metrics.accuracy", "params.nestimators"]])
Search dengan filter
runs = mlflow.searchruns(
experimentnames=["my-ml-experiment"],
filterstring="metrics.accuracy > 0.9 and params.nestimators = '100'",
orderby=["metrics.accuracy DESC"],
maxresults=10
)
2. Compare Runs
# Dapatkan multiple runs
runids = ["run1id", "run2id", "run3id"]
comparisondf = mlflow.searchruns(
filterstring=f"runid IN ({','.join([f\"'{r}'\" for r in runids])})"
)
Bandingkan metrics
print(comparisondf[["runid", "metrics.accuracy", "metrics.f1score"]])
3. Dapatkan Best Run
# Dapatkan run terbaik berdasarkan metric
bestrun = mlflow.searchruns(
experimentnames=["my-ml-experiment"],
filterstring="metrics.accuracy > 0",
orderby=["metrics.accuracy DESC"],
maxresults=1
).iloc[0]
print(f"Best run: {bestrun['runid']}")
print(f"Best accuracy: {bestrun['metrics.accuracy']}")
Training Jobs dengan MLflow
1. Submit Training Job
from azure.ai.ml import command, Input
Definisikan training job
trainingjob = command(
code="./src",
command="python trainwithmlflow.py --data ${{inputs.data}}",
inputs={
"data": Input(type="urifile", path="azureml:training-data:1")
},
environment="AzureML-sklearn-1.0-ubuntu20.04-py38-cpu@latest",
compute="cpu-cluster",
experimentname="mlflow-training"
)
Submit job
returnedjob = mlclient.jobs.createorupdate(trainingjob)
print(f"Job disubmit: {returnedjob.name}")
2. Training Script dengan MLflow
# src/trainwithmlflow.py
import argparse
import mlflow
import pandas as pd
from sklearn.ensemble import RandomForestClassifier
from sklearn.modelselection import traintestsplit
from sklearn.metrics import accuracyscore
def main():
parser = argparse.ArgumentParser()
parser.addargument("--data", type=str, required=True)
args = parser.parseargs()
# MLflow autologging
mlflow.autolog()
# Load data
df = pd.readcsv(args.data)
X = df.drop("target", axis=1)
y = df["target"]
Xtrain, Xtest, ytrain, ytest = traintestsplit(
X, y, testsize=0.2, randomstate=42
)
# Train model
with mlflow.startrun():
model = RandomForestClassifier(nestimators=100, maxdepth=10)
model.fit(Xtrain, ytrain)
# Log metrics tambahan
predictions = model.predict(Xtest)
accuracy = accuracyscore(ytest, predictions)
mlflow.logmetric("testaccuracy", accuracy)
# Register model
mlflow.sklearn.logmodel(
model,
"model",
registeredmodelname="training-job-model"
)
if name == "main":
main()
Best Practices
1. Organisasi Experiments
# Gunakan nama experiment yang bermakna
mlflow.setexperiment("project/team/model-type")
Gunakan tags untuk organisasi
with mlflow.startrun(tags={
"developer": "john",
"modeltype": "classification",
"datasetversion": "v2"
}):
# Kode training
pass
2. Log Informasi Komprehensif
with mlflow.startrun():
# Log info data
mlflow.log
param("trainingsamples", len(Xtrain))
mlflow.logparam("testsamples", len(Xtest))
mlflow.logparam("features", Xtrain.shape[1])
# Log preprocessing
mlflow.logparam("scaler", "StandardScaler")
mlflow.logparam("featureselection", "SelectKBest")
# Log environment
mlflow.logparam("pythonversion", "3.9")
mlflow.logparam("sklearnversion", sklearn.version)
# Training dan logging
# ...
3. Dokumentasi Model
# Tambahkan deskripsi model
client.updateregisteredmodel(
name="production-classifier",
description="Random Forest classifier untuk prediksi customer churn"
)
Tambahkan deskripsi versi
client.updatemodelversion(
name="production-classifier",
version="1",
description="Versi awal ditraining dengan data Q1 2024"
)
Kesimpulan
MLflow di Azure menyediakan:
Key takeaways:
- Gunakan autologging untuk setup cepat
- Organisasi experiments dengan nama dan tags
- Register models untuk production
- Gunakan stages untuk lifecycle model
- Deploy MLflow models langsung ke Azure ML