Complete Azure MLflow Integration Tutorial: Experiment Tracking and Model Management
Azure Machine Learning provides native MLflow integration for experiment tracking, model versioning, and deployment. This tutorial covers using MLflow with Azure ML for comprehensive ML lifecycle management.
Why MLflow on Azure?
Key Benefits:- Native integration: Seamless Azure ML connectivity
- Open standard: Portable across platforms
- Unified tracking: Experiments, models, artifacts
- Easy deployment: Deploy MLflow models directly
- Collaboration: Share experiments across teams
- Tracking: Log experiments and metrics
- Projects: Package ML code
- Models: Model versioning and deployment
- Registry: Centralized model store
Prerequisites
pip install mlflow azureml-mlflow azure-ai-ml azure-identity
Azure CLI
az login
Setup
1. Connect to Azure ML
from azure.ai.ml import MLClient
from azure.identity import DefaultAzureCredential
import mlflow
Connect to workspace
mlclient = MLClient(
credential=DefaultAzureCredential(),
subscriptionid="your-subscription-id",
resourcegroupname="my-resource-group",
workspacename="my-ml-workspace"
)
Get MLflow tracking URI
trackinguri = mlclient.workspaces.get().mlflowtrackinguri
print(f"Tracking URI: {trackinguri}")
Set tracking URI
mlflow.settrackinguri(trackinguri)
2. Configure Authentication
import os
Set Azure credentials for MLflow
os.environ["AZURETENANTID"] = "your-tenant-id"
os.environ["AZURECLIENTID"] = "your-client-id"
os.environ["AZURECLIENTSECRET"] = "your-client-secret"
Or use DefaultAzureCredential
from azure.identity import DefaultAzureCredential
credential = DefaultAzureCredential()
Experiment Tracking
1. Create and Set Experiment
import mlflow
Set experiment
mlflow.setexperiment("my-ml-experiment")
Or create with tags
experiment = mlflow.createexperiment(
name="classification-experiment",
tags={
"team": "data-science",
"project": "customer-churn"
}
)
2. Log Parameters and Metrics
import mlflow
from sklearn.ensemble import RandomForestClassifier
from sklearn.modelselection import traintestsplit
from sklearn.metrics import accuracyscore, f1score, precisionscore, recallscore
Start run
with mlflow.startrun(runname="random-forest-v1"):
# Log parameters
mlflow.logparam("nestimators", 100)
mlflow.logparam("maxdepth", 10)
mlflow.logparam("randomstate", 42)
# Train model
model = RandomForestClassifier(
nestimators=100,
maxdepth=10,
randomstate=42
)
model.fit(Xtrain, ytrain)
# Predictions
predictions = model.predict(Xtest)
# Log metrics
mlflow.logmetric("accuracy", accuracyscore(ytest, predictions))
mlflow.logmetric("f1score", f1score(ytest, predictions))
mlflow.logmetric("precision", precisionscore(ytest, predictions))
mlflow.logmetric("recall", recallscore(ytest, predictions))
print("Run completed")
3. Log Artifacts
import matplotlib.pyplot as plt
from sklearn.metrics import confusionmatrix, ConfusionMatrixDisplay
with mlflow.startrun():
# Train and predict
model.fit(Xtrain, ytrain)
predictions = model.predict(Xtest)
# Create confusion matrix plot
cm = confusionmatrix(ytest, predictions)
disp = ConfusionMatrixDisplay(confusionmatrix=cm)
disp.plot()
plt.savefig("confusionmatrix.png")
# Log artifact
mlflow.logartifact("confusionmatrix.png")
# Log directory of artifacts
mlflow.logartifacts("./plots", artifactpath="visualizations")
# Log text file
with open("modelinfo.txt", "w") as f:
f.write(f"Model: RandomForest\nFeatures: {Xtrain.shape[1]}")
mlflow.logartifact("modelinfo.txt")
4. Autologging
import mlflow.sklearn
Enable autologging for sklearn
mlflow.sklearn.autolog()
with mlflow.startrun():
model = RandomForestClassifier(nestimators=100, maxdepth=10)
model.fit(Xtrain, ytrain)
# All parameters, metrics, and model are logged automatically
Autologging for other frameworks
mlflow.tensorflow.autolog()
mlflow.pytorch.autolog()
mlflow.xgboost.autolog()
mlflow.lightgbm.autolog()
Model Logging
1. Log Sklearn Model
import mlflow.sklearn
with mlflow.startrun():
# Train model
model = RandomForestClassifier(nestimators=100)
model.fit(Xtrain, ytrain)
# Log model
mlflow.sklearn.logmodel(
model,
artifactpath="model",
registeredmodelname="sklearn-classifier"
)
# Get run info
runid = mlflow.activerun().info.runid
print(f"Run ID: {runid}")
2. Log PyTorch Model
import mlflow.pytorch
import torch
import torch.nn as nn
class SimpleNN(nn.Module):
def init(self, inputsize, hiddensize, outputsize):
super().init()
self.fc1 = nn.Linear(inputsize, hiddensize)
self.fc2 = nn.Linear(hiddensize, outputsize)
self.relu = nn.ReLU()
def forward(self, x):
x = self.relu(self.fc1(x))
return self.fc2(x)
with mlflow.startrun():
# Create and train model
model = SimpleNN(10, 50, 2)
# ... training code
# Log model
mlflow.pytorch.logmodel(
model,
artifactpath="pytorch-model",
registeredmodelname="pytorch-classifier"
)
3. Log with Signature
from mlflow.models.signature import infersignature
import pandas as pd
with mlflow.startrun():
# Train model
model.fit(Xtrain, ytrain)
predictions = model.predict(Xtest)
# Infer signature
signature = infersignature(Xtrain, predictions)
# Log with signature
mlflow.sklearn.logmodel(
model,
artifactpath="model",
signature=signature,
inputexample=Xtrain[:5]
)
Model Registry
1. Register Model
import mlflow
Register model from run
modeluri = f"runs:/{runid}/model"
result = mlflow.registermodel(
modeluri=modeluri,
name="production-classifier"
)
print(f"Model version: {result.version}")
2. Manage Model Versions
from mlflow.tracking import MlflowClient
client = MlflowClient()
Get model details
model = client.getregisteredmodel("production-classifier")
print(f"Model: {model.name}")
print(f"Latest versions: {model.latestversions}")
Get specific version
modelversion = client.getmodelversion(
name="production-classifier",
version="1"
)
print(f"Version: {modelversion.version}")
print(f"Stage: {modelversion.currentstage}")
3. Transition Model Stage
# Transition to staging
client.transitionmodelversionstage(
name="production-classifier",
version="1",
stage="Staging"
)
Transition to production
client.transitionmodelversionstage(
name="production-classifier",
version="1",
stage="Production",
archiveexistingversions=True
)
Archive model
client.transitionmodelversionstage(
name="production-classifier",
version="1",
stage="Archived"
)
4. Model Aliases
# Set alias
client.setregisteredmodelalias(
name="production-classifier",
alias="champion",
version="2"
)
Get model by alias
modeluri = "models:/production-classifier@champion"
model = mlflow.pyfunc.loadmodel(modeluri)
Delete alias
client.deleteregisteredmodelalias(
name="production-classifier",
alias="champion"
)
Load and Use Models
1. Load Model from Registry
import mlflow.pyfunc
Load latest production model
model = mlflow.pyfunc.loadmodel("models:/production-classifier/Production")
Load specific version
model = mlflow.pyfunc.loadmodel("models:/production-classifier/1")
Load by alias
model = mlflow.pyfunc.loadmodel("models:/production-classifier@champion")
Make predictions
predictions = model.predict(Xtest)
2. Load Model from Run
# Load from specific run
modeluri = f"runs:/{runid}/model"
model = mlflow.sklearn.loadmodel(modeluri)
predictions = model.predict(Xtest)
Deploy MLflow Models
1. Deploy to Azure ML Online Endpoint
from azure.ai.ml.entities import (
ManagedOnlineEndpoint,
ManagedOnlineDeployment,
Model
)
from azure.ai.ml.constants import AssetTypes
Register MLflow model in Azure ML
model = Model(
path=f"runs:/{runid}/model",
name="mlflow-classifier",
type=AssetTypes.MLFLOWMODEL
)
registeredmodel = mlclient.models.createorupdate(model)
Create endpoint
endpoint = ManagedOnlineEndpoint(
name="mlflow-endpoint",
authmode="key"
)
mlclient.onlineendpoints.begincreateorupdate(endpoint).result()
Create deployment (no scoring script needed for MLflow models)
deployment = ManagedOnlineDeployment(
name="blue",
endpointname="mlflow-endpoint",
model=f"azureml:{registeredmodel.name}:{registeredmodel.version}",
instancetype="StandardDS3v2",
instancecount=1
)
mlclient.onlinedeployments.begincreateorupdate(deployment).result()
Set traffic
endpoint.traffic = {"blue": 100}
mlclient.onlineendpoints.begincreateorupdate(endpoint).result()
2. Test Deployed Model
import json
Prepare data
testdata = {"inputdata": Xtest[:5].tolist()}
Invoke endpoint
response = mlclient.onlineendpoints.invoke(
endpointname="mlflow-endpoint",
requestfile=json.dumps(testdata)
)
print(f"Predictions: {response}")
Query Experiments
1. Search Runs
import mlflow
Search all runs
runs = mlflow.searchruns(
experimentnames=["my-ml-experiment"]
)
print(runs[["runid", "metrics.accuracy", "params.nestimators"]])
Search with filter
runs = mlflow.searchruns(
experimentnames=["my-ml-experiment"],
filterstring="metrics.accuracy > 0.9 and params.nestimators = '100'",
orderby=["metrics.accuracy DESC"],
maxresults=10
)
2. Compare Runs
# Get multiple runs
runids = ["run1id", "run2id", "run3id"]
comparisondf = mlflow.searchruns(
filterstring=f"runid IN ({','.join([f\"'{r}'\" for r in runids])})"
)
Compare metrics
print(comparisondf[["runid", "metrics.accuracy", "metrics.f1score"]])
3. Get Best Run
# Get best run by metric
bestrun = mlflow.searchruns(
experimentnames=["my-ml-experiment"],
filterstring="metrics.accuracy > 0",
orderby=["metrics.accuracy DESC"],
maxresults=1
).iloc[0]
print(f"Best run: {bestrun['runid']}")
print(f"Best accuracy: {bestrun['metrics.accuracy']}")
Training Jobs with MLflow
1. Submit Training Job
from azure.ai.ml import command, Input
Define training job
trainingjob = command(
code="./src",
command="python trainwithmlflow.py --data ${{inputs.data}}",
inputs={
"data": Input(type="urifile", path="azureml:training-data:1")
},
environment="AzureML-sklearn-1.0-ubuntu20.04-py38-cpu@latest",
compute="cpu-cluster",
experimentname="mlflow-training"
)
Submit job
returnedjob = mlclient.jobs.createorupdate(trainingjob)
print(f"Job submitted: {returnedjob.name}")
2. Training Script with MLflow
# src/trainwithmlflow.py
import argparse
import mlflow
import pandas as pd
from sklearn.ensemble import RandomForestClassifier
from sklearn.modelselection import traintestsplit
from sklearn.metrics import accuracyscore
def main():
parser = argparse.ArgumentParser()
parser.addargument("--data", type=str, required=True)
args = parser.parseargs()
# MLflow autologging
mlflow.autolog()
# Load data
df = pd.readcsv(args.data)
X = df.drop("target", axis=1)
y = df["target"]
Xtrain, Xtest, ytrain, ytest = traintestsplit(
X, y, testsize=0.2, randomstate=42
)
# Train model
with mlflow.startrun():
model = RandomForestClassifier(nestimators=100, maxdepth=10)
model.fit(Xtrain, ytrain)
# Log additional metrics
predictions = model.predict(Xtest)
accuracy = accuracyscore(ytest, predictions)
mlflow.logmetric("testaccuracy", accuracy)
# Register model
mlflow.sklearn.logmodel(
model,
"model",
registeredmodelname="training-job-model"
)
if name == "main":
main()
Best Practices
1. Organize Experiments
# Use meaningful experiment names
mlflow.setexperiment("project/team/model-type")
Use tags for organization
with mlflow.startrun(tags={
"developer": "john",
"modeltype": "classification",
"datasetversion": "v2"
}):
# Training code
pass
2. Log Comprehensive Information
with mlflow.startrun():
# Log data info
mlflow.log
param("trainingsamples", len(Xtrain))
mlflow.logparam("testsamples", len(Xtest))
mlflow.logparam("features", Xtrain.shape[1])
# Log preprocessing
mlflow.logparam("scaler", "StandardScaler")
mlflow.logparam("featureselection", "SelectKBest")
# Log environment
mlflow.logparam("pythonversion", "3.9")
mlflow.logparam("sklearnversion", sklearn.version)
# Training and logging
# ...
3. Model Documentation
# Add model description
client.updateregisteredmodel(
name="production-classifier",
description="Random Forest classifier for customer churn prediction"
)
Add version description
client.updatemodelversion(
name="production-classifier",
version="1",
description="Initial version trained on Q1 2024 data"
)
Conclusion
MLflow on Azure provides:
Key takeaways:
- Use autologging for quick setup
- Organize experiments with names and tags
- Register models for production
- Use stages for model lifecycle
- Deploy MLflow models directly to Azure ML