Complete Azure Machine Learning Tutorial: End-to-End ML on Azure
Azure Machine Learning is a cloud-based platform for building, training, and deploying machine learning models. It provides a comprehensive MLOps environment with tools for the entire ML lifecycle.
Why Azure Machine Learning?
Key Benefits:- Unified platform: Complete ML lifecycle management
- AutoML: Automated machine learning capabilities
- Enterprise-ready: Security, compliance, and governance
- Flexible compute: From notebooks to GPU clusters
- Integration: Seamless Azure ecosystem connectivity
- Workspaces
- Compute instances and clusters
- Datastores and datasets
- Experiments and runs
- Models and endpoints
Prerequisites
pip install azure-ai-ml azure-identity
Azure CLI
az login
az extension add -n ml
Quick Start
1. Create Workspace
from azure.ai.ml import MLClient
from azure.identity import DefaultAzureCredential
from azure.ai.ml.entities import Workspace
Authenticate
credential = DefaultAzureCredential()
Create workspace
workspace = Workspace(
name="my-ml-workspace",
location="eastus",
displayname="ML Workspace",
description="Azure ML workspace for ML projects"
)
Create ML client
mlclient = MLClient(
credential=credential,
subscriptionid="your-subscription-id",
resourcegroupname="my-resource-group"
)
Create workspace
mlclient.workspaces.begincreateorupdate(workspace).result()
print(f"Workspace created: {workspace.name}")
2. Connect to Existing Workspace
from azure.ai.ml import MLClient
from azure.identity import DefaultAzureCredential
mlclient = MLClient(
credential=DefaultAzureCredential(),
subscriptionid="your-subscription-id",
resourcegroupname="my-resource-group",
workspacename="my-ml-workspace"
)
print(f"Connected to workspace: {mlclient.workspacename}")
Compute Resources
1. Create Compute Instance
from azure.ai.ml.entities import ComputeInstance
computeinstance = ComputeInstance(
name="my-compute-instance",
size="StandardDS3v2",
idletimebeforeshutdownminutes=60
)
mlclient.compute.begincreateorupdate(computeinstance).result()
print("Compute instance created")
2. Create Compute Cluster
from azure.ai.ml.entities import AmlCompute
computecluster = AmlCompute(
name="cpu-cluster",
type="amlcompute",
size="StandardDS3v2",
mininstances=0,
maxinstances=4,
idletimebeforescaledown=120
)
mlclient.compute.begincreateorupdate(computecluster).result()
print("Compute cluster created")
3. GPU Cluster
gpucluster = AmlCompute(
name="gpu-cluster",
type="amlcompute",
size="Standard
NC6",
mininstances=0,
maxinstances=2,
idletimebeforescaledown=300
)
mlclient.compute.begincreateorupdate(gpucluster).result()
Data Management
1. Register Datastore
from azure.ai.ml.entities import AzureBlobDatastore
datastore = AzureBlobDatastore(
name="my-blob-datastore",
accountname="mystorageaccount",
containername="ml-data",
credentials={
"accountkey": "your-account-key"
}
)
mlclient.datastores.createorupdate(datastore)
print("Datastore registered")
2. Create Dataset
from azure.ai.ml.entities import Data
from azure.ai.ml.constants import AssetTypes
File dataset
filedata = Data(
name="training-data",
path="azureml://datastores/my-blob-datastore/paths/data/train.csv",
type=AssetTypes.URIFILE,
description="Training dataset"
)
mlclient.data.createorupdate(filedata)
Folder dataset
folderdata = Data(
name="image-dataset",
path="azureml://datastores/my-blob-datastore/paths/images/",
type=AssetTypes.URIFOLDER,
description="Image dataset"
)
mlclient.data.createorupdate(folderdata)
3. Access Data in Jobs
from azure.ai.ml import Input
Use in command job
jobinputs = {
"trainingdata": Input(
type="urifile",
path="azureml://datastores/my-blob-datastore/paths/data/train.csv"
)
}
Training Jobs
1. Command Job
from azure.ai.ml import command, Input, Output
Define training job
trainingjob = command(
code="./src",
command="python train.py --data ${{inputs.data}} --output ${{outputs.model}}",
inputs={
"data": Input(
type="urifile",
path="azureml://datastores/workspaceblobstore/paths/data/train.csv"
)
},
outputs={
"model": Output(type="urifolder", path="azureml://datastores/workspaceblobstore/paths/models/")
},
environment="AzureML-sklearn-1.0-ubuntu20.04-py38-cpu@latest",
compute="cpu-cluster",
displayname="sklearn-training",
experimentname="my-experiment"
)
Submit job
returnedjob = mlclient.jobs.createorupdate(trainingjob)
print(f"Job submitted: {returnedjob.name}")
Wait for completion
mlclient.jobs.stream(returnedjob.name)
2. Training Script
# src/train.py
import argparse
import pandas as pd
from sklearn.ensemble import RandomForestClassifier
from sklearn.modelselection import traintestsplit
from sklearn.metrics import accuracyscore
import joblib
import os
import mlflow
def main():
parser = argparse.ArgumentParser()
parser.addargument("--data", type=str, required=True)
parser.addargument("--output", type=str, required=True)
args = parser.parseargs()
# Enable MLflow autologging
mlflow.autolog()
# Load data
df = pd.readcsv(args.data)
X = df.drop("target", axis=1)
y = df["target"]
# Split data
Xtrain, Xtest, ytrain, ytest = traintestsplit(
X, y, testsize=0.2, randomstate=42
)
# Train model
model = RandomForestClassifier(nestimators=100, randomstate=42)
model.fit(Xtrain, ytrain)
# Evaluate
predictions = model.predict(Xtest)
accuracy = accuracyscore(ytest, predictions)
print(f"Accuracy: {accuracy}")
# Save model
os.makedirs(args.output, existok=True)
joblib.dump(model, os.path.join(args.output, "model.joblib"))
if name == "main":
main()
3. Custom Environment
from azure.ai.ml.entities import Environment, BuildContext
From conda specification
env = Environment(
name="sklearn-env",
description="Scikit-learn environment",
condafile="./environment.yml",
image="mcr.microsoft.com/azureml/openmpi4.1.0-ubuntu20.04:latest"
)
mlclient.environments.createorupdate(env)
From Dockerfile
dockerenv = Environment(
name="custom-env",
build=BuildContext(path="./docker-context"),
description="Custom Docker environment"
)
mlclient.environments.createorupdate(dockerenv)
AutoML
1. Classification
from azure.ai.ml import automl, Input
Configure AutoML classification
classificationjob = automl.classification(
compute="cpu-cluster",
experimentname="automl-classification",
trainingdata=Input(type="mltable", path="./data/train"),
targetcolumnname="target",
primarymetric="accuracy",
ncrossvalidations=5,
enablemodelexplainability=True
)
Set limits
classificationjob.setlimits(
timeoutminutes=60,
trialtimeoutminutes=20,
maxtrials=20,
maxconcurrenttrials=4
)
Set training settings
classificationjob.settraining(
enablestackensemble=True,
enablevoteensemble=True
)
Submit job
returnedjob = mlclient.jobs.createorupdate(classificationjob)
2. Regression
regressionjob = automl.regression(
compute="cpu-cluster",
experimentname="automl-regression",
trainingdata=Input(type="mltable", path="./data/train"),
targetcolumnname="price",
primarymetric="r2score",
ncrossvalidations=5
)
regressionjob.setlimits(
timeoutminutes=120,
trialtimeoutminutes=30,
maxtrials=30
)
mlclient.jobs.createorupdate(regressionjob)
3. Forecasting
forecastingjob = automl.forecasting(
compute="cpu-cluster",
experiment
name="automl-forecasting",
trainingdata=Input(type="mltable", path="./data/timeseries"),
targetcolumnname="sales",
primarymetric="normalizedrootmeansquarederror",
ncrossvalidations=3
)
Configure forecasting settings
forecastingjob.setforecastsettings(
timecolumnname="date",
forecasthorizon=30,
frequency="D"
)
mlclient.jobs.createorupdate(forecastingjob)
Model Registration
1. Register Model
from azure.ai.ml.entities import Model
from azure.ai.ml.constants import AssetTypes
Register from job output
model = Model(
path=f"azureml://jobs/{returnedjob.name}/outputs/model/",
name="sklearn-classifier",
description="Random Forest classifier",
type=AssetTypes.CUSTOMMODEL
)
registeredmodel = mlclient.models.createorupdate(model)
print(f"Model registered: {registeredmodel.name}:{registeredmodel.version}")
2. Register MLflow Model
mlflowmodel = Model(
path="runs:/run
id/model",
name="mlflow-model",
type=AssetTypes.MLFLOWMODEL,
description="MLflow logged model"
)
mlclient.models.createorupdate(mlflowmodel)
3. List Models
# List all models
models = mlclient.models.list()
for model in models:
print(f"{model.name}: {model.latestversion}")
Get specific version
model = mlclient.models.get("sklearn-classifier", version="1")
print(f"Model path: {model.path}")
Model Deployment
1. Online Endpoint
from azure.ai.ml.entities import (
ManagedOnlineEndpoint,
ManagedOnlineDeployment,
Model,
CodeConfiguration
)
Create endpoint
endpoint = ManagedOnlineEndpoint(
name="sklearn-endpoint",
description="Sklearn model endpoint",
authmode="key"
)
mlclient.onlineendpoints.begincreateorupdate(endpoint).result()
print("Endpoint created")
Create deployment
deployment = ManagedOnlineDeployment(
name="blue",
endpointname="sklearn-endpoint",
model="sklearn-classifier:1",
instancetype="StandardDS3v2",
instancecount=1,
codeconfiguration=CodeConfiguration(
code="./scoring",
scoringscript="score.py"
),
environment="AzureML-sklearn-1.0-ubuntu20.04-py38-cpu@latest"
)
mlclient.onlinedeployments.begincreateorupdate(deployment).result()
Set traffic
endpoint.traffic = {"blue": 100}
mlclient.onlineendpoints.begincreateorupdate(endpoint).result()
2. Scoring Script
# scoring/score.py
import json
import joblib
import numpy as np
import os
def init():
global model
modelpath = os.path.join(os.getenv("AZUREMLMODELDIR"), "model.joblib")
model = joblib.load(modelpath)
def run(rawdata):
try:
data = json.loads(rawdata)
features = np.array(data["features"])
predictions = model.predict(features)
return {"predictions": predictions.tolist()}
except Exception as e:
return {"error": str(e)}
3. Test Endpoint
import json
Prepare test data
testdata = {
"features": [[5.1, 3.5, 1.4, 0.2], [6.2, 3.4, 5.4, 2.3]]
}
Invoke endpoint
response = mlclient.onlineendpoints.invoke(
endpointname="sklearn-endpoint",
requestfile=json.dumps(testdata)
)
print(f"Response: {response}")
Batch Endpoints
1. Create Batch Endpoint
from azure.ai.ml.entities import BatchEndpoint, BatchDeployment
Create endpoint
batchendpoint = BatchEndpoint(
name="batch-sklearn-endpoint",
description="Batch inference endpoint"
)
mlclient.batchendpoints.begincreateorupdate(batchendpoint).result()
Create deployment
batchdeployment = BatchDeployment(
name="batch-deployment",
endpointname="batch-sklearn-endpoint",
model="sklearn-classifier:1",
compute="cpu-cluster",
instancecount=2,
maxconcurrencyperinstance=2,
minibatchsize=10,
outputaction="appendrow",
outputfilename="predictions.csv",
codeconfiguration=CodeConfiguration(
code="./batch-scoring",
scoringscript="batchscore.py"
),
environment="AzureML-sklearn-1.0-ubuntu20.04-py38-cpu@latest"
)
mlclient.batchdeployments.begincreateorupdate(batchdeployment).result()
2. Invoke Batch Endpoint
from azure.ai.ml import Input
Start batch job
job = mlclient.batchendpoints.invoke(
endpointname="batch-sklearn-endpoint",
input=Input(
path="azureml://datastores/workspaceblobstore/paths/batch-data/",
type="urifolder"
)
)
print(f"Batch job started: {job.name}")
Monitor job
mlclient.jobs.stream(job.name)
MLflow Integration
1. Track Experiments
import mlflow
from azure.ai.ml import MLClient
Set tracking URI
mlclient = MLClient.fromconfig()
mlflow.settrackinguri(mlclient.workspaces.get().mlflowtrackinguri)
Start experiment
mlflow.setexperiment("my-experiment")
with mlflow.startrun():
# Log parameters
mlflow.logparam("learningrate", 0.01)
mlflow.logparam("epochs", 100)
# Train model
# ...
# Log metrics
mlflow.logmetric("accuracy", 0.95)
mlflow.logmetric("f1score", 0.93)
# Log model
mlflow.sklearn.logmodel(model, "model")
# Log artifacts
mlflow.logartifact("plots/confusionmatrix.png")
2. Query Runs
# Search runs
runs = mlflow.searchruns(
experimentnames=["my-experiment"],
filterstring="metrics.accuracy > 0.9",
orderby=["metrics.accuracy DESC"]
)
print(runs[["runid", "metrics.accuracy", "params.learningrate"]])
Best Practices
1. Resource Cleanup
# Delete endpoint
mlclient.onlineendpoints.begindelete("sklearn-endpoint").result()
Delete compute
mlclient.compute.begindelete("cpu-cluster").result()
Delete model
mlclient.models.archive("sklearn-classifier")
2. Cost Management
# Auto-scale compute cluster
cluster = AmlCompute(
name="cost-optimized-cluster",
size="StandardDS3v2",
mininstances=0, # Scale to zero when idle
maxinstances=4,
idletimebeforescaledown=120,
tier="LowPriority" # Use spot instances
)
Conclusion
Azure Machine Learning provides:
Key takeaways:
- Use compute clusters for scalability
- Leverage AutoML for rapid prototyping
- Track experiments with MLflow
- Deploy with managed endpoints
- Monitor costs with auto-scaling