Complete Azure ML Managed Endpoints Tutorial: Production Model Deployment
Azure ML Managed Endpoints provide a fully managed solution for deploying machine learning models at scale. They handle infrastructure, scaling, security, and monitoring automatically.
Why Managed Endpoints?
Key Benefits:- Fully managed: No infrastructure to manage
- Auto-scaling: Scale based on traffic
- Blue-green deployments: Safe rollouts
- Built-in monitoring: Metrics and logging
- Security: Authentication and network isolation
- Online endpoints: Real-time inference
- Batch endpoints: Large-scale batch processing
Prerequisites
pip install azure-ai-ml azure-identity
Azure CLI
az login
az extension add -n ml
Online Endpoints
1. Create Online Endpoint
from azure.ai.ml import MLClient
from azure.ai.ml.entities import ManagedOnlineEndpoint
from azure.identity import DefaultAzureCredential
mlclient = MLClient(
credential=DefaultAzureCredential(),
subscriptionid="your-subscription-id",
resourcegroupname="my-resource-group",
workspacename="my-ml-workspace"
)
Create endpoint
endpoint = ManagedOnlineEndpoint(
name="my-online-endpoint",
description="Online endpoint for real-time inference",
authmode="key", # or "amltoken"
tags={"environment": "production"}
)
mlclient.onlineendpoints.begincreateorupdate(endpoint).result()
print(f"Endpoint created: {endpoint.name}")
2. Create Deployment
from azure.ai.ml.entities import (
ManagedOnlineDeployment,
Model,
Environment,
CodeConfiguration
)
Create deployment
bluedeployment = ManagedOnlineDeployment(
name="blue",
endpointname="my-online-endpoint",
model=Model(path="./model"),
codeconfiguration=CodeConfiguration(
code="./scoring",
scoringscript="score.py"
),
environment=Environment(
condafile="./environment.yml",
image="mcr.microsoft.com/azureml/openmpi4.1.0-ubuntu20.04:latest"
),
instancetype="StandardDS3v2",
instancecount=1
)
mlclient.onlinedeployments.begincreateorupdate(bluedeployment).result()
print("Deployment created")
3. Scoring Script
# scoring/score.py
import json
import joblib
import numpy as np
import os
import logging
def init():
"""Initialize model on startup."""
global model
modelpath = os.path.join(os.getenv("AZUREMLMODELDIR"), "model.joblib")
model = joblib.load(modelpath)
logging.info("Model loaded successfully")
def run(rawdata):
"""Run inference on incoming data."""
try:
data = json.loads(rawdata)
features = np.array(data["features"])
# Run prediction
predictions = model.predict(features)
probabilities = model.predictproba(features)
return {
"predictions": predictions.tolist(),
"probabilities": probabilities.tolist()
}
except Exception as e:
logging.error(f"Error: {str(e)}")
return {"error": str(e)}
4. Set Traffic
# Route all traffic to blue deployment
endpoint.traffic = {"blue": 100}
mlclient.onlineendpoints.begincreateorupdate(endpoint).result()
Get endpoint details
endpoint = mlclient.onlineendpoints.get("my-online-endpoint")
print(f"Scoring URI: {endpoint.scoringuri}")
print(f"Traffic: {endpoint.traffic}")
5. Test Endpoint
import json
Test data
testdata = {
"features": [[5.1, 3.5, 1.4, 0.2], [6.2, 3.4, 5.4, 2.3]]
}
Invoke endpoint
response = mlclient.onlineendpoints.invoke(
endpointname="my-online-endpoint",
requestfile=json.dumps(testdata)
)
print(f"Response: {response}")
6. Invoke with REST API
import requests
Get endpoint details
endpoint = mlclient.onlineendpoints.get("my-online-endpoint")
scoringuri = endpoint.scoringuri
Get key
keys = mlclient.onlineendpoints.getkeys("my-online-endpoint")
apikey = keys.primarykey
Make request
headers = {
"Content-Type": "application/json",
"Authorization": f"Bearer {apikey}"
}
data = {"features": [[5.1, 3.5, 1.4, 0.2]]}
response = requests.post(scoringuri, headers=headers, json=data)
print(response.json())
Blue-Green Deployments
1. Create Green Deployment
# Create new deployment with updated model
greendeployment = ManagedOnlineDeployment(
name="green",
endpointname="my-online-endpoint",
model=Model(path="./modelv2"),
codeconfiguration=CodeConfiguration(
code="./scoring",
scoringscript="score.py"
),
environment=Environment(
condafile="./environment.yml",
image="mcr.microsoft.com/azureml/openmpi4.1.0-ubuntu20.04:latest"
),
instancetype="StandardDS3v2",
instancecount=1
)
mlclient.onlinedeployments.begincreateorupdate(greendeployment).result()
2. Gradual Traffic Shift
# Start with 10% traffic to green
endpoint.traffic = {"blue": 90, "green": 10}
mlclient.onlineendpoints.begincreateorupdate(endpoint).result()
Monitor metrics, then increase
endpoint.traffic = {"blue": 50, "green": 50}
mlclient.onlineendpoints.begincreateorupdate(endpoint).result()
Full rollout to green
endpoint.traffic = {"blue": 0, "green": 100}
mlclient.onlineendpoints.begincreateorupdate(endpoint).result()
3. Rollback
# Rollback to blue if issues detected
endpoint.traffic = {"blue": 100, "green": 0}
mlclient.onlineendpoints.begincreateorupdate(endpoint).result()
Delete failed deployment
mlclient.onlinedeployments.begindelete(
endpointname="my-online-endpoint",
deploymentname="green"
).result()
Auto-Scaling
1. Configure Auto-Scaling
from azure.ai.ml.entities import (
ManagedOnlineDeployment,
OnlineRequestSettings,
ProbeSettings
)
Deployment with scaling configuration
scaleddeployment = ManagedOnlineDeployment(
name="scaled",
endpointname="my-online-endpoint",
model=Model(path="./model"),
codeconfiguration=CodeConfiguration(
code="./scoring",
scoringscript="score.py"
),
environment=Environment(
condafile="./environment.yml",
image="mcr.microsoft.com/azureml/openmpi4.1.0-ubuntu20.04:latest"
),
instancetype="StandardDS3v2",
instancecount=2,
requestsettings=OnlineRequestSettings(
requesttimeoutms=90000,
maxconcurrentrequestsperinstance=10
),
livenessprobe=ProbeSettings(
initialdelay=10,
period=10,
failurethreshold=30
),
readinessprobe=ProbeSettings(
initialdelay=10,
period=10,
failurethreshold=30
)
)
mlclient.onlinedeployments.begincreateorupdate(scaleddeployment).result()
2. Azure Monitor Auto-Scale
from azure.mgmt.monitor import MonitorManagementClient
from azure.mgmt.monitor.models import (
AutoscaleSettingResource,
AutoscaleProfile,
ScaleRule,
MetricTrigger,
ScaleAction,
ScaleCapacity
)
Configure auto-scale via Azure Monitor
monitorclient = MonitorManagementClient(
credential=DefaultAzureCredential(),
subscriptionid="your-subscription-id"
)
autoscalesetting = AutoscaleSettingResource(
location="eastus",
profiles=[
AutoscaleProfile(
name="auto-scale-profile",
capacity=ScaleCapacity(
minimum="1",
maximum="10",
default="2"
),
rules=[
ScaleRule(
metrictrigger=MetricTrigger(
metricname="RequestsPerInstance",
metricresourceuri="/subscriptions/.../endpoints/my-endpoint",
timegrain="PT1M",
statistic="Average",
timewindow="PT5M",
timeaggregation="Average",
operator="GreaterThan",
threshold=100
),
scaleaction=ScaleAction(
direction="Increase",
type="ChangeCount",
value="1",
cooldown="PT5M"
)
),
ScaleRule(
metrictrigger=MetricTrigger(
metricname="RequestsPerInstance",
operator="LessThan",
threshold=20
),
scaleaction=ScaleAction(
direction="Decrease",
type="ChangeCount",
value="1",
cooldown="PT10M"
)
)
]
)
],
targetresourceuri="/subscriptions/.../deployments/scaled"
)
Batch Endpoints
1. Create Batch Endpoint
from azure.ai.ml.entities import BatchEndpoint
batchendpoint = BatchEndpoint(
name="my-batch-endpoint",
description="Batch endpoint for large-scale inference"
)
mlclient.batchendpoints.begincreateorupdate(batchendpoint).result()
print("Batch endpoint created")
2. Create Batch Deployment
from azure.ai.ml.entities import (
BatchDeployment,
BatchRetrySettings,
CodeConfiguration
)
batchdeployment = BatchDeployment(
name="batch-scorer",
endpointname="my-batch-endpoint",
model=Model(path="./model"),
codeconfiguration=CodeConfiguration(
code="./batch-scoring",
scoringscript="batchscore.py"
),
environment=Environment(
condafile="./environment.yml",
image="mcr.microsoft.com/azureml/openmpi4.1.0-ubuntu20.04:latest"
),
compute="cpu-cluster",
instancecount=2,
maxconcurrencyperinstance=2,
minibatchsize=10,
outputaction="appendrow",
outputfilename="predictions.csv",
retrysettings=BatchRetrySettings(
maxretries=3,
timeout=300
),
logginglevel="info"
)
mlclient.batchdeployments.begincreateorupdate(batchdeployment).result()
3. Batch Scoring Script
# batch-scoring/batchscore.py
import os
import pandas as pd
import joblib
import logging
def init():
"""Initialize model."""
global model
model
path = os.path.join(os.getenv("AZUREMLMODELDIR"), "model.joblib")
model = joblib.load(modelpath)
logging.info("Model loaded")
def run(minibatch):
"""Process mini-batch of files."""
results = []
for filepath in minibatch:
# Read data
df = pd.readcsv(filepath)
# Predict
predictions = model.predict(df.values)
# Add predictions to results
for i, pred in enumerate(predictions):
results.append({
"file": os.path.basename(filepath),
"index": i,
"prediction": pred
})
return pd.DataFrame(results)
4. Invoke Batch Endpoint
from azure.ai.ml import Input
from azure.ai.ml.constants import AssetTypes
Start batch job
job = mlclient.batchendpoints.invoke(
endpointname="my-batch-endpoint",
inputs=Input(
path="azureml://datastores/workspaceblobstore/paths/batch-data/",
type=AssetTypes.URIFOLDER
)
)
print(f"Batch job started: {job.name}")
Monitor job
mlclient.jobs.stream(job.name)
Get output location
job = mlclient.jobs.get(job.name)
print(f"Output: {job.outputs}")
Monitoring
1. Get Logs
# Get deployment logs
logs = mlclient.onlinedeployments.getlogs(
endpointname="my-online-endpoint",
deploymentname="blue",
lines=100
)
print(logs)
2. Metrics
from azure.monitor.query import MetricsQueryClient
from datetime import datetime, timedelta
metricsclient = MetricsQueryClient(credential=DefaultAzureCredential())
Query metrics
response = metricsclient.queryresource(
resourceuri="/subscriptions/.../endpoints/my-online-endpoint",
metricnames=["RequestsPerMinute", "RequestLatency", "RequestsSucceeded"],
timespan=timedelta(hours=1)
)
for metric in response.metrics:
print(f"{metric.name}: {metric.timeseries[0].data[-1].average}")
3. Application Insights
# Enable Application Insights in deployment
deployment = ManagedOnlineDeployment(
name="monitored",
endpointname="my-online-endpoint",
model=Model(path="./model"),
# ... other config
appinsightsenabled=True
)
Security
1. Network Isolation
# Create endpoint with private endpoint
privateendpoint = ManagedOnlineEndpoint(
name="private-endpoint",
description="Private endpoint",
authmode="key",
publicnetworkaccess="disabled" # Disable public access
)
mlclient.onlineendpoints.begincreateorupdate(privateendpoint).result()
2. Managed Identity
from azure.ai.ml.entities import ManagedIdentityConfiguration
Create endpoint with managed identity
identityendpoint = ManagedOnlineEndpoint(
name="identity-endpoint",
authmode="amltoken",
identity=ManagedIdentityConfiguration(
type="SystemAssigned"
)
)
mlclient.onlineendpoints.begincreateorupdate(identityendpoint).result()
Best Practices
1. Health Checks
# Configure probes properly
deployment = ManagedOnlineDeployment(
name="healthy",
# ...
livenessprobe=ProbeSettings(
initialdelay=30, # Wait for model loading
period=10,
timeout=2,
failurethreshold=3
),
readinessprobe=ProbeSettings(
initialdelay=30,
period=10,
timeout=2,
successthreshold=1,
failurethreshold=3
)
)
2. Resource Cleanup
# Delete deployment
mlclient.onlinedeployments.begindelete(
endpointname="my-online-endpoint",
deploymentname="blue"
).result()
Delete endpoint
mlclient.onlineendpoints.begindelete("my-online-endpoint").result()
Conclusion
Azure ML Managed Endpoints provide:
Key takeaways:
- Use online endpoints for real-time inference
- Use batch endpoints for large-scale processing
- Implement blue-green deployments for safe rollouts
- Configure auto-scaling for traffic variations
- Monitor performance and health continuously