Complete Azure OpenAI Service Tutorial: Enterprise AI with GPT Models
Azure OpenAI Service provides REST API access to OpenAI's powerful language models including GPT-4, GPT-3.5-Turbo, and embedding models. It combines OpenAI's capabilities with Azure's enterprise security and compliance.
Why Azure OpenAI?
Key Benefits:- Enterprise security: Azure security and compliance
- Data privacy: Your data stays in your Azure subscription
- Regional availability: Deploy in multiple Azure regions
- Integration: Native Azure service integration
- Responsible AI: Built-in content filtering
- GPT-4 and GPT-4 Turbo
- GPT-3.5-Turbo
- DALL-E 3
- Embeddings (text-embedding-ada-002)
- Whisper
Prerequisites
pip install openai azure-identity
Azure CLI
az login
Setup
1. Create Azure OpenAI Resource
from azure.mgmt.cognitiveservices import CognitiveServicesManagementClient
from azure.identity import DefaultAzureCredential
credential = DefaultAzureCredential()
client = CognitiveServicesManagementClient(
credential=credential,
subscriptionid="your-subscription-id"
)
Create resource
resource = client.accounts.begincreate(
resourcegroupname="my-resource-group",
accountname="my-openai-resource",
account={
"location": "eastus",
"kind": "OpenAI",
"sku": {"name": "S0"},
"properties": {}
}
).result()
print(f"Resource created: {resource.name}")
2. Deploy Model
# Deploy GPT-4 model
deployment = client.deployments.begincreateorupdate(
resourcegroupname="my-resource-group",
accountname="my-openai-resource",
deploymentname="gpt-4-deployment",
deployment={
"sku": {"name": "Standard", "capacity": 10},
"properties": {
"model": {
"format": "OpenAI",
"name": "gpt-4",
"version": "0613"
}
}
}
).result()
print(f"Deployment created: {deployment.name}")
3. Connect to Azure OpenAI
from openai import AzureOpenAI
client = AzureOpenAI(
apikey="your-api-key",
apiversion="2024-02-01",
azureendpoint="https://my-openai-resource.openai.azure.com"
)
Or use Azure Identity
from azure.identity import DefaultAzureCredential, getbearertokenprovider
tokenprovider = getbearertokenprovider(
DefaultAzureCredential(),
"https://cognitiveservices.azure.com/.default"
)
client = AzureOpenAI(
azureadtokenprovider=tokenprovider,
apiversion="2024-02-01",
azureendpoint="https://my-openai-resource.openai.azure.com"
)
Chat Completions
1. Basic Chat
response = client.chat.completions.create(
model="gpt-4-deployment",
messages=[
{"role": "system", "content": "You are a helpful assistant."},
{"role": "user", "content": "What is machine learning?"}
]
)
print(response.choices[0].message.content)
2. Multi-turn Conversation
class ChatBot:
def init(self, client, deploymentname, systemprompt):
self.client = client
self.deploymentname = deploymentname
self.messages = [{"role": "system", "content": systemprompt}]
def chat(self, usermessage):
self.messages.append({"role": "user", "content": usermessage})
response = self.client.chat.completions.create(
model=self.deploymentname,
messages=self.messages,
temperature=0.7,
maxtokens=1000
)
assistantmessage = response.choices[0].message.content
self.messages.append({"role": "assistant", "content": assistantmessage})
return assistantmessage
def clearhistory(self):
self.messages = [self.messages[0]] # Keep system prompt
Usage
bot = ChatBot(client, "gpt-4-deployment", "You are an ML expert.")
print(bot.chat("What is deep learning?"))
print(bot.chat("Can you give an example?"))
3. Streaming Response
def streamchat(messages):
stream = client.chat.completions.create(
model="gpt-4-deployment",
messages=messages,
stream=True
)
for chunk in stream:
if chunk.choices[0].delta.content:
print(chunk.choices[0].delta.content, end="", flush=True)
print()
streamchat([
{"role": "user", "content": "Write a poem about AI"}
])
4. Function Calling
import json
Define functions
tools = [
{
"type": "function",
"function": {
"name": "getweather",
"description": "Get weather for a location",
"parameters": {
"type": "object",
"properties": {
"location": {
"type": "string",
"description": "City name"
},
"unit": {
"type": "string",
"enum": ["celsius", "fahrenheit"]
}
},
"required": ["location"]
}
}
}
]
Call with function
response = client.chat.completions.create(
model="gpt-4-deployment",
messages=[
{"role": "user", "content": "What's the weather in Tokyo?"}
],
tools=tools,
toolchoice="auto"
)
Check if function was called
message = response.choices[0].message
if message.toolcalls:
toolcall = message.toolcalls[0]
functionname = toolcall.function.name
arguments = json.loads(toolcall.function.arguments)
print(f"Function: {functionname}")
print(f"Arguments: {arguments}")
# Simulate function response
functionresponse = {"temperature": 22, "condition": "sunny"}
# Continue conversation with function result
messages = [
{"role": "user", "content": "What's the weather in Tokyo?"},
message,
{
"role": "tool",
"toolcallid": toolcall.id,
"content": json.dumps(functionresponse)
}
]
finalresponse = client.chat.completions.create(
model="gpt-4-deployment",
messages=messages
)
print(finalresponse.choices[0].message.content)
Embeddings
1. Generate Embeddings
def getembedding(text):
response = client.embeddings.create(
model="text-embedding-ada-002",
input=text
)
return response.data[0].embedding
Single text
embedding = getembedding("Machine learning is fascinating")
print(f"Embedding dimension: {len(embedding)}")
Multiple texts
texts = ["Hello world", "AI is powerful", "Data science"]
response = client.embeddings.create(
model="text-embedding-ada-002",
input=texts
)
embeddings = [item.embedding for item in response.data]
2. Semantic Search
import numpy as np
def cosinesimilarity(a, b):
return np.dot(a, b) / (np.linalg.norm(a) np.linalg.norm(b))
class SemanticSearch:
def init(self, client, deploymentname):
self.client = client
self.deploymentname = deploymentname
self.documents = []
self.embeddings = []
def adddocuments(self, documents):
response = self.client.embeddings.create(
model=self.deploymentname,
input=documents
)
for i, doc in enumerate(documents):
self.documents.append(doc)
self.embeddings.append(response.data[i].embedding)
def search(self, query, topk=3):
queryresponse = self.client.embeddings.create(
model=self.deploymentname,
input=query
)
queryembedding = queryresponse.data[0].embedding
similarities = [
cosinesimilarity(queryembedding, emb)
for emb in self.embeddings
]
topindices = np.argsort(similarities)[-topk:][::-1]
return [(self.documents[i], similarities[i]) for i in topindices]
Usage
search = SemanticSearch(client, "text-embedding-ada-002")
search.adddocuments([
"Machine learning is a subset of AI",
"Deep learning uses neural networks",
"Python is popular for data science"
])
results = search.search("What is ML?")
for doc, score in results:
print(f"{score:.3f}: {doc}")
RAG (Retrieval Augmented Generation)
1. Simple RAG
class RAGSystem:
def init(self, client, chatdeployment, embeddingdeployment):
self.client = client
self.chatdeployment = chatdeployment
self.embeddingdeployment = embeddingdeployment
self.documents = []
self.embeddings = []
def adddocuments(self, documents):
response = self.client.embeddings.create(
model=self.embeddingdeployment,
input=documents
)
for i, doc in enumerate(documents):
self.documents.append(doc)
self.embeddings.append(response.data[i].embedding)
def retrieve(self, query, topk=3):
queryresponse = self.client.embeddings.create(
model=self.embeddingdeployment,
input=query
)
queryembedding = queryresponse.data[0].embedding
similarities = [
cosinesimilarity(queryembedding, emb)
for emb in self.embeddings
]
topindices = np.argsort(similarities)[-topk:][::-1]
return [self.documents[i] for i in topindices]
def query(self, question):
# Retrieve relevant documents
relevantdocs = self.retrieve(question)
context = "\n".join(relevantdocs)
# Generate answer
response = self.client.chat.completions.create(
model=self.chatdeployment,
messages=[
{
"role": "system",
"content": f"Answer based on this context:\n{context}"
},
{"role": "user", "content": question}
]
)
return response.choices[0].message.content
Usage
rag = RAGSystem(client, "gpt-4-deployment", "text-embedding-ada-002")
rag.adddocuments([
"Our company was founded in 2020",
"We have offices in New York and London",
"Our main product is an AI platform"
])
answer = rag.query("When was the company founded?")
print(answer)
2. RAG with Azure AI Search
from azure.search.documents import SearchClient
from azure.core.credentials import AzureKeyCredential
searchclient = SearchClient(
endpoint="https://my-search.search.windows.net",
indexname="documents",
credential=AzureKeyCredential("search-api-key")
)
def ragwithsearch(question):
# Search for relevant documents
results = searchclient.search(
searchtext=question,
top=5,
select=["content", "title"]
)
# Build context
context = "\n".join([doc["content"] for doc in results])
# Generate answer
response = client.chat.completions.create(
model="gpt-4-deployment",
messages=[
{
"role": "system",
"content": f"Answer based on:\n{context}"
},
{"role": "user", "content": question}
]
)
return response.choices[0].message.content
Image Generation with DALL-E
1. Generate Images
response = client.images.generate(
model="dall-e-3",
prompt="A futuristic city with flying cars, digital art style",
n=1,
size="1024x1024",
quality="hd",
style="vivid"
)
imageurl = response.data[0].url
print(f"Image URL: {imageurl}")
Download image
import requests
imageresponse = requests.get(imageurl)
with open("generatedimage.png", "wb") as f:
f.write(imageresponse.content)
2. Image Variations
# Generate variation of existing image
with open("originalimage.png", "rb") as f:
response = client.images.createvariation(
model="dall-e-2",
image=f,
n=1,
size="1024x1024"
)
variationurl = response.data[0].url
Vision with GPT-4V
1. Analyze Image
import base64
def encodeimage(imagepath):
with open(imagepath, "rb") as f:
return base64.b64encode(f.read()).decode()
response = client.chat.completions.create(
model="gpt-4-vision-deployment",
messages=[
{
"role": "user",
"content": [
{"type": "text", "text": "What's in this image?"},
{
"type": "imageurl",
"imageurl": {
"url": f"data:image/png;base64,{encodeimage('image.png')}"
}
}
]
}
],
maxtokens=500
)
print(response.choices[0].message.content)
2. Analyze Multiple Images
response = client.chat.completions.create(
model="gpt-4-vision-deployment",
messages=[
{
"role": "user",
"content": [
{"type": "text", "text": "Compare these two images:"},
{
"type": "imageurl",
"imageurl": {"url": f"data:image/png;base64,{encodeimage('image1.png')}"}
},
{
"type": "imageurl",
"imageurl": {"url": f"data:image/png;base64,{encodeimage('image2.png')}"}
}
]
}
]
)
Best Practices
1. Error Handling
from openai import APIError, RateLimitError
import time
def callwithretry(func, maxretries=3):
for attempt in range(maxretries):
try:
return func()
except RateLimitError:
waittime = 2 attempt
print(f"Rate limited, waiting {waittime}s...")
time.sleep(waittime)
except APIError as e:
print(f"API error: {e}")
raise
raise Exception("Max retries exceeded")
Usage
result = callwithretry(
lambda: client.chat.completions.create(
model="gpt-4-deployment",
messages=[{"role": "user", "content": "Hello"}]
)
)
2. Token Management
import tiktoken
def counttokens(text, model="gpt-4"):
encoding = tiktoken.encodingformodel(model)
return len(encoding.encode(text))
def truncatetotokens(text, maxtokens, model="gpt-4"):
encoding = tiktoken.encodingformodel(model)
tokens = encoding.encode(text)
if len(tokens) > maxtokens:
tokens = tokens[:maxtokens]
return encoding.decode(tokens)
Check token count
text = "Your long text here..."
print(f"Tokens: {counttokens(text)}")
Conclusion
Azure OpenAI Service provides:
Key takeaways:
- Use appropriate models for tasks
- Implement proper error handling
- Manage tokens effectively
- Use RAG for domain knowledge
- Monitor usage and costs