Complete AWS Bedrock Tutorial: Managed Generative AI on AWS
Amazon Bedrock is a fully managed service that provides access to foundation models (FMs) from leading AI companies through a unified API. It enables building generative AI applications without managing infrastructure.
Why AWS Bedrock?
Key Benefits:- Multiple FMs: Access Claude, Llama, Titan, and more
- Fully managed: No infrastructure to manage
- Secure: Data privacy and VPC support
- Customizable: Fine-tune models with your data
- Integrated: Native AWS service integration
- Anthropic Claude (Claude 3, Claude 2)
- Meta Llama 2
- Amazon Titan
- AI21 Labs Jurassic
- Cohere Command
- Stability AI (images)
Prerequisites
pip install boto3
Configure AWS CLI
aws configure
Enable Bedrock model access in AWS Console
Quick Start
1. Basic Text Generation
import boto3
import json
Create Bedrock runtime client
bedrock = boto3.client(
servicename="bedrock-runtime",
regionname="us-east-1"
)
Invoke Claude model
def generatetext(prompt):
body = json.dumps({
"anthropicversion": "bedrock-2023-05-31",
"maxtokens": 1024,
"messages": [
{"role": "user", "content": prompt}
]
})
response = bedrock.invokemodel(
modelId="anthropic.claude-3-sonnet-20240229-v1:0",
body=body
)
result = json.loads(response["body"].read())
return result["content"][0]["text"]
Generate text
response = generatetext("Explain machine learning in simple terms.")
print(response)
2. Streaming Response
def generatetextstreaming(prompt):
body = json.dumps({
"anthropic
version": "bedrock-2023-05-31",
"maxtokens": 1024,
"messages": [
{"role": "user", "content": prompt}
]
})
response = bedrock.invokemodelwithresponsestream(
modelId="anthropic.claude-3-sonnet-20240229-v1:0",
body=body
)
for event in response["body"]:
chunk = json.loads(event["chunk"]["bytes"])
if chunk["type"] == "contentblockdelta":
print(chunk["delta"]["text"], end="", flush=True)
generatetextstreaming("Write a short poem about AI.")
Working with Different Models
1. Amazon Titan
def invoketitan(prompt):
body = json.dumps({
"inputText": prompt,
"textGenerationConfig": {
"maxTokenCount": 1024,
"temperature": 0.7,
"topP": 0.9
}
})
response = bedrock.invokemodel(
modelId="amazon.titan-text-express-v1",
body=body
)
result = json.loads(response["body"].read())
return result["results"][0]["outputText"]
response = invoketitan("What is cloud computing?")
print(response)
2. Meta Llama 2
def invokellama(prompt):
body = json.dumps({
"prompt": f"[INST] {prompt} [/INST]",
"max
genlen": 512,
"temperature": 0.7,
"top
p": 0.9
})
response = bedrock.invokemodel(
modelId="meta.llama2-70b-chat-v1",
body=body
)
result = json.loads(response["body"].read())
return result["generation"]
response = invokellama("Explain neural networks.")
print(response)
3. Cohere Command
def invokecohere(prompt):
body = json.dumps({
"prompt": prompt,
"max
tokens": 500,
"temperature": 0.7
})
response = bedrock.invokemodel(
modelId="cohere.command-text-v14",
body=body
)
result = json.loads(response["body"].read())
return result["generations"][0]["text"]
Embeddings
1. Titan Embeddings
def getembeddings(text):
body = json.dumps({
"inputText": text
})
response = bedrock.invokemodel(
modelId="amazon.titan-embed-text-v1",
body=body
)
result = json.loads(response["body"].read())
return result["embedding"]
Get embeddings
embedding = getembeddings("Machine learning is fascinating.")
print(f"Embedding dimension: {len(embedding)}")
2. Cohere Embeddings
def getcohereembeddings(texts):
body = json.dumps({
"texts": texts,
"inputtype": "searchdocument"
})
response = bedrock.invokemodel(
modelId="cohere.embed-english-v3",
body=body
)
result = json.loads(response["body"].read())
return result["embeddings"]
embeddings = getcohereembeddings(["Hello world", "AI is powerful"])
Image Generation
1. Stability AI
import base64
def generateimage(prompt):
body = json.dumps({
"textprompts": [{"text": prompt}],
"cfgscale": 7,
"steps": 50,
"seed": 42
})
response = bedrock.invokemodel(
modelId="stability.stable-diffusion-xl-v1",
body=body
)
result = json.loads(response["body"].read())
imagedata = base64.b64decode(result["artifacts"][0]["base64"])
with open("generatedimage.png", "wb") as f:
f.write(imagedata)
return "generatedimage.png"
imagepath = generateimage("A futuristic city with flying cars")
print(f"Image saved to: {imagepath}")
2. Amazon Titan Image
def generatetitanimage(prompt):
body = json.dumps({
"taskType": "TEXTIMAGE",
"textToImageParams": {
"text": prompt
},
"imageGenerationConfig": {
"numberOfImages": 1,
"quality": "standard",
"height": 512,
"width": 512
}
})
response = bedrock.invokemodel(
modelId="amazon.titan-image-generator-v1",
body=body
)
result = json.loads(response["body"].read())
return base64.b64decode(result["images"][0])
Conversation and Chat
1. Multi-turn Conversation
class BedrockChat:
def init(self, modelid="anthropic.claude-3-sonnet-20240229-v1:0"):
self.bedrock = boto3.client("bedrock-runtime", regionname="us-east-1")
self.modelid = modelid
self.messages = []
self.systemprompt = None
def setsystemprompt(self, prompt):
self.systemprompt = prompt
def chat(self, usermessage):
self.messages.append({"role": "user", "content": usermessage})
body = {
"anthropicversion": "bedrock-2023-05-31",
"maxtokens": 1024,
"messages": self.messages
}
if self.systemprompt:
body["system"] = self.systemprompt
response = self.bedrock.invokemodel(
modelId=self.modelid,
body=json.dumps(body)
)
result = json.loads(response["body"].read())
assistantmessage = result["content"][0]["text"]
self.messages.append({"role": "assistant", "content": assistantmessage})
return assistantmessage
def clearhistory(self):
self.messages = []
Usage
chat = BedrockChat()
chat.setsystemprompt("You are a helpful AI assistant specializing in ML.")
response1 = chat.chat("What is deep learning?")
print(f"Assistant: {response1}\n")
response2 = chat.chat("Can you give me an example?")
print(f"Assistant: {response2}")
RAG with Bedrock
1. Knowledge Bases
bedrockagent = boto3.client("bedrock-agent-runtime", regionname="us-east-1")
def queryknowledgebase(query, kbid):
response = bedrockagent.retrieveandgenerate(
input={"text": query},
retrieveAndGenerateConfiguration={
"type": "KNOWLEDGEBASE",
"knowledgeBaseConfiguration": {
"knowledgeBaseId": kbid,
"modelArn": "arn:aws:bedrock:us-east-1::foundation-model/anthropic.claude-3-sonnet-20240229-v1:0"
}
}
)
return response["output"]["text"]
Query knowledge base
answer = queryknowledgebase(
"What are the main features of our product?",
"KNOWLEDGEBASEID"
)
print(answer)
2. Custom RAG Pipeline
import numpy as np
class BedrockRAG:
def init(self):
self.bedrock = boto3.client("bedrock-runtime", regionname="us-east-1")
self.documents = []
self.embeddings = []
def adddocuments(self, documents):
for doc in documents:
embedding = self.getembedding(doc)
self.documents.append(doc)
self.embeddings.append(embedding)
def getembedding(self, text):
body = json.dumps({"inputText": text})
response = self.bedrock.invokemodel(
modelId="amazon.titan-embed-text-v1",
body=body
)
result = json.loads(response["body"].read())
return np.array(result["embedding"])
def cosinesimilarity(self, a, b):
return np.dot(a, b) / (np.linalg.norm(a) * np.linalg.norm(b))
def retrieve(self, query, topk=3):
queryembedding = self.getembedding(query)
similarities = [
self.cosinesimilarity(queryembedding, emb)
for emb in self.embeddings
]
topindices = np.argsort(similarities)[-topk:][::-1]
return [self.documents[i] for i in topindices]
def query(self, question):
relevantdocs = self.retrieve(question)
context = "\n".join(relevantdocs)
prompt = f"""Based on the following context, answer the question.
Context:
{context}
Question: {question}
Answer:"""
body = json.dumps({
"anthropicversion": "bedrock-2023-05-31",
"maxtokens": 1024,
"messages": [{"role": "user", "content": prompt}]
})
response = self.bedrock.invokemodel(
modelId="anthropic.claude-3-sonnet-20240229-v1:0",
body=body
)
result = json.loads(response["body"].read())
return result["content"][0]["text"]
Usage
rag = BedrockRAG()
rag.adddocuments([
"Our product supports Python 3.8 and above.",
"The API rate limit is 1000 requests per minute.",
"Premium users get priority support."
])
answer = rag.query("What Python versions are supported?")
print(answer)
Bedrock Agents
1. Create Agent
bedrockagentclient = boto3.client("bedrock-agent", regionname="us-east-1")
def createagent(agentname, instructions):
response = bedrockagentclient.createagent(
agentName=agentname,
foundationModel="anthropic.claude-3-sonnet-20240229-v1:0",
instruction=instructions,
agentResourceRoleArn="arn:aws:iam::123456789:role/BedrockAgentRole"
)
return response["agent"]["agentId"]
agentid = createagent(
"customer-service-agent",
"You are a customer service agent. Help users with their inquiries."
)
2. Invoke Agent
def invokeagent(agentid, sessionid, prompt):
response = bedrock
agent.invokeagent(
agentId=agent
id,
agentAliasId="TSTALIASID",
sessionId=sessionid,
inputText=prompt
)
completion = ""
for event in response["completion"]:
if "chunk" in event:
completion += event["chunk"]["bytes"].decode()
return completion
Model Customization
1. Fine-tuning
bedrockclient = boto3.client("bedrock", regionname="us-east-1")
def create
finetuningjob(jobname, modelid, trainingdatauri):
response = bedrockclient.createmodelcustomizationjob(
jobName=jobname,
customModelName=f"custom-{jobname}",
roleArn="arn:aws:iam::123456789:role/BedrockFineTuningRole",
baseModelIdentifier=modelid,
trainingDataConfig={
"s3Uri": trainingdatauri
},
outputDataConfig={
"s3Uri": "s3://bucket/fine-tuned-models/"
},
hyperParameters={
"epochCount": "3",
"batchSize": "8",
"learningRate": "0.00001"
}
)
return response["jobArn"]
2. Continued Pre-training
def createcontinuedpretrainingjob(jobname, modelid, datauri):
response = bedrock
client.createmodelcustomizationjob(
jobName=job
name,
customModelName=f"pretrained-{jobname}",
roleArn="arn:aws:iam::123456789:role/BedrockRole",
baseModelIdentifier=modelid,
customizationType="CONTINUEDPRETRAINING",
trainingDataConfig={
"s3Uri": datauri
},
outputDataConfig={
"s3Uri": "s3://bucket/pretrained-models/"
}
)
return response["jobArn"]
Guardrails
1. Create Guardrail
def createguardrail(name, blockedtopics, wordfilters):
response = bedrockclient.createguardrail(
name=name,
description="Content filtering guardrail",
topicPolicyConfig={
"topicsConfig": [
{
"name": topic,
"definition": f"Content related to {topic}",
"type": "DENY"
}
for topic in blockedtopics
]
},
wordPolicyConfig={
"wordsConfig": [
{"text": word} for word in wordfilters
]
},
blockedInputMessaging="Your request was blocked.",
blockedOutputsMessaging="Response was filtered."
)
return response["guardrailId"]
guardrailid = createguardrail(
"content-filter",
["violence", "illegal activities"],
["profanity1", "profanity2"]
)
2. Use Guardrail
def invokewithguardrail(prompt, guardrailid):
body = json.dumps({
"anthropic
version": "bedrock-2023-05-31",
"maxtokens": 1024,
"messages": [{"role": "user", "content": prompt}]
})
response = bedrock.invokemodel(
modelId="anthropic.claude-3-sonnet-20240229-v1:0",
body=body,
guardrailIdentifier=guardrailid,
guardrailVersion="DRAFT"
)
return json.loads(response["body"].read())
Best Practices
1. Error Handling
from botocore.exceptions import ClientError
def safeinvoke(prompt, modelid):
try:
body = json.dumps({
"anthropicversion": "bedrock-2023-05-31",
"maxtokens": 1024,
"messages": [{"role": "user", "content": prompt}]
})
response = bedrock.invokemodel(modelId=modelid, body=body)
return json.loads(response["body"].read())
except ClientError as e:
errorcode = e.response["Error"]["Code"]
if errorcode == "ThrottlingException":
print("Rate limited, retrying...")
elif errorcode == "ModelNotReadyException":
print("Model not ready")
raise
2. Cost Optimization
# Use appropriate model for task
def selectmodel(taskcomplexity):
if taskcomplexity == "simple":
return "amazon.titan-text-lite-v1"
elif taskcomplexity == "medium":
return "anthropic.claude-3-haiku-20240307-v1:0"
else:
return "anthropic.claude-3-sonnet-20240229-v1:0"
Conclusion
AWS Bedrock provides:
Key takeaways:
- Choose model based on use case
- Implement proper error handling
- Use guardrails for content filtering
- Leverage knowledge bases for RAG
- Monitor costs and usage