Complete AWS Bedrock Tutorial: Foundation Models on AWS

# Tutorial Lengkap AWS Bedrock: Managed Generative AI di AWS Amazon Bedrock adalah layanan terkelola penuh yang menyediakan akses ke foundation models (FMs) dari perusahaan AI terkemuka melalui API t...

By Ruby Abdullah · · tutorial
AWSBedrockLLMFoundation ModelsGenerative AIClaude

Complete AWS Bedrock Tutorial: Managed Generative AI on AWS

Amazon Bedrock is a fully managed service that provides access to foundation models (FMs) from leading AI companies through a unified API. It enables building generative AI applications without managing infrastructure.

Why AWS Bedrock?

Key Benefits:
  • Multiple FMs: Access Claude, Llama, Titan, and more
  • Fully managed: No infrastructure to manage
  • Secure: Data privacy and VPC support
  • Customizable: Fine-tune models with your data
  • Integrated: Native AWS service integration

Available Models:
  • Anthropic Claude (Claude 3, Claude 2)
  • Meta Llama 2
  • Amazon Titan
  • AI21 Labs Jurassic
  • Cohere Command
  • Stability AI (images)

Prerequisites

pip install boto3

Configure AWS CLI

aws configure

Enable Bedrock model access in AWS Console

Quick Start

1. Basic Text Generation

import boto3

import json

Create Bedrock runtime client

bedrock = boto3.client(

servicename="bedrock-runtime",

regionname="us-east-1"

)

Invoke Claude model

def generatetext(prompt):

body = json.dumps({

"anthropicversion": "bedrock-2023-05-31",

"maxtokens": 1024,

"messages": [

{"role": "user", "content": prompt}

]

})

response = bedrock.invokemodel(

modelId="anthropic.claude-3-sonnet-20240229-v1:0",

body=body

)

result = json.loads(response["body"].read())

return result["content"][0]["text"]

Generate text

response = generatetext("Explain machine learning in simple terms.")

print(response)

2. Streaming Response

def generatetextstreaming(prompt):

body = json.dumps({

"anthropicversion": "bedrock-2023-05-31",

"maxtokens": 1024,

"messages": [

{"role": "user", "content": prompt}

]

})

response = bedrock.invokemodelwithresponsestream(

modelId="anthropic.claude-3-sonnet-20240229-v1:0",

body=body

)

for event in response["body"]:

chunk = json.loads(event["chunk"]["bytes"])

if chunk["type"] == "contentblockdelta":

print(chunk["delta"]["text"], end="", flush=True)

generatetextstreaming("Write a short poem about AI.")

Working with Different Models

1. Amazon Titan

def invoketitan(prompt):

body = json.dumps({

"inputText": prompt,

"textGenerationConfig": {

"maxTokenCount": 1024,

"temperature": 0.7,

"topP": 0.9

}

})

response = bedrock.invokemodel(

modelId="amazon.titan-text-express-v1",

body=body

)

result = json.loads(response["body"].read())

return result["results"][0]["outputText"]

response = invoketitan("What is cloud computing?")

print(response)

2. Meta Llama 2

def invokellama(prompt):

body = json.dumps({

"prompt": f"[INST] {prompt} [/INST]",

"maxgenlen": 512,

"temperature": 0.7,

"topp": 0.9

})

response = bedrock.invokemodel(

modelId="meta.llama2-70b-chat-v1",

body=body

)

result = json.loads(response["body"].read())

return result["generation"]

response = invokellama("Explain neural networks.")

print(response)

3. Cohere Command

def invokecohere(prompt):

body = json.dumps({

"prompt": prompt,

"maxtokens": 500,

"temperature": 0.7

})

response = bedrock.invokemodel(

modelId="cohere.command-text-v14",

body=body

)

result = json.loads(response["body"].read())

return result["generations"][0]["text"]

Embeddings

1. Titan Embeddings

def getembeddings(text):

body = json.dumps({

"inputText": text

})

response = bedrock.invokemodel(

modelId="amazon.titan-embed-text-v1",

body=body

)

result = json.loads(response["body"].read())

return result["embedding"]

Get embeddings

embedding = getembeddings("Machine learning is fascinating.")

print(f"Embedding dimension: {len(embedding)}")

2. Cohere Embeddings

def getcohereembeddings(texts):

body = json.dumps({

"texts": texts,

"inputtype": "searchdocument"

})

response = bedrock.invokemodel(

modelId="cohere.embed-english-v3",

body=body

)

result = json.loads(response["body"].read())

return result["embeddings"]

embeddings = getcohereembeddings(["Hello world", "AI is powerful"])

Image Generation

1. Stability AI

import base64

def generateimage(prompt):

body = json.dumps({

"textprompts": [{"text": prompt}],

"cfgscale": 7,

"steps": 50,

"seed": 42

})

response = bedrock.invokemodel(

modelId="stability.stable-diffusion-xl-v1",

body=body

)

result = json.loads(response["body"].read())

imagedata = base64.b64decode(result["artifacts"][0]["base64"])

with open("generatedimage.png", "wb") as f:

f.write(imagedata)

return "generatedimage.png"

imagepath = generateimage("A futuristic city with flying cars")

print(f"Image saved to: {imagepath}")

2. Amazon Titan Image

def generatetitanimage(prompt):

body = json.dumps({

"taskType": "TEXTIMAGE",

"textToImageParams": {

"text": prompt

},

"imageGenerationConfig": {

"numberOfImages": 1,

"quality": "standard",

"height": 512,

"width": 512

}

})

response = bedrock.invokemodel(

modelId="amazon.titan-image-generator-v1",

body=body

)

result = json.loads(response["body"].read())

return base64.b64decode(result["images"][0])

Conversation and Chat

1. Multi-turn Conversation

class BedrockChat:

def init(self, modelid="anthropic.claude-3-sonnet-20240229-v1:0"):

self.bedrock = boto3.client("bedrock-runtime", regionname="us-east-1")

self.modelid = modelid

self.messages = []

self.systemprompt = None

def setsystemprompt(self, prompt):

self.systemprompt = prompt

def chat(self, usermessage):

self.messages.append({"role": "user", "content": usermessage})

body = {

"anthropicversion": "bedrock-2023-05-31",

"maxtokens": 1024,

"messages": self.messages

}

if self.systemprompt:

body["system"] = self.systemprompt

response = self.bedrock.invokemodel(

modelId=self.modelid,

body=json.dumps(body)

)

result = json.loads(response["body"].read())

assistantmessage = result["content"][0]["text"]

self.messages.append({"role": "assistant", "content": assistantmessage})

return assistantmessage

def clearhistory(self):

self.messages = []

Usage

chat = BedrockChat()

chat.setsystemprompt("You are a helpful AI assistant specializing in ML.")

response1 = chat.chat("What is deep learning?")

print(f"Assistant: {response1}\n")

response2 = chat.chat("Can you give me an example?")

print(f"Assistant: {response2}")

RAG with Bedrock

1. Knowledge Bases

bedrockagent = boto3.client("bedrock-agent-runtime", regionname="us-east-1")

def queryknowledgebase(query, kbid):

response = bedrockagent.retrieveandgenerate(

input={"text": query},

retrieveAndGenerateConfiguration={

"type": "KNOWLEDGEBASE",

"knowledgeBaseConfiguration": {

"knowledgeBaseId": kbid,

"modelArn": "arn:aws:bedrock:us-east-1::foundation-model/anthropic.claude-3-sonnet-20240229-v1:0"

}

}

)

return response["output"]["text"]

Query knowledge base

answer = queryknowledgebase(

"What are the main features of our product?",

"KNOWLEDGEBASEID"

)

print(answer)

2. Custom RAG Pipeline

import numpy as np

class BedrockRAG:

def init(self):

self.bedrock = boto3.client("bedrock-runtime", regionname="us-east-1")

self.documents = []

self.embeddings = []

def adddocuments(self, documents):

for doc in documents:

embedding = self.getembedding(doc)

self.documents.append(doc)

self.embeddings.append(embedding)

def getembedding(self, text):

body = json.dumps({"inputText": text})

response = self.bedrock.invokemodel(

modelId="amazon.titan-embed-text-v1",

body=body

)

result = json.loads(response["body"].read())

return np.array(result["embedding"])

def cosinesimilarity(self, a, b):

return np.dot(a, b) / (np.linalg.norm(a) * np.linalg.norm(b))

def retrieve(self, query, topk=3):

queryembedding = self.getembedding(query)

similarities = [

self.cosinesimilarity(queryembedding, emb)

for emb in self.embeddings

]

topindices = np.argsort(similarities)[-topk:][::-1]

return [self.documents[i] for i in topindices]

def query(self, question):

relevantdocs = self.retrieve(question)

context = "\n".join(relevantdocs)

prompt = f"""Based on the following context, answer the question.

Context:

{context}

Question: {question}

Answer:"""

body = json.dumps({

"anthropicversion": "bedrock-2023-05-31",

"maxtokens": 1024,

"messages": [{"role": "user", "content": prompt}]

})

response = self.bedrock.invokemodel(

modelId="anthropic.claude-3-sonnet-20240229-v1:0",

body=body

)

result = json.loads(response["body"].read())

return result["content"][0]["text"]

Usage

rag = BedrockRAG()

rag.adddocuments([

"Our product supports Python 3.8 and above.",

"The API rate limit is 1000 requests per minute.",

"Premium users get priority support."

])

answer = rag.query("What Python versions are supported?")

print(answer)

Bedrock Agents

1. Create Agent

bedrockagentclient = boto3.client("bedrock-agent", regionname="us-east-1")

def createagent(agentname, instructions):

response = bedrockagentclient.createagent(

agentName=agentname,

foundationModel="anthropic.claude-3-sonnet-20240229-v1:0",

instruction=instructions,

agentResourceRoleArn="arn:aws:iam::123456789:role/BedrockAgentRole"

)

return response["agent"]["agentId"]

agentid = createagent(

"customer-service-agent",

"You are a customer service agent. Help users with their inquiries."

)

2. Invoke Agent

def invokeagent(agentid, sessionid, prompt):

response = bedrockagent.invokeagent(

agentId=agentid,

agentAliasId="TSTALIASID",

sessionId=sessionid,

inputText=prompt

)

completion = ""

for event in response["completion"]:

if "chunk" in event:

completion += event["chunk"]["bytes"].decode()

return completion

Model Customization

1. Fine-tuning

bedrockclient = boto3.client("bedrock", regionname="us-east-1")

def createfinetuningjob(jobname, modelid, trainingdatauri):

response = bedrockclient.createmodelcustomizationjob(

jobName=jobname,

customModelName=f"custom-{jobname}",

roleArn="arn:aws:iam::123456789:role/BedrockFineTuningRole",

baseModelIdentifier=modelid,

trainingDataConfig={

"s3Uri": trainingdatauri

},

outputDataConfig={

"s3Uri": "s3://bucket/fine-tuned-models/"

},

hyperParameters={

"epochCount": "3",

"batchSize": "8",

"learningRate": "0.00001"

}

)

return response["jobArn"]

2. Continued Pre-training

def createcontinuedpretrainingjob(jobname, modelid, datauri):

response = bedrockclient.createmodelcustomizationjob(

jobName=jobname,

customModelName=f"pretrained-{jobname}",

roleArn="arn:aws:iam::123456789:role/BedrockRole",

baseModelIdentifier=modelid,

customizationType="CONTINUEDPRETRAINING",

trainingDataConfig={

"s3Uri": datauri

},

outputDataConfig={

"s3Uri": "s3://bucket/pretrained-models/"

}

)

return response["jobArn"]

Guardrails

1. Create Guardrail

def createguardrail(name, blockedtopics, wordfilters):

response = bedrockclient.createguardrail(

name=name,

description="Content filtering guardrail",

topicPolicyConfig={

"topicsConfig": [

{

"name": topic,

"definition": f"Content related to {topic}",

"type": "DENY"

}

for topic in blockedtopics

]

},

wordPolicyConfig={

"wordsConfig": [

{"text": word} for word in wordfilters

]

},

blockedInputMessaging="Your request was blocked.",

blockedOutputsMessaging="Response was filtered."

)

return response["guardrailId"]

guardrailid = createguardrail(

"content-filter",

["violence", "illegal activities"],

["profanity1", "profanity2"]

)

2. Use Guardrail

def invokewithguardrail(prompt, guardrailid):

body = json.dumps({

"anthropicversion": "bedrock-2023-05-31",

"maxtokens": 1024,

"messages": [{"role": "user", "content": prompt}]

})

response = bedrock.invokemodel(

modelId="anthropic.claude-3-sonnet-20240229-v1:0",

body=body,

guardrailIdentifier=guardrailid,

guardrailVersion="DRAFT"

)

return json.loads(response["body"].read())

Best Practices

1. Error Handling

from botocore.exceptions import ClientError

def safeinvoke(prompt, modelid):

try:

body = json.dumps({

"anthropicversion": "bedrock-2023-05-31",

"maxtokens": 1024,

"messages": [{"role": "user", "content": prompt}]

})

response = bedrock.invokemodel(modelId=modelid, body=body)

return json.loads(response["body"].read())

except ClientError as e:

errorcode = e.response["Error"]["Code"]

if errorcode == "ThrottlingException":

print("Rate limited, retrying...")

elif errorcode == "ModelNotReadyException":

print("Model not ready")

raise

2. Cost Optimization

# Use appropriate model for task

def selectmodel(taskcomplexity):

if taskcomplexity == "simple":

return "amazon.titan-text-lite-v1"

elif taskcomplexity == "medium":

return "anthropic.claude-3-haiku-20240307-v1:0"

else:

return "anthropic.claude-3-sonnet-20240229-v1:0"

Conclusion

AWS Bedrock provides:

  • Multiple FMs: Access to leading AI models
  • Unified API: Consistent interface
  • Customization: Fine-tuning capabilities
  • Security: Enterprise-grade security
  • Integration: AWS ecosystem
  • Key takeaways:

    • Choose model based on use case
    • Implement proper error handling
    • Use guardrails for content filtering
    • Leverage knowledge bases for RAG
    • Monitor costs and usage

    Related Articles

    Complete Azure OpenAI Service Tutorial: GPT and LLMs on Azure

    Tutorial Lengkap Azure OpenAI Service: Enterprise AI dengan Model GPT Azure OpenAI Service menyediakan akses REST API ke...

    DSPy: Stop Hand-Tuning Prompts, Let the Compiler Optimize Them

    DSPy: Berhenti Ngoprek Prompt Manual, Biarkan Compiler yang Optimasi Halo temen-temen, kali ini aku mau ngenalin satu li...

    Inspect AI: The LLM Evaluation Framework from the UK AI Safety Institute

    Inspect AI: Framework Evaluasi LLM dari UK AI Safety Institute yang Wajib Kamu Coba Temen-temen, kalau kamu udah mulai s...

    Complete Braintrust Tutorial: Evaluate, Test, and Improve Your LLM Applications

    Tutorial Lengkap Braintrust: Evaluasi, Testing, dan Improve Aplikasi LLM Halo temen-temen, di tutorial kali ini aku mau ...