Complete Streamlit Advanced Tutorial: Build Production-Ready ML Apps
Streamlit is a powerful Python library for building interactive web applications for machine learning and data science. This advanced tutorial covers production patterns, performance optimization, and enterprise features.
Why Streamlit for Production?
Streamlit Advantages:- Rapid development: Build apps in hours, not days
- Pure Python: No frontend knowledge required
- Interactive widgets: Rich UI components
- Easy deployment: Streamlit Cloud, Docker, Kubernetes
- Active ecosystem: Community components and integrations
- ML model demos and dashboards
- Data exploration tools
- Internal analytics apps
- Customer-facing applications
- Prototyping and MVPs
Installation
pip install streamlit
With additional features
pip install streamlit-extras
pip install streamlit-aggrid
pip install plotly
Verify installation
streamlit --version
App Architecture
1. Multi-Page Apps
# pages/1Home.py
import streamlit as st
st.set
pageconfig(
page
title="ML Dashboard",
pageicon="🤖",
layout="wide",
)
st.title("Welcome to ML Dashboard")
st.write("Navigate using the sidebar")
# pages/2DataExplorer.py
import streamlit as st
import pandas as pd
st.title("Data Explorer")
uploaded
file = st.fileuploader("Upload CSV", type="csv")
if uploaded
file:
df = pd.readcsv(uploadedfile)
st.dataframe(df)
# pages/3ModelInference.py
import streamlit as st
st.title("Model Inference")
Model inference code here
2. Session State Management
import streamlit as st
Initialize session state
if 'counter' not in st.sessionstate:
st.sessionstate.counter = 0
if 'userdata' not in st.sessionstate:
st.sessionstate.userdata = {}
Update session state
def increment():
st.sessionstate.counter += 1
st.button("Increment", onclick=increment)
st.write(f"Counter: {st.sessionstate.counter}")
Store user data
name = st.textinput("Name", key="nameinput")
if name:
st.sessionstate.userdata['name'] = name
Access across pages
st.write(st.sessionstate.userdata)
3. Callbacks and Events
import streamlit as st
Callback function
def onsubmit():
st.sessionstate.submitted = True
st.sessionstate.result = f"Hello, {st.sessionstate.namefield}!"
Form with callback
with st.form("myform"):
st.textinput("Name", key="namefield")
submitted = st.formsubmitbutton("Submit", onclick=onsubmit)
if st.sessionstate.get('submitted'):
st.success(st.sessionstate.result)
Multiple callbacks
def clearform():
st.sessionstate.namefield = ""
st.sessionstate.submitted = False
st.button("Clear", onclick=clearform)
Caching and Performance
1. Cache Data
import streamlit as st
import pandas as pd
@st.cachedata(ttl=3600) # Cache for 1 hour
def loaddata(url):
"""Load and cache data"""
return pd.readcsv(url)
@st.cachedata(showspinner="Loading data...")
def expensivecomputation(df):
"""Expensive computation with spinner"""
# Simulate long computation
import time
time.sleep(5)
return df.describe()
Use cached functions
df = loaddata("https://example.com/data.csv")
stats = expensivecomputation(df)
2. Cache Resources
import streamlit as st
from transformers import pipeline
import joblib
@st.cacheresource
def loadmodel():
"""Cache ML model (singleton)"""
return joblib.load("model.joblib")
@st.cacheresource
def loadllm():
"""Cache LLM pipeline"""
return pipeline("text-generation", model="gpt2")
Models loaded once and reused
model = loadmodel()
llm = loadllm()
3. Cache Configuration
import streamlit as st
Cache with hash functions
@st.cachedata(hashfuncs={pd.DataFrame: lambda x: x.tojson()})
def processdataframe(df):
return df.groupby('category').sum()
Cache with max entries
@st.cachedata(maxentries=100)
def getuserdata(userid):
return fetchfromdatabase(userid)
Clear cache
if st.button("Clear Cache"):
st.cachedata.clear()
st.cacheresource.clear()
Advanced Components
1. Interactive Tables with AgGrid
import streamlit as st
from staggrid import AgGrid, GridOptionsBuilder
import pandas as pd
df = pd.readcsv("data.csv")
Configure grid options
gb = GridOptionsBuilder.fromdataframe(df)
gb.configurepagination(paginationAutoPageSize=True)
gb.configureselection('multiple', usecheckbox=True)
gb.configuresidebar()
gridoptions = gb.build()
Display grid
gridresponse = AgGrid(
df,
gridOptions=gridoptions,
enableenterprisemodules=True,
theme='streamlit'
)
Get selected rows
selectedrows = gridresponse['selectedrows']
if selectedrows:
st.write("Selected:", selectedrows)
2. Interactive Charts with Plotly
import streamlit as st
import plotly.express as px
import plotly.graphobjects as go
Interactive scatter plot
fig = px.scatter(
df,
x="feature1",
y="feature2",
color="category",
size="value",
hoverdata=["name"]
)
st.plotlychart(fig, usecontainerwidth=True)
Interactive line chart with range slider
fig = go.Figure()
fig.addtrace(go.Scatter(x=dates, y=values, mode='lines'))
fig.updatelayout(
xaxis=dict(rangeslider=dict(visible=True)),
title="Time Series with Range Slider"
)
st.plotlychart(fig)
Callback on chart selection
selectedpoints = st.plotlychart(fig, onselect="rerun")
if selectedpoints:
st.write("Selected points:", selectedpoints)
3. Custom Components
import streamlit as st
import streamlit.components.v1 as components
Embed HTML/JS
components.html(
"""
// Chart.js code here
""",
height=400
)
Embed iframe
components.iframe("https://example.com/dashboard", height=600)
Real-time Updates
1. Auto-refresh
import streamlit as st
import time
Auto-refresh every 5 seconds
stautorefresh = st.empty()
with stautorefresh:
st.write(f"Last updated: {time.strftime('%H:%M:%S')}")
# Your real-time data here
Manual refresh
if st.button("Refresh"):
st.rerun()
2. Streaming Updates
import streamlit as st
import time
Streaming text
def streamresponse():
response = "This is a streaming response..."
for word in response.split():
yield word + " "
time.sleep(0.1)
st.writestream(streamresponse)
Streaming with placeholder
placeholder = st.empty()
for i in range(100):
placeholder.metric("Progress", f"{i}%")
time.sleep(0.1)
3. WebSocket Updates
import streamlit as st
import asyncio
import websockets
@st.cacheresource
def getwebsocketdata():
"""Connect to WebSocket and return data"""
async def connect():
async with websockets.connect("wss://example.com/ws") as ws:
return await ws.recv()
return asyncio.run(connect())
Display real-time data
data = getwebsocketdata()
st.write(data)
Authentication and Security
1. Basic Authentication
import streamlit as st
import hashlib
def checkpassword():
"""Returns True if user has correct password"""
def passwordentered():
if st.sessionstate["username"] in st.secrets["passwords"] and \
hashlib.sha256(st.sessionstate["password"].encode()).hexdigest() == \
st.secrets["passwords"][st.sessionstate["username"]]:
st.sessionstate["passwordcorrect"] = True
del st.sessionstate["password"]
else:
st.sessionstate["passwordcorrect"] = False
if "passwordcorrect" not in st.sessionstate:
st.textinput("Username", key="username")
st.textinput("Password", type="password", key="password")
st.button("Login", onclick=passwordentered)
return False
elif not st.sessionstate["passwordcorrect"]:
st.textinput("Username", key="username")
st.textinput("Password", type="password", key="password")
st.button("Login", onclick=passwordentered)
st.error("Invalid username or password")
return False
return True
if checkpassword():
st.write("Welcome to the app!")
2. Secrets Management
# .streamlit/secrets.toml
[passwords]
admin = "5e884898da28047d91..." # hashed password
[database]
host = "localhost"
port = 5432
user = "appuser"
password = "secretpassword"
[apikeys]
openai = "sk-..."
import streamlit as st
Access secrets
dbhost = st.secrets["database"]["host"]
apikey = st.secrets["apikeys"]["openai"]
3. Role-based Access
import streamlit as st
Define user roles
USERS = {
"admin": {"password": "admin123", "role": "admin"},
"analyst": {"password": "analyst123", "role": "analyst"},
"viewer": {"password": "viewer123", "role": "viewer"},
}
def getuserrole():
if "userrole" not in st.sessionstate:
return None
return st.sessionstate.userrole
def requirerole(allowedroles):
"""Decorator to check user role"""
role = getuserrole()
if role not in allowedroles:
st.error("You don't have permission to access this feature")
st.stop()
Usage
role = getuserrole()
if role == "admin":
st.write("Admin panel")
# Show admin features
elif role == "analyst":
st.write("Analyst dashboard")
# Show analyst features
else:
st.write("View-only dashboard")
Database Integration
1. SQL Connection
import streamlit as st
import pandas as pd
@st.cacheresource
def getconnection():
return st.connection("postgresql", type="sql")
conn = getconnection()
Query data
df = conn.query("SELECT FROM users WHERE active = true", ttl=600)
st.dataframe(df)
Parameterized query
userid = st.numberinput("User ID", minvalue=1)
user = conn.query(
"SELECT FROM users WHERE id = :id",
params={"id": userid}
)
2. MongoDB Integration
import streamlit as st
from pymongo import MongoClient
@st.cacheresource
def getmongoclient():
return MongoClient(st.secrets["mongodb"]["uri"])
client = getmongoclient()
db = client["mydb"]
collection = db["users"]
Query data
users = list(collection.find({"active": True}))
st.write(users)
Insert data
if st.button("Add User"):
collection.insertone({"name": "New User", "active": True})
st.success("User added!")
3. Redis for Caching
import streamlit as st
import redis
import json
@st.cacheresource
def getredis():
return redis.Redis(
host=st.secrets["redis"]["host"],
port=st.secrets["redis"]["port"],
password=st.secrets["redis"]["password"]
)
r = getredis()
Cache expensive computation
def getcacheddata(key):
cached = r.get(key)
if cached:
return json.loads(cached)
# Compute if not cached
data = expensivecomputation()
r.setex(key, 3600, json.dumps(data)) # Cache for 1 hour
return data
ML Model Integration
1. Model Inference Dashboard
import streamlit as st
import pandas as pd
import joblib
@st.cacheresource
def loadmodel():
return joblib.load("model.joblib")
model = loadmodel()
st.title("ML Model Inference")
Input features
col1, col2 = st.columns(2)
with col1:
feature1 = st.numberinput("Feature 1", value=0.0)
feature2 = st.numberinput("Feature 2", value=0.0)
with col2:
feature3 = st.selectbox("Category", ["A", "B", "C"])
feature4 = st.slider("Score", 0, 100, 50)
Predict
if st.button("Predict"):
features = pd.DataFrame([[feature1, feature2, feature3, feature4]],
columns=["f1", "f2", "category", "score"])
prediction = model.predict(features)[0]
probability = model.predictproba(features)[0]
st.success(f"Prediction: {prediction}")
st.write("Probabilities:", dict(zip(model.classes, probability)))
2. Batch Prediction
import streamlit as st
import pandas as pd
st.title("Batch Prediction")
uploadedfile = st.fileuploader("Upload CSV for prediction", type="csv")
if uploadedfile:
df = pd.readcsv(uploadedfile)
st.write("Preview:", df.head())
if st.button("Run Batch Prediction"):
with st.spinner("Running predictions..."):
predictions = model.predict(df)
df["prediction"] = predictions
st.success(f"Completed {len(df)} predictions")
st.dataframe(df)
# Download results
csv = df.tocsv(index=False)
st.downloadbutton(
"Download Results",
csv,
"predictions.csv",
"text/csv"
)
3. LLM Chat Interface
import streamlit as st
from openai import OpenAI
st.title("AI Chat Assistant")
client = OpenAI(apikey=st.secrets["openai"]["apikey"])
Initialize chat history
if "messages" not in st.sessionstate:
st.sessionstate.messages = []
Display chat history
for message in st.sessionstate.messages:
with st.chatmessage(message["role"]):
st.markdown(message["content"])
Chat input
if prompt := st.chatinput("What's on your mind?"):
st.sessionstate.messages.append({"role": "user", "content": prompt})
with st.chatmessage("user"):
st.markdown(prompt)
with st.chatmessage("assistant"):
stream = client.chat.completions.create(
model="gpt-4o-mini",
messages=st.sessionstate.messages,
stream=True,
)
response = st.writestream(stream)
st.sessionstate.messages.append({"role": "assistant", "content": response})
Deployment
1. Docker Deployment
# Dockerfile
FROM python:3.10-slim
WORKDIR /app
COPY requirements.txt .
RUN pip install -r requirements.txt
COPY . .
EXPOSE 8501
CMD ["streamlit", "run", "app.py", "--server.port=8501", "--server.address=0.0.0.0"]
# docker-compose.yml
version: '3.8'
services:
streamlit:
build: .
ports:
- "8501:8501"
volumes:
- ./data:/app/data
environment:
- STREAMLITSERVERHEADLESS=true
2. Kubernetes Deployment
# k8s-deployment.yaml
apiVersion: apps/v1
kind: Deployment
metadata:
name: streamlit-app
spec:
replicas: 3
selector:
matchLabels:
app: streamlit
template:
metadata:
labels:
app: streamlit
spec:
containers:
- name: streamlit
image: myregistry/streamlit-app:latest
ports:
- containerPort: 8501
resources:
requests:
memory: "512Mi"
cpu: "250m"
limits:
memory: "1Gi"
cpu: "500m"
apiVersion: v1
kind: Service
metadata:
name: streamlit-service
spec:
selector:
app: streamlit
ports:
- port: 80
targetPort: 8501
type: LoadBalancer
3. Configuration
# .streamlit/config.toml
[server]
headless = true
port = 8501
enableCORS = false
maxUploadSize = 200
[theme]
primaryColor = "#FF4B4B"
backgroundColor = "#FFFFFF"
secondaryBackgroundColor = "#F0F2F6"
textColor = "#262730"
font = "sans serif"
[browser]
gatherUsageStats = false
Best Practices
1. Error Handling
import streamlit as st
def safeoperation():
try:
result = riskyoperation()
return result
except ValueError as e:
st.error(f"Invalid value: {e}")
except ConnectionError:
st.error("Connection failed. Please try again.")
except Exception as e:
st.exception(e) # Show full traceback
return None
Global error handler
try:
main()
except Exception as e:
st.error("An unexpected error occurred")
st.exception(e)
2. Logging
import streamlit as st
import logging
logging.basicConfig(level=logging.INFO)
logger = logging.getLogger(name)
def processdata(df):
logger.info(f"Processing {len(df)} rows")
# Process data
logger.info("Processing complete")
return df
Log user actions
if st.button("Run Analysis"):
logger.info(f"User started analysis at {datetime.now()}")
3. Testing
# testapp.py
from streamlit.testing.v1 import AppTest
def testapploads():
at = AppTest.fromfile("app.py")
at.run()
assert not at.exception
def testbuttonclick():
at = AppTest.fromfile("app.py")
at.run()
at.button[0].click().run()
assert at.success[0].value == "Success!"
Conclusion
Streamlit is production-ready for ML applications with:
Key takeaways:
- Use caching for expensive operations
- Implement proper authentication
- Structure apps with multi-page architecture
- Handle errors gracefully
- Monitor and log user actions