Stable Diffusion: A Comprehensive Tutorial
Table of Contents
Introduction
Stable Diffusion is a latent diffusion model that generates high-quality images from text descriptions. Unlike earlier diffusion models that operated directly in pixel space, Stable Diffusion works in a compressed latent space, making it significantly faster and more memory-efficient. This tutorial covers everything from basic text-to-image generation to advanced techniques like ControlNet and DreamBooth fine-tuning.
The Hugging Face Diffusers library provides the most convenient interface for working with Stable Diffusion models, and this tutorial uses it extensively.
Prerequisites
# Install required packages
pip install diffusers transformers accelerate torch torchvision
pip install safetensors xformers
pip install Pillow numpy
pip install peft # For fine-tuning with LoRA/DreamBooth
System requirements:
- Python 3.9 or higher
- NVIDIA GPU with at least 8 GB VRAM (16 GB recommended for fine-tuning)
- CUDA 11.8 or higher
- At least 16 GB system RAM
Verify your setup:
import torch
import diffusers
print(f"PyTorch version: {torch.version}")
print(f"CUDA available: {torch.cuda.isavailable()}")
print(f"GPU: {torch.cuda.getdevicename(0) if torch.cuda.isavailable() else 'N/A'}")
print(f"Diffusers version: {diffusers.version}")
Understanding Stable Diffusion Architecture
Stable Diffusion consists of three main components:
The generation process works as follows: start with random noise in latent space, then iteratively denoise it using the U-Net, guided by the text conditioning from CLIP.
Text-to-Image Generation
Basic Generation
from diffusers import StableDiffusionPipeline, DPMSolverMultistepScheduler
import torch
def createpipeline(modelid="stabilityai/stable-diffusion-2-1"):
"""Initialize the Stable Diffusion pipeline."""
pipe = StableDiffusionPipeline.frompretrained(
modelid,
torchdtype=torch.float16,
safetychecker=None,
)
pipe.scheduler = DPMSolverMultistepScheduler.fromconfig(pipe.scheduler.config)
pipe = pipe.to("cuda")
# Enable memory optimizations
pipe.enableattentionslicing()
# pipe.enablexformersmemoryefficientattention() # If xformers installed
return pipe
pipe = createpipeline()
Generate an image
prompt = "A serene mountain landscape at sunset, photorealistic, 8k resolution"
negativeprompt = "blurry, low quality, distorted, deformed"
image = pipe(
prompt=prompt,
negativeprompt=negativeprompt,
numinferencesteps=30,
guidancescale=7.5,
width=768,
height=768,
).images[0]
image.save("mountainsunset.png")
Batch Generation with Different Seeds
import torch
def generatevariations(pipe, prompt, negativeprompt="", numimages=4, seedstart=42):
"""Generate multiple variations of the same prompt with different seeds."""
images = []
for i in range(numimages):
generator = torch.Generator(device="cuda").manualseed(seedstart + i)
image = pipe(
prompt=prompt,
negativeprompt=negativeprompt,
numinferencesteps=30,
guidancescale=7.5,
generator=generator,
).images[0]
images.append(image)
image.save(f"variation{seedstart + i}.png")
return images
images = generatevariations(
pipe,
prompt="A futuristic city with flying cars, cyberpunk style",
negativeprompt="ugly, blurry, low quality",
numimages=4
)
Using SDXL for Higher Quality
from diffusers import StableDiffusionXLPipeline, StableDiffusionXLRefinerPipeline
def createsdxlpipeline():
"""Create an SDXL pipeline with optional refiner."""
base = StableDiffusionXLPipeline.frompretrained(
"stabilityai/stable-diffusion-xl-base-1.0",
torchdtype=torch.float16,
variant="fp16",
usesafetensors=True,
).to("cuda")
refiner = StableDiffusionXLRefinerPipeline.frompretrained(
"stabilityai/stable-diffusion-xl-refiner-1.0",
torchdtype=torch.float16,
variant="fp16",
usesafetensors=True,
).to("cuda")
return base, refiner
def generatewithrefiner(base, refiner, prompt, negativeprompt=""):
"""Two-stage generation: base model + refiner."""
# Base generation (run for 80% of steps)
image = base(
prompt=prompt,
negativeprompt=negativeprompt,
numinferencesteps=40,
denoisingend=0.8,
outputtype="latent",
).images
# Refiner (run for remaining 20% of steps)
refined = refiner(
prompt=prompt,
negativeprompt=negativeprompt,
numinferencesteps=40,
denoisingstart=0.8,
image=image,
).images[0]
return refined
Prompt Engineering for Images
Effective prompts are crucial for generating high-quality images. Here are key strategies:
Prompt Structure
# Template: [Subject] [Style] [Details] [Quality modifiers]
prompts = {
"portrait": (
"Portrait of a young woman with auburn hair, "
"soft natural lighting, shallow depth of field, "
"Canon EOS R5, 85mm lens, f/1.4, "
"photorealistic, highly detailed, 8k"
),
"landscape": (
"Vast alpine meadow with wildflowers in the foreground, "
"snow-capped mountains in the background, golden hour lighting, "
"dramatic clouds, landscape photography, "
"National Geographic style, ultra sharp, 4k"
),
"conceptart": (
"Ancient floating temple above the clouds, "
"intricate stone carvings, mystical atmosphere, "
"volumetric lighting, concept art, "
"trending on ArtStation, digital painting by Greg Rutkowski"
),
}
Negative prompt template
negativeprompts = {
"general": (
"ugly, deformed, noisy, blurry, low contrast, "
"watermark, text, signature, out of frame, "
"worst quality, low quality, jpeg artifacts"
),
"portrait": (
"ugly, deformed, noisy, blurry, bad anatomy, "
"extra limbs, extra fingers, poorly drawn hands, "
"poorly drawn face, mutation, disfigured"
),
}
Prompt Weighting
from diffusers import StableDiffusionPipeline
from compel import Compel
pipe = StableDiffusionPipeline.frompretrained(
"stabilityai/stable-diffusion-2-1",
torchdtype=torch.float16
).to("cuda")
compel = Compel(tokenizer=pipe.tokenizer, textencoder=pipe.textencoder)
Use ++ for stronger emphasis, -- for weaker
prompt = "a red++ cat sitting on a blue-- chair, photorealistic"
conditioning = compel(prompt)
image = pipe(
promptembeds=conditioning,
numinferencesteps=30,
).images[0]
Image-to-Image Generation
Transform existing images using text guidance:
from diffusers import StableDiffusionImg2ImgPipeline
from PIL import Image
def img2imgtransform(inputpath, prompt, strength=0.75):
"""
Transform an existing image based on a text prompt.
strength: How much to transform (0.0 = no change, 1.0 = complete regeneration)
"""
pipe = StableDiffusionImg2ImgPipeline.frompretrained(
"stabilityai/stable-diffusion-2-1",
torchdtype=torch.float16,
).to("cuda")
initimage = Image.open(inputpath).convert("RGB")
initimage = initimage.resize((768, 768))
result = pipe(
prompt=prompt,
image=initimage,
strength=strength,
guidancescale=7.5,
numinferencesteps=30,
).images[0]
return result
Example: Turn a sketch into a painting
result = img2imgtransform(
"sketch.png",
prompt="A beautiful oil painting of a forest, masterpiece, detailed brushstrokes",
strength=0.6
)
result.save("paintingfromsketch.png")
Inpainting
Replace or modify specific regions of an image:
from diffusers import StableDiffusionInpaintPipeline
from PIL import Image
import numpy as np
def inpaintregion(imagepath, maskpath, prompt):
"""
Inpaint a masked region of an image.
White pixels in the mask indicate areas to regenerate.
"""
pipe = StableDiffusionInpaintPipeline.frompretrained(
"stabilityai/stable-diffusion-2-inpainting",
torchdtype=torch.float16,
).to("cuda")
image = Image.open(imagepath).convert("RGB").resize((512, 512))
mask = Image.open(maskpath).convert("RGB").resize((512, 512))
result = pipe(
prompt=prompt,
image=image,
maskimage=mask,
guidancescale=7.5,
numinferencesteps=30,
).images[0]
return result
def createmaskprogrammatically(imagesize, regions):
"""Create an inpainting mask from rectangular regions."""
mask = np.zeros((imagesize, 3), dtype=np.uint8)
for x1, y1, x2, y2 in regions:
mask[y1:y2, x1:x2] = 255
return Image.fromarray(mask)
Example
mask = createmaskprogrammatically((512, 512), [(100, 100, 300, 300)])
mask.save("mask.png")
result = inpaintregion("photo.png", "mask.png", "A golden retriever sitting here")
result.save("inpainted.png")
ControlNet for Guided Generation
ControlNet adds spatial conditioning to Stable Diffusion, allowing precise control over the output structure.
from diffusers import StableDiffusionControlNetPipeline, ControlNetModel
from diffusers.utils import loadimage
import cv2
import numpy as np
from PIL import Image
def createcannycontrolnetpipeline():
"""Create a ControlNet pipeline conditioned on Canny edges."""
controlnet = ControlNetModel.frompretrained(
"lllyasviel/sd-controlnet-canny",
torchdtype=torch.float16,
)
pipe = StableDiffusionControlNetPipeline.frompretrained(
"runwayml/stable-diffusion-v1-5",
controlnet=controlnet,
torchdtype=torch.float16,
).to("cuda")
pipe.enableattentionslicing()
return pipe
def generatefromedges(pipe, imagepath, prompt):
"""Generate an image guided by Canny edges of the input."""
# Extract Canny edges
image = cv2.imread(imagepath)
edges = cv2.Canny(image, 100, 200)
edges = np.stack([edges] 3, axis=-1)
controlimage = Image.fromarray(edges)
result = pipe(
prompt=prompt,
image=controlimage,
numinferencesteps=30,
guidancescale=7.5,
).images[0]
return result
Usage
pipe = createcannycontrolnetpipeline()
result = generatefromedges(
pipe,
"architecture.jpg",
"A beautiful modern house, photorealistic, golden hour lighting"
)
result.save("controlledoutput.png")
Using Multiple ControlNets
from diffusers import StableDiffusionControlNetPipeline, ControlNetModel
def multicontrolnetpipeline():
"""Combine Canny edge and depth ControlNets."""
controlnets = [
ControlNetModel.frompretrained(
"lllyasviel/sd-controlnet-canny", torchdtype=torch.float16
),
ControlNetModel.frompretrained(
"lllyasviel/sd-controlnet-depth", torchdtype=torch.float16
),
]
pipe = StableDiffusionControlNetPipeline.frompretrained(
"runwayml/stable-diffusion-v1-5",
controlnet=controlnets,
torchdtype=torch.float16,
).to("cuda")
return pipe
Fine-tuning with DreamBooth
DreamBooth allows you to personalize Stable Diffusion by teaching it new concepts from just a few images.
# DreamBooth training script (run from command line)
"""
accelerate launch traindreambooth.py \
--pretrainedmodelnameorpath="stabilityai/stable-diffusion-2-1" \
--instancedatadir="./mydogphotos" \
--outputdir="./dreamboothmodel" \
--instanceprompt="a photo of sks dog" \
--resolution=512 \
--trainbatchsize=1 \
--gradientaccumulationsteps=1 \
--learningrate=5e-6 \
--lrscheduler="constant" \
--lrwarmupsteps=0 \
--maxtrainsteps=800 \
--use8bitadam \
--mixedprecision="fp16"
"""
Using DreamBooth with LoRA for efficient fine-tuning
"""
accelerate launch traindreamboothlora.py \
--pretrainedmodelnameorpath="stabilityai/stable-diffusion-2-1" \
--instancedatadir="./mydogphotos" \
--outputdir="./dreamboothlora" \
--instanceprompt="a photo of sks dog" \
--resolution=512 \
--trainbatchsize=1 \
--gradientaccumulationsteps=1 \
--learningrate=1e-4 \
--lrscheduler="constant" \
--lrwarmupsteps=0 \
--maxtrainsteps=500 \
--rank=4
"""
Loading and using the fine-tuned model
from diffusers import StableDiffusionPipeline
def loaddreamboothmodel(modelpath):
"""Load a DreamBooth fine-tuned model."""
pipe = StableDiffusionPipeline.frompretrained(
modelpath,
torchdtype=torch.float16,
).to("cuda")
return pipe
def loadloramodel(basemodelid, lorapath):
"""Load a LoRA fine-tuned model."""
pipe = StableDiffusionPipeline.frompretrained(
basemodelid,
torchdtype=torch.float16,
).to("cuda")
pipe.loadloraweights(lorapath)
return pipe
Generate with the fine-tuned model
pipe = loadloramodel(
"stabilityai/stable-diffusion-2-1",
"./dreamboothlora"
)
image = pipe("a photo of sks dog wearing a birthday hat").images[0]
image.save("mydogbirthday.png")
Deployment Strategies
FastAPI Server
from fastapi import FastAPI, HTTPException
from pydantic import BaseModel
from diffusers import StableDiffusionPipeline
import torch
import io
import base64
from PIL import Image
app = FastAPI()
Load model at startup
pipe = StableDiffusionPipeline.frompretrained(
"stabilityai/stable-diffusion-2-1",
torchdtype=torch.float16,
).to("cuda")
class GenerationRequest(BaseModel):
prompt: str
negativeprompt: str = ""
numinferencesteps: int = 30
guidancescale: float = 7.5
width: int = 512
height: int = 512
seed: int = -1
@app.post("/generate")
async def generateimage(request: GenerationRequest):
try:
generator = None
if request.seed >= 0:
generator = torch.Generator(device="cuda").manualseed(request.seed)
image = pipe(
prompt=request.prompt,
negativeprompt=request.negativeprompt,
numinferencesteps=request.numinferencesteps,
guidancescale=request.guidancescale,
width=request.width,
height=request.height,
generator=generator,
).images[0]
# Convert to base64
buffer = io.BytesIO()
image.save(buffer, format="PNG")
imgstr = base64.b64encode(buffer.getvalue()).decode()
return {"image": imgstr, "seed": request.seed}
except Exception as e:
raise HTTPException(statuscode=500, detail=str(e))
Optimizations for Production
import torch
from diffusers import StableDiffusionPipeline
def createoptimizedpipeline(modelid):
"""Create a production-optimized pipeline."""
pipe = StableDiffusionPipeline.frompretrained(
modelid,
torchdtype=torch.float16,
usesafetensors=True,
).to("cuda")
# Compile the UNet for faster inference (PyTorch 2.0+)
pipe.unet = torch.compile(pipe.unet, mode="reduce-overhead", fullgraph=True)
# Enable memory-efficient attention
pipe.enablexformersmemoryefficientattention()
# Enable VAE slicing for large batch sizes
pipe.enablevaeslicing()
return pipe
Best Practices
torch.float16 for inference to halve memory usage and improve speed without noticeable quality loss.Conclusion
Stable Diffusion is a versatile and powerful tool for image generation, with applications ranging from creative content creation to industrial design and prototyping. This tutorial covered the core techniques: basic generation, prompt engineering, img2img, inpainting, ControlNet, and DreamBooth fine-tuning. By combining these techniques and following the deployment strategies outlined, you can build production-ready image generation systems.
The field evolves rapidly. Keep an eye on newer model architectures like Stable Diffusion 3, FLUX, and emerging control mechanisms for even better results.