Weaviate: Vector Database with Integrated AI Modules
Weaviate is an open-source vector database designed to store data objects along with their vector embeddings. What makes Weaviate unique is its ability to integrate AI modules directly into the database, enabling auto-vectorization, semantic search, and even generative AI without additional infrastructure.
In this tutorial, we will learn how to use Weaviate from installation, schema definition, vector and hybrid search, to building a semantic product search engine with auto-vectorization and generative answers.
Why Weaviate?
Weaviate offers several advantages over other vector databases:
- Integrated AI Modules: Vectorizer and generative modules built directly into the database
- Auto-Vectorization: Data is automatically vectorized on insertion without manual preprocessing
- Hybrid Search: Combines vector (semantic) and keyword (BM25) search
- GraphQL API: Flexible and powerful query interface
- Multi-Tenancy: Data isolation for multi-tenant applications
- Scalability: Supports horizontal scaling for large datasets
- Integration Ecosystem: Compatible with LangChain, LlamaIndex, and other AI frameworks
Installation
Using Docker (Recommended for Development)
Create a docker-compose.yml file:
version: '3.4'
services:
weaviate:
image: cr.weaviate.io/semitechnologies/weaviate:1.25.0
restart: on-failure:0
ports:
- "8080:8080"
- "50051:50051"
environment:
QUERYDEFAULTSLIMIT: 25
AUTHENTICATIONANONYMOUSACCESSENABLED: 'true'
PERSISTENCEDATAPATH: '/var/lib/weaviate'
DEFAULTVECTORIZERMODULE: 'text2vec-openai'
ENABLEMODULES: 'text2vec-openai,generative-openai'
OPENAIAPIKEY: 'sk-your-openai-api-key'
CLUSTERHOSTNAME: 'node1'
volumes:
- weaviatedata:/var/lib/weaviate
volumes:
weaviatedata:
Start Weaviate:
docker-compose up -d
Using Weaviate Cloud (WCD)
For production deployments, you can use Weaviate Cloud:
Python Client Installation
pip install weaviate-client
Connecting to Weaviate
import weaviate
from weaviate.classes.init import Auth
Connect to local instance (Docker)
client = weaviate.connecttolocal()
Connect to Weaviate Cloud
client = weaviate.connecttoweaviatecloud(
clusterurl="https://your-cluster.weaviate.network",
authcredentials=Auth.apikey("your-wcd-api-key"),
headers={
"X-OpenAI-Api-Key": "sk-your-openai-api-key"
}
)
Verify connection
print(client.isready()) # True if successful
Schema Definition
Schema in Weaviate defines the data structure, including properties and vectorizer configuration.
Creating a Collection (Class)
import weaviate
import weaviate.classes.config as wc
client = weaviate.connecttolocal()
Create a simple collection
client.collections.create(
name="Article",
description="Blog article collection",
vectorizerconfig=wc.Configure.Vectorizer.text2vecopenai(
model="text-embedding-3-small",
),
generativeconfig=wc.Configure.Generative.openai(
model="gpt-4",
),
properties=[
wc.Property(
name="title",
datatype=wc.DataType.TEXT,
description="Article title",
),
wc.Property(
name="content",
datatype=wc.DataType.TEXT,
description="Article content",
),
wc.Property(
name="author",
datatype=wc.DataType.TEXT,
description="Author name",
skipvectorization=True, # Not vectorized
),
wc.Property(
name="publisheddate",
datatype=wc.DataType.DATE,
description="Publication date",
skipvectorization=True,
),
wc.Property(
name="tags",
datatype=wc.DataType.TEXTARRAY,
description="Article tags",
),
wc.Property(
name="viewcount",
datatype=wc.DataType.INT,
description="Number of views",
skipvectorization=True,
),
],
)
print("Collection 'Article' created successfully!")
Alternative Vectorizer Configurations
# Using text2vec-transformers (self-hosted)
client.collections.create(
name="Document",
vectorizerconfig=wc.Configure.Vectorizer.text2vectransformers(),
properties=[
wc.Property(name="text", datatype=wc.DataType.TEXT),
wc.Property(name="source", datatype=wc.DataType.TEXT),
],
)
Using text2vec-cohere
client.collections.create(
name="SearchIndex",
vectorizerconfig=wc.Configure.Vectorizer.text2veccohere(
model="embed-multilingual-v3.0",
),
properties=[
wc.Property(name="content", datatype=wc.DataType.TEXT),
wc.Property(name="language", datatype=wc.DataType.TEXT),
],
)
Viewing and Deleting Collections
# View all collections
collections = client.collections.listall()
for name, config in collections.items():
print(f"Collection: {name}")
Delete a collection
client.collections.delete("Article")
Data Import (Batch)
Weaviate supports batch import for efficient data insertion.
Single Object Import
articles = client.collections.get("Article")
Insert a single object
articleuuid = articles.data.insert(
properties={
"title": "Introduction to Machine Learning",
"content": "Machine learning is a branch of AI that enables computers to learn from data...",
"author": "Ruby Abdullah",
"publisheddate": "2024-01-15T00:00:00Z",
"tags": ["machine-learning", "ai", "tutorial"],
"viewcount": 1500,
}
)
print(f"Inserted with UUID: {articleuuid}")
Batch Import
articles = client.collections.get("Article")
Sample data
sampledata = [
{
"title": "Deep Learning with PyTorch",
"content": "PyTorch is a popular deep learning framework developed by Meta AI...",
"author": "Ruby Abdullah",
"publisheddate": "2024-02-10T00:00:00Z",
"tags": ["deep-learning", "pytorch", "tutorial"],
"viewcount": 2300,
},
{
"title": "Natural Language Processing for Beginners",
"content": "NLP is a field of AI that focuses on the interaction between computers and human language...",
"author": "Ruby Abdullah",
"publisheddate": "2024-03-05T00:00:00Z",
"tags": ["nlp", "ai", "beginner"],
"viewcount": 1800,
},
{
"title": "Computer Vision with OpenCV",
"content": "OpenCV is an open-source library for computer vision and image processing...",
"author": "Ruby Abdullah",
"publisheddate": "2024-04-20T00:00:00Z",
"tags": ["computer-vision", "opencv", "tutorial"],
"viewcount": 2100,
},
{
"title": "Reinforcement Learning Basics",
"content": "Reinforcement learning is a machine learning paradigm where agents learn through trial and error...",
"author": "Ruby Abdullah",
"publisheddate": "2024-05-15T00:00:00Z",
"tags": ["reinforcement-learning", "ai", "tutorial"],
"viewcount": 950,
},
]
Batch insert
with articles.batch.dynamic() as batch:
for item in sampledata:
batch.addobject(properties=item)
Check batch results
failed = articles.batch.failedobjects
if failed:
print(f"Failed objects: {len(failed)}")
for obj in failed:
print(f" Error: {obj.message}")
else:
print(f"All {len(sampledata)} objects inserted successfully!")
Import with Custom Vector
import numpy as np
articles = client.collections.get("Article")
Insert with custom vector
articles.data.insert(
properties={
"title": "Custom Vector Article",
"content": "An article with a manually provided vector...",
"author": "Ruby Abdullah",
},
vector=np.random.rand(1536).tolist(), # Custom embedding
)
Vector Search
Weaviate provides several vector search methods.
nearText Search
Search based on semantic similarity with a text query.
articles = client.collections.get("Article")
Semantic search
response = articles.query.neartext(
query="tutorial on learning artificial intelligence",
limit=3,
returnmetadata=wc.query.MetadataQuery(
distance=True,
certainty=True,
),
)
for obj in response.objects:
print(f"Title: {obj.properties['title']}")
print(f"Distance: {obj.metadata.distance:.4f}")
print(f"Certainty: {obj.metadata.certainty:.4f}")
print()
nearVector Search
Search based on an existing vector embedding.
import numpy as np
articles = client.collections.get("Article")
Generate or retrieve vector from another source
queryvector = np.random.rand(1536).tolist()
response = articles.query.nearvector(
nearvector=queryvector,
limit=5,
returnmetadata=wq.MetadataQuery(distance=True),
)
for obj in response.objects:
print(f"Title: {obj.properties['title']}")
print(f"Distance: {obj.metadata.distance:.4f}")
nearObject Search
Search for objects similar to another object already in the database.
articles = client.collections.get("Article")
Find objects similar to a specific object
response = articles.query.nearobject(
nearobject=targetuuid, # Reference object UUID
limit=5,
returnmetadata=wq.MetadataQuery(distance=True),
)
for obj in response.objects:
print(f"Title: {obj.properties['title']}")
print(f"Distance: {obj.metadata.distance:.4f}")
Hybrid Search
Hybrid search combines vector (semantic) and keyword (BM25) search for more accurate results.
import weaviate.classes.query as wq
articles = client.collections.get("Article")
Hybrid search
response = articles.query.hybrid(
query="deep learning framework python",
alpha=0.5, # 0 = pure BM25, 1 = pure vector search
limit=5,
returnmetadata=wq.MetadataQuery(score=True, explainscore=True),
)
for obj in response.objects:
print(f"Title: {obj.properties['title']}")
print(f"Score: {obj.metadata.score:.4f}")
print(f"Explain: {obj.metadata.explainscore}")
print()
Adjusting Alpha Weight
# Emphasize keyword matching
responsekeyword = articles.query.hybrid(
query="PyTorch tutorial",
alpha=0.25, # 75% BM25, 25% vector
limit=5,
)
Emphasize semantic similarity
responsesemantic = articles.query.hybrid(
query="how to build AI models",
alpha=0.75, # 25% BM25, 75% vector
limit=5,
)
Filters
Weaviate supports various filters to narrow down search results.
import weaviate.classes.query as wq
articles = client.collections.get("Article")
Filter by property
response = articles.query.neartext(
query="AI tutorial",
limit=10,
filters=wq.Filter.byproperty("author").equal("Ruby Abdullah"),
)
Filter with comparison operators
response = articles.query.neartext(
query="machine learning",
limit=10,
filters=wq.Filter.byproperty("viewcount").greaterthan(1000),
)
Filter with AND
response = articles.query.neartext(
query="deep learning",
limit=10,
filters=(
wq.Filter.byproperty("author").equal("Ruby Abdullah") &
wq.Filter.byproperty("viewcount").greaterthan(1000)
),
)
Filter with OR
response = articles.query.neartext(
query="programming",
limit=10,
filters=(
wq.Filter.byproperty("tags").containsany(["python", "tutorial"]) |
wq.Filter.byproperty("viewcount").greaterthan(2000)
),
)
Filter by date
from datetime import datetime
response = articles.query.neartext(
query="AI tutorial",
limit=10,
filters=wq.Filter.byproperty("publisheddate").greaterthan(
datetime(2024, 3, 1)
),
)
Generative Modules
Generative modules allow you to use LLMs directly within Weaviate queries.
Single Prompt (Per Object)
articles = client.collections.get("Article")
response = articles.generate.neartext(
query="machine learning for beginners",
limit=3,
singleprompt="Create a brief summary (2-3 sentences) of the following article: {title} - {content}",
)
for obj in response.objects:
print(f"Title: {obj.properties['title']}")
print(f"Summary: {obj.generated}")
print()
Grouped Task (All Objects)
articles = client.collections.get("Article")
response = articles.generate.neartext(
query="programming tutorial",
limit=5,
groupedtask="Based on the following articles, create a recommended learning path for a beginner who wants to learn AI. Order from most basic to advanced.",
)
Generative result from all found objects
print("Learning Path Recommendation:")
print(response.generated)
Generative with Hybrid Search
articles = client.collections.get("Article")
response = articles.generate.hybrid(
query="python data science",
alpha=0.5,
limit=3,
singleprompt="Explain why the article '{title}' is relevant for beginner data scientists.",
groupedtask="Compare the three articles above and determine which is most suitable for beginners.",
)
for obj in response.objects:
print(f"Title: {obj.properties['title']}")
print(f"Relevance: {obj.generated}")
print()
print(f"\nComparison: {response.generated}")
Multi-Tenancy
Multi-tenancy enables data isolation per tenant within a single collection.
Enabling Multi-Tenancy
import weaviate.classes.config as wc
Create collection with multi-tenancy
client.collections.create(
name="CustomerData",
multitenancyconfig=wc.Configure.multitenancy(
enabled=True,
autotenantcreation=True,
),
vectorizerconfig=wc.Configure.Vectorizer.text2vecopenai(),
properties=[
wc.Property(name="name", datatype=wc.DataType.TEXT),
wc.Property(name="description", datatype=wc.DataType.TEXT),
wc.Property(name="category", datatype=wc.DataType.TEXT),
],
)
Managing Tenants
from weaviate.classes.tenants import Tenant, TenantActivityStatus
collection = client.collections.get("CustomerData")
Add tenants
collection.tenants.create([
Tenant(name="companya"),
Tenant(name="companyb"),
Tenant(name="companyc"),
])
View all tenants
tenants = collection.tenants.get()
for name, tenant in tenants.items():
print(f"Tenant: {name}, Status: {tenant.activitystatus}")
Per-Tenant Data Operations
# Access data for a specific tenant
tenanta = client.collections.get("CustomerData").withtenant("companya")
Insert data for tenant A
tenanta.data.insert(
properties={
"name": "Product X",
"description": "Premium product for Company A",
"category": "premium",
}
)
Query data only for tenant A
response = tenanta.query.neartext(
query="premium product",
limit=5,
)
Tenant B data won't appear in tenant A queries
tenantb = client.collections.get("CustomerData").withtenant("companyb")
tenantb.data.insert(
properties={
"name": "Product Y",
"description": "Basic product for Company B",
"category": "basic",
}
)
Backup and Restore
Creating Backups
# Backup all collections
result = client.backup.create(
backupid="backup-2024-01-15",
backend="filesystem",
waitforcompletion=True,
)
print(f"Backup status: {result.status}")
Backup specific collections
result = client.backup.create(
backupid="backup-articles-only",
backend="filesystem",
includecollections=["Article"],
waitforcompletion=True,
)
Restoring from Backup
# Restore all collections
result = client.backup.restore(
backupid="backup-2024-01-15",
backend="filesystem",
waitforcompletion=True,
)
print(f"Restore status: {result.status}")
Restore specific collections
result = client.backup.restore(
backupid="backup-articles-only",
backend="filesystem",
includecollections=["Article"],
waitforcompletion=True,
)
Integration with LangChain
from langchainweaviate import WeaviateVectorStore
from langchainopenai import OpenAIEmbeddings
import weaviate
Connect to Weaviate
client = weaviate.connecttolocal()
Create vector store
embeddings = OpenAIEmbeddings(model="text-embedding-3-small")
vectorstore = WeaviateVectorStore(
client=client,
indexname="LangChainDocs",
textkey="content",
embedding=embeddings,
)
Add documents
from langchain.schema import Document
docs = [
Document(pagecontent="Python is a popular programming language", metadata={"source": "intro"}),
Document(pagecontent="Machine learning uses data to make predictions", metadata={"source": "ml"}),
]
vectorstore.adddocuments(docs)
Search
results = vectorstore.similaritysearch(
query="programming language for AI",
k=3,
)
for doc in results:
print(f"Content: {doc.pagecontent}")
print(f"Metadata: {doc.metadata}")
print()
As a retriever for RAG
from langchainopenai import ChatOpenAI
from langchain.chains import RetrievalQA
llm = ChatOpenAI(model="gpt-4", temperature=0)
retriever = vectorstore.asretriever(searchkwargs={"k": 3})
qachain = RetrievalQA.fromchaintype(
llm=llm,
chaintype="stuff",
retriever=retriever,
)
answer = qachain.invoke("What is machine learning?")
print(answer["result"])
Integration with LlamaIndex
from llamaindex.core import VectorStoreIndex, StorageContext
from llama
index.vectorstores.weaviate import WeaviateVectorStore
import weaviate
Connect
client = weaviate.connect
tolocal()
Create vector store
vector
store = WeaviateVectorStore(
weaviateclient=client,
indexname="LlamaIndexDocs",
)
Create storage context
storagecontext = StorageContext.fromdefaults(
vectorstore=vectorstore,
)
Build index from documents
from llamaindex.core import Document
documents = [
Document(text="Weaviate is a powerful vector database"),
Document(text="LlamaIndex simplifies building RAG applications"),
]
index = VectorStoreIndex.fromdocuments(
documents,
storagecontext=storagecontext,
)
Query
queryengine = index.asqueryengine()
response = queryengine.query("What is Weaviate?")
print(response)
Practical Example: Semantic Product Search Engine
Let's build a product search engine with auto-vectorization and generative answers.
Product Collection Setup
import weaviate
import weaviate.classes.config as wc
client = weaviate.connecttolocal()
Delete collection if it exists
if client.collections.exists("Product"):
client.collections.delete("Product")
Create product collection
client.collections.create(
name="Product",
description="E-commerce product catalog",
vectorizerconfig=wc.Configure.Vectorizer.text2vecopenai(
model="text-embedding-3-small",
),
generativeconfig=wc.Configure.Generative.openai(
model="gpt-4",
),
properties=[
wc.Property(
name="name",
datatype=wc.DataType.TEXT,
description="Product name",
),
wc.Property(
name="description",
datatype=wc.DataType.TEXT,
description="Product description",
),
wc.Property(
name="category",
datatype=wc.DataType.TEXT,
description="Product category",
),
wc.Property(
name="price",
datatype=wc.DataType.NUMBER,
description="Product price",
skipvectorization=True,
),
wc.Property(
name="brand",
datatype=wc.DataType.TEXT,
description="Product brand",
),
wc.Property(
name="specs",
datatype=wc.DataType.TEXT,
description="Product specifications",
),
wc.Property(
name="rating",
datatype=wc.DataType.NUMBER,
description="Product rating (1-5)",
skipvectorization=True,
),
wc.Property(
name="stock",
datatype=wc.DataType.INT,
description="Available stock",
skipvectorization=True,
),
],
)
print("Collection 'Product' created successfully!")
Importing Product Data
products = client.collections.get("Product")
sampleproducts = [
{
"name": "Gaming Laptop ProMax X15",
"description": "High-performance gaming laptop with 15.6-inch 144Hz display, perfect for AAA gaming and content creation. Equipped with advanced cooling system.",
"category": "Laptop",
"price": 1299.99,
"brand": "ProMax",
"specs": "Intel i7-13700H, RTX 4060, 16GB DDR5, 512GB NVMe SSD, 15.6\" FHD 144Hz",
"rating": 4.5,
"stock": 25,
},
{
"name": "NoiseBlock Pro Wireless Headphones",
"description": "Premium wireless headphones with best-in-class Active Noise Cancellation. Battery lasts up to 30 hours of use.",
"category": "Audio",
"price": 249.99,
"brand": "NoiseBlock",
"specs": "ANC, Bluetooth 5.3, 30hr battery, 40mm drivers, LDAC/AAC codec",
"rating": 4.7,
"stock": 50,
},
{
"name": "UltraVision 5G Smartphone",
"description": "Flagship smartphone with 200MP camera and 6.7-inch AMOLED display. Supports 5G for ultra-fast connectivity.",
"category": "Smartphone",
"price": 899.99,
"brand": "UltraVision",
"specs": "Snapdragon 8 Gen 3, 12GB RAM, 256GB Storage, 200MP Camera, 6.7\" AMOLED 120Hz",
"rating": 4.6,
"stock": 100,
},
{
"name": "TypeMaster RGB Mechanical Keyboard",
"description": "Full-size mechanical keyboard with Cherry MX Blue switches, hot-swappable, and per-key RGB lighting. Ideal for programmers and gamers.",
"category": "Accessories",
"price": 119.99,
"brand": "TypeMaster",
"specs": "Cherry MX Blue, Hot-swap, RGB per-key, PBT keycaps, USB-C, N-key rollover",
"rating": 4.4,
"stock": 75,
},
{
"name": "ScreenPro 4K UltraWide Monitor",
"description": "34-inch ultrawide monitor with 4K resolution for maximum productivity. IPS panel with high color accuracy for designers.",
"category": "Monitor",
"price": 649.99,
"brand": "ScreenPro",
"specs": "34\" IPS UltraWide, 3440x1440, 100% sRGB, USB-C PD 65W, HDR400",
"rating": 4.3,
"stock": 30,
},
{
"name": "CreativeTab Pro 12 Tablet",
"description": "Tablet with stylus pen for digital drawing and note-taking. 12-inch screen with paper-like display technology.",
"category": "Tablet",
"price": 579.99,
"brand": "CreativeTab",
"specs": "12\" Paper-like display, Stylus 4096 levels, 8GB RAM, 128GB, Android 14",
"rating": 4.2,
"stock": 40,
},
]
Batch import
with products.batch.dynamic() as batch:
for product in sampleproducts:
batch.addobject(properties=product)
print(f"Successfully imported {len(sampleproducts)} products!")
Semantic Product Search
import weaviate.classes.query as wq
products = client.collections.get("Product")
Search: user searches with natural language
queries = [
"laptop for playing heavy games",
"noise-canceling earphones for working from a cafe",
"phone with great camera for photography",
"comfortable keyboard for coding",
"large screen for graphic design",
]
for query in queries:
print(f"\nQuery: '{query}'")
print("-" * 50)
response = products.query.neartext(
query=query,
limit=2,
returnmetadata=wq.MetadataQuery(distance=True),
)
for obj in response.objects:
p = obj.properties
print(f" {p['name']} - ${p['price']:,.2f}")
print(f" Distance: {obj.metadata.distance:.4f}")
Hybrid Search with Filters
products = client.collections.get("Product")
Hybrid search + price filter
response = products.query.hybrid(
query="devices for remote work productivity",
alpha=0.6,
limit=5,
filters=wq.Filter.byproperty("price").lessthan(700),
returnmetadata=wq.MetadataQuery(score=True),
)
print("Products for remote work (budget < $700):")
for obj in response.objects:
p = obj.properties
print(f" {p['name']} - ${p['price']:,.2f} (score: {obj.metadata.score:.4f})")
Generative Recommendations
products = client.collections.get("Product")
Search + personalized recommendations
response = products.generate.neartext(
query="complete setup for a freelance programmer",
limit=4,
singleprompt="Explain in 1-2 sentences why '{name}' is suitable for a freelance programmer. Price: ${price}",
groupedtask="""Based on the products above, create a complete setup recommendation
for a freelance programmer with a $2,500 budget.
Include the total price and reasoning for each item selection.""",
)
print("=== Per-Product Recommendations ===\n")
for obj in response.objects:
p = obj.properties
print(f"Product: {p['name']}")
print(f"Price: ${p['price']:,.2f}")
print(f"Recommendation: {obj.generated}")
print()
print("=== Complete Setup Recommendation ===\n")
print(response.generated)
Complete Product Search Class
import weaviate
import weaviate.classes.query as wq
from dataclasses import dataclass
from typing import Optional
@dataclass
class SearchResult:
name: str
description: str
category: str
price: float
brand: str
rating: float
score: float = 0.0
recommendation: str = ""
class ProductSearchEngine:
def init(self, client: weaviate.WeaviateClient):
self.client = client
self.products = client.collections.get("Product")
def semanticsearch(
self,
query: str,
limit: int = 5,
minrating: Optional[float] = None,
maxprice: Optional[float] = None,
category: Optional[str] = None,
) -> list[SearchResult]:
"""Semantic search with optional filters."""
filters = []
if minrating:
filters.append(
wq.Filter.byproperty("rating").greaterorequal(minrating)
)
if maxprice:
filters.append(
wq.Filter.byproperty("price").lessorequal(maxprice)
)
if category:
filters.append(
wq.Filter.byproperty("category").equal(category)
)
combinedfilter = None
if filters:
combinedfilter = filters[0]
for f in filters[1:]:
combinedfilter = combinedfilter & f
response = self.products.query.neartext(
query=query,
limit=limit,
filters=combinedfilter,
returnmetadata=wq.MetadataQuery(distance=True),
)
results = []
for obj in response.objects:
p = obj.properties
results.append(SearchResult(
name=p["name"],
description=p["description"],
category=p["category"],
price=p["price"],
brand=p["brand"],
rating=p["rating"],
score=1 - obj.metadata.distance,
))
return results
def hybridsearch(
self,
query: str,
alpha: float = 0.5,
limit: int = 5,
) -> list[SearchResult]:
"""Hybrid search (semantic + keyword)."""
response = self.products.query.hybrid(
query=query,
alpha=alpha,
limit=limit,
returnmetadata=wq.MetadataQuery(score=True),
)
results = []
for obj in response.objects:
p = obj.properties
results.append(SearchResult(
name=p["name"],
description=p["description"],
category=p["category"],
price=p["price"],
brand=p["brand"],
rating=p["rating"],
score=obj.metadata.score,
))
return results
def askrecommendation(
self,
query: str,
limit: int = 3,
) -> tuple[list[SearchResult], str]:
"""Search with generative recommendations."""
response = self.products.generate.neartext(
query=query,
limit=limit,
groupedtask=f"Based on the products found, provide the best recommendation for: {query}. Explain your reasoning.",
)
results = []
for obj in response.objects:
p = obj.properties
results.append(SearchResult(
name=p["name"],
description=p["description"],
category=p["category"],
price=p["price"],
brand=p["brand"],
rating=p["rating"],
))
return results, response.generated
def findsimilar(self, productuuid: str, limit: int = 3) -> list[SearchResult]:
"""Find similar products."""
response = self.products.query.nearobject(
nearobject=productuuid,
limit=limit,
returnmetadata=wq.MetadataQuery(distance=True),
)
results = []
for obj in response.objects:
p = obj.properties
results.append(SearchResult(
name=p["name"],
description=p["description"],
category=p["category"],
price=p["price"],
brand=p["brand"],
rating=p["rating"],
score=1 - obj.metadata.distance,
))
return results
Usage
if name == "main":
client = weaviate.connecttolocal()
engine = ProductSearchEngine(client)
# Semantic search
print("=== Semantic Search ===")
results = engine.semanticsearch(
query="devices for gaming",
maxprice=1500,
minrating=4.0,
)
for r in results:
print(f" {r.name} - ${r.price:,.2f} (rating: {r.rating})")
# Hybrid search
print("\n=== Hybrid Search ===")
results = engine.hybridsearch(
query="mechanical keyboard RGB",
alpha=0.3,
)
for r in results:
print(f" {r.name} - Score: {r.score:.4f}")
# Generative recommendation
print("\n=== AI Recommendation ===")
results, recommendation = engine.askrecommendation(
query="complete setup for a content creator with a $2000 budget"
)
print(f"Recommendation: {recommendation}")
client.close()