Complete Vertex AI Feature Store Tutorial: Centralized Feature Management
Vertex AI Feature Store is a centralized repository for organizing, storing, and serving ML features. It enables feature reuse, reduces training-serving skew, and provides consistent feature access across teams.
Why Feature Store?
Key Benefits:- Centralized features: Single source of truth
- Feature reuse: Share features across models
- Low latency serving: Fast online feature retrieval
- Consistency: Same features for training and serving
- Time travel: Point-in-time feature lookups
Prerequisites
pip install google-cloud-aiplatform
gcloud auth login
gcloud config set project your-project-id
Setup
1. Initialize Vertex AI
from google.cloud import aiplatform
aiplatform.init(project="your-project-id", location="us-central1")
2. Create Feature Store
# Create feature store
featurestore = aiplatform.Featurestore.create(
featurestoreid="myfeaturestore",
onlinestorefixednodecount=1
)
print(f"Feature store created: {featurestore.resourcename}")
Entity Types
1. Create Entity Type
# Create customer entity type
customerentity = featurestore.createentitytype(
entitytypeid="customer",
description="Customer entity for churn prediction"
)
Create product entity type
productentity = featurestore.createentitytype(
entitytypeid="product",
description="Product entity for recommendation"
)
2. List Entity Types
entitytypes = featurestore.listentitytypes()
for et in entitytypes:
print(f"{et.entitytypeid}: {et.description}")
Features
1. Create Features
# Create features for customer entity
customerentity.createfeature(
featureid="age",
valuetype="INT64",
description="Customer age"
)
customerentity.createfeature(
featureid="tenuremonths",
valuetype="INT64",
description="Months as customer"
)
customerentity.createfeature(
featureid="monthlycharges",
valuetype="DOUBLE",
description="Monthly charges"
)
customerentity.createfeature(
featureid="totalcharges",
valuetype="DOUBLE",
description="Total charges to date"
)
customerentity.createfeature(
featureid="contracttype",
valuetype="STRING",
description="Type of contract"
)
2. Batch Create Features
# Create multiple features at once
featuresconfig = {
"age": {"valuetype": "INT64", "description": "Customer age"},
"tenuremonths": {"valuetype": "INT64", "description": "Tenure in months"},
"monthlycharges": {"valuetype": "DOUBLE", "description": "Monthly charges"},
"totalcharges": {"valuetype": "DOUBLE", "description": "Total charges"},
"contracttype": {"valuetype": "STRING", "description": "Contract type"}
}
customerentity.batchcreatefeatures(featuresconfig)
Ingesting Features
1. Ingest from BigQuery
# Ingest features from BigQuery
customerentity.ingestfrombq(
featureids=["age", "tenuremonths", "monthlycharges", "totalcharges"],
featuretime="updatetime",
bqsourceuri="bq://project.dataset.customerfeatures",
entityidfield="customerid"
)
2. Ingest from DataFrame
import pandas as pd
from datetime import datetime
Create feature dataframe
df = pd.DataFrame({
"customerid": ["C001", "C002", "C003"],
"age": [25, 35, 45],
"tenuremonths": [12, 24, 36],
"monthlycharges": [50.0, 75.0, 100.0],
"totalcharges": [600.0, 1800.0, 3600.0],
"featuretimestamp": [datetime.now()] 3
})
Ingest features
customerentity.ingestfromdf(
featureids=["age", "tenuremonths", "monthlycharges", "totalcharges"],
featuretime="featuretimestamp",
dfsource=df,
entityidfield="customerid"
)
3. Ingest from GCS
# Ingest from GCS Avro files
customerentity.ingestfromgcs(
featureids=["age", "tenuremonths", "monthlycharges"],
featuretime="timestamp",
gcssourceuris=["gs://bucket/features/.avro"],
gcssourcetype="avro",
entityidfield="customerid"
)
Reading Features
1. Online Serving
# Read features for single entity
features = customerentity.read(
entityids=["C001"],
featureids=["age", "tenuremonths", "monthlycharges"]
)
print(features)
2. Batch Serving
# Read features for multiple entities
featuresdf = customerentity.batchservetodf(
bqdestinationoutputuri="bq://project.dataset.featuresoutput",
servingfeatureids={
"customer": ["age", "tenuremonths", "monthlycharges", "totalcharges"]
},
readinstancesuri="bq://project.dataset.entityids"
)
3. Point-in-Time Lookup
# Read features at specific timestamps
import pandas as pd
instancesdf = pd.DataFrame({
"customerid": ["C001", "C002", "C003"],
"timestamp": [
"2024-01-01 00:00:00",
"2024-01-15 00:00:00",
"2024-02-01 00:00:00"
]
})
Batch serve with point-in-time
featurestore.batchservetodf(
servingfeatureids={
"customer": ["age", "tenuremonths", "monthlycharges"]
},
readinstancesdf=instancesdf,
passthroughfields=["customerid", "timestamp"]
)
Training with Feature Store
1. Create Training Dataset
from google.cloud import aiplatform
Define training query
trainingdata = featurestore.batchservetobq(
bqdestinationoutputuri="bq://project.dataset.trainingdata",
servingfeatureids={
"customer": ["age", "tenuremonths", "monthlycharges", "totalcharges"]
},
readinstancesuri="bq://project.dataset.traininginstances"
)
Create Vertex AI dataset
dataset = aiplatform.TabularDataset.create(
displayname="churn-training-dataset",
bqsource="bq://project.dataset.trainingdata"
)
2. Training Pipeline Integration
from kfp import dsl
from kfp.dsl import component
@component(packagestoinstall=["google-cloud-aiplatform"])
def fetchfeatures(
project: str,
location: str,
featurestoreid: str,
entitytypeid: str,
featureids: list,
instancesuri: str,
outputuri: str
):
from google.cloud import aiplatform
aiplatform.init(project=project, location=location)
fs = aiplatform.Featurestore(featurestoreid)
entitytype = fs.getentitytype(entitytypeid)
entitytype.batchservetobq(
bqdestinationoutputuri=outputuri,
servingfeatureids=featureids,
readinstancesuri=instancesuri
)
@dsl.pipeline
def trainingpipeline():
# Fetch features
fetchtask = fetchfeatures(
project="your-project",
location="us-central1",
featurestoreid="myfeaturestore",
entitytypeid="customer",
featureids=["age", "tenuremonths", "monthlycharges"],
instancesuri="bq://project.dataset.instances",
outputuri="bq://project.dataset.trainingfeatures"
)
# Train model
traintask = trainmodel(datauri=fetchtask.outputs["outputuri"])
Online Serving
1. Configure Online Store
# Update feature store for online serving
featurestore = aiplatform.Featurestore(
featurestorename="myfeaturestore"
)
Scale online serving nodes
featurestore.update(onlinestorefixednodecount=3)
2. Serve Features Online
# Read features online
onlinefeatures = customerentity.read(
entityids=["C001", "C002"],
featureids=["age", "tenuremonths", "monthlycharges"]
)
for entityid, features in onlinefeatures.items():
print(f"Entity {entityid}:")
for featureid, value in features.items():
print(f" {featureid}: {value}")
3. Streaming Ingestion
# Stream features for real-time updates
import time
def streamfeatures(entityid, features):
customerentity.writefeaturevalues(
entityid=entityid,
featurevalues=features
)
Example: Update features in real-time
streamfeatures("C001", {
"monthlycharges": 55.0,
"totalcharges": 655.0
})
Feature Monitoring
1. Enable Monitoring
# Create monitoring job
monitoringjob = featurestore.createmonitoringjob(
displayname="feature-monitoring",
entitytypeids=["customer"],
featureids=["age", "tenuremonths", "monthlycharges"],
objectiveconfig={
"trainingdataset": "bq://project.dataset.trainingdata",
"targetfield": "churn"
},
schedule="0 0 *" # Daily
)
2. View Monitoring Results
# Get monitoring stats
stats = customerentity.getfeaturemonitoringstats(
featureid="monthlycharges",
starttime="2024-01-01",
endtime="2024-02-01"
)
for stat in stats:
print(f"Date: {stat.date}")
print(f"Mean: {stat.mean}")
print(f"Stddev: {stat.stddev}")
Feature Groups (V2)
1. Create Feature Group
from google.cloud.aiplatformv1beta1 import (
FeatureRegistryServiceClient,
FeatureGroup,
BigQuerySource
)
client = FeatureRegistryServiceClient()
feature
group = FeatureGroup(
bigquery=BigQuerySource(
uri="bq://project.dataset.customerfeatures",
entityidcolumns=["customerid"]
),
description="Customer features from BigQuery"
)
operation = client.createfeaturegroup(
parent=f"projects/{project}/locations/{location}",
featuregroupid="customerfeatures",
featuregroup=featuregroup
)
result = operation.result()
print(f"Feature group created: {result.name}")
2. Create Feature View
from google.cloud.aiplatformv1beta1 import FeatureView
feature
view = FeatureView(
bigquerysource=BigQuerySource(
uri="bq://project.dataset.customerfeatures",
entityidcolumns=["customerid"]
),
featureregistrysource={
"featuregroups": {
"customerfeatures": {
"featureids": ["age", "tenuremonths", "monthlycharges"]
}
}
}
)
Create feature view for online serving
client.createfeatureview(
parent=f"projects/{project}/locations/{location}/featureOnlineStores/{storeid}",
featureviewid="customerview",
featureview=featureview
)
Best Practices
1. Feature Naming
# Use consistent naming conventions
features = {
"customerageyears": {"valuetype": "INT64"},
"customertenuremonths": {"valuetype": "INT64"},
"customermonthlychargesusd": {"valuetype": "DOUBLE"},
"customercontracttypecat": {"valuetype": "STRING"}
}
2. Resource Cleanup
# Delete features
customerentity.deletefeatures(["oldfeature"])
Delete entity type
customerentity.delete()
Delete feature store
featurestore.delete(force=True)
Conclusion
Vertex AI Feature Store provides:
Key takeaways:
- Organize features by entity types
- Use batch ingestion for efficiency
- Enable online serving for real-time
- Monitor feature distributions
- Use consistent naming conventions