MLOps CI/CD: Automating Model Pipelines with Vertex AI and Cloud Build
Manual ML model deployment is error-prone and doesn't scale. This guide builds an automated CI/CD pipeline that trains, evaluates, and deploys models with Vertex AI Pipelines triggered by Cloud Build, with quality gates that prevent degraded models from reaching production.
In traditional software engineering, CI/CD is table stakes. Every PR triggers a build, every merge to main deploys to staging, every production release goes through automated testing.
In machine learning, manual deployment is still surprisingly common. A data scientist trains a model in a notebook, and an engineer manually copies it to a server. This process has no version control, no automated testing, no rollback mechanism, and no audit trail.
MLOps CI/CD applies software engineering discipline to the model lifecycle. Every change to training code triggers an automated pipeline. Models that don't meet quality thresholds are blocked from deployment. Production deployments happen through canary releases with automated rollback.
The MLOps CI/CD Architecture
Code Repository (GitHub/Cloud Source Repositories)
|
| merge to main
v
Cloud Build Trigger
|
| Step 1: Run unit tests
| Step 2: Build serving container
| Step 3: Submit Vertex AI Pipeline
|
v
Vertex AI Pipeline
|
| Component 1: Data validation
| Component 2: Model training
| Component 3: Model evaluation
| Component 4: Quality gate (pass/fail)
| Component 5: Register model (on pass)
|
v
Canary Endpoint Deployment
| Stage 1: 10% traffic to new model
| Stage 2: Automated smoke tests
| Stage 3: Full production rollout
Training Pipeline with Quality Gates
# pipeline_with_gates.py
from kfp import dsl, compiler
from kfp.dsl import Dataset, Model, Input, Output, Metrics
@dsl.component(
base_image="python:3.11",
packages_to_install=["pandas", "scikit-learn", "xgboost"],
)
def evaluate_model(
model: Input[Model],
test_dataset: Input[Dataset],
min_auc_threshold: float,
current_production_auc: float,
metrics: Output[Metrics],
) -> bool:
import pandas as pd
from xgboost import XGBClassifier
from sklearn.metrics import roc_auc_score
import json
df = pd.read_parquet(test_dataset.path)
feature_cols = [c for c in df.columns if c not in ["churned", "customer_id", "snapshot_date"]]
trained_model = XGBClassifier()
trained_model.load_model(f"{model.path}/model.json") # XGBoost native format
auc = roc_auc_score(df["churned"], trained_model.predict_proba(df[feature_cols])[:, 1])
metrics.log_metric("evaluation_auc", auc)
metrics.log_metric("threshold", min_auc_threshold)
metrics.log_metric("current_production_auc", current_production_auc)
# Dual gate: absolute minimum AND no regression vs production
should_deploy = auc >= min_auc_threshold and auc >= (current_production_auc - 0.02)
metrics.log_metric("deploy_decision", 1 if should_deploy else 0)
print(f"AUC: {auc:.4f}, Gate: {min_auc_threshold}, Production: {current_production_auc:.4f}")
print(f"Deploy decision: {should_deploy}")
return should_deploy
@dsl.component(
base_image="python:3.11",
packages_to_install=["google-cloud-aiplatform"],
)
def conditional_register(
model: Input[Model],
should_deploy: bool,
project_id: str,
region: str,
model_name: str,
serving_container: str,
commit_sha: str,
):
if not should_deploy:
print("Quality gate failed — model will not be registered")
return
from google.cloud import aiplatform
aiplatform.init(project=project_id, location=region)
vertex_model = aiplatform.Model.upload(
display_name=model_name,
artifact_uri=model.uri,
serving_container_image_uri=serving_container,
labels={"commit_sha": commit_sha[:8], "auto_deployed": "true"},
)
print(f"Model registered: {vertex_model.resource_name}")
@dsl.pipeline(name="churn-cicd-pipeline")
def churn_cicd_pipeline(
project_id: str,
region: str,
commit_sha: str,
min_auc_threshold: float = 0.82,
current_production_auc: float = 0.85,
test_size: float = 0.2,
):
extract_task = extract_training_data(project_id=project_id)
split_task = split_dataset(
dataset=extract_task.outputs["output_dataset"],
test_size=test_size,
)
train_task = train_model(
train_dataset=split_task.outputs["train_dataset"],
n_estimators=200,
)
eval_task = evaluate_model(
model=train_task.outputs["output_model"],
test_dataset=split_task.outputs["test_dataset"],
min_auc_threshold=min_auc_threshold,
current_production_auc=current_production_auc,
)
conditional_register(
model=train_task.outputs["output_model"],
should_deploy=eval_task.output,
project_id=project_id,
region=region,
model_name="churn-predictor",
serving_container=f"europe-west4-docker.pkg.dev/{project_id}/ml-models/churn-server:{commit_sha}",
commit_sha=commit_sha,
)
Cloud Build Configuration
# cloudbuild.yaml
steps:
- name: python:3.11
id: unit-tests
entrypoint: bash
args:
- -c
- |
pip install -q -r requirements.txt -r requirements-test.txt
python -m pytest tests/unit/ -v --tb=short
python -m pytest tests/pipeline/ -v --tb=short
- name: python:3.11
id: validate-pipeline
entrypoint: bash
args:
- -c
- |
pip install -q kfp google-cloud-aiplatform
python pipeline_with_gates.py --validate
- name: gcr.io/cloud-builders/docker
id: build-serving-container
args:
- build
- -t
- europe-west4-docker.pkg.dev/$PROJECT_ID/ml-models/churn-server:$COMMIT_SHA
- -f
- serving/Dockerfile
- serving/
waitFor: [unit-tests]
- name: gcr.io/cloud-builders/docker
id: push-serving-container
args:
- push
- europe-west4-docker.pkg.dev/$PROJECT_ID/ml-models/churn-server:$COMMIT_SHA
waitFor: [build-serving-container]
- name: python:3.11
id: submit-training-pipeline
entrypoint: bash
args:
- -c
- |
pip install -q google-cloud-aiplatform kfp
python -c "
from google.cloud import aiplatform
from pipeline_with_gates import churn_cicd_pipeline
from kfp import compiler
compiler.Compiler().compile(churn_cicd_pipeline, '/tmp/pipeline.yaml')
aiplatform.init(project='$PROJECT_ID', location='europe-west4')
job = aiplatform.PipelineJob(
display_name='cicd-$COMMIT_SHA',
template_path='/tmp/pipeline.yaml',
pipeline_root='gs://$PROJECT_ID-pipelines/churn/',
parameter_values={
'project_id': '$PROJECT_ID',
'region': 'europe-west4',
'commit_sha': '$COMMIT_SHA',
},
)
job.submit(service_account='vertex-pipelines-sa@$PROJECT_ID.iam.gserviceaccount.com')
job.wait()
if job.state.name != 'PIPELINE_STATE_SUCCEEDED':
raise Exception(f'Pipeline failed: {job.state.name}')
"
waitFor: [push-serving-container, validate-pipeline]
options:
logging: CLOUD_LOGGING_ONLY
machineType: E2_HIGHCPU_8
timeout: 7200s
Creating the Cloud Build Trigger
gcloud builds triggers create github --name=churn-mlops-pipeline --repo-name=my-ml-repo --repo-owner=my-org --branch-pattern=^main$ --build-config=cloudbuild.yaml --include-files=training/**,serving/**,pipeline_with_gates.py --service-account=projects/my-project/serviceAccounts/cloud-build-sa@my-project.iam.gserviceaccount.com
Fetching Current Production Metrics for the Quality Gate
def get_production_model_auc(model_display_name: str, project_id: str, region: str) -> float:
from google.cloud import aiplatform
aiplatform.init(project=project_id, location=region)
models = aiplatform.Model.list(
filter=f'display_name="{model_display_name}" AND labels.production="true"',
order_by="create_time desc",
)
if not models:
return 0.0 # No production model — any model passes the gate
return float(models[0].labels.get("auc", "0"))
Canary Deployment with Vertex AI Endpoints
def canary_deploy(new_model_name: str, endpoint_name: str, initial_traffic_pct: int = 10):
from google.cloud import aiplatform
endpoint = aiplatform.Endpoint.list(
filter=f'display_name="{endpoint_name}"'
)[0]
new_model = aiplatform.Model.list(
filter=f'display_name="{new_model_name}"',
order_by="create_time desc",
)[0]
endpoint.deploy(
model=new_model,
deployed_model_display_name=f"{new_model_name}-canary",
machine_type="n1-standard-4",
min_replica_count=1,
max_replica_count=5,
traffic_percentage=initial_traffic_pct,
)
print(f"Deployed {new_model_name} at {initial_traffic_pct}% traffic")
print("Monitor metrics for 15-30 minutes before promoting to full traffic")
Pipeline Failure Alerting
# Sink pipeline failures to BigQuery for analysis
gcloud logging sinks create mlops-failures bigquery.googleapis.com/projects/my-project/datasets/mlops_logs --log-filter='resource.type="aiplatform.googleapis.com/PipelineJob" AND severity="ERROR"'
This automation makes model deployment as reliable as software deployment. Teams can merge model improvements with confidence that regressions are caught before they reach users.
For the training pipeline design, see our Vertex AI MLOps guide. For Gemini model applications, see our Vertex AI Gemini enterprise guide.