GCP Terraform Best Practices: Infrastructure as Code at Enterprise Scale

Terraform is the de facto standard for managing GCP infrastructure as code. This guide covers Google provider configuration, module design for reusable GCP patterns, state management with Cloud Storage backend, policy as code with Sentinel and OPA, and CI/CD integration with Cloud Build.

Terraform has become the standard tool for managing GCP infrastructure. If you're clicking through the Cloud Console to provision resources, you're accumulating infrastructure debt — configurations that can't be reviewed, reproduced in a new region, or audited for security compliance.

Infrastructure as Code (IaC) with Terraform solves these problems. But Terraform for a personal project and Terraform for an enterprise GCP environment are different disciplines. At enterprise scale, you need thoughtful module design, shared state management, policy enforcement, and CI/CD automation.

This guide covers the patterns that separate maintainable enterprise Terraform from unmaintainable spaghetti infrastructure code.

Google Provider Configuration

# providers.tf
terraform {
  required_version = ">= 1.5.0"

  required_providers {
    google = {
      source  = "hashicorp/google"
      version = "~> 5.0"
    }
    google-beta = {
      source  = "hashicorp/google-beta"
      version = "~> 5.0"
    }
  }

  # Remote state in Cloud Storage
  backend "gcs" {
    bucket = "my-project-terraform-state"
    prefix = "production"
  }
}

provider "google" {
  project = var.project_id
  region  = var.region

  # Use Workload Identity for CI/CD instead of service account keys
  # When running in Cloud Build, no credential configuration needed —
  # Cloud Build authenticates automatically
}

provider "google-beta" {
  project = var.project_id
  region  = var.region
}
# versions.tf — pin all provider versions for reproducibility
terraform {
  required_providers {
    google = {
      source  = "hashicorp/google"
      version = "5.15.0"  # Pin exact version, not just major.minor
    }
  }
}

Exact version pinning prevents infrastructure surprises when provider updates include breaking changes.

Module Design Principles

Modules should have a single, clear responsibility. Avoid "super modules" that provision an entire application stack in one module — they're hard to test, hard to version, and create too many dependencies.

modules/
├── gke-cluster/          # Module: GKE cluster only
│   ├── main.tf
│   ├── variables.tf
│   ├── outputs.tf
│   └── versions.tf
├── gke-node-pool/        # Module: GKE node pool
├── cloud-sql-postgres/   # Module: Cloud SQL Postgres instance
├── vpc-network/          # Module: VPC and subnets
├── cloud-armor-waf/      # Module: Cloud Armor security policy
└── service-account/      # Module: IAM service account with bindings

Well-Designed Module Example

# modules/cloud-sql-postgres/main.tf

resource "google_sql_database_instance" "primary" {
  name             = var.instance_name
  database_version = "POSTGRES_15"
  region           = var.region
  project          = var.project_id

  deletion_protection = var.deletion_protection

  settings {
    tier                        = var.machine_type
    availability_type           = var.high_availability ? "REGIONAL" : "ZONAL"
    disk_type                   = "PD_SSD"
    disk_size                   = var.disk_size_gb
    disk_autoresize             = true
    disk_autoresize_limit       = var.disk_autoresize_limit_gb

    backup_configuration {
      enabled                        = true
      start_time                     = var.backup_start_time
      point_in_time_recovery_enabled = true
      transaction_log_retention_days = 7
      backup_retention_settings {
        retained_backups = 30
        retention_unit   = "COUNT"
      }
    }

    ip_configuration {
      ipv4_enabled                                  = false  # No public IP
      private_network                               = var.vpc_id
      enable_private_path_for_google_cloud_services = true
    }

    insights_config {
      query_insights_enabled  = true
      query_plans_per_minute  = 5
      query_string_length     = 1024
      record_application_tags = true
      record_client_address   = true
    }

    maintenance_window {
      day          = 7  # Sunday
      hour         = 3  # 3am
      update_track = "stable"
    }

    database_flags {
      name  = "log_min_duration_statement"
      value = "1000"  # Log queries slower than 1 second
    }
  }

  lifecycle {
    prevent_destroy = true  # Prevent accidental deletion
  }
}

resource "google_sql_database" "main" {
  name     = var.database_name
  instance = google_sql_database_instance.primary.name
  project  = var.project_id
}

resource "google_sql_user" "app_user" {
  name     = var.app_username
  instance = google_sql_database_instance.primary.name
  password = var.app_password  # Source from Secret Manager, not hardcoded
  project  = var.project_id
}
# modules/cloud-sql-postgres/variables.tf
variable "instance_name" {
  description = "Name of the Cloud SQL instance"
  type        = string
  validation {
    condition     = can(regex("^[a-z][a-z0-9-]{0,93}[a-z0-9]$", var.instance_name))
    error_message = "Instance name must start with letter, contain only lowercase letters, numbers, and hyphens."
  }
}

variable "machine_type" {
  description = "Cloud SQL machine type (e.g., db-custom-4-15360)"
  type        = string
  default     = "db-custom-2-7680"
}

variable "high_availability" {
  description = "Enable high availability (regional) deployment"
  type        = bool
  default     = true
}

variable "deletion_protection" {
  description = "Enable deletion protection (set to false before destroying)"
  type        = bool
  default     = true
}
# modules/cloud-sql-postgres/outputs.tf
output "connection_name" {
  description = "Cloud SQL instance connection name for Cloud SQL Auth Proxy"
  value       = google_sql_database_instance.primary.connection_name
}

output "private_ip_address" {
  description = "Private IP address for VPC-internal connections"
  value       = google_sql_database_instance.primary.private_ip_address
}

output "instance_name" {
  description = "Cloud SQL instance name"
  value       = google_sql_database_instance.primary.name
}

Remote State and State Locking

Always use remote state for team environments. Local state files cause merge conflicts and make collaboration impossible.

# Create the state bucket
gsutil mb -l europe-west4 gs://my-project-terraform-state
gsutil versioning set on gs://my-project-terraform-state
gsutil lifecycle set lifecycle.json gs://my-project-terraform-state

# lifecycle.json: keep 30 state versions, delete older ones
cat > lifecycle.json << 'EOF'
{
  "lifecycle": {
    "rule": [{
      "action": {"type": "Delete"},
      "condition": {
        "numNewerVersions": 30,
        "isLive": false
      }
    }]
  }
}
EOF
# backend.tf — configure remote state
terraform {
  backend "gcs" {
    bucket  = "my-project-terraform-state"
    prefix  = "environments/production"
  }
}

Cloud Storage state backend provides automatic state locking — Terraform writes a lock file to GCS before modifying state, preventing concurrent applies that would corrupt state.

Workspace Strategy for Multiple Environments

Terraform workspaces allow the same configuration to manage multiple environments (dev, staging, prod) with different variable values:

# Create and switch workspaces
terraform workspace new staging
terraform workspace new production
terraform workspace select production

# Show current workspace
terraform workspace show
# main.tf — use workspace-based configuration
locals {
  env_config = {
    staging = {
      machine_type   = "db-custom-2-7680"
      disk_size_gb   = 100
      ha_enabled     = false
    }
    production = {
      machine_type   = "db-custom-8-30720"
      disk_size_gb   = 500
      ha_enabled     = true
    }
  }

  config = local.env_config[terraform.workspace]
}

module "cloud_sql" {
  source       = "../../modules/cloud-sql-postgres"
  instance_name = "${var.project_id}-db-${terraform.workspace}"
  machine_type  = local.config.machine_type
  disk_size_gb  = local.config.disk_size_gb
  high_availability = local.config.ha_enabled
}

CI/CD Integration with Cloud Build

# cloudbuild-terraform.yaml
steps:
# Step 1: Initialize Terraform
- name: hashicorp/terraform:1.5.0
  id: terraform-init
  entrypoint: terraform
  args: [init, -backend-config=bucket=my-project-terraform-state]

# Step 2: Validate configuration
- name: hashicorp/terraform:1.5.0
  id: terraform-validate
  entrypoint: terraform
  args: [validate]
  waitFor: [terraform-init]

# Step 3: Plan (on PR — for review)
- name: hashicorp/terraform:1.5.0
  id: terraform-plan
  entrypoint: bash
  args:
  - -c
  - |
    terraform plan       -var="project_id=$PROJECT_ID"       -var="region=europe-west4"       -out=/workspace/tfplan       -no-color 2>&1 | tee /workspace/plan.txt
  waitFor: [terraform-validate]

# Step 4: Apply (on merge to main — production only)
- name: hashicorp/terraform:1.5.0
  id: terraform-apply
  entrypoint: terraform
  args:
  - apply
  - -auto-approve
  - /workspace/tfplan
  waitFor: [terraform-plan]

options:
  logging: CLOUD_LOGGING_ONLY
# Create Cloud Build trigger for Terraform plan on PRs
gcloud builds triggers create github   --name=terraform-plan   --repo-name=infrastructure   --repo-owner=my-org   --pull-request-pattern=.*   --build-config=cloudbuild-terraform.yaml   --comment-control=COMMENTS_ENABLED

# Create trigger for apply on merge to main
gcloud builds triggers create github   --name=terraform-apply   --repo-name=infrastructure   --repo-owner=my-org   --branch-pattern=^main$   --build-config=cloudbuild-terraform.yaml   --service-account=projects/my-project/serviceAccounts/terraform-sa@my-project.iam.gserviceaccount.com

Policy as Code with OPA

Open Policy Agent (OPA) enforces organizational policies before Terraform applies:

# policies/gcp_security.rego
package terraform.gcp.security

import future.keywords

# Rule: Cloud SQL instances must have deletion protection enabled
deny[msg] {
  resource := input.resource_changes[_]
  resource.type == "google_sql_database_instance"
  resource.change.after.settings[_].deletion_protection == false

  msg := sprintf(
    "Cloud SQL instance '%s' must have deletion_protection = true",
    [resource.address]
  )
}

# Rule: No public IP on Cloud SQL
deny[msg] {
  resource := input.resource_changes[_]
  resource.type == "google_sql_database_instance"
  resource.change.after.settings[_].ip_configuration[_].ipv4_enabled == true

  msg := sprintf(
    "Cloud SQL instance '%s' must not have a public IP (ipv4_enabled = false)",
    [resource.address]
  )
}

# Rule: GKE clusters must have private nodes
deny[msg] {
  resource := input.resource_changes[_]
  resource.type == "google_container_cluster"
  not resource.change.after.private_cluster_config[_].enable_private_nodes

  msg := sprintf(
    "GKE cluster '%s' must enable private nodes",
    [resource.address]
  )
}
# Add OPA policy check to Cloud Build pipeline
- name: openpolicyagent/opa:latest
  id: policy-check
  entrypoint: bash
  args:
  - -c
  - |
    terraform show -json /workspace/tfplan > /workspace/plan.json
    opa eval       --data policies/       --input /workspace/plan.json       --format pretty       "data.terraform.gcp.security.deny" | tee /workspace/policy_results.txt

    # Fail if any policies are violated
    if grep -q '"strings"' /workspace/policy_results.txt; then
      echo "Policy violations detected:"
      cat /workspace/policy_results.txt
      exit 1
    fi
  waitFor: [terraform-plan]

For GKE infrastructure managed with Terraform, see our GKE production configuration guide. For the Anthos multi-cluster patterns enabled by Terraform, see our Anthos hybrid cloud guide.