CloudWatch Cost Management and Optimization

Reduce CloudWatch costs by optimizing metric collection, log retention, and implementing efficient monitoring strategies.

CloudWatch costs can grow significantly with scale. This guide covers strategies to optimize observability spending while maintaining effective monitoring.

CloudWatch Cost Components

Cost Breakdown

def estimate_cloudwatch_costs(metrics_count, log_gb, dashboard_count):
    costs = {
        'custom_metrics': {
            'standard_resolution': metrics_count * 0.30,  # Per metric per month
            'high_resolution': metrics_count * 3.00  # Per metric per month
        },
        'logs': {
            'ingestion': log_gb * 0.50,  # Per GB ingested
            'storage': log_gb * 0.03,  # Per GB stored per month
            'insights_queries': 0  # Per GB scanned
        },
        'dashboards': dashboard_count * 3.00,  # Per dashboard per month
        'alarms': {
            'standard': 0.10,  # Per alarm per month
            'high_resolution': 0.30  # Per alarm per month
        }
    }
    
    return costs

Metric Optimization

Reduce Custom Metrics

import boto3

def consolidate_metrics():
    cloudwatch = boto3.client('cloudwatch')
    
    # Use dimensions instead of separate metrics
    cloudwatch.put_metric_data(
        Namespace='MyApplication',
        MetricData=[
            {
                'MetricName': 'RequestCount',
                'Dimensions': [
                    {'Name': 'Service', 'Value': 'API'},
                    {'Name': 'Environment', 'Value': 'Production'},
                    {'Name': 'Endpoint', 'Value': '/users'}
                ],
                'Value': 1,
                'Unit': 'Count'
            }
        ]
    )

def use_embedded_metrics():
    # Use CloudWatch Embedded Metric Format
    import json
    
    metric_log = {
        '_aws': {
            'Timestamp': int(time.time() * 1000),
            'CloudWatchMetrics': [{
                'Namespace': 'MyApp',
                'Dimensions': [['Service', 'Environment']],
                'Metrics': [
                    {'Name': 'RequestLatency', 'Unit': 'Milliseconds'},
                    {'Name': 'RequestCount', 'Unit': 'Count'}
                ]
            }]
        },
        'Service': 'API',
        'Environment': 'Production',
        'RequestLatency': 45.2,
        'RequestCount': 1
    }
    
    print(json.dumps(metric_log))

Standard vs High Resolution

# Use standard resolution (1-minute) unless required
StandardResolutionAlarm:
  Type: AWS::CloudWatch::Alarm
  Properties:
    MetricName: CPUUtilization
    Namespace: AWS/EC2
    Period: 60  # 1 minute - standard resolution
    Statistic: Average
    Threshold: 80
    # Cost: $0.10/month

# Reserve high resolution for critical metrics only
HighResolutionAlarm:
  Type: AWS::CloudWatch::Alarm
  Properties:
    MetricName: TransactionLatency
    Namespace: MyApp
    Period: 10  # 10 seconds - high resolution
    ExtendedStatistic: p99
    Threshold: 500
    # Cost: $0.30/month

Log Optimization

Retention Policies

def set_cost_effective_retention():
    logs = boto3.client('logs')
    
    retention_policies = {
        '/aws/lambda/': 7,      # Lambda logs: 7 days
        '/ecs/': 14,            # ECS logs: 14 days
        '/application/debug': 3, # Debug logs: 3 days
        '/application/prod': 30, # Production: 30 days
        '/audit/': 365          # Audit logs: 1 year
    }
    
    log_groups = logs.describe_log_groups()
    
    for group in log_groups['logGroups']:
        name = group['logGroupName']
        
        for prefix, days in retention_policies.items():
            if name.startswith(prefix):
                logs.put_retention_policy(
                    logGroupName=name,
                    retentionInDays=days
                )
                break

def archive_to_s3():
    # Export logs to S3 for long-term storage
    logs = boto3.client('logs')
    
    logs.create_export_task(
        logGroupName='/application/prod',
        fromTime=int((datetime.now() - timedelta(days=30)).timestamp() * 1000),
        to=int(datetime.now().timestamp() * 1000),
        destination='log-archive-bucket',
        destinationPrefix='cloudwatch-logs/'
    )

Log Filtering

def filter_logs_at_source():
    # In Lambda, filter before logging
    import logging
    
    logger = logging.getLogger()
    logger.setLevel(logging.INFO)  # Not DEBUG
    
    # Only log actionable information
    def log_request(event, response_time, status):
        if status >= 400 or response_time > 1000:
            logger.info({
                'path': event['path'],
                'status': status,
                'response_time': response_time
            })

Dashboard Optimization

Consolidate Dashboards

def create_efficient_dashboard():
    cloudwatch = boto3.client('cloudwatch')
    
    # One dashboard with multiple widgets instead of multiple dashboards
    dashboard_body = {
        'widgets': [
            {
                'type': 'metric',
                'x': 0, 'y': 0, 'width': 12, 'height': 6,
                'properties': {
                    'metrics': [
                        ['AWS/EC2', 'CPUUtilization', 'AutoScalingGroupName', 'prod-asg'],
                        ['AWS/ECS', 'CPUUtilization', 'ClusterName', 'prod-cluster'],
                        ['AWS/Lambda', 'Duration', 'FunctionName', 'api-handler']
                    ],
                    'period': 300,
                    'stat': 'Average',
                    'title': 'Compute Metrics'
                }
            }
        ]
    }
    
    cloudwatch.put_dashboard(
        DashboardName='ConsolidatedMonitoring',
        DashboardBody=json.dumps(dashboard_body)
    )

Cost Monitoring

def monitor_cloudwatch_costs():
    ce = boto3.client('ce')
    
    response = ce.get_cost_and_usage(
        TimePeriod={
            'Start': (datetime.now() - timedelta(days=30)).strftime('%Y-%m-%d'),
            'End': datetime.now().strftime('%Y-%m-%d')
        },
        Granularity='DAILY',
        Metrics=['UnblendedCost'],
        Filter={
            'Dimensions': {
                'Key': 'SERVICE',
                'Values': ['Amazon CloudWatch']
            }
        },
        GroupBy=[
            {'Type': 'DIMENSION', 'Key': 'USAGE_TYPE'}
        ]
    )
    
    return response

Working with Warqline

We are a cloud engineering consultancy and an official AWS and Google Cloud partner. If you are running this in production and want a second pair of eyes, we scope work in a free 45-minute technical call: you describe what you are running and what worries you, and we tell you what we would look at first.

Talk to an engineer

Conclusion

CloudWatch cost optimization requires balancing observability needs with spending. Focus on standard resolution metrics, implement log retention policies, and consolidate dashboards for significant savings.