CloudWatch Cost Management and Optimization
Reduce CloudWatch costs by optimizing metric collection, log retention, and implementing efficient monitoring strategies.
CloudWatch costs can grow significantly with scale. This guide covers strategies to optimize observability spending while maintaining effective monitoring.
CloudWatch Cost Components
Cost Breakdown
def estimate_cloudwatch_costs(metrics_count, log_gb, dashboard_count):
costs = {
'custom_metrics': {
'standard_resolution': metrics_count * 0.30, # Per metric per month
'high_resolution': metrics_count * 3.00 # Per metric per month
},
'logs': {
'ingestion': log_gb * 0.50, # Per GB ingested
'storage': log_gb * 0.03, # Per GB stored per month
'insights_queries': 0 # Per GB scanned
},
'dashboards': dashboard_count * 3.00, # Per dashboard per month
'alarms': {
'standard': 0.10, # Per alarm per month
'high_resolution': 0.30 # Per alarm per month
}
}
return costs
Metric Optimization
Reduce Custom Metrics
import boto3
def consolidate_metrics():
cloudwatch = boto3.client('cloudwatch')
# Use dimensions instead of separate metrics
cloudwatch.put_metric_data(
Namespace='MyApplication',
MetricData=[
{
'MetricName': 'RequestCount',
'Dimensions': [
{'Name': 'Service', 'Value': 'API'},
{'Name': 'Environment', 'Value': 'Production'},
{'Name': 'Endpoint', 'Value': '/users'}
],
'Value': 1,
'Unit': 'Count'
}
]
)
def use_embedded_metrics():
# Use CloudWatch Embedded Metric Format
import json
metric_log = {
'_aws': {
'Timestamp': int(time.time() * 1000),
'CloudWatchMetrics': [{
'Namespace': 'MyApp',
'Dimensions': [['Service', 'Environment']],
'Metrics': [
{'Name': 'RequestLatency', 'Unit': 'Milliseconds'},
{'Name': 'RequestCount', 'Unit': 'Count'}
]
}]
},
'Service': 'API',
'Environment': 'Production',
'RequestLatency': 45.2,
'RequestCount': 1
}
print(json.dumps(metric_log))
Standard vs High Resolution
# Use standard resolution (1-minute) unless required
StandardResolutionAlarm:
Type: AWS::CloudWatch::Alarm
Properties:
MetricName: CPUUtilization
Namespace: AWS/EC2
Period: 60 # 1 minute - standard resolution
Statistic: Average
Threshold: 80
# Cost: $0.10/month
# Reserve high resolution for critical metrics only
HighResolutionAlarm:
Type: AWS::CloudWatch::Alarm
Properties:
MetricName: TransactionLatency
Namespace: MyApp
Period: 10 # 10 seconds - high resolution
ExtendedStatistic: p99
Threshold: 500
# Cost: $0.30/month
Log Optimization
Retention Policies
def set_cost_effective_retention():
logs = boto3.client('logs')
retention_policies = {
'/aws/lambda/': 7, # Lambda logs: 7 days
'/ecs/': 14, # ECS logs: 14 days
'/application/debug': 3, # Debug logs: 3 days
'/application/prod': 30, # Production: 30 days
'/audit/': 365 # Audit logs: 1 year
}
log_groups = logs.describe_log_groups()
for group in log_groups['logGroups']:
name = group['logGroupName']
for prefix, days in retention_policies.items():
if name.startswith(prefix):
logs.put_retention_policy(
logGroupName=name,
retentionInDays=days
)
break
def archive_to_s3():
# Export logs to S3 for long-term storage
logs = boto3.client('logs')
logs.create_export_task(
logGroupName='/application/prod',
fromTime=int((datetime.now() - timedelta(days=30)).timestamp() * 1000),
to=int(datetime.now().timestamp() * 1000),
destination='log-archive-bucket',
destinationPrefix='cloudwatch-logs/'
)
Log Filtering
def filter_logs_at_source():
# In Lambda, filter before logging
import logging
logger = logging.getLogger()
logger.setLevel(logging.INFO) # Not DEBUG
# Only log actionable information
def log_request(event, response_time, status):
if status >= 400 or response_time > 1000:
logger.info({
'path': event['path'],
'status': status,
'response_time': response_time
})
Dashboard Optimization
Consolidate Dashboards
def create_efficient_dashboard():
cloudwatch = boto3.client('cloudwatch')
# One dashboard with multiple widgets instead of multiple dashboards
dashboard_body = {
'widgets': [
{
'type': 'metric',
'x': 0, 'y': 0, 'width': 12, 'height': 6,
'properties': {
'metrics': [
['AWS/EC2', 'CPUUtilization', 'AutoScalingGroupName', 'prod-asg'],
['AWS/ECS', 'CPUUtilization', 'ClusterName', 'prod-cluster'],
['AWS/Lambda', 'Duration', 'FunctionName', 'api-handler']
],
'period': 300,
'stat': 'Average',
'title': 'Compute Metrics'
}
}
]
}
cloudwatch.put_dashboard(
DashboardName='ConsolidatedMonitoring',
DashboardBody=json.dumps(dashboard_body)
)
Cost Monitoring
def monitor_cloudwatch_costs():
ce = boto3.client('ce')
response = ce.get_cost_and_usage(
TimePeriod={
'Start': (datetime.now() - timedelta(days=30)).strftime('%Y-%m-%d'),
'End': datetime.now().strftime('%Y-%m-%d')
},
Granularity='DAILY',
Metrics=['UnblendedCost'],
Filter={
'Dimensions': {
'Key': 'SERVICE',
'Values': ['Amazon CloudWatch']
}
},
GroupBy=[
{'Type': 'DIMENSION', 'Key': 'USAGE_TYPE'}
]
)
return response
Working with Warqline
We are a cloud engineering consultancy and an official AWS and Google Cloud partner. If you are running this in production and want a second pair of eyes, we scope work in a free 45-minute technical call: you describe what you are running and what worries you, and we tell you what we would look at first.
Conclusion
CloudWatch cost optimization requires balancing observability needs with spending. Focus on standard resolution metrics, implement log retention policies, and consolidate dashboards for significant savings.