Compose EKS monitoring modules (#115)

* Move modules around

* Update amp billing source

* Merge Java monitoring to EKS

* Update docs

* Merge nginx pattern

* Pre-commit

* Add save and test URL output

* Move EKS dependencies to EKS monitoring module

* update docs

* Update examples and docs

* Add java doc

* Add NGINX doc

* Update nginx doc

* Fix amp monitoring example path

* Fix pre-commit

* Todo: move to main after merge

* Update docs, fix tags
This commit is contained in:
Rodrigue Koffi
2023-02-20 18:36:08 +01:00
committed by GitHub
parent fe83579997
commit daed34db80
99 changed files with 887 additions and 1477 deletions
@@ -0,0 +1,55 @@
# Observability Pattern for Amazon Managed Prometheus
This module provides an automated experience around Observability for AMP (Amazon Managed Prometheus) workspaces.
It provides the following resources:
- AWS Managed Grafana Dashboard
- Cloudwatch data source to monitor AMP usage and alert metrics.
Note: The Billing widget of the dashboard requires [CloudWatch Billing Alerts](https://docs.aws.amazon.com/AmazonCloudWatch/latest/monitoring/monitor_estimated_charges_with_cloudwatch.html) to be enabled.
- CloudWatch alarms for AMP service quotas.
<!-- BEGIN_TF_DOCS -->
## Requirements
| Name | Version |
|------|---------|
| <a name="requirement_terraform"></a> [terraform](#requirement\_terraform) | >= 1.1.0, < 1.3.0 |
| <a name="requirement_aws"></a> [aws](#requirement\_aws) | >= 4.0.0 |
| <a name="requirement_grafana"></a> [grafana](#requirement\_grafana) | >= 1.25.0 |
## Providers
| Name | Version |
|------|---------|
| <a name="provider_aws"></a> [aws](#provider\_aws) | >= 4.0.0 |
| <a name="provider_grafana"></a> [grafana](#provider\_grafana) | >= 1.25.0 |
## Modules
No modules.
## Resources
| Name | Type |
|------|------|
| [aws_cloudwatch_metric_alarm.active-series-metrics](https://registry.terraform.io/providers/hashicorp/aws/latest/docs/resources/cloudwatch_metric_alarm) | resource |
| [aws_cloudwatch_metric_alarm.ingestion_rate](https://registry.terraform.io/providers/hashicorp/aws/latest/docs/resources/cloudwatch_metric_alarm) | resource |
| [grafana_dashboard.this](https://registry.terraform.io/providers/grafana/grafana/latest/docs/resources/dashboard) | resource |
| [grafana_data_source.cloudwatch](https://registry.terraform.io/providers/grafana/grafana/latest/docs/resources/data_source) | resource |
## Inputs
| Name | Description | Type | Default | Required |
|------|-------------|------|---------|:--------:|
| <a name="input_active_series_threshold"></a> [active\_series\_threshold](#input\_active\_series\_threshold) | Threshold for active series metric alarm | `number` | `1000000` | no |
| <a name="input_aws_region"></a> [aws\_region](#input\_aws\_region) | AWS Region | `string` | n/a | yes |
| <a name="input_dashboards_folder_id"></a> [dashboards\_folder\_id](#input\_dashboards\_folder\_id) | Grafana folder ID for automatic dashboards | `string` | n/a | yes |
| <a name="input_ingestion_rate_threshold"></a> [ingestion\_rate\_threshold](#input\_ingestion\_rate\_threshold) | Threshold for active series metric alarm | `number` | `70000` | no |
| <a name="input_managed_prometheus_workspace_id"></a> [managed\_prometheus\_workspace\_id](#input\_managed\_prometheus\_workspace\_id) | Amazon Managed Service for Prometheus Workspace ID to create Alarms for | `string` | n/a | yes |
## Outputs
No outputs.
<!-- END_TF_DOCS -->
@@ -0,0 +1,59 @@
#CloudWatch Alerts on AMP Usage
resource "aws_cloudwatch_metric_alarm" "active_series_metrics" {
for_each = local.amp_list
alarm_name = "active-series-metrics"
comparison_operator = "GreaterThanOrEqualToThreshold"
evaluation_periods = "2"
threshold = var.active_series_threshold
alarm_description = "This metric monitors AMP active series metrics"
insufficient_data_actions = []
metric_query {
id = "m1"
return_data = true
metric {
metric_name = "ResourceCount"
namespace = "AWS/Usage"
period = "120"
stat = "Average"
unit = "None"
dimensions = {
Type = "Resource"
ResourceId = each.key
Resource = "ActiveSeries"
Service = "Prometheus"
Class = "None"
}
}
}
}
resource "aws_cloudwatch_metric_alarm" "ingestion_rate" {
for_each = local.amp_list
alarm_name = "ingestion_rate"
comparison_operator = "GreaterThanOrEqualToThreshold"
evaluation_periods = "2"
threshold = var.ingestion_rate_threshold
alarm_description = "This metric monitors AMP ingestion rate"
insufficient_data_actions = []
metric_query {
id = "m1"
return_data = true
metric {
metric_name = "ResourceCount"
namespace = "AWS/Usage"
period = "120"
stat = "Average"
unit = "None"
dimensions = {
Type = "Resource"
ResourceId = each.key
Resource = "IngestionRate"
Service = "Prometheus"
Class = "None"
}
}
}
}
@@ -0,0 +1,32 @@
resource "aws_cloudwatch_metric_alarm" "amp_billing_anomaly_detection" {
alarm_name = "amp_billing_anomaly"
comparison_operator = "GreaterThanUpperThreshold"
evaluation_periods = "2"
threshold_metric_id = "e1"
alarm_description = "This metric monitors ec2 cpu utilization"
insufficient_data_actions = []
metric_query {
id = "e1"
expression = "ANOMALY_DETECTION_BAND(m1)"
label = "Expected AMP Charges"
return_data = "true"
}
metric_query {
id = "m1"
return_data = "true"
metric {
metric_name = "Estimated Charges"
namespace = "AWS/Billing"
period = "21600"
stat = "Maximum"
unit = "Count"
dimensions = {
ServiceName = "Prometheus"
Currencty = "USD"
}
}
}
}
@@ -0,0 +1,14 @@
terraform {
required_version = ">= 1.1.0"
required_providers {
aws = {
source = "hashicorp/aws"
version = ">= 4.0.0"
}
grafana = {
source = "grafana/grafana"
version = ">= 1.25.0"
}
}
}
@@ -0,0 +1,795 @@
{
"annotations": {
"list": [
{
"builtIn": 1,
"datasource": "-- Grafana --",
"enable": true,
"hide": true,
"iconColor": "rgba(0, 211, 255, 1)",
"name": "Annotations & Alerts",
"target": {
"limit": 100,
"matchAny": false,
"tags": [],
"type": "dashboard"
},
"type": "dashboard"
}
]
},
"description": "Dashboard for Amazon Managed Prometheus",
"editable": true,
"fiscalYearStartMonth": 0,
"graphTooltip": 0,
"id": 51,
"iteration": 1666292684202,
"links": [],
"liveNow": false,
"panels": [
{
"gridPos": {
"h": 7,
"w": 5,
"x": 0,
"y": 0
},
"id": 16,
"options": {
"content": "# Ingestion Usage Metrics\n\nMetrics relating to ingestion usage of the AMP service",
"mode": "markdown"
},
"pluginVersion": "8.4.7",
"title": "Usage",
"type": "text"
},
{
"datasource": {
"type": "cloudwatch",
"uid": "$datasource"
},
"fieldConfig": {
"defaults": {
"color": {
"mode": "palette-classic"
},
"custom": {
"axisLabel": "",
"axisPlacement": "auto",
"barAlignment": 0,
"drawStyle": "line",
"fillOpacity": 0,
"gradientMode": "none",
"hideFrom": {
"legend": false,
"tooltip": false,
"viz": false
},
"lineInterpolation": "linear",
"lineWidth": 1,
"pointSize": 5,
"scaleDistribution": {
"type": "linear"
},
"showPoints": "auto",
"spanNulls": false,
"stacking": {
"group": "A",
"mode": "none"
},
"thresholdsStyle": {
"mode": "off"
}
},
"mappings": [],
"thresholds": {
"mode": "absolute",
"steps": [
{
"color": "green",
"value": null
},
{
"color": "red",
"value": 80
}
]
}
},
"overrides": []
},
"gridPos": {
"h": 7,
"w": 9,
"x": 5,
"y": 0
},
"id": 6,
"options": {
"legend": {
"calcs": [],
"displayMode": "list",
"placement": "bottom"
},
"tooltip": {
"mode": "single",
"sort": "none"
}
},
"targets": [
{
"alias": "",
"datasource": {
"type": "cloudwatch",
"uid": "$datasource"
},
"dimensions": {},
"expression": "SELECT SUM(ResourceCount) FROM SCHEMA(\"AWS/Usage\", Class,Resource,ResourceId,Service,Type) WHERE Type = 'Resource' AND ResourceId = '$WorkspaceID' AND Resource = 'ActiveSeries' AND Service = 'Prometheus' AND Class = 'None'",
"id": "",
"matchExact": true,
"metricEditorMode": 1,
"metricName": "",
"metricQueryType": 0,
"namespace": "",
"period": "",
"queryMode": "Metrics",
"refId": "A",
"region": "default",
"sqlExpression": "",
"statistic": "Average"
}
],
"title": "Active Series Metrics",
"type": "timeseries"
},
{
"datasource": {
"type": "cloudwatch",
"uid": "$datasource"
},
"fieldConfig": {
"defaults": {
"color": {
"mode": "palette-classic"
},
"custom": {
"axisLabel": "",
"axisPlacement": "auto",
"barAlignment": 0,
"drawStyle": "line",
"fillOpacity": 0,
"gradientMode": "none",
"hideFrom": {
"legend": false,
"tooltip": false,
"viz": false
},
"lineInterpolation": "linear",
"lineWidth": 1,
"pointSize": 5,
"scaleDistribution": {
"type": "linear"
},
"showPoints": "auto",
"spanNulls": false,
"stacking": {
"group": "A",
"mode": "none"
},
"thresholdsStyle": {
"mode": "off"
}
},
"mappings": [],
"thresholds": {
"mode": "absolute",
"steps": [
{
"color": "green",
"value": null
},
{
"color": "red",
"value": 80
}
]
}
},
"overrides": []
},
"gridPos": {
"h": 7,
"w": 9,
"x": 14,
"y": 0
},
"id": 2,
"options": {
"legend": {
"calcs": [],
"displayMode": "list",
"placement": "bottom"
},
"tooltip": {
"mode": "single",
"sort": "none"
}
},
"targets": [
{
"alias": "",
"datasource": {
"type": "cloudwatch",
"uid": "$datasource"
},
"dimensions": {},
"expression": "SELECT AVG(ResourceCount) FROM SCHEMA(\"AWS/Usage\", Class,Resource,ResourceId,Service,Type) WHERE Type = 'Resource' AND ResourceId = '$WorkspaceID' AND Resource = 'IngestionRate' AND Service = 'Prometheus' AND Class = 'None'",
"id": "",
"matchExact": true,
"metricEditorMode": 1,
"metricName": "",
"metricQueryType": 0,
"namespace": "",
"period": "",
"queryMode": "Metrics",
"refId": "A",
"region": "default",
"sqlExpression": "",
"statistic": "Average"
}
],
"title": "Workspace Ingestion Rate",
"type": "timeseries"
},
{
"gridPos": {
"h": 8,
"w": 5,
"x": 0,
"y": 7
},
"id": 22,
"options": {
"content": "# Billing\n\nContains information relating to the cost of AMP\n\n",
"mode": "markdown"
},
"pluginVersion": "8.4.7",
"title": "Billing",
"type": "text"
},
{
"datasource": {
"type": "cloudwatch",
"uid": "$datasource"
},
"fieldConfig": {
"defaults": {
"color": {
"mode": "palette-classic"
},
"custom": {
"axisLabel": "",
"axisPlacement": "auto",
"barAlignment": 0,
"drawStyle": "line",
"fillOpacity": 0,
"gradientMode": "none",
"hideFrom": {
"legend": false,
"tooltip": false,
"viz": false
},
"lineInterpolation": "linear",
"lineWidth": 1,
"pointSize": 5,
"scaleDistribution": {
"type": "linear"
},
"showPoints": "auto",
"spanNulls": false,
"stacking": {
"group": "A",
"mode": "none"
},
"thresholdsStyle": {
"mode": "off"
}
},
"mappings": [],
"thresholds": {
"mode": "absolute",
"steps": [
{
"color": "green",
"value": null
},
{
"color": "red",
"value": 80
}
]
}
},
"overrides": []
},
"gridPos": {
"h": 8,
"w": 18,
"x": 5,
"y": 7
},
"id": 24,
"options": {
"legend": {
"calcs": [],
"displayMode": "list",
"placement": "bottom"
},
"tooltip": {
"mode": "single",
"sort": "none"
}
},
"targets": [
{
"alias": "",
"datasource": {
"type": "cloudwatch",
"uid": "$datasource"
},
"dimensions": {},
"expression": "SELECT SUM(EstimatedCharges) FROM SCHEMA(\"AWS/Billing\", Currency,ServiceName) WHERE ServiceName = 'AmazonPrometheus'",
"id": "",
"matchExact": true,
"metricEditorMode": 1,
"metricName": "",
"metricQueryType": 0,
"namespace": "",
"period": "",
"queryMode": "Metrics",
"refId": "A",
"region": "default",
"sqlExpression": "",
"statistic": "Average"
}
],
"title": "Sum of Estimated AMP Charges (total)",
"type": "timeseries"
},
{
"gridPos": {
"h": 9,
"w": 5,
"x": 0,
"y": 15
},
"id": 14,
"options": {
"content": "# Alert Usage Metrics\n\nMetrics associated with Alertmanager Alert Usage",
"mode": "markdown"
},
"pluginVersion": "8.4.7",
"title": "Alerts",
"type": "text"
},
{
"datasource": {
"type": "cloudwatch",
"uid": "$datasource"
},
"fieldConfig": {
"defaults": {
"color": {
"mode": "thresholds"
},
"mappings": [],
"thresholds": {
"mode": "absolute",
"steps": [
{
"color": "green",
"value": null
},
{
"color": "red",
"value": 1
}
]
}
},
"overrides": []
},
"gridPos": {
"h": 9,
"w": 4,
"x": 5,
"y": 15
},
"id": 4,
"options": {
"colorMode": "value",
"graphMode": "area",
"justifyMode": "auto",
"orientation": "auto",
"reduceOptions": {
"calcs": [
"lastNotNull"
],
"fields": "",
"values": false
},
"textMode": "auto"
},
"pluginVersion": "8.4.7",
"targets": [
{
"alias": "",
"datasource": {
"type": "cloudwatch",
"uid": "$datasource"
},
"dimensions": {},
"expression": "SELECT AVG(ResourceCount) FROM SCHEMA(\"AWS/Usage\", Class,Resource,ResourceId,Service,Type) WHERE Type = 'Resource' AND ResourceId = '$WorkspaceID' AND Resource = 'ActiveAlerts' AND Service = 'Prometheus' AND Class = 'None'",
"id": "",
"matchExact": true,
"metricEditorMode": 1,
"metricName": "",
"metricQueryType": 0,
"namespace": "",
"period": "",
"queryMode": "Metrics",
"refId": "A",
"region": "default",
"sqlExpression": "",
"statistic": "Average"
}
],
"title": "Active Alerts",
"type": "stat"
},
{
"datasource": {
"type": "cloudwatch",
"uid": "$datasource"
},
"fieldConfig": {
"defaults": {
"color": {
"mode": "thresholds"
},
"mappings": [],
"thresholds": {
"mode": "absolute",
"steps": [
{
"color": "green",
"value": null
},
{
"color": "red",
"value": 1
}
]
}
},
"overrides": []
},
"gridPos": {
"h": 9,
"w": 5,
"x": 9,
"y": 15
},
"id": 12,
"options": {
"colorMode": "value",
"graphMode": "area",
"justifyMode": "auto",
"orientation": "auto",
"reduceOptions": {
"calcs": [
"lastNotNull"
],
"fields": "",
"values": false
},
"textMode": "auto"
},
"pluginVersion": "8.4.7",
"targets": [
{
"alias": "",
"datasource": {
"type": "cloudwatch",
"uid": "$datasource"
},
"dimensions": {},
"expression": "SELECT AVG(AlertManagerNotificationsFailed) FROM SCHEMA(\"AWS/Prometheus\", Workspace) WHERE Workspace = '$WorkspaceID'",
"id": "",
"matchExact": true,
"metricEditorMode": 1,
"metricName": "",
"metricQueryType": 0,
"namespace": "",
"period": "",
"queryMode": "Metrics",
"refId": "A",
"region": "default",
"sqlExpression": "",
"statistic": "Average"
}
],
"title": "Alert Manager Notifications Failed",
"type": "stat"
},
{
"datasource": {
"type": "cloudwatch",
"uid": "$datasource"
},
"fieldConfig": {
"defaults": {
"color": {
"mode": "thresholds"
},
"mappings": [],
"thresholds": {
"mode": "absolute",
"steps": [
{
"color": "green",
"value": null
}
]
}
},
"overrides": []
},
"gridPos": {
"h": 9,
"w": 4,
"x": 14,
"y": 15
},
"id": 10,
"options": {
"colorMode": "value",
"graphMode": "area",
"justifyMode": "auto",
"orientation": "auto",
"reduceOptions": {
"calcs": [
"lastNotNull"
],
"fields": "",
"values": false
},
"textMode": "auto"
},
"pluginVersion": "8.4.7",
"targets": [
{
"alias": "",
"datasource": {
"type": "cloudwatch",
"uid": "$datasource"
},
"dimensions": {},
"expression": "SELECT AVG(AlertManagerAlertsReceived) FROM SCHEMA(\"AWS/Prometheus\", Workspace) WHERE Workspace = '$WorkspaceID'",
"id": "",
"matchExact": true,
"metricEditorMode": 1,
"metricName": "",
"metricQueryType": 0,
"namespace": "",
"period": "",
"queryMode": "Metrics",
"refId": "A",
"region": "default",
"sqlExpression": "",
"statistic": "Average"
}
],
"title": "Alert Manager Alerts Received",
"type": "stat"
},
{
"datasource": {
"type": "cloudwatch",
"uid": "$datasource"
},
"fieldConfig": {
"defaults": {
"color": {
"mode": "thresholds"
},
"mappings": [],
"thresholds": {
"mode": "absolute",
"steps": [
{
"color": "green",
"value": null
},
{
"color": "red",
"value": 80
}
]
}
},
"overrides": []
},
"gridPos": {
"h": 9,
"w": 5,
"x": 18,
"y": 15
},
"id": 8,
"options": {
"colorMode": "value",
"graphMode": "area",
"justifyMode": "auto",
"orientation": "auto",
"reduceOptions": {
"calcs": [
"lastNotNull"
],
"fields": "",
"values": false
},
"textMode": "auto"
},
"pluginVersion": "8.4.7",
"targets": [
{
"alias": "",
"datasource": {
"type": "cloudwatch",
"uid": "$datasource"
},
"dimensions": {},
"expression": "SELECT AVG(ResourceCount) FROM SCHEMA(\"AWS/Usage\", Class,Resource,ResourceId,Service,Type) WHERE Type = 'Resource' AND ResourceId = '$WorkspaceID' AND Resource = 'SizeOfAlerts' AND Service = 'Prometheus' AND Class = 'None'",
"id": "",
"matchExact": true,
"metricEditorMode": 1,
"metricName": "",
"metricQueryType": 0,
"namespace": "",
"period": "",
"queryMode": "Metrics",
"refId": "A",
"region": "default",
"sqlExpression": "",
"statistic": "Average"
}
],
"title": "Size of Alerts",
"type": "stat"
},
{
"gridPos": {
"h": 7,
"w": 5,
"x": 0,
"y": 24
},
"id": 20,
"options": {
"content": "# AMP Vended Logs\n\nLast 25 log events from AMP Vended Logs for alert and rule evaluation",
"mode": "markdown"
},
"pluginVersion": "8.4.7",
"title": "AMP Logs",
"type": "text"
},
{
"datasource": {
"type": "cloudwatch",
"uid": "$datasource"
},
"gridPos": {
"h": 7,
"w": 18,
"x": 5,
"y": 24
},
"id": 18,
"options": {
"dedupStrategy": "none",
"enableLogDetails": true,
"prettifyLogMessage": false,
"showCommonLabels": false,
"showLabels": false,
"showTime": false,
"sortOrder": "Descending",
"wrapLogMessage": false
},
"targets": [
{
"datasource": {
"type": "cloudwatch",
"uid": "$datasource"
},
"expression": "fields @timestamp, @message\n| sort @timestamp desc\n| limit 25",
"id": "",
"logGroupNames": [
"/aws/vendedlogs/amp"
],
"namespace": "",
"queryMode": "Logs",
"refId": "A",
"region": "default",
"statsGroups": []
}
],
"timeFrom": "6h",
"timeShift": "6h",
"title": "AMP Vended Logs",
"type": "logs"
}
],
"refresh": "",
"schemaVersion": 35,
"style": "dark",
"tags": [],
"templating": {
"list": [
{
"current": {
"selected": true,
"text": [
"ws-e8b003eb-0528-4208-b31c-edf4598d5f66"
],
"value": [
"ws-e8b003eb-0528-4208-b31c-edf4598d5f66"
]
},
"datasource": {
"type": "cloudwatch",
"uid": "$datasource"
},
"definition": "dimension_values(default,AWS/Prometheus,RuleEvaluations,Workspace)",
"hide": 0,
"includeAll": false,
"multi": true,
"name": "WorkspaceID",
"options": [],
"query": "dimension_values(default,AWS/Prometheus,RuleEvaluations,Workspace)",
"refresh": 1,
"regex": "",
"skipUrlSync": false,
"sort": 0,
"type": "query"
},
{
"current": {
"selected": false,
"text": "Amazon CloudWatch us-west-2",
"value": "Amazon CloudWatch us-west-2"
},
"hide": 0,
"includeAll": false,
"multi": false,
"name": "datasource",
"options": [],
"query": "cloudwatch",
"refresh": 1,
"regex": "",
"skipUrlSync": false,
"type": "datasource"
}
]
},
"time": {
"from": "now-6h",
"to": "now"
},
"timepicker": {},
"timezone": "",
"title": "AMP Accelerator Dashboard",
"uid": "",
"version": 1,
"weekStart": ""
}
@@ -0,0 +1,35 @@
provider "aws" {
region = "us-east-1"
alias = "billing_region"
}
locals {
name = "aws-observability-accelerator-cloudwatch"
amp_list = toset(split(",", var.managed_prometheus_workspace_ids))
}
resource "grafana_data_source" "cloudwatch" {
type = "cloudwatch"
name = local.name
# Giving priority to Managed Prometheus datasources
is_default = false
json_data {
default_region = var.aws_region
sigv4_auth = true
sigv4_auth_type = "workspace-iam-role"
sigv4_region = var.aws_region
}
}
resource "grafana_dashboard" "this" {
folder = var.dashboards_folder_id
config_json = file("${path.module}/dashboards/amp-dashboard.json")
}
module "billing" {
source = "./billing"
providers = {
aws = aws.billing_region
}
}
@@ -0,0 +1,4 @@
output "grafana_dashboard_urls" {
value = [grafana_dashboard.this.url]
description = "URLs for dashboards created"
}
@@ -0,0 +1,26 @@
variable "dashboards_folder_id" {
description = "Grafana folder ID for automatic dashboards"
type = string
}
variable "aws_region" {
description = "AWS Region"
type = string
}
variable "managed_prometheus_workspace_ids" {
description = "Amazon Managed Service for Prometheus Workspace ID to create Alarms for"
type = string
}
variable "active_series_threshold" {
description = "Threshold for active series metric alarm"
type = number
default = 1000000
}
variable "ingestion_rate_threshold" {
description = "Threshold for active series metric alarm"
type = number
default = 70000
}
@@ -0,0 +1,14 @@
terraform {
required_version = ">= 1.1.0"
required_providers {
aws = {
source = "hashicorp/aws"
version = ">= 4.0.0"
}
grafana = {
source = "grafana/grafana"
version = ">= 1.25.0"
}
}
}