From 07545aca27c05f576ad8f53b1d6a7837bd4f08a5 Mon Sep 17 00:00:00 2001 From: Kevin Lewin <97046295+lewinkedrs@users.noreply.github.com> Date: Fri, 9 Dec 2022 10:49:55 -0500 Subject: [PATCH] AMP Observability Pattern (#55) * working first run * removing core module dependencies * adding CW datasource * alarms MVP * readmes * Adding Screenshot * Adding billing note * adding billing module * Revert "adding billing module" This reverts commit 40d667e37db1036cd71a471ef2fde83ec02aaa13. reverting * adding billing module * Updating Screenshot * resolving feedback * removing unused modules * fmt * Support for tf 1.3.x * removing unused variables * support alarms for multiple workspaces * Updating Readme * amp to managed prometheus * sub-module * Fix pre-commit Co-authored-by: Rodrigue Koffi --- .../managed-prometheus-monitoring/README.md | 151 ++++ .../managed-prometheus-monitoring/main.tf | 29 + .../managed-prometheus-monitoring/outputs.tf | 4 + .../variables.tf | 20 + .../managed-prometheus-monitoring/versions.tf | 14 + .../workloads/infra/dashboards/cluster.json | 2 +- .../workloads/infra/dashboards/kubelet.json | 2 +- .../infra/dashboards/namespace-workloads.json | 2 +- .../infra/dashboards/nodeexporter-nodes.json | 2 +- modules/workloads/infra/dashboards/nodes.json | 2 +- .../workloads/infra/dashboards/workloads.json | 2 +- .../managed-prometheus-monitoring/README.md | 55 ++ .../managed-prometheus-monitoring/alarms.tf | 59 ++ .../billing/main.tf | 32 + .../billing/outputs.tf | 0 .../billing/variables.tf | 0 .../billing/versions.tf | 14 + .../dashboards/amp-dashboard.json | 795 ++++++++++++++++++ .../managed-prometheus-monitoring/locals.tf | 0 .../managed-prometheus-monitoring/main.tf | 33 + .../managed-prometheus-monitoring/outputs.tf | 4 + .../variables.tf | 26 + .../managed-prometheus-monitoring/versions.tf | 14 + 23 files changed, 1256 insertions(+), 6 deletions(-) create mode 100644 examples/managed-prometheus-monitoring/README.md create mode 100644 examples/managed-prometheus-monitoring/main.tf create mode 100644 examples/managed-prometheus-monitoring/outputs.tf create mode 100644 examples/managed-prometheus-monitoring/variables.tf create mode 100644 examples/managed-prometheus-monitoring/versions.tf create mode 100644 modules/workloads/managed-prometheus-monitoring/README.md create mode 100644 modules/workloads/managed-prometheus-monitoring/alarms.tf create mode 100644 modules/workloads/managed-prometheus-monitoring/billing/main.tf create mode 100644 modules/workloads/managed-prometheus-monitoring/billing/outputs.tf create mode 100644 modules/workloads/managed-prometheus-monitoring/billing/variables.tf create mode 100644 modules/workloads/managed-prometheus-monitoring/billing/versions.tf create mode 100644 modules/workloads/managed-prometheus-monitoring/dashboards/amp-dashboard.json create mode 100644 modules/workloads/managed-prometheus-monitoring/locals.tf create mode 100644 modules/workloads/managed-prometheus-monitoring/main.tf create mode 100644 modules/workloads/managed-prometheus-monitoring/outputs.tf create mode 100644 modules/workloads/managed-prometheus-monitoring/variables.tf create mode 100644 modules/workloads/managed-prometheus-monitoring/versions.tf diff --git a/examples/managed-prometheus-monitoring/README.md b/examples/managed-prometheus-monitoring/README.md new file mode 100644 index 0000000..2f18016 --- /dev/null +++ b/examples/managed-prometheus-monitoring/README.md @@ -0,0 +1,151 @@ +# Existing Managed Prometheus Workspace Observability Pattern + +This example demonstrates how to use the AWS Observability Accelerator Terraform +modules with Amazon Managed Prometheus (AMP) workspace monitoring enabled. + +The current example deploys a dashboard into an existing Amazon Managed Grafana (AMG) workspace to provide observability over an existing AMP workspace. It also deploys CloudWatch alarms for AMP usage service limits. + +## Prerequisites + +Ensure that you have the following tools installed locally: + +1. [aws cli](https://docs.aws.amazon.com/cli/latest/userguide/getting-started-install.html) +2. [terraform](https://learn.hashicorp.com/tutorials/terraform/install-cli) + +It is also required to have existing AMP and Grafana workspaces. These could be created through the [other example modules](../) in this repository. + +## Setup + +This example uses a local terraform state. If you need states to be saved remotely, +on Amazon S3 for example, visit the [terraform remote states](https://www.terraform.io/language/state/remote) documentation + +1. **Clone the repo using the command below** + +```sh +git clone https://github.com/aws-observability/terraform-aws-observability-accelerator.git +``` + +2. **Initialize terraform** + +```sh +cd examples/amp-monitoring +terraform init +``` + +3. **AWS Region** + +Specify the AWS Region where the resources will be deployed. Edit the `terraform.tfvars` file and modify `aws_region="..."`. You can also use environement variables `export TF_VAR_aws_region=xxx`. + +4. **Amazon Managed Service for Prometheus workspace** + +If you have an existing workspace, add `managed_prometheus_workspace_id=ws-xxx` +or use an environment variable `export TF_VAR_managed_prometheus_workspace_id=ws-xxx`. + +If you would like to create CloudWatch alarms for multiple workspaces in a region you can pass them in a comma seperated string. + +`managed_prometheus_workspace_id = "ws-xxx,ws-xxx"` + +You can use the following export command to create alarms for all of the workspaces in a region. + +```sh +export TF_VAR_managed_prometheus_workspace_id=$(aws amp list-workspaces --query 'workspaces[].workspaceId' --output text | sed -E 's/\t/,/g') +``` + +5. **Amazon Managed Grafana workspace** + +Use an existing workspace, add `managed_grafana_workspace_id=g-xxx` +or use an environment variable `export TF_VAR_managed_grafana_workspace_id=g-xxx`. + +6. **Grafana API Key** + +Amazon Managed Service for Grafana provides a control plane API for generating Grafana API keys. We will provide to Terraform +a short lived API key to run the `apply` or `destroy` command. +Ensure you have necessary IAM permissions (`CreateWorkspaceApiKey, DeleteWorkspaceApiKey`) + +```sh +export TF_VAR_grafana_api_key=`aws grafana create-workspace-api-key --key-name "observability-accelerator-$(date +%s)" --key-role ADMIN --seconds-to-live 1200 --workspace-id $TF_VAR_managed_grafana_workspace_id --query key --output text` +``` + +## Deploy + +```sh +terraform apply -var-file=terraform.tfvars +``` + +or if you had only setup environment variables, run + +```sh +terraform apply +``` + +## Visualization + +1. **Cloudwatch datasource on Grafana** + +Open your Grafana workspace and under Configuration -> Data sources, you should see `aws-observability-accelerator-cloudwatch`. Open and click `Save & test`. You should see a notification confirming that the CloudWatch datasource is ready to be used on Grafana. + +2. **Grafana dashboards** + +Go to the Dashboards panel of your Grafana workspace. You should see a list of dashboards under the `AMP Monitoring Dashboards` folder. + +Open the `AMP Accelerator Dashboard` to see a visualization of the AMP workspace. + +Screen Shot 2022-10-11 at 2 16 17 PM + +3. **Amazon Managed Service for Prometheus CloudWatch Alarms.** + +Open the CloudWatch console and click `Alarms` > `All Alarms` to review the service limit alarms. + +image + +In us-east-1 region an alarm is created for billing. This alarm utilizes anomaly detection to detect anomalies in the Estimated Charges billing metric. + +image + + + + +## Requirements + +| Name | Version | +|------|---------| +| [terraform](#requirement\_terraform) | >= 1.1.0, < 1.3.0 | +| [aws](#requirement\_aws) | >= 4.0.0 | +| [grafana](#requirement\_grafana) | >= 1.25.0 | + +## Providers + +| Name | Version | +|------|---------| +| [aws](#provider\_aws) | 4.36.1 | +| [grafana](#provider\_grafana) | 1.30.0 | + +## Modules + +| Name | Source | Version | +|------|--------|---------| +| [amp\_monitor](#module\_amp\_monitor) | ../../modules/workloads/amp-monitoring | n/a | +| [billing](#module\_billing) | ../../modules/Billing | n/a | + +## Resources + +| Name | Type | +|------|------| +| [grafana_folder.this](https://registry.terraform.io/providers/grafana/grafana/latest/docs/resources/folder) | resource | +| [aws_grafana_workspace.this](https://registry.terraform.io/providers/hashicorp/aws/latest/docs/data-sources/grafana_workspace) | data source | + +## Inputs + +| Name | Description | Type | Default | Required | +|------|-------------|------|---------|:--------:| +| [aws\_region](#input\_aws\_region) | AWS Region | `string` | n/a | yes | +| [grafana\_api\_key](#input\_grafana\_api\_key) | API key for authorizing the Grafana provider to make changes to Amazon Managed Grafana | `string` | n/a | yes | +| [managed\_grafana\_workspace\_id](#input\_managed\_grafana\_workspace\_id) | Amazon Managed Grafana (AMG) workspace ID | `string` | n/a | yes | +| [managed\_prometheus\_workspace\_id](#input\_managed\_prometheus\_workspace\_id) | Amazon Managed Service for Prometheus Workspace ID to create Alarms for | `string` | n/a | yes | + +## Outputs + +| Name | Description | +|------|-------------| +| [grafana\_dashboards\_folder\_id](#output\_grafana\_dashboards\_folder\_id) | Grafana folder ID for automatic dashboards. Required by workload modules | + diff --git a/examples/managed-prometheus-monitoring/main.tf b/examples/managed-prometheus-monitoring/main.tf new file mode 100644 index 0000000..ee246ef --- /dev/null +++ b/examples/managed-prometheus-monitoring/main.tf @@ -0,0 +1,29 @@ +provider "aws" { + region = local.region +} + +provider "grafana" { + url = local.amg_ws_endpoint + auth = var.grafana_api_key +} + +data "aws_grafana_workspace" "this" { + count = var.managed_grafana_workspace_id == "" ? 0 : 1 + workspace_id = var.managed_grafana_workspace_id +} + +locals { + region = var.aws_region + amg_ws_endpoint = "https://${data.aws_grafana_workspace.this[0].endpoint}" +} + +resource "grafana_folder" "this" { + title = "Amazon Managed Prometheus monitoring dashboards" +} + +module "managed_prometheus_monitoring" { + source = "../../modules/workloads/managed-prometheus-monitoring" + dashboards_folder_id = resource.grafana_folder.this.id + aws_region = local.region + managed_prometheus_workspace_ids = var.managed_prometheus_workspace_ids +} diff --git a/examples/managed-prometheus-monitoring/outputs.tf b/examples/managed-prometheus-monitoring/outputs.tf new file mode 100644 index 0000000..af9a5e7 --- /dev/null +++ b/examples/managed-prometheus-monitoring/outputs.tf @@ -0,0 +1,4 @@ +output "grafana_dashboard_urls" { + description = "URLs for dashboards created" + value = module.managed_prometheus_monitoring.grafana_dashboard_urls +} diff --git a/examples/managed-prometheus-monitoring/variables.tf b/examples/managed-prometheus-monitoring/variables.tf new file mode 100644 index 0000000..98226f5 --- /dev/null +++ b/examples/managed-prometheus-monitoring/variables.tf @@ -0,0 +1,20 @@ +variable "grafana_api_key" { + description = "API key for authorizing the Grafana provider to make changes to Amazon Managed Grafana" + type = string + sensitive = true +} + +variable "aws_region" { + description = "AWS Region" + type = string +} + +variable "managed_prometheus_workspace_ids" { + description = "Amazon Managed Service for Prometheus Workspace IDs to create Alarms for" + type = string +} + +variable "managed_grafana_workspace_id" { + description = "Amazon Managed Grafana workspace ID" + type = string +} diff --git a/examples/managed-prometheus-monitoring/versions.tf b/examples/managed-prometheus-monitoring/versions.tf new file mode 100644 index 0000000..ac93e75 --- /dev/null +++ b/examples/managed-prometheus-monitoring/versions.tf @@ -0,0 +1,14 @@ +terraform { + required_version = ">= 1.1.0" + + required_providers { + aws = { + source = "hashicorp/aws" + version = ">= 4.0.0" + } + grafana = { + source = "grafana/grafana" + version = ">= 1.25.0" + } + } +} diff --git a/modules/workloads/infra/dashboards/cluster.json b/modules/workloads/infra/dashboards/cluster.json index 3d6947b..5f6196c 100644 --- a/modules/workloads/infra/dashboards/cluster.json +++ b/modules/workloads/infra/dashboards/cluster.json @@ -2905,4 +2905,4 @@ "uid": "efa86fd1d0c121a26444b636a3f509a8", "version": 3, "weekStart": "" -} \ No newline at end of file +} diff --git a/modules/workloads/infra/dashboards/kubelet.json b/modules/workloads/infra/dashboards/kubelet.json index 754b207..8dcbf3c 100644 --- a/modules/workloads/infra/dashboards/kubelet.json +++ b/modules/workloads/infra/dashboards/kubelet.json @@ -2233,4 +2233,4 @@ "uid": "3138fa155d5915769fbded898ac09fd9", "version": 19, "weekStart": "" -} \ No newline at end of file +} diff --git a/modules/workloads/infra/dashboards/namespace-workloads.json b/modules/workloads/infra/dashboards/namespace-workloads.json index 085423b..011a2f0 100644 --- a/modules/workloads/infra/dashboards/namespace-workloads.json +++ b/modules/workloads/infra/dashboards/namespace-workloads.json @@ -2657,4 +2657,4 @@ "uid": "a87fb0d919ec0ea5f6543124e16c42a5", "version": 2, "weekStart": "" -} \ No newline at end of file +} diff --git a/modules/workloads/infra/dashboards/nodeexporter-nodes.json b/modules/workloads/infra/dashboards/nodeexporter-nodes.json index 4400b87..e5e27ff 100644 --- a/modules/workloads/infra/dashboards/nodeexporter-nodes.json +++ b/modules/workloads/infra/dashboards/nodeexporter-nodes.json @@ -1279,4 +1279,4 @@ "uid": "v8yDYJqnz", "version": 18, "weekStart": "" -} \ No newline at end of file +} diff --git a/modules/workloads/infra/dashboards/nodes.json b/modules/workloads/infra/dashboards/nodes.json index 55f825c..8a32402 100644 --- a/modules/workloads/infra/dashboards/nodes.json +++ b/modules/workloads/infra/dashboards/nodes.json @@ -1480,4 +1480,4 @@ "uid": "200ac8fdbfbb74b39aff88118e4d1c2c", "version": 7, "weekStart": "" -} \ No newline at end of file +} diff --git a/modules/workloads/infra/dashboards/workloads.json b/modules/workloads/infra/dashboards/workloads.json index c3b9f1f..4378a38 100644 --- a/modules/workloads/infra/dashboards/workloads.json +++ b/modules/workloads/infra/dashboards/workloads.json @@ -2295,4 +2295,4 @@ "uid": "a164a7f0339f99e89cea5cb47e9be617", "version": 7, "weekStart": "" -} \ No newline at end of file +} diff --git a/modules/workloads/managed-prometheus-monitoring/README.md b/modules/workloads/managed-prometheus-monitoring/README.md new file mode 100644 index 0000000..4ec39b4 --- /dev/null +++ b/modules/workloads/managed-prometheus-monitoring/README.md @@ -0,0 +1,55 @@ +# Observability Pattern for Amazon Managed Prometheus + +This module provides an automated experience around Observability for AMP (Amazon Managed Prometheus) workspaces. +It provides the following resources: + +- AWS Managed Grafana Dashboard +- Cloudwatch data source to monitor AMP usage and alert metrics. + +Note: The Billing widget of the dashboard requires [CloudWatch Billing Alerts](https://docs.aws.amazon.com/AmazonCloudWatch/latest/monitoring/monitor_estimated_charges_with_cloudwatch.html) to be enabled. + +- CloudWatch alarms for AMP service quotas. + + +## Requirements + +| Name | Version | +|------|---------| +| [terraform](#requirement\_terraform) | >= 1.1.0, < 1.3.0 | +| [aws](#requirement\_aws) | >= 4.0.0 | +| [grafana](#requirement\_grafana) | >= 1.25.0 | + +## Providers + +| Name | Version | +|------|---------| +| [aws](#provider\_aws) | >= 4.0.0 | +| [grafana](#provider\_grafana) | >= 1.25.0 | + +## Modules + +No modules. + +## Resources + +| Name | Type | +|------|------| +| [aws_cloudwatch_metric_alarm.active-series-metrics](https://registry.terraform.io/providers/hashicorp/aws/latest/docs/resources/cloudwatch_metric_alarm) | resource | +| [aws_cloudwatch_metric_alarm.ingestion_rate](https://registry.terraform.io/providers/hashicorp/aws/latest/docs/resources/cloudwatch_metric_alarm) | resource | +| [grafana_dashboard.this](https://registry.terraform.io/providers/grafana/grafana/latest/docs/resources/dashboard) | resource | +| [grafana_data_source.cloudwatch](https://registry.terraform.io/providers/grafana/grafana/latest/docs/resources/data_source) | resource | + +## Inputs + +| Name | Description | Type | Default | Required | +|------|-------------|------|---------|:--------:| +| [active\_series\_threshold](#input\_active\_series\_threshold) | Threshold for active series metric alarm | `number` | `1000000` | no | +| [aws\_region](#input\_aws\_region) | AWS Region | `string` | n/a | yes | +| [dashboards\_folder\_id](#input\_dashboards\_folder\_id) | Grafana folder ID for automatic dashboards | `string` | n/a | yes | +| [ingestion\_rate\_threshold](#input\_ingestion\_rate\_threshold) | Threshold for active series metric alarm | `number` | `70000` | no | +| [managed\_prometheus\_workspace\_id](#input\_managed\_prometheus\_workspace\_id) | Amazon Managed Service for Prometheus Workspace ID to create Alarms for | `string` | n/a | yes | + +## Outputs + +No outputs. + diff --git a/modules/workloads/managed-prometheus-monitoring/alarms.tf b/modules/workloads/managed-prometheus-monitoring/alarms.tf new file mode 100644 index 0000000..3a6eba8 --- /dev/null +++ b/modules/workloads/managed-prometheus-monitoring/alarms.tf @@ -0,0 +1,59 @@ +#CloudWatch Alerts on AMP Usage +resource "aws_cloudwatch_metric_alarm" "active_series_metrics" { + for_each = local.amp_list + alarm_name = "active-series-metrics" + comparison_operator = "GreaterThanOrEqualToThreshold" + evaluation_periods = "2" + threshold = var.active_series_threshold + alarm_description = "This metric monitors AMP active series metrics" + insufficient_data_actions = [] + metric_query { + id = "m1" + return_data = true + metric { + metric_name = "ResourceCount" + namespace = "AWS/Usage" + period = "120" + stat = "Average" + unit = "None" + + dimensions = { + Type = "Resource" + ResourceId = each.key + Resource = "ActiveSeries" + Service = "Prometheus" + Class = "None" + } + } + } +} + +resource "aws_cloudwatch_metric_alarm" "ingestion_rate" { + for_each = local.amp_list + alarm_name = "ingestion_rate" + comparison_operator = "GreaterThanOrEqualToThreshold" + evaluation_periods = "2" + threshold = var.ingestion_rate_threshold + alarm_description = "This metric monitors AMP ingestion rate" + insufficient_data_actions = [] + metric_query { + id = "m1" + return_data = true + + metric { + metric_name = "ResourceCount" + namespace = "AWS/Usage" + period = "120" + stat = "Average" + unit = "None" + + dimensions = { + Type = "Resource" + ResourceId = each.key + Resource = "IngestionRate" + Service = "Prometheus" + Class = "None" + } + } + } +} diff --git a/modules/workloads/managed-prometheus-monitoring/billing/main.tf b/modules/workloads/managed-prometheus-monitoring/billing/main.tf new file mode 100644 index 0000000..c351580 --- /dev/null +++ b/modules/workloads/managed-prometheus-monitoring/billing/main.tf @@ -0,0 +1,32 @@ +resource "aws_cloudwatch_metric_alarm" "amp_billing_anomaly_detection" { + alarm_name = "amp_billing_anomaly" + comparison_operator = "GreaterThanUpperThreshold" + evaluation_periods = "2" + threshold_metric_id = "e1" + alarm_description = "This metric monitors ec2 cpu utilization" + insufficient_data_actions = [] + + metric_query { + id = "e1" + expression = "ANOMALY_DETECTION_BAND(m1)" + label = "Expected AMP Charges" + return_data = "true" + } + + metric_query { + id = "m1" + return_data = "true" + metric { + metric_name = "Estimated Charges" + namespace = "AWS/Billing" + period = "21600" + stat = "Maximum" + unit = "Count" + + dimensions = { + ServiceName = "Prometheus" + Currencty = "USD" + } + } + } +} diff --git a/modules/workloads/managed-prometheus-monitoring/billing/outputs.tf b/modules/workloads/managed-prometheus-monitoring/billing/outputs.tf new file mode 100644 index 0000000..e69de29 diff --git a/modules/workloads/managed-prometheus-monitoring/billing/variables.tf b/modules/workloads/managed-prometheus-monitoring/billing/variables.tf new file mode 100644 index 0000000..e69de29 diff --git a/modules/workloads/managed-prometheus-monitoring/billing/versions.tf b/modules/workloads/managed-prometheus-monitoring/billing/versions.tf new file mode 100644 index 0000000..ac93e75 --- /dev/null +++ b/modules/workloads/managed-prometheus-monitoring/billing/versions.tf @@ -0,0 +1,14 @@ +terraform { + required_version = ">= 1.1.0" + + required_providers { + aws = { + source = "hashicorp/aws" + version = ">= 4.0.0" + } + grafana = { + source = "grafana/grafana" + version = ">= 1.25.0" + } + } +} diff --git a/modules/workloads/managed-prometheus-monitoring/dashboards/amp-dashboard.json b/modules/workloads/managed-prometheus-monitoring/dashboards/amp-dashboard.json new file mode 100644 index 0000000..66681a9 --- /dev/null +++ b/modules/workloads/managed-prometheus-monitoring/dashboards/amp-dashboard.json @@ -0,0 +1,795 @@ +{ + "annotations": { + "list": [ + { + "builtIn": 1, + "datasource": "-- Grafana --", + "enable": true, + "hide": true, + "iconColor": "rgba(0, 211, 255, 1)", + "name": "Annotations & Alerts", + "target": { + "limit": 100, + "matchAny": false, + "tags": [], + "type": "dashboard" + }, + "type": "dashboard" + } + ] + }, + "description": "Dashboard for Amazon Managed Prometheus", + "editable": true, + "fiscalYearStartMonth": 0, + "graphTooltip": 0, + "id": 51, + "iteration": 1666292684202, + "links": [], + "liveNow": false, + "panels": [ + { + "gridPos": { + "h": 7, + "w": 5, + "x": 0, + "y": 0 + }, + "id": 16, + "options": { + "content": "# Ingestion Usage Metrics\n\nMetrics relating to ingestion usage of the AMP service", + "mode": "markdown" + }, + "pluginVersion": "8.4.7", + "title": "Usage", + "type": "text" + }, + { + "datasource": { + "type": "cloudwatch", + "uid": "$datasource" + }, + "fieldConfig": { + "defaults": { + "color": { + "mode": "palette-classic" + }, + "custom": { + "axisLabel": "", + "axisPlacement": "auto", + "barAlignment": 0, + "drawStyle": "line", + "fillOpacity": 0, + "gradientMode": "none", + "hideFrom": { + "legend": false, + "tooltip": false, + "viz": false + }, + "lineInterpolation": "linear", + "lineWidth": 1, + "pointSize": 5, + "scaleDistribution": { + "type": "linear" + }, + "showPoints": "auto", + "spanNulls": false, + "stacking": { + "group": "A", + "mode": "none" + }, + "thresholdsStyle": { + "mode": "off" + } + }, + "mappings": [], + "thresholds": { + "mode": "absolute", + "steps": [ + { + "color": "green", + "value": null + }, + { + "color": "red", + "value": 80 + } + ] + } + }, + "overrides": [] + }, + "gridPos": { + "h": 7, + "w": 9, + "x": 5, + "y": 0 + }, + "id": 6, + "options": { + "legend": { + "calcs": [], + "displayMode": "list", + "placement": "bottom" + }, + "tooltip": { + "mode": "single", + "sort": "none" + } + }, + "targets": [ + { + "alias": "", + "datasource": { + "type": "cloudwatch", + "uid": "$datasource" + }, + "dimensions": {}, + "expression": "SELECT SUM(ResourceCount) FROM SCHEMA(\"AWS/Usage\", Class,Resource,ResourceId,Service,Type) WHERE Type = 'Resource' AND ResourceId = '$WorkspaceID' AND Resource = 'ActiveSeries' AND Service = 'Prometheus' AND Class = 'None'", + "id": "", + "matchExact": true, + "metricEditorMode": 1, + "metricName": "", + "metricQueryType": 0, + "namespace": "", + "period": "", + "queryMode": "Metrics", + "refId": "A", + "region": "default", + "sqlExpression": "", + "statistic": "Average" + } + ], + "title": "Active Series Metrics", + "type": "timeseries" + }, + { + "datasource": { + "type": "cloudwatch", + "uid": "$datasource" + }, + "fieldConfig": { + "defaults": { + "color": { + "mode": "palette-classic" + }, + "custom": { + "axisLabel": "", + "axisPlacement": "auto", + "barAlignment": 0, + "drawStyle": "line", + "fillOpacity": 0, + "gradientMode": "none", + "hideFrom": { + "legend": false, + "tooltip": false, + "viz": false + }, + "lineInterpolation": "linear", + "lineWidth": 1, + "pointSize": 5, + "scaleDistribution": { + "type": "linear" + }, + "showPoints": "auto", + "spanNulls": false, + "stacking": { + "group": "A", + "mode": "none" + }, + "thresholdsStyle": { + "mode": "off" + } + }, + "mappings": [], + "thresholds": { + "mode": "absolute", + "steps": [ + { + "color": "green", + "value": null + }, + { + "color": "red", + "value": 80 + } + ] + } + }, + "overrides": [] + }, + "gridPos": { + "h": 7, + "w": 9, + "x": 14, + "y": 0 + }, + "id": 2, + "options": { + "legend": { + "calcs": [], + "displayMode": "list", + "placement": "bottom" + }, + "tooltip": { + "mode": "single", + "sort": "none" + } + }, + "targets": [ + { + "alias": "", + "datasource": { + "type": "cloudwatch", + "uid": "$datasource" + }, + "dimensions": {}, + "expression": "SELECT AVG(ResourceCount) FROM SCHEMA(\"AWS/Usage\", Class,Resource,ResourceId,Service,Type) WHERE Type = 'Resource' AND ResourceId = '$WorkspaceID' AND Resource = 'IngestionRate' AND Service = 'Prometheus' AND Class = 'None'", + "id": "", + "matchExact": true, + "metricEditorMode": 1, + "metricName": "", + "metricQueryType": 0, + "namespace": "", + "period": "", + "queryMode": "Metrics", + "refId": "A", + "region": "default", + "sqlExpression": "", + "statistic": "Average" + } + ], + "title": "Workspace Ingestion Rate", + "type": "timeseries" + }, + { + "gridPos": { + "h": 8, + "w": 5, + "x": 0, + "y": 7 + }, + "id": 22, + "options": { + "content": "# Billing\n\nContains information relating to the cost of AMP\n\n", + "mode": "markdown" + }, + "pluginVersion": "8.4.7", + "title": "Billing", + "type": "text" + }, + { + "datasource": { + "type": "cloudwatch", + "uid": "$datasource" + }, + "fieldConfig": { + "defaults": { + "color": { + "mode": "palette-classic" + }, + "custom": { + "axisLabel": "", + "axisPlacement": "auto", + "barAlignment": 0, + "drawStyle": "line", + "fillOpacity": 0, + "gradientMode": "none", + "hideFrom": { + "legend": false, + "tooltip": false, + "viz": false + }, + "lineInterpolation": "linear", + "lineWidth": 1, + "pointSize": 5, + "scaleDistribution": { + "type": "linear" + }, + "showPoints": "auto", + "spanNulls": false, + "stacking": { + "group": "A", + "mode": "none" + }, + "thresholdsStyle": { + "mode": "off" + } + }, + "mappings": [], + "thresholds": { + "mode": "absolute", + "steps": [ + { + "color": "green", + "value": null + }, + { + "color": "red", + "value": 80 + } + ] + } + }, + "overrides": [] + }, + "gridPos": { + "h": 8, + "w": 18, + "x": 5, + "y": 7 + }, + "id": 24, + "options": { + "legend": { + "calcs": [], + "displayMode": "list", + "placement": "bottom" + }, + "tooltip": { + "mode": "single", + "sort": "none" + } + }, + "targets": [ + { + "alias": "", + "datasource": { + "type": "cloudwatch", + "uid": "$datasource" + }, + "dimensions": {}, + "expression": "SELECT SUM(EstimatedCharges) FROM SCHEMA(\"AWS/Billing\", Currency,ServiceName) WHERE ServiceName = 'AmazonPrometheus'", + "id": "", + "matchExact": true, + "metricEditorMode": 1, + "metricName": "", + "metricQueryType": 0, + "namespace": "", + "period": "", + "queryMode": "Metrics", + "refId": "A", + "region": "default", + "sqlExpression": "", + "statistic": "Average" + } + ], + "title": "Sum of Estimated AMP Charges (total)", + "type": "timeseries" + }, + { + "gridPos": { + "h": 9, + "w": 5, + "x": 0, + "y": 15 + }, + "id": 14, + "options": { + "content": "# Alert Usage Metrics\n\nMetrics associated with Alertmanager Alert Usage", + "mode": "markdown" + }, + "pluginVersion": "8.4.7", + "title": "Alerts", + "type": "text" + }, + { + "datasource": { + "type": "cloudwatch", + "uid": "$datasource" + }, + "fieldConfig": { + "defaults": { + "color": { + "mode": "thresholds" + }, + "mappings": [], + "thresholds": { + "mode": "absolute", + "steps": [ + { + "color": "green", + "value": null + }, + { + "color": "red", + "value": 1 + } + ] + } + }, + "overrides": [] + }, + "gridPos": { + "h": 9, + "w": 4, + "x": 5, + "y": 15 + }, + "id": 4, + "options": { + "colorMode": "value", + "graphMode": "area", + "justifyMode": "auto", + "orientation": "auto", + "reduceOptions": { + "calcs": [ + "lastNotNull" + ], + "fields": "", + "values": false + }, + "textMode": "auto" + }, + "pluginVersion": "8.4.7", + "targets": [ + { + "alias": "", + "datasource": { + "type": "cloudwatch", + "uid": "$datasource" + }, + "dimensions": {}, + "expression": "SELECT AVG(ResourceCount) FROM SCHEMA(\"AWS/Usage\", Class,Resource,ResourceId,Service,Type) WHERE Type = 'Resource' AND ResourceId = '$WorkspaceID' AND Resource = 'ActiveAlerts' AND Service = 'Prometheus' AND Class = 'None'", + "id": "", + "matchExact": true, + "metricEditorMode": 1, + "metricName": "", + "metricQueryType": 0, + "namespace": "", + "period": "", + "queryMode": "Metrics", + "refId": "A", + "region": "default", + "sqlExpression": "", + "statistic": "Average" + } + ], + "title": "Active Alerts", + "type": "stat" + }, + { + "datasource": { + "type": "cloudwatch", + "uid": "$datasource" + }, + "fieldConfig": { + "defaults": { + "color": { + "mode": "thresholds" + }, + "mappings": [], + "thresholds": { + "mode": "absolute", + "steps": [ + { + "color": "green", + "value": null + }, + { + "color": "red", + "value": 1 + } + ] + } + }, + "overrides": [] + }, + "gridPos": { + "h": 9, + "w": 5, + "x": 9, + "y": 15 + }, + "id": 12, + "options": { + "colorMode": "value", + "graphMode": "area", + "justifyMode": "auto", + "orientation": "auto", + "reduceOptions": { + "calcs": [ + "lastNotNull" + ], + "fields": "", + "values": false + }, + "textMode": "auto" + }, + "pluginVersion": "8.4.7", + "targets": [ + { + "alias": "", + "datasource": { + "type": "cloudwatch", + "uid": "$datasource" + }, + "dimensions": {}, + "expression": "SELECT AVG(AlertManagerNotificationsFailed) FROM SCHEMA(\"AWS/Prometheus\", Workspace) WHERE Workspace = '$WorkspaceID'", + "id": "", + "matchExact": true, + "metricEditorMode": 1, + "metricName": "", + "metricQueryType": 0, + "namespace": "", + "period": "", + "queryMode": "Metrics", + "refId": "A", + "region": "default", + "sqlExpression": "", + "statistic": "Average" + } + ], + "title": "Alert Manager Notifications Failed", + "type": "stat" + }, + { + "datasource": { + "type": "cloudwatch", + "uid": "$datasource" + }, + "fieldConfig": { + "defaults": { + "color": { + "mode": "thresholds" + }, + "mappings": [], + "thresholds": { + "mode": "absolute", + "steps": [ + { + "color": "green", + "value": null + } + ] + } + }, + "overrides": [] + }, + "gridPos": { + "h": 9, + "w": 4, + "x": 14, + "y": 15 + }, + "id": 10, + "options": { + "colorMode": "value", + "graphMode": "area", + "justifyMode": "auto", + "orientation": "auto", + "reduceOptions": { + "calcs": [ + "lastNotNull" + ], + "fields": "", + "values": false + }, + "textMode": "auto" + }, + "pluginVersion": "8.4.7", + "targets": [ + { + "alias": "", + "datasource": { + "type": "cloudwatch", + "uid": "$datasource" + }, + "dimensions": {}, + "expression": "SELECT AVG(AlertManagerAlertsReceived) FROM SCHEMA(\"AWS/Prometheus\", Workspace) WHERE Workspace = '$WorkspaceID'", + "id": "", + "matchExact": true, + "metricEditorMode": 1, + "metricName": "", + "metricQueryType": 0, + "namespace": "", + "period": "", + "queryMode": "Metrics", + "refId": "A", + "region": "default", + "sqlExpression": "", + "statistic": "Average" + } + ], + "title": "Alert Manager Alerts Received", + "type": "stat" + }, + { + "datasource": { + "type": "cloudwatch", + "uid": "$datasource" + }, + "fieldConfig": { + "defaults": { + "color": { + "mode": "thresholds" + }, + "mappings": [], + "thresholds": { + "mode": "absolute", + "steps": [ + { + "color": "green", + "value": null + }, + { + "color": "red", + "value": 80 + } + ] + } + }, + "overrides": [] + }, + "gridPos": { + "h": 9, + "w": 5, + "x": 18, + "y": 15 + }, + "id": 8, + "options": { + "colorMode": "value", + "graphMode": "area", + "justifyMode": "auto", + "orientation": "auto", + "reduceOptions": { + "calcs": [ + "lastNotNull" + ], + "fields": "", + "values": false + }, + "textMode": "auto" + }, + "pluginVersion": "8.4.7", + "targets": [ + { + "alias": "", + "datasource": { + "type": "cloudwatch", + "uid": "$datasource" + }, + "dimensions": {}, + "expression": "SELECT AVG(ResourceCount) FROM SCHEMA(\"AWS/Usage\", Class,Resource,ResourceId,Service,Type) WHERE Type = 'Resource' AND ResourceId = '$WorkspaceID' AND Resource = 'SizeOfAlerts' AND Service = 'Prometheus' AND Class = 'None'", + "id": "", + "matchExact": true, + "metricEditorMode": 1, + "metricName": "", + "metricQueryType": 0, + "namespace": "", + "period": "", + "queryMode": "Metrics", + "refId": "A", + "region": "default", + "sqlExpression": "", + "statistic": "Average" + } + ], + "title": "Size of Alerts", + "type": "stat" + }, + { + "gridPos": { + "h": 7, + "w": 5, + "x": 0, + "y": 24 + }, + "id": 20, + "options": { + "content": "# AMP Vended Logs\n\nLast 25 log events from AMP Vended Logs for alert and rule evaluation", + "mode": "markdown" + }, + "pluginVersion": "8.4.7", + "title": "AMP Logs", + "type": "text" + }, + { + "datasource": { + "type": "cloudwatch", + "uid": "$datasource" + }, + "gridPos": { + "h": 7, + "w": 18, + "x": 5, + "y": 24 + }, + "id": 18, + "options": { + "dedupStrategy": "none", + "enableLogDetails": true, + "prettifyLogMessage": false, + "showCommonLabels": false, + "showLabels": false, + "showTime": false, + "sortOrder": "Descending", + "wrapLogMessage": false + }, + "targets": [ + { + "datasource": { + "type": "cloudwatch", + "uid": "$datasource" + }, + "expression": "fields @timestamp, @message\n| sort @timestamp desc\n| limit 25", + "id": "", + "logGroupNames": [ + "/aws/vendedlogs/amp" + ], + "namespace": "", + "queryMode": "Logs", + "refId": "A", + "region": "default", + "statsGroups": [] + } + ], + "timeFrom": "6h", + "timeShift": "6h", + "title": "AMP Vended Logs", + "type": "logs" + } + ], + "refresh": "", + "schemaVersion": 35, + "style": "dark", + "tags": [], + "templating": { + "list": [ + { + "current": { + "selected": true, + "text": [ + "ws-e8b003eb-0528-4208-b31c-edf4598d5f66" + ], + "value": [ + "ws-e8b003eb-0528-4208-b31c-edf4598d5f66" + ] + }, + "datasource": { + "type": "cloudwatch", + "uid": "$datasource" + }, + "definition": "dimension_values(default,AWS/Prometheus,RuleEvaluations,Workspace)", + "hide": 0, + "includeAll": false, + "multi": true, + "name": "WorkspaceID", + "options": [], + "query": "dimension_values(default,AWS/Prometheus,RuleEvaluations,Workspace)", + "refresh": 1, + "regex": "", + "skipUrlSync": false, + "sort": 0, + "type": "query" + }, + { + "current": { + "selected": false, + "text": "Amazon CloudWatch us-west-2", + "value": "Amazon CloudWatch us-west-2" + }, + "hide": 0, + "includeAll": false, + "multi": false, + "name": "datasource", + "options": [], + "query": "cloudwatch", + "refresh": 1, + "regex": "", + "skipUrlSync": false, + "type": "datasource" + } + ] + }, + "time": { + "from": "now-6h", + "to": "now" + }, + "timepicker": {}, + "timezone": "", + "title": "AMP Accelerator Dashboard", + "uid": "", + "version": 1, + "weekStart": "" +} diff --git a/modules/workloads/managed-prometheus-monitoring/locals.tf b/modules/workloads/managed-prometheus-monitoring/locals.tf new file mode 100644 index 0000000..e69de29 diff --git a/modules/workloads/managed-prometheus-monitoring/main.tf b/modules/workloads/managed-prometheus-monitoring/main.tf new file mode 100644 index 0000000..937d8a6 --- /dev/null +++ b/modules/workloads/managed-prometheus-monitoring/main.tf @@ -0,0 +1,33 @@ +provider "aws" { + region = "us-east-1" + alias = "billing_region" +} + +locals { + name = "aws-observability-accelerator-cloudwatch" + amp_list = toset(split(",", var.managed_prometheus_workspace_ids)) +} + +resource "grafana_data_source" "cloudwatch" { + type = "cloudwatch" + name = local.name + is_default = true + json_data { + default_region = var.aws_region + sigv4_auth = true + sigv4_auth_type = "workspace-iam-role" + sigv4_region = var.aws_region + } +} + +resource "grafana_dashboard" "this" { + folder = var.dashboards_folder_id + config_json = file("${path.module}/dashboards/amp-dashboard.json") +} + +module "billing" { + source = "../../workloads/managed-prometheus-monitoring/billing" + providers = { + aws = aws.billing_region + } +} diff --git a/modules/workloads/managed-prometheus-monitoring/outputs.tf b/modules/workloads/managed-prometheus-monitoring/outputs.tf new file mode 100644 index 0000000..f031a73 --- /dev/null +++ b/modules/workloads/managed-prometheus-monitoring/outputs.tf @@ -0,0 +1,4 @@ +output "grafana_dashboard_urls" { + value = [grafana_dashboard.this.url] + description = "URLs for dashboards created" +} diff --git a/modules/workloads/managed-prometheus-monitoring/variables.tf b/modules/workloads/managed-prometheus-monitoring/variables.tf new file mode 100644 index 0000000..50a376d --- /dev/null +++ b/modules/workloads/managed-prometheus-monitoring/variables.tf @@ -0,0 +1,26 @@ +variable "dashboards_folder_id" { + description = "Grafana folder ID for automatic dashboards" + type = string +} + +variable "aws_region" { + description = "AWS Region" + type = string +} + +variable "managed_prometheus_workspace_ids" { + description = "Amazon Managed Service for Prometheus Workspace ID to create Alarms for" + type = string +} + +variable "active_series_threshold" { + description = "Threshold for active series metric alarm" + type = number + default = 1000000 +} + +variable "ingestion_rate_threshold" { + description = "Threshold for active series metric alarm" + type = number + default = 70000 +} diff --git a/modules/workloads/managed-prometheus-monitoring/versions.tf b/modules/workloads/managed-prometheus-monitoring/versions.tf new file mode 100644 index 0000000..ac93e75 --- /dev/null +++ b/modules/workloads/managed-prometheus-monitoring/versions.tf @@ -0,0 +1,14 @@ +terraform { + required_version = ">= 1.1.0" + + required_providers { + aws = { + source = "hashicorp/aws" + version = ">= 4.0.0" + } + grafana = { + source = "grafana/grafana" + version = ">= 1.25.0" + } + } +}