From 9f0b90ab2ffbcda094736cb58263d2a0130b4f4d Mon Sep 17 00:00:00 2001 From: Kevin Lewin <97046295+lewinkedrs@users.noreply.github.com> Date: Tue, 27 Sep 2022 10:00:55 -0400 Subject: [PATCH] Nginx module source (#35) * working module * metrics work * Add dynamic targets * dashboards and rules * readme and outputs * updating readme * Update scrape config - Drop unused go metrics - Drop empty labels - Add host, container and namespace labels * Update tags and query labels Co-authored-by: EC2 Default User Co-authored-by: Rodrigue Koffi --- examples/existing-cluster-nginx/README.md | 241 +++++ examples/existing-cluster-nginx/main.tf | 105 +++ examples/existing-cluster-nginx/outputs.tf | 24 + examples/existing-cluster-nginx/variables.tf | 39 + .../main.tf | 3 +- modules/workloads/nginx/dashboards.tf | 5 + .../dashboards/{default.json => nginx.json} | 868 +++++++++--------- modules/workloads/nginx/locals.tf | 31 + modules/workloads/nginx/main.tf | 13 +- .../templates/opentelemetrycollector.yaml | 79 +- modules/workloads/nginx/rules.tf | 44 + modules/workloads/nginx/variables.tf | 128 ++- modules/workloads/nginx/versions.tf | 4 + 13 files changed, 1076 insertions(+), 508 deletions(-) create mode 100644 examples/existing-cluster-nginx/README.md create mode 100644 examples/existing-cluster-nginx/main.tf create mode 100644 examples/existing-cluster-nginx/outputs.tf create mode 100644 examples/existing-cluster-nginx/variables.tf create mode 100644 modules/workloads/nginx/dashboards.tf rename modules/workloads/nginx/dashboards/{default.json => nginx.json} (70%) create mode 100644 modules/workloads/nginx/locals.tf create mode 100644 modules/workloads/nginx/rules.tf diff --git a/examples/existing-cluster-nginx/README.md b/examples/existing-cluster-nginx/README.md new file mode 100644 index 0000000..bad5e9c --- /dev/null +++ b/examples/existing-cluster-nginx/README.md @@ -0,0 +1,241 @@ +# Existing Cluster with the AWS Observability accelerator base module and Nginx monitoring + + +This example demonstrates how to use the AWS Observability Accelerator Terraform +modules with Nginx monitoring enabled. +The current example deploys the [AWS Distro for OpenTelemetry Operator](https://docs.aws.amazon.com/eks/latest/userguide/opentelemetry.html) for Amazon EKS with its requirements and make use of existing +Amazon Managed Service for Prometheus and Amazon Managed Grafana workspaces. + +It is based on the `nginx module`, one of our [workload modules](../../modules/workloads/) +to provide an existing EKS cluster with an OpenTelemetry collector, +curated Grafana dashboards, Prometheus alerting and recording rules with multiple +configuration options on the cluster infrastructure. + + +## Prerequisites + +Ensure that you have the following tools installed locally: + +1. [aws cli](https://docs.aws.amazon.com/cli/latest/userguide/getting-started-install.html) +2. [kubectl](https://kubernetes.io/docs/tasks/tools/) +3. [terraform](https://learn.hashicorp.com/tutorials/terraform/install-cli) + + +## Setup + +This example uses a local terraform state. If you need states to be saved remotely, +on Amazon S3 for example, visit the [terraform remote states](https://www.terraform.io/language/state/remote) documentation + +1. Clone the repo using the command below + +``` +git clone https://github.com/aws-observability/terraform-aws-observability-accelerator.git +``` + +2. Initialize terraform + +```console +cd examples/existing-cluster-nginx +terraform init +``` + +3. AWS Region + +Specify the AWS Region where the resources will be deployed. Edit the `terraform.tfvars` file and modify `aws_region="..."`. You can also use environement variables `export TF_VAR_aws_region=xxx`. + +4. Amazon EKS Cluster + +To run this example, you need to provide your EKS cluster name. +If you don't have a cluster ready, visit [this example](https://github.com/aws-ia/terraform-aws-eks-blueprints/tree/main/examples/eks-cluster-with-new-vpc) +first to create a new one. + +Add your cluster name for `eks_cluster_id="..."` to the `terraform.tfvars` or use an environment variable `export TF_VAR_eks_cluster_id=xxx`. + +5. Amazon Managed Service for Prometheus workspace (optional) + +If you have an existing workspace, add `managed_prometheus_workspace_id=ws-xxx` +or use an environment variable `export TF_VAR_managed_prometheus_workspace_id=ws-xxx`. + +If you don't specify anything a new workspace will be created for you. + +6. Amazon Managed Grafana workspace + +If you have an existing workspace, add `managed_grafana_workspace_id=g-xxx` +or use an environment variable `export TF_VAR_managed_grafana_workspace_id=g-xxx`. + +7. Grafana API Key + +- Give admin access to the SSO user you set up when creating the Amazon Managed Grafana Workspace: +- In the AWS Console, navigate to Amazon Grafana. In the left navigation bar, click **All workspaces**, then click on the workspace name you are using for this example. +- Under **Authentication** within **AWS Single Sign-On (SSO)**, click **Configure users and user groups** +- Check the box next to the SSO user you created and click **Make admin** +- From the workspace in the AWS console, click on the `Grafana workspace URL` to open the workspace +- If you don't see the gear icon in the left navigation bar, log out and log back in. +- Click on the gear icon, then click on the **API keys** tab. +- Click **Add API key**, fill in the _Key name_ field and select _Admin_ as the Role. +- Copy your API key into `terraform.tfvars` under the `grafana_api_key` variable (`grafana_api_key="xxx"`) or set as an environment variable on your CLI (`export TF_VAR_grafana_api_key="xxx"`) + + +## Deploy + +```sh +terraform apply -var-file=terraform.tfvars +``` + +or if you had setup environment variables, run + +```sh +terraform apply +``` + +## Visualization + +1. Prometheus datasource on Grafana + +Open your Grafana workspace and under Configuration -> Data sources, you should see `aws-observability-accelerator`. Open and click `Save & test`. You should see a notification confirming that the Amazon Managed Service for Prometheus workspace is ready to be used on Grafana. + +2. Grafana dashboards + +Go to the Dashboards panel of your Grafana workspace. You should see a list of dashboards under the `Observability Accelerator Dashboards` + +image + +Open the NGINX dashboard and you should be able to view its visualization + +image + +2. Amazon Managed Service for Prometheus rules and alerts + +Open the Amazon Managed Service for Prometheus console and view the details of your workspace. Under the `Rules management` tab, you should find new rules deployed. + +image + + +To setup your alert receiver, with Amazon SNS, follow [this documentation](https://docs.aws.amazon.com/prometheus/latest/userguide/AMP-alertmanager-receiver.html) + +## Deploy an Example Application to Visualize + +In this section we will deploy sample application and extract metrics using AWS OpenTelemetry collector + +1. Add the helm incubator repo: + +```sh +helm repo add ingress-nginx https://kubernetes.github.io/ingress-nginx +``` + +2. Enter the following command to create a new namespace: + +```sh +kubectl create namespace nginx-ingress-sample +``` + +3. Enter the following commands to install NGINX: + +```sh +helm install my-nginx ingress-nginx/ingress-nginx \ +--namespace nginx-ingress-sample \ +--set controller.metrics.enabled=true \ +--set-string controller.metrics.service.annotations."prometheus\.io/port"="10254" \ +--set-string controller.metrics.service.annotations."prometheus\.io/scrape"="true" +``` + +4. Set an EXTERNAL-IP variable to the value of the EXTERNAL-IP column in the row of the NGINX ingress controller. + +```sh +EXTERNAL_IP=your-nginx-controller-external-ip +``` + +5. Start some sample NGINX traffic by entering the following command. + +```sh +SAMPLE_TRAFFIC_NAMESPACE=nginx-sample-traffic +curl https://raw.githubusercontent.com/aws-samples/amazon-cloudwatch-container-insights/master/k8s-deployment-manifest-templates/deployment-mode/service/cwagent-prometheus/sample_traffic/nginx-traffic/nginx-traffic-sample.yaml | +sed "s/{{external_ip}}/$EXTERNAL_IP/g" | +sed "s/{{namespace}}/$SAMPLE_TRAFFIC_NAMESPACE/g" | +kubectl apply -f - +``` + +4. Verify if the application is running + +```sh +kubectl get pods -n nginx-ingress-sample +``` + +#### Visualize the Application's dashboard + +Log back into your Managed Grafana Workspace and navigate to the dashboard side panel, click on `Observability Accelerator Dashboards` Folder and open the `NGINX` Dashboard. + +## Destroy + +To teardown and remove the resources created in this example: + +```sh +terraform destroy +``` + +## Advanced configuration + +1. Cross-region Amazon Managed Prometheus workspace + +If your existing Amazon Managed Prometheus workspace is in another AWS Region, +add this `managed_prometheus_region=xxx` and `managed_prometheus_workspace_id=ws-xxx`. + +2. Cross-region Amazon Managed Grafana workspace + +If your existing Amazon Managed Prometheus workspace is in another AWS Region, +add this `managed_prometheus_region=xxx` and `managed_prometheus_workspace_id=ws-xxx`. + + + +## Requirements + +| Name | Version | +|------|---------| +| [terraform](#requirement\_terraform) | >= 1.0.0 | +| [aws](#requirement\_aws) | >= 4.0.0 | +| [grafana](#requirement\_grafana) | >= 1.25.0 | +| [grafana](#requirement\_grafana) | >= 1.25.0 | +| [helm](#requirement\_helm) | >= 2.4.1 | +| [kubectl](#requirement\_kubectl) | >= 1.14 | +| [kubernetes](#requirement\_kubernetes) | >= 2.10 | + +## Providers + +| Name | Version | +|------|---------| +| [aws](#provider\_aws) | >= 4.0.0 | + +## Modules + +| Name | Source | Version | +|------|--------|---------| +| [eks\_observability\_accelerator](#module\_eks\_observability\_accelerator) | ../../ | n/a | +| [workloads\_nginx](#module\_workloads\_nginx) | ../../modules/workloads/nginx | n/a | + +## Resources + +| Name | Type | +|------|------| +| [aws_eks_cluster.this](https://registry.terraform.io/providers/hashicorp/aws/latest/docs/data-sources/eks_cluster) | data source | +| [aws_eks_cluster_auth.this](https://registry.terraform.io/providers/hashicorp/aws/latest/docs/data-sources/eks_cluster_auth) | data source | + +## Inputs + +| Name | Description | Type | Default | Required | +|------|-------------|------|---------|:--------:| +| [aws\_region](#input\_aws\_region) | AWS Region | `string` | n/a | yes | +| [eks\_cluster\_id](#input\_eks\_cluster\_id) | Name of the EKS cluster | `string` | n/a | yes | +| [grafana\_api\_key](#input\_grafana\_api\_key) | API key for authorizing the Grafana provider to make changes to Amazon Managed Grafana | `string` | `""` | no | +| [managed\_grafana\_workspace\_id](#input\_managed\_grafana\_workspace\_id) | Amazon Managed Grafana Workspace ID | `string` | `""` | no | +| [managed\_prometheus\_workspace\_id](#input\_managed\_prometheus\_workspace\_id) | Amazon Managed Service for Prometheus Workspace ID | `string` | `""` | no | + +## Outputs + +| Name | Description | +|------|-------------| +| [aws\_region](#output\_aws\_region) | AWS Region | +| [eks\_cluster\_id](#output\_eks\_cluster\_id) | EKS Cluster Id | +| [eks\_cluster\_version](#output\_eks\_cluster\_version) | EKS Cluster version | +| [managed\_prometheus\_workspace\_endpoint](#output\_managed\_prometheus\_workspace\_endpoint) | Amazon Managed Prometheus workspace endpoint | +| [managed\_prometheus\_workspace\_id](#output\_managed\_prometheus\_workspace\_id) | Amazon Managed Prometheus workspace ID | + diff --git a/examples/existing-cluster-nginx/main.tf b/examples/existing-cluster-nginx/main.tf new file mode 100644 index 0000000..1bad0fb --- /dev/null +++ b/examples/existing-cluster-nginx/main.tf @@ -0,0 +1,105 @@ +provider "aws" { + region = local.region +} + +data "aws_eks_cluster_auth" "this" { + name = var.eks_cluster_id +} + +data "aws_eks_cluster" "this" { + name = var.eks_cluster_id +} + +provider "kubernetes" { + host = local.eks_cluster_endpoint + cluster_ca_certificate = base64decode(data.aws_eks_cluster.this.certificate_authority[0].data) + token = data.aws_eks_cluster_auth.this.token +} + +provider "helm" { + kubernetes { + host = local.eks_cluster_endpoint + cluster_ca_certificate = base64decode(data.aws_eks_cluster.this.certificate_authority[0].data) + token = data.aws_eks_cluster_auth.this.token + } +} + +terraform { + required_providers { + grafana = { + source = "grafana/grafana" + version = ">= 1.25.0" + } + } +} + +locals { + name = basename(path.cwd) + region = var.aws_region + + eks_oidc_issuer_url = replace(data.aws_eks_cluster.this.identity[0].oidc[0].issuer, "https://", "") + eks_cluster_endpoint = data.aws_eks_cluster.this.endpoint + eks_cluster_version = data.aws_eks_cluster.this.version + + create_new_workspace = var.managed_prometheus_workspace_id == "" ? true : false + + tags = { + Source = "github.com/aws-observability/terraform-aws-observability-accelerator" + } +} + +module "eks_observability_accelerator" { + # source = "aws-observability/terrarom-aws-observability-accelerator" + source = "../../" + + aws_region = var.aws_region + eks_cluster_id = var.eks_cluster_id + + # deploys AWS Distro for OpenTelemetry operator into the cluster + enable_amazon_eks_adot = true + + # reusing existing certificate manager? defaults to true + enable_cert_manager = true + + # creates a new AMP workspace, defaults to true + enable_managed_prometheus = local.create_new_workspace + + # reusing existing AMP if specified + managed_prometheus_workspace_id = var.managed_prometheus_workspace_id + managed_prometheus_workspace_region = null # defaults to the current region, useful for cross region scenarios (same account) + + # sets up the AMP alert manager at the workspace level + enable_alertmanager = true + + # reusing existing Amazon Managed Grafana workspace + enable_managed_grafana = false + managed_grafana_workspace_id = var.managed_grafana_workspace_id + grafana_api_key = var.grafana_api_key + + tags = local.tags +} + +provider "grafana" { + url = module.eks_observability_accelerator.managed_grafana_workspace_endpoint + auth = var.grafana_api_key +} + +//* +module "workloads_nginx" { + source = "../../modules/workloads/nginx" + + eks_cluster_id = module.eks_observability_accelerator.eks_cluster_id + + dashboards_folder_id = module.eks_observability_accelerator.grafana_dashboards_folder_id + managed_prometheus_workspace_id = module.eks_observability_accelerator.managed_prometheus_workspace_id + + managed_prometheus_workspace_endpoint = module.eks_observability_accelerator.managed_prometheus_workspace_endpoint + managed_prometheus_workspace_region = module.eks_observability_accelerator.managed_prometheus_workspace_region + + tags = local.tags + + depends_on = [ + module.eks_observability_accelerator + ] +} +//*/ diff --git a/examples/existing-cluster-nginx/outputs.tf b/examples/existing-cluster-nginx/outputs.tf new file mode 100644 index 0000000..a0aa57e --- /dev/null +++ b/examples/existing-cluster-nginx/outputs.tf @@ -0,0 +1,24 @@ +output "eks_cluster_id" { + description = "EKS Cluster Id" + value = module.eks_observability_accelerator.eks_cluster_id +} + +output "aws_region" { + description = "AWS Region" + value = module.eks_observability_accelerator.aws_region +} + +output "eks_cluster_version" { + description = "EKS Cluster version" + value = module.eks_observability_accelerator.eks_cluster_version +} + +output "managed_prometheus_workspace_endpoint" { + description = "Amazon Managed Prometheus workspace endpoint" + value = module.eks_observability_accelerator.managed_prometheus_workspace_endpoint +} + +output "managed_prometheus_workspace_id" { + description = "Amazon Managed Prometheus workspace ID" + value = module.eks_observability_accelerator.managed_prometheus_workspace_id +} \ No newline at end of file diff --git a/examples/existing-cluster-nginx/variables.tf b/examples/existing-cluster-nginx/variables.tf new file mode 100644 index 0000000..d0b11b2 --- /dev/null +++ b/examples/existing-cluster-nginx/variables.tf @@ -0,0 +1,39 @@ +variable "eks_cluster_id" { + description = "EKS Cluster Id" + type = string +} +variable "aws_region" { + description = "AWS Region" + type = string +} +variable "managed_prometheus_workspace_id" { + description = "Amazon Managed Service for Prometheus (AMP) workspace ID" + type = string + default = "" +} +variable "managed_prometheus_endpoint" { + description = "AMP workspace ID" + type = string + default = "" +} +variable "managed_prometheus_region" { + description = "Region where AMP is deployed" + type = string + default = "" +} +variable "managed_grafana_workspace_id" { + description = "Amazon Managed Grafana (AMG) workspace ID" + type = string + default = "" +} +variable "grafana_endpoint" { + description = "AMG endpoint" + type = string + default = null +} +variable "grafana_api_key" { + description = "API key for authorizing the Grafana provider to make changes to Amazon Managed Grafana" + type = string + default = "" + sensitive = true +} \ No newline at end of file diff --git a/examples/existing-cluster-with-base-and-infra/main.tf b/examples/existing-cluster-with-base-and-infra/main.tf index 08b0350..ba55e3d 100644 --- a/examples/existing-cluster-with-base-and-infra/main.tf +++ b/examples/existing-cluster-with-base-and-infra/main.tf @@ -76,7 +76,6 @@ provider "grafana" { module "workloads_infra" { source = "../../modules/workloads/infra" - # source = "aws-observability/terrarom-aws-observability-accelerator/workloads/infra" eks_cluster_id = module.eks_observability_accelerator.eks_cluster_id @@ -87,7 +86,7 @@ module "workloads_infra" { managed_prometheus_workspace_region = module.eks_observability_accelerator.managed_prometheus_workspace_region tags = local.tags - + depends_on = [ module.eks_observability_accelerator ] diff --git a/modules/workloads/nginx/dashboards.tf b/modules/workloads/nginx/dashboards.tf new file mode 100644 index 0000000..b923ef8 --- /dev/null +++ b/modules/workloads/nginx/dashboards.tf @@ -0,0 +1,5 @@ +resource "grafana_dashboard" "workloads" { + count = var.enable_dashboards ? 1 : 0 + folder = var.dashboards_folder_id + config_json = file("${path.module}/dashboards/nginx.json") +} \ No newline at end of file diff --git a/modules/workloads/nginx/dashboards/default.json b/modules/workloads/nginx/dashboards/nginx.json similarity index 70% rename from modules/workloads/nginx/dashboards/default.json rename to modules/workloads/nginx/dashboards/nginx.json index 98ecd4f..9cda933 100644 --- a/modules/workloads/nginx/dashboards/default.json +++ b/modules/workloads/nginx/dashboards/nginx.json @@ -1,34 +1,4 @@ { - "__inputs": [ - { - "name": "DS_PROMETHEUS", - "label": "Prometheus", - "description": "", - "type": "datasource", - "pluginId": "prometheus", - "pluginName": "Prometheus" - } - ], - "__requires": [ - { - "type": "grafana", - "id": "grafana", - "name": "Grafana", - "version": "5.2.1" - }, - { - "type": "datasource", - "id": "prometheus", - "name": "Prometheus", - "version": "5.0.0" - }, - { - "type": "panel", - "id": "singlestat", - "name": "Singlestat", - "version": "5.0.0" - } - ], "annotations": { "list": [ { @@ -38,10 +8,16 @@ "hide": true, "iconColor": "rgba(0, 211, 255, 1)", "name": "Annotations & Alerts", + "target": { + "limit": 100, + "matchAny": false, + "tags": [], + "type": "dashboard" + }, "type": "dashboard" }, { - "datasource": "${DS_PROMETHEUS}", + "datasource": "$datasource", "enable": true, "expr": "sum(changes(nginx_ingress_controller_config_last_reload_successful_timestamp_seconds{instance!=\"unknown\",controller_class=~\"$controller_class\",namespace=~\"$namespace\"}[30s])) by (controller_class)", "hide": false, @@ -57,71 +33,73 @@ } ] }, + "description": "Ingress-nginx supports a rich collection of prometheus metrics. If you have prometheus and grafana installed on your cluster then prometheus will already be scraping this data due to the scrape annotation on the deployment.", "editable": true, + "fiscalYearStartMonth": 0, "gnetId": 9614, "graphTooltip": 0, - "iteration": 1534359654832, + "id": 8, + "iteration": 1663262425658, "links": [], + "liveNow": false, "panels": [ { - "cacheTimeout": null, - "colorBackground": false, - "colorValue": false, - "colors": [ - "rgba(245, 54, 54, 0.9)", - "rgba(237, 129, 40, 0.89)", - "rgba(50, 172, 45, 0.97)" - ], - "datasource": "${DS_PROMETHEUS}", - "format": "ops", - "gauge": { - "maxValue": 100, - "minValue": 0, - "show": false, - "thresholdLabels": false, - "thresholdMarkers": true + "datasource": { + "type": "prometheus", + "uid": "$datasource" }, - "gridPos": { - "h": 3, - "w": 6, - "x": 0, - "y": 0 + "fieldConfig": { + "defaults": { + "color": { + "fixedColor": "rgb(31, 120, 193)", + "mode": "fixed" + }, + "mappings": [ + { + "options": { + "match": "null", + "result": { + "text": "N/A" + } + }, + "type": "special" + } + ], + "thresholds": { + "mode": "absolute", + "steps": [ + { + "color": "green", + "value": null + }, + { + "color": "red", + "value": 80 + } + ] + }, + "unit": "ops" + }, + "overrides": [] }, "id": 20, - "interval": null, "links": [], - "mappingType": 1, - "mappingTypes": [ - { - "name": "value to text", - "value": 1 - }, - { - "name": "range to text", - "value": 2 - } - ], "maxDataPoints": 100, - "nullPointMode": "connected", - "nullText": null, - "postfix": "", - "postfixFontSize": "50%", - "prefix": "", - "prefixFontSize": "50%", - "rangeMaps": [ - { - "from": "null", - "text": "N/A", - "to": "null" - } - ], - "sparkline": { - "fillColor": "rgba(31, 118, 189, 0.18)", - "full": true, - "lineColor": "rgb(31, 120, 193)", - "show": true + "options": { + "colorMode": "none", + "graphMode": "area", + "justifyMode": "auto", + "orientation": "horizontal", + "reduceOptions": { + "calcs": [ + "mean" + ], + "fields": "", + "values": false + }, + "textMode": "auto" }, - "tableColumn": "", + "pluginVersion": "8.4.7", "targets": [ { "expr": "round(sum(irate(nginx_ingress_controller_requests{controller_pod=~\"$controller\",controller_class=~\"$controller_class\",namespace=~\"$namespace\"}[2m])), 0.001)", @@ -131,37 +109,47 @@ "step": 4 } ], - "thresholds": "", "title": "Controller Request Volume", - "transparent": false, - "type": "singlestat", - "valueFontSize": "80%", - "valueMaps": [ - { - "op": "=", - "text": "N/A", - "value": "null" - } - ], - "valueName": "avg" + "type": "stat" }, { - "cacheTimeout": null, - "colorBackground": false, - "colorValue": false, - "colors": [ - "rgba(245, 54, 54, 0.9)", - "rgba(237, 129, 40, 0.89)", - "rgba(50, 172, 45, 0.97)" - ], - "datasource": "${DS_PROMETHEUS}", - "format": "none", - "gauge": { - "maxValue": 100, - "minValue": 0, - "show": false, - "thresholdLabels": false, - "thresholdMarkers": true + "datasource": { + "type": "prometheus", + "uid": "$datasource" + }, + "fieldConfig": { + "defaults": { + "color": { + "fixedColor": "rgb(31, 120, 193)", + "mode": "fixed" + }, + "mappings": [ + { + "options": { + "match": "null", + "result": { + "text": "N/A" + } + }, + "type": "special" + } + ], + "thresholds": { + "mode": "absolute", + "steps": [ + { + "color": "green", + "value": null + }, + { + "color": "red", + "value": 80 + } + ] + }, + "unit": "none" + }, + "overrides": [] }, "gridPos": { "h": 3, @@ -170,40 +158,23 @@ "y": 0 }, "id": 82, - "interval": null, "links": [], - "mappingType": 1, - "mappingTypes": [ - { - "name": "value to text", - "value": 1 - }, - { - "name": "range to text", - "value": 2 - } - ], "maxDataPoints": 100, - "nullPointMode": "connected", - "nullText": null, - "postfix": "", - "postfixFontSize": "50%", - "prefix": "", - "prefixFontSize": "50%", - "rangeMaps": [ - { - "from": "null", - "text": "N/A", - "to": "null" - } - ], - "sparkline": { - "fillColor": "rgba(31, 118, 189, 0.18)", - "full": true, - "lineColor": "rgb(31, 120, 193)", - "show": true + "options": { + "colorMode": "none", + "graphMode": "area", + "justifyMode": "auto", + "orientation": "horizontal", + "reduceOptions": { + "calcs": [ + "mean" + ], + "fields": "", + "values": false + }, + "textMode": "auto" }, - "tableColumn": "", + "pluginVersion": "8.4.7", "targets": [ { "expr": "sum(avg_over_time(nginx_ingress_controller_nginx_process_connections{controller_pod=~\"$controller\",controller_class=~\"$controller_class\",controller_namespace=~\"$namespace\"}[2m]))", @@ -214,37 +185,51 @@ "step": 4 } ], - "thresholds": "", "title": "Controller Connections", - "transparent": false, - "type": "singlestat", - "valueFontSize": "80%", - "valueMaps": [ - { - "op": "=", - "text": "N/A", - "value": "null" - } - ], - "valueName": "avg" + "type": "stat" }, { - "cacheTimeout": null, - "colorBackground": false, - "colorValue": false, - "colors": [ - "rgba(245, 54, 54, 0.9)", - "rgba(237, 129, 40, 0.89)", - "rgba(50, 172, 45, 0.97)" - ], - "datasource": "${DS_PROMETHEUS}", - "format": "percentunit", - "gauge": { - "maxValue": 100, - "minValue": 80, - "show": false, - "thresholdLabels": false, - "thresholdMarkers": false + "datasource": { + "type": "prometheus", + "uid": "$datasource" + }, + "fieldConfig": { + "defaults": { + "color": { + "fixedColor": "rgb(31, 120, 193)", + "mode": "fixed" + }, + "mappings": [ + { + "options": { + "match": "null", + "result": { + "text": "N/A" + } + }, + "type": "special" + } + ], + "thresholds": { + "mode": "absolute", + "steps": [ + { + "color": "rgba(245, 54, 54, 0.9)", + "value": null + }, + { + "color": "rgba(237, 129, 40, 0.89)", + "value": 95 + }, + { + "color": "rgba(50, 172, 45, 0.97)", + "value": 99 + } + ] + }, + "unit": "percentunit" + }, + "overrides": [] }, "gridPos": { "h": 3, @@ -253,40 +238,23 @@ "y": 0 }, "id": 21, - "interval": null, "links": [], - "mappingType": 1, - "mappingTypes": [ - { - "name": "value to text", - "value": 1 - }, - { - "name": "range to text", - "value": 2 - } - ], "maxDataPoints": 100, - "nullPointMode": "connected", - "nullText": null, - "postfix": "", - "postfixFontSize": "50%", - "prefix": "", - "prefixFontSize": "50%", - "rangeMaps": [ - { - "from": "null", - "text": "N/A", - "to": "null" - } - ], - "sparkline": { - "fillColor": "rgba(31, 118, 189, 0.18)", - "full": true, - "lineColor": "rgb(31, 120, 193)", - "show": true + "options": { + "colorMode": "none", + "graphMode": "area", + "justifyMode": "auto", + "orientation": "horizontal", + "reduceOptions": { + "calcs": [ + "mean" + ], + "fields": "", + "values": false + }, + "textMode": "auto" }, - "tableColumn": "", + "pluginVersion": "8.4.7", "targets": [ { "expr": "sum(rate(nginx_ingress_controller_requests{controller_pod=~\"$controller\",controller_class=~\"$controller_class\",namespace=~\"$namespace\",status!~\"[4-5].*\"}[2m])) / sum(rate(nginx_ingress_controller_requests{controller_pod=~\"$controller\",controller_class=~\"$controller_class\",namespace=~\"$namespace\"}[2m]))", @@ -296,38 +264,48 @@ "step": 4 } ], - "thresholds": "95, 99, 99.5", "title": "Controller Success Rate (non-4|5xx responses)", - "transparent": false, - "type": "singlestat", - "valueFontSize": "80%", - "valueMaps": [ - { - "op": "=", - "text": "N/A", - "value": "null" - } - ], - "valueName": "avg" + "type": "stat" }, { - "cacheTimeout": null, - "colorBackground": false, - "colorValue": false, - "colors": [ - "rgba(245, 54, 54, 0.9)", - "rgba(237, 129, 40, 0.89)", - "rgba(50, 172, 45, 0.97)" - ], - "datasource": "${DS_PROMETHEUS}", - "decimals": 0, - "format": "none", - "gauge": { - "maxValue": 100, - "minValue": 0, - "show": false, - "thresholdLabels": false, - "thresholdMarkers": true + "datasource": { + "type": "prometheus", + "uid": "$datasource" + }, + "fieldConfig": { + "defaults": { + "color": { + "fixedColor": "rgb(31, 120, 193)", + "mode": "fixed" + }, + "decimals": 0, + "mappings": [ + { + "options": { + "match": "null", + "result": { + "text": "N/A" + } + }, + "type": "special" + } + ], + "thresholds": { + "mode": "absolute", + "steps": [ + { + "color": "green", + "value": null + }, + { + "color": "red", + "value": 80 + } + ] + }, + "unit": "none" + }, + "overrides": [] }, "gridPos": { "h": 3, @@ -336,40 +314,23 @@ "y": 0 }, "id": 81, - "interval": null, "links": [], - "mappingType": 1, - "mappingTypes": [ - { - "name": "value to text", - "value": 1 - }, - { - "name": "range to text", - "value": 2 - } - ], "maxDataPoints": 100, - "nullPointMode": "connected", - "nullText": null, - "postfix": "", - "postfixFontSize": "50%", - "prefix": "", - "prefixFontSize": "50%", - "rangeMaps": [ - { - "from": "null", - "text": "N/A", - "to": "null" - } - ], - "sparkline": { - "fillColor": "rgba(31, 118, 189, 0.18)", - "full": true, - "lineColor": "rgb(31, 120, 193)", - "show": true + "options": { + "colorMode": "none", + "graphMode": "area", + "justifyMode": "auto", + "orientation": "horizontal", + "reduceOptions": { + "calcs": [ + "mean" + ], + "fields": "", + "values": false + }, + "textMode": "auto" }, - "tableColumn": "", + "pluginVersion": "8.4.7", "targets": [ { "expr": "avg(nginx_ingress_controller_success{controller_pod=~\"$controller\",controller_class=~\"$controller_class\",controller_namespace=~\"$namespace\"})", @@ -380,38 +341,48 @@ "step": 4 } ], - "thresholds": "", "title": "Config Reloads", - "transparent": false, - "type": "singlestat", - "valueFontSize": "80%", - "valueMaps": [ - { - "op": "=", - "text": "N/A", - "value": "null" - } - ], - "valueName": "avg" + "type": "stat" }, { - "cacheTimeout": null, - "colorBackground": false, - "colorValue": false, - "colors": [ - "rgba(245, 54, 54, 0.9)", - "rgba(237, 129, 40, 0.89)", - "rgba(50, 172, 45, 0.97)" - ], - "datasource": "${DS_PROMETHEUS}", - "decimals": 0, - "format": "none", - "gauge": { - "maxValue": 100, - "minValue": 0, - "show": false, - "thresholdLabels": false, - "thresholdMarkers": true + "datasource": { + "type": "prometheus", + "uid": "$datasource" + }, + "fieldConfig": { + "defaults": { + "color": { + "fixedColor": "rgb(31, 120, 193)", + "mode": "fixed" + }, + "decimals": 0, + "mappings": [ + { + "options": { + "match": "null", + "result": { + "text": "N/A" + } + }, + "type": "special" + } + ], + "thresholds": { + "mode": "absolute", + "steps": [ + { + "color": "green", + "value": null + }, + { + "color": "red", + "value": 80 + } + ] + }, + "unit": "none" + }, + "overrides": [] }, "gridPos": { "h": 3, @@ -420,40 +391,23 @@ "y": 0 }, "id": 83, - "interval": null, "links": [], - "mappingType": 1, - "mappingTypes": [ - { - "name": "value to text", - "value": 1 - }, - { - "name": "range to text", - "value": 2 - } - ], "maxDataPoints": 100, - "nullPointMode": "connected", - "nullText": null, - "postfix": "", - "postfixFontSize": "50%", - "prefix": "", - "prefixFontSize": "50%", - "rangeMaps": [ - { - "from": "null", - "text": "N/A", - "to": "null" - } - ], - "sparkline": { - "fillColor": "rgba(31, 118, 189, 0.18)", - "full": true, - "lineColor": "rgb(31, 120, 193)", - "show": true + "options": { + "colorMode": "none", + "graphMode": "area", + "justifyMode": "auto", + "orientation": "horizontal", + "reduceOptions": { + "calcs": [ + "mean" + ], + "fields": "", + "values": false + }, + "textMode": "auto" }, - "tableColumn": "", + "pluginVersion": "8.4.7", "targets": [ { "expr": "count(nginx_ingress_controller_config_last_reload_successful{controller_pod=~\"$controller\",controller_namespace=~\"$namespace\"} == 0)", @@ -464,30 +418,23 @@ "step": 4 } ], - "thresholds": "", "title": "Last Config Failed", - "transparent": false, - "type": "singlestat", - "valueFontSize": "80%", - "valueMaps": [ - { - "op": "=", - "text": "N/A", - "value": "null" - } - ], - "valueName": "avg" + "type": "stat" }, { "aliasColors": {}, "bars": false, "dashLength": 10, "dashes": false, - "datasource": "${DS_PROMETHEUS}", + "datasource": { + "type": "prometheus", + "uid": "$datasource" + }, "decimals": 2, "editable": true, "error": false, "fill": 1, + "fillGradient": 0, "grid": {}, "gridPos": { "h": 7, @@ -496,6 +443,7 @@ "y": 3 }, "height": "200px", + "hiddenSeries": false, "id": 86, "isNew": true, "legend": { @@ -518,11 +466,14 @@ "linewidth": 2, "links": [], "nullPointMode": "connected", + "options": { + "alertThreshold": true + }, "percentage": false, + "pluginVersion": "8.4.7", "pointradius": 5, "points": false, "renderer": "flot", - "repeat": null, "repeatDirection": "h", "seriesOverrides": [], "spaceLength": 10, @@ -543,8 +494,7 @@ } ], "thresholds": [], - "timeFrom": null, - "timeShift": null, + "timeRegions": [], "title": "Ingress Request Volume", "tooltip": { "msResolution": false, @@ -552,36 +502,26 @@ "sort": 2, "value_type": "cumulative" }, - "transparent": false, "type": "graph", "xaxis": { - "buckets": null, "mode": "time", - "name": null, "show": true, "values": [] }, "yaxes": [ { "format": "reqps", - "label": null, "logBase": 1, - "max": null, - "min": null, "show": true }, { "format": "Bps", - "label": null, "logBase": 1, - "max": null, - "min": null, "show": false } ], "yaxis": { - "align": false, - "alignLevel": null + "align": false } }, { @@ -593,11 +533,15 @@ "bars": false, "dashLength": 10, "dashes": false, - "datasource": "${DS_PROMETHEUS}", + "datasource": { + "type": "prometheus", + "uid": "$datasource" + }, "decimals": 2, "editable": false, "error": false, "fill": 0, + "fillGradient": 0, "grid": {}, "gridPos": { "h": 7, @@ -605,6 +549,7 @@ "x": 12, "y": 3 }, + "hiddenSeries": false, "id": 87, "isNew": true, "legend": { @@ -627,7 +572,11 @@ "linewidth": 2, "links": [], "nullPointMode": "connected", + "options": { + "alertThreshold": true + }, "percentage": false, + "pluginVersion": "8.4.7", "pointradius": 5, "points": false, "renderer": "flot", @@ -649,8 +598,7 @@ } ], "thresholds": [], - "timeFrom": null, - "timeShift": null, + "timeRegions": [], "title": "Ingress Success Rate (non-4|5xx responses)", "tooltip": { "msResolution": false, @@ -660,33 +608,24 @@ }, "type": "graph", "xaxis": { - "buckets": null, "mode": "time", - "name": null, "show": true, "values": [] }, "yaxes": [ { "format": "percentunit", - "label": null, "logBase": 1, - "max": null, - "min": null, "show": true }, { "format": "short", - "label": null, "logBase": 1, - "max": null, - "min": null, "show": false } ], "yaxis": { - "align": false, - "alignLevel": null + "align": false } }, { @@ -694,11 +633,15 @@ "bars": false, "dashLength": 10, "dashes": false, - "datasource": "${DS_PROMETHEUS}", + "datasource": { + "type": "prometheus", + "uid": "$datasource" + }, "decimals": 2, "editable": true, "error": false, "fill": 1, + "fillGradient": 0, "grid": {}, "gridPos": { "h": 6, @@ -707,6 +650,7 @@ "y": 10 }, "height": "200px", + "hiddenSeries": false, "id": 32, "isNew": true, "legend": { @@ -727,7 +671,11 @@ "linewidth": 2, "links": [], "nullPointMode": "connected", + "options": { + "alertThreshold": true + }, "percentage": false, + "pluginVersion": "8.4.7", "pointradius": 5, "points": false, "renderer": "flot", @@ -760,8 +708,7 @@ } ], "thresholds": [], - "timeFrom": null, - "timeShift": null, + "timeRegions": [], "title": "Network I/O pressure", "tooltip": { "msResolution": false, @@ -769,36 +716,26 @@ "sort": 0, "value_type": "cumulative" }, - "transparent": false, "type": "graph", "xaxis": { - "buckets": null, "mode": "time", - "name": null, "show": true, "values": [] }, "yaxes": [ { "format": "Bps", - "label": null, "logBase": 1, - "max": null, - "min": null, "show": true }, { "format": "Bps", - "label": null, "logBase": 1, - "max": null, - "min": null, "show": false } ], "yaxis": { - "align": false, - "alignLevel": null + "align": false } }, { @@ -810,11 +747,15 @@ "bars": false, "dashLength": 10, "dashes": false, - "datasource": "${DS_PROMETHEUS}", + "datasource": { + "type": "prometheus", + "uid": "$datasource" + }, "decimals": 2, "editable": false, "error": false, "fill": 0, + "fillGradient": 0, "grid": {}, "gridPos": { "h": 6, @@ -822,6 +763,7 @@ "x": 8, "y": 10 }, + "hiddenSeries": false, "id": 77, "isNew": true, "legend": { @@ -842,7 +784,11 @@ "linewidth": 2, "links": [], "nullPointMode": "connected", + "options": { + "alertThreshold": true + }, "percentage": false, + "pluginVersion": "8.4.7", "pointradius": 5, "points": false, "renderer": "flot", @@ -864,8 +810,7 @@ } ], "thresholds": [], - "timeFrom": null, - "timeShift": null, + "timeRegions": [], "title": "Average Memory Usage", "tooltip": { "msResolution": false, @@ -875,33 +820,24 @@ }, "type": "graph", "xaxis": { - "buckets": null, "mode": "time", - "name": null, "show": true, "values": [] }, "yaxes": [ { "format": "bytes", - "label": null, "logBase": 1, - "max": null, - "min": null, "show": true }, { "format": "short", - "label": null, "logBase": 1, - "max": null, - "min": null, "show": false } ], "yaxis": { - "align": false, - "alignLevel": null + "align": false } }, { @@ -912,11 +848,15 @@ "bars": false, "dashLength": 10, "dashes": false, - "datasource": "${DS_PROMETHEUS}", + "datasource": { + "type": "prometheus", + "uid": "$datasource" + }, "decimals": 3, "editable": false, "error": false, "fill": 0, + "fillGradient": 0, "grid": {}, "gridPos": { "h": 6, @@ -925,6 +865,7 @@ "y": 10 }, "height": "", + "hiddenSeries": false, "id": 79, "isNew": true, "legend": { @@ -935,8 +876,6 @@ "min": false, "rightSide": false, "show": false, - "sort": null, - "sortDesc": null, "total": false, "values": true }, @@ -944,7 +883,11 @@ "linewidth": 2, "links": [], "nullPointMode": "connected", + "options": { + "alertThreshold": true + }, "percentage": false, + "pluginVersion": "8.4.7", "pointradius": 5, "points": false, "renderer": "flot", @@ -972,8 +915,7 @@ "op": "gt" } ], - "timeFrom": null, - "timeShift": null, + "timeRegions": [], "title": "Average CPU Usage", "tooltip": { "msResolution": true, @@ -981,12 +923,9 @@ "sort": 2, "value_type": "cumulative" }, - "transparent": false, "type": "graph", "xaxis": { - "buckets": null, "mode": "time", - "name": null, "show": true, "values": [] }, @@ -995,27 +934,24 @@ "format": "none", "label": "cores", "logBase": 1, - "max": null, - "min": null, "show": true }, { "format": "short", - "label": null, "logBase": 1, - "max": null, - "min": null, "show": true } ], "yaxis": { - "align": false, - "alignLevel": null + "align": false } }, { "columns": [], - "datasource": "${DS_PROMETHEUS}", + "datasource": { + "type": "prometheus", + "uid": "$datasource" + }, "fontSize": "100%", "gridPos": { "h": 8, @@ -1027,7 +963,6 @@ "id": 75, "links": [], "pageSize": 7, - "repeat": null, "repeatDirection": "h", "scroll": true, "showHeader": true, @@ -1038,7 +973,7 @@ "styles": [ { "alias": "Ingress", - "colorMode": null, + "align": "auto", "colors": [ "rgba(245, 54, 54, 0.9)", "rgba(237, 129, 40, 0.89)", @@ -1055,7 +990,7 @@ }, { "alias": "Requests", - "colorMode": null, + "align": "auto", "colors": [ "rgba(245, 54, 54, 0.9)", "rgba(237, 129, 40, 0.89)", @@ -1072,7 +1007,7 @@ }, { "alias": "Errors", - "colorMode": null, + "align": "auto", "colors": [ "rgba(245, 54, 54, 0.9)", "rgba(237, 129, 40, 0.89)", @@ -1087,7 +1022,7 @@ }, { "alias": "P50 Latency", - "colorMode": null, + "align": "auto", "colors": [ "rgba(245, 54, 54, 0.9)", "rgba(237, 129, 40, 0.89)", @@ -1103,7 +1038,7 @@ }, { "alias": "P90 Latency", - "colorMode": null, + "align": "auto", "colors": [ "rgba(245, 54, 54, 0.9)", "rgba(237, 129, 40, 0.89)", @@ -1118,7 +1053,7 @@ }, { "alias": "P99 Latency", - "colorMode": null, + "align": "auto", "colors": [ "rgba(245, 54, 54, 0.9)", "rgba(237, 129, 40, 0.89)", @@ -1133,7 +1068,7 @@ }, { "alias": "IN", - "colorMode": null, + "align": "auto", "colors": [ "rgba(245, 54, 54, 0.9)", "rgba(237, 129, 40, 0.89)", @@ -1150,7 +1085,7 @@ }, { "alias": "", - "colorMode": null, + "align": "auto", "colors": [ "rgba(245, 54, 54, 0.9)", "rgba(237, 129, 40, 0.89)", @@ -1165,7 +1100,7 @@ }, { "alias": "OUT", - "colorMode": null, + "align": "auto", "colors": [ "rgba(245, 54, 54, 0.9)", "rgba(237, 129, 40, 0.89)", @@ -1227,11 +1162,9 @@ "refId": "G" } ], - "timeFrom": null, "title": "Ingress Percentile Response Times and Transfer Rates", "transform": "table", - "transparent": false, - "type": "table" + "type": "table-old" }, { "columns": [ @@ -1240,7 +1173,10 @@ "value": "current" } ], - "datasource": "${DS_PROMETHEUS}", + "datasource": { + "type": "prometheus", + "uid": "$datasource" + }, "fontSize": "100%", "gridPos": { "h": 8, @@ -1261,12 +1197,14 @@ "styles": [ { "alias": "Time", + "align": "auto", "dateFormat": "YYYY-MM-DD HH:mm:ss", "pattern": "Time", "type": "date" }, { "alias": "TTL", + "align": "auto", "colorMode": "cell", "colors": [ "rgba(245, 54, 54, 0.9)", @@ -1285,7 +1223,7 @@ }, { "alias": "", - "colorMode": null, + "align": "auto", "colors": [ "rgba(245, 54, 54, 0.9)", "rgba(237, 129, 40, 0.89)", @@ -1311,36 +1249,45 @@ ], "title": "Ingress Certificate Expiry", "transform": "timeseries_aggregations", - "type": "table" + "type": "table-old" } ], "refresh": "5s", - "schemaVersion": 16, + "schemaVersion": 35, "style": "dark", "tags": [ - "nginx" + "nginx", + "workloads" ], "templating": { "list": [ { "allValue": ".*", "current": { + "selected": false, "text": "All", "value": "$__all" }, - "datasource": "${DS_PROMETHEUS}", + "datasource": { + "type": "prometheus", + "uid": "$datasource" + }, + "definition": "", "hide": 0, "includeAll": true, "label": "Namespace", "multi": false, "name": "namespace", "options": [], - "query": "label_values(nginx_ingress_controller_config_hash, controller_namespace)", + "query": { + "query": "label_values(nginx_ingress_controller_config_hash, controller_namespace)", + "refId": "aws-observability-accelerator-namespace-Variable-Query" + }, "refresh": 1, "regex": "", + "skipUrlSync": false, "sort": 0, "tagValuesQuery": "", - "tags": [], "tagsQuery": "", "type": "query", "useTags": false @@ -1348,22 +1295,30 @@ { "allValue": ".*", "current": { + "selected": false, "text": "All", "value": "$__all" }, - "datasource": "${DS_PROMETHEUS}", + "datasource": { + "type": "prometheus", + "uid": "$datasource" + }, + "definition": "", "hide": 0, "includeAll": true, "label": "Controller Class", "multi": false, "name": "controller_class", "options": [], - "query": "label_values(nginx_ingress_controller_config_hash{namespace=~\"$namespace\"}, controller_class) ", + "query": { + "query": "label_values(nginx_ingress_controller_config_hash{namespace=~\"$namespace\"}, controller_class) ", + "refId": "aws-observability-accelerator-controller_class-Variable-Query" + }, "refresh": 1, "regex": "", + "skipUrlSync": false, "sort": 0, "tagValuesQuery": "", - "tags": [], "tagsQuery": "", "type": "query", "useTags": false @@ -1371,22 +1326,30 @@ { "allValue": ".*", "current": { + "selected": false, "text": "All", "value": "$__all" }, - "datasource": "${DS_PROMETHEUS}", + "datasource": { + "type": "prometheus", + "uid": "$datasource" + }, + "definition": "label_values(nginx_ingress_controller_config_hash{namespace=~\"$namespace\",controller_class=~\"$controller_class\"}, controller_pod) ", "hide": 0, "includeAll": true, "label": "Controller", "multi": false, "name": "controller", "options": [], - "query": "label_values(nginx_ingress_controller_config_hash{namespace=~\"$namespace\",controller_class=~\"$controller_class\"}, controller_pod) ", + "query": { + "query": "label_values(nginx_ingress_controller_config_hash{namespace=~\"$namespace\",controller_class=~\"$controller_class\"}, controller_pod) ", + "refId": "aws-observability-accelerator-controller-Variable-Query" + }, "refresh": 1, "regex": "", + "skipUrlSync": false, "sort": 0, "tagValuesQuery": "", - "tags": [], "tagsQuery": "", "type": "query", "useTags": false @@ -1394,26 +1357,51 @@ { "allValue": ".*", "current": { - "tags": [], + "selected": false, "text": "All", "value": "$__all" }, - "datasource": "${DS_PROMETHEUS}", + "datasource": { + "type": "prometheus", + "uid": "$datasource" + }, + "definition": "label_values(nginx_ingress_controller_requests{namespace=~\"$namespace\",controller_class=~\"$controller_class\",controller_pod=~\"$controller\"}, ingress) ", "hide": 0, "includeAll": true, "label": "Ingress", "multi": false, "name": "ingress", "options": [], - "query": "label_values(nginx_ingress_controller_requests{namespace=~\"$namespace\",controller_class=~\"$controller_class\",controller=~\"$controller\"}, ingress) ", + "query": { + "query": "label_values(nginx_ingress_controller_requests{namespace=~\"$namespace\",controller_class=~\"$controller_class\",controller_pod=~\"$controller\"}, ingress) ", + "refId": "StandardVariableQuery" + }, "refresh": 1, "regex": "", + "skipUrlSync": false, "sort": 2, "tagValuesQuery": "", - "tags": [], "tagsQuery": "", "type": "query", "useTags": false + }, + { + "current": { + "selected": false, + "text": "aws-observability-accelerator", + "value": "aws-observability-accelerator" + }, + "description": "datasource for prometheus", + "hide": 0, + "includeAll": false, + "multi": false, + "name": "datasource", + "options": [], + "query": "prometheus", + "refresh": 1, + "regex": "", + "skipUrlSync": false, + "type": "datasource" } ] }, @@ -1447,8 +1435,8 @@ ] }, "timezone": "browser", - "title": "NGINX Ingress controller", + "title": "NGINX", "uid": "nginx", - "version": 1, - "description": "Ingress-nginx supports a rich collection of prometheus metrics. If you have prometheus and grafana installed on your cluster then prometheus will already be scraping this data due to the scrape annotation on the deployment." -} + "version": 5, + "weekStart": "" +} \ No newline at end of file diff --git a/modules/workloads/nginx/locals.tf b/modules/workloads/nginx/locals.tf new file mode 100644 index 0000000..180156b --- /dev/null +++ b/modules/workloads/nginx/locals.tf @@ -0,0 +1,31 @@ +data "aws_partition" "current" {} + +data "aws_caller_identity" "current" {} + +data "aws_region" "current" {} + +data "aws_eks_cluster" "eks_cluster" { + name = var.eks_cluster_id +} + +locals { + name = "adot-collector-nginx" + namespace = try(var.config.helm_config.namespace, local.name) + + eks_oidc_issuer_url = replace(data.aws_eks_cluster.eks_cluster.identity[0].oidc[0].issuer, "https://", "") + eks_cluster_endpoint = data.aws_eks_cluster.eks_cluster.endpoint + + context = { + aws_caller_identity_account_id = data.aws_caller_identity.current.account_id + aws_caller_identity_arn = data.aws_caller_identity.current.arn + aws_eks_cluster_endpoint = local.eks_cluster_endpoint + aws_partition_id = data.aws_partition.current.partition + aws_region_name = data.aws_region.current.name + eks_cluster_id = var.eks_cluster_id + eks_oidc_issuer_url = local.eks_oidc_issuer_url + eks_oidc_provider_arn = "arn:${data.aws_partition.current.partition}:iam::${data.aws_caller_identity.current.account_id}:oidc-provider/${local.eks_oidc_issuer_url}" + tags = var.tags + irsa_iam_role_path = var.irsa_iam_role_path + irsa_iam_permissions_boundary = var.irsa_iam_permissions_boundary + } +} diff --git a/modules/workloads/nginx/main.tf b/modules/workloads/nginx/main.tf index 6710656..176e9ba 100644 --- a/modules/workloads/nginx/main.tf +++ b/modules/workloads/nginx/main.tf @@ -1,10 +1,3 @@ -locals { - name = "adot-collector-nginx" - namespace = try(var.helm_config.namespace, local.name) -} - -data "aws_partition" "current" {} - module "helm_addon" { source = "github.com/aws-ia/terraform-aws-eks-blueprints/modules/kubernetes-addons/helm-addon" @@ -22,11 +15,11 @@ module "helm_addon" { set_values = [ { name = "ampurl" - value = "${var.amazon_prometheus_workspace_endpoint}api/v1/remote_write" + value = "${var.managed_prometheus_workspace_endpoint}api/v1/remote_write" }, { name = "region" - value = var.amazon_prometheus_workspace_region + value = var.managed_prometheus_workspace_region }, { name = "prometheusMetricsEndpoint" @@ -58,5 +51,5 @@ module "helm_addon" { irsa_iam_policies = ["arn:${data.aws_partition.current.partition}:iam::aws:policy/AmazonPrometheusRemoteWriteAccess"] } - addon_context = var.addon_context + addon_context = local.context } diff --git a/modules/workloads/nginx/otel-config/templates/opentelemetrycollector.yaml b/modules/workloads/nginx/otel-config/templates/opentelemetrycollector.yaml index aaefbdd..1eefba0 100644 --- a/modules/workloads/nginx/otel-config/templates/opentelemetrycollector.yaml +++ b/modules/workloads/nginx/otel-config/templates/opentelemetrycollector.yaml @@ -3,7 +3,7 @@ kind: OpenTelemetryCollector metadata: name: adot spec: - image: public.ecr.aws/aws-observability/aws-otel-collector:latest + image: public.ecr.aws/aws-observability/aws-otel-collector:v0.21.1 mode: deployment serviceAccount: adot-collector-nginx config: | @@ -13,55 +13,56 @@ spec: global: scrape_interval: {{ .Values.scrapeInterval }} scrape_timeout: {{ .Values.scrapeTimeout }} - scrape_configs: - - job_name: 'kubernetes-pod-nginx' - sample_limit: {{ .Values.scrapeSampleLimit }} - metrics_path: /{{ .Values.prometheusMetricsEndpoint }} - kubernetes_sd_configs: - - role: pod - relabel_configs: - - source_labels: [ __address__ ] - action: keep - regex: '.*:9404$' - - action: labelmap - regex: __meta_kubernetes_pod_label_(.+) - - action: replace - source_labels: [ __meta_kubernetes_namespace ] - target_label: Namespace - - source_labels: [ __meta_kubernetes_pod_name ] - action: replace - target_label: pod_name - - action: replace - source_labels: [ __meta_kubernetes_pod_container_name ] - target_label: container_name - - action: replace - source_labels: [ __meta_kubernetes_pod_controller_kind ] - target_label: pod_controller_kind - - action: replace - source_labels: [ __meta_kubernetes_pod_phase ] - target_label: pod_controller_phase - metric_relabel_configs: - - source_labels: [ __name__ ] - regex: 'jvm_gc_collection_seconds.*' - action: drop + - job_name: 'kubernetes-pod-nginx' + sample_limit: {{ .Values.scrapeSampleLimit }} + metrics_path: /{{ .Values.prometheusMetricsEndpoint }} + kubernetes_sd_configs: + - role: pod + relabel_configs: + - source_labels: [ __address__ ] + action: keep + regex: '.*:10254$' + - source_labels: [__meta_kubernetes_pod_container_name] + target_label: container + action: replace + - source_labels: [__meta_kubernetes_pod_node_name] + target_label: host + action: replace + - source_labels: [__meta_kubernetes_namespace] + target_label: namespace + action: replace + metric_relabel_configs: + - source_labels: [__name__] + regex: 'go_memstats.*' + action: drop + - source_labels: [__name__] + regex: 'go_gc.*' + action: drop + - source_labels: [__name__] + regex: 'go_threads' + action: drop + - regex: exported_host + action: labeldrop exporters: - awsprometheusremotewrite: + prometheusremotewrite: endpoint: {{ .Values.ampurl }} - aws_auth: - region: {{ .Values.region }} - service: "aps" + auth: + authenticator: sigv4auth logging: - loglevel: info + loglevel: debug extensions: + sigv4auth: + region: {{ .Values.region }} + service: "aps" health_check: pprof: endpoint: :1888 zpages: endpoint: :55679 service: - extensions: [pprof, zpages, health_check] + extensions: [pprof, zpages, health_check, sigv4auth] pipelines: metrics: receivers: [prometheus] - exporters: [logging, awsprometheusremotewrite] + exporters: [logging, prometheusremotewrite] diff --git a/modules/workloads/nginx/rules.tf b/modules/workloads/nginx/rules.tf new file mode 100644 index 0000000..dab099e --- /dev/null +++ b/modules/workloads/nginx/rules.tf @@ -0,0 +1,44 @@ +# Prioritize recording rules over alerting rules for limits (10) + +################################################################################################################################################ +# Recording rules ############################################################################################################################## +################################################################################################################################################ + +resource "aws_prometheus_rule_group_namespace" "recording_rules" { + count = var.enable_recording_rules ? 1 : 0 + name = "acclerator-nginx-rules" + workspace_id = var.managed_prometheus_workspace_id + data = < 5 + for: 1m + labels: + severity: critical + annotations: + summary: Nginx high HTTP 4xx error rate (instance {{ $labels.instance }}) + description: "Too many HTTP requests with status 4xx (> 5%)\n VALUE = {{ $value }}\n LABELS = {{ $labels }}" + - name: Nginx-HTTP-5xx-error-rate + rules: + - alert: metric:alerting_rule + expr: sum(rate(nginx_http_requests_total{status=~"^5.."}[1m])) / sum(rate(nginx_http_requests_total[1m])) * 100 > 5 + for: 1m + labels: + severity: critical + annotations: + summary: Nginx high HTTP 5xx error rate (instance {{ $labels.instance }}) + description: "Too many HTTP requests with status 5xx (> 5%)\n VALUE = {{ $value }}\n LABELS = {{ $labels }}" + - name: Nginx-high-latency + rules: + - alert: metric:alerting_rule + expr: histogram_quantile(0.99, sum(rate(nginx_http_request_duration_seconds_bucket[2m])) by (host, node)) > 3 + for: 2m + labels: + severity: warning + annotations: + summary: Nginx latency high (instance {{ $labels.instance }}) + description: "Nginx p99 latency is higher than 3 seconds\n VALUE = {{ $value }}\n LABELS = {{ $labels }}" +EOF +} \ No newline at end of file diff --git a/modules/workloads/nginx/variables.tf b/modules/workloads/nginx/variables.tf index b7d13c3..59e9f40 100644 --- a/modules/workloads/nginx/variables.tf +++ b/modules/workloads/nginx/variables.tf @@ -1,34 +1,128 @@ + +variable "eks_cluster_id" { + description = "EKS Cluster Id" + type = string +} + variable "helm_config" { description = "Helm Config for Prometheus" type = any default = {} } -variable "amazon_prometheus_workspace_endpoint" { +variable "irsa_iam_role_path" { + description = "IAM role path for IRSA roles" + type = string + default = "/" +} + +variable "irsa_iam_permissions_boundary" { + description = "IAM permissions boundary for IRSA roles" + type = string + default = "" +} + +variable "managed_prometheus_workspace_endpoint" { description = "Amazon Managed Prometheus Workspace Endpoint" type = string default = null } +variable "managed_prometheus_workspace_id" { + description = "Amazon Managed Prometheus Workspace ID" + type = string + default = null +} -variable "amazon_prometheus_workspace_region" { +variable "managed_prometheus_workspace_region" { description = "Amazon Managed Prometheus Workspace's Region" type = string default = null } -variable "addon_context" { - description = "Input configuration for the addon" - type = object({ - aws_caller_identity_account_id = string - aws_caller_identity_arn = string - aws_eks_cluster_endpoint = string - aws_partition_id = string - aws_region_name = string - eks_cluster_id = string - eks_oidc_issuer_url = string - eks_oidc_provider_arn = string - irsa_iam_permissions_boundary = string - irsa_iam_role_path = string - tags = map(string) - }) +variable "dashboards_folder_id" { + type = string } + +variable "enable_recording_rules" { + type = bool + default = true +} + +variable "enable_alerting_rules" { + type = bool + default = true +} + +variable "enable_dashboards" { + type = bool + default = true +} + +variable "enable_kube_state_metrics" { + type = bool + default = true +} + +variable "enable_node_exporter" { + type = bool + default = true +} + +variable "config" { + type = object({ + helm_config = map(any) + + kms_create_namespace = bool + ksm_k8s_namespace = string + ksm_helm_chart_name = string + ksm_helm_chart_version = string + ksm_helm_release_name = string + ksm_helm_repo_url = string + ksm_helm_settings = map(string) + ksm_helm_values = map(any) + + ne_create_namespace = bool + ne_k8s_namespace = string + ne_helm_chart_name = string + ne_helm_chart_version = string + ne_helm_release_name = string + ne_helm_repo_url = string + ne_helm_settings = map(string) + ne_helm_values = map(any) + + }) + + default = { + enable_kube_state_metrics = true + enable_node_exporter = true + + helm_config = {} + + kms_create_namespace = true + ksm_helm_chart_name = "kube-state-metrics" + ksm_helm_chart_version = "4.9.2" + ksm_helm_release_name = "kube-state-metrics" + ksm_helm_repo_url = "https://prometheus-community.github.io/helm-charts" + ksm_helm_settings = {} + ksm_helm_values = {} + ksm_k8s_namespace = "kube-system" + + ne_create_namespace = true + ne_k8s_namespace = "prometheus-node-exporter" + ne_helm_chart_name = "prometheus-node-exporter" + ne_helm_chart_version = "2.0.3" + ne_helm_release_name = "prometheus-node-exporter" + ne_helm_repo_url = "https://prometheus-community.github.io/helm-charts" + ne_helm_settings = {} + ne_helm_values = {} + } + nullable = false +} + +variable "tags" { + description = "Additional tags (e.g. `map('BusinessUnit`,`XYZ`)" + type = map(string) + default = {} +} + + diff --git a/modules/workloads/nginx/versions.tf b/modules/workloads/nginx/versions.tf index 024e519..628087a 100644 --- a/modules/workloads/nginx/versions.tf +++ b/modules/workloads/nginx/versions.tf @@ -10,5 +10,9 @@ terraform { source = "hashicorp/kubernetes" version = ">= 2.10" } + grafana = { + source = "grafana/grafana" + version = ">= 1.25.0" + } } }