refactor module with all the required files

This commit is contained in:
Vara Bonthu
2022-08-11 12:47:54 +01:00
committed by Rodrigue Koffi
parent a99936ba7e
commit 09ef0d6c17
68 changed files with 452 additions and 230 deletions
@@ -41,8 +41,7 @@ the ADOT Operator.
| Name | Source | Version |
|------|--------|---------|
| <a name="module_cert_manager"></a> [cert\_manager](#module\_cert\_manager) | ../cert-manager | n/a |
| <a name="module_helm_addon"></a> [helm\_addon](#module\_helm\_addon) | ../helm-addon | n/a |
| <a name="module_cert_manager"></a> [cert\_manager](#module\_cert\_manager) | github.com/aws-ia/terraform-aws-eks-blueprints/modules/kubernetes-addons/cert-manager | n/a |
## Resources
@@ -62,9 +61,9 @@ the ADOT Operator.
|------|-------------|------|---------|:--------:|
| <a name="input_addon_config"></a> [addon\_config](#input\_addon\_config) | Amazon EKS Managed CoreDNS Add-on config | `any` | `{}` | no |
| <a name="input_addon_context"></a> [addon\_context](#input\_addon\_context) | Input configuration for the addon | <pre>object({<br> aws_caller_identity_account_id = string<br> aws_caller_identity_arn = string<br> aws_eks_cluster_endpoint = string<br> aws_partition_id = string<br> aws_region_name = string<br> eks_cluster_id = string<br> eks_oidc_issuer_url = string<br> eks_oidc_provider_arn = string<br> irsa_iam_role_path = string<br> tags = map(string)<br> })</pre> | n/a | yes |
| <a name="input_enable_amazon_eks_adot"></a> [enable\_amazon\_eks\_adot](#input\_enable\_amazon\_eks\_adot) | Enable Amazon EKS ADOT add-on | `bool` | `true` | no |
| <a name="input_enable_opentelemetry_operator"></a> [enable\_opentelemetry\_operator](#input\_enable\_opentelemetry\_operator) | Enable opentelemetry operator addon | `bool` | `false` | no |
| <a name="input_enable_cert_manager"></a> [enable\_cert\_manager](#input\_enable\_cert\_manager) | n/a | `bool` | `true` | no |
| <a name="input_helm_config"></a> [helm\_config](#input\_helm\_config) | Helm provider config for ADOT Operator AddOn | `any` | `{}` | no |
| <a name="input_kubernetes_version"></a> [kubernetes\_version](#input\_kubernetes\_version) | n/a | `string` | n/a | yes |
## Outputs
+171
View File
@@ -0,0 +1,171 @@
#---------------------------------------------------------------
# Observability Resources
#---------------------------------------------------------------
module "managed_grafana" {
source = "terraform-aws-modules/managed-service-grafana/aws"
version = "~> 1.3"
# Workspace
name = local.name
stack_set_name = local.name
data_sources = ["PROMETHEUS"]
associate_license = false
# # Role associations
# Pending https://github.com/hashicorp/terraform-provider-aws/issues/24166
# role_associations = {
# "ADMIN" = {
# "group_ids" = []
# "user_ids" = []
# }
# "EDITOR" = {
# "group_ids" = []
# "user_ids" = []
# }
# }
tags = local.tags
}
resource "grafana_data_source" "prometheus" {
type = "prometheus"
name = "amp"
is_default = true
url = module.managed_prometheus.workspace_prometheus_endpoint
json_data {
http_method = "GET"
sigv4_auth = true
sigv4_auth_type = "workspace-iam-role"
sigv4_region = local.region
}
}
resource "grafana_folder" "this" {
title = "Observability"
}
resource "grafana_dashboard" "this" {
folder = grafana_folder.this.id
config_json = file("${path.module}/dashboards/default.json")
}
module "managed_prometheus" {
source = "terraform-aws-modules/managed-service-prometheus/aws"
version = "~> 2.1"
workspace_alias = local.name
alert_manager_definition = <<-EOT
alertmanager_config: |
route:
receiver: 'default'
receivers:
- name: 'default'
EOT
rule_group_namespaces = {
haproxy = {
name = "haproxy_rules"
data = <<-EOT
groups:
- name: obsa-haproxy-down-alert
rules:
- alert: HA_proxy_down
expr: haproxy_up == 0
for: 0m
labels:
severity: critical
annotations:
summary: HAProxy down (instance {{ $labels.instance }})
description: "HAProxy down\n VALUE = {{ $value }}\n LABELS = {{ $labels }}"
- name: obsa-haproxy-http4xx-error-alert
rules:
- alert: Ha_proxy_High_Http4xx_ErrorRate_Backend
expr: sum by (backend) (rate(haproxy_server_http_responses_total{code="4xx"}[1m])) / sum by (backend) (rate(haproxy_server_http_responses_total[1m]) * 100) > 5
for: 1m
labels:
severity: critical
annotations:
summary: HAProxy high HTTP 4xx error rate backend (instance {{ $labels.instance }})
description: "Too many HTTP requests with status 4xx (> 5%) on backend {{ $labels.fqdn }}/{{ $labels.backend }}\n VALUE = {{ $value }}\n LABELS = {{ $labels }}"
- name: obsa-haproxy-http5xx-error-alert
rules:
- alert: Ha_proxy_High_Http5xx_ErrorRate_Backend
expr: sum by (backend) (rate(haproxy_server_http_responses_total{code="5xx"}[1m])) / sum by (backend) (rate(haproxy_server_http_responses_total[1m]) * 100) > 5
for: 1m
labels:
severity: critical
annotations:
summary: HAProxy high HTTP 5xx error rate backend (instance {{ $labels.instance }})
description: "Too many HTTP requests with status 5xx (> 5%) on backend {{ $labels.fqdn }}/{{ $labels.backend }}\n VALUE = {{ $value }}\n LABELS = {{ $labels }}"
- name: obsa-haproxy-Http4xx-ErrorRate-Server-alert
rules:
- alert: Ha_proxy_High_Http4xx_ErrorRate_Server
expr: sum by (server) (rate(haproxy_server_http_responses_total{code="4xx"}[1m])) / sum by (server) (rate(haproxy_server_http_responses_total[1m]) * 100) > 5
for: 1m
labels:
severity: critical
annotations:
summary: HAProxy high HTTP 4xx error rate server (instance {{ $labels.instance }})
description: "Too many HTTP requests with status 4xx (> 5%) on server {{ $labels.server }}\n VALUE = {{ $value }}\n LABELS = {{ $labels }}"
- name: obsa-haproxy-Http5xx-ErrorRate-Server-alert
rules:
- alert: Ha_proxy_High_Http5xx_ErrorRate_Server
expr: sum by (server) (rate(haproxy_server_http_responses_total{code="5xx"}[1m])) / sum by (server) (rate(haproxy_server_http_responses_total[1m]) * 100) > 5
for: 1m
labels:
severity: critical
annotations:
summary: HAProxy high HTTP 5xx error rate server (instance {{ $labels.instance }})
description: "Too many HTTP requests with status 5xx (> 5%) on server {{ $labels.server }}\n VALUE = {{ $value }}\n LABELS = {{ $labels }}"
EOT
}
}
tags = local.tags
}
#---------------------------------------------------------------
# Sample Application
#---------------------------------------------------------------
# https://github.com/haproxy-ingress/charts/tree/master/haproxy-ingress
resource "helm_release" "haproxy_ingress" {
namespace = "haproxy-ingress"
create_namespace = true
name = "haproxy-ingress"
repository = "https://haproxy-ingress.github.io/charts"
chart = "haproxy-ingress"
version = "0.13.7"
set {
name = "defaultBackend.enabled"
value = true
}
set {
name = "controller.stats.enabled"
value = true
}
set {
name = "controller.metrics.enabled"
value = true
}
set {
name = "controller.metrics.service.annotations.prometheus\\.io/port"
value = 9101
type = "string"
}
set {
name = "controller.metrics.service.annotations.prometheus\\.io/scrape"
value = true
type = "string"
}
}
+96
View File
@@ -0,0 +1,96 @@
# Observability Pattern for Java/JMX
This module provides an automated experience around Observability for Nginx workloads.
It provides the following resources:
- AWS Distro For OpenTelemetry Operator and Collector
- AWS Managed Grafana Dashboard and data source
- Alerts and recording rules with AWS Managed Service for Prometheus
<!-- BEGINNING OF PRE-COMMIT-TERRAFORM DOCS HOOK -->
## Requirements
| Name | Version |
|------|---------|
| <a name="requirement_grafana"></a> [grafana](#requirement\_grafana) | 1.25.0 |
## Providers
| Name | Version |
|------|---------|
| <a name="provider_aws"></a> [aws](#provider\_aws) | n/a |
| <a name="provider_grafana"></a> [grafana](#provider\_grafana) | 1.25.0 |
| <a name="provider_helm"></a> [helm](#provider\_helm) | n/a |
## Modules
| Name | Source | Version |
|------|--------|---------|
| <a name="module_helm_addon"></a> [helm\_addon](#module\_helm\_addon) | github.com/aws-ia/terraform-aws-eks-blueprints/modules/kubernetes-addons/helm-addon | n/a |
## Resources
| Name | Type |
|------|------|
| [aws_prometheus_rule_group_namespace.api-availability](https://registry.terraform.io/providers/hashicorp/aws/latest/docs/resources/prometheus_rule_group_namespace) | resource |
| [aws_prometheus_rule_group_namespace.apibr](https://registry.terraform.io/providers/hashicorp/aws/latest/docs/resources/prometheus_rule_group_namespace) | resource |
| [aws_prometheus_rule_group_namespace.apihg](https://registry.terraform.io/providers/hashicorp/aws/latest/docs/resources/prometheus_rule_group_namespace) | resource |
| [aws_prometheus_rule_group_namespace.kubelet](https://registry.terraform.io/providers/hashicorp/aws/latest/docs/resources/prometheus_rule_group_namespace) | resource |
| [aws_prometheus_rule_group_namespace.kubeprom-recording](https://registry.terraform.io/providers/hashicorp/aws/latest/docs/resources/prometheus_rule_group_namespace) | resource |
| [aws_prometheus_rule_group_namespace.kubepromnr](https://registry.terraform.io/providers/hashicorp/aws/latest/docs/resources/prometheus_rule_group_namespace) | resource |
| [aws_prometheus_rule_group_namespace.kubescheduler](https://registry.terraform.io/providers/hashicorp/aws/latest/docs/resources/prometheus_rule_group_namespace) | resource |
| [aws_prometheus_rule_group_namespace.kubprom](https://registry.terraform.io/providers/hashicorp/aws/latest/docs/resources/prometheus_rule_group_namespace) | resource |
| [aws_prometheus_rule_group_namespace.ne](https://registry.terraform.io/providers/hashicorp/aws/latest/docs/resources/prometheus_rule_group_namespace) | resource |
| [aws_prometheus_rule_group_namespace.noderules](https://registry.terraform.io/providers/hashicorp/aws/latest/docs/resources/prometheus_rule_group_namespace) | resource |
| [grafana_dashboard.alertmanager](https://registry.terraform.io/providers/grafana/grafana/1.25.0/docs/resources/dashboard) | resource |
| [grafana_dashboard.apis](https://registry.terraform.io/providers/grafana/grafana/1.25.0/docs/resources/dashboard) | resource |
| [grafana_dashboard.cluster](https://registry.terraform.io/providers/grafana/grafana/1.25.0/docs/resources/dashboard) | resource |
| [grafana_dashboard.clusternw](https://registry.terraform.io/providers/grafana/grafana/1.25.0/docs/resources/dashboard) | resource |
| [grafana_dashboard.controller](https://registry.terraform.io/providers/grafana/grafana/1.25.0/docs/resources/dashboard) | resource |
| [grafana_dashboard.coredns](https://registry.terraform.io/providers/grafana/grafana/1.25.0/docs/resources/dashboard) | resource |
| [grafana_dashboard.etcd](https://registry.terraform.io/providers/grafana/grafana/1.25.0/docs/resources/dashboard) | resource |
| [grafana_dashboard.grafana](https://registry.terraform.io/providers/grafana/grafana/1.25.0/docs/resources/dashboard) | resource |
| [grafana_dashboard.kubelet](https://registry.terraform.io/providers/grafana/grafana/1.25.0/docs/resources/dashboard) | resource |
| [grafana_dashboard.macos](https://registry.terraform.io/providers/grafana/grafana/1.25.0/docs/resources/dashboard) | resource |
| [grafana_dashboard.necluster](https://registry.terraform.io/providers/grafana/grafana/1.25.0/docs/resources/dashboard) | resource |
| [grafana_dashboard.nenode](https://registry.terraform.io/providers/grafana/grafana/1.25.0/docs/resources/dashboard) | resource |
| [grafana_dashboard.nenodeuse](https://registry.terraform.io/providers/grafana/grafana/1.25.0/docs/resources/dashboard) | resource |
| [grafana_dashboard.nodes](https://registry.terraform.io/providers/grafana/grafana/1.25.0/docs/resources/dashboard) | resource |
| [grafana_dashboard.nsnw](https://registry.terraform.io/providers/grafana/grafana/1.25.0/docs/resources/dashboard) | resource |
| [grafana_dashboard.nsnwworkload](https://registry.terraform.io/providers/grafana/grafana/1.25.0/docs/resources/dashboard) | resource |
| [grafana_dashboard.nspods](https://registry.terraform.io/providers/grafana/grafana/1.25.0/docs/resources/dashboard) | resource |
| [grafana_dashboard.nsworkload](https://registry.terraform.io/providers/grafana/grafana/1.25.0/docs/resources/dashboard) | resource |
| [grafana_dashboard.nwworload](https://registry.terraform.io/providers/grafana/grafana/1.25.0/docs/resources/dashboard) | resource |
| [grafana_dashboard.podnetwork](https://registry.terraform.io/providers/grafana/grafana/1.25.0/docs/resources/dashboard) | resource |
| [grafana_dashboard.pods](https://registry.terraform.io/providers/grafana/grafana/1.25.0/docs/resources/dashboard) | resource |
| [grafana_dashboard.prometheus](https://registry.terraform.io/providers/grafana/grafana/1.25.0/docs/resources/dashboard) | resource |
| [grafana_dashboard.proxy](https://registry.terraform.io/providers/grafana/grafana/1.25.0/docs/resources/dashboard) | resource |
| [grafana_dashboard.pv](https://registry.terraform.io/providers/grafana/grafana/1.25.0/docs/resources/dashboard) | resource |
| [grafana_dashboard.scheduler](https://registry.terraform.io/providers/grafana/grafana/1.25.0/docs/resources/dashboard) | resource |
| [grafana_dashboard.workloads](https://registry.terraform.io/providers/grafana/grafana/1.25.0/docs/resources/dashboard) | resource |
| [helm_release.kube_state_metrics](https://registry.terraform.io/providers/hashicorp/helm/latest/docs/resources/release) | resource |
| [helm_release.prometheus_node_exporter](https://registry.terraform.io/providers/hashicorp/helm/latest/docs/resources/release) | resource |
| [aws_caller_identity.current](https://registry.terraform.io/providers/hashicorp/aws/latest/docs/data-sources/caller_identity) | data source |
| [aws_eks_cluster.eks_cluster](https://registry.terraform.io/providers/hashicorp/aws/latest/docs/data-sources/eks_cluster) | data source |
| [aws_partition.current](https://registry.terraform.io/providers/hashicorp/aws/latest/docs/data-sources/partition) | data source |
| [aws_region.current](https://registry.terraform.io/providers/hashicorp/aws/latest/docs/data-sources/region) | data source |
## Inputs
| Name | Description | Type | Default | Required |
|------|-------------|------|---------|:--------:|
| <a name="input_config"></a> [config](#input\_config) | n/a | <pre>object({<br> helm_config = map(any)<br><br> enable_kube_state_metrics = bool<br> kms_create_namespace = bool<br> ksm_k8s_namespace = string<br> ksm_helm_chart_name = string<br> ksm_helm_chart_version = string<br> ksm_helm_release_name = string<br> ksm_helm_repo_url = string<br> ksm_helm_settings = map(string)<br> ksm_helm_values = map(any)<br><br> enable_node_exporter = bool<br> ne_create_namespace = bool<br> ne_k8s_namespace = string<br> ne_helm_chart_name = string<br> ne_helm_chart_version = string<br> ne_helm_release_name = string<br> ne_helm_repo_url = string<br> ne_helm_settings = map(string)<br> ne_helm_values = map(any)<br><br> enable_dashboards = bool<br><br> enable_recording_rules = bool<br> })</pre> | <pre>{<br> "enable_dashboards": true,<br> "enable_kube_state_metrics": true,<br> "enable_node_exporter": true,<br> "enable_recording_rules": true,<br> "helm_config": {},<br> "kms_create_namespace": true,<br> "ksm_helm_chart_name": "kube-state-metrics",<br> "ksm_helm_chart_version": "4.9.2",<br> "ksm_helm_release_name": "kube-state-metrics",<br> "ksm_helm_repo_url": "https://prometheus-community.github.io/helm-charts",<br> "ksm_helm_settings": {},<br> "ksm_helm_values": {},<br> "ksm_k8s_namespace": "kube-system",<br> "ne_create_namespace": true,<br> "ne_helm_chart_name": "prometheus-node-exporter",<br> "ne_helm_chart_version": "2.0.3",<br> "ne_helm_release_name": "prometheus-node-exporter",<br> "ne_helm_repo_url": "https://prometheus-community.github.io/helm-charts",<br> "ne_helm_settings": {},<br> "ne_helm_values": {},<br> "ne_k8s_namespace": "prometheus-node-exporter"<br>}</pre> | no |
| <a name="input_dashboards_folder_id"></a> [dashboards\_folder\_id](#input\_dashboards\_folder\_id) | n/a | `string` | n/a | yes |
| <a name="input_eks_cluster_id"></a> [eks\_cluster\_id](#input\_eks\_cluster\_id) | EKS Cluster Id | `string` | n/a | yes |
| <a name="input_helm_config"></a> [helm\_config](#input\_helm\_config) | Helm Config for Prometheus | `any` | `{}` | no |
| <a name="input_irsa_iam_permissions_boundary"></a> [irsa\_iam\_permissions\_boundary](#input\_irsa\_iam\_permissions\_boundary) | IAM permissions boundary for IRSA roles | `string` | `""` | no |
| <a name="input_irsa_iam_role_path"></a> [irsa\_iam\_role\_path](#input\_irsa\_iam\_role\_path) | IAM role path for IRSA roles | `string` | `"/"` | no |
| <a name="input_managed_prometheus_workspace_endpoint"></a> [managed\_prometheus\_workspace\_endpoint](#input\_managed\_prometheus\_workspace\_endpoint) | Amazon Managed Prometheus Workspace Endpoint | `string` | `null` | no |
| <a name="input_managed_prometheus_workspace_id"></a> [managed\_prometheus\_workspace\_id](#input\_managed\_prometheus\_workspace\_id) | Amazon Managed Prometheus Workspace ID | `string` | `null` | no |
| <a name="input_managed_prometheus_workspace_region"></a> [managed\_prometheus\_workspace\_region](#input\_managed\_prometheus\_workspace\_region) | Amazon Managed Prometheus Workspace's Region | `string` | `null` | no |
| <a name="input_tags"></a> [tags](#input\_tags) | Additional tags (e.g. `map('BusinessUnit`,`XYZ`) | `map(string)` | `{}` | no |
## Outputs
No outputs.
<!-- END OF PRE-COMMIT-TERRAFORM DOCS HOOK -->
+155
View File
@@ -0,0 +1,155 @@
resource "grafana_dashboard" "alertmanager" {
count = var.config.enable_dashboards ? 1 : 0
folder = var.dashboards_folder_id
config_json = file("${path.module}/dashboards/alertmanager.json")
}
resource "grafana_dashboard" "workloads" {
count = var.config.enable_dashboards ? 1 : 0
folder = var.dashboards_folder_id
config_json = file("${path.module}/dashboards/workloads.json")
}
resource "grafana_dashboard" "scheduler" {
count = var.config.enable_dashboards ? 1 : 0
folder = var.dashboards_folder_id
config_json = file("${path.module}/dashboards/scheduler.json")
}
resource "grafana_dashboard" "proxy" {
count = var.config.enable_dashboards ? 1 : 0
folder = var.dashboards_folder_id
config_json = file("${path.module}/dashboards/proxy.json")
}
resource "grafana_dashboard" "prometheus" {
count = var.config.enable_dashboards ? 1 : 0
folder = var.dashboards_folder_id
config_json = file("${path.module}/dashboards/prometheus.json")
}
resource "grafana_dashboard" "podnetwork" {
count = var.config.enable_dashboards ? 1 : 0
folder = var.dashboards_folder_id
config_json = file("${path.module}/dashboards/pods-networking.json")
}
resource "grafana_dashboard" "pods" {
count = var.config.enable_dashboards ? 1 : 0
folder = var.dashboards_folder_id
config_json = file("${path.module}/dashboards/pods.json")
}
resource "grafana_dashboard" "pv" {
count = var.config.enable_dashboards ? 1 : 0
folder = var.dashboards_folder_id
config_json = file("${path.module}/dashboards/pesistentvolumes.json")
}
resource "grafana_dashboard" "nodes" {
count = var.config.enable_dashboards ? 1 : 0
folder = var.dashboards_folder_id
config_json = file("${path.module}/dashboards/nodes.json")
}
resource "grafana_dashboard" "necluster" {
count = var.config.enable_dashboards ? 1 : 0
folder = var.dashboards_folder_id
config_json = file("${path.module}/dashboards/nodeexpoter-use-cluster.json")
}
resource "grafana_dashboard" "nenodeuse" {
count = var.config.enable_dashboards ? 1 : 0
folder = var.dashboards_folder_id
config_json = file("${path.module}/dashboards/nodeexporter-use-node.json")
}
resource "grafana_dashboard" "nenode" {
count = var.config.enable_dashboards ? 1 : 0
folder = var.dashboards_folder_id
config_json = file("${path.module}/dashboards/nodeexporter-nodes.json")
}
resource "grafana_dashboard" "nwworload" {
count = var.config.enable_dashboards ? 1 : 0
folder = var.dashboards_folder_id
config_json = file("${path.module}/dashboards/networking-workloads.json")
}
resource "grafana_dashboard" "nsworkload" {
count = var.config.enable_dashboards ? 1 : 0
folder = var.dashboards_folder_id
config_json = file("${path.module}/dashboards/namespace-workloads.json")
}
resource "grafana_dashboard" "nspods" {
count = var.config.enable_dashboards ? 1 : 0
folder = var.dashboards_folder_id
config_json = file("${path.module}/dashboards/namespace-pods.json")
}
resource "grafana_dashboard" "nsnwworkload" {
count = var.config.enable_dashboards ? 1 : 0
folder = var.dashboards_folder_id
config_json = file("${path.module}/dashboards/namespace-nw-workloads.json")
}
resource "grafana_dashboard" "nsnw" {
count = var.config.enable_dashboards ? 1 : 0
folder = var.dashboards_folder_id
config_json = file("${path.module}/dashboards/namespace-networking.json")
}
resource "grafana_dashboard" "macos" {
count = var.config.enable_dashboards ? 1 : 0
folder = var.dashboards_folder_id
config_json = file("${path.module}/dashboards/macos.json")
}
resource "grafana_dashboard" "kubelet" {
count = var.config.enable_dashboards ? 1 : 0
folder = var.dashboards_folder_id
config_json = file("${path.module}/dashboards/kubelet.json")
}
resource "grafana_dashboard" "grafana" {
count = var.config.enable_dashboards ? 1 : 0
folder = var.dashboards_folder_id
config_json = file("${path.module}/dashboards/grafana.json")
}
resource "grafana_dashboard" "etcd" {
count = var.config.enable_dashboards ? 1 : 0
folder = var.dashboards_folder_id
config_json = file("${path.module}/dashboards/etcd.json")
}
resource "grafana_dashboard" "coredns" {
count = var.config.enable_dashboards ? 1 : 0
folder = var.dashboards_folder_id
config_json = file("${path.module}/dashboards/coredns.json")
}
resource "grafana_dashboard" "controller" {
count = var.config.enable_dashboards ? 1 : 0
folder = var.dashboards_folder_id
config_json = file("${path.module}/dashboards/controller.json")
}
resource "grafana_dashboard" "clusternw" {
count = var.config.enable_dashboards ? 1 : 0
folder = var.dashboards_folder_id
config_json = file("${path.module}/dashboards/cluster-networking.json")
}
resource "grafana_dashboard" "cluster" {
count = var.config.enable_dashboards ? 1 : 0
folder = var.dashboards_folder_id
config_json = file("${path.module}/dashboards/cluster.json")
}
resource "grafana_dashboard" "apis" {
count = var.config.enable_dashboards ? 1 : 0
folder = var.dashboards_folder_id
config_json = file("${path.module}/dashboards/apiserver.json")
}
@@ -0,0 +1,626 @@
{
"annotations": {
"list": [
{
"builtIn": 1,
"datasource": {
"type": "grafana",
"uid": "-- Grafana --"
},
"enable": true,
"hide": true,
"iconColor": "rgba(0, 211, 255, 1)",
"name": "Annotations & Alerts",
"target": {
"limit": 100,
"matchAny": false,
"tags": [],
"type": "dashboard"
},
"type": "dashboard"
}
]
},
"editable": false,
"fiscalYearStartMonth": 0,
"graphTooltip": 1,
"id": 22,
"iteration": 1656210900773,
"links": [],
"liveNow": false,
"panels": [
{
"collapsed": false,
"datasource": {
"type": "prometheus",
"uid": "prometheus"
},
"gridPos": {
"h": 1,
"w": 24,
"x": 0,
"y": 0
},
"id": 6,
"panels": [],
"targets": [
{
"datasource": {
"type": "prometheus",
"uid": "prometheus"
},
"refId": "A"
}
],
"title": "Alerts",
"type": "row"
},
{
"aliasColors": {},
"bars": false,
"dashLength": 10,
"dashes": false,
"datasource": {
"uid": "$datasource"
},
"fill": 1,
"fillGradient": 0,
"gridPos": {
"h": 7,
"w": 12,
"x": 0,
"y": 1
},
"hiddenSeries": false,
"id": 2,
"legend": {
"alignAsTable": false,
"avg": false,
"current": false,
"max": false,
"min": false,
"rightSide": false,
"show": false,
"total": false,
"values": false
},
"lines": true,
"linewidth": 1,
"links": [],
"nullPointMode": "null",
"options": {
"alertThreshold": true
},
"percentage": false,
"pluginVersion": "9.0.1",
"pointradius": 5,
"points": false,
"renderer": "flot",
"seriesOverrides": [],
"spaceLength": 10,
"stack": true,
"steppedLine": false,
"targets": [
{
"datasource": {
"uid": "$datasource"
},
"expr": "sum(alertmanager_alerts{namespace=~\"$namespace\",service=~\"$service\"}) by (namespace,service,instance)",
"format": "time_series",
"intervalFactor": 2,
"legendFormat": "{{instance}}",
"refId": "A"
}
],
"thresholds": [],
"timeRegions": [],
"title": "Alerts",
"tooltip": {
"shared": true,
"sort": 0,
"value_type": "individual"
},
"type": "graph",
"xaxis": {
"mode": "time",
"show": true,
"values": []
},
"yaxes": [
{
"format": "none",
"logBase": 1,
"show": true
},
{
"format": "none",
"logBase": 1,
"show": true
}
],
"yaxis": {
"align": false
}
},
{
"aliasColors": {},
"bars": false,
"dashLength": 10,
"dashes": false,
"datasource": {
"uid": "$datasource"
},
"fill": 1,
"fillGradient": 0,
"gridPos": {
"h": 7,
"w": 12,
"x": 12,
"y": 1
},
"hiddenSeries": false,
"id": 3,
"legend": {
"alignAsTable": false,
"avg": false,
"current": false,
"max": false,
"min": false,
"rightSide": false,
"show": false,
"total": false,
"values": false
},
"lines": true,
"linewidth": 1,
"links": [],
"nullPointMode": "null",
"options": {
"alertThreshold": true
},
"percentage": false,
"pluginVersion": "9.0.1",
"pointradius": 5,
"points": false,
"renderer": "flot",
"seriesOverrides": [],
"spaceLength": 10,
"stack": true,
"steppedLine": false,
"targets": [
{
"datasource": {
"uid": "$datasource"
},
"expr": "sum(rate(alertmanager_alerts_received_total{namespace=~\"$namespace\",service=~\"$service\"}[$__rate_interval])) by (namespace,service,instance)",
"format": "time_series",
"intervalFactor": 2,
"legendFormat": "{{instance}} Received",
"refId": "A"
},
{
"datasource": {
"uid": "$datasource"
},
"expr": "sum(rate(alertmanager_alerts_invalid_total{namespace=~\"$namespace\",service=~\"$service\"}[$__rate_interval])) by (namespace,service,instance)",
"format": "time_series",
"intervalFactor": 2,
"legendFormat": "{{instance}} Invalid",
"refId": "B"
}
],
"thresholds": [],
"timeRegions": [],
"title": "Alerts receive rate",
"tooltip": {
"shared": true,
"sort": 0,
"value_type": "individual"
},
"type": "graph",
"xaxis": {
"mode": "time",
"show": true,
"values": []
},
"yaxes": [
{
"format": "ops",
"logBase": 1,
"show": true
},
{
"format": "ops",
"logBase": 1,
"show": true
}
],
"yaxis": {
"align": false
}
},
{
"collapsed": false,
"datasource": {
"type": "prometheus",
"uid": "prometheus"
},
"gridPos": {
"h": 1,
"w": 24,
"x": 0,
"y": 8
},
"id": 7,
"panels": [],
"targets": [
{
"datasource": {
"type": "prometheus",
"uid": "prometheus"
},
"refId": "A"
}
],
"title": "Notifications",
"type": "row"
},
{
"aliasColors": {},
"bars": false,
"dashLength": 10,
"dashes": false,
"datasource": {
"uid": "$datasource"
},
"fill": 1,
"fillGradient": 0,
"gridPos": {
"h": 7,
"w": 6,
"x": 0,
"y": 9
},
"hiddenSeries": false,
"id": 4,
"legend": {
"alignAsTable": false,
"avg": false,
"current": false,
"max": false,
"min": false,
"rightSide": false,
"show": false,
"total": false,
"values": false
},
"lines": true,
"linewidth": 1,
"links": [],
"nullPointMode": "null",
"options": {
"alertThreshold": true
},
"percentage": false,
"pluginVersion": "9.0.1",
"pointradius": 5,
"points": false,
"renderer": "flot",
"repeat": "integration",
"seriesOverrides": [],
"spaceLength": 10,
"stack": true,
"steppedLine": false,
"targets": [
{
"datasource": {
"uid": "$datasource"
},
"expr": "sum(rate(alertmanager_notifications_total{namespace=~\"$namespace\",service=~\"$service\", integration=\"$integration\"}[$__rate_interval])) by (integration,namespace,service,instance)",
"format": "time_series",
"intervalFactor": 2,
"legendFormat": "{{instance}} Total",
"refId": "A"
},
{
"datasource": {
"uid": "$datasource"
},
"expr": "sum(rate(alertmanager_notifications_failed_total{namespace=~\"$namespace\",service=~\"$service\", integration=\"$integration\"}[$__rate_interval])) by (integration,namespace,service,instance)",
"format": "time_series",
"intervalFactor": 2,
"legendFormat": "{{instance}} Failed",
"refId": "B"
}
],
"thresholds": [],
"timeRegions": [],
"title": "$integration: Notifications Send Rate",
"tooltip": {
"shared": true,
"sort": 0,
"value_type": "individual"
},
"type": "graph",
"xaxis": {
"mode": "time",
"show": true,
"values": []
},
"yaxes": [
{
"format": "ops",
"logBase": 1,
"show": true
},
{
"format": "ops",
"logBase": 1,
"show": true
}
],
"yaxis": {
"align": false
}
},
{
"aliasColors": {},
"bars": false,
"dashLength": 10,
"dashes": false,
"datasource": {
"uid": "$datasource"
},
"fill": 1,
"fillGradient": 0,
"gridPos": {
"h": 7,
"w": 6,
"x": 0,
"y": 16
},
"hiddenSeries": false,
"id": 5,
"legend": {
"alignAsTable": false,
"avg": false,
"current": false,
"max": false,
"min": false,
"rightSide": false,
"show": false,
"total": false,
"values": false
},
"lines": true,
"linewidth": 1,
"links": [],
"nullPointMode": "null",
"options": {
"alertThreshold": true
},
"percentage": false,
"pluginVersion": "9.0.1",
"pointradius": 5,
"points": false,
"renderer": "flot",
"repeat": "integration",
"seriesOverrides": [],
"spaceLength": 10,
"stack": false,
"steppedLine": false,
"targets": [
{
"datasource": {
"uid": "$datasource"
},
"expr": "histogram_quantile(0.99,\n sum(rate(alertmanager_notification_latency_seconds_bucket{namespace=~\"$namespace\",service=~\"$service\", integration=\"$integration\"}[$__rate_interval])) by (le,namespace,service,instance)\n) \n",
"format": "time_series",
"intervalFactor": 2,
"legendFormat": "{{instance}} 99th Percentile",
"refId": "A"
},
{
"datasource": {
"uid": "$datasource"
},
"expr": "histogram_quantile(0.50,\n sum(rate(alertmanager_notification_latency_seconds_bucket{namespace=~\"$namespace\",service=~\"$service\", integration=\"$integration\"}[$__rate_interval])) by (le,namespace,service,instance)\n) \n",
"format": "time_series",
"intervalFactor": 2,
"legendFormat": "{{instance}} Median",
"refId": "B"
},
{
"datasource": {
"uid": "$datasource"
},
"expr": "sum(rate(alertmanager_notification_latency_seconds_sum{namespace=~\"$namespace\",service=~\"$service\", integration=\"$integration\"}[$__rate_interval])) by (namespace,service,instance)\n/\nsum(rate(alertmanager_notification_latency_seconds_count{namespace=~\"$namespace\",service=~\"$service\", integration=\"$integration\"}[$__rate_interval])) by (namespace,service,instance)\n",
"format": "time_series",
"intervalFactor": 2,
"legendFormat": "{{instance}} Average",
"refId": "C"
}
],
"thresholds": [],
"timeRegions": [],
"title": "$integration: Notification Duration",
"tooltip": {
"shared": true,
"sort": 0,
"value_type": "individual"
},
"type": "graph",
"xaxis": {
"mode": "time",
"show": true,
"values": []
},
"yaxes": [
{
"format": "s",
"logBase": 1,
"show": true
},
{
"format": "s",
"logBase": 1,
"show": true
}
],
"yaxis": {
"align": false
}
}
],
"refresh": "30s",
"schemaVersion": 36,
"style": "dark",
"tags": [
"alertmanager-mixin"
],
"templating": {
"list": [
{
"current": {
"selected": false,
"text": "Prometheus",
"value": "Prometheus"
},
"hide": 0,
"includeAll": false,
"label": "Data Source",
"multi": false,
"name": "datasource",
"options": [],
"query": "prometheus",
"refresh": 1,
"regex": "",
"skipUrlSync": false,
"type": "datasource"
},
{
"current": {
"selected": false,
"text": "default",
"value": "default"
},
"datasource": {
"type": "prometheus",
"uid": "$datasource"
},
"definition": "",
"hide": 0,
"includeAll": false,
"label": "namespace",
"multi": false,
"name": "namespace",
"options": [],
"query": {
"query": "label_values(alertmanager_alerts, namespace)",
"refId": "Prometheus-namespace-Variable-Query"
},
"refresh": 2,
"regex": "",
"skipUrlSync": false,
"sort": 1,
"tagValuesQuery": "",
"tagsQuery": "",
"type": "query",
"useTags": false
},
{
"current": {
"selected": false,
"text": "kube-prometheus-stack-alertmanager",
"value": "kube-prometheus-stack-alertmanager"
},
"datasource": {
"type": "prometheus",
"uid": "$datasource"
},
"definition": "",
"hide": 0,
"includeAll": false,
"label": "service",
"multi": false,
"name": "service",
"options": [],
"query": {
"query": "label_values(alertmanager_alerts, service)",
"refId": "Prometheus-service-Variable-Query"
},
"refresh": 2,
"regex": "",
"skipUrlSync": false,
"sort": 1,
"tagValuesQuery": "",
"tagsQuery": "",
"type": "query",
"useTags": false
},
{
"current": {
"selected": false,
"text": "All",
"value": "$__all"
},
"datasource": {
"type": "prometheus",
"uid": "$datasource"
},
"definition": "",
"hide": 2,
"includeAll": true,
"multi": false,
"name": "integration",
"options": [],
"query": {
"query": "label_values(alertmanager_notifications_total{integration=~\".*\"}, integration)",
"refId": "Prometheus-integration-Variable-Query"
},
"refresh": 2,
"regex": "",
"skipUrlSync": false,
"sort": 1,
"tagValuesQuery": "",
"tagsQuery": "",
"type": "query",
"useTags": false
}
]
},
"time": {
"from": "now-1h",
"to": "now"
},
"timepicker": {
"refresh_intervals": [
"5s",
"10s",
"30s",
"1m",
"5m",
"15m",
"30m",
"1h",
"2h",
"1d"
],
"time_options": [
"5m",
"15m",
"1h",
"6h",
"12h",
"24h",
"2d",
"7d",
"30d"
]
},
"timezone": "utc",
"title": "Alertmanager / Overview",
"uid": "alertmanager-overview",
"version": 1,
"weekStart": ""
}
File diff suppressed because it is too large Load Diff
File diff suppressed because it is too large Load Diff
File diff suppressed because it is too large Load Diff
File diff suppressed because it is too large Load Diff
File diff suppressed because it is too large Load Diff
File diff suppressed because it is too large Load Diff
@@ -0,0 +1,563 @@
{
"annotations": {
"list": [
{
"builtIn": 1,
"datasource": {
"type": "datasource",
"uid": "grafana"
},
"enable": true,
"hide": true,
"iconColor": "rgba(0, 211, 255, 1)",
"name": "Annotations & Alerts",
"target": {
"limit": 100,
"matchAny": false,
"tags": [],
"type": "dashboard"
},
"type": "dashboard"
}
]
},
"editable": true,
"fiscalYearStartMonth": 0,
"graphTooltip": 0,
"id": 15,
"iteration": 1656211285882,
"links": [],
"liveNow": false,
"panels": [
{
"datasource": {
"uid": "$datasource"
},
"fieldConfig": {
"defaults": {
"mappings": [],
"noValue": "0",
"thresholds": {
"mode": "absolute",
"steps": [
{
"color": "green",
"value": null
},
{
"color": "red",
"value": 80
}
]
}
},
"overrides": []
},
"gridPos": {
"h": 5,
"w": 6,
"x": 0,
"y": 0
},
"id": 6,
"options": {
"colorMode": "value",
"graphMode": "area",
"justifyMode": "auto",
"orientation": "auto",
"reduceOptions": {
"calcs": [
"mean"
],
"fields": "",
"values": false
},
"text": {},
"textMode": "auto"
},
"pluginVersion": "9.0.1",
"targets": [
{
"datasource": {
"uid": "$datasource"
},
"expr": "grafana_alerting_result_total{job=~\"$job\", instance=~\"$instance\", state=\"alerting\"}",
"instant": true,
"interval": "",
"legendFormat": "",
"refId": "A"
}
],
"title": "Firing Alerts",
"type": "stat"
},
{
"datasource": {
"uid": "$datasource"
},
"fieldConfig": {
"defaults": {
"mappings": [],
"thresholds": {
"mode": "absolute",
"steps": [
{
"color": "green",
"value": null
},
{
"color": "red",
"value": 80
}
]
}
},
"overrides": []
},
"gridPos": {
"h": 5,
"w": 6,
"x": 6,
"y": 0
},
"id": 8,
"options": {
"colorMode": "value",
"graphMode": "area",
"justifyMode": "auto",
"orientation": "auto",
"reduceOptions": {
"calcs": [
"mean"
],
"fields": "",
"values": false
},
"text": {},
"textMode": "auto"
},
"pluginVersion": "9.0.1",
"targets": [
{
"datasource": {
"uid": "$datasource"
},
"expr": "sum(grafana_stat_totals_dashboard{job=~\"$job\", instance=~\"$instance\"})",
"interval": "",
"legendFormat": "",
"refId": "A"
}
],
"title": "Dashboards",
"type": "stat"
},
{
"datasource": {
"uid": "$datasource"
},
"fieldConfig": {
"defaults": {
"custom": {
"displayMode": "auto",
"inspect": false
},
"mappings": [],
"thresholds": {
"mode": "absolute",
"steps": [
{
"color": "green",
"value": null
},
{
"color": "red",
"value": 80
}
]
}
},
"overrides": []
},
"gridPos": {
"h": 5,
"w": 12,
"x": 12,
"y": 0
},
"id": 10,
"options": {
"footer": {
"fields": "",
"reducer": [
"sum"
],
"show": false
},
"showHeader": true
},
"pluginVersion": "9.0.1",
"targets": [
{
"datasource": {
"uid": "$datasource"
},
"expr": "grafana_build_info{job=~\"$job\", instance=~\"$instance\"}",
"instant": true,
"interval": "",
"legendFormat": "",
"refId": "A"
}
],
"title": "Build Info",
"transformations": [
{
"id": "labelsToFields",
"options": {}
},
{
"id": "merge",
"options": {}
},
{
"id": "organize",
"options": {
"excludeByName": {
"Time": true,
"Value": true,
"branch": true,
"container": true,
"goversion": true,
"namespace": true,
"pod": true,
"revision": true
},
"indexByName": {
"Time": 7,
"Value": 11,
"branch": 4,
"container": 8,
"edition": 2,
"goversion": 6,
"instance": 1,
"job": 0,
"namespace": 9,
"pod": 10,
"revision": 5,
"version": 3
},
"renameByName": {}
}
}
],
"type": "table"
},
{
"aliasColors": {},
"bars": false,
"dashLength": 10,
"dashes": false,
"datasource": {
"uid": "$datasource"
},
"fieldConfig": {
"defaults": {
"links": []
},
"overrides": []
},
"fill": 1,
"fillGradient": 0,
"gridPos": {
"h": 8,
"w": 12,
"x": 0,
"y": 5
},
"hiddenSeries": false,
"id": 2,
"legend": {
"avg": false,
"current": false,
"max": false,
"min": false,
"show": true,
"total": false,
"values": false
},
"lines": true,
"linewidth": 1,
"nullPointMode": "null",
"options": {
"alertThreshold": true
},
"percentage": false,
"pluginVersion": "9.0.1",
"pointradius": 2,
"points": false,
"renderer": "flot",
"seriesOverrides": [],
"spaceLength": 10,
"stack": true,
"steppedLine": false,
"targets": [
{
"datasource": {
"uid": "$datasource"
},
"expr": "sum by (status_code) (irate(grafana_http_request_duration_seconds_count{job=~\"$job\", instance=~\"$instance\"}[1m])) ",
"interval": "",
"legendFormat": "{{status_code}}",
"refId": "A"
}
],
"thresholds": [],
"timeRegions": [],
"title": "RPS",
"tooltip": {
"shared": true,
"sort": 0,
"value_type": "individual"
},
"type": "graph",
"xaxis": {
"mode": "time",
"show": true,
"values": []
},
"yaxes": [
{
"$$hashKey": "object:157",
"format": "reqps",
"logBase": 1,
"show": true
},
{
"$$hashKey": "object:158",
"format": "short",
"logBase": 1,
"show": false
}
],
"yaxis": {
"align": false
}
},
{
"aliasColors": {},
"bars": false,
"dashLength": 10,
"dashes": false,
"datasource": {
"uid": "$datasource"
},
"fieldConfig": {
"defaults": {
"links": []
},
"overrides": []
},
"fill": 1,
"fillGradient": 0,
"gridPos": {
"h": 8,
"w": 12,
"x": 12,
"y": 5
},
"hiddenSeries": false,
"id": 4,
"legend": {
"avg": false,
"current": false,
"max": false,
"min": false,
"show": true,
"total": false,
"values": false
},
"lines": true,
"linewidth": 1,
"nullPointMode": "null",
"options": {
"alertThreshold": true
},
"percentage": false,
"pluginVersion": "9.0.1",
"pointradius": 2,
"points": false,
"renderer": "flot",
"seriesOverrides": [],
"spaceLength": 10,
"stack": false,
"steppedLine": false,
"targets": [
{
"datasource": {
"uid": "$datasource"
},
"exemplar": true,
"expr": "histogram_quantile(0.99, sum(irate(grafana_http_request_duration_seconds_bucket{instance=~\"$instance\", job=~\"$job\"}[$__rate_interval])) by (le)) * 1",
"interval": "",
"legendFormat": "99th Percentile",
"refId": "A"
},
{
"datasource": {
"uid": "$datasource"
},
"exemplar": true,
"expr": "histogram_quantile(0.50, sum(irate(grafana_http_request_duration_seconds_bucket{instance=~\"$instance\", job=~\"$job\"}[$__rate_interval])) by (le)) * 1",
"interval": "",
"legendFormat": "50th Percentile",
"refId": "B"
},
{
"datasource": {
"uid": "$datasource"
},
"exemplar": true,
"expr": "sum(irate(grafana_http_request_duration_seconds_sum{instance=~\"$instance\", job=~\"$job\"}[$__rate_interval])) * 1 / sum(irate(grafana_http_request_duration_seconds_count{instance=~\"$instance\", job=~\"$job\"}[$__rate_interval]))",
"interval": "",
"legendFormat": "Average",
"refId": "C"
}
],
"thresholds": [],
"timeRegions": [],
"title": "Request Latency",
"tooltip": {
"shared": true,
"sort": 0,
"value_type": "individual"
},
"type": "graph",
"xaxis": {
"mode": "time",
"show": true,
"values": []
},
"yaxes": [
{
"$$hashKey": "object:210",
"format": "ms",
"logBase": 1,
"show": true
},
{
"$$hashKey": "object:211",
"format": "short",
"logBase": 1,
"show": true
}
],
"yaxis": {
"align": false
}
}
],
"schemaVersion": 36,
"style": "dark",
"tags": [],
"templating": {
"list": [
{
"current": {
"selected": false,
"text": "Prometheus",
"value": "Prometheus"
},
"hide": 0,
"includeAll": false,
"multi": false,
"name": "datasource",
"options": [],
"query": "prometheus",
"queryValue": "",
"refresh": 1,
"regex": "",
"skipUrlSync": false,
"type": "datasource"
},
{
"allValue": ".*",
"current": {
"selected": false,
"text": "All",
"value": "$__all"
},
"datasource": {
"uid": "$datasource"
},
"definition": "label_values(grafana_build_info, job)",
"hide": 0,
"includeAll": true,
"multi": true,
"name": "job",
"options": [],
"query": {
"query": "label_values(grafana_build_info, job)",
"refId": "Billing Admin-job-Variable-Query"
},
"refresh": 1,
"regex": "",
"skipUrlSync": false,
"sort": 0,
"tagValuesQuery": "",
"tagsQuery": "",
"type": "query",
"useTags": false
},
{
"allValue": ".*",
"current": {
"selected": false,
"text": "All",
"value": "$__all"
},
"datasource": {
"uid": "$datasource"
},
"definition": "label_values(grafana_build_info, instance)",
"hide": 0,
"includeAll": true,
"multi": true,
"name": "instance",
"options": [],
"query": {
"query": "label_values(grafana_build_info, instance)",
"refId": "Billing Admin-instance-Variable-Query"
},
"refresh": 1,
"regex": "",
"skipUrlSync": false,
"sort": 0,
"tagValuesQuery": "",
"tagsQuery": "",
"type": "query",
"useTags": false
}
]
},
"time": {
"from": "now-6h",
"to": "now"
},
"timepicker": {
"refresh_intervals": [
"10s",
"30s",
"1m",
"5m",
"15m",
"30m",
"1h",
"2h",
"1d"
]
},
"timezone": "utc",
"title": "Grafana Overview",
"uid": "6be0s85Mk",
"version": 1,
"weekStart": ""
}
File diff suppressed because it is too large Load Diff
File diff suppressed because it is too large Load Diff
File diff suppressed because it is too large Load Diff
File diff suppressed because it is too large Load Diff
File diff suppressed because it is too large Load Diff
File diff suppressed because it is too large Load Diff
File diff suppressed because it is too large Load Diff
File diff suppressed because it is too large Load Diff
File diff suppressed because it is too large Load Diff
File diff suppressed because it is too large Load Diff
File diff suppressed because it is too large Load Diff
@@ -0,0 +1,550 @@
{
"annotations": {
"list": [
{
"builtIn": 1,
"datasource": {
"type": "grafana",
"uid": "-- Grafana --"
},
"enable": true,
"hide": true,
"iconColor": "rgba(0, 211, 255, 1)",
"name": "Annotations & Alerts",
"target": {
"limit": 100,
"matchAny": false,
"tags": [],
"type": "dashboard"
},
"type": "dashboard"
}
]
},
"editable": false,
"fiscalYearStartMonth": 0,
"graphTooltip": 0,
"id": 18,
"iteration": 1656212469966,
"links": [],
"liveNow": false,
"panels": [
{
"aliasColors": {},
"bars": false,
"dashLength": 10,
"dashes": false,
"datasource": {
"uid": "$datasource"
},
"fill": 1,
"fillGradient": 0,
"gridPos": {
"h": 7,
"w": 18,
"x": 0,
"y": 0
},
"hiddenSeries": false,
"id": 2,
"interval": "1m",
"legend": {
"alignAsTable": true,
"avg": true,
"current": true,
"max": true,
"min": true,
"rightSide": true,
"show": true,
"total": false,
"values": true
},
"lines": true,
"linewidth": 1,
"links": [],
"nullPointMode": "null",
"options": {
"alertThreshold": true
},
"percentage": false,
"pluginVersion": "9.0.1",
"pointradius": 5,
"points": false,
"renderer": "flot",
"seriesOverrides": [],
"spaceLength": 10,
"stack": true,
"steppedLine": false,
"targets": [
{
"datasource": {
"uid": "$datasource"
},
"expr": "(\n sum without(instance, node) (topk(1, (kubelet_volume_stats_capacity_bytes{cluster=\"$cluster\", job=\"kubelet\", metrics_path=\"/metrics\", namespace=\"$namespace\", persistentvolumeclaim=\"$volume\"})))\n -\n sum without(instance, node) (topk(1, (kubelet_volume_stats_available_bytes{cluster=\"$cluster\", job=\"kubelet\", metrics_path=\"/metrics\", namespace=\"$namespace\", persistentvolumeclaim=\"$volume\"})))\n)\n",
"format": "time_series",
"intervalFactor": 1,
"legendFormat": "Used Space",
"refId": "A"
},
{
"datasource": {
"uid": "$datasource"
},
"expr": "sum without(instance, node) (topk(1, (kubelet_volume_stats_available_bytes{cluster=\"$cluster\", job=\"kubelet\", metrics_path=\"/metrics\", namespace=\"$namespace\", persistentvolumeclaim=\"$volume\"})))\n",
"format": "time_series",
"intervalFactor": 1,
"legendFormat": "Free Space",
"refId": "B"
}
],
"thresholds": [],
"timeRegions": [],
"title": "Volume Space Usage",
"tooltip": {
"shared": false,
"sort": 0,
"value_type": "individual"
},
"type": "graph",
"xaxis": {
"mode": "time",
"show": true,
"values": []
},
"yaxes": [
{
"format": "bytes",
"logBase": 1,
"min": 0,
"show": true
},
{
"format": "bytes",
"logBase": 1,
"min": 0,
"show": true
}
],
"yaxis": {
"align": false
}
},
{
"datasource": {
"uid": "$datasource"
},
"fieldConfig": {
"defaults": {
"color": {
"mode": "thresholds"
},
"mappings": [
{
"options": {
"match": "null",
"result": {
"text": "N/A"
}
},
"type": "special"
}
],
"max": 100,
"min": 0,
"thresholds": {
"mode": "absolute",
"steps": [
{
"color": "rgba(50, 172, 45, 0.97)",
"value": null
},
{
"color": "rgba(237, 129, 40, 0.89)",
"value": 80
},
{
"color": "rgba(245, 54, 54, 0.9)",
"value": 90
}
]
},
"unit": "percent"
},
"overrides": []
},
"gridPos": {
"h": 7,
"w": 6,
"x": 18,
"y": 0
},
"id": 3,
"interval": "1m",
"links": [],
"maxDataPoints": 100,
"options": {
"orientation": "horizontal",
"reduceOptions": {
"calcs": [
"lastNotNull"
],
"fields": "",
"values": false
},
"showThresholdLabels": false,
"showThresholdMarkers": true
},
"pluginVersion": "9.0.1",
"targets": [
{
"datasource": {
"uid": "$datasource"
},
"expr": "max without(instance,node) (\n(\n topk(1, kubelet_volume_stats_capacity_bytes{cluster=\"$cluster\", job=\"kubelet\", metrics_path=\"/metrics\", namespace=\"$namespace\", persistentvolumeclaim=\"$volume\"})\n -\n topk(1, kubelet_volume_stats_available_bytes{cluster=\"$cluster\", job=\"kubelet\", metrics_path=\"/metrics\", namespace=\"$namespace\", persistentvolumeclaim=\"$volume\"})\n)\n/\ntopk(1, kubelet_volume_stats_capacity_bytes{cluster=\"$cluster\", job=\"kubelet\", metrics_path=\"/metrics\", namespace=\"$namespace\", persistentvolumeclaim=\"$volume\"})\n* 100)\n",
"format": "time_series",
"intervalFactor": 2,
"legendFormat": "",
"refId": "A"
}
],
"title": "Volume Space Usage",
"type": "gauge"
},
{
"aliasColors": {},
"bars": false,
"dashLength": 10,
"dashes": false,
"datasource": {
"uid": "$datasource"
},
"fill": 1,
"fillGradient": 0,
"gridPos": {
"h": 7,
"w": 18,
"x": 0,
"y": 7
},
"hiddenSeries": false,
"id": 4,
"interval": "1m",
"legend": {
"alignAsTable": true,
"avg": true,
"current": true,
"max": true,
"min": true,
"rightSide": true,
"show": true,
"total": false,
"values": true
},
"lines": true,
"linewidth": 1,
"links": [],
"nullPointMode": "null",
"options": {
"alertThreshold": true
},
"percentage": false,
"pluginVersion": "9.0.1",
"pointradius": 5,
"points": false,
"renderer": "flot",
"seriesOverrides": [],
"spaceLength": 10,
"stack": true,
"steppedLine": false,
"targets": [
{
"datasource": {
"uid": "$datasource"
},
"expr": "sum without(instance, node) (topk(1, (kubelet_volume_stats_inodes_used{cluster=\"$cluster\", job=\"kubelet\", metrics_path=\"/metrics\", namespace=\"$namespace\", persistentvolumeclaim=\"$volume\"})))\n",
"format": "time_series",
"intervalFactor": 1,
"legendFormat": "Used inodes",
"refId": "A"
},
{
"datasource": {
"uid": "$datasource"
},
"expr": "(\n sum without(instance, node) (topk(1, (kubelet_volume_stats_inodes{cluster=\"$cluster\", job=\"kubelet\", metrics_path=\"/metrics\", namespace=\"$namespace\", persistentvolumeclaim=\"$volume\"})))\n -\n sum without(instance, node) (topk(1, (kubelet_volume_stats_inodes_used{cluster=\"$cluster\", job=\"kubelet\", metrics_path=\"/metrics\", namespace=\"$namespace\", persistentvolumeclaim=\"$volume\"})))\n)\n",
"format": "time_series",
"intervalFactor": 1,
"legendFormat": " Free inodes",
"refId": "B"
}
],
"thresholds": [],
"timeRegions": [],
"title": "Volume inodes Usage",
"tooltip": {
"shared": false,
"sort": 0,
"value_type": "individual"
},
"type": "graph",
"xaxis": {
"mode": "time",
"show": true,
"values": []
},
"yaxes": [
{
"format": "none",
"logBase": 1,
"min": 0,
"show": true
},
{
"format": "none",
"logBase": 1,
"min": 0,
"show": true
}
],
"yaxis": {
"align": false
}
},
{
"datasource": {
"uid": "$datasource"
},
"fieldConfig": {
"defaults": {
"color": {
"mode": "thresholds"
},
"mappings": [
{
"options": {
"match": "null",
"result": {
"text": "N/A"
}
},
"type": "special"
}
],
"max": 100,
"min": 0,
"thresholds": {
"mode": "absolute",
"steps": [
{
"color": "rgba(50, 172, 45, 0.97)",
"value": null
},
{
"color": "rgba(237, 129, 40, 0.89)",
"value": 80
},
{
"color": "rgba(245, 54, 54, 0.9)",
"value": 90
}
]
},
"unit": "percent"
},
"overrides": []
},
"gridPos": {
"h": 7,
"w": 6,
"x": 18,
"y": 7
},
"id": 5,
"interval": "1m",
"links": [],
"maxDataPoints": 100,
"options": {
"orientation": "horizontal",
"reduceOptions": {
"calcs": [
"lastNotNull"
],
"fields": "",
"values": false
},
"showThresholdLabels": false,
"showThresholdMarkers": true
},
"pluginVersion": "9.0.1",
"targets": [
{
"datasource": {
"uid": "$datasource"
},
"expr": "max without(instance,node) (\ntopk(1, kubelet_volume_stats_inodes_used{cluster=\"$cluster\", job=\"kubelet\", metrics_path=\"/metrics\", namespace=\"$namespace\", persistentvolumeclaim=\"$volume\"})\n/\ntopk(1, kubelet_volume_stats_inodes{cluster=\"$cluster\", job=\"kubelet\", metrics_path=\"/metrics\", namespace=\"$namespace\", persistentvolumeclaim=\"$volume\"})\n* 100)\n",
"format": "time_series",
"intervalFactor": 2,
"legendFormat": "",
"refId": "A"
}
],
"title": "Volume inodes Usage",
"type": "gauge"
}
],
"refresh": "10s",
"schemaVersion": 36,
"style": "dark",
"tags": [
"kubernetes-mixin"
],
"templating": {
"list": [
{
"current": {
"selected": false,
"text": "default",
"value": "default"
},
"hide": 0,
"includeAll": false,
"label": "Data Source",
"multi": false,
"name": "datasource",
"options": [],
"query": "prometheus",
"refresh": 1,
"regex": "",
"skipUrlSync": false,
"type": "datasource"
},
{
"current": {
"isNone": true,
"selected": false,
"text": "None",
"value": ""
},
"datasource": {
"type": "prometheus",
"uid": "$datasource"
},
"definition": "",
"hide": 2,
"includeAll": false,
"label": "cluster",
"multi": false,
"name": "cluster",
"options": [],
"query": {
"query": "label_values(kubelet_volume_stats_capacity_bytes{job=\"kubelet\", metrics_path=\"/metrics\"}, cluster)",
"refId": "Prometheus-cluster-Variable-Query"
},
"refresh": 2,
"regex": "",
"skipUrlSync": false,
"sort": 1,
"tagValuesQuery": "",
"tagsQuery": "",
"type": "query",
"useTags": false
},
{
"current": {
"isNone": true,
"selected": false,
"text": "None",
"value": ""
},
"datasource": {
"type": "prometheus",
"uid": "$datasource"
},
"definition": "",
"hide": 0,
"includeAll": false,
"label": "Namespace",
"multi": false,
"name": "namespace",
"options": [],
"query": {
"query": "label_values(kubelet_volume_stats_capacity_bytes{cluster=\"$cluster\", job=\"kubelet\", metrics_path=\"/metrics\"}, namespace)",
"refId": "Prometheus-namespace-Variable-Query"
},
"refresh": 2,
"regex": "",
"skipUrlSync": false,
"sort": 1,
"tagValuesQuery": "",
"tagsQuery": "",
"type": "query",
"useTags": false
},
{
"current": {
"isNone": true,
"selected": false,
"text": "None",
"value": ""
},
"datasource": {
"type": "prometheus",
"uid": "$datasource"
},
"definition": "",
"hide": 0,
"includeAll": false,
"label": "PersistentVolumeClaim",
"multi": false,
"name": "volume",
"options": [],
"query": {
"query": "label_values(kubelet_volume_stats_capacity_bytes{cluster=\"$cluster\", job=\"kubelet\", metrics_path=\"/metrics\", namespace=\"$namespace\"}, persistentvolumeclaim)",
"refId": "Prometheus-volume-Variable-Query"
},
"refresh": 2,
"regex": "",
"skipUrlSync": false,
"sort": 1,
"tagValuesQuery": "",
"tagsQuery": "",
"type": "query",
"useTags": false
}
]
},
"time": {
"from": "now-7d",
"to": "now"
},
"timepicker": {
"refresh_intervals": [
"5s",
"10s",
"30s",
"1m",
"5m",
"15m",
"30m",
"1h",
"2h",
"1d"
],
"time_options": [
"5m",
"15m",
"1h",
"6h",
"12h",
"24h",
"2d",
"7d",
"30d"
]
},
"timezone": "utc",
"title": "Kubernetes / Persistent Volumes",
"uid": "919b92a8e8041bd567af9edab12c840c",
"version": 1,
"weekStart": ""
}
File diff suppressed because it is too large Load Diff
File diff suppressed because it is too large Load Diff
File diff suppressed because it is too large Load Diff
File diff suppressed because it is too large Load Diff
File diff suppressed because it is too large Load Diff
File diff suppressed because it is too large Load Diff
+31
View File
@@ -0,0 +1,31 @@
data "aws_partition" "current" {}
data "aws_caller_identity" "current" {}
data "aws_region" "current" {}
data "aws_eks_cluster" "eks_cluster" {
name = var.eks_cluster_id
}
locals {
name = "adot-collector-kubeprometheus"
namespace = try(var.config.helm_config.namespace, local.name)
eks_oidc_issuer_url = replace(data.aws_eks_cluster.eks_cluster.identity[0].oidc[0].issuer, "https://", "")
eks_cluster_endpoint = data.aws_eks_cluster.eks_cluster.endpoint
context = {
aws_caller_identity_account_id = data.aws_caller_identity.current.account_id
aws_caller_identity_arn = data.aws_caller_identity.current.arn
aws_eks_cluster_endpoint = local.eks_cluster_endpoint
aws_partition_id = data.aws_partition.current.partition
aws_region_name = data.aws_region.current.name
eks_cluster_id = var.eks_cluster_id
eks_oidc_issuer_url = local.eks_oidc_issuer_url
eks_oidc_provider_arn = "arn:${data.aws_partition.current.partition}:iam::${data.aws_caller_identity.current.account_id}:oidc-provider/${local.eks_oidc_issuer_url}"
tags = var.tags
irsa_iam_role_path = var.irsa_iam_role_path
irsa_iam_permissions_boundary = var.irsa_iam_permissions_boundary
}
}
+100
View File
@@ -0,0 +1,100 @@
terraform {
required_providers {
grafana = {
source = "grafana/grafana"
version = "1.25.0"
}
}
}
resource "helm_release" "kube_state_metrics" {
count = var.config.enable_kube_state_metrics ? 1 : 0
chart = var.config.ksm_helm_chart_name
create_namespace = var.config.kms_create_namespace
namespace = var.config.ksm_k8s_namespace
name = var.config.ksm_helm_release_name
version = var.config.ksm_helm_chart_version
repository = var.config.ksm_helm_repo_url
dynamic "set" {
for_each = var.config.ksm_helm_settings
content {
name = set.key
value = set.value
}
}
}
resource "helm_release" "prometheus_node_exporter" {
count = var.config.enable_node_exporter ? 1 : 0
chart = var.config.ne_helm_chart_name
create_namespace = var.config.ne_create_namespace
namespace = var.config.ne_k8s_namespace
name = var.config.ne_helm_release_name
version = var.config.ne_helm_chart_version
repository = var.config.ne_helm_repo_url
dynamic "set" {
for_each = var.config.ne_helm_settings
content {
name = set.key
value = set.value
}
}
}
module "helm_addon" {
source = "github.com/aws-ia/terraform-aws-eks-blueprints/modules/kubernetes-addons/helm-addon"
helm_config = merge(
{
name = local.name
chart = "${path.module}/otel-config"
version = "0.2.0"
namespace = local.namespace
description = "ADOT helm Chart deployment configuration"
},
var.helm_config
)
set_values = [
{
name = "ampurl"
value = "${var.managed_prometheus_workspace_endpoint}api/v1/remote_write"
},
{
name = "region"
value = var.managed_prometheus_workspace_region
},
{
name = "prometheusMetricsEndpoint"
value = "metrics"
},
{
name = "prometheusMetricsPort"
value = 8888
},
{
name = "scrapeInterval"
value = "15s"
},
{
name = "scrapeTimeout"
value = "10s"
},
{
name = "scrapeSampleLimit"
value = 1000
}
]
irsa_config = {
create_kubernetes_namespace = true
kubernetes_namespace = local.namespace
create_kubernetes_service_account = true
kubernetes_service_account = try(var.config.helm_config.service_account, local.name)
irsa_iam_policies = ["arn:${data.aws_partition.current.partition}:iam::aws:policy/AmazonPrometheusRemoteWriteAccess"]
}
addon_context = local.context
}
@@ -0,0 +1,6 @@
apiVersion: v2
name: opentelemetry
description: A Helm chart to install otel operator
type: application
version: 0.2.0
appVersion: v0.1.0
@@ -0,0 +1,29 @@
apiVersion: rbac.authorization.k8s.io/v1
kind: ClusterRole
metadata:
name: otel-prometheus-role
rules:
- apiGroups:
- ""
resources:
- nodes
- nodes/proxy
- services
- endpoints
- pods
verbs:
- get
- list
- watch
- apiGroups:
- extensions
resources:
- ingresses
verbs:
- get
- list
- watch
- nonResourceURLs:
- /metrics
verbs:
- get
@@ -0,0 +1,12 @@
apiVersion: rbac.authorization.k8s.io/v1
kind: ClusterRoleBinding
metadata:
name: otel-prometheus-role-binding
roleRef:
apiGroup: rbac.authorization.k8s.io
kind: ClusterRole
name: otel-prometheus-role
subjects:
- kind: ServiceAccount
name: adot-collector-kubeprometheus
namespace: adot-collector-kubeprometheus
File diff suppressed because it is too large Load Diff
@@ -0,0 +1,7 @@
ampurl: ${amp_url}
region: ${region}
prometheusMetricsEndpoint: ${prometheus_metrics_endpoint}
prometheusMetricsPort: ${prometheus_metrics_port}
scrapeInterval: ${scrape_interval}
scrapeTimeout: ${scrape_timeout}
scrapeSampleLimit: ${scrape_sample_limit}
View File
File diff suppressed because it is too large Load Diff
+107
View File
@@ -0,0 +1,107 @@
variable "eks_cluster_id" {
description = "EKS Cluster Id"
type = string
}
variable "helm_config" {
description = "Helm Config for Prometheus"
type = any
default = {}
}
variable "irsa_iam_role_path" {
description = "IAM role path for IRSA roles"
type = string
default = "/"
}
variable "irsa_iam_permissions_boundary" {
description = "IAM permissions boundary for IRSA roles"
type = string
default = ""
}
variable "managed_prometheus_workspace_endpoint" {
description = "Amazon Managed Prometheus Workspace Endpoint"
type = string
default = null
}
variable "managed_prometheus_workspace_id" {
description = "Amazon Managed Prometheus Workspace ID"
type = string
default = null
}
variable "managed_prometheus_workspace_region" {
description = "Amazon Managed Prometheus Workspace's Region"
type = string
default = null
}
variable "dashboards_folder_id" {
type = string
}
variable "config" {
type = object({
helm_config = map(any)
enable_kube_state_metrics = bool
kms_create_namespace = bool
ksm_k8s_namespace = string
ksm_helm_chart_name = string
ksm_helm_chart_version = string
ksm_helm_release_name = string
ksm_helm_repo_url = string
ksm_helm_settings = map(string)
ksm_helm_values = map(any)
enable_node_exporter = bool
ne_create_namespace = bool
ne_k8s_namespace = string
ne_helm_chart_name = string
ne_helm_chart_version = string
ne_helm_release_name = string
ne_helm_repo_url = string
ne_helm_settings = map(string)
ne_helm_values = map(any)
enable_dashboards = bool
enable_recording_rules = bool
})
default = {
enable_kube_state_metrics = true
enable_node_exporter = true
enable_dashboards = true
enable_recording_rules = true
helm_config = {}
kms_create_namespace = true
ksm_helm_chart_name = "kube-state-metrics"
ksm_helm_chart_version = "4.9.2"
ksm_helm_release_name = "kube-state-metrics"
ksm_helm_repo_url = "https://prometheus-community.github.io/helm-charts"
ksm_helm_settings = {}
ksm_helm_values = {}
ksm_k8s_namespace = "kube-system"
ne_create_namespace = true
ne_k8s_namespace = "prometheus-node-exporter"
ne_helm_chart_name = "prometheus-node-exporter"
ne_helm_chart_version = "2.0.3"
ne_helm_release_name = "prometheus-node-exporter"
ne_helm_repo_url = "https://prometheus-community.github.io/helm-charts"
ne_helm_settings = {}
ne_helm_values = {}
}
nullable = false
}
variable "tags" {
description = "Additional tags (e.g. `map('BusinessUnit`,`XYZ`)"
type = map(string)
default = {}
}
File diff suppressed because it is too large Load Diff
+103
View File
@@ -0,0 +1,103 @@
locals {
name = "adot-collector-java"
namespace = try(var.helm_config.namespace, local.name)
}
terraform {
required_providers {
grafana = {
source = "grafana/grafana"
version = "1.25.0"
}
}
}
data "aws_partition" "current" {}
# deploys collector
module "helm_addon" {
source = "github.com/aws-ia/terraform-aws-eks-blueprints/modules/kubernetes-addons/helm-addon"
helm_config = merge(
{
name = local.name
chart = "${path.module}/otel-config"
version = "0.2.0"
namespace = local.namespace
description = "ADOT helm Chart deployment configuration"
},
var.helm_config
)
set_values = [
{
name = "ampurl"
value = "${var.amp_endpoint}api/v1/remote_write"
},
{
name = "region"
value = var.amp_region
},
{
name = "prometheusMetricsEndpoint"
value = "metrics"
},
{
name = "prometheusMetricsPort"
value = 8888
},
{
name = "scrapeInterval"
value = "15s"
},
{
name = "scrapeTimeout"
value = "10s"
},
{
name = "scrapeSampleLimit"
value = 1000
}
]
irsa_config = {
create_kubernetes_namespace = try(var.helm_config["create_namespace"], true)
kubernetes_namespace = local.namespace
create_kubernetes_service_account = true
kubernetes_service_account = try(var.helm_config.service_account, local.name)
irsa_iam_policies = ["arn:${data.aws_partition.current.partition}:iam::aws:policy/AmazonPrometheusRemoteWriteAccess"]
}
addon_context = var.addon_context
}
resource "aws_prometheus_rule_group_namespace" "this" {
count = var.enable_recording_rules ? 1 : 0
name = "java_rules"
workspace_id = var.amp_id
data = <<EOF
groups:
- name: default-metric
rules:
- record: metric:recording_rule
expr: avg(rate(container_cpu_usage_seconds_total[5m]))
- name: default-alert
rules:
- alert: metric:alerting_rule
expr: jvm_memory_bytes_used{job="java", area="heap"} / jvm_memory_bytes_max * 100 > 80
for: 1m
labels:
severity: warning
annotations:
summary: "JVM heap warning"
description: "JVM heap of instance `{{$labels.instance}}` from application `{{$labels.application}}` is above 80% for one minute. (current=`{{$value}}%`)"
EOF
}
resource "grafana_dashboard" "this" {
folder = var.dashboards_folder_id
config_json = file("${path.module}/dashboards/default.json")
}
@@ -0,0 +1,6 @@
apiVersion: v2
name: opentelemetry
description: A Helm chart to install otel operator
type: application
version: 0.2.0
appVersion: v0.1.0
@@ -0,0 +1,29 @@
apiVersion: rbac.authorization.k8s.io/v1
kind: ClusterRole
metadata:
name: otel-prometheus-role
rules:
- apiGroups:
- ""
resources:
- nodes
- nodes/proxy
- services
- endpoints
- pods
verbs:
- get
- list
- watch
- apiGroups:
- extensions
resources:
- ingresses
verbs:
- get
- list
- watch
- nonResourceURLs:
- /metrics
verbs:
- get
@@ -0,0 +1,12 @@
apiVersion: rbac.authorization.k8s.io/v1
kind: ClusterRoleBinding
metadata:
name: otel-prometheus-role-binding
roleRef:
apiGroup: rbac.authorization.k8s.io
kind: ClusterRole
name: otel-prometheus-role
subjects:
- kind: ServiceAccount
name: adot-collector-java
namespace: adot-collector-java
@@ -0,0 +1,67 @@
apiVersion: opentelemetry.io/v1alpha1
kind: OpenTelemetryCollector
metadata:
name: adot
spec:
image: public.ecr.aws/aws-observability/aws-otel-collector:latest
mode: deployment
serviceAccount: adot-collector-java
config: |
receivers:
prometheus:
config:
global:
scrape_interval: {{ .Values.scrapeInterval }}
scrape_timeout: {{ .Values.scrapeTimeout }}
scrape_configs:
- job_name: 'kubernetes-pod-jmx'
sample_limit: {{ .Values.scrapeSampleLimit }}
metrics_path: /{{ .Values.prometheusMetricsEndpoint }}
kubernetes_sd_configs:
- role: pod
relabel_configs:
- source_labels: [ __address__ ]
action: keep
regex: '.*:9404$'
- action: labelmap
regex: __meta_kubernetes_pod_label_(.+)
- action: replace
source_labels: [ __meta_kubernetes_namespace ]
target_label: Namespace
- source_labels: [ __meta_kubernetes_pod_name ]
action: replace
target_label: pod_name
- action: replace
source_labels: [ __meta_kubernetes_pod_container_name ]
target_label: container_name
- action: replace
source_labels: [ __meta_kubernetes_pod_controller_kind ]
target_label: pod_controller_kind
- action: replace
source_labels: [ __meta_kubernetes_pod_phase ]
target_label: pod_controller_phase
metric_relabel_configs:
- source_labels: [ __name__ ]
regex: 'jvm_gc_collection_seconds.*'
action: drop
exporters:
awsprometheusremotewrite:
endpoint: {{ .Values.ampurl }}
aws_auth:
region: {{ .Values.region }}
service: "aps"
logging:
loglevel: info
extensions:
health_check:
pprof:
endpoint: :1888
zpages:
endpoint: :55679
service:
extensions: [pprof, zpages, health_check]
pipelines:
metrics:
receivers: [prometheus]
exporters: [logging, awsprometheusremotewrite]
@@ -0,0 +1,7 @@
ampurl: ${amp_url}
region: ${region}
prometheusMetricsEndpoint: ${prometheus_metrics_endpoint}
prometheusMetricsPort: ${prometheus_metrics_port}
scrapeInterval: ${scrape_interval}
scrapeTimeout: ${scrape_timeout}
scrapeSampleLimit: ${scrape_sample_limit}
+48
View File
@@ -0,0 +1,48 @@
variable "enable_recording_rules" {
description = "Enable AMP recording rules"
type = bool
default = true
}
variable "amp_endpoint" {
description = "Amazon Managed Prometheus endpoint"
type = string
}
variable "amp_id" {
description = "Managed Prometheus workspace id"
type = string
}
variable "helm_config" {
description = "Helm Config for Prometheus"
type = any
default = {}
}
variable "amp_region" {
description = "Amazon Managed Prometheus Workspace's Region"
type = string
default = null
}
variable "dashboards_folder_id" {
type = string
}
variable "addon_context" {
description = "Input configuration for the addon"
type = object({
aws_caller_identity_account_id = string
aws_caller_identity_arn = string
aws_eks_cluster_endpoint = string
aws_partition_id = string
aws_region_name = string
eks_cluster_id = string
eks_oidc_issuer_url = string
eks_oidc_provider_arn = string
irsa_iam_permissions_boundary = string
irsa_iam_role_path = string
tags = map(string)
})
}