Split core and workloads modules

This commit is contained in:
Rodrigue Koffi
2022-08-09 12:35:14 +02:00
parent cd827f6fb9
commit 5b5d5e7dd1
59 changed files with 119 additions and 191 deletions
-172
View File
@@ -1,172 +0,0 @@
#---------------------------------------------------------------
# Observability Resources
#---------------------------------------------------------------
module "managed_grafana" {
source = "terraform-aws-modules/managed-service-grafana/aws"
version = "~> 1.3"
# Workspace
name = local.name
stack_set_name = local.name
data_sources = ["PROMETHEUS"]
associate_license = false
# # Role associations
# Pending https://github.com/hashicorp/terraform-provider-aws/issues/24166
# role_associations = {
# "ADMIN" = {
# "group_ids" = []
# "user_ids" = []
# }
# "EDITOR" = {
# "group_ids" = []
# "user_ids" = []
# }
# }
tags = local.tags
}
resource "grafana_data_source" "prometheus" {
type = "prometheus"
name = "amp"
is_default = true
url = module.managed_prometheus.workspace_prometheus_endpoint
json_data {
http_method = "GET"
sigv4_auth = true
sigv4_auth_type = "workspace-iam-role"
sigv4_region = local.region
}
}
resource "grafana_folder" "this" {
title = "Observability"
}
resource "grafana_dashboard" "this" {
folder = grafana_folder.this.id
config_json = file("${path.module}/dashboards/default.json")
}
module "managed_prometheus" {
source = "terraform-aws-modules/managed-service-prometheus/aws"
version = "~> 2.1"
workspace_alias = local.name
alert_manager_definition = <<-EOT
alertmanager_config: |
route:
receiver: 'default'
receivers:
- name: 'default'
EOT
rule_group_namespaces = {
haproxy = {
name = "haproxy_rules"
data = <<-EOT
groups:
- name: obsa-haproxy-down-alert
rules:
- alert: HA_proxy_down
expr: haproxy_up == 0
for: 0m
labels:
severity: critical
annotations:
summary: HAProxy down (instance {{ $labels.instance }})
description: "HAProxy down\n VALUE = {{ $value }}\n LABELS = {{ $labels }}"
- name: obsa-haproxy-http4xx-error-alert
rules:
- alert: Ha_proxy_High_Http4xx_ErrorRate_Backend
expr: sum by (backend) (rate(haproxy_server_http_responses_total{code="4xx"}[1m])) / sum by (backend) (rate(haproxy_server_http_responses_total[1m]) * 100) > 5
for: 1m
labels:
severity: critical
annotations:
summary: HAProxy high HTTP 4xx error rate backend (instance {{ $labels.instance }})
description: "Too many HTTP requests with status 4xx (> 5%) on backend {{ $labels.fqdn }}/{{ $labels.backend }}\n VALUE = {{ $value }}\n LABELS = {{ $labels }}"
- name: obsa-haproxy-http5xx-error-alert
rules:
- alert: Ha_proxy_High_Http5xx_ErrorRate_Backend
expr: sum by (backend) (rate(haproxy_server_http_responses_total{code="5xx"}[1m])) / sum by (backend) (rate(haproxy_server_http_responses_total[1m]) * 100) > 5
for: 1m
labels:
severity: critical
annotations:
summary: HAProxy high HTTP 5xx error rate backend (instance {{ $labels.instance }})
description: "Too many HTTP requests with status 5xx (> 5%) on backend {{ $labels.fqdn }}/{{ $labels.backend }}\n VALUE = {{ $value }}\n LABELS = {{ $labels }}"
- name: obsa-haproxy-Http4xx-ErrorRate-Server-alert
rules:
- alert: Ha_proxy_High_Http4xx_ErrorRate_Server
expr: sum by (server) (rate(haproxy_server_http_responses_total{code="4xx"}[1m])) / sum by (server) (rate(haproxy_server_http_responses_total[1m]) * 100) > 5
for: 1m
labels:
severity: critical
annotations:
summary: HAProxy high HTTP 4xx error rate server (instance {{ $labels.instance }})
description: "Too many HTTP requests with status 4xx (> 5%) on server {{ $labels.server }}\n VALUE = {{ $value }}\n LABELS = {{ $labels }}"
- name: obsa-haproxy-Http5xx-ErrorRate-Server-alert
rules:
- alert: Ha_proxy_High_Http5xx_ErrorRate_Server
expr: sum by (server) (rate(haproxy_server_http_responses_total{code="5xx"}[1m])) / sum by (server) (rate(haproxy_server_http_responses_total[1m]) * 100) > 5
for: 1m
labels:
severity: critical
annotations:
summary: HAProxy high HTTP 5xx error rate server (instance {{ $labels.instance }})
description: "Too many HTTP requests with status 5xx (> 5%) on server {{ $labels.server }}\n VALUE = {{ $value }}\n LABELS = {{ $labels }}"
EOT
}
}
tags = local.tags
}
#---------------------------------------------------------------
# Sample Application
#---------------------------------------------------------------
# https://github.com/haproxy-ingress/charts/tree/master/haproxy-ingress
resource "helm_release" "haproxy_ingress" {
namespace = "haproxy-ingress"
create_namespace = true
name = "haproxy-ingress"
repository = "https://haproxy-ingress.github.io/charts"
chart = "haproxy-ingress"
version = "0.13.7"
set {
name = "defaultBackend.enabled"
value = true
}
set {
name = "controller.stats.enabled"
value = true
}
set {
name = "controller.metrics.enabled"
value = true
}
set {
name = "controller.metrics.service.annotations.prometheus\\.io/port"
value = 9101
type = "string"
}
set {
name = "controller.metrics.service.annotations.prometheus\\.io/scrape"
value = true
type = "string"
}
}
File diff suppressed because it is too large Load Diff
-104
View File
@@ -1,104 +0,0 @@
locals {
name = "adot-collector-java"
namespace = try(var.helm_config.namespace, local.name)
}
terraform {
required_providers {
grafana = {
source = "grafana/grafana"
version = "1.24.0"
}
}
}
data "aws_partition" "current" {}
# deploys collector
module "helm_addon" {
source = "github.com/aws-ia/terraform-aws-eks-blueprints/modules/kubernetes-addons/helm-addon"
helm_config = merge(
{
name = local.name
chart = "${path.module}/otel-config"
version = "0.2.0"
namespace = local.namespace
description = "ADOT helm Chart deployment configuration"
},
var.helm_config
)
set_values = [
{
name = "ampurl"
value = "${var.amp_endpoint}api/v1/remote_write"
},
{
name = "region"
value = var.amp_region
},
{
name = "prometheusMetricsEndpoint"
value = "metrics"
},
{
name = "prometheusMetricsPort"
value = 8888
},
{
name = "scrapeInterval"
value = "15s"
},
{
name = "scrapeTimeout"
value = "10s"
},
{
name = "scrapeSampleLimit"
value = 1000
}
]
irsa_config = {
create_kubernetes_namespace = try(var.helm_config["create_namespace"], true)
kubernetes_namespace = local.namespace
create_kubernetes_service_account = true
kubernetes_service_account = try(var.helm_config.service_account, local.name)
irsa_iam_policies = ["arn:${data.aws_partition.current.partition}:iam::aws:policy/AmazonPrometheusRemoteWriteAccess"]
}
addon_context = var.addon_context
}
resource "aws_prometheus_rule_group_namespace" "this" {
count = var.enable_recording_rules ? 1 : 0
name = "java_rules"
workspace_id = var.amp_id
data = <<EOF
groups:
- name: default-metric
rules:
- record: metric:recording_rule
expr: avg(rate(container_cpu_usage_seconds_total[5m]))
- name: default-alert
rules:
- alert: metric:alerting_rule
expr: jvm_memory_bytes_used{job="java", area="heap"} / jvm_memory_bytes_max * 100 > 80
for: 1m
labels:
severity: warning
annotations:
summary: "JVM heap warning"
description: "JVM heap of instance `{{$labels.instance}}` from application `{{$labels.application}}` is above 80% for one minute. (current=`{{$value}}%`)"
EOF
}
resource "grafana_dashboard" "this" {
folder = var.dashboards_folder_id
config_json = file("${path.module}/dashboards/default.json")
}
@@ -1,6 +0,0 @@
apiVersion: v2
name: opentelemetry
description: A Helm chart to install otel operator
type: application
version: 0.2.0
appVersion: v0.1.0
@@ -1,29 +0,0 @@
apiVersion: rbac.authorization.k8s.io/v1
kind: ClusterRole
metadata:
name: otel-prometheus-role
rules:
- apiGroups:
- ""
resources:
- nodes
- nodes/proxy
- services
- endpoints
- pods
verbs:
- get
- list
- watch
- apiGroups:
- extensions
resources:
- ingresses
verbs:
- get
- list
- watch
- nonResourceURLs:
- /metrics
verbs:
- get
@@ -1,12 +0,0 @@
apiVersion: rbac.authorization.k8s.io/v1
kind: ClusterRoleBinding
metadata:
name: otel-prometheus-role-binding
roleRef:
apiGroup: rbac.authorization.k8s.io
kind: ClusterRole
name: otel-prometheus-role
subjects:
- kind: ServiceAccount
name: adot-collector-java
namespace: adot-collector-java
@@ -1,67 +0,0 @@
apiVersion: opentelemetry.io/v1alpha1
kind: OpenTelemetryCollector
metadata:
name: adot
spec:
image: public.ecr.aws/aws-observability/aws-otel-collector:latest
mode: deployment
serviceAccount: adot-collector-java
config: |
receivers:
prometheus:
config:
global:
scrape_interval: {{ .Values.scrapeInterval }}
scrape_timeout: {{ .Values.scrapeTimeout }}
scrape_configs:
- job_name: 'kubernetes-pod-jmx'
sample_limit: {{ .Values.scrapeSampleLimit }}
metrics_path: /{{ .Values.prometheusMetricsEndpoint }}
kubernetes_sd_configs:
- role: pod
relabel_configs:
- source_labels: [ __address__ ]
action: keep
regex: '.*:9404$'
- action: labelmap
regex: __meta_kubernetes_pod_label_(.+)
- action: replace
source_labels: [ __meta_kubernetes_namespace ]
target_label: Namespace
- source_labels: [ __meta_kubernetes_pod_name ]
action: replace
target_label: pod_name
- action: replace
source_labels: [ __meta_kubernetes_pod_container_name ]
target_label: container_name
- action: replace
source_labels: [ __meta_kubernetes_pod_controller_kind ]
target_label: pod_controller_kind
- action: replace
source_labels: [ __meta_kubernetes_pod_phase ]
target_label: pod_controller_phase
metric_relabel_configs:
- source_labels: [ __name__ ]
regex: 'jvm_gc_collection_seconds.*'
action: drop
exporters:
awsprometheusremotewrite:
endpoint: {{ .Values.ampurl }}
aws_auth:
region: {{ .Values.region }}
service: "aps"
logging:
loglevel: info
extensions:
health_check:
pprof:
endpoint: :1888
zpages:
endpoint: :55679
service:
extensions: [pprof, zpages, health_check]
pipelines:
metrics:
receivers: [prometheus]
exporters: [logging, awsprometheusremotewrite]
@@ -1,7 +0,0 @@
ampurl: ${amp_url}
region: ${region}
prometheusMetricsEndpoint: ${prometheus_metrics_endpoint}
prometheusMetricsPort: ${prometheus_metrics_port}
scrapeInterval: ${scrape_interval}
scrapeTimeout: ${scrape_timeout}
scrapeSampleLimit: ${scrape_sample_limit}
-48
View File
@@ -1,48 +0,0 @@
variable "enable_recording_rules" {
description = "Enable AMP recording rules"
type = bool
default = true
}
variable "amp_endpoint" {
description = "Amazon Managed Prometheus endpoint"
type = string
}
variable "amp_id" {
description = "Managed Prometheus workspace id"
type = string
}
variable "helm_config" {
description = "Helm Config for Prometheus"
type = any
default = {}
}
variable "amp_region" {
description = "Amazon Managed Prometheus Workspace's Region"
type = string
default = null
}
variable "dashboards_folder_id" {
type = string
}
variable "addon_context" {
description = "Input configuration for the addon"
type = object({
aws_caller_identity_account_id = string
aws_caller_identity_arn = string
aws_eks_cluster_endpoint = string
aws_partition_id = string
aws_region_name = string
eks_cluster_id = string
eks_oidc_issuer_url = string
eks_oidc_provider_arn = string
irsa_iam_permissions_boundary = string
irsa_iam_role_path = string
tags = map(string)
})
}
@@ -1,49 +0,0 @@
# Observability Pattern for Java/JMX
This module provides an automated experience around Observability for Nginx workloads.
It provides the following resources:
- AWS Distro For OpenTelemetry Operator and Collector
- AWS Managed Grafana Dashboard and data source
- Alerts and recording rules with AWS Managed Service for Prometheus
<!-- BEGINNING OF PRE-COMMIT-TERRAFORM DOCS HOOK -->
## Requirements
| Name | Version |
|------|---------|
| <a name="requirement_terraform"></a> [terraform](#requirement\_terraform) | >= 1.0.0 |
| <a name="requirement_aws"></a> [aws](#requirement\_aws) | >= 3.72 |
| <a name="requirement_kubernetes"></a> [kubernetes](#requirement\_kubernetes) | >= 2.10 |
## Providers
| Name | Version |
|------|---------|
| <a name="provider_aws"></a> [aws](#provider\_aws) | >= 3.72 |
## Modules
| Name | Source | Version |
|------|--------|---------|
| <a name="module_helm_addon"></a> [helm\_addon](#module\_helm\_addon) | ../helm-addon | n/a |
## Resources
| Name | Type |
|------|------|
| [aws_partition.current](https://registry.terraform.io/providers/hashicorp/aws/latest/docs/data-sources/partition) | data source |
## Inputs
| Name | Description | Type | Default | Required |
|------|-------------|------|---------|:--------:|
| <a name="input_addon_context"></a> [addon\_context](#input\_addon\_context) | Input configuration for the addon | <pre>object({<br> aws_caller_identity_account_id = string<br> aws_caller_identity_arn = string<br> aws_eks_cluster_endpoint = string<br> aws_partition_id = string<br> aws_region_name = string<br> eks_cluster_id = string<br> eks_oidc_issuer_url = string<br> eks_oidc_provider_arn = string<br> irsa_iam_permissions_boundary = string<br> irsa_iam_role_path = string<br> tags = map(string)<br> })</pre> | n/a | yes |
| <a name="input_amazon_prometheus_workspace_endpoint"></a> [amazon\_prometheus\_workspace\_endpoint](#input\_amazon\_prometheus\_workspace\_endpoint) | Amazon Managed Prometheus Workspace Endpoint | `string` | `null` | no |
| <a name="input_amazon_prometheus_workspace_region"></a> [amazon\_prometheus\_workspace\_region](#input\_amazon\_prometheus\_workspace\_region) | Amazon Managed Prometheus Workspace's Region | `string` | `null` | no |
| <a name="input_helm_config"></a> [helm\_config](#input\_helm\_config) | Helm Config for Prometheus | `any` | `{}` | no |
## Outputs
No outputs.
<!-- END OF PRE-COMMIT-TERRAFORM DOCS HOOK -->
@@ -1,155 +0,0 @@
resource "grafana_dashboard" "alertmanager" {
count = var.config.enable_dashboards ? 1 : 0
folder = var.dashboards_folder_id
config_json = file("${path.module}/dashboards/alertmanager.json")
}
resource "grafana_dashboard" "workloads" {
count = var.config.enable_dashboards ? 1 : 0
folder = var.dashboards_folder_id
config_json = file("${path.module}/dashboards/workloads.json")
}
resource "grafana_dashboard" "scheduler" {
count = var.config.enable_dashboards ? 1 : 0
folder = var.dashboards_folder_id
config_json = file("${path.module}/dashboards/scheduler.json")
}
resource "grafana_dashboard" "proxy" {
count = var.config.enable_dashboards ? 1 : 0
folder = var.dashboards_folder_id
config_json = file("${path.module}/dashboards/proxy.json")
}
resource "grafana_dashboard" "prometheus" {
count = var.config.enable_dashboards ? 1 : 0
folder = var.dashboards_folder_id
config_json = file("${path.module}/dashboards/prometheus.json")
}
resource "grafana_dashboard" "podnetwork" {
count = var.config.enable_dashboards ? 1 : 0
folder = var.dashboards_folder_id
config_json = file("${path.module}/dashboards/pods-networking.json")
}
resource "grafana_dashboard" "pods" {
count = var.config.enable_dashboards ? 1 : 0
folder = var.dashboards_folder_id
config_json = file("${path.module}/dashboards/pods.json")
}
resource "grafana_dashboard" "pv" {
count = var.config.enable_dashboards ? 1 : 0
folder = var.dashboards_folder_id
config_json = file("${path.module}/dashboards/pesistentvolumes.json")
}
resource "grafana_dashboard" "nodes" {
count = var.config.enable_dashboards ? 1 : 0
folder = var.dashboards_folder_id
config_json = file("${path.module}/dashboards/nodes.json")
}
resource "grafana_dashboard" "necluster" {
count = var.config.enable_dashboards ? 1 : 0
folder = var.dashboards_folder_id
config_json = file("${path.module}/dashboards/nodeexpoter-use-cluster.json")
}
resource "grafana_dashboard" "nenodeuse" {
count = var.config.enable_dashboards ? 1 : 0
folder = var.dashboards_folder_id
config_json = file("${path.module}/dashboards/nodeexporter-use-node.json")
}
resource "grafana_dashboard" "nenode" {
count = var.config.enable_dashboards ? 1 : 0
folder = var.dashboards_folder_id
config_json = file("${path.module}/dashboards/nodeexporter-nodes.json")
}
resource "grafana_dashboard" "nwworload" {
count = var.config.enable_dashboards ? 1 : 0
folder = var.dashboards_folder_id
config_json = file("${path.module}/dashboards/networking-workloads.json")
}
resource "grafana_dashboard" "nsworkload" {
count = var.config.enable_dashboards ? 1 : 0
folder = var.dashboards_folder_id
config_json = file("${path.module}/dashboards/namespace-workloads.json")
}
resource "grafana_dashboard" "nspods" {
count = var.config.enable_dashboards ? 1 : 0
folder = var.dashboards_folder_id
config_json = file("${path.module}/dashboards/namespace-pods.json")
}
resource "grafana_dashboard" "nsnwworkload" {
count = var.config.enable_dashboards ? 1 : 0
folder = var.dashboards_folder_id
config_json = file("${path.module}/dashboards/namespace-nw-workloads.json")
}
resource "grafana_dashboard" "nsnw" {
count = var.config.enable_dashboards ? 1 : 0
folder = var.dashboards_folder_id
config_json = file("${path.module}/dashboards/namespace-networking.json")
}
resource "grafana_dashboard" "macos" {
count = var.config.enable_dashboards ? 1 : 0
folder = var.dashboards_folder_id
config_json = file("${path.module}/dashboards/macos.json")
}
resource "grafana_dashboard" "kubelet" {
count = var.config.enable_dashboards ? 1 : 0
folder = var.dashboards_folder_id
config_json = file("${path.module}/dashboards/kubelet.json")
}
resource "grafana_dashboard" "grafana" {
count = var.config.enable_dashboards ? 1 : 0
folder = var.dashboards_folder_id
config_json = file("${path.module}/dashboards/grafana.json")
}
resource "grafana_dashboard" "etcd" {
count = var.config.enable_dashboards ? 1 : 0
folder = var.dashboards_folder_id
config_json = file("${path.module}/dashboards/etcd.json")
}
resource "grafana_dashboard" "coredns" {
count = var.config.enable_dashboards ? 1 : 0
folder = var.dashboards_folder_id
config_json = file("${path.module}/dashboards/coredns.json")
}
resource "grafana_dashboard" "controller" {
count = var.config.enable_dashboards ? 1 : 0
folder = var.dashboards_folder_id
config_json = file("${path.module}/dashboards/controller.json")
}
resource "grafana_dashboard" "clusternw" {
count = var.config.enable_dashboards ? 1 : 0
folder = var.dashboards_folder_id
config_json = file("${path.module}/dashboards/cluster-networking.json")
}
resource "grafana_dashboard" "cluster" {
count = var.config.enable_dashboards ? 1 : 0
folder = var.dashboards_folder_id
config_json = file("${path.module}/dashboards/cluster.json")
}
resource "grafana_dashboard" "apis" {
count = var.config.enable_dashboards ? 1 : 0
folder = var.dashboards_folder_id
config_json = file("${path.module}/dashboards/apiserver.json")
}
@@ -1,626 +0,0 @@
{
"annotations": {
"list": [
{
"builtIn": 1,
"datasource": {
"type": "grafana",
"uid": "-- Grafana --"
},
"enable": true,
"hide": true,
"iconColor": "rgba(0, 211, 255, 1)",
"name": "Annotations & Alerts",
"target": {
"limit": 100,
"matchAny": false,
"tags": [],
"type": "dashboard"
},
"type": "dashboard"
}
]
},
"editable": false,
"fiscalYearStartMonth": 0,
"graphTooltip": 1,
"id": 22,
"iteration": 1656210900773,
"links": [],
"liveNow": false,
"panels": [
{
"collapsed": false,
"datasource": {
"type": "prometheus",
"uid": "prometheus"
},
"gridPos": {
"h": 1,
"w": 24,
"x": 0,
"y": 0
},
"id": 6,
"panels": [],
"targets": [
{
"datasource": {
"type": "prometheus",
"uid": "prometheus"
},
"refId": "A"
}
],
"title": "Alerts",
"type": "row"
},
{
"aliasColors": {},
"bars": false,
"dashLength": 10,
"dashes": false,
"datasource": {
"uid": "$datasource"
},
"fill": 1,
"fillGradient": 0,
"gridPos": {
"h": 7,
"w": 12,
"x": 0,
"y": 1
},
"hiddenSeries": false,
"id": 2,
"legend": {
"alignAsTable": false,
"avg": false,
"current": false,
"max": false,
"min": false,
"rightSide": false,
"show": false,
"total": false,
"values": false
},
"lines": true,
"linewidth": 1,
"links": [],
"nullPointMode": "null",
"options": {
"alertThreshold": true
},
"percentage": false,
"pluginVersion": "9.0.1",
"pointradius": 5,
"points": false,
"renderer": "flot",
"seriesOverrides": [],
"spaceLength": 10,
"stack": true,
"steppedLine": false,
"targets": [
{
"datasource": {
"uid": "$datasource"
},
"expr": "sum(alertmanager_alerts{namespace=~\"$namespace\",service=~\"$service\"}) by (namespace,service,instance)",
"format": "time_series",
"intervalFactor": 2,
"legendFormat": "{{instance}}",
"refId": "A"
}
],
"thresholds": [],
"timeRegions": [],
"title": "Alerts",
"tooltip": {
"shared": true,
"sort": 0,
"value_type": "individual"
},
"type": "graph",
"xaxis": {
"mode": "time",
"show": true,
"values": []
},
"yaxes": [
{
"format": "none",
"logBase": 1,
"show": true
},
{
"format": "none",
"logBase": 1,
"show": true
}
],
"yaxis": {
"align": false
}
},
{
"aliasColors": {},
"bars": false,
"dashLength": 10,
"dashes": false,
"datasource": {
"uid": "$datasource"
},
"fill": 1,
"fillGradient": 0,
"gridPos": {
"h": 7,
"w": 12,
"x": 12,
"y": 1
},
"hiddenSeries": false,
"id": 3,
"legend": {
"alignAsTable": false,
"avg": false,
"current": false,
"max": false,
"min": false,
"rightSide": false,
"show": false,
"total": false,
"values": false
},
"lines": true,
"linewidth": 1,
"links": [],
"nullPointMode": "null",
"options": {
"alertThreshold": true
},
"percentage": false,
"pluginVersion": "9.0.1",
"pointradius": 5,
"points": false,
"renderer": "flot",
"seriesOverrides": [],
"spaceLength": 10,
"stack": true,
"steppedLine": false,
"targets": [
{
"datasource": {
"uid": "$datasource"
},
"expr": "sum(rate(alertmanager_alerts_received_total{namespace=~\"$namespace\",service=~\"$service\"}[$__rate_interval])) by (namespace,service,instance)",
"format": "time_series",
"intervalFactor": 2,
"legendFormat": "{{instance}} Received",
"refId": "A"
},
{
"datasource": {
"uid": "$datasource"
},
"expr": "sum(rate(alertmanager_alerts_invalid_total{namespace=~\"$namespace\",service=~\"$service\"}[$__rate_interval])) by (namespace,service,instance)",
"format": "time_series",
"intervalFactor": 2,
"legendFormat": "{{instance}} Invalid",
"refId": "B"
}
],
"thresholds": [],
"timeRegions": [],
"title": "Alerts receive rate",
"tooltip": {
"shared": true,
"sort": 0,
"value_type": "individual"
},
"type": "graph",
"xaxis": {
"mode": "time",
"show": true,
"values": []
},
"yaxes": [
{
"format": "ops",
"logBase": 1,
"show": true
},
{
"format": "ops",
"logBase": 1,
"show": true
}
],
"yaxis": {
"align": false
}
},
{
"collapsed": false,
"datasource": {
"type": "prometheus",
"uid": "prometheus"
},
"gridPos": {
"h": 1,
"w": 24,
"x": 0,
"y": 8
},
"id": 7,
"panels": [],
"targets": [
{
"datasource": {
"type": "prometheus",
"uid": "prometheus"
},
"refId": "A"
}
],
"title": "Notifications",
"type": "row"
},
{
"aliasColors": {},
"bars": false,
"dashLength": 10,
"dashes": false,
"datasource": {
"uid": "$datasource"
},
"fill": 1,
"fillGradient": 0,
"gridPos": {
"h": 7,
"w": 6,
"x": 0,
"y": 9
},
"hiddenSeries": false,
"id": 4,
"legend": {
"alignAsTable": false,
"avg": false,
"current": false,
"max": false,
"min": false,
"rightSide": false,
"show": false,
"total": false,
"values": false
},
"lines": true,
"linewidth": 1,
"links": [],
"nullPointMode": "null",
"options": {
"alertThreshold": true
},
"percentage": false,
"pluginVersion": "9.0.1",
"pointradius": 5,
"points": false,
"renderer": "flot",
"repeat": "integration",
"seriesOverrides": [],
"spaceLength": 10,
"stack": true,
"steppedLine": false,
"targets": [
{
"datasource": {
"uid": "$datasource"
},
"expr": "sum(rate(alertmanager_notifications_total{namespace=~\"$namespace\",service=~\"$service\", integration=\"$integration\"}[$__rate_interval])) by (integration,namespace,service,instance)",
"format": "time_series",
"intervalFactor": 2,
"legendFormat": "{{instance}} Total",
"refId": "A"
},
{
"datasource": {
"uid": "$datasource"
},
"expr": "sum(rate(alertmanager_notifications_failed_total{namespace=~\"$namespace\",service=~\"$service\", integration=\"$integration\"}[$__rate_interval])) by (integration,namespace,service,instance)",
"format": "time_series",
"intervalFactor": 2,
"legendFormat": "{{instance}} Failed",
"refId": "B"
}
],
"thresholds": [],
"timeRegions": [],
"title": "$integration: Notifications Send Rate",
"tooltip": {
"shared": true,
"sort": 0,
"value_type": "individual"
},
"type": "graph",
"xaxis": {
"mode": "time",
"show": true,
"values": []
},
"yaxes": [
{
"format": "ops",
"logBase": 1,
"show": true
},
{
"format": "ops",
"logBase": 1,
"show": true
}
],
"yaxis": {
"align": false
}
},
{
"aliasColors": {},
"bars": false,
"dashLength": 10,
"dashes": false,
"datasource": {
"uid": "$datasource"
},
"fill": 1,
"fillGradient": 0,
"gridPos": {
"h": 7,
"w": 6,
"x": 0,
"y": 16
},
"hiddenSeries": false,
"id": 5,
"legend": {
"alignAsTable": false,
"avg": false,
"current": false,
"max": false,
"min": false,
"rightSide": false,
"show": false,
"total": false,
"values": false
},
"lines": true,
"linewidth": 1,
"links": [],
"nullPointMode": "null",
"options": {
"alertThreshold": true
},
"percentage": false,
"pluginVersion": "9.0.1",
"pointradius": 5,
"points": false,
"renderer": "flot",
"repeat": "integration",
"seriesOverrides": [],
"spaceLength": 10,
"stack": false,
"steppedLine": false,
"targets": [
{
"datasource": {
"uid": "$datasource"
},
"expr": "histogram_quantile(0.99,\n sum(rate(alertmanager_notification_latency_seconds_bucket{namespace=~\"$namespace\",service=~\"$service\", integration=\"$integration\"}[$__rate_interval])) by (le,namespace,service,instance)\n) \n",
"format": "time_series",
"intervalFactor": 2,
"legendFormat": "{{instance}} 99th Percentile",
"refId": "A"
},
{
"datasource": {
"uid": "$datasource"
},
"expr": "histogram_quantile(0.50,\n sum(rate(alertmanager_notification_latency_seconds_bucket{namespace=~\"$namespace\",service=~\"$service\", integration=\"$integration\"}[$__rate_interval])) by (le,namespace,service,instance)\n) \n",
"format": "time_series",
"intervalFactor": 2,
"legendFormat": "{{instance}} Median",
"refId": "B"
},
{
"datasource": {
"uid": "$datasource"
},
"expr": "sum(rate(alertmanager_notification_latency_seconds_sum{namespace=~\"$namespace\",service=~\"$service\", integration=\"$integration\"}[$__rate_interval])) by (namespace,service,instance)\n/\nsum(rate(alertmanager_notification_latency_seconds_count{namespace=~\"$namespace\",service=~\"$service\", integration=\"$integration\"}[$__rate_interval])) by (namespace,service,instance)\n",
"format": "time_series",
"intervalFactor": 2,
"legendFormat": "{{instance}} Average",
"refId": "C"
}
],
"thresholds": [],
"timeRegions": [],
"title": "$integration: Notification Duration",
"tooltip": {
"shared": true,
"sort": 0,
"value_type": "individual"
},
"type": "graph",
"xaxis": {
"mode": "time",
"show": true,
"values": []
},
"yaxes": [
{
"format": "s",
"logBase": 1,
"show": true
},
{
"format": "s",
"logBase": 1,
"show": true
}
],
"yaxis": {
"align": false
}
}
],
"refresh": "30s",
"schemaVersion": 36,
"style": "dark",
"tags": [
"alertmanager-mixin"
],
"templating": {
"list": [
{
"current": {
"selected": false,
"text": "Prometheus",
"value": "Prometheus"
},
"hide": 0,
"includeAll": false,
"label": "Data Source",
"multi": false,
"name": "datasource",
"options": [],
"query": "prometheus",
"refresh": 1,
"regex": "",
"skipUrlSync": false,
"type": "datasource"
},
{
"current": {
"selected": false,
"text": "default",
"value": "default"
},
"datasource": {
"type": "prometheus",
"uid": "$datasource"
},
"definition": "",
"hide": 0,
"includeAll": false,
"label": "namespace",
"multi": false,
"name": "namespace",
"options": [],
"query": {
"query": "label_values(alertmanager_alerts, namespace)",
"refId": "Prometheus-namespace-Variable-Query"
},
"refresh": 2,
"regex": "",
"skipUrlSync": false,
"sort": 1,
"tagValuesQuery": "",
"tagsQuery": "",
"type": "query",
"useTags": false
},
{
"current": {
"selected": false,
"text": "kube-prometheus-stack-alertmanager",
"value": "kube-prometheus-stack-alertmanager"
},
"datasource": {
"type": "prometheus",
"uid": "$datasource"
},
"definition": "",
"hide": 0,
"includeAll": false,
"label": "service",
"multi": false,
"name": "service",
"options": [],
"query": {
"query": "label_values(alertmanager_alerts, service)",
"refId": "Prometheus-service-Variable-Query"
},
"refresh": 2,
"regex": "",
"skipUrlSync": false,
"sort": 1,
"tagValuesQuery": "",
"tagsQuery": "",
"type": "query",
"useTags": false
},
{
"current": {
"selected": false,
"text": "All",
"value": "$__all"
},
"datasource": {
"type": "prometheus",
"uid": "$datasource"
},
"definition": "",
"hide": 2,
"includeAll": true,
"multi": false,
"name": "integration",
"options": [],
"query": {
"query": "label_values(alertmanager_notifications_total{integration=~\".*\"}, integration)",
"refId": "Prometheus-integration-Variable-Query"
},
"refresh": 2,
"regex": "",
"skipUrlSync": false,
"sort": 1,
"tagValuesQuery": "",
"tagsQuery": "",
"type": "query",
"useTags": false
}
]
},
"time": {
"from": "now-1h",
"to": "now"
},
"timepicker": {
"refresh_intervals": [
"5s",
"10s",
"30s",
"1m",
"5m",
"15m",
"30m",
"1h",
"2h",
"1d"
],
"time_options": [
"5m",
"15m",
"1h",
"6h",
"12h",
"24h",
"2d",
"7d",
"30d"
]
},
"timezone": "utc",
"title": "Alertmanager / Overview",
"uid": "alertmanager-overview",
"version": 1,
"weekStart": ""
}
File diff suppressed because it is too large Load Diff
File diff suppressed because it is too large Load Diff
File diff suppressed because it is too large Load Diff
File diff suppressed because it is too large Load Diff
File diff suppressed because it is too large Load Diff
File diff suppressed because it is too large Load Diff
@@ -1,563 +0,0 @@
{
"annotations": {
"list": [
{
"builtIn": 1,
"datasource": {
"type": "datasource",
"uid": "grafana"
},
"enable": true,
"hide": true,
"iconColor": "rgba(0, 211, 255, 1)",
"name": "Annotations & Alerts",
"target": {
"limit": 100,
"matchAny": false,
"tags": [],
"type": "dashboard"
},
"type": "dashboard"
}
]
},
"editable": true,
"fiscalYearStartMonth": 0,
"graphTooltip": 0,
"id": 15,
"iteration": 1656211285882,
"links": [],
"liveNow": false,
"panels": [
{
"datasource": {
"uid": "$datasource"
},
"fieldConfig": {
"defaults": {
"mappings": [],
"noValue": "0",
"thresholds": {
"mode": "absolute",
"steps": [
{
"color": "green",
"value": null
},
{
"color": "red",
"value": 80
}
]
}
},
"overrides": []
},
"gridPos": {
"h": 5,
"w": 6,
"x": 0,
"y": 0
},
"id": 6,
"options": {
"colorMode": "value",
"graphMode": "area",
"justifyMode": "auto",
"orientation": "auto",
"reduceOptions": {
"calcs": [
"mean"
],
"fields": "",
"values": false
},
"text": {},
"textMode": "auto"
},
"pluginVersion": "9.0.1",
"targets": [
{
"datasource": {
"uid": "$datasource"
},
"expr": "grafana_alerting_result_total{job=~\"$job\", instance=~\"$instance\", state=\"alerting\"}",
"instant": true,
"interval": "",
"legendFormat": "",
"refId": "A"
}
],
"title": "Firing Alerts",
"type": "stat"
},
{
"datasource": {
"uid": "$datasource"
},
"fieldConfig": {
"defaults": {
"mappings": [],
"thresholds": {
"mode": "absolute",
"steps": [
{
"color": "green",
"value": null
},
{
"color": "red",
"value": 80
}
]
}
},
"overrides": []
},
"gridPos": {
"h": 5,
"w": 6,
"x": 6,
"y": 0
},
"id": 8,
"options": {
"colorMode": "value",
"graphMode": "area",
"justifyMode": "auto",
"orientation": "auto",
"reduceOptions": {
"calcs": [
"mean"
],
"fields": "",
"values": false
},
"text": {},
"textMode": "auto"
},
"pluginVersion": "9.0.1",
"targets": [
{
"datasource": {
"uid": "$datasource"
},
"expr": "sum(grafana_stat_totals_dashboard{job=~\"$job\", instance=~\"$instance\"})",
"interval": "",
"legendFormat": "",
"refId": "A"
}
],
"title": "Dashboards",
"type": "stat"
},
{
"datasource": {
"uid": "$datasource"
},
"fieldConfig": {
"defaults": {
"custom": {
"displayMode": "auto",
"inspect": false
},
"mappings": [],
"thresholds": {
"mode": "absolute",
"steps": [
{
"color": "green",
"value": null
},
{
"color": "red",
"value": 80
}
]
}
},
"overrides": []
},
"gridPos": {
"h": 5,
"w": 12,
"x": 12,
"y": 0
},
"id": 10,
"options": {
"footer": {
"fields": "",
"reducer": [
"sum"
],
"show": false
},
"showHeader": true
},
"pluginVersion": "9.0.1",
"targets": [
{
"datasource": {
"uid": "$datasource"
},
"expr": "grafana_build_info{job=~\"$job\", instance=~\"$instance\"}",
"instant": true,
"interval": "",
"legendFormat": "",
"refId": "A"
}
],
"title": "Build Info",
"transformations": [
{
"id": "labelsToFields",
"options": {}
},
{
"id": "merge",
"options": {}
},
{
"id": "organize",
"options": {
"excludeByName": {
"Time": true,
"Value": true,
"branch": true,
"container": true,
"goversion": true,
"namespace": true,
"pod": true,
"revision": true
},
"indexByName": {
"Time": 7,
"Value": 11,
"branch": 4,
"container": 8,
"edition": 2,
"goversion": 6,
"instance": 1,
"job": 0,
"namespace": 9,
"pod": 10,
"revision": 5,
"version": 3
},
"renameByName": {}
}
}
],
"type": "table"
},
{
"aliasColors": {},
"bars": false,
"dashLength": 10,
"dashes": false,
"datasource": {
"uid": "$datasource"
},
"fieldConfig": {
"defaults": {
"links": []
},
"overrides": []
},
"fill": 1,
"fillGradient": 0,
"gridPos": {
"h": 8,
"w": 12,
"x": 0,
"y": 5
},
"hiddenSeries": false,
"id": 2,
"legend": {
"avg": false,
"current": false,
"max": false,
"min": false,
"show": true,
"total": false,
"values": false
},
"lines": true,
"linewidth": 1,
"nullPointMode": "null",
"options": {
"alertThreshold": true
},
"percentage": false,
"pluginVersion": "9.0.1",
"pointradius": 2,
"points": false,
"renderer": "flot",
"seriesOverrides": [],
"spaceLength": 10,
"stack": true,
"steppedLine": false,
"targets": [
{
"datasource": {
"uid": "$datasource"
},
"expr": "sum by (status_code) (irate(grafana_http_request_duration_seconds_count{job=~\"$job\", instance=~\"$instance\"}[1m])) ",
"interval": "",
"legendFormat": "{{status_code}}",
"refId": "A"
}
],
"thresholds": [],
"timeRegions": [],
"title": "RPS",
"tooltip": {
"shared": true,
"sort": 0,
"value_type": "individual"
},
"type": "graph",
"xaxis": {
"mode": "time",
"show": true,
"values": []
},
"yaxes": [
{
"$$hashKey": "object:157",
"format": "reqps",
"logBase": 1,
"show": true
},
{
"$$hashKey": "object:158",
"format": "short",
"logBase": 1,
"show": false
}
],
"yaxis": {
"align": false
}
},
{
"aliasColors": {},
"bars": false,
"dashLength": 10,
"dashes": false,
"datasource": {
"uid": "$datasource"
},
"fieldConfig": {
"defaults": {
"links": []
},
"overrides": []
},
"fill": 1,
"fillGradient": 0,
"gridPos": {
"h": 8,
"w": 12,
"x": 12,
"y": 5
},
"hiddenSeries": false,
"id": 4,
"legend": {
"avg": false,
"current": false,
"max": false,
"min": false,
"show": true,
"total": false,
"values": false
},
"lines": true,
"linewidth": 1,
"nullPointMode": "null",
"options": {
"alertThreshold": true
},
"percentage": false,
"pluginVersion": "9.0.1",
"pointradius": 2,
"points": false,
"renderer": "flot",
"seriesOverrides": [],
"spaceLength": 10,
"stack": false,
"steppedLine": false,
"targets": [
{
"datasource": {
"uid": "$datasource"
},
"exemplar": true,
"expr": "histogram_quantile(0.99, sum(irate(grafana_http_request_duration_seconds_bucket{instance=~\"$instance\", job=~\"$job\"}[$__rate_interval])) by (le)) * 1",
"interval": "",
"legendFormat": "99th Percentile",
"refId": "A"
},
{
"datasource": {
"uid": "$datasource"
},
"exemplar": true,
"expr": "histogram_quantile(0.50, sum(irate(grafana_http_request_duration_seconds_bucket{instance=~\"$instance\", job=~\"$job\"}[$__rate_interval])) by (le)) * 1",
"interval": "",
"legendFormat": "50th Percentile",
"refId": "B"
},
{
"datasource": {
"uid": "$datasource"
},
"exemplar": true,
"expr": "sum(irate(grafana_http_request_duration_seconds_sum{instance=~\"$instance\", job=~\"$job\"}[$__rate_interval])) * 1 / sum(irate(grafana_http_request_duration_seconds_count{instance=~\"$instance\", job=~\"$job\"}[$__rate_interval]))",
"interval": "",
"legendFormat": "Average",
"refId": "C"
}
],
"thresholds": [],
"timeRegions": [],
"title": "Request Latency",
"tooltip": {
"shared": true,
"sort": 0,
"value_type": "individual"
},
"type": "graph",
"xaxis": {
"mode": "time",
"show": true,
"values": []
},
"yaxes": [
{
"$$hashKey": "object:210",
"format": "ms",
"logBase": 1,
"show": true
},
{
"$$hashKey": "object:211",
"format": "short",
"logBase": 1,
"show": true
}
],
"yaxis": {
"align": false
}
}
],
"schemaVersion": 36,
"style": "dark",
"tags": [],
"templating": {
"list": [
{
"current": {
"selected": false,
"text": "Prometheus",
"value": "Prometheus"
},
"hide": 0,
"includeAll": false,
"multi": false,
"name": "datasource",
"options": [],
"query": "prometheus",
"queryValue": "",
"refresh": 1,
"regex": "",
"skipUrlSync": false,
"type": "datasource"
},
{
"allValue": ".*",
"current": {
"selected": false,
"text": "All",
"value": "$__all"
},
"datasource": {
"uid": "$datasource"
},
"definition": "label_values(grafana_build_info, job)",
"hide": 0,
"includeAll": true,
"multi": true,
"name": "job",
"options": [],
"query": {
"query": "label_values(grafana_build_info, job)",
"refId": "Billing Admin-job-Variable-Query"
},
"refresh": 1,
"regex": "",
"skipUrlSync": false,
"sort": 0,
"tagValuesQuery": "",
"tagsQuery": "",
"type": "query",
"useTags": false
},
{
"allValue": ".*",
"current": {
"selected": false,
"text": "All",
"value": "$__all"
},
"datasource": {
"uid": "$datasource"
},
"definition": "label_values(grafana_build_info, instance)",
"hide": 0,
"includeAll": true,
"multi": true,
"name": "instance",
"options": [],
"query": {
"query": "label_values(grafana_build_info, instance)",
"refId": "Billing Admin-instance-Variable-Query"
},
"refresh": 1,
"regex": "",
"skipUrlSync": false,
"sort": 0,
"tagValuesQuery": "",
"tagsQuery": "",
"type": "query",
"useTags": false
}
]
},
"time": {
"from": "now-6h",
"to": "now"
},
"timepicker": {
"refresh_intervals": [
"10s",
"30s",
"1m",
"5m",
"15m",
"30m",
"1h",
"2h",
"1d"
]
},
"timezone": "utc",
"title": "Grafana Overview",
"uid": "6be0s85Mk",
"version": 1,
"weekStart": ""
}
File diff suppressed because it is too large Load Diff
File diff suppressed because it is too large Load Diff
File diff suppressed because it is too large Load Diff
File diff suppressed because it is too large Load Diff
File diff suppressed because it is too large Load Diff
File diff suppressed because it is too large Load Diff
File diff suppressed because it is too large Load Diff
File diff suppressed because it is too large Load Diff
File diff suppressed because it is too large Load Diff
File diff suppressed because it is too large Load Diff
File diff suppressed because it is too large Load Diff
@@ -1,550 +0,0 @@
{
"annotations": {
"list": [
{
"builtIn": 1,
"datasource": {
"type": "grafana",
"uid": "-- Grafana --"
},
"enable": true,
"hide": true,
"iconColor": "rgba(0, 211, 255, 1)",
"name": "Annotations & Alerts",
"target": {
"limit": 100,
"matchAny": false,
"tags": [],
"type": "dashboard"
},
"type": "dashboard"
}
]
},
"editable": false,
"fiscalYearStartMonth": 0,
"graphTooltip": 0,
"id": 18,
"iteration": 1656212469966,
"links": [],
"liveNow": false,
"panels": [
{
"aliasColors": {},
"bars": false,
"dashLength": 10,
"dashes": false,
"datasource": {
"uid": "$datasource"
},
"fill": 1,
"fillGradient": 0,
"gridPos": {
"h": 7,
"w": 18,
"x": 0,
"y": 0
},
"hiddenSeries": false,
"id": 2,
"interval": "1m",
"legend": {
"alignAsTable": true,
"avg": true,
"current": true,
"max": true,
"min": true,
"rightSide": true,
"show": true,
"total": false,
"values": true
},
"lines": true,
"linewidth": 1,
"links": [],
"nullPointMode": "null",
"options": {
"alertThreshold": true
},
"percentage": false,
"pluginVersion": "9.0.1",
"pointradius": 5,
"points": false,
"renderer": "flot",
"seriesOverrides": [],
"spaceLength": 10,
"stack": true,
"steppedLine": false,
"targets": [
{
"datasource": {
"uid": "$datasource"
},
"expr": "(\n sum without(instance, node) (topk(1, (kubelet_volume_stats_capacity_bytes{cluster=\"$cluster\", job=\"kubelet\", metrics_path=\"/metrics\", namespace=\"$namespace\", persistentvolumeclaim=\"$volume\"})))\n -\n sum without(instance, node) (topk(1, (kubelet_volume_stats_available_bytes{cluster=\"$cluster\", job=\"kubelet\", metrics_path=\"/metrics\", namespace=\"$namespace\", persistentvolumeclaim=\"$volume\"})))\n)\n",
"format": "time_series",
"intervalFactor": 1,
"legendFormat": "Used Space",
"refId": "A"
},
{
"datasource": {
"uid": "$datasource"
},
"expr": "sum without(instance, node) (topk(1, (kubelet_volume_stats_available_bytes{cluster=\"$cluster\", job=\"kubelet\", metrics_path=\"/metrics\", namespace=\"$namespace\", persistentvolumeclaim=\"$volume\"})))\n",
"format": "time_series",
"intervalFactor": 1,
"legendFormat": "Free Space",
"refId": "B"
}
],
"thresholds": [],
"timeRegions": [],
"title": "Volume Space Usage",
"tooltip": {
"shared": false,
"sort": 0,
"value_type": "individual"
},
"type": "graph",
"xaxis": {
"mode": "time",
"show": true,
"values": []
},
"yaxes": [
{
"format": "bytes",
"logBase": 1,
"min": 0,
"show": true
},
{
"format": "bytes",
"logBase": 1,
"min": 0,
"show": true
}
],
"yaxis": {
"align": false
}
},
{
"datasource": {
"uid": "$datasource"
},
"fieldConfig": {
"defaults": {
"color": {
"mode": "thresholds"
},
"mappings": [
{
"options": {
"match": "null",
"result": {
"text": "N/A"
}
},
"type": "special"
}
],
"max": 100,
"min": 0,
"thresholds": {
"mode": "absolute",
"steps": [
{
"color": "rgba(50, 172, 45, 0.97)",
"value": null
},
{
"color": "rgba(237, 129, 40, 0.89)",
"value": 80
},
{
"color": "rgba(245, 54, 54, 0.9)",
"value": 90
}
]
},
"unit": "percent"
},
"overrides": []
},
"gridPos": {
"h": 7,
"w": 6,
"x": 18,
"y": 0
},
"id": 3,
"interval": "1m",
"links": [],
"maxDataPoints": 100,
"options": {
"orientation": "horizontal",
"reduceOptions": {
"calcs": [
"lastNotNull"
],
"fields": "",
"values": false
},
"showThresholdLabels": false,
"showThresholdMarkers": true
},
"pluginVersion": "9.0.1",
"targets": [
{
"datasource": {
"uid": "$datasource"
},
"expr": "max without(instance,node) (\n(\n topk(1, kubelet_volume_stats_capacity_bytes{cluster=\"$cluster\", job=\"kubelet\", metrics_path=\"/metrics\", namespace=\"$namespace\", persistentvolumeclaim=\"$volume\"})\n -\n topk(1, kubelet_volume_stats_available_bytes{cluster=\"$cluster\", job=\"kubelet\", metrics_path=\"/metrics\", namespace=\"$namespace\", persistentvolumeclaim=\"$volume\"})\n)\n/\ntopk(1, kubelet_volume_stats_capacity_bytes{cluster=\"$cluster\", job=\"kubelet\", metrics_path=\"/metrics\", namespace=\"$namespace\", persistentvolumeclaim=\"$volume\"})\n* 100)\n",
"format": "time_series",
"intervalFactor": 2,
"legendFormat": "",
"refId": "A"
}
],
"title": "Volume Space Usage",
"type": "gauge"
},
{
"aliasColors": {},
"bars": false,
"dashLength": 10,
"dashes": false,
"datasource": {
"uid": "$datasource"
},
"fill": 1,
"fillGradient": 0,
"gridPos": {
"h": 7,
"w": 18,
"x": 0,
"y": 7
},
"hiddenSeries": false,
"id": 4,
"interval": "1m",
"legend": {
"alignAsTable": true,
"avg": true,
"current": true,
"max": true,
"min": true,
"rightSide": true,
"show": true,
"total": false,
"values": true
},
"lines": true,
"linewidth": 1,
"links": [],
"nullPointMode": "null",
"options": {
"alertThreshold": true
},
"percentage": false,
"pluginVersion": "9.0.1",
"pointradius": 5,
"points": false,
"renderer": "flot",
"seriesOverrides": [],
"spaceLength": 10,
"stack": true,
"steppedLine": false,
"targets": [
{
"datasource": {
"uid": "$datasource"
},
"expr": "sum without(instance, node) (topk(1, (kubelet_volume_stats_inodes_used{cluster=\"$cluster\", job=\"kubelet\", metrics_path=\"/metrics\", namespace=\"$namespace\", persistentvolumeclaim=\"$volume\"})))\n",
"format": "time_series",
"intervalFactor": 1,
"legendFormat": "Used inodes",
"refId": "A"
},
{
"datasource": {
"uid": "$datasource"
},
"expr": "(\n sum without(instance, node) (topk(1, (kubelet_volume_stats_inodes{cluster=\"$cluster\", job=\"kubelet\", metrics_path=\"/metrics\", namespace=\"$namespace\", persistentvolumeclaim=\"$volume\"})))\n -\n sum without(instance, node) (topk(1, (kubelet_volume_stats_inodes_used{cluster=\"$cluster\", job=\"kubelet\", metrics_path=\"/metrics\", namespace=\"$namespace\", persistentvolumeclaim=\"$volume\"})))\n)\n",
"format": "time_series",
"intervalFactor": 1,
"legendFormat": " Free inodes",
"refId": "B"
}
],
"thresholds": [],
"timeRegions": [],
"title": "Volume inodes Usage",
"tooltip": {
"shared": false,
"sort": 0,
"value_type": "individual"
},
"type": "graph",
"xaxis": {
"mode": "time",
"show": true,
"values": []
},
"yaxes": [
{
"format": "none",
"logBase": 1,
"min": 0,
"show": true
},
{
"format": "none",
"logBase": 1,
"min": 0,
"show": true
}
],
"yaxis": {
"align": false
}
},
{
"datasource": {
"uid": "$datasource"
},
"fieldConfig": {
"defaults": {
"color": {
"mode": "thresholds"
},
"mappings": [
{
"options": {
"match": "null",
"result": {
"text": "N/A"
}
},
"type": "special"
}
],
"max": 100,
"min": 0,
"thresholds": {
"mode": "absolute",
"steps": [
{
"color": "rgba(50, 172, 45, 0.97)",
"value": null
},
{
"color": "rgba(237, 129, 40, 0.89)",
"value": 80
},
{
"color": "rgba(245, 54, 54, 0.9)",
"value": 90
}
]
},
"unit": "percent"
},
"overrides": []
},
"gridPos": {
"h": 7,
"w": 6,
"x": 18,
"y": 7
},
"id": 5,
"interval": "1m",
"links": [],
"maxDataPoints": 100,
"options": {
"orientation": "horizontal",
"reduceOptions": {
"calcs": [
"lastNotNull"
],
"fields": "",
"values": false
},
"showThresholdLabels": false,
"showThresholdMarkers": true
},
"pluginVersion": "9.0.1",
"targets": [
{
"datasource": {
"uid": "$datasource"
},
"expr": "max without(instance,node) (\ntopk(1, kubelet_volume_stats_inodes_used{cluster=\"$cluster\", job=\"kubelet\", metrics_path=\"/metrics\", namespace=\"$namespace\", persistentvolumeclaim=\"$volume\"})\n/\ntopk(1, kubelet_volume_stats_inodes{cluster=\"$cluster\", job=\"kubelet\", metrics_path=\"/metrics\", namespace=\"$namespace\", persistentvolumeclaim=\"$volume\"})\n* 100)\n",
"format": "time_series",
"intervalFactor": 2,
"legendFormat": "",
"refId": "A"
}
],
"title": "Volume inodes Usage",
"type": "gauge"
}
],
"refresh": "10s",
"schemaVersion": 36,
"style": "dark",
"tags": [
"kubernetes-mixin"
],
"templating": {
"list": [
{
"current": {
"selected": false,
"text": "default",
"value": "default"
},
"hide": 0,
"includeAll": false,
"label": "Data Source",
"multi": false,
"name": "datasource",
"options": [],
"query": "prometheus",
"refresh": 1,
"regex": "",
"skipUrlSync": false,
"type": "datasource"
},
{
"current": {
"isNone": true,
"selected": false,
"text": "None",
"value": ""
},
"datasource": {
"type": "prometheus",
"uid": "$datasource"
},
"definition": "",
"hide": 2,
"includeAll": false,
"label": "cluster",
"multi": false,
"name": "cluster",
"options": [],
"query": {
"query": "label_values(kubelet_volume_stats_capacity_bytes{job=\"kubelet\", metrics_path=\"/metrics\"}, cluster)",
"refId": "Prometheus-cluster-Variable-Query"
},
"refresh": 2,
"regex": "",
"skipUrlSync": false,
"sort": 1,
"tagValuesQuery": "",
"tagsQuery": "",
"type": "query",
"useTags": false
},
{
"current": {
"isNone": true,
"selected": false,
"text": "None",
"value": ""
},
"datasource": {
"type": "prometheus",
"uid": "$datasource"
},
"definition": "",
"hide": 0,
"includeAll": false,
"label": "Namespace",
"multi": false,
"name": "namespace",
"options": [],
"query": {
"query": "label_values(kubelet_volume_stats_capacity_bytes{cluster=\"$cluster\", job=\"kubelet\", metrics_path=\"/metrics\"}, namespace)",
"refId": "Prometheus-namespace-Variable-Query"
},
"refresh": 2,
"regex": "",
"skipUrlSync": false,
"sort": 1,
"tagValuesQuery": "",
"tagsQuery": "",
"type": "query",
"useTags": false
},
{
"current": {
"isNone": true,
"selected": false,
"text": "None",
"value": ""
},
"datasource": {
"type": "prometheus",
"uid": "$datasource"
},
"definition": "",
"hide": 0,
"includeAll": false,
"label": "PersistentVolumeClaim",
"multi": false,
"name": "volume",
"options": [],
"query": {
"query": "label_values(kubelet_volume_stats_capacity_bytes{cluster=\"$cluster\", job=\"kubelet\", metrics_path=\"/metrics\", namespace=\"$namespace\"}, persistentvolumeclaim)",
"refId": "Prometheus-volume-Variable-Query"
},
"refresh": 2,
"regex": "",
"skipUrlSync": false,
"sort": 1,
"tagValuesQuery": "",
"tagsQuery": "",
"type": "query",
"useTags": false
}
]
},
"time": {
"from": "now-7d",
"to": "now"
},
"timepicker": {
"refresh_intervals": [
"5s",
"10s",
"30s",
"1m",
"5m",
"15m",
"30m",
"1h",
"2h",
"1d"
],
"time_options": [
"5m",
"15m",
"1h",
"6h",
"12h",
"24h",
"2d",
"7d",
"30d"
]
},
"timezone": "utc",
"title": "Kubernetes / Persistent Volumes",
"uid": "919b92a8e8041bd567af9edab12c840c",
"version": 1,
"weekStart": ""
}
File diff suppressed because it is too large Load Diff
File diff suppressed because it is too large Load Diff
File diff suppressed because it is too large Load Diff
File diff suppressed because it is too large Load Diff
File diff suppressed because it is too large Load Diff
File diff suppressed because it is too large Load Diff
-106
View File
@@ -1,106 +0,0 @@
locals {
name = "adot-collector-kubeprometheus"
namespace = try(var.config.helm_config.namespace, local.name)
}
terraform {
required_providers {
grafana = {
source = "grafana/grafana"
version = "1.24.0"
}
}
}
resource "helm_release" "kube_state_metrics" {
count = var.config.enable_kube_state_metrics ? 1 : 0
chart = var.config.ksm_helm_chart_name
create_namespace = var.config.kms_create_namespace
namespace = var.config.ksm_k8s_namespace
name = var.config.ksm_helm_release_name
version = var.config.ksm_helm_chart_version
repository = var.config.ksm_helm_repo_url
dynamic "set" {
for_each = var.config.ksm_helm_settings
content {
name = set.key
value = set.value
}
}
}
resource "helm_release" "prometheus_node_exporter" {
count = var.config.enable_node_exporter ? 1 : 0
chart = var.config.ne_helm_chart_name
create_namespace = var.config.ne_create_namespace
namespace = var.config.ne_k8s_namespace
name = var.config.ne_helm_release_name
version = var.config.ne_helm_chart_version
repository = var.config.ne_helm_repo_url
dynamic "set" {
for_each = var.config.ne_helm_settings
content {
name = set.key
value = set.value
}
}
}
data "aws_partition" "current" {}
module "helm_addon" {
source = "github.com/aws-ia/terraform-aws-eks-blueprints/modules/kubernetes-addons/helm-addon"
helm_config = merge(
{
name = local.name
chart = "${path.module}/otel-config"
version = "0.2.0"
namespace = local.namespace
description = "ADOT helm Chart deployment configuration"
},
var.helm_config
)
set_values = [
{
name = "ampurl"
value = "${var.amp_endpoint}api/v1/remote_write"
},
{
name = "region"
value = var.amp_region
},
{
name = "prometheusMetricsEndpoint"
value = "metrics"
},
{
name = "prometheusMetricsPort"
value = 8888
},
{
name = "scrapeInterval"
value = "15s"
},
{
name = "scrapeTimeout"
value = "10s"
},
{
name = "scrapeSampleLimit"
value = 1000
}
]
irsa_config = {
create_kubernetes_namespace = true
kubernetes_namespace = local.namespace
create_kubernetes_service_account = true
kubernetes_service_account = try(var.config.helm_config.service_account, local.name)
irsa_iam_policies = ["arn:${data.aws_partition.current.partition}:iam::aws:policy/AmazonPrometheusRemoteWriteAccess"]
}
addon_context = var.addon_context
}
@@ -1,6 +0,0 @@
apiVersion: v2
name: opentelemetry
description: A Helm chart to install otel operator
type: application
version: 0.2.0
appVersion: v0.1.0
@@ -1,29 +0,0 @@
apiVersion: rbac.authorization.k8s.io/v1
kind: ClusterRole
metadata:
name: otel-prometheus-role
rules:
- apiGroups:
- ""
resources:
- nodes
- nodes/proxy
- services
- endpoints
- pods
verbs:
- get
- list
- watch
- apiGroups:
- extensions
resources:
- ingresses
verbs:
- get
- list
- watch
- nonResourceURLs:
- /metrics
verbs:
- get
@@ -1,12 +0,0 @@
apiVersion: rbac.authorization.k8s.io/v1
kind: ClusterRoleBinding
metadata:
name: otel-prometheus-role-binding
roleRef:
apiGroup: rbac.authorization.k8s.io
kind: ClusterRole
name: otel-prometheus-role
subjects:
- kind: ServiceAccount
name: adot-collector-kubeprometheus
namespace: adot-collector-kubeprometheus
@@ -1,7 +0,0 @@
ampurl: ${amp_url}
region: ${region}
prometheusMetricsEndpoint: ${prometheus_metrics_endpoint}
prometheusMetricsPort: ${prometheus_metrics_port}
scrapeInterval: ${scrape_interval}
scrapeTimeout: ${scrape_timeout}
scrapeSampleLimit: ${scrape_sample_limit}
@@ -1,100 +0,0 @@
# ADOT variable
variable "helm_config" {
description = "Helm Config for Prometheus"
type = any
default = {}
}
variable "amp_endpoint" {
description = "Amazon Managed Prometheus Workspace Endpoint"
type = string
default = null
}
variable "amp_id" {
description = "Amazon Managed Prometheus Workspace ID"
type = string
default = null
}
variable "amp_region" {
description = "Amazon Managed Prometheus Workspace's Region"
type = string
default = null
}
variable "dashboards_folder_id" {
type = string
}
variable "addon_context" {
description = "Input configuration for the addon"
type = object({
aws_caller_identity_account_id = string
aws_caller_identity_arn = string
aws_eks_cluster_endpoint = string
aws_partition_id = string
aws_region_name = string
eks_cluster_id = string
eks_oidc_issuer_url = string
eks_oidc_provider_arn = string
irsa_iam_permissions_boundary = string
irsa_iam_role_path = string
tags = map(string)
})
}
variable "config" {
type = object({
helm_config = map(any)
enable_kube_state_metrics = bool
kms_create_namespace = bool
ksm_k8s_namespace = string
ksm_helm_chart_name = string
ksm_helm_chart_version = string
ksm_helm_release_name = string
ksm_helm_repo_url = string
ksm_helm_settings = map(string)
ksm_helm_values = map(any)
enable_node_exporter = bool
ne_create_namespace = bool
ne_k8s_namespace = string
ne_helm_chart_name = string
ne_helm_chart_version = string
ne_helm_release_name = string
ne_helm_repo_url = string
ne_helm_settings = map(string)
ne_helm_values = map(any)
enable_dashboards = bool
})
default = {
enable_kube_state_metrics = true
enable_node_exporter = true
enable_dashboards = true
helm_config = {}
kms_create_namespace = true
ksm_helm_chart_name = "kube-state-metrics"
ksm_helm_chart_version = "4.9.2"
ksm_helm_release_name = "kube-state-metrics"
ksm_helm_repo_url = "https://prometheus-community.github.io/helm-charts"
ksm_helm_settings = {}
ksm_helm_values = {}
ksm_k8s_namespace = "kube-system"
ne_create_namespace = true
ne_k8s_namespace = "prometheus-node-exporter"
ne_helm_chart_name = "prometheus-node-exporter"
ne_helm_chart_version = "2.0.3"
ne_helm_release_name = "prometheus-node-exporter"
ne_helm_repo_url = "https://prometheus-community.github.io/helm-charts"
ne_helm_settings = {}
ne_helm_values = {}
}
nullable = false
}