Nginx module source (#35)

* working module

* metrics work

* Add dynamic targets

* dashboards and rules

* readme and outputs

* updating readme

* Update scrape config

- Drop unused go metrics
- Drop empty labels
- Add host, container and namespace labels

* Update tags and query labels

Co-authored-by: EC2 Default User <ec2-user@ip-172-31-9-72.us-west-2.compute.internal>
Co-authored-by: Rodrigue Koffi <bonclay7@users.noreply.github.com>
This commit is contained in:
Kevin Lewin
2022-09-27 10:00:55 -04:00
committed by GitHub
parent d446f82955
commit 9f0b90ab2f
13 changed files with 1076 additions and 508 deletions
+5
View File
@@ -0,0 +1,5 @@
resource "grafana_dashboard" "workloads" {
count = var.enable_dashboards ? 1 : 0
folder = var.dashboards_folder_id
config_json = file("${path.module}/dashboards/nginx.json")
}
+31
View File
@@ -0,0 +1,31 @@
data "aws_partition" "current" {}
data "aws_caller_identity" "current" {}
data "aws_region" "current" {}
data "aws_eks_cluster" "eks_cluster" {
name = var.eks_cluster_id
}
locals {
name = "adot-collector-nginx"
namespace = try(var.config.helm_config.namespace, local.name)
eks_oidc_issuer_url = replace(data.aws_eks_cluster.eks_cluster.identity[0].oidc[0].issuer, "https://", "")
eks_cluster_endpoint = data.aws_eks_cluster.eks_cluster.endpoint
context = {
aws_caller_identity_account_id = data.aws_caller_identity.current.account_id
aws_caller_identity_arn = data.aws_caller_identity.current.arn
aws_eks_cluster_endpoint = local.eks_cluster_endpoint
aws_partition_id = data.aws_partition.current.partition
aws_region_name = data.aws_region.current.name
eks_cluster_id = var.eks_cluster_id
eks_oidc_issuer_url = local.eks_oidc_issuer_url
eks_oidc_provider_arn = "arn:${data.aws_partition.current.partition}:iam::${data.aws_caller_identity.current.account_id}:oidc-provider/${local.eks_oidc_issuer_url}"
tags = var.tags
irsa_iam_role_path = var.irsa_iam_role_path
irsa_iam_permissions_boundary = var.irsa_iam_permissions_boundary
}
}
+3 -10
View File
@@ -1,10 +1,3 @@
locals {
name = "adot-collector-nginx"
namespace = try(var.helm_config.namespace, local.name)
}
data "aws_partition" "current" {}
module "helm_addon" {
source = "github.com/aws-ia/terraform-aws-eks-blueprints/modules/kubernetes-addons/helm-addon"
@@ -22,11 +15,11 @@ module "helm_addon" {
set_values = [
{
name = "ampurl"
value = "${var.amazon_prometheus_workspace_endpoint}api/v1/remote_write"
value = "${var.managed_prometheus_workspace_endpoint}api/v1/remote_write"
},
{
name = "region"
value = var.amazon_prometheus_workspace_region
value = var.managed_prometheus_workspace_region
},
{
name = "prometheusMetricsEndpoint"
@@ -58,5 +51,5 @@ module "helm_addon" {
irsa_iam_policies = ["arn:${data.aws_partition.current.partition}:iam::aws:policy/AmazonPrometheusRemoteWriteAccess"]
}
addon_context = var.addon_context
addon_context = local.context
}
@@ -3,7 +3,7 @@ kind: OpenTelemetryCollector
metadata:
name: adot
spec:
image: public.ecr.aws/aws-observability/aws-otel-collector:latest
image: public.ecr.aws/aws-observability/aws-otel-collector:v0.21.1
mode: deployment
serviceAccount: adot-collector-nginx
config: |
@@ -13,55 +13,56 @@ spec:
global:
scrape_interval: {{ .Values.scrapeInterval }}
scrape_timeout: {{ .Values.scrapeTimeout }}
scrape_configs:
- job_name: 'kubernetes-pod-nginx'
sample_limit: {{ .Values.scrapeSampleLimit }}
metrics_path: /{{ .Values.prometheusMetricsEndpoint }}
kubernetes_sd_configs:
- role: pod
relabel_configs:
- source_labels: [ __address__ ]
action: keep
regex: '.*:9404$'
- action: labelmap
regex: __meta_kubernetes_pod_label_(.+)
- action: replace
source_labels: [ __meta_kubernetes_namespace ]
target_label: Namespace
- source_labels: [ __meta_kubernetes_pod_name ]
action: replace
target_label: pod_name
- action: replace
source_labels: [ __meta_kubernetes_pod_container_name ]
target_label: container_name
- action: replace
source_labels: [ __meta_kubernetes_pod_controller_kind ]
target_label: pod_controller_kind
- action: replace
source_labels: [ __meta_kubernetes_pod_phase ]
target_label: pod_controller_phase
metric_relabel_configs:
- source_labels: [ __name__ ]
regex: 'jvm_gc_collection_seconds.*'
action: drop
- job_name: 'kubernetes-pod-nginx'
sample_limit: {{ .Values.scrapeSampleLimit }}
metrics_path: /{{ .Values.prometheusMetricsEndpoint }}
kubernetes_sd_configs:
- role: pod
relabel_configs:
- source_labels: [ __address__ ]
action: keep
regex: '.*:10254$'
- source_labels: [__meta_kubernetes_pod_container_name]
target_label: container
action: replace
- source_labels: [__meta_kubernetes_pod_node_name]
target_label: host
action: replace
- source_labels: [__meta_kubernetes_namespace]
target_label: namespace
action: replace
metric_relabel_configs:
- source_labels: [__name__]
regex: 'go_memstats.*'
action: drop
- source_labels: [__name__]
regex: 'go_gc.*'
action: drop
- source_labels: [__name__]
regex: 'go_threads'
action: drop
- regex: exported_host
action: labeldrop
exporters:
awsprometheusremotewrite:
prometheusremotewrite:
endpoint: {{ .Values.ampurl }}
aws_auth:
region: {{ .Values.region }}
service: "aps"
auth:
authenticator: sigv4auth
logging:
loglevel: info
loglevel: debug
extensions:
sigv4auth:
region: {{ .Values.region }}
service: "aps"
health_check:
pprof:
endpoint: :1888
zpages:
endpoint: :55679
service:
extensions: [pprof, zpages, health_check]
extensions: [pprof, zpages, health_check, sigv4auth]
pipelines:
metrics:
receivers: [prometheus]
exporters: [logging, awsprometheusremotewrite]
exporters: [logging, prometheusremotewrite]
+44
View File
@@ -0,0 +1,44 @@
# Prioritize recording rules over alerting rules for limits (10)
################################################################################################################################################
# Recording rules ##############################################################################################################################
################################################################################################################################################
resource "aws_prometheus_rule_group_namespace" "recording_rules" {
count = var.enable_recording_rules ? 1 : 0
name = "acclerator-nginx-rules"
workspace_id = var.managed_prometheus_workspace_id
data = <<EOF
groups:
- name: Nginx-HTTP-4xx-error-rate
rules:
- alert: metric:alerting_rule
expr: sum(rate(nginx_http_requests_total{status=~"^4.."}[1m])) / sum(rate(nginx_http_requests_total[1m])) * 100 > 5
for: 1m
labels:
severity: critical
annotations:
summary: Nginx high HTTP 4xx error rate (instance {{ $labels.instance }})
description: "Too many HTTP requests with status 4xx (> 5%)\n VALUE = {{ $value }}\n LABELS = {{ $labels }}"
- name: Nginx-HTTP-5xx-error-rate
rules:
- alert: metric:alerting_rule
expr: sum(rate(nginx_http_requests_total{status=~"^5.."}[1m])) / sum(rate(nginx_http_requests_total[1m])) * 100 > 5
for: 1m
labels:
severity: critical
annotations:
summary: Nginx high HTTP 5xx error rate (instance {{ $labels.instance }})
description: "Too many HTTP requests with status 5xx (> 5%)\n VALUE = {{ $value }}\n LABELS = {{ $labels }}"
- name: Nginx-high-latency
rules:
- alert: metric:alerting_rule
expr: histogram_quantile(0.99, sum(rate(nginx_http_request_duration_seconds_bucket[2m])) by (host, node)) > 3
for: 2m
labels:
severity: warning
annotations:
summary: Nginx latency high (instance {{ $labels.instance }})
description: "Nginx p99 latency is higher than 3 seconds\n VALUE = {{ $value }}\n LABELS = {{ $labels }}"
EOF
}
+111 -17
View File
@@ -1,34 +1,128 @@
variable "eks_cluster_id" {
description = "EKS Cluster Id"
type = string
}
variable "helm_config" {
description = "Helm Config for Prometheus"
type = any
default = {}
}
variable "amazon_prometheus_workspace_endpoint" {
variable "irsa_iam_role_path" {
description = "IAM role path for IRSA roles"
type = string
default = "/"
}
variable "irsa_iam_permissions_boundary" {
description = "IAM permissions boundary for IRSA roles"
type = string
default = ""
}
variable "managed_prometheus_workspace_endpoint" {
description = "Amazon Managed Prometheus Workspace Endpoint"
type = string
default = null
}
variable "managed_prometheus_workspace_id" {
description = "Amazon Managed Prometheus Workspace ID"
type = string
default = null
}
variable "amazon_prometheus_workspace_region" {
variable "managed_prometheus_workspace_region" {
description = "Amazon Managed Prometheus Workspace's Region"
type = string
default = null
}
variable "addon_context" {
description = "Input configuration for the addon"
type = object({
aws_caller_identity_account_id = string
aws_caller_identity_arn = string
aws_eks_cluster_endpoint = string
aws_partition_id = string
aws_region_name = string
eks_cluster_id = string
eks_oidc_issuer_url = string
eks_oidc_provider_arn = string
irsa_iam_permissions_boundary = string
irsa_iam_role_path = string
tags = map(string)
})
variable "dashboards_folder_id" {
type = string
}
variable "enable_recording_rules" {
type = bool
default = true
}
variable "enable_alerting_rules" {
type = bool
default = true
}
variable "enable_dashboards" {
type = bool
default = true
}
variable "enable_kube_state_metrics" {
type = bool
default = true
}
variable "enable_node_exporter" {
type = bool
default = true
}
variable "config" {
type = object({
helm_config = map(any)
kms_create_namespace = bool
ksm_k8s_namespace = string
ksm_helm_chart_name = string
ksm_helm_chart_version = string
ksm_helm_release_name = string
ksm_helm_repo_url = string
ksm_helm_settings = map(string)
ksm_helm_values = map(any)
ne_create_namespace = bool
ne_k8s_namespace = string
ne_helm_chart_name = string
ne_helm_chart_version = string
ne_helm_release_name = string
ne_helm_repo_url = string
ne_helm_settings = map(string)
ne_helm_values = map(any)
})
default = {
enable_kube_state_metrics = true
enable_node_exporter = true
helm_config = {}
kms_create_namespace = true
ksm_helm_chart_name = "kube-state-metrics"
ksm_helm_chart_version = "4.9.2"
ksm_helm_release_name = "kube-state-metrics"
ksm_helm_repo_url = "https://prometheus-community.github.io/helm-charts"
ksm_helm_settings = {}
ksm_helm_values = {}
ksm_k8s_namespace = "kube-system"
ne_create_namespace = true
ne_k8s_namespace = "prometheus-node-exporter"
ne_helm_chart_name = "prometheus-node-exporter"
ne_helm_chart_version = "2.0.3"
ne_helm_release_name = "prometheus-node-exporter"
ne_helm_repo_url = "https://prometheus-community.github.io/helm-charts"
ne_helm_settings = {}
ne_helm_values = {}
}
nullable = false
}
variable "tags" {
description = "Additional tags (e.g. `map('BusinessUnit`,`XYZ`)"
type = map(string)
default = {}
}
+4
View File
@@ -10,5 +10,9 @@ terraform {
source = "hashicorp/kubernetes"
version = ">= 2.10"
}
grafana = {
source = "grafana/grafana"
version = ">= 1.25.0"
}
}
}