mirror of
https://github.com/storytold/terraform-aws-observability-accelerator.git
synced 2026-10-09 00:09:43 +00:00
Nginx module source (#35)
* working module * metrics work * Add dynamic targets * dashboards and rules * readme and outputs * updating readme * Update scrape config - Drop unused go metrics - Drop empty labels - Add host, container and namespace labels * Update tags and query labels Co-authored-by: EC2 Default User <ec2-user@ip-172-31-9-72.us-west-2.compute.internal> Co-authored-by: Rodrigue Koffi <bonclay7@users.noreply.github.com>
This commit is contained in:
@@ -0,0 +1,5 @@
|
||||
resource "grafana_dashboard" "workloads" {
|
||||
count = var.enable_dashboards ? 1 : 0
|
||||
folder = var.dashboards_folder_id
|
||||
config_json = file("${path.module}/dashboards/nginx.json")
|
||||
}
|
||||
+428
-440
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,31 @@
|
||||
data "aws_partition" "current" {}
|
||||
|
||||
data "aws_caller_identity" "current" {}
|
||||
|
||||
data "aws_region" "current" {}
|
||||
|
||||
data "aws_eks_cluster" "eks_cluster" {
|
||||
name = var.eks_cluster_id
|
||||
}
|
||||
|
||||
locals {
|
||||
name = "adot-collector-nginx"
|
||||
namespace = try(var.config.helm_config.namespace, local.name)
|
||||
|
||||
eks_oidc_issuer_url = replace(data.aws_eks_cluster.eks_cluster.identity[0].oidc[0].issuer, "https://", "")
|
||||
eks_cluster_endpoint = data.aws_eks_cluster.eks_cluster.endpoint
|
||||
|
||||
context = {
|
||||
aws_caller_identity_account_id = data.aws_caller_identity.current.account_id
|
||||
aws_caller_identity_arn = data.aws_caller_identity.current.arn
|
||||
aws_eks_cluster_endpoint = local.eks_cluster_endpoint
|
||||
aws_partition_id = data.aws_partition.current.partition
|
||||
aws_region_name = data.aws_region.current.name
|
||||
eks_cluster_id = var.eks_cluster_id
|
||||
eks_oidc_issuer_url = local.eks_oidc_issuer_url
|
||||
eks_oidc_provider_arn = "arn:${data.aws_partition.current.partition}:iam::${data.aws_caller_identity.current.account_id}:oidc-provider/${local.eks_oidc_issuer_url}"
|
||||
tags = var.tags
|
||||
irsa_iam_role_path = var.irsa_iam_role_path
|
||||
irsa_iam_permissions_boundary = var.irsa_iam_permissions_boundary
|
||||
}
|
||||
}
|
||||
@@ -1,10 +1,3 @@
|
||||
locals {
|
||||
name = "adot-collector-nginx"
|
||||
namespace = try(var.helm_config.namespace, local.name)
|
||||
}
|
||||
|
||||
data "aws_partition" "current" {}
|
||||
|
||||
module "helm_addon" {
|
||||
source = "github.com/aws-ia/terraform-aws-eks-blueprints/modules/kubernetes-addons/helm-addon"
|
||||
|
||||
@@ -22,11 +15,11 @@ module "helm_addon" {
|
||||
set_values = [
|
||||
{
|
||||
name = "ampurl"
|
||||
value = "${var.amazon_prometheus_workspace_endpoint}api/v1/remote_write"
|
||||
value = "${var.managed_prometheus_workspace_endpoint}api/v1/remote_write"
|
||||
},
|
||||
{
|
||||
name = "region"
|
||||
value = var.amazon_prometheus_workspace_region
|
||||
value = var.managed_prometheus_workspace_region
|
||||
},
|
||||
{
|
||||
name = "prometheusMetricsEndpoint"
|
||||
@@ -58,5 +51,5 @@ module "helm_addon" {
|
||||
irsa_iam_policies = ["arn:${data.aws_partition.current.partition}:iam::aws:policy/AmazonPrometheusRemoteWriteAccess"]
|
||||
}
|
||||
|
||||
addon_context = var.addon_context
|
||||
addon_context = local.context
|
||||
}
|
||||
|
||||
@@ -3,7 +3,7 @@ kind: OpenTelemetryCollector
|
||||
metadata:
|
||||
name: adot
|
||||
spec:
|
||||
image: public.ecr.aws/aws-observability/aws-otel-collector:latest
|
||||
image: public.ecr.aws/aws-observability/aws-otel-collector:v0.21.1
|
||||
mode: deployment
|
||||
serviceAccount: adot-collector-nginx
|
||||
config: |
|
||||
@@ -13,55 +13,56 @@ spec:
|
||||
global:
|
||||
scrape_interval: {{ .Values.scrapeInterval }}
|
||||
scrape_timeout: {{ .Values.scrapeTimeout }}
|
||||
|
||||
scrape_configs:
|
||||
- job_name: 'kubernetes-pod-nginx'
|
||||
sample_limit: {{ .Values.scrapeSampleLimit }}
|
||||
metrics_path: /{{ .Values.prometheusMetricsEndpoint }}
|
||||
kubernetes_sd_configs:
|
||||
- role: pod
|
||||
relabel_configs:
|
||||
- source_labels: [ __address__ ]
|
||||
action: keep
|
||||
regex: '.*:9404$'
|
||||
- action: labelmap
|
||||
regex: __meta_kubernetes_pod_label_(.+)
|
||||
- action: replace
|
||||
source_labels: [ __meta_kubernetes_namespace ]
|
||||
target_label: Namespace
|
||||
- source_labels: [ __meta_kubernetes_pod_name ]
|
||||
action: replace
|
||||
target_label: pod_name
|
||||
- action: replace
|
||||
source_labels: [ __meta_kubernetes_pod_container_name ]
|
||||
target_label: container_name
|
||||
- action: replace
|
||||
source_labels: [ __meta_kubernetes_pod_controller_kind ]
|
||||
target_label: pod_controller_kind
|
||||
- action: replace
|
||||
source_labels: [ __meta_kubernetes_pod_phase ]
|
||||
target_label: pod_controller_phase
|
||||
metric_relabel_configs:
|
||||
- source_labels: [ __name__ ]
|
||||
regex: 'jvm_gc_collection_seconds.*'
|
||||
action: drop
|
||||
- job_name: 'kubernetes-pod-nginx'
|
||||
sample_limit: {{ .Values.scrapeSampleLimit }}
|
||||
metrics_path: /{{ .Values.prometheusMetricsEndpoint }}
|
||||
kubernetes_sd_configs:
|
||||
- role: pod
|
||||
relabel_configs:
|
||||
- source_labels: [ __address__ ]
|
||||
action: keep
|
||||
regex: '.*:10254$'
|
||||
- source_labels: [__meta_kubernetes_pod_container_name]
|
||||
target_label: container
|
||||
action: replace
|
||||
- source_labels: [__meta_kubernetes_pod_node_name]
|
||||
target_label: host
|
||||
action: replace
|
||||
- source_labels: [__meta_kubernetes_namespace]
|
||||
target_label: namespace
|
||||
action: replace
|
||||
metric_relabel_configs:
|
||||
- source_labels: [__name__]
|
||||
regex: 'go_memstats.*'
|
||||
action: drop
|
||||
- source_labels: [__name__]
|
||||
regex: 'go_gc.*'
|
||||
action: drop
|
||||
- source_labels: [__name__]
|
||||
regex: 'go_threads'
|
||||
action: drop
|
||||
- regex: exported_host
|
||||
action: labeldrop
|
||||
exporters:
|
||||
awsprometheusremotewrite:
|
||||
prometheusremotewrite:
|
||||
endpoint: {{ .Values.ampurl }}
|
||||
aws_auth:
|
||||
region: {{ .Values.region }}
|
||||
service: "aps"
|
||||
auth:
|
||||
authenticator: sigv4auth
|
||||
logging:
|
||||
loglevel: info
|
||||
loglevel: debug
|
||||
extensions:
|
||||
sigv4auth:
|
||||
region: {{ .Values.region }}
|
||||
service: "aps"
|
||||
health_check:
|
||||
pprof:
|
||||
endpoint: :1888
|
||||
zpages:
|
||||
endpoint: :55679
|
||||
service:
|
||||
extensions: [pprof, zpages, health_check]
|
||||
extensions: [pprof, zpages, health_check, sigv4auth]
|
||||
pipelines:
|
||||
metrics:
|
||||
receivers: [prometheus]
|
||||
exporters: [logging, awsprometheusremotewrite]
|
||||
exporters: [logging, prometheusremotewrite]
|
||||
|
||||
@@ -0,0 +1,44 @@
|
||||
# Prioritize recording rules over alerting rules for limits (10)
|
||||
|
||||
################################################################################################################################################
|
||||
# Recording rules ##############################################################################################################################
|
||||
################################################################################################################################################
|
||||
|
||||
resource "aws_prometheus_rule_group_namespace" "recording_rules" {
|
||||
count = var.enable_recording_rules ? 1 : 0
|
||||
name = "acclerator-nginx-rules"
|
||||
workspace_id = var.managed_prometheus_workspace_id
|
||||
data = <<EOF
|
||||
groups:
|
||||
- name: Nginx-HTTP-4xx-error-rate
|
||||
rules:
|
||||
- alert: metric:alerting_rule
|
||||
expr: sum(rate(nginx_http_requests_total{status=~"^4.."}[1m])) / sum(rate(nginx_http_requests_total[1m])) * 100 > 5
|
||||
for: 1m
|
||||
labels:
|
||||
severity: critical
|
||||
annotations:
|
||||
summary: Nginx high HTTP 4xx error rate (instance {{ $labels.instance }})
|
||||
description: "Too many HTTP requests with status 4xx (> 5%)\n VALUE = {{ $value }}\n LABELS = {{ $labels }}"
|
||||
- name: Nginx-HTTP-5xx-error-rate
|
||||
rules:
|
||||
- alert: metric:alerting_rule
|
||||
expr: sum(rate(nginx_http_requests_total{status=~"^5.."}[1m])) / sum(rate(nginx_http_requests_total[1m])) * 100 > 5
|
||||
for: 1m
|
||||
labels:
|
||||
severity: critical
|
||||
annotations:
|
||||
summary: Nginx high HTTP 5xx error rate (instance {{ $labels.instance }})
|
||||
description: "Too many HTTP requests with status 5xx (> 5%)\n VALUE = {{ $value }}\n LABELS = {{ $labels }}"
|
||||
- name: Nginx-high-latency
|
||||
rules:
|
||||
- alert: metric:alerting_rule
|
||||
expr: histogram_quantile(0.99, sum(rate(nginx_http_request_duration_seconds_bucket[2m])) by (host, node)) > 3
|
||||
for: 2m
|
||||
labels:
|
||||
severity: warning
|
||||
annotations:
|
||||
summary: Nginx latency high (instance {{ $labels.instance }})
|
||||
description: "Nginx p99 latency is higher than 3 seconds\n VALUE = {{ $value }}\n LABELS = {{ $labels }}"
|
||||
EOF
|
||||
}
|
||||
@@ -1,34 +1,128 @@
|
||||
|
||||
variable "eks_cluster_id" {
|
||||
description = "EKS Cluster Id"
|
||||
type = string
|
||||
}
|
||||
|
||||
variable "helm_config" {
|
||||
description = "Helm Config for Prometheus"
|
||||
type = any
|
||||
default = {}
|
||||
}
|
||||
|
||||
variable "amazon_prometheus_workspace_endpoint" {
|
||||
variable "irsa_iam_role_path" {
|
||||
description = "IAM role path for IRSA roles"
|
||||
type = string
|
||||
default = "/"
|
||||
}
|
||||
|
||||
variable "irsa_iam_permissions_boundary" {
|
||||
description = "IAM permissions boundary for IRSA roles"
|
||||
type = string
|
||||
default = ""
|
||||
}
|
||||
|
||||
variable "managed_prometheus_workspace_endpoint" {
|
||||
description = "Amazon Managed Prometheus Workspace Endpoint"
|
||||
type = string
|
||||
default = null
|
||||
}
|
||||
variable "managed_prometheus_workspace_id" {
|
||||
description = "Amazon Managed Prometheus Workspace ID"
|
||||
type = string
|
||||
default = null
|
||||
}
|
||||
|
||||
variable "amazon_prometheus_workspace_region" {
|
||||
variable "managed_prometheus_workspace_region" {
|
||||
description = "Amazon Managed Prometheus Workspace's Region"
|
||||
type = string
|
||||
default = null
|
||||
}
|
||||
|
||||
variable "addon_context" {
|
||||
description = "Input configuration for the addon"
|
||||
type = object({
|
||||
aws_caller_identity_account_id = string
|
||||
aws_caller_identity_arn = string
|
||||
aws_eks_cluster_endpoint = string
|
||||
aws_partition_id = string
|
||||
aws_region_name = string
|
||||
eks_cluster_id = string
|
||||
eks_oidc_issuer_url = string
|
||||
eks_oidc_provider_arn = string
|
||||
irsa_iam_permissions_boundary = string
|
||||
irsa_iam_role_path = string
|
||||
tags = map(string)
|
||||
})
|
||||
variable "dashboards_folder_id" {
|
||||
type = string
|
||||
}
|
||||
|
||||
variable "enable_recording_rules" {
|
||||
type = bool
|
||||
default = true
|
||||
}
|
||||
|
||||
variable "enable_alerting_rules" {
|
||||
type = bool
|
||||
default = true
|
||||
}
|
||||
|
||||
variable "enable_dashboards" {
|
||||
type = bool
|
||||
default = true
|
||||
}
|
||||
|
||||
variable "enable_kube_state_metrics" {
|
||||
type = bool
|
||||
default = true
|
||||
}
|
||||
|
||||
variable "enable_node_exporter" {
|
||||
type = bool
|
||||
default = true
|
||||
}
|
||||
|
||||
variable "config" {
|
||||
type = object({
|
||||
helm_config = map(any)
|
||||
|
||||
kms_create_namespace = bool
|
||||
ksm_k8s_namespace = string
|
||||
ksm_helm_chart_name = string
|
||||
ksm_helm_chart_version = string
|
||||
ksm_helm_release_name = string
|
||||
ksm_helm_repo_url = string
|
||||
ksm_helm_settings = map(string)
|
||||
ksm_helm_values = map(any)
|
||||
|
||||
ne_create_namespace = bool
|
||||
ne_k8s_namespace = string
|
||||
ne_helm_chart_name = string
|
||||
ne_helm_chart_version = string
|
||||
ne_helm_release_name = string
|
||||
ne_helm_repo_url = string
|
||||
ne_helm_settings = map(string)
|
||||
ne_helm_values = map(any)
|
||||
|
||||
})
|
||||
|
||||
default = {
|
||||
enable_kube_state_metrics = true
|
||||
enable_node_exporter = true
|
||||
|
||||
helm_config = {}
|
||||
|
||||
kms_create_namespace = true
|
||||
ksm_helm_chart_name = "kube-state-metrics"
|
||||
ksm_helm_chart_version = "4.9.2"
|
||||
ksm_helm_release_name = "kube-state-metrics"
|
||||
ksm_helm_repo_url = "https://prometheus-community.github.io/helm-charts"
|
||||
ksm_helm_settings = {}
|
||||
ksm_helm_values = {}
|
||||
ksm_k8s_namespace = "kube-system"
|
||||
|
||||
ne_create_namespace = true
|
||||
ne_k8s_namespace = "prometheus-node-exporter"
|
||||
ne_helm_chart_name = "prometheus-node-exporter"
|
||||
ne_helm_chart_version = "2.0.3"
|
||||
ne_helm_release_name = "prometheus-node-exporter"
|
||||
ne_helm_repo_url = "https://prometheus-community.github.io/helm-charts"
|
||||
ne_helm_settings = {}
|
||||
ne_helm_values = {}
|
||||
}
|
||||
nullable = false
|
||||
}
|
||||
|
||||
variable "tags" {
|
||||
description = "Additional tags (e.g. `map('BusinessUnit`,`XYZ`)"
|
||||
type = map(string)
|
||||
default = {}
|
||||
}
|
||||
|
||||
|
||||
|
||||
@@ -10,5 +10,9 @@ terraform {
|
||||
source = "hashicorp/kubernetes"
|
||||
version = ">= 2.10"
|
||||
}
|
||||
grafana = {
|
||||
source = "grafana/grafana"
|
||||
version = ">= 1.25.0"
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user