💡Setup alertmanager and recording rules

Activate Alertmanager at workspace level, recording rules per workload
This commit is contained in:
Rodrigue Koffi
2022-07-27 23:37:00 +02:00
parent b7e909c92d
commit 333d46cccc
6 changed files with 82 additions and 18 deletions
+28
View File
@@ -7,6 +7,7 @@ locals {
data "aws_partition" "current" {}
# deploys collector
module "helm_addon" {
source = "github.com/aws-ia/terraform-aws-eks-blueprints/modules/kubernetes-addons/helm-addon"
@@ -63,3 +64,30 @@ module "helm_addon" {
addon_context = var.addon_context
}
resource "aws_prometheus_rule_group_namespace" "this" {
count = var.enable_recording_rules ? 1 : 0
name = "java_rules"
workspace_id = var.amp_id
data = <<EOF
groups:
- name: default-metric
rules:
- record: metric:recording_rule
expr: avg(rate(container_cpu_usage_seconds_total[5m]))
- name: default-alert
rules:
- alert: metric:alerting_rule
expr: jvm_memory_bytes_used{job="java", area="heap"} / jvm_memory_bytes_max * 100 > 80
for: 1m
labels:
severity: warning
annotations:
summary: "JVM heap warning"
description: "JVM heap of instance `{{$labels.instance}}` from application `{{$labels.application}}` is above 80% for one minute. (current=`{{$value}}%`)"
EOF
}
# dashboard
+4 -5
View File
@@ -1,8 +1,7 @@
variable "java" {
default = {
a = ""
b = ""
}
variable "enable_recording_rules" {
description = "Enable AMP recording rules"
type = bool
default = true
}
variable "amp_endpoint" {