mirror of
https://github.com/storytold/terraform-aws-observability-accelerator.git
synced 2026-10-09 00:09:43 +00:00
initial commit
This commit is contained in:
committed by
Rodrigue Koffi
parent
0262a8a45d
commit
d654fd36ae
@@ -1,29 +0,0 @@
|
|||||||
<!-- BEGIN_TF_DOCS -->
|
|
||||||
## Requirements
|
|
||||||
|
|
||||||
| Name | Version |
|
|
||||||
|------|---------|
|
|
||||||
| <a name="requirement_terraform"></a> [terraform](#requirement\_terraform) | >= 0.14.0 |
|
|
||||||
| <a name="requirement_aws"></a> [aws](#requirement\_aws) | >= 3.72.0 |
|
|
||||||
| <a name="requirement_awscc"></a> [awscc](#requirement\_awscc) | >= 0.11.0 |
|
|
||||||
|
|
||||||
## Providers
|
|
||||||
|
|
||||||
No providers.
|
|
||||||
|
|
||||||
## Modules
|
|
||||||
|
|
||||||
No modules.
|
|
||||||
|
|
||||||
## Resources
|
|
||||||
|
|
||||||
No resources.
|
|
||||||
|
|
||||||
## Inputs
|
|
||||||
|
|
||||||
No inputs.
|
|
||||||
|
|
||||||
## Outputs
|
|
||||||
|
|
||||||
No outputs.
|
|
||||||
<!-- END_TF_DOCS -->
|
|
||||||
@@ -1,5 +0,0 @@
|
|||||||
#####################################################################################
|
|
||||||
# Terraform module examples are meant to show an _example_ on how to use a module
|
|
||||||
# per use-case. The code below should not be copied directly but referenced in order
|
|
||||||
# to build your own root module that invokes this module
|
|
||||||
#####################################################################################
|
|
||||||
@@ -1,21 +0,0 @@
|
|||||||
terraform {
|
|
||||||
required_version = ">= 0.14.0"
|
|
||||||
required_providers {
|
|
||||||
aws = {
|
|
||||||
source = "hashicorp/aws"
|
|
||||||
version = ">= 3.72.0"
|
|
||||||
}
|
|
||||||
awscc = {
|
|
||||||
source = "hashicorp/awscc"
|
|
||||||
version = ">= 0.11.0"
|
|
||||||
}
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
provider "awscc" {
|
|
||||||
user_agent = [{
|
|
||||||
product_name = "terraform-awscc-"
|
|
||||||
product_version = "0.0.1"
|
|
||||||
comment = "V1/AWS-D69B4015/<github repo id>"
|
|
||||||
}]
|
|
||||||
}
|
|
||||||
@@ -0,0 +1,24 @@
|
|||||||
|
#---------------------------------------------------------------
|
||||||
|
# EKS Blueprints
|
||||||
|
#---------------------------------------------------------------
|
||||||
|
|
||||||
|
module "eks_blueprints" {
|
||||||
|
source = "../../.."
|
||||||
|
|
||||||
|
cluster_name = local.name
|
||||||
|
cluster_version = "1.22"
|
||||||
|
|
||||||
|
vpc_id = module.vpc.vpc_id
|
||||||
|
private_subnet_ids = module.vpc.private_subnets
|
||||||
|
|
||||||
|
managed_node_groups = {
|
||||||
|
t3_l = {
|
||||||
|
node_group_name = "managed-ondemand"
|
||||||
|
instance_types = ["t3.large"]
|
||||||
|
min_size = 2
|
||||||
|
subnet_ids = module.vpc.private_subnets
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
tags = local.tags
|
||||||
|
}
|
||||||
@@ -0,0 +1,35 @@
|
|||||||
|
module "vpc" {
|
||||||
|
source = "terraform-aws-modules/vpc/aws"
|
||||||
|
version = "~> 3.0"
|
||||||
|
|
||||||
|
name = local.name
|
||||||
|
cidr = local.vpc_cidr
|
||||||
|
|
||||||
|
azs = local.azs
|
||||||
|
public_subnets = [for k, v in local.azs : cidrsubnet(local.vpc_cidr, 8, k)]
|
||||||
|
private_subnets = [for k, v in local.azs : cidrsubnet(local.vpc_cidr, 8, k + 10)]
|
||||||
|
|
||||||
|
enable_nat_gateway = true
|
||||||
|
single_nat_gateway = true
|
||||||
|
enable_dns_hostnames = true
|
||||||
|
|
||||||
|
# Manage so we can name
|
||||||
|
manage_default_network_acl = true
|
||||||
|
default_network_acl_tags = { Name = "${local.name}-default" }
|
||||||
|
manage_default_route_table = true
|
||||||
|
default_route_table_tags = { Name = "${local.name}-default" }
|
||||||
|
manage_default_security_group = true
|
||||||
|
default_security_group_tags = { Name = "${local.name}-default" }
|
||||||
|
|
||||||
|
public_subnet_tags = {
|
||||||
|
"kubernetes.io/cluster/${local.name}" = "shared"
|
||||||
|
"kubernetes.io/role/elb" = 1
|
||||||
|
}
|
||||||
|
|
||||||
|
private_subnet_tags = {
|
||||||
|
"kubernetes.io/cluster/${local.name}" = "shared"
|
||||||
|
"kubernetes.io/role/internal-elb" = 1
|
||||||
|
}
|
||||||
|
|
||||||
|
tags = local.tags
|
||||||
|
}
|
||||||
@@ -0,0 +1,67 @@
|
|||||||
|
|
||||||
|
|
||||||
|
module "eks_observability_accelerator" {
|
||||||
|
source = "aws-ia/aws-observability-accelerator/terraform/eks"
|
||||||
|
|
||||||
|
# -- or use an existing cluster
|
||||||
|
eks_cluster_id = var.eks_cluster_id
|
||||||
|
|
||||||
|
# enable managed add-on for ADOT. Do we enforce this or let users
|
||||||
|
# have their own configs for OTEL operator
|
||||||
|
enable_amazon_eks_adot = true
|
||||||
|
|
||||||
|
# -- or enable opentelemetry operator
|
||||||
|
enable_open_telemetry_operator = true
|
||||||
|
open_telemetry_operator_config = map() // custom config
|
||||||
|
|
||||||
|
# deploy selected workloads by count indexing
|
||||||
|
|
||||||
|
# this creates a new AMP workspace
|
||||||
|
create_managed_prometheus_workspace = true
|
||||||
|
|
||||||
|
enable_haproxy = true
|
||||||
|
haproxy_config = {
|
||||||
|
amp_endpoint = module/amp.endpoint
|
||||||
|
grafana_endpoint = module.grafana.endpoint
|
||||||
|
}
|
||||||
|
|
||||||
|
enable_java = true
|
||||||
|
java_config = {
|
||||||
|
amp_endpoint = ""
|
||||||
|
grafana_endpoint = ""
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
# -- or use an existing one
|
||||||
|
# seems like https://github.com/terraform-aws-modules/terraform-aws-managed-service-prometheus
|
||||||
|
# supports importing
|
||||||
|
amp_workspace_alias = var.amp_alias
|
||||||
|
|
||||||
|
# enable rules and alerts
|
||||||
|
enable_alert_manager = true
|
||||||
|
|
||||||
|
# -- or provide custom alerts definition
|
||||||
|
prometheus_custom_alert_rule = var.prometheus_custom_alert_rule
|
||||||
|
|
||||||
|
|
||||||
|
# create grafana workspace, and customer to deal with authentication later
|
||||||
|
create_managed_grafana_workspace = true
|
||||||
|
grafana_auth_provider = var.grafana_auth_provider //SAML or AWS_SSO
|
||||||
|
grafana_account_access_type = var.grafana_account_access_type // CURRENT_ACCOUNT or ORGANIZATION
|
||||||
|
grafana_permission_type = var.grafana_permission_type // SERVICE_MANAGED or CUSTOMER_MANAGED
|
||||||
|
grafana_permission_role_arn = var.grafana_permission_role_arn // if CUSTOMER_MANAGED
|
||||||
|
|
||||||
|
# -- or using existing amg workspace. so we can use API for keys
|
||||||
|
managed_grafana_workspace_id = var.managed_grafana_workspace_id
|
||||||
|
|
||||||
|
}
|
||||||
|
|
||||||
|
module "amp" {
|
||||||
|
|
||||||
|
}
|
||||||
|
|
||||||
|
module "grafana" {
|
||||||
|
|
||||||
|
}
|
||||||
@@ -0,0 +1,17 @@
|
|||||||
|
|
||||||
|
# DONT create the resources
|
||||||
|
# VPC and supporting resources
|
||||||
|
# EKS and Managed node groups
|
||||||
|
|
||||||
|
#
|
||||||
|
|
||||||
|
module "java" {
|
||||||
|
source = ""
|
||||||
|
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
module "haproxy" {
|
||||||
|
source = ""
|
||||||
|
|
||||||
|
}
|
||||||
@@ -0,0 +1,172 @@
|
|||||||
|
|
||||||
|
#---------------------------------------------------------------
|
||||||
|
# Observability Resources
|
||||||
|
#---------------------------------------------------------------
|
||||||
|
|
||||||
|
module "managed_grafana" {
|
||||||
|
source = "terraform-aws-modules/managed-service-grafana/aws"
|
||||||
|
version = "~> 1.3"
|
||||||
|
|
||||||
|
# Workspace
|
||||||
|
name = local.name
|
||||||
|
stack_set_name = local.name
|
||||||
|
data_sources = ["PROMETHEUS"]
|
||||||
|
associate_license = false
|
||||||
|
|
||||||
|
# # Role associations
|
||||||
|
# Pending https://github.com/hashicorp/terraform-provider-aws/issues/24166
|
||||||
|
# role_associations = {
|
||||||
|
# "ADMIN" = {
|
||||||
|
# "group_ids" = []
|
||||||
|
# "user_ids" = []
|
||||||
|
# }
|
||||||
|
# "EDITOR" = {
|
||||||
|
# "group_ids" = []
|
||||||
|
# "user_ids" = []
|
||||||
|
# }
|
||||||
|
# }
|
||||||
|
|
||||||
|
tags = local.tags
|
||||||
|
}
|
||||||
|
|
||||||
|
resource "grafana_data_source" "prometheus" {
|
||||||
|
type = "prometheus"
|
||||||
|
name = "amp"
|
||||||
|
is_default = true
|
||||||
|
url = module.managed_prometheus.workspace_prometheus_endpoint
|
||||||
|
|
||||||
|
json_data {
|
||||||
|
http_method = "GET"
|
||||||
|
sigv4_auth = true
|
||||||
|
sigv4_auth_type = "workspace-iam-role"
|
||||||
|
sigv4_region = local.region
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
resource "grafana_folder" "this" {
|
||||||
|
title = "Observability"
|
||||||
|
}
|
||||||
|
|
||||||
|
resource "grafana_dashboard" "this" {
|
||||||
|
folder = grafana_folder.this.id
|
||||||
|
config_json = file("${path.module}/dashboards/default.json")
|
||||||
|
}
|
||||||
|
|
||||||
|
module "managed_prometheus" {
|
||||||
|
source = "terraform-aws-modules/managed-service-prometheus/aws"
|
||||||
|
version = "~> 2.1"
|
||||||
|
|
||||||
|
workspace_alias = local.name
|
||||||
|
|
||||||
|
alert_manager_definition = <<-EOT
|
||||||
|
alertmanager_config: |
|
||||||
|
route:
|
||||||
|
receiver: 'default'
|
||||||
|
receivers:
|
||||||
|
- name: 'default'
|
||||||
|
EOT
|
||||||
|
|
||||||
|
rule_group_namespaces = {
|
||||||
|
haproxy = {
|
||||||
|
name = "haproxy_rules"
|
||||||
|
data = <<-EOT
|
||||||
|
groups:
|
||||||
|
- name: obsa-haproxy-down-alert
|
||||||
|
rules:
|
||||||
|
- alert: HA_proxy_down
|
||||||
|
expr: haproxy_up == 0
|
||||||
|
for: 0m
|
||||||
|
labels:
|
||||||
|
severity: critical
|
||||||
|
annotations:
|
||||||
|
summary: HAProxy down (instance {{ $labels.instance }})
|
||||||
|
description: "HAProxy down\n VALUE = {{ $value }}\n LABELS = {{ $labels }}"
|
||||||
|
- name: obsa-haproxy-http4xx-error-alert
|
||||||
|
rules:
|
||||||
|
- alert: Ha_proxy_High_Http4xx_ErrorRate_Backend
|
||||||
|
expr: sum by (backend) (rate(haproxy_server_http_responses_total{code="4xx"}[1m])) / sum by (backend) (rate(haproxy_server_http_responses_total[1m]) * 100) > 5
|
||||||
|
for: 1m
|
||||||
|
labels:
|
||||||
|
severity: critical
|
||||||
|
annotations:
|
||||||
|
summary: HAProxy high HTTP 4xx error rate backend (instance {{ $labels.instance }})
|
||||||
|
description: "Too many HTTP requests with status 4xx (> 5%) on backend {{ $labels.fqdn }}/{{ $labels.backend }}\n VALUE = {{ $value }}\n LABELS = {{ $labels }}"
|
||||||
|
- name: obsa-haproxy-http5xx-error-alert
|
||||||
|
rules:
|
||||||
|
- alert: Ha_proxy_High_Http5xx_ErrorRate_Backend
|
||||||
|
expr: sum by (backend) (rate(haproxy_server_http_responses_total{code="5xx"}[1m])) / sum by (backend) (rate(haproxy_server_http_responses_total[1m]) * 100) > 5
|
||||||
|
for: 1m
|
||||||
|
labels:
|
||||||
|
severity: critical
|
||||||
|
annotations:
|
||||||
|
summary: HAProxy high HTTP 5xx error rate backend (instance {{ $labels.instance }})
|
||||||
|
description: "Too many HTTP requests with status 5xx (> 5%) on backend {{ $labels.fqdn }}/{{ $labels.backend }}\n VALUE = {{ $value }}\n LABELS = {{ $labels }}"
|
||||||
|
- name: obsa-haproxy-Http4xx-ErrorRate-Server-alert
|
||||||
|
rules:
|
||||||
|
- alert: Ha_proxy_High_Http4xx_ErrorRate_Server
|
||||||
|
expr: sum by (server) (rate(haproxy_server_http_responses_total{code="4xx"}[1m])) / sum by (server) (rate(haproxy_server_http_responses_total[1m]) * 100) > 5
|
||||||
|
for: 1m
|
||||||
|
labels:
|
||||||
|
severity: critical
|
||||||
|
annotations:
|
||||||
|
summary: HAProxy high HTTP 4xx error rate server (instance {{ $labels.instance }})
|
||||||
|
description: "Too many HTTP requests with status 4xx (> 5%) on server {{ $labels.server }}\n VALUE = {{ $value }}\n LABELS = {{ $labels }}"
|
||||||
|
- name: obsa-haproxy-Http5xx-ErrorRate-Server-alert
|
||||||
|
rules:
|
||||||
|
- alert: Ha_proxy_High_Http5xx_ErrorRate_Server
|
||||||
|
expr: sum by (server) (rate(haproxy_server_http_responses_total{code="5xx"}[1m])) / sum by (server) (rate(haproxy_server_http_responses_total[1m]) * 100) > 5
|
||||||
|
for: 1m
|
||||||
|
labels:
|
||||||
|
severity: critical
|
||||||
|
annotations:
|
||||||
|
summary: HAProxy high HTTP 5xx error rate server (instance {{ $labels.instance }})
|
||||||
|
description: "Too many HTTP requests with status 5xx (> 5%) on server {{ $labels.server }}\n VALUE = {{ $value }}\n LABELS = {{ $labels }}"
|
||||||
|
EOT
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
tags = local.tags
|
||||||
|
}
|
||||||
|
|
||||||
|
#---------------------------------------------------------------
|
||||||
|
# Sample Application
|
||||||
|
#---------------------------------------------------------------
|
||||||
|
|
||||||
|
# https://github.com/haproxy-ingress/charts/tree/master/haproxy-ingress
|
||||||
|
resource "helm_release" "haproxy_ingress" {
|
||||||
|
namespace = "haproxy-ingress"
|
||||||
|
create_namespace = true
|
||||||
|
|
||||||
|
name = "haproxy-ingress"
|
||||||
|
repository = "https://haproxy-ingress.github.io/charts"
|
||||||
|
chart = "haproxy-ingress"
|
||||||
|
version = "0.13.7"
|
||||||
|
|
||||||
|
set {
|
||||||
|
name = "defaultBackend.enabled"
|
||||||
|
value = true
|
||||||
|
}
|
||||||
|
|
||||||
|
set {
|
||||||
|
name = "controller.stats.enabled"
|
||||||
|
value = true
|
||||||
|
}
|
||||||
|
|
||||||
|
set {
|
||||||
|
name = "controller.metrics.enabled"
|
||||||
|
value = true
|
||||||
|
}
|
||||||
|
|
||||||
|
set {
|
||||||
|
name = "controller.metrics.service.annotations.prometheus\\.io/port"
|
||||||
|
value = 9101
|
||||||
|
type = "string"
|
||||||
|
}
|
||||||
|
|
||||||
|
set {
|
||||||
|
name = "controller.metrics.service.annotations.prometheus\\.io/scrape"
|
||||||
|
value = true
|
||||||
|
type = "string"
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
@@ -0,0 +1,6 @@
|
|||||||
|
variable "java" {
|
||||||
|
default = {
|
||||||
|
a = ""
|
||||||
|
b = ""
|
||||||
|
}
|
||||||
|
}
|
||||||
Reference in New Issue
Block a user