mirror of
https://github.com/storytold/terraform-aws-observability-accelerator.git
synced 2026-10-09 00:09:43 +00:00
initial commit
This commit is contained in:
committed by
Rodrigue Koffi
parent
0262a8a45d
commit
d654fd36ae
@@ -1,29 +0,0 @@
|
||||
<!-- BEGIN_TF_DOCS -->
|
||||
## Requirements
|
||||
|
||||
| Name | Version |
|
||||
|------|---------|
|
||||
| <a name="requirement_terraform"></a> [terraform](#requirement\_terraform) | >= 0.14.0 |
|
||||
| <a name="requirement_aws"></a> [aws](#requirement\_aws) | >= 3.72.0 |
|
||||
| <a name="requirement_awscc"></a> [awscc](#requirement\_awscc) | >= 0.11.0 |
|
||||
|
||||
## Providers
|
||||
|
||||
No providers.
|
||||
|
||||
## Modules
|
||||
|
||||
No modules.
|
||||
|
||||
## Resources
|
||||
|
||||
No resources.
|
||||
|
||||
## Inputs
|
||||
|
||||
No inputs.
|
||||
|
||||
## Outputs
|
||||
|
||||
No outputs.
|
||||
<!-- END_TF_DOCS -->
|
||||
@@ -1,5 +0,0 @@
|
||||
#####################################################################################
|
||||
# Terraform module examples are meant to show an _example_ on how to use a module
|
||||
# per use-case. The code below should not be copied directly but referenced in order
|
||||
# to build your own root module that invokes this module
|
||||
#####################################################################################
|
||||
@@ -1,21 +0,0 @@
|
||||
terraform {
|
||||
required_version = ">= 0.14.0"
|
||||
required_providers {
|
||||
aws = {
|
||||
source = "hashicorp/aws"
|
||||
version = ">= 3.72.0"
|
||||
}
|
||||
awscc = {
|
||||
source = "hashicorp/awscc"
|
||||
version = ">= 0.11.0"
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
provider "awscc" {
|
||||
user_agent = [{
|
||||
product_name = "terraform-awscc-"
|
||||
product_version = "0.0.1"
|
||||
comment = "V1/AWS-D69B4015/<github repo id>"
|
||||
}]
|
||||
}
|
||||
@@ -0,0 +1,24 @@
|
||||
#---------------------------------------------------------------
|
||||
# EKS Blueprints
|
||||
#---------------------------------------------------------------
|
||||
|
||||
module "eks_blueprints" {
|
||||
source = "../../.."
|
||||
|
||||
cluster_name = local.name
|
||||
cluster_version = "1.22"
|
||||
|
||||
vpc_id = module.vpc.vpc_id
|
||||
private_subnet_ids = module.vpc.private_subnets
|
||||
|
||||
managed_node_groups = {
|
||||
t3_l = {
|
||||
node_group_name = "managed-ondemand"
|
||||
instance_types = ["t3.large"]
|
||||
min_size = 2
|
||||
subnet_ids = module.vpc.private_subnets
|
||||
}
|
||||
}
|
||||
|
||||
tags = local.tags
|
||||
}
|
||||
@@ -0,0 +1,35 @@
|
||||
module "vpc" {
|
||||
source = "terraform-aws-modules/vpc/aws"
|
||||
version = "~> 3.0"
|
||||
|
||||
name = local.name
|
||||
cidr = local.vpc_cidr
|
||||
|
||||
azs = local.azs
|
||||
public_subnets = [for k, v in local.azs : cidrsubnet(local.vpc_cidr, 8, k)]
|
||||
private_subnets = [for k, v in local.azs : cidrsubnet(local.vpc_cidr, 8, k + 10)]
|
||||
|
||||
enable_nat_gateway = true
|
||||
single_nat_gateway = true
|
||||
enable_dns_hostnames = true
|
||||
|
||||
# Manage so we can name
|
||||
manage_default_network_acl = true
|
||||
default_network_acl_tags = { Name = "${local.name}-default" }
|
||||
manage_default_route_table = true
|
||||
default_route_table_tags = { Name = "${local.name}-default" }
|
||||
manage_default_security_group = true
|
||||
default_security_group_tags = { Name = "${local.name}-default" }
|
||||
|
||||
public_subnet_tags = {
|
||||
"kubernetes.io/cluster/${local.name}" = "shared"
|
||||
"kubernetes.io/role/elb" = 1
|
||||
}
|
||||
|
||||
private_subnet_tags = {
|
||||
"kubernetes.io/cluster/${local.name}" = "shared"
|
||||
"kubernetes.io/role/internal-elb" = 1
|
||||
}
|
||||
|
||||
tags = local.tags
|
||||
}
|
||||
@@ -0,0 +1,67 @@
|
||||
|
||||
|
||||
module "eks_observability_accelerator" {
|
||||
source = "aws-ia/aws-observability-accelerator/terraform/eks"
|
||||
|
||||
# -- or use an existing cluster
|
||||
eks_cluster_id = var.eks_cluster_id
|
||||
|
||||
# enable managed add-on for ADOT. Do we enforce this or let users
|
||||
# have their own configs for OTEL operator
|
||||
enable_amazon_eks_adot = true
|
||||
|
||||
# -- or enable opentelemetry operator
|
||||
enable_open_telemetry_operator = true
|
||||
open_telemetry_operator_config = map() // custom config
|
||||
|
||||
# deploy selected workloads by count indexing
|
||||
|
||||
# this creates a new AMP workspace
|
||||
create_managed_prometheus_workspace = true
|
||||
|
||||
enable_haproxy = true
|
||||
haproxy_config = {
|
||||
amp_endpoint = module/amp.endpoint
|
||||
grafana_endpoint = module.grafana.endpoint
|
||||
}
|
||||
|
||||
enable_java = true
|
||||
java_config = {
|
||||
amp_endpoint = ""
|
||||
grafana_endpoint = ""
|
||||
}
|
||||
|
||||
|
||||
|
||||
|
||||
# -- or use an existing one
|
||||
# seems like https://github.com/terraform-aws-modules/terraform-aws-managed-service-prometheus
|
||||
# supports importing
|
||||
amp_workspace_alias = var.amp_alias
|
||||
|
||||
# enable rules and alerts
|
||||
enable_alert_manager = true
|
||||
|
||||
# -- or provide custom alerts definition
|
||||
prometheus_custom_alert_rule = var.prometheus_custom_alert_rule
|
||||
|
||||
|
||||
# create grafana workspace, and customer to deal with authentication later
|
||||
create_managed_grafana_workspace = true
|
||||
grafana_auth_provider = var.grafana_auth_provider //SAML or AWS_SSO
|
||||
grafana_account_access_type = var.grafana_account_access_type // CURRENT_ACCOUNT or ORGANIZATION
|
||||
grafana_permission_type = var.grafana_permission_type // SERVICE_MANAGED or CUSTOMER_MANAGED
|
||||
grafana_permission_role_arn = var.grafana_permission_role_arn // if CUSTOMER_MANAGED
|
||||
|
||||
# -- or using existing amg workspace. so we can use API for keys
|
||||
managed_grafana_workspace_id = var.managed_grafana_workspace_id
|
||||
|
||||
}
|
||||
|
||||
module "amp" {
|
||||
|
||||
}
|
||||
|
||||
module "grafana" {
|
||||
|
||||
}
|
||||
@@ -0,0 +1,17 @@
|
||||
|
||||
# DONT create the resources
|
||||
# VPC and supporting resources
|
||||
# EKS and Managed node groups
|
||||
|
||||
#
|
||||
|
||||
module "java" {
|
||||
source = ""
|
||||
|
||||
}
|
||||
|
||||
|
||||
module "haproxy" {
|
||||
source = ""
|
||||
|
||||
}
|
||||
@@ -0,0 +1,172 @@
|
||||
|
||||
#---------------------------------------------------------------
|
||||
# Observability Resources
|
||||
#---------------------------------------------------------------
|
||||
|
||||
module "managed_grafana" {
|
||||
source = "terraform-aws-modules/managed-service-grafana/aws"
|
||||
version = "~> 1.3"
|
||||
|
||||
# Workspace
|
||||
name = local.name
|
||||
stack_set_name = local.name
|
||||
data_sources = ["PROMETHEUS"]
|
||||
associate_license = false
|
||||
|
||||
# # Role associations
|
||||
# Pending https://github.com/hashicorp/terraform-provider-aws/issues/24166
|
||||
# role_associations = {
|
||||
# "ADMIN" = {
|
||||
# "group_ids" = []
|
||||
# "user_ids" = []
|
||||
# }
|
||||
# "EDITOR" = {
|
||||
# "group_ids" = []
|
||||
# "user_ids" = []
|
||||
# }
|
||||
# }
|
||||
|
||||
tags = local.tags
|
||||
}
|
||||
|
||||
resource "grafana_data_source" "prometheus" {
|
||||
type = "prometheus"
|
||||
name = "amp"
|
||||
is_default = true
|
||||
url = module.managed_prometheus.workspace_prometheus_endpoint
|
||||
|
||||
json_data {
|
||||
http_method = "GET"
|
||||
sigv4_auth = true
|
||||
sigv4_auth_type = "workspace-iam-role"
|
||||
sigv4_region = local.region
|
||||
}
|
||||
}
|
||||
|
||||
resource "grafana_folder" "this" {
|
||||
title = "Observability"
|
||||
}
|
||||
|
||||
resource "grafana_dashboard" "this" {
|
||||
folder = grafana_folder.this.id
|
||||
config_json = file("${path.module}/dashboards/default.json")
|
||||
}
|
||||
|
||||
module "managed_prometheus" {
|
||||
source = "terraform-aws-modules/managed-service-prometheus/aws"
|
||||
version = "~> 2.1"
|
||||
|
||||
workspace_alias = local.name
|
||||
|
||||
alert_manager_definition = <<-EOT
|
||||
alertmanager_config: |
|
||||
route:
|
||||
receiver: 'default'
|
||||
receivers:
|
||||
- name: 'default'
|
||||
EOT
|
||||
|
||||
rule_group_namespaces = {
|
||||
haproxy = {
|
||||
name = "haproxy_rules"
|
||||
data = <<-EOT
|
||||
groups:
|
||||
- name: obsa-haproxy-down-alert
|
||||
rules:
|
||||
- alert: HA_proxy_down
|
||||
expr: haproxy_up == 0
|
||||
for: 0m
|
||||
labels:
|
||||
severity: critical
|
||||
annotations:
|
||||
summary: HAProxy down (instance {{ $labels.instance }})
|
||||
description: "HAProxy down\n VALUE = {{ $value }}\n LABELS = {{ $labels }}"
|
||||
- name: obsa-haproxy-http4xx-error-alert
|
||||
rules:
|
||||
- alert: Ha_proxy_High_Http4xx_ErrorRate_Backend
|
||||
expr: sum by (backend) (rate(haproxy_server_http_responses_total{code="4xx"}[1m])) / sum by (backend) (rate(haproxy_server_http_responses_total[1m]) * 100) > 5
|
||||
for: 1m
|
||||
labels:
|
||||
severity: critical
|
||||
annotations:
|
||||
summary: HAProxy high HTTP 4xx error rate backend (instance {{ $labels.instance }})
|
||||
description: "Too many HTTP requests with status 4xx (> 5%) on backend {{ $labels.fqdn }}/{{ $labels.backend }}\n VALUE = {{ $value }}\n LABELS = {{ $labels }}"
|
||||
- name: obsa-haproxy-http5xx-error-alert
|
||||
rules:
|
||||
- alert: Ha_proxy_High_Http5xx_ErrorRate_Backend
|
||||
expr: sum by (backend) (rate(haproxy_server_http_responses_total{code="5xx"}[1m])) / sum by (backend) (rate(haproxy_server_http_responses_total[1m]) * 100) > 5
|
||||
for: 1m
|
||||
labels:
|
||||
severity: critical
|
||||
annotations:
|
||||
summary: HAProxy high HTTP 5xx error rate backend (instance {{ $labels.instance }})
|
||||
description: "Too many HTTP requests with status 5xx (> 5%) on backend {{ $labels.fqdn }}/{{ $labels.backend }}\n VALUE = {{ $value }}\n LABELS = {{ $labels }}"
|
||||
- name: obsa-haproxy-Http4xx-ErrorRate-Server-alert
|
||||
rules:
|
||||
- alert: Ha_proxy_High_Http4xx_ErrorRate_Server
|
||||
expr: sum by (server) (rate(haproxy_server_http_responses_total{code="4xx"}[1m])) / sum by (server) (rate(haproxy_server_http_responses_total[1m]) * 100) > 5
|
||||
for: 1m
|
||||
labels:
|
||||
severity: critical
|
||||
annotations:
|
||||
summary: HAProxy high HTTP 4xx error rate server (instance {{ $labels.instance }})
|
||||
description: "Too many HTTP requests with status 4xx (> 5%) on server {{ $labels.server }}\n VALUE = {{ $value }}\n LABELS = {{ $labels }}"
|
||||
- name: obsa-haproxy-Http5xx-ErrorRate-Server-alert
|
||||
rules:
|
||||
- alert: Ha_proxy_High_Http5xx_ErrorRate_Server
|
||||
expr: sum by (server) (rate(haproxy_server_http_responses_total{code="5xx"}[1m])) / sum by (server) (rate(haproxy_server_http_responses_total[1m]) * 100) > 5
|
||||
for: 1m
|
||||
labels:
|
||||
severity: critical
|
||||
annotations:
|
||||
summary: HAProxy high HTTP 5xx error rate server (instance {{ $labels.instance }})
|
||||
description: "Too many HTTP requests with status 5xx (> 5%) on server {{ $labels.server }}\n VALUE = {{ $value }}\n LABELS = {{ $labels }}"
|
||||
EOT
|
||||
}
|
||||
}
|
||||
|
||||
tags = local.tags
|
||||
}
|
||||
|
||||
#---------------------------------------------------------------
|
||||
# Sample Application
|
||||
#---------------------------------------------------------------
|
||||
|
||||
# https://github.com/haproxy-ingress/charts/tree/master/haproxy-ingress
|
||||
resource "helm_release" "haproxy_ingress" {
|
||||
namespace = "haproxy-ingress"
|
||||
create_namespace = true
|
||||
|
||||
name = "haproxy-ingress"
|
||||
repository = "https://haproxy-ingress.github.io/charts"
|
||||
chart = "haproxy-ingress"
|
||||
version = "0.13.7"
|
||||
|
||||
set {
|
||||
name = "defaultBackend.enabled"
|
||||
value = true
|
||||
}
|
||||
|
||||
set {
|
||||
name = "controller.stats.enabled"
|
||||
value = true
|
||||
}
|
||||
|
||||
set {
|
||||
name = "controller.metrics.enabled"
|
||||
value = true
|
||||
}
|
||||
|
||||
set {
|
||||
name = "controller.metrics.service.annotations.prometheus\\.io/port"
|
||||
value = 9101
|
||||
type = "string"
|
||||
}
|
||||
|
||||
set {
|
||||
name = "controller.metrics.service.annotations.prometheus\\.io/scrape"
|
||||
value = true
|
||||
type = "string"
|
||||
}
|
||||
}
|
||||
|
||||
@@ -0,0 +1,6 @@
|
||||
variable "java" {
|
||||
default = {
|
||||
a = ""
|
||||
b = ""
|
||||
}
|
||||
}
|
||||
Reference in New Issue
Block a user