initial commit

This commit is contained in:
Vara Bonthu
2022-07-22 08:17:01 +01:00
committed by Rodrigue Koffi
parent 0262a8a45d
commit d654fd36ae
12 changed files with 321 additions and 55 deletions
View File
-29
View File
@@ -1,29 +0,0 @@
<!-- BEGIN_TF_DOCS -->
## Requirements
| Name | Version |
|------|---------|
| <a name="requirement_terraform"></a> [terraform](#requirement\_terraform) | >= 0.14.0 |
| <a name="requirement_aws"></a> [aws](#requirement\_aws) | >= 3.72.0 |
| <a name="requirement_awscc"></a> [awscc](#requirement\_awscc) | >= 0.11.0 |
## Providers
No providers.
## Modules
No modules.
## Resources
No resources.
## Inputs
No inputs.
## Outputs
No outputs.
<!-- END_TF_DOCS -->
-5
View File
@@ -1,5 +0,0 @@
#####################################################################################
# Terraform module examples are meant to show an _example_ on how to use a module
# per use-case. The code below should not be copied directly but referenced in order
# to build your own root module that invokes this module
#####################################################################################
View File
-21
View File
@@ -1,21 +0,0 @@
terraform {
required_version = ">= 0.14.0"
required_providers {
aws = {
source = "hashicorp/aws"
version = ">= 3.72.0"
}
awscc = {
source = "hashicorp/awscc"
version = ">= 0.11.0"
}
}
}
provider "awscc" {
user_agent = [{
product_name = "terraform-awscc-"
product_version = "0.0.1"
comment = "V1/AWS-D69B4015/<github repo id>"
}]
}
View File
+24
View File
@@ -0,0 +1,24 @@
#---------------------------------------------------------------
# EKS Blueprints
#---------------------------------------------------------------
module "eks_blueprints" {
source = "../../.."
cluster_name = local.name
cluster_version = "1.22"
vpc_id = module.vpc.vpc_id
private_subnet_ids = module.vpc.private_subnets
managed_node_groups = {
t3_l = {
node_group_name = "managed-ondemand"
instance_types = ["t3.large"]
min_size = 2
subnet_ids = module.vpc.private_subnets
}
}
tags = local.tags
}
+35
View File
@@ -0,0 +1,35 @@
module "vpc" {
source = "terraform-aws-modules/vpc/aws"
version = "~> 3.0"
name = local.name
cidr = local.vpc_cidr
azs = local.azs
public_subnets = [for k, v in local.azs : cidrsubnet(local.vpc_cidr, 8, k)]
private_subnets = [for k, v in local.azs : cidrsubnet(local.vpc_cidr, 8, k + 10)]
enable_nat_gateway = true
single_nat_gateway = true
enable_dns_hostnames = true
# Manage so we can name
manage_default_network_acl = true
default_network_acl_tags = { Name = "${local.name}-default" }
manage_default_route_table = true
default_route_table_tags = { Name = "${local.name}-default" }
manage_default_security_group = true
default_security_group_tags = { Name = "${local.name}-default" }
public_subnet_tags = {
"kubernetes.io/cluster/${local.name}" = "shared"
"kubernetes.io/role/elb" = 1
}
private_subnet_tags = {
"kubernetes.io/cluster/${local.name}" = "shared"
"kubernetes.io/role/internal-elb" = 1
}
tags = local.tags
}
+67
View File
@@ -0,0 +1,67 @@
module "eks_observability_accelerator" {
source = "aws-ia/aws-observability-accelerator/terraform/eks"
# -- or use an existing cluster
eks_cluster_id = var.eks_cluster_id
# enable managed add-on for ADOT. Do we enforce this or let users
# have their own configs for OTEL operator
enable_amazon_eks_adot = true
# -- or enable opentelemetry operator
enable_open_telemetry_operator = true
open_telemetry_operator_config = map() // custom config
# deploy selected workloads by count indexing
# this creates a new AMP workspace
create_managed_prometheus_workspace = true
enable_haproxy = true
haproxy_config = {
amp_endpoint = module/amp.endpoint
grafana_endpoint = module.grafana.endpoint
}
enable_java = true
java_config = {
amp_endpoint = ""
grafana_endpoint = ""
}
# -- or use an existing one
# seems like https://github.com/terraform-aws-modules/terraform-aws-managed-service-prometheus
# supports importing
amp_workspace_alias = var.amp_alias
# enable rules and alerts
enable_alert_manager = true
# -- or provide custom alerts definition
prometheus_custom_alert_rule = var.prometheus_custom_alert_rule
# create grafana workspace, and customer to deal with authentication later
create_managed_grafana_workspace = true
grafana_auth_provider = var.grafana_auth_provider //SAML or AWS_SSO
grafana_account_access_type = var.grafana_account_access_type // CURRENT_ACCOUNT or ORGANIZATION
grafana_permission_type = var.grafana_permission_type // SERVICE_MANAGED or CUSTOMER_MANAGED
grafana_permission_role_arn = var.grafana_permission_role_arn // if CUSTOMER_MANAGED
# -- or using existing amg workspace. so we can use API for keys
managed_grafana_workspace_id = var.managed_grafana_workspace_id
}
module "amp" {
}
module "grafana" {
}
+17
View File
@@ -0,0 +1,17 @@
# DONT create the resources
# VPC and supporting resources
# EKS and Managed node groups
#
module "java" {
source = ""
}
module "haproxy" {
source = ""
}
@@ -0,0 +1,172 @@
#---------------------------------------------------------------
# Observability Resources
#---------------------------------------------------------------
module "managed_grafana" {
source = "terraform-aws-modules/managed-service-grafana/aws"
version = "~> 1.3"
# Workspace
name = local.name
stack_set_name = local.name
data_sources = ["PROMETHEUS"]
associate_license = false
# # Role associations
# Pending https://github.com/hashicorp/terraform-provider-aws/issues/24166
# role_associations = {
# "ADMIN" = {
# "group_ids" = []
# "user_ids" = []
# }
# "EDITOR" = {
# "group_ids" = []
# "user_ids" = []
# }
# }
tags = local.tags
}
resource "grafana_data_source" "prometheus" {
type = "prometheus"
name = "amp"
is_default = true
url = module.managed_prometheus.workspace_prometheus_endpoint
json_data {
http_method = "GET"
sigv4_auth = true
sigv4_auth_type = "workspace-iam-role"
sigv4_region = local.region
}
}
resource "grafana_folder" "this" {
title = "Observability"
}
resource "grafana_dashboard" "this" {
folder = grafana_folder.this.id
config_json = file("${path.module}/dashboards/default.json")
}
module "managed_prometheus" {
source = "terraform-aws-modules/managed-service-prometheus/aws"
version = "~> 2.1"
workspace_alias = local.name
alert_manager_definition = <<-EOT
alertmanager_config: |
route:
receiver: 'default'
receivers:
- name: 'default'
EOT
rule_group_namespaces = {
haproxy = {
name = "haproxy_rules"
data = <<-EOT
groups:
- name: obsa-haproxy-down-alert
rules:
- alert: HA_proxy_down
expr: haproxy_up == 0
for: 0m
labels:
severity: critical
annotations:
summary: HAProxy down (instance {{ $labels.instance }})
description: "HAProxy down\n VALUE = {{ $value }}\n LABELS = {{ $labels }}"
- name: obsa-haproxy-http4xx-error-alert
rules:
- alert: Ha_proxy_High_Http4xx_ErrorRate_Backend
expr: sum by (backend) (rate(haproxy_server_http_responses_total{code="4xx"}[1m])) / sum by (backend) (rate(haproxy_server_http_responses_total[1m]) * 100) > 5
for: 1m
labels:
severity: critical
annotations:
summary: HAProxy high HTTP 4xx error rate backend (instance {{ $labels.instance }})
description: "Too many HTTP requests with status 4xx (> 5%) on backend {{ $labels.fqdn }}/{{ $labels.backend }}\n VALUE = {{ $value }}\n LABELS = {{ $labels }}"
- name: obsa-haproxy-http5xx-error-alert
rules:
- alert: Ha_proxy_High_Http5xx_ErrorRate_Backend
expr: sum by (backend) (rate(haproxy_server_http_responses_total{code="5xx"}[1m])) / sum by (backend) (rate(haproxy_server_http_responses_total[1m]) * 100) > 5
for: 1m
labels:
severity: critical
annotations:
summary: HAProxy high HTTP 5xx error rate backend (instance {{ $labels.instance }})
description: "Too many HTTP requests with status 5xx (> 5%) on backend {{ $labels.fqdn }}/{{ $labels.backend }}\n VALUE = {{ $value }}\n LABELS = {{ $labels }}"
- name: obsa-haproxy-Http4xx-ErrorRate-Server-alert
rules:
- alert: Ha_proxy_High_Http4xx_ErrorRate_Server
expr: sum by (server) (rate(haproxy_server_http_responses_total{code="4xx"}[1m])) / sum by (server) (rate(haproxy_server_http_responses_total[1m]) * 100) > 5
for: 1m
labels:
severity: critical
annotations:
summary: HAProxy high HTTP 4xx error rate server (instance {{ $labels.instance }})
description: "Too many HTTP requests with status 4xx (> 5%) on server {{ $labels.server }}\n VALUE = {{ $value }}\n LABELS = {{ $labels }}"
- name: obsa-haproxy-Http5xx-ErrorRate-Server-alert
rules:
- alert: Ha_proxy_High_Http5xx_ErrorRate_Server
expr: sum by (server) (rate(haproxy_server_http_responses_total{code="5xx"}[1m])) / sum by (server) (rate(haproxy_server_http_responses_total[1m]) * 100) > 5
for: 1m
labels:
severity: critical
annotations:
summary: HAProxy high HTTP 5xx error rate server (instance {{ $labels.instance }})
description: "Too many HTTP requests with status 5xx (> 5%) on server {{ $labels.server }}\n VALUE = {{ $value }}\n LABELS = {{ $labels }}"
EOT
}
}
tags = local.tags
}
#---------------------------------------------------------------
# Sample Application
#---------------------------------------------------------------
# https://github.com/haproxy-ingress/charts/tree/master/haproxy-ingress
resource "helm_release" "haproxy_ingress" {
namespace = "haproxy-ingress"
create_namespace = true
name = "haproxy-ingress"
repository = "https://haproxy-ingress.github.io/charts"
chart = "haproxy-ingress"
version = "0.13.7"
set {
name = "defaultBackend.enabled"
value = true
}
set {
name = "controller.stats.enabled"
value = true
}
set {
name = "controller.metrics.enabled"
value = true
}
set {
name = "controller.metrics.service.annotations.prometheus\\.io/port"
value = 9101
type = "string"
}
set {
name = "controller.metrics.service.annotations.prometheus\\.io/scrape"
value = true
type = "string"
}
}
+6
View File
@@ -0,0 +1,6 @@
variable "java" {
default = {
a = ""
b = ""
}
}