mirror of
https://github.com/storytold/terraform-aws-observability-accelerator.git
synced 2026-10-09 00:09:43 +00:00
@@ -0,0 +1,65 @@
|
||||
name: 🐞 Bug Report
|
||||
title: "[Bug]: <title>"
|
||||
description: Create a report to help us improve
|
||||
labels: ["bug", "triage"]
|
||||
body:
|
||||
- type: markdown
|
||||
attributes:
|
||||
value: |
|
||||
### How to write a good bug report?
|
||||
|
||||
- Respect the issue template as much as possible.
|
||||
- The title should be short and descriptive.
|
||||
- Explain the conditions which led you to report this issue and the context.
|
||||
- The context should lead to something, an idea or a problem that you’re facing.
|
||||
- Remain clear and concise.
|
||||
- Format your messages to help the reader focus on what matters and understand the structure of your message, use [Markdown syntax](https://help.github.com/articles/github-flavored-markdown)
|
||||
|
||||
- type: checkboxes
|
||||
id: terms
|
||||
attributes:
|
||||
label: Welcome to Amazon EKS Blueprints!
|
||||
options:
|
||||
- label: Yes, I've searched similar issues on [GitHub](https://github.com/aws-observability/terraform-aws-observability-accelerator/issues) and didn't find any.
|
||||
required: true
|
||||
|
||||
- type: input
|
||||
attributes:
|
||||
label: Amazon EKS Blueprints Release version
|
||||
description: |
|
||||
`latest` is not considered as a valid version.
|
||||
Enter release number!
|
||||
placeholder: Your version here.
|
||||
validations:
|
||||
required: true
|
||||
|
||||
- type: textarea
|
||||
attributes:
|
||||
label: What is your environment, configuration and the example used?
|
||||
description: |
|
||||
Terraform version, link to example used or your main.tf content etc.
|
||||
|
||||
Use [Markdown syntax](https://help.github.com/articles/github-flavored-markdown) if needed.
|
||||
placeholder: Add information here.
|
||||
validations:
|
||||
required: true
|
||||
|
||||
- type: textarea
|
||||
attributes:
|
||||
label: What did you do and What did you see instead?
|
||||
description: |
|
||||
Provide error details and the expected details.
|
||||
|
||||
Use [Markdown syntax](https://help.github.com/articles/github-flavored-markdown) if needed.
|
||||
placeholder: Add information here.
|
||||
validations:
|
||||
required: true
|
||||
|
||||
- type: textarea
|
||||
attributes:
|
||||
label: Additional Information
|
||||
description: Use [Markdown syntax](https://help.github.com/articles/github-flavored-markdown) if needed.
|
||||
placeholder: Add information here.
|
||||
render: shell
|
||||
validations:
|
||||
required: false
|
||||
@@ -0,0 +1 @@
|
||||
blank_issues_enabled: false
|
||||
@@ -0,0 +1,23 @@
|
||||
---
|
||||
name: Feature request
|
||||
about: Suggest an idea for this project
|
||||
title: '[FEATURE] <title>'
|
||||
labels: 'feature-request'
|
||||
assignees: ''
|
||||
|
||||
---
|
||||
|
||||
#### Is your feature request related to a problem? Please describe
|
||||
A clear and concise description of what the problem is. Ex. I'm always frustrated when [...]
|
||||
|
||||
|
||||
#### Describe the solution you'd like
|
||||
A clear and concise description of what you want to happen.
|
||||
|
||||
|
||||
#### Describe alternatives you've considered
|
||||
A clear and concise description of any alternative solutions or features you've considered.
|
||||
|
||||
|
||||
#### Additional context
|
||||
Add any other context or screenshots about the feature request here.
|
||||
@@ -0,0 +1,24 @@
|
||||
---
|
||||
name: Question
|
||||
about: I have a Question
|
||||
title: '[QUESTION] <title>'
|
||||
labels: 'question'
|
||||
assignees: ''
|
||||
|
||||
---
|
||||
|
||||
#### Please describe your question here
|
||||
<!-- Provide as much information as possible to explain your question -->
|
||||
|
||||
|
||||
#### Provide link to the example related to the question
|
||||
<!-- Please provide the link to the example related to this question from this repo -->
|
||||
|
||||
|
||||
#### Additional context
|
||||
<!-- Add any other context or screenshots about the question here -->
|
||||
|
||||
|
||||
#### More
|
||||
|
||||
- [ ] Yes, I have checked the repo for existing issues before raising this question
|
||||
@@ -0,0 +1,30 @@
|
||||
|
||||
### What does this PR do?
|
||||
|
||||
<!-- A brief description of the change being made with this pull request. -->
|
||||
|
||||
🛑 Please open an issue first to discuss any significant work and flesh out details/direction - we would hate for your time to be wasted. Consult the CONTRIBUTING guide for submitting pull-requests.
|
||||
|
||||
|
||||
### Motivation
|
||||
|
||||
<!-- What inspired you to submit this pull request? -->
|
||||
|
||||
|
||||
### More
|
||||
|
||||
- [ ] Yes, I have tested the PR using my local account setup (Provide any test evidence report under Additional Notes)
|
||||
- [ ] Yes, I have added a new example under [examples](https://github.com/aws-observability/terraform-aws-eks-blueprints/tree/main/examples) to support my PR
|
||||
- [ ] Yes, I have created another PR for add-ons under [add-ons](https://github.com/aws-samples/eks-blueprints-add-ons) repo (if applicable)
|
||||
- [ ] Yes, I have updated the [docs](https://github.com/aws-observability/terraform-aws-eks-blueprints/tree/main/docs) for this feature
|
||||
- [ ] Yes, I ran `pre-commit run -a` with this PR
|
||||
|
||||
|
||||
**Note**: Not all the PRs required examples and docs except a new pattern or add-on added.
|
||||
|
||||
### For Moderators
|
||||
- [ ] E2E Test successfully complete before merge?
|
||||
|
||||
### Additional Notes
|
||||
|
||||
<!-- Anything else we should know when reviewing? -->
|
||||
+46
@@ -0,0 +1,46 @@
|
||||
.DS_Store
|
||||
.idea
|
||||
.build
|
||||
|
||||
# Local .terraform directories
|
||||
**/.terraform/*
|
||||
|
||||
# Terraform lockfile
|
||||
.terraform.lock.hcl
|
||||
|
||||
# .tfstate files
|
||||
*.tfstate
|
||||
*.tfstate.*
|
||||
*.tfplan
|
||||
|
||||
# Crash log files
|
||||
crash.log
|
||||
|
||||
# Exclude all .tfvars files, which are likely to contain sentitive data, such as
|
||||
# password, private keys, and other secrets. These should not be part of version
|
||||
# control as they are data points which are potentially sensitive and subject
|
||||
# to change depending on the environment.
|
||||
*.tfvars
|
||||
|
||||
# Ignore override files as they are usually used to override resources locally and so
|
||||
# are not checked in
|
||||
override.tf
|
||||
override.tf.json
|
||||
*_override.tf
|
||||
*_override.tf.json
|
||||
|
||||
# Ignore CLI configuration files
|
||||
.terraformrc
|
||||
terraform.rc
|
||||
|
||||
# Locals
|
||||
kubeconfig*
|
||||
kube-config*
|
||||
local_tf_state/
|
||||
.vscode
|
||||
.gitallowed
|
||||
site
|
||||
.env*
|
||||
|
||||
# Checks
|
||||
.tfsec
|
||||
@@ -0,0 +1,40 @@
|
||||
repos:
|
||||
- repo: https://github.com/pre-commit/pre-commit-hooks
|
||||
rev: v4.3.0
|
||||
hooks:
|
||||
- id: trailing-whitespace
|
||||
args: ['--markdown-linebreak-ext=md']
|
||||
- id: end-of-file-fixer
|
||||
- id: check-merge-conflict
|
||||
- id: detect-private-key
|
||||
- id: detect-aws-credentials
|
||||
args: ['--allow-missing-credentials']
|
||||
- repo: https://github.com/antonbabenko/pre-commit-terraform
|
||||
rev: v1.74.1
|
||||
hooks:
|
||||
- id: terraform_fmt
|
||||
- id: terraform_docs
|
||||
args:
|
||||
- '--args=--lockfile=false'
|
||||
- id: terraform_validate
|
||||
exclude: deploy
|
||||
- id: terraform_tflint
|
||||
args:
|
||||
- '--args=--only=terraform_deprecated_interpolation'
|
||||
- '--args=--only=terraform_deprecated_index'
|
||||
- '--args=--only=terraform_unused_declarations'
|
||||
- '--args=--only=terraform_comment_syntax'
|
||||
- '--args=--only=terraform_documented_outputs'
|
||||
- '--args=--only=terraform_documented_variables'
|
||||
- '--args=--only=terraform_typed_variables'
|
||||
- '--args=--only=terraform_module_pinned_source'
|
||||
- '--args=--only=terraform_naming_convention'
|
||||
- '--args=--only=terraform_required_version'
|
||||
- '--args=--only=terraform_required_providers'
|
||||
- '--args=--only=terraform_standard_module_structure'
|
||||
- '--args=--only=terraform_workspace_remote'
|
||||
- id: terraform_tfsec
|
||||
files: ^examples/ # only scan `examples/*` which are the implementation
|
||||
args:
|
||||
- --args=--config-file=__GIT_WORKING_DIR__/tfsec.yaml
|
||||
- --args=--concise-output
|
||||
+66
@@ -0,0 +1,66 @@
|
||||
# https://github.com/terraform-linters/tflint/blob/master/docs/user-guide/module-inspection.md
|
||||
# borrowed & modified indefinitely from https://github.com/ksatirli/building-infrastructure-you-can-mostly-trust/blob/main/.tflint.hcl
|
||||
|
||||
plugin "aws" {
|
||||
enabled = true
|
||||
version = "0.14.0"
|
||||
source = "github.com/terraform-linters/tflint-ruleset-aws"
|
||||
}
|
||||
|
||||
config {
|
||||
module = true
|
||||
force = false
|
||||
}
|
||||
|
||||
rule "terraform_required_providers" {
|
||||
enabled = true
|
||||
}
|
||||
|
||||
rule "terraform_required_version" {
|
||||
enabled = true
|
||||
}
|
||||
|
||||
rule "terraform_naming_convention" {
|
||||
enabled = true
|
||||
format = "snake_case"
|
||||
}
|
||||
|
||||
rule "terraform_typed_variables" {
|
||||
enabled = true
|
||||
}
|
||||
|
||||
rule "terraform_unused_declarations" {
|
||||
enabled = true
|
||||
}
|
||||
|
||||
rule "terraform_comment_syntax" {
|
||||
enabled = true
|
||||
}
|
||||
|
||||
rule "terraform_deprecated_index" {
|
||||
enabled = true
|
||||
}
|
||||
|
||||
rule "terraform_deprecated_interpolation" {
|
||||
enabled = true
|
||||
}
|
||||
|
||||
rule "terraform_documented_outputs" {
|
||||
enabled = true
|
||||
}
|
||||
|
||||
rule "terraform_documented_variables" {
|
||||
enabled = true
|
||||
}
|
||||
|
||||
rule "terraform_module_pinned_source" {
|
||||
enabled = true
|
||||
}
|
||||
|
||||
rule "terraform_standard_module_structure" {
|
||||
enabled = true
|
||||
}
|
||||
|
||||
rule "terraform_workspace_remote" {
|
||||
enabled = true
|
||||
}
|
||||
+15
@@ -0,0 +1,15 @@
|
||||
# Who is using AWS Observability Accelerator for Terraform?
|
||||
|
||||
AWS Observability Accelerator for Terraform has a variety of users and use cases to configure and manage Observability on EKS/ECS clusters.
|
||||
Many customers want to learn from others who have already implemented AWS Observability Accelerator in their environments.
|
||||
|
||||
The following is a self-reported list of users to help identify adoption and points of contact.
|
||||
|
||||
## Add yourself
|
||||
|
||||
If you are using AWS Observability Accelerator please consider adding yourself as a user by opening a pull request to this file.
|
||||
|
||||
## Adopters (Alphabetical)
|
||||
|
||||
| Organization | Description | Contacts | Link |
|
||||
| --- | --- | --- | --- |
|
||||
@@ -0,0 +1,4 @@
|
||||
# Require approvals from someone in the owner team before merging
|
||||
# More information here: https://docs.github.com/en/repositories/managing-your-repositorys-settings-and-features/customizing-your-repository/about-code-owners
|
||||
|
||||
* @aws-observability/aws-observability-accelerator
|
||||
@@ -1,4 +1,3 @@
|
||||
|
||||
Apache License
|
||||
Version 2.0, January 2004
|
||||
http://www.apache.org/licenses/
|
||||
@@ -173,3 +172,30 @@
|
||||
defend, and hold each Contributor harmless for any liability
|
||||
incurred by, or claims asserted against, such Contributor by reason
|
||||
of your accepting any such warranty or additional liability.
|
||||
|
||||
END OF TERMS AND CONDITIONS
|
||||
|
||||
APPENDIX: How to apply the Apache License to your work.
|
||||
|
||||
To apply the Apache License to your work, attach the following
|
||||
boilerplate notice, with the fields enclosed by brackets "[]"
|
||||
replaced with your own identifying information. (Don't include
|
||||
the brackets!) The text should be enclosed in the appropriate
|
||||
comment syntax for the file format. We also recommend that a
|
||||
file or class name and description of purpose be included on the
|
||||
same "printed page" as the copyright notice for easier
|
||||
identification within third-party archives.
|
||||
|
||||
Copyright [yyyy] [name of copyright owner]
|
||||
|
||||
Licensed under the Apache License, Version 2.0 (the "License");
|
||||
you may not use this file except in compliance with the License.
|
||||
You may obtain a copy of the License at
|
||||
|
||||
http://www.apache.org/licenses/LICENSE-2.0
|
||||
|
||||
Unless required by applicable law or agreed to in writing, software
|
||||
distributed under the License is distributed on an "AS IS" BASIS,
|
||||
WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
See the License for the specific language governing permissions and
|
||||
limitations under the License.
|
||||
|
||||
@@ -0,0 +1,7 @@
|
||||
Copyright 2016-2022 Amazon.com, Inc. or its affiliates. All Rights Reserved.
|
||||
|
||||
Licensed under the Apache License, Version 2.0 (the "License"). You may not use this file except in compliance with the License. A copy of the License is located at
|
||||
|
||||
http://aws.amazon.com/apache2.0/
|
||||
|
||||
or in the "license" file accompanying this file. This file is distributed on an "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. See the License for the specific language governing permissions and limitations under the License.
|
||||
@@ -1,17 +1,187 @@
|
||||
## My Project
|
||||
# AWS Observability Accelerator for Terraform
|
||||
|
||||
TODO: Fill this README out!
|
||||
Welcome to the AWS Observability Accelerator for Terraform!
|
||||
|
||||
Be sure to:
|
||||
The AWS Observability accelerator for Terraform is a set of modules to help you
|
||||
configure Observability for your Amazon EKS clusters with AWS Observability services.
|
||||
This project proposes a core module to bootstrap your cluster with the AWS Distro for
|
||||
OpenTelemetry (ADOT) Operator for EKS, Amazon Managed Service for Prometheus,
|
||||
Amazon Managed Grafana. Additionally we have a set of workloads modules to
|
||||
leverage curated ADOT collector configurations, Grafana dashboards,
|
||||
Prometheus recording rules and alerts.
|
||||
|
||||
* Change the title in this README
|
||||
* Edit your repository description on GitHub
|
||||
We will be leveraging [EKS Blueprints](https://github.com/aws-ia/terraform-aws-eks-blueprints)
|
||||
repository to deploy the solution.
|
||||
|
||||
## Security
|
||||
## Getting started
|
||||
|
||||
To quickstart with a complete workflow and view Aamzon EKS infrastructure dashboards, visit the [existing cluster with base and module example](./examples/existing-cluster-with-base-and-infra/)
|
||||
|
||||
## How it works
|
||||
|
||||
The sections below demonstrate how you can leverage AWS Observability Accelerator
|
||||
to enable monitoring to an existing EKS cluster.
|
||||
|
||||
### Base Module
|
||||
|
||||
The base module allows you to configure the AWS Observability services for your cluster and
|
||||
the AWS Distro for OpenTelemetry (ADOT) Operator as the signals collection mechanism.
|
||||
|
||||
This is the minimum configuration to have a new Managed Grafana Workspace, Amazon Managed
|
||||
Service for Prometheus Workspace, ADOT Operator deployed for you and ready to receive your
|
||||
data.
|
||||
|
||||
```hcl
|
||||
module "eks_observability_accelerator" {
|
||||
source = "aws-observability/terrarom-aws-observability-accelerator"
|
||||
aws_region = "eu-west-1"
|
||||
eks_cluster_id = "my-eks-cluster"
|
||||
}
|
||||
```
|
||||
|
||||
You can optionally reuse existing Workspaces:
|
||||
|
||||
```hcl
|
||||
module "eks_observability_accelerator" {
|
||||
source = "aws-observability/terrarom-aws-observability-accelerator"
|
||||
aws_region = "eu-west-1"
|
||||
eks_cluster_id = "my-eks-cluster"
|
||||
|
||||
# prevents creation of a new Amazon Managed Prometheus workspace
|
||||
enable_managed_prometheus = false
|
||||
|
||||
# reusing existing Amazon Managed Prometheus Workspace
|
||||
managed_prometheus_workspace_id = "ws-abcd123..."
|
||||
|
||||
# prevents creation of a new Amazon Managed Grafana workspace
|
||||
enable_managed_grafana = false
|
||||
|
||||
managed_grafana_workspace_id = "g-abcdef123"
|
||||
grafana_api_key = var.grafana_api_key
|
||||
}
|
||||
```
|
||||
|
||||
View all the configuration options in the module documentation below.
|
||||
|
||||
### Workload modules
|
||||
|
||||
[Workloads modules](./modules/workloads) are provided, which essentially provide curated
|
||||
metrics collection, alerting rule and Grafana dashboards.
|
||||
|
||||
|
||||
#### Infrastructure monitoring
|
||||
|
||||
```hcl
|
||||
module "workloads_infra" {
|
||||
source = "aws-observability/terrarom-aws-observability-accelerator/workloads/infra"
|
||||
|
||||
eks_cluster_id = module.eks_observability_accelerator.eks_cluster_id
|
||||
|
||||
dashboards_folder_id = module.eks_observability_accelerator.grafana_dashboards_folder_id
|
||||
managed_prometheus_workspace_id = module.eks_observability_accelerator.managed_prometheus_workspace_id
|
||||
|
||||
managed_prometheus_workspace_endpoint = module.eks_observability_accelerator.managed_prometheus_workspace_endpoint
|
||||
managed_prometheus_workspace_region = module.eks_observability_accelerator.managed_prometheus_workspace_region
|
||||
}
|
||||
```
|
||||
|
||||
Grafana Dashboards
|
||||
|
||||
<img width="1719" alt="image" src="https://user-images.githubusercontent.com/10175027/187661363-608cdfcf-ed13-4ddd-a198-e761b78d2291.png">
|
||||
|
||||
Check the the [complete example](./examples/existing-cluster-with-base-and-infra/)
|
||||
|
||||
## Motivation
|
||||
|
||||
Kubernetes is a powerful and extensible container orchestration technology that allows you to deploy and manage containerized applications at scale. The extensible nature of Kubernetes also allows you to use a wide range of popular open-source tools, commonly referred to as add-ons, in Kubernetes clusters. With such a large number of tools and design choices available, building a tailored EKS cluster that meets your application’s specific needs can take a significant amount of time. It involves integrating a wide range of open-source tools and AWS services and requires deep expertise in AWS and Kubernetes.
|
||||
|
||||
AWS customers have asked for examples that demonstrate how to integrate the landscape of Kubernetes tools and make it easy for them to provision complete, opinionated EKS clusters that meet specific application requirements. Customers can use AWS Observability Accelerator to configure and deploy purpose built EKS clusters, and start onboarding workloads in days, rather than months.
|
||||
|
||||
## Support & Feedback
|
||||
|
||||
AWS Observability Accelerator for Terraform is maintained by AWS Solution Architects. It is not part of an AWS service and support is provided best-effort by the AWS Observability Accelerator community.
|
||||
|
||||
To post feedback, submit feature ideas, or report bugs, please use the [Issues](https://github.com/aws-observability/terraform-aws-observability-accelerator/issues) section of this GitHub repo.
|
||||
|
||||
If you are interested in contributing to EKS Blueprints, see the [Contribution guide](https://github.com/aws-observability/terraform-aws-observability-accelerator/blob/main/CONTRIBUTING.md).
|
||||
|
||||
---
|
||||
|
||||
<!-- BEGINNING OF PRE-COMMIT-TERRAFORM DOCS HOOK -->
|
||||
## Requirements
|
||||
|
||||
| Name | Version |
|
||||
|------|---------|
|
||||
| <a name="requirement_terraform"></a> [terraform](#requirement\_terraform) | >= 0.14.0 |
|
||||
| <a name="requirement_aws"></a> [aws](#requirement\_aws) | >= 4.0.0 |
|
||||
| <a name="requirement_awscc"></a> [awscc](#requirement\_awscc) | >= 0.24.0 |
|
||||
| <a name="requirement_grafana"></a> [grafana](#requirement\_grafana) | 1.25.0 |
|
||||
|
||||
## Providers
|
||||
|
||||
| Name | Version |
|
||||
|------|---------|
|
||||
| <a name="provider_aws"></a> [aws](#provider\_aws) | >= 4.0.0 |
|
||||
| <a name="provider_grafana"></a> [grafana](#provider\_grafana) | 1.25.0 |
|
||||
|
||||
## Modules
|
||||
|
||||
| Name | Source | Version |
|
||||
|------|--------|---------|
|
||||
| <a name="module_managed_grafana"></a> [managed\_grafana](#module\_managed\_grafana) | terraform-aws-modules/managed-service-grafana/aws | ~> 1.3 |
|
||||
| <a name="module_operator"></a> [operator](#module\_operator) | ./modules/add-ons/adot-operator | n/a |
|
||||
|
||||
## Resources
|
||||
|
||||
| Name | Type |
|
||||
|------|------|
|
||||
| [aws_prometheus_alert_manager_definition.this](https://registry.terraform.io/providers/hashicorp/aws/latest/docs/resources/prometheus_alert_manager_definition) | resource |
|
||||
| [aws_prometheus_workspace.this](https://registry.terraform.io/providers/hashicorp/aws/latest/docs/resources/prometheus_workspace) | resource |
|
||||
| [grafana_data_source.amp](https://registry.terraform.io/providers/grafana/grafana/1.25.0/docs/resources/data_source) | resource |
|
||||
| [grafana_folder.this](https://registry.terraform.io/providers/grafana/grafana/1.25.0/docs/resources/folder) | resource |
|
||||
| [aws_caller_identity.current](https://registry.terraform.io/providers/hashicorp/aws/latest/docs/data-sources/caller_identity) | data source |
|
||||
| [aws_eks_cluster.eks_cluster](https://registry.terraform.io/providers/hashicorp/aws/latest/docs/data-sources/eks_cluster) | data source |
|
||||
| [aws_grafana_workspace.this](https://registry.terraform.io/providers/hashicorp/aws/latest/docs/data-sources/grafana_workspace) | data source |
|
||||
| [aws_partition.current](https://registry.terraform.io/providers/hashicorp/aws/latest/docs/data-sources/partition) | data source |
|
||||
| [aws_region.current](https://registry.terraform.io/providers/hashicorp/aws/latest/docs/data-sources/region) | data source |
|
||||
|
||||
## Inputs
|
||||
|
||||
| Name | Description | Type | Default | Required |
|
||||
|------|-------------|------|---------|:--------:|
|
||||
| <a name="input_aws_region"></a> [aws\_region](#input\_aws\_region) | AWS Region | `string` | n/a | yes |
|
||||
| <a name="input_eks_cluster_id"></a> [eks\_cluster\_id](#input\_eks\_cluster\_id) | Name of the EKS cluster | `string` | n/a | yes |
|
||||
| <a name="input_enable_alertmanager"></a> [enable\_alertmanager](#input\_enable\_alertmanager) | Creates Amazon Managed Service for Prometheus AlertManager for all workloads | `bool` | `false` | no |
|
||||
| <a name="input_enable_amazon_eks_adot"></a> [enable\_amazon\_eks\_adot](#input\_enable\_amazon\_eks\_adot) | Enables the ADOT Operator on the EKS Cluster | `bool` | `true` | no |
|
||||
| <a name="input_enable_cert_manager"></a> [enable\_cert\_manager](#input\_enable\_cert\_manager) | Allow reusing an existing installation of cert-manager | `bool` | `true` | no |
|
||||
| <a name="input_enable_managed_grafana"></a> [enable\_managed\_grafana](#input\_enable\_managed\_grafana) | Creates a new Amazon Managed Grafana Workspace | `bool` | `true` | no |
|
||||
| <a name="input_enable_managed_prometheus"></a> [enable\_managed\_prometheus](#input\_enable\_managed\_prometheus) | Creates a new Amazon Managed Service for Prometheus Workspace | `bool` | `true` | no |
|
||||
| <a name="input_grafana_api_key"></a> [grafana\_api\_key](#input\_grafana\_api\_key) | Grafana API key for the Amazon Managed Grafana workspace | `string` | `null` | no |
|
||||
| <a name="input_irsa_iam_permissions_boundary"></a> [irsa\_iam\_permissions\_boundary](#input\_irsa\_iam\_permissions\_boundary) | IAM permissions boundary for IRSA roles | `string` | `""` | no |
|
||||
| <a name="input_irsa_iam_role_path"></a> [irsa\_iam\_role\_path](#input\_irsa\_iam\_role\_path) | IAM role path for IRSA roles | `string` | `"/"` | no |
|
||||
| <a name="input_managed_grafana_workspace_id"></a> [managed\_grafana\_workspace\_id](#input\_managed\_grafana\_workspace\_id) | Amazon Managed Grafana Workspace ID | `string` | `""` | no |
|
||||
| <a name="input_managed_prometheus_workspace_id"></a> [managed\_prometheus\_workspace\_id](#input\_managed\_prometheus\_workspace\_id) | Amazon Managed Service for Prometheus Workspace ID | `string` | `""` | no |
|
||||
| <a name="input_managed_prometheus_workspace_region"></a> [managed\_prometheus\_workspace\_region](#input\_managed\_prometheus\_workspace\_region) | Region where Amazon Managed Service for Prometheus is deployed | `string` | `null` | no |
|
||||
| <a name="input_tags"></a> [tags](#input\_tags) | Additional tags (e.g. `map('BusinessUnit`,`XYZ`) | `map(string)` | `{}` | no |
|
||||
|
||||
## Outputs
|
||||
|
||||
| Name | Description |
|
||||
|------|-------------|
|
||||
| <a name="output_aws_region"></a> [aws\_region](#output\_aws\_region) | EKS Cluster Id |
|
||||
| <a name="output_eks_cluster_id"></a> [eks\_cluster\_id](#output\_eks\_cluster\_id) | EKS Cluster Id |
|
||||
| <a name="output_eks_cluster_version"></a> [eks\_cluster\_version](#output\_eks\_cluster\_version) | EKS Cluster version |
|
||||
| <a name="output_grafana_dashboards_folder_id"></a> [grafana\_dashboards\_folder\_id](#output\_grafana\_dashboards\_folder\_id) | Grafana folder ID for automatic dashboards. Required by workload modules |
|
||||
| <a name="output_managed_grafana_workspace_endpoint"></a> [managed\_grafana\_workspace\_endpoint](#output\_managed\_grafana\_workspace\_endpoint) | Amazon Managed Grafana workspace endpoint |
|
||||
| <a name="output_managed_prometheus_workspace_endpoint"></a> [managed\_prometheus\_workspace\_endpoint](#output\_managed\_prometheus\_workspace\_endpoint) | Amazon Managed Prometheus workspace endpoint |
|
||||
| <a name="output_managed_prometheus_workspace_id"></a> [managed\_prometheus\_workspace\_id](#output\_managed\_prometheus\_workspace\_id) | Amazon Managed Prometheus workspace ID |
|
||||
| <a name="output_managed_prometheus_workspace_region"></a> [managed\_prometheus\_workspace\_region](#output\_managed\_prometheus\_workspace\_region) | Amazon Managed Prometheus workspace region |
|
||||
<!-- END OF PRE-COMMIT-TERRAFORM DOCS HOOK -->
|
||||
|
||||
## Contributing
|
||||
|
||||
See [CONTRIBUTING](CONTRIBUTING.md#security-issue-notifications) for more information.
|
||||
|
||||
## License
|
||||
|
||||
This project is licensed under the Apache-2.0 License.
|
||||
|
||||
Apache-2.0 Licensed. See [LICENSE](https://github.com/aws-observability/terraform-aws-eks-blueprints/blob/main/LICENSE).
|
||||
|
||||
@@ -0,0 +1,19 @@
|
||||
# AWS Observability Accelerator for Terraform
|
||||
|
||||

|
||||
|
||||
Welcome to AWS Observability Accelerator for Terraform!
|
||||
|
||||
|
||||
## What is AWS Observability Accelerator for Terraform
|
||||
|
||||
|
||||
## Examples
|
||||
|
||||
|
||||
## Workshop
|
||||
|
||||
|
||||
## Motivation
|
||||
|
||||
## What can I do with this Solution?
|
||||
@@ -0,0 +1,99 @@
|
||||
# EKS Cluster Deployment with new VPC
|
||||
|
||||
Note: This example is a subset from [this EKS Blueprint example](https://github.com/aws-ia/terraform-aws-eks-blueprints/tree/main/examples/eks-cluster-with-new-vpc)
|
||||
|
||||
This example deploys the following Basic EKS Cluster with VPC
|
||||
|
||||
- Creates a new sample VPC, 3 Private Subnets and 3 Public Subnets
|
||||
- Creates Internet gateway for Public Subnets and NAT Gateway for Private Subnets
|
||||
- Creates EKS Cluster Control plane with one managed node group
|
||||
|
||||
## How to Deploy
|
||||
|
||||
### Prerequisites
|
||||
|
||||
Ensure that you have installed the following tools in your Mac or Windows Laptop before start working with this module and run Terraform Plan and Apply
|
||||
|
||||
1. [AWS CLI](https://docs.aws.amazon.com/cli/latest/userguide/install-cliv2.html)
|
||||
2. [Kubectl](https://Kubernetes.io/docs/tasks/tools/)
|
||||
3. [Terraform](https://learn.hashicorp.com/tutorials/terraform/install-cli)
|
||||
|
||||
### Minimum IAM Policy
|
||||
|
||||
> **Note**: The policy resource is set as `*` to allow all resources, this is not a recommended practice.
|
||||
|
||||
You can find the policy [here](min-iam-policy.json)
|
||||
|
||||
|
||||
### Deployment Steps
|
||||
|
||||
#### Step 1: Clone the repo using the command below
|
||||
|
||||
```sh
|
||||
git clone https://github.com/aws-observability/terraform-aws-observability-accelerator.git
|
||||
```
|
||||
|
||||
#### Step 2: Run Terraform INIT
|
||||
|
||||
Initialize a working directory with configuration files
|
||||
|
||||
```sh
|
||||
cd examples/eks-cluster-with-vpc/
|
||||
terraform init
|
||||
```
|
||||
|
||||
#### Step 3: Run Terraform PLAN
|
||||
|
||||
Verify the resources created by this execution
|
||||
|
||||
```sh
|
||||
export TF_VAR_aws_region=<ENTER YOUR REGION> # Select your own region
|
||||
terraform plan
|
||||
```
|
||||
|
||||
#### Step 4: Finally, Terraform APPLY
|
||||
|
||||
**Deploy the pattern**
|
||||
|
||||
```sh
|
||||
terraform apply
|
||||
```
|
||||
|
||||
Enter `yes` to apply.
|
||||
|
||||
### Configure `kubectl` and test cluster
|
||||
|
||||
EKS Cluster details can be extracted from terraform output or from AWS Console to get the name of cluster.
|
||||
This following command used to update the `kubeconfig` in your local machine where you run kubectl commands to interact with your EKS Cluster.
|
||||
|
||||
#### Step 5: Run `update-kubeconfig` command
|
||||
|
||||
`~/.kube/config` file gets updated with cluster details and certificate from the below command
|
||||
|
||||
aws eks --region <enter-your-region> update-kubeconfig --name <cluster-name>
|
||||
|
||||
#### Step 6: List all the worker nodes by running the command below
|
||||
|
||||
kubectl get nodes
|
||||
|
||||
#### Step 7: List all the pods running in `kube-system` namespace
|
||||
|
||||
kubectl get pods -n kube-system
|
||||
|
||||
## Cleanup
|
||||
|
||||
To clean up your environment, destroy the Terraform modules in reverse order.
|
||||
|
||||
Destroy the Kubernetes Add-ons, EKS cluster with Node groups and VPC
|
||||
|
||||
```sh
|
||||
terraform destroy -target="module.eks_blueprints_kubernetes_addons" -auto-approve
|
||||
terraform destroy -target="module.eks_blueprints" -auto-approve
|
||||
terraform destroy -target="module.vpc" -auto-approve
|
||||
```
|
||||
|
||||
Finally, destroy any additional resources that are not in the above modules
|
||||
|
||||
```sh
|
||||
terraform destroy -auto-approve
|
||||
```
|
||||
@@ -0,0 +1,119 @@
|
||||
provider "aws" {
|
||||
region = local.region
|
||||
}
|
||||
|
||||
provider "kubernetes" {
|
||||
host = module.eks_blueprints.eks_cluster_endpoint
|
||||
cluster_ca_certificate = base64decode(module.eks_blueprints.eks_cluster_certificate_authority_data)
|
||||
token = data.aws_eks_cluster_auth.this.token
|
||||
}
|
||||
|
||||
provider "helm" {
|
||||
kubernetes {
|
||||
host = module.eks_blueprints.eks_cluster_endpoint
|
||||
cluster_ca_certificate = base64decode(module.eks_blueprints.eks_cluster_certificate_authority_data)
|
||||
token = data.aws_eks_cluster_auth.this.token
|
||||
}
|
||||
}
|
||||
|
||||
data "aws_eks_cluster_auth" "this" {
|
||||
name = module.eks_blueprints.eks_cluster_id
|
||||
}
|
||||
|
||||
data "aws_availability_zones" "available" {}
|
||||
|
||||
locals {
|
||||
name = basename(path.cwd)
|
||||
cluster_name = coalesce(var.cluster_name, local.name)
|
||||
region = var.aws_region
|
||||
|
||||
vpc_cidr = "10.0.0.0/16"
|
||||
azs = slice(data.aws_availability_zones.available.names, 0, 3)
|
||||
|
||||
tags = {
|
||||
Blueprint = local.name
|
||||
GithubRepo = "github.com/aws-observability/terraform-aws-observability-accelerator"
|
||||
}
|
||||
}
|
||||
|
||||
#---------------------------------------------------------------
|
||||
# EKS Blueprints
|
||||
#---------------------------------------------------------------
|
||||
|
||||
module "eks_blueprints" {
|
||||
source = "github.com/aws-ia/terraform-aws-eks-blueprints"
|
||||
|
||||
cluster_name = local.cluster_name
|
||||
cluster_version = "1.23"
|
||||
|
||||
vpc_id = module.vpc.vpc_id
|
||||
private_subnet_ids = module.vpc.private_subnets
|
||||
|
||||
managed_node_groups = {
|
||||
mg_5 = {
|
||||
node_group_name = "managed-ondemand"
|
||||
instance_types = ["t3.xlarge"]
|
||||
min_size = 2
|
||||
subnet_ids = module.vpc.private_subnets
|
||||
}
|
||||
}
|
||||
|
||||
tags = local.tags
|
||||
}
|
||||
|
||||
module "eks_blueprints_kubernetes_addons" {
|
||||
source = "github.com/aws-ia/terraform-aws-eks-blueprints/modules/kubernetes-addons"
|
||||
|
||||
eks_cluster_id = module.eks_blueprints.eks_cluster_id
|
||||
eks_cluster_endpoint = module.eks_blueprints.eks_cluster_endpoint
|
||||
eks_oidc_provider = module.eks_blueprints.oidc_provider
|
||||
eks_cluster_version = module.eks_blueprints.eks_cluster_version
|
||||
|
||||
# EKS Managed Add-ons
|
||||
enable_amazon_eks_vpc_cni = true
|
||||
enable_amazon_eks_coredns = true
|
||||
enable_amazon_eks_kube_proxy = true
|
||||
enable_amazon_eks_aws_ebs_csi_driver = true
|
||||
|
||||
tags = local.tags
|
||||
}
|
||||
|
||||
#---------------------------------------------------------------
|
||||
# Supporting Resources
|
||||
#---------------------------------------------------------------
|
||||
|
||||
module "vpc" {
|
||||
source = "terraform-aws-modules/vpc/aws"
|
||||
version = "~> 3.0"
|
||||
|
||||
name = local.name
|
||||
cidr = local.vpc_cidr
|
||||
|
||||
azs = local.azs
|
||||
public_subnets = [for k, v in local.azs : cidrsubnet(local.vpc_cidr, 8, k)]
|
||||
private_subnets = [for k, v in local.azs : cidrsubnet(local.vpc_cidr, 8, k + 10)]
|
||||
|
||||
enable_nat_gateway = true
|
||||
single_nat_gateway = true
|
||||
enable_dns_hostnames = true
|
||||
|
||||
# Manage so we can name
|
||||
manage_default_network_acl = true
|
||||
default_network_acl_tags = { Name = "${local.name}-default" }
|
||||
manage_default_route_table = true
|
||||
default_route_table_tags = { Name = "${local.name}-default" }
|
||||
manage_default_security_group = true
|
||||
default_security_group_tags = { Name = "${local.name}-default" }
|
||||
|
||||
public_subnet_tags = {
|
||||
"kubernetes.io/cluster/${local.cluster_name}" = "shared"
|
||||
"kubernetes.io/role/elb" = 1
|
||||
}
|
||||
|
||||
private_subnet_tags = {
|
||||
"kubernetes.io/cluster/${local.cluster_name}" = "shared"
|
||||
"kubernetes.io/role/internal-elb" = 1
|
||||
}
|
||||
|
||||
tags = local.tags
|
||||
}
|
||||
@@ -0,0 +1,105 @@
|
||||
{
|
||||
"Version": "2012-10-17",
|
||||
"Statement": [
|
||||
{
|
||||
"Effect": "Allow",
|
||||
"Action": [
|
||||
"ec2:AllocateAddress",
|
||||
"ec2:AssociateRouteTable",
|
||||
"ec2:AttachInternetGateway",
|
||||
"ec2:AuthorizeSecurityGroupEgress",
|
||||
"ec2:AuthorizeSecurityGroupIngress",
|
||||
"ec2:CreateInternetGateway",
|
||||
"ec2:CreateNatGateway",
|
||||
"ec2:CreateNetworkAclEntry",
|
||||
"ec2:CreateRoute",
|
||||
"ec2:CreateRouteTable",
|
||||
"ec2:CreateSecurityGroup",
|
||||
"ec2:CreateSubnet",
|
||||
"ec2:CreateTags",
|
||||
"ec2:CreateVpc",
|
||||
"ec2:DeleteInternetGateway",
|
||||
"ec2:DeleteNatGateway",
|
||||
"ec2:DeleteNetworkAclEntry",
|
||||
"ec2:DeleteRoute",
|
||||
"ec2:DeleteRouteTable",
|
||||
"ec2:DeleteSecurityGroup",
|
||||
"ec2:DeleteSubnet",
|
||||
"ec2:DeleteTags",
|
||||
"ec2:DeleteVpc",
|
||||
"ec2:DescribeAccountAttributes",
|
||||
"ec2:DescribeAddresses",
|
||||
"ec2:DescribeAvailabilityZones",
|
||||
"ec2:DescribeInternetGateways",
|
||||
"ec2:DescribeNatGateways",
|
||||
"ec2:DescribeNetworkAcls",
|
||||
"ec2:DescribeNetworkInterfaces",
|
||||
"ec2:DescribeRouteTables",
|
||||
"ec2:DescribeSecurityGroups",
|
||||
"ec2:DescribeSubnets",
|
||||
"ec2:DescribeTags",
|
||||
"ec2:DescribeVpcAttribute",
|
||||
"ec2:DescribeVpcClassicLink",
|
||||
"ec2:DescribeVpcClassicLinkDnsSupport",
|
||||
"ec2:DescribeVpcs",
|
||||
"ec2:DetachInternetGateway",
|
||||
"ec2:DisassociateRouteTable",
|
||||
"ec2:ModifySubnetAttribute",
|
||||
"ec2:ModifyVpcAttribute",
|
||||
"ec2:ReleaseAddress",
|
||||
"ec2:RevokeSecurityGroupEgress",
|
||||
"ec2:RevokeSecurityGroupIngress",
|
||||
"eks:CreateAddon",
|
||||
"eks:CreateCluster",
|
||||
"eks:CreateNodegroup",
|
||||
"eks:DeleteAddon",
|
||||
"eks:DeleteCluster",
|
||||
"eks:DeleteNodegroup",
|
||||
"eks:DescribeAddon",
|
||||
"eks:DescribeAddonVersions",
|
||||
"eks:DescribeCluster",
|
||||
"eks:DescribeNodegroup",
|
||||
"iam:AddRoleToInstanceProfile",
|
||||
"iam:AttachRolePolicy",
|
||||
"iam:CreateInstanceProfile",
|
||||
"iam:CreateOpenIDConnectProvider",
|
||||
"iam:CreatePolicy",
|
||||
"iam:CreateRole",
|
||||
"iam:CreateServiceLinkedRole",
|
||||
"iam:DeleteInstanceProfile",
|
||||
"iam:DeleteOpenIDConnectProvider",
|
||||
"iam:DeletePolicy",
|
||||
"iam:DeleteRole",
|
||||
"iam:DetachRolePolicy",
|
||||
"iam:GetInstanceProfile",
|
||||
"iam:GetOpenIDConnectProvider",
|
||||
"iam:GetPolicy",
|
||||
"iam:GetPolicyVersion",
|
||||
"iam:GetRole",
|
||||
"iam:ListAttachedRolePolicies",
|
||||
"iam:ListInstanceProfilesForRole",
|
||||
"iam:ListPolicyVersions",
|
||||
"iam:ListRolePolicies",
|
||||
"iam:PassRole",
|
||||
"iam:RemoveRoleFromInstanceProfile",
|
||||
"iam:TagInstanceProfile",
|
||||
"kms:CreateAlias",
|
||||
"kms:CreateKey",
|
||||
"kms:DeleteAlias",
|
||||
"kms:DescribeKey",
|
||||
"kms:EnableKeyRotation",
|
||||
"kms:GetKeyPolicy",
|
||||
"kms:GetKeyRotationStatus",
|
||||
"kms:ListAliases",
|
||||
"kms:ListResourceTags",
|
||||
"kms:PutKeyPolicy",
|
||||
"kms:ScheduleKeyDeletion",
|
||||
"kms:TagResource",
|
||||
"s3:GetObject",
|
||||
"s3:ListBucket",
|
||||
"s3:PutObject"
|
||||
],
|
||||
"Resource": "*"
|
||||
}
|
||||
]
|
||||
}
|
||||
@@ -0,0 +1,49 @@
|
||||
output "vpc_private_subnet_cidr" {
|
||||
description = "VPC private subnet CIDR"
|
||||
value = module.vpc.private_subnets_cidr_blocks
|
||||
}
|
||||
|
||||
output "vpc_public_subnet_cidr" {
|
||||
description = "VPC public subnet CIDR"
|
||||
value = module.vpc.public_subnets_cidr_blocks
|
||||
}
|
||||
|
||||
output "vpc_cidr" {
|
||||
description = "VPC CIDR"
|
||||
value = module.vpc.vpc_cidr_block
|
||||
}
|
||||
|
||||
output "eks_cluster_id" {
|
||||
description = "EKS cluster ID"
|
||||
value = module.eks_blueprints.eks_cluster_id
|
||||
}
|
||||
|
||||
output "eks_managed_nodegroups" {
|
||||
description = "EKS managed node groups"
|
||||
value = module.eks_blueprints.managed_node_groups
|
||||
}
|
||||
|
||||
output "eks_managed_nodegroup_ids" {
|
||||
description = "EKS managed node group ids"
|
||||
value = module.eks_blueprints.managed_node_groups_id
|
||||
}
|
||||
|
||||
output "eks_managed_nodegroup_arns" {
|
||||
description = "EKS managed node group arns"
|
||||
value = module.eks_blueprints.managed_node_group_arn
|
||||
}
|
||||
|
||||
output "eks_managed_nodegroup_role_name" {
|
||||
description = "EKS managed node group role name"
|
||||
value = module.eks_blueprints.managed_node_group_iam_role_names
|
||||
}
|
||||
|
||||
output "eks_managed_nodegroup_status" {
|
||||
description = "EKS managed node group status"
|
||||
value = module.eks_blueprints.managed_node_groups_status
|
||||
}
|
||||
|
||||
output "configure_kubectl" {
|
||||
description = "Configure kubectl: make sure you're logged in with the correct AWS profile and run the following command to update your kubeconfig"
|
||||
value = module.eks_blueprints.configure_kubectl
|
||||
}
|
||||
@@ -0,0 +1,9 @@
|
||||
variable "cluster_name" {
|
||||
description = "Name of cluster - used by Terratest for e2e test automation"
|
||||
type = string
|
||||
default = ""
|
||||
}
|
||||
variable "aws_region" {
|
||||
description = "AWS Region"
|
||||
type = string
|
||||
}
|
||||
@@ -0,0 +1,26 @@
|
||||
terraform {
|
||||
required_version = ">= 1.0.0"
|
||||
|
||||
required_providers {
|
||||
aws = {
|
||||
source = "hashicorp/aws"
|
||||
version = ">= 4.0.0"
|
||||
}
|
||||
kubernetes = {
|
||||
source = "hashicorp/kubernetes"
|
||||
version = ">= 2.10"
|
||||
}
|
||||
kubectl = {
|
||||
source = "gavinbunney/kubectl"
|
||||
version = ">= 1.14"
|
||||
}
|
||||
helm = {
|
||||
source = "hashicorp/helm"
|
||||
version = ">= 2.4.1"
|
||||
}
|
||||
grafana = {
|
||||
source = "grafana/grafana"
|
||||
version = ">= 1.25.0"
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,182 @@
|
||||
# Existing Cluster with the AWS Observability accelerator base module and Infrastructure monitoring
|
||||
|
||||
|
||||
This example demonstrates how to use the AWS Observability Accelerator Terraform
|
||||
modules with Infrastructure monitoring enabled.
|
||||
The current example deploys the [AWS Distro for OpenTelemetry Operator](https://docs.aws.amazon.com/eks/latest/userguide/opentelemetry.html) for Amazon EKS with its requirements and make use of existing
|
||||
Amazon Managed Service for Prometheus and Amazon Managed Grafana workspaces.
|
||||
|
||||
It is based on the `infrastructure monitoring`, one of our [workloads modules](../../modules/workloads/)
|
||||
to provide an existing EKS cluster with an OpenTelemetry collector,
|
||||
curated Grafana dashboards, Prometheus alerting and recording rules with multiple
|
||||
configuration options on the cluster infrastructure.
|
||||
|
||||
|
||||
## Prerequisites
|
||||
|
||||
Ensure that you have the following tools installed locally:
|
||||
|
||||
1. [aws cli](https://docs.aws.amazon.com/cli/latest/userguide/getting-started-install.html)
|
||||
2. [kubectl](https://kubernetes.io/docs/tasks/tools/)
|
||||
3. [terraform](https://learn.hashicorp.com/tutorials/terraform/install-cli)
|
||||
|
||||
|
||||
## Setup
|
||||
|
||||
This example uses a local terraform state. If you need states to be saved remotely,
|
||||
on Amazon S3 for example, visit the [terraform remote states](https://www.terraform.io/language/state/remote) documentation
|
||||
|
||||
1. Clone the repo using the command below
|
||||
|
||||
```
|
||||
git clone https://github.com/aws-observability/terraform-aws-observability-accelerator.git
|
||||
```
|
||||
|
||||
2. Initialize terraform
|
||||
|
||||
```console
|
||||
cd examples/existing-cluster-with-base-and-infra
|
||||
terraform init
|
||||
```
|
||||
|
||||
3. AWS Region
|
||||
|
||||
Specify the AWS Region where the resources will be deployed. Edit the `terraform.tfvars` file and modify `aws_region="..."`. You can also use environement variables `export TF_VAR_aws_region=xxx`.
|
||||
|
||||
4. Amazon EKS Cluster
|
||||
|
||||
To run this example, you need to provide your EKS cluster name.
|
||||
If you don't have a cluster ready, visit [this example](https://github.com/aws-ia/terraform-aws-eks-blueprints/tree/main/examples/eks-cluster-with-new-vpc)
|
||||
first to create a new one.
|
||||
|
||||
Add your cluster name for `eks_cluster_id="..."` to the `terraform.tfvars` or use an environment variable `export TF_VAR_eks_cluster_id=xxx`.
|
||||
|
||||
5. Amazon Managed Service for Prometheus workspace (optional)
|
||||
|
||||
If you have an existing workspace, add `managed_prometheus_workspace_id=ws-xxx`
|
||||
or use an environment variable `export TF_VAR_managed_prometheus_workspace_id=ws-xxx`.
|
||||
|
||||
If you don't specify anything a new workspace will be created for you.
|
||||
|
||||
6. Amazon Managed Grafana workspace
|
||||
|
||||
If you have an existing workspace, add `managed_grafana_workspace_id=g-xxx`
|
||||
or use an environment variable `export TF_VAR_managed_grafana_workspace_id=g-xxx`.
|
||||
|
||||
7. Grafana API Key
|
||||
|
||||
- Give admin access to the SSO user you set up when creating the Amazon Managed Grafana Workspace:
|
||||
- In the AWS Console, navigate to Amazon Grafana. In the left navigation bar, click **All workspaces**, then click on the workspace name you are using for this example.
|
||||
- Under **Authentication** within **AWS Single Sign-On (SSO)**, click **Configure users and user groups**
|
||||
- Check the box next to the SSO user you created and click **Make admin**
|
||||
- From the workspace in the AWS console, click on the `Grafana workspace URL` to open the workspace
|
||||
- If you don't see the gear icon in the left navigation bar, log out and log back in.
|
||||
- Click on the gear icon, then click on the **API keys** tab.
|
||||
- Click **Add API key**, fill in the _Key name_ field and select _Admin_ as the Role.
|
||||
- Copy your API key into `terraform.tfvars` under the `grafana_api_key` variable (`grafana_api_key="xxx"`) or set as an environment variable on your CLI (`export TF_VAR_grafana_api_key="xxx"`)
|
||||
|
||||
|
||||
## Deploy
|
||||
|
||||
```sh
|
||||
terraform apply -var-file=terraform.tfvars
|
||||
```
|
||||
|
||||
or if you had setup environment variables, run
|
||||
|
||||
```sh
|
||||
terraform apply
|
||||
```
|
||||
|
||||
## Visualization
|
||||
|
||||
1. Prometheus datasource on Grafana
|
||||
|
||||
Open your Grafana workspace and under Configuration -> Data sources, you should see `aws-observability-accelerator`. Open and click `Save & test`. You should see a notification confirming that the Amazon Managed Service for Prometheus workspace is ready to be used on Grafana.
|
||||
|
||||
2. Grafana dashboards
|
||||
|
||||
Go to the Dashboards panel of your Grafana workspace. You should see a list of dashboards under the `Observability Accelerator Dashboards`
|
||||
|
||||
<img width="830" alt="image" src="https://user-images.githubusercontent.com/10175027/188886724-f566fd55-018e-4352-abc0-4c9470f87694.png">
|
||||
|
||||
|
||||
Open a specific dashboard and you should be able to view its visualization
|
||||
|
||||
<img width="1721" alt="Screenshot 2022-08-30 at 20 01 32" src="https://user-images.githubusercontent.com/10175027/187515925-67864dd1-2b35-4be0-a15e-1e36805e8b29.png">
|
||||
|
||||
2. Amazon Managed Service for Prometheus rules and alerts
|
||||
|
||||
Open the Amazon Managed Service for Prometheus console and view the details of your workspace. Under the `Rules management` tab, you should find new rules deployed.
|
||||
|
||||
<img width="1629" alt="image" src="https://user-images.githubusercontent.com/10175027/189301297-4865e75d-2d71-434f-b5d0-9750b3533632.png">
|
||||
|
||||
|
||||
To setup your alert receiver, with Amazon SNS, follow [this documentation](https://docs.aws.amazon.com/prometheus/latest/userguide/AMP-alertmanager-receiver.html)
|
||||
|
||||
## Advanced configuration
|
||||
|
||||
1. Cross-region Amazon Managed Prometheus workspace
|
||||
|
||||
If your existing Amazon Managed Prometheus workspace is in another AWS Region,
|
||||
add this `managed_prometheus_region=xxx` and `managed_prometheus_workspace_id=ws-xxx`.
|
||||
|
||||
2. Cross-region Amazon Managed Grafana workspace
|
||||
|
||||
If your existing Amazon Managed Prometheus workspace is in another AWS Region,
|
||||
add this `managed_prometheus_region=xxx` and `managed_prometheus_workspace_id=ws-xxx`.
|
||||
|
||||
|
||||
<!-- BEGINNING OF PRE-COMMIT-TERRAFORM DOCS HOOK -->
|
||||
## Requirements
|
||||
|
||||
| Name | Version |
|
||||
|------|---------|
|
||||
| <a name="requirement_terraform"></a> [terraform](#requirement\_terraform) | >= 1.0.0 |
|
||||
| <a name="requirement_aws"></a> [aws](#requirement\_aws) | >= 4.0.0 |
|
||||
| <a name="requirement_grafana"></a> [grafana](#requirement\_grafana) | >= 1.25.0 |
|
||||
| <a name="requirement_grafana"></a> [grafana](#requirement\_grafana) | >= 1.25.0 |
|
||||
| <a name="requirement_helm"></a> [helm](#requirement\_helm) | >= 2.4.1 |
|
||||
| <a name="requirement_kubectl"></a> [kubectl](#requirement\_kubectl) | >= 1.14 |
|
||||
| <a name="requirement_kubernetes"></a> [kubernetes](#requirement\_kubernetes) | >= 2.10 |
|
||||
|
||||
## Providers
|
||||
|
||||
| Name | Version |
|
||||
|------|---------|
|
||||
| <a name="provider_aws"></a> [aws](#provider\_aws) | >= 4.0.0 |
|
||||
|
||||
## Modules
|
||||
|
||||
| Name | Source | Version |
|
||||
|------|--------|---------|
|
||||
| <a name="module_eks_observability_accelerator"></a> [eks\_observability\_accelerator](#module\_eks\_observability\_accelerator) | ../../ | n/a |
|
||||
| <a name="module_workloads_infra"></a> [workloads\_infra](#module\_workloads\_infra) | ../../modules/workloads/infra | n/a |
|
||||
|
||||
## Resources
|
||||
|
||||
| Name | Type |
|
||||
|------|------|
|
||||
| [aws_eks_cluster.this](https://registry.terraform.io/providers/hashicorp/aws/latest/docs/data-sources/eks_cluster) | data source |
|
||||
| [aws_eks_cluster_auth.this](https://registry.terraform.io/providers/hashicorp/aws/latest/docs/data-sources/eks_cluster_auth) | data source |
|
||||
|
||||
## Inputs
|
||||
|
||||
| Name | Description | Type | Default | Required |
|
||||
|------|-------------|------|---------|:--------:|
|
||||
| <a name="input_aws_region"></a> [aws\_region](#input\_aws\_region) | AWS Region | `string` | n/a | yes |
|
||||
| <a name="input_eks_cluster_id"></a> [eks\_cluster\_id](#input\_eks\_cluster\_id) | Name of the EKS cluster | `string` | n/a | yes |
|
||||
| <a name="input_grafana_api_key"></a> [grafana\_api\_key](#input\_grafana\_api\_key) | API key for authorizing the Grafana provider to make changes to Amazon Managed Grafana | `string` | `""` | no |
|
||||
| <a name="input_managed_grafana_workspace_id"></a> [managed\_grafana\_workspace\_id](#input\_managed\_grafana\_workspace\_id) | Amazon Managed Grafana Workspace ID | `string` | `""` | no |
|
||||
| <a name="input_managed_prometheus_workspace_id"></a> [managed\_prometheus\_workspace\_id](#input\_managed\_prometheus\_workspace\_id) | Amazon Managed Service for Prometheus Workspace ID | `string` | `""` | no |
|
||||
|
||||
## Outputs
|
||||
|
||||
| Name | Description |
|
||||
|------|-------------|
|
||||
| <a name="output_aws_region"></a> [aws\_region](#output\_aws\_region) | AWS Region |
|
||||
| <a name="output_eks_cluster_id"></a> [eks\_cluster\_id](#output\_eks\_cluster\_id) | EKS Cluster Id |
|
||||
| <a name="output_eks_cluster_version"></a> [eks\_cluster\_version](#output\_eks\_cluster\_version) | EKS Cluster version |
|
||||
| <a name="output_managed_prometheus_workspace_endpoint"></a> [managed\_prometheus\_workspace\_endpoint](#output\_managed\_prometheus\_workspace\_endpoint) | Amazon Managed Prometheus workspace endpoint |
|
||||
| <a name="output_managed_prometheus_workspace_id"></a> [managed\_prometheus\_workspace\_id](#output\_managed\_prometheus\_workspace\_id) | Amazon Managed Prometheus workspace ID |
|
||||
<!-- END OF PRE-COMMIT-TERRAFORM DOCS HOOK -->
|
||||
@@ -0,0 +1,94 @@
|
||||
provider "aws" {
|
||||
region = local.region
|
||||
}
|
||||
|
||||
data "aws_eks_cluster_auth" "this" {
|
||||
name = var.eks_cluster_id
|
||||
}
|
||||
|
||||
data "aws_eks_cluster" "this" {
|
||||
name = var.eks_cluster_id
|
||||
}
|
||||
|
||||
provider "kubernetes" {
|
||||
host = local.eks_cluster_endpoint
|
||||
cluster_ca_certificate = base64decode(data.aws_eks_cluster.this.certificate_authority[0].data)
|
||||
token = data.aws_eks_cluster_auth.this.token
|
||||
}
|
||||
|
||||
provider "helm" {
|
||||
kubernetes {
|
||||
host = local.eks_cluster_endpoint
|
||||
cluster_ca_certificate = base64decode(data.aws_eks_cluster.this.certificate_authority[0].data)
|
||||
token = data.aws_eks_cluster_auth.this.token
|
||||
}
|
||||
}
|
||||
|
||||
locals {
|
||||
region = var.aws_region
|
||||
eks_cluster_endpoint = data.aws_eks_cluster.this.endpoint
|
||||
create_new_workspace = var.managed_prometheus_workspace_id == "" ? true : false
|
||||
tags = {
|
||||
Source = "github.com/aws-observability/terraform-aws-observability-accelerator"
|
||||
}
|
||||
}
|
||||
|
||||
# deploys the base module
|
||||
module "eks_observability_accelerator" {
|
||||
# source = "aws-observability/terrarom-aws-observability-accelerator"
|
||||
source = "../../"
|
||||
|
||||
aws_region = var.aws_region
|
||||
eks_cluster_id = var.eks_cluster_id
|
||||
|
||||
# deploys AWS Distro for OpenTelemetry operator into the cluster
|
||||
enable_amazon_eks_adot = true
|
||||
|
||||
# reusing existing certificate manager? defaults to true
|
||||
enable_cert_manager = true
|
||||
|
||||
# creates a new Amazon Managed Prometheus workspace, defaults to true
|
||||
enable_managed_prometheus = local.create_new_workspace
|
||||
|
||||
# reusing existing Amazon Managed Prometheus if specified
|
||||
managed_prometheus_workspace_id = var.managed_prometheus_workspace_id
|
||||
managed_prometheus_workspace_region = null # defaults to the current region, useful for cross region scenarios (same account)
|
||||
|
||||
# sets up the Amazon Managed Prometheus alert manager at the workspace level
|
||||
enable_alertmanager = true
|
||||
|
||||
# reusing existing Amazon Managed Grafana workspace
|
||||
enable_managed_grafana = false
|
||||
managed_grafana_workspace_id = var.managed_grafana_workspace_id
|
||||
grafana_api_key = var.grafana_api_key
|
||||
|
||||
tags = local.tags
|
||||
}
|
||||
|
||||
# https://www.terraform.io/language/modules/develop/providers
|
||||
# A module intended to be called by one or more other modules must not contain
|
||||
# any provider blocks.
|
||||
# This allows forcing dependency between base and workloads module
|
||||
provider "grafana" {
|
||||
url = module.eks_observability_accelerator.managed_grafana_workspace_endpoint
|
||||
auth = var.grafana_api_key
|
||||
}
|
||||
|
||||
module "workloads_infra" {
|
||||
source = "../../modules/workloads/infra"
|
||||
# source = "aws-observability/terrarom-aws-observability-accelerator/workloads/infra"
|
||||
|
||||
eks_cluster_id = module.eks_observability_accelerator.eks_cluster_id
|
||||
|
||||
dashboards_folder_id = module.eks_observability_accelerator.grafana_dashboards_folder_id
|
||||
managed_prometheus_workspace_id = module.eks_observability_accelerator.managed_prometheus_workspace_id
|
||||
|
||||
managed_prometheus_workspace_endpoint = module.eks_observability_accelerator.managed_prometheus_workspace_endpoint
|
||||
managed_prometheus_workspace_region = module.eks_observability_accelerator.managed_prometheus_workspace_region
|
||||
|
||||
tags = local.tags
|
||||
|
||||
depends_on = [
|
||||
module.eks_observability_accelerator
|
||||
]
|
||||
}
|
||||
@@ -0,0 +1,24 @@
|
||||
output "eks_cluster_id" {
|
||||
description = "EKS Cluster Id"
|
||||
value = module.eks_observability_accelerator.eks_cluster_id
|
||||
}
|
||||
|
||||
output "aws_region" {
|
||||
description = "AWS Region"
|
||||
value = module.eks_observability_accelerator.aws_region
|
||||
}
|
||||
|
||||
output "eks_cluster_version" {
|
||||
description = "EKS Cluster version"
|
||||
value = module.eks_observability_accelerator.eks_cluster_version
|
||||
}
|
||||
|
||||
output "managed_prometheus_workspace_endpoint" {
|
||||
description = "Amazon Managed Prometheus workspace endpoint"
|
||||
value = module.eks_observability_accelerator.managed_prometheus_workspace_endpoint
|
||||
}
|
||||
|
||||
output "managed_prometheus_workspace_id" {
|
||||
description = "Amazon Managed Prometheus workspace ID"
|
||||
value = module.eks_observability_accelerator.managed_prometheus_workspace_id
|
||||
}
|
||||
@@ -0,0 +1,14 @@
|
||||
# (mandatory) AWS Region where your resources will be located
|
||||
aws_region = ""
|
||||
|
||||
# (mandatory) EKS Cluster name
|
||||
eks_cluster_id = ""
|
||||
|
||||
# (optional) Leave it empty for a new workspace to be created
|
||||
managed_prometheus_workspace_id = ""
|
||||
|
||||
# (mandatory) Amazon Managed Grafana Workspace ID: ex: g-abc123
|
||||
managed_grafana_workspace_id = ""
|
||||
|
||||
# (mandatory) Grafana API Key - https://docs.aws.amazon.com/grafana/latest/userguide/API_key_console.html
|
||||
grafana_api_key = ""
|
||||
@@ -0,0 +1,24 @@
|
||||
variable "eks_cluster_id" {
|
||||
description = "Name of the EKS cluster"
|
||||
type = string
|
||||
}
|
||||
variable "aws_region" {
|
||||
description = "AWS Region"
|
||||
type = string
|
||||
}
|
||||
variable "managed_prometheus_workspace_id" {
|
||||
description = "Amazon Managed Service for Prometheus Workspace ID"
|
||||
type = string
|
||||
default = ""
|
||||
}
|
||||
variable "managed_grafana_workspace_id" {
|
||||
description = "Amazon Managed Grafana Workspace ID"
|
||||
type = string
|
||||
default = ""
|
||||
}
|
||||
variable "grafana_api_key" {
|
||||
description = "API key for authorizing the Grafana provider to make changes to Amazon Managed Grafana"
|
||||
type = string
|
||||
default = ""
|
||||
sensitive = true
|
||||
}
|
||||
@@ -0,0 +1,25 @@
|
||||
terraform {
|
||||
required_version = ">= 1.0.0"
|
||||
required_providers {
|
||||
aws = {
|
||||
source = "hashicorp/aws"
|
||||
version = ">= 4.0.0"
|
||||
}
|
||||
kubernetes = {
|
||||
source = "hashicorp/kubernetes"
|
||||
version = ">= 2.10"
|
||||
}
|
||||
kubectl = {
|
||||
source = "gavinbunney/kubectl"
|
||||
version = ">= 1.14"
|
||||
}
|
||||
helm = {
|
||||
source = "hashicorp/helm"
|
||||
version = ">= 2.4.1"
|
||||
}
|
||||
grafana = {
|
||||
source = "grafana/grafana"
|
||||
version = ">= 1.25.0"
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,45 @@
|
||||
data "aws_partition" "current" {}
|
||||
|
||||
data "aws_caller_identity" "current" {}
|
||||
|
||||
data "aws_region" "current" {}
|
||||
|
||||
data "aws_eks_cluster" "eks_cluster" {
|
||||
name = var.eks_cluster_id
|
||||
}
|
||||
|
||||
data "aws_grafana_workspace" "this" {
|
||||
count = var.managed_grafana_workspace_id == "" ? 0 : 1
|
||||
workspace_id = var.managed_grafana_workspace_id
|
||||
}
|
||||
|
||||
locals {
|
||||
eks_oidc_issuer_url = replace(data.aws_eks_cluster.eks_cluster.identity[0].oidc[0].issuer, "https://", "")
|
||||
eks_cluster_endpoint = data.aws_eks_cluster.eks_cluster.endpoint
|
||||
eks_cluster_version = data.aws_eks_cluster.eks_cluster.version
|
||||
|
||||
# if region is not passed, we assume the current one
|
||||
amp_ws_region = coalesce(var.managed_prometheus_workspace_region, data.aws_region.current.name)
|
||||
amp_ws_id = var.enable_managed_prometheus ? aws_prometheus_workspace.this[0].id : var.managed_prometheus_workspace_id
|
||||
amp_ws_endpoint = "https://aps-workspaces.${local.amp_ws_region}.amazonaws.com/workspaces/${local.amp_ws_id}/"
|
||||
|
||||
# if grafana_workspace_id is supplied, we infer the endpoint from
|
||||
# computed region, else we create a new workspace
|
||||
amg_ws_endpoint = var.managed_grafana_workspace_id == "" ? "https://${module.managed_grafana[0].workspace_endpoint}" : "https://${data.aws_grafana_workspace.this[0].endpoint}"
|
||||
|
||||
context = {
|
||||
aws_caller_identity_account_id = data.aws_caller_identity.current.account_id
|
||||
aws_caller_identity_arn = data.aws_caller_identity.current.arn
|
||||
aws_eks_cluster_endpoint = local.eks_cluster_endpoint
|
||||
aws_partition_id = data.aws_partition.current.partition
|
||||
aws_region_name = data.aws_region.current.name
|
||||
eks_cluster_id = var.eks_cluster_id
|
||||
eks_oidc_issuer_url = local.eks_oidc_issuer_url
|
||||
eks_oidc_provider_arn = "arn:${data.aws_partition.current.partition}:iam::${data.aws_caller_identity.current.account_id}:oidc-provider/${local.eks_oidc_issuer_url}"
|
||||
tags = var.tags
|
||||
irsa_iam_role_path = var.irsa_iam_role_path
|
||||
irsa_iam_permissions_boundary = var.irsa_iam_permissions_boundary
|
||||
}
|
||||
|
||||
name = "aws-observability-accelerator"
|
||||
}
|
||||
@@ -0,0 +1,66 @@
|
||||
module "operator" {
|
||||
source = "./modules/add-ons/adot-operator"
|
||||
count = var.enable_amazon_eks_adot ? 1 : 0
|
||||
|
||||
enable_cert_manager = var.enable_cert_manager
|
||||
kubernetes_version = local.eks_cluster_version
|
||||
addon_context = local.context
|
||||
}
|
||||
|
||||
resource "aws_prometheus_workspace" "this" {
|
||||
count = var.enable_managed_prometheus ? 1 : 0
|
||||
|
||||
alias = local.name
|
||||
tags = var.tags
|
||||
}
|
||||
|
||||
resource "aws_prometheus_alert_manager_definition" "this" {
|
||||
count = var.enable_alertmanager ? 1 : 0
|
||||
|
||||
workspace_id = local.amp_ws_id
|
||||
|
||||
definition = <<EOF
|
||||
alertmanager_config: |
|
||||
route:
|
||||
receiver: 'default'
|
||||
receivers:
|
||||
- name: 'default'
|
||||
EOF
|
||||
}
|
||||
|
||||
module "managed_grafana" {
|
||||
count = var.enable_managed_grafana ? 1 : 0
|
||||
source = "terraform-aws-modules/managed-service-grafana/aws"
|
||||
version = "~> 1.3"
|
||||
|
||||
# Workspace
|
||||
name = local.name
|
||||
stack_set_name = local.name
|
||||
data_sources = ["PROMETHEUS"]
|
||||
associate_license = false
|
||||
|
||||
tags = var.tags
|
||||
}
|
||||
|
||||
provider "grafana" {
|
||||
url = local.amg_ws_endpoint
|
||||
auth = var.grafana_api_key
|
||||
}
|
||||
|
||||
resource "grafana_data_source" "amp" {
|
||||
type = "prometheus"
|
||||
name = local.name
|
||||
is_default = true
|
||||
url = local.amp_ws_endpoint
|
||||
json_data {
|
||||
http_method = "GET"
|
||||
sigv4_auth = true
|
||||
sigv4_auth_type = "workspace-iam-role"
|
||||
sigv4_region = local.amp_ws_region
|
||||
}
|
||||
}
|
||||
|
||||
# dashboards
|
||||
resource "grafana_folder" "this" {
|
||||
title = "Observability Accelerator Dashboards"
|
||||
}
|
||||
@@ -0,0 +1,71 @@
|
||||
# AWS OpenTelemetry Operator
|
||||
|
||||
[AWS Distro for OpenTelemetry (ADOT)](https://aws-otel.github.io/) is a secure,
|
||||
production-ready, AWS-supported distribution of the OpenTelemetry project.
|
||||
Part of the Cloud Native Computing Foundation, OpenTelemetry provides open
|
||||
source APIs, libraries, and agents to collect distributed traces and metrics
|
||||
for application monitoring.
|
||||
|
||||
This modules deploys either the
|
||||
[AWS Managed ADOT OpenTelemetry Operator for EKS](https://aws.amazon.com/about-aws/whats-new/2022/04/eks-opentelemetry-operator-now-available/)
|
||||
or [the OpenTelemetry Operator](https://github.com/open-telemetry/opentelemetry-helm-charts)
|
||||
through helm.
|
||||
The OpenTelemetry Operator is an implementation of a Kubernetes Operator.
|
||||
A Kubernetes Operator is a method of packaging, deploying and managing a
|
||||
Kubernetes-native application, which is both deployed on Kubernetes and
|
||||
managed using the Kubernetes APIs and kubectl tooling. The Kubernetes Operator
|
||||
is a custom controller, which introduces new object types through Custom Resource
|
||||
Definition (CRD), an extension mechanism in Kubernetes.
|
||||
In this case, the CRD that is managed by the OpenTelemetry Operator is the Collector.
|
||||
|
||||
> :warning: We do install [cert-manager](https://cert-manager.io/) as a [hard requirement](https://docs.aws.amazon.com/eks/latest/userguide/opentelemetry.html) for
|
||||
the ADOT Operator.
|
||||
|
||||
<!-- BEGINNING OF PRE-COMMIT-TERRAFORM DOCS HOOK -->
|
||||
## Requirements
|
||||
|
||||
| Name | Version |
|
||||
|------|---------|
|
||||
| <a name="requirement_terraform"></a> [terraform](#requirement\_terraform) | >= 1.0.0 |
|
||||
| <a name="requirement_aws"></a> [aws](#requirement\_aws) | >= 3.72 |
|
||||
| <a name="requirement_kubernetes"></a> [kubernetes](#requirement\_kubernetes) | >= 2.10 |
|
||||
|
||||
## Providers
|
||||
|
||||
| Name | Version |
|
||||
|------|---------|
|
||||
| <a name="provider_aws"></a> [aws](#provider\_aws) | >= 3.72 |
|
||||
| <a name="provider_kubernetes"></a> [kubernetes](#provider\_kubernetes) | >= 2.10 |
|
||||
|
||||
## Modules
|
||||
|
||||
| Name | Source | Version |
|
||||
|------|--------|---------|
|
||||
| <a name="module_cert_manager"></a> [cert\_manager](#module\_cert\_manager) | github.com/aws-ia/terraform-aws-eks-blueprints/modules/kubernetes-addons/cert-manager | n/a |
|
||||
|
||||
## Resources
|
||||
|
||||
| Name | Type |
|
||||
|------|------|
|
||||
| [aws_eks_addon.adot](https://registry.terraform.io/providers/hashicorp/aws/latest/docs/resources/eks_addon) | resource |
|
||||
| [kubernetes_cluster_role_binding_v1.adot](https://registry.terraform.io/providers/hashicorp/kubernetes/latest/docs/resources/cluster_role_binding_v1) | resource |
|
||||
| [kubernetes_cluster_role_v1.adot](https://registry.terraform.io/providers/hashicorp/kubernetes/latest/docs/resources/cluster_role_v1) | resource |
|
||||
| [kubernetes_namespace_v1.adot](https://registry.terraform.io/providers/hashicorp/kubernetes/latest/docs/resources/namespace_v1) | resource |
|
||||
| [kubernetes_role_binding_v1.adot](https://registry.terraform.io/providers/hashicorp/kubernetes/latest/docs/resources/role_binding_v1) | resource |
|
||||
| [kubernetes_role_v1.adot](https://registry.terraform.io/providers/hashicorp/kubernetes/latest/docs/resources/role_v1) | resource |
|
||||
| [aws_eks_addon_version.this](https://registry.terraform.io/providers/hashicorp/aws/latest/docs/data-sources/eks_addon_version) | data source |
|
||||
|
||||
## Inputs
|
||||
|
||||
| Name | Description | Type | Default | Required |
|
||||
|------|-------------|------|---------|:--------:|
|
||||
| <a name="input_addon_config"></a> [addon\_config](#input\_addon\_config) | Amazon EKS Managed CoreDNS Add-on config | `any` | `{}` | no |
|
||||
| <a name="input_addon_context"></a> [addon\_context](#input\_addon\_context) | Input configuration for the addon | <pre>object({<br> aws_caller_identity_account_id = string<br> aws_caller_identity_arn = string<br> aws_eks_cluster_endpoint = string<br> aws_partition_id = string<br> aws_region_name = string<br> eks_cluster_id = string<br> eks_oidc_issuer_url = string<br> eks_oidc_provider_arn = string<br> irsa_iam_role_path = string<br> tags = map(string)<br> })</pre> | n/a | yes |
|
||||
| <a name="input_enable_cert_manager"></a> [enable\_cert\_manager](#input\_enable\_cert\_manager) | Enable cert-manager, a requirement for ADOT Operator | `bool` | `true` | no |
|
||||
| <a name="input_helm_config"></a> [helm\_config](#input\_helm\_config) | Helm provider config for ADOT Operator AddOn | `any` | <pre>{<br> "version": "v1.8.2"<br>}</pre> | no |
|
||||
| <a name="input_kubernetes_version"></a> [kubernetes\_version](#input\_kubernetes\_version) | EKS Cluster version | `string` | n/a | yes |
|
||||
|
||||
## Outputs
|
||||
|
||||
No outputs.
|
||||
<!-- END OF PRE-COMMIT-TERRAFORM DOCS HOOK -->
|
||||
@@ -0,0 +1,6 @@
|
||||
locals {
|
||||
name = "adot"
|
||||
eks_addon_role_name = "eks:addon-manager"
|
||||
eks_addon_clusterrole_name = "eks:addon-manager-otel"
|
||||
addon_namespace = "opentelemetry-operator-system"
|
||||
}
|
||||
@@ -0,0 +1,276 @@
|
||||
module "cert_manager" {
|
||||
source = "github.com/aws-ia/terraform-aws-eks-blueprints/modules/kubernetes-addons/cert-manager"
|
||||
count = var.enable_cert_manager ? 1 : 0
|
||||
|
||||
helm_config = var.helm_config
|
||||
addon_context = var.addon_context
|
||||
}
|
||||
|
||||
resource "kubernetes_namespace_v1" "adot" {
|
||||
|
||||
metadata {
|
||||
# If using EKS addon, namespace must be "opentelemetry-operator-system"
|
||||
name = local.addon_namespace
|
||||
|
||||
labels = {
|
||||
# Prerequisite for EKS addon
|
||||
"control-plane" = "controller-manager"
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
data "aws_eks_addon_version" "this" {
|
||||
addon_name = local.name
|
||||
kubernetes_version = try(var.addon_config.kubernetes_version, var.kubernetes_version)
|
||||
most_recent = try(var.addon_config.most_recent, true)
|
||||
}
|
||||
|
||||
resource "aws_eks_addon" "adot" {
|
||||
cluster_name = var.addon_context.eks_cluster_id
|
||||
addon_name = local.name
|
||||
addon_version = try(var.addon_config.addon_version, data.aws_eks_addon_version.this.version)
|
||||
resolve_conflicts = try(var.addon_config.resolve_conflicts, "OVERWRITE")
|
||||
service_account_role_arn = try(var.addon_config.service_account_role_arn, null)
|
||||
preserve = try(var.addon_config.preserve, true)
|
||||
|
||||
tags = merge(
|
||||
var.addon_context.tags,
|
||||
try(var.addon_config.tags, {}),
|
||||
# implicit dependency with roles
|
||||
{
|
||||
RoleVersion = try(kubernetes_role_v1.adot.metadata[0].resource_version, ""),
|
||||
ClusterRoleVersion = try(kubernetes_cluster_role_v1.adot.metadata[0].resource_version, "")
|
||||
}
|
||||
)
|
||||
|
||||
depends_on = [module.cert_manager]
|
||||
}
|
||||
|
||||
resource "kubernetes_role_v1" "adot" {
|
||||
|
||||
metadata {
|
||||
name = local.eks_addon_role_name
|
||||
namespace = kubernetes_namespace_v1.adot.metadata[0].name
|
||||
}
|
||||
|
||||
rule {
|
||||
api_groups = [""]
|
||||
resources = ["serviceaccounts"]
|
||||
resource_names = ["opentelemetry-operator-controller-manager"]
|
||||
verbs = ["create", "delete", "get", "list", "patch", "update", "watch"]
|
||||
}
|
||||
rule {
|
||||
api_groups = ["rbac.authorization.k8s.io"]
|
||||
resources = ["roles"]
|
||||
resource_names = ["opentelemetry-operator-leader-election-role"]
|
||||
verbs = ["create", "delete", "get", "list", "patch", "update", "watch"]
|
||||
}
|
||||
rule {
|
||||
api_groups = ["rbac.authorization.k8s.io"]
|
||||
resources = ["rolebindings"]
|
||||
resource_names = ["opentelemetry-operator-leader-election-rolebinding"]
|
||||
verbs = ["create", "delete", "get", "list", "patch", "update", "watch"]
|
||||
}
|
||||
rule {
|
||||
api_groups = [""]
|
||||
resources = ["services"]
|
||||
resource_names = ["opentelemetry-operator-controller-manager-metrics-service", "opentelemetry-operator-webhook-service"]
|
||||
verbs = ["create", "delete", "get", "list", "patch", "update", "watch"]
|
||||
}
|
||||
rule {
|
||||
api_groups = ["apps"]
|
||||
resources = ["deployments"]
|
||||
resource_names = ["opentelemetry-operator-controller-manager"]
|
||||
verbs = ["create", "delete", "get", "list", "patch", "update", "watch"]
|
||||
}
|
||||
rule {
|
||||
api_groups = ["cert-manager.io"]
|
||||
resources = ["certificates", "issuers"]
|
||||
resource_names = ["opentelemetry-operator-serving-cert", "opentelemetry-operator-selfsigned-issuer"]
|
||||
verbs = ["create", "delete", "get", "list", "patch", "update", "watch"]
|
||||
}
|
||||
rule {
|
||||
api_groups = [""]
|
||||
resources = ["configmaps"]
|
||||
verbs = ["create", "delete", "get", "list", "patch", "update", "watch"]
|
||||
}
|
||||
rule {
|
||||
api_groups = [""]
|
||||
resources = ["configmaps/status"]
|
||||
verbs = ["get", "update", "patch"]
|
||||
}
|
||||
rule {
|
||||
api_groups = [""]
|
||||
resources = ["events"]
|
||||
verbs = ["create", "patch"]
|
||||
}
|
||||
rule {
|
||||
api_groups = [""]
|
||||
resources = ["pods"]
|
||||
verbs = ["list"]
|
||||
}
|
||||
}
|
||||
|
||||
resource "kubernetes_role_binding_v1" "adot" {
|
||||
|
||||
metadata {
|
||||
name = local.eks_addon_role_name
|
||||
namespace = kubernetes_namespace_v1.adot.metadata[0].name
|
||||
}
|
||||
|
||||
subject {
|
||||
kind = "User"
|
||||
name = local.eks_addon_role_name
|
||||
api_group = "rbac.authorization.k8s.io"
|
||||
}
|
||||
role_ref {
|
||||
api_group = "rbac.authorization.k8s.io"
|
||||
kind = "Role"
|
||||
name = local.eks_addon_role_name
|
||||
}
|
||||
}
|
||||
|
||||
resource "kubernetes_cluster_role_v1" "adot" {
|
||||
|
||||
metadata {
|
||||
name = local.eks_addon_clusterrole_name
|
||||
}
|
||||
|
||||
rule {
|
||||
api_groups = ["apiextensions.k8s.io"]
|
||||
resources = ["customresourcedefinitions"]
|
||||
resource_names = ["opentelemetrycollectors.opentelemetry.io", "instrumentations.opentelemetry.io"]
|
||||
verbs = ["create", "delete", "get", "list", "patch", "update", "watch"]
|
||||
}
|
||||
rule {
|
||||
api_groups = [""]
|
||||
resources = ["namespaces"]
|
||||
resource_names = [kubernetes_namespace_v1.adot.metadata[0].name]
|
||||
verbs = ["create", "delete", "get", "list", "patch", "update", "watch"]
|
||||
}
|
||||
rule {
|
||||
api_groups = ["rbac.authorization.k8s.io"]
|
||||
resources = ["clusterroles"]
|
||||
resource_names = ["opentelemetry-operator-manager-role", "opentelemetry-operator-metrics-reader", "opentelemetry-operator-proxy-role"]
|
||||
verbs = ["create", "delete", "get", "list", "patch", "update", "watch"]
|
||||
}
|
||||
rule {
|
||||
api_groups = ["rbac.authorization.k8s.io"]
|
||||
resources = ["clusterrolebindings"]
|
||||
resource_names = ["opentelemetry-operator-manager-rolebinding", "opentelemetry-operator-proxy-rolebinding"]
|
||||
verbs = ["create", "delete", "get", "list", "patch", "update", "watch"]
|
||||
}
|
||||
rule {
|
||||
api_groups = ["admissionregistration.k8s.io"]
|
||||
resources = ["mutatingwebhookconfigurations", "validatingwebhookconfigurations"]
|
||||
resource_names = ["opentelemetry-operator-mutating-webhook-configuration", "opentelemetry-operator-validating-webhook-configuration"]
|
||||
verbs = ["create", "delete", "get", "list", "patch", "update", "watch"]
|
||||
}
|
||||
rule {
|
||||
non_resource_urls = ["/metrics"]
|
||||
verbs = ["get"]
|
||||
}
|
||||
rule {
|
||||
api_groups = [""]
|
||||
resources = ["configmaps"]
|
||||
verbs = ["create", "delete", "get", "list", "patch", "update", "watch"]
|
||||
}
|
||||
rule {
|
||||
api_groups = [""]
|
||||
resources = ["events"]
|
||||
verbs = ["create", "patch"]
|
||||
}
|
||||
rule {
|
||||
api_groups = [""]
|
||||
resources = ["namespaces"]
|
||||
verbs = ["list", "watch"]
|
||||
}
|
||||
rule {
|
||||
api_groups = [""]
|
||||
resources = ["serviceaccounts"]
|
||||
verbs = ["create", "delete", "get", "list", "patch", "update", "watch"]
|
||||
}
|
||||
rule {
|
||||
api_groups = [""]
|
||||
resources = ["services"]
|
||||
verbs = ["create", "delete", "get", "list", "patch", "update", "watch"]
|
||||
}
|
||||
rule {
|
||||
api_groups = ["apps"]
|
||||
resources = ["daemonsets"]
|
||||
verbs = ["create", "delete", "get", "list", "patch", "update", "watch"]
|
||||
}
|
||||
rule {
|
||||
api_groups = ["apps"]
|
||||
resources = ["deployments"]
|
||||
verbs = ["create", "delete", "get", "list", "patch", "update", "watch"]
|
||||
}
|
||||
rule {
|
||||
api_groups = ["apps"]
|
||||
resources = ["replicasets"]
|
||||
verbs = ["create", "delete", "get", "list", "patch", "update", "watch"]
|
||||
}
|
||||
rule {
|
||||
api_groups = ["apps"]
|
||||
resources = ["statefulsets"]
|
||||
verbs = ["create", "delete", "get", "list", "patch", "update", "watch"]
|
||||
}
|
||||
rule {
|
||||
api_groups = ["autoscaling"]
|
||||
resources = ["horizontalpodautoscalers"]
|
||||
verbs = ["create", "delete", "get", "list", "patch", "update", "watch"]
|
||||
}
|
||||
rule {
|
||||
api_groups = ["coordination.k8s.io"]
|
||||
resources = ["leases"]
|
||||
verbs = ["create", "get", "list", "update"]
|
||||
}
|
||||
rule {
|
||||
api_groups = ["opentelemetry.io"]
|
||||
resources = ["opentelemetrycollectors"]
|
||||
verbs = ["create", "delete", "get", "list", "patch", "update", "watch"]
|
||||
}
|
||||
rule {
|
||||
api_groups = ["opentelemetry.io"]
|
||||
resources = ["opentelemetrycollectors/finalizers"]
|
||||
verbs = ["get", "patch", "update"]
|
||||
}
|
||||
rule {
|
||||
api_groups = ["opentelemetry.io"]
|
||||
resources = ["opentelemetrycollectors/status"]
|
||||
verbs = ["get", "patch", "update"]
|
||||
}
|
||||
rule {
|
||||
api_groups = ["opentelemetry.io"]
|
||||
resources = ["instrumentations"]
|
||||
verbs = ["get", "list", "patch", "update", "watch"]
|
||||
}
|
||||
rule {
|
||||
api_groups = ["authentication.k8s.io"]
|
||||
resources = ["tokenreviews"]
|
||||
verbs = ["create"]
|
||||
}
|
||||
rule {
|
||||
api_groups = ["authorization.k8s.io"]
|
||||
resources = ["subjectaccessreviews"]
|
||||
verbs = ["create"]
|
||||
}
|
||||
}
|
||||
|
||||
resource "kubernetes_cluster_role_binding_v1" "adot" {
|
||||
|
||||
metadata {
|
||||
name = local.eks_addon_clusterrole_name
|
||||
}
|
||||
subject {
|
||||
kind = "User"
|
||||
name = local.eks_addon_role_name
|
||||
api_group = "rbac.authorization.k8s.io"
|
||||
}
|
||||
role_ref {
|
||||
api_group = "rbac.authorization.k8s.io"
|
||||
kind = "ClusterRole"
|
||||
name = local.eks_addon_clusterrole_name
|
||||
}
|
||||
|
||||
}
|
||||
@@ -0,0 +1,38 @@
|
||||
variable "helm_config" {
|
||||
description = "Helm provider config for ADOT Operator AddOn"
|
||||
type = any
|
||||
default = { version = "v1.8.2" }
|
||||
}
|
||||
|
||||
variable "addon_context" {
|
||||
description = "Input configuration for the addon"
|
||||
type = object({
|
||||
aws_caller_identity_account_id = string
|
||||
aws_caller_identity_arn = string
|
||||
aws_eks_cluster_endpoint = string
|
||||
aws_partition_id = string
|
||||
aws_region_name = string
|
||||
eks_cluster_id = string
|
||||
eks_oidc_issuer_url = string
|
||||
eks_oidc_provider_arn = string
|
||||
irsa_iam_role_path = string
|
||||
tags = map(string)
|
||||
})
|
||||
}
|
||||
|
||||
variable "enable_cert_manager" {
|
||||
description = "Enable cert-manager, a requirement for ADOT Operator"
|
||||
type = bool
|
||||
default = true
|
||||
}
|
||||
|
||||
variable "kubernetes_version" {
|
||||
description = "EKS Cluster version"
|
||||
type = string
|
||||
}
|
||||
|
||||
variable "addon_config" {
|
||||
description = "Amazon EKS Managed CoreDNS Add-on config"
|
||||
type = any
|
||||
default = {}
|
||||
}
|
||||
@@ -0,0 +1,14 @@
|
||||
terraform {
|
||||
required_version = ">= 1.0.0"
|
||||
|
||||
required_providers {
|
||||
aws = {
|
||||
source = "hashicorp/aws"
|
||||
version = ">= 3.72"
|
||||
}
|
||||
kubernetes = {
|
||||
source = "hashicorp/kubernetes"
|
||||
version = ">= 2.10"
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,171 @@
|
||||
|
||||
#---------------------------------------------------------------
|
||||
# Observability Resources
|
||||
#---------------------------------------------------------------
|
||||
|
||||
module "managed_grafana" {
|
||||
source = "terraform-aws-modules/managed-service-grafana/aws"
|
||||
version = "~> 1.3"
|
||||
|
||||
# Workspace
|
||||
name = local.name
|
||||
stack_set_name = local.name
|
||||
data_sources = ["PROMETHEUS"]
|
||||
associate_license = false
|
||||
|
||||
# # Role associations
|
||||
# Pending https://github.com/hashicorp/terraform-provider-aws/issues/24166
|
||||
# role_associations = {
|
||||
# "ADMIN" = {
|
||||
# "group_ids" = []
|
||||
# "user_ids" = []
|
||||
# }
|
||||
# "EDITOR" = {
|
||||
# "group_ids" = []
|
||||
# "user_ids" = []
|
||||
# }
|
||||
# }
|
||||
|
||||
tags = local.tags
|
||||
}
|
||||
|
||||
resource "grafana_data_source" "prometheus" {
|
||||
type = "prometheus"
|
||||
name = "amp"
|
||||
is_default = true
|
||||
url = module.managed_prometheus.workspace_prometheus_endpoint
|
||||
|
||||
json_data {
|
||||
http_method = "GET"
|
||||
sigv4_auth = true
|
||||
sigv4_auth_type = "workspace-iam-role"
|
||||
sigv4_region = local.region
|
||||
}
|
||||
}
|
||||
|
||||
resource "grafana_folder" "this" {
|
||||
title = "Observability"
|
||||
}
|
||||
|
||||
resource "grafana_dashboard" "this" {
|
||||
folder = grafana_folder.this.id
|
||||
config_json = file("${path.module}/dashboards/default.json")
|
||||
}
|
||||
|
||||
module "managed_prometheus" {
|
||||
source = "terraform-aws-modules/managed-service-prometheus/aws"
|
||||
version = "~> 2.1"
|
||||
|
||||
workspace_alias = local.name
|
||||
|
||||
alert_manager_definition = <<-EOT
|
||||
alertmanager_config: |
|
||||
route:
|
||||
receiver: 'default'
|
||||
receivers:
|
||||
- name: 'default'
|
||||
EOT
|
||||
|
||||
rule_group_namespaces = {
|
||||
haproxy = {
|
||||
name = "haproxy_rules"
|
||||
data = <<-EOT
|
||||
groups:
|
||||
- name: obsa-haproxy-down-alert
|
||||
rules:
|
||||
- alert: HA_proxy_down
|
||||
expr: haproxy_up == 0
|
||||
for: 0m
|
||||
labels:
|
||||
severity: critical
|
||||
annotations:
|
||||
summary: HAProxy down (instance {{ $labels.instance }})
|
||||
description: "HAProxy down\n VALUE = {{ $value }}\n LABELS = {{ $labels }}"
|
||||
- name: obsa-haproxy-http4xx-error-alert
|
||||
rules:
|
||||
- alert: Ha_proxy_High_Http4xx_ErrorRate_Backend
|
||||
expr: sum by (backend) (rate(haproxy_server_http_responses_total{code="4xx"}[1m])) / sum by (backend) (rate(haproxy_server_http_responses_total[1m]) * 100) > 5
|
||||
for: 1m
|
||||
labels:
|
||||
severity: critical
|
||||
annotations:
|
||||
summary: HAProxy high HTTP 4xx error rate backend (instance {{ $labels.instance }})
|
||||
description: "Too many HTTP requests with status 4xx (> 5%) on backend {{ $labels.fqdn }}/{{ $labels.backend }}\n VALUE = {{ $value }}\n LABELS = {{ $labels }}"
|
||||
- name: obsa-haproxy-http5xx-error-alert
|
||||
rules:
|
||||
- alert: Ha_proxy_High_Http5xx_ErrorRate_Backend
|
||||
expr: sum by (backend) (rate(haproxy_server_http_responses_total{code="5xx"}[1m])) / sum by (backend) (rate(haproxy_server_http_responses_total[1m]) * 100) > 5
|
||||
for: 1m
|
||||
labels:
|
||||
severity: critical
|
||||
annotations:
|
||||
summary: HAProxy high HTTP 5xx error rate backend (instance {{ $labels.instance }})
|
||||
description: "Too many HTTP requests with status 5xx (> 5%) on backend {{ $labels.fqdn }}/{{ $labels.backend }}\n VALUE = {{ $value }}\n LABELS = {{ $labels }}"
|
||||
- name: obsa-haproxy-Http4xx-ErrorRate-Server-alert
|
||||
rules:
|
||||
- alert: Ha_proxy_High_Http4xx_ErrorRate_Server
|
||||
expr: sum by (server) (rate(haproxy_server_http_responses_total{code="4xx"}[1m])) / sum by (server) (rate(haproxy_server_http_responses_total[1m]) * 100) > 5
|
||||
for: 1m
|
||||
labels:
|
||||
severity: critical
|
||||
annotations:
|
||||
summary: HAProxy high HTTP 4xx error rate server (instance {{ $labels.instance }})
|
||||
description: "Too many HTTP requests with status 4xx (> 5%) on server {{ $labels.server }}\n VALUE = {{ $value }}\n LABELS = {{ $labels }}"
|
||||
- name: obsa-haproxy-Http5xx-ErrorRate-Server-alert
|
||||
rules:
|
||||
- alert: Ha_proxy_High_Http5xx_ErrorRate_Server
|
||||
expr: sum by (server) (rate(haproxy_server_http_responses_total{code="5xx"}[1m])) / sum by (server) (rate(haproxy_server_http_responses_total[1m]) * 100) > 5
|
||||
for: 1m
|
||||
labels:
|
||||
severity: critical
|
||||
annotations:
|
||||
summary: HAProxy high HTTP 5xx error rate server (instance {{ $labels.instance }})
|
||||
description: "Too many HTTP requests with status 5xx (> 5%) on server {{ $labels.server }}\n VALUE = {{ $value }}\n LABELS = {{ $labels }}"
|
||||
EOT
|
||||
}
|
||||
}
|
||||
|
||||
tags = local.tags
|
||||
}
|
||||
|
||||
#---------------------------------------------------------------
|
||||
# Sample Application
|
||||
#---------------------------------------------------------------
|
||||
|
||||
# https://github.com/haproxy-ingress/charts/tree/master/haproxy-ingress
|
||||
resource "helm_release" "haproxy_ingress" {
|
||||
namespace = "haproxy-ingress"
|
||||
create_namespace = true
|
||||
|
||||
name = "haproxy-ingress"
|
||||
repository = "https://haproxy-ingress.github.io/charts"
|
||||
chart = "haproxy-ingress"
|
||||
version = "0.13.7"
|
||||
|
||||
set {
|
||||
name = "defaultBackend.enabled"
|
||||
value = true
|
||||
}
|
||||
|
||||
set {
|
||||
name = "controller.stats.enabled"
|
||||
value = true
|
||||
}
|
||||
|
||||
set {
|
||||
name = "controller.metrics.enabled"
|
||||
value = true
|
||||
}
|
||||
|
||||
set {
|
||||
name = "controller.metrics.service.annotations.prometheus\\.io/port"
|
||||
value = 9101
|
||||
type = "string"
|
||||
}
|
||||
|
||||
set {
|
||||
name = "controller.metrics.service.annotations.prometheus\\.io/scrape"
|
||||
value = true
|
||||
type = "string"
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,86 @@
|
||||
# Infrastructure monitoring
|
||||
|
||||
This module provides EKS cluster monitoring with the following resources:
|
||||
|
||||
- AWS Distro For OpenTelemetry Operator and Collector
|
||||
- AWS Managed Grafana Dashboard and data source
|
||||
- Alerts and recording rules with AWS Managed Service for Prometheus
|
||||
|
||||
This module is inspired from the open source [kube-prometheus-stack](https://github.com/prometheus-community/helm-charts/tree/main/charts/kube-prometheus-stack)
|
||||
|
||||
<!-- BEGINNING OF PRE-COMMIT-TERRAFORM DOCS HOOK -->
|
||||
## Requirements
|
||||
|
||||
| Name | Version |
|
||||
|------|---------|
|
||||
| <a name="requirement_terraform"></a> [terraform](#requirement\_terraform) | >= 1.0.0 |
|
||||
| <a name="requirement_aws"></a> [aws](#requirement\_aws) | >= 4.0.0 |
|
||||
| <a name="requirement_grafana"></a> [grafana](#requirement\_grafana) | >= 1.25.0 |
|
||||
| <a name="requirement_helm"></a> [helm](#requirement\_helm) | >= 2.4.1 |
|
||||
| <a name="requirement_kubectl"></a> [kubectl](#requirement\_kubectl) | >= 1.14 |
|
||||
| <a name="requirement_kubernetes"></a> [kubernetes](#requirement\_kubernetes) | >= 2.10 |
|
||||
|
||||
## Providers
|
||||
|
||||
| Name | Version |
|
||||
|------|---------|
|
||||
| <a name="provider_aws"></a> [aws](#provider\_aws) | >= 4.0.0 |
|
||||
| <a name="provider_grafana"></a> [grafana](#provider\_grafana) | >= 1.25.0 |
|
||||
| <a name="provider_helm"></a> [helm](#provider\_helm) | >= 2.4.1 |
|
||||
|
||||
## Modules
|
||||
|
||||
| Name | Source | Version |
|
||||
|------|--------|---------|
|
||||
| <a name="module_helm_addon"></a> [helm\_addon](#module\_helm\_addon) | github.com/aws-ia/terraform-aws-eks-blueprints/modules/kubernetes-addons/helm-addon | n/a |
|
||||
|
||||
## Resources
|
||||
|
||||
| Name | Type |
|
||||
|------|------|
|
||||
| [aws_prometheus_rule_group_namespace.alerting_rules](https://registry.terraform.io/providers/hashicorp/aws/latest/docs/resources/prometheus_rule_group_namespace) | resource |
|
||||
| [aws_prometheus_rule_group_namespace.recording_rules](https://registry.terraform.io/providers/hashicorp/aws/latest/docs/resources/prometheus_rule_group_namespace) | resource |
|
||||
| [grafana_dashboard.cluster](https://registry.terraform.io/providers/grafana/grafana/latest/docs/resources/dashboard) | resource |
|
||||
| [grafana_dashboard.clusternw](https://registry.terraform.io/providers/grafana/grafana/latest/docs/resources/dashboard) | resource |
|
||||
| [grafana_dashboard.kubelet](https://registry.terraform.io/providers/grafana/grafana/latest/docs/resources/dashboard) | resource |
|
||||
| [grafana_dashboard.nodes](https://registry.terraform.io/providers/grafana/grafana/latest/docs/resources/dashboard) | resource |
|
||||
| [grafana_dashboard.nsnw](https://registry.terraform.io/providers/grafana/grafana/latest/docs/resources/dashboard) | resource |
|
||||
| [grafana_dashboard.nsnwworkload](https://registry.terraform.io/providers/grafana/grafana/latest/docs/resources/dashboard) | resource |
|
||||
| [grafana_dashboard.nspods](https://registry.terraform.io/providers/grafana/grafana/latest/docs/resources/dashboard) | resource |
|
||||
| [grafana_dashboard.nsworkload](https://registry.terraform.io/providers/grafana/grafana/latest/docs/resources/dashboard) | resource |
|
||||
| [grafana_dashboard.nwworload](https://registry.terraform.io/providers/grafana/grafana/latest/docs/resources/dashboard) | resource |
|
||||
| [grafana_dashboard.podnetwork](https://registry.terraform.io/providers/grafana/grafana/latest/docs/resources/dashboard) | resource |
|
||||
| [grafana_dashboard.pods](https://registry.terraform.io/providers/grafana/grafana/latest/docs/resources/dashboard) | resource |
|
||||
| [grafana_dashboard.workloads](https://registry.terraform.io/providers/grafana/grafana/latest/docs/resources/dashboard) | resource |
|
||||
| [helm_release.kube_state_metrics](https://registry.terraform.io/providers/hashicorp/helm/latest/docs/resources/release) | resource |
|
||||
| [helm_release.prometheus_node_exporter](https://registry.terraform.io/providers/hashicorp/helm/latest/docs/resources/release) | resource |
|
||||
| [aws_caller_identity.current](https://registry.terraform.io/providers/hashicorp/aws/latest/docs/data-sources/caller_identity) | data source |
|
||||
| [aws_eks_cluster.eks_cluster](https://registry.terraform.io/providers/hashicorp/aws/latest/docs/data-sources/eks_cluster) | data source |
|
||||
| [aws_partition.current](https://registry.terraform.io/providers/hashicorp/aws/latest/docs/data-sources/partition) | data source |
|
||||
| [aws_region.current](https://registry.terraform.io/providers/hashicorp/aws/latest/docs/data-sources/region) | data source |
|
||||
|
||||
## Inputs
|
||||
|
||||
| Name | Description | Type | Default | Required |
|
||||
|------|-------------|------|---------|:--------:|
|
||||
| <a name="input_dashboards_folder_id"></a> [dashboards\_folder\_id](#input\_dashboards\_folder\_id) | Grafana folder ID for automatic dashboards | `string` | n/a | yes |
|
||||
| <a name="input_eks_cluster_id"></a> [eks\_cluster\_id](#input\_eks\_cluster\_id) | EKS Cluster Id | `string` | n/a | yes |
|
||||
| <a name="input_enable_alerting_rules"></a> [enable\_alerting\_rules](#input\_enable\_alerting\_rules) | Enables or disables Managed Prometheus alerting rules | `bool` | `true` | no |
|
||||
| <a name="input_enable_dashboards"></a> [enable\_dashboards](#input\_enable\_dashboards) | Enables or disables curated dashboards | `bool` | `true` | no |
|
||||
| <a name="input_enable_kube_state_metrics"></a> [enable\_kube\_state\_metrics](#input\_enable\_kube\_state\_metrics) | Enables or disables Kube State metrics exporter. Disabling this might affect some data in the dashboards | `bool` | `true` | no |
|
||||
| <a name="input_enable_node_exporter"></a> [enable\_node\_exporter](#input\_enable\_node\_exporter) | Enables or disables Node exporter. Disabling this might affect some data in the dashboards | `bool` | `true` | no |
|
||||
| <a name="input_enable_recording_rules"></a> [enable\_recording\_rules](#input\_enable\_recording\_rules) | Enables or disables Managed Prometheus recording rules. Disabling this might affect some data in the dashboards | `bool` | `true` | no |
|
||||
| <a name="input_helm_config"></a> [helm\_config](#input\_helm\_config) | Helm Config for Prometheus | `any` | `{}` | no |
|
||||
| <a name="input_irsa_iam_permissions_boundary"></a> [irsa\_iam\_permissions\_boundary](#input\_irsa\_iam\_permissions\_boundary) | IAM permissions boundary for IRSA roles | `string` | `""` | no |
|
||||
| <a name="input_irsa_iam_role_path"></a> [irsa\_iam\_role\_path](#input\_irsa\_iam\_role\_path) | IAM role path for IRSA roles | `string` | `"/"` | no |
|
||||
| <a name="input_ksm_config"></a> [ksm\_config](#input\_ksm\_config) | Kube State metrics configuration | <pre>object({<br> create_namespace = bool<br> k8s_namespace = string<br> helm_chart_name = string<br> helm_chart_version = string<br> helm_release_name = string<br> helm_repo_url = string<br> helm_settings = map(string)<br> helm_values = map(any)<br> })</pre> | <pre>{<br> "create_namespace": true,<br> "helm_chart_name": "kube-state-metrics",<br> "helm_chart_version": "4.16.0",<br> "helm_release_name": "kube-state-metrics",<br> "helm_repo_url": "https://prometheus-community.github.io/helm-charts",<br> "helm_settings": {},<br> "helm_values": {},<br> "k8s_namespace": "kube-system"<br>}</pre> | no |
|
||||
| <a name="input_managed_prometheus_workspace_endpoint"></a> [managed\_prometheus\_workspace\_endpoint](#input\_managed\_prometheus\_workspace\_endpoint) | Amazon Managed Prometheus Workspace Endpoint | `string` | `null` | no |
|
||||
| <a name="input_managed_prometheus_workspace_id"></a> [managed\_prometheus\_workspace\_id](#input\_managed\_prometheus\_workspace\_id) | Amazon Managed Prometheus Workspace ID | `string` | `null` | no |
|
||||
| <a name="input_managed_prometheus_workspace_region"></a> [managed\_prometheus\_workspace\_region](#input\_managed\_prometheus\_workspace\_region) | Amazon Managed Prometheus Workspace's Region | `string` | `null` | no |
|
||||
| <a name="input_ne_config"></a> [ne\_config](#input\_ne\_config) | Node exporter configuration | <pre>object({<br> create_namespace = bool<br> k8s_namespace = string<br> helm_chart_name = string<br> helm_chart_version = string<br> helm_release_name = string<br> helm_repo_url = string<br> helm_settings = map(string)<br> helm_values = map(any)<br> })</pre> | <pre>{<br> "create_namespace": true,<br> "helm_chart_name": "prometheus-node-exporter",<br> "helm_chart_version": "2.0.3",<br> "helm_release_name": "prometheus-node-exporter",<br> "helm_repo_url": "https://prometheus-community.github.io/helm-charts",<br> "helm_settings": {},<br> "helm_values": {},<br> "k8s_namespace": "prometheus-node-exporter"<br>}</pre> | no |
|
||||
| <a name="input_tags"></a> [tags](#input\_tags) | Additional tags (e.g. `map('BusinessUnit`,`XYZ`) | `map(string)` | `{}` | no |
|
||||
|
||||
## Outputs
|
||||
|
||||
No outputs.
|
||||
<!-- END OF PRE-COMMIT-TERRAFORM DOCS HOOK -->
|
||||
@@ -0,0 +1,755 @@
|
||||
################################################################################################################################################
|
||||
# Alerting rules ###############################################################################################################################
|
||||
################################################################################################################################################
|
||||
|
||||
resource "aws_prometheus_rule_group_namespace" "alerting_rules" {
|
||||
count = var.enable_alerting_rules ? 1 : 0
|
||||
|
||||
name = "accelerator-infra-alerting"
|
||||
workspace_id = var.managed_prometheus_workspace_id
|
||||
data = <<EOF
|
||||
groups:
|
||||
- name: infra-alerts-01
|
||||
rules:
|
||||
- alert: NodeNetworkInterfaceFlapping
|
||||
expr: changes(node_network_up{device!~"veth.+",job="node-exporter"}[2m]) > 2
|
||||
for: 2m
|
||||
labels:
|
||||
severity: warning
|
||||
annotations:
|
||||
description: Network interface "{{ $labels.device }}" changing its up status often on node-exporter {{ $labels.namespace }}/{{ $labels.pod }}
|
||||
summary: Network interface is often changing its status
|
||||
- alert: NodeFilesystemSpaceFillingUp
|
||||
expr: (node_filesystem_avail_bytes{fstype!="",job="node-exporter"} / node_filesystem_size_bytes{fstype!="",job="node-exporter"} * 100 < 15 and predict_linear(node_filesystem_avail_bytes{fstype!="",job="node-exporter"}[6h], 24 * 60 * 60) < 0 and node_filesystem_readonly{fstype!="",job="node-exporter"} == 0)
|
||||
for: 1h
|
||||
labels:
|
||||
severity: warning
|
||||
annotations:
|
||||
description: Filesystem on {{ $labels.device }} at {{ $labels.instance }} has only {{ printf "%.2f" $value }}% available space left and is filling up.
|
||||
summary: Filesystem is predicted to run out of space within the next 24 hours.
|
||||
- alert: NodeFilesystemSpaceFillingUp
|
||||
expr: (node_filesystem_avail_bytes{fstype!="",job="node-exporter"} / node_filesystem_size_bytes{fstype!="",job="node-exporter"} * 100 < 10 and predict_linear(node_filesystem_avail_bytes{fstype!="",job="node-exporter"}[6h], 4 * 60 * 60) < 0 and node_filesystem_readonly{fstype!="",job="node-exporter"} == 0)
|
||||
for: 1h
|
||||
labels:
|
||||
severity: critical
|
||||
annotations:
|
||||
description: Filesystem on {{ $labels.device }} at {{ $labels.instance }} has only {{ printf "%.2f" $value }}% available space left and is filling up fast.
|
||||
summary: Filesystem is predicted to run out of space within the next 4 hours.
|
||||
- alert: NodeFilesystemAlmostOutOfSpace
|
||||
expr: (node_filesystem_avail_bytes{fstype!="",job="node-exporter"} / node_filesystem_size_bytes{fstype!="",job="node-exporter"} * 100 < 3 and node_filesystem_readonly{fstype!="",job="node-exporter"} == 0)
|
||||
for: 30m
|
||||
labels:
|
||||
severity: warning
|
||||
annotations:
|
||||
description: Filesystem on {{ $labels.device }} at {{ $labels.instance }} has only {{ printf "%.2f" $value }}% available space left.
|
||||
summary: Filesystem has less than 3% space left.
|
||||
- alert: NodeFilesystemAlmostOutOfSpace
|
||||
expr: (node_filesystem_avail_bytes{fstype!="",job="node-exporter"} / node_filesystem_size_bytes{fstype!="",job="node-exporter"} * 100 < 5 and node_filesystem_readonly{fstype!="",job="node-exporter"} == 0)
|
||||
for: 30m
|
||||
labels:
|
||||
severity: critical
|
||||
annotations:
|
||||
description: Filesystem on {{ $labels.device }} at {{ $labels.instance }} has only {{ printf "%.2f" $value }}% available space left.
|
||||
summary: Filesystem has less than 5% space left.
|
||||
- alert: NodeFilesystemFilesFillingUp
|
||||
expr: (node_filesystem_files_free{fstype!="",job="node-exporter"} / node_filesystem_files{fstype!="",job="node-exporter"} * 100 < 40 and predict_linear(node_filesystem_files_free{fstype!="",job="node-exporter"}[6h], 24 * 60 * 60) < 0 and node_filesystem_readonly{fstype!="",job="node-exporter"} == 0)
|
||||
for: 1h
|
||||
labels:
|
||||
severity: warning
|
||||
annotations:
|
||||
description: Filesystem on {{ $labels.device }} at {{ $labels.instance }} has only {{ printf "%.2f" $value }}% available inodes left and is filling up.
|
||||
summary: Filesystem is predicted to run out of inodes within the next 24 hours.
|
||||
- alert: NodeFilesystemFilesFillingUp
|
||||
expr: (node_filesystem_files_free{fstype!="",job="node-exporter"} / node_filesystem_files{fstype!="",job="node-exporter"} * 100 < 20 and predict_linear(node_filesystem_files_free{fstype!="",job="node-exporter"}[6h], 4 * 60 * 60) < 0 and node_filesystem_readonly{fstype!="",job="node-exporter"} == 0)
|
||||
for: 1h
|
||||
labels:
|
||||
severity: critical
|
||||
annotations:
|
||||
description: Filesystem on {{ $labels.device }} at {{ $labels.instance }} has only {{ printf "%.2f" $value }}% available inodes left and is filling up fast.
|
||||
summary: Filesystem is predicted to run out of inodes within the next 4 hours.
|
||||
- alert: NodeFilesystemAlmostOutOfFiles
|
||||
expr: (node_filesystem_files_free{fstype!="",job="node-exporter"} / node_filesystem_files{fstype!="",job="node-exporter"} * 100 < 5 and node_filesystem_readonly{fstype!="",job="node-exporter"} == 0)
|
||||
for: 1h
|
||||
labels:
|
||||
severity: warning
|
||||
annotations:
|
||||
description: Filesystem on {{ $labels.device }} at {{ $labels.instance }} has only {{ printf "%.2f" $value }}% available inodes left.
|
||||
summary: Filesystem has less than 5% inodes left.
|
||||
- alert: NodeFilesystemAlmostOutOfFiles
|
||||
expr: (node_filesystem_files_free{fstype!="",job="node-exporter"} / node_filesystem_files{fstype!="",job="node-exporter"} * 100 < 3 and node_filesystem_readonly{fstype!="",job="node-exporter"} == 0)
|
||||
for: 1h
|
||||
labels:
|
||||
severity: critical
|
||||
annotations:
|
||||
description: Filesystem on {{ $labels.device }} at {{ $labels.instance }} has only {{ printf "%.2f" $value }}% available inodes left.
|
||||
summary: Filesystem has less than 3% inodes left.
|
||||
- alert: NodeNetworkReceiveErrs
|
||||
expr: rate(node_network_receive_errs_total[2m]) / rate(node_network_receive_packets_total[2m]) > 0.01
|
||||
for: 1h
|
||||
labels:
|
||||
severity: warning
|
||||
annotations:
|
||||
description: The {{ $labels.instance }} interface {{ $labels.device }} has encountered {{ printf "%.0f" $value }} receive errors in the last two minutes.
|
||||
summary: Network interface is reporting many receive errors.
|
||||
- alert: NodeNetworkTransmitErrs
|
||||
expr: rate(node_network_transmit_errs_total[2m]) / rate(node_network_transmit_packets_total[2m]) > 0.01
|
||||
for: 1h
|
||||
labels:
|
||||
severity: warning
|
||||
annotations:
|
||||
description: The {{ $labels.instance }} interface {{ $labels.device }} has encountered {{ printf "%.0f" $value }} transmit errors in the last two minutes.
|
||||
summary: Network interface is reporting many transmit errors.
|
||||
- alert: NodeHighNumberConntrackEntriesUsed
|
||||
expr: (node_nf_conntrack_entries / node_nf_conntrack_entries_limit) > 0.75
|
||||
labels:
|
||||
severity: warning
|
||||
annotations:
|
||||
description: The {{ $value | humanizePercentage }} of conntrack entries are used.
|
||||
summary: Number of conntrack are getting close to the limit.
|
||||
- alert: NodeTextFileCollectorScrapeError
|
||||
expr: node_textfile_scrape_error{job="node-exporter"} == 1
|
||||
labels:
|
||||
severity: warning
|
||||
annotations:
|
||||
description: Node Exporter text file collector failed to scrape.
|
||||
summary: Node Exporter text file collector failed to scrape.
|
||||
- alert: NodeClockSkewDetected
|
||||
expr: (node_timex_offset_seconds > 0.05 and deriv(node_timex_offset_seconds[5m]) >= 0) or (node_timex_offset_seconds < -0.05 and deriv(node_timex_offset_seconds[5m]) <= 0)
|
||||
for: 10m
|
||||
labels:
|
||||
severity: warning
|
||||
annotations:
|
||||
description: Clock on {{ $labels.instance }} is out of sync by more than 300s. Ensure NTP is configured correctly on this host.
|
||||
summary: Clock skew detected.
|
||||
- alert: NodeClockNotSynchronising
|
||||
expr: min_over_time(node_timex_sync_status[5m]) == 0 and node_timex_maxerror_seconds >= 16
|
||||
for: 10m
|
||||
labels:
|
||||
severity: warning
|
||||
annotations:
|
||||
description: Clock on {{ $labels.instance }} is not synchronising. Ensure NTP is configured on this host.
|
||||
summary: Clock not synchronising.
|
||||
- alert: NodeRAIDDegraded
|
||||
expr: node_md_disks_required - ignoring(state) (node_md_disks{state="active"}) > 0
|
||||
for: 15m
|
||||
labels:
|
||||
severity: critical
|
||||
annotations:
|
||||
description: RAID array '{{ $labels.device }}' on {{ $labels.instance }} is in degraded state due to one or more disks failures. Number of spare drives is insufficient to fix issue automatically.
|
||||
summary: RAID Array is degraded
|
||||
- alert: NodeRAIDDiskFailure
|
||||
expr: node_md_disks{state="failed"} > 0
|
||||
labels:
|
||||
severity: warning
|
||||
annotations:
|
||||
description: At least one device in RAID array on {{ $labels.instance }} failed. Array '{{ $labels.device }}' needs attention and possibly a disk swap.
|
||||
summary: Failed device in RAID array
|
||||
- alert: NodeFileDescriptorLimit
|
||||
expr: (node_filefd_allocated{job="node-exporter"} * 100 / node_filefd_maximum{job="node-exporter"} > 70)
|
||||
for: 15m
|
||||
labels:
|
||||
severity: warning
|
||||
annotations:
|
||||
description: File descriptors limit at {{ $labels.instance }} is currently at {{ printf "%.2f" $value }}%.
|
||||
summary: Kernel is predicted to exhaust file descriptors limit soon.
|
||||
- alert: NodeFileDescriptorLimit
|
||||
expr: (node_filefd_allocated{job="node-exporter"} * 100 / node_filefd_maximum{job="node-exporter"} > 90)
|
||||
for: 15m
|
||||
labels:
|
||||
severity: critical
|
||||
annotations:
|
||||
description: File descriptors limit at {{ $labels.instance }} is currently at {{ printf "%.2f" $value }}%.
|
||||
summary: Kernel is predicted to exhaust file descriptors limit soon.
|
||||
- alert: KubeSchedulerDown
|
||||
expr: absent(up{job="kube-scheduler"} == 1)
|
||||
for: 15m
|
||||
labels:
|
||||
severity: critical
|
||||
annotations:
|
||||
description: KubeScheduler has disappeared from Prometheus target discovery.
|
||||
summary: Target disappeared from Prometheus target discovery.
|
||||
- name: infra-alerts-02
|
||||
rules:
|
||||
- alert: KubeNodeNotReady
|
||||
expr: kube_node_status_condition{condition="Ready",job="kube-state-metrics",status="true"} == 0
|
||||
for: 15m
|
||||
labels:
|
||||
severity: warning
|
||||
annotations:
|
||||
description: The {{ $labels.node }} has been unready for more than 15 minutes.
|
||||
summary: Node is not ready.
|
||||
- alert: KubeNodeUnreachable
|
||||
expr: (kube_node_spec_taint{effect="NoSchedule",job="kube-state-metrics",key="node.kubernetes.io/unreachable"} unless ignoring(key, value) kube_node_spec_taint{job="kube-state-metrics",key=~"ToBeDeletedByClusterAutoscaler|cloud.google.com/impending-node-termination|aws-node-termination-handler/spot-itn"}) == 1
|
||||
for: 15m
|
||||
labels:
|
||||
severity: warning
|
||||
annotations:
|
||||
description: The {{ $labels.node }} is unreachable and some workloads may be rescheduled.
|
||||
summary: Node is unreachable.
|
||||
- alert: KubeletTooManyPods
|
||||
expr: count by(cluster, node) ((kube_pod_status_phase{job="kube-state-metrics",phase="Running"} == 1) * on(instance, pod, namespace, cluster) group_left(node) topk by(instance, pod, namespace, cluster) (1, kube_pod_info{job="kube-state-metrics"})) / max by(cluster, node) (kube_node_status_capacity{job="kube-state-metrics",resource="pods"} != 1) > 0.95
|
||||
for: 15m
|
||||
labels:
|
||||
severity: info
|
||||
annotations:
|
||||
description: Kubelet '{{ $labels.node }}' is running at {{ $value | humanizePercentage }} of its Pod capacity.
|
||||
summary: Kubelet is running at capacity.
|
||||
- alert: KubeNodeReadinessFlapping
|
||||
expr: sum by(cluster, node) (changes(kube_node_status_condition{condition="Ready",status="true"}[15m])) > 2
|
||||
for: 15m
|
||||
labels:
|
||||
severity: warning
|
||||
annotations:
|
||||
description: The readiness status of node {{ $labels.node }} has changed {{ $value }} times in the last 15 minutes.
|
||||
summary: Node readiness status is flapping.
|
||||
- alert: KubeletPlegDurationHigh
|
||||
expr: node_quantile:kubelet_pleg_relist_duration_seconds:histogram_quantile{quantile="0.99"} >= 10
|
||||
for: 5m
|
||||
labels:
|
||||
severity: warning
|
||||
annotations:
|
||||
description: The Kubelet Pod Lifecycle Event Generator has a 99th percentile duration of {{ $value }} seconds on node {{ $labels.node }}.
|
||||
summary: Kubelet Pod Lifecycle Event Generator is taking too long to relist.
|
||||
- alert: KubeletPodStartUpLatencyHigh
|
||||
expr: histogram_quantile(0.99, sum by(cluster, instance, le) (rate(kubelet_pod_worker_duration_seconds_bucket{job="kubelet"}[5m]))) * on(cluster, instance) group_left(node) kubelet_node_name{job="kubelet"} > 60
|
||||
for: 15m
|
||||
labels:
|
||||
severity: warning
|
||||
annotations:
|
||||
description: Kubelet Pod startup 99th percentile latency is {{ $value }} seconds on node {{ $labels.node }}.
|
||||
summary: Kubelet Pod startup latency is too high.
|
||||
- alert: KubeletClientCertificateExpiration
|
||||
expr: kubelet_certificate_manager_client_ttl_seconds < 604800
|
||||
labels:
|
||||
severity: warning
|
||||
annotations:
|
||||
description: Client certificate for Kubelet on node {{ $labels.node }} expires in {{ $value | humanizeDuration }}.
|
||||
summary: Kubelet client certificate is about to expire.
|
||||
- alert: KubeletClientCertificateExpiration
|
||||
expr: kubelet_certificate_manager_client_ttl_seconds < 86400
|
||||
labels:
|
||||
severity: critical
|
||||
annotations:
|
||||
description: Client certificate for Kubelet on node {{ $labels.node }} expires in {{ $value | humanizeDuration }}.
|
||||
summary: Kubelet client certificate is about to expire.
|
||||
- alert: KubeletServerCertificateExpiration
|
||||
expr: kubelet_certificate_manager_server_ttl_seconds < 604800
|
||||
labels:
|
||||
severity: warning
|
||||
annotations:
|
||||
description: Server certificate for Kubelet on node {{ $labels.node }} expires in {{ $value | humanizeDuration }}.
|
||||
summary: Kubelet server certificate is about to expire.
|
||||
- alert: KubeletServerCertificateExpiration
|
||||
expr: kubelet_certificate_manager_server_ttl_seconds < 86400
|
||||
labels:
|
||||
severity: critical
|
||||
annotations:
|
||||
description: Server certificate for Kubelet on node {{ $labels.node }} expires in {{ $value | humanizeDuration }}.
|
||||
summary: Kubelet server certificate is about to expire.
|
||||
- alert: KubeletClientCertificateRenewalErrors
|
||||
expr: increase(kubelet_certificate_manager_client_expiration_renew_errors[5m]) > 0
|
||||
for: 15m
|
||||
labels:
|
||||
severity: warning
|
||||
annotations:
|
||||
description: Kubelet on node {{ $labels.node }} has failed to renew its client certificate ({{ $value | humanize }} errors in the last 5 minutes).
|
||||
summary: Kubelet has failed to renew its client certificate.
|
||||
- alert: KubeletServerCertificateRenewalErrors
|
||||
expr: increase(kubelet_server_expiration_renew_errors[5m]) > 0
|
||||
for: 15m
|
||||
labels:
|
||||
severity: warning
|
||||
annotations:
|
||||
description: Kubelet on node {{ $labels.node }} has failed to renew its server certificate ({{ $value | humanize }} errors in the last 5 minutes).
|
||||
summary: Kubelet has failed to renew its server certificate.
|
||||
- alert: KubeletDown
|
||||
expr: absent(up{job="kubelet"} == 1)
|
||||
for: 15m
|
||||
labels:
|
||||
severity: critical
|
||||
annotations:
|
||||
description: Kubelet has disappeared from Prometheus target discovery.
|
||||
summary: Target disappeared from Prometheus target discovery.
|
||||
- alert: KubeProxyDown
|
||||
expr: absent(up{job="kube-proxy"} == 1)
|
||||
for: 15m
|
||||
labels:
|
||||
severity: critical
|
||||
annotations:
|
||||
description: KubeProxy has disappeared from Prometheus target discovery.
|
||||
summary: Target disappeared from Prometheus target discovery.
|
||||
- alert: KubeVersionMismatch
|
||||
expr: count by(cluster) (count by(git_version, cluster) (label_replace(kubernetes_build_info{job!~"kube-dns|coredns"}, "git_version", "$1", "git_version", "(v[0-9]*.[0-9]*).*"))) > 1
|
||||
for: 15m
|
||||
labels:
|
||||
severity: warning
|
||||
annotations:
|
||||
description: There are {{ $value }} different semantic versions of Kubernetes components running.
|
||||
summary: Different semantic versions of Kubernetes components running.
|
||||
- alert: KubeClientErrors
|
||||
expr: (sum by(cluster, instance, job, namespace) (rate(rest_client_requests_total{code=~"5.."}[5m])) / sum by(cluster, instance, job, namespace) (rate(rest_client_requests_total[5m]))) > 0.01
|
||||
for: 15m
|
||||
labels:
|
||||
severity: warning
|
||||
annotations:
|
||||
description: Kubernetes API server client '{{ $labels.job }}/{{ $labels.instance }}' is experiencing {{ $value | humanizePercentage }} errors.'
|
||||
summary: Kubernetes API server client is experiencing errors.
|
||||
- alert: KubeControllerManagerDown
|
||||
expr: absent(up{job="kube-controller-manager"} == 1)
|
||||
for: 15m
|
||||
labels:
|
||||
severity: critical
|
||||
annotations:
|
||||
description: KubeControllerManager has disappeared from Prometheus target discovery.
|
||||
summary: Target disappeared from Prometheus target discovery.
|
||||
- alert: KubeClientCertificateExpiration
|
||||
expr: apiserver_client_certificate_expiration_seconds_count{job="apiserver"} > 0 and on(job) histogram_quantile(0.01, sum by(job, le) (rate(apiserver_client_certificate_expiration_seconds_bucket{job="apiserver"}[5m]))) < 604800
|
||||
labels:
|
||||
severity: warning
|
||||
annotations:
|
||||
description: A client certificate used to authenticate to kubernetes apiserver is expiring in less than 7.0 days.
|
||||
summary: Client certificate is about to expire.
|
||||
- alert: KubeClientCertificateExpiration
|
||||
expr: apiserver_client_certificate_expiration_seconds_count{job="apiserver"} > 0 and on(job) histogram_quantile(0.01, sum by(job, le) (rate(apiserver_client_certificate_expiration_seconds_bucket{job="apiserver"}[5m]))) < 86400
|
||||
labels:
|
||||
severity: critical
|
||||
annotations:
|
||||
description: A client certificate used to authenticate to kubernetes apiserver is expiring in less than 24.0 hours.
|
||||
summary: Client certificate is about to expire.
|
||||
- alert: KubeAggregatedAPIErrors
|
||||
expr: sum by(name, namespace, cluster) (increase(aggregator_unavailable_apiservice_total[10m])) > 4
|
||||
labels:
|
||||
severity: warning
|
||||
annotations:
|
||||
description: Kubernetes aggregated API {{ $labels.name }}/{{ $labels.namespace }} has reported errors. It has appeared unavailable {{ $value | humanize }} times averaged over the past 10m.
|
||||
summary: Kubernetes aggregated API has reported errors.
|
||||
- name: infra-alerts-03
|
||||
rules:
|
||||
- alert: KubeAggregatedAPIDown
|
||||
expr: (1 - max by(name, namespace, cluster) (avg_over_time(aggregator_unavailable_apiservice[10m]))) * 100 < 85
|
||||
for: 5m
|
||||
labels:
|
||||
severity: warning
|
||||
annotations:
|
||||
description: Kubernetes aggregated API {{ $labels.name }}/{{ $labels.namespace }} has been only {{ $value | humanize }}% available over the last 10m.
|
||||
summary: Kubernetes aggregated API is down.
|
||||
- alert: KubeAPIDown
|
||||
expr: absent(up{job="apiserver"} == 1)
|
||||
for: 15m
|
||||
labels:
|
||||
severity: critical
|
||||
annotations:
|
||||
description: KubeAPI has disappeared from Prometheus target discovery.
|
||||
summary: Target disappeared from Prometheus target discovery.
|
||||
- alert: KubeAPITerminatedRequests
|
||||
expr: sum(rate(apiserver_request_terminations_total{job="apiserver"}[10m])) / (sum(rate(apiserver_request_total{job="apiserver"}[10m])) + sum(rate(apiserver_request_terminations_total{job="apiserver"}[10m]))) > 0.2
|
||||
for: 5m
|
||||
labels:
|
||||
severity: warning
|
||||
annotations:
|
||||
description: The kubernetes apiserver has terminated {{ $value | humanizePercentage }} of its incoming requests.
|
||||
summary: The kubernetes apiserver has terminated {{ $value | humanizePercentage }} of its incoming requests.
|
||||
- alert: KubePersistentVolumeFillingUp
|
||||
expr: (kubelet_volume_stats_available_bytes{job="kubelet",namespace=~".*"} / kubelet_volume_stats_capacity_bytes{job="kubelet",namespace=~".*"}) < 0.03 and kubelet_volume_stats_used_bytes{job="kubelet",namespace=~".*"} > 0 unless on(namespace, persistentvolumeclaim) kube_persistentvolumeclaim_access_mode{access_mode="ReadOnlyMany"} == 1 unless on(namespace, persistentvolumeclaim) kube_persistentvolumeclaim_labels{label_excluded_from_alerts="true"} == 1
|
||||
for: 1m
|
||||
labels:
|
||||
severity: critical
|
||||
annotations:
|
||||
description: The PersistentVolume claimed by {{ $labels.persistentvolumeclaim }} in Namespace {{ $labels.namespace }} is only {{ $value | humanizePercentage }} free.
|
||||
summary: PersistentVolume is filling up.
|
||||
- alert: KubePersistentVolumeFillingUp
|
||||
expr: (kubelet_volume_stats_available_bytes{job="kubelet",namespace=~".*"} / kubelet_volume_stats_capacity_bytes{job="kubelet",namespace=~".*"}) < 0.15 and kubelet_volume_stats_used_bytes{job="kubelet",namespace=~".*"} > 0 and predict_linear(kubelet_volume_stats_available_bytes{job="kubelet",namespace=~".*"}[6h], 4 * 24 * 3600) < 0 unless on(namespace, persistentvolumeclaim) kube_persistentvolumeclaim_access_mode{access_mode="ReadOnlyMany"} == 1 unless on(namespace, persistentvolumeclaim) kube_persistentvolumeclaim_labels{label_excluded_from_alerts="true"} == 1
|
||||
for: 1h
|
||||
labels:
|
||||
severity: warning
|
||||
annotations:
|
||||
description: Based on recent sampling, the PersistentVolume claimed by {{ $labels.persistentvolumeclaim }} in Namespace {{ $labels.namespace }} is expected to fill up within four days.
|
||||
summary: PersistentVolume is filling up.
|
||||
- alert: KubePersistentVolumeInodesFillingUp
|
||||
expr: (kubelet_volume_stats_inodes_free{job="kubelet",namespace=~".*"} / kubelet_volume_stats_inodes{job="kubelet",namespace=~".*"}) < 0.03 and kubelet_volume_stats_inodes_used{job="kubelet",namespace=~".*"} > 0 unless on(namespace, persistentvolumeclaim) kube_persistentvolumeclaim_access_mode{access_mode="ReadOnlyMany"} == 1 unless on(namespace, persistentvolumeclaim) kube_persistentvolumeclaim_labels{label_excluded_from_alerts="true"} == 1
|
||||
for: 1m
|
||||
labels:
|
||||
severity: critical
|
||||
annotations:
|
||||
description: The PersistentVolume claimed by {{ $labels.persistentvolumeclaim }} in Namespace {{ $labels.namespace }} only has {{ $value | humanizePercentage }} free inodes.
|
||||
summary: PersistentVolumeInodes is filling up.
|
||||
- alert: KubePersistentVolumeInodesFillingUp
|
||||
expr: (kubelet_volume_stats_inodes_free{job="kubelet",namespace=~".*"} / kubelet_volume_stats_inodes{job="kubelet",namespace=~".*"}) < 0.15 and kubelet_volume_stats_inodes_used{job="kubelet",namespace=~".*"} > 0 and predict_linear(kubelet_volume_stats_inodes_free{job="kubelet",namespace=~".*"}[6h], 4 * 24 * 3600) < 0 unless on(namespace, persistentvolumeclaim) kube_persistentvolumeclaim_access_mode{access_mode="ReadOnlyMany"} == 1 unless on(namespace, persistentvolumeclaim) kube_persistentvolumeclaim_labels{label_excluded_from_alerts="true"} == 1
|
||||
for: 1h
|
||||
labels:
|
||||
severity: warning
|
||||
annotations:
|
||||
description: Based on recent sampling, the PersistentVolume claimed by {{ $labels.persistentvolumeclaim }} in Namespace {{ $labels.namespace }} is expected to run out of inodes within four days. Currently {{ $value | humanizePercentage }} of its inodes are free.
|
||||
summary: PersistentVolumeInodes are filling up.
|
||||
- alert: KubePersistentVolumeErrors
|
||||
expr: kube_persistentvolume_status_phase{job="kube-state-metrics",phase=~"Failed|Pending"} > 0
|
||||
for: 5m
|
||||
labels:
|
||||
severity: critical
|
||||
annotations:
|
||||
description: The persistent volume {{ $labels.persistentvolume }} has status {{ $labels.phase }}.
|
||||
summary: PersistentVolume is having issues with provisioning.
|
||||
- alert: KubeCPUOvercommit
|
||||
expr: sum(namespace_cpu:kube_pod_container_resource_requests:sum) - (sum(kube_node_status_allocatable{resource="cpu"}) - max(kube_node_status_allocatable{resource="cpu"})) > 0 and (sum(kube_node_status_allocatable{resource="cpu"}) - max(kube_node_status_allocatable{resource="cpu"})) > 0
|
||||
for: 10m
|
||||
labels:
|
||||
severity: warning
|
||||
annotations:
|
||||
description: Cluster has overcommitted CPU resource requests for Pods by {{ $value }} CPU shares and cannot tolerate node failure.
|
||||
summary: Cluster has overcommitted CPU resource requests.
|
||||
- alert: KubeMemoryOvercommit
|
||||
expr: sum(namespace_memory:kube_pod_container_resource_requests:sum) - (sum(kube_node_status_allocatable{resource="memory"}) - max(kube_node_status_allocatable{resource="memory"})) > 0 and (sum(kube_node_status_allocatable{resource="memory"}) - max(kube_node_status_allocatable{resource="memory"})) > 0
|
||||
for: 10m
|
||||
labels:
|
||||
severity: warning
|
||||
annotations:
|
||||
description: Cluster has overcommitted memory resource requests for Pods by {{ $value | humanize }} bytes and cannot tolerate node failure.
|
||||
summary: Cluster has overcommitted memory resource requests.
|
||||
- alert: KubeCPUQuotaOvercommit
|
||||
expr: sum(min without(resource) (kube_resourcequota{job="kube-state-metrics",resource=~"(cpu|requests.cpu)",type="hard"})) / sum(kube_node_status_allocatable{job="kube-state-metrics",resource="cpu"}) > 1.5
|
||||
for: 5m
|
||||
labels:
|
||||
severity: warning
|
||||
annotations:
|
||||
description: Cluster has overcommitted CPU resource requests for Namespaces.
|
||||
summary: Cluster has overcommitted CPU resource requests.
|
||||
- alert: KubeMemoryQuotaOvercommit
|
||||
expr: sum(min without(resource) (kube_resourcequota{job="kube-state-metrics",resource=~"(memory|requests.memory)",type="hard"})) / sum(kube_node_status_allocatable{job="kube-state-metrics",resource="memory"}) > 1.5
|
||||
for: 5m
|
||||
labels:
|
||||
severity: warning
|
||||
annotations:
|
||||
description: Cluster has overcommitted memory resource requests for Namespaces.
|
||||
summary: Cluster has overcommitted memory resource requests.
|
||||
- alert: KubeQuotaAlmostFull
|
||||
expr: kube_resourcequota{job="kube-state-metrics",type="used"} / ignoring(instance, job, type) (kube_resourcequota{job="kube-state-metrics",type="hard"} > 0) > 0.9 < 1
|
||||
for: 15m
|
||||
labels:
|
||||
severity: info
|
||||
annotations:
|
||||
description: Namespace {{ $labels.namespace }} is using {{ $value | humanizePercentage }} of its {{ $labels.resource }} quota.
|
||||
summary: Namespace quota is going to be full.
|
||||
- alert: KubeQuotaFullyUsed
|
||||
expr: kube_resourcequota{job="kube-state-metrics",type="used"} / ignoring(instance, job, type) (kube_resourcequota{job="kube-state-metrics",type="hard"} > 0) == 1
|
||||
for: 15m
|
||||
labels:
|
||||
severity: info
|
||||
annotations:
|
||||
description: Namespace {{ $labels.namespace }} is using {{ $value | humanizePercentage }} of its {{ $labels.resource }} quota.
|
||||
summary: Namespace quota is fully used.
|
||||
- alert: KubeQuotaExceeded
|
||||
expr: kube_resourcequota{job="kube-state-metrics",type="used"} / ignoring(instance, job, type) (kube_resourcequota{job="kube-state-metrics",type="hard"} > 0) > 1
|
||||
for: 15m
|
||||
labels:
|
||||
severity: warning
|
||||
annotations:
|
||||
description: Namespace {{ $labels.namespace }} is using {{ $value | humanizePercentage }} of its {{ $labels.resource }} quota.
|
||||
summary: Namespace quota has exceeded the limits.
|
||||
- alert: CPUThrottlingHigh
|
||||
expr: sum by(container, pod, namespace) (increase(container_cpu_cfs_throttled_periods_total{container!=""}[5m])) / sum by(container, pod, namespace) (increase(container_cpu_cfs_periods_total[5m])) > (25 / 100)
|
||||
for: 15m
|
||||
labels:
|
||||
severity: info
|
||||
annotations:
|
||||
description: The {{ $value | humanizePercentage }} throttling of CPU in namespace {{ $labels.namespace }} for container {{ $labels.container }} in pod {{ $labels.pod }}.
|
||||
summary: Processes experience elevated CPU throttling.
|
||||
- alert: KubePodCrashLooping
|
||||
expr: max_over_time(kube_pod_container_status_waiting_reason{job="kube-state-metrics",namespace=~".*",reason="CrashLoopBackOff"}[5m]) >= 1
|
||||
for: 15m
|
||||
labels:
|
||||
severity: warning
|
||||
annotations:
|
||||
description: Pod {{ $labels.namespace }}/{{ $labels.pod }} ({{ $labels.container }}) is in waiting state (reason:"CrashLoopBackOff").
|
||||
summary: Pod is crash looping.
|
||||
- alert: KubePodNotReady
|
||||
expr: sum by(namespace, pod, cluster) (max by(namespace, pod, cluster) (kube_pod_status_phase{job="kube-state-metrics",namespace=~".*",phase=~"Pending|Unknown"}) * on(namespace, pod, cluster) group_left(owner_kind) topk by(namespace, pod, cluster) (1, max by(namespace, pod, owner_kind, cluster) (kube_pod_owner{owner_kind!="Job"}))) > 0
|
||||
for: 15m
|
||||
labels:
|
||||
severity: warning
|
||||
annotations:
|
||||
description: Pod {{ $labels.namespace }}/{{ $labels.pod }} ({{ $labels.container }}) has been in a non-ready state for longer than 15 minutes.
|
||||
summary: Pod has been in a non-ready state for more than 15 minutes.
|
||||
- alert: KubeDeploymentGenerationMismatch
|
||||
expr: kube_deployment_status_observed_generation{job="kube-state-metrics",namespace=~".*"} != kube_deployment_metadata_generation{job="kube-state-metrics",namespace=~".*"}
|
||||
for: 15m
|
||||
labels:
|
||||
severity: warning
|
||||
annotations:
|
||||
description: Deployment generation for {{ $labels.namespace }}/{{ $labels.deployment }} does not match, this indicates that the Deployment has failed but has not been rolled back.
|
||||
summary: Deployment generation mismatch due to possible roll-back
|
||||
- alert: KubeDeploymentReplicasMismatch
|
||||
expr: (kube_deployment_spec_replicas{job="kube-state-metrics",namespace=~".*"} > kube_deployment_status_replicas_available{job="kube-state-metrics",namespace=~".*"}) and (changes(kube_deployment_status_replicas_updated{job="kube-state-metrics",namespace=~".*"}[10m]) == 0)
|
||||
for: 15m
|
||||
labels:
|
||||
severity: warning
|
||||
annotations:
|
||||
description: Deployment {{ $labels.namespace }}/{{ $labels.deployment }} has not matched the expected number of replicas for longer than 15 minutes.
|
||||
summary: Deployment has not matched the expected number of replicas.
|
||||
- name: infra-alerts-04
|
||||
rules:
|
||||
- alert: KubeStatefulSetReplicasMismatch
|
||||
expr: (kube_statefulset_status_replicas_ready{job="kube-state-metrics",namespace=~".*"} != kube_statefulset_status_replicas{job="kube-state-metrics",namespace=~".*"}) and (changes(kube_statefulset_status_replicas_updated{job="kube-state-metrics",namespace=~".*"}[10m]) == 0)
|
||||
for: 15m
|
||||
labels:
|
||||
severity: warning
|
||||
annotations:
|
||||
description: StatefulSet {{ $labels.namespace }}/{{ $labels.statefulset }} has not matched the expected number of replicas for longer than 15 minutes.
|
||||
summary: Deployment has not matched the expected number of replicas.
|
||||
- alert: KubeStatefulSetGenerationMismatch
|
||||
expr: kube_statefulset_status_observed_generation{job="kube-state-metrics",namespace=~".*"} != kube_statefulset_metadata_generation{job="kube-state-metrics",namespace=~".*"}
|
||||
for: 15m
|
||||
labels:
|
||||
severity: warning
|
||||
annotations:
|
||||
description: StatefulSet generation for {{ $labels.namespace }}/{{ $labels.statefulset }} does not match, this indicates that the StatefulSet has failed but has not been rolled back.
|
||||
summary: StatefulSet generation mismatch due to possible roll-back
|
||||
- alert: KubeStatefulSetUpdateNotRolledOut
|
||||
expr: (max without(revision) (kube_statefulset_status_current_revision{job="kube-state-metrics",namespace=~".*"} unless kube_statefulset_status_update_revision{job="kube-state-metrics",namespace=~".*"}) * (kube_statefulset_replicas{job="kube-state-metrics",namespace=~".*"} != kube_statefulset_status_replicas_updated{job="kube-state-metrics",namespace=~".*"})) and (changes(kube_statefulset_status_replicas_updated{job="kube-state-metrics",namespace=~".*"}[5m]) == 0)
|
||||
for: 15m
|
||||
labels:
|
||||
severity: warning
|
||||
annotations:
|
||||
description: StatefulSet {{ $labels.namespace }}/{{ $labels.statefulset }} update has not been rolled out.
|
||||
summary: StatefulSet update has not been rolled out.
|
||||
- alert: KubeDaemonSetRolloutStuck
|
||||
expr: ((kube_daemonset_status_current_number_scheduled{job="kube-state-metrics",namespace=~".*"} != kube_daemonset_status_desired_number_scheduled{job="kube-state-metrics",namespace=~".*"}) or (kube_daemonset_status_number_misscheduled{job="kube-state-metrics",namespace=~".*"} != 0) or (kube_daemonset_status_updated_number_scheduled{job="kube-state-metrics",namespace=~".*"} != kube_daemonset_status_desired_number_scheduled{job="kube-state-metrics",namespace=~".*"}) or (kube_daemonset_status_number_available{job="kube-state-metrics",namespace=~".*"} != kube_daemonset_status_desired_number_scheduled{job="kube-state-metrics",namespace=~".*"})) and (changes(kube_daemonset_status_updated_number_scheduled{job="kube-state-metrics",namespace=~".*"}[5m]) == 0)
|
||||
for: 15m
|
||||
labels:
|
||||
severity: warning
|
||||
annotations:
|
||||
description: DaemonSet {{ $labels.namespace }}/{{ $labels.daemonset }} has not finished or progressed for at least 15 minutes.
|
||||
summary: DaemonSet rollout is stuck.
|
||||
- alert: KubeContainerWaiting
|
||||
expr: sum by(namespace, pod, container, cluster) (kube_pod_container_status_waiting_reason{job="kube-state-metrics",namespace=~".*"}) > 0
|
||||
for: 1h
|
||||
labels:
|
||||
severity: warning
|
||||
annotations:
|
||||
description: Pod/{{ $labels.pod }} in namespace {{ $labels.namespace }} on container {{ $labels.container}} has been in waiting state for longer than 1 hour.
|
||||
summary: Pod container waiting longer than 1 hour
|
||||
- alert: KubeDaemonSetNotScheduled
|
||||
expr: kube_daemonset_status_desired_number_scheduled{job="kube-state-metrics",namespace=~".*"} - kube_daemonset_status_current_number_scheduled{job="kube-state-metrics",namespace=~".*"} > 0
|
||||
for: 10m
|
||||
labels:
|
||||
severity: warning
|
||||
annotations:
|
||||
description: The {{ $value }} Pods of DaemonSet {{ $labels.namespace }}/{{ $labels.daemonset }} are not scheduled.
|
||||
summary: DaemonSet pods are not scheduled.
|
||||
- alert: KubeDaemonSetMisScheduled
|
||||
expr: kube_daemonset_status_number_misscheduled{job="kube-state-metrics",namespace=~".*"} > 0
|
||||
for: 15m
|
||||
labels:
|
||||
severity: warning
|
||||
annotations:
|
||||
description: The {{ $value }} Pods of DaemonSet {{ $labels.namespace }}/{{ $labels.daemonset }} are running where they are not supposed to run.
|
||||
summary: DaemonSet pods are misscheduled.
|
||||
- alert: KubeJobNotCompleted
|
||||
expr: time() - max by(namespace, job_name, cluster) (kube_job_status_start_time{job="kube-state-metrics",namespace=~".*"} and kube_job_status_active{job="kube-state-metrics",namespace=~".*"} > 0) > 43200
|
||||
labels:
|
||||
severity: warning
|
||||
annotations:
|
||||
description: Job {{ $labels.namespace }}/{{ $labels.job_name }} is taking more than {{ "43200" | humanizeDuration }} to complete.
|
||||
summary: Job did not complete in time
|
||||
- alert: KubeJobFailed
|
||||
expr: kube_job_failed{job="kube-state-metrics",namespace=~".*"} > 0
|
||||
for: 15m
|
||||
labels:
|
||||
severity: warning
|
||||
annotations:
|
||||
description: Job {{ $labels.namespace }}/{{ $labels.job_name }} failed to complete. Removing failed job after investigation should clear this alert.
|
||||
summary: Job failed to complete.
|
||||
- alert: KubeHpaReplicasMismatch
|
||||
expr: (kube_horizontalpodautoscaler_status_desired_replicas{job="kube-state-metrics",namespace=~".*"} != kube_horizontalpodautoscaler_status_current_replicas{job="kube-state-metrics",namespace=~".*"}) and (kube_horizontalpodautoscaler_status_current_replicas{job="kube-state-metrics",namespace=~".*"} > kube_horizontalpodautoscaler_spec_min_replicas{job="kube-state-metrics",namespace=~".*"}) and (kube_horizontalpodautoscaler_status_current_replicas{job="kube-state-metrics",namespace=~".*"} < kube_horizontalpodautoscaler_spec_max_replicas{job="kube-state-metrics",namespace=~".*"}) and changes(kube_horizontalpodautoscaler_status_current_replicas{job="kube-state-metrics",namespace=~".*"}[15m]) == 0
|
||||
for: 15m
|
||||
labels:
|
||||
severity: warning
|
||||
annotations:
|
||||
description: HPA {{ $labels.namespace }}/{{ $labels.horizontalpodautoscaler }} has not matched the desired number of replicas for longer than 15 minutes.
|
||||
summary: HPA has not matched descired number of replicas.
|
||||
- alert: KubeHpaMaxedOut
|
||||
expr: kube_horizontalpodautoscaler_status_current_replicas{job="kube-state-metrics",namespace=~".*"} == kube_horizontalpodautoscaler_spec_max_replicas{job="kube-state-metrics",namespace=~".*"}
|
||||
for: 15m
|
||||
labels:
|
||||
severity: warning
|
||||
annotations:
|
||||
description: HPA {{ $labels.namespace }}/{{ $labels.horizontalpodautoscaler }} has been running at max replicas for longer than 15 minutes.
|
||||
summary: HPA is running at max replicas
|
||||
- alert: KubeStateMetricsListErrors
|
||||
expr: (sum(rate(kube_state_metrics_list_total{job="kube-state-metrics",result="error"}[5m])) / sum(rate(kube_state_metrics_list_total{job="kube-state-metrics"}[5m]))) > 0.01
|
||||
for: 15m
|
||||
labels:
|
||||
severity: critical
|
||||
annotations:
|
||||
description: kube-state-metrics is experiencing errors at an elevated rate in list operations. This is likely causing it to not be able to expose metrics about Kubernetes objects or at all.
|
||||
summary: kube-state-metrics is experiencing errors in list operations.
|
||||
- alert: KubeStateMetricsWatchErrors
|
||||
expr: (sum(rate(kube_state_metrics_watch_total{job="kube-state-metrics",result="error"}[5m])) / sum(rate(kube_state_metrics_watch_total{job="kube-state-metrics"}[5m]))) > 0.01
|
||||
for: 15m
|
||||
labels:
|
||||
severity: critical
|
||||
annotations:
|
||||
description: kube-state-metrics is experiencing errors at an elevated rate in list operations. This is likely causing it to not be able to expose metrics about Kubernetes objects or at all.
|
||||
summary: kube-state-metrics is experiencing errors in watch operations.
|
||||
- alert: KubeStateMetricsShardingMismatch
|
||||
expr: stdvar(kube_state_metrics_total_shards{job="kube-state-metrics"}) != 0
|
||||
for: 15m
|
||||
labels:
|
||||
severity: critical
|
||||
annotations:
|
||||
description: kube-state-metrics pods are running with different --total-shards configuration, some Kubernetes objects may be exposed multiple times or not exposed at all.
|
||||
summary: kube-state-metrics sharding is misconfigured.
|
||||
- alert: KubeStateMetricsShardsMissing
|
||||
expr: 2 ^ max(kube_state_metrics_total_shards{job="kube-state-metrics"}) - 1 - sum(2 ^ max by(shard_ordinal) (kube_state_metrics_shard_ordinal{job="kube-state-metrics"})) != 0
|
||||
for: 15m
|
||||
labels:
|
||||
severity: critical
|
||||
annotations:
|
||||
description: kube-state-metrics shards are missing, some Kubernetes objects are not being exposed.
|
||||
summary: kube-state-metrics shards are missing.
|
||||
- alert: KubeAPIErrorBudgetBurn
|
||||
expr: sum(apiserver_request:burnrate1h) > (14.4 * 0.01) and sum(apiserver_request:burnrate5m) > (14.4 * 0.01)
|
||||
for: 2m
|
||||
labels:
|
||||
long: 1h
|
||||
severity: critical
|
||||
short: 5m
|
||||
annotations:
|
||||
description: The API server is burning too much error budget.
|
||||
summary: The API server is burning too much error budget.
|
||||
- alert: KubeAPIErrorBudgetBurn
|
||||
expr: sum(apiserver_request:burnrate6h) > (6 * 0.01) and sum(apiserver_request:burnrate30m) > (6 * 0.01)
|
||||
for: 15m
|
||||
labels:
|
||||
long: 6h
|
||||
severity: critical
|
||||
short: 30m
|
||||
annotations:
|
||||
description: The API server is burning too much error budget.
|
||||
summary: The API server is burning too much error budget.
|
||||
- alert: KubeAPIErrorBudgetBurn
|
||||
expr: sum(apiserver_request:burnrate1d) > (3 * 0.01) and sum(apiserver_request:burnrate2h) > (3 * 0.01)
|
||||
for: 1d
|
||||
labels:
|
||||
long: 1d
|
||||
severity: warning
|
||||
short: 2h
|
||||
annotations:
|
||||
description: The API server is burning too much error budget.
|
||||
summary: The API server is burning too much error budget.
|
||||
- alert: KubeAPIErrorBudgetBurn
|
||||
expr: sum(apiserver_request:burnrate3d) > (1 * 0.01) and sum(apiserver_request:burnrate6h) > (1 * 0.01)
|
||||
for: 3h
|
||||
labels:
|
||||
long: 3d
|
||||
severity: warning
|
||||
short: 6h
|
||||
annotations:
|
||||
description: The API server is burning too much error budget.
|
||||
summary: The API server is burning too much error budget.
|
||||
- alert: TargetDown
|
||||
expr: 100 * (count by(job, namespace, service) (up == 0) / count by(job, namespace, service) (up)) > 10
|
||||
for: 10m
|
||||
labels:
|
||||
severity: warning
|
||||
annotations:
|
||||
description: The {{ printf "%.4g" $value }}% of the {{ $labels.job }}/{{ $labels.service }} targets in {{ $labels.namespace }} namespace are down.
|
||||
- name: infra-alerts-05
|
||||
rules:
|
||||
- alert: Watchdog
|
||||
expr: vector(1)
|
||||
labels:
|
||||
severity: none
|
||||
annotations:
|
||||
description: This is an alert meant to ensure that the entire alerting pipeline is functional. This alert is always firing, therefore it should always be firing in Alertmanager and always fire against a receiver. There are integrations with various notification mechanisms that send a notification when this alert is not firing. For example the "DeadMansSnitch" integration in PagerDuty.
|
||||
- alert: InfoInhibitor
|
||||
expr: ALERTS{severity="info"} == 1 unless on(namespace) ALERTS{alertname!="InfoInhibitor",alertstate="firing",severity=~"warning|critical"} == 1
|
||||
labels:
|
||||
severity: none
|
||||
annotations:
|
||||
description: This is an alert that is used to inhibit info alerts. By themselves, the info-level alerts are sometimes very noisy, but they are relevant when combined with other alerts. This alert fires whenever there's a severity="info" alert, and stops firing when another alert with a severity of 'warning' or 'critical' starts firing on the same namespace. This alert should be routed to a null receiver and configured to inhibit alerts with severity="info".
|
||||
- alert: etcdInsufficientMembers
|
||||
expr: sum by(job) (up{job=~".*etcd.*"} == bool 1) < ((count by(job) (up{job=~".*etcd.*"}) + 1) / 2)
|
||||
for: 3m
|
||||
labels:
|
||||
severity: critical
|
||||
annotations:
|
||||
message: etcd cluster "{{ $labels.job }}":insufficient members ({{ $value }}).
|
||||
- alert: etcdHighNumberOfLeaderChanges
|
||||
expr: rate(etcd_server_leader_changes_seen_total{job=~".*etcd.*"}[15m]) > 3
|
||||
for: 15m
|
||||
labels:
|
||||
severity: warning
|
||||
annotations:
|
||||
message: etcd cluster "{{ $labels.job }}":instance {{ $labels.instance }} has seen {{ $value }} leader changes within the last hour.
|
||||
- alert: etcdNoLeader
|
||||
expr: etcd_server_has_leader{job=~".*etcd.*"} == 0
|
||||
for: 1m
|
||||
labels:
|
||||
severity: critical
|
||||
annotations:
|
||||
message: message:etcd cluster "{{ $labels.job }}":member {{ $labels.instance }} has no leader.
|
||||
- alert: etcdHighNumberOfFailedGRPCRequests
|
||||
expr: 100 * sum by(job, instance, grpc_service, grpc_method) (rate(grpc_server_handled_total{grpc_code!="OK",job=~".*etcd.*"}[5m])) / sum by(job, instance, grpc_service, grpc_method) (rate(grpc_server_handled_total{job=~".*etcd.*"}[5m])) > 1
|
||||
for: 10m
|
||||
labels:
|
||||
severity: warning
|
||||
annotations:
|
||||
message: etcd cluster "{{ $labels.job }}":{{ $value }}% of requests for {{ $labels.grpc_method }} failed on etcd instance {{ $labels.instance }}.
|
||||
- alert: etcdGRPCRequestsSlow
|
||||
expr: histogram_quantile(0.99, sum by(job, instance, grpc_service, grpc_method, le) (rate(grpc_server_handling_seconds_bucket{grpc_type="unary",job=~".*etcd.*"}[5m]))) > 0.15
|
||||
for: 10m
|
||||
labels:
|
||||
severity: critical
|
||||
annotations:
|
||||
message: etcd cluster "{{ $labels.job }}":gRPC requests to {{ $labels.grpc_method }} are taking {{ $value }}s on etcd instance {{ $labels.instance }}.
|
||||
- alert: etcdMemberCommunicationSlow
|
||||
expr: histogram_quantile(0.99, rate(etcd_network_peer_round_trip_time_seconds_bucket{job=~".*etcd.*"}[5m])) > 0.15
|
||||
for: 10m
|
||||
labels:
|
||||
severity: warning
|
||||
annotations:
|
||||
message: message:etcd cluster "{{ $labels.job }}":member communication with {{ $labels.To }} is taking {{ $value }}s on etcd instance {{ $labels.instance }}.
|
||||
- alert: etcdHighNumberOfFailedProposals
|
||||
expr: rate(etcd_server_proposals_failed_total{job=~".*etcd.*"}[15m]) > 5
|
||||
for: 15m
|
||||
labels:
|
||||
severity: warning
|
||||
annotations:
|
||||
message: etcd cluster "{{ $labels.job }}":{{ $value }} proposal failures within the last hour on etcd instance {{ $labels.instance }}.
|
||||
- alert: etcdHighFsyncDurations
|
||||
expr: histogram_quantile(0.99, rate(etcd_disk_wal_fsync_duration_seconds_bucket{job=~".*etcd.*"}[5m])) > 0.5
|
||||
for: 10m
|
||||
labels:
|
||||
severity: warning
|
||||
annotations:
|
||||
message: etcd cluster "{{ $labels.job }}":99th percentile fync durations are {{ $value }}s on etcd instance {{ $labels.instance }}.
|
||||
- alert: etcdHighCommitDurations
|
||||
expr: histogram_quantile(0.99, rate(etcd_disk_backend_commit_duration_seconds_bucket{job=~".*etcd.*"}[5m])) > 0.25
|
||||
for: 10m
|
||||
labels:
|
||||
severity: warning
|
||||
annotations:
|
||||
message: etcd cluster "{{ $labels.job }}":99th percentile commit durations {{ $value }}s on etcd instance {{ $labels.instance }}.
|
||||
- alert: etcdHighNumberOfFailedHTTPRequests
|
||||
expr: sum by(method) (rate(etcd_http_failed_total{code!="404",job=~".*etcd.*"}[5m])) / sum by(method) (rate(etcd_http_received_total{job=~".*etcd.*"}[5m])) > 0.01
|
||||
for: 10m
|
||||
labels:
|
||||
severity: warning
|
||||
annotations:
|
||||
message: The {{ $value }}% of requests for {{ $labels.method }} failed on etcd instance {{ $labels.instance }}
|
||||
- alert: etcdHighNumberOfFailedHTTPRequests
|
||||
expr: sum by(method) (rate(etcd_http_failed_total{code!="404",job=~".*etcd.*"}[5m])) / sum by(method) (rate(etcd_http_received_total{job=~".*etcd.*"}[5m])) > 0.05
|
||||
for: 10m
|
||||
labels:
|
||||
severity: warning
|
||||
annotations:
|
||||
message: The {{ $value }}% of requests for {{ $labels.method }} failed on etcd instance {{ $labels.instance }}.
|
||||
- alert: etcdHTTPRequestsSlow
|
||||
expr: histogram_quantile(0.99, rate(etcd_http_successful_duration_seconds_bucket[5m])) > 0.15
|
||||
for: 10m
|
||||
labels:
|
||||
severity: warning
|
||||
annotations:
|
||||
message: etcd instance {{ $labels.instance }} HTTP requests to {{ $labels.method }} are slow.
|
||||
EOF
|
||||
}
|
||||
@@ -0,0 +1,30 @@
|
||||
resource "grafana_dashboard" "workloads" {
|
||||
count = var.enable_dashboards ? 1 : 0
|
||||
folder = var.dashboards_folder_id
|
||||
config_json = file("${path.module}/dashboards/workloads.json")
|
||||
}
|
||||
|
||||
resource "grafana_dashboard" "nodes" {
|
||||
count = var.enable_dashboards ? 1 : 0
|
||||
folder = var.dashboards_folder_id
|
||||
config_json = file("${path.module}/dashboards/nodes.json")
|
||||
}
|
||||
|
||||
resource "grafana_dashboard" "nsworkload" {
|
||||
count = var.enable_dashboards ? 1 : 0
|
||||
folder = var.dashboards_folder_id
|
||||
config_json = file("${path.module}/dashboards/namespace-workloads.json")
|
||||
}
|
||||
|
||||
|
||||
resource "grafana_dashboard" "kubelet" {
|
||||
count = var.enable_dashboards ? 1 : 0
|
||||
folder = var.dashboards_folder_id
|
||||
config_json = file("${path.module}/dashboards/kubelet.json")
|
||||
}
|
||||
|
||||
resource "grafana_dashboard" "cluster" {
|
||||
count = var.enable_dashboards ? 1 : 0
|
||||
folder = var.dashboards_folder_id
|
||||
config_json = file("${path.module}/dashboards/cluster.json")
|
||||
}
|
||||
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,31 @@
|
||||
data "aws_partition" "current" {}
|
||||
|
||||
data "aws_caller_identity" "current" {}
|
||||
|
||||
data "aws_region" "current" {}
|
||||
|
||||
data "aws_eks_cluster" "eks_cluster" {
|
||||
name = var.eks_cluster_id
|
||||
}
|
||||
|
||||
locals {
|
||||
name = "adot-collector-kubeprometheus"
|
||||
namespace = try(var.helm_config.namespace, local.name)
|
||||
|
||||
eks_oidc_issuer_url = replace(data.aws_eks_cluster.eks_cluster.identity[0].oidc[0].issuer, "https://", "")
|
||||
eks_cluster_endpoint = data.aws_eks_cluster.eks_cluster.endpoint
|
||||
|
||||
context = {
|
||||
aws_caller_identity_account_id = data.aws_caller_identity.current.account_id
|
||||
aws_caller_identity_arn = data.aws_caller_identity.current.arn
|
||||
aws_eks_cluster_endpoint = local.eks_cluster_endpoint
|
||||
aws_partition_id = data.aws_partition.current.partition
|
||||
aws_region_name = data.aws_region.current.name
|
||||
eks_cluster_id = var.eks_cluster_id
|
||||
eks_oidc_issuer_url = local.eks_oidc_issuer_url
|
||||
eks_oidc_provider_arn = "arn:${data.aws_partition.current.partition}:iam::${data.aws_caller_identity.current.account_id}:oidc-provider/${local.eks_oidc_issuer_url}"
|
||||
tags = var.tags
|
||||
irsa_iam_role_path = var.irsa_iam_role_path
|
||||
irsa_iam_permissions_boundary = var.irsa_iam_permissions_boundary
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,95 @@
|
||||
resource "helm_release" "kube_state_metrics" {
|
||||
count = var.enable_kube_state_metrics ? 1 : 0
|
||||
chart = var.ksm_config.helm_chart_name
|
||||
create_namespace = var.ksm_config.create_namespace
|
||||
namespace = var.ksm_config.k8s_namespace
|
||||
name = var.ksm_config.helm_release_name
|
||||
version = var.ksm_config.helm_chart_version
|
||||
repository = var.ksm_config.helm_repo_url
|
||||
|
||||
dynamic "set" {
|
||||
for_each = var.ksm_config.helm_settings
|
||||
content {
|
||||
name = set.key
|
||||
value = set.value
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
resource "helm_release" "prometheus_node_exporter" {
|
||||
count = var.enable_node_exporter ? 1 : 0
|
||||
chart = var.ne_config.helm_chart_name
|
||||
create_namespace = var.ne_config.create_namespace
|
||||
namespace = var.ne_config.k8s_namespace
|
||||
name = var.ne_config.helm_release_name
|
||||
version = var.ne_config.helm_chart_version
|
||||
repository = var.ne_config.helm_repo_url
|
||||
|
||||
dynamic "set" {
|
||||
for_each = var.ne_config.helm_settings
|
||||
content {
|
||||
name = set.key
|
||||
value = set.value
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
module "helm_addon" {
|
||||
source = "github.com/aws-ia/terraform-aws-eks-blueprints/modules/kubernetes-addons/helm-addon"
|
||||
|
||||
helm_config = merge(
|
||||
{
|
||||
name = local.name
|
||||
chart = "${path.module}/otel-config"
|
||||
version = "0.2.0"
|
||||
namespace = local.namespace
|
||||
description = "ADOT helm Chart deployment configuration"
|
||||
},
|
||||
var.helm_config
|
||||
)
|
||||
|
||||
set_values = [
|
||||
{
|
||||
name = "ampurl"
|
||||
value = "${var.managed_prometheus_workspace_endpoint}api/v1/remote_write"
|
||||
},
|
||||
{
|
||||
name = "region"
|
||||
value = var.managed_prometheus_workspace_region
|
||||
},
|
||||
{
|
||||
name = "prometheusMetricsEndpoint"
|
||||
value = "metrics"
|
||||
},
|
||||
{
|
||||
name = "prometheusMetricsPort"
|
||||
value = 8888
|
||||
},
|
||||
{
|
||||
name = "scrapeInterval"
|
||||
value = "15s"
|
||||
},
|
||||
{
|
||||
name = "scrapeTimeout"
|
||||
value = "10s"
|
||||
},
|
||||
{
|
||||
name = "scrapeSampleLimit"
|
||||
value = 1000
|
||||
},
|
||||
{
|
||||
name = "ekscluster"
|
||||
value = local.context.eks_cluster_id
|
||||
},
|
||||
]
|
||||
|
||||
irsa_config = {
|
||||
create_kubernetes_namespace = true
|
||||
kubernetes_namespace = local.namespace
|
||||
create_kubernetes_service_account = true
|
||||
kubernetes_service_account = try(var.helm_config.service_account, local.name)
|
||||
irsa_iam_policies = ["arn:${data.aws_partition.current.partition}:iam::aws:policy/AmazonPrometheusRemoteWriteAccess"]
|
||||
}
|
||||
|
||||
addon_context = local.context
|
||||
}
|
||||
@@ -0,0 +1,6 @@
|
||||
apiVersion: v2
|
||||
name: opentelemetry
|
||||
description: A Helm chart to install otel operator
|
||||
type: application
|
||||
version: 0.2.0
|
||||
appVersion: v0.1.0
|
||||
@@ -0,0 +1,29 @@
|
||||
apiVersion: rbac.authorization.k8s.io/v1
|
||||
kind: ClusterRole
|
||||
metadata:
|
||||
name: otel-prometheus-role
|
||||
rules:
|
||||
- apiGroups:
|
||||
- ""
|
||||
resources:
|
||||
- nodes
|
||||
- nodes/proxy
|
||||
- services
|
||||
- endpoints
|
||||
- pods
|
||||
verbs:
|
||||
- get
|
||||
- list
|
||||
- watch
|
||||
- apiGroups:
|
||||
- extensions
|
||||
resources:
|
||||
- ingresses
|
||||
verbs:
|
||||
- get
|
||||
- list
|
||||
- watch
|
||||
- nonResourceURLs:
|
||||
- /metrics
|
||||
verbs:
|
||||
- get
|
||||
@@ -0,0 +1,12 @@
|
||||
apiVersion: rbac.authorization.k8s.io/v1
|
||||
kind: ClusterRoleBinding
|
||||
metadata:
|
||||
name: otel-prometheus-role-binding
|
||||
roleRef:
|
||||
apiGroup: rbac.authorization.k8s.io
|
||||
kind: ClusterRole
|
||||
name: otel-prometheus-role
|
||||
subjects:
|
||||
- kind: ServiceAccount
|
||||
name: adot-collector-kubeprometheus
|
||||
namespace: adot-collector-kubeprometheus
|
||||
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,8 @@
|
||||
ampurl: ${amp_url}
|
||||
region: ${region}
|
||||
prometheusMetricsEndpoint: ${prometheus_metrics_endpoint}
|
||||
prometheusMetricsPort: ${prometheus_metrics_port}
|
||||
scrapeInterval: ${scrape_interval}
|
||||
scrapeTimeout: ${scrape_timeout}
|
||||
scrapeSampleLimit: ${scrape_sample_limit}
|
||||
ekscluster: ${eks_cluster}
|
||||
@@ -0,0 +1,242 @@
|
||||
# Prioritize recording rules over alerting rules for limits (10)
|
||||
|
||||
################################################################################################################################################
|
||||
# Recording rules ##############################################################################################################################
|
||||
################################################################################################################################################
|
||||
|
||||
resource "aws_prometheus_rule_group_namespace" "recording_rules" {
|
||||
count = var.enable_recording_rules ? 1 : 0
|
||||
name = "accelerator-infra-rules"
|
||||
workspace_id = var.managed_prometheus_workspace_id
|
||||
data = <<EOF
|
||||
groups:
|
||||
- name: infra-rules-01
|
||||
rules:
|
||||
- record: "node_namespace_pod:kube_pod_info:"
|
||||
expr: topk by(cluster, namespace, pod) (1, max by(cluster, node, namespace, pod) (label_replace(kube_pod_info{job="kube-state-metrics",node!=""}, "pod", "$1", "pod", "(.*)")))
|
||||
- record: node:node_num_cpu:sum
|
||||
expr: count by(cluster, node) (sum by(node, cpu) (node_cpu_seconds_total{job="node-exporter"} * on(namespace, pod) group_left(node) topk by(namespace, pod) (1, node_namespace_pod:kube_pod_info:)))
|
||||
- record: :node_memory_MemAvailable_bytes:sum
|
||||
expr: sum by(cluster) (node_memory_MemAvailable_bytes{job="node-exporter"} or (node_memory_Buffers_bytes{job="node-exporter"} + node_memory_Cached_bytes{job="node-exporter"} + node_memory_MemFree_bytes{job="node-exporter"} + node_memory_Slab_bytes{job="node-exporter"}))
|
||||
- record: cluster:node_cpu:ratio_rate5m
|
||||
expr: sum by (cluster) (rate(node_cpu_seconds_total{job="node-exporter",mode!="idle",mode!="iowait",mode!="steal"}[5m])) / count by (cluster) (sum by(cluster, instance, cpu) (node_cpu_seconds_total{job="node-exporter"}))
|
||||
- record: node_quantile:kubelet_pleg_relist_duration_seconds:histogram_quantile
|
||||
expr: histogram_quantile(0.99, sum by(cluster, instance, le) (rate(kubelet_pleg_relist_duration_seconds_bucket[5m])) * on(cluster, instance) group_left(node) kubelet_node_name{job="kubelet"})
|
||||
labels:
|
||||
quantile: 0.99
|
||||
- record: node_quantile:kubelet_pleg_relist_duration_seconds:histogram_quantile
|
||||
expr: histogram_quantile(0.9, sum by(cluster, instance, le) (rate(kubelet_pleg_relist_duration_seconds_bucket[5m])) * on(cluster, instance) group_left(node) kubelet_node_name{job="kubelet"})
|
||||
labels:
|
||||
quantile: 0.9
|
||||
- record: node_quantile:kubelet_pleg_relist_duration_seconds:histogram_quantile
|
||||
expr: histogram_quantile(0.5, sum by(cluster, instance, le) (rate(kubelet_pleg_relist_duration_seconds_bucket[5m])) * on(cluster, instance) group_left(node) kubelet_node_name{job="kubelet"})
|
||||
labels:
|
||||
quantile: 0.5
|
||||
- record: instance:node_num_cpu:sum
|
||||
expr: count without(cpu, mode) (node_cpu_seconds_total{job="node-exporter",mode="idle"})
|
||||
- record: instance:node_cpu_utilisation:rate5m
|
||||
expr: 1 - avg without(cpu) (sum without(mode) (rate(node_cpu_seconds_total{job="node-exporter",mode=~"idle|iowait|steal"}[5m])))
|
||||
- record: instance:node_load1_per_cpu:ratio
|
||||
expr: (node_load1{job="node-exporter"} / instance:node_num_cpu:sum{job="node-exporter"})
|
||||
- record: instance:node_memory_utilisation:ratio
|
||||
expr: 1 - ((node_memory_MemAvailable_bytes{job="node-exporter"} or (node_memory_Buffers_bytes{job="node-exporter"} + node_memory_Cached_bytes{job="node-exporter"} + node_memory_MemFree_bytes{job="node-exporter"} + node_memory_Slab_bytes{job="node-exporter"})) / node_memory_MemTotal_bytes{job="node-exporter"})
|
||||
- record: instance:node_vmstat_pgmajfault:rate5m
|
||||
expr: rate(node_vmstat_pgmajfault{job="node-exporter"}[5m])
|
||||
- record: instance_device:node_disk_io_time_seconds:rate5m
|
||||
expr: rate(node_disk_io_time_seconds_total{device=~"mmcblk.p.+|.*nvme.+|rbd.+|sd.+|vd.+|xvd.+|dm-.+|dasd.+",job="node-exporter"}[5m])
|
||||
- record: instance_device:node_disk_io_time_weighted_seconds:rate5m
|
||||
expr: rate(node_disk_io_time_weighted_seconds_total{device=~"mmcblk.p.+|.*nvme.+|rbd.+|sd.+|vd.+|xvd.+|dm-.+|dasd.+",job="node-exporter"}[5m])
|
||||
- record: instance:node_network_receive_bytes_excluding_lo:rate5m
|
||||
expr: sum without(device) (rate(node_network_receive_bytes_total{device!="lo",job="node-exporter"}[5m]))
|
||||
- record: instance:node_network_transmit_bytes_excluding_lo:rate5m
|
||||
expr: sum without(device) (rate(node_network_transmit_bytes_total{device!="lo",job="node-exporter"}[5m]))
|
||||
- record: instance:node_network_receive_drop_excluding_lo:rate5m
|
||||
expr: sum without(device) (rate(node_network_receive_drop_total{device!="lo",job="node-exporter"}[5m]))
|
||||
- record: instance:node_network_transmit_drop_excluding_lo:rate5m
|
||||
expr: sum without(device) (rate(node_network_transmit_drop_total{device!="lo",job="node-exporter"}[5m]))
|
||||
- record: cluster_quantile:scheduler_e2e_scheduling_duration_seconds:histogram_quantile
|
||||
expr: histogram_quantile(0.99, sum without(instance, pod) (rate(scheduler_e2e_scheduling_duration_seconds_bucket{job="kube-scheduler"}[5m])))
|
||||
labels:
|
||||
quantile: 0.99
|
||||
- record: cluster_quantile:scheduler_scheduling_algorithm_duration_seconds:histogram_quantile
|
||||
expr: histogram_quantile(0.99, sum without(instance, pod) (rate(scheduler_scheduling_algorithm_duration_seconds_bucket{job="kube-scheduler"}[5m])))
|
||||
labels:
|
||||
quantile: 0.99
|
||||
- name: infra-rules-02
|
||||
rules:
|
||||
- record: cluster_quantile:scheduler_binding_duration_seconds:histogram_quantile
|
||||
expr: histogram_quantile(0.99, sum without(instance, pod) (rate(scheduler_binding_duration_seconds_bucket{job="kube-scheduler"}[5m])))
|
||||
labels:
|
||||
quantile: 0.99
|
||||
- record: cluster_quantile:scheduler_e2e_scheduling_duration_seconds:histogram_quantile
|
||||
expr: histogram_quantile(0.9, sum without(instance, pod) (rate(scheduler_e2e_scheduling_duration_seconds_bucket{job="kube-scheduler"}[5m])))
|
||||
labels:
|
||||
quantile: 0.9
|
||||
- record: cluster_quantile:scheduler_scheduling_algorithm_duration_seconds:histogram_quantile
|
||||
expr: histogram_quantile(0.9, sum without(instance, pod) (rate(scheduler_scheduling_algorithm_duration_seconds_bucket{job="kube-scheduler"}[5m])))
|
||||
labels:
|
||||
quantile: 0.9
|
||||
- record: cluster_quantile:scheduler_binding_duration_seconds:histogram_quantile
|
||||
expr: histogram_quantile(0.9, sum without(instance, pod) (rate(scheduler_binding_duration_seconds_bucket{job="kube-scheduler"}[5m])))
|
||||
labels:
|
||||
quantile: 0.9
|
||||
- record: cluster_quantile:scheduler_e2e_scheduling_duration_seconds:histogram_quantile
|
||||
expr: histogram_quantile(0.5, sum without(instance, pod) (rate(scheduler_e2e_scheduling_duration_seconds_bucket{job="kube-scheduler"}[5m])))
|
||||
labels:
|
||||
quantile: 0.5
|
||||
- record: cluster_quantile:scheduler_scheduling_algorithm_duration_seconds:histogram_quantile
|
||||
expr: histogram_quantile(0.5, sum without(instance, pod) (rate(scheduler_scheduling_algorithm_duration_seconds_bucket{job="kube-scheduler"}[5m])))
|
||||
labels:
|
||||
quantile: 0.5
|
||||
- record: cluster_quantile:scheduler_binding_duration_seconds:histogram_quantile
|
||||
expr: histogram_quantile(0.5, sum without(instance, pod) (rate(scheduler_binding_duration_seconds_bucket{job="kube-scheduler"}[5m])))
|
||||
labels:
|
||||
quantile: 0.5
|
||||
- record: instance:node_cpu:rate:sum
|
||||
expr: sum by(instance) (rate(node_cpu_seconds_total{mode!="idle",mode!="iowait",mode!="steal"}[3m]))
|
||||
- record: instance:node_network_receive_bytes:rate:sum
|
||||
expr: sum by(instance) (rate(node_network_receive_bytes_total[3m]))
|
||||
- record: instance:node_network_transmit_bytes:rate:sum
|
||||
expr: sum by(instance) (rate(node_network_transmit_bytes_total[3m]))
|
||||
- record: instance:node_cpu:ratio
|
||||
expr: sum without(cpu, mode) (rate(node_cpu_seconds_total{mode!="idle",mode!="iowait",mode!="steal"}[5m])) / on(instance) group_left() count by(instance) (sum by(instance, cpu) (node_cpu_seconds_total))
|
||||
- record: cluster:node_cpu:sum_rate5m
|
||||
expr: sum(rate(node_cpu_seconds_total{mode!="idle",mode!="iowait",mode!="steal"}[5m]))
|
||||
- record: cluster:node_cpu:ratio
|
||||
expr: cluster:node_cpu:sum_rate5m / count(sum by(instance, cpu) (node_cpu_seconds_total))
|
||||
- record: count:up1
|
||||
expr: count without(instance, pod, node) (up == 1)
|
||||
- record: count:up0
|
||||
expr: count without(instance, pod, node) (up == 0)
|
||||
- record: cluster_quantile:apiserver_request_slo_duration_seconds:histogram_quantile
|
||||
expr: histogram_quantile(0.99, sum by(cluster, le, resource) (rate(apiserver_request_slo_duration_seconds_bucket{job="apiserver",subresource!~"proxy|attach|log|exec|portforward",verb=~"LIST|GET"}[5m]))) > 0
|
||||
labels:
|
||||
quantile: 0.99
|
||||
verb: read
|
||||
- record: cluster_quantile:apiserver_request_slo_duration_seconds:histogram_quantile
|
||||
expr: histogram_quantile(0.99, sum by(cluster, le, resource) (rate(apiserver_request_slo_duration_seconds_bucket{job="apiserver",subresource!~"proxy|attach|log|exec|portforward",verb=~"POST|PUT|PATCH|DELETE"}[5m]))) > 0
|
||||
labels:
|
||||
quantile: 0.99
|
||||
verb: write
|
||||
- record: apiserver_request:burnrate1d
|
||||
expr: ((sum by(cluster) (rate(apiserver_request_slo_duration_seconds_count{job="apiserver",subresource!~"proxy|attach|log|exec|portforward",verb=~"LIST|GET"}[1d])) - ((sum by(cluster) (rate(apiserver_request_slo_duration_seconds_bucket{job="apiserver",le="1",scope=~"resource|",subresource!~"proxy|attach|log|exec|portforward",verb=~"LIST|GET"}[1d])) or vector(0)) + sum by(cluster) (rate(apiserver_request_slo_duration_seconds_bucket{job="apiserver",le="5",scope="namespace",subresource!~"proxy|attach|log|exec|portforward",verb=~"LIST|GET"}[1d])) + sum by(cluster) (rate(apiserver_request_slo_duration_seconds_bucket{job="apiserver",le="30",scope="cluster",subresource!~"proxy|attach|log|exec|portforward",verb=~"LIST|GET"}[1d])))) + sum by(cluster) (rate(apiserver_request_total{code=~"5..",job="apiserver",verb=~"LIST|GET"}[1d]))) / sum by(cluster) (rate(apiserver_request_total{job="apiserver",verb=~"LIST|GET"}[1d]))
|
||||
labels:
|
||||
verb: read
|
||||
- record: apiserver_request:burnrate1h
|
||||
expr: ((sum by(cluster) (rate(apiserver_request_slo_duration_seconds_count{job="apiserver",subresource!~"proxy|attach|log|exec|portforward",verb=~"LIST|GET"}[1h])) - ((sum by(cluster) (rate(apiserver_request_slo_duration_seconds_bucket{job="apiserver",le="1",scope=~"resource|",subresource!~"proxy|attach|log|exec|portforward",verb=~"LIST|GET"}[1h])) or vector(0)) + sum by(cluster) (rate(apiserver_request_slo_duration_seconds_bucket{job="apiserver",le="5",scope="namespace",subresource!~"proxy|attach|log|exec|portforward",verb=~"LIST|GET"}[1h])) + sum by(cluster) (rate(apiserver_request_slo_duration_seconds_bucket{job="apiserver",le="30",scope="cluster",subresource!~"proxy|attach|log|exec|portforward",verb=~"LIST|GET"}[1h])))) + sum by(cluster) (rate(apiserver_request_total{code=~"5..",job="apiserver",verb=~"LIST|GET"}[1h]))) / sum by(cluster) (rate(apiserver_request_total{job="apiserver",verb=~"LIST|GET"}[1h]))
|
||||
labels:
|
||||
verb: read
|
||||
- record: apiserver_request:burnrate2h
|
||||
expr: ((sum by(cluster) (rate(apiserver_request_slo_duration_seconds_count{job="apiserver",subresource!~"proxy|attach|log|exec|portforward",verb=~"LIST|GET"}[2h])) - ((sum by(cluster) (rate(apiserver_request_slo_duration_seconds_bucket{job="apiserver",le="1",scope=~"resource|",subresource!~"proxy|attach|log|exec|portforward",verb=~"LIST|GET"}[2h])) or vector(0)) + sum by(cluster) (rate(apiserver_request_slo_duration_seconds_bucket{job="apiserver",le="5",scope="namespace",subresource!~"proxy|attach|log|exec|portforward",verb=~"LIST|GET"}[2h])) + sum by(cluster) (rate(apiserver_request_slo_duration_seconds_bucket{job="apiserver",le="30",scope="cluster",subresource!~"proxy|attach|log|exec|portforward",verb=~"LIST|GET"}[2h])))) + sum by(cluster) (rate(apiserver_request_total{code=~"5..",job="apiserver",verb=~"LIST|GET"}[2h]))) / sum by(cluster) (rate(apiserver_request_total{job="apiserver",verb=~"LIST|GET"}[2h]))
|
||||
labels:
|
||||
verb: read
|
||||
- name: infra-rules-03
|
||||
rules:
|
||||
- record: apiserver_request:burnrate30m
|
||||
expr: ((sum by(cluster) (rate(apiserver_request_slo_duration_seconds_count{job="apiserver",subresource!~"proxy|attach|log|exec|portforward",verb=~"LIST|GET"}[30m])) - ((sum by(cluster) (rate(apiserver_request_slo_duration_seconds_bucket{job="apiserver",le="1",scope=~"resource|",subresource!~"proxy|attach|log|exec|portforward",verb=~"LIST|GET"}[30m])) or vector(0)) + sum by(cluster) (rate(apiserver_request_slo_duration_seconds_bucket{job="apiserver",le="5",scope="namespace",subresource!~"proxy|attach|log|exec|portforward",verb=~"LIST|GET"}[30m])) + sum by(cluster) (rate(apiserver_request_slo_duration_seconds_bucket{job="apiserver",le="30",scope="cluster",subresource!~"proxy|attach|log|exec|portforward",verb=~"LIST|GET"}[30m])))) + sum by(cluster) (rate(apiserver_request_total{code=~"5..",job="apiserver",verb=~"LIST|GET"}[30m]))) / sum by(cluster) (rate(apiserver_request_total{job="apiserver",verb=~"LIST|GET"}[30m]))
|
||||
labels:
|
||||
verb: read
|
||||
- record: apiserver_request:burnrate3d
|
||||
expr: ((sum by(cluster) (rate(apiserver_request_slo_duration_seconds_count{job="apiserver",subresource!~"proxy|attach|log|exec|portforward",verb=~"LIST|GET"}[3d])) - ((sum by(cluster) (rate(apiserver_request_slo_duration_seconds_bucket{job="apiserver",le="1",scope=~"resource|",subresource!~"proxy|attach|log|exec|portforward",verb=~"LIST|GET"}[3d])) or vector(0)) + sum by(cluster) (rate(apiserver_request_slo_duration_seconds_bucket{job="apiserver",le="5",scope="namespace",subresource!~"proxy|attach|log|exec|portforward",verb=~"LIST|GET"}[3d])) + sum by(cluster) (rate(apiserver_request_slo_duration_seconds_bucket{job="apiserver",le="30",scope="cluster",subresource!~"proxy|attach|log|exec|portforward",verb=~"LIST|GET"}[3d])))) + sum by(cluster) (rate(apiserver_request_total{code=~"5..",job="apiserver",verb=~"LIST|GET"}[3d]))) / sum by(cluster) (rate(apiserver_request_total{job="apiserver",verb=~"LIST|GET"}[3d]))
|
||||
labels:
|
||||
verb: read
|
||||
- record: apiserver_request:burnrate5m
|
||||
expr: ((sum by(cluster) (rate(apiserver_request_slo_duration_seconds_count{job="apiserver",subresource!~"proxy|attach|log|exec|portforward",verb=~"LIST|GET"}[5m])) - ((sum by(cluster) (rate(apiserver_request_slo_duration_seconds_bucket{job="apiserver",le="1",scope=~"resource|",subresource!~"proxy|attach|log|exec|portforward",verb=~"LIST|GET"}[5m])) or vector(0)) + sum by(cluster) (rate(apiserver_request_slo_duration_seconds_bucket{job="apiserver",le="5",scope="namespace",subresource!~"proxy|attach|log|exec|portforward",verb=~"LIST|GET"}[5m])) + sum by(cluster) (rate(apiserver_request_slo_duration_seconds_bucket{job="apiserver",le="30",scope="cluster",subresource!~"proxy|attach|log|exec|portforward",verb=~"LIST|GET"}[5m])))) + sum by(cluster) (rate(apiserver_request_total{code=~"5..",job="apiserver",verb=~"LIST|GET"}[5m]))) / sum by(cluster) (rate(apiserver_request_total{job="apiserver",verb=~"LIST|GET"}[5m]))
|
||||
labels:
|
||||
verb: read
|
||||
- record: apiserver_request:burnrate6h
|
||||
expr: ((sum by(cluster) (rate(apiserver_request_slo_duration_seconds_count{job="apiserver",subresource!~"proxy|attach|log|exec|portforward",verb=~"LIST|GET"}[6h])) - ((sum by(cluster) (rate(apiserver_request_slo_duration_seconds_bucket{job="apiserver",le="1",scope=~"resource|",subresource!~"proxy|attach|log|exec|portforward",verb=~"LIST|GET"}[6h])) or vector(0)) + sum by(cluster) (rate(apiserver_request_slo_duration_seconds_bucket{job="apiserver",le="5",scope="namespace",subresource!~"proxy|attach|log|exec|portforward",verb=~"LIST|GET"}[6h])) + sum by(cluster) (rate(apiserver_request_slo_duration_seconds_bucket{job="apiserver",le="30",scope="cluster",subresource!~"proxy|attach|log|exec|portforward",verb=~"LIST|GET"}[6h])))) + sum by(cluster) (rate(apiserver_request_total{code=~"5..",job="apiserver",verb=~"LIST|GET"}[6h]))) / sum by(cluster) (rate(apiserver_request_total{job="apiserver",verb=~"LIST|GET"}[6h]))
|
||||
labels:
|
||||
verb: read
|
||||
- record: apiserver_request:burnrate1d
|
||||
expr: ((sum by(cluster) (rate(apiserver_request_slo_duration_seconds_count{job="apiserver",subresource!~"proxy|attach|log|exec|portforward",verb=~"POST|PUT|PATCH|DELETE"}[1d])) - sum by(cluster) (rate(apiserver_request_slo_duration_seconds_bucket{job="apiserver",le="1",subresource!~"proxy|attach|log|exec|portforward",verb=~"POST|PUT|PATCH|DELETE"}[1d]))) + sum by(cluster) (rate(apiserver_request_total{code=~"5..",job="apiserver",verb=~"POST|PUT|PATCH|DELETE"}[1d]))) / sum by(cluster) (rate(apiserver_request_total{job="apiserver",verb=~"POST|PUT|PATCH|DELETE"}[1d]))
|
||||
labels:
|
||||
verb: read
|
||||
- record: apiserver_request:burnrate1d
|
||||
expr: ((sum by(cluster) (rate(apiserver_request_slo_duration_seconds_count{job="apiserver",subresource!~"proxy|attach|log|exec|portforward",verb=~"LIST|GET"}[1d])) - ((sum by(cluster) (rate(apiserver_request_slo_duration_seconds_bucket{job="apiserver",le="1",scope=~"resource|",subresource!~"proxy|attach|log|exec|portforward",verb=~"LIST|GET"}[1d])) or vector(0)) + sum by(cluster) (rate(apiserver_request_slo_duration_seconds_bucket{job="apiserver",le="5",scope="namespace",subresource!~"proxy|attach|log|exec|portforward",verb=~"LIST|GET"}[1d])) + sum by(cluster) (rate(apiserver_request_slo_duration_seconds_bucket{job="apiserver",le="30",scope="cluster",subresource!~"proxy|attach|log|exec|portforward",verb=~"LIST|GET"}[1d])))) + sum by(cluster) (rate(apiserver_request_total{code=~"5..",job="apiserver",verb=~"LIST|GET"}[1d]))) / sum by(cluster) (rate(apiserver_request_total{job="apiserver",verb=~"LIST|GET"}[1d]))
|
||||
labels:
|
||||
verb: write
|
||||
- record: apiserver_request:burnrate1h
|
||||
expr: ((sum by(cluster) (rate(apiserver_request_slo_duration_seconds_count{job="apiserver",subresource!~"proxy|attach|log|exec|portforward",verb=~"POST|PUT|PATCH|DELETE"}[1h])) - sum by(cluster) (rate(apiserver_request_slo_duration_seconds_bucket{job="apiserver",le="1",subresource!~"proxy|attach|log|exec|portforward",verb=~"POST|PUT|PATCH|DELETE"}[1h]))) + sum by(cluster) (rate(apiserver_request_total{code=~"5..",job="apiserver",verb=~"POST|PUT|PATCH|DELETE"}[1h]))) / sum by(cluster) (rate(apiserver_request_total{job="apiserver",verb=~"POST|PUT|PATCH|DELETE"}[1h]))
|
||||
labels:
|
||||
verb: write
|
||||
- record: apiserver_request:burnrate2h
|
||||
expr: ((sum by(cluster) (rate(apiserver_request_slo_duration_seconds_count{job="apiserver",subresource!~"proxy|attach|log|exec|portforward",verb=~"POST|PUT|PATCH|DELETE"}[2h])) - sum by(cluster) (rate(apiserver_request_slo_duration_seconds_bucket{job="apiserver",le="1",subresource!~"proxy|attach|log|exec|portforward",verb=~"POST|PUT|PATCH|DELETE"}[2h]))) + sum by(cluster) (rate(apiserver_request_total{code=~"5..",job="apiserver",verb=~"POST|PUT|PATCH|DELETE"}[2h]))) / sum by(cluster) (rate(apiserver_request_total{job="apiserver",verb=~"POST|PUT|PATCH|DELETE"}[2h]))
|
||||
labels:
|
||||
verb: write
|
||||
- record: apiserver_request:burnrate30m
|
||||
expr: ((sum by(cluster) (rate(apiserver_request_slo_duration_seconds_count{job="apiserver",subresource!~"proxy|attach|log|exec|portforward",verb=~"POST|PUT|PATCH|DELETE"}[30m])) - sum by(cluster) (rate(apiserver_request_slo_duration_seconds_bucket{job="apiserver",le="1",subresource!~"proxy|attach|log|exec|portforward",verb=~"POST|PUT|PATCH|DELETE"}[30m]))) + sum by(cluster) (rate(apiserver_request_total{code=~"5..",job="apiserver",verb=~"POST|PUT|PATCH|DELETE"}[30m]))) / sum by(cluster) (rate(apiserver_request_total{job="apiserver",verb=~"POST|PUT|PATCH|DELETE"}[30m]))
|
||||
labels:
|
||||
verb: write
|
||||
- record: apiserver_request:burnrate3d
|
||||
expr: ((sum by(cluster) (rate(apiserver_request_slo_duration_seconds_count{job="apiserver",subresource!~"proxy|attach|log|exec|portforward",verb=~"POST|PUT|PATCH|DELETE"}[3d])) - sum by(cluster) (rate(apiserver_request_slo_duration_seconds_bucket{job="apiserver",le="1",subresource!~"proxy|attach|log|exec|portforward",verb=~"POST|PUT|PATCH|DELETE"}[3d]))) + sum by(cluster) (rate(apiserver_request_total{code=~"5..",job="apiserver",verb=~"POST|PUT|PATCH|DELETE"}[3d]))) / sum by(cluster) (rate(apiserver_request_total{job="apiserver",verb=~"POST|PUT|PATCH|DELETE"}[3d]))
|
||||
labels:
|
||||
verb: write
|
||||
- record: apiserver_request:burnrate5m
|
||||
expr: ((sum by(cluster) (rate(apiserver_request_slo_duration_seconds_count{job="apiserver",subresource!~"proxy|attach|log|exec|portforward",verb=~"POST|PUT|PATCH|DELETE"}[5m])) - sum by(cluster) (rate(apiserver_request_slo_duration_seconds_bucket{job="apiserver",le="1",subresource!~"proxy|attach|log|exec|portforward",verb=~"POST|PUT|PATCH|DELETE"}[5m]))) + sum by(cluster) (rate(apiserver_request_total{code=~"5..",job="apiserver",verb=~"POST|PUT|PATCH|DELETE"}[5m]))) / sum by(cluster) (rate(apiserver_request_total{job="apiserver",verb=~"POST|PUT|PATCH|DELETE"}[5m]))
|
||||
labels:
|
||||
verb: write
|
||||
- record: apiserver_request:burnrate6h
|
||||
expr: ((sum by(cluster) (rate(apiserver_request_slo_duration_seconds_count{job="apiserver",subresource!~"proxy|attach|log|exec|portforward",verb=~"POST|PUT|PATCH|DELETE"}[6h])) - sum by(cluster) (rate(apiserver_request_slo_duration_seconds_bucket{job="apiserver",le="1",subresource!~"proxy|attach|log|exec|portforward",verb=~"POST|PUT|PATCH|DELETE"}[6h]))) + sum by(cluster) (rate(apiserver_request_total{code=~"5..",job="apiserver",verb=~"POST|PUT|PATCH|DELETE"}[6h]))) / sum by(cluster) (rate(apiserver_request_total{job="apiserver",verb=~"POST|PUT|PATCH|DELETE"}[6h]))
|
||||
labels:
|
||||
verb: write
|
||||
- record: code_verb:apiserver_request_total:increase30d
|
||||
expr: avg_over_time(code_verb:apiserver_request_total:increase1h[30d]) * 24 * 30
|
||||
- record: code:apiserver_request_total:increase30d
|
||||
expr: sum by(cluster, code) (code_verb:apiserver_request_total:increase30d{verb=~"LIST|GET"})
|
||||
labels:
|
||||
verb: read
|
||||
- record: code:apiserver_request_total:increase30d
|
||||
expr: sum by(cluster, code) (code_verb:apiserver_request_total:increase30d{verb=~"POST|PUT|PATCH|DELETE"})
|
||||
labels:
|
||||
verb: write
|
||||
- record: cluster_verb_scope:apiserver_request_slo_duration_seconds_count:increase1h
|
||||
expr: sum by(cluster, verb, scope) (increase(apiserver_request_slo_duration_seconds_count[1h]))
|
||||
- record: cluster_verb_scope:apiserver_request_slo_duration_seconds_count:increase30d
|
||||
expr: sum by(cluster, verb, scope) (avg_over_time(cluster_verb_scope:apiserver_request_slo_duration_seconds_count:increase1h[30d]) * 24 * 30)
|
||||
- record: node_namespace_pod_container:container_cpu_usage_seconds_total:sum_irate
|
||||
expr: sum by(cluster, namespace, pod, container) (irate(container_cpu_usage_seconds_total{image!="",job="kubelet"}[5m])) * on(cluster, namespace, pod) group_left(node) topk by(cluster, namespace, pod) (1, max by(cluster, namespace, pod, node) (kube_pod_info{node!=""}))
|
||||
- record: node_namespace_pod_container:container_memory_working_set_bytes
|
||||
expr: container_memory_working_set_bytes{image!="",job="kubelet"} * on(namespace, pod) group_left(node) topk by(namespace, pod) (1, max by(namespace, pod, node) (kube_pod_info{node!=""}))
|
||||
- record: node_namespace_pod_container:container_memory_rss
|
||||
expr: container_memory_rss{image!="",job="kubelet"} * on(namespace, pod) group_left(node) topk by(namespace, pod) (1, max by(namespace, pod, node) (kube_pod_info{node!=""}))
|
||||
- name: infra-rules-04
|
||||
rules:
|
||||
- record: node_namespace_pod_container:container_memory_cache
|
||||
expr: container_memory_cache{image!="",job="kubelet"} * on(namespace, pod) group_left(node) topk by(namespace, pod) (1, max by(namespace, pod, node) (kube_pod_info{node!=""}))
|
||||
- record: node_namespace_pod_container:container_memory_swap
|
||||
expr: container_memory_swap{image!="",job="kubelet"} * on(namespace, pod) group_left(node) topk by(namespace, pod) (1, max by(namespace, pod, node) (kube_pod_info{node!=""}))
|
||||
- record: cluster:namespace:pod_memory:active:kube_pod_container_resource_requests
|
||||
expr: kube_pod_container_resource_requests{job="kube-state-metrics",resource="memory"} * on(namespace, pod, cluster) group_left() max by(namespace, pod, cluster) ((kube_pod_status_phase{phase=~"Pending|Running"} == 1))
|
||||
- record: namespace_memory:kube_pod_container_resource_requests:sum
|
||||
expr: sum by(namespace, cluster) (sum by(namespace, pod, cluster) (max by(namespace, pod, container, cluster) (kube_pod_container_resource_requests{job="kube-state-metrics",resource="memory"}) * on(namespace, pod, cluster) group_left() max by(namespace, pod, cluster) (kube_pod_status_phase{phase=~"Pending|Running"} == 1)))
|
||||
- record: cluster:namespace:pod_cpu:active:kube_pod_container_resource_requests
|
||||
expr: kube_pod_container_resource_requests{job="kube-state-metrics",resource="cpu"} * on(namespace, pod, cluster) group_left() max by(namespace, pod, cluster) ((kube_pod_status_phase{phase=~"Pending|Running"} == 1))
|
||||
- record: namespace_cpu:kube_pod_container_resource_requests:sum
|
||||
expr: sum by(namespace, cluster) (sum by(namespace, pod, cluster) (max by(namespace, pod, container, cluster) (kube_pod_container_resource_requests{job="kube-state-metrics",resource="cpu"}) * on(namespace, pod, cluster) group_left() max by(namespace, pod, cluster) (kube_pod_status_phase{phase=~"Pending|Running"} == 1)))
|
||||
- record: cluster:namespace:pod_memory:active:kube_pod_container_resource_limits
|
||||
expr: kube_pod_container_resource_limits{job="kube-state-metrics",resource="memory"} * on(namespace, pod, cluster) group_left() max by(namespace, pod, cluster) ((kube_pod_status_phase{phase=~"Pending|Running"} == 1))
|
||||
- record: namespace_memory:kube_pod_container_resource_limits:sum
|
||||
expr: sum by(namespace, cluster) (sum by(namespace, pod, cluster) (max by(namespace, pod, container, cluster) (kube_pod_container_resource_limits{job="kube-state-metrics",resource="memory"}) * on(namespace, pod, cluster) group_left() max by(namespace, pod, cluster) (kube_pod_status_phase{phase=~"Pending|Running"} == 1)))
|
||||
- record: cluster:namespace:pod_cpu:active:kube_pod_container_resource_limits
|
||||
expr: kube_pod_container_resource_limits{job="kube-state-metrics",resource="cpu"} * on(namespace, pod, cluster) group_left() max by(namespace, pod, cluster) ((kube_pod_status_phase{phase=~"Pending|Running"} == 1))
|
||||
- record: namespace_cpu:kube_pod_container_resource_limits:sum
|
||||
expr: sum by(namespace, cluster) (sum by(namespace, pod, cluster) (max by(namespace, pod, container, cluster) (kube_pod_container_resource_limits{job="kube-state-metrics",resource="cpu"}) * on(namespace, pod, cluster) group_left() max by(namespace, pod, cluster) (kube_pod_status_phase{phase=~"Pending|Running"} == 1)))
|
||||
- record: namespace_workload_pod:kube_pod_owner:relabel
|
||||
expr: max by(cluster, namespace, workload, pod) (label_replace(label_replace(kube_pod_owner{job="kube-state-metrics",owner_kind="ReplicaSet"}, "replicaset", "$1", "owner_name", "(.*)") * on(replicaset, namespace) group_left(owner_name) topk by(replicaset, namespace) (1, max by(replicaset, namespace, owner_name) (kube_replicaset_owner{job="kube-state-metrics"})), "workload", "$1", "owner_name", "(.*)"))
|
||||
labels:
|
||||
workload_type: deployment
|
||||
- record: namespace_workload_pod:kube_pod_owner:relabel
|
||||
expr: max by(cluster, namespace, workload, pod) (label_replace(kube_pod_owner{job="kube-state-metrics",owner_kind="DaemonSet"}, "workload", "$1", "owner_name", "(.*)"))
|
||||
labels:
|
||||
workload_type: daemonset
|
||||
- record: namespace_workload_pod:kube_pod_owner:relabel
|
||||
expr: max by(cluster, namespace, workload, pod) (label_replace(kube_pod_owner{job="kube-state-metrics",owner_kind="StatefulSet"}, "workload", "$1", "owner_name", "(.*)"))
|
||||
labels:
|
||||
workload_type: statefulset
|
||||
- record: namespace_workload_pod:kube_pod_owner:relabel
|
||||
expr: max by(cluster, namespace, workload, pod) (label_replace(kube_pod_owner{job="kube-state-metrics",owner_kind="Job"}, "workload", "$1", "owner_name", "(.*)"))
|
||||
labels:
|
||||
workload_type: job
|
||||
EOF
|
||||
}
|
||||
@@ -0,0 +1,131 @@
|
||||
variable "eks_cluster_id" {
|
||||
description = "EKS Cluster Id"
|
||||
type = string
|
||||
}
|
||||
|
||||
variable "helm_config" {
|
||||
description = "Helm Config for Prometheus"
|
||||
type = any
|
||||
default = {}
|
||||
}
|
||||
|
||||
variable "irsa_iam_role_path" {
|
||||
description = "IAM role path for IRSA roles"
|
||||
type = string
|
||||
default = "/"
|
||||
}
|
||||
|
||||
variable "irsa_iam_permissions_boundary" {
|
||||
description = "IAM permissions boundary for IRSA roles"
|
||||
type = string
|
||||
default = ""
|
||||
}
|
||||
|
||||
variable "managed_prometheus_workspace_endpoint" {
|
||||
description = "Amazon Managed Prometheus Workspace Endpoint"
|
||||
type = string
|
||||
default = null
|
||||
}
|
||||
variable "managed_prometheus_workspace_id" {
|
||||
description = "Amazon Managed Prometheus Workspace ID"
|
||||
type = string
|
||||
default = null
|
||||
}
|
||||
|
||||
variable "managed_prometheus_workspace_region" {
|
||||
description = "Amazon Managed Prometheus Workspace's Region"
|
||||
type = string
|
||||
default = null
|
||||
}
|
||||
|
||||
variable "dashboards_folder_id" {
|
||||
description = "Grafana folder ID for automatic dashboards"
|
||||
type = string
|
||||
}
|
||||
|
||||
variable "enable_recording_rules" {
|
||||
description = "Enables or disables Managed Prometheus recording rules. Disabling this might affect some data in the dashboards"
|
||||
type = bool
|
||||
default = true
|
||||
}
|
||||
|
||||
variable "enable_alerting_rules" {
|
||||
description = "Enables or disables Managed Prometheus alerting rules"
|
||||
type = bool
|
||||
default = true
|
||||
}
|
||||
|
||||
variable "enable_dashboards" {
|
||||
description = "Enables or disables curated dashboards"
|
||||
type = bool
|
||||
default = true
|
||||
}
|
||||
|
||||
variable "enable_kube_state_metrics" {
|
||||
description = "Enables or disables Kube State metrics exporter. Disabling this might affect some data in the dashboards"
|
||||
type = bool
|
||||
default = true
|
||||
}
|
||||
|
||||
variable "ksm_config" {
|
||||
description = "Kube State metrics configuration"
|
||||
type = object({
|
||||
create_namespace = bool
|
||||
k8s_namespace = string
|
||||
helm_chart_name = string
|
||||
helm_chart_version = string
|
||||
helm_release_name = string
|
||||
helm_repo_url = string
|
||||
helm_settings = map(string)
|
||||
helm_values = map(any)
|
||||
})
|
||||
|
||||
default = {
|
||||
create_namespace = true
|
||||
helm_chart_name = "kube-state-metrics"
|
||||
helm_chart_version = "4.16.0"
|
||||
helm_release_name = "kube-state-metrics"
|
||||
helm_repo_url = "https://prometheus-community.github.io/helm-charts"
|
||||
helm_settings = {}
|
||||
helm_values = {}
|
||||
k8s_namespace = "kube-system"
|
||||
}
|
||||
nullable = false
|
||||
}
|
||||
|
||||
variable "enable_node_exporter" {
|
||||
description = "Enables or disables Node exporter. Disabling this might affect some data in the dashboards"
|
||||
type = bool
|
||||
default = true
|
||||
}
|
||||
|
||||
variable "ne_config" {
|
||||
description = "Node exporter configuration"
|
||||
type = object({
|
||||
create_namespace = bool
|
||||
k8s_namespace = string
|
||||
helm_chart_name = string
|
||||
helm_chart_version = string
|
||||
helm_release_name = string
|
||||
helm_repo_url = string
|
||||
helm_settings = map(string)
|
||||
helm_values = map(any)
|
||||
})
|
||||
|
||||
default = {
|
||||
create_namespace = true
|
||||
helm_chart_name = "prometheus-node-exporter"
|
||||
helm_chart_version = "2.0.3"
|
||||
helm_release_name = "prometheus-node-exporter"
|
||||
helm_repo_url = "https://prometheus-community.github.io/helm-charts"
|
||||
helm_settings = {}
|
||||
helm_values = {}
|
||||
k8s_namespace = "prometheus-node-exporter"
|
||||
}
|
||||
nullable = false
|
||||
}
|
||||
variable "tags" {
|
||||
description = "Additional tags (e.g. `map('BusinessUnit`,`XYZ`)"
|
||||
type = map(string)
|
||||
default = {}
|
||||
}
|
||||
@@ -0,0 +1,25 @@
|
||||
terraform {
|
||||
required_version = ">= 1.0.0"
|
||||
required_providers {
|
||||
aws = {
|
||||
source = "hashicorp/aws"
|
||||
version = ">= 4.0.0"
|
||||
}
|
||||
kubernetes = {
|
||||
source = "hashicorp/kubernetes"
|
||||
version = ">= 2.10"
|
||||
}
|
||||
kubectl = {
|
||||
source = "gavinbunney/kubectl"
|
||||
version = ">= 1.14"
|
||||
}
|
||||
helm = {
|
||||
source = "hashicorp/helm"
|
||||
version = ">= 2.4.1"
|
||||
}
|
||||
grafana = {
|
||||
source = "grafana/grafana"
|
||||
version = ">= 1.25.0"
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,57 @@
|
||||
# Java based workloads monitoring
|
||||
|
||||
This module provides monitoring for Java based workloads with the following resources:
|
||||
|
||||
- AWS Distro For OpenTelemetry Operator and Collector
|
||||
- AWS Managed Grafana Dashboard and data source
|
||||
- Alerts and recording rules with AWS Managed Service for Prometheus
|
||||
|
||||
<!-- BEGINNING OF PRE-COMMIT-TERRAFORM DOCS HOOK -->
|
||||
## Requirements
|
||||
|
||||
| Name | Version |
|
||||
|------|---------|
|
||||
| <a name="requirement_terraform"></a> [terraform](#requirement\_terraform) | >= 1.0.0 |
|
||||
| <a name="requirement_aws"></a> [aws](#requirement\_aws) | >= 4.0.0 |
|
||||
| <a name="requirement_grafana"></a> [grafana](#requirement\_grafana) | >= 1.25.0 |
|
||||
| <a name="requirement_helm"></a> [helm](#requirement\_helm) | >= 2.4.1 |
|
||||
| <a name="requirement_kubectl"></a> [kubectl](#requirement\_kubectl) | >= 1.14 |
|
||||
| <a name="requirement_kubernetes"></a> [kubernetes](#requirement\_kubernetes) | >= 2.10 |
|
||||
|
||||
## Providers
|
||||
|
||||
| Name | Version |
|
||||
|------|---------|
|
||||
| <a name="provider_aws"></a> [aws](#provider\_aws) | >= 4.0.0 |
|
||||
| <a name="provider_grafana"></a> [grafana](#provider\_grafana) | >= 1.25.0 |
|
||||
|
||||
## Modules
|
||||
|
||||
| Name | Source | Version |
|
||||
|------|--------|---------|
|
||||
| <a name="module_helm_addon"></a> [helm\_addon](#module\_helm\_addon) | github.com/aws-ia/terraform-aws-eks-blueprints/modules/kubernetes-addons/helm-addon | n/a |
|
||||
|
||||
## Resources
|
||||
|
||||
| Name | Type |
|
||||
|------|------|
|
||||
| [aws_prometheus_rule_group_namespace.this](https://registry.terraform.io/providers/hashicorp/aws/latest/docs/resources/prometheus_rule_group_namespace) | resource |
|
||||
| [grafana_dashboard.this](https://registry.terraform.io/providers/grafana/grafana/latest/docs/resources/dashboard) | resource |
|
||||
| [aws_partition.current](https://registry.terraform.io/providers/hashicorp/aws/latest/docs/data-sources/partition) | data source |
|
||||
|
||||
## Inputs
|
||||
|
||||
| Name | Description | Type | Default | Required |
|
||||
|------|-------------|------|---------|:--------:|
|
||||
| <a name="input_addon_context"></a> [addon\_context](#input\_addon\_context) | Input configuration for the addon | <pre>object({<br> aws_caller_identity_account_id = string<br> aws_caller_identity_arn = string<br> aws_eks_cluster_endpoint = string<br> aws_partition_id = string<br> aws_region_name = string<br> eks_cluster_id = string<br> eks_oidc_issuer_url = string<br> eks_oidc_provider_arn = string<br> irsa_iam_permissions_boundary = string<br> irsa_iam_role_path = string<br> tags = map(string)<br> })</pre> | n/a | yes |
|
||||
| <a name="input_amp_endpoint"></a> [amp\_endpoint](#input\_amp\_endpoint) | Amazon Managed Prometheus endpoint | `string` | n/a | yes |
|
||||
| <a name="input_amp_id"></a> [amp\_id](#input\_amp\_id) | Managed Prometheus workspace id | `string` | n/a | yes |
|
||||
| <a name="input_amp_region"></a> [amp\_region](#input\_amp\_region) | Amazon Managed Prometheus Workspace's Region | `string` | `null` | no |
|
||||
| <a name="input_dashboards_folder_id"></a> [dashboards\_folder\_id](#input\_dashboards\_folder\_id) | Grafana folder ID for automatic dashboards | `string` | n/a | yes |
|
||||
| <a name="input_enable_recording_rules"></a> [enable\_recording\_rules](#input\_enable\_recording\_rules) | Enable AMP recording rules | `bool` | `true` | no |
|
||||
| <a name="input_helm_config"></a> [helm\_config](#input\_helm\_config) | Helm Config for Prometheus | `any` | `{}` | no |
|
||||
|
||||
## Outputs
|
||||
|
||||
No outputs.
|
||||
<!-- END OF PRE-COMMIT-TERRAFORM DOCS HOOK -->
|
||||
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,94 @@
|
||||
locals {
|
||||
name = "adot-collector-java"
|
||||
namespace = try(var.helm_config.namespace, local.name)
|
||||
}
|
||||
|
||||
data "aws_partition" "current" {}
|
||||
|
||||
# deploys collector
|
||||
module "helm_addon" {
|
||||
source = "github.com/aws-ia/terraform-aws-eks-blueprints/modules/kubernetes-addons/helm-addon"
|
||||
|
||||
|
||||
helm_config = merge(
|
||||
{
|
||||
name = local.name
|
||||
chart = "${path.module}/otel-config"
|
||||
version = "0.2.0"
|
||||
namespace = local.namespace
|
||||
description = "ADOT helm Chart deployment configuration"
|
||||
},
|
||||
var.helm_config
|
||||
)
|
||||
|
||||
set_values = [
|
||||
{
|
||||
name = "ampurl"
|
||||
value = "${var.amp_endpoint}api/v1/remote_write"
|
||||
},
|
||||
{
|
||||
name = "region"
|
||||
value = var.amp_region
|
||||
},
|
||||
{
|
||||
name = "prometheusMetricsEndpoint"
|
||||
value = "metrics"
|
||||
},
|
||||
{
|
||||
name = "prometheusMetricsPort"
|
||||
value = 8888
|
||||
},
|
||||
{
|
||||
name = "scrapeInterval"
|
||||
value = "15s"
|
||||
},
|
||||
{
|
||||
name = "scrapeTimeout"
|
||||
value = "10s"
|
||||
},
|
||||
{
|
||||
name = "scrapeSampleLimit"
|
||||
value = 1000
|
||||
}
|
||||
]
|
||||
|
||||
irsa_config = {
|
||||
create_kubernetes_namespace = try(var.helm_config["create_namespace"], true)
|
||||
kubernetes_namespace = local.namespace
|
||||
create_kubernetes_service_account = true
|
||||
kubernetes_service_account = try(var.helm_config.service_account, local.name)
|
||||
irsa_iam_policies = ["arn:${data.aws_partition.current.partition}:iam::aws:policy/AmazonPrometheusRemoteWriteAccess"]
|
||||
}
|
||||
|
||||
addon_context = var.addon_context
|
||||
}
|
||||
|
||||
|
||||
resource "aws_prometheus_rule_group_namespace" "this" {
|
||||
count = var.enable_recording_rules ? 1 : 0
|
||||
|
||||
name = "java_rules"
|
||||
workspace_id = var.amp_id
|
||||
data = <<EOF
|
||||
groups:
|
||||
- name: default-metric
|
||||
rules:
|
||||
- record: metric:recording_rule
|
||||
expr: avg(rate(container_cpu_usage_seconds_total[5m]))
|
||||
- name: default-alert
|
||||
rules:
|
||||
- alert: metric:alerting_rule
|
||||
expr: jvm_memory_bytes_used{job="java", area="heap"} / jvm_memory_bytes_max * 100 > 80
|
||||
for: 1m
|
||||
labels:
|
||||
severity: warning
|
||||
annotations:
|
||||
summary: "JVM heap warning"
|
||||
description: "JVM heap of instance `{{$labels.instance}}` from application `{{$labels.application}}` is above 80% for one minute. (current=`{{$value}}%`)"
|
||||
EOF
|
||||
}
|
||||
|
||||
resource "grafana_dashboard" "this" {
|
||||
folder = var.dashboards_folder_id
|
||||
config_json = file("${path.module}/dashboards/default.json")
|
||||
}
|
||||
@@ -0,0 +1,6 @@
|
||||
apiVersion: v2
|
||||
name: opentelemetry
|
||||
description: A Helm chart to install otel operator
|
||||
type: application
|
||||
version: 0.2.0
|
||||
appVersion: v0.1.0
|
||||
@@ -0,0 +1,29 @@
|
||||
apiVersion: rbac.authorization.k8s.io/v1
|
||||
kind: ClusterRole
|
||||
metadata:
|
||||
name: otel-prometheus-role
|
||||
rules:
|
||||
- apiGroups:
|
||||
- ""
|
||||
resources:
|
||||
- nodes
|
||||
- nodes/proxy
|
||||
- services
|
||||
- endpoints
|
||||
- pods
|
||||
verbs:
|
||||
- get
|
||||
- list
|
||||
- watch
|
||||
- apiGroups:
|
||||
- extensions
|
||||
resources:
|
||||
- ingresses
|
||||
verbs:
|
||||
- get
|
||||
- list
|
||||
- watch
|
||||
- nonResourceURLs:
|
||||
- /metrics
|
||||
verbs:
|
||||
- get
|
||||
@@ -0,0 +1,12 @@
|
||||
apiVersion: rbac.authorization.k8s.io/v1
|
||||
kind: ClusterRoleBinding
|
||||
metadata:
|
||||
name: otel-prometheus-role-binding
|
||||
roleRef:
|
||||
apiGroup: rbac.authorization.k8s.io
|
||||
kind: ClusterRole
|
||||
name: otel-prometheus-role
|
||||
subjects:
|
||||
- kind: ServiceAccount
|
||||
name: adot-collector-java
|
||||
namespace: adot-collector-java
|
||||
@@ -0,0 +1,67 @@
|
||||
apiVersion: opentelemetry.io/v1alpha1
|
||||
kind: OpenTelemetryCollector
|
||||
metadata:
|
||||
name: adot
|
||||
spec:
|
||||
image: public.ecr.aws/aws-observability/aws-otel-collector:latest
|
||||
mode: deployment
|
||||
serviceAccount: adot-collector-java
|
||||
config: |
|
||||
receivers:
|
||||
prometheus:
|
||||
config:
|
||||
global:
|
||||
scrape_interval: {{ .Values.scrapeInterval }}
|
||||
scrape_timeout: {{ .Values.scrapeTimeout }}
|
||||
|
||||
scrape_configs:
|
||||
- job_name: 'kubernetes-pod-jmx'
|
||||
sample_limit: {{ .Values.scrapeSampleLimit }}
|
||||
metrics_path: /{{ .Values.prometheusMetricsEndpoint }}
|
||||
kubernetes_sd_configs:
|
||||
- role: pod
|
||||
relabel_configs:
|
||||
- source_labels: [ __address__ ]
|
||||
action: keep
|
||||
regex: '.*:9404$'
|
||||
- action: labelmap
|
||||
regex: __meta_kubernetes_pod_label_(.+)
|
||||
- action: replace
|
||||
source_labels: [ __meta_kubernetes_namespace ]
|
||||
target_label: Namespace
|
||||
- source_labels: [ __meta_kubernetes_pod_name ]
|
||||
action: replace
|
||||
target_label: pod_name
|
||||
- action: replace
|
||||
source_labels: [ __meta_kubernetes_pod_container_name ]
|
||||
target_label: container_name
|
||||
- action: replace
|
||||
source_labels: [ __meta_kubernetes_pod_controller_kind ]
|
||||
target_label: pod_controller_kind
|
||||
- action: replace
|
||||
source_labels: [ __meta_kubernetes_pod_phase ]
|
||||
target_label: pod_controller_phase
|
||||
metric_relabel_configs:
|
||||
- source_labels: [ __name__ ]
|
||||
regex: 'jvm_gc_collection_seconds.*'
|
||||
action: drop
|
||||
exporters:
|
||||
awsprometheusremotewrite:
|
||||
endpoint: {{ .Values.ampurl }}
|
||||
aws_auth:
|
||||
region: {{ .Values.region }}
|
||||
service: "aps"
|
||||
logging:
|
||||
loglevel: info
|
||||
extensions:
|
||||
health_check:
|
||||
pprof:
|
||||
endpoint: :1888
|
||||
zpages:
|
||||
endpoint: :55679
|
||||
service:
|
||||
extensions: [pprof, zpages, health_check]
|
||||
pipelines:
|
||||
metrics:
|
||||
receivers: [prometheus]
|
||||
exporters: [logging, awsprometheusremotewrite]
|
||||
@@ -0,0 +1,7 @@
|
||||
ampurl: ${amp_url}
|
||||
region: ${region}
|
||||
prometheusMetricsEndpoint: ${prometheus_metrics_endpoint}
|
||||
prometheusMetricsPort: ${prometheus_metrics_port}
|
||||
scrapeInterval: ${scrape_interval}
|
||||
scrapeTimeout: ${scrape_timeout}
|
||||
scrapeSampleLimit: ${scrape_sample_limit}
|
||||
@@ -0,0 +1,49 @@
|
||||
variable "enable_recording_rules" {
|
||||
description = "Enable AMP recording rules"
|
||||
type = bool
|
||||
default = true
|
||||
}
|
||||
|
||||
variable "amp_endpoint" {
|
||||
description = "Amazon Managed Prometheus endpoint"
|
||||
type = string
|
||||
}
|
||||
|
||||
variable "amp_id" {
|
||||
description = "Managed Prometheus workspace id"
|
||||
type = string
|
||||
}
|
||||
|
||||
variable "helm_config" {
|
||||
description = "Helm Config for Prometheus"
|
||||
type = any
|
||||
default = {}
|
||||
}
|
||||
|
||||
variable "amp_region" {
|
||||
description = "Amazon Managed Prometheus Workspace's Region"
|
||||
type = string
|
||||
default = null
|
||||
}
|
||||
|
||||
variable "dashboards_folder_id" {
|
||||
description = "Grafana folder ID for automatic dashboards"
|
||||
type = string
|
||||
}
|
||||
|
||||
variable "addon_context" {
|
||||
description = "Input configuration for the addon"
|
||||
type = object({
|
||||
aws_caller_identity_account_id = string
|
||||
aws_caller_identity_arn = string
|
||||
aws_eks_cluster_endpoint = string
|
||||
aws_partition_id = string
|
||||
aws_region_name = string
|
||||
eks_cluster_id = string
|
||||
eks_oidc_issuer_url = string
|
||||
eks_oidc_provider_arn = string
|
||||
irsa_iam_permissions_boundary = string
|
||||
irsa_iam_role_path = string
|
||||
tags = map(string)
|
||||
})
|
||||
}
|
||||
@@ -0,0 +1,25 @@
|
||||
terraform {
|
||||
required_version = ">= 1.0.0"
|
||||
required_providers {
|
||||
aws = {
|
||||
source = "hashicorp/aws"
|
||||
version = ">= 4.0.0"
|
||||
}
|
||||
kubernetes = {
|
||||
source = "hashicorp/kubernetes"
|
||||
version = ">= 2.10"
|
||||
}
|
||||
kubectl = {
|
||||
source = "gavinbunney/kubectl"
|
||||
version = ">= 1.14"
|
||||
}
|
||||
helm = {
|
||||
source = "hashicorp/helm"
|
||||
version = ">= 2.4.1"
|
||||
}
|
||||
grafana = {
|
||||
source = "grafana/grafana"
|
||||
version = ">= 1.25.0"
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,49 @@
|
||||
# Observability Pattern for Nginx
|
||||
|
||||
This module provides an automated experience around Observability for Nginx workloads.
|
||||
It provides the following resources:
|
||||
|
||||
- AWS Distro For OpenTelemetry Operator and Collector
|
||||
- AWS Managed Grafana Dashboard and data source
|
||||
- Alerts and recording rules with AWS Managed Service for Prometheus
|
||||
|
||||
<!-- BEGINNING OF PRE-COMMIT-TERRAFORM DOCS HOOK -->
|
||||
## Requirements
|
||||
|
||||
| Name | Version |
|
||||
|------|---------|
|
||||
| <a name="requirement_terraform"></a> [terraform](#requirement\_terraform) | >= 1.0.0 |
|
||||
| <a name="requirement_aws"></a> [aws](#requirement\_aws) | >= 4.0.0 |
|
||||
| <a name="requirement_kubernetes"></a> [kubernetes](#requirement\_kubernetes) | >= 2.10 |
|
||||
|
||||
## Providers
|
||||
|
||||
| Name | Version |
|
||||
|------|---------|
|
||||
| <a name="provider_aws"></a> [aws](#provider\_aws) | >= 4.0.0 |
|
||||
|
||||
## Modules
|
||||
|
||||
| Name | Source | Version |
|
||||
|------|--------|---------|
|
||||
| <a name="module_helm_addon"></a> [helm\_addon](#module\_helm\_addon) | ../helm-addon | n/a |
|
||||
|
||||
## Resources
|
||||
|
||||
| Name | Type |
|
||||
|------|------|
|
||||
| [aws_partition.current](https://registry.terraform.io/providers/hashicorp/aws/latest/docs/data-sources/partition) | data source |
|
||||
|
||||
## Inputs
|
||||
|
||||
| Name | Description | Type | Default | Required |
|
||||
|------|-------------|------|---------|:--------:|
|
||||
| <a name="input_addon_context"></a> [addon\_context](#input\_addon\_context) | Input configuration for the addon | <pre>object({<br> aws_caller_identity_account_id = string<br> aws_caller_identity_arn = string<br> aws_eks_cluster_endpoint = string<br> aws_partition_id = string<br> aws_region_name = string<br> eks_cluster_id = string<br> eks_oidc_issuer_url = string<br> eks_oidc_provider_arn = string<br> irsa_iam_permissions_boundary = string<br> irsa_iam_role_path = string<br> tags = map(string)<br> })</pre> | n/a | yes |
|
||||
| <a name="input_amazon_prometheus_workspace_endpoint"></a> [amazon\_prometheus\_workspace\_endpoint](#input\_amazon\_prometheus\_workspace\_endpoint) | Amazon Managed Prometheus Workspace Endpoint | `string` | `null` | no |
|
||||
| <a name="input_amazon_prometheus_workspace_region"></a> [amazon\_prometheus\_workspace\_region](#input\_amazon\_prometheus\_workspace\_region) | Amazon Managed Prometheus Workspace's Region | `string` | `null` | no |
|
||||
| <a name="input_helm_config"></a> [helm\_config](#input\_helm\_config) | Helm Config for Prometheus | `any` | `{}` | no |
|
||||
|
||||
## Outputs
|
||||
|
||||
No outputs.
|
||||
<!-- END OF PRE-COMMIT-TERRAFORM DOCS HOOK -->
|
||||
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,62 @@
|
||||
locals {
|
||||
name = "adot-collector-nginx"
|
||||
namespace = try(var.helm_config.namespace, local.name)
|
||||
}
|
||||
|
||||
data "aws_partition" "current" {}
|
||||
|
||||
module "helm_addon" {
|
||||
source = "github.com/aws-ia/terraform-aws-eks-blueprints/modules/kubernetes-addons/helm-addon"
|
||||
|
||||
helm_config = merge(
|
||||
{
|
||||
name = local.name
|
||||
chart = "${path.module}/otel-config"
|
||||
version = "0.2.0"
|
||||
namespace = local.namespace
|
||||
description = "ADOT helm Chart deployment configuration"
|
||||
},
|
||||
var.helm_config
|
||||
)
|
||||
|
||||
set_values = [
|
||||
{
|
||||
name = "ampurl"
|
||||
value = "${var.amazon_prometheus_workspace_endpoint}api/v1/remote_write"
|
||||
},
|
||||
{
|
||||
name = "region"
|
||||
value = var.amazon_prometheus_workspace_region
|
||||
},
|
||||
{
|
||||
name = "prometheusMetricsEndpoint"
|
||||
value = "metrics"
|
||||
},
|
||||
{
|
||||
name = "prometheusMetricsPort"
|
||||
value = 8888
|
||||
},
|
||||
{
|
||||
name = "scrapeInterval"
|
||||
value = "15s"
|
||||
},
|
||||
{
|
||||
name = "scrapeTimeout"
|
||||
value = "10s"
|
||||
},
|
||||
{
|
||||
name = "scrapeSampleLimit"
|
||||
value = 1000
|
||||
}
|
||||
]
|
||||
|
||||
irsa_config = {
|
||||
create_kubernetes_namespace = try(var.helm_config["create_namespace"], true)
|
||||
kubernetes_namespace = local.namespace
|
||||
create_kubernetes_service_account = true
|
||||
kubernetes_service_account = try(var.helm_config.service_account, local.name)
|
||||
irsa_iam_policies = ["arn:${data.aws_partition.current.partition}:iam::aws:policy/AmazonPrometheusRemoteWriteAccess"]
|
||||
}
|
||||
|
||||
addon_context = var.addon_context
|
||||
}
|
||||
@@ -0,0 +1,6 @@
|
||||
apiVersion: v2
|
||||
name: opentelemetry
|
||||
description: A Helm chart to install otel operator
|
||||
type: application
|
||||
version: 0.2.0
|
||||
appVersion: v0.1.0
|
||||
@@ -0,0 +1,29 @@
|
||||
apiVersion: rbac.authorization.k8s.io/v1
|
||||
kind: ClusterRole
|
||||
metadata:
|
||||
name: otel-prometheus-role
|
||||
rules:
|
||||
- apiGroups:
|
||||
- ""
|
||||
resources:
|
||||
- nodes
|
||||
- nodes/proxy
|
||||
- services
|
||||
- endpoints
|
||||
- pods
|
||||
verbs:
|
||||
- get
|
||||
- list
|
||||
- watch
|
||||
- apiGroups:
|
||||
- extensions
|
||||
resources:
|
||||
- ingresses
|
||||
verbs:
|
||||
- get
|
||||
- list
|
||||
- watch
|
||||
- nonResourceURLs:
|
||||
- /metrics
|
||||
verbs:
|
||||
- get
|
||||
@@ -0,0 +1,12 @@
|
||||
apiVersion: rbac.authorization.k8s.io/v1
|
||||
kind: ClusterRoleBinding
|
||||
metadata:
|
||||
name: otel-prometheus-role-binding
|
||||
roleRef:
|
||||
apiGroup: rbac.authorization.k8s.io
|
||||
kind: ClusterRole
|
||||
name: otel-prometheus-role
|
||||
subjects:
|
||||
- kind: ServiceAccount
|
||||
name: adot-collector-nginx
|
||||
namespace: adot-collector-nginx
|
||||
@@ -0,0 +1,67 @@
|
||||
apiVersion: opentelemetry.io/v1alpha1
|
||||
kind: OpenTelemetryCollector
|
||||
metadata:
|
||||
name: adot
|
||||
spec:
|
||||
image: public.ecr.aws/aws-observability/aws-otel-collector:latest
|
||||
mode: deployment
|
||||
serviceAccount: adot-collector-nginx
|
||||
config: |
|
||||
receivers:
|
||||
prometheus:
|
||||
config:
|
||||
global:
|
||||
scrape_interval: {{ .Values.scrapeInterval }}
|
||||
scrape_timeout: {{ .Values.scrapeTimeout }}
|
||||
|
||||
scrape_configs:
|
||||
- job_name: 'kubernetes-pod-nginx'
|
||||
sample_limit: {{ .Values.scrapeSampleLimit }}
|
||||
metrics_path: /{{ .Values.prometheusMetricsEndpoint }}
|
||||
kubernetes_sd_configs:
|
||||
- role: pod
|
||||
relabel_configs:
|
||||
- source_labels: [ __address__ ]
|
||||
action: keep
|
||||
regex: '.*:9404$'
|
||||
- action: labelmap
|
||||
regex: __meta_kubernetes_pod_label_(.+)
|
||||
- action: replace
|
||||
source_labels: [ __meta_kubernetes_namespace ]
|
||||
target_label: Namespace
|
||||
- source_labels: [ __meta_kubernetes_pod_name ]
|
||||
action: replace
|
||||
target_label: pod_name
|
||||
- action: replace
|
||||
source_labels: [ __meta_kubernetes_pod_container_name ]
|
||||
target_label: container_name
|
||||
- action: replace
|
||||
source_labels: [ __meta_kubernetes_pod_controller_kind ]
|
||||
target_label: pod_controller_kind
|
||||
- action: replace
|
||||
source_labels: [ __meta_kubernetes_pod_phase ]
|
||||
target_label: pod_controller_phase
|
||||
metric_relabel_configs:
|
||||
- source_labels: [ __name__ ]
|
||||
regex: 'jvm_gc_collection_seconds.*'
|
||||
action: drop
|
||||
exporters:
|
||||
awsprometheusremotewrite:
|
||||
endpoint: {{ .Values.ampurl }}
|
||||
aws_auth:
|
||||
region: {{ .Values.region }}
|
||||
service: "aps"
|
||||
logging:
|
||||
loglevel: info
|
||||
extensions:
|
||||
health_check:
|
||||
pprof:
|
||||
endpoint: :1888
|
||||
zpages:
|
||||
endpoint: :55679
|
||||
service:
|
||||
extensions: [pprof, zpages, health_check]
|
||||
pipelines:
|
||||
metrics:
|
||||
receivers: [prometheus]
|
||||
exporters: [logging, awsprometheusremotewrite]
|
||||
@@ -0,0 +1,7 @@
|
||||
ampurl: ${amp_url}
|
||||
region: ${region}
|
||||
prometheusMetricsEndpoint: ${prometheus_metrics_endpoint}
|
||||
prometheusMetricsPort: ${prometheus_metrics_port}
|
||||
scrapeInterval: ${scrape_interval}
|
||||
scrapeTimeout: ${scrape_timeout}
|
||||
scrapeSampleLimit: ${scrape_sample_limit}
|
||||
@@ -0,0 +1,34 @@
|
||||
variable "helm_config" {
|
||||
description = "Helm Config for Prometheus"
|
||||
type = any
|
||||
default = {}
|
||||
}
|
||||
|
||||
variable "amazon_prometheus_workspace_endpoint" {
|
||||
description = "Amazon Managed Prometheus Workspace Endpoint"
|
||||
type = string
|
||||
default = null
|
||||
}
|
||||
|
||||
variable "amazon_prometheus_workspace_region" {
|
||||
description = "Amazon Managed Prometheus Workspace's Region"
|
||||
type = string
|
||||
default = null
|
||||
}
|
||||
|
||||
variable "addon_context" {
|
||||
description = "Input configuration for the addon"
|
||||
type = object({
|
||||
aws_caller_identity_account_id = string
|
||||
aws_caller_identity_arn = string
|
||||
aws_eks_cluster_endpoint = string
|
||||
aws_partition_id = string
|
||||
aws_region_name = string
|
||||
eks_cluster_id = string
|
||||
eks_oidc_issuer_url = string
|
||||
eks_oidc_provider_arn = string
|
||||
irsa_iam_permissions_boundary = string
|
||||
irsa_iam_role_path = string
|
||||
tags = map(string)
|
||||
})
|
||||
}
|
||||
@@ -0,0 +1,14 @@
|
||||
terraform {
|
||||
required_version = ">= 1.0.0"
|
||||
|
||||
required_providers {
|
||||
aws = {
|
||||
source = "hashicorp/aws"
|
||||
version = ">= 4.0.0"
|
||||
}
|
||||
kubernetes = {
|
||||
source = "hashicorp/kubernetes"
|
||||
version = ">= 2.10"
|
||||
}
|
||||
}
|
||||
}
|
||||
+39
@@ -0,0 +1,39 @@
|
||||
output "eks_cluster_id" {
|
||||
description = "EKS Cluster Id"
|
||||
value = var.eks_cluster_id
|
||||
}
|
||||
|
||||
output "aws_region" {
|
||||
description = "EKS Cluster Id"
|
||||
value = var.aws_region
|
||||
}
|
||||
|
||||
output "eks_cluster_version" {
|
||||
description = "EKS Cluster version"
|
||||
value = data.aws_eks_cluster.eks_cluster.version
|
||||
}
|
||||
|
||||
output "managed_prometheus_workspace_endpoint" {
|
||||
description = "Amazon Managed Prometheus workspace endpoint"
|
||||
value = local.amp_ws_endpoint
|
||||
}
|
||||
|
||||
output "managed_prometheus_workspace_id" {
|
||||
description = "Amazon Managed Prometheus workspace ID"
|
||||
value = local.amp_ws_id
|
||||
}
|
||||
|
||||
output "managed_prometheus_workspace_region" {
|
||||
description = "Amazon Managed Prometheus workspace region"
|
||||
value = local.amp_ws_region
|
||||
}
|
||||
|
||||
output "managed_grafana_workspace_endpoint" {
|
||||
description = "Amazon Managed Grafana workspace endpoint"
|
||||
value = local.amg_ws_endpoint
|
||||
}
|
||||
|
||||
output "grafana_dashboards_folder_id" {
|
||||
description = "Grafana folder ID for automatic dashboards. Required by workload modules"
|
||||
value = grafana_folder.this.id
|
||||
}
|
||||
@@ -0,0 +1,21 @@
|
||||
package test
|
||||
|
||||
import (
|
||||
"testing"
|
||||
|
||||
"github.com/gruntwork-io/terratest/modules/terraform"
|
||||
)
|
||||
|
||||
func TestExamplesBasic(t *testing.T) {
|
||||
|
||||
terraformOptions := &terraform.Options{
|
||||
TerraformDir: "../examples/basic",
|
||||
// Vars: map[string]interface{}{
|
||||
// "myvar": "test",
|
||||
// "mylistvar": []string{"list_item_1"},
|
||||
// },
|
||||
}
|
||||
|
||||
defer terraform.Destroy(t, terraformOptions)
|
||||
terraform.InitAndApply(t, terraformOptions)
|
||||
}
|
||||
@@ -0,0 +1,8 @@
|
||||
exclude:
|
||||
- aws-observabilitym-no-policy-wildcards # Wildcards required in addon IAM policies
|
||||
- aws-vpc-no-excessive-port-access # VPC settings left up to user implementation for recommended practices
|
||||
- aws-vpc-no-public-ingress-acl # VPC settings left up to user implementation for recommended practices
|
||||
- aws-eks-no-public-cluster-access-to-cidr # Public access enabled for better example usability, users are recommended to disable if possible
|
||||
- aws-eks-no-public-cluster-access # Public access enabled for better example usability, users are recommended to disable if possible
|
||||
- aws-eks-encrypt-secrets # Module defaults to encrypting secrets with CMK, but this is not hardcoded and therefore a spurious error
|
||||
- aws-vpc-no-public-egress-sgr # Added in v1.22
|
||||
@@ -0,0 +1,80 @@
|
||||
variable "eks_cluster_id" {
|
||||
description = "Name of the EKS cluster"
|
||||
type = string
|
||||
}
|
||||
|
||||
variable "aws_region" {
|
||||
description = "AWS Region"
|
||||
type = string
|
||||
}
|
||||
|
||||
variable "irsa_iam_role_path" {
|
||||
description = "IAM role path for IRSA roles"
|
||||
type = string
|
||||
default = "/"
|
||||
}
|
||||
|
||||
variable "irsa_iam_permissions_boundary" {
|
||||
description = "IAM permissions boundary for IRSA roles"
|
||||
type = string
|
||||
default = ""
|
||||
}
|
||||
|
||||
variable "enable_amazon_eks_adot" {
|
||||
description = "Enables the ADOT Operator on the EKS Cluster"
|
||||
type = bool
|
||||
default = true
|
||||
}
|
||||
|
||||
variable "enable_cert_manager" {
|
||||
description = "Allow reusing an existing installation of cert-manager"
|
||||
type = bool
|
||||
default = true
|
||||
}
|
||||
|
||||
variable "enable_managed_prometheus" {
|
||||
description = "Creates a new Amazon Managed Service for Prometheus Workspace"
|
||||
type = bool
|
||||
default = true
|
||||
}
|
||||
|
||||
variable "managed_prometheus_workspace_id" {
|
||||
description = "Amazon Managed Service for Prometheus Workspace ID"
|
||||
type = string
|
||||
default = ""
|
||||
}
|
||||
|
||||
variable "managed_prometheus_workspace_region" {
|
||||
description = "Region where Amazon Managed Service for Prometheus is deployed"
|
||||
type = string
|
||||
default = null
|
||||
}
|
||||
|
||||
variable "enable_alertmanager" {
|
||||
description = "Creates Amazon Managed Service for Prometheus AlertManager for all workloads"
|
||||
type = bool
|
||||
default = false
|
||||
}
|
||||
|
||||
variable "enable_managed_grafana" {
|
||||
description = "Creates a new Amazon Managed Grafana Workspace"
|
||||
type = bool
|
||||
default = true
|
||||
}
|
||||
|
||||
variable "managed_grafana_workspace_id" {
|
||||
description = "Amazon Managed Grafana Workspace ID"
|
||||
type = string
|
||||
default = ""
|
||||
}
|
||||
variable "grafana_api_key" {
|
||||
description = "Grafana API key for the Amazon Managed Grafana workspace"
|
||||
type = string
|
||||
default = null
|
||||
}
|
||||
|
||||
variable "tags" {
|
||||
description = "Additional tags (e.g. `map('BusinessUnit`,`XYZ`)"
|
||||
type = map(string)
|
||||
default = {}
|
||||
}
|
||||
+17
@@ -0,0 +1,17 @@
|
||||
terraform {
|
||||
required_version = ">= 0.14.0"
|
||||
required_providers {
|
||||
aws = {
|
||||
source = "hashicorp/aws"
|
||||
version = ">= 4.0.0"
|
||||
}
|
||||
awscc = {
|
||||
source = "hashicorp/awscc"
|
||||
version = ">= 0.24.0"
|
||||
}
|
||||
grafana = {
|
||||
source = "grafana/grafana"
|
||||
version = "1.25.0"
|
||||
}
|
||||
}
|
||||
}
|
||||
Reference in New Issue
Block a user