Node exporter dashboards (#43)

* Added dashboards for node exporter
* Drop use cluster and node
* Add nodename to ne metrics
* Add account ID and region

Co-authored-by: Venkataraman <vikrvenk@88665a367666.ant.amazon.com>
Co-authored-by: Rodrigue Koffi <bonclay7@users.noreply.github.com>
This commit is contained in:
Vikram Venkataraman
2022-10-24 11:40:44 -04:00
committed by GitHub
parent 2e6ac046d8
commit a42b1751e9
8 changed files with 1307 additions and 4 deletions
+1 -1
View File
@@ -8,7 +8,7 @@ on:
workflow_dispatch:
concurrency:
group: '${{ github.workflow }} @ ${{ github.event.pull_request.head.label || github.head_ref || github.ref }}'
group: "${{ github.workflow }} @ ${{ github.event.pull_request.head.label || github.head_ref || github.ref }}"
cancel-in-progress: true
jobs:
+1 -1
View File
@@ -137,4 +137,4 @@ jobs:
with:
terraform-version: ${{ steps.minMax.outputs.maxVersion }}
terraform-docs-version: ${{ env.TERRAFORM_DOCS_VERSION }}
tflint-version: ${{ env.TFLINT_VERSION }}
tflint-version: ${{ env.TFLINT_VERSION }}
+1 -1
View File
@@ -30,4 +30,4 @@ jobs:
with no activity. Remove stale label or comment or this issue will be closed in 10 days
stale-pr-message: |
This PR has been automatically marked as stale because it has been open 30 days
with no activity. Remove stale label or comment or this PR will be closed in 10 days
with no activity. Remove stale label or comment or this PR will be closed in 10 days
+1
View File
@@ -42,6 +42,7 @@ This module is inspired from the open source [kube-prometheus-stack](https://git
| [aws_prometheus_rule_group_namespace.recording_rules](https://registry.terraform.io/providers/hashicorp/aws/latest/docs/resources/prometheus_rule_group_namespace) | resource |
| [grafana_dashboard.cluster](https://registry.terraform.io/providers/grafana/grafana/latest/docs/resources/dashboard) | resource |
| [grafana_dashboard.kubelet](https://registry.terraform.io/providers/grafana/grafana/latest/docs/resources/dashboard) | resource |
| [grafana_dashboard.nodeexp_nodes](https://registry.terraform.io/providers/grafana/grafana/latest/docs/resources/dashboard) | resource |
| [grafana_dashboard.nodes](https://registry.terraform.io/providers/grafana/grafana/latest/docs/resources/dashboard) | resource |
| [grafana_dashboard.nsworkload](https://registry.terraform.io/providers/grafana/grafana/latest/docs/resources/dashboard) | resource |
| [grafana_dashboard.workloads](https://registry.terraform.io/providers/grafana/grafana/latest/docs/resources/dashboard) | resource |
+6
View File
@@ -28,3 +28,9 @@ resource "grafana_dashboard" "cluster" {
folder = var.dashboards_folder_id
config_json = file("${path.module}/dashboards/cluster.json")
}
resource "grafana_dashboard" "nodeexp_nodes" {
count = var.enable_dashboards ? 1 : 0
folder = var.dashboards_folder_id
config_json = file("${path.module}/dashboards/nodeexporter-nodes.json")
}
File diff suppressed because it is too large Load Diff
+5 -1
View File
@@ -41,7 +41,7 @@ module "helm_addon" {
{
name = local.name
chart = "${path.module}/otel-config"
version = "0.3.0"
version = "0.3.1"
namespace = local.namespace
description = "ADOT helm Chart deployment configuration"
},
@@ -69,6 +69,10 @@ module "helm_addon" {
name = "globalScrapeTimeout"
value = var.prometheus_config.global_scrape_timeout
},
{
name = "accountId"
value = local.context.aws_caller_identity_account_id
},
]
irsa_config = {
@@ -15,6 +15,8 @@ spec:
scrape_timeout: {{ .Values.globalScrapeTimeout }}
external_labels:
cluster: {{ .Values.ekscluster }}
account_id: {{ .Values.accountId }}
region: {{ .Values.region }}
scrape_configs:
- job_name: 'kubernetes-kubelet'
scrape_interval: {{ .Values.globalScrapeInterval }}
@@ -1552,6 +1554,14 @@ spec:
- job_name: 'node-exporter'
kubernetes_sd_configs:
- role: endpoints
ec2_sd_configs:
relabel_configs:
- source_labels: [ __address__ ]
action: keep
regex: '.*:9100$'
- action: replace
source_labels: [__meta_kubernetes_endpoint_node_name]
target_label: nodename
exporters:
prometheusremotewrite:
endpoint: {{ .Values.ampurl }}