From 6032c6da23b74824a335f3ee001efbd0ddcfe8e6 Mon Sep 17 00:00:00 2001 From: Rodrigue Koffi Date: Tue, 4 Oct 2022 18:20:14 +0200 Subject: [PATCH] feat: Default scrape interval (#49) * Enable higher but configurable scrape intervals * Use global parameters per job level * Remove scrape limit --- .../main.tf | 6 ++ modules/workloads/infra/README.md | 5 +- modules/workloads/infra/main.tf | 30 +++----- .../templates/opentelemetrycollector.yaml | 74 ++++++++----------- .../workloads/infra/otel-config/values.yaml | 45 +++++++++-- modules/workloads/infra/variables.tf | 27 +++++++ modules/workloads/java/main.tf | 1 - 7 files changed, 117 insertions(+), 71 deletions(-) diff --git a/examples/existing-cluster-with-base-and-infra/main.tf b/examples/existing-cluster-with-base-and-infra/main.tf index f150d81..42512b5 100644 --- a/examples/existing-cluster-with-base-and-infra/main.tf +++ b/examples/existing-cluster-with-base-and-infra/main.tf @@ -85,6 +85,12 @@ module "workloads_infra" { managed_prometheus_workspace_endpoint = module.eks_observability_accelerator.managed_prometheus_workspace_endpoint managed_prometheus_workspace_region = module.eks_observability_accelerator.managed_prometheus_workspace_region + # optional, defaults to 60s interval and 15s timeout + prometheus_config = { + global_scrape_interval = "60s" + global_scrape_timeout = "15s" + } + tags = local.tags depends_on = [ diff --git a/modules/workloads/infra/README.md b/modules/workloads/infra/README.md index b981291..3361d86 100644 --- a/modules/workloads/infra/README.md +++ b/modules/workloads/infra/README.md @@ -66,11 +66,12 @@ This module is inspired from the open source [kube-prometheus-stack](https://git | [helm\_config](#input\_helm\_config) | Helm Config for Prometheus | `any` | `{}` | no | | [irsa\_iam\_permissions\_boundary](#input\_irsa\_iam\_permissions\_boundary) | IAM permissions boundary for IRSA roles | `string` | `""` | no | | [irsa\_iam\_role\_path](#input\_irsa\_iam\_role\_path) | IAM role path for IRSA roles | `string` | `"/"` | no | -| [ksm\_config](#input\_ksm\_config) | Kube State metrics configuration |
object({
create_namespace = bool
k8s_namespace = string
helm_chart_name = string
helm_chart_version = string
helm_release_name = string
helm_repo_url = string
helm_settings = map(string)
helm_values = map(any)
})
|
{
"create_namespace": true,
"helm_chart_name": "kube-state-metrics",
"helm_chart_version": "4.16.0",
"helm_release_name": "kube-state-metrics",
"helm_repo_url": "https://prometheus-community.github.io/helm-charts",
"helm_settings": {},
"helm_values": {},
"k8s_namespace": "kube-system"
}
| no | +| [ksm\_config](#input\_ksm\_config) | Kube State metrics configuration |
object({
create_namespace = bool
k8s_namespace = string
helm_chart_name = string
helm_chart_version = string
helm_release_name = string
helm_repo_url = string
helm_settings = map(string)
helm_values = map(any)

scrape_interval = string
scrape_timeout = string
})
|
{
"create_namespace": true,
"helm_chart_name": "kube-state-metrics",
"helm_chart_version": "4.16.0",
"helm_release_name": "kube-state-metrics",
"helm_repo_url": "https://prometheus-community.github.io/helm-charts",
"helm_settings": {},
"helm_values": {},
"k8s_namespace": "kube-system",
"scrape_interval": "60s",
"scrape_timeout": "15s"
}
| no | | [managed\_prometheus\_workspace\_endpoint](#input\_managed\_prometheus\_workspace\_endpoint) | Amazon Managed Prometheus Workspace Endpoint | `string` | `null` | no | | [managed\_prometheus\_workspace\_id](#input\_managed\_prometheus\_workspace\_id) | Amazon Managed Prometheus Workspace ID | `string` | `null` | no | | [managed\_prometheus\_workspace\_region](#input\_managed\_prometheus\_workspace\_region) | Amazon Managed Prometheus Workspace's Region | `string` | `null` | no | -| [ne\_config](#input\_ne\_config) | Node exporter configuration |
object({
create_namespace = bool
k8s_namespace = string
helm_chart_name = string
helm_chart_version = string
helm_release_name = string
helm_repo_url = string
helm_settings = map(string)
helm_values = map(any)
})
|
{
"create_namespace": true,
"helm_chart_name": "prometheus-node-exporter",
"helm_chart_version": "2.0.3",
"helm_release_name": "prometheus-node-exporter",
"helm_repo_url": "https://prometheus-community.github.io/helm-charts",
"helm_settings": {},
"helm_values": {},
"k8s_namespace": "prometheus-node-exporter"
}
| no | +| [ne\_config](#input\_ne\_config) | Node exporter configuration |
object({
create_namespace = bool
k8s_namespace = string
helm_chart_name = string
helm_chart_version = string
helm_release_name = string
helm_repo_url = string
helm_settings = map(string)
helm_values = map(any)

scrape_interval = string
scrape_timeout = string
})
|
{
"create_namespace": true,
"helm_chart_name": "prometheus-node-exporter",
"helm_chart_version": "2.0.3",
"helm_release_name": "prometheus-node-exporter",
"helm_repo_url": "https://prometheus-community.github.io/helm-charts",
"helm_settings": {},
"helm_values": {},
"k8s_namespace": "prometheus-node-exporter",
"scrape_interval": "60s",
"scrape_timeout": "60s"
}
| no | +| [prometheus\_config](#input\_prometheus\_config) | Controls default values such as scrape interval, timeouts and ports globally |
object({
global_scrape_interval = string
global_scrape_timeout = string
})
|
{
"global_scrape_interval": "60s",
"global_scrape_timeout": "15s"
}
| no | | [tags](#input\_tags) | Additional tags (e.g. `map('BusinessUnit`,`XYZ`) | `map(string)` | `{}` | no | ## Outputs diff --git a/modules/workloads/infra/main.tf b/modules/workloads/infra/main.tf index b5dbb49..8bb3845 100644 --- a/modules/workloads/infra/main.tf +++ b/modules/workloads/infra/main.tf @@ -41,7 +41,7 @@ module "helm_addon" { { name = local.name chart = "${path.module}/otel-config" - version = "0.2.0" + version = "0.3.0" namespace = local.namespace description = "ADOT helm Chart deployment configuration" }, @@ -57,30 +57,18 @@ module "helm_addon" { name = "region" value = var.managed_prometheus_workspace_region }, - { - name = "prometheusMetricsEndpoint" - value = "metrics" - }, - { - name = "prometheusMetricsPort" - value = 8888 - }, - { - name = "scrapeInterval" - value = "15s" - }, - { - name = "scrapeTimeout" - value = "10s" - }, - { - name = "scrapeSampleLimit" - value = 1000 - }, { name = "ekscluster" value = local.context.eks_cluster_id }, + { + name = "globalScrapeInterval" + value = var.prometheus_config.global_scrape_interval + }, + { + name = "globalScrapeTimeout" + value = var.prometheus_config.global_scrape_timeout + }, ] irsa_config = { diff --git a/modules/workloads/infra/otel-config/templates/opentelemetrycollector.yaml b/modules/workloads/infra/otel-config/templates/opentelemetrycollector.yaml index 7641775..ee3b923 100644 --- a/modules/workloads/infra/otel-config/templates/opentelemetrycollector.yaml +++ b/modules/workloads/infra/otel-config/templates/opentelemetrycollector.yaml @@ -11,12 +11,14 @@ spec: prometheus: config: global: - scrape_interval: {{ .Values.scrapeInterval }} - scrape_timeout: {{ .Values.scrapeTimeout }} + scrape_interval: {{ .Values.globalScrapeInterval }} + scrape_timeout: {{ .Values.globalScrapeTimeout }} external_labels: cluster: {{ .Values.ekscluster }} scrape_configs: - job_name: 'kubernetes-kubelet' + scrape_interval: {{ .Values.globalScrapeInterval }} + scrape_timeout: {{ .Values.globalScrapeTimeout }} scheme: https tls_config: ca_file: /var/run/secrets/kubernetes.io/serviceaccount/ca.crt @@ -52,9 +54,8 @@ spec: replacement: /api/v1/nodes/$${1}/proxy/metrics/cadvisor - job_name: serviceMonitor/default/kube-prometheus-stack-prometheus-node-exporter/0 honor_timestamps: true - scrape_interval: 30s - scrape_timeout: 10s - metrics_path: /metrics + scrape_interval: {{ .Values.globalScrapeInterval }} + scrape_timeout: {{ .Values.globalScrapeInterval }} scheme: http follow_redirects: true enable_http2: true @@ -156,9 +157,8 @@ spec: - default - job_name: serviceMonitor/default/kube-prometheus-stack-prometheus/0 honor_timestamps: true - scrape_interval: 30s - scrape_timeout: 10s - metrics_path: /metrics + scrape_interval: {{ .Values.globalScrapeInterval }} + scrape_timeout: {{ .Values.globalScrapeTimeout }} scheme: http follow_redirects: true enable_http2: true @@ -260,9 +260,8 @@ spec: - job_name: serviceMonitor/default/kube-prometheus-stack-operator/0 honor_labels: true honor_timestamps: true - scrape_interval: 30s - scrape_timeout: 10s - metrics_path: /metrics + scrape_interval: {{ .Values.globalScrapeInterval }} + scrape_timeout: {{ .Values.globalScrapeTimeout }} scheme: https tls_config: ca_file: /var/run/secrets/kubernetes.io/serviceaccount/ca.crt @@ -362,8 +361,8 @@ spec: - job_name: serviceMonitor/default/kube-prometheus-stack-kubelet/2 honor_labels: true honor_timestamps: true - scrape_interval: 30s - scrape_timeout: 10s + scrape_interval: {{ .Values.globalScrapeInterval }} + scrape_timeout: {{ .Values.globalScrapeTimeout }} metrics_path: /metrics/probes scheme: https authorization: @@ -479,8 +478,8 @@ spec: - job_name: serviceMonitor/default/kube-prometheus-stack-kubelet/1 honor_labels: true honor_timestamps: true - scrape_interval: 30s - scrape_timeout: 10s + scrape_interval: {{ .Values.globalScrapeInterval }} + scrape_timeout: {{ .Values.globalScrapeTimeout }} metrics_path: /metrics/cadvisor scheme: https authorization: @@ -596,9 +595,8 @@ spec: - job_name: serviceMonitor/default/kube-prometheus-stack-kubelet/0 honor_labels: true honor_timestamps: true - scrape_interval: 30s - scrape_timeout: 10s - metrics_path: /metrics + scrape_interval: {{ .Values.globalScrapeInterval }} + scrape_timeout: {{ .Values.globalScrapeTimeout }} scheme: https authorization: type: Bearer @@ -713,9 +711,8 @@ spec: - job_name: serviceMonitor/default/kube-prometheus-stack-kube-state-metrics/0 honor_labels: true honor_timestamps: true - scrape_interval: 30s - scrape_timeout: 10s - metrics_path: /metrics + scrape_interval: {{ .Values.globalScrapeInterval }} + scrape_timeout: {{ .Values.globalScrapeTimeout }} scheme: http follow_redirects: true enable_http2: true @@ -817,9 +814,8 @@ spec: - default - job_name: serviceMonitor/default/kube-prometheus-stack-kube-scheduler/0 honor_timestamps: true - scrape_interval: 30s - scrape_timeout: 10s - metrics_path: /metrics + scrape_interval: {{ .Values.globalScrapeInterval }} + scrape_timeout: {{ .Values.globalScrapeTimeout }} scheme: http authorization: type: Bearer @@ -924,9 +920,8 @@ spec: - kube-system - job_name: serviceMonitor/default/kube-prometheus-stack-kube-proxy/0 honor_timestamps: true - scrape_interval: 30s - scrape_timeout: 10s - metrics_path: /metrics + scrape_interval: {{ .Values.globalScrapeInterval }} + scrape_timeout: {{ .Values.globalScrapeTimeout }} scheme: http authorization: type: Bearer @@ -1031,9 +1026,8 @@ spec: - kube-system - job_name: serviceMonitor/default/kube-prometheus-stack-kube-etcd/0 honor_timestamps: true - scrape_interval: 30s - scrape_timeout: 10s - metrics_path: /metrics + scrape_interval: {{ .Values.globalScrapeInterval }} + scrape_timeout: {{ .Values.globalScrapeTimeout }} scheme: http authorization: type: Bearer @@ -1138,9 +1132,8 @@ spec: - kube-system - job_name: serviceMonitor/default/kube-prometheus-stack-kube-controller-manager/0 honor_timestamps: true - scrape_interval: 30s - scrape_timeout: 10s - metrics_path: /metrics + scrape_interval: {{ .Values.globalScrapeInterval }} + scrape_timeout: {{ .Values.globalScrapeTimeout }} scheme: http authorization: type: Bearer @@ -1245,9 +1238,8 @@ spec: - kube-system - job_name: serviceMonitor/default/kube-prometheus-stack-coredns/0 honor_timestamps: true - scrape_interval: 30s - scrape_timeout: 10s - metrics_path: /metrics + scrape_interval: {{ .Values.globalScrapeInterval }} + scrape_timeout: {{ .Values.globalScrapeTimeout }} scheme: http authorization: type: Bearer @@ -1350,9 +1342,8 @@ spec: - kube-system - job_name: serviceMonitor/default/kube-prometheus-stack-apiserver/0 honor_timestamps: true - scrape_interval: 30s - scrape_timeout: 10s - metrics_path: /metrics + scrape_interval: {{ .Values.globalScrapeInterval }} + scrape_timeout: {{ .Values.globalScrapeTimeout }} scheme: https authorization: type: Bearer @@ -1455,9 +1446,8 @@ spec: - default - job_name: serviceMonitor/default/kube-prometheus-stack-alertmanager/0 honor_timestamps: true - scrape_interval: 30s - scrape_timeout: 10s - metrics_path: /metrics + scrape_interval: {{ .Values.globalScrapeInterval }} + scrape_timeout: {{ .Values.globalScrapeTimeout }} scheme: http follow_redirects: true enable_http2: true diff --git a/modules/workloads/infra/otel-config/values.yaml b/modules/workloads/infra/otel-config/values.yaml index 43383d0..10c83c5 100644 --- a/modules/workloads/infra/otel-config/values.yaml +++ b/modules/workloads/infra/otel-config/values.yaml @@ -1,8 +1,43 @@ ampurl: ${amp_url} region: ${region} -prometheusMetricsEndpoint: ${prometheus_metrics_endpoint} -prometheusMetricsPort: ${prometheus_metrics_port} -scrapeInterval: ${scrape_interval} -scrapeTimeout: ${scrape_timeout} -scrapeSampleLimit: ${scrape_sample_limit} ekscluster: ${eks_cluster} + +globalScrapeTimeout: ${global_scrape_timeout} +globalScrapeSampleLimit: ${global_scrape_sample_limit} +# TODO: enable after terraform 1.3 as defaults will be optional + +#nodeExporterScrapeInterval: ${node_exporter_scrape_interval} +#nodeExporterScrapeTimeout: ${node_exporter_scrape_timeout} + +#kubeletScrapeInterval: ${kubelet_scrape_interval} +#kubeletScrapeTimeout: ${kubelet_scrape_timeout} + +#operatorScrapeInterval: ${operator_scrape_interval} +#operatorScrapeTimeout: ${operator_scrape_timeout} + +#kubeletScrapeInterval: ${operator_scrape_interval} +#kubeletScrapeTimeout: ${operator_scrape_timeout} + +#ksmScrapeInterval: ${ksm_scrape_interval} +#ksmScrapeTimeout: ${ksm_scrape_timeout} + +#schedulerScrapeInterval: ${scheduler_scrape_interval} +#schedulerScrapeTimeout: ${scheduler_scrape_timeout} + +#proxyScrapeInterval: ${proxy_scrape_interval} +#proxyScrapeTimeout: ${proxy_scrape_timeout} + +#etcdScrapeInterval: ${etcd_scrape_interval} +#etcdScrapeTimeout: ${etcd_scrape_timeout} + +#controllerManagerScrapeInterval: ${controller_manager_scrape_interval} +#controllerManagerScrapeTimeout: ${controller_manager_scrape_timeout} + +#corednsScrapeInterval: ${coredns_scrape_interval} +#corednsScrapeTimeout: ${coredns_scrape_timeout} + +#apiserverScrapeInterval: ${apiserver_scrape_interval} +#apiserverScrapeTimeout: ${apiserver_scrape_timeout} + +#alertmanagerScrapeInterval: ${alertmnager_scrape_interval} +#alertmanagerScrapeTimeout: ${alertmnager_scrape_timeout} diff --git a/modules/workloads/infra/variables.tf b/modules/workloads/infra/variables.tf index 5d65aa2..7bed974 100644 --- a/modules/workloads/infra/variables.tf +++ b/modules/workloads/infra/variables.tf @@ -78,6 +78,9 @@ variable "ksm_config" { helm_repo_url = string helm_settings = map(string) helm_values = map(any) + + scrape_interval = string + scrape_timeout = string }) default = { @@ -89,6 +92,9 @@ variable "ksm_config" { helm_settings = {} helm_values = {} k8s_namespace = "kube-system" + + scrape_interval = "60s" + scrape_timeout = "15s" } nullable = false } @@ -110,6 +116,9 @@ variable "ne_config" { helm_repo_url = string helm_settings = map(string) helm_values = map(any) + + scrape_interval = string + scrape_timeout = string }) default = { @@ -121,11 +130,29 @@ variable "ne_config" { helm_settings = {} helm_values = {} k8s_namespace = "prometheus-node-exporter" + + scrape_interval = "60s" + scrape_timeout = "60s" } nullable = false } + variable "tags" { description = "Additional tags (e.g. `map('BusinessUnit`,`XYZ`)" type = map(string) default = {} } + +variable "prometheus_config" { + description = "Controls default values such as scrape interval, timeouts and ports globally" + type = object({ + global_scrape_interval = string + global_scrape_timeout = string + }) + + default = { + global_scrape_interval = "60s" + global_scrape_timeout = "15s" + } + nullable = false +} diff --git a/modules/workloads/java/main.tf b/modules/workloads/java/main.tf index 0a15492..92101a4 100644 --- a/modules/workloads/java/main.tf +++ b/modules/workloads/java/main.tf @@ -9,7 +9,6 @@ data "aws_partition" "current" {} module "helm_addon" { source = "github.com/aws-ia/terraform-aws-eks-blueprints//modules/kubernetes-addons/helm-addon?ref=v4.8.1" - helm_config = merge( { name = local.name