From 4019db6cd794aeebf2fcdeb120f9bbc5a63de86d Mon Sep 17 00:00:00 2001 From: Rodrigue Koffi Date: Thu, 30 Mar 2023 15:48:40 +0200 Subject: [PATCH] Support Fluentbit to CloudWatch logs (#140) * Import and customize fluenbit add-on * Enable fluent bit logs * Bump helm addon version * Dropping account id as it seems to create scraping errors * Create separate log groups per namespace * Apply pre-commit * Remove conflicting global label * Add config object for logs * Enable logs in examples * Add logs docs * Fix broken link * Add screenshots * Update docs * Typos --- README.md | 61 +++++++++++------ docs/concepts.md | 4 +- docs/eks/logs.md | 67 +++++++++++++++++++ docs/index.md | 22 +++--- examples/existing-cluster-java/main.tf | 2 + examples/existing-cluster-nginx/main.tf | 2 + .../main.tf | 2 + mkdocs.yml | 1 + modules/eks-monitoring/README.md | 10 ++- .../add-ons/aws-for-fluentbit/README.md | 54 +++++++++++++++ .../add-ons/aws-for-fluentbit/data.tf | 21 ++++++ .../add-ons/aws-for-fluentbit/locals.tf | 47 +++++++++++++ .../add-ons/aws-for-fluentbit/main.tf | 15 +++++ .../add-ons/aws-for-fluentbit/outputs.tf | 19 ++++++ .../add-ons/aws-for-fluentbit/values.yaml | 16 +++++ .../add-ons/aws-for-fluentbit/variables.tf | 40 +++++++++++ .../add-ons/aws-for-fluentbit/versions.tf | 10 +++ modules/eks-monitoring/main.tf | 10 ++- .../templates/opentelemetrycollector.yaml | 1 - modules/eks-monitoring/variables.tf | 18 +++++ 20 files changed, 386 insertions(+), 36 deletions(-) create mode 100644 docs/eks/logs.md create mode 100644 modules/eks-monitoring/add-ons/aws-for-fluentbit/README.md create mode 100644 modules/eks-monitoring/add-ons/aws-for-fluentbit/data.tf create mode 100644 modules/eks-monitoring/add-ons/aws-for-fluentbit/locals.tf create mode 100644 modules/eks-monitoring/add-ons/aws-for-fluentbit/main.tf create mode 100644 modules/eks-monitoring/add-ons/aws-for-fluentbit/outputs.tf create mode 100644 modules/eks-monitoring/add-ons/aws-for-fluentbit/values.yaml create mode 100644 modules/eks-monitoring/add-ons/aws-for-fluentbit/variables.tf create mode 100644 modules/eks-monitoring/add-ons/aws-for-fluentbit/versions.tf diff --git a/README.md b/README.md index a2052df..d8a1f3d 100644 --- a/README.md +++ b/README.md @@ -4,13 +4,14 @@ Welcome to the AWS Observability Accelerator for Terraform! -The AWS Observability Accelerator for Terraform is a set of opinionated modules to -help you set up observability for your AWS environments with +The AWS Observability Accelerator for Terraform is a set of opinionated modules +to help you set up observability for your AWS environments with AWS-managed observability services such as Amazon Managed Service for Prometheus, -Amazon Managed Grafana and AWS Distro for OpenTelemetry (ADOT). +Amazon Managed Grafana, AWS Distro for OpenTelemetry (ADOT) and Amazon CloudWatch. -We provide curated metrics, traces collection, alerting rules and Grafana dashboards -for your EKS infrastructure, Java/JMX, NGINX based workloads and custom applications. +We provide curated metrics, logs, traces collection, alerting rules and Grafana +dashboards for your EKS infrastructure, Java/JMX, NGINX based workloads and +your custom applications. You also can monitor your Amazon Managed Service for Prometheus workspaces ingestion, costs, active series with [this module](./modules/managed-prometheus-monitoring). @@ -42,18 +43,20 @@ v2+ releases introduces couple of breaking changes compared to previous versions ### Base Module -The base module allows you to configure the AWS Observability services for your cluster and -the AWS Distro for OpenTelemetry (ADOT) Operator as the signals collection mechanism. +The base module allows you to configure the AWS Observability services for your +cluster and the AWS Distro for OpenTelemetry (ADOT) Operator as the signals +collection mechanism. -This is the minimum configuration to have a new Amazon Managed Service for Prometheus Workspace -and ADOT Operator deployed for you and ready to receive your data. -The base module serve as an anchor to the workload modules and cannot run on its own. +This is the minimum configuration to have a new Amazon Managed Service for +Prometheus Workspace and ADOT Operator deployed for you and ready to receive +your data. The base module serve as an anchor to the workload modules and +cannot run on its own. ```hcl module "aws_observability_accelerator" { # use release tags and check for the latest versions # https://github.com/aws-observability/terraform-aws-observability-accelerator/releases - source = "github.com/aws-observability/terraform-aws-observability-accelerator?ref=v1.6.1" + source = "github.com/aws-observability/terraform-aws-observability-accelerator?ref=v2.1.0" aws_region = "eu-west-1" eks_cluster_id = "my-eks-cluster" @@ -70,7 +73,7 @@ You can optionally reuse an existing Amazon Managed Servce for Prometheus Worksp module "aws_observability_accelerator" { # use release tags and check for the latest versions # https://github.com/aws-observability/terraform-aws-observability-accelerator/releases - source = "github.com/aws-observability/terraform-aws-observability-accelerator?ref=v1.6.1" + source = "github.com/aws-observability/terraform-aws-observability-accelerator?ref=v2.1.0" aws_region = "eu-west-1" eks_cluster_id = "my-eks-cluster" @@ -91,13 +94,13 @@ View all the configuration options in the module documentation below. ### Workload modules [Workloads modules](./modules) are provided, which essentially provide curated -metrics collection, alerting rules and Grafana dashboards. +metrics, logs, traces collection, alerting rules and Grafana dashboards. -#### Infrastructure monitoring +#### Amazon EKS monitoring ```hcl -module "workloads_infra" { - source = "aws-observability/terraform-aws-observability-accelerator/workloads/infra" +module "eks_monitoring" { + source = "github.com/aws-observability/terraform-aws-observability-accelerator//modules/eks-monitoring?ref=v2.1.0" eks_cluster_id = module.eks_observability_accelerator.eks_cluster_id @@ -106,6 +109,9 @@ module "workloads_infra" { managed_prometheus_workspace_endpoint = module.eks_observability_accelerator.managed_prometheus_workspace_endpoint managed_prometheus_workspace_region = module.eks_observability_accelerator.managed_prometheus_workspace_region + + enable_logs = true + enable_tracing = true } ``` @@ -118,17 +124,30 @@ Check the the [complete example](./examples/existing-cluster-with-base-and-infra ## Motivation -Kubernetes is a powerful and extensible container orchestration technology that allows you to deploy and manage containerized applications at scale. The extensible nature of Kubernetes also allows you to use a wide range of popular open-source tools, commonly referred to as add-ons, in Kubernetes clusters. With such a large number of tools and design choices available, building a tailored EKS cluster that meets your application’s specific needs can take a significant amount of time. It involves integrating a wide range of open-source tools and AWS services and requires deep expertise in AWS and Kubernetes. +To gain deep visibility into your workloads and environments, AWS proposes a +set of secure, scalable, highly available, production-grade managed open +source services such as Amazon Managed Service for Prometheus, Amazon Managed +Grafana and Amazon OpenSearch. + +AWS customers have asked for best-practices and guidance to collect metrics, logs +and traces from their containerized applications and microservices with ease of +deployment. Customers can use the AWS Observability Accelerator to configure their +metrics and traces collection, leveraging [AWS Distro for OpenTelemetry](https://aws-otel.github.io/), +to have opinionated dashboards and alerts available in only minutes. -AWS customers have asked for examples that demonstrate how to integrate the landscape of Kubernetes tools and make it easy for them to provision complete, opinionated EKS clusters that meet specific application requirements. Customers can use AWS Observability Accelerator to configure and deploy purpose built EKS clusters, and start onboarding workloads in days, rather than months. ## Support & Feedback -AWS Observability Accelerator for Terraform is maintained by AWS Solution Architects. It is not part of an AWS service and support is provided best-effort by the AWS Observability Accelerator community. +AWS Observability Accelerator for Terraform is maintained by AWS Solution +Architects. It is not part of an AWS service and support is provided best-effort +by the AWS Observability Accelerator community. -To post feedback, submit feature ideas, or report bugs, please use the [Issues](https://github.com/aws-observability/terraform-aws-observability-accelerator/issues) section of this GitHub repo. +To post feedback, submit feature ideas, or report bugs, please use the +[Issues](https://github.com/aws-observability/terraform-aws-observability-accelerator/issues) +section of this GitHub repo. -If you are interested in contributing, see the [Contribution guide](https://github.com/aws-observability/terraform-aws-observability-accelerator/blob/main/CONTRIBUTING.md). +If you are interested in contributing, see the +[Contribution guide](https://github.com/aws-observability/terraform-aws-observability-accelerator/blob/main/CONTRIBUTING.md). --- diff --git a/docs/concepts.md b/docs/concepts.md index fa7f567..fa5b40f 100644 --- a/docs/concepts.md +++ b/docs/concepts.md @@ -123,4 +123,6 @@ classDiagram ## Getting started with AWS Observability services -If you are new to AWS Observability services, or want to dive deeper into them, check our [One Observability Workshop](https://catalog.workshops.aws/observability/) for a hands-on experience in a self-paced environement or at an AWS venue. +If you are new to AWS Observability services, or want to dive deeper into them, +check our [One Observability Workshop](https://catalog.workshops.aws/observability/) +for a hands-on experience in a self-paced environement or at an AWS venue. diff --git a/docs/eks/logs.md b/docs/eks/logs.md new file mode 100644 index 0000000..7cc33f9 --- /dev/null +++ b/docs/eks/logs.md @@ -0,0 +1,67 @@ +# Viewing Logs + +By default, we deploy a FluentBit daemon set in the cluster to collect worker +logs for all namespaces. Logs collection can be disabled with +`enable_logs = false`. Logs are collected and exported to Amazon CloudWatch Logs, +which enables you to centralize the logs from all of your systems, applications, +and AWS services that you use, in a single, highly scalable service. + +Further configuration options are available in the [module documentation](https://github.com/aws-observability/terraform-aws-observability-accelerator/tree/main/modules/eks-monitoring#inputs). +This guide shows how you can leverage CloudWatch Logs in Amazon Managed Grafana +for your cluster and application logs. + +## Using CloudWatch Logs as data source in Grafana + +Follow [the documentation](https://docs.aws.amazon.com/grafana/latest/userguide/using-amazon-cloudwatch-in-AMG.html) +to enable Amazon CloudWatch as a data source. Make sure to provide permissions. + +!!! tip + If you created your workspace with our [provided example](https://aws-observability.github.io/terraform-aws-observability-accelerator/helpers/managed-grafana/), + Amazon CloudWatch data source has already been setup for you. + +All logs are delivered in the following CloudWatch Log groups naming pattern: +`/aws/eks/observability-accelerator/{cluster-name}/{namespace}`. Log streams +follow `{container-name}.{pod-name}`. In Grafana, querying and analyzing logs +is done with [CloudWatch Logs Insights](https://docs.aws.amazon.com/AmazonCloudWatch/latest/logs/AnalyzingLogData.html) + +### Example - ADOT collector logs + +Select one or many log groups and run the following query. The example below, +queries AWS Distro for OpenTelemetry (ADOT) logs + +```console +fields @timestamp, log +| order @timestamp desc +| limit 100 +``` + +Screenshot 2023-03-27 at 19 08 35 + + +### Example - Using time series visualizations + +[CloudWatch Logs syntax](https://docs.aws.amazon.com/AmazonCloudWatch/latest/logs/CWL_QuerySyntax.html) +provide powerful functions to extract data from your logs. The `stats()` +function allows you to calculate aggregate statistics with log field values. +This is useful to have visualization on non-metric data from your applications. + +In the example below, we use the following query to graph the number of metrics +collected by the ADOT collector + +```console +fields @timestamp, log +| parse log /"#metrics": (?\d+)}/ +| stats avg(metrics_count) by bin(5m) +| limit 100 +``` + +!!! tip + You can add logs in your dashboards with logs panel types or time series + depending on your query results type. + +image + +!!! warning + Querying CloudWatch logs will incur costs per GB scanned. Use small time + windows and limits in your queries. Checkout the CloudWatch + [pricing page](https://aws.amazon.com/cloudwatch/pricing/) for more infos. diff --git a/docs/index.md b/docs/index.md index f6260d9..dab1b01 100644 --- a/docs/index.md +++ b/docs/index.md @@ -5,32 +5,36 @@ Welcome to the AWS Observability Accelerator for Terraform! The AWS Observability Accelerator for Terraform is a set of opinionated modules to help you set up observability for your AWS environments with AWS-managed observability services such as Amazon Managed Service for Prometheus, -Amazon Managed Grafana and AWS Distro for OpenTelemetry (ADOT). +Amazon Managed Grafana, AWS Distro for OpenTelemetry (ADOT) and Amazon CloudWatch. -We provide curated metrics, traces collection, alerting rules and Grafana dashboards -for your EKS infrastructure, Java/JMX, NGINX based workloads and custom applications. +We provide curated metrics, logs, traces collection, alerting rules and Grafana +dashboards for your EKS infrastructure, Java/JMX, NGINX based workloads and +your custom applications. + +You also can monitor your Amazon Managed Service for Prometheus workspaces ingestion, +costs, active series with [this module](https://aws-observability.github.io/terraform-aws-observability-accelerator/workloads/managed-prometheus/). image ## Getting started -This project provides a set of Terraform modules to enable metrics and traces collection, -dashboards and alerts for monitoring: +This project provides a set of Terraform modules to enable metrics, logs and +traces collection, dashboards and alerts for monitoring: -- Amazon EKS clusters infrastructure +- Amazon EKS clusters infrastructure and applications - NGINX workloads (running on Amazon EKS) - Java/JMX workloads (running on Amazon EKS) - Amazon Managed Service for Prometheus workspaces with Amazon CloudWatch -These modules can be directly configured in your existing Terraform configurations or ready -to be deployed in our packaged +These modules can be directly configured in your existing Terraform +configurations or ready to be deployed in our packaged [examples](https://github.com/aws-observability/terraform-aws-observability-accelerator/tree/main/examples) !!! tip We have supporting examples for quick setup such as: - Creating a new Amazon EKS cluster and a VPC - - Creating and configure an Amazon Managed Grafana workspace with SSO (coming soon) + - Creating and configure an Amazon Managed Grafana workspace with SSO ## Motivation diff --git a/examples/existing-cluster-java/main.tf b/examples/existing-cluster-java/main.tf index 1d3cb48..cb59200 100644 --- a/examples/existing-cluster-java/main.tf +++ b/examples/existing-cluster-java/main.tf @@ -84,6 +84,8 @@ module "eks_monitoring" { scrape_sample_limit = 2000 } + enable_logs = true + tags = local.tags depends_on = [ diff --git a/examples/existing-cluster-nginx/main.tf b/examples/existing-cluster-nginx/main.tf index 9d3c373..893f268 100644 --- a/examples/existing-cluster-nginx/main.tf +++ b/examples/existing-cluster-nginx/main.tf @@ -73,6 +73,8 @@ module "eks_monitoring" { managed_prometheus_workspace_endpoint = module.aws_observability_accelerator.managed_prometheus_workspace_endpoint managed_prometheus_workspace_region = module.aws_observability_accelerator.managed_prometheus_workspace_region + enable_logs = true + tags = local.tags depends_on = [ diff --git a/examples/existing-cluster-with-base-and-infra/main.tf b/examples/existing-cluster-with-base-and-infra/main.tf index bdd4648..2bee73d 100644 --- a/examples/existing-cluster-with-base-and-infra/main.tf +++ b/examples/existing-cluster-with-base-and-infra/main.tf @@ -89,6 +89,8 @@ module "eks_monitoring" { global_scrape_timeout = "15s" } + enable_logs = true + tags = local.tags depends_on = [ diff --git a/mkdocs.yml b/mkdocs.yml index 3eae2b7..92784f2 100644 --- a/mkdocs.yml +++ b/mkdocs.yml @@ -29,6 +29,7 @@ nav: - Infrastructure monitoring: eks/index.md - Java/JMX: eks/java.md - Nginx: eks/nginx.md + - Viewing logs: eks/logs.md - Teardown: eks/destroy.md - Monitoring Managed Service for Prometheus Workspaces: workloads/managed-prometheus.md - Supporting Examples: diff --git a/modules/eks-monitoring/README.md b/modules/eks-monitoring/README.md index 899e9b3..9615f9c 100644 --- a/modules/eks-monitoring/README.md +++ b/modules/eks-monitoring/README.md @@ -2,11 +2,12 @@ This module provides EKS cluster monitoring with the following resources: -- AWS Distro For OpenTelemetry Operator and Collector +- AWS Distro For OpenTelemetry Operator and Collector for Metrics and Traces +- Logs with [AWS for FluentBit](https://github.com/aws/aws-for-fluent-bit) - AWS Managed Grafana Dashboard and data source - Alerts and recording rules with AWS Managed Service for Prometheus -This module is inspired from the open source [kube-prometheus-stack](https://github.com/prometheus-community/helm-charts/tree/main/charts/kube-prometheus-stack) +This module makes use of the open source [kube-prometheus-stack](https://github.com/prometheus-community/helm-charts/tree/main/charts/kube-prometheus-stack) ## Requirements @@ -32,7 +33,8 @@ This module is inspired from the open source [kube-prometheus-stack](https://git | Name | Source | Version | |------|--------|---------| -| [helm\_addon](#module\_helm\_addon) | github.com/aws-ia/terraform-aws-eks-blueprints//modules/kubernetes-addons/helm-addon | v4.13.1 | +| [fluentbit\_logs](#module\_fluentbit\_logs) | ./add-ons/aws-for-fluentbit | n/a | +| [helm\_addon](#module\_helm\_addon) | github.com/aws-ia/terraform-aws-eks-blueprints//modules/kubernetes-addons/helm-addon | v4.26.0 | | [java\_monitoring](#module\_java\_monitoring) | ./patterns/java | n/a | | [nginx\_monitoring](#module\_nginx\_monitoring) | ./patterns/nginx | n/a | | [operator](#module\_operator) | ./add-ons/adot-operator | n/a | @@ -70,6 +72,7 @@ This module is inspired from the open source [kube-prometheus-stack](https://git | [enable\_dashboards](#input\_enable\_dashboards) | Enables or disables curated dashboards | `bool` | `true` | no | | [enable\_java](#input\_enable\_java) | Enable Java workloads monitoring, alerting and default dashboards | `bool` | `false` | no | | [enable\_kube\_state\_metrics](#input\_enable\_kube\_state\_metrics) | Enables or disables Kube State metrics exporter. Disabling this might affect some data in the dashboards | `bool` | `true` | no | +| [enable\_logs](#input\_enable\_logs) | Using AWS For FluentBit to collect cluster and application logs to Amazon CloudWatch | `bool` | `true` | no | | [enable\_nginx](#input\_enable\_nginx) | Enable NGINX workloads monitoring, alerting and default dashboards | `bool` | `false` | no | | [enable\_node\_exporter](#input\_enable\_node\_exporter) | Enables or disables Node exporter. Disabling this might affect some data in the dashboards | `bool` | `true` | no | | [enable\_tracing](#input\_enable\_tracing) | (Experimental) Enables tracing with AWS X-Ray. This changes the deploy mode of the collector to daemon set. Requirement: adot add-on <= 0.58-build.0 | `bool` | `false` | no | @@ -78,6 +81,7 @@ This module is inspired from the open source [kube-prometheus-stack](https://git | [irsa\_iam\_role\_path](#input\_irsa\_iam\_role\_path) | IAM role path for IRSA roles | `string` | `"/"` | no | | [java\_config](#input\_java\_config) | Configuration object for Java/JMX monitoring |
object({
enable_alerting_rules = bool
scrape_sample_limit = number
})
|
{
"enable_alerting_rules": true,
"scrape_sample_limit": 1000
}
| no | | [ksm\_config](#input\_ksm\_config) | Kube State metrics configuration |
object({
create_namespace = bool
k8s_namespace = string
helm_chart_name = string
helm_chart_version = string
helm_release_name = string
helm_repo_url = string
helm_settings = map(string)
helm_values = map(any)

scrape_interval = string
scrape_timeout = string
})
|
{
"create_namespace": true,
"helm_chart_name": "kube-state-metrics",
"helm_chart_version": "4.24.0",
"helm_release_name": "kube-state-metrics",
"helm_repo_url": "https://prometheus-community.github.io/helm-charts",
"helm_settings": {},
"helm_values": {},
"k8s_namespace": "kube-system",
"scrape_interval": "60s",
"scrape_timeout": "15s"
}
| no | +| [logs\_config](#input\_logs\_config) | Configuration object for logs collection |
object({
cw_log_retention_days = number
})
|
{
"cw_log_retention_days": 90
}
| no | | [managed\_prometheus\_workspace\_endpoint](#input\_managed\_prometheus\_workspace\_endpoint) | Amazon Managed Prometheus Workspace Endpoint | `string` | `""` | no | | [managed\_prometheus\_workspace\_id](#input\_managed\_prometheus\_workspace\_id) | Amazon Managed Prometheus Workspace ID | `string` | `null` | no | | [managed\_prometheus\_workspace\_region](#input\_managed\_prometheus\_workspace\_region) | Amazon Managed Prometheus Workspace's Region | `string` | `null` | no | diff --git a/modules/eks-monitoring/add-ons/aws-for-fluentbit/README.md b/modules/eks-monitoring/add-ons/aws-for-fluentbit/README.md new file mode 100644 index 0000000..8a5b846 --- /dev/null +++ b/modules/eks-monitoring/add-ons/aws-for-fluentbit/README.md @@ -0,0 +1,54 @@ +# AWS for Fluent Bit + +Fluent Bit is an open source Log Processor and Forwarder which allows you to collect any data like metrics and logs from different sources, enrich them with filters and send them to multiple destinations. +AWS provides a Fluent Bit image with plugins for CloudWatch Logs, Kinesis Data Firehose, Kinesis Data Stream and Amazon OpenSearch Service. + +This add-on is configured to stream the worker node logs to CloudWatch Logs by default. It can be configured to stream the logs to additional destinations like Kinesis Data Firehose, Kinesis Data Streams and Amazon OpenSearch Service by passing the custom `values.yaml`. +See this [Helm Chart](https://github.com/aws/eks-charts/tree/master/stable/aws-for-fluent-bit) for more details. + + +## Requirements + +| Name | Version | +|------|---------| +| [terraform](#requirement\_terraform) | >= 1.0.0 | +| [aws](#requirement\_aws) | >= 3.72 | + +## Providers + +| Name | Version | +|------|---------| +| [aws](#provider\_aws) | >= 3.72 | + +## Modules + +| Name | Source | Version | +|------|--------|---------| +| [helm\_addon](#module\_helm\_addon) | github.com/aws-ia/terraform-aws-eks-blueprints//modules/kubernetes-addons/helm-addon | v4.26.0 | + +## Resources + +| Name | Type | +|------|------| +| [aws_iam_policy.aws_for_fluent_bit](https://registry.terraform.io/providers/hashicorp/aws/latest/docs/resources/iam_policy) | resource | +| [aws_iam_policy_document.irsa](https://registry.terraform.io/providers/hashicorp/aws/latest/docs/data-sources/iam_policy_document) | data source | + +## Inputs + +| Name | Description | Type | Default | Required | +|------|-------------|------|---------|:--------:| +| [addon\_context](#input\_addon\_context) | Input configuration for the addon |
object({
aws_caller_identity_account_id = string
aws_caller_identity_arn = string
aws_eks_cluster_endpoint = string
aws_partition_id = string
aws_region_name = string
eks_cluster_id = string
eks_oidc_issuer_url = string
eks_oidc_provider_arn = string
tags = map(string)
irsa_iam_role_path = string
irsa_iam_permissions_boundary = string
})
| n/a | yes | +| [cw\_log\_retention\_days](#input\_cw\_log\_retention\_days) | FluentBit CloudWatch Log group retention period | `number` | `90` | no | +| [helm\_config](#input\_helm\_config) | Helm provider config aws\_for\_fluent\_bit. | `any` | `{}` | no | +| [irsa\_policies](#input\_irsa\_policies) | Additional IAM policies for a IAM role for service accounts | `list(string)` | `[]` | no | +| [manage\_via\_gitops](#input\_manage\_via\_gitops) | Determines if the add-on should be managed via GitOps. | `bool` | `false` | no | + +## Outputs + +| Name | Description | +|------|-------------| +| [irsa\_arn](#output\_irsa\_arn) | IAM role ARN for the service account | +| [irsa\_name](#output\_irsa\_name) | IAM role name for the service account | +| [release\_metadata](#output\_release\_metadata) | Map of attributes of the Helm release metadata | +| [service\_account](#output\_service\_account) | Name of Kubernetes service account | + diff --git a/modules/eks-monitoring/add-ons/aws-for-fluentbit/data.tf b/modules/eks-monitoring/add-ons/aws-for-fluentbit/data.tf new file mode 100644 index 0000000..1da5b7f --- /dev/null +++ b/modules/eks-monitoring/add-ons/aws-for-fluentbit/data.tf @@ -0,0 +1,21 @@ +data "aws_iam_policy_document" "irsa" { + statement { + sid = "PutLogEvents" + effect = "Allow" + resources = ["arn:${var.addon_context.aws_partition_id}:logs:${var.addon_context.aws_region_name}:${var.addon_context.aws_caller_identity_account_id}:log-group:*:log-stream:*"] + actions = ["logs:PutLogEvents"] + } + + statement { + sid = "CreateCWLogs" + effect = "Allow" + resources = ["arn:${var.addon_context.aws_partition_id}:logs:${var.addon_context.aws_region_name}:${var.addon_context.aws_caller_identity_account_id}:log-group:*"] + + actions = [ + "logs:CreateLogGroup", + "logs:CreateLogStream", + "logs:DescribeLogGroups", + "logs:DescribeLogStreams", + ] + } +} diff --git a/modules/eks-monitoring/add-ons/aws-for-fluentbit/locals.tf b/modules/eks-monitoring/add-ons/aws-for-fluentbit/locals.tf new file mode 100644 index 0000000..a53ce88 --- /dev/null +++ b/modules/eks-monitoring/add-ons/aws-for-fluentbit/locals.tf @@ -0,0 +1,47 @@ +locals { + name = "aws-for-fluent-bit" + service_account = try(var.helm_config.service_account, "${local.name}-sa") + + set_values = [ + { + name = "serviceAccount.name" + value = local.service_account + }, + { + name = "serviceAccount.create" + value = false + } + ] + + # https://github.com/aws/eks-charts/blob/master/stable/aws-for-fluent-bit/Chart.yaml + default_helm_config = { + name = local.name + chart = local.name + repository = "https://aws.github.io/eks-charts" + version = "0.1.24" + namespace = local.name + values = local.default_helm_values + description = "aws-for-fluentbit Helm Chart deployment configuration" + } + + helm_config = merge( + local.default_helm_config, + var.helm_config + ) + + default_helm_values = [templatefile("${path.module}/values.yaml", { + aws_region = var.addon_context.aws_region_name + cluster_name = var.addon_context.eks_cluster_id + log_retention_days = var.cw_log_retention_days + service_account = local.service_account + })] + + irsa_config = { + kubernetes_namespace = local.helm_config["namespace"] + kubernetes_service_account = local.service_account + create_kubernetes_namespace = try(local.helm_config["create_namespace"], true) + create_kubernetes_service_account = true + create_service_account_secret_token = try(local.helm_config["create_service_account_secret_token"], false) + irsa_iam_policies = concat([aws_iam_policy.aws_for_fluent_bit.arn], var.irsa_policies) + } +} diff --git a/modules/eks-monitoring/add-ons/aws-for-fluentbit/main.tf b/modules/eks-monitoring/add-ons/aws-for-fluentbit/main.tf new file mode 100644 index 0000000..e93fcec --- /dev/null +++ b/modules/eks-monitoring/add-ons/aws-for-fluentbit/main.tf @@ -0,0 +1,15 @@ +module "helm_addon" { + source = "github.com/aws-ia/terraform-aws-eks-blueprints//modules/kubernetes-addons/helm-addon?ref=v4.26.0" + manage_via_gitops = var.manage_via_gitops + set_values = local.set_values + helm_config = local.helm_config + irsa_config = local.irsa_config + addon_context = var.addon_context +} + +resource "aws_iam_policy" "aws_for_fluent_bit" { + name = "${var.addon_context.eks_cluster_id}-fluentbit" + description = "IAM Policy for AWS for FluentBit" + policy = data.aws_iam_policy_document.irsa.json + tags = var.addon_context.tags +} diff --git a/modules/eks-monitoring/add-ons/aws-for-fluentbit/outputs.tf b/modules/eks-monitoring/add-ons/aws-for-fluentbit/outputs.tf new file mode 100644 index 0000000..37b305f --- /dev/null +++ b/modules/eks-monitoring/add-ons/aws-for-fluentbit/outputs.tf @@ -0,0 +1,19 @@ +output "release_metadata" { + description = "Map of attributes of the Helm release metadata" + value = module.helm_addon.release_metadata +} + +output "irsa_arn" { + description = "IAM role ARN for the service account" + value = module.helm_addon.irsa_arn +} + +output "irsa_name" { + description = "IAM role name for the service account" + value = module.helm_addon.irsa_name +} + +output "service_account" { + description = "Name of Kubernetes service account" + value = module.helm_addon.service_account +} diff --git a/modules/eks-monitoring/add-ons/aws-for-fluentbit/values.yaml b/modules/eks-monitoring/add-ons/aws-for-fluentbit/values.yaml new file mode 100644 index 0000000..899063b --- /dev/null +++ b/modules/eks-monitoring/add-ons/aws-for-fluentbit/values.yaml @@ -0,0 +1,16 @@ +serviceAccount: + create: false + name: ${service_account} + +cloudWatch: + enabled: false + +cloudWatchLogs: + enabled: true + region: ${aws_region} + # logGroupName is a fallback to failed parsing + logGroupName: /aws/eks/observability-accelerator/workloads + logGroupTemplate: /aws/eks/observability-accelerator/${cluster_name}/$kubernetes['namespace_name'] + logStreamTemplate: $kubernetes['container_name'].$kubernetes['pod_name'] + log_key: log + log_retention_days: ${log_retention_days} diff --git a/modules/eks-monitoring/add-ons/aws-for-fluentbit/variables.tf b/modules/eks-monitoring/add-ons/aws-for-fluentbit/variables.tf new file mode 100644 index 0000000..2e4ef43 --- /dev/null +++ b/modules/eks-monitoring/add-ons/aws-for-fluentbit/variables.tf @@ -0,0 +1,40 @@ +variable "helm_config" { + description = "Helm provider config aws_for_fluent_bit." + type = any + default = {} +} + +variable "cw_log_retention_days" { + description = "FluentBit CloudWatch Log group retention period" + type = number + default = 90 +} + +variable "manage_via_gitops" { + type = bool + description = "Determines if the add-on should be managed via GitOps." + default = false +} + +variable "irsa_policies" { + description = "Additional IAM policies for a IAM role for service accounts" + type = list(string) + default = [] +} + +variable "addon_context" { + description = "Input configuration for the addon" + type = object({ + aws_caller_identity_account_id = string + aws_caller_identity_arn = string + aws_eks_cluster_endpoint = string + aws_partition_id = string + aws_region_name = string + eks_cluster_id = string + eks_oidc_issuer_url = string + eks_oidc_provider_arn = string + tags = map(string) + irsa_iam_role_path = string + irsa_iam_permissions_boundary = string + }) +} diff --git a/modules/eks-monitoring/add-ons/aws-for-fluentbit/versions.tf b/modules/eks-monitoring/add-ons/aws-for-fluentbit/versions.tf new file mode 100644 index 0000000..f92f41b --- /dev/null +++ b/modules/eks-monitoring/add-ons/aws-for-fluentbit/versions.tf @@ -0,0 +1,10 @@ +terraform { + required_version = ">= 1.0.0" + + required_providers { + aws = { + source = "hashicorp/aws" + version = ">= 3.72" + } + } +} diff --git a/modules/eks-monitoring/main.tf b/modules/eks-monitoring/main.tf index 80d51b7..9373a33 100644 --- a/modules/eks-monitoring/main.tf +++ b/modules/eks-monitoring/main.tf @@ -44,7 +44,7 @@ resource "helm_release" "prometheus_node_exporter" { } module "helm_addon" { - source = "github.com/aws-ia/terraform-aws-eks-blueprints//modules/kubernetes-addons/helm-addon?ref=v4.13.1" + source = "github.com/aws-ia/terraform-aws-eks-blueprints//modules/kubernetes-addons/helm-addon?ref=v4.26.0" helm_config = merge( { @@ -169,3 +169,11 @@ module "nginx_monitoring" { enable_alerting_rules = var.nginx_config.enable_alerting_rules dashboards_folder_id = var.dashboards_folder_id } + +module "fluentbit_logs" { + source = "./add-ons/aws-for-fluentbit" + count = var.enable_logs ? 1 : 0 + + cw_log_retention_days = var.logs_config.cw_log_retention_days + addon_context = local.context +} diff --git a/modules/eks-monitoring/otel-config/templates/opentelemetrycollector.yaml b/modules/eks-monitoring/otel-config/templates/opentelemetrycollector.yaml index 21f6ab4..0f32f58 100644 --- a/modules/eks-monitoring/otel-config/templates/opentelemetrycollector.yaml +++ b/modules/eks-monitoring/otel-config/templates/opentelemetrycollector.yaml @@ -40,7 +40,6 @@ spec: scrape_timeout: {{ .Values.globalScrapeTimeout }} external_labels: cluster: {{ .Values.ekscluster }} - account_id: {{ .Values.accountId }} region: {{ .Values.region }} scrape_configs: - job_name: 'kubernetes-kubelet' diff --git a/modules/eks-monitoring/variables.tf b/modules/eks-monitoring/variables.tf index 0de7829..383f6f5 100644 --- a/modules/eks-monitoring/variables.tf +++ b/modules/eks-monitoring/variables.tf @@ -247,3 +247,21 @@ variable "nginx_config" { prometheus_metrics_endpoint = "metrics" } } + +variable "enable_logs" { + description = "Using AWS For FluentBit to collect cluster and application logs to Amazon CloudWatch" + type = bool + default = true +} + +variable "logs_config" { + description = "Configuration object for logs collection" + type = object({ + cw_log_retention_days = number + }) + + default = { + # Valid values are [1, 3, 5, 7, 14, 30, 60, 90, 120, 150, 180, 365, 400, 545, 731, 1827, 3653] + cw_log_retention_days = 90 + } +}