From 26371d5210bf4f080330d812b8a686bd4b088133 Mon Sep 17 00:00:00 2001 From: Rodrigue Koffi Date: Mon, 7 Aug 2023 20:59:18 +0200 Subject: [PATCH] Improve tracing (#195) * Dropping daemonset option * Update adot loglevel config * Enable tracing by default * Init tracing docs * Add tracing docs * Update tracing.md * Update collector endpoint * Fix tracing instructions --- docs/eks/tracing.md | 208 ++++++++++++++++++ mkdocs.yml | 3 + modules/eks-monitoring/README.md | 4 +- .../templates/opentelemetrycollector.yaml | 87 +------- modules/eks-monitoring/variables.tf | 8 +- 5 files changed, 218 insertions(+), 92 deletions(-) create mode 100644 docs/eks/tracing.md diff --git a/docs/eks/tracing.md b/docs/eks/tracing.md new file mode 100644 index 0000000..424131d --- /dev/null +++ b/docs/eks/tracing.md @@ -0,0 +1,208 @@ +# Tracing on Amazon EKS + +[Distributed tracing](https://aws-observability.github.io/observability-best-practices/signals/traces/) +helps you have end-to-end visibility between transactions in distributed nodes. +The `eks-monitoring` module is configured by default to collect traces into +[AWS X-Ray](https://docs.aws.amazon.com/xray/latest/devguide/aws-xray.html). + +The AWS Distro for OpenTelemetry collector is configured to receive traces +in the OTLP format (OTLP receiver), using the OpenTelemetry SDK or +auto-instrumentation agents. + +!!! note + To disable the tracing configuration, set up `enable_tracing = false` in + the [module configuration](https://github.com/aws-observability/terraform-aws-observability-accelerator/tree/main/modules/eks-monitoring#input_enable_tracing) + + +## Instrumentation + +Let's take a [sample application](https://github.com/aws-observability/aws-otel-community/tree/master/sample-apps/go-sample-app) +that is already instrumented with the OpenTelemetry SDK. + +!!! note + To learn more about instrumenting with OpenTelemetry, please visit the + [OpenTelemetry documentation](https://opentelemetry.io/docs/instrumentation/) + for your programming language. + +Cloning the repo + +```console +git clone https://github.com/aws-observability/aws-otel-community.git +cd aws-otel-community/sample-apps/go-sample-app +``` + + +Highlighting code sections + + +## Deploying on Amazon EKS + +Using the sample application, we will build a container image, create and push +an image on Amazon ECR. We will use a Kubernetes manifest to deploy to an EKS +cluster. + +!!! warning + The following steps require that you have an EKS cluster ready. To deploy + an EKS cluster, please visit [our example](https://aws-observability.github.io/terraform-aws-observability-accelerator/helpers/new-eks-cluster/). + +### Building container image + + +=== "amd64 linux" + + ``` console + docker build -t go-sample-app . + ``` + +=== "cross platform build" + + ``` bash + docker buildx build -t go-sample-app . --platform=linux/amd64 + ``` + +### Publishing on Amazon ECR + + +=== "using docker" + + ``` console + export ECR_REPOSITORY_URI=$(aws ecr create-repository --repository go-sample-app --query repository.repositoryUri --output text) + aws ecr get-login-password --region $AWS_REGION | docker login --username AWS --password-stdin $ECR_REPOSITORY_URI + docker tag go-sample-app:latest "${ECR_REPOSITORY_URI}:latest" + docker push "${ECR_REPOSITORY_URI}:latest" + ``` + + +## Deploying on Amazon EKS + + +``` yaml title="eks.yaml" linenums="1" +apiVersion: apps/v1 +kind: Deployment +metadata: + name: go-sample-app + namespace: default +spec: + replicas: 2 + selector: + matchLabels: + app: go-sample-app + template: + metadata: + labels: + app: go-sample-app + spec: + containers: + - name: go-sample-app + image: "${ECR_REPOSITORY_URI}:latest" # make sure to replace this variable + imagePullPolicy: Always + env: + - name: OTEL_EXPORTER_OTLP_TRACES_ENDPOINT + value: adot-collector.adot-collector-kubeprometheus.svc.cluster.local:4317 + resources: + limits: + cpu: 300m + memory: 300Mi + requests: + cpu: 100m + memory: 180Mi + ports: + - containerPort: 8080 +--- +apiVersion: v1 +kind: Service +metadata: + name: go-sample-app + namespace: default + labels: + app: go-sample-app +spec: + ports: + - protocol: TCP + port: 8080 + targetPort: 8080 + selector: + app: go-sample-app +--- +apiVersion: v1 +kind: Service +metadata: + name: go-sample-app + namespace: default +spec: + type: ClusterIP + selector: + app: go-sample-app + ports: + - protocol: TCP + port: 8080 + targetPort: 8080 +``` + +### Deploying and testing + +With the Kubernetes manifest ready, run: + +```bash +kubectl apply -f eks.yaml +``` + +You should see the pods running with the command: + +```console +kubectl get pods +NAME READY STATUS RESTARTS AGE +go-sample-app-67c48ff8c6-bdw74 1/1 Running 0 4s +go-sample-app-67c48ff8c6-t6k2j 1/1 Running 0 4s +``` + +To simulate some traffic you can forward the service port to your local host +and test a few queries + +```console +kubectl port-forward deployment/go-sample-app 8080:8080 +``` + +Test a few endpoints + +``` +curl http://localhost:8080/ +curl http://localhost:8080/outgoing-http-call +curl http://localhost:8080/aws-sdk-call +curl http://localhost:8080/outgoing-sampleapp +``` + +## Visualizing traces + +As this is a basic example, the service map doesn't have a lot of nodes, +but this shows you how to setup tracing in your application and deploying +it on Amazon EKS using the `eks-monitoring` module. + +With Flux and Grafana Operator, the `eks-monitoring` module configures +an AWS X-Ray data source on your provided Grafana workspace. Open the +Grafana explorer view and select the X-Ray data source. If you type the query +below, and select `Trace List` for **Query Type**, you should see the list +of traces occured in the selected timeframe. + +Screenshot 2023-07-20 at 21 42 30 + +You can add the service map to a dashbaord, for example a service focused +dashbaord. You can click on any of the traces to view a node map and the traces +details. + +There is a button that can take you the CloudWatch console to view the same +data. If your logs are stored on CloudWatch Logs, this page can present +all the logs in the trace details page. The CloudWatch Log Group name should +be added to the trace as an attribute. +Read more about this in our [One Observability Workshop](https://catalog.workshops.aws/observability/en-US/use-cases/trace-to-logs-java-instrumentation/concepts) + +![CloudWatch service map](https://user-images.githubusercontent.com/10175027/254973349-1028f428-c2ef-4bd2-8114-0d0961d7cdd8.png) + + +## Resoures + +- [AWS Observability Best Practices](https://aws-observability.github.io/observability-best-practices/) +- [One Observability Workshop](https://catalog.workshops.aws/observability/en-US/) +- [AWS Distro for OpenTelemetry documentation](https://aws-otel.github.io/docs/introduction) +- [AWS X-Ray user guide](https://docs.aws.amazon.com/xray/latest/devguide/aws-xray.html) +- [OpenTelemetry documentation](https://opentelemetry.io/docs/what-is-opentelemetry/) diff --git a/mkdocs.yml b/mkdocs.yml index 5071b4d..ddbcb4e 100644 --- a/mkdocs.yml +++ b/mkdocs.yml @@ -32,6 +32,7 @@ nav: - Nginx: eks/nginx.md - Istio: eks/istio.md - Viewing logs: eks/logs.md + - Tracing: eks/tracing.md - Teardown: eks/destroy.md - Monitoring Managed Service for Prometheus Workspaces: workloads/managed-prometheus.md - Supporting Examples: @@ -47,6 +48,8 @@ markdown_extensions: - codehilite - footnotes - pymdownx.critic + - pymdownx.tabbed: + alternate_style: true - pymdownx.superfences: custom_fences: - name: mermaid diff --git a/modules/eks-monitoring/README.md b/modules/eks-monitoring/README.md index bd66cd8..5c0a868 100644 --- a/modules/eks-monitoring/README.md +++ b/modules/eks-monitoring/README.md @@ -66,7 +66,7 @@ See examples using this Terraform modules in the **Amazon EKS** section of [this | Name | Description | Type | Default | Required | |------|-------------|------|---------|:--------:| -| [adot\_loglevel](#input\_adot\_loglevel) | Verbosity level for ADOT collector logs | `string` | `"warn"` | no | +| [adot\_loglevel](#input\_adot\_loglevel) | Verbosity level for ADOT collector logs. This accepts (detailed\|normal\|basic), see https://aws-otel.github.io/docs/components/misc-exporters for mor infos. | `string` | `"normal"` | no | | [custom\_metrics\_config](#input\_custom\_metrics\_config) | Configuration object to enable custom metrics collection |
map(object({
enableBasicAuth = bool
path = string
basicAuthUsername = string
basicAuthPassword = string
ports = string
droppedSeriesPrefixes = string
}))
| `null` | no | | [eks\_cluster\_id](#input\_eks\_cluster\_id) | EKS Cluster Id | `string` | n/a | yes | | [enable\_alerting\_rules](#input\_enable\_alerting\_rules) | Enables or disables Managed Prometheus alerting rules | `bool` | `true` | no | @@ -84,7 +84,7 @@ See examples using this Terraform modules in the **Amazon EKS** section of [this | [enable\_nginx](#input\_enable\_nginx) | Enable NGINX workloads monitoring, alerting and default dashboards | `bool` | `false` | no | | [enable\_node\_exporter](#input\_enable\_node\_exporter) | Enables or disables Node exporter. Disabling this might affect some data in the dashboards | `bool` | `true` | no | | [enable\_recording\_rules](#input\_enable\_recording\_rules) | Enables or disables Managed Prometheus recording rules | `bool` | `true` | no | -| [enable\_tracing](#input\_enable\_tracing) | (Experimental) Enables tracing with AWS X-Ray. This changes the deploy mode of the collector to daemon set. Requirement: adot add-on <= 0.58-build.0 | `bool` | `false` | no | +| [enable\_tracing](#input\_enable\_tracing) | Enables tracing with OTLP traces receiver to X-Ray | `bool` | `true` | no | | [flux\_config](#input\_flux\_config) | FluxCD configuration |
object({
create_namespace = bool
k8s_namespace = string
helm_chart_name = string
helm_chart_version = string
helm_release_name = string
helm_repo_url = string
helm_settings = map(string)
helm_values = map(any)
})
|
{
"create_namespace": true,
"helm_chart_name": "flux2",
"helm_chart_version": "2.7.0",
"helm_release_name": "observability-fluxcd-addon",
"helm_repo_url": "https://fluxcd-community.github.io/helm-charts",
"helm_settings": {},
"helm_values": {},
"k8s_namespace": "flux-system"
}
| no | | [flux\_gitrepository\_branch](#input\_flux\_gitrepository\_branch) | Flux GitRepository Branch | `string` | `"main"` | no | | [flux\_gitrepository\_name](#input\_flux\_gitrepository\_name) | Flux GitRepository name | `string` | `"aws-observability-accelerator"` | no | diff --git a/modules/eks-monitoring/otel-config/templates/opentelemetrycollector.yaml b/modules/eks-monitoring/otel-config/templates/opentelemetrycollector.yaml index bfbd938..53a2dd9 100644 --- a/modules/eks-monitoring/otel-config/templates/opentelemetrycollector.yaml +++ b/modules/eks-monitoring/otel-config/templates/opentelemetrycollector.yaml @@ -3,12 +3,7 @@ kind: OpenTelemetryCollector metadata: name: adot spec: - {{ if .Values.enableTracing }} - mode: daemonset - hostNetwork: true - {{ else }} mode: deployment - {{ end }} serviceAccount: adot-collector-kubeprometheus env: - name: "K8S_NODE_NAME" @@ -56,11 +51,6 @@ spec: regex: (.+) target_label: __metrics_path__ replacement: /api/v1/nodes/$${1}/proxy/metrics - {{ if .Values.enableTracing }} - - action: keep - regex: $K8S_NODE_NAME - source_labels: [__meta_kubernetes_node_name] - {{ end }} - job_name: 'kubelet' scheme: https tls_config: @@ -78,11 +68,6 @@ spec: regex: (.+) target_label: __metrics_path__ replacement: /api/v1/nodes/$${1}/proxy/metrics/cadvisor - {{ if .Values.enableTracing }} - - action: keep - regex: $K8S_NODE_NAME - source_labels: [__meta_kubernetes_node_name] - {{ end }} - job_name: 'kube-admin' scheme: https tls_config: @@ -195,11 +180,6 @@ spec: regex: "0" replacement: $$1 action: keep - {{ if .Values.enableTracing }} - - action: keep - regex: $K8S_NODE_NAME - source_labels: [__meta_kubernetes_endpoint_node_name] - {{ end }} kubernetes_sd_configs: - role: endpoints kubeconfig_file: "" @@ -302,11 +282,6 @@ spec: regex: "0" replacement: $$1 action: keep - {{ if .Values.enableTracing }} - - action: keep - regex: $K8S_NODE_NAME - source_labels: [__meta_kubernetes_endpoint_node_name] - {{ end }} kubernetes_sd_configs: - role: endpoints kubeconfig_file: "" @@ -408,11 +383,6 @@ spec: regex: "0" replacement: $$1 action: keep - {{ if .Values.enableTracing }} - - action: keep - regex: $K8S_NODE_NAME - source_labels: [__meta_kubernetes_endpoint_node_name] - {{ end }} kubernetes_sd_configs: - role: endpoints kubeconfig_file: "" @@ -530,11 +500,6 @@ spec: regex: "0" replacement: $$1 action: keep - {{ if .Values.enableTracing }} - - action: keep - regex: $K8S_NODE_NAME - source_labels: [__meta_kubernetes_endpoint_node_name] - {{ end }} kubernetes_sd_configs: - role: endpoints kubeconfig_file: "" @@ -652,11 +617,6 @@ spec: regex: "0" replacement: $$1 action: keep - {{ if .Values.enableTracing }} - - action: keep - regex: $K8S_NODE_NAME - source_labels: [__meta_kubernetes_endpoint_node_name] - {{ end }} kubernetes_sd_configs: - role: endpoints kubeconfig_file: "" @@ -773,11 +733,6 @@ spec: regex: "0" replacement: $$1 action: keep - {{ if .Values.enableTracing }} - - action: keep - regex: $K8S_NODE_NAME - source_labels: [__meta_kubernetes_endpoint_node_name] - {{ end }} kubernetes_sd_configs: - role: endpoints kubeconfig_file: "" @@ -882,11 +837,6 @@ spec: regex: "0" replacement: $$1 action: keep - {{ if .Values.enableTracing }} - - action: keep - regex: $K8S_NODE_NAME - source_labels: [__meta_kubernetes_endpoint_node_name] - {{ end }} kubernetes_sd_configs: - role: endpoints kubeconfig_file: "" @@ -993,11 +943,6 @@ spec: regex: "0" replacement: $$1 action: keep - {{ if .Values.enableTracing }} - - action: keep - regex: $K8S_NODE_NAME - source_labels: [__meta_kubernetes_endpoint_node_name] - {{ end }} kubernetes_sd_configs: - role: endpoints kubeconfig_file: "" @@ -1104,11 +1049,6 @@ spec: regex: "0" replacement: $$1 action: keep - {{ if .Values.enableTracing }} - - action: keep - regex: $K8S_NODE_NAME - source_labels: [__meta_kubernetes_endpoint_node_name] - {{ end }} kubernetes_sd_configs: - role: endpoints kubeconfig_file: "" @@ -1215,11 +1155,6 @@ spec: regex: "0" replacement: $$1 action: keep - {{ if .Values.enableTracing }} - - action: keep - regex: $K8S_NODE_NAME - source_labels: [__meta_kubernetes_endpoint_node_name] - {{ end }} kubernetes_sd_configs: - role: endpoints kubeconfig_file: "" @@ -1326,11 +1261,6 @@ spec: regex: "0" replacement: $$1 action: keep - {{ if .Values.enableTracing }} - - action: keep - regex: $K8S_NODE_NAME - source_labels: [__meta_kubernetes_endpoint_node_name] - {{ end }} kubernetes_sd_configs: - role: endpoints kubeconfig_file: "" @@ -1431,11 +1361,6 @@ spec: regex: "0" replacement: $$1 action: keep - {{ if .Values.enableTracing }} - - action: keep - regex: $K8S_NODE_NAME - source_labels: [__meta_kubernetes_endpoint_node_name] - {{ end }} kubernetes_sd_configs: - role: endpoints kubeconfig_file: "" @@ -1459,11 +1384,6 @@ spec: - action: replace source_labels: [__meta_kubernetes_endpoint_node_name] target_label: nodename - {{ if .Values.enableTracing }} - - action: keep - regex: $K8S_NODE_NAME - source_labels: [__meta_kubernetes_endpoint_node_name] - {{ end }} {{ if .Values.enableCustomMetrics }} {{- range $k, $v := fromYaml .Values.customMetrics }} - job_name: "{{ $k }}" @@ -1494,11 +1414,6 @@ spec: - action: replace source_labels: [__meta_kubernetes_pod_controller_kind] target_label: pod_controller_kind - {{ if $.Values.enableTracing }} - - action: keep - regex: $K8S_NODE_NAME - source_labels: [__meta_kubernetes_pod_node_name] - {{ end }} metric_relabel_configs: - source_labels: [ __name__ ] regex: '{{ $v.droppedSeriesPrefixes }}' @@ -1644,7 +1559,7 @@ spec: resource_to_telemetry_conversion: enabled: true logging: - loglevel: {{ .Values.adotLoglevel }} + verbosity: {{ .Values.adotLoglevel }} extensions: sigv4auth: region: {{ .Values.region }} diff --git a/modules/eks-monitoring/variables.tf b/modules/eks-monitoring/variables.tf index 6d64c13..d51b24a 100644 --- a/modules/eks-monitoring/variables.tf +++ b/modules/eks-monitoring/variables.tf @@ -34,9 +34,9 @@ variable "irsa_iam_permissions_boundary" { } variable "adot_loglevel" { - description = "Verbosity level for ADOT collector logs" + description = "Verbosity level for ADOT collector logs. This accepts (detailed|normal|basic), see https://aws-otel.github.io/docs/components/misc-exporters for mor infos." type = string - default = "warn" + default = "normal" } variable "managed_prometheus_workspace_endpoint" { @@ -202,9 +202,9 @@ variable "prometheus_config" { } variable "enable_tracing" { - description = "(Experimental) Enables tracing with AWS X-Ray. This changes the deploy mode of the collector to daemon set. Requirement: adot add-on <= 0.58-build.0" + description = "Enables tracing with OTLP traces receiver to X-Ray" type = bool - default = false + default = true } variable "tracing_config" {