From 6c3b7367a3bc1ca58af51f453c3de91abd570840 Mon Sep 17 00:00:00 2001 From: Nick Lathe Date: Thu, 17 Sep 2026 11:27:48 -0700 Subject: [PATCH] Update collector metrics Signed-off-by: Nick Lathe --- apps/monitoring/README.md | 13 ++++- apps/monitoring/chart/Chart.yaml | 2 +- apps/monitoring/chart/files/config.alloy | 60 +++++++++++++++++++++--- apps/monitoring/chart/values.yaml | 24 ++++++++-- 4 files changed, 87 insertions(+), 12 deletions(-) diff --git a/apps/monitoring/README.md b/apps/monitoring/README.md index 67473af5..a249266a 100644 --- a/apps/monitoring/README.md +++ b/apps/monitoring/README.md @@ -14,8 +14,17 @@ existing **Amazon Managed Service for Prometheus (AMP)** workspace. IAM role managed through Crossplane, scoped to writing metrics to the selected AMP workspace. -Collected metrics cover node readiness, workload replicas, pod status and -restarts, container CPU and memory usage, and collector health. +Collected metrics cover node readiness and pressure conditions, node pool +labels, allocatable capacity, workload replicas and rollout progress, HPA +state, pod phase, restarts, waiting and exit reasons, container CPU and memory +usage against requests and limits, CPU throttling, OOM kills, kubelet pod +counts and evictions, and collector health. The node root cgroup is kept from +cAdvisor so whole-node usage is available without node-exporter. + +The Alloy keep-list and the kube-state-metrics allowlist are the contract with +the Grafana dashboards in the `infrastructure` repo +(`observability/dashboards/grafana/src/dashboards/kubernetes/`). Add a metric to +both when a panel needs it; drop it from both when nothing reads it. ## Configuration diff --git a/apps/monitoring/chart/Chart.yaml b/apps/monitoring/chart/Chart.yaml index 7d7c6292..89824c09 100644 --- a/apps/monitoring/chart/Chart.yaml +++ b/apps/monitoring/chart/Chart.yaml @@ -2,7 +2,7 @@ apiVersion: v2 name: monitoring description: Kubernetes metrics sent to the existing Amazon Managed Service for Prometheus workspace type: application -version: 0.1.2 +version: 0.1.3 dependencies: - name: alloy repository: https://grafana.github.io/helm-charts diff --git a/apps/monitoring/chart/files/config.alloy b/apps/monitoring/chart/files/config.alloy index 6ba6d785..93e1dd17 100644 --- a/apps/monitoring/chart/files/config.alloy +++ b/apps/monitoring/chart/files/config.alloy @@ -74,16 +74,62 @@ prometheus.scrape "alloy" { } prometheus.relabel "v1" { - // Only the metrics consumed by the overview and rollout checks reach AMP. + // Only metrics the Kubernetes dashboards and alerts read reach AMP + // (infrastructure: observability/dashboards/grafana/src/dashboards/kubernetes/). + // kube_* names must also be on the kube-state-metrics metricAllowlist in values.yaml. rule { source_labels = ["__name__"] - regex = "up|scrape_samples_scraped|scrape_samples_post_metric_relabeling|scrape_duration_seconds|kube_node_info|kube_node_status_condition|kube_node_status_allocatable|kube_pod_info|kube_pod_status_phase|kube_pod_container_status_restarts_total|kube_pod_container_resource_requests|kube_pod_container_resource_limits|kube_deployment_spec_replicas|kube_deployment_status_replicas_available|kube_statefulset_replicas|kube_statefulset_status_replicas_ready|kube_daemonset_status_desired_number_scheduled|kube_daemonset_status_number_ready|kubelet_node_name|container_cpu_usage_seconds_total|container_memory_working_set_bytes|process_resident_memory_bytes|process_cpu_seconds_total|prometheus_remote_storage_samples_pending|prometheus_remote_storage_samples_failed_total|prometheus_remote_storage_samples_retried_total|prometheus_remote_storage_samples_total|prometheus_remote_storage_queue_highest_sent_timestamp_seconds" + regex = string.join([ + // Scrape health + "up", + "scrape_samples_post_metric_relabeling", + // kube-state-metrics: nodes + "kube_node_info", + "kube_node_labels", + "kube_node_status_condition", + "kube_node_status_allocatable", + // kube-state-metrics: pods + "kube_pod_info", + "kube_pod_status_phase", + "kube_pod_container_status_restarts_total", + "kube_pod_container_status_waiting_reason", + "kube_pod_container_status_last_terminated_reason", + "kube_pod_container_resource_requests", + "kube_pod_container_resource_limits", + // kube-state-metrics: workloads + autoscaling + "kube_deployment_spec_replicas", + "kube_deployment_status_replicas_available", + "kube_deployment_status_replicas_updated", + "kube_statefulset_replicas", + "kube_statefulset_status_replicas_ready", + "kube_statefulset_status_replicas_updated", + "kube_horizontalpodautoscaler_spec_max_replicas", + "kube_horizontalpodautoscaler_status_current_replicas", + "kube_horizontalpodautoscaler_status_desired_replicas", + // cAdvisor: usage, throttling, OOM kills (per container and node root cgroup) + "container_cpu_usage_seconds_total", + "container_memory_working_set_bytes", + "container_cpu_cfs_periods_total", + "container_cpu_cfs_throttled_periods_total", + "container_oom_events_total", + // kubelet + "kubelet_running_pods", + "kubelet_evictions", + // Alloy self: collector memory and remote-write health + "process_resident_memory_bytes", + "prometheus_remote_storage_samples_pending", + "prometheus_remote_storage_samples_failed_total", + "prometheus_remote_storage_samples_retried_total", + "prometheus_remote_storage_queue_highest_sent_timestamp_seconds", + ], "|") action = "keep" } - // Discard cgroup aggregates; keep real Kubernetes containers only. + // Discard intermediate cgroup aggregates (pod sandboxes, kubepods slices) so + // per-container sums count each container once. The node root cgroup + // (id="/") is kept: it is whole-node usage without needing node-exporter. rule { - source_labels = ["__name__", "container"] - regex = "container_(cpu_usage_seconds_total|memory_working_set_bytes);(|POD)" + source_labels = ["__name__", "container", "id"] + regex = "container_.*;(|POD);/.+" action = "drop" } rule { @@ -93,9 +139,11 @@ prometheus.relabel "v1" { replacement = "$1" } // Preserve series identity (including cgroup id and cpu); no arbitrary pod labels. + // reason: waiting / last-terminated reason. horizontalpodautoscaler: HPA name. + // label_karpenter_sh_nodepool: from kube_node_labels. eviction_signal: kubelet_evictions. rule { action = "labelkeep" - regex = "__name__|job|instance|source|node|namespace|pod|uid|container|id|cpu|condition|status|phase|resource|unit|deployment|statefulset|daemonset|component_id|component_path|remote_name|url" + regex = "__name__|job|instance|source|node|namespace|pod|uid|container|id|cpu|condition|status|phase|reason|resource|unit|deployment|statefulset|horizontalpodautoscaler|label_karpenter_sh_nodepool|eviction_signal|component_id|component_path|remote_name|url" } forward_to = [prometheus.remote_write.amp.receiver] } diff --git a/apps/monitoring/chart/values.yaml b/apps/monitoring/chart/values.yaml index 4f0a8fcd..96a7b0f5 100644 --- a/apps/monitoring/chart/values.yaml +++ b/apps/monitoring/chart/values.yaml @@ -92,22 +92,40 @@ kube-state-metrics: autosharding: enabled: false prometheusScrape: false - collectors: [nodes, pods, deployments, statefulsets, daemonsets] + # Every metric below must also be on the Alloy keep-list in files/config.alloy, + # and every label it introduces on the Alloy labelkeep, or it never reaches AMP. + # No DaemonSets run on Auto Mode (system daemons are AWS-managed), so that + # collector is off. + collectors: [nodes, pods, deployments, statefulsets, horizontalpodautoscalers] metricAllowlist: + # Nodes: inventory, Ready/pressure conditions, allocatable (capacity denominators), pool label. - kube_node_info + - kube_node_labels - kube_node_status_condition - kube_node_status_allocatable + # Pods: placement, phase, restarts, waiting/exit reasons, requests and limits. - kube_pod_info - kube_pod_status_phase - kube_pod_container_status_restarts_total + - kube_pod_container_status_waiting_reason + - kube_pod_container_status_last_terminated_reason - kube_pod_container_resource_requests - kube_pod_container_resource_limits + # Workloads: desired vs available/ready, and updated for rollout progress. - kube_deployment_spec_replicas - kube_deployment_status_replicas_available + - kube_deployment_status_replicas_updated - kube_statefulset_replicas - kube_statefulset_status_replicas_ready - - kube_daemonset_status_desired_number_scheduled - - kube_daemonset_status_number_ready + - kube_statefulset_status_replicas_updated + # Autoscaling: current vs desired vs ceiling. + - kube_horizontalpodautoscaler_spec_max_replicas + - kube_horizontalpodautoscaler_status_current_replicas + - kube_horizontalpodautoscaler_status_desired_replicas + # kube_node_labels only carries labels named here (as label_). + # karpenter.sh/nodepool → label_karpenter_sh_nodepool: system / general-purpose / frontend. + metricLabelsAllowlist: + - nodes=[karpenter.sh/nodepool] nodeSelector: kubernetes.io/os: linux tolerations: []