diff --git a/terraform/azure/aks/main.tf b/terraform/azure/aks/main.tf index 90fb4e0f4..0722a5108 100644 --- a/terraform/azure/aks/main.tf +++ b/terraform/azure/aks/main.tf @@ -67,12 +67,29 @@ locals { service_account_name = "cloudwatch-agent" cwagent_role_name = "cwa-aks-integ-role-${module.common.testing_id}" + # test_mode == "containerinsights" swaps the OTLP load-gen topology for the CI node+cluster topology. + is_ci = var.test_mode == "containerinsights" + # Must match serviceName in test/azure/aks/aks_test.go -- the test derives the expected log stream # and the trace query filter from it. load_gen_service_name = "aks-otlp-test-service" load_gen_duration_seconds = 180 } +# Couple test_mode and test_dir so a caller can't run one suite against the other's topology +# (e.g. test_mode=containerinsights with the default:otel test_dir). Fails the plan on a mismatch. +resource "terraform_data" "validate_test_mode_dir" { + lifecycle { + precondition { + condition = ( + (var.test_mode == "otlp" && var.test_dir == "./test/azure/aks") || + (var.test_mode == "containerinsights" && var.test_dir == "./test/azure/aks/containerinsights") + ) + error_message = "test_dir must match test_mode: \"otlp\" -> \"./test/azure/aks\", \"containerinsights\" -> \"./test/azure/aks/containerinsights\"." + } + } +} + data "aws_iam_policy_document" "cwagent_assume_role" { statement { effect = "Allow" @@ -149,6 +166,24 @@ resource "kubernetes_cluster_role" "cwagent" { resources = ["jobs"] verbs = ["list", "watch", "get"] } + + # Container Insights only: kubelet-scrape RBAC so the AKS kubelet authorizer allows + # kubeletstats (/stats/summary) and cadvisor (/metrics/cadvisor), plus node /metrics. + dynamic "rule" { + for_each = local.is_ci ? [1] : [] + content { + api_groups = [""] + resources = ["nodes/proxy", "nodes/stats", "nodes/metrics"] + verbs = ["get", "list", "watch"] + } + } + dynamic "rule" { + for_each = local.is_ci ? [1] : [] + content { + non_resource_urls = ["/metrics"] + verbs = ["get", "list", "watch"] + } + } } resource "kubernetes_cluster_role_binding" "cwagent" { @@ -253,9 +288,14 @@ resource "kubernetes_daemon_set_v1" "cwagent" { name = "RUN_IN_AKS" value = "True" } - env { - name = "USE_DEFAULT_CONFIG" - value = "otel" + # OTLP mode uses the built-in default:otel config. CI mode drops it so the agent + # translates the JSON mounted at /etc/cwagentconfig below. + dynamic "env" { + for_each = local.is_ci ? [] : [1] + content { + name = "USE_DEFAULT_CONFIG" + value = "otel" + } } env { name = "K8S_NODE_NAME" @@ -284,6 +324,32 @@ resource "kubernetes_daemon_set_v1" "cwagent" { mount_path = "/rootfs" read_only = true } + dynamic "volume_mount" { + for_each = local.is_ci ? [1] : [] + content { + name = "cwagentconfig" + mount_path = "/etc/cwagentconfig" + read_only = true + } + } + dynamic "volume_mount" { + for_each = local.is_ci ? [1] : [] + content { + name = "agent-client-cert" + mount_path = "/etc/amazon-cloudwatch-observability-agent-client-cert" + read_only = true + } + } + # Container Insights only: host container logs for the filelog receiver + # (/var/log/containers/*.log -> /var/log/pods). + dynamic "volume_mount" { + for_each = local.is_ci ? [1] : [] + content { + name = "varlog" + mount_path = "/var/log" + read_only = true + } + } } volume { @@ -304,6 +370,33 @@ resource "kubernetes_daemon_set_v1" "cwagent" { path = "/" } } + dynamic "volume" { + for_each = local.is_ci ? [1] : [] + content { + name = "cwagentconfig" + config_map { + name = kubernetes_config_map.ci_node[0].metadata[0].name + } + } + } + dynamic "volume" { + for_each = local.is_ci ? [1] : [] + content { + name = "varlog" + host_path { + path = "/var/log" + } + } + } + dynamic "volume" { + for_each = local.is_ci ? [1] : [] + content { + name = "agent-client-cert" + secret { + secret_name = kubernetes_secret.ci_scrape_ca[0].metadata[0].name + } + } + } } } } @@ -319,6 +412,7 @@ resource "kubernetes_daemon_set_v1" "cwagent" { ##################################################################### # See otlp_load_generator.sh for what the payloads carry and why. resource "kubernetes_job_v1" "otlp_load" { + count = local.is_ci ? 0 : 1 metadata { name = "otlp-load-generator" namespace = kubernetes_namespace.cwagent.metadata[0].name @@ -375,10 +469,11 @@ resource "null_resource" "agent_diagnostics" { command = <<-EOT kubectl --kubeconfig='${local_sensitive_file.kubeconfig.filename}' get pods -n amazon-cloudwatch -o wide || true kubectl --kubeconfig='${local_sensitive_file.kubeconfig.filename}' logs -n amazon-cloudwatch -l app=cloudwatch-agent --tail=200 --prefix || true + kubectl --kubeconfig='${local_sensitive_file.kubeconfig.filename}' logs -n amazon-cloudwatch -l app=cloudwatch-agent-cluster-scraper --tail=200 --prefix || true EOT } - depends_on = [kubernetes_job_v1.otlp_load] + depends_on = [kubernetes_daemon_set_v1.cwagent, kubernetes_job_v1.otlp_load] } ##################################################################### @@ -401,5 +496,462 @@ resource "null_resource" "integration_test" { } } - depends_on = [kubernetes_job_v1.otlp_load, null_resource.agent_diagnostics] + depends_on = [kubernetes_daemon_set_v1.cwagent, kubernetes_job_v1.otlp_load, kubernetes_deployment_v1.cluster_scraper, null_resource.ci_extra_manifests, null_resource.agent_diagnostics] +} + +##################################################################### +# Container Insights topology (test_mode == "containerinsights"). +# All resources below are count-gated so OTLP-mode applies are unchanged. +##################################################################### + +# Agent JSON configs, cluster_name placeholder replaced with the real cluster name. +resource "kubernetes_config_map" "ci_node" { + count = local.is_ci ? 1 : 0 + metadata { + name = "cwagentconfig" + namespace = kubernetes_namespace.cwagent.metadata[0].name + } + data = { + "cwagentconfig.json" = replace( + file("${path.module}/../../../${var.test_dir}/resources/ci_node.json"), + "AKS_CLUSTER_NAME", azurerm_kubernetes_cluster.cwagent.name, + ) + } +} + +resource "kubernetes_config_map" "ci_cluster" { + count = local.is_ci ? 1 : 0 + metadata { + name = "cwagentconfig-cluster-scraper" + namespace = kubernetes_namespace.cwagent.metadata[0].name + } + data = { + "cwagentconfig.json" = replace( + file("${path.module}/../../../${var.test_dir}/resources/ci_cluster.json"), + "AKS_CLUSTER_NAME", azurerm_kubernetes_cluster.cwagent.name, + ) + } +} + +##################################################################### +# Self-signed CA + server cert for the KSM / node-exporter TLS scrapes. +# The agent trusts the CA (mounted at the two agent cert paths); KSM and +# node-exporter serve the leaf via exporter-toolkit web-config. +##################################################################### +resource "tls_private_key" "ci_ca" { + count = local.is_ci ? 1 : 0 + algorithm = "RSA" + rsa_bits = 2048 +} + +resource "tls_self_signed_cert" "ci_ca" { + count = local.is_ci ? 1 : 0 + private_key_pem = tls_private_key.ci_ca[0].private_key_pem + is_ca_certificate = true + subject { + common_name = "cwa-aks-ci-ca" + } + validity_period_hours = 24 + allowed_uses = ["cert_signing", "crl_signing"] +} + +resource "tls_private_key" "ci_server" { + count = local.is_ci ? 1 : 0 + algorithm = "RSA" + rsa_bits = 2048 +} + +resource "tls_cert_request" "ci_server" { + count = local.is_ci ? 1 : 0 + private_key_pem = tls_private_key.ci_server[0].private_key_pem + subject { + common_name = "kube-state-metrics.${local.namespace}.svc" + } + dns_names = [ + "kube-state-metrics", + "kube-state-metrics.${local.namespace}.svc", + "node-exporter-service", + "node-exporter-service.${local.namespace}.svc", + ] +} + +resource "tls_locally_signed_cert" "ci_server" { + count = local.is_ci ? 1 : 0 + cert_request_pem = tls_cert_request.ci_server[0].cert_request_pem + ca_private_key_pem = tls_private_key.ci_ca[0].private_key_pem + ca_cert_pem = tls_self_signed_cert.ci_ca[0].cert_pem + validity_period_hours = 24 + allowed_uses = ["server_auth"] +} + +# CA the agent trusts. Mounted at BOTH agent cert paths (node uses the +# -client-cert path for node-exporter; cluster-scraper uses -cert for KSM). +resource "kubernetes_secret" "ci_scrape_ca" { + count = local.is_ci ? 1 : 0 + metadata { + name = "ci-scrape-ca" + namespace = kubernetes_namespace.cwagent.metadata[0].name + } + data = { + "tls-ca.crt" = tls_self_signed_cert.ci_ca[0].cert_pem + } +} + +# Server leaf + exporter-toolkit web-config shared by KSM and node-exporter. +resource "kubernetes_secret" "ci_server_cert" { + count = local.is_ci ? 1 : 0 + metadata { + name = "ci-server-cert" + namespace = kubernetes_namespace.cwagent.metadata[0].name + } + data = { + "server.crt" = tls_locally_signed_cert.ci_server[0].cert_pem + "server.key" = tls_private_key.ci_server[0].private_key_pem + "web-config.yaml" = <<-EOT + tls_server_config: + cert_file: /tls/server.crt + key_file: /tls/server.key + EOT + } +} + +##################################################################### +# kube-state-metrics (cluster metrics: kube_node_info / kube_pod_info) +##################################################################### +resource "kubernetes_cluster_role" "ksm" { + count = local.is_ci ? 1 : 0 + metadata { + name = "cwa-aks-ci-ksm-${module.common.testing_id}" + } + rule { + api_groups = [""] + resources = ["pods", "nodes", "namespaces", "services", "endpoints"] + verbs = ["list", "watch"] + } + rule { + api_groups = ["apps"] + resources = ["deployments", "replicasets", "daemonsets", "statefulsets"] + verbs = ["list", "watch"] + } + rule { + api_groups = ["batch"] + resources = ["jobs", "cronjobs"] + verbs = ["list", "watch"] + } +} + +resource "kubernetes_cluster_role_binding" "ksm" { + count = local.is_ci ? 1 : 0 + metadata { + name = "cwa-aks-ci-ksm-${module.common.testing_id}" + } + role_ref { + api_group = "rbac.authorization.k8s.io" + kind = "ClusterRole" + name = kubernetes_cluster_role.ksm[0].metadata[0].name + } + subject { + kind = "ServiceAccount" + name = kubernetes_service_account.cwagent.metadata[0].name + namespace = local.namespace + } +} + +resource "kubernetes_deployment_v1" "ksm" { + count = local.is_ci ? 1 : 0 + metadata { + name = "kube-state-metrics" + namespace = local.namespace + labels = { app = "kube-state-metrics" } + } + spec { + replicas = 1 + selector { + match_labels = { app = "kube-state-metrics" } + } + template { + metadata { + labels = { app = "kube-state-metrics" } + } + spec { + service_account_name = kubernetes_service_account.cwagent.metadata[0].name + image_pull_secrets { + name = kubernetes_secret.ecr_pull.metadata[0].name + } + container { + name = "kube-state-metrics" + image = "registry.k8s.io/kube-state-metrics/kube-state-metrics:v2.13.0" + args = [ + "--port=8443", + "--tls-config=/web/web-config.yaml", + ] + port { + name = "https" + container_port = 8443 + } + volume_mount { + name = "tls" + mount_path = "/tls" + read_only = true + } + volume_mount { + name = "web" + mount_path = "/web" + read_only = true + } + } + volume { + name = "tls" + secret { + secret_name = kubernetes_secret.ci_server_cert[0].metadata[0].name + } + } + volume { + name = "web" + secret { + secret_name = kubernetes_secret.ci_server_cert[0].metadata[0].name + items { + key = "web-config.yaml" + path = "web-config.yaml" + } + } + } + } + } + } +} + +resource "kubernetes_service" "ksm" { + count = local.is_ci ? 1 : 0 + metadata { + name = "kube-state-metrics" + namespace = local.namespace + } + spec { + selector = { app = "kube-state-metrics" } + port { + name = "https" + port = 8443 + target_port = 8443 + } + } +} + +##################################################################### +# node-exporter (node metrics: node_cpu_seconds_total, node_memory_*) +##################################################################### +resource "kubernetes_daemon_set_v1" "node_exporter" { + count = local.is_ci ? 1 : 0 + metadata { + name = "node-exporter" + namespace = local.namespace + labels = { app = "node-exporter" } + } + spec { + selector { + match_labels = { app = "node-exporter" } + } + template { + metadata { + labels = { app = "node-exporter" } + } + spec { + host_network = true + host_pid = true + image_pull_secrets { + name = kubernetes_secret.ecr_pull.metadata[0].name + } + container { + name = "node-exporter" + image = "quay.io/prometheus/node-exporter:v1.8.2" + args = [ + "--web.listen-address=:9487", + "--web.config.file=/web/web-config.yaml", + "--path.rootfs=/host/root", + ] + port { + name = "https" + container_port = 9487 + } + volume_mount { + name = "tls" + mount_path = "/tls" + read_only = true + } + volume_mount { + name = "web" + mount_path = "/web" + read_only = true + } + volume_mount { + name = "root" + mount_path = "/host/root" + read_only = true + mount_propagation = "HostToContainer" + } + } + volume { + name = "tls" + secret { + secret_name = kubernetes_secret.ci_server_cert[0].metadata[0].name + } + } + volume { + name = "web" + secret { + secret_name = kubernetes_secret.ci_server_cert[0].metadata[0].name + items { + key = "web-config.yaml" + path = "web-config.yaml" + } + } + } + volume { + name = "root" + host_path { + path = "/" + } + } + } + } + } +} + +resource "kubernetes_service" "node_exporter" { + count = local.is_ci ? 1 : 0 + metadata { + name = "node-exporter-service" + namespace = local.namespace + } + spec { + selector = { app = "node-exporter" } + port { + name = "https" + port = 9487 + target_port = 9487 + } + } +} + +##################################################################### +# Cluster-scraper agent Deployment: translates ci_cluster.json (apiserver, +# KSM, keda/karpenter). Mounts the CA at the KSM scrape path. +##################################################################### +resource "kubernetes_deployment_v1" "cluster_scraper" { + count = local.is_ci ? 1 : 0 + metadata { + name = "cloudwatch-agent-cluster-scraper" + namespace = local.namespace + labels = { app = "cloudwatch-agent-cluster-scraper" } + } + spec { + replicas = 1 + selector { + match_labels = { app = "cloudwatch-agent-cluster-scraper" } + } + template { + metadata { + labels = { app = "cloudwatch-agent-cluster-scraper" } + } + spec { + service_account_name = kubernetes_service_account.cwagent.metadata[0].name + image_pull_secrets { + name = kubernetes_secret.ecr_pull.metadata[0].name + } + container { + name = "cloudwatch-agent" + image = "${local.cwagent_image_repo}:${var.cwagent_image_tag}" + image_pull_policy = "Always" + + env { + name = "AWS_REGION" + value = var.region + } + env { + name = "AWS_WEB_IDENTITY_TOKEN_FILE" + value = "/var/run/secrets/aws/token" + } + env { + name = "AWS_ROLE_ARN" + value = aws_iam_role.cwagent.arn + } + env { + name = "RUN_IN_CONTAINER" + value = "True" + } + env { + name = "RUN_IN_AKS" + value = "True" + } + env { + name = "K8S_NODE_NAME" + value_from { + field_ref { + field_path = "spec.nodeName" + } + } + } + + volume_mount { + name = "aws-token" + mount_path = "/var/run/secrets/aws" + read_only = true + } + volume_mount { + name = "cwagentconfig" + mount_path = "/etc/cwagentconfig" + read_only = true + } + volume_mount { + name = "agent-cert" + mount_path = "/etc/amazon-cloudwatch-observability-agent-cert" + read_only = true + } + } + + volume { + name = "aws-token" + projected { + sources { + service_account_token { + audience = "sts.amazonaws.com" + expiration_seconds = 86400 + path = "token" + } + } + } + } + volume { + name = "cwagentconfig" + config_map { + name = kubernetes_config_map.ci_cluster[0].metadata[0].name + } + } + volume { + name = "agent-cert" + secret { + secret_name = kubernetes_secret.ci_scrape_ca[0].metadata[0].name + } + } + } + } + } + + depends_on = [ + kubernetes_cluster_role_binding.cwagent, + aws_iam_role_policy_attachment.cwagent_server_policy, + ] +} + +##################################################################### +# keda/karpenter stub emitters + sample app (applied via kubectl). +##################################################################### +resource "null_resource" "ci_extra_manifests" { + count = local.is_ci ? 1 : 0 + provisioner "local-exec" { + command = <<-EOT + kubectl --kubeconfig='${local_sensitive_file.kubeconfig.filename}' apply -f '${path.module}/../../../${var.test_dir}/resources/keda_karpenter.yaml' + EOT + } + depends_on = [local_sensitive_file.kubeconfig] } diff --git a/terraform/azure/aks/variables.tf b/terraform/azure/aks/variables.tf index 590b284e9..553f514e1 100644 --- a/terraform/azure/aks/variables.tf +++ b/terraform/azure/aks/variables.tf @@ -11,6 +11,19 @@ variable "test_dir" { default = "./test/azure/aks" } +# Selects which suite this apply deploys: +# "otlp" -> default:otel DaemonSet + OTLP load generator (existing behavior) +# "containerinsights" -> node DaemonSet + cluster-scraper Deployment translating CI configs +# Defaults to otlp so existing OTLP runs are unchanged. +variable "test_mode" { + type = string + default = "otlp" + validation { + condition = contains(["otlp", "containerinsights"], var.test_mode) + error_message = "test_mode must be \"otlp\" or \"containerinsights\"." + } +} + variable "cwa_github_sha" { type = string default = "" diff --git a/test/azure/aks/containerinsights/aks_containerinsights_test.go b/test/azure/aks/containerinsights/aks_containerinsights_test.go new file mode 100644 index 000000000..38db580a3 --- /dev/null +++ b/test/azure/aks/containerinsights/aks_containerinsights_test.go @@ -0,0 +1,153 @@ +// Copyright Amazon.com, Inc. or its affiliates. All Rights Reserved. +// SPDX-License-Identifier: MIT + +//go:build integration + +// Package containerinsights validates that the CloudWatch Agent translates the OTEL Container +// Insights JSON config on AKS and delivers metrics/logs to CloudWatch over the AKS +// workload-identity -> STS web-identity chain. Terraform mounts ci_node.json (role=node, +// logs enabled) and ci_cluster.json (role=cluster, keda+karpenter) into the agent workloads +// (no USE_DEFAULT_CONFIG), so the agent's own translator builds the pipelines. +package containerinsights + +import ( + "flag" + "fmt" + "log" + "os" + "testing" + "time" + + "github.com/stretchr/testify/require" + + "github.com/aws/amazon-cloudwatch-agent-test/environment" + "github.com/aws/amazon-cloudwatch-agent-test/test/otel_collect/otlpvalidation" + "github.com/aws/amazon-cloudwatch-agent-test/test/status" + "github.com/aws/amazon-cloudwatch-agent-test/util/awsservice" +) + +// validationWindow bounds every CloudWatch lookup (scrape + export + ingestion latency). +const validationWindow = 10 * time.Minute + +var env *environment.MetaData + +func TestMain(m *testing.M) { + environment.RegisterEnvironmentMetaDataFlags() + flag.Parse() + env = environment.GetEnvironmentMetaData() + // AKSClusterName is the k8s.cluster.name stamped on all telemetry; scopes this run. + if env.AKSClusterName == "" { + fmt.Fprintln(os.Stderr, "aksClusterName flag is required to scope telemetry to this cluster") + os.Exit(1) + } + os.Exit(m.Run()) +} + +// role=node: cadvisor / kubeletstats / node_exporter. +var nodeMetrics = []string{ + "container_cpu_usage_seconds_total", + "container_memory_working_set_bytes", + "k8s.node.cpu.usage", + "k8s.node.memory.working_set", + "k8s.pod.cpu.usage", + "node_cpu_seconds_total", + "node_memory_MemAvailable_bytes", +} + +// role=cluster: apiserver + kube-state-metrics. +var clusterMetrics = []string{ + "apiserver_request_total", + "kube_node_info", + "kube_pod_info", +} + +// role=cluster keda/karpenter solution pipelines (scraped from the stub emitters). +var kedaMetrics = []string{"keda_scaler_active", "keda_scaledobject_paused"} +var karpenterMetrics = []string{"karpenter_nodes_total", "karpenter_pods_state"} + +func TestAKSContainerInsights(t *testing.T) { + deadline := time.Now().Add(validationWindow) + + t.Run("NodeMetrics", func(t *testing.T) { validateMetrics(t, nodeMetrics, deadline) }) + t.Run("ClusterMetrics", func(t *testing.T) { validateMetrics(t, clusterMetrics, deadline) }) + t.Run("KedaMetrics", func(t *testing.T) { validateMetrics(t, kedaMetrics, deadline) }) + t.Run("KarpenterMetrics", func(t *testing.T) { validateMetrics(t, karpenterMetrics, deadline) }) + t.Run("NodeLogs", testNodeApplicationLogs) +} + +// validateMetrics asserts each metric is present for this cluster. cloud.platform=azure.aks +// proves the agent ran the RUN_IN_AKS translation path, not a hardcoded EKS/EC2 one. +func validateMetrics(t *testing.T, metrics []string, deadline time.Time) { + labels := map[string]string{ + "@resource.k8s.cluster.name": env.AKSClusterName, + "@resource.cloud.platform": "azure.aks", + } + + // ValidateOtlpMetricsWithLabels retries internally (~90s); wrap it in a bounded poll so a + // slow-to-propagate category keeps checking until the shared deadline instead of failing early. + const pollInterval = 15 * time.Second + allSuccessful := func(g status.TestGroupResult) bool { + if len(g.TestResults) == 0 { + return false + } + for _, r := range g.TestResults { + if r.Status != status.SUCCESSFUL { + return false + } + } + return true + } + + var group status.TestGroupResult + for { + group = otlpvalidation.ValidateOtlpMetricsWithLabels(t.Name(), env.Region, metrics, labels) + if allSuccessful(group) || !time.Now().Before(deadline) { + break + } + time.Sleep(pollInterval) + } + + for _, r := range group.TestResults { + r := r + t.Run(r.Name, func(t *testing.T) { + require.Equal(t, status.SUCCESSFUL, r.Status, + "metric %s (cluster=%s): %v", r.Name, env.AKSClusterName, r.Reason) + }) + } +} + +// testNodeApplicationLogs asserts the role=node logs pipeline delivered application logs. +// Cleans up the group only on success; on failure it is left as debugging evidence. +func testNodeApplicationLogs(t *testing.T) { + logGroup := fmt.Sprintf("/aws/otel/containerinsights/%s/application", env.AKSClusterName) + + succeeded := false + defer func() { + if succeeded { + awsservice.DeleteLogGroup(logGroup) + } + }() + + const maxRetries = 4 + const retryInterval = 30 * time.Second + var lastErr error + for attempt := 1; attempt <= maxRetries; attempt++ { + streams := awsservice.GetLogStreamNames(logGroup) + if len(streams) > 0 { + since := time.Now().Add(-validationWindow) + until := time.Now() + lastErr = awsservice.ValidateLogs(logGroup, streams[0], &since, &until, awsservice.AssertLogsNotEmpty()) + if lastErr == nil { + succeeded = true + return + } + } else { + lastErr = fmt.Errorf("no log streams in %s yet", logGroup) + } + log.Printf("[AKS_CI_Logs] attempt %d: %v", attempt, lastErr) + if attempt < maxRetries { + time.Sleep(retryInterval) + } + } + require.NoError(t, lastErr, "validating application logs in %s", logGroup) +} diff --git a/test/azure/aks/containerinsights/resources/ci_cluster.json b/test/azure/aks/containerinsights/resources/ci_cluster.json new file mode 100644 index 000000000..d64a90cc6 --- /dev/null +++ b/test/azure/aks/containerinsights/resources/ci_cluster.json @@ -0,0 +1,21 @@ +{ + "opentelemetry": { + "cluster_name": "AKS_CLUSTER_NAME", + "collect": { + "container_insights": { + "collection_interval": 30, + "role": "cluster", + "solutions": { + "keda": { + "enabled": true, + "namespace": "keda" + }, + "karpenter": { + "enabled": true, + "namespace": "karpenter" + } + } + } + } + } +} diff --git a/test/azure/aks/containerinsights/resources/ci_node.json b/test/azure/aks/containerinsights/resources/ci_node.json new file mode 100644 index 000000000..5d1624b6e --- /dev/null +++ b/test/azure/aks/containerinsights/resources/ci_node.json @@ -0,0 +1,14 @@ +{ + "opentelemetry": { + "cluster_name": "AKS_CLUSTER_NAME", + "collect": { + "container_insights": { + "collection_interval": 30, + "role": "node", + "logs": { + "enabled": true + } + } + } + } +} diff --git a/test/azure/aks/containerinsights/resources/keda_karpenter.yaml b/test/azure/aks/containerinsights/resources/keda_karpenter.yaml new file mode 100644 index 000000000..064b4f81e --- /dev/null +++ b/test/azure/aks/containerinsights/resources/keda_karpenter.yaml @@ -0,0 +1,157 @@ +# Lightweight KEDA/Karpenter stubs for the Container Insights integ tests. +# +# These are NOT the real operators. Each is a pod serving a static Prometheus +# /metrics endpoint (from a ConfigMap) that emits the metric names the +# cluster-scraper's solutions pipelines expect, in the keda/karpenter namespaces +# with the app.kubernetes.io/name labels the solutions config selects on. This +# validates the scrape that is agent-translated OTEL pipeline to CloudWatch path +# (metric presence), without provisioning real KEDA/Karpenter infrastructure. +apiVersion: v1 +kind: Namespace +metadata: + name: keda +--- +apiVersion: v1 +kind: Namespace +metadata: + name: karpenter +--- +apiVersion: v1 +kind: ConfigMap +metadata: + name: metrics + namespace: keda +data: + metrics: | + # TYPE keda_scaler_active gauge + keda_scaler_active{scaledObject="demo"} 1 + # TYPE keda_scaledobject_paused gauge + keda_scaledobject_paused{scaledObject="demo"} 0 +--- +apiVersion: apps/v1 +kind: Deployment +metadata: + name: keda-operator + namespace: keda + labels: + app.kubernetes.io/name: keda-operator +spec: + replicas: 1 + selector: + matchLabels: + app.kubernetes.io/name: keda-operator + template: + metadata: + labels: + app.kubernetes.io/name: keda-operator + spec: + containers: + - name: metrics + image: public.ecr.aws/amazonlinux/amazonlinux:2023 + command: ["/bin/sh", "-c"] + args: + - | + command -v python3 >/dev/null 2>&1 || dnf install -y python3 >/dev/null 2>&1 + exec python3 -c ' + import http.server, socketserver, socket + body = open("/etc/metrics/metrics", "rb").read() + class H(http.server.BaseHTTPRequestHandler): + def do_GET(self): + self.send_response(200) + self.send_header("Content-Type", "text/plain; version=0.0.4; charset=utf-8") + self.send_header("Content-Length", str(len(body))) + self.end_headers() + self.wfile.write(body) + def log_message(self, *a): + pass + class Server(socketserver.TCPServer): + address_family = socket.AF_INET6 + def server_bind(self): + self.socket.setsockopt(socket.IPPROTO_IPV6, socket.IPV6_V6ONLY, 0) + super().server_bind() + try: + httpd = Server(("::", 80), H) + except OSError: + httpd = socketserver.TCPServer(("", 80), H) + httpd.serve_forever() + ' + ports: + - name: metrics + containerPort: 80 + volumeMounts: + - name: metrics + mountPath: /etc/metrics + volumes: + - name: metrics + configMap: + name: metrics +--- +apiVersion: v1 +kind: ConfigMap +metadata: + name: metrics + namespace: karpenter +data: + metrics: | + # TYPE karpenter_nodes_total gauge + karpenter_nodes_total{nodepool="default"} 3 + # TYPE karpenter_pods_state gauge + karpenter_pods_state{phase="Running"} 12 +--- +apiVersion: apps/v1 +kind: Deployment +metadata: + name: karpenter + namespace: karpenter + labels: + app.kubernetes.io/name: karpenter +spec: + replicas: 1 + selector: + matchLabels: + app.kubernetes.io/name: karpenter + template: + metadata: + labels: + app.kubernetes.io/name: karpenter + spec: + containers: + - name: metrics + image: public.ecr.aws/amazonlinux/amazonlinux:2023 + command: ["/bin/sh", "-c"] + args: + - | + command -v python3 >/dev/null 2>&1 || dnf install -y python3 >/dev/null 2>&1 + exec python3 -c ' + import http.server, socketserver, socket + body = open("/etc/metrics/metrics", "rb").read() + class H(http.server.BaseHTTPRequestHandler): + def do_GET(self): + self.send_response(200) + self.send_header("Content-Type", "text/plain; version=0.0.4; charset=utf-8") + self.send_header("Content-Length", str(len(body))) + self.end_headers() + self.wfile.write(body) + def log_message(self, *a): + pass + class Server(socketserver.TCPServer): + address_family = socket.AF_INET6 + def server_bind(self): + self.socket.setsockopt(socket.IPPROTO_IPV6, socket.IPV6_V6ONLY, 0) + super().server_bind() + try: + httpd = Server(("::", 80), H) + except OSError: + httpd = socketserver.TCPServer(("", 80), H) + httpd.serve_forever() + ' + ports: + - name: http-metrics + containerPort: 80 + volumeMounts: + - name: metrics + mountPath: /etc/metrics + volumes: + - name: metrics + configMap: + name: metrics diff --git a/test/e2e/containerinsights/resources/keda_karpenter.yaml b/test/e2e/containerinsights/resources/keda_karpenter.yaml index c0417d8a9..66679fdbc 100644 --- a/test/e2e/containerinsights/resources/keda_karpenter.yaml +++ b/test/e2e/containerinsights/resources/keda_karpenter.yaml @@ -1,11 +1,11 @@ -# Lightweight KEDA/Karpenter stubs for Container Insights e2e. +# Lightweight KEDA/Karpenter stubs for the Container Insights integ tests. # -# These are NOT the real operators. Each is an nginx pod serving a static -# Prometheus /metrics endpoint (from a ConfigMap) that emits the metric names the +# These are NOT the real operators. Each is a pod serving a static Prometheus +# /metrics endpoint (from a ConfigMap) that emits the metric names the # cluster-scraper's solutions pipelines expect, in the keda/karpenter namespaces # with the app.kubernetes.io/name labels the solutions config selects on. This -# validates the scrape -> OTEL pipeline -> CloudWatch path (metric presence), -# without provisioning real KEDA/Karpenter infrastructure (IAM, provisioners, etc). +# validates the scrape, that is agent-translated OTEL pipeline to CloudWatch path +# (metric presence), without provisioning real KEDA/Karpenter infrastructure. apiVersion: v1 kind: Namespace metadata: