From 354f03452e6f0d128991ace4f95cd0cabb61185e Mon Sep 17 00:00:00 2001 From: Fareedah Hafis Date: Thu, 10 Sep 2026 14:37:11 +0000 Subject: [PATCH 1/2] Mark otel performance suite WIP, drop lower bounds, and set up stats experimentation --- generator/test_case_generator.go | 2 + .../eks/daemon/otel-performance/variables.tf | 7 +- test/otel/performance/performance_test.go | 50 +++++------ test/otel/performance/regression_test.go | 49 +++++------ .../resources/performance_thresholds.json | 4 +- test/otel/performance/setup_test.go | 86 ++++++++++++++++++- 6 files changed, 133 insertions(+), 65 deletions(-) diff --git a/generator/test_case_generator.go b/generator/test_case_generator.go index aee64a39..accaa3f3 100644 --- a/generator/test_case_generator.go +++ b/generator/test_case_generator.go @@ -538,8 +538,10 @@ var testTypeToTestConfig = map[string][]testConfig{ testDir: "./test/otel/performance", terraformDir: "terraform/eks/daemon/otel-performance", targets: map[string]map[string]struct{}{"arc": {"amd64": {}}}, + instanceType: "t3.medium", ami: "AL2023_x86_64_STANDARD", k8sVersion: "1.35", + wip: true, }, }, "eks_deployment": { diff --git a/terraform/eks/daemon/otel-performance/variables.tf b/terraform/eks/daemon/otel-performance/variables.tf index 037a5707..3e958b68 100644 --- a/terraform/eks/daemon/otel-performance/variables.tf +++ b/terraform/eks/daemon/otel-performance/variables.tf @@ -17,11 +17,8 @@ variable "cwagent_image_repo" { } variable "cwagent_image_tag" { - type = string - validation { - condition = length(var.cwagent_image_tag) > 0 - error_message = "cwagent_image_tag must be set; it is used as the regression baseline commit key." - } + type = string + default = "" } variable "helm_chart_branch" { diff --git a/test/otel/performance/performance_test.go b/test/otel/performance/performance_test.go index 3c1fa7be..02f4fd79 100644 --- a/test/otel/performance/performance_test.go +++ b/test/otel/performance/performance_test.go @@ -116,14 +116,13 @@ func podTypeLabel(podType string) string { } } -// isAllValuesWithinBound returns the average of values and an error if the set -// is empty, contains a negative value, or the average falls outside the -// threshold band [threshold*(1-errorBound), threshold*(1+errorBound)]. -func isAllValuesWithinBound(values []float64, threshold float64, errorBound float64) (float64, error) { +// checkAgainstUpperBound returns the configured statistic and errors if the set +// is empty, contains NaN or a negative value, or the statistic exceeds the upper +// bound. No lower bound: a healthy pod can idle near zero over the window. +func checkAgainstUpperBound(values []float64, threshold float64, errorBound float64, stat string) (float64, error) { if len(values) == 0 { return 0, fmt.Errorf("no values found") } - totalSum := 0.0 for _, value := range values { if math.IsNaN(value) { return 0, fmt.Errorf("values contain NaN") @@ -131,15 +130,13 @@ func isAllValuesWithinBound(values []float64, threshold float64, errorBound floa if value < 0 && threshold >= 0 { return 0, fmt.Errorf("values are not all greater than or equal to zero") } - totalSum += value } - avg := totalSum / float64(len(values)) + value := summaryStat(values, stat) upperBound := threshold * (1 + errorBound) - lowerBound := threshold * (1 - errorBound) - if threshold > 0 && (avg > upperBound || avg < lowerBound) { - return avg, fmt.Errorf("average value %f is not within bound [%f, %f]", avg, lowerBound, upperBound) + if threshold > 0 && value > upperBound { + return value, fmt.Errorf("%s value %f exceeds upper bound %f", stat, value, upperBound) } - return avg, nil + return value, nil } // TestPerformanceThresholds checks each agent pod's resource usage against the @@ -173,15 +170,14 @@ func TestPerformanceThresholds(t *testing.T) { for _, m := range thresholds.Metrics { for _, mpt := range m.PodThresholds { if mpt.PodFilter == pt.PodFilter { - lower := mpt.Threshold * (1 - thresholds.ErrorBound) upper := mpt.Threshold * (1 + thresholds.ErrorBound) switch m.Name { case "k8s.pod.cpu.utilization": - t.Logf(" (%s): Node CPU safe range: ±%.0f%% of %.2f%% of Node allocatable CPU [%.4f%%, %.4f%%]", - label, thresholds.ErrorBound*100, mpt.Threshold, lower, upper) + t.Logf(" (%s): Node CPU upper bound: %.2f%% +%.0f%% of Node allocatable CPU (%.4f%%)", + label, mpt.Threshold, thresholds.ErrorBound*100, upper) case "k8s.pod.memory.working_set": - t.Logf(" (%s): Node memory safe range: ±%.0f%% of %.2f%% of Node allocatable memory [%.4f%%, %.4f%%]", - label, thresholds.ErrorBound*100, mpt.Threshold, lower, upper) + t.Logf(" (%s): Node memory upper bound: %.2f%% +%.0f%% of Node allocatable memory (%.4f%%)", + label, mpt.Threshold, thresholds.ErrorBound*100, upper) } } } @@ -244,30 +240,28 @@ func TestPerformanceThresholds(t *testing.T) { pctValues[i] = (val / denominator) * 100 } - // Check if values are within bounds and log all the results. + // Log all stats side by side for comparison, then gate on the + // configured one (metric.Stat). Drop the all-stats line once chosen. label := podTypeLabel(podType) - avg, err := isAllValuesWithinBound(pctValues, threshold, thresholds.ErrorBound) + t.Logf(" %s (%s) [%% of %s]: %s", label, podName, resourceLabel, allStatsString(pctValues)) + value, err := checkAgainstUpperBound(pctValues, threshold, thresholds.ErrorBound, metric.Stat) - lowerBound := threshold * (1 - thresholds.ErrorBound) upperBound := threshold * (1 + thresholds.ErrorBound) - t.Logf(" %s (%s): Using %.4f%% of %s", label, podName, avg, resourceLabel) + t.Logf(" %s (%s): Gating on %s = %.4f%% of %s", label, podName, metric.Stat, value, resourceLabel) if err != nil { - if avg < lowerBound { - t.Logf(" %.4f%% %sBELOW%s range [%.4f%%, %.4f%%], Threshold Test: %sFAIL%s", - avg, colorRed, colorReset, lowerBound, upperBound, colorRed, colorReset) - } else if avg > upperBound { - t.Logf(" %.4f%% %sABOVE%s range [%.4f%%, %.4f%%], Threshold Test: %sFAIL%s", - avg, colorRed, colorReset, lowerBound, upperBound, colorRed, colorReset) + if value > upperBound { + t.Logf(" %.4f%% %sABOVE%s upper bound %.4f%%, Threshold Test: %sFAIL%s", + value, colorRed, colorReset, upperBound, colorRed, colorReset) } else { t.Logf(" %s%s%s", colorRed, err.Error(), colorReset) } failures = append(failures, fmt.Sprintf( "%s (%s) [%s]: %s", label, podName, metric.Name, err.Error())) } else { - t.Logf(" %.4f%% %sWITHIN%s range [%.4f%%, %.4f%%], Threshold Test: %sPASS%s", - avg, colorGreen, colorReset, lowerBound, upperBound, colorGreen, colorReset) + t.Logf(" %.4f%% %sWITHIN%s upper bound %.4f%%, Threshold Test: %sPASS%s", + value, colorGreen, colorReset, upperBound, colorGreen, colorReset) } t.Log("") } diff --git a/test/otel/performance/regression_test.go b/test/otel/performance/regression_test.go index bbe116b4..beb12eb0 100644 --- a/test/otel/performance/regression_test.go +++ b/test/otel/performance/regression_test.go @@ -26,14 +26,12 @@ import ( // DynamoDB table identifiers and regression thresholds. const ( - tableName = "CWAPerformanceMetrics" - useCase = "otel-containerinsights" - serviceName = "AmazonCloudWatchAgent" + tableName = "CWAPerformanceMetrics" + useCase = "otel-containerinsights" + serviceName = "AmazonCloudWatchAgent" + // Statistic used to reduce each pod's series to one value for the baseline. + regressionStat = "p95" maxRegressionPercent = 30.0 - // A drop larger than this fails too: a big decrease is either breakage/a - // missing metric or a large optimization — both warrant a human look and a - // deliberate re-baseline rather than silently lowering the bar. - maxDropPercent = 50.0 ) // PerfResult stores the worst-case (max) performance values for a run. @@ -91,9 +89,10 @@ const ( colorReset = "\033[0m" ) -// compareAndReport logs the comparison and reports whether it passed. A regression above the -// threshold fails the test; a zero baseline is treated as a corrupt/missing -// row and also fails, since agent usage is never legitimately zero. +// compareAndReport logs the comparison and reports whether it passed. Only +// growth above the threshold fails; a decrease always passes. A zero baseline is +// treated as a corrupt/missing row and also fails, since agent usage is never +// legitimately zero. func compareAndReport(t *testing.T, metricName string, previous, current float64, unit string) bool { t.Helper() if previous == 0 { @@ -103,15 +102,8 @@ func compareAndReport(t *testing.T, metricName string, previous, current float64 } changePercent := ((current - previous) / previous) * 100 if changePercent <= 0 { + // A decrease never fails: less usage is not a regression. dropPercent := -changePercent - if dropPercent > maxDropPercent { - t.Logf(" Current %s usage is %.1f%% %sLESS%s than last known usage (%.4f %s -> %.4f %s)", - metricName, dropPercent, colorRed, colorReset, previous, unit, current, unit) - t.Errorf(" %.1f%% drop > %.0f%% drop threshold — investigate (breakage/missing metric) or re-baseline if intentional, Regression Test: %sFAIL%s", - dropPercent, maxDropPercent, colorRed, colorReset) - t.Log("") - return false - } t.Logf(" Current %s usage is %.1f%% %sLESS%s than last known usage (%.4f %s -> %.4f %s)", metricName, dropPercent, colorGreen, colorReset, previous, unit, current, unit) t.Logf(" Regression Test: %sPASS%s", colorGreen, colorReset) @@ -174,16 +166,16 @@ func collectCurrentResults(t *testing.T) PerfResult { for _, v := range series.Values { require.False(t, math.IsNaN(v), "CPU series for %s contains a NaN sample", podName) } - _, max := calcStats(series.Values) + stat := summaryStat(series.Values, regressionStat) if isDaemonSetPod(podName) { sawDaemonSetCPU = true - if max > result.DaemonSetCPUMax { - result.DaemonSetCPUMax = max + if stat > result.DaemonSetCPUMax { + result.DaemonSetCPUMax = stat } } else { sawScraperCPU = true - if max > result.ScraperCPUMax { - result.ScraperCPUMax = max + if stat > result.ScraperCPUMax { + result.ScraperCPUMax = stat } } } @@ -196,17 +188,16 @@ func collectCurrentResults(t *testing.T) PerfResult { for _, v := range series.Values { require.False(t, math.IsNaN(v), "memory series for %s contains a NaN sample", podName) } - _, max := calcStats(series.Values) - maxMB := max / (1024 * 1024) + statMB := summaryStat(series.Values, regressionStat) / (1024 * 1024) if isDaemonSetPod(podName) { sawDaemonSetMem = true - if maxMB > result.DaemonSetMemMaxMB { - result.DaemonSetMemMaxMB = maxMB + if statMB > result.DaemonSetMemMaxMB { + result.DaemonSetMemMaxMB = statMB } } else { sawScraperMem = true - if maxMB > result.ScraperMemMaxMB { - result.ScraperMemMaxMB = maxMB + if statMB > result.ScraperMemMaxMB { + result.ScraperMemMaxMB = statMB } } } diff --git a/test/otel/performance/resources/performance_thresholds.json b/test/otel/performance/resources/performance_thresholds.json index e32af92d..0cf6f115 100644 --- a/test/otel/performance/resources/performance_thresholds.json +++ b/test/otel/performance/resources/performance_thresholds.json @@ -4,7 +4,7 @@ { "name": "k8s.pod.cpu.utilization", "unit": "percent_of_node", - "stat": "Average", + "stat": "p95", "pod_thresholds": [ { "pod_filter": "daemonset", @@ -21,7 +21,7 @@ { "name": "k8s.pod.memory.working_set", "unit": "percent_of_node", - "stat": "Average", + "stat": "p95", "pod_thresholds": [ { "pod_filter": "daemonset", diff --git a/test/otel/performance/setup_test.go b/test/otel/performance/setup_test.go index 3785e979..e911ee1d 100644 --- a/test/otel/performance/setup_test.go +++ b/test/otel/performance/setup_test.go @@ -9,7 +9,10 @@ import ( "context" "flag" "fmt" + "math" "os" + "sort" + "strings" "sync" "testing" "time" @@ -55,7 +58,7 @@ func fetchSharedMetrics(t *testing.T) *podMetricData { ctx := context.Background() end := time.Now() start := end.Add(-queryRangeMinutes * time.Minute) - step := 30 * time.Second + step := 1 * time.Second // Escape the cluster name before interpolating, matching the shared // helper used by the other otel suites (kubeletstats/cadvisor/gpu). @@ -101,6 +104,87 @@ func calcStats(values []float64) (float64, float64) { return avg, max } +// summaryStat reduces a series to a single value using the named statistic +// (max, p90, p95, p99, es90, es95, or average). Driven by the "stat" field in the config. +func summaryStat(values []float64, stat string) float64 { + if len(values) == 0 { + return 0 + } + switch strings.ToLower(strings.TrimSpace(stat)) { + case "max": + _, max := calcStats(values) + return max + case "p90": + return percentile(values, 90) + case "p95": + return percentile(values, 95) + case "p99": + return percentile(values, 99) + case "es90": + return expectedShortfall(values, 90) + case "es95": + return expectedShortfall(values, 95) + case "average", "avg", "mean", "": + avg, _ := calcStats(values) + return avg + default: + avg, _ := calcStats(values) + return avg + } +} + +// allStatsString formats avg/max/p95/p99/es95 for one series, for side-by-side +// comparison while deciding which statistic to gate on. Remove once chosen. +func allStatsString(values []float64) string { + avg, max := calcStats(values) + return fmt.Sprintf("avg=%.4f max=%.4f p95=%.4f p99=%.4f es95=%.4f", + avg, max, percentile(values, 95), percentile(values, 99), expectedShortfall(values, 95)) +} + +// expectedShortfall returns the mean of the samples at or above the p-th +// percentile (CVaR / tail-conditional mean) — a smoothed view of the worst tail +// that is steadier than p95 but still moves when the tail genuinely shifts. +func expectedShortfall(values []float64, p float64) float64 { + if len(values) == 0 { + return 0 + } + sorted := append([]float64(nil), values...) + sort.Float64s(sorted) + cutoff := percentile(values, p) + var sum float64 + var count int + for _, v := range sorted { + if v >= cutoff { + sum += v + count++ + } + } + if count == 0 { + return cutoff + } + return sum / float64(count) +} + +// percentile returns the linearly-interpolated p-th percentile (0-100) of values. +func percentile(values []float64, p float64) float64 { + if len(values) == 0 { + return 0 + } + sorted := append([]float64(nil), values...) + sort.Float64s(sorted) + if len(sorted) == 1 { + return sorted[0] + } + rank := (p / 100) * float64(len(sorted)-1) + lo := int(math.Floor(rank)) + hi := int(math.Ceil(rank)) + if lo == hi { + return sorted[lo] + } + frac := rank - float64(lo) + return sorted[lo]*(1-frac) + sorted[hi]*frac +} + // TestMain resolves the region, cluster name, and account ID into a config and // sets up the shared metrics client before running the suite. func TestMain(m *testing.M) { From cf60005f5fb7c62a70e00d5d91658464f3ed287b Mon Sep 17 00:00:00 2001 From: Fareedah Hafis Date: Fri, 11 Sep 2026 11:19:06 +0000 Subject: [PATCH 2/2] Skip all-zero series and fail on a zero stat in the otel performance tests --- test/otel/performance/performance_test.go | 12 ++++++++++++ test/otel/performance/regression_test.go | 6 ++++++ test/otel/performance/setup_test.go | 12 ++++++++++++ 3 files changed, 30 insertions(+) diff --git a/test/otel/performance/performance_test.go b/test/otel/performance/performance_test.go index 02f4fd79..8201a646 100644 --- a/test/otel/performance/performance_test.go +++ b/test/otel/performance/performance_test.go @@ -132,6 +132,11 @@ func checkAgainstUpperBound(values []float64, threshold float64, errorBound floa } } value := summaryStat(values, stat) + // A resulting statistic of 0 means no data was collected for this pod — + // that is a broken run, not a low-but-healthy reading, so fail. + if value == 0 { + return 0, fmt.Errorf("%s value is 0 — no data collected for this pod", stat) + } upperBound := threshold * (1 + errorBound) if threshold > 0 && value > upperBound { return value, fmt.Errorf("%s value %f exceeds upper bound %f", stat, value, upperBound) @@ -219,6 +224,13 @@ func TestPerformanceThresholds(t *testing.T) { if podType == "" { continue } + + // An all-zero series means no data was collected for this pod in the + // window; skip it so it doesn't count as a (failing) observation. A + // whole pod class going missing is still caught by the presence check. + if isAllZero(series.Values) { + continue + } expectedClasses[podType] = true var denominator float64 diff --git a/test/otel/performance/regression_test.go b/test/otel/performance/regression_test.go index beb12eb0..db6b706c 100644 --- a/test/otel/performance/regression_test.go +++ b/test/otel/performance/regression_test.go @@ -166,6 +166,9 @@ func collectCurrentResults(t *testing.T) PerfResult { for _, v := range series.Values { require.False(t, math.IsNaN(v), "CPU series for %s contains a NaN sample", podName) } + if isAllZero(series.Values) { + continue + } stat := summaryStat(series.Values, regressionStat) if isDaemonSetPod(podName) { sawDaemonSetCPU = true @@ -188,6 +191,9 @@ func collectCurrentResults(t *testing.T) PerfResult { for _, v := range series.Values { require.False(t, math.IsNaN(v), "memory series for %s contains a NaN sample", podName) } + if isAllZero(series.Values) { + continue + } statMB := summaryStat(series.Values, regressionStat) / (1024 * 1024) if isDaemonSetPod(podName) { sawDaemonSetMem = true diff --git a/test/otel/performance/setup_test.go b/test/otel/performance/setup_test.go index e911ee1d..53cf130e 100644 --- a/test/otel/performance/setup_test.go +++ b/test/otel/performance/setup_test.go @@ -87,6 +87,18 @@ func fetchSharedMetrics(t *testing.T) *podMetricData { return sharedMetrics } +// isAllZero reports whether every value in the series is zero. An all-zero +// series means no data was collected for that pod in the window; scoring it +// would drag the stats artificially low, so callers skip these series. +func isAllZero(values []float64) bool { + for _, v := range values { + if v != 0 { + return false + } + } + return true +} + // calcStats computes the average and maximum from a series of data points. // performance_test.go uses the average and regression_test.go uses the max. func calcStats(values []float64) (float64, float64) {