Skip to content
Draft
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
15 changes: 15 additions & 0 deletions README.md
Original file line number Diff line number Diff line change
Expand Up @@ -26,6 +26,7 @@ featuring advanced flamegraph analysis tools.
- [Usage](#usage)
- [Managing the stack](#managing-the-stack)
- [Local running and development](#local-running-and-development)
- [End-to-end test harness](#end-to-end-test-harness)


## System Overview
Expand Down Expand Up @@ -55,6 +56,7 @@ same backend services and storage layer:
```mermaid
%%{init: {
'theme': 'base',
'flowchart': { 'defaultRenderer': 'elk' },
'themeVariables': {
'primaryColor': '#ffffff',
'primaryBorderColor': '#cbd5e1',
Expand Down Expand Up @@ -388,3 +390,16 @@ To develop the project, it may be useful to run each component locally, see rele
- [webapp (backend and frontend)](src/gprofiler/README.md)
- [gprofiler_flamedb_rest](src/gprofiler_flamedb_rest/README.md)
- [gprofiler_indexer](src/gprofiler_indexer/README.md)

## End-to-end test harness
To run the full stack **plus a real gProfiler agent** locally (LocalStack S3/SQS
included) and exercise the workload-level profiling flow end to end — with API
acceptance tests (AT-S1..S15) and Playwright UI tests — see
[deploy/E2E_HARNESS.md](deploy/E2E_HARNESS.md).

```bash
cd deploy
make -f Makefile.e2e e2e-up-src # studio + sample workload + source-built agent
make -f Makefile.e2e e2e-test # API acceptance suite
make -f Makefile.e2e e2e-ui-test # Playwright UI suite
```
342 changes: 342 additions & 0 deletions deploy/E2E_HARNESS.md

Large diffs are not rendered by default.

114 changes: 114 additions & 0 deletions deploy/Makefile.e2e
Original file line number Diff line number Diff line change
@@ -0,0 +1,114 @@
# E2E harness for Performance Studio + a real gProfiler agent.
#
# Usage:
# make -f Makefile.e2e e2e-up # bring up studio + sample app + agent
# make -f Makefile.e2e e2e-status # show the agent in workload_status
# make -f Makefile.e2e e2e-start # submit a host-scope start (SERVICE/HOST overridable)
# make -f Makefile.e2e e2e-stop # submit a host-scope stop
# make -f Makefile.e2e e2e-agent-logs
# make -f Makefile.e2e e2e-test # run in-network pytest acceptance suite
# make -f Makefile.e2e e2e-down # remove only the agent + sample app (studio stays up)
# make -f Makefile.e2e e2e-down-all # tear the whole stack down (destructive)

SHELL := /bin/bash

COMPOSE := docker compose -f docker-compose.yml -f docker-compose.e2e.yml --profile with-clickhouse
NGINX := https://localhost:4433
AUTH := admin:admin
SERVICE ?= e2e-sample-app
# Compose default network. Derived from the project name (the deploy/ dir), so
# override NETWORK=<name>_default if you set COMPOSE_PROJECT_NAME.
NETWORK ?= $(notdir $(CURDIR))_default

# Source-build settings (sibling agent repo).
AGENT_REPO ?= ../../gprofiler
AGENT_ARCH ?= x86_64
AGENT_SRC_IMAGE ?= gprofiler-e2e-agent:src

# Placeholder token so compose can parse before a real one is minted; the agent
# is re-created with the real token in e2e-up.
export GPROFILER_TOKEN ?= bootstrap

.PHONY: e2e-up e2e-up-src e2e-agent-build e2e-token e2e-status e2e-start e2e-stop e2e-agent-logs e2e-test e2e-down e2e-down-all e2e-flamegraph

## Build the agent image FROM SOURCE using the fast (no-staticx) build (~1-2 min
## warm cache). The --fast exe is glibc-dynamic, so it's wrapped in a glibc base.
e2e-agent-build:
@echo ">> building agent executable from source (--fast, no staticx)"
cd $(AGENT_REPO) && scripts/build_$(AGENT_ARCH)_executable.sh --fast
@echo ">> wrapping exe in glibc base -> $(AGENT_SRC_IMAGE)"
docker build -f e2e/agent-glibc.Dockerfile --build-arg ARCH=$(AGENT_ARCH) -t $(AGENT_SRC_IMAGE) $(AGENT_REPO)

## Build the agent from source, then bring the whole harness up with it.
e2e-up-src: e2e-agent-build
GPROFILER_IMAGE=$(AGENT_SRC_IMAGE) $(MAKE) -f Makefile.e2e e2e-up

## Mint (or fetch) a valid profiler token from the running studio.
e2e-token:
@curl -sk $(NGINX)/api/api_key -u $(AUTH) \
| python3 -c "import sys,json;print(json.load(sys.stdin)['apiKey'])"

## Bring the whole stack up, then re-create the agent with a real token.
e2e-up:
@echo ">> starting studio + sample app + agent (bootstrap token)"
$(COMPOSE) up -d --build
@echo ">> waiting for studio API..."
@for i in $$(seq 1 30); do \
if curl -skf $(NGINX)/api/api_key -u $(AUTH) >/dev/null 2>&1; then break; fi; \
sleep 2; \
done
@tok=$$(curl -sk $(NGINX)/api/api_key -u $(AUTH) | python3 -c "import sys,json;print(json.load(sys.stdin)['apiKey'])"); \
echo ">> minting token: $$tok"; \
GPROFILER_TOKEN=$$tok $(COMPOSE) up -d --force-recreate gprofiler-agent
@echo ">> up. Try: make -f Makefile.e2e e2e-status"

## Show the agent's live inventory row.
e2e-status:
@curl -sk "$(NGINX)/api/metrics/profiling/workload_status" -u $(AUTH) \
| python3 -m json.tool

## Submit a host-scope START for $(SERVICE) (auto-discovers the reporting host).
e2e-start:
@host=$$(curl -sk "$(NGINX)/api/metrics/profiling/host_status?service_name=$(SERVICE)" -u $(AUTH) \
| python3 -c "import sys,json;print(json.load(sys.stdin)['hosts'][0]['hostname'])"); \
echo ">> start on host=$$host"; \
curl -sk -X POST "$(NGINX)/api/metrics/profile_request" -u $(AUTH) -H "Content-Type: application/json" \
-d "{\"service_name\":\"$(SERVICE)\",\"request_type\":\"start\",\"continuous\":false,\"duration\":30,\"frequency\":11,\"profiling_mode\":\"cpu\",\"target_scope\":\"host\",\"target_hosts\":{\"$$host\":[]},\"target_entities\":[{\"service_name\":\"$(SERVICE)\",\"hostname\":\"$$host\"}],\"additional_args\":{}}" \
| python3 -m json.tool

## Submit a host-scope STOP for $(SERVICE) (regression check: no PIDs at host level).
e2e-stop:
@host=$$(curl -sk "$(NGINX)/api/metrics/profiling/host_status?service_name=$(SERVICE)" -u $(AUTH) \
| python3 -c "import sys,json;print(json.load(sys.stdin)['hosts'][0]['hostname'])"); \
echo ">> stop on host=$$host"; \
curl -sk -X POST "$(NGINX)/api/metrics/profile_request" -u $(AUTH) -H "Content-Type: application/json" \
-d "{\"service_name\":\"$(SERVICE)\",\"request_type\":\"stop\",\"continuous\":false,\"duration\":30,\"frequency\":11,\"profiling_mode\":\"cpu\",\"target_scope\":\"host\",\"stop_level\":\"host\",\"target_hosts\":{\"$$host\":[]},\"target_entities\":[{\"service_name\":\"$(SERVICE)\",\"hostname\":\"$$host\"}],\"additional_args\":{}}" \
| python3 -m json.tool

## List the flamegraph artifacts produced in (local)S3.
e2e-flamegraph:
@docker run --rm --network $(NETWORK) \
-e AWS_ACCESS_KEY_ID=test -e AWS_SECRET_ACCESS_KEY=test -e AWS_DEFAULT_REGION=us-east-1 \
amazon/aws-cli:2.15.0 --endpoint-url http://localstack:4566 \
s3 ls s3://performance-studio-bucket/$${S3_PATH_PREFIX:+$${S3_PATH_PREFIX}/}products/$(SERVICE)/ --recursive

e2e-agent-logs:
@docker logs -f gprofiler-ps-e2e-agent

## Run the in-network acceptance suite (pytest) against the live stack.
e2e-test:
$(COMPOSE) --profile test build e2e-tests
$(COMPOSE) --profile test run --rm e2e-tests

## Run the Playwright UI acceptance suite (in-network, official browser image).
e2e-ui-test:
$(COMPOSE) --profile ui build e2e-ui-tests
$(COMPOSE) --profile ui run --rm e2e-ui-tests

## Remove only the harness-added services; studio keeps running.
e2e-down:
$(COMPOSE) rm -sf gprofiler-agent e2e-sample-app e2e-tests e2e-ui-tests

## Tear everything down (studio included). Destructive.
e2e-down-all:
$(COMPOSE) down
116 changes: 116 additions & 0 deletions deploy/docker-compose.e2e.yml
Original file line number Diff line number Diff line change
@@ -0,0 +1,116 @@
#
# E2E overlay for the Performance Studio + gProfiler agent test harness.
#
# Layer this on top of the base stack:
#
# docker compose -f docker-compose.yml -f docker-compose.e2e.yml \
# --profile with-clickhouse up -d --build
#
# It adds:
# * e2e-sample-app - a deterministic CPU-burning workload to profile
# * gprofiler-agent - a real gProfiler agent in dynamic/heartbeat mode, pointed
# at the INTERNAL webapp (bypassing nginx basic auth), that
# reports workload inventory and executes start/stop commands
# * e2e-tests - an in-network pytest runner (profile "test"; on demand)
#
# The agent profiles only the sample app's PID namespace, so it never touches the
# host's processes.
#
services:
# A deterministic target the agent can profile (py-spy sees a hot Python frame).
e2e-sample-app:
image: python:3.11-slim
container_name: gprofiler-ps-e2e-sample-app
restart: unless-stopped
command:
- python3
- -c
- |
import time
def hot_loop():
total = 0
for i in range(5_000_000):
total += (i * i) % 7
return total
while True:
hot_loop()
time.sleep(0.01)

# gProfiler agent in dynamic profiling (heartbeat) mode.
# Defaults to the upstream public image; override with GPROFILER_IMAGE, or build
# from source with `make -f Makefile.e2e e2e-up-src` (see E2E_HARNESS.md).
gprofiler-agent:
image: ${GPROFILER_IMAGE:-intel/gprofiler:latest}
container_name: gprofiler-ps-e2e-agent
restart: unless-stopped
entrypoint: ["/gprofiler"]
# Point at the internal webapp service (no nginx basic auth on the internal
# port); the backend does not validate the bearer token locally.
command:
# --server-host: profile upload + health check; --api-server: heartbeat/metrics.
# Both are served by the internal webapp (no nginx basic auth on that port).
- --server-host=http://webapp
- --api-server=http://webapp
# A valid profiler token is required for upload/health-check. Mint one from
# the running studio (GET /api/api_key) and export GPROFILER_TOKEN before up.
- "--token=${GPROFILER_TOKEN:?set GPROFILER_TOKEN first - use make e2e-up}"
- --service-name=e2e-sample-app
- --upload-results
- --enable-heartbeat-server
- --heartbeat-interval=10
- --perf-mode=none
- --output-dir=/tmp/gprofiler_output
- --dont-send-logs
# Agent shares the sample app's PID namespace (not host-init), so skip the
# init-namespace guard.
- --disable-pidns-check
privileged: true
# Share the sample app's PID namespace so the agent profiles only that
# workload (isolated from the host).
pid: "service:e2e-sample-app"
environment:
- GPROFILER_IN_CONTAINER=1
- POD_NAMESPACE=e2e
- POD_NAME=gprofiler-agent
depends_on:
- webapp
- e2e-sample-app

# In-network acceptance/UI test runner. Kept behind the "test" profile so it
# only runs on demand (e.g. `docker compose ... --profile test run --rm e2e-tests`).
e2e-tests:
build:
context: ../src
dockerfile: tests/e2e/Dockerfile
container_name: gprofiler-ps-e2e-tests
profiles: ["test"]
environment:
# Reach services by their compose network names.
- E2E_BASE_URL=http://webapp
- E2E_NGINX_URL=https://nginx-load-balancer
- E2E_BASIC_AUTH_USER=admin
- E2E_BASIC_AUTH_PASSWORD=admin
- GPROFILER_POSTGRES_HOST=$POSTGRES_HOST
- GPROFILER_POSTGRES_PORT=$POSTGRES_PORT
- GPROFILER_POSTGRES_USERNAME=$POSTGRES_USER
- GPROFILER_POSTGRES_PASSWORD=$POSTGRES_PASSWORD
- GPROFILER_POSTGRES_DB_NAME=$POSTGRES_DB
depends_on:
- webapp
- db_postgres

# Playwright UI acceptance runner (profile "ui"; on demand). Reaches the
# console via the internal nginx service with basic auth over self-signed TLS.
e2e-ui-tests:
build:
context: ../src/tests/playwright
dockerfile: Dockerfile
container_name: gprofiler-ps-e2e-ui-tests
profiles: ["ui"]
environment:
- E2E_UI_URL=https://nginx-load-balancer
- E2E_BASIC_AUTH_USER=admin
- E2E_BASIC_AUTH_PASSWORD=admin
- E2E_UI_SERVICE=e2e-sample-app
depends_on:
- nginx-load-balancer
19 changes: 19 additions & 0 deletions deploy/e2e/agent-glibc.Dockerfile
Original file line number Diff line number Diff line change
@@ -0,0 +1,19 @@
# Wraps a locally-built gProfiler executable in a glibc base for the e2e harness.
#
# Build context MUST be the gprofiler agent repo root (so build/<arch>/gprofiler
# is present). Produce that exe first with the fast (no-staticx) build:
#
# cd ../../gprofiler && scripts/build_x86_64_executable.sh --fast
#
# The --fast build skips staticx, so the exe is glibc-dynamic and needs a glibc
# base (ubuntu) rather than the alpine base in the repo's container.Dockerfile.
ARG ARCH=x86_64
FROM ubuntu:22.04

ARG ARCH
ENV GPROFILER_IN_CONTAINER=1

COPY build/${ARCH}/gprofiler /gprofiler
RUN chmod +x /gprofiler

ENTRYPOINT ["/gprofiler"]
Loading
Loading