diff --git a/.github/workflows/01-ci.yml b/.github/workflows/01-ci.yml index 8206db43..451df1a2 100644 --- a/.github/workflows/01-ci.yml +++ b/.github/workflows/01-ci.yml @@ -1,17 +1,24 @@ name: 01 - CI on: - # Trigger the workflow on push to main branch + # ========================================================= + # Pull Request Validation + # ========================================================= + pull_request: + branches: + - main + + # ========================================================= + # Main Branch CI + # ========================================================= push: branches: - main - - # Manual trigger for the workflow - workflow_dispatch: + # Allow controlled manual execution from GitHub Actions. + workflow_dispatch: jobs: - # ========================================================= # Backend Tests # ========================================================= @@ -81,7 +88,6 @@ jobs: AZURE_STORAGE_CONTAINER_NAME: "" steps: - - name: Checkout repository uses: actions/checkout@v4 @@ -101,20 +107,164 @@ jobs: run: | pytest -v + # ========================================================= + # Frontend Tests + # ========================================================= + frontend-test: + name: Test Frontend + runs-on: ubuntu-latest + + env: + VITE_USER_SERVICE_URL: http://localhost:8001 + VITE_STUDENT_SERVICE_URL: http://localhost:8002 + VITE_LECTURER_SERVICE_URL: http://localhost:8003 + VITE_COURSE_SERVICE_URL: http://localhost:8004 + VITE_ENROLLMENT_SERVICE_URL: http://localhost:8005 + + steps: + - name: Checkout repository + uses: actions/checkout@v4 + + - name: Set up Node.js + uses: actions/setup-node@v4 + with: + node-version: "20" + cache: "npm" + cache-dependency-path: frontend/package-lock.json + + - name: Install dependencies + working-directory: frontend + run: | + npm ci + + - name: Run frontend tests + working-directory: frontend + run: | + npm run test -- --run + + # ========================================================= + # Terraform Validation, Plan and Provisioning + # Task 10.2D - Infrastructure as Code integration + # + # Pull Request: + # Init -> Format -> Validate -> Plan + # + # Main/Manual: + # Init -> Format -> Validate -> Plan -> Apply + # ========================================================= + terraform-check: + name: Terraform Validate, Plan and Apply + runs-on: ubuntu-latest + + # terraform.tfvars is intentionally not committed. + # Week10 values are supplied through TF_VAR_* variables. + env: + TF_VAR_resource_group_name: "sit722-week10-rg" + TF_VAR_location: "Australia East" + TF_VAR_acr_name: "sit722week10acr224848845" + TF_VAR_aks_name: "sit722-week10-aks" + TF_VAR_storage_account_name: "sit722w10storage224848" + TF_VAR_aks_node_count: "1" + TF_VAR_aks_vm_size: "Standard_D2s_v3" + + steps: + - name: Checkout repository + uses: actions/checkout@v4 + + - name: Set up Terraform + uses: hashicorp/setup-terraform@v3 + + - name: Login to Azure + uses: azure/login@v3 + with: + creds: ${{ secrets.AZURE_CREDENTIALS }} + + # ------------------------------------------------------- + # Persistent Terraform state + # ------------------------------------------------------- + - name: Terraform Init + working-directory: terraform + run: | + terraform init \ + -input=false \ + -backend-config="resource_group_name=sit722-week10-tfstate-rg" \ + -backend-config="storage_account_name=sit722w10tfstate224848" \ + -backend-config="container_name=tfstate" \ + -backend-config="key=week10-10.2d.tfstate" + + - name: Terraform Format Check + working-directory: terraform + run: | + terraform fmt -check + + - name: Terraform Validate + working-directory: terraform + run: | + terraform validate + + # ------------------------------------------------------- + # Generate a saved Terraform plan. + # ------------------------------------------------------- + - name: Terraform Plan + working-directory: terraform + run: | + terraform plan \ + -input=false \ + -no-color \ + -out=tfplan + + # ------------------------------------------------------- + # Infrastructure provisioning + # + # Pull requests NEVER provision infrastructure. + # Apply occurs only for main push or controlled + # workflow_dispatch execution. + # ------------------------------------------------------- + - name: Terraform Apply + if: > + github.event_name == 'push' || + github.event_name == 'workflow_dispatch' + working-directory: terraform + run: | + terraform apply \ + -input=false \ + -auto-approve \ + tfplan + + - name: Display Terraform Outputs + if: > + github.event_name == 'push' || + github.event_name == 'workflow_dispatch' + working-directory: terraform + run: | + echo "===== TERRAFORM OUTPUTS =====" + terraform output + + echo + echo "===== TERRAFORM STATE =====" + terraform state list # ========================================================= - # Build and Push Docker Images + # Build, Security Scan and Push Docker Images + # Task 10.2D - Docker Scout integration # ========================================================= - build-and-push: - name: Build and Push ${{ matrix.image }} + build-scan-and-push: + name: Build, Scan and Push ${{ matrix.image }} runs-on: ubuntu-latest - # All backend tests must pass before this job starts + # Application tests AND Terraform infrastructure + # provisioning/validation must succeed before images + # are published. needs: - backend-test + - frontend-test + - terraform-check - # Build and push when code is pushed to main or manually triggered - if: github.event_name == 'push' || github.event_name == 'workflow_dispatch' + # Pull requests perform validation only. + # Images are published for main or controlled manual runs. + if: > + github.event_name == 'push' || + github.event_name == 'workflow_dispatch' strategy: fail-fast: false @@ -140,19 +290,28 @@ jobs: image: koalatech-enrollment-service steps: - - name: Checkout repository uses: actions/checkout@v4 + # ------------------------------------------------------- + # Azure authentication + # ------------------------------------------------------- - name: Login to Azure - uses: azure/login@v2 + uses: azure/login@v3 with: creds: ${{ secrets.AZURE_CREDENTIALS }} + # ------------------------------------------------------- + # Authenticate with Azure Container Registry + # ------------------------------------------------------- - name: Login to Azure Container Registry run: | - az acr login --name ${{ vars.ACR_NAME }} + az acr login \ + --name ${{ vars.ACR_NAME }} + # ------------------------------------------------------- + # Build image locally before security analysis + # ------------------------------------------------------- - name: Build Docker image run: | docker build \ @@ -160,7 +319,56 @@ jobs: -t ${{ vars.ACR_LOGIN_SERVER }}/${{ matrix.image }}:${{ github.sha }} \ ./${{ matrix.service }} + # ------------------------------------------------------- + # Docker Hub authentication for Docker Scout + # + # Required repository secrets: + # DOCKERHUB_USERNAME + # DOCKERHUB_TOKEN + # ------------------------------------------------------- + - name: Login to Docker Hub for Docker Scout + uses: docker/login-action@v4 + with: + username: ${{ secrets.DOCKERHUB_USERNAME }} + password: ${{ secrets.DOCKERHUB_TOKEN }} + + # ------------------------------------------------------- + # Docker Scout vulnerability analysis BEFORE ACR push + # + # exit-code is false so identified vulnerabilities are + # reported without preventing the assessment pipeline + # from continuing. Remediation is demonstrated separately. + # ------------------------------------------------------- + - name: Docker Scout CVE scan + uses: docker/scout-action@v1 + with: + command: cves + image: local://${{ vars.ACR_LOGIN_SERVER }}/${{ matrix.image }}:${{ github.sha }} + only-severities: critical,high + write-comment: false + exit-code: false + github-token: ${{ secrets.GITHUB_TOKEN }} + + # ------------------------------------------------------- + # Obtain Docker Scout remediation guidance + # ------------------------------------------------------- + - name: Docker Scout remediation recommendations + uses: docker/scout-action@v1 + with: + command: recommendations + image: local://${{ vars.ACR_LOGIN_SERVER }}/${{ matrix.image }}:${{ github.sha }} + write-comment: false + github-token: ${{ secrets.GITHUB_TOKEN }} + + # ------------------------------------------------------- + # Publish security-scanned image to ACR + # ------------------------------------------------------- - name: Push Docker image with commit SHA run: | docker push \ ${{ vars.ACR_LOGIN_SERVER }}/${{ matrix.image }}:${{ github.sha }} + + - name: Display published image + run: | + echo "Published image:" + echo "${{ vars.ACR_LOGIN_SERVER }}/${{ matrix.image }}:${{ github.sha }}" \ No newline at end of file diff --git a/.github/workflows/02-deploy-staging.yml b/.github/workflows/02-deploy-staging.yml index 006bb70d..d0d2dc0c 100644 --- a/.github/workflows/02-deploy-staging.yml +++ b/.github/workflows/02-deploy-staging.yml @@ -11,7 +11,7 @@ on: jobs: deploy-staging: - name: Deploy to Staging + name: Deploy Application and Monitoring to Staging runs-on: ubuntu-latest if: > @@ -67,85 +67,271 @@ jobs: --dry-run=client \ -o yaml | kubectl apply -f - - - name: Apply Kubernetes manifests + - name: Apply staging Kubernetes manifests run: | - kubectl apply \ - -f kubernetes/staging/ + kubectl apply -f kubernetes/staging/ - - name: Update frontend image + - name: Deploy tested application images + shell: bash run: | + IMAGE_TAG="${{ github.event.workflow_run.head_sha }}" + kubectl set image deployment/frontend \ - frontend=${{ vars.ACR_LOGIN_SERVER }}/koalatech-frontend:${{ github.event.workflow_run.head_sha }} \ + frontend=${{ vars.ACR_LOGIN_SERVER }}/koalatech-frontend:$IMAGE_TAG \ -n staging - - name: Update user-service image - run: | kubectl set image deployment/user-service \ - user-service=${{ vars.ACR_LOGIN_SERVER }}/koalatech-user-service:${{ github.event.workflow_run.head_sha }} \ + user-service=${{ vars.ACR_LOGIN_SERVER }}/koalatech-user-service:$IMAGE_TAG \ -n staging - - name: Update student-service image - run: | kubectl set image deployment/student-service \ - student-service=${{ vars.ACR_LOGIN_SERVER }}/koalatech-student-service:${{ github.event.workflow_run.head_sha }} \ + student-service=${{ vars.ACR_LOGIN_SERVER }}/koalatech-student-service:$IMAGE_TAG \ -n staging - - name: Update lecturer-service image - run: | kubectl set image deployment/lecturer-service \ - lecturer-service=${{ vars.ACR_LOGIN_SERVER }}/koalatech-lecturer-service:${{ github.event.workflow_run.head_sha }} \ + lecturer-service=${{ vars.ACR_LOGIN_SERVER }}/koalatech-lecturer-service:$IMAGE_TAG \ -n staging - - name: Update course-service image - run: | kubectl set image deployment/course-service \ - course-service=${{ vars.ACR_LOGIN_SERVER }}/koalatech-course-service:${{ github.event.workflow_run.head_sha }} \ + course-service=${{ vars.ACR_LOGIN_SERVER }}/koalatech-course-service:$IMAGE_TAG \ -n staging - - name: Update enrollment-service image - run: | kubectl set image deployment/enrollment-service \ - enrollment-service=${{ vars.ACR_LOGIN_SERVER }}/koalatech-enrollment-service:${{ github.event.workflow_run.head_sha }} \ + enrollment-service=${{ vars.ACR_LOGIN_SERVER }}/koalatech-enrollment-service:$IMAGE_TAG \ -n staging - - name: Wait for frontend rollout + - name: Verify application rollouts + shell: bash run: | - kubectl rollout status deployment/frontend \ - -n staging \ - --timeout=300s + for deployment in \ + frontend \ + user-service \ + student-service \ + lecturer-service \ + course-service \ + enrollment-service + do + echo "Checking rollout: $deployment" - - name: Wait for user-service rollout + kubectl rollout status deployment/$deployment \ + -n staging \ + --timeout=300s + done + + - name: Create monitoring namespace run: | - kubectl rollout status deployment/user-service \ - -n staging \ - --timeout=300s + kubectl apply \ + -f kubernetes/monitoring/01-namespace.yaml - - name: Wait for student-service rollout + - name: Create Grafana admin secret run: | - kubectl rollout status deployment/student-service \ - -n staging \ - --timeout=300s + kubectl create secret generic grafana-admin \ + --namespace monitoring \ + --from-literal=password="${{ secrets.GRAFANA_ADMIN_PASSWORD }}" \ + --dry-run=client \ + -o yaml | kubectl apply -f - - - name: Wait for lecturer-service rollout + - name: Deploy Prometheus and Grafana monitoring run: | - kubectl rollout status deployment/lecturer-service \ - -n staging \ - --timeout=300s + kubectl apply \ + -f kubernetes/monitoring/ - - name: Wait for course-service rollout + - name: Verify monitoring rollouts + shell: bash run: | - kubectl rollout status deployment/course-service \ - -n staging \ - --timeout=300s + for deployment in \ + kube-state-metrics \ + prometheus \ + grafana + do + echo "Checking monitoring deployment: $deployment" + + kubectl rollout status deployment/$deployment \ + -n monitoring \ + --timeout=300s + done - - name: Wait for enrollment-service rollout + - name: Verify Prometheus targets + shell: bash run: | - kubectl rollout status deployment/enrollment-service \ - -n staging \ - --timeout=300s + kubectl port-forward \ + -n monitoring \ + svc/prometheus \ + 9090:9090 > /tmp/prometheus-port-forward.log 2>&1 & + + PF_PID=$! + + cleanup() { + kill "$PF_PID" 2>/dev/null || true + } + + trap cleanup EXIT + + echo "Waiting for Prometheus API..." + + for attempt in {1..20} + do + if curl --fail --silent \ + http://127.0.0.1:9090/-/ready > /dev/null + then + echo "Prometheus API is ready." + break + fi + + if [ "$attempt" -eq 20 ]; then + echo "Prometheus did not become ready." + cat /tmp/prometheus-port-forward.log + exit 1 + fi + + echo "Prometheus API not ready yet - attempt ${attempt}/20" + sleep 3 + done + + echo + echo "Waiting for all required Prometheus targets to become healthy..." + + for attempt in {1..12} + do + echo + echo "Target health check ${attempt}/12" + + curl --fail --silent \ + http://127.0.0.1:9090/api/v1/targets \ + > /tmp/prometheus-targets.json + + if python3 - <<'PYTHON' + import json + + with open("/tmp/prometheus-targets.json") as file: + data = json.load(file) + + targets = data["data"]["activeTargets"] + + required_jobs = { + "user-service", + "student-service", + "lecturer-service", + "course-service", + "enrollment-service", + "kube-state-metrics", + } + + target_status = {} + + for target in targets: + labels = target.get("labels", {}) + job = labels.get("job", "unknown") + health = target.get("health", "unknown") + scrape_url = target.get("scrapeUrl", "") + last_error = target.get("lastError", "") + + if job in required_jobs: + target_status[job] = health + + print( + f"{job:25} " + f"health={health:8} " + f"{scrape_url}" + ) + + if last_error: + print(f" error: {last_error}") + + healthy_jobs = { + job + for job, health in target_status.items() + if health == "up" + } + + missing = required_jobs - healthy_jobs + + if missing: + print() + print( + "Targets not healthy yet: " + + ", ".join(sorted(missing)) + ) + raise SystemExit(1) + + print() + print("All required Prometheus targets are healthy.") + PYTHON + then + echo + echo "Prometheus monitoring validation successful." + exit 0 + fi + + if [ "$attempt" -eq 12 ]; then + echo + echo "Required Prometheus targets did not become healthy within 120 seconds." + exit 1 + fi + + echo "Waiting 10 seconds before retry..." + sleep 10 + done + + - name: Verify Grafana health + shell: bash + run: | + kubectl port-forward \ + -n monitoring \ + svc/grafana \ + 3000:3000 > /tmp/grafana-port-forward.log 2>&1 & + + PF_PID=$! + + cleanup() { + kill "$PF_PID" 2>/dev/null || true + } + + trap cleanup EXIT + + echo "Waiting for Grafana..." + + for attempt in {1..20} + do + if curl --fail --silent \ + http://127.0.0.1:3000/api/health + then + echo + echo "Grafana is healthy." + exit 0 + fi + + sleep 3 + done + + echo "Grafana health validation failed." + cat /tmp/grafana-port-forward.log + exit 1 - name: Show staging resources + if: always() run: | - kubectl get pods -n staging + echo "===== STAGING PODS =====" + kubectl get pods -n staging -o wide + + echo + echo "===== STAGING SERVICES =====" kubectl get services -n staging - kubectl get pvc -n staging \ No newline at end of file + + echo + echo "===== STAGING PVCs =====" + kubectl get pvc -n staging + + - name: Show monitoring resources + if: always() + run: | + echo "===== MONITORING DEPLOYMENTS =====" + kubectl get deployments -n monitoring + + echo + echo "===== MONITORING PODS =====" + kubectl get pods -n monitoring -o wide + + echo + echo "===== MONITORING SERVICES =====" + kubectl get services -n monitoring diff --git a/.github/workflows/04-deploy-production.yml b/.github/workflows/04-deploy-production.yml index 969d646b..5bda7b89 100644 --- a/.github/workflows/04-deploy-production.yml +++ b/.github/workflows/04-deploy-production.yml @@ -2,15 +2,14 @@ name: 04 - Deploy to Production on: workflow_dispatch: - inputs: - image_tag: - description: "Tested image SHA to deploy" - required: true - type: string jobs: deploy-production: name: Deploy to Production + + # Production deployment is manual during the Week 10 task. + if: ${{ github.event_name == 'workflow_dispatch' }} + runs-on: ubuntu-latest environment: @@ -32,6 +31,60 @@ jobs: --name ${{ vars.AKS_CLUSTER_NAME }} \ --overwrite-existing + # Read the exact frontend image that has already passed + # deployment and smoke testing in staging. + - name: Get tested image SHA from staging + id: tested-image + shell: bash + run: | + TESTED_IMAGE=$(kubectl get deployment frontend \ + -n staging \ + -o jsonpath='{.spec.template.spec.containers[0].image}') + + IMAGE_TAG="${TESTED_IMAGE##*:}" + + echo "Tested staging image: $TESTED_IMAGE" + echo "Promoting image tag: $IMAGE_TAG" + + if [ -z "$IMAGE_TAG" ]; then + echo "Unable to determine tested staging image tag." + exit 1 + fi + + echo "image_tag=$IMAGE_TAG" >> "$GITHUB_OUTPUT" + + # Verify that all six application deployments in staging + # are running the exact same tested release. + - name: Verify staging release consistency + shell: bash + run: | + IMAGE_TAG="${{ steps.tested-image.outputs.image_tag }}" + + for deployment in \ + frontend \ + user-service \ + student-service \ + lecturer-service \ + course-service \ + enrollment-service + do + IMAGE=$(kubectl get deployment "$deployment" \ + -n staging \ + -o jsonpath='{.spec.template.spec.containers[0].image}') + + echo "$deployment -> $IMAGE" + + case "$IMAGE" in + *:"$IMAGE_TAG") + echo "$deployment uses tested release $IMAGE_TAG" + ;; + *) + echo "$deployment does not use tested release $IMAGE_TAG" + exit 1 + ;; + esac + done + - name: Create production namespace run: | kubectl create namespace production \ @@ -63,80 +116,59 @@ jobs: - name: Apply Kubernetes manifests run: | - kubectl apply \ - -f kubernetes/production/ + kubectl apply -f kubernetes/production/ - - name: Update frontend image + - name: Promote tested images to production + shell: bash run: | + IMAGE_TAG="${{ steps.tested-image.outputs.image_tag }}" + + echo "Promoting tested release $IMAGE_TAG to production" + kubectl set image deployment/frontend \ - frontend=${{ vars.ACR_LOGIN_SERVER }}/koalatech-frontend:${{ inputs.image_tag }} \ + frontend=${{ vars.ACR_LOGIN_SERVER }}/koalatech-frontend:$IMAGE_TAG \ -n production - - name: Update user-service image - run: | kubectl set image deployment/user-service \ - user-service=${{ vars.ACR_LOGIN_SERVER }}/koalatech-user-service:${{ inputs.image_tag }} \ + user-service=${{ vars.ACR_LOGIN_SERVER }}/koalatech-user-service:$IMAGE_TAG \ -n production - - name: Update student-service image - run: | kubectl set image deployment/student-service \ - student-service=${{ vars.ACR_LOGIN_SERVER }}/koalatech-student-service:${{ inputs.image_tag }} \ + student-service=${{ vars.ACR_LOGIN_SERVER }}/koalatech-student-service:$IMAGE_TAG \ -n production - - name: Update lecturer-service image - run: | kubectl set image deployment/lecturer-service \ - lecturer-service=${{ vars.ACR_LOGIN_SERVER }}/koalatech-lecturer-service:${{ inputs.image_tag }} \ + lecturer-service=${{ vars.ACR_LOGIN_SERVER }}/koalatech-lecturer-service:$IMAGE_TAG \ -n production - - name: Update course-service image - run: | kubectl set image deployment/course-service \ - course-service=${{ vars.ACR_LOGIN_SERVER }}/koalatech-course-service:${{ inputs.image_tag }} \ + course-service=${{ vars.ACR_LOGIN_SERVER }}/koalatech-course-service:$IMAGE_TAG \ -n production - - name: Update enrollment-service image - run: | kubectl set image deployment/enrollment-service \ - enrollment-service=${{ vars.ACR_LOGIN_SERVER }}/koalatech-enrollment-service:${{ inputs.image_tag }} \ + enrollment-service=${{ vars.ACR_LOGIN_SERVER }}/koalatech-enrollment-service:$IMAGE_TAG \ -n production - - name: Wait for frontend rollout - run: | - kubectl rollout status deployment/frontend \ - -n production \ - --timeout=300s - - - name: Wait for user-service rollout - run: | - kubectl rollout status deployment/user-service \ - -n production \ - --timeout=300s - - - name: Wait for student-service rollout - run: | - kubectl rollout status deployment/student-service \ - -n production \ - --timeout=300s - - - name: Wait for lecturer-service rollout - run: | - kubectl rollout status deployment/lecturer-service \ - -n production \ - --timeout=300s - - - name: Wait for course-service rollout - run: | - kubectl rollout status deployment/course-service \ - -n production \ - --timeout=300s - - - name: Wait for enrollment-service rollout - run: | - kubectl rollout status deployment/enrollment-service \ - -n production \ - --timeout=300s + - name: Verify production rollouts + shell: bash + run: | + for deployment in \ + frontend \ + user-service \ + student-service \ + lecturer-service \ + course-service \ + enrollment-service + do + kubectl rollout status deployment/$deployment \ + -n production \ + --timeout=300s + done + + - name: Verify production image versions + run: | + kubectl get deployments -n production \ + -o custom-columns='DEPLOYMENT:.metadata.name,IMAGE:.spec.template.spec.containers[*].image' - name: Show production resources run: | diff --git a/.github/workflows/05-canary-deployment.yml b/.github/workflows/05-canary-deployment.yml new file mode 100644 index 00000000..a4034b28 --- /dev/null +++ b/.github/workflows/05-canary-deployment.yml @@ -0,0 +1,313 @@ +name: 05 - Canary Deployment and Automatic Rollback + +on: + workflow_dispatch: + inputs: + scenario: + description: "Canary demonstration scenario" + required: true + default: healthy + type: choice + options: + - healthy + - faulty + +permissions: + contents: read + +env: + STAGING_NAMESPACE: staging + CANARY_DEPLOYMENT: frontend-canary + CANARY_SERVICE: frontend-canary + STABLE_DEPLOYMENT: frontend + +jobs: + canary-deployment: + name: Canary Deployment, Validation and Recovery + runs-on: ubuntu-latest + + steps: + - name: Checkout repository + uses: actions/checkout@v4 + + - name: Azure login + uses: azure/login@v3 + with: + creds: ${{ secrets.AZURE_CREDENTIALS }} + + - name: Get AKS credentials + shell: bash + run: | + az aks get-credentials \ + --resource-group "${{ vars.AKS_RESOURCE_GROUP }}" \ + --name "${{ vars.AKS_CLUSTER_NAME }}" \ + --overwrite-existing + + kubectl get nodes + + - name: Capture current stable release + id: stable + shell: bash + run: | + STABLE_IMAGE=$(kubectl get deployment "${STABLE_DEPLOYMENT}" \ + -n "${STAGING_NAMESPACE}" \ + -o jsonpath='{.spec.template.spec.containers[0].image}') + + echo "Current stable image: ${STABLE_IMAGE}" + echo "image=${STABLE_IMAGE}" >> "$GITHUB_OUTPUT" + + - name: Verify stable application before canary + shell: bash + run: | + kubectl rollout status deployment/"${STABLE_DEPLOYMENT}" \ + -n "${STAGING_NAMESPACE}" \ + --timeout=180s + + STABLE_IP=$(kubectl get service frontend \ + -n "${STAGING_NAMESPACE}" \ + -o jsonpath='{.status.loadBalancer.ingress[0].ip}') + + if [ -z "${STABLE_IP}" ]; then + STABLE_IP=$(kubectl get service frontend \ + -n "${STAGING_NAMESPACE}" \ + -o jsonpath='{.status.loadBalancer.ingress[0].hostname}') + fi + + echo "Stable frontend address: ${STABLE_IP}" + + curl --fail \ + --silent \ + --show-error \ + --retry 10 \ + --retry-delay 5 \ + --retry-all-errors \ + "http://${STABLE_IP}/" \ + > /dev/null + + echo "Stable application is healthy before canary deployment." + + - name: Select canary image + id: candidate + shell: bash + run: | + if [ "${{ inputs.scenario }}" = "healthy" ]; then + CANARY_IMAGE="${{ steps.stable.outputs.image }}" + echo "Healthy scenario selected." + echo "Candidate image: ${CANARY_IMAGE}" + else + CANARY_IMAGE="nginx:1.27-alpine" + echo "Faulty scenario selected." + echo "The canary will be configured with an invalid health endpoint." + echo "Candidate image: ${CANARY_IMAGE}" + fi + + echo "image=${CANARY_IMAGE}" >> "$GITHUB_OUTPUT" + + - name: Deploy canary candidate + shell: bash + run: | + sed \ + "s|CANARY_IMAGE|${{ steps.candidate.outputs.image }}|g" \ + kubernetes/canary/frontend-canary-deployment.yaml \ + > /tmp/frontend-canary-deployment.yaml + + if [ "${{ inputs.scenario }}" = "faulty" ]; then + echo "Injecting deliberately invalid health endpoint for rollback demonstration." + + sed -i \ + 's|path: /$|path: /deliberately-faulty-health-endpoint|g' \ + /tmp/frontend-canary-deployment.yaml + fi + + kubectl apply \ + -f /tmp/frontend-canary-deployment.yaml + + kubectl apply \ + -f kubernetes/canary/frontend-canary-service.yaml + + echo + echo "Stable and canary deployments:" + kubectl get deployments \ + -n "${STAGING_NAMESPACE}" \ + -l app=frontend \ + -o wide + + - name: Wait for canary rollout + id: rollout + continue-on-error: true + shell: bash + run: | + kubectl rollout status deployment/"${CANARY_DEPLOYMENT}" \ + -n "${STAGING_NAMESPACE}" \ + --timeout=120s + + - name: Validate canary release + id: validation + if: steps.rollout.outcome == 'success' + continue-on-error: true + shell: bash + run: | + echo "Starting automated canary health validation..." + + kubectl run canary-health-check \ + --rm \ + -i \ + --restart=Never \ + --image=curlimages/curl:8.12.1 \ + -n "${STAGING_NAMESPACE}" \ + -- \ + curl \ + --fail \ + --silent \ + --show-error \ + --max-time 10 \ + "http://${CANARY_SERVICE}/" \ + > /dev/null + + echo "Canary HTTP health validation passed." + + echo "Canary candidate passed all validation checks." + + - name: Promote healthy canary + if: >- + inputs.scenario == 'healthy' && + steps.rollout.outcome == 'success' && + steps.validation.outcome == 'success' + shell: bash + run: | + echo "Promoting validated canary image to stable deployment..." + + kubectl set image deployment/"${STABLE_DEPLOYMENT}" \ + frontend="${{ steps.candidate.outputs.image }}" \ + -n "${STAGING_NAMESPACE}" + + kubectl rollout status deployment/"${STABLE_DEPLOYMENT}" \ + -n "${STAGING_NAMESPACE}" \ + --timeout=180s + + echo + echo "Canary promotion completed successfully." + + - name: Automatic rollback and recovery + if: >- + inputs.scenario == 'faulty' || + steps.rollout.outcome == 'failure' || + steps.validation.outcome == 'failure' + shell: bash + run: | + echo "Canary validation failed." + echo "Automatic recovery has been triggered." + echo + echo "Stable deployment was never replaced." + echo "Removing failed canary workload..." + + kubectl delete deployment "${CANARY_DEPLOYMENT}" \ + -n "${STAGING_NAMESPACE}" \ + --ignore-not-found=true + + kubectl delete service "${CANARY_SERVICE}" \ + -n "${STAGING_NAMESPACE}" \ + --ignore-not-found=true + + echo + echo "Verifying original stable deployment..." + + kubectl rollout status deployment/"${STABLE_DEPLOYMENT}" \ + -n "${STAGING_NAMESPACE}" \ + --timeout=180s + + CURRENT_IMAGE=$(kubectl get deployment "${STABLE_DEPLOYMENT}" \ + -n "${STAGING_NAMESPACE}" \ + -o jsonpath='{.spec.template.spec.containers[0].image}') + + echo "Original stable image : ${{ steps.stable.outputs.image }}" + echo "Current stable image : ${CURRENT_IMAGE}" + + if [ "${CURRENT_IMAGE}" != "${{ steps.stable.outputs.image }}" ]; then + echo "Stable image changed unexpectedly." + exit 1 + fi + + STABLE_IP=$(kubectl get service frontend \ + -n "${STAGING_NAMESPACE}" \ + -o jsonpath='{.status.loadBalancer.ingress[0].ip}') + + if [ -z "${STABLE_IP}" ]; then + STABLE_IP=$(kubectl get service frontend \ + -n "${STAGING_NAMESPACE}" \ + -o jsonpath='{.status.loadBalancer.ingress[0].hostname}') + fi + + curl --fail \ + --silent \ + --show-error \ + --retry 10 \ + --retry-delay 5 \ + --retry-all-errors \ + "http://${STABLE_IP}/" \ + > /dev/null + + echo + echo "AUTOMATIC RECOVERY SUCCESSFUL." + echo "Failed canary removed." + echo "Original stable release remains healthy." + + - name: Remove successful canary + if: >- + inputs.scenario == 'healthy' && + steps.rollout.outcome == 'success' && + steps.validation.outcome == 'success' + shell: bash + run: | + echo "Removing temporary canary resources after promotion..." + + kubectl delete deployment "${CANARY_DEPLOYMENT}" \ + -n "${STAGING_NAMESPACE}" \ + --ignore-not-found=true + + kubectl delete service "${CANARY_SERVICE}" \ + -n "${STAGING_NAMESPACE}" \ + --ignore-not-found=true + + - name: Final deployment evidence + if: always() + shell: bash + run: | + echo "============================================" + echo "FINAL CANARY DEPLOYMENT STATE" + echo "============================================" + + kubectl get deployments \ + -n "${STAGING_NAMESPACE}" \ + -o wide + + echo + kubectl get pods \ + -n "${STAGING_NAMESPACE}" \ + -o wide + + echo + kubectl get services \ + -n "${STAGING_NAMESPACE}" + + - name: Final scenario result + if: always() + shell: bash + run: | + if [ "${{ inputs.scenario }}" = "healthy" ]; then + if [ "${{ steps.rollout.outcome }}" != "success" ] || \ + [ "${{ steps.validation.outcome }}" != "success" ]; then + echo "Healthy canary scenario failed." + exit 1 + fi + + echo "============================================" + echo "HEALTHY CANARY SUCCESSFULLY PROMOTED" + echo "============================================" + else + echo "============================================" + echo "FAULTY CANARY REJECTED" + echo "STABLE RELEASE RETAINED" + echo "AUTOMATIC RECOVERY SUCCESSFUL" + echo "============================================" + fi diff --git a/.gitignore b/.gitignore index 4299cd1a..bf811163 100644 --- a/.gitignore +++ b/.gitignore @@ -1,199 +1,12 @@ -# Byte-compiled / optimized / DLL files -__pycache__/ -*.py[cod] -*$py.class - -# C extensions -*.so - -# MacOS -.DS_Store - -# Distribution / packaging -.Python -build/ -develop-eggs/ -dist/ -downloads/ -eggs/ -.eggs/ -lib/ -lib64/ -parts/ -sdist/ -var/ -wheels/ -share/python-wheels/ -*.egg-info/ -.installed.cfg -*.egg -MANIFEST - -# PyInstaller -# Usually these files are written by a python script from a template -# before PyInstaller builds the exe, so as to inject date/other infos into it. -*.manifest -*.spec - -# Installer logs -pip-log.txt -pip-delete-this-directory.txt - -# Unit test / coverage reports -htmlcov/ -.tox/ -.nox/ -.coverage -.coverage.* -.cache -nosetests.xml -coverage.xml -*.cover -*.py,cover -.hypothesis/ -.pytest_cache/ -cover/ - -# Translations -*.mo -*.pot - -# Django stuff: -*.log -local_settings.py -db.sqlite3 -db.sqlite3-journal - -# Flask stuff: -instance/ -.webassets-cache - -# Scrapy stuff: -.scrapy - -# Sphinx documentation -docs/_build/ - -# PyBuilder -.pybuilder/ -target/ - -# Jupyter Notebook -.ipynb_checkpoints - -# IPython -profile_default/ -ipython_config.py - -# pyenv -# For a library or package, you might want to ignore these files since the code is -# intended to run in multiple environments; otherwise, check them in: -# .python-version - -# pipenv -# According to pypa/pipenv#598, it is recommended to include Pipfile.lock in version control. -# However, in case of collaboration, if having platform-specific dependencies or dependencies -# having no cross-platform support, pipenv may install dependencies that don't work, or not -# install all needed dependencies. -#Pipfile.lock - -# UV -# Similar to Pipfile.lock, it is generally recommended to include uv.lock in version control. -# This is especially recommended for binary packages to ensure reproducibility, and is more -# commonly ignored for libraries. -#uv.lock - -# poetry -# Similar to Pipfile.lock, it is generally recommended to include poetry.lock in version control. -# This is especially recommended for binary packages to ensure reproducibility, and is more -# commonly ignored for libraries. -# https://python-poetry.org/docs/basic-usage/#commit-your-poetrylock-file-to-version-control -#poetry.lock +terraform/.terraform/ +terraform/*.tfstate +terraform/*.tfstate.* +terraform/*.tfplan +terraform/terraform.tfvars -# pdm -# Similar to Pipfile.lock, it is generally recommended to include pdm.lock in version control. -#pdm.lock -# pdm stores project-wide configurations in .pdm.toml, but it is recommended to not include it -# in version control. -# https://pdm.fming.dev/latest/usage/project/#working-with-version-control -.pdm.toml -.pdm-python -.pdm-build/ +# Node.js dependencies +node_modules/ -# PEP 582; used by e.g. github.com/David-OConnor/pyflow and github.com/pdm-project/pdm -__pypackages__/ - -# Celery stuff -celerybeat-schedule -celerybeat.pid - -# SageMath parsed files -*.sage.py - -# Environments -.env -.venv -env/ -venv/ -ENV/ -env.bak/ -venv.bak/ - -# Spyder project settings -.spyderproject -.spyproject - -# Rope project settings -.ropeproject - -# mkdocs documentation -/site - -# mypy -.mypy_cache/ -.dmypy.json -dmypy.json - -# Pyre type checker -.pyre/ - -# pytype static type analyzer -.pytype/ - -# Cython debug symbols -cython_debug/ - -# PyCharm -# JetBrains specific template is maintained in a separate JetBrains.gitignore that can -# be found at https://github.com/github/gitignore/blob/main/Global/JetBrains.gitignore -# and can be added to the global gitignore or merged into this file. For a more nuclear -# option (not recommended) you can uncomment the following to ignore the entire idea folder. -#.idea/ - -# Abstra -# Abstra is an AI-powered process automation framework. -# Ignore directories containing user credentials, local state, and settings. -# Learn more at https://abstra.io/docs -.abstra/ - -# Visual Studio Code -# Visual Studio Code specific template is maintained in a separate VisualStudioCode.gitignore -# that can be found at https://github.com/github/gitignore/blob/main/Global/VisualStudioCode.gitignore -# and can be added to the global gitignore or merged into this file. However, if you prefer, -# you could uncomment the following to ignore the enitre vscode folder -# .vscode/ - -# Ruff stuff: -.ruff_cache/ - -# PyPI configuration file -.pypirc - -# Cursor -# Cursor is an AI-powered code editor. `.cursorignore` specifies files/directories to -# exclude from AI features like autocomplete and code analysis. Recommended for sensitive data -# refer to https://docs.cursor.com/context/ignore-files -.cursorignore -.cursorindexingignore - -node_modules/ \ No newline at end of file +# Python generated files +__pycache__/ +*.py[cod] diff --git a/course-service/app/main.py b/course-service/app/main.py index 7c8cbfb2..6d0670ec 100644 --- a/course-service/app/main.py +++ b/course-service/app/main.py @@ -2,7 +2,13 @@ import time from contextlib import asynccontextmanager -from fastapi import FastAPI +from fastapi import FastAPI, Request, Response +from prometheus_client import ( + CONTENT_TYPE_LATEST, + Counter, + Histogram, + generate_latest, +) from sqlalchemy.exc import OperationalError from app.db import Base, engine @@ -16,6 +22,20 @@ logger = logging.getLogger(__name__) +SERVICE_NAME = "course-service" + +HTTP_REQUESTS_TOTAL = Counter( + "koalatech_http_requests_total", + "Total number of HTTP requests.", + ["service", "method", "path", "status_code"], +) + +HTTP_REQUEST_DURATION_SECONDS = Histogram( + "koalatech_http_request_duration_seconds", + "HTTP request duration in seconds.", + ["service", "method", "path"], +) + def initialise_database() -> None: maximum_attempts = 10 @@ -64,6 +84,34 @@ async def lifespan(_: FastAPI): ) +@app.middleware("http") +async def prometheus_metrics( + request: Request, + call_next, +): + start_time = time.perf_counter() + response = await call_next(request) + duration = time.perf_counter() - start_time + + route = request.scope.get("route") + path = getattr(route, "path", request.url.path) + + HTTP_REQUESTS_TOTAL.labels( + service=SERVICE_NAME, + method=request.method, + path=path, + status_code=str(response.status_code), + ).inc() + + HTTP_REQUEST_DURATION_SECONDS.labels( + service=SERVICE_NAME, + method=request.method, + path=path, + ).observe(duration) + + return response + + app.include_router(courses.router) @@ -80,5 +128,16 @@ def root() -> dict[str, str]: def health_check() -> dict[str, str]: return { "status": "healthy", - "service": "course-service", - } \ No newline at end of file + "service": SERVICE_NAME, + } + + +@app.get( + "/metrics", + include_in_schema=False, +) +def metrics() -> Response: + return Response( + content=generate_latest(), + media_type=CONTENT_TYPE_LATEST, + ) diff --git a/course-service/requirements.txt b/course-service/requirements.txt index 58a1178e..dbf225c5 100644 --- a/course-service/requirements.txt +++ b/course-service/requirements.txt @@ -6,4 +6,5 @@ pydantic==2.11.7 PyJWT==2.10.1 python-dotenv==1.0.1 pytest==8.4.1 -httpx==0.28.1 \ No newline at end of file +httpx==0.28.1 +prometheus-client==0.23.1 diff --git a/enrollment-service/app/main.py b/enrollment-service/app/main.py index 868063be..1b33edd3 100644 --- a/enrollment-service/app/main.py +++ b/enrollment-service/app/main.py @@ -2,7 +2,13 @@ import time from contextlib import asynccontextmanager -from fastapi import FastAPI +from fastapi import FastAPI, Request, Response +from prometheus_client import ( + CONTENT_TYPE_LATEST, + Counter, + Histogram, + generate_latest, +) from sqlalchemy.exc import OperationalError from app.db import Base, engine @@ -16,6 +22,20 @@ logger = logging.getLogger(__name__) +SERVICE_NAME = "enrollment-service" + +HTTP_REQUESTS_TOTAL = Counter( + "koalatech_http_requests_total", + "Total number of HTTP requests.", + ["service", "method", "path", "status_code"], +) + +HTTP_REQUEST_DURATION_SECONDS = Histogram( + "koalatech_http_request_duration_seconds", + "HTTP request duration in seconds.", + ["service", "method", "path"], +) + def initialise_database() -> None: maximum_attempts = 10 @@ -64,6 +84,34 @@ async def lifespan(_: FastAPI): ) +@app.middleware("http") +async def prometheus_metrics( + request: Request, + call_next, +): + start_time = time.perf_counter() + response = await call_next(request) + duration = time.perf_counter() - start_time + + route = request.scope.get("route") + path = getattr(route, "path", request.url.path) + + HTTP_REQUESTS_TOTAL.labels( + service=SERVICE_NAME, + method=request.method, + path=path, + status_code=str(response.status_code), + ).inc() + + HTTP_REQUEST_DURATION_SECONDS.labels( + service=SERVICE_NAME, + method=request.method, + path=path, + ).observe(duration) + + return response + + app.include_router(enrollments.router) @@ -80,5 +128,16 @@ def root() -> dict[str, str]: def health_check() -> dict[str, str]: return { "status": "healthy", - "service": "enrollment-service", - } \ No newline at end of file + "service": SERVICE_NAME, + } + + +@app.get( + "/metrics", + include_in_schema=False, +) +def metrics() -> Response: + return Response( + content=generate_latest(), + media_type=CONTENT_TYPE_LATEST, + ) diff --git a/enrollment-service/requirements.txt b/enrollment-service/requirements.txt index 58a1178e..dbf225c5 100644 --- a/enrollment-service/requirements.txt +++ b/enrollment-service/requirements.txt @@ -6,4 +6,5 @@ pydantic==2.11.7 PyJWT==2.10.1 python-dotenv==1.0.1 pytest==8.4.1 -httpx==0.28.1 \ No newline at end of file +httpx==0.28.1 +prometheus-client==0.23.1 diff --git a/frontend/src/pages/Login.jsx b/frontend/src/pages/Login.jsx index 1132741c..2918a615 100644 --- a/frontend/src/pages/Login.jsx +++ b/frontend/src/pages/Login.jsx @@ -98,11 +98,17 @@ const Login = () => { Sign in to continue + + Successfully deployed automatically through GitHub Actions + + {error && ( { ) ).toBeInTheDocument(); + expect( + screen.getByText( + "Successfully deployed automatically through GitHub Actions" + ) + ).toBeInTheDocument(); + expect( screen.getByRole("textbox", { name: /username/i, @@ -53,7 +59,6 @@ describe("Login page", () => { expect( screen.getByLabelText(/password/i) ).toBeInTheDocument(); - expect( screen.getByRole("button", { name: /login/i, diff --git a/kubernetes/canary/frontend-canary-deployment.yaml b/kubernetes/canary/frontend-canary-deployment.yaml new file mode 100644 index 00000000..d050c19c --- /dev/null +++ b/kubernetes/canary/frontend-canary-deployment.yaml @@ -0,0 +1,42 @@ +apiVersion: apps/v1 +kind: Deployment +metadata: + name: frontend-canary + namespace: staging + labels: + app: frontend + track: canary +spec: + replicas: 1 + selector: + matchLabels: + app: frontend + track: canary + template: + metadata: + labels: + app: frontend + track: canary + spec: + containers: + - name: frontend + image: CANARY_IMAGE + imagePullPolicy: Always + ports: + - containerPort: 80 + readinessProbe: + httpGet: + path: / + port: 80 + initialDelaySeconds: 5 + periodSeconds: 5 + timeoutSeconds: 3 + failureThreshold: 6 + livenessProbe: + httpGet: + path: / + port: 80 + initialDelaySeconds: 10 + periodSeconds: 10 + timeoutSeconds: 3 + failureThreshold: 3 diff --git a/kubernetes/canary/frontend-canary-service.yaml b/kubernetes/canary/frontend-canary-service.yaml new file mode 100644 index 00000000..3b9237fd --- /dev/null +++ b/kubernetes/canary/frontend-canary-service.yaml @@ -0,0 +1,17 @@ +apiVersion: v1 +kind: Service +metadata: + name: frontend-canary + namespace: staging + labels: + app: frontend + track: canary +spec: + type: ClusterIP + selector: + app: frontend + track: canary + ports: + - name: http + port: 80 + targetPort: 80 diff --git a/kubernetes/monitoring/01-namespace.yaml b/kubernetes/monitoring/01-namespace.yaml new file mode 100644 index 00000000..d3252360 --- /dev/null +++ b/kubernetes/monitoring/01-namespace.yaml @@ -0,0 +1,4 @@ +apiVersion: v1 +kind: Namespace +metadata: + name: monitoring diff --git a/kubernetes/monitoring/02-prometheus-rbac.yaml b/kubernetes/monitoring/02-prometheus-rbac.yaml new file mode 100644 index 00000000..bb268018 --- /dev/null +++ b/kubernetes/monitoring/02-prometheus-rbac.yaml @@ -0,0 +1,46 @@ +apiVersion: v1 +kind: ServiceAccount +metadata: + name: prometheus + namespace: monitoring + +--- +apiVersion: rbac.authorization.k8s.io/v1 +kind: ClusterRole +metadata: + name: prometheus +rules: + - apiGroups: [""] + resources: + - nodes + - nodes/proxy + - services + - endpoints + - pods + verbs: + - get + - list + - watch + - apiGroups: + - extensions + - networking.k8s.io + resources: + - ingresses + verbs: + - get + - list + - watch + +--- +apiVersion: rbac.authorization.k8s.io/v1 +kind: ClusterRoleBinding +metadata: + name: prometheus +roleRef: + apiGroup: rbac.authorization.k8s.io + kind: ClusterRole + name: prometheus +subjects: + - kind: ServiceAccount + name: prometheus + namespace: monitoring diff --git a/kubernetes/monitoring/03-prometheus-config.yaml b/kubernetes/monitoring/03-prometheus-config.yaml new file mode 100644 index 00000000..d40f0350 --- /dev/null +++ b/kubernetes/monitoring/03-prometheus-config.yaml @@ -0,0 +1,52 @@ +apiVersion: v1 +kind: ConfigMap +metadata: + name: prometheus-config + namespace: monitoring +data: + prometheus.yml: | + global: + scrape_interval: 15s + evaluation_interval: 15s + + scrape_configs: + + - job_name: prometheus + static_configs: + - targets: + - localhost:9090 + + - job_name: user-service + metrics_path: /metrics + static_configs: + - targets: + - user-service.staging.svc.cluster.local:8000 + + - job_name: student-service + metrics_path: /metrics + static_configs: + - targets: + - student-service.staging.svc.cluster.local:8000 + + - job_name: lecturer-service + metrics_path: /metrics + static_configs: + - targets: + - lecturer-service.staging.svc.cluster.local:8000 + + - job_name: course-service + metrics_path: /metrics + static_configs: + - targets: + - course-service.staging.svc.cluster.local:8000 + + - job_name: enrollment-service + metrics_path: /metrics + static_configs: + - targets: + - enrollment-service.staging.svc.cluster.local:8000 + + - job_name: kube-state-metrics + static_configs: + - targets: + - kube-state-metrics.monitoring.svc.cluster.local:8080 diff --git a/kubernetes/monitoring/04-prometheus.yaml b/kubernetes/monitoring/04-prometheus.yaml new file mode 100644 index 00000000..69bed06a --- /dev/null +++ b/kubernetes/monitoring/04-prometheus.yaml @@ -0,0 +1,56 @@ +apiVersion: apps/v1 +kind: Deployment +metadata: + name: prometheus + namespace: monitoring +spec: + replicas: 1 + selector: + matchLabels: + app: prometheus + template: + metadata: + labels: + app: prometheus + spec: + serviceAccountName: prometheus + containers: + - name: prometheus + image: prom/prometheus:v3.5.0 + imagePullPolicy: IfNotPresent + args: + - --config.file=/etc/prometheus/prometheus.yml + - --storage.tsdb.path=/prometheus + - --storage.tsdb.retention.time=6h + ports: + - name: http + containerPort: 9090 + resources: + requests: + cpu: 50m + memory: 128Mi + limits: + cpu: 250m + memory: 384Mi + volumeMounts: + - name: prometheus-config + mountPath: /etc/prometheus + volumes: + - name: prometheus-config + configMap: + name: prometheus-config + +--- +apiVersion: v1 +kind: Service +metadata: + name: prometheus + namespace: monitoring +spec: + selector: + app: prometheus + ports: + - name: http + port: 9090 + targetPort: 9090 + type: ClusterIP diff --git a/kubernetes/monitoring/05-kube-state-metrics.yaml b/kubernetes/monitoring/05-kube-state-metrics.yaml new file mode 100644 index 00000000..d7e4e15d --- /dev/null +++ b/kubernetes/monitoring/05-kube-state-metrics.yaml @@ -0,0 +1,108 @@ +apiVersion: v1 +kind: ServiceAccount +metadata: + name: kube-state-metrics + namespace: monitoring + +--- +apiVersion: rbac.authorization.k8s.io/v1 +kind: ClusterRole +metadata: + name: kube-state-metrics +rules: + - apiGroups: [""] + resources: + - configmaps + - secrets + - nodes + - pods + - services + - resourcequotas + - replicationcontrollers + - limitranges + - persistentvolumeclaims + - persistentvolumes + - namespaces + - endpoints + verbs: + - list + - watch + - apiGroups: + - apps + resources: + - statefulsets + - daemonsets + - deployments + - replicasets + verbs: + - list + - watch + - apiGroups: + - batch + resources: + - cronjobs + - jobs + verbs: + - list + - watch + +--- +apiVersion: rbac.authorization.k8s.io/v1 +kind: ClusterRoleBinding +metadata: + name: kube-state-metrics +roleRef: + apiGroup: rbac.authorization.k8s.io + kind: ClusterRole + name: kube-state-metrics +subjects: + - kind: ServiceAccount + name: kube-state-metrics + namespace: monitoring + +--- +apiVersion: apps/v1 +kind: Deployment +metadata: + name: kube-state-metrics + namespace: monitoring +spec: + replicas: 1 + selector: + matchLabels: + app: kube-state-metrics + template: + metadata: + labels: + app: kube-state-metrics + spec: + serviceAccountName: kube-state-metrics + containers: + - name: kube-state-metrics + image: registry.k8s.io/kube-state-metrics/kube-state-metrics:v2.17.0 + imagePullPolicy: IfNotPresent + ports: + - name: http + containerPort: 8080 + resources: + requests: + cpu: 20m + memory: 32Mi + limits: + cpu: 100m + memory: 128Mi + +--- +apiVersion: v1 +kind: Service +metadata: + name: kube-state-metrics + namespace: monitoring +spec: + selector: + app: kube-state-metrics + ports: + - name: http + port: 8080 + targetPort: 8080 + type: ClusterIP diff --git a/kubernetes/monitoring/06-grafana-datasource.yaml b/kubernetes/monitoring/06-grafana-datasource.yaml new file mode 100644 index 00000000..3aa1e3be --- /dev/null +++ b/kubernetes/monitoring/06-grafana-datasource.yaml @@ -0,0 +1,17 @@ +apiVersion: v1 +kind: ConfigMap +metadata: + name: grafana-datasource + namespace: monitoring +data: + datasource.yaml: | + apiVersion: 1 + + datasources: + - name: Prometheus + uid: prometheus + type: prometheus + access: proxy + url: http://prometheus.monitoring.svc.cluster.local:9090 + isDefault: true + editable: false diff --git a/kubernetes/monitoring/07-grafana-dashboard.yaml b/kubernetes/monitoring/07-grafana-dashboard.yaml new file mode 100644 index 00000000..7ede90f3 --- /dev/null +++ b/kubernetes/monitoring/07-grafana-dashboard.yaml @@ -0,0 +1,157 @@ +apiVersion: v1 +kind: ConfigMap +metadata: + name: grafana-dashboard + namespace: monitoring +data: + koalatech-monitoring.json: | + { + "annotations": { + "list": [] + }, + "editable": true, + "panels": [ + { + "type": "stat", + "title": "HTTP Requests by Backend Service", + "datasource": { + "type": "prometheus", + "uid": "prometheus" + }, + "targets": [ + { + "expr": "sum by (service) (koalatech_http_requests_total)", + "legendFormat": "{{service}}", + "refId": "A" + } + ], + "gridPos": { + "h": 8, + "w": 12, + "x": 0, + "y": 0 + } + }, + { + "type": "timeseries", + "title": "Backend Request Rate", + "datasource": { + "type": "prometheus", + "uid": "prometheus" + }, + "targets": [ + { + "expr": "sum by (service) (rate(koalatech_http_requests_total[5m]))", + "legendFormat": "{{service}}", + "refId": "A" + } + ], + "gridPos": { + "h": 8, + "w": 12, + "x": 12, + "y": 0 + } + }, + { + "type": "timeseries", + "title": "Backend P95 Request Duration", + "datasource": { + "type": "prometheus", + "uid": "prometheus" + }, + "targets": [ + { + "expr": "histogram_quantile(0.95, sum by (le, service) (rate(koalatech_http_request_duration_seconds_bucket[5m])))", + "legendFormat": "{{service}}", + "refId": "A" + } + ], + "gridPos": { + "h": 8, + "w": 12, + "x": 0, + "y": 8 + } + }, + { + "type": "stat", + "title": "Running Staging Pods", + "datasource": { + "type": "prometheus", + "uid": "prometheus" + }, + "targets": [ + { + "expr": "sum(kube_pod_status_phase{namespace=\"staging\",phase=\"Running\"})", + "refId": "A" + } + ], + "gridPos": { + "h": 8, + "w": 6, + "x": 12, + "y": 8 + } + }, + { + "type": "stat", + "title": "Available Staging Deployment Replicas", + "datasource": { + "type": "prometheus", + "uid": "prometheus" + }, + "targets": [ + { + "expr": "sum(kube_deployment_status_replicas_available{namespace=\"staging\"})", + "refId": "A" + } + ], + "gridPos": { + "h": 8, + "w": 6, + "x": 18, + "y": 8 + } + }, + { + "type": "timeseries", + "title": "Staging Container Restarts", + "datasource": { + "type": "prometheus", + "uid": "prometheus" + }, + "targets": [ + { + "expr": "sum by (pod) (kube_pod_container_status_restarts_total{namespace=\"staging\"})", + "legendFormat": "{{pod}}", + "refId": "A" + } + ], + "gridPos": { + "h": 8, + "w": 24, + "x": 0, + "y": 16 + } + } + ], + "refresh": "10s", + "schemaVersion": 41, + "tags": [ + "SIT722", + "KoalaTech", + "10.2D" + ], + "templating": { + "list": [] + }, + "time": { + "from": "now-15m", + "to": "now" + }, + "timezone": "browser", + "title": "KoalaTech - SIT722 Task 10.2D", + "uid": "koalatech-10-2d", + "version": 1 + } diff --git a/kubernetes/monitoring/08-grafana-dashboard-provider.yaml b/kubernetes/monitoring/08-grafana-dashboard-provider.yaml new file mode 100644 index 00000000..1fc6dd16 --- /dev/null +++ b/kubernetes/monitoring/08-grafana-dashboard-provider.yaml @@ -0,0 +1,19 @@ +apiVersion: v1 +kind: ConfigMap +metadata: + name: grafana-dashboard-provider + namespace: monitoring +data: + dashboard.yaml: | + apiVersion: 1 + + providers: + - name: KoalaTech + orgId: 1 + folder: SIT722 + type: file + disableDeletion: false + updateIntervalSeconds: 10 + allowUiUpdates: true + options: + path: /var/lib/grafana/dashboards diff --git a/kubernetes/monitoring/09-grafana.yaml b/kubernetes/monitoring/09-grafana.yaml new file mode 100644 index 00000000..ad604e60 --- /dev/null +++ b/kubernetes/monitoring/09-grafana.yaml @@ -0,0 +1,71 @@ +apiVersion: apps/v1 +kind: Deployment +metadata: + name: grafana + namespace: monitoring +spec: + replicas: 1 + selector: + matchLabels: + app: grafana + template: + metadata: + labels: + app: grafana + spec: + containers: + - name: grafana + image: grafana/grafana:12.1.1 + imagePullPolicy: IfNotPresent + ports: + - name: http + containerPort: 3000 + env: + - name: GF_SECURITY_ADMIN_USER + value: admin + - name: GF_SECURITY_ADMIN_PASSWORD + valueFrom: + secretKeyRef: + name: grafana-admin + key: password + - name: GF_USERS_ALLOW_SIGN_UP + value: "false" + resources: + requests: + cpu: 50m + memory: 128Mi + limits: + cpu: 250m + memory: 256Mi + volumeMounts: + - name: datasource + mountPath: /etc/grafana/provisioning/datasources + - name: dashboard-provider + mountPath: /etc/grafana/provisioning/dashboards + - name: dashboards + mountPath: /var/lib/grafana/dashboards + volumes: + - name: datasource + configMap: + name: grafana-datasource + - name: dashboard-provider + configMap: + name: grafana-dashboard-provider + - name: dashboards + configMap: + name: grafana-dashboard + +--- +apiVersion: v1 +kind: Service +metadata: + name: grafana + namespace: monitoring +spec: + selector: + app: grafana + ports: + - name: http + port: 3000 + targetPort: 3000 + type: ClusterIP diff --git a/kubernetes/staging/06-enrollment-db.yaml b/kubernetes/staging/06-enrollment-db.yaml index 3a94e0ee..557982be 100644 --- a/kubernetes/staging/06-enrollment-db.yaml +++ b/kubernetes/staging/06-enrollment-db.yaml @@ -1,16 +1,3 @@ -apiVersion: v1 -kind: PersistentVolumeClaim -metadata: - name: enrollment-db-pvc - namespace: staging -spec: - accessModes: - - ReadWriteOnce - resources: - requests: - storage: 1Gi - ---- apiVersion: apps/v1 kind: Deployment metadata: @@ -51,8 +38,7 @@ spec: mountPath: /var/lib/postgresql/data volumes: - name: enrollment-db-storage - persistentVolumeClaim: - claimName: enrollment-db-pvc + emptyDir: {} --- apiVersion: v1 @@ -65,4 +51,4 @@ spec: app: enrollment-db ports: - port: 5432 - targetPort: 5432 \ No newline at end of file + targetPort: 5432 diff --git a/kubernetes/staging/07-user-service.yaml b/kubernetes/staging/07-user-service.yaml index 7984533d..0b9ff098 100644 --- a/kubernetes/staging/07-user-service.yaml +++ b/kubernetes/staging/07-user-service.yaml @@ -15,7 +15,7 @@ spec: spec: containers: - name: user-service - image: placeholder + image: sit722week08acr224848845.azurecr.io/koalatech-user-service:latest imagePullPolicy: Always ports: - containerPort: 8000 diff --git a/kubernetes/staging/08-student-service.yaml b/kubernetes/staging/08-student-service.yaml index 244ed083..dbb6716c 100644 --- a/kubernetes/staging/08-student-service.yaml +++ b/kubernetes/staging/08-student-service.yaml @@ -15,7 +15,7 @@ spec: spec: containers: - name: student-service - image: placeholder + image: sit722week08acr224848845.azurecr.io/koalatech-student-service:latest imagePullPolicy: Always ports: - containerPort: 8000 diff --git a/kubernetes/staging/09-lecturer-service.yaml b/kubernetes/staging/09-lecturer-service.yaml index 48a611f6..48719a81 100644 --- a/kubernetes/staging/09-lecturer-service.yaml +++ b/kubernetes/staging/09-lecturer-service.yaml @@ -15,7 +15,7 @@ spec: spec: containers: - name: lecturer-service - image: placeholder + image: sit722week08acr224848845.azurecr.io/koalatech-lecturer-service:latest imagePullPolicy: Always ports: - containerPort: 8000 diff --git a/kubernetes/staging/10-course-service.yaml b/kubernetes/staging/10-course-service.yaml index 652cd06b..f0b1dbee 100644 --- a/kubernetes/staging/10-course-service.yaml +++ b/kubernetes/staging/10-course-service.yaml @@ -15,7 +15,7 @@ spec: spec: containers: - name: course-service - image: placeholder + image: sit722week08acr224848845.azurecr.io/koalatech-course-service:latest imagePullPolicy: Always ports: - containerPort: 8000 diff --git a/kubernetes/staging/11-enrollment-service.yaml b/kubernetes/staging/11-enrollment-service.yaml index 87a21d4f..275a286c 100644 --- a/kubernetes/staging/11-enrollment-service.yaml +++ b/kubernetes/staging/11-enrollment-service.yaml @@ -15,7 +15,7 @@ spec: spec: containers: - name: enrollment-service - image: placeholder + image: sit722week08acr224848845.azurecr.io/koalatech-enrollment-service:latest imagePullPolicy: Always ports: - containerPort: 8000 diff --git a/kubernetes/staging/12-frontend.yaml b/kubernetes/staging/12-frontend.yaml index 91a438da..fd8f353e 100644 --- a/kubernetes/staging/12-frontend.yaml +++ b/kubernetes/staging/12-frontend.yaml @@ -15,7 +15,7 @@ spec: spec: containers: - name: frontend - image: placeholder + image: sit722week08acr224848845.azurecr.io/koalatech-frontend:latest imagePullPolicy: Always ports: - containerPort: 80 diff --git a/lecturer-service/app/main.py b/lecturer-service/app/main.py index 62972d48..028676e8 100644 --- a/lecturer-service/app/main.py +++ b/lecturer-service/app/main.py @@ -2,7 +2,13 @@ import time from contextlib import asynccontextmanager -from fastapi import FastAPI +from fastapi import FastAPI, Request, Response +from prometheus_client import ( + CONTENT_TYPE_LATEST, + Counter, + Histogram, + generate_latest, +) from sqlalchemy.exc import OperationalError from app.db import Base, engine @@ -17,6 +23,20 @@ logger = logging.getLogger(__name__) +SERVICE_NAME = "lecturer-service" + +HTTP_REQUESTS_TOTAL = Counter( + "koalatech_http_requests_total", + "Total number of HTTP requests.", + ["service", "method", "path", "status_code"], +) + +HTTP_REQUEST_DURATION_SECONDS = Histogram( + "koalatech_http_request_duration_seconds", + "HTTP request duration in seconds.", + ["service", "method", "path"], +) + def initialise_database() -> None: maximum_attempts = 10 @@ -78,6 +98,34 @@ async def lifespan(_: FastAPI): ) +@app.middleware("http") +async def prometheus_metrics( + request: Request, + call_next, +): + start_time = time.perf_counter() + response = await call_next(request) + duration = time.perf_counter() - start_time + + route = request.scope.get("route") + path = getattr(route, "path", request.url.path) + + HTTP_REQUESTS_TOTAL.labels( + service=SERVICE_NAME, + method=request.method, + path=path, + status_code=str(response.status_code), + ).inc() + + HTTP_REQUEST_DURATION_SECONDS.labels( + service=SERVICE_NAME, + method=request.method, + path=path, + ).observe(duration) + + return response + + app.include_router(lecturers.router) @@ -94,5 +142,16 @@ def root() -> dict[str, str]: def health_check() -> dict[str, str]: return { "status": "healthy", - "service": "lecturer-service", - } \ No newline at end of file + "service": SERVICE_NAME, + } + + +@app.get( + "/metrics", + include_in_schema=False, +) +def metrics() -> Response: + return Response( + content=generate_latest(), + media_type=CONTENT_TYPE_LATEST, + ) diff --git a/lecturer-service/requirements.txt b/lecturer-service/requirements.txt index feeedd2d..72befaf6 100644 --- a/lecturer-service/requirements.txt +++ b/lecturer-service/requirements.txt @@ -8,4 +8,5 @@ python-multipart==0.0.20 python-dotenv==1.0.1 azure-storage-blob==12.26.0 pytest==8.4.1 -httpx==0.28.1 \ No newline at end of file +httpx==0.28.1 +prometheus-client==0.23.1 diff --git a/student-service/app/main.py b/student-service/app/main.py index 0d85f2ad..3a29e8a4 100644 --- a/student-service/app/main.py +++ b/student-service/app/main.py @@ -2,7 +2,13 @@ import time from contextlib import asynccontextmanager -from fastapi import FastAPI +from fastapi import FastAPI, Request, Response +from prometheus_client import ( + CONTENT_TYPE_LATEST, + Counter, + Histogram, + generate_latest, +) from sqlalchemy.exc import OperationalError from app.db import Base, engine @@ -17,6 +23,20 @@ logger = logging.getLogger(__name__) +SERVICE_NAME = "student-service" + +HTTP_REQUESTS_TOTAL = Counter( + "koalatech_http_requests_total", + "Total number of HTTP requests.", + ["service", "method", "path", "status_code"], +) + +HTTP_REQUEST_DURATION_SECONDS = Histogram( + "koalatech_http_request_duration_seconds", + "HTTP request duration in seconds.", + ["service", "method", "path"], +) + def initialise_database() -> None: maximum_attempts = 10 @@ -79,13 +99,38 @@ async def lifespan(_: FastAPI): ) +@app.middleware("http") +async def prometheus_metrics( + request: Request, + call_next, +): + start_time = time.perf_counter() + response = await call_next(request) + duration = time.perf_counter() - start_time + + route = request.scope.get("route") + path = getattr(route, "path", request.url.path) + + HTTP_REQUESTS_TOTAL.labels( + service=SERVICE_NAME, + method=request.method, + path=path, + status_code=str(response.status_code), + ).inc() + + HTTP_REQUEST_DURATION_SECONDS.labels( + service=SERVICE_NAME, + method=request.method, + path=path, + ).observe(duration) + + return response + + app.include_router(students.router) -@app.get( - "/", - tags=["Health"], -) +@app.get("/", tags=["Health"]) def root() -> dict[str, str]: return { "message": ( @@ -94,12 +139,20 @@ def root() -> dict[str, str]: } -@app.get( - "/health", - tags=["Health"], -) +@app.get("/health", tags=["Health"]) def health_check() -> dict[str, str]: return { "status": "healthy", - "service": "student-service", - } \ No newline at end of file + "service": SERVICE_NAME, + } + + +@app.get( + "/metrics", + include_in_schema=False, +) +def metrics() -> Response: + return Response( + content=generate_latest(), + media_type=CONTENT_TYPE_LATEST, + ) diff --git a/student-service/requirements.txt b/student-service/requirements.txt index feeedd2d..72befaf6 100644 --- a/student-service/requirements.txt +++ b/student-service/requirements.txt @@ -8,4 +8,5 @@ python-multipart==0.0.20 python-dotenv==1.0.1 azure-storage-blob==12.26.0 pytest==8.4.1 -httpx==0.28.1 \ No newline at end of file +httpx==0.28.1 +prometheus-client==0.23.1 diff --git a/terraform/.terraform.lock.hcl b/terraform/.terraform.lock.hcl new file mode 100644 index 00000000..1c302601 --- /dev/null +++ b/terraform/.terraform.lock.hcl @@ -0,0 +1,22 @@ +# This file is maintained automatically by "terraform init". +# Manual edits may be lost in future updates. + +provider "registry.terraform.io/hashicorp/azurerm" { + version = "4.81.0" + constraints = "~> 4.0" + hashes = [ + "h1:XhToZua4gtih1Kv8RdStcfND83G4Tmb6GZFT4jEUhDU=", + "zh:0732e7b74264ddfa2b90ba69d01c283d3cbae9f72ed3e506c6ac92529fed7fd3", + "zh:12afb524e232fe4e3d6161927724af5dfa4831d71edd9c174917ca9b7377bfae", + "zh:169d619ae202c4145e02fb706fb7c3679445ab3e3ff722edbf89597517a8c92e", + "zh:6beb95a3ef2f2d9c76abaa48e5450e90686a3fb6a47f1cb0ff7c5e94b6960151", + "zh:705e075fb5ffc4bf66fd7cbabf1a65007a41621e80030a2c158a4c83b6046216", + "zh:78d5eefdd9e494defcb3c68d282b8f96630502cac21d1ea161f53cfe9bb483b3", + "zh:79a8d17fefe647040fcb9ee8821a4f09f395427c4fd49493489b9a93a9a1038e", + "zh:8cc3f900b3774c0ae37ae42365c4579a199cf9e5edf88e476fdf5ab1048f84ea", + "zh:dec373b9390fa95e257291acd018ed65a7d512b428645d35e22cdbe8b245a08b", + "zh:e60f1e9fb45df6defade2855ed6e68547409ea75d30655c556adb0c08579749b", + "zh:f901d12ec82f3f8b5880a27b5cbcd7bd0d97e60c9367a2d7ed82fdd1157b39ff", + "zh:facf68ea5bf0f2b8ba720e7fba5f86492e1d4c591100460bb91c3f79f391f4b6", + ] +} diff --git a/terraform/backend.tf b/terraform/backend.tf new file mode 100644 index 00000000..6602f206 --- /dev/null +++ b/terraform/backend.tf @@ -0,0 +1,3 @@ +terraform { + backend "azurerm" {} +} diff --git a/terraform/container_registry.tf b/terraform/container_registry.tf new file mode 100644 index 00000000..36266676 --- /dev/null +++ b/terraform/container_registry.tf @@ -0,0 +1,15 @@ +resource "azurerm_container_registry" "acr" { + name = var.acr_name + resource_group_name = azurerm_resource_group.rg.name + location = azurerm_resource_group.rg.location + + sku = "Basic" + admin_enabled = false + + tags = { + project = "SIT722" + task = "10.2D" + environment = "week10" + purpose = "infrastructure-security-monitoring" + } +} diff --git a/terraform/kubernetes_service.tf b/terraform/kubernetes_service.tf new file mode 100644 index 00000000..ab5e73e4 --- /dev/null +++ b/terraform/kubernetes_service.tf @@ -0,0 +1,34 @@ +resource "azurerm_kubernetes_cluster" "aks" { + name = var.aks_name + location = azurerm_resource_group.rg.location + resource_group_name = azurerm_resource_group.rg.name + + dns_prefix = var.aks_name + + default_node_pool { + name = "default" + node_count = var.aks_node_count + vm_size = var.aks_vm_size + } + + identity { + type = "SystemAssigned" + } + + role_based_access_control_enabled = true + + tags = { + project = "SIT722" + task = "10.2D" + environment = "week10" + purpose = "infrastructure-security-monitoring" + } +} + +# Allow AKS to pull container images from ACR. +resource "azurerm_role_assignment" "acr_pull" { + principal_id = azurerm_kubernetes_cluster.aks.kubelet_identity[0].object_id + role_definition_name = "AcrPull" + scope = azurerm_container_registry.acr.id + skip_service_principal_aad_check = true +} diff --git a/terraform/outputs.tf b/terraform/outputs.tf new file mode 100644 index 00000000..283cdcb2 --- /dev/null +++ b/terraform/outputs.tf @@ -0,0 +1,30 @@ +output "resource_group_name" { + description = "Azure Resource Group name" + value = azurerm_resource_group.rg.name +} + +output "acr_name" { + description = "Azure Container Registry name" + value = azurerm_container_registry.acr.name +} + +output "acr_login_server" { + description = "Azure Container Registry login server" + value = azurerm_container_registry.acr.login_server +} + +output "aks_cluster_name" { + description = "AKS cluster name" + value = azurerm_kubernetes_cluster.aks.name +} + +output "storage_account_name" { + description = "Azure Storage Account name" + value = azurerm_storage_account.storage.name +} + +output "storage_connection_string" { + description = "Azure Storage Account connection string" + value = azurerm_storage_account.storage.primary_connection_string + sensitive = true +} \ No newline at end of file diff --git a/terraform/resource_group.tf b/terraform/resource_group.tf new file mode 100644 index 00000000..75d42f14 --- /dev/null +++ b/terraform/resource_group.tf @@ -0,0 +1,11 @@ +resource "azurerm_resource_group" "rg" { + name = var.resource_group_name + location = var.location + + tags = { + project = "SIT722" + task = "10.2D" + environment = "week10" + purpose = "infrastructure-security-monitoring" + } +} diff --git a/terraform/storage_account.tf b/terraform/storage_account.tf new file mode 100644 index 00000000..e9d9eaa8 --- /dev/null +++ b/terraform/storage_account.tf @@ -0,0 +1,17 @@ +resource "azurerm_storage_account" "storage" { + name = var.storage_account_name + resource_group_name = azurerm_resource_group.rg.name + location = azurerm_resource_group.rg.location + + account_tier = "Standard" + account_replication_type = "LRS" + + min_tls_version = "TLS1_2" + + tags = { + project = "SIT722" + task = "10.2D" + environment = "week10" + purpose = "infrastructure-security-monitoring" + } +} diff --git a/terraform/variables.tf b/terraform/variables.tf new file mode 100644 index 00000000..42291769 --- /dev/null +++ b/terraform/variables.tf @@ -0,0 +1,42 @@ +variable "resource_group_name" { + description = "Name of the Azure Resource Group" + type = string +} + +variable "location" { + description = "Azure region where resources will be deployed" + type = string + default = "Australia East" +} + +variable "acr_name" { + description = "Globally unique Azure Container Registry name" + type = string +} + +variable "aks_name" { + description = "Name of the Azure Kubernetes Service cluster" + type = string +} + +variable "storage_account_name" { + description = "Globally unique Azure Storage Account name" + type = string +} + +variable "aks_node_count" { + description = "Number of AKS worker nodes" + type = number + default = 1 + + validation { + condition = var.aks_node_count >= 1 + error_message = "AKS must contain at least one worker node." + } +} + +variable "aks_vm_size" { + description = "VM size used by the AKS default node pool" + type = string + default = "Standard_D2s_v3" +} diff --git a/terraform/versions.tf b/terraform/versions.tf new file mode 100644 index 00000000..841bbb65 --- /dev/null +++ b/terraform/versions.tf @@ -0,0 +1,14 @@ +terraform { + required_version = ">= 1.5.0" + + required_providers { + azurerm = { + source = "hashicorp/azurerm" + version = "~> 4.0" + } + } +} + +provider "azurerm" { + features {} +} \ No newline at end of file diff --git a/user-service/app/main.py b/user-service/app/main.py index 1b4769a6..19c65332 100644 --- a/user-service/app/main.py +++ b/user-service/app/main.py @@ -3,7 +3,13 @@ import time from contextlib import asynccontextmanager -from fastapi import FastAPI +from fastapi import FastAPI, Request, Response +from prometheus_client import ( + CONTENT_TYPE_LATEST, + Counter, + Histogram, + generate_latest, +) from sqlalchemy import select from sqlalchemy.exc import OperationalError from sqlalchemy.orm import Session @@ -21,6 +27,20 @@ logger = logging.getLogger(__name__) +SERVICE_NAME = "user-service" + +HTTP_REQUESTS_TOTAL = Counter( + "koalatech_http_requests_total", + "Total number of HTTP requests.", + ["service", "method", "path", "status_code"], +) + +HTTP_REQUEST_DURATION_SECONDS = Histogram( + "koalatech_http_request_duration_seconds", + "HTTP request duration in seconds.", + ["service", "method", "path"], +) + def initialise_database() -> None: maximum_attempts = 10 @@ -116,6 +136,36 @@ async def lifespan(_: FastAPI): ) +@app.middleware("http") +async def prometheus_metrics( + request: Request, + call_next, +): + start_time = time.perf_counter() + + response = await call_next(request) + + duration = time.perf_counter() - start_time + + route = request.scope.get("route") + path = getattr(route, "path", request.url.path) + + HTTP_REQUESTS_TOTAL.labels( + service=SERVICE_NAME, + method=request.method, + path=path, + status_code=str(response.status_code), + ).inc() + + HTTP_REQUEST_DURATION_SECONDS.labels( + service=SERVICE_NAME, + method=request.method, + path=path, + ).observe(duration) + + return response + + app.include_router(auth.router) app.include_router(users.router) @@ -137,5 +187,16 @@ def root() -> dict[str, str]: def health_check() -> dict[str, str]: return { "status": "healthy", - "service": "user-service", - } \ No newline at end of file + "service": SERVICE_NAME, + } + + +@app.get( + "/metrics", + include_in_schema=False, +) +def metrics() -> Response: + return Response( + content=generate_latest(), + media_type=CONTENT_TYPE_LATEST, + ) diff --git a/user-service/requirements.txt b/user-service/requirements.txt index 02679ad8..a52eb85a 100644 --- a/user-service/requirements.txt +++ b/user-service/requirements.txt @@ -5,7 +5,8 @@ psycopg2-binary==2.9.10 pydantic[email]==2.11.7 PyJWT pwdlib[argon2]==0.2.1 -python-multipart==0.0.20 +python-multipart==0.0.30 pytest==8.4.1 httpx==0.28.1 -python-dotenv==1.0.1 \ No newline at end of file +python-dotenv==1.0.1 +prometheus-client==0.23.1