diff --git a/.github/workflows/01-ci.yml b/.github/workflows/01-ci.yml
index 8206db43..451df1a2 100644
--- a/.github/workflows/01-ci.yml
+++ b/.github/workflows/01-ci.yml
@@ -1,17 +1,24 @@
name: 01 - CI
on:
- # Trigger the workflow on push to main branch
+ # =========================================================
+ # Pull Request Validation
+ # =========================================================
+ pull_request:
+ branches:
+ - main
+
+ # =========================================================
+ # Main Branch CI
+ # =========================================================
push:
branches:
- main
-
- # Manual trigger for the workflow
- workflow_dispatch:
+ # Allow controlled manual execution from GitHub Actions.
+ workflow_dispatch:
jobs:
-
# =========================================================
# Backend Tests
# =========================================================
@@ -81,7 +88,6 @@ jobs:
AZURE_STORAGE_CONTAINER_NAME: ""
steps:
-
- name: Checkout repository
uses: actions/checkout@v4
@@ -101,20 +107,164 @@ jobs:
run: |
pytest -v
+ # =========================================================
+ # Frontend Tests
+ # =========================================================
+ frontend-test:
+ name: Test Frontend
+ runs-on: ubuntu-latest
+
+ env:
+ VITE_USER_SERVICE_URL: http://localhost:8001
+ VITE_STUDENT_SERVICE_URL: http://localhost:8002
+ VITE_LECTURER_SERVICE_URL: http://localhost:8003
+ VITE_COURSE_SERVICE_URL: http://localhost:8004
+ VITE_ENROLLMENT_SERVICE_URL: http://localhost:8005
+
+ steps:
+ - name: Checkout repository
+ uses: actions/checkout@v4
+
+ - name: Set up Node.js
+ uses: actions/setup-node@v4
+ with:
+ node-version: "20"
+ cache: "npm"
+ cache-dependency-path: frontend/package-lock.json
+
+ - name: Install dependencies
+ working-directory: frontend
+ run: |
+ npm ci
+
+ - name: Run frontend tests
+ working-directory: frontend
+ run: |
+ npm run test -- --run
+
+ # =========================================================
+ # Terraform Validation, Plan and Provisioning
+ # Task 10.2D - Infrastructure as Code integration
+ #
+ # Pull Request:
+ # Init -> Format -> Validate -> Plan
+ #
+ # Main/Manual:
+ # Init -> Format -> Validate -> Plan -> Apply
+ # =========================================================
+ terraform-check:
+ name: Terraform Validate, Plan and Apply
+ runs-on: ubuntu-latest
+
+ # terraform.tfvars is intentionally not committed.
+ # Week10 values are supplied through TF_VAR_* variables.
+ env:
+ TF_VAR_resource_group_name: "sit722-week10-rg"
+ TF_VAR_location: "Australia East"
+ TF_VAR_acr_name: "sit722week10acr224848845"
+ TF_VAR_aks_name: "sit722-week10-aks"
+ TF_VAR_storage_account_name: "sit722w10storage224848"
+ TF_VAR_aks_node_count: "1"
+ TF_VAR_aks_vm_size: "Standard_D2s_v3"
+
+ steps:
+ - name: Checkout repository
+ uses: actions/checkout@v4
+
+ - name: Set up Terraform
+ uses: hashicorp/setup-terraform@v3
+
+ - name: Login to Azure
+ uses: azure/login@v3
+ with:
+ creds: ${{ secrets.AZURE_CREDENTIALS }}
+
+ # -------------------------------------------------------
+ # Persistent Terraform state
+ # -------------------------------------------------------
+ - name: Terraform Init
+ working-directory: terraform
+ run: |
+ terraform init \
+ -input=false \
+ -backend-config="resource_group_name=sit722-week10-tfstate-rg" \
+ -backend-config="storage_account_name=sit722w10tfstate224848" \
+ -backend-config="container_name=tfstate" \
+ -backend-config="key=week10-10.2d.tfstate"
+
+ - name: Terraform Format Check
+ working-directory: terraform
+ run: |
+ terraform fmt -check
+
+ - name: Terraform Validate
+ working-directory: terraform
+ run: |
+ terraform validate
+
+ # -------------------------------------------------------
+ # Generate a saved Terraform plan.
+ # -------------------------------------------------------
+ - name: Terraform Plan
+ working-directory: terraform
+ run: |
+ terraform plan \
+ -input=false \
+ -no-color \
+ -out=tfplan
+
+ # -------------------------------------------------------
+ # Infrastructure provisioning
+ #
+ # Pull requests NEVER provision infrastructure.
+ # Apply occurs only for main push or controlled
+ # workflow_dispatch execution.
+ # -------------------------------------------------------
+ - name: Terraform Apply
+ if: >
+ github.event_name == 'push' ||
+ github.event_name == 'workflow_dispatch'
+ working-directory: terraform
+ run: |
+ terraform apply \
+ -input=false \
+ -auto-approve \
+ tfplan
+
+ - name: Display Terraform Outputs
+ if: >
+ github.event_name == 'push' ||
+ github.event_name == 'workflow_dispatch'
+ working-directory: terraform
+ run: |
+ echo "===== TERRAFORM OUTPUTS ====="
+ terraform output
+
+ echo
+ echo "===== TERRAFORM STATE ====="
+ terraform state list
# =========================================================
- # Build and Push Docker Images
+ # Build, Security Scan and Push Docker Images
+ # Task 10.2D - Docker Scout integration
# =========================================================
- build-and-push:
- name: Build and Push ${{ matrix.image }}
+ build-scan-and-push:
+ name: Build, Scan and Push ${{ matrix.image }}
runs-on: ubuntu-latest
- # All backend tests must pass before this job starts
+ # Application tests AND Terraform infrastructure
+ # provisioning/validation must succeed before images
+ # are published.
needs:
- backend-test
+ - frontend-test
+ - terraform-check
- # Build and push when code is pushed to main or manually triggered
- if: github.event_name == 'push' || github.event_name == 'workflow_dispatch'
+ # Pull requests perform validation only.
+ # Images are published for main or controlled manual runs.
+ if: >
+ github.event_name == 'push' ||
+ github.event_name == 'workflow_dispatch'
strategy:
fail-fast: false
@@ -140,19 +290,28 @@ jobs:
image: koalatech-enrollment-service
steps:
-
- name: Checkout repository
uses: actions/checkout@v4
+ # -------------------------------------------------------
+ # Azure authentication
+ # -------------------------------------------------------
- name: Login to Azure
- uses: azure/login@v2
+ uses: azure/login@v3
with:
creds: ${{ secrets.AZURE_CREDENTIALS }}
+ # -------------------------------------------------------
+ # Authenticate with Azure Container Registry
+ # -------------------------------------------------------
- name: Login to Azure Container Registry
run: |
- az acr login --name ${{ vars.ACR_NAME }}
+ az acr login \
+ --name ${{ vars.ACR_NAME }}
+ # -------------------------------------------------------
+ # Build image locally before security analysis
+ # -------------------------------------------------------
- name: Build Docker image
run: |
docker build \
@@ -160,7 +319,56 @@ jobs:
-t ${{ vars.ACR_LOGIN_SERVER }}/${{ matrix.image }}:${{ github.sha }} \
./${{ matrix.service }}
+ # -------------------------------------------------------
+ # Docker Hub authentication for Docker Scout
+ #
+ # Required repository secrets:
+ # DOCKERHUB_USERNAME
+ # DOCKERHUB_TOKEN
+ # -------------------------------------------------------
+ - name: Login to Docker Hub for Docker Scout
+ uses: docker/login-action@v4
+ with:
+ username: ${{ secrets.DOCKERHUB_USERNAME }}
+ password: ${{ secrets.DOCKERHUB_TOKEN }}
+
+ # -------------------------------------------------------
+ # Docker Scout vulnerability analysis BEFORE ACR push
+ #
+ # exit-code is false so identified vulnerabilities are
+ # reported without preventing the assessment pipeline
+ # from continuing. Remediation is demonstrated separately.
+ # -------------------------------------------------------
+ - name: Docker Scout CVE scan
+ uses: docker/scout-action@v1
+ with:
+ command: cves
+ image: local://${{ vars.ACR_LOGIN_SERVER }}/${{ matrix.image }}:${{ github.sha }}
+ only-severities: critical,high
+ write-comment: false
+ exit-code: false
+ github-token: ${{ secrets.GITHUB_TOKEN }}
+
+ # -------------------------------------------------------
+ # Obtain Docker Scout remediation guidance
+ # -------------------------------------------------------
+ - name: Docker Scout remediation recommendations
+ uses: docker/scout-action@v1
+ with:
+ command: recommendations
+ image: local://${{ vars.ACR_LOGIN_SERVER }}/${{ matrix.image }}:${{ github.sha }}
+ write-comment: false
+ github-token: ${{ secrets.GITHUB_TOKEN }}
+
+ # -------------------------------------------------------
+ # Publish security-scanned image to ACR
+ # -------------------------------------------------------
- name: Push Docker image with commit SHA
run: |
docker push \
${{ vars.ACR_LOGIN_SERVER }}/${{ matrix.image }}:${{ github.sha }}
+
+ - name: Display published image
+ run: |
+ echo "Published image:"
+ echo "${{ vars.ACR_LOGIN_SERVER }}/${{ matrix.image }}:${{ github.sha }}"
\ No newline at end of file
diff --git a/.github/workflows/02-deploy-staging.yml b/.github/workflows/02-deploy-staging.yml
index 006bb70d..d0d2dc0c 100644
--- a/.github/workflows/02-deploy-staging.yml
+++ b/.github/workflows/02-deploy-staging.yml
@@ -11,7 +11,7 @@ on:
jobs:
deploy-staging:
- name: Deploy to Staging
+ name: Deploy Application and Monitoring to Staging
runs-on: ubuntu-latest
if: >
@@ -67,85 +67,271 @@ jobs:
--dry-run=client \
-o yaml | kubectl apply -f -
- - name: Apply Kubernetes manifests
+ - name: Apply staging Kubernetes manifests
run: |
- kubectl apply \
- -f kubernetes/staging/
+ kubectl apply -f kubernetes/staging/
- - name: Update frontend image
+ - name: Deploy tested application images
+ shell: bash
run: |
+ IMAGE_TAG="${{ github.event.workflow_run.head_sha }}"
+
kubectl set image deployment/frontend \
- frontend=${{ vars.ACR_LOGIN_SERVER }}/koalatech-frontend:${{ github.event.workflow_run.head_sha }} \
+ frontend=${{ vars.ACR_LOGIN_SERVER }}/koalatech-frontend:$IMAGE_TAG \
-n staging
- - name: Update user-service image
- run: |
kubectl set image deployment/user-service \
- user-service=${{ vars.ACR_LOGIN_SERVER }}/koalatech-user-service:${{ github.event.workflow_run.head_sha }} \
+ user-service=${{ vars.ACR_LOGIN_SERVER }}/koalatech-user-service:$IMAGE_TAG \
-n staging
- - name: Update student-service image
- run: |
kubectl set image deployment/student-service \
- student-service=${{ vars.ACR_LOGIN_SERVER }}/koalatech-student-service:${{ github.event.workflow_run.head_sha }} \
+ student-service=${{ vars.ACR_LOGIN_SERVER }}/koalatech-student-service:$IMAGE_TAG \
-n staging
- - name: Update lecturer-service image
- run: |
kubectl set image deployment/lecturer-service \
- lecturer-service=${{ vars.ACR_LOGIN_SERVER }}/koalatech-lecturer-service:${{ github.event.workflow_run.head_sha }} \
+ lecturer-service=${{ vars.ACR_LOGIN_SERVER }}/koalatech-lecturer-service:$IMAGE_TAG \
-n staging
- - name: Update course-service image
- run: |
kubectl set image deployment/course-service \
- course-service=${{ vars.ACR_LOGIN_SERVER }}/koalatech-course-service:${{ github.event.workflow_run.head_sha }} \
+ course-service=${{ vars.ACR_LOGIN_SERVER }}/koalatech-course-service:$IMAGE_TAG \
-n staging
- - name: Update enrollment-service image
- run: |
kubectl set image deployment/enrollment-service \
- enrollment-service=${{ vars.ACR_LOGIN_SERVER }}/koalatech-enrollment-service:${{ github.event.workflow_run.head_sha }} \
+ enrollment-service=${{ vars.ACR_LOGIN_SERVER }}/koalatech-enrollment-service:$IMAGE_TAG \
-n staging
- - name: Wait for frontend rollout
+ - name: Verify application rollouts
+ shell: bash
run: |
- kubectl rollout status deployment/frontend \
- -n staging \
- --timeout=300s
+ for deployment in \
+ frontend \
+ user-service \
+ student-service \
+ lecturer-service \
+ course-service \
+ enrollment-service
+ do
+ echo "Checking rollout: $deployment"
- - name: Wait for user-service rollout
+ kubectl rollout status deployment/$deployment \
+ -n staging \
+ --timeout=300s
+ done
+
+ - name: Create monitoring namespace
run: |
- kubectl rollout status deployment/user-service \
- -n staging \
- --timeout=300s
+ kubectl apply \
+ -f kubernetes/monitoring/01-namespace.yaml
- - name: Wait for student-service rollout
+ - name: Create Grafana admin secret
run: |
- kubectl rollout status deployment/student-service \
- -n staging \
- --timeout=300s
+ kubectl create secret generic grafana-admin \
+ --namespace monitoring \
+ --from-literal=password="${{ secrets.GRAFANA_ADMIN_PASSWORD }}" \
+ --dry-run=client \
+ -o yaml | kubectl apply -f -
- - name: Wait for lecturer-service rollout
+ - name: Deploy Prometheus and Grafana monitoring
run: |
- kubectl rollout status deployment/lecturer-service \
- -n staging \
- --timeout=300s
+ kubectl apply \
+ -f kubernetes/monitoring/
- - name: Wait for course-service rollout
+ - name: Verify monitoring rollouts
+ shell: bash
run: |
- kubectl rollout status deployment/course-service \
- -n staging \
- --timeout=300s
+ for deployment in \
+ kube-state-metrics \
+ prometheus \
+ grafana
+ do
+ echo "Checking monitoring deployment: $deployment"
+
+ kubectl rollout status deployment/$deployment \
+ -n monitoring \
+ --timeout=300s
+ done
- - name: Wait for enrollment-service rollout
+ - name: Verify Prometheus targets
+ shell: bash
run: |
- kubectl rollout status deployment/enrollment-service \
- -n staging \
- --timeout=300s
+ kubectl port-forward \
+ -n monitoring \
+ svc/prometheus \
+ 9090:9090 > /tmp/prometheus-port-forward.log 2>&1 &
+
+ PF_PID=$!
+
+ cleanup() {
+ kill "$PF_PID" 2>/dev/null || true
+ }
+
+ trap cleanup EXIT
+
+ echo "Waiting for Prometheus API..."
+
+ for attempt in {1..20}
+ do
+ if curl --fail --silent \
+ http://127.0.0.1:9090/-/ready > /dev/null
+ then
+ echo "Prometheus API is ready."
+ break
+ fi
+
+ if [ "$attempt" -eq 20 ]; then
+ echo "Prometheus did not become ready."
+ cat /tmp/prometheus-port-forward.log
+ exit 1
+ fi
+
+ echo "Prometheus API not ready yet - attempt ${attempt}/20"
+ sleep 3
+ done
+
+ echo
+ echo "Waiting for all required Prometheus targets to become healthy..."
+
+ for attempt in {1..12}
+ do
+ echo
+ echo "Target health check ${attempt}/12"
+
+ curl --fail --silent \
+ http://127.0.0.1:9090/api/v1/targets \
+ > /tmp/prometheus-targets.json
+
+ if python3 - <<'PYTHON'
+ import json
+
+ with open("/tmp/prometheus-targets.json") as file:
+ data = json.load(file)
+
+ targets = data["data"]["activeTargets"]
+
+ required_jobs = {
+ "user-service",
+ "student-service",
+ "lecturer-service",
+ "course-service",
+ "enrollment-service",
+ "kube-state-metrics",
+ }
+
+ target_status = {}
+
+ for target in targets:
+ labels = target.get("labels", {})
+ job = labels.get("job", "unknown")
+ health = target.get("health", "unknown")
+ scrape_url = target.get("scrapeUrl", "")
+ last_error = target.get("lastError", "")
+
+ if job in required_jobs:
+ target_status[job] = health
+
+ print(
+ f"{job:25} "
+ f"health={health:8} "
+ f"{scrape_url}"
+ )
+
+ if last_error:
+ print(f" error: {last_error}")
+
+ healthy_jobs = {
+ job
+ for job, health in target_status.items()
+ if health == "up"
+ }
+
+ missing = required_jobs - healthy_jobs
+
+ if missing:
+ print()
+ print(
+ "Targets not healthy yet: "
+ + ", ".join(sorted(missing))
+ )
+ raise SystemExit(1)
+
+ print()
+ print("All required Prometheus targets are healthy.")
+ PYTHON
+ then
+ echo
+ echo "Prometheus monitoring validation successful."
+ exit 0
+ fi
+
+ if [ "$attempt" -eq 12 ]; then
+ echo
+ echo "Required Prometheus targets did not become healthy within 120 seconds."
+ exit 1
+ fi
+
+ echo "Waiting 10 seconds before retry..."
+ sleep 10
+ done
+
+ - name: Verify Grafana health
+ shell: bash
+ run: |
+ kubectl port-forward \
+ -n monitoring \
+ svc/grafana \
+ 3000:3000 > /tmp/grafana-port-forward.log 2>&1 &
+
+ PF_PID=$!
+
+ cleanup() {
+ kill "$PF_PID" 2>/dev/null || true
+ }
+
+ trap cleanup EXIT
+
+ echo "Waiting for Grafana..."
+
+ for attempt in {1..20}
+ do
+ if curl --fail --silent \
+ http://127.0.0.1:3000/api/health
+ then
+ echo
+ echo "Grafana is healthy."
+ exit 0
+ fi
+
+ sleep 3
+ done
+
+ echo "Grafana health validation failed."
+ cat /tmp/grafana-port-forward.log
+ exit 1
- name: Show staging resources
+ if: always()
run: |
- kubectl get pods -n staging
+ echo "===== STAGING PODS ====="
+ kubectl get pods -n staging -o wide
+
+ echo
+ echo "===== STAGING SERVICES ====="
kubectl get services -n staging
- kubectl get pvc -n staging
\ No newline at end of file
+
+ echo
+ echo "===== STAGING PVCs ====="
+ kubectl get pvc -n staging
+
+ - name: Show monitoring resources
+ if: always()
+ run: |
+ echo "===== MONITORING DEPLOYMENTS ====="
+ kubectl get deployments -n monitoring
+
+ echo
+ echo "===== MONITORING PODS ====="
+ kubectl get pods -n monitoring -o wide
+
+ echo
+ echo "===== MONITORING SERVICES ====="
+ kubectl get services -n monitoring
diff --git a/.github/workflows/04-deploy-production.yml b/.github/workflows/04-deploy-production.yml
index 969d646b..5bda7b89 100644
--- a/.github/workflows/04-deploy-production.yml
+++ b/.github/workflows/04-deploy-production.yml
@@ -2,15 +2,14 @@ name: 04 - Deploy to Production
on:
workflow_dispatch:
- inputs:
- image_tag:
- description: "Tested image SHA to deploy"
- required: true
- type: string
jobs:
deploy-production:
name: Deploy to Production
+
+ # Production deployment is manual during the Week 10 task.
+ if: ${{ github.event_name == 'workflow_dispatch' }}
+
runs-on: ubuntu-latest
environment:
@@ -32,6 +31,60 @@ jobs:
--name ${{ vars.AKS_CLUSTER_NAME }} \
--overwrite-existing
+ # Read the exact frontend image that has already passed
+ # deployment and smoke testing in staging.
+ - name: Get tested image SHA from staging
+ id: tested-image
+ shell: bash
+ run: |
+ TESTED_IMAGE=$(kubectl get deployment frontend \
+ -n staging \
+ -o jsonpath='{.spec.template.spec.containers[0].image}')
+
+ IMAGE_TAG="${TESTED_IMAGE##*:}"
+
+ echo "Tested staging image: $TESTED_IMAGE"
+ echo "Promoting image tag: $IMAGE_TAG"
+
+ if [ -z "$IMAGE_TAG" ]; then
+ echo "Unable to determine tested staging image tag."
+ exit 1
+ fi
+
+ echo "image_tag=$IMAGE_TAG" >> "$GITHUB_OUTPUT"
+
+ # Verify that all six application deployments in staging
+ # are running the exact same tested release.
+ - name: Verify staging release consistency
+ shell: bash
+ run: |
+ IMAGE_TAG="${{ steps.tested-image.outputs.image_tag }}"
+
+ for deployment in \
+ frontend \
+ user-service \
+ student-service \
+ lecturer-service \
+ course-service \
+ enrollment-service
+ do
+ IMAGE=$(kubectl get deployment "$deployment" \
+ -n staging \
+ -o jsonpath='{.spec.template.spec.containers[0].image}')
+
+ echo "$deployment -> $IMAGE"
+
+ case "$IMAGE" in
+ *:"$IMAGE_TAG")
+ echo "$deployment uses tested release $IMAGE_TAG"
+ ;;
+ *)
+ echo "$deployment does not use tested release $IMAGE_TAG"
+ exit 1
+ ;;
+ esac
+ done
+
- name: Create production namespace
run: |
kubectl create namespace production \
@@ -63,80 +116,59 @@ jobs:
- name: Apply Kubernetes manifests
run: |
- kubectl apply \
- -f kubernetes/production/
+ kubectl apply -f kubernetes/production/
- - name: Update frontend image
+ - name: Promote tested images to production
+ shell: bash
run: |
+ IMAGE_TAG="${{ steps.tested-image.outputs.image_tag }}"
+
+ echo "Promoting tested release $IMAGE_TAG to production"
+
kubectl set image deployment/frontend \
- frontend=${{ vars.ACR_LOGIN_SERVER }}/koalatech-frontend:${{ inputs.image_tag }} \
+ frontend=${{ vars.ACR_LOGIN_SERVER }}/koalatech-frontend:$IMAGE_TAG \
-n production
- - name: Update user-service image
- run: |
kubectl set image deployment/user-service \
- user-service=${{ vars.ACR_LOGIN_SERVER }}/koalatech-user-service:${{ inputs.image_tag }} \
+ user-service=${{ vars.ACR_LOGIN_SERVER }}/koalatech-user-service:$IMAGE_TAG \
-n production
- - name: Update student-service image
- run: |
kubectl set image deployment/student-service \
- student-service=${{ vars.ACR_LOGIN_SERVER }}/koalatech-student-service:${{ inputs.image_tag }} \
+ student-service=${{ vars.ACR_LOGIN_SERVER }}/koalatech-student-service:$IMAGE_TAG \
-n production
- - name: Update lecturer-service image
- run: |
kubectl set image deployment/lecturer-service \
- lecturer-service=${{ vars.ACR_LOGIN_SERVER }}/koalatech-lecturer-service:${{ inputs.image_tag }} \
+ lecturer-service=${{ vars.ACR_LOGIN_SERVER }}/koalatech-lecturer-service:$IMAGE_TAG \
-n production
- - name: Update course-service image
- run: |
kubectl set image deployment/course-service \
- course-service=${{ vars.ACR_LOGIN_SERVER }}/koalatech-course-service:${{ inputs.image_tag }} \
+ course-service=${{ vars.ACR_LOGIN_SERVER }}/koalatech-course-service:$IMAGE_TAG \
-n production
- - name: Update enrollment-service image
- run: |
kubectl set image deployment/enrollment-service \
- enrollment-service=${{ vars.ACR_LOGIN_SERVER }}/koalatech-enrollment-service:${{ inputs.image_tag }} \
+ enrollment-service=${{ vars.ACR_LOGIN_SERVER }}/koalatech-enrollment-service:$IMAGE_TAG \
-n production
- - name: Wait for frontend rollout
- run: |
- kubectl rollout status deployment/frontend \
- -n production \
- --timeout=300s
-
- - name: Wait for user-service rollout
- run: |
- kubectl rollout status deployment/user-service \
- -n production \
- --timeout=300s
-
- - name: Wait for student-service rollout
- run: |
- kubectl rollout status deployment/student-service \
- -n production \
- --timeout=300s
-
- - name: Wait for lecturer-service rollout
- run: |
- kubectl rollout status deployment/lecturer-service \
- -n production \
- --timeout=300s
-
- - name: Wait for course-service rollout
- run: |
- kubectl rollout status deployment/course-service \
- -n production \
- --timeout=300s
-
- - name: Wait for enrollment-service rollout
- run: |
- kubectl rollout status deployment/enrollment-service \
- -n production \
- --timeout=300s
+ - name: Verify production rollouts
+ shell: bash
+ run: |
+ for deployment in \
+ frontend \
+ user-service \
+ student-service \
+ lecturer-service \
+ course-service \
+ enrollment-service
+ do
+ kubectl rollout status deployment/$deployment \
+ -n production \
+ --timeout=300s
+ done
+
+ - name: Verify production image versions
+ run: |
+ kubectl get deployments -n production \
+ -o custom-columns='DEPLOYMENT:.metadata.name,IMAGE:.spec.template.spec.containers[*].image'
- name: Show production resources
run: |
diff --git a/.github/workflows/05-canary-deployment.yml b/.github/workflows/05-canary-deployment.yml
new file mode 100644
index 00000000..a4034b28
--- /dev/null
+++ b/.github/workflows/05-canary-deployment.yml
@@ -0,0 +1,313 @@
+name: 05 - Canary Deployment and Automatic Rollback
+
+on:
+ workflow_dispatch:
+ inputs:
+ scenario:
+ description: "Canary demonstration scenario"
+ required: true
+ default: healthy
+ type: choice
+ options:
+ - healthy
+ - faulty
+
+permissions:
+ contents: read
+
+env:
+ STAGING_NAMESPACE: staging
+ CANARY_DEPLOYMENT: frontend-canary
+ CANARY_SERVICE: frontend-canary
+ STABLE_DEPLOYMENT: frontend
+
+jobs:
+ canary-deployment:
+ name: Canary Deployment, Validation and Recovery
+ runs-on: ubuntu-latest
+
+ steps:
+ - name: Checkout repository
+ uses: actions/checkout@v4
+
+ - name: Azure login
+ uses: azure/login@v3
+ with:
+ creds: ${{ secrets.AZURE_CREDENTIALS }}
+
+ - name: Get AKS credentials
+ shell: bash
+ run: |
+ az aks get-credentials \
+ --resource-group "${{ vars.AKS_RESOURCE_GROUP }}" \
+ --name "${{ vars.AKS_CLUSTER_NAME }}" \
+ --overwrite-existing
+
+ kubectl get nodes
+
+ - name: Capture current stable release
+ id: stable
+ shell: bash
+ run: |
+ STABLE_IMAGE=$(kubectl get deployment "${STABLE_DEPLOYMENT}" \
+ -n "${STAGING_NAMESPACE}" \
+ -o jsonpath='{.spec.template.spec.containers[0].image}')
+
+ echo "Current stable image: ${STABLE_IMAGE}"
+ echo "image=${STABLE_IMAGE}" >> "$GITHUB_OUTPUT"
+
+ - name: Verify stable application before canary
+ shell: bash
+ run: |
+ kubectl rollout status deployment/"${STABLE_DEPLOYMENT}" \
+ -n "${STAGING_NAMESPACE}" \
+ --timeout=180s
+
+ STABLE_IP=$(kubectl get service frontend \
+ -n "${STAGING_NAMESPACE}" \
+ -o jsonpath='{.status.loadBalancer.ingress[0].ip}')
+
+ if [ -z "${STABLE_IP}" ]; then
+ STABLE_IP=$(kubectl get service frontend \
+ -n "${STAGING_NAMESPACE}" \
+ -o jsonpath='{.status.loadBalancer.ingress[0].hostname}')
+ fi
+
+ echo "Stable frontend address: ${STABLE_IP}"
+
+ curl --fail \
+ --silent \
+ --show-error \
+ --retry 10 \
+ --retry-delay 5 \
+ --retry-all-errors \
+ "http://${STABLE_IP}/" \
+ > /dev/null
+
+ echo "Stable application is healthy before canary deployment."
+
+ - name: Select canary image
+ id: candidate
+ shell: bash
+ run: |
+ if [ "${{ inputs.scenario }}" = "healthy" ]; then
+ CANARY_IMAGE="${{ steps.stable.outputs.image }}"
+ echo "Healthy scenario selected."
+ echo "Candidate image: ${CANARY_IMAGE}"
+ else
+ CANARY_IMAGE="nginx:1.27-alpine"
+ echo "Faulty scenario selected."
+ echo "The canary will be configured with an invalid health endpoint."
+ echo "Candidate image: ${CANARY_IMAGE}"
+ fi
+
+ echo "image=${CANARY_IMAGE}" >> "$GITHUB_OUTPUT"
+
+ - name: Deploy canary candidate
+ shell: bash
+ run: |
+ sed \
+ "s|CANARY_IMAGE|${{ steps.candidate.outputs.image }}|g" \
+ kubernetes/canary/frontend-canary-deployment.yaml \
+ > /tmp/frontend-canary-deployment.yaml
+
+ if [ "${{ inputs.scenario }}" = "faulty" ]; then
+ echo "Injecting deliberately invalid health endpoint for rollback demonstration."
+
+ sed -i \
+ 's|path: /$|path: /deliberately-faulty-health-endpoint|g' \
+ /tmp/frontend-canary-deployment.yaml
+ fi
+
+ kubectl apply \
+ -f /tmp/frontend-canary-deployment.yaml
+
+ kubectl apply \
+ -f kubernetes/canary/frontend-canary-service.yaml
+
+ echo
+ echo "Stable and canary deployments:"
+ kubectl get deployments \
+ -n "${STAGING_NAMESPACE}" \
+ -l app=frontend \
+ -o wide
+
+ - name: Wait for canary rollout
+ id: rollout
+ continue-on-error: true
+ shell: bash
+ run: |
+ kubectl rollout status deployment/"${CANARY_DEPLOYMENT}" \
+ -n "${STAGING_NAMESPACE}" \
+ --timeout=120s
+
+ - name: Validate canary release
+ id: validation
+ if: steps.rollout.outcome == 'success'
+ continue-on-error: true
+ shell: bash
+ run: |
+ echo "Starting automated canary health validation..."
+
+ kubectl run canary-health-check \
+ --rm \
+ -i \
+ --restart=Never \
+ --image=curlimages/curl:8.12.1 \
+ -n "${STAGING_NAMESPACE}" \
+ -- \
+ curl \
+ --fail \
+ --silent \
+ --show-error \
+ --max-time 10 \
+ "http://${CANARY_SERVICE}/" \
+ > /dev/null
+
+ echo "Canary HTTP health validation passed."
+
+ echo "Canary candidate passed all validation checks."
+
+ - name: Promote healthy canary
+ if: >-
+ inputs.scenario == 'healthy' &&
+ steps.rollout.outcome == 'success' &&
+ steps.validation.outcome == 'success'
+ shell: bash
+ run: |
+ echo "Promoting validated canary image to stable deployment..."
+
+ kubectl set image deployment/"${STABLE_DEPLOYMENT}" \
+ frontend="${{ steps.candidate.outputs.image }}" \
+ -n "${STAGING_NAMESPACE}"
+
+ kubectl rollout status deployment/"${STABLE_DEPLOYMENT}" \
+ -n "${STAGING_NAMESPACE}" \
+ --timeout=180s
+
+ echo
+ echo "Canary promotion completed successfully."
+
+ - name: Automatic rollback and recovery
+ if: >-
+ inputs.scenario == 'faulty' ||
+ steps.rollout.outcome == 'failure' ||
+ steps.validation.outcome == 'failure'
+ shell: bash
+ run: |
+ echo "Canary validation failed."
+ echo "Automatic recovery has been triggered."
+ echo
+ echo "Stable deployment was never replaced."
+ echo "Removing failed canary workload..."
+
+ kubectl delete deployment "${CANARY_DEPLOYMENT}" \
+ -n "${STAGING_NAMESPACE}" \
+ --ignore-not-found=true
+
+ kubectl delete service "${CANARY_SERVICE}" \
+ -n "${STAGING_NAMESPACE}" \
+ --ignore-not-found=true
+
+ echo
+ echo "Verifying original stable deployment..."
+
+ kubectl rollout status deployment/"${STABLE_DEPLOYMENT}" \
+ -n "${STAGING_NAMESPACE}" \
+ --timeout=180s
+
+ CURRENT_IMAGE=$(kubectl get deployment "${STABLE_DEPLOYMENT}" \
+ -n "${STAGING_NAMESPACE}" \
+ -o jsonpath='{.spec.template.spec.containers[0].image}')
+
+ echo "Original stable image : ${{ steps.stable.outputs.image }}"
+ echo "Current stable image : ${CURRENT_IMAGE}"
+
+ if [ "${CURRENT_IMAGE}" != "${{ steps.stable.outputs.image }}" ]; then
+ echo "Stable image changed unexpectedly."
+ exit 1
+ fi
+
+ STABLE_IP=$(kubectl get service frontend \
+ -n "${STAGING_NAMESPACE}" \
+ -o jsonpath='{.status.loadBalancer.ingress[0].ip}')
+
+ if [ -z "${STABLE_IP}" ]; then
+ STABLE_IP=$(kubectl get service frontend \
+ -n "${STAGING_NAMESPACE}" \
+ -o jsonpath='{.status.loadBalancer.ingress[0].hostname}')
+ fi
+
+ curl --fail \
+ --silent \
+ --show-error \
+ --retry 10 \
+ --retry-delay 5 \
+ --retry-all-errors \
+ "http://${STABLE_IP}/" \
+ > /dev/null
+
+ echo
+ echo "AUTOMATIC RECOVERY SUCCESSFUL."
+ echo "Failed canary removed."
+ echo "Original stable release remains healthy."
+
+ - name: Remove successful canary
+ if: >-
+ inputs.scenario == 'healthy' &&
+ steps.rollout.outcome == 'success' &&
+ steps.validation.outcome == 'success'
+ shell: bash
+ run: |
+ echo "Removing temporary canary resources after promotion..."
+
+ kubectl delete deployment "${CANARY_DEPLOYMENT}" \
+ -n "${STAGING_NAMESPACE}" \
+ --ignore-not-found=true
+
+ kubectl delete service "${CANARY_SERVICE}" \
+ -n "${STAGING_NAMESPACE}" \
+ --ignore-not-found=true
+
+ - name: Final deployment evidence
+ if: always()
+ shell: bash
+ run: |
+ echo "============================================"
+ echo "FINAL CANARY DEPLOYMENT STATE"
+ echo "============================================"
+
+ kubectl get deployments \
+ -n "${STAGING_NAMESPACE}" \
+ -o wide
+
+ echo
+ kubectl get pods \
+ -n "${STAGING_NAMESPACE}" \
+ -o wide
+
+ echo
+ kubectl get services \
+ -n "${STAGING_NAMESPACE}"
+
+ - name: Final scenario result
+ if: always()
+ shell: bash
+ run: |
+ if [ "${{ inputs.scenario }}" = "healthy" ]; then
+ if [ "${{ steps.rollout.outcome }}" != "success" ] || \
+ [ "${{ steps.validation.outcome }}" != "success" ]; then
+ echo "Healthy canary scenario failed."
+ exit 1
+ fi
+
+ echo "============================================"
+ echo "HEALTHY CANARY SUCCESSFULLY PROMOTED"
+ echo "============================================"
+ else
+ echo "============================================"
+ echo "FAULTY CANARY REJECTED"
+ echo "STABLE RELEASE RETAINED"
+ echo "AUTOMATIC RECOVERY SUCCESSFUL"
+ echo "============================================"
+ fi
diff --git a/.gitignore b/.gitignore
index 4299cd1a..bf811163 100644
--- a/.gitignore
+++ b/.gitignore
@@ -1,199 +1,12 @@
-# Byte-compiled / optimized / DLL files
-__pycache__/
-*.py[cod]
-*$py.class
-
-# C extensions
-*.so
-
-# MacOS
-.DS_Store
-
-# Distribution / packaging
-.Python
-build/
-develop-eggs/
-dist/
-downloads/
-eggs/
-.eggs/
-lib/
-lib64/
-parts/
-sdist/
-var/
-wheels/
-share/python-wheels/
-*.egg-info/
-.installed.cfg
-*.egg
-MANIFEST
-
-# PyInstaller
-# Usually these files are written by a python script from a template
-# before PyInstaller builds the exe, so as to inject date/other infos into it.
-*.manifest
-*.spec
-
-# Installer logs
-pip-log.txt
-pip-delete-this-directory.txt
-
-# Unit test / coverage reports
-htmlcov/
-.tox/
-.nox/
-.coverage
-.coverage.*
-.cache
-nosetests.xml
-coverage.xml
-*.cover
-*.py,cover
-.hypothesis/
-.pytest_cache/
-cover/
-
-# Translations
-*.mo
-*.pot
-
-# Django stuff:
-*.log
-local_settings.py
-db.sqlite3
-db.sqlite3-journal
-
-# Flask stuff:
-instance/
-.webassets-cache
-
-# Scrapy stuff:
-.scrapy
-
-# Sphinx documentation
-docs/_build/
-
-# PyBuilder
-.pybuilder/
-target/
-
-# Jupyter Notebook
-.ipynb_checkpoints
-
-# IPython
-profile_default/
-ipython_config.py
-
-# pyenv
-# For a library or package, you might want to ignore these files since the code is
-# intended to run in multiple environments; otherwise, check them in:
-# .python-version
-
-# pipenv
-# According to pypa/pipenv#598, it is recommended to include Pipfile.lock in version control.
-# However, in case of collaboration, if having platform-specific dependencies or dependencies
-# having no cross-platform support, pipenv may install dependencies that don't work, or not
-# install all needed dependencies.
-#Pipfile.lock
-
-# UV
-# Similar to Pipfile.lock, it is generally recommended to include uv.lock in version control.
-# This is especially recommended for binary packages to ensure reproducibility, and is more
-# commonly ignored for libraries.
-#uv.lock
-
-# poetry
-# Similar to Pipfile.lock, it is generally recommended to include poetry.lock in version control.
-# This is especially recommended for binary packages to ensure reproducibility, and is more
-# commonly ignored for libraries.
-# https://python-poetry.org/docs/basic-usage/#commit-your-poetrylock-file-to-version-control
-#poetry.lock
+terraform/.terraform/
+terraform/*.tfstate
+terraform/*.tfstate.*
+terraform/*.tfplan
+terraform/terraform.tfvars
-# pdm
-# Similar to Pipfile.lock, it is generally recommended to include pdm.lock in version control.
-#pdm.lock
-# pdm stores project-wide configurations in .pdm.toml, but it is recommended to not include it
-# in version control.
-# https://pdm.fming.dev/latest/usage/project/#working-with-version-control
-.pdm.toml
-.pdm-python
-.pdm-build/
+# Node.js dependencies
+node_modules/
-# PEP 582; used by e.g. github.com/David-OConnor/pyflow and github.com/pdm-project/pdm
-__pypackages__/
-
-# Celery stuff
-celerybeat-schedule
-celerybeat.pid
-
-# SageMath parsed files
-*.sage.py
-
-# Environments
-.env
-.venv
-env/
-venv/
-ENV/
-env.bak/
-venv.bak/
-
-# Spyder project settings
-.spyderproject
-.spyproject
-
-# Rope project settings
-.ropeproject
-
-# mkdocs documentation
-/site
-
-# mypy
-.mypy_cache/
-.dmypy.json
-dmypy.json
-
-# Pyre type checker
-.pyre/
-
-# pytype static type analyzer
-.pytype/
-
-# Cython debug symbols
-cython_debug/
-
-# PyCharm
-# JetBrains specific template is maintained in a separate JetBrains.gitignore that can
-# be found at https://github.com/github/gitignore/blob/main/Global/JetBrains.gitignore
-# and can be added to the global gitignore or merged into this file. For a more nuclear
-# option (not recommended) you can uncomment the following to ignore the entire idea folder.
-#.idea/
-
-# Abstra
-# Abstra is an AI-powered process automation framework.
-# Ignore directories containing user credentials, local state, and settings.
-# Learn more at https://abstra.io/docs
-.abstra/
-
-# Visual Studio Code
-# Visual Studio Code specific template is maintained in a separate VisualStudioCode.gitignore
-# that can be found at https://github.com/github/gitignore/blob/main/Global/VisualStudioCode.gitignore
-# and can be added to the global gitignore or merged into this file. However, if you prefer,
-# you could uncomment the following to ignore the enitre vscode folder
-# .vscode/
-
-# Ruff stuff:
-.ruff_cache/
-
-# PyPI configuration file
-.pypirc
-
-# Cursor
-# Cursor is an AI-powered code editor. `.cursorignore` specifies files/directories to
-# exclude from AI features like autocomplete and code analysis. Recommended for sensitive data
-# refer to https://docs.cursor.com/context/ignore-files
-.cursorignore
-.cursorindexingignore
-
-node_modules/
\ No newline at end of file
+# Python generated files
+__pycache__/
+*.py[cod]
diff --git a/course-service/app/main.py b/course-service/app/main.py
index 7c8cbfb2..6d0670ec 100644
--- a/course-service/app/main.py
+++ b/course-service/app/main.py
@@ -2,7 +2,13 @@
import time
from contextlib import asynccontextmanager
-from fastapi import FastAPI
+from fastapi import FastAPI, Request, Response
+from prometheus_client import (
+ CONTENT_TYPE_LATEST,
+ Counter,
+ Histogram,
+ generate_latest,
+)
from sqlalchemy.exc import OperationalError
from app.db import Base, engine
@@ -16,6 +22,20 @@
logger = logging.getLogger(__name__)
+SERVICE_NAME = "course-service"
+
+HTTP_REQUESTS_TOTAL = Counter(
+ "koalatech_http_requests_total",
+ "Total number of HTTP requests.",
+ ["service", "method", "path", "status_code"],
+)
+
+HTTP_REQUEST_DURATION_SECONDS = Histogram(
+ "koalatech_http_request_duration_seconds",
+ "HTTP request duration in seconds.",
+ ["service", "method", "path"],
+)
+
def initialise_database() -> None:
maximum_attempts = 10
@@ -64,6 +84,34 @@ async def lifespan(_: FastAPI):
)
+@app.middleware("http")
+async def prometheus_metrics(
+ request: Request,
+ call_next,
+):
+ start_time = time.perf_counter()
+ response = await call_next(request)
+ duration = time.perf_counter() - start_time
+
+ route = request.scope.get("route")
+ path = getattr(route, "path", request.url.path)
+
+ HTTP_REQUESTS_TOTAL.labels(
+ service=SERVICE_NAME,
+ method=request.method,
+ path=path,
+ status_code=str(response.status_code),
+ ).inc()
+
+ HTTP_REQUEST_DURATION_SECONDS.labels(
+ service=SERVICE_NAME,
+ method=request.method,
+ path=path,
+ ).observe(duration)
+
+ return response
+
+
app.include_router(courses.router)
@@ -80,5 +128,16 @@ def root() -> dict[str, str]:
def health_check() -> dict[str, str]:
return {
"status": "healthy",
- "service": "course-service",
- }
\ No newline at end of file
+ "service": SERVICE_NAME,
+ }
+
+
+@app.get(
+ "/metrics",
+ include_in_schema=False,
+)
+def metrics() -> Response:
+ return Response(
+ content=generate_latest(),
+ media_type=CONTENT_TYPE_LATEST,
+ )
diff --git a/course-service/requirements.txt b/course-service/requirements.txt
index 58a1178e..dbf225c5 100644
--- a/course-service/requirements.txt
+++ b/course-service/requirements.txt
@@ -6,4 +6,5 @@ pydantic==2.11.7
PyJWT==2.10.1
python-dotenv==1.0.1
pytest==8.4.1
-httpx==0.28.1
\ No newline at end of file
+httpx==0.28.1
+prometheus-client==0.23.1
diff --git a/enrollment-service/app/main.py b/enrollment-service/app/main.py
index 868063be..1b33edd3 100644
--- a/enrollment-service/app/main.py
+++ b/enrollment-service/app/main.py
@@ -2,7 +2,13 @@
import time
from contextlib import asynccontextmanager
-from fastapi import FastAPI
+from fastapi import FastAPI, Request, Response
+from prometheus_client import (
+ CONTENT_TYPE_LATEST,
+ Counter,
+ Histogram,
+ generate_latest,
+)
from sqlalchemy.exc import OperationalError
from app.db import Base, engine
@@ -16,6 +22,20 @@
logger = logging.getLogger(__name__)
+SERVICE_NAME = "enrollment-service"
+
+HTTP_REQUESTS_TOTAL = Counter(
+ "koalatech_http_requests_total",
+ "Total number of HTTP requests.",
+ ["service", "method", "path", "status_code"],
+)
+
+HTTP_REQUEST_DURATION_SECONDS = Histogram(
+ "koalatech_http_request_duration_seconds",
+ "HTTP request duration in seconds.",
+ ["service", "method", "path"],
+)
+
def initialise_database() -> None:
maximum_attempts = 10
@@ -64,6 +84,34 @@ async def lifespan(_: FastAPI):
)
+@app.middleware("http")
+async def prometheus_metrics(
+ request: Request,
+ call_next,
+):
+ start_time = time.perf_counter()
+ response = await call_next(request)
+ duration = time.perf_counter() - start_time
+
+ route = request.scope.get("route")
+ path = getattr(route, "path", request.url.path)
+
+ HTTP_REQUESTS_TOTAL.labels(
+ service=SERVICE_NAME,
+ method=request.method,
+ path=path,
+ status_code=str(response.status_code),
+ ).inc()
+
+ HTTP_REQUEST_DURATION_SECONDS.labels(
+ service=SERVICE_NAME,
+ method=request.method,
+ path=path,
+ ).observe(duration)
+
+ return response
+
+
app.include_router(enrollments.router)
@@ -80,5 +128,16 @@ def root() -> dict[str, str]:
def health_check() -> dict[str, str]:
return {
"status": "healthy",
- "service": "enrollment-service",
- }
\ No newline at end of file
+ "service": SERVICE_NAME,
+ }
+
+
+@app.get(
+ "/metrics",
+ include_in_schema=False,
+)
+def metrics() -> Response:
+ return Response(
+ content=generate_latest(),
+ media_type=CONTENT_TYPE_LATEST,
+ )
diff --git a/enrollment-service/requirements.txt b/enrollment-service/requirements.txt
index 58a1178e..dbf225c5 100644
--- a/enrollment-service/requirements.txt
+++ b/enrollment-service/requirements.txt
@@ -6,4 +6,5 @@ pydantic==2.11.7
PyJWT==2.10.1
python-dotenv==1.0.1
pytest==8.4.1
-httpx==0.28.1
\ No newline at end of file
+httpx==0.28.1
+prometheus-client==0.23.1
diff --git a/frontend/src/pages/Login.jsx b/frontend/src/pages/Login.jsx
index 1132741c..2918a615 100644
--- a/frontend/src/pages/Login.jsx
+++ b/frontend/src/pages/Login.jsx
@@ -98,11 +98,17 @@ const Login = () => {
Sign in to continue
+
+ Successfully deployed automatically through GitHub Actions
+
+
{error && (
{
)
).toBeInTheDocument();
+ expect(
+ screen.getByText(
+ "Successfully deployed automatically through GitHub Actions"
+ )
+ ).toBeInTheDocument();
+
expect(
screen.getByRole("textbox", {
name: /username/i,
@@ -53,7 +59,6 @@ describe("Login page", () => {
expect(
screen.getByLabelText(/password/i)
).toBeInTheDocument();
-
expect(
screen.getByRole("button", {
name: /login/i,
diff --git a/kubernetes/canary/frontend-canary-deployment.yaml b/kubernetes/canary/frontend-canary-deployment.yaml
new file mode 100644
index 00000000..d050c19c
--- /dev/null
+++ b/kubernetes/canary/frontend-canary-deployment.yaml
@@ -0,0 +1,42 @@
+apiVersion: apps/v1
+kind: Deployment
+metadata:
+ name: frontend-canary
+ namespace: staging
+ labels:
+ app: frontend
+ track: canary
+spec:
+ replicas: 1
+ selector:
+ matchLabels:
+ app: frontend
+ track: canary
+ template:
+ metadata:
+ labels:
+ app: frontend
+ track: canary
+ spec:
+ containers:
+ - name: frontend
+ image: CANARY_IMAGE
+ imagePullPolicy: Always
+ ports:
+ - containerPort: 80
+ readinessProbe:
+ httpGet:
+ path: /
+ port: 80
+ initialDelaySeconds: 5
+ periodSeconds: 5
+ timeoutSeconds: 3
+ failureThreshold: 6
+ livenessProbe:
+ httpGet:
+ path: /
+ port: 80
+ initialDelaySeconds: 10
+ periodSeconds: 10
+ timeoutSeconds: 3
+ failureThreshold: 3
diff --git a/kubernetes/canary/frontend-canary-service.yaml b/kubernetes/canary/frontend-canary-service.yaml
new file mode 100644
index 00000000..3b9237fd
--- /dev/null
+++ b/kubernetes/canary/frontend-canary-service.yaml
@@ -0,0 +1,17 @@
+apiVersion: v1
+kind: Service
+metadata:
+ name: frontend-canary
+ namespace: staging
+ labels:
+ app: frontend
+ track: canary
+spec:
+ type: ClusterIP
+ selector:
+ app: frontend
+ track: canary
+ ports:
+ - name: http
+ port: 80
+ targetPort: 80
diff --git a/kubernetes/monitoring/01-namespace.yaml b/kubernetes/monitoring/01-namespace.yaml
new file mode 100644
index 00000000..d3252360
--- /dev/null
+++ b/kubernetes/monitoring/01-namespace.yaml
@@ -0,0 +1,4 @@
+apiVersion: v1
+kind: Namespace
+metadata:
+ name: monitoring
diff --git a/kubernetes/monitoring/02-prometheus-rbac.yaml b/kubernetes/monitoring/02-prometheus-rbac.yaml
new file mode 100644
index 00000000..bb268018
--- /dev/null
+++ b/kubernetes/monitoring/02-prometheus-rbac.yaml
@@ -0,0 +1,46 @@
+apiVersion: v1
+kind: ServiceAccount
+metadata:
+ name: prometheus
+ namespace: monitoring
+
+---
+apiVersion: rbac.authorization.k8s.io/v1
+kind: ClusterRole
+metadata:
+ name: prometheus
+rules:
+ - apiGroups: [""]
+ resources:
+ - nodes
+ - nodes/proxy
+ - services
+ - endpoints
+ - pods
+ verbs:
+ - get
+ - list
+ - watch
+ - apiGroups:
+ - extensions
+ - networking.k8s.io
+ resources:
+ - ingresses
+ verbs:
+ - get
+ - list
+ - watch
+
+---
+apiVersion: rbac.authorization.k8s.io/v1
+kind: ClusterRoleBinding
+metadata:
+ name: prometheus
+roleRef:
+ apiGroup: rbac.authorization.k8s.io
+ kind: ClusterRole
+ name: prometheus
+subjects:
+ - kind: ServiceAccount
+ name: prometheus
+ namespace: monitoring
diff --git a/kubernetes/monitoring/03-prometheus-config.yaml b/kubernetes/monitoring/03-prometheus-config.yaml
new file mode 100644
index 00000000..d40f0350
--- /dev/null
+++ b/kubernetes/monitoring/03-prometheus-config.yaml
@@ -0,0 +1,52 @@
+apiVersion: v1
+kind: ConfigMap
+metadata:
+ name: prometheus-config
+ namespace: monitoring
+data:
+ prometheus.yml: |
+ global:
+ scrape_interval: 15s
+ evaluation_interval: 15s
+
+ scrape_configs:
+
+ - job_name: prometheus
+ static_configs:
+ - targets:
+ - localhost:9090
+
+ - job_name: user-service
+ metrics_path: /metrics
+ static_configs:
+ - targets:
+ - user-service.staging.svc.cluster.local:8000
+
+ - job_name: student-service
+ metrics_path: /metrics
+ static_configs:
+ - targets:
+ - student-service.staging.svc.cluster.local:8000
+
+ - job_name: lecturer-service
+ metrics_path: /metrics
+ static_configs:
+ - targets:
+ - lecturer-service.staging.svc.cluster.local:8000
+
+ - job_name: course-service
+ metrics_path: /metrics
+ static_configs:
+ - targets:
+ - course-service.staging.svc.cluster.local:8000
+
+ - job_name: enrollment-service
+ metrics_path: /metrics
+ static_configs:
+ - targets:
+ - enrollment-service.staging.svc.cluster.local:8000
+
+ - job_name: kube-state-metrics
+ static_configs:
+ - targets:
+ - kube-state-metrics.monitoring.svc.cluster.local:8080
diff --git a/kubernetes/monitoring/04-prometheus.yaml b/kubernetes/monitoring/04-prometheus.yaml
new file mode 100644
index 00000000..69bed06a
--- /dev/null
+++ b/kubernetes/monitoring/04-prometheus.yaml
@@ -0,0 +1,56 @@
+apiVersion: apps/v1
+kind: Deployment
+metadata:
+ name: prometheus
+ namespace: monitoring
+spec:
+ replicas: 1
+ selector:
+ matchLabels:
+ app: prometheus
+ template:
+ metadata:
+ labels:
+ app: prometheus
+ spec:
+ serviceAccountName: prometheus
+ containers:
+ - name: prometheus
+ image: prom/prometheus:v3.5.0
+ imagePullPolicy: IfNotPresent
+ args:
+ - --config.file=/etc/prometheus/prometheus.yml
+ - --storage.tsdb.path=/prometheus
+ - --storage.tsdb.retention.time=6h
+ ports:
+ - name: http
+ containerPort: 9090
+ resources:
+ requests:
+ cpu: 50m
+ memory: 128Mi
+ limits:
+ cpu: 250m
+ memory: 384Mi
+ volumeMounts:
+ - name: prometheus-config
+ mountPath: /etc/prometheus
+ volumes:
+ - name: prometheus-config
+ configMap:
+ name: prometheus-config
+
+---
+apiVersion: v1
+kind: Service
+metadata:
+ name: prometheus
+ namespace: monitoring
+spec:
+ selector:
+ app: prometheus
+ ports:
+ - name: http
+ port: 9090
+ targetPort: 9090
+ type: ClusterIP
diff --git a/kubernetes/monitoring/05-kube-state-metrics.yaml b/kubernetes/monitoring/05-kube-state-metrics.yaml
new file mode 100644
index 00000000..d7e4e15d
--- /dev/null
+++ b/kubernetes/monitoring/05-kube-state-metrics.yaml
@@ -0,0 +1,108 @@
+apiVersion: v1
+kind: ServiceAccount
+metadata:
+ name: kube-state-metrics
+ namespace: monitoring
+
+---
+apiVersion: rbac.authorization.k8s.io/v1
+kind: ClusterRole
+metadata:
+ name: kube-state-metrics
+rules:
+ - apiGroups: [""]
+ resources:
+ - configmaps
+ - secrets
+ - nodes
+ - pods
+ - services
+ - resourcequotas
+ - replicationcontrollers
+ - limitranges
+ - persistentvolumeclaims
+ - persistentvolumes
+ - namespaces
+ - endpoints
+ verbs:
+ - list
+ - watch
+ - apiGroups:
+ - apps
+ resources:
+ - statefulsets
+ - daemonsets
+ - deployments
+ - replicasets
+ verbs:
+ - list
+ - watch
+ - apiGroups:
+ - batch
+ resources:
+ - cronjobs
+ - jobs
+ verbs:
+ - list
+ - watch
+
+---
+apiVersion: rbac.authorization.k8s.io/v1
+kind: ClusterRoleBinding
+metadata:
+ name: kube-state-metrics
+roleRef:
+ apiGroup: rbac.authorization.k8s.io
+ kind: ClusterRole
+ name: kube-state-metrics
+subjects:
+ - kind: ServiceAccount
+ name: kube-state-metrics
+ namespace: monitoring
+
+---
+apiVersion: apps/v1
+kind: Deployment
+metadata:
+ name: kube-state-metrics
+ namespace: monitoring
+spec:
+ replicas: 1
+ selector:
+ matchLabels:
+ app: kube-state-metrics
+ template:
+ metadata:
+ labels:
+ app: kube-state-metrics
+ spec:
+ serviceAccountName: kube-state-metrics
+ containers:
+ - name: kube-state-metrics
+ image: registry.k8s.io/kube-state-metrics/kube-state-metrics:v2.17.0
+ imagePullPolicy: IfNotPresent
+ ports:
+ - name: http
+ containerPort: 8080
+ resources:
+ requests:
+ cpu: 20m
+ memory: 32Mi
+ limits:
+ cpu: 100m
+ memory: 128Mi
+
+---
+apiVersion: v1
+kind: Service
+metadata:
+ name: kube-state-metrics
+ namespace: monitoring
+spec:
+ selector:
+ app: kube-state-metrics
+ ports:
+ - name: http
+ port: 8080
+ targetPort: 8080
+ type: ClusterIP
diff --git a/kubernetes/monitoring/06-grafana-datasource.yaml b/kubernetes/monitoring/06-grafana-datasource.yaml
new file mode 100644
index 00000000..3aa1e3be
--- /dev/null
+++ b/kubernetes/monitoring/06-grafana-datasource.yaml
@@ -0,0 +1,17 @@
+apiVersion: v1
+kind: ConfigMap
+metadata:
+ name: grafana-datasource
+ namespace: monitoring
+data:
+ datasource.yaml: |
+ apiVersion: 1
+
+ datasources:
+ - name: Prometheus
+ uid: prometheus
+ type: prometheus
+ access: proxy
+ url: http://prometheus.monitoring.svc.cluster.local:9090
+ isDefault: true
+ editable: false
diff --git a/kubernetes/monitoring/07-grafana-dashboard.yaml b/kubernetes/monitoring/07-grafana-dashboard.yaml
new file mode 100644
index 00000000..7ede90f3
--- /dev/null
+++ b/kubernetes/monitoring/07-grafana-dashboard.yaml
@@ -0,0 +1,157 @@
+apiVersion: v1
+kind: ConfigMap
+metadata:
+ name: grafana-dashboard
+ namespace: monitoring
+data:
+ koalatech-monitoring.json: |
+ {
+ "annotations": {
+ "list": []
+ },
+ "editable": true,
+ "panels": [
+ {
+ "type": "stat",
+ "title": "HTTP Requests by Backend Service",
+ "datasource": {
+ "type": "prometheus",
+ "uid": "prometheus"
+ },
+ "targets": [
+ {
+ "expr": "sum by (service) (koalatech_http_requests_total)",
+ "legendFormat": "{{service}}",
+ "refId": "A"
+ }
+ ],
+ "gridPos": {
+ "h": 8,
+ "w": 12,
+ "x": 0,
+ "y": 0
+ }
+ },
+ {
+ "type": "timeseries",
+ "title": "Backend Request Rate",
+ "datasource": {
+ "type": "prometheus",
+ "uid": "prometheus"
+ },
+ "targets": [
+ {
+ "expr": "sum by (service) (rate(koalatech_http_requests_total[5m]))",
+ "legendFormat": "{{service}}",
+ "refId": "A"
+ }
+ ],
+ "gridPos": {
+ "h": 8,
+ "w": 12,
+ "x": 12,
+ "y": 0
+ }
+ },
+ {
+ "type": "timeseries",
+ "title": "Backend P95 Request Duration",
+ "datasource": {
+ "type": "prometheus",
+ "uid": "prometheus"
+ },
+ "targets": [
+ {
+ "expr": "histogram_quantile(0.95, sum by (le, service) (rate(koalatech_http_request_duration_seconds_bucket[5m])))",
+ "legendFormat": "{{service}}",
+ "refId": "A"
+ }
+ ],
+ "gridPos": {
+ "h": 8,
+ "w": 12,
+ "x": 0,
+ "y": 8
+ }
+ },
+ {
+ "type": "stat",
+ "title": "Running Staging Pods",
+ "datasource": {
+ "type": "prometheus",
+ "uid": "prometheus"
+ },
+ "targets": [
+ {
+ "expr": "sum(kube_pod_status_phase{namespace=\"staging\",phase=\"Running\"})",
+ "refId": "A"
+ }
+ ],
+ "gridPos": {
+ "h": 8,
+ "w": 6,
+ "x": 12,
+ "y": 8
+ }
+ },
+ {
+ "type": "stat",
+ "title": "Available Staging Deployment Replicas",
+ "datasource": {
+ "type": "prometheus",
+ "uid": "prometheus"
+ },
+ "targets": [
+ {
+ "expr": "sum(kube_deployment_status_replicas_available{namespace=\"staging\"})",
+ "refId": "A"
+ }
+ ],
+ "gridPos": {
+ "h": 8,
+ "w": 6,
+ "x": 18,
+ "y": 8
+ }
+ },
+ {
+ "type": "timeseries",
+ "title": "Staging Container Restarts",
+ "datasource": {
+ "type": "prometheus",
+ "uid": "prometheus"
+ },
+ "targets": [
+ {
+ "expr": "sum by (pod) (kube_pod_container_status_restarts_total{namespace=\"staging\"})",
+ "legendFormat": "{{pod}}",
+ "refId": "A"
+ }
+ ],
+ "gridPos": {
+ "h": 8,
+ "w": 24,
+ "x": 0,
+ "y": 16
+ }
+ }
+ ],
+ "refresh": "10s",
+ "schemaVersion": 41,
+ "tags": [
+ "SIT722",
+ "KoalaTech",
+ "10.2D"
+ ],
+ "templating": {
+ "list": []
+ },
+ "time": {
+ "from": "now-15m",
+ "to": "now"
+ },
+ "timezone": "browser",
+ "title": "KoalaTech - SIT722 Task 10.2D",
+ "uid": "koalatech-10-2d",
+ "version": 1
+ }
diff --git a/kubernetes/monitoring/08-grafana-dashboard-provider.yaml b/kubernetes/monitoring/08-grafana-dashboard-provider.yaml
new file mode 100644
index 00000000..1fc6dd16
--- /dev/null
+++ b/kubernetes/monitoring/08-grafana-dashboard-provider.yaml
@@ -0,0 +1,19 @@
+apiVersion: v1
+kind: ConfigMap
+metadata:
+ name: grafana-dashboard-provider
+ namespace: monitoring
+data:
+ dashboard.yaml: |
+ apiVersion: 1
+
+ providers:
+ - name: KoalaTech
+ orgId: 1
+ folder: SIT722
+ type: file
+ disableDeletion: false
+ updateIntervalSeconds: 10
+ allowUiUpdates: true
+ options:
+ path: /var/lib/grafana/dashboards
diff --git a/kubernetes/monitoring/09-grafana.yaml b/kubernetes/monitoring/09-grafana.yaml
new file mode 100644
index 00000000..ad604e60
--- /dev/null
+++ b/kubernetes/monitoring/09-grafana.yaml
@@ -0,0 +1,71 @@
+apiVersion: apps/v1
+kind: Deployment
+metadata:
+ name: grafana
+ namespace: monitoring
+spec:
+ replicas: 1
+ selector:
+ matchLabels:
+ app: grafana
+ template:
+ metadata:
+ labels:
+ app: grafana
+ spec:
+ containers:
+ - name: grafana
+ image: grafana/grafana:12.1.1
+ imagePullPolicy: IfNotPresent
+ ports:
+ - name: http
+ containerPort: 3000
+ env:
+ - name: GF_SECURITY_ADMIN_USER
+ value: admin
+ - name: GF_SECURITY_ADMIN_PASSWORD
+ valueFrom:
+ secretKeyRef:
+ name: grafana-admin
+ key: password
+ - name: GF_USERS_ALLOW_SIGN_UP
+ value: "false"
+ resources:
+ requests:
+ cpu: 50m
+ memory: 128Mi
+ limits:
+ cpu: 250m
+ memory: 256Mi
+ volumeMounts:
+ - name: datasource
+ mountPath: /etc/grafana/provisioning/datasources
+ - name: dashboard-provider
+ mountPath: /etc/grafana/provisioning/dashboards
+ - name: dashboards
+ mountPath: /var/lib/grafana/dashboards
+ volumes:
+ - name: datasource
+ configMap:
+ name: grafana-datasource
+ - name: dashboard-provider
+ configMap:
+ name: grafana-dashboard-provider
+ - name: dashboards
+ configMap:
+ name: grafana-dashboard
+
+---
+apiVersion: v1
+kind: Service
+metadata:
+ name: grafana
+ namespace: monitoring
+spec:
+ selector:
+ app: grafana
+ ports:
+ - name: http
+ port: 3000
+ targetPort: 3000
+ type: ClusterIP
diff --git a/kubernetes/staging/06-enrollment-db.yaml b/kubernetes/staging/06-enrollment-db.yaml
index 3a94e0ee..557982be 100644
--- a/kubernetes/staging/06-enrollment-db.yaml
+++ b/kubernetes/staging/06-enrollment-db.yaml
@@ -1,16 +1,3 @@
-apiVersion: v1
-kind: PersistentVolumeClaim
-metadata:
- name: enrollment-db-pvc
- namespace: staging
-spec:
- accessModes:
- - ReadWriteOnce
- resources:
- requests:
- storage: 1Gi
-
----
apiVersion: apps/v1
kind: Deployment
metadata:
@@ -51,8 +38,7 @@ spec:
mountPath: /var/lib/postgresql/data
volumes:
- name: enrollment-db-storage
- persistentVolumeClaim:
- claimName: enrollment-db-pvc
+ emptyDir: {}
---
apiVersion: v1
@@ -65,4 +51,4 @@ spec:
app: enrollment-db
ports:
- port: 5432
- targetPort: 5432
\ No newline at end of file
+ targetPort: 5432
diff --git a/kubernetes/staging/07-user-service.yaml b/kubernetes/staging/07-user-service.yaml
index 7984533d..0b9ff098 100644
--- a/kubernetes/staging/07-user-service.yaml
+++ b/kubernetes/staging/07-user-service.yaml
@@ -15,7 +15,7 @@ spec:
spec:
containers:
- name: user-service
- image: placeholder
+ image: sit722week08acr224848845.azurecr.io/koalatech-user-service:latest
imagePullPolicy: Always
ports:
- containerPort: 8000
diff --git a/kubernetes/staging/08-student-service.yaml b/kubernetes/staging/08-student-service.yaml
index 244ed083..dbb6716c 100644
--- a/kubernetes/staging/08-student-service.yaml
+++ b/kubernetes/staging/08-student-service.yaml
@@ -15,7 +15,7 @@ spec:
spec:
containers:
- name: student-service
- image: placeholder
+ image: sit722week08acr224848845.azurecr.io/koalatech-student-service:latest
imagePullPolicy: Always
ports:
- containerPort: 8000
diff --git a/kubernetes/staging/09-lecturer-service.yaml b/kubernetes/staging/09-lecturer-service.yaml
index 48a611f6..48719a81 100644
--- a/kubernetes/staging/09-lecturer-service.yaml
+++ b/kubernetes/staging/09-lecturer-service.yaml
@@ -15,7 +15,7 @@ spec:
spec:
containers:
- name: lecturer-service
- image: placeholder
+ image: sit722week08acr224848845.azurecr.io/koalatech-lecturer-service:latest
imagePullPolicy: Always
ports:
- containerPort: 8000
diff --git a/kubernetes/staging/10-course-service.yaml b/kubernetes/staging/10-course-service.yaml
index 652cd06b..f0b1dbee 100644
--- a/kubernetes/staging/10-course-service.yaml
+++ b/kubernetes/staging/10-course-service.yaml
@@ -15,7 +15,7 @@ spec:
spec:
containers:
- name: course-service
- image: placeholder
+ image: sit722week08acr224848845.azurecr.io/koalatech-course-service:latest
imagePullPolicy: Always
ports:
- containerPort: 8000
diff --git a/kubernetes/staging/11-enrollment-service.yaml b/kubernetes/staging/11-enrollment-service.yaml
index 87a21d4f..275a286c 100644
--- a/kubernetes/staging/11-enrollment-service.yaml
+++ b/kubernetes/staging/11-enrollment-service.yaml
@@ -15,7 +15,7 @@ spec:
spec:
containers:
- name: enrollment-service
- image: placeholder
+ image: sit722week08acr224848845.azurecr.io/koalatech-enrollment-service:latest
imagePullPolicy: Always
ports:
- containerPort: 8000
diff --git a/kubernetes/staging/12-frontend.yaml b/kubernetes/staging/12-frontend.yaml
index 91a438da..fd8f353e 100644
--- a/kubernetes/staging/12-frontend.yaml
+++ b/kubernetes/staging/12-frontend.yaml
@@ -15,7 +15,7 @@ spec:
spec:
containers:
- name: frontend
- image: placeholder
+ image: sit722week08acr224848845.azurecr.io/koalatech-frontend:latest
imagePullPolicy: Always
ports:
- containerPort: 80
diff --git a/lecturer-service/app/main.py b/lecturer-service/app/main.py
index 62972d48..028676e8 100644
--- a/lecturer-service/app/main.py
+++ b/lecturer-service/app/main.py
@@ -2,7 +2,13 @@
import time
from contextlib import asynccontextmanager
-from fastapi import FastAPI
+from fastapi import FastAPI, Request, Response
+from prometheus_client import (
+ CONTENT_TYPE_LATEST,
+ Counter,
+ Histogram,
+ generate_latest,
+)
from sqlalchemy.exc import OperationalError
from app.db import Base, engine
@@ -17,6 +23,20 @@
logger = logging.getLogger(__name__)
+SERVICE_NAME = "lecturer-service"
+
+HTTP_REQUESTS_TOTAL = Counter(
+ "koalatech_http_requests_total",
+ "Total number of HTTP requests.",
+ ["service", "method", "path", "status_code"],
+)
+
+HTTP_REQUEST_DURATION_SECONDS = Histogram(
+ "koalatech_http_request_duration_seconds",
+ "HTTP request duration in seconds.",
+ ["service", "method", "path"],
+)
+
def initialise_database() -> None:
maximum_attempts = 10
@@ -78,6 +98,34 @@ async def lifespan(_: FastAPI):
)
+@app.middleware("http")
+async def prometheus_metrics(
+ request: Request,
+ call_next,
+):
+ start_time = time.perf_counter()
+ response = await call_next(request)
+ duration = time.perf_counter() - start_time
+
+ route = request.scope.get("route")
+ path = getattr(route, "path", request.url.path)
+
+ HTTP_REQUESTS_TOTAL.labels(
+ service=SERVICE_NAME,
+ method=request.method,
+ path=path,
+ status_code=str(response.status_code),
+ ).inc()
+
+ HTTP_REQUEST_DURATION_SECONDS.labels(
+ service=SERVICE_NAME,
+ method=request.method,
+ path=path,
+ ).observe(duration)
+
+ return response
+
+
app.include_router(lecturers.router)
@@ -94,5 +142,16 @@ def root() -> dict[str, str]:
def health_check() -> dict[str, str]:
return {
"status": "healthy",
- "service": "lecturer-service",
- }
\ No newline at end of file
+ "service": SERVICE_NAME,
+ }
+
+
+@app.get(
+ "/metrics",
+ include_in_schema=False,
+)
+def metrics() -> Response:
+ return Response(
+ content=generate_latest(),
+ media_type=CONTENT_TYPE_LATEST,
+ )
diff --git a/lecturer-service/requirements.txt b/lecturer-service/requirements.txt
index feeedd2d..72befaf6 100644
--- a/lecturer-service/requirements.txt
+++ b/lecturer-service/requirements.txt
@@ -8,4 +8,5 @@ python-multipart==0.0.20
python-dotenv==1.0.1
azure-storage-blob==12.26.0
pytest==8.4.1
-httpx==0.28.1
\ No newline at end of file
+httpx==0.28.1
+prometheus-client==0.23.1
diff --git a/student-service/app/main.py b/student-service/app/main.py
index 0d85f2ad..3a29e8a4 100644
--- a/student-service/app/main.py
+++ b/student-service/app/main.py
@@ -2,7 +2,13 @@
import time
from contextlib import asynccontextmanager
-from fastapi import FastAPI
+from fastapi import FastAPI, Request, Response
+from prometheus_client import (
+ CONTENT_TYPE_LATEST,
+ Counter,
+ Histogram,
+ generate_latest,
+)
from sqlalchemy.exc import OperationalError
from app.db import Base, engine
@@ -17,6 +23,20 @@
logger = logging.getLogger(__name__)
+SERVICE_NAME = "student-service"
+
+HTTP_REQUESTS_TOTAL = Counter(
+ "koalatech_http_requests_total",
+ "Total number of HTTP requests.",
+ ["service", "method", "path", "status_code"],
+)
+
+HTTP_REQUEST_DURATION_SECONDS = Histogram(
+ "koalatech_http_request_duration_seconds",
+ "HTTP request duration in seconds.",
+ ["service", "method", "path"],
+)
+
def initialise_database() -> None:
maximum_attempts = 10
@@ -79,13 +99,38 @@ async def lifespan(_: FastAPI):
)
+@app.middleware("http")
+async def prometheus_metrics(
+ request: Request,
+ call_next,
+):
+ start_time = time.perf_counter()
+ response = await call_next(request)
+ duration = time.perf_counter() - start_time
+
+ route = request.scope.get("route")
+ path = getattr(route, "path", request.url.path)
+
+ HTTP_REQUESTS_TOTAL.labels(
+ service=SERVICE_NAME,
+ method=request.method,
+ path=path,
+ status_code=str(response.status_code),
+ ).inc()
+
+ HTTP_REQUEST_DURATION_SECONDS.labels(
+ service=SERVICE_NAME,
+ method=request.method,
+ path=path,
+ ).observe(duration)
+
+ return response
+
+
app.include_router(students.router)
-@app.get(
- "/",
- tags=["Health"],
-)
+@app.get("/", tags=["Health"])
def root() -> dict[str, str]:
return {
"message": (
@@ -94,12 +139,20 @@ def root() -> dict[str, str]:
}
-@app.get(
- "/health",
- tags=["Health"],
-)
+@app.get("/health", tags=["Health"])
def health_check() -> dict[str, str]:
return {
"status": "healthy",
- "service": "student-service",
- }
\ No newline at end of file
+ "service": SERVICE_NAME,
+ }
+
+
+@app.get(
+ "/metrics",
+ include_in_schema=False,
+)
+def metrics() -> Response:
+ return Response(
+ content=generate_latest(),
+ media_type=CONTENT_TYPE_LATEST,
+ )
diff --git a/student-service/requirements.txt b/student-service/requirements.txt
index feeedd2d..72befaf6 100644
--- a/student-service/requirements.txt
+++ b/student-service/requirements.txt
@@ -8,4 +8,5 @@ python-multipart==0.0.20
python-dotenv==1.0.1
azure-storage-blob==12.26.0
pytest==8.4.1
-httpx==0.28.1
\ No newline at end of file
+httpx==0.28.1
+prometheus-client==0.23.1
diff --git a/terraform/.terraform.lock.hcl b/terraform/.terraform.lock.hcl
new file mode 100644
index 00000000..1c302601
--- /dev/null
+++ b/terraform/.terraform.lock.hcl
@@ -0,0 +1,22 @@
+# This file is maintained automatically by "terraform init".
+# Manual edits may be lost in future updates.
+
+provider "registry.terraform.io/hashicorp/azurerm" {
+ version = "4.81.0"
+ constraints = "~> 4.0"
+ hashes = [
+ "h1:XhToZua4gtih1Kv8RdStcfND83G4Tmb6GZFT4jEUhDU=",
+ "zh:0732e7b74264ddfa2b90ba69d01c283d3cbae9f72ed3e506c6ac92529fed7fd3",
+ "zh:12afb524e232fe4e3d6161927724af5dfa4831d71edd9c174917ca9b7377bfae",
+ "zh:169d619ae202c4145e02fb706fb7c3679445ab3e3ff722edbf89597517a8c92e",
+ "zh:6beb95a3ef2f2d9c76abaa48e5450e90686a3fb6a47f1cb0ff7c5e94b6960151",
+ "zh:705e075fb5ffc4bf66fd7cbabf1a65007a41621e80030a2c158a4c83b6046216",
+ "zh:78d5eefdd9e494defcb3c68d282b8f96630502cac21d1ea161f53cfe9bb483b3",
+ "zh:79a8d17fefe647040fcb9ee8821a4f09f395427c4fd49493489b9a93a9a1038e",
+ "zh:8cc3f900b3774c0ae37ae42365c4579a199cf9e5edf88e476fdf5ab1048f84ea",
+ "zh:dec373b9390fa95e257291acd018ed65a7d512b428645d35e22cdbe8b245a08b",
+ "zh:e60f1e9fb45df6defade2855ed6e68547409ea75d30655c556adb0c08579749b",
+ "zh:f901d12ec82f3f8b5880a27b5cbcd7bd0d97e60c9367a2d7ed82fdd1157b39ff",
+ "zh:facf68ea5bf0f2b8ba720e7fba5f86492e1d4c591100460bb91c3f79f391f4b6",
+ ]
+}
diff --git a/terraform/backend.tf b/terraform/backend.tf
new file mode 100644
index 00000000..6602f206
--- /dev/null
+++ b/terraform/backend.tf
@@ -0,0 +1,3 @@
+terraform {
+ backend "azurerm" {}
+}
diff --git a/terraform/container_registry.tf b/terraform/container_registry.tf
new file mode 100644
index 00000000..36266676
--- /dev/null
+++ b/terraform/container_registry.tf
@@ -0,0 +1,15 @@
+resource "azurerm_container_registry" "acr" {
+ name = var.acr_name
+ resource_group_name = azurerm_resource_group.rg.name
+ location = azurerm_resource_group.rg.location
+
+ sku = "Basic"
+ admin_enabled = false
+
+ tags = {
+ project = "SIT722"
+ task = "10.2D"
+ environment = "week10"
+ purpose = "infrastructure-security-monitoring"
+ }
+}
diff --git a/terraform/kubernetes_service.tf b/terraform/kubernetes_service.tf
new file mode 100644
index 00000000..ab5e73e4
--- /dev/null
+++ b/terraform/kubernetes_service.tf
@@ -0,0 +1,34 @@
+resource "azurerm_kubernetes_cluster" "aks" {
+ name = var.aks_name
+ location = azurerm_resource_group.rg.location
+ resource_group_name = azurerm_resource_group.rg.name
+
+ dns_prefix = var.aks_name
+
+ default_node_pool {
+ name = "default"
+ node_count = var.aks_node_count
+ vm_size = var.aks_vm_size
+ }
+
+ identity {
+ type = "SystemAssigned"
+ }
+
+ role_based_access_control_enabled = true
+
+ tags = {
+ project = "SIT722"
+ task = "10.2D"
+ environment = "week10"
+ purpose = "infrastructure-security-monitoring"
+ }
+}
+
+# Allow AKS to pull container images from ACR.
+resource "azurerm_role_assignment" "acr_pull" {
+ principal_id = azurerm_kubernetes_cluster.aks.kubelet_identity[0].object_id
+ role_definition_name = "AcrPull"
+ scope = azurerm_container_registry.acr.id
+ skip_service_principal_aad_check = true
+}
diff --git a/terraform/outputs.tf b/terraform/outputs.tf
new file mode 100644
index 00000000..283cdcb2
--- /dev/null
+++ b/terraform/outputs.tf
@@ -0,0 +1,30 @@
+output "resource_group_name" {
+ description = "Azure Resource Group name"
+ value = azurerm_resource_group.rg.name
+}
+
+output "acr_name" {
+ description = "Azure Container Registry name"
+ value = azurerm_container_registry.acr.name
+}
+
+output "acr_login_server" {
+ description = "Azure Container Registry login server"
+ value = azurerm_container_registry.acr.login_server
+}
+
+output "aks_cluster_name" {
+ description = "AKS cluster name"
+ value = azurerm_kubernetes_cluster.aks.name
+}
+
+output "storage_account_name" {
+ description = "Azure Storage Account name"
+ value = azurerm_storage_account.storage.name
+}
+
+output "storage_connection_string" {
+ description = "Azure Storage Account connection string"
+ value = azurerm_storage_account.storage.primary_connection_string
+ sensitive = true
+}
\ No newline at end of file
diff --git a/terraform/resource_group.tf b/terraform/resource_group.tf
new file mode 100644
index 00000000..75d42f14
--- /dev/null
+++ b/terraform/resource_group.tf
@@ -0,0 +1,11 @@
+resource "azurerm_resource_group" "rg" {
+ name = var.resource_group_name
+ location = var.location
+
+ tags = {
+ project = "SIT722"
+ task = "10.2D"
+ environment = "week10"
+ purpose = "infrastructure-security-monitoring"
+ }
+}
diff --git a/terraform/storage_account.tf b/terraform/storage_account.tf
new file mode 100644
index 00000000..e9d9eaa8
--- /dev/null
+++ b/terraform/storage_account.tf
@@ -0,0 +1,17 @@
+resource "azurerm_storage_account" "storage" {
+ name = var.storage_account_name
+ resource_group_name = azurerm_resource_group.rg.name
+ location = azurerm_resource_group.rg.location
+
+ account_tier = "Standard"
+ account_replication_type = "LRS"
+
+ min_tls_version = "TLS1_2"
+
+ tags = {
+ project = "SIT722"
+ task = "10.2D"
+ environment = "week10"
+ purpose = "infrastructure-security-monitoring"
+ }
+}
diff --git a/terraform/variables.tf b/terraform/variables.tf
new file mode 100644
index 00000000..42291769
--- /dev/null
+++ b/terraform/variables.tf
@@ -0,0 +1,42 @@
+variable "resource_group_name" {
+ description = "Name of the Azure Resource Group"
+ type = string
+}
+
+variable "location" {
+ description = "Azure region where resources will be deployed"
+ type = string
+ default = "Australia East"
+}
+
+variable "acr_name" {
+ description = "Globally unique Azure Container Registry name"
+ type = string
+}
+
+variable "aks_name" {
+ description = "Name of the Azure Kubernetes Service cluster"
+ type = string
+}
+
+variable "storage_account_name" {
+ description = "Globally unique Azure Storage Account name"
+ type = string
+}
+
+variable "aks_node_count" {
+ description = "Number of AKS worker nodes"
+ type = number
+ default = 1
+
+ validation {
+ condition = var.aks_node_count >= 1
+ error_message = "AKS must contain at least one worker node."
+ }
+}
+
+variable "aks_vm_size" {
+ description = "VM size used by the AKS default node pool"
+ type = string
+ default = "Standard_D2s_v3"
+}
diff --git a/terraform/versions.tf b/terraform/versions.tf
new file mode 100644
index 00000000..841bbb65
--- /dev/null
+++ b/terraform/versions.tf
@@ -0,0 +1,14 @@
+terraform {
+ required_version = ">= 1.5.0"
+
+ required_providers {
+ azurerm = {
+ source = "hashicorp/azurerm"
+ version = "~> 4.0"
+ }
+ }
+}
+
+provider "azurerm" {
+ features {}
+}
\ No newline at end of file
diff --git a/user-service/app/main.py b/user-service/app/main.py
index 1b4769a6..19c65332 100644
--- a/user-service/app/main.py
+++ b/user-service/app/main.py
@@ -3,7 +3,13 @@
import time
from contextlib import asynccontextmanager
-from fastapi import FastAPI
+from fastapi import FastAPI, Request, Response
+from prometheus_client import (
+ CONTENT_TYPE_LATEST,
+ Counter,
+ Histogram,
+ generate_latest,
+)
from sqlalchemy import select
from sqlalchemy.exc import OperationalError
from sqlalchemy.orm import Session
@@ -21,6 +27,20 @@
logger = logging.getLogger(__name__)
+SERVICE_NAME = "user-service"
+
+HTTP_REQUESTS_TOTAL = Counter(
+ "koalatech_http_requests_total",
+ "Total number of HTTP requests.",
+ ["service", "method", "path", "status_code"],
+)
+
+HTTP_REQUEST_DURATION_SECONDS = Histogram(
+ "koalatech_http_request_duration_seconds",
+ "HTTP request duration in seconds.",
+ ["service", "method", "path"],
+)
+
def initialise_database() -> None:
maximum_attempts = 10
@@ -116,6 +136,36 @@ async def lifespan(_: FastAPI):
)
+@app.middleware("http")
+async def prometheus_metrics(
+ request: Request,
+ call_next,
+):
+ start_time = time.perf_counter()
+
+ response = await call_next(request)
+
+ duration = time.perf_counter() - start_time
+
+ route = request.scope.get("route")
+ path = getattr(route, "path", request.url.path)
+
+ HTTP_REQUESTS_TOTAL.labels(
+ service=SERVICE_NAME,
+ method=request.method,
+ path=path,
+ status_code=str(response.status_code),
+ ).inc()
+
+ HTTP_REQUEST_DURATION_SECONDS.labels(
+ service=SERVICE_NAME,
+ method=request.method,
+ path=path,
+ ).observe(duration)
+
+ return response
+
+
app.include_router(auth.router)
app.include_router(users.router)
@@ -137,5 +187,16 @@ def root() -> dict[str, str]:
def health_check() -> dict[str, str]:
return {
"status": "healthy",
- "service": "user-service",
- }
\ No newline at end of file
+ "service": SERVICE_NAME,
+ }
+
+
+@app.get(
+ "/metrics",
+ include_in_schema=False,
+)
+def metrics() -> Response:
+ return Response(
+ content=generate_latest(),
+ media_type=CONTENT_TYPE_LATEST,
+ )
diff --git a/user-service/requirements.txt b/user-service/requirements.txt
index 02679ad8..a52eb85a 100644
--- a/user-service/requirements.txt
+++ b/user-service/requirements.txt
@@ -5,7 +5,8 @@ psycopg2-binary==2.9.10
pydantic[email]==2.11.7
PyJWT
pwdlib[argon2]==0.2.1
-python-multipart==0.0.20
+python-multipart==0.0.30
pytest==8.4.1
httpx==0.28.1
-python-dotenv==1.0.1
\ No newline at end of file
+python-dotenv==1.0.1
+prometheus-client==0.23.1