From 68bb81661a3f91f045b85c8bf7d0ba27ad6b960c Mon Sep 17 00:00:00 2001 From: Ashan Indika Date: Tue, 8 Sep 2026 02:54:03 +1000 Subject: [PATCH 01/13] Add Week 08 Terraform infrastructure and trigger CD pipeline --- .gitignore | 204 +------------------------------- terraform/.terraform.lock.hcl | 22 ++++ terraform/container_registry.tf | 14 +++ terraform/kubernetes_service.tf | 33 ++++++ terraform/outputs.tf | 30 +++++ terraform/resource_group.tf | 10 ++ terraform/storage_account.tf | 16 +++ terraform/variables.tf | 36 ++++++ terraform/versions.tf | 14 +++ 9 files changed, 180 insertions(+), 199 deletions(-) create mode 100644 terraform/.terraform.lock.hcl create mode 100644 terraform/container_registry.tf create mode 100644 terraform/kubernetes_service.tf create mode 100644 terraform/outputs.tf create mode 100644 terraform/resource_group.tf create mode 100644 terraform/storage_account.tf create mode 100644 terraform/variables.tf create mode 100644 terraform/versions.tf diff --git a/.gitignore b/.gitignore index 4299cd1a..0e0abc1d 100644 --- a/.gitignore +++ b/.gitignore @@ -1,199 +1,5 @@ -# Byte-compiled / optimized / DLL files -__pycache__/ -*.py[cod] -*$py.class - -# C extensions -*.so - -# MacOS -.DS_Store - -# Distribution / packaging -.Python -build/ -develop-eggs/ -dist/ -downloads/ -eggs/ -.eggs/ -lib/ -lib64/ -parts/ -sdist/ -var/ -wheels/ -share/python-wheels/ -*.egg-info/ -.installed.cfg -*.egg -MANIFEST - -# PyInstaller -# Usually these files are written by a python script from a template -# before PyInstaller builds the exe, so as to inject date/other infos into it. -*.manifest -*.spec - -# Installer logs -pip-log.txt -pip-delete-this-directory.txt - -# Unit test / coverage reports -htmlcov/ -.tox/ -.nox/ -.coverage -.coverage.* -.cache -nosetests.xml -coverage.xml -*.cover -*.py,cover -.hypothesis/ -.pytest_cache/ -cover/ - -# Translations -*.mo -*.pot - -# Django stuff: -*.log -local_settings.py -db.sqlite3 -db.sqlite3-journal - -# Flask stuff: -instance/ -.webassets-cache - -# Scrapy stuff: -.scrapy - -# Sphinx documentation -docs/_build/ - -# PyBuilder -.pybuilder/ -target/ - -# Jupyter Notebook -.ipynb_checkpoints - -# IPython -profile_default/ -ipython_config.py - -# pyenv -# For a library or package, you might want to ignore these files since the code is -# intended to run in multiple environments; otherwise, check them in: -# .python-version - -# pipenv -# According to pypa/pipenv#598, it is recommended to include Pipfile.lock in version control. -# However, in case of collaboration, if having platform-specific dependencies or dependencies -# having no cross-platform support, pipenv may install dependencies that don't work, or not -# install all needed dependencies. -#Pipfile.lock - -# UV -# Similar to Pipfile.lock, it is generally recommended to include uv.lock in version control. -# This is especially recommended for binary packages to ensure reproducibility, and is more -# commonly ignored for libraries. -#uv.lock - -# poetry -# Similar to Pipfile.lock, it is generally recommended to include poetry.lock in version control. -# This is especially recommended for binary packages to ensure reproducibility, and is more -# commonly ignored for libraries. -# https://python-poetry.org/docs/basic-usage/#commit-your-poetrylock-file-to-version-control -#poetry.lock - -# pdm -# Similar to Pipfile.lock, it is generally recommended to include pdm.lock in version control. -#pdm.lock -# pdm stores project-wide configurations in .pdm.toml, but it is recommended to not include it -# in version control. -# https://pdm.fming.dev/latest/usage/project/#working-with-version-control -.pdm.toml -.pdm-python -.pdm-build/ - -# PEP 582; used by e.g. github.com/David-OConnor/pyflow and github.com/pdm-project/pdm -__pypackages__/ - -# Celery stuff -celerybeat-schedule -celerybeat.pid - -# SageMath parsed files -*.sage.py - -# Environments -.env -.venv -env/ -venv/ -ENV/ -env.bak/ -venv.bak/ - -# Spyder project settings -.spyderproject -.spyproject - -# Rope project settings -.ropeproject - -# mkdocs documentation -/site - -# mypy -.mypy_cache/ -.dmypy.json -dmypy.json - -# Pyre type checker -.pyre/ - -# pytype static type analyzer -.pytype/ - -# Cython debug symbols -cython_debug/ - -# PyCharm -# JetBrains specific template is maintained in a separate JetBrains.gitignore that can -# be found at https://github.com/github/gitignore/blob/main/Global/JetBrains.gitignore -# and can be added to the global gitignore or merged into this file. For a more nuclear -# option (not recommended) you can uncomment the following to ignore the entire idea folder. -#.idea/ - -# Abstra -# Abstra is an AI-powered process automation framework. -# Ignore directories containing user credentials, local state, and settings. -# Learn more at https://abstra.io/docs -.abstra/ - -# Visual Studio Code -# Visual Studio Code specific template is maintained in a separate VisualStudioCode.gitignore -# that can be found at https://github.com/github/gitignore/blob/main/Global/VisualStudioCode.gitignore -# and can be added to the global gitignore or merged into this file. However, if you prefer, -# you could uncomment the following to ignore the enitre vscode folder -# .vscode/ - -# Ruff stuff: -.ruff_cache/ - -# PyPI configuration file -.pypirc - -# Cursor -# Cursor is an AI-powered code editor. `.cursorignore` specifies files/directories to -# exclude from AI features like autocomplete and code analysis. Recommended for sensitive data -# refer to https://docs.cursor.com/context/ignore-files -.cursorignore -.cursorindexingignore - -node_modules/ \ No newline at end of file +terraform/.terraform/ +terraform/*.tfstate +terraform/*.tfstate.* +terraform/*.tfplan +terraform/terraform.tfvars diff --git a/terraform/.terraform.lock.hcl b/terraform/.terraform.lock.hcl new file mode 100644 index 00000000..1c302601 --- /dev/null +++ b/terraform/.terraform.lock.hcl @@ -0,0 +1,22 @@ +# This file is maintained automatically by "terraform init". +# Manual edits may be lost in future updates. + +provider "registry.terraform.io/hashicorp/azurerm" { + version = "4.81.0" + constraints = "~> 4.0" + hashes = [ + "h1:XhToZua4gtih1Kv8RdStcfND83G4Tmb6GZFT4jEUhDU=", + "zh:0732e7b74264ddfa2b90ba69d01c283d3cbae9f72ed3e506c6ac92529fed7fd3", + "zh:12afb524e232fe4e3d6161927724af5dfa4831d71edd9c174917ca9b7377bfae", + "zh:169d619ae202c4145e02fb706fb7c3679445ab3e3ff722edbf89597517a8c92e", + "zh:6beb95a3ef2f2d9c76abaa48e5450e90686a3fb6a47f1cb0ff7c5e94b6960151", + "zh:705e075fb5ffc4bf66fd7cbabf1a65007a41621e80030a2c158a4c83b6046216", + "zh:78d5eefdd9e494defcb3c68d282b8f96630502cac21d1ea161f53cfe9bb483b3", + "zh:79a8d17fefe647040fcb9ee8821a4f09f395427c4fd49493489b9a93a9a1038e", + "zh:8cc3f900b3774c0ae37ae42365c4579a199cf9e5edf88e476fdf5ab1048f84ea", + "zh:dec373b9390fa95e257291acd018ed65a7d512b428645d35e22cdbe8b245a08b", + "zh:e60f1e9fb45df6defade2855ed6e68547409ea75d30655c556adb0c08579749b", + "zh:f901d12ec82f3f8b5880a27b5cbcd7bd0d97e60c9367a2d7ed82fdd1157b39ff", + "zh:facf68ea5bf0f2b8ba720e7fba5f86492e1d4c591100460bb91c3f79f391f4b6", + ] +} diff --git a/terraform/container_registry.tf b/terraform/container_registry.tf new file mode 100644 index 00000000..32f23508 --- /dev/null +++ b/terraform/container_registry.tf @@ -0,0 +1,14 @@ +resource "azurerm_container_registry" "acr" { + name = var.acr_name + resource_group_name = azurerm_resource_group.rg.name + location = azurerm_resource_group.rg.location + + sku = "Basic" + admin_enabled = false + + tags = { + project = "SIT722" + task = "8.1P" + environment = "week08" + } +} \ No newline at end of file diff --git a/terraform/kubernetes_service.tf b/terraform/kubernetes_service.tf new file mode 100644 index 00000000..a73a9eec --- /dev/null +++ b/terraform/kubernetes_service.tf @@ -0,0 +1,33 @@ +resource "azurerm_kubernetes_cluster" "aks" { + name = var.aks_name + location = azurerm_resource_group.rg.location + resource_group_name = azurerm_resource_group.rg.name + + dns_prefix = var.aks_name + + default_node_pool { + name = "default" + node_count = var.aks_node_count + vm_size = var.aks_vm_size + } + + identity { + type = "SystemAssigned" + } + + role_based_access_control_enabled = true + + tags = { + project = "SIT722" + task = "8.1P" + environment = "week08" + } +} + +# Allow AKS to pull container images from ACR +resource "azurerm_role_assignment" "acr_pull" { + principal_id = azurerm_kubernetes_cluster.aks.kubelet_identity[0].object_id + role_definition_name = "AcrPull" + scope = azurerm_container_registry.acr.id + skip_service_principal_aad_check = true +} \ No newline at end of file diff --git a/terraform/outputs.tf b/terraform/outputs.tf new file mode 100644 index 00000000..283cdcb2 --- /dev/null +++ b/terraform/outputs.tf @@ -0,0 +1,30 @@ +output "resource_group_name" { + description = "Azure Resource Group name" + value = azurerm_resource_group.rg.name +} + +output "acr_name" { + description = "Azure Container Registry name" + value = azurerm_container_registry.acr.name +} + +output "acr_login_server" { + description = "Azure Container Registry login server" + value = azurerm_container_registry.acr.login_server +} + +output "aks_cluster_name" { + description = "AKS cluster name" + value = azurerm_kubernetes_cluster.aks.name +} + +output "storage_account_name" { + description = "Azure Storage Account name" + value = azurerm_storage_account.storage.name +} + +output "storage_connection_string" { + description = "Azure Storage Account connection string" + value = azurerm_storage_account.storage.primary_connection_string + sensitive = true +} \ No newline at end of file diff --git a/terraform/resource_group.tf b/terraform/resource_group.tf new file mode 100644 index 00000000..76431283 --- /dev/null +++ b/terraform/resource_group.tf @@ -0,0 +1,10 @@ +resource "azurerm_resource_group" "rg" { + name = var.resource_group_name + location = var.location + + tags = { + project = "SIT722" + task = "8.1P" + environment = "week08" + } +} \ No newline at end of file diff --git a/terraform/storage_account.tf b/terraform/storage_account.tf new file mode 100644 index 00000000..89531535 --- /dev/null +++ b/terraform/storage_account.tf @@ -0,0 +1,16 @@ +resource "azurerm_storage_account" "storage" { + name = var.storage_account_name + resource_group_name = azurerm_resource_group.rg.name + location = azurerm_resource_group.rg.location + + account_tier = "Standard" + account_replication_type = "LRS" + + min_tls_version = "TLS1_2" + + tags = { + project = "SIT722" + task = "8.1P" + environment = "week08" + } +} \ No newline at end of file diff --git a/terraform/variables.tf b/terraform/variables.tf new file mode 100644 index 00000000..45f8d61b --- /dev/null +++ b/terraform/variables.tf @@ -0,0 +1,36 @@ +variable "resource_group_name" { + description = "Name of the Azure Resource Group" + type = string +} + +variable "location" { + description = "Azure region where resources will be deployed" + type = string + default = "Australia East" +} + +variable "acr_name" { + description = "Globally unique Azure Container Registry name" + type = string +} + +variable "aks_name" { + description = "Name of the Azure Kubernetes Service cluster" + type = string +} + +variable "storage_account_name" { + description = "Globally unique Azure Storage Account name" + type = string +} + +variable "aks_node_count" { + description = "Number of AKS worker nodes" + type = number + default = 3 +} + +variable "aks_vm_size" { + description = "VM size used by the AKS default node pool" + type = string +} \ No newline at end of file diff --git a/terraform/versions.tf b/terraform/versions.tf new file mode 100644 index 00000000..841bbb65 --- /dev/null +++ b/terraform/versions.tf @@ -0,0 +1,14 @@ +terraform { + required_version = ">= 1.5.0" + + required_providers { + azurerm = { + source = "hashicorp/azurerm" + version = "~> 4.0" + } + } +} + +provider "azurerm" { + features {} +} \ No newline at end of file From 86c5ea095df56d1100a61afdc464edee00d0b761 Mon Sep 17 00:00:00 2001 From: Ashan Indika Date: Fri, 18 Sep 2026 16:11:58 +1000 Subject: [PATCH 02/13] Implement CI and staging deployment updates for 9.3C --- .github/workflows/01-ci.yml | 71 ++++++++++++++++--- kubernetes/staging/07-user-service.yaml | 2 +- kubernetes/staging/08-student-service.yaml | 2 +- kubernetes/staging/09-lecturer-service.yaml | 2 +- kubernetes/staging/10-course-service.yaml | 2 +- kubernetes/staging/11-enrollment-service.yaml | 2 +- kubernetes/staging/12-frontend.yaml | 2 +- 7 files changed, 69 insertions(+), 14 deletions(-) diff --git a/.github/workflows/01-ci.yml b/.github/workflows/01-ci.yml index 8206db43..188d1a32 100644 --- a/.github/workflows/01-ci.yml +++ b/.github/workflows/01-ci.yml @@ -1,12 +1,25 @@ name: 01 - CI on: - # Trigger the workflow on push to main branch + # ========================================================= + # Pull Request Validation + # ========================================================= + # Run tests when a pull request targets main. + # Docker images are NOT built/pushed for pull requests. + pull_request: + branches: + - main + + # ========================================================= + # Main Branch CI + # ========================================================= + # After a pull request is merged into main, run tests, + # build the Docker images and push them to ACR. push: branches: - main - - # Manual trigger for the workflow + + # Allow the workflow to be started manually if required. workflow_dispatch: @@ -102,6 +115,36 @@ jobs: pytest -v + # ========================================================= + # Frontend Tests + # ========================================================= + frontend-test: + name: Test Frontend + runs-on: ubuntu-latest + + steps: + + - name: Checkout repository + uses: actions/checkout@v4 + + - name: Set up Node.js + uses: actions/setup-node@v4 + with: + node-version: "20" + cache: "npm" + cache-dependency-path: frontend/package-lock.json + + - name: Install dependencies + working-directory: frontend + run: | + npm ci + + - name: Run frontend tests + working-directory: frontend + run: | + npm run test -- --run + + # ========================================================= # Build and Push Docker Images # ========================================================= @@ -109,12 +152,18 @@ jobs: name: Build and Push ${{ matrix.image }} runs-on: ubuntu-latest - # All backend tests must pass before this job starts + # Both backend and frontend tests must pass before + # Docker images can be built and pushed. needs: - backend-test + - frontend-test - # Build and push when code is pushed to main or manually triggered - if: github.event_name == 'push' || github.event_name == 'workflow_dispatch' + # Do not build/push images for pull requests. + # Build/push only after changes reach main or when + # the workflow is manually triggered. + if: > + github.event_name == 'push' || + github.event_name == 'workflow_dispatch' strategy: fail-fast: false @@ -145,13 +194,14 @@ jobs: uses: actions/checkout@v4 - name: Login to Azure - uses: azure/login@v2 + uses: azure/login@v3 with: creds: ${{ secrets.AZURE_CREDENTIALS }} - name: Login to Azure Container Registry run: | - az acr login --name ${{ vars.ACR_NAME }} + az acr login \ + --name ${{ vars.ACR_NAME }} - name: Build Docker image run: | @@ -164,3 +214,8 @@ jobs: run: | docker push \ ${{ vars.ACR_LOGIN_SERVER }}/${{ matrix.image }}:${{ github.sha }} + + - name: Display published image + run: | + echo "Published image:" + echo "${{ vars.ACR_LOGIN_SERVER }}/${{ matrix.image }}:${{ github.sha }}" \ No newline at end of file diff --git a/kubernetes/staging/07-user-service.yaml b/kubernetes/staging/07-user-service.yaml index 7984533d..0b9ff098 100644 --- a/kubernetes/staging/07-user-service.yaml +++ b/kubernetes/staging/07-user-service.yaml @@ -15,7 +15,7 @@ spec: spec: containers: - name: user-service - image: placeholder + image: sit722week08acr224848845.azurecr.io/koalatech-user-service:latest imagePullPolicy: Always ports: - containerPort: 8000 diff --git a/kubernetes/staging/08-student-service.yaml b/kubernetes/staging/08-student-service.yaml index 244ed083..dbb6716c 100644 --- a/kubernetes/staging/08-student-service.yaml +++ b/kubernetes/staging/08-student-service.yaml @@ -15,7 +15,7 @@ spec: spec: containers: - name: student-service - image: placeholder + image: sit722week08acr224848845.azurecr.io/koalatech-student-service:latest imagePullPolicy: Always ports: - containerPort: 8000 diff --git a/kubernetes/staging/09-lecturer-service.yaml b/kubernetes/staging/09-lecturer-service.yaml index 48a611f6..48719a81 100644 --- a/kubernetes/staging/09-lecturer-service.yaml +++ b/kubernetes/staging/09-lecturer-service.yaml @@ -15,7 +15,7 @@ spec: spec: containers: - name: lecturer-service - image: placeholder + image: sit722week08acr224848845.azurecr.io/koalatech-lecturer-service:latest imagePullPolicy: Always ports: - containerPort: 8000 diff --git a/kubernetes/staging/10-course-service.yaml b/kubernetes/staging/10-course-service.yaml index 652cd06b..f0b1dbee 100644 --- a/kubernetes/staging/10-course-service.yaml +++ b/kubernetes/staging/10-course-service.yaml @@ -15,7 +15,7 @@ spec: spec: containers: - name: course-service - image: placeholder + image: sit722week08acr224848845.azurecr.io/koalatech-course-service:latest imagePullPolicy: Always ports: - containerPort: 8000 diff --git a/kubernetes/staging/11-enrollment-service.yaml b/kubernetes/staging/11-enrollment-service.yaml index 87a21d4f..275a286c 100644 --- a/kubernetes/staging/11-enrollment-service.yaml +++ b/kubernetes/staging/11-enrollment-service.yaml @@ -15,7 +15,7 @@ spec: spec: containers: - name: enrollment-service - image: placeholder + image: sit722week08acr224848845.azurecr.io/koalatech-enrollment-service:latest imagePullPolicy: Always ports: - containerPort: 8000 diff --git a/kubernetes/staging/12-frontend.yaml b/kubernetes/staging/12-frontend.yaml index 91a438da..fd8f353e 100644 --- a/kubernetes/staging/12-frontend.yaml +++ b/kubernetes/staging/12-frontend.yaml @@ -15,7 +15,7 @@ spec: spec: containers: - name: frontend - image: placeholder + image: sit722week08acr224848845.azurecr.io/koalatech-frontend:latest imagePullPolicy: Always ports: - containerPort: 80 From 0cc552e9e002408e1d7ee0c57de815c86d0650bd Mon Sep 17 00:00:00 2001 From: Ashan Indika Date: Fri, 18 Sep 2026 20:27:40 +1000 Subject: [PATCH 03/13] Add frontend service URLs for CI tests --- .github/workflows/01-ci.yml | 20 +++++++++++--------- 1 file changed, 11 insertions(+), 9 deletions(-) diff --git a/.github/workflows/01-ci.yml b/.github/workflows/01-ci.yml index 188d1a32..e9efbf82 100644 --- a/.github/workflows/01-ci.yml +++ b/.github/workflows/01-ci.yml @@ -22,9 +22,7 @@ on: # Allow the workflow to be started manually if required. workflow_dispatch: - jobs: - # ========================================================= # Backend Tests # ========================================================= @@ -94,7 +92,6 @@ jobs: AZURE_STORAGE_CONTAINER_NAME: "" steps: - - name: Checkout repository uses: actions/checkout@v4 @@ -114,7 +111,6 @@ jobs: run: | pytest -v - # ========================================================= # Frontend Tests # ========================================================= @@ -122,8 +118,17 @@ jobs: name: Test Frontend runs-on: ubuntu-latest - steps: + # Test-only service URLs required by the Vite frontend. + # These values allow frontend modules to initialise during + # unit testing without requiring deployed backend services. + env: + VITE_USER_SERVICE_URL: http://localhost:8001 + VITE_STUDENT_SERVICE_URL: http://localhost:8002 + VITE_LECTURER_SERVICE_URL: http://localhost:8003 + VITE_COURSE_SERVICE_URL: http://localhost:8004 + VITE_ENROLLMENT_SERVICE_URL: http://localhost:8005 + steps: - name: Checkout repository uses: actions/checkout@v4 @@ -143,8 +148,6 @@ jobs: working-directory: frontend run: | npm run test -- --run - - # ========================================================= # Build and Push Docker Images # ========================================================= @@ -189,7 +192,6 @@ jobs: image: koalatech-enrollment-service steps: - - name: Checkout repository uses: actions/checkout@v4 @@ -218,4 +220,4 @@ jobs: - name: Display published image run: | echo "Published image:" - echo "${{ vars.ACR_LOGIN_SERVER }}/${{ matrix.image }}:${{ github.sha }}" \ No newline at end of file + echo "${{ vars.ACR_LOGIN_SERVER }}/${{ matrix.image }}:${{ github.sha }}" From cf1de24257486d9e0ac9c32a0c7f115a5c5400ac Mon Sep 17 00:00:00 2001 From: Ashan Indika Date: Fri, 18 Sep 2026 21:56:10 +1000 Subject: [PATCH 04/13] Automate tested release promotion to production --- .github/workflows/04-deploy-production.yml | 160 +++++++++++++-------- 1 file changed, 100 insertions(+), 60 deletions(-) diff --git a/.github/workflows/04-deploy-production.yml b/.github/workflows/04-deploy-production.yml index 969d646b..d0cd26a4 100644 --- a/.github/workflows/04-deploy-production.yml +++ b/.github/workflows/04-deploy-production.yml @@ -1,16 +1,21 @@ name: 04 - Deploy to Production on: - workflow_dispatch: - inputs: - image_tag: - description: "Tested image SHA to deploy" - required: true - type: string + workflow_run: + workflows: + - "03 - Test Staging" + types: + - completed + branches: + - main jobs: deploy-production: name: Deploy to Production + + # Production is promoted only after the staging smoke test succeeds. + if: ${{ github.event.workflow_run.conclusion == 'success' }} + runs-on: ubuntu-latest environment: @@ -19,6 +24,8 @@ jobs: steps: - name: Checkout repository uses: actions/checkout@v4 + with: + ref: ${{ github.event.workflow_run.head_sha }} - name: Login to Azure uses: azure/login@v3 @@ -32,6 +39,60 @@ jobs: --name ${{ vars.AKS_CLUSTER_NAME }} \ --overwrite-existing + # Read the exact frontend image that has already passed + # deployment and smoke testing in staging. + - name: Get tested image SHA from staging + id: tested-image + shell: bash + run: | + TESTED_IMAGE=$(kubectl get deployment frontend \ + -n staging \ + -o jsonpath='{.spec.template.spec.containers[0].image}') + + IMAGE_TAG="${TESTED_IMAGE##*:}" + + echo "Tested staging image: $TESTED_IMAGE" + echo "Promoting image tag: $IMAGE_TAG" + + if [ -z "$IMAGE_TAG" ]; then + echo "Unable to determine tested staging image tag." + exit 1 + fi + + echo "image_tag=$IMAGE_TAG" >> "$GITHUB_OUTPUT" + + # Verify that all six application deployments in staging + # are running the exact same tested release. + - name: Verify staging release consistency + shell: bash + run: | + IMAGE_TAG="${{ steps.tested-image.outputs.image_tag }}" + + for deployment in \ + frontend \ + user-service \ + student-service \ + lecturer-service \ + course-service \ + enrollment-service + do + IMAGE=$(kubectl get deployment "$deployment" \ + -n staging \ + -o jsonpath='{.spec.template.spec.containers[0].image}') + + echo "$deployment -> $IMAGE" + + case "$IMAGE" in + *:"$IMAGE_TAG") + echo "$deployment uses tested release $IMAGE_TAG" + ;; + *) + echo "$deployment does not use tested release $IMAGE_TAG" + exit 1 + ;; + esac + done + - name: Create production namespace run: | kubectl create namespace production \ @@ -63,80 +124,59 @@ jobs: - name: Apply Kubernetes manifests run: | - kubectl apply \ - -f kubernetes/production/ + kubectl apply -f kubernetes/production/ - - name: Update frontend image + - name: Promote tested images to production + shell: bash run: | + IMAGE_TAG="${{ steps.tested-image.outputs.image_tag }}" + + echo "Promoting tested release $IMAGE_TAG to production" + kubectl set image deployment/frontend \ - frontend=${{ vars.ACR_LOGIN_SERVER }}/koalatech-frontend:${{ inputs.image_tag }} \ + frontend=${{ vars.ACR_LOGIN_SERVER }}/koalatech-frontend:$IMAGE_TAG \ -n production - - name: Update user-service image - run: | kubectl set image deployment/user-service \ - user-service=${{ vars.ACR_LOGIN_SERVER }}/koalatech-user-service:${{ inputs.image_tag }} \ + user-service=${{ vars.ACR_LOGIN_SERVER }}/koalatech-user-service:$IMAGE_TAG \ -n production - - name: Update student-service image - run: | kubectl set image deployment/student-service \ - student-service=${{ vars.ACR_LOGIN_SERVER }}/koalatech-student-service:${{ inputs.image_tag }} \ + student-service=${{ vars.ACR_LOGIN_SERVER }}/koalatech-student-service:$IMAGE_TAG \ -n production - - name: Update lecturer-service image - run: | kubectl set image deployment/lecturer-service \ - lecturer-service=${{ vars.ACR_LOGIN_SERVER }}/koalatech-lecturer-service:${{ inputs.image_tag }} \ + lecturer-service=${{ vars.ACR_LOGIN_SERVER }}/koalatech-lecturer-service:$IMAGE_TAG \ -n production - - name: Update course-service image - run: | kubectl set image deployment/course-service \ - course-service=${{ vars.ACR_LOGIN_SERVER }}/koalatech-course-service:${{ inputs.image_tag }} \ + course-service=${{ vars.ACR_LOGIN_SERVER }}/koalatech-course-service:$IMAGE_TAG \ -n production - - name: Update enrollment-service image - run: | kubectl set image deployment/enrollment-service \ - enrollment-service=${{ vars.ACR_LOGIN_SERVER }}/koalatech-enrollment-service:${{ inputs.image_tag }} \ + enrollment-service=${{ vars.ACR_LOGIN_SERVER }}/koalatech-enrollment-service:$IMAGE_TAG \ -n production - - name: Wait for frontend rollout - run: | - kubectl rollout status deployment/frontend \ - -n production \ - --timeout=300s - - - name: Wait for user-service rollout - run: | - kubectl rollout status deployment/user-service \ - -n production \ - --timeout=300s - - - name: Wait for student-service rollout - run: | - kubectl rollout status deployment/student-service \ - -n production \ - --timeout=300s - - - name: Wait for lecturer-service rollout - run: | - kubectl rollout status deployment/lecturer-service \ - -n production \ - --timeout=300s - - - name: Wait for course-service rollout - run: | - kubectl rollout status deployment/course-service \ - -n production \ - --timeout=300s - - - name: Wait for enrollment-service rollout - run: | - kubectl rollout status deployment/enrollment-service \ - -n production \ - --timeout=300s + - name: Verify production rollouts + shell: bash + run: | + for deployment in \ + frontend \ + user-service \ + student-service \ + lecturer-service \ + course-service \ + enrollment-service + do + kubectl rollout status deployment/$deployment \ + -n production \ + --timeout=300s + done + + - name: Verify production image versions + run: | + kubectl get deployments -n production \ + -o custom-columns='DEPLOYMENT:.metadata.name,IMAGE:.spec.template.spec.containers[*].image' - name: Show production resources run: | From a58087b1443835a6e64bac2cef0f09d8659b4e2a Mon Sep 17 00:00:00 2001 From: Ashan Indika Date: Fri, 18 Sep 2026 23:10:08 +1000 Subject: [PATCH 05/13] Add visible continuous deployment demonstration --- .gitignore | 3 +++ frontend/src/pages/Login.jsx | 8 +++++++- frontend/src/test/Login.test.jsx | 7 ++++++- 3 files changed, 16 insertions(+), 2 deletions(-) diff --git a/.gitignore b/.gitignore index 0e0abc1d..875ed826 100644 --- a/.gitignore +++ b/.gitignore @@ -3,3 +3,6 @@ terraform/*.tfstate terraform/*.tfstate.* terraform/*.tfplan terraform/terraform.tfvars + +# Node.js dependencies +node_modules/ diff --git a/frontend/src/pages/Login.jsx b/frontend/src/pages/Login.jsx index 1132741c..2918a615 100644 --- a/frontend/src/pages/Login.jsx +++ b/frontend/src/pages/Login.jsx @@ -98,11 +98,17 @@ const Login = () => { Sign in to continue + + Successfully deployed automatically through GitHub Actions + + {error && ( { ) ).toBeInTheDocument(); + expect( + screen.getByText( + "Successfully deployed automatically through GitHub Actions" + ) + ).toBeInTheDocument(); + expect( screen.getByRole("textbox", { name: /username/i, @@ -53,7 +59,6 @@ describe("Login page", () => { expect( screen.getByLabelText(/password/i) ).toBeInTheDocument(); - expect( screen.getByRole("button", { name: /login/i, From 991bd2a51b34551fe797d84ceb8964e0f3f0708a Mon Sep 17 00:00:00 2001 From: Ashan Indika Date: Fri, 25 Sep 2026 02:53:11 +1000 Subject: [PATCH 06/13] feat: optimise Terraform infrastructure for Task 10.2D --- terraform/container_registry.tf | 7 ++++--- terraform/kubernetes_service.tf | 9 +++++---- terraform/resource_group.tf | 7 ++++--- terraform/storage_account.tf | 7 ++++--- terraform/variables.tf | 10 ++++++++-- 5 files changed, 25 insertions(+), 15 deletions(-) diff --git a/terraform/container_registry.tf b/terraform/container_registry.tf index 32f23508..36266676 100644 --- a/terraform/container_registry.tf +++ b/terraform/container_registry.tf @@ -8,7 +8,8 @@ resource "azurerm_container_registry" "acr" { tags = { project = "SIT722" - task = "8.1P" - environment = "week08" + task = "10.2D" + environment = "week10" + purpose = "infrastructure-security-monitoring" } -} \ No newline at end of file +} diff --git a/terraform/kubernetes_service.tf b/terraform/kubernetes_service.tf index a73a9eec..ab5e73e4 100644 --- a/terraform/kubernetes_service.tf +++ b/terraform/kubernetes_service.tf @@ -19,15 +19,16 @@ resource "azurerm_kubernetes_cluster" "aks" { tags = { project = "SIT722" - task = "8.1P" - environment = "week08" + task = "10.2D" + environment = "week10" + purpose = "infrastructure-security-monitoring" } } -# Allow AKS to pull container images from ACR +# Allow AKS to pull container images from ACR. resource "azurerm_role_assignment" "acr_pull" { principal_id = azurerm_kubernetes_cluster.aks.kubelet_identity[0].object_id role_definition_name = "AcrPull" scope = azurerm_container_registry.acr.id skip_service_principal_aad_check = true -} \ No newline at end of file +} diff --git a/terraform/resource_group.tf b/terraform/resource_group.tf index 76431283..75d42f14 100644 --- a/terraform/resource_group.tf +++ b/terraform/resource_group.tf @@ -4,7 +4,8 @@ resource "azurerm_resource_group" "rg" { tags = { project = "SIT722" - task = "8.1P" - environment = "week08" + task = "10.2D" + environment = "week10" + purpose = "infrastructure-security-monitoring" } -} \ No newline at end of file +} diff --git a/terraform/storage_account.tf b/terraform/storage_account.tf index 89531535..e9d9eaa8 100644 --- a/terraform/storage_account.tf +++ b/terraform/storage_account.tf @@ -10,7 +10,8 @@ resource "azurerm_storage_account" "storage" { tags = { project = "SIT722" - task = "8.1P" - environment = "week08" + task = "10.2D" + environment = "week10" + purpose = "infrastructure-security-monitoring" } -} \ No newline at end of file +} diff --git a/terraform/variables.tf b/terraform/variables.tf index 45f8d61b..42291769 100644 --- a/terraform/variables.tf +++ b/terraform/variables.tf @@ -27,10 +27,16 @@ variable "storage_account_name" { variable "aks_node_count" { description = "Number of AKS worker nodes" type = number - default = 3 + default = 1 + + validation { + condition = var.aks_node_count >= 1 + error_message = "AKS must contain at least one worker node." + } } variable "aks_vm_size" { description = "VM size used by the AKS default node pool" type = string -} \ No newline at end of file + default = "Standard_D2s_v3" +} From ffcdd5033b44470582b2315bb1696bd707057243 Mon Sep 17 00:00:00 2001 From: Ashan Indika Date: Fri, 25 Sep 2026 03:41:46 +1000 Subject: [PATCH 07/13] feat: integrate Terraform validation and Docker Scout security scanning --- .github/workflows/01-ci.yml | 103 ++++++++++++++++++++++++++++------ user-service/requirements.txt | 4 +- 2 files changed, 89 insertions(+), 18 deletions(-) diff --git a/.github/workflows/01-ci.yml b/.github/workflows/01-ci.yml index e9efbf82..9fa43764 100644 --- a/.github/workflows/01-ci.yml +++ b/.github/workflows/01-ci.yml @@ -4,8 +4,6 @@ on: # ========================================================= # Pull Request Validation # ========================================================= - # Run tests when a pull request targets main. - # Docker images are NOT built/pushed for pull requests. pull_request: branches: - main @@ -13,13 +11,11 @@ on: # ========================================================= # Main Branch CI # ========================================================= - # After a pull request is merged into main, run tests, - # build the Docker images and push them to ACR. push: branches: - main - # Allow the workflow to be started manually if required. + # Allow controlled manual execution from GitHub Actions. workflow_dispatch: jobs: @@ -118,9 +114,6 @@ jobs: name: Test Frontend runs-on: ubuntu-latest - # Test-only service URLs required by the Vite frontend. - # These values allow frontend modules to initialise during - # unit testing without requiring deployed backend services. env: VITE_USER_SERVICE_URL: http://localhost:8001 VITE_STUDENT_SERVICE_URL: http://localhost:8002 @@ -148,22 +141,77 @@ jobs: working-directory: frontend run: | npm run test -- --run + + # ========================================================= + # Terraform Validation and Plan + # Task 10.2D - Infrastructure as Code integration + # ========================================================= + terraform-check: + name: Terraform Validation and Plan + runs-on: ubuntu-latest + + # terraform.tfvars is intentionally not committed. + # Non-sensitive Week10 values are supplied using TF_VAR_*. + env: + TF_VAR_resource_group_name: "sit722-week10-rg" + TF_VAR_location: "Australia East" + TF_VAR_acr_name: "sit722week10acr224848845" + TF_VAR_aks_name: "sit722-week10-aks" + TF_VAR_storage_account_name: "sit722w10storage224848" + TF_VAR_aks_node_count: "1" + TF_VAR_aks_vm_size: "Standard_D2s_v3" + + steps: + - name: Checkout repository + uses: actions/checkout@v4 + + - name: Set up Terraform + uses: hashicorp/setup-terraform@v3 + + - name: Login to Azure + uses: azure/login@v3 + with: + creds: ${{ secrets.AZURE_CREDENTIALS }} + + - name: Terraform Init + working-directory: terraform + run: | + terraform init -input=false + + - name: Terraform Format Check + working-directory: terraform + run: | + terraform fmt -check + + - name: Terraform Validate + working-directory: terraform + run: | + terraform validate + + - name: Terraform Plan + working-directory: terraform + run: | + terraform plan \ + -input=false \ + -no-color + # ========================================================= - # Build and Push Docker Images + # Build, Security Scan and Push Docker Images + # Task 10.2D - Docker Scout integration # ========================================================= - build-and-push: - name: Build and Push ${{ matrix.image }} + build-scan-and-push: + name: Build, Scan and Push ${{ matrix.image }} runs-on: ubuntu-latest - # Both backend and frontend tests must pass before - # Docker images can be built and pushed. + # Application tests AND Terraform validation must succeed + # before container images are published. needs: - backend-test - frontend-test + - terraform-check - # Do not build/push images for pull requests. - # Build/push only after changes reach main or when - # the workflow is manually triggered. + # Pull requests perform validation only. + # Images are published for main or controlled manual runs. if: > github.event_name == 'push' || github.event_name == 'workflow_dispatch' @@ -212,6 +260,29 @@ jobs: -t ${{ vars.ACR_LOGIN_SERVER }}/${{ matrix.image }}:${{ github.sha }} \ ./${{ matrix.service }} + # ------------------------------------------------------- + # Docker Scout vulnerability analysis occurs BEFORE push. + # + # The scan is currently evidence/reporting mode while the + # actual Task 10.2D vulnerability is identified and + # remediated. The image will subsequently be rescanned. + # ------------------------------------------------------- + - name: Docker Scout CVE scan + uses: docker/scout-action@v1 + with: + command: cves + image: local://${{ vars.ACR_LOGIN_SERVER }}/${{ matrix.image }}:${{ github.sha }} + only-severities: critical,high + write-comment: false + exit-code: false + + - name: Docker Scout remediation recommendations + uses: docker/scout-action@v1 + with: + command: recommendations + image: local://${{ vars.ACR_LOGIN_SERVER }}/${{ matrix.image }}:${{ github.sha }} + write-comment: false + - name: Push Docker image with commit SHA run: | docker push \ diff --git a/user-service/requirements.txt b/user-service/requirements.txt index 02679ad8..f07adff6 100644 --- a/user-service/requirements.txt +++ b/user-service/requirements.txt @@ -5,7 +5,7 @@ psycopg2-binary==2.9.10 pydantic[email]==2.11.7 PyJWT pwdlib[argon2]==0.2.1 -python-multipart==0.0.20 +python-multipart==0.0.30 pytest==8.4.1 httpx==0.28.1 -python-dotenv==1.0.1 \ No newline at end of file +python-dotenv==1.0.1 From 1c532a4d7909c11016558cde60deade1cfdf5339 Mon Sep 17 00:00:00 2001 From: Ashan Indika Date: Fri, 25 Sep 2026 04:26:09 +1000 Subject: [PATCH 08/13] feat: add Prometheus Grafana and Kubernetes monitoring for Task 10.2D --- .github/workflows/02-deploy-staging.yml | 247 ++++++++++++++---- .github/workflows/04-deploy-production.yml | 14 +- .gitignore | 4 + course-service/app/main.py | 65 ++++- course-service/requirements.txt | 3 +- enrollment-service/app/main.py | 65 ++++- enrollment-service/requirements.txt | 3 +- kubernetes/monitoring/01-namespace.yaml | 4 + kubernetes/monitoring/02-prometheus-rbac.yaml | 46 ++++ .../monitoring/03-prometheus-config.yaml | 52 ++++ kubernetes/monitoring/04-prometheus.yaml | 56 ++++ .../monitoring/05-kube-state-metrics.yaml | 108 ++++++++ .../monitoring/06-grafana-datasource.yaml | 17 ++ .../monitoring/07-grafana-dashboard.yaml | 157 +++++++++++ .../08-grafana-dashboard-provider.yaml | 19 ++ kubernetes/monitoring/09-grafana.yaml | 71 +++++ lecturer-service/app/main.py | 65 ++++- lecturer-service/requirements.txt | 3 +- student-service/app/main.py | 75 +++++- student-service/requirements.txt | 3 +- user-service/app/main.py | 67 ++++- user-service/requirements.txt | 1 + 22 files changed, 1060 insertions(+), 85 deletions(-) create mode 100644 kubernetes/monitoring/01-namespace.yaml create mode 100644 kubernetes/monitoring/02-prometheus-rbac.yaml create mode 100644 kubernetes/monitoring/03-prometheus-config.yaml create mode 100644 kubernetes/monitoring/04-prometheus.yaml create mode 100644 kubernetes/monitoring/05-kube-state-metrics.yaml create mode 100644 kubernetes/monitoring/06-grafana-datasource.yaml create mode 100644 kubernetes/monitoring/07-grafana-dashboard.yaml create mode 100644 kubernetes/monitoring/08-grafana-dashboard-provider.yaml create mode 100644 kubernetes/monitoring/09-grafana.yaml diff --git a/.github/workflows/02-deploy-staging.yml b/.github/workflows/02-deploy-staging.yml index 006bb70d..502d0d63 100644 --- a/.github/workflows/02-deploy-staging.yml +++ b/.github/workflows/02-deploy-staging.yml @@ -11,7 +11,7 @@ on: jobs: deploy-staging: - name: Deploy to Staging + name: Deploy Application and Monitoring to Staging runs-on: ubuntu-latest if: > @@ -67,85 +67,238 @@ jobs: --dry-run=client \ -o yaml | kubectl apply -f - - - name: Apply Kubernetes manifests + - name: Apply staging Kubernetes manifests run: | - kubectl apply \ - -f kubernetes/staging/ + kubectl apply -f kubernetes/staging/ - - name: Update frontend image + - name: Deploy tested application images + shell: bash run: | + IMAGE_TAG="${{ github.event.workflow_run.head_sha }}" + kubectl set image deployment/frontend \ - frontend=${{ vars.ACR_LOGIN_SERVER }}/koalatech-frontend:${{ github.event.workflow_run.head_sha }} \ + frontend=${{ vars.ACR_LOGIN_SERVER }}/koalatech-frontend:$IMAGE_TAG \ -n staging - - name: Update user-service image - run: | kubectl set image deployment/user-service \ - user-service=${{ vars.ACR_LOGIN_SERVER }}/koalatech-user-service:${{ github.event.workflow_run.head_sha }} \ + user-service=${{ vars.ACR_LOGIN_SERVER }}/koalatech-user-service:$IMAGE_TAG \ -n staging - - name: Update student-service image - run: | kubectl set image deployment/student-service \ - student-service=${{ vars.ACR_LOGIN_SERVER }}/koalatech-student-service:${{ github.event.workflow_run.head_sha }} \ + student-service=${{ vars.ACR_LOGIN_SERVER }}/koalatech-student-service:$IMAGE_TAG \ -n staging - - name: Update lecturer-service image - run: | kubectl set image deployment/lecturer-service \ - lecturer-service=${{ vars.ACR_LOGIN_SERVER }}/koalatech-lecturer-service:${{ github.event.workflow_run.head_sha }} \ + lecturer-service=${{ vars.ACR_LOGIN_SERVER }}/koalatech-lecturer-service:$IMAGE_TAG \ -n staging - - name: Update course-service image - run: | kubectl set image deployment/course-service \ - course-service=${{ vars.ACR_LOGIN_SERVER }}/koalatech-course-service:${{ github.event.workflow_run.head_sha }} \ + course-service=${{ vars.ACR_LOGIN_SERVER }}/koalatech-course-service:$IMAGE_TAG \ -n staging - - name: Update enrollment-service image - run: | kubectl set image deployment/enrollment-service \ - enrollment-service=${{ vars.ACR_LOGIN_SERVER }}/koalatech-enrollment-service:${{ github.event.workflow_run.head_sha }} \ + enrollment-service=${{ vars.ACR_LOGIN_SERVER }}/koalatech-enrollment-service:$IMAGE_TAG \ -n staging - - name: Wait for frontend rollout + - name: Verify application rollouts + shell: bash run: | - kubectl rollout status deployment/frontend \ - -n staging \ - --timeout=300s + for deployment in \ + frontend \ + user-service \ + student-service \ + lecturer-service \ + course-service \ + enrollment-service + do + echo "Checking rollout: $deployment" + + kubectl rollout status deployment/$deployment \ + -n staging \ + --timeout=300s + done - - name: Wait for user-service rollout + - name: Create monitoring namespace run: | - kubectl rollout status deployment/user-service \ - -n staging \ - --timeout=300s + kubectl apply \ + -f kubernetes/monitoring/01-namespace.yaml - - name: Wait for student-service rollout + - name: Create Grafana admin secret run: | - kubectl rollout status deployment/student-service \ - -n staging \ - --timeout=300s + kubectl create secret generic grafana-admin \ + --namespace monitoring \ + --from-literal=password="${{ secrets.GRAFANA_ADMIN_PASSWORD }}" \ + --dry-run=client \ + -o yaml | kubectl apply -f - - - name: Wait for lecturer-service rollout + - name: Deploy Prometheus and Grafana monitoring run: | - kubectl rollout status deployment/lecturer-service \ - -n staging \ - --timeout=300s + kubectl apply \ + -f kubernetes/monitoring/ - - name: Wait for course-service rollout + - name: Verify monitoring rollouts + shell: bash run: | - kubectl rollout status deployment/course-service \ - -n staging \ - --timeout=300s + for deployment in \ + kube-state-metrics \ + prometheus \ + grafana + do + echo "Checking monitoring deployment: $deployment" + + kubectl rollout status deployment/$deployment \ + -n monitoring \ + --timeout=300s + done + + - name: Verify Prometheus targets + shell: bash + run: | + kubectl port-forward \ + -n monitoring \ + svc/prometheus \ + 9090:9090 > /tmp/prometheus-port-forward.log 2>&1 & + + PF_PID=$! + + cleanup() { + kill "$PF_PID" 2>/dev/null || true + } + + trap cleanup EXIT + + echo "Waiting for Prometheus API..." + + for attempt in {1..20} + do + if curl --fail --silent \ + http://127.0.0.1:9090/-/ready > /dev/null + then + echo "Prometheus is ready." + break + fi + + if [ "$attempt" -eq 20 ]; then + echo "Prometheus did not become ready." + cat /tmp/prometheus-port-forward.log + exit 1 + fi + + sleep 3 + done + + echo + echo "Prometheus targets:" + curl --fail --silent \ + http://127.0.0.1:9090/api/v1/targets \ + > /tmp/prometheus-targets.json - - name: Wait for enrollment-service rollout + python3 - <<'PY' + import json + + with open("/tmp/prometheus-targets.json") as file: + data = json.load(file) + + targets = data["data"]["activeTargets"] + + for target in targets: + labels = target.get("labels", {}) + job = labels.get("job", "unknown") + health = target.get("health", "unknown") + scrape_url = target.get("scrapeUrl", "") + + print( + f"{job:25} " + f"health={health:8} " + f"{scrape_url}" + ) + + required_jobs = { + "user-service", + "student-service", + "lecturer-service", + "course-service", + "enrollment-service", + "kube-state-metrics", + } + + healthy_jobs = { + target.get("labels", {}).get("job") + for target in targets + if target.get("health") == "up" + } + + missing = required_jobs - healthy_jobs + + if missing: + raise SystemExit( + "Required Prometheus targets are not healthy: " + + ", ".join(sorted(missing)) + ) + + print() + print("All required Prometheus targets are healthy.") + PY + + - name: Verify Grafana health + shell: bash run: | - kubectl rollout status deployment/enrollment-service \ - -n staging \ - --timeout=300s + kubectl port-forward \ + -n monitoring \ + svc/grafana \ + 3000:3000 > /tmp/grafana-port-forward.log 2>&1 & + + PF_PID=$! + + cleanup() { + kill "$PF_PID" 2>/dev/null || true + } + + trap cleanup EXIT + + echo "Waiting for Grafana..." + + for attempt in {1..20} + do + if curl --fail --silent \ + http://127.0.0.1:3000/api/health + then + echo + echo "Grafana is healthy." + exit 0 + fi + + sleep 3 + done + + echo "Grafana health validation failed." + cat /tmp/grafana-port-forward.log + exit 1 - name: Show staging resources + if: always() run: | - kubectl get pods -n staging + echo "===== STAGING PODS =====" + kubectl get pods -n staging -o wide + + echo + echo "===== STAGING SERVICES =====" kubectl get services -n staging - kubectl get pvc -n staging \ No newline at end of file + + echo + echo "===== STAGING PVCs =====" + kubectl get pvc -n staging + + - name: Show monitoring resources + if: always() + run: | + echo "===== MONITORING DEPLOYMENTS =====" + kubectl get deployments -n monitoring + + echo + echo "===== MONITORING PODS =====" + kubectl get pods -n monitoring -o wide + + echo + echo "===== MONITORING SERVICES =====" + kubectl get services -n monitoring diff --git a/.github/workflows/04-deploy-production.yml b/.github/workflows/04-deploy-production.yml index d0cd26a4..5bda7b89 100644 --- a/.github/workflows/04-deploy-production.yml +++ b/.github/workflows/04-deploy-production.yml @@ -1,20 +1,14 @@ name: 04 - Deploy to Production on: - workflow_run: - workflows: - - "03 - Test Staging" - types: - - completed - branches: - - main + workflow_dispatch: jobs: deploy-production: name: Deploy to Production - # Production is promoted only after the staging smoke test succeeds. - if: ${{ github.event.workflow_run.conclusion == 'success' }} + # Production deployment is manual during the Week 10 task. + if: ${{ github.event_name == 'workflow_dispatch' }} runs-on: ubuntu-latest @@ -24,8 +18,6 @@ jobs: steps: - name: Checkout repository uses: actions/checkout@v4 - with: - ref: ${{ github.event.workflow_run.head_sha }} - name: Login to Azure uses: azure/login@v3 diff --git a/.gitignore b/.gitignore index 875ed826..bf811163 100644 --- a/.gitignore +++ b/.gitignore @@ -6,3 +6,7 @@ terraform/terraform.tfvars # Node.js dependencies node_modules/ + +# Python generated files +__pycache__/ +*.py[cod] diff --git a/course-service/app/main.py b/course-service/app/main.py index 7c8cbfb2..6d0670ec 100644 --- a/course-service/app/main.py +++ b/course-service/app/main.py @@ -2,7 +2,13 @@ import time from contextlib import asynccontextmanager -from fastapi import FastAPI +from fastapi import FastAPI, Request, Response +from prometheus_client import ( + CONTENT_TYPE_LATEST, + Counter, + Histogram, + generate_latest, +) from sqlalchemy.exc import OperationalError from app.db import Base, engine @@ -16,6 +22,20 @@ logger = logging.getLogger(__name__) +SERVICE_NAME = "course-service" + +HTTP_REQUESTS_TOTAL = Counter( + "koalatech_http_requests_total", + "Total number of HTTP requests.", + ["service", "method", "path", "status_code"], +) + +HTTP_REQUEST_DURATION_SECONDS = Histogram( + "koalatech_http_request_duration_seconds", + "HTTP request duration in seconds.", + ["service", "method", "path"], +) + def initialise_database() -> None: maximum_attempts = 10 @@ -64,6 +84,34 @@ async def lifespan(_: FastAPI): ) +@app.middleware("http") +async def prometheus_metrics( + request: Request, + call_next, +): + start_time = time.perf_counter() + response = await call_next(request) + duration = time.perf_counter() - start_time + + route = request.scope.get("route") + path = getattr(route, "path", request.url.path) + + HTTP_REQUESTS_TOTAL.labels( + service=SERVICE_NAME, + method=request.method, + path=path, + status_code=str(response.status_code), + ).inc() + + HTTP_REQUEST_DURATION_SECONDS.labels( + service=SERVICE_NAME, + method=request.method, + path=path, + ).observe(duration) + + return response + + app.include_router(courses.router) @@ -80,5 +128,16 @@ def root() -> dict[str, str]: def health_check() -> dict[str, str]: return { "status": "healthy", - "service": "course-service", - } \ No newline at end of file + "service": SERVICE_NAME, + } + + +@app.get( + "/metrics", + include_in_schema=False, +) +def metrics() -> Response: + return Response( + content=generate_latest(), + media_type=CONTENT_TYPE_LATEST, + ) diff --git a/course-service/requirements.txt b/course-service/requirements.txt index 58a1178e..dbf225c5 100644 --- a/course-service/requirements.txt +++ b/course-service/requirements.txt @@ -6,4 +6,5 @@ pydantic==2.11.7 PyJWT==2.10.1 python-dotenv==1.0.1 pytest==8.4.1 -httpx==0.28.1 \ No newline at end of file +httpx==0.28.1 +prometheus-client==0.23.1 diff --git a/enrollment-service/app/main.py b/enrollment-service/app/main.py index 868063be..1b33edd3 100644 --- a/enrollment-service/app/main.py +++ b/enrollment-service/app/main.py @@ -2,7 +2,13 @@ import time from contextlib import asynccontextmanager -from fastapi import FastAPI +from fastapi import FastAPI, Request, Response +from prometheus_client import ( + CONTENT_TYPE_LATEST, + Counter, + Histogram, + generate_latest, +) from sqlalchemy.exc import OperationalError from app.db import Base, engine @@ -16,6 +22,20 @@ logger = logging.getLogger(__name__) +SERVICE_NAME = "enrollment-service" + +HTTP_REQUESTS_TOTAL = Counter( + "koalatech_http_requests_total", + "Total number of HTTP requests.", + ["service", "method", "path", "status_code"], +) + +HTTP_REQUEST_DURATION_SECONDS = Histogram( + "koalatech_http_request_duration_seconds", + "HTTP request duration in seconds.", + ["service", "method", "path"], +) + def initialise_database() -> None: maximum_attempts = 10 @@ -64,6 +84,34 @@ async def lifespan(_: FastAPI): ) +@app.middleware("http") +async def prometheus_metrics( + request: Request, + call_next, +): + start_time = time.perf_counter() + response = await call_next(request) + duration = time.perf_counter() - start_time + + route = request.scope.get("route") + path = getattr(route, "path", request.url.path) + + HTTP_REQUESTS_TOTAL.labels( + service=SERVICE_NAME, + method=request.method, + path=path, + status_code=str(response.status_code), + ).inc() + + HTTP_REQUEST_DURATION_SECONDS.labels( + service=SERVICE_NAME, + method=request.method, + path=path, + ).observe(duration) + + return response + + app.include_router(enrollments.router) @@ -80,5 +128,16 @@ def root() -> dict[str, str]: def health_check() -> dict[str, str]: return { "status": "healthy", - "service": "enrollment-service", - } \ No newline at end of file + "service": SERVICE_NAME, + } + + +@app.get( + "/metrics", + include_in_schema=False, +) +def metrics() -> Response: + return Response( + content=generate_latest(), + media_type=CONTENT_TYPE_LATEST, + ) diff --git a/enrollment-service/requirements.txt b/enrollment-service/requirements.txt index 58a1178e..dbf225c5 100644 --- a/enrollment-service/requirements.txt +++ b/enrollment-service/requirements.txt @@ -6,4 +6,5 @@ pydantic==2.11.7 PyJWT==2.10.1 python-dotenv==1.0.1 pytest==8.4.1 -httpx==0.28.1 \ No newline at end of file +httpx==0.28.1 +prometheus-client==0.23.1 diff --git a/kubernetes/monitoring/01-namespace.yaml b/kubernetes/monitoring/01-namespace.yaml new file mode 100644 index 00000000..d3252360 --- /dev/null +++ b/kubernetes/monitoring/01-namespace.yaml @@ -0,0 +1,4 @@ +apiVersion: v1 +kind: Namespace +metadata: + name: monitoring diff --git a/kubernetes/monitoring/02-prometheus-rbac.yaml b/kubernetes/monitoring/02-prometheus-rbac.yaml new file mode 100644 index 00000000..bb268018 --- /dev/null +++ b/kubernetes/monitoring/02-prometheus-rbac.yaml @@ -0,0 +1,46 @@ +apiVersion: v1 +kind: ServiceAccount +metadata: + name: prometheus + namespace: monitoring + +--- +apiVersion: rbac.authorization.k8s.io/v1 +kind: ClusterRole +metadata: + name: prometheus +rules: + - apiGroups: [""] + resources: + - nodes + - nodes/proxy + - services + - endpoints + - pods + verbs: + - get + - list + - watch + - apiGroups: + - extensions + - networking.k8s.io + resources: + - ingresses + verbs: + - get + - list + - watch + +--- +apiVersion: rbac.authorization.k8s.io/v1 +kind: ClusterRoleBinding +metadata: + name: prometheus +roleRef: + apiGroup: rbac.authorization.k8s.io + kind: ClusterRole + name: prometheus +subjects: + - kind: ServiceAccount + name: prometheus + namespace: monitoring diff --git a/kubernetes/monitoring/03-prometheus-config.yaml b/kubernetes/monitoring/03-prometheus-config.yaml new file mode 100644 index 00000000..d40f0350 --- /dev/null +++ b/kubernetes/monitoring/03-prometheus-config.yaml @@ -0,0 +1,52 @@ +apiVersion: v1 +kind: ConfigMap +metadata: + name: prometheus-config + namespace: monitoring +data: + prometheus.yml: | + global: + scrape_interval: 15s + evaluation_interval: 15s + + scrape_configs: + + - job_name: prometheus + static_configs: + - targets: + - localhost:9090 + + - job_name: user-service + metrics_path: /metrics + static_configs: + - targets: + - user-service.staging.svc.cluster.local:8000 + + - job_name: student-service + metrics_path: /metrics + static_configs: + - targets: + - student-service.staging.svc.cluster.local:8000 + + - job_name: lecturer-service + metrics_path: /metrics + static_configs: + - targets: + - lecturer-service.staging.svc.cluster.local:8000 + + - job_name: course-service + metrics_path: /metrics + static_configs: + - targets: + - course-service.staging.svc.cluster.local:8000 + + - job_name: enrollment-service + metrics_path: /metrics + static_configs: + - targets: + - enrollment-service.staging.svc.cluster.local:8000 + + - job_name: kube-state-metrics + static_configs: + - targets: + - kube-state-metrics.monitoring.svc.cluster.local:8080 diff --git a/kubernetes/monitoring/04-prometheus.yaml b/kubernetes/monitoring/04-prometheus.yaml new file mode 100644 index 00000000..69bed06a --- /dev/null +++ b/kubernetes/monitoring/04-prometheus.yaml @@ -0,0 +1,56 @@ +apiVersion: apps/v1 +kind: Deployment +metadata: + name: prometheus + namespace: monitoring +spec: + replicas: 1 + selector: + matchLabels: + app: prometheus + template: + metadata: + labels: + app: prometheus + spec: + serviceAccountName: prometheus + containers: + - name: prometheus + image: prom/prometheus:v3.5.0 + imagePullPolicy: IfNotPresent + args: + - --config.file=/etc/prometheus/prometheus.yml + - --storage.tsdb.path=/prometheus + - --storage.tsdb.retention.time=6h + ports: + - name: http + containerPort: 9090 + resources: + requests: + cpu: 50m + memory: 128Mi + limits: + cpu: 250m + memory: 384Mi + volumeMounts: + - name: prometheus-config + mountPath: /etc/prometheus + volumes: + - name: prometheus-config + configMap: + name: prometheus-config + +--- +apiVersion: v1 +kind: Service +metadata: + name: prometheus + namespace: monitoring +spec: + selector: + app: prometheus + ports: + - name: http + port: 9090 + targetPort: 9090 + type: ClusterIP diff --git a/kubernetes/monitoring/05-kube-state-metrics.yaml b/kubernetes/monitoring/05-kube-state-metrics.yaml new file mode 100644 index 00000000..d7e4e15d --- /dev/null +++ b/kubernetes/monitoring/05-kube-state-metrics.yaml @@ -0,0 +1,108 @@ +apiVersion: v1 +kind: ServiceAccount +metadata: + name: kube-state-metrics + namespace: monitoring + +--- +apiVersion: rbac.authorization.k8s.io/v1 +kind: ClusterRole +metadata: + name: kube-state-metrics +rules: + - apiGroups: [""] + resources: + - configmaps + - secrets + - nodes + - pods + - services + - resourcequotas + - replicationcontrollers + - limitranges + - persistentvolumeclaims + - persistentvolumes + - namespaces + - endpoints + verbs: + - list + - watch + - apiGroups: + - apps + resources: + - statefulsets + - daemonsets + - deployments + - replicasets + verbs: + - list + - watch + - apiGroups: + - batch + resources: + - cronjobs + - jobs + verbs: + - list + - watch + +--- +apiVersion: rbac.authorization.k8s.io/v1 +kind: ClusterRoleBinding +metadata: + name: kube-state-metrics +roleRef: + apiGroup: rbac.authorization.k8s.io + kind: ClusterRole + name: kube-state-metrics +subjects: + - kind: ServiceAccount + name: kube-state-metrics + namespace: monitoring + +--- +apiVersion: apps/v1 +kind: Deployment +metadata: + name: kube-state-metrics + namespace: monitoring +spec: + replicas: 1 + selector: + matchLabels: + app: kube-state-metrics + template: + metadata: + labels: + app: kube-state-metrics + spec: + serviceAccountName: kube-state-metrics + containers: + - name: kube-state-metrics + image: registry.k8s.io/kube-state-metrics/kube-state-metrics:v2.17.0 + imagePullPolicy: IfNotPresent + ports: + - name: http + containerPort: 8080 + resources: + requests: + cpu: 20m + memory: 32Mi + limits: + cpu: 100m + memory: 128Mi + +--- +apiVersion: v1 +kind: Service +metadata: + name: kube-state-metrics + namespace: monitoring +spec: + selector: + app: kube-state-metrics + ports: + - name: http + port: 8080 + targetPort: 8080 + type: ClusterIP diff --git a/kubernetes/monitoring/06-grafana-datasource.yaml b/kubernetes/monitoring/06-grafana-datasource.yaml new file mode 100644 index 00000000..3aa1e3be --- /dev/null +++ b/kubernetes/monitoring/06-grafana-datasource.yaml @@ -0,0 +1,17 @@ +apiVersion: v1 +kind: ConfigMap +metadata: + name: grafana-datasource + namespace: monitoring +data: + datasource.yaml: | + apiVersion: 1 + + datasources: + - name: Prometheus + uid: prometheus + type: prometheus + access: proxy + url: http://prometheus.monitoring.svc.cluster.local:9090 + isDefault: true + editable: false diff --git a/kubernetes/monitoring/07-grafana-dashboard.yaml b/kubernetes/monitoring/07-grafana-dashboard.yaml new file mode 100644 index 00000000..7ede90f3 --- /dev/null +++ b/kubernetes/monitoring/07-grafana-dashboard.yaml @@ -0,0 +1,157 @@ +apiVersion: v1 +kind: ConfigMap +metadata: + name: grafana-dashboard + namespace: monitoring +data: + koalatech-monitoring.json: | + { + "annotations": { + "list": [] + }, + "editable": true, + "panels": [ + { + "type": "stat", + "title": "HTTP Requests by Backend Service", + "datasource": { + "type": "prometheus", + "uid": "prometheus" + }, + "targets": [ + { + "expr": "sum by (service) (koalatech_http_requests_total)", + "legendFormat": "{{service}}", + "refId": "A" + } + ], + "gridPos": { + "h": 8, + "w": 12, + "x": 0, + "y": 0 + } + }, + { + "type": "timeseries", + "title": "Backend Request Rate", + "datasource": { + "type": "prometheus", + "uid": "prometheus" + }, + "targets": [ + { + "expr": "sum by (service) (rate(koalatech_http_requests_total[5m]))", + "legendFormat": "{{service}}", + "refId": "A" + } + ], + "gridPos": { + "h": 8, + "w": 12, + "x": 12, + "y": 0 + } + }, + { + "type": "timeseries", + "title": "Backend P95 Request Duration", + "datasource": { + "type": "prometheus", + "uid": "prometheus" + }, + "targets": [ + { + "expr": "histogram_quantile(0.95, sum by (le, service) (rate(koalatech_http_request_duration_seconds_bucket[5m])))", + "legendFormat": "{{service}}", + "refId": "A" + } + ], + "gridPos": { + "h": 8, + "w": 12, + "x": 0, + "y": 8 + } + }, + { + "type": "stat", + "title": "Running Staging Pods", + "datasource": { + "type": "prometheus", + "uid": "prometheus" + }, + "targets": [ + { + "expr": "sum(kube_pod_status_phase{namespace=\"staging\",phase=\"Running\"})", + "refId": "A" + } + ], + "gridPos": { + "h": 8, + "w": 6, + "x": 12, + "y": 8 + } + }, + { + "type": "stat", + "title": "Available Staging Deployment Replicas", + "datasource": { + "type": "prometheus", + "uid": "prometheus" + }, + "targets": [ + { + "expr": "sum(kube_deployment_status_replicas_available{namespace=\"staging\"})", + "refId": "A" + } + ], + "gridPos": { + "h": 8, + "w": 6, + "x": 18, + "y": 8 + } + }, + { + "type": "timeseries", + "title": "Staging Container Restarts", + "datasource": { + "type": "prometheus", + "uid": "prometheus" + }, + "targets": [ + { + "expr": "sum by (pod) (kube_pod_container_status_restarts_total{namespace=\"staging\"})", + "legendFormat": "{{pod}}", + "refId": "A" + } + ], + "gridPos": { + "h": 8, + "w": 24, + "x": 0, + "y": 16 + } + } + ], + "refresh": "10s", + "schemaVersion": 41, + "tags": [ + "SIT722", + "KoalaTech", + "10.2D" + ], + "templating": { + "list": [] + }, + "time": { + "from": "now-15m", + "to": "now" + }, + "timezone": "browser", + "title": "KoalaTech - SIT722 Task 10.2D", + "uid": "koalatech-10-2d", + "version": 1 + } diff --git a/kubernetes/monitoring/08-grafana-dashboard-provider.yaml b/kubernetes/monitoring/08-grafana-dashboard-provider.yaml new file mode 100644 index 00000000..1fc6dd16 --- /dev/null +++ b/kubernetes/monitoring/08-grafana-dashboard-provider.yaml @@ -0,0 +1,19 @@ +apiVersion: v1 +kind: ConfigMap +metadata: + name: grafana-dashboard-provider + namespace: monitoring +data: + dashboard.yaml: | + apiVersion: 1 + + providers: + - name: KoalaTech + orgId: 1 + folder: SIT722 + type: file + disableDeletion: false + updateIntervalSeconds: 10 + allowUiUpdates: true + options: + path: /var/lib/grafana/dashboards diff --git a/kubernetes/monitoring/09-grafana.yaml b/kubernetes/monitoring/09-grafana.yaml new file mode 100644 index 00000000..ad604e60 --- /dev/null +++ b/kubernetes/monitoring/09-grafana.yaml @@ -0,0 +1,71 @@ +apiVersion: apps/v1 +kind: Deployment +metadata: + name: grafana + namespace: monitoring +spec: + replicas: 1 + selector: + matchLabels: + app: grafana + template: + metadata: + labels: + app: grafana + spec: + containers: + - name: grafana + image: grafana/grafana:12.1.1 + imagePullPolicy: IfNotPresent + ports: + - name: http + containerPort: 3000 + env: + - name: GF_SECURITY_ADMIN_USER + value: admin + - name: GF_SECURITY_ADMIN_PASSWORD + valueFrom: + secretKeyRef: + name: grafana-admin + key: password + - name: GF_USERS_ALLOW_SIGN_UP + value: "false" + resources: + requests: + cpu: 50m + memory: 128Mi + limits: + cpu: 250m + memory: 256Mi + volumeMounts: + - name: datasource + mountPath: /etc/grafana/provisioning/datasources + - name: dashboard-provider + mountPath: /etc/grafana/provisioning/dashboards + - name: dashboards + mountPath: /var/lib/grafana/dashboards + volumes: + - name: datasource + configMap: + name: grafana-datasource + - name: dashboard-provider + configMap: + name: grafana-dashboard-provider + - name: dashboards + configMap: + name: grafana-dashboard + +--- +apiVersion: v1 +kind: Service +metadata: + name: grafana + namespace: monitoring +spec: + selector: + app: grafana + ports: + - name: http + port: 3000 + targetPort: 3000 + type: ClusterIP diff --git a/lecturer-service/app/main.py b/lecturer-service/app/main.py index 62972d48..028676e8 100644 --- a/lecturer-service/app/main.py +++ b/lecturer-service/app/main.py @@ -2,7 +2,13 @@ import time from contextlib import asynccontextmanager -from fastapi import FastAPI +from fastapi import FastAPI, Request, Response +from prometheus_client import ( + CONTENT_TYPE_LATEST, + Counter, + Histogram, + generate_latest, +) from sqlalchemy.exc import OperationalError from app.db import Base, engine @@ -17,6 +23,20 @@ logger = logging.getLogger(__name__) +SERVICE_NAME = "lecturer-service" + +HTTP_REQUESTS_TOTAL = Counter( + "koalatech_http_requests_total", + "Total number of HTTP requests.", + ["service", "method", "path", "status_code"], +) + +HTTP_REQUEST_DURATION_SECONDS = Histogram( + "koalatech_http_request_duration_seconds", + "HTTP request duration in seconds.", + ["service", "method", "path"], +) + def initialise_database() -> None: maximum_attempts = 10 @@ -78,6 +98,34 @@ async def lifespan(_: FastAPI): ) +@app.middleware("http") +async def prometheus_metrics( + request: Request, + call_next, +): + start_time = time.perf_counter() + response = await call_next(request) + duration = time.perf_counter() - start_time + + route = request.scope.get("route") + path = getattr(route, "path", request.url.path) + + HTTP_REQUESTS_TOTAL.labels( + service=SERVICE_NAME, + method=request.method, + path=path, + status_code=str(response.status_code), + ).inc() + + HTTP_REQUEST_DURATION_SECONDS.labels( + service=SERVICE_NAME, + method=request.method, + path=path, + ).observe(duration) + + return response + + app.include_router(lecturers.router) @@ -94,5 +142,16 @@ def root() -> dict[str, str]: def health_check() -> dict[str, str]: return { "status": "healthy", - "service": "lecturer-service", - } \ No newline at end of file + "service": SERVICE_NAME, + } + + +@app.get( + "/metrics", + include_in_schema=False, +) +def metrics() -> Response: + return Response( + content=generate_latest(), + media_type=CONTENT_TYPE_LATEST, + ) diff --git a/lecturer-service/requirements.txt b/lecturer-service/requirements.txt index feeedd2d..72befaf6 100644 --- a/lecturer-service/requirements.txt +++ b/lecturer-service/requirements.txt @@ -8,4 +8,5 @@ python-multipart==0.0.20 python-dotenv==1.0.1 azure-storage-blob==12.26.0 pytest==8.4.1 -httpx==0.28.1 \ No newline at end of file +httpx==0.28.1 +prometheus-client==0.23.1 diff --git a/student-service/app/main.py b/student-service/app/main.py index 0d85f2ad..3a29e8a4 100644 --- a/student-service/app/main.py +++ b/student-service/app/main.py @@ -2,7 +2,13 @@ import time from contextlib import asynccontextmanager -from fastapi import FastAPI +from fastapi import FastAPI, Request, Response +from prometheus_client import ( + CONTENT_TYPE_LATEST, + Counter, + Histogram, + generate_latest, +) from sqlalchemy.exc import OperationalError from app.db import Base, engine @@ -17,6 +23,20 @@ logger = logging.getLogger(__name__) +SERVICE_NAME = "student-service" + +HTTP_REQUESTS_TOTAL = Counter( + "koalatech_http_requests_total", + "Total number of HTTP requests.", + ["service", "method", "path", "status_code"], +) + +HTTP_REQUEST_DURATION_SECONDS = Histogram( + "koalatech_http_request_duration_seconds", + "HTTP request duration in seconds.", + ["service", "method", "path"], +) + def initialise_database() -> None: maximum_attempts = 10 @@ -79,13 +99,38 @@ async def lifespan(_: FastAPI): ) +@app.middleware("http") +async def prometheus_metrics( + request: Request, + call_next, +): + start_time = time.perf_counter() + response = await call_next(request) + duration = time.perf_counter() - start_time + + route = request.scope.get("route") + path = getattr(route, "path", request.url.path) + + HTTP_REQUESTS_TOTAL.labels( + service=SERVICE_NAME, + method=request.method, + path=path, + status_code=str(response.status_code), + ).inc() + + HTTP_REQUEST_DURATION_SECONDS.labels( + service=SERVICE_NAME, + method=request.method, + path=path, + ).observe(duration) + + return response + + app.include_router(students.router) -@app.get( - "/", - tags=["Health"], -) +@app.get("/", tags=["Health"]) def root() -> dict[str, str]: return { "message": ( @@ -94,12 +139,20 @@ def root() -> dict[str, str]: } -@app.get( - "/health", - tags=["Health"], -) +@app.get("/health", tags=["Health"]) def health_check() -> dict[str, str]: return { "status": "healthy", - "service": "student-service", - } \ No newline at end of file + "service": SERVICE_NAME, + } + + +@app.get( + "/metrics", + include_in_schema=False, +) +def metrics() -> Response: + return Response( + content=generate_latest(), + media_type=CONTENT_TYPE_LATEST, + ) diff --git a/student-service/requirements.txt b/student-service/requirements.txt index feeedd2d..72befaf6 100644 --- a/student-service/requirements.txt +++ b/student-service/requirements.txt @@ -8,4 +8,5 @@ python-multipart==0.0.20 python-dotenv==1.0.1 azure-storage-blob==12.26.0 pytest==8.4.1 -httpx==0.28.1 \ No newline at end of file +httpx==0.28.1 +prometheus-client==0.23.1 diff --git a/user-service/app/main.py b/user-service/app/main.py index 1b4769a6..19c65332 100644 --- a/user-service/app/main.py +++ b/user-service/app/main.py @@ -3,7 +3,13 @@ import time from contextlib import asynccontextmanager -from fastapi import FastAPI +from fastapi import FastAPI, Request, Response +from prometheus_client import ( + CONTENT_TYPE_LATEST, + Counter, + Histogram, + generate_latest, +) from sqlalchemy import select from sqlalchemy.exc import OperationalError from sqlalchemy.orm import Session @@ -21,6 +27,20 @@ logger = logging.getLogger(__name__) +SERVICE_NAME = "user-service" + +HTTP_REQUESTS_TOTAL = Counter( + "koalatech_http_requests_total", + "Total number of HTTP requests.", + ["service", "method", "path", "status_code"], +) + +HTTP_REQUEST_DURATION_SECONDS = Histogram( + "koalatech_http_request_duration_seconds", + "HTTP request duration in seconds.", + ["service", "method", "path"], +) + def initialise_database() -> None: maximum_attempts = 10 @@ -116,6 +136,36 @@ async def lifespan(_: FastAPI): ) +@app.middleware("http") +async def prometheus_metrics( + request: Request, + call_next, +): + start_time = time.perf_counter() + + response = await call_next(request) + + duration = time.perf_counter() - start_time + + route = request.scope.get("route") + path = getattr(route, "path", request.url.path) + + HTTP_REQUESTS_TOTAL.labels( + service=SERVICE_NAME, + method=request.method, + path=path, + status_code=str(response.status_code), + ).inc() + + HTTP_REQUEST_DURATION_SECONDS.labels( + service=SERVICE_NAME, + method=request.method, + path=path, + ).observe(duration) + + return response + + app.include_router(auth.router) app.include_router(users.router) @@ -137,5 +187,16 @@ def root() -> dict[str, str]: def health_check() -> dict[str, str]: return { "status": "healthy", - "service": "user-service", - } \ No newline at end of file + "service": SERVICE_NAME, + } + + +@app.get( + "/metrics", + include_in_schema=False, +) +def metrics() -> Response: + return Response( + content=generate_latest(), + media_type=CONTENT_TYPE_LATEST, + ) diff --git a/user-service/requirements.txt b/user-service/requirements.txt index f07adff6..a52eb85a 100644 --- a/user-service/requirements.txt +++ b/user-service/requirements.txt @@ -9,3 +9,4 @@ python-multipart==0.0.30 pytest==8.4.1 httpx==0.28.1 python-dotenv==1.0.1 +prometheus-client==0.23.1 From 5ca2776a9c6984b5801d039a95bbd040e6c8df0b Mon Sep 17 00:00:00 2001 From: Ashan Indika Date: Fri, 25 Sep 2026 10:12:37 +1000 Subject: [PATCH 09/13] feat: provision Week10 infrastructure through GitHub Actions --- .github/workflows/01-ci.yml | 72 ++++++++++++++++++++++++++++++------- terraform/backend.tf | 3 ++ 2 files changed, 62 insertions(+), 13 deletions(-) create mode 100644 terraform/backend.tf diff --git a/.github/workflows/01-ci.yml b/.github/workflows/01-ci.yml index 9fa43764..ac6fdcba 100644 --- a/.github/workflows/01-ci.yml +++ b/.github/workflows/01-ci.yml @@ -143,15 +143,21 @@ jobs: npm run test -- --run # ========================================================= - # Terraform Validation and Plan + # Terraform Validation, Plan and Provisioning # Task 10.2D - Infrastructure as Code integration + # + # Pull Request: + # Init -> Format -> Validate -> Plan + # + # Main/Manual: + # Init -> Format -> Validate -> Plan -> Apply # ========================================================= terraform-check: - name: Terraform Validation and Plan + name: Terraform Validate, Plan and Apply runs-on: ubuntu-latest # terraform.tfvars is intentionally not committed. - # Non-sensitive Week10 values are supplied using TF_VAR_*. + # Week10 values are supplied through TF_VAR_* variables. env: TF_VAR_resource_group_name: "sit722-week10-rg" TF_VAR_location: "Australia East" @@ -173,10 +179,18 @@ jobs: with: creds: ${{ secrets.AZURE_CREDENTIALS }} + # ------------------------------------------------------- + # Persistent Terraform state + # ------------------------------------------------------- - name: Terraform Init working-directory: terraform run: | - terraform init -input=false + terraform init \ + -input=false \ + -backend-config="resource_group_name=sit722-week10-tfstate-rg" \ + -backend-config="storage_account_name=sit722w10tfstate224848" \ + -backend-config="container_name=tfstate" \ + -backend-config="key=week10-10.2d.tfstate" - name: Terraform Format Check working-directory: terraform @@ -188,12 +202,47 @@ jobs: run: | terraform validate + # ------------------------------------------------------- + # Generate a saved Terraform plan. + # ------------------------------------------------------- - name: Terraform Plan working-directory: terraform run: | terraform plan \ -input=false \ - -no-color + -no-color \ + -out=tfplan + + # ------------------------------------------------------- + # Infrastructure provisioning + # + # Pull requests NEVER provision infrastructure. + # Apply occurs only for main push or controlled + # workflow_dispatch execution. + # ------------------------------------------------------- + - name: Terraform Apply + if: > + github.event_name == 'push' || + github.event_name == 'workflow_dispatch' + working-directory: terraform + run: | + terraform apply \ + -input=false \ + -auto-approve \ + tfplan + + - name: Display Terraform Outputs + if: > + github.event_name == 'push' || + github.event_name == 'workflow_dispatch' + working-directory: terraform + run: | + echo "===== TERRAFORM OUTPUTS =====" + terraform output + + echo + echo "===== TERRAFORM STATE =====" + terraform state list # ========================================================= # Build, Security Scan and Push Docker Images @@ -203,8 +252,9 @@ jobs: name: Build, Scan and Push ${{ matrix.image }} runs-on: ubuntu-latest - # Application tests AND Terraform validation must succeed - # before container images are published. + # Application tests AND Terraform infrastructure + # provisioning/validation must succeed before images + # are published. needs: - backend-test - frontend-test @@ -261,11 +311,7 @@ jobs: ./${{ matrix.service }} # ------------------------------------------------------- - # Docker Scout vulnerability analysis occurs BEFORE push. - # - # The scan is currently evidence/reporting mode while the - # actual Task 10.2D vulnerability is identified and - # remediated. The image will subsequently be rescanned. + # Docker Scout vulnerability analysis BEFORE push # ------------------------------------------------------- - name: Docker Scout CVE scan uses: docker/scout-action@v1 @@ -291,4 +337,4 @@ jobs: - name: Display published image run: | echo "Published image:" - echo "${{ vars.ACR_LOGIN_SERVER }}/${{ matrix.image }}:${{ github.sha }}" + echo "${{ vars.ACR_LOGIN_SERVER }}/${{ matrix.image }}:${{ github.sha }}" \ No newline at end of file diff --git a/terraform/backend.tf b/terraform/backend.tf new file mode 100644 index 00000000..6602f206 --- /dev/null +++ b/terraform/backend.tf @@ -0,0 +1,3 @@ +terraform { + backend "azurerm" {} +} From bdb49cb3622cc2062c22930355b113e374d9f19b Mon Sep 17 00:00:00 2001 From: Ashan Indika Date: Fri, 25 Sep 2026 12:29:31 +1000 Subject: [PATCH 10/13] fix: authenticate Docker Scout in CI pipeline --- .github/workflows/01-ci.yml | 36 +++++++++++++++++++++++++++++++++++- 1 file changed, 35 insertions(+), 1 deletion(-) diff --git a/.github/workflows/01-ci.yml b/.github/workflows/01-ci.yml index ac6fdcba..451df1a2 100644 --- a/.github/workflows/01-ci.yml +++ b/.github/workflows/01-ci.yml @@ -293,16 +293,25 @@ jobs: - name: Checkout repository uses: actions/checkout@v4 + # ------------------------------------------------------- + # Azure authentication + # ------------------------------------------------------- - name: Login to Azure uses: azure/login@v3 with: creds: ${{ secrets.AZURE_CREDENTIALS }} + # ------------------------------------------------------- + # Authenticate with Azure Container Registry + # ------------------------------------------------------- - name: Login to Azure Container Registry run: | az acr login \ --name ${{ vars.ACR_NAME }} + # ------------------------------------------------------- + # Build image locally before security analysis + # ------------------------------------------------------- - name: Build Docker image run: | docker build \ @@ -311,7 +320,24 @@ jobs: ./${{ matrix.service }} # ------------------------------------------------------- - # Docker Scout vulnerability analysis BEFORE push + # Docker Hub authentication for Docker Scout + # + # Required repository secrets: + # DOCKERHUB_USERNAME + # DOCKERHUB_TOKEN + # ------------------------------------------------------- + - name: Login to Docker Hub for Docker Scout + uses: docker/login-action@v4 + with: + username: ${{ secrets.DOCKERHUB_USERNAME }} + password: ${{ secrets.DOCKERHUB_TOKEN }} + + # ------------------------------------------------------- + # Docker Scout vulnerability analysis BEFORE ACR push + # + # exit-code is false so identified vulnerabilities are + # reported without preventing the assessment pipeline + # from continuing. Remediation is demonstrated separately. # ------------------------------------------------------- - name: Docker Scout CVE scan uses: docker/scout-action@v1 @@ -321,14 +347,22 @@ jobs: only-severities: critical,high write-comment: false exit-code: false + github-token: ${{ secrets.GITHUB_TOKEN }} + # ------------------------------------------------------- + # Obtain Docker Scout remediation guidance + # ------------------------------------------------------- - name: Docker Scout remediation recommendations uses: docker/scout-action@v1 with: command: recommendations image: local://${{ vars.ACR_LOGIN_SERVER }}/${{ matrix.image }}:${{ github.sha }} write-comment: false + github-token: ${{ secrets.GITHUB_TOKEN }} + # ------------------------------------------------------- + # Publish security-scanned image to ACR + # ------------------------------------------------------- - name: Push Docker image with commit SHA run: | docker push \ From 4398ccd631070d34109fcf200c0da00bb85424c2 Mon Sep 17 00:00:00 2001 From: Ashan Indika Date: Fri, 25 Sep 2026 14:06:25 +1000 Subject: [PATCH 11/13] fix: use ephemeral storage for enrollment database in staging --- kubernetes/staging/06-enrollment-db.yaml | 18 ++---------------- 1 file changed, 2 insertions(+), 16 deletions(-) diff --git a/kubernetes/staging/06-enrollment-db.yaml b/kubernetes/staging/06-enrollment-db.yaml index 3a94e0ee..557982be 100644 --- a/kubernetes/staging/06-enrollment-db.yaml +++ b/kubernetes/staging/06-enrollment-db.yaml @@ -1,16 +1,3 @@ -apiVersion: v1 -kind: PersistentVolumeClaim -metadata: - name: enrollment-db-pvc - namespace: staging -spec: - accessModes: - - ReadWriteOnce - resources: - requests: - storage: 1Gi - ---- apiVersion: apps/v1 kind: Deployment metadata: @@ -51,8 +38,7 @@ spec: mountPath: /var/lib/postgresql/data volumes: - name: enrollment-db-storage - persistentVolumeClaim: - claimName: enrollment-db-pvc + emptyDir: {} --- apiVersion: v1 @@ -65,4 +51,4 @@ spec: app: enrollment-db ports: - port: 5432 - targetPort: 5432 \ No newline at end of file + targetPort: 5432 From 9bdbc494ae0e91dbf5c5f3394ff8ee6d48aa4e20 Mon Sep 17 00:00:00 2001 From: Ashan Indika Date: Fri, 25 Sep 2026 14:24:14 +1000 Subject: [PATCH 12/13] fix: retry Prometheus target validation during staging deployment --- .github/workflows/02-deploy-staging.yml | 81 +++++++++++++++++-------- 1 file changed, 57 insertions(+), 24 deletions(-) diff --git a/.github/workflows/02-deploy-staging.yml b/.github/workflows/02-deploy-staging.yml index 502d0d63..d0d2dc0c 100644 --- a/.github/workflows/02-deploy-staging.yml +++ b/.github/workflows/02-deploy-staging.yml @@ -174,7 +174,7 @@ jobs: if curl --fail --silent \ http://127.0.0.1:9090/-/ready > /dev/null then - echo "Prometheus is ready." + echo "Prometheus API is ready." break fi @@ -184,16 +184,23 @@ jobs: exit 1 fi + echo "Prometheus API not ready yet - attempt ${attempt}/20" sleep 3 done echo - echo "Prometheus targets:" - curl --fail --silent \ - http://127.0.0.1:9090/api/v1/targets \ - > /tmp/prometheus-targets.json + echo "Waiting for all required Prometheus targets to become healthy..." - python3 - <<'PY' + for attempt in {1..12} + do + echo + echo "Target health check ${attempt}/12" + + curl --fail --silent \ + http://127.0.0.1:9090/api/v1/targets \ + > /tmp/prometheus-targets.json + + if python3 - <<'PYTHON' import json with open("/tmp/prometheus-targets.json") as file: @@ -201,18 +208,6 @@ jobs: targets = data["data"]["activeTargets"] - for target in targets: - labels = target.get("labels", {}) - job = labels.get("job", "unknown") - health = target.get("health", "unknown") - scrape_url = target.get("scrapeUrl", "") - - print( - f"{job:25} " - f"health={health:8} " - f"{scrape_url}" - ) - required_jobs = { "user-service", "student-service", @@ -222,23 +217,61 @@ jobs: "kube-state-metrics", } + target_status = {} + + for target in targets: + labels = target.get("labels", {}) + job = labels.get("job", "unknown") + health = target.get("health", "unknown") + scrape_url = target.get("scrapeUrl", "") + last_error = target.get("lastError", "") + + if job in required_jobs: + target_status[job] = health + + print( + f"{job:25} " + f"health={health:8} " + f"{scrape_url}" + ) + + if last_error: + print(f" error: {last_error}") + healthy_jobs = { - target.get("labels", {}).get("job") - for target in targets - if target.get("health") == "up" + job + for job, health in target_status.items() + if health == "up" } missing = required_jobs - healthy_jobs if missing: - raise SystemExit( - "Required Prometheus targets are not healthy: " + print() + print( + "Targets not healthy yet: " + ", ".join(sorted(missing)) ) + raise SystemExit(1) print() print("All required Prometheus targets are healthy.") - PY + PYTHON + then + echo + echo "Prometheus monitoring validation successful." + exit 0 + fi + + if [ "$attempt" -eq 12 ]; then + echo + echo "Required Prometheus targets did not become healthy within 120 seconds." + exit 1 + fi + + echo "Waiting 10 seconds before retry..." + sleep 10 + done - name: Verify Grafana health shell: bash From 0c981ef6d4c908159fb234ef3981d2b0d733ec7d Mon Sep 17 00:00:00 2001 From: Ashan Indika Date: Fri, 25 Sep 2026 15:32:09 +1000 Subject: [PATCH 13/13] feat: add automated canary deployment validation and recovery --- .github/workflows/05-canary-deployment.yml | 313 ++++++++++++++++++ .../canary/frontend-canary-deployment.yaml | 42 +++ .../canary/frontend-canary-service.yaml | 17 + 3 files changed, 372 insertions(+) create mode 100644 .github/workflows/05-canary-deployment.yml create mode 100644 kubernetes/canary/frontend-canary-deployment.yaml create mode 100644 kubernetes/canary/frontend-canary-service.yaml diff --git a/.github/workflows/05-canary-deployment.yml b/.github/workflows/05-canary-deployment.yml new file mode 100644 index 00000000..a4034b28 --- /dev/null +++ b/.github/workflows/05-canary-deployment.yml @@ -0,0 +1,313 @@ +name: 05 - Canary Deployment and Automatic Rollback + +on: + workflow_dispatch: + inputs: + scenario: + description: "Canary demonstration scenario" + required: true + default: healthy + type: choice + options: + - healthy + - faulty + +permissions: + contents: read + +env: + STAGING_NAMESPACE: staging + CANARY_DEPLOYMENT: frontend-canary + CANARY_SERVICE: frontend-canary + STABLE_DEPLOYMENT: frontend + +jobs: + canary-deployment: + name: Canary Deployment, Validation and Recovery + runs-on: ubuntu-latest + + steps: + - name: Checkout repository + uses: actions/checkout@v4 + + - name: Azure login + uses: azure/login@v3 + with: + creds: ${{ secrets.AZURE_CREDENTIALS }} + + - name: Get AKS credentials + shell: bash + run: | + az aks get-credentials \ + --resource-group "${{ vars.AKS_RESOURCE_GROUP }}" \ + --name "${{ vars.AKS_CLUSTER_NAME }}" \ + --overwrite-existing + + kubectl get nodes + + - name: Capture current stable release + id: stable + shell: bash + run: | + STABLE_IMAGE=$(kubectl get deployment "${STABLE_DEPLOYMENT}" \ + -n "${STAGING_NAMESPACE}" \ + -o jsonpath='{.spec.template.spec.containers[0].image}') + + echo "Current stable image: ${STABLE_IMAGE}" + echo "image=${STABLE_IMAGE}" >> "$GITHUB_OUTPUT" + + - name: Verify stable application before canary + shell: bash + run: | + kubectl rollout status deployment/"${STABLE_DEPLOYMENT}" \ + -n "${STAGING_NAMESPACE}" \ + --timeout=180s + + STABLE_IP=$(kubectl get service frontend \ + -n "${STAGING_NAMESPACE}" \ + -o jsonpath='{.status.loadBalancer.ingress[0].ip}') + + if [ -z "${STABLE_IP}" ]; then + STABLE_IP=$(kubectl get service frontend \ + -n "${STAGING_NAMESPACE}" \ + -o jsonpath='{.status.loadBalancer.ingress[0].hostname}') + fi + + echo "Stable frontend address: ${STABLE_IP}" + + curl --fail \ + --silent \ + --show-error \ + --retry 10 \ + --retry-delay 5 \ + --retry-all-errors \ + "http://${STABLE_IP}/" \ + > /dev/null + + echo "Stable application is healthy before canary deployment." + + - name: Select canary image + id: candidate + shell: bash + run: | + if [ "${{ inputs.scenario }}" = "healthy" ]; then + CANARY_IMAGE="${{ steps.stable.outputs.image }}" + echo "Healthy scenario selected." + echo "Candidate image: ${CANARY_IMAGE}" + else + CANARY_IMAGE="nginx:1.27-alpine" + echo "Faulty scenario selected." + echo "The canary will be configured with an invalid health endpoint." + echo "Candidate image: ${CANARY_IMAGE}" + fi + + echo "image=${CANARY_IMAGE}" >> "$GITHUB_OUTPUT" + + - name: Deploy canary candidate + shell: bash + run: | + sed \ + "s|CANARY_IMAGE|${{ steps.candidate.outputs.image }}|g" \ + kubernetes/canary/frontend-canary-deployment.yaml \ + > /tmp/frontend-canary-deployment.yaml + + if [ "${{ inputs.scenario }}" = "faulty" ]; then + echo "Injecting deliberately invalid health endpoint for rollback demonstration." + + sed -i \ + 's|path: /$|path: /deliberately-faulty-health-endpoint|g' \ + /tmp/frontend-canary-deployment.yaml + fi + + kubectl apply \ + -f /tmp/frontend-canary-deployment.yaml + + kubectl apply \ + -f kubernetes/canary/frontend-canary-service.yaml + + echo + echo "Stable and canary deployments:" + kubectl get deployments \ + -n "${STAGING_NAMESPACE}" \ + -l app=frontend \ + -o wide + + - name: Wait for canary rollout + id: rollout + continue-on-error: true + shell: bash + run: | + kubectl rollout status deployment/"${CANARY_DEPLOYMENT}" \ + -n "${STAGING_NAMESPACE}" \ + --timeout=120s + + - name: Validate canary release + id: validation + if: steps.rollout.outcome == 'success' + continue-on-error: true + shell: bash + run: | + echo "Starting automated canary health validation..." + + kubectl run canary-health-check \ + --rm \ + -i \ + --restart=Never \ + --image=curlimages/curl:8.12.1 \ + -n "${STAGING_NAMESPACE}" \ + -- \ + curl \ + --fail \ + --silent \ + --show-error \ + --max-time 10 \ + "http://${CANARY_SERVICE}/" \ + > /dev/null + + echo "Canary HTTP health validation passed." + + echo "Canary candidate passed all validation checks." + + - name: Promote healthy canary + if: >- + inputs.scenario == 'healthy' && + steps.rollout.outcome == 'success' && + steps.validation.outcome == 'success' + shell: bash + run: | + echo "Promoting validated canary image to stable deployment..." + + kubectl set image deployment/"${STABLE_DEPLOYMENT}" \ + frontend="${{ steps.candidate.outputs.image }}" \ + -n "${STAGING_NAMESPACE}" + + kubectl rollout status deployment/"${STABLE_DEPLOYMENT}" \ + -n "${STAGING_NAMESPACE}" \ + --timeout=180s + + echo + echo "Canary promotion completed successfully." + + - name: Automatic rollback and recovery + if: >- + inputs.scenario == 'faulty' || + steps.rollout.outcome == 'failure' || + steps.validation.outcome == 'failure' + shell: bash + run: | + echo "Canary validation failed." + echo "Automatic recovery has been triggered." + echo + echo "Stable deployment was never replaced." + echo "Removing failed canary workload..." + + kubectl delete deployment "${CANARY_DEPLOYMENT}" \ + -n "${STAGING_NAMESPACE}" \ + --ignore-not-found=true + + kubectl delete service "${CANARY_SERVICE}" \ + -n "${STAGING_NAMESPACE}" \ + --ignore-not-found=true + + echo + echo "Verifying original stable deployment..." + + kubectl rollout status deployment/"${STABLE_DEPLOYMENT}" \ + -n "${STAGING_NAMESPACE}" \ + --timeout=180s + + CURRENT_IMAGE=$(kubectl get deployment "${STABLE_DEPLOYMENT}" \ + -n "${STAGING_NAMESPACE}" \ + -o jsonpath='{.spec.template.spec.containers[0].image}') + + echo "Original stable image : ${{ steps.stable.outputs.image }}" + echo "Current stable image : ${CURRENT_IMAGE}" + + if [ "${CURRENT_IMAGE}" != "${{ steps.stable.outputs.image }}" ]; then + echo "Stable image changed unexpectedly." + exit 1 + fi + + STABLE_IP=$(kubectl get service frontend \ + -n "${STAGING_NAMESPACE}" \ + -o jsonpath='{.status.loadBalancer.ingress[0].ip}') + + if [ -z "${STABLE_IP}" ]; then + STABLE_IP=$(kubectl get service frontend \ + -n "${STAGING_NAMESPACE}" \ + -o jsonpath='{.status.loadBalancer.ingress[0].hostname}') + fi + + curl --fail \ + --silent \ + --show-error \ + --retry 10 \ + --retry-delay 5 \ + --retry-all-errors \ + "http://${STABLE_IP}/" \ + > /dev/null + + echo + echo "AUTOMATIC RECOVERY SUCCESSFUL." + echo "Failed canary removed." + echo "Original stable release remains healthy." + + - name: Remove successful canary + if: >- + inputs.scenario == 'healthy' && + steps.rollout.outcome == 'success' && + steps.validation.outcome == 'success' + shell: bash + run: | + echo "Removing temporary canary resources after promotion..." + + kubectl delete deployment "${CANARY_DEPLOYMENT}" \ + -n "${STAGING_NAMESPACE}" \ + --ignore-not-found=true + + kubectl delete service "${CANARY_SERVICE}" \ + -n "${STAGING_NAMESPACE}" \ + --ignore-not-found=true + + - name: Final deployment evidence + if: always() + shell: bash + run: | + echo "============================================" + echo "FINAL CANARY DEPLOYMENT STATE" + echo "============================================" + + kubectl get deployments \ + -n "${STAGING_NAMESPACE}" \ + -o wide + + echo + kubectl get pods \ + -n "${STAGING_NAMESPACE}" \ + -o wide + + echo + kubectl get services \ + -n "${STAGING_NAMESPACE}" + + - name: Final scenario result + if: always() + shell: bash + run: | + if [ "${{ inputs.scenario }}" = "healthy" ]; then + if [ "${{ steps.rollout.outcome }}" != "success" ] || \ + [ "${{ steps.validation.outcome }}" != "success" ]; then + echo "Healthy canary scenario failed." + exit 1 + fi + + echo "============================================" + echo "HEALTHY CANARY SUCCESSFULLY PROMOTED" + echo "============================================" + else + echo "============================================" + echo "FAULTY CANARY REJECTED" + echo "STABLE RELEASE RETAINED" + echo "AUTOMATIC RECOVERY SUCCESSFUL" + echo "============================================" + fi diff --git a/kubernetes/canary/frontend-canary-deployment.yaml b/kubernetes/canary/frontend-canary-deployment.yaml new file mode 100644 index 00000000..d050c19c --- /dev/null +++ b/kubernetes/canary/frontend-canary-deployment.yaml @@ -0,0 +1,42 @@ +apiVersion: apps/v1 +kind: Deployment +metadata: + name: frontend-canary + namespace: staging + labels: + app: frontend + track: canary +spec: + replicas: 1 + selector: + matchLabels: + app: frontend + track: canary + template: + metadata: + labels: + app: frontend + track: canary + spec: + containers: + - name: frontend + image: CANARY_IMAGE + imagePullPolicy: Always + ports: + - containerPort: 80 + readinessProbe: + httpGet: + path: / + port: 80 + initialDelaySeconds: 5 + periodSeconds: 5 + timeoutSeconds: 3 + failureThreshold: 6 + livenessProbe: + httpGet: + path: / + port: 80 + initialDelaySeconds: 10 + periodSeconds: 10 + timeoutSeconds: 3 + failureThreshold: 3 diff --git a/kubernetes/canary/frontend-canary-service.yaml b/kubernetes/canary/frontend-canary-service.yaml new file mode 100644 index 00000000..3b9237fd --- /dev/null +++ b/kubernetes/canary/frontend-canary-service.yaml @@ -0,0 +1,17 @@ +apiVersion: v1 +kind: Service +metadata: + name: frontend-canary + namespace: staging + labels: + app: frontend + track: canary +spec: + type: ClusterIP + selector: + app: frontend + track: canary + ports: + - name: http + port: 80 + targetPort: 80