diff --git a/.deploystack/docker-run.txt b/.deploystack/docker-run.txt
deleted file mode 100644
index 65c2310..0000000
--- a/.deploystack/docker-run.txt
+++ /dev/null
@@ -1 +0,0 @@
-docker run --rm -p 8000:8000 darrenofficial/dpaste:latest
diff --git a/.dockerignore b/.dockerignore
index 4ed9321..9a14336 100644
--- a/.dockerignore
+++ b/.dockerignore
@@ -14,7 +14,6 @@ venv
Dockerfile*
docker-compose*
.dockerignore
-.travis.yml
.gitattributes
docs
terraform
@@ -22,4 +21,7 @@ monitoring
scripts
*.md
!setup.cfg
-!README.md
\ No newline at end of file
+!README.md.hadolint.yaml
+.gitleaksignore
+.trivyignore
+ruff.toml
diff --git a/.github/ISSUE_TEMPLATE/bug_report.md b/.github/ISSUE_TEMPLATE/bug_report.md
deleted file mode 100644
index 9cb6ab4..0000000
--- a/.github/ISSUE_TEMPLATE/bug_report.md
+++ /dev/null
@@ -1,27 +0,0 @@
----
-name: โ ๏ธ Bug
-about: Report a problem or defect
----
-
-## Description
-
-
-
-## Environment
-
-URL:
-
-
-
-## Steps to reproduce
-
-1.
-2.
-3. ...
-
-## Expected result
-
-
-## Actual result
-
-
diff --git a/.github/ISSUE_TEMPLATE/feature.md b/.github/ISSUE_TEMPLATE/feature.md
deleted file mode 100644
index 66a5d1f..0000000
--- a/.github/ISSUE_TEMPLATE/feature.md
+++ /dev/null
@@ -1,12 +0,0 @@
----
-name: โจ Feature
-about: Request new functionality or modifications to existing functionality
----
-
-## Description
-
-## Acceptance Criteria
-
-
-
-
diff --git a/.github/ISSUE_TEMPLATE/task.md b/.github/ISSUE_TEMPLATE/task.md
deleted file mode 100644
index 6ed1d24..0000000
--- a/.github/ISSUE_TEMPLATE/task.md
+++ /dev/null
@@ -1,8 +0,0 @@
----
-name: ๐ท Task
-about: Behind-the-scenes work (upgrades, technical debt cleanup, etc.)
----
-
-## Description
-
-## Reason for task
diff --git a/.github/dependabot.yml b/.github/dependabot.yml
index 91abb11..2c60f1a 100644
--- a/.github/dependabot.yml
+++ b/.github/dependabot.yml
@@ -1,11 +1,31 @@
-# To get started with Dependabot version updates, you'll need to specify which
-# package ecosystems to update and where the package manifests are located.
-# Please see the documentation for all configuration options:
-# https://docs.github.com/github/administering-a-repository/configuration-options-for-dependency-updates
-
+# Weekly dependency update PRs. Every PR runs the full CI (tests, lint, image
+# build + smoke test + Trivy) before it can be merged.
version: 2
updates:
- - package-ecosystem: "pip" # See documentation for possible values
- directory: "/" # Location of package manifests
+ # Keeps the SHA-pinned actions in .github/workflows current
+ - package-ecosystem: github-actions
+ directory: /
+ schedule:
+ interval: weekly
+ groups:
+ github-actions:
+ patterns: ["*"]
+
+ # Base images in Dockerfile.hardened
+ - package-ecosystem: docker
+ directory: /
+ schedule:
+ interval: weekly
+
+ # AWS / random provider versions (terraform/.terraform.lock.hcl)
+ - package-ecosystem: terraform
+ directory: /terraform
+ schedule:
+ interval: weekly
+
+ # Python dependencies of the inherited dpaste application
+ - package-ecosystem: pip
+ directory: /
schedule:
- interval: "weekly"
+ interval: weekly
+ open-pull-requests-limit: 5
diff --git a/.github/workflows/cd.yml b/.github/workflows/cd.yml
index ef39b2b..7f4e2da 100644
--- a/.github/workflows/cd.yml
+++ b/.github/workflows/cd.yml
@@ -2,7 +2,9 @@
# CloudPulse - CD pipeline
# Build -> Trivy gate -> push to ECR -> deploy to EC2 via SSM Run Command
# -> HTTPS smoke test. Runs on merges to master that change the app,
-# the image or the deploy script; can also be started manually.
+# the image or the on-instance scripts; can also be started manually.
+# AWS credentials are only configured after the image has passed the scan,
+# so the build and third-party scanner never run with cloud access.
# ============================================================
name: CD
@@ -18,6 +20,7 @@ on:
- "package.json"
- "package-lock.json"
- "scripts/deploy.sh"
+ - "scripts/backup.sh"
- ".github/workflows/cd.yml"
workflow_dispatch:
@@ -38,33 +41,22 @@ jobs:
build-scan-push:
name: Build, scan, push
runs-on: ubuntu-latest
+ timeout-minutes: 20
outputs:
image_tag: ${{ steps.meta.outputs.tag }}
steps:
- - uses: actions/checkout@v4
+ - uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4.4.0
- name: Image tag = short commit SHA
id: meta
run: echo "tag=${GITHUB_SHA::7}" >> "$GITHUB_OUTPUT"
- # AWS Academy session credentials (OIDC is blocked in the Learner Lab).
- # They expire with the lab session; refreshed by scripts/refresh-github-aws-secrets.ps1
- - name: Configure AWS credentials
- uses: aws-actions/configure-aws-credentials@v4
- with:
- aws-access-key-id: ${{ secrets.AWS_ACCESS_KEY_ID }}
- aws-secret-access-key: ${{ secrets.AWS_SECRET_ACCESS_KEY }}
- aws-session-token: ${{ secrets.AWS_SESSION_TOKEN }}
- aws-region: ${{ env.AWS_REGION }}
-
- - name: Log in to Amazon ECR
- id: ecr
- uses: aws-actions/amazon-ecr-login@v2
-
- - uses: docker/setup-buildx-action@v3
+ - uses: docker/setup-buildx-action@8d2750c68a42422c14e847fe6c8ac0403b4cbd6f # v3.12.0
+ # provenance/sbom off: an attestation turns the push into an image index,
+ # which ECR basic scanning cannot scan
- name: Build image
- uses: docker/build-push-action@v6
+ uses: docker/build-push-action@10e90e3645eae34f1e60eeb005ba3a3d33f178e8 # v6.19.2
with:
context: .
file: Dockerfile.hardened
@@ -72,41 +64,59 @@ jobs:
push: false
provenance: false
sbom: false
- tags: ${{ steps.ecr.outputs.registry }}/${{ env.ECR_REPOSITORY }}:${{ steps.meta.outputs.tag }}
+ tags: ${{ env.ECR_REPOSITORY }}:${{ steps.meta.outputs.tag }}
# Same policy as CI: fail on fixable CRITICAL/HIGH. Nothing is pushed if this fails.
- name: Trivy scan (release gate)
- uses: aquasecurity/trivy-action@master
+ uses: aquasecurity/trivy-action@ed142fd0673e97e23eac54620cfb913e5ce36c25 # v0.36.0
with:
- image-ref: ${{ steps.ecr.outputs.registry }}/${{ env.ECR_REPOSITORY }}:${{ steps.meta.outputs.tag }}
+ image-ref: ${{ env.ECR_REPOSITORY }}:${{ steps.meta.outputs.tag }}
format: table
exit-code: "1"
severity: CRITICAL,HIGH
ignore-unfixed: true
trivyignores: .trivyignore
+ # AWS Academy session credentials (OIDC is blocked in the Learner Lab).
+ # They expire with the lab session; refreshed by scripts/refresh-github-aws-secrets.ps1
+ - name: Configure AWS credentials
+ uses: aws-actions/configure-aws-credentials@7474bc4690e29a8392af63c5b98e7449536d5c3a # v4.3.1
+ with:
+ aws-access-key-id: ${{ secrets.AWS_ACCESS_KEY_ID }}
+ aws-secret-access-key: ${{ secrets.AWS_SECRET_ACCESS_KEY }}
+ aws-session-token: ${{ secrets.AWS_SESSION_TOKEN }}
+ aws-region: ${{ env.AWS_REGION }}
+
+ - name: Log in to Amazon ECR
+ id: ecr
+ uses: aws-actions/amazon-ecr-login@03f1aad4c6c7ffd436567f42f9384779290529bd # v2.1.7
+
# ECR tags are immutable: a re-run for the same commit must not fail on push
- name: Push image to ECR
+ env:
+ REGISTRY: ${{ steps.ecr.outputs.registry }}
+ TAG: ${{ steps.meta.outputs.tag }}
run: |
- TAG="${{ steps.meta.outputs.tag }}"
if aws ecr describe-images --repository-name "$ECR_REPOSITORY" --image-ids imageTag="$TAG" >/dev/null 2>&1; then
echo "Tag $TAG already exists in ECR (immutable) - skipping push"
else
- docker push "${{ steps.ecr.outputs.registry }}/${{ env.ECR_REPOSITORY }}:$TAG"
+ docker tag "$ECR_REPOSITORY:$TAG" "$REGISTRY/$ECR_REPOSITORY:$TAG"
+ docker push "$REGISTRY/$ECR_REPOSITORY:$TAG"
fi
deploy:
name: Deploy to EC2 via SSM
needs: build-scan-push
runs-on: ubuntu-latest
+ timeout-minutes: 15
environment:
name: production
url: ${{ steps.deploy.outputs.app_url }}
steps:
- - uses: actions/checkout@v4
+ - uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4.4.0
- name: Configure AWS credentials
- uses: aws-actions/configure-aws-credentials@v4
+ uses: aws-actions/configure-aws-credentials@7474bc4690e29a8392af63c5b98e7449536d5c3a # v4.3.1
with:
aws-access-key-id: ${{ secrets.AWS_ACCESS_KEY_ID }}
aws-secret-access-key: ${{ secrets.AWS_SECRET_ACCESS_KEY }}
@@ -129,13 +139,17 @@ jobs:
--query "Reservations[0].Instances[0].PublicIpAddress" --output text)
echo "Deploying tag ${IMAGE_TAG} to ${INSTANCE_ID} (${PUBLIC_IP})"
- # Ship this commit's deploy.sh with the command, so the instance
- # always runs the reviewed version from Git.
- SCRIPT_B64=$(base64 -w0 scripts/deploy.sh)
- jq -n --arg b64 "$SCRIPT_B64" --arg tag "$IMAGE_TAG" '{commands: [
+ # Ship this commit's deploy.sh and backup.sh with the command, so the
+ # instance always runs the reviewed versions from Git (gzip keeps the
+ # SSM parameter small).
+ DEPLOY_B64=$(gzip -9c scripts/deploy.sh | base64 -w0)
+ BACKUP_B64=$(gzip -9c scripts/backup.sh | base64 -w0)
+ jq -n --arg deploy "$DEPLOY_B64" --arg backup "$BACKUP_B64" --arg tag "$IMAGE_TAG" '{commands: [
"set -e",
- "echo \($b64) | base64 -d > /opt/cloudpulse/deploy.sh",
- "chmod 0755 /opt/cloudpulse/deploy.sh",
+ "mkdir -p /opt/cloudpulse",
+ "echo \($deploy) | base64 -d | gunzip > /opt/cloudpulse/deploy.sh",
+ "echo \($backup) | base64 -d | gunzip > /opt/cloudpulse/backup.sh",
+ "chmod 0755 /opt/cloudpulse/deploy.sh /opt/cloudpulse/backup.sh",
"/opt/cloudpulse/deploy.sh \($tag)"
]}' > ssm-params.json
@@ -174,4 +188,4 @@ jobs:
APP_URL: ${{ steps.deploy.outputs.app_url }}
run: |
curl -sS --fail --retry 10 --retry-delay 5 --retry-all-errors \
- -o /dev/null -w "GET / -> HTTP %{http_code} in %{time_total}s\n" "${APP_URL}/"
\ No newline at end of file
+ -o /dev/null -w "GET / -> HTTP %{http_code} in %{time_total}s\n" "${APP_URL}/"
diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml
index fc56319..dda76a6 100644
--- a/.github/workflows/ci.yml
+++ b/.github/workflows/ci.yml
@@ -1,3 +1,14 @@
+# ============================================================
+# CloudPulse - CI
+# Every pull request and push to master:
+# test pytest on the same Python version as the runtime image
+# lint ruff (blocking)
+# docker-build-scan hadolint -> build -> container smoke test -> Trivy
+# secret-scan gitleaks over the full Git history
+# test, lint and docker-build-scan are required checks in branch protection:
+# do not rename these job IDs.
+# Third-party actions are pinned to full commit SHAs (Dependabot keeps them current).
+# ============================================================
name: CI
on:
@@ -6,15 +17,25 @@ on:
pull_request:
branches: [master]
+permissions:
+ contents: read
+
+concurrency:
+ group: ci-${{ github.ref }}
+ cancel-in-progress: ${{ github.event_name == 'pull_request' }}
+
jobs:
test:
runs-on: ubuntu-latest
+ timeout-minutes: 15
steps:
- - uses: actions/checkout@v4
+ - uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4.4.0
- - uses: actions/setup-python@v5
+ - uses: actions/setup-python@a26af69be951a213d495a4c3e4e4022e16d87065 # v5.6.0
with:
- python-version: "3.10"
+ python-version: "3.10" # same as the runtime image (python:3.10-slim)
+ cache: pip
+ cache-dependency-path: setup.cfg
- name: Install dependencies
run: pip install -e ".[dev]"
@@ -24,43 +45,96 @@ jobs:
lint:
runs-on: ubuntu-latest
+ timeout-minutes: 10
steps:
- - uses: actions/checkout@v4
+ - uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4.4.0
- - uses: actions/setup-python@v5
+ - uses: actions/setup-python@a26af69be951a213d495a4c3e4e4022e16d87065 # v5.6.0
with:
python-version: "3.10"
- name: Install ruff
- run: pip install ruff
+ run: pip install ruff==0.15.11
+ # Rules and the one ignored upstream finding are in ruff.toml
- name: Run linter
run: ruff check dpaste/
- continue-on-error: true
docker-build-scan:
runs-on: ubuntu-latest
+ timeout-minutes: 20
+ env:
+ IMAGE: cloudpulse:ci-${{ github.sha }}
steps:
- - uses: actions/checkout@v4
+ - uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4.4.0
+
+ # Binary download verified against its published SHA-256
+ - name: Lint Dockerfile (hadolint)
+ run: |
+ curl -sSfL -o /tmp/hadolint \
+ https://github.com/hadolint/hadolint/releases/download/v2.14.0/hadolint-linux-x86_64
+ echo "6bf226944684f56c84dd014e8b979d27425c0148f61b3bd99bcc6f39e9dc5a47 /tmp/hadolint" | sha256sum -c -
+ chmod +x /tmp/hadolint
+ /tmp/hadolint Dockerfile.hardened
- name: Set up Docker Buildx
- uses: docker/setup-buildx-action@v3
+ uses: docker/setup-buildx-action@8d2750c68a42422c14e847fe6c8ac0403b4cbd6f # v3.12.0
- name: Build hardened image
- uses: docker/build-push-action@v6
+ uses: docker/build-push-action@10e90e3645eae34f1e60eeb005ba3a3d33f178e8 # v6.19.2
with:
context: .
file: Dockerfile.hardened
push: false
load: true
- tags: cloudpulse:ci-${{ github.sha }}
+ tags: ${{ env.IMAGE }}
+
+ # Run the image the way production does: non-root, DEBUG off, health check green,
+ # page served, API accepts a snippet.
+ - name: Container smoke test
+ run: |
+ docker run -d --name smoke -p 8000:8000 \
+ -e SECRET_KEY=ci-smoke-test-only -e DEBUG=False -e ALLOWED_HOSTS=localhost \
+ "$IMAGE"
+ status=starting
+ for _ in $(seq 1 30); do
+ status=$(docker inspect -f '{{.State.Health.Status}}' smoke)
+ if [ "$status" = healthy ] || [ "$status" = unhealthy ]; then break; fi
+ sleep 3
+ done
+ echo "Health status: $status"
+ if [ "$status" != healthy ]; then docker logs smoke; exit 1; fi
+ uid=$(docker exec smoke id -u)
+ echo "Container user id: $uid"
+ if [ "$uid" = "0" ]; then echo "::error::container runs as root"; exit 1; fi
+ curl -sSf -o /dev/null -w "GET / -> HTTP %{http_code}\n" http://localhost:8000/
+ curl -sSf -X POST -d "content=ci smoke test&lexer=_text" http://localhost:8000/api/
+ echo
+ docker rm -f smoke >/dev/null
- name: Run Trivy vulnerability scanner
- uses: aquasecurity/trivy-action@master
+ uses: aquasecurity/trivy-action@ed142fd0673e97e23eac54620cfb913e5ce36c25 # v0.36.0
with:
- image-ref: cloudpulse:ci-${{ github.sha }}
+ image-ref: ${{ env.IMAGE }}
format: table
- exit-code: 1
+ exit-code: "1"
severity: CRITICAL,HIGH
ignore-unfixed: true
- trivyignores: .trivyignore
\ No newline at end of file
+ trivyignores: .trivyignore
+
+ secret-scan:
+ runs-on: ubuntu-latest
+ timeout-minutes: 10
+ steps:
+ - uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4.4.0
+ with:
+ fetch-depth: 0 # full history: a secret deleted later is still leaked
+
+ # Accepted historical findings are listed in .gitleaksignore
+ - name: gitleaks
+ run: |
+ curl -sSfL -o /tmp/gitleaks.tar.gz \
+ https://github.com/gitleaks/gitleaks/releases/download/v8.30.1/gitleaks_8.30.1_linux_x64.tar.gz
+ echo "551f6fc83ea457d62a0d98237cbad105af8d557003051f41f3e7ca7b3f2470eb /tmp/gitleaks.tar.gz" | sha256sum -c -
+ tar -xzf /tmp/gitleaks.tar.gz -C /tmp gitleaks
+ /tmp/gitleaks git --no-banner --redact --verbose .
diff --git a/.github/workflows/docker.yml.disabled b/.github/workflows/docker.yml.disabled
deleted file mode 100644
index a804cdb..0000000
--- a/.github/workflows/docker.yml.disabled
+++ /dev/null
@@ -1,64 +0,0 @@
-name: Docker CICD
-
-# Only run when theres something on master
-on:
- push:
- branches:
- - 'master'
-
-jobs:
- docker:
- runs-on: ubuntu-latest
- steps:
- - name: Checkout
- uses: actions/checkout@v2
-
- - name: Prepare
- id: prep
- run: |
- DOCKER_IMAGE=${{ secrets.DOCKER_USERNAME }}/${GITHUB_REPOSITORY#*/}
- VERSION=latest
- SHORTREF=${GITHUB_SHA::8}
-
- # If this is git tag, use the tag name as a docker tag
- if [[ $GITHUB_REF == refs/tags/* ]]; then
- VERSION=${GITHUB_REF#refs/tags/v}
- fi
- TAGS="${DOCKER_IMAGE}:${VERSION},${DOCKER_IMAGE}:${SHORTREF}"
-
- # If the VERSION looks like a version number, assume that
- # this is the most recent version of the image and also
- # tag it 'latest'.
- if [[ $VERSION =~ ^[0-9]{1,3}\.[0-9]{1,3}\.[0-9]{1,3}$ ]]; then
- TAGS="$TAGS,${DOCKER_IMAGE}:latest"
- fi
-
- # Set output parameters.
- echo ::set-output name=tags::${TAGS}
- echo ::set-output name=docker_image::${DOCKER_IMAGE}
-
- - name: Set up QEMU
- uses: docker/setup-qemu-action@master
- with:
- platforms: all
-
- - name: Set up Docker Buildx
- id: buildx
- uses: docker/setup-buildx-action@master
-
- - name: Login to DockerHub
- if: github.event_name != 'pull_request'
- uses: docker/login-action@v2
- with:
- username: ${{ secrets.DOCKER_USERNAME }}
- password: ${{ secrets.DOCKER_PASSWORD }}
-
- - name: Build
- uses: docker/build-push-action@v3
- with:
- builder: ${{ steps.buildx.outputs.name }}
- context: .
- file: ./Dockerfile
- platforms: linux/amd64,linux/arm64,linux/ppc64le
- push: true
- tags: ${{ steps.prep.outputs.tags }}
diff --git a/.github/workflows/terraform.yml b/.github/workflows/terraform.yml
index 2f38b97..58df7c8 100644
--- a/.github/workflows/terraform.yml
+++ b/.github/workflows/terraform.yml
@@ -1,6 +1,6 @@
# ============================================================
# CloudPulse - Terraform pipeline
-# PR: fmt -> validate -> Checkov IaC scan -> plan (shown in the run summary)
+# PR: fmt -> validate -> TFLint -> Checkov IaC scan -> plan (shown in the run summary)
# Merge: plan -> MANUAL APPROVAL (environment "infrastructure") -> apply the exact saved plan
# ============================================================
name: Terraform
@@ -42,9 +42,9 @@ jobs:
name: fmt, validate, Checkov
runs-on: ubuntu-latest
steps:
- - uses: actions/checkout@v4
+ - uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4.4.0
- - uses: hashicorp/setup-terraform@v3
+ - uses: hashicorp/setup-terraform@b9cd54a3c349d3f38e8881555d616ced269862dd # v3.1.2
with:
terraform_version: ${{ env.TF_VERSION }}
terraform_wrapper: false
@@ -57,9 +57,19 @@ jobs:
terraform init -backend=false
terraform validate
+ # Catches what validate does not: unused variables/outputs, missing version
+ # constraints, deprecated syntax. Binary verified against its published SHA-256.
+ - name: TFLint
+ run: |
+ curl -sSfL -o /tmp/tflint.zip \
+ https://github.com/terraform-linters/tflint/releases/download/v0.64.0/tflint_linux_amd64.zip
+ echo "cca9d13e2e1d7a2c627af60ff899a3c9b74212899416aeb96ec764d2ef954537 /tmp/tflint.zip" | sha256sum -c -
+ unzip -q /tmp/tflint.zip -d /tmp
+ /tmp/tflint --recursive --format compact
+
# Fails on any finding that is neither fixed nor skipped with a justification in the code
- name: Checkov IaC security scan
- uses: bridgecrewio/checkov-action@v12
+ uses: bridgecrewio/checkov-action@444c9db6fa75e2d9c19ebf1fde7322089be9009e # v12.3125.0
with:
directory: terraform
framework: terraform
@@ -72,16 +82,16 @@ jobs:
outputs:
has_changes: ${{ steps.plan.outputs.has_changes }}
steps:
- - uses: actions/checkout@v4
+ - uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4.4.0
- - uses: hashicorp/setup-terraform@v3
+ - uses: hashicorp/setup-terraform@b9cd54a3c349d3f38e8881555d616ced269862dd # v3.1.2
with:
terraform_version: ${{ env.TF_VERSION }}
terraform_wrapper: false
# AWS Academy session credentials (OIDC is blocked in the Learner Lab)
- name: Configure AWS credentials
- uses: aws-actions/configure-aws-credentials@v4
+ uses: aws-actions/configure-aws-credentials@7474bc4690e29a8392af63c5b98e7449536d5c3a # v4.3.1
with:
aws-access-key-id: ${{ secrets.AWS_ACCESS_KEY_ID }}
aws-secret-access-key: ${{ secrets.AWS_SECRET_ACCESS_KEY }}
@@ -120,7 +130,7 @@ jobs:
# The apply job applies exactly this reviewed plan (short retention: plan files can contain sensitive values)
- name: Upload plan
if: github.event_name != 'pull_request' && steps.plan.outputs.has_changes == 'true'
- uses: actions/upload-artifact@v4
+ uses: actions/upload-artifact@ea165f8d65b6e75b540449e92b4886f43607fa02 # v4.6.2
with:
name: tfplan
path: terraform/tfplan
@@ -133,15 +143,15 @@ jobs:
runs-on: ubuntu-latest
environment: infrastructure
steps:
- - uses: actions/checkout@v4
+ - uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4.4.0
- - uses: hashicorp/setup-terraform@v3
+ - uses: hashicorp/setup-terraform@b9cd54a3c349d3f38e8881555d616ced269862dd # v3.1.2
with:
terraform_version: ${{ env.TF_VERSION }}
terraform_wrapper: false
- name: Configure AWS credentials
- uses: aws-actions/configure-aws-credentials@v4
+ uses: aws-actions/configure-aws-credentials@7474bc4690e29a8392af63c5b98e7449536d5c3a # v4.3.1
with:
aws-access-key-id: ${{ secrets.AWS_ACCESS_KEY_ID }}
aws-secret-access-key: ${{ secrets.AWS_SECRET_ACCESS_KEY }}
@@ -152,7 +162,7 @@ jobs:
run: terraform init
- name: Download reviewed plan
- uses: actions/download-artifact@v4
+ uses: actions/download-artifact@d3f86a106a0bac45b974a628896c90dbdf5c8093 # v4.3.0
with:
name: tfplan
path: terraform
diff --git a/.gitignore b/.gitignore
index bd967a6..e8d37ae 100644
--- a/.gitignore
+++ b/.gitignore
@@ -9,3 +9,9 @@ node_modules
**/__pycache__/
# Generated monitoring target (changes with the EC2 IP)
monitoring/prometheus/targets.d/*.yml
+# Local secrets and generated files that must never be committed
+.env
+.env.*
+*.pem
+*.tfplan
+ssm-params.json
diff --git a/.gitleaksignore b/.gitleaksignore
new file mode 100644
index 0000000..cecb8e4
--- /dev/null
+++ b/.gitleaksignore
@@ -0,0 +1,5 @@
+# gitleaks allowlist (fingerprint = commit:file:rule:line)
+# Upstream dpaste commit from 2013: a Coveralls repo token in .coveralls.yml.
+# The file has since been deleted upstream; the token belonged to the original
+# dpaste project's Coveralls account, not to this repository.
+902efaed97a0b5107aaefcf6859e5beb6b9e9a7a:.coveralls.yml:generic-api-key:2
diff --git a/.hadolint.yaml b/.hadolint.yaml
new file mode 100644
index 0000000..c54bc5c
--- /dev/null
+++ b/.hadolint.yaml
@@ -0,0 +1,9 @@
+# hadolint configuration (used by the docker-build-scan CI job)
+failure-threshold: warning
+ignored:
+ # Pinning apt package versions breaks the build as soon as Debian publishes a
+ # security update (old versions leave the mirror). Instead, images are rebuilt
+ # on every change and Trivy blocks any fixable CRITICAL/HIGH vulnerability.
+ - DL3008
+ # pip itself is upgraded in the throwaway build stage only.
+ - DL3013
diff --git a/.node-version b/.node-version
index 82c2191..a45fd52 100644
--- a/.node-version
+++ b/.node-version
@@ -1 +1 @@
-12.22.12
+24
diff --git a/.python-version b/.python-version
index 3149984..c8cfe39 100644
--- a/.python-version
+++ b/.python-version
@@ -1,4 +1 @@
-3.8.0
-3.7.5
-3.6.9
-3.5.8
+3.10
diff --git a/.travis.yml b/.travis.yml
deleted file mode 100644
index 107c193..0000000
--- a/.travis.yml
+++ /dev/null
@@ -1,15 +0,0 @@
-language: python
-dist: xenial
-matrix:
- include:
- - python: 3.6
- - python: 3.7
- - python: 3.8
-
-install: pip install tox-travis coverage codacy-coverage
-
-script: tox
-
-after_success:
- - coverage xml
- - python-codacy-coverage -r coverage.xml
diff --git a/.trivyignore b/.trivyignore
index 6c51fb1..226f005 100644
--- a/.trivyignore
+++ b/.trivyignore
@@ -1,4 +1,4 @@
-๏ปฟ# Python vulnerabilities vendored inside setuptools โ cannot be
+# Python vulnerabilities vendored inside setuptools โ cannot be
# independently upgraded without forking the base image's Python.
# These are build-tool internals, not application runtime code.
# Tracked: will be resolved when python:3.10-slim updates setuptools.
diff --git a/CODE_OF_CONDUCT.md b/CODE_OF_CONDUCT.md
deleted file mode 100644
index d73c361..0000000
--- a/CODE_OF_CONDUCT.md
+++ /dev/null
@@ -1,128 +0,0 @@
-# Contributor Covenant Code of Conduct
-
-## Our Pledge
-
-We as members, contributors, and leaders pledge to make participation in our
-community a harassment-free experience for everyone, regardless of age, body
-size, visible or invisible disability, ethnicity, sex characteristics, gender
-identity and expression, level of experience, education, socio-economic status,
-nationality, personal appearance, race, religion, or sexual identity
-and orientation.
-
-We pledge to act and interact in ways that contribute to an open, welcoming,
-diverse, inclusive, and healthy community.
-
-## Our Standards
-
-Examples of behavior that contributes to a positive environment for our
-community include:
-
-* Demonstrating empathy and kindness toward other people
-* Being respectful of differing opinions, viewpoints, and experiences
-* Giving and gracefully accepting constructive feedback
-* Accepting responsibility and apologizing to those affected by our mistakes,
- and learning from the experience
-* Focusing on what is best not just for us as individuals, but for the
- overall community
-
-Examples of unacceptable behavior include:
-
-* The use of sexualized language or imagery, and sexual attention or
- advances of any kind
-* Trolling, insulting or derogatory comments, and personal or political attacks
-* Public or private harassment
-* Publishing others' private information, such as a physical or email
- address, without their explicit permission
-* Other conduct which could reasonably be considered inappropriate in a
- professional setting
-
-## Enforcement Responsibilities
-
-Community leaders are responsible for clarifying and enforcing our standards of
-acceptable behavior and will take appropriate and fair corrective action in
-response to any behavior that they deem inappropriate, threatening, offensive,
-or harmful.
-
-Community leaders have the right and responsibility to remove, edit, or reject
-comments, commits, code, wiki edits, issues, and other contributions that are
-not aligned to this Code of Conduct, and will communicate reasons for moderation
-decisions when appropriate.
-
-## Scope
-
-This Code of Conduct applies within all community spaces, and also applies when
-an individual is officially representing the community in public spaces.
-Examples of representing our community include using an official e-mail address,
-posting via an official social media account, or acting as an appointed
-representative at an online or offline event.
-
-## Enforcement
-
-Instances of abusive, harassing, or otherwise unacceptable behavior may be
-reported to the community leaders responsible for enforcement at
-legal@dpaste.org.
-All complaints will be reviewed and investigated promptly and fairly.
-
-All community leaders are obligated to respect the privacy and security of the
-reporter of any incident.
-
-## Enforcement Guidelines
-
-Community leaders will follow these Community Impact Guidelines in determining
-the consequences for any action they deem in violation of this Code of Conduct:
-
-### 1. Correction
-
-**Community Impact**: Use of inappropriate language or other behavior deemed
-unprofessional or unwelcome in the community.
-
-**Consequence**: A private, written warning from community leaders, providing
-clarity around the nature of the violation and an explanation of why the
-behavior was inappropriate. A public apology may be requested.
-
-### 2. Warning
-
-**Community Impact**: A violation through a single incident or series
-of actions.
-
-**Consequence**: A warning with consequences for continued behavior. No
-interaction with the people involved, including unsolicited interaction with
-those enforcing the Code of Conduct, for a specified period of time. This
-includes avoiding interactions in community spaces as well as external channels
-like social media. Violating these terms may lead to a temporary or
-permanent ban.
-
-### 3. Temporary Ban
-
-**Community Impact**: A serious violation of community standards, including
-sustained inappropriate behavior.
-
-**Consequence**: A temporary ban from any sort of interaction or public
-communication with the community for a specified period of time. No public or
-private interaction with the people involved, including unsolicited interaction
-with those enforcing the Code of Conduct, is allowed during this period.
-Violating these terms may lead to a permanent ban.
-
-### 4. Permanent Ban
-
-**Community Impact**: Demonstrating a pattern of violation of community
-standards, including sustained inappropriate behavior, harassment of an
-individual, or aggression toward or disparagement of classes of individuals.
-
-**Consequence**: A permanent ban from any sort of public interaction within
-the community.
-
-## Attribution
-
-This Code of Conduct is adapted from the [Contributor Covenant][homepage],
-version 2.0, available at
-https://www.contributor-covenant.org/version/2/0/code_of_conduct.html.
-
-Community Impact Guidelines were inspired by [Mozilla's code of conduct
-enforcement ladder](https://github.com/mozilla/diversity).
-
-[homepage]: https://www.contributor-covenant.org
-
-For answers to common questions about this code of conduct, see the FAQ at
-https://www.contributor-covenant.org/faq. Translations are available at
-https://www.contributor-covenant.org/translations.
diff --git a/CONTRIBUTING.md b/CONTRIBUTING.md
deleted file mode 100644
index 6bfa41b..0000000
--- a/CONTRIBUTING.md
+++ /dev/null
@@ -1,3 +0,0 @@
-- Write in camelCase, not snake_case.
-- Do not push to master/main without testing your changes first, make a branch
- if you have to.
diff --git a/Dockerfile b/Dockerfile
deleted file mode 100644
index 71711ef..0000000
--- a/Dockerfile
+++ /dev/null
@@ -1,57 +0,0 @@
-FROM node:lts as staticfiles
-
-ARG BUILD_EXTRAS=production
-
-RUN echo "\nโน๏ธ Building staticfiles with "${BUILD_EXTRAS}" dependencies.\n"
-
-WORKDIR /app
-
-# Install the JS dependencies
-COPY package.json package-lock.json Makefile ./
-
-RUN if [ "$BUILD_EXTRAS" = "dev" ] ; then npm install --ignore-scripts ; else npm ci --ignore-scripts ; fi
-
-# Copy the client/ directory and compile them. The Python application
-# doesn't need to exist yet.
-COPY client ./client
-
-RUN mkdir -p dpaste/static
-RUN make css
-RUN make js
-
-# ------------------------------------------------
-
-FROM python:3.10 as build
-
-ARG BUILD_EXTRAS=production
-
-ENV PORT=8000
-
-RUN echo "\nโน๏ธ Building Django project with "${BUILD_EXTRAS}" dependencies.\n"
-
-WORKDIR /app
-
-# Upgrade pip, the Image one is quite old.
-RUN pip install -U pip
-
-# Copy the dpaste staticfiles to this image
-COPY --from=staticfiles /app /app/
-
-# Copy only the files necessary to install the dpaste project as an editable
-# package. This improves caching.
-COPY setup.py setup.cfg ./
-COPY dpaste/__init__.py dpaste/
-RUN pip install -e .[${BUILD_EXTRAS}]
-
-# Copy the rest of the application code
-COPY . .
-
-# Collect all static files once.
-RUN ./manage.py collectstatic --noinput
-
-# By default run it with pyuwsgi, which is a great production ready
-# server. For development, docker-compose will override it to use the
-# regular Django runserver.
-CMD ./manage.py migrate --noinput && ./manage.py pyuwsgi --http=:${PORT} --logger file:/var/log/uwsgi.log
-
-EXPOSE ${PORT}
diff --git a/Dockerfile.hardened b/Dockerfile.hardened
index e633a69..562b7fa 100644
--- a/Dockerfile.hardened
+++ b/Dockerfile.hardened
@@ -5,7 +5,8 @@
# ============================================================
# Stage 1: Build static files (CSS/JS) with Node
-FROM node:lts-slim AS staticfiles
+# Major version pinned: a floating "lts" tag would silently jump to the next LTS.
+FROM node:24-slim AS staticfiles
WORKDIR /app
RUN apt-get update && apt-get install -y --no-install-recommends make && rm -rf /var/lib/apt/lists/*
@@ -32,7 +33,7 @@ COPY --from=staticfiles /app /app/
# Install Python dependencies
COPY setup.py setup.cfg ./
COPY dpaste/__init__.py dpaste/
-RUN pip install --no-cache-dir -e .[production]
+RUN pip install --no-cache-dir -e ".[production]"
# Copy application code
COPY . .
@@ -46,6 +47,11 @@ RUN rm -rf node_modules
# Stage 3: Production runtime
FROM python:3.10-slim AS runtime
+LABEL org.opencontainers.image.title="cloudpulse-dpaste" \
+ org.opencontainers.image.description="dpaste (MIT, DarrenOfficial/dpaste) in a hardened runtime image" \
+ org.opencontainers.image.source="https://github.com/rayenmabrouk/cloudpulse" \
+ org.opencontainers.image.licenses="MIT"
+
# Install runtime dependencies and curl for health check
RUN apt-get update && \
apt-get install -y --no-install-recommends libpq5 curl libexpat1 && \
@@ -64,8 +70,10 @@ COPY --from=build /app /app
# Create directory for SQLite database and set permissions
RUN mkdir -p /data && chown -R dpaste:dpaste /app /data
-ENV PORT=8000
-ENV DATABASE_URL=sqlite:////data/dpaste.sqlite
+ENV PORT=8000 \
+ DATABASE_URL=sqlite:////data/dpaste.sqlite \
+ PYTHONUNBUFFERED=1 \
+ PYTHONDONTWRITEBYTECODE=1
# Switch to non-root user
USER dpaste
@@ -74,7 +82,7 @@ EXPOSE ${PORT}
# Health check
HEALTHCHECK --interval=30s --timeout=5s --start-period=10s --retries=3 \
- CMD curl -f http://localhost:${PORT}/ || exit 1
+ CMD ["sh", "-c", "curl -fsS -o /dev/null http://localhost:${PORT}/ || exit 1"]
# Use exec form for proper signal handling
CMD ["sh", "-c", "python manage.py migrate --noinput && python manage.py pyuwsgi --http=:${PORT} --logger file:/dev/null"]
\ No newline at end of file
diff --git a/README.md b/README.md
index ee7f98b..32bbf85 100644
--- a/README.md
+++ b/README.md
@@ -1,24 +1,263 @@
-Dpaste
----
-
-[](https://github.com/DarrenOfficial/dpaste/actions/workflows/python.yml)
-[](https://hub.docker.com/r/darrenofficial/dpaste)
-
+# CloudPulse
-----
+An existing open-source web application (dpaste) run on AWS the way a Cloud/DevOps team would: containerised, provisioned with Terraform, delivered by gated CI/CD, secured, backed up and monitored, on a deliberately low-cost architecture.
-๐ Full documentation on [https://docs.dpaste.org](https://docs.dpaste.org)
+[](https://github.com/rayenmabrouk/cloudpulse/actions/workflows/ci.yml)
+[](https://github.com/rayenmabrouk/cloudpulse/actions/workflows/cd.yml)
+[](https://github.com/rayenmabrouk/cloudpulse/actions/workflows/terraform.yml)
+> **No permanent URL:** the deployment runs in an AWS Academy Learner Lab, which stops the instance between lab sessions and assigns a new public IP each time. The [evidence](#evidence) section shows the running system.
-dpaste is a [pastebin](https://en.wikipedia.org/wiki/Pastebin) application written in [Python](https://www.python.org/) using the [Django](https://www.djangoproject.com/) framework. You can find a live installation on [dpaste.org.](https://dpaste.org)
+## Architecture
-The project is intended to run standalone as any regular Django Project, but it's also possible to install it into an existing project as a typical Django application.
+```mermaid
+flowchart LR
+ dev["Developer"] -->|pull request| gh["GitHub"]
+ gh --> ci["CI: tests, ruff, hadolint,
image build + smoke test,
Trivy, gitleaks"]
+ gh --> tfp["Terraform pipeline: fmt, validate,
TFLint, Checkov, plan,
manual approval, apply"]
+ gh --> cd["CD: build, Trivy gate,
push, deploy, HTTPS smoke test"]
+ user["User"] -->|HTTPS 443| igw
+ mon["Prometheus + Grafana
(local)"] -.->|HTTPS probe| igw
+ subgraph aws["AWS us-east-1"]
+ state[("S3
Terraform state")]
+ ecr[("ECR
immutable tags,
scan on push, lifecycle")]
+ subgraph vpc["VPC 10.0.0.0/16"]
+ igw["Internet gateway"]
+ subgraph subnet["Public subnet 10.0.1.0/24 - security group: 80, 443 (22 optional)"]
+ ec2["EC2 t3.micro, Amazon Linux 2023
Caddy :443 - TLS, Let's Encrypt
dpaste container :8000, non-root
SQLite on a Docker volume"]
+ end
+ end
+ ssm[("SSM Parameter Store
SECRET_KEY (SecureString),
backup bucket name")]
+ cw[("CloudWatch
container logs, 5xx metric,
3 alarms")]
+ bak[("S3
daily SQLite backups")]
+ sns["SNS email
(optional)"]
+ end
-The code is open source and available on Github: [https://github.com/darrenofficial/dpaste](https://github.com/darrenofficial/dpaste). If you found bugs, have problems or ideas with the project or the website installation, please create an *Issue* there.
+ tfp -->|state + lock file| state
+ cd -->|push sha-tagged image| ecr
+ cd -->|SSM Run Command| ec2
+ igw --> ec2
+ ec2 -->|pull, instance role| ecr
+ ec2 -->|read at deploy| ssm
+ ec2 -->|awslogs driver| cw
+ ec2 -->|daily backup| bak
+ cw -.->|alarm state change| sns
+ mon -.->|metrics| cw
+```
-โ ๏ธ dpaste requires at a minimum Python 3.9 and Django 3.2.
+Everything in the AWS box is created by Terraform, except the Terraform state bucket (bootstrapped by a script, because Terraform cannot store its state in a bucket it has not created yet) and the IAM instance profile (pre-created by AWS Academy).
+## What I built
-dpaste.org: https://dpaste.org/
-pastebin: https://en.wikipedia.org/wiki/Pastebin
+**The application is not mine.** The workload is [dpaste](https://github.com/DarrenOfficial/dpaste) (v3.5, MIT License), a Django pastebin by Martin Mahner and Darren Nathanael. I did not write `dpaste/`, `client/`, `manage.py`, `setup.*`, `package*.json` or `Makefile`. The original README is kept in [`docs/upstream-dpaste-README.md`](docs/upstream-dpaste-README.md). A Cloud/DevOps engineer usually receives an application from a development team and builds everything around it; this repository is that "everything around it".
+
+| Area | What | Where |
+|---|---|---|
+| Container | 3-stage build (Node assets, Python deps, slim runtime), non-root `dpaste` user, health check. Measured: **369 MB** on disk, **~77 MB** compressed in ECR | `Dockerfile.hardened`, `.dockerignore`, `docker-compose.yml` |
+| Infrastructure as Code | Terraform, 3 modules: networking (VPC, subnet, IGW, routes, security groups), compute (EC2, ECR, SSM, backup bucket), monitoring (logs, metric filter, alarms, SNS) | `terraform/` |
+| Remote state | S3 backend: versioned, encrypted, public access blocked, TLS-only, S3-native locking | `terraform/backend.tf`, `scripts/bootstrap-tfstate.ps1` |
+| Deployment | Pull from ECR with the instance role, secret from SSM, health check, **automatic rollback** to the previous image | `scripts/deploy.sh` |
+| HTTPS | Caddy reverse proxy with automatic Let's Encrypt certificates; the app port is bound to localhost only | `scripts/deploy.sh` |
+| Backups | Daily online SQLite backup to S3, restore with integrity check | `scripts/backup.sh` |
+| CI | Tests, lint, Dockerfile lint, image build + container smoke test, Trivy, secret scan | `.github/workflows/ci.yml` |
+| CD | Build -> Trivy gate -> ECR -> deploy through **SSM Run Command** (no SSH keys in CI) -> HTTPS smoke test | `.github/workflows/cd.yml` |
+| Infra pipeline | PR: fmt, validate, TFLint, Checkov, plan. Merge: plan -> **manual approval** -> apply the saved plan | `.github/workflows/terraform.yml` |
+| Observability | CloudWatch logs, HTTP 5xx metric and alarm, status and CPU alarms; local Prometheus + blackbox exporter + Grafana probing production | `terraform/modules/monitoring/`, `monitoring/` |
+| Operations docs | Runbook, troubleshooting (15 real issues), cost analysis | `docs/` |
+
+## Technology stack
+
+AWS (VPC, EC2, ECR, S3, SSM, CloudWatch, SNS) ยท Terraform ยท Docker ยท Caddy ยท GitHub Actions ยท Trivy ยท Checkov ยท TFLint ยท hadolint ยท gitleaks ยท Prometheus ยท Grafana ยท Bash / PowerShell
+
+## Infrastructure
+
+| Component | Why it exists |
+|---|---|
+| **VPC + public subnet + internet gateway** | Isolated network with a single public subnet. No NAT gateway: the only instance needs a public IP anyway, and NAT would cost more than the rest of the stack combined. |
+| **Security group** | Inbound 443 (HTTPS) and 80 (HTTP -> HTTPS redirect and Let's Encrypt challenge) only. SSH (22) is **off by default** and can be opened to a single `/32` for break-glass access. The default security group is emptied so nothing can use it by accident. |
+| **EC2 t3.micro (Amazon Linux 2023)** | Runs Docker: Caddy and the dpaste container. IMDSv2 required, encrypted gp3 root volume, 1-minute monitoring. |
+| **ECR** | Private registry. Tags are the commit SHA and **immutable**, scan on push, lifecycle policy keeps the 10 newest images. |
+| **SSM Parameter Store** | Django `SECRET_KEY` generated by Terraform and stored as a `SecureString`; read by the instance at deploy time, never in Git, the image or CI. |
+| **SSM Run Command** | How CD deploys and how operators run commands: no inbound port, no SSH key distributed, every command audited. |
+| **S3 (state)** | Terraform remote state with versioning, encryption and native lock file. |
+| **S3 (backups)** | Daily SQLite backups: versioned, SSE-S3, TLS-only, public access blocked, 14-day expiry. The EBS volume is otherwise the only copy of the data. |
+| **CloudWatch** | Container logs from dpaste and Caddy (`awslogs` driver, 7-day retention), a metric filter counting HTTP 5xx in Caddy's access logs, and three alarms (below). |
+| **SNS (optional)** | Email notification for alarm state changes when `alarm_email` is set. |
+
+**Alarms and what they detect**
+
+| Alarm | Fires when | Detects |
+|---|---|---|
+| `cloudpulse-status-check` | EC2 status check fails for 2 x 5 min | Host, network or OS failure |
+| `cloudpulse-cpu-high` | CPU > 80 % for 10 min | Sustained load; a t3.micro will run out of CPU credits |
+| `cloudpulse-http-5xx` | >= 5 HTTP 5xx responses in 5 min | Application failure. When the dpaste container is down, Caddy answers **502**, which the EC2 status check cannot see |
+
+## Security
+
+- **Network:** only 80/443 are public; the app listens on `127.0.0.1:8000` and a private Docker network; SSH closed unless explicitly enabled (and `0.0.0.0/0` is rejected by variable validation).
+- **Access:** deployments and shell access through SSM, not SSH. IMDSv2 required (mitigates SSRF credential theft).
+- **Secrets:** generated by Terraform, stored encrypted in SSM, written on the instance to a root-only env file (`umask 077`). The ECR credential helper avoids storing a registry token on disk. No credentials in Git (checked by gitleaks over the full history on every PR).
+- **Encryption:** EBS, both S3 buckets and CloudWatch Logs encrypted at rest; S3 bucket policies deny non-TLS requests; public HTTPS with Let's Encrypt.
+- **Container:** non-root user, minimal runtime image (no compilers, no Node), health check, dpaste's own CSP / CSRF / secure-cookie / clickjacking headers left enabled.
+- **Supply chain:** third-party GitHub Actions pinned to commit SHAs, downloaded tool binaries verified by SHA-256, Dependabot for actions, base images, Terraform providers and pip. CD configures AWS credentials only **after** the image has passed the Trivy scan.
+- **Gates:** Trivy blocks any fixable CRITICAL/HIGH vulnerability in CI and CD; Checkov (47 passed, 0 failed, 17 skipped, each skip justified inline) and TFLint gate Terraform; hadolint gates the Dockerfile; infrastructure changes need my approval before apply; `master` is branch-protected.
+- **Accepted risks:** see [AWS Academy limitations](#aws-academy-limitations). `.trivyignore` lists four CVEs in the base image's bundled setuptools (build tooling, not application code). ECR's own scan also reports CRITICAL/HIGH findings in Debian base packages that have no fixed version yet; Trivy's policy blocks only fixable ones, and a base-image rebuild picks up fixes as Debian releases them.
+
+**IAM.** The Learner Lab forbids creating IAM roles, so the instance uses the broad, pre-created `LabRole`. In a real account the instance role would only allow: `ecr:GetAuthorizationToken` plus pull actions on this repository, `ssm:GetParameter` on `/cloudpulse/*`, `logs:CreateLogStream`/`PutLogEvents` on the log group, `s3:PutObject`/`GetObject`/`ListBucket` on the backup bucket, and `AmazonSSMManagedInstanceCore`. GitHub Actions would assume an OIDC role scoped to this repository instead of using session credentials.
+
+## CI/CD
+
+| Change | Path to production |
+|---|---|
+| Application, image or on-instance scripts | PR -> **CI** -> merge -> **CD**: build `cloudpulse:`, Trivy gate, push to ECR, send `deploy.sh` + `backup.sh` via SSM, health check with automatic rollback, HTTPS smoke test |
+| Infrastructure (`terraform/`) | PR -> CI + **Terraform checks** (fmt, validate, TFLint, Checkov, plan in the run summary) -> merge -> plan -> **waits for approval** -> apply exactly the reviewed plan |
+
+**CI jobs** (every PR and push): `test` (pytest, Python 3.10 like the image), `lint` (ruff, blocking), `docker-build-scan` (hadolint, build, run the container and require: health check `healthy`, non-root UID, `GET /` 200, API accepts a snippet; then Trivy), `secret-scan` (gitleaks, full history). `test`, `lint` and `docker-build-scan` are required checks.
+
+**Why deployment is only partly automated:** the Learner Lab issues session credentials that expire after about 4 hours and cannot create an OIDC provider. CD and the Terraform pipeline read those credentials from GitHub secrets, refreshed at the start of each lab session with `scripts/refresh-github-aws-secrets.ps1`. Outside a lab session, CD and Terraform runs fail at the AWS login step; CI does not need AWS and always runs.
+
+## Local development
+
+Requires Docker. From the repository root:
+
+```bash
+docker compose up --build -d # hardened image, http://localhost:8000
+docker compose ps # STATUS shows (healthy) after ~30 s
+curl -X POST -d "content=hello&lexer=python" http://localhost:8000/api/ # returns the snippet URL
+docker compose down # add -v to delete the local database volume
+```
+
+Tests without Docker (Python 3.10, the image's version):
+
+```bash
+python3.10 -m venv .venv && . .venv/bin/activate # Windows: .venv\Scripts\activate
+pip install -e ".[dev]"
+pytest dpaste/ # 51 tests
+ruff check dpaste/
+```
+
+On Windows PowerShell 5, use `curl.exe` (plain `curl` is an alias for `Invoke-WebRequest`).
+
+Monitoring stack (Prometheus, blackbox exporter, Grafana): see [runbook section 6](docs/deployment-runbook.md#6-monitoring-stack-local-verified).
+
+## AWS deployment
+
+Full procedure, including GitHub environments and branch protection: [`docs/deployment-runbook.md`](docs/deployment-runbook.md). Short version (PowerShell, repository root, Terraform >= 1.10, AWS CLI v2, GitHub CLI):
+
+```powershell
+# 1. Every lab session: Start Lab, paste credentials into ~/.aws/credentials, then
+aws sts get-caller-identity
+powershell -ExecutionPolicy Bypass -File scripts\refresh-github-aws-secrets.ps1
+
+# 2. Once: remote state bucket
+powershell -ExecutionPolicy Bypass -File scripts\bootstrap-tfstate.ps1
+
+# 3. Infrastructure (optional settings in terraform.tfvars, see terraform.tfvars.example)
+terraform -chdir=terraform init
+terraform -chdir=terraform plan -out=tfplan
+terraform -chdir=terraform apply tfplan
+
+# 4. Build, scan, push and deploy the current master through CD
+gh workflow run cd.yml --repo rayenmabrouk/cloudpulse
+terraform -chdir=terraform output app_url
+```
+
+In a different AWS account, pass the state bucket at init time: `terraform init -backend-config="bucket=cloudpulse-tfstate-"`.
+
+## Verification
+
+| What | Command | Expected |
+|---|---|---|
+| Terraform formatting | `terraform -chdir=terraform fmt -check -recursive` | no output |
+| Terraform validity | `terraform -chdir=terraform init -backend=false` then `terraform -chdir=terraform validate` | `Success!` |
+| Terraform lint / security | `tflint --chdir=terraform --recursive` and `checkov -d terraform --framework terraform --quiet` | no issues; 0 failed checks |
+| Drift | `terraform -chdir=terraform plan` | `No changes` |
+| Dockerfile | `hadolint Dockerfile.hardened` | no warnings |
+| Container health | `docker inspect -f "{{.State.Health.Status}}" ` | `healthy` |
+| Non-root | `docker compose exec app id -u` | not `0` |
+| Deployed app | `curl.exe -sI (terraform -chdir=terraform output -raw app_url)` | `HTTP/1.1 200`, valid certificate |
+| Instance managed by SSM | `aws ssm describe-instance-information --query "InstanceInformationList[].[InstanceId,PingStatus]"` | `Online` |
+| Logs | `aws logs tail /cloudpulse/dpaste --since 15m` | dpaste startup lines and Caddy JSON access logs |
+| Alarms | `aws cloudwatch describe-alarms --alarm-name-prefix cloudpulse --query "MetricAlarms[].[AlarmName,StateValue]" --output table` | 3 alarms, `OK` |
+| Backups | `aws s3 ls s3://$(terraform -chdir=terraform output -raw backup_bucket_name)/sqlite/` | one object per day |
+| Local monitoring | http://localhost:9090/targets and Grafana http://localhost:3000 | local and production probes `UP` |
+
+**Verified in AWS** (executed and observed, not just configured): CD deploys through SSM; **automatic rollback** after a deliberately broken image (production kept serving 200); data persisted across container replacement; Terraform apply paused for approval and applied the exact plan; `DpasteDown` fired in Prometheus/Grafana when the container stopped and resolved after restart; CloudWatch received container logs.
+
+**Added after that verification and not yet exercised in AWS:** SQLite backups to S3 and restore, the HTTP 5xx metric filter and alarm, SNS notifications, the ECR lifecycle policy, the optional SSH rule, and the updated CI/CD workflows. The runbook marks them `[not yet verified]`.
+
+## Cost considerations
+
+The whole stack is about **$16 per month if left running 24/7** (us-east-1 on-demand prices); in practice the Learner Lab stops the instance between sessions, so actual spend is a fraction of that. Breakdown and decisions: [`docs/cost.md`](docs/cost.md).
+
+Resources that incur charges: the **EC2 instance** (~$7.60/month), its **public IPv4 address** ($3.65), the **EBS volume** ($1.60), **detailed monitoring** (~$2.10), **3 alarms + 1 custom metric** (~$0.60), CloudWatch Logs ingestion, and ECR / S3 storage (cents). Not used, deliberately: NAT gateway, load balancer, RDS, EKS, Route 53, customer-managed KMS keys.
+
+To stop all charges: `terraform -chdir=terraform destroy` (ECR images and the backup bucket are force-deleted; the state bucket is removed by hand, see the runbook).
+
+## AWS Academy limitations
+
+| Limitation | Consequence | In a real account |
+|---|---|---|
+| Session credentials expire after ~4 h | GitHub secrets refreshed every session; CD/Terraform runs fail outside sessions | OIDC role for GitHub Actions |
+| `iam:CreateRole` and `iam:CreateOpenIDConnectProvider` denied | Broad `LabRole` instead of a least-privilege instance role; no OIDC; no VPC flow logs, S3 replication or DLM snapshot policies (all need a service role) | Dedicated least-privilege roles |
+| Instance stopped between sessions, new public IP on start | sslip.io hostname changes; a redeploy regenerates the certificate and `ALLOWED_HOSTS` | Elastic IP + Route 53 domain |
+| Regions limited to us-east-1 / us-west-2, small instance types only | Single region, t3.micro | Region chosen for users/data residency |
+| $50 total credit | No load balancer, NAT or managed database | See engineering decisions |
+
+## Project structure
+
+```
+.
+โโโ Dockerfile.hardened # 3-stage, non-root production image
+โโโ docker-compose.yml # local run of the hardened image
+โโโ terraform/
+โ โโโ main.tf, variables.tf, outputs.tf, providers.tf, backend.tf
+โ โโโ modules/
+โ โโโ networking/ # VPC, subnet, IGW, routes, security groups
+โ โโโ compute/ # EC2, ECR + lifecycle, SSM secret, backup bucket
+โ โโโ monitoring/ # log group, 5xx metric filter, alarms, SNS
+โโโ scripts/
+โ โโโ deploy.sh # runs on EC2: pull, secret, health check, rollback, Caddy, timers
+โ โโโ backup.sh # runs on EC2: SQLite backup / list / restore
+โ โโโ bootstrap-tfstate.ps1 # one-time state bucket creation
+โ โโโ refresh-github-aws-secrets.ps1
+โโโ monitoring/ # Prometheus, blackbox exporter, Grafana (local)
+โโโ .github/workflows/ # ci.yml, cd.yml, terraform.yml
+โโโ docs/ # runbook, troubleshooting, cost, screenshots
+โโโ dpaste/, client/, ... # inherited application (not my code)
+```
+
+## Engineering decisions
+
+- **EC2 + Docker instead of EKS or ECS.** One small container does not justify a $73/month control plane or a load balancer. The deploy script provides what matters at this scale: immutable images, health checks and automatic rollback.
+- **SQLite instead of RDS (or DynamoDB).** dpaste is built on Django's relational ORM and designed for SQLite; RDS would roughly double the cost and DynamoDB would mean rewriting the data layer. The trade-off is a stateful instance, which is why daily S3 backups exist. Production would use RDS PostgreSQL and a stateless app tier.
+- **SSM instead of SSH.** No inbound port, no key to distribute to CI, commands logged. SSH remains available as an opt-in break-glass rule restricted to one IP.
+- **Caddy on the instance instead of ALB + ACM.** Automatic Let's Encrypt HTTPS for free; HTTPS was required because dpaste sets secure cookies (CSRF fails over plain HTTP).
+- **S3-native state locking instead of a DynamoDB lock table.** Terraform >= 1.10 locks with a lock file in the state bucket; one fewer resource, and DynamoDB-based locking is deprecated.
+- **Multi-stage build, non-root runtime.** Node and compilers stay in build stages; the runtime image holds only Python, the app and its static files, and runs as an unprivileged user.
+- **Terraform with a gated pipeline instead of console clicks.** Every change is reviewed as a plan, scanned, approved, and the exact reviewed plan is applied.
+- **Public subnet, no NAT.** Saves ~$32/month; the security group and localhost-bound app limit exposure. Production would use private subnets behind a load balancer, with VPC endpoints.
+- **Black-box monitoring.** Prometheus probes the public URL (availability, latency, TLS expiry) without changing the application; CloudWatch covers the AWS side.
+- **Backups to S3 from the instance instead of EBS snapshot policies.** Data Lifecycle Manager needs its own IAM role, which the lab forbids; the SQLite online-backup API gives a consistent copy without stopping the app.
+
+## How this was built
+
+I built this project with an AI assistant (Claude) as a pair programmer. It proposed designs and drafted code, commands and parts of this documentation. I ran the commands, reviewed the Terraform plans and pull requests, worked through the failures, and made the final decisions.
+
+## Evidence
+
+Screenshots are in [`docs/screenshots/`](docs/screenshots/).
+
+| | |
+|---|---|
+| HTTPS site with valid certificate |  |
+| Grafana: probes + CloudWatch |  |
+| CD pipeline run |  |
+| Terraform apply waiting for approval |  |
+| Automatic rollback: container started by the rollback, same database |  |
+
+## License
+
+- dpaste application code: MIT License, (c) the dpaste authors (see [`LICENSE`](LICENSE)).
+- Infrastructure, pipelines, scripts and documentation added in this repository: MIT License, (c) Rayen Mabrouk.
diff --git a/docs/cost.md b/docs/cost.md
new file mode 100644
index 0000000..291ee82
--- /dev/null
+++ b/docs/cost.md
@@ -0,0 +1,45 @@
+# Cost analysis
+
+Last updated: 2026-09-24
+
+The project runs in an **AWS Academy Learner Lab** with a **$50 credit budget**. Academy stops the EC2 instance when a lab session ends, so the real spend is far below a 24/7 estimate.
+
+Actual spend is shown in Vocareum ("Used $X of $50").
+
+## What a 24/7 month would cost (estimate)
+
+Estimates use public **us-east-1 on-demand** prices at the time of writing; check the [AWS pricing pages](https://aws.amazon.com/pricing/) for current values.
+
+| Resource | Basis | ~ USD / month |
+|---|---|---|
+| EC2 t3.micro | $0.0104 / hour x 730 h | 7.59 |
+| Public IPv4 address | $0.005 / hour x 730 h | 3.65 |
+| EBS gp3, 20 GB | $0.08 / GB-month | 1.60 |
+| EC2 detailed monitoring | 7 metrics x $0.30 | 2.10 |
+| CloudWatch alarms | 3 x $0.10 | 0.30 |
+| CloudWatch custom metric (5xx metric filter) | 1 x $0.30 | 0.30 |
+| CloudWatch Logs | < 1 GB ingested, 7-day retention | < 0.50 |
+| ECR storage | 10 images kept by the lifecycle policy, ~0.8 GB x $0.10 | < 0.10 |
+| S3 state bucket | a few KB, versioned | ~0 |
+| S3 backup bucket | 14 daily compressed SQLite backups, KB-sized | ~0 |
+| SNS email notifications (optional) | within the free tier (1,000 emails / month) | 0 |
+| SSM Parameter Store | standard parameter | 0 |
+| Data transfer out | a pastebin demo, well under 1 GB | < 0.10 |
+| **Total** | | **~ $16** |
+
+GitHub Actions minutes are free for public repositories. Let's Encrypt certificates and sslip.io DNS are free.
+
+## Deliberate cost decisions
+
+| Not used | Typical monthly cost | Why skipped | Production alternative |
+|---|---|---|---|
+| Application Load Balancer | ~$16+ | single instance; Caddy terminates TLS | ALB + ACM certificate |
+| NAT gateway | ~$32+ | instance sits in a public subnet | private subnet + NAT or VPC endpoints |
+| RDS | ~$12+ (smallest) | dpaste is designed for SQLite | RDS PostgreSQL, stateless app tier |
+| Route 53 hosted zone + domain | ~$0.50 + domain | sslip.io hostname | Route 53 + Elastic IP |
+| EKS | ~$73 control plane | far beyond the need | EKS or ECS Fargate when there are several services |
+| Customer-managed KMS keys | $1 / key | AWS-managed encryption is already on | CMKs where key policy control is required |
+
+## Where the money would go first
+
+Detailed monitoring (~$2.10) was enabled on purpose: 1-minute metrics make the CPU alarm react faster. On a tight budget it is the first thing to turn off, followed by the public IPv4 charge (only avoidable with a load balancer or IPv6).
\ No newline at end of file
diff --git a/docs/deployment-runbook.md b/docs/deployment-runbook.md
new file mode 100644
index 0000000..207f00e
--- /dev/null
+++ b/docs/deployment-runbook.md
@@ -0,0 +1,294 @@
+# Deployment runbook
+
+Last updated: 2026-09-24
+
+How to operate CloudPulse on AWS, from an empty AWS Academy account to a running, monitored deployment.
+All commands are for **Windows PowerShell 5.1** (the environment this project was built on), run from the repository root unless stated otherwise.
+
+Each step is marked:
+- **[verified]**: executed and observed working during the build
+- **[not yet verified]**: written from the design, still to be tested end to end
+
+---
+
+## 0. Prerequisites
+
+| Tool | Version used | Install |
+|---|---|---|
+| Git | any recent | `winget install -e --id Git.Git` |
+| Docker Desktop | any recent | `winget install -e --id Docker.DockerDesktop` |
+| Terraform | 1.16.2 (**>= 1.10 required** for S3-native locking) | `winget install -e --id Hashicorp.Terraform` |
+| AWS CLI | v2 | `winget install -e --id Amazon.AWSCLI` |
+| GitHub CLI | any recent | `winget install -e --id GitHub.cli`, then `gh auth login` |
+| Session Manager plugin (optional, for interactive shells) | any recent | `winget install -e --id Amazon.SessionManagerPlugin` |
+
+AWS access: an **AWS Academy Learner Lab** (region `us-east-1`). The lab provides the `LabInstanceProfile` and the `vockey` key pair used by Terraform (both configurable: `instance_profile_name`, `ssh_key_name`).
+
+---
+
+## 1. At the start of every lab session [verified]
+
+Academy credentials expire when the lab session ends (about 4 hours), and the EC2 instance is stopped between sessions.
+
+1. In Vocareum: **Start Lab**, wait for the green dot, open **AWS Details -> AWS CLI: Show**.
+2. Write the credentials **without a BOM** (Windows PowerShell 5's `Set-Content -Encoding UTF8` adds a BOM that the AWS SDK cannot parse):
+
+ ```powershell
+ $content = @"
+ [default]
+ aws_access_key_id=PASTE
+ aws_secret_access_key=PASTE
+ aws_session_token=PASTE
+ "@
+ [System.IO.File]::WriteAllText("$HOME\.aws\credentials", $content)
+ ```
+
+3. One-time only, the region:
+
+ ```powershell
+ [System.IO.File]::WriteAllText("$HOME\.aws\config", "[default]`nregion = us-east-1`noutput = json`n")
+ ```
+
+4. Verify, then push the new credentials to GitHub Actions:
+
+ ```powershell
+ aws sts get-caller-identity
+ powershell -ExecutionPolicy Bypass -File scripts\refresh-github-aws-secrets.ps1
+ ```
+
+5. If the AWS Console shows `explicit deny ... voc-cancel-cred`, the console tab is from an older session: close all console tabs and reopen the console from Vocareum.
+
+### After a lab restart: redeploy [not yet verified]
+
+When the instance starts again it gets a **new public IP**, so the sslip.io hostname changes. The containers restart automatically, but Caddy's certificate and Django's `ALLOWED_HOSTS` still refer to the old hostname. Redeploy so `deploy.sh` regenerates both:
+
+```powershell
+terraform -chdir=terraform apply -refresh-only -auto-approve # refresh outputs (new IP) in state
+terraform -chdir=terraform output app_url
+gh workflow run cd.yml --repo rayenmabrouk/cloudpulse # rebuild + redeploy through CD
+```
+
+If your home IP changed, update `terraform/terraform.tfvars` and the `TF_VAR_ALLOWED_SSH_CIDR` secret, then change the SSH rule through a pull request.
+
+---
+
+## 2. First-time setup from zero
+
+### 2.1 Remote state bucket [verified]
+
+The state bucket is created outside Terraform (Terraform cannot store its state in a bucket it has not created yet). The script is idempotent:
+
+```powershell
+powershell -ExecutionPolicy Bypass -File scripts\bootstrap-tfstate.ps1
+```
+
+It creates `cloudpulse-tfstate-` with versioning, SSE-S3 encryption, all public access blocked and a TLS-only bucket policy.
+
+### 2.2 Infrastructure [verified]
+
+```powershell
+Copy-Item terraform\terraform.tfvars.example terraform\terraform.tfvars
+# optional in terraform.tfvars:
+# allowed_ssh_cidr = "/32" # break-glass SSH; empty = port 22 closed
+# alarm_email = "you@example.com" # alarm notifications (confirm the SNS email)
+cd terraform
+terraform init
+terraform plan -out=tfplan
+terraform apply tfplan
+terraform output
+cd ..
+```
+
+This creates the VPC, subnet, internet gateway, security groups, EC2 instance (Docker installed by `user_data`), ECR repository with lifecycle policy, SSM `SecureString` parameter with a generated Django `SECRET_KEY`, the SQLite backup bucket and its SSM parameter, CloudWatch log group, 5xx metric filter and alarms (and an SNS topic if `alarm_email` is set).
+
+**[not yet verified]** The ECR lifecycle policy, backup bucket, 5xx metric filter/alarm, SNS topic and the optional SSH rule were added after the verified build. The first plan after pulling these changes should show only additions (ECR lifecycle policy, S3 bucket and its settings, SSM parameter, metric filter, alarm) and in-place tag updates, **no replacement**. Stop and investigate if it proposes to replace the instance.
+
+After this first bootstrap, **infrastructure changes go through pull requests** and the Terraform pipeline (section 4.2).
+
+### 2.3 GitHub configuration [verified]
+
+```powershell
+# AWS credentials for CI/CD (repeat every lab session)
+powershell -ExecutionPolicy Bypass -File scripts\refresh-github-aws-secrets.ps1
+
+# Optional: SSH CIDR for the Terraform pipeline (must match terraform.tfvars).
+# Without this secret the pipeline plans with SSH closed.
+gh secret set TF_VAR_ALLOWED_SSH_CIDR --repo rayenmabrouk/cloudpulse --body "/32"
+# To close port 22 again: gh secret delete TF_VAR_ALLOWED_SSH_CIDR --repo rayenmabrouk/cloudpulse
+```
+
+Environments (created with `gh api`, see the PR history for exact commands):
+- `production`: deployments allowed from `master` only (used by CD)
+- `infrastructure`: `master` only **and a required reviewer** (used by Terraform apply)
+
+Branch protection on `master`: pull request required, required checks `test`, `lint`, `docker-build-scan`, branch up to date, enforced for admins, no force-push or deletion. The `secret-scan` job also runs on every PR; add it to the required checks in the branch protection settings.
+
+### 2.4 First deployment [verified]
+
+The first release was deployed manually to validate `deploy.sh`, then all later releases went through CD. To deploy the current `master` through CD:
+
+```powershell
+gh workflow run cd.yml --repo rayenmabrouk/cloudpulse
+```
+
+**[not yet verified]** The manual trigger (`workflow_dispatch`) runs the same jobs as a push but has not been exercised yet; every verified CD run was triggered by a merge.
+
+Get the URL:
+
+```powershell
+terraform -chdir=terraform output app_url
+```
+
+---
+
+## 3. What a deployment does (`scripts/deploy.sh`) [verified]
+
+Runs on the instance as root. CD ships the commit's `deploy.sh` and `backup.sh` to `/opt/cloudpulse/` in the SSM command, then runs `deploy.sh `:
+
+1. Reads account ID, region and public IP from **IMDSv2** (session token required).
+2. Authenticates to ECR with the **ECR credential helper** via the instance role (falls back to `docker login` if the helper is unavailable).
+3. Pulls `cloudpulse:`.
+4. Reads `SECRET_KEY` from SSM Parameter Store and writes a root-only env file (`umask 077`) with `DEBUG=False` and `ALLOWED_HOSTS=,localhost,127.0.0.1`.
+5. Replaces the `dpaste` container on a private Docker network, published on `127.0.0.1:8000` only, SQLite on the `dpaste_data` volume, logs to CloudWatch.
+6. Health-checks `http://localhost:8000/` for up to 60 s. **On failure it restarts the previous image.**
+7. Starts or reloads **Caddy** (ports 80/443, automatic Let's Encrypt certificate, JSON access logs to CloudWatch).
+8. Installs the daily `cloudpulse-cleanup.timer` and **[not yet verified]** `cloudpulse-backup.timer` (03:00 UTC).
+9. **[not yet verified]** Removes older release images from the instance disk, keeping the running image and the previous one (rollback target).
+
+---
+
+## 4. Day-to-day changes
+
+### 4.1 Application, image or deploy script [verified]
+
+```powershell
+git checkout -b feature/my-change
+# edit, commit
+git push -u origin feature/my-change
+gh pr create --repo rayenmabrouk/cloudpulse --base master --fill
+gh pr checks --repo rayenmabrouk/cloudpulse --watch
+gh pr merge --repo rayenmabrouk/cloudpulse --merge --delete-branch
+```
+
+The merge triggers CD when it touches `dpaste/`, `client/`, `Dockerfile.hardened`, `setup.*`, `package*.json`, `scripts/deploy.sh`, `scripts/backup.sh` or `.github/workflows/cd.yml`. Refresh the AWS secrets first (section 1), otherwise CD fails at the AWS login step. Watch it:
+
+```powershell
+$RUN = gh run list --repo rayenmabrouk/cloudpulse --workflow cd.yml --limit 1 --json databaseId --jq ".[0].databaseId"
+gh run watch $RUN --repo rayenmabrouk/cloudpulse --exit-status
+```
+
+### 4.2 Infrastructure [verified]
+
+Same PR flow for changes under `terraform/`. On the PR, the Terraform workflow runs `fmt`, `validate`, TFLint, Checkov and `plan` (the plan is in the run summary). After merge:
+
+1. Actions -> the Terraform run -> the **Apply (manual approval)** job waits.
+2. Review the plan in the run summary.
+3. **Review deployments -> infrastructure -> Approve and deploy.**
+4. Verify locally: `terraform -chdir=terraform plan` should report **No changes**.
+
+A Checkov finding must be either fixed or skipped inline with a justification (`# checkov:skip=:`).
+
+---
+
+## 5. Operations
+
+### Run a command on the server without SSH [verified]
+
+```powershell
+$ID = aws ec2 describe-instances --filters "Name=tag:Name,Values=cloudpulse-server" "Name=instance-state-name,Values=running" --query "Reservations[0].Instances[0].InstanceId" --output text
+[System.IO.File]::WriteAllText("$env:TEMP\cmd.json", '{"commands":["docker ps --format ''{{.Names}} {{.Image}} {{.Status}}''"]}')
+$CMD = aws ssm send-command --instance-ids $ID --document-name AWS-RunShellScript --parameters "file://$env:TEMP\cmd.json" --query Command.CommandId --output text
+Start-Sleep -Seconds 5
+aws ssm get-command-invocation --command-id $CMD --instance-id $ID --query StandardOutputContent --output text
+```
+
+The AWS CLI on Windows crashes when the output contains emoji (dpaste's startup banner). Append `| tr -cd '\11\12\40-\176'` to the remote command to strip them.
+
+### Interactive shell without SSH [not yet verified]
+
+Requires the Session Manager plugin:
+
+```powershell
+aws ssm start-session --target (terraform -chdir=terraform output -raw instance_id)
+```
+
+### Logs [verified]
+
+```powershell
+aws logs tail /cloudpulse/dpaste --since 15m # app + Caddy
+aws logs tail /cloudpulse/dpaste --since 1h --filter-pattern "migrations"
+```
+
+### Manual rollback to a specific release [verified] (automatic rollback verified; manual uses the same script)
+
+ECR keeps the 10 most recent images (lifecycle policy). List releases, then deploy an older tag through SSM:
+
+```powershell
+aws ecr describe-images --repository-name cloudpulse --query "sort_by(imageDetails,&imagePushedAt)[].imageTags[0]" --output text
+# then send-command with: /opt/cloudpulse/deploy.sh
+```
+
+### Snippet cleanup [verified]
+
+Runs daily via `cloudpulse-cleanup.timer`. Run it now: `systemctl start cloudpulse-cleanup.service` (through SSM), then `journalctl -u cloudpulse-cleanup.service -n 5`.
+
+### Backups and restore [not yet verified]
+
+`cloudpulse-backup.timer` runs `/opt/cloudpulse/backup.sh backup` daily at 03:00 UTC (and at boot if a run was missed while the lab was stopped). It uses SQLite's online backup API, so dpaste keeps serving, and uploads `sqlite/dpaste-.sqlite.gz` to the backup bucket (kept 14 days).
+
+Through SSM (section "Run a command on the server"), run one of:
+
+```bash
+/opt/cloudpulse/backup.sh backup # back up now
+/opt/cloudpulse/backup.sh list # list backups
+/opt/cloudpulse/backup.sh restore sqlite/dpaste-.sqlite.gz
+journalctl -u cloudpulse-backup.service -n 20 # last run
+```
+
+`restore` downloads the backup, refuses it unless `PRAGMA integrity_check` returns `ok`, takes a fresh backup of the current database, stops dpaste, replaces the database file in the `dpaste_data` volume, restarts dpaste and waits for it to be healthy. From your machine: `aws s3 ls s3://$(terraform -chdir=terraform output -raw backup_bucket_name)/sqlite/`.
+
+### Alarms [not yet verified for the 5xx alarm and SNS]
+
+```powershell
+aws cloudwatch describe-alarms --alarm-name-prefix cloudpulse --query "MetricAlarms[].[AlarmName,StateValue]" --output table
+```
+
+- `cloudpulse-status-check`: EC2 status check failing (host / OS).
+- `cloudpulse-cpu-high`: CPU > 80 % for 10 minutes.
+- `cloudpulse-http-5xx`: 5 or more HTTP 5xx answered by Caddy in 5 minutes. Test it by stopping dpaste through SSM (`docker stop dpaste`), loading the site a few times (Caddy answers 502), then `docker start dpaste`.
+
+With `alarm_email` set, every state change is emailed (confirm the SNS subscription email first).
+
+### SSH (break-glass) [verified]
+
+Only when `allowed_ssh_cidr` is set (port 22 is closed otherwise), from that IP, with the Academy key (`labsuser.pem`, permissions restricted with `icacls`):
+
+```powershell
+ssh -i "$HOME\.ssh\labsuser.pem" "ec2-user@$(terraform -chdir=terraform output -raw instance_public_ip)"
+```
+
+---
+
+## 6. Monitoring stack (local) [verified]
+
+```powershell
+cd monitoring
+powershell -ExecutionPolicy Bypass -File set-production-target.ps1
+docker compose up -d --build
+cd ..
+```
+
+- Grafana: http://localhost:3000 (admin / `GRAFANA_ADMIN_PASSWORD`, default `cloudpulse-local`) -> Dashboards -> CloudPulse
+- Prometheus: http://localhost:9090 (targets, alerts)
+- The CloudWatch panels use `~/.aws` (read-only mount) and stop working when the lab session credentials expire.
+
+---
+
+## 7. Teardown [not yet verified]
+
+```powershell
+terraform -chdir=terraform plan -destroy -out=destroy.tfplan
+terraform -chdir=terraform apply destroy.tfplan
+```
+
+`force_delete = true` on the ECR repository removes its images, and `force_destroy = true` on the backup bucket removes the backups (download any you want to keep first). The state bucket is not managed by Terraform: empty all object **versions** and delete it manually afterwards, only once the state is no longer needed.
\ No newline at end of file
diff --git a/docs/screenshots/01-github-fork-dpaste.png b/docs/screenshots/01-github-fork-dpaste.png
new file mode 100644
index 0000000..bb44984
Binary files /dev/null and b/docs/screenshots/01-github-fork-dpaste.png differ
diff --git a/docs/screenshots/02-docker-build-hardened.png b/docs/screenshots/02-docker-build-hardened.png
new file mode 100644
index 0000000..b7ecd29
Binary files /dev/null and b/docs/screenshots/02-docker-build-hardened.png differ
diff --git a/docs/screenshots/03-dpaste-local-docker.png b/docs/screenshots/03-dpaste-local-docker.png
new file mode 100644
index 0000000..de7ce34
Binary files /dev/null and b/docs/screenshots/03-dpaste-local-docker.png differ
diff --git a/docs/screenshots/04-api-security-headers.png b/docs/screenshots/04-api-security-headers.png
new file mode 100644
index 0000000..fbbcbbb
Binary files /dev/null and b/docs/screenshots/04-api-security-headers.png differ
diff --git a/docs/screenshots/05-ci-first-runs-failing.png b/docs/screenshots/05-ci-first-runs-failing.png
new file mode 100644
index 0000000..be2a53b
Binary files /dev/null and b/docs/screenshots/05-ci-first-runs-failing.png differ
diff --git a/docs/screenshots/06-ci-trivy-failure-annotations.png b/docs/screenshots/06-ci-trivy-failure-annotations.png
new file mode 100644
index 0000000..95cfc9e
Binary files /dev/null and b/docs/screenshots/06-ci-trivy-failure-annotations.png differ
diff --git a/docs/screenshots/07-ci-green-after-fix.png b/docs/screenshots/07-ci-green-after-fix.png
new file mode 100644
index 0000000..b2d8c97
Binary files /dev/null and b/docs/screenshots/07-ci-green-after-fix.png differ
diff --git a/docs/screenshots/08-ci-run-success-3-jobs.png b/docs/screenshots/08-ci-run-success-3-jobs.png
new file mode 100644
index 0000000..546f182
Binary files /dev/null and b/docs/screenshots/08-ci-run-success-3-jobs.png differ
diff --git a/docs/screenshots/10-academy-learner-lab.png b/docs/screenshots/10-academy-learner-lab.png
new file mode 100644
index 0000000..dbe91a6
Binary files /dev/null and b/docs/screenshots/10-academy-learner-lab.png differ
diff --git a/docs/screenshots/11-academy-console-home.png b/docs/screenshots/11-academy-console-home.png
new file mode 100644
index 0000000..b5db141
Binary files /dev/null and b/docs/screenshots/11-academy-console-home.png differ
diff --git a/docs/screenshots/12-academy-ec2-dashboard.png b/docs/screenshots/12-academy-ec2-dashboard.png
new file mode 100644
index 0000000..e7510cb
Binary files /dev/null and b/docs/screenshots/12-academy-ec2-dashboard.png differ
diff --git a/docs/screenshots/13-academy-ecr.png b/docs/screenshots/13-academy-ecr.png
new file mode 100644
index 0000000..bbabfb8
Binary files /dev/null and b/docs/screenshots/13-academy-ecr.png differ
diff --git a/docs/screenshots/14-academy-vpc.png b/docs/screenshots/14-academy-vpc.png
new file mode 100644
index 0000000..80ee200
Binary files /dev/null and b/docs/screenshots/14-academy-vpc.png differ
diff --git a/docs/screenshots/15-academy-s3.png b/docs/screenshots/15-academy-s3.png
new file mode 100644
index 0000000..6a13983
Binary files /dev/null and b/docs/screenshots/15-academy-s3.png differ
diff --git a/docs/screenshots/16-academy-systems-manager.png b/docs/screenshots/16-academy-systems-manager.png
new file mode 100644
index 0000000..1312601
Binary files /dev/null and b/docs/screenshots/16-academy-systems-manager.png differ
diff --git a/docs/screenshots/17-academy-parameter-store.png b/docs/screenshots/17-academy-parameter-store.png
new file mode 100644
index 0000000..46a3728
Binary files /dev/null and b/docs/screenshots/17-academy-parameter-store.png differ
diff --git a/docs/screenshots/18-academy-cloudwatch.png b/docs/screenshots/18-academy-cloudwatch.png
new file mode 100644
index 0000000..a13fdec
Binary files /dev/null and b/docs/screenshots/18-academy-cloudwatch.png differ
diff --git a/docs/screenshots/19-academy-iam.png b/docs/screenshots/19-academy-iam.png
new file mode 100644
index 0000000..10099a6
Binary files /dev/null and b/docs/screenshots/19-academy-iam.png differ
diff --git a/docs/screenshots/20-academy-labrole.png b/docs/screenshots/20-academy-labrole.png
new file mode 100644
index 0000000..ad95944
Binary files /dev/null and b/docs/screenshots/20-academy-labrole.png differ
diff --git a/docs/screenshots/21-academy-labrole-policies.png b/docs/screenshots/21-academy-labrole-policies.png
new file mode 100644
index 0000000..e201269
Binary files /dev/null and b/docs/screenshots/21-academy-labrole-policies.png differ
diff --git a/docs/screenshots/30-terraform-init.png b/docs/screenshots/30-terraform-init.png
new file mode 100644
index 0000000..6c87436
Binary files /dev/null and b/docs/screenshots/30-terraform-init.png differ
diff --git a/docs/screenshots/31-terraform-apply-complete.png b/docs/screenshots/31-terraform-apply-complete.png
new file mode 100644
index 0000000..deb7a11
Binary files /dev/null and b/docs/screenshots/31-terraform-apply-complete.png differ
diff --git a/docs/screenshots/32-console-voc-cancel-cred-denied.png b/docs/screenshots/32-console-voc-cancel-cred-denied.png
new file mode 100644
index 0000000..68a3f42
Binary files /dev/null and b/docs/screenshots/32-console-voc-cancel-cred-denied.png differ
diff --git a/docs/screenshots/33-ec2-instance-running.png b/docs/screenshots/33-ec2-instance-running.png
new file mode 100644
index 0000000..8759cf4
Binary files /dev/null and b/docs/screenshots/33-ec2-instance-running.png differ
diff --git a/docs/screenshots/34-ecr-repo-created.png b/docs/screenshots/34-ecr-repo-created.png
new file mode 100644
index 0000000..4a22bea
Binary files /dev/null and b/docs/screenshots/34-ecr-repo-created.png differ
diff --git a/docs/screenshots/35-vpc-created.png b/docs/screenshots/35-vpc-created.png
new file mode 100644
index 0000000..4d35821
Binary files /dev/null and b/docs/screenshots/35-vpc-created.png differ
diff --git a/docs/screenshots/36-ec2-bootstrap-verified.png b/docs/screenshots/36-ec2-bootstrap-verified.png
new file mode 100644
index 0000000..f0edbc3
Binary files /dev/null and b/docs/screenshots/36-ec2-bootstrap-verified.png differ
diff --git a/docs/screenshots/37-pr-base-upstream-trap.png b/docs/screenshots/37-pr-base-upstream-trap.png
new file mode 100644
index 0000000..bda00ac
Binary files /dev/null and b/docs/screenshots/37-pr-base-upstream-trap.png differ
diff --git a/docs/screenshots/38-pr1-terraform.png b/docs/screenshots/38-pr1-terraform.png
new file mode 100644
index 0000000..3a1c444
Binary files /dev/null and b/docs/screenshots/38-pr1-terraform.png differ
diff --git a/docs/screenshots/39-pr1-checks-running.png b/docs/screenshots/39-pr1-checks-running.png
new file mode 100644
index 0000000..00e3c66
Binary files /dev/null and b/docs/screenshots/39-pr1-checks-running.png differ
diff --git a/docs/screenshots/40-local-prod-mode.png b/docs/screenshots/40-local-prod-mode.png
new file mode 100644
index 0000000..06a8a20
Binary files /dev/null and b/docs/screenshots/40-local-prod-mode.png differ
diff --git a/docs/screenshots/41-local-prod-snippet.png b/docs/screenshots/41-local-prod-snippet.png
new file mode 100644
index 0000000..b9e6aa2
Binary files /dev/null and b/docs/screenshots/41-local-prod-snippet.png differ
diff --git a/docs/screenshots/42-deploy-script-healthy.png b/docs/screenshots/42-deploy-script-healthy.png
new file mode 100644
index 0000000..c1d4470
Binary files /dev/null and b/docs/screenshots/42-deploy-script-healthy.png differ
diff --git a/docs/screenshots/43-cloudwatch-logs-cli.png b/docs/screenshots/43-cloudwatch-logs-cli.png
new file mode 100644
index 0000000..dc20d04
Binary files /dev/null and b/docs/screenshots/43-cloudwatch-logs-cli.png differ
diff --git a/docs/screenshots/44-csrf-403-over-http.png b/docs/screenshots/44-csrf-403-over-http.png
new file mode 100644
index 0000000..50b2a86
Binary files /dev/null and b/docs/screenshots/44-csrf-403-over-http.png differ
diff --git a/docs/screenshots/45-https-snippet-created.png b/docs/screenshots/45-https-snippet-created.png
new file mode 100644
index 0000000..04e246e
Binary files /dev/null and b/docs/screenshots/45-https-snippet-created.png differ
diff --git a/docs/screenshots/50-github-environment-production.png b/docs/screenshots/50-github-environment-production.png
new file mode 100644
index 0000000..7257e95
Binary files /dev/null and b/docs/screenshots/50-github-environment-production.png differ
diff --git a/docs/screenshots/51-github-actions-secrets.png b/docs/screenshots/51-github-actions-secrets.png
new file mode 100644
index 0000000..9818ba1
Binary files /dev/null and b/docs/screenshots/51-github-actions-secrets.png differ
diff --git a/docs/screenshots/52-cd-run-success.png b/docs/screenshots/52-cd-run-success.png
new file mode 100644
index 0000000..7b34c29
Binary files /dev/null and b/docs/screenshots/52-cd-run-success.png differ
diff --git a/docs/screenshots/53-cd-pr-checks.png b/docs/screenshots/53-cd-pr-checks.png
new file mode 100644
index 0000000..430c597
Binary files /dev/null and b/docs/screenshots/53-cd-pr-checks.png differ
diff --git a/docs/screenshots/54-tf-pr-checks.png b/docs/screenshots/54-tf-pr-checks.png
new file mode 100644
index 0000000..b2d666c
Binary files /dev/null and b/docs/screenshots/54-tf-pr-checks.png differ
diff --git a/docs/screenshots/55-tf-apply-waiting-approval.png b/docs/screenshots/55-tf-apply-waiting-approval.png
new file mode 100644
index 0000000..e3aa8ae
Binary files /dev/null and b/docs/screenshots/55-tf-apply-waiting-approval.png differ
diff --git a/docs/screenshots/60-monitoring-stack-up.png b/docs/screenshots/60-monitoring-stack-up.png
new file mode 100644
index 0000000..e200caf
Binary files /dev/null and b/docs/screenshots/60-monitoring-stack-up.png differ
diff --git a/docs/screenshots/61-prometheus-targets.png b/docs/screenshots/61-prometheus-targets.png
new file mode 100644
index 0000000..417efdc
Binary files /dev/null and b/docs/screenshots/61-prometheus-targets.png differ
diff --git a/docs/screenshots/62-prometheus-alerts.png b/docs/screenshots/62-prometheus-alerts.png
new file mode 100644
index 0000000..ff191f4
Binary files /dev/null and b/docs/screenshots/62-prometheus-alerts.png differ
diff --git a/docs/screenshots/63-grafana-dashboard.png b/docs/screenshots/63-grafana-dashboard.png
new file mode 100644
index 0000000..e0dbb13
Binary files /dev/null and b/docs/screenshots/63-grafana-dashboard.png differ
diff --git a/docs/screenshots/64-grafana-alert-firing.png b/docs/screenshots/64-grafana-alert-firing.png
new file mode 100644
index 0000000..5621e68
Binary files /dev/null and b/docs/screenshots/64-grafana-alert-firing.png differ
diff --git a/docs/screenshots/65-grafana-dashboard-final.png b/docs/screenshots/65-grafana-dashboard-final.png
new file mode 100644
index 0000000..130bee0
Binary files /dev/null and b/docs/screenshots/65-grafana-dashboard-final.png differ
diff --git a/docs/screenshots/70-final-ec2-instance.png b/docs/screenshots/70-final-ec2-instance.png
new file mode 100644
index 0000000..ff8d589
Binary files /dev/null and b/docs/screenshots/70-final-ec2-instance.png differ
diff --git a/docs/screenshots/71-final-security-group.png b/docs/screenshots/71-final-security-group.png
new file mode 100644
index 0000000..c6c869c
Binary files /dev/null and b/docs/screenshots/71-final-security-group.png differ
diff --git a/docs/screenshots/72-final-ecr-images.png b/docs/screenshots/72-final-ecr-images.png
new file mode 100644
index 0000000..c6beff5
Binary files /dev/null and b/docs/screenshots/72-final-ecr-images.png differ
diff --git a/docs/screenshots/73-final-ssm-parameter.png b/docs/screenshots/73-final-ssm-parameter.png
new file mode 100644
index 0000000..423bab5
Binary files /dev/null and b/docs/screenshots/73-final-ssm-parameter.png differ
diff --git a/docs/screenshots/74-final-s3-state-versions.png b/docs/screenshots/74-final-s3-state-versions.png
new file mode 100644
index 0000000..67baf1c
Binary files /dev/null and b/docs/screenshots/74-final-s3-state-versions.png differ
diff --git a/docs/screenshots/75-final-cloudwatch-overview.png b/docs/screenshots/75-final-cloudwatch-overview.png
new file mode 100644
index 0000000..190dcdc
Binary files /dev/null and b/docs/screenshots/75-final-cloudwatch-overview.png differ
diff --git a/docs/screenshots/76-final-cloudwatch-alarms.png b/docs/screenshots/76-final-cloudwatch-alarms.png
new file mode 100644
index 0000000..a80ff7a
Binary files /dev/null and b/docs/screenshots/76-final-cloudwatch-alarms.png differ
diff --git a/docs/screenshots/77-final-site-https.png b/docs/screenshots/77-final-site-https.png
new file mode 100644
index 0000000..cd4bbe0
Binary files /dev/null and b/docs/screenshots/77-final-site-https.png differ
diff --git a/docs/screenshots/78-final-cloudwatch-log-streams.png b/docs/screenshots/78-final-cloudwatch-log-streams.png
new file mode 100644
index 0000000..bbeeb5c
Binary files /dev/null and b/docs/screenshots/78-final-cloudwatch-log-streams.png differ
diff --git a/docs/screenshots/79-rollback-container-logs.png b/docs/screenshots/79-rollback-container-logs.png
new file mode 100644
index 0000000..448bebd
Binary files /dev/null and b/docs/screenshots/79-rollback-container-logs.png differ
diff --git a/docs/troubleshooting.md b/docs/troubleshooting.md
new file mode 100644
index 0000000..dcd118b
--- /dev/null
+++ b/docs/troubleshooting.md
@@ -0,0 +1,81 @@
+# Troubleshooting
+
+Last updated: 2026-09-23
+
+Every problem below actually happened while building CloudPulse. Each entry gives the symptom, the root cause and the fix.
+
+## Windows / PowerShell
+
+### 1. `terraform init`: "Invalid character encoding"
+- **Symptom:** `Invalid character encoding` / `Unterminated template string` in `outputs.tf`.
+- **Cause:** an em-dash in a description was written by PowerShell in a legacy encoding, and `` inside a string looked like a template.
+- **Fix:** keep `.tf` files ASCII-only.
+
+### 2. Terraform / AWS CLI: "No valid credential sources found"
+- **Symptom:** credentials file present but not read.
+- **Cause:** `Set-Content -Encoding UTF8` in Windows PowerShell 5 writes a **byte order mark**; the AWS SDK cannot parse the first line.
+- **Fix:** `[System.IO.File]::WriteAllText($path, $content)` (UTF-8 without BOM).
+
+### 3. `docker login` to ECR: `400 Bad Request`
+- **Cause:** piping `aws ecr get-login-password` into `docker login` in PowerShell 5 re-encodes the token.
+- **Fix:** run the pipe in cmd: `cmd /c "aws ecr get-login-password | docker login --username AWS --password-stdin "`.
+
+### 4. AWS CLI: `'charmap' codec can't encode character`
+- **Cause:** remote output contained an emoji (dpaste's startup banner) and the CLI prints with the Windows console code page; `PYTHONUTF8` is ignored by the bundled Python.
+- **Fix:** strip non-ASCII on the server: `... | tr -cd '\11\12\40-\176'`.
+
+### 5. `terraform plan` wants to modify `user_data` although nothing changed
+- **Symptom:** in-place update of the instance (which would restart it) after switching Git branches.
+- **Cause:** `user_data.sh` switched between CRLF and LF line endings; Terraform hashes the bytes.
+- **Fix:** `.gitattributes` with `*.sh text eol=lf` (and `*.tf`, `*.hcl`), then a one-time apply. Lesson learned: the repository already had a `.gitattributes`; **append** to existing config files instead of overwriting them.
+
+### 6. AWS CLI: "You must specify a region"
+- **Fix:** create `~/.aws/config` with `region = us-east-1`.
+
+## AWS Academy
+
+### 7. Console: `explicit deny ... policy/voc-cancel-cred`
+- **Cause:** a console tab opened in a previous lab session; Vocareum revokes older sessions.
+- **Fix:** close all console tabs and reopen the console from the Vocareum AWS link.
+
+### 8. OIDC for GitHub Actions: `AccessDenied` on `iam:CreateOpenIDConnectProvider`
+- **Cause:** the Learner Lab does not allow creating IAM identity providers or roles.
+- **Fix / trade-off:** use the lab's short-lived session credentials as GitHub secrets, refreshed by `scripts/refresh-github-aws-secrets.ps1` each session.
+
+## Terraform
+
+### 9. `InvalidBlockDeviceMapping: Volume of size 20GB is smaller than snapshot ... expect size >= 30GB`
+- **Cause:** the AMI filter `al2023-ami-*-x86_64` also matched `al2023-ami-ecs-hvm-...` (ECS-optimized, 30 GB snapshot) and `most_recent` picked it. The login banner said "Amazon Linux 2023 (ECS Optimized)".
+- **Fix:** filter `al2023-ami-2023.*-x86_64`. The replacement plan also updated both CloudWatch alarms (their `InstanceId` dimension), which is why plans are reviewed.
+
+### 10. Checkov passes but ignores some files
+- **Symptom:** `Parsing errors: 3` and CloudWatch resources never scanned.
+- **Cause:** three files contained byte `0x97` (a Windows-1252 em-dash) in comments. Terraform tolerated it; Checkov could not parse the files and skipped them.
+- **Fix:** convert every `.tf` file to ASCII.
+
+## Application / deployment
+
+### 11. `403 Forbidden - CSRF verification failed` when creating a snippet
+- **Cause:** dpaste sets `CSRF_COOKIE_SECURE = True` and `SESSION_COOKIE_SECURE = True`. Over plain HTTP the browser drops these cookies. It worked locally because browsers treat `localhost` as secure.
+- **Fix:** serve over HTTPS (Caddy + Let's Encrypt on an sslip.io hostname). dpaste already sets `SECURE_PROXY_SSL_HEADER`, so it trusts `X-Forwarded-Proto` from the proxy. Disabling the secure cookies was rejected.
+
+### 12. Snippets would have been lost on every redeploy
+- **Cause:** the deploy script mounted the volume at `/db`, but the image's `DATABASE_URL` is `sqlite:////data/dpaste.sqlite`.
+- **Fix:** read the path from the image (`docker image inspect --format '{{json .Config.Env}}'`) and mount at `/data`. Caught before the first deploy.
+
+### 13. ECR scan status `None`, three entries for one image
+- **Cause:** Docker Buildx adds a provenance attestation, so the push is an image index, which ECR basic scanning does not handle.
+- **Fix:** build with `--provenance=false --sbom=false`.
+
+### 14. `docker login` warning: password stored unencrypted in `/root/.docker/config.json`
+- **Fix:** Amazon ECR credential helper; `config.json` now only contains `credHelpers`.
+
+## Monitoring
+
+### 15. cAdvisor running but container panels empty
+- **Symptom:** only one anonymous series; logs show `failed to identify the read-write layer ID ... layerdb/mounts/...: no such file or directory`.
+- **Cause:** Docker Desktop uses the containerd image store; cAdvisor v0.49 expects the classic `overlay2` layout.
+- **Fix:** removed cAdvisor; the Grafana dashboard shows production EC2 CPU and network from CloudWatch instead.
+
+### Grafana: "Invalid username or password"
+- **Fix:** `docker exec cloudpulse-grafana grafana cli admin reset-admin-password `. Repeated failures trigger a short lockout.
\ No newline at end of file
diff --git a/docs/upstream-dpaste-README.md b/docs/upstream-dpaste-README.md
new file mode 100644
index 0000000..ee7f98b
--- /dev/null
+++ b/docs/upstream-dpaste-README.md
@@ -0,0 +1,24 @@
+Dpaste
+---
+
+[](https://github.com/DarrenOfficial/dpaste/actions/workflows/python.yml)
+[](https://hub.docker.com/r/darrenofficial/dpaste)
+
+
+----
+
+๐ Full documentation on [https://docs.dpaste.org](https://docs.dpaste.org)
+
+
+dpaste is a [pastebin](https://en.wikipedia.org/wiki/Pastebin) application written in [Python](https://www.python.org/) using the [Django](https://www.djangoproject.com/) framework. You can find a live installation on [dpaste.org.](https://dpaste.org)
+
+The project is intended to run standalone as any regular Django Project, but it's also possible to install it into an existing project as a typical Django application.
+
+
+The code is open source and available on Github: [https://github.com/darrenofficial/dpaste](https://github.com/darrenofficial/dpaste). If you found bugs, have problems or ideas with the project or the website installation, please create an *Issue* there.
+
+โ ๏ธ dpaste requires at a minimum Python 3.9 and Django 3.2.
+
+
+dpaste.org: https://dpaste.org/
+pastebin: https://en.wikipedia.org/wiki/Pastebin
diff --git a/minimal.docker-compose.yml b/minimal.docker-compose.yml
deleted file mode 100644
index b6c28e3..0000000
--- a/minimal.docker-compose.yml
+++ /dev/null
@@ -1,12 +0,0 @@
-services:
- dpaste:
- container_name: dpaste
- image: darrenofficial/dpaste:latest
- restart: unless-stopped
- environment:
- DATABASE_URL: sqlite:////db/dpaste.sqlite
- PORT: 8000
- volumes:
- - ./data/db:/db
- ports:
- - "8000:8000"
diff --git a/monitoring/docker-compose.yml b/monitoring/docker-compose.yml
index dcd3a7a..853d82f 100644
--- a/monitoring/docker-compose.yml
+++ b/monitoring/docker-compose.yml
@@ -32,7 +32,6 @@ services:
- ./blackbox/blackbox.yml:/etc/blackbox/blackbox.yml:ro
restart: unless-stopped
-
prometheus:
image: prom/prometheus:v3.2.1
container_name: cloudpulse-prometheus
@@ -58,7 +57,8 @@ services:
- ./grafana/dashboards:/var/lib/grafana/dashboards:ro
- grafana_data:/var/lib/grafana
# AWS Academy session credentials, read-only, for the CloudWatch datasource
- - ${USERPROFILE}/.aws:/usr/share/grafana/.aws:ro
+ # (HOME on Linux/macOS, USERPROFILE on Windows)
+ - ${HOME:-${USERPROFILE}}/.aws:/usr/share/grafana/.aws:ro
ports:
- "127.0.0.1:3000:3000"
depends_on:
diff --git a/renovate.json b/renovate.json
deleted file mode 100644
index f45d8f1..0000000
--- a/renovate.json
+++ /dev/null
@@ -1,5 +0,0 @@
-{
- "extends": [
- "config:base"
- ]
-}
diff --git a/ruff.toml b/ruff.toml
new file mode 100644
index 0000000..bd9ba67
--- /dev/null
+++ b/ruff.toml
@@ -0,0 +1,8 @@
+# Lint configuration for the inherited dpaste code (CI "lint" job, blocking).
+# Default rule set (pyflakes + pycodestyle errors). The dpaste sources are
+# upstream code, so style-only findings are ignored rather than rewriting them.
+extend-exclude = ["dpaste/migrations"]
+
+[lint]
+# E741 "ambiguous variable name" (single-letter names like `l`) in upstream views/highlighting
+ignore = ["E741"]
diff --git a/scripts/backup.sh b/scripts/backup.sh
new file mode 100755
index 0000000..47657c1
--- /dev/null
+++ b/scripts/backup.sh
@@ -0,0 +1,106 @@
+#!/bin/bash
+# CloudPulse SQLite backup / restore - runs ON the EC2 instance as root.
+# Shipped to /opt/cloudpulse/backup.sh by the CD pipeline; deploy.sh installs a
+# daily systemd timer that runs "backup.sh backup".
+#
+# backup.sh backup online backup of the dpaste database -> S3
+# backup.sh list list available backups
+# backup.sh restore replace the live database with a backup from S3
+#
+# The bucket name comes from SSM Parameter Store (/cloudpulse/backup/bucket,
+# created by Terraform). Credentials come from the instance role.
+set -euo pipefail
+
+CONTAINER="dpaste"
+DB_PATH="/data/dpaste.sqlite" # matches DATABASE_URL baked into the image
+VOLUME="dpaste_data"
+PREFIX="sqlite"
+BUCKET_PARAM="/cloudpulse/backup/bucket"
+STATE_DIR="/opt/cloudpulse"
+
+log() { echo "[backup $(date -u +%H:%M:%S)] $*"; }
+
+# Region from IMDSv2 (IMDSv1 is disabled on the instance)
+IMDS="http://169.254.169.254/latest"
+TOKEN=$(curl -sf -X PUT "${IMDS}/api/token" -H "X-aws-ec2-metadata-token-ttl-seconds: 60")
+REGION=$(curl -sf -H "X-aws-ec2-metadata-token: ${TOKEN}" "${IMDS}/meta-data/placement/region")
+BUCKET=$(aws ssm get-parameter --region "${REGION}" --name "${BUCKET_PARAM}" \
+ --query Parameter.Value --output text)
+
+WORK=$(mktemp -d)
+trap 'rm -rf "${WORK}"' EXIT
+
+backup() {
+ local stamp key
+ stamp=$(date -u +%Y%m%dT%H%M%SZ)
+ key="${PREFIX}/dpaste-${stamp}.sqlite.gz"
+
+ # SQLite online backup API: a consistent copy while dpaste keeps serving
+ # (copying the file directly could capture a half-written page).
+ docker exec "${CONTAINER}" python -c "
+import sqlite3
+src = sqlite3.connect('${DB_PATH}')
+dst = sqlite3.connect('/data/.backup.sqlite')
+src.backup(dst)
+dst.close(); src.close()
+"
+ docker cp "${CONTAINER}:/data/.backup.sqlite" "${WORK}/dpaste.sqlite"
+ docker exec "${CONTAINER}" rm -f /data/.backup.sqlite
+ gzip -9 "${WORK}/dpaste.sqlite"
+
+ # The bucket enforces TLS and encrypts at rest (SSE-S3)
+ aws s3 cp --region "${REGION}" --only-show-errors \
+ "${WORK}/dpaste.sqlite.gz" "s3://${BUCKET}/${key}"
+ log "Uploaded s3://${BUCKET}/${key} ($(stat -c %s "${WORK}/dpaste.sqlite.gz") bytes)"
+}
+
+list() {
+ aws s3 ls --region "${REGION}" "s3://${BUCKET}/${PREFIX}/"
+}
+
+restore() {
+ local key="${1:?usage: backup.sh restore }"
+ local image
+ image=$(cat "${STATE_DIR}/current_image")
+
+ aws s3 cp --region "${REGION}" --only-show-errors "s3://${BUCKET}/${key}" "${WORK}/restore.sqlite.gz"
+ gunzip "${WORK}/restore.sqlite.gz"
+
+ # Refuse to restore a corrupt file (throwaway container: no network, read-only;
+ # root only because the mktemp work directory is root-owned 0700)
+ docker run --rm --network none --read-only -u 0 -v "${WORK}:/restore:ro" \
+ --entrypoint python "${image}" -c "
+import sqlite3, sys
+r = sqlite3.connect('file:/restore/restore.sqlite?mode=ro', uri=True).execute('PRAGMA integrity_check').fetchone()[0]
+sys.exit(0 if r == 'ok' else 'integrity_check: ' + r)
+"
+ log "Integrity check passed"
+
+ # Keep a copy of the current database before overwriting it
+ backup
+
+ docker stop "${CONTAINER}" >/dev/null
+ # Copy into the volume with the app image itself (no extra image pulled),
+ # as root only for this step, then hand the file back to the dpaste user.
+ docker run --rm --network none -u 0 -v "${VOLUME}:/data" -v "${WORK}:/restore:ro" \
+ --entrypoint sh "${image}" -c \
+ "cp /restore/restore.sqlite ${DB_PATH} && chown dpaste:dpaste ${DB_PATH} && rm -f ${DB_PATH}-journal"
+ docker start "${CONTAINER}" >/dev/null
+
+ for _ in $(seq 1 30); do
+ if curl -sf -o /dev/null "http://localhost:8000/"; then
+ log "Restored ${key}; dpaste healthy"
+ return 0
+ fi
+ sleep 2
+ done
+ log "dpaste NOT healthy after restore"
+ return 1
+}
+
+case "${1:-backup}" in
+ backup) backup ;;
+ list) list ;;
+ restore) shift; restore "$@" ;;
+ *) echo "usage: backup.sh [backup|list|restore ]" >&2; exit 2 ;;
+esac
diff --git a/scripts/deploy.sh b/scripts/deploy.sh
index 1b74584..07b318d 100644
--- a/scripts/deploy.sh
+++ b/scripts/deploy.sh
@@ -1,15 +1,15 @@
#!/bin/bash
-# CloudPulse deploy script - runs ON the EC2 instance
-# (manually over SSH for now, via SSM Run Command from the CD pipeline).
+# CloudPulse deploy script - runs ON the EC2 instance as root
+# (sent by the CD pipeline through SSM Run Command; can also be run by hand).
# - Pulls a dpaste image tag from ECR and replaces the running container
# - Health-checks it and rolls back to the previous image on failure
# - Runs Caddy as the TLS-terminating reverse proxy: automatic HTTPS via
# Let's Encrypt on .sslip.io
+# - Installs daily systemd timers: expired-snippet cleanup and SQLite backup to S3
# Usage: sudo deploy.sh
set -euo pipefail
IMAGE_TAG="${1:?usage: deploy.sh }"
-REGION="us-east-1"
REPO="cloudpulse"
CONTAINER="dpaste"
APP_PORT=8000
@@ -30,6 +30,7 @@ IMDS="http://169.254.169.254/latest"
TOKEN=$(curl -sf -X PUT "${IMDS}/api/token" -H "X-aws-ec2-metadata-token-ttl-seconds: 300")
meta() { curl -sf -H "X-aws-ec2-metadata-token: ${TOKEN}" "${IMDS}/$1"; }
ACCOUNT_ID=$(meta dynamic/instance-identity/document | grep -oP '"accountId"\s*:\s*"\K[0-9]+')
+REGION=$(meta meta-data/placement/region)
PUBLIC_IP=$(meta meta-data/public-ipv4)
APP_HOST="${PUBLIC_IP//./-}.sslip.io"
@@ -161,13 +162,60 @@ EOF
log "Cleanup timer active"
}
+# Daily online SQLite backup to S3 (scripts/backup.sh, shipped next to this script
+# by the CD pipeline). Non-fatal: a backup problem must not fail a healthy deploy.
+ensure_backup_timer() {
+ if [ ! -x "${STATE_DIR}/backup.sh" ]; then
+ log "WARNING: ${STATE_DIR}/backup.sh missing - backup timer not installed"
+ return 0
+ fi
+ cat > /etc/systemd/system/cloudpulse-backup.service < /etc/systemd/system/cloudpulse-backup.timer </dev/null 2>&1; then
+ log "Backup timer active"
+ else
+ log "WARNING: could not enable backup timer"
+ fi
+}
+
+# Remove older release images from the instance disk; keep the running image and
+# the previous one (the rollback target). ECR keeps the full history.
+prune_old_images() {
+ docker images "${REGISTRY}/${REPO}" --format '{{.Repository}}:{{.Tag}}' \
+ | grep -vxF -e "${IMAGE}" -e "${PREVIOUS_IMAGE:-none}" \
+ | xargs -r docker rmi >/dev/null 2>&1 || true
+ docker image prune -f >/dev/null
+}
+
start_container "${IMAGE}"
if healthy; then
log "Healthy: ${IMAGE}"
echo "${IMAGE}" > "${STATE_DIR}/current_image"
ensure_proxy
ensure_cleanup_timer
- docker image prune -f >/dev/null
+ ensure_backup_timer
+ prune_old_images
log "Live at https://${APP_HOST}"
exit 0
fi
diff --git a/terraform/backend.tf b/terraform/backend.tf
index 2802be8..e4888bf 100644
--- a/terraform/backend.tf
+++ b/terraform/backend.tf
@@ -1,6 +1,10 @@
# Remote state in S3 (bucket bootstrapped by scripts/bootstrap-tfstate.ps1):
# versioned, SSE-S3 encrypted, public access blocked, TLS-only bucket policy.
# use_lockfile = S3-native state locking (Terraform >= 1.10); no DynamoDB table needed.
+#
+# Backend blocks cannot use variables. The bucket name contains the AWS account ID;
+# to use another account, override it at init time instead of editing this file:
+# terraform init -backend-config="bucket=cloudpulse-tfstate-"
terraform {
backend "s3" {
bucket = "cloudpulse-tfstate-530008597446"
@@ -9,4 +13,4 @@ terraform {
encrypt = true
use_lockfile = true
}
-}
\ No newline at end of file
+}
diff --git a/terraform/main.tf b/terraform/main.tf
index b39a3b5..822b960 100644
--- a/terraform/main.tf
+++ b/terraform/main.tf
@@ -1,7 +1,6 @@
# ============================================================
-# CloudPulse - Root Module
-# Wires together: networking ? compute ? monitoring
-# Infrastructure: Rayen Mabrouk
+# CloudPulse - Root module
+# Wires together: networking -> compute -> monitoring
# ============================================================
module "networking" {
@@ -14,15 +13,21 @@ module "networking" {
module "compute" {
source = "./modules/compute"
- project_name = var.project_name
- aws_region = var.aws_region
- subnet_id = module.networking.public_subnet_id
- security_group_id = module.networking.app_security_group_id
+ project_name = var.project_name
+ subnet_id = module.networking.public_subnet_id
+ security_group_id = module.networking.app_security_group_id
+ instance_type = var.instance_type
+ key_name = var.ssh_key_name
+ instance_profile_name = var.instance_profile_name
+ ecr_images_to_keep = var.ecr_images_to_keep
+ backup_retention_days = var.backup_retention_days
}
module "monitoring" {
source = "./modules/monitoring"
- project_name = var.project_name
- instance_id = module.compute.instance_id
+ project_name = var.project_name
+ instance_id = module.compute.instance_id
+ log_retention_days = var.log_retention_days
+ alarm_email = var.alarm_email
}
diff --git a/terraform/modules/compute/backups.tf b/terraform/modules/compute/backups.tf
new file mode 100644
index 0000000..30d82d6
--- /dev/null
+++ b/terraform/modules/compute/backups.tf
@@ -0,0 +1,125 @@
+# ============================================================
+# SQLite backups
+# dpaste stores its data in SQLite on the instance's EBS volume. Without a copy
+# elsewhere, losing the instance means losing every snippet. scripts/backup.sh
+# (run daily by a systemd timer that deploy.sh installs) takes an online SQLite
+# backup and uploads it here. Restore: scripts/backup.sh restore .
+# ============================================================
+
+data "aws_caller_identity" "current" {}
+
+resource "aws_s3_bucket" "backups" {
+ # checkov:skip=CKV_AWS_18:access logging needs a second log bucket; backups are written only by the instance role and every object is versioned
+ # checkov:skip=CKV_AWS_144:cross-region replication needs an IAM replication role, which cannot be created in AWS Academy
+ # checkov:skip=CKV_AWS_145:SSE-S3 (AES-256) encryption; a customer-managed KMS key adds cost with no benefit in a single-account lab
+ # checkov:skip=CKV2_AWS_62:no consumer for event notifications
+ bucket = "${var.project_name}-backups-${data.aws_caller_identity.current.account_id}"
+
+ # Lab teardown must be able to delete the bucket with its objects.
+ # In a real environment this would be false (and backups would be replicated).
+ force_destroy = true
+
+ tags = {
+ Name = "${var.project_name}-backups"
+ }
+}
+
+resource "aws_s3_bucket_public_access_block" "backups" {
+ bucket = aws_s3_bucket.backups.id
+
+ block_public_acls = true
+ block_public_policy = true
+ ignore_public_acls = true
+ restrict_public_buckets = true
+}
+
+resource "aws_s3_bucket_versioning" "backups" {
+ bucket = aws_s3_bucket.backups.id
+
+ versioning_configuration {
+ status = "Enabled"
+ }
+}
+
+resource "aws_s3_bucket_server_side_encryption_configuration" "backups" {
+ bucket = aws_s3_bucket.backups.id
+
+ rule {
+ apply_server_side_encryption_by_default {
+ sse_algorithm = "AES256"
+ }
+ bucket_key_enabled = true
+ }
+}
+
+resource "aws_s3_bucket_lifecycle_configuration" "backups" {
+ bucket = aws_s3_bucket.backups.id
+
+ rule {
+ id = "expire-old-backups"
+ status = "Enabled"
+
+ filter {}
+
+ expiration {
+ days = var.backup_retention_days
+ }
+
+ noncurrent_version_expiration {
+ noncurrent_days = 7
+ }
+
+ abort_incomplete_multipart_upload {
+ days_after_initiation = 1
+ }
+ }
+
+ # Lifecycle rules on a versioned bucket must be created after versioning
+ depends_on = [aws_s3_bucket_versioning.backups]
+}
+
+# Deny any request that is not made over TLS
+data "aws_iam_policy_document" "backups_tls_only" {
+ statement {
+ sid = "DenyInsecureTransport"
+ effect = "Deny"
+ actions = ["s3:*"]
+ resources = [
+ aws_s3_bucket.backups.arn,
+ "${aws_s3_bucket.backups.arn}/*",
+ ]
+
+ principals {
+ type = "*"
+ identifiers = ["*"]
+ }
+
+ condition {
+ test = "Bool"
+ variable = "aws:SecureTransport"
+ values = ["false"]
+ }
+ }
+}
+
+resource "aws_s3_bucket_policy" "backups" {
+ bucket = aws_s3_bucket.backups.id
+ policy = data.aws_iam_policy_document.backups_tls_only.json
+
+ # Applying a bucket policy while the public access block is being created can fail
+ depends_on = [aws_s3_bucket_public_access_block.backups]
+}
+
+# The backup script discovers the bucket through Parameter Store instead of
+# hard-coding a naming convention.
+resource "aws_ssm_parameter" "backup_bucket" {
+ # checkov:skip=CKV2_AWS_34:bucket name is not a secret, a plain String parameter is intended
+ name = "/${var.project_name}/backup/bucket"
+ description = "S3 bucket used by scripts/backup.sh"
+ type = "String"
+ value = aws_s3_bucket.backups.bucket
+
+ tags = {
+ Name = "${var.project_name}-backup-bucket"
+ }
+}
diff --git a/terraform/modules/compute/main.tf b/terraform/modules/compute/main.tf
index 330cbbd..f60f260 100644
--- a/terraform/modules/compute/main.tf
+++ b/terraform/modules/compute/main.tf
@@ -1,6 +1,6 @@
# ============================================================
-# CloudPulse ? Compute Module
-# Creates: ECR repository, EC2 instance with Docker
+# CloudPulse - Compute module
+# Creates: ECR repository (+ lifecycle policy), EC2 instance with Docker
# ============================================================
# --- Find latest Amazon Linux 2023 AMI ---
@@ -19,9 +19,10 @@ data "aws_ami" "amazon_linux" {
}
}
-# --- Reference pre-existing Academy LabInstanceProfile ---
+# --- Pre-existing instance profile (AWS Academy: LabInstanceProfile -> LabRole) ---
+# The Learner Lab denies iam:CreateRole, so a least-privilege role cannot be created here.
data "aws_iam_instance_profile" "lab" {
- name = "LabInstanceProfile"
+ name = var.instance_profile_name
}
# --- ECR Repository ---
@@ -40,6 +41,39 @@ resource "aws_ecr_repository" "app" {
}
}
+# Tags are immutable commit SHAs, so images would pile up forever without this.
+# Keeps the newest N images (enough to roll back several releases) and drops
+# untagged leftovers after a day.
+resource "aws_ecr_lifecycle_policy" "app" {
+ repository = aws_ecr_repository.app.name
+
+ policy = jsonencode({
+ rules = [
+ {
+ rulePriority = 1
+ description = "Expire untagged images after 1 day"
+ selection = {
+ tagStatus = "untagged"
+ countType = "sinceImagePushed"
+ countUnit = "days"
+ countNumber = 1
+ }
+ action = { type = "expire" }
+ },
+ {
+ rulePriority = 2
+ description = "Keep only the ${var.ecr_images_to_keep} most recent images"
+ selection = {
+ tagStatus = "any"
+ countType = "imageCountMoreThan"
+ countNumber = var.ecr_images_to_keep
+ }
+ action = { type = "expire" }
+ }
+ ]
+ })
+}
+
# --- EC2 Instance ---
resource "aws_instance" "app" {
# checkov:skip=CKV_AWS_135:t3 instance types are EBS-optimized by default; the flag does not apply
@@ -51,7 +85,7 @@ resource "aws_instance" "app" {
key_name = var.key_name
monitoring = true # 1-minute CloudWatch metrics for faster alarms
- # Enforce IMDSv2 ? prevents SSRF token theft
+ # Enforce IMDSv2 - mitigates SSRF-based credential theft
metadata_options {
http_endpoint = "enabled"
http_tokens = "required"
diff --git a/terraform/modules/compute/outputs.tf b/terraform/modules/compute/outputs.tf
index 48d9c1d..4a19381 100644
--- a/terraform/modules/compute/outputs.tf
+++ b/terraform/modules/compute/outputs.tf
@@ -17,3 +17,13 @@ output "ecr_repository_url" {
description = "ECR repository URL"
value = aws_ecr_repository.app.repository_url
}
+
+output "backup_bucket_name" {
+ description = "S3 bucket holding SQLite backups"
+ value = aws_s3_bucket.backups.bucket
+}
+
+output "secret_key_parameter_name" {
+ description = "SSM parameter name of the Django SECRET_KEY"
+ value = aws_ssm_parameter.django_secret_key.name
+}
diff --git a/terraform/modules/compute/variables.tf b/terraform/modules/compute/variables.tf
index 7ea330e..8ae3384 100644
--- a/terraform/modules/compute/variables.tf
+++ b/terraform/modules/compute/variables.tf
@@ -3,11 +3,6 @@ variable "project_name" {
type = string
}
-variable "aws_region" {
- description = "AWS region"
- type = string
-}
-
variable "subnet_id" {
description = "Subnet ID for the EC2 instance"
type = string
@@ -25,13 +20,25 @@ variable "instance_type" {
}
variable "key_name" {
- description = "SSH key pair name"
+ description = "Existing EC2 key pair name (changing it replaces the instance)"
type = string
default = "vockey"
}
-variable "app_port" {
- description = "Application port"
+variable "instance_profile_name" {
+ description = "Existing IAM instance profile attached to the instance"
+ type = string
+ default = "LabInstanceProfile"
+}
+
+variable "ecr_images_to_keep" {
+ description = "Number of most recent images kept in ECR"
+ type = number
+ default = 10
+}
+
+variable "backup_retention_days" {
+ description = "Days a SQLite backup object is kept"
type = number
- default = 8000
+ default = 14
}
diff --git a/terraform/modules/compute/versions.tf b/terraform/modules/compute/versions.tf
index 11b2b8a..3a4346d 100644
--- a/terraform/modules/compute/versions.tf
+++ b/terraform/modules/compute/versions.tf
@@ -1,5 +1,11 @@
terraform {
+ required_version = ">= 1.10"
+
required_providers {
+ aws = {
+ source = "hashicorp/aws"
+ version = "~> 5.0"
+ }
random = {
source = "hashicorp/random"
version = "~> 3.6"
diff --git a/terraform/modules/monitoring/main.tf b/terraform/modules/monitoring/main.tf
index 9b8aade..0ff986b 100644
--- a/terraform/modules/monitoring/main.tf
+++ b/terraform/modules/monitoring/main.tf
@@ -1,47 +1,110 @@
# ============================================================
-# CloudPulse - Monitoring Module
-# Creates: CloudWatch log group, CPU alarm, status check alarm
+# CloudPulse - Monitoring module
+# Creates: CloudWatch log group, 5xx metric filter, three alarms
+# (instance status, CPU, HTTP 5xx), optional SNS email notifications
# ============================================================
resource "aws_cloudwatch_log_group" "app" {
# checkov:skip=CKV_AWS_158:CloudWatch Logs encrypts log data at rest by default; a customer-managed key would add a key policy to maintain
- # checkov:skip=CKV_AWS_338:7-day retention chosen deliberately to limit cost in a lab environment
- name = "/cloudpulse/dpaste"
- retention_in_days = 7
+ # checkov:skip=CKV_AWS_338:short retention chosen deliberately to limit cost in a lab environment
+ name = "/${var.project_name}/dpaste"
+ retention_in_days = var.log_retention_days
tags = {
Name = "${var.project_name}-logs"
}
}
-resource "aws_cloudwatch_metric_alarm" "cpu_high" {
- alarm_name = "${var.project_name}-cpu-high"
+# --- Optional alarm notifications ---
+# Created only when alarm_email is set. The subscription must be confirmed from the email.
+resource "aws_sns_topic" "alarms" {
+ # checkov:skip=CKV_AWS_26:CloudWatch alarms cannot publish to a topic encrypted with the AWS-managed aws/sns key, and a customer-managed key is not worth its cost here; messages only contain alarm metadata
+ count = var.alarm_email == "" ? 0 : 1
+ name = "${var.project_name}-alarms"
+
+ tags = {
+ Name = "${var.project_name}-alarms"
+ }
+}
+
+resource "aws_sns_topic_subscription" "alarm_email" {
+ count = var.alarm_email == "" ? 0 : 1
+ topic_arn = aws_sns_topic.alarms[0].arn
+ protocol = "email"
+ endpoint = var.alarm_email
+}
+
+locals {
+ alarm_actions = aws_sns_topic.alarms[*].arn
+}
+
+# --- Instance health: EC2 system or instance status check failing ---
+resource "aws_cloudwatch_metric_alarm" "status_check" {
+ alarm_name = "${var.project_name}-status-check"
comparison_operator = "GreaterThanThreshold"
evaluation_periods = 2
- metric_name = "CPUUtilization"
+ metric_name = "StatusCheckFailed"
namespace = "AWS/EC2"
period = 300
- statistic = "Average"
- threshold = 80
- alarm_description = "CPU utilization exceeds 80% for 10 minutes"
+ statistic = "Maximum"
+ threshold = 0
+ alarm_description = "EC2 instance status check failed (hardware, network or OS problem)"
+ alarm_actions = local.alarm_actions
+ ok_actions = local.alarm_actions
dimensions = {
InstanceId = var.instance_id
}
}
-resource "aws_cloudwatch_metric_alarm" "status_check" {
- alarm_name = "${var.project_name}-status-check"
+# --- Capacity: sustained CPU on a burstable instance ---
+resource "aws_cloudwatch_metric_alarm" "cpu_high" {
+ alarm_name = "${var.project_name}-cpu-high"
comparison_operator = "GreaterThanThreshold"
evaluation_periods = 2
- metric_name = "StatusCheckFailed"
+ metric_name = "CPUUtilization"
namespace = "AWS/EC2"
period = 300
- statistic = "Maximum"
- threshold = 0
- alarm_description = "EC2 instance status check failed"
+ statistic = "Average"
+ threshold = 80
+ alarm_description = "CPU utilization above 80% for 10 minutes (a t3.micro will soon run out of CPU credits)"
+ alarm_actions = local.alarm_actions
+ ok_actions = local.alarm_actions
dimensions = {
InstanceId = var.instance_id
}
}
+
+# --- Application health: HTTP 5xx answered by Caddy ---
+# Caddy writes one JSON access-log line per request to the same log group.
+# A 502 is what users get when the dpaste container is down or crashing, so this
+# catches application failures that the EC2 status check cannot see.
+resource "aws_cloudwatch_log_metric_filter" "http_5xx" {
+ name = "${var.project_name}-http-5xx"
+ log_group_name = aws_cloudwatch_log_group.app.name
+ pattern = "{ $.status >= 500 }"
+
+ metric_transformation {
+ name = "Http5xxCount"
+ namespace = "CloudPulse"
+ value = "1"
+ default_value = "0"
+ unit = "Count"
+ }
+}
+
+resource "aws_cloudwatch_metric_alarm" "http_5xx" {
+ alarm_name = "${var.project_name}-http-5xx"
+ comparison_operator = "GreaterThanOrEqualToThreshold"
+ evaluation_periods = 1
+ metric_name = aws_cloudwatch_log_metric_filter.http_5xx.metric_transformation[0].name
+ namespace = aws_cloudwatch_log_metric_filter.http_5xx.metric_transformation[0].namespace
+ period = 300
+ statistic = "Sum"
+ threshold = 5
+ treat_missing_data = "notBreaching" # no traffic is not an error
+ alarm_description = "5 or more HTTP 5xx responses in 5 minutes (502 = dpaste container down behind Caddy)"
+ alarm_actions = local.alarm_actions
+ ok_actions = local.alarm_actions
+}
diff --git a/terraform/modules/monitoring/outputs.tf b/terraform/modules/monitoring/outputs.tf
index 918dbe4..4ad46b2 100644
--- a/terraform/modules/monitoring/outputs.tf
+++ b/terraform/modules/monitoring/outputs.tf
@@ -2,3 +2,12 @@ output "log_group_name" {
description = "CloudWatch log group name"
value = aws_cloudwatch_log_group.app.name
}
+
+output "alarm_names" {
+ description = "Names of the CloudWatch alarms"
+ value = [
+ aws_cloudwatch_metric_alarm.status_check.alarm_name,
+ aws_cloudwatch_metric_alarm.cpu_high.alarm_name,
+ aws_cloudwatch_metric_alarm.http_5xx.alarm_name,
+ ]
+}
diff --git a/terraform/modules/monitoring/variables.tf b/terraform/modules/monitoring/variables.tf
index 02ee3b7..26f4b2b 100644
--- a/terraform/modules/monitoring/variables.tf
+++ b/terraform/modules/monitoring/variables.tf
@@ -7,3 +7,15 @@ variable "instance_id" {
description = "EC2 instance ID to monitor"
type = string
}
+
+variable "log_retention_days" {
+ description = "CloudWatch Logs retention in days"
+ type = number
+ default = 7
+}
+
+variable "alarm_email" {
+ description = "Email notified on alarm state changes; empty string = no SNS topic"
+ type = string
+ default = ""
+}
diff --git a/terraform/modules/monitoring/versions.tf b/terraform/modules/monitoring/versions.tf
new file mode 100644
index 0000000..f5e9311
--- /dev/null
+++ b/terraform/modules/monitoring/versions.tf
@@ -0,0 +1,10 @@
+terraform {
+ required_version = ">= 1.10"
+
+ required_providers {
+ aws = {
+ source = "hashicorp/aws"
+ version = "~> 5.0"
+ }
+ }
+}
diff --git a/terraform/modules/networking/main.tf b/terraform/modules/networking/main.tf
index dae483d..9f93de8 100644
--- a/terraform/modules/networking/main.tf
+++ b/terraform/modules/networking/main.tf
@@ -71,13 +71,17 @@ resource "aws_security_group" "app" {
description = "Security group for CloudPulse application"
vpc_id = aws_vpc.main.id
- # SSH - restricted to your IP only
- ingress {
- description = "SSH from allowed IP"
- from_port = 22
- to_port = 22
- protocol = "tcp"
- cidr_blocks = [var.allowed_ssh_cidr]
+ # SSH - break-glass only, from a single CIDR, and only when allowed_ssh_cidr is set.
+ # Normal operations (deploys, shell access) go through SSM, which needs no inbound port.
+ dynamic "ingress" {
+ for_each = var.allowed_ssh_cidr == "" ? [] : [var.allowed_ssh_cidr]
+ content {
+ description = "SSH from allowed IP (break-glass)"
+ from_port = 22
+ to_port = 22
+ protocol = "tcp"
+ cidr_blocks = [ingress.value]
+ }
}
# HTTP - Caddy redirects to HTTPS and answers Let's Encrypt HTTP-01 challenges
@@ -119,4 +123,4 @@ resource "aws_default_security_group" "default" {
tags = {
Name = "${var.project_name}-default-sg-locked"
}
-}
\ No newline at end of file
+}
diff --git a/terraform/modules/networking/outputs.tf b/terraform/modules/networking/outputs.tf
index fcb8bc3..353e7d9 100644
--- a/terraform/modules/networking/outputs.tf
+++ b/terraform/modules/networking/outputs.tf
@@ -12,3 +12,8 @@ output "app_security_group_id" {
description = "ID of the application security group"
value = aws_security_group.app.id
}
+
+output "ssh_enabled" {
+ description = "Whether the security group allows SSH"
+ value = var.allowed_ssh_cidr != ""
+}
diff --git a/terraform/modules/networking/variables.tf b/terraform/modules/networking/variables.tf
index 690f31f..a68bfcf 100644
--- a/terraform/modules/networking/variables.tf
+++ b/terraform/modules/networking/variables.tf
@@ -16,12 +16,7 @@ variable "public_subnet_cidr" {
}
variable "allowed_ssh_cidr" {
- description = "CIDR block allowed to SSH into EC2"
+ description = "CIDR allowed to SSH into the instance; empty string = no SSH rule"
type = string
-}
-
-variable "app_port" {
- description = "Application port exposed by the container"
- type = number
- default = 8000
+ default = ""
}
diff --git a/terraform/modules/networking/versions.tf b/terraform/modules/networking/versions.tf
new file mode 100644
index 0000000..f5e9311
--- /dev/null
+++ b/terraform/modules/networking/versions.tf
@@ -0,0 +1,10 @@
+terraform {
+ required_version = ">= 1.10"
+
+ required_providers {
+ aws = {
+ source = "hashicorp/aws"
+ version = "~> 5.0"
+ }
+ }
+}
diff --git a/terraform/outputs.tf b/terraform/outputs.tf
index 62805df..754aa01 100644
--- a/terraform/outputs.tf
+++ b/terraform/outputs.tf
@@ -1,19 +1,59 @@
+output "app_url" {
+ description = "Public HTTPS URL (sslip.io hostname derived from the instance public IP)"
+ value = "https://${replace(module.compute.instance_public_ip, ".", "-")}.sslip.io"
+}
+
+output "instance_id" {
+ description = "EC2 instance ID (target for SSM Run Command and Session Manager)"
+ value = module.compute.instance_id
+}
+
output "instance_public_ip" {
- description = "Public IP of the EC2 instance running dpaste"
+ description = "Public IP of the EC2 instance (changes when the Learner Lab restarts the instance)"
value = module.compute.instance_public_ip
}
+output "ssm_session_command" {
+ description = "Open a shell on the instance without SSH (requires the Session Manager plugin)"
+ value = "aws ssm start-session --target ${module.compute.instance_id} --region ${var.aws_region}"
+}
+
output "ecr_repository_url" {
- description = "ECR repository URL for Docker image push"
+ description = "ECR repository URL for image pushes"
value = module.compute.ecr_repository_url
}
+output "backup_bucket_name" {
+ description = "S3 bucket holding the daily SQLite backups"
+ value = module.compute.backup_bucket_name
+}
+
+output "secret_key_parameter_name" {
+ description = "SSM Parameter Store name of the Django SECRET_KEY (the value is never output)"
+ value = module.compute.secret_key_parameter_name
+}
+
+output "vpc_id" {
+ description = "VPC ID"
+ value = module.networking.vpc_id
+}
+
+output "app_security_group_id" {
+ description = "Security group attached to the instance"
+ value = module.networking.app_security_group_id
+}
+
+output "ssh_enabled" {
+ description = "Whether port 22 is open to allowed_ssh_cidr"
+ value = module.networking.ssh_enabled
+}
+
output "log_group_name" {
- description = "CloudWatch log group for container logs"
+ description = "CloudWatch log group for container logs (dpaste + Caddy)"
value = module.monitoring.log_group_name
}
-output "app_url" {
- description = "Public HTTPS URL (sslip.io hostname derived from the instance IP)"
- value = "https://${replace(module.compute.instance_public_ip, ".", "-")}.sslip.io"
-}
\ No newline at end of file
+output "alarm_names" {
+ description = "CloudWatch alarms watching the deployment"
+ value = module.monitoring.alarm_names
+}
diff --git a/terraform/providers.tf b/terraform/providers.tf
index 8772ef3..0cf5b26 100644
--- a/terraform/providers.tf
+++ b/terraform/providers.tf
@@ -1,9 +1,9 @@
# ============================================================
-# CloudPulse - Terraform Provider Configuration
-# Infrastructure: Rayen Mabrouk
+# CloudPulse - Terraform and provider configuration
# ============================================================
terraform {
+ # >= 1.10 is required for S3-native state locking (use_lockfile in backend.tf)
required_version = ">= 1.10"
required_providers {
@@ -11,9 +11,24 @@ terraform {
source = "hashicorp/aws"
version = "~> 5.0"
}
+ random = {
+ source = "hashicorp/random"
+ version = "~> 3.6"
+ }
}
}
provider "aws" {
region = var.aws_region
+
+ # Applied to every taggable resource: cost allocation, ownership, and a quick
+ # way to tell Terraform-managed resources from console-created ones.
+ default_tags {
+ tags = {
+ Project = var.project_name
+ Environment = var.environment
+ ManagedBy = "terraform"
+ Repository = "github.com/rayenmabrouk/cloudpulse"
+ }
+ }
}
diff --git a/terraform/terraform.tfvars.example b/terraform/terraform.tfvars.example
index 8a2804c..889b48b 100644
--- a/terraform/terraform.tfvars.example
+++ b/terraform/terraform.tfvars.example
@@ -1,3 +1,13 @@
-aws_region = "us-east-1"
-project_name = "cloudpulse"
-allowed_ssh_cidr = "YOUR_PUBLIC_IP/32"
+# Copy to terraform.tfvars (git-ignored) and adjust. Every value has a default,
+# so this file is only needed to change them.
+
+aws_region = "us-east-1"
+project_name = "cloudpulse"
+
+# Break-glass SSH from a single IP. Leave empty to keep port 22 closed
+# (day-to-day operations use SSM Run Command / Session Manager).
+allowed_ssh_cidr = ""
+# allowed_ssh_cidr = "203.0.113.10/32"
+
+# Receive an email when an alarm fires (confirm the SNS subscription email first).
+alarm_email = ""
diff --git a/terraform/variables.tf b/terraform/variables.tf
index 4404e8e..e437d98 100644
--- a/terraform/variables.tf
+++ b/terraform/variables.tf
@@ -1,20 +1,90 @@
# ============================================================
-# CloudPulse - Root Variables
+# CloudPulse - Root variables
+# Defaults match the AWS Academy Learner Lab this project runs in.
# ============================================================
variable "aws_region" {
- description = "AWS region for all resources"
+ description = "AWS region for all resources (the Learner Lab allows us-east-1 and us-west-2)"
type = string
default = "us-east-1"
}
variable "project_name" {
- description = "Project name used for resource naming and tagging"
+ description = "Project name used for resource names and tags"
type = string
default = "cloudpulse"
+
+ validation {
+ condition = can(regex("^[a-z][a-z0-9-]{2,20}$", var.project_name))
+ error_message = "project_name must be 3-21 characters: lowercase letters, digits and hyphens, starting with a letter."
+ }
+}
+
+variable "environment" {
+ description = "Environment name, used in tags"
+ type = string
+ default = "dev"
}
variable "allowed_ssh_cidr" {
- description = "CIDR block allowed to SSH into EC2 (your IP)"
+ description = "Single IPv4 CIDR allowed to SSH (break-glass only; operations use SSM). Empty string = port 22 closed."
+ type = string
+ default = ""
+
+ validation {
+ condition = var.allowed_ssh_cidr == "" || (can(cidrhost(var.allowed_ssh_cidr, 0)) && var.allowed_ssh_cidr != "0.0.0.0/0")
+ error_message = "allowed_ssh_cidr must be empty (SSH disabled) or a valid CIDR such as 203.0.113.10/32. 0.0.0.0/0 is refused."
+ }
+}
+
+variable "instance_type" {
+ description = "EC2 instance type (the Learner Lab limits instance sizes)"
+ type = string
+ default = "t3.micro"
+}
+
+variable "ssh_key_name" {
+ description = "Existing EC2 key pair name (vockey is pre-created in the Learner Lab). Changing it replaces the instance."
+ type = string
+ default = "vockey"
+}
+
+variable "instance_profile_name" {
+ description = "Existing IAM instance profile for the EC2 instance. The Learner Lab forbids creating IAM roles, so the pre-created LabInstanceProfile is used."
+ type = string
+ default = "LabInstanceProfile"
+}
+
+variable "ecr_images_to_keep" {
+ description = "Number of most recent images kept in ECR; older ones are expired by a lifecycle policy"
+ type = number
+ default = 10
+
+ validation {
+ condition = var.ecr_images_to_keep >= 2
+ error_message = "Keep at least 2 images so the previous release stays available for rollback."
+ }
+}
+
+variable "backup_retention_days" {
+ description = "Days a SQLite backup is kept in the backup bucket before it expires"
+ type = number
+ default = 14
+}
+
+variable "log_retention_days" {
+ description = "CloudWatch Logs retention for container logs"
+ type = number
+ default = 7
+
+ validation {
+ condition = contains([1, 3, 5, 7, 14, 30, 60, 90, 120, 150, 180, 365], var.log_retention_days)
+ error_message = "log_retention_days must be a value CloudWatch Logs accepts (1, 3, 5, 7, 14, 30, 60, 90, ...)."
+ }
+}
+
+variable "alarm_email" {
+ description = "Email address notified when an alarm changes state (SNS). Empty string = no notifications."
type = string
+ default = ""
}
diff --git a/tox.ini b/tox.ini
deleted file mode 100644
index 5928e8b..0000000
--- a/tox.ini
+++ /dev/null
@@ -1,44 +0,0 @@
-[tox]
-toxworkdir=/tmp/tox/dpaste
-skip_missing_interpreters=True
-envlist=
- readme
- coverage_setup
- py{38,39,310,311,312}-django-{32,40,41,42,50}
- coverage_report
-
-[testenv]
-install_command =
- pip install {opts} {packages}
-commands=
- pytest dpaste
-deps=
- # Django versions
- django-32: django>=3.2,<4.0
- django-40: django>=4.0,<4.1
- django-41: django>=4.1,<4.2
- django-42: django>=4.2,<5.0
- django-50: django>=5.0,<5.1
-
-[testenv:coverage_setup]
-skip_install = True
-deps = coverage
-basepython = python3.8
-commands = coverage erase
-
-[testenv:coverage_report]
-skip_install = True
-deps = coverage
-basepython = python3.8
-commands=
- coverage report
- coverage html
-
-[testenv:readme]
-skip_install = True
-deps =
- docutils
- Pygments
-commands =
- python -m docutils.core --writer=html5 --report=info --halt=warning README.rst /dev/null
- python -m docutils.core --writer=html5 --report=info --halt=warning CHANGELOG.rst /dev/null