diff --git a/.github/workflows/pr.preview-deploy.yml b/.github/workflows/pr.preview-deploy.yml deleted file mode 100644 index 9d2038621..000000000 --- a/.github/workflows/pr.preview-deploy.yml +++ /dev/null @@ -1,430 +0,0 @@ -name: Preview Deploy - -on: - pull_request: - types: [labeled] - workflow_dispatch: - inputs: - pr_number: - description: "PR number to simulate (for manual testing)" - required: true - default: "1" - m365: - description: "Include PowerShell service for M365 scans" - required: false - default: "false" - -permissions: - contents: read - pull-requests: write - issues: write - packages: write - -jobs: - # ───────────────────────────────────────────────────────────────────────────── - # Job 0 — Gate: validate trigger, resolve PR number, remove consumed label - # ───────────────────────────────────────────────────────────────────────────── - gate: - runs-on: ubuntu-latest - if: >- - github.event_name == 'workflow_dispatch' || - (github.event_name == 'pull_request' && - (github.event.label.name == 'deploy-preview' || - github.event.label.name == 'deploy-preview-m365')) - outputs: - pr-number: ${{ steps.pr.outputs.number }} - short-sha: ${{ steps.sha.outputs.short }} - m365: ${{ steps.pr.outputs.m365 }} - - steps: - - name: Resolve PR number - id: pr - run: | - if [ "${{ github.event_name }}" = "pull_request" ]; then - echo "number=${{ github.event.number }}" >> "$GITHUB_OUTPUT" - if [ "${{ github.event.label.name }}" = "deploy-preview-m365" ]; then - echo "m365=true" >> "$GITHUB_OUTPUT" - else - echo "m365=false" >> "$GITHUB_OUTPUT" - fi - else - echo "number=${{ inputs.pr_number }}" >> "$GITHUB_OUTPUT" - echo "m365=${{ inputs.m365 }}" >> "$GITHUB_OUTPUT" - fi - - - name: Resolve short SHA - id: sha - run: | - if [ "${{ github.event_name }}" = "pull_request" ]; then - echo "short=$(echo '${{ github.event.pull_request.head.sha }}' | cut -c1-7)" >> "$GITHUB_OUTPUT" - else - echo "short=$(echo '${{ github.sha }}' | cut -c1-7)" >> "$GITHUB_OUTPUT" - fi - - - name: Remove triggering label - if: github.event_name == 'pull_request' - uses: actions/github-script@v7 - with: - script: | - await github.rest.issues.removeLabel({ - owner: context.repo.owner, - repo: context.repo.repo, - issue_number: context.payload.pull_request.number, - name: context.payload.label.name, - }); - - # ───────────────────────────────────────────────────────────────────────────── - # Job 1 — Build backend and worker images, push to GHCR - # Uses GitHub Actions cache (type=gha) for layer reuse so no Docker Hub token needed - # Only GITHUB_TOKEN is required (packages: write permission above) - # ───────────────────────────────────────────────────────────────────────────── - build-images: - runs-on: ubuntu-latest - needs: gate - - steps: - - name: Checkout - uses: actions/checkout@v4 - - - name: Set up Docker Buildx - uses: docker/setup-buildx-action@v3 - - - name: Log in to GHCR - uses: docker/login-action@v3 - with: - registry: ghcr.io - username: ${{ github.actor }} - password: ${{ secrets.GITHUB_TOKEN }} - - - name: Set lowercase image names - run: | - echo "BACKEND_IMAGE=ghcr.io/${GITHUB_REPOSITORY,,}/backend:pr-${{ needs.gate.outputs.pr-number }}" >> "$GITHUB_ENV" - echo "WORKER_IMAGE=ghcr.io/${GITHUB_REPOSITORY,,}/worker:pr-${{ needs.gate.outputs.pr-number }}" >> "$GITHUB_ENV" - - - name: Build and push backend image - uses: docker/build-push-action@v6 - with: - context: . - file: backend-api/Dockerfile - push: true - tags: ${{ env.BACKEND_IMAGE }} - cache-from: type=gha,scope=backend - cache-to: type=gha,mode=max,scope=backend - - - name: Build and push worker image - uses: docker/build-push-action@v6 - with: - context: ./engine - file: engine/Dockerfile - push: true - tags: ${{ env.WORKER_IMAGE }} - cache-from: type=gha,scope=worker - cache-to: type=gha,mode=max,scope=worker - - - name: Set lowercase PowerShell image name - if: needs.gate.outputs.m365 == 'true' - run: | - echo "POWERSHELL_IMAGE=ghcr.io/${GITHUB_REPOSITORY,,}/powershell:pr-${{ needs.gate.outputs.pr-number }}" >> "$GITHUB_ENV" - - - name: Build and push PowerShell service image - if: needs.gate.outputs.m365 == 'true' - uses: docker/build-push-action@v6 - with: - context: ./engine/powershell - file: engine/powershell/Dockerfile - platforms: linux/amd64 - push: true - tags: ${{ env.POWERSHELL_IMAGE }} - cache-from: type=gha,scope=powershell - cache-to: type=gha,mode=max,scope=powershell - - # ───────────────────────────────────────────────────────────────────────────── - # Job 2 — Run the full stack on the Actions runner and expose via Cloudflare - # Quick Tunnels (trycloudflare.com, no account, no token required). - # - # Architecture: - # 1. Both CF tunnels start immediately (they serve 502 until origin is ready) - # 2. Tunnel URLs are captured so the backend and frontend are configured once - # with the correct public URLs, no container restarts needed - # 3. The job sleeps (keeping everything alive) until cancelled by teardown - # or the 6-hour GitHub Actions job timeout is reached - # ───────────────────────────────────────────────────────────────────────────── - run-preview: - runs-on: ubuntu-latest - needs: [gate, build-images] - timeout-minutes: 360 - - steps: - - name: Checkout - uses: actions/checkout@v4 - - - name: Log in to GHCR - uses: docker/login-action@v3 - with: - registry: ghcr.io - username: ${{ github.actor }} - password: ${{ secrets.GITHUB_TOKEN }} - - - name: Set lowercase image names - run: | - echo "BACKEND_IMAGE=ghcr.io/${GITHUB_REPOSITORY,,}/backend:pr-${{ needs.gate.outputs.pr-number }}" >> "$GITHUB_ENV" - echo "WORKER_IMAGE=ghcr.io/${GITHUB_REPOSITORY,,}/worker:pr-${{ needs.gate.outputs.pr-number }}" >> "$GITHUB_ENV" - echo "PREVIEW_ENC_KEY=${{ secrets.PREVIEW_ENCRYPTION_KEY }}" >> "$GITHUB_ENV" - echo "PREVIEW_SECRET_KEY=$(openssl rand -hex 32)" >> "$GITHUB_ENV" - - - name: Set lowercase PowerShell image name - if: needs.gate.outputs.m365 == 'true' - run: | - echo "POWERSHELL_IMAGE=ghcr.io/${GITHUB_REPOSITORY,,}/powershell:pr-${{ needs.gate.outputs.pr-number }}" >> "$GITHUB_ENV" - - - name: Pull pre-built images from GHCR - run: | - docker pull "$BACKEND_IMAGE" - docker pull "$WORKER_IMAGE" - if [ "${{ needs.gate.outputs.m365 }}" = "true" ]; then - docker pull "$POWERSHELL_IMAGE" - fi - - - name: Install cloudflared - run: | - curl -fsSL \ - https://github.com/cloudflare/cloudflared/releases/latest/download/cloudflared-linux-amd64.deb \ - -o /tmp/cloudflared.deb - sudo dpkg -i /tmp/cloudflared.deb - - - name: Start Cloudflare Quick Tunnels and capture URLs - id: tunnels - run: | - # Start both tunnels immediately — CF generates URLs before the origin is up, - # so we can configure the backend and frontend correctly from the first start. - cloudflared tunnel --url http://localhost:8000 \ - --logfile /tmp/cf-backend.log \ - --no-autoupdate 2>/dev/null & - - cloudflared tunnel --url http://localhost:3000 \ - --logfile /tmp/cf-frontend.log \ - --no-autoupdate 2>/dev/null & - - BACKEND_URL="" - FRONTEND_URL="" - for i in $(seq 1 30); do - sleep 2 - if [ -z "$BACKEND_URL" ]; then - BACKEND_URL=$(grep -o 'https://[a-z0-9-]*\.trycloudflare\.com' /tmp/cf-backend.log 2>/dev/null | head -1 || true) - fi - if [ -z "$FRONTEND_URL" ]; then - FRONTEND_URL=$(grep -o 'https://[a-z0-9-]*\.trycloudflare\.com' /tmp/cf-frontend.log 2>/dev/null | head -1 || true) - fi - if [ -n "$BACKEND_URL" ] && [ -n "$FRONTEND_URL" ]; then - break - fi - done - - if [ -z "$BACKEND_URL" ]; then - echo "ERROR: Could not get backend tunnel URL" - cat /tmp/cf-backend.log - exit 1 - fi - if [ -z "$FRONTEND_URL" ]; then - echo "ERROR: Could not get frontend tunnel URL" - cat /tmp/cf-frontend.log - exit 1 - fi - - echo "Backend tunnel: $BACKEND_URL" - echo "Frontend tunnel: $FRONTEND_URL" - echo "backend-url=$BACKEND_URL" >> "$GITHUB_OUTPUT" - echo "frontend-url=$FRONTEND_URL" >> "$GITHUB_OUTPUT" - - - name: Start infrastructure services - run: COMPOSE_PROJECT_NAME=autoaudit docker compose up -d db redis opa - - - name: Wait for infrastructure healthchecks - run: | - echo "Waiting for PostgreSQL..." - timeout 60 bash -c 'until docker exec autoaudit-db pg_isready -U autoaudit; do sleep 2; done' - - echo "Waiting for Redis..." - timeout 30 bash -c 'until docker exec autoaudit-redis redis-cli ping | grep -q PONG; do sleep 2; done' - - echo "Waiting for OPA..." - timeout 30 bash -c 'until curl -sf http://localhost:8181/health > /dev/null; do sleep 2; done' - - - name: Start PowerShell service - if: needs.gate.outputs.m365 == 'true' - run: | - docker run -d \ - --name autoaudit-powershell-service \ - --network autoaudit_default \ - -p 8001:8001 \ - "$POWERSHELL_IMAGE" - - - name: Wait for PowerShell service health - if: needs.gate.outputs.m365 == 'true' - run: timeout 120 bash -c 'until curl -sf http://localhost:8001/health > /dev/null; do sleep 3; done' - - - name: Start backend container - run: | - docker run -d \ - --name autoaudit-backend-api \ - --network autoaudit_default \ - -p 8000:8000 \ - -e APP_ENV=preview \ - -e DATABASE_URL=postgresql+asyncpg://autoaudit:autoaudit_dev_password@db:5432/autoaudit \ - -e SECRET_KEY="$PREVIEW_SECRET_KEY" \ - -e ENCRYPTION_KEY="$PREVIEW_ENC_KEY" \ - -e REDIS_URL=redis://redis:6379 \ - -e OPA_URL=http://opa:8181 \ - -e BACKEND_PUBLIC_URL="${{ steps.tunnels.outputs.backend-url }}" \ - -e FRONTEND_URL="${{ steps.tunnels.outputs.frontend-url }}" \ - -v "${{ github.workspace }}/engine/policies:/app/policies:ro" \ - "$BACKEND_IMAGE" - - - name: Wait for backend health (includes Alembic migrations) - run: | - echo "Waiting for backend API to be ready..." - timeout 120 bash -c ' - until curl -sf http://localhost:8000/health > /dev/null 2>&1 || \ - curl -sf http://localhost:8000/ > /dev/null 2>&1; do - sleep 3 - done - ' - echo "Backend is ready." - - - name: Setup Node.js - uses: actions/setup-node@v4 - with: - node-version: 20 - cache: npm - cache-dependency-path: frontend/package-lock.json - - - name: Build frontend - working-directory: frontend - run: | - npm ci - VITE_API_URL="${{ steps.tunnels.outputs.backend-url }}" npm run build - - - name: Serve frontend static files - working-directory: frontend - run: | - npx serve -s dist -l 3000 & - sleep 3 - - - name: Start worker container - run: | - PS_URL="" - if [ "${{ needs.gate.outputs.m365 }}" = "true" ]; then - PS_URL="-e POWERSHELL_SERVICE_URL=http://autoaudit-powershell-service:8001" - fi - docker run -d \ - --name autoaudit-worker \ - --network autoaudit_default \ - -e DATABASE_URL=postgresql+asyncpg://autoaudit:autoaudit_dev_password@db:5432/autoaudit \ - -e REDIS_URL=redis://redis:6379 \ - -e OPA_URL=http://opa:8181 \ - -e ENCRYPTION_KEY="$PREVIEW_ENC_KEY" \ - $PS_URL \ - -v "${{ github.workspace }}/engine:/app/engine:ro" \ - -v "${{ github.workspace }}/engine/policies:/app/policies:ro" \ - "$WORKER_IMAGE" - - - name: Post or update PR comment - uses: actions/github-script@v7 - with: - script: | - const prNumber = parseInt('${{ needs.gate.outputs.pr-number }}'); - const backendUrl = '${{ steps.tunnels.outputs.backend-url }}'; - const frontendUrl = '${{ steps.tunnels.outputs.frontend-url }}'; - const runId = '${{ github.run_id }}'; - const sha = '${{ github.sha }}'.slice(0, 7); - - const m365 = '${{ needs.gate.outputs.m365 }}' === 'true'; - const rows = [ - `| Frontend | ${frontendUrl} |`, - `| Backend | ${backendUrl} |`, - ]; - if (m365) rows.push('| PowerShell | running (M365 enabled) |'); - - const body = [ - '', - ``, - '## Preview Environment', - '', - '| | URL |', - '|---|---|', - ...rows, - '', - `> Running on GitHub Actions · expires in up to 6 hours · SHA: ${sha}`, - `> Add \`teardown-preview\` label to stop early · [Workflow run](${context.serverUrl}/${context.repo.owner}/${context.repo.repo}/actions/runs/${runId})` - ].join('\n'); - - const comments = await github.rest.issues.listComments({ - owner: context.repo.owner, - repo: context.repo.repo, - issue_number: prNumber, - }); - - const existing = comments.data.find(c => - c.user.login === 'github-actions[bot]' && - c.body.includes('') - ); - - if (existing) { - await github.rest.issues.updateComment({ - owner: context.repo.owner, - repo: context.repo.repo, - comment_id: existing.id, - body, - }); - } else { - await github.rest.issues.createComment({ - owner: context.repo.owner, - repo: context.repo.repo, - issue_number: prNumber, - body, - }); - } - - - name: Keep-alive (preview stays up until cancelled or 6h timeout) - run: | - echo "Preview is live." - echo " Frontend: ${{ steps.tunnels.outputs.frontend-url }}" - echo " Backend: ${{ steps.tunnels.outputs.backend-url }}" - echo "Sleeping until cancelled by teardown workflow or 6-hour timeout..." - for i in $(seq 1 360); do - sleep 60 - done - echo "6-hour timeout reached. Preview shutting down." - - - name: Mark PR comment as expired - if: always() - uses: actions/github-script@v7 - with: - script: | - const prNumber = parseInt('${{ needs.gate.outputs.pr-number }}'); - const runId = '${{ github.run_id }}'; - const comments = await github.rest.issues.listComments({ - owner: context.repo.owner, - repo: context.repo.repo, - issue_number: prNumber, - }); - const existing = comments.data.find(c => - c.user.login === 'github-actions[bot]' && - c.body.includes('') - ); - if (!existing) return; - await github.rest.issues.updateComment({ - owner: context.repo.owner, - repo: context.repo.repo, - comment_id: existing.id, - body: [ - '', - ``, - '## Preview Environment', - '', - '> **Preview has expired.** The tunnel URLs above no longer resolve.', - '> Add `deploy-preview` to spin up a fresh environment.', - '', - `> [Workflow run](${context.serverUrl}/${context.repo.owner}/${context.repo.repo}/actions/runs/${runId})` - ].join('\n'), - }); diff --git a/.github/workflows/pr.preview-instructions.yml b/.github/workflows/pr.preview-instructions.yml deleted file mode 100644 index 70172862d..000000000 --- a/.github/workflows/pr.preview-instructions.yml +++ /dev/null @@ -1,32 +0,0 @@ -name: Preview Instructions - -on: - pull_request: - types: [opened, reopened] - -permissions: - pull-requests: write - -jobs: - post-instructions: - runs-on: ubuntu-latest - steps: - - name: Post preview environment instructions - uses: peter-evans/create-or-update-comment@v4 - with: - issue-number: ${{ github.event.number }} - body: | - - ## Preview Environment - - A preview environment can be spun up on demand for this PR. - - | Action | Label | Includes | - |---|---|---| - | **Spin up preview** | `deploy-preview` | Frontend, backend, database, Redis, OPA, worker | - | **Spin up preview with M365** | `deploy-preview-m365` | Everything above + PowerShell service for Exchange/Teams scan testing | - | **Tear down preview** | `teardown-preview` | Stops the environment early | - - > The environment will also be torn down automatically when the PR is closed or merged. - > Preview URLs will appear in a follow-up comment once the deploy completes (~5–8 min). - > M365 scans require real tenant credentials added through the frontend UI. diff --git a/.github/workflows/pr.preview-teardown.yml b/.github/workflows/pr.preview-teardown.yml deleted file mode 100644 index 18449a2de..000000000 --- a/.github/workflows/pr.preview-teardown.yml +++ /dev/null @@ -1,118 +0,0 @@ -name: Preview Teardown - -on: - pull_request: - types: [closed, labeled] - workflow_dispatch: - inputs: - pr_number: - description: "PR number to tear down (for manual testing)" - required: true - default: "1" - -permissions: - contents: read - pull-requests: write - issues: write - actions: write - -jobs: - teardown: - runs-on: ubuntu-latest - if: >- - github.event_name == 'workflow_dispatch' || - (github.event_name == 'pull_request' && github.event.action == 'closed') || - (github.event_name == 'pull_request' && github.event.action == 'labeled' && - github.event.label.name == 'teardown-preview') - - steps: - - name: Resolve PR number - id: pr - run: | - if [ "${{ github.event_name }}" = "pull_request" ]; then - echo "number=${{ github.event.number }}" >> "$GITHUB_OUTPUT" - else - echo "number=${{ inputs.pr_number }}" >> "$GITHUB_OUTPUT" - fi - - - name: Remove teardown-preview label - if: github.event_name == 'pull_request' && github.event.action == 'labeled' - uses: actions/github-script@v7 - with: - script: | - await github.rest.issues.removeLabel({ - owner: context.repo.owner, - repo: context.repo.repo, - issue_number: context.payload.pull_request.number, - name: 'teardown-preview' - }); - - # Known risk: listComments only returns the first page so teardown may miss the preview comment on PRs with >30 comments. - - name: Find preview comment and extract run ID - id: find-run - uses: actions/github-script@v7 - with: - script: | - const prNumber = parseInt('${{ steps.pr.outputs.number }}'); - const comments = await github.rest.issues.listComments({ - owner: context.repo.owner, - repo: context.repo.repo, - issue_number: prNumber, - }); - - const preview = comments.data.find(c => - c.user.login === 'github-actions[bot]' && - c.body.includes('') - ); - - if (!preview) { - core.setOutput('run-id', ''); - core.setOutput('comment-id', ''); - console.log('No preview comment found for PR #' + prNumber); - return; - } - - const match = preview.body.match(//); - const runId = match ? match[1] : ''; - core.setOutput('run-id', runId); - core.setOutput('comment-id', String(preview.id)); - console.log('Found run ID: ' + runId + ', comment ID: ' + preview.id); - - - name: Cancel preview workflow run - if: steps.find-run.outputs.run-id != '' - run: | - RUN_ID="${{ steps.find-run.outputs.run-id }}" - echo "Cancelling workflow run $RUN_ID..." - HTTP_STATUS=$(curl -s -o /dev/null -w "%{http_code}" \ - -X POST \ - -H "Authorization: Bearer ${{ secrets.GITHUB_TOKEN }}" \ - -H "Accept: application/vnd.github+json" \ - -H "X-GitHub-Api-Version: 2022-11-28" \ - "https://api.github.com/repos/${{ github.repository }}/actions/runs/${RUN_ID}/cancel") - echo "Cancel API response status: $HTTP_STATUS" - # 202 = accepted, 409 = already completed — both are fine - if [ "$HTTP_STATUS" = "202" ] || [ "$HTTP_STATUS" = "409" ]; then - echo "Run cancelled successfully (or was already completed)." - else - echo "WARNING: Unexpected status $HTTP_STATUS — run may not have been cancelled." - fi - - - name: Update PR comment — teardown complete - if: steps.find-run.outputs.comment-id != '' - uses: actions/github-script@v7 - with: - script: | - await github.rest.issues.updateComment({ - owner: context.repo.owner, - repo: context.repo.repo, - comment_id: parseInt('${{ steps.find-run.outputs.comment-id }}'), - body: [ - '', - '', - '## Preview Environment', - '', - 'Preview environment for this PR has been **torn down**.', - '', - `> [Workflow run](${context.serverUrl}/${context.repo.owner}/${context.repo.repo}/actions/runs/${{ github.run_id }})` - ].join('\n'), - }); diff --git a/docs/DevSecOps/workflow-documentation.md b/docs/DevSecOps/workflow-documentation.md index f7b4e11d5..b11a4621b 100644 --- a/docs/DevSecOps/workflow-documentation.md +++ b/docs/DevSecOps/workflow-documentation.md @@ -2,6 +2,8 @@ Reference for anyone picking up the project who wants to understand what runs in `.github/workflows/`, when it runs, and why. +There are fourteen workflow files. This page names all fourteen; if you add one, add it here. + --- ## Naming Convention @@ -10,46 +12,69 @@ Workflow files use a prefix to group them by purpose: | Prefix | Purpose | |---|---| -| `ci.` | Code quality checks on PRs and pushes to main | +| `ci.` | Code quality and security checks on PRs and pushes to main | | `ops.` | Scheduled or operational jobs not tied to code review | | `pr.` | PR metadata management | --- -## CI Workflows +## What runs, and when -| Workflow | Watches | -|---|---| -| `ci.backend-api.yml` | `backend-api/**` | -| `ci.frontend.yml` | `frontend/**` | -| `ci.engine.yml` | `engine/**` | -| `ci.security.yml` | `security/**` | +| Workflow | Triggers | Path filter | +|---|---|---| +| `ci.backend-api.yml` | push `main`, PR to `main`, weekly | `backend-api/**` | +| `ci.frontend.yml` | push `main`, PR to `main`, weekly | `frontend/**` | +| `ci.engine.yml` | push `main`, PR to `main`, weekly | `engine/**` | +| `ci.security.yml` | push `main`, PR to `main`, weekly | `security/**` | +| `ci.compliance.yml` | push `main`, PR to any branch | `policy/**`, `docker-compose.yml`, `.github/workflows/**` | +| `ci.grype.yml` | push `main`, PR to `main`, weekly | `frontend/**`, `backend-api/**`, `engine/**` | +| `ci.gitleaks.yml` | push and PR on `main` and `staging`, weekly | none | +| `ci.opa-eval.yml` | push `engine-development`, **every** PR, manual | none | +| `ci.validate-alerts.yml` | push on any branch, PR to any branch | `infrastructure/monitoring/alerts/**` | +| `ops.branch-cleanup.yml` | PR to `main` closed | none | +| `ops.collector.yml` | manual only, and **hard-disabled** | none | +| `ops.short-test.yml` | push, but the filter never matches (see below) | `.github/workflows/short-test.yml` | +| `ops.workflow-cleanup.yml` | weekly, manual | none | +| `pr.size-warning.yml` | PR to `main` opened / synchronized / reopened | none | + +A path-filtered workflow does not start when a PR touches nothing inside its filter, and does not appear in the Actions tab for that PR. The weekly schedule carries no path filter, so a scheduled run executes regardless of what changed. + +Three workflows omit a branch filter on `pull_request`: `ci.compliance.yml` and `ci.validate-alerts.yml`, which are still narrowed by their path filters, and `ci.opa-eval.yml`, which has neither. `ci.opa-eval.yml` therefore runs on every pull request in the repository. + +--- -### When they run +## CodeQL and lint workflows -Each workflow triggers on pull requests and pushes to `main`, but only when files inside its watched directory changed. If a PR only touches `frontend/`, the backend, engine, and security workflows never start and do not appear in the Actions tab. +`ci.backend-api.yml`, `ci.frontend.yml`, `ci.engine.yml` and `ci.security.yml` share a shape. -The weekly schedule bypasses path filtering and runs everything regardless. +**`analyze`** runs CodeQL static analysis and reports findings to the GitHub Security tab. The backend-api workflow also installs and runs Bandit in the same job, as a Python-specific check. -### Jobs +**`run-lint`** runs Super Linter in `ci.backend-api.yml`, `ci.engine.yml` and `ci.security.yml`, with several linters disabled to keep it focused on the languages used in each area. `ci.frontend.yml` does not use Super Linter: its lint job runs `npm run lint`, which `frontend/package.json` defines as `oxlint`. -When triggered, two jobs run in parallel: +**A test job**, in three of the four: -**`analyze`** runs CodeQL static analysis and reports findings to the GitHub Security tab. The backend-api workflow also runs Bandit inside the same job as a Python-specific check. +| Workflow | Job | What it runs | +|---|---|---| +| `ci.backend-api.yml` | `test` | `pytest` under `uv`, in `backend-api/` | +| `ci.engine.yml` | `test` | `pytest tests/test_wiring.py` only — structural checks, not the engine suite | +| `ci.frontend.yml` | `build-and-test` | `tsc --noEmit`, `npm test`, `npm run build` | +| `ci.security.yml` | — | no test job | -**`run-lint`** runs Super Linter across changed files. Several linters are disabled to keep it focused on the languages used in each area. +`ci.engine.yml` runs one test file. The Rego policies under `engine/policies/` are not evaluated by any workflow in this repository — see `ci.opa-eval.yml` below. ### PR status comment -After both jobs finish, a `report` job posts a table on the PR showing which jobs passed or failed with a link to the run logs. On subsequent pushes the comment updates in place. +After the other jobs finish, a `report` job posts a table on the PR showing which jobs passed or failed, with a link to the run logs. On subsequent pushes the comment updates in place. ``` analyze ─┐ - ├─→ report -run-lint ─┘ +run-lint ─┼─→ report +test ─┘ ``` -The report job only runs on PR events, not on push or schedule. +The report job is `if: always() && github.event_name == 'pull_request'`, so it runs whatever the other jobs did, and never on push or schedule. The link to the run logs appears only when something failed. + +The comment does not always arrive. `ci.backend-api.yml` and `ci.security.yml` detect a pull request from a fork and skip the comment with a log line; `ci.engine.yml` and `ci.frontend.yml` have no such guard and attempt the write, which a fork's read-only token cannot perform. ### `ci.security.yml` @@ -59,43 +84,86 @@ The `security/` directory contains the TPRM Scanner from T2 2025 and is no longe ## Grype Dependency Scan -**`ci.grype.yml`** triggers on PRs and pushes to `main` when files change inside `frontend/**`, `backend-api/**`, or `engine/**`, and on a weekly schedule. +**`ci.grype.yml`** triggers on PRs and pushes to `main` when files change inside `frontend/**`, `backend-api/**` or `engine/**`, and weekly. + +It runs `anchore/scan-action@v6` as a **directory** scan over `path: "."`, checking dependency files (`pyproject.toml`, `requirements.txt`, `package-lock.json`) against public vulnerability databases without needing a built image. Results upload to the GitHub Security tab as a SARIF report. -Grype walks the repo and checks dependency files (`pyproject.toml`, `requirements.txt`, `package-lock.json`) against public vulnerability databases without needing a Docker build. Results upload to the GitHub Security tab as a SARIF report. +**`fail-build` is `true`, with `severity-cutoff: critical`.** A critical finding fails the check; anything at high or below is reported and does not. An earlier version of this page said `fail-build` was false — it is not, and a contributor who assumed otherwise would be surprised by a red check. Whether a failing check actually blocks a merge is a branch-protection setting, which is not visible in this repository. -`fail-build` is false so findings are surfaced for review rather than blocking merges. +The workflow keeps a commented-out image scan from before Docker Hub was decommissioned. A directory scan cannot see a base image or an OS package layer, so nothing in this repository scans a built image. Note: the Security tab upload requires GitHub Advanced Security, which is free for public repos and requires a paid plan for private ones. --- -## Ops Workflows +## Secret scanning + +**`ci.gitleaks.yml`** runs on pushes and pull requests targeting `main` and `staging`, and weekly on Mondays. It checks out the full history (`fetch-depth: 0`) and scans all of it (`--log-opts="--all"`), so a secret removed in a later commit is still found. + +It installs a pinned gitleaks (8.30.1) and verifies the release checksum before using it. Findings are matched against `.gitleaks-baseline.json` at the repository root; anything not in that baseline exits non-zero and **fails the build**. Adding a finding to the baseline is how you accept it, and the baseline is reviewable in the diff. + +--- + +## Compose and workflow policy + +**`ci.compliance.yml`** runs Conftest over the Rego policies in `policy/`. It verifies the policy unit tests and the compliant fixtures — those steps gate — and then reports findings against the live `docker-compose.yml` and `.github/workflows/` with `--no-fail`, which do not. + +It is the only `ci.` workflow whose `pull_request` trigger has no branch filter beyond its paths, and it watches `.github/workflows/**`, so editing any workflow file runs it. + +--- + +## Other CI workflows + +**`ci.opa-eval.yml`** does not evaluate the benchmark policies. It runs `engine/legacy/engine/aggregator.py` and `engine/legacy/engine/json_to_pdf.py` over the legacy engine, and uploads a JSON report and a PDF as artifacts. It does not read `engine/policies/`. + +No policy verdict from it can fail the workflow: `aggregator.py` catches a non-zero `opa` exit and records the stderr into the report JSON rather than raising. A crash in either script, or a missing artifact, still fails the job. + +Nothing in `.github/workflows/` runs `opa check`, `opa test` or `opa eval` against the CIS or Essential Eight policy corpus. + +**`ci.validate-alerts.yml`** validates the Prometheus alerting rules under `infrastructure/monitoring/alerts/`. An earlier version of this page called it a conceptual piece that "does not do anything meaningful"; it does. It runs `infrastructure/monitoring/alerts/tests/promtool_validation.sh`, which is `set -euo pipefail` and calls `promtool check rules`, so a syntax error in a validated rule fails the build. + +Three limits are worth knowing: + +- The script does not validate the directory. It globs five filename patterns — `*alerts.yaml`, `*_errors.yaml`, `*health.yaml`, `*utilisation.yaml`, `*security.yaml` — non-recursively. Every rule file present today matches one, but a new `custom_rules.yaml` would trigger the workflow and escape validation. +- It installs promtool with `apt-get install -y prometheus`, unpinned, so the tool version is whatever the runner image offers that day. +- `promtool check rules` is a syntax check. It does not test that a threshold fires when it should, and it does not check that an alert reads a metric the application actually emits. + +--- + +## Ops workflows + +**`ops.branch-cleanup.yml`** deletes the head branch after a pull request into `main` is merged. It only acts on branches in this repository — not forks — and never on `main`. + +**`ops.collector.yml`** is **hard-disabled and runs nothing.** Its push trigger is commented out, leaving only `workflow_dispatch`, and the job itself carries `if: false` so that even a manual run does nothing. Its header records the two conditions for re-enabling it: the `GCP_CREDENTIALS` secret is a long-lived service-account JSON key and must be replaced with Workload Identity Federation, and the workflow auto-commits live GCP infrastructure data — IAM policies, firewall rules, SQL/BigQuery/Dataproc details, DNS zones — into `engine/test-configs/` on every run. Do not restore the trigger before both are addressed. + +**`ops.workflow-cleanup.yml`** runs weekly, on Sundays at 00:00 UTC, **as a dry run**: the scheduled step is `node cleanup-workflows.js --dryRun=true`, which reports and deletes nothing. Actual deletion requires a manual `workflow_dispatch` with `dry_run=false` **and** `confirm=DELETE`; without the confirmation the job refuses and exits 1. -**`ops.collector.yml`** runs the audit engine collector on the `engine-development` branch when changes are pushed there. +**It is not a retention policy.** The workflow passes `RETENTION_DAYS: 30` and the dispatch input is described as "Delete runs older than this many days", but `tools/workflow-cleanup/cleanup-workflows.js` never reads that variable. It selects by how long a run *took*: under 10 seconds is deleted, 10 seconds to 2 minutes is deleted, 2 minutes or more is kept. It also fetches a single page of 20 runs with no pagination. -**`ops.workflow-cleanup.yml`** runs on a schedule to delete old workflow run history. +So a real deletion run would remove recent short runs and keep old long ones, which is not what the input description promises. Nothing here prunes by age, and this workflow should not be cited as evidence that run history is retained for any period. -**`ops.short-test.yml`** a minimal workflow used to verify cleanup is working. Only triggers when its own file changes. +**`ops.short-test.yml`** is the canary used to check that the cleanup workflow has something to clean. It is meant to trigger only when its own file changes, but its filter names `.github/workflows/short-test.yml` while the file is `.github/workflows/ops.short-test.yml`: commit `155f82aa` applied the `ops.` prefix and left the filter behind. The filter now names a path that does not exist, so editing the workflow does not run it. Anyone who adds a file at the old path would trigger it by accident. --- -## Other CI Workflows +## PR workflows -**`ci.opa-eval.yml`** evaluates OPA policies used by the audit engine. Runs on `engine-development` and on any pull request. +**`pr.size-warning.yml`** flags unusually large pull requests: above 30 changed files or 500 changed lines it emits a `core.warning` annotation and writes a job summary with the numbers. It does **not** post a PR comment, and it does not block the PR. `docs/DevSecOps/pr-size-warning.md` has the reasoning behind the thresholds. -**`ci.validate-alerts.yml`** was intended to validate Prometheus alerting rules under `infrastructure/monitoring/alerts/`. It was left by a previous team member as a conceptual piece and does not do anything meaningful in its current state. Left in place for reference. +The three `pr.preview-*` workflows — deploy, instructions and teardown — were removed. `git log --diff-filter=D -- .github/workflows/pr.preview-deploy.yml` has the original. --- -## Weekly Scheduled Scans +## Scheduled runs -| Workflow | Schedule | Purpose | +| Workflow | Schedule (UTC) | Purpose | |---|---|---| -| `ci.backend-api.yml` | Saturdays 23:32 UTC | CodeQL and lint scan of backend-api | -| `ci.frontend.yml` | Saturdays 23:32 UTC | CodeQL and lint scan of frontend | -| `ci.engine.yml` | Saturdays 23:32 UTC | CodeQL and lint scan of engine | -| `ci.security.yml` | Saturdays 23:32 UTC | CodeQL and lint scan of security | -| `ci.grype.yml` | Thursdays 20:37 UTC | Dependency vulnerability scan | -| `ops.workflow-cleanup.yml` | Saturdays 23:32 UTC | Cleans up old workflow run history | +| `ci.backend-api.yml` | Saturdays 23:32 | CodeQL and lint scan of backend-api | +| `ci.frontend.yml` | Saturdays 23:32 | CodeQL and lint scan of frontend | +| `ci.engine.yml` | Saturdays 23:32 | CodeQL and lint scan of engine | +| `ci.security.yml` | Saturdays 23:32 | CodeQL and lint scan of security | +| `ci.grype.yml` | Thursdays 20:37 | Dependency vulnerability scan | +| `ci.gitleaks.yml` | Mondays 03:00 | Full-history secret scan | +| `ops.workflow-cleanup.yml` | Sundays 00:00 | Dry run only — reports which of the 20 most recent runs it would delete, and deletes nothing | Scheduled runs do not post PR comments. diff --git a/engine/collectors/README.md b/engine/collectors/README.md index 704e55d61..08555c3b7 100644 --- a/engine/collectors/README.md +++ b/engine/collectors/README.md @@ -15,14 +15,21 @@ A collector is a simple class with one job: call an API and return structured da ## Writing a new collector -### 1. Create the collector class +### 1. Pick the client -Inherit from `BaseDataCollector` and implement the `collect` method: +There are two, and they are not interchangeable. Of the 48 collectors registered in `registry.py`, **26 take a `GraphClient` and 22 take a `PowerShellClient`.** Which one you get is decided by your collector's registered ID, not by your class — see [The two clients](#the-two-clients) below — so choose before you write anything. + +Use Graph if the setting has a Graph endpoint. Use PowerShell if it does not: every Exchange Online and SharePoint Online setting AutoAudit reads today comes through a cmdlet. + +### 2. Create the collector class + +For a Graph collector, inherit from `BaseDataCollector`: ```python from typing import Any -from engine.collectors.base import BaseDataCollector -from engine.collectors.graph_client import GraphClient + +from collectors.base import BaseDataCollector +from collectors.graph_client import GraphClient class MyNewCollector(BaseDataCollector): @@ -39,22 +46,55 @@ class MyNewCollector(BaseDataCollector): } ``` -### 2. Place it in the right folder +For a PowerShell collector, inherit from `BasePowerShellCollector` and annotate the client accordingly. All 22 PowerShell collectors do this, and no Graph collector does: + +```python +from typing import Any + +from collectors.powershell_base import BasePowerShellCollector +from collectors.powershell_client import PowerShellClient + + +class MyPowerShellCollector(BasePowerShellCollector): + """One-line description of what this collects.""" + + async def collect(self, client: PowerShellClient) -> dict[str, Any]: + policies = await client.run_cmdlet("ExchangeOnline", "Get-SomePolicy") + + # The service returns None, a single object, or a list. Normalise it, + # and leave the interpretation to the policy. + if policies is None: + policies = [] + elif isinstance(policies, dict): + policies = [policies] + + return {"policies": policies, "total_policies": len(policies)} +``` + +Imports are rooted at `collectors.`, not `engine.collectors.`: `engine/` is the import root for the worker and the tests. + +### 3. Place it in the right folder Collectors are organized by service and domain: ``` collectors/ - entra/ # Microsoft Entra ID (Azure AD) - roles/ # Role and admin management - conditional_access/ - m365/ # Microsoft 365 services - exchange/ # Exchange Online - sharepoint/ # SharePoint Online - teams/ # Microsoft Teams + entra/ # Microsoft Entra ID (Azure AD) - Graph + applications/ authentication/ conditional_access/ devices/ + domains/ governance/ groups/ policies/ roles/ users/ + exchange/ # Exchange Online - PowerShell, except dns/ + audit/ authentication/ dns/ mailbox/ organization/ + protection/ transport/ + sharepoint/ # SharePoint Online - PowerShell + pnp/ + _pending/ # Written but not registered; not run by a scan + compliance/ fabric/ teams/ + compliance/ m365/ teams/ # package stubs, no collectors in them ``` -### 3. Register it +`_pending/` holds collectors that exist but are not in `registry.py`. Nothing there runs. The five Teams collectors and the two Purview compliance collectors live there, so no registered collector connects to the `Teams` or `Compliance` PowerShell modules. Teams *settings* are not entirely uncovered: `exchange.protection.teams_protection_policy` reads `Get-TeamsProtectionPolicy` through the `ExchangeOnline` module. The bare `compliance/`, `m365/` and `teams/` packages at the top level hold only an `__init__.py`. + +### 4. Register it Why do we need to register these? @@ -68,7 +108,7 @@ You could dynamically import based on the string path, but that's fragile and ha Add your collector to `registry.py`: ```python -from engine.collectors.entra.roles.my_new import MyNewCollector +from collectors.entra.roles.my_new import MyNewCollector DATA_COLLECTORS: dict[str, type[BaseDataCollector]] = { # ... existing collectors ... @@ -76,9 +116,11 @@ DATA_COLLECTORS: dict[str, type[BaseDataCollector]] = { } ``` -The ID follows the folder path: `entra.roles.my_new` maps to `entra/roles/my_new.py`. +The ID follows the folder path by convention: `entra.roles.my_new` maps to `entra/roles/my_new.py`. It is a convention, not an invariant — `entra.conditional_access.policies` is registered against `entra/conditional_access/conditional_access_policies.py` — but follow it unless you have a reason not to. + +**The ID is also what selects your client at runtime**, so a PowerShell collector registered under an `entra.` ID will be handed a `GraphClient` and fail. -### 4. Reference it in metadata.json +### 5. Reference it in metadata.json In the benchmark's `metadata.json`, set the `data_collector_id` for your control: @@ -102,9 +144,30 @@ In the benchmark's `metadata.json`, set the `data_collector_id` for your control **Async all the way.** Collectors are async. Use `await` for all API calls. This lets us run multiple collectors concurrently during scans. -## The Graph client +## The two clients -`GraphClient` handles Microsoft Graph API authentication and requests. It's shared across all Microsoft collectors (Entra, M365 services). +An earlier version of this page said `GraphClient` was "shared across all Microsoft collectors (Entra, M365 services)". It is not, and the difference is nearly half the registry: 26 of the 48 registered collectors take a `GraphClient`, and the other 22 take a `PowerShellClient`. + +`engine/worker/tasks.py` picks between them by **collector ID prefix**, not by anything on your class: + +```python +if collector_id.startswith( + ("exchange.", "compliance.", "sharepoint.pnp.") +) and not collector_id.startswith("exchange.dns."): + client = PowerShellClient(...) +else: + client = GraphClient(...) +``` + +The collector is instantiated first, by ID, and the client is then constructed and passed to `collect()`. That rule and the `collect` annotations agree for all 48 registered collectors today, but the rule is what actually runs: change one without the other and the collector gets the wrong client. + +`exchange.dns.dns_security_records` is the carve-out: it lives under `exchange/` but reads tenant domains from Graph and then resolves SPF and DMARC over DNS, so it takes a `GraphClient`. + +Of the 22 PowerShell collectors, 21 are `exchange.*` and reach the `ExchangeOnline` module; one is `sharepoint.pnp.tenant` and reaches `SharePointOnline`. + +### `GraphClient` + +Handles Microsoft Graph API authentication and requests. Every `entra.*` collector uses it, including the device-management ones that read Intune endpoints under `/deviceManagement`, as does `exchange.dns.dns_security_records`. ```python # Basic GET request @@ -122,6 +185,25 @@ data = await client.get("/some/beta/endpoint", beta=True) The client handles: - OAuth token acquisition via MSAL -- Token caching and refresh +- Token caching — note that the cached token is returned without an expiry check, so it is caching, not refresh - Pagination with @odata.nextLink - Both v1.0 and beta Graph endpoints + +`get_all_pages` follows `@odata.nextLink` up to `max_pages`, which defaults to 100, and returns what it has without raising if there are more. If you expect an endpoint to exceed that, pass a higher `max_pages` rather than assuming the list is complete. + +### `PowerShellClient` + +Runs a cmdlet through the PowerShell service, for settings Graph does not expose. + +```python +# module, cmdlet, then cmdlet parameters as keyword arguments +config = await client.run_cmdlet("ExchangeOnline", "Get-OrganizationConfig") +policy = await client.run_cmdlet("ExchangeOnline", "Get-SafeLinksPolicy", Identity=name) +``` + +The client handles: +- Access-token acquisition via MSAL. The token authenticates the *connection* command the service runs — `Connect-ExchangeOnline`, and `Connect-MicrosoftTeams` with `-AccessTokens` — not the data cmdlet itself +- Dispatch to the PowerShell HTTP service when `POWERSHELL_SERVICE_URL` is set, and to a local Docker container otherwise +- `SharePointOnline`, which takes a different path entirely: it requires the HTTP service, has no Docker fallback, and authenticates with `Connect-PnPOnline` using a certificate, so the client raises unless both `sharepoint_admin_url` and `certificate_alias` are configured + +A cmdlet returns `None`, a single object, or a list depending on how many results there are. Normalise that in the collector rather than in the policy.