From 4c4a0a0655fb5fbdd605f195249f1519a1211dc9 Mon Sep 17 00:00:00 2001 From: Josh Hufford Date: Mon, 21 Sep 2026 22:22:08 -0400 Subject: [PATCH] Make staging and production deployment hosts independent --- .editorconfig | 5 + .gitattributes | 1 + .github/actions/validate/action.yml | 7 ++ .github/workflows/_deploy.yml | 30 +++-- .github/workflows/cd-edge.yml | 105 ++++++++++++---- deploy/edge/Caddyfile | 43 ------- deploy/edge/Caddyfile.production | 23 ++++ deploy/edge/Caddyfile.staging | 19 +++ deploy/edge/compose.production.yml | 4 + deploy/edge/compose.staging.yml | 4 + deploy/edge/compose.yml | 2 - docs/foundation-checklist.md | 10 ++ docs/quality/testing.md | 17 ++- docs/release/README.md | 6 + docs/setup/deployment-host-key.md | 114 ++++++++--------- docs/setup/github.md | 15 ++- docs/setup/hosting.md | 155 ++++++++++++++++++----- scripts/apply-edge.sh | 80 +++++++++++- scripts/check-edge-health.sh | 18 +++ scripts/deploy-app.sh | 4 +- scripts/render-edge.sh | 46 +++++++ scripts/resolve-edge-profile.sh | 28 +++++ scripts/verify-edge-profile.sh | 50 ++++++++ tests/deploy-edge/run.sh | 188 +++++++++++++++++++++++++++- tests/deploy-topology/run.sh | 183 +++++++++++++++++++++++++++ 25 files changed, 971 insertions(+), 186 deletions(-) delete mode 100644 deploy/edge/Caddyfile create mode 100644 deploy/edge/Caddyfile.production create mode 100644 deploy/edge/Caddyfile.staging create mode 100644 deploy/edge/compose.production.yml create mode 100644 deploy/edge/compose.staging.yml create mode 100644 scripts/check-edge-health.sh create mode 100644 scripts/render-edge.sh create mode 100644 scripts/resolve-edge-profile.sh create mode 100644 scripts/verify-edge-profile.sh create mode 100644 tests/deploy-topology/run.sh diff --git a/.editorconfig b/.editorconfig index a850735..5b29542 100644 --- a/.editorconfig +++ b/.editorconfig @@ -188,3 +188,8 @@ indent_size = 2 end_of_line = lf [*.{cmd,bat}] end_of_line = crlf + +# Caddy's formatter uses tabs for site blocks. +[Caddyfile*] +indent_style = tab +end_of_line = lf diff --git a/.gitattributes b/.gitattributes index bd56cc7..74718db 100644 --- a/.gitattributes +++ b/.gitattributes @@ -16,6 +16,7 @@ *.css text eol=lf *.js text eol=lf *.ts text eol=lf +Caddyfile* text eol=lf # Scripts *.sh text eol=lf diff --git a/.github/actions/validate/action.yml b/.github/actions/validate/action.yml index 20f620b..8781cfc 100644 --- a/.github/actions/validate/action.yml +++ b/.github/actions/validate/action.yml @@ -75,6 +75,13 @@ runs: set -o pipefail bash tests/deploy-edge/run.sh 2>&1 | tee artifacts/validation/deploy-edge.log + - name: Test shared and separate host configuration without a Docker daemon + shell: bash + working-directory: ${{ inputs.working-directory }} + run: | + set -o pipefail + bash tests/deploy-topology/run.sh 2>&1 | tee artifacts/validation/deploy-topology.log + - name: Test application rollback without Docker or SSH shell: bash working-directory: ${{ inputs.working-directory }} diff --git a/.github/workflows/_deploy.yml b/.github/workflows/_deploy.yml index 2a4e63e..759b9f3 100644 --- a/.github/workflows/_deploy.yml +++ b/.github/workflows/_deploy.yml @@ -160,14 +160,23 @@ jobs: packages: write steps: + - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 + with: + ref: ${{ needs.resolve-source.outputs.protected-revision }} + persist-credentials: false + - name: Check deployment configuration + id: target env: + DEPLOY_ENVIRONMENT: ${{ inputs.environment-slug }} + EDGE_PROFILE: ${{ vars.EDGE_PROFILE }} DEPLOY_HOST: ${{ secrets.DEPLOY_HOST }} DEPLOY_KNOWN_HOSTS: ${{ vars.DEPLOY_KNOWN_HOSTS }} DEPLOY_USER: ${{ secrets.DEPLOY_USER }} DEPLOY_SSH_KEY: ${{ secrets.DEPLOY_SSH_KEY }} run: | set -euo pipefail + bash scripts/resolve-edge-profile.sh "$DEPLOY_ENVIRONMENT" "$EDGE_PROFILE" >> "$GITHUB_OUTPUT" for name in DEPLOY_HOST DEPLOY_KNOWN_HOSTS DEPLOY_USER DEPLOY_SSH_KEY; do if [ -z "${!name}" ]; then echo "::error::${name} is unavailable to the deployment job." @@ -177,11 +186,6 @@ jobs: [[ "$DEPLOY_HOST" =~ ^[a-zA-Z0-9.-]+$ ]] || { echo '::error::DEPLOY_HOST is invalid.'; exit 1; } [[ "$DEPLOY_USER" =~ ^[a-zA-Z_][a-zA-Z0-9_-]*$ ]] || { echo '::error::DEPLOY_USER is invalid.'; exit 1; } - - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 - with: - ref: ${{ needs.resolve-source.outputs.protected-revision }} - persist-credentials: false - - uses: actions/download-artifact@3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c # v8.0.1 with: name: release-${{ inputs.environment-slug }} @@ -267,8 +271,14 @@ jobs: chmod 644 ~/.ssh/known_hosts ssh-keygen -l -f ~/.ssh/known_hosts - - name: Check production before staging update - if: ${{ inputs.environment-slug == 'staging' }} + - name: Verify the host edge profile before transferring a release + env: + EDGE_PROFILE: ${{ steps.target.outputs.profile }} + run: | + ssh deployment bash -s -- /srv/opengamebuilder/edge "$EDGE_PROFILE" < scripts/verify-edge-profile.sh + + - name: Check colocated production before staging update + if: ${{ steps.target.outputs.check-production == 'true' }} env: PRODUCTION_URL: ${{ vars.PRODUCTION_SMOKE_TEST_URL }} run: | @@ -301,7 +311,7 @@ jobs: release_id="$2" mkdir -p "$app_dir/incoming/$release_id" docker network inspect ogb-edge >/dev/null 2>&1 || { - echo "Shared edge network is missing; apply the edge configuration first." >&2 + echo "Host edge network is missing; apply this host's edge configuration first." >&2 exit 1 } EOF @@ -356,8 +366,8 @@ jobs: echo 'Intentional staging failure: the recovery step must restore the previous release.' >&2 exit 1 - - name: Check production after staging update - if: ${{ inputs.environment-slug == 'staging' }} + - name: Check colocated production after staging update + if: ${{ steps.target.outputs.check-production == 'true' }} env: PRODUCTION_URL: ${{ vars.PRODUCTION_SMOKE_TEST_URL }} run: | diff --git a/.github/workflows/cd-edge.yml b/.github/workflows/cd-edge.yml index b978dd3..f108d31 100644 --- a/.github/workflows/cd-edge.yml +++ b/.github/workflows/cd-edge.yml @@ -1,10 +1,24 @@ -name: 🌐 CD Shared Edge +name: 🌐 CD Edge -# Run from main when the shared Caddyfile or edge Compose definition changes. -# This is the only workflow that writes shared edge files or updates Caddy. +# One edge per host. Its profile explicitly selects the environments it serves. on: workflow_dispatch: + inputs: + environment: + description: "GitHub environment providing this host's credentials (shared edge requires production)" + required: true + type: choice + options: + - production + - staging + default: production + allow-profile-change: + description: "Production-approved migration: allow changing an already marked host's edge profile" + required: false + type: boolean + default: false +# Serialize edge maintenance, including two environments pointing at one host. concurrency: group: cd-shared-edge cancel-in-progress: false @@ -23,25 +37,55 @@ jobs: DISPATCH_REF: ${{ github.ref }} run: | if [ "$DISPATCH_REF" != 'refs/heads/main' ]; then - echo "Shared edge updates must be dispatched from main, not $DISPATCH_REF." >&2 + echo "Edge updates must be dispatched from main, not $DISPATCH_REF." >&2 exit 1 fi - apply: - name: Validate and apply shared edge + authorize-profile-change: + name: Authorize host profile migration needs: require-main-dispatch + if: ${{ inputs.allow-profile-change }} runs-on: ubuntu-latest - timeout-minutes: 15 + timeout-minutes: 5 environment: production steps: - - name: Check deployment configuration + - name: Record production authorization + run: echo 'Production environment authorization obtained for this host profile migration.' + + apply: + name: Validate and apply host edge + needs: + - require-main-dispatch + - authorize-profile-change + # Normal isolated staging does not enter or depend on the production environment. + # A migration must have completed its separate production approval gate. + if: ${{ !cancelled() && needs.require-main-dispatch.result == 'success' && (needs.authorize-profile-change.result == 'success' || (!inputs.allow-profile-change && needs.authorize-profile-change.result == 'skipped')) }} + runs-on: ubuntu-latest + timeout-minutes: 15 + environment: ${{ inputs.environment }} + steps: + - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 + with: + persist-credentials: false + + - name: Check deployment configuration and edge ownership + id: target env: + DEPLOY_ENVIRONMENT: ${{ inputs.environment }} + EDGE_PROFILE: ${{ vars.EDGE_PROFILE }} DEPLOY_HOST: ${{ secrets.DEPLOY_HOST }} DEPLOY_KNOWN_HOSTS: ${{ vars.DEPLOY_KNOWN_HOSTS }} DEPLOY_USER: ${{ secrets.DEPLOY_USER }} DEPLOY_SSH_KEY: ${{ secrets.DEPLOY_SSH_KEY }} + STAGING_URL: ${{ vars.STAGING_SMOKE_TEST_URL }} + PRODUCTION_URL: ${{ vars.PRODUCTION_SMOKE_TEST_URL }} run: | set -euo pipefail + bash scripts/resolve-edge-profile.sh "$DEPLOY_ENVIRONMENT" "$EDGE_PROFILE" >> "$GITHUB_OUTPUT" + if [[ "$DEPLOY_ENVIRONMENT" != production && "$EDGE_PROFILE" == shared ]]; then + echo '::error::Shared edge updates require the production environment.' + exit 1 + fi for name in DEPLOY_HOST DEPLOY_KNOWN_HOSTS DEPLOY_USER DEPLOY_SSH_KEY; do if [ -z "${!name}" ]; then echo "::error::${name} is unavailable to the edge job." @@ -50,10 +94,17 @@ jobs: done [[ "$DEPLOY_HOST" =~ ^[a-zA-Z0-9.-]+$ ]] || { echo '::error::DEPLOY_HOST is invalid.'; exit 1; } [[ "$DEPLOY_USER" =~ ^[a-zA-Z_][a-zA-Z0-9_-]*$ ]] || { echo '::error::DEPLOY_USER is invalid.'; exit 1; } + if [[ "$EDGE_PROFILE" == shared || "$EDGE_PROFILE" == staging ]]; then + [[ "$STAGING_URL" =~ ^https://[a-zA-Z0-9.-]+/?$ ]] || { echo '::error::STAGING_SMOKE_TEST_URL must be an HTTPS origin.'; exit 1; } + fi + if [[ "$EDGE_PROFILE" == shared || "$EDGE_PROFILE" == production ]]; then + [[ "$PRODUCTION_URL" =~ ^https://[a-zA-Z0-9.-]+/?$ ]] || { echo '::error::PRODUCTION_SMOKE_TEST_URL must be an HTTPS origin.'; exit 1; } + fi - - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 - with: - persist-credentials: false + - name: Render only this host's environments + env: + EDGE_PROFILE: ${{ steps.target.outputs.profile }} + run: bash scripts/render-edge.sh "$EDGE_PROFILE" artifacts/edge-candidate - name: Set up SSH env: @@ -86,30 +137,30 @@ jobs: - name: Transfer candidate and apply it env: CANDIDATE_DIR: /srv/opengamebuilder/edge-candidates/${{ github.run_id }}-${{ github.run_attempt }} + DEPLOY_ENVIRONMENT: ${{ inputs.environment }} + ALLOW_PROFILE_CHANGE: ${{ inputs.allow-profile-change }} run: | set -euo pipefail - ssh deployment \ - mkdir -p "$CANDIDATE_DIR" - rsync -az \ - ./deploy/edge/compose.yml ./deploy/edge/Caddyfile \ + ssh deployment mkdir -p "$CANDIDATE_DIR" + rsync -az artifacts/edge-candidate/compose.yml artifacts/edge-candidate/Caddyfile \ deployment:"$CANDIDATE_DIR/" ssh deployment \ - bash -s -- "$CANDIDATE_DIR" < ./scripts/apply-edge.sh + bash -s -- "$CANDIDATE_DIR" "$DEPLOY_ENVIRONMENT" "$ALLOW_PROFILE_CHANGE" < scripts/apply-edge.sh - - name: Check both sites after edge update + - name: Check selected edge routes on the deployed host env: + EDGE_ENVIRONMENTS: ${{ steps.target.outputs.environments }} STAGING_URL: ${{ vars.STAGING_SMOKE_TEST_URL }} PRODUCTION_URL: ${{ vars.PRODUCTION_SMOKE_TEST_URL }} run: | set -euo pipefail - for name in STAGING_URL PRODUCTION_URL; do - url="${!name}" - if [ -z "$url" ]; then - echo "::error::${name} is not configured." - exit 1 - fi - curl --fail --show-error --silent --location \ - --connect-timeout 10 --max-time 30 --retry-max-time 180 \ - --retry 5 --retry-delay 5 --retry-all-errors \ - "${url%/}/api/alive" >/dev/null + for environment in $EDGE_ENVIRONMENTS; do + case "$environment" in + staging) url="$STAGING_URL" ;; + production) url="$PRODUCTION_URL" ;; + *) echo 'Unexpected edge environment.' >&2; exit 1 ;; + esac + # Probe the host reached through pinned SSH, not a possibly old DNS target. + # /health is edge readiness; initial setup does not need an API deployed yet. + ssh deployment bash -s -- "$environment" "$url" < scripts/check-edge-health.sh done diff --git a/deploy/edge/Caddyfile b/deploy/edge/Caddyfile deleted file mode 100644 index c904339..0000000 --- a/deploy/edge/Caddyfile +++ /dev/null @@ -1,43 +0,0 @@ -www.opengamebuilder.com { - redir https://opengamebuilder.com{uri} permanent -} - -opengamebuilder.com { - encode zstd gzip - - handle /health { - respond "ok production" 200 - } - - handle /api/* { - reverse_proxy api-production:8080 - } - - handle { - root * /srv/opengamebuilder/production/web - try_files {path} /index.html - file_server { - precompressed br gzip - } - } -} - -staging.opengamebuilder.com { - encode zstd gzip - - handle /health { - respond "ok staging" 200 - } - - handle /api/* { - reverse_proxy api-staging:8080 - } - - handle { - root * /srv/opengamebuilder/staging/web - try_files {path} /index.html - file_server { - precompressed br gzip - } - } -} diff --git a/deploy/edge/Caddyfile.production b/deploy/edge/Caddyfile.production new file mode 100644 index 0000000..e5f58c5 --- /dev/null +++ b/deploy/edge/Caddyfile.production @@ -0,0 +1,23 @@ +www.opengamebuilder.com { + redir https://opengamebuilder.com{uri} permanent +} + +opengamebuilder.com { + encode zstd gzip + + handle /health { + respond "ok production" 200 + } + + handle /api/* { + reverse_proxy api-production:8080 + } + + handle { + root * /srv/opengamebuilder/production/web + try_files {path} /index.html + file_server { + precompressed br gzip + } + } +} diff --git a/deploy/edge/Caddyfile.staging b/deploy/edge/Caddyfile.staging new file mode 100644 index 0000000..da93b01 --- /dev/null +++ b/deploy/edge/Caddyfile.staging @@ -0,0 +1,19 @@ +staging.opengamebuilder.com { + encode zstd gzip + + handle /health { + respond "ok staging" 200 + } + + handle /api/* { + reverse_proxy api-staging:8080 + } + + handle { + root * /srv/opengamebuilder/staging/web + try_files {path} /index.html + file_server { + precompressed br gzip + } + } +} diff --git a/deploy/edge/compose.production.yml b/deploy/edge/compose.production.yml new file mode 100644 index 0000000..3828820 --- /dev/null +++ b/deploy/edge/compose.production.yml @@ -0,0 +1,4 @@ +services: + caddy: + volumes: + - ../production/web:/srv/opengamebuilder/production/web:ro diff --git a/deploy/edge/compose.staging.yml b/deploy/edge/compose.staging.yml new file mode 100644 index 0000000..8521fcf --- /dev/null +++ b/deploy/edge/compose.staging.yml @@ -0,0 +1,4 @@ +services: + caddy: + volumes: + - ../staging/web:/srv/opengamebuilder/staging/web:ro diff --git a/deploy/edge/compose.yml b/deploy/edge/compose.yml index 8faf8c2..69c6b23 100644 --- a/deploy/edge/compose.yml +++ b/deploy/edge/compose.yml @@ -9,8 +9,6 @@ services: - "443:443" volumes: - ./Caddyfile:/etc/caddy/Caddyfile:ro - - ../production/web:/srv/opengamebuilder/production/web:ro - - ../staging/web:/srv/opengamebuilder/staging/web:ro - caddy_data:/data - caddy_config:/config networks: diff --git a/docs/foundation-checklist.md b/docs/foundation-checklist.md index 41cd211..65157e2 100644 --- a/docs/foundation-checklist.md +++ b/docs/foundation-checklist.md @@ -240,6 +240,16 @@ Live staging deployment and production availability read-back remain to be verified after the protected workflow change is merged; no edge or application deployment was run for this local implementation. +**Host independence follow-up:** deployment environments explicitly select a +`shared`, `staging`, or `production` edge profile. Each host owns its own network, +proxy, and certificate volumes; an isolated profile contains no routes or web +mounts for the other environment. Only shared staging deployments probe +production availability. Profile changes require explicit production approval, +and application preflight checks the installed profile before transferring a +release. See [hosting setup](setup/hosting.md#host-edge-changes) for adoption and +future separation. Local regression/configuration validation does not establish +live separate-host acceptance; no host migration is performed by this change. + ### 13. Make builds portable and promote identifiable artifacts - [x] Prefer deployed frontend requests to the current origin's `/api` rather than diff --git a/docs/quality/testing.md b/docs/quality/testing.md index b0d0f6a..3eb46d5 100644 --- a/docs/quality/testing.md +++ b/docs/quality/testing.md @@ -71,7 +71,7 @@ repositories, mocks GitHub CLI calls and Git pushes, and never publishes a branch, tag, or release. CI and deployment validation run it after the solution tests and upload its log on failure. -The shared-edge apply script also has an isolated Bash test: +The host-edge apply script also has an isolated Bash test: ```pwsh & 'C:\Program Files\Git\bin\bash.exe' tests/deploy-edge/run.sh @@ -80,10 +80,25 @@ The shared-edge apply script also has an isolated Bash test: It mocks Docker and verifies invalid-candidate rejection, Caddyfile-only reload, failed-reload restoration, intentional Compose updates, first-time setup, and recovery of a stopped edge even when the candidate files are unchanged. +It also verifies environment/profile ownership and explicit migration approval. CI runs it without SSH, Docker, deployment credentials, or service changes. Live staging and production availability must still be checked by the deployment smoke tests after the workflow change reaches `main`. +Host topology has a separate gate requiring the **Docker Compose CLI**, but not +a running Docker daemon, SSH, or deployment credentials: + +```pwsh +& 'C:\Program Files\Git\bin\bash.exe' tests/deploy-topology/run.sh +``` + +It renders shared, staging-only, and production-only candidates and checks real +Compose normalization, image pins, stable certificate volumes, relative mounts, +and absence of the other environment from isolated profiles. Invalid/missing +profiles fail closed. Mocked curl tests ensure the edge readiness check targets +the selected SSH host even before DNS cutover. CI/deployment validation run it; +ordinary .NET tests still do not require Docker. + Application activation and rollback have an isolated host-script test: ```pwsh diff --git a/docs/release/README.md b/docs/release/README.md index e02c937..fdad51f 100644 --- a/docs/release/README.md +++ b/docs/release/README.md @@ -9,6 +9,12 @@ workflows you need to know about: | 🚀 **CD Production** | Manually dispatched (usually from `main`) with a `ref` input | Validate, build, test, deploy to production, smoke test, tag, release, follow-up PR | | 🩹 **Prepare Patch** | Manually dispatched | Create `patch/vX.Y.(Z+1)` branch and a version-bump PR off the latest tag | +Host infrastructure is separate: **🌐 CD Edge** updates the selected host's +explicit edge profile, not an application release. Staging and production may +use separate hosts and SSH keys. Configure their `EDGE_PROFILE` variables before +deploying; see [hosting setup](../setup/hosting.md). Shared-host staging keeps +the production-health guard; isolated staging has no production dependency. + The single source of truth for the version is `` in `Directory.Build.props`. See [versioning.md](./versioning.md). diff --git a/docs/setup/deployment-host-key.md b/docs/setup/deployment-host-key.md index f72d8c3..cb6a57c 100644 --- a/docs/setup/deployment-host-key.md +++ b/docs/setup/deployment-host-key.md @@ -1,9 +1,11 @@ # Pin the deployment server's SSH host key Use this guide from Windows PowerShell on the computer where you already log in -to the Hetzner server with OpenSSH. It transfers the server identity trusted by -that computer into GitHub's deployment configuration. The normal path needs no -root password, password reset, server restart, or SSH configuration change. +to the selected deployment server with OpenSSH. It transfers the server identity +trusted by that computer into one GitHub environment's deployment configuration. +Staging and production may use the same server or different servers; complete +this guide separately for each environment. The normal path needs no root +password, password reset, server restart, or SSH configuration change. ## What we are doing @@ -47,13 +49,18 @@ WSL, its saved server keys may be in a different store. An unknown-host error from Windows OpenSSH does not justify accepting a new key; verify or transfer the existing trusted entry first. -## 2. Supply your existing connection details +## 2. Select one environment and supply its existing connection details ```powershell $repo = 'OpenGameBuilder/opengamebuilder' +$deploymentEnvironment = (Read-Host 'Deployment environment to configure (staging or production)').Trim().ToLowerInvariant() +if ($deploymentEnvironment -notin @('staging', 'production')) { + throw 'Select staging or production. This run configures only that environment.' +} + $sshTarget = Read-Host 'Your usual SSH destination (for example root@SERVER_IP or an SSH config alias)' $sshIdentity = Read-Host 'Private-key file path if you normally use ssh -i (otherwise press Enter)' -$deployHost = Read-Host 'Exact DEPLOY_HOST value used by GitHub (server IP or DNS name)' +$deployHost = Read-Host "Exact DEPLOY_HOST value for $deploymentEnvironment in GitHub (server IP or DNS name)" if ($deployHost -notmatch '^[a-zA-Z0-9.-]+$') { throw 'Enter only the deployment server IP or DNS name, without user, URL scheme, path, or port.' @@ -70,16 +77,18 @@ if ($sshIdentity) { ``` For example, if your normal command is `ssh -i C:\Keys\hetzner root@203.0.113.10`, -enter `root@203.0.113.10` at the first prompt and `C:\Keys\hetzner` at the second. +enter `root@203.0.113.10` for the SSH destination and `C:\Keys\hetzner` for the key path. If you normally type `ssh ogb`, enter `ogb` and leave the key path empty. Enter paths without surrounding quotation marks. Use your existing administrative login; you do not need GitHub's private deployment key. -`DEPLOY_HOST` is the SSH server address you configured in GitHub, which may differ -from the public website address or a local SSH alias. GitHub cannot reveal an -existing secret value; use your deployment setup records. The workflow currently -uses SSH port 22. The server reached by `$sshTarget` must be the same server as -`$deployHost`; check the IP in Hetzner's server overview if necessary. +`DEPLOY_HOST` is the SSH server address configured in the selected GitHub +environment, which may differ from the public website address or a local SSH +alias. GitHub cannot reveal an existing secret value; use your deployment setup +records. The workflow currently uses SSH port 22. The server reached by +`$sshTarget` must be the same server as `$deployHost`; check the IP in Hetzner's +server overview if necessary. Do not reuse the other environment's connection +details unless you have confirmed that they identify the intended server. ## 3. Read the public key through the existing trusted connection @@ -123,68 +132,59 @@ That complete line is the GitHub variable value. The second output includes its `SHA256:...` fingerprint, a short identifier to retain for later comparison. The fingerprint alone is not a usable `known_hosts` entry. -## 4. Save the entry in GitHub +## 4. Save the entry only in the selected GitHub environment -The current hosting design puts staging and production on the same server. -If **both environments have the same `DEPLOY_HOST`**, run: +This command changes only the environment selected in step 2: ```powershell -gh variable set DEPLOY_KNOWN_HOSTS --repo $repo --env staging --body $deployKnownHosts -if ($LASTEXITCODE -ne 0) { throw 'Could not save the staging variable.' } - -gh variable set DEPLOY_KNOWN_HOSTS --repo $repo --env production --body $deployKnownHosts -if ($LASTEXITCODE -ne 0) { throw 'Could not save the production variable.' } +gh variable set DEPLOY_KNOWN_HOSTS --repo $repo --env $deploymentEnvironment --body $deployKnownHosts +if ($LASTEXITCODE -ne 0) { throw "Could not save the $deploymentEnvironment variable." } ``` -These commands create or update **environment variables**, as required by the -workflow's `vars.DEPLOY_KNOWN_HOSTS` reference. They do not deploy anything. +This creates or updates an **environment variable**, as required by the +workflow's `vars.DEPLOY_KNOWN_HOSTS` reference. It does not deploy anything. [GitHub CLI documents the environment and value flags](https://cli.github.com/manual/gh_variable_set). -If production uses a different hostname for the **same server**, replace the -production command above with: - -```powershell -$productionHost = Read-Host 'Exact production DEPLOY_HOST for this same server' -if ($productionHost -notmatch '^[a-zA-Z0-9.-]+$') { throw 'Invalid server address.' } -$productionKnownHosts = "$productionHost $($keyParts[0]) $($keyParts[1])" -gh variable set DEPLOY_KNOWN_HOSTS --repo $repo --env production --body $productionKnownHosts -if ($LASTEXITCODE -ne 0) { throw 'Could not save the production variable.' } -``` - -If the environments are on different servers, repeat steps 2-3 against the other -server and save that server's entry only to its own environment. +After completing step 5, repeat steps 2-5 for the other environment with its own +connection details. If both environments intentionally use the same physical +server, they may have the same public host key and fingerprint. The host field +in each entry must still match that environment's exact `DEPLOY_HOST`, including +when two DNS names identify the same server. Separate servers must each have +their own host key read and trusted; do not copy one server's entry to the other. For a browser-based alternative, copy `$deployKnownHosts` and use repository -**Settings → Environments → staging → Environment variables → Add variable**, -with name `DEPLOY_KNOWN_HOSTS`. Repeat under production with its matching entry. +**Settings → Environments → the environment selected in step 2 → Environment +variables → Add variable**, with name `DEPLOY_KNOWN_HOSTS`. Do not update the +other environment during this run. [GitHub documents this settings path](https://docs.github.com/en/actions/how-tos/write-workflows/choose-what-workflows-do/use-variables#creating-configuration-variables-for-an-environment). ## 5. Read back what GitHub stored ```powershell -$stagingSaved = gh api "repos/$repo/environments/staging/variables/DEPLOY_KNOWN_HOSTS" --jq .value -if ($LASTEXITCODE -ne 0) { throw 'Could not read the staging variable.' } -$productionSaved = gh api "repos/$repo/environments/production/variables/DEPLOY_KNOWN_HOSTS" --jq .value -if ($LASTEXITCODE -ne 0) { throw 'Could not read the production variable.' } - -$stagingSaved -$stagingSaved | ssh-keygen -E sha256 -lf - -if ($LASTEXITCODE -ne 0) { throw 'Invalid staging host-key entry.' } -$productionSaved -$productionSaved | ssh-keygen -E sha256 -lf - -if ($LASTEXITCODE -ne 0) { throw 'Invalid production host-key entry.' } -``` +$savedKnownHosts = gh api "repos/$repo/environments/$deploymentEnvironment/variables/DEPLOY_KNOWN_HOSTS" --jq .value +if ($LASTEXITCODE -ne 0) { throw "Could not read the $deploymentEnvironment variable." } +if ($savedKnownHosts -ne $deployKnownHosts) { + throw 'The saved entry does not match the reviewed entry from step 3.' +} -Check each hostname against its environment's `DEPLOY_HOST`. For a shared server, -both fingerprints must equal the fingerprint from step 3. This verifies the -public configuration was transferred correctly. The deployment login itself is -verified by the later staging run, using GitHub's existing deployment secret. +$savedKnownHosts +$savedKnownHosts | ssh-keygen -E sha256 -lf - +if ($LASTEXITCODE -ne 0) { throw "Invalid $deploymentEnvironment host-key entry." } +``` -The new workflow takes effect after this branch passes CI and is merged. A push -to `main` already triggers CD Staging, so no extra manual deployment command is -needed for this setup. In that run, **Set up SSH** should print the same -fingerprint and the deployment must pass strict SSH checking and its smoke test. -Keep the checklist's SSH/staging acceptance open until those checks pass. +Check the hostname against the selected environment's `DEPLOY_HOST` and the +fingerprint against step 3. This verifies that this environment's public +configuration was transferred correctly; it does not prove a deployment login +or change the other environment's configuration. + +Verify the deployment login during the next authorized deployment to this +environment, using GitHub's existing deployment secret. A push to `main` +normally triggers CD Staging; production deployment is separately dispatched +and approved. Do not dispatch a production release solely to save a host key. +In the selected environment's run, **Set up SSH** should print the same +fingerprint, and subsequent remote steps must pass strict SSH checking and the +deployment smoke test. Record acceptance separately for each environment; a +passing staging run does not verify production's SSH configuration. ## If the existing SSH host key is not trusted diff --git a/docs/setup/github.md b/docs/setup/github.md index 9815e0f..e8234b6 100644 --- a/docs/setup/github.md +++ b/docs/setup/github.md @@ -213,13 +213,20 @@ waiting jobs, `main`-only is not an absolute restriction against an administrato Revisit self-review and bypass when a second trusted release operator exists. Each environment holds only its own `DEPLOY_HOST`, `DEPLOY_USER`, and -`DEPLOY_SSH_KEY` secrets, plus a `DEPLOY_KNOWN_HOSTS` environment variable. -The variable contains the complete OpenSSH `known_hosts` entry for that +`DEPLOY_SSH_KEY` secrets, plus `DEPLOY_KNOWN_HOSTS` and `EDGE_PROFILE` environment +variables. The hosts and accounts may be the same or different. Set `EDGE_PROFILE` +explicitly: `shared` in both environments for a shared server, or `staging` and +`production` respectively for separate servers. There is no default. The edge +workflow selects the environment whose credentials to use; a shared edge or +profile migration requires production approval. See +[host profiles and migration](hosting.md#host-edge-changes). +`DEPLOY_KNOWN_HOSTS` contains the complete OpenSSH `known_hosts` entry for that environment's `DEPLOY_HOST`; the public host key is configuration, not a secret. Follow the [deployment host-key guide](deployment-host-key.md) to read the public key through an existing SSH connection that strictly checks a saved, trusted -server key, construct the entry, set both environment variables, and read them -back. This carries forward the administrator's existing trust decision; it is +server key, construct the entry, set the selected environment's host-key variable, +and read it back. Repeat for the other environment and its host. This carries +forward the administrator's existing trust decision; it is not independent verification of an unverified first connection. A provider console or another authenticated independent channel is needed when the saved key is missing, changed without explanation, or untrusted. Do not populate the diff --git a/docs/setup/hosting.md b/docs/setup/hosting.md index 088576e..37caf27 100644 --- a/docs/setup/hosting.md +++ b/docs/setup/hosting.md @@ -1,9 +1,25 @@ # Hosting setup -Production uses the Compose files in `deploy/production` and `deploy/edge`; -staging uses `deploy/staging` and the same edge. The shared Caddy service routes -both sites and reads their published web files. The `ogb-edge` Docker network -connects Caddy to both API services. Aspire is only the local launcher. +Staging and production are independent deployment targets. Each GitHub +environment supplies its own SSH host, account, key, and trusted host-key entry. +They can point at the same server or different servers. Aspire is only the local +launcher; application hosting uses `deploy/staging` and `deploy/production`. + +Each server runs **one host-local Caddy edge**, explicitly configured through +the `EDGE_PROFILE` variable on its GitHub deployment environment: + +| Layout | Staging environment's `EDGE_PROFILE` | Production environment's `EDGE_PROFILE` | +| --- | --- | --- | +| Both applications on one server (current layout) | `shared` | `shared` | +| Separate servers | `staging` | `production` | + +There is no default. Missing, unknown, or mismatched profiles fail closed. +`shared` is an intentional configuration, not an inferred relationship between +hosts. Each host has its own `ogb-edge` Docker network and Caddy certificate +volumes; the identical names on separate machines do not connect those machines. +An isolated profile has no routes, web mounts, or API aliases for the other +environment. Shared hosting still shares host/Docker privileges and outage risk; +it is not a security isolation boundary. ## Runtime trust boundaries @@ -26,7 +42,7 @@ Caddy terminates HTTPS and replaces client-supplied `X-Forwarded-For`, activation, `scripts/deploy-app.sh` reads the actual `ogb-edge` IPAM subnets and passes only those CIDRs to the API. Forwarded-header middleware accepts one hop from those networks and runs before HTTPS redirection. It does not trust every -private address or an arbitrary direct client. If the shared network has no IPAM +private address or an arbitrary direct client. If the host network has no IPAM subnet, activation fails before the candidate starts. The host deployment account is separate from the API process identity. It needs @@ -35,14 +51,20 @@ key, and the Docker operations used by the reviewed deployment scripts. Docker daemon access is host-privileged; do not reuse this account or key for application traffic, interactive contributor access, or unrelated automation. -## Shared edge changes +## Host edge changes + +**🌐 CD Edge** (`.github/workflows/cd-edge.yml`) owns edge updates. Dispatch it +**from `main`**, selecting the GitHub environment that supplies the target host's +credentials and approval rules. An edge serving both environments must be +dispatched through **production**, never staging. Edge updates queue rather than +cancelling one another, even when two environments happen to share a host. -The shared edge is owned by **🌐 CD Shared Edge** (`.github/workflows/cd-edge.yml`). -After a reviewed change to `deploy/edge/Caddyfile` or `deploy/edge/compose.yml` -reaches `main`, dispatch that workflow **from `main`**. It uses the `production` -environment's reviewer approval and deployment credentials, and queues edge -updates without cancelling an update already in progress. Run it once before -the first application deployment to create the shared network and Caddy service. +`scripts/render-edge.sh` combines the common pinned `deploy/edge/compose.yml` +with the selected `compose..yml` mount overrides and +`Caddyfile.` site fragments. It produces standalone `compose.yml` +and `Caddyfile` candidates with an explicit profile marker. Only those rendered +files are transferred; do not deploy the common Compose base by itself. +Run the edge workflow before the first application deployment on a new host. The workflow transfers candidate files to a separate directory on the host. `scripts/apply-edge.sh` checks the candidate Compose definition, pulls its Caddy @@ -52,24 +74,90 @@ files remain in place. A Caddyfile-only update copies the validated file into the existing bind mount and calls `caddy reload`; it does not recreate Caddy. An intentional Compose change runs `docker compose up -d`, which may recreate the service. A failed reload restores the previous files and attempts to reload -the previous configuration. The workflow checks both API liveness URLs after -the update. Inspect the job log and host state if activation or recovery fails; +the previous configuration. The workflow checks each selected site's `/health` +over HTTPS **on the SSH target host**, forcing its hostname to loopback while +still validating its TLS certificate. This tests edge readiness without needing +an application installed, and cannot accidentally validate an old DNS target. +DNS and certificate issuance must be ready for this check to pass; a new edge +may be installed while this check fails during a planned DNS cutover. It is not +application acceptance: each app deployment separately checks the real API and +browser revision. Inspect the job log and host state if activation or recovery fails; do not assume a failed job automatically restored service availability. An unchanged candidate leaves a running Caddy container alone, but starts it if it is stopped. This cannot repair a port conflict with another host process. -The staging and production application workflows update only their own release -directory and API service. They require the shared edge network to exist and -never sync or restart Caddy. Staging deployments queue instead of cancelling an -in-flight deployment. Production `/api/alive` is checked before and after a -staging update; a failed preflight stops the update. +The installed profile marker must agree with an application's configured +profile before release files are transferred. A normal edge update also refuses +to change an installed profile. Intentional migrations require +`allow-profile-change` and **production approval** in a separate authorization +job, then use the selected environment's own credentials and approval rules to +apply the change. Normal isolated staging updates do not enter the production +environment. Existing unmarked edge files +from the previous shared-only setup may be adopted only as `shared` through +production; they cannot silently become an isolated edge. + +Application workflows update only their own release directory and API. They +require their host's edge network and never sync or restart Caddy. Staging +deployments queue instead of cancelling an in-flight deployment. Production +`/api/alive` is checked before and after staging updates **only when staging's +profile is `shared`**. Isolated staging neither requires production's URL nor +depends on production being available. + +### Adopt explicit profiles on the current server + +Before merging this workflow change, set the environment variable +`EDGE_PROFILE=shared` in **both** GitHub environments under **Settings > +Environments > staging/production > Environment variables**. Keep their existing +SSH secrets and verified host-key entries. These are configuration changes, not +an instruction to recreate the server or reset a password. + +After merge, run **CD Edge** from `main`, select `production`, leave +`allow-profile-change` off, and approve it. This adopts the existing shared +layout while preserving the Compose project and certificate-volume names. +The old unmarked shared layout remains recognizable by application preflight +during this one-time adoption. Confirm the host-local HTTPS checks and the next +staging application's browser check pass. + +### Later: move staging to its own server + +1. Provision the new host with Docker/Compose, curl, a deployment account with + write access under `/srv/opengamebuilder`, and ports 80/443 available for its + edge. Do not run a competing native Caddy service. +2. Verify the new server's SSH host key through the + [host-key guide](deployment-host-key.md). Change **only staging's** SSH + secrets/host-key variable to the new host and set its `EDGE_PROFILE=staging`. +3. Coordinate DNS and certificate issuance for `staging.opengamebuilder.com`. + Run **CD Edge** from `main` with environment `staging`, then **CD Staging**. + A first edge check may need rerunning after DNS/certificates are ready; it + deliberately does not accept a successful response from the old server. +4. Verify the new host's edge, API revision, and browser flow. Then set + **production's** `EDGE_PROFILE=production` and dispatch **CD Edge** through + production with `allow-profile-change` enabled. This deliberately removes the + obsolete staging route/mount from the old host; coordinate the brief proxy + interruption. Keep the old staging data until the cutover is accepted. + +Pause conflicting deployment runs during the move. Moving credentials/profile +alone does not migrate application data, DNS, certificates, or release history. +Retain a documented recovery plan; no migration or cleanup is performed by a +normal application deployment. + +To move **production** instead, provision and verify its new host, set +production's own SSH configuration and `EDGE_PROFILE=production`, then run its +edge and application workflows and verify the cutover. Once production is +accepted on the new host, leave staging's SSH configuration pointing at the old +host, set staging's `EDGE_PROFILE=staging`, and dispatch **CD Edge** with +environment `staging` and `allow-profile-change` enabled. Its separate +production approval authorizes removing the old production route/mount; the +apply job still uses staging's credentials to reach the old host. You do not +need to point production credentials back at that host. The same DNS, +certificate, data-retention, and recovery precautions apply in either direction. ### Host Caddy conflicts and read-only verification Only the Docker edge should own this server's ports 80 and 443. A separately installed `caddy.service` can start at boot, occupy those ports, and prevent `ogb-edge-caddy-1` from starting. An API container being up does not establish -that either public site is reachable. +that a public site is reachable. In an existing **root SSH session on the Hetzner server**, from any directory, these commands inspect state without changing services or printing environment @@ -79,10 +167,13 @@ variables or private keys: systemctl is-enabled caddy systemctl is-active caddy ss -ltnp '( sport = :80 or sport = :443 )' -docker inspect --format '{{.Name}} image={{.Config.Image}} status={{.State.Status}} restarts={{.RestartCount}}' ogb-edge-caddy-1 ogb-staging-api-1 ogb-production-api-1 +docker ps --all --filter name=ogb- --format 'table {{.Names}}\t{{.Image}}\t{{.Status}}' +docker inspect --format '{{.Name}} image={{.Config.Image}} status={{.State.Status}} restarts={{.RestartCount}}' ogb-edge-caddy-1 docker logs --since 2h --tail 150 ogb-edge-caddy-1 2>&1 -docker logs --since 2h --tail 150 ogb-staging-api-1 2>&1 -docker logs --since 2h --tail 150 ogb-production-api-1 2>&1 +# Choose an application actually deployed on this host; repeat for the other +# only if this is a shared host. +environment=staging +docker logs --since 2h --tail 150 "ogb-${environment}-api-1" 2>&1 ``` The native Caddy service should be disabled/inactive (or not installed). @@ -92,23 +183,23 @@ logs may contain client addresses or request data; redact sensitive content before sharing them. If inspection confirms the native service is the conflicting, superseded OGB -proxy, coordinate a short interruption for **both sites**, then run on the host: +proxy, coordinate a short interruption for **the sites on this host**, then run: ```bash systemctl disable --now caddy docker start ogb-edge-caddy-1 -curl --fail --show-error https://opengamebuilder.com/api/alive -curl --fail --show-error https://staging.opengamebuilder.com/api/alive ``` Do not stop a service hosting unrelated sites. No root-password reset or web console is needed when the existing SSH session works. Restarting the container restores its existing image; it does **not** apply a new image pin. After an edge -change is reviewed and merged, use GitHub **Actions > CD Shared Edge > Run -workflow**, select `main`, and approve the production environment. A Compose -image change can recreate the shared proxy and briefly interrupt both sites. +change is reviewed and merged, use GitHub **Actions > CD Edge > Run workflow**, +select `main` and the correct environment (production for a shared host), and +approve it. A Compose image change can briefly interrupt the sites on that host. Verify the running image reference matches `deploy/edge/compose.yml` and that -both workflow liveness checks pass. +the selected host's edge checks and its deployed applications' smoke tests pass. + +## Application artifacts After source validation, the package job produces one Release web archive. After environment approval, deployment verifies that archive and builds and @@ -183,8 +274,8 @@ not a zero-downtime atomic swap of the API and web processes. The section 14 staging failure and rollback rehearsal passed on 2026-09-20; see the [checklist evidence](../foundation-checklist.md#14-make-rollout-atomic-and-rollback-explicit). -To recover an edge change, fix the candidate on `main` and dispatch **CD Shared -Edge** again. For an urgent host-side recovery, use the last known-good edge +To recover an edge change, fix the candidate on `main` and dispatch **CD Edge** +for the affected host again. For an urgent host-side recovery, use the last known-good edge files retained in the host's `.rollback.*` directory after a failed activation, validate them with `caddy validate`, and reload Caddy. Do not restart or recreate Caddy for a Caddyfile-only correction. Coordinate host-side changes diff --git a/scripts/apply-edge.sh b/scripts/apply-edge.sh index 459f3df..e1988e8 100644 --- a/scripts/apply-edge.sh +++ b/scripts/apply-edge.sh @@ -3,14 +3,80 @@ set -euo pipefail candidate_dir="${1:?Candidate directory is required}" +deployment_environment="${2:?Deployment environment is required}" +# The workflow obtains separate production-environment approval before setting +# this flag; SSH still targets the selected deployment environment's host. +allow_profile_change="${3:-false}" +if [[ "$deployment_environment" != staging && "$deployment_environment" != production ]]; then + echo "Expected deployment environment staging or production." >&2 + exit 1 +fi +if [[ "$allow_profile_change" != true && "$allow_profile_change" != false ]]; then + echo "Expected allow-profile-change to be true or false." >&2 + exit 1 +fi +if (( $# > 3 )); then + echo "Usage: apply-edge.sh [allow-profile-change=false]" >&2 + exit 1 +fi edge_dir="${EDGE_DIR:-/srv/opengamebuilder/edge}" candidate_compose="${candidate_dir}/compose.yml" candidate_caddyfile="${candidate_dir}/Caddyfile" active_compose="${edge_dir}/compose.yml" active_caddyfile="${edge_dir}/Caddyfile" -test -f "$candidate_compose" -test -f "$candidate_caddyfile" +if [[ ! -f "$candidate_compose" || ! -f "$candidate_caddyfile" ]]; then + echo "Edge candidate requires compose.yml and Caddyfile." >&2 + exit 1 +fi + +# Keep this script self-contained: the workflow streams it over SSH. The +# reserved marker must occur exactly once and be the entire first line. +read_edge_profile() { + local caddyfile="$1" marker_count first_line + marker_count="$(awk 'tolower($0) ~ /ogb-edge-profile/ { count++ } END { print count+0 }' "$caddyfile")" + if [[ "$marker_count" == 0 ]]; then + printf '%s\n' legacy + return + fi + first_line="$(head -n 1 "$caddyfile")" + if [[ "$marker_count" != 1 || ! "$first_line" =~ ^\#\ ogb-edge-profile:\ (shared|staging|production)$ ]]; then + echo "Invalid edge profile marker in ${caddyfile}; expected one exact marker on the first line." >&2 + return 1 + fi + printf '%s\n' "${BASH_REMATCH[1]}" +} + +candidate_profile="$(read_edge_profile "$candidate_caddyfile")" +if [[ "$candidate_profile" == legacy ]]; then + echo "Candidate Caddyfile requires an edge profile marker." >&2 + exit 1 +fi +if [[ "$candidate_profile" != shared && "$candidate_profile" != "$deployment_environment" ]]; then + echo "Candidate profile ${candidate_profile} does not include deployment environment ${deployment_environment}." >&2 + exit 1 +fi +if [[ "$candidate_profile" == shared && "$deployment_environment" != production ]]; then + echo "Shared edge updates require the production environment." >&2 + exit 1 +fi + +if [[ -f "$active_caddyfile" ]]; then + active_profile="$(read_edge_profile "$active_caddyfile")" + if [[ "$active_profile" == legacy ]]; then + # The pre-profile deployment format always served both environments. + if [[ "$candidate_profile" != shared || "$deployment_environment" != production ]]; then + echo "Unmarked legacy edge can only adopt the shared profile through production." >&2 + exit 1 + fi + elif [[ "$active_profile" != "$candidate_profile" ]] && + [[ "$allow_profile_change" != true ]]; then + echo "Changing the installed edge profile requires allow-profile-change=true after production approval." >&2 + exit 1 + fi +fi + +# Reject profile mistakes before directory creation, image pulls, or service work. mkdir -p "$edge_dir" # Compose resolves relative bind mounts from the permanent edge directory. @@ -26,6 +92,16 @@ docker run --rm --entrypoint caddy \ --mount "type=bind,src=${candidate_caddyfile},dst=/etc/caddy/Caddyfile,readonly" \ "$image" validate --config /etc/caddy/Caddyfile +# Create only this profile's web roots as the deploying user, before Docker can +# create missing bind-mount directories as root during first-time edge setup. +case "$candidate_profile" in + shared) edge_environments=(staging production) ;; + staging|production) edge_environments=("$candidate_profile") ;; +esac +for environment in "${edge_environments[@]}"; do + mkdir -p "${edge_dir}/../${environment}/web" +done + # Nothing above this line modifies the active edge files or service. compose_changed=true config_changed=true diff --git a/scripts/check-edge-health.sh b/scripts/check-edge-health.sh new file mode 100644 index 0000000..f74ecac --- /dev/null +++ b/scripts/check-edge-health.sh @@ -0,0 +1,18 @@ +#!/usr/bin/env bash +set -euo pipefail + +# Stream this script over pinned SSH. Force TLS to the edge on that host so +# pre-cutover DNS cannot accidentally validate the old server instead. +environment="${1:-}" +base_url="${2:-}" +[[ "$environment" == staging || "$environment" == production ]] || { echo 'Unknown edge environment.' >&2; exit 1; } +[[ "$base_url" =~ ^https://([a-zA-Z0-9.-]+)/?$ ]] || { echo 'Expected an HTTPS origin without a port or path.' >&2; exit 1; } +hostname="${BASH_REMATCH[1]}" +response="$(curl --fail --show-error --silent \ + --noproxy '*' \ + --resolve "${hostname}:443:127.0.0.1" \ + --connect-timeout 10 --max-time 30 --retry-max-time 180 \ + --retry 5 --retry-delay 5 --retry-all-errors \ + "${base_url%/}/health")" +[[ "$response" == "ok $environment" ]] || { echo "Unexpected ${environment} edge health response." >&2; exit 1; } +echo "Verified ${environment} HTTPS edge on this host." diff --git a/scripts/deploy-app.sh b/scripts/deploy-app.sh index 5fea1be..d2cfd19 100644 --- a/scripts/deploy-app.sh +++ b/scripts/deploy-app.sh @@ -115,12 +115,12 @@ web_sha="$(manifest_value "$manifest" WEB_SHA256)" [[ "$web_sha" =~ ^[0-9a-f]{64}$ ]] || die 'invalid web checksum' echo "$web_sha $incoming/web-release.tar.gz" | sha256sum --check --status || die 'web archive checksum mismatch' -docker network inspect ogb-edge >/dev/null || die 'shared edge network is missing' +docker network inspect ogb-edge >/dev/null || die 'host edge network is missing' mapfile -t proxy_networks < <( docker network inspect ogb-edge \ --format '{{range .IPAM.Config}}{{if .Subnet}}{{println .Subnet}}{{end}}{{end}}' ) -[[ "${#proxy_networks[@]}" -gt 0 ]] || die 'shared edge network has no configured subnet' +[[ "${#proxy_networks[@]}" -gt 0 ]] || die 'host edge network has no configured subnet' printf -v trusted_proxy_networks '%s;' "${proxy_networks[@]}" trusted_proxy_networks="${trusted_proxy_networks%;}" diff --git a/scripts/render-edge.sh b/scripts/render-edge.sh new file mode 100644 index 0000000..0d1a2e3 --- /dev/null +++ b/scripts/render-edge.sh @@ -0,0 +1,46 @@ +#!/usr/bin/env bash +# Render an explicit host profile into two standalone edge deployment files. +# This only uses the Compose CLI; it does not contact Docker Engine or start services. +set -euo pipefail + +if [[ "$#" != 2 || -z "$2" ]]; then + echo "Usage: render-edge.sh " >&2 + exit 1 +fi + +profile="$1" +case "$profile" in + shared) environments=(production staging) ;; + staging|production) environments=("$profile") ;; + *) echo "Unknown edge profile '$profile'; expected shared, staging, or production." >&2; exit 1 ;; +esac + +repo_root="$(cd "$(dirname "${BASH_SOURCE[0]}")/.." && pwd)" +source_dir="$repo_root/deploy/edge" +mkdir -p -- "$2" +candidate_dir="$(cd "$2" && pwd)" +if [[ "$candidate_dir" == "$source_dir" ]]; then + echo "The candidate directory must not overwrite the edge source templates." >&2 + exit 1 +fi + +temporary_dir="$(mktemp -d "$candidate_dir/.render.XXXXXXXX")" +trap 'rm -r -- "$temporary_dir"' EXIT +compose=(docker compose -f "$source_dir/compose.yml") +for environment in "${environments[@]}"; do + compose+=(-f "$source_dir/compose.$environment.yml") +done + +# Retain relative mounts so the result can be transferred to a different host +# and applied with that host's permanent edge directory as its project directory. +printf '# ogb-edge-profile: %s\n' "$profile" > "$temporary_dir/compose.yml" +"${compose[@]}" config --no-path-resolution >> "$temporary_dir/compose.yml" +printf '# ogb-edge-profile: %s\n' "$profile" > "$temporary_dir/Caddyfile" +for environment in "${environments[@]}"; do + printf '\n' >> "$temporary_dir/Caddyfile" + cat "$source_dir/Caddyfile.$environment" >> "$temporary_dir/Caddyfile" +done + +mv -- "$temporary_dir/compose.yml" "$candidate_dir/compose.yml" +mv -- "$temporary_dir/Caddyfile" "$candidate_dir/Caddyfile" +printf 'Rendered %s edge profile in %s\n' "$profile" "$candidate_dir" diff --git a/scripts/resolve-edge-profile.sh b/scripts/resolve-edge-profile.sh new file mode 100644 index 0000000..fc526e5 --- /dev/null +++ b/scripts/resolve-edge-profile.sh @@ -0,0 +1,28 @@ +#!/usr/bin/env bash +# Validate the host topology selected by a GitHub deployment environment. +set -euo pipefail + +if [[ "$#" != 2 ]]; then + echo "Usage: resolve-edge-profile.sh " >&2 + exit 1 +fi + +environment="$1" +profile="$2" +case "$environment:$profile" in + staging:staging|production:production|staging:shared|production:shared) ;; + *) + echo "EDGE_PROFILE must be explicitly set to '$environment' or 'shared' for a staging or production deployment environment." >&2 + exit 1 + ;; +esac + +check_production=false +if [[ "$environment:$profile" == staging:shared ]]; then + check_production=true +fi +environments="$profile" +if [[ "$profile" == shared ]]; then + environments="production staging" +fi +printf 'profile=%s\ncheck-production=%s\nenvironments=%s\n' "$profile" "$check_production" "$environments" diff --git a/scripts/verify-edge-profile.sh b/scripts/verify-edge-profile.sh new file mode 100644 index 0000000..3c49e2d --- /dev/null +++ b/scripts/verify-edge-profile.sh @@ -0,0 +1,50 @@ +#!/usr/bin/env bash +# Read-only application preflight for the edge installed on a deployment host. +set -euo pipefail + +edge_dir="${1:?Edge directory is required}" +expected_profile="${2:?Expected edge profile is required}" +if [[ "$expected_profile" != shared && "$expected_profile" != staging && "$expected_profile" != production ]]; then + echo "Expected edge profile shared staging or production." >&2 + exit 1 +fi +if (( $# != 2 )); then + echo "Usage: verify-edge-profile.sh " >&2 + exit 1 +fi +if [[ ! -f "$edge_dir/compose.yml" || ! -f "$edge_dir/Caddyfile" ]]; then + echo "Edge installation requires compose.yml and Caddyfile in ${edge_dir}." >&2 + exit 1 +fi + +# Self-contained for `ssh ... bash -s`: match the same reserved first-line +# marker accepted by apply-edge.sh without Docker calls or filesystem writes. +read_edge_profile() { + local caddyfile="$1" marker_count first_line + marker_count="$(awk 'tolower($0) ~ /ogb-edge-profile/ { count++ } END { print count+0 }' "$caddyfile")" + if [[ "$marker_count" == 0 ]]; then + printf '%s\n' legacy + return + fi + first_line="$(head -n 1 "$caddyfile")" + if [[ "$marker_count" != 1 || ! "$first_line" =~ ^\#\ ogb-edge-profile:\ (shared|staging|production)$ ]]; then + echo "Invalid edge profile marker in ${caddyfile}; expected one exact marker on the first line." >&2 + return 1 + fi + printf '%s\n' "${BASH_REMATCH[1]}" +} + +active_profile="$(read_edge_profile "$edge_dir/Caddyfile")" +if [[ "$active_profile" == legacy ]]; then + # Existing installations predate the marker and use the original shared edge. + if [[ "$expected_profile" != shared ]]; then + echo "Unmarked legacy edge is only valid for the shared profile." >&2 + exit 1 + fi + echo "Verified legacy shared edge profile." +elif [[ "$active_profile" != "$expected_profile" ]]; then + echo "Installed edge profile ${active_profile} does not match expected profile ${expected_profile}." >&2 + exit 1 +else + echo "Verified edge profile: ${active_profile}." +fi diff --git a/tests/deploy-edge/run.sh b/tests/deploy-edge/run.sh index 46acadc..6935222 100644 --- a/tests/deploy-edge/run.sh +++ b/tests/deploy-edge/run.sh @@ -17,25 +17,199 @@ case "$*" in *'validate --config'*) [[ "${FAIL_VALIDATE:-0}" != 1 ]] ;; *'exec -T caddy caddy reload'*) [[ "${FAIL_RELOAD:-0}" != 1 ]] ;; *'ps -q caddy') if [[ "${CADDY_RUNNING:-1}" == 1 ]]; then printf '%s\n' caddy-id; fi ;; - *'up -d') [[ "${FAIL_UP:-0}" != 1 ]] ;; + *'up -d') + for environment in ${EXPECTED_WEB_ENVIRONMENTS:-}; do + [[ -d "$EDGE_DIR/../$environment/web" ]] || exit 1 + done + [[ "${FAIL_UP:-0}" != 1 ]] + ;; *'network inspect ogb-edge') [[ "${NETWORK_EXISTS:-1}" == 1 ]] ;; esac EOF chmod +x "$test_root/bin/docker" fail() { echo "FAIL: $*" >&2; exit 1; } +write_profile() { + local default_config='example.com { respond "ok" }' + printf '# ogb-edge-profile: %s\n%s\n' "$2" "${3:-$default_config}" > "$1" +} reset_fixture() { rm -f "$DOCKER_CALLS" "$EDGE_DIR/compose.yml" "$EDGE_DIR/Caddyfile" + rm -rf "$test_root/staging" "$test_root/production" printf 'name: ogb-edge\nservices:\n caddy:\n image: caddy:2\n' > "$EDGE_DIR/compose.yml" - printf 'old.example.com { respond "old" }\n' > "$EDGE_DIR/Caddyfile" + write_profile "$EDGE_DIR/Caddyfile" shared 'old.example.com { respond "old" }' cp "$EDGE_DIR/compose.yml" "$test_root/candidate/compose.yml" - printf 'new.example.com { respond "new" }\n' > "$test_root/candidate/Caddyfile" - unset FAIL_VALIDATE FAIL_RELOAD FAIL_UP + write_profile "$test_root/candidate/Caddyfile" shared 'new.example.com { respond "new" }' + unset FAIL_VALIDATE FAIL_RELOAD FAIL_UP EXPECTED_WEB_ENVIRONMENTS export CADDY_RUNNING=1 NETWORK_EXISTS=1 } -run_apply() { bash "$repo_root/scripts/apply-edge.sh" "$test_root/candidate"; } +run_apply() { bash "$repo_root/scripts/apply-edge.sh" "$test_root/candidate" "${1:-production}" "${2:-false}"; } +run_verify() { bash "$repo_root/scripts/verify-edge-profile.sh" "$EDGE_DIR" "$1"; } assert_called() { grep -Fq -- "$1" "$DOCKER_CALLS" || fail "missing Docker call: $1"; } assert_not_called() { if grep -Fq -- "$1" "$DOCKER_CALLS"; then fail "unexpected Docker call: $1"; fi; } +assert_rejected_without_mutation() { + local expected_error="$1" + shift + cp "$EDGE_DIR/compose.yml" "$test_root/expected-compose.yml" + cp "$EDGE_DIR/Caddyfile" "$test_root/expected-Caddyfile" + if "$@" > "$test_root/rejection.log" 2>&1; then fail 'unsafe edge operation was accepted'; fi + grep -Fq "$expected_error" "$test_root/rejection.log" || fail "expected error: $expected_error" + [[ ! -s "$DOCKER_CALLS" ]] || fail 'rejected profile operation called Docker' + [[ ! -e "$test_root/staging" && ! -e "$test_root/production" ]] || fail 'rejected profile operation created application directories' + cmp -s "$test_root/expected-compose.yml" "$EDGE_DIR/compose.yml" || fail 'rejected profile operation changed Compose file' + cmp -s "$test_root/expected-Caddyfile" "$EDGE_DIR/Caddyfile" || fail 'rejected profile operation changed Caddyfile' +} + +reset_fixture +assert_rejected_without_mutation 'Deployment environment is required' \ + bash "$repo_root/scripts/apply-edge.sh" "$test_root/candidate" +assert_rejected_without_mutation 'Expected deployment environment staging or production' run_apply development +assert_rejected_without_mutation 'Expected allow-profile-change to be true or false' run_apply production yes +echo 'PASS edge application requires a valid environment and explicit boolean override' + +for missing_file in compose.yml Caddyfile; do + reset_fixture + rm "$test_root/candidate/$missing_file" + assert_rejected_without_mutation 'Edge candidate requires compose.yml and Caddyfile' run_apply +done +echo 'PASS incomplete candidates do not touch Docker or application directories' + +reset_fixture +printf 'unmarked.example.com { respond "old" }\n' > "$test_root/candidate/Caddyfile" +assert_rejected_without_mutation 'Candidate Caddyfile requires an edge profile marker' run_apply +for marker in \ + '# ogb-edge-profile: invalid' \ + '# ogb-edge-profile: shared ' \ + '# ogb-edge-profile:shared' \ + $'# another comment\n# ogb-edge-profile: shared' \ + $'# ogb-edge-profile: shared\n# ogb-edge-profile: shared'; do + printf '%s\nexample.com { respond "ok" }\n' "$marker" > "$test_root/candidate/Caddyfile" + assert_rejected_without_mutation 'Invalid edge profile marker' run_apply +done +echo 'PASS missing malformed misplaced and duplicate candidate markers are rejected' + +reset_fixture +write_profile "$test_root/candidate/Caddyfile" staging +assert_rejected_without_mutation 'does not include deployment environment production' run_apply production +assert_rejected_without_mutation 'does not include deployment environment production' run_apply production true +write_profile "$test_root/candidate/Caddyfile" production +assert_rejected_without_mutation 'does not include deployment environment staging' run_apply staging +assert_rejected_without_mutation 'does not include deployment environment staging' run_apply staging true +write_profile "$test_root/candidate/Caddyfile" shared +assert_rejected_without_mutation 'Shared edge updates require the production environment' run_apply staging +assert_rejected_without_mutation 'Shared edge updates require the production environment' run_apply staging true +echo 'PASS candidate scope must match its environment and shared edge needs production approval' + +for active_profile in shared production; do + reset_fixture + write_profile "$EDGE_DIR/Caddyfile" "$active_profile" + write_profile "$test_root/candidate/Caddyfile" staging + assert_rejected_without_mutation 'Changing the installed edge profile requires' run_apply staging + run_apply staging true >/dev/null + grep -Fxq '# ogb-edge-profile: staging' "$EDGE_DIR/Caddyfile" || fail 'approved profile change did not install staging marker' + assert_called 'exec -T caddy caddy reload' +done +echo 'PASS staging can replace shared or production edge profiles only after explicit production authorization' + +for active_profile in shared staging; do + reset_fixture + write_profile "$EDGE_DIR/Caddyfile" "$active_profile" + write_profile "$test_root/candidate/Caddyfile" production + assert_rejected_without_mutation 'Changing the installed edge profile requires' run_apply production + run_apply production true >/dev/null + grep -Fxq '# ogb-edge-profile: production' "$EDGE_DIR/Caddyfile" || fail 'approved profile change did not install production marker' + assert_called 'exec -T caddy caddy reload' +done +reset_fixture +write_profile "$EDGE_DIR/Caddyfile" production +assert_rejected_without_mutation 'Changing the installed edge profile requires' run_apply production +run_apply production true >/dev/null +grep -Fxq '# ogb-edge-profile: shared' "$EDGE_DIR/Caddyfile" || fail 'approved profile change did not install shared marker' +echo 'PASS changing an installed profile requires explicit production-authorized migration' + +reset_fixture +printf 'legacy.example.com { respond "old" }\n' > "$EDGE_DIR/Caddyfile" +write_profile "$test_root/candidate/Caddyfile" staging +assert_rejected_without_mutation 'Unmarked legacy edge can only adopt the shared profile through production' run_apply staging true +write_profile "$test_root/candidate/Caddyfile" production +assert_rejected_without_mutation 'Unmarked legacy edge can only adopt the shared profile through production' run_apply production true +write_profile "$test_root/candidate/Caddyfile" shared +run_apply production >/dev/null +grep -Fxq '# ogb-edge-profile: shared' "$EDGE_DIR/Caddyfile" || fail 'legacy shared edge was not marked' +echo 'PASS unmarked legacy edge can only adopt shared profile through production' + +for marker in \ + '# ogb-edge-profile: invalid' \ + $'# another comment\n# ogb-edge-profile: shared' \ + $'# ogb-edge-profile: shared\n# ogb-edge-profile: production'; do + reset_fixture + printf '%s\nexample.com { respond "ok" }\n' "$marker" > "$EDGE_DIR/Caddyfile" + assert_rejected_without_mutation 'Invalid edge profile marker' run_apply production true +done +echo 'PASS malformed installed profiles cannot be overridden' + +for profile in staging production; do + reset_fixture + rm "$EDGE_DIR/compose.yml" "$EDGE_DIR/Caddyfile" + write_profile "$test_root/candidate/Caddyfile" "$profile" + export CADDY_RUNNING=0 NETWORK_EXISTS=0 EXPECTED_WEB_ENVIRONMENTS="$profile" + run_apply "$profile" >/dev/null + grep -Fxq "# ogb-edge-profile: $profile" "$EDGE_DIR/Caddyfile" || fail 'fresh host received incorrect profile' + [[ -d "$test_root/$profile/web" ]] || fail 'fresh edge setup did not create its web root' + if [[ "$profile" == staging ]]; then other_profile=production; else other_profile=staging; fi + [[ ! -e "$test_root/$other_profile" ]] || fail 'isolated edge setup created the other environment directory' + assert_called 'up -d' +done +echo 'PASS fresh hosts install isolated edge profiles and create only their own web roots' + +for profile in production staging; do + reset_fixture + write_profile "$test_root/candidate/Caddyfile" "$profile" + export FAIL_RELOAD=1 + if run_apply "$profile" true >/dev/null 2>&1; then fail 'failed profile migration was reported as success'; fi + assert_called 'exec -T caddy caddy reload' + grep -Fxq '# ogb-edge-profile: shared' "$EDGE_DIR/Caddyfile" || fail 'failed migration did not restore profile marker' + grep -Fq 'old.example.com' "$EDGE_DIR/Caddyfile" || fail 'failed migration did not restore previous configuration' +done +echo 'PASS failed migration restores the installed profile marker with its configuration' + +for profile in shared staging production; do + reset_fixture + write_profile "$EDGE_DIR/Caddyfile" "$profile" + run_verify "$profile" >/dev/null + for expected_profile in shared staging production; do + if [[ "$expected_profile" != "$profile" ]]; then + assert_rejected_without_mutation 'does not match expected profile' run_verify "$expected_profile" + fi + done + [[ ! -s "$DOCKER_CALLS" ]] || fail 'profile verification called Docker' +done +echo 'PASS application preflight checks exact marked profile without Docker or mutations' + +reset_fixture +printf 'legacy.example.com { respond "old" }\n' > "$EDGE_DIR/Caddyfile" +run_verify shared >/dev/null +assert_rejected_without_mutation 'Unmarked legacy edge is only valid for the shared profile' run_verify staging +assert_rejected_without_mutation 'Unmarked legacy edge is only valid for the shared profile' run_verify production +assert_rejected_without_mutation 'Expected edge profile shared staging or production' run_verify invalid +echo 'PASS preflight accepts legacy unmarked edge only for explicitly expected shared profile' + +for marker in \ + '# ogb-edge-profile: invalid' \ + $'# another comment\n# ogb-edge-profile: shared' \ + $'# ogb-edge-profile: shared\n# ogb-edge-profile: shared'; do + reset_fixture + printf '%s\nexample.com { respond "ok" }\n' "$marker" > "$EDGE_DIR/Caddyfile" + assert_rejected_without_mutation 'Invalid edge profile marker' run_verify shared +done +for missing_file in compose.yml Caddyfile; do + reset_fixture + rm "$EDGE_DIR/$missing_file" + if run_verify shared > "$test_root/rejection.log" 2>&1; then fail 'incomplete edge installation passed preflight'; fi + grep -Fq 'Edge installation requires compose.yml and Caddyfile' "$test_root/rejection.log" || fail 'preflight did not report incomplete installation' + [[ ! -s "$DOCKER_CALLS" ]] || fail 'preflight called Docker for incomplete installation' +done +echo 'PASS preflight rejects malformed markers and incomplete edge installations' reset_fixture export FAIL_VALIDATE=1 CADDY_RUNNING=0 NETWORK_EXISTS=0 @@ -44,6 +218,7 @@ grep -Fq 'old.example.com' "$EDGE_DIR/Caddyfile" || fail 'invalid candidate repl assert_not_called 'exec -T caddy caddy reload' assert_not_called 'up -d' assert_not_called 'network create ogb-edge' +[[ ! -e "$test_root/staging" && ! -e "$test_root/production" ]] || fail 'invalid candidate created application directories' echo 'PASS invalid candidate preserves active edge' reset_fixture @@ -104,8 +279,9 @@ echo 'PASS failed Compose update restores previous files' reset_fixture rm "$EDGE_DIR/compose.yml" "$EDGE_DIR/Caddyfile" -export CADDY_RUNNING=0 NETWORK_EXISTS=0 +export CADDY_RUNNING=0 NETWORK_EXISTS=0 EXPECTED_WEB_ENVIRONMENTS='staging production' run_apply >/dev/null assert_called 'network create ogb-edge' assert_called 'up -d' +[[ -d "$test_root/staging/web" && -d "$test_root/production/web" ]] || fail 'shared edge setup did not create both web roots' echo 'PASS initial edge setup creates the shared network and service' diff --git a/tests/deploy-topology/run.sh b/tests/deploy-topology/run.sh new file mode 100644 index 0000000..592bd21 --- /dev/null +++ b/tests/deploy-topology/run.sh @@ -0,0 +1,183 @@ +#!/usr/bin/env bash +set -euo pipefail + +repo_root="$(cd "$(dirname "${BASH_SOURCE[0]}")/../.." && pwd)" +test_root="$(mktemp -d)" +trap 'rm -rf -- "$test_root"' EXIT +resolver="$repo_root/scripts/resolve-edge-profile.sh" +renderer="$repo_root/scripts/render-edge.sh" + +fail() { echo "FAIL: $*" >&2; exit 1; } +assert_contains() { grep -Fq -- "$2" "$1" || fail "$1 is missing $2"; } +assert_absent() { if grep -Fq -- "$2" "$1"; then fail "$1 unexpectedly contains $2"; fi; } +assert_rejected() { if "$@" >/dev/null 2>&1; then fail "unexpectedly accepted: $*"; fi; } + +for environment in staging production; do + expected="$(printf 'profile=%s\ncheck-production=false\nenvironments=%s' "$environment" "$environment")" + [[ "$(bash "$resolver" "$environment" "$environment")" == "$expected" ]] || fail "wrong isolated outputs for $environment" + check_production=false + if [[ "$environment" == staging ]]; then check_production=true; fi + expected="$(printf 'profile=shared\ncheck-production=%s\nenvironments=production staging' "$check_production")" + [[ "$(bash "$resolver" "$environment" shared)" == "$expected" ]] || fail "wrong shared outputs for $environment" +done +assert_rejected bash "$resolver" +assert_rejected bash "$resolver" staging +assert_rejected bash "$resolver" staging "" +assert_rejected bash "$resolver" production "" +assert_rejected bash "$resolver" staging production +assert_rejected bash "$resolver" production staging +assert_rejected bash "$resolver" preview shared +assert_rejected bash "$resolver" staging Shared +assert_rejected bash "$resolver" staging shared extra +echo 'PASS edge topology is explicit and must match the deployment environment' + +# Exercise host-local health probes without making HTTP requests. Each argument +# is logged separately so redirects, TLS bypasses, and the pinned destination are +# checked independently of shell quoting. +mkdir -p "$test_root/bin" +export CURL_CALLS="$test_root/curl-calls" +cat > "$test_root/bin/curl" <<'EOF' +#!/usr/bin/env bash +printf '%s\n' "$@" >> "$CURL_CALLS" +printf '%s' "${CURL_RESPONSE:-}" +exit "${CURL_EXIT_CODE:-0}" +EOF +chmod +x "$test_root/bin/curl" +export PATH="$test_root/bin:$PATH" +health_checker="$repo_root/scripts/check-edge-health.sh" + +assert_rejected bash "$health_checker" +assert_rejected bash "$health_checker" preview https://opengamebuilder.com +assert_rejected bash "$health_checker" staging http://staging.opengamebuilder.com +assert_rejected bash "$health_checker" staging https://staging.opengamebuilder.com/path +assert_rejected bash "$health_checker" staging https://staging.opengamebuilder.com:8443 +assert_rejected bash "$health_checker" staging https://staging.opengamebuilder.com/?redirect=other +[[ ! -e "$CURL_CALLS" ]] || fail 'invalid edge health input invoked curl' +echo 'PASS edge health rejects invalid environments and non-HTTPS origins before curl' + +for environment in production staging; do + hostname=opengamebuilder.com + if [[ "$environment" == staging ]]; then hostname=staging.opengamebuilder.com; fi + export CURL_RESPONSE="ok $environment" CURL_EXIT_CODE=0 + rm -f "$CURL_CALLS" + bash "$health_checker" "$environment" "https://$hostname/" >/dev/null + mapfile -t curl_arguments < "$CURL_CALLS" + resolve_count=0 + noproxy_count=0 + for ((index = 0; index < ${#curl_arguments[@]}; index++)); do + case "${curl_arguments[$index]}" in + --noproxy) + ((noproxy_count += 1)) + [[ "${curl_arguments[$((index + 1))]:-}" == '*' ]] || fail 'health probe can use an external proxy' + ;; + --resolve) + ((resolve_count += 1)) + [[ "${curl_arguments[$((index + 1))]:-}" == "$hostname:443:127.0.0.1" ]] || fail 'health probe does not target the deployed host' + ;; + --location|--location-trusted|-L|--insecure|-k) fail 'health probe follows redirects or bypasses TLS validation' ;; + esac + done + [[ "$resolve_count" == 1 ]] || fail 'health probe must specify exactly one host-local resolution' + [[ "$noproxy_count" == 1 ]] || fail 'health probe must bypass external proxies' + [[ "${curl_arguments[-1]}" == "https://$hostname/health" ]] || fail 'health probe requested an unexpected URL' + export CURL_RESPONSE="ok $environment unexpected" + assert_rejected bash "$health_checker" "$environment" "https://$hostname" + export CURL_RESPONSE="ok $environment" CURL_EXIT_CODE=22 + assert_rejected bash "$health_checker" "$environment" "https://$hostname" +done +echo 'PASS edge health verifies host-local HTTPS and exact selected-environment responses' + +# Keep workflow wiring covered alongside the behavior of its shell helpers. +mapfile -t production_checks < <(awk ' + /- name: Check colocated production (before|after) staging update$/ { + getline; sub(/^[[:space:]]+/, ""); print + } +' "$repo_root/.github/workflows/_deploy.yml") +expected_condition="if: \${{ steps.target.outputs.check-production == 'true' }}" +[[ "${#production_checks[@]}" == 2 ]] || fail 'expected both production health guards' +for condition in "${production_checks[@]}"; do + [[ "$condition" == "$expected_condition" ]] || fail 'production health guard is not topology-dependent' +done +awk ' + /- name: Check selected edge routes on the deployed host$/ { selected = 1; next } + selected && /- name:/ { exit } + selected { print } +' "$repo_root/.github/workflows/cd-edge.yml" > "$test_root/edge-health-step" +assert_contains "$test_root/edge-health-step" 'EDGE_ENVIRONMENTS: ${{ steps.target.outputs.environments }}' +assert_contains "$test_root/edge-health-step" 'for environment in $EDGE_ENVIRONMENTS; do' +assert_contains "$test_root/edge-health-step" 'ssh deployment bash -s -- "$environment" "$url" < scripts/check-edge-health.sh' +echo 'PASS deployment workflows gate production checks and host-local probes by the selected topology' + +for job in authorize-profile-change apply; do + awk -v job="$job" ' + $0 == " " job ":" { selected = 1; next } + selected && /^ [a-zA-Z0-9_-]+:/ { exit } + selected { print } + ' "$repo_root/.github/workflows/cd-edge.yml" > "$test_root/$job-job" +done +assert_contains "$test_root/authorize-profile-change-job" 'if: ${{ inputs.allow-profile-change }}' +assert_contains "$test_root/authorize-profile-change-job" 'environment: production' +assert_contains "$test_root/apply-job" ' - require-main-dispatch' +assert_contains "$test_root/apply-job" ' - authorize-profile-change' +assert_contains "$test_root/apply-job" "if: \${{ !cancelled() && needs.require-main-dispatch.result == 'success' && (needs.authorize-profile-change.result == 'success' || (!inputs.allow-profile-change && needs.authorize-profile-change.result == 'skipped')) }}" +assert_contains "$test_root/apply-job" 'environment: ${{ inputs.environment }}' +echo 'PASS profile migrations require production approval without gating normal isolated staging on production' + +assert_rejected bash "$renderer" +assert_rejected bash "$renderer" shared +assert_rejected bash "$renderer" "" "$test_root/invalid" +assert_rejected bash "$renderer" unknown "$test_root/invalid" +assert_rejected bash "$renderer" staging "" +assert_rejected bash "$renderer" staging "$test_root/invalid" extra +assert_rejected bash "$renderer" shared "$repo_root/deploy/edge" +echo 'PASS renderer rejects missing or invalid profiles and source overwrite' + +pinned_image="$(docker compose -f "$repo_root/deploy/edge/compose.yml" config --images)" +[[ "$pinned_image" =~ ^caddy:2@sha256:[a-f0-9]{64}$ ]] || fail 'common edge image is not digest-pinned' +for profile in shared staging production; do + candidate="$test_root/$profile candidate" + bash "$renderer" "$profile" "$candidate" >/dev/null + for file in compose.yml Caddyfile; do + [[ "$(head -n 1 "$candidate/$file")" == "# ogb-edge-profile: $profile" ]] || fail "$file lacks its exact profile marker" + done + docker compose -f "$candidate/compose.yml" config --quiet + [[ "$(docker compose -f "$candidate/compose.yml" config --images)" == "$pinned_image" ]] || fail "$profile changed the pinned image" + [[ "$(docker compose -f "$candidate/compose.yml" config --services)" == caddy ]] || fail "$profile contains unexpected services" + assert_contains "$candidate/compose.yml" 'name: ogb-edge' + assert_contains "$candidate/compose.yml" 'external: true' + assert_contains "$candidate/compose.yml" 'name: ogb-edge_caddy_data' + assert_contains "$candidate/compose.yml" 'name: ogb-edge_caddy_config' + assert_contains "$candidate/compose.yml" 'source: ./Caddyfile' + assert_contains "$candidate/compose.yml" 'target: /etc/caddy/Caddyfile' + assert_absent "$candidate/compose.yml" "$repo_root" + for environment in production staging; do + if [[ "$profile" == shared || "$profile" == "$environment" ]]; then + assert_contains "$candidate/compose.yml" "source: ../$environment/web" + assert_contains "$candidate/compose.yml" "target: /srv/opengamebuilder/$environment/web" + assert_contains "$candidate/Caddyfile" "reverse_proxy api-$environment:8080" + assert_contains "$candidate/Caddyfile" "respond \"ok $environment\" 200" + assert_contains "$candidate/Caddyfile" "root * /srv/opengamebuilder/$environment/web" + else + assert_absent "$candidate/compose.yml" "$environment" + assert_absent "$candidate/Caddyfile" "$environment" + fi + done + if [[ "$profile" == staging ]]; then + assert_absent "$candidate/Caddyfile" 'www.opengamebuilder.com' + if grep -Fxq 'opengamebuilder.com {' "$candidate/Caddyfile"; then fail 'staging includes the production site'; fi + else + assert_contains "$candidate/Caddyfile" 'www.opengamebuilder.com {' + assert_contains "$candidate/Caddyfile" 'redir https://opengamebuilder.com{uri} permanent' + fi + if [[ "$profile" == production ]]; then + assert_absent "$candidate/Caddyfile" 'staging.opengamebuilder.com' + else + assert_contains "$candidate/Caddyfile" 'staging.opengamebuilder.com {' + fi + cp "$candidate/compose.yml" "$candidate/previous-compose.yml" + cp "$candidate/Caddyfile" "$candidate/previous-Caddyfile" + bash "$renderer" "$profile" "$candidate" >/dev/null + cmp -s "$candidate/previous-compose.yml" "$candidate/compose.yml" || fail "$profile Compose rendering is not deterministic" + cmp -s "$candidate/previous-Caddyfile" "$candidate/Caddyfile" || fail "$profile Caddy rendering is not deterministic" + echo "PASS $profile edge renders a portable pinned candidate with only its selected environments" +done