Enforce password byte cap and throttle invite resends #595
Workflow file for this run
This file contains hidden or bidirectional Unicode text that may be interpreted or compiled differently than what appears below. To review, open the file in an editor that reveals hidden Unicode characters.
Learn more about bidirectional Unicode characters
| name: CI | |
| on: | |
| # No branch filter on pull_request, deliberately. | |
| # | |
| # Filtering to [main, staging] meant a PR based on another PR's branch — a | |
| # stack — got no CI at all. `Static analysis (zizmor)` uses a bare | |
| # `pull_request:` trigger because it is a required status check, so it still | |
| # reported and still passed, and GitHub then read the PR as CLEAN and | |
| # mergeable. #249 sat that way: 618 lines presenting as green with a workflow | |
| # linter as its only signal and no test run behind it. | |
| # | |
| # A check that reports success without evaluating anything is worse than no | |
| # check, so this now runs on every pull request whatever its base. | |
| pull_request: | |
| # `push` keeps the filter. `main` is the known-good line and deploys | |
| # production; `staging` is the integration line and deploys staging. Railway's | |
| # deploy triggers wait for a CI check suite on the pushed commit, so both | |
| # branches must run CI here — dropping one would leave its deploys waiting on a | |
| # check that never arrives. No other branch deploys, so no other branch needs a | |
| # push-triggered suite; its PR run covers it. | |
| push: | |
| branches: [main, staging] | |
| permissions: | |
| contents: read | |
| # Read-only, and only so the format check can ask which files a PR touches. | |
| pull-requests: read | |
| concurrency: | |
| group: ${{ github.workflow }}-${{ github.ref }} | |
| cancel-in-progress: true | |
| jobs: | |
| lint-typecheck-test: | |
| name: Lint, Typecheck & Test | |
| runs-on: ubuntu-latest | |
| timeout-minutes: 15 | |
| # A real database, because the incident this workflow now guards against was | |
| # invisible without one: `prisma validate` parses the schema and `migrate | |
| # deploy` only consults the _prisma_migrations bookkeeping table, so a | |
| # migration recorded as applied but never executed passes both. Production | |
| # carried exactly that state for months and it only surfaced as nine days of | |
| # "1/1 replicas never became healthy" in August 2026. | |
| services: | |
| postgres: | |
| image: postgres:18-alpine | |
| env: | |
| POSTGRES_USER: ci | |
| POSTGRES_PASSWORD: ci | |
| POSTGRES_DB: ci | |
| ports: | |
| - 5432:5432 | |
| options: >- | |
| --health-cmd "pg_isready -U ci" | |
| --health-interval 5s | |
| --health-timeout 5s | |
| --health-retries 10 | |
| steps: | |
| - name: Checkout | |
| uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 | |
| with: | |
| persist-credentials: false | |
| - name: Setup pnpm | |
| uses: pnpm/action-setup@ea17c68df8912ef543352723c149a84f56e3d413 # v6.1.0 | |
| - name: Setup Node.js | |
| uses: actions/setup-node@820762786026740c76f36085b0efc47a31fe5020 # v7.0.0 | |
| with: | |
| node-version: 24 | |
| cache: 'pnpm' | |
| - name: Install dependencies | |
| run: pnpm install --frozen-lockfile | |
| - name: Generate Prisma client | |
| run: pnpm db:generate | |
| - name: Verify Prisma schema and migrations | |
| env: | |
| DATABASE_URL: 'postgresql://ci:ci@127.0.0.1:5432/ci' | |
| PGPASSWORD: ci | |
| run: | | |
| set -e | |
| # Invoke the CLI directly, the way apps/worker/start.sh does. `pnpm exec` | |
| # remaps ANY non-zero child status to 1 (ERR_PNPM_RECURSIVE_EXEC_FIRST_FAIL), | |
| # which destroys the 0/2/other contract these assertions depend on — the | |
| # first run of this job proved it, reporting real drift (exit 2) as a | |
| # tool failure (exit 1). | |
| PRISMA="node $(cd packages/outpost && node -p "require.resolve('prisma/build/index.js')")" | |
| echo "using $PRISMA" | |
| $PRISMA validate --schema packages/outpost/db/prisma/schema.prisma | |
| test -f packages/outpost/db/prisma/migrations/migration_lock.toml | |
| # Apply every migration to a real, empty database, then ask whether the | |
| # resulting schema actually matches schema.prisma. | |
| # | |
| # This is the check that was missing. `migrate deploy` reports success | |
| # from the _prisma_migrations table alone — it never inspects the schema | |
| # — so a migration recorded as applied but never effective is invisible | |
| # to it, and `prisma validate` only parses the file. Production ran in | |
| # exactly that state (0001_init recorded as applied, its SystemConfig | |
| # table absent) until it surfaced as nine days of undiagnosable deploy | |
| # failures in August 2026. Running the same `migrate diff` the worker's | |
| # start.sh uses closes the loop: if the migrations cannot reproduce | |
| # schema.prisma from scratch, this fails here instead of at 3am. | |
| $PRISMA migrate deploy --schema packages/outpost/db/prisma/schema.prisma | |
| $PRISMA migrate diff \ | |
| --from-schema-datasource packages/outpost/db/prisma/schema.prisma \ | |
| --to-schema-datamodel packages/outpost/db/prisma/schema.prisma \ | |
| --exit-code | |
| echo "Migrations reproduce schema.prisma exactly." | |
| # Re-running `migrate deploy` proves nothing on its own: the migration | |
| # rows already exist, so it applies zero SQL and would pass even if the | |
| # migration body were `SELECT 1/0`. That is the same bookkeeping-only | |
| # blind spot this whole job exists to close. To actually exercise | |
| # CREATE TABLE IF NOT EXISTS, the row has to be removed while the | |
| # correctly-shaped table stays. | |
| psql -h 127.0.0.1 -U ci -d ci -q -c \ | |
| "DELETE FROM _prisma_migrations WHERE migration_name = '20260812220000_create_missing_systemconfig_table';" | |
| $PRISMA migrate deploy --schema packages/outpost/db/prisma/schema.prisma | |
| echo "Repair migration re-runs cleanly against an already-correct table." | |
| # Positive control on the thing every assertion below rests on: that | |
| # this invocation style returns the child's exit code unchanged. | |
| set +e | |
| sh -c 'exit 2'; propagated=$? | |
| set -e | |
| if [ "$propagated" -ne 2 ]; then | |
| echo "FATAL: exit codes are not propagating (got $propagated for a deliberate 2)." | |
| exit 1 | |
| fi | |
| - name: Verify the SystemConfig repair path | |
| env: | |
| DATABASE_URL: 'postgresql://ci:ci@127.0.0.1:5432/ci' | |
| PGPASSWORD: ci | |
| run: | | |
| set -e | |
| PRISMA="node $(cd packages/outpost && node -p "require.resolve('prisma/build/index.js')")" | |
| # Reconstruct the broken production state and prove the repair fixes it. | |
| # The step above only covers a fresh database, which was never the broken | |
| # case — production had 0001_init RECORDED AS APPLIED while the table it | |
| # declares was absent, and that is the one scenario this migration exists | |
| # for. Without this, nothing tests it. | |
| psql -h 127.0.0.1 -U ci -d ci -c 'DROP TABLE IF EXISTS "SystemConfig";' | |
| psql -h 127.0.0.1 -U ci -d ci -c \ | |
| "DELETE FROM _prisma_migrations WHERE migration_name = '20260812220000_create_missing_systemconfig_table';" | |
| # The database now looks exactly like production did on 2026-08-07: | |
| # migrations recorded, table missing. The guard must see it... | |
| # Assert exit 2 SPECIFICALLY. `migrate diff --exit-code` returns 0 for | |
| # identical, 2 for a difference, and 1 (or other) when the CLI itself | |
| # failed — so `if ! ...` would accept a broken CLI or a lost connection | |
| # as proof the guard works, having never compared anything. start.sh | |
| # treats that distinction as load-bearing; the test for it must too. | |
| set +e | |
| $PRISMA migrate diff \ | |
| --from-schema-datasource packages/outpost/db/prisma/schema.prisma \ | |
| --to-schema-datamodel packages/outpost/db/prisma/schema.prisma \ | |
| --exit-code | |
| drift_status=$? | |
| set -e | |
| if [ "$drift_status" -eq 0 ]; then | |
| echo "FATAL: drift guard did not detect a missing SystemConfig table." | |
| echo "That is the exact blind spot this PR exists to close." | |
| exit 1 | |
| elif [ "$drift_status" -ne 2 ]; then | |
| echo "FATAL: migrate diff exited $drift_status — it never compared the database." | |
| exit 1 | |
| fi | |
| echo "Drift guard detects the missing table (exit 2)." | |
| # ...and the repair migration must fix it. | |
| $PRISMA migrate deploy --schema packages/outpost/db/prisma/schema.prisma | |
| $PRISMA migrate diff \ | |
| --from-schema-datasource packages/outpost/db/prisma/schema.prisma \ | |
| --to-schema-datamodel packages/outpost/db/prisma/schema.prisma \ | |
| --exit-code | |
| echo "Repair migration restores SystemConfig." | |
| # A table present but mis-shaped must FAIL the migration loudly rather | |
| # than no-op and be recorded as applied — the "recorded but not | |
| # effective" state that caused the incident. | |
| psql -h 127.0.0.1 -U ci -d ci -c 'DROP TABLE "SystemConfig";' | |
| # Wrong TYPE, not just a missing column name — the assertion must check | |
| # shape, since a table with the right three names but wrong types is | |
| # still a database that does not match schema.prisma. | |
| psql -h 127.0.0.1 -U ci -d ci -c 'CREATE TABLE "SystemConfig" ("key" TEXT PRIMARY KEY, "value" INTEGER NOT NULL, "updatedAt" TIMESTAMP(3) NOT NULL);' | |
| psql -h 127.0.0.1 -U ci -d ci -c \ | |
| "DELETE FROM _prisma_migrations WHERE migration_name = '20260812220000_create_missing_systemconfig_table';" | |
| set +e | |
| deploy_out=$($PRISMA migrate deploy --schema packages/outpost/db/prisma/schema.prisma 2>&1) | |
| deploy_status=$? | |
| set -e | |
| echo "$deploy_out" | |
| if [ "$deploy_status" -eq 0 ]; then | |
| echo "FATAL: migration reported success against a mis-shaped SystemConfig." | |
| exit 1 | |
| fi | |
| # Any non-zero would also cover "database unreachable", which would pass | |
| # this test without the assertion ever firing. Require our own message. | |
| # Match a stable fragment of the migration's own RAISE, not the whole | |
| # sentence: the previous run failed here because the assertion text was | |
| # reworded and this pattern was not, so a correctly-firing guard read as | |
| # "failed for the wrong reason". | |
| case "$deploy_out" in | |
| *"Repairing it needs an ALTER"*) ;; | |
| *) | |
| echo "FATAL: deploy failed, but not via the migration's own guard." | |
| exit 1 | |
| ;; | |
| esac | |
| echo "Migration fails loudly on a mis-shaped SystemConfig." | |
| # Leave the database usable. These assertions deliberately corrupt it | |
| # (mis-shaped table plus a failed P3009 row), and they run before Build | |
| # and Test — so the first DB-touching test added to this job would | |
| # otherwise inherit a knowingly-broken schema. | |
| psql -h 127.0.0.1 -U ci -d ci -q -c 'DROP SCHEMA public CASCADE; CREATE SCHEMA public;' | |
| $PRISMA migrate deploy --schema packages/outpost/db/prisma/schema.prisma | |
| echo "CI database restored." | |
| - name: Verify the worker startup guard | |
| run: | | |
| set -e | |
| # Execute start.sh itself against a stub CLI. Nothing previously ran | |
| # this script -- CI re-implemented its prisma calls in bash -- which is | |
| # why a busybox-vs-GNU timeout exit-code difference sat in it unnoticed. | |
| # Every branch that decides whether the worker may boot is asserted here. | |
| stub=$(mktemp); chmod +x "$stub" | |
| printf '#!/bin/sh\nif [ "$1" = migrate ] && [ "$2" = diff ]; then exit "${DIFF_RC:-0}"; fi\nexit "${DEPLOY_RC:-0}"\n' > "$stub" | |
| script=$(mktemp) | |
| sed 's|^exec node apps/worker/dist/index.js|echo STARTED|' apps/worker/start.sh > "$script" | |
| check() { # label expect_exit expect_started env... | |
| label=$1 want_rc=$2 want_started=$3; shift 3 | |
| # set +e around the call: half these cases are SUPPOSED to exit | |
| # non-zero, and this step runs under `bash -e`, so the assignment | |
| # would abort the whole step on the first expected failure. | |
| set +e | |
| out=$(env OUTPOST_PRISMA_CMD="$stub" "$@" sh "$script" 2>&1) | |
| rc=$? | |
| set -e | |
| case "$out" in *STARTED*) got=yes ;; *) got=no ;; esac | |
| if [ "$rc" != "$want_rc" ] || [ "$got" != "$want_started" ]; then | |
| echo "FATAL: $label -> exit=$rc started=$got (wanted exit=$want_rc started=$want_started)" | |
| echo "$out" | |
| exit 1 | |
| fi | |
| echo " ok: $label" | |
| } | |
| check "clean schema boots" 0 yes DIFF_RC=0 | |
| check "drift blocks the boot" 1 no DIFF_RC=2 | |
| check "drift + override boots" 0 yes DIFF_RC=2 OUTPOST_ALLOW_SCHEMA_DRIFT=1 | |
| check "drift + override=on boots" 0 yes DIFF_RC=2 OUTPOST_ALLOW_SCHEMA_DRIFT=on | |
| check "drift + unknown override blocks" 1 no DIFF_RC=2 OUTPOST_ALLOW_SCHEMA_DRIFT=banana | |
| check "drift + override=0 blocks" 1 no DIFF_RC=2 OUTPOST_ALLOW_SCHEMA_DRIFT=0 | |
| # A check that could not RUN must be overridable too, or a CLI-level | |
| # breakage is an unbreakable crash loop under restartPolicyType=ALWAYS. | |
| check "tool failure blocks" 1 no DIFF_RC=1 | |
| check "tool failure + override boots" 0 yes DIFF_RC=1 OUTPOST_ALLOW_SCHEMA_DRIFT=1 | |
| check "migrate deploy failure blocks" 1 no DEPLOY_RC=1 | |
| # A bad timeout value must not fail the step closed. | |
| check "non-numeric timeout still boots" 0 yes DIFF_RC=0 OUTPOST_STARTUP_STEP_TIMEOUT=abc | |
| check "timeout=0 means unbounded" 0 yes DIFF_RC=0 OUTPOST_STARTUP_STEP_TIMEOUT=0 | |
| # Diagnostics must reach stderr, and an overridden boot must not page. | |
| set +e | |
| env OUTPOST_PRISMA_CMD="$stub" DIFF_RC=2 sh "$script" 2>/dev/null | grep -q FATAL | |
| leaked=$? | |
| env OUTPOST_PRISMA_CMD="$stub" DIFF_RC=1 OUTPOST_ALLOW_SCHEMA_DRIFT=1 sh "$script" 2>&1 >/dev/null | grep -q FATAL | |
| paged=$? | |
| set -e | |
| if [ "$leaked" -eq 0 ]; then | |
| echo "FATAL: diagnostics leaked to stdout." | |
| exit 1 | |
| fi | |
| if [ "$paged" -eq 0 ]; then | |
| echo "FATAL: an overridden boot emitted FATAL and would page." | |
| exit 1 | |
| fi | |
| echo "Startup guard behaves correctly across all branches." | |
| - name: Verify the pinned Prisma CLI matches the lockfile | |
| run: | | |
| set -e | |
| # apps/worker/start.sh gates the worker's boot on this CLI's `migrate | |
| # diff` exit code, so a CLI that has drifted from @prisma/client can | |
| # report false drift and block every deploy with a message pointing at | |
| # the database. Keep the Dockerfile pin and the lockfile in lockstep. | |
| set -o pipefail | |
| resolved=$(node "$(cd packages/outpost && node -p "require.resolve('prisma/build/index.js')")" --version | sed -n 's/^prisma *: *\([0-9][^ ]*\).*/\1/p' | head -1) | |
| if [ -z "$resolved" ]; then | |
| echo "FATAL: could not parse a version from 'prisma --version'; its output format may have changed." | |
| exit 1 | |
| fi | |
| echo "lockfile resolves prisma=$resolved" | |
| # BOTH images, not just the worker's. web and worker apply/verify | |
| # migrations against the same database, so a bump to either alone | |
| # recreates the CLI skew this check exists to prevent. Anchored on ^RUN | |
| # so a comment quoting the command cannot shadow the real line. | |
| for dockerfile in apps/worker/Dockerfile apps/web/Dockerfile; do | |
| pinned=$(sed -n 's/^RUN .*--no-save prisma@\([0-9][^ ]*\).*/\1/p' "$dockerfile" | head -1) | |
| if [ -z "$pinned" ]; then | |
| echo "FATAL: could not read a Prisma pin out of $dockerfile." | |
| exit 1 | |
| fi | |
| if [ "$resolved" != "$pinned" ]; then | |
| echo "FATAL: $dockerfile pins prisma@$pinned but the lockfile resolves $resolved." | |
| echo "Update the Dockerfile pins and the lockfile together." | |
| exit 1 | |
| fi | |
| echo " $dockerfile pins $pinned — matches." | |
| done | |
| - name: Build | |
| run: pnpm build | |
| - name: Lint | |
| run: pnpm lint | |
| # Formatting is checked on the files this change touches, not repo-wide. | |
| # | |
| # 311 files predate any enforcement, and reformatting them in one commit | |
| # would collide with every open PR while burying the change that | |
| # actually mattered. Checking the diff enforces the convention from here | |
| # on and lets the backlog be cleared on its own schedule. | |
| # | |
| # Push builds have no base to diff against, so this only runs on pull | |
| # requests. `templates/` is excluded via .prettierignore — those markdown | |
| # files are rendered into customer emails, so reflowing them is a content | |
| # change wearing a whitespace diff. | |
| # The file list comes from the API, not from git. `actions/checkout` runs | |
| # with `persist-credentials: false` and a shallow clone, so the base | |
| # commit is absent locally and `git fetch` has no credentials to go get | |
| # it — it fails with "could not read Username for 'https://github.com'". | |
| # Asking the API avoids both problems and needs no extra checkout depth. | |
| # | |
| # Everything reaches the script through `env`. Interpolating `${{ }}` | |
| # directly into `run` is the template-injection shape zizmor flags. | |
| - name: Format check (changed files) | |
| if: github.event_name == 'pull_request' | |
| env: | |
| GH_TOKEN: ${{ github.token }} | |
| REPO: ${{ github.repository }} | |
| PR_NUMBER: ${{ github.event.pull_request.number }} | |
| run: | | |
| # `|| true` on grep: no formattable files in the diff is a pass, | |
| # and grep exits 1 when it matches nothing, which `set -e` would | |
| # otherwise turn into a failed build. | |
| files=$(gh api --paginate "repos/$REPO/pulls/$PR_NUMBER/files" \ | |
| --jq '.[] | select(.status != "removed") | .filename' \ | |
| | grep -E '\.(ts|tsx|js|jsx|json|md)$' || true) | |
| if [ -z "$files" ]; then | |
| echo "No formattable files changed." | |
| exit 0 | |
| fi | |
| echo "Checking:" | |
| echo "$files" | sed 's/^/ /' | |
| # NUL-delimited so a path containing a space stays one argument. | |
| # Word-splitting would hand prettier a nonexistent path, and | |
| # --no-error-on-unmatched-pattern would then skip the real file in | |
| # silence. `tr | xargs -0` rather than `xargs -d` because -d is | |
| # GNU-only, and a line that cannot be run locally is how the first | |
| # version of this step shipped broken. | |
| # | |
| # That flag covers genuinely absent paths only — prettier 3 does | |
| # not error on a named file that .prettierignore covers, it just | |
| # reports nothing for it. | |
| echo "$files" | tr '\n' '\0' | xargs -0 npx prettier --no-error-on-unmatched-pattern --check | |
| - name: Typecheck | |
| run: pnpm typecheck | |
| - name: Test | |
| run: pnpm test | |
| # Railway auto-deploys on push to main via GitHub integration. | |
| # No explicit deploy step needed — Railway watches the repo directly. | |
| # | |
| # IMPORTANT: Configure branch protection on `main` to require the | |
| # "Lint, Typecheck & Test" check to pass before merging. This ensures | |
| # Railway only auto-deploys code that has passed CI. | |
| # | |
| # To switch to CLI-driven deploys instead, uncomment the job below | |
| # and set the RAILWAY_TOKEN repository secret. | |
| # | |
| # deploy: | |
| # name: Deploy to Railway | |
| # runs-on: ubuntu-latest | |
| # needs: lint-typecheck-test | |
| # if: github.ref == 'refs/heads/main' && github.event_name == 'push' | |
| # | |
| # steps: | |
| # - name: Checkout | |
| # uses: actions/checkout@df4cb1c069e1874edd31b4311f1884172cec0e10 # v6.0.3 | |
| # | |
| # - name: Install Railway CLI | |
| # run: npm i -g @railway/cli | |
| # | |
| # - name: Deploy to Railway | |
| # env: | |
| # RAILWAY_TOKEN: ${{ secrets.RAILWAY_TOKEN }} | |
| # run: railway up --detach |