@@ -201,24 +201,15 @@ jobs:
201201 run : |
202202 git fetch --no-tags --depth=1 origin "$BASE_SHA" 2>/dev/null || true
203203 node scripts/check-screenshots.mts --plan --base "$BASE_SHA"
204- # e2e shard fan-out is runner-dependent. The GitHub-hosted nightly keeps 8
205- # shards to stay under the 2-core/7GB OOM ceiling (see the e2e job comment).
206- # PR/push runs on the self-hosted Linux Docker `copse-e2e` fleet (see #422),
207- # where 8 shards just means paying the (mostly-cached but non-trivial, and
208- # occasionally cold) dep restore + Electron setup up to 8× across the fleet.
209- # Fewer, larger shards there cut that repeated setup without duplicating any
210- # spec: wdio `--shard current/total` and the `subset` round-robin both
211- # partition the specs disjointly, so a lower total just makes each slice
212- # bigger. Tune the self-hosted count below to your fleet size.
204+ # Keep every full e2e run at 8 shards. Both runner classes have a tight
205+ # memory ceiling: GitHub-hosted is 7 GB and each self-hosted `copse-e2e`
206+ # container is capped at 6 GB. Packing the suite into 3 shards makes each
207+ # container launch too many sequential Electron sessions; accumulated
208+ # Electron/gortex processes then OOM-crash unrelated late-running specs.
209+ # `--shard` and the subset round-robin still partition specs disjointly.
213210 - id : shards
214- env :
215- EVENT : ${{ github.event_name }}
216211 run : |
217- if [ "$EVENT" = "schedule" ]; then
218- { echo "total=8"; echo "list=[1,2,3,4,5,6,7,8]"; } >> "$GITHUB_OUTPUT"
219- else
220- { echo "total=3"; echo "list=[1,2,3]"; } >> "$GITHUB_OUTPUT"
221- fi
212+ { echo "total=8"; echo "list=[1,2,3,4,5,6,7,8]"; } >> "$GITHUB_OUTPUT"
222213
223214 # Auto-fix the PR branch: run every autofix we have — ESLint `--fix` then
224215 # Prettier `--write` — and commit the result back onto the head branch. This
@@ -455,19 +446,12 @@ jobs:
455446 # slice of specs via wdio's `--shard current/total`. The shared `dist` artifact
456447 # from the `build` job is unpacked here rather than rebuilt per shard.
457448 #
458- # Shard count is runner-dependent (computed by the precheck `shards` step):
459- # - hosted nightly: 8. Each shard runs its slice as one sequential wdio
460- # process, relaunching Electron per spec. On the 2-core/7GB runner, once a
461- # shard runs too many specs the accumulated electron/gortex processes
462- # exhaust it and a later spec's Electron fails to boot or OOM-crashes the
463- # runner ("shutdown signal"). #339's data showed 8 stays under that limit
464- # even in the densest spec range.
465- # - self-hosted PR/push: 3. These run on the Linux Docker `copse-e2e`
466- # runners (see #422), which have the headroom 8 was protecting against, so
467- # the only thing 8 buys there is paying dep restore + Electron setup up to
468- # 8× across the fleet. Fewer, larger shards cut that repeated (and
469- # sometimes cold) setup — the dominant wall-clock cost — without
470- # duplicating any spec. Tune to the `copse-e2e` runner count.
449+ # Every full run uses 8 shards. Each shard runs its slice as one sequential
450+ # wdio process, relaunching Electron per spec. Both the hosted runner (7 GB)
451+ # and self-hosted container (6 GB cap) eventually exhaust memory when a shard
452+ # launches too many Electron/gortex processes; a later unrelated spec then
453+ # fails to boot or OOM-crashes ("shutdown signal"). #339's data showed that
454+ # 8 shards stays under that limit even in the densest spec range.
471455 # The per-attempt `timeout` is 480s (step cap 26 min for 3 attempts) so attempt
472456 # 1 finishes instead of being SIGKILL'd mid-retry, letting attempt 2 relaunch.
473457 strategy :
@@ -506,9 +490,11 @@ jobs:
506490 # The Electron/Chromedriver e2e session occasionally fails to start on the
507491 # runner ("DevToolsActivePort file doesn't exist" / "POST /session" timeout).
508492 # The hang wedges the whole wdio run, so in-process retries don't recover —
509- # only a fresh xvfb+Electron launch does. Retry the entire shard, and wrap
510- # each attempt in `timeout` so a wedged session is killed and retried
511- # (a plain hang would otherwise never exit and stall the job).
493+ # only a fresh xvfb+Electron launch does. Retry the entire shard, cleaning
494+ # orphaned session processes before each attempt and when the step exits.
495+ # The runners are one job per PID-isolated container, so this cannot touch
496+ # another job. Wrap each attempt in `timeout` so a plain hang cannot stall
497+ # the job indefinitely.
512498 # Oracle gate (see the precheck job's plan step): `full` shards the whole
513499 # suite; `subset` distributes only the planned specs round-robin across the
514500 # same 8 shards (so no shard exceeds the per-shard memory budget the count
@@ -550,7 +536,25 @@ jobs:
550536 # fall back to no inner timeout (the step `timeout-minutes` still caps
551537 # a wedged run) so the command can't die with "timeout: not found".
552538 TIMEOUT="$(command -v timeout || command -v gtimeout || true)"
539+
540+ cleanup_e2e_processes() {
541+ # Bracketed patterns avoid matching this shell's own command line.
542+ # A failed deleteSession can orphan any of these; self-hosted runner
543+ # containers are reused, so clean before attempt 1 as well as retries.
544+ pkill -TERM -f '[e]lectron/dist/electron' 2>/dev/null || true
545+ pkill -TERM -f '[e]lectron-chromedriver/bin/chromedriver' 2>/dev/null || true
546+ pkill -TERM -f '[X]vfb' 2>/dev/null || true
547+ pkill -TERM -f '[g]ortex' 2>/dev/null || true
548+ sleep 1
549+ pkill -KILL -f '[e]lectron/dist/electron' 2>/dev/null || true
550+ pkill -KILL -f '[e]lectron-chromedriver/bin/chromedriver' 2>/dev/null || true
551+ pkill -KILL -f '[X]vfb' 2>/dev/null || true
552+ pkill -KILL -f '[g]ortex' 2>/dev/null || true
553+ }
554+
555+ trap cleanup_e2e_processes EXIT
553556 for attempt in 1 2 3; do
557+ cleanup_e2e_processes
554558 echo "::group::e2e shard ${{ matrix.shard }} attempt $attempt"
555559 if ${TIMEOUT:+$TIMEOUT -k 15 480} npm run test:e2e:ci -- $SHARD_ARGS $SPEC_ARGS; then
556560 echo "::endgroup::"
@@ -965,15 +969,33 @@ jobs:
965969 # have their own explicit safe-tier result below: they may pass only after
966970 # `precheck` actually ran on a GitHub-hosted runner, never merely because all
967971 # code-executing jobs were skipped.
972+ #
973+ # Concurrency supersession: top-level `cancel-in-progress` cancels in-flight
974+ # jobs when a newer push/sync arrives on the same ref. Those cancelled jobs
975+ # must not paint the superseded SHA's `CI Passed` red — tip CI is the gate
976+ # that matters. We detect supersession by comparing this run's head SHA to
977+ # the live tip; genuine cancels of the tip (manual cancel, runner death with
978+ # no tip move) still fail. Sibling-matrix cancels after a real shard failure
979+ # still fail because `ANY_FAILURE` is true first.
968980 ci-passed :
969981 name : CI Passed
970982 if : always()
971983 needs : [precheck, check, bench, build, e2e]
972984 # Tiny aggregate gate — keep on hosted so a saturated self-hosted check
973985 # fleet cannot leave a green pipeline without its required status check.
974986 runs-on : ubuntu-latest
987+ permissions :
988+ contents : read
989+ pull-requests : read
975990 steps :
976991 - name : Check upstream job results
992+ env :
993+ GH_TOKEN : ${{ github.token }}
994+ REPOSITORY : ${{ github.repository }}
995+ EVENT_NAME : ${{ github.event_name }}
996+ REF_NAME : ${{ github.ref_name }}
997+ RUN_HEAD_SHA : ${{ github.event.pull_request.head.sha || github.sha }}
998+ PR_NUMBER : ${{ github.event.pull_request.number || '' }}
977999 run : |
9781000 FORK_PR=${{ github.event_name == 'pull_request' && github.event.pull_request.head.repo.full_name != github.repository }}
9791001 if $FORK_PR; then
@@ -987,7 +1009,7 @@ jobs:
9871009 fi
9881010 MODE="${{ needs.precheck.outputs.mode }}"
9891011 ANY_FAILURE=${{ contains(needs.*.result, 'failure') }}
990- ANY_BAD =${{ contains(needs.*.result, 'failure') || contains(needs.*.result, 'cancelled') }}
1012+ ANY_CANCELLED =${{ contains(needs.*.result, 'cancelled') }}
9911013 CORE_OK=${{ needs.precheck.result == 'success' }}
9921014 # Skip-mode runs (screenshots-only pushes, e.g. the e2e screenshot
9931015 # bot's own commits) intentionally do not execute check/build/e2e;
@@ -1003,8 +1025,32 @@ jobs:
10031025 echo "All CI jobs passed (heavy jobs intentionally skipped: mode=skip)."
10041026 exit 0
10051027 fi
1006- if $ANY_BAD; then
1007- echo "One or more CI jobs failed or were cancelled:"
1028+ if $ANY_FAILURE; then
1029+ echo "One or more CI jobs failed:"
1030+ echo '${{ toJSON(needs) }}'
1031+ exit 1
1032+ fi
1033+ if $ANY_CANCELLED; then
1034+ # Fail closed if we cannot resolve tip — a tip cancel must stay red.
1035+ if [ -n "$PR_NUMBER" ]; then
1036+ TIP_SHA=$(gh api "repos/${REPOSITORY}/pulls/${PR_NUMBER}" --jq .head.sha) || {
1037+ echo "Could not resolve PR tip SHA; treating cancel as failure."
1038+ echo '${{ toJSON(needs) }}'
1039+ exit 1
1040+ }
1041+ else
1042+ TIP_SHA=$(gh api "repos/${REPOSITORY}/commits/${REF_NAME}" --jq .sha) || {
1043+ echo "Could not resolve branch tip SHA; treating cancel as failure."
1044+ echo '${{ toJSON(needs) }}'
1045+ exit 1
1046+ }
1047+ fi
1048+ if [ -n "$TIP_SHA" ] && [ "$TIP_SHA" != "$RUN_HEAD_SHA" ]; then
1049+ echo "Run superseded by newer tip ${TIP_SHA} (this SHA=${RUN_HEAD_SHA}); cancelled jobs are concurrency noise."
1050+ echo '${{ toJSON(needs) }}'
1051+ exit 0
1052+ fi
1053+ echo "One or more CI jobs were cancelled while still at tip (${RUN_HEAD_SHA}):"
10081054 echo '${{ toJSON(needs) }}'
10091055 exit 1
10101056 fi
0 commit comments