Skip to content

Commit 4d8e4bd

Browse files
Merge branch 'main' into claude/roadmap-thread-reopen-w80epb
2 parents 2ce65ed + ff3dacc commit 4d8e4bd

229 files changed

Lines changed: 13194 additions & 1602 deletions

File tree

Some content is hidden

Large Commits have some content hidden by default. Use the searchbox below for content that may be hidden.

‎.github/actions/setup/action.yml‎

Lines changed: 1 addition & 1 deletion
Original file line numberDiff line numberDiff line change
@@ -72,7 +72,7 @@ runs:
7272
# node_modules cache hits next time. So this layer is pure waste on the hit path
7373
# and redundant on the miss path.
7474
- uses: actions/setup-node@v6
75-
with: { node-version: '22' }
75+
with: { node-version-file: '.nvmrc' }
7676
# ripgrep backs the app's code-search services at runtime; the unit suite
7777
# and the e2e app both exercise them. GitHub-hosted runners have passwordless
7878
# sudo + apt, so install it there. Self-hosted runners typically have neither

‎.github/workflows/ci.yml‎

Lines changed: 81 additions & 35 deletions
Original file line numberDiff line numberDiff line change
@@ -201,24 +201,15 @@ jobs:
201201
run: |
202202
git fetch --no-tags --depth=1 origin "$BASE_SHA" 2>/dev/null || true
203203
node scripts/check-screenshots.mts --plan --base "$BASE_SHA"
204-
# e2e shard fan-out is runner-dependent. The GitHub-hosted nightly keeps 8
205-
# shards to stay under the 2-core/7GB OOM ceiling (see the e2e job comment).
206-
# PR/push runs on the self-hosted Linux Docker `copse-e2e` fleet (see #422),
207-
# where 8 shards just means paying the (mostly-cached but non-trivial, and
208-
# occasionally cold) dep restore + Electron setup up to 8× across the fleet.
209-
# Fewer, larger shards there cut that repeated setup without duplicating any
210-
# spec: wdio `--shard current/total` and the `subset` round-robin both
211-
# partition the specs disjointly, so a lower total just makes each slice
212-
# bigger. Tune the self-hosted count below to your fleet size.
204+
# Keep every full e2e run at 8 shards. Both runner classes have a tight
205+
# memory ceiling: GitHub-hosted is 7 GB and each self-hosted `copse-e2e`
206+
# container is capped at 6 GB. Packing the suite into 3 shards makes each
207+
# container launch too many sequential Electron sessions; accumulated
208+
# Electron/gortex processes then OOM-crash unrelated late-running specs.
209+
# `--shard` and the subset round-robin still partition specs disjointly.
213210
- id: shards
214-
env:
215-
EVENT: ${{ github.event_name }}
216211
run: |
217-
if [ "$EVENT" = "schedule" ]; then
218-
{ echo "total=8"; echo "list=[1,2,3,4,5,6,7,8]"; } >> "$GITHUB_OUTPUT"
219-
else
220-
{ echo "total=3"; echo "list=[1,2,3]"; } >> "$GITHUB_OUTPUT"
221-
fi
212+
{ echo "total=8"; echo "list=[1,2,3,4,5,6,7,8]"; } >> "$GITHUB_OUTPUT"
222213
223214
# Auto-fix the PR branch: run every autofix we have — ESLint `--fix` then
224215
# Prettier `--write` — and commit the result back onto the head branch. This
@@ -455,19 +446,12 @@ jobs:
455446
# slice of specs via wdio's `--shard current/total`. The shared `dist` artifact
456447
# from the `build` job is unpacked here rather than rebuilt per shard.
457448
#
458-
# Shard count is runner-dependent (computed by the precheck `shards` step):
459-
# - hosted nightly: 8. Each shard runs its slice as one sequential wdio
460-
# process, relaunching Electron per spec. On the 2-core/7GB runner, once a
461-
# shard runs too many specs the accumulated electron/gortex processes
462-
# exhaust it and a later spec's Electron fails to boot or OOM-crashes the
463-
# runner ("shutdown signal"). #339's data showed 8 stays under that limit
464-
# even in the densest spec range.
465-
# - self-hosted PR/push: 3. These run on the Linux Docker `copse-e2e`
466-
# runners (see #422), which have the headroom 8 was protecting against, so
467-
# the only thing 8 buys there is paying dep restore + Electron setup up to
468-
# 8× across the fleet. Fewer, larger shards cut that repeated (and
469-
# sometimes cold) setup — the dominant wall-clock cost — without
470-
# duplicating any spec. Tune to the `copse-e2e` runner count.
449+
# Every full run uses 8 shards. Each shard runs its slice as one sequential
450+
# wdio process, relaunching Electron per spec. Both the hosted runner (7 GB)
451+
# and self-hosted container (6 GB cap) eventually exhaust memory when a shard
452+
# launches too many Electron/gortex processes; a later unrelated spec then
453+
# fails to boot or OOM-crashes ("shutdown signal"). #339's data showed that
454+
# 8 shards stays under that limit even in the densest spec range.
471455
# The per-attempt `timeout` is 480s (step cap 26 min for 3 attempts) so attempt
472456
# 1 finishes instead of being SIGKILL'd mid-retry, letting attempt 2 relaunch.
473457
strategy:
@@ -506,9 +490,11 @@ jobs:
506490
# The Electron/Chromedriver e2e session occasionally fails to start on the
507491
# runner ("DevToolsActivePort file doesn't exist" / "POST /session" timeout).
508492
# The hang wedges the whole wdio run, so in-process retries don't recover —
509-
# only a fresh xvfb+Electron launch does. Retry the entire shard, and wrap
510-
# each attempt in `timeout` so a wedged session is killed and retried
511-
# (a plain hang would otherwise never exit and stall the job).
493+
# only a fresh xvfb+Electron launch does. Retry the entire shard, cleaning
494+
# orphaned session processes before each attempt and when the step exits.
495+
# The runners are one job per PID-isolated container, so this cannot touch
496+
# another job. Wrap each attempt in `timeout` so a plain hang cannot stall
497+
# the job indefinitely.
512498
# Oracle gate (see the precheck job's plan step): `full` shards the whole
513499
# suite; `subset` distributes only the planned specs round-robin across the
514500
# same 8 shards (so no shard exceeds the per-shard memory budget the count
@@ -550,7 +536,25 @@ jobs:
550536
# fall back to no inner timeout (the step `timeout-minutes` still caps
551537
# a wedged run) so the command can't die with "timeout: not found".
552538
TIMEOUT="$(command -v timeout || command -v gtimeout || true)"
539+
540+
cleanup_e2e_processes() {
541+
# Bracketed patterns avoid matching this shell's own command line.
542+
# A failed deleteSession can orphan any of these; self-hosted runner
543+
# containers are reused, so clean before attempt 1 as well as retries.
544+
pkill -TERM -f '[e]lectron/dist/electron' 2>/dev/null || true
545+
pkill -TERM -f '[e]lectron-chromedriver/bin/chromedriver' 2>/dev/null || true
546+
pkill -TERM -f '[X]vfb' 2>/dev/null || true
547+
pkill -TERM -f '[g]ortex' 2>/dev/null || true
548+
sleep 1
549+
pkill -KILL -f '[e]lectron/dist/electron' 2>/dev/null || true
550+
pkill -KILL -f '[e]lectron-chromedriver/bin/chromedriver' 2>/dev/null || true
551+
pkill -KILL -f '[X]vfb' 2>/dev/null || true
552+
pkill -KILL -f '[g]ortex' 2>/dev/null || true
553+
}
554+
555+
trap cleanup_e2e_processes EXIT
553556
for attempt in 1 2 3; do
557+
cleanup_e2e_processes
554558
echo "::group::e2e shard ${{ matrix.shard }} attempt $attempt"
555559
if ${TIMEOUT:+$TIMEOUT -k 15 480} npm run test:e2e:ci -- $SHARD_ARGS $SPEC_ARGS; then
556560
echo "::endgroup::"
@@ -965,15 +969,33 @@ jobs:
965969
# have their own explicit safe-tier result below: they may pass only after
966970
# `precheck` actually ran on a GitHub-hosted runner, never merely because all
967971
# code-executing jobs were skipped.
972+
#
973+
# Concurrency supersession: top-level `cancel-in-progress` cancels in-flight
974+
# jobs when a newer push/sync arrives on the same ref. Those cancelled jobs
975+
# must not paint the superseded SHA's `CI Passed` red — tip CI is the gate
976+
# that matters. We detect supersession by comparing this run's head SHA to
977+
# the live tip; genuine cancels of the tip (manual cancel, runner death with
978+
# no tip move) still fail. Sibling-matrix cancels after a real shard failure
979+
# still fail because `ANY_FAILURE` is true first.
968980
ci-passed:
969981
name: CI Passed
970982
if: always()
971983
needs: [precheck, check, bench, build, e2e]
972984
# Tiny aggregate gate — keep on hosted so a saturated self-hosted check
973985
# fleet cannot leave a green pipeline without its required status check.
974986
runs-on: ubuntu-latest
987+
permissions:
988+
contents: read
989+
pull-requests: read
975990
steps:
976991
- name: Check upstream job results
992+
env:
993+
GH_TOKEN: ${{ github.token }}
994+
REPOSITORY: ${{ github.repository }}
995+
EVENT_NAME: ${{ github.event_name }}
996+
REF_NAME: ${{ github.ref_name }}
997+
RUN_HEAD_SHA: ${{ github.event.pull_request.head.sha || github.sha }}
998+
PR_NUMBER: ${{ github.event.pull_request.number || '' }}
977999
run: |
9781000
FORK_PR=${{ github.event_name == 'pull_request' && github.event.pull_request.head.repo.full_name != github.repository }}
9791001
if $FORK_PR; then
@@ -987,7 +1009,7 @@ jobs:
9871009
fi
9881010
MODE="${{ needs.precheck.outputs.mode }}"
9891011
ANY_FAILURE=${{ contains(needs.*.result, 'failure') }}
990-
ANY_BAD=${{ contains(needs.*.result, 'failure') || contains(needs.*.result, 'cancelled') }}
1012+
ANY_CANCELLED=${{ contains(needs.*.result, 'cancelled') }}
9911013
CORE_OK=${{ needs.precheck.result == 'success' }}
9921014
# Skip-mode runs (screenshots-only pushes, e.g. the e2e screenshot
9931015
# bot's own commits) intentionally do not execute check/build/e2e;
@@ -1003,8 +1025,32 @@ jobs:
10031025
echo "All CI jobs passed (heavy jobs intentionally skipped: mode=skip)."
10041026
exit 0
10051027
fi
1006-
if $ANY_BAD; then
1007-
echo "One or more CI jobs failed or were cancelled:"
1028+
if $ANY_FAILURE; then
1029+
echo "One or more CI jobs failed:"
1030+
echo '${{ toJSON(needs) }}'
1031+
exit 1
1032+
fi
1033+
if $ANY_CANCELLED; then
1034+
# Fail closed if we cannot resolve tip — a tip cancel must stay red.
1035+
if [ -n "$PR_NUMBER" ]; then
1036+
TIP_SHA=$(gh api "repos/${REPOSITORY}/pulls/${PR_NUMBER}" --jq .head.sha) || {
1037+
echo "Could not resolve PR tip SHA; treating cancel as failure."
1038+
echo '${{ toJSON(needs) }}'
1039+
exit 1
1040+
}
1041+
else
1042+
TIP_SHA=$(gh api "repos/${REPOSITORY}/commits/${REF_NAME}" --jq .sha) || {
1043+
echo "Could not resolve branch tip SHA; treating cancel as failure."
1044+
echo '${{ toJSON(needs) }}'
1045+
exit 1
1046+
}
1047+
fi
1048+
if [ -n "$TIP_SHA" ] && [ "$TIP_SHA" != "$RUN_HEAD_SHA" ]; then
1049+
echo "Run superseded by newer tip ${TIP_SHA} (this SHA=${RUN_HEAD_SHA}); cancelled jobs are concurrency noise."
1050+
echo '${{ toJSON(needs) }}'
1051+
exit 0
1052+
fi
1053+
echo "One or more CI jobs were cancelled while still at tip (${RUN_HEAD_SHA}):"
10081054
echo '${{ toJSON(needs) }}'
10091055
exit 1
10101056
fi

‎.github/workflows/release-mac.yml‎

Lines changed: 17 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -31,7 +31,24 @@ concurrency:
3131
cancel-in-progress: false
3232

3333
jobs:
34+
# Keep a deliberately uncached install in the release path. The shared setup
35+
# action restores a complete node_modules tree when possible, which is useful
36+
# for packaging but can hide an undeclared/transitive production dependency.
37+
# This gate exercises the lockfile from scratch with install scripts disabled,
38+
# then proves the production bundle is self-contained before packaging starts.
39+
verify-clean-release-build:
40+
runs-on: ubuntu-latest
41+
permissions:
42+
contents: read
43+
steps:
44+
- uses: actions/checkout@v7.0.0
45+
- uses: actions/setup-node@v6
46+
with: { node-version-file: '.nvmrc' }
47+
- name: Verify clean release build
48+
run: npm ci --ignore-scripts && npm run build:release
49+
3450
build-mac:
51+
needs: verify-clean-release-build
3552
# Apple-Silicon GitHub-hosted runner; electron-builder downloads the x64
3653
# Electron to cross-build the Intel artifacts in the same run.
3754
runs-on: macos-14

‎AGENTS.md‎

Lines changed: 6 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -195,6 +195,12 @@ decisions) and `permission-gate.test.ts` (gate wiring + MCP decisions).
195195
- `npm test` runs Node's test runner over `src/**/*.test.ts` (esbuild-bundled into `dist-test/`).
196196
- `npm run test:e2e` is WebdriverIO + `@wdio/electron-service` and needs a display. It passes under
197197
`npm run test:e2e` on this headless VM (WDIO auto-starts Xvfb on Linux).
198+
- **Don't block the machine on e2e while iterating.** When a remote e2e host is available
199+
(`.tmp/remote-e2e/host.json` exists, or one can be provisioned), prefer
200+
`npm run e2e:remote -- run --detach` — it snapshots the working tree (uncommitted changes
201+
included), runs the oracle-selected specs on a cloud container, and leaves results in
202+
`.tmp/remote-e2e/runs/<run-id>/`; finish with `e2e:remote -- wait <run-id>` and keep working in
203+
between. See [`ci-runners/README.md`](ci-runners/README.md#remote-e2e-dev-hosts-npm-run-e2eremote).
198204

199205
For _which tier a test belongs in_ — favour unit/component tests, reserve e2e for broad validation
200206
and real-runtime checks (sizing/rendering, Monaco, terminal, webview, main IPC) — read

‎CHANGELOG.md‎

Lines changed: 32 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,32 @@
1+
# Changelog and release notes
2+
3+
The canonical changelog is the set of
4+
[GitHub Releases](https://github.com/copse-dev/agent-pane/releases). Published
5+
release notes are owned and maintained with the GitHub Release; this file records
6+
the release-note process and the current unreleased summary rather than copying
7+
every published entry.
8+
9+
## Unreleased
10+
11+
- Added general-availability security, support, privacy/data-flow, release, and
12+
recovery documentation.
13+
14+
## Release-note process
15+
16+
For every release:
17+
18+
1. Draft the GitHub Release from the matching `v<version>` tag.
19+
2. Turn merged changes into user-facing notes, grouped into features, fixes,
20+
security/privacy changes, and developer changes as applicable.
21+
3. State the supported OS and architectures, known issues, data migrations, and
22+
recovery implications. Copse supports forward fixes only; do not recommend a
23+
downgrade.
24+
4. Link issues or pull requests that provide important detail without exposing
25+
confidential security-report information.
26+
5. Review the notes with the artifacts, then publish them as part of the GitHub
27+
Release.
28+
6. Reset the `Unreleased` section here after publication. Do not mirror the
29+
published notes into a second historical list in this file.
30+
31+
The complete shipping procedure is in
32+
[docs/release-checklist.md](docs/release-checklist.md).

‎Makefile‎

Lines changed: 45 additions & 4 deletions
Original file line numberDiff line numberDiff line change
@@ -67,9 +67,30 @@ BUILD_STAMP := $(STAMP_DIR)/build.stamp
6767

6868
# Source that, when changed, should trigger a rebuild of `dist/`. Evaluated when
6969
# the Makefile is parsed; new files are picked up on the next `make` invocation.
70-
BUILD_SRC := $(shell find src packages scripts -type f 2>/dev/null) \
70+
# `assets` is included because scripts/build.mts copies it into dist/ wholesale.
71+
BUILD_SRC_DIRS := src packages scripts assets
72+
BUILD_SRC := $(shell find $(BUILD_SRC_DIRS) -type f 2>/dev/null) \
7173
package.json tsconfig.json tsconfig.node.json tsconfig.web.json
7274

75+
# find-based prerequisites only notice files that still exist. A branch switch
76+
# that merely deletes or renames a module leaves every surviving file older
77+
# than the stamp, so make declares dist/ "up to date" while the bundle still
78+
# contains the deleted code. Fingerprint the file *list* into a stamp that is
79+
# refreshed (at parse time) whenever the list changes; ordinary mtime logic
80+
# then forces the rebuild.
81+
SRC_LIST_STAMP := $(STAMP_DIR)/src-list.stamp
82+
SRC_LIST_SUM := $(shell find $(BUILD_SRC_DIRS) -type f 2>/dev/null | sort | cksum | cut -d' ' -f1)
83+
ifneq ($(SRC_LIST_SUM),$(shell cat $(SRC_LIST_STAMP) 2>/dev/null))
84+
$(shell mkdir -p $(STAMP_DIR) && echo '$(SRC_LIST_SUM)' > $(SRC_LIST_STAMP))
85+
endif
86+
87+
# The build stamp only proves a build *ran* — not that dist/ still holds its
88+
# output. `npm run dev` writes watch-mode bundles (a partial set, dev flags)
89+
# into the same dist/, and a hand-deleted dist/ leaves the stamp behind. If the
90+
# main bundle is missing or newer than the stamp, dist/ was touched outside
91+
# make: drop the stamp so the next build re-syncs it.
92+
$(shell if [ -f $(BUILD_STAMP) ] && { [ ! -f dist/main/index.js ] || [ dist/main/index.js -nt $(BUILD_STAMP) ]; }; then rm -f $(BUILD_STAMP); fi)
93+
7394
# ----------------------------------------------------------------------------
7495
# Help
7596
# ----------------------------------------------------------------------------
@@ -248,12 +269,32 @@ check-node:
248269
# terminal fails to launch. The flag is scoped to this one invocation, so it
249270
# doesn't touch your global config. See the "Hardened npm profiles" section of
250271
# the README.
272+
#
273+
# On macOS, `npm ci` after a lockfile change (e.g. switching branches) can die
274+
# with ENOTEMPTY/EBUSY/EPERM while pruning stale packages — Spotlight, Finder,
275+
# or an editor briefly holds a file inside a directory npm is rmdir-ing. The
276+
# tree it leaves behind is half-pruned, so every rerun fails the same way until
277+
# node_modules is removed by hand. Detect that failure class from npm's output,
278+
# wipe node_modules, and retry once; any other failure (or a second one) still
279+
# aborts loudly.
280+
NPM_CI := npm ci --ignore-scripts=false
281+
251282
.PHONY: deps
252283
deps: $(DEPS_STAMP)
253284

254285
$(DEPS_STAMP): package-lock.json package.json | $(STAMP_DIR) check-node
255286
@echo "==> Dependencies out of date — running 'npm ci' (scripts forced on)…"
256-
@$(USE_NVM); npm ci --ignore-scripts=false
287+
@$(USE_NVM); \
288+
log="$(STAMP_DIR)/npm-ci.log"; \
289+
if ! $(NPM_CI) 2>&1 | tee "$$log"; then \
290+
if grep -qE 'code (ENOTEMPTY|EBUSY|EPERM)' "$$log"; then \
291+
echo "==> npm ci hit a filesystem race pruning node_modules — wiping it and retrying…"; \
292+
rm -rf node_modules; \
293+
$(NPM_CI); \
294+
else \
295+
exit 1; \
296+
fi; \
297+
fi
257298
touch $(DEPS_STAMP)
258299

259300
# --- build ------------------------------------------------------------------
@@ -262,7 +303,7 @@ $(DEPS_STAMP): package-lock.json package.json | $(STAMP_DIR) check-node
262303
.PHONY: build
263304
build: $(BUILD_STAMP)
264305

265-
$(BUILD_STAMP): $(DEPS_STAMP) $(BUILD_SRC) | $(STAMP_DIR)
306+
$(BUILD_STAMP): $(DEPS_STAMP) $(BUILD_SRC) $(SRC_LIST_STAMP) | $(STAMP_DIR)
266307
@echo "==> Source changed — clearing dist/ and rebuilding…"
267308
rm -rf dist
268309
@$(USE_NVM); npm run build
@@ -279,4 +320,4 @@ run: build
279320
.PHONY: clean
280321
clean:
281322
@echo "==> Removing dist/ and build/deps stamps…"
282-
rm -rf dist $(DEPS_STAMP) $(BUILD_STAMP)
323+
rm -rf dist $(DEPS_STAMP) $(BUILD_STAMP) $(SRC_LIST_STAMP)

0 commit comments

Comments
 (0)