Repository navigation
Expand file tree
/
Copy pathcommon.sh
More file actions
473 lines (434 loc) · 20.8 KB
/
Copy pathcommon.sh
File metadata and controls
473 lines (434 loc) · 20.8 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
342
343
344
345
346
347
348
349
350
351
352
353
354
355
356
357
358
359
360
361
362
363
364
365
366
367
368
369
370
371
372
373
374
375
376
377
378
379
380
381
382
383
384
385
386
387
388
389
390
391
392
393
394
395
396
397
398
399
400
401
402
403
404
405
406
407
408
409
410
411
412
413
414
415
416
417
418
419
420
421
422
423
424
425
426
427
428
429
430
431
432
433
434
435
436
437
438
439
440
441
442
443
444
445
446
447
448
449
450
451
452
453
454
455
456
457
458
459
460
461
462
463
464
465
466
467
468
469
470
471
472
473
# shellcheck shell=bash
# common.sh — constants, configuration, and helpers shared by every part.
#
# Sourced by bin/runpool. Targets bash 3.2, which is what stock macOS ships,
# so nothing here may use associative arrays, mapfile, or ${var^^}.
# ---------------------------------------------------------------------------
# Configuration
# ---------------------------------------------------------------------------
# Everything installation-specific lives in a config file outside the repo, so
# a checkout carries no personal values. Every setting has a working default.
RUNPOOL_CONFIG="${RUNPOOL_CONFIG:-${XDG_CONFIG_HOME:-${HOME}/.config}/runpool/config}"
# Precedence is environment, then config file, then the defaults below.
#
# The config file uses plain assignments, so sourcing it would otherwise
# clobber an explicit override and there would be no way to point a single
# invocation at a scratch directory. That matters for anything testing this
# on a machine that already has a config, the Homebrew test block included.
_rp_env_BASE="${RUNPOOL_BASE:-}"
_rp_env_POOLS_FILE="${RUNPOOL_POOLS_FILE:-}"
_rp_env_LOG_DIR="${RUNPOOL_LOG_DIR:-}"
_rp_env_LABEL_NS="${RUNPOOL_LABEL_NS:-}"
_rp_env_IDLE_SECS="${RUNPOOL_IDLE_SECS:-}"
_rp_env_LOAD_WARN="${RUNPOOL_LOAD_WARN:-}"
_rp_env_NOTIFY_CMD="${RUNPOOL_NOTIFY_CMD:-}"
_rp_env_JOB_HOOK="${RUNPOOL_JOB_HOOK:-}"
_rp_env_TELEMETRY="${RUNPOOL_TELEMETRY:-}"
# 'set -a' exports everything the config assigns, which matters because a
# notifier or job hook runs as a child process and cannot see a variable that
# was only set. The config may also define settings runpool knows nothing about
# by name, such as an endpoint and token for whatever RUNPOOL_NOTIFY_CMD points
# at, so they cannot be exported individually.
set -a
# shellcheck disable=SC1090
[ -f "${RUNPOOL_CONFIG}" ] && . "${RUNPOOL_CONFIG}"
set +a
[ -n "${_rp_env_BASE}" ] && RUNPOOL_BASE="${_rp_env_BASE}"
[ -n "${_rp_env_POOLS_FILE}" ] && RUNPOOL_POOLS_FILE="${_rp_env_POOLS_FILE}"
[ -n "${_rp_env_LOG_DIR}" ] && RUNPOOL_LOG_DIR="${_rp_env_LOG_DIR}"
[ -n "${_rp_env_LABEL_NS}" ] && RUNPOOL_LABEL_NS="${_rp_env_LABEL_NS}"
[ -n "${_rp_env_IDLE_SECS}" ] && RUNPOOL_IDLE_SECS="${_rp_env_IDLE_SECS}"
[ -n "${_rp_env_LOAD_WARN}" ] && RUNPOOL_LOAD_WARN="${_rp_env_LOAD_WARN}"
[ -n "${_rp_env_NOTIFY_CMD}" ] && RUNPOOL_NOTIFY_CMD="${_rp_env_NOTIFY_CMD}"
[ -n "${_rp_env_JOB_HOOK}" ] && RUNPOOL_JOB_HOOK="${_rp_env_JOB_HOOK}"
[ -n "${_rp_env_TELEMETRY}" ] && RUNPOOL_TELEMETRY="${_rp_env_TELEMETRY}"
unset _rp_env_BASE _rp_env_POOLS_FILE _rp_env_LOG_DIR _rp_env_LABEL_NS \
_rp_env_IDLE_SECS _rp_env_LOAD_WARN _rp_env_NOTIFY_CMD _rp_env_JOB_HOOK \
_rp_env_TELEMETRY
# Restored values need exporting again: the restore above is a plain assignment
# and happens after 'set -a' was turned off.
export RUNPOOL_BASE RUNPOOL_LOG_DIR RUNPOOL_LABEL_NS RUNPOOL_IDLE_SECS \
RUNPOOL_LOAD_WARN RUNPOOL_NOTIFY_CMD RUNPOOL_JOB_HOOK RUNPOOL_TELEMETRY \
RUNPOOL_CONFIG RUNPOOL_POOLS_FILE
# Where runners, pool definitions, launch agents and state live. Never the
# repository: it holds registration credentials.
RUNPOOL_BASE="${RUNPOOL_BASE:-${XDG_DATA_HOME:-${HOME}/.local/share}/runpool}"
RUNPOOL_POOL_DIR="${RUNPOOL_BASE}/pools"
RUNPOOL_AGENT_DIR="${RUNPOOL_BASE}/agents"
RUNPOOL_STATE_DIR="${RUNPOOL_BASE}/state"
RUNPOOL_LOG_DIR="${RUNPOOL_LOG_DIR:-${HOME}/Library/Logs/runpool}"
RUNPOOL_LOG="${RUNPOOL_LOG_DIR}/runpool.log"
RUNPOOL_ACTIVITY="${RUNPOOL_STATE_DIR}/activity"
RUNPOOL_PAUSE_FLAG="${RUNPOOL_STATE_DIR}/paused"
# shellcheck disable=SC2034 # read by lib/scheduler.sh
RUNPOOL_LAST_CLEAN="${RUNPOOL_STATE_DIR}/last-clean"
# The pools `runpool apply` reconciles to. Beside the config rather than under
# RUNPOOL_BASE, because it is written by a person and copied between machines,
# while everything under the base is runtime state this tool owns. Derived from
# XDG_CONFIG_HOME directly and not from RUNPOOL_CONFIG, so pointing the config
# at /dev/null to isolate a test does not also lose the pools file.
# shellcheck disable=SC2034 # read by lib/apply.sh
RUNPOOL_POOLS_FILE="${RUNPOOL_POOLS_FILE:-${XDG_CONFIG_HOME:-${HOME}/.config}/runpool/pools}"
# Prefix for every launch-agent label this tool owns. Configurable so two
# installations on one machine cannot collide.
RUNPOOL_LABEL_NS="${RUNPOOL_LABEL_NS:-runpool}"
# Stand a pool down after this many seconds with no job running anywhere.
# Restarting a runner is cheap and needs no re-registration, so a short grace
# only avoids churn between rapid pushes inside one working session.
RUNPOOL_IDLE_SECS="${RUNPOOL_IDLE_SECS:-1200}"
# Warn above this 1-minute load average while jobs are running. The contention
# incidents this exists to catch ran at 163 and 184 on a 14-core machine.
#
# Six times core count is a starting point, not a safe default at any size.
# A pool of N runners each forking a worker per core produces roughly N times
# core count when it is simply busy, so once a pool passes six runners the
# default fires on entirely healthy work. Raise it to suit the pool.
RUNPOOL_LOAD_WARN="${RUNPOOL_LOAD_WARN:-$(( $(sysctl -n hw.ncpu 2>/dev/null || echo 8) * 6 ))}"
# Optional command receiving one JSON object on stdin whenever something is
# worth reporting. Unset by default: runpool works fully without a notifier and
# never grows one of its own. See lib/notify.sh and contrib/notify-webhook.sh.
RUNPOOL_NOTIFY_CMD="${RUNPOOL_NOTIFY_CMD:-}"
# Record one line per job to <base>/telemetry/jobs.jsonl, so the right runner
# count can be measured rather than argued about. Off by default. Timings and
# machine state only, nothing about the code, and it never leaves the machine.
RUNPOOL_TELEMETRY="${RUNPOOL_TELEMETRY:-0}"
mkdir -p "${RUNPOOL_POOL_DIR}" "${RUNPOOL_AGENT_DIR}" \
"${RUNPOOL_STATE_DIR}" "${RUNPOOL_LOG_DIR}" 2>/dev/null || true
# ---------------------------------------------------------------------------
# Output
# ---------------------------------------------------------------------------
_rp_log() {
echo "[$(date -u +%Y-%m-%dT%H:%M:%SZ)] $*" | tee -a "${RUNPOOL_LOG}" >&2
}
_rp_err() { echo "runpool: $*" >&2; }
_rp_require() {
command -v "$1" >/dev/null 2>&1 || { _rp_err "missing dependency: $1"; return 1; }
}
_rp_now() { date +%s; }
# Absolute path to the installed executable, for launch agents to invoke. A
# launchd job gets none of the interactive shell, so the path must be resolved
# at install time rather than looked up on PATH later.
_rp_self_path() { echo "${RUNPOOL_SELF:-$0}"; }
# ---------------------------------------------------------------------------
# Pools
# ---------------------------------------------------------------------------
_rp_pool_conf() { echo "${RUNPOOL_POOL_DIR}/$1.conf"; }
# A pool name becomes four different things: a config file path, a runner
# directory, a launchd label, and a bare string in the status JSON. It is
# constrained here to what is safe in all four, which is also what lets
# _rp_status_json assemble JSON without escaping anything.
#
# A 'case' glob rather than a bash regex, because stock bash 3.2 treats a
# quoted and an unquoted right-hand side of =~ differently and the difference
# is easy to get wrong. '.' and '..' pass a character-class test and are still
# path hazards, so they are rejected by name.
_rp_valid_pool_name() {
case "$1" in
''|.|..) return 1 ;;
*[!A-Za-z0-9._-]*) return 1 ;;
esac
return 0
}
# A GitHub owner or repository name, by the same reasoning and the same rule.
# These matter because `apply` is the first thing to write POOL_TARGET and
# POOL_WATCH from a file rather than from a command line, and _rp_status_json
# prints both without escaping on the stated grounds that they are GitHub
# identifiers. This is what keeps that stated assumption true.
_rp_valid_gh_name() {
case "$1" in
''|.|..) return 1 ;;
*[!A-Za-z0-9._-]*) return 1 ;;
esac
return 0
}
# OWNER/REPO: exactly one slash, with a valid name either side. The rejecting
# patterns come first so that '/x', 'x/' and 'a/b/c' never reach the split.
_rp_valid_gh_repo() {
case "$1" in
*/*/*|/*|*/) return 1 ;;
*/*) _rp_valid_gh_name "${1%%/*}" && _rp_valid_gh_name "${1#*/}" ;;
*) return 1 ;;
esac
}
# A runner count. Every caller used to test only '' and non-digits, which let
# through two values that then failed somewhere else and blamed the wrong
# thing:
#
# '007' passes a digits-only test, is written to POOL_COUNT verbatim, and
# comes back out of _rp_status_json as "count":007 — which Python and Node
# both reject, taking any wrapper reading that JSON down with it.
#
# A twenty-digit count also passes, and then `[ "${count}" -ge 1 ]` prints
# its own 'integer expression expected' with an internal path in it before
# the caller reports some unrelated reason.
#
# The upper bound is what keeps `[ -ge ]` off a value libc cannot parse. Four
# digits is far past anything a single Mac can host, so the bound costs nothing
# real and the failure it prevents is a raw shell error.
_rp_valid_count() {
case "$1" in
''|*[!0-9]*) return 1 ;; # empty or not all digits
0*) return 1 ;; # leading zero, and plain '0' with it
esac
[ "${#1}" -le 4 ] || return 1
return 0
}
# The one sentence every caller prints when _rp_valid_count says no. Kept here
# so the pools file and the command line cannot drift into describing the same
# rule differently.
_rp_count_rule() { echo "a runner count is a whole number from 1 to 9999, written without a leading zero"; }
# Load POOL_* for pool $1 into the caller's scope. POOL_WATCH is set only on
# org pools, so every reader still uses "${POOL_WATCH:-}".
#
# EVERY POOL_* is cleared first, not just POOL_WATCH. Clearing one of them was
# enough while nothing loaded more than one pool per process; `apply` reconciles
# a whole file in one, and a config missing a field then inherited the previous
# pool's value and got planned against it. A pool silently taking on its
# neighbour's count, directory or labels is worse than any error.
#
# The four fields below are dereferenced without a default all over this tool —
# POOL_COUNT in arithmetic, POOL_DIR as a path prefix — so a config that
# survived sourcing but defines none of them is refused here rather than
# somewhere further on.
# shellcheck disable=SC2034 # every POOL_* here is read by another lib/ fragment
_rp_load_pool() {
local conf; conf="$(_rp_pool_conf "$1")"
if [ ! -f "${conf}" ]; then
_rp_err "unknown pool: $1 (see 'runpool pools')"; return 1
fi
POOL_NAME=""; POOL_SCOPE=""; POOL_TARGET=""; POOL_COUNT=""
POOL_DIR=""; POOL_LABELS=""; POOL_WATCH=""
# shellcheck disable=SC1090
. "${conf}" || { _rp_err "pool '$1': could not read ${conf}"; return 1; }
if [ -z "${POOL_SCOPE}" ] || [ -z "${POOL_TARGET}" ] || \
[ -z "${POOL_COUNT}" ] || [ -z "${POOL_DIR}" ]; then
_rp_err "pool '$1': ${conf} is incomplete (needs POOL_SCOPE, POOL_TARGET, POOL_COUNT and POOL_DIR)"
return 1
fi
}
# find, not a glob, so an empty pool directory stays silent rather than
# expanding to a literal '*.conf' or aborting under zsh's nomatch.
_rp_pool_names() {
find "${RUNPOOL_POOL_DIR}" -maxdepth 1 -name '*.conf' 2>/dev/null \
| while read -r f; do basename "${f}" .conf; done
}
_rp_label() { echo "${RUNPOOL_LABEL_NS}.$1.$2"; }
# In-flight jobs for the pool rooted at $1. Matches "<dir>/runner-" so a pool
# whose directory contains another pool's cannot count its neighbour's work.
_rp_busy_in() {
ps -Ao command= 2>/dev/null | grep -F "$1/runner-" | grep -c "Runner.Worker" | tr -d ' '
}
_rp_agent_loaded() { launchctl list "$1" >/dev/null 2>&1; }
# '>|' rather than '>': a shell with noclobber set refuses to truncate an
# existing file, which silently stopped this timestamp updating and left the
# idle sweep reading a frozen clock.
_rp_touch_activity() { _rp_now >| "${RUNPOOL_ACTIVITY}"; }
_rp_paused() { [ -f "${RUNPOOL_PAUSE_FLAG}" ]; }
# GitHub API prefix for a pool's scope: orgs/<org> or repos/<owner>/<repo>.
_rp_scope_path() {
if [ "$1" = "org" ]; then echo "/orgs/$2"; else echo "/repos/$2"; fi
}
# Whether organisation $1's DEFAULT runner group lets public repositories use
# its runners. Echoes 'true', 'false' or 'unknown'.
#
# 'unknown' is a third answer and not a synonym for 'false': reading runner
# groups needs admin:org, and a caller that folded the two together would
# report a permission problem as an all-clear.
#
# GitHub owns this control at organisation scope. The setting defaults to false
# and runners land in the default group because config.sh is never passed
# --runnergroup, so RunPool reads it and reports it and does nothing else. In
# particular it does NOT enumerate the organisation's public repositories to
# re-derive the same answer; SECURITY.md and AGENTS.md both state why the
# repository and organisation cases are deliberately asymmetric.
#
# Shared rather than inline because it now has two callers that must agree:
# `register` reads it once when a pool is created, and `doctor` reads it on
# every run — the setting can be switched on long after the pool exists.
_rp_org_allows_public() {
local pub
pub=$(gh api "/orgs/$1/actions/runner-groups" \
--jq '[.runner_groups[] | select(.default == true) | .allows_public_repositories][0]' 2>/dev/null)
case "${pub}" in
true) echo true ;;
false) echo false ;;
*) echo unknown ;;
esac
}
# ---------------------------------------------------------------------------
# Runner binary
# ---------------------------------------------------------------------------
# Fetch the latest osx-arm64 runner tarball once and echo its local path.
_rp_fetch_runner_tarball() {
local out url digest path tmp jqf attempt sum
# '[.]' matches a literal dot without a backslash, which keeps this filter
# safe to carry through shells that mangle escapes. The digest comes back as
# "sha256:..." and is empty on a release that does not publish one.
jqf='[.assets[] | select(.name | test("osx-arm64.*[.]tar[.]gz$"))
| "\(.browser_download_url) \(.digest // "")"][0] // ""'
# releases/latest intermittently returns empty under secondary rate limiting,
# so retry with backoff. Once cached this is skipped entirely.
for attempt in 1 2 3 4 5; do
out=$(gh api repos/actions/runner/releases/latest --jq "${jqf}" 2>/dev/null)
[ -n "${out}" ] && break
sleep $(( attempt * 2 ))
done
url="${out%% *}"; digest="${out#* }"
[ -n "${url}" ] || { _rp_err "could not resolve the osx-arm64 runner tarball after retries"; return 1; }
path="${RUNPOOL_BASE}/.cache/${url##*/}"
mkdir -p "${RUNPOOL_BASE}/.cache" 2>/dev/null
if [ ! -f "${path}" ]; then
_rp_log "downloading runner: ${url##*/}"
# '-f' so an HTTP error is a failure. Without it curl writes the error body
# to the output path and exits 0, and the "already cached" test above then
# trusts that file forever: every later tar fails and nothing says why.
#
# Downloaded under a temporary name in the same directory and moved into
# place only once it is complete and verified, so an interrupted fetch
# cannot leave a partial file behind either.
tmp="${path}.part.$$"
curl -fsSL "${url}" -o "${tmp}" || {
rm -f "${tmp}"; _rp_err "download failed: ${url}"; return 1; }
# The release publishes a sha256 and shasum is stock on macOS, so verifying
# costs one field in the filter above and no new dependency. A release
# without a digest is skipped rather than refused.
if [ -n "${digest}" ]; then
sum="$(shasum -a 256 "${tmp}" 2>/dev/null | awk '{print $1}')"
if [ "${sum}" != "${digest#sha256:}" ]; then
rm -f "${tmp}"
_rp_err "checksum mismatch on ${url##*/}: expected ${digest#sha256:}, got ${sum:-none}"
return 1
fi
fi
mv -f "${tmp}" "${path}" || { rm -f "${tmp}"; return 1; }
fi
echo "${path}"
}
# Deregister one runner from GitHub using the agent id it recorded itself.
#
# Two earlier approaches failed, and both failed quietly. 'config.sh remove'
# takes no --unattended flag, so the runner binary rejected the arguments and
# exited having done nothing. Matching by a derived name misses any pool whose
# runners were registered under different names. The agent id is written by the
# runner and is exact. The file carries a UTF-8 BOM, hence grep rather than jq.
# $1 runner_dir $2 scope $3 target
_rp_deregister_runner() {
local runner_dir="$1" scope="$2" target="$3" id url
[ -f "${runner_dir}/.runner" ] || return 0 # never registered
id=$(grep -o '"agentId"[[:space:]]*:[[:space:]]*[0-9]*' "${runner_dir}/.runner" 2>/dev/null \
| grep -o '[0-9]*$')
[ -n "${id}" ] || { _rp_err "no agentId in ${runner_dir}/.runner — cannot deregister"; return 1; }
url="$(_rp_scope_path "${scope}" "${target}")/actions/runners/${id}"
gh api -X DELETE "${url}" >/dev/null 2>&1 || {
_rp_err "DEREGISTER FAILED: runner id ${id} is still registered on ${target}. Remove it with: gh api -X DELETE ${url}"
return 1
}
return 0
}
# ---------------------------------------------------------------------------
# Launch agents
# ---------------------------------------------------------------------------
# The runner invokes a job hook with no arguments and gives no indication of
# which phase it is, so a single script cannot tell "started" from "completed".
# Generate a one-line wrapper per phase that passes it in. Written into the
# runtime directory rather than the repo, because the path to the user's hook
# is installation-specific.
_rp_write_hook_wrappers() {
[ -n "${RUNPOOL_JOB_HOOK:-}" ] || return 0
local dir="${RUNPOOL_BASE}/hooks" phase
mkdir -p "${dir}" 2>/dev/null
for phase in started completed; do
cat >| "${dir}/${phase}.sh" <<WRAP
#!/bin/sh
exec "${RUNPOOL_JOB_HOOK}" ${phase}
WRAP
chmod +x "${dir}/${phase}.sh" 2>/dev/null
done
}
# Write the on-demand launch agent for one runner. $1 label, $2 runner_dir.
#
# Agents are written outside ~/Library/LaunchAgents on purpose, so that macOS
# never starts a runner at login. Pools come up because something asked, or
# because the tick saw queued work.
_rp_write_plist() {
local label="$1" dir="$2" plist="${RUNPOOL_AGENT_DIR}/$1.plist" hook=""
[ -n "${RUNPOOL_JOB_HOOK:-}" ] && { _rp_write_hook_wrappers; hook="${RUNPOOL_BASE}/hooks"; }
{
cat <<PLIST
<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE plist PUBLIC "-//Apple//DTD PLIST 1.0//EN" "http://www.apple.com/DTDs/PropertyList-1.0.dtd">
<plist version="1.0">
<dict>
<key>Label</key>
<string>${label}</string>
<key>ProgramArguments</key>
<array>
<string>${dir}/run.sh</string>
</array>
<key>WorkingDirectory</key>
<string>${dir}</string>
<key>EnvironmentVariables</key>
<dict>
<key>PATH</key>
<string>/opt/homebrew/bin:/usr/local/bin:/usr/bin:/bin:/usr/sbin:/sbin</string>
<key>npm_config_store_dir</key>
<string>${dir}/.pnpm-store</string>
<key>npm_config_cache</key>
<string>${dir}/.npm-cache</string>
<key>TMPDIR</key>
<string>${dir}/tmp</string>
PLIST
if [ -n "${hook}" ]; then
cat <<PLIST
<key>ACTIONS_RUNNER_HOOK_JOB_STARTED</key>
<string>${hook}/started.sh</string>
<key>ACTIONS_RUNNER_HOOK_JOB_COMPLETED</key>
<string>${hook}/completed.sh</string>
PLIST
fi
cat <<PLIST
</dict>
<key>RunAtLoad</key>
<true/>
<key>KeepAlive</key>
<true/>
<key>ThrottleInterval</key>
<integer>10</integer>
<key>StandardOutPath</key>
<string>${RUNPOOL_LOG_DIR}/${label}.log</string>
<key>StandardErrorPath</key>
<string>${RUNPOOL_LOG_DIR}/${label}.log</string>
</dict>
</plist>
PLIST
} >| "${plist}"
}
# Per-runner package caches and temp are load-bearing, not tidiness. All
# runners share one HOME, so without npm_config_store_dir and npm_config_cache
# concurrent installs collide and pnpm fails with a reflink error. TMPDIR is
# redirected because the shared /var/folders temp is used by the OS and every
# running app, so a job that leaks there cannot be swept safely; one test suite
# once left 633k directories and 178GB behind. Pointing each runner at its own
# temp makes that leak collectable, which is what 'runpool clean' collects.
#
# TMPDIR is deliberately NOT dot-led, unlike the caches beside it. Library code
# roots things at os.tmpdir() without knowing where that points, and a dotted
# component in the absolute path silently disables anything applying
# dotfile-ignore rules to it. A file watcher rooted there ignored its whole tree
# and failed every run of one repository's suite for eight days, load
# independent, before anyone traced it to the path. The caches keep their dots
# because only npm and pnpm read them, and neither walks its own cache.
# Regenerate every pool's agents from its config. A running pool picks the
# change up on its next down/up; a stopped pool on its next up.
_rp_rewrite_plists() {
local p i
for p in $(_rp_pool_names); do
_rp_load_pool "${p}" || continue
i=1
while [ "${i}" -le "${POOL_COUNT}" ]; do
_rp_write_plist "$(_rp_label "${p}" "${i}")" "${POOL_DIR}/runner-${i}"
i=$(( i + 1 ))
done
_rp_log "rewrote agents for pool '${p}'"
done
}