Repository navigation
Scoring: make the world reset unit a choice (ADR 0020), make M4/M9 selectable, and measure them #17
This file contains hidden or bidirectional Unicode text that may be interpreted or compiled differently than what appears below. To review, open the file in an editor that reveals hidden Unicode characters.
Learn more about bidirectional Unicode characters
| # Full-chain check: deploy every venue to a bare anvil, dump the state, and prove a backtest runs | |
| # against that dump. | |
| # | |
| # This is the only thing that would have caught the GMX localhost market bug (#42's `1502616`), where | |
| # `npm run deploy` had been dying at GM seeding -- nothing in the fast CI touches the deployer, and | |
| # the state dump is gitignored, so no per-PR job can exercise this path. | |
| # | |
| # ~25-40 minutes, dominated by the gmx-synthetics clone + yarn install and the hardhat deploy, so it | |
| # only runs when something it actually covers changes. | |
| name: deploy + backtest | |
| on: | |
| workflow_dispatch: | |
| pull_request: | |
| branches: [main] | |
| paths: | |
| - "deployer/**" | |
| - "scripts/genStateDump.ts" | |
| - "scripts/genLocalConstants.ts" | |
| - "core/src/cli/backtest.ts" | |
| - "core/src/backtest/**" | |
| - "sdk/src/constants.ts" | |
| - "sdk/src/constants.local.ts" | |
| - "config/regimes/**" | |
| # Scoring. A run's value series cannot be regression-tested any other way: tx timing and | |
| # ordering are non-deterministic (ADR 0005), so a golden run does not exist and unit tests can | |
| # only pin individual functions. This job is the only thing that puts the scorer in front of | |
| # real venue state across all five venues. #44 changed the scorer and skipped this job. | |
| - "core/src/realtime/reconstruct.ts" | |
| - "sdk/src/valuation.ts" | |
| - "sdk/src/protocols/**" | |
| - ".github/workflows/deploy-backtest.yml" | |
| concurrency: | |
| group: deploy-backtest-${{ github.ref }} | |
| cancel-in-progress: true | |
| jobs: | |
| deploy-and-backtest: | |
| runs-on: ubuntu-latest | |
| timeout-minutes: 60 | |
| steps: | |
| - uses: actions/checkout@v4 | |
| - uses: actions/setup-node@v4 | |
| with: | |
| node-version: 22 | |
| cache: npm | |
| - uses: foundry-rs/foundry-toolchain@v1 | |
| with: | |
| version: stable | |
| - run: npm ci | |
| # The mock oracles and PriceFeed the run deploys at setup. The deployer's own `forge build` | |
| # writes to deployer/out and does not cover these. | |
| - run: npm run build:contracts | |
| - name: Set up the deployer | |
| working-directory: deployer | |
| run: | | |
| npm install | |
| forge build | |
| cp .env.example .env | |
| # anvil is started by this workflow, not by the deployer, so gen:state-dump can talk to the | |
| # same chain after the deploy process has exited. | |
| sed -i 's/^MANAGE_ANVIL=.*/MANAGE_ANVIL=false/' .env | |
| ./scripts/setup-vendors.sh | |
| - name: Start anvil | |
| working-directory: deployer | |
| run: | | |
| npm run anvil > "$RUNNER_TEMP/anvil.log" 2>&1 & | |
| for _ in $(seq 1 60); do | |
| cast block-number --rpc-url http://127.0.0.1:8545 >/dev/null 2>&1 && exit 0 | |
| sleep 2 | |
| done | |
| echo "anvil did not come up"; cat "$RUNNER_TEMP/anvil.log"; exit 1 | |
| - name: Deploy all venues | |
| working-directory: deployer | |
| run: npm run deploy -- --keep-fresh | |
| - name: Generate constants and the state dump | |
| run: | | |
| npm run gen:local-constants | |
| npm run gen:state-dump | |
| # A scenario is (regime, seed) and the regime YAML carries no seed, so --seed is required | |
| # (ADR 0017 §1). --blocks shortens the run for CI; the regime's own 360 would take 12 minutes. | |
| - name: Backtest against the dump | |
| run: npm run backtest -- --regime calm --seed 101 --blocks 12 --seconds 90 | |
| # The liquidity pull (issue #52) is the one part of the stress overlay that writes to the | |
| # venues instead of to the price, and `calm` does not carry it. It is also the part most | |
| # exposed to what the deployer actually produced: the Balancer join had to be rebuilt because | |
| # this deployment's WeightedPool does not implement ALL_TOKENS_IN_FOR_EXACT_BPT_OUT, and that | |
| # class of mismatch is invisible until a real pool refuses the call. | |
| # | |
| # 80 blocks, not the 12 used above: the pull's trapezoid is 21 blocks and its start is drawn | |
| # from windowFrac's upper bound (0.7), so a shorter run cannot hold the window and fails fast | |
| # rather than silently mis-composing the pair (see EventSchedule's alignWith check). | |
| - name: Replay the crash regime (liquidity pull across all three venues) | |
| run: | | |
| npm run backtest -- --regime crash --seed 606 --blocks 80 | |
| node -e ' | |
| const { readdirSync, readFileSync } = require("node:fs"); | |
| const dir = readdirSync("runs").filter((d) => /^\d{4}-/.test(d)).sort().at(-1); | |
| const events = readFileSync(`runs/${dir}/events.jsonl`, "utf8") | |
| .split("\n").filter(Boolean).map((l) => JSON.parse(l)); | |
| const pulls = events.filter((e) => e.type === "stress_liquidity_pull"); | |
| const venues = new Set(pulls.map((e) => e.venue)); | |
| for (const v of ["uniswap", "balancer", "curve"]) | |
| if (!venues.has(v)) throw new Error(`no depth was withdrawn from ${v}`); | |
| const bad = events.filter((e) => | |
| ["stress_liquidity_pull_failed", "stress_liquidity_pull_reverted", | |
| "stress_liquidity_task_failed", "stress_liquidity_teardown_failed"].includes(e.type)); | |
| if (bad.length) throw new Error(`liquidity pull writes failed: ${JSON.stringify(bad.slice(0, 3))}`); | |
| // The window has to end with the venues where it started, or the next run on a shared | |
| // chain trades a pool this event drained. | |
| const incomplete = events.filter((e) => e.type === "stress_liquidity_restore_incomplete"); | |
| if (incomplete.length) throw new Error(`depth was not restored: ${JSON.stringify(incomplete)}`); | |
| const restored = events.filter((e) => e.type === "stress_liquidity_restored"); | |
| if (restored.length < 3) throw new Error(`expected a restore per venue, got ${restored.length}`); | |
| console.log(`liquidity pull ok: ${pulls.length} writes across ${[...venues].join(", ")}`); | |
| ' | |
| # The matrix path is what the competition actually runs, and it is a different code path from | |
| # a single --regime run: it writes matrix.json / standings.json and has to survive a scenario | |
| # that fails without abandoning the rest. Two scenarios is enough to exercise the loop. | |
| - name: Replay a small scenario matrix | |
| run: | | |
| cat > "$RUNNER_TEMP/ci-scenarios.yaml" <<'YAML' | |
| regimes: [calm] | |
| seeds: [101, 202] | |
| YAML | |
| npm run backtest -- --scenarios "$RUNNER_TEMP/ci-scenarios.yaml" --blocks 12 --seconds 90 | |
| node -e ' | |
| const { readdirSync, readFileSync } = require("node:fs"); | |
| const dir = readdirSync("runs").filter((d) => d.startsWith("matrix-")).sort().at(-1); | |
| if (!dir) throw new Error("no matrix directory was produced"); | |
| const m = JSON.parse(readFileSync(`runs/${dir}/matrix.json`, "utf8")); | |
| const s = JSON.parse(readFileSync(`runs/${dir}/standings.json`, "utf8")); | |
| const failed = m.scenarios.filter((x) => !x.agents); | |
| if (failed.length) throw new Error(`scenarios produced no result: ${JSON.stringify(failed)}`); | |
| if (m.scenarios.length !== 2) throw new Error(`expected 2 scenarios, got ${m.scenarios.length}`); | |
| if (!s.agents?.length) throw new Error("standings ranked nobody"); | |
| // Both metrics must survive into the matrix, since the scoring rule is expected to | |
| // change and matrix.json is what makes a finished run re-scorable (ADR 0017 §4). | |
| for (const sc of m.scenarios) | |
| for (const a of sc.agents) | |
| if (!("netPnlUsdc" in a) || !("alphaUsdc" in a)) | |
| throw new Error(`${sc.regime}#${sc.seed} ${a.id} is missing a metric`); | |
| console.log(`matrix ok: ${m.scenarios.length} scenarios, ${s.agents.length} ranked`); | |
| ' | |
| # The backtest exits 0 even when the scorer quietly read nothing, which is the failure mode | |
| # this whole job exists to catch, so assert on the run's own output. | |
| - name: Check the run is healthy | |
| run: | | |
| node -e ' | |
| const { readdirSync, readFileSync } = require("node:fs"); | |
| const runs = readdirSync("runs").filter((d) => /^\d{4}-/.test(d)).sort(); | |
| const id = runs.at(-1); | |
| if (!id) throw new Error("no run directory was produced"); | |
| const s = JSON.parse(readFileSync(`runs/${id}/summary.json`, "utf8")); | |
| const problems = []; | |
| if (s.valueSeries?.failedReads !== 0) | |
| problems.push(`failedReads=${s.valueSeries?.failedReads}`); | |
| if ((s.violations ?? []).length) problems.push(`violations=${JSON.stringify(s.violations)}`); | |
| if (!s.agents?.length) problems.push("no agents were scored"); | |
| for (const a of s.agents ?? []) { | |
| if (!Number.isFinite(a.finalValueUsdc) || a.finalValueUsdc <= 0) | |
| problems.push(`${a.id} finalValueUsdc=${a.finalValueUsdc}`); | |
| } | |
| const unpriced = s.valueSeries?.unpricedHoldings ?? []; | |
| if (unpriced.length) console.log(`note: ${unpriced.length} unpriced holding(s):`, JSON.stringify(unpriced)); | |
| if (problems.length) throw new Error(`unhealthy run ${id}: ${problems.join(", ")}`); | |
| console.log(`run ${id} ok: ${s.agents.length} agents, ${s.blocksProcessed} blocks, failedReads 0`); | |
| ' | |
| - name: Upload the run and deploy logs | |
| if: always() | |
| uses: actions/upload-artifact@v4 | |
| with: | |
| name: deploy-backtest-output | |
| path: | | |
| runs/ | |
| deployer/deployments/deployments.json | |
| backtest/state/manifest.json | |
| ${{ runner.temp }}/anvil.log | |
| retention-days: 7 |