fix(ci): unblock @next/swc, lockfile-keyed node_modules, per-image runner sizing (#5945)

* fix(ci): key node_modules sticky disk on the lockfile hash

* improvement(ci): per-image Blacksmith runner sizing + cold-build memory preflight

* fix(deps): exclude @next/swc binaries from the release-age gate

* fix(deps): pin @next/swc binaries so frozen installs get a compiler

* docs(ci): explain ARM runner sizing rationale
This commit is contained in:
Waleed
2026-07-24 16:12:22 -07:00
committed by GitHub
parent 17d77795b4
commit d64739cf4f
6 changed files with 90 additions and 23 deletions
+22 -3
View File
@@ -104,7 +104,7 @@ jobs:
name: Build Dev ECR
needs: [detect-version, migrate-dev]
if: github.event_name == 'push' && github.ref == 'refs/heads/dev'
runs-on: ${{ (vars.CI_PROVIDER == '' || vars.CI_PROVIDER == 'blacksmith') && 'blacksmith-8vcpu-ubuntu-2404' || matrix.gh_runner }}
runs-on: ${{ (vars.CI_PROVIDER == '' || vars.CI_PROVIDER == 'blacksmith') && matrix.bs_runner || matrix.gh_runner }}
timeout-minutes: 30
permissions:
contents: read
@@ -115,18 +115,25 @@ jobs:
include:
# Only the app image needs the paid 8-core/32 GB runner: next build
# exhausts the free 16 GB one (exit 137). The others build in <5 min.
# bs_runner mirrors that per-image sizing on Blacksmith — a single
# pinned tier put every image on 8 vCPU, where the non-app builds idle
# at 12-15% CPU and under 10% memory.
- dockerfile: ./docker/app.Dockerfile
ecr_repo_secret: ECR_APP
gh_runner: linux-x64-8-core
bs_runner: blacksmith-8vcpu-ubuntu-2404
- dockerfile: ./docker/db.Dockerfile
ecr_repo_secret: ECR_MIGRATIONS
gh_runner: ubuntu-latest
bs_runner: blacksmith-2vcpu-ubuntu-2404
- dockerfile: ./docker/realtime.Dockerfile
ecr_repo_secret: ECR_REALTIME
gh_runner: ubuntu-latest
bs_runner: blacksmith-4vcpu-ubuntu-2404
- dockerfile: ./docker/pii.Dockerfile
ecr_repo_secret: ECR_PII
gh_runner: ubuntu-latest
bs_runner: blacksmith-4vcpu-ubuntu-2404
steps:
- name: Checkout code
uses: actions/checkout@df4cb1c069e1874edd31b4311f1884172cec0e10 # v6
@@ -215,7 +222,7 @@ jobs:
if: >-
github.event_name == 'push' &&
(github.ref == 'refs/heads/main' || github.ref == 'refs/heads/staging')
runs-on: ${{ (vars.CI_PROVIDER == '' || vars.CI_PROVIDER == 'blacksmith') && 'blacksmith-8vcpu-ubuntu-2404' || matrix.gh_runner }}
runs-on: ${{ (vars.CI_PROVIDER == '' || vars.CI_PROVIDER == 'blacksmith') && matrix.bs_runner || matrix.gh_runner }}
timeout-minutes: 30
permissions:
contents: read
@@ -229,18 +236,22 @@ jobs:
ghcr_image: ghcr.io/simstudioai/simstudio
ecr_repo_secret: ECR_APP
gh_runner: linux-x64-8-core
bs_runner: blacksmith-8vcpu-ubuntu-2404
- dockerfile: ./docker/db.Dockerfile
ghcr_image: ghcr.io/simstudioai/migrations
ecr_repo_secret: ECR_MIGRATIONS
gh_runner: ubuntu-latest
bs_runner: blacksmith-2vcpu-ubuntu-2404
- dockerfile: ./docker/realtime.Dockerfile
ghcr_image: ghcr.io/simstudioai/realtime
ecr_repo_secret: ECR_REALTIME
gh_runner: ubuntu-latest
bs_runner: blacksmith-4vcpu-ubuntu-2404
- dockerfile: ./docker/pii.Dockerfile
ghcr_image: ghcr.io/simstudioai/pii
ecr_repo_secret: ECR_PII
gh_runner: ubuntu-latest
bs_runner: blacksmith-4vcpu-ubuntu-2404
steps:
- name: Checkout code
uses: actions/checkout@df4cb1c069e1874edd31b4311f1884172cec0e10 # v6
@@ -387,7 +398,7 @@ jobs:
# never moves a documented tag.
build-ghcr-arm64:
name: Build ARM64 (GHCR Only)
runs-on: ${{ (vars.CI_PROVIDER == '' || vars.CI_PROVIDER == 'blacksmith') && 'blacksmith-8vcpu-ubuntu-2404-arm' || matrix.gh_runner }}
runs-on: ${{ (vars.CI_PROVIDER == '' || vars.CI_PROVIDER == 'blacksmith') && matrix.bs_runner || matrix.gh_runner }}
timeout-minutes: 30
if: github.event_name == 'push' && github.ref == 'refs/heads/main'
permissions:
@@ -396,19 +407,27 @@ jobs:
strategy:
fail-fast: false
matrix:
# Non-app images sit at 4 vCPU rather than the finer x64 split: the ARM
# sizing data is job-level (8 -> 4 for the whole matrix), not per-image,
# and this job only runs on push to main — an unprovisioned label would
# hang a release in `queued` rather than fail a PR.
include:
- dockerfile: ./docker/app.Dockerfile
image: ghcr.io/simstudioai/simstudio
gh_runner: linux-arm64-8-core
bs_runner: blacksmith-8vcpu-ubuntu-2404-arm
- dockerfile: ./docker/db.Dockerfile
image: ghcr.io/simstudioai/migrations
gh_runner: ubuntu-24.04-arm
bs_runner: blacksmith-4vcpu-ubuntu-2404-arm
- dockerfile: ./docker/realtime.Dockerfile
image: ghcr.io/simstudioai/realtime
gh_runner: ubuntu-24.04-arm
bs_runner: blacksmith-4vcpu-ubuntu-2404-arm
- dockerfile: ./docker/pii.Dockerfile
image: ghcr.io/simstudioai/pii
gh_runner: ubuntu-24.04-arm
bs_runner: blacksmith-4vcpu-ubuntu-2404-arm
steps:
- name: Checkout code
+2 -1
View File
@@ -10,7 +10,8 @@ permissions:
jobs:
process-docs-embeddings:
name: Process Documentation Embeddings
runs-on: ${{ (vars.CI_PROVIDER == '' || vars.CI_PROVIDER == 'blacksmith') && 'blacksmith-8vcpu-ubuntu-2404' || 'ubuntu-latest' }}
# Network-bound on the embeddings API: ~9% CPU and ~3% peak memory on 8 vCPU.
runs-on: ${{ (vars.CI_PROVIDER == '' || vars.CI_PROVIDER == 'blacksmith') && 'blacksmith-2vcpu-ubuntu-2404' || 'ubuntu-latest' }}
timeout-minutes: 30
if: github.ref == 'refs/heads/main'
+34 -15
View File
@@ -31,6 +31,13 @@ jobs:
# namespace on top: untrusted fork runs must never share a cache with
# push runs (whose caches feed production image builds) or with trusted
# internal-PR runs.
#
# node_modules also keys on the lockfile hash: a sticky disk is a mutable
# volume, and `bun install --frozen-lockfile` adds what the lockfile needs
# without pruning what it dropped, so branches on different lockfiles were
# contaminating each other (a stale @next/swc 16.2.6 outlived the 16.2.11
# bump). The bun and Turbo caches are content/hash-addressed, so they stay
# shared — that is what keeps a fresh node_modules disk cheap to fill.
- name: Mount Bun cache
uses: ./.github/actions/cache-mount
with:
@@ -42,7 +49,7 @@ jobs:
uses: ./.github/actions/cache-mount
with:
provider: ${{ vars.CI_PROVIDER }}
key: ${{ github.repository }}-node-modules-${{ github.event_name }}${{ github.event.pull_request.head.repo.fork && '-fork' || '' }}
key: ${{ github.repository }}-node-modules-${{ github.event_name }}${{ github.event.pull_request.head.repo.fork && '-fork' || '' }}-${{ hashFiles('bun.lock') }}
path: ./node_modules
- name: Mount Turbo cache
@@ -193,19 +200,16 @@ jobs:
# Next.js production build, in parallel with lint + tests. Sticky disks are
# cloned from the last committed snapshot per job and committed last-writer-
# wins, so concurrent mounts are safe. The bun/node_modules disks are shared
# with test-build (identical content from the same lockfile — LWW loss is
# harmless), but the Turbo cache gets its own key: with a shared key, only
# the last committer's new entries survive each run, so the test and build
# Turbo entries would evict each other nondeterministically.
# Same next build as the app image, so it needs a large runner. Peak RSS
# tracks Turbo/Turbopack cache warmth, and it is the COLD case that sizes the
# runner: warm peaks ~12 GB, partial ~28 GB, and a cold full rebuild peaked
# 51 GB. The 8vcpu tier only has 30.4 GB, so cold-cache runs OOM-killed the VM
# (oom_count 1, memory p100 ~30.5 GB, ~99% of the tier) — the job burned 8-15
# min and died, while warm runs finished under 8 min. NODE_OPTIONS'
# --max-old-space-size only caps Node's JS heap; the bulk is native Turbopack
# worker memory, so the cap cannot prevent this. 16vcpu = 60.8 GB fits the
# 51 GB cold peak with headroom. Keep the test job on 8vcpu; its memory is fine.
# with test-build (the lockfile-hashed key means they only ever share when the
# dependency tree really is identical, so LWW loss is harmless), but the Turbo
# cache gets its own key: with a shared key, only the last committer's new
# entries survive each run, so the test and build Turbo entries would evict
# each other nondeterministically.
#
# Runner is sized for the COLD-cache build, which is what OOM-killed the 8vcpu
# tier (23 kills / 1074 runs at 98% of its 30.4 GB): warm peaks ~12 GB, cold
# peaked 51 GB. NODE_OPTIONS' --max-old-space-size caps only Node's JS heap,
# not the native Turbopack workers that dominate, so it cannot prevent this.
build:
name: Build App
runs-on: ${{ (vars.CI_PROVIDER == '' || vars.CI_PROVIDER == 'blacksmith') && 'blacksmith-16vcpu-ubuntu-2404' || 'linux-x64-8-core' }}
@@ -236,7 +240,7 @@ jobs:
uses: ./.github/actions/cache-mount
with:
provider: ${{ vars.CI_PROVIDER }}
key: ${{ github.repository }}-node-modules-${{ github.event_name }}${{ github.event.pull_request.head.repo.fork && '-fork' || '' }}
key: ${{ github.repository }}-node-modules-${{ github.event_name }}${{ github.event.pull_request.head.repo.fork && '-fork' || '' }}-${{ hashFiles('bun.lock') }}
path: ./node_modules
- name: Mount Turbo cache
@@ -258,6 +262,21 @@ jobs:
key: ${{ github.repository }}-nextjs-cache-${{ github.event_name }}${{ github.event.pull_request.head.repo.fork && '-fork' || '' }}
path: ./apps/sim/.next/cache
# Running out of RAM kills the whole VM and surfaces only as "the runner
# has received a shutdown signal" — no mention of memory, ~12 min in. Warn
# with the real numbers so that failure is a one-line diagnosis instead of
# a mystery. Warn, never fail: a warm build peaks ~12 GB and a partial one
# ~28 GB, so a 32 GB runner still completes plenty of builds, and the
# GitHub fallback is the break-glass path — degrading it to a guaranteed
# failure would be worse than the risk this flags.
- name: Check runner memory headroom
run: |
TOTAL_GB=$(awk '/MemTotal/ {printf "%d", $2/1048576}' /proc/meminfo)
echo "Runner memory: ${TOTAL_GB} GB"
if [ "$TOTAL_GB" -lt 40 ]; then
echo "::warning::Runner has ${TOTAL_GB} GB. A cold-cache build peaks ~51 GB, so this run may be OOM-killed (reported only as 'the runner has received a shutdown signal'). Warm/partial builds should still fit."
fi
- name: Install dependencies
run: bun install --frozen-lockfile