mirror of
https://github.com/simstudioai/sim.git
synced 2026-09-24 15:45:35 +08:00
fix(ci): unblock @next/swc, lockfile-keyed node_modules, per-image runner sizing (#5945)
* fix(ci): key node_modules sticky disk on the lockfile hash * improvement(ci): per-image Blacksmith runner sizing + cold-build memory preflight * fix(deps): exclude @next/swc binaries from the release-age gate * fix(deps): pin @next/swc binaries so frozen installs get a compiler * docs(ci): explain ARM runner sizing rationale
This commit is contained in:
@@ -104,7 +104,7 @@ jobs:
|
||||
name: Build Dev ECR
|
||||
needs: [detect-version, migrate-dev]
|
||||
if: github.event_name == 'push' && github.ref == 'refs/heads/dev'
|
||||
runs-on: ${{ (vars.CI_PROVIDER == '' || vars.CI_PROVIDER == 'blacksmith') && 'blacksmith-8vcpu-ubuntu-2404' || matrix.gh_runner }}
|
||||
runs-on: ${{ (vars.CI_PROVIDER == '' || vars.CI_PROVIDER == 'blacksmith') && matrix.bs_runner || matrix.gh_runner }}
|
||||
timeout-minutes: 30
|
||||
permissions:
|
||||
contents: read
|
||||
@@ -115,18 +115,25 @@ jobs:
|
||||
include:
|
||||
# Only the app image needs the paid 8-core/32 GB runner: next build
|
||||
# exhausts the free 16 GB one (exit 137). The others build in <5 min.
|
||||
# bs_runner mirrors that per-image sizing on Blacksmith — a single
|
||||
# pinned tier put every image on 8 vCPU, where the non-app builds idle
|
||||
# at 12-15% CPU and under 10% memory.
|
||||
- dockerfile: ./docker/app.Dockerfile
|
||||
ecr_repo_secret: ECR_APP
|
||||
gh_runner: linux-x64-8-core
|
||||
bs_runner: blacksmith-8vcpu-ubuntu-2404
|
||||
- dockerfile: ./docker/db.Dockerfile
|
||||
ecr_repo_secret: ECR_MIGRATIONS
|
||||
gh_runner: ubuntu-latest
|
||||
bs_runner: blacksmith-2vcpu-ubuntu-2404
|
||||
- dockerfile: ./docker/realtime.Dockerfile
|
||||
ecr_repo_secret: ECR_REALTIME
|
||||
gh_runner: ubuntu-latest
|
||||
bs_runner: blacksmith-4vcpu-ubuntu-2404
|
||||
- dockerfile: ./docker/pii.Dockerfile
|
||||
ecr_repo_secret: ECR_PII
|
||||
gh_runner: ubuntu-latest
|
||||
bs_runner: blacksmith-4vcpu-ubuntu-2404
|
||||
steps:
|
||||
- name: Checkout code
|
||||
uses: actions/checkout@df4cb1c069e1874edd31b4311f1884172cec0e10 # v6
|
||||
@@ -215,7 +222,7 @@ jobs:
|
||||
if: >-
|
||||
github.event_name == 'push' &&
|
||||
(github.ref == 'refs/heads/main' || github.ref == 'refs/heads/staging')
|
||||
runs-on: ${{ (vars.CI_PROVIDER == '' || vars.CI_PROVIDER == 'blacksmith') && 'blacksmith-8vcpu-ubuntu-2404' || matrix.gh_runner }}
|
||||
runs-on: ${{ (vars.CI_PROVIDER == '' || vars.CI_PROVIDER == 'blacksmith') && matrix.bs_runner || matrix.gh_runner }}
|
||||
timeout-minutes: 30
|
||||
permissions:
|
||||
contents: read
|
||||
@@ -229,18 +236,22 @@ jobs:
|
||||
ghcr_image: ghcr.io/simstudioai/simstudio
|
||||
ecr_repo_secret: ECR_APP
|
||||
gh_runner: linux-x64-8-core
|
||||
bs_runner: blacksmith-8vcpu-ubuntu-2404
|
||||
- dockerfile: ./docker/db.Dockerfile
|
||||
ghcr_image: ghcr.io/simstudioai/migrations
|
||||
ecr_repo_secret: ECR_MIGRATIONS
|
||||
gh_runner: ubuntu-latest
|
||||
bs_runner: blacksmith-2vcpu-ubuntu-2404
|
||||
- dockerfile: ./docker/realtime.Dockerfile
|
||||
ghcr_image: ghcr.io/simstudioai/realtime
|
||||
ecr_repo_secret: ECR_REALTIME
|
||||
gh_runner: ubuntu-latest
|
||||
bs_runner: blacksmith-4vcpu-ubuntu-2404
|
||||
- dockerfile: ./docker/pii.Dockerfile
|
||||
ghcr_image: ghcr.io/simstudioai/pii
|
||||
ecr_repo_secret: ECR_PII
|
||||
gh_runner: ubuntu-latest
|
||||
bs_runner: blacksmith-4vcpu-ubuntu-2404
|
||||
steps:
|
||||
- name: Checkout code
|
||||
uses: actions/checkout@df4cb1c069e1874edd31b4311f1884172cec0e10 # v6
|
||||
@@ -387,7 +398,7 @@ jobs:
|
||||
# never moves a documented tag.
|
||||
build-ghcr-arm64:
|
||||
name: Build ARM64 (GHCR Only)
|
||||
runs-on: ${{ (vars.CI_PROVIDER == '' || vars.CI_PROVIDER == 'blacksmith') && 'blacksmith-8vcpu-ubuntu-2404-arm' || matrix.gh_runner }}
|
||||
runs-on: ${{ (vars.CI_PROVIDER == '' || vars.CI_PROVIDER == 'blacksmith') && matrix.bs_runner || matrix.gh_runner }}
|
||||
timeout-minutes: 30
|
||||
if: github.event_name == 'push' && github.ref == 'refs/heads/main'
|
||||
permissions:
|
||||
@@ -396,19 +407,27 @@ jobs:
|
||||
strategy:
|
||||
fail-fast: false
|
||||
matrix:
|
||||
# Non-app images sit at 4 vCPU rather than the finer x64 split: the ARM
|
||||
# sizing data is job-level (8 -> 4 for the whole matrix), not per-image,
|
||||
# and this job only runs on push to main — an unprovisioned label would
|
||||
# hang a release in `queued` rather than fail a PR.
|
||||
include:
|
||||
- dockerfile: ./docker/app.Dockerfile
|
||||
image: ghcr.io/simstudioai/simstudio
|
||||
gh_runner: linux-arm64-8-core
|
||||
bs_runner: blacksmith-8vcpu-ubuntu-2404-arm
|
||||
- dockerfile: ./docker/db.Dockerfile
|
||||
image: ghcr.io/simstudioai/migrations
|
||||
gh_runner: ubuntu-24.04-arm
|
||||
bs_runner: blacksmith-4vcpu-ubuntu-2404-arm
|
||||
- dockerfile: ./docker/realtime.Dockerfile
|
||||
image: ghcr.io/simstudioai/realtime
|
||||
gh_runner: ubuntu-24.04-arm
|
||||
bs_runner: blacksmith-4vcpu-ubuntu-2404-arm
|
||||
- dockerfile: ./docker/pii.Dockerfile
|
||||
image: ghcr.io/simstudioai/pii
|
||||
gh_runner: ubuntu-24.04-arm
|
||||
bs_runner: blacksmith-4vcpu-ubuntu-2404-arm
|
||||
|
||||
steps:
|
||||
- name: Checkout code
|
||||
|
||||
@@ -10,7 +10,8 @@ permissions:
|
||||
jobs:
|
||||
process-docs-embeddings:
|
||||
name: Process Documentation Embeddings
|
||||
runs-on: ${{ (vars.CI_PROVIDER == '' || vars.CI_PROVIDER == 'blacksmith') && 'blacksmith-8vcpu-ubuntu-2404' || 'ubuntu-latest' }}
|
||||
# Network-bound on the embeddings API: ~9% CPU and ~3% peak memory on 8 vCPU.
|
||||
runs-on: ${{ (vars.CI_PROVIDER == '' || vars.CI_PROVIDER == 'blacksmith') && 'blacksmith-2vcpu-ubuntu-2404' || 'ubuntu-latest' }}
|
||||
timeout-minutes: 30
|
||||
if: github.ref == 'refs/heads/main'
|
||||
|
||||
|
||||
@@ -31,6 +31,13 @@ jobs:
|
||||
# namespace on top: untrusted fork runs must never share a cache with
|
||||
# push runs (whose caches feed production image builds) or with trusted
|
||||
# internal-PR runs.
|
||||
#
|
||||
# node_modules also keys on the lockfile hash: a sticky disk is a mutable
|
||||
# volume, and `bun install --frozen-lockfile` adds what the lockfile needs
|
||||
# without pruning what it dropped, so branches on different lockfiles were
|
||||
# contaminating each other (a stale @next/swc 16.2.6 outlived the 16.2.11
|
||||
# bump). The bun and Turbo caches are content/hash-addressed, so they stay
|
||||
# shared — that is what keeps a fresh node_modules disk cheap to fill.
|
||||
- name: Mount Bun cache
|
||||
uses: ./.github/actions/cache-mount
|
||||
with:
|
||||
@@ -42,7 +49,7 @@ jobs:
|
||||
uses: ./.github/actions/cache-mount
|
||||
with:
|
||||
provider: ${{ vars.CI_PROVIDER }}
|
||||
key: ${{ github.repository }}-node-modules-${{ github.event_name }}${{ github.event.pull_request.head.repo.fork && '-fork' || '' }}
|
||||
key: ${{ github.repository }}-node-modules-${{ github.event_name }}${{ github.event.pull_request.head.repo.fork && '-fork' || '' }}-${{ hashFiles('bun.lock') }}
|
||||
path: ./node_modules
|
||||
|
||||
- name: Mount Turbo cache
|
||||
@@ -193,19 +200,16 @@ jobs:
|
||||
# Next.js production build, in parallel with lint + tests. Sticky disks are
|
||||
# cloned from the last committed snapshot per job and committed last-writer-
|
||||
# wins, so concurrent mounts are safe. The bun/node_modules disks are shared
|
||||
# with test-build (identical content from the same lockfile — LWW loss is
|
||||
# harmless), but the Turbo cache gets its own key: with a shared key, only
|
||||
# the last committer's new entries survive each run, so the test and build
|
||||
# Turbo entries would evict each other nondeterministically.
|
||||
# Same next build as the app image, so it needs a large runner. Peak RSS
|
||||
# tracks Turbo/Turbopack cache warmth, and it is the COLD case that sizes the
|
||||
# runner: warm peaks ~12 GB, partial ~28 GB, and a cold full rebuild peaked
|
||||
# 51 GB. The 8vcpu tier only has 30.4 GB, so cold-cache runs OOM-killed the VM
|
||||
# (oom_count 1, memory p100 ~30.5 GB, ~99% of the tier) — the job burned 8-15
|
||||
# min and died, while warm runs finished under 8 min. NODE_OPTIONS'
|
||||
# --max-old-space-size only caps Node's JS heap; the bulk is native Turbopack
|
||||
# worker memory, so the cap cannot prevent this. 16vcpu = 60.8 GB fits the
|
||||
# 51 GB cold peak with headroom. Keep the test job on 8vcpu; its memory is fine.
|
||||
# with test-build (the lockfile-hashed key means they only ever share when the
|
||||
# dependency tree really is identical, so LWW loss is harmless), but the Turbo
|
||||
# cache gets its own key: with a shared key, only the last committer's new
|
||||
# entries survive each run, so the test and build Turbo entries would evict
|
||||
# each other nondeterministically.
|
||||
#
|
||||
# Runner is sized for the COLD-cache build, which is what OOM-killed the 8vcpu
|
||||
# tier (23 kills / 1074 runs at 98% of its 30.4 GB): warm peaks ~12 GB, cold
|
||||
# peaked 51 GB. NODE_OPTIONS' --max-old-space-size caps only Node's JS heap,
|
||||
# not the native Turbopack workers that dominate, so it cannot prevent this.
|
||||
build:
|
||||
name: Build App
|
||||
runs-on: ${{ (vars.CI_PROVIDER == '' || vars.CI_PROVIDER == 'blacksmith') && 'blacksmith-16vcpu-ubuntu-2404' || 'linux-x64-8-core' }}
|
||||
@@ -236,7 +240,7 @@ jobs:
|
||||
uses: ./.github/actions/cache-mount
|
||||
with:
|
||||
provider: ${{ vars.CI_PROVIDER }}
|
||||
key: ${{ github.repository }}-node-modules-${{ github.event_name }}${{ github.event.pull_request.head.repo.fork && '-fork' || '' }}
|
||||
key: ${{ github.repository }}-node-modules-${{ github.event_name }}${{ github.event.pull_request.head.repo.fork && '-fork' || '' }}-${{ hashFiles('bun.lock') }}
|
||||
path: ./node_modules
|
||||
|
||||
- name: Mount Turbo cache
|
||||
@@ -258,6 +262,21 @@ jobs:
|
||||
key: ${{ github.repository }}-nextjs-cache-${{ github.event_name }}${{ github.event.pull_request.head.repo.fork && '-fork' || '' }}
|
||||
path: ./apps/sim/.next/cache
|
||||
|
||||
# Running out of RAM kills the whole VM and surfaces only as "the runner
|
||||
# has received a shutdown signal" — no mention of memory, ~12 min in. Warn
|
||||
# with the real numbers so that failure is a one-line diagnosis instead of
|
||||
# a mystery. Warn, never fail: a warm build peaks ~12 GB and a partial one
|
||||
# ~28 GB, so a 32 GB runner still completes plenty of builds, and the
|
||||
# GitHub fallback is the break-glass path — degrading it to a guaranteed
|
||||
# failure would be worse than the risk this flags.
|
||||
- name: Check runner memory headroom
|
||||
run: |
|
||||
TOTAL_GB=$(awk '/MemTotal/ {printf "%d", $2/1048576}' /proc/meminfo)
|
||||
echo "Runner memory: ${TOTAL_GB} GB"
|
||||
if [ "$TOTAL_GB" -lt 40 ]; then
|
||||
echo "::warning::Runner has ${TOTAL_GB} GB. A cold-cache build peaks ~51 GB, so this run may be OOM-killed (reported only as 'the runner has received a shutdown signal'). Warm/partial builds should still fit."
|
||||
fi
|
||||
|
||||
- name: Install dependencies
|
||||
run: bun install --frozen-lockfile
|
||||
|
||||
|
||||
Reference in New Issue
Block a user