Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
73 changes: 73 additions & 0 deletions .github/workflows/auto-merge.yml
Original file line number Diff line number Diff line change
@@ -0,0 +1,73 @@
# Auto-merge — nobody is in the merge loop.
#
# Green, ready PRs merge themselves and deploy themselves. The owner does not
# review PRs, and background-job agent sessions are barred from merging by hand,
# so the policy lives in scripts/ci/auto-merge-sweep.sh (read it — it defines
# exactly what "ready" means and how a PR is held back).
#
# Two triggers, deliberately:
# workflow_run — merges within seconds of CI going green (the common path).
# schedule — a safety net. Catches PRs whose checks finished while this
# workflow was failing/disabled, and PRs whose last check was
# an external status that reported after CI. Without it, a PR
# that went green "off-cycle" waits forever.
#
# To stop all of this: delete this file, or add a `hold` label to a PR.

name: Auto-merge

on:
workflow_run:
workflows: ['CI']
types: [completed]
# HOURLY, not */10 as in the rest of the fleet — this repo is PRIVATE, so
# Actions minutes are billed. Every other repo running this sweep is public,
# where scheduled runs are free. At */10 the safety net alone would burn
# ~4,300 billed minutes a month (144 runs/day, each rounded up to a minute)
# against a 2,000-minute free allowance, and it would spend most of them
# discovering there are no PRs. Hourly costs ~720.
#
# This barely slows anything down: the schedule is only the safety net. The
# common path is workflow_run above, which fires within seconds of CI going
# green. Make this repo public and */10 becomes free again.
schedule:
- cron: '0 * * * *'
workflow_dispatch: {}

permissions:
contents: write # merge the PR
pull-requests: write # read PR state, delete the branch
actions: write # dispatch the re-arm workflows
# Required because THIS repo is private. The sweep reads statusCheckRollup to
# decide whether a PR is green; a public repo answers that with no explicit
# scope, which is why the copy of this file going around the fleet omits these
# two lines and still works in ~20 public repos. On a private repo the same
# query fails with 'Resource not accessible by integration' and the sweep
# merges nothing, ever — see ivy-portal, where that went unnoticed because a
# repo that merges nothing looks like a repo with nothing to merge.
checks: read # check-run conclusions (CI jobs)
statuses: read # commit statuses (external reporters)

# Never let two sweeps merge concurrently — they would race on the same PRs.
concurrency:
group: auto-merge
cancel-in-progress: false

jobs:
sweep:
runs-on: ubuntu-latest
timeout-minutes: 10
steps:
- uses: actions/checkout@v4

- name: Merge every green, ready PR
env:
GH_TOKEN: ${{ secrets.GITHUB_TOKEN }}
GH_REPO: ${{ github.repository }}
BASE_BRANCH: ${{ github.event.repository.default_branch }}
CI_WORKFLOW: ci.yml
# Everything that runs on push and therefore would NOT fire after a
# merge made with the default GITHUB_TOKEN. Keep this in sync when a
# push-triggered workflow is added.
REARM_WORKFLOWS: ci.yml
run: bash scripts/ci/auto-merge-sweep.sh
69 changes: 69 additions & 0 deletions .github/workflows/ci.yml
Original file line number Diff line number Diff line change
@@ -0,0 +1,69 @@
name: CI

# Nothing has ever run this repo's checks. `npm run verify` (typecheck + lint +
# vitest) existed from the seed commit and was only ever run by hand, which for
# an agent-built project means: not run. This is the gate that makes it real,
# and it is also what lets auto-merge work at all — the sweep refuses to merge a
# PR with zero checks, so without CI nothing here could ever ship itself.

on:
# Load-bearing for the merge train: a merge made by auto-merge.yml uses the
# default GITHUB_TOKEN, and a push made with that token does NOT trigger
# workflows. The sweep dispatches this explicitly afterwards to verify the
# merge on main; without this trigger that re-arm silently fails.
workflow_dispatch: {}
push:
branches: [main]
pull_request:

concurrency:
group: ${{ github.workflow }}-${{ github.ref }}
cancel-in-progress: true

jobs:
ci:
runs-on: ubuntu-latest

# `next build` needs a real database. The home page reads DATABASE_URL at
# module scope, so page-data collection dies with "DATABASE_URL is not set"
# on a bare runner — verified before writing this. A schema-only Postgres is
# enough: the build only needs the queries to run, not to return rows.
services:
postgres:
image: postgres:17
env:
POSTGRES_PASSWORD: ci
POSTGRES_DB: hamstercheek
ports: ['5432:5432']
options: >-
--health-cmd pg_isready --health-interval 5s
--health-timeout 5s --health-retries 10

env:
DATABASE_URL: postgres://postgres:ci@localhost:5432/hamstercheek

steps:
- uses: actions/checkout@v4

- name: Setup Node
uses: actions/setup-node@v4
with:
node-version: 22
cache: npm

- name: Install dependencies
run: npm ci

# ONE canonical bundle — the exact `npm run verify` you run locally, so a
# green local verify cannot diverge from a green CI. Needs no database:
# the unit tests are pure.
- name: Verify (typecheck + lint + test)
run: npm run verify

- name: Apply migrations (the build queries this DB)
run: npm run db:migrate

# Catches the class verify cannot: code that typechecks but fails to
# compile or blows up while Next collects page data.
- name: Build
run: npm run build
208 changes: 208 additions & 0 deletions scripts/ci/auto-merge-sweep.sh
Original file line number Diff line number Diff line change
@@ -0,0 +1,208 @@
#!/usr/bin/env bash
#
# Merge every open PR that is ready and fully green, then re-arm CI/CD.
#
# WHY THIS EXISTS
# ---------------
# Nobody reviews PRs on this fleet — the owner explicitly does not want to be in
# the merge loop, and background-job agent sessions are barred from merging by
# hand. So the policy lives here, in the repo, where it is visible, revocable,
# and applies uniformly to every PR instead of depending on who opened it.
#
# THE POLICY
# merge a PR <=> it is not a draft
# AND carries no hold label
# AND has at least one check
# AND every check has finished green
# AND GitHub reports it cleanly mergeable
#
# Anything else is left alone for the next sweep. Nothing here forces a merge:
# a red or pending PR simply waits, and a draft waits forever. To hold a ready
# PR back, mark it a draft or add one of the hold labels below.
#
# ONE PR PER SWEEP, OLDEST FIRST, AND ONLY ONTO A GREEN BASE
# ----------------------------------------------------------
# A PR's checks prove *that PR against the base it branched from* — not against
# the other PRs sitting next to it. Merging a batch in one pass would put a
# combination onto the base that nothing ever built. So this script merges at
# most one PR, then hands control back to CI: the merge train advances one car
# per sweep, and every car is verified on the base before the next one couples.
#
# For the same reason it refuses to merge while the base's CI is red or still
# running. Red base => stop adding changes until it is fixed; running CI => the
# answer is not in yet. Both simply defer to the next sweep.
#
# THE RE-ARM (do not remove)
# A push made with the default GITHUB_TOKEN does NOT trigger workflows. Both
# CI and the deploy workflow here run on push, so a merge from this script
# would otherwise land on the base branch and never build or ship. Worse, the
# green-base guard above keys on "a CI run exists for the current tip" — with
# no CI run ever produced by an automated merge, the very next sweep would
# block forever. The explicit workflow_dispatch calls at the end restore both.
#
# REARM_WORKFLOWS is set by .github/workflows/auto-merge.yml and lists exactly
# the workflows that would otherwise have fired on push.

set -euo pipefail

REPO="${GH_REPO:?GH_REPO must be set}"
BASE_BRANCH="${BASE_BRANCH:-main}"
CI_WORKFLOW="${CI_WORKFLOW:-ci.yml}"
REARM_WORKFLOWS="${REARM_WORKFLOWS:-$CI_WORKFLOW}"

# A PR wearing any of these is never merged automatically.
HOLD_LABELS='["hold","no-automerge","do-not-merge","wip"]'

echo "[auto-merge] sweeping open PRs against ${BASE_BRANCH} in ${REPO}"

# Never add changes to a base that is red or mid-verification.
#
# The run has to belong to the CURRENT tip of the base branch. Checking only
# "the latest CI run" is a trap: right after a merge, the newest run is still
# the *previous* commit's — and it is green — so the guard would wave through a
# second merge onto a commit nothing has verified yet. That is exactly the
# batching this script exists to prevent.
base_sha=$(gh api "repos/${REPO}/commits/${BASE_BRANCH}" --jq '.sha')
base_ci=$(gh run list --repo "$REPO" --workflow "$CI_WORKFLOW" --branch "$BASE_BRANCH" --limit 1 \
--json status,conclusion,headSha --jq '.[0] // empty')

if [ -z "$base_ci" ]; then
echo "[auto-merge] no CI history for ${BASE_BRANCH} — proceeding"
else
base_status=$(printf '%s' "$base_ci" | jq -r '.status')
base_conclusion=$(printf '%s' "$base_ci" | jq -r '.conclusion // ""')
base_ci_sha=$(printf '%s' "$base_ci" | jq -r '.headSha')

if [ "$base_ci_sha" != "$base_sha" ]; then
echo "[auto-merge] ${BASE_BRANCH} is at ${base_sha:0:8} but the newest CI run is for ${base_ci_sha:0:8} — waiting for CI to catch up"
exit 0
fi
if [ "$base_status" != "completed" ]; then
echo "[auto-merge] ${BASE_BRANCH} CI is still running — deferring to the next sweep"
exit 0
fi
if [ "$base_conclusion" != "success" ]; then
echo "[auto-merge] ${BASE_BRANCH} CI is ${base_conclusion} — refusing to merge onto a broken base" >&2
exit 0
fi
fi

prs_json=$(gh pr list --repo "$REPO" --state open --base "$BASE_BRANCH" --limit 50 \
--json number,title,isDraft,mergeable,mergeStateStatus,labels,statusCheckRollup)

count=$(printf '%s' "$prs_json" | jq 'length')
if [ "$count" -eq 0 ]; then
echo "[auto-merge] no open PRs"
exit 0
fi

merged_any=0

# OLDEST FIRST. `gh pr list` returns newest-first, and this loop merges the
# first eligible PR and stops — so the newest green PR wins every sweep and an
# older one can wait indefinitely. Observed in maonakamoto/fleetcrown on
# 2026-08-06: consecutive sweeps merged the two newest PRs while three older
# green ones were never even evaluated. With several agent sessions opening PRs
# continuously, "newest wins" is starvation, and it starves the PR whose checks
# were proven against the most now-stale base.
#
# PR numbers increase monotonically with creation, so sorting ascending is FIFO.
for number in $(printf '%s' "$prs_json" | jq -r 'sort_by(.number) | .[].number'); do
pr=$(printf '%s' "$prs_json" | jq -c --argjson n "$number" '.[] | select(.number == $n)')
title=$(printf '%s' "$pr" | jq -r '.title')

# A rollup entry is either a CheckRun (status + conclusion) or a commit
# StatusContext (state) — external services report as the latter.
verdict=$(printf '%s' "$pr" | jq -r --argjson hold "$HOLD_LABELS" '
def ok:
if has("state") then (.state == "SUCCESS")
else ((.status == "COMPLETED")
and ((.conclusion // "") | test("^(SUCCESS|NEUTRAL|SKIPPED)$"))) end;
def pending:
if has("state") then (.state == "PENDING")
else (.status != "COMPLETED") end;

. as $pr
| (($pr.statusCheckRollup) // []) as $checks
| if $pr.isDraft then "skip: draft"
elif ([$pr.labels[]?.name] | any(. as $l | $hold | index($l) != null))
then "skip: hold label"
elif ($checks | length) == 0 then "skip: no checks reported yet"
elif ($checks | map(pending) | any) then "skip: checks still running"
elif (($checks | map(ok) | all) | not) then "skip: checks not green"
else "merge" end
')

if [ "$verdict" != "merge" ]; then
echo "[auto-merge] #${number} ${verdict} — ${title}"

# A CANCELLED check is not a verdict, it is noise: CI workflows in this
# fleet use `concurrency: cancel-in-progress`, so an unrelated newer run on
# the same ref can kill a PR's build. Nothing ever re-runs it, the PR is
# never green, and it would sit in this queue forever. Re-run it and let a
# later sweep judge the real result. Genuine failures are left alone; only a
# run with no real failure is retried.
if [ "$verdict" = "skip: checks not green" ]; then
retry_urls=$(printf '%s' "$pr" | jq -r '
[ .statusCheckRollup[]?
| select(has("state") | not)
| select((.conclusion // "") == "CANCELLED")
| .detailsUrl ] as $cancelled
| [ .statusCheckRollup[]?
| select(((.conclusion // .state // "")
| test("^(FAILURE|TIMED_OUT|ACTION_REQUIRED|STARTUP_FAILURE|ERROR)$"))) ] as $failed
| if ($failed | length) == 0 then $cancelled[] else empty end
')
for url in $retry_urls; do
run_id=$(printf '%s' "$url" | grep -oE '/runs/[0-9]+' | grep -oE '[0-9]+' || true)
[ -z "$run_id" ] && continue
echo "[auto-merge] #${number} re-running cancelled run ${run_id}"
gh run rerun "$run_id" --repo "$REPO" || echo "[auto-merge] #${number} could not re-run ${run_id}" >&2
done
fi
continue
fi

# Mergeability is computed lazily by GitHub and is invalidated every time the
# base branch moves — so right after a merge (exactly when this workflow runs)
# every PR reports UNKNOWN. Poll until GitHub has an answer instead of
# treating "not computed yet" as "not mergeable"; otherwise the fast path can
# never merge anything and the whole train falls back to the cron.
mergeable=""
state=""
for attempt in 1 2 3 4 5 6; do
fresh=$(gh pr view "$number" --repo "$REPO" --json mergeable,mergeStateStatus)
mergeable=$(printf '%s' "$fresh" | jq -r '.mergeable')
state=$(printf '%s' "$fresh" | jq -r '.mergeStateStatus')
[ "$mergeable" != "UNKNOWN" ] && break
echo "[auto-merge] #${number} mergeability not computed yet (attempt ${attempt}) — waiting"
sleep 5
done

if [ "$mergeable" != "MERGEABLE" ]; then
echo "[auto-merge] #${number} skip: not mergeable (${mergeable}/${state}) — ${title}"
continue
fi

echo "[auto-merge] #${number} green and ready — merging: ${title}"
if gh pr merge "$number" --repo "$REPO" --squash --delete-branch; then
merged_any=1
echo "[auto-merge] #${number} merged"
# One car per sweep: let CI verify this on the base before the next couples.
break
else
# Losing a race (someone merged first, or the base moved underneath) is
# normal; the next sweep re-evaluates from fresh state.
echo "[auto-merge] #${number} merge failed — leaving for the next sweep" >&2
fi
done

if [ "$merged_any" -eq 1 ]; then
for wf in $REARM_WORKFLOWS; do
echo "[auto-merge] re-arming ${wf} on ${BASE_BRANCH}"
gh workflow run "$wf" --repo "$REPO" --ref "$BASE_BRANCH" \
|| echo "[auto-merge] could not dispatch ${wf} — is workflow_dispatch declared?" >&2
done
else
echo "[auto-merge] nothing merged; no re-arm needed"
fi
Loading