diff --git a/.github/workflows/_report-unattended-failure.yml b/.github/workflows/_report-unattended-failure.yml new file mode 100644 index 000000000..c60a29967 --- /dev/null +++ b/.github/workflows/_report-unattended-failure.yml @@ -0,0 +1,84 @@ +name: "Report an unattended failure (reusable)" + +# Records a failed run that nobody is watching on a GitHub issue. Called as the +# last job of fuzz.yml, bench-eql.yml, macro-expand-eql.yml and test-eql.yml. +# +# WHY. A scheduled run, or a push to main, has no pull request to turn red. +# GitHub mails a scheduled run's failure to one person — whoever last edited +# the cron line — and a push run's to whoever pushed. Everyone else learns of +# it only by opening the Actions tab. An issue notifies the repository's +# watchers and stays open until someone deals with it. +# +# ONE ISSUE PER WORKFLOW AND BRANCH, NOT PER RUN. The issue is found by its +# exact title, which names both. If one is open, the failure is added to it as +# a comment, so a workflow that keeps failing nightly builds one thread rather +# than a pile of duplicates. Once it is closed, the next failure opens a new +# one. +# +# The caller decides when this runs and grants the permission, because a called +# workflow cannot raise its caller's token: +# +# report: +# needs: [] +# if: >- +# failure() && github.event_name != 'pull_request' && +# github.event_name != 'workflow_dispatch' +# permissions: +# issues: write +# uses: ./.github/workflows/_report-unattended-failure.yml +# with: +# guidance: +# +# A pull request shows its own failure, and whoever dispatches a run by hand is +# watching it (and may have pointed it at any branch). Every other event is +# reported, so a trigger added later is covered without editing the condition. +# +# `needs` must list every other job in the workflow: `failure()` only sees the +# jobs a job needs, so a failure in one left out is not reported. +# scripts/__tests__/unattended-failure-report.test.mjs checks this, and that +# every scheduled workflow either calls this file or is listed there with the +# reason it does not. +on: + workflow_call: + inputs: + guidance: + description: "What to do about the failure. Goes in the issue body and in every comment." + required: true + type: string + +permissions: + contents: read + +jobs: + report: + name: Open or update the failure issue + runs-on: ubuntu-latest + timeout-minutes: 5 + permissions: + issues: write + steps: + - name: Open or update the failure issue + env: + GH_TOKEN: ${{ github.token }} + GH_REPO: ${{ github.repository }} + # `github.workflow` in a called workflow is the CALLER's name. + TITLE: "${{ github.workflow }} failed on ${{ github.ref_name }}" + EVENT: ${{ github.event_name }} + SHA: ${{ github.sha }} + RUN_URL: ${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }} + GUIDANCE: ${{ inputs.guidance }} + run: | + set -euo pipefail + body="The \`$EVENT\` run at $SHA failed: $RUN_URL + + $GUIDANCE" + # `--search` matches words, not the whole title; the jq filter + # demands an exact match so a similarly named workflow's issue is + # never commented on. + issue=$(gh issue list --state open --search "in:title \"$TITLE\"" \ + --json number,title --jq "map(select(.title == env.TITLE)) | .[0].number // empty") + if [ -n "$issue" ]; then + gh issue comment "$issue" --body "$body" + else + gh issue create --title "$TITLE" --label needs-triage --body "$body" + fi diff --git a/.github/workflows/bench-eql.yml b/.github/workflows/bench-eql.yml index 8d46945ae..d858c427f 100644 --- a/.github/workflows/bench-eql.yml +++ b/.github/workflows/bench-eql.yml @@ -180,3 +180,23 @@ jobs: active_rust_toolchain=$(rustup show active-toolchain | cut -d' ' -f1) rustup component add --toolchain "${active_rust_toolchain}" rustfmt clippy mise run --output prefix test:bench --postgres "${POSTGRES_VERSION}" + + # Every run but a pull request or a manual dispatch is one nobody is + # watching, with no PR to turn red, so record its failure on an issue. See + # _report-unattended-failure.yml; `needs` must name every other job. + report: + needs: + - bench + if: >- + failure() && github.event_name != 'pull_request' && + github.event_name != 'workflow_dispatch' + permissions: + issues: write + uses: ./.github/workflows/_report-unattended-failure.yml + with: + guidance: >- + The EQL benchmark run failed. If the first failing step is the + `require-cs-secrets` pre-flight, a CipherStash credential was rotated + or cleared; otherwise a benchmark, regression or scale test broke, and + on a push run the commit it ran is the first suspect. Close this issue + once a run on main passes again. diff --git a/.github/workflows/fuzz.yml b/.github/workflows/fuzz.yml index baa8e8cd8..e21153c6c 100644 --- a/.github/workflows/fuzz.yml +++ b/.github/workflows/fuzz.yml @@ -189,3 +189,25 @@ jobs: name: fuzz-artifacts-${{ matrix.slug }} path: ${{ matrix.dir }}/fuzz/artifacts/** if-no-files-found: ignore + + # Every run but a pull request or a manual dispatch is one nobody is + # watching, with no PR to turn red, so record its failure on an issue. See + # _report-unattended-failure.yml; `needs` must name every other job. + report: + needs: + - fuzz-regression + - fuzz-campaign + if: >- + failure() && github.event_name != 'pull_request' && + github.event_name != 'workflow_dispatch' + permissions: + issues: write + uses: ./.github/workflows/_report-unattended-failure.yml + with: + guidance: >- + A fuzz campaign found an input that crashes one of the targets. The + reproducer is attached to the run as a `fuzz-artifacts-` + artifact. Replay it with `cargo +nightly fuzz run + ` (docs/fuzzing.md), fix the bug, add the input to + that target's seed corpus so every PR replays it, then close this + issue. diff --git a/.github/workflows/macro-expand-eql.yml b/.github/workflows/macro-expand-eql.yml index f0949d16b..c5c04ba14 100644 --- a/.github/workflows/macro-expand-eql.yml +++ b/.github/workflows/macro-expand-eql.yml @@ -127,3 +127,23 @@ jobs: tests/sqlx/snapshots/text_expanded.rs \ tests/sqlx/snapshots/boolean_expanded.rs \ || { echo "Expansion snapshot stale — run 'mise run test:matrix:expand' (needs the pinned nightly) and commit."; exit 1; } + + # Every run but a pull request or a manual dispatch is one nobody is + # watching, with no PR to turn red, so record its failure on an issue. See + # _report-unattended-failure.yml; `needs` must name every other job. + report: + needs: + - macro-expand + if: >- + failure() && github.event_name != 'pull_request' && + github.event_name != 'workflow_dispatch' + permissions: + issues: write + uses: ./.github/workflows/_report-unattended-failure.yml + with: + guidance: >- + A `cargo expand` matrix snapshot no longer matches its committed copy, + most likely because a change to the matrix macro bodies merged without + regenerating them. Run `mise run test:matrix:expand` in + `packages/eql`, review and commit the snapshot diff, then close this + issue. diff --git a/.github/workflows/test-eql.yml b/.github/workflows/test-eql.yml index a75d02a28..7ba25a735 100644 --- a/.github/workflows/test-eql.yml +++ b/.github/workflows/test-eql.yml @@ -1028,3 +1028,38 @@ jobs: esac done echo "ci-required: all needed jobs passed or were skipped" + + # Every run but a pull request or a manual dispatch is one nobody is + # watching, with no PR to turn red, so record its failure on an issue. See + # _report-unattended-failure.yml; `needs` must name every other job. + report: + needs: + - changes + - setup + - build-archive + - test + - validate + - schema + - rust-crates + - codegen + - self-contained-v3 + - matrix-coverage + - splinter + - docs-static + - known-failures + - doc-anchors + - e2e + - ci-required + if: >- + failure() && github.event_name != 'pull_request' && + github.event_name != 'workflow_dispatch' + permissions: + issues: write + uses: ./.github/workflows/_report-unattended-failure.yml + with: + guidance: >- + The full EQL matrix (PG 14-17) failed. On a push run the commit it ran + is the first suspect; a failure seen only on the nightly run usually + means a dependency, toolchain or image moved underneath an unchanged + tree. Find the failing job in the run, fix it, then close this issue + once a run on main passes again. diff --git a/docs/fuzzing.md b/docs/fuzzing.md index e1d7ace9e..f77cfd3b5 100644 --- a/docs/fuzzing.md +++ b/docs/fuzzing.md @@ -125,6 +125,9 @@ Two jobs with deliberately different roles: persisted across runs via `actions/cache` (write-once key + prefix `restore-keys`) so coverage compounds, minimized with `cargo fuzz cmin` to stay small, and any crash reproducer is uploaded as an artifact. + A failed scheduled run opens an issue titled `Fuzz (crates) failed on + main`, or comments on it if it is still open, linking the run + (`.github/workflows/_report-unattended-failure.yml`). The `pull_request` trigger is path-filtered to `packages/stack-auth/**`, `packages/stack-kms/**`, `packages/stack-encrypt/**`, the root `Cargo.toml`, diff --git a/scripts/__tests__/lib/expressions.mjs b/scripts/__tests__/lib/expressions.mjs index 34eef8ea9..910d0a7d0 100644 --- a/scripts/__tests__/lib/expressions.mjs +++ b/scripts/__tests__/lib/expressions.mjs @@ -135,14 +135,32 @@ function lookup(path, context) { * `cancelled()` is a fact about the run, not about any job, so it comes from * the caller's `run` argument and is false unless that says `cancelled: true`. * - * Everything else still throws. `success()` and `failure()` depend on the - * results of the whole `needs:` chain, which the contexts here do not carry, - * and `contains()` / `startsWith` take arguments the parser below deliberately - * cannot evaluate. + * `failure()` depends on the results of the job's `needs:` chain, which the + * contexts here do not carry, so it has no default: it comes from the caller's + * `run.failure`, and throws when the caller did not say. The unattended-failure + * report jobs are `failure() && `; a guard asking whether such a + * job runs on an event passes `{ failure: true }`, holding the status open the + * way `PERMISSIVE_NEEDS` holds the job outputs open, so what it sees is the + * event gate. + * + * Everything else still throws. `success()` depends on the same `needs:` + * results, and `contains()` / `startsWith` take arguments the parser below + * deliberately cannot evaluate. */ const SUPPORTED_FUNCTIONS = new Map([ ['always', () => true], ['cancelled', (run) => run.cancelled === true], + [ + 'failure', + (run) => { + if (typeof run.failure !== 'boolean') { + throw new UnsupportedExpression( + 'failure() depends on the needs: results; pass run.failure to say which', + ) + } + return run.failure + }, + ], ]) /** @@ -247,7 +265,7 @@ export function unwrap(condition) { return (match ? match[1] : trimmed).trim() } -/** `context` is `{ github, needs, vars }`; `run` is `{ cancelled }`. */ +/** `context` is `{ github, needs, vars }`; `run` is `{ cancelled, failure }`. */ export function runsWhen(condition, context, run = {}) { return evaluate(unwrap(condition), context, run) } diff --git a/scripts/__tests__/unattended-failure-report.test.mjs b/scripts/__tests__/unattended-failure-report.test.mjs new file mode 100644 index 000000000..8cb3443dd --- /dev/null +++ b/scripts/__tests__/unattended-failure-report.test.mjs @@ -0,0 +1,125 @@ +import { describe, expect, it } from 'vitest' +import { expr } from './lib/expressions.mjs' +import { readWorkflow, workflowFiles } from './lib/workflows.mjs' + +/** + * A scheduled workflow that fails must tell someone. + * + * WHAT HAPPENED. The nightly fuzz campaign uploaded each crash reproducer as an + * artifact and stopped there. A scheduled run has no pull request to turn red, + * and GitHub mails its failure to one person — whoever last edited the cron + * line — so a crash would have sat in the Actions tab until someone happened to + * look. test-eql.yml, macro-expand-eql.yml and bench-eql.yml had the same gap, + * as did the push-to-main runs of the first and last; only + * musl-build-image.yml opened an issue. + * + * THE FIX is `_report-unattended-failure.yml`, called as the last job of each + * of those workflows. This file pins the ways that call can quietly stop + * working: + * + * 1. A new scheduled workflow lands without it. So scheduled workflows are + * DISCOVERED by scanning the directory, and each must either call the + * reporter or be listed in `REPORTS_ELSEWHERE` with its reason. + * 2. A job is added to a workflow but not to the report job's `needs`. The + * report job's `failure()` only sees the jobs it needs, so a failure in the + * new job would go unreported, silently. So `needs` must be every other job. + * 3. The condition is narrowed back to `github.event_name == 'schedule'`, + * which drops push-to-main failures and fails shut on any trigger added + * later (eql-matrix-triggers.test.mjs forbids that shape in test-eql.yml + * for the same reason). So the condition is held to one spelling. + */ + +const REPORTER = './.github/workflows/_report-unattended-failure.yml' + +/** The report job's condition, as the parsed workflow holds it. */ +const CONDITION = + "failure() && github.event_name != 'pull_request' && github.event_name != 'workflow_dispatch'" + +/** + * Scheduled workflows that do not call the reporter, with the reason. An + * equality, not a floor: an entry for a workflow that has since started + * calling the reporter, or stopped being scheduled, fails too. + */ +const REPORTS_ELSEWHERE = { + // CodeQL uploads its findings to code scanning, where they raise alerts on + // their own. + '.github/workflows/codeql.yml': 'findings go to code scanning', + // Same: `fail-on-vuln: false`, and the scan's SARIF goes to code scanning. + '.github/workflows/osv-scanner.yml': 'findings go to code scanning', + // Opens its own issue from a step inside its single job, pinned by + // musl-build-image.test.mjs. + '.github/workflows/musl-build-image.yml': 'opens its own issue', +} + +/** + * The workflows known to call the reporter today. A minimum, so the discovery + * below cannot pass by finding nothing. + */ +const EXPECTED_REPORTERS = [ + '.github/workflows/bench-eql.yml', + '.github/workflows/fuzz.yml', + '.github/workflows/macro-expand-eql.yml', + '.github/workflows/test-eql.yml', +] + +function triggers(wf) { + return wf?.on ?? wf?.[true] ?? {} +} + +const workflows = workflowFiles().map((path) => ({ + path, + wf: readWorkflow(path), +})) + +function reportJobs(wf) { + return Object.entries(wf?.jobs ?? {}).filter( + ([, job]) => job?.uses === REPORTER, + ) +} + +const reporters = workflows.filter(({ wf }) => reportJobs(wf).length > 0) + +describe('unattended workflow failures are reported', () => { + it('discovers the known reporters', () => { + expect(reporters.map(({ path }) => path)).toEqual( + expect.arrayContaining(EXPECTED_REPORTERS), + ) + }) + + it('every scheduled workflow calls the reporter or says why not', () => { + const silent = workflows + .filter(({ wf }) => triggers(wf).schedule) + .filter(({ wf }) => reportJobs(wf).length === 0) + .map(({ path }) => path) + .sort() + expect(silent).toEqual(Object.keys(REPORTS_ELSEWHERE).sort()) + }) + + for (const { path, wf } of reporters) { + it(`${path} reports a failure in any of its jobs`, () => { + const jobs = reportJobs(wf) + expect(jobs).toHaveLength(1) + const [[name, job]] = jobs + const others = Object.keys(wf.jobs) + .filter((other) => other !== name) + .sort() + expect([job.needs].flat().sort()).toEqual(others) + expect(job.if).toBe(CONDITION) + expect(job.permissions).toEqual({ issues: 'write' }) + expect(String(job.with?.guidance ?? '').trim()).not.toBe('') + }) + } + + it('the reporter keeps one open issue per workflow and branch', () => { + const wf = readWorkflow(REPORTER.slice(2)) + expect(triggers(wf).workflow_call?.inputs?.guidance?.required).toBe(true) + const steps = Object.values(wf.jobs).flatMap((job) => job?.steps ?? []) + const run = steps.map((step) => String(step?.run ?? '')).join('\n') + expect(run).toContain('gh issue create') + expect(run).toContain('gh issue comment') + // The issue is found by its title, so the title must name the caller. + const env = Object.assign({}, ...steps.map((step) => step?.env ?? {})) + expect(env.TITLE).toContain(expr('github.workflow')) + expect(env.TITLE).toContain(expr('github.ref_name')) + }) +}) diff --git a/scripts/__tests__/workflow-dispatch-job-conditions.test.mjs b/scripts/__tests__/workflow-dispatch-job-conditions.test.mjs index 59172aa0a..8c79d1ad3 100644 --- a/scripts/__tests__/workflow-dispatch-job-conditions.test.mjs +++ b/scripts/__tests__/workflow-dispatch-job-conditions.test.mjs @@ -104,6 +104,13 @@ const DISPATCH_SKIPPED_JOBS = [ '.github/workflows/release.yml / prerelease-eql-docs', '.github/workflows/release.yml / prerelease-eql-npm', '.github/workflows/release.yml / prerelease-eql-sql', + // The unattended-failure reporters. Whoever dispatches a run is watching it, + // and may have pointed it at any branch, so a dispatched failure must not + // open an issue against main. See _report-unattended-failure.yml. + '.github/workflows/bench-eql.yml / report', + '.github/workflows/fuzz.yml / report', + '.github/workflows/macro-expand-eql.yml / report', + '.github/workflows/test-eql.yml / report', ] /** @@ -363,8 +370,11 @@ describe('a declared workflow_dispatch actually dispatches', () => { it('skips exactly the jobs declared unreachable by a plain dispatch', () => { // Both directions, so the list cannot become a place findings go to be // forgotten. + // `failure()` held true, like the job outputs in PERMISSIVE_NEEDS: a + // report job skipping because nothing failed is not a dispatch skip. const skipped = DISPATCHABLE_CONDITIONS.filter( - ({ condition }) => !runsWhen(condition, CONTEXTS.workflow_dispatch), + ({ condition }) => + !runsWhen(condition, CONTEXTS.workflow_dispatch, { failure: true }), ).map((entry) => entry.id) expect( @@ -378,7 +388,7 @@ describe('a declared workflow_dispatch actually dispatches', () => { )) { it(`${id} runs on a manual dispatch`, () => { expect( - runsWhen(condition, CONTEXTS.workflow_dispatch), + runsWhen(condition, CONTEXTS.workflow_dispatch, { failure: true }), `This job is skipped when the workflow is dispatched by hand, so its declared \`workflow_dispatch:\` trigger does nothing: the run is created and reports success having executed no job.\n if: ${condition}\nGate on the case that genuinely cannot run — a fork pull request has no secrets — rather than enumerating the events that can: \`github.event_name != 'pull_request' || \`.`, ).toBe(true) }) @@ -517,6 +527,18 @@ describe('the expression evaluator this guard depends on', () => { ) }) + it('reads `failure()` from the run state, and refuses to guess it', () => { + // Unlike `cancelled()`, there is no safe default: whether a job's `needs:` + // failed is the whole question a report job asks. + for (const context of Object.values(CONTEXTS)) { + expect(runsWhen('failure()', context, { failure: true })).toBe(true) + expect(runsWhen('failure()', context, { failure: false })).toBe(false) + } + expect(() => runsWhen('failure()', CONTEXTS.push)).toThrow( + UnsupportedExpression, + ) + }) + it('applies GitHub null coercion rather than JavaScript equality', () => { // `null == 'cipherstash/stack'` is `0 == NaN`. This is the coercion that // turned a missing `pull_request` payload into a silent `false` instead of