diff --git a/.github/workflows/codeql-analysis.yml b/.github/workflows/codeql-analysis.yml index 2c173a337c..92e9708c81 100644 --- a/.github/workflows/codeql-analysis.yml +++ b/.github/workflows/codeql-analysis.yml @@ -9,8 +9,11 @@ on: schedule: - cron: '0 2 * * 6' +# Only pull-request runs may be superseded -- see the note in unit-tests.yml. A +# ref-keyed group drops a *pending* run when a newer one joins it, which for a +# security scan means a commit reaching main with no analysis on record. concurrency: - group: codeql-${{ github.ref }} + group: codeql-${{ github.event_name == 'pull_request' && github.ref || github.run_id }} cancel-in-progress: ${{ github.event_name == 'pull_request' }} jobs: diff --git a/.github/workflows/create-release.yml b/.github/workflows/create-release.yml index 7ade36e146..a92cfe5b72 100644 --- a/.github/workflows/create-release.yml +++ b/.github/workflows/create-release.yml @@ -20,6 +20,13 @@ name: Create Release # before either had tagged, which is a double-publish race. Serialise instead: # never cancel a release run, just make the second wait and find the tag # already there. +# Deliberately still keyed on the ref, unlike the validation workflows, and the +# distinction is the point: for them a dropped run means a commit nobody checked, +# which is unrecoverable, whereas here a dropped run means a publish that did not +# happen twice. This workflow tags, uploads to PyPI and uploads to Anaconda, and +# unlike cron-conda and cron-vendor it has *no* job-level concurrency to fall back +# on, so this group is the only thing serialising it. Keying it per run would let +# two runs tag and publish concurrently, which is the worse failure. concurrency: group: create-release-${{ github.ref }} cancel-in-progress: false diff --git a/.github/workflows/cron-conda.yml b/.github/workflows/cron-conda.yml index b7baf93ec2..8a9a5dc7dc 100644 --- a/.github/workflows/cron-conda.yml +++ b/.github/workflows/cron-conda.yml @@ -10,7 +10,13 @@ on: branches: [main] # This workflow commits and pushes, so a cancelled run can leave the bump -# half-applied -- queue superseded runs rather than killing them. +# half-applied. Kept keyed on the ref, unlike the validation workflows: this +# publishes to Anaconda, and letting two runs upload the same version +# concurrently is worse than dropping a superseded duplicate. +# +# Note this group is not what serialises the repository writes -- the +# `conda-update` job below carries its own `repository-maintenance` group, shared +# with cron-vendor, and that is what actually stops two jobs committing at once. concurrency: group: conda-update-${{ github.ref }} cancel-in-progress: false diff --git a/.github/workflows/cron-vendor.yml b/.github/workflows/cron-vendor.yml index 8cf79e015e..509c3fb013 100644 --- a/.github/workflows/cron-vendor.yml +++ b/.github/workflows/cron-vendor.yml @@ -7,7 +7,14 @@ on: branches: [main] # This workflow commits and pushes, so a cancelled run can leave the bump -# half-applied -- queue superseded runs rather than killing them. +# half-applied. Kept keyed on the ref, unlike the validation workflows: a dropped +# run here just means a crawl that did not happen twice, and the next schedule +# picks the registries up again, whereas a dropped validation run means a commit +# nobody ever checked. +# +# Note this group is not what serialises the repository writes -- the +# `vendor-update` job below carries its own `repository-maintenance` group, shared +# with cron-conda, and that is what actually stops two jobs committing at once. concurrency: group: vendor-update-${{ github.ref }} cancel-in-progress: false diff --git a/.github/workflows/deploy-pages.yml b/.github/workflows/deploy-pages.yml index 0ed33d625b..5e4c601aa3 100644 --- a/.github/workflows/deploy-pages.yml +++ b/.github/workflows/deploy-pages.yml @@ -14,6 +14,12 @@ on: permissions: contents: write +# Deliberately still keyed on the ref, unlike the validation workflows. This one +# publishes a *latest state* rather than a verdict on a commit, so coalescing is +# correct: when two pushes land close together the pending build is dropped and +# the newer one publishes, and deploying the older commit's docs after the newer +# ones would actively be wrong. GitHub Pages also permits only one live +# deployment, so serialising here is a requirement rather than a saving. concurrency: group: pages-${{ github.ref }} cancel-in-progress: ${{ github.event_name == 'pull_request' }} diff --git a/.github/workflows/python-compatibility.yml b/.github/workflows/python-compatibility.yml index 893f9f0b6c..d1ee883a1c 100644 --- a/.github/workflows/python-compatibility.yml +++ b/.github/workflows/python-compatibility.yml @@ -11,8 +11,15 @@ on: permissions: contents: read +# Only pull-request runs may be superseded. Keying every event on the ref (as +# this did) coalesces all of main's pushes into one group, and GitHub cancels a +# *pending* run whenever a newer one joins the group -- `cancel-in-progress: +# false` protects a run that has already started, never one still queued. So a +# commit merged while its predecessor was still running lost its validation +# entirely. Keying non-PR events on the run id gives each one its own group, +# which is what "validate every commit" actually requires. concurrency: - group: python-compatibility-${{ github.ref }} + group: python-compatibility-${{ github.event_name == 'pull_request' && github.ref || github.run_id }} cancel-in-progress: ${{ github.event_name == 'pull_request' }} jobs: diff --git a/.github/workflows/unit-tests.yml b/.github/workflows/unit-tests.yml index 8d3335f539..124278bd05 100644 --- a/.github/workflows/unit-tests.yml +++ b/.github/workflows/unit-tests.yml @@ -21,14 +21,22 @@ on: permissions: contents: read -# `github.workflow` is the *caller's* workflow name inside a called workflow, -# so the `unit-tests-` prefix is what stops a called run from ever sharing a -# group with the caller that invoked it -- if they shared one, the callee -# would cancel its own caller. Only pull-request runs are cancellable: on -# main, on a tag, or on the release path a cancelled run is worse than a slow -# one. +# Only pull-request runs may be superseded. Keying every event on the ref (as +# this did) coalesces all of main's pushes into one group, and GitHub cancels a +# *pending* run whenever a newer one joins the group -- `cancel-in-progress: +# false` protects a run that has already started, never one still queued. Merging +# two PRs a minute apart therefore left the first commit with no unit tests at +# all: measured on run 34970682137 for e0f0160a3, which finished `cancelled` with +# zero jobs one second after the next push's run was created. Keying non-PR +# events on the run id gives each commit its own group. +# +# `github.workflow` is the *caller's* workflow name inside a called workflow, and +# `github.run_id` is the caller's run id too, so the literal `unit-tests-` prefix +# is what stops a called run from sharing a group with the caller that invoked it +# -- if they shared one, the callee would sit pending behind its own caller and +# deadlock. concurrency: - group: unit-tests-${{ github.workflow }}-${{ github.ref }} + group: unit-tests-${{ github.workflow }}-${{ github.event_name == 'pull_request' && github.ref || github.run_id }} cancel-in-progress: ${{ github.event_name == 'pull_request' }} jobs: