From 249381b2493d2157e369f2ca7307feecaf17819c Mon Sep 17 00:00:00 2001 From: Jason Ross Date: Fri, 3 Jul 2026 17:08:37 -0500 Subject: [PATCH] ci(runners): P0 & broken-target preemption; P1-P9 yield without bumping Only P0 preempts in-progress runs (emergency reservation). P1-P9 no longer cancel lower-priority runs; instead traffic-control holds back (bounded poll, kept under timeout-minutes) while strictly-higher-priority PRs still have active/queued CI runs, so their heavy jobs reach the runner queue first. New `broken` label forces effective priority below P9 (sentinel 10): a broken PR never preempts (even if also labelled P0 -- broken wins) and always yields, and because its run is wasted, ANY higher-priority PR (not just P0) may cancel its in-progress run to reclaim the runner. Net rule: a strictly-lower run is cancelled iff (self is P0) OR (target is broken); otherwise yield. All existing safety preserved: never main/push runs, never our own run, never an equal-or-higher-priority PR; PR-controlled strings via env/jq only; continue-on-error + set +e + always exit 0; traffic-control stays a non-required best-effort job and ci-passed is unchanged. Validated with actionlint and a mocked-gh + fake-clock logic harness. Co-Authored-By: Claude Opus 4.8 --- .github/workflows/ci.yml | 222 +++++++++++++++++++++++++++++++-------- 1 file changed, 178 insertions(+), 44 deletions(-) diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index 4ad1d7d..30dd301 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -22,17 +22,50 @@ env: jobs: # ── Priority-based runner orchestration ────────────────────────────────────── # Runs FIRST (the heavy jobs below all `needs: traffic-control`). It reads THIS - # PR's P0–P9 label — P0 = highest priority, P9 = lowest, default P5 when the PR - # carries no P-label — and PREEMPTS: it cancels the in-progress / queued CI runs - # of strictly-LOWER-priority OTHER open PRs, freeing their runners for this - # higher-priority PR. A preempted PR simply re-runs on its next push / autoupdate - # rebase. + # PR's P0–P9 label (and the special `broken` label) to order runner access. + # Effective priority: `broken` => 10 (BOTTOM, below P9), overriding any P0–P9; + # else the lowest-numbered P0–P9 label present (P0 = highest); else default P5. + # + # • P0 = EMERGENCY ONLY (app broken in production / emergency security update). + # P0 PREEMPTS: it cancels the in-progress / queued CI runs of ALL strictly- + # LOWER-priority OTHER open PRs to grab their runners immediately. A preempted + # PR simply re-runs on its next push / autoupdate rebase. + # + # • `broken` = STUCK/FAILING PR — a MANUALLY-applied signal (maintainer / repo + # owner only) meaning "deprioritise to the bottom so others aren't blocked + # behind it while it's being fixed." Its effective priority is 10, so it NEVER + # preempts (even if it's also labelled P0 — `broken` wins; a stuck PR can't be + # an emergency merge) and ALWAYS yields: every other PR, even lower P-levels, + # advances ahead of it. And because a broken PR's run is wasted (it can't + # merge), ANY higher-priority PR — not just P0 — MAY cancel its in-progress run + # to reclaim the runner. Removing the label restores its normal P-priority. + # (Example: a P3 PR with failing CI was making lower-priority PRs wait behind + # it; marking it `broken` lets them proceed — and reclaim its runner.) + # + # • P1–P9 = YIELD WITHOUT BUMPING. They NEVER cancel a NON-broken lower-priority + # run that is already going — a higher-priority PR does not evict it, it just + # takes the next free slot. Mechanism: a bounded hold-back. This job polls and + # defers (up to HOLD_BACK_BUDGET_SECONDS, kept well under timeout-minutes) while + # any strictly-higher-priority OTHER open PR still has an active/queued CI run, + # so that PR's heavy jobs reach the runner queue ahead of this PR's. When the + # budget elapses it proceeds anyway (a PR never blocks itself). + # + # Net preemption rule — a strictly-lower-priority OTHER PR's active run is cancelled + # iff (THIS PR is P0) OR (that PR is `broken`); otherwise it is left to run and we + # yield. So: P0 preempts ALL lower runs; ANY PR preempts lower `broken` runs; P1–P9 + # never preempt a non-broken run. # # Hard safety rules, all enforced in the script below: # • never cancels a run on main / a push event (filters --event pull_request); # • never cancels THIS PR's own run (skips self by PR number + run id); - # • never cancels an equal-or-higher-priority PR (only prio > self); - # • only strictly-lower-priority OTHER open PRs' active runs are cancelled. + # • never cancels an equal-or-higher-priority PR (only strictly-lower, prio > self); + # • P1–P9 cancel NO non-broken run — they only wait (bounded), then proceed. + # + # Honest limitation: GitHub Actions has no native priority queue and assigns + # runners roughly FIFO, so the hold-back is a BEST-EFFORT head-start, not a hard + # guarantee — under sustained contention the bounded wait can expire before a + # higher-priority PR drains. The waiting job also occupies a (cheap, short-lived) + # runner meanwhile, which is exactly why the wait is kept bounded. # # It is deliberately NOT a merge-gate check: it is absent from `ci-passed`'s # needs, every API call is guarded, the script always exits 0, and the step is @@ -43,18 +76,23 @@ jobs: traffic-control: name: Traffic control (runner priority) runs-on: ubuntu-latest - timeout-minutes: 5 + timeout-minutes: 6 # hard backstop; the P1–P9 hold-back budget below stays well under this permissions: - actions: write # cancel workflow runs on lower-priority PRs + actions: write # cancel lower-priority runs (P0 emergencies + broken targets) pull-requests: read # read PR P0–P9 labels env: GH_TOKEN: ${{ github.token }} GH_REPO: ${{ github.repository }} SELF_PR: ${{ github.event.pull_request.number }} + # P1–P9 bounded hold-back knobs. BUDGET must stay comfortably below + # timeout-minutes so the poll loop always exits 0 before the hard job timeout + # fires — a timed-out job would skip the heavy jobs and fail `ci-passed`. + HOLD_BACK_BUDGET_SECONDS: "180" + HOLD_BACK_POLL_SECONDS: "15" steps: # No checkout: this job only calls the gh CLI (auto-configured from GH_TOKEN / # GH_REPO), so it needs neither the repo contents nor the default contents:read. - - name: Preempt lower-priority PR runs + - name: Apply runner priority (P0/broken preempt; P1–P9 hold back) continue-on-error: true # belt-and-suspenders: never let this fail the run run: | # GitHub invokes run steps with `bash -eo pipefail`. Disable errexit so a @@ -64,21 +102,40 @@ jobs: set +e if [ "${GITHUB_EVENT_NAME:-}" != "pull_request" ] || [ -z "${SELF_PR:-}" ]; then - echo "Not a pull_request event (or no PR number) — nothing to preempt." + echo "Not a pull_request event (or no PR number) — nothing to do." exit 0 fi - # Effective priority (0–9) of a labels JSON array read on stdin: the - # highest-priority (lowest-numbered) P0–P9 label present, else 5. + # Effective priority of a labels JSON array read on stdin: a `broken` label + # => 10 (bottom, below P9), overriding any P0–P9; else the highest-priority + # (lowest-numbered) P0–P9 label present; else 5. prio_of() { - jq -r '[ .[] | .name | select(test("^P[0-9]$")) | ltrimstr("P") | tonumber ] - | if length == 0 then 5 else min end' 2>/dev/null + jq -r 'if any(.[]; .name == "broken") then 10 + else ([ .[] | .name | select(test("^P[0-9]$")) | ltrimstr("P") | tonumber ] + | if length == 0 then 5 else min end) end' 2>/dev/null } - # One snapshot of every open PR (number, head branch, labels). - if ! gh pr list --state open --limit 300 \ - --json number,headRefName,labels > open_prs.json 2>err.txt; then - echo "::warning::Could not list open PRs — skipping preemption. $(cat err.txt 2>/dev/null)" + # Snapshot of every open PR (number, head branch, labels) to stdout. + list_open_prs() { + gh pr list --state open --limit 300 --json number,headRefName,labels + } + + # Active (non-completed) CI run ids on head branch $1 — PR events only, never + # main. Shared by the preemption pass (ids to cancel) and the hold-back + # (presence => keep waiting). + active_run_ids_for_head() { + gh run list --workflow ci.yml --branch "$1" --event pull_request \ + --limit 100 --json databaseId,status,headBranch,event 2>/dev/null \ + | jq -r '.[] + | select(.event == "pull_request") + | select(.headBranch != "main") + | select(.status != "completed") + | .databaseId' 2>/dev/null + } + + # One initial snapshot, used to read THIS PR's own priority. + if ! list_open_prs > open_prs.json 2>err.txt; then + echo "::warning::Could not list open PRs — skipping. $(cat err.txt 2>/dev/null)" exit 0 fi @@ -86,40 +143,51 @@ jobs: '([ .[] | select(.number == $pr) | .labels ] | .[0]) // []' open_prs.json 2>/dev/null) self_prio=$(printf '%s' "${self_labels:-[]}" | prio_of) case "$self_prio" in ''|*[!0-9]*) self_prio=5 ;; esac - echo "This PR #$SELF_PR has effective priority P$self_prio (P0 = highest, P9 = lowest)." + if [ "$self_prio" -ge 10 ]; then prio_label="broken (below P9, bottom)"; else prio_label="P$self_prio"; fi + echo "This PR #$SELF_PR effective priority: $prio_label (P0 = highest/emergency, P9 = lowest, 'broken' = bottom)." - if [ "$self_prio" -ge 9 ]; then - echo "P$self_prio is the lowest tier — no strictly-lower-priority PRs to preempt." - exit 0 - fi - - # "numberheadprio" for every OTHER open PR. + # ── PASS 1: PREEMPTION — cancel a strictly-lower OTHER PR's active runs ── + # A strictly-lower-priority (prio > self) OTHER PR's active CI run is + # cancelled iff keeping it running is wasteful, i.e. EITHER: + # • THIS PR is P0 (emergency — reclaim every lower runner now), OR + # • that OTHER PR is `broken` (its run can't merge, so ANY higher-priority + # PR — not just P0 — may reclaim its runner). + # Otherwise (we're P1–P9 and the target isn't broken) we DON'T cancel; we + # only yield to genuinely-higher-priority PRs in PASS 2. + # Emit "numberheadpriobroken(0|1)" for every OTHER open PR + # (broken => effective prio 10, the bottom, overriding any P0–P9 label). jq -r --argjson self "$SELF_PR" ' .[] | select(.number != $self) + | (any(.labels[]; .name == "broken")) as $b | [ .number, .headRefName, - ([ .labels[] | .name | select(test("^P[0-9]$")) | ltrimstr("P") | tonumber ] - | if length == 0 then 5 else min end) ] + (if $b then 10 + else ([ .labels[] | .name | select(test("^P[0-9]$")) | ltrimstr("P") | tonumber ] + | if length == 0 then 5 else min end) end), + (if $b then 1 else 0 end) ] | @tsv' open_prs.json 2>/dev/null > others.tsv cancelled_total=0 - while IFS=$'\t' read -r num head prio; do + while IFS=$'\t' read -r num head prio isbroken; do [ -n "${num:-}" ] || continue case "$prio" in ''|*[!0-9]*) prio=5 ;; esac + [ "$isbroken" = "1" ] || isbroken=0 + # Never touch an equal-or-higher-priority PR — only strictly lower. if [ "$prio" -le "$self_prio" ]; then echo "· PR #$num (P$prio): equal-or-higher priority — left untouched." continue fi - echo "· PR #$num (P$prio, head '$head'): strictly lower priority — checking for active CI runs." - # Active (non-completed) CI runs on that PR's head branch, PR events only. - run_ids=$(gh run list --workflow ci.yml --branch "$head" --event pull_request \ - --limit 100 --json databaseId,status,headBranch,event 2>/dev/null \ - | jq -r '.[] - | select(.event == "pull_request") - | select(.headBranch != "main") - | select(.status != "completed") - | .databaseId' 2>/dev/null) + # Strictly lower, but only a P0 self OR a broken target is preemptible. + if [ "$self_prio" -ne 0 ] && [ "$isbroken" != "1" ]; then + echo "· PR #$num (P$prio): strictly lower, not broken, and we're not P0 — left to run (we don't cancel it)." + continue + fi + + if [ "$self_prio" -eq 0 ]; then reason="P0 emergency"; else reason="target is 'broken'"; fi + tag="P$prio"; [ "$isbroken" = "1" ] && tag="broken" + echo "· PR #$num ($tag, head '$head'): preemptible ($reason) — checking for active CI runs." + run_ids=$(active_run_ids_for_head "$head") if [ -z "$run_ids" ]; then echo " no active CI runs." @@ -137,13 +205,79 @@ jobs: fi done <<< "$run_ids" done < others.tsv + echo "Preemption pass complete — cancelled $cancelled_total run(s)." - echo "Preemption pass complete — cancelled $cancelled_total lower-priority run(s)." + # ── PASS 2: BOUNDED HOLD-BACK — yield to strictly-higher, cancel NOTHING ── + # P0 is top priority: nothing outranks an emergency, so it never yields. + if [ "$self_prio" -eq 0 ]; then + echo "P0 emergency — not yielding; proceeding immediately." + exit 0 + fi + + # Defer this PR's heavy jobs (which `needs: traffic-control`) while any + # strictly-higher-priority OTHER open PR still has an active/queued CI run, + # so those heavy jobs reach the runner queue first. Cancel NOTHING here. + # Bounded by budget; on expiry proceed regardless (never block ourselves, + # never hit the hard job timeout). Fail-open: any API hiccup => stop waiting. + case "$HOLD_BACK_BUDGET_SECONDS" in ''|*[!0-9]*) HOLD_BACK_BUDGET_SECONDS=180 ;; esac + case "$HOLD_BACK_POLL_SECONDS" in ''|*[!0-9]*) HOLD_BACK_POLL_SECONDS=15 ;; esac + deadline=$(( $(date +%s) + HOLD_BACK_BUDGET_SECONDS )) + echo "$prio_label — holding back up to ${HOLD_BACK_BUDGET_SECONDS}s for strictly-higher-priority PRs (no cancellation)." + + while :; do + remaining=$(( deadline - $(date +%s) )) + if [ "$remaining" -le 0 ]; then + echo "Hold-back budget elapsed — proceeding; higher-priority PRs got their head start." + break + fi + + # Refresh so newly opened higher-priority PRs are seen mid-wait. + if ! list_open_prs > open_prs.json 2>err.txt; then + echo "::warning::Could not refresh open PRs — proceeding. $(cat err.txt 2>/dev/null)" + break + fi + + # Strictly-higher-priority OTHER PRs (broken => 10, so a broken PR is never + # higher than a non-broken one): "numberheadprio". + jq -r --argjson self "$SELF_PR" --argjson me "$self_prio" ' + .[] | select(.number != $self) + | { n: .number, h: .headRefName, + p: (if any(.labels[]; .name == "broken") then 10 + else ([ .labels[] | .name | select(test("^P[0-9]$")) | ltrimstr("P") | tonumber ] + | if length == 0 then 5 else min end) end) } + | select(.p < $me) + | [ .n, .h, .p ] | @tsv' open_prs.json 2>/dev/null > higher.tsv + + if [ ! -s higher.tsv ]; then + echo "No strictly-higher-priority open PRs — proceeding." + break + fi + + blockers="" + while IFS=$'\t' read -r num head prio; do + [ -n "${num:-}" ] || continue + if [ -n "$(active_run_ids_for_head "$head")" ]; then + blockers="$blockers #$num(P$prio)" + fi + done < higher.tsv + + if [ -z "$blockers" ]; then + echo "No strictly-higher-priority PR has active CI runs — proceeding." + break + fi + + sleep_s="$HOLD_BACK_POLL_SECONDS" + [ "$remaining" -lt "$sleep_s" ] && sleep_s="$remaining" + echo "Yielding to strictly-higher-priority PR(s) with active CI:${blockers} — re-checking in ${sleep_s}s (${remaining}s budget left)." + [ "$sleep_s" -gt 0 ] && sleep "$sleep_s" + done + + echo "Hold-back complete — this PR's heavy jobs may now start." exit 0 debug-build: name: Debug build - needs: traffic-control # order after runner-priority preemption + needs: traffic-control # order after runner-priority orchestration (P0/broken preempt; P1–P9 hold-back) # x86_64: Linux-arm64 runners can't set up this SDK — android-actions/setup-android's sdkmanager # fails (exit 1) on the android-37.0 preview platform, and the emulator package has no arm64-Linux # build. Build/unit-test results are host-arch-independent anyway (R8/AGP/JVM); real arm64 @@ -180,7 +314,7 @@ jobs: unit-tests: name: Unit tests - needs: traffic-control # order after runner-priority preemption + needs: traffic-control # order after runner-priority orchestration (P0/broken preempt; P1–P9 hold-back) runs-on: ubuntu-latest steps: - name: Check out source @@ -229,7 +363,7 @@ jobs: static-analysis: name: Static analysis - needs: traffic-control # order after runner-priority preemption + needs: traffic-control # order after runner-priority orchestration (P0/broken preempt; P1–P9 hold-back) runs-on: ubuntu-latest steps: - name: Check out source @@ -267,7 +401,7 @@ jobs: e2e: name: E2E - needs: traffic-control # order after runner-priority preemption + needs: traffic-control # order after runner-priority orchestration (P0/broken preempt; P1–P9 hold-back) runs-on: ubuntu-latest strategy: fail-fast: false @@ -378,7 +512,7 @@ jobs: # 37 into the main `e2e` matrix and delete this job. e2e-preview: name: E2E (API 37 preview) - needs: traffic-control # order after runner-priority preemption + needs: traffic-control # order after runner-priority orchestration (P0/broken preempt; P1–P9 hold-back) runs-on: ubuntu-latest timeout-minutes: 35 env: