Repository navigation
Auto-retry stalled runs #3206
This file contains hidden or bidirectional Unicode text that may be interpreted or compiled differently than what appears below. To review, open the file in an editor that reveals hidden Unicode characters.
Learn more about bidirectional Unicode characters
| # Watchdog: auto-retry runs killed by runner stalls. | |
| # | |
| # Self-hosted runners occasionally drop communication mid-job — the build itself is | |
| # green, but the job dies at a random later step (e.g. "Logs:", "Upload to servers") | |
| # and the whole run is marked failed. Those stalls hit different stages on different | |
| # runners, so no per-step hardening fixes them; the only robust recovery is a run-level | |
| # re-run of just the failed jobs. | |
| # | |
| # This fires when a build run finishes 'failure' and re-runs ONLY its failed jobs, once. | |
| # Guards: | |
| # * run_attempt < 10 -> up to 10 attempts total. GitHub-side flakiness comes in | |
| # bursts, so give it some budget before a run is left red; | |
| # a genuine failure still surfaces once the budget's spent. | |
| # * > 50% of jobs green -> a minority of failures against a mostly-green run looks | |
| # like stalls -> retry. If half or more of the run's jobs | |
| # failed it looks systemic (bad commit / infra outage) -> do | |
| # NOT mass-retry; leave it red and warn for a human. | |
| # | |
| # Note: we wait for the run to finish rather than cancelling it mid-flight. The build | |
| # chunks are fail-fast:false, so the good jobs complete and produce artifacts even while | |
| # a few stall; cancelling would discard that work and turn every in-progress/queued job | |
| # into 'cancelled', which "re-run failed" would then restart wholesale. | |
| name: "Auto-retry stalled runs" | |
| on: | |
| workflow_run: | |
| workflows: | |
| - "Build Nightly Images" | |
| - "Build All Artifacts" | |
| - "Build All Stable Artifacts" | |
| - "Build Apps Images" | |
| - "Build Community Images" | |
| - "Build Standard Support Images" | |
| types: [ completed ] | |
| # Manual trigger: re-run the failed jobs of a specific run on demand. | |
| workflow_dispatch: | |
| inputs: | |
| run_id: | |
| description: "Run ID to re-run failed jobs for (the numeric id from the run URL)" | |
| type: string | |
| required: true | |
| ignore_threshold: | |
| description: "Re-run even if more than the stall threshold failed (force)" | |
| type: boolean | |
| default: false | |
| permissions: | |
| actions: write # required to re-run jobs | |
| jobs: | |
| retry: | |
| name: "Re-run failed jobs (stall recovery)" | |
| # Auto: only on a build run that failed, while retry budget is left (so a real | |
| # failure still surfaces once spent). Manual: always runs (the operator chose this run). | |
| if: ${{ github.event_name == 'workflow_dispatch' || (github.event.workflow_run.conclusion == 'failure' && github.event.workflow_run.run_attempt < 10) }} | |
| runs-on: ubuntu-latest | |
| env: | |
| GH_TOKEN: ${{ secrets.ACCESS_TOKEN || github.token }} | |
| # Auto-retry only when the run is mostly green: more than this percentage of | |
| # jobs must have succeeded. At or below it, the failures look systemic rather | |
| # than like runner stalls, so leave it red (raise/lower to taste). | |
| MIN_SUCCESS_PERCENT: "50" | |
| steps: | |
| - name: "Re-run failed jobs if the failure count looks like stalls" | |
| run: | | |
| set -euo pipefail | |
| # Resolve the target run + whether to bypass the threshold, per trigger. | |
| if [ "${{ github.event_name }}" = "workflow_dispatch" ]; then | |
| RUN_ID="${{ inputs.run_id }}" | |
| IGNORE_THRESHOLD="${{ inputs.ignore_threshold }}" | |
| echo "Manual dispatch -> run ${RUN_ID} (ignore_threshold=${IGNORE_THRESHOLD})." | |
| else | |
| RUN_ID="${{ github.event.workflow_run.id }}" | |
| IGNORE_THRESHOLD="false" | |
| echo "Auto -> run '${{ github.event.workflow_run.name }}' #${RUN_ID} (attempt ${{ github.event.workflow_run.run_attempt }})." | |
| fi | |
| # Count total and failed jobs in the latest attempt (paginate; 200+ jobs | |
| # possible). One object per page, then slurp+sum across pages. | |
| jobs_json="$(gh api --paginate "repos/${GITHUB_REPOSITORY}/actions/runs/${RUN_ID}/jobs" \ | |
| --jq '{total: (.jobs | length), failed: ([.jobs[] | select(.conclusion == "failure")] | length)}')" | |
| total="$(jq -s '[.[].total] | add // 0' <<<"$jobs_json")" | |
| failed="$(jq -s '[.[].failed] | add // 0' <<<"$jobs_json")" | |
| succeeded=$(( total - failed )) | |
| echo "Run ${RUN_ID} -> ${succeeded}/${total} jobs succeeded, ${failed} failed." | |
| if [ "$failed" -eq 0 ]; then | |
| echo "No failed jobs — nothing to re-run." | |
| exit 0 | |
| fi | |
| # Systemic guard: only retry when the run is mostly green — strictly more | |
| # than MIN_SUCCESS_PERCENT% of jobs succeeded. Integer math (no bc): | |
| # succeeded*100 > MIN_SUCCESS_PERCENT*total -> retry. | |
| if [ "$IGNORE_THRESHOLD" != "true" ] && [ "$total" -gt 0 ] \ | |
| && [ $(( succeeded * 100 )) -le $(( MIN_SUCCESS_PERCENT * total )) ]; then | |
| pct=$(( succeeded * 100 / total )) | |
| echo "::warning title=Not auto-retrying::only ${pct}% of jobs succeeded (${succeeded}/${total}); need > ${MIN_SUCCESS_PERCENT}%. Looks systemic, not runner stalls. Leaving it red — re-run manually with ignore_threshold=true to force. Inspect run ${RUN_ID}." | |
| exit 0 | |
| fi | |
| echo "Mostly green (${succeeded}/${total}) — re-running ${failed} failed job(s)." | |
| gh run rerun "${RUN_ID}" --failed --repo "${GITHUB_REPOSITORY}" |