From 9f25c89994600c32e21443fb1e5968a577a407bf Mon Sep 17 00:00:00 2001 From: Hanif Koh Date: Sat, 10 Oct 2026 10:25:19 +0800 Subject: [PATCH] Hold Pull Request macOS Builds Until a Runner Is Free A pull request's macOS build now waits in one repo-wide line, from a Linux job, until the macOS runners every active Build all run holds or still needs leave two free. A run is let in whole: its first arch takes both runners and its second goes straight through, since letting arches in one at a time deadlocks with every runner held by a run waiting for its other arch. Push, nightly and manual runs reserve two runners until their macOS work is done, and a pull request run keeps its two until its universal build and tests finish, so jobs that never pass the line are accounted for. Queued jobs count as busy. API errors and a four-hour limit let the build in rather than block it. Co-authored-by: raistlin7447 --- .github/workflows/build_check_cache.yml | 33 +++- scripts/ci_macos_admission.py | 197 +++++++++++++++++++ scripts/tests/test_ci_macos_admission.py | 239 +++++++++++++++++++++++ 3 files changed, 467 insertions(+), 2 deletions(-) create mode 100644 scripts/ci_macos_admission.py create mode 100644 scripts/tests/test_ci_macos_admission.py diff --git a/.github/workflows/build_check_cache.yml b/.github/workflows/build_check_cache.yml index 46244dcb23..9ca270ba6d 100644 --- a/.github/workflows/build_check_cache.yml +++ b/.github/workflows/build_check_cache.yml @@ -131,10 +131,39 @@ jobs: key: ${{ needs.check_cache.outputs.cache-key }} lookup-only: true + # Hosted macOS runners are scarce, so a pull request's macOS build waits until + # one is free that push and nightly builds do not need. See the script. + macos_admission: + name: Wait for a macOS runner + needs: [check_cache, wait_for_deps] + if: ${{ !cancelled() && needs.check_cache.result == 'success' && startsWith(inputs.os, 'macos-') && github.event_name == 'pull_request' }} + runs-on: ubuntu-24.04 + env: + WAIT_MINUTES: 240 + # Must exceed WAIT_MINUTES. + timeout-minutes: 250 + # One line for every arch of every pull request, and only its first job polls. + concurrency: + group: macos-admission + queue: max + steps: + - name: Checkout + uses: actions/checkout@v7 + with: + sparse-checkout: scripts/ci_macos_admission.py + sparse-checkout-cone-mode: false + + - name: wait for a macOS runner + env: + GH_TOKEN: ${{ github.token }} + MACOS_RUNNER_LIMIT: ${{ vars.MACOS_RUNNER_LIMIT }} + run: python3 scripts/ci_macos_admission.py + build_deps: # call next step name: Build Deps - needs: [check_cache, wait_for_deps] - # A failed wait counts as no cache, so the deps are built here. + needs: [check_cache, wait_for_deps, macos_admission] + # A failed wait counts as no cache, so the deps are built here. A failed + # admission still builds rather than skip macOS. if: ${{ !cancelled() && needs.check_cache.result == 'success' }} uses: ./.github/workflows/build_deps.yml with: diff --git a/scripts/ci_macos_admission.py b/scripts/ci_macos_admission.py new file mode 100644 index 0000000000..278ca3bc65 --- /dev/null +++ b/scripts/ci_macos_admission.py @@ -0,0 +1,197 @@ +#!/usr/bin/env python3 +"""Holds a pull request's macOS build until a hosted macOS runner is free for it. + +Runs from a Linux job in build_check_cache.yml, once per macOS arch, in a single +repo-wide line (a concurrency group with queue: max), so only the job at the +front of the line polls. A pull request run is let in whole: its first arch waits +until the runners every active Build all run holds or still needs, plus RESERVE +for this run, fit in MACOS_RUNNER_LIMIT, and its second arch then goes straight +through. Letting arches in one at a time would deadlock, with every runner held +by a run waiting for a second one for its other arch. + +- A push, nightly or manual run holds RESERVE runners until its macOS work is + done. The jobs API lists only the jobs a run has reached, so what it still + needs cannot be counted and is reserved instead. +- A pull request run that was let in holds RESERVE runners until its macOS work + is done, which covers its later app build, universal build and tests that + never pass through the line. +- A run holds at least the macOS jobs it has queued or running. +- In the minutes around the nightly's cron time, RESERVE runners are held for it + until its run appears. + +Any API error repeated FAILURES_BEFORE_ADMIT times, and the WAIT_MINUTES limit, +let the arch in, so a fault here never blocks pull requests. + +Environment: GH_TOKEN, REPO, WORKFLOW_REF, GITHUB_RUN_ID, and optionally MACOS_RUNNER_LIMIT, +WAIT_MINUTES, POLL_SECONDS and GITHUB_API_URL. +""" + +import datetime +import json +import os +import sys +import time +import urllib.error +import urllib.parse +import urllib.request + +# A Build all run builds arm64 and x86_64 at the same time, then the universal +# build and the macOS tests at the same time. +RESERVE = 2 + +# Must match the job names in build_all.yml and build_check_cache.yml. +GATE_SUFFIX = " / Wait for a macOS runner" +FINAL_JOBS = {"Build macOS Universal", "macOS arm64"} +# Must match the cron in build_all.yml. +NIGHTLY_UTC = datetime.time(2, 15) +NIGHTLY_REPO = "OrcaSlicer/OrcaSlicer" +NIGHTLY_LEAD = datetime.timedelta(minutes=10) +NIGHTLY_GRACE = datetime.timedelta(minutes=45) + +ACTIVE = {"queued", "in_progress", "waiting", "pending", "requested"} +FAILURES_BEFORE_ADMIT = 3 + + +def is_macos(job): + return any(label.startswith("macos-") for label in job.get("labels") or []) + + +def macos_done(jobs): + """True once the universal build and the macOS tests have finished or been skipped. + Both start together when the arch builds finish.""" + # A skipped caller job is listed under its own name, a started one as " / ". + final = [job for job in jobs if job["name"].split(" / ")[0] in FINAL_JOBS] + return bool(final) and all(job["status"] == "completed" for job in final) + + +def admitted(jobs): + """True once one of the run's arches was let in.""" + return any(job["name"].endswith(GATE_SUFFIX) and job["conclusion"] == "success" for job in jobs) + + +def run_demand(run, jobs): + """macOS runners a run holds or still needs.""" + active = sum(1 for job in jobs if is_macos(job) and job["status"] in ACTIVE) + if macos_done(jobs): + return active + if run["event"] == "pull_request" and not admitted(jobs): + return active + return max(active, RESERVE) + + + +def nightly_window(now): + """The window around today's nightly cron time, as (start, end) in UTC.""" + cron = datetime.datetime.combine(now.date(), NIGHTLY_UTC, tzinfo=datetime.timezone.utc) + return cron - NIGHTLY_LEAD, cron + NIGHTLY_GRACE + + +def parse_time(value): + return datetime.datetime.fromisoformat(value.replace("Z", "+00:00")) + + +class Api: + def __init__(self, token, url="https://api.github.com"): + self.token = token + self.url = url.rstrip("/") + + def get(self, path, **params): + query = urllib.parse.urlencode(params) + request = urllib.request.Request(f"{self.url}/{path}?{query}", headers={ + "Accept": "application/vnd.github+json", + "Authorization": f"Bearer {self.token}", + "X-GitHub-Api-Version": "2022-11-28", + }) + with urllib.request.urlopen(request, timeout=30) as response: + return json.load(response) + + +def list_runs(api, repo, workflow, **params): + runs, page = [], 1 + while True: + body = api.get(f"repos/{repo}/actions/workflows/{workflow}/runs", + per_page=100, page=page, **params) + runs += body["workflow_runs"] + if len(runs) >= body["total_count"] or not body["workflow_runs"]: + return runs + page += 1 + + +def latest_run(api, repo, workflow, **params): + runs = api.get(f"repos/{repo}/actions/workflows/{workflow}/runs", + per_page=1, **params)["workflow_runs"] + return runs[0] if runs else None + + +def measure(api, repo, workflow, run_id, now): + """Total macOS runners held or needed, one line per run that holds any, and + whether run_id was already let in.""" + total, lines, here = 0, [], False + # A run is listed as queued whenever one of its jobs waits for a runner, so a + # queued run can hold runners too. A run can move between the two lists + # between the calls. + runs = {run["id"]: run for status in ("in_progress", "queued") + for run in list_runs(api, repo, workflow, status=status)} + for run in runs.values(): + jobs = api.get(f"repos/{repo}/actions/runs/{run['id']}/jobs", + filter="latest", per_page=100)["jobs"] + demand = run_demand(run, jobs) + here = here or (run["id"] == run_id and admitted(jobs)) + if demand: + total += demand + lines.append(f" {demand} run {run['id']} ({run['event']}, {run['head_branch']})") + + # build_all.yml runs the nightly only in the main repository. + start, end = nightly_window(now) + if repo == NIGHTLY_REPO and start <= now < end: + last = latest_run(api, repo, workflow, event="schedule") + if not last or parse_time(last["created_at"]) < start: + total += RESERVE + lines.append(f" {RESERVE} the nightly, due at {NIGHTLY_UTC:%H:%M} UTC") + return total, lines, here + + +def wait(measure_now, limit, wait_minutes, poll_seconds, + clock=time.monotonic, sleep=time.sleep, log=print): + """Polls until this run fits. Returns the reason it was let in.""" + deadline = clock() + wait_minutes * 60 + failures = 0 + while True: + try: + total, lines, here = measure_now() + except (urllib.error.URLError, OSError, ValueError, KeyError, TypeError) as error: + failures += 1 + log(f"::warning title=macOS admission::Could not read the queue ({error}).") + if failures >= FAILURES_BEFORE_ADMIT: + return "the queue could not be read" + else: + failures = 0 + if here: + return "this run already holds its runners" + log(f"{total} of {limit} macOS runners held or needed:") + for line in lines: + log(line) + if total + RESERVE <= limit: + return "a runner is free" + if clock() + poll_seconds >= deadline: + return f"it waited {wait_minutes} minutes" + sleep(poll_seconds) + + +def main(): + repo = os.environ["REPO"] + workflow = os.environ["WORKFLOW_REF"].split("@")[0].rsplit("/", 1)[-1] + api = Api(os.environ["GH_TOKEN"], os.environ.get("GITHUB_API_URL", "https://api.github.com")) + run_id = int(os.environ["GITHUB_RUN_ID"]) + limit = int(os.environ.get("MACOS_RUNNER_LIMIT") or 5) + reason = wait( + lambda: measure(api, repo, workflow, run_id, datetime.datetime.now(datetime.timezone.utc)), + limit, + wait_minutes=int(os.environ.get("WAIT_MINUTES") or 240), + poll_seconds=int(os.environ.get("POLL_SECONDS") or 180), + ) + print(f"Letting this macOS build in: {reason}.") + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/scripts/tests/test_ci_macos_admission.py b/scripts/tests/test_ci_macos_admission.py new file mode 100644 index 0000000000..ec70267b48 --- /dev/null +++ b/scripts/tests/test_ci_macos_admission.py @@ -0,0 +1,239 @@ +#!/usr/bin/env python3 +"""Tests for scripts/ci_macos_admission.py (stdlib unittest, no external deps). + +Run from the repo root: python -m unittest discover -s scripts/tests -v +""" + +import datetime +import os +import sys +import unittest +import urllib.error + +sys.path.insert(0, os.path.abspath(os.path.join(os.path.dirname(__file__), ".."))) + +import ci_macos_admission as admission # noqa: E402 + +UTC = datetime.timezone.utc +NOON = datetime.datetime(2026, 10, 10, 12, 0, tzinfo=UTC) + + +def job(name, status="completed", conclusion="success", labels=("macos-15",)): + return {"name": name, "status": status, "conclusion": conclusion, "labels": list(labels)} + + +def gate(arch, status="completed", conclusion="success"): + return job(f"build_macos_arch ({arch}) / Wait for a macOS runner", status, conclusion, + labels=("ubuntu-24.04",)) + + +def build(arch, status="in_progress"): + return job(f"build_macos_arch ({arch}) / Build Deps / Build OrcaSlicer / Build OrcaSlicer", + status, None if status != "completed" else "success") + + +CHECK_CACHE = job("build_macos_arch (arm64) / Check Cache", labels=("ubuntu-24.04",)) +UNIVERSAL_DONE = job("Build macOS Universal / Build OrcaSlicer") +TESTS_DONE = job("macOS arm64 / Unit Tests") +PUSH = {"id": 1, "event": "push", "head_branch": "main", "status": "in_progress"} +PR = {"id": 2, "event": "pull_request", "head_branch": "topic", "status": "in_progress"} + + +class MacosDoneTest(unittest.TestCase): + def test_not_done_before_the_final_jobs_are_listed(self): + self.assertFalse(admission.macos_done([CHECK_CACHE, build("arm64", "completed")])) + + def test_not_done_while_a_final_job_runs(self): + running = job("macOS arm64 / Unit Tests", "in_progress", None) + self.assertFalse(admission.macos_done([UNIVERSAL_DONE, running])) + + def test_done_when_the_final_jobs_finished(self): + self.assertTrue(admission.macos_done([build("arm64", "completed"), UNIVERSAL_DONE, TESTS_DONE])) + + def test_done_when_the_final_jobs_were_skipped(self): + # A skipped caller job is listed under its own name, without labels. + skipped = [job("Build macOS Universal", conclusion="skipped", labels=()), + job("macOS arm64", conclusion="skipped", labels=())] + self.assertTrue(admission.macos_done(skipped)) + + +class RunDemandTest(unittest.TestCase): + def test_push_reserves_before_its_macos_jobs_are_listed(self): + self.assertEqual(admission.run_demand(PUSH, []), admission.RESERVE) + self.assertEqual(admission.run_demand(PUSH, [CHECK_CACHE]), admission.RESERVE) + + def test_push_keeps_its_reserve_between_stages(self): + jobs = [build("arm64", "completed"), build("x86_64", "completed")] + self.assertEqual(admission.run_demand(PUSH, jobs), admission.RESERVE) + + def test_push_releases_when_macos_is_done(self): + jobs = [build("arm64", "completed"), UNIVERSAL_DONE, TESTS_DONE] + self.assertEqual(admission.run_demand(PUSH, jobs), 0) + + def test_queued_jobs_count_as_busy(self): + jobs = [build("arm64", "queued")] + self.assertEqual(admission.run_demand(PR, jobs), 1) + + def test_pull_request_holds_nothing_before_it_is_let_in(self): + self.assertEqual(admission.run_demand(PR, [CHECK_CACHE, gate("arm64", "in_progress", None)]), 0) + + def test_pull_request_holds_both_runners_once_let_in(self): + # The other arch may still be waiting for deps, and goes straight through. + self.assertEqual(admission.run_demand(PR, [gate("arm64"), gate("x86_64", "in_progress", None)]), + admission.RESERVE) + + def test_pull_request_keeps_its_runners_for_its_later_jobs(self): + # Both arch builds are done and the universal build and tests are not + # listed yet: they will need both runners without passing the line. + jobs = [gate("arm64"), gate("x86_64"), build("arm64", "completed"), build("x86_64", "completed")] + self.assertEqual(admission.run_demand(PR, jobs), 2) + + def test_pull_request_releases_when_macos_is_done(self): + jobs = [gate("arm64"), gate("x86_64"), UNIVERSAL_DONE, TESTS_DONE] + self.assertEqual(admission.run_demand(PR, jobs), 0) + + def test_pull_request_without_the_line_holds_what_it_runs(self): + # A run started from a workflow without the line. + self.assertEqual(admission.run_demand(PR, [build("arm64"), build("x86_64", "queued")]), 2) + + def test_finished_and_non_macos_jobs_are_not_counted(self): + jobs = [build("arm64", "completed"), CHECK_CACHE, + job("build_linux (ubuntu-24.04) / Check Cache", "in_progress", None, ("ubuntu-24.04",))] + self.assertEqual(admission.run_demand(PR, jobs), 0) + + +class FakeApi: + """Answers GET requests from a dict of path -> list of pages (or one body).""" + + def __init__(self, responses): + self.responses = responses + self.calls = [] + + def get(self, path, **params): + self.calls.append((path, params)) + if path.endswith("/runs") and "status" in params: + runs = self.responses.get(("runs", params["status"]), []) + start = (params["page"] - 1) * params["per_page"] + return {"total_count": len(runs), "workflow_runs": runs[start:start + params["per_page"]]} + if path.endswith("/runs"): + return {"workflow_runs": self.responses.get(("runs", params.get("event")), [])[:1]} + return {"jobs": self.responses[path]} + + +def jobs_path(run_id): + return f"repos/o/r/actions/runs/{run_id}/jobs" + + +class MeasureTest(unittest.TestCase): + def test_sums_in_progress_and_queued_runs(self): + queued_pr = dict(PR, id=3, status="queued") + api = FakeApi({ + ("runs", "in_progress"): [PUSH, PR], + ("runs", "queued"): [queued_pr], + jobs_path(1): [CHECK_CACHE], + jobs_path(2): [gate("arm64"), build("arm64")], + jobs_path(3): [gate("arm64"), gate("x86_64"), build("arm64", "queued")], + }) + total, lines, here = admission.measure(api, "o/r", "build_all.yml", 99, NOON) + self.assertEqual(total, 3 * admission.RESERVE) + self.assertEqual(len(lines), 3) + self.assertFalse(here) + + def test_reads_every_page_of_runs(self): + runs = [dict(PR, id=i) for i in range(150)] + responses = {("runs", "in_progress"): runs} + responses.update({jobs_path(i): [gate("arm64")] for i in range(150)}) + total, _, _ = admission.measure(FakeApi(responses), "o/r", "build_all.yml", 999, NOON) + self.assertEqual(total, 150 * admission.RESERVE) + + def test_reserves_for_the_nightly_until_its_run_appears(self): + due = datetime.datetime(2026, 10, 10, 2, 10, tzinfo=UTC) + yesterday = {"created_at": "2026-10-09T02:20:00Z"} + total, lines, _ = admission.measure( + FakeApi({("runs", "schedule"): [yesterday]}), "OrcaSlicer/OrcaSlicer", "build_all.yml", 99, due) + self.assertEqual(total, admission.RESERVE) + self.assertIn("nightly", lines[0]) + + def test_no_nightly_reserve_once_its_run_exists(self): + due = datetime.datetime(2026, 10, 10, 2, 30, tzinfo=UTC) + today = {"created_at": "2026-10-10T02:21:00Z"} + total, _, _ = admission.measure( + FakeApi({("runs", "schedule"): [today]}), "OrcaSlicer/OrcaSlicer", "build_all.yml", 99, due) + self.assertEqual(total, 0) + + def test_no_nightly_reserve_in_a_fork(self): + due = datetime.datetime(2026, 10, 10, 2, 10, tzinfo=UTC) + total, _, _ = admission.measure(FakeApi({}), "fork/OrcaSlicer", "build_all.yml", 99, due) + self.assertEqual(total, 0) + + def test_counts_a_run_in_both_lists_once(self): + api = FakeApi({ + ("runs", "in_progress"): [PUSH], + ("runs", "queued"): [dict(PUSH, status="queued")], + jobs_path(1): [], + }) + total, _, _ = admission.measure(api, "o/r", "build_all.yml", 99, NOON) + self.assertEqual(total, admission.RESERVE) + + def test_knows_when_its_own_run_was_let_in(self): + # Every runner is held by runs that were let in whole, so none waits for + # a runner for its second arch. + responses = {("runs", "in_progress"): [dict(PR, id=i) for i in (1, 2)]} + responses.update({jobs_path(i): [gate("arm64"), gate("x86_64", "in_progress", None)] for i in (1, 2)}) + total, _, here = admission.measure(FakeApi(responses), "o/r", "build_all.yml", 2, NOON) + self.assertEqual(total, 2 * admission.RESERVE) + self.assertTrue(here) + + def test_no_nightly_lookup_outside_its_window(self): + api = FakeApi({}) + admission.measure(api, "o/r", "build_all.yml", 99, NOON) + self.assertFalse(any("event" in params for _, params in api.calls)) + + +class WaitTest(unittest.TestCase): + def run_wait(self, results, limit=5, wait_minutes=60, poll_seconds=120): + results = iter(results) + now = [0.0] + + def measure_now(): + result = next(results) + if isinstance(result, Exception): + raise result + if result == "here": + return 99, [], True + return result, [], False + + def sleep(seconds): + now[0] += seconds + + reason = admission.wait(measure_now, limit, wait_minutes, poll_seconds, + clock=lambda: now[0], sleep=sleep, log=lambda _: None) + return reason, now[0] + + def test_lets_in_when_both_runners_fit(self): + self.assertEqual(self.run_wait([3]), ("a runner is free", 0)) + + def test_waits_while_full(self): + self.assertEqual(self.run_wait([4, 5, 3]), ("a runner is free", 240)) + + def test_second_arch_goes_straight_through(self): + self.assertEqual(self.run_wait(["here"]), ("this run already holds its runners", 0)) + + def test_lets_in_after_repeated_api_errors(self): + error = urllib.error.URLError("rate limited") + reason, _ = self.run_wait([error, error, error]) + self.assertEqual(reason, "the queue could not be read") + + def test_a_good_read_resets_the_error_count(self): + error = urllib.error.URLError("rate limited") + reason, _ = self.run_wait([error, error, 5, error, error, 3]) + self.assertEqual(reason, "a runner is free") + + def test_lets_in_at_the_time_limit(self): + reason, elapsed = self.run_wait([5] * 100, wait_minutes=10, poll_seconds=120) + self.assertEqual(reason, "it waited 10 minutes") + self.assertLess(elapsed, 600) + + +if __name__ == "__main__": + unittest.main()