Hold Pull Request macOS Builds Until a Runner Is Free

A pull request's macOS build now waits in one repo-wide line, from a
Linux job, until the macOS runners every active Build all run holds or
still needs leave two free. A run is let in whole: its first arch takes
both runners and its second goes straight through, since letting arches
in one at a time deadlocks with every runner held by a run waiting for
its other arch. Push, nightly and manual runs reserve two runners until
their macOS work is done, and a pull request run keeps its two until its
universal build and tests finish, so jobs that never pass the line are
accounted for. Queued jobs count as busy. API errors and a four-hour
limit let the build in rather than block it.

Co-authored-by: raistlin7447 <kris.austin@gmail.com>
This commit is contained in:
Hanif Koh
2026-10-11 05:49:13 +08:00
co-authored by raistlin7447
parent d208e45736
commit 9f25c89994
3 changed files with 467 additions and 2 deletions
+197
View File
@@ -0,0 +1,197 @@
#!/usr/bin/env python3
"""Holds a pull request's macOS build until a hosted macOS runner is free for it.
Runs from a Linux job in build_check_cache.yml, once per macOS arch, in a single
repo-wide line (a concurrency group with queue: max), so only the job at the
front of the line polls. A pull request run is let in whole: its first arch waits
until the runners every active Build all run holds or still needs, plus RESERVE
for this run, fit in MACOS_RUNNER_LIMIT, and its second arch then goes straight
through. Letting arches in one at a time would deadlock, with every runner held
by a run waiting for a second one for its other arch.
- A push, nightly or manual run holds RESERVE runners until its macOS work is
done. The jobs API lists only the jobs a run has reached, so what it still
needs cannot be counted and is reserved instead.
- A pull request run that was let in holds RESERVE runners until its macOS work
is done, which covers its later app build, universal build and tests that
never pass through the line.
- A run holds at least the macOS jobs it has queued or running.
- In the minutes around the nightly's cron time, RESERVE runners are held for it
until its run appears.
Any API error repeated FAILURES_BEFORE_ADMIT times, and the WAIT_MINUTES limit,
let the arch in, so a fault here never blocks pull requests.
Environment: GH_TOKEN, REPO, WORKFLOW_REF, GITHUB_RUN_ID, and optionally MACOS_RUNNER_LIMIT,
WAIT_MINUTES, POLL_SECONDS and GITHUB_API_URL.
"""
import datetime
import json
import os
import sys
import time
import urllib.error
import urllib.parse
import urllib.request
# A Build all run builds arm64 and x86_64 at the same time, then the universal
# build and the macOS tests at the same time.
RESERVE = 2
# Must match the job names in build_all.yml and build_check_cache.yml.
GATE_SUFFIX = " / Wait for a macOS runner"
FINAL_JOBS = {"Build macOS Universal", "macOS arm64"}
# Must match the cron in build_all.yml.
NIGHTLY_UTC = datetime.time(2, 15)
NIGHTLY_REPO = "OrcaSlicer/OrcaSlicer"
NIGHTLY_LEAD = datetime.timedelta(minutes=10)
NIGHTLY_GRACE = datetime.timedelta(minutes=45)
ACTIVE = {"queued", "in_progress", "waiting", "pending", "requested"}
FAILURES_BEFORE_ADMIT = 3
def is_macos(job):
return any(label.startswith("macos-") for label in job.get("labels") or [])
def macos_done(jobs):
"""True once the universal build and the macOS tests have finished or been skipped.
Both start together when the arch builds finish."""
# A skipped caller job is listed under its own name, a started one as "<name> / <job>".
final = [job for job in jobs if job["name"].split(" / ")[0] in FINAL_JOBS]
return bool(final) and all(job["status"] == "completed" for job in final)
def admitted(jobs):
"""True once one of the run's arches was let in."""
return any(job["name"].endswith(GATE_SUFFIX) and job["conclusion"] == "success" for job in jobs)
def run_demand(run, jobs):
"""macOS runners a run holds or still needs."""
active = sum(1 for job in jobs if is_macos(job) and job["status"] in ACTIVE)
if macos_done(jobs):
return active
if run["event"] == "pull_request" and not admitted(jobs):
return active
return max(active, RESERVE)
def nightly_window(now):
"""The window around today's nightly cron time, as (start, end) in UTC."""
cron = datetime.datetime.combine(now.date(), NIGHTLY_UTC, tzinfo=datetime.timezone.utc)
return cron - NIGHTLY_LEAD, cron + NIGHTLY_GRACE
def parse_time(value):
return datetime.datetime.fromisoformat(value.replace("Z", "+00:00"))
class Api:
def __init__(self, token, url="https://api.github.com"):
self.token = token
self.url = url.rstrip("/")
def get(self, path, **params):
query = urllib.parse.urlencode(params)
request = urllib.request.Request(f"{self.url}/{path}?{query}", headers={
"Accept": "application/vnd.github+json",
"Authorization": f"Bearer {self.token}",
"X-GitHub-Api-Version": "2022-11-28",
})
with urllib.request.urlopen(request, timeout=30) as response:
return json.load(response)
def list_runs(api, repo, workflow, **params):
runs, page = [], 1
while True:
body = api.get(f"repos/{repo}/actions/workflows/{workflow}/runs",
per_page=100, page=page, **params)
runs += body["workflow_runs"]
if len(runs) >= body["total_count"] or not body["workflow_runs"]:
return runs
page += 1
def latest_run(api, repo, workflow, **params):
runs = api.get(f"repos/{repo}/actions/workflows/{workflow}/runs",
per_page=1, **params)["workflow_runs"]
return runs[0] if runs else None
def measure(api, repo, workflow, run_id, now):
"""Total macOS runners held or needed, one line per run that holds any, and
whether run_id was already let in."""
total, lines, here = 0, [], False
# A run is listed as queued whenever one of its jobs waits for a runner, so a
# queued run can hold runners too. A run can move between the two lists
# between the calls.
runs = {run["id"]: run for status in ("in_progress", "queued")
for run in list_runs(api, repo, workflow, status=status)}
for run in runs.values():
jobs = api.get(f"repos/{repo}/actions/runs/{run['id']}/jobs",
filter="latest", per_page=100)["jobs"]
demand = run_demand(run, jobs)
here = here or (run["id"] == run_id and admitted(jobs))
if demand:
total += demand
lines.append(f" {demand} run {run['id']} ({run['event']}, {run['head_branch']})")
# build_all.yml runs the nightly only in the main repository.
start, end = nightly_window(now)
if repo == NIGHTLY_REPO and start <= now < end:
last = latest_run(api, repo, workflow, event="schedule")
if not last or parse_time(last["created_at"]) < start:
total += RESERVE
lines.append(f" {RESERVE} the nightly, due at {NIGHTLY_UTC:%H:%M} UTC")
return total, lines, here
def wait(measure_now, limit, wait_minutes, poll_seconds,
clock=time.monotonic, sleep=time.sleep, log=print):
"""Polls until this run fits. Returns the reason it was let in."""
deadline = clock() + wait_minutes * 60
failures = 0
while True:
try:
total, lines, here = measure_now()
except (urllib.error.URLError, OSError, ValueError, KeyError, TypeError) as error:
failures += 1
log(f"::warning title=macOS admission::Could not read the queue ({error}).")
if failures >= FAILURES_BEFORE_ADMIT:
return "the queue could not be read"
else:
failures = 0
if here:
return "this run already holds its runners"
log(f"{total} of {limit} macOS runners held or needed:")
for line in lines:
log(line)
if total + RESERVE <= limit:
return "a runner is free"
if clock() + poll_seconds >= deadline:
return f"it waited {wait_minutes} minutes"
sleep(poll_seconds)
def main():
repo = os.environ["REPO"]
workflow = os.environ["WORKFLOW_REF"].split("@")[0].rsplit("/", 1)[-1]
api = Api(os.environ["GH_TOKEN"], os.environ.get("GITHUB_API_URL", "https://api.github.com"))
run_id = int(os.environ["GITHUB_RUN_ID"])
limit = int(os.environ.get("MACOS_RUNNER_LIMIT") or 5)
reason = wait(
lambda: measure(api, repo, workflow, run_id, datetime.datetime.now(datetime.timezone.utc)),
limit,
wait_minutes=int(os.environ.get("WAIT_MINUTES") or 240),
poll_seconds=int(os.environ.get("POLL_SECONDS") or 180),
)
print(f"Letting this macOS build in: {reason}.")
if __name__ == "__main__":
sys.exit(main())
+239
View File
@@ -0,0 +1,239 @@
#!/usr/bin/env python3
"""Tests for scripts/ci_macos_admission.py (stdlib unittest, no external deps).
Run from the repo root: python -m unittest discover -s scripts/tests -v
"""
import datetime
import os
import sys
import unittest
import urllib.error
sys.path.insert(0, os.path.abspath(os.path.join(os.path.dirname(__file__), "..")))
import ci_macos_admission as admission # noqa: E402
UTC = datetime.timezone.utc
NOON = datetime.datetime(2026, 10, 10, 12, 0, tzinfo=UTC)
def job(name, status="completed", conclusion="success", labels=("macos-15",)):
return {"name": name, "status": status, "conclusion": conclusion, "labels": list(labels)}
def gate(arch, status="completed", conclusion="success"):
return job(f"build_macos_arch ({arch}) / Wait for a macOS runner", status, conclusion,
labels=("ubuntu-24.04",))
def build(arch, status="in_progress"):
return job(f"build_macos_arch ({arch}) / Build Deps / Build OrcaSlicer / Build OrcaSlicer",
status, None if status != "completed" else "success")
CHECK_CACHE = job("build_macos_arch (arm64) / Check Cache", labels=("ubuntu-24.04",))
UNIVERSAL_DONE = job("Build macOS Universal / Build OrcaSlicer")
TESTS_DONE = job("macOS arm64 / Unit Tests")
PUSH = {"id": 1, "event": "push", "head_branch": "main", "status": "in_progress"}
PR = {"id": 2, "event": "pull_request", "head_branch": "topic", "status": "in_progress"}
class MacosDoneTest(unittest.TestCase):
def test_not_done_before_the_final_jobs_are_listed(self):
self.assertFalse(admission.macos_done([CHECK_CACHE, build("arm64", "completed")]))
def test_not_done_while_a_final_job_runs(self):
running = job("macOS arm64 / Unit Tests", "in_progress", None)
self.assertFalse(admission.macos_done([UNIVERSAL_DONE, running]))
def test_done_when_the_final_jobs_finished(self):
self.assertTrue(admission.macos_done([build("arm64", "completed"), UNIVERSAL_DONE, TESTS_DONE]))
def test_done_when_the_final_jobs_were_skipped(self):
# A skipped caller job is listed under its own name, without labels.
skipped = [job("Build macOS Universal", conclusion="skipped", labels=()),
job("macOS arm64", conclusion="skipped", labels=())]
self.assertTrue(admission.macos_done(skipped))
class RunDemandTest(unittest.TestCase):
def test_push_reserves_before_its_macos_jobs_are_listed(self):
self.assertEqual(admission.run_demand(PUSH, []), admission.RESERVE)
self.assertEqual(admission.run_demand(PUSH, [CHECK_CACHE]), admission.RESERVE)
def test_push_keeps_its_reserve_between_stages(self):
jobs = [build("arm64", "completed"), build("x86_64", "completed")]
self.assertEqual(admission.run_demand(PUSH, jobs), admission.RESERVE)
def test_push_releases_when_macos_is_done(self):
jobs = [build("arm64", "completed"), UNIVERSAL_DONE, TESTS_DONE]
self.assertEqual(admission.run_demand(PUSH, jobs), 0)
def test_queued_jobs_count_as_busy(self):
jobs = [build("arm64", "queued")]
self.assertEqual(admission.run_demand(PR, jobs), 1)
def test_pull_request_holds_nothing_before_it_is_let_in(self):
self.assertEqual(admission.run_demand(PR, [CHECK_CACHE, gate("arm64", "in_progress", None)]), 0)
def test_pull_request_holds_both_runners_once_let_in(self):
# The other arch may still be waiting for deps, and goes straight through.
self.assertEqual(admission.run_demand(PR, [gate("arm64"), gate("x86_64", "in_progress", None)]),
admission.RESERVE)
def test_pull_request_keeps_its_runners_for_its_later_jobs(self):
# Both arch builds are done and the universal build and tests are not
# listed yet: they will need both runners without passing the line.
jobs = [gate("arm64"), gate("x86_64"), build("arm64", "completed"), build("x86_64", "completed")]
self.assertEqual(admission.run_demand(PR, jobs), 2)
def test_pull_request_releases_when_macos_is_done(self):
jobs = [gate("arm64"), gate("x86_64"), UNIVERSAL_DONE, TESTS_DONE]
self.assertEqual(admission.run_demand(PR, jobs), 0)
def test_pull_request_without_the_line_holds_what_it_runs(self):
# A run started from a workflow without the line.
self.assertEqual(admission.run_demand(PR, [build("arm64"), build("x86_64", "queued")]), 2)
def test_finished_and_non_macos_jobs_are_not_counted(self):
jobs = [build("arm64", "completed"), CHECK_CACHE,
job("build_linux (ubuntu-24.04) / Check Cache", "in_progress", None, ("ubuntu-24.04",))]
self.assertEqual(admission.run_demand(PR, jobs), 0)
class FakeApi:
"""Answers GET requests from a dict of path -> list of pages (or one body)."""
def __init__(self, responses):
self.responses = responses
self.calls = []
def get(self, path, **params):
self.calls.append((path, params))
if path.endswith("/runs") and "status" in params:
runs = self.responses.get(("runs", params["status"]), [])
start = (params["page"] - 1) * params["per_page"]
return {"total_count": len(runs), "workflow_runs": runs[start:start + params["per_page"]]}
if path.endswith("/runs"):
return {"workflow_runs": self.responses.get(("runs", params.get("event")), [])[:1]}
return {"jobs": self.responses[path]}
def jobs_path(run_id):
return f"repos/o/r/actions/runs/{run_id}/jobs"
class MeasureTest(unittest.TestCase):
def test_sums_in_progress_and_queued_runs(self):
queued_pr = dict(PR, id=3, status="queued")
api = FakeApi({
("runs", "in_progress"): [PUSH, PR],
("runs", "queued"): [queued_pr],
jobs_path(1): [CHECK_CACHE],
jobs_path(2): [gate("arm64"), build("arm64")],
jobs_path(3): [gate("arm64"), gate("x86_64"), build("arm64", "queued")],
})
total, lines, here = admission.measure(api, "o/r", "build_all.yml", 99, NOON)
self.assertEqual(total, 3 * admission.RESERVE)
self.assertEqual(len(lines), 3)
self.assertFalse(here)
def test_reads_every_page_of_runs(self):
runs = [dict(PR, id=i) for i in range(150)]
responses = {("runs", "in_progress"): runs}
responses.update({jobs_path(i): [gate("arm64")] for i in range(150)})
total, _, _ = admission.measure(FakeApi(responses), "o/r", "build_all.yml", 999, NOON)
self.assertEqual(total, 150 * admission.RESERVE)
def test_reserves_for_the_nightly_until_its_run_appears(self):
due = datetime.datetime(2026, 10, 10, 2, 10, tzinfo=UTC)
yesterday = {"created_at": "2026-10-09T02:20:00Z"}
total, lines, _ = admission.measure(
FakeApi({("runs", "schedule"): [yesterday]}), "OrcaSlicer/OrcaSlicer", "build_all.yml", 99, due)
self.assertEqual(total, admission.RESERVE)
self.assertIn("nightly", lines[0])
def test_no_nightly_reserve_once_its_run_exists(self):
due = datetime.datetime(2026, 10, 10, 2, 30, tzinfo=UTC)
today = {"created_at": "2026-10-10T02:21:00Z"}
total, _, _ = admission.measure(
FakeApi({("runs", "schedule"): [today]}), "OrcaSlicer/OrcaSlicer", "build_all.yml", 99, due)
self.assertEqual(total, 0)
def test_no_nightly_reserve_in_a_fork(self):
due = datetime.datetime(2026, 10, 10, 2, 10, tzinfo=UTC)
total, _, _ = admission.measure(FakeApi({}), "fork/OrcaSlicer", "build_all.yml", 99, due)
self.assertEqual(total, 0)
def test_counts_a_run_in_both_lists_once(self):
api = FakeApi({
("runs", "in_progress"): [PUSH],
("runs", "queued"): [dict(PUSH, status="queued")],
jobs_path(1): [],
})
total, _, _ = admission.measure(api, "o/r", "build_all.yml", 99, NOON)
self.assertEqual(total, admission.RESERVE)
def test_knows_when_its_own_run_was_let_in(self):
# Every runner is held by runs that were let in whole, so none waits for
# a runner for its second arch.
responses = {("runs", "in_progress"): [dict(PR, id=i) for i in (1, 2)]}
responses.update({jobs_path(i): [gate("arm64"), gate("x86_64", "in_progress", None)] for i in (1, 2)})
total, _, here = admission.measure(FakeApi(responses), "o/r", "build_all.yml", 2, NOON)
self.assertEqual(total, 2 * admission.RESERVE)
self.assertTrue(here)
def test_no_nightly_lookup_outside_its_window(self):
api = FakeApi({})
admission.measure(api, "o/r", "build_all.yml", 99, NOON)
self.assertFalse(any("event" in params for _, params in api.calls))
class WaitTest(unittest.TestCase):
def run_wait(self, results, limit=5, wait_minutes=60, poll_seconds=120):
results = iter(results)
now = [0.0]
def measure_now():
result = next(results)
if isinstance(result, Exception):
raise result
if result == "here":
return 99, [], True
return result, [], False
def sleep(seconds):
now[0] += seconds
reason = admission.wait(measure_now, limit, wait_minutes, poll_seconds,
clock=lambda: now[0], sleep=sleep, log=lambda _: None)
return reason, now[0]
def test_lets_in_when_both_runners_fit(self):
self.assertEqual(self.run_wait([3]), ("a runner is free", 0))
def test_waits_while_full(self):
self.assertEqual(self.run_wait([4, 5, 3]), ("a runner is free", 240))
def test_second_arch_goes_straight_through(self):
self.assertEqual(self.run_wait(["here"]), ("this run already holds its runners", 0))
def test_lets_in_after_repeated_api_errors(self):
error = urllib.error.URLError("rate limited")
reason, _ = self.run_wait([error, error, error])
self.assertEqual(reason, "the queue could not be read")
def test_a_good_read_resets_the_error_count(self):
error = urllib.error.URLError("rate limited")
reason, _ = self.run_wait([error, error, 5, error, error, 3])
self.assertEqual(reason, "a runner is free")
def test_lets_in_at_the_time_limit(self):
reason, elapsed = self.run_wait([5] * 100, wait_minutes=10, poll_seconds=120)
self.assertEqual(reason, "it waited 10 minutes")
self.assertLess(elapsed, 600)
if __name__ == "__main__":
unittest.main()