diff --git a/.github/workflows/build_check_cache.yml b/.github/workflows/build_check_cache.yml index 46244dcb23..9ca270ba6d 100644 --- a/.github/workflows/build_check_cache.yml +++ b/.github/workflows/build_check_cache.yml @@ -131,10 +131,39 @@ jobs: key: ${{ needs.check_cache.outputs.cache-key }} lookup-only: true + # Hosted macOS runners are scarce, so a pull request's macOS build waits until + # one is free that push and nightly builds do not need. See the script. + macos_admission: + name: Wait for a macOS runner + needs: [check_cache, wait_for_deps] + if: ${{ !cancelled() && needs.check_cache.result == 'success' && startsWith(inputs.os, 'macos-') && github.event_name == 'pull_request' }} + runs-on: ubuntu-24.04 + env: + WAIT_MINUTES: 240 + # Must exceed WAIT_MINUTES. + timeout-minutes: 250 + # One line for every arch of every pull request, and only its first job polls. + concurrency: + group: macos-admission + queue: max + steps: + - name: Checkout + uses: actions/checkout@v7 + with: + sparse-checkout: scripts/ci_macos_admission.py + sparse-checkout-cone-mode: false + + - name: wait for a macOS runner + env: + GH_TOKEN: ${{ github.token }} + MACOS_RUNNER_LIMIT: ${{ vars.MACOS_RUNNER_LIMIT }} + run: python3 scripts/ci_macos_admission.py + build_deps: # call next step name: Build Deps - needs: [check_cache, wait_for_deps] - # A failed wait counts as no cache, so the deps are built here. + needs: [check_cache, wait_for_deps, macos_admission] + # A failed wait counts as no cache, so the deps are built here. A failed + # admission still builds rather than skip macOS. if: ${{ !cancelled() && needs.check_cache.result == 'success' }} uses: ./.github/workflows/build_deps.yml with: diff --git a/scripts/ci_macos_admission.py b/scripts/ci_macos_admission.py new file mode 100644 index 0000000000..278ca3bc65 --- /dev/null +++ b/scripts/ci_macos_admission.py @@ -0,0 +1,197 @@ +#!/usr/bin/env python3 +"""Holds a pull request's macOS build until a hosted macOS runner is free for it. + +Runs from a Linux job in build_check_cache.yml, once per macOS arch, in a single +repo-wide line (a concurrency group with queue: max), so only the job at the +front of the line polls. A pull request run is let in whole: its first arch waits +until the runners every active Build all run holds or still needs, plus RESERVE +for this run, fit in MACOS_RUNNER_LIMIT, and its second arch then goes straight +through. Letting arches in one at a time would deadlock, with every runner held +by a run waiting for a second one for its other arch. + +- A push, nightly or manual run holds RESERVE runners until its macOS work is + done. The jobs API lists only the jobs a run has reached, so what it still + needs cannot be counted and is reserved instead. +- A pull request run that was let in holds RESERVE runners until its macOS work + is done, which covers its later app build, universal build and tests that + never pass through the line. +- A run holds at least the macOS jobs it has queued or running. +- In the minutes around the nightly's cron time, RESERVE runners are held for it + until its run appears. + +Any API error repeated FAILURES_BEFORE_ADMIT times, and the WAIT_MINUTES limit, +let the arch in, so a fault here never blocks pull requests. + +Environment: GH_TOKEN, REPO, WORKFLOW_REF, GITHUB_RUN_ID, and optionally MACOS_RUNNER_LIMIT, +WAIT_MINUTES, POLL_SECONDS and GITHUB_API_URL. +""" + +import datetime +import json +import os +import sys +import time +import urllib.error +import urllib.parse +import urllib.request + +# A Build all run builds arm64 and x86_64 at the same time, then the universal +# build and the macOS tests at the same time. +RESERVE = 2 + +# Must match the job names in build_all.yml and build_check_cache.yml. +GATE_SUFFIX = " / Wait for a macOS runner" +FINAL_JOBS = {"Build macOS Universal", "macOS arm64"} +# Must match the cron in build_all.yml. +NIGHTLY_UTC = datetime.time(2, 15) +NIGHTLY_REPO = "OrcaSlicer/OrcaSlicer" +NIGHTLY_LEAD = datetime.timedelta(minutes=10) +NIGHTLY_GRACE = datetime.timedelta(minutes=45) + +ACTIVE = {"queued", "in_progress", "waiting", "pending", "requested"} +FAILURES_BEFORE_ADMIT = 3 + + +def is_macos(job): + return any(label.startswith("macos-") for label in job.get("labels") or []) + + +def macos_done(jobs): + """True once the universal build and the macOS tests have finished or been skipped. + Both start together when the arch builds finish.""" + # A skipped caller job is listed under its own name, a started one as " / ". + final = [job for job in jobs if job["name"].split(" / ")[0] in FINAL_JOBS] + return bool(final) and all(job["status"] == "completed" for job in final) + + +def admitted(jobs): + """True once one of the run's arches was let in.""" + return any(job["name"].endswith(GATE_SUFFIX) and job["conclusion"] == "success" for job in jobs) + + +def run_demand(run, jobs): + """macOS runners a run holds or still needs.""" + active = sum(1 for job in jobs if is_macos(job) and job["status"] in ACTIVE) + if macos_done(jobs): + return active + if run["event"] == "pull_request" and not admitted(jobs): + return active + return max(active, RESERVE) + + + +def nightly_window(now): + """The window around today's nightly cron time, as (start, end) in UTC.""" + cron = datetime.datetime.combine(now.date(), NIGHTLY_UTC, tzinfo=datetime.timezone.utc) + return cron - NIGHTLY_LEAD, cron + NIGHTLY_GRACE + + +def parse_time(value): + return datetime.datetime.fromisoformat(value.replace("Z", "+00:00")) + + +class Api: + def __init__(self, token, url="https://api.github.com"): + self.token = token + self.url = url.rstrip("/") + + def get(self, path, **params): + query = urllib.parse.urlencode(params) + request = urllib.request.Request(f"{self.url}/{path}?{query}", headers={ + "Accept": "application/vnd.github+json", + "Authorization": f"Bearer {self.token}", + "X-GitHub-Api-Version": "2022-11-28", + }) + with urllib.request.urlopen(request, timeout=30) as response: + return json.load(response) + + +def list_runs(api, repo, workflow, **params): + runs, page = [], 1 + while True: + body = api.get(f"repos/{repo}/actions/workflows/{workflow}/runs", + per_page=100, page=page, **params) + runs += body["workflow_runs"] + if len(runs) >= body["total_count"] or not body["workflow_runs"]: + return runs + page += 1 + + +def latest_run(api, repo, workflow, **params): + runs = api.get(f"repos/{repo}/actions/workflows/{workflow}/runs", + per_page=1, **params)["workflow_runs"] + return runs[0] if runs else None + + +def measure(api, repo, workflow, run_id, now): + """Total macOS runners held or needed, one line per run that holds any, and + whether run_id was already let in.""" + total, lines, here = 0, [], False + # A run is listed as queued whenever one of its jobs waits for a runner, so a + # queued run can hold runners too. A run can move between the two lists + # between the calls. + runs = {run["id"]: run for status in ("in_progress", "queued") + for run in list_runs(api, repo, workflow, status=status)} + for run in runs.values(): + jobs = api.get(f"repos/{repo}/actions/runs/{run['id']}/jobs", + filter="latest", per_page=100)["jobs"] + demand = run_demand(run, jobs) + here = here or (run["id"] == run_id and admitted(jobs)) + if demand: + total += demand + lines.append(f" {demand} run {run['id']} ({run['event']}, {run['head_branch']})") + + # build_all.yml runs the nightly only in the main repository. + start, end = nightly_window(now) + if repo == NIGHTLY_REPO and start <= now < end: + last = latest_run(api, repo, workflow, event="schedule") + if not last or parse_time(last["created_at"]) < start: + total += RESERVE + lines.append(f" {RESERVE} the nightly, due at {NIGHTLY_UTC:%H:%M} UTC") + return total, lines, here + + +def wait(measure_now, limit, wait_minutes, poll_seconds, + clock=time.monotonic, sleep=time.sleep, log=print): + """Polls until this run fits. Returns the reason it was let in.""" + deadline = clock() + wait_minutes * 60 + failures = 0 + while True: + try: + total, lines, here = measure_now() + except (urllib.error.URLError, OSError, ValueError, KeyError, TypeError) as error: + failures += 1 + log(f"::warning title=macOS admission::Could not read the queue ({error}).") + if failures >= FAILURES_BEFORE_ADMIT: + return "the queue could not be read" + else: + failures = 0 + if here: + return "this run already holds its runners" + log(f"{total} of {limit} macOS runners held or needed:") + for line in lines: + log(line) + if total + RESERVE <= limit: + return "a runner is free" + if clock() + poll_seconds >= deadline: + return f"it waited {wait_minutes} minutes" + sleep(poll_seconds) + + +def main(): + repo = os.environ["REPO"] + workflow = os.environ["WORKFLOW_REF"].split("@")[0].rsplit("/", 1)[-1] + api = Api(os.environ["GH_TOKEN"], os.environ.get("GITHUB_API_URL", "https://api.github.com")) + run_id = int(os.environ["GITHUB_RUN_ID"]) + limit = int(os.environ.get("MACOS_RUNNER_LIMIT") or 5) + reason = wait( + lambda: measure(api, repo, workflow, run_id, datetime.datetime.now(datetime.timezone.utc)), + limit, + wait_minutes=int(os.environ.get("WAIT_MINUTES") or 240), + poll_seconds=int(os.environ.get("POLL_SECONDS") or 180), + ) + print(f"Letting this macOS build in: {reason}.") + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/scripts/tests/test_ci_macos_admission.py b/scripts/tests/test_ci_macos_admission.py new file mode 100644 index 0000000000..ec70267b48 --- /dev/null +++ b/scripts/tests/test_ci_macos_admission.py @@ -0,0 +1,239 @@ +#!/usr/bin/env python3 +"""Tests for scripts/ci_macos_admission.py (stdlib unittest, no external deps). + +Run from the repo root: python -m unittest discover -s scripts/tests -v +""" + +import datetime +import os +import sys +import unittest +import urllib.error + +sys.path.insert(0, os.path.abspath(os.path.join(os.path.dirname(__file__), ".."))) + +import ci_macos_admission as admission # noqa: E402 + +UTC = datetime.timezone.utc +NOON = datetime.datetime(2026, 10, 10, 12, 0, tzinfo=UTC) + + +def job(name, status="completed", conclusion="success", labels=("macos-15",)): + return {"name": name, "status": status, "conclusion": conclusion, "labels": list(labels)} + + +def gate(arch, status="completed", conclusion="success"): + return job(f"build_macos_arch ({arch}) / Wait for a macOS runner", status, conclusion, + labels=("ubuntu-24.04",)) + + +def build(arch, status="in_progress"): + return job(f"build_macos_arch ({arch}) / Build Deps / Build OrcaSlicer / Build OrcaSlicer", + status, None if status != "completed" else "success") + + +CHECK_CACHE = job("build_macos_arch (arm64) / Check Cache", labels=("ubuntu-24.04",)) +UNIVERSAL_DONE = job("Build macOS Universal / Build OrcaSlicer") +TESTS_DONE = job("macOS arm64 / Unit Tests") +PUSH = {"id": 1, "event": "push", "head_branch": "main", "status": "in_progress"} +PR = {"id": 2, "event": "pull_request", "head_branch": "topic", "status": "in_progress"} + + +class MacosDoneTest(unittest.TestCase): + def test_not_done_before_the_final_jobs_are_listed(self): + self.assertFalse(admission.macos_done([CHECK_CACHE, build("arm64", "completed")])) + + def test_not_done_while_a_final_job_runs(self): + running = job("macOS arm64 / Unit Tests", "in_progress", None) + self.assertFalse(admission.macos_done([UNIVERSAL_DONE, running])) + + def test_done_when_the_final_jobs_finished(self): + self.assertTrue(admission.macos_done([build("arm64", "completed"), UNIVERSAL_DONE, TESTS_DONE])) + + def test_done_when_the_final_jobs_were_skipped(self): + # A skipped caller job is listed under its own name, without labels. + skipped = [job("Build macOS Universal", conclusion="skipped", labels=()), + job("macOS arm64", conclusion="skipped", labels=())] + self.assertTrue(admission.macos_done(skipped)) + + +class RunDemandTest(unittest.TestCase): + def test_push_reserves_before_its_macos_jobs_are_listed(self): + self.assertEqual(admission.run_demand(PUSH, []), admission.RESERVE) + self.assertEqual(admission.run_demand(PUSH, [CHECK_CACHE]), admission.RESERVE) + + def test_push_keeps_its_reserve_between_stages(self): + jobs = [build("arm64", "completed"), build("x86_64", "completed")] + self.assertEqual(admission.run_demand(PUSH, jobs), admission.RESERVE) + + def test_push_releases_when_macos_is_done(self): + jobs = [build("arm64", "completed"), UNIVERSAL_DONE, TESTS_DONE] + self.assertEqual(admission.run_demand(PUSH, jobs), 0) + + def test_queued_jobs_count_as_busy(self): + jobs = [build("arm64", "queued")] + self.assertEqual(admission.run_demand(PR, jobs), 1) + + def test_pull_request_holds_nothing_before_it_is_let_in(self): + self.assertEqual(admission.run_demand(PR, [CHECK_CACHE, gate("arm64", "in_progress", None)]), 0) + + def test_pull_request_holds_both_runners_once_let_in(self): + # The other arch may still be waiting for deps, and goes straight through. + self.assertEqual(admission.run_demand(PR, [gate("arm64"), gate("x86_64", "in_progress", None)]), + admission.RESERVE) + + def test_pull_request_keeps_its_runners_for_its_later_jobs(self): + # Both arch builds are done and the universal build and tests are not + # listed yet: they will need both runners without passing the line. + jobs = [gate("arm64"), gate("x86_64"), build("arm64", "completed"), build("x86_64", "completed")] + self.assertEqual(admission.run_demand(PR, jobs), 2) + + def test_pull_request_releases_when_macos_is_done(self): + jobs = [gate("arm64"), gate("x86_64"), UNIVERSAL_DONE, TESTS_DONE] + self.assertEqual(admission.run_demand(PR, jobs), 0) + + def test_pull_request_without_the_line_holds_what_it_runs(self): + # A run started from a workflow without the line. + self.assertEqual(admission.run_demand(PR, [build("arm64"), build("x86_64", "queued")]), 2) + + def test_finished_and_non_macos_jobs_are_not_counted(self): + jobs = [build("arm64", "completed"), CHECK_CACHE, + job("build_linux (ubuntu-24.04) / Check Cache", "in_progress", None, ("ubuntu-24.04",))] + self.assertEqual(admission.run_demand(PR, jobs), 0) + + +class FakeApi: + """Answers GET requests from a dict of path -> list of pages (or one body).""" + + def __init__(self, responses): + self.responses = responses + self.calls = [] + + def get(self, path, **params): + self.calls.append((path, params)) + if path.endswith("/runs") and "status" in params: + runs = self.responses.get(("runs", params["status"]), []) + start = (params["page"] - 1) * params["per_page"] + return {"total_count": len(runs), "workflow_runs": runs[start:start + params["per_page"]]} + if path.endswith("/runs"): + return {"workflow_runs": self.responses.get(("runs", params.get("event")), [])[:1]} + return {"jobs": self.responses[path]} + + +def jobs_path(run_id): + return f"repos/o/r/actions/runs/{run_id}/jobs" + + +class MeasureTest(unittest.TestCase): + def test_sums_in_progress_and_queued_runs(self): + queued_pr = dict(PR, id=3, status="queued") + api = FakeApi({ + ("runs", "in_progress"): [PUSH, PR], + ("runs", "queued"): [queued_pr], + jobs_path(1): [CHECK_CACHE], + jobs_path(2): [gate("arm64"), build("arm64")], + jobs_path(3): [gate("arm64"), gate("x86_64"), build("arm64", "queued")], + }) + total, lines, here = admission.measure(api, "o/r", "build_all.yml", 99, NOON) + self.assertEqual(total, 3 * admission.RESERVE) + self.assertEqual(len(lines), 3) + self.assertFalse(here) + + def test_reads_every_page_of_runs(self): + runs = [dict(PR, id=i) for i in range(150)] + responses = {("runs", "in_progress"): runs} + responses.update({jobs_path(i): [gate("arm64")] for i in range(150)}) + total, _, _ = admission.measure(FakeApi(responses), "o/r", "build_all.yml", 999, NOON) + self.assertEqual(total, 150 * admission.RESERVE) + + def test_reserves_for_the_nightly_until_its_run_appears(self): + due = datetime.datetime(2026, 10, 10, 2, 10, tzinfo=UTC) + yesterday = {"created_at": "2026-10-09T02:20:00Z"} + total, lines, _ = admission.measure( + FakeApi({("runs", "schedule"): [yesterday]}), "OrcaSlicer/OrcaSlicer", "build_all.yml", 99, due) + self.assertEqual(total, admission.RESERVE) + self.assertIn("nightly", lines[0]) + + def test_no_nightly_reserve_once_its_run_exists(self): + due = datetime.datetime(2026, 10, 10, 2, 30, tzinfo=UTC) + today = {"created_at": "2026-10-10T02:21:00Z"} + total, _, _ = admission.measure( + FakeApi({("runs", "schedule"): [today]}), "OrcaSlicer/OrcaSlicer", "build_all.yml", 99, due) + self.assertEqual(total, 0) + + def test_no_nightly_reserve_in_a_fork(self): + due = datetime.datetime(2026, 10, 10, 2, 10, tzinfo=UTC) + total, _, _ = admission.measure(FakeApi({}), "fork/OrcaSlicer", "build_all.yml", 99, due) + self.assertEqual(total, 0) + + def test_counts_a_run_in_both_lists_once(self): + api = FakeApi({ + ("runs", "in_progress"): [PUSH], + ("runs", "queued"): [dict(PUSH, status="queued")], + jobs_path(1): [], + }) + total, _, _ = admission.measure(api, "o/r", "build_all.yml", 99, NOON) + self.assertEqual(total, admission.RESERVE) + + def test_knows_when_its_own_run_was_let_in(self): + # Every runner is held by runs that were let in whole, so none waits for + # a runner for its second arch. + responses = {("runs", "in_progress"): [dict(PR, id=i) for i in (1, 2)]} + responses.update({jobs_path(i): [gate("arm64"), gate("x86_64", "in_progress", None)] for i in (1, 2)}) + total, _, here = admission.measure(FakeApi(responses), "o/r", "build_all.yml", 2, NOON) + self.assertEqual(total, 2 * admission.RESERVE) + self.assertTrue(here) + + def test_no_nightly_lookup_outside_its_window(self): + api = FakeApi({}) + admission.measure(api, "o/r", "build_all.yml", 99, NOON) + self.assertFalse(any("event" in params for _, params in api.calls)) + + +class WaitTest(unittest.TestCase): + def run_wait(self, results, limit=5, wait_minutes=60, poll_seconds=120): + results = iter(results) + now = [0.0] + + def measure_now(): + result = next(results) + if isinstance(result, Exception): + raise result + if result == "here": + return 99, [], True + return result, [], False + + def sleep(seconds): + now[0] += seconds + + reason = admission.wait(measure_now, limit, wait_minutes, poll_seconds, + clock=lambda: now[0], sleep=sleep, log=lambda _: None) + return reason, now[0] + + def test_lets_in_when_both_runners_fit(self): + self.assertEqual(self.run_wait([3]), ("a runner is free", 0)) + + def test_waits_while_full(self): + self.assertEqual(self.run_wait([4, 5, 3]), ("a runner is free", 240)) + + def test_second_arch_goes_straight_through(self): + self.assertEqual(self.run_wait(["here"]), ("this run already holds its runners", 0)) + + def test_lets_in_after_repeated_api_errors(self): + error = urllib.error.URLError("rate limited") + reason, _ = self.run_wait([error, error, error]) + self.assertEqual(reason, "the queue could not be read") + + def test_a_good_read_resets_the_error_count(self): + error = urllib.error.URLError("rate limited") + reason, _ = self.run_wait([error, error, 5, error, error, 3]) + self.assertEqual(reason, "a runner is free") + + def test_lets_in_at_the_time_limit(self): + reason, elapsed = self.run_wait([5] * 100, wait_minutes=10, poll_seconds=120) + self.assertEqual(reason, "it waited 10 minutes") + self.assertLess(elapsed, 600) + + +if __name__ == "__main__": + unittest.main()