Files
OrcaSlicer/scripts/ci_macos_admission.py
T
Hanif Kohandraistlin7447 dca9b96ab2 Let Each macOS Arch In on Its Own Without Holding an Idle Runner
Letting a pull request run in whole deadlocks when its second arch
reaches the line behind another pull request: the run holds two
runners, the pull request in front waits for them, and the second arch
never gets to check. Each arch now waits on its own. A run with one
arch let in holds a runner only until that arch's app build is done,
and a run with both holds two until its universal build and tests
finish, so whoever is first in line always gets in eventually.

Co-authored-by: raistlin7447 <kris.austin@gmail.com>
2026-10-11 05:49:13 +08:00

240 lines
9.9 KiB
Python

#!/usr/bin/env python3
"""Holds a pull request's macOS build until a hosted macOS runner is free for it.
Runs from a Linux job in build_check_cache.yml, once per macOS arch, in a single
repo-wide line (a concurrency group with queue: max), so only the job at the
front of the line polls. It lets its arch in when the runners every active Build
all run holds or still needs, plus what letting this arch in adds to its own
run, fit in MACOS_RUNNER_LIMIT:
- A push, nightly or manual run holds RESERVE runners until its macOS work is
done. The jobs API lists only the jobs a run has reached, so what it still
needs cannot be counted and is reserved instead.
- A pull request run with both arches let in holds RESERVE runners until its
macOS work is done, which covers its universal build and tests that never
pass through the line and start only after both arch builds.
- A pull request run with one arch let in holds one runner until that arch's
app build is done, and none after. A run holding a runner while it waits for
its other arch would deadlock: the other arch can sit in the line behind a
pull request that waits for that runner.
- A run holds at least the macOS jobs it has queued or running.
- In the minutes around the nightly's cron time, RESERVE runners are held for it
until its run appears.
A pull request labelled macos-priority when its run starts waits in a separate
line under the same rule, and the normal line counts every priority arch still
waiting as holding a runner, so the next free runners go to it.
Any API error repeated FAILURES_BEFORE_ADMIT times, and the WAIT_MINUTES limit,
let the arch in, so a fault here never blocks pull requests.
Environment: GH_TOKEN, REPO, WORKFLOW_REF, GITHUB_RUN_ID, ARCH, and optionally
PRIORITY, MACOS_RUNNER_LIMIT, WAIT_MINUTES, POLL_SECONDS and GITHUB_API_URL.
"""
import datetime
import json
import os
import sys
import time
import urllib.error
import urllib.parse
import urllib.request
# A Build all run builds arm64 and x86_64 at the same time, then the universal
# build and the macOS tests at the same time.
RESERVE = 2
# Must match the job names in build_all.yml and build_check_cache.yml.
ARCH_PREFIX = "build_macos_arch ("
GATE = "Wait for a macOS runner"
PRIORITY_GATE = GATE + " (priority)"
FINAL_JOBS = {"Build macOS Universal", "macOS arm64"}
# Must match the cron in build_all.yml.
NIGHTLY_UTC = datetime.time(2, 15)
NIGHTLY_REPO = "OrcaSlicer/OrcaSlicer"
NIGHTLY_LEAD = datetime.timedelta(minutes=10)
NIGHTLY_GRACE = datetime.timedelta(minutes=45)
ACTIVE = {"queued", "in_progress", "waiting", "pending", "requested"}
FAILURES_BEFORE_ADMIT = 3
def is_macos(job):
return any(label.startswith("macos-") for label in job.get("labels") or [])
def macos_done(jobs):
"""True once the universal build and the macOS tests have finished or been skipped.
Both start together when the arch builds finish."""
# A skipped caller job is listed under its own name, a started one as "<name> / <job>".
final = [job for job in jobs if job["name"].split(" / ")[0] in FINAL_JOBS]
return bool(final) and all(job["status"] == "completed" for job in final)
def gates(jobs, names=(GATE, PRIORITY_GATE)):
return [job for job in jobs if job["name"].split(" / ")[-1] in names]
def arch_of(job):
name = job["name"]
return name[len(ARCH_PREFIX):name.find(")")] if name.startswith(ARCH_PREFIX) else None
def admitted_arches(jobs):
return {arch_of(job) for job in gates(jobs) if job["conclusion"] == "success"}
def arch_built(jobs, arch):
"""True once the arch's app build finished, or one of its jobs failed or was cancelled."""
own = [job for job in jobs if arch_of(job) == arch]
return any(job["conclusion"] in ("failure", "cancelled") for job in own) or \
any(" / Build OrcaSlicer" in job["name"] and job["status"] == "completed" for job in own)
def priority_waiting(jobs):
"""How many of the run's arches wait in the priority line."""
return sum(1 for job in gates(jobs, (PRIORITY_GATE,)) if job["status"] != "completed")
def run_demand(run, jobs, yield_to_priority=False, letting_in=None):
"""macOS runners a run holds or still needs. With yield_to_priority, each
priority arch still waiting counts as holding one. letting_in counts that arch
as let in."""
# A run with no jobs that is pending waits behind another run of its
# concurrency group, which holds the runners for both.
if not jobs and run["status"] in ("pending", "waiting"):
return 0
active = sum(1 for job in jobs if is_macos(job) and job["status"] in ACTIVE)
if macos_done(jobs):
return active
if run["event"] != "pull_request":
return max(active, RESERVE)
arches = admitted_arches(jobs) | ({letting_in} if letting_in else set())
if len(arches) >= RESERVE:
return max(active, RESERVE)
held = sum(1 for arch in arches if not arch_built(jobs, arch))
if yield_to_priority:
held += priority_waiting(jobs)
return max(active, held)
def nightly_window(now):
"""The window around today's nightly cron time, as (start, end) in UTC."""
cron = datetime.datetime.combine(now.date(), NIGHTLY_UTC, tzinfo=datetime.timezone.utc)
return cron - NIGHTLY_LEAD, cron + NIGHTLY_GRACE
def parse_time(value):
return datetime.datetime.fromisoformat(value.replace("Z", "+00:00"))
class Api:
def __init__(self, token, url="https://api.github.com"):
self.token = token
self.url = url.rstrip("/")
def get(self, path, **params):
query = urllib.parse.urlencode(params)
request = urllib.request.Request(f"{self.url}/{path}?{query}", headers={
"Accept": "application/vnd.github+json",
"Authorization": f"Bearer {self.token}",
"X-GitHub-Api-Version": "2022-11-28",
})
with urllib.request.urlopen(request, timeout=30) as response:
return json.load(response)
def list_runs(api, repo, workflow, **params):
runs, page = [], 1
while True:
body = api.get(f"repos/{repo}/actions/workflows/{workflow}/runs",
per_page=100, page=page, **params)
runs += body["workflow_runs"]
if len(runs) >= body["total_count"] or not body["workflow_runs"]:
return runs
page += 1
def latest_run(api, repo, workflow, **params):
runs = api.get(f"repos/{repo}/actions/workflows/{workflow}/runs",
per_page=1, **params)["workflow_runs"]
return runs[0] if runs else None
def measure(api, repo, workflow, run_id, now, priority=False, arch=None):
"""Total macOS runners held or needed, one line per run that holds any, and
how many more letting arch in adds to run_id. The priority line does not
count the priority runs waiting behind it."""
total, lines, need = 0, [], 1
# A run is listed as queued whenever one of its jobs waits for a runner, and
# as pending whenever one waits in a concurrency group, so runs in any of these
# can hold runners. A run can move between the lists between the calls.
runs = {run["id"]: run for status in ("in_progress", "queued", "pending", "waiting")
for run in list_runs(api, repo, workflow, status=status)}
for run in runs.values():
jobs = api.get(f"repos/{repo}/actions/runs/{run['id']}/jobs",
filter="latest", per_page=100)["jobs"]
own = run["id"] == run_id
demand = run_demand(run, jobs, yield_to_priority=not priority and not own)
if own:
need = max(1, run_demand(run, jobs, letting_in=arch) - demand)
if demand:
total += demand
waiting = ", priority, waiting" if priority_waiting(jobs) and not own else ""
lines.append(f" {demand} run {run['id']} ({run['event']}, {run['head_branch']}{waiting})")
# build_all.yml runs the nightly only in the main repository.
start, end = nightly_window(now)
if repo == NIGHTLY_REPO and start <= now < end:
last = latest_run(api, repo, workflow, event="schedule")
if not last or parse_time(last["created_at"]) < start:
total += RESERVE
lines.append(f" {RESERVE} the nightly, due at {NIGHTLY_UTC:%H:%M} UTC")
return total, lines, need
def wait(measure_now, limit, wait_minutes, poll_seconds,
clock=time.monotonic, sleep=time.sleep, log=print):
"""Polls until this arch fits. Returns the reason it was let in."""
deadline = clock() + wait_minutes * 60
failures = 0
while True:
try:
total, lines, need = measure_now()
except (urllib.error.URLError, OSError, ValueError, KeyError, TypeError) as error:
failures += 1
log(f"::warning title=macOS admission::Could not read the queue ({error}).")
if failures >= FAILURES_BEFORE_ADMIT:
return "the queue could not be read"
else:
failures = 0
log(f"{total} of {limit} macOS runners held or needed, and this arch needs {need}:")
for line in lines:
log(line)
if total + need <= limit:
return "a runner is free"
if clock() + poll_seconds >= deadline:
return f"it waited {wait_minutes} minutes"
sleep(poll_seconds)
def main():
repo = os.environ["REPO"]
workflow = os.environ["WORKFLOW_REF"].split("@")[0].rsplit("/", 1)[-1]
api = Api(os.environ["GH_TOKEN"], os.environ.get("GITHUB_API_URL", "https://api.github.com"))
run_id = int(os.environ["GITHUB_RUN_ID"])
limit = int(os.environ.get("MACOS_RUNNER_LIMIT") or 5)
reason = wait(
lambda: measure(api, repo, workflow, run_id, datetime.datetime.now(datetime.timezone.utc),
priority=os.environ.get("PRIORITY") == "true", arch=os.environ["ARCH"]),
limit,
wait_minutes=int(os.environ.get("WAIT_MINUTES") or 240),
poll_seconds=int(os.environ.get("POLL_SECONDS") or 180),
)
print(f"Letting this macOS build in: {reason}.")
if __name__ == "__main__":
sys.exit(main())