mirror of
https://github.com/OrcaSlicer/OrcaSlicer.git
synced 2026-10-11 01:41:03 +00:00
Let Each macOS Arch In on Its Own Without Holding an Idle Runner
Letting a pull request run in whole deadlocks when its second arch reaches the line behind another pull request: the run holds two runners, the pull request in front waits for them, and the second arch never gets to check. Each arch now waits on its own. A run with one arch let in holds a runner only until that arch's app build is done, and a run with both holds two until its universal build and tests finish, so whoever is first in line always gets in eventually. Co-authored-by: raistlin7447 <kris.austin@gmail.com>
This commit is contained in:
co-authored by
raistlin7447
parent
83e0253dc8
commit
dca9b96ab2
@@ -3,31 +3,33 @@
|
||||
|
||||
Runs from a Linux job in build_check_cache.yml, once per macOS arch, in a single
|
||||
repo-wide line (a concurrency group with queue: max), so only the job at the
|
||||
front of the line polls. A pull request run is let in whole: its first arch waits
|
||||
until the runners every active Build all run holds or still needs, plus RESERVE
|
||||
for this run, fit in MACOS_RUNNER_LIMIT, and its second arch then goes straight
|
||||
through. Letting arches in one at a time would deadlock, with every runner held
|
||||
by a run waiting for a second one for its other arch.
|
||||
front of the line polls. It lets its arch in when the runners every active Build
|
||||
all run holds or still needs, plus what letting this arch in adds to its own
|
||||
run, fit in MACOS_RUNNER_LIMIT:
|
||||
|
||||
- A push, nightly or manual run holds RESERVE runners until its macOS work is
|
||||
done. The jobs API lists only the jobs a run has reached, so what it still
|
||||
needs cannot be counted and is reserved instead.
|
||||
- A pull request run that was let in holds RESERVE runners until its macOS work
|
||||
is done, which covers its later app build, universal build and tests that
|
||||
never pass through the line.
|
||||
- A pull request run with both arches let in holds RESERVE runners until its
|
||||
macOS work is done, which covers its universal build and tests that never
|
||||
pass through the line and start only after both arch builds.
|
||||
- A pull request run with one arch let in holds one runner until that arch's
|
||||
app build is done, and none after. A run holding a runner while it waits for
|
||||
its other arch would deadlock: the other arch can sit in the line behind a
|
||||
pull request that waits for that runner.
|
||||
- A run holds at least the macOS jobs it has queued or running.
|
||||
- In the minutes around the nightly's cron time, RESERVE runners are held for it
|
||||
until its run appears.
|
||||
|
||||
A pull request labelled macos-priority when its run starts waits in a separate
|
||||
line under the same rule, and the normal line counts every priority run still
|
||||
waiting as holding RESERVE, so the next free runners go to it.
|
||||
line under the same rule, and the normal line counts every priority arch still
|
||||
waiting as holding a runner, so the next free runners go to it.
|
||||
|
||||
Any API error repeated FAILURES_BEFORE_ADMIT times, and the WAIT_MINUTES limit,
|
||||
let the arch in, so a fault here never blocks pull requests.
|
||||
|
||||
Environment: GH_TOKEN, REPO, WORKFLOW_REF, GITHUB_RUN_ID, and optionally PRIORITY,
|
||||
MACOS_RUNNER_LIMIT, WAIT_MINUTES, POLL_SECONDS and GITHUB_API_URL.
|
||||
Environment: GH_TOKEN, REPO, WORKFLOW_REF, GITHUB_RUN_ID, ARCH, and optionally
|
||||
PRIORITY, MACOS_RUNNER_LIMIT, WAIT_MINUTES, POLL_SECONDS and GITHUB_API_URL.
|
||||
"""
|
||||
|
||||
import datetime
|
||||
@@ -44,6 +46,7 @@ import urllib.request
|
||||
RESERVE = 2
|
||||
|
||||
# Must match the job names in build_all.yml and build_check_cache.yml.
|
||||
ARCH_PREFIX = "build_macos_arch ("
|
||||
GATE = "Wait for a macOS runner"
|
||||
PRIORITY_GATE = GATE + " (priority)"
|
||||
FINAL_JOBS = {"Build macOS Universal", "macOS arm64"}
|
||||
@@ -73,18 +76,31 @@ def gates(jobs, names=(GATE, PRIORITY_GATE)):
|
||||
return [job for job in jobs if job["name"].split(" / ")[-1] in names]
|
||||
|
||||
|
||||
def admitted(jobs):
|
||||
"""True once one of the run's arches was let in."""
|
||||
return any(job["conclusion"] == "success" for job in gates(jobs))
|
||||
def arch_of(job):
|
||||
name = job["name"]
|
||||
return name[len(ARCH_PREFIX):name.find(")")] if name.startswith(ARCH_PREFIX) else None
|
||||
|
||||
|
||||
def admitted_arches(jobs):
|
||||
return {arch_of(job) for job in gates(jobs) if job["conclusion"] == "success"}
|
||||
|
||||
|
||||
def arch_built(jobs, arch):
|
||||
"""True once the arch's app build finished, or one of its jobs failed or was cancelled."""
|
||||
own = [job for job in jobs if arch_of(job) == arch]
|
||||
return any(job["conclusion"] in ("failure", "cancelled") for job in own) or \
|
||||
any(" / Build OrcaSlicer" in job["name"] and job["status"] == "completed" for job in own)
|
||||
|
||||
|
||||
def priority_waiting(jobs):
|
||||
return not admitted(jobs) and any(job["status"] != "completed" for job in gates(jobs, (PRIORITY_GATE,)))
|
||||
"""How many of the run's arches wait in the priority line."""
|
||||
return sum(1 for job in gates(jobs, (PRIORITY_GATE,)) if job["status"] != "completed")
|
||||
|
||||
|
||||
def run_demand(run, jobs, yield_to_priority=False):
|
||||
"""macOS runners a run holds or still needs. With yield_to_priority, a priority
|
||||
run still waiting counts as holding RESERVE."""
|
||||
def run_demand(run, jobs, yield_to_priority=False, letting_in=None):
|
||||
"""macOS runners a run holds or still needs. With yield_to_priority, each
|
||||
priority arch still waiting counts as holding one. letting_in counts that arch
|
||||
as let in."""
|
||||
# A run with no jobs that is pending waits behind another run of its
|
||||
# concurrency group, which holds the runners for both.
|
||||
if not jobs and run["status"] in ("pending", "waiting"):
|
||||
@@ -92,10 +108,15 @@ def run_demand(run, jobs, yield_to_priority=False):
|
||||
active = sum(1 for job in jobs if is_macos(job) and job["status"] in ACTIVE)
|
||||
if macos_done(jobs):
|
||||
return active
|
||||
if run["event"] == "pull_request" and not admitted(jobs):
|
||||
return max(active, RESERVE) if yield_to_priority and priority_waiting(jobs) else active
|
||||
return max(active, RESERVE)
|
||||
|
||||
if run["event"] != "pull_request":
|
||||
return max(active, RESERVE)
|
||||
arches = admitted_arches(jobs) | ({letting_in} if letting_in else set())
|
||||
if len(arches) >= RESERVE:
|
||||
return max(active, RESERVE)
|
||||
held = sum(1 for arch in arches if not arch_built(jobs, arch))
|
||||
if yield_to_priority:
|
||||
held += priority_waiting(jobs)
|
||||
return max(active, held)
|
||||
|
||||
|
||||
def nightly_window(now):
|
||||
@@ -141,11 +162,11 @@ def latest_run(api, repo, workflow, **params):
|
||||
return runs[0] if runs else None
|
||||
|
||||
|
||||
def measure(api, repo, workflow, run_id, now, priority=False):
|
||||
def measure(api, repo, workflow, run_id, now, priority=False, arch=None):
|
||||
"""Total macOS runners held or needed, one line per run that holds any, and
|
||||
whether run_id was already let in. The priority line does not count the
|
||||
priority runs waiting behind it."""
|
||||
total, lines, here = 0, [], False
|
||||
how many more letting arch in adds to run_id. The priority line does not
|
||||
count the priority runs waiting behind it."""
|
||||
total, lines, need = 0, [], 1
|
||||
# A run is listed as queued whenever one of its jobs waits for a runner, and
|
||||
# as pending whenever one waits in a concurrency group, so runs in any of these
|
||||
# can hold runners. A run can move between the lists between the calls.
|
||||
@@ -154,11 +175,13 @@ def measure(api, repo, workflow, run_id, now, priority=False):
|
||||
for run in runs.values():
|
||||
jobs = api.get(f"repos/{repo}/actions/runs/{run['id']}/jobs",
|
||||
filter="latest", per_page=100)["jobs"]
|
||||
demand = run_demand(run, jobs, yield_to_priority=not priority and run["id"] != run_id)
|
||||
here = here or (run["id"] == run_id and admitted(jobs))
|
||||
own = run["id"] == run_id
|
||||
demand = run_demand(run, jobs, yield_to_priority=not priority and not own)
|
||||
if own:
|
||||
need = max(1, run_demand(run, jobs, letting_in=arch) - demand)
|
||||
if demand:
|
||||
total += demand
|
||||
waiting = ", priority, waiting" if priority_waiting(jobs) else ""
|
||||
waiting = ", priority, waiting" if priority_waiting(jobs) and not own else ""
|
||||
lines.append(f" {demand} run {run['id']} ({run['event']}, {run['head_branch']}{waiting})")
|
||||
|
||||
# build_all.yml runs the nightly only in the main repository.
|
||||
@@ -168,17 +191,17 @@ def measure(api, repo, workflow, run_id, now, priority=False):
|
||||
if not last or parse_time(last["created_at"]) < start:
|
||||
total += RESERVE
|
||||
lines.append(f" {RESERVE} the nightly, due at {NIGHTLY_UTC:%H:%M} UTC")
|
||||
return total, lines, here
|
||||
return total, lines, need
|
||||
|
||||
|
||||
def wait(measure_now, limit, wait_minutes, poll_seconds,
|
||||
clock=time.monotonic, sleep=time.sleep, log=print):
|
||||
"""Polls until this run fits. Returns the reason it was let in."""
|
||||
"""Polls until this arch fits. Returns the reason it was let in."""
|
||||
deadline = clock() + wait_minutes * 60
|
||||
failures = 0
|
||||
while True:
|
||||
try:
|
||||
total, lines, here = measure_now()
|
||||
total, lines, need = measure_now()
|
||||
except (urllib.error.URLError, OSError, ValueError, KeyError, TypeError) as error:
|
||||
failures += 1
|
||||
log(f"::warning title=macOS admission::Could not read the queue ({error}).")
|
||||
@@ -186,12 +209,10 @@ def wait(measure_now, limit, wait_minutes, poll_seconds,
|
||||
return "the queue could not be read"
|
||||
else:
|
||||
failures = 0
|
||||
if here:
|
||||
return "this run already holds its runners"
|
||||
log(f"{total} of {limit} macOS runners held or needed:")
|
||||
log(f"{total} of {limit} macOS runners held or needed, and this arch needs {need}:")
|
||||
for line in lines:
|
||||
log(line)
|
||||
if total + RESERVE <= limit:
|
||||
if total + need <= limit:
|
||||
return "a runner is free"
|
||||
if clock() + poll_seconds >= deadline:
|
||||
return f"it waited {wait_minutes} minutes"
|
||||
@@ -206,7 +227,7 @@ def main():
|
||||
limit = int(os.environ.get("MACOS_RUNNER_LIMIT") or 5)
|
||||
reason = wait(
|
||||
lambda: measure(api, repo, workflow, run_id, datetime.datetime.now(datetime.timezone.utc),
|
||||
priority=os.environ.get("PRIORITY") == "true"),
|
||||
priority=os.environ.get("PRIORITY") == "true", arch=os.environ["ARCH"]),
|
||||
limit,
|
||||
wait_minutes=int(os.environ.get("WAIT_MINUTES") or 240),
|
||||
poll_seconds=int(os.environ.get("POLL_SECONDS") or 180),
|
||||
|
||||
@@ -78,10 +78,20 @@ class RunDemandTest(unittest.TestCase):
|
||||
def test_pull_request_holds_nothing_before_it_is_let_in(self):
|
||||
self.assertEqual(admission.run_demand(PR, [CHECK_CACHE, gate("arm64", "in_progress", None)]), 0)
|
||||
|
||||
def test_pull_request_holds_both_runners_once_let_in(self):
|
||||
# The other arch may still be waiting for deps, and goes straight through.
|
||||
self.assertEqual(admission.run_demand(PR, [gate("arm64"), gate("x86_64", "in_progress", None)]),
|
||||
admission.RESERVE)
|
||||
def test_one_arch_let_in_holds_a_runner_until_it_is_built(self):
|
||||
# Including before its first macOS job is listed, and between its stages.
|
||||
self.assertEqual(admission.run_demand(PR, [gate("arm64")]), 1)
|
||||
self.assertEqual(admission.run_demand(PR, [gate("arm64"), build("arm64")]), 1)
|
||||
|
||||
def test_one_arch_let_in_holds_nothing_once_it_is_built(self):
|
||||
# Its other arch may be in the line behind a pull request waiting for this
|
||||
# runner, so holding it would deadlock.
|
||||
jobs = [gate("arm64"), build("arm64", "completed"), gate("x86_64", "pending", None)]
|
||||
self.assertEqual(admission.run_demand(PR, jobs), 0)
|
||||
|
||||
def test_one_arch_let_in_holds_nothing_once_it_failed(self):
|
||||
failed = job("build_macos_arch (arm64) / Build Deps / Build Deps", conclusion="failure")
|
||||
self.assertEqual(admission.run_demand(PR, [gate("arm64"), failed]), 0)
|
||||
|
||||
def test_pull_request_keeps_its_runners_for_its_later_jobs(self):
|
||||
# Both arch builds are done and the universal build and tests are not
|
||||
@@ -116,10 +126,10 @@ class PriorityTest(unittest.TestCase):
|
||||
waiting = [gate("arm64", "in_progress", None), gate("x86_64", "pending", None)]
|
||||
self.assertEqual(admission.run_demand(PR, waiting, yield_to_priority=True), 0)
|
||||
|
||||
def test_a_priority_run_let_in_holds_its_runners(self):
|
||||
def test_a_priority_run_half_let_in_still_counts_its_waiting_arch(self):
|
||||
jobs = [gate("arm64", priority=True), gate("x86_64", "pending", None, priority=True)]
|
||||
self.assertEqual(admission.run_demand(PR, jobs), admission.RESERVE)
|
||||
self.assertEqual(admission.run_demand(PR, jobs, yield_to_priority=True), admission.RESERVE)
|
||||
self.assertEqual(admission.run_demand(PR, jobs), 1)
|
||||
self.assertEqual(admission.run_demand(PR, jobs, yield_to_priority=True), 2)
|
||||
|
||||
def test_skipped_priority_jobs_of_other_platforms_are_not_waiting(self):
|
||||
jobs = [job("build_linux (ubuntu-24.04) / Wait for a macOS runner (priority)",
|
||||
@@ -136,8 +146,8 @@ class PriorityTest(unittest.TestCase):
|
||||
|
||||
def test_own_waiting_run_is_not_counted(self):
|
||||
api = FakeApi({("runs", "pending"): [dict(PR, status="pending")], jobs_path(2): self.WAITING})
|
||||
total, _, here = admission.measure(api, "o/r", "build_all.yml", 2, NOON)
|
||||
self.assertEqual((total, here), (0, False))
|
||||
total, _, need = admission.measure(api, "o/r", "build_all.yml", 2, NOON, arch="arm64")
|
||||
self.assertEqual((total, need), (0, 1))
|
||||
|
||||
|
||||
class FakeApi:
|
||||
@@ -172,17 +182,16 @@ class MeasureTest(unittest.TestCase):
|
||||
jobs_path(2): [gate("arm64"), build("arm64")],
|
||||
jobs_path(3): [gate("arm64"), gate("x86_64"), build("arm64", "queued")],
|
||||
})
|
||||
total, lines, here = admission.measure(api, "o/r", "build_all.yml", 99, NOON)
|
||||
self.assertEqual(total, 3 * admission.RESERVE)
|
||||
total, lines, _ = admission.measure(api, "o/r", "build_all.yml", 99, NOON)
|
||||
self.assertEqual(total, admission.RESERVE + 1 + admission.RESERVE)
|
||||
self.assertEqual(len(lines), 3)
|
||||
self.assertFalse(here)
|
||||
|
||||
def test_counts_a_pending_run_that_holds_runners(self):
|
||||
# One of its jobs waits in a concurrency group while its macOS builds run.
|
||||
pending_pr = dict(PR, status="pending")
|
||||
api = FakeApi({("runs", "pending"): [pending_pr], jobs_path(2): [gate("arm64"), build("arm64")]})
|
||||
total, _, _ = admission.measure(api, "o/r", "build_all.yml", 99, NOON)
|
||||
self.assertEqual(total, admission.RESERVE)
|
||||
self.assertEqual(total, 1)
|
||||
|
||||
def test_a_push_waiting_behind_another_holds_nothing(self):
|
||||
api = FakeApi({("runs", "pending"): [dict(PUSH, status="pending")], jobs_path(1): []})
|
||||
@@ -194,7 +203,7 @@ class MeasureTest(unittest.TestCase):
|
||||
responses = {("runs", "in_progress"): runs}
|
||||
responses.update({jobs_path(i): [gate("arm64")] for i in range(150)})
|
||||
total, _, _ = admission.measure(FakeApi(responses), "o/r", "build_all.yml", 999, NOON)
|
||||
self.assertEqual(total, 150 * admission.RESERVE)
|
||||
self.assertEqual(total, 150)
|
||||
|
||||
def test_reserves_for_the_nightly_until_its_run_appears(self):
|
||||
due = datetime.datetime(2026, 10, 10, 2, 10, tzinfo=UTC)
|
||||
@@ -225,14 +234,34 @@ class MeasureTest(unittest.TestCase):
|
||||
total, _, _ = admission.measure(api, "o/r", "build_all.yml", 99, NOON)
|
||||
self.assertEqual(total, admission.RESERVE)
|
||||
|
||||
def test_knows_when_its_own_run_was_let_in(self):
|
||||
# Every runner is held by runs that were let in whole, so none waits for
|
||||
# a runner for its second arch.
|
||||
responses = {("runs", "in_progress"): [dict(PR, id=i) for i in (1, 2)]}
|
||||
responses.update({jobs_path(i): [gate("arm64"), gate("x86_64", "in_progress", None)] for i in (1, 2)})
|
||||
total, _, here = admission.measure(FakeApi(responses), "o/r", "build_all.yml", 2, NOON)
|
||||
self.assertEqual(total, 2 * admission.RESERVE)
|
||||
self.assertTrue(here)
|
||||
def own(self, jobs, arch):
|
||||
api = FakeApi({("runs", "in_progress"): [PR], jobs_path(2): jobs})
|
||||
total, _, need = admission.measure(api, "o/r", "build_all.yml", 2, NOON, arch=arch)
|
||||
return total, need
|
||||
|
||||
def test_first_arch_needs_one(self):
|
||||
self.assertEqual(self.own([gate("arm64", "in_progress", None)], "arm64"), (0, 1))
|
||||
|
||||
def test_second_arch_needs_one_while_the_first_builds(self):
|
||||
jobs = [gate("arm64"), build("arm64"), gate("x86_64", "in_progress", None)]
|
||||
self.assertEqual(self.own(jobs, "x86_64"), (1, 1))
|
||||
|
||||
def test_second_arch_needs_both_once_the_first_is_built(self):
|
||||
# Letting it in makes the run hold both for its universal build and tests.
|
||||
jobs = [gate("arm64"), build("arm64", "completed"), gate("x86_64", "in_progress", None)]
|
||||
self.assertEqual(self.own(jobs, "x86_64"), (0, 2))
|
||||
|
||||
def test_a_run_whose_other_arch_waits_behind_this_one_does_not_block_it(self):
|
||||
# The line is A-arm64, B-arm64, A-x86_64. A's arm64 is built and its
|
||||
# x86_64 is pending behind B, so A must not hold a runner B waits for.
|
||||
other = dict(PR, id=1, status="pending")
|
||||
api = FakeApi({
|
||||
("runs", "pending"): [other, dict(PR, status="pending")],
|
||||
jobs_path(1): [gate("arm64"), build("arm64", "completed"), gate("x86_64", "pending", None)],
|
||||
jobs_path(2): [gate("arm64", "in_progress", None)],
|
||||
})
|
||||
total, _, need = admission.measure(api, "o/r", "build_all.yml", 2, NOON, arch="arm64")
|
||||
self.assertEqual((total, need), (0, 1))
|
||||
|
||||
def test_no_nightly_lookup_outside_its_window(self):
|
||||
api = FakeApi({})
|
||||
@@ -249,9 +278,8 @@ class WaitTest(unittest.TestCase):
|
||||
result = next(results)
|
||||
if isinstance(result, Exception):
|
||||
raise result
|
||||
if result == "here":
|
||||
return 99, [], True
|
||||
return result, [], False
|
||||
total, need = result if isinstance(result, tuple) else (result, 1)
|
||||
return total, [], need
|
||||
|
||||
def sleep(seconds):
|
||||
now[0] += seconds
|
||||
@@ -260,14 +288,14 @@ class WaitTest(unittest.TestCase):
|
||||
clock=lambda: now[0], sleep=sleep, log=lambda _: None)
|
||||
return reason, now[0]
|
||||
|
||||
def test_lets_in_when_both_runners_fit(self):
|
||||
self.assertEqual(self.run_wait([3]), ("a runner is free", 0))
|
||||
def test_lets_in_when_a_runner_fits(self):
|
||||
self.assertEqual(self.run_wait([4]), ("a runner is free", 0))
|
||||
|
||||
def test_waits_while_full(self):
|
||||
self.assertEqual(self.run_wait([4, 5, 3]), ("a runner is free", 240))
|
||||
self.assertEqual(self.run_wait([5, 6, 4]), ("a runner is free", 240))
|
||||
|
||||
def test_second_arch_goes_straight_through(self):
|
||||
self.assertEqual(self.run_wait(["here"]), ("this run already holds its runners", 0))
|
||||
def test_waits_for_what_this_arch_needs(self):
|
||||
self.assertEqual(self.run_wait([(4, 2), (3, 2)]), ("a runner is free", 120))
|
||||
|
||||
def test_lets_in_after_repeated_api_errors(self):
|
||||
error = urllib.error.URLError("rate limited")
|
||||
|
||||
Reference in New Issue
Block a user