From 1c2eb586b2ac64d822a15fae4ae0a242112f1ffa Mon Sep 17 00:00:00 2001 From: Joseph Mearman Date: Wed, 30 Sep 2026 09:25:06 +0100 Subject: [PATCH] feat(arc): let a sized profile's runner pods overflow onto burst nodes A scale-set profile can now set burst: the label a cluster autoscaler's burst nodes carry, tolerations for their taint, and how many runners those nodes hold at most. The runner pods then trade the eligibility nodeSelector for a required node affinity accepting either the eligibility label or the burst label, with a preferred term that keeps them on the fixed nodes while those have room. The listener keeps the eligibility nodeSelector. burst.max_runners is added to the ceiling sizing derives, uniform or measured, so ARC creates more runner pods than the fixed nodes hold and the extra ones pend until the autoscaler adds a node. A profile without burst renders exactly as before. --- README.md | 2 +- plugins/filter/arc.py | 78 +++++++++++++++++-- roles/github_runner_arc/README.md | 22 +++++- roles/github_runner_arc/defaults/main.yml | 2 +- .../github_runner_arc/meta/argument_specs.yml | 3 +- .../templates/scale-set-overlay.yaml.j2 | 17 ++-- tests/scale_set_overlay/render.yml | 14 ++++ tests/unit/test_arc_filters.py | 69 ++++++++++++++++ tests/unit/test_scale_set_overlay.py | 76 ++++++++++++++++++ 9 files changed, 266 insertions(+), 17 deletions(-) create mode 100644 tests/scale_set_overlay/render.yml create mode 100644 tests/unit/test_scale_set_overlay.py diff --git a/README.md b/README.md index e1b7bef..ddfa492 100644 --- a/README.md +++ b/README.md @@ -39,7 +39,7 @@ The playbook runs up to four roles, each with its own README: `github_runner_clu Which hosts are k3s servers is worked out from the inventory, not set per host. The `github_runner_cluster` inventory group lists the cluster's hosts in order: the first N become servers (control-plane/etcd members running `k3s server`, genuine voting members of the raft quorum), where N is the largest odd number no greater than both the number of hosts and `github_runner_cluster_max_servers` (5 by default), so 1 or 2 hosts give 1 server, 3 or 4 give 3, and 5 or more give 5. The first server bootstraps the cluster (`--cluster-init`, embedded etcd) and every other host joins it, the rest as agents (`k3s agent`, schedulable capacity with no control-plane role). Join URLs and every server's `--tls-san` list are derived from each server's node name and `github_runner_cluster_tailnet_domain`. A host can still set `github_runner_cluster_node_role`, `github_runner_cluster_bootstrap`, `github_runner_cluster_server_url` or `github_runner_cluster_tls_sans` to override its part, and the run fails with an explanation if the result has an even number of servers, none at all, or more than one bootstrap host; `roles/github_runner_cluster/defaults/main.yml` describes each. Installing the ARC controller, each org's runner scale set, and the fleet-health platform is not tied to a node role at all: each is driven purely by whether the installing host's own `github_runner_arc_orgs`/`github_runner_arc_heartbeat_gist_id` is set, so any server host can be the one Ansible installs against, and moving which host does the installing is a config change (which host's `host_vars` sets it), not a role change. See the [Architecture](#architecture) section for why 3 servers is the minimum sane count for genuine HA (2 is worse than 1 for quorum math, not better). To add a server or agent, add the host to the `github_runner_cluster` group in the fleet's inventory and give it its own `host_vars/.yml`; keep the bootstrap host first. -`github_runner_arc_orgs[].scale_set_profiles` is a list of `{suffix, values_file?, node_selector?, runs_on_label?, autoscale?, sizing?}` entries (see `roles/github_runner_arc/README.md` for every key) - one Helm release and one namespace per entry, pooled under the org's own shared `runs-on` label via `scaleSetLabels` (not `runnerScaleSetName`, which is a scale set's own unique GitHub-side registration identity, not just a label - two releases sharing it collide outright) unless a profile sets its own `runs_on_label` to deliberately opt out of that pool - e.g. a dedicated image-building profile ordinary CI jobs have no business landing on. `node_selector` is optional per profile, not mandatory: omitting it leaves that profile's pods genuinely unpinned, letting the Kubernetes scheduler place them on whichever node in the fleet actually has room - real failover if one machine goes down, not just capacity pooling. Values files are rendered on the control node from `github_runner_arc_values_dir`, a directory in the fleet's own repository, so no fleet host needs them. ExaDev's own fleet uses two profiles: an ordinary CI profile unpinned across the whole fleet (sized with Guaranteed QoS for the smallest machine's real capacity), and a second, separately-labelled, Docker-in-Docker profile pinned to one host for building the runner image - a deliberate isolation choice, not a capacity constraint. +`github_runner_arc_orgs[].scale_set_profiles` is a list of `{suffix, values_file?, node_selector?, runs_on_label?, autoscale?, sizing?, burst?}` entries (see `roles/github_runner_arc/README.md` for every key) - one Helm release and one namespace per entry, pooled under the org's own shared `runs-on` label via `scaleSetLabels` (not `runnerScaleSetName`, which is a scale set's own unique GitHub-side registration identity, not just a label - two releases sharing it collide outright) unless a profile sets its own `runs_on_label` to deliberately opt out of that pool - e.g. a dedicated image-building profile ordinary CI jobs have no business landing on. `node_selector` is optional per profile, not mandatory: omitting it leaves that profile's pods genuinely unpinned, letting the Kubernetes scheduler place them on whichever node in the fleet actually has room - real failover if one machine goes down, not just capacity pooling. Values files are rendered on the control node from `github_runner_arc_values_dir`, a directory in the fleet's own repository, so no fleet host needs them. ExaDev's own fleet uses two profiles: an ordinary CI profile unpinned across the whole fleet (sized with Guaranteed QoS for the smallest machine's real capacity), and a second, separately-labelled, Docker-in-Docker profile pinned to one host for building the runner image - a deliberate isolation choice, not a capacity constraint. ## Build, test, and smoke-test diff --git a/plugins/filter/arc.py b/plugins/filter/arc.py index bdceaf6..abb8cf7 100644 --- a/plugins/filter/arc.py +++ b/plugins/filter/arc.py @@ -1,4 +1,4 @@ -"""Filters for the github_runner_arc role: expanding orgs into scale-set profiles, deriving a runner ceiling from node capacity (uniform figures, or each node's measured free capacity), and splitting an image reference.""" +"""Filters for the github_runner_arc role: expanding orgs into scale-set profiles, placing their runner pods (on the eligible nodes, or overflowing onto burst nodes), deriving a runner ceiling from node capacity (uniform figures, or each node's measured free capacity), and splitting an image reference.""" from __future__ import annotations @@ -68,6 +68,12 @@ # The prefix of the taints Kubernetes itself puts on a node for a passing condition (not ready, unreachable, under pressure, cordoned). The ceiling outlasts the moment it is measured, so a node in such a state still counts, as it will when it recovers; only a node's labels and deliberately added taints decide whether it is eligible. _CONDITION_TAINT_PREFIX = "node.kubernetes.io/" +# A profile's burst keys (see _burst), and the fields of a Kubernetes toleration (core/v1 Toleration). +_BURST_KEYS = frozenset({"node_label_key", "node_label_values", "tolerations", "max_runners"}) +_TOLERATION_KEYS = frozenset({"key", "operator", "value", "effect", "tolerationSeconds"}) +# The highest weight a preferred node affinity term may carry (1-100 in the Kubernetes API), so keeping runner pods on the fixed nodes outweighs any other preference the scheduler scores. +_PREFER_FIXED_NODES_WEIGHT = 100 + def _decimal(value: Any) -> Decimal | None: """Return value as an exact Decimal, or None when it is not a finite number.""" @@ -365,6 +371,59 @@ def _explicit_max_runners(profile: Mapping[str, Any], where: str, errors: list[s return int(number) +def _burst(profile: Mapping[str, Any], where: str, errors: list[str]) -> dict[str, Any] | None: + """Return a profile's burst settings with defaults filled in, None when it sets none, recording each problem in errors.""" + burst = profile.get("burst") + if burst is None: + return None + if not isinstance(burst, Mapping): + errors.append(f"{where}.burst must be a mapping") + return None + errors.extend(f"{where}.burst has an unknown key '{key}'" for key in sorted(set(burst) - _BURST_KEYS)) + label_key = burst.get("node_label_key") + if not isinstance(label_key, str) or not label_key: + errors.append(f"{where}.burst.node_label_key must be a non-empty string") + label_values = burst.get("node_label_values") or [] + if not isinstance(label_values, Sequence) or isinstance(label_values, (str, bytes)) or not all(isinstance(value, str) and value for value in label_values): + errors.append(f"{where}.burst.node_label_values must be a list of non-empty strings (empty or unset accepts any value)") + label_values = [] + tolerations = burst.get("tolerations") or [] + if not isinstance(tolerations, Sequence) or isinstance(tolerations, (str, bytes)) or not all(isinstance(toleration, Mapping) for toleration in tolerations): + errors.append(f"{where}.burst.tolerations must be a list of Kubernetes tolerations") + tolerations = [] + for toleration_index, toleration in enumerate(tolerations): + errors.extend(f"{where}.burst.tolerations[{toleration_index}] has an unknown key '{key}'" for key in sorted(set(toleration) - _TOLERATION_KEYS)) + max_runners = _decimal(burst.get("max_runners")) + if max_runners is None or max_runners < 0 or max_runners != max_runners.to_integral_value(): + errors.append(f"{where}.burst.max_runners must be a whole number of at least 0 (got {burst.get('max_runners')!r})") + max_runners = Decimal(0) + return {"node_label_key": label_key, "node_label_values": list(label_values), "tolerations": [dict(toleration) for toleration in tolerations], "max_runners": int(max_runners)} + + +def arc_runner_placement(profile: Mapping[str, Any], eligibility: Mapping[str, Any]) -> dict[str, Any]: + """Return where a profile's runner pods may be scheduled, as pod spec fields: nodeSelector, and with burst also affinity and tolerations. + + Without burst the pods are held by nodeSelector to nodes carrying the eligibility label and the profile's node_selector. With burst the eligibility label moves into a required node affinity that also accepts a node carrying the burst label, so a pod that finds no room on an eligible node stays Pending until a burst node can take it, which is what makes Cluster Autoscaler add one. A preferred term keeps the pods off burst nodes while an eligible node has room. The profile's node_selector still applies to every node, and the tolerations let the pods past the burst nodes' taint. + + Args: + profile: an expanded profile from arc_profiles. eligibility: the eligibility label as {key: value}, empty when the role sets none. + + Returns: + A dict of pod spec fields for the runner pod template. + """ + selector = {key: str(value) for key, value in eligibility.items()} + burst = profile["burst"] + if burst is None: + return {"nodeSelector": {**selector, **profile["node_selector"]}} + burst_node = {"key": burst["node_label_key"], "operator": "In", "values": burst["node_label_values"]} if burst["node_label_values"] else {"key": burst["node_label_key"], "operator": "Exists"} + node_affinity: dict[str, Any] = {"preferredDuringSchedulingIgnoredDuringExecution": [{"weight": _PREFER_FIXED_NODES_WEIGHT, "preference": {"matchExpressions": [{"key": burst["node_label_key"], "operator": "DoesNotExist"}]}}]} + # With no eligibility label every untainted node is already allowed, and the tolerations alone add the burst nodes. + if selector: + eligible_node = [{"key": key, "operator": "In", "values": [value]} for key, value in selector.items()] + node_affinity["requiredDuringSchedulingIgnoredDuringExecution"] = {"nodeSelectorTerms": [{"matchExpressions": eligible_node}, {"matchExpressions": [burst_node]}]} + return {"nodeSelector": dict(profile["node_selector"]), "affinity": {"nodeAffinity": node_affinity}, "tolerations": burst["tolerations"]} + + def _expand_profile(org_name: str, app_secret: str, index: int, profile: Any, errors: list[str]) -> dict[str, Any] | None: """Describe one scale-set profile of an org, or record why it cannot be described.""" where = f"{org_name} scale_set_profiles[{index}]" @@ -387,11 +446,16 @@ def _expand_profile(org_name: str, app_secret: str, index: int, profile: Any, er sizing_result = arc_max_runners(profile["sizing"]) errors.extend(f"{where}: {message}" for message in sizing_result["errors"]) explicit_max_runners = _explicit_max_runners(profile, where, errors) - # The static maxRunners the role sets on the release: the profile's own max_runners wins; otherwise sizing supplies it, except on the autoscaled profile, whose sizing sets the autoscaler's ceiling rather than its floor. + burst = _burst(profile, where, errors) + # Burst runners are added to a ceiling the role derives from the fixed nodes, so they need sizing on a profile that sets its own maxRunners: an explicit max_runners is final, and the autoscaler's usage-driven ceiling has no notion of nodes that do not exist yet. + if burst is not None and (sizing_result is None or explicit_max_runners is not None or autoscale): + errors.append(f"{where}: burst needs sizing, and no max_runners or autoscale, since its max_runners is added to the ceiling sizing derives") + burst_runners = burst["max_runners"] if burst is not None else 0 + # The static maxRunners the role sets on the release: the profile's own max_runners wins; otherwise sizing supplies it, plus any burst runners, except on the autoscaled profile, whose sizing sets the autoscaler's ceiling rather than its floor. A measured sizing's result is known only at install time, so its max_runners stays None here and the burst runners are added then. if explicit_max_runners is not None: max_runners = explicit_max_runners elif sizing_result is not None and not sizing_result["errors"] and not autoscale: - max_runners = sizing_result["max_runners"] + max_runners = None if sizing_result["max_runners"] is None else sizing_result["max_runners"] + burst_runners else: max_runners = None namespace = _optional_name(profile, "namespace", where, errors) or f"arc-runners-{org_lower}{suffix}" @@ -409,6 +473,8 @@ def _expand_profile(org_name: str, app_secret: str, index: int, profile: Any, er "node_selector": dict(node_selector), "autoscale": autoscale, "sizing": sizing_result, + "burst": burst, + "burst_runners": burst_runners, "max_runners": max_runners, "label": f"{org_name}{suffix}", "settings": dict(profile), @@ -428,10 +494,10 @@ def arc_profiles(orgs: Sequence[Any], require_app_id: bool = True) -> dict[str, Args: require_app_id: whether each org must set app_id. The role needs it only when it writes the App Secret; otherwise the App's id is in the existing App Secret, and an app_id that is set is only checked against it. - orgs: github_runner_arc_orgs entries, each with name, app_id (see require_app_id), image and a non-empty scale_set_profiles list, at most one of private_key and private_key_op_reference, and optionally app_secret_name. A profile may override its namespace and release_name, and set scale_set_labels in place of runs_on_label. + orgs: github_runner_arc_orgs entries, each with name, app_id (see require_app_id), image and a non-empty scale_set_profiles list, at most one of private_key and private_key_op_reference, and optionally app_secret_name. A profile may override its namespace and release_name, and set scale_set_labels in place of runs_on_label. A profile with sizing may set burst (node_label_key, node_label_values, tolerations, max_runners) to let its runner pods overflow onto burst nodes; see arc_runner_placement. Returns: - A dict with ``errors`` (messages, empty when valid), ``profiles`` (one dict per profile with org_name, suffix, namespace, release, target (namespace/release, joined), app_secret, scale_set_labels, values_file, node_selector, autoscale, sizing, max_runners (the static maxRunners the role sets, or None), label and settings, the profile's own keys) and ``autoscaled`` (every profile flagged autoscale, sharing one autoscaler-managed memory pool across their combined scale sets; empty when none are). + A dict with ``errors`` (messages, empty when valid), ``profiles`` (one dict per profile with org_name, suffix, namespace, release, target (namespace/release, joined), app_secret, scale_set_labels, values_file, node_selector, autoscale, sizing, burst (the burst settings with defaults filled in, or None), burst_runners (burst.max_runners, 0 without burst), max_runners (the static maxRunners the role sets, including burst_runners, or None), label and settings, the profile's own keys) and ``autoscaled`` (every profile flagged autoscale, sharing one autoscaler-managed memory pool across their combined scale sets; empty when none are). """ errors: list[str] = [] profiles: list[dict[str, Any]] = [] @@ -524,4 +590,4 @@ class FilterModule: def filters(self) -> dict[str, Any]: """Return the filters this plugin provides.""" - return {"arc_profiles": arc_profiles, "arc_max_runners": arc_max_runners, "arc_measured_max_runners": arc_measured_max_runners, "arc_image_ref": arc_image_ref} + return {"arc_profiles": arc_profiles, "arc_max_runners": arc_max_runners, "arc_measured_max_runners": arc_measured_max_runners, "arc_image_ref": arc_image_ref, "arc_runner_placement": arc_runner_placement} diff --git a/roles/github_runner_arc/README.md b/roles/github_runner_arc/README.md index 7cdb49e..89f1869 100644 --- a/roles/github_runner_arc/README.md +++ b/roles/github_runner_arc/README.md @@ -6,7 +6,7 @@ It runs in two ways. Inside `playbooks/site.yml`, against a k3s server host that ## Validation -`tasks/validate.yml` checks every input that needs neither secrets nor a cluster (a measured sizing's node figures are read later, when the role installs the profile): each org's fields, profile names (derived or overridden) that would collide or make invalid Kubernetes names, two of an org's profiles sharing a release name, a `max_runners` that is not a whole number, the controller's release name and namespace, the same org configured on two hosts, more than one autoscaled profile, values files that do not exist, the autoscaler's pool settings, sizing inputs, and the controller, probe and node label settings. Both playbooks run it in a play of its own before anything changes; the role runs it itself when included some other way. It also bootstraps the heartbeat gist when `github_runner_arc_heartbeat_bootstrap_gist` is on and no host in the inventory has one: it creates the gist with the control node's `gh` session and stops, printing the id to record. +`tasks/validate.yml` checks every input that needs neither secrets nor a cluster (a measured sizing's node figures are read later, when the role installs the profile): each org's fields, profile names (derived or overridden) that would collide or make invalid Kubernetes names, two of an org's profiles sharing a release name, a `max_runners` that is not a whole number, the controller's release name and namespace, the same org configured on two hosts, more than one autoscaled profile, values files that do not exist, the autoscaler's pool settings, sizing inputs, burst settings, and the controller, probe and node label settings. Both playbooks run it in a play of its own before anything changes; the role runs it itself when included some other way. It also bootstraps the heartbeat gist when `github_runner_arc_heartbeat_bootstrap_gist` is on and no host in the inventory has one: it creates the gist with the control node's `gh` session and stops, printing the id to record. Before touching the cluster the role then checks the secrets it is about to write (each org's App key looks like a PEM key, the pull credential and heartbeat token are set) and proves the pull credential can read every image it installs from `github_runner_arc_image_pull_registry`, by getting a pull-scoped registry token and fetching each image's manifest. With an App-sourced pull Secret (see below) it mints each org's token and proves it the same way, before installing anything. @@ -22,6 +22,7 @@ Before touching the cluster the role then checks the secrets it is about to writ - `scale_set_labels`: the whole `scaleSetLabels` list, in place of `runs_on_label`. An empty list sets no `scaleSetLabels`, so jobs target the scale set by its release name. - `autoscale`: `true` on at most one profile in the whole inventory makes it the scale set the autoscaler manages. - `sizing`: derive a runner ceiling from node capacity, given as uniform figures or measured from each eligible node (see Sizing). On an ordinary profile without `max_runners` it sets `maxRunners`; on the autoscaled profile it sets the autoscaler's ceiling, leaving `maxRunners` as the floor. + - `burst`: let the runner pods overflow onto nodes a cluster autoscaler adds on demand, and count those nodes' runners into `maxRunners` (see [Burst nodes](#burst-nodes)). Needs `sizing`, and no `max_runners` or `autoscale`. - `github_runner_arc_values_dir`: the control-node directory relative `values_file` paths are read from. - `github_runner_arc_kubeconfig_path`: the kubeconfig on the host the role runs against. Defaults to `github_runner_cluster`'s kubeconfig; empty means `KUBECONFIG` or `~/.kube/config`. - `github_runner_arc_controller_chart_version`, `github_runner_arc_scaleset_chart_version`: chart pins; unset installs the latest chart. @@ -76,6 +77,25 @@ A profile's own `max_runners` still takes precedence, and on the autoscaled prof The filter behind it is `exadev.github_runner.arc_max_runners`. +## Burst nodes + +A fixed pool can overflow onto nodes that exist only while there is work for them, such as cloud instances a cluster autoscaler (Kubernetes' Cluster Autoscaler, for one) launches when pods are Pending and removes once they are idle. Such nodes usually carry a label and a matching `NoSchedule` taint so nothing else lands on them, and no eligibility label, so measured sizing never counts them. A profile's `burst` lets its runner pods onto them: + +```yaml +burst: + node_label_key: example.com/ci-burst # the label every burst node carries + node_label_values: [standard] # optional; unset accepts any value of the label + tolerations: # optional; for the burst nodes' taint + - key: example.com/ci-burst + operator: Exists + effect: NoSchedule + max_runners: 12 # how many runners the burst nodes hold at most +``` + +The runner pods then lose the eligibility `nodeSelector`, and get instead a required node affinity accepting a node that carries either the eligibility label or the burst label, a preferred one (at the highest weight) for nodes without the burst label, so the scheduler keeps using the fixed pool while it has room, and the tolerations. The profile's own `node_selector` still applies everywhere. The listener keeps the eligibility `nodeSelector`, so it never runs on a node that can disappear. Without an eligibility label the runner pods may already go on any untainted node, so only the tolerations and the preference are added. + +`burst.max_runners` is added to the ceiling `sizing` derives from the fixed nodes (uniform or measured, safety margin included), so ARC creates more runner pods than the fixed nodes hold; the extra pods stay Pending, which is what makes the autoscaler add a burst node. The safety margin is not applied to it, since burst nodes carry nothing else. Derive it from what the autoscaler may launch: each node group's maximum size times the runners one of its nodes holds (`exadev.github_runner.arc_max_runners` with the node's allocatable figures works that out). The placement comes from the `exadev.github_runner.arc_runner_placement` filter. + ## Image pull Secret from the GitHub App When the runner image is a private package owned by the org, `github_runner_arc_image_pull_secret_source: app` pulls it with the runners' own GitHub App rather than a registry token tied to a person. The App needs the `packages: read` permission (add it to `github_runner_arc_app_setup_permissions` when creating the App, or to an existing App's settings), approved by the organisation on the App's installation. diff --git a/roles/github_runner_arc/defaults/main.yml b/roles/github_runner_arc/defaults/main.yml index 1e2828c..42c8f91 100644 --- a/roles/github_runner_arc/defaults/main.yml +++ b/roles/github_runner_arc/defaults/main.yml @@ -1,7 +1,7 @@ --- # ARC (Actions Runner Controller) in an existing Kubernetes cluster: the controller, each org's runner scale sets, and the fleet-health platform (heartbeat and autoscaler). Each part is gated on this host's own github_runner_arc_orgs or github_runner_arc_heartbeat_gist_id, so any host that can reach the cluster's API can be the one Ansible installs from: a k3s server host brought up by github_runner_cluster, or localhost with a kubeconfig (playbooks/arc.yml). On a host github_runner_cluster made an agent, this role installs nothing. -# Orgs whose runner scale sets this host installs. Each entry: name, app_id (needed only when the role writes the App Secret; otherwise read from the existing Secret, and checked against it when set), image, private_key (the App's PEM private key, from any source Ansible reads) or private_key_op_reference (an op:// reference the 1Password adapter resolves into private_key), optionally installation_id (resolved from the App when empty), optionally app_secret_name (the App Secret's name in each of the org's namespaces, default -github-app), and scale_set_profiles, a list of {suffix, namespace?, release_name?, values_file?, max_runners?, node_selector?, runs_on_label?, scale_set_labels?, autoscale?, sizing?}. sizing takes uniform node figures, or measured: true to read each eligible node's free capacity from the cluster at install time. Each profile becomes one namespace (namespace, default arc-runners-), two Secrets in it, and one Helm release (release_name, default -runners); setting both overrides lets the role take over scale sets an existing install created under its own names. See the README for every profile key. +# Orgs whose runner scale sets this host installs. Each entry: name, app_id (needed only when the role writes the App Secret; otherwise read from the existing Secret, and checked against it when set), image, private_key (the App's PEM private key, from any source Ansible reads) or private_key_op_reference (an op:// reference the 1Password adapter resolves into private_key), optionally installation_id (resolved from the App when empty), optionally app_secret_name (the App Secret's name in each of the org's namespaces, default -github-app), and scale_set_profiles, a list of {suffix, namespace?, release_name?, values_file?, max_runners?, node_selector?, runs_on_label?, scale_set_labels?, autoscale?, sizing?, burst?}. sizing takes uniform node figures, or measured: true to read each eligible node's free capacity from the cluster at install time. burst ({node_label_key, node_label_values?, tolerations?, max_runners}) lets a sized profile's runner pods overflow onto nodes carrying node_label_key, which a cluster autoscaler adds when pods are Pending, and adds max_runners to the sized maxRunners. Each profile becomes one namespace (namespace, default arc-runners-), two Secrets in it, and one Helm release (release_name, default -runners); setting both overrides lets the role take over scale sets an existing install created under its own names. See the README for every profile key. github_runner_arc_orgs: [] # Directory on the control node that a relative scale_set_profiles[].values_file is read from. Values files are Jinja templates rendered on the control node, so the target host needs no checkout. A profile with no values_file uses the role's own templates/runner-scale-set-values.yaml.j2. diff --git a/roles/github_runner_arc/meta/argument_specs.yml b/roles/github_runner_arc/meta/argument_specs.yml index 8c736df..84f9dec 100644 --- a/roles/github_runner_arc/meta/argument_specs.yml +++ b/roles/github_runner_arc/meta/argument_specs.yml @@ -10,10 +10,11 @@ argument_specs: default: [] description: - Orgs whose runner scale sets this host installs. Each entry has name, app_id (needed only when the role writes the App Secret; otherwise the existing Secret supplies it), image, private_key or private_key_op_reference, and optionally installation_id and app_secret_name (the App Secret's name, default -github-app). - - Each entry's scale_set_profiles is a non-empty list of profiles, each with suffix and optionally namespace (default arc-runners-), release_name (default -runners), values_file, max_runners, min_runners, container_mode, resources, node_selector, runs_on_label, scale_set_labels, autoscale and sizing. + - Each entry's scale_set_profiles is a non-empty list of profiles, each with suffix and optionally namespace (default arc-runners-), release_name (default -runners), values_file, max_runners, min_runners, container_mode, resources, node_selector, runs_on_label, scale_set_labels, autoscale, sizing and burst. - max_runners, when set, is the release's maxRunners and takes precedence over its values file and its sizing. - scale_set_labels replaces runs_on_label with the whole list of scaleSetLabels; an empty list sets none, so jobs target the release name. - sizing derives a runner ceiling from uniform node figures, or with measured true from each eligible node's allocatable CPU and memory less its current requests and reserve_cpu and reserve_memory_gib. + - burst, on a profile with sizing and without max_runners or autoscale, lets its runner pods overflow onto nodes carrying burst.node_label_key (any value, or one of burst.node_label_values), with burst.tolerations for their taint, and adds burst.max_runners to the sized maxRunners. github_runner_arc_values_dir: type: str default: "" diff --git a/roles/github_runner_arc/templates/scale-set-overlay.yaml.j2 b/roles/github_runner_arc/templates/scale-set-overlay.yaml.j2 index 3780b5b..ce7a8e4 100644 --- a/roles/github_runner_arc/templates/scale-set-overlay.yaml.j2 +++ b/roles/github_runner_arc/templates/scale-set-overlay.yaml.j2 @@ -1,24 +1,27 @@ -{# The values the role itself owns on every scale-set release, rendered on the control node and merged over the profile's values (nested mappings merge, lists are replaced, as Helm does). With node eligibility, probes, max_runners and sizing all off it holds only the profile's nodeSelector. A measured sizing's result is read from github_runner_arc_measured_sizing, which tasks/measure_sizing.yml fills in. `profile` is the expanded profile from plugins/filter/arc.py. #} -{% set node_selector = ({github_runner_arc_node_label_key: github_runner_arc_node_label_value | string} if github_runner_arc_node_label_key | length > 0 else {}) | combine(profile.node_selector) %} +{# The values the role itself owns on every scale-set release, rendered on the control node and merged over the profile's values (nested mappings merge, lists are replaced, as Helm does). With node eligibility, probes, max_runners, sizing and burst all off it holds only the profile's nodeSelector. A measured sizing's result is read from github_runner_arc_measured_sizing, which tasks/measure_sizing.yml fills in. `profile` is the expanded profile from plugins/filter/arc.py. #} +{% set eligibility = {github_runner_arc_node_label_key: github_runner_arc_node_label_value | string} if github_runner_arc_node_label_key | length > 0 else {} %} {% if profile.settings.max_runners is defined and profile.settings.max_runners is not none %} # The profile's own max_runners, which takes precedence over its values file and its sizing. maxRunners: {{ profile.max_runners }} {% elif profile.label in (github_runner_arc_measured_sizing | default({})) and not profile.autoscale %} -# Measured from the eligible nodes' free capacity: {{ github_runner_arc_measured_sizing[profile.label].nodes | map(attribute='runners') | list }} per node, {{ github_runner_arc_measured_sizing[profile.label].theoretical }} in all, less the safety margin. -maxRunners: {{ github_runner_arc_measured_sizing[profile.label].max_runners }} +# Measured from the eligible nodes' free capacity: {{ github_runner_arc_measured_sizing[profile.label].nodes | map(attribute='runners') | list }} per node, {{ github_runner_arc_measured_sizing[profile.label].theoretical }} in all, less the safety margin{% if profile.burst is not none %}, plus {{ profile.burst_runners }} on burst nodes{% endif %}. +maxRunners: {{ github_runner_arc_measured_sizing[profile.label].max_runners + profile.burst_runners }} {% elif profile.max_runners is not none %} -# Derived from the profile's sizing: {{ profile.sizing.per_node }} per node ({{ profile.sizing.per_node_by_cpu }} by CPU, {{ profile.sizing.per_node_by_memory }} by memory) across the nodes, {{ profile.sizing.theoretical }} in theory, less the safety margin. +# Derived from the profile's sizing: {{ profile.sizing.per_node }} per node ({{ profile.sizing.per_node_by_cpu }} by CPU, {{ profile.sizing.per_node_by_memory }} by memory) across the nodes, {{ profile.sizing.theoretical }} in theory, less the safety margin{% if profile.burst is not none %}, plus {{ profile.burst_runners }} on burst nodes{% endif %}. maxRunners: {{ profile.max_runners }} {% endif %} +# The runner pods' placement (plugins/filter/arc.py's arc_runner_placement): a nodeSelector for the eligibility label, or with burst a node affinity that also accepts burst nodes, and tolerations for their taint. nodeSelector and affinity merge over the values file's own mappings; tolerations, a list, replaces the values file's. template: spec: - nodeSelector: {{ node_selector | to_json }} +{% for field, value in (profile | exadev.github_runner.arc_runner_placement(eligibility)).items() %} + {{ field }}: {{ value | to_json }} +{% endfor %} {% if github_runner_arc_node_label_key | length > 0 or github_runner_arc_listener_probes_enabled | bool %} # The chart puts no nodeSelector and no probe on the listener. The AutoscalingRunnerSet needs containers whenever listenerTemplate.spec is set, and a container named listener is merged into the real listener rather than added as a sidecar. listenerTemplate: spec: {% if github_runner_arc_node_label_key | length > 0 %} - nodeSelector: {{ {github_runner_arc_node_label_key: github_runner_arc_node_label_value | string} | to_json }} + nodeSelector: {{ eligibility | to_json }} {% endif %} containers: - name: listener diff --git a/tests/scale_set_overlay/render.yml b/tests/scale_set_overlay/render.yml new file mode 100644 index 0000000..b7831e9 --- /dev/null +++ b/tests/scale_set_overlay/render.yml @@ -0,0 +1,14 @@ +# Renders the github_runner_arc role's scale-set overlay (templates/scale-set-overlay.yaml.j2) for one expanded profile, with the role's defaults and constants, for tests/unit/test_scale_set_overlay.py to read back. Takes profile (an expanded profile from the arc_profiles filter) and overlay_output as extra variables; a role variable such as the eligibility label or github_runner_arc_measured_sizing is overridden with an extra variable of its own name, since the defaults file loaded here outranks play variables. Needs ANSIBLE_COLLECTIONS_PATH to include playbooks/collections, so the template finds this checkout's filters. +- name: Render the scale-set overlay + hosts: localhost + connection: local + gather_facts: false + vars_files: + - ../../roles/github_runner_arc/defaults/main.yml + - ../../roles/github_runner_arc/vars/main.yml + tasks: + - name: Write the overlay + ansible.builtin.copy: + content: "{{ lookup('ansible.builtin.template', playbook_dir ~ '/../../roles/github_runner_arc/templates/scale-set-overlay.yaml.j2') }}" + dest: "{{ overlay_output }}" + mode: "0644" diff --git a/tests/unit/test_arc_filters.py b/tests/unit/test_arc_filters.py index ff57068..3a8be37 100644 --- a/tests/unit/test_arc_filters.py +++ b/tests/unit/test_arc_filters.py @@ -344,6 +344,75 @@ def test_invalid_input_is_reported(self) -> None: self.assertEqual(bad["errors"], ["node a: not a Kubernetes quantity: 'lots'"]) +BURST = {"node_label_key": "example.com/ci-burst", "tolerations": [{"key": "example.com/ci-burst", "operator": "Exists", "effect": "NoSchedule"}], "max_runners": 6} + + +class BurstTest(unittest.TestCase): + def test_burst_runners_are_added_to_the_sized_ceiling(self) -> None: + result = arc.arc_profiles([org(profiles=[{"sizing": SIZING, "burst": BURST}, {"suffix": "-m", "sizing": MEASURED, "burst": BURST}])]) + self.assertEqual(result["errors"], []) + uniform, measured = result["profiles"] + self.assertEqual((uniform["max_runners"], uniform["burst_runners"]), (24 + 6, 6)) + # A measured ceiling is known only at install time; the overlay adds burst_runners to it then. + self.assertEqual((measured["max_runners"], measured["burst_runners"]), (None, 6)) + self.assertEqual(uniform["burst"], {"node_label_key": "example.com/ci-burst", "node_label_values": [], "tolerations": BURST["tolerations"], "max_runners": 6}) + + def test_a_profile_without_burst_has_no_burst_runners(self) -> None: + profile = arc.arc_profiles([org(profiles=[{"sizing": SIZING}])])["profiles"][0] + self.assertEqual((profile["burst"], profile["burst_runners"], profile["max_runners"]), (None, 0, 24)) + + def test_burst_needs_sizing_and_no_explicit_or_autoscaled_ceiling(self) -> None: + for profile in ({"values_file": "v.yaml", "burst": BURST}, {"max_runners": 3, "sizing": SIZING, "burst": BURST}, {"autoscale": True, "values_file": "v.yaml", "sizing": SIZING, "burst": BURST}): + with self.subTest(profile=profile): + errors = arc.arc_profiles([org(profiles=[profile])])["errors"] + self.assertIn("Example scale_set_profiles[0]: burst needs sizing, and no max_runners or autoscale, since its max_runners is added to the ceiling sizing derives", errors) + + def test_invalid_burst_settings_are_errors(self) -> None: + cases = { + "must be a mapping": "yes", + "node_label_key must be a non-empty string": {**BURST, "node_label_key": ""}, + "node_label_values must be a list of non-empty strings": {**BURST, "node_label_values": "standard"}, + "tolerations must be a list of Kubernetes tolerations": {**BURST, "tolerations": ["NoSchedule"]}, + "tolerations[0] has an unknown key 'efect'": {**BURST, "tolerations": [{"key": "k", "efect": "NoSchedule"}]}, + "max_runners must be a whole number of at least 0": {**BURST, "max_runners": 1.5}, + "burst has an unknown key 'nodes'": {**BURST, "nodes": 4}, + } + for message, burst in cases.items(): + with self.subTest(message=message): + errors = arc.arc_profiles([org(profiles=[{"sizing": SIZING, "burst": burst}])])["errors"] + self.assertTrue(any(message in error for error in errors), errors) + missing = {key: value for key, value in BURST.items() if key != "max_runners"} + self.assertTrue(any("burst.max_runners must be a whole number" in error for error in arc.arc_profiles([org(profiles=[{"sizing": SIZING, "burst": missing}])])["errors"])) + + +class RunnerPlacementTest(unittest.TestCase): + ELIGIBLE = {"example.com/ci-eligible": True} + + def placement(self, profile: dict[str, Any], eligibility: dict[str, Any]) -> dict[str, Any]: + expanded = arc.arc_profiles([org(profiles=[{"sizing": SIZING, **profile}])]) + self.assertEqual(expanded["errors"], []) + result: dict[str, Any] = arc.arc_runner_placement(expanded["profiles"][0], eligibility) + return result + + def test_without_burst_the_eligibility_label_is_a_node_selector(self) -> None: + self.assertEqual(self.placement({"node_selector": {"kubernetes.io/arch": "amd64"}}, self.ELIGIBLE), {"nodeSelector": {"example.com/ci-eligible": "True", "kubernetes.io/arch": "amd64"}}) + self.assertEqual(self.placement({}, {}), {"nodeSelector": {}}) + + def test_burst_accepts_an_eligible_or_a_burst_node_and_prefers_the_eligible_one(self) -> None: + placement = self.placement({"node_selector": {"kubernetes.io/arch": "amd64"}, "burst": {**BURST, "node_label_values": ["standard"]}}, {"example.com/ci-eligible": "true"}) + self.assertEqual(placement["nodeSelector"], {"kubernetes.io/arch": "amd64"}) + self.assertEqual(placement["tolerations"], BURST["tolerations"]) + affinity = placement["affinity"]["nodeAffinity"] + self.assertEqual(affinity["requiredDuringSchedulingIgnoredDuringExecution"], {"nodeSelectorTerms": [{"matchExpressions": [{"key": "example.com/ci-eligible", "operator": "In", "values": ["true"]}]}, {"matchExpressions": [{"key": "example.com/ci-burst", "operator": "In", "values": ["standard"]}]}]}) + self.assertEqual(affinity["preferredDuringSchedulingIgnoredDuringExecution"], [{"weight": 100, "preference": {"matchExpressions": [{"key": "example.com/ci-burst", "operator": "DoesNotExist"}]}}]) + + def test_burst_without_an_eligibility_label_only_adds_the_tolerations_and_preference(self) -> None: + placement = self.placement({"burst": BURST}, {}) + self.assertEqual(placement["nodeSelector"], {}) + self.assertNotIn("requiredDuringSchedulingIgnoredDuringExecution", placement["affinity"]["nodeAffinity"]) + self.assertEqual(placement["tolerations"], BURST["tolerations"]) + + class ImageRefTest(unittest.TestCase): def test_splits_registry_repository_and_tag(self) -> None: cases = { diff --git a/tests/unit/test_scale_set_overlay.py b/tests/unit/test_scale_set_overlay.py new file mode 100644 index 0000000..f36d59e --- /dev/null +++ b/tests/unit/test_scale_set_overlay.py @@ -0,0 +1,76 @@ +"""Tests for the github_runner_arc role's scale-set overlay, rendered by ansible-playbook from tests/scale_set_overlay/render.yml: the runner pods' placement and maxRunners, with and without burst nodes.""" + +from __future__ import annotations + +import importlib.util +import json +import os +import subprocess +import tempfile +import unittest +from pathlib import Path +from typing import Any + +import yaml + +ROOT = Path(__file__).resolve().parents[2] +PLUGIN = ROOT / "plugins" / "filter" / "arc.py" +RENDER = ROOT / "tests" / "scale_set_overlay" / "render.yml" +_spec = importlib.util.spec_from_file_location("arc_filters", PLUGIN) +assert _spec is not None and _spec.loader is not None +arc = importlib.util.module_from_spec(_spec) +_spec.loader.exec_module(arc) + +ELIGIBLE = "example.com/ci-eligible" +BURST_KEY = "example.com/ci-burst" +BURST = {"node_label_key": BURST_KEY, "tolerations": [{"key": BURST_KEY, "operator": "Exists", "effect": "NoSchedule"}], "max_runners": 6} +SIZING = {"node_allocatable_cpu": 8, "node_allocatable_memory_gib": 16, "node_count": 3, "pod_cpu_request": 1, "pod_memory_request_gib": 2} +MEASURED = {"measured": True, "pod_cpu_request": 1, "pod_memory_request_gib": 2} + + +def expanded(profile: dict[str, Any]) -> dict[str, Any]: + """Return the one profile arc_profiles expands from profile, failing on any error.""" + result = arc.arc_profiles([{"name": "Example", "image": "ghcr.io/example/runner:1", "scale_set_profiles": [profile]}], require_app_id=False) + assert result["errors"] == [], result["errors"] + return result["profiles"][0] + + +def render(profile: dict[str, Any], **role_vars: Any) -> dict[str, Any]: + """Render the overlay for profile with ansible-playbook and return it parsed.""" + with tempfile.TemporaryDirectory() as work: + output = Path(work) / "overlay.yaml" + extra = Path(work) / "extra.json" + extra.write_text(json.dumps({"profile": profile, "overlay_output": str(output), "github_runner_arc_listener_probes_enabled": False, **role_vars})) + env = {**os.environ, "ANSIBLE_COLLECTIONS_PATH": str(ROOT / "playbooks" / "collections"), "ANSIBLE_LOCALHOST_WARNING": "false", "ANSIBLE_INVENTORY_UNPARSED_WARNING": "false"} + subprocess.run([os.environ.get("ANSIBLE_PLAYBOOK", "ansible-playbook"), str(RENDER), "-e", f"@{extra}"], check=True, env=env, cwd=work, capture_output=True, text=True) + rendered: dict[str, Any] = yaml.safe_load(output.read_text()) + return rendered + + +class OverlayTest(unittest.TestCase): + def test_without_burst_the_runner_pods_keep_the_eligibility_node_selector(self) -> None: + overlay = render(expanded({"sizing": SIZING, "node_selector": {"kubernetes.io/arch": "amd64"}}), github_runner_arc_node_label_key=ELIGIBLE, github_runner_arc_node_label_value="true") + self.assertEqual(overlay["template"]["spec"], {"nodeSelector": {ELIGIBLE: "true", "kubernetes.io/arch": "amd64"}}) + self.assertEqual(overlay["maxRunners"], 24) + self.assertEqual(overlay["listenerTemplate"]["spec"]["nodeSelector"], {ELIGIBLE: "true"}) + + def test_burst_places_the_runner_pods_by_affinity_and_adds_the_burst_runners(self) -> None: + overlay = render(expanded({"sizing": SIZING, "burst": BURST}), github_runner_arc_node_label_key=ELIGIBLE, github_runner_arc_node_label_value="true") + spec = overlay["template"]["spec"] + self.assertEqual(spec["nodeSelector"], {}) + self.assertEqual(spec["tolerations"], BURST["tolerations"]) + terms = spec["affinity"]["nodeAffinity"]["requiredDuringSchedulingIgnoredDuringExecution"]["nodeSelectorTerms"] + self.assertEqual(terms, [{"matchExpressions": [{"key": ELIGIBLE, "operator": "In", "values": ["true"]}]}, {"matchExpressions": [{"key": BURST_KEY, "operator": "Exists"}]}]) + self.assertEqual(overlay["maxRunners"], 24 + BURST["max_runners"]) + # The listener must stay on a fixed node, since a burst node can be removed under it. + self.assertEqual(overlay["listenerTemplate"]["spec"]["nodeSelector"], {ELIGIBLE: "true"}) + + def test_burst_runners_are_added_to_a_measured_ceiling(self) -> None: + profile = expanded({"sizing": MEASURED, "burst": BURST}) + overlay = render(profile, github_runner_arc_measured_sizing={profile["label"]: {"nodes": [{"runners": 3}, {"runners": 2}], "theoretical": 5, "max_runners": 4}}) + self.assertEqual(overlay["maxRunners"], 4 + BURST["max_runners"]) + self.assertEqual(overlay["template"]["spec"]["tolerations"], BURST["tolerations"]) + + +if __name__ == "__main__": + unittest.main()