diff --git a/.github/workflows/litestream.yml b/.github/workflows/litestream.yml index 22b5523..aac9939 100644 --- a/.github/workflows/litestream.yml +++ b/.github/workflows/litestream.yml @@ -35,7 +35,7 @@ jobs: - uses: actions/checkout@v7 - name: Install the pinned Docker Compose - # Always the pinned version, not whatever the runner has: ubuntu-latest's own Compose recreated every node container when the role ran a second time, which the pinned version does not. The user plugin directory takes precedence over the system one. + # Always the pinned version, not whatever the runner has: ubuntu-latest has shipped Compose releases the cluster role refuses (see its tasks/main.yml). The user plugin directory takes precedence over the system one. run: | set -euo pipefail docker version --format 'Docker Engine {{.Server.Version}}' diff --git a/.github/workflows/mesh-integration.yml b/.github/workflows/mesh-integration.yml index 9ac0151..392f33c 100644 --- a/.github/workflows/mesh-integration.yml +++ b/.github/workflows/mesh-integration.yml @@ -1,6 +1,6 @@ name: Mesh integration -# Brings up three k3s nodes in Docker and joins them through Headscale with the cluster role (tests/mesh/run.sh), once per scenario: headscale_hosted, headscale_in_cluster, and the headscale_existing policy check. Only on changes that can affect how nodes join the mesh. +# Brings up three k3s nodes in Docker and joins them through Headscale with the cluster role (tests/mesh/run.sh), once per scenario: headscale_hosted, headscale_in_cluster, and the headscale_existing policy check, plus the role's refusal of a Docker Compose release that recreates the containers it has just built. Only on changes that can affect how nodes join the mesh. # # Runs on GitHub-hosted ubuntu-latest, whose Docker daemon runs the privileged containers the test needs (k3s, and Tailscale inside it). The work directory sits under runner.temp, which the job always has write access to. @@ -25,20 +25,26 @@ permissions: jobs: mesh: - name: Mesh (${{ matrix.scenario }}) + name: Mesh (${{ matrix.scenario }}, Compose ${{ matrix.compose }}) runs-on: ubuntu-latest timeout-minutes: 45 strategy: fail-fast: false matrix: - scenario: [existing, hosted, in_cluster] - env: - COMPOSE_VERSION: v5.5.1 + # The hosted scenario also runs on Compose 2.39.0, the first release after the range the cluster role refuses (docker/compose#13047), so its rerun check proves that release keeps every node running. compose_faulty runs on 2.38.2, the last release in that range, and checks the role refuses it. + include: + - {scenario: existing, compose: v5.5.1} + - {scenario: hosted, compose: v5.5.1} + - {scenario: in_cluster, compose: v5.5.1} + - {scenario: hosted, compose: v2.39.0} + - {scenario: compose_faulty, compose: v2.38.2} steps: - uses: actions/checkout@v7 - - name: Install the pinned Docker Compose - # Always the pinned version, not whatever the runner has: ubuntu-latest's own Compose recreated every node container when the role ran a second time, which the pinned version does not. The user plugin directory takes precedence over the system one. + - name: Install Docker Compose ${{ matrix.compose }} + # Always the matrix's version, not whatever the runner has, so a runner image update never changes what a scenario tests. The user plugin directory takes precedence over the system one. + env: + COMPOSE_VERSION: ${{ matrix.compose }} run: | set -euo pipefail docker version --format 'Docker Engine {{.Server.Version}}' diff --git a/README.md b/README.md index b5a5a9f..359152f 100644 --- a/README.md +++ b/README.md @@ -6,7 +6,7 @@ There are two ways to deploy. A fleet's hosts are managed from a control machine with Ansible, from the fleet's own inventory (see [Ansible](#ansible) below). `bootstrap.sh`, described here, brings up a single machine from a checkout of this repository without a control node or inventory. It runs the same Ansible role against the machine itself, so both paths deploy identically. -Prerequisites on the host: Docker (or Colima) with the Compose plugin, `python3`, `jq`, `openssl`, and `gh`. On macOS, the playbook installs `helm`, `kubectl` and Docker through Homebrew itself. Elsewhere, install `helm` and `kubectl` from the host's package manager first. +Prerequisites on the host: Docker (or Colima) with the Compose plugin, any release but 2.37.1 to 2.38.x (the cluster role refuses those; see Docker Compose versions in `roles/github_runner_cluster/README.md`), `python3`, `jq`, `openssl`, and `gh`. On macOS, the playbook installs `helm`, `kubectl` and Docker through Homebrew itself. Elsewhere, install `helm` and `kubectl` from the host's package manager first. 1. **Create a GitHub App** on the organisation with **Self-hosted runners: read & write** permission, and install it on the organisation. `playbooks/github_app_setup.yml` does this through GitHub's manifest flow (see `roles/github_runner_arc/README.md`). 2. **Configure the environment.** Copy `.env.bootstrap.example` to `.env.bootstrap` and fill in the following (the file's own comments cover the rest, such as joining an existing cluster or running as an agent): diff --git a/roles/github_runner_cluster/README.md b/roles/github_runner_cluster/README.md index ae387fd..bef4617 100644 --- a/roles/github_runner_cluster/README.md +++ b/roles/github_runner_cluster/README.md @@ -4,6 +4,10 @@ Runs a k3s cluster under Docker Compose on each host, from the `docker-compose.y Which hosts are servers is decided automatically from the inventory group named by `github_runner_cluster_group`, in inventory order. The first N hosts are servers, where N is the largest odd number no greater than both the number of hosts and `github_runner_cluster_max_servers`: 1 or 2 hosts give 1 server, 3 or 4 give 3, and 5 or more give 5. The first server bootstraps the cluster, the other servers join it, and the remaining hosts join it as agents. The group is used rather than the play's hosts so that `--limit` never changes the result. The logic lives in `tasks/topology.yml` and the collection's `exadev.github_runner.cluster_topology` filter (`plugins/filter/cluster_topology.py`), and publishes `github_runner_cluster_effective_node_role`, `github_runner_cluster_effective_bootstrap`, `github_runner_cluster_effective_server_url` and `github_runner_cluster_effective_tls_sans` for each host. +## Docker Compose versions + +The role reads `docker compose version` first and fails, before changing anything on the host, if it is 2.37.1 or later but older than 2.39.0. Those releases create a container from an image they have just built without recording that image's ID on it ([docker/compose#13047](https://github.com/docker/compose/pull/13047)), so the next run sees a different image and recreates the container, which restarts every node on the run after the one that built the k3s image. Releases before 2.37.1 and from 2.39.0 on keep the containers. GitHub's ubuntu-latest runner image has shipped 2.38.2, so a CI job that runs the role there needs a different Compose installed first, as `.github/workflows/mesh-integration.yml` does. + ## Mesh providers `github_runner_cluster_mesh` picks the provider. Each one lives in `tasks/mesh/.yml` and publishes the same facts, so nothing else in the role depends on which is in use: @@ -124,4 +128,4 @@ or read from 1Password by adding `exadev.github_runner.github_runner_secrets_one `tests/litestream/run.sh` runs the `headscale_hosted` provider with Litestream and a warm standby against an S3-compatible store on this machine's Docker, registers a real Tailscale client, promotes the standby and checks that users, nodes and pre-auth keys survive, that the client reconnects and that the promoted server keeps replicating. It also checks that `headscale_in_cluster` refuses to run without Litestream. `.github/workflows/litestream.yml` runs it on pull requests that touch the Headscale providers. -`tests/mesh/run.sh` brings up three k3s nodes in Docker on one machine and joins them through Headscale with this role, for the `headscale_hosted` and `headscale_in_cluster` providers, and checks the `headscale_existing` policy check against a real server; `.github/workflows/mesh-integration.yml` runs it on pull requests that touch the role. +`tests/mesh/run.sh` brings up three k3s nodes in Docker on one machine and joins them through Headscale with this role, for the `headscale_hosted` and `headscale_in_cluster` providers, and checks the `headscale_existing` policy check against a real server. Each cluster scenario runs the role a second time and fails if any node container restarted. The `compose_faulty` scenario checks the role refuses a Compose release that has that fault. `.github/workflows/mesh-integration.yml` runs them on pull requests that touch the role, the hosted scenario on 2.39.0, the first fixed release, as well as on a current one. diff --git a/roles/github_runner_cluster/tasks/main.yml b/roles/github_runner_cluster/tasks/main.yml index 16fc0b3..0049895 100644 --- a/roles/github_runner_cluster/tasks/main.yml +++ b/roles/github_runner_cluster/tasks/main.yml @@ -1,6 +1,31 @@ --- # Brings this host up as a k3s node under Docker Compose: a server (bootstrapping the cluster or joining it) or an agent, as tasks/topology.yml decides from the cluster group. Safe to rerun: the .env is templated and docker compose converges to it. +- name: Read this host's Docker Compose version + # First, before anything that runs Compose (the mesh providers start Compose projects too), for the check below. The CLI plugin's own version, which no Docker context or daemon changes. Read-only, so it runs under --check as well. + ansible.builtin.command: docker compose version --short + changed_when: false + check_mode: false + register: github_runner_cluster_compose_version + +- name: Fail if this host's Docker Compose recreates the containers it has just built + # Compose 2.37.1 to 2.38.x build a service's image through Bake, key the build result by service name instead of image name, and so create the service's container without the com.docker.compose.image label (docker/compose#13047, fixed in 2.39.0). The next `up` sees that label differ from the image and recreates the container: every node restarts, and k3s with it, on the run after the one that built the k3s image. Releases before and after that range keep the container. An output with no version number in it fails too, rather than letting the check pass by default. + vars: + github_runner_cluster_compose_faulty_from: "2.37.1" + github_runner_cluster_compose_fixed_in: "2.39.0" + github_runner_cluster_compose_version_number: "{{ github_runner_cluster_compose_version.stdout | regex_search('[0-9]+\\.[0-9]+\\.[0-9]+') }}" + ansible.builtin.fail: + msg: >- + Docker Compose on this host reports version "{{ github_runner_cluster_compose_version.stdout | trim }}". + Compose releases from {{ github_runner_cluster_compose_faulty_from }} up to but not including + {{ github_runner_cluster_compose_fixed_in }} recreate every node container, restarting k3s, on the run after the one + that built the k3s image (docker/compose#13047). Upgrade the Docker Compose + plugin on this host to {{ github_runner_cluster_compose_fixed_in }} or later and run again. + when: >- + github_runner_cluster_compose_version.stdout is not search('[0-9]+\\.[0-9]+\\.[0-9]+') + or (github_runner_cluster_compose_version_number is version(github_runner_cluster_compose_faulty_from, '>=') + and github_runner_cluster_compose_version_number is version(github_runner_cluster_compose_fixed_in, '<')) + - name: Decide this host's part in the cluster ansible.builtin.include_tasks: topology.yml diff --git a/tests/mesh/run.sh b/tests/mesh/run.sh index 7787699..640c076 100755 --- a/tests/mesh/run.sh +++ b/tests/mesh/run.sh @@ -1,5 +1,5 @@ #!/usr/bin/env bash -# Mesh integration test for the github_runner_cluster role: stands up three k3s nodes in Docker on one machine, each in its own copy of the Compose project, and has the role join them through Headscale. Scenarios: hosted headscale_hosted: the role runs Headscale under Compose beside the first node. in_cluster headscale_in_cluster: the role runs Headscale inside the cluster, on the bootstrap server. existing headscale_existing: the policy check fails, with the entry to add, against a server whose policy lacks autoApprovers, and passes once it has one. Every scenario uses automatic server selection (three hosts, no overrides, so three servers). The cluster scenarios assert that etcd has three members, that every node is Ready, and that pods on different nodes reach each other over the mesh. +# Mesh integration test for the github_runner_cluster role: stands up three k3s nodes in Docker on one machine, each in its own copy of the Compose project, and has the role join them through Headscale. Scenarios: hosted headscale_hosted: the role runs Headscale under Compose beside the first node. in_cluster headscale_in_cluster: the role runs Headscale inside the cluster, on the bootstrap server. existing headscale_existing: the policy check fails, with the entry to add, against a server whose policy lacks autoApprovers, and passes once it has one. Every scenario uses automatic server selection (three hosts, no overrides, so three servers). compose_faulty the role refuses a Docker Compose release that recreates the containers it has just built, before writing anything (run it under such a release; it is not in the default set). The cluster scenarios assert that etcd has three members, that every node is Ready, and that pods on different nodes reach each other over the mesh. # # Usage: tests/mesh/run.sh ... (default: all three). Needs Docker with Compose, and Python 3 with ansible-core (ANSIBLE_PLAYBOOK overrides which ansible-playbook runs). Creates only Docker objects named grtest-*, and removes them again on exit unless GRTEST_KEEP=1. set -euo pipefail @@ -274,6 +274,37 @@ scenario_cluster() { fi } +# The role must refuse a Docker Compose release that recreates the containers it has just built (see the check in the cluster role's tasks/main.yml) before it writes anything to the host. The workflow installs such a release for this scenario; every other scenario runs under a good one and so also shows the check lets that through. +scenario_compose_faulty() { + local output node + reset_environment + node="$(node_name 1)" + mkdir -p "${work}/${node}" + cat > "${work}/inventory.yml" <&1); then + echo "$output"; fail "the role ran with a Compose release it should refuse" + fi + echo "$output" | grep -F "Fail if this host's Docker Compose recreates the containers it has just built" >/dev/null || { echo "$output"; fail "the role failed, but not at the Compose version check"; } + echo "$output" | grep -F 'Upgrade the Docker Compose plugin on this host' >/dev/null || { echo "$output"; fail "the failure did not say to upgrade Compose"; } + [ -z "$(ls -A "${work}/${node}")" ] || fail "the role wrote to the cluster directory before refusing: $(ls -A "${work}/${node}")" + echo "$output" | grep -F 'Upgrade the Docker Compose plugin' | head -n 1 + log "It did" +} + scenario_existing() { reset_environment local fixtures="${repo_root}/tests/mesh/fixtures/headscale-existing" api_url api_key output @@ -322,6 +353,7 @@ for scenario in "${scenarios[@]}"; do case "$scenario" in hosted | in_cluster) scenario_cluster "$scenario" ;; existing) scenario_existing ;; + compose_faulty) scenario_compose_faulty ;; # Re-checks a cluster a previous run kept with GRTEST_KEEP=1 and the same GRTEST_WORK, without rebuilding it. assert) assert_cluster ;; *) fail "unknown scenario ${scenario}" ;;