diff --git a/.github/workflows/litestream.yml b/.github/workflows/litestream.yml index aac9939..8fa9ad6 100644 --- a/.github/workflows/litestream.yml +++ b/.github/workflows/litestream.yml @@ -1,6 +1,6 @@ name: Headscale Litestream -# Runs tests/litestream/run.sh: the headscale_hosted provider with Litestream and a warm standby against an S3-compatible store, a real Tailscale client, promotion of the standby, and the headscale_in_cluster guard. Only on changes that can affect them. +# Runs tests/litestream/run.sh: the headscale_hosted provider with Litestream and a warm standby against an S3-compatible store, a real Tailscale client, promotion of the standby, and the headscale_in_cluster guard, once as is and once with the Headscale hosts pinning github_runner_cluster_docker_context while the current Docker context points at nothing. Only on changes that can affect them. # # Runs on GitHub-hosted ubuntu-latest, which has the Docker daemon the test needs. The work directory sits under runner.temp, which the job always has write access to. @@ -9,6 +9,9 @@ on: paths: - roles/github_runner_cluster/tasks/mesh/headscale_*.yml - roles/github_runner_cluster/tasks/headscale_promote.yml + - roles/github_runner_cluster/tasks/docker_context.yml + - plugins/filter/docker.py + - tests/lib/** - roles/github_runner_cluster/templates/headscale/** - roles/github_runner_cluster/files/headscale/** - roles/github_runner_cluster/defaults/main.yml @@ -26,9 +29,14 @@ permissions: jobs: litestream: - name: Litestream and standby promotion + name: Litestream and standby promotion${{ matrix.pin && ' (context pinned)' || '' }} runs-on: ubuntu-latest timeout-minutes: 30 + strategy: + fail-fast: false + matrix: + # Pinned: the Headscale hosts pin github_runner_cluster_docker_context while the current Docker context points at nothing (see tests/litestream/run.sh). + pin: [false, true] env: COMPOSE_VERSION: v5.5.1 steps: @@ -67,6 +75,7 @@ jobs: env: GRTEST_WORK: ${{ runner.temp }}/grtest-ls GRTEST_EXTRA_COLLECTIONS: ${{ runner.temp }}/collections + GRTEST_PIN_CONTEXT: ${{ matrix.pin && '1' || '0' }} run: tests/litestream/run.sh required-checks: diff --git a/.github/workflows/mesh-integration.yml b/.github/workflows/mesh-integration.yml index 392f33c..824bbe0 100644 --- a/.github/workflows/mesh-integration.yml +++ b/.github/workflows/mesh-integration.yml @@ -1,6 +1,6 @@ name: Mesh integration -# Brings up three k3s nodes in Docker and joins them through Headscale with the cluster role (tests/mesh/run.sh), once per scenario: headscale_hosted, headscale_in_cluster, and the headscale_existing policy check, plus the role's refusal of a Docker Compose release that recreates the containers it has just built. Only on changes that can affect how nodes join the mesh. +# Brings up three k3s nodes in Docker and joins them through Headscale with the cluster role (tests/mesh/run.sh), once per scenario: headscale_hosted, headscale_in_cluster, and the headscale_existing policy check, plus the role's refusal of a Docker Compose release that recreates the containers it has just built. headscale_in_cluster also runs with github_runner_cluster_docker_context pinned while the current Docker context points at nothing, which covers the cluster role's own docker calls and the recovery playbook's, and context_checks checks the role refuses a pin it cannot honour. Only on changes that can affect how nodes join the mesh. # # Runs on GitHub-hosted ubuntu-latest, whose Docker daemon runs the privileged containers the test needs (k3s, and Tailscale inside it). The work directory sits under runner.temp, which the job always has write access to. @@ -13,6 +13,7 @@ on: - k3s/** - playbooks/recover_in_cluster_mesh.yml - tests/mesh/** + - tests/lib/** - .github/workflows/mesh-integration.yml workflow_dispatch: @@ -25,7 +26,7 @@ permissions: jobs: mesh: - name: Mesh (${{ matrix.scenario }}, Compose ${{ matrix.compose }}) + name: Mesh (${{ matrix.scenario }}, Compose ${{ matrix.compose }}${{ matrix.pin && ', context pinned' || '' }}) runs-on: ubuntu-latest timeout-minutes: 45 strategy: @@ -38,6 +39,8 @@ jobs: - {scenario: in_cluster, compose: v5.5.1} - {scenario: hosted, compose: v2.39.0} - {scenario: compose_faulty, compose: v2.38.2} + - {scenario: in_cluster, compose: v5.5.1, pin: true} + - {scenario: context_checks, compose: v5.5.1} steps: - uses: actions/checkout@v7 @@ -76,6 +79,7 @@ jobs: env: ANSIBLE_COLLECTIONS_PATH: ${{ runner.temp }}/collections GRTEST_WORK: ${{ runner.temp }}/grtest-mesh + GRTEST_PIN_CONTEXT: ${{ matrix.pin && '1' || '0' }} run: tests/mesh/run.sh "${{ matrix.scenario }}" required-checks: diff --git a/playbooks/recover_in_cluster_mesh.yml b/playbooks/recover_in_cluster_mesh.yml index 2b36d01..65249ad 100644 --- a/playbooks/recover_in_cluster_mesh.yml +++ b/playbooks/recover_in_cluster_mesh.yml @@ -27,6 +27,12 @@ msg: github_runner_cluster_mesh is {{ github_runner_cluster_mesh }}; this playbook only applies to headscale_in_cluster. when: github_runner_cluster_mesh != 'headscale_in_cluster' + - name: Resolve the Docker context the recovery drives, the same one the cluster role uses on this host + # See the cluster role's docker_context.yml: the host's pinned github_runner_cluster_docker_context, or its current context. Every docker command below runs with it. + ansible.builtin.include_role: + name: exadev.github_runner.github_runner_cluster + tasks_from: docker_context + - name: Record where Headscale's files live ansible.builtin.set_fact: github_runner_cluster_recover_headscale_dir: "{{ github_runner_cluster_headscale_dir or github_runner_cluster_dir ~ '/headscale' }}" @@ -36,11 +42,13 @@ - name: Restart the k3s container, which starts k3s off the mesh until Headscale answers ansible.builtin.command: argv: [docker, restart, "{{ github_runner_cluster_container_name }}"] + environment: "{{ github_runner_cluster_docker_cli_environment }}" changed_when: true - name: Wait for the Headscale static pod to answer ansible.builtin.command: argv: "{{ github_runner_cluster_recover_probe }}" + environment: "{{ github_runner_cluster_docker_cli_environment }}" changed_when: false failed_when: false register: github_runner_cluster_recover_answers @@ -54,6 +62,7 @@ - name: Find the k3s container's data volume, which holds Headscale's state ansible.builtin.command: argv: [docker, inspect, --format, "{{ '{{' }} range .Mounts {{ '}}' }}{{ '{{' }} if eq .Destination \"/var/lib/rancher/k3s\" {{ '}}' }}{{ '{{' }} .Name {{ '}}' }}{{ '{{' }} end {{ '}}' }}{{ '{{' }} end {{ '}}' }}", "{{ github_runner_cluster_container_name }}"] + environment: "{{ github_runner_cluster_docker_cli_environment }}" changed_when: false register: github_runner_cluster_recover_volume @@ -77,11 +86,13 @@ - "type=volume,src={{ github_runner_cluster_recover_volume.stdout | trim }},dst=/var/lib/headscale,volume-subpath=github-runner-headscale" - "docker.io/headscale/headscale:v{{ github_runner_cluster_headscale_version }}" - serve + environment: "{{ github_runner_cluster_docker_cli_environment }}" changed_when: true - name: Wait for the temporary Headscale server to answer ansible.builtin.command: argv: "{{ github_runner_cluster_recover_probe }}" + environment: "{{ github_runner_cluster_docker_cli_environment }}" changed_when: false failed_when: false register: github_runner_cluster_recover_rescue_answers @@ -97,6 +108,7 @@ - name: Wait for this node to rejoin the cluster on the mesh ansible.builtin.command: argv: [docker, exec, "{{ github_runner_cluster_container_name }}", kubectl, wait, --for=condition=Ready, "node/{{ github_runner_cluster_topology.hosts[inventory_hostname].node_name }}", --timeout=10s] + environment: "{{ github_runner_cluster_docker_cli_environment }}" changed_when: false register: github_runner_cluster_recover_ready until: github_runner_cluster_recover_ready.rc == 0 @@ -107,6 +119,7 @@ - name: Read the temporary Headscale server's log, for a recovery that did not bring the node back ansible.builtin.command: argv: [docker, logs, --tail, "40", "{{ github_runner_cluster_recover_rescue }}"] + environment: "{{ github_runner_cluster_docker_cli_environment }}" changed_when: false failed_when: false register: github_runner_cluster_recover_rescue_log @@ -120,12 +133,14 @@ - name: Remove the temporary Headscale server, so the static pod can take its port back ansible.builtin.command: argv: [docker, rm, --force, "{{ github_runner_cluster_recover_rescue }}"] + environment: "{{ github_runner_cluster_docker_cli_environment }}" changed_when: true failed_when: false - name: Wait for the Headscale static pod to answer again ansible.builtin.command: argv: "{{ github_runner_cluster_recover_probe }}" + environment: "{{ github_runner_cluster_docker_cli_environment }}" changed_when: false register: github_runner_cluster_recover_final until: github_runner_cluster_recover_final.rc == 0 diff --git a/plugins/filter/docker.py b/plugins/filter/docker.py new file mode 100644 index 0000000..95ddd5f --- /dev/null +++ b/plugins/filter/docker.py @@ -0,0 +1,32 @@ +"""Filters the github_runner_cluster role uses to point the docker CLI at the same daemon as its community.docker module calls.""" + +from __future__ import annotations + +from typing import Any + +from ansible.errors import AnsibleFilterError + + +def docker_cli_environment(context: Any) -> dict[str, str]: + """Return the environment that makes the docker CLI use a pinned Docker context. + + DOCKER_CONTEXT selects the context for that one command, and for the Compose plugin it runs, without touching the host's current context. With no context pinned the environment is empty, so the CLI keeps following the host's current context or DOCKER_HOST. + + :param context: the pinned context name, or an empty value for none. :returns: ``{"DOCKER_CONTEXT": context}``, or an empty mapping. :raises AnsibleFilterError: if the value is not a string. + """ + if context is None or context == "": + return {} + if not isinstance(context, str): + raise AnsibleFilterError(f"A Docker context name must be a string, not {type(context).__name__}: {context!r}.") + name = context.strip() + if name == "": + return {} + return {"DOCKER_CONTEXT": name} + + +class FilterModule: + """Registers the Docker filters with Ansible.""" + + def filters(self) -> dict[str, Any]: + """Return the filters this plugin provides.""" + return {"docker_cli_environment": docker_cli_environment} diff --git a/roles/github_runner_cluster/README.md b/roles/github_runner_cluster/README.md index bef4617..54064fb 100644 --- a/roles/github_runner_cluster/README.md +++ b/roles/github_runner_cluster/README.md @@ -8,6 +8,12 @@ Which hosts are servers is decided automatically from the inventory group named The role reads `docker compose version` first and fails, before changing anything on the host, if it is 2.37.1 or later but older than 2.39.0. Those releases create a container from an image they have just built without recording that image's ID on it ([docker/compose#13047](https://github.com/docker/compose/pull/13047)), so the next run sees a different image and recreates the container, which restarts every node on the run after the one that built the k3s image. Releases before 2.37.1 and from 2.39.0 on keep the containers. GitHub's ubuntu-latest runner image has shipped 2.38.2, so a CI job that runs the role there needs a different Compose installed first, as `.github/workflows/mesh-integration.yml` does. +## Docker contexts + +By default the role drives whichever Docker daemon the host's current Docker context points at (`docker context show`), or the one `DOCKER_HOST` names when that is set. That is right for a host with one Docker engine, but the current context is shared, mutable state that other software changes. The case that matters: a Mac that runs its node in Colima and also has Docker Desktop installed. Launching Docker Desktop switches the current context to `desktop-linux` without asking, so the next run of the role would find no node in Docker Desktop and build a second, empty k3s node there, alongside the real one in Colima, instead of managing it. + +Set `github_runner_cluster_docker_context` in the host's variables to pin the context by name (`colima` in that example) on any host with more than one engine. Every `community.docker` module call in the role then passes it as `cli_context`, and every `docker` command the role runs (Compose, `exec`, `inspect`, `volume`, image pulls, and the commands in `playbooks/recover_in_cluster_mesh.yml` and `playbooks/promote_headscale_standby.yml`) runs with `DOCKER_CONTEXT` set to it, so the modules and the CLI always reach the same daemon. The role checks the pin before it touches the host and fails if the context does not exist there, or if `DOCKER_HOST` is also set, since the two would name the daemon in conflicting ways. It never creates a context and never changes the host's current context; `docker context show` on the host reads the same before and after a run. The Headscale tasks that act on another host (the `headscale_hosted` server and standby, and the promotion playbook) use that host's own value. Run `docker context show` on each host to see what it currently points at before choosing a value. + ## Mesh providers `github_runner_cluster_mesh` picks the provider. Each one lives in `tasks/mesh/.yml` and publishes the same facts, so nothing else in the role depends on which is in use: @@ -87,6 +93,7 @@ Each site is its own cluster: its own inventory group, bootstrap server and mesh - `github_runner_cluster_headscale_*`: the Headscale providers' settings, documented in `defaults/main.yml`. - `github_runner_cluster_node_name`: the k3s node name and mesh hostname, defaulting to the inventory name. - Per-host overrides: `github_runner_cluster_node_role` (`server` or `agent`), `github_runner_cluster_bootstrap` (true picks the bootstrap host, false excludes a host), `github_runner_cluster_server_url` (join this server instead, for a cluster whose servers are outside the inventory), `github_runner_cluster_tls_sans`, `github_runner_cluster_node_address`. The run fails if the result has an even number of servers, no servers, or more than one bootstrap host. +- `github_runner_cluster_docker_context` (default empty): pins the Docker context this host's node runs in (see [Docker contexts](#docker-contexts)). - `github_runner_cluster_compose_files`: extra Compose files for the host's k3s project, merged after `docker-compose.yml`, as paths relative to `github_runner_cluster_dir` or absolute. Several nodes on one Docker host each need their own `github_runner_cluster_dir` and `github_runner_cluster_compose_project`. - `github_runner_cluster_wait_for_nodes`: wait for every node to be Ready afterwards, clearing stale node-password registrations if a node fails to rejoin (installs the kubernetes Python client through `github_runner_k8s_client`). @@ -128,4 +135,4 @@ or read from 1Password by adding `exadev.github_runner.github_runner_secrets_one `tests/litestream/run.sh` runs the `headscale_hosted` provider with Litestream and a warm standby against an S3-compatible store on this machine's Docker, registers a real Tailscale client, promotes the standby and checks that users, nodes and pre-auth keys survive, that the client reconnects and that the promoted server keeps replicating. It also checks that `headscale_in_cluster` refuses to run without Litestream. `.github/workflows/litestream.yml` runs it on pull requests that touch the Headscale providers. -`tests/mesh/run.sh` brings up three k3s nodes in Docker on one machine and joins them through Headscale with this role, for the `headscale_hosted` and `headscale_in_cluster` providers, and checks the `headscale_existing` policy check against a real server. Each cluster scenario runs the role a second time and fails if any node container restarted. The `compose_faulty` scenario checks the role refuses a Compose release that has that fault. `.github/workflows/mesh-integration.yml` runs them on pull requests that touch the role, the hosted scenario on 2.39.0, the first fixed release, as well as on a current one. +`tests/mesh/run.sh` brings up three k3s nodes in Docker on one machine and joins them through Headscale with this role, for the `headscale_hosted` and `headscale_in_cluster` providers, and checks the `headscale_existing` policy check against a real server. Each cluster scenario runs the role a second time and fails if any node container restarted. The `compose_faulty` scenario checks the role refuses a Compose release that has that fault. With `GRTEST_PIN_CONTEXT=1` the cluster scenarios pin `github_runner_cluster_docker_context` while the current Docker context the playbooks see points at a socket that does not exist, so any docker call that ignored the pin fails, and they check the current context is unchanged afterwards; `tests/litestream/run.sh` takes the same setting for the Headscale hosts it delegates to and the promotion playbook. The `context_checks` scenario checks the role refuses a pinned context that does not exist, and one set alongside `DOCKER_HOST`, before writing anything. `.github/workflows/mesh-integration.yml` runs them on pull requests that touch the role, the hosted scenario on 2.39.0, the first fixed release, as well as on a current one. diff --git a/roles/github_runner_cluster/defaults/main.yml b/roles/github_runner_cluster/defaults/main.yml index 4bcd71f..0b5feb1 100644 --- a/roles/github_runner_cluster/defaults/main.yml +++ b/roles/github_runner_cluster/defaults/main.yml @@ -109,6 +109,8 @@ github_runner_cluster_compose_file_list: >- {{ ['docker-compose.yml'] + (['docker-compose.mesh.yml'] if github_runner_cluster_mesh == 'headscale_in_cluster' and github_runner_cluster_effective_bootstrap | default(false) | bool else []) + github_runner_cluster_compose_files }} +# The Docker context this host's node runs in, pinned by name (for example "colima"). Empty, the default, follows the host's current Docker context, or DOCKER_HOST when that is set. When set, every community.docker module call in the role passes it as cli_context and every docker CLI command runs with DOCKER_CONTEXT set to it, so both always reach the same daemon, whatever the host's current context is at the time. The role fails before touching the host if the context does not exist there, or if DOCKER_HOST is set as well, and it never changes the host's current context. Set it on any host with more than one Docker engine installed: on a Mac running its node in Colima with Docker Desktop also installed, merely launching Docker Desktop makes desktop-linux the current context, and an unpinned run would then build a second, empty node in Docker Desktop instead of managing the real one. Per host, like every value here that names something on the host; the Headscale tasks read it from the inventory of the host they act on. +github_runner_cluster_docker_context: "" # The node's k3s container, as docker-compose.yml names it. The headscale_in_cluster provider runs commands inside it; override it only alongside an extra Compose file that renames the container. github_runner_cluster_container_name: github-runner-k3s diff --git a/roles/github_runner_cluster/tasks/docker_context.yml b/roles/github_runner_cluster/tasks/docker_context.yml new file mode 100644 index 0000000..710cfb9 --- /dev/null +++ b/roles/github_runner_cluster/tasks/docker_context.yml @@ -0,0 +1,63 @@ +--- +# Checks and resolves the Docker daemon the role drives on one host, github_runner_cluster_docker_target (the host this play runs on unless the caller names another, as the Headscale tasks do for a host they delegate to). Include it immediately before the docker tasks for that host, since a later include for a different host replaces what it publishes. +# +# Publishes two facts. github_runner_cluster_docker_cli_context is what every community.docker module call for that host passes as cli_context: the host's github_runner_cluster_docker_context when set, otherwise the host's current CLI context, or empty when DOCKER_HOST is set, since the module then takes the daemon from DOCKER_HOST and rejects cli_context alongside it ("parameters are mutually exclusive: docker_host|cli_context", as on a Docker-in-Docker CI runner). github_runner_cluster_docker_cli_environment is the environment every docker CLI call for that host runs with: DOCKER_CONTEXT set to the pinned context, or nothing when no context is pinned, so the CLI follows the host's current context or DOCKER_HOST exactly as the modules do. It never changes the host's own current context. + +- name: Take the Docker context pinned for the host whose Docker the next tasks drive, {{ github_runner_cluster_docker_target | default(inventory_hostname) }} + # The host's own value when it is this host, so that a value set in play vars or by set_fact counts too; another host's through hostvars, which never sees role defaults, hence its empty default. + ansible.builtin.set_fact: + github_runner_cluster_docker_pinned_context: >- + {{ (github_runner_cluster_docker_context + if (github_runner_cluster_docker_target | default(inventory_hostname)) == inventory_hostname + else hostvars[github_runner_cluster_docker_target].github_runner_cluster_docker_context | default('')) | trim }} + +- name: Read how the host chooses its Docker daemon, {{ github_runner_cluster_docker_target | default(inventory_hostname) }} + # Read-only, and every docker task after it depends on its result, so it runs under --check too. Lists the host's context names when the pinned one is missing, for the failure below; the names are not secret, unlike the endpoints `docker context ls` would also print. + ansible.builtin.shell: | + if [ -n "${DOCKER_HOST:-}" ]; then echo docker_host=set; fi + if [ -n "$pinned" ]; then + if docker context inspect "$pinned" >/dev/null 2>&1; then + echo pinned=present + else + echo pinned=missing + for name in $(docker context ls --quiet); do echo "available=$name"; done + fi + elif [ -z "${DOCKER_HOST:-}" ]; then + current=$(docker context show) || exit 1 + echo "current=$current" + fi + environment: + pinned: "{{ github_runner_cluster_docker_pinned_context }}" + delegate_to: "{{ github_runner_cluster_docker_target | default(inventory_hostname) }}" + changed_when: false + check_mode: false + register: github_runner_cluster_docker_probe + +- name: Fail if the pinned Docker context and DOCKER_HOST are both set + # Either names the daemon, and they can name different ones, so neither is allowed to win silently. + ansible.builtin.fail: + msg: >- + github_runner_cluster_docker_context is set to '{{ github_runner_cluster_docker_pinned_context }}' for + {{ github_runner_cluster_docker_target | default(inventory_hostname) }}, but DOCKER_HOST is also set in the environment Ansible + runs docker in there. They conflict: unset DOCKER_HOST for that user, or leave github_runner_cluster_docker_context empty to + use DOCKER_HOST. + when: + - github_runner_cluster_docker_pinned_context | length > 0 + - "'docker_host=set' in github_runner_cluster_docker_probe.stdout_lines" + +- name: Fail if the pinned Docker context does not exist + ansible.builtin.fail: + msg: >- + github_runner_cluster_docker_context names the Docker context '{{ github_runner_cluster_docker_pinned_context }}', which does + not exist on {{ github_runner_cluster_docker_target | default(inventory_hostname) }} (its contexts: + {{ github_runner_cluster_docker_probe.stdout_lines | select('match', '^available=') | map('regex_replace', '^available=', '') | join(', ') }}). + Correct the variable, or create the context with `docker context create`. The role never creates a context or changes the + current one. + when: "'pinned=missing' in github_runner_cluster_docker_probe.stdout_lines" + +- name: Publish the context and environment the docker tasks use on {{ github_runner_cluster_docker_target | default(inventory_hostname) }} + ansible.builtin.set_fact: + github_runner_cluster_docker_cli_context: >- + {{ github_runner_cluster_docker_pinned_context + or (github_runner_cluster_docker_probe.stdout_lines | select('match', '^current=') | map('regex_replace', '^current=', '') | first | default('')) }} + github_runner_cluster_docker_cli_environment: "{{ github_runner_cluster_docker_pinned_context | exadev.github_runner.docker_cli_environment }}" diff --git a/roles/github_runner_cluster/tasks/headscale_promote.yml b/roles/github_runner_cluster/tasks/headscale_promote.yml index 05a12cb..1d91911 100644 --- a/roles/github_runner_cluster/tasks/headscale_promote.yml +++ b/roles/github_runner_cluster/tasks/headscale_promote.yml @@ -12,13 +12,17 @@ ansible.builtin.include_tasks: mesh/headscale_hosted_hosts.yml - name: Stop the old server, so that it neither answers nodes nor writes to the replica again - # Headscale first, then its sidecar, which copies Headscale's last writes to the replica as it stops. The docker CLI rather than the Compose module, which would need the host's Docker context read first, on a host that may not answer. A stopped container stays stopped across reboots under the unless-stopped restart policy. + # Headscale first, then its sidecar, which copies Headscale's last writes to the replica as it stops. The docker CLI rather than the Compose module, which would need the host's Docker context read first, on a host that may not answer; for the same reason that host's pinned github_runner_cluster_docker_context is taken from its inventory rather than checked there first (a missing one shows up as this task's error). A stopped container stays stopped across reboots under the unless-stopped restart policy. ansible.builtin.command: argv: >- {{ ['docker', 'compose', '--project-directory', github_runner_cluster_headscale_host_dir] + (['docker-compose.yml'] + github_runner_cluster_headscale_host_compose_files) | map('regex_replace', '^', '--file=') | list + ['stop', item] }} chdir: "{{ github_runner_cluster_headscale_host_dir }}" + environment: >- + {{ ((github_runner_cluster_docker_context if github_runner_cluster_headscale_host_name == inventory_hostname + else hostvars[github_runner_cluster_headscale_host_name].github_runner_cluster_docker_context | default('')) | trim) + | exadev.github_runner.docker_cli_environment }} loop: [headscale, litestream] delegate_to: "{{ github_runner_cluster_headscale_host_name }}" ignore_unreachable: true @@ -49,12 +53,11 @@ - name: Promote the standby delegate_to: "{{ github_runner_cluster_headscale_standby_name }}" block: - - name: Determine the standby's Docker CLI context - # Empty when DOCKER_HOST is set: the Compose module then takes the daemon from DOCKER_HOST itself, and rejects cli_context alongside it ("parameters are mutually exclusive: docker_host|cli_context", as on a Docker-in-Docker CI runner). The docker CLI calls below need neither, since the CLI follows its own context and DOCKER_HOST. - ansible.builtin.shell: if [ -z "${DOCKER_HOST:-}" ]; then docker context show; fi - changed_when: false - check_mode: false - register: github_runner_cluster_headscale_promote_context + - name: Resolve the standby's Docker context + # See docker_context.yml: the standby's pinned github_runner_cluster_docker_context, or its current context. The docker CLI calls below take the same context through DOCKER_CONTEXT. + ansible.builtin.include_tasks: docker_context.yml + vars: + github_runner_cluster_docker_target: "{{ github_runner_cluster_headscale_standby_name }}" - name: Stop following the replica and remove the follower # Removed, not just stopped: its restart policy would otherwise bring it back once the restore below has replaced the database it was following, and it cannot resume from there. @@ -64,6 +67,7 @@ + (['docker-compose.yml'] + github_runner_cluster_headscale_standby_compose_files) | map('regex_replace', '^', '--file=') | list + ['--profile', 'standby', 'rm', '--stop', '--force', 'litestream-follow'] }} chdir: "{{ github_runner_cluster_headscale_standby_dir }}" + environment: "{{ github_runner_cluster_docker_cli_environment }}" changed_when: true - name: Restore the replica's latest state, replacing anything the standby holds @@ -74,6 +78,7 @@ + (['docker-compose.yml'] + github_runner_cluster_headscale_standby_compose_files) | map('regex_replace', '^', '--file=') | list + ['--profile', 'active', 'run', '--rm', '--no-deps', '-e', 'GITHUB_RUNNER_LITESTREAM_FORCE=1', 'litestream-restore'] }} chdir: "{{ github_runner_cluster_headscale_standby_dir }}" + environment: "{{ github_runner_cluster_docker_cli_environment }}" changed_when: true register: github_runner_cluster_headscale_promote_restore @@ -81,7 +86,7 @@ community.docker.docker_compose_v2: project_src: "{{ github_runner_cluster_headscale_standby_dir }}" files: "{{ ['docker-compose.yml'] + github_runner_cluster_headscale_standby_compose_files }}" - cli_context: "{{ (github_runner_cluster_headscale_promote_context.stdout | trim) or omit }}" + cli_context: "{{ github_runner_cluster_docker_cli_context or omit }}" profiles: [active] state: present wait: true @@ -90,6 +95,7 @@ - name: Check the promoted server answers with the restored nodes ansible.builtin.command: argv: [docker, exec, "{{ github_runner_cluster_headscale_standby_project }}", headscale, nodes, list, --output, json] + environment: "{{ github_runner_cluster_docker_cli_environment }}" changed_when: false register: github_runner_cluster_headscale_promote_nodes until: github_runner_cluster_headscale_promote_nodes.rc == 0 diff --git a/roles/github_runner_cluster/tasks/main.yml b/roles/github_runner_cluster/tasks/main.yml index 0049895..8c23c68 100644 --- a/roles/github_runner_cluster/tasks/main.yml +++ b/roles/github_runner_cluster/tasks/main.yml @@ -1,9 +1,14 @@ --- # Brings this host up as a k3s node under Docker Compose: a server (bootstrapping the cluster or joining it) or an agent, as tasks/topology.yml decides from the cluster group. Safe to rerun: the .env is templated and docker compose converges to it. +- name: Check which Docker daemon the role drives on this host + # First, so that a pinned context that does not exist, or one set alongside DOCKER_HOST, stops the run before anything touches the host. + ansible.builtin.include_tasks: docker_context.yml + - name: Read this host's Docker Compose version - # First, before anything that runs Compose (the mesh providers start Compose projects too), for the check below. The CLI plugin's own version, which no Docker context or daemon changes. Read-only, so it runs under --check as well. + # Before anything that runs Compose (the mesh providers start Compose projects too), for the check below. The CLI plugin's own version, which no Docker context or daemon changes; it still runs with the pinned context, like every docker command in the role. Read-only, so it runs under --check as well. ansible.builtin.command: docker compose version --short + environment: "{{ github_runner_cluster_docker_cli_environment }}" changed_when: false check_mode: false register: github_runner_cluster_compose_version @@ -59,14 +64,9 @@ # Must complete before either branch below templates .env, since both env.j2 and env-agent.j2 read the mesh facts it publishes. See tasks/mesh/main.yml for the provider interface. ansible.builtin.include_tasks: mesh/main.yml -- name: Determine the active Docker CLI context - # community.docker.docker_compose_v2 defaults to unix:///var/run/docker.sock when neither docker_host nor cli_context is given, ignoring whatever context `docker` itself is actually configured to use, so a host running Colima (whose docker.sock lives under ~/.colima/default/, not /var/run) fails outright without this. Read the host's own current context rather than hardcoding "colima", since a Docker Desktop host would need "desktop-linux" instead. - # Empty when DOCKER_HOST is set: the module then takes the daemon from DOCKER_HOST itself, and rejects cli_context alongside it ("parameters are mutually exclusive: docker_host|cli_context", as on a Docker-in-Docker CI runner). - ansible.builtin.shell: if [ -z "${DOCKER_HOST:-}" ]; then docker context show; fi - changed_when: false - # Read-only, and every docker_compose_v2 task below depends on its output, so it has to run under --check too: a skipped result leaves cli_context empty and the dry run targets the wrong daemon. - check_mode: false - register: github_runner_cluster_docker_context +- name: Resolve this host's Docker context again for the Compose tasks below + # community.docker.docker_compose_v2 defaults to unix:///var/run/docker.sock when neither docker_host nor cli_context is given, ignoring whatever context `docker` itself is actually configured to use, so a host running Colima (whose docker.sock lives under ~/.colima/default/, not /var/run) fails outright without a context. docker_context.yml takes the pinned github_runner_cluster_docker_context, or else reads the host's own current context rather than hardcoding "colima", since a Docker Desktop host would need "desktop-linux" instead. Again here, because the mesh step above resolves it for the Headscale host it delegates to, which may be another host. + ansible.builtin.include_tasks: docker_context.yml - name: Set up and start this host as a k3s server, bootstrapping or joining when: github_runner_cluster_effective_node_role == "server" @@ -95,7 +95,7 @@ project_src: "{{ github_runner_cluster_dir }}" project_name: "{{ github_runner_cluster_compose_project }}" profiles: ["server"] - cli_context: "{{ (github_runner_cluster_docker_context.stdout | trim) or omit }}" + cli_context: "{{ github_runner_cluster_docker_cli_context or omit }}" files: "{{ github_runner_cluster_compose_file_list if github_runner_cluster_compose_file_list | length > 1 else omit }}" state: present @@ -114,7 +114,7 @@ project_src: "{{ github_runner_cluster_dir }}" project_name: "{{ github_runner_cluster_compose_project }}" profiles: ["agent"] - cli_context: "{{ (github_runner_cluster_docker_context.stdout | trim) or omit }}" + cli_context: "{{ github_runner_cluster_docker_cli_context or omit }}" files: "{{ github_runner_cluster_compose_file_list if github_runner_cluster_compose_file_list | length > 1 else omit }}" state: present diff --git a/roles/github_runner_cluster/tasks/mesh/headscale_hosted_server.yml b/roles/github_runner_cluster/tasks/mesh/headscale_hosted_server.yml index 17ab746..0fd980e 100644 --- a/roles/github_runner_cluster/tasks/mesh/headscale_hosted_server.yml +++ b/roles/github_runner_cluster/tasks/mesh/headscale_hosted_server.yml @@ -30,20 +30,21 @@ delegate_to: "{{ github_runner_cluster_headscale_server_host }}" when: github_runner_cluster_headscale_litestream | bool - - name: Determine the Headscale host's Docker CLI context - # See the same step in tasks/main.yml for why the context is read rather than assumed. - # Empty when DOCKER_HOST is set: the Compose module then takes the daemon from DOCKER_HOST itself, and rejects cli_context alongside it ("parameters are mutually exclusive: docker_host|cli_context", as on a Docker-in-Docker CI runner). The docker CLI calls below need neither, since the CLI follows its own context and DOCKER_HOST. - ansible.builtin.shell: if [ -z "${DOCKER_HOST:-}" ]; then docker context show; fi - changed_when: false - check_mode: false - register: github_runner_cluster_headscale_docker_context + - name: Resolve the Headscale host's Docker context + # See docker_context.yml: that host's pinned github_runner_cluster_docker_context, or its current context. The docker CLI calls below take the same context through DOCKER_CONTEXT. + ansible.builtin.include_tasks: + file: docker_context.yml + apply: + run_once: true + vars: + github_runner_cluster_docker_target: "{{ github_runner_cluster_headscale_server_host }}" - name: Start the project, recreating it when its configuration, policy or Litestream settings changed # The configuration is a bind mount each container reads only at start, so a changed file needs a new container to take effect. On a standby only the follower's standby profile starts; the server's active profile stays down until promotion. community.docker.docker_compose_v2: project_src: "{{ github_runner_cluster_headscale_server_dir }}" files: "{{ ['docker-compose.yml'] + github_runner_cluster_headscale_server_compose_files }}" - cli_context: "{{ (github_runner_cluster_headscale_docker_context.stdout | trim) or omit }}" + cli_context: "{{ github_runner_cluster_docker_cli_context or omit }}" state: present profiles: "{{ ['standby'] if github_runner_cluster_headscale_as_standby | bool else omit }}" remove_orphans: true @@ -56,6 +57,7 @@ - name: Look for the follower's copy of the database, left from when this host was the standby ansible.builtin.command: argv: [docker, volume, inspect, "{{ github_runner_cluster_headscale_project }}_headscale-standby"] + environment: "{{ github_runner_cluster_docker_cli_environment }}" register: github_runner_cluster_headscale_standby_volume changed_when: false failed_when: false @@ -65,6 +67,7 @@ - name: Remove the follower's copy, which a server never reads ansible.builtin.command: argv: [docker, volume, rm, "{{ github_runner_cluster_headscale_project }}_headscale-standby"] + environment: "{{ github_runner_cluster_docker_cli_environment }}" changed_when: true when: - not github_runner_cluster_headscale_as_standby | bool @@ -79,6 +82,7 @@ + ['exec', '-T', 'litestream', 'cat', '/var/lib/headscale/' ~ item] }} # Relative Compose file paths resolve from here, as they do for the Compose module above. chdir: "{{ github_runner_cluster_headscale_server_dir }}" + environment: "{{ github_runner_cluster_docker_cli_environment }}" loop: "{{ ['noise_private.key'] + (['derp_server_private.key'] if github_runner_cluster_headscale_derp == 'embedded' else []) }}" changed_when: false check_mode: false diff --git a/roles/github_runner_cluster/tasks/mesh/headscale_in_cluster.yml b/roles/github_runner_cluster/tasks/mesh/headscale_in_cluster.yml index 816a1b7..4ef1a76 100644 --- a/roles/github_runner_cluster/tasks/mesh/headscale_in_cluster.yml +++ b/roles/github_runner_cluster/tasks/mesh/headscale_in_cluster.yml @@ -96,19 +96,15 @@ dest: "{{ github_runner_cluster_dir }}/docker-compose.mesh.yml" mode: "0644" - - name: Determine the active Docker CLI context - # Empty when DOCKER_HOST is set: the module then takes the daemon from DOCKER_HOST itself, and rejects cli_context alongside it ("parameters are mutually exclusive: docker_host|cli_context", as on a Docker-in-Docker CI runner). - ansible.builtin.shell: if [ -z "${DOCKER_HOST:-}" ]; then docker context show; fi - changed_when: false - check_mode: false - register: github_runner_cluster_docker_context + - name: Resolve the bootstrap server's Docker context + # See docker_context.yml: the pinned github_runner_cluster_docker_context, or the host's current context. + ansible.builtin.include_tasks: docker_context.yml - name: Keep the Headscale image in the bootstrap host's own Docker, for recovery # playbooks/recover_in_cluster_mesh.yml runs a temporary Headscale container from this image when the node cannot rejoin, which is exactly when pulling it may not work; kubelet's copy inside k3s is in a different image store. Confirmed in CI: the recovery's own pull timed out against Docker Hub. ansible.builtin.shell: | if docker image inspect "$image" >/dev/null 2>&1; then echo present; else docker pull --quiet "$image"; fi - environment: - image: "docker.io/headscale/headscale:v{{ github_runner_cluster_headscale_version }}" + environment: "{{ github_runner_cluster_docker_cli_environment | combine({'image': 'docker.io/headscale/headscale:v' ~ github_runner_cluster_headscale_version}) }}" register: github_runner_cluster_headscale_image changed_when: github_runner_cluster_headscale_image.stdout != 'present' until: github_runner_cluster_headscale_image.rc == 0 @@ -118,6 +114,7 @@ - name: Ask whether Headscale answers inside the bootstrap server's k3s container ansible.builtin.command: argv: [docker, exec, "{{ github_runner_cluster_container_name }}", wget, -q, -T, "5", -O, /dev/null, "http://127.0.0.1:9090/metrics"] + environment: "{{ github_runner_cluster_docker_cli_environment }}" changed_when: false failed_when: false check_mode: false @@ -139,7 +136,7 @@ project_src: "{{ github_runner_cluster_dir }}" project_name: "{{ github_runner_cluster_compose_project }}" profiles: ["server"] - cli_context: "{{ (github_runner_cluster_docker_context.stdout | trim) or omit }}" + cli_context: "{{ github_runner_cluster_docker_cli_context or omit }}" files: "{{ github_runner_cluster_compose_file_list }}" state: present @@ -147,6 +144,7 @@ # First start: k3s starts off the mesh, kubelet pulls the Headscale image and starts the static pod. ansible.builtin.command: argv: [docker, exec, "{{ github_runner_cluster_container_name }}", wget, -q, -T, "5", -O, /dev/null, "http://127.0.0.1:9090/metrics"] + environment: "{{ github_runner_cluster_docker_cli_environment }}" changed_when: false check_mode: false register: github_runner_cluster_headscale_wait @@ -166,6 +164,7 @@ - exec crictl exec "$(crictl ps -q --name '^litestream$' --state running | head -n 1)" cat "/var/lib/headscale/$1" - read-key - "{{ item }}" + environment: "{{ github_runner_cluster_docker_cli_environment }}" loop: "{{ ['noise_private.key'] + (['derp_server_private.key'] if github_runner_cluster_headscale_derp == 'embedded' else []) }}" changed_when: false check_mode: false diff --git a/roles/github_runner_cluster/tasks/mesh/headscale_server_keys.yml b/roles/github_runner_cluster/tasks/mesh/headscale_server_keys.yml index be57910..77b87fa 100644 --- a/roles/github_runner_cluster/tasks/mesh/headscale_server_keys.yml +++ b/roles/github_runner_cluster/tasks/mesh/headscale_server_keys.yml @@ -38,9 +38,20 @@ delay: 2 when: github_runner_cluster_headscale_api_key | length > 0 + - name: Resolve the Headscale host's Docker context, for its CLI + # The CLI is a docker command on that host (github_runner_cluster_headscale_cli), so it takes that host's pinned github_runner_cluster_docker_context; see docker_context.yml. + ansible.builtin.include_tasks: + file: docker_context.yml + apply: + run_once: true + vars: + github_runner_cluster_docker_target: "{{ github_runner_cluster_headscale_host_name }}" + when: github_runner_cluster_headscale_api_key | length == 0 or github_runner_cluster_headscale_api_probe.status | default(0) == 401 + - name: Create an API key with the server's own CLI ansible.builtin.command: argv: "{{ github_runner_cluster_headscale_cli + ['apikeys', 'create', '--expiration', github_runner_cluster_headscale_join_key_days ~ 'd'] }}" + environment: "{{ github_runner_cluster_docker_cli_environment }}" delegate_to: "{{ github_runner_cluster_headscale_host_name }}" changed_when: true no_log: true diff --git a/tests/lib/docker_context.sh b/tests/lib/docker_context.sh new file mode 100644 index 0000000..b6ce011 --- /dev/null +++ b/tests/lib/docker_context.sh @@ -0,0 +1,46 @@ +#!/usr/bin/env bash +# Shared by the integration harnesses, sourced rather than run. Proves the cluster role's github_runner_cluster_docker_context pin wins over the host's current Docker context: it builds a throwaway Docker CLI configuration whose current context is a decoy pointing at a socket that does not exist, plus a context named for the pin that points at this machine's real daemon. Only the ansible-playbook runs see that configuration, through DOCKER_CONFIG; the harness's own docker commands keep the real one, whose current context is never changed. Any docker call the role made without the pin would reach the decoy and fail. + +grtest_pinned_context=grtest-pinned +grtest_decoy_context=grtest-decoy + +# Usage: grtest_setup_pinned_contexts . Creates the configuration there and checks the decoy really is the current context and the pinned one really reaches the daemon. +grtest_setup_pinned_contexts() { + local config="$1" real_config="${DOCKER_CONFIG:-${HOME}/.docker}" endpoint + endpoint="$(docker context inspect --format '{{.Endpoints.docker.Host}}')" + mkdir -p "$config" + # Keep finding the CLI plugins (Compose above all) the real configuration finds, and nothing else from it: its credentials stay out. + if [ -d "${real_config}/cli-plugins" ]; then + ln -s "${real_config}/cli-plugins" "${config}/cli-plugins" + fi + python3 - "${real_config}/config.json" "${config}/config.json" <<'EOF' +import json, sys +try: + with open(sys.argv[1]) as source: + real = json.load(source) +except FileNotFoundError: + real = {} +kept = {key: real[key] for key in ("cliPluginsExtraDirs",) if key in real} +with open(sys.argv[2], "w") as target: + json.dump(kept, target) +EOF + DOCKER_CONFIG="$config" docker context create "$grtest_pinned_context" --docker "host=${endpoint}" >/dev/null + DOCKER_CONFIG="$config" docker context create "$grtest_decoy_context" --docker "host=unix://${config}/decoy.sock" >/dev/null + DOCKER_CONFIG="$config" docker context use "$grtest_decoy_context" >/dev/null 2>&1 + if DOCKER_CONFIG="$config" docker version --format '{{.Server.Version}}' >/dev/null 2>&1; then + echo "The decoy context reached a daemon, so it proves nothing" >&2 + return 1 + fi + DOCKER_CONFIG="$config" docker --context "$grtest_pinned_context" version --format '{{.Server.Version}}' >/dev/null + DOCKER_CONFIG="$config" docker compose version >/dev/null +} + +# Usage: grtest_assert_current_context_unchanged . Fails unless the throwaway configuration's current context is still the decoy, i.e. the role did not switch it. +grtest_assert_current_context_unchanged() { + local current + current="$(DOCKER_CONFIG="$1" docker context show)" + if [ "$current" != "$grtest_decoy_context" ]; then + echo "The role changed the current Docker context from ${grtest_decoy_context} to ${current}" >&2 + return 1 + fi +} diff --git a/tests/litestream/run.sh b/tests/litestream/run.sh index b149028..7622a21 100755 --- a/tests/litestream/run.sh +++ b/tests/litestream/run.sh @@ -3,11 +3,16 @@ # # Everything the Tailscale client, the S3 store and Litestream talk to is on an internal Docker network with no route out, and Headscale serves a DERP map of its own, so nothing contacts Tailscale or any other outside service. A network alias on each project's TLS proxy stands in for the DNS name the promotion playbook tells the operator to move: only the running server's proxy answers it. # +# With GRTEST_PIN_CONTEXT=1, the Headscale hosts pin github_runner_cluster_docker_context while the current Docker context the playbooks see points at nothing (tests/lib/docker_context.sh), so every docker call the mesh step and the promotion make on those hosts, which they reach by delegation, must take the pin from that host's own inventory; afterwards the current context must be unchanged. +# # Usage: tests/litestream/run.sh. Needs Docker with Compose, openssl, and Python 3 with ansible-core and the community.docker collection (ANSIBLE_PLAYBOOK overrides which ansible-playbook runs, and GRTEST_EXTRA_COLLECTIONS adds a collections directory, e.g. where community.docker was installed). Creates only Docker objects named grtest-ls-*, and removes them on exit unless GRTEST_KEEP=1. set -euo pipefail repo_root="$(cd "$(dirname "$0")/../.." && pwd)" here="${repo_root}/tests/litestream" +# shellcheck source=tests/lib/docker_context.sh +. "${repo_root}/tests/lib/docker_context.sh" +pin_context="${GRTEST_PIN_CONTEXT:-0}" ansible_playbook="${ANSIBLE_PLAYBOOK:-ansible-playbook}" network=grtest-ls-net s3=grtest-ls-s3 @@ -133,15 +138,24 @@ litestream_version() { } run_playbook() { - local playbook="$1" + local playbook="$1" config=() shift - ANSIBLE_COLLECTIONS_PATH="${repo_root}/playbooks/collections${GRTEST_EXTRA_COLLECTIONS:+:${GRTEST_EXTRA_COLLECTIONS}}" ANSIBLE_HOST_KEY_CHECKING=False ANSIBLE_RETRY_FILES_ENABLED=False \ + if [ "$pin_context" = 1 ]; then config=("DOCKER_CONFIG=${work}/docker-config"); fi + env "${config[@]}" ANSIBLE_COLLECTIONS_PATH="${repo_root}/playbooks/collections${GRTEST_EXTRA_COLLECTIONS:+:${GRTEST_EXTRA_COLLECTIONS}}" ANSIBLE_HOST_KEY_CHECKING=False ANSIBLE_RETRY_FILES_ENABLED=False \ "$ansible_playbook" -i "${work}/inventory.yml" "$playbook" "$@" /dev/null @@ -233,10 +247,12 @@ all: headscale: hosts: primary: + ${pin_inventory} github_runner_cluster_headscale_dir: ${work}/${primary} github_runner_cluster_headscale_container_name: ${primary} github_runner_cluster_headscale_compose_files: [${work}/${primary}-override.yml] standby: + ${pin_inventory} github_runner_cluster_headscale_dir: ${work}/${standby} github_runner_cluster_headscale_container_name: ${standby} github_runner_cluster_headscale_compose_files: [${work}/${standby}-override.yml] @@ -305,4 +321,8 @@ fi [ "$(snapshot "$standby" | python3 -c 'import json,sys; print(len(json.load(sys.stdin)["users"]))')" -ge 2 ] || fail "the promoted server lost users when the role reran" wait_for 60 client_online_on "$standby" || fail "the client is not online after the role reran on the promoted server" +if [ "$pin_context" = 1 ]; then + grtest_assert_current_context_unchanged "${work}/docker-config" || fail "the playbooks changed the current Docker context" +fi + log "PASS: promotion took $((promoted - promote_started))s; the client was back online $((reconnected - promote_started))s after promotion started" diff --git a/tests/mesh/run.sh b/tests/mesh/run.sh index 640c076..75370c6 100755 --- a/tests/mesh/run.sh +++ b/tests/mesh/run.sh @@ -1,10 +1,15 @@ #!/usr/bin/env bash # Mesh integration test for the github_runner_cluster role: stands up three k3s nodes in Docker on one machine, each in its own copy of the Compose project, and has the role join them through Headscale. Scenarios: hosted headscale_hosted: the role runs Headscale under Compose beside the first node. in_cluster headscale_in_cluster: the role runs Headscale inside the cluster, on the bootstrap server. existing headscale_existing: the policy check fails, with the entry to add, against a server whose policy lacks autoApprovers, and passes once it has one. Every scenario uses automatic server selection (three hosts, no overrides, so three servers). compose_faulty the role refuses a Docker Compose release that recreates the containers it has just built, before writing anything (run it under such a release; it is not in the default set). The cluster scenarios assert that etcd has three members, that every node is Ready, and that pods on different nodes reach each other over the mesh. # +# With GRTEST_PIN_CONTEXT=1, the cluster scenarios run the role, and the in_cluster recovery playbook, with github_runner_cluster_docker_context pinned while the current Docker context they see points at nothing (tests/lib/docker_context.sh), so any docker call that ignored the pin fails the run; afterwards the current context must be unchanged. context_checks the role refuses a pinned context that does not exist, and a pinned context alongside DOCKER_HOST, before writing anything (not in the default set). +# # Usage: tests/mesh/run.sh ... (default: all three). Needs Docker with Compose, and Python 3 with ansible-core (ANSIBLE_PLAYBOOK overrides which ansible-playbook runs). Creates only Docker objects named grtest-*, and removes them again on exit unless GRTEST_KEEP=1. set -euo pipefail repo_root="$(cd "$(dirname "$0")/../.." && pwd)" +# shellcheck source=tests/lib/docker_context.sh +. "${repo_root}/tests/lib/docker_context.sh" +pin_context="${GRTEST_PIN_CONTEXT:-0}" ansible_playbook="${ANSIBLE_PLAYBOOK:-ansible-playbook}" network=grtest-mesh subnet_prefix=172.31.250 @@ -16,6 +21,8 @@ busybox_image=busybox:1.37 reach_attempts=8 work="${GRTEST_WORK:-$(mktemp -d "${TMPDIR:-/tmp}/grtest-mesh.XXXXXX")}" k3s_token="grtest-$(od -An -N12 -tx1 /dev/urandom | tr -d ' \n')" +# The Docker CLI configuration the ansible-playbook runs see when the context is pinned; see tests/lib/docker_context.sh. +docker_config="${work}/docker-config" log() { echo "==> $*"; } fail() { @@ -66,6 +73,13 @@ teardown() { rm -rf "$work" } +# Runs ansible-playbook from ansible/, with the throwaway Docker configuration when the context is pinned. +ansible_run() { + local config=() + if [ "$pin_context" = 1 ]; then config=("DOCKER_CONFIG=${docker_config}"); fi + (cd "${repo_root}/ansible" && env "${config[@]}" ANSIBLE_COLLECTIONS_PATH="${repo_root}/playbooks/collections:${ANSIBLE_COLLECTIONS_PATH:-}" "$ansible_playbook" "$@") +} + cleanup() { if [ "${GRTEST_KEEP:-}" = 1 ]; then log "Keeping the test environment in ${work} (GRTEST_KEEP=1)" @@ -80,6 +94,10 @@ reset_environment() { teardown mkdir -p "$work" docker network create --subnet "${subnet_prefix}.0/24" "$network" >/dev/null + if [ "$pin_context" = 1 ]; then + log "Pinning the Docker context ${grtest_pinned_context}, with the current context pointing at nothing" + grtest_setup_pinned_contexts "$docker_config" + fi } # Each node gets its own cluster directory, which the role fills with the Compose file and k3s build context, plus an override that renames its container, drops the host port every node would otherwise publish, and puts it on the shared test network at a fixed address. @@ -143,6 +161,9 @@ EOF echo " github_runner_cluster_headscale_compose_files: [${work}/headscale-grtest.yml]" echo " github_runner_cluster_compose_files: [compose.grtest.yml]" echo " github_runner_cluster_k3s_token: ${k3s_token}" + if [ "$pin_context" = 1 ]; then + echo " github_runner_cluster_docker_context: ${grtest_pinned_context}" + fi # Litestream has its own test (tests/litestream); this one exercises the in-cluster bootstrap on node-local state, which the role only allows once acknowledged. if [ "$scenario" = in_cluster ]; then echo " github_runner_cluster_headscale_accept_node_local_state: true" @@ -159,8 +180,9 @@ EOF run_role() { log "Running the cluster role" - (cd "${repo_root}/ansible" && ANSIBLE_COLLECTIONS_PATH="${repo_root}/playbooks/collections:${ANSIBLE_COLLECTIONS_PATH:-}" \ - "$ansible_playbook" -i "${work}/inventory.yml" "${repo_root}/tests/mesh/cluster.yml") + # Returns the playbook's own status, which a caller testing it (compose_faulty) relies on, errexit being off there. + ansible_run -i "${work}/inventory.yml" "${repo_root}/tests/mesh/cluster.yml" || return + if [ "$pin_context" = 1 ]; then grtest_assert_current_context_unchanged "$docker_config" || fail "the role changed the current Docker context"; fi } kubectl_on() { docker exec -i "$(container "$1")" kubectl "${@:2}"; } @@ -267,19 +289,16 @@ scenario_cluster() { assert_cluster if [ "$scenario" = in_cluster ]; then log "Cold-restarting the bootstrap server, which cannot rejoin a three-server cluster on its own, then recovering it" - (cd "${repo_root}/ansible" && ANSIBLE_COLLECTIONS_PATH="${repo_root}/playbooks/collections:${ANSIBLE_COLLECTIONS_PATH:-}" \ - "$ansible_playbook" -i "${work}/inventory.yml" "${repo_root}/playbooks/recover_in_cluster_mesh.yml") \ + ansible_run -i "${work}/inventory.yml" "${repo_root}/playbooks/recover_in_cluster_mesh.yml" \ || fail "the recovery playbook did not bring the bootstrap server back" + if [ "$pin_context" = 1 ]; then grtest_assert_current_context_unchanged "$docker_config" || fail "the recovery playbook changed the current Docker context"; fi assert_cluster fi } -# The role must refuse a Docker Compose release that recreates the containers it has just built (see the check in the cluster role's tasks/main.yml) before it writes anything to the host. The workflow installs such a release for this scenario; every other scenario runs under a good one and so also shows the check lets that through. -scenario_compose_faulty() { - local output node - reset_environment - node="$(node_name 1)" - mkdir -p "${work}/${node}" +# One node and only the settings the role reads before its first check fails, for the scenarios that expect it to stop before touching the host. +write_minimal_inventory() { + local node="$1" cat > "${work}/inventory.yml" <&1); then echo "$output"; fail "the role ran with a Compose release it should refuse" @@ -305,6 +333,35 @@ EOF log "It did" } +# The role must refuse a pinned context that does not exist, and a pinned context alongside DOCKER_HOST, before it writes anything to the host. +scenario_context_checks() { + local output node + reset_environment + # reset_environment has already made it when every scenario runs pinned. + [ "$pin_context" = 1 ] || grtest_setup_pinned_contexts "$docker_config" + node="$(node_name 1)" + mkdir -p "${work}/${node}" + write_minimal_inventory "$node" + + log "A pinned context that does not exist must stop the role before it touches the host" + if output=$(pin_context=1 ansible_run -i "${work}/inventory.yml" "${repo_root}/tests/mesh/cluster.yml" -e github_runner_cluster_docker_context=grtest-missing 2>&1); then + echo "$output"; fail "the role ran with a pinned context that does not exist" + fi + echo "$output" | grep -F "names the Docker context 'grtest-missing', which does not exist" >/dev/null || { echo "$output"; fail "the failure did not say the pinned context is missing"; } + echo "$output" | grep -F "$grtest_pinned_context" >/dev/null || { echo "$output"; fail "the failure did not list the contexts that do exist"; } + [ -z "$(ls -A "${work}/${node}")" ] || fail "the role wrote to the cluster directory before refusing: $(ls -A "${work}/${node}")" + log "It did" + + log "A pinned context alongside DOCKER_HOST must stop the role before it touches the host" + if output=$(pin_context=1 DOCKER_HOST="unix://${docker_config}/decoy.sock" ansible_run -i "${work}/inventory.yml" "${repo_root}/tests/mesh/cluster.yml" -e "github_runner_cluster_docker_context=${grtest_pinned_context}" 2>&1); then + echo "$output"; fail "the role ran with both a pinned context and DOCKER_HOST" + fi + echo "$output" | grep -F 'but DOCKER_HOST is also set' >/dev/null || { echo "$output"; fail "the failure did not name the conflict with DOCKER_HOST"; } + [ -z "$(ls -A "${work}/${node}")" ] || fail "the role wrote to the cluster directory before refusing: $(ls -A "${work}/${node}")" + grtest_assert_current_context_unchanged "$docker_config" || fail "the role changed the current Docker context" + log "It did" +} + scenario_existing() { reset_environment local fixtures="${repo_root}/tests/mesh/fixtures/headscale-existing" api_url api_key output @@ -354,6 +411,7 @@ for scenario in "${scenarios[@]}"; do hosted | in_cluster) scenario_cluster "$scenario" ;; existing) scenario_existing ;; compose_faulty) scenario_compose_faulty ;; + context_checks) scenario_context_checks ;; # Re-checks a cluster a previous run kept with GRTEST_KEEP=1 and the same GRTEST_WORK, without rebuilding it. assert) assert_cluster ;; *) fail "unknown scenario ${scenario}" ;; diff --git a/tests/unit/test_docker.py b/tests/unit/test_docker.py new file mode 100644 index 0000000..801b4df --- /dev/null +++ b/tests/unit/test_docker.py @@ -0,0 +1,38 @@ +"""Tests for the filter that points the docker CLI at the Docker context the cluster role pins.""" + +from __future__ import annotations + +import importlib.util +import unittest +from pathlib import Path + +PLUGIN = Path(__file__).resolve().parents[2] / "plugins" / "filter" / "docker.py" +_spec = importlib.util.spec_from_file_location("docker_filters", PLUGIN) +assert _spec is not None and _spec.loader is not None +docker = importlib.util.module_from_spec(_spec) +_spec.loader.exec_module(docker) + + +class DockerCliEnvironmentTest(unittest.TestCase): + def test_a_pinned_context_sets_docker_context(self) -> None: + self.assertEqual(docker.docker_cli_environment("colima"), {"DOCKER_CONTEXT": "colima"}) + + def test_surrounding_whitespace_is_not_part_of_the_name(self) -> None: + self.assertEqual(docker.docker_cli_environment(" colima\n"), {"DOCKER_CONTEXT": "colima"}) + + def test_no_pinned_context_leaves_the_environment_alone(self) -> None: + # Empty, so the CLI keeps following the host's current context or DOCKER_HOST. + for value in ("", " ", None): + with self.subTest(value=value): + self.assertEqual(docker.docker_cli_environment(value), {}) + + def test_a_value_that_is_not_a_name_is_an_error(self) -> None: + with self.assertRaises(docker.AnsibleFilterError): + docker.docker_cli_environment(["colima"]) + + def test_the_filter_is_registered(self) -> None: + self.assertIs(docker.FilterModule().filters()["docker_cli_environment"], docker.docker_cli_environment) + + +if __name__ == "__main__": + unittest.main()